@cogitator-ai/core 0.24.0 → 0.26.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +531 -167
- package/dist/agent-tool.d.ts +4 -3
- package/dist/agent-tool.d.ts.map +1 -1
- package/dist/agent-tool.js +15 -2
- package/dist/agent-tool.js.map +1 -1
- package/dist/agent.d.ts.map +1 -1
- package/dist/agent.js +3 -1
- package/dist/agent.js.map +1 -1
- package/dist/cache/cache-key.d.ts +1 -0
- package/dist/cache/cache-key.d.ts.map +1 -1
- package/dist/cache/cache-key.js +2 -1
- package/dist/cache/cache-key.js.map +1 -1
- package/dist/cache/storage/memory.d.ts +8 -1
- package/dist/cache/storage/memory.d.ts.map +1 -1
- package/dist/cache/storage/memory.js +8 -1
- package/dist/cache/storage/memory.js.map +1 -1
- package/dist/cache/storage/redis.d.ts +4 -0
- package/dist/cache/storage/redis.d.ts.map +1 -1
- package/dist/cache/storage/redis.js +17 -18
- package/dist/cache/storage/redis.js.map +1 -1
- package/dist/cache/tool-cache.d.ts +2 -0
- package/dist/cache/tool-cache.d.ts.map +1 -1
- package/dist/cache/tool-cache.js +4 -2
- package/dist/cache/tool-cache.js.map +1 -1
- package/dist/causal/inference/counterfactual.d.ts +2 -0
- package/dist/causal/inference/counterfactual.d.ts.map +1 -1
- package/dist/causal/inference/counterfactual.js +24 -3
- package/dist/causal/inference/counterfactual.js.map +1 -1
- package/dist/causal/inference/expression.d.ts +19 -0
- package/dist/causal/inference/expression.d.ts.map +1 -0
- package/dist/causal/inference/expression.js +177 -0
- package/dist/causal/inference/expression.js.map +1 -0
- package/dist/cogitator/initializers.d.ts +4 -3
- package/dist/cogitator/initializers.d.ts.map +1 -1
- package/dist/cogitator/initializers.js +100 -46
- package/dist/cogitator/initializers.js.map +1 -1
- package/dist/cogitator/message-builder.d.ts.map +1 -1
- package/dist/cogitator/message-builder.js +5 -1
- package/dist/cogitator/message-builder.js.map +1 -1
- package/dist/cogitator/prompts.d.ts.map +1 -1
- package/dist/cogitator/prompts.js +5 -1
- package/dist/cogitator/prompts.js.map +1 -1
- package/dist/cogitator/response-format.d.ts +7 -2
- package/dist/cogitator/response-format.d.ts.map +1 -1
- package/dist/cogitator/response-format.js +45 -7
- package/dist/cogitator/response-format.js.map +1 -1
- package/dist/cogitator/run-limiter.d.ts.map +1 -1
- package/dist/cogitator/run-limiter.js +5 -1
- package/dist/cogitator/run-limiter.js.map +1 -1
- package/dist/cogitator/tool-executor.d.ts +8 -2
- package/dist/cogitator/tool-executor.d.ts.map +1 -1
- package/dist/cogitator/tool-executor.js +54 -5
- package/dist/cogitator/tool-executor.js.map +1 -1
- package/dist/constitutional/constitutional-ai.d.ts +7 -0
- package/dist/constitutional/constitutional-ai.d.ts.map +1 -1
- package/dist/constitutional/constitutional-ai.js +11 -0
- package/dist/constitutional/constitutional-ai.js.map +1 -1
- package/dist/constitutional/tool-guard.d.ts +9 -0
- package/dist/constitutional/tool-guard.d.ts.map +1 -1
- package/dist/constitutional/tool-guard.js +25 -20
- package/dist/constitutional/tool-guard.js.map +1 -1
- package/dist/cost-routing/cost-estimator.d.ts.map +1 -1
- package/dist/cost-routing/cost-estimator.js +5 -1
- package/dist/cost-routing/cost-estimator.js.map +1 -1
- package/dist/cost-routing/cost-router.d.ts +13 -0
- package/dist/cost-routing/cost-router.d.ts.map +1 -1
- package/dist/cost-routing/cost-router.js +22 -1
- package/dist/cost-routing/cost-router.js.map +1 -1
- package/dist/cost-routing/model-selector.d.ts +13 -0
- package/dist/cost-routing/model-selector.d.ts.map +1 -1
- package/dist/cost-routing/model-selector.js +25 -16
- package/dist/cost-routing/model-selector.js.map +1 -1
- package/dist/index.d.ts +3 -3
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +2 -2
- package/dist/index.js.map +1 -1
- package/dist/learning/ab-testing.d.ts +1 -1
- package/dist/learning/ab-testing.d.ts.map +1 -1
- package/dist/learning/ab-testing.js +6 -28
- package/dist/learning/ab-testing.js.map +1 -1
- package/dist/learning/agent-optimizer.d.ts +2 -0
- package/dist/learning/agent-optimizer.d.ts.map +1 -1
- package/dist/learning/agent-optimizer.js +20 -2
- package/dist/learning/agent-optimizer.js.map +1 -1
- package/dist/learning/auto-optimizer.d.ts +11 -0
- package/dist/learning/auto-optimizer.d.ts.map +1 -1
- package/dist/learning/auto-optimizer.js +47 -16
- package/dist/learning/auto-optimizer.js.map +1 -1
- package/dist/learning/metrics.d.ts +8 -1
- package/dist/learning/metrics.d.ts.map +1 -1
- package/dist/learning/metrics.js +48 -16
- package/dist/learning/metrics.js.map +1 -1
- package/dist/learning/postgres-trace-store.d.ts +3 -1
- package/dist/learning/postgres-trace-store.d.ts.map +1 -1
- package/dist/learning/postgres-trace-store.js +27 -4
- package/dist/learning/postgres-trace-store.js.map +1 -1
- package/dist/learning/prompt-logger.d.ts +2 -2
- package/dist/learning/prompt-logger.d.ts.map +1 -1
- package/dist/learning/prompt-logger.js.map +1 -1
- package/dist/learning/trace-builder.d.ts.map +1 -1
- package/dist/learning/trace-builder.js +1 -0
- package/dist/learning/trace-builder.js.map +1 -1
- package/dist/llm/anthropic.d.ts +1 -0
- package/dist/llm/anthropic.d.ts.map +1 -1
- package/dist/llm/anthropic.js +6 -1
- package/dist/llm/anthropic.js.map +1 -1
- package/dist/llm/base.d.ts +2 -2
- package/dist/llm/base.d.ts.map +1 -1
- package/dist/llm/bedrock.d.ts.map +1 -1
- package/dist/llm/bedrock.js +3 -1
- package/dist/llm/bedrock.js.map +1 -1
- package/dist/llm/debug.d.ts +2 -2
- package/dist/llm/debug.d.ts.map +1 -1
- package/dist/llm/debug.js.map +1 -1
- package/dist/llm/google.d.ts.map +1 -1
- package/dist/llm/google.js +56 -6
- package/dist/llm/google.js.map +1 -1
- package/dist/llm/openai-compatible-base.d.ts +4 -0
- package/dist/llm/openai-compatible-base.d.ts.map +1 -1
- package/dist/llm/openai-compatible-base.js +38 -13
- package/dist/llm/openai-compatible-base.js.map +1 -1
- package/dist/llm/openai-responses.js +1 -1
- package/dist/llm/openai-responses.js.map +1 -1
- package/dist/llm/retry.d.ts +2 -2
- package/dist/llm/retry.d.ts.map +1 -1
- package/dist/llm/retry.js.map +1 -1
- package/dist/logger.d.ts +11 -2
- package/dist/logger.d.ts.map +1 -1
- package/dist/logger.js +35 -6
- package/dist/logger.js.map +1 -1
- package/dist/observability/langfuse.d.ts +1 -0
- package/dist/observability/langfuse.d.ts.map +1 -1
- package/dist/observability/langfuse.js.map +1 -1
- package/dist/observability/opentelemetry.d.ts +1 -0
- package/dist/observability/opentelemetry.d.ts.map +1 -1
- package/dist/observability/opentelemetry.js.map +1 -1
- package/dist/reasoning/thought-tree.d.ts +28 -4
- package/dist/reasoning/thought-tree.d.ts.map +1 -1
- package/dist/reasoning/thought-tree.js +173 -110
- package/dist/reasoning/thought-tree.js.map +1 -1
- package/dist/runtime.d.ts +39 -5
- package/dist/runtime.d.ts.map +1 -1
- package/dist/runtime.js +158 -50
- package/dist/runtime.js.map +1 -1
- package/dist/security/pii.d.ts +2 -2
- package/dist/security/pii.d.ts.map +1 -1
- package/dist/security/pii.js.map +1 -1
- package/dist/time-travel/checkpoint-store.d.ts +6 -1
- package/dist/time-travel/checkpoint-store.d.ts.map +1 -1
- package/dist/time-travel/checkpoint-store.js +15 -18
- package/dist/time-travel/checkpoint-store.js.map +1 -1
- package/dist/time-travel/replayer.d.ts +0 -1
- package/dist/time-travel/replayer.d.ts.map +1 -1
- package/dist/time-travel/replayer.js +12 -19
- package/dist/time-travel/replayer.js.map +1 -1
- package/dist/time-travel/time-travel.d.ts +0 -1
- package/dist/time-travel/time-travel.d.ts.map +1 -1
- package/dist/time-travel/time-travel.js +15 -14
- package/dist/time-travel/time-travel.js.map +1 -1
- package/dist/tool.d.ts +7 -0
- package/dist/tool.d.ts.map +1 -1
- package/dist/tool.js +10 -0
- package/dist/tool.js.map +1 -1
- package/dist/tools/index.d.ts +3 -416
- package/dist/tools/index.d.ts.map +1 -1
- package/dist/tools/index.js +1 -0
- package/dist/tools/index.js.map +1 -1
- package/dist/tools/memory-tools.d.ts +4 -4
- package/dist/tools/memory-tools.d.ts.map +1 -1
- package/dist/tools/memory-tools.js +2 -2
- package/dist/tools/memory-tools.js.map +1 -1
- package/dist/tools/scheduler-tools.d.ts +4 -4
- package/dist/tools/scheduler-tools.d.ts.map +1 -1
- package/dist/tools/scheduler-tools.js +2 -2
- package/dist/tools/scheduler-tools.js.map +1 -1
- package/dist/tools/sql-query.d.ts.map +1 -1
- package/dist/tools/sql-query.js +2 -1
- package/dist/tools/sql-query.js.map +1 -1
- package/dist/tools/vector-search.d.ts.map +1 -1
- package/dist/tools/vector-search.js +12 -3
- package/dist/tools/vector-search.js.map +1 -1
- package/package.json +6 -6
package/README.md
CHANGED
|
@@ -8,6 +8,19 @@ Core runtime for Cogitator AI agents. Build and run LLM-powered agents with tool
|
|
|
8
8
|
pnpm add @cogitator-ai/core zod
|
|
9
9
|
```
|
|
10
10
|
|
|
11
|
+
Optional peer dependencies, installed only for the features that use them:
|
|
12
|
+
|
|
13
|
+
| Package | Needed for |
|
|
14
|
+
| --------------------------------- | --------------------------------------------------------------- |
|
|
15
|
+
| `@aws-sdk/client-bedrock-runtime` | `BedrockBackend` (`bedrock/...` models) |
|
|
16
|
+
| `@cogitator-ai/sandbox` | Tools with `sandbox: { type: 'docker' \| 'wasm' }` |
|
|
17
|
+
| `pg` | `sqlQuery` / `vectorSearch` on PostgreSQL, `PostgresTraceStore` |
|
|
18
|
+
| `better-sqlite3` | `sqlQuery` on SQLite |
|
|
19
|
+
| `nodemailer` | `sendEmail` over SMTP |
|
|
20
|
+
| `langfuse` | `LangfuseExporter` |
|
|
21
|
+
|
|
22
|
+
Full documentation: [cogitator.app/docs](https://cogitator.app/docs) — start with [Agents](https://cogitator.app/docs/core/agents) and [Cogitator](https://cogitator.app/docs/core/cogitator).
|
|
23
|
+
|
|
11
24
|
## Quick Start
|
|
12
25
|
|
|
13
26
|
```typescript
|
|
@@ -48,10 +61,13 @@ console.log(result.output);
|
|
|
48
61
|
|
|
49
62
|
## Features
|
|
50
63
|
|
|
51
|
-
- **Multi-Provider LLM Support** - Ollama, OpenAI, Anthropic, Google, vLLM
|
|
52
|
-
- **Type-Safe Tools** - Zod-validated tool definitions
|
|
53
|
-
- **Streaming Responses** - Real-time token streaming
|
|
54
|
-
- **
|
|
64
|
+
- **Multi-Provider LLM Support** - Ollama, OpenAI, Anthropic, Google, Azure OpenAI, Bedrock, vLLM, Mistral, Groq, Together, DeepSeek
|
|
65
|
+
- **Type-Safe Tools** - Zod-validated tool definitions, `toolset()` for typed tool tuples
|
|
66
|
+
- **Streaming Responses** - Real-time token and reasoning streaming
|
|
67
|
+
- **Structured Output** - `responseFormat` with a Zod schema, validated and repaired once on mismatch
|
|
68
|
+
- **Handoffs & Approvals** - Pass a conversation to another agent; pause runs for human approval and resume them later
|
|
69
|
+
- **Prompt Versions & A/B Tests** - Versioned instructions per agent with `cog.prompts`
|
|
70
|
+
- **Memory Integration** - In-memory, Redis, PostgreSQL, SQLite, MongoDB and Qdrant adapters
|
|
55
71
|
- **26 Built-in Tools** - Web search, SQL, email, GitHub, filesystem, and more
|
|
56
72
|
- **Reflection Engine** - Self-improvement through tool call analysis
|
|
57
73
|
- **Tree-of-Thought** - Advanced reasoning with branch exploration
|
|
@@ -113,23 +129,30 @@ const cog = new Cogitator({
|
|
|
113
129
|
});
|
|
114
130
|
```
|
|
115
131
|
|
|
132
|
+
The runtime does not read provider keys from the environment: pass them under `providers` (or build the config with `loadConfig()` from `@cogitator-ai/config`, which reads `OPENAI_API_KEY`, `ANTHROPIC_API_KEY` and the rest). Ollama defaults to `http://localhost:11434`. A model string without a known provider prefix runs on `llm.defaultProvider` (Ollama when unset). An agent without `model` uses `llm.defaultModel`. `llm.backends` registers backends of your own by name (`model: 'my-backend/some-model'`), and `llm.retry` (2 retries with exponential backoff by default, `false` to disable) applies to every backend the runtime creates. Azure (`endpoint`, `apiKey`, `apiVersion`, `deployment`), Bedrock (`region`, credentials) and Mistral / Groq / Together / DeepSeek (`apiKey`) are configured the same way under `providers`.
|
|
133
|
+
|
|
134
|
+
See [LLM Backends](https://cogitator.app/docs/core/llm-backends) for every provider's options.
|
|
135
|
+
|
|
116
136
|
### Provider Notes
|
|
117
137
|
|
|
118
138
|
- **OpenAI** — the official backend uses the Responses API and defaults to `gpt-6.1-sol`. Requests are stateless (`store: false`); reasoning items are round-tripped between tool-call turns via `ToolCall.replay`. Reasoning models (o-series, GPT-5+) get no `temperature` / `top_p`. Requests with stop sequences fall back to Chat Completions (the Responses API has no stop parameter). OpenAI-compatible providers (Azure, Mistral, Groq, Together, DeepSeek, vLLM, custom `baseUrl`) stay on Chat Completions. Force either path with `providers.openai.api: 'responses' | 'chat-completions'`. Usage includes `cachedInputTokens` and `reasoningTokens` when reported.
|
|
119
139
|
- **Anthropic** — defaults to `claude-sonnet-5-5`. Sampling params are omitted for Claude 4.7+, 5.x and Fable (they reject non-default values); Claude 4.0 – 4.6 get at most one of `temperature` / `top_p` (`temperature` wins). `json_schema` uses native structured outputs on Claude 4.5+. Forced tool choice falls back to `auto` with a system-prompt instruction and a one-time warning on Opus/Sonnet 5.5 and Fable.
|
|
120
140
|
- **Bedrock** — Claude models follow the same sampling and tool-choice rules; `json_object` and `json_schema` response formats are supported (schema enforced via `outputConfig.textFormat` on Claude 4.5 – 4.6, system-prompt instruction otherwise).
|
|
141
|
+
- **Google** — Gemini has no `null` schema type, so `.nullable()` fields in response schemas and tool parameters are sent as `nullable: true`.
|
|
121
142
|
|
|
122
143
|
### Direct Backend Usage
|
|
123
144
|
|
|
124
145
|
```typescript
|
|
125
146
|
import { createLLMBackend, parseModel } from '@cogitator-ai/core';
|
|
126
147
|
|
|
127
|
-
const
|
|
128
|
-
|
|
148
|
+
const { provider, model } = parseModel('openai/gpt-6.1-sol'); // { provider: 'openai', model: 'gpt-6.1-sol' }
|
|
149
|
+
|
|
150
|
+
const backend = createLLMBackend(provider ?? 'ollama', {
|
|
151
|
+
providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } },
|
|
129
152
|
});
|
|
130
153
|
|
|
131
154
|
const response = await backend.chat({
|
|
132
|
-
model
|
|
155
|
+
model,
|
|
133
156
|
messages: [
|
|
134
157
|
{ role: 'system', content: 'You are helpful.' },
|
|
135
158
|
{ role: 'user', content: 'Hello!' },
|
|
@@ -158,6 +181,8 @@ registerLLMBackend(myPlugin);
|
|
|
158
181
|
const backend = createLLMBackendFromPlugin('my-provider', { apiKey: '...' });
|
|
159
182
|
```
|
|
160
183
|
|
|
184
|
+
A backend's `provider` field is an `LLMBackendProvider`: a built-in provider name or one of your own, such as the plugin's provider or its key in `llm.backends`.
|
|
185
|
+
|
|
161
186
|
### LLM Debug Wrapper
|
|
162
187
|
|
|
163
188
|
Wrap any backend for request/response logging:
|
|
@@ -174,25 +199,28 @@ const debugBackend = withDebug(backend, {
|
|
|
174
199
|
### LLM Error Handling
|
|
175
200
|
|
|
176
201
|
```typescript
|
|
177
|
-
import { LLMError
|
|
202
|
+
import { LLMError } from '@cogitator-ai/core';
|
|
178
203
|
|
|
179
204
|
try {
|
|
180
205
|
await backend.chat(request);
|
|
181
206
|
} catch (error) {
|
|
182
207
|
if (error instanceof LLMError) {
|
|
183
|
-
console.log('Provider:', error.provider);
|
|
184
|
-
console.log('
|
|
185
|
-
console.log('
|
|
208
|
+
console.log('Provider:', error.provider, error.model);
|
|
209
|
+
console.log('Code:', error.code); // ErrorCode, e.g. LLM_RATE_LIMITED
|
|
210
|
+
console.log('HTTP status from the provider:', error.details?.statusCode);
|
|
211
|
+
console.log('Retryable:', error.retryable, 'after', error.retryAfter, 'ms');
|
|
186
212
|
}
|
|
187
213
|
}
|
|
188
214
|
```
|
|
189
215
|
|
|
216
|
+
`llmUnavailable`, `llmTimeout`, `llmInvalidResponse`, `llmConfigError` and `wrapSDKError` build these errors in your own backends; `withLLMRetry(backend, options)` / `RetryingBackend` add retries with `Retry-After` support to any backend.
|
|
217
|
+
|
|
190
218
|
---
|
|
191
219
|
|
|
192
220
|
## Agent Configuration
|
|
193
221
|
|
|
194
222
|
```typescript
|
|
195
|
-
import { Agent } from '@cogitator-ai/core';
|
|
223
|
+
import { Agent, calculator, webSearch } from '@cogitator-ai/core';
|
|
196
224
|
|
|
197
225
|
const agent = new Agent({
|
|
198
226
|
id: 'custom-id',
|
|
@@ -207,6 +235,7 @@ const agent = new Agent({
|
|
|
207
235
|
maxIterations: 15,
|
|
208
236
|
timeout: 120_000,
|
|
209
237
|
stopSequences: ['DONE'],
|
|
238
|
+
// responseFormat, reasoning, handoffs, skills, description are covered below
|
|
210
239
|
});
|
|
211
240
|
|
|
212
241
|
// Clone with modifications
|
|
@@ -217,6 +246,63 @@ const variant = agent.clone({
|
|
|
217
246
|
});
|
|
218
247
|
```
|
|
219
248
|
|
|
249
|
+
### Skills and Serialization
|
|
250
|
+
|
|
251
|
+
A skill bundles tools with the instructions for using them; `skills` merges them into the agent:
|
|
252
|
+
|
|
253
|
+
```typescript
|
|
254
|
+
import { Agent, defineSkill, httpRequest, ToolRegistry } from '@cogitator-ai/core';
|
|
255
|
+
|
|
256
|
+
const apiSkill = defineSkill({
|
|
257
|
+
name: 'http-api',
|
|
258
|
+
version: '1.0.0',
|
|
259
|
+
description: 'Call JSON APIs',
|
|
260
|
+
tools: [httpRequest],
|
|
261
|
+
instructions: 'Prefer GET requests and summarize responses.',
|
|
262
|
+
env: ['API_TOKEN'], // checked by validateSkill()
|
|
263
|
+
});
|
|
264
|
+
|
|
265
|
+
const agent = new Agent({
|
|
266
|
+
name: 'integrator',
|
|
267
|
+
model: 'openai/gpt-5.5',
|
|
268
|
+
instructions: 'Answer with data from the API.',
|
|
269
|
+
skills: [apiSkill],
|
|
270
|
+
});
|
|
271
|
+
|
|
272
|
+
const snapshot = agent.serialize(); // plain JSON; tools are stored by name
|
|
273
|
+
const registry = new ToolRegistry();
|
|
274
|
+
registry.register(httpRequest);
|
|
275
|
+
const restored = Agent.deserialize(snapshot, { toolRegistry: registry });
|
|
276
|
+
```
|
|
277
|
+
|
|
278
|
+
See [Agents](https://cogitator.app/docs/core/agents).
|
|
279
|
+
|
|
280
|
+
### Structured Output
|
|
281
|
+
|
|
282
|
+
`responseFormat` asks the model for JSON. With a Zod schema the answer is validated and parsed into `result.structured`:
|
|
283
|
+
|
|
284
|
+
```typescript
|
|
285
|
+
import { Agent, Cogitator } from '@cogitator-ai/core';
|
|
286
|
+
import { z } from 'zod';
|
|
287
|
+
|
|
288
|
+
const Weather = z.object({ city: z.string(), celsius: z.number() });
|
|
289
|
+
|
|
290
|
+
const extractor = new Agent({
|
|
291
|
+
name: 'extractor',
|
|
292
|
+
model: 'openai/gpt-5.5',
|
|
293
|
+
instructions: 'Extract the weather report.',
|
|
294
|
+
responseFormat: { type: 'json_schema', schema: Weather }, // or { type: 'json' } for any JSON
|
|
295
|
+
});
|
|
296
|
+
|
|
297
|
+
const cog = new Cogitator({
|
|
298
|
+
llm: { providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } } },
|
|
299
|
+
});
|
|
300
|
+
const result = await cog.run(extractor, { input: 'Paris, 21 degrees' });
|
|
301
|
+
const weather = Weather.parse(result.structured);
|
|
302
|
+
```
|
|
303
|
+
|
|
304
|
+
When the final answer does not fit the schema, the run asks the model once more with the validation problem (for example `celsius: expected number, received string`) and does not save the rejected answer to the thread; if the retry fails too, `structured` is `undefined`. Streamed runs keep the first answer, since the client has already seen it. JSON wrapped in prose or code fences is still read. See [Structured Outputs](https://cogitator.app/docs/core/structured-outputs).
|
|
305
|
+
|
|
220
306
|
### Reasoning and Prompt Caching
|
|
221
307
|
|
|
222
308
|
`reasoning` sets how hard a reasoning model thinks, in one vocabulary for every provider (Anthropic adaptive thinking and effort, OpenAI `reasoning.effort`, Gemini thinking levels or budgets, Ollama `think`), and can ask for a readable summary:
|
|
@@ -266,6 +352,46 @@ const weatherTool = tool({
|
|
|
266
352
|
});
|
|
267
353
|
```
|
|
268
354
|
|
|
355
|
+
`tool()` also takes `category`, `tags`, `sideEffects`, `requiresApproval`, `timeout` and `sandbox`. A parameter with `.default()` is optional in the JSON Schema the model sees; `execute` receives the default when the model leaves it out. Tools passed to `new Agent({ tools })` lose their individual parameter types in a plain array; `toolset(...tools)` keeps them as a typed tuple that agents still accept:
|
|
356
|
+
|
|
357
|
+
```typescript
|
|
358
|
+
import { tool, toolset } from '@cogitator-ai/core';
|
|
359
|
+
import { z } from 'zod';
|
|
360
|
+
|
|
361
|
+
function createSearchTools() {
|
|
362
|
+
return toolset(
|
|
363
|
+
tool({
|
|
364
|
+
name: 'search',
|
|
365
|
+
description: 'Search the catalog',
|
|
366
|
+
parameters: z.object({ query: z.string() }),
|
|
367
|
+
execute: async ({ query }) => ({ hits: [query] }),
|
|
368
|
+
}),
|
|
369
|
+
tool({
|
|
370
|
+
name: 'fetch_item',
|
|
371
|
+
description: 'Fetch one item',
|
|
372
|
+
parameters: z.object({ id: z.number() }),
|
|
373
|
+
execute: async ({ id }) => ({ id }),
|
|
374
|
+
})
|
|
375
|
+
);
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
const [search, fetchItem] = createSearchTools();
|
|
379
|
+
await search.execute({ query: 'lamp' }, ctx); // typed as { query: string }
|
|
380
|
+
```
|
|
381
|
+
|
|
382
|
+
A result object with a base64 image in `image` or `imageBase64` (PNG, JPEG, GIF or WebP, plain or as a `data:` URL) reaches the model as an image, with the rest of the result as JSON, so a vision model sees a screenshot instead of its base64 text. Anthropic, Bedrock and the OpenAI Responses API get the image inside the tool result, Google after the turn's function responses, Ollama in the tool message's `images`, and Chat Completions backends (OpenAI-compatible, Azure) in a user message after the turn's tool messages:
|
|
383
|
+
|
|
384
|
+
```typescript
|
|
385
|
+
const screenshot = tool({
|
|
386
|
+
name: 'screenshot',
|
|
387
|
+
description: 'Capture the dashboard as a PNG',
|
|
388
|
+
parameters: z.object({}),
|
|
389
|
+
execute: async () => ({ page: 'dashboard', image: (await capture()).toString('base64') }),
|
|
390
|
+
});
|
|
391
|
+
```
|
|
392
|
+
|
|
393
|
+
See [Tools](https://cogitator.app/docs/core/tools) and [Custom Tools](https://cogitator.app/docs/tools/custom-tools).
|
|
394
|
+
|
|
269
395
|
### Handoffs
|
|
270
396
|
|
|
271
397
|
`handoffs` lets an agent pass the conversation to another one: each target becomes a `transfer_to_<name>` tool, and the rest of the run goes on as the target — its instructions, tools and model — with the whole conversation:
|
|
@@ -273,7 +399,7 @@ const weatherTool = tool({
|
|
|
273
399
|
```typescript
|
|
274
400
|
const triage = new Agent({
|
|
275
401
|
name: 'triage',
|
|
276
|
-
model,
|
|
402
|
+
model: 'openai/gpt-5.5',
|
|
277
403
|
instructions: 'Hand the customer to the right specialist.',
|
|
278
404
|
handoffs: [billing, techSupport],
|
|
279
405
|
});
|
|
@@ -283,6 +409,8 @@ result.handoffs; // [{ from: 'triage', to: 'billing', reason }]
|
|
|
283
409
|
result.finalAgent; // 'billing'
|
|
284
410
|
```
|
|
285
411
|
|
|
412
|
+
A handoff can also be `{ agent, toolName, description }` to name the tool or describe when to use it. `onHandoff` on the run options reports each switch as it happens.
|
|
413
|
+
|
|
286
414
|
### Approvals
|
|
287
415
|
|
|
288
416
|
A tool with `requiresApproval` (`true` or a function of its arguments) never runs without a person's decision. Decide inline with `onApproval`, or let the run pause and resume it later:
|
|
@@ -301,6 +429,8 @@ if (result.status === 'paused') {
|
|
|
301
429
|
|
|
302
430
|
Nothing of the paused turn runs until every call in it is decided. Paused runs live in the thread's memory (or process memory, or your `runCheckpoints` store), so a resume can come after a restart; a new message on the thread instead declines the waiting calls.
|
|
303
431
|
|
|
432
|
+
`cog.resume()` takes the thread id or the returned `result.checkpoint`; `defaultDecision` answers every call `decisions` leaves out. To decide while the run waits, pass `onApproval: (request) => ({ approved: true })` (or return `'pause'`) to `cog.run()`. See [Tool Approvals](https://cogitator.app/docs/tools/approvals).
|
|
433
|
+
|
|
304
434
|
### PII Masking
|
|
305
435
|
|
|
306
436
|
`security.pii` replaces emails, phones, card numbers (Luhn-checked), IBANs, SSNs, IP addresses, API keys and your own patterns with placeholders before every LLM request, so the provider never sees them:
|
|
@@ -317,7 +447,7 @@ const cog = new Cogitator({
|
|
|
317
447
|
});
|
|
318
448
|
```
|
|
319
449
|
|
|
320
|
-
In `mask` mode the answer, its stream and tool call arguments get the real values back, so `send_email({ to: '[EMAIL_1]' })` reaches the tool as the real address. `PiiMasker`, `PiiVault` and `withPiiMasking` work outside a run too.
|
|
450
|
+
In `mask` mode the answer, its stream and tool call arguments get the real values back, so `send_email({ to: '[EMAIL_1]' })` reaches the tool as the real address. `detect` limits the built-in kinds (`PII_TYPES`: `email`, `phone`, `credit_card`, `iban`, `ssn`, `ip_address`, `api_key`). `PiiMasker`, `PiiVault` and `withPiiMasking` work outside a run too. See [Security](https://cogitator.app/docs/advanced/security).
|
|
321
451
|
|
|
322
452
|
### Tool Context
|
|
323
453
|
|
|
@@ -327,7 +457,11 @@ Every tool receives a context object:
|
|
|
327
457
|
interface ToolContext {
|
|
328
458
|
agentId: string;
|
|
329
459
|
runId: string;
|
|
330
|
-
signal: AbortSignal;
|
|
460
|
+
signal: AbortSignal; // aborted on run cancel or tool timeout
|
|
461
|
+
threadId?: string;
|
|
462
|
+
userId?: string; // the run's userId
|
|
463
|
+
channelType?: string;
|
|
464
|
+
channelId?: string;
|
|
331
465
|
}
|
|
332
466
|
```
|
|
333
467
|
|
|
@@ -336,6 +470,9 @@ interface ToolContext {
|
|
|
336
470
|
Execute tools in isolated Docker or WASM environments:
|
|
337
471
|
|
|
338
472
|
```typescript
|
|
473
|
+
import { tool } from '@cogitator-ai/core';
|
|
474
|
+
import { z } from 'zod';
|
|
475
|
+
|
|
339
476
|
const shellTool = tool({
|
|
340
477
|
name: 'run_shell',
|
|
341
478
|
description: 'Execute shell commands safely',
|
|
@@ -351,6 +488,10 @@ const shellTool = tool({
|
|
|
351
488
|
});
|
|
352
489
|
```
|
|
353
490
|
|
|
491
|
+
A Docker-sandboxed tool does not call `execute`: the sandbox runs the `command` argument with `sh -c` (plus optional `cwd` / `env` arguments) and returns its output. A WASM tool gets its arguments as JSON on stdin and its JSON stdout is parsed as the result. Sandboxing needs `@cogitator-ai/sandbox` installed (options go in `new Cogitator({ sandbox })`).
|
|
492
|
+
|
|
493
|
+
When Docker is unavailable (or `@cogitator-ai/sandbox` is missing), a Docker-sandboxed tool runs its command directly on the host with a warning; set `sandbox.allowNativeFallback: false` to make those calls fail instead. WASM tools never fall back to the host: they run their own `execute` only when `@cogitator-ai/sandbox` is missing or fails to start, and a WASM sandbox that cannot load the module returns an error. Each Docker execution gets a fresh container (the pool keeps them warm); `sandbox.pool.reuseContainers: true` reuses containers between executions with the same settings, which is faster but lets files and processes leak from one execution to the next.
|
|
494
|
+
|
|
354
495
|
`timeout` is enforced for every tool: native tools get an aborted `context.signal` and the model receives a `Tool "<name>" timed out after <ms>ms` error; sandboxed tools forward it to the sandbox executor. The sandbox is initialized lazily on the first sandboxed call, and that call already runs inside it.
|
|
355
496
|
|
|
356
497
|
### Tool Registry
|
|
@@ -433,25 +574,38 @@ Not part of `builtinTools`. `createAnalyzeImageTool` takes an `llm` backend; the
|
|
|
433
574
|
| `createGenerateSpeechTool` | `generateSpeech` | `gpt-4o-mini-tts` |
|
|
434
575
|
|
|
435
576
|
```typescript
|
|
436
|
-
import {
|
|
577
|
+
import { Agent, builtinTools } from '@cogitator-ai/core';
|
|
437
578
|
|
|
438
579
|
const agent = new Agent({
|
|
439
580
|
name: 'utility-agent',
|
|
440
581
|
instructions: 'Use your tools to help users',
|
|
441
582
|
model: 'openai/gpt-6.1-sol',
|
|
442
|
-
tools: builtinTools,
|
|
583
|
+
tools: builtinTools, // Tool[]
|
|
443
584
|
});
|
|
444
585
|
```
|
|
445
586
|
|
|
587
|
+
#### Assistant Tool Factories
|
|
588
|
+
|
|
589
|
+
Also not in `builtinTools`, these build tools around your own stores. `createMemoryTools` and `createSchedulerTools` return typed tuples (see `toolset()`), so destructured tools keep their parameter types:
|
|
590
|
+
|
|
591
|
+
| Factory | Tools |
|
|
592
|
+
| -------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- |
|
|
593
|
+
| `createMemoryTools({ graphAdapter, agentId, coreFacts?, embeddingFn? })` | `remember`, `recall`, `forget` on a knowledge graph |
|
|
594
|
+
| `createSchedulerTools({ store, defaultChannel?, defaultUserId? })` | `schedule_task`, `list_tasks`, `cancel_task` on a `TimerStore` |
|
|
595
|
+
| `createDeviceTools()`, `createCapabilitiesTool(doc)`, `createSelfTools({ toolsDir })` / `loadCustomTools(dir)` | Device info, a capabilities description, and tools an agent writes to a directory |
|
|
596
|
+
|
|
446
597
|
### Web Search Tool
|
|
447
598
|
|
|
448
599
|
Search the web using Tavily, Brave, or Serper APIs:
|
|
449
600
|
|
|
450
601
|
```typescript
|
|
451
|
-
import { webSearch } from '@cogitator-ai/core';
|
|
602
|
+
import { Agent, webSearch } from '@cogitator-ai/core';
|
|
452
603
|
|
|
453
604
|
// Auto-detects from TAVILY_API_KEY, BRAVE_API_KEY, or SERPER_API_KEY
|
|
454
605
|
const agent = new Agent({
|
|
606
|
+
name: 'assistant',
|
|
607
|
+
instructions: 'Use your tools to help users',
|
|
608
|
+
model: 'openai/gpt-5.5',
|
|
455
609
|
tools: [webSearch],
|
|
456
610
|
});
|
|
457
611
|
|
|
@@ -464,13 +618,16 @@ const agent = new Agent({
|
|
|
464
618
|
Extract content from web pages:
|
|
465
619
|
|
|
466
620
|
```typescript
|
|
467
|
-
import { webScrape } from '@cogitator-ai/core';
|
|
621
|
+
import { Agent, webScrape } from '@cogitator-ai/core';
|
|
468
622
|
|
|
469
623
|
const agent = new Agent({
|
|
624
|
+
name: 'assistant',
|
|
625
|
+
instructions: 'Use your tools to help users',
|
|
626
|
+
model: 'openai/gpt-5.5',
|
|
470
627
|
tools: [webScrape],
|
|
471
628
|
});
|
|
472
629
|
|
|
473
|
-
// Supports
|
|
630
|
+
// Supports simple selectors (tag, .class, #id), text/markdown/html output, link/image extraction
|
|
474
631
|
```
|
|
475
632
|
|
|
476
633
|
### SQL Query Tool
|
|
@@ -478,9 +635,12 @@ const agent = new Agent({
|
|
|
478
635
|
Execute SQL queries against PostgreSQL or SQLite:
|
|
479
636
|
|
|
480
637
|
```typescript
|
|
481
|
-
import { sqlQuery } from '@cogitator-ai/core';
|
|
638
|
+
import { Agent, sqlQuery } from '@cogitator-ai/core';
|
|
482
639
|
|
|
483
640
|
const agent = new Agent({
|
|
641
|
+
name: 'assistant',
|
|
642
|
+
instructions: 'Use your tools to help users',
|
|
643
|
+
model: 'openai/gpt-5.5',
|
|
484
644
|
tools: [sqlQuery],
|
|
485
645
|
});
|
|
486
646
|
|
|
@@ -498,14 +658,17 @@ Read-only queries on PostgreSQL run inside `BEGIN TRANSACTION READ ONLY` (always
|
|
|
498
658
|
Semantic search using embeddings with pgvector:
|
|
499
659
|
|
|
500
660
|
```typescript
|
|
501
|
-
import { vectorSearch } from '@cogitator-ai/core';
|
|
661
|
+
import { Agent, vectorSearch } from '@cogitator-ai/core';
|
|
502
662
|
|
|
503
663
|
const agent = new Agent({
|
|
664
|
+
name: 'assistant',
|
|
665
|
+
instructions: 'Use your tools to help users',
|
|
666
|
+
model: 'openai/gpt-5.5',
|
|
504
667
|
tools: [vectorSearch],
|
|
505
668
|
});
|
|
506
669
|
|
|
507
670
|
// Embedding providers: OpenAI, Ollama, Google
|
|
508
|
-
// Auto-detects from OPENAI_API_KEY, OLLAMA_BASE_URL / OLLAMA_HOST, or GOOGLE_API_KEY
|
|
671
|
+
// Auto-detects from OPENAI_API_KEY, OLLAMA_BASE_URL / OLLAMA_URL / OLLAMA_HOST, or GOOGLE_API_KEY
|
|
509
672
|
// Default models: text-embedding-3-small, nomic-embed-text, gemini-embedding-001
|
|
510
673
|
```
|
|
511
674
|
|
|
@@ -516,9 +679,12 @@ const agent = new Agent({
|
|
|
516
679
|
Send emails via Resend API or SMTP:
|
|
517
680
|
|
|
518
681
|
```typescript
|
|
519
|
-
import { sendEmail } from '@cogitator-ai/core';
|
|
682
|
+
import { Agent, sendEmail } from '@cogitator-ai/core';
|
|
520
683
|
|
|
521
684
|
const agent = new Agent({
|
|
685
|
+
name: 'assistant',
|
|
686
|
+
instructions: 'Use your tools to help users',
|
|
687
|
+
model: 'openai/gpt-5.5',
|
|
522
688
|
tools: [sendEmail],
|
|
523
689
|
});
|
|
524
690
|
|
|
@@ -534,9 +700,12 @@ const agent = new Agent({
|
|
|
534
700
|
Interact with GitHub repositories:
|
|
535
701
|
|
|
536
702
|
```typescript
|
|
537
|
-
import { githubApi } from '@cogitator-ai/core';
|
|
703
|
+
import { Agent, githubApi } from '@cogitator-ai/core';
|
|
538
704
|
|
|
539
705
|
const agent = new Agent({
|
|
706
|
+
name: 'assistant',
|
|
707
|
+
instructions: 'Use your tools to help users',
|
|
708
|
+
model: 'openai/gpt-5.5',
|
|
540
709
|
tools: [githubApi],
|
|
541
710
|
});
|
|
542
711
|
|
|
@@ -553,8 +722,15 @@ const agent = new Agent({
|
|
|
553
722
|
```typescript
|
|
554
723
|
import { Cogitator, Agent } from '@cogitator-ai/core';
|
|
555
724
|
|
|
556
|
-
const cog = new Cogitator(
|
|
557
|
-
|
|
725
|
+
const cog = new Cogitator({
|
|
726
|
+
llm: { providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } } },
|
|
727
|
+
});
|
|
728
|
+
const agent = new Agent({
|
|
729
|
+
name: 'analyst',
|
|
730
|
+
instructions: 'Analyze data.',
|
|
731
|
+
model: 'openai/gpt-5.5',
|
|
732
|
+
});
|
|
733
|
+
const controller = new AbortController();
|
|
558
734
|
|
|
559
735
|
const result = await cog.run(agent, {
|
|
560
736
|
input: 'Analyze this data...',
|
|
@@ -562,16 +738,24 @@ const result = await cog.run(agent, {
|
|
|
562
738
|
audio: [{ data: base64Wav, format: 'wav' }],
|
|
563
739
|
|
|
564
740
|
threadId: 'thread_123',
|
|
565
|
-
|
|
741
|
+
userId: 'user_456', // owns the thread, scopes memory, reaches tools as context.userId
|
|
742
|
+
threadAccess: 'owner', // default; 'shared' lets any user continue the thread
|
|
743
|
+
context: { task: 'analysis' }, // extra values for the system prompt
|
|
566
744
|
|
|
567
745
|
timeout: 60000,
|
|
746
|
+
signal: controller.signal,
|
|
568
747
|
stream: true,
|
|
569
748
|
onToken: (token) => process.stdout.write(token),
|
|
749
|
+
onReasoning: (delta) => process.stdout.write(delta),
|
|
750
|
+
reasoning: { effort: 'low' }, // overrides the agent's reasoning for this run
|
|
570
751
|
|
|
571
752
|
useMemory: true,
|
|
572
753
|
loadHistory: true,
|
|
573
754
|
saveHistory: true,
|
|
755
|
+
parallelToolCalls: false, // default: tool calls of one turn run one after another
|
|
574
756
|
|
|
757
|
+
onApproval: (request) => ({ approved: request.toolName !== 'delete_account' }),
|
|
758
|
+
onHandoff: (handoff) => console.log(`${handoff.from} -> ${handoff.to}`),
|
|
575
759
|
onToolCall: (call) => console.log('Tool:', call.name),
|
|
576
760
|
onToolResult: (result) => console.log('Result:', result.result),
|
|
577
761
|
onSpan: (span) => console.log('Span:', span.name),
|
|
@@ -589,16 +773,28 @@ const result = await cog.run(agent, {
|
|
|
589
773
|
```typescript
|
|
590
774
|
interface RunResult {
|
|
591
775
|
output: string;
|
|
776
|
+
structured?: unknown; // parsed output when the agent has a responseFormat
|
|
592
777
|
runId: string;
|
|
593
778
|
agentId: string;
|
|
594
779
|
threadId: string;
|
|
780
|
+
modelUsed?: string; // differs from agent.model when cost routing picked another
|
|
595
781
|
usage: {
|
|
596
782
|
inputTokens: number;
|
|
597
783
|
outputTokens: number;
|
|
598
784
|
totalTokens: number;
|
|
599
785
|
cost: number;
|
|
600
786
|
duration: number;
|
|
787
|
+
reasoningTokens?: number;
|
|
788
|
+
cachedInputTokens?: number;
|
|
789
|
+
cacheWriteTokens?: number;
|
|
601
790
|
};
|
|
791
|
+
reasoning?: string; // reasoning summary, with reasoning.summary
|
|
792
|
+
prompt?: RunPrompt; // versioned instructions / A/B variant used
|
|
793
|
+
handoffs?: HandoffEvent[];
|
|
794
|
+
finalAgent?: string;
|
|
795
|
+
status?: 'completed' | 'paused';
|
|
796
|
+
pendingApprovals?: ToolApprovalRequest[];
|
|
797
|
+
checkpoint?: RunCheckpoint; // pass to cog.resume()
|
|
602
798
|
toolCalls: ToolCall[];
|
|
603
799
|
messages: Message[];
|
|
604
800
|
trace: { traceId: string; spans: Span[] };
|
|
@@ -607,6 +803,8 @@ interface RunResult {
|
|
|
607
803
|
}
|
|
608
804
|
```
|
|
609
805
|
|
|
806
|
+
All fields are `readonly`. The run timeout comes from the run, the agent, `limits.defaultTimeout`, or 120 s.
|
|
807
|
+
|
|
610
808
|
---
|
|
611
809
|
|
|
612
810
|
## Memory Integration
|
|
@@ -649,6 +847,21 @@ const cog = new Cogitator({
|
|
|
649
847
|
});
|
|
650
848
|
```
|
|
651
849
|
|
|
850
|
+
`adapter` also accepts `'sqlite'` (`sqlite.path`), `'mongodb'` (`mongodb.uri`) and `'redis'` (`redis.url`, or `redis.host` + `redis.port`, or `redis.cluster`). Qdrant stores embeddings, not threads: configure it in `memory.qdrant` next to a thread adapter, together with `memory.embedding` and `memory.contextBuilder.includeSemanticContext`, and semantic context is retrieved from it. The Postgres adapter sizes its vector column to the `memory.embedding` model. Runs with a `threadId` load history from and save messages to the adapter.
|
|
851
|
+
|
|
852
|
+
The adapter connects on the first run. To read threads before that (for example in an API route), use `getMemory()`, which connects it on first use; `cog.memory` stays `undefined` until something connected it:
|
|
853
|
+
|
|
854
|
+
```typescript
|
|
855
|
+
import { unwrap } from '@cogitator-ai/memory';
|
|
856
|
+
|
|
857
|
+
const memory = await cog.getMemory(); // undefined when memory is not configured
|
|
858
|
+
if (memory) {
|
|
859
|
+
const entries = unwrap(await memory.getEntries({ threadId: 'thread_123', limit: 20 }));
|
|
860
|
+
}
|
|
861
|
+
```
|
|
862
|
+
|
|
863
|
+
See [Memory](https://cogitator.app/docs/memory) and [Memory Adapters](https://cogitator.app/docs/memory/adapters).
|
|
864
|
+
|
|
652
865
|
---
|
|
653
866
|
|
|
654
867
|
## Reflection Engine
|
|
@@ -656,7 +869,7 @@ const cog = new Cogitator({
|
|
|
656
869
|
Enable self-improvement through reflection on tool calls and runs:
|
|
657
870
|
|
|
658
871
|
```typescript
|
|
659
|
-
import { Cogitator
|
|
872
|
+
import { Cogitator } from '@cogitator-ai/core';
|
|
660
873
|
|
|
661
874
|
const cog = new Cogitator({
|
|
662
875
|
reflection: {
|
|
@@ -683,13 +896,16 @@ console.log('Learned insights:', insights);
|
|
|
683
896
|
```typescript
|
|
684
897
|
import { ReflectionEngine, InMemoryInsightStore, createLLMBackend } from '@cogitator-ai/core';
|
|
685
898
|
|
|
686
|
-
const backend = createLLMBackend('openai', {
|
|
899
|
+
const backend = createLLMBackend('openai', {
|
|
900
|
+
providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } },
|
|
901
|
+
});
|
|
687
902
|
const insightStore = new InMemoryInsightStore();
|
|
688
903
|
|
|
689
904
|
const engine = new ReflectionEngine({
|
|
690
905
|
llm: backend,
|
|
691
906
|
insightStore,
|
|
692
907
|
config: {
|
|
908
|
+
enabled: true,
|
|
693
909
|
reflectAfterToolCall: true,
|
|
694
910
|
minConfidenceToStore: 0.7,
|
|
695
911
|
},
|
|
@@ -701,6 +917,8 @@ if (result.shouldAdjustStrategy) {
|
|
|
701
917
|
}
|
|
702
918
|
```
|
|
703
919
|
|
|
920
|
+
`reflectOnError`, `reflectOnRun`, `getRelevantInsights` and `getSummary(agentId)` cover the other stages. See [Reflection](https://cogitator.app/docs/advanced/reflection).
|
|
921
|
+
|
|
704
922
|
---
|
|
705
923
|
|
|
706
924
|
## Tree-of-Thought Reasoning
|
|
@@ -708,38 +926,49 @@ if (result.shouldAdjustStrategy) {
|
|
|
708
926
|
Explore multiple reasoning paths before deciding:
|
|
709
927
|
|
|
710
928
|
```typescript
|
|
711
|
-
import { ThoughtTreeExecutor
|
|
929
|
+
import { ThoughtTreeExecutor } from '@cogitator-ai/core';
|
|
712
930
|
|
|
713
|
-
const executor = new ThoughtTreeExecutor(
|
|
714
|
-
|
|
931
|
+
const executor = new ThoughtTreeExecutor(cog, {
|
|
932
|
+
branchFactor: 3,
|
|
715
933
|
maxDepth: 3,
|
|
716
934
|
explorationStrategy: 'best-first',
|
|
717
|
-
|
|
935
|
+
confidenceThreshold: 0.3,
|
|
718
936
|
});
|
|
719
937
|
|
|
720
|
-
const result = await executor.
|
|
721
|
-
|
|
722
|
-
|
|
938
|
+
const result = await executor.explore(agent, 'Solve this complex problem...', {
|
|
939
|
+
timeout: 60_000,
|
|
940
|
+
onProgress: (stats) => console.log('Explored:', stats.exploredNodes),
|
|
723
941
|
});
|
|
724
942
|
|
|
725
|
-
console.log('
|
|
726
|
-
console.log(
|
|
943
|
+
console.log('Answer:', result.output);
|
|
944
|
+
console.log(
|
|
945
|
+
'Best path:',
|
|
946
|
+
result.bestPath.map((node) => node.branch.thought)
|
|
947
|
+
);
|
|
948
|
+
console.log('Nodes in tree:', result.tree.nodes.size);
|
|
727
949
|
console.log('Stats:', result.stats);
|
|
728
950
|
```
|
|
729
951
|
|
|
730
952
|
### ToT Configuration
|
|
731
953
|
|
|
954
|
+
Every field is optional; defaults are in `DEFAULT_TOT_CONFIG`:
|
|
955
|
+
|
|
732
956
|
```typescript
|
|
733
|
-
const executor = new ThoughtTreeExecutor(
|
|
734
|
-
|
|
735
|
-
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
|
|
739
|
-
|
|
957
|
+
const executor = new ThoughtTreeExecutor(cog, {
|
|
958
|
+
branchFactor: 3, // branches generated per node
|
|
959
|
+
beamWidth: 2, // best candidates queued per expanded node
|
|
960
|
+
maxDepth: 5,
|
|
961
|
+
explorationStrategy: 'beam', // 'beam' | 'best-first' | 'dfs'
|
|
962
|
+
confidenceThreshold: 0.3, // prune branches scored below
|
|
963
|
+
terminationConfidence: 0.8, // stop once a branch reaches this
|
|
964
|
+
maxTotalNodes: 50, // executed nodes
|
|
965
|
+
maxIterationsPerBranch: 3, // iteration cap for each branch run
|
|
966
|
+
onBranchEvaluated: (branch, score) => console.log(branch.thought, score.composite),
|
|
740
967
|
});
|
|
741
968
|
```
|
|
742
969
|
|
|
970
|
+
Candidates beyond `beamWidth` stay pending, so a failed branch backtracks to the next best one. `beam` runs the tree level by level, `best-first` always runs the node whose own branch scored highest, and `dfs` goes deep first. The executor's own generation, evaluation and synthesis calls use the agent's routed model and count into `usage` (tokens and cost) and `stats`. See [Tree-of-Thought](https://cogitator.app/docs/advanced/reasoning).
|
|
971
|
+
|
|
743
972
|
---
|
|
744
973
|
|
|
745
974
|
## Agent Optimizer (Learning)
|
|
@@ -751,10 +980,10 @@ import { AgentOptimizer, InMemoryTraceStore } from '@cogitator-ai/core';
|
|
|
751
980
|
|
|
752
981
|
const traceStore = new InMemoryTraceStore();
|
|
753
982
|
const optimizer = new AgentOptimizer({
|
|
754
|
-
llm:
|
|
983
|
+
llm: cog.getLLMBackend('openai/gpt-6.1-sol'),
|
|
755
984
|
model: 'gpt-6.1-sol',
|
|
756
985
|
traceStore,
|
|
757
|
-
cogitator,
|
|
986
|
+
cogitator: cog,
|
|
758
987
|
});
|
|
759
988
|
|
|
760
989
|
const result = await optimizer.compile(
|
|
@@ -779,35 +1008,65 @@ import {
|
|
|
779
1008
|
createSuccessMetric,
|
|
780
1009
|
createExactMatchMetric,
|
|
781
1010
|
createContainsMetric,
|
|
782
|
-
MetricEvaluator,
|
|
783
1011
|
} from '@cogitator-ai/core';
|
|
784
1012
|
|
|
785
|
-
const successMetric = createSuccessMetric();
|
|
786
|
-
|
|
787
|
-
const
|
|
788
|
-
|
|
789
|
-
const containsMetric = createContainsMetric(['error', 'failed'], { negate: true });
|
|
1013
|
+
const successMetric = createSuccessMetric(); // 1 when no tool call failed
|
|
1014
|
+
const exactMatch = createExactMatchMetric(); // compares the output (or a field path) with `expected`
|
|
1015
|
+
const containsMetric = createContainsMetric(['refund', 'order']); // share of keywords found, passes at >= 0.5
|
|
790
1016
|
```
|
|
791
1017
|
|
|
1018
|
+
Each is a `MetricFn` (`(trace, expected?) => MetricResult`); `MetricEvaluator` combines built-in and custom metrics, including LLM-judged ones.
|
|
1019
|
+
|
|
792
1020
|
### Demo Selection
|
|
793
1021
|
|
|
794
1022
|
```typescript
|
|
795
|
-
import { DemoSelector } from '@cogitator-ai/core';
|
|
1023
|
+
import { DemoSelector, InMemoryTraceStore } from '@cogitator-ai/core';
|
|
796
1024
|
|
|
797
1025
|
const selector = new DemoSelector({
|
|
798
|
-
|
|
799
|
-
maxDemos: 5,
|
|
1026
|
+
traceStore: new InMemoryTraceStore(),
|
|
1027
|
+
maxDemos: 5, // per agent (default 10)
|
|
1028
|
+
minScore: 0.8, // traces scoring lower are not used as demos
|
|
1029
|
+
diversityWeight: 0.3,
|
|
800
1030
|
});
|
|
801
1031
|
|
|
802
|
-
|
|
1032
|
+
await selector.addDemo(trace);
|
|
1033
|
+
const demos = await selector.selectDemos(agent.id, currentInput, 3);
|
|
1034
|
+
const fewShot = selector.formatDemosForPrompt(demos);
|
|
803
1035
|
```
|
|
804
1036
|
|
|
1037
|
+
See [Learning](https://cogitator.app/docs/advanced/learning).
|
|
1038
|
+
|
|
805
1039
|
---
|
|
806
1040
|
|
|
807
1041
|
## Prompt Auto-Optimization
|
|
808
1042
|
|
|
809
1043
|
Capture prompts, run A/B tests, monitor performance, and automatically optimize agent instructions.
|
|
810
1044
|
|
|
1045
|
+
### Prompt Versions
|
|
1046
|
+
|
|
1047
|
+
`cog.prompts` versions an agent's instructions on top of the ones in code. A deployed version is used by every following run of that agent, and each run's outcome is recorded against the version (or A/B variant) it used:
|
|
1048
|
+
|
|
1049
|
+
```typescript
|
|
1050
|
+
const v2 = await cog.prompts.deploy(writer, 'You write release notes. Lead with what changed.');
|
|
1051
|
+
|
|
1052
|
+
const result = await cog.run(writer, { input, threadId });
|
|
1053
|
+
result.prompt; // { key: 'writer', versionId: v2.id, version: 2 }
|
|
1054
|
+
|
|
1055
|
+
await cog.prompts.rollbackTo(writer); // previous version, recorded as a new one
|
|
1056
|
+
await cog.prompts.history(writer); // newest first, with metrics
|
|
1057
|
+
|
|
1058
|
+
await cog.prompts.startABTest(writer, {
|
|
1059
|
+
name: 'shorter notes',
|
|
1060
|
+
treatment: 'You write release notes in at most five bullet points.',
|
|
1061
|
+
treatmentAllocation: 0.3, // share of threads
|
|
1062
|
+
minSampleSize: 50,
|
|
1063
|
+
});
|
|
1064
|
+
```
|
|
1065
|
+
|
|
1066
|
+
Versions are kept per agent `id` when set, else per `name`. They live in process memory unless `new Cogitator({ prompts: { versions, abTests } })` gets durable stores (`PostgresTraceStore` provides both via `instructionVersions()` and `abTests()`); `prompts.score` scores runs (default: 1 for a completed run) and `prompts.autoDeployWinner` deploys a significant A/B winner. See [Prompt Versions](https://cogitator.app/docs/advanced/prompt-versions).
|
|
1067
|
+
|
|
1068
|
+
The classes below are the building blocks `cog.prompts` uses, available for your own pipelines.
|
|
1069
|
+
|
|
811
1070
|
### Prompt Logger
|
|
812
1071
|
|
|
813
1072
|
Wrap any LLM backend to capture all prompts:
|
|
@@ -819,11 +1078,14 @@ const store = new PostgresTraceStore({
|
|
|
819
1078
|
connectionString: process.env.DATABASE_URL!,
|
|
820
1079
|
});
|
|
821
1080
|
|
|
1081
|
+
await store.connect();
|
|
1082
|
+
|
|
822
1083
|
const wrappedBackend = wrapWithPromptLogger(openaiBackend, store, {
|
|
823
|
-
|
|
1084
|
+
captureContent: true,
|
|
824
1085
|
captureTools: true,
|
|
825
|
-
captureResponse: true,
|
|
826
1086
|
});
|
|
1087
|
+
|
|
1088
|
+
wrappedBackend.setContext({ runId, agentId: agent.id, threadId });
|
|
827
1089
|
```
|
|
828
1090
|
|
|
829
1091
|
### A/B Testing Framework
|
|
@@ -880,9 +1142,9 @@ const monitor = new PromptMonitor({
|
|
|
880
1142
|
|
|
881
1143
|
const alerts = monitor.recordExecution(trace);
|
|
882
1144
|
|
|
883
|
-
const metrics = monitor.getCurrentMetrics('agent-1');
|
|
884
|
-
console.log('Avg score:', metrics
|
|
885
|
-
console.log('P95 latency:', metrics
|
|
1145
|
+
const metrics = monitor.getCurrentMetrics('agent-1'); // null before any execution
|
|
1146
|
+
console.log('Avg score:', metrics?.avgScore);
|
|
1147
|
+
console.log('P95 latency:', metrics?.p95Latency);
|
|
886
1148
|
```
|
|
887
1149
|
|
|
888
1150
|
### Rollback Manager
|
|
@@ -937,6 +1199,8 @@ const optimizer = new AutoOptimizer({
|
|
|
937
1199
|
await optimizer.recordExecution(trace);
|
|
938
1200
|
```
|
|
939
1201
|
|
|
1202
|
+
It optimizes the agent's deployed version, so deploy one first. To serve its A/B tests on live traffic, give `new Cogitator({ prompts: { abTests, versions } })` the same stores as `abTesting` and `rollbackManager` and the agent an explicit `id`: the Cogitator then assigns variants per thread and records the results, traces carry the variant (`trace.prompt`) so the optimizer does not count them twice, and the optimizer deploys the winner (leave `prompts.autoDeployWinner` off). `PostgresTraceStore` backs all of it: `traces()` for `AgentOptimizer`, `abTests()` and `instructionVersions()` for the rest. See [Learning](https://cogitator.app/docs/advanced/learning#ab-tests-on-live-traffic).
|
|
1203
|
+
|
|
940
1204
|
---
|
|
941
1205
|
|
|
942
1206
|
## Time Travel Debugging
|
|
@@ -949,10 +1213,10 @@ import { TimeTravel, InMemoryCheckpointStore } from '@cogitator-ai/core';
|
|
|
949
1213
|
const timeTravel = new TimeTravel(cogitator);
|
|
950
1214
|
|
|
951
1215
|
const result = await cogitator.run(agent, { input: 'Original task...' });
|
|
952
|
-
const checkpoints = await timeTravel.checkpointAll(result, 'original');
|
|
1216
|
+
const checkpoints = await timeTravel.checkpointAll(result, 'original'); // one per tool call
|
|
953
1217
|
|
|
954
1218
|
const replayResult = await timeTravel.replayLive(agent, checkpoints[2].id);
|
|
955
|
-
console.log('Replayed from
|
|
1219
|
+
console.log('Replayed from the third tool call:', replayResult.output);
|
|
956
1220
|
|
|
957
1221
|
const forkResult = await timeTravel.fork(agent, checkpoints[2].id, {
|
|
958
1222
|
input: 'Modified task...',
|
|
@@ -1006,7 +1270,7 @@ const liveReplay = await timeTravel.replayLive(agent, checkpointId, {
|
|
|
1006
1270
|
});
|
|
1007
1271
|
```
|
|
1008
1272
|
|
|
1009
|
-
Mocked tool results are keyed by tool name (in deterministic replays also by call id): the tool answers with the given value and never runs. Tools in `skipTools` are removed from the replayed agent. Checkpoints, replays and forks store their traces in the trace store, so `compare()` and `compareWithOriginal()` can read them.
|
|
1273
|
+
Checkpoints are anchored on tool calls: checkpoint `i` holds the conversation after `i` tool calls, stopping before the result of the call it is anchored on. `stepsReplayed` is that index, `stepsExecuted` counts the replay's tool calls, and `divergedAt` uses the original run's numbering. Mocked tool results are keyed by tool name (in deterministic replays also by call id): the tool answers with the given value and never runs. Tools in `skipTools` are removed from the replayed agent. Checkpoints, replays and forks store their traces in the trace store, so `compare()` and `compareWithOriginal()` can read them.
|
|
1010
1274
|
|
|
1011
1275
|
---
|
|
1012
1276
|
|
|
@@ -1043,8 +1307,8 @@ const engine = new CausalInferenceEngine(graph);
|
|
|
1043
1307
|
```typescript
|
|
1044
1308
|
const identifiable = engine.isIdentifiable('X', 'Y');
|
|
1045
1309
|
if (identifiable.identifiable) {
|
|
1046
|
-
console.log('Effect is identifiable
|
|
1047
|
-
console.log('Adjustment set:', identifiable.adjustmentSet);
|
|
1310
|
+
console.log('Effect is identifiable:', identifiable.reason); // e.g. backdoor criterion
|
|
1311
|
+
console.log('Adjustment set:', identifiable.adjustmentSet?.variables);
|
|
1048
1312
|
}
|
|
1049
1313
|
```
|
|
1050
1314
|
|
|
@@ -1054,11 +1318,11 @@ if (identifiable.identifiable) {
|
|
|
1054
1318
|
const effect = engine.computeInterventionalEffect({
|
|
1055
1319
|
target: 'Y',
|
|
1056
1320
|
interventions: { X: 1 },
|
|
1057
|
-
|
|
1321
|
+
conditions: { Z: 0.5 },
|
|
1058
1322
|
});
|
|
1059
1323
|
|
|
1060
1324
|
console.log('Expected effect:', effect.effect);
|
|
1061
|
-
console.log('
|
|
1325
|
+
console.log('Formula:', effect.formula, 'identifiable:', effect.isIdentifiable);
|
|
1062
1326
|
```
|
|
1063
1327
|
|
|
1064
1328
|
### Counterfactual Reasoning
|
|
@@ -1077,6 +1341,8 @@ console.log('Factual value:', result.factualValue);
|
|
|
1077
1341
|
console.log('Counterfactual value:', result.counterfactualValue);
|
|
1078
1342
|
```
|
|
1079
1343
|
|
|
1344
|
+
Counterfactuals use the nodes' structural equations (`withEquation()` on the builder): `linear`, `logistic`, `polynomial`, or `custom` with a safe arithmetic expression over the parent ids in `customFn` (`'2 * price - log(demand)'`; `+ - * / ^`, parentheses, `abs exp log sqrt pow min max tanh sigmoid`). Nodes without an equation keep their factual values.
|
|
1345
|
+
|
|
1080
1346
|
### D-Separation Analysis
|
|
1081
1347
|
|
|
1082
1348
|
```typescript
|
|
@@ -1098,16 +1364,19 @@ import { CausalExtractor, CausalHypothesisGenerator } from '@cogitator-ai/core';
|
|
|
1098
1364
|
|
|
1099
1365
|
const extractor = new CausalExtractor({ llmBackend: backend });
|
|
1100
1366
|
|
|
1101
|
-
const
|
|
1102
|
-
{ name: 'database_query', arguments: { table: 'users' } },
|
|
1367
|
+
const { nodes, edges } = await extractor.extractFromToolResult(
|
|
1368
|
+
{ id: 'call_1', name: 'database_query', arguments: { table: 'users' } },
|
|
1103
1369
|
{ rows: 100, cached: true },
|
|
1104
|
-
{
|
|
1370
|
+
{ taskDescription: 'Count active users' },
|
|
1371
|
+
graph
|
|
1105
1372
|
);
|
|
1106
1373
|
|
|
1107
1374
|
const generator = new CausalHypothesisGenerator({ llmBackend: backend });
|
|
1108
1375
|
const hypotheses = await generator.generateFromFailure(trace, { agentId: 'agent-1' });
|
|
1109
1376
|
```
|
|
1110
1377
|
|
|
1378
|
+
`CausalReasoner` ties these together for agents (`predictEffect`, `explainCause`, `planForGoal`, `evaluateToolCall`, `analyzeErrorCausally`). See [Causal Reasoning](https://cogitator.app/docs/advanced/causal-reasoning).
|
|
1379
|
+
|
|
1111
1380
|
---
|
|
1112
1381
|
|
|
1113
1382
|
## Error Handling & Resilience
|
|
@@ -1117,6 +1386,8 @@ const hypotheses = await generator.generateFromFailure(trace, { agentId: 'agent-
|
|
|
1117
1386
|
Agent runs retry failed LLM calls on their own: rate limits, 5xx, timeouts and dropped connections, with exponential backoff and the provider's `Retry-After`. Streams are retried only before the first chunk. Tune or turn this off with `llm.retry`:
|
|
1118
1387
|
|
|
1119
1388
|
```typescript
|
|
1389
|
+
import { Cogitator, GoogleBackend, withLLMRetry } from '@cogitator-ai/core';
|
|
1390
|
+
|
|
1120
1391
|
const cog = new Cogitator({
|
|
1121
1392
|
llm: {
|
|
1122
1393
|
retry: { maxRetries: 3, maxRetryAfter: 30_000, onRetry: (e) => console.warn(e) },
|
|
@@ -1125,7 +1396,9 @@ const cog = new Cogitator({
|
|
|
1125
1396
|
});
|
|
1126
1397
|
|
|
1127
1398
|
// A backend used on its own
|
|
1128
|
-
const backend = withLLMRetry(new GoogleBackend({ apiKey }), {
|
|
1399
|
+
const backend = withLLMRetry(new GoogleBackend({ apiKey: process.env.GOOGLE_API_KEY! }), {
|
|
1400
|
+
maxRetries: 3,
|
|
1401
|
+
});
|
|
1129
1402
|
```
|
|
1130
1403
|
|
|
1131
1404
|
### Retry with Backoff
|
|
@@ -1154,31 +1427,66 @@ const response = await retryableFetch('https://api.example.com');
|
|
|
1154
1427
|
import { CircuitBreaker, CircuitBreakerRegistry } from '@cogitator-ai/core';
|
|
1155
1428
|
|
|
1156
1429
|
const breaker = new CircuitBreaker({
|
|
1157
|
-
|
|
1430
|
+
failureThreshold: 5,
|
|
1158
1431
|
resetTimeout: 30000,
|
|
1159
|
-
|
|
1432
|
+
halfOpenRequests: 3,
|
|
1433
|
+
onStateChange: (from, to) => console.log(`Circuit ${from} -> ${to}`),
|
|
1160
1434
|
});
|
|
1161
1435
|
|
|
1162
|
-
|
|
1163
|
-
|
|
1164
|
-
|
|
1165
|
-
|
|
1166
|
-
} catch (error) {
|
|
1167
|
-
breaker.recordFailure();
|
|
1168
|
-
throw error;
|
|
1169
|
-
}
|
|
1170
|
-
}
|
|
1436
|
+
// Throws CIRCUIT_OPEN while open; counts successes and failures (by default only retryable errors, see isFailure)
|
|
1437
|
+
const result = await breaker.execute(() => riskyOperation());
|
|
1438
|
+
|
|
1439
|
+
console.log(breaker.getState(), breaker.getStats());
|
|
1171
1440
|
|
|
1172
|
-
|
|
1173
|
-
|
|
1441
|
+
const registry = new CircuitBreakerRegistry({ failureThreshold: 3 });
|
|
1442
|
+
const apiBreaker = registry.get('payments-api');
|
|
1443
|
+
```
|
|
1444
|
+
|
|
1445
|
+
### Fallback Patterns
|
|
1446
|
+
|
|
1447
|
+
```typescript
|
|
1448
|
+
import {
|
|
1449
|
+
CircuitBreakerRegistry,
|
|
1450
|
+
withFallback,
|
|
1451
|
+
withGracefulDegradation,
|
|
1452
|
+
createLLMFallbackExecutor,
|
|
1453
|
+
} from '@cogitator-ai/core';
|
|
1454
|
+
|
|
1455
|
+
const result = await withFallback({
|
|
1456
|
+
primary: () => primaryCall(),
|
|
1457
|
+
fallbacks: [
|
|
1458
|
+
{ name: 'secondary', fn: () => fallbackCall() },
|
|
1459
|
+
{ name: 'cache', fn: () => cachedResult() },
|
|
1460
|
+
],
|
|
1461
|
+
retry: { maxRetries: 2 },
|
|
1462
|
+
onFallback: (from, to, error) => console.warn(`${from} -> ${to}: ${error.message}`),
|
|
1174
1463
|
});
|
|
1464
|
+
|
|
1465
|
+
const degraded = await withGracefulDegradation(() => fullFeatureCall(), {
|
|
1466
|
+
defaultValue: [],
|
|
1467
|
+
onDegraded: (error) => console.warn('Degraded:', error.message),
|
|
1468
|
+
});
|
|
1469
|
+
|
|
1470
|
+
const executeWithFallback = createLLMFallbackExecutor(
|
|
1471
|
+
{
|
|
1472
|
+
providers: [
|
|
1473
|
+
{ provider: 'openai', model: 'gpt-6.1-sol' },
|
|
1474
|
+
{ provider: 'anthropic', model: 'claude-sonnet-5-5' },
|
|
1475
|
+
{ provider: 'ollama', model: 'llama3.3:70b' },
|
|
1476
|
+
],
|
|
1477
|
+
},
|
|
1478
|
+
new CircuitBreakerRegistry()
|
|
1479
|
+
);
|
|
1480
|
+
const response = await executeWithFallback((provider, model) =>
|
|
1481
|
+
cog.getLLMBackend(`${provider}/${model}`).chat({ model, messages })
|
|
1482
|
+
);
|
|
1175
1483
|
```
|
|
1176
1484
|
|
|
1177
1485
|
---
|
|
1178
1486
|
|
|
1179
1487
|
## Prompt Injection Detection
|
|
1180
1488
|
|
|
1181
|
-
Protect your agents from jailbreak attempts, prompt injections, and other adversarial inputs
|
|
1489
|
+
Protect your agents from jailbreak attempts, prompt injections, and other adversarial inputs. See [Security](https://cogitator.app/docs/advanced/security).
|
|
1182
1490
|
|
|
1183
1491
|
```typescript
|
|
1184
1492
|
import { Cogitator, PromptInjectionDetector } from '@cogitator-ai/core';
|
|
@@ -1273,7 +1581,7 @@ const stats = detector.getStats();
|
|
|
1273
1581
|
|
|
1274
1582
|
## Tool Caching
|
|
1275
1583
|
|
|
1276
|
-
Cache tool results to avoid redundant API calls with exact or semantic matching
|
|
1584
|
+
Cache tool results to avoid redundant API calls with exact or semantic matching. See [Tool Caching](https://cogitator.app/docs/tools/tool-caching).
|
|
1277
1585
|
|
|
1278
1586
|
### Exact Match Caching
|
|
1279
1587
|
|
|
@@ -1310,18 +1618,17 @@ Similar queries hit the cache based on embedding similarity:
|
|
|
1310
1618
|
|
|
1311
1619
|
```typescript
|
|
1312
1620
|
import { withCache } from '@cogitator-ai/core';
|
|
1313
|
-
import
|
|
1621
|
+
import { OpenAIEmbeddingService } from '@cogitator-ai/memory';
|
|
1314
1622
|
|
|
1315
|
-
|
|
1316
|
-
|
|
1317
|
-
|
|
1318
|
-
dimensions: 1536,
|
|
1623
|
+
// Any EmbeddingService ({ embed, embedBatch, dimensions, model }) works
|
|
1624
|
+
const embeddingService = new OpenAIEmbeddingService({
|
|
1625
|
+
apiKey: process.env.OPENAI_API_KEY!,
|
|
1319
1626
|
model: 'text-embedding-3-small',
|
|
1320
|
-
};
|
|
1627
|
+
});
|
|
1321
1628
|
|
|
1322
1629
|
const cachedSearch = withCache(webSearch, {
|
|
1323
1630
|
strategy: 'semantic',
|
|
1324
|
-
similarity: 0.95,
|
|
1631
|
+
similarity: 0.95, // 95% similarity threshold
|
|
1325
1632
|
ttl: '1h',
|
|
1326
1633
|
maxSize: 1000,
|
|
1327
1634
|
storage: 'memory',
|
|
@@ -1334,10 +1641,13 @@ await cachedSearch.execute({ query: 'Paris weather forecast' }, ctx); // semanti
|
|
|
1334
1641
|
|
|
1335
1642
|
### Redis Storage
|
|
1336
1643
|
|
|
1337
|
-
For production with persistence:
|
|
1644
|
+
For production with persistence. `redisClient` takes any `RedisClientLike`; an ioredis client and a `createRedisClient()` client from `@cogitator-ai/redis` (standalone or cluster) fit as they are:
|
|
1338
1645
|
|
|
1339
1646
|
```typescript
|
|
1340
|
-
import { withCache
|
|
1647
|
+
import { withCache } from '@cogitator-ai/core';
|
|
1648
|
+
import { Redis } from 'ioredis';
|
|
1649
|
+
|
|
1650
|
+
const redis = new Redis(process.env.REDIS_URL!);
|
|
1341
1651
|
|
|
1342
1652
|
const cachedTool = withCache(webSearch, {
|
|
1343
1653
|
strategy: 'semantic',
|
|
@@ -1345,16 +1655,18 @@ const cachedTool = withCache(webSearch, {
|
|
|
1345
1655
|
ttl: '1h',
|
|
1346
1656
|
maxSize: 1000,
|
|
1347
1657
|
storage: 'redis',
|
|
1348
|
-
redisClient:
|
|
1658
|
+
redisClient: redis,
|
|
1349
1659
|
keyPrefix: 'myapp:cache',
|
|
1350
1660
|
embeddingService,
|
|
1351
1661
|
});
|
|
1352
1662
|
```
|
|
1353
1663
|
|
|
1664
|
+
Keys live under `keyPrefix` (a `:` is appended when missing and never doubled): entries at `<prefix>:entry:<cache key>`, the LRU order in `<prefix>:lru` and the entry count in `<prefix>:counter`. Cached tools sharing a prefix share one LRU and `maxSize`; `cache.clear()` deletes every key under the prefix.
|
|
1665
|
+
|
|
1354
1666
|
### Cache Management
|
|
1355
1667
|
|
|
1356
1668
|
```typescript
|
|
1357
|
-
const cached = withCache(
|
|
1669
|
+
const cached = withCache(searchTool, config);
|
|
1358
1670
|
|
|
1359
1671
|
// Get statistics
|
|
1360
1672
|
const stats = cached.cache.stats();
|
|
@@ -1375,44 +1687,30 @@ await cached.cache.warmup([
|
|
|
1375
1687
|
### Cache Callbacks
|
|
1376
1688
|
|
|
1377
1689
|
```typescript
|
|
1378
|
-
const cached = withCache(
|
|
1690
|
+
const cached = withCache(searchTool, {
|
|
1379
1691
|
strategy: 'exact',
|
|
1380
1692
|
ttl: '1h',
|
|
1381
1693
|
maxSize: 100,
|
|
1382
1694
|
storage: 'memory',
|
|
1383
|
-
onHit: (key
|
|
1384
|
-
onMiss: (key
|
|
1695
|
+
onHit: (key) => console.log('Cache hit:', key),
|
|
1696
|
+
onMiss: (key) => console.log('Cache miss:', key),
|
|
1385
1697
|
onEvict: (key) => console.log('Evicted:', key),
|
|
1386
1698
|
});
|
|
1387
1699
|
```
|
|
1388
1700
|
|
|
1389
|
-
|
|
1701
|
+
`onEvict` fires for entries removed by `cache.invalidate()` and for entries evicted to stay under `maxSize`, in memory and in Redis.
|
|
1390
1702
|
|
|
1391
|
-
|
|
1703
|
+
To build a storage yourself, `createToolCacheStorage()` takes the same `onEvict`, called with the key of each entry evicted to make room:
|
|
1392
1704
|
|
|
1393
1705
|
```typescript
|
|
1394
|
-
import {
|
|
1395
|
-
withFallback,
|
|
1396
|
-
withGracefulDegradation,
|
|
1397
|
-
createLLMFallbackExecutor,
|
|
1398
|
-
} from '@cogitator-ai/core';
|
|
1399
|
-
|
|
1400
|
-
const result = await withFallback(
|
|
1401
|
-
() => primaryCall(),
|
|
1402
|
-
() => fallbackCall()
|
|
1403
|
-
);
|
|
1404
|
-
|
|
1405
|
-
const degraded = await withGracefulDegradation(
|
|
1406
|
-
() => fullFeatureCall(),
|
|
1407
|
-
[() => reducedFeatureCall(), () => minimalCall(), () => cachedResult()]
|
|
1408
|
-
);
|
|
1706
|
+
import { createToolCacheStorage } from '@cogitator-ai/core';
|
|
1409
1707
|
|
|
1410
|
-
const
|
|
1411
|
-
|
|
1412
|
-
|
|
1413
|
-
|
|
1414
|
-
|
|
1415
|
-
|
|
1708
|
+
const storage = createToolCacheStorage('redis', {
|
|
1709
|
+
redisClient: redis,
|
|
1710
|
+
keyPrefix: 'myapp:cache',
|
|
1711
|
+
maxSize: 5000,
|
|
1712
|
+
onEvict: (key) => console.log('Evicted:', key),
|
|
1713
|
+
});
|
|
1416
1714
|
```
|
|
1417
1715
|
|
|
1418
1716
|
---
|
|
@@ -1422,38 +1720,47 @@ const response = await llmExecutor.chat(request);
|
|
|
1422
1720
|
Built-in content safety with input/output filtering, tool guards, and critique-revision loops:
|
|
1423
1721
|
|
|
1424
1722
|
```typescript
|
|
1425
|
-
import {
|
|
1723
|
+
import {
|
|
1724
|
+
Cogitator,
|
|
1725
|
+
ConstitutionalAI,
|
|
1726
|
+
createConstitution,
|
|
1727
|
+
DEFAULT_PRINCIPLES,
|
|
1728
|
+
} from '@cogitator-ai/core';
|
|
1426
1729
|
|
|
1427
1730
|
const constitutional = new ConstitutionalAI({
|
|
1428
1731
|
llm: backend,
|
|
1429
1732
|
constitution: createConstitution([...DEFAULT_PRINCIPLES, customPrinciple]),
|
|
1430
|
-
config: {
|
|
1733
|
+
config: { strictMode: true },
|
|
1431
1734
|
});
|
|
1432
1735
|
|
|
1433
|
-
const
|
|
1434
|
-
|
|
1435
|
-
|
|
1436
|
-
console.log('Blocked:', inputResult.harmCategories);
|
|
1736
|
+
const inputResult = await constitutional.filterInput('user message');
|
|
1737
|
+
if (!inputResult.allowed) {
|
|
1738
|
+
console.log('Blocked:', inputResult.blockedReason, inputResult.harmScores);
|
|
1437
1739
|
}
|
|
1438
1740
|
|
|
1439
|
-
const
|
|
1440
|
-
const
|
|
1741
|
+
const outputResult = await constitutional.filterOutput('agent response', messages);
|
|
1742
|
+
const guardResult = await constitutional.guardTool(
|
|
1743
|
+
deleteFileTool,
|
|
1744
|
+
{ path: '/etc/passwd' },
|
|
1745
|
+
toolContext
|
|
1746
|
+
);
|
|
1747
|
+
console.log(guardResult.approved, guardResult.riskLevel, guardResult.reason);
|
|
1441
1748
|
|
|
1442
|
-
const
|
|
1443
|
-
const guardResult = await toolGuard.evaluate({
|
|
1444
|
-
name: 'delete_file',
|
|
1445
|
-
arguments: { path: '/etc/passwd' },
|
|
1446
|
-
});
|
|
1749
|
+
const revision = await constitutional.critiqueAndRevise('draft answer', messages);
|
|
1447
1750
|
|
|
1448
|
-
// Integrated with Cogitator runtime
|
|
1751
|
+
// Integrated with the Cogitator runtime: on when `guardrails` is set (unless enabled: false)
|
|
1449
1752
|
const cog = new Cogitator({
|
|
1450
1753
|
guardrails: {
|
|
1451
|
-
|
|
1452
|
-
|
|
1754
|
+
model: 'openai/gpt-6-luna', // judge model; default: llm.defaultModel, else the first run's agent model
|
|
1755
|
+
filterToolResults: true,
|
|
1453
1756
|
},
|
|
1454
1757
|
});
|
|
1455
1758
|
```
|
|
1456
1759
|
|
|
1760
|
+
Fields left out take `DEFAULT_GUARDRAIL_CONFIG` (input, output and tool-call filtering plus critique-revision on). `InputFilter`, `OutputFilter`, `ToolGuard` and `CritiqueReviser` are the layers `ConstitutionalAI` uses; each takes `{ config, constitution }` plus `llm` for the LLM-backed ones. `cog.getGuardrails()` and `cog.setConstitution()` work before the first run when `guardrails.model` or `llm.defaultModel` names the judge model; a constitution set earlier applies once the guardrails are built.
|
|
1761
|
+
|
|
1762
|
+
With `strictMode`, every call of a tool with `sideEffects` needs approval and goes through the run's [approval flow](#approvals) like a `requiresApproval` tool: `onApproval`, then `guardrails.onToolApproval`, else the run pauses. `ToolGuard` fails closed: a call that needs approval and was not approved by that flow or by `onToolApproval` is denied. See [Constitutional AI](https://cogitator.app/docs/advanced/constitutional-ai).
|
|
1763
|
+
|
|
1457
1764
|
---
|
|
1458
1765
|
|
|
1459
1766
|
## Cost-Aware Routing
|
|
@@ -1461,35 +1768,43 @@ const cog = new Cogitator({
|
|
|
1461
1768
|
Automatically route tasks to the optimal model based on complexity, budget, and latency requirements:
|
|
1462
1769
|
|
|
1463
1770
|
```typescript
|
|
1464
|
-
import {
|
|
1771
|
+
import { Cogitator, CostAwareRouter } from '@cogitator-ai/core';
|
|
1465
1772
|
|
|
1466
1773
|
const router = new CostAwareRouter({
|
|
1467
1774
|
config: {
|
|
1468
1775
|
enabled: true,
|
|
1469
|
-
|
|
1470
|
-
|
|
1471
|
-
|
|
1776
|
+
budget: {
|
|
1777
|
+
maxCostPerRun: 0.5,
|
|
1778
|
+
maxCostPerDay: 10.0,
|
|
1779
|
+
warningThreshold: 0.8,
|
|
1780
|
+
onBudgetWarning: (current, limit) => console.warn(`$${current} of $${limit}`),
|
|
1472
1781
|
},
|
|
1473
1782
|
},
|
|
1474
1783
|
});
|
|
1475
1784
|
|
|
1476
|
-
const
|
|
1477
|
-
|
|
1478
|
-
|
|
1479
|
-
|
|
1480
|
-
});
|
|
1481
|
-
console.log('Use model:', recommendation.model);
|
|
1482
|
-
console.log('Estimated cost:', recommendation.estimatedCost);
|
|
1785
|
+
const requirements = router.analyzeTask('Say hello'); // complexity, reasoning, speed, cost sensitivity
|
|
1786
|
+
const recommendation = await router.recommendModel('Say hello');
|
|
1787
|
+
console.log('Use model:', recommendation.provider, recommendation.modelId);
|
|
1788
|
+
console.log('Estimated cost:', recommendation.estimatedCost, recommendation.reasons);
|
|
1483
1789
|
|
|
1484
1790
|
// Integrated with Cogitator
|
|
1485
1791
|
const cog = new Cogitator({
|
|
1486
1792
|
costRouting: {
|
|
1487
1793
|
enabled: true,
|
|
1488
|
-
|
|
1794
|
+
autoSelectModel: true, // pick the model per run; result.modelUsed tells which
|
|
1795
|
+
budget: { maxCostPerDay: 10.0 },
|
|
1489
1796
|
},
|
|
1490
1797
|
});
|
|
1798
|
+
|
|
1799
|
+
cog.getCostSummary(); // tracked costs, also before the first run
|
|
1800
|
+
const estimate = await cog.estimateCost({ agent, input: 'Summarize this report' });
|
|
1801
|
+
console.log(estimate.expectedCost);
|
|
1491
1802
|
```
|
|
1492
1803
|
|
|
1804
|
+
With `costRouting.enabled`, every run is checked against `budget`, with or without `autoSelectModel`: the cost is estimated from the task's complexity and the model's price, and a run over a limit throws a `CogitatorError` with code `BUDGET_EXCEEDED` (HTTP 429). `autoSelectModel` only picks models of providers the runtime can call (`llm.backends`, plugins, `llm.defaultProvider`, the agent's own provider, or `llm.providers` entries with credentials) and keeps the agent's model when none fits. The router exposes the same steps: `recommendAvailableModel(input, isProviderAvailable)` (undefined when no provider fits) and `checkRunBudget(input, model)`.
|
|
1805
|
+
|
|
1806
|
+
See [Cost Routing](https://cogitator.app/docs/advanced/cost-routing).
|
|
1807
|
+
|
|
1493
1808
|
---
|
|
1494
1809
|
|
|
1495
1810
|
## Context Management
|
|
@@ -1499,15 +1814,16 @@ Automatic context window management with compression strategies:
|
|
|
1499
1814
|
```typescript
|
|
1500
1815
|
const cog = new Cogitator({
|
|
1501
1816
|
context: {
|
|
1502
|
-
strategy: 'hybrid',
|
|
1503
|
-
compressionThreshold: 0.8,
|
|
1817
|
+
strategy: 'hybrid', // 'truncate' | 'sliding-window' | 'summarize' | 'hybrid'
|
|
1818
|
+
compressionThreshold: 0.8, // compress above 80% of the usable window
|
|
1819
|
+
outputReserve: 0.15, // share of the model's context kept for the answer
|
|
1820
|
+
windowSize: 10, // recent messages kept verbatim by sliding-window / hybrid
|
|
1821
|
+
summaryModel: 'openai/gpt-6-luna', // model that writes summaries
|
|
1504
1822
|
},
|
|
1505
1823
|
});
|
|
1506
1824
|
```
|
|
1507
1825
|
|
|
1508
|
-
Setting `context` turns compression on; pass `enabled: false` to keep the config but switch it off.
|
|
1509
|
-
|
|
1510
|
-
Available strategies: `TruncateStrategy`, `SlidingWindowStrategy`, `SummarizeStrategy`, `HybridStrategy`.
|
|
1826
|
+
Setting `context` turns compression on; pass `enabled: false` to keep the config but switch it off. The strategies are also exported as classes (`TruncateStrategy`, `SlidingWindowStrategy`, `SummarizeStrategy`, `HybridStrategy`) and `ContextManager` can be used on its own. See [Context Management](https://cogitator.app/docs/advanced/context-management).
|
|
1511
1827
|
|
|
1512
1828
|
---
|
|
1513
1829
|
|
|
@@ -1522,14 +1838,35 @@ const langfuse = createLangfuseExporter({
|
|
|
1522
1838
|
publicKey: process.env.LANGFUSE_PUBLIC_KEY!,
|
|
1523
1839
|
secretKey: process.env.LANGFUSE_SECRET_KEY!,
|
|
1524
1840
|
baseUrl: 'https://cloud.langfuse.com',
|
|
1841
|
+
enabled: true,
|
|
1525
1842
|
});
|
|
1526
1843
|
|
|
1844
|
+
await langfuse.init(); // needs the optional `langfuse` package
|
|
1845
|
+
|
|
1527
1846
|
const otlp = createOTLPExporter({
|
|
1528
1847
|
endpoint: 'http://localhost:4318/v1/traces',
|
|
1529
1848
|
headers: { Authorization: 'Bearer ...' },
|
|
1849
|
+
serviceName: 'my-agents',
|
|
1850
|
+
enabled: true,
|
|
1851
|
+
});
|
|
1852
|
+
otlp.start(); // flushes every 5 seconds
|
|
1853
|
+
|
|
1854
|
+
let runId = '';
|
|
1855
|
+
const result = await cog.run(agent, {
|
|
1856
|
+
input: 'Analyze this data...',
|
|
1857
|
+
onRunStart: (data) => {
|
|
1858
|
+
runId = data.runId;
|
|
1859
|
+
langfuse.onRunStart({ ...data, agentName: agent.name });
|
|
1860
|
+
},
|
|
1861
|
+
onToolCall: (call) => langfuse.onToolCall(runId, call),
|
|
1862
|
+
onToolResult: (toolResult) => langfuse.onToolResult(runId, toolResult),
|
|
1863
|
+
onSpan: (span) => otlp.exportSpan(runId, span),
|
|
1864
|
+
onRunComplete: (runResult) => langfuse.onRunComplete(runResult),
|
|
1530
1865
|
});
|
|
1531
1866
|
```
|
|
1532
1867
|
|
|
1868
|
+
Both exporters do nothing unless `enabled: true` is set, so you can build them unconditionally and switch them per environment. They are not attached automatically; wire them to the run callbacks as above. See [Observability](https://cogitator.app/docs/deployment/observability).
|
|
1869
|
+
|
|
1533
1870
|
---
|
|
1534
1871
|
|
|
1535
1872
|
## Agent as Tool
|
|
@@ -1549,7 +1886,10 @@ const researcher = new Agent({
|
|
|
1549
1886
|
const researchTool = agentAsTool(cog, researcher, {
|
|
1550
1887
|
name: 'research',
|
|
1551
1888
|
description: 'Delegate research tasks to a specialist agent',
|
|
1889
|
+
timeout: 60_000,
|
|
1552
1890
|
includeUsage: true,
|
|
1891
|
+
includeToolCalls: false,
|
|
1892
|
+
onApproval: () => ({ approved: false, reason: 'Not allowed in delegated runs' }),
|
|
1553
1893
|
});
|
|
1554
1894
|
|
|
1555
1895
|
const manager = new Agent({
|
|
@@ -1560,17 +1900,19 @@ const manager = new Agent({
|
|
|
1560
1900
|
});
|
|
1561
1901
|
```
|
|
1562
1902
|
|
|
1903
|
+
Tool calls of the inner agent that need approval are declined unless `onApproval` decides them, since a delegated run cannot pause for a person; an `onApproval` that returns `'pause'` declines too. If the inner run pauses anyway, the tool returns `success: false` with an `error` naming the tools that waited. To hand the conversation over instead of calling a sub-agent, use [handoffs](#handoffs). See [Agent as Tool](https://cogitator.app/docs/tools/agent-as-tool).
|
|
1904
|
+
|
|
1563
1905
|
---
|
|
1564
1906
|
|
|
1565
1907
|
## Logging
|
|
1566
1908
|
|
|
1567
1909
|
```typescript
|
|
1568
|
-
import {
|
|
1910
|
+
import { getLogger, setLogger, createLogger, createLoggerFromConfig } from '@cogitator-ai/core';
|
|
1569
1911
|
|
|
1570
1912
|
const logger = createLogger({
|
|
1571
|
-
level: 'debug',
|
|
1572
|
-
|
|
1573
|
-
|
|
1913
|
+
level: 'debug', // default 'info'; the default logger reads LOG_LEVEL
|
|
1914
|
+
format: 'json', // 'pretty' (default) or 'json'
|
|
1915
|
+
output: (entry, formatted) => process.stderr.write(formatted + '\n'),
|
|
1574
1916
|
});
|
|
1575
1917
|
|
|
1576
1918
|
setLogger(logger);
|
|
@@ -1581,6 +1923,24 @@ getLogger().warn('Rate limited', { retryAfter: 60 });
|
|
|
1581
1923
|
getLogger().error('Failed', { error: 'Connection timeout' });
|
|
1582
1924
|
```
|
|
1583
1925
|
|
|
1926
|
+
`new Cogitator({ logging })` installs a logger built from the config with `createLoggerFromConfig()`. It is process-wide, so with several runtimes the last one created with `logging` wins:
|
|
1927
|
+
|
|
1928
|
+
```typescript
|
|
1929
|
+
import { Cogitator, createLoggerFromConfig, setLogger } from '@cogitator-ai/core';
|
|
1930
|
+
|
|
1931
|
+
const cog = new Cogitator({
|
|
1932
|
+
logging: {
|
|
1933
|
+
level: 'warn', // 'debug' | 'info' | 'warn' | 'error' | 'silent'
|
|
1934
|
+
destination: 'file', // appends JSON lines to filePath
|
|
1935
|
+
filePath: './cogitator.log',
|
|
1936
|
+
},
|
|
1937
|
+
});
|
|
1938
|
+
|
|
1939
|
+
setLogger(createLoggerFromConfig({ level: 'silent' })); // the same, without a runtime
|
|
1940
|
+
```
|
|
1941
|
+
|
|
1942
|
+
Without `filePath`, or where there is no file system, `destination: 'file'` logs to the console with a warning.
|
|
1943
|
+
|
|
1584
1944
|
---
|
|
1585
1945
|
|
|
1586
1946
|
## Type Reference
|
|
@@ -1611,11 +1971,11 @@ import type {
|
|
|
1611
1971
|
import type {
|
|
1612
1972
|
LLMBackend,
|
|
1613
1973
|
LLMProvider,
|
|
1974
|
+
LLMBackendProvider, // LLMProvider or the name of your own backend
|
|
1614
1975
|
LLMConfig,
|
|
1615
1976
|
ChatRequest,
|
|
1616
1977
|
ChatResponse,
|
|
1617
1978
|
ChatStreamChunk,
|
|
1618
|
-
ChatUsage,
|
|
1619
1979
|
LLMErrorContext,
|
|
1620
1980
|
LLMDebugOptions,
|
|
1621
1981
|
LLMPlugin,
|
|
@@ -1698,14 +2058,17 @@ try {
|
|
|
1698
2058
|
} catch (error) {
|
|
1699
2059
|
if (error instanceof LLMError) {
|
|
1700
2060
|
console.log('Provider:', error.provider);
|
|
1701
|
-
console.log('
|
|
2061
|
+
console.log('Provider status:', error.details?.statusCode);
|
|
1702
2062
|
} else if (error instanceof CogitatorError) {
|
|
1703
|
-
console.log('Code:', error.code);
|
|
1704
|
-
console.log('
|
|
2063
|
+
console.log('Code:', error.code); // an ErrorCode, e.g. ErrorCode.THREAD_ACCESS_DENIED
|
|
2064
|
+
console.log('HTTP status:', error.statusCode);
|
|
2065
|
+
console.log('Retryable:', isRetryableError(error), 'retry in', getRetryDelay(error), 'ms');
|
|
1705
2066
|
}
|
|
1706
2067
|
}
|
|
1707
2068
|
```
|
|
1708
2069
|
|
|
2070
|
+
Runs that hit their `timeout` throw `RUN_TIMEOUT` (HTTP 504, `Run timed out after <n>ms`), runs over the cost-routing budget `BUDGET_EXCEEDED` (429), and input or output blocked by guardrails (`Input blocked: …` / `Output blocked: …`) `LLM_CONTENT_FILTERED` (400), so server adapters return those statuses and messages.
|
|
2071
|
+
|
|
1709
2072
|
---
|
|
1710
2073
|
|
|
1711
2074
|
## Examples
|
|
@@ -1714,6 +2077,7 @@ try {
|
|
|
1714
2077
|
|
|
1715
2078
|
```typescript
|
|
1716
2079
|
import { Cogitator, Agent, tool } from '@cogitator-ai/core';
|
|
2080
|
+
import { z } from 'zod';
|
|
1717
2081
|
|
|
1718
2082
|
const webSearch = tool({
|
|
1719
2083
|
name: 'web_search',
|