@cogitator-ai/core 0.25.0 → 0.26.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (169) hide show
  1. package/README.md +531 -167
  2. package/dist/agent-tool.d.ts +4 -3
  3. package/dist/agent-tool.d.ts.map +1 -1
  4. package/dist/agent-tool.js +15 -2
  5. package/dist/agent-tool.js.map +1 -1
  6. package/dist/agent.d.ts.map +1 -1
  7. package/dist/agent.js +3 -1
  8. package/dist/agent.js.map +1 -1
  9. package/dist/cache/cache-key.d.ts +1 -0
  10. package/dist/cache/cache-key.d.ts.map +1 -1
  11. package/dist/cache/cache-key.js +2 -1
  12. package/dist/cache/cache-key.js.map +1 -1
  13. package/dist/cache/storage/memory.d.ts +8 -1
  14. package/dist/cache/storage/memory.d.ts.map +1 -1
  15. package/dist/cache/storage/memory.js +8 -1
  16. package/dist/cache/storage/memory.js.map +1 -1
  17. package/dist/cache/storage/redis.d.ts +4 -0
  18. package/dist/cache/storage/redis.d.ts.map +1 -1
  19. package/dist/cache/storage/redis.js +17 -18
  20. package/dist/cache/storage/redis.js.map +1 -1
  21. package/dist/cache/tool-cache.d.ts +2 -0
  22. package/dist/cache/tool-cache.d.ts.map +1 -1
  23. package/dist/cache/tool-cache.js +4 -2
  24. package/dist/cache/tool-cache.js.map +1 -1
  25. package/dist/causal/inference/counterfactual.d.ts +2 -0
  26. package/dist/causal/inference/counterfactual.d.ts.map +1 -1
  27. package/dist/causal/inference/counterfactual.js +24 -3
  28. package/dist/causal/inference/counterfactual.js.map +1 -1
  29. package/dist/causal/inference/expression.d.ts +19 -0
  30. package/dist/causal/inference/expression.d.ts.map +1 -0
  31. package/dist/causal/inference/expression.js +177 -0
  32. package/dist/causal/inference/expression.js.map +1 -0
  33. package/dist/cogitator/initializers.d.ts +4 -3
  34. package/dist/cogitator/initializers.d.ts.map +1 -1
  35. package/dist/cogitator/initializers.js +100 -46
  36. package/dist/cogitator/initializers.js.map +1 -1
  37. package/dist/cogitator/message-builder.d.ts.map +1 -1
  38. package/dist/cogitator/message-builder.js +5 -1
  39. package/dist/cogitator/message-builder.js.map +1 -1
  40. package/dist/cogitator/prompts.d.ts.map +1 -1
  41. package/dist/cogitator/prompts.js +5 -1
  42. package/dist/cogitator/prompts.js.map +1 -1
  43. package/dist/cogitator/run-limiter.d.ts.map +1 -1
  44. package/dist/cogitator/run-limiter.js +5 -1
  45. package/dist/cogitator/run-limiter.js.map +1 -1
  46. package/dist/cogitator/tool-executor.d.ts +8 -2
  47. package/dist/cogitator/tool-executor.d.ts.map +1 -1
  48. package/dist/cogitator/tool-executor.js +54 -5
  49. package/dist/cogitator/tool-executor.js.map +1 -1
  50. package/dist/constitutional/constitutional-ai.d.ts +7 -0
  51. package/dist/constitutional/constitutional-ai.d.ts.map +1 -1
  52. package/dist/constitutional/constitutional-ai.js +11 -0
  53. package/dist/constitutional/constitutional-ai.js.map +1 -1
  54. package/dist/constitutional/tool-guard.d.ts +9 -0
  55. package/dist/constitutional/tool-guard.d.ts.map +1 -1
  56. package/dist/constitutional/tool-guard.js +25 -20
  57. package/dist/constitutional/tool-guard.js.map +1 -1
  58. package/dist/cost-routing/cost-estimator.d.ts.map +1 -1
  59. package/dist/cost-routing/cost-estimator.js +5 -1
  60. package/dist/cost-routing/cost-estimator.js.map +1 -1
  61. package/dist/cost-routing/cost-router.d.ts +13 -0
  62. package/dist/cost-routing/cost-router.d.ts.map +1 -1
  63. package/dist/cost-routing/cost-router.js +22 -1
  64. package/dist/cost-routing/cost-router.js.map +1 -1
  65. package/dist/cost-routing/model-selector.d.ts +13 -0
  66. package/dist/cost-routing/model-selector.d.ts.map +1 -1
  67. package/dist/cost-routing/model-selector.js +25 -16
  68. package/dist/cost-routing/model-selector.js.map +1 -1
  69. package/dist/index.d.ts +2 -2
  70. package/dist/index.d.ts.map +1 -1
  71. package/dist/index.js +1 -1
  72. package/dist/index.js.map +1 -1
  73. package/dist/learning/ab-testing.d.ts +1 -1
  74. package/dist/learning/ab-testing.d.ts.map +1 -1
  75. package/dist/learning/ab-testing.js +6 -28
  76. package/dist/learning/ab-testing.js.map +1 -1
  77. package/dist/learning/agent-optimizer.d.ts +2 -0
  78. package/dist/learning/agent-optimizer.d.ts.map +1 -1
  79. package/dist/learning/agent-optimizer.js +20 -2
  80. package/dist/learning/agent-optimizer.js.map +1 -1
  81. package/dist/learning/auto-optimizer.d.ts +11 -0
  82. package/dist/learning/auto-optimizer.d.ts.map +1 -1
  83. package/dist/learning/auto-optimizer.js +47 -16
  84. package/dist/learning/auto-optimizer.js.map +1 -1
  85. package/dist/learning/metrics.d.ts +8 -1
  86. package/dist/learning/metrics.d.ts.map +1 -1
  87. package/dist/learning/metrics.js +48 -16
  88. package/dist/learning/metrics.js.map +1 -1
  89. package/dist/learning/postgres-trace-store.d.ts +3 -1
  90. package/dist/learning/postgres-trace-store.d.ts.map +1 -1
  91. package/dist/learning/postgres-trace-store.js +27 -4
  92. package/dist/learning/postgres-trace-store.js.map +1 -1
  93. package/dist/learning/prompt-logger.d.ts +2 -2
  94. package/dist/learning/prompt-logger.d.ts.map +1 -1
  95. package/dist/learning/prompt-logger.js.map +1 -1
  96. package/dist/learning/trace-builder.d.ts.map +1 -1
  97. package/dist/learning/trace-builder.js +1 -0
  98. package/dist/learning/trace-builder.js.map +1 -1
  99. package/dist/llm/anthropic.d.ts +1 -0
  100. package/dist/llm/anthropic.d.ts.map +1 -1
  101. package/dist/llm/anthropic.js +6 -1
  102. package/dist/llm/anthropic.js.map +1 -1
  103. package/dist/llm/base.d.ts +2 -2
  104. package/dist/llm/base.d.ts.map +1 -1
  105. package/dist/llm/bedrock.d.ts.map +1 -1
  106. package/dist/llm/bedrock.js +3 -1
  107. package/dist/llm/bedrock.js.map +1 -1
  108. package/dist/llm/debug.d.ts +2 -2
  109. package/dist/llm/debug.d.ts.map +1 -1
  110. package/dist/llm/debug.js.map +1 -1
  111. package/dist/llm/google.d.ts.map +1 -1
  112. package/dist/llm/google.js +56 -6
  113. package/dist/llm/google.js.map +1 -1
  114. package/dist/llm/openai-compatible-base.d.ts +4 -0
  115. package/dist/llm/openai-compatible-base.d.ts.map +1 -1
  116. package/dist/llm/openai-compatible-base.js +38 -13
  117. package/dist/llm/openai-compatible-base.js.map +1 -1
  118. package/dist/llm/openai-responses.js +1 -1
  119. package/dist/llm/openai-responses.js.map +1 -1
  120. package/dist/llm/retry.d.ts +2 -2
  121. package/dist/llm/retry.d.ts.map +1 -1
  122. package/dist/llm/retry.js.map +1 -1
  123. package/dist/logger.d.ts +11 -2
  124. package/dist/logger.d.ts.map +1 -1
  125. package/dist/logger.js +35 -6
  126. package/dist/logger.js.map +1 -1
  127. package/dist/observability/langfuse.d.ts +1 -0
  128. package/dist/observability/langfuse.d.ts.map +1 -1
  129. package/dist/observability/langfuse.js.map +1 -1
  130. package/dist/observability/opentelemetry.d.ts +1 -0
  131. package/dist/observability/opentelemetry.d.ts.map +1 -1
  132. package/dist/observability/opentelemetry.js.map +1 -1
  133. package/dist/reasoning/thought-tree.d.ts +28 -4
  134. package/dist/reasoning/thought-tree.d.ts.map +1 -1
  135. package/dist/reasoning/thought-tree.js +173 -110
  136. package/dist/reasoning/thought-tree.js.map +1 -1
  137. package/dist/runtime.d.ts +29 -3
  138. package/dist/runtime.d.ts.map +1 -1
  139. package/dist/runtime.js +141 -34
  140. package/dist/runtime.js.map +1 -1
  141. package/dist/security/pii.d.ts +2 -2
  142. package/dist/security/pii.d.ts.map +1 -1
  143. package/dist/security/pii.js.map +1 -1
  144. package/dist/time-travel/checkpoint-store.d.ts +6 -1
  145. package/dist/time-travel/checkpoint-store.d.ts.map +1 -1
  146. package/dist/time-travel/checkpoint-store.js +15 -18
  147. package/dist/time-travel/checkpoint-store.js.map +1 -1
  148. package/dist/time-travel/replayer.d.ts +0 -1
  149. package/dist/time-travel/replayer.d.ts.map +1 -1
  150. package/dist/time-travel/replayer.js +12 -19
  151. package/dist/time-travel/replayer.js.map +1 -1
  152. package/dist/time-travel/time-travel.d.ts +0 -1
  153. package/dist/time-travel/time-travel.d.ts.map +1 -1
  154. package/dist/time-travel/time-travel.js +15 -14
  155. package/dist/time-travel/time-travel.js.map +1 -1
  156. package/dist/tool.d.ts.map +1 -1
  157. package/dist/tool.js +1 -0
  158. package/dist/tool.js.map +1 -1
  159. package/dist/tools/index.d.ts +3 -416
  160. package/dist/tools/index.d.ts.map +1 -1
  161. package/dist/tools/index.js +1 -0
  162. package/dist/tools/index.js.map +1 -1
  163. package/dist/tools/sql-query.d.ts.map +1 -1
  164. package/dist/tools/sql-query.js +2 -1
  165. package/dist/tools/sql-query.js.map +1 -1
  166. package/dist/tools/vector-search.d.ts.map +1 -1
  167. package/dist/tools/vector-search.js +12 -3
  168. package/dist/tools/vector-search.js.map +1 -1
  169. package/package.json +5 -5
package/README.md CHANGED
@@ -8,6 +8,19 @@ Core runtime for Cogitator AI agents. Build and run LLM-powered agents with tool
8
8
  pnpm add @cogitator-ai/core zod
9
9
  ```
10
10
 
11
+ Optional peer dependencies, installed only for the features that use them:
12
+
13
+ | Package | Needed for |
14
+ | --------------------------------- | --------------------------------------------------------------- |
15
+ | `@aws-sdk/client-bedrock-runtime` | `BedrockBackend` (`bedrock/...` models) |
16
+ | `@cogitator-ai/sandbox` | Tools with `sandbox: { type: 'docker' \| 'wasm' }` |
17
+ | `pg` | `sqlQuery` / `vectorSearch` on PostgreSQL, `PostgresTraceStore` |
18
+ | `better-sqlite3` | `sqlQuery` on SQLite |
19
+ | `nodemailer` | `sendEmail` over SMTP |
20
+ | `langfuse` | `LangfuseExporter` |
21
+
22
+ Full documentation: [cogitator.app/docs](https://cogitator.app/docs) — start with [Agents](https://cogitator.app/docs/core/agents) and [Cogitator](https://cogitator.app/docs/core/cogitator).
23
+
11
24
  ## Quick Start
12
25
 
13
26
  ```typescript
@@ -48,10 +61,13 @@ console.log(result.output);
48
61
 
49
62
  ## Features
50
63
 
51
- - **Multi-Provider LLM Support** - Ollama, OpenAI, Anthropic, Google, vLLM
52
- - **Type-Safe Tools** - Zod-validated tool definitions
53
- - **Streaming Responses** - Real-time token streaming
54
- - **Memory Integration** - Redis, PostgreSQL, in-memory adapters
64
+ - **Multi-Provider LLM Support** - Ollama, OpenAI, Anthropic, Google, Azure OpenAI, Bedrock, vLLM, Mistral, Groq, Together, DeepSeek
65
+ - **Type-Safe Tools** - Zod-validated tool definitions, `toolset()` for typed tool tuples
66
+ - **Streaming Responses** - Real-time token and reasoning streaming
67
+ - **Structured Output** - `responseFormat` with a Zod schema, validated and repaired once on mismatch
68
+ - **Handoffs & Approvals** - Pass a conversation to another agent; pause runs for human approval and resume them later
69
+ - **Prompt Versions & A/B Tests** - Versioned instructions per agent with `cog.prompts`
70
+ - **Memory Integration** - In-memory, Redis, PostgreSQL, SQLite, MongoDB and Qdrant adapters
55
71
  - **26 Built-in Tools** - Web search, SQL, email, GitHub, filesystem, and more
56
72
  - **Reflection Engine** - Self-improvement through tool call analysis
57
73
  - **Tree-of-Thought** - Advanced reasoning with branch exploration
@@ -113,23 +129,30 @@ const cog = new Cogitator({
113
129
  });
114
130
  ```
115
131
 
132
+ The runtime does not read provider keys from the environment: pass them under `providers` (or build the config with `loadConfig()` from `@cogitator-ai/config`, which reads `OPENAI_API_KEY`, `ANTHROPIC_API_KEY` and the rest). Ollama defaults to `http://localhost:11434`. A model string without a known provider prefix runs on `llm.defaultProvider` (Ollama when unset). An agent without `model` uses `llm.defaultModel`. `llm.backends` registers backends of your own by name (`model: 'my-backend/some-model'`), and `llm.retry` (2 retries with exponential backoff by default, `false` to disable) applies to every backend the runtime creates. Azure (`endpoint`, `apiKey`, `apiVersion`, `deployment`), Bedrock (`region`, credentials) and Mistral / Groq / Together / DeepSeek (`apiKey`) are configured the same way under `providers`.
133
+
134
+ See [LLM Backends](https://cogitator.app/docs/core/llm-backends) for every provider's options.
135
+
116
136
  ### Provider Notes
117
137
 
118
138
  - **OpenAI** — the official backend uses the Responses API and defaults to `gpt-6.1-sol`. Requests are stateless (`store: false`); reasoning items are round-tripped between tool-call turns via `ToolCall.replay`. Reasoning models (o-series, GPT-5+) get no `temperature` / `top_p`. Requests with stop sequences fall back to Chat Completions (the Responses API has no stop parameter). OpenAI-compatible providers (Azure, Mistral, Groq, Together, DeepSeek, vLLM, custom `baseUrl`) stay on Chat Completions. Force either path with `providers.openai.api: 'responses' | 'chat-completions'`. Usage includes `cachedInputTokens` and `reasoningTokens` when reported.
119
139
  - **Anthropic** — defaults to `claude-sonnet-5-5`. Sampling params are omitted for Claude 4.7+, 5.x and Fable (they reject non-default values); Claude 4.0 – 4.6 get at most one of `temperature` / `top_p` (`temperature` wins). `json_schema` uses native structured outputs on Claude 4.5+. Forced tool choice falls back to `auto` with a system-prompt instruction and a one-time warning on Opus/Sonnet 5.5 and Fable.
120
140
  - **Bedrock** — Claude models follow the same sampling and tool-choice rules; `json_object` and `json_schema` response formats are supported (schema enforced via `outputConfig.textFormat` on Claude 4.5 – 4.6, system-prompt instruction otherwise).
141
+ - **Google** — Gemini has no `null` schema type, so `.nullable()` fields in response schemas and tool parameters are sent as `nullable: true`.
121
142
 
122
143
  ### Direct Backend Usage
123
144
 
124
145
  ```typescript
125
146
  import { createLLMBackend, parseModel } from '@cogitator-ai/core';
126
147
 
127
- const backend = createLLMBackend('openai', {
128
- providers: { openai: { apiKey: process.env.OPENAI_API_KEY } },
148
+ const { provider, model } = parseModel('openai/gpt-6.1-sol'); // { provider: 'openai', model: 'gpt-6.1-sol' }
149
+
150
+ const backend = createLLMBackend(provider ?? 'ollama', {
151
+ providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } },
129
152
  });
130
153
 
131
154
  const response = await backend.chat({
132
- model: 'gpt-6.1-sol',
155
+ model,
133
156
  messages: [
134
157
  { role: 'system', content: 'You are helpful.' },
135
158
  { role: 'user', content: 'Hello!' },
@@ -158,6 +181,8 @@ registerLLMBackend(myPlugin);
158
181
  const backend = createLLMBackendFromPlugin('my-provider', { apiKey: '...' });
159
182
  ```
160
183
 
184
+ A backend's `provider` field is an `LLMBackendProvider`: a built-in provider name or one of your own, such as the plugin's provider or its key in `llm.backends`.
185
+
161
186
  ### LLM Debug Wrapper
162
187
 
163
188
  Wrap any backend for request/response logging:
@@ -174,25 +199,28 @@ const debugBackend = withDebug(backend, {
174
199
  ### LLM Error Handling
175
200
 
176
201
  ```typescript
177
- import { LLMError, llmUnavailable, llmTimeout } from '@cogitator-ai/core';
202
+ import { LLMError } from '@cogitator-ai/core';
178
203
 
179
204
  try {
180
205
  await backend.chat(request);
181
206
  } catch (error) {
182
207
  if (error instanceof LLMError) {
183
- console.log('Provider:', error.provider);
184
- console.log('Status:', error.statusCode);
185
- console.log('Retryable:', error.retryable);
208
+ console.log('Provider:', error.provider, error.model);
209
+ console.log('Code:', error.code); // ErrorCode, e.g. LLM_RATE_LIMITED
210
+ console.log('HTTP status from the provider:', error.details?.statusCode);
211
+ console.log('Retryable:', error.retryable, 'after', error.retryAfter, 'ms');
186
212
  }
187
213
  }
188
214
  ```
189
215
 
216
+ `llmUnavailable`, `llmTimeout`, `llmInvalidResponse`, `llmConfigError` and `wrapSDKError` build these errors in your own backends; `withLLMRetry(backend, options)` / `RetryingBackend` add retries with `Retry-After` support to any backend.
217
+
190
218
  ---
191
219
 
192
220
  ## Agent Configuration
193
221
 
194
222
  ```typescript
195
- import { Agent } from '@cogitator-ai/core';
223
+ import { Agent, calculator, webSearch } from '@cogitator-ai/core';
196
224
 
197
225
  const agent = new Agent({
198
226
  id: 'custom-id',
@@ -207,6 +235,7 @@ const agent = new Agent({
207
235
  maxIterations: 15,
208
236
  timeout: 120_000,
209
237
  stopSequences: ['DONE'],
238
+ // responseFormat, reasoning, handoffs, skills, description are covered below
210
239
  });
211
240
 
212
241
  // Clone with modifications
@@ -217,6 +246,63 @@ const variant = agent.clone({
217
246
  });
218
247
  ```
219
248
 
249
+ ### Skills and Serialization
250
+
251
+ A skill bundles tools with the instructions for using them; `skills` merges them into the agent:
252
+
253
+ ```typescript
254
+ import { Agent, defineSkill, httpRequest, ToolRegistry } from '@cogitator-ai/core';
255
+
256
+ const apiSkill = defineSkill({
257
+ name: 'http-api',
258
+ version: '1.0.0',
259
+ description: 'Call JSON APIs',
260
+ tools: [httpRequest],
261
+ instructions: 'Prefer GET requests and summarize responses.',
262
+ env: ['API_TOKEN'], // checked by validateSkill()
263
+ });
264
+
265
+ const agent = new Agent({
266
+ name: 'integrator',
267
+ model: 'openai/gpt-5.5',
268
+ instructions: 'Answer with data from the API.',
269
+ skills: [apiSkill],
270
+ });
271
+
272
+ const snapshot = agent.serialize(); // plain JSON; tools are stored by name
273
+ const registry = new ToolRegistry();
274
+ registry.register(httpRequest);
275
+ const restored = Agent.deserialize(snapshot, { toolRegistry: registry });
276
+ ```
277
+
278
+ See [Agents](https://cogitator.app/docs/core/agents).
279
+
280
+ ### Structured Output
281
+
282
+ `responseFormat` asks the model for JSON. With a Zod schema the answer is validated and parsed into `result.structured`:
283
+
284
+ ```typescript
285
+ import { Agent, Cogitator } from '@cogitator-ai/core';
286
+ import { z } from 'zod';
287
+
288
+ const Weather = z.object({ city: z.string(), celsius: z.number() });
289
+
290
+ const extractor = new Agent({
291
+ name: 'extractor',
292
+ model: 'openai/gpt-5.5',
293
+ instructions: 'Extract the weather report.',
294
+ responseFormat: { type: 'json_schema', schema: Weather }, // or { type: 'json' } for any JSON
295
+ });
296
+
297
+ const cog = new Cogitator({
298
+ llm: { providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } } },
299
+ });
300
+ const result = await cog.run(extractor, { input: 'Paris, 21 degrees' });
301
+ const weather = Weather.parse(result.structured);
302
+ ```
303
+
304
+ When the final answer does not fit the schema, the run asks the model once more with the validation problem (for example `celsius: expected number, received string`) and does not save the rejected answer to the thread; if the retry fails too, `structured` is `undefined`. Streamed runs keep the first answer, since the client has already seen it. JSON wrapped in prose or code fences is still read. See [Structured Outputs](https://cogitator.app/docs/core/structured-outputs).
305
+
220
306
  ### Reasoning and Prompt Caching
221
307
 
222
308
  `reasoning` sets how hard a reasoning model thinks, in one vocabulary for every provider (Anthropic adaptive thinking and effort, OpenAI `reasoning.effort`, Gemini thinking levels or budgets, Ollama `think`), and can ask for a readable summary:
@@ -266,6 +352,46 @@ const weatherTool = tool({
266
352
  });
267
353
  ```
268
354
 
355
+ `tool()` also takes `category`, `tags`, `sideEffects`, `requiresApproval`, `timeout` and `sandbox`. A parameter with `.default()` is optional in the JSON Schema the model sees; `execute` receives the default when the model leaves it out. Tools passed to `new Agent({ tools })` lose their individual parameter types in a plain array; `toolset(...tools)` keeps them as a typed tuple that agents still accept:
356
+
357
+ ```typescript
358
+ import { tool, toolset } from '@cogitator-ai/core';
359
+ import { z } from 'zod';
360
+
361
+ function createSearchTools() {
362
+ return toolset(
363
+ tool({
364
+ name: 'search',
365
+ description: 'Search the catalog',
366
+ parameters: z.object({ query: z.string() }),
367
+ execute: async ({ query }) => ({ hits: [query] }),
368
+ }),
369
+ tool({
370
+ name: 'fetch_item',
371
+ description: 'Fetch one item',
372
+ parameters: z.object({ id: z.number() }),
373
+ execute: async ({ id }) => ({ id }),
374
+ })
375
+ );
376
+ }
377
+
378
+ const [search, fetchItem] = createSearchTools();
379
+ await search.execute({ query: 'lamp' }, ctx); // typed as { query: string }
380
+ ```
381
+
382
+ A result object with a base64 image in `image` or `imageBase64` (PNG, JPEG, GIF or WebP, plain or as a `data:` URL) reaches the model as an image, with the rest of the result as JSON, so a vision model sees a screenshot instead of its base64 text. Anthropic, Bedrock and the OpenAI Responses API get the image inside the tool result, Google after the turn's function responses, Ollama in the tool message's `images`, and Chat Completions backends (OpenAI-compatible, Azure) in a user message after the turn's tool messages:
383
+
384
+ ```typescript
385
+ const screenshot = tool({
386
+ name: 'screenshot',
387
+ description: 'Capture the dashboard as a PNG',
388
+ parameters: z.object({}),
389
+ execute: async () => ({ page: 'dashboard', image: (await capture()).toString('base64') }),
390
+ });
391
+ ```
392
+
393
+ See [Tools](https://cogitator.app/docs/core/tools) and [Custom Tools](https://cogitator.app/docs/tools/custom-tools).
394
+
269
395
  ### Handoffs
270
396
 
271
397
  `handoffs` lets an agent pass the conversation to another one: each target becomes a `transfer_to_<name>` tool, and the rest of the run goes on as the target — its instructions, tools and model — with the whole conversation:
@@ -273,7 +399,7 @@ const weatherTool = tool({
273
399
  ```typescript
274
400
  const triage = new Agent({
275
401
  name: 'triage',
276
- model,
402
+ model: 'openai/gpt-5.5',
277
403
  instructions: 'Hand the customer to the right specialist.',
278
404
  handoffs: [billing, techSupport],
279
405
  });
@@ -283,6 +409,8 @@ result.handoffs; // [{ from: 'triage', to: 'billing', reason }]
283
409
  result.finalAgent; // 'billing'
284
410
  ```
285
411
 
412
+ A handoff can also be `{ agent, toolName, description }` to name the tool or describe when to use it. `onHandoff` on the run options reports each switch as it happens.
413
+
286
414
  ### Approvals
287
415
 
288
416
  A tool with `requiresApproval` (`true` or a function of its arguments) never runs without a person's decision. Decide inline with `onApproval`, or let the run pause and resume it later:
@@ -301,6 +429,8 @@ if (result.status === 'paused') {
301
429
 
302
430
  Nothing of the paused turn runs until every call in it is decided. Paused runs live in the thread's memory (or process memory, or your `runCheckpoints` store), so a resume can come after a restart; a new message on the thread instead declines the waiting calls.
303
431
 
432
+ `cog.resume()` takes the thread id or the returned `result.checkpoint`; `defaultDecision` answers every call `decisions` leaves out. To decide while the run waits, pass `onApproval: (request) => ({ approved: true })` (or return `'pause'`) to `cog.run()`. See [Tool Approvals](https://cogitator.app/docs/tools/approvals).
433
+
304
434
  ### PII Masking
305
435
 
306
436
  `security.pii` replaces emails, phones, card numbers (Luhn-checked), IBANs, SSNs, IP addresses, API keys and your own patterns with placeholders before every LLM request, so the provider never sees them:
@@ -317,7 +447,7 @@ const cog = new Cogitator({
317
447
  });
318
448
  ```
319
449
 
320
- In `mask` mode the answer, its stream and tool call arguments get the real values back, so `send_email({ to: '[EMAIL_1]' })` reaches the tool as the real address. `PiiMasker`, `PiiVault` and `withPiiMasking` work outside a run too.
450
+ In `mask` mode the answer, its stream and tool call arguments get the real values back, so `send_email({ to: '[EMAIL_1]' })` reaches the tool as the real address. `detect` limits the built-in kinds (`PII_TYPES`: `email`, `phone`, `credit_card`, `iban`, `ssn`, `ip_address`, `api_key`). `PiiMasker`, `PiiVault` and `withPiiMasking` work outside a run too. See [Security](https://cogitator.app/docs/advanced/security).
321
451
 
322
452
  ### Tool Context
323
453
 
@@ -327,7 +457,11 @@ Every tool receives a context object:
327
457
  interface ToolContext {
328
458
  agentId: string;
329
459
  runId: string;
330
- signal: AbortSignal;
460
+ signal: AbortSignal; // aborted on run cancel or tool timeout
461
+ threadId?: string;
462
+ userId?: string; // the run's userId
463
+ channelType?: string;
464
+ channelId?: string;
331
465
  }
332
466
  ```
333
467
 
@@ -336,6 +470,9 @@ interface ToolContext {
336
470
  Execute tools in isolated Docker or WASM environments:
337
471
 
338
472
  ```typescript
473
+ import { tool } from '@cogitator-ai/core';
474
+ import { z } from 'zod';
475
+
339
476
  const shellTool = tool({
340
477
  name: 'run_shell',
341
478
  description: 'Execute shell commands safely',
@@ -351,6 +488,10 @@ const shellTool = tool({
351
488
  });
352
489
  ```
353
490
 
491
+ A Docker-sandboxed tool does not call `execute`: the sandbox runs the `command` argument with `sh -c` (plus optional `cwd` / `env` arguments) and returns its output. A WASM tool gets its arguments as JSON on stdin and its JSON stdout is parsed as the result. Sandboxing needs `@cogitator-ai/sandbox` installed (options go in `new Cogitator({ sandbox })`).
492
+
493
+ When Docker is unavailable (or `@cogitator-ai/sandbox` is missing), a Docker-sandboxed tool runs its command directly on the host with a warning; set `sandbox.allowNativeFallback: false` to make those calls fail instead. WASM tools never fall back to the host: they run their own `execute` only when `@cogitator-ai/sandbox` is missing or fails to start, and a WASM sandbox that cannot load the module returns an error. Each Docker execution gets a fresh container (the pool keeps them warm); `sandbox.pool.reuseContainers: true` reuses containers between executions with the same settings, which is faster but lets files and processes leak from one execution to the next.
494
+
354
495
  `timeout` is enforced for every tool: native tools get an aborted `context.signal` and the model receives a `Tool "<name>" timed out after <ms>ms` error; sandboxed tools forward it to the sandbox executor. The sandbox is initialized lazily on the first sandboxed call, and that call already runs inside it.
355
496
 
356
497
  ### Tool Registry
@@ -433,25 +574,38 @@ Not part of `builtinTools`. `createAnalyzeImageTool` takes an `llm` backend; the
433
574
  | `createGenerateSpeechTool` | `generateSpeech` | `gpt-4o-mini-tts` |
434
575
 
435
576
  ```typescript
436
- import { builtinTools, calculator, datetime } from '@cogitator-ai/core';
577
+ import { Agent, builtinTools } from '@cogitator-ai/core';
437
578
 
438
579
  const agent = new Agent({
439
580
  name: 'utility-agent',
440
581
  instructions: 'Use your tools to help users',
441
582
  model: 'openai/gpt-6.1-sol',
442
- tools: builtinTools,
583
+ tools: builtinTools, // Tool[]
443
584
  });
444
585
  ```
445
586
 
587
+ #### Assistant Tool Factories
588
+
589
+ Also not in `builtinTools`, these build tools around your own stores. `createMemoryTools` and `createSchedulerTools` return typed tuples (see `toolset()`), so destructured tools keep their parameter types:
590
+
591
+ | Factory | Tools |
592
+ | -------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------- |
593
+ | `createMemoryTools({ graphAdapter, agentId, coreFacts?, embeddingFn? })` | `remember`, `recall`, `forget` on a knowledge graph |
594
+ | `createSchedulerTools({ store, defaultChannel?, defaultUserId? })` | `schedule_task`, `list_tasks`, `cancel_task` on a `TimerStore` |
595
+ | `createDeviceTools()`, `createCapabilitiesTool(doc)`, `createSelfTools({ toolsDir })` / `loadCustomTools(dir)` | Device info, a capabilities description, and tools an agent writes to a directory |
596
+
446
597
  ### Web Search Tool
447
598
 
448
599
  Search the web using Tavily, Brave, or Serper APIs:
449
600
 
450
601
  ```typescript
451
- import { webSearch } from '@cogitator-ai/core';
602
+ import { Agent, webSearch } from '@cogitator-ai/core';
452
603
 
453
604
  // Auto-detects from TAVILY_API_KEY, BRAVE_API_KEY, or SERPER_API_KEY
454
605
  const agent = new Agent({
606
+ name: 'assistant',
607
+ instructions: 'Use your tools to help users',
608
+ model: 'openai/gpt-5.5',
455
609
  tools: [webSearch],
456
610
  });
457
611
 
@@ -464,13 +618,16 @@ const agent = new Agent({
464
618
  Extract content from web pages:
465
619
 
466
620
  ```typescript
467
- import { webScrape } from '@cogitator-ai/core';
621
+ import { Agent, webScrape } from '@cogitator-ai/core';
468
622
 
469
623
  const agent = new Agent({
624
+ name: 'assistant',
625
+ instructions: 'Use your tools to help users',
626
+ model: 'openai/gpt-5.5',
470
627
  tools: [webScrape],
471
628
  });
472
629
 
473
- // Supports CSS selectors, text/markdown/html output, link/image extraction
630
+ // Supports simple selectors (tag, .class, #id), text/markdown/html output, link/image extraction
474
631
  ```
475
632
 
476
633
  ### SQL Query Tool
@@ -478,9 +635,12 @@ const agent = new Agent({
478
635
  Execute SQL queries against PostgreSQL or SQLite:
479
636
 
480
637
  ```typescript
481
- import { sqlQuery } from '@cogitator-ai/core';
638
+ import { Agent, sqlQuery } from '@cogitator-ai/core';
482
639
 
483
640
  const agent = new Agent({
641
+ name: 'assistant',
642
+ instructions: 'Use your tools to help users',
643
+ model: 'openai/gpt-5.5',
484
644
  tools: [sqlQuery],
485
645
  });
486
646
 
@@ -498,14 +658,17 @@ Read-only queries on PostgreSQL run inside `BEGIN TRANSACTION READ ONLY` (always
498
658
  Semantic search using embeddings with pgvector:
499
659
 
500
660
  ```typescript
501
- import { vectorSearch } from '@cogitator-ai/core';
661
+ import { Agent, vectorSearch } from '@cogitator-ai/core';
502
662
 
503
663
  const agent = new Agent({
664
+ name: 'assistant',
665
+ instructions: 'Use your tools to help users',
666
+ model: 'openai/gpt-5.5',
504
667
  tools: [vectorSearch],
505
668
  });
506
669
 
507
670
  // Embedding providers: OpenAI, Ollama, Google
508
- // Auto-detects from OPENAI_API_KEY, OLLAMA_BASE_URL / OLLAMA_HOST, or GOOGLE_API_KEY
671
+ // Auto-detects from OPENAI_API_KEY, OLLAMA_BASE_URL / OLLAMA_URL / OLLAMA_HOST, or GOOGLE_API_KEY
509
672
  // Default models: text-embedding-3-small, nomic-embed-text, gemini-embedding-001
510
673
  ```
511
674
 
@@ -516,9 +679,12 @@ const agent = new Agent({
516
679
  Send emails via Resend API or SMTP:
517
680
 
518
681
  ```typescript
519
- import { sendEmail } from '@cogitator-ai/core';
682
+ import { Agent, sendEmail } from '@cogitator-ai/core';
520
683
 
521
684
  const agent = new Agent({
685
+ name: 'assistant',
686
+ instructions: 'Use your tools to help users',
687
+ model: 'openai/gpt-5.5',
522
688
  tools: [sendEmail],
523
689
  });
524
690
 
@@ -534,9 +700,12 @@ const agent = new Agent({
534
700
  Interact with GitHub repositories:
535
701
 
536
702
  ```typescript
537
- import { githubApi } from '@cogitator-ai/core';
703
+ import { Agent, githubApi } from '@cogitator-ai/core';
538
704
 
539
705
  const agent = new Agent({
706
+ name: 'assistant',
707
+ instructions: 'Use your tools to help users',
708
+ model: 'openai/gpt-5.5',
540
709
  tools: [githubApi],
541
710
  });
542
711
 
@@ -553,8 +722,15 @@ const agent = new Agent({
553
722
  ```typescript
554
723
  import { Cogitator, Agent } from '@cogitator-ai/core';
555
724
 
556
- const cog = new Cogitator();
557
- const agent = new Agent({/* ... */});
725
+ const cog = new Cogitator({
726
+ llm: { providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } } },
727
+ });
728
+ const agent = new Agent({
729
+ name: 'analyst',
730
+ instructions: 'Analyze data.',
731
+ model: 'openai/gpt-5.5',
732
+ });
733
+ const controller = new AbortController();
558
734
 
559
735
  const result = await cog.run(agent, {
560
736
  input: 'Analyze this data...',
@@ -562,16 +738,24 @@ const result = await cog.run(agent, {
562
738
  audio: [{ data: base64Wav, format: 'wav' }],
563
739
 
564
740
  threadId: 'thread_123',
565
- context: { userId: 'user_456', task: 'analysis' },
741
+ userId: 'user_456', // owns the thread, scopes memory, reaches tools as context.userId
742
+ threadAccess: 'owner', // default; 'shared' lets any user continue the thread
743
+ context: { task: 'analysis' }, // extra values for the system prompt
566
744
 
567
745
  timeout: 60000,
746
+ signal: controller.signal,
568
747
  stream: true,
569
748
  onToken: (token) => process.stdout.write(token),
749
+ onReasoning: (delta) => process.stdout.write(delta),
750
+ reasoning: { effort: 'low' }, // overrides the agent's reasoning for this run
570
751
 
571
752
  useMemory: true,
572
753
  loadHistory: true,
573
754
  saveHistory: true,
755
+ parallelToolCalls: false, // default: tool calls of one turn run one after another
574
756
 
757
+ onApproval: (request) => ({ approved: request.toolName !== 'delete_account' }),
758
+ onHandoff: (handoff) => console.log(`${handoff.from} -> ${handoff.to}`),
575
759
  onToolCall: (call) => console.log('Tool:', call.name),
576
760
  onToolResult: (result) => console.log('Result:', result.result),
577
761
  onSpan: (span) => console.log('Span:', span.name),
@@ -589,16 +773,28 @@ const result = await cog.run(agent, {
589
773
  ```typescript
590
774
  interface RunResult {
591
775
  output: string;
776
+ structured?: unknown; // parsed output when the agent has a responseFormat
592
777
  runId: string;
593
778
  agentId: string;
594
779
  threadId: string;
780
+ modelUsed?: string; // differs from agent.model when cost routing picked another
595
781
  usage: {
596
782
  inputTokens: number;
597
783
  outputTokens: number;
598
784
  totalTokens: number;
599
785
  cost: number;
600
786
  duration: number;
787
+ reasoningTokens?: number;
788
+ cachedInputTokens?: number;
789
+ cacheWriteTokens?: number;
601
790
  };
791
+ reasoning?: string; // reasoning summary, with reasoning.summary
792
+ prompt?: RunPrompt; // versioned instructions / A/B variant used
793
+ handoffs?: HandoffEvent[];
794
+ finalAgent?: string;
795
+ status?: 'completed' | 'paused';
796
+ pendingApprovals?: ToolApprovalRequest[];
797
+ checkpoint?: RunCheckpoint; // pass to cog.resume()
602
798
  toolCalls: ToolCall[];
603
799
  messages: Message[];
604
800
  trace: { traceId: string; spans: Span[] };
@@ -607,6 +803,8 @@ interface RunResult {
607
803
  }
608
804
  ```
609
805
 
806
+ All fields are `readonly`. The run timeout comes from the run, the agent, `limits.defaultTimeout`, or 120 s.
807
+
610
808
  ---
611
809
 
612
810
  ## Memory Integration
@@ -649,6 +847,21 @@ const cog = new Cogitator({
649
847
  });
650
848
  ```
651
849
 
850
+ `adapter` also accepts `'sqlite'` (`sqlite.path`), `'mongodb'` (`mongodb.uri`) and `'redis'` (`redis.url`, or `redis.host` + `redis.port`, or `redis.cluster`). Qdrant stores embeddings, not threads: configure it in `memory.qdrant` next to a thread adapter, together with `memory.embedding` and `memory.contextBuilder.includeSemanticContext`, and semantic context is retrieved from it. The Postgres adapter sizes its vector column to the `memory.embedding` model. Runs with a `threadId` load history from and save messages to the adapter.
851
+
852
+ The adapter connects on the first run. To read threads before that (for example in an API route), use `getMemory()`, which connects it on first use; `cog.memory` stays `undefined` until something connected it:
853
+
854
+ ```typescript
855
+ import { unwrap } from '@cogitator-ai/memory';
856
+
857
+ const memory = await cog.getMemory(); // undefined when memory is not configured
858
+ if (memory) {
859
+ const entries = unwrap(await memory.getEntries({ threadId: 'thread_123', limit: 20 }));
860
+ }
861
+ ```
862
+
863
+ See [Memory](https://cogitator.app/docs/memory) and [Memory Adapters](https://cogitator.app/docs/memory/adapters).
864
+
652
865
  ---
653
866
 
654
867
  ## Reflection Engine
@@ -656,7 +869,7 @@ const cog = new Cogitator({
656
869
  Enable self-improvement through reflection on tool calls and runs:
657
870
 
658
871
  ```typescript
659
- import { Cogitator, ReflectionEngine, InMemoryInsightStore } from '@cogitator-ai/core';
872
+ import { Cogitator } from '@cogitator-ai/core';
660
873
 
661
874
  const cog = new Cogitator({
662
875
  reflection: {
@@ -683,13 +896,16 @@ console.log('Learned insights:', insights);
683
896
  ```typescript
684
897
  import { ReflectionEngine, InMemoryInsightStore, createLLMBackend } from '@cogitator-ai/core';
685
898
 
686
- const backend = createLLMBackend('openai', { apiKey: '...' });
899
+ const backend = createLLMBackend('openai', {
900
+ providers: { openai: { apiKey: process.env.OPENAI_API_KEY! } },
901
+ });
687
902
  const insightStore = new InMemoryInsightStore();
688
903
 
689
904
  const engine = new ReflectionEngine({
690
905
  llm: backend,
691
906
  insightStore,
692
907
  config: {
908
+ enabled: true,
693
909
  reflectAfterToolCall: true,
694
910
  minConfidenceToStore: 0.7,
695
911
  },
@@ -701,6 +917,8 @@ if (result.shouldAdjustStrategy) {
701
917
  }
702
918
  ```
703
919
 
920
+ `reflectOnError`, `reflectOnRun`, `getRelevantInsights` and `getSummary(agentId)` cover the other stages. See [Reflection](https://cogitator.app/docs/advanced/reflection).
921
+
704
922
  ---
705
923
 
706
924
  ## Tree-of-Thought Reasoning
@@ -708,38 +926,49 @@ if (result.shouldAdjustStrategy) {
708
926
  Explore multiple reasoning paths before deciding:
709
927
 
710
928
  ```typescript
711
- import { ThoughtTreeExecutor, BranchGenerator, BranchEvaluator } from '@cogitator-ai/core';
929
+ import { ThoughtTreeExecutor } from '@cogitator-ai/core';
712
930
 
713
- const executor = new ThoughtTreeExecutor(cogitator, {
714
- maxBranches: 5,
931
+ const executor = new ThoughtTreeExecutor(cog, {
932
+ branchFactor: 3,
715
933
  maxDepth: 3,
716
934
  explorationStrategy: 'best-first',
717
- pruneThreshold: 0.3,
935
+ confidenceThreshold: 0.3,
718
936
  });
719
937
 
720
- const result = await executor.run(agent, {
721
- input: 'Solve this complex problem...',
722
- explorationBudget: 10,
938
+ const result = await executor.explore(agent, 'Solve this complex problem...', {
939
+ timeout: 60_000,
940
+ onProgress: (stats) => console.log('Explored:', stats.exploredNodes),
723
941
  });
724
942
 
725
- console.log('Best path:', result.bestPath);
726
- console.log('All branches explored:', result.tree.branches.length);
943
+ console.log('Answer:', result.output);
944
+ console.log(
945
+ 'Best path:',
946
+ result.bestPath.map((node) => node.branch.thought)
947
+ );
948
+ console.log('Nodes in tree:', result.tree.nodes.size);
727
949
  console.log('Stats:', result.stats);
728
950
  ```
729
951
 
730
952
  ### ToT Configuration
731
953
 
954
+ Every field is optional; defaults are in `DEFAULT_TOT_CONFIG`:
955
+
732
956
  ```typescript
733
- const executor = new ThoughtTreeExecutor(cogitator, {
734
- maxBranches: 5,
735
- maxDepth: 3,
736
- explorationStrategy: 'breadth-first',
737
- pruneThreshold: 0.3,
738
- branchTemperature: 0.8,
739
- evaluationModel: 'openai/gpt-6-luna',
957
+ const executor = new ThoughtTreeExecutor(cog, {
958
+ branchFactor: 3, // branches generated per node
959
+ beamWidth: 2, // best candidates queued per expanded node
960
+ maxDepth: 5,
961
+ explorationStrategy: 'beam', // 'beam' | 'best-first' | 'dfs'
962
+ confidenceThreshold: 0.3, // prune branches scored below
963
+ terminationConfidence: 0.8, // stop once a branch reaches this
964
+ maxTotalNodes: 50, // executed nodes
965
+ maxIterationsPerBranch: 3, // iteration cap for each branch run
966
+ onBranchEvaluated: (branch, score) => console.log(branch.thought, score.composite),
740
967
  });
741
968
  ```
742
969
 
970
+ Candidates beyond `beamWidth` stay pending, so a failed branch backtracks to the next best one. `beam` runs the tree level by level, `best-first` always runs the node whose own branch scored highest, and `dfs` goes deep first. The executor's own generation, evaluation and synthesis calls use the agent's routed model and count into `usage` (tokens and cost) and `stats`. See [Tree-of-Thought](https://cogitator.app/docs/advanced/reasoning).
971
+
743
972
  ---
744
973
 
745
974
  ## Agent Optimizer (Learning)
@@ -751,10 +980,10 @@ import { AgentOptimizer, InMemoryTraceStore } from '@cogitator-ai/core';
751
980
 
752
981
  const traceStore = new InMemoryTraceStore();
753
982
  const optimizer = new AgentOptimizer({
754
- llm: cogitator.getLLMBackend('openai/gpt-6.1-sol'),
983
+ llm: cog.getLLMBackend('openai/gpt-6.1-sol'),
755
984
  model: 'gpt-6.1-sol',
756
985
  traceStore,
757
- cogitator,
986
+ cogitator: cog,
758
987
  });
759
988
 
760
989
  const result = await optimizer.compile(
@@ -779,35 +1008,65 @@ import {
779
1008
  createSuccessMetric,
780
1009
  createExactMatchMetric,
781
1010
  createContainsMetric,
782
- MetricEvaluator,
783
1011
  } from '@cogitator-ai/core';
784
1012
 
785
- const successMetric = createSuccessMetric();
786
-
787
- const exactMatch = createExactMatchMetric();
788
-
789
- const containsMetric = createContainsMetric(['error', 'failed'], { negate: true });
1013
+ const successMetric = createSuccessMetric(); // 1 when no tool call failed
1014
+ const exactMatch = createExactMatchMetric(); // compares the output (or a field path) with `expected`
1015
+ const containsMetric = createContainsMetric(['refund', 'order']); // share of keywords found, passes at >= 0.5
790
1016
  ```
791
1017
 
1018
+ Each is a `MetricFn` (`(trace, expected?) => MetricResult`); `MetricEvaluator` combines built-in and custom metrics, including LLM-judged ones.
1019
+
792
1020
  ### Demo Selection
793
1021
 
794
1022
  ```typescript
795
- import { DemoSelector } from '@cogitator-ai/core';
1023
+ import { DemoSelector, InMemoryTraceStore } from '@cogitator-ai/core';
796
1024
 
797
1025
  const selector = new DemoSelector({
798
- strategy: 'diverse',
799
- maxDemos: 5,
1026
+ traceStore: new InMemoryTraceStore(),
1027
+ maxDemos: 5, // per agent (default 10)
1028
+ minScore: 0.8, // traces scoring lower are not used as demos
1029
+ diversityWeight: 0.3,
800
1030
  });
801
1031
 
802
- const selectedDemos = selector.select(allDemos, currentInput);
1032
+ await selector.addDemo(trace);
1033
+ const demos = await selector.selectDemos(agent.id, currentInput, 3);
1034
+ const fewShot = selector.formatDemosForPrompt(demos);
803
1035
  ```
804
1036
 
1037
+ See [Learning](https://cogitator.app/docs/advanced/learning).
1038
+
805
1039
  ---
806
1040
 
807
1041
  ## Prompt Auto-Optimization
808
1042
 
809
1043
  Capture prompts, run A/B tests, monitor performance, and automatically optimize agent instructions.
810
1044
 
1045
+ ### Prompt Versions
1046
+
1047
+ `cog.prompts` versions an agent's instructions on top of the ones in code. A deployed version is used by every following run of that agent, and each run's outcome is recorded against the version (or A/B variant) it used:
1048
+
1049
+ ```typescript
1050
+ const v2 = await cog.prompts.deploy(writer, 'You write release notes. Lead with what changed.');
1051
+
1052
+ const result = await cog.run(writer, { input, threadId });
1053
+ result.prompt; // { key: 'writer', versionId: v2.id, version: 2 }
1054
+
1055
+ await cog.prompts.rollbackTo(writer); // previous version, recorded as a new one
1056
+ await cog.prompts.history(writer); // newest first, with metrics
1057
+
1058
+ await cog.prompts.startABTest(writer, {
1059
+ name: 'shorter notes',
1060
+ treatment: 'You write release notes in at most five bullet points.',
1061
+ treatmentAllocation: 0.3, // share of threads
1062
+ minSampleSize: 50,
1063
+ });
1064
+ ```
1065
+
1066
+ Versions are kept per agent `id` when set, else per `name`. They live in process memory unless `new Cogitator({ prompts: { versions, abTests } })` gets durable stores (`PostgresTraceStore` provides both via `instructionVersions()` and `abTests()`); `prompts.score` scores runs (default: 1 for a completed run) and `prompts.autoDeployWinner` deploys a significant A/B winner. See [Prompt Versions](https://cogitator.app/docs/advanced/prompt-versions).
1067
+
1068
+ The classes below are the building blocks `cog.prompts` uses, available for your own pipelines.
1069
+
811
1070
  ### Prompt Logger
812
1071
 
813
1072
  Wrap any LLM backend to capture all prompts:
@@ -819,11 +1078,14 @@ const store = new PostgresTraceStore({
819
1078
  connectionString: process.env.DATABASE_URL!,
820
1079
  });
821
1080
 
1081
+ await store.connect();
1082
+
822
1083
  const wrappedBackend = wrapWithPromptLogger(openaiBackend, store, {
823
- captureSystemPrompt: true,
1084
+ captureContent: true,
824
1085
  captureTools: true,
825
- captureResponse: true,
826
1086
  });
1087
+
1088
+ wrappedBackend.setContext({ runId, agentId: agent.id, threadId });
827
1089
  ```
828
1090
 
829
1091
  ### A/B Testing Framework
@@ -880,9 +1142,9 @@ const monitor = new PromptMonitor({
880
1142
 
881
1143
  const alerts = monitor.recordExecution(trace);
882
1144
 
883
- const metrics = monitor.getCurrentMetrics('agent-1');
884
- console.log('Avg score:', metrics.avgScore);
885
- console.log('P95 latency:', metrics.p95Latency);
1145
+ const metrics = monitor.getCurrentMetrics('agent-1'); // null before any execution
1146
+ console.log('Avg score:', metrics?.avgScore);
1147
+ console.log('P95 latency:', metrics?.p95Latency);
886
1148
  ```
887
1149
 
888
1150
  ### Rollback Manager
@@ -937,6 +1199,8 @@ const optimizer = new AutoOptimizer({
937
1199
  await optimizer.recordExecution(trace);
938
1200
  ```
939
1201
 
1202
+ It optimizes the agent's deployed version, so deploy one first. To serve its A/B tests on live traffic, give `new Cogitator({ prompts: { abTests, versions } })` the same stores as `abTesting` and `rollbackManager` and the agent an explicit `id`: the Cogitator then assigns variants per thread and records the results, traces carry the variant (`trace.prompt`) so the optimizer does not count them twice, and the optimizer deploys the winner (leave `prompts.autoDeployWinner` off). `PostgresTraceStore` backs all of it: `traces()` for `AgentOptimizer`, `abTests()` and `instructionVersions()` for the rest. See [Learning](https://cogitator.app/docs/advanced/learning#ab-tests-on-live-traffic).
1203
+
940
1204
  ---
941
1205
 
942
1206
  ## Time Travel Debugging
@@ -949,10 +1213,10 @@ import { TimeTravel, InMemoryCheckpointStore } from '@cogitator-ai/core';
949
1213
  const timeTravel = new TimeTravel(cogitator);
950
1214
 
951
1215
  const result = await cogitator.run(agent, { input: 'Original task...' });
952
- const checkpoints = await timeTravel.checkpointAll(result, 'original');
1216
+ const checkpoints = await timeTravel.checkpointAll(result, 'original'); // one per tool call
953
1217
 
954
1218
  const replayResult = await timeTravel.replayLive(agent, checkpoints[2].id);
955
- console.log('Replayed from step 2:', replayResult.output);
1219
+ console.log('Replayed from the third tool call:', replayResult.output);
956
1220
 
957
1221
  const forkResult = await timeTravel.fork(agent, checkpoints[2].id, {
958
1222
  input: 'Modified task...',
@@ -1006,7 +1270,7 @@ const liveReplay = await timeTravel.replayLive(agent, checkpointId, {
1006
1270
  });
1007
1271
  ```
1008
1272
 
1009
- Mocked tool results are keyed by tool name (in deterministic replays also by call id): the tool answers with the given value and never runs. Tools in `skipTools` are removed from the replayed agent. Checkpoints, replays and forks store their traces in the trace store, so `compare()` and `compareWithOriginal()` can read them.
1273
+ Checkpoints are anchored on tool calls: checkpoint `i` holds the conversation after `i` tool calls, stopping before the result of the call it is anchored on. `stepsReplayed` is that index, `stepsExecuted` counts the replay's tool calls, and `divergedAt` uses the original run's numbering. Mocked tool results are keyed by tool name (in deterministic replays also by call id): the tool answers with the given value and never runs. Tools in `skipTools` are removed from the replayed agent. Checkpoints, replays and forks store their traces in the trace store, so `compare()` and `compareWithOriginal()` can read them.
1010
1274
 
1011
1275
  ---
1012
1276
 
@@ -1043,8 +1307,8 @@ const engine = new CausalInferenceEngine(graph);
1043
1307
  ```typescript
1044
1308
  const identifiable = engine.isIdentifiable('X', 'Y');
1045
1309
  if (identifiable.identifiable) {
1046
- console.log('Effect is identifiable via:', identifiable.method);
1047
- console.log('Adjustment set:', identifiable.adjustmentSet);
1310
+ console.log('Effect is identifiable:', identifiable.reason); // e.g. backdoor criterion
1311
+ console.log('Adjustment set:', identifiable.adjustmentSet?.variables);
1048
1312
  }
1049
1313
  ```
1050
1314
 
@@ -1054,11 +1318,11 @@ if (identifiable.identifiable) {
1054
1318
  const effect = engine.computeInterventionalEffect({
1055
1319
  target: 'Y',
1056
1320
  interventions: { X: 1 },
1057
- observed: { Z: 0.5 },
1321
+ conditions: { Z: 0.5 },
1058
1322
  });
1059
1323
 
1060
1324
  console.log('Expected effect:', effect.effect);
1061
- console.log('Confidence:', effect.confidence);
1325
+ console.log('Formula:', effect.formula, 'identifiable:', effect.isIdentifiable);
1062
1326
  ```
1063
1327
 
1064
1328
  ### Counterfactual Reasoning
@@ -1077,6 +1341,8 @@ console.log('Factual value:', result.factualValue);
1077
1341
  console.log('Counterfactual value:', result.counterfactualValue);
1078
1342
  ```
1079
1343
 
1344
+ Counterfactuals use the nodes' structural equations (`withEquation()` on the builder): `linear`, `logistic`, `polynomial`, or `custom` with a safe arithmetic expression over the parent ids in `customFn` (`'2 * price - log(demand)'`; `+ - * / ^`, parentheses, `abs exp log sqrt pow min max tanh sigmoid`). Nodes without an equation keep their factual values.
1345
+
1080
1346
  ### D-Separation Analysis
1081
1347
 
1082
1348
  ```typescript
@@ -1098,16 +1364,19 @@ import { CausalExtractor, CausalHypothesisGenerator } from '@cogitator-ai/core';
1098
1364
 
1099
1365
  const extractor = new CausalExtractor({ llmBackend: backend });
1100
1366
 
1101
- const relations = await extractor.extractFromToolResult(
1102
- { name: 'database_query', arguments: { table: 'users' } },
1367
+ const { nodes, edges } = await extractor.extractFromToolResult(
1368
+ { id: 'call_1', name: 'database_query', arguments: { table: 'users' } },
1103
1369
  { rows: 100, cached: true },
1104
- { agentId: 'agent-1' }
1370
+ { taskDescription: 'Count active users' },
1371
+ graph
1105
1372
  );
1106
1373
 
1107
1374
  const generator = new CausalHypothesisGenerator({ llmBackend: backend });
1108
1375
  const hypotheses = await generator.generateFromFailure(trace, { agentId: 'agent-1' });
1109
1376
  ```
1110
1377
 
1378
+ `CausalReasoner` ties these together for agents (`predictEffect`, `explainCause`, `planForGoal`, `evaluateToolCall`, `analyzeErrorCausally`). See [Causal Reasoning](https://cogitator.app/docs/advanced/causal-reasoning).
1379
+
1111
1380
  ---
1112
1381
 
1113
1382
  ## Error Handling & Resilience
@@ -1117,6 +1386,8 @@ const hypotheses = await generator.generateFromFailure(trace, { agentId: 'agent-
1117
1386
  Agent runs retry failed LLM calls on their own: rate limits, 5xx, timeouts and dropped connections, with exponential backoff and the provider's `Retry-After`. Streams are retried only before the first chunk. Tune or turn this off with `llm.retry`:
1118
1387
 
1119
1388
  ```typescript
1389
+ import { Cogitator, GoogleBackend, withLLMRetry } from '@cogitator-ai/core';
1390
+
1120
1391
  const cog = new Cogitator({
1121
1392
  llm: {
1122
1393
  retry: { maxRetries: 3, maxRetryAfter: 30_000, onRetry: (e) => console.warn(e) },
@@ -1125,7 +1396,9 @@ const cog = new Cogitator({
1125
1396
  });
1126
1397
 
1127
1398
  // A backend used on its own
1128
- const backend = withLLMRetry(new GoogleBackend({ apiKey }), { maxRetries: 3 });
1399
+ const backend = withLLMRetry(new GoogleBackend({ apiKey: process.env.GOOGLE_API_KEY! }), {
1400
+ maxRetries: 3,
1401
+ });
1129
1402
  ```
1130
1403
 
1131
1404
  ### Retry with Backoff
@@ -1154,31 +1427,66 @@ const response = await retryableFetch('https://api.example.com');
1154
1427
  import { CircuitBreaker, CircuitBreakerRegistry } from '@cogitator-ai/core';
1155
1428
 
1156
1429
  const breaker = new CircuitBreaker({
1157
- threshold: 5,
1430
+ failureThreshold: 5,
1158
1431
  resetTimeout: 30000,
1159
- successThreshold: 2,
1432
+ halfOpenRequests: 3,
1433
+ onStateChange: (from, to) => console.log(`Circuit ${from} -> ${to}`),
1160
1434
  });
1161
1435
 
1162
- if (breaker.canExecute()) {
1163
- try {
1164
- const result = await riskyOperation();
1165
- breaker.recordSuccess();
1166
- } catch (error) {
1167
- breaker.recordFailure();
1168
- throw error;
1169
- }
1170
- }
1436
+ // Throws CIRCUIT_OPEN while open; counts successes and failures (by default only retryable errors, see isFailure)
1437
+ const result = await breaker.execute(() => riskyOperation());
1438
+
1439
+ console.log(breaker.getState(), breaker.getStats());
1171
1440
 
1172
- breaker.onStateChange((state) => {
1173
- console.log('Circuit state:', state);
1441
+ const registry = new CircuitBreakerRegistry({ failureThreshold: 3 });
1442
+ const apiBreaker = registry.get('payments-api');
1443
+ ```
1444
+
1445
+ ### Fallback Patterns
1446
+
1447
+ ```typescript
1448
+ import {
1449
+ CircuitBreakerRegistry,
1450
+ withFallback,
1451
+ withGracefulDegradation,
1452
+ createLLMFallbackExecutor,
1453
+ } from '@cogitator-ai/core';
1454
+
1455
+ const result = await withFallback({
1456
+ primary: () => primaryCall(),
1457
+ fallbacks: [
1458
+ { name: 'secondary', fn: () => fallbackCall() },
1459
+ { name: 'cache', fn: () => cachedResult() },
1460
+ ],
1461
+ retry: { maxRetries: 2 },
1462
+ onFallback: (from, to, error) => console.warn(`${from} -> ${to}: ${error.message}`),
1174
1463
  });
1464
+
1465
+ const degraded = await withGracefulDegradation(() => fullFeatureCall(), {
1466
+ defaultValue: [],
1467
+ onDegraded: (error) => console.warn('Degraded:', error.message),
1468
+ });
1469
+
1470
+ const executeWithFallback = createLLMFallbackExecutor(
1471
+ {
1472
+ providers: [
1473
+ { provider: 'openai', model: 'gpt-6.1-sol' },
1474
+ { provider: 'anthropic', model: 'claude-sonnet-5-5' },
1475
+ { provider: 'ollama', model: 'llama3.3:70b' },
1476
+ ],
1477
+ },
1478
+ new CircuitBreakerRegistry()
1479
+ );
1480
+ const response = await executeWithFallback((provider, model) =>
1481
+ cog.getLLMBackend(`${provider}/${model}`).chat({ model, messages })
1482
+ );
1175
1483
  ```
1176
1484
 
1177
1485
  ---
1178
1486
 
1179
1487
  ## Prompt Injection Detection
1180
1488
 
1181
- Protect your agents from jailbreak attempts, prompt injections, and other adversarial inputs:
1489
+ Protect your agents from jailbreak attempts, prompt injections, and other adversarial inputs. See [Security](https://cogitator.app/docs/advanced/security).
1182
1490
 
1183
1491
  ```typescript
1184
1492
  import { Cogitator, PromptInjectionDetector } from '@cogitator-ai/core';
@@ -1273,7 +1581,7 @@ const stats = detector.getStats();
1273
1581
 
1274
1582
  ## Tool Caching
1275
1583
 
1276
- Cache tool results to avoid redundant API calls with exact or semantic matching:
1584
+ Cache tool results to avoid redundant API calls with exact or semantic matching. See [Tool Caching](https://cogitator.app/docs/tools/tool-caching).
1277
1585
 
1278
1586
  ### Exact Match Caching
1279
1587
 
@@ -1310,18 +1618,17 @@ Similar queries hit the cache based on embedding similarity:
1310
1618
 
1311
1619
  ```typescript
1312
1620
  import { withCache } from '@cogitator-ai/core';
1313
- import type { EmbeddingService } from '@cogitator-ai/types';
1621
+ import { OpenAIEmbeddingService } from '@cogitator-ai/memory';
1314
1622
 
1315
- const embeddingService: EmbeddingService = {
1316
- embed: async (text) => openai.embeddings.create({ input: text }),
1317
- embedBatch: async (texts) => /* ... */,
1318
- dimensions: 1536,
1623
+ // Any EmbeddingService ({ embed, embedBatch, dimensions, model }) works
1624
+ const embeddingService = new OpenAIEmbeddingService({
1625
+ apiKey: process.env.OPENAI_API_KEY!,
1319
1626
  model: 'text-embedding-3-small',
1320
- };
1627
+ });
1321
1628
 
1322
1629
  const cachedSearch = withCache(webSearch, {
1323
1630
  strategy: 'semantic',
1324
- similarity: 0.95, // 95% similarity threshold
1631
+ similarity: 0.95, // 95% similarity threshold
1325
1632
  ttl: '1h',
1326
1633
  maxSize: 1000,
1327
1634
  storage: 'memory',
@@ -1334,10 +1641,13 @@ await cachedSearch.execute({ query: 'Paris weather forecast' }, ctx); // semanti
1334
1641
 
1335
1642
  ### Redis Storage
1336
1643
 
1337
- For production with persistence:
1644
+ For production with persistence. `redisClient` takes any `RedisClientLike`; an ioredis client and a `createRedisClient()` client from `@cogitator-ai/redis` (standalone or cluster) fit as they are:
1338
1645
 
1339
1646
  ```typescript
1340
- import { withCache, RedisToolCacheStorage } from '@cogitator-ai/core';
1647
+ import { withCache } from '@cogitator-ai/core';
1648
+ import { Redis } from 'ioredis';
1649
+
1650
+ const redis = new Redis(process.env.REDIS_URL!);
1341
1651
 
1342
1652
  const cachedTool = withCache(webSearch, {
1343
1653
  strategy: 'semantic',
@@ -1345,16 +1655,18 @@ const cachedTool = withCache(webSearch, {
1345
1655
  ttl: '1h',
1346
1656
  maxSize: 1000,
1347
1657
  storage: 'redis',
1348
- redisClient: redisClient, // ioredis compatible client
1658
+ redisClient: redis,
1349
1659
  keyPrefix: 'myapp:cache',
1350
1660
  embeddingService,
1351
1661
  });
1352
1662
  ```
1353
1663
 
1664
+ Keys live under `keyPrefix` (a `:` is appended when missing and never doubled): entries at `<prefix>:entry:<cache key>`, the LRU order in `<prefix>:lru` and the entry count in `<prefix>:counter`. Cached tools sharing a prefix share one LRU and `maxSize`; `cache.clear()` deletes every key under the prefix.
1665
+
1354
1666
  ### Cache Management
1355
1667
 
1356
1668
  ```typescript
1357
- const cached = withCache(tool, config);
1669
+ const cached = withCache(searchTool, config);
1358
1670
 
1359
1671
  // Get statistics
1360
1672
  const stats = cached.cache.stats();
@@ -1375,44 +1687,30 @@ await cached.cache.warmup([
1375
1687
  ### Cache Callbacks
1376
1688
 
1377
1689
  ```typescript
1378
- const cached = withCache(tool, {
1690
+ const cached = withCache(searchTool, {
1379
1691
  strategy: 'exact',
1380
1692
  ttl: '1h',
1381
1693
  maxSize: 100,
1382
1694
  storage: 'memory',
1383
- onHit: (key, params) => console.log('Cache hit:', key),
1384
- onMiss: (key, params) => console.log('Cache miss:', key),
1695
+ onHit: (key) => console.log('Cache hit:', key),
1696
+ onMiss: (key) => console.log('Cache miss:', key),
1385
1697
  onEvict: (key) => console.log('Evicted:', key),
1386
1698
  });
1387
1699
  ```
1388
1700
 
1389
- ---
1701
+ `onEvict` fires for entries removed by `cache.invalidate()` and for entries evicted to stay under `maxSize`, in memory and in Redis.
1390
1702
 
1391
- ### Fallback Patterns
1703
+ To build a storage yourself, `createToolCacheStorage()` takes the same `onEvict`, called with the key of each entry evicted to make room:
1392
1704
 
1393
1705
  ```typescript
1394
- import {
1395
- withFallback,
1396
- withGracefulDegradation,
1397
- createLLMFallbackExecutor,
1398
- } from '@cogitator-ai/core';
1399
-
1400
- const result = await withFallback(
1401
- () => primaryCall(),
1402
- () => fallbackCall()
1403
- );
1404
-
1405
- const degraded = await withGracefulDegradation(
1406
- () => fullFeatureCall(),
1407
- [() => reducedFeatureCall(), () => minimalCall(), () => cachedResult()]
1408
- );
1706
+ import { createToolCacheStorage } from '@cogitator-ai/core';
1409
1707
 
1410
- const llmExecutor = createLLMFallbackExecutor([
1411
- { provider: 'openai', model: 'gpt-6.1-sol' },
1412
- { provider: 'anthropic', model: 'claude-sonnet-5-5' },
1413
- { provider: 'ollama', model: 'llama3.3:70b' },
1414
- ]);
1415
- const response = await llmExecutor.chat(request);
1708
+ const storage = createToolCacheStorage('redis', {
1709
+ redisClient: redis,
1710
+ keyPrefix: 'myapp:cache',
1711
+ maxSize: 5000,
1712
+ onEvict: (key) => console.log('Evicted:', key),
1713
+ });
1416
1714
  ```
1417
1715
 
1418
1716
  ---
@@ -1422,38 +1720,47 @@ const response = await llmExecutor.chat(request);
1422
1720
  Built-in content safety with input/output filtering, tool guards, and critique-revision loops:
1423
1721
 
1424
1722
  ```typescript
1425
- import { ConstitutionalAI, InputFilter, OutputFilter, ToolGuard } from '@cogitator-ai/core';
1723
+ import {
1724
+ Cogitator,
1725
+ ConstitutionalAI,
1726
+ createConstitution,
1727
+ DEFAULT_PRINCIPLES,
1728
+ } from '@cogitator-ai/core';
1426
1729
 
1427
1730
  const constitutional = new ConstitutionalAI({
1428
1731
  llm: backend,
1429
1732
  constitution: createConstitution([...DEFAULT_PRINCIPLES, customPrinciple]),
1430
- config: { enabled: true },
1733
+ config: { strictMode: true },
1431
1734
  });
1432
1735
 
1433
- const inputFilter = new InputFilter({ llm: backend });
1434
- const inputResult = await inputFilter.evaluate('user message');
1435
- if (inputResult.isHarmful) {
1436
- console.log('Blocked:', inputResult.harmCategories);
1736
+ const inputResult = await constitutional.filterInput('user message');
1737
+ if (!inputResult.allowed) {
1738
+ console.log('Blocked:', inputResult.blockedReason, inputResult.harmScores);
1437
1739
  }
1438
1740
 
1439
- const outputFilter = new OutputFilter({ llm: backend });
1440
- const outputResult = await outputFilter.evaluate('agent response');
1741
+ const outputResult = await constitutional.filterOutput('agent response', messages);
1742
+ const guardResult = await constitutional.guardTool(
1743
+ deleteFileTool,
1744
+ { path: '/etc/passwd' },
1745
+ toolContext
1746
+ );
1747
+ console.log(guardResult.approved, guardResult.riskLevel, guardResult.reason);
1441
1748
 
1442
- const toolGuard = new ToolGuard({ llm: backend, strictMode: true });
1443
- const guardResult = await toolGuard.evaluate({
1444
- name: 'delete_file',
1445
- arguments: { path: '/etc/passwd' },
1446
- });
1749
+ const revision = await constitutional.critiqueAndRevise('draft answer', messages);
1447
1750
 
1448
- // Integrated with Cogitator runtime
1751
+ // Integrated with the Cogitator runtime: on when `guardrails` is set (unless enabled: false)
1449
1752
  const cog = new Cogitator({
1450
1753
  guardrails: {
1451
- enabled: true,
1452
- model: 'openai/gpt-6-luna',
1754
+ model: 'openai/gpt-6-luna', // judge model; default: llm.defaultModel, else the first run's agent model
1755
+ filterToolResults: true,
1453
1756
  },
1454
1757
  });
1455
1758
  ```
1456
1759
 
1760
+ Fields left out take `DEFAULT_GUARDRAIL_CONFIG` (input, output and tool-call filtering plus critique-revision on). `InputFilter`, `OutputFilter`, `ToolGuard` and `CritiqueReviser` are the layers `ConstitutionalAI` uses; each takes `{ config, constitution }` plus `llm` for the LLM-backed ones. `cog.getGuardrails()` and `cog.setConstitution()` work before the first run when `guardrails.model` or `llm.defaultModel` names the judge model; a constitution set earlier applies once the guardrails are built.
1761
+
1762
+ With `strictMode`, every call of a tool with `sideEffects` needs approval and goes through the run's [approval flow](#approvals) like a `requiresApproval` tool: `onApproval`, then `guardrails.onToolApproval`, else the run pauses. `ToolGuard` fails closed: a call that needs approval and was not approved by that flow or by `onToolApproval` is denied. See [Constitutional AI](https://cogitator.app/docs/advanced/constitutional-ai).
1763
+
1457
1764
  ---
1458
1765
 
1459
1766
  ## Cost-Aware Routing
@@ -1461,35 +1768,43 @@ const cog = new Cogitator({
1461
1768
  Automatically route tasks to the optimal model based on complexity, budget, and latency requirements:
1462
1769
 
1463
1770
  ```typescript
1464
- import { CostAwareRouter, TaskAnalyzer, CostTracker, BudgetEnforcer } from '@cogitator-ai/core';
1771
+ import { Cogitator, CostAwareRouter } from '@cogitator-ai/core';
1465
1772
 
1466
1773
  const router = new CostAwareRouter({
1467
1774
  config: {
1468
1775
  enabled: true,
1469
- budgets: {
1470
- daily: 10.0,
1471
- monthly: 200.0,
1776
+ budget: {
1777
+ maxCostPerRun: 0.5,
1778
+ maxCostPerDay: 10.0,
1779
+ warningThreshold: 0.8,
1780
+ onBudgetWarning: (current, limit) => console.warn(`$${current} of $${limit}`),
1472
1781
  },
1473
1782
  },
1474
1783
  });
1475
1784
 
1476
- const recommendation = await router.route({
1477
- input: 'Simple greeting',
1478
- complexity: 'low',
1479
- speedPreference: 'fast',
1480
- });
1481
- console.log('Use model:', recommendation.model);
1482
- console.log('Estimated cost:', recommendation.estimatedCost);
1785
+ const requirements = router.analyzeTask('Say hello'); // complexity, reasoning, speed, cost sensitivity
1786
+ const recommendation = await router.recommendModel('Say hello');
1787
+ console.log('Use model:', recommendation.provider, recommendation.modelId);
1788
+ console.log('Estimated cost:', recommendation.estimatedCost, recommendation.reasons);
1483
1789
 
1484
1790
  // Integrated with Cogitator
1485
1791
  const cog = new Cogitator({
1486
1792
  costRouting: {
1487
1793
  enabled: true,
1488
- budgets: { daily: 10.0 },
1794
+ autoSelectModel: true, // pick the model per run; result.modelUsed tells which
1795
+ budget: { maxCostPerDay: 10.0 },
1489
1796
  },
1490
1797
  });
1798
+
1799
+ cog.getCostSummary(); // tracked costs, also before the first run
1800
+ const estimate = await cog.estimateCost({ agent, input: 'Summarize this report' });
1801
+ console.log(estimate.expectedCost);
1491
1802
  ```
1492
1803
 
1804
+ With `costRouting.enabled`, every run is checked against `budget`, with or without `autoSelectModel`: the cost is estimated from the task's complexity and the model's price, and a run over a limit throws a `CogitatorError` with code `BUDGET_EXCEEDED` (HTTP 429). `autoSelectModel` only picks models of providers the runtime can call (`llm.backends`, plugins, `llm.defaultProvider`, the agent's own provider, or `llm.providers` entries with credentials) and keeps the agent's model when none fits. The router exposes the same steps: `recommendAvailableModel(input, isProviderAvailable)` (undefined when no provider fits) and `checkRunBudget(input, model)`.
1805
+
1806
+ See [Cost Routing](https://cogitator.app/docs/advanced/cost-routing).
1807
+
1493
1808
  ---
1494
1809
 
1495
1810
  ## Context Management
@@ -1499,15 +1814,16 @@ Automatic context window management with compression strategies:
1499
1814
  ```typescript
1500
1815
  const cog = new Cogitator({
1501
1816
  context: {
1502
- strategy: 'hybrid',
1503
- compressionThreshold: 0.8,
1817
+ strategy: 'hybrid', // 'truncate' | 'sliding-window' | 'summarize' | 'hybrid'
1818
+ compressionThreshold: 0.8, // compress above 80% of the usable window
1819
+ outputReserve: 0.15, // share of the model's context kept for the answer
1820
+ windowSize: 10, // recent messages kept verbatim by sliding-window / hybrid
1821
+ summaryModel: 'openai/gpt-6-luna', // model that writes summaries
1504
1822
  },
1505
1823
  });
1506
1824
  ```
1507
1825
 
1508
- Setting `context` turns compression on; pass `enabled: false` to keep the config but switch it off.
1509
-
1510
- Available strategies: `TruncateStrategy`, `SlidingWindowStrategy`, `SummarizeStrategy`, `HybridStrategy`.
1826
+ Setting `context` turns compression on; pass `enabled: false` to keep the config but switch it off. The strategies are also exported as classes (`TruncateStrategy`, `SlidingWindowStrategy`, `SummarizeStrategy`, `HybridStrategy`) and `ContextManager` can be used on its own. See [Context Management](https://cogitator.app/docs/advanced/context-management).
1511
1827
 
1512
1828
  ---
1513
1829
 
@@ -1522,14 +1838,35 @@ const langfuse = createLangfuseExporter({
1522
1838
  publicKey: process.env.LANGFUSE_PUBLIC_KEY!,
1523
1839
  secretKey: process.env.LANGFUSE_SECRET_KEY!,
1524
1840
  baseUrl: 'https://cloud.langfuse.com',
1841
+ enabled: true,
1525
1842
  });
1526
1843
 
1844
+ await langfuse.init(); // needs the optional `langfuse` package
1845
+
1527
1846
  const otlp = createOTLPExporter({
1528
1847
  endpoint: 'http://localhost:4318/v1/traces',
1529
1848
  headers: { Authorization: 'Bearer ...' },
1849
+ serviceName: 'my-agents',
1850
+ enabled: true,
1851
+ });
1852
+ otlp.start(); // flushes every 5 seconds
1853
+
1854
+ let runId = '';
1855
+ const result = await cog.run(agent, {
1856
+ input: 'Analyze this data...',
1857
+ onRunStart: (data) => {
1858
+ runId = data.runId;
1859
+ langfuse.onRunStart({ ...data, agentName: agent.name });
1860
+ },
1861
+ onToolCall: (call) => langfuse.onToolCall(runId, call),
1862
+ onToolResult: (toolResult) => langfuse.onToolResult(runId, toolResult),
1863
+ onSpan: (span) => otlp.exportSpan(runId, span),
1864
+ onRunComplete: (runResult) => langfuse.onRunComplete(runResult),
1530
1865
  });
1531
1866
  ```
1532
1867
 
1868
+ Both exporters do nothing unless `enabled: true` is set, so you can build them unconditionally and switch them per environment. They are not attached automatically; wire them to the run callbacks as above. See [Observability](https://cogitator.app/docs/deployment/observability).
1869
+
1533
1870
  ---
1534
1871
 
1535
1872
  ## Agent as Tool
@@ -1549,7 +1886,10 @@ const researcher = new Agent({
1549
1886
  const researchTool = agentAsTool(cog, researcher, {
1550
1887
  name: 'research',
1551
1888
  description: 'Delegate research tasks to a specialist agent',
1889
+ timeout: 60_000,
1552
1890
  includeUsage: true,
1891
+ includeToolCalls: false,
1892
+ onApproval: () => ({ approved: false, reason: 'Not allowed in delegated runs' }),
1553
1893
  });
1554
1894
 
1555
1895
  const manager = new Agent({
@@ -1560,17 +1900,19 @@ const manager = new Agent({
1560
1900
  });
1561
1901
  ```
1562
1902
 
1903
+ Tool calls of the inner agent that need approval are declined unless `onApproval` decides them, since a delegated run cannot pause for a person; an `onApproval` that returns `'pause'` declines too. If the inner run pauses anyway, the tool returns `success: false` with an `error` naming the tools that waited. To hand the conversation over instead of calling a sub-agent, use [handoffs](#handoffs). See [Agent as Tool](https://cogitator.app/docs/tools/agent-as-tool).
1904
+
1563
1905
  ---
1564
1906
 
1565
1907
  ## Logging
1566
1908
 
1567
1909
  ```typescript
1568
- import { Logger, getLogger, setLogger, createLogger } from '@cogitator-ai/core';
1910
+ import { getLogger, setLogger, createLogger, createLoggerFromConfig } from '@cogitator-ai/core';
1569
1911
 
1570
1912
  const logger = createLogger({
1571
- level: 'debug',
1572
- prefix: '[MyApp]',
1573
- timestamps: true,
1913
+ level: 'debug', // default 'info'; the default logger reads LOG_LEVEL
1914
+ format: 'json', // 'pretty' (default) or 'json'
1915
+ output: (entry, formatted) => process.stderr.write(formatted + '\n'),
1574
1916
  });
1575
1917
 
1576
1918
  setLogger(logger);
@@ -1581,6 +1923,24 @@ getLogger().warn('Rate limited', { retryAfter: 60 });
1581
1923
  getLogger().error('Failed', { error: 'Connection timeout' });
1582
1924
  ```
1583
1925
 
1926
+ `new Cogitator({ logging })` installs a logger built from the config with `createLoggerFromConfig()`. It is process-wide, so with several runtimes the last one created with `logging` wins:
1927
+
1928
+ ```typescript
1929
+ import { Cogitator, createLoggerFromConfig, setLogger } from '@cogitator-ai/core';
1930
+
1931
+ const cog = new Cogitator({
1932
+ logging: {
1933
+ level: 'warn', // 'debug' | 'info' | 'warn' | 'error' | 'silent'
1934
+ destination: 'file', // appends JSON lines to filePath
1935
+ filePath: './cogitator.log',
1936
+ },
1937
+ });
1938
+
1939
+ setLogger(createLoggerFromConfig({ level: 'silent' })); // the same, without a runtime
1940
+ ```
1941
+
1942
+ Without `filePath`, or where there is no file system, `destination: 'file'` logs to the console with a warning.
1943
+
1584
1944
  ---
1585
1945
 
1586
1946
  ## Type Reference
@@ -1611,11 +1971,11 @@ import type {
1611
1971
  import type {
1612
1972
  LLMBackend,
1613
1973
  LLMProvider,
1974
+ LLMBackendProvider, // LLMProvider or the name of your own backend
1614
1975
  LLMConfig,
1615
1976
  ChatRequest,
1616
1977
  ChatResponse,
1617
1978
  ChatStreamChunk,
1618
- ChatUsage,
1619
1979
  LLMErrorContext,
1620
1980
  LLMDebugOptions,
1621
1981
  LLMPlugin,
@@ -1698,14 +2058,17 @@ try {
1698
2058
  } catch (error) {
1699
2059
  if (error instanceof LLMError) {
1700
2060
  console.log('Provider:', error.provider);
1701
- console.log('Status:', error.statusCode);
2061
+ console.log('Provider status:', error.details?.statusCode);
1702
2062
  } else if (error instanceof CogitatorError) {
1703
- console.log('Code:', error.code);
1704
- console.log('Retryable:', isRetryableError(error));
2063
+ console.log('Code:', error.code); // an ErrorCode, e.g. ErrorCode.THREAD_ACCESS_DENIED
2064
+ console.log('HTTP status:', error.statusCode);
2065
+ console.log('Retryable:', isRetryableError(error), 'retry in', getRetryDelay(error), 'ms');
1705
2066
  }
1706
2067
  }
1707
2068
  ```
1708
2069
 
2070
+ Runs that hit their `timeout` throw `RUN_TIMEOUT` (HTTP 504, `Run timed out after <n>ms`), runs over the cost-routing budget `BUDGET_EXCEEDED` (429), and input or output blocked by guardrails (`Input blocked: …` / `Output blocked: …`) `LLM_CONTENT_FILTERED` (400), so server adapters return those statuses and messages.
2071
+
1709
2072
  ---
1710
2073
 
1711
2074
  ## Examples
@@ -1714,6 +2077,7 @@ try {
1714
2077
 
1715
2078
  ```typescript
1716
2079
  import { Cogitator, Agent, tool } from '@cogitator-ai/core';
2080
+ import { z } from 'zod';
1717
2081
 
1718
2082
  const webSearch = tool({
1719
2083
  name: 'web_search',