vel-ai 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. vel_ai-0.2.0/PKG-INFO +896 -0
  2. vel_ai-0.2.0/README.md +856 -0
  3. vel_ai-0.2.0/pyproject.toml +65 -0
  4. vel_ai-0.2.0/setup.cfg +4 -0
  5. vel_ai-0.2.0/tests/test_memory.py +342 -0
  6. vel_ai-0.2.0/tests/test_memory_context.py +309 -0
  7. vel_ai-0.2.0/tests/test_message_converter.py +513 -0
  8. vel_ai-0.2.0/tests/test_message_translation.py +411 -0
  9. vel_ai-0.2.0/tests/test_prompts.py +594 -0
  10. vel_ai-0.2.0/tests/test_rlm.py +401 -0
  11. vel_ai-0.2.0/tests/test_thinking.py +438 -0
  12. vel_ai-0.2.0/tests/test_tool_loop.py +213 -0
  13. vel_ai-0.2.0/tests/test_tool_use_behavior.py +183 -0
  14. vel_ai-0.2.0/vel/__init__.py +114 -0
  15. vel_ai-0.2.0/vel/agent.py +1479 -0
  16. vel_ai-0.2.0/vel/core/__init__.py +101 -0
  17. vel_ai-0.2.0/vel/core/context.py +542 -0
  18. vel_ai-0.2.0/vel/core/file_output.py +99 -0
  19. vel_ai-0.2.0/vel/core/guardrails.py +231 -0
  20. vel_ai-0.2.0/vel/core/hooks.py +145 -0
  21. vel_ai-0.2.0/vel/core/json_stream_parser.py +305 -0
  22. vel_ai-0.2.0/vel/core/reducer.py +42 -0
  23. vel_ai-0.2.0/vel/core/structured_output.py +206 -0
  24. vel_ai-0.2.0/vel/core/tool_behavior.py +84 -0
  25. vel_ai-0.2.0/vel/events.py +706 -0
  26. vel_ai-0.2.0/vel/memory/__init__.py +40 -0
  27. vel_ai-0.2.0/vel/memory/fact_store.py +71 -0
  28. vel_ai-0.2.0/vel/memory/strategy_reasoningbank.py +135 -0
  29. vel_ai-0.2.0/vel/prompts/__init__.py +88 -0
  30. vel_ai-0.2.0/vel/prompts/context_manager.py +176 -0
  31. vel_ai-0.2.0/vel/prompts/formatters.py +384 -0
  32. vel_ai-0.2.0/vel/prompts/manager.py +207 -0
  33. vel_ai-0.2.0/vel/prompts/registry.py +217 -0
  34. vel_ai-0.2.0/vel/prompts/template.py +262 -0
  35. vel_ai-0.2.0/vel/providers/__init__.py +86 -0
  36. vel_ai-0.2.0/vel/providers/anthropic.py +302 -0
  37. vel_ai-0.2.0/vel/providers/base.py +70 -0
  38. vel_ai-0.2.0/vel/providers/google.py +276 -0
  39. vel_ai-0.2.0/vel/providers/message_translator.py +841 -0
  40. vel_ai-0.2.0/vel/providers/openai.py +579 -0
  41. vel_ai-0.2.0/vel/providers/translators.py +1344 -0
  42. vel_ai-0.2.0/vel/rlm/__init__.py +24 -0
  43. vel_ai-0.2.0/vel/rlm/budget.py +177 -0
  44. vel_ai-0.2.0/vel/rlm/config.py +101 -0
  45. vel_ai-0.2.0/vel/rlm/context_store.py +395 -0
  46. vel_ai-0.2.0/vel/rlm/controller.py +695 -0
  47. vel_ai-0.2.0/vel/rlm/prompts.py +202 -0
  48. vel_ai-0.2.0/vel/rlm/scratchpad.py +193 -0
  49. vel_ai-0.2.0/vel/rlm/tools.py +354 -0
  50. vel_ai-0.2.0/vel/rlm/utils.py +258 -0
  51. vel_ai-0.2.0/vel/thinking/__init__.py +11 -0
  52. vel_ai-0.2.0/vel/thinking/config.py +97 -0
  53. vel_ai-0.2.0/vel/thinking/controller.py +540 -0
  54. vel_ai-0.2.0/vel/thinking/prompts.py +74 -0
  55. vel_ai-0.2.0/vel/tools/__init__.py +3 -0
  56. vel_ai-0.2.0/vel/tools/registry.py +260 -0
  57. vel_ai-0.2.0/vel/tools/schema_generator.py +228 -0
  58. vel_ai-0.2.0/vel/utils/__init__.py +14 -0
  59. vel_ai-0.2.0/vel/utils/async_queue.py +66 -0
  60. vel_ai-0.2.0/vel/utils/message_converter.py +385 -0
  61. vel_ai-0.2.0/vel/utils/message_reducer.py +437 -0
  62. vel_ai-0.2.0/vel_ai.egg-info/PKG-INFO +896 -0
  63. vel_ai-0.2.0/vel_ai.egg-info/SOURCES.txt +64 -0
  64. vel_ai-0.2.0/vel_ai.egg-info/dependency_links.txt +1 -0
  65. vel_ai-0.2.0/vel_ai.egg-info/requires.txt +17 -0
  66. vel_ai-0.2.0/vel_ai.egg-info/top_level.txt +1 -0
vel_ai-0.2.0/PKG-INFO ADDED
@@ -0,0 +1,896 @@
1
+ Metadata-Version: 2.4
2
+ Name: vel-ai
3
+ Version: 0.2.0
4
+ Summary: 12-Factor inspired AI agent runtime with streaming responses
5
+ Author: Richard Scheiwe
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/rscheiwe/vel
8
+ Project-URL: Documentation, https://rscheiwe.github.io/vel
9
+ Project-URL: Repository, https://github.com/rscheiwe/vel
10
+ Project-URL: Issues, https://github.com/rscheiwe/vel/issues
11
+ Keywords: ai,agent,llm,openai,anthropic,gemini,streaming,12-factor
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ Requires-Dist: httpx>=0.27.0
25
+ Requires-Dist: anyio>=4.4.0
26
+ Requires-Dist: pydantic>=2.8.0
27
+ Requires-Dist: jsonschema>=4.22.0
28
+ Requires-Dist: python-dotenv>=1.0.1
29
+ Requires-Dist: tenacity>=8.2.3
30
+ Requires-Dist: jinja2>=3.1.0
31
+ Requires-Dist: openai>=1.54.0
32
+ Requires-Dist: google-generativeai>=0.8.0
33
+ Requires-Dist: anthropic>=0.39.0
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=8.3.2; extra == "dev"
36
+ Requires-Dist: pytest-asyncio>=0.23.7; extra == "dev"
37
+ Requires-Dist: ruff>=0.6.9; extra == "dev"
38
+ Requires-Dist: black>=24.8.0; extra == "dev"
39
+ Requires-Dist: mypy>=1.11.0; extra == "dev"
40
+
41
+ # VEL
42
+
43
+ ## Agent Runtime (12-Factor Agents Aligned)
44
+
45
+ A production-ready AI agent runtime aligned with [12-Factor Agent principles](https://github.com/humanlayer/12-factor-agents) by Dex and contributors. Built for reliability, scalability, and maintainability with streaming responses, multiple LLM providers, and event-driven architecture.
46
+
47
+ ## Features
48
+
49
+ - **Dual Execution Modes**: Streaming (SSE) and non-streaming (JSON) responses
50
+ - **Multiple LLM Providers**: OpenAI, Google Gemini, and Anthropic Claude with plug-and-play architecture
51
+ - **RLM (Recursive Language Model)**: Handle 5MB+ documents through iterative reasoning, context probing, and budget-controlled execution
52
+ - **Generation Configuration**: Full control over model parameters (temperature, max_tokens, top_p, etc.) with per-run override support - matches Vercel AI SDK flexibility
53
+ - **Stream Protocol**: Vercel AI SDK **V5 UI Stream Protocol** compatible - works seamlessly with React `useChat()` and frontend components (100% parity)
54
+ - Exact event naming (`tool-call`, `tool-result`, etc.)
55
+ - Custom `data-*` events with transient flag for RAG citations, progress tracking, and analytics
56
+ - Response metadata (token usage tracking)
57
+ - Source events (citations and grounding)
58
+ - File events (inline data support)
59
+ - Reasoning events (OpenAI o1/o3 chain-of-thought streaming)
60
+ - Anthropic thinking blocks
61
+ - Enhanced error details
62
+ - **Message Aggregation**: MessageReducer for converting streaming events (text, reasoning, tools) to Vercel AI SDK message format
63
+ - **Tool System**: JSON schema-validated tools with async support
64
+ - **Flexible Prompts**: Jinja2 templating with XML formatting, environment-based configuration, and version control
65
+ - **Message Format Compatibility**: Works with Vercel AI SDK's `convertToModelMessages()` and includes Python converter for UIMessage → ModelMessage
66
+ - **Automatic Provider Translation**: Converts ModelMessage format to provider-specific formats (OpenAI/Anthropic/Gemini) automatically
67
+
68
+ ## Message Format Compatibility
69
+
70
+ Vel supports multiple message format patterns for seamless integration:
71
+
72
+ ### React Frontend + Vel Backend
73
+ ```typescript
74
+ // Frontend (React with Vercel AI SDK)
75
+ import { useChat, convertToModelMessages } from 'ai';
76
+
77
+ const { messages } = useChat();
78
+ const modelMessages = convertToModelMessages(messages);
79
+
80
+ fetch('/api/chat', {
81
+ body: JSON.stringify({ messages: modelMessages })
82
+ });
83
+ ```
84
+
85
+ ```python
86
+ # Backend (FastAPI with Vel)
87
+ from vel import Agent
88
+
89
+ @app.post("/api/chat")
90
+ async def chat(request: dict):
91
+ agent = Agent(
92
+ id='chat',
93
+ model={'provider': 'openai', 'model': 'gpt-4o'}
94
+ )
95
+
96
+ # Vel translates ModelMessage → OpenAI format automatically
97
+ response = await agent.run({'messages': request['messages']})
98
+ return {'response': response}
99
+ ```
100
+
101
+ ### Python-Only Applications
102
+ ```python
103
+ from vel import Agent
104
+ from vel.utils import convert_to_model_messages
105
+
106
+ # Option 1: Build ModelMessages manually
107
+ messages = [
108
+ {'role': 'user', 'content': 'Hello'},
109
+ {'role': 'assistant', 'content': 'Hi!'}
110
+ ]
111
+
112
+ # Option 2: Convert UIMessages from database
113
+ ui_messages = db.get_conversation(user_id)
114
+ messages = convert_to_model_messages(ui_messages)
115
+
116
+ # Use with any provider - translation is automatic
117
+ agent = Agent(
118
+ id='chat',
119
+ model={'provider': 'anthropic', 'model': 'claude-3-5-sonnet-20241022'}
120
+ )
121
+ response = await agent.run({'messages': messages})
122
+ ```
123
+
124
+ **See [Message Formats Documentation](https://rscheiwe.github.io/vel/message-formats) for detailed patterns and examples.**
125
+
126
+ ## Documentation
127
+
128
+ **📚 [Complete Documentation](https://rscheiwe.github.io/vel)**
129
+
130
+ - [Getting Started](https://rscheiwe.github.io/vel/getting-started) - Installation and quick start
131
+ - [Message Formats](https://rscheiwe.github.io/vel/message-formats) - UIMessage, ModelMessage, and automatic provider translation
132
+ - [Session Management](https://rscheiwe.github.io/vel/sessions) - Multi-turn conversations
133
+ - [RLM (Recursive Language Model)](https://rscheiwe.github.io/vel/rlm) - Long context support (5MB+) with iterative reasoning
134
+ - [Prompt Templates](https://rscheiwe.github.io/vel/prompts) - Flexible prompt management with Jinja2 and XML
135
+ - [Providers](https://rscheiwe.github.io/vel/providers) - OpenAI, Gemini, and Claude configuration
136
+ - [Tools](https://rscheiwe.github.io/vel/tools) - Custom tool creation
137
+ - [Stream Protocol](https://rscheiwe.github.io/vel/stream-protocol) - Event streaming reference
138
+ - [Event Translators](https://rscheiwe.github.io/vel/event-translators) - Protocol adapter architecture and usage guide
139
+ - [Using Translators Directly](https://rscheiwe.github.io/vel/using-translators) - Custom orchestration with frontend compatibility
140
+ - [Memory System](https://rscheiwe.github.io/vel/memory) - Optional memory with Fact Store and ReasoningBank
141
+ - [API Reference](https://rscheiwe.github.io/vel/api-reference) - Complete API docs
142
+ - [12-Factor Alignment](https://rscheiwe.github.io/vel/12-factor-alignment) - Production-ready agent principles
143
+ - [Stream Protocol Parity](PARITY_STATUS.md) - Vercel AI SDK V5 UI Stream Protocol compatibility status (100% parity)
144
+
145
+ ## Project Structure
146
+
147
+ ```
148
+ vel/
149
+ ├── providers/ # LLM provider implementations (OpenAI, Gemini, Anthropic)
150
+ ├── rlm/ # RLM (Recursive Language Model) for long context support
151
+ ├── tools/ # Tool registry and specifications
152
+ ├── prompts/ # Prompt templates with Jinja2 and XML formatting
153
+ ├── core/ # State management, reducer, context
154
+ ├── events.py # Stream protocol event definitions
155
+ └── agent.py # Main Agent class
156
+ ```
157
+
158
+ ## Architecture
159
+
160
+ Vel uses a **two-layer architecture** based on the Single Responsibility Principle:
161
+
162
+ ### Layer 1: Translators (Protocol Adapters)
163
+
164
+ - **Job:** Convert provider-specific → standard protocol
165
+ - **Scope:** Single LLM response stream
166
+ - **Stateful:** Only tracks current response (text blocks, tool calls)
167
+ - **Reusable:** Works with any orchestrator (Vel Agent, Mesh, LangGraph, custom)
168
+
169
+ ### Layer 2: Agent (Orchestrator)
170
+
171
+ - **Job:** Multi-step execution, tool calling, context management
172
+ - **Scope:** Full agentic workflow
173
+ - **Stateful:** Sessions, context, run history
174
+ - **Opinionated:** Implements specific orchestration pattern
175
+
176
+ This separation enables **composability**: use Agent for turnkey workflows, or use Event Translators directly with custom orchestrators. See [Event Translators](https://rscheiwe.github.io/vel/event-translators) for complete architecture details and integration examples.
177
+
178
+ ## Installation
179
+
180
+ ```bash
181
+ # Clone and install
182
+ git clone <repo-url>
183
+ cd vel
184
+ pip install -e .
185
+
186
+ # Set up environment
187
+ cp .env.example .env
188
+ # Edit .env with your API keys
189
+ ```
190
+
191
+ ## ⚠️ Deprecation Notice
192
+
193
+ **Global tool registration is deprecated in v0.3.0 and will be removed in v2.0.**
194
+
195
+ **Old (deprecated):**
196
+ ```python
197
+ from vel import ToolSpec, register_tool
198
+
199
+ register_tool(ToolSpec(...)) # ⚠️ DEPRECATED
200
+ agent = Agent(tools=['tool_name']) # ⚠️ DEPRECATED
201
+ ```
202
+
203
+ **New (recommended):**
204
+ ```python
205
+ from vel import ToolSpec
206
+
207
+ tool = ToolSpec.from_function(your_function)
208
+ agent = Agent(tools=[tool]) # ✅ No registration needed!
209
+ ```
210
+
211
+ See [Migration Guide](#migration-guide-global-registry--instance-tools-v20) below for details.
212
+
213
+ ---
214
+
215
+ ## Quick Start
216
+
217
+ ### API Key Configuration
218
+
219
+ Vel supports two ways to provide API keys:
220
+
221
+ **1. Environment Variables (recommended for development)**
222
+ ```bash
223
+ export OPENAI_API_KEY='sk-...'
224
+ export ANTHROPIC_API_KEY='sk-ant-...'
225
+ export GOOGLE_API_KEY='...'
226
+ ```
227
+
228
+ **2. Explicit API Keys (recommended for libraries/production)**
229
+ ```python
230
+ agent = Agent(
231
+ id='my-agent',
232
+ model={
233
+ 'provider': 'openai',
234
+ 'model': 'gpt-4o',
235
+ 'api_key': 'sk-...' # Override environment variable
236
+ }
237
+ )
238
+ ```
239
+
240
+ This makes Vel suitable for:
241
+ - **Applications**: Use environment variables
242
+ - **Libraries**: Pass API keys programmatically
243
+ - **Multi-tenant**: Different agents can use different API keys
244
+
245
+ ### Python SDK
246
+
247
+ ```python
248
+ import asyncio
249
+ from vel import Agent, ToolSpec
250
+
251
+ # Define a tool
252
+ def get_weather(city: str) -> dict:
253
+ """Get weather for a city."""
254
+ return {'temp': 72, 'condition': 'sunny'}
255
+
256
+ weather_tool = ToolSpec.from_function(get_weather)
257
+
258
+ async def main():
259
+ # Option 1: Use environment variable (OPENAI_API_KEY)
260
+ agent = Agent(
261
+ id='chat-general:v1',
262
+ model={'provider': 'openai', 'model': 'gpt-4o'},
263
+ tools=[weather_tool], # Pass ToolSpec directly
264
+ policies={'max_steps': 8}
265
+ )
266
+
267
+ # Option 2: Explicit API key
268
+ agent = Agent(
269
+ id='chat-general:v1',
270
+ model={'provider': 'openai', 'model': 'gpt-4o', 'api_key': 'sk-...'},
271
+ tools=[weather_tool], # Pass ToolSpec directly
272
+ policies={'max_steps': 8}
273
+ )
274
+
275
+ # Non-streaming mode
276
+ answer = await agent.run({'message': 'What is the weather?'})
277
+ print(answer)
278
+
279
+ # Streaming mode
280
+ async for event in agent.run_stream({'message': 'Tell me a story'}):
281
+ print(event)
282
+
283
+ if __name__ == '__main__':
284
+ asyncio.run(main())
285
+ ```
286
+
287
+ ## Stream Protocol
288
+
289
+ Vel uses the [Vercel AI SDK V5 UI Stream Protocol](https://ai-sdk.dev/docs/ai-sdk-ui/stream-protocol) for frontend-compatible event streaming:
290
+
291
+ - `text-start`, `text-delta`, `text-end` - Text content chunks
292
+ - `reasoning-start`, `reasoning-delta`, `reasoning-end` - Reasoning/chain-of-thought (o1/o3 models)
293
+ - `tool-input-start`, `tool-input-delta` - Tool input streaming
294
+ - `tool-input-available` - Complete tool input ready for execution
295
+ - `tool-output-available` - Tool execution result
296
+ - `start-step`, `finish-step` - Multi-step agent progress
297
+ - `data-*` - Custom application events (notifications, progress, metrics) with transient flag support
298
+ - `response-metadata` - Token usage and model info
299
+ - `source` - Citations and grounding (Gemini)
300
+ - `file` - Inline file attachments
301
+ - `error`, `finish-message` - Error handling and completion
302
+
303
+ **Frontend Compatible:** Works seamlessly with React's `useChat()`, `useCompletion()`, and other Vercel AI SDK frontend components. Each provider translates native events into V5-compatible standardized events.
304
+
305
+ #### Enhanced Error Handling
306
+
307
+ Vel automatically surfaces detailed error information without requiring manual print statements. Error events include:
308
+
309
+ ```python
310
+ {
311
+ 'type': 'error',
312
+ 'error': 'max_tokens must be greater than thinking.budget_tokens',
313
+ 'errorCode': 'invalid_request_error',
314
+ 'errorType': 'InvalidRequestError',
315
+ 'statusCode': 400,
316
+ 'provider': 'anthropic',
317
+ 'details': {
318
+ 'type': 'error',
319
+ 'message': 'max_tokens must be greater than thinking.budget_tokens'
320
+ }
321
+ }
322
+ ```
323
+
324
+ **Automatic Logging:** Errors are automatically logged with full context:
325
+ ```python
326
+ # Errors are logged automatically
327
+ agent = Agent(id='agent:v1', model={'provider': 'openai', 'model': 'gpt-4o'})
328
+
329
+ # If an error occurs, it's logged with full context
330
+ # No manual print statements needed!
331
+ async for event in agent.run_stream({'message': 'test'}):
332
+ if event['type'] == 'error':
333
+ # Full error context is available in the event
334
+ print(f"Error from {event['provider']}: {event['error']}")
335
+ if event.get('statusCode'):
336
+ print(f"HTTP {event['statusCode']}")
337
+ ```
338
+
339
+ **Python Logging:** Configure logging to see detailed error traces:
340
+ ```python
341
+ import logging
342
+ logging.basicConfig(level=logging.ERROR)
343
+ # vel.agent logger will now output detailed error information
344
+ ```
345
+
346
+ ### Message Aggregation
347
+
348
+ **MessageReducer** aggregates streaming events into structured messages (Vercel AI SDK format):
349
+
350
+ ```python
351
+ from vel import Agent, MessageReducer
352
+
353
+ # Create reducer
354
+ reducer = MessageReducer()
355
+ reducer.add_user_message("What's the weather in San Francisco?")
356
+
357
+ # Stream agent response
358
+ agent = Agent(
359
+ id='weather-agent',
360
+ model={'provider': 'openai', 'model': 'gpt-4o'},
361
+ tools=['get_weather']
362
+ )
363
+
364
+ async for event in agent.run_stream({'message': "What's the weather in SF?"}):
365
+ reducer.process_event(event)
366
+
367
+ # Get Vercel AI SDK compatible messages
368
+ messages = reducer.get_messages()
369
+ # [
370
+ # {user message},
371
+ # {assistant message with parts: [tool-call, tool-result, text]}
372
+ # ]
373
+
374
+ # Use messages however you need (store in DB, return to client, etc.)
375
+ print(messages)
376
+ ```
377
+
378
+ **With Reasoning (o1/o3 models):**
379
+
380
+ ```python
381
+ # Create reducer for reasoning model
382
+ reducer = MessageReducer()
383
+ reducer.add_user_message("What is sqrt(169)?")
384
+
385
+ agent = Agent(
386
+ id='reasoning-agent',
387
+ model={'provider': 'openai-responses', 'model': 'o1'}
388
+ )
389
+
390
+ async for event in agent.run_stream({'message': 'What is sqrt(169)?'}):
391
+ reducer.process_event(event)
392
+
393
+ messages = reducer.get_messages()
394
+ # assistant message parts: [
395
+ # {'type': 'start-step'},
396
+ # {'type': 'reasoning', 'text': '', 'state': 'done', 'providerMetadata': {...}},
397
+ # {'type': 'text', 'text': 'The answer is 13', 'state': 'done'}
398
+ # ]
399
+ ```
400
+
401
+ **Features:**
402
+ - ✓ Vercel AI SDK `useChat` hook compatible
403
+ - ✓ Aggregates text, reasoning, tool calls, and results into parts array
404
+ - ✓ Reasoning parts with provider metadata (o1/o3 models)
405
+ - ✓ Provider metadata (OpenAI message/call IDs)
406
+ - ✓ Custom message IDs and metadata support
407
+
408
+ See [Message Aggregation docs](https://rscheiwe.github.io/vel/stream-protocol#message-aggregation) for complete details.
409
+
410
+ ## Providers
411
+
412
+ ### OpenAI
413
+
414
+ ```python
415
+ agent = Agent(
416
+ id='my-agent',
417
+ model={'provider': 'openai', 'model': 'gpt-4o'}
418
+ )
419
+ ```
420
+
421
+ ### Google Gemini
422
+
423
+ ```python
424
+ agent = Agent(
425
+ id='my-agent',
426
+ model={'provider': 'google', 'model': 'gemini-1.5-pro'}
427
+ )
428
+ ```
429
+
430
+ ### Anthropic Claude
431
+
432
+ ```python
433
+ agent = Agent(
434
+ id='my-agent',
435
+ model={'provider': 'anthropic', 'model': 'claude-sonnet-4-20250514'}
436
+ )
437
+ ```
438
+
439
+ ### Reasoning Models (o1/o3)
440
+
441
+ Vel supports OpenAI's reasoning models. **Use the Responses API provider** for reasoning event indicators:
442
+
443
+ ```python
444
+ agent = Agent(
445
+ id='reasoning-agent',
446
+ model={
447
+ 'provider': 'openai-responses', # Use Responses API for reasoning events
448
+ 'model': 'o1' # or 'o1-mini', 'o3-mini'
449
+ }
450
+ )
451
+
452
+ async for event in agent.run_stream({'message': 'Solve: sqrt(169)'}):
453
+ if event['type'] == 'reasoning-start':
454
+ print("🧠 Reasoning begins...")
455
+ elif event['type'] == 'reasoning-delta':
456
+ # Note: OpenAI often encrypts reasoning content, so deltas may be empty
457
+ delta = event.get('delta', '')
458
+ if delta:
459
+ print(f"💭 {delta}", end='', flush=True)
460
+ elif event['type'] == 'reasoning-end':
461
+ print("\n✅ Reasoning complete")
462
+ elif event['type'] == 'text-delta':
463
+ print(event['delta'], end='', flush=True)
464
+ ```
465
+
466
+ **Event Flow**:
467
+ 1. `reasoning-start` - Reasoning block begins
468
+ 2. `reasoning-delta` - Reasoning content (often empty/encrypted by OpenAI)
469
+ 3. `reasoning-end` - Reasoning block ends
470
+ 4. `text-start` → `text-delta`* → `text-end` - Final answer
471
+
472
+ **Note**: OpenAI encrypts reasoning content for o1/o3 models in most cases. You'll receive `reasoning-start` and `reasoning-end` events to indicate reasoning occurred, but `reasoning-delta` events may be empty. This matches the AI SDK behavior.
473
+
474
+ **See**: [examples/responses_api.py](examples/responses_api.py) for Responses API examples, [examples/reasoning_o1.py](examples/reasoning_o1.py) for Chat Completions API
475
+
476
+ ## Session Management (Multi-Turn Conversations)
477
+
478
+ Sessions enable multi-turn conversations where the agent remembers context across multiple calls.
479
+
480
+ ### Basic Session Usage
481
+
482
+ ```python
483
+ agent = Agent(
484
+ id='my-agent',
485
+ model={'provider': 'openai', 'model': 'gpt-4o'}
486
+ )
487
+
488
+ # Multi-turn conversation - same session_id = shared history
489
+ session_id = 'user-123'
490
+
491
+ answer1 = await agent.run({'message': 'My name is Alice'}, session_id=session_id)
492
+ # "Hello Alice! How can I help you?"
493
+
494
+ answer2 = await agent.run({'message': 'What is my name?'}, session_id=session_id)
495
+ # "Your name is Alice."
496
+
497
+ # Note: Sessions are in-memory. For persistent storage, save/load messages yourself.
498
+ ```
499
+
500
+ ### Message History Modes
501
+
502
+ Control how much conversation history is retained:
503
+
504
+ ```python
505
+ from vel import ContextManager, StatelessContextManager
506
+
507
+ # Full message history (default)
508
+ agent = Agent(..., context_manager=ContextManager())
509
+
510
+ # No message history (stateless)
511
+ agent = Agent(..., context_manager=StatelessContextManager())
512
+
513
+ # Limited history (last 10 messages)
514
+ agent = Agent(..., context_manager=ContextManager(max_history=10))
515
+
516
+ # Custom logic
517
+ class CustomContextManager(ContextManager):
518
+ def messages_for_llm(self, run_id: str, session_id: Optional[str] = None):
519
+ # Your custom retrieval (e.g., RAG, summarization)
520
+ return your_logic()
521
+
522
+ agent = Agent(..., context_manager=CustomContextManager())
523
+ ```
524
+
525
+ See `examples/context_modes.py` for a full demonstration.
526
+
527
+ ## Generation Configuration
528
+
529
+ Control model behavior with fine-grained generation parameters. Matches the flexibility of Vercel AI SDK's `streamText()` function.
530
+
531
+ ### Agent-Level Configuration
532
+
533
+ Set default generation parameters when creating an agent:
534
+
535
+ ```python
536
+ from vel import Agent
537
+
538
+ agent = Agent(
539
+ id='my-agent',
540
+ model={'provider': 'openai', 'model': 'gpt-4o'},
541
+ generation_config={
542
+ 'temperature': 0.7, # Creativity (0-2)
543
+ 'max_tokens': 500, # Output limit
544
+ 'top_p': 0.9, # Nucleus sampling
545
+ 'presence_penalty': 0.6, # Encourage new topics (OpenAI)
546
+ 'frequency_penalty': 0.3,# Reduce repetition (OpenAI)
547
+ 'stop': ['END'], # Stop sequences
548
+ 'seed': 42 # Reproducible outputs (OpenAI, Anthropic)
549
+ }
550
+ )
551
+ ```
552
+
553
+ ### Per-Run Override
554
+
555
+ Override generation config for specific runs:
556
+
557
+ ```python
558
+ # Use agent's default config
559
+ result1 = await agent.run({'message': 'Write a creative story'})
560
+
561
+ # Override for deterministic response
562
+ result2 = await agent.run(
563
+ {'message': 'What is 2+2?'},
564
+ generation_config={'temperature': 0} # Override to 0 for this run only
565
+ )
566
+
567
+ # Works with streaming too
568
+ async for event in agent.run_stream(
569
+ {'message': 'Explain AI'},
570
+ generation_config={'max_tokens': 100} # Brief response
571
+ ):
572
+ print(event)
573
+ ```
574
+
575
+ ### Supported Parameters
576
+
577
+ #### Common (All Providers)
578
+ - `temperature` - Sampling temperature (0-2, default varies by provider)
579
+ - `max_tokens` - Maximum output tokens
580
+ - `top_p` - Nucleus sampling (0-1)
581
+ - `stop` - Stop sequences (list of strings)
582
+
583
+ #### OpenAI
584
+ - `presence_penalty` - Penalize new tokens (-2 to 2)
585
+ - `frequency_penalty` - Penalize repeated tokens (-2 to 2)
586
+ - `seed` - Reproducibility seed (integer)
587
+ - `logit_bias` - Token probability adjustments (dict)
588
+
589
+ #### Anthropic
590
+ - `top_k` - Top-K sampling (integer)
591
+ - `stop_sequences` - Alternative to `stop` (list of strings)
592
+
593
+ #### Google Gemini
594
+ - `top_k` - Top-K sampling (integer)
595
+ - `max_output_tokens` - Alternative to `max_tokens` (integer)
596
+ - `stop_sequences` - Alternative to `stop` (list of strings)
597
+
598
+ ### Examples
599
+
600
+ #### Deterministic Code Generation
601
+ ```python
602
+ agent = Agent(
603
+ id='code-gen',
604
+ model={'provider': 'openai', 'model': 'gpt-4o'},
605
+ generation_config={
606
+ 'temperature': 0,
607
+ 'seed': 42, # Same output every time
608
+ 'max_tokens': 2000
609
+ }
610
+ )
611
+ ```
612
+
613
+ #### Creative Writing
614
+ ```python
615
+ agent = Agent(
616
+ id='creative',
617
+ model={'provider': 'anthropic', 'model': 'claude-sonnet-4-20250514'},
618
+ generation_config={
619
+ 'temperature': 0.9, # High creativity
620
+ 'top_p': 0.95,
621
+ 'top_k': 50,
622
+ 'max_tokens': 4000
623
+ }
624
+ )
625
+ ```
626
+
627
+ #### Concise Responses
628
+ ```python
629
+ agent = Agent(
630
+ id='brief',
631
+ model={'provider': 'google', 'model': 'gemini-1.5-pro'},
632
+ generation_config={
633
+ 'max_tokens': 100,
634
+ 'temperature': 0.7,
635
+ 'stop_sequences': ['\n\n'] # Stop at double newline
636
+ }
637
+ )
638
+ ```
639
+
640
+ See `examples/generation_config_example.py` for comprehensive examples.
641
+
642
+ ## RLM (Recursive Language Model) - Long Context Support
643
+
644
+ RLM is a middleware that enables agents to handle very long contexts (5MB+) through recursive reasoning and iterative context probing.
645
+
646
+ ### How It Works
647
+
648
+ Instead of loading the entire context into the prompt, RLM:
649
+ 1. **Probes context iteratively** using tools (search, read, summarize)
650
+ 2. **Accumulates notes** in a scratchpad
651
+ 3. **Reasons recursively** until reaching a FINAL() answer
652
+ 4. **Enforces budgets** for cost and performance control
653
+
654
+ ```python
655
+ from vel import Agent
656
+
657
+ # Enable RLM for long-context reasoning
658
+ agent = Agent(
659
+ id='doc-analyzer:v1',
660
+ model={'provider': 'openai', 'model': 'gpt-4o-mini'},
661
+ rlm={
662
+ 'enabled': True,
663
+ 'depth': 1, # Allow recursive sub-queries
664
+ 'control_model': {'provider': 'openai', 'model': 'gpt-4o-mini'},
665
+ 'writer_model': {'provider': 'openai', 'model': 'gpt-4o'}, # Optional
666
+ 'budgets': {
667
+ 'max_steps_root': 12,
668
+ 'max_tokens_total': 120000,
669
+ 'max_cost_usd': 0.50
670
+ }
671
+ }
672
+ )
673
+
674
+ # Use with large documents (5MB+)
675
+ with open('large_document.txt') as f:
676
+ large_doc = f.read()
677
+
678
+ answer = await agent.run(
679
+ input={'message': 'Summarize the key findings and recommendations.'},
680
+ context_refs=large_doc # RLM activates automatically
681
+ )
682
+ ```
683
+
684
+ ### Key Features
685
+
686
+ - **No context window limits** - Handle documents beyond model limits
687
+ - **Cost efficient** - Use cheap models for iteration, strong models for synthesis
688
+ - **Budget controls** - Hard limits on steps, tokens, and cost
689
+ - **Streaming support** - Emit RLM events (probes, notes, budget status)
690
+ - **REPL-style execution** - Optional `python_exec` for complex data processing (disabled by default)
691
+
692
+ ### Tools
693
+
694
+ RLM provides three tools for context interaction:
695
+
696
+ - **context_probe** - Safe search/read/summarize operations (always enabled)
697
+ - **rlm_call** - Spawn recursive sub-queries for decomposition
698
+ - **python_exec** - Execute Python code with CONTEXT variable (⚠️ security risk, disabled by default)
699
+
700
+ ### Documentation
701
+
702
+ See the [complete RLM guide](https://rscheiwe.github.io/vel/rlm) for:
703
+ - Detailed architecture and control flow
704
+ - Configuration options and tuning
705
+ - Security considerations for `python_exec`
706
+ - Streaming events
707
+ - Examples and best practices
708
+
709
+ ### Example Output
710
+
711
+ ```bash
712
+ python examples/rlm_basic.py
713
+ ```
714
+
715
+ Inspired by [Alex Zhang's RLM approach](https://alexzhang13.github.io/blog/2025/rlm/).
716
+
717
+ ## Configuration
718
+
719
+ Environment variables (see `.env.example`):
720
+
721
+ ```bash
722
+ # OpenAI
723
+ OPENAI_API_KEY=sk-...
724
+ OPENAI_API_BASE=https://api.openai.com/v1
725
+
726
+ # Google Gemini
727
+ GOOGLE_API_KEY=...
728
+
729
+ # Anthropic Claude
730
+ ANTHROPIC_API_KEY=sk-ant-...
731
+
732
+ # Runner mode
733
+ VEL_RUNNER=local-async
734
+ ```
735
+
736
+ ## Examples
737
+
738
+ Vel includes comprehensive examples demonstrating various patterns:
739
+
740
+ **Core Examples:**
741
+ - `examples/quickstart.py` - Basic agent usage (streaming & non-streaming)
742
+ - `examples/rlm_basic.py` - RLM for long contexts (5MB+ documents)
743
+ - `examples/message_reducer_example.py` - MessageReducer for message aggregation
744
+ - `examples/custom_data_events.py` - Custom data-* events with transient flag
745
+ - `examples/context_modes.py` - Different context management strategies
746
+ - `examples/generation_config_example.py` - Model parameter control
747
+ - `examples/prompt_templates.py` - Prompt template system
748
+
749
+ **Multi-Step Agent Examples:**
750
+ - `examples/multi_step_simple.py` - Basic multi-step pattern (websearch + news)
751
+ - `examples/multi_step_analysis.py` - Problem analysis with analyze tool
752
+ - `examples/multi_step_decision.py` - Decision-making with decide tool
753
+ - `examples/multi_step_complex.py` - Complex reasoning with all tools
754
+ - `examples/comprehensive_multi_step_agent.py` - Full multi-step demonstration
755
+
756
+ **Run with:**
757
+ ```bash
758
+ python examples/quickstart.py
759
+ python examples/message_reducer_example.py
760
+ python examples/multi_step_simple.py
761
+ ```
762
+
763
+ Or use VS Code debug configurations (see `.vscode/launch.json`).
764
+
765
+ ## Development
766
+
767
+ ```bash
768
+ # Install dev dependencies
769
+ pip install -e ".[dev]"
770
+
771
+ # Run tests
772
+ pytest
773
+
774
+ # Format code
775
+ black vel/
776
+ ruff check vel/
777
+
778
+ # Type checking
779
+ mypy vel/
780
+ ```
781
+
782
+ ## Architecture
783
+
784
+ Vel is designed following the [12-Factor Agent principles](https://github.com/humanlayer/12-factor-agents) (by Dex and contributors) for production-ready AI applications. See our [implementation guide](docs/12-factor-alignment.md) for details.
785
+
786
+ - **Agent**: Main orchestrator with dual execution modes (streaming/non-streaming)
787
+ - **RLM**: Middleware for long-context reasoning (5MB+) with iterative probing and budget controls
788
+ - **ContextManager**: Message history layer for conversation turns (configurable: full/stateless/limited)
789
+ - **Reducer**: Pure function for state transitions and effect generation (stateless, reproducible)
790
+ - **Providers**: LLM-specific implementations with stream protocol translation
791
+ - **Tools**: Validated, async-capable function execution (structured outputs)
792
+ - **Memory** (optional): Fact store and ReasoningBank for long-term structured data and strategy learning
793
+
794
+ **Key Principles:**
795
+
796
+ - ✓ Own your prompts - Direct control, no abstractions
797
+ - ✓ Own your context window - Custom context managers
798
+ - ✓ Stateless reducer - Predictable, reproducible behavior
799
+ - ✓ Small, focused agents - Composable design
800
+
801
+ ## TODO
802
+
803
+ - [ ] Add features from OpenAI Agent SDK (tool responses, e.g.)
804
+ - [ ] Test Gemini tool calling
805
+ - [ ] Finish Postgres integration
806
+ - [ ] Add knowledge-graph memory layer
807
+ - [ ] Add example of how to create Vel agents via a tool
808
+ - [ ] Add guardrails
809
+ - [ ] Stress test RLM with real-world large documents
810
+ - [x] ~~Update ReasoningBank to include e2e implementation as described in Google's paper~~ (Phase 1 complete, see `docs/Memory/reasoningbank-phase2-roadmap.md` for Phase 2)
811
+ - [x] ~~Add RLM (Recursive Language Model) support for long contexts~~ (Complete - see `docs/rlm.md`)
812
+
813
+ ## Migration Guide: Global Registry → Instance Tools (v2.0)
814
+
815
+ **Status:** Global tool registration is deprecated in v0.3.0 and will be removed in v2.0.
816
+
817
+ ### What's Changing
818
+
819
+ **Before (v0.x - Deprecated):**
820
+ ```python
821
+ from vel import ToolSpec, register_tool, Agent
822
+
823
+ # Register globally
824
+ tool = ToolSpec(name='get_weather', input_schema={...}, output_schema={...}, handler=my_handler)
825
+ register_tool(tool) # ⚠️ DEPRECATED
826
+
827
+ # Use by string
828
+ agent = Agent(tools=['get_weather']) # ⚠️ DEPRECATED
829
+ ```
830
+
831
+ **After (v2.0 - Recommended):**
832
+ ```python
833
+ from vel import ToolSpec, Agent
834
+
835
+ # Define function
836
+ def get_weather(city: str) -> dict:
837
+ """Get weather for a city."""
838
+ return {'temp': 72, 'condition': 'sunny'}
839
+
840
+ # Wrap in ToolSpec (auto-generates schemas)
841
+ tool = ToolSpec.from_function(get_weather)
842
+
843
+ # Pass directly to agent
844
+ agent = Agent(tools=[tool]) # ✅ No registration needed!
845
+ ```
846
+
847
+ ### Why?
848
+
849
+ 1. **No Global State** - Tools scoped to agent instances
850
+ 2. **Type Safety** - No string magic, IDE autocomplete works
851
+ 3. **Better Testing** - No need to mock global registries
852
+ 4. **Runtime Tools** - Create tools dynamically (perfect for UIs)
853
+ 5. **Industry Standard** - Matches OpenAI Agents SDK pattern
854
+
855
+ ### Migration Steps
856
+
857
+ 1. **Replace `register_tool()` calls:**
858
+ ```python
859
+ # Before
860
+ register_tool(ToolSpec(...))
861
+
862
+ # After
863
+ tool = ToolSpec.from_function(your_function)
864
+ ```
865
+
866
+ 2. **Update Agent initialization:**
867
+ ```python
868
+ # Before
869
+ agent = Agent(tools=['tool_name'])
870
+
871
+ # After
872
+ agent = Agent(tools=[tool])
873
+ ```
874
+
875
+ 3. **For shared tools, define once and reuse:**
876
+ ```python
877
+ shared_tool = ToolSpec.from_function(my_function)
878
+ agent1 = Agent(tools=[shared_tool])
879
+ agent2 = Agent(tools=[shared_tool])
880
+ ```
881
+
882
+ ### Timeline
883
+
884
+ - **v0.3.0** (Current): Deprecation warnings added, old code still works
885
+ - **v1.x**: Warnings continue, old code still works
886
+ - **v2.0**: Breaking changes - `register_tool()` removed, `Agent` only accepts `List[ToolSpec]`
887
+
888
+ ### Examples
889
+
890
+ See `examples/dynamic_tools.py` for complete migration examples.
891
+
892
+ ---
893
+
894
+ ## License
895
+
896
+ MIT