llm.rb 15.3.0 → 15.4.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +148 -3
- data/README.md +185 -71
- data/bin/llm.rb +9 -2
- data/data/alibaba.json +47 -2
- data/data/anthropic.json +67 -0
- data/data/bedrock.json +1828 -425
- data/data/deepinfra.json +148 -16
- data/data/deepseek.json +7 -4
- data/data/google.json +3 -3
- data/data/mistral.json +42 -0
- data/data/moonshot.json +1 -1
- data/data/openai.json +186 -0
- data/data/openrouter.json +1783 -440
- data/data/xai.json +53 -20
- data/data/zai.json +90 -4
- data/docs/deepdive/advanced/compaction.md +1 -2
- data/docs/deepdive/advanced/context.md +214 -1
- data/docs/deepdive/advanced/guard.md +9 -57
- data/docs/deepdive/features/builtin_tools.md +14 -16
- data/docs/deepdive/features/console.md +5 -0
- data/docs/deepdive/features/database.md +85 -10
- data/docs/deepdive/fundamentals/agents.md +7 -8
- data/docs/deepdive/fundamentals/providers.md +45 -5
- data/docs/deepdive/fundamentals/schema.md +73 -0
- data/docs/deepdive/fundamentals/tools.md +80 -27
- data/docs/deepdive/media/audio.md +8 -19
- data/docs/deepdive/media/images.md +8 -10
- data/docs/deepdive/media/ocr.md +1 -3
- data/docs/deepdive/reference/cost.md +48 -0
- data/docs/deepdive/reference/tracer.md +76 -0
- data/docs/deepdive.md +1 -1
- data/lib/llm/active_record/message.rb +113 -0
- data/lib/llm/active_record.rb +1 -0
- data/lib/llm/agent.rb +28 -19
- data/lib/llm/console/buffer.rb +9 -1
- data/lib/llm/console.rb +6 -1
- data/lib/llm/context/deserializer.rb +6 -1
- data/lib/llm/context.rb +8 -4
- data/lib/llm/guard.rb +2 -8
- data/lib/llm/provider.rb +16 -0
- data/lib/llm/providers/alibaba.rb +15 -0
- data/lib/llm/providers/anthropic/error_handler.rb +5 -2
- data/lib/llm/providers/anthropic/files.rb +12 -12
- data/lib/llm/providers/anthropic/models.rb +2 -2
- data/lib/llm/providers/anthropic.rb +2 -2
- data/lib/llm/providers/bedrock/error_handler.rb +3 -2
- data/lib/llm/providers/bedrock/models.rb +5 -3
- data/lib/llm/providers/bedrock.rb +2 -2
- data/lib/llm/providers/deepinfra/audio.rb +4 -4
- data/lib/llm/providers/deepinfra/images.rb +4 -4
- data/lib/llm/providers/google/error_handler.rb +5 -2
- data/lib/llm/providers/google/files.rb +10 -10
- data/lib/llm/providers/google/images.rb +2 -2
- data/lib/llm/providers/google/models.rb +2 -2
- data/lib/llm/providers/google.rb +4 -4
- data/lib/llm/providers/ollama/error_handler.rb +5 -2
- data/lib/llm/providers/ollama/models.rb +2 -2
- data/lib/llm/providers/ollama.rb +4 -4
- data/lib/llm/providers/openai/audio.rb +6 -6
- data/lib/llm/providers/openai/error_handler.rb +5 -2
- data/lib/llm/providers/openai/files.rb +10 -10
- data/lib/llm/providers/openai/images.rb +4 -4
- data/lib/llm/providers/openai/models.rb +2 -2
- data/lib/llm/providers/openai/moderations.rb +2 -2
- data/lib/llm/providers/openai/responses.rb +6 -6
- data/lib/llm/providers/openai/vector_stores.rb +22 -22
- data/lib/llm/providers/openai.rb +4 -4
- data/lib/llm/providers/xai/images.rb +4 -4
- data/lib/llm/tracer/telemetry.rb +4 -4
- data/lib/llm/tracer.rb +11 -3
- data/lib/llm/transport/execution.rb +8 -4
- data/lib/llm/version.rb +1 -1
- data/llm.gemspec +0 -6
- metadata +3 -5
- data/lib/llm/guard/loop.rb +0 -89
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 4647447cce1ccfc7e4f062cfa46b6dd0363ffd1a86537200a7877daff983cda3
|
|
4
|
+
data.tar.gz: 31502899d4a6443b893a48baf18c75d02311920166a6d791d671fa33600e93b9
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: ee244a4b0d7d317632e1ee2f3dc44b82926086a71f2ded6ba7e3189eff9636dba2eed5f97d4ed6c83140584d3807b4872303a20b369a123be2336c32671758a1
|
|
7
|
+
data.tar.gz: 7ed45ea9770afabb6522b6939c9d333c28a5892e40e32de271d03a8c5fb3f9fa15b5c2ace19e74ac80342bc65e04a7b5edb48b6799a612f5754b8ccccd33bdaf
|
data/CHANGELOG.md
CHANGED
|
@@ -17,6 +17,150 @@
|
|
|
17
17
|
|
|
18
18
|
*No unreleased changes yet. Check back after the next release.*
|
|
19
19
|
|
|
20
|
+
## v15.4.1
|
|
21
|
+
|
|
22
|
+
Changes since `v15.4.0`.
|
|
23
|
+
|
|
24
|
+
This release removes the gemspec's post-install message, so installing the gem
|
|
25
|
+
no longer prints the r.uby.dev notice. It also refreshes the model registry
|
|
26
|
+
with current model listings, limits, and pricing.
|
|
27
|
+
|
|
28
|
+
### Core
|
|
29
|
+
|
|
30
|
+
* **remove the gemspec post install message** <br>
|
|
31
|
+
The gemspec no longer sets `post_install_message`, so installing the
|
|
32
|
+
gem no longer prints the r.uby.dev website notice.
|
|
33
|
+
|
|
34
|
+
### Registry
|
|
35
|
+
|
|
36
|
+
* **refresh model metadata** <br>
|
|
37
|
+
Update `data/` with current model listings, limits, and pricing for the
|
|
38
|
+
Alibaba, Bedrock, DeepSeek, Google, Moonshot, and OpenRouter registries.
|
|
39
|
+
Bedrock adds the Kimi K3 and GPT-6 Sol and GPT-6 Luna families in both
|
|
40
|
+
the global and US regions and raises the context limit to 1M tokens for
|
|
41
|
+
two models, while Moonshot raises the Kimi K3 output limit to 1M tokens.
|
|
42
|
+
OpenRouter adds `qwen/qwen3.8-max-prime`, `z-ai/glm-5.3-prime`,
|
|
43
|
+
`upstage/solar-mini4`, the Aion 3.5 models, and `stealth/space-bunny-alpha`,
|
|
44
|
+
drops `mistralai/devstral-2512` and a free Ling 3.0 Flash VL entry, and
|
|
45
|
+
reprices several DeepSeek and Mistral models.
|
|
46
|
+
|
|
47
|
+
## v15.4.0
|
|
48
|
+
|
|
49
|
+
Changes since `v15.3.0`.
|
|
50
|
+
|
|
51
|
+
This release saves token usage with the context state, moves the retry budget
|
|
52
|
+
onto the provider, groups a turn's spans under one trace, and lets an agent
|
|
53
|
+
declare `name`, `description`, `path`, and `tool_budget` with a block. It also
|
|
54
|
+
gives every tracer request an id, adds `LLM::ActiveRecord::Message`, reads
|
|
55
|
+
`AGENTS.md` as the console's system prompt, removes `LLM::Guard::Loop`, and
|
|
56
|
+
refreshes the model registry.
|
|
57
|
+
|
|
58
|
+
### Core
|
|
59
|
+
|
|
60
|
+
* **context: save token usage with the state** <br>
|
|
61
|
+
[`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h-instance_method)
|
|
62
|
+
now writes `context_used` and `context_window`, so a saved state can be
|
|
63
|
+
inspected without loading the runtime. They are never read back, and a payload
|
|
64
|
+
without them still loads.
|
|
65
|
+
|
|
66
|
+
### Provider
|
|
67
|
+
|
|
68
|
+
* **provider: decide the retry budget on the provider** <br>
|
|
69
|
+
[`LLM::Provider#retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#retry_budget-instance_method)
|
|
70
|
+
returns how many times a rate-limited request is retried before the error is
|
|
71
|
+
raised, and defaults to 5. An agent that sets no `retry_budget` of its own
|
|
72
|
+
now takes the budget from its provider instead of special-casing Alibaba, so
|
|
73
|
+
[`LLM::Alibaba`](https://r.uby.dev/api-docs/llm.rb/LLM/Alibaba.html)
|
|
74
|
+
still retries 8 times and a provider that recovers from rate limits slowly can
|
|
75
|
+
return a higher budget. An explicit `retry_budget:` still takes precedence.
|
|
76
|
+
|
|
77
|
+
### Agent
|
|
78
|
+
|
|
79
|
+
* **agent: group a turn's spans into one trace** <br>
|
|
80
|
+
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) now opens a
|
|
81
|
+
`llm.turn` trace group around every turn, so all spans a turn produces share
|
|
82
|
+
one trace id. It uses the agent's tracer when it has one, the provider's
|
|
83
|
+
otherwise; previously a provider-wide tracer split one turn across traces.
|
|
84
|
+
|
|
85
|
+
* **agent: declare `name`, `description`, `path`, and `tool_budget` with a block** <br>
|
|
86
|
+
A block passed to any of these four class-level setters was previously
|
|
87
|
+
ignored, because the call was read as a getter. Now
|
|
88
|
+
[`LLM::Agent.name`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#name-class_method)
|
|
89
|
+
and [`LLM::Agent.description`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#description-class_method)
|
|
90
|
+
store the block, and an instance resolves it against itself, or against its
|
|
91
|
+
ORM record when it is bound to one. `name` and `description` also accept a
|
|
92
|
+
`Symbol` or a `Proc`, and the class-level reader still returns what was
|
|
93
|
+
configured, so read them on an instance for the resolved string.
|
|
94
|
+
|
|
95
|
+
### Console
|
|
96
|
+
|
|
97
|
+
* **console: use `AGENTS.md` as the system prompt** <br>
|
|
98
|
+
`bin/llm.rb` now looks for `AGENTS.md` in the current working directory when
|
|
99
|
+
it boots, and when the file exists its contents become the agent's
|
|
100
|
+
instructions for the session. The instructions are injected once, so a
|
|
101
|
+
resumed session that already has a system message keeps the one it has.
|
|
102
|
+
|
|
103
|
+
* **console: stop a cancelled turn from crashing the console** <br>
|
|
104
|
+
A turn cancelled with Esc can still have chunks queued, and they arrive after
|
|
105
|
+
the buffer that renders them is closed. The streaming path then tried to
|
|
106
|
+
replace a row that no longer existed and raised.
|
|
107
|
+
[`LLM::Console::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Console/Buffer.html)
|
|
108
|
+
now reports whether it has a row to replace through `open?`, the stream drops
|
|
109
|
+
chunks that arrive for a closed buffer, and `replace` keeps the current text
|
|
110
|
+
instead of raising.
|
|
111
|
+
|
|
112
|
+
### Guard
|
|
113
|
+
|
|
114
|
+
* **guard: remove `LLM::Guard::Loop`** <br>
|
|
115
|
+
`LLM::Guard::Loop` is removed. It stopped repeated tool-call patterns, but
|
|
116
|
+
could also interrupt a loop that was making progress, so it did not hold up as
|
|
117
|
+
a default. Agents no longer enable a guard of their own; bound the loop with
|
|
118
|
+
`tool_budget` instead.
|
|
119
|
+
|
|
120
|
+
### ActiveRecord
|
|
121
|
+
|
|
122
|
+
* **activerecord: add `LLM::ActiveRecord::Message`** <br>
|
|
123
|
+
[`LLM::ActiveRecord::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord/Message.html)
|
|
124
|
+
is a virtual model over the JSONB column that holds an agent's state, so
|
|
125
|
+
messages can be filtered, ordered, and counted in SQL instead of in memory.
|
|
126
|
+
`for(agent:)` returns an
|
|
127
|
+
[`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
|
|
128
|
+
with `id`, `role`, `content`, `tools`, and `position` columns; a row's
|
|
129
|
+
`unwrap!` returns the message. The class never materializes as a table.
|
|
130
|
+
|
|
131
|
+
### Tracer
|
|
132
|
+
|
|
133
|
+
* **tracer: give every request an id** <br>
|
|
134
|
+
[`LLM::Tracer#on_request_start`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#on_request_start-instance_method)
|
|
135
|
+
now takes a `request_id:`, a UUIDv7 that the runtime mints when a request
|
|
136
|
+
begins and passes to `on_request_finish` and `on_request_error` for that same
|
|
137
|
+
request. A turn can make many requests, so `trace_group_id` groups a turn but
|
|
138
|
+
not the events inside one request; the id lets a tracer correlate a request's
|
|
139
|
+
start, finish, and error events. The keyword is required, so a subclass that
|
|
140
|
+
overrides these hooks must accept it.
|
|
141
|
+
|
|
142
|
+
* **tracer: record the model the provider answers with** <br>
|
|
143
|
+
[`LLM::Tracer::Telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Telemetry.html)
|
|
144
|
+
now sets `gen_ai.response.model` from the model on the response instead of the
|
|
145
|
+
model requested, so a span reports the model a provider actually served. A
|
|
146
|
+
router model such as `openrouter/auto` records the model it resolved to,
|
|
147
|
+
while `gen_ai.request.model` keeps the name that was asked for.
|
|
148
|
+
|
|
149
|
+
### Registry
|
|
150
|
+
|
|
151
|
+
* **refresh model metadata** <br>
|
|
152
|
+
Update `data/` with current pricing, limits, and capabilities for the
|
|
153
|
+
Alibaba, Anthropic, Bedrock, DeepInfra, DeepSeek, Mistral, OpenAI,
|
|
154
|
+
OpenRouter, xAI, and Z.ai registries. Claude Opus 5.5 reaches Anthropic,
|
|
155
|
+
Bedrock, and OpenRouter, OpenAI adds GPT-6 Sol and GPT-6 Luna, Bedrock adds
|
|
156
|
+
Gemma 4 and more regional Claude Sonnet 4 and GPT-5.6 entries, and
|
|
157
|
+
OpenRouter adds the Xiaomi MiMo V2.6, Nex N2.5, Cohere Command A+, and
|
|
158
|
+
Qwen3.8 Omni Flash models. Z.ai adds `glm-4.6v-flash` and `glm-5.3-flashx`,
|
|
159
|
+
xAI adds Grok 4.7, DeepSeek adds a `low` reasoning effort to
|
|
160
|
+
`deepseek-v4-pro` and deprecates `deepseek-v4-flash` and
|
|
161
|
+
`deepseek-v4-flash-vision-exp`, DeepInfra reprices `tencent/Hy3`, and
|
|
162
|
+
OpenRouter drops `anthropic/claude-opus-4` and `kwaipilot/kat-coder-pro-v2`.
|
|
163
|
+
|
|
20
164
|
## v15.3.0
|
|
21
165
|
|
|
22
166
|
Changes since `v15.2.2`.
|
|
@@ -290,9 +434,10 @@ and a `-v` switch to the CLI, and refreshes the model registry.
|
|
|
290
434
|
The shared [`LLM::Tool::Utils`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Utils.html)
|
|
291
435
|
module now requires the `test-cmd.rb` gem (at `~> 2.7.1`) itself and
|
|
292
436
|
exposes the `spawn` and `wait` helpers, so any tool that includes
|
|
293
|
-
`Utils` gets command spawning without requiring `exec` directly.
|
|
294
|
-
`
|
|
295
|
-
|
|
437
|
+
`Utils` gets command spawning without requiring `exec` directly.
|
|
438
|
+
`Exec` and `ReadFile` include `Utils`, and the tools that shell out
|
|
439
|
+
(`Git`, `Rg`, `Mkdir`, `Ruby`, and `Bundle`) route through `Exec`,
|
|
440
|
+
so they all get the same bounded output.
|
|
296
441
|
|
|
297
442
|
* **tools: route `git`, `rg`, `mkdir`, and `ruby` through `exec`** <br>
|
|
298
443
|
`LLM::Tool::Git`, `LLM::Tool::Rg`, `LLM::Tool::Mkdir`, and
|
data/README.md
CHANGED
|
@@ -19,10 +19,13 @@ on CRuby. It has zero runtime dependencies by default, supports
|
|
|
19
19
|
concurrent and parallel tool execution and has a single coherent API
|
|
20
20
|
that spans 14+ providers.
|
|
21
21
|
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
22
|
+
It is possible to see llm.rb in action on the
|
|
23
|
+
[the r.uby.dev website](https://r.uby.dev) where
|
|
24
|
+
I am working on building an agentic platform that
|
|
25
|
+
users can use to manage multiple agents that are
|
|
26
|
+
specialized in different areas, and have access to
|
|
27
|
+
different services (eg GitHub, etc). Check it out if
|
|
28
|
+
curious. Still in early development.
|
|
26
29
|
|
|
27
30
|
## Install
|
|
28
31
|
|
|
@@ -52,7 +55,7 @@ an invalid state that would lead to API-level errors. For example,
|
|
|
52
55
|
when a tool call is interrupted it could leave an unanswered tool
|
|
53
56
|
call that a model will reject on the next turn. The runtime takes
|
|
54
57
|
care of this by pruning orphaned tool calls and ensuring that the
|
|
55
|
-
tool loop always remains valid.
|
|
58
|
+
tool loop always remains valid.
|
|
56
59
|
|
|
57
60
|
```ruby
|
|
58
61
|
require "llm"
|
|
@@ -141,7 +144,9 @@ call them on your behalf, and they're one of the most powerful features
|
|
|
141
144
|
for extending the feature set or abilities of a model.
|
|
142
145
|
|
|
143
146
|
The runtime also ships with a catalog of built-in tools for
|
|
144
|
-
filesystem, search, and shell operations
|
|
147
|
+
filesystem, search, and shell operations, and providers expose
|
|
148
|
+
platform-native tools such as web search and code execution that run
|
|
149
|
+
on the provider's side.
|
|
145
150
|
|
|
146
151
|
```ruby
|
|
147
152
|
class ReadFile < LLM::Tool
|
|
@@ -276,13 +281,18 @@ for agents.
|
|
|
276
281
|
|
|
277
282
|
##### Installation
|
|
278
283
|
|
|
279
|
-
The console is distributed with llm.rb
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
experience:
|
|
284
|
+
The console is distributed with llm.rb but it requires a number
|
|
285
|
+
of optional dependencies to be installed separately. The following
|
|
286
|
+
gems provide the full experience:
|
|
283
287
|
|
|
284
288
|
gem install unicode-display_width curses kramdown xchan.rb test-cmd.rb
|
|
285
289
|
|
|
290
|
+
For convenience it is also possible to just use the following, it
|
|
291
|
+
is a metagem that depends on llm.rb and all the dependencies it requires
|
|
292
|
+
to run the console:
|
|
293
|
+
|
|
294
|
+
gem install llm-shell
|
|
295
|
+
|
|
286
296
|
##### Persistence
|
|
287
297
|
|
|
288
298
|
the `path:` option can be set on an agent for automatic persistence
|
|
@@ -355,29 +365,32 @@ for both Rack-based / Rails-based applications. On databases
|
|
|
355
365
|
where it is supported, such as PostgreSQL, the column can be optimized by using
|
|
356
366
|
the `jsonb` type.
|
|
357
367
|
|
|
358
|
-
The following example is based on the agent used to power the
|
|
359
|
-
[r.uby.dev chatbot](https://r.uby.dev).
|
|
360
|
-
|
|
361
368
|
```ruby
|
|
362
369
|
require "active_record"
|
|
363
370
|
require "llm"
|
|
364
371
|
require "llm/active_record"
|
|
365
372
|
|
|
366
|
-
|
|
373
|
+
##
|
|
374
|
+
# The Robert agent.
|
|
375
|
+
class Robert < ActiveRecord::Base
|
|
367
376
|
acts_as_agent(format: :jsonb) do |agent|
|
|
368
|
-
agent.set name: "
|
|
369
|
-
description: "
|
|
370
|
-
|
|
377
|
+
agent.set name: "robert",
|
|
378
|
+
description: "robert is an agent that has access to the official " \
|
|
379
|
+
"r.uby.dev GitHub repositories. He can access the repositories " \
|
|
380
|
+
"to answer your question(s) about r.uby.dev projects.",
|
|
381
|
+
instructions: proc { File.read(File.join(__dir__, "robert", "prompt.md")) },
|
|
371
382
|
tools: :tools,
|
|
372
|
-
concurrency: :async
|
|
373
|
-
end
|
|
383
|
+
concurrency: :async,
|
|
374
384
|
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
385
|
+
##
|
|
386
|
+
# The maximum number of tool calls per-turn.
|
|
387
|
+
tool_budget: 25,
|
|
378
388
|
|
|
379
|
-
|
|
380
|
-
|
|
389
|
+
##
|
|
390
|
+
# The default tracer that all agents have associated
|
|
391
|
+
# with them. The tracer exports a trace to a couple of
|
|
392
|
+
# SQL tables.
|
|
393
|
+
tracer: proc { Raven::Tracer::SQL.new(llm, agent: self) }
|
|
381
394
|
end
|
|
382
395
|
|
|
383
396
|
##
|
|
@@ -398,10 +411,6 @@ class Raven < ActiveRecord::Base
|
|
|
398
411
|
|
|
399
412
|
private
|
|
400
413
|
|
|
401
|
-
def set_provider
|
|
402
|
-
LLM.deepseek
|
|
403
|
-
end
|
|
404
|
-
|
|
405
414
|
def allowlist
|
|
406
415
|
%w[
|
|
407
416
|
get_commit
|
|
@@ -420,19 +429,18 @@ class Raven < ActiveRecord::Base
|
|
|
420
429
|
end
|
|
421
430
|
end
|
|
422
431
|
|
|
423
|
-
agent =
|
|
432
|
+
agent = Robert.create!
|
|
424
433
|
|
|
425
434
|
##
|
|
426
435
|
# Every call to `talk` automatically persists
|
|
427
|
-
# to the database
|
|
428
|
-
|
|
429
|
-
agent.research_issues
|
|
436
|
+
# to the database.
|
|
437
|
+
agent.talk "what's new on the llm.rb repository?"
|
|
430
438
|
|
|
431
439
|
##
|
|
432
440
|
# The conversation was persisted to database. A
|
|
433
441
|
# fresh instance restores it and continues where
|
|
434
442
|
# we left off
|
|
435
|
-
agent =
|
|
443
|
+
agent = Robert.find(agent.id).talk "and what about roda-llm?"
|
|
436
444
|
|
|
437
445
|
##
|
|
438
446
|
# Start an agent console.
|
|
@@ -440,6 +448,100 @@ agent = Raven.find(agent.id).tap(&:research_codebase)
|
|
|
440
448
|
# The console does not persist back to the database.
|
|
441
449
|
agent.console
|
|
442
450
|
```
|
|
451
|
+
</details>
|
|
452
|
+
<details>
|
|
453
|
+
<summary> SQL optimizations </summary>
|
|
454
|
+
<br>
|
|
455
|
+
|
|
456
|
+
In a database environment the runtime optimizes for
|
|
457
|
+
the PostgreSQL database and its builtin support for
|
|
458
|
+
the `jsonb` column type. An agent fits in a single
|
|
459
|
+
column, on a single row, and that column carries
|
|
460
|
+
everything it has done: messages, tool calls,
|
|
461
|
+
context usage, and so on. It works well in practice
|
|
462
|
+
and means you can store an agent almost anywhere.
|
|
463
|
+
|
|
464
|
+
For scenarios where performance matters most the runtime
|
|
465
|
+
ships with virtual ActiveRecord classes that never materialize
|
|
466
|
+
in your database but provide a SQL view into the column where
|
|
467
|
+
an agent stores its runtime state. They return
|
|
468
|
+
[`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
|
|
469
|
+
objects, so the filtering happens in the database.
|
|
470
|
+
|
|
471
|
+
```ruby
|
|
472
|
+
class Agent < ActiveRecord::Base
|
|
473
|
+
acts_as_agent(format: :jsonb) do |agent|
|
|
474
|
+
agent.set name: "activerecord agent"
|
|
475
|
+
end
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
##
|
|
479
|
+
# Find an instance of your agent
|
|
480
|
+
agent = Agent.find_by(id: 1)
|
|
481
|
+
|
|
482
|
+
##
|
|
483
|
+
# Returns a relation over the agent's messages.
|
|
484
|
+
# It is scoped to the agent, and it yields one
|
|
485
|
+
# instance of LLM::ActiveRecord::Message per
|
|
486
|
+
# message the agent has produced.
|
|
487
|
+
messages = LLM::ActiveRecord::Message.for(agent:)
|
|
488
|
+
|
|
489
|
+
##
|
|
490
|
+
# The relation chains like any other
|
|
491
|
+
messages.where(role: "assistant")
|
|
492
|
+
.order(position: :desc)
|
|
493
|
+
.limit(10)
|
|
494
|
+
|
|
495
|
+
##
|
|
496
|
+
# Count, too
|
|
497
|
+
messages.count
|
|
498
|
+
```
|
|
499
|
+
|
|
500
|
+
**Schema**
|
|
501
|
+
|
|
502
|
+
Each row carries a message, flattened into columns:
|
|
503
|
+
|
|
504
|
+
| column | contents |
|
|
505
|
+
| --- | --- |
|
|
506
|
+
| `agent_id` | the agent a message belongs to |
|
|
507
|
+
| `id` | the message id |
|
|
508
|
+
| `role` | the message role |
|
|
509
|
+
| `content` | the message content |
|
|
510
|
+
| `tools` | the tool calls a message carries |
|
|
511
|
+
| `position` | the position of a message in the conversation |
|
|
512
|
+
| `data` | the whole message, as the runtime stores it |
|
|
513
|
+
|
|
514
|
+
**Indexes**
|
|
515
|
+
|
|
516
|
+
The queries the view runs are already covered. They expand
|
|
517
|
+
one agent, found by primary key, so they are index scans.
|
|
518
|
+
There is nothing to add for
|
|
519
|
+
`LLM::ActiveRecord::Message.for(agent:)`.
|
|
520
|
+
|
|
521
|
+
The queries you write on top of it are not. Once a question
|
|
522
|
+
is asked of every agent, the column is expanded row by row
|
|
523
|
+
and no index helps the view itself. Index the column for
|
|
524
|
+
those questions instead:
|
|
525
|
+
|
|
526
|
+
```sql
|
|
527
|
+
CREATE INDEX index_agents_on_data
|
|
528
|
+
ON agents USING gin (data jsonb_path_ops);
|
|
529
|
+
|
|
530
|
+
CREATE INDEX index_agents_on_context_used
|
|
531
|
+
ON agents (((data ->> 'context_used')::int));
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
The first serves containment (`@>`) and path queries over
|
|
535
|
+
the state as a whole. The second serves a scalar key, and
|
|
536
|
+
the runtime already writes `context_used` and
|
|
537
|
+
`context_window` at the top level, so "sessions over 80%
|
|
538
|
+
full" becomes cheap. Both assume `format: :jsonb`.
|
|
539
|
+
|
|
540
|
+
**However:** an agent's whole conversation lives in one
|
|
541
|
+
value, so every save rewrites it, and a GIN index is
|
|
542
|
+
maintained with it. Prefer an index on a key or two over
|
|
543
|
+
the whole column.
|
|
544
|
+
|
|
443
545
|
</details>
|
|
444
546
|
|
|
445
547
|
<details><summary>MCP</summary>
|
|
@@ -534,10 +636,9 @@ even answer for it. Because it runs before the tool, anything
|
|
|
534
636
|
it intercepts never executes. Policy, validation, quotas, and
|
|
535
637
|
cost ceilings all live here.
|
|
536
638
|
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
by default, so agents get loop protection out of the box. To
|
|
639
|
+
Agents and contexts use
|
|
640
|
+
[`LLM::Guard::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Null.html)
|
|
641
|
+
by default, so a guard only runs when you configure one. To
|
|
541
642
|
write your own guard, subclass
|
|
542
643
|
[`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
|
|
543
644
|
and implement
|
|
@@ -737,11 +838,16 @@ is also distributed with llm.rb.
|
|
|
737
838
|
```ruby
|
|
738
839
|
llm = LLM.openai
|
|
739
840
|
llm = LLM.anthropic
|
|
841
|
+
llm = LLM.google
|
|
740
842
|
llm = LLM.deepseek
|
|
741
|
-
llm = LLM.
|
|
843
|
+
llm = LLM.deepinfra
|
|
844
|
+
llm = LLM.xai
|
|
845
|
+
llm = LLM.zai
|
|
742
846
|
llm = LLM.moonshot
|
|
743
847
|
llm = LLM.openrouter
|
|
848
|
+
llm = LLM.alibaba # also: LLM.aliyun
|
|
744
849
|
llm = LLM.mistral
|
|
850
|
+
llm = LLM.bedrock
|
|
745
851
|
```
|
|
746
852
|
</details>
|
|
747
853
|
<details>
|
|
@@ -755,10 +861,14 @@ key at all.
|
|
|
755
861
|
```ruby
|
|
756
862
|
llm = LLM.openai(key: ENV["OPENAI_API_KEY"])
|
|
757
863
|
llm = LLM.anthropic(key: ENV["ANTHROPIC_API_KEY"])
|
|
864
|
+
llm = LLM.google(key: ENV["GOOGLE_API_KEY"])
|
|
758
865
|
llm = LLM.deepseek(key: ENV["DEEPSEEK_API_KEY"])
|
|
759
|
-
llm = LLM.
|
|
866
|
+
llm = LLM.deepinfra(key: ENV["DEEPINFRA_API_KEY"])
|
|
867
|
+
llm = LLM.xai(key: ENV["XAI_API_KEY"])
|
|
868
|
+
llm = LLM.zai(key: ENV["ZHIPU_API_KEY"])
|
|
760
869
|
llm = LLM.moonshot(key: ENV["MOONSHOT_API_KEY"])
|
|
761
870
|
llm = LLM.openrouter(key: ENV["OPENROUTER_API_KEY"])
|
|
871
|
+
llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
|
|
762
872
|
llm = LLM.mistral(key: ENV["MISTRAL_API_KEY"])
|
|
763
873
|
```
|
|
764
874
|
</details>
|
|
@@ -919,22 +1029,16 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
|
|
|
919
1029
|
<details>
|
|
920
1030
|
<summary>Where can I see llm.rb in action?</summary>
|
|
921
1031
|
<br>
|
|
922
|
-
<p>
|
|
923
1032
|
|
|
924
|
-
The [r.uby.dev](https://r.uby.dev) website
|
|
925
|
-
|
|
926
|
-
to this very GitHub repository. It is designed to help
|
|
927
|
-
you learn and troubleshoot llm.rb.
|
|
928
|
-
</p>
|
|
1033
|
+
The [r.uby.dev](https://r.uby.dev) website.
|
|
1034
|
+
|
|
929
1035
|
</details>
|
|
930
1036
|
<details>
|
|
931
1037
|
<summary>What about local LLM support?</summary>
|
|
932
1038
|
<br>
|
|
933
|
-
|
|
1039
|
+
|
|
934
1040
|
The following providers can be run used with models that
|
|
935
|
-
are running on your own hardware.
|
|
936
|
-
tested but not my main driver:
|
|
937
|
-
</p>
|
|
1041
|
+
are running on your own hardware.
|
|
938
1042
|
|
|
939
1043
|
* Ollama
|
|
940
1044
|
* Llamacpp
|
|
@@ -943,29 +1047,27 @@ tested but not my main driver:
|
|
|
943
1047
|
<details>
|
|
944
1048
|
<summary>I have a limited budget. What should I do?</summary>
|
|
945
1049
|
<br>
|
|
946
|
-
|
|
1050
|
+
|
|
947
1051
|
There are a few options. The first option is to host
|
|
948
1052
|
your own model, and use the ollama or llamacpp
|
|
949
1053
|
providers. This can be difficult though because
|
|
950
1054
|
a capable model requires hardware that can
|
|
951
1055
|
match it. If you have the ability to self-host,
|
|
952
1056
|
this would be my first option.
|
|
953
|
-
|
|
954
|
-
<p>
|
|
1057
|
+
|
|
955
1058
|
The second option is DeepSeek. <br>
|
|
956
1059
|
The deepseek-v4-flash model costs pennies to use. <br>
|
|
957
1060
|
And llm.rb has been optimized for deepseek. For example,
|
|
958
1061
|
DeepSeek does not have image generation capabilities
|
|
959
1062
|
but on the llm.rb runtime it does (vector graphics only,
|
|
960
1063
|
though).
|
|
961
|
-
|
|
962
|
-
<p>
|
|
1064
|
+
|
|
963
1065
|
The same is true for structured outputs. DeepSeek does
|
|
964
1066
|
not support structured outputs in the same way as OpenAI or
|
|
965
1067
|
Google, but the llm.rb runtime makes it appear as
|
|
966
1068
|
though it does, through the `json_object` response
|
|
967
1069
|
type.
|
|
968
|
-
|
|
1070
|
+
|
|
969
1071
|
If you're on a budget, DeepSeek is hard to beat.
|
|
970
1072
|
</details>
|
|
971
1073
|
<details>
|
|
@@ -988,23 +1090,35 @@ web</a>.
|
|
|
988
1090
|
<summary>Who maintains llm.rb?</summary>
|
|
989
1091
|
<br>
|
|
990
1092
|
|
|
991
|
-
The llm.rb project
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
environments.
|
|
1093
|
+
The llm.rb project was started more than three
|
|
1094
|
+
years ago by
|
|
1095
|
+
[@0x1eef](https://github.com/0x1eef) and
|
|
1096
|
+
[@antaz](https://github.com/0x1eef). The primary
|
|
1097
|
+
maintainer is [@0x1eef](https://github.com/0x1eef).
|
|
1098
|
+
Over those three years multiple other contributors have
|
|
1099
|
+
contributed to llm.rb as well, and new contributors are
|
|
1100
|
+
always welcome.
|
|
1101
|
+
</details>
|
|
1001
1102
|
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1103
|
+
<details>
|
|
1104
|
+
<summary>How well tested is llm.rb?</summary>
|
|
1105
|
+
<br>
|
|
1005
1106
|
|
|
1006
|
-
|
|
1007
|
-
|
|
1107
|
+
It is battle tested daily.
|
|
1108
|
+
|
|
1109
|
+
The console that is distributed with llm.rb is used
|
|
1110
|
+
to build llm.rb so there is a healthy, active feedback
|
|
1111
|
+
loop. It also powers the [r.uby.dev](https://r.uby.dev)
|
|
1112
|
+
website where multiple llm.rb agents are deployed with
|
|
1113
|
+
the help of [roda-llm](https://github.com/r-uby-dev/roda-llm).
|
|
1114
|
+
I'm also aware of at least one production Rails deployment
|
|
1115
|
+
at a large-ish company.
|
|
1116
|
+
|
|
1117
|
+
And this git repository includes llm.rb agents that help me
|
|
1118
|
+
maintain the documentation and perform other repository
|
|
1119
|
+
maintainence. The feedback loop is constant. Outside of that
|
|
1120
|
+
there is a large test suite that covers live requests (recorded
|
|
1121
|
+
by VCR) and database interactions.
|
|
1008
1122
|
</details>
|
|
1009
1123
|
|
|
1010
1124
|
## See also
|
|
@@ -1019,7 +1133,7 @@ be hosted within a Rails application or other Rack-based applications.
|
|
|
1019
1133
|
|
|
1020
1134
|
The [docs/](docs/) directory contains the full documentation and
|
|
1021
1135
|
the chatbot can find the answers to your questions there. Or you
|
|
1022
|
-
can read them yourself.
|
|
1136
|
+
can read them yourself. :)
|
|
1023
1137
|
|
|
1024
1138
|
## License
|
|
1025
1139
|
|
data/bin/llm.rb
CHANGED
|
@@ -193,6 +193,12 @@ def main(argv)
|
|
|
193
193
|
end
|
|
194
194
|
end
|
|
195
195
|
|
|
196
|
+
##
|
|
197
|
+
# Let's use `AGENTS.md` as the system prompt
|
|
198
|
+
agent_opts = {}
|
|
199
|
+
sysprompt = File.join(Dir.getwd, "AGENTS.md")
|
|
200
|
+
File.file?(sysprompt) ? agent_opts.merge!(instructions: File.read(sysprompt)) : {}
|
|
201
|
+
|
|
196
202
|
##
|
|
197
203
|
# No provider has been given.
|
|
198
204
|
# Try to infer one.
|
|
@@ -255,8 +261,9 @@ def main(argv)
|
|
|
255
261
|
##
|
|
256
262
|
# Let's go!
|
|
257
263
|
concurrency ||= :sequential
|
|
258
|
-
path
|
|
259
|
-
|
|
264
|
+
path = temp ? nil : data[Dir.getwd]
|
|
265
|
+
params = agent_opts.merge(model:, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
266
|
+
agent = LLM::Agent.new(llm, params)
|
|
260
267
|
agent.console
|
|
261
268
|
rescue Interrupt
|
|
262
269
|
warn "llm.rb: Bye!"
|
data/data/alibaba.json
CHANGED
|
@@ -38,7 +38,7 @@
|
|
|
38
38
|
"open_weights": false,
|
|
39
39
|
"limit": {
|
|
40
40
|
"context": 1000000,
|
|
41
|
-
"output":
|
|
41
|
+
"output": 131072
|
|
42
42
|
},
|
|
43
43
|
"cost": {
|
|
44
44
|
"input": 2.5,
|
|
@@ -1193,6 +1193,51 @@
|
|
|
1193
1193
|
"output_audio": 15.11
|
|
1194
1194
|
}
|
|
1195
1195
|
},
|
|
1196
|
+
"kimi-k3": {
|
|
1197
|
+
"id": "kimi-k3",
|
|
1198
|
+
"name": "Kimi K3",
|
|
1199
|
+
"description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
|
|
1200
|
+
"family": "kimi-k3",
|
|
1201
|
+
"attachment": true,
|
|
1202
|
+
"reasoning": true,
|
|
1203
|
+
"reasoning_options": [
|
|
1204
|
+
{
|
|
1205
|
+
"type": "effort",
|
|
1206
|
+
"values": [
|
|
1207
|
+
"low",
|
|
1208
|
+
"high",
|
|
1209
|
+
"max"
|
|
1210
|
+
]
|
|
1211
|
+
}
|
|
1212
|
+
],
|
|
1213
|
+
"tool_call": true,
|
|
1214
|
+
"interleaved": {
|
|
1215
|
+
"field": "reasoning_content"
|
|
1216
|
+
},
|
|
1217
|
+
"structured_output": true,
|
|
1218
|
+
"temperature": false,
|
|
1219
|
+
"release_date": "2026-07-16",
|
|
1220
|
+
"last_updated": "2026-07-16",
|
|
1221
|
+
"modalities": {
|
|
1222
|
+
"input": [
|
|
1223
|
+
"text",
|
|
1224
|
+
"image"
|
|
1225
|
+
],
|
|
1226
|
+
"output": [
|
|
1227
|
+
"text"
|
|
1228
|
+
]
|
|
1229
|
+
},
|
|
1230
|
+
"open_weights": true,
|
|
1231
|
+
"limit": {
|
|
1232
|
+
"context": 1048576,
|
|
1233
|
+
"output": 1048576
|
|
1234
|
+
},
|
|
1235
|
+
"cost": {
|
|
1236
|
+
"input": 3,
|
|
1237
|
+
"output": 15,
|
|
1238
|
+
"cache_read": 0.3
|
|
1239
|
+
}
|
|
1240
|
+
},
|
|
1196
1241
|
"qwen2-5-vl-72b-instruct": {
|
|
1197
1242
|
"id": "qwen2-5-vl-72b-instruct",
|
|
1198
1243
|
"name": "Qwen2.5-VL 72B Instruct",
|
|
@@ -1803,7 +1848,7 @@
|
|
|
1803
1848
|
"open_weights": false,
|
|
1804
1849
|
"limit": {
|
|
1805
1850
|
"context": 1000000,
|
|
1806
|
-
"output":
|
|
1851
|
+
"output": 131072
|
|
1807
1852
|
},
|
|
1808
1853
|
"cost": {
|
|
1809
1854
|
"input": 0.5,
|