llm.rb 15.3.0 → 15.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +121 -3
- data/README.md +180 -64
- data/bin/llm.rb +9 -2
- data/data/alibaba.json +45 -0
- data/data/anthropic.json +67 -0
- data/data/bedrock.json +1466 -397
- data/data/deepinfra.json +148 -16
- data/data/deepseek.json +3 -0
- data/data/mistral.json +42 -0
- data/data/openai.json +186 -0
- data/data/openrouter.json +1538 -378
- data/data/xai.json +53 -20
- data/data/zai.json +90 -4
- data/docs/deepdive/advanced/compaction.md +1 -2
- data/docs/deepdive/advanced/context.md +214 -1
- data/docs/deepdive/advanced/guard.md +9 -57
- data/docs/deepdive/features/builtin_tools.md +14 -16
- data/docs/deepdive/features/console.md +5 -0
- data/docs/deepdive/features/database.md +85 -10
- data/docs/deepdive/fundamentals/agents.md +7 -8
- data/docs/deepdive/fundamentals/providers.md +45 -5
- data/docs/deepdive/fundamentals/schema.md +73 -0
- data/docs/deepdive/fundamentals/tools.md +80 -27
- data/docs/deepdive/media/audio.md +8 -19
- data/docs/deepdive/media/images.md +8 -10
- data/docs/deepdive/media/ocr.md +1 -3
- data/docs/deepdive/reference/cost.md +48 -0
- data/docs/deepdive/reference/tracer.md +76 -0
- data/docs/deepdive.md +1 -1
- data/lib/llm/active_record/message.rb +113 -0
- data/lib/llm/active_record.rb +1 -0
- data/lib/llm/agent.rb +28 -19
- data/lib/llm/console/buffer.rb +9 -1
- data/lib/llm/console.rb +6 -1
- data/lib/llm/context/deserializer.rb +6 -1
- data/lib/llm/context.rb +8 -4
- data/lib/llm/guard.rb +2 -8
- data/lib/llm/provider.rb +16 -0
- data/lib/llm/providers/alibaba.rb +15 -0
- data/lib/llm/providers/anthropic/error_handler.rb +5 -2
- data/lib/llm/providers/anthropic/files.rb +12 -12
- data/lib/llm/providers/anthropic/models.rb +2 -2
- data/lib/llm/providers/anthropic.rb +2 -2
- data/lib/llm/providers/bedrock/error_handler.rb +3 -2
- data/lib/llm/providers/bedrock/models.rb +5 -3
- data/lib/llm/providers/bedrock.rb +2 -2
- data/lib/llm/providers/deepinfra/audio.rb +4 -4
- data/lib/llm/providers/deepinfra/images.rb +4 -4
- data/lib/llm/providers/google/error_handler.rb +5 -2
- data/lib/llm/providers/google/files.rb +10 -10
- data/lib/llm/providers/google/images.rb +2 -2
- data/lib/llm/providers/google/models.rb +2 -2
- data/lib/llm/providers/google.rb +4 -4
- data/lib/llm/providers/ollama/error_handler.rb +5 -2
- data/lib/llm/providers/ollama/models.rb +2 -2
- data/lib/llm/providers/ollama.rb +4 -4
- data/lib/llm/providers/openai/audio.rb +6 -6
- data/lib/llm/providers/openai/error_handler.rb +5 -2
- data/lib/llm/providers/openai/files.rb +10 -10
- data/lib/llm/providers/openai/images.rb +4 -4
- data/lib/llm/providers/openai/models.rb +2 -2
- data/lib/llm/providers/openai/moderations.rb +2 -2
- data/lib/llm/providers/openai/responses.rb +6 -6
- data/lib/llm/providers/openai/vector_stores.rb +22 -22
- data/lib/llm/providers/openai.rb +4 -4
- data/lib/llm/providers/xai/images.rb +4 -4
- data/lib/llm/tracer/telemetry.rb +4 -4
- data/lib/llm/tracer.rb +11 -3
- data/lib/llm/transport/execution.rb +8 -4
- data/lib/llm/version.rb +1 -1
- data/llm.gemspec +2 -2
- metadata +5 -5
- data/lib/llm/guard/loop.rb +0 -89
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: de1f4eeee4762548e75eac2b5d4112de49cf446944b30c545b76e62750aad6c3
|
|
4
|
+
data.tar.gz: 4abc1d2d707717fde266b48bfa626f06d553514761653db8c3cd190cfcffb864
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: c28cadd2a412e2623b472d3be7814aa51a462076782fc847bc73cdc57a601a630d909bb9b40c29d31dd7107b487bf169bdeb245e9803581df0d90d642694bc16
|
|
7
|
+
data.tar.gz: 2fd3c1905d2db4212958e98e537db76d37f0f30f5cee4dda3cbf351c7f0fffe1ce6eaa5b38553e80b9441b029e2edf78c4e9ea75c200b731638dbbe49bd947df
|
data/CHANGELOG.md
CHANGED
|
@@ -17,6 +17,123 @@
|
|
|
17
17
|
|
|
18
18
|
*No unreleased changes yet. Check back after the next release.*
|
|
19
19
|
|
|
20
|
+
## v15.4.0
|
|
21
|
+
|
|
22
|
+
Changes since `v15.3.0`.
|
|
23
|
+
|
|
24
|
+
This release saves token usage with the context state, moves the retry budget
|
|
25
|
+
onto the provider, groups a turn's spans under one trace, and lets an agent
|
|
26
|
+
declare `name`, `description`, `path`, and `tool_budget` with a block. It also
|
|
27
|
+
gives every tracer request an id, adds `LLM::ActiveRecord::Message`, reads
|
|
28
|
+
`AGENTS.md` as the console's system prompt, removes `LLM::Guard::Loop`, and
|
|
29
|
+
refreshes the model registry.
|
|
30
|
+
|
|
31
|
+
### Core
|
|
32
|
+
|
|
33
|
+
* **context: save token usage with the state** <br>
|
|
34
|
+
[`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h-instance_method)
|
|
35
|
+
now writes `context_used` and `context_window`, so a saved state can be
|
|
36
|
+
inspected without loading the runtime. They are never read back, and a payload
|
|
37
|
+
without them still loads.
|
|
38
|
+
|
|
39
|
+
### Provider
|
|
40
|
+
|
|
41
|
+
* **provider: decide the retry budget on the provider** <br>
|
|
42
|
+
[`LLM::Provider#retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#retry_budget-instance_method)
|
|
43
|
+
returns how many times a rate-limited request is retried before the error is
|
|
44
|
+
raised, and defaults to 5. An agent that sets no `retry_budget` of its own
|
|
45
|
+
now takes the budget from its provider instead of special-casing Alibaba, so
|
|
46
|
+
[`LLM::Alibaba`](https://r.uby.dev/api-docs/llm.rb/LLM/Alibaba.html)
|
|
47
|
+
still retries 8 times and a provider that recovers from rate limits slowly can
|
|
48
|
+
return a higher budget. An explicit `retry_budget:` still takes precedence.
|
|
49
|
+
|
|
50
|
+
### Agent
|
|
51
|
+
|
|
52
|
+
* **agent: group a turn's spans into one trace** <br>
|
|
53
|
+
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) now opens a
|
|
54
|
+
`llm.turn` trace group around every turn, so all spans a turn produces share
|
|
55
|
+
one trace id. It uses the agent's tracer when it has one, the provider's
|
|
56
|
+
otherwise; previously a provider-wide tracer split one turn across traces.
|
|
57
|
+
|
|
58
|
+
* **agent: declare `name`, `description`, `path`, and `tool_budget` with a block** <br>
|
|
59
|
+
A block passed to any of these four class-level setters was previously
|
|
60
|
+
ignored, because the call was read as a getter. Now
|
|
61
|
+
[`LLM::Agent.name`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#name-class_method)
|
|
62
|
+
and [`LLM::Agent.description`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#description-class_method)
|
|
63
|
+
store the block, and an instance resolves it against itself, or against its
|
|
64
|
+
ORM record when it is bound to one. `name` and `description` also accept a
|
|
65
|
+
`Symbol` or a `Proc`, and the class-level reader still returns what was
|
|
66
|
+
configured, so read them on an instance for the resolved string.
|
|
67
|
+
|
|
68
|
+
### Console
|
|
69
|
+
|
|
70
|
+
* **console: use `AGENTS.md` as the system prompt** <br>
|
|
71
|
+
`bin/llm.rb` now looks for `AGENTS.md` in the current working directory when
|
|
72
|
+
it boots, and when the file exists its contents become the agent's
|
|
73
|
+
instructions for the session. The instructions are injected once, so a
|
|
74
|
+
resumed session that already has a system message keeps the one it has.
|
|
75
|
+
|
|
76
|
+
* **console: stop a cancelled turn from crashing the console** <br>
|
|
77
|
+
A turn cancelled with Esc can still have chunks queued, and they arrive after
|
|
78
|
+
the buffer that renders them is closed. The streaming path then tried to
|
|
79
|
+
replace a row that no longer existed and raised.
|
|
80
|
+
[`LLM::Console::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Console/Buffer.html)
|
|
81
|
+
now reports whether it has a row to replace through `open?`, the stream drops
|
|
82
|
+
chunks that arrive for a closed buffer, and `replace` keeps the current text
|
|
83
|
+
instead of raising.
|
|
84
|
+
|
|
85
|
+
### Guard
|
|
86
|
+
|
|
87
|
+
* **guard: remove `LLM::Guard::Loop`** <br>
|
|
88
|
+
`LLM::Guard::Loop` is removed. It stopped repeated tool-call patterns, but
|
|
89
|
+
could also interrupt a loop that was making progress, so it did not hold up as
|
|
90
|
+
a default. Agents no longer enable a guard of their own; bound the loop with
|
|
91
|
+
`tool_budget` instead.
|
|
92
|
+
|
|
93
|
+
### ActiveRecord
|
|
94
|
+
|
|
95
|
+
* **activerecord: add `LLM::ActiveRecord::Message`** <br>
|
|
96
|
+
[`LLM::ActiveRecord::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord/Message.html)
|
|
97
|
+
is a virtual model over the JSONB column that holds an agent's state, so
|
|
98
|
+
messages can be filtered, ordered, and counted in SQL instead of in memory.
|
|
99
|
+
`for(agent:)` returns an
|
|
100
|
+
[`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
|
|
101
|
+
with `id`, `role`, `content`, `tools`, and `position` columns; a row's
|
|
102
|
+
`unwrap!` returns the message. The class never materializes as a table.
|
|
103
|
+
|
|
104
|
+
### Tracer
|
|
105
|
+
|
|
106
|
+
* **tracer: give every request an id** <br>
|
|
107
|
+
[`LLM::Tracer#on_request_start`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#on_request_start-instance_method)
|
|
108
|
+
now takes a `request_id:`, a UUIDv7 that the runtime mints when a request
|
|
109
|
+
begins and passes to `on_request_finish` and `on_request_error` for that same
|
|
110
|
+
request. A turn can make many requests, so `trace_group_id` groups a turn but
|
|
111
|
+
not the events inside one request; the id lets a tracer correlate a request's
|
|
112
|
+
start, finish, and error events. The keyword is required, so a subclass that
|
|
113
|
+
overrides these hooks must accept it.
|
|
114
|
+
|
|
115
|
+
* **tracer: record the model the provider answers with** <br>
|
|
116
|
+
[`LLM::Tracer::Telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Telemetry.html)
|
|
117
|
+
now sets `gen_ai.response.model` from the model on the response instead of the
|
|
118
|
+
model requested, so a span reports the model a provider actually served. A
|
|
119
|
+
router model such as `openrouter/auto` records the model it resolved to,
|
|
120
|
+
while `gen_ai.request.model` keeps the name that was asked for.
|
|
121
|
+
|
|
122
|
+
### Registry
|
|
123
|
+
|
|
124
|
+
* **refresh model metadata** <br>
|
|
125
|
+
Update `data/` with current pricing, limits, and capabilities for the
|
|
126
|
+
Alibaba, Anthropic, Bedrock, DeepInfra, DeepSeek, Mistral, OpenAI,
|
|
127
|
+
OpenRouter, xAI, and Z.ai registries. Claude Opus 5.5 reaches Anthropic,
|
|
128
|
+
Bedrock, and OpenRouter, OpenAI adds GPT-6 Sol and GPT-6 Luna, Bedrock adds
|
|
129
|
+
Gemma 4 and more regional Claude Sonnet 4 and GPT-5.6 entries, and
|
|
130
|
+
OpenRouter adds the Xiaomi MiMo V2.6, Nex N2.5, Cohere Command A+, and
|
|
131
|
+
Qwen3.8 Omni Flash models. Z.ai adds `glm-4.6v-flash` and `glm-5.3-flashx`,
|
|
132
|
+
xAI adds Grok 4.7, DeepSeek adds a `low` reasoning effort to
|
|
133
|
+
`deepseek-v4-pro` and deprecates `deepseek-v4-flash` and
|
|
134
|
+
`deepseek-v4-flash-vision-exp`, DeepInfra reprices `tencent/Hy3`, and
|
|
135
|
+
OpenRouter drops `anthropic/claude-opus-4` and `kwaipilot/kat-coder-pro-v2`.
|
|
136
|
+
|
|
20
137
|
## v15.3.0
|
|
21
138
|
|
|
22
139
|
Changes since `v15.2.2`.
|
|
@@ -290,9 +407,10 @@ and a `-v` switch to the CLI, and refreshes the model registry.
|
|
|
290
407
|
The shared [`LLM::Tool::Utils`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Utils.html)
|
|
291
408
|
module now requires the `test-cmd.rb` gem (at `~> 2.7.1`) itself and
|
|
292
409
|
exposes the `spawn` and `wait` helpers, so any tool that includes
|
|
293
|
-
`Utils` gets command spawning without requiring `exec` directly.
|
|
294
|
-
`
|
|
295
|
-
|
|
410
|
+
`Utils` gets command spawning without requiring `exec` directly.
|
|
411
|
+
`Exec` and `ReadFile` include `Utils`, and the tools that shell out
|
|
412
|
+
(`Git`, `Rg`, `Mkdir`, `Ruby`, and `Bundle`) route through `Exec`,
|
|
413
|
+
so they all get the same bounded output.
|
|
296
414
|
|
|
297
415
|
* **tools: route `git`, `rg`, `mkdir`, and `ruby` through `exec`** <br>
|
|
298
416
|
`LLM::Tool::Git`, `LLM::Tool::Rg`, `LLM::Tool::Mkdir`, and
|
data/README.md
CHANGED
|
@@ -52,7 +52,7 @@ an invalid state that would lead to API-level errors. For example,
|
|
|
52
52
|
when a tool call is interrupted it could leave an unanswered tool
|
|
53
53
|
call that a model will reject on the next turn. The runtime takes
|
|
54
54
|
care of this by pruning orphaned tool calls and ensuring that the
|
|
55
|
-
tool loop always remains valid.
|
|
55
|
+
tool loop always remains valid.
|
|
56
56
|
|
|
57
57
|
```ruby
|
|
58
58
|
require "llm"
|
|
@@ -141,7 +141,9 @@ call them on your behalf, and they're one of the most powerful features
|
|
|
141
141
|
for extending the feature set or abilities of a model.
|
|
142
142
|
|
|
143
143
|
The runtime also ships with a catalog of built-in tools for
|
|
144
|
-
filesystem, search, and shell operations
|
|
144
|
+
filesystem, search, and shell operations, and providers expose
|
|
145
|
+
platform-native tools such as web search and code execution that run
|
|
146
|
+
on the provider's side.
|
|
145
147
|
|
|
146
148
|
```ruby
|
|
147
149
|
class ReadFile < LLM::Tool
|
|
@@ -276,13 +278,18 @@ for agents.
|
|
|
276
278
|
|
|
277
279
|
##### Installation
|
|
278
280
|
|
|
279
|
-
The console is distributed with llm.rb
|
|
280
|
-
|
|
281
|
-
|
|
282
|
-
experience:
|
|
281
|
+
The console is distributed with llm.rb but it requires a number
|
|
282
|
+
of optional dependencies to be installed separately. The following
|
|
283
|
+
gems provide the full experience:
|
|
283
284
|
|
|
284
285
|
gem install unicode-display_width curses kramdown xchan.rb test-cmd.rb
|
|
285
286
|
|
|
287
|
+
For convenience it is also possible to just use the following, it
|
|
288
|
+
is a metagem that depends on llm.rb and all the dependencies it requires
|
|
289
|
+
to run the console:
|
|
290
|
+
|
|
291
|
+
gem install llm-shell
|
|
292
|
+
|
|
286
293
|
##### Persistence
|
|
287
294
|
|
|
288
295
|
the `path:` option can be set on an agent for automatic persistence
|
|
@@ -363,21 +370,27 @@ require "active_record"
|
|
|
363
370
|
require "llm"
|
|
364
371
|
require "llm/active_record"
|
|
365
372
|
|
|
366
|
-
|
|
373
|
+
##
|
|
374
|
+
# The Robert agent.
|
|
375
|
+
class Robert < ActiveRecord::Base
|
|
367
376
|
acts_as_agent(format: :jsonb) do |agent|
|
|
368
|
-
agent.set name: "
|
|
369
|
-
description: "
|
|
370
|
-
|
|
377
|
+
agent.set name: "robert",
|
|
378
|
+
description: "robert is an agent that has access to the official " \
|
|
379
|
+
"r.uby.dev GitHub repositories. He can access the repositories " \
|
|
380
|
+
"to answer your question(s) about r.uby.dev projects.",
|
|
381
|
+
instructions: proc { File.read(File.join(__dir__, "robert", "prompt.md")) },
|
|
371
382
|
tools: :tools,
|
|
372
|
-
concurrency: :async
|
|
373
|
-
end
|
|
383
|
+
concurrency: :async,
|
|
374
384
|
|
|
375
|
-
|
|
376
|
-
|
|
377
|
-
|
|
385
|
+
##
|
|
386
|
+
# The maximum number of tool calls per-turn.
|
|
387
|
+
tool_budget: 25,
|
|
378
388
|
|
|
379
|
-
|
|
380
|
-
|
|
389
|
+
##
|
|
390
|
+
# The default tracer that all agents have associated
|
|
391
|
+
# with them. The tracer exports a trace to a couple of
|
|
392
|
+
# SQL tables.
|
|
393
|
+
tracer: proc { Raven::Tracer::SQL.new(llm, agent: self) }
|
|
381
394
|
end
|
|
382
395
|
|
|
383
396
|
##
|
|
@@ -398,10 +411,6 @@ class Raven < ActiveRecord::Base
|
|
|
398
411
|
|
|
399
412
|
private
|
|
400
413
|
|
|
401
|
-
def set_provider
|
|
402
|
-
LLM.deepseek
|
|
403
|
-
end
|
|
404
|
-
|
|
405
414
|
def allowlist
|
|
406
415
|
%w[
|
|
407
416
|
get_commit
|
|
@@ -420,19 +429,18 @@ class Raven < ActiveRecord::Base
|
|
|
420
429
|
end
|
|
421
430
|
end
|
|
422
431
|
|
|
423
|
-
agent =
|
|
432
|
+
agent = Robert.create!
|
|
424
433
|
|
|
425
434
|
##
|
|
426
435
|
# Every call to `talk` automatically persists
|
|
427
|
-
# to the database
|
|
428
|
-
|
|
429
|
-
agent.research_issues
|
|
436
|
+
# to the database.
|
|
437
|
+
agent.talk "what's new on the llm.rb repository?"
|
|
430
438
|
|
|
431
439
|
##
|
|
432
440
|
# The conversation was persisted to database. A
|
|
433
441
|
# fresh instance restores it and continues where
|
|
434
442
|
# we left off
|
|
435
|
-
agent =
|
|
443
|
+
agent = Robert.find(agent.id).talk "and what about roda-llm?"
|
|
436
444
|
|
|
437
445
|
##
|
|
438
446
|
# Start an agent console.
|
|
@@ -440,6 +448,100 @@ agent = Raven.find(agent.id).tap(&:research_codebase)
|
|
|
440
448
|
# The console does not persist back to the database.
|
|
441
449
|
agent.console
|
|
442
450
|
```
|
|
451
|
+
</details>
|
|
452
|
+
<details>
|
|
453
|
+
<summary> SQL optimizations </summary>
|
|
454
|
+
<br>
|
|
455
|
+
|
|
456
|
+
In a database environment the runtime optimizes for
|
|
457
|
+
the PostgreSQL database and its builtin support for
|
|
458
|
+
the `jsonb` column type. An agent fits in a single
|
|
459
|
+
column, on a single row, and that column carries
|
|
460
|
+
everything it has done: messages, tool calls,
|
|
461
|
+
context usage, and so on. It works well in practice
|
|
462
|
+
and means you can store an agent almost anywhere.
|
|
463
|
+
|
|
464
|
+
For scenarios where performance matters most the runtime
|
|
465
|
+
ships with virtual ActiveRecord classes that never materialize
|
|
466
|
+
in your database but provide a SQL view into the column where
|
|
467
|
+
an agent stores its runtime state. They return
|
|
468
|
+
[`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
|
|
469
|
+
objects, so the filtering happens in the database.
|
|
470
|
+
|
|
471
|
+
```ruby
|
|
472
|
+
class Agent < ActiveRecord::Base
|
|
473
|
+
acts_as_agent(format: :jsonb) do |agent|
|
|
474
|
+
agent.set name: "activerecord agent"
|
|
475
|
+
end
|
|
476
|
+
end
|
|
477
|
+
|
|
478
|
+
##
|
|
479
|
+
# Find an instance of your agent
|
|
480
|
+
agent = Agent.find_by(id: 1)
|
|
481
|
+
|
|
482
|
+
##
|
|
483
|
+
# Returns a relation over the agent's messages.
|
|
484
|
+
# It is scoped to the agent, and it yields one
|
|
485
|
+
# instance of LLM::ActiveRecord::Message per
|
|
486
|
+
# message the agent has produced.
|
|
487
|
+
messages = LLM::ActiveRecord::Message.for(agent:)
|
|
488
|
+
|
|
489
|
+
##
|
|
490
|
+
# The relation chains like any other
|
|
491
|
+
messages.where(role: "assistant")
|
|
492
|
+
.order(position: :desc)
|
|
493
|
+
.limit(10)
|
|
494
|
+
|
|
495
|
+
##
|
|
496
|
+
# Count, too
|
|
497
|
+
messages.count
|
|
498
|
+
```
|
|
499
|
+
|
|
500
|
+
**Schema**
|
|
501
|
+
|
|
502
|
+
Each row carries a message, flattened into columns:
|
|
503
|
+
|
|
504
|
+
| column | contents |
|
|
505
|
+
| --- | --- |
|
|
506
|
+
| `agent_id` | the agent a message belongs to |
|
|
507
|
+
| `id` | the message id |
|
|
508
|
+
| `role` | the message role |
|
|
509
|
+
| `content` | the message content |
|
|
510
|
+
| `tools` | the tool calls a message carries |
|
|
511
|
+
| `position` | the position of a message in the conversation |
|
|
512
|
+
| `data` | the whole message, as the runtime stores it |
|
|
513
|
+
|
|
514
|
+
**Indexes**
|
|
515
|
+
|
|
516
|
+
The queries the view runs are already covered. They expand
|
|
517
|
+
one agent, found by primary key, so they are index scans.
|
|
518
|
+
There is nothing to add for
|
|
519
|
+
`LLM::ActiveRecord::Message.for(agent:)`.
|
|
520
|
+
|
|
521
|
+
The queries you write on top of it are not. Once a question
|
|
522
|
+
is asked of every agent, the column is expanded row by row
|
|
523
|
+
and no index helps the view itself. Index the column for
|
|
524
|
+
those questions instead:
|
|
525
|
+
|
|
526
|
+
```sql
|
|
527
|
+
CREATE INDEX index_agents_on_data
|
|
528
|
+
ON agents USING gin (data jsonb_path_ops);
|
|
529
|
+
|
|
530
|
+
CREATE INDEX index_agents_on_context_used
|
|
531
|
+
ON agents (((data ->> 'context_used')::int));
|
|
532
|
+
```
|
|
533
|
+
|
|
534
|
+
The first serves containment (`@>`) and path queries over
|
|
535
|
+
the state as a whole. The second serves a scalar key, and
|
|
536
|
+
the runtime already writes `context_used` and
|
|
537
|
+
`context_window` at the top level, so "sessions over 80%
|
|
538
|
+
full" becomes cheap. Both assume `format: :jsonb`.
|
|
539
|
+
|
|
540
|
+
**However:** an agent's whole conversation lives in one
|
|
541
|
+
value, so every save rewrites it, and a GIN index is
|
|
542
|
+
maintained with it. Prefer an index on a key or two over
|
|
543
|
+
the whole column.
|
|
544
|
+
|
|
443
545
|
</details>
|
|
444
546
|
|
|
445
547
|
<details><summary>MCP</summary>
|
|
@@ -534,10 +636,9 @@ even answer for it. Because it runs before the tool, anything
|
|
|
534
636
|
it intercepts never executes. Policy, validation, quotas, and
|
|
535
637
|
cost ceilings all live here.
|
|
536
638
|
|
|
537
|
-
|
|
538
|
-
|
|
539
|
-
|
|
540
|
-
by default, so agents get loop protection out of the box. To
|
|
639
|
+
Agents and contexts use
|
|
640
|
+
[`LLM::Guard::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Null.html)
|
|
641
|
+
by default, so a guard only runs when you configure one. To
|
|
541
642
|
write your own guard, subclass
|
|
542
643
|
[`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
|
|
543
644
|
and implement
|
|
@@ -737,11 +838,16 @@ is also distributed with llm.rb.
|
|
|
737
838
|
```ruby
|
|
738
839
|
llm = LLM.openai
|
|
739
840
|
llm = LLM.anthropic
|
|
841
|
+
llm = LLM.google
|
|
740
842
|
llm = LLM.deepseek
|
|
741
|
-
llm = LLM.
|
|
843
|
+
llm = LLM.deepinfra
|
|
844
|
+
llm = LLM.xai
|
|
845
|
+
llm = LLM.zai
|
|
742
846
|
llm = LLM.moonshot
|
|
743
847
|
llm = LLM.openrouter
|
|
848
|
+
llm = LLM.alibaba # also: LLM.aliyun
|
|
744
849
|
llm = LLM.mistral
|
|
850
|
+
llm = LLM.bedrock
|
|
745
851
|
```
|
|
746
852
|
</details>
|
|
747
853
|
<details>
|
|
@@ -755,10 +861,14 @@ key at all.
|
|
|
755
861
|
```ruby
|
|
756
862
|
llm = LLM.openai(key: ENV["OPENAI_API_KEY"])
|
|
757
863
|
llm = LLM.anthropic(key: ENV["ANTHROPIC_API_KEY"])
|
|
864
|
+
llm = LLM.google(key: ENV["GOOGLE_API_KEY"])
|
|
758
865
|
llm = LLM.deepseek(key: ENV["DEEPSEEK_API_KEY"])
|
|
759
|
-
llm = LLM.
|
|
866
|
+
llm = LLM.deepinfra(key: ENV["DEEPINFRA_API_KEY"])
|
|
867
|
+
llm = LLM.xai(key: ENV["XAI_API_KEY"])
|
|
868
|
+
llm = LLM.zai(key: ENV["ZHIPU_API_KEY"])
|
|
760
869
|
llm = LLM.moonshot(key: ENV["MOONSHOT_API_KEY"])
|
|
761
870
|
llm = LLM.openrouter(key: ENV["OPENROUTER_API_KEY"])
|
|
871
|
+
llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
|
|
762
872
|
llm = LLM.mistral(key: ENV["MISTRAL_API_KEY"])
|
|
763
873
|
```
|
|
764
874
|
</details>
|
|
@@ -919,22 +1029,18 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
|
|
|
919
1029
|
<details>
|
|
920
1030
|
<summary>Where can I see llm.rb in action?</summary>
|
|
921
1031
|
<br>
|
|
922
|
-
<p>
|
|
923
1032
|
|
|
924
|
-
The [r.uby.dev](https://r.uby.dev) website
|
|
925
|
-
|
|
926
|
-
|
|
927
|
-
|
|
928
|
-
</p>
|
|
1033
|
+
The [r.uby.dev](https://r.uby.dev) website deploys
|
|
1034
|
+
multiple llm.rb agents that guests can interact with
|
|
1035
|
+
and there is even an agent that is connected to this
|
|
1036
|
+
GitHub repository.
|
|
929
1037
|
</details>
|
|
930
1038
|
<details>
|
|
931
1039
|
<summary>What about local LLM support?</summary>
|
|
932
1040
|
<br>
|
|
933
|
-
|
|
1041
|
+
|
|
934
1042
|
The following providers can be run used with models that
|
|
935
|
-
are running on your own hardware.
|
|
936
|
-
tested but not my main driver:
|
|
937
|
-
</p>
|
|
1043
|
+
are running on your own hardware.
|
|
938
1044
|
|
|
939
1045
|
* Ollama
|
|
940
1046
|
* Llamacpp
|
|
@@ -943,29 +1049,27 @@ tested but not my main driver:
|
|
|
943
1049
|
<details>
|
|
944
1050
|
<summary>I have a limited budget. What should I do?</summary>
|
|
945
1051
|
<br>
|
|
946
|
-
|
|
1052
|
+
|
|
947
1053
|
There are a few options. The first option is to host
|
|
948
1054
|
your own model, and use the ollama or llamacpp
|
|
949
1055
|
providers. This can be difficult though because
|
|
950
1056
|
a capable model requires hardware that can
|
|
951
1057
|
match it. If you have the ability to self-host,
|
|
952
1058
|
this would be my first option.
|
|
953
|
-
|
|
954
|
-
<p>
|
|
1059
|
+
|
|
955
1060
|
The second option is DeepSeek. <br>
|
|
956
1061
|
The deepseek-v4-flash model costs pennies to use. <br>
|
|
957
1062
|
And llm.rb has been optimized for deepseek. For example,
|
|
958
1063
|
DeepSeek does not have image generation capabilities
|
|
959
1064
|
but on the llm.rb runtime it does (vector graphics only,
|
|
960
1065
|
though).
|
|
961
|
-
|
|
962
|
-
<p>
|
|
1066
|
+
|
|
963
1067
|
The same is true for structured outputs. DeepSeek does
|
|
964
1068
|
not support structured outputs in the same way as OpenAI or
|
|
965
1069
|
Google, but the llm.rb runtime makes it appear as
|
|
966
1070
|
though it does, through the `json_object` response
|
|
967
1071
|
type.
|
|
968
|
-
|
|
1072
|
+
|
|
969
1073
|
If you're on a budget, DeepSeek is hard to beat.
|
|
970
1074
|
</details>
|
|
971
1075
|
<details>
|
|
@@ -988,23 +1092,35 @@ web</a>.
|
|
|
988
1092
|
<summary>Who maintains llm.rb?</summary>
|
|
989
1093
|
<br>
|
|
990
1094
|
|
|
991
|
-
The llm.rb project
|
|
992
|
-
|
|
993
|
-
|
|
994
|
-
|
|
995
|
-
|
|
996
|
-
|
|
997
|
-
|
|
998
|
-
|
|
999
|
-
|
|
1000
|
-
environments.
|
|
1095
|
+
The llm.rb project was started more than three
|
|
1096
|
+
years ago by
|
|
1097
|
+
[@0x1eef](https://github.com/0x1eef) and
|
|
1098
|
+
[@antaz](https://github.com/0x1eef). The primary
|
|
1099
|
+
maintainer is [@0x1eef](https://github.com/0x1eef).
|
|
1100
|
+
Over those three years multiple other contributors have
|
|
1101
|
+
contributed to llm.rb as well, and new contributors are
|
|
1102
|
+
always welcome.
|
|
1103
|
+
</details>
|
|
1001
1104
|
|
|
1002
|
-
|
|
1003
|
-
|
|
1004
|
-
|
|
1105
|
+
<details>
|
|
1106
|
+
<summary>How well tested is llm.rb?</summary>
|
|
1107
|
+
<br>
|
|
1005
1108
|
|
|
1006
|
-
|
|
1007
|
-
|
|
1109
|
+
It is battle tested daily.
|
|
1110
|
+
|
|
1111
|
+
The console that is distributed with llm.rb is used
|
|
1112
|
+
to build llm.rb so there is a healthy, active feedback
|
|
1113
|
+
loop. It also powers the [r.uby.dev](https://r.uby.dev)
|
|
1114
|
+
website where multiple llm.rb agents are deployed with
|
|
1115
|
+
the help of [roda-llm](https://github.com/r-uby-dev/roda-llm).
|
|
1116
|
+
I'm also aware of at least one production Rails deployment
|
|
1117
|
+
at a large-ish company.
|
|
1118
|
+
|
|
1119
|
+
And this git repository includes llm.rb agents that help me
|
|
1120
|
+
maintain the documentation and perform other repository
|
|
1121
|
+
maintainence. The feedback loop is constant. Outside of that
|
|
1122
|
+
there is a large test suite that covers live requests (recorded
|
|
1123
|
+
by VCR) and database interactions.
|
|
1008
1124
|
</details>
|
|
1009
1125
|
|
|
1010
1126
|
## See also
|
|
@@ -1019,7 +1135,7 @@ be hosted within a Rails application or other Rack-based applications.
|
|
|
1019
1135
|
|
|
1020
1136
|
The [docs/](docs/) directory contains the full documentation and
|
|
1021
1137
|
the chatbot can find the answers to your questions there. Or you
|
|
1022
|
-
can read them yourself.
|
|
1138
|
+
can read them yourself. :)
|
|
1023
1139
|
|
|
1024
1140
|
## License
|
|
1025
1141
|
|
data/bin/llm.rb
CHANGED
|
@@ -193,6 +193,12 @@ def main(argv)
|
|
|
193
193
|
end
|
|
194
194
|
end
|
|
195
195
|
|
|
196
|
+
##
|
|
197
|
+
# Let's use `AGENTS.md` as the system prompt
|
|
198
|
+
agent_opts = {}
|
|
199
|
+
sysprompt = File.join(Dir.getwd, "AGENTS.md")
|
|
200
|
+
File.file?(sysprompt) ? agent_opts.merge!(instructions: File.read(sysprompt)) : {}
|
|
201
|
+
|
|
196
202
|
##
|
|
197
203
|
# No provider has been given.
|
|
198
204
|
# Try to infer one.
|
|
@@ -255,8 +261,9 @@ def main(argv)
|
|
|
255
261
|
##
|
|
256
262
|
# Let's go!
|
|
257
263
|
concurrency ||= :sequential
|
|
258
|
-
path
|
|
259
|
-
|
|
264
|
+
path = temp ? nil : data[Dir.getwd]
|
|
265
|
+
params = agent_opts.merge(model:, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
266
|
+
agent = LLM::Agent.new(llm, params)
|
|
260
267
|
agent.console
|
|
261
268
|
rescue Interrupt
|
|
262
269
|
warn "llm.rb: Bye!"
|
data/data/alibaba.json
CHANGED
|
@@ -1193,6 +1193,51 @@
|
|
|
1193
1193
|
"output_audio": 15.11
|
|
1194
1194
|
}
|
|
1195
1195
|
},
|
|
1196
|
+
"kimi-k3": {
|
|
1197
|
+
"id": "kimi-k3",
|
|
1198
|
+
"name": "Kimi K3",
|
|
1199
|
+
"description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
|
|
1200
|
+
"family": "kimi-k3",
|
|
1201
|
+
"attachment": true,
|
|
1202
|
+
"reasoning": true,
|
|
1203
|
+
"reasoning_options": [
|
|
1204
|
+
{
|
|
1205
|
+
"type": "effort",
|
|
1206
|
+
"values": [
|
|
1207
|
+
"low",
|
|
1208
|
+
"high",
|
|
1209
|
+
"max"
|
|
1210
|
+
]
|
|
1211
|
+
}
|
|
1212
|
+
],
|
|
1213
|
+
"tool_call": true,
|
|
1214
|
+
"interleaved": {
|
|
1215
|
+
"field": "reasoning_content"
|
|
1216
|
+
},
|
|
1217
|
+
"structured_output": true,
|
|
1218
|
+
"temperature": false,
|
|
1219
|
+
"release_date": "2026-07-16",
|
|
1220
|
+
"last_updated": "2026-07-16",
|
|
1221
|
+
"modalities": {
|
|
1222
|
+
"input": [
|
|
1223
|
+
"text",
|
|
1224
|
+
"image"
|
|
1225
|
+
],
|
|
1226
|
+
"output": [
|
|
1227
|
+
"text"
|
|
1228
|
+
]
|
|
1229
|
+
},
|
|
1230
|
+
"open_weights": true,
|
|
1231
|
+
"limit": {
|
|
1232
|
+
"context": 1048576,
|
|
1233
|
+
"output": 1048576
|
|
1234
|
+
},
|
|
1235
|
+
"cost": {
|
|
1236
|
+
"input": 3,
|
|
1237
|
+
"output": 15,
|
|
1238
|
+
"cache_read": 0.3
|
|
1239
|
+
}
|
|
1240
|
+
},
|
|
1196
1241
|
"qwen2-5-vl-72b-instruct": {
|
|
1197
1242
|
"id": "qwen2-5-vl-72b-instruct",
|
|
1198
1243
|
"name": "Qwen2.5-VL 72B Instruct",
|