llm.rb 15.3.0 → 15.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (76) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +148 -3
  3. data/README.md +185 -71
  4. data/bin/llm.rb +9 -2
  5. data/data/alibaba.json +47 -2
  6. data/data/anthropic.json +67 -0
  7. data/data/bedrock.json +1828 -425
  8. data/data/deepinfra.json +148 -16
  9. data/data/deepseek.json +7 -4
  10. data/data/google.json +3 -3
  11. data/data/mistral.json +42 -0
  12. data/data/moonshot.json +1 -1
  13. data/data/openai.json +186 -0
  14. data/data/openrouter.json +1783 -440
  15. data/data/xai.json +53 -20
  16. data/data/zai.json +90 -4
  17. data/docs/deepdive/advanced/compaction.md +1 -2
  18. data/docs/deepdive/advanced/context.md +214 -1
  19. data/docs/deepdive/advanced/guard.md +9 -57
  20. data/docs/deepdive/features/builtin_tools.md +14 -16
  21. data/docs/deepdive/features/console.md +5 -0
  22. data/docs/deepdive/features/database.md +85 -10
  23. data/docs/deepdive/fundamentals/agents.md +7 -8
  24. data/docs/deepdive/fundamentals/providers.md +45 -5
  25. data/docs/deepdive/fundamentals/schema.md +73 -0
  26. data/docs/deepdive/fundamentals/tools.md +80 -27
  27. data/docs/deepdive/media/audio.md +8 -19
  28. data/docs/deepdive/media/images.md +8 -10
  29. data/docs/deepdive/media/ocr.md +1 -3
  30. data/docs/deepdive/reference/cost.md +48 -0
  31. data/docs/deepdive/reference/tracer.md +76 -0
  32. data/docs/deepdive.md +1 -1
  33. data/lib/llm/active_record/message.rb +113 -0
  34. data/lib/llm/active_record.rb +1 -0
  35. data/lib/llm/agent.rb +28 -19
  36. data/lib/llm/console/buffer.rb +9 -1
  37. data/lib/llm/console.rb +6 -1
  38. data/lib/llm/context/deserializer.rb +6 -1
  39. data/lib/llm/context.rb +8 -4
  40. data/lib/llm/guard.rb +2 -8
  41. data/lib/llm/provider.rb +16 -0
  42. data/lib/llm/providers/alibaba.rb +15 -0
  43. data/lib/llm/providers/anthropic/error_handler.rb +5 -2
  44. data/lib/llm/providers/anthropic/files.rb +12 -12
  45. data/lib/llm/providers/anthropic/models.rb +2 -2
  46. data/lib/llm/providers/anthropic.rb +2 -2
  47. data/lib/llm/providers/bedrock/error_handler.rb +3 -2
  48. data/lib/llm/providers/bedrock/models.rb +5 -3
  49. data/lib/llm/providers/bedrock.rb +2 -2
  50. data/lib/llm/providers/deepinfra/audio.rb +4 -4
  51. data/lib/llm/providers/deepinfra/images.rb +4 -4
  52. data/lib/llm/providers/google/error_handler.rb +5 -2
  53. data/lib/llm/providers/google/files.rb +10 -10
  54. data/lib/llm/providers/google/images.rb +2 -2
  55. data/lib/llm/providers/google/models.rb +2 -2
  56. data/lib/llm/providers/google.rb +4 -4
  57. data/lib/llm/providers/ollama/error_handler.rb +5 -2
  58. data/lib/llm/providers/ollama/models.rb +2 -2
  59. data/lib/llm/providers/ollama.rb +4 -4
  60. data/lib/llm/providers/openai/audio.rb +6 -6
  61. data/lib/llm/providers/openai/error_handler.rb +5 -2
  62. data/lib/llm/providers/openai/files.rb +10 -10
  63. data/lib/llm/providers/openai/images.rb +4 -4
  64. data/lib/llm/providers/openai/models.rb +2 -2
  65. data/lib/llm/providers/openai/moderations.rb +2 -2
  66. data/lib/llm/providers/openai/responses.rb +6 -6
  67. data/lib/llm/providers/openai/vector_stores.rb +22 -22
  68. data/lib/llm/providers/openai.rb +4 -4
  69. data/lib/llm/providers/xai/images.rb +4 -4
  70. data/lib/llm/tracer/telemetry.rb +4 -4
  71. data/lib/llm/tracer.rb +11 -3
  72. data/lib/llm/transport/execution.rb +8 -4
  73. data/lib/llm/version.rb +1 -1
  74. data/llm.gemspec +0 -6
  75. metadata +3 -5
  76. data/lib/llm/guard/loop.rb +0 -89
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: a461691c878ffef8343a8f1ae2dad4298ebb631d379d901110ef8ab8641d9c68
4
- data.tar.gz: 95bb85e3debb178ebbe6949cd3cafbb3625d6f20a72d617f08f1a600354911ba
3
+ metadata.gz: 4647447cce1ccfc7e4f062cfa46b6dd0363ffd1a86537200a7877daff983cda3
4
+ data.tar.gz: 31502899d4a6443b893a48baf18c75d02311920166a6d791d671fa33600e93b9
5
5
  SHA512:
6
- metadata.gz: c9b6e723d1bc16d9cc515a9ae2a3fd34072ce88024804d5b7b0e713cff0b08c26e78e2008641763bbc9a3aabea6f262023fa1f243d9fe155f8d8fdf58215d24f
7
- data.tar.gz: e125d9936d34b3550fb504d281760b73bc617aa4b5d42f1f4d1e3b9b3edd98a448c4a7509d010c62b67b1f57def5f3a9f1e088786b860e6d8f08c94bf64aa0f8
6
+ metadata.gz: ee244a4b0d7d317632e1ee2f3dc44b82926086a71f2ded6ba7e3189eff9636dba2eed5f97d4ed6c83140584d3807b4872303a20b369a123be2336c32671758a1
7
+ data.tar.gz: 7ed45ea9770afabb6522b6939c9d333c28a5892e40e32de271d03a8c5fb3f9fa15b5c2ace19e74ac80342bc65e04a7b5edb48b6799a612f5754b8ccccd33bdaf
data/CHANGELOG.md CHANGED
@@ -17,6 +17,150 @@
17
17
 
18
18
  *No unreleased changes yet. Check back after the next release.*
19
19
 
20
+ ## v15.4.1
21
+
22
+ Changes since `v15.4.0`.
23
+
24
+ This release removes the gemspec's post-install message, so installing the gem
25
+ no longer prints the r.uby.dev notice. It also refreshes the model registry
26
+ with current model listings, limits, and pricing.
27
+
28
+ ### Core
29
+
30
+ * **remove the gemspec post install message** <br>
31
+ The gemspec no longer sets `post_install_message`, so installing the
32
+ gem no longer prints the r.uby.dev website notice.
33
+
34
+ ### Registry
35
+
36
+ * **refresh model metadata** <br>
37
+ Update `data/` with current model listings, limits, and pricing for the
38
+ Alibaba, Bedrock, DeepSeek, Google, Moonshot, and OpenRouter registries.
39
+ Bedrock adds the Kimi K3 and GPT-6 Sol and GPT-6 Luna families in both
40
+ the global and US regions and raises the context limit to 1M tokens for
41
+ two models, while Moonshot raises the Kimi K3 output limit to 1M tokens.
42
+ OpenRouter adds `qwen/qwen3.8-max-prime`, `z-ai/glm-5.3-prime`,
43
+ `upstage/solar-mini4`, the Aion 3.5 models, and `stealth/space-bunny-alpha`,
44
+ drops `mistralai/devstral-2512` and a free Ling 3.0 Flash VL entry, and
45
+ reprices several DeepSeek and Mistral models.
46
+
47
+ ## v15.4.0
48
+
49
+ Changes since `v15.3.0`.
50
+
51
+ This release saves token usage with the context state, moves the retry budget
52
+ onto the provider, groups a turn's spans under one trace, and lets an agent
53
+ declare `name`, `description`, `path`, and `tool_budget` with a block. It also
54
+ gives every tracer request an id, adds `LLM::ActiveRecord::Message`, reads
55
+ `AGENTS.md` as the console's system prompt, removes `LLM::Guard::Loop`, and
56
+ refreshes the model registry.
57
+
58
+ ### Core
59
+
60
+ * **context: save token usage with the state** <br>
61
+ [`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h-instance_method)
62
+ now writes `context_used` and `context_window`, so a saved state can be
63
+ inspected without loading the runtime. They are never read back, and a payload
64
+ without them still loads.
65
+
66
+ ### Provider
67
+
68
+ * **provider: decide the retry budget on the provider** <br>
69
+ [`LLM::Provider#retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#retry_budget-instance_method)
70
+ returns how many times a rate-limited request is retried before the error is
71
+ raised, and defaults to 5. An agent that sets no `retry_budget` of its own
72
+ now takes the budget from its provider instead of special-casing Alibaba, so
73
+ [`LLM::Alibaba`](https://r.uby.dev/api-docs/llm.rb/LLM/Alibaba.html)
74
+ still retries 8 times and a provider that recovers from rate limits slowly can
75
+ return a higher budget. An explicit `retry_budget:` still takes precedence.
76
+
77
+ ### Agent
78
+
79
+ * **agent: group a turn's spans into one trace** <br>
80
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) now opens a
81
+ `llm.turn` trace group around every turn, so all spans a turn produces share
82
+ one trace id. It uses the agent's tracer when it has one, the provider's
83
+ otherwise; previously a provider-wide tracer split one turn across traces.
84
+
85
+ * **agent: declare `name`, `description`, `path`, and `tool_budget` with a block** <br>
86
+ A block passed to any of these four class-level setters was previously
87
+ ignored, because the call was read as a getter. Now
88
+ [`LLM::Agent.name`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#name-class_method)
89
+ and [`LLM::Agent.description`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#description-class_method)
90
+ store the block, and an instance resolves it against itself, or against its
91
+ ORM record when it is bound to one. `name` and `description` also accept a
92
+ `Symbol` or a `Proc`, and the class-level reader still returns what was
93
+ configured, so read them on an instance for the resolved string.
94
+
95
+ ### Console
96
+
97
+ * **console: use `AGENTS.md` as the system prompt** <br>
98
+ `bin/llm.rb` now looks for `AGENTS.md` in the current working directory when
99
+ it boots, and when the file exists its contents become the agent's
100
+ instructions for the session. The instructions are injected once, so a
101
+ resumed session that already has a system message keeps the one it has.
102
+
103
+ * **console: stop a cancelled turn from crashing the console** <br>
104
+ A turn cancelled with Esc can still have chunks queued, and they arrive after
105
+ the buffer that renders them is closed. The streaming path then tried to
106
+ replace a row that no longer existed and raised.
107
+ [`LLM::Console::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Console/Buffer.html)
108
+ now reports whether it has a row to replace through `open?`, the stream drops
109
+ chunks that arrive for a closed buffer, and `replace` keeps the current text
110
+ instead of raising.
111
+
112
+ ### Guard
113
+
114
+ * **guard: remove `LLM::Guard::Loop`** <br>
115
+ `LLM::Guard::Loop` is removed. It stopped repeated tool-call patterns, but
116
+ could also interrupt a loop that was making progress, so it did not hold up as
117
+ a default. Agents no longer enable a guard of their own; bound the loop with
118
+ `tool_budget` instead.
119
+
120
+ ### ActiveRecord
121
+
122
+ * **activerecord: add `LLM::ActiveRecord::Message`** <br>
123
+ [`LLM::ActiveRecord::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord/Message.html)
124
+ is a virtual model over the JSONB column that holds an agent's state, so
125
+ messages can be filtered, ordered, and counted in SQL instead of in memory.
126
+ `for(agent:)` returns an
127
+ [`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
128
+ with `id`, `role`, `content`, `tools`, and `position` columns; a row's
129
+ `unwrap!` returns the message. The class never materializes as a table.
130
+
131
+ ### Tracer
132
+
133
+ * **tracer: give every request an id** <br>
134
+ [`LLM::Tracer#on_request_start`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#on_request_start-instance_method)
135
+ now takes a `request_id:`, a UUIDv7 that the runtime mints when a request
136
+ begins and passes to `on_request_finish` and `on_request_error` for that same
137
+ request. A turn can make many requests, so `trace_group_id` groups a turn but
138
+ not the events inside one request; the id lets a tracer correlate a request's
139
+ start, finish, and error events. The keyword is required, so a subclass that
140
+ overrides these hooks must accept it.
141
+
142
+ * **tracer: record the model the provider answers with** <br>
143
+ [`LLM::Tracer::Telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Telemetry.html)
144
+ now sets `gen_ai.response.model` from the model on the response instead of the
145
+ model requested, so a span reports the model a provider actually served. A
146
+ router model such as `openrouter/auto` records the model it resolved to,
147
+ while `gen_ai.request.model` keeps the name that was asked for.
148
+
149
+ ### Registry
150
+
151
+ * **refresh model metadata** <br>
152
+ Update `data/` with current pricing, limits, and capabilities for the
153
+ Alibaba, Anthropic, Bedrock, DeepInfra, DeepSeek, Mistral, OpenAI,
154
+ OpenRouter, xAI, and Z.ai registries. Claude Opus 5.5 reaches Anthropic,
155
+ Bedrock, and OpenRouter, OpenAI adds GPT-6 Sol and GPT-6 Luna, Bedrock adds
156
+ Gemma 4 and more regional Claude Sonnet 4 and GPT-5.6 entries, and
157
+ OpenRouter adds the Xiaomi MiMo V2.6, Nex N2.5, Cohere Command A+, and
158
+ Qwen3.8 Omni Flash models. Z.ai adds `glm-4.6v-flash` and `glm-5.3-flashx`,
159
+ xAI adds Grok 4.7, DeepSeek adds a `low` reasoning effort to
160
+ `deepseek-v4-pro` and deprecates `deepseek-v4-flash` and
161
+ `deepseek-v4-flash-vision-exp`, DeepInfra reprices `tencent/Hy3`, and
162
+ OpenRouter drops `anthropic/claude-opus-4` and `kwaipilot/kat-coder-pro-v2`.
163
+
20
164
  ## v15.3.0
21
165
 
22
166
  Changes since `v15.2.2`.
@@ -290,9 +434,10 @@ and a `-v` switch to the CLI, and refreshes the model registry.
290
434
  The shared [`LLM::Tool::Utils`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Utils.html)
291
435
  module now requires the `test-cmd.rb` gem (at `~> 2.7.1`) itself and
292
436
  exposes the `spawn` and `wait` helpers, so any tool that includes
293
- `Utils` gets command spawning without requiring `exec` directly. The
294
- `Git`, `Mkdir`, `Rg`, `Ruby`, `Exec`, and `Bundle` tools all
295
- inherit their bounded-output protections from this shared runner.
437
+ `Utils` gets command spawning without requiring `exec` directly.
438
+ `Exec` and `ReadFile` include `Utils`, and the tools that shell out
439
+ (`Git`, `Rg`, `Mkdir`, `Ruby`, and `Bundle`) route through `Exec`,
440
+ so they all get the same bounded output.
296
441
 
297
442
  * **tools: route `git`, `rg`, `mkdir`, and `ruby` through `exec`** <br>
298
443
  `LLM::Tool::Git`, `LLM::Tool::Rg`, `LLM::Tool::Mkdir`, and
data/README.md CHANGED
@@ -19,10 +19,13 @@ on CRuby. It has zero runtime dependencies by default, supports
19
19
  concurrent and parallel tool execution and has a single coherent API
20
20
  that spans 14+ providers.
21
21
 
22
- The most effective way to learn about llm.rb is to ask [the r.uby.dev chatbot](https://r.uby.dev)
23
- a question. It is connected to the llm.rb GitHub repository, backed by
24
- ActiveRecord and uses the builtin MCP feature to connect to GitHub. The chatbot
25
- is an llm.rb agent that is deployed with [roda-llm](https://github.com/r-uby-dev/roda-llm#readme).
22
+ It is possible to see llm.rb in action on the
23
+ [the r.uby.dev website](https://r.uby.dev) where
24
+ I am working on building an agentic platform that
25
+ users can use to manage multiple agents that are
26
+ specialized in different areas, and have access to
27
+ different services (eg GitHub, etc). Check it out if
28
+ curious. Still in early development.
26
29
 
27
30
  ## Install
28
31
 
@@ -52,7 +55,7 @@ an invalid state that would lead to API-level errors. For example,
52
55
  when a tool call is interrupted it could leave an unanswered tool
53
56
  call that a model will reject on the next turn. The runtime takes
54
57
  care of this by pruning orphaned tool calls and ensuring that the
55
- tool loop always remains valid.
58
+ tool loop always remains valid.
56
59
 
57
60
  ```ruby
58
61
  require "llm"
@@ -141,7 +144,9 @@ call them on your behalf, and they're one of the most powerful features
141
144
  for extending the feature set or abilities of a model.
142
145
 
143
146
  The runtime also ships with a catalog of built-in tools for
144
- filesystem, search, and shell operations.
147
+ filesystem, search, and shell operations, and providers expose
148
+ platform-native tools such as web search and code execution that run
149
+ on the provider's side.
145
150
 
146
151
  ```ruby
147
152
  class ReadFile < LLM::Tool
@@ -276,13 +281,18 @@ for agents.
276
281
 
277
282
  ##### Installation
278
283
 
279
- The console is distributed with llm.rb so you don't have to install
280
- a separate gem but it requires a number of optional dependencies
281
- to be installed separately. The following gems provide the full
282
- experience:
284
+ The console is distributed with llm.rb but it requires a number
285
+ of optional dependencies to be installed separately. The following
286
+ gems provide the full experience:
283
287
 
284
288
  gem install unicode-display_width curses kramdown xchan.rb test-cmd.rb
285
289
 
290
+ For convenience it is also possible to just use the following, it
291
+ is a metagem that depends on llm.rb and all the dependencies it requires
292
+ to run the console:
293
+
294
+ gem install llm-shell
295
+
286
296
  ##### Persistence
287
297
 
288
298
  the `path:` option can be set on an agent for automatic persistence
@@ -355,29 +365,32 @@ for both Rack-based / Rails-based applications. On databases
355
365
  where it is supported, such as PostgreSQL, the column can be optimized by using
356
366
  the `jsonb` type.
357
367
 
358
- The following example is based on the agent used to power the
359
- [r.uby.dev chatbot](https://r.uby.dev).
360
-
361
368
  ```ruby
362
369
  require "active_record"
363
370
  require "llm"
364
371
  require "llm/active_record"
365
372
 
366
- class Raven < ActiveRecord::Base
373
+ ##
374
+ # The Robert agent.
375
+ class Robert < ActiveRecord::Base
367
376
  acts_as_agent(format: :jsonb) do |agent|
368
- agent.set name: "raven",
369
- description: "a chatbot for the r.uby.dev website",
370
- instructions: proc { File.read(File.join(__dir__, "raven", "prompt.md")) },
377
+ agent.set name: "robert",
378
+ description: "robert is an agent that has access to the official " \
379
+ "r.uby.dev GitHub repositories. He can access the repositories " \
380
+ "to answer your question(s) about r.uby.dev projects.",
381
+ instructions: proc { File.read(File.join(__dir__, "robert", "prompt.md")) },
371
382
  tools: :tools,
372
- concurrency: :async
373
- end
383
+ concurrency: :async,
374
384
 
375
- def research_issues
376
- talk("research open pull requests on r-uby-dev/llm")
377
- end
385
+ ##
386
+ # The maximum number of tool calls per-turn.
387
+ tool_budget: 25,
378
388
 
379
- def research_codebase
380
- talk("research the codebase on r-uby-dev/llm")
389
+ ##
390
+ # The default tracer that all agents have associated
391
+ # with them. The tracer exports a trace to a couple of
392
+ # SQL tables.
393
+ tracer: proc { Raven::Tracer::SQL.new(llm, agent: self) }
381
394
  end
382
395
 
383
396
  ##
@@ -398,10 +411,6 @@ class Raven < ActiveRecord::Base
398
411
 
399
412
  private
400
413
 
401
- def set_provider
402
- LLM.deepseek
403
- end
404
-
405
414
  def allowlist
406
415
  %w[
407
416
  get_commit
@@ -420,19 +429,18 @@ class Raven < ActiveRecord::Base
420
429
  end
421
430
  end
422
431
 
423
- agent = Raven.create!
432
+ agent = Robert.create!
424
433
 
425
434
  ##
426
435
  # Every call to `talk` automatically persists
427
- # to the database (under the hood research_issues
428
- # calls the talk method)
429
- agent.research_issues
436
+ # to the database.
437
+ agent.talk "what's new on the llm.rb repository?"
430
438
 
431
439
  ##
432
440
  # The conversation was persisted to database. A
433
441
  # fresh instance restores it and continues where
434
442
  # we left off
435
- agent = Raven.find(agent.id).tap(&:research_codebase)
443
+ agent = Robert.find(agent.id).talk "and what about roda-llm?"
436
444
 
437
445
  ##
438
446
  # Start an agent console.
@@ -440,6 +448,100 @@ agent = Raven.find(agent.id).tap(&:research_codebase)
440
448
  # The console does not persist back to the database.
441
449
  agent.console
442
450
  ```
451
+ </details>
452
+ <details>
453
+ <summary> SQL optimizations </summary>
454
+ <br>
455
+
456
+ In a database environment the runtime optimizes for
457
+ the PostgreSQL database and its builtin support for
458
+ the `jsonb` column type. An agent fits in a single
459
+ column, on a single row, and that column carries
460
+ everything it has done: messages, tool calls,
461
+ context usage, and so on. It works well in practice
462
+ and means you can store an agent almost anywhere.
463
+
464
+ For scenarios where performance matters most the runtime
465
+ ships with virtual ActiveRecord classes that never materialize
466
+ in your database but provide a SQL view into the column where
467
+ an agent stores its runtime state. They return
468
+ [`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
469
+ objects, so the filtering happens in the database.
470
+
471
+ ```ruby
472
+ class Agent < ActiveRecord::Base
473
+ acts_as_agent(format: :jsonb) do |agent|
474
+ agent.set name: "activerecord agent"
475
+ end
476
+ end
477
+
478
+ ##
479
+ # Find an instance of your agent
480
+ agent = Agent.find_by(id: 1)
481
+
482
+ ##
483
+ # Returns a relation over the agent's messages.
484
+ # It is scoped to the agent, and it yields one
485
+ # instance of LLM::ActiveRecord::Message per
486
+ # message the agent has produced.
487
+ messages = LLM::ActiveRecord::Message.for(agent:)
488
+
489
+ ##
490
+ # The relation chains like any other
491
+ messages.where(role: "assistant")
492
+ .order(position: :desc)
493
+ .limit(10)
494
+
495
+ ##
496
+ # Count, too
497
+ messages.count
498
+ ```
499
+
500
+ **Schema**
501
+
502
+ Each row carries a message, flattened into columns:
503
+
504
+ | column | contents |
505
+ | --- | --- |
506
+ | `agent_id` | the agent a message belongs to |
507
+ | `id` | the message id |
508
+ | `role` | the message role |
509
+ | `content` | the message content |
510
+ | `tools` | the tool calls a message carries |
511
+ | `position` | the position of a message in the conversation |
512
+ | `data` | the whole message, as the runtime stores it |
513
+
514
+ **Indexes**
515
+
516
+ The queries the view runs are already covered. They expand
517
+ one agent, found by primary key, so they are index scans.
518
+ There is nothing to add for
519
+ `LLM::ActiveRecord::Message.for(agent:)`.
520
+
521
+ The queries you write on top of it are not. Once a question
522
+ is asked of every agent, the column is expanded row by row
523
+ and no index helps the view itself. Index the column for
524
+ those questions instead:
525
+
526
+ ```sql
527
+ CREATE INDEX index_agents_on_data
528
+ ON agents USING gin (data jsonb_path_ops);
529
+
530
+ CREATE INDEX index_agents_on_context_used
531
+ ON agents (((data ->> 'context_used')::int));
532
+ ```
533
+
534
+ The first serves containment (`@>`) and path queries over
535
+ the state as a whole. The second serves a scalar key, and
536
+ the runtime already writes `context_used` and
537
+ `context_window` at the top level, so "sessions over 80%
538
+ full" becomes cheap. Both assume `format: :jsonb`.
539
+
540
+ **However:** an agent's whole conversation lives in one
541
+ value, so every save rewrites it, and a GIN index is
542
+ maintained with it. Prefer an index on a key or two over
543
+ the whole column.
544
+
443
545
  </details>
444
546
 
445
547
  <details><summary>MCP</summary>
@@ -534,10 +636,9 @@ even answer for it. Because it runs before the tool, anything
534
636
  it intercepts never executes. Policy, validation, quotas, and
535
637
  cost ceilings all live here.
536
638
 
537
- [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
538
- enables
539
- [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
540
- by default, so agents get loop protection out of the box. To
639
+ Agents and contexts use
640
+ [`LLM::Guard::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Null.html)
641
+ by default, so a guard only runs when you configure one. To
541
642
  write your own guard, subclass
542
643
  [`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
543
644
  and implement
@@ -737,11 +838,16 @@ is also distributed with llm.rb.
737
838
  ```ruby
738
839
  llm = LLM.openai
739
840
  llm = LLM.anthropic
841
+ llm = LLM.google
740
842
  llm = LLM.deepseek
741
- llm = LLM.alibaba # also: LLM.aliyun
843
+ llm = LLM.deepinfra
844
+ llm = LLM.xai
845
+ llm = LLM.zai
742
846
  llm = LLM.moonshot
743
847
  llm = LLM.openrouter
848
+ llm = LLM.alibaba # also: LLM.aliyun
744
849
  llm = LLM.mistral
850
+ llm = LLM.bedrock
745
851
  ```
746
852
  </details>
747
853
  <details>
@@ -755,10 +861,14 @@ key at all.
755
861
  ```ruby
756
862
  llm = LLM.openai(key: ENV["OPENAI_API_KEY"])
757
863
  llm = LLM.anthropic(key: ENV["ANTHROPIC_API_KEY"])
864
+ llm = LLM.google(key: ENV["GOOGLE_API_KEY"])
758
865
  llm = LLM.deepseek(key: ENV["DEEPSEEK_API_KEY"])
759
- llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
866
+ llm = LLM.deepinfra(key: ENV["DEEPINFRA_API_KEY"])
867
+ llm = LLM.xai(key: ENV["XAI_API_KEY"])
868
+ llm = LLM.zai(key: ENV["ZHIPU_API_KEY"])
760
869
  llm = LLM.moonshot(key: ENV["MOONSHOT_API_KEY"])
761
870
  llm = LLM.openrouter(key: ENV["OPENROUTER_API_KEY"])
871
+ llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
762
872
  llm = LLM.mistral(key: ENV["MISTRAL_API_KEY"])
763
873
  ```
764
874
  </details>
@@ -919,22 +1029,16 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
919
1029
  <details>
920
1030
  <summary>Where can I see llm.rb in action?</summary>
921
1031
  <br>
922
- <p>
923
1032
 
924
- The [r.uby.dev](https://r.uby.dev) website is powered
925
- by llm.rb and its builtin MCP feature. It is connected
926
- to this very GitHub repository. It is designed to help
927
- you learn and troubleshoot llm.rb.
928
- </p>
1033
+ The [r.uby.dev](https://r.uby.dev) website.
1034
+
929
1035
  </details>
930
1036
  <details>
931
1037
  <summary>What about local LLM support?</summary>
932
1038
  <br>
933
- <p>
1039
+
934
1040
  The following providers can be run used with models that
935
- are running on your own hardware. They're reasonably well
936
- tested but not my main driver:
937
- </p>
1041
+ are running on your own hardware.
938
1042
 
939
1043
  * Ollama
940
1044
  * Llamacpp
@@ -943,29 +1047,27 @@ tested but not my main driver:
943
1047
  <details>
944
1048
  <summary>I have a limited budget. What should I do?</summary>
945
1049
  <br>
946
- <p>
1050
+
947
1051
  There are a few options. The first option is to host
948
1052
  your own model, and use the ollama or llamacpp
949
1053
  providers. This can be difficult though because
950
1054
  a capable model requires hardware that can
951
1055
  match it. If you have the ability to self-host,
952
1056
  this would be my first option.
953
- </p>
954
- <p>
1057
+
955
1058
  The second option is DeepSeek. <br>
956
1059
  The deepseek-v4-flash model costs pennies to use. <br>
957
1060
  And llm.rb has been optimized for deepseek. For example,
958
1061
  DeepSeek does not have image generation capabilities
959
1062
  but on the llm.rb runtime it does (vector graphics only,
960
1063
  though).
961
- </p>
962
- <p>
1064
+
963
1065
  The same is true for structured outputs. DeepSeek does
964
1066
  not support structured outputs in the same way as OpenAI or
965
1067
  Google, but the llm.rb runtime makes it appear as
966
1068
  though it does, through the `json_object` response
967
1069
  type.
968
- </p>
1070
+
969
1071
  If you're on a budget, DeepSeek is hard to beat.
970
1072
  </details>
971
1073
  <details>
@@ -988,23 +1090,35 @@ web</a>.
988
1090
  <summary>Who maintains llm.rb?</summary>
989
1091
  <br>
990
1092
 
991
- The llm.rb project is maintained primarily by one
992
- person. llm.rb has been in active development for more
993
- than three years and over that time multiple other
994
- contributors have contributed to llm.rb as well. New
995
- contributors are always welcome.
996
-
997
- I use the console that is distributed with llm.rb to build
998
- llm.rb itself so there is a healthy feedback loop and
999
- llm.rb has also been battle tested in production
1000
- environments.
1093
+ The llm.rb project was started more than three
1094
+ years ago by
1095
+ [@0x1eef](https://github.com/0x1eef) and
1096
+ [@antaz](https://github.com/0x1eef). The primary
1097
+ maintainer is [@0x1eef](https://github.com/0x1eef).
1098
+ Over those three years multiple other contributors have
1099
+ contributed to llm.rb as well, and new contributors are
1100
+ always welcome.
1101
+ </details>
1001
1102
 
1002
- I have also also written llm.rb agents within the
1003
- repository that help me maintain the documentation,
1004
- and backport changes to the mruby-llm runtime as well.
1103
+ <details>
1104
+ <summary>How well tested is llm.rb?</summary>
1105
+ <br>
1005
1106
 
1006
- I am constantly focused on improving llm.rb by using
1007
- it as my primary driver for development.
1107
+ It is battle tested daily.
1108
+
1109
+ The console that is distributed with llm.rb is used
1110
+ to build llm.rb so there is a healthy, active feedback
1111
+ loop. It also powers the [r.uby.dev](https://r.uby.dev)
1112
+ website where multiple llm.rb agents are deployed with
1113
+ the help of [roda-llm](https://github.com/r-uby-dev/roda-llm).
1114
+ I'm also aware of at least one production Rails deployment
1115
+ at a large-ish company.
1116
+
1117
+ And this git repository includes llm.rb agents that help me
1118
+ maintain the documentation and perform other repository
1119
+ maintainence. The feedback loop is constant. Outside of that
1120
+ there is a large test suite that covers live requests (recorded
1121
+ by VCR) and database interactions.
1008
1122
  </details>
1009
1123
 
1010
1124
  ## See also
@@ -1019,7 +1133,7 @@ be hosted within a Rails application or other Rack-based applications.
1019
1133
 
1020
1134
  The [docs/](docs/) directory contains the full documentation and
1021
1135
  the chatbot can find the answers to your questions there. Or you
1022
- can read them yourself.
1136
+ can read them yourself. :)
1023
1137
 
1024
1138
  ## License
1025
1139
 
data/bin/llm.rb CHANGED
@@ -193,6 +193,12 @@ def main(argv)
193
193
  end
194
194
  end
195
195
 
196
+ ##
197
+ # Let's use `AGENTS.md` as the system prompt
198
+ agent_opts = {}
199
+ sysprompt = File.join(Dir.getwd, "AGENTS.md")
200
+ File.file?(sysprompt) ? agent_opts.merge!(instructions: File.read(sysprompt)) : {}
201
+
196
202
  ##
197
203
  # No provider has been given.
198
204
  # Try to infer one.
@@ -255,8 +261,9 @@ def main(argv)
255
261
  ##
256
262
  # Let's go!
257
263
  concurrency ||= :sequential
258
- path = temp ? nil : data[Dir.getwd]
259
- agent = LLM::Agent.new(llm, model:, path:, concurrency:, tools: LLM::Tool.subclasses)
264
+ path = temp ? nil : data[Dir.getwd]
265
+ params = agent_opts.merge(model:, path:, concurrency:, tools: LLM::Tool.subclasses)
266
+ agent = LLM::Agent.new(llm, params)
260
267
  agent.console
261
268
  rescue Interrupt
262
269
  warn "llm.rb: Bye!"
data/data/alibaba.json CHANGED
@@ -38,7 +38,7 @@
38
38
  "open_weights": false,
39
39
  "limit": {
40
40
  "context": 1000000,
41
- "output": 65536
41
+ "output": 131072
42
42
  },
43
43
  "cost": {
44
44
  "input": 2.5,
@@ -1193,6 +1193,51 @@
1193
1193
  "output_audio": 15.11
1194
1194
  }
1195
1195
  },
1196
+ "kimi-k3": {
1197
+ "id": "kimi-k3",
1198
+ "name": "Kimi K3",
1199
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
1200
+ "family": "kimi-k3",
1201
+ "attachment": true,
1202
+ "reasoning": true,
1203
+ "reasoning_options": [
1204
+ {
1205
+ "type": "effort",
1206
+ "values": [
1207
+ "low",
1208
+ "high",
1209
+ "max"
1210
+ ]
1211
+ }
1212
+ ],
1213
+ "tool_call": true,
1214
+ "interleaved": {
1215
+ "field": "reasoning_content"
1216
+ },
1217
+ "structured_output": true,
1218
+ "temperature": false,
1219
+ "release_date": "2026-07-16",
1220
+ "last_updated": "2026-07-16",
1221
+ "modalities": {
1222
+ "input": [
1223
+ "text",
1224
+ "image"
1225
+ ],
1226
+ "output": [
1227
+ "text"
1228
+ ]
1229
+ },
1230
+ "open_weights": true,
1231
+ "limit": {
1232
+ "context": 1048576,
1233
+ "output": 1048576
1234
+ },
1235
+ "cost": {
1236
+ "input": 3,
1237
+ "output": 15,
1238
+ "cache_read": 0.3
1239
+ }
1240
+ },
1196
1241
  "qwen2-5-vl-72b-instruct": {
1197
1242
  "id": "qwen2-5-vl-72b-instruct",
1198
1243
  "name": "Qwen2.5-VL 72B Instruct",
@@ -1803,7 +1848,7 @@
1803
1848
  "open_weights": false,
1804
1849
  "limit": {
1805
1850
  "context": 1000000,
1806
- "output": 65536
1851
+ "output": 131072
1807
1852
  },
1808
1853
  "cost": {
1809
1854
  "input": 0.5,