llm.rb 15.3.0 → 15.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +121 -3
  3. data/README.md +180 -64
  4. data/bin/llm.rb +9 -2
  5. data/data/alibaba.json +45 -0
  6. data/data/anthropic.json +67 -0
  7. data/data/bedrock.json +1466 -397
  8. data/data/deepinfra.json +148 -16
  9. data/data/deepseek.json +3 -0
  10. data/data/mistral.json +42 -0
  11. data/data/openai.json +186 -0
  12. data/data/openrouter.json +1538 -378
  13. data/data/xai.json +53 -20
  14. data/data/zai.json +90 -4
  15. data/docs/deepdive/advanced/compaction.md +1 -2
  16. data/docs/deepdive/advanced/context.md +214 -1
  17. data/docs/deepdive/advanced/guard.md +9 -57
  18. data/docs/deepdive/features/builtin_tools.md +14 -16
  19. data/docs/deepdive/features/console.md +5 -0
  20. data/docs/deepdive/features/database.md +85 -10
  21. data/docs/deepdive/fundamentals/agents.md +7 -8
  22. data/docs/deepdive/fundamentals/providers.md +45 -5
  23. data/docs/deepdive/fundamentals/schema.md +73 -0
  24. data/docs/deepdive/fundamentals/tools.md +80 -27
  25. data/docs/deepdive/media/audio.md +8 -19
  26. data/docs/deepdive/media/images.md +8 -10
  27. data/docs/deepdive/media/ocr.md +1 -3
  28. data/docs/deepdive/reference/cost.md +48 -0
  29. data/docs/deepdive/reference/tracer.md +76 -0
  30. data/docs/deepdive.md +1 -1
  31. data/lib/llm/active_record/message.rb +113 -0
  32. data/lib/llm/active_record.rb +1 -0
  33. data/lib/llm/agent.rb +28 -19
  34. data/lib/llm/console/buffer.rb +9 -1
  35. data/lib/llm/console.rb +6 -1
  36. data/lib/llm/context/deserializer.rb +6 -1
  37. data/lib/llm/context.rb +8 -4
  38. data/lib/llm/guard.rb +2 -8
  39. data/lib/llm/provider.rb +16 -0
  40. data/lib/llm/providers/alibaba.rb +15 -0
  41. data/lib/llm/providers/anthropic/error_handler.rb +5 -2
  42. data/lib/llm/providers/anthropic/files.rb +12 -12
  43. data/lib/llm/providers/anthropic/models.rb +2 -2
  44. data/lib/llm/providers/anthropic.rb +2 -2
  45. data/lib/llm/providers/bedrock/error_handler.rb +3 -2
  46. data/lib/llm/providers/bedrock/models.rb +5 -3
  47. data/lib/llm/providers/bedrock.rb +2 -2
  48. data/lib/llm/providers/deepinfra/audio.rb +4 -4
  49. data/lib/llm/providers/deepinfra/images.rb +4 -4
  50. data/lib/llm/providers/google/error_handler.rb +5 -2
  51. data/lib/llm/providers/google/files.rb +10 -10
  52. data/lib/llm/providers/google/images.rb +2 -2
  53. data/lib/llm/providers/google/models.rb +2 -2
  54. data/lib/llm/providers/google.rb +4 -4
  55. data/lib/llm/providers/ollama/error_handler.rb +5 -2
  56. data/lib/llm/providers/ollama/models.rb +2 -2
  57. data/lib/llm/providers/ollama.rb +4 -4
  58. data/lib/llm/providers/openai/audio.rb +6 -6
  59. data/lib/llm/providers/openai/error_handler.rb +5 -2
  60. data/lib/llm/providers/openai/files.rb +10 -10
  61. data/lib/llm/providers/openai/images.rb +4 -4
  62. data/lib/llm/providers/openai/models.rb +2 -2
  63. data/lib/llm/providers/openai/moderations.rb +2 -2
  64. data/lib/llm/providers/openai/responses.rb +6 -6
  65. data/lib/llm/providers/openai/vector_stores.rb +22 -22
  66. data/lib/llm/providers/openai.rb +4 -4
  67. data/lib/llm/providers/xai/images.rb +4 -4
  68. data/lib/llm/tracer/telemetry.rb +4 -4
  69. data/lib/llm/tracer.rb +11 -3
  70. data/lib/llm/transport/execution.rb +8 -4
  71. data/lib/llm/version.rb +1 -1
  72. data/llm.gemspec +2 -2
  73. metadata +5 -5
  74. data/lib/llm/guard/loop.rb +0 -89
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: a461691c878ffef8343a8f1ae2dad4298ebb631d379d901110ef8ab8641d9c68
4
- data.tar.gz: 95bb85e3debb178ebbe6949cd3cafbb3625d6f20a72d617f08f1a600354911ba
3
+ metadata.gz: de1f4eeee4762548e75eac2b5d4112de49cf446944b30c545b76e62750aad6c3
4
+ data.tar.gz: 4abc1d2d707717fde266b48bfa626f06d553514761653db8c3cd190cfcffb864
5
5
  SHA512:
6
- metadata.gz: c9b6e723d1bc16d9cc515a9ae2a3fd34072ce88024804d5b7b0e713cff0b08c26e78e2008641763bbc9a3aabea6f262023fa1f243d9fe155f8d8fdf58215d24f
7
- data.tar.gz: e125d9936d34b3550fb504d281760b73bc617aa4b5d42f1f4d1e3b9b3edd98a448c4a7509d010c62b67b1f57def5f3a9f1e088786b860e6d8f08c94bf64aa0f8
6
+ metadata.gz: c28cadd2a412e2623b472d3be7814aa51a462076782fc847bc73cdc57a601a630d909bb9b40c29d31dd7107b487bf169bdeb245e9803581df0d90d642694bc16
7
+ data.tar.gz: 2fd3c1905d2db4212958e98e537db76d37f0f30f5cee4dda3cbf351c7f0fffe1ce6eaa5b38553e80b9441b029e2edf78c4e9ea75c200b731638dbbe49bd947df
data/CHANGELOG.md CHANGED
@@ -17,6 +17,123 @@
17
17
 
18
18
  *No unreleased changes yet. Check back after the next release.*
19
19
 
20
+ ## v15.4.0
21
+
22
+ Changes since `v15.3.0`.
23
+
24
+ This release saves token usage with the context state, moves the retry budget
25
+ onto the provider, groups a turn's spans under one trace, and lets an agent
26
+ declare `name`, `description`, `path`, and `tool_budget` with a block. It also
27
+ gives every tracer request an id, adds `LLM::ActiveRecord::Message`, reads
28
+ `AGENTS.md` as the console's system prompt, removes `LLM::Guard::Loop`, and
29
+ refreshes the model registry.
30
+
31
+ ### Core
32
+
33
+ * **context: save token usage with the state** <br>
34
+ [`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h-instance_method)
35
+ now writes `context_used` and `context_window`, so a saved state can be
36
+ inspected without loading the runtime. They are never read back, and a payload
37
+ without them still loads.
38
+
39
+ ### Provider
40
+
41
+ * **provider: decide the retry budget on the provider** <br>
42
+ [`LLM::Provider#retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#retry_budget-instance_method)
43
+ returns how many times a rate-limited request is retried before the error is
44
+ raised, and defaults to 5. An agent that sets no `retry_budget` of its own
45
+ now takes the budget from its provider instead of special-casing Alibaba, so
46
+ [`LLM::Alibaba`](https://r.uby.dev/api-docs/llm.rb/LLM/Alibaba.html)
47
+ still retries 8 times and a provider that recovers from rate limits slowly can
48
+ return a higher budget. An explicit `retry_budget:` still takes precedence.
49
+
50
+ ### Agent
51
+
52
+ * **agent: group a turn's spans into one trace** <br>
53
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) now opens a
54
+ `llm.turn` trace group around every turn, so all spans a turn produces share
55
+ one trace id. It uses the agent's tracer when it has one, the provider's
56
+ otherwise; previously a provider-wide tracer split one turn across traces.
57
+
58
+ * **agent: declare `name`, `description`, `path`, and `tool_budget` with a block** <br>
59
+ A block passed to any of these four class-level setters was previously
60
+ ignored, because the call was read as a getter. Now
61
+ [`LLM::Agent.name`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#name-class_method)
62
+ and [`LLM::Agent.description`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#description-class_method)
63
+ store the block, and an instance resolves it against itself, or against its
64
+ ORM record when it is bound to one. `name` and `description` also accept a
65
+ `Symbol` or a `Proc`, and the class-level reader still returns what was
66
+ configured, so read them on an instance for the resolved string.
67
+
68
+ ### Console
69
+
70
+ * **console: use `AGENTS.md` as the system prompt** <br>
71
+ `bin/llm.rb` now looks for `AGENTS.md` in the current working directory when
72
+ it boots, and when the file exists its contents become the agent's
73
+ instructions for the session. The instructions are injected once, so a
74
+ resumed session that already has a system message keeps the one it has.
75
+
76
+ * **console: stop a cancelled turn from crashing the console** <br>
77
+ A turn cancelled with Esc can still have chunks queued, and they arrive after
78
+ the buffer that renders them is closed. The streaming path then tried to
79
+ replace a row that no longer existed and raised.
80
+ [`LLM::Console::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Console/Buffer.html)
81
+ now reports whether it has a row to replace through `open?`, the stream drops
82
+ chunks that arrive for a closed buffer, and `replace` keeps the current text
83
+ instead of raising.
84
+
85
+ ### Guard
86
+
87
+ * **guard: remove `LLM::Guard::Loop`** <br>
88
+ `LLM::Guard::Loop` is removed. It stopped repeated tool-call patterns, but
89
+ could also interrupt a loop that was making progress, so it did not hold up as
90
+ a default. Agents no longer enable a guard of their own; bound the loop with
91
+ `tool_budget` instead.
92
+
93
+ ### ActiveRecord
94
+
95
+ * **activerecord: add `LLM::ActiveRecord::Message`** <br>
96
+ [`LLM::ActiveRecord::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord/Message.html)
97
+ is a virtual model over the JSONB column that holds an agent's state, so
98
+ messages can be filtered, ordered, and counted in SQL instead of in memory.
99
+ `for(agent:)` returns an
100
+ [`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
101
+ with `id`, `role`, `content`, `tools`, and `position` columns; a row's
102
+ `unwrap!` returns the message. The class never materializes as a table.
103
+
104
+ ### Tracer
105
+
106
+ * **tracer: give every request an id** <br>
107
+ [`LLM::Tracer#on_request_start`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#on_request_start-instance_method)
108
+ now takes a `request_id:`, a UUIDv7 that the runtime mints when a request
109
+ begins and passes to `on_request_finish` and `on_request_error` for that same
110
+ request. A turn can make many requests, so `trace_group_id` groups a turn but
111
+ not the events inside one request; the id lets a tracer correlate a request's
112
+ start, finish, and error events. The keyword is required, so a subclass that
113
+ overrides these hooks must accept it.
114
+
115
+ * **tracer: record the model the provider answers with** <br>
116
+ [`LLM::Tracer::Telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Telemetry.html)
117
+ now sets `gen_ai.response.model` from the model on the response instead of the
118
+ model requested, so a span reports the model a provider actually served. A
119
+ router model such as `openrouter/auto` records the model it resolved to,
120
+ while `gen_ai.request.model` keeps the name that was asked for.
121
+
122
+ ### Registry
123
+
124
+ * **refresh model metadata** <br>
125
+ Update `data/` with current pricing, limits, and capabilities for the
126
+ Alibaba, Anthropic, Bedrock, DeepInfra, DeepSeek, Mistral, OpenAI,
127
+ OpenRouter, xAI, and Z.ai registries. Claude Opus 5.5 reaches Anthropic,
128
+ Bedrock, and OpenRouter, OpenAI adds GPT-6 Sol and GPT-6 Luna, Bedrock adds
129
+ Gemma 4 and more regional Claude Sonnet 4 and GPT-5.6 entries, and
130
+ OpenRouter adds the Xiaomi MiMo V2.6, Nex N2.5, Cohere Command A+, and
131
+ Qwen3.8 Omni Flash models. Z.ai adds `glm-4.6v-flash` and `glm-5.3-flashx`,
132
+ xAI adds Grok 4.7, DeepSeek adds a `low` reasoning effort to
133
+ `deepseek-v4-pro` and deprecates `deepseek-v4-flash` and
134
+ `deepseek-v4-flash-vision-exp`, DeepInfra reprices `tencent/Hy3`, and
135
+ OpenRouter drops `anthropic/claude-opus-4` and `kwaipilot/kat-coder-pro-v2`.
136
+
20
137
  ## v15.3.0
21
138
 
22
139
  Changes since `v15.2.2`.
@@ -290,9 +407,10 @@ and a `-v` switch to the CLI, and refreshes the model registry.
290
407
  The shared [`LLM::Tool::Utils`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Utils.html)
291
408
  module now requires the `test-cmd.rb` gem (at `~> 2.7.1`) itself and
292
409
  exposes the `spawn` and `wait` helpers, so any tool that includes
293
- `Utils` gets command spawning without requiring `exec` directly. The
294
- `Git`, `Mkdir`, `Rg`, `Ruby`, `Exec`, and `Bundle` tools all
295
- inherit their bounded-output protections from this shared runner.
410
+ `Utils` gets command spawning without requiring `exec` directly.
411
+ `Exec` and `ReadFile` include `Utils`, and the tools that shell out
412
+ (`Git`, `Rg`, `Mkdir`, `Ruby`, and `Bundle`) route through `Exec`,
413
+ so they all get the same bounded output.
296
414
 
297
415
  * **tools: route `git`, `rg`, `mkdir`, and `ruby` through `exec`** <br>
298
416
  `LLM::Tool::Git`, `LLM::Tool::Rg`, `LLM::Tool::Mkdir`, and
data/README.md CHANGED
@@ -52,7 +52,7 @@ an invalid state that would lead to API-level errors. For example,
52
52
  when a tool call is interrupted it could leave an unanswered tool
53
53
  call that a model will reject on the next turn. The runtime takes
54
54
  care of this by pruning orphaned tool calls and ensuring that the
55
- tool loop always remains valid.
55
+ tool loop always remains valid.
56
56
 
57
57
  ```ruby
58
58
  require "llm"
@@ -141,7 +141,9 @@ call them on your behalf, and they're one of the most powerful features
141
141
  for extending the feature set or abilities of a model.
142
142
 
143
143
  The runtime also ships with a catalog of built-in tools for
144
- filesystem, search, and shell operations.
144
+ filesystem, search, and shell operations, and providers expose
145
+ platform-native tools such as web search and code execution that run
146
+ on the provider's side.
145
147
 
146
148
  ```ruby
147
149
  class ReadFile < LLM::Tool
@@ -276,13 +278,18 @@ for agents.
276
278
 
277
279
  ##### Installation
278
280
 
279
- The console is distributed with llm.rb so you don't have to install
280
- a separate gem but it requires a number of optional dependencies
281
- to be installed separately. The following gems provide the full
282
- experience:
281
+ The console is distributed with llm.rb but it requires a number
282
+ of optional dependencies to be installed separately. The following
283
+ gems provide the full experience:
283
284
 
284
285
  gem install unicode-display_width curses kramdown xchan.rb test-cmd.rb
285
286
 
287
+ For convenience it is also possible to just use the following, it
288
+ is a metagem that depends on llm.rb and all the dependencies it requires
289
+ to run the console:
290
+
291
+ gem install llm-shell
292
+
286
293
  ##### Persistence
287
294
 
288
295
  the `path:` option can be set on an agent for automatic persistence
@@ -363,21 +370,27 @@ require "active_record"
363
370
  require "llm"
364
371
  require "llm/active_record"
365
372
 
366
- class Raven < ActiveRecord::Base
373
+ ##
374
+ # The Robert agent.
375
+ class Robert < ActiveRecord::Base
367
376
  acts_as_agent(format: :jsonb) do |agent|
368
- agent.set name: "raven",
369
- description: "a chatbot for the r.uby.dev website",
370
- instructions: proc { File.read(File.join(__dir__, "raven", "prompt.md")) },
377
+ agent.set name: "robert",
378
+ description: "robert is an agent that has access to the official " \
379
+ "r.uby.dev GitHub repositories. He can access the repositories " \
380
+ "to answer your question(s) about r.uby.dev projects.",
381
+ instructions: proc { File.read(File.join(__dir__, "robert", "prompt.md")) },
371
382
  tools: :tools,
372
- concurrency: :async
373
- end
383
+ concurrency: :async,
374
384
 
375
- def research_issues
376
- talk("research open pull requests on r-uby-dev/llm")
377
- end
385
+ ##
386
+ # The maximum number of tool calls per-turn.
387
+ tool_budget: 25,
378
388
 
379
- def research_codebase
380
- talk("research the codebase on r-uby-dev/llm")
389
+ ##
390
+ # The default tracer that all agents have associated
391
+ # with them. The tracer exports a trace to a couple of
392
+ # SQL tables.
393
+ tracer: proc { Raven::Tracer::SQL.new(llm, agent: self) }
381
394
  end
382
395
 
383
396
  ##
@@ -398,10 +411,6 @@ class Raven < ActiveRecord::Base
398
411
 
399
412
  private
400
413
 
401
- def set_provider
402
- LLM.deepseek
403
- end
404
-
405
414
  def allowlist
406
415
  %w[
407
416
  get_commit
@@ -420,19 +429,18 @@ class Raven < ActiveRecord::Base
420
429
  end
421
430
  end
422
431
 
423
- agent = Raven.create!
432
+ agent = Robert.create!
424
433
 
425
434
  ##
426
435
  # Every call to `talk` automatically persists
427
- # to the database (under the hood research_issues
428
- # calls the talk method)
429
- agent.research_issues
436
+ # to the database.
437
+ agent.talk "what's new on the llm.rb repository?"
430
438
 
431
439
  ##
432
440
  # The conversation was persisted to database. A
433
441
  # fresh instance restores it and continues where
434
442
  # we left off
435
- agent = Raven.find(agent.id).tap(&:research_codebase)
443
+ agent = Robert.find(agent.id).talk "and what about roda-llm?"
436
444
 
437
445
  ##
438
446
  # Start an agent console.
@@ -440,6 +448,100 @@ agent = Raven.find(agent.id).tap(&:research_codebase)
440
448
  # The console does not persist back to the database.
441
449
  agent.console
442
450
  ```
451
+ </details>
452
+ <details>
453
+ <summary> SQL optimizations </summary>
454
+ <br>
455
+
456
+ In a database environment the runtime optimizes for
457
+ the PostgreSQL database and its builtin support for
458
+ the `jsonb` column type. An agent fits in a single
459
+ column, on a single row, and that column carries
460
+ everything it has done: messages, tool calls,
461
+ context usage, and so on. It works well in practice
462
+ and means you can store an agent almost anywhere.
463
+
464
+ For scenarios where performance matters most the runtime
465
+ ships with virtual ActiveRecord classes that never materialize
466
+ in your database but provide a SQL view into the column where
467
+ an agent stores its runtime state. They return
468
+ [`ActiveRecord::Relation`](https://api.rubyonrails.org/classes/ActiveRecord/Relation.html)
469
+ objects, so the filtering happens in the database.
470
+
471
+ ```ruby
472
+ class Agent < ActiveRecord::Base
473
+ acts_as_agent(format: :jsonb) do |agent|
474
+ agent.set name: "activerecord agent"
475
+ end
476
+ end
477
+
478
+ ##
479
+ # Find an instance of your agent
480
+ agent = Agent.find_by(id: 1)
481
+
482
+ ##
483
+ # Returns a relation over the agent's messages.
484
+ # It is scoped to the agent, and it yields one
485
+ # instance of LLM::ActiveRecord::Message per
486
+ # message the agent has produced.
487
+ messages = LLM::ActiveRecord::Message.for(agent:)
488
+
489
+ ##
490
+ # The relation chains like any other
491
+ messages.where(role: "assistant")
492
+ .order(position: :desc)
493
+ .limit(10)
494
+
495
+ ##
496
+ # Count, too
497
+ messages.count
498
+ ```
499
+
500
+ **Schema**
501
+
502
+ Each row carries a message, flattened into columns:
503
+
504
+ | column | contents |
505
+ | --- | --- |
506
+ | `agent_id` | the agent a message belongs to |
507
+ | `id` | the message id |
508
+ | `role` | the message role |
509
+ | `content` | the message content |
510
+ | `tools` | the tool calls a message carries |
511
+ | `position` | the position of a message in the conversation |
512
+ | `data` | the whole message, as the runtime stores it |
513
+
514
+ **Indexes**
515
+
516
+ The queries the view runs are already covered. They expand
517
+ one agent, found by primary key, so they are index scans.
518
+ There is nothing to add for
519
+ `LLM::ActiveRecord::Message.for(agent:)`.
520
+
521
+ The queries you write on top of it are not. Once a question
522
+ is asked of every agent, the column is expanded row by row
523
+ and no index helps the view itself. Index the column for
524
+ those questions instead:
525
+
526
+ ```sql
527
+ CREATE INDEX index_agents_on_data
528
+ ON agents USING gin (data jsonb_path_ops);
529
+
530
+ CREATE INDEX index_agents_on_context_used
531
+ ON agents (((data ->> 'context_used')::int));
532
+ ```
533
+
534
+ The first serves containment (`@>`) and path queries over
535
+ the state as a whole. The second serves a scalar key, and
536
+ the runtime already writes `context_used` and
537
+ `context_window` at the top level, so "sessions over 80%
538
+ full" becomes cheap. Both assume `format: :jsonb`.
539
+
540
+ **However:** an agent's whole conversation lives in one
541
+ value, so every save rewrites it, and a GIN index is
542
+ maintained with it. Prefer an index on a key or two over
543
+ the whole column.
544
+
443
545
  </details>
444
546
 
445
547
  <details><summary>MCP</summary>
@@ -534,10 +636,9 @@ even answer for it. Because it runs before the tool, anything
534
636
  it intercepts never executes. Policy, validation, quotas, and
535
637
  cost ceilings all live here.
536
638
 
537
- [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
538
- enables
539
- [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
540
- by default, so agents get loop protection out of the box. To
639
+ Agents and contexts use
640
+ [`LLM::Guard::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Null.html)
641
+ by default, so a guard only runs when you configure one. To
541
642
  write your own guard, subclass
542
643
  [`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
543
644
  and implement
@@ -737,11 +838,16 @@ is also distributed with llm.rb.
737
838
  ```ruby
738
839
  llm = LLM.openai
739
840
  llm = LLM.anthropic
841
+ llm = LLM.google
740
842
  llm = LLM.deepseek
741
- llm = LLM.alibaba # also: LLM.aliyun
843
+ llm = LLM.deepinfra
844
+ llm = LLM.xai
845
+ llm = LLM.zai
742
846
  llm = LLM.moonshot
743
847
  llm = LLM.openrouter
848
+ llm = LLM.alibaba # also: LLM.aliyun
744
849
  llm = LLM.mistral
850
+ llm = LLM.bedrock
745
851
  ```
746
852
  </details>
747
853
  <details>
@@ -755,10 +861,14 @@ key at all.
755
861
  ```ruby
756
862
  llm = LLM.openai(key: ENV["OPENAI_API_KEY"])
757
863
  llm = LLM.anthropic(key: ENV["ANTHROPIC_API_KEY"])
864
+ llm = LLM.google(key: ENV["GOOGLE_API_KEY"])
758
865
  llm = LLM.deepseek(key: ENV["DEEPSEEK_API_KEY"])
759
- llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
866
+ llm = LLM.deepinfra(key: ENV["DEEPINFRA_API_KEY"])
867
+ llm = LLM.xai(key: ENV["XAI_API_KEY"])
868
+ llm = LLM.zai(key: ENV["ZHIPU_API_KEY"])
760
869
  llm = LLM.moonshot(key: ENV["MOONSHOT_API_KEY"])
761
870
  llm = LLM.openrouter(key: ENV["OPENROUTER_API_KEY"])
871
+ llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
762
872
  llm = LLM.mistral(key: ENV["MISTRAL_API_KEY"])
763
873
  ```
764
874
  </details>
@@ -919,22 +1029,18 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
919
1029
  <details>
920
1030
  <summary>Where can I see llm.rb in action?</summary>
921
1031
  <br>
922
- <p>
923
1032
 
924
- The [r.uby.dev](https://r.uby.dev) website is powered
925
- by llm.rb and its builtin MCP feature. It is connected
926
- to this very GitHub repository. It is designed to help
927
- you learn and troubleshoot llm.rb.
928
- </p>
1033
+ The [r.uby.dev](https://r.uby.dev) website deploys
1034
+ multiple llm.rb agents that guests can interact with
1035
+ and there is even an agent that is connected to this
1036
+ GitHub repository.
929
1037
  </details>
930
1038
  <details>
931
1039
  <summary>What about local LLM support?</summary>
932
1040
  <br>
933
- <p>
1041
+
934
1042
  The following providers can be run used with models that
935
- are running on your own hardware. They're reasonably well
936
- tested but not my main driver:
937
- </p>
1043
+ are running on your own hardware.
938
1044
 
939
1045
  * Ollama
940
1046
  * Llamacpp
@@ -943,29 +1049,27 @@ tested but not my main driver:
943
1049
  <details>
944
1050
  <summary>I have a limited budget. What should I do?</summary>
945
1051
  <br>
946
- <p>
1052
+
947
1053
  There are a few options. The first option is to host
948
1054
  your own model, and use the ollama or llamacpp
949
1055
  providers. This can be difficult though because
950
1056
  a capable model requires hardware that can
951
1057
  match it. If you have the ability to self-host,
952
1058
  this would be my first option.
953
- </p>
954
- <p>
1059
+
955
1060
  The second option is DeepSeek. <br>
956
1061
  The deepseek-v4-flash model costs pennies to use. <br>
957
1062
  And llm.rb has been optimized for deepseek. For example,
958
1063
  DeepSeek does not have image generation capabilities
959
1064
  but on the llm.rb runtime it does (vector graphics only,
960
1065
  though).
961
- </p>
962
- <p>
1066
+
963
1067
  The same is true for structured outputs. DeepSeek does
964
1068
  not support structured outputs in the same way as OpenAI or
965
1069
  Google, but the llm.rb runtime makes it appear as
966
1070
  though it does, through the `json_object` response
967
1071
  type.
968
- </p>
1072
+
969
1073
  If you're on a budget, DeepSeek is hard to beat.
970
1074
  </details>
971
1075
  <details>
@@ -988,23 +1092,35 @@ web</a>.
988
1092
  <summary>Who maintains llm.rb?</summary>
989
1093
  <br>
990
1094
 
991
- The llm.rb project is maintained primarily by one
992
- person. llm.rb has been in active development for more
993
- than three years and over that time multiple other
994
- contributors have contributed to llm.rb as well. New
995
- contributors are always welcome.
996
-
997
- I use the console that is distributed with llm.rb to build
998
- llm.rb itself so there is a healthy feedback loop and
999
- llm.rb has also been battle tested in production
1000
- environments.
1095
+ The llm.rb project was started more than three
1096
+ years ago by
1097
+ [@0x1eef](https://github.com/0x1eef) and
1098
+ [@antaz](https://github.com/0x1eef). The primary
1099
+ maintainer is [@0x1eef](https://github.com/0x1eef).
1100
+ Over those three years multiple other contributors have
1101
+ contributed to llm.rb as well, and new contributors are
1102
+ always welcome.
1103
+ </details>
1001
1104
 
1002
- I have also also written llm.rb agents within the
1003
- repository that help me maintain the documentation,
1004
- and backport changes to the mruby-llm runtime as well.
1105
+ <details>
1106
+ <summary>How well tested is llm.rb?</summary>
1107
+ <br>
1005
1108
 
1006
- I am constantly focused on improving llm.rb by using
1007
- it as my primary driver for development.
1109
+ It is battle tested daily.
1110
+
1111
+ The console that is distributed with llm.rb is used
1112
+ to build llm.rb so there is a healthy, active feedback
1113
+ loop. It also powers the [r.uby.dev](https://r.uby.dev)
1114
+ website where multiple llm.rb agents are deployed with
1115
+ the help of [roda-llm](https://github.com/r-uby-dev/roda-llm).
1116
+ I'm also aware of at least one production Rails deployment
1117
+ at a large-ish company.
1118
+
1119
+ And this git repository includes llm.rb agents that help me
1120
+ maintain the documentation and perform other repository
1121
+ maintainence. The feedback loop is constant. Outside of that
1122
+ there is a large test suite that covers live requests (recorded
1123
+ by VCR) and database interactions.
1008
1124
  </details>
1009
1125
 
1010
1126
  ## See also
@@ -1019,7 +1135,7 @@ be hosted within a Rails application or other Rack-based applications.
1019
1135
 
1020
1136
  The [docs/](docs/) directory contains the full documentation and
1021
1137
  the chatbot can find the answers to your questions there. Or you
1022
- can read them yourself.
1138
+ can read them yourself. :)
1023
1139
 
1024
1140
  ## License
1025
1141
 
data/bin/llm.rb CHANGED
@@ -193,6 +193,12 @@ def main(argv)
193
193
  end
194
194
  end
195
195
 
196
+ ##
197
+ # Let's use `AGENTS.md` as the system prompt
198
+ agent_opts = {}
199
+ sysprompt = File.join(Dir.getwd, "AGENTS.md")
200
+ File.file?(sysprompt) ? agent_opts.merge!(instructions: File.read(sysprompt)) : {}
201
+
196
202
  ##
197
203
  # No provider has been given.
198
204
  # Try to infer one.
@@ -255,8 +261,9 @@ def main(argv)
255
261
  ##
256
262
  # Let's go!
257
263
  concurrency ||= :sequential
258
- path = temp ? nil : data[Dir.getwd]
259
- agent = LLM::Agent.new(llm, model:, path:, concurrency:, tools: LLM::Tool.subclasses)
264
+ path = temp ? nil : data[Dir.getwd]
265
+ params = agent_opts.merge(model:, path:, concurrency:, tools: LLM::Tool.subclasses)
266
+ agent = LLM::Agent.new(llm, params)
260
267
  agent.console
261
268
  rescue Interrupt
262
269
  warn "llm.rb: Bye!"
data/data/alibaba.json CHANGED
@@ -1193,6 +1193,51 @@
1193
1193
  "output_audio": 15.11
1194
1194
  }
1195
1195
  },
1196
+ "kimi-k3": {
1197
+ "id": "kimi-k3",
1198
+ "name": "Kimi K3",
1199
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
1200
+ "family": "kimi-k3",
1201
+ "attachment": true,
1202
+ "reasoning": true,
1203
+ "reasoning_options": [
1204
+ {
1205
+ "type": "effort",
1206
+ "values": [
1207
+ "low",
1208
+ "high",
1209
+ "max"
1210
+ ]
1211
+ }
1212
+ ],
1213
+ "tool_call": true,
1214
+ "interleaved": {
1215
+ "field": "reasoning_content"
1216
+ },
1217
+ "structured_output": true,
1218
+ "temperature": false,
1219
+ "release_date": "2026-07-16",
1220
+ "last_updated": "2026-07-16",
1221
+ "modalities": {
1222
+ "input": [
1223
+ "text",
1224
+ "image"
1225
+ ],
1226
+ "output": [
1227
+ "text"
1228
+ ]
1229
+ },
1230
+ "open_weights": true,
1231
+ "limit": {
1232
+ "context": 1048576,
1233
+ "output": 1048576
1234
+ },
1235
+ "cost": {
1236
+ "input": 3,
1237
+ "output": 15,
1238
+ "cache_read": 0.3
1239
+ }
1240
+ },
1196
1241
  "qwen2-5-vl-72b-instruct": {
1197
1242
  "id": "qwen2-5-vl-72b-instruct",
1198
1243
  "name": "Qwen2.5-VL 72B Instruct",