llm.rb 15.0.3 → 15.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +433 -3
- data/README.md +188 -71
- data/bin/llm.rb +50 -7
- data/data/alibaba.json +912 -823
- data/data/anthropic.json +234 -187
- data/data/bedrock.json +3702 -2058
- data/data/deepinfra.json +1288 -951
- data/data/deepseek.json +87 -53
- data/data/google.json +670 -670
- data/data/mistral.json +501 -460
- data/data/moonshot.json +43 -248
- data/data/openai.json +1008 -914
- data/data/openrouter.json +14417 -0
- data/data/xai.json +213 -201
- data/data/zai.json +242 -149
- data/docs/deepdive/advanced/compaction.md +5 -5
- data/docs/deepdive/advanced/context.md +8 -6
- data/docs/deepdive/advanced/guard.md +2 -2
- data/docs/deepdive/features/builtin_tools.md +93 -22
- data/docs/deepdive/features/{repl.md → console.md} +28 -28
- data/docs/deepdive/features/database.md +3 -3
- data/docs/deepdive/fundamentals/agents.md +13 -12
- data/docs/deepdive/fundamentals/providers.md +91 -6
- data/docs/deepdive/fundamentals/skills.md +14 -6
- data/docs/deepdive/fundamentals/stream.md +4 -4
- data/docs/deepdive/fundamentals/tools.md +63 -31
- data/docs/deepdive/reference/cost.md +2 -2
- data/docs/deepdive/reference/model_registry.md +2 -2
- data/docs/deepdive/reference/tracer.md +15 -13
- data/docs/deepdive.md +2 -2
- data/lib/llm/active_record/acts_as_agent.rb +9 -5
- data/lib/llm/agent.rb +40 -15
- data/lib/llm/{repl → console}/bar.rb +3 -3
- data/lib/llm/{repl → console}/buffer.rb +24 -9
- data/lib/llm/{repl → console}/color.rb +2 -2
- data/lib/llm/{repl → console}/command.rb +12 -12
- data/lib/llm/{repl → console}/commands/exit.rb +4 -4
- data/lib/llm/{repl → console}/commands/help.rb +1 -1
- data/lib/llm/{repl/commands/compact.rb → console/commands/keep.rb} +11 -9
- data/lib/llm/{repl → console}/commands/model.rb +2 -2
- data/lib/llm/{repl → console}/input/cache.rb +2 -2
- data/lib/llm/{repl → console}/input/char.rb +2 -2
- data/lib/llm/{repl → console}/input/row.rb +1 -1
- data/lib/llm/{repl → console}/input.rb +18 -10
- data/lib/llm/console/markdown/parser.rb +78 -0
- data/lib/llm/{repl → console}/markdown/table.rb +8 -5
- data/lib/llm/{repl → console}/markdown.rb +13 -30
- data/lib/llm/console/node.rb +69 -0
- data/lib/llm/{repl → console}/status.rb +11 -11
- data/lib/llm/{repl → console}/stream.rb +36 -9
- data/lib/llm/{repl → console}/walker.rb +1 -1
- data/lib/llm/{repl → console}/window.rb +17 -17
- data/lib/llm/{repl.rb → console.rb} +39 -19
- data/lib/llm/context/deserializer.rb +2 -1
- data/lib/llm/context.rb +29 -13
- data/lib/llm/cost.rb +13 -0
- data/lib/llm/function/async/reactor.rb +20 -1
- data/lib/llm/function/fork/task.rb +14 -10
- data/lib/llm/function.rb +1 -1
- data/lib/llm/json_adapter.rb +40 -28
- data/lib/llm/message.rb +7 -0
- data/lib/llm/provider.rb +31 -10
- data/lib/llm/providers/alibaba.rb +1 -1
- data/lib/llm/providers/anthropic.rb +1 -1
- data/lib/llm/providers/bedrock/models.rb +2 -2
- data/lib/llm/providers/bedrock.rb +1 -1
- data/lib/llm/providers/deepseek.rb +1 -1
- data/lib/llm/providers/google.rb +1 -1
- data/lib/llm/providers/ollama.rb +1 -1
- data/lib/llm/providers/openai/responses.rb +2 -1
- data/lib/llm/providers/openai.rb +2 -1
- data/lib/llm/providers/openrouter.rb +87 -0
- data/lib/llm/schema/leaf.rb +34 -2
- data/lib/llm/schema.rb +4 -2
- data/lib/llm/sequel/agent.rb +9 -5
- data/lib/llm/skill.rb +7 -1
- data/lib/llm/stream.rb +8 -3
- data/lib/llm/tool/param.rb +5 -1
- data/lib/llm/tool.rb +5 -0
- data/lib/llm/tools/bundle.rb +53 -0
- data/lib/llm/tools/edit-file.rb +7 -2
- data/lib/llm/tools/exec.rb +78 -0
- data/lib/llm/tools/git.rb +27 -26
- data/lib/llm/tools/mkdir.rb +12 -19
- data/lib/llm/tools/read_file.rb +69 -9
- data/lib/llm/tools/rg.rb +20 -24
- data/lib/llm/tools/ruby.rb +17 -25
- data/lib/llm/tools/utils.rb +75 -2
- data/lib/llm/tools/write_file.rb +4 -1
- data/lib/llm/tracer/logger.rb +2 -2
- data/lib/llm/tracer/pretty_logger.rb +4 -4
- data/lib/llm/tracer/telemetry.rb +2 -2
- data/lib/llm/tracer.rb +33 -0
- data/lib/llm/transport/curb.rb +5 -3
- data/lib/llm/transport/http.rb +5 -2
- data/lib/llm/transport/persistent_http.rb +6 -4
- data/lib/llm/transport/utils.rb +8 -6
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +18 -12
- data/llm.gemspec +8 -8
- metadata +80 -37
- data/lib/llm/repl/node.rb +0 -44
- data/lib/llm/tools/shell.rb +0 -55
data/README.md
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
<p align="center">
|
|
2
2
|
<a href="https://r.uby.dev">
|
|
3
3
|
<img
|
|
4
|
-
src="
|
|
4
|
+
src="rubydev.svg"
|
|
5
5
|
width="400"
|
|
6
6
|
height="200"
|
|
7
7
|
border="0"
|
|
@@ -10,18 +10,16 @@
|
|
|
10
10
|
</a>
|
|
11
11
|
</p>
|
|
12
12
|
|
|
13
|
-
>
|
|
13
|
+
> [r.uby.dev](https://r.uby.dev/llm) project.
|
|
14
14
|
|
|
15
15
|
Welcome to the canonical llm.rb repository.
|
|
16
16
|
|
|
17
17
|
llm.rb is an advanced runtime for building agentic AI applications
|
|
18
18
|
on CRuby. It has zero runtime dependencies by default, supports
|
|
19
19
|
concurrent and parallel tool execution and has a single coherent API
|
|
20
|
-
that spans
|
|
21
|
-
REPL, builtin MCP/A2A support and the database integrations all build
|
|
22
|
-
on the same three concepts: providers, contexts, and agents.
|
|
20
|
+
that spans 14+ providers.
|
|
23
21
|
|
|
24
|
-
The
|
|
22
|
+
The easiest way to learn about llm.rb is to ask [the r.uby.dev chatbot](https://r.uby.dev)
|
|
25
23
|
a question. It is connected to the llm.rb GitHub repository, backed by
|
|
26
24
|
ActiveRecord and uses the builtin MCP feature to connect to GitHub. All
|
|
27
25
|
answers are grounded in the llm.rb source code.
|
|
@@ -108,8 +106,8 @@ class MyStream < LLM::Stream
|
|
|
108
106
|
def on_skill_return(agent, skill, result)
|
|
109
107
|
end
|
|
110
108
|
|
|
111
|
-
# A request was rate limited and will be retried.
|
|
112
|
-
def
|
|
109
|
+
# A request was rate limited or timed out and will be retried.
|
|
110
|
+
def on_retry(error, attempt)
|
|
113
111
|
end
|
|
114
112
|
end
|
|
115
113
|
|
|
@@ -204,6 +202,12 @@ with the `:fork` and `:ractor` strategies. The
|
|
|
204
202
|
The `:fork` strategy also provides a separate process that offers
|
|
205
203
|
isolation from its parent.
|
|
206
204
|
|
|
205
|
+
A couple of concurrency strategies require optional, opt-in dependencies.
|
|
206
|
+
The `async` strategy requires the [async](https://github.com/socketry/async)
|
|
207
|
+
gem and the `fork` strategy requires the [xchan.rb](https://github.com/r-uby-dev/xchan.rb)
|
|
208
|
+
gem. The `fiber` strategy requires a scheduler (`Fiber.scheduler`) but by
|
|
209
|
+
default Ruby does not provide one.
|
|
210
|
+
|
|
207
211
|
```ruby
|
|
208
212
|
require "llm"
|
|
209
213
|
require "llm/tools"
|
|
@@ -241,37 +245,35 @@ end
|
|
|
241
245
|
```
|
|
242
246
|
</details>
|
|
243
247
|
<details>
|
|
244
|
-
<summary>Console (<code>binding.
|
|
248
|
+
<summary>Console (<code>binding.irb</code> for agents)</summary>
|
|
245
249
|
<br>
|
|
246
250
|
|
|
247
|
-
The [LLM::Agent#
|
|
248
|
-
method drops you into a highly capable
|
|
251
|
+
The [LLM::Agent#console](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#console-instance_method)
|
|
252
|
+
method drops you into a highly capable interactive console
|
|
249
253
|
that is built on top of curses. It can help you debug agents,
|
|
250
254
|
test your tools, connect to MCP servers, and even A2A agents.
|
|
251
|
-
The
|
|
255
|
+
The console stands out because it connects to the surrounding
|
|
252
256
|
runtime and it can be extended by your code. Think of it as
|
|
253
|
-
`binding.
|
|
257
|
+
`binding.irb` but for agents.
|
|
254
258
|
|
|
255
259
|
##### Demo
|
|
256
260
|
|
|
257
|
-
[
|
|
258
|
-
|
|
259
|
-

|
|
261
|
+

|
|
260
262
|
|
|
261
263
|
|
|
262
264
|
##### Installation
|
|
263
265
|
|
|
264
|
-
The
|
|
266
|
+
The console is distributed with llm.rb so you don't have to install
|
|
265
267
|
a separate gem but it requires a number of optional dependencies
|
|
266
268
|
to be installed separately. The following gems provide the full
|
|
267
269
|
experience:
|
|
268
270
|
|
|
269
|
-
gem install curses kramdown xchan.rb test-cmd.rb
|
|
271
|
+
gem install unicode-display_width curses kramdown xchan.rb test-cmd.rb
|
|
270
272
|
|
|
271
273
|
##### Persistence
|
|
272
274
|
|
|
273
|
-
|
|
274
|
-
across
|
|
275
|
+
the `path:` option can be set on an agent for automatic persistence
|
|
276
|
+
across console sessions. The `tools:` option attaches extra tools
|
|
275
277
|
for the duration of the session. Recall previous turns with Ctrl+P and
|
|
276
278
|
Ctrl+N.
|
|
277
279
|
|
|
@@ -281,21 +283,26 @@ require "llm/tools"
|
|
|
281
283
|
|
|
282
284
|
llm = LLM.deepseek(key: ENV["KEY"])
|
|
283
285
|
agent = LLM::Agent.new(llm, name: "my-agent", path: "agent.json")
|
|
284
|
-
agent.
|
|
286
|
+
agent.console(tools: LLM::Tool.subclasses)
|
|
285
287
|
```
|
|
286
288
|
|
|
287
289
|
##### CLI
|
|
288
290
|
|
|
289
291
|
The `llm.rb` executable is available on your PATH after installation.
|
|
290
|
-
It starts a
|
|
292
|
+
It starts a console session from any directory. The CLI auto-detects your
|
|
291
293
|
provider from standard environment variables (`DEEPSEEK_API_KEY`,
|
|
292
294
|
`OPENAI_API_KEY`, `ANTHROPIC_API_KEY`, etc.). Persistent sessions are
|
|
293
295
|
stored under `~/.llm.rb/` and restored automatically on your next visit.
|
|
294
296
|
|
|
295
297
|
```bash
|
|
296
|
-
llm.rb # auto-detect from $
|
|
298
|
+
llm.rb # auto-detect from $PROVIDER_API_KEY
|
|
297
299
|
llm.rb -p openai # use OpenAI explicitly
|
|
300
|
+
llm.rb -m gpt-5.6 # use a model other than the provider default
|
|
301
|
+
llm.rb -c thread # run tool calls on a separate thread
|
|
302
|
+
llm.rb -n curb # use libcurl as the HTTP transport
|
|
303
|
+
llm.rb -x 900 # read timeout of 15 minutes
|
|
298
304
|
llm.rb -t # temporary session, no persistence
|
|
305
|
+
llm.rb -v # print the version
|
|
299
306
|
```
|
|
300
307
|
</details>
|
|
301
308
|
<details>
|
|
@@ -335,53 +342,90 @@ for both Rack-based / Rails-based applications. On databases
|
|
|
335
342
|
where it is supported, such as PostgreSQL, the column can be optimized by using
|
|
336
343
|
the `jsonb` type.
|
|
337
344
|
|
|
345
|
+
The following example is based on the agent used to power the
|
|
346
|
+
[r.uby.dev chatbot](https://r.uby.dev).
|
|
347
|
+
|
|
338
348
|
```ruby
|
|
339
349
|
require "active_record"
|
|
340
350
|
require "llm"
|
|
341
351
|
require "llm/active_record"
|
|
342
352
|
|
|
343
|
-
class
|
|
344
|
-
acts_as_agent do |agent|
|
|
345
|
-
agent.set name: "
|
|
346
|
-
|
|
347
|
-
|
|
353
|
+
class Raven < ActiveRecord::Base
|
|
354
|
+
acts_as_agent(format: :jsonb) do |agent|
|
|
355
|
+
agent.set name: "raven",
|
|
356
|
+
description: "a chatbot for the r.uby.dev website",
|
|
357
|
+
instructions: proc { File.read(File.join(__dir__, "raven", "prompt.md")) },
|
|
358
|
+
tools: :tools,
|
|
359
|
+
concurrency: :async
|
|
348
360
|
end
|
|
349
361
|
|
|
350
|
-
def
|
|
351
|
-
talk("
|
|
362
|
+
def research_issues
|
|
363
|
+
talk("research open pull requests on r-uby-dev/llm")
|
|
352
364
|
end
|
|
353
365
|
|
|
354
|
-
def
|
|
355
|
-
talk("
|
|
366
|
+
def research_codebase
|
|
367
|
+
talk("research the codebase on r-uby-dev/llm")
|
|
356
368
|
end
|
|
357
369
|
|
|
358
|
-
|
|
370
|
+
##
|
|
371
|
+
# @return [LLM::MCP]
|
|
372
|
+
def github
|
|
373
|
+
@github ||= LLM::MCP.http(
|
|
374
|
+
url: "https://api.githubcopilot.com/mcp/",
|
|
375
|
+
headers: {"Authorization" => "Bearer #{ENV['GITHUB_RUBYDEV_PAT']}"},
|
|
376
|
+
transport: :net_http_persistent
|
|
377
|
+
)
|
|
378
|
+
end
|
|
359
379
|
|
|
360
380
|
##
|
|
361
|
-
#
|
|
362
|
-
|
|
381
|
+
# @return [Array<LLM::Tool>]
|
|
382
|
+
def tools
|
|
383
|
+
github.tools.select { allowlist.include?(_1.name.to_s) }
|
|
384
|
+
end
|
|
385
|
+
|
|
386
|
+
private
|
|
387
|
+
|
|
363
388
|
def set_provider
|
|
364
|
-
LLM.deepseek
|
|
389
|
+
LLM.deepseek
|
|
365
390
|
end
|
|
366
391
|
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
392
|
+
def allowlist
|
|
393
|
+
%w[
|
|
394
|
+
get_commit
|
|
395
|
+
get_file_contents
|
|
396
|
+
list_branches
|
|
397
|
+
list_commits
|
|
398
|
+
search_code
|
|
399
|
+
search_commits
|
|
400
|
+
search_repositories
|
|
401
|
+
search_issues
|
|
402
|
+
pull_request_read
|
|
403
|
+
list_pull_requests
|
|
404
|
+
list_issues
|
|
405
|
+
issue_read
|
|
406
|
+
].freeze
|
|
372
407
|
end
|
|
373
408
|
end
|
|
374
409
|
|
|
375
|
-
|
|
376
|
-
email.draft_reply!
|
|
410
|
+
agent = Raven.create!
|
|
377
411
|
|
|
378
412
|
##
|
|
379
|
-
#
|
|
380
|
-
#
|
|
381
|
-
#
|
|
382
|
-
|
|
383
|
-
|
|
384
|
-
|
|
413
|
+
# Every call to `talk` automatically persists
|
|
414
|
+
# to the database (under the hood research_issues
|
|
415
|
+
# calls the talk method)
|
|
416
|
+
agent.research_issues
|
|
417
|
+
|
|
418
|
+
##
|
|
419
|
+
# The conversation was persisted to database. A
|
|
420
|
+
# fresh instance restores it and continues where
|
|
421
|
+
# we left off
|
|
422
|
+
agent = Raven.find(agent.id).tap(&:research_codebase)
|
|
423
|
+
|
|
424
|
+
##
|
|
425
|
+
# Start an agent console.
|
|
426
|
+
# Query agent's state, debug, etc.
|
|
427
|
+
# The console does not persist back to the database.
|
|
428
|
+
agent.console
|
|
385
429
|
```
|
|
386
430
|
</details>
|
|
387
431
|
|
|
@@ -491,15 +535,15 @@ the call, or `nil` to let it run:
|
|
|
491
535
|
```ruby
|
|
492
536
|
class PolicyGuard < LLM::Guard
|
|
493
537
|
def call(function:)
|
|
494
|
-
if function.name == "
|
|
538
|
+
if function.name == "exec"
|
|
495
539
|
function.return(error: true, type: "policy_error",
|
|
496
|
-
message: "
|
|
540
|
+
message: "exec is disabled")
|
|
497
541
|
end
|
|
498
542
|
end
|
|
499
543
|
end
|
|
500
544
|
|
|
501
545
|
llm = LLM.deepseek(key: ENV["KEY"])
|
|
502
|
-
agent = LLM::Agent.new(llm, tools: [
|
|
546
|
+
agent = LLM::Agent.new(llm, tools: [LLM::Tool::Exec, ReadFile], guard: PolicyGuard)
|
|
503
547
|
```
|
|
504
548
|
</details>
|
|
505
549
|
|
|
@@ -566,7 +610,8 @@ agent.talk "Hello"
|
|
|
566
610
|
|
|
567
611
|
Rate-limited requests are retried automatically by default. Agents
|
|
568
612
|
retry a 429 up to five times with a growing backoff before giving
|
|
569
|
-
up, so most request failures resolve on their own.
|
|
613
|
+
up, so most request failures resolve on their own. Connection and
|
|
614
|
+
read timeouts are retried the same way. Set `retry_budget`
|
|
570
615
|
to change the number of retries, or `retry_budget: 0` to disable
|
|
571
616
|
them.
|
|
572
617
|
|
|
@@ -585,21 +630,28 @@ agent.talk "Hello"
|
|
|
585
630
|
<summary>Observability</summary>
|
|
586
631
|
<br>
|
|
587
632
|
|
|
588
|
-
|
|
589
|
-
requests, tool calls, and other
|
|
590
|
-
|
|
591
|
-
observability backend. All built-in tracers
|
|
592
|
-
so switching between them means changing a
|
|
633
|
+
It is possible to trace what an agent is doing by attaching a
|
|
634
|
+
tracer. A tracer can hook into requests, tool calls, and other
|
|
635
|
+
runtime events to debug an agent, provide insights, monitor latency,
|
|
636
|
+
or export spans to an observability backend. All built-in tracers
|
|
637
|
+
share one interface, so switching between them means changing a
|
|
638
|
+
factory method:
|
|
593
639
|
|
|
594
|
-
* [`LLM::Tracer
|
|
595
|
-
* [`LLM::Tracer
|
|
640
|
+
* [`LLM::Tracer.pretty_logger`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#pretty_logger-class_method): human-readable single-line logs to stderr, ideal during development.
|
|
641
|
+
* [`LLM::Tracer.telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#telemetry-class_method):
|
|
596
642
|
exports spans via OTLP for OpenTelemetry in production.
|
|
597
|
-
* [`LLM::Tracer
|
|
643
|
+
* [`LLM::Tracer.logger`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html#logger-class_method):
|
|
598
644
|
structured JSON to stdout or a file.
|
|
599
645
|
|
|
646
|
+
It is also possible to create your own tracer by creating a subclass
|
|
647
|
+
of [`LLM::Tracer`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html)
|
|
648
|
+
that implements a number of callbacks that cover an agent's lifecycle.
|
|
649
|
+
The tracer feature provides visibility into what the runtime is doing,
|
|
650
|
+
and the tracer API lets other code hook into that feature.
|
|
651
|
+
|
|
600
652
|
```ruby
|
|
601
653
|
llm = LLM.deepseek(key: ENV["KEY"])
|
|
602
|
-
agent = LLM::Agent.new(llm, tracer: LLM::Tracer
|
|
654
|
+
agent = LLM::Agent.new(llm, tracer: LLM::Tracer.pretty_logger(llm))
|
|
603
655
|
agent.talk "Hello"
|
|
604
656
|
```
|
|
605
657
|
</details>
|
|
@@ -624,7 +676,7 @@ class Agent < LLM::Agent
|
|
|
624
676
|
set name: "sysadmin",
|
|
625
677
|
description: "system administration agent",
|
|
626
678
|
model: "deepseek-v4-pro",
|
|
627
|
-
tools: [LLM::Tool::
|
|
679
|
+
tools: [LLM::Tool::Exec]
|
|
628
680
|
end
|
|
629
681
|
|
|
630
682
|
llm = LLM.deepseek(key: ENV["KEY"])
|
|
@@ -653,6 +705,7 @@ change.
|
|
|
653
705
|
* **xAI** (`LLM.xai`)
|
|
654
706
|
* **Z.ai** (`LLM.zai`)
|
|
655
707
|
* **Moonshot (Kimi)** (`LLM.moonshot`)
|
|
708
|
+
* **OpenRouter** (`LLM.openrouter`)
|
|
656
709
|
* **Alibaba (Qwen3)** (`LLM.alibaba`, also `LLM.aliyun`)
|
|
657
710
|
* **Mistral** (`LLM.mistral`)
|
|
658
711
|
* **AWS Bedrock** (`LLM.bedrock`)
|
|
@@ -674,6 +727,7 @@ llm = LLM.anthropic
|
|
|
674
727
|
llm = LLM.deepseek
|
|
675
728
|
llm = LLM.alibaba # also: LLM.aliyun
|
|
676
729
|
llm = LLM.moonshot
|
|
730
|
+
llm = LLM.openrouter
|
|
677
731
|
llm = LLM.mistral
|
|
678
732
|
```
|
|
679
733
|
</details>
|
|
@@ -691,6 +745,7 @@ llm = LLM.anthropic(key: ENV["ANTHROPIC_API_KEY"])
|
|
|
691
745
|
llm = LLM.deepseek(key: ENV["DEEPSEEK_API_KEY"])
|
|
692
746
|
llm = LLM.alibaba(key: ENV["DASHSCOPE_API_KEY"]) # also: LLM.aliyun
|
|
693
747
|
llm = LLM.moonshot(key: ENV["MOONSHOT_API_KEY"])
|
|
748
|
+
llm = LLM.openrouter(key: ENV["OPENROUTER_API_KEY"])
|
|
694
749
|
llm = LLM.mistral(key: ENV["MISTRAL_API_KEY"])
|
|
695
750
|
```
|
|
696
751
|
</details>
|
|
@@ -709,7 +764,7 @@ require "llm"
|
|
|
709
764
|
|
|
710
765
|
llm = LLM.openai
|
|
711
766
|
registry = llm.registry # => LLM::Provider#registry
|
|
712
|
-
cheapest = registry.models.sort.first # => LLM::Model
|
|
767
|
+
cheapest = registry.models.sort.first # => LLM::Registry::Model
|
|
713
768
|
cheapest.id # => "text-embedding-3-small"
|
|
714
769
|
cheapest.context_window # => 8191
|
|
715
770
|
cheapest.structured_output? # => false
|
|
@@ -734,6 +789,51 @@ llm = LLM.deepseek(
|
|
|
734
789
|
```
|
|
735
790
|
</details>
|
|
736
791
|
|
|
792
|
+
<details>
|
|
793
|
+
<summary>Timeouts</summary>
|
|
794
|
+
<br>
|
|
795
|
+
|
|
796
|
+
Providers accept two timeouts:
|
|
797
|
+
|
|
798
|
+
* `connect_timeout` - opening the connection. Defaults to 5 seconds.
|
|
799
|
+
* `read_timeout` - waiting for a response on an idle connection.
|
|
800
|
+
Defaults to 600 seconds (10 minutes).
|
|
801
|
+
|
|
802
|
+
The longer read timeout leaves room for slow reasoning models and
|
|
803
|
+
local models. The legacy `timeout:` option remains as a shorthand for
|
|
804
|
+
`read_timeout`. Timeouts are retriable:
|
|
805
|
+
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
|
|
806
|
+
retries a timed out request up to its
|
|
807
|
+
[`retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#retry_budget-class_method)
|
|
808
|
+
(five by default), so a dropped connection or a slow first token
|
|
809
|
+
is often something we can recover from.
|
|
810
|
+
|
|
811
|
+
```ruby
|
|
812
|
+
llm = LLM.deepseek(
|
|
813
|
+
connect_timeout: 5, # opening the connection
|
|
814
|
+
read_timeout: 600 # waiting for the next bytes
|
|
815
|
+
)
|
|
816
|
+
```
|
|
817
|
+
</details>
|
|
818
|
+
|
|
819
|
+
<details>
|
|
820
|
+
<summary>Headers</summary>
|
|
821
|
+
<br>
|
|
822
|
+
|
|
823
|
+
Providers can accept a custom set of headers with
|
|
824
|
+
the [`LLM::Provider#with`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#with-instance_method) method.
|
|
825
|
+
For example, you could set a custom User-Agent header,
|
|
826
|
+
or provide headers that carry special meaning to
|
|
827
|
+
certain providers (eg OpenAI, OpenRouter).
|
|
828
|
+
|
|
829
|
+
```ruby
|
|
830
|
+
llm = LLM.openrouter
|
|
831
|
+
llm = llm.with("HTTP-Referer" => "https://example.com")
|
|
832
|
+
llm = llm.with("X-OpenRouter-Title" => "Example App")
|
|
833
|
+
```
|
|
834
|
+
|
|
835
|
+
</details>
|
|
836
|
+
|
|
737
837
|
### RAG
|
|
738
838
|
|
|
739
839
|
Most providers offer an embedding model that can be
|
|
@@ -803,6 +903,17 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
|
|
|
803
903
|
|
|
804
904
|
## FAQ
|
|
805
905
|
|
|
906
|
+
<details>
|
|
907
|
+
<summary>Where can I see llm.rb in action?</summary>
|
|
908
|
+
<br>
|
|
909
|
+
<p>
|
|
910
|
+
|
|
911
|
+
The [r.uby.dev](https://r.uby.dev) website is powered
|
|
912
|
+
by llm.rb and its builtin MCP feature. It is connected
|
|
913
|
+
to this very GitHub repository. It is designed to help
|
|
914
|
+
you learn and troubleshoot llm.rb.
|
|
915
|
+
</p>
|
|
916
|
+
</details>
|
|
806
917
|
<details>
|
|
807
918
|
<summary>What about local LLM support?</summary>
|
|
808
919
|
<br>
|
|
@@ -870,7 +981,7 @@ than three years and over that time multiple other
|
|
|
870
981
|
contributors have contributed to llm.rb as well. New
|
|
871
982
|
contributors are always welcome.
|
|
872
983
|
|
|
873
|
-
I use the
|
|
984
|
+
I use the console that is distributed with llm.rb to build
|
|
874
985
|
llm.rb itself so there is a healthy feedback loop and
|
|
875
986
|
llm.rb has also been battle tested in production
|
|
876
987
|
environments.
|
|
@@ -883,13 +994,19 @@ I am constantly focused on improving llm.rb by using
|
|
|
883
994
|
it as my primary driver for development.
|
|
884
995
|
</details>
|
|
885
996
|
|
|
886
|
-
##
|
|
997
|
+
## See also
|
|
998
|
+
|
|
999
|
+
The [roda-llm](https://github.com/r-uby-dev/roda-llm#readme) project
|
|
1000
|
+
is how I deploy multiple ActiveRecord-backed llm.rb agents over HTTP.
|
|
1001
|
+
Each agent has an identical interface at a unique path that provide
|
|
1002
|
+
CRUD operations and stream support (via SSE - Server Side Events).
|
|
1003
|
+
It lets you focus on implementing agents rather than the glue that
|
|
1004
|
+
brings them together. It is implemented as a Roda plugin that could
|
|
1005
|
+
be hosted within a Rails application or other Rack-based applications.
|
|
887
1006
|
|
|
888
|
-
The [
|
|
889
|
-
|
|
890
|
-
|
|
891
|
-
directory contains the full documentation and the chatbot
|
|
892
|
-
can find the answers to your questions there.
|
|
1007
|
+
The [docs/](docs/) directory contains the full documentation and
|
|
1008
|
+
the chatbot can find the answers to your questions there. Or you
|
|
1009
|
+
can read them yourself.
|
|
893
1010
|
|
|
894
1011
|
## License
|
|
895
1012
|
|
data/bin/llm.rb
CHANGED
|
@@ -42,6 +42,10 @@ def wrap(text, width)
|
|
|
42
42
|
end
|
|
43
43
|
end
|
|
44
44
|
|
|
45
|
+
def version
|
|
46
|
+
warn "llm.rb v#{LLM::VERSION}"
|
|
47
|
+
end
|
|
48
|
+
|
|
45
49
|
def help
|
|
46
50
|
prog = File.basename($PROGRAM_NAME)
|
|
47
51
|
warn ""
|
|
@@ -49,16 +53,21 @@ def help
|
|
|
49
53
|
warn ""
|
|
50
54
|
warn "Options:"
|
|
51
55
|
warn " -p PROVIDER Choose a provider"
|
|
56
|
+
warn " -m MODEL Choose a model"
|
|
52
57
|
warn " -c STRATEGY Concurrency strategy for tool calls (eg thread, async, fork)"
|
|
53
58
|
warn " -n TRANSPORT HTTP transports - net-http (default), net-http-persistent, and curb"
|
|
59
|
+
warn " -x TIMEOUT The default read timeout (in seconds)"
|
|
54
60
|
warn " -t Temporary session that doesn't persist to disk"
|
|
61
|
+
warn " -v Print version information"
|
|
55
62
|
warn " -h Show this help"
|
|
56
63
|
warn ""
|
|
57
64
|
warn "Examples:"
|
|
58
65
|
warn " #{prog} # auto-detect provider from $PROVIDER_API_KEY"
|
|
59
66
|
warn " #{prog} -p openai # use OpenAI"
|
|
67
|
+
warn " #{prog} -m gpt-5.6 # use a model other than the provider default"
|
|
60
68
|
warn " #{prog} -n curb # use libcurl"
|
|
61
69
|
warn " #{prog} -c thread # run tool calls on a separate thread"
|
|
70
|
+
warn " #{prog} -x 900 # read timeout of 15mins"
|
|
62
71
|
warn " #{prog} -h # this help"
|
|
63
72
|
warn ""
|
|
64
73
|
end
|
|
@@ -70,12 +79,12 @@ def loaderror(ex)
|
|
|
70
79
|
warn ""
|
|
71
80
|
warn_title "Missing dependency: #{gem}"
|
|
72
81
|
warn ""
|
|
73
|
-
warn " The
|
|
82
|
+
warn " The console needs this gem, but it's not installed."
|
|
74
83
|
warn ""
|
|
75
84
|
wrapped "Fix: gem install #{gem}", " "
|
|
76
85
|
wrapped "Or: bundle add #{gem}", " "
|
|
77
86
|
warn ""
|
|
78
|
-
warn " Tip: If you don't need the
|
|
87
|
+
warn " Tip: If you don't need the console, you can use the"
|
|
79
88
|
wrapped "library directly with: require \"llm\"", " "
|
|
80
89
|
warn ""
|
|
81
90
|
warn " ───────────────────────────────────────────────────────"
|
|
@@ -94,6 +103,11 @@ def fatal(ex)
|
|
|
94
103
|
wrapped detail, " "
|
|
95
104
|
end
|
|
96
105
|
warn ""
|
|
106
|
+
warn " Backtrace:"
|
|
107
|
+
ex.backtrace.drop(1).first(3).each do |line|
|
|
108
|
+
wrapped line, " "
|
|
109
|
+
end
|
|
110
|
+
warn ""
|
|
97
111
|
warn " This is an unexpected error. If it keeps happening,"
|
|
98
112
|
warn " consider opening an issue at"
|
|
99
113
|
warn " https://github.com/r-uby-dev/llm/issues"
|
|
@@ -111,7 +125,7 @@ def main(argv)
|
|
|
111
125
|
# Make sure the dependencies are satisified first
|
|
112
126
|
begin
|
|
113
127
|
require "llm/tools"
|
|
114
|
-
require "llm/
|
|
128
|
+
require "llm/console"
|
|
115
129
|
rescue LLM::LoadError => ex
|
|
116
130
|
loaderror(ex)
|
|
117
131
|
exit 1
|
|
@@ -126,6 +140,9 @@ def main(argv)
|
|
|
126
140
|
when '-h'
|
|
127
141
|
help
|
|
128
142
|
exit 0
|
|
143
|
+
when '-v'
|
|
144
|
+
version
|
|
145
|
+
exit 0
|
|
129
146
|
when '-t'
|
|
130
147
|
temp = true
|
|
131
148
|
when '-c'
|
|
@@ -153,6 +170,22 @@ def main(argv)
|
|
|
153
170
|
help
|
|
154
171
|
exit 1
|
|
155
172
|
end
|
|
173
|
+
when '-m'
|
|
174
|
+
model = argv.shift
|
|
175
|
+
if model.nil?
|
|
176
|
+
warn "llm.rb: -m switch requires an argument"
|
|
177
|
+
help
|
|
178
|
+
exit 1
|
|
179
|
+
end
|
|
180
|
+
when '-x'
|
|
181
|
+
timeout = argv.shift
|
|
182
|
+
if timeout.nil?
|
|
183
|
+
warn "llm.rb: -x switch requires an argument"
|
|
184
|
+
help
|
|
185
|
+
exit 1
|
|
186
|
+
else
|
|
187
|
+
timeout = Integer(timeout)
|
|
188
|
+
end
|
|
156
189
|
else
|
|
157
190
|
warn "llm.rb: unknown option #{option}"
|
|
158
191
|
help
|
|
@@ -164,9 +197,10 @@ def main(argv)
|
|
|
164
197
|
# No provider has been given.
|
|
165
198
|
# Try to infer one.
|
|
166
199
|
transport ||= :net_http
|
|
200
|
+
options = timeout ? {timeout:, transport:} : {transport:}
|
|
167
201
|
if provider.nil?
|
|
168
202
|
llm = providers.filter_map do
|
|
169
|
-
LLM.method(_1).call(
|
|
203
|
+
LLM.method(_1).call(**options)
|
|
170
204
|
rescue ArgumentError
|
|
171
205
|
end.first
|
|
172
206
|
if llm.nil?
|
|
@@ -175,7 +209,7 @@ def main(argv)
|
|
|
175
209
|
end
|
|
176
210
|
else
|
|
177
211
|
begin
|
|
178
|
-
llm = LLM.method(provider).call(
|
|
212
|
+
llm = LLM.method(provider).call(**options)
|
|
179
213
|
rescue ArgumentError
|
|
180
214
|
warn "llm.rb: set credentials for #{provider}"
|
|
181
215
|
exit 1
|
|
@@ -209,12 +243,21 @@ def main(argv)
|
|
|
209
243
|
File.binwrite file, JSON.pretty_generate(data)
|
|
210
244
|
end
|
|
211
245
|
|
|
246
|
+
if model
|
|
247
|
+
if not llm.registry.keys.include?(model)
|
|
248
|
+
warn "llm.rb: #{model} is not a valid #{llm.name} model"
|
|
249
|
+
exit 1
|
|
250
|
+
end
|
|
251
|
+
else
|
|
252
|
+
model = llm.default_model
|
|
253
|
+
end
|
|
254
|
+
|
|
212
255
|
##
|
|
213
256
|
# Let's go!
|
|
214
257
|
concurrency ||= :sequential
|
|
215
258
|
path = temp ? nil : data[Dir.getwd]
|
|
216
|
-
agent = LLM::Agent.new(llm, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
217
|
-
agent.
|
|
259
|
+
agent = LLM::Agent.new(llm, model:, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
260
|
+
agent.console
|
|
218
261
|
rescue Interrupt
|
|
219
262
|
warn "llm.rb: Bye!"
|
|
220
263
|
rescue => ex
|