llm.rb 13.1.0 β 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +320 -0
- data/README.md +340 -31
- data/bin/llm.rb +36 -12
- data/data/anthropic.json +206 -263
- data/data/bedrock.json +2138 -1860
- data/data/deepinfra.json +1003 -624
- data/data/deepseek.json +38 -34
- data/data/google.json +1079 -371
- data/data/mistral.json +448 -368
- data/data/moonshot.json +384 -0
- data/data/openai.json +974 -1343
- data/data/xai.json +154 -126
- data/data/zai.json +191 -191
- data/lib/llm/agent.rb +47 -14
- data/lib/llm/context.rb +71 -88
- data/lib/llm/cost.rb +23 -17
- data/lib/llm/error.rb +0 -8
- data/lib/llm/function/async/task.rb +2 -0
- data/lib/llm/function/fiber/task.rb +2 -0
- data/lib/llm/function/fork/task.rb +2 -0
- data/lib/llm/function/ractor/task.rb +2 -0
- data/lib/llm/function/sequential/group.rb +4 -1
- data/lib/llm/function/sequential/task.rb +1 -1
- data/lib/llm/function/task.rb +4 -0
- data/lib/llm/function/thread/task.rb +2 -0
- data/lib/llm/function.rb +32 -4
- data/lib/llm/guard/loop.rb +89 -0
- data/lib/llm/guard/null.rb +19 -0
- data/lib/llm/guard.rb +61 -0
- data/lib/llm/provider.rb +36 -0
- data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
- data/lib/llm/providers/anthropic.rb +1 -8
- data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
- data/lib/llm/providers/bedrock.rb +1 -8
- data/lib/llm/providers/google/stream_parser.rb +1 -0
- data/lib/llm/providers/google.rb +1 -8
- data/lib/llm/providers/moonshot.rb +76 -0
- data/lib/llm/providers/ollama.rb +1 -8
- data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
- data/lib/llm/providers/openai/responses.rb +6 -8
- data/lib/llm/providers/openai/stream_parser.rb +1 -0
- data/lib/llm/providers/openai.rb +3 -10
- data/lib/llm/repl/bar.rb +4 -3
- data/lib/llm/repl/buffer.rb +42 -15
- data/lib/llm/repl/color.rb +78 -0
- data/lib/llm/repl/input/char.rb +46 -0
- data/lib/llm/repl/input/row.rb +39 -0
- data/lib/llm/repl/input.rb +251 -66
- data/lib/llm/repl/markdown/table.rb +6 -2
- data/lib/llm/repl/markdown.rb +31 -5
- data/lib/llm/repl/status.rb +38 -3
- data/lib/llm/repl/stream.rb +16 -4
- data/lib/llm/repl/walker.rb +3 -2
- data/lib/llm/repl/window.rb +25 -5
- data/lib/llm/repl.rb +29 -13
- data/lib/llm/stream.rb +8 -7
- data/lib/llm/tool.rb +29 -0
- data/lib/llm/transformer/null.rb +21 -0
- data/lib/llm/transformer.rb +55 -0
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +12 -2
- data/llm.gemspec +1 -0
- data/resources/deepdive/advanced/cancellation.md +74 -0
- data/resources/deepdive/advanced/compaction.md +83 -0
- data/resources/deepdive/advanced/context.md +267 -0
- data/resources/deepdive/advanced/guard.md +371 -0
- data/resources/deepdive/advanced/tracer.md +180 -0
- data/resources/deepdive/advanced/transformer.md +67 -0
- data/resources/deepdive/advanced/transports.md +45 -0
- data/resources/deepdive/everything_else/audio.md +122 -0
- data/resources/deepdive/everything_else/cost.md +99 -0
- data/resources/deepdive/everything_else/images.md +89 -0
- data/resources/deepdive/everything_else/object.md +108 -0
- data/resources/deepdive/everything_else/ocr.md +48 -0
- data/resources/deepdive/fundamentals/agents.md +202 -0
- data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
- data/resources/deepdive/fundamentals/concurrency.md +104 -0
- data/resources/deepdive/fundamentals/database.md +449 -0
- data/resources/deepdive/fundamentals/embeddings.md +157 -0
- data/resources/deepdive/fundamentals/repl.md +87 -0
- data/resources/deepdive/fundamentals/schema.md +61 -0
- data/resources/deepdive/fundamentals/skills.md +106 -0
- data/resources/deepdive/fundamentals/stream.md +110 -0
- data/resources/deepdive/fundamentals/tools.md +265 -0
- data/resources/deepdive/protocols/a2a.md +106 -0
- data/resources/deepdive/protocols/mcp.md +111 -0
- data/resources/deepdive.md +7 -1
- metadata +36 -3
- data/lib/llm/loop_guard.rb +0 -107
data/README.md
CHANGED
|
@@ -15,30 +15,153 @@
|
|
|
15
15
|
Welcome to the canonical llm.rb repository.
|
|
16
16
|
|
|
17
17
|
llm.rb is an advanced runtime for building capable AI applications
|
|
18
|
-
on CRuby.
|
|
19
|
-
|
|
18
|
+
on CRuby. It has zero runtime dependencies by default, and a single
|
|
19
|
+
coherent API that spans 12+ providers. Streaming, tools, guards,
|
|
20
|
+
compaction, the REPL, builtin MCP/A2A support and the database
|
|
21
|
+
integrations all build on the same three concepts: providers,
|
|
22
|
+
contexts, and agents.
|
|
23
|
+
|
|
24
|
+
Once you learn the fundamentals, everything else falls into place
|
|
25
|
+
naturally. Some features, such as ActiveRecord support, require
|
|
20
26
|
optional dependencies that are opt-in.
|
|
21
27
|
|
|
22
|
-
When you want to learn more than what the README covers, checkout
|
|
23
|
-
the [deepdive.md](https://r.uby.dev/llm/deepdive/).
|
|
24
|
-
|
|
25
28
|
## Features
|
|
26
29
|
|
|
27
|
-
|
|
28
|
-
Gemini, Mistral, DeepSeek, DeepInfra, xAI, Z.ai,
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
30
|
+
One runtime, 12+ providers. The same API drives OpenAI, Anthropic,
|
|
31
|
+
Google Gemini, Moonshot (kimi), Mistral, DeepSeek, DeepInfra, xAI, Z.ai,
|
|
32
|
+
AWS Bedrock, Ollama, and llama.cpp, so switching models or providers
|
|
33
|
+
can be done with minimal code change.
|
|
34
|
+
|
|
35
|
+
<details>
|
|
36
|
+
<summary><b>Agents</b></summary>
|
|
37
|
+
|
|
38
|
+
* **First-class support** <br>
|
|
39
|
+
llm.rb is designed to build agents. They can be attached to a
|
|
40
|
+
terminal-based read-eval-print loop (repl), persisted to disk
|
|
41
|
+
or a database column, run tools concurrently and be safely
|
|
42
|
+
interrupted.
|
|
43
|
+
|
|
44
|
+
* **Builtin REPL** <br>
|
|
45
|
+
A curses-based TUI for talking to an agent interactively. It
|
|
46
|
+
renders markdown, shows a live status line with context usage
|
|
47
|
+
and running cost, and recalls previous turns, so a
|
|
48
|
+
conversation survives a restart.
|
|
49
|
+
|
|
50
|
+
* **Persistence** <br>
|
|
51
|
+
Set `path:` and the agent saves its conversation to disk
|
|
52
|
+
automatically. ActiveRecord and Sequel support keep the same
|
|
53
|
+
state in a single database column, so you pick the storage
|
|
54
|
+
and the API stays identical.
|
|
55
|
+
|
|
56
|
+
</details>
|
|
57
|
+
|
|
58
|
+
<details>
|
|
59
|
+
<summary><b>MCP & A2A</b></summary>
|
|
60
|
+
|
|
61
|
+
* **MCP** <br>
|
|
62
|
+
The Model Context Protocol is first-class. Point an MCP client
|
|
63
|
+
at any tool server over stdio or HTTP, and its tools translate
|
|
64
|
+
into local `LLM::Tool` subclasses, with the same tracing and
|
|
65
|
+
error handling.
|
|
66
|
+
|
|
67
|
+
* **A2A** <br>
|
|
68
|
+
The Agent 2 Agent protocol is first-class. Point an A2A client
|
|
69
|
+
at another agent over HTTP or JSON-RPC, and call its skills
|
|
70
|
+
exactly like local tools.
|
|
32
71
|
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
72
|
+
</details>
|
|
73
|
+
|
|
74
|
+
<details>
|
|
75
|
+
<summary><b>ORM</b></summary>
|
|
76
|
+
|
|
77
|
+
* **ActiveRecord** <br>
|
|
78
|
+
Add `acts_as_agent` to a model and the agent state lives in a
|
|
79
|
+
single database column, saved after every turn and restored
|
|
80
|
+
on load. Works in Rack and Rails apps, with `jsonb` on
|
|
81
|
+
PostgreSQL.
|
|
82
|
+
|
|
83
|
+
* **Sequel** <br>
|
|
84
|
+
Add `plugin :agent` to a Sequel model for the same single-
|
|
85
|
+
column persistence, with the `pg_json` extension loaded
|
|
86
|
+
automatically on PostgreSQL.
|
|
87
|
+
|
|
88
|
+
</details>
|
|
89
|
+
|
|
90
|
+
<details>
|
|
91
|
+
<summary><b>RAG</b></summary>
|
|
92
|
+
|
|
93
|
+
* **RAG, out of the box** <br>
|
|
94
|
+
Embeddings, OCR, and OpenAI's vector stores API come first-
|
|
95
|
+
class. Ground answers in your own documents, with vectors in
|
|
96
|
+
a managed store or in your own database such as sqlite-vec
|
|
97
|
+
or pgvector.
|
|
98
|
+
|
|
99
|
+
</details>
|
|
100
|
+
|
|
101
|
+
<details>
|
|
102
|
+
<summary><b>Runtime</b></summary>
|
|
36
103
|
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
104
|
+
* **Streaming** <br>
|
|
105
|
+
Streaming is first-class, with structured callbacks for
|
|
106
|
+
content, reasoning, and tool calls. Tools can start while
|
|
107
|
+
the model is still talking, so the first result lands
|
|
108
|
+
before the response finishes.
|
|
109
|
+
|
|
110
|
+
* **Concurrency** <br>
|
|
111
|
+
Six ways to run tools: sequential, threads, async, fibers,
|
|
112
|
+
forks, and ractors. Plus three HTTP backends, so you pick
|
|
113
|
+
the concurrency model that fits the workload, not the other
|
|
114
|
+
way around.
|
|
115
|
+
|
|
116
|
+
* **Interruption** <br>
|
|
117
|
+
Cancel an in-flight request or a running tool at any moment,
|
|
118
|
+
on any transport or concurrency strategy. A stuck call never
|
|
119
|
+
leaves a thread running that you can't stop.
|
|
120
|
+
|
|
121
|
+
</details>
|
|
122
|
+
|
|
123
|
+
<details>
|
|
124
|
+
<summary><b>Provider extras</b></summary>
|
|
125
|
+
|
|
126
|
+
* **DeepSeek-optimized** <br>
|
|
127
|
+
DeepSeek is the most cost-effective option for API users, and the
|
|
128
|
+
runtime closes its gaps: [`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
|
|
129
|
+
makes structured outputs work despite no official API, and
|
|
130
|
+
`images.create`/`edit` produce SVG vector graphics.
|
|
131
|
+
</details>
|
|
132
|
+
|
|
133
|
+
<details>
|
|
134
|
+
<summary><b>Portable</b></summary>
|
|
135
|
+
|
|
136
|
+
* **mruby-llm** <br>
|
|
137
|
+
The same runtime runs on mruby as
|
|
138
|
+
[mruby-llm](https://github.com/r-uby-dev/mruby-llm), with an
|
|
139
|
+
almost identical interface and the same set of capabilities.
|
|
140
|
+
|
|
141
|
+
</details>
|
|
142
|
+
|
|
143
|
+
<details>
|
|
144
|
+
<summary><b>Everything else</b></summary>
|
|
145
|
+
|
|
146
|
+
* **Skills** <br>
|
|
147
|
+
Write a SKILL.md, get a tool. The runtime spawns a
|
|
148
|
+
disposable subagent with the skill's instructions and tool
|
|
149
|
+
set for one turn, then discards it. Fresh and stateless
|
|
150
|
+
every call.
|
|
151
|
+
|
|
152
|
+
* **A unified plugin family** <br>
|
|
153
|
+
Compactors, transformers, and guards all share one
|
|
154
|
+
interface. Context management, message rewriting, and tool
|
|
155
|
+
supervision (policy, quotas, loop detection) plug in the
|
|
156
|
+
same way and compose freely.
|
|
157
|
+
|
|
158
|
+
* **Cost and usage tracking** <br>
|
|
159
|
+
Every context tracks its own cost and token usage, per turn.
|
|
160
|
+
Break the spend down by input, output, cache, and reasoning,
|
|
161
|
+
so the exact cost of any conversation is visible at a
|
|
162
|
+
glance.
|
|
163
|
+
|
|
164
|
+
</details>
|
|
42
165
|
|
|
43
166
|
## Install
|
|
44
167
|
|
|
@@ -54,8 +177,9 @@ The
|
|
|
54
177
|
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
|
|
55
178
|
class is the default high-level interface,
|
|
56
179
|
and it is recommended for most use-cases. It manages tool execution
|
|
57
|
-
automatically
|
|
58
|
-
|
|
180
|
+
automatically and
|
|
181
|
+
[guards against infinite loops](https://r.uby.dev/llm/deepdive/advanced/guard),
|
|
182
|
+
manages conversation state, and much more.
|
|
59
183
|
|
|
60
184
|
```ruby
|
|
61
185
|
require "llm"
|
|
@@ -71,7 +195,8 @@ agent.talk "Hello world"
|
|
|
71
195
|
is a class-level DSL that accepts a Hash of properties. Each key resolves to a
|
|
72
196
|
corresponding class accessor: `name`, `description`, `model`, `tools`,
|
|
73
197
|
`instructions`, `schema`, `stream`, `tracer`, `concurrency`, `confirm`,
|
|
74
|
-
`path`, and `
|
|
198
|
+
`path`, `skills`, and `tool_budget`. All options are optional; zero or
|
|
199
|
+
more can be set.
|
|
75
200
|
An error is raised for unknown keys so that typos are caught early.
|
|
76
201
|
|
|
77
202
|
```ruby
|
|
@@ -123,6 +248,11 @@ sometimes that can be useful, but usually for advanced use-cases.
|
|
|
123
248
|
If you're new to llm.rb, try
|
|
124
249
|
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) first.
|
|
125
250
|
|
|
251
|
+
Every context tracks its own token usage and estimated cost. After any
|
|
252
|
+
turn, you can read the cost breakdown through
|
|
253
|
+
[`LLM::Context#cost`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#cost-instance_method),
|
|
254
|
+
and the REPL shows the running total in the status line.
|
|
255
|
+
|
|
126
256
|
```ruby
|
|
127
257
|
require "llm"
|
|
128
258
|
|
|
@@ -140,6 +270,11 @@ an optional set of typed parameters. <br> The model can choose to
|
|
|
140
270
|
call them on your behalf, and they're one of the most powerful features
|
|
141
271
|
for extending the feature set or abilities of a model.
|
|
142
272
|
|
|
273
|
+
The runtime also ships with a catalog of built-in tools for
|
|
274
|
+
filesystem, search, and shell operations. See the
|
|
275
|
+
[deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/builtin_tools)
|
|
276
|
+
for details.
|
|
277
|
+
|
|
143
278
|
```ruby
|
|
144
279
|
class ReadFile < LLM::Tool
|
|
145
280
|
name "read-file"
|
|
@@ -153,12 +288,39 @@ class ReadFile < LLM::Tool
|
|
|
153
288
|
end
|
|
154
289
|
```
|
|
155
290
|
|
|
291
|
+
##### set
|
|
292
|
+
|
|
293
|
+
[`LLM::Tool.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool.html#set-class_method)
|
|
294
|
+
is an alternative way to define tool properties using a Hash. It works
|
|
295
|
+
the same way as
|
|
296
|
+
[`LLM::Agent.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#set-class_method)
|
|
297
|
+
and accepts the same keys that the individual methods do: `name`,
|
|
298
|
+
`description`, `parameters`, `required`, and `defaults`:
|
|
299
|
+
|
|
300
|
+
```ruby
|
|
301
|
+
class MathTool < LLM::Tool
|
|
302
|
+
set name: "math",
|
|
303
|
+
description: "Performs arithmetic",
|
|
304
|
+
parameters: [
|
|
305
|
+
[:x, Integer, "first number" , {required: true}],
|
|
306
|
+
[:y, Integer, "second number", {default: 0}]
|
|
307
|
+
]
|
|
308
|
+
|
|
309
|
+
def call(x:, y: 0)
|
|
310
|
+
{result: x + y}
|
|
311
|
+
end
|
|
312
|
+
end
|
|
313
|
+
```
|
|
314
|
+
|
|
156
315
|
#### LLM::Stream
|
|
157
316
|
|
|
158
317
|
Streams can be simple IO objects or subclasses of
|
|
159
318
|
[`LLM::Stream`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html)
|
|
160
319
|
with structured callbacks for content,
|
|
161
320
|
reasoning, tool calls, tool returns, and compaction.
|
|
321
|
+
Streams can also observe message transformers, which rewrite
|
|
322
|
+
outgoing messages before they reach the provider (see the
|
|
323
|
+
[deepdive.md](https://r.uby.dev/llm/deepdive/advanced/transformer)).
|
|
162
324
|
|
|
163
325
|
```ruby
|
|
164
326
|
class MyStream < LLM::Stream
|
|
@@ -180,9 +342,12 @@ agent.talk "Explain Ruby fibers."
|
|
|
180
342
|
|
|
181
343
|
[`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
|
|
182
344
|
subclasses produce typed, structured
|
|
183
|
-
output from any model call. Pass a schema to
|
|
184
|
-
`LLM::
|
|
185
|
-
|
|
345
|
+
output from any model call. Pass a schema to
|
|
346
|
+
[`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk-instance_method),
|
|
347
|
+
[`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk-instance_method),
|
|
348
|
+
or
|
|
349
|
+
[`LLM::Provider#complete`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#complete-instance_method)
|
|
350
|
+
to receive validated JSON instead of free text. Schemas work alongside tools and streams.
|
|
186
351
|
|
|
187
352
|
[`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
|
|
188
353
|
can define objects, arrays, enums, nested schemas,
|
|
@@ -217,10 +382,16 @@ res.content! # => {city: "Paris", temperature: 15.0, conditions: "Cloudy"}
|
|
|
217
382
|
|
|
218
383
|
The [LLM::Agent#repl](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#repl-instance_method)
|
|
219
384
|
method drops you into a curses-based TUI for talking to an
|
|
220
|
-
agent interactively.
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
385
|
+
agent interactively. It renders markdown directly in the
|
|
386
|
+
terminal and shows a live status line with context usage,
|
|
387
|
+
running cost, and the current tool call. A second thread keeps
|
|
388
|
+
the UI responsive while the model works. Think of it as
|
|
389
|
+
`binding.pry` but for agents.
|
|
390
|
+
|
|
391
|
+
Set `path:` on the agent for automatic persistence across REPL
|
|
392
|
+
sessions. The `tools:` option attaches extra tools for the
|
|
393
|
+
duration of the session. Recall previous turns with Ctrl+P and
|
|
394
|
+
Ctrl+N. For the full reference, see the
|
|
224
395
|
[REPL section](https://r.uby.dev/llm/deepdive/fundamentals/repl) in the
|
|
225
396
|
deepdive.
|
|
226
397
|
|
|
@@ -268,6 +439,22 @@ agent = LLM::Agent.new(llm, stream: $stdout, tools: mcp.tools)
|
|
|
268
439
|
agent.talk "Run the tool"
|
|
269
440
|
```
|
|
270
441
|
|
|
442
|
+
##### Persistent connections
|
|
443
|
+
|
|
444
|
+
Set `persistent: true` on HTTP transports to reuse connections
|
|
445
|
+
across requests. This uses
|
|
446
|
+
[`Net::HTTP::Persistent`](https://github.com/drbrain/net-http-persistent)
|
|
447
|
+
under the hood and avoids opening a new TCP connection for every
|
|
448
|
+
request:
|
|
449
|
+
|
|
450
|
+
```ruby
|
|
451
|
+
mcp = LLM::MCP.http(
|
|
452
|
+
url: "https://api.githubcopilot.com/mcp/",
|
|
453
|
+
headers: {"Authorization" => "Bearer #{ENV.fetch('GITHUB_PAT')}"},
|
|
454
|
+
persistent: true
|
|
455
|
+
)
|
|
456
|
+
```
|
|
457
|
+
|
|
271
458
|
#### LLM::A2A
|
|
272
459
|
|
|
273
460
|
The Agent 2 Agent (A2A) protocol has first-class support
|
|
@@ -287,6 +474,52 @@ agent = LLM::Agent.new(llm, stream: $stdout, tools: a2a.skills)
|
|
|
287
474
|
agent.talk "Run the skill"
|
|
288
475
|
```
|
|
289
476
|
|
|
477
|
+
##### Persistent connections
|
|
478
|
+
|
|
479
|
+
Set `persistent: true` on HTTP transports to reuse connections
|
|
480
|
+
across requests. This uses
|
|
481
|
+
[`Net::HTTP::Persistent`](https://github.com/drbrain/net-http-persistent)
|
|
482
|
+
under the hood and avoids opening a new TCP connection for every
|
|
483
|
+
request:
|
|
484
|
+
|
|
485
|
+
```ruby
|
|
486
|
+
a2a = LLM::A2A.rest(url: "https://agent.example.com", persistent: true)
|
|
487
|
+
a2a = LLM::A2A.jsonrpc(url: "https://agent.example.com", persistent: true)
|
|
488
|
+
```
|
|
489
|
+
|
|
490
|
+
#### LLM::Guard
|
|
491
|
+
|
|
492
|
+
[`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
|
|
493
|
+
is the hook that sees every tool call before it runs. A guard
|
|
494
|
+
can let a call through, cancel it, block it with an error, or
|
|
495
|
+
even answer for it. Because it runs before the tool, anything
|
|
496
|
+
it intercepts never executes. Policy, validation, quotas, and
|
|
497
|
+
cost ceilings all live here.
|
|
498
|
+
|
|
499
|
+
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
|
|
500
|
+
enables
|
|
501
|
+
[`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
|
|
502
|
+
by default, so agents get loop protection out of the box. To
|
|
503
|
+
write your own guard, subclass
|
|
504
|
+
[`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
|
|
505
|
+
and implement
|
|
506
|
+
[`LLM::Guard#call`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html#call-instance_method).
|
|
507
|
+
The pending call arrives as `function:`. Return a value to close
|
|
508
|
+
the call, or `nil` to let it run:
|
|
509
|
+
|
|
510
|
+
```ruby
|
|
511
|
+
class PolicyGuard < LLM::Guard
|
|
512
|
+
def call(function:)
|
|
513
|
+
if function.name == "shell"
|
|
514
|
+
function.return(error: true, type: "policy_error",
|
|
515
|
+
message: "shell is disabled")
|
|
516
|
+
end
|
|
517
|
+
end
|
|
518
|
+
end
|
|
519
|
+
|
|
520
|
+
agent = LLM::Agent.new(llm, guard: PolicyGuard)
|
|
521
|
+
```
|
|
522
|
+
|
|
290
523
|
#### LLM::Skill
|
|
291
524
|
|
|
292
525
|
A skill turns a markdown file into a callable tool. When the model
|
|
@@ -296,7 +529,7 @@ one turn and returns the result, then is discarded. Each call
|
|
|
296
529
|
is fresh and stateless. For a deeper explanation see the
|
|
297
530
|
[deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/skills).
|
|
298
531
|
|
|
299
|
-
|
|
532
|
+
##### SKILL.md
|
|
300
533
|
|
|
301
534
|
```markdown
|
|
302
535
|
---
|
|
@@ -309,7 +542,7 @@ Collect the recent git log, analyze each commit,
|
|
|
309
542
|
and write a summary to summary.txt.
|
|
310
543
|
```
|
|
311
544
|
|
|
312
|
-
|
|
545
|
+
##### agent.rb
|
|
313
546
|
|
|
314
547
|
```ruby
|
|
315
548
|
require "llm"
|
|
@@ -330,7 +563,8 @@ or PostgreSQL's [pg-vector](https://github.com/pgvector/pgvector).
|
|
|
330
563
|
|
|
331
564
|
llm.rb also includes support for OpenAI's vector store API. It
|
|
332
565
|
provides a vector database as a HTTP service but we won't cover
|
|
333
|
-
that here.
|
|
566
|
+
that here. For a deeper explanation see the
|
|
567
|
+
[deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/embeddings).
|
|
334
568
|
|
|
335
569
|
```ruby
|
|
336
570
|
require "llm"
|
|
@@ -417,6 +651,44 @@ agent = Agent.create!
|
|
|
417
651
|
agent.talk "perform research"
|
|
418
652
|
```
|
|
419
653
|
|
|
654
|
+
#### Images
|
|
655
|
+
|
|
656
|
+
A handful of providers can generate images from a text prompt.
|
|
657
|
+
OpenAI, Google, xAI, and DeepInfra all support it. The API is
|
|
658
|
+
the same across providers:
|
|
659
|
+
|
|
660
|
+
```ruby
|
|
661
|
+
require "llm"
|
|
662
|
+
|
|
663
|
+
llm = LLM.openai(key: ENV["KEY"])
|
|
664
|
+
res = llm.images.create(prompt: "a dog on a rocket to the moon")
|
|
665
|
+
IO.copy_stream res.images[0], "rocket.png"
|
|
666
|
+
```
|
|
667
|
+
|
|
668
|
+
##### DeepSeek
|
|
669
|
+
|
|
670
|
+
DeepSeek does not have a dedicated image model, but the runtime
|
|
671
|
+
generates SVG vector graphics through its text model. Each
|
|
672
|
+
generation produces a valid SVG document that can be converted
|
|
673
|
+
to PNG with tools like `rsvg-convert`. Pass an existing agent
|
|
674
|
+
to maintain a session across generations:
|
|
675
|
+
|
|
676
|
+
```ruby
|
|
677
|
+
require "llm"
|
|
678
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
679
|
+
|
|
680
|
+
##
|
|
681
|
+
# First generation
|
|
682
|
+
res = llm.images.create(prompt: "a rocket on the moon")
|
|
683
|
+
IO.copy_stream res.images[0], "rocket.svg"
|
|
684
|
+
|
|
685
|
+
##
|
|
686
|
+
# Refine with follow-up prompts (shares context)
|
|
687
|
+
res = llm.images.create(prompt: "add a dog next to the rocket",
|
|
688
|
+
agent: res.agent)
|
|
689
|
+
IO.copy_stream res.images[0], "rocket-with-dog.svg"
|
|
690
|
+
```
|
|
691
|
+
|
|
420
692
|
## FAQ
|
|
421
693
|
|
|
422
694
|
<details>
|
|
@@ -437,6 +709,7 @@ In no particular order:
|
|
|
437
709
|
πΊπΈ Anthropic <br>
|
|
438
710
|
π¨π³ DeepSeek <br>
|
|
439
711
|
π¨π³ zAI <br>
|
|
712
|
+
π¨π³ Moonshot AI (Kimi) <br>
|
|
440
713
|
πͺπΊ Mistral <br>
|
|
441
714
|
|
|
442
715
|
**Weights**
|
|
@@ -448,6 +721,7 @@ In no particular order:
|
|
|
448
721
|
πΊπΈ AWS bedrock <br>
|
|
449
722
|
π¨π³ DeepSeek <br>
|
|
450
723
|
π¨π³ zAI <br>
|
|
724
|
+
π¨π³ Moonshot AI (Kimi) <br>
|
|
451
725
|
πͺπΊ Mistral <br>
|
|
452
726
|
|
|
453
727
|
**Local**
|
|
@@ -512,6 +786,41 @@ wasn't possible to cover every feature without the README becoming a small book.
|
|
|
512
786
|
The [r.uby.dev](https://r.uby.dev) homepage also includes more learning material
|
|
513
787
|
and resources.
|
|
514
788
|
|
|
789
|
+
## Developers
|
|
790
|
+
|
|
791
|
+
The llm.rb project is quite large and maintained primarily by one
|
|
792
|
+
person. It would be near impossible for me to maintain both the codebase
|
|
793
|
+
and its documentation, especially the [deepdive.md](https://r.uby.dev/llm/deepdive/)
|
|
794
|
+
so I have written agents that maintain the documentation assets and that
|
|
795
|
+
allows me to put more focus on the code.
|
|
796
|
+
|
|
797
|
+
The following agents are available for those tasks, and all of them
|
|
798
|
+
use the most cost effective option: DeepSeek. Feel free to use them
|
|
799
|
+
in your own fork.
|
|
800
|
+
|
|
801
|
+
```sh
|
|
802
|
+
##
|
|
803
|
+
# Maintains the deepdive and API docs
|
|
804
|
+
rake agents:scribe:yardoc
|
|
805
|
+
rake agents:scribe:coverage
|
|
806
|
+
rake agents:scribe:regressions
|
|
807
|
+
rake agents:scribe:style
|
|
808
|
+
|
|
809
|
+
##
|
|
810
|
+
# Maintains the release
|
|
811
|
+
rake agents:dexter:changelog
|
|
812
|
+
rake agents:dexter:release
|
|
813
|
+
|
|
814
|
+
##
|
|
815
|
+
# Maintains mruby-llm backports
|
|
816
|
+
rake agents:mruby:research
|
|
817
|
+
rake agents:mruby:implement
|
|
818
|
+
|
|
819
|
+
##
|
|
820
|
+
# Refresh the data/ registry
|
|
821
|
+
rake models.dev:download
|
|
822
|
+
```
|
|
823
|
+
|
|
515
824
|
## License
|
|
516
825
|
|
|
517
826
|
This software is released under the terms of the MIT license. <br>
|
data/bin/llm.rb
CHANGED
|
@@ -1,7 +1,9 @@
|
|
|
1
1
|
#!/usr/bin/env ruby
|
|
2
2
|
|
|
3
3
|
require "llm"
|
|
4
|
+
require "json"
|
|
4
5
|
require "fileutils"
|
|
6
|
+
require "securerandom"
|
|
5
7
|
|
|
6
8
|
##
|
|
7
9
|
# utils
|
|
@@ -76,20 +78,18 @@ def main(argv)
|
|
|
76
78
|
temp = true
|
|
77
79
|
when '-p'
|
|
78
80
|
provider = argv.shift
|
|
81
|
+
if provider.nil?
|
|
82
|
+
warn "llm.rb: -p switch requires an argument"
|
|
83
|
+
help
|
|
84
|
+
exit 1
|
|
85
|
+
end
|
|
79
86
|
else
|
|
80
87
|
warn "llm.rb: unknown option #{option}"
|
|
88
|
+
help
|
|
89
|
+
exit 1
|
|
81
90
|
end
|
|
82
91
|
end
|
|
83
92
|
|
|
84
|
-
##
|
|
85
|
-
# Setup the home directory
|
|
86
|
-
# But only if the `-t` switch has not been provided
|
|
87
|
-
if temp.nil?
|
|
88
|
-
home = File.join(Dir.home, ".llm.rb")
|
|
89
|
-
FileUtils.mkdir_p(File.join(home, Dir.getwd))
|
|
90
|
-
session = File.join(home, Dir.getwd, "session.json")
|
|
91
|
-
end
|
|
92
|
-
|
|
93
93
|
##
|
|
94
94
|
# No provider has been given.
|
|
95
95
|
# Try to infer one.
|
|
@@ -102,19 +102,43 @@ def main(argv)
|
|
|
102
102
|
provider, = key.split("_")
|
|
103
103
|
end
|
|
104
104
|
end
|
|
105
|
+
provider = provider.downcase
|
|
106
|
+
|
|
107
|
+
##
|
|
108
|
+
# Setup the filesystem where <provider>.json maps
|
|
109
|
+
# the current working directory to a session file,
|
|
110
|
+
# and where the session file is stored in
|
|
111
|
+
# `~/.llm.rb/<provider>/<uuid>.json`.
|
|
112
|
+
# This can be skipped with the `-t` option.
|
|
113
|
+
if temp.nil?
|
|
114
|
+
home = File.join(Dir.home, ".llm.rb")
|
|
115
|
+
file = File.join(home, "#{provider}.json")
|
|
116
|
+
parent = File.join(home, provider)
|
|
117
|
+
|
|
118
|
+
FileUtils.mkdir_p(parent)
|
|
119
|
+
FileUtils.touch(file)
|
|
120
|
+
|
|
121
|
+
if File.size(file).zero?
|
|
122
|
+
data = LLM::Object.from({})
|
|
123
|
+
File.binwrite file, JSON.pretty_generate(data)
|
|
124
|
+
else
|
|
125
|
+
data = LLM::Object.from JSON.parse(File.read(file))
|
|
126
|
+
end
|
|
127
|
+
data[Dir.getwd] ||= File.join(parent, "#{SecureRandom.uuid}.json")
|
|
128
|
+
end
|
|
105
129
|
|
|
106
130
|
##
|
|
107
131
|
# We're ready to start the REPL
|
|
108
132
|
# This should always succeed unless -p gave garbage
|
|
109
|
-
provider = provider.downcase
|
|
110
133
|
if LLM.respond_to?(provider)
|
|
111
134
|
key ||= "#{provider.upcase}_API_KEY"
|
|
112
135
|
if ENV[key].nil? || ENV[key].to_s.empty?
|
|
113
136
|
warn "llm.rb: set #{key} to use #{provider}"
|
|
114
137
|
exit 1
|
|
115
138
|
end
|
|
116
|
-
llm
|
|
117
|
-
|
|
139
|
+
llm = LLM.method(provider).call(key: ENV[key])
|
|
140
|
+
path = temp ? nil : data[Dir.getwd]
|
|
141
|
+
agent = LLM::Agent.new(llm, path:, tools: LLM::Tool.subclasses)
|
|
118
142
|
agent.repl
|
|
119
143
|
else
|
|
120
144
|
warn "llm.rb: #{provider} was not recognized"
|