llm.rb 13.1.0 β†’ 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (90) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +320 -0
  3. data/README.md +340 -31
  4. data/bin/llm.rb +36 -12
  5. data/data/anthropic.json +206 -263
  6. data/data/bedrock.json +2138 -1860
  7. data/data/deepinfra.json +1003 -624
  8. data/data/deepseek.json +38 -34
  9. data/data/google.json +1079 -371
  10. data/data/mistral.json +448 -368
  11. data/data/moonshot.json +384 -0
  12. data/data/openai.json +974 -1343
  13. data/data/xai.json +154 -126
  14. data/data/zai.json +191 -191
  15. data/lib/llm/agent.rb +47 -14
  16. data/lib/llm/context.rb +71 -88
  17. data/lib/llm/cost.rb +23 -17
  18. data/lib/llm/error.rb +0 -8
  19. data/lib/llm/function/async/task.rb +2 -0
  20. data/lib/llm/function/fiber/task.rb +2 -0
  21. data/lib/llm/function/fork/task.rb +2 -0
  22. data/lib/llm/function/ractor/task.rb +2 -0
  23. data/lib/llm/function/sequential/group.rb +4 -1
  24. data/lib/llm/function/sequential/task.rb +1 -1
  25. data/lib/llm/function/task.rb +4 -0
  26. data/lib/llm/function/thread/task.rb +2 -0
  27. data/lib/llm/function.rb +32 -4
  28. data/lib/llm/guard/loop.rb +89 -0
  29. data/lib/llm/guard/null.rb +19 -0
  30. data/lib/llm/guard.rb +61 -0
  31. data/lib/llm/provider.rb +36 -0
  32. data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
  33. data/lib/llm/providers/anthropic.rb +1 -8
  34. data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
  35. data/lib/llm/providers/bedrock.rb +1 -8
  36. data/lib/llm/providers/google/stream_parser.rb +1 -0
  37. data/lib/llm/providers/google.rb +1 -8
  38. data/lib/llm/providers/moonshot.rb +76 -0
  39. data/lib/llm/providers/ollama.rb +1 -8
  40. data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
  41. data/lib/llm/providers/openai/responses.rb +6 -8
  42. data/lib/llm/providers/openai/stream_parser.rb +1 -0
  43. data/lib/llm/providers/openai.rb +3 -10
  44. data/lib/llm/repl/bar.rb +4 -3
  45. data/lib/llm/repl/buffer.rb +42 -15
  46. data/lib/llm/repl/color.rb +78 -0
  47. data/lib/llm/repl/input/char.rb +46 -0
  48. data/lib/llm/repl/input/row.rb +39 -0
  49. data/lib/llm/repl/input.rb +251 -66
  50. data/lib/llm/repl/markdown/table.rb +6 -2
  51. data/lib/llm/repl/markdown.rb +31 -5
  52. data/lib/llm/repl/status.rb +38 -3
  53. data/lib/llm/repl/stream.rb +16 -4
  54. data/lib/llm/repl/walker.rb +3 -2
  55. data/lib/llm/repl/window.rb +25 -5
  56. data/lib/llm/repl.rb +29 -13
  57. data/lib/llm/stream.rb +8 -7
  58. data/lib/llm/tool.rb +29 -0
  59. data/lib/llm/transformer/null.rb +21 -0
  60. data/lib/llm/transformer.rb +55 -0
  61. data/lib/llm/version.rb +1 -1
  62. data/lib/llm.rb +12 -2
  63. data/llm.gemspec +1 -0
  64. data/resources/deepdive/advanced/cancellation.md +74 -0
  65. data/resources/deepdive/advanced/compaction.md +83 -0
  66. data/resources/deepdive/advanced/context.md +267 -0
  67. data/resources/deepdive/advanced/guard.md +371 -0
  68. data/resources/deepdive/advanced/tracer.md +180 -0
  69. data/resources/deepdive/advanced/transformer.md +67 -0
  70. data/resources/deepdive/advanced/transports.md +45 -0
  71. data/resources/deepdive/everything_else/audio.md +122 -0
  72. data/resources/deepdive/everything_else/cost.md +99 -0
  73. data/resources/deepdive/everything_else/images.md +89 -0
  74. data/resources/deepdive/everything_else/object.md +108 -0
  75. data/resources/deepdive/everything_else/ocr.md +48 -0
  76. data/resources/deepdive/fundamentals/agents.md +202 -0
  77. data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
  78. data/resources/deepdive/fundamentals/concurrency.md +104 -0
  79. data/resources/deepdive/fundamentals/database.md +449 -0
  80. data/resources/deepdive/fundamentals/embeddings.md +157 -0
  81. data/resources/deepdive/fundamentals/repl.md +87 -0
  82. data/resources/deepdive/fundamentals/schema.md +61 -0
  83. data/resources/deepdive/fundamentals/skills.md +106 -0
  84. data/resources/deepdive/fundamentals/stream.md +110 -0
  85. data/resources/deepdive/fundamentals/tools.md +265 -0
  86. data/resources/deepdive/protocols/a2a.md +106 -0
  87. data/resources/deepdive/protocols/mcp.md +111 -0
  88. data/resources/deepdive.md +7 -1
  89. metadata +36 -3
  90. data/lib/llm/loop_guard.rb +0 -107
data/README.md CHANGED
@@ -15,30 +15,153 @@
15
15
  Welcome to the canonical llm.rb repository.
16
16
 
17
17
  llm.rb is an advanced runtime for building capable AI applications
18
- on CRuby. By default it has zero runtime dependencies although certain
19
- functionality (such as ActiveRecord support) require
18
+ on CRuby. It has zero runtime dependencies by default, and a single
19
+ coherent API that spans 12+ providers. Streaming, tools, guards,
20
+ compaction, the REPL, builtin MCP/A2A support and the database
21
+ integrations all build on the same three concepts: providers,
22
+ contexts, and agents.
23
+
24
+ Once you learn the fundamentals, everything else falls into place
25
+ naturally. Some features, such as ActiveRecord support, require
20
26
  optional dependencies that are opt-in.
21
27
 
22
- When you want to learn more than what the README covers, checkout
23
- the [deepdive.md](https://r.uby.dev/llm/deepdive/).
24
-
25
28
  ## Features
26
29
 
27
- The runtime supports OpenAI, OpenAI-compatible endpoints, Anthropic, Google
28
- Gemini, Mistral, DeepSeek, DeepInfra, xAI, Z.ai, AWS Bedrock, Ollama, and llama.cpp.
29
- It has first-class support for streaming, tool calls, MCP
30
- and A2A, embeddings, vector stores, OCR, context compaction,
31
- and the RAG pattern.
30
+ One runtime, 12+ providers. The same API drives OpenAI, Anthropic,
31
+ Google Gemini, Moonshot (kimi), Mistral, DeepSeek, DeepInfra, xAI, Z.ai,
32
+ AWS Bedrock, Ollama, and llama.cpp, so switching models or providers
33
+ can be done with minimal code change.
34
+
35
+ <details>
36
+ <summary><b>Agents</b></summary>
37
+
38
+ * **First-class support** <br>
39
+ llm.rb is designed to build agents. They can be attached to a
40
+ terminal-based read-eval-print loop (repl), persisted to disk
41
+ or a database column, run tools concurrently and be safely
42
+ interrupted.
43
+
44
+ * **Builtin REPL** <br>
45
+ A curses-based TUI for talking to an agent interactively. It
46
+ renders markdown, shows a live status line with context usage
47
+ and running cost, and recalls previous turns, so a
48
+ conversation survives a restart.
49
+
50
+ * **Persistence** <br>
51
+ Set `path:` and the agent saves its conversation to disk
52
+ automatically. ActiveRecord and Sequel support keep the same
53
+ state in a single database column, so you pick the storage
54
+ and the API stays identical.
55
+
56
+ </details>
57
+
58
+ <details>
59
+ <summary><b>MCP &amp; A2A</b></summary>
60
+
61
+ * **MCP** <br>
62
+ The Model Context Protocol is first-class. Point an MCP client
63
+ at any tool server over stdio or HTTP, and its tools translate
64
+ into local `LLM::Tool` subclasses, with the same tracing and
65
+ error handling.
66
+
67
+ * **A2A** <br>
68
+ The Agent 2 Agent protocol is first-class. Point an A2A client
69
+ at another agent over HTTP or JSON-RPC, and call its skills
70
+ exactly like local tools.
32
71
 
33
- There are multiple HTTP backends to choose from, tools can be run concurrently
34
- or in parallel via threads, async tasks, fibers, ractors, and fork, and it is
35
- also possible to make a tool call while the model is still streaming.
72
+ </details>
73
+
74
+ <details>
75
+ <summary><b>ORM</b></summary>
76
+
77
+ * **ActiveRecord** <br>
78
+ Add `acts_as_agent` to a model and the agent state lives in a
79
+ single database column, saved after every turn and restored
80
+ on load. Works in Rack and Rails apps, with `jsonb` on
81
+ PostgreSQL.
82
+
83
+ * **Sequel** <br>
84
+ Add `plugin :agent` to a Sequel model for the same single-
85
+ column persistence, with the `pg_json` extension loaded
86
+ automatically on PostgreSQL.
87
+
88
+ </details>
89
+
90
+ <details>
91
+ <summary><b>RAG</b></summary>
92
+
93
+ * **RAG, out of the box** <br>
94
+ Embeddings, OCR, and OpenAI's vector stores API come first-
95
+ class. Ground answers in your own documents, with vectors in
96
+ a managed store or in your own database such as sqlite-vec
97
+ or pgvector.
98
+
99
+ </details>
100
+
101
+ <details>
102
+ <summary><b>Runtime</b></summary>
36
103
 
37
- The runtime builds on top of three core concepts: providers, contexts, and agents,
38
- so once you learn the fundamentals, everything else falls into place naturally. And once
39
- you learn llm.rb, you will also be able to use
40
- <a href="https://r.uby.dev/mruby-llm">mruby-llm</a> and
41
- <a href="https://r.uby.dev/wasm-llm">wasm-llm</a> because the API is pretty much identical.
104
+ * **Streaming** <br>
105
+ Streaming is first-class, with structured callbacks for
106
+ content, reasoning, and tool calls. Tools can start while
107
+ the model is still talking, so the first result lands
108
+ before the response finishes.
109
+
110
+ * **Concurrency** <br>
111
+ Six ways to run tools: sequential, threads, async, fibers,
112
+ forks, and ractors. Plus three HTTP backends, so you pick
113
+ the concurrency model that fits the workload, not the other
114
+ way around.
115
+
116
+ * **Interruption** <br>
117
+ Cancel an in-flight request or a running tool at any moment,
118
+ on any transport or concurrency strategy. A stuck call never
119
+ leaves a thread running that you can't stop.
120
+
121
+ </details>
122
+
123
+ <details>
124
+ <summary><b>Provider extras</b></summary>
125
+
126
+ * **DeepSeek-optimized** <br>
127
+ DeepSeek is the most cost-effective option for API users, and the
128
+ runtime closes its gaps: [`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
129
+ makes structured outputs work despite no official API, and
130
+ `images.create`/`edit` produce SVG vector graphics.
131
+ </details>
132
+
133
+ <details>
134
+ <summary><b>Portable</b></summary>
135
+
136
+ * **mruby-llm** <br>
137
+ The same runtime runs on mruby as
138
+ [mruby-llm](https://github.com/r-uby-dev/mruby-llm), with an
139
+ almost identical interface and the same set of capabilities.
140
+
141
+ </details>
142
+
143
+ <details>
144
+ <summary><b>Everything else</b></summary>
145
+
146
+ * **Skills** <br>
147
+ Write a SKILL.md, get a tool. The runtime spawns a
148
+ disposable subagent with the skill's instructions and tool
149
+ set for one turn, then discards it. Fresh and stateless
150
+ every call.
151
+
152
+ * **A unified plugin family** <br>
153
+ Compactors, transformers, and guards all share one
154
+ interface. Context management, message rewriting, and tool
155
+ supervision (policy, quotas, loop detection) plug in the
156
+ same way and compose freely.
157
+
158
+ * **Cost and usage tracking** <br>
159
+ Every context tracks its own cost and token usage, per turn.
160
+ Break the spend down by input, output, cache, and reasoning,
161
+ so the exact cost of any conversation is visible at a
162
+ glance.
163
+
164
+ </details>
42
165
 
43
166
  ## Install
44
167
 
@@ -54,8 +177,9 @@ The
54
177
  [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
55
178
  class is the default high-level interface,
56
179
  and it is recommended for most use-cases. It manages tool execution
57
- automatically, guards against infinite loops, manages conversation
58
- state, and much more.
180
+ automatically and
181
+ [guards against infinite loops](https://r.uby.dev/llm/deepdive/advanced/guard),
182
+ manages conversation state, and much more.
59
183
 
60
184
  ```ruby
61
185
  require "llm"
@@ -71,7 +195,8 @@ agent.talk "Hello world"
71
195
  is a class-level DSL that accepts a Hash of properties. Each key resolves to a
72
196
  corresponding class accessor: `name`, `description`, `model`, `tools`,
73
197
  `instructions`, `schema`, `stream`, `tracer`, `concurrency`, `confirm`,
74
- `path`, and `skills`. All options are optional; zero or more can be set.
198
+ `path`, `skills`, and `tool_budget`. All options are optional; zero or
199
+ more can be set.
75
200
  An error is raised for unknown keys so that typos are caught early.
76
201
 
77
202
  ```ruby
@@ -123,6 +248,11 @@ sometimes that can be useful, but usually for advanced use-cases.
123
248
  If you're new to llm.rb, try
124
249
  [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) first.
125
250
 
251
+ Every context tracks its own token usage and estimated cost. After any
252
+ turn, you can read the cost breakdown through
253
+ [`LLM::Context#cost`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#cost-instance_method),
254
+ and the REPL shows the running total in the status line.
255
+
126
256
  ```ruby
127
257
  require "llm"
128
258
 
@@ -140,6 +270,11 @@ an optional set of typed parameters. <br> The model can choose to
140
270
  call them on your behalf, and they're one of the most powerful features
141
271
  for extending the feature set or abilities of a model.
142
272
 
273
+ The runtime also ships with a catalog of built-in tools for
274
+ filesystem, search, and shell operations. See the
275
+ [deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/builtin_tools)
276
+ for details.
277
+
143
278
  ```ruby
144
279
  class ReadFile < LLM::Tool
145
280
  name "read-file"
@@ -153,12 +288,39 @@ class ReadFile < LLM::Tool
153
288
  end
154
289
  ```
155
290
 
291
+ ##### set
292
+
293
+ [`LLM::Tool.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool.html#set-class_method)
294
+ is an alternative way to define tool properties using a Hash. It works
295
+ the same way as
296
+ [`LLM::Agent.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#set-class_method)
297
+ and accepts the same keys that the individual methods do: `name`,
298
+ `description`, `parameters`, `required`, and `defaults`:
299
+
300
+ ```ruby
301
+ class MathTool < LLM::Tool
302
+ set name: "math",
303
+ description: "Performs arithmetic",
304
+ parameters: [
305
+ [:x, Integer, "first number" , {required: true}],
306
+ [:y, Integer, "second number", {default: 0}]
307
+ ]
308
+
309
+ def call(x:, y: 0)
310
+ {result: x + y}
311
+ end
312
+ end
313
+ ```
314
+
156
315
  #### LLM::Stream
157
316
 
158
317
  Streams can be simple IO objects or subclasses of
159
318
  [`LLM::Stream`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html)
160
319
  with structured callbacks for content,
161
320
  reasoning, tool calls, tool returns, and compaction.
321
+ Streams can also observe message transformers, which rewrite
322
+ outgoing messages before they reach the provider (see the
323
+ [deepdive.md](https://r.uby.dev/llm/deepdive/advanced/transformer)).
162
324
 
163
325
  ```ruby
164
326
  class MyStream < LLM::Stream
@@ -180,9 +342,12 @@ agent.talk "Explain Ruby fibers."
180
342
 
181
343
  [`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
182
344
  subclasses produce typed, structured
183
- output from any model call. Pass a schema to `LLM::Context#talk`,
184
- `LLM::Agent#talk`, or `LLM::Provider#complete` to receive validated
185
- JSON instead of free text. Schemas work alongside tools and streams.
345
+ output from any model call. Pass a schema to
346
+ [`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk-instance_method),
347
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk-instance_method),
348
+ or
349
+ [`LLM::Provider#complete`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#complete-instance_method)
350
+ to receive validated JSON instead of free text. Schemas work alongside tools and streams.
186
351
 
187
352
  [`LLM::Schema`](https://r.uby.dev/api-docs/llm.rb/LLM/Schema.html)
188
353
  can define objects, arrays, enums, nested schemas,
@@ -217,10 +382,16 @@ res.content! # => {city: "Paris", temperature: 15.0, conditions: "Cloudy"}
217
382
 
218
383
  The [LLM::Agent#repl](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#repl-instance_method)
219
384
  method drops you into a curses-based TUI for talking to an
220
- agent interactively. Set `path:` on the agent for automatic
221
- persistence across REPL sessions. The `tools:` option attaches
222
- extra tools for the duration of the session. It is like
223
- `binding.pry` but for agents. For the full reference see the
385
+ agent interactively. It renders markdown directly in the
386
+ terminal and shows a live status line with context usage,
387
+ running cost, and the current tool call. A second thread keeps
388
+ the UI responsive while the model works. Think of it as
389
+ `binding.pry` but for agents.
390
+
391
+ Set `path:` on the agent for automatic persistence across REPL
392
+ sessions. The `tools:` option attaches extra tools for the
393
+ duration of the session. Recall previous turns with Ctrl+P and
394
+ Ctrl+N. For the full reference, see the
224
395
  [REPL section](https://r.uby.dev/llm/deepdive/fundamentals/repl) in the
225
396
  deepdive.
226
397
 
@@ -268,6 +439,22 @@ agent = LLM::Agent.new(llm, stream: $stdout, tools: mcp.tools)
268
439
  agent.talk "Run the tool"
269
440
  ```
270
441
 
442
+ ##### Persistent connections
443
+
444
+ Set `persistent: true` on HTTP transports to reuse connections
445
+ across requests. This uses
446
+ [`Net::HTTP::Persistent`](https://github.com/drbrain/net-http-persistent)
447
+ under the hood and avoids opening a new TCP connection for every
448
+ request:
449
+
450
+ ```ruby
451
+ mcp = LLM::MCP.http(
452
+ url: "https://api.githubcopilot.com/mcp/",
453
+ headers: {"Authorization" => "Bearer #{ENV.fetch('GITHUB_PAT')}"},
454
+ persistent: true
455
+ )
456
+ ```
457
+
271
458
  #### LLM::A2A
272
459
 
273
460
  The Agent 2 Agent (A2A) protocol has first-class support
@@ -287,6 +474,52 @@ agent = LLM::Agent.new(llm, stream: $stdout, tools: a2a.skills)
287
474
  agent.talk "Run the skill"
288
475
  ```
289
476
 
477
+ ##### Persistent connections
478
+
479
+ Set `persistent: true` on HTTP transports to reuse connections
480
+ across requests. This uses
481
+ [`Net::HTTP::Persistent`](https://github.com/drbrain/net-http-persistent)
482
+ under the hood and avoids opening a new TCP connection for every
483
+ request:
484
+
485
+ ```ruby
486
+ a2a = LLM::A2A.rest(url: "https://agent.example.com", persistent: true)
487
+ a2a = LLM::A2A.jsonrpc(url: "https://agent.example.com", persistent: true)
488
+ ```
489
+
490
+ #### LLM::Guard
491
+
492
+ [`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
493
+ is the hook that sees every tool call before it runs. A guard
494
+ can let a call through, cancel it, block it with an error, or
495
+ even answer for it. Because it runs before the tool, anything
496
+ it intercepts never executes. Policy, validation, quotas, and
497
+ cost ceilings all live here.
498
+
499
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
500
+ enables
501
+ [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
502
+ by default, so agents get loop protection out of the box. To
503
+ write your own guard, subclass
504
+ [`LLM::Guard`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html)
505
+ and implement
506
+ [`LLM::Guard#call`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard.html#call-instance_method).
507
+ The pending call arrives as `function:`. Return a value to close
508
+ the call, or `nil` to let it run:
509
+
510
+ ```ruby
511
+ class PolicyGuard < LLM::Guard
512
+ def call(function:)
513
+ if function.name == "shell"
514
+ function.return(error: true, type: "policy_error",
515
+ message: "shell is disabled")
516
+ end
517
+ end
518
+ end
519
+
520
+ agent = LLM::Agent.new(llm, guard: PolicyGuard)
521
+ ```
522
+
290
523
  #### LLM::Skill
291
524
 
292
525
  A skill turns a markdown file into a callable tool. When the model
@@ -296,7 +529,7 @@ one turn and returns the result, then is discarded. Each call
296
529
  is fresh and stateless. For a deeper explanation see the
297
530
  [deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/skills).
298
531
 
299
- **SKILL.md**
532
+ ##### SKILL.md
300
533
 
301
534
  ```markdown
302
535
  ---
@@ -309,7 +542,7 @@ Collect the recent git log, analyze each commit,
309
542
  and write a summary to summary.txt.
310
543
  ```
311
544
 
312
- **agent.rb**
545
+ ##### agent.rb
313
546
 
314
547
  ```ruby
315
548
  require "llm"
@@ -330,7 +563,8 @@ or PostgreSQL's [pg-vector](https://github.com/pgvector/pgvector).
330
563
 
331
564
  llm.rb also includes support for OpenAI's vector store API. It
332
565
  provides a vector database as a HTTP service but we won't cover
333
- that here.
566
+ that here. For a deeper explanation see the
567
+ [deepdive.md](https://r.uby.dev/llm/deepdive/fundamentals/embeddings).
334
568
 
335
569
  ```ruby
336
570
  require "llm"
@@ -417,6 +651,44 @@ agent = Agent.create!
417
651
  agent.talk "perform research"
418
652
  ```
419
653
 
654
+ #### Images
655
+
656
+ A handful of providers can generate images from a text prompt.
657
+ OpenAI, Google, xAI, and DeepInfra all support it. The API is
658
+ the same across providers:
659
+
660
+ ```ruby
661
+ require "llm"
662
+
663
+ llm = LLM.openai(key: ENV["KEY"])
664
+ res = llm.images.create(prompt: "a dog on a rocket to the moon")
665
+ IO.copy_stream res.images[0], "rocket.png"
666
+ ```
667
+
668
+ ##### DeepSeek
669
+
670
+ DeepSeek does not have a dedicated image model, but the runtime
671
+ generates SVG vector graphics through its text model. Each
672
+ generation produces a valid SVG document that can be converted
673
+ to PNG with tools like `rsvg-convert`. Pass an existing agent
674
+ to maintain a session across generations:
675
+
676
+ ```ruby
677
+ require "llm"
678
+ llm = LLM.deepseek(key: ENV["KEY"])
679
+
680
+ ##
681
+ # First generation
682
+ res = llm.images.create(prompt: "a rocket on the moon")
683
+ IO.copy_stream res.images[0], "rocket.svg"
684
+
685
+ ##
686
+ # Refine with follow-up prompts (shares context)
687
+ res = llm.images.create(prompt: "add a dog next to the rocket",
688
+ agent: res.agent)
689
+ IO.copy_stream res.images[0], "rocket-with-dog.svg"
690
+ ```
691
+
420
692
  ## FAQ
421
693
 
422
694
  <details>
@@ -437,6 +709,7 @@ In no particular order:
437
709
  πŸ‡ΊπŸ‡Έ Anthropic <br>
438
710
  πŸ‡¨πŸ‡³ DeepSeek <br>
439
711
  πŸ‡¨πŸ‡³ zAI <br>
712
+ πŸ‡¨πŸ‡³ Moonshot AI (Kimi) <br>
440
713
  πŸ‡ͺπŸ‡Ί Mistral <br>
441
714
 
442
715
  **Weights**
@@ -448,6 +721,7 @@ In no particular order:
448
721
  πŸ‡ΊπŸ‡Έ AWS bedrock <br>
449
722
  πŸ‡¨πŸ‡³ DeepSeek <br>
450
723
  πŸ‡¨πŸ‡³ zAI <br>
724
+ πŸ‡¨πŸ‡³ Moonshot AI (Kimi) <br>
451
725
  πŸ‡ͺπŸ‡Ί Mistral <br>
452
726
 
453
727
  **Local**
@@ -512,6 +786,41 @@ wasn't possible to cover every feature without the README becoming a small book.
512
786
  The [r.uby.dev](https://r.uby.dev) homepage also includes more learning material
513
787
  and resources.
514
788
 
789
+ ## Developers
790
+
791
+ The llm.rb project is quite large and maintained primarily by one
792
+ person. It would be near impossible for me to maintain both the codebase
793
+ and its documentation, especially the [deepdive.md](https://r.uby.dev/llm/deepdive/)
794
+ so I have written agents that maintain the documentation assets and that
795
+ allows me to put more focus on the code.
796
+
797
+ The following agents are available for those tasks, and all of them
798
+ use the most cost effective option: DeepSeek. Feel free to use them
799
+ in your own fork.
800
+
801
+ ```sh
802
+ ##
803
+ # Maintains the deepdive and API docs
804
+ rake agents:scribe:yardoc
805
+ rake agents:scribe:coverage
806
+ rake agents:scribe:regressions
807
+ rake agents:scribe:style
808
+
809
+ ##
810
+ # Maintains the release
811
+ rake agents:dexter:changelog
812
+ rake agents:dexter:release
813
+
814
+ ##
815
+ # Maintains mruby-llm backports
816
+ rake agents:mruby:research
817
+ rake agents:mruby:implement
818
+
819
+ ##
820
+ # Refresh the data/ registry
821
+ rake models.dev:download
822
+ ```
823
+
515
824
  ## License
516
825
 
517
826
  This software is released under the terms of the MIT license. <br>
data/bin/llm.rb CHANGED
@@ -1,7 +1,9 @@
1
1
  #!/usr/bin/env ruby
2
2
 
3
3
  require "llm"
4
+ require "json"
4
5
  require "fileutils"
6
+ require "securerandom"
5
7
 
6
8
  ##
7
9
  # utils
@@ -76,20 +78,18 @@ def main(argv)
76
78
  temp = true
77
79
  when '-p'
78
80
  provider = argv.shift
81
+ if provider.nil?
82
+ warn "llm.rb: -p switch requires an argument"
83
+ help
84
+ exit 1
85
+ end
79
86
  else
80
87
  warn "llm.rb: unknown option #{option}"
88
+ help
89
+ exit 1
81
90
  end
82
91
  end
83
92
 
84
- ##
85
- # Setup the home directory
86
- # But only if the `-t` switch has not been provided
87
- if temp.nil?
88
- home = File.join(Dir.home, ".llm.rb")
89
- FileUtils.mkdir_p(File.join(home, Dir.getwd))
90
- session = File.join(home, Dir.getwd, "session.json")
91
- end
92
-
93
93
  ##
94
94
  # No provider has been given.
95
95
  # Try to infer one.
@@ -102,19 +102,43 @@ def main(argv)
102
102
  provider, = key.split("_")
103
103
  end
104
104
  end
105
+ provider = provider.downcase
106
+
107
+ ##
108
+ # Setup the filesystem where <provider>.json maps
109
+ # the current working directory to a session file,
110
+ # and where the session file is stored in
111
+ # `~/.llm.rb/<provider>/<uuid>.json`.
112
+ # This can be skipped with the `-t` option.
113
+ if temp.nil?
114
+ home = File.join(Dir.home, ".llm.rb")
115
+ file = File.join(home, "#{provider}.json")
116
+ parent = File.join(home, provider)
117
+
118
+ FileUtils.mkdir_p(parent)
119
+ FileUtils.touch(file)
120
+
121
+ if File.size(file).zero?
122
+ data = LLM::Object.from({})
123
+ File.binwrite file, JSON.pretty_generate(data)
124
+ else
125
+ data = LLM::Object.from JSON.parse(File.read(file))
126
+ end
127
+ data[Dir.getwd] ||= File.join(parent, "#{SecureRandom.uuid}.json")
128
+ end
105
129
 
106
130
  ##
107
131
  # We're ready to start the REPL
108
132
  # This should always succeed unless -p gave garbage
109
- provider = provider.downcase
110
133
  if LLM.respond_to?(provider)
111
134
  key ||= "#{provider.upcase}_API_KEY"
112
135
  if ENV[key].nil? || ENV[key].to_s.empty?
113
136
  warn "llm.rb: set #{key} to use #{provider}"
114
137
  exit 1
115
138
  end
116
- llm = LLM.method(provider).call(key: ENV[key])
117
- agent = LLM::Agent.new(llm, path: temp ? nil : session, tools: LLM::Tool.subclasses)
139
+ llm = LLM.method(provider).call(key: ENV[key])
140
+ path = temp ? nil : data[Dir.getwd]
141
+ agent = LLM::Agent.new(llm, path:, tools: LLM::Tool.subclasses)
118
142
  agent.repl
119
143
  else
120
144
  warn "llm.rb: #{provider} was not recognized"