llm.rb 13.0.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +505 -14
  3. data/README.md +484 -50
  4. data/bin/llm.rb +148 -0
  5. data/data/anthropic.json +206 -263
  6. data/data/bedrock.json +2138 -1860
  7. data/data/deepinfra.json +1003 -624
  8. data/data/deepseek.json +38 -34
  9. data/data/google.json +1079 -371
  10. data/data/mistral.json +448 -368
  11. data/data/moonshot.json +384 -0
  12. data/data/openai.json +974 -1343
  13. data/data/xai.json +154 -126
  14. data/data/zai.json +191 -191
  15. data/lib/llm/agent.rb +123 -20
  16. data/lib/llm/context.rb +71 -88
  17. data/lib/llm/cost.rb +23 -17
  18. data/lib/llm/error.rb +0 -8
  19. data/lib/llm/function/array.rb +3 -3
  20. data/lib/llm/function/async/task.rb +2 -0
  21. data/lib/llm/function/fiber/task.rb +2 -0
  22. data/lib/llm/function/fork/task.rb +2 -0
  23. data/lib/llm/function/ractor/task.rb +2 -0
  24. data/lib/llm/function/sequential/group.rb +4 -1
  25. data/lib/llm/function/sequential/task.rb +1 -1
  26. data/lib/llm/function/task.rb +4 -0
  27. data/lib/llm/function/thread/task.rb +2 -0
  28. data/lib/llm/function.rb +33 -6
  29. data/lib/llm/guard/loop.rb +89 -0
  30. data/lib/llm/guard/null.rb +19 -0
  31. data/lib/llm/guard.rb +61 -0
  32. data/lib/llm/provider.rb +36 -0
  33. data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
  34. data/lib/llm/providers/anthropic.rb +2 -9
  35. data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
  36. data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
  37. data/lib/llm/providers/bedrock.rb +1 -8
  38. data/lib/llm/providers/google/stream_parser.rb +1 -0
  39. data/lib/llm/providers/google.rb +1 -8
  40. data/lib/llm/providers/mistral.rb +1 -1
  41. data/lib/llm/providers/moonshot.rb +76 -0
  42. data/lib/llm/providers/ollama.rb +2 -9
  43. data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
  44. data/lib/llm/providers/openai/responses.rb +7 -9
  45. data/lib/llm/providers/openai/stream_parser.rb +1 -0
  46. data/lib/llm/providers/openai.rb +4 -11
  47. data/lib/llm/repl/bar.rb +4 -3
  48. data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
  49. data/lib/llm/repl/color.rb +78 -0
  50. data/lib/llm/repl/command.rb +12 -5
  51. data/lib/llm/repl/commands/compact.rb +2 -2
  52. data/lib/llm/repl/commands/help.rb +3 -5
  53. data/lib/llm/repl/input/char.rb +46 -0
  54. data/lib/llm/repl/input/row.rb +39 -0
  55. data/lib/llm/repl/input.rb +251 -66
  56. data/lib/llm/repl/markdown/table.rb +11 -3
  57. data/lib/llm/repl/markdown.rb +34 -8
  58. data/lib/llm/repl/node.rb +37 -0
  59. data/lib/llm/repl/status.rb +42 -7
  60. data/lib/llm/repl/stream.rb +18 -6
  61. data/lib/llm/repl/walker.rb +3 -2
  62. data/lib/llm/repl/window.rb +54 -35
  63. data/lib/llm/repl.rb +74 -32
  64. data/lib/llm/skill.rb +20 -4
  65. data/lib/llm/stream.rb +8 -7
  66. data/lib/llm/tool.rb +29 -0
  67. data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
  68. data/lib/llm/tools/git.rb +3 -0
  69. data/lib/llm/tools/mkdir.rb +3 -0
  70. data/lib/llm/tools/rg.rb +3 -0
  71. data/lib/llm/tools/ruby.rb +46 -0
  72. data/lib/llm/tools/shell.rb +3 -0
  73. data/lib/llm/tracer/pretty_logger.rb +127 -0
  74. data/lib/llm/tracer.rb +1 -0
  75. data/lib/llm/transformer/null.rb +21 -0
  76. data/lib/llm/transformer.rb +55 -0
  77. data/lib/llm/version.rb +1 -1
  78. data/lib/llm.rb +12 -2
  79. data/llm.gemspec +9 -2
  80. data/resources/deepdive/advanced/cancellation.md +74 -0
  81. data/resources/deepdive/advanced/compaction.md +83 -0
  82. data/resources/deepdive/advanced/context.md +267 -0
  83. data/resources/deepdive/advanced/guard.md +371 -0
  84. data/resources/deepdive/advanced/tracer.md +180 -0
  85. data/resources/deepdive/advanced/transformer.md +67 -0
  86. data/resources/deepdive/advanced/transports.md +45 -0
  87. data/resources/deepdive/everything_else/audio.md +122 -0
  88. data/resources/deepdive/everything_else/cost.md +99 -0
  89. data/resources/deepdive/everything_else/images.md +89 -0
  90. data/resources/deepdive/everything_else/object.md +108 -0
  91. data/resources/deepdive/everything_else/ocr.md +48 -0
  92. data/resources/deepdive/fundamentals/agents.md +202 -0
  93. data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
  94. data/resources/deepdive/fundamentals/concurrency.md +104 -0
  95. data/resources/deepdive/fundamentals/database.md +449 -0
  96. data/resources/deepdive/fundamentals/embeddings.md +157 -0
  97. data/resources/deepdive/fundamentals/repl.md +87 -0
  98. data/resources/deepdive/fundamentals/schema.md +61 -0
  99. data/resources/deepdive/fundamentals/skills.md +106 -0
  100. data/resources/deepdive/fundamentals/stream.md +110 -0
  101. data/resources/deepdive/fundamentals/tools.md +265 -0
  102. data/resources/deepdive/protocols/a2a.md +106 -0
  103. data/resources/deepdive/protocols/mcp.md +111 -0
  104. data/resources/deepdive.md +58 -1792
  105. metadata +51 -7
  106. data/lib/llm/loop_guard.rb +0 -107
@@ -0,0 +1,449 @@
1
+
2
+ ## Database
3
+ ### Introduction
4
+
5
+ #### Overview
6
+
7
+ Persistence lets an agent outlive a single session. The conversation
8
+ history, model name, and compaction status are serialized as JSON
9
+ that can be stored in a file, a database column, or transmitted
10
+ over a network. Four storage options are available:
11
+
12
+ - **Automatic filesystem persistence**: set `path:` on an agent
13
+ for transparent auto-save after every turn (recommended for
14
+ most file-based use-cases)
15
+ - **Filesystem**: save and restore from a JSON file on disk
16
+ - **ActiveRecord**: persist state in a database column using
17
+ [`acts_as_agent`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord.html#acts_as_agent-instance_method)
18
+ - **Sequel**: persist state in a database column using `plugin :agent`
19
+
20
+ All three use the same serialization mechanism under the hood.
21
+
22
+ #### How it works
23
+
24
+ [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
25
+ implements
26
+ [`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h)
27
+ and
28
+ [`LLM::Context#to_json`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_json)
29
+ for serialization and
30
+ [`LLM::Context#restore`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#restore)
31
+ for deserialization. Save writes the current state, restore loads
32
+ it back and picks up where the conversation left off. The ORM
33
+ wrappers automate this; each
34
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
35
+ or
36
+ [`LLM::Agent#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#ask)
37
+ call persists the
38
+ updated state back to the column automatically.
39
+
40
+ #### Why would I use it?
41
+
42
+ Without persistence, every agent starts with a blank conversation.
43
+ Persistence enables long-running agents that survive process
44
+ restarts, debugging sessions that resume mid-investigation, and
45
+ conversation history that can be queried alongside application
46
+ data.
47
+
48
+ #### Notes
49
+
50
+ The saved JSON can be stored in a file, a database column, or
51
+ transmitted over the network. The ORM integrations use the same
52
+ underlying serialization as filesystem persistence.
53
+
54
+ ### Automatic filesystem persistence
55
+
56
+ #### Overview
57
+
58
+ The [`LLM::Agent#path`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#path)
59
+ attribute provides transparent auto-persistence. Set a file path once
60
+ and the agent restores conversation history from that file on startup
61
+ and saves it back after every
62
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
63
+ or
64
+ [`LLM::Agent#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#ask)
65
+ turn. No manual
66
+ [`LLM::Agent#save`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#save)/
67
+ [`LLM::Agent#restore`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#restore)
68
+ calls needed.
69
+
70
+ The feature is available on subclasses via the class DSL and on
71
+ direct instances via the `path:` keyword argument.
72
+
73
+ #### How it works
74
+
75
+ When a `path` is set, the agent loads existing state from the file
76
+ during initialization. After each turn
77
+ ([`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
78
+ or
79
+ [`LLM::Agent#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#ask)),
80
+ the updated
81
+ state is written back automatically. If the file does not exist yet,
82
+ the agent starts with a blank conversation and creates the file on
83
+ the first save.
84
+
85
+ The path can be set using any of these approaches:
86
+
87
+ ##### Class DSL
88
+
89
+ ```ruby
90
+ class PersistentAgent < LLM::Agent
91
+ set model: "deepseek-v4-pro",
92
+ path: "session.json"
93
+ end
94
+
95
+ # state saved automatically to session.json
96
+ llm = LLM.deepseek(key: ENV["KEY"])
97
+ agent = PersistentAgent.new(llm)
98
+ agent.talk "remember my name is robert"
99
+
100
+ # restored from session.json; prints "robert"
101
+ agent = PersistentAgent.new(llm)
102
+ agent.talk "what's my name?"
103
+ ```
104
+
105
+ ##### Keyword argument
106
+
107
+ ```ruby
108
+ # state saved automatically
109
+ llm = LLM.deepseek(key: ENV["KEY"])
110
+ agent = LLM::Agent.new(llm, path: "session.json", stream: $stdout)
111
+ agent.talk "remember my name is robert"
112
+ ```
113
+
114
+ ##### Lazy resolution
115
+
116
+ ```ruby
117
+ class DynamicAgent < LLM::Agent
118
+ set path: -> { "sessions/#{name}.json" }
119
+ end
120
+ ```
121
+
122
+ #### Why would I use it?
123
+
124
+ Auto-persistence eliminates boilerplate. You set the path once and
125
+ forget about serialization entirely. The agent picks up where it
126
+ left off across process restarts, REPL sessions, or debugging runs
127
+ without a single
128
+ [`LLM::Agent#save`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#save)
129
+ or
130
+ [`LLM::Agent#restore`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#restore)
131
+ call in your code.
132
+
133
+ #### Notes
134
+
135
+ The auto-path feature is a strict superset of manual filesystem
136
+ persistence. If you need fine-grained control over when state is
137
+ saved (e.g. batch several turns before persisting), use the manual
138
+ [`LLM::Agent#save`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#save)
139
+ and
140
+ [`LLM::Agent#restore`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#restore)
141
+ methods described in the Filesystem section below. The `path`
142
+ attribute delegates to the same underlying serialization.
143
+
144
+ ### Filesystem
145
+
146
+ #### Overview
147
+
148
+ A conversation that ends when the process exits is not very useful.
149
+ Serialization saves the context (message history, model name,
150
+ compaction status) as JSON that can be stored in a string, a
151
+ file, or a database column. Restore it later, in a different
152
+ process or on a different machine, and pick up where you left off.
153
+
154
+ #### How it works
155
+
156
+ [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
157
+ implements
158
+ [`LLM::Context#to_h`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_h)
159
+ and
160
+ [`LLM::Context#to_json`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#to_json)
161
+ for serialization
162
+ and
163
+ [`LLM::Context#restore`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#restore)
164
+ for deserialization. The serialized state includes
165
+ the message history, model name, and compaction status. Save and
166
+ restore work with file paths or in-memory strings. You can also
167
+ serialize to a JSON string for database storage or network
168
+ transmission:
169
+
170
+ ##### Save to a file
171
+
172
+ ```ruby
173
+ llm = LLM.deepseek(key: ENV["KEY"])
174
+ agent = LLM::Agent.new(llm)
175
+ agent.talk "remember my name is robert"
176
+ agent.save(path: "agent.json")
177
+ ```
178
+
179
+ ##### Restore from a file
180
+
181
+ ```ruby
182
+ llm = LLM.deepseek(key: ENV["KEY"])
183
+ agent = LLM::Agent.new(llm, stream: $stdout)
184
+ agent.restore(path: "agent.json")
185
+ agent.talk "what's my name?"
186
+ ```
187
+
188
+ ##### Serialize to a JSON string
189
+
190
+ ```ruby
191
+ llm = LLM.deepseek(key: ENV["KEY"])
192
+ agent = LLM::Agent.new(llm)
193
+ agent.talk "remember my name is robert"
194
+ json = agent.to_json
195
+ agent = LLM::Agent.new(llm)
196
+ agent.restore(string: json)
197
+ ```
198
+
199
+ #### Why would I use it?
200
+
201
+ Filesystem persistence is the foundation for long-running agents
202
+ that outlive a single session. Save a debugging session
203
+ mid-investigation and resume it tomorrow. Store a completed
204
+ conversation as evidence or training data. Pass context between
205
+ processes, between machines, or through a job queue.
206
+
207
+ #### Notes
208
+
209
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
210
+ delegates serialization to its internal context. The
211
+ saved JSON can be stored in a file, a database column, or
212
+ transmitted over the network.
213
+
214
+ ### ActiveRecord
215
+
216
+ #### Overview
217
+
218
+ ActiveRecord models use
219
+ [`LLM::ActiveRecord#acts_as_agent`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord.html#acts_as_agent-instance_method)
220
+ to install the agent wrapper. The method accepts a block for
221
+ configuring agent defaults
222
+ and an options hash for storage format. Most defaults such as
223
+ `model`, `tools`, `instructions`, `schema`, and `concurrency` can
224
+ be set through the block. Tracer and stream can also be set here
225
+ rather than through legacy convention methods.
226
+
227
+ #### How it works
228
+
229
+ When you want to add agent persistence to an ActiveRecord model,
230
+ call
231
+ [`LLM::ActiveRecord#acts_as_agent`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord.html#acts_as_agent-instance_method)
232
+ in the model class. The `data` column stores
233
+ the full agent state (conversation history, model name, compaction
234
+ status) as JSON. On first call, a fresh agent is created and the
235
+ conversation starts from scratch.
236
+
237
+ On subsequent calls, the stored state is restored and the
238
+ conversation continues. Every
239
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
240
+ or
241
+ [`LLM::Agent#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#ask)
242
+ call persists the
243
+ updated state back to the column automatically.
244
+
245
+ The `data_column:` option lets you use a different column name.
246
+ The `format:` option controls the storage type. Use `:string` for
247
+ a text column (any database) or `:jsonb` for a native PostgreSQL
248
+ JSONB column (recommended for PostgreSQL). For `:jsonb`,
249
+ ActiveRecord handles JSON typecasting automatically.
250
+
251
+ The block yields an
252
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
253
+ instance. Set agent defaults in the block using the
254
+ [`LLM::Agent.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#set-class_method)
255
+ method or individual accessors. Anything you can set on an agent
256
+ can be
257
+ configured here -- model, tools, instructions, schema, concurrency,
258
+ tracer, stream, and confirm:
259
+
260
+ Note that `tracer` and `stream` are set directly in the block
261
+ rather than through legacy `set_tracer` / `set_context` convention
262
+ methods. The block style is the recommended approach for all agent
263
+ configuration. Only `set_provider` is required as a private method.
264
+
265
+ ##### Migration with TEXT column
266
+
267
+ ```ruby
268
+ class CreateAgents < ActiveRecord::Migration[7.1]
269
+ def change
270
+ create_table :agents do |t|
271
+ t.string :name
272
+ t.text :data # stores the serialized agent state
273
+ t.timestamps
274
+ end
275
+ end
276
+ end
277
+ ```
278
+
279
+ ##### Migration with JSONB column
280
+
281
+ ```ruby
282
+ class CreateAgents < ActiveRecord::Migration[7.1]
283
+ def change
284
+ create_table :agents do |t|
285
+ t.string :name
286
+ t.jsonb :data # native JSON storage, supports indexing
287
+ t.timestamps
288
+ end
289
+ end
290
+ end
291
+ ```
292
+
293
+ ##### Model setup
294
+
295
+ ```ruby
296
+ require "active_record"
297
+ require "llm"
298
+ require "llm/active_record"
299
+
300
+ class Agent < ApplicationRecord
301
+ acts_as_agent(format: :jsonb) do |agent|
302
+ agent.model "deepseek-v4-pro"
303
+ agent.instructions "solve the user's query"
304
+ agent.tools [Research, FinalizeResearch, ActOnResearch]
305
+ agent.tracer LLM::Tracer::Logger.new(llm, io: $stdout)
306
+ end
307
+
308
+ private
309
+
310
+ def set_provider
311
+ LLM.deepseek(key: ENV["KEY"])
312
+ end
313
+ end
314
+
315
+ agent = Agent.create!(name: "researcher")
316
+ agent.talk "perform research"
317
+ # state is saved to the data column automatically
318
+ ```
319
+
320
+ #### Why would I use it?
321
+
322
+ The ActiveRecord wrapper integrates with your existing models.
323
+ Agent state is automatically persisted after each conversation turn
324
+ using the same `create!`, `save!`, and query methods you already
325
+ use.
326
+
327
+ #### Notes
328
+
329
+ The `format` option defaults to `:string`. Use `:json` or `:jsonb`
330
+ for PostgreSQL. JSONB is recommended; it supports indexing and
331
+ is more efficient for querying. The `data_column` option defaults
332
+ to `:data` -- create a migration to add a `TEXT` or `JSONB` column
333
+ to your table.
334
+
335
+ The model requires a `set_provider` private method that returns an
336
+ [`LLM::Provider`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html)
337
+ instance. Legacy `set_context` and `set_tracer` convention methods
338
+ also work for backwards compatibility, but the block style
339
+ (`agent.tracer ...`, `agent.stream ...`) is preferred for all
340
+ agent-level configuration.
341
+
342
+ ### Sequel
343
+
344
+ #### Overview
345
+
346
+ Sequel models use `plugin :agent` to install the agent wrapper.
347
+ The plugin accepts the same options as
348
+ [`acts_as_agent`](https://r.uby.dev/api-docs/llm.rb/LLM/ActiveRecord.html#acts_as_agent-instance_method)
349
+ and follows the same conventions for provider resolution and state
350
+ persistence.
351
+ On first call, a fresh agent is created and the conversation starts
352
+ from scratch. On subsequent calls, the stored state is restored and
353
+ the conversation continues automatically.
354
+
355
+ #### How it works
356
+
357
+ When you want to add agent persistence to a Sequel model, register
358
+ the plugin in the model class. The `data` column stores
359
+ the full agent state as JSON, same structure as ActiveRecord.
360
+ On first call, a fresh agent is created. On subsequent calls,
361
+ the stored state is restored and the conversation continues.
362
+ Every
363
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
364
+ or
365
+ [`LLM::Agent#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#ask)
366
+ call persists the updated state.
367
+
368
+ The `data_column:` and `format:` options work identically to
369
+ ActiveRecord. For `:jsonb`, Sequel loads the `pg_json` extension
370
+ automatically and handles JSON typecasting. The block yields an
371
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
372
+ instance. Configure agent defaults in the block -- model, tools,
373
+ instructions, tracer, stream, and confirm all go here:
374
+
375
+ As with ActiveRecord, `tracer` and `stream` are configured in the
376
+ block rather than through legacy convention methods. The block
377
+ style is the recommended approach.
378
+
379
+ ##### Migration with TEXT column
380
+
381
+ ```ruby
382
+ DB.create_table :agents do
383
+ primary_key :id
384
+ String :name
385
+ String :data # stores the serialized agent state
386
+ DateTime :created_at
387
+ DateTime :updated_at
388
+ end
389
+ ```
390
+
391
+ ##### Migration with JSONB column
392
+
393
+ ```ruby
394
+ DB.create_table :agents do
395
+ primary_key :id
396
+ String :name
397
+ column :data, :jsonb # native JSON storage
398
+ DateTime :created_at
399
+ DateTime :updated_at
400
+ end
401
+ ```
402
+
403
+ ##### Model setup
404
+
405
+ ```ruby
406
+ require "sequel"
407
+ require "llm"
408
+ require "llm/sequel/plugin"
409
+
410
+ class Agent < Sequel::Model
411
+ plugin(:agent, format: :jsonb) do |agent|
412
+ agent.model "deepseek-v4-pro"
413
+ agent.instructions "solve the user's query"
414
+ agent.tools [Research, FinalizeResearch, ActOnResearch]
415
+ agent.tracer LLM::Tracer::Logger.new(llm, io: $stdout)
416
+ end
417
+
418
+ private
419
+
420
+ def set_provider
421
+ LLM.deepseek(key: ENV["KEY"])
422
+ end
423
+ end
424
+
425
+ agent = Agent.create(name: "researcher")
426
+ agent.talk "perform research"
427
+ # state is saved to the data column automatically
428
+ ```
429
+
430
+ #### Why would I use it?
431
+
432
+ The Sequel plugin integrates with your existing models.
433
+ It follows the same conventions as ActiveRecord, so
434
+ switching between the two requires minimal code changes.
435
+
436
+ #### Notes
437
+
438
+ The `format` option defaults to `:string`. Use `:json` or `:jsonb`
439
+ for PostgreSQL. Sequel loads the `pg_json` extension automatically
440
+ and wraps JSON values for native storage. JSONB is recommended.
441
+ The `data_column` option defaults to `:data`.
442
+
443
+ The model requires a `set_provider` private method that returns an
444
+ [`LLM::Provider`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html)
445
+ instance. Legacy `set_context` and `set_tracer` convention methods
446
+ also work for backwards compatibility, but configuring tracer,
447
+ stream, and other agent options in the block is the preferred
448
+ approach.
449
+
@@ -0,0 +1,157 @@
1
+
2
+ ## Embeddings
3
+
4
+ ### Introduction
5
+
6
+ #### Overview
7
+
8
+ Embeddings convert text into numerical vectors that capture semantic
9
+ meaning. Similar texts produce vectors that are close together in the
10
+ vector space, enabling semantic search, clustering, and classification.
11
+ Most providers offer embedding models through an
12
+ [`LLM::Provider#embed`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#embed)
13
+ method.
14
+
15
+ llm.rb also includes support for OpenAI's vector store API, which
16
+ provides a managed vector database as an HTTP service.
17
+
18
+ #### How it works
19
+
20
+ When you want to generate an embedding vector for a piece of text,
21
+ call
22
+ [`LLM::Provider#embed`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#embed)
23
+ on any provider that supports it. The method accepts an input string
24
+ or array of strings and returns a response with the resulting
25
+ embeddings, which can be stored in a vector database and used for
26
+ similarity search at query time:
27
+
28
+ ```ruby
29
+ require "llm"
30
+
31
+ llm = LLM.openai(key: ENV["KEY"])
32
+ body = "llm.rb is Ruby's capable AI runtime."
33
+ embedding = llm.embed([body]).embeddings.first
34
+ ```
35
+
36
+ #### Why would I use it?
37
+
38
+ Embeddings are the foundation for Retrieval-Augmented Generation (RAG).
39
+ Store embeddings for your documents, then retrieve the most relevant
40
+ chunks at query time and feed them into the model as context. This
41
+ pattern is one of the most common AI workflows.
42
+
43
+ #### Notes
44
+
45
+ Not all providers support embeddings. The following providers have
46
+ an `embed` method:
47
+
48
+ - OpenAI (`text-embedding-3-small`)
49
+ - DeepInfra (`BAAI/bge-m3`)
50
+ - Ollama (`qwen3:latest`)
51
+ - Google (`gemini-embedding-2`)
52
+ - Mistral (`mistral-embed`)
53
+
54
+ ### Vector stores (OpenAI)
55
+
56
+ #### Overview
57
+
58
+ OpenAI's vector store API provides a managed vector database as an
59
+ HTTP service. You upload files, add them to a vector store, and
60
+ search across them. The runtime handles chunking, embedding, and
61
+ indexing on OpenAI's side.
62
+
63
+ #### How it works
64
+
65
+ When you want to create a vector store, upload files, and search
66
+ across them, you can call methods on
67
+ [`LLM::OpenAI::VectorStores`](https://r.uby.dev/api-docs/llm.rb/LLM/OpenAI/VectorStores.html).
68
+ Upload files first, then create the store with
69
+ [`LLM::OpenAI::VectorStores#create_and_poll`](https://r.uby.dev/api-docs/llm.rb/LLM/OpenAI/VectorStores.html#create_and_poll-instance_method)
70
+ to
71
+ wait until indexing completes:
72
+
73
+ ```ruby
74
+ llm = LLM.openai(key: ENV["KEY"])
75
+
76
+ # Upload files first
77
+ files = %w[report.pdf manual.pdf].map { |f| llm.files.create(file: f) }
78
+
79
+ # Create a vector store with the files and wait for indexing
80
+ store = llm.vector_stores.create_and_poll(
81
+ name: "documents",
82
+ file_ids: files.map(&:id)
83
+ )
84
+
85
+ # Search across the store
86
+ results = llm.vector_stores.search(
87
+ vector: store,
88
+ query: "What is the deadline?",
89
+ max_results: 5
90
+ )
91
+ results.each do |chunk|
92
+ text = chunk.content.map { |c| c.text }.join
93
+ puts "[score=#{chunk.score}] #{text[0..120]}..."
94
+ end
95
+ ```
96
+
97
+ #### Why would I use it?
98
+
99
+ OpenAI's vector store API removes the operational overhead of running
100
+ your own vector database. There is no infrastructure to manage, no
101
+ indexing pipeline to maintain, and no embedding model to configure.
102
+ The trade-off is vendor lock-in and per-request pricing.
103
+
104
+ #### Notes
105
+
106
+ Vector stores require the `openai` provider. Other providers such as
107
+ DeepInfra, Google, and Mistral expose an embedding method but no
108
+ vector store API. For a self-hosted vector database, pair
109
+ [`LLM::Provider#embed`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#embed)
110
+ with sqlite-vec or pgvector.
111
+
112
+ ### Local vector search
113
+
114
+ #### Overview
115
+
116
+ Local vector search combines
117
+ [`LLM::Provider#embed`](https://r.uby.dev/api-docs/llm.rb/LLM/Provider.html#embed)
118
+ with a local vector database such as
119
+ [sqlite-vec](https://github.com/asg017/sqlite-vec) or
120
+ [pgvector](https://github.com/pgvector/pgvector). This approach keeps
121
+ your data on your own infrastructure.
122
+
123
+ #### How it works
124
+
125
+ When you want to store embeddings in a local database and search
126
+ them later, generate vectors with any provider, store the embedding
127
+ alongside the content, and query with a similarity search:
128
+
129
+ ```ruby
130
+ require "llm"
131
+
132
+ llm = LLM.deepseek(key: ENV["KEY"])
133
+
134
+ # Store: embed text and save it with a vector column
135
+ body = "llm.rb is Ruby's capable AI runtime."
136
+ embedding = llm.embed([body]).embeddings.first
137
+ Document.create!(title: "llm.rb", body:, embedding:)
138
+
139
+ # Query: embed the question and find similar documents
140
+ query = "What is llm.rb?"
141
+ query_embedding = llm.embed([query]).embeddings.first
142
+ results = Document.order(Arel.sql("embedding <=> ?", query_embedding)).limit(5)
143
+ ```
144
+
145
+ #### Why would I use it?
146
+
147
+ Local vector search gives you full control over your data, no
148
+ per-query costs beyond your embedding provider, and the ability
149
+ to use any LLM provider for the generation step. It pairs well
150
+ with local providers like Ollama for a fully self-hosted stack.
151
+
152
+ #### Notes
153
+
154
+ The exact query syntax depends on your vector database. sqlite-vec
155
+ uses `vec_distance_L2` for Euclidean distance, while pgvector uses
156
+ the `<=>` operator for cosine distance. Check your database
157
+ documentation for the correct similarity function.
@@ -0,0 +1,87 @@
1
+
2
+ ## REPL
3
+
4
+ ### Introduction
5
+
6
+ #### Overview
7
+
8
+ The REPL drops you into a curses-based TUI for talking to an agent
9
+ interactively. It has a scrollable transcript that renders markdown,
10
+ a multi-line input area, and a status bar showing context usage and
11
+ cost. The UI thread stays responsive while a second thread communicates
12
+ with the model. Think of it as `binding.pry` but for agents.
13
+
14
+ #### How it works
15
+
16
+ The REPL runs on two threads: one for the curses UI (input handling,
17
+ transcript rendering, status bar) and one for model communication.
18
+
19
+ The `name:` option labels the agent in the prompt. The `path:`
20
+ option persists state across sessions. The `tools:` option attaches
21
+ extra tools for the session.
22
+
23
+ Commands start with `/` and are dispatched to registered
24
+ [`LLM::Command`](https://r.uby.dev/api-docs/llm.rb/LLM/Repl/Command.html)
25
+ subclasses. Type `/compact` to free context window
26
+ space, `/exit` to leave.
27
+
28
+ When characters arrive faster than a threshold, the REPL detects
29
+ paste mode. In paste mode, pressing Enter inserts a newline instead
30
+ of submitting.
31
+
32
+ Start a session with:
33
+
34
+ ```ruby
35
+ require "llm"
36
+ require "llm/tools"
37
+
38
+ llm = LLM.deepseek(key: ENV["KEY"])
39
+ agent = LLM::Agent.new(llm, name: "my-agent", path: "session.json")
40
+ agent.repl(tools: LLM::Tool.subclasses)
41
+ ```
42
+
43
+ #### Why would I use it?
44
+
45
+ The REPL gives you an interactive environment to test agents, debug
46
+ tool calls, and inspect conversation state without writing a
47
+ separate UI. Drop in after running an agent to confirm it did what
48
+ you expected. Inspect what went wrong when it did not. Keep talking
49
+ to the same agent while its state is still intact. It is
50
+ `binding.pry` but for agents.
51
+
52
+ #### Notes
53
+
54
+ The REPL requires the `curses` and `kramdown` gems. By default the
55
+ tracer is disabled during the session. Set `tracer: true` to keep
56
+ it active.
57
+
58
+ Commands use the same vocabulary as tools: declare a name,
59
+ description, and parameters with `parameter` and `required`.
60
+ Subclassing an existing command inherits its name, description,
61
+ and parameters. This is how `/quit` is an alias of `/exit`.
62
+
63
+ ```ruby
64
+ class Greeter < LLM::Command
65
+ name "greet"
66
+ description "Greets the given name"
67
+ parameter :name, String, "The person's name"
68
+ required %i[name]
69
+
70
+ def call(name:)
71
+ write "Welcome #{name}!\n"
72
+ end
73
+ end
74
+ ```
75
+
76
+ The input area supports several keyboard shortcuts:
77
+
78
+ | Key | Action |
79
+ |---|---|
80
+ | `Enter` | Submit the current prompt |
81
+ | `Ctrl+A` | Jump to the start of the line |
82
+ | `Ctrl+E` | Jump to the end of the line |
83
+ | `Ctrl+P` / `Ctrl+N` | Recall previous / next user message |
84
+ | `Tab` | Complete `/command` names |
85
+ | `Esc` | Cancel the current request |
86
+ | `Up` / `Down` | Scroll the transcript one line |
87
+ | `PgUp` / `PgDn` | Scroll the transcript by one page |