llm.rb 12.5.1 → 13.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +482 -0
  3. data/LICENSE +21 -93
  4. data/README.md +49 -159
  5. data/data/deepinfra.json +3 -0
  6. data/data/xai.json +1 -1
  7. data/lib/llm/a2a.rb +1 -1
  8. data/lib/llm/active_record/acts_as_agent.rb +32 -0
  9. data/lib/llm/active_record/acts_as_llm.rb +6 -6
  10. data/lib/llm/agent.rb +101 -26
  11. data/lib/llm/buffer.rb +85 -3
  12. data/lib/llm/compactor/null.rb +19 -0
  13. data/lib/llm/compactor/truncate.rb +80 -0
  14. data/lib/llm/compactor.rb +42 -124
  15. data/lib/llm/context.rb +33 -37
  16. data/lib/llm/contract.rb +4 -25
  17. data/lib/llm/function/array.rb +15 -14
  18. data/lib/llm/function/async/group.rb +54 -0
  19. data/lib/llm/function/async/reactor.rb +48 -0
  20. data/lib/llm/function/async/task.rb +83 -0
  21. data/lib/llm/function/fiber/group.rb +46 -0
  22. data/lib/llm/function/fiber/task.rb +62 -0
  23. data/lib/llm/function/{fork_group.rb → fork/group.rb} +14 -5
  24. data/lib/llm/function/fork/job.rb +4 -3
  25. data/lib/llm/function/fork/task.rb +20 -10
  26. data/lib/llm/function/group.rb +40 -0
  27. data/lib/llm/function/{ractor_group.rb → ractor/group.rb} +13 -5
  28. data/lib/llm/function/ractor/job.rb +19 -3
  29. data/lib/llm/function/ractor/mailbox.rb +9 -0
  30. data/lib/llm/function/ractor/task.rb +24 -15
  31. data/lib/llm/function/{call_group.rb → sequential/group.rb} +17 -8
  32. data/lib/llm/function/sequential/task.rb +49 -0
  33. data/lib/llm/function/task.rb +25 -37
  34. data/lib/llm/function/thread/group.rb +46 -0
  35. data/lib/llm/function/thread/task.rb +60 -0
  36. data/lib/llm/function/tracing.rb +2 -0
  37. data/lib/llm/function.rb +56 -64
  38. data/lib/llm/loop_guard.rb +1 -2
  39. data/lib/llm/mcp.rb +22 -0
  40. data/lib/llm/object.rb +2 -1
  41. data/lib/llm/provider.rb +6 -3
  42. data/lib/llm/providers/google.rb +2 -2
  43. data/lib/llm/repl/command.rb +35 -8
  44. data/lib/llm/repl/commands/compact.rb +33 -0
  45. data/lib/llm/repl/input.rb +80 -15
  46. data/lib/llm/repl/markdown/table.rb +76 -0
  47. data/lib/llm/repl/markdown.rb +31 -1
  48. data/lib/llm/repl/status.rb +1 -1
  49. data/lib/llm/repl/stream.rb +10 -3
  50. data/lib/llm/repl/transcript.rb +1 -1
  51. data/lib/llm/repl/walker.rb +46 -0
  52. data/lib/llm/repl.rb +18 -12
  53. data/lib/llm/response.rb +10 -0
  54. data/lib/llm/schema/leaf.rb +5 -0
  55. data/lib/llm/schema/object.rb +11 -5
  56. data/lib/llm/sequel/agent.rb +32 -0
  57. data/lib/llm/sequel/plugin.rb +6 -6
  58. data/lib/llm/stream.rb +24 -17
  59. data/lib/llm/tool/param.rb +12 -0
  60. data/lib/llm/tool.rb +20 -4
  61. data/lib/llm/tools/chdir.rb +0 -2
  62. data/lib/llm/tools/git.rb +8 -4
  63. data/lib/llm/tools/mkdir.rb +1 -1
  64. data/lib/llm/tools/pwd.rb +0 -2
  65. data/lib/llm/tools/read_file.rb +0 -2
  66. data/lib/llm/tools/rg.rb +8 -4
  67. data/lib/llm/tools/shell.rb +8 -4
  68. data/lib/llm/tools/utils.rb +31 -0
  69. data/lib/llm/version.rb +1 -1
  70. data/lib/llm.rb +25 -5
  71. data/llm.gemspec +3 -3
  72. data/resources/deepdive.md +693 -57
  73. metadata +24 -13
  74. data/lib/llm/function/call_task.rb +0 -46
  75. data/lib/llm/function/fiber_group.rb +0 -105
  76. data/lib/llm/function/task_group.rb +0 -97
  77. data/lib/llm/function/thread_group.rb +0 -102
@@ -43,6 +43,12 @@ features that didn't make it into the homepage documentation.
43
43
 
44
44
  - [LLM::Tool](#llmtool)
45
45
  - [Errors](#errors)
46
+ - [Confirmation](#confirmation)
47
+ - [Manual tool loop](#manual-tool-loop)
48
+ - [Executing](#executing)
49
+ - [Per-tool confirmation](#per-tool-confirmation)
50
+ - [Full loop](#full-loop)
51
+ - [Trade-offs](#trade-offs)
46
52
  </details>
47
53
 
48
54
  <details>
@@ -67,10 +73,34 @@ features that didn't make it into the homepage documentation.
67
73
  - [LLM::Stream](#llmstream)
68
74
  </details>
69
75
 
76
+ <details>
77
+ <summary>Concurrency</summary>
78
+
79
+ - [Overview](#overview)
80
+ - [sequential](#sequential)
81
+ - [thread](#thread)
82
+ - [fiber](#fiber)
83
+ - [async](#async)
84
+ - [fork](#fork)
85
+ - [ractor](#ractor)
86
+ - [Quick reference](#quick-reference)
87
+ </details>
88
+
89
+ <details>
90
+ <summary>Context Compaction</summary>
91
+
92
+ - [Configuration](#configuration)
93
+ - [Standalone usage](#standalone-usage)
94
+ - [Strategies](#strategies)
95
+ - [Manual compaction](#manual-compaction)
96
+ - [Lifecycle callbacks](#lifecycle-callbacks)
97
+ </details>
98
+
70
99
  <details>
71
100
  <summary>Cancellation</summary>
72
101
 
73
102
  - [Cancel a request](#cancel-a-request)
103
+ - [Tool interrupts](#tool-interrupts)
74
104
  </details>
75
105
 
76
106
  <details>
@@ -92,7 +122,7 @@ features that didn't make it into the homepage documentation.
92
122
  <summary>REPL</summary>
93
123
 
94
124
  - [LLM::Agent](#llmagent)
95
- - [State](#state)
125
+ - [Persistence](#persistence)
96
126
  - [Tools](#tools)
97
127
  - [Skills](#skills-1)
98
128
  - [Tracer](#tracer-1)
@@ -133,6 +163,12 @@ features that didn't make it into the homepage documentation.
133
163
  - [translation](#translation)
134
164
  </details>
135
165
 
166
+ <details>
167
+ <summary>OCR</summary>
168
+
169
+ - [Mistral](#mistral)
170
+ </details>
171
+
136
172
  **Protocols**
137
173
 
138
174
  <details>
@@ -157,7 +193,7 @@ class, and it is built on top of
157
193
  [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html) -
158
194
  the heart of the runtime. An agent manages the tool loop automatically,
159
195
  implements a tool loop guard for misbehaving models, and
160
- it can use five different concurrency strategies to execute
196
+ it can use six different concurrency strategies to execute
161
197
  tools.
162
198
 
163
199
  An agent can be a subclass of
@@ -188,11 +224,11 @@ be simpler without it.
188
224
 
189
225
  ```ruby
190
226
  class Agent < LLM::Agent
191
- model "deepseek-v4-pro"
192
- tools { [DoResearch, FinalizeResearch, ActOnResearch] }
193
- stream { $stdout }
194
- tracer :set_tracer
195
- concurrency :fork
227
+ set model: "deepseek-v4-pro",
228
+ tools: [DoResearch, FinalizeResearch, ActOnResearch],
229
+ stream: -> { $stdout },
230
+ tracer: :set_tracer,
231
+ concurrency: :fork
196
232
 
197
233
  def research!
198
234
  talk "start the research"
@@ -248,6 +284,14 @@ The model then proceeds to process the tool's response,
248
284
  and then might generate its own response, or perhaps call
249
285
  another tool.
250
286
 
287
+ There is exactly one rule: a tool call must always produce
288
+ a tool response. If a tool raises an exception, the runtime
289
+ rescues it and returns a structured error to the model
290
+ instead. The conversation never enters an invalid state
291
+ because of a crashed tool &mdash; the model always has
292
+ something to work with. This is by design. Keeping the
293
+ tool loop alive is the highest priority.
294
+
251
295
  #### LLM::Tool
252
296
 
253
297
  A tool can be defined by subclassing
@@ -262,7 +306,7 @@ serve a user's query.
262
306
  require "llm"
263
307
  require "shellwords"
264
308
 
265
- class Shell < LLM::Shell
309
+ class Shell < LLM::Tool
266
310
  name "shell"
267
311
  description "execute a shell command"
268
312
  parameter :name, String, "the command's name"
@@ -270,8 +314,8 @@ class Shell < LLM::Shell
270
314
  required %i[name]
271
315
  defaults arguments: []
272
316
 
273
- def call(name:, arguments:)
274
- out = `#{name.shellscape} #{arguments.map(&:shellescape).join(" ")}`
317
+ def call(name:, arguments: [])
318
+ out = `#{name.shellescape} #{arguments.map(&:shellescape).join(" ")}`
275
319
  {ok: $?.success?, out:}
276
320
  end
277
321
  end
@@ -283,15 +327,9 @@ agent.talk "What files are in the current working directory?"
283
327
 
284
328
  #### Errors
285
329
 
286
- Exceptions that might be raised by a tool are automatically
287
- rescued and returned to the model as a structured error.
288
- Otherwise &ndash; the conversation's history could be left
289
- in an invalid state.
290
-
291
- That's because a tool call must complete with a tool response,
292
- that's the only valid response a model expects, so even in the
293
- case of an error, something must be returned that communicates
294
- what happened.
330
+ Exceptions raised by a tool are automatically rescued and
331
+ returned to the model as a structured error. The model sees
332
+ something like this:
295
333
 
296
334
  ```ruby
297
335
  class Error < LLM::Tool
@@ -307,6 +345,196 @@ class Error < LLM::Tool
307
345
  end
308
346
  ```
309
347
 
348
+ The runtime wraps the exception into `{error: true, kind: "RuntimeError",
349
+ message: "boom"}` and returns it to the model as the tool response. From
350
+ the model's perspective the tool completed &mdash; it just completed with
351
+ an error. The model can read the error, decide what went wrong, and try
352
+ something else. The conversation stays valid.
353
+
354
+ You can also handle errors yourself inside `call`. Rescue the exception
355
+ and return whatever shape makes sense for your tool:
356
+
357
+ ```ruby
358
+ class Shell < LLM::Tool
359
+ name "shell"
360
+ description "execute a shell command"
361
+
362
+ def call(name:, arguments: [])
363
+ out = `#{name} #{arguments.join(" ")}`
364
+ {ok: $?.success?, out:}
365
+ rescue Errno::ENOENT
366
+ {ok: false, error: "command not found: #{name}"}
367
+ end
368
+ end
369
+ ```
370
+
371
+ The model receives `{ok: false, error: "command not found: ls"}` and can
372
+ react accordingly &mdash; maybe it corrects the command name and tries
373
+ again. This is often better than letting the runtime's generic error
374
+ wrapper speak for you, because you can provide domain-specific detail
375
+ that helps the model recover.
376
+
377
+ The principle is the same either way: **return something**. A tool call
378
+ must complete with a tool response. If you don't return a value, and you
379
+ don't raise, the runtime has nothing to send back and the conversation
380
+ is stuck. As long as you return a Hash (or anything the model can
381
+ interpret), the tool loop continues.
382
+
383
+ #### Confirmation
384
+
385
+ Tools that perform destructive actions can be gated behind
386
+ explicit confirmation. List their names in
387
+ [`LLM::Agent.confirm`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#confirm-class_method)
388
+ to block execution until you override `on_tool_confirmation`.
389
+
390
+ The default handler cancels the tool. Override it per-agent to
391
+ prompt the user, log the decision, or auto-approve certain tools.
392
+
393
+ ```ruby
394
+ class AdminAgent < LLM::Agent
395
+ set confirm: %w[delete destroy shutdown]
396
+
397
+ def on_tool_confirmation(fn, strategy)
398
+ print "Run #{fn.name} with #{fn.arguments}? [y/N] "
399
+ $stdin.gets&.match?(/\Ay\z/i) ? wait(strategy) : fn.cancel
400
+ end
401
+ end
402
+
403
+ llm = LLM.deepseek(key: ENV["KEY"])
404
+ agent = AdminAgent.new(llm)
405
+ ```
406
+
407
+ Confirmation also accepts a Symbol for lazy resolution, which
408
+ allows the list of confirmed tools to change per-instance:
409
+
410
+ ```ruby
411
+ class AdaptiveAgent < LLM::Agent
412
+ set confirm: :tools_that_need_confirmation
413
+
414
+ def tools_that_need_confirmation
415
+ some_condition ? %w[delete destroy] : %w[delete]
416
+ end
417
+ end
418
+ ```
419
+
420
+ ## Manual tool loop
421
+
422
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) manages
423
+ the tool loop automatically — it calls the model, checks for tool calls,
424
+ executes them, sends results back, and repeats until the model responds
425
+ with text. You can bypass this and drive the loop yourself for finer
426
+ control.
427
+
428
+ Start with a bare [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
429
+ (no agent):
430
+
431
+ ```ruby
432
+ require "llm"
433
+
434
+ llm = LLM.deepseek(key: ENV["KEY"])
435
+ ctx = LLM::Context.new(llm)
436
+ ```
437
+
438
+ Send a message and check whether the model wants to call tools:
439
+
440
+ ```ruby
441
+ res = ctx.talk "What's the weather in Tokyo?"
442
+
443
+ if ctx.pending_functions?
444
+ puts "Model requested #{ctx.pending_functions.size} tool(s)"
445
+ else
446
+ puts res.text
447
+ end
448
+ ```
449
+
450
+ ### Executing
451
+
452
+ When the model asks to call tools, call `ctx.wait(:strategy)` to run
453
+ them. `ctx.wait` picks up pending functions, spawns them, waits for
454
+ results, and records them back in the context. Every strategy is
455
+ supported: `:sequential`, `:thread`, `:fiber`, `:async`, `:fork`,
456
+ and `:ractor`.
457
+
458
+ ```ruby
459
+ require "llm"
460
+
461
+ llm = LLM.deepseek(key: ENV["KEY"])
462
+ ctx = LLM::Context.new(llm)
463
+
464
+ ctx.talk("What's the weather in Tokyo?")
465
+ ctx.talk ctx.wait(:thread)
466
+ ```
467
+
468
+ ### Per-tool confirmation
469
+
470
+ Because `pending_functions` returns a regular array, you can inspect
471
+ each function before execution. Call `ctx.wait(:thread)` to execute
472
+ all pending tools — but you can also selectively exclude functions or
473
+ run individual ones through `fn.task(:thread).wait` for ad-hoc execution
474
+ that bypasses guards and streaming hooks:
475
+
476
+ ```ruby
477
+ results = ctx.pending_functions.map do |fn|
478
+ print "Run #{fn.name} with #{fn.arguments}? [y/N] "
479
+ if $stdin.gets&.match?(/\Ay\z/i)
480
+ fn.task(:thread).wait
481
+ else
482
+ fn.cancel(reason: "user declined")
483
+ end
484
+ end
485
+ ctx.talk(results)
486
+ ```
487
+
488
+ This pattern is what
489
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)'s
490
+ built-in confirmation feature uses internally, but doing it manually
491
+ gives you full control — route the decision through a web socket, a
492
+ background job, or a multi-user approval flow.
493
+
494
+ ### Full loop
495
+
496
+ A complete manual tool loop looks like this:
497
+
498
+ ```ruby
499
+ require "llm"
500
+
501
+ llm = LLM.deepseek(key: ENV["KEY"])
502
+ ctx = LLM::Context.new(llm)
503
+ res = nil
504
+
505
+ loop do
506
+ res = ctx.talk("What's the weather in Tokyo?")
507
+ break unless ctx.pending_functions?
508
+
509
+ results = ctx.wait(:thread)
510
+ ctx.talk(results)
511
+ end
512
+
513
+ puts res.content
514
+ ```
515
+
516
+ The loop exits when the model responds with text rather than tool
517
+ calls. You can extend it with timeouts, user confirmation gates, or
518
+ custom error handling for each tool.
519
+
520
+ ### Trade-offs
521
+
522
+ Manual control is more code but gives you:
523
+
524
+ - **Arbitrary pre-execution checks** — inspect, rewrite, or skip tool
525
+ calls before they run.
526
+ - **Custom confirmation flows** — async approval over HTTP, Slack,
527
+ or email instead of the built-in terminal prompt.
528
+ - **Different strategies per tool** — run one tool on a thread and
529
+ another in a forked process within the same turn.
530
+ - **Fine-grained error recovery** — rescue per-tool failures and
531
+ decide which results to feed back.
532
+
533
+ [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html) handles
534
+ all of this automatically and is the right choice for most applications.
535
+ Drop down to the manual loop when you need control that the agent
536
+ abstraction doesn't expose.
537
+
310
538
  ## Skills
311
539
 
312
540
  The skill concept is borrowed from tools like Claude and
@@ -571,11 +799,11 @@ class Stream < LLM::Stream
571
799
  def on_tool_return(tool, result)
572
800
  end
573
801
 
574
- def on_compaction(ctx, compactor)
802
+ def on_compaction(compactor)
575
803
  # this callback is called *before* a compact happens
576
804
  end
577
805
 
578
- def on_compaction_finish(ctx, compactor)
806
+ def on_compaction_finish(compactor)
579
807
  # this callback is called *after* a compact happens
580
808
  end
581
809
  end
@@ -583,6 +811,244 @@ end
583
811
 
584
812
  [Back to top](#table-of-contents)
585
813
 
814
+ ## Concurrency
815
+
816
+ llm.rb supports six concurrency strategies for tool execution &ndash;
817
+ `:sequential`, `:thread`, `:fiber`, `:async`, `:fork`, and `:ractor`.
818
+ Each one implements the same interface &mdash; `spawn`, `wait`, `alive?`,
819
+ `interrupt!` &mdash; so the caller never has to care which strategy is
820
+ behind a given task.
821
+
822
+ Choose a strategy per-agent or per-call:
823
+
824
+ ```ruby
825
+ ## Per-agent — every tool loop uses :fork
826
+ agent = LLM::Agent.new(llm, concurrency: :fork, tools: [...])
827
+
828
+ ## Per-call — run a single tool on a thread
829
+ fn = FetchStocks.function
830
+ fn.task(:thread).wait
831
+ ```
832
+
833
+ Interruption is reliable across all six. No matter the backing &mdash;
834
+ thread, fiber, process, ractor &mdash; `LLM::Interrupt` reaches the
835
+ tool and it can rescue, clean up, and either re-raise to cancel the
836
+ turn or return a value to continue.
837
+
838
+ #### sequential
839
+
840
+ The default. Tools run one at a time on the calling thread. No
841
+ concurrency, no overhead. `spawn` is a no-op &mdash; execution happens
842
+ in `wait`. `alive?` always returns `false`.
843
+
844
+ Best for simple agents with a tool or two, debugging, or when tool
845
+ order matters.
846
+
847
+ #### thread
848
+
849
+ Each tool runs in its own `Thread`. The thread is created lazily &mdash;
850
+ you can build a task, pass it around, and decide when to run it.
851
+ Threads have `report_on_exception` disabled so errors surface through
852
+ `wait` rather than stderr.
853
+
854
+ Interruption raises `LLM::Interrupt` directly on the tool's thread,
855
+ which stops it mid-flight.
856
+
857
+ Best for IO-bound tools &mdash; HTTP calls, database queries. CRuby
858
+ releases the GVL during blocking IO, so you get real concurrency.
859
+
860
+ #### fiber
861
+
862
+ Each tool runs in a scheduler-backed `Fiber` via `Fiber.schedule`.
863
+ Requires `Fiber.scheduler` &mdash; raises `ArgumentError` without one.
864
+ Fibers yield cooperatively at IO boundaries, so this pairs well with
865
+ async libraries that set a scheduler.
866
+
867
+ Interruption raises `LLM::Interrupt` on the fiber, which stops at
868
+ the next yield point.
869
+
870
+ Best for IO-bound tools inside an async framework. Much lighter than
871
+ threads.
872
+
873
+ #### async
874
+
875
+ Each tool runs as an `Async::Task` inside a managed background
876
+ reactor. A dedicated thread runs an `Async::Reactor` event loop.
877
+ Work is submitted through a thread-safe `Queue` inbox and consumed
878
+ by the reactor. All fibers stay on one thread &mdash; no shared-memory
879
+ contention between them.
880
+
881
+ The reactor is created on demand and shared across all tasks in a
882
+ group. When `Group#wait` is called, tasks are submitted, the reactor
883
+ runs them concurrently, and results are bridged back to the caller
884
+ through per-task queues. The reactor is torn down after `wait`
885
+ completes.
886
+
887
+ Interruption pushes an `LLM::Interrupt` sentinel into the task's
888
+ result queue instead of using `Fiber#raise` &mdash; cleaner, and it
889
+ avoids surprising the reactor's internal fibers.
890
+
891
+ Best for IO-bound tools when you want Async's structured concurrency
892
+ model without running your whole application inside a reactor. The
893
+ reactor is self-contained &mdash; your main thread stays synchronous.
894
+ Requires the `async` gem.
895
+
896
+ #### fork
897
+
898
+ Each tool runs in a forked child process. Communication uses
899
+ [`xchan`](https://github.com/1robertrb/xchan.rb) (marshal-based
900
+ channels): the parent sends control messages, the child sends
901
+ results back. Each child is a separate OS process with its own
902
+ memory space &mdash; a crash in the tool cannot touch the parent.
903
+
904
+ `Fork::Task` checks liveness with `Process.waitpid(WNOHANG)` and
905
+ delivers interrupts as messages over the control channel. The child
906
+ raises `LLM::Interrupt` on `Thread.main` when it receives the
907
+ interrupt message. Tracer callbacks fire in both parent and child.
908
+
909
+ Best for process isolation &mdash; shell commands, native extensions,
910
+ anything you don't want touching the parent's memory. True parallelism
911
+ too, since there's no GVL in separate processes. Requires the
912
+ `xchan` gem.
913
+
914
+ #### ractor
915
+
916
+ Each class-based tool runs in a Ruby `Ractor`. `Ractor::Task`
917
+ coordinates through `Ractor::Mailbox`. Interruption sends a message
918
+ through the mailbox; a listener thread inside the ractor raises
919
+ `LLM::Interrupt` on `Thread.main`.
920
+
921
+ Ractors have restrictions: only class-based tools are supported (no
922
+ blocks, skills, or MCP tools), and arguments must be
923
+ ractor-shareable. The runtime raises `LLM::RactorError` early if
924
+ you try to run an unsupported tool type.
925
+
926
+ Best for CPU-bound tools, true parallelism without the overhead of
927
+ forking full processes. More restrictive than `:fork` but lighter.
928
+
929
+ #### Quick reference
930
+
931
+ | Strategy | Backing | Parallel? | Isolation? | Requires |
932
+ |---|---|---|---|---|
933
+ | `:sequential` | direct call | No | No | &mdash; |
934
+ | `:thread` | `Thread` | IO only (GVL) | No | &mdash; |
935
+ | `:fiber` | `Fiber.schedule` | Cooperative | No | `Fiber.scheduler` |
936
+ | `:async` | `Async::Reactor` on bg thread | Cooperative | No | `async` gem |
937
+ | `:fork` | `Kernel.fork` | Yes (process) | Yes (memory) | `xchan` gem |
938
+ | `:ractor` | `Ractor` | Yes (CPU) | Limited | &mdash; |
939
+
940
+ [Back to top](#table-of-contents)
941
+
942
+ ## Context Compaction
943
+
944
+ Long-running conversations consume tokens. Without intervention, every turn
945
+ pushes toward the model's context window limit, at which point the provider
946
+ rejects the request.
947
+
948
+ llm.rb provides compaction through pluggable strategies. All strategies
949
+ inherit from [`LLM::Compactor`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor.html)
950
+ and are invoked automatically before each `ctx.talk(...)` call.
951
+
952
+ ### Configuration
953
+
954
+ Pass a compactor class and options when creating a context or agent:
955
+
956
+ ```ruby
957
+ ctx = LLM::Context.new(
958
+ llm,
959
+ compactor: LLM::Compactor::Truncate,
960
+ compactor_options: {keep: 64}
961
+ )
962
+
963
+ # LLM::Agent accepts the same options
964
+ agent = LLM::Agent.new(
965
+ llm,
966
+ compactor: LLM::Compactor::Truncate,
967
+ compactor_options: {keep: 128}
968
+ )
969
+ ```
970
+
971
+ The compactor runs automatically before every `talk` call. This keeps the
972
+ conversation constantly alive — there is no chance of exhausting the context
973
+ window because old messages are dropped before they accumulate. The trade-off
974
+ is that dropped messages are gone, so information may be lost. Set `keep` to
975
+ a higher number to retain more context at the cost of slower accumulation.
976
+
977
+ The default is [`LLM::Compactor::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Null.html)
978
+ — compaction is disabled unless you opt in.
979
+
980
+ ### Standalone usage
981
+
982
+ A compactor can be used independently of a context or agent:
983
+
984
+ ```ruby
985
+ compactor = LLM::Compactor::Truncate.new(agent)
986
+ compactor.call(keep: 200) # or ctx, agent, etc.
987
+ ```
988
+
989
+ This is useful for one-off compaction outside the automatic per-turn cycle,
990
+ or when you want to compact on a different schedule.
991
+
992
+ ### Strategies
993
+
994
+ **[`LLM::Compactor::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Null.html)**
995
+ — the default. Does nothing.
996
+
997
+ **[`LLM::Compactor::Truncate`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Truncate.html)**
998
+ — drops the oldest messages, keeping only the N most recent.
999
+
1000
+ - **Fast** — no network call, no LLM overhead. Operates entirely in memory
1001
+ with a single pass over the message list.
1002
+ - **No dependencies** — works offline, has no model or API requirements, and
1003
+ introduces no additional cost.
1004
+ - **Tool-loop safe** — when a tool result (return) falls at the truncation
1005
+ boundary, the corresponding tool call is kept. Without this, the
1006
+ conversation would contain an orphaned result with no matching call,
1007
+ causing API-level errors on the next turn.
1008
+
1009
+ The `keep:` parameter accepts either an integer count or a percentage
1010
+ string like `"80%"`, which keeps approximately 80% of the most recent
1011
+ messages. This is useful when you want to trim proportionally rather
1012
+ than to an absolute number.
1013
+
1014
+ ```ruby
1015
+ ctx = LLM::Context.new(
1016
+ llm,
1017
+ compactor: LLM::Compactor::Truncate,
1018
+ compactor_options: {keep: 128}
1019
+ )
1020
+ ```
1021
+
1022
+ ### Manual compaction
1023
+
1024
+ The REPL provides a `/compact` command that invokes Truncate on the current
1025
+ agent's context:
1026
+
1027
+ ```
1028
+ /compact # keep last 128 messages
1029
+ /compact 50 # keep last 50 messages
1030
+ /compact 75% # keep approximately 75% of messages
1031
+ ```
1032
+
1033
+ ### Lifecycle callbacks
1034
+
1035
+ Both strategies call stream hooks so the UI can show progress:
1036
+
1037
+ ```ruby
1038
+ def on_compaction(compactor)
1039
+ # called before compaction begins
1040
+ end
1041
+
1042
+ def on_compaction_finish(compactor)
1043
+ # called after compaction completes
1044
+ end
1045
+ ```
1046
+
1047
+ The context's `compacted?` flag is `true` between compaction and the next
1048
+ model response.
1049
+
1050
+ [Back to top](#table-of-contents)
1051
+
586
1052
  ## Serialization
587
1053
 
588
1054
  The [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
@@ -792,20 +1258,32 @@ for a number of different reasons, most often because the
792
1258
  user made a mistake, or the model is making a mistake and
793
1259
  the user wants to cancel the action.
794
1260
 
795
- The runtime has built-in support for cancellation. So for
796
- example it is possible to cancel a request on the main
797
- thread from a secondary thread. A number of things happen
798
- when a request is cancelled. First the request is cancelled
799
- at the transport level, and each transport handles it a little
800
- differently. The net effect in every case is that the connection
801
- is closed.
802
-
803
- The runtime then notifies the rest of the system. so for example,
804
- if a tool was running, it will receive the `on_interrupt` / `on_cancel`
805
- callback that lets the tool do any necessary cleanup, or execute its own
806
- cancellation plan. Tools that were pending (not yet run but requetsed to
807
- run) are cancelled through
808
- [`LLM::Function#cancel`](https://r.uby.dev/api-docs/llm.rb/LLM/Function.html#cancel-instance_method).
1261
+ The runtime has built-in support for cancellation. Call
1262
+ `agent.cancel!` or `ctx.cancel!` from any thread and two
1263
+ things happen at once:
1264
+
1265
+ [`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
1266
+ is raised on the thread where `talk` is running, so the
1267
+ caller can rescue it and know the request was cancelled.
1268
+
1269
+ At the same time, `LLM::Interrupt` is raised on every tool
1270
+ that is currently executing &mdash; regardless of which
1271
+ concurrency strategy it's using. A tool running in a thread
1272
+ gets it on that thread. A tool in a fiber gets it on that
1273
+ fiber. A tool in a forked process gets it via a message
1274
+ over the xchan control channel. The
1275
+ delivery mechanism depends on the strategy, but the effect is
1276
+ the same: the tool can rescue `LLM::Interrupt`, clean up
1277
+ resources, close connections, flush buffers, and either
1278
+ re-raise to abort or return a partial result.
1279
+
1280
+ Pending tools &mdash; those the model requested but that haven't
1281
+ started running yet &mdash; are cancelled through
1282
+ [`LLM::Function#cancel`](https://r.uby.dev/api-docs/llm.rb/LLM/Function.html#cancel-instance_method)
1283
+ without ever being executed.
1284
+
1285
+ The transport layer also cancels the in-flight HTTP request,
1286
+ closing the connection to the provider.
809
1287
 
810
1288
  ```ruby
811
1289
  require "llm"
@@ -828,6 +1306,58 @@ rescue LLM::Interrupt
828
1306
  end
829
1307
  ```
830
1308
 
1309
+ #### Tool interrupts
1310
+
1311
+ When a running tool is interrupted &ndash; for example the user presses
1312
+ ESC in the [REPL](#repl) &ndash; the runtime raises
1313
+ [`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
1314
+ on the tool's execution context. This behavior is uniform across
1315
+ all six concurrency strategies (`:sequential`, `:thread`, `:fiber`,
1316
+ `:async`, `:fork`, and `:ractor`).
1317
+
1318
+ A tool has two choices:
1319
+
1320
+ **Re-raise** `LLM::Interrupt` to cancel the entire turn. The
1321
+ exception propagates out of the tool loop and the request is
1322
+ aborted. This is the default when you don't rescue the exception.
1323
+
1324
+ ```ruby
1325
+ def call
1326
+ # ... do work ...
1327
+ rescue LLM::Interrupt
1328
+ cleanup
1329
+ raise # cancel the turn
1330
+ end
1331
+ ```
1332
+
1333
+ **Return a value** to continue the tool loop. The model receives
1334
+ the result and decides what to do next, aware that the tool was
1335
+ interrupted.
1336
+
1337
+ ```ruby
1338
+ def call
1339
+ # ... do work ...
1340
+ rescue LLM::Interrupt
1341
+ cleanup
1342
+ {ok: false, reason: "interrupted"} # continue the loop
1343
+ end
1344
+ ```
1345
+
1346
+ The right choice depends on the situation. A hard cancel aborts
1347
+ the request outright &ndash; useful when continuing would produce
1348
+ garbage. Returning a value lets the model adapt, which can be
1349
+ helpful when the interrupt is temporary (e.g. a timeout).
1350
+
1351
+ The `:ractor` strategy delivers the interrupt through ractor
1352
+ message passing &mdash; a listener thread inside the tool ractor
1353
+ receives the interrupt message and raises `LLM::Interrupt` on the
1354
+ ractor's main thread. The end result is the same as every other
1355
+ strategy: the tool can rescue, clean up, and decide.
1356
+ The `:fork` strategy delivers the interrupt via a message
1357
+ over the xchan control channel, which a listener thread in the
1358
+ child process picks up and raises on `Thread.main`. All other strategies
1359
+ raise the exception directly on the executing thread or fiber.
1360
+
831
1361
  [Back to top](#table-of-contents)
832
1362
 
833
1363
  ## Tracer
@@ -916,15 +1446,20 @@ This feature requires that the [curses](https://github.com/ruby/curses)
916
1446
  and [kramdown](https://github.com/gettalong/kramdown) libraries are
917
1447
  installed and available to require.
918
1448
 
1449
+ The `name:` option labels the agent throughout the TUI &mdash;
1450
+ useful when working with multiple agents. The `path:` option
1451
+ persists state across sessions. The `tools:` option attaches
1452
+ extra tools for the duration of the session.
1453
+
919
1454
  ```ruby
920
1455
  require "llm"
921
1456
 
922
1457
  llm = LLM.deepseek(key: ENV["KEY"])
923
- agent = LLM::Agent.new(llm)
924
- agent.repl
1458
+ agent = LLM::Agent.new(llm, name: "my-agent")
1459
+ agent.repl(path: "session.json", tools: LLM::Tool.subclasses)
925
1460
  ```
926
1461
 
927
- #### State
1462
+ #### Persistence
928
1463
 
929
1464
  The `path:` option accepts a file path where runtime state
930
1465
  is read from and written to. This lets you resume a
@@ -949,6 +1484,16 @@ agent = LLM::Agent.new(llm)
949
1484
  agent.repl(tools: [Debugger])
950
1485
  ```
951
1486
 
1487
+ Load every built-in tool with `LLM::Tool.subclasses`:
1488
+
1489
+ ```ruby
1490
+ require "llm/tools"
1491
+
1492
+ llm = LLM.deepseek(key: ENV["KEY"])
1493
+ agent = LLM::Agent.new(llm)
1494
+ agent.repl(tools: LLM::Tool.subclasses)
1495
+ ```
1496
+
952
1497
  #### Skills
953
1498
 
954
1499
  The read-eval-print loop also accepts a `skills` option.
@@ -977,7 +1522,11 @@ agent.repl(tracer: true, tools: [Debugger])
977
1522
 
978
1523
  #### Input
979
1524
 
980
- The input area supports several keyboard shortcuts:
1525
+ The input area supports several keyboard shortcuts.
1526
+ When characters arrive faster than a threshold the REPL
1527
+ detects that text is being pasted rather than typed. In
1528
+ paste mode pressing `Enter` inserts a newline instead of
1529
+ submitting, allowing multi-line prompts.
981
1530
 
982
1531
  | Key | Action |
983
1532
  |---|---|
@@ -988,36 +1537,91 @@ The input area supports several keyboard shortcuts:
988
1537
  | `Ctrl+K` | Erase from cursor to the end of the line |
989
1538
  | `Ctrl+Y` | Paste previously killed text |
990
1539
  | `Ctrl+D` | Delete the character at the cursor |
1540
+ | `Ctrl+P` | Recall the previous user message |
1541
+ | `Ctrl+N` | Recall the next user message |
991
1542
  | `Left / Right` | Move the cursor |
992
- | `Up / Down` | Scroll the transcript |
993
-
994
- When characters arrive faster than a threshold the REPL
995
- detects that text is being pasted rather than typed. In
996
- paste mode pressing `Enter` inserts a newline instead of
997
- submitting, allowing multi-line prompts.
1543
+ | `Up / Down` | Scroll the transcript one line |
1544
+ | `PgUp` / `PgDn` | Scroll the transcript by one page |
1545
+ | `Tab` | Complete `/command` names |
1546
+ | `Esc` | Cancel the current request |
998
1547
 
999
1548
  #### Commands
1000
1549
 
1001
1550
  Commands are recognized by a `/` prefix on the input line.
1002
- The built-in command is `/exit`, which leaves the REPL.
1551
+ Type `/compact` to free context window space by dropping the
1552
+ oldest messages. Type `/exit` to leave the REPL.
1553
+
1003
1554
  The [`LLM::Repl::Command`](https://r.uby.dev/api-docs/llm.rb/LLM/Repl/Command.html)
1004
- class can be subclassed to add custom commands &ndash;
1005
- any subclass that defines a `name`, `description`, and
1006
- `call` method is automatically registered and available.
1555
+ class is intentionally similar to [`LLM::Tool`](#llmtool) and
1556
+ [`LLM::Schema`](#schema) in its interface &ndash; you declare a
1557
+ name, description, and parameters with the same vocabulary.
1558
+ A subclass is automatically registered and available as `/name`.
1559
+
1560
+ ##### Parameters
1561
+
1562
+ Parameters are declared with `parameter :name, Type, "description"`
1563
+ and marked required with `required %i[name]`. The `call` method
1564
+ receives them as keyword arguments matching the parameter names.
1565
+ Parameters without a user-supplied value fall back to the
1566
+ method signature's default.
1007
1567
 
1008
1568
  ```ruby
1009
- class LLM::Repl::Command
1010
- class Clear < self
1011
- name "clear"
1012
- description "clears the transcript"
1013
-
1014
- def call
1015
- system("clear") || system("cls")
1016
- end
1569
+ class Greeter < LLM::Command
1570
+ name "greet"
1571
+ description "Greets the given name"
1572
+ parameter :name, String, "The person's name"
1573
+ required %i[name]
1574
+
1575
+ def call(name:)
1576
+ write "Welcome #{name}!\n"
1017
1577
  end
1018
1578
  end
1019
1579
  ```
1020
1580
 
1581
+ ##### Output
1582
+
1583
+ A command writes to the transcript with `write(str, who:)`.
1584
+ The `who:` label is rendered in bold. It defaults to
1585
+ `command(name): ` where `name` is the command's registered
1586
+ name.
1587
+
1588
+ ```ruby
1589
+ def call(name:)
1590
+ write("Greetings #{name}!\n")
1591
+ end
1592
+ ```
1593
+
1594
+ ##### Help
1595
+
1596
+ The built-in [`help`](https://r.uby.dev/api-docs/llm.rb/LLM/Repl/Command.html#help-class_method)
1597
+ class method formats the name, description, and parameter list
1598
+ automatically. Use it from inside a command or via `/help`.
1599
+
1600
+ ```ruby
1601
+ class Greeter < LLM::Command
1602
+ name "greet"
1603
+ # ...
1604
+ end
1605
+
1606
+ # /help greet displays:
1607
+ # Command: greet
1608
+ # Description: Greets the given name
1609
+ #
1610
+ # Parameters:
1611
+ # name [String] - The person's name (required)
1612
+ ```
1613
+
1614
+ ##### Aliases
1615
+
1616
+ Subclassing an existing command inherits its name, description,
1617
+ and parameters. This is how `/quit` is an alias of `/exit`:
1618
+
1619
+ ```ruby
1620
+ class Quit < LLM::Repl::Command::Exit
1621
+ name "quit"
1622
+ end
1623
+ ```
1624
+
1021
1625
  [Back to top](#table-of-contents)
1022
1626
 
1023
1627
  ## Images
@@ -1182,3 +1786,35 @@ res.text # => "Good day"
1182
1786
  ```
1183
1787
 
1184
1788
  [Back to top](#table-of-contents)
1789
+
1790
+ ## OCR
1791
+
1792
+ Optical Character Recognition extracts text from images and
1793
+ documents.
1794
+
1795
+ #### Mistral
1796
+
1797
+ Mistral is the only provider that currently supports OCR
1798
+ through its dedicated API endpoint. The `ocr` method accepts
1799
+ either an `image_url:` or a `document_url:` parameter.
1800
+ Document URLs can point to PDFs. The response exposes pages
1801
+ through `res.pages`, where each page has a `markdown` field
1802
+ containing the extracted text.
1803
+
1804
+ ```ruby
1805
+ require "llm"
1806
+
1807
+ llm = LLM.mistral(key: ENV["KEY"])
1808
+
1809
+ ##
1810
+ # Extract text from an image
1811
+ res = llm.ocr(image_url: "https://example.com/photo.png")
1812
+ res.pages.each { |page| puts page.markdown }
1813
+
1814
+ ##
1815
+ # Extract text from a PDF
1816
+ res = llm.ocr(document_url: "https://example.com/report.pdf")
1817
+ res.pages.each { |page| puts page.markdown }
1818
+ ```
1819
+
1820
+ [Back to top](#table-of-contents)