llm.rb 15.2.2 → 15.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +197 -3
  3. data/README.md +180 -64
  4. data/bin/llm.rb +9 -2
  5. data/data/alibaba.json +45 -0
  6. data/data/anthropic.json +67 -0
  7. data/data/bedrock.json +1466 -397
  8. data/data/deepinfra.json +148 -16
  9. data/data/deepseek.json +3 -0
  10. data/data/mistral.json +42 -0
  11. data/data/openai.json +186 -0
  12. data/data/openrouter.json +1538 -378
  13. data/data/xai.json +53 -20
  14. data/data/zai.json +90 -4
  15. data/docs/deepdive/advanced/compaction.md +1 -2
  16. data/docs/deepdive/advanced/context.md +214 -1
  17. data/docs/deepdive/advanced/guard.md +9 -57
  18. data/docs/deepdive/features/builtin_tools.md +14 -16
  19. data/docs/deepdive/features/console.md +5 -0
  20. data/docs/deepdive/features/database.md +85 -10
  21. data/docs/deepdive/fundamentals/agents.md +7 -8
  22. data/docs/deepdive/fundamentals/providers.md +45 -5
  23. data/docs/deepdive/fundamentals/schema.md +73 -0
  24. data/docs/deepdive/fundamentals/tools.md +80 -27
  25. data/docs/deepdive/media/audio.md +8 -19
  26. data/docs/deepdive/media/images.md +8 -10
  27. data/docs/deepdive/media/ocr.md +1 -3
  28. data/docs/deepdive/reference/cost.md +48 -0
  29. data/docs/deepdive/reference/tracer.md +76 -0
  30. data/docs/deepdive.md +1 -1
  31. data/lib/llm/active_record/message.rb +113 -0
  32. data/lib/llm/active_record.rb +1 -0
  33. data/lib/llm/agent.rb +44 -19
  34. data/lib/llm/console/buffer.rb +9 -1
  35. data/lib/llm/console.rb +6 -1
  36. data/lib/llm/context/deserializer.rb +10 -3
  37. data/lib/llm/context.rb +62 -26
  38. data/lib/llm/guard.rb +2 -8
  39. data/lib/llm/message.rb +18 -7
  40. data/lib/llm/provider.rb +74 -16
  41. data/lib/llm/providers/alibaba.rb +15 -0
  42. data/lib/llm/providers/anthropic/error_handler.rb +5 -2
  43. data/lib/llm/providers/anthropic/files.rb +12 -12
  44. data/lib/llm/providers/anthropic/models.rb +2 -2
  45. data/lib/llm/providers/anthropic.rb +5 -3
  46. data/lib/llm/providers/bedrock/error_handler.rb +3 -2
  47. data/lib/llm/providers/bedrock/models.rb +5 -3
  48. data/lib/llm/providers/bedrock.rb +5 -3
  49. data/lib/llm/providers/deepinfra/audio.rb +4 -4
  50. data/lib/llm/providers/deepinfra/images.rb +4 -4
  51. data/lib/llm/providers/google/error_handler.rb +5 -2
  52. data/lib/llm/providers/google/files.rb +10 -10
  53. data/lib/llm/providers/google/images.rb +2 -2
  54. data/lib/llm/providers/google/models.rb +2 -2
  55. data/lib/llm/providers/google.rb +7 -7
  56. data/lib/llm/providers/mistral.rb +3 -1
  57. data/lib/llm/providers/ollama/error_handler.rb +5 -2
  58. data/lib/llm/providers/ollama/models.rb +2 -2
  59. data/lib/llm/providers/ollama.rb +7 -5
  60. data/lib/llm/providers/openai/audio.rb +6 -6
  61. data/lib/llm/providers/openai/error_handler.rb +5 -2
  62. data/lib/llm/providers/openai/files.rb +10 -10
  63. data/lib/llm/providers/openai/images.rb +4 -4
  64. data/lib/llm/providers/openai/models.rb +2 -2
  65. data/lib/llm/providers/openai/moderations.rb +2 -2
  66. data/lib/llm/providers/openai/request_adapter.rb +1 -1
  67. data/lib/llm/providers/openai/responses.rb +8 -8
  68. data/lib/llm/providers/openai/vector_stores.rb +22 -22
  69. data/lib/llm/providers/openai.rb +10 -8
  70. data/lib/llm/providers/xai/images.rb +4 -4
  71. data/lib/llm/schema.rb +24 -0
  72. data/lib/llm/tracer/telemetry.rb +4 -4
  73. data/lib/llm/tracer.rb +11 -3
  74. data/lib/llm/transport/execution.rb +8 -4
  75. data/lib/llm/utils.rb +13 -0
  76. data/lib/llm/version.rb +1 -1
  77. data/llm.gemspec +2 -2
  78. metadata +5 -5
  79. data/lib/llm/guard/loop.rb +0 -89
data/data/xai.json CHANGED
@@ -370,17 +370,30 @@
370
370
  "output": 0
371
371
  }
372
372
  },
373
- "grok-imagine-image-2.0": {
374
- "id": "grok-imagine-image-2.0",
375
- "name": "Grok Imagine Image 2.0",
376
- "description": "Image model for prompt-driven generation, editing, and visual design workflows",
373
+ "grok-4.6": {
374
+ "id": "grok-4.6",
375
+ "name": "Grok 4.6",
376
+ "description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
377
377
  "family": "grok",
378
378
  "attachment": true,
379
- "reasoning": false,
380
- "tool_call": false,
381
- "temperature": false,
382
- "release_date": "2026-08-07",
383
- "last_updated": "2026-08-07",
379
+ "reasoning": true,
380
+ "reasoning_options": [
381
+ {
382
+ "type": "effort",
383
+ "values": [
384
+ "low",
385
+ "medium",
386
+ "high",
387
+ "xhigh"
388
+ ]
389
+ }
390
+ ],
391
+ "tool_call": true,
392
+ "structured_output": true,
393
+ "temperature": true,
394
+ "knowledge": "2026-02-01",
395
+ "release_date": "2026-08-12",
396
+ "last_updated": "2026-08-12",
384
397
  "modalities": {
385
398
  "input": [
386
399
  "text",
@@ -388,19 +401,39 @@
388
401
  "pdf"
389
402
  ],
390
403
  "output": [
391
- "image",
392
- "pdf"
404
+ "text"
393
405
  ]
394
406
  },
395
407
  "open_weights": false,
396
408
  "limit": {
397
- "context": 64000,
398
- "output": 0
409
+ "context": 500000,
410
+ "output": 500000
411
+ },
412
+ "cost": {
413
+ "input": 2,
414
+ "output": 6,
415
+ "cache_read": 0.5,
416
+ "tiers": [
417
+ {
418
+ "input": 4,
419
+ "output": 12,
420
+ "cache_read": 1,
421
+ "tier": {
422
+ "type": "context",
423
+ "size": 200000
424
+ }
425
+ }
426
+ ],
427
+ "context_over_200k": {
428
+ "input": 4,
429
+ "output": 12,
430
+ "cache_read": 1
431
+ }
399
432
  }
400
433
  },
401
- "grok-4.6": {
402
- "id": "grok-4.6",
403
- "name": "Grok 4.6",
434
+ "grok-4.7": {
435
+ "id": "grok-4.7",
436
+ "name": "Grok 4.7",
404
437
  "description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
405
438
  "family": "grok",
406
439
  "attachment": true,
@@ -419,9 +452,9 @@
419
452
  "tool_call": true,
420
453
  "structured_output": true,
421
454
  "temperature": true,
422
- "knowledge": "2026-02-01",
423
- "release_date": "2026-08-12",
424
- "last_updated": "2026-08-12",
455
+ "knowledge": "2026-05",
456
+ "release_date": "2026-09-21",
457
+ "last_updated": "2026-09-21",
425
458
  "modalities": {
426
459
  "input": [
427
460
  "text",
@@ -462,7 +495,7 @@
462
495
  "grok-imagine-image-quality": {
463
496
  "id": "grok-imagine-image-quality",
464
497
  "name": "Grok Imagine Image Quality",
465
- "description": "Image model for prompt-driven generation, editing, and visual design workflows",
498
+ "description": "Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows",
466
499
  "family": "grok",
467
500
  "attachment": true,
468
501
  "reasoning": false,
data/data/zai.json CHANGED
@@ -240,11 +240,59 @@
240
240
  "cache_write": 0
241
241
  }
242
242
  },
243
+ "glm-5.3-flashx": {
244
+ "id": "glm-5.3-flashx",
245
+ "name": "GLM-5.3-FlashX",
246
+ "description": "High-speed GLM-5.3-Flash serving option for coding and agent workflows",
247
+ "family": "glm-flash",
248
+ "attachment": true,
249
+ "reasoning": true,
250
+ "reasoning_options": [
251
+ {
252
+ "type": "effort",
253
+ "values": [
254
+ "low",
255
+ "high",
256
+ "max"
257
+ ]
258
+ }
259
+ ],
260
+ "tool_call": true,
261
+ "interleaved": {
262
+ "field": "reasoning_content"
263
+ },
264
+ "structured_output": true,
265
+ "temperature": true,
266
+ "release_date": "2026-09-18",
267
+ "last_updated": "2026-09-18",
268
+ "modalities": {
269
+ "input": [
270
+ "text",
271
+ "image",
272
+ "video",
273
+ "pdf"
274
+ ],
275
+ "output": [
276
+ "text"
277
+ ]
278
+ },
279
+ "open_weights": true,
280
+ "limit": {
281
+ "context": 1000000,
282
+ "output": 131072
283
+ },
284
+ "cost": {
285
+ "input": 0.37,
286
+ "output": 1.25,
287
+ "cache_read": 0.075,
288
+ "cache_write": 0
289
+ }
290
+ },
243
291
  "glm-5.3-flash": {
244
292
  "id": "glm-5.3-flash",
245
293
  "name": "GLM-5.3-Flash",
246
294
  "description": "Native multimodal GLM model for efficient coding and long-horizon agent tasks",
247
- "family": "glm",
295
+ "family": "glm-flash",
248
296
  "attachment": true,
249
297
  "reasoning": true,
250
298
  "reasoning_options": [
@@ -282,9 +330,9 @@
282
330
  "output": 131072
283
331
  },
284
332
  "cost": {
285
- "input": 0.075,
286
- "output": 0.25,
287
- "cache_read": 0.015,
333
+ "input": 0.15,
334
+ "output": 0.5,
335
+ "cache_read": 0.03,
288
336
  "cache_write": 0
289
337
  }
290
338
  },
@@ -605,6 +653,44 @@
605
653
  "cache_write": 0
606
654
  }
607
655
  },
656
+ "glm-4.6v-flash": {
657
+ "id": "glm-4.6v-flash",
658
+ "name": "GLM-4.6V-Flash",
659
+ "description": "Lightweight GLM vision model for visual reasoning, documents, and multimodal agents",
660
+ "family": "glm-flash",
661
+ "attachment": true,
662
+ "reasoning": true,
663
+ "reasoning_options": [
664
+ {
665
+ "type": "toggle"
666
+ }
667
+ ],
668
+ "tool_call": true,
669
+ "temperature": true,
670
+ "release_date": "2025-12-08",
671
+ "last_updated": "2025-12-08",
672
+ "modalities": {
673
+ "input": [
674
+ "text",
675
+ "image",
676
+ "video"
677
+ ],
678
+ "output": [
679
+ "text"
680
+ ]
681
+ },
682
+ "open_weights": true,
683
+ "limit": {
684
+ "context": 128000,
685
+ "output": 32768
686
+ },
687
+ "cost": {
688
+ "input": 0,
689
+ "output": 0,
690
+ "cache_read": 0,
691
+ "cache_write": 0
692
+ }
693
+ },
608
694
  "glm-4.7-flash": {
609
695
  "id": "glm-4.7-flash",
610
696
  "name": "GLM-4.7-Flash",
@@ -47,11 +47,10 @@ compactor.call(keep: 200)
47
47
 
48
48
  ##### The `/keep` command
49
49
 
50
- The console provides a `/keep` command that accepts a count or
50
+ The console provides a `/keep` command that requires a count or
51
51
  percentage:
52
52
 
53
53
  ```ruby
54
- # /keep # keep last 128 messages
55
54
  # /keep 50 # keep last 50 messages
56
55
  # /keep 75% # keep approximately 75% of messages
57
56
  ```
@@ -82,7 +82,60 @@ Each retry sleeps a growing interval (2s, 4s, 6s, ...) and notifies
82
82
  the stream through
83
83
  [`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry-instance_method)
84
84
  before trying again. An `LLM::Agent` enables a budget of 5 by
85
- default, so most users never touch this directly.
85
+ default (or 8 on the Alibaba provider, which rate limits more
86
+ often), so most users never touch this directly.
87
+
88
+ ### Identity
89
+
90
+ #### Overview
91
+
92
+ A context has an id and a creation time, and so does each message it
93
+ holds. The id is a UUIDv7 string, a UUID version that encodes its own
94
+ creation timestamp, so an id sorts by the order it was created and
95
+ carries the time it was made.
96
+
97
+ #### How it works
98
+
99
+ Read the id and creation time from a context, an agent, or a message:
100
+
101
+ ```ruby
102
+ require "llm"
103
+
104
+ llm = LLM.deepseek(key: ENV["KEY"])
105
+ ctx = LLM::Context.new(llm)
106
+ ctx.talk "Hello"
107
+ ctx.id # => "01932f5a-..." (UUIDv7)
108
+ ctx.created_at # => 2026-09-11 04:21:07 UTC
109
+ ctx.messages.first.id
110
+ ```
111
+
112
+ [`LLM::Agent#id`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#id-instance_method)
113
+ and
114
+ [`LLM::Agent#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#created_at-instance_method)
115
+ delegate to the context the agent wraps.
116
+ [`LLM::Context#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#created_at-instance_method)
117
+ and
118
+ [`LLM::Message#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html#created_at-instance_method)
119
+ are read from the id rather than stored, so they return `nil` when
120
+ the id is not a UUIDv7 string.
121
+
122
+ #### Why would I use it?
123
+
124
+ An id gives a conversation or a message a stable name you can log,
125
+ correlate, or look up. Because the id is a UUIDv7, the same value
126
+ also answers when the object was created, so sorting ids sorts by
127
+ creation order and no separate timestamp column is needed.
128
+
129
+ #### Notes
130
+
131
+ The id is generated once and saved with the runtime state, so it
132
+ survives a save and a restore. A payload written before ids existed
133
+ has none, so its object is restored with a fresh id.
134
+ [`LLM::Message#==`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html#==-instance_method)
135
+ ignores the id, so a difference in creation time alone does not make
136
+ two messages unequal.
137
+ [`LLM::Utils.timestamp`](https://r.uby.dev/api-docs/llm.rb/LLM/Utils.html#timestamp-instance_method)
138
+ is the shared method that decodes a UUIDv7 timestamp.
86
139
 
87
140
  ### Manual loop
88
141
 
@@ -280,3 +333,163 @@ The mechanism is the same across all six concurrency strategies.
280
333
  The `:ractor` strategy delivers the interrupt through ractor
281
334
  message passing. The `:fork` strategy delivers it via xchan.
282
335
 
336
+ ### Messages
337
+
338
+ #### Overview
339
+
340
+ [`LLM::Context#messages`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#messages-instance_method)
341
+ returns an
342
+ [`LLM::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Buffer.html),
343
+ an ordered, array-like collection of the conversation's
344
+ [`LLM::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html)
345
+ objects. Read it to inspect or filter a conversation, and edit it to
346
+ shape what the model sees next.
347
+
348
+ #### How it works
349
+
350
+ A buffer is `Enumerable`, so it supports `each`, `find`, `map`,
351
+ `select`, and the rest. It also offers the array methods a long
352
+ conversation needs, including `first`, `last`, `take`, `drop`,
353
+ `shift`, `pop`, `slice!`, `select!`, `reject!`, and `clear`:
354
+
355
+ ```ruby
356
+ require "llm"
357
+
358
+ llm = LLM.deepseek(key: ENV["KEY"])
359
+ ctx = LLM::Context.new(llm)
360
+ ctx.talk "Hello"
361
+
362
+ ctx.messages.size # => 2
363
+ ctx.messages.first # => the user message
364
+ ctx.messages.last # => the assistant message
365
+ ctx.messages.each { |m| puts "#{m.role}: #{m.content}" }
366
+ ```
367
+
368
+ Because a message carries its own role, filtering by role is a normal
369
+ `select`:
370
+
371
+ ```ruby
372
+ ctx.messages.select(&:assistant?)
373
+ ```
374
+
375
+ #### Why would I use it?
376
+
377
+ Reading the buffer gives you the conversation as data, so you can log
378
+ it, count tokens against it, or render it in your own UI. Editing the
379
+ buffer lets you drop or keep specific messages without rebuilding the
380
+ conversation.
381
+
382
+ #### Notes
383
+
384
+ Changing the buffer changes the next request, so edits are best made
385
+ between turns. The compaction topic covers the built-in, bounded way
386
+ to trim a long conversation.
387
+
388
+ ### Prompt
389
+
390
+ #### Overview
391
+
392
+ [`LLM::Prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Prompt.html)
393
+ composes a single request from several role-aware messages. A prompt
394
+ is not just a string: it is an ordered list of messages with explicit
395
+ roles, so one turn can carry a system message, a user message, and
396
+ anything else the model supports.
397
+
398
+ #### How it works
399
+
400
+ Call
401
+ [`LLM::Context#prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#prompt-instance_method)
402
+ with a block, then pass the result to
403
+ [`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk).
404
+ Inside the block, `system`, `user`, and `developer` append a message
405
+ with the matching role, and `talk` appends one with an explicit
406
+ role. The provider resolves each role to its provider-specific name:
407
+
408
+ ```ruby
409
+ require "llm"
410
+
411
+ llm = LLM.deepseek(key: ENV["KEY"])
412
+ ctx = LLM::Context.new(llm)
413
+
414
+ prompt = ctx.prompt do
415
+ system "Your task is to assist the user"
416
+ user "Hello. Can you assist me?"
417
+ end
418
+
419
+ res = ctx.talk(prompt)
420
+ ```
421
+
422
+ The block receives the prompt object when it takes an argument, and
423
+ otherwise runs in the prompt's context:
424
+ [`LLM::Prompt#to_a`](https://r.uby.dev/api-docs/llm.rb/LLM/Prompt.html#to_a)
425
+ returns the messages in order, and two prompts are equal when their
426
+ messages match.
427
+
428
+ #### Why would I use it?
429
+
430
+ A prompt keeps the roles of a multi-part request explicit, and it is
431
+ an object you can build, pass around, and compare before it is sent.
432
+ Use it when a turn needs more than one role, or when the same prompt
433
+ is composed in more than one place.
434
+
435
+ #### Notes
436
+
437
+ [`LLM::Agent#prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#prompt-instance_method)
438
+ delegates to the context it wraps, so an agent accepts a prompt
439
+ wherever it accepts a string. `LLM::Context#build_prompt` is an alias
440
+ kept for compatibility.
441
+
442
+ ### Attachments
443
+
444
+ #### Overview
445
+
446
+ A message can carry files alongside its text. Pass file paths with the
447
+ `with:` option of
448
+ [`LLM::Context#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#ask-instance_method),
449
+ or tag a value explicitly with
450
+ [`LLM::Context#local_file`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#local_file-instance_method),
451
+ [`LLM::Context#image_url`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#image_url-instance_method),
452
+ or
453
+ [`LLM::Context#remote_file`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#remote_file-instance_method).
454
+
455
+ #### How it works
456
+
457
+ `ask` is a shorthand for a turn that returns a response. Pass the
458
+ prompt and, optionally, files to attach with `with:`. It also accepts
459
+ a `stream:` target or a block for streaming:
460
+
461
+ ```ruby
462
+ require "llm"
463
+
464
+ llm = LLM.deepseek(key: ENV["KEY"])
465
+ ctx = LLM::Context.new(llm)
466
+
467
+ res = ctx.ask "What is in this photo?", with: "photo.jpg"
468
+ res = ctx.ask "Summarize these", with: ["one.pdf", "two.pdf"]
469
+ ```
470
+
471
+ The three helpers tag a value so the runtime knows how to send it, for
472
+ turns built with `talk`:
473
+
474
+ ```ruby
475
+ ctx.talk ["Describe this", ctx.local_file("/images/photo.png")]
476
+ ctx.talk ["Describe this", ctx.image_url("https://example.com/photo.png")]
477
+ ctx.talk ["Describe this", ctx.remote_file(res)]
478
+ ```
479
+
480
+ `local_file` reads a path from disk, `image_url` passes a URL the
481
+ provider fetches, and `remote_file` reuses a file a previous response
482
+ produced.
483
+
484
+ #### Why would I use it?
485
+
486
+ Attachments let one turn carry an image, a PDF, or another file for the
487
+ model to read, instead of pasting its contents into the prompt.
488
+
489
+ #### Notes
490
+
491
+ Which files a model accepts depends on the provider and the model.
492
+ `ask` is a shorthand over `talk`: it builds the same prompt and returns
493
+ the same `LLM::Response`, so anything that works with `talk` works with
494
+ `ask`. An agent delegates all four methods to the context it wraps.
495
+
@@ -9,9 +9,8 @@
9
9
  is the superclass for context-level supervisors. A guard is bound
10
10
  to a context and inspects each pending tool call before it runs.
11
11
  It can let the call through, cancel it, block it with an error, or
12
- answer for it with a synthesized result. Beyond loop detection,
13
- guards handle policy, validation, quotas, cost control, caching,
14
- and approval workflows.
12
+ answer for it with a synthesized result. Guards handle policy,
13
+ validation, quotas, cost control, caching, and approval workflows.
15
14
 
16
15
  #### How it works
17
16
 
@@ -74,11 +73,13 @@ onto every function the context binds, so it runs whenever a task is
74
73
  spawned, including tool calls a stream queues itself during a
75
74
  streaming turn. A blocked call yields its return without executing.
76
75
 
77
- The runtime binds a guard instance to the context and stamps it onto
78
- the functions it resolves, so a guard cannot carry state in instance
79
- variables between calls. Anything a guard needs to remember, like
80
- how many calls already ran, must come from the conversation
81
- (`messages`) or from class-level state.
76
+ The runtime creates a guard instance for each batch of tool calls and
77
+ stamps it onto the functions in that batch, so the same instance sees
78
+ every call in a batch. State in an instance variable therefore lasts
79
+ for the batch, but not from one batch to the next. Anything a guard
80
+ needs to remember across batches, like how many calls already ran,
81
+ must come from the conversation (`messages`) or from class-level
82
+ state.
82
83
 
83
84
  ### Inspect
84
85
 
@@ -320,52 +321,3 @@ guard is not a replacement for
320
321
  [`LLM::Agent.tool_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#tool_budget-class_method),
321
322
  which caps the number of tool calls in a single turn. The two
322
323
  compose: the budget caps call count, and a guard enforces cost.
323
-
324
- ### Loop
325
-
326
- #### Overview
327
-
328
- [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
329
- is the built-in loop-detection guard. It reduces each assistant
330
- tool call to a `[tool name, arguments]` signature and checks whether
331
- the tail of the sequence is repeating.
332
-
333
- #### How it works
334
-
335
- When you want to detect repeated tool-call patterns, enable
336
- [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
337
- and tune the `threshold:` option, which is the number of repeated
338
- patterns required before the guard intervenes (default `3`). When
339
- the guard detects a repeat, it returns an in-band
340
- [`LLM::Function::Return`](https://r.uby.dev/api-docs/llm.rb/LLM/Function/Return.html)
341
- with type `"guard_error"` and a message that tells the model it is
342
- stuck and should change approach:
343
-
344
- ```ruby
345
- ctx = LLM::Context.new(
346
- llm,
347
- guard: LLM::Guard::Loop,
348
- guard_options: {threshold: 2}
349
- )
350
- ctx.talk "Research the market", tools: [FetchNews, FetchStocks]
351
- ```
352
-
353
- #### Why would I use it?
354
-
355
- Loop detection matters for long, autonomous agent runs. Without it,
356
- a model that repeats a tool call with the same arguments can
357
- bounce between calls forever. The guard turns that into a bounded
358
- conversation: after the threshold, the model receives a message
359
- telling it to stop and try a different strategy.
360
-
361
- #### Notes
362
-
363
- [`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
364
- enables
365
- [`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
366
- by default, so agents get loop protection without configuration.
367
- A custom guard can be passed through the `guard:` option to replace
368
- the loop guard entirely. Guards and the agent's tool budget
369
- complement each other: a guard blocks work that looks stuck, while
370
- [`LLM::Agent.tool_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#tool_budget-class_method)
371
- caps the total number of tool calls in a single turn.
@@ -126,7 +126,7 @@ method with a `name:`.
126
126
 
127
127
  | Tool | Name | Parameters | Purpose |
128
128
  |---|---|---|---|
129
- | [`LLM::Tool::Rg`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Rg.html) | `rg` | `patterns`, `path`, `timeout` | Recursively search for lines matching patterns |
129
+ | [`LLM::Tool::Rg`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Rg.html) | `rg` | `patterns`, `path`, `timeout`, `max_count`, `max_bytes` | Recursively search for lines matching patterns |
130
130
  | [`LLM::Tool::Which`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Which.html) | `which` | `name` | Locate an executable on the system PATH |
131
131
 
132
132
  #### Why would I use it?
@@ -152,14 +152,13 @@ executable with the given name. When no match is found it returns
152
152
 
153
153
  The command tools run real subprocesses: arbitrary commands through
154
154
  `exec`, Ruby code through `ruby`, commands inside a Bundler context
155
- through `bundle-exec`, and a fixed set of git subcommands through
155
+ through `bundle`, and a fixed set of git subcommands through
156
156
  `git`. All of them accept a `timeout:` and kill the child process
157
157
  when the model interrupts the turn.
158
158
 
159
159
  ```ruby
160
160
  LLM::Tool::Exec.new.call(
161
- name: "bundle",
162
- arguments: ["exec", "rspec", "spec/llm"],
161
+ arguments: ["bundle", "exec", "rspec", "spec/llm"],
163
162
  timeout: 30
164
163
  )
165
164
  ```
@@ -168,14 +167,14 @@ LLM::Tool::Exec.new.call(
168
167
 
169
168
  When you want to run a command and capture its output, call the
170
169
  [`LLM::Tool::Exec#call`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html#call-instance_method)
171
- method with a `name:` and optional `arguments:`. The `git` tool
172
- takes a single `arguments:` array whose first element is a
173
- subcommand from a fixed set, the `ruby` tool runs
174
- its code in a fresh process, and the `bundle-exec` tool runs a
175
- command under the project's Bundler context. `bundle-exec` inherits
176
- the `BUNDLE_GEMFILE` environment variable when set, or defaults to a
177
- `Gemfile` in the current working directory, so the model can run
178
- project tools like `rspec` or `rake` with the right gems loaded:
170
+ method with an `arguments:` array whose first element is the command
171
+ name. The `git` tool takes the same array, with a subcommand from a
172
+ fixed set as its first element, the `ruby` tool runs its code in a
173
+ fresh process, and the `bundle` tool runs a command under the
174
+ project's Bundler context. `bundle` uses the `BUNDLE_GEMFILE`
175
+ environment variable when set, or a `Gemfile` in the current working
176
+ directory otherwise, so the model can run project tools like `rspec`
177
+ or `rake` with the right gems loaded:
179
178
 
180
179
  ```ruby
181
180
  LLM::Tool::Bundle.new.call(
@@ -186,10 +185,10 @@ LLM::Tool::Bundle.new.call(
186
185
 
187
186
  | Tool | Name | Parameters | Purpose |
188
187
  |---|---|---|---|
189
- | [`LLM::Tool::Exec`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html) | `exec` | `name`, `arguments`, `timeout` | Run a command without a shell |
188
+ | [`LLM::Tool::Exec`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html) | `exec` | `arguments`, `timeout` | Run a command without a shell |
190
189
  | [`LLM::Tool::Git`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Git.html) | `git` | `arguments`, `timeout` | Run a fixed set of git subcommands |
191
190
  | [`LLM::Tool::Ruby`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Ruby.html) | `ruby` | `code`, `timeout` | Run a string of Ruby code |
192
- | [`LLM::Tool::BundleExec`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/BundleExec.html) | `bundle-exec` | `name`, `arguments`, `timeout` | Run a command through `bundle exec` |
191
+ | [`LLM::Tool::Bundle`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Bundle.html) | `bundle` | `arguments`, `timeout` | Run a command through `bundle` |
193
192
 
194
193
  #### Why would I use it?
195
194
 
@@ -241,8 +240,7 @@ format the result itself:
241
240
 
242
241
  ```ruby
243
242
  LLM::Tool::Exec.new.call(
244
- name: "bundle",
245
- arguments: ["exec", "rspec"],
243
+ arguments: ["bundle", "exec", "rspec"],
246
244
  max_bytes: 20_000
247
245
  )
248
246
  ```
@@ -58,6 +58,11 @@ The console requires the `curses` and `kramdown` gems. By default the
58
58
  tracer is disabled during the session. Set `tracer: true` to keep
59
59
  it active.
60
60
 
61
+ The loop was previously named the REPL. `LLM::Repl` and
62
+ `LLM::Agent#repl` still resolve to `LLM::Console` and
63
+ `LLM::Agent#console`, so existing code keeps working, but new code
64
+ should use the `console` names.
65
+
61
66
  The user-message label is exposed through
62
67
  [`LLM::Console#sender`](https://r.uby.dev/api-docs/llm.rb/LLM/Console.html#sender-instance_method),
63
68
  which defaults to `"You"`. The