llm.rb 15.2.2 → 15.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +197 -3
- data/README.md +180 -64
- data/bin/llm.rb +9 -2
- data/data/alibaba.json +45 -0
- data/data/anthropic.json +67 -0
- data/data/bedrock.json +1466 -397
- data/data/deepinfra.json +148 -16
- data/data/deepseek.json +3 -0
- data/data/mistral.json +42 -0
- data/data/openai.json +186 -0
- data/data/openrouter.json +1538 -378
- data/data/xai.json +53 -20
- data/data/zai.json +90 -4
- data/docs/deepdive/advanced/compaction.md +1 -2
- data/docs/deepdive/advanced/context.md +214 -1
- data/docs/deepdive/advanced/guard.md +9 -57
- data/docs/deepdive/features/builtin_tools.md +14 -16
- data/docs/deepdive/features/console.md +5 -0
- data/docs/deepdive/features/database.md +85 -10
- data/docs/deepdive/fundamentals/agents.md +7 -8
- data/docs/deepdive/fundamentals/providers.md +45 -5
- data/docs/deepdive/fundamentals/schema.md +73 -0
- data/docs/deepdive/fundamentals/tools.md +80 -27
- data/docs/deepdive/media/audio.md +8 -19
- data/docs/deepdive/media/images.md +8 -10
- data/docs/deepdive/media/ocr.md +1 -3
- data/docs/deepdive/reference/cost.md +48 -0
- data/docs/deepdive/reference/tracer.md +76 -0
- data/docs/deepdive.md +1 -1
- data/lib/llm/active_record/message.rb +113 -0
- data/lib/llm/active_record.rb +1 -0
- data/lib/llm/agent.rb +44 -19
- data/lib/llm/console/buffer.rb +9 -1
- data/lib/llm/console.rb +6 -1
- data/lib/llm/context/deserializer.rb +10 -3
- data/lib/llm/context.rb +62 -26
- data/lib/llm/guard.rb +2 -8
- data/lib/llm/message.rb +18 -7
- data/lib/llm/provider.rb +74 -16
- data/lib/llm/providers/alibaba.rb +15 -0
- data/lib/llm/providers/anthropic/error_handler.rb +5 -2
- data/lib/llm/providers/anthropic/files.rb +12 -12
- data/lib/llm/providers/anthropic/models.rb +2 -2
- data/lib/llm/providers/anthropic.rb +5 -3
- data/lib/llm/providers/bedrock/error_handler.rb +3 -2
- data/lib/llm/providers/bedrock/models.rb +5 -3
- data/lib/llm/providers/bedrock.rb +5 -3
- data/lib/llm/providers/deepinfra/audio.rb +4 -4
- data/lib/llm/providers/deepinfra/images.rb +4 -4
- data/lib/llm/providers/google/error_handler.rb +5 -2
- data/lib/llm/providers/google/files.rb +10 -10
- data/lib/llm/providers/google/images.rb +2 -2
- data/lib/llm/providers/google/models.rb +2 -2
- data/lib/llm/providers/google.rb +7 -7
- data/lib/llm/providers/mistral.rb +3 -1
- data/lib/llm/providers/ollama/error_handler.rb +5 -2
- data/lib/llm/providers/ollama/models.rb +2 -2
- data/lib/llm/providers/ollama.rb +7 -5
- data/lib/llm/providers/openai/audio.rb +6 -6
- data/lib/llm/providers/openai/error_handler.rb +5 -2
- data/lib/llm/providers/openai/files.rb +10 -10
- data/lib/llm/providers/openai/images.rb +4 -4
- data/lib/llm/providers/openai/models.rb +2 -2
- data/lib/llm/providers/openai/moderations.rb +2 -2
- data/lib/llm/providers/openai/request_adapter.rb +1 -1
- data/lib/llm/providers/openai/responses.rb +8 -8
- data/lib/llm/providers/openai/vector_stores.rb +22 -22
- data/lib/llm/providers/openai.rb +10 -8
- data/lib/llm/providers/xai/images.rb +4 -4
- data/lib/llm/schema.rb +24 -0
- data/lib/llm/tracer/telemetry.rb +4 -4
- data/lib/llm/tracer.rb +11 -3
- data/lib/llm/transport/execution.rb +8 -4
- data/lib/llm/utils.rb +13 -0
- data/lib/llm/version.rb +1 -1
- data/llm.gemspec +2 -2
- metadata +5 -5
- data/lib/llm/guard/loop.rb +0 -89
data/data/xai.json
CHANGED
|
@@ -370,17 +370,30 @@
|
|
|
370
370
|
"output": 0
|
|
371
371
|
}
|
|
372
372
|
},
|
|
373
|
-
"grok-
|
|
374
|
-
"id": "grok-
|
|
375
|
-
"name": "Grok
|
|
376
|
-
"description": "
|
|
373
|
+
"grok-4.6": {
|
|
374
|
+
"id": "grok-4.6",
|
|
375
|
+
"name": "Grok 4.6",
|
|
376
|
+
"description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
|
|
377
377
|
"family": "grok",
|
|
378
378
|
"attachment": true,
|
|
379
|
-
"reasoning":
|
|
380
|
-
"
|
|
381
|
-
|
|
382
|
-
|
|
383
|
-
|
|
379
|
+
"reasoning": true,
|
|
380
|
+
"reasoning_options": [
|
|
381
|
+
{
|
|
382
|
+
"type": "effort",
|
|
383
|
+
"values": [
|
|
384
|
+
"low",
|
|
385
|
+
"medium",
|
|
386
|
+
"high",
|
|
387
|
+
"xhigh"
|
|
388
|
+
]
|
|
389
|
+
}
|
|
390
|
+
],
|
|
391
|
+
"tool_call": true,
|
|
392
|
+
"structured_output": true,
|
|
393
|
+
"temperature": true,
|
|
394
|
+
"knowledge": "2026-02-01",
|
|
395
|
+
"release_date": "2026-08-12",
|
|
396
|
+
"last_updated": "2026-08-12",
|
|
384
397
|
"modalities": {
|
|
385
398
|
"input": [
|
|
386
399
|
"text",
|
|
@@ -388,19 +401,39 @@
|
|
|
388
401
|
"pdf"
|
|
389
402
|
],
|
|
390
403
|
"output": [
|
|
391
|
-
"
|
|
392
|
-
"pdf"
|
|
404
|
+
"text"
|
|
393
405
|
]
|
|
394
406
|
},
|
|
395
407
|
"open_weights": false,
|
|
396
408
|
"limit": {
|
|
397
|
-
"context":
|
|
398
|
-
"output":
|
|
409
|
+
"context": 500000,
|
|
410
|
+
"output": 500000
|
|
411
|
+
},
|
|
412
|
+
"cost": {
|
|
413
|
+
"input": 2,
|
|
414
|
+
"output": 6,
|
|
415
|
+
"cache_read": 0.5,
|
|
416
|
+
"tiers": [
|
|
417
|
+
{
|
|
418
|
+
"input": 4,
|
|
419
|
+
"output": 12,
|
|
420
|
+
"cache_read": 1,
|
|
421
|
+
"tier": {
|
|
422
|
+
"type": "context",
|
|
423
|
+
"size": 200000
|
|
424
|
+
}
|
|
425
|
+
}
|
|
426
|
+
],
|
|
427
|
+
"context_over_200k": {
|
|
428
|
+
"input": 4,
|
|
429
|
+
"output": 12,
|
|
430
|
+
"cache_read": 1
|
|
431
|
+
}
|
|
399
432
|
}
|
|
400
433
|
},
|
|
401
|
-
"grok-4.
|
|
402
|
-
"id": "grok-4.
|
|
403
|
-
"name": "Grok 4.
|
|
434
|
+
"grok-4.7": {
|
|
435
|
+
"id": "grok-4.7",
|
|
436
|
+
"name": "Grok 4.7",
|
|
404
437
|
"description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
|
|
405
438
|
"family": "grok",
|
|
406
439
|
"attachment": true,
|
|
@@ -419,9 +452,9 @@
|
|
|
419
452
|
"tool_call": true,
|
|
420
453
|
"structured_output": true,
|
|
421
454
|
"temperature": true,
|
|
422
|
-
"knowledge": "2026-
|
|
423
|
-
"release_date": "2026-
|
|
424
|
-
"last_updated": "2026-
|
|
455
|
+
"knowledge": "2026-05",
|
|
456
|
+
"release_date": "2026-09-21",
|
|
457
|
+
"last_updated": "2026-09-21",
|
|
425
458
|
"modalities": {
|
|
426
459
|
"input": [
|
|
427
460
|
"text",
|
|
@@ -462,7 +495,7 @@
|
|
|
462
495
|
"grok-imagine-image-quality": {
|
|
463
496
|
"id": "grok-imagine-image-quality",
|
|
464
497
|
"name": "Grok Imagine Image Quality",
|
|
465
|
-
"description": "
|
|
498
|
+
"description": "Higher-fidelity Grok Imagine image model for prompt-driven generation, editing, and visual design workflows",
|
|
466
499
|
"family": "grok",
|
|
467
500
|
"attachment": true,
|
|
468
501
|
"reasoning": false,
|
data/data/zai.json
CHANGED
|
@@ -240,11 +240,59 @@
|
|
|
240
240
|
"cache_write": 0
|
|
241
241
|
}
|
|
242
242
|
},
|
|
243
|
+
"glm-5.3-flashx": {
|
|
244
|
+
"id": "glm-5.3-flashx",
|
|
245
|
+
"name": "GLM-5.3-FlashX",
|
|
246
|
+
"description": "High-speed GLM-5.3-Flash serving option for coding and agent workflows",
|
|
247
|
+
"family": "glm-flash",
|
|
248
|
+
"attachment": true,
|
|
249
|
+
"reasoning": true,
|
|
250
|
+
"reasoning_options": [
|
|
251
|
+
{
|
|
252
|
+
"type": "effort",
|
|
253
|
+
"values": [
|
|
254
|
+
"low",
|
|
255
|
+
"high",
|
|
256
|
+
"max"
|
|
257
|
+
]
|
|
258
|
+
}
|
|
259
|
+
],
|
|
260
|
+
"tool_call": true,
|
|
261
|
+
"interleaved": {
|
|
262
|
+
"field": "reasoning_content"
|
|
263
|
+
},
|
|
264
|
+
"structured_output": true,
|
|
265
|
+
"temperature": true,
|
|
266
|
+
"release_date": "2026-09-18",
|
|
267
|
+
"last_updated": "2026-09-18",
|
|
268
|
+
"modalities": {
|
|
269
|
+
"input": [
|
|
270
|
+
"text",
|
|
271
|
+
"image",
|
|
272
|
+
"video",
|
|
273
|
+
"pdf"
|
|
274
|
+
],
|
|
275
|
+
"output": [
|
|
276
|
+
"text"
|
|
277
|
+
]
|
|
278
|
+
},
|
|
279
|
+
"open_weights": true,
|
|
280
|
+
"limit": {
|
|
281
|
+
"context": 1000000,
|
|
282
|
+
"output": 131072
|
|
283
|
+
},
|
|
284
|
+
"cost": {
|
|
285
|
+
"input": 0.37,
|
|
286
|
+
"output": 1.25,
|
|
287
|
+
"cache_read": 0.075,
|
|
288
|
+
"cache_write": 0
|
|
289
|
+
}
|
|
290
|
+
},
|
|
243
291
|
"glm-5.3-flash": {
|
|
244
292
|
"id": "glm-5.3-flash",
|
|
245
293
|
"name": "GLM-5.3-Flash",
|
|
246
294
|
"description": "Native multimodal GLM model for efficient coding and long-horizon agent tasks",
|
|
247
|
-
"family": "glm",
|
|
295
|
+
"family": "glm-flash",
|
|
248
296
|
"attachment": true,
|
|
249
297
|
"reasoning": true,
|
|
250
298
|
"reasoning_options": [
|
|
@@ -282,9 +330,9 @@
|
|
|
282
330
|
"output": 131072
|
|
283
331
|
},
|
|
284
332
|
"cost": {
|
|
285
|
-
"input": 0.
|
|
286
|
-
"output": 0.
|
|
287
|
-
"cache_read": 0.
|
|
333
|
+
"input": 0.15,
|
|
334
|
+
"output": 0.5,
|
|
335
|
+
"cache_read": 0.03,
|
|
288
336
|
"cache_write": 0
|
|
289
337
|
}
|
|
290
338
|
},
|
|
@@ -605,6 +653,44 @@
|
|
|
605
653
|
"cache_write": 0
|
|
606
654
|
}
|
|
607
655
|
},
|
|
656
|
+
"glm-4.6v-flash": {
|
|
657
|
+
"id": "glm-4.6v-flash",
|
|
658
|
+
"name": "GLM-4.6V-Flash",
|
|
659
|
+
"description": "Lightweight GLM vision model for visual reasoning, documents, and multimodal agents",
|
|
660
|
+
"family": "glm-flash",
|
|
661
|
+
"attachment": true,
|
|
662
|
+
"reasoning": true,
|
|
663
|
+
"reasoning_options": [
|
|
664
|
+
{
|
|
665
|
+
"type": "toggle"
|
|
666
|
+
}
|
|
667
|
+
],
|
|
668
|
+
"tool_call": true,
|
|
669
|
+
"temperature": true,
|
|
670
|
+
"release_date": "2025-12-08",
|
|
671
|
+
"last_updated": "2025-12-08",
|
|
672
|
+
"modalities": {
|
|
673
|
+
"input": [
|
|
674
|
+
"text",
|
|
675
|
+
"image",
|
|
676
|
+
"video"
|
|
677
|
+
],
|
|
678
|
+
"output": [
|
|
679
|
+
"text"
|
|
680
|
+
]
|
|
681
|
+
},
|
|
682
|
+
"open_weights": true,
|
|
683
|
+
"limit": {
|
|
684
|
+
"context": 128000,
|
|
685
|
+
"output": 32768
|
|
686
|
+
},
|
|
687
|
+
"cost": {
|
|
688
|
+
"input": 0,
|
|
689
|
+
"output": 0,
|
|
690
|
+
"cache_read": 0,
|
|
691
|
+
"cache_write": 0
|
|
692
|
+
}
|
|
693
|
+
},
|
|
608
694
|
"glm-4.7-flash": {
|
|
609
695
|
"id": "glm-4.7-flash",
|
|
610
696
|
"name": "GLM-4.7-Flash",
|
|
@@ -47,11 +47,10 @@ compactor.call(keep: 200)
|
|
|
47
47
|
|
|
48
48
|
##### The `/keep` command
|
|
49
49
|
|
|
50
|
-
The console provides a `/keep` command that
|
|
50
|
+
The console provides a `/keep` command that requires a count or
|
|
51
51
|
percentage:
|
|
52
52
|
|
|
53
53
|
```ruby
|
|
54
|
-
# /keep # keep last 128 messages
|
|
55
54
|
# /keep 50 # keep last 50 messages
|
|
56
55
|
# /keep 75% # keep approximately 75% of messages
|
|
57
56
|
```
|
|
@@ -82,7 +82,60 @@ Each retry sleeps a growing interval (2s, 4s, 6s, ...) and notifies
|
|
|
82
82
|
the stream through
|
|
83
83
|
[`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry-instance_method)
|
|
84
84
|
before trying again. An `LLM::Agent` enables a budget of 5 by
|
|
85
|
-
default
|
|
85
|
+
default (or 8 on the Alibaba provider, which rate limits more
|
|
86
|
+
often), so most users never touch this directly.
|
|
87
|
+
|
|
88
|
+
### Identity
|
|
89
|
+
|
|
90
|
+
#### Overview
|
|
91
|
+
|
|
92
|
+
A context has an id and a creation time, and so does each message it
|
|
93
|
+
holds. The id is a UUIDv7 string, a UUID version that encodes its own
|
|
94
|
+
creation timestamp, so an id sorts by the order it was created and
|
|
95
|
+
carries the time it was made.
|
|
96
|
+
|
|
97
|
+
#### How it works
|
|
98
|
+
|
|
99
|
+
Read the id and creation time from a context, an agent, or a message:
|
|
100
|
+
|
|
101
|
+
```ruby
|
|
102
|
+
require "llm"
|
|
103
|
+
|
|
104
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
105
|
+
ctx = LLM::Context.new(llm)
|
|
106
|
+
ctx.talk "Hello"
|
|
107
|
+
ctx.id # => "01932f5a-..." (UUIDv7)
|
|
108
|
+
ctx.created_at # => 2026-09-11 04:21:07 UTC
|
|
109
|
+
ctx.messages.first.id
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
[`LLM::Agent#id`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#id-instance_method)
|
|
113
|
+
and
|
|
114
|
+
[`LLM::Agent#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#created_at-instance_method)
|
|
115
|
+
delegate to the context the agent wraps.
|
|
116
|
+
[`LLM::Context#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#created_at-instance_method)
|
|
117
|
+
and
|
|
118
|
+
[`LLM::Message#created_at`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html#created_at-instance_method)
|
|
119
|
+
are read from the id rather than stored, so they return `nil` when
|
|
120
|
+
the id is not a UUIDv7 string.
|
|
121
|
+
|
|
122
|
+
#### Why would I use it?
|
|
123
|
+
|
|
124
|
+
An id gives a conversation or a message a stable name you can log,
|
|
125
|
+
correlate, or look up. Because the id is a UUIDv7, the same value
|
|
126
|
+
also answers when the object was created, so sorting ids sorts by
|
|
127
|
+
creation order and no separate timestamp column is needed.
|
|
128
|
+
|
|
129
|
+
#### Notes
|
|
130
|
+
|
|
131
|
+
The id is generated once and saved with the runtime state, so it
|
|
132
|
+
survives a save and a restore. A payload written before ids existed
|
|
133
|
+
has none, so its object is restored with a fresh id.
|
|
134
|
+
[`LLM::Message#==`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html#==-instance_method)
|
|
135
|
+
ignores the id, so a difference in creation time alone does not make
|
|
136
|
+
two messages unequal.
|
|
137
|
+
[`LLM::Utils.timestamp`](https://r.uby.dev/api-docs/llm.rb/LLM/Utils.html#timestamp-instance_method)
|
|
138
|
+
is the shared method that decodes a UUIDv7 timestamp.
|
|
86
139
|
|
|
87
140
|
### Manual loop
|
|
88
141
|
|
|
@@ -280,3 +333,163 @@ The mechanism is the same across all six concurrency strategies.
|
|
|
280
333
|
The `:ractor` strategy delivers the interrupt through ractor
|
|
281
334
|
message passing. The `:fork` strategy delivers it via xchan.
|
|
282
335
|
|
|
336
|
+
### Messages
|
|
337
|
+
|
|
338
|
+
#### Overview
|
|
339
|
+
|
|
340
|
+
[`LLM::Context#messages`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#messages-instance_method)
|
|
341
|
+
returns an
|
|
342
|
+
[`LLM::Buffer`](https://r.uby.dev/api-docs/llm.rb/LLM/Buffer.html),
|
|
343
|
+
an ordered, array-like collection of the conversation's
|
|
344
|
+
[`LLM::Message`](https://r.uby.dev/api-docs/llm.rb/LLM/Message.html)
|
|
345
|
+
objects. Read it to inspect or filter a conversation, and edit it to
|
|
346
|
+
shape what the model sees next.
|
|
347
|
+
|
|
348
|
+
#### How it works
|
|
349
|
+
|
|
350
|
+
A buffer is `Enumerable`, so it supports `each`, `find`, `map`,
|
|
351
|
+
`select`, and the rest. It also offers the array methods a long
|
|
352
|
+
conversation needs, including `first`, `last`, `take`, `drop`,
|
|
353
|
+
`shift`, `pop`, `slice!`, `select!`, `reject!`, and `clear`:
|
|
354
|
+
|
|
355
|
+
```ruby
|
|
356
|
+
require "llm"
|
|
357
|
+
|
|
358
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
359
|
+
ctx = LLM::Context.new(llm)
|
|
360
|
+
ctx.talk "Hello"
|
|
361
|
+
|
|
362
|
+
ctx.messages.size # => 2
|
|
363
|
+
ctx.messages.first # => the user message
|
|
364
|
+
ctx.messages.last # => the assistant message
|
|
365
|
+
ctx.messages.each { |m| puts "#{m.role}: #{m.content}" }
|
|
366
|
+
```
|
|
367
|
+
|
|
368
|
+
Because a message carries its own role, filtering by role is a normal
|
|
369
|
+
`select`:
|
|
370
|
+
|
|
371
|
+
```ruby
|
|
372
|
+
ctx.messages.select(&:assistant?)
|
|
373
|
+
```
|
|
374
|
+
|
|
375
|
+
#### Why would I use it?
|
|
376
|
+
|
|
377
|
+
Reading the buffer gives you the conversation as data, so you can log
|
|
378
|
+
it, count tokens against it, or render it in your own UI. Editing the
|
|
379
|
+
buffer lets you drop or keep specific messages without rebuilding the
|
|
380
|
+
conversation.
|
|
381
|
+
|
|
382
|
+
#### Notes
|
|
383
|
+
|
|
384
|
+
Changing the buffer changes the next request, so edits are best made
|
|
385
|
+
between turns. The compaction topic covers the built-in, bounded way
|
|
386
|
+
to trim a long conversation.
|
|
387
|
+
|
|
388
|
+
### Prompt
|
|
389
|
+
|
|
390
|
+
#### Overview
|
|
391
|
+
|
|
392
|
+
[`LLM::Prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Prompt.html)
|
|
393
|
+
composes a single request from several role-aware messages. A prompt
|
|
394
|
+
is not just a string: it is an ordered list of messages with explicit
|
|
395
|
+
roles, so one turn can carry a system message, a user message, and
|
|
396
|
+
anything else the model supports.
|
|
397
|
+
|
|
398
|
+
#### How it works
|
|
399
|
+
|
|
400
|
+
Call
|
|
401
|
+
[`LLM::Context#prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#prompt-instance_method)
|
|
402
|
+
with a block, then pass the result to
|
|
403
|
+
[`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk).
|
|
404
|
+
Inside the block, `system`, `user`, and `developer` append a message
|
|
405
|
+
with the matching role, and `talk` appends one with an explicit
|
|
406
|
+
role. The provider resolves each role to its provider-specific name:
|
|
407
|
+
|
|
408
|
+
```ruby
|
|
409
|
+
require "llm"
|
|
410
|
+
|
|
411
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
412
|
+
ctx = LLM::Context.new(llm)
|
|
413
|
+
|
|
414
|
+
prompt = ctx.prompt do
|
|
415
|
+
system "Your task is to assist the user"
|
|
416
|
+
user "Hello. Can you assist me?"
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
res = ctx.talk(prompt)
|
|
420
|
+
```
|
|
421
|
+
|
|
422
|
+
The block receives the prompt object when it takes an argument, and
|
|
423
|
+
otherwise runs in the prompt's context:
|
|
424
|
+
[`LLM::Prompt#to_a`](https://r.uby.dev/api-docs/llm.rb/LLM/Prompt.html#to_a)
|
|
425
|
+
returns the messages in order, and two prompts are equal when their
|
|
426
|
+
messages match.
|
|
427
|
+
|
|
428
|
+
#### Why would I use it?
|
|
429
|
+
|
|
430
|
+
A prompt keeps the roles of a multi-part request explicit, and it is
|
|
431
|
+
an object you can build, pass around, and compare before it is sent.
|
|
432
|
+
Use it when a turn needs more than one role, or when the same prompt
|
|
433
|
+
is composed in more than one place.
|
|
434
|
+
|
|
435
|
+
#### Notes
|
|
436
|
+
|
|
437
|
+
[`LLM::Agent#prompt`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#prompt-instance_method)
|
|
438
|
+
delegates to the context it wraps, so an agent accepts a prompt
|
|
439
|
+
wherever it accepts a string. `LLM::Context#build_prompt` is an alias
|
|
440
|
+
kept for compatibility.
|
|
441
|
+
|
|
442
|
+
### Attachments
|
|
443
|
+
|
|
444
|
+
#### Overview
|
|
445
|
+
|
|
446
|
+
A message can carry files alongside its text. Pass file paths with the
|
|
447
|
+
`with:` option of
|
|
448
|
+
[`LLM::Context#ask`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#ask-instance_method),
|
|
449
|
+
or tag a value explicitly with
|
|
450
|
+
[`LLM::Context#local_file`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#local_file-instance_method),
|
|
451
|
+
[`LLM::Context#image_url`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#image_url-instance_method),
|
|
452
|
+
or
|
|
453
|
+
[`LLM::Context#remote_file`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#remote_file-instance_method).
|
|
454
|
+
|
|
455
|
+
#### How it works
|
|
456
|
+
|
|
457
|
+
`ask` is a shorthand for a turn that returns a response. Pass the
|
|
458
|
+
prompt and, optionally, files to attach with `with:`. It also accepts
|
|
459
|
+
a `stream:` target or a block for streaming:
|
|
460
|
+
|
|
461
|
+
```ruby
|
|
462
|
+
require "llm"
|
|
463
|
+
|
|
464
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
465
|
+
ctx = LLM::Context.new(llm)
|
|
466
|
+
|
|
467
|
+
res = ctx.ask "What is in this photo?", with: "photo.jpg"
|
|
468
|
+
res = ctx.ask "Summarize these", with: ["one.pdf", "two.pdf"]
|
|
469
|
+
```
|
|
470
|
+
|
|
471
|
+
The three helpers tag a value so the runtime knows how to send it, for
|
|
472
|
+
turns built with `talk`:
|
|
473
|
+
|
|
474
|
+
```ruby
|
|
475
|
+
ctx.talk ["Describe this", ctx.local_file("/images/photo.png")]
|
|
476
|
+
ctx.talk ["Describe this", ctx.image_url("https://example.com/photo.png")]
|
|
477
|
+
ctx.talk ["Describe this", ctx.remote_file(res)]
|
|
478
|
+
```
|
|
479
|
+
|
|
480
|
+
`local_file` reads a path from disk, `image_url` passes a URL the
|
|
481
|
+
provider fetches, and `remote_file` reuses a file a previous response
|
|
482
|
+
produced.
|
|
483
|
+
|
|
484
|
+
#### Why would I use it?
|
|
485
|
+
|
|
486
|
+
Attachments let one turn carry an image, a PDF, or another file for the
|
|
487
|
+
model to read, instead of pasting its contents into the prompt.
|
|
488
|
+
|
|
489
|
+
#### Notes
|
|
490
|
+
|
|
491
|
+
Which files a model accepts depends on the provider and the model.
|
|
492
|
+
`ask` is a shorthand over `talk`: it builds the same prompt and returns
|
|
493
|
+
the same `LLM::Response`, so anything that works with `talk` works with
|
|
494
|
+
`ask`. An agent delegates all four methods to the context it wraps.
|
|
495
|
+
|
|
@@ -9,9 +9,8 @@
|
|
|
9
9
|
is the superclass for context-level supervisors. A guard is bound
|
|
10
10
|
to a context and inspects each pending tool call before it runs.
|
|
11
11
|
It can let the call through, cancel it, block it with an error, or
|
|
12
|
-
answer for it with a synthesized result.
|
|
13
|
-
|
|
14
|
-
and approval workflows.
|
|
12
|
+
answer for it with a synthesized result. Guards handle policy,
|
|
13
|
+
validation, quotas, cost control, caching, and approval workflows.
|
|
15
14
|
|
|
16
15
|
#### How it works
|
|
17
16
|
|
|
@@ -74,11 +73,13 @@ onto every function the context binds, so it runs whenever a task is
|
|
|
74
73
|
spawned, including tool calls a stream queues itself during a
|
|
75
74
|
streaming turn. A blocked call yields its return without executing.
|
|
76
75
|
|
|
77
|
-
The runtime
|
|
78
|
-
the functions
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
76
|
+
The runtime creates a guard instance for each batch of tool calls and
|
|
77
|
+
stamps it onto the functions in that batch, so the same instance sees
|
|
78
|
+
every call in a batch. State in an instance variable therefore lasts
|
|
79
|
+
for the batch, but not from one batch to the next. Anything a guard
|
|
80
|
+
needs to remember across batches, like how many calls already ran,
|
|
81
|
+
must come from the conversation (`messages`) or from class-level
|
|
82
|
+
state.
|
|
82
83
|
|
|
83
84
|
### Inspect
|
|
84
85
|
|
|
@@ -320,52 +321,3 @@ guard is not a replacement for
|
|
|
320
321
|
[`LLM::Agent.tool_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#tool_budget-class_method),
|
|
321
322
|
which caps the number of tool calls in a single turn. The two
|
|
322
323
|
compose: the budget caps call count, and a guard enforces cost.
|
|
323
|
-
|
|
324
|
-
### Loop
|
|
325
|
-
|
|
326
|
-
#### Overview
|
|
327
|
-
|
|
328
|
-
[`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
|
|
329
|
-
is the built-in loop-detection guard. It reduces each assistant
|
|
330
|
-
tool call to a `[tool name, arguments]` signature and checks whether
|
|
331
|
-
the tail of the sequence is repeating.
|
|
332
|
-
|
|
333
|
-
#### How it works
|
|
334
|
-
|
|
335
|
-
When you want to detect repeated tool-call patterns, enable
|
|
336
|
-
[`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
|
|
337
|
-
and tune the `threshold:` option, which is the number of repeated
|
|
338
|
-
patterns required before the guard intervenes (default `3`). When
|
|
339
|
-
the guard detects a repeat, it returns an in-band
|
|
340
|
-
[`LLM::Function::Return`](https://r.uby.dev/api-docs/llm.rb/LLM/Function/Return.html)
|
|
341
|
-
with type `"guard_error"` and a message that tells the model it is
|
|
342
|
-
stuck and should change approach:
|
|
343
|
-
|
|
344
|
-
```ruby
|
|
345
|
-
ctx = LLM::Context.new(
|
|
346
|
-
llm,
|
|
347
|
-
guard: LLM::Guard::Loop,
|
|
348
|
-
guard_options: {threshold: 2}
|
|
349
|
-
)
|
|
350
|
-
ctx.talk "Research the market", tools: [FetchNews, FetchStocks]
|
|
351
|
-
```
|
|
352
|
-
|
|
353
|
-
#### Why would I use it?
|
|
354
|
-
|
|
355
|
-
Loop detection matters for long, autonomous agent runs. Without it,
|
|
356
|
-
a model that repeats a tool call with the same arguments can
|
|
357
|
-
bounce between calls forever. The guard turns that into a bounded
|
|
358
|
-
conversation: after the threshold, the model receives a message
|
|
359
|
-
telling it to stop and try a different strategy.
|
|
360
|
-
|
|
361
|
-
#### Notes
|
|
362
|
-
|
|
363
|
-
[`LLM::Agent`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html)
|
|
364
|
-
enables
|
|
365
|
-
[`LLM::Guard::Loop`](https://r.uby.dev/api-docs/llm.rb/LLM/Guard/Loop.html)
|
|
366
|
-
by default, so agents get loop protection without configuration.
|
|
367
|
-
A custom guard can be passed through the `guard:` option to replace
|
|
368
|
-
the loop guard entirely. Guards and the agent's tool budget
|
|
369
|
-
complement each other: a guard blocks work that looks stuck, while
|
|
370
|
-
[`LLM::Agent.tool_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#tool_budget-class_method)
|
|
371
|
-
caps the total number of tool calls in a single turn.
|
|
@@ -126,7 +126,7 @@ method with a `name:`.
|
|
|
126
126
|
|
|
127
127
|
| Tool | Name | Parameters | Purpose |
|
|
128
128
|
|---|---|---|---|
|
|
129
|
-
| [`LLM::Tool::Rg`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Rg.html) | `rg` | `patterns`, `path`, `timeout` | Recursively search for lines matching patterns |
|
|
129
|
+
| [`LLM::Tool::Rg`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Rg.html) | `rg` | `patterns`, `path`, `timeout`, `max_count`, `max_bytes` | Recursively search for lines matching patterns |
|
|
130
130
|
| [`LLM::Tool::Which`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Which.html) | `which` | `name` | Locate an executable on the system PATH |
|
|
131
131
|
|
|
132
132
|
#### Why would I use it?
|
|
@@ -152,14 +152,13 @@ executable with the given name. When no match is found it returns
|
|
|
152
152
|
|
|
153
153
|
The command tools run real subprocesses: arbitrary commands through
|
|
154
154
|
`exec`, Ruby code through `ruby`, commands inside a Bundler context
|
|
155
|
-
through `bundle
|
|
155
|
+
through `bundle`, and a fixed set of git subcommands through
|
|
156
156
|
`git`. All of them accept a `timeout:` and kill the child process
|
|
157
157
|
when the model interrupts the turn.
|
|
158
158
|
|
|
159
159
|
```ruby
|
|
160
160
|
LLM::Tool::Exec.new.call(
|
|
161
|
-
|
|
162
|
-
arguments: ["exec", "rspec", "spec/llm"],
|
|
161
|
+
arguments: ["bundle", "exec", "rspec", "spec/llm"],
|
|
163
162
|
timeout: 30
|
|
164
163
|
)
|
|
165
164
|
```
|
|
@@ -168,14 +167,14 @@ LLM::Tool::Exec.new.call(
|
|
|
168
167
|
|
|
169
168
|
When you want to run a command and capture its output, call the
|
|
170
169
|
[`LLM::Tool::Exec#call`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html#call-instance_method)
|
|
171
|
-
method with
|
|
172
|
-
takes
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
170
|
+
method with an `arguments:` array whose first element is the command
|
|
171
|
+
name. The `git` tool takes the same array, with a subcommand from a
|
|
172
|
+
fixed set as its first element, the `ruby` tool runs its code in a
|
|
173
|
+
fresh process, and the `bundle` tool runs a command under the
|
|
174
|
+
project's Bundler context. `bundle` uses the `BUNDLE_GEMFILE`
|
|
175
|
+
environment variable when set, or a `Gemfile` in the current working
|
|
176
|
+
directory otherwise, so the model can run project tools like `rspec`
|
|
177
|
+
or `rake` with the right gems loaded:
|
|
179
178
|
|
|
180
179
|
```ruby
|
|
181
180
|
LLM::Tool::Bundle.new.call(
|
|
@@ -186,10 +185,10 @@ LLM::Tool::Bundle.new.call(
|
|
|
186
185
|
|
|
187
186
|
| Tool | Name | Parameters | Purpose |
|
|
188
187
|
|---|---|---|---|
|
|
189
|
-
| [`LLM::Tool::Exec`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html) | `exec` | `
|
|
188
|
+
| [`LLM::Tool::Exec`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Exec.html) | `exec` | `arguments`, `timeout` | Run a command without a shell |
|
|
190
189
|
| [`LLM::Tool::Git`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Git.html) | `git` | `arguments`, `timeout` | Run a fixed set of git subcommands |
|
|
191
190
|
| [`LLM::Tool::Ruby`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Ruby.html) | `ruby` | `code`, `timeout` | Run a string of Ruby code |
|
|
192
|
-
| [`LLM::Tool::
|
|
191
|
+
| [`LLM::Tool::Bundle`](https://r.uby.dev/api-docs/llm.rb/LLM/Tool/Bundle.html) | `bundle` | `arguments`, `timeout` | Run a command through `bundle` |
|
|
193
192
|
|
|
194
193
|
#### Why would I use it?
|
|
195
194
|
|
|
@@ -241,8 +240,7 @@ format the result itself:
|
|
|
241
240
|
|
|
242
241
|
```ruby
|
|
243
242
|
LLM::Tool::Exec.new.call(
|
|
244
|
-
|
|
245
|
-
arguments: ["exec", "rspec"],
|
|
243
|
+
arguments: ["bundle", "exec", "rspec"],
|
|
246
244
|
max_bytes: 20_000
|
|
247
245
|
)
|
|
248
246
|
```
|
|
@@ -58,6 +58,11 @@ The console requires the `curses` and `kramdown` gems. By default the
|
|
|
58
58
|
tracer is disabled during the session. Set `tracer: true` to keep
|
|
59
59
|
it active.
|
|
60
60
|
|
|
61
|
+
The loop was previously named the REPL. `LLM::Repl` and
|
|
62
|
+
`LLM::Agent#repl` still resolve to `LLM::Console` and
|
|
63
|
+
`LLM::Agent#console`, so existing code keeps working, but new code
|
|
64
|
+
should use the `console` names.
|
|
65
|
+
|
|
61
66
|
The user-message label is exposed through
|
|
62
67
|
[`LLM::Console#sender`](https://r.uby.dev/api-docs/llm.rb/LLM/Console.html#sender-instance_method),
|
|
63
68
|
which defaults to `"You"`. The
|