llm.rb 12.3.1 → 12.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +395 -0
  3. data/README.md +103 -16
  4. data/data/anthropic.json +249 -249
  5. data/data/bedrock.json +2038 -1879
  6. data/data/deepinfra.json +591 -591
  7. data/data/deepseek.json +31 -31
  8. data/data/google.json +332 -329
  9. data/data/mistral.json +381 -381
  10. data/data/openai.json +1132 -1132
  11. data/data/xai.json +138 -138
  12. data/data/zai.json +165 -165
  13. data/lib/llm/agent.rb +12 -7
  14. data/lib/llm/buffer.rb +15 -0
  15. data/lib/llm/context/deserializer.rb +2 -5
  16. data/lib/llm/context.rb +4 -5
  17. data/lib/llm/function.rb +2 -35
  18. data/lib/llm/message.rb +12 -2
  19. data/lib/llm/provider.rb +11 -1
  20. data/lib/llm/providers/anthropic/stream_parser.rb +3 -12
  21. data/lib/llm/providers/anthropic.rb +11 -0
  22. data/lib/llm/providers/bedrock/stream_parser.rb +4 -13
  23. data/lib/llm/providers/google/stream_parser.rb +3 -12
  24. data/lib/llm/providers/google.rb +7 -0
  25. data/lib/llm/providers/mistral.rb +11 -0
  26. data/lib/llm/providers/ollama/stream_parser.rb +4 -4
  27. data/lib/llm/providers/ollama.rb +11 -0
  28. data/lib/llm/providers/openai/responses/stream_parser.rb +4 -14
  29. data/lib/llm/providers/openai/responses.rb +10 -0
  30. data/lib/llm/providers/openai/stream_parser.rb +5 -15
  31. data/lib/llm/providers/openai.rb +11 -0
  32. data/lib/llm/repl/command.rb +201 -0
  33. data/lib/llm/repl/commands/exit.rb +23 -0
  34. data/lib/llm/repl/commands/help.rb +24 -0
  35. data/lib/llm/repl/input.rb +118 -35
  36. data/lib/llm/repl/stream.rb +43 -8
  37. data/lib/llm/repl/transcript.rb +6 -0
  38. data/lib/llm/repl/window.rb +28 -6
  39. data/lib/llm/repl.rb +125 -27
  40. data/lib/llm/stream.rb +2 -7
  41. data/lib/llm/tools/ls.rb +30 -0
  42. data/lib/llm/tools/which.rb +38 -0
  43. data/lib/llm/version.rb +1 -1
  44. data/resources/deepdive.md +67 -1
  45. metadata +6 -1
data/data/zai.json CHANGED
@@ -48,43 +48,6 @@
48
48
  "cache_write": 0
49
49
  }
50
50
  },
51
- "glm-4.5v": {
52
- "id": "glm-4.5v",
53
- "name": "GLM-4.5V",
54
- "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
55
- "family": "glm",
56
- "attachment": true,
57
- "reasoning": true,
58
- "reasoning_options": [
59
- {
60
- "type": "toggle"
61
- }
62
- ],
63
- "tool_call": true,
64
- "temperature": true,
65
- "knowledge": "2025-04",
66
- "release_date": "2025-08-11",
67
- "last_updated": "2025-08-11",
68
- "modalities": {
69
- "input": [
70
- "text",
71
- "image",
72
- "video"
73
- ],
74
- "output": [
75
- "text"
76
- ]
77
- },
78
- "open_weights": true,
79
- "limit": {
80
- "context": 64000,
81
- "output": 16384
82
- },
83
- "cost": {
84
- "input": 0.6,
85
- "output": 1.8
86
- }
87
- },
88
51
  "glm-4.5": {
89
52
  "id": "glm-4.5",
90
53
  "name": "GLM-4.5",
@@ -122,11 +85,11 @@
122
85
  "cache_write": 0
123
86
  }
124
87
  },
125
- "glm-4.7-flashx": {
126
- "id": "glm-4.7-flashx",
127
- "name": "GLM-4.7-FlashX",
128
- "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
129
- "family": "glm-flash",
88
+ "glm-5-turbo": {
89
+ "id": "glm-5-turbo",
90
+ "name": "GLM-5-Turbo",
91
+ "description": "Faster GLM-5 lane for coding agents that need lower latency",
92
+ "family": "glm",
130
93
  "attachment": false,
131
94
  "reasoning": true,
132
95
  "reasoning_options": [
@@ -135,10 +98,13 @@
135
98
  }
136
99
  ],
137
100
  "tool_call": true,
101
+ "interleaved": {
102
+ "field": "reasoning_content"
103
+ },
104
+ "structured_output": true,
138
105
  "temperature": true,
139
- "knowledge": "2025-04",
140
- "release_date": "2026-01-19",
141
- "last_updated": "2026-01-19",
106
+ "release_date": "2026-03-16",
107
+ "last_updated": "2026-03-16",
142
108
  "modalities": {
143
109
  "input": [
144
110
  "text"
@@ -147,23 +113,23 @@
147
113
  "text"
148
114
  ]
149
115
  },
150
- "open_weights": true,
116
+ "open_weights": false,
151
117
  "limit": {
152
118
  "context": 200000,
153
119
  "output": 131072
154
120
  },
155
121
  "cost": {
156
- "input": 0.07,
157
- "output": 0.4,
158
- "cache_read": 0.01,
122
+ "input": 1.2,
123
+ "output": 4,
124
+ "cache_read": 0.24,
159
125
  "cache_write": 0
160
126
  }
161
127
  },
162
- "glm-5.1": {
163
- "id": "glm-5.1",
164
- "name": "GLM-5.1",
165
- "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
166
- "family": "glm",
128
+ "glm-4.7-flashx": {
129
+ "id": "glm-4.7-flashx",
130
+ "name": "GLM-4.7-FlashX",
131
+ "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
132
+ "family": "glm-flash",
167
133
  "attachment": false,
168
134
  "reasoning": true,
169
135
  "reasoning_options": [
@@ -172,13 +138,10 @@
172
138
  }
173
139
  ],
174
140
  "tool_call": true,
175
- "interleaved": {
176
- "field": "reasoning_content"
177
- },
178
- "structured_output": true,
179
141
  "temperature": true,
180
- "release_date": "2026-04-07",
181
- "last_updated": "2026-04-07",
142
+ "knowledge": "2025-04",
143
+ "release_date": "2026-01-19",
144
+ "last_updated": "2026-01-19",
182
145
  "modalities": {
183
146
  "input": [
184
147
  "text"
@@ -193,17 +156,17 @@
193
156
  "output": 131072
194
157
  },
195
158
  "cost": {
196
- "input": 1.4,
197
- "output": 4.4,
198
- "cache_read": 0.26,
159
+ "input": 0.07,
160
+ "output": 0.4,
161
+ "cache_read": 0.01,
199
162
  "cache_write": 0
200
163
  }
201
164
  },
202
- "glm-4.6": {
203
- "id": "glm-4.6",
204
- "name": "GLM-4.6",
205
- "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
206
- "family": "glm",
165
+ "glm-4.5-air": {
166
+ "id": "glm-4.5-air",
167
+ "name": "GLM-4.5-Air",
168
+ "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
169
+ "family": "glm-air",
207
170
  "attachment": false,
208
171
  "reasoning": true,
209
172
  "reasoning_options": [
@@ -214,8 +177,8 @@
214
177
  "tool_call": true,
215
178
  "temperature": true,
216
179
  "knowledge": "2025-04",
217
- "release_date": "2025-09-30",
218
- "last_updated": "2025-09-30",
180
+ "release_date": "2025-07-28",
181
+ "last_updated": "2025-07-28",
219
182
  "modalities": {
220
183
  "input": [
221
184
  "text"
@@ -226,40 +189,33 @@
226
189
  },
227
190
  "open_weights": true,
228
191
  "limit": {
229
- "context": 204800,
230
- "output": 131072
192
+ "context": 131072,
193
+ "output": 98304
231
194
  },
232
195
  "cost": {
233
- "input": 0.6,
234
- "output": 2.2,
235
- "cache_read": 0.11,
196
+ "input": 0.2,
197
+ "output": 1.1,
198
+ "cache_read": 0.03,
236
199
  "cache_write": 0
237
200
  }
238
201
  },
239
- "glm-5.2": {
240
- "id": "glm-5.2",
241
- "name": "GLM-5.2",
242
- "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
243
- "family": "glm",
202
+ "glm-4.5-flash": {
203
+ "id": "glm-4.5-flash",
204
+ "name": "GLM-4.5-Flash",
205
+ "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
206
+ "family": "glm-flash",
244
207
  "attachment": false,
245
208
  "reasoning": true,
246
209
  "reasoning_options": [
247
210
  {
248
- "type": "effort",
249
- "values": [
250
- "high",
251
- "max"
252
- ]
211
+ "type": "toggle"
253
212
  }
254
213
  ],
255
214
  "tool_call": true,
256
- "interleaved": {
257
- "field": "reasoning_content"
258
- },
259
- "structured_output": true,
260
215
  "temperature": true,
261
- "release_date": "2026-06-13",
262
- "last_updated": "2026-06-13",
216
+ "knowledge": "2025-04",
217
+ "release_date": "2025-07-28",
218
+ "last_updated": "2025-07-28",
263
219
  "modalities": {
264
220
  "input": [
265
221
  "text"
@@ -270,13 +226,13 @@
270
226
  },
271
227
  "open_weights": true,
272
228
  "limit": {
273
- "context": 1000000,
274
- "output": 131072
229
+ "context": 131072,
230
+ "output": 98304
275
231
  },
276
232
  "cost": {
277
- "input": 1.4,
278
- "output": 4.4,
279
- "cache_read": 0.26,
233
+ "input": 0,
234
+ "output": 0,
235
+ "cache_read": 0,
280
236
  "cache_write": 0
281
237
  }
282
238
  },
@@ -317,10 +273,10 @@
317
273
  "output": 0.9
318
274
  }
319
275
  },
320
- "glm-5v-turbo": {
321
- "id": "glm-5v-turbo",
322
- "name": "GLM-5V-Turbo",
323
- "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
276
+ "glm-4.5v": {
277
+ "id": "glm-4.5v",
278
+ "name": "GLM-4.5V",
279
+ "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
324
280
  "family": "glm",
325
281
  "attachment": true,
326
282
  "reasoning": true,
@@ -330,40 +286,35 @@
330
286
  }
331
287
  ],
332
288
  "tool_call": true,
333
- "interleaved": {
334
- "field": "reasoning_content"
335
- },
336
289
  "temperature": true,
337
- "release_date": "2026-04-01",
338
- "last_updated": "2026-04-01",
290
+ "knowledge": "2025-04",
291
+ "release_date": "2025-08-11",
292
+ "last_updated": "2025-08-11",
339
293
  "modalities": {
340
294
  "input": [
341
295
  "text",
342
296
  "image",
343
- "video",
344
- "pdf"
297
+ "video"
345
298
  ],
346
299
  "output": [
347
300
  "text"
348
301
  ]
349
302
  },
350
- "open_weights": false,
303
+ "open_weights": true,
351
304
  "limit": {
352
- "context": 200000,
353
- "output": 131072
305
+ "context": 64000,
306
+ "output": 16384
354
307
  },
355
308
  "cost": {
356
- "input": 1.2,
357
- "output": 4,
358
- "cache_read": 0.24,
359
- "cache_write": 0
309
+ "input": 0.6,
310
+ "output": 1.8
360
311
  }
361
312
  },
362
- "glm-4.5-air": {
363
- "id": "glm-4.5-air",
364
- "name": "GLM-4.5-Air",
365
- "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
366
- "family": "glm-air",
313
+ "glm-4.7-flash": {
314
+ "id": "glm-4.7-flash",
315
+ "name": "GLM-4.7-Flash",
316
+ "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
317
+ "family": "glm-flash",
367
318
  "attachment": false,
368
319
  "reasoning": true,
369
320
  "reasoning_options": [
@@ -374,8 +325,8 @@
374
325
  "tool_call": true,
375
326
  "temperature": true,
376
327
  "knowledge": "2025-04",
377
- "release_date": "2025-07-28",
378
- "last_updated": "2025-07-28",
328
+ "release_date": "2026-01-19",
329
+ "last_updated": "2026-01-19",
379
330
  "modalities": {
380
331
  "input": [
381
332
  "text"
@@ -386,33 +337,40 @@
386
337
  },
387
338
  "open_weights": true,
388
339
  "limit": {
389
- "context": 131072,
390
- "output": 98304
340
+ "context": 200000,
341
+ "output": 131072
391
342
  },
392
343
  "cost": {
393
- "input": 0.2,
394
- "output": 1.1,
395
- "cache_read": 0.03,
344
+ "input": 0,
345
+ "output": 0,
346
+ "cache_read": 0,
396
347
  "cache_write": 0
397
348
  }
398
349
  },
399
- "glm-4.7-flash": {
400
- "id": "glm-4.7-flash",
401
- "name": "GLM-4.7-Flash",
402
- "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
403
- "family": "glm-flash",
350
+ "glm-5.2": {
351
+ "id": "glm-5.2",
352
+ "name": "GLM-5.2",
353
+ "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
354
+ "family": "glm",
404
355
  "attachment": false,
405
356
  "reasoning": true,
406
357
  "reasoning_options": [
407
358
  {
408
- "type": "toggle"
359
+ "type": "effort",
360
+ "values": [
361
+ "high",
362
+ "max"
363
+ ]
409
364
  }
410
365
  ],
411
366
  "tool_call": true,
367
+ "interleaved": {
368
+ "field": "reasoning_content"
369
+ },
370
+ "structured_output": true,
412
371
  "temperature": true,
413
- "knowledge": "2025-04",
414
- "release_date": "2026-01-19",
415
- "last_updated": "2026-01-19",
372
+ "release_date": "2026-06-13",
373
+ "last_updated": "2026-06-13",
416
374
  "modalities": {
417
375
  "input": [
418
376
  "text"
@@ -423,22 +381,22 @@
423
381
  },
424
382
  "open_weights": true,
425
383
  "limit": {
426
- "context": 200000,
384
+ "context": 1000000,
427
385
  "output": 131072
428
386
  },
429
387
  "cost": {
430
- "input": 0,
431
- "output": 0,
432
- "cache_read": 0,
388
+ "input": 1.4,
389
+ "output": 4.4,
390
+ "cache_read": 0.26,
433
391
  "cache_write": 0
434
392
  }
435
393
  },
436
- "glm-4.5-flash": {
437
- "id": "glm-4.5-flash",
438
- "name": "GLM-4.5-Flash",
439
- "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
440
- "family": "glm-flash",
441
- "attachment": false,
394
+ "glm-5v-turbo": {
395
+ "id": "glm-5v-turbo",
396
+ "name": "GLM-5V-Turbo",
397
+ "description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
398
+ "family": "glm",
399
+ "attachment": true,
442
400
  "reasoning": true,
443
401
  "reasoning_options": [
444
402
  {
@@ -446,27 +404,32 @@
446
404
  }
447
405
  ],
448
406
  "tool_call": true,
407
+ "interleaved": {
408
+ "field": "reasoning_content"
409
+ },
449
410
  "temperature": true,
450
- "knowledge": "2025-04",
451
- "release_date": "2025-07-28",
452
- "last_updated": "2025-07-28",
411
+ "release_date": "2026-04-01",
412
+ "last_updated": "2026-04-01",
453
413
  "modalities": {
454
414
  "input": [
455
- "text"
415
+ "text",
416
+ "image",
417
+ "video",
418
+ "pdf"
456
419
  ],
457
420
  "output": [
458
421
  "text"
459
422
  ]
460
423
  },
461
- "open_weights": true,
424
+ "open_weights": false,
462
425
  "limit": {
463
- "context": 131072,
464
- "output": 98304
426
+ "context": 200000,
427
+ "output": 131072
465
428
  },
466
429
  "cost": {
467
- "input": 0,
468
- "output": 0,
469
- "cache_read": 0,
430
+ "input": 1.2,
431
+ "output": 4,
432
+ "cache_read": 0.24,
470
433
  "cache_write": 0
471
434
  }
472
435
  },
@@ -509,10 +472,10 @@
509
472
  "cache_write": 0
510
473
  }
511
474
  },
512
- "glm-5-turbo": {
513
- "id": "glm-5-turbo",
514
- "name": "GLM-5-Turbo",
515
- "description": "Faster GLM-5 lane for coding agents that need lower latency",
475
+ "glm-5.1": {
476
+ "id": "glm-5.1",
477
+ "name": "GLM-5.1",
478
+ "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
516
479
  "family": "glm",
517
480
  "attachment": false,
518
481
  "reasoning": true,
@@ -527,8 +490,8 @@
527
490
  },
528
491
  "structured_output": true,
529
492
  "temperature": true,
530
- "release_date": "2026-03-16",
531
- "last_updated": "2026-03-16",
493
+ "release_date": "2026-04-07",
494
+ "last_updated": "2026-04-07",
532
495
  "modalities": {
533
496
  "input": [
534
497
  "text"
@@ -537,15 +500,52 @@
537
500
  "text"
538
501
  ]
539
502
  },
540
- "open_weights": false,
503
+ "open_weights": true,
541
504
  "limit": {
542
505
  "context": 200000,
543
506
  "output": 131072
544
507
  },
545
508
  "cost": {
546
- "input": 1.2,
547
- "output": 4,
548
- "cache_read": 0.24,
509
+ "input": 1.4,
510
+ "output": 4.4,
511
+ "cache_read": 0.26,
512
+ "cache_write": 0
513
+ }
514
+ },
515
+ "glm-4.6": {
516
+ "id": "glm-4.6",
517
+ "name": "GLM-4.6",
518
+ "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
519
+ "family": "glm",
520
+ "attachment": false,
521
+ "reasoning": true,
522
+ "reasoning_options": [
523
+ {
524
+ "type": "toggle"
525
+ }
526
+ ],
527
+ "tool_call": true,
528
+ "temperature": true,
529
+ "knowledge": "2025-04",
530
+ "release_date": "2025-09-30",
531
+ "last_updated": "2025-09-30",
532
+ "modalities": {
533
+ "input": [
534
+ "text"
535
+ ],
536
+ "output": [
537
+ "text"
538
+ ]
539
+ },
540
+ "open_weights": true,
541
+ "limit": {
542
+ "context": 204800,
543
+ "output": 131072
544
+ },
545
+ "cost": {
546
+ "input": 0.6,
547
+ "output": 2.2,
548
+ "cache_read": 0.11,
549
549
  "cache_write": 0
550
550
  }
551
551
  }
data/lib/llm/agent.rb CHANGED
@@ -266,6 +266,7 @@ module LLM
266
266
  def functions
267
267
  @tracer ? @llm.with_tracer(@tracer) { @ctx.functions } : @ctx.functions
268
268
  end
269
+ alias_method :pending_functions, :functions
269
270
 
270
271
  ##
271
272
  # @see LLM::Context#returns
@@ -398,15 +399,18 @@ module LLM
398
399
  # By default this method disables the tracer for
399
400
  # the duration of the repl session, and restores
400
401
  # it afterwards.
401
- # @param [Boolean] tracer
402
- # When true, the tracer is kept alive during the
403
- # repl session. Default is false.
402
+ # @param [String] path
403
+ # The path to a file where runtime state is read
404
+ # from, and written to
404
405
  # @param [Array<LLM::Tool>] tools
405
406
  # Extra tools to attach for the repl session
406
407
  # @param [Array<String>] skills
407
408
  # Extra skills to attach for the repl session
409
+ # @param [Boolean] tracer
410
+ # When true, the tracer is kept alive during the
411
+ # repl session. Default is false.
408
412
  # @return [void]
409
- def repl(tracer: false, trace: nil, tools: [], skills: [])
413
+ def repl(path: nil, tools: [], skills: [], tracer: false, trace: nil)
410
414
  if trace != nil
411
415
  warn "llm.rb: trace option is deprecated, use tracer instead"
412
416
  tracer = trace
@@ -416,9 +420,9 @@ module LLM
416
420
  self.tracer = nil
417
421
  end
418
422
  require_relative "repl" unless defined?(::LLM::Repl)
419
- LLM::Repl.new(agent: self, tools:, skills:).start
423
+ LLM::Repl.new(agent: self, path:, tools:, skills:).start
420
424
  ensure
421
- if !trace
425
+ if !tracer
422
426
  self.tracer = previous
423
427
  end
424
428
  end
@@ -460,9 +464,10 @@ module LLM
460
464
 
461
465
  ##
462
466
  # @param (see LLM::Context#deserialize)
463
- # @return (see LLM::Context#deserialize)
467
+ # @return [LLM::Agent]
464
468
  def deserialize(**kw)
465
469
  @ctx.deserialize(**kw)
470
+ self
466
471
  end
467
472
  alias_method :restore, :deserialize
468
473
 
data/lib/llm/buffer.rb CHANGED
@@ -69,6 +69,21 @@ module LLM
69
69
  n.nil? ? @messages.last : @messages.last(n)
70
70
  end
71
71
 
72
+ ##
73
+ # Slice a portion of the internal buffer in-place
74
+ # @return [void]
75
+ def slice!(...)
76
+ @messages.slice!(...)
77
+ nil
78
+ end
79
+
80
+ ##
81
+ # Pop the last element from the tail of the buffer
82
+ # @return [void]
83
+ def pop
84
+ @messages.pop
85
+ end
86
+
72
87
  ##
73
88
  # @param [[LLM::Message]] item
74
89
  # A message to add to the buffer
@@ -31,9 +31,8 @@ class LLM::Context
31
31
  end
32
32
  alias_method :restore, :deserialize
33
33
 
34
- ##
35
- # @param [Hash] payload
36
- # @return [LLM::Message]
34
+ private
35
+
37
36
  def deserialize_message(payload)
38
37
  tool_calls = deserialize_tool_calls(payload["tools"])
39
38
  returns = deserialize_returns(payload["content"]) if returns.nil?
@@ -46,8 +45,6 @@ class LLM::Context
46
45
  LLM::Message.new(payload["role"], content, extra)
47
46
  end
48
47
 
49
- private
50
-
51
48
  def deserialize_content(content)
52
49
  case content
53
50
  when Array
data/lib/llm/context.rb CHANGED
@@ -257,6 +257,7 @@ module LLM
257
257
  end
258
258
  end.extend(LLM::Function::Array)
259
259
  end
260
+ alias_method :pending_functions, :functions
260
261
 
261
262
  ##
262
263
  # Returns whether there is pending tool work in this context.
@@ -334,13 +335,11 @@ module LLM
334
335
  # This is inspired by Go's context cancellation model.
335
336
  # @return [nil]
336
337
  def interrupt!
337
- pending = functions.to_a
338
338
  llm.interrupt!(@owner)
339
339
  queue&.interrupt!
340
- return if pending.empty?
341
- pending.each(&:interrupt!)
342
- returns = pending.map { _1.cancel(reason: "function call cancelled") }
343
- @messages << LLM::Message.new(@llm.tool_role, returns)
340
+ functions.each(&:interrupt!)
341
+ @queue = nil
342
+ @owner = nil
344
343
  nil
345
344
  end
346
345
  alias_method :cancel!, :interrupt!