llm.rb 12.3.1 → 12.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +395 -0
- data/README.md +103 -16
- data/data/anthropic.json +249 -249
- data/data/bedrock.json +2038 -1879
- data/data/deepinfra.json +591 -591
- data/data/deepseek.json +31 -31
- data/data/google.json +332 -329
- data/data/mistral.json +381 -381
- data/data/openai.json +1132 -1132
- data/data/xai.json +138 -138
- data/data/zai.json +165 -165
- data/lib/llm/agent.rb +12 -7
- data/lib/llm/buffer.rb +15 -0
- data/lib/llm/context/deserializer.rb +2 -5
- data/lib/llm/context.rb +4 -5
- data/lib/llm/function.rb +2 -35
- data/lib/llm/message.rb +12 -2
- data/lib/llm/provider.rb +11 -1
- data/lib/llm/providers/anthropic/stream_parser.rb +3 -12
- data/lib/llm/providers/anthropic.rb +11 -0
- data/lib/llm/providers/bedrock/stream_parser.rb +4 -13
- data/lib/llm/providers/google/stream_parser.rb +3 -12
- data/lib/llm/providers/google.rb +7 -0
- data/lib/llm/providers/mistral.rb +11 -0
- data/lib/llm/providers/ollama/stream_parser.rb +4 -4
- data/lib/llm/providers/ollama.rb +11 -0
- data/lib/llm/providers/openai/responses/stream_parser.rb +4 -14
- data/lib/llm/providers/openai/responses.rb +10 -0
- data/lib/llm/providers/openai/stream_parser.rb +5 -15
- data/lib/llm/providers/openai.rb +11 -0
- data/lib/llm/repl/command.rb +201 -0
- data/lib/llm/repl/commands/exit.rb +23 -0
- data/lib/llm/repl/commands/help.rb +24 -0
- data/lib/llm/repl/input.rb +118 -35
- data/lib/llm/repl/stream.rb +43 -8
- data/lib/llm/repl/transcript.rb +6 -0
- data/lib/llm/repl/window.rb +28 -6
- data/lib/llm/repl.rb +125 -27
- data/lib/llm/stream.rb +2 -7
- data/lib/llm/tools/ls.rb +30 -0
- data/lib/llm/tools/which.rb +38 -0
- data/lib/llm/version.rb +1 -1
- data/resources/deepdive.md +67 -1
- metadata +6 -1
data/data/zai.json
CHANGED
|
@@ -48,43 +48,6 @@
|
|
|
48
48
|
"cache_write": 0
|
|
49
49
|
}
|
|
50
50
|
},
|
|
51
|
-
"glm-4.5v": {
|
|
52
|
-
"id": "glm-4.5v",
|
|
53
|
-
"name": "GLM-4.5V",
|
|
54
|
-
"description": "GLM vision model for visual reasoning, documents, and multimodal agents",
|
|
55
|
-
"family": "glm",
|
|
56
|
-
"attachment": true,
|
|
57
|
-
"reasoning": true,
|
|
58
|
-
"reasoning_options": [
|
|
59
|
-
{
|
|
60
|
-
"type": "toggle"
|
|
61
|
-
}
|
|
62
|
-
],
|
|
63
|
-
"tool_call": true,
|
|
64
|
-
"temperature": true,
|
|
65
|
-
"knowledge": "2025-04",
|
|
66
|
-
"release_date": "2025-08-11",
|
|
67
|
-
"last_updated": "2025-08-11",
|
|
68
|
-
"modalities": {
|
|
69
|
-
"input": [
|
|
70
|
-
"text",
|
|
71
|
-
"image",
|
|
72
|
-
"video"
|
|
73
|
-
],
|
|
74
|
-
"output": [
|
|
75
|
-
"text"
|
|
76
|
-
]
|
|
77
|
-
},
|
|
78
|
-
"open_weights": true,
|
|
79
|
-
"limit": {
|
|
80
|
-
"context": 64000,
|
|
81
|
-
"output": 16384
|
|
82
|
-
},
|
|
83
|
-
"cost": {
|
|
84
|
-
"input": 0.6,
|
|
85
|
-
"output": 1.8
|
|
86
|
-
}
|
|
87
|
-
},
|
|
88
51
|
"glm-4.5": {
|
|
89
52
|
"id": "glm-4.5",
|
|
90
53
|
"name": "GLM-4.5",
|
|
@@ -122,11 +85,11 @@
|
|
|
122
85
|
"cache_write": 0
|
|
123
86
|
}
|
|
124
87
|
},
|
|
125
|
-
"glm-
|
|
126
|
-
"id": "glm-
|
|
127
|
-
"name": "GLM-
|
|
128
|
-
"description": "
|
|
129
|
-
"family": "glm
|
|
88
|
+
"glm-5-turbo": {
|
|
89
|
+
"id": "glm-5-turbo",
|
|
90
|
+
"name": "GLM-5-Turbo",
|
|
91
|
+
"description": "Faster GLM-5 lane for coding agents that need lower latency",
|
|
92
|
+
"family": "glm",
|
|
130
93
|
"attachment": false,
|
|
131
94
|
"reasoning": true,
|
|
132
95
|
"reasoning_options": [
|
|
@@ -135,10 +98,13 @@
|
|
|
135
98
|
}
|
|
136
99
|
],
|
|
137
100
|
"tool_call": true,
|
|
101
|
+
"interleaved": {
|
|
102
|
+
"field": "reasoning_content"
|
|
103
|
+
},
|
|
104
|
+
"structured_output": true,
|
|
138
105
|
"temperature": true,
|
|
139
|
-
"
|
|
140
|
-
"
|
|
141
|
-
"last_updated": "2026-01-19",
|
|
106
|
+
"release_date": "2026-03-16",
|
|
107
|
+
"last_updated": "2026-03-16",
|
|
142
108
|
"modalities": {
|
|
143
109
|
"input": [
|
|
144
110
|
"text"
|
|
@@ -147,23 +113,23 @@
|
|
|
147
113
|
"text"
|
|
148
114
|
]
|
|
149
115
|
},
|
|
150
|
-
"open_weights":
|
|
116
|
+
"open_weights": false,
|
|
151
117
|
"limit": {
|
|
152
118
|
"context": 200000,
|
|
153
119
|
"output": 131072
|
|
154
120
|
},
|
|
155
121
|
"cost": {
|
|
156
|
-
"input":
|
|
157
|
-
"output":
|
|
158
|
-
"cache_read": 0.
|
|
122
|
+
"input": 1.2,
|
|
123
|
+
"output": 4,
|
|
124
|
+
"cache_read": 0.24,
|
|
159
125
|
"cache_write": 0
|
|
160
126
|
}
|
|
161
127
|
},
|
|
162
|
-
"glm-
|
|
163
|
-
"id": "glm-
|
|
164
|
-
"name": "GLM-
|
|
165
|
-
"description": "
|
|
166
|
-
"family": "glm",
|
|
128
|
+
"glm-4.7-flashx": {
|
|
129
|
+
"id": "glm-4.7-flashx",
|
|
130
|
+
"name": "GLM-4.7-FlashX",
|
|
131
|
+
"description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
|
|
132
|
+
"family": "glm-flash",
|
|
167
133
|
"attachment": false,
|
|
168
134
|
"reasoning": true,
|
|
169
135
|
"reasoning_options": [
|
|
@@ -172,13 +138,10 @@
|
|
|
172
138
|
}
|
|
173
139
|
],
|
|
174
140
|
"tool_call": true,
|
|
175
|
-
"interleaved": {
|
|
176
|
-
"field": "reasoning_content"
|
|
177
|
-
},
|
|
178
|
-
"structured_output": true,
|
|
179
141
|
"temperature": true,
|
|
180
|
-
"
|
|
181
|
-
"
|
|
142
|
+
"knowledge": "2025-04",
|
|
143
|
+
"release_date": "2026-01-19",
|
|
144
|
+
"last_updated": "2026-01-19",
|
|
182
145
|
"modalities": {
|
|
183
146
|
"input": [
|
|
184
147
|
"text"
|
|
@@ -193,17 +156,17 @@
|
|
|
193
156
|
"output": 131072
|
|
194
157
|
},
|
|
195
158
|
"cost": {
|
|
196
|
-
"input":
|
|
197
|
-
"output":
|
|
198
|
-
"cache_read": 0.
|
|
159
|
+
"input": 0.07,
|
|
160
|
+
"output": 0.4,
|
|
161
|
+
"cache_read": 0.01,
|
|
199
162
|
"cache_write": 0
|
|
200
163
|
}
|
|
201
164
|
},
|
|
202
|
-
"glm-4.
|
|
203
|
-
"id": "glm-4.
|
|
204
|
-
"name": "GLM-4.
|
|
205
|
-
"description": "
|
|
206
|
-
"family": "glm",
|
|
165
|
+
"glm-4.5-air": {
|
|
166
|
+
"id": "glm-4.5-air",
|
|
167
|
+
"name": "GLM-4.5-Air",
|
|
168
|
+
"description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
|
|
169
|
+
"family": "glm-air",
|
|
207
170
|
"attachment": false,
|
|
208
171
|
"reasoning": true,
|
|
209
172
|
"reasoning_options": [
|
|
@@ -214,8 +177,8 @@
|
|
|
214
177
|
"tool_call": true,
|
|
215
178
|
"temperature": true,
|
|
216
179
|
"knowledge": "2025-04",
|
|
217
|
-
"release_date": "2025-
|
|
218
|
-
"last_updated": "2025-
|
|
180
|
+
"release_date": "2025-07-28",
|
|
181
|
+
"last_updated": "2025-07-28",
|
|
219
182
|
"modalities": {
|
|
220
183
|
"input": [
|
|
221
184
|
"text"
|
|
@@ -226,40 +189,33 @@
|
|
|
226
189
|
},
|
|
227
190
|
"open_weights": true,
|
|
228
191
|
"limit": {
|
|
229
|
-
"context":
|
|
230
|
-
"output":
|
|
192
|
+
"context": 131072,
|
|
193
|
+
"output": 98304
|
|
231
194
|
},
|
|
232
195
|
"cost": {
|
|
233
|
-
"input": 0.
|
|
234
|
-
"output":
|
|
235
|
-
"cache_read": 0.
|
|
196
|
+
"input": 0.2,
|
|
197
|
+
"output": 1.1,
|
|
198
|
+
"cache_read": 0.03,
|
|
236
199
|
"cache_write": 0
|
|
237
200
|
}
|
|
238
201
|
},
|
|
239
|
-
"glm-5
|
|
240
|
-
"id": "glm-5
|
|
241
|
-
"name": "GLM-5
|
|
242
|
-
"description": "
|
|
243
|
-
"family": "glm",
|
|
202
|
+
"glm-4.5-flash": {
|
|
203
|
+
"id": "glm-4.5-flash",
|
|
204
|
+
"name": "GLM-4.5-Flash",
|
|
205
|
+
"description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
|
|
206
|
+
"family": "glm-flash",
|
|
244
207
|
"attachment": false,
|
|
245
208
|
"reasoning": true,
|
|
246
209
|
"reasoning_options": [
|
|
247
210
|
{
|
|
248
|
-
"type": "
|
|
249
|
-
"values": [
|
|
250
|
-
"high",
|
|
251
|
-
"max"
|
|
252
|
-
]
|
|
211
|
+
"type": "toggle"
|
|
253
212
|
}
|
|
254
213
|
],
|
|
255
214
|
"tool_call": true,
|
|
256
|
-
"interleaved": {
|
|
257
|
-
"field": "reasoning_content"
|
|
258
|
-
},
|
|
259
|
-
"structured_output": true,
|
|
260
215
|
"temperature": true,
|
|
261
|
-
"
|
|
262
|
-
"
|
|
216
|
+
"knowledge": "2025-04",
|
|
217
|
+
"release_date": "2025-07-28",
|
|
218
|
+
"last_updated": "2025-07-28",
|
|
263
219
|
"modalities": {
|
|
264
220
|
"input": [
|
|
265
221
|
"text"
|
|
@@ -270,13 +226,13 @@
|
|
|
270
226
|
},
|
|
271
227
|
"open_weights": true,
|
|
272
228
|
"limit": {
|
|
273
|
-
"context":
|
|
274
|
-
"output":
|
|
229
|
+
"context": 131072,
|
|
230
|
+
"output": 98304
|
|
275
231
|
},
|
|
276
232
|
"cost": {
|
|
277
|
-
"input":
|
|
278
|
-
"output":
|
|
279
|
-
"cache_read": 0
|
|
233
|
+
"input": 0,
|
|
234
|
+
"output": 0,
|
|
235
|
+
"cache_read": 0,
|
|
280
236
|
"cache_write": 0
|
|
281
237
|
}
|
|
282
238
|
},
|
|
@@ -317,10 +273,10 @@
|
|
|
317
273
|
"output": 0.9
|
|
318
274
|
}
|
|
319
275
|
},
|
|
320
|
-
"glm-5v
|
|
321
|
-
"id": "glm-5v
|
|
322
|
-
"name": "GLM-5V
|
|
323
|
-
"description": "
|
|
276
|
+
"glm-4.5v": {
|
|
277
|
+
"id": "glm-4.5v",
|
|
278
|
+
"name": "GLM-4.5V",
|
|
279
|
+
"description": "GLM vision model for visual reasoning, documents, and multimodal agents",
|
|
324
280
|
"family": "glm",
|
|
325
281
|
"attachment": true,
|
|
326
282
|
"reasoning": true,
|
|
@@ -330,40 +286,35 @@
|
|
|
330
286
|
}
|
|
331
287
|
],
|
|
332
288
|
"tool_call": true,
|
|
333
|
-
"interleaved": {
|
|
334
|
-
"field": "reasoning_content"
|
|
335
|
-
},
|
|
336
289
|
"temperature": true,
|
|
337
|
-
"
|
|
338
|
-
"
|
|
290
|
+
"knowledge": "2025-04",
|
|
291
|
+
"release_date": "2025-08-11",
|
|
292
|
+
"last_updated": "2025-08-11",
|
|
339
293
|
"modalities": {
|
|
340
294
|
"input": [
|
|
341
295
|
"text",
|
|
342
296
|
"image",
|
|
343
|
-
"video"
|
|
344
|
-
"pdf"
|
|
297
|
+
"video"
|
|
345
298
|
],
|
|
346
299
|
"output": [
|
|
347
300
|
"text"
|
|
348
301
|
]
|
|
349
302
|
},
|
|
350
|
-
"open_weights":
|
|
303
|
+
"open_weights": true,
|
|
351
304
|
"limit": {
|
|
352
|
-
"context":
|
|
353
|
-
"output":
|
|
305
|
+
"context": 64000,
|
|
306
|
+
"output": 16384
|
|
354
307
|
},
|
|
355
308
|
"cost": {
|
|
356
|
-
"input":
|
|
357
|
-
"output":
|
|
358
|
-
"cache_read": 0.24,
|
|
359
|
-
"cache_write": 0
|
|
309
|
+
"input": 0.6,
|
|
310
|
+
"output": 1.8
|
|
360
311
|
}
|
|
361
312
|
},
|
|
362
|
-
"glm-4.
|
|
363
|
-
"id": "glm-4.
|
|
364
|
-
"name": "GLM-4.
|
|
365
|
-
"description": "
|
|
366
|
-
"family": "glm-
|
|
313
|
+
"glm-4.7-flash": {
|
|
314
|
+
"id": "glm-4.7-flash",
|
|
315
|
+
"name": "GLM-4.7-Flash",
|
|
316
|
+
"description": "Budget GLM lane for fast coding help, routing, and everyday automation",
|
|
317
|
+
"family": "glm-flash",
|
|
367
318
|
"attachment": false,
|
|
368
319
|
"reasoning": true,
|
|
369
320
|
"reasoning_options": [
|
|
@@ -374,8 +325,8 @@
|
|
|
374
325
|
"tool_call": true,
|
|
375
326
|
"temperature": true,
|
|
376
327
|
"knowledge": "2025-04",
|
|
377
|
-
"release_date": "
|
|
378
|
-
"last_updated": "
|
|
328
|
+
"release_date": "2026-01-19",
|
|
329
|
+
"last_updated": "2026-01-19",
|
|
379
330
|
"modalities": {
|
|
380
331
|
"input": [
|
|
381
332
|
"text"
|
|
@@ -386,33 +337,40 @@
|
|
|
386
337
|
},
|
|
387
338
|
"open_weights": true,
|
|
388
339
|
"limit": {
|
|
389
|
-
"context":
|
|
390
|
-
"output":
|
|
340
|
+
"context": 200000,
|
|
341
|
+
"output": 131072
|
|
391
342
|
},
|
|
392
343
|
"cost": {
|
|
393
|
-
"input": 0
|
|
394
|
-
"output":
|
|
395
|
-
"cache_read": 0
|
|
344
|
+
"input": 0,
|
|
345
|
+
"output": 0,
|
|
346
|
+
"cache_read": 0,
|
|
396
347
|
"cache_write": 0
|
|
397
348
|
}
|
|
398
349
|
},
|
|
399
|
-
"glm-
|
|
400
|
-
"id": "glm-
|
|
401
|
-
"name": "GLM-
|
|
402
|
-
"description": "
|
|
403
|
-
"family": "glm
|
|
350
|
+
"glm-5.2": {
|
|
351
|
+
"id": "glm-5.2",
|
|
352
|
+
"name": "GLM-5.2",
|
|
353
|
+
"description": "Open flagship GLM for long-horizon coding agents and million-token context work",
|
|
354
|
+
"family": "glm",
|
|
404
355
|
"attachment": false,
|
|
405
356
|
"reasoning": true,
|
|
406
357
|
"reasoning_options": [
|
|
407
358
|
{
|
|
408
|
-
"type": "
|
|
359
|
+
"type": "effort",
|
|
360
|
+
"values": [
|
|
361
|
+
"high",
|
|
362
|
+
"max"
|
|
363
|
+
]
|
|
409
364
|
}
|
|
410
365
|
],
|
|
411
366
|
"tool_call": true,
|
|
367
|
+
"interleaved": {
|
|
368
|
+
"field": "reasoning_content"
|
|
369
|
+
},
|
|
370
|
+
"structured_output": true,
|
|
412
371
|
"temperature": true,
|
|
413
|
-
"
|
|
414
|
-
"
|
|
415
|
-
"last_updated": "2026-01-19",
|
|
372
|
+
"release_date": "2026-06-13",
|
|
373
|
+
"last_updated": "2026-06-13",
|
|
416
374
|
"modalities": {
|
|
417
375
|
"input": [
|
|
418
376
|
"text"
|
|
@@ -423,22 +381,22 @@
|
|
|
423
381
|
},
|
|
424
382
|
"open_weights": true,
|
|
425
383
|
"limit": {
|
|
426
|
-
"context":
|
|
384
|
+
"context": 1000000,
|
|
427
385
|
"output": 131072
|
|
428
386
|
},
|
|
429
387
|
"cost": {
|
|
430
|
-
"input":
|
|
431
|
-
"output":
|
|
432
|
-
"cache_read": 0,
|
|
388
|
+
"input": 1.4,
|
|
389
|
+
"output": 4.4,
|
|
390
|
+
"cache_read": 0.26,
|
|
433
391
|
"cache_write": 0
|
|
434
392
|
}
|
|
435
393
|
},
|
|
436
|
-
"glm-
|
|
437
|
-
"id": "glm-
|
|
438
|
-
"name": "GLM-
|
|
439
|
-
"description": "
|
|
440
|
-
"family": "glm
|
|
441
|
-
"attachment":
|
|
394
|
+
"glm-5v-turbo": {
|
|
395
|
+
"id": "glm-5v-turbo",
|
|
396
|
+
"name": "GLM-5V-Turbo",
|
|
397
|
+
"description": "Fast GLM vision model for screenshots, documents, and multimodal agent tasks",
|
|
398
|
+
"family": "glm",
|
|
399
|
+
"attachment": true,
|
|
442
400
|
"reasoning": true,
|
|
443
401
|
"reasoning_options": [
|
|
444
402
|
{
|
|
@@ -446,27 +404,32 @@
|
|
|
446
404
|
}
|
|
447
405
|
],
|
|
448
406
|
"tool_call": true,
|
|
407
|
+
"interleaved": {
|
|
408
|
+
"field": "reasoning_content"
|
|
409
|
+
},
|
|
449
410
|
"temperature": true,
|
|
450
|
-
"
|
|
451
|
-
"
|
|
452
|
-
"last_updated": "2025-07-28",
|
|
411
|
+
"release_date": "2026-04-01",
|
|
412
|
+
"last_updated": "2026-04-01",
|
|
453
413
|
"modalities": {
|
|
454
414
|
"input": [
|
|
455
|
-
"text"
|
|
415
|
+
"text",
|
|
416
|
+
"image",
|
|
417
|
+
"video",
|
|
418
|
+
"pdf"
|
|
456
419
|
],
|
|
457
420
|
"output": [
|
|
458
421
|
"text"
|
|
459
422
|
]
|
|
460
423
|
},
|
|
461
|
-
"open_weights":
|
|
424
|
+
"open_weights": false,
|
|
462
425
|
"limit": {
|
|
463
|
-
"context":
|
|
464
|
-
"output":
|
|
426
|
+
"context": 200000,
|
|
427
|
+
"output": 131072
|
|
465
428
|
},
|
|
466
429
|
"cost": {
|
|
467
|
-
"input":
|
|
468
|
-
"output":
|
|
469
|
-
"cache_read": 0,
|
|
430
|
+
"input": 1.2,
|
|
431
|
+
"output": 4,
|
|
432
|
+
"cache_read": 0.24,
|
|
470
433
|
"cache_write": 0
|
|
471
434
|
}
|
|
472
435
|
},
|
|
@@ -509,10 +472,10 @@
|
|
|
509
472
|
"cache_write": 0
|
|
510
473
|
}
|
|
511
474
|
},
|
|
512
|
-
"glm-5
|
|
513
|
-
"id": "glm-5
|
|
514
|
-
"name": "GLM-5
|
|
515
|
-
"description": "
|
|
475
|
+
"glm-5.1": {
|
|
476
|
+
"id": "glm-5.1",
|
|
477
|
+
"name": "GLM-5.1",
|
|
478
|
+
"description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
|
|
516
479
|
"family": "glm",
|
|
517
480
|
"attachment": false,
|
|
518
481
|
"reasoning": true,
|
|
@@ -527,8 +490,8 @@
|
|
|
527
490
|
},
|
|
528
491
|
"structured_output": true,
|
|
529
492
|
"temperature": true,
|
|
530
|
-
"release_date": "2026-
|
|
531
|
-
"last_updated": "2026-
|
|
493
|
+
"release_date": "2026-04-07",
|
|
494
|
+
"last_updated": "2026-04-07",
|
|
532
495
|
"modalities": {
|
|
533
496
|
"input": [
|
|
534
497
|
"text"
|
|
@@ -537,15 +500,52 @@
|
|
|
537
500
|
"text"
|
|
538
501
|
]
|
|
539
502
|
},
|
|
540
|
-
"open_weights":
|
|
503
|
+
"open_weights": true,
|
|
541
504
|
"limit": {
|
|
542
505
|
"context": 200000,
|
|
543
506
|
"output": 131072
|
|
544
507
|
},
|
|
545
508
|
"cost": {
|
|
546
|
-
"input": 1.
|
|
547
|
-
"output": 4,
|
|
548
|
-
"cache_read": 0.
|
|
509
|
+
"input": 1.4,
|
|
510
|
+
"output": 4.4,
|
|
511
|
+
"cache_read": 0.26,
|
|
512
|
+
"cache_write": 0
|
|
513
|
+
}
|
|
514
|
+
},
|
|
515
|
+
"glm-4.6": {
|
|
516
|
+
"id": "glm-4.6",
|
|
517
|
+
"name": "GLM-4.6",
|
|
518
|
+
"description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
|
|
519
|
+
"family": "glm",
|
|
520
|
+
"attachment": false,
|
|
521
|
+
"reasoning": true,
|
|
522
|
+
"reasoning_options": [
|
|
523
|
+
{
|
|
524
|
+
"type": "toggle"
|
|
525
|
+
}
|
|
526
|
+
],
|
|
527
|
+
"tool_call": true,
|
|
528
|
+
"temperature": true,
|
|
529
|
+
"knowledge": "2025-04",
|
|
530
|
+
"release_date": "2025-09-30",
|
|
531
|
+
"last_updated": "2025-09-30",
|
|
532
|
+
"modalities": {
|
|
533
|
+
"input": [
|
|
534
|
+
"text"
|
|
535
|
+
],
|
|
536
|
+
"output": [
|
|
537
|
+
"text"
|
|
538
|
+
]
|
|
539
|
+
},
|
|
540
|
+
"open_weights": true,
|
|
541
|
+
"limit": {
|
|
542
|
+
"context": 204800,
|
|
543
|
+
"output": 131072
|
|
544
|
+
},
|
|
545
|
+
"cost": {
|
|
546
|
+
"input": 0.6,
|
|
547
|
+
"output": 2.2,
|
|
548
|
+
"cache_read": 0.11,
|
|
549
549
|
"cache_write": 0
|
|
550
550
|
}
|
|
551
551
|
}
|
data/lib/llm/agent.rb
CHANGED
|
@@ -266,6 +266,7 @@ module LLM
|
|
|
266
266
|
def functions
|
|
267
267
|
@tracer ? @llm.with_tracer(@tracer) { @ctx.functions } : @ctx.functions
|
|
268
268
|
end
|
|
269
|
+
alias_method :pending_functions, :functions
|
|
269
270
|
|
|
270
271
|
##
|
|
271
272
|
# @see LLM::Context#returns
|
|
@@ -398,15 +399,18 @@ module LLM
|
|
|
398
399
|
# By default this method disables the tracer for
|
|
399
400
|
# the duration of the repl session, and restores
|
|
400
401
|
# it afterwards.
|
|
401
|
-
# @param [
|
|
402
|
-
#
|
|
403
|
-
#
|
|
402
|
+
# @param [String] path
|
|
403
|
+
# The path to a file where runtime state is read
|
|
404
|
+
# from, and written to
|
|
404
405
|
# @param [Array<LLM::Tool>] tools
|
|
405
406
|
# Extra tools to attach for the repl session
|
|
406
407
|
# @param [Array<String>] skills
|
|
407
408
|
# Extra skills to attach for the repl session
|
|
409
|
+
# @param [Boolean] tracer
|
|
410
|
+
# When true, the tracer is kept alive during the
|
|
411
|
+
# repl session. Default is false.
|
|
408
412
|
# @return [void]
|
|
409
|
-
def repl(
|
|
413
|
+
def repl(path: nil, tools: [], skills: [], tracer: false, trace: nil)
|
|
410
414
|
if trace != nil
|
|
411
415
|
warn "llm.rb: trace option is deprecated, use tracer instead"
|
|
412
416
|
tracer = trace
|
|
@@ -416,9 +420,9 @@ module LLM
|
|
|
416
420
|
self.tracer = nil
|
|
417
421
|
end
|
|
418
422
|
require_relative "repl" unless defined?(::LLM::Repl)
|
|
419
|
-
LLM::Repl.new(agent: self, tools:, skills:).start
|
|
423
|
+
LLM::Repl.new(agent: self, path:, tools:, skills:).start
|
|
420
424
|
ensure
|
|
421
|
-
if !
|
|
425
|
+
if !tracer
|
|
422
426
|
self.tracer = previous
|
|
423
427
|
end
|
|
424
428
|
end
|
|
@@ -460,9 +464,10 @@ module LLM
|
|
|
460
464
|
|
|
461
465
|
##
|
|
462
466
|
# @param (see LLM::Context#deserialize)
|
|
463
|
-
# @return
|
|
467
|
+
# @return [LLM::Agent]
|
|
464
468
|
def deserialize(**kw)
|
|
465
469
|
@ctx.deserialize(**kw)
|
|
470
|
+
self
|
|
466
471
|
end
|
|
467
472
|
alias_method :restore, :deserialize
|
|
468
473
|
|
data/lib/llm/buffer.rb
CHANGED
|
@@ -69,6 +69,21 @@ module LLM
|
|
|
69
69
|
n.nil? ? @messages.last : @messages.last(n)
|
|
70
70
|
end
|
|
71
71
|
|
|
72
|
+
##
|
|
73
|
+
# Slice a portion of the internal buffer in-place
|
|
74
|
+
# @return [void]
|
|
75
|
+
def slice!(...)
|
|
76
|
+
@messages.slice!(...)
|
|
77
|
+
nil
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
##
|
|
81
|
+
# Pop the last element from the tail of the buffer
|
|
82
|
+
# @return [void]
|
|
83
|
+
def pop
|
|
84
|
+
@messages.pop
|
|
85
|
+
end
|
|
86
|
+
|
|
72
87
|
##
|
|
73
88
|
# @param [[LLM::Message]] item
|
|
74
89
|
# A message to add to the buffer
|
|
@@ -31,9 +31,8 @@ class LLM::Context
|
|
|
31
31
|
end
|
|
32
32
|
alias_method :restore, :deserialize
|
|
33
33
|
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
# @return [LLM::Message]
|
|
34
|
+
private
|
|
35
|
+
|
|
37
36
|
def deserialize_message(payload)
|
|
38
37
|
tool_calls = deserialize_tool_calls(payload["tools"])
|
|
39
38
|
returns = deserialize_returns(payload["content"]) if returns.nil?
|
|
@@ -46,8 +45,6 @@ class LLM::Context
|
|
|
46
45
|
LLM::Message.new(payload["role"], content, extra)
|
|
47
46
|
end
|
|
48
47
|
|
|
49
|
-
private
|
|
50
|
-
|
|
51
48
|
def deserialize_content(content)
|
|
52
49
|
case content
|
|
53
50
|
when Array
|
data/lib/llm/context.rb
CHANGED
|
@@ -257,6 +257,7 @@ module LLM
|
|
|
257
257
|
end
|
|
258
258
|
end.extend(LLM::Function::Array)
|
|
259
259
|
end
|
|
260
|
+
alias_method :pending_functions, :functions
|
|
260
261
|
|
|
261
262
|
##
|
|
262
263
|
# Returns whether there is pending tool work in this context.
|
|
@@ -334,13 +335,11 @@ module LLM
|
|
|
334
335
|
# This is inspired by Go's context cancellation model.
|
|
335
336
|
# @return [nil]
|
|
336
337
|
def interrupt!
|
|
337
|
-
pending = functions.to_a
|
|
338
338
|
llm.interrupt!(@owner)
|
|
339
339
|
queue&.interrupt!
|
|
340
|
-
|
|
341
|
-
|
|
342
|
-
|
|
343
|
-
@messages << LLM::Message.new(@llm.tool_role, returns)
|
|
340
|
+
functions.each(&:interrupt!)
|
|
341
|
+
@queue = nil
|
|
342
|
+
@owner = nil
|
|
344
343
|
nil
|
|
345
344
|
end
|
|
346
345
|
alias_method :cancel!, :interrupt!
|