llm.rb 13.0.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +505 -14
  3. data/README.md +484 -50
  4. data/bin/llm.rb +148 -0
  5. data/data/anthropic.json +206 -263
  6. data/data/bedrock.json +2138 -1860
  7. data/data/deepinfra.json +1003 -624
  8. data/data/deepseek.json +38 -34
  9. data/data/google.json +1079 -371
  10. data/data/mistral.json +448 -368
  11. data/data/moonshot.json +384 -0
  12. data/data/openai.json +974 -1343
  13. data/data/xai.json +154 -126
  14. data/data/zai.json +191 -191
  15. data/lib/llm/agent.rb +123 -20
  16. data/lib/llm/context.rb +71 -88
  17. data/lib/llm/cost.rb +23 -17
  18. data/lib/llm/error.rb +0 -8
  19. data/lib/llm/function/array.rb +3 -3
  20. data/lib/llm/function/async/task.rb +2 -0
  21. data/lib/llm/function/fiber/task.rb +2 -0
  22. data/lib/llm/function/fork/task.rb +2 -0
  23. data/lib/llm/function/ractor/task.rb +2 -0
  24. data/lib/llm/function/sequential/group.rb +4 -1
  25. data/lib/llm/function/sequential/task.rb +1 -1
  26. data/lib/llm/function/task.rb +4 -0
  27. data/lib/llm/function/thread/task.rb +2 -0
  28. data/lib/llm/function.rb +33 -6
  29. data/lib/llm/guard/loop.rb +89 -0
  30. data/lib/llm/guard/null.rb +19 -0
  31. data/lib/llm/guard.rb +61 -0
  32. data/lib/llm/provider.rb +36 -0
  33. data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
  34. data/lib/llm/providers/anthropic.rb +2 -9
  35. data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
  36. data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
  37. data/lib/llm/providers/bedrock.rb +1 -8
  38. data/lib/llm/providers/google/stream_parser.rb +1 -0
  39. data/lib/llm/providers/google.rb +1 -8
  40. data/lib/llm/providers/mistral.rb +1 -1
  41. data/lib/llm/providers/moonshot.rb +76 -0
  42. data/lib/llm/providers/ollama.rb +2 -9
  43. data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
  44. data/lib/llm/providers/openai/responses.rb +7 -9
  45. data/lib/llm/providers/openai/stream_parser.rb +1 -0
  46. data/lib/llm/providers/openai.rb +4 -11
  47. data/lib/llm/repl/bar.rb +4 -3
  48. data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
  49. data/lib/llm/repl/color.rb +78 -0
  50. data/lib/llm/repl/command.rb +12 -5
  51. data/lib/llm/repl/commands/compact.rb +2 -2
  52. data/lib/llm/repl/commands/help.rb +3 -5
  53. data/lib/llm/repl/input/char.rb +46 -0
  54. data/lib/llm/repl/input/row.rb +39 -0
  55. data/lib/llm/repl/input.rb +251 -66
  56. data/lib/llm/repl/markdown/table.rb +11 -3
  57. data/lib/llm/repl/markdown.rb +34 -8
  58. data/lib/llm/repl/node.rb +37 -0
  59. data/lib/llm/repl/status.rb +42 -7
  60. data/lib/llm/repl/stream.rb +18 -6
  61. data/lib/llm/repl/walker.rb +3 -2
  62. data/lib/llm/repl/window.rb +54 -35
  63. data/lib/llm/repl.rb +74 -32
  64. data/lib/llm/skill.rb +20 -4
  65. data/lib/llm/stream.rb +8 -7
  66. data/lib/llm/tool.rb +29 -0
  67. data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
  68. data/lib/llm/tools/git.rb +3 -0
  69. data/lib/llm/tools/mkdir.rb +3 -0
  70. data/lib/llm/tools/rg.rb +3 -0
  71. data/lib/llm/tools/ruby.rb +46 -0
  72. data/lib/llm/tools/shell.rb +3 -0
  73. data/lib/llm/tracer/pretty_logger.rb +127 -0
  74. data/lib/llm/tracer.rb +1 -0
  75. data/lib/llm/transformer/null.rb +21 -0
  76. data/lib/llm/transformer.rb +55 -0
  77. data/lib/llm/version.rb +1 -1
  78. data/lib/llm.rb +12 -2
  79. data/llm.gemspec +9 -2
  80. data/resources/deepdive/advanced/cancellation.md +74 -0
  81. data/resources/deepdive/advanced/compaction.md +83 -0
  82. data/resources/deepdive/advanced/context.md +267 -0
  83. data/resources/deepdive/advanced/guard.md +371 -0
  84. data/resources/deepdive/advanced/tracer.md +180 -0
  85. data/resources/deepdive/advanced/transformer.md +67 -0
  86. data/resources/deepdive/advanced/transports.md +45 -0
  87. data/resources/deepdive/everything_else/audio.md +122 -0
  88. data/resources/deepdive/everything_else/cost.md +99 -0
  89. data/resources/deepdive/everything_else/images.md +89 -0
  90. data/resources/deepdive/everything_else/object.md +108 -0
  91. data/resources/deepdive/everything_else/ocr.md +48 -0
  92. data/resources/deepdive/fundamentals/agents.md +202 -0
  93. data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
  94. data/resources/deepdive/fundamentals/concurrency.md +104 -0
  95. data/resources/deepdive/fundamentals/database.md +449 -0
  96. data/resources/deepdive/fundamentals/embeddings.md +157 -0
  97. data/resources/deepdive/fundamentals/repl.md +87 -0
  98. data/resources/deepdive/fundamentals/schema.md +61 -0
  99. data/resources/deepdive/fundamentals/skills.md +106 -0
  100. data/resources/deepdive/fundamentals/stream.md +110 -0
  101. data/resources/deepdive/fundamentals/tools.md +265 -0
  102. data/resources/deepdive/protocols/a2a.md +106 -0
  103. data/resources/deepdive/protocols/mcp.md +111 -0
  104. data/resources/deepdive.md +58 -1792
  105. metadata +51 -7
  106. data/lib/llm/loop_guard.rb +0 -107
data/data/zai.json CHANGED
@@ -8,10 +8,47 @@
8
8
  "name": "Z.AI",
9
9
  "doc": "https://docs.z.ai/guides/overview/pricing",
10
10
  "models": {
11
- "glm-4.7": {
12
- "id": "glm-4.7",
13
- "name": "GLM-4.7",
14
- "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
11
+ "glm-4.6v": {
12
+ "id": "glm-4.6v",
13
+ "name": "GLM-4.6V",
14
+ "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
15
+ "family": "glm",
16
+ "attachment": true,
17
+ "reasoning": true,
18
+ "reasoning_options": [
19
+ {
20
+ "type": "toggle"
21
+ }
22
+ ],
23
+ "tool_call": true,
24
+ "temperature": true,
25
+ "knowledge": "2025-04",
26
+ "release_date": "2025-12-08",
27
+ "last_updated": "2025-12-08",
28
+ "modalities": {
29
+ "input": [
30
+ "text",
31
+ "image",
32
+ "video"
33
+ ],
34
+ "output": [
35
+ "text"
36
+ ]
37
+ },
38
+ "open_weights": true,
39
+ "limit": {
40
+ "context": 128000,
41
+ "output": 32768
42
+ },
43
+ "cost": {
44
+ "input": 0.3,
45
+ "output": 0.9
46
+ }
47
+ },
48
+ "glm-5": {
49
+ "id": "glm-5",
50
+ "name": "GLM-5",
51
+ "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
15
52
  "family": "glm",
16
53
  "attachment": false,
17
54
  "reasoning": true,
@@ -25,9 +62,8 @@
25
62
  "field": "reasoning_content"
26
63
  },
27
64
  "temperature": true,
28
- "knowledge": "2025-04",
29
- "release_date": "2025-12-22",
30
- "last_updated": "2025-12-22",
65
+ "release_date": "2026-02-12",
66
+ "last_updated": "2026-02-12",
31
67
  "modalities": {
32
68
  "input": [
33
69
  "text"
@@ -42,17 +78,17 @@
42
78
  "output": 131072
43
79
  },
44
80
  "cost": {
45
- "input": 0.6,
46
- "output": 2.2,
47
- "cache_read": 0.11,
81
+ "input": 1,
82
+ "output": 3.2,
83
+ "cache_read": 0.2,
48
84
  "cache_write": 0
49
85
  }
50
86
  },
51
- "glm-4.5": {
52
- "id": "glm-4.5",
53
- "name": "GLM-4.5",
54
- "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
55
- "family": "glm",
87
+ "glm-4.5-air": {
88
+ "id": "glm-4.5-air",
89
+ "name": "GLM-4.5-Air",
90
+ "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
91
+ "family": "glm-air",
56
92
  "attachment": false,
57
93
  "reasoning": true,
58
94
  "reasoning_options": [
@@ -79,16 +115,16 @@
79
115
  "output": 98304
80
116
  },
81
117
  "cost": {
82
- "input": 0.6,
83
- "output": 2.2,
84
- "cache_read": 0.11,
118
+ "input": 0.2,
119
+ "output": 1.1,
120
+ "cache_read": 0.03,
85
121
  "cache_write": 0
86
122
  }
87
123
  },
88
- "glm-5-turbo": {
89
- "id": "glm-5-turbo",
90
- "name": "GLM-5-Turbo",
91
- "description": "Faster GLM-5 lane for coding agents that need lower latency",
124
+ "glm-5.1": {
125
+ "id": "glm-5.1",
126
+ "name": "GLM-5.1",
127
+ "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
92
128
  "family": "glm",
93
129
  "attachment": false,
94
130
  "reasoning": true,
@@ -103,8 +139,8 @@
103
139
  },
104
140
  "structured_output": true,
105
141
  "temperature": true,
106
- "release_date": "2026-03-16",
107
- "last_updated": "2026-03-16",
142
+ "release_date": "2026-04-07",
143
+ "last_updated": "2026-04-07",
108
144
  "modalities": {
109
145
  "input": [
110
146
  "text"
@@ -113,22 +149,22 @@
113
149
  "text"
114
150
  ]
115
151
  },
116
- "open_weights": false,
152
+ "open_weights": true,
117
153
  "limit": {
118
154
  "context": 200000,
119
155
  "output": 131072
120
156
  },
121
157
  "cost": {
122
- "input": 1.2,
123
- "output": 4,
124
- "cache_read": 0.24,
158
+ "input": 1.4,
159
+ "output": 4.4,
160
+ "cache_read": 0.26,
125
161
  "cache_write": 0
126
162
  }
127
163
  },
128
- "glm-4.7-flashx": {
129
- "id": "glm-4.7-flashx",
130
- "name": "GLM-4.7-FlashX",
131
- "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
164
+ "glm-4.7-flash": {
165
+ "id": "glm-4.7-flash",
166
+ "name": "GLM-4.7-Flash",
167
+ "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
132
168
  "family": "glm-flash",
133
169
  "attachment": false,
134
170
  "reasoning": true,
@@ -156,29 +192,36 @@
156
192
  "output": 131072
157
193
  },
158
194
  "cost": {
159
- "input": 0.07,
160
- "output": 0.4,
161
- "cache_read": 0.01,
195
+ "input": 0,
196
+ "output": 0,
197
+ "cache_read": 0,
162
198
  "cache_write": 0
163
199
  }
164
200
  },
165
- "glm-4.5-air": {
166
- "id": "glm-4.5-air",
167
- "name": "GLM-4.5-Air",
168
- "description": "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents",
169
- "family": "glm-air",
201
+ "glm-5.2": {
202
+ "id": "glm-5.2",
203
+ "name": "GLM-5.2",
204
+ "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
205
+ "family": "glm",
170
206
  "attachment": false,
171
207
  "reasoning": true,
172
208
  "reasoning_options": [
173
209
  {
174
- "type": "toggle"
210
+ "type": "effort",
211
+ "values": [
212
+ "high",
213
+ "max"
214
+ ]
175
215
  }
176
216
  ],
177
217
  "tool_call": true,
218
+ "interleaved": {
219
+ "field": "reasoning_content"
220
+ },
221
+ "structured_output": true,
178
222
  "temperature": true,
179
- "knowledge": "2025-04",
180
- "release_date": "2025-07-28",
181
- "last_updated": "2025-07-28",
223
+ "release_date": "2026-06-13",
224
+ "last_updated": "2026-06-13",
182
225
  "modalities": {
183
226
  "input": [
184
227
  "text"
@@ -189,19 +232,19 @@
189
232
  },
190
233
  "open_weights": true,
191
234
  "limit": {
192
- "context": 131072,
193
- "output": 98304
235
+ "context": 1000000,
236
+ "output": 131072
194
237
  },
195
238
  "cost": {
196
- "input": 0.2,
197
- "output": 1.1,
198
- "cache_read": 0.03,
239
+ "input": 1.4,
240
+ "output": 4.4,
241
+ "cache_read": 0.26,
199
242
  "cache_write": 0
200
243
  }
201
244
  },
202
- "glm-4.5-flash": {
203
- "id": "glm-4.5-flash",
204
- "name": "GLM-4.5-Flash",
245
+ "glm-4.7-flashx": {
246
+ "id": "glm-4.7-flashx",
247
+ "name": "GLM-4.7-FlashX",
205
248
  "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
206
249
  "family": "glm-flash",
207
250
  "attachment": false,
@@ -214,8 +257,8 @@
214
257
  "tool_call": true,
215
258
  "temperature": true,
216
259
  "knowledge": "2025-04",
217
- "release_date": "2025-07-28",
218
- "last_updated": "2025-07-28",
260
+ "release_date": "2026-01-19",
261
+ "last_updated": "2026-01-19",
219
262
  "modalities": {
220
263
  "input": [
221
264
  "text"
@@ -226,22 +269,22 @@
226
269
  },
227
270
  "open_weights": true,
228
271
  "limit": {
229
- "context": 131072,
230
- "output": 98304
272
+ "context": 200000,
273
+ "output": 131072
231
274
  },
232
275
  "cost": {
233
- "input": 0,
234
- "output": 0,
235
- "cache_read": 0,
276
+ "input": 0.07,
277
+ "output": 0.4,
278
+ "cache_read": 0.01,
236
279
  "cache_write": 0
237
280
  }
238
281
  },
239
- "glm-4.6v": {
240
- "id": "glm-4.6v",
241
- "name": "GLM-4.6V",
242
- "description": "GLM vision model for visual reasoning, documents, and multimodal agents",
282
+ "glm-4.6": {
283
+ "id": "glm-4.6",
284
+ "name": "GLM-4.6",
285
+ "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
243
286
  "family": "glm",
244
- "attachment": true,
287
+ "attachment": false,
245
288
  "reasoning": true,
246
289
  "reasoning_options": [
247
290
  {
@@ -251,13 +294,11 @@
251
294
  "tool_call": true,
252
295
  "temperature": true,
253
296
  "knowledge": "2025-04",
254
- "release_date": "2025-12-08",
255
- "last_updated": "2025-12-08",
297
+ "release_date": "2025-09-30",
298
+ "last_updated": "2025-09-30",
256
299
  "modalities": {
257
300
  "input": [
258
- "text",
259
- "image",
260
- "video"
301
+ "text"
261
302
  ],
262
303
  "output": [
263
304
  "text"
@@ -265,12 +306,51 @@
265
306
  },
266
307
  "open_weights": true,
267
308
  "limit": {
268
- "context": 128000,
269
- "output": 32768
309
+ "context": 204800,
310
+ "output": 131072
270
311
  },
271
312
  "cost": {
272
- "input": 0.3,
273
- "output": 0.9
313
+ "input": 0.6,
314
+ "output": 2.2,
315
+ "cache_read": 0.11,
316
+ "cache_write": 0
317
+ }
318
+ },
319
+ "glm-4.5": {
320
+ "id": "glm-4.5",
321
+ "name": "GLM-4.5",
322
+ "description": "Hybrid-reasoning GLM release that made the 4.5 line broadly useful",
323
+ "family": "glm",
324
+ "attachment": false,
325
+ "reasoning": true,
326
+ "reasoning_options": [
327
+ {
328
+ "type": "toggle"
329
+ }
330
+ ],
331
+ "tool_call": true,
332
+ "temperature": true,
333
+ "knowledge": "2025-04",
334
+ "release_date": "2025-07-28",
335
+ "last_updated": "2025-07-28",
336
+ "modalities": {
337
+ "input": [
338
+ "text"
339
+ ],
340
+ "output": [
341
+ "text"
342
+ ]
343
+ },
344
+ "open_weights": true,
345
+ "limit": {
346
+ "context": 131072,
347
+ "output": 98304
348
+ },
349
+ "cost": {
350
+ "input": 0.6,
351
+ "output": 2.2,
352
+ "cache_read": 0.11,
353
+ "cache_write": 0
274
354
  }
275
355
  },
276
356
  "glm-4.5v": {
@@ -310,11 +390,11 @@
310
390
  "output": 1.8
311
391
  }
312
392
  },
313
- "glm-4.7-flash": {
314
- "id": "glm-4.7-flash",
315
- "name": "GLM-4.7-Flash",
316
- "description": "Budget GLM lane for fast coding help, routing, and everyday automation",
317
- "family": "glm-flash",
393
+ "glm-4.7": {
394
+ "id": "glm-4.7",
395
+ "name": "GLM-4.7",
396
+ "description": "Mature GLM model for dependable coding, reasoning, and structured agent tasks",
397
+ "family": "glm",
318
398
  "attachment": false,
319
399
  "reasoning": true,
320
400
  "reasoning_options": [
@@ -323,10 +403,13 @@
323
403
  }
324
404
  ],
325
405
  "tool_call": true,
406
+ "interleaved": {
407
+ "field": "reasoning_content"
408
+ },
326
409
  "temperature": true,
327
410
  "knowledge": "2025-04",
328
- "release_date": "2026-01-19",
329
- "last_updated": "2026-01-19",
411
+ "release_date": "2025-12-22",
412
+ "last_updated": "2025-12-22",
330
413
  "modalities": {
331
414
  "input": [
332
415
  "text"
@@ -337,30 +420,26 @@
337
420
  },
338
421
  "open_weights": true,
339
422
  "limit": {
340
- "context": 200000,
423
+ "context": 204800,
341
424
  "output": 131072
342
425
  },
343
426
  "cost": {
344
- "input": 0,
345
- "output": 0,
346
- "cache_read": 0,
427
+ "input": 0.6,
428
+ "output": 2.2,
429
+ "cache_read": 0.11,
347
430
  "cache_write": 0
348
431
  }
349
432
  },
350
- "glm-5.2": {
351
- "id": "glm-5.2",
352
- "name": "GLM-5.2",
353
- "description": "Open flagship GLM for long-horizon coding agents and million-token context work",
433
+ "glm-5-turbo": {
434
+ "id": "glm-5-turbo",
435
+ "name": "GLM-5-Turbo",
436
+ "description": "Faster GLM-5 lane for coding agents that need lower latency",
354
437
  "family": "glm",
355
438
  "attachment": false,
356
439
  "reasoning": true,
357
440
  "reasoning_options": [
358
441
  {
359
- "type": "effort",
360
- "values": [
361
- "high",
362
- "max"
363
- ]
442
+ "type": "toggle"
364
443
  }
365
444
  ],
366
445
  "tool_call": true,
@@ -369,8 +448,8 @@
369
448
  },
370
449
  "structured_output": true,
371
450
  "temperature": true,
372
- "release_date": "2026-06-13",
373
- "last_updated": "2026-06-13",
451
+ "release_date": "2026-03-16",
452
+ "last_updated": "2026-03-16",
374
453
  "modalities": {
375
454
  "input": [
376
455
  "text"
@@ -379,15 +458,15 @@
379
458
  "text"
380
459
  ]
381
460
  },
382
- "open_weights": true,
461
+ "open_weights": false,
383
462
  "limit": {
384
- "context": 1000000,
463
+ "context": 200000,
385
464
  "output": 131072
386
465
  },
387
466
  "cost": {
388
- "input": 1.4,
389
- "output": 4.4,
390
- "cache_read": 0.26,
467
+ "input": 1.2,
468
+ "output": 4,
469
+ "cache_read": 0.24,
391
470
  "cache_write": 0
392
471
  }
393
472
  },
@@ -433,90 +512,11 @@
433
512
  "cache_write": 0
434
513
  }
435
514
  },
436
- "glm-5": {
437
- "id": "glm-5",
438
- "name": "GLM-5",
439
- "description": "General GLM flagship for coding, analysis, and tool-heavy engineering workflows",
440
- "family": "glm",
441
- "attachment": false,
442
- "reasoning": true,
443
- "reasoning_options": [
444
- {
445
- "type": "toggle"
446
- }
447
- ],
448
- "tool_call": true,
449
- "interleaved": {
450
- "field": "reasoning_content"
451
- },
452
- "temperature": true,
453
- "release_date": "2026-02-12",
454
- "last_updated": "2026-02-12",
455
- "modalities": {
456
- "input": [
457
- "text"
458
- ],
459
- "output": [
460
- "text"
461
- ]
462
- },
463
- "open_weights": true,
464
- "limit": {
465
- "context": 204800,
466
- "output": 131072
467
- },
468
- "cost": {
469
- "input": 1,
470
- "output": 3.2,
471
- "cache_read": 0.2,
472
- "cache_write": 0
473
- }
474
- },
475
- "glm-5.1": {
476
- "id": "glm-5.1",
477
- "name": "GLM-5.1",
478
- "description": "Strong GLM coding model for agentic engineering, terminals, and repository generation",
479
- "family": "glm",
480
- "attachment": false,
481
- "reasoning": true,
482
- "reasoning_options": [
483
- {
484
- "type": "toggle"
485
- }
486
- ],
487
- "tool_call": true,
488
- "interleaved": {
489
- "field": "reasoning_content"
490
- },
491
- "structured_output": true,
492
- "temperature": true,
493
- "release_date": "2026-04-07",
494
- "last_updated": "2026-04-07",
495
- "modalities": {
496
- "input": [
497
- "text"
498
- ],
499
- "output": [
500
- "text"
501
- ]
502
- },
503
- "open_weights": true,
504
- "limit": {
505
- "context": 200000,
506
- "output": 131072
507
- },
508
- "cost": {
509
- "input": 1.4,
510
- "output": 4.4,
511
- "cache_read": 0.26,
512
- "cache_write": 0
513
- }
514
- },
515
- "glm-4.6": {
516
- "id": "glm-4.6",
517
- "name": "GLM-4.6",
518
- "description": "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks",
519
- "family": "glm",
515
+ "glm-4.5-flash": {
516
+ "id": "glm-4.5-flash",
517
+ "name": "GLM-4.5-Flash",
518
+ "description": "Efficient GLM model for fast reasoning, coding, and agent workflows",
519
+ "family": "glm-flash",
520
520
  "attachment": false,
521
521
  "reasoning": true,
522
522
  "reasoning_options": [
@@ -527,8 +527,8 @@
527
527
  "tool_call": true,
528
528
  "temperature": true,
529
529
  "knowledge": "2025-04",
530
- "release_date": "2025-09-30",
531
- "last_updated": "2025-09-30",
530
+ "release_date": "2025-07-28",
531
+ "last_updated": "2025-07-28",
532
532
  "modalities": {
533
533
  "input": [
534
534
  "text"
@@ -539,13 +539,13 @@
539
539
  },
540
540
  "open_weights": true,
541
541
  "limit": {
542
- "context": 204800,
543
- "output": 131072
542
+ "context": 131072,
543
+ "output": 98304
544
544
  },
545
545
  "cost": {
546
- "input": 0.6,
547
- "output": 2.2,
548
- "cache_read": 0.11,
546
+ "input": 0,
547
+ "output": 0,
548
+ "cache_read": 0,
549
549
  "cache_write": 0
550
550
  }
551
551
  }