llm.rb 15.0.2 → 15.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +161 -1
- data/README.md +134 -85
- data/bin/llm.rb +38 -3
- data/data/bedrock.json +250 -0
- data/data/deepinfra.json +162 -0
- data/data/deepseek.json +50 -0
- data/data/google.json +19 -21
- data/data/openrouter.json +14280 -0
- data/data/xai.json +28 -16
- data/docs/deepdive/advanced/context.md +8 -6
- data/docs/deepdive/fundamentals/agents.md +11 -10
- data/docs/deepdive/fundamentals/stream.md +4 -4
- data/lib/llm/active_record.rb +1 -1
- data/lib/llm/agent.rb +6 -0
- data/lib/llm/context.rb +28 -13
- data/lib/llm/cost.rb +13 -0
- data/lib/llm/function/fork/task.rb +14 -10
- data/lib/llm/provider.rb +29 -8
- data/lib/llm/providers/anthropic.rb +1 -1
- data/lib/llm/providers/bedrock/models.rb +2 -2
- data/lib/llm/providers/bedrock.rb +1 -1
- data/lib/llm/providers/google.rb +1 -1
- data/lib/llm/providers/ollama.rb +1 -1
- data/lib/llm/providers/openai/responses.rb +2 -1
- data/lib/llm/providers/openai.rb +1 -1
- data/lib/llm/providers/openrouter.rb +87 -0
- data/lib/llm/repl/buffer.rb +20 -5
- data/lib/llm/repl/input.rb +5 -4
- data/lib/llm/repl/markdown/table.rb +5 -2
- data/lib/llm/repl/node.rb +26 -1
- data/lib/llm/repl/stream.rb +30 -3
- data/lib/llm/repl/window.rb +7 -7
- data/lib/llm/repl.rb +1 -0
- data/lib/llm/skill.rb +7 -1
- data/lib/llm/stream.rb +8 -3
- data/lib/llm/tools/git.rb +2 -2
- data/lib/llm/tools/mkdir.rb +2 -2
- data/lib/llm/tools/rg.rb +2 -2
- data/lib/llm/tools/ruby.rb +2 -2
- data/lib/llm/tools/shell.rb +2 -2
- data/lib/llm/tools/utils.rb +1 -1
- data/lib/llm/transport/curb.rb +5 -3
- data/lib/llm/transport/http.rb +5 -2
- data/lib/llm/transport/persistent_http.rb +6 -4
- data/lib/llm/transport/utils.rb +8 -6
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +16 -1
- data/llm.gemspec +2 -1
- metadata +19 -3
data/bin/llm.rb
CHANGED
|
@@ -49,16 +49,20 @@ def help
|
|
|
49
49
|
warn ""
|
|
50
50
|
warn "Options:"
|
|
51
51
|
warn " -p PROVIDER Choose a provider"
|
|
52
|
+
warn " -m MODEL Choose a model"
|
|
52
53
|
warn " -c STRATEGY Concurrency strategy for tool calls (eg thread, async, fork)"
|
|
53
54
|
warn " -n TRANSPORT HTTP transports - net-http (default), net-http-persistent, and curb"
|
|
55
|
+
warn " -x TIMEOUT The default read timeout (in seconds)"
|
|
54
56
|
warn " -t Temporary session that doesn't persist to disk"
|
|
55
57
|
warn " -h Show this help"
|
|
56
58
|
warn ""
|
|
57
59
|
warn "Examples:"
|
|
58
60
|
warn " #{prog} # auto-detect provider from $PROVIDER_API_KEY"
|
|
59
61
|
warn " #{prog} -p openai # use OpenAI"
|
|
62
|
+
warn " #{prog} -m gpt-5.6 # use a model other than the provider default"
|
|
60
63
|
warn " #{prog} -n curb # use libcurl"
|
|
61
64
|
warn " #{prog} -c thread # run tool calls on a separate thread"
|
|
65
|
+
warn " #{prog} -x 900 # read timeout of 15mins"
|
|
62
66
|
warn " #{prog} -h # this help"
|
|
63
67
|
warn ""
|
|
64
68
|
end
|
|
@@ -94,6 +98,11 @@ def fatal(ex)
|
|
|
94
98
|
wrapped detail, " "
|
|
95
99
|
end
|
|
96
100
|
warn ""
|
|
101
|
+
warn " Backtrace:"
|
|
102
|
+
ex.backtrace.drop(1).first(3).each do |line|
|
|
103
|
+
wrapped line, " "
|
|
104
|
+
end
|
|
105
|
+
warn ""
|
|
97
106
|
warn " This is an unexpected error. If it keeps happening,"
|
|
98
107
|
warn " consider opening an issue at"
|
|
99
108
|
warn " https://github.com/r-uby-dev/llm/issues"
|
|
@@ -153,6 +162,22 @@ def main(argv)
|
|
|
153
162
|
help
|
|
154
163
|
exit 1
|
|
155
164
|
end
|
|
165
|
+
when '-m'
|
|
166
|
+
model = argv.shift
|
|
167
|
+
if model.nil?
|
|
168
|
+
warn "llm.rb: -m switch requires an argument"
|
|
169
|
+
help
|
|
170
|
+
exit 1
|
|
171
|
+
end
|
|
172
|
+
when '-x'
|
|
173
|
+
timeout = argv.shift
|
|
174
|
+
if timeout.nil?
|
|
175
|
+
warn "llm.rb: -x switch requires an argument"
|
|
176
|
+
help
|
|
177
|
+
exit 1
|
|
178
|
+
else
|
|
179
|
+
timeout = Integer(timeout)
|
|
180
|
+
end
|
|
156
181
|
else
|
|
157
182
|
warn "llm.rb: unknown option #{option}"
|
|
158
183
|
help
|
|
@@ -164,9 +189,10 @@ def main(argv)
|
|
|
164
189
|
# No provider has been given.
|
|
165
190
|
# Try to infer one.
|
|
166
191
|
transport ||= :net_http
|
|
192
|
+
options = timeout ? {timeout:, transport:} : {transport:}
|
|
167
193
|
if provider.nil?
|
|
168
194
|
llm = providers.filter_map do
|
|
169
|
-
LLM.method(_1).call(
|
|
195
|
+
LLM.method(_1).call(**options)
|
|
170
196
|
rescue ArgumentError
|
|
171
197
|
end.first
|
|
172
198
|
if llm.nil?
|
|
@@ -175,7 +201,7 @@ def main(argv)
|
|
|
175
201
|
end
|
|
176
202
|
else
|
|
177
203
|
begin
|
|
178
|
-
llm = LLM.method(provider).call(
|
|
204
|
+
llm = LLM.method(provider).call(**options)
|
|
179
205
|
rescue ArgumentError
|
|
180
206
|
warn "llm.rb: set credentials for #{provider}"
|
|
181
207
|
exit 1
|
|
@@ -209,11 +235,20 @@ def main(argv)
|
|
|
209
235
|
File.binwrite file, JSON.pretty_generate(data)
|
|
210
236
|
end
|
|
211
237
|
|
|
238
|
+
if model
|
|
239
|
+
if not llm.registry.keys.include?(model)
|
|
240
|
+
warn "llm.rb: #{model} is not a valid #{llm.name} model"
|
|
241
|
+
exit 1
|
|
242
|
+
end
|
|
243
|
+
else
|
|
244
|
+
model = llm.default_model
|
|
245
|
+
end
|
|
246
|
+
|
|
212
247
|
##
|
|
213
248
|
# Let's go!
|
|
214
249
|
concurrency ||= :sequential
|
|
215
250
|
path = temp ? nil : data[Dir.getwd]
|
|
216
|
-
agent = LLM::Agent.new(llm, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
251
|
+
agent = LLM::Agent.new(llm, model:, path:, concurrency:, tools: LLM::Tool.subclasses)
|
|
217
252
|
agent.repl
|
|
218
253
|
rescue Interrupt
|
|
219
254
|
warn "llm.rb: Bye!"
|
data/data/bedrock.json
CHANGED
|
@@ -435,6 +435,73 @@
|
|
|
435
435
|
"cache_write": 4.125
|
|
436
436
|
}
|
|
437
437
|
},
|
|
438
|
+
"global.openai.gpt-5.6-sol": {
|
|
439
|
+
"id": "global.openai.gpt-5.6-sol",
|
|
440
|
+
"name": "GPT-5.6 Sol (Global)",
|
|
441
|
+
"description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows",
|
|
442
|
+
"family": "gpt-sol",
|
|
443
|
+
"attachment": true,
|
|
444
|
+
"reasoning": true,
|
|
445
|
+
"reasoning_options": [
|
|
446
|
+
{
|
|
447
|
+
"type": "effort",
|
|
448
|
+
"values": [
|
|
449
|
+
"none",
|
|
450
|
+
"low",
|
|
451
|
+
"medium",
|
|
452
|
+
"high",
|
|
453
|
+
"xhigh",
|
|
454
|
+
"max"
|
|
455
|
+
]
|
|
456
|
+
}
|
|
457
|
+
],
|
|
458
|
+
"tool_call": true,
|
|
459
|
+
"structured_output": true,
|
|
460
|
+
"temperature": false,
|
|
461
|
+
"knowledge": "2026-02-16",
|
|
462
|
+
"release_date": "2026-07-09",
|
|
463
|
+
"last_updated": "2026-07-09",
|
|
464
|
+
"modalities": {
|
|
465
|
+
"input": [
|
|
466
|
+
"text",
|
|
467
|
+
"image",
|
|
468
|
+
"pdf"
|
|
469
|
+
],
|
|
470
|
+
"output": [
|
|
471
|
+
"text"
|
|
472
|
+
]
|
|
473
|
+
},
|
|
474
|
+
"open_weights": false,
|
|
475
|
+
"limit": {
|
|
476
|
+
"context": 1050000,
|
|
477
|
+
"input": 922000,
|
|
478
|
+
"output": 128000
|
|
479
|
+
},
|
|
480
|
+
"cost": {
|
|
481
|
+
"input": 5.5,
|
|
482
|
+
"output": 33,
|
|
483
|
+
"cache_read": 0.55,
|
|
484
|
+
"cache_write": 6.875,
|
|
485
|
+
"tiers": [
|
|
486
|
+
{
|
|
487
|
+
"input": 11,
|
|
488
|
+
"output": 49.5,
|
|
489
|
+
"cache_read": 1.1,
|
|
490
|
+
"cache_write": 13.75,
|
|
491
|
+
"tier": {
|
|
492
|
+
"type": "context",
|
|
493
|
+
"size": 272000
|
|
494
|
+
}
|
|
495
|
+
}
|
|
496
|
+
],
|
|
497
|
+
"context_over_200k": {
|
|
498
|
+
"input": 11,
|
|
499
|
+
"output": 49.5,
|
|
500
|
+
"cache_read": 1.1,
|
|
501
|
+
"cache_write": 13.75
|
|
502
|
+
}
|
|
503
|
+
}
|
|
504
|
+
},
|
|
438
505
|
"jp.anthropic.claude-sonnet-4-6": {
|
|
439
506
|
"id": "jp.anthropic.claude-sonnet-4-6",
|
|
440
507
|
"name": "Claude Sonnet 4.6 (JP)",
|
|
@@ -1907,6 +1974,55 @@
|
|
|
1907
1974
|
"cache_write": 6.25
|
|
1908
1975
|
}
|
|
1909
1976
|
},
|
|
1977
|
+
"xai.grok-4.6": {
|
|
1978
|
+
"id": "xai.grok-4.6",
|
|
1979
|
+
"name": "Grok 4.6",
|
|
1980
|
+
"description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
|
|
1981
|
+
"family": "grok",
|
|
1982
|
+
"attachment": true,
|
|
1983
|
+
"reasoning": true,
|
|
1984
|
+
"reasoning_options": [
|
|
1985
|
+
{
|
|
1986
|
+
"type": "effort",
|
|
1987
|
+
"values": [
|
|
1988
|
+
"low",
|
|
1989
|
+
"medium",
|
|
1990
|
+
"high",
|
|
1991
|
+
"xhigh"
|
|
1992
|
+
]
|
|
1993
|
+
}
|
|
1994
|
+
],
|
|
1995
|
+
"tool_call": true,
|
|
1996
|
+
"structured_output": true,
|
|
1997
|
+
"temperature": true,
|
|
1998
|
+
"knowledge": "2026-02-01",
|
|
1999
|
+
"release_date": "2026-08-12",
|
|
2000
|
+
"last_updated": "2026-08-18",
|
|
2001
|
+
"modalities": {
|
|
2002
|
+
"input": [
|
|
2003
|
+
"text",
|
|
2004
|
+
"image"
|
|
2005
|
+
],
|
|
2006
|
+
"output": [
|
|
2007
|
+
"text"
|
|
2008
|
+
]
|
|
2009
|
+
},
|
|
2010
|
+
"open_weights": false,
|
|
2011
|
+
"limit": {
|
|
2012
|
+
"context": 500000,
|
|
2013
|
+
"output": 500000
|
|
2014
|
+
},
|
|
2015
|
+
"provider": {
|
|
2016
|
+
"npm": "@ai-sdk/amazon-bedrock/mantle",
|
|
2017
|
+
"api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1",
|
|
2018
|
+
"shape": "responses"
|
|
2019
|
+
},
|
|
2020
|
+
"cost": {
|
|
2021
|
+
"input": 2.2,
|
|
2022
|
+
"output": 6.6,
|
|
2023
|
+
"cache_read": 0.55
|
|
2024
|
+
}
|
|
2025
|
+
},
|
|
1910
2026
|
"anthropic.claude-sonnet-4-6": {
|
|
1911
2027
|
"id": "anthropic.claude-sonnet-4-6",
|
|
1912
2028
|
"name": "Claude Sonnet 4.6",
|
|
@@ -3167,6 +3283,73 @@
|
|
|
3167
3283
|
"cache_read": 0.2
|
|
3168
3284
|
}
|
|
3169
3285
|
},
|
|
3286
|
+
"global.openai.gpt-5.6-luna": {
|
|
3287
|
+
"id": "global.openai.gpt-5.6-luna",
|
|
3288
|
+
"name": "GPT-5.6 Luna (Global)",
|
|
3289
|
+
"description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads",
|
|
3290
|
+
"family": "gpt-luna",
|
|
3291
|
+
"attachment": true,
|
|
3292
|
+
"reasoning": true,
|
|
3293
|
+
"reasoning_options": [
|
|
3294
|
+
{
|
|
3295
|
+
"type": "effort",
|
|
3296
|
+
"values": [
|
|
3297
|
+
"none",
|
|
3298
|
+
"low",
|
|
3299
|
+
"medium",
|
|
3300
|
+
"high",
|
|
3301
|
+
"xhigh",
|
|
3302
|
+
"max"
|
|
3303
|
+
]
|
|
3304
|
+
}
|
|
3305
|
+
],
|
|
3306
|
+
"tool_call": true,
|
|
3307
|
+
"structured_output": true,
|
|
3308
|
+
"temperature": false,
|
|
3309
|
+
"knowledge": "2026-02-16",
|
|
3310
|
+
"release_date": "2026-07-09",
|
|
3311
|
+
"last_updated": "2026-07-09",
|
|
3312
|
+
"modalities": {
|
|
3313
|
+
"input": [
|
|
3314
|
+
"text",
|
|
3315
|
+
"image",
|
|
3316
|
+
"pdf"
|
|
3317
|
+
],
|
|
3318
|
+
"output": [
|
|
3319
|
+
"text"
|
|
3320
|
+
]
|
|
3321
|
+
},
|
|
3322
|
+
"open_weights": false,
|
|
3323
|
+
"limit": {
|
|
3324
|
+
"context": 1050000,
|
|
3325
|
+
"input": 922000,
|
|
3326
|
+
"output": 128000
|
|
3327
|
+
},
|
|
3328
|
+
"cost": {
|
|
3329
|
+
"input": 0.22,
|
|
3330
|
+
"output": 1.32,
|
|
3331
|
+
"cache_read": 0.022,
|
|
3332
|
+
"cache_write": 0.275,
|
|
3333
|
+
"tiers": [
|
|
3334
|
+
{
|
|
3335
|
+
"input": 0.44,
|
|
3336
|
+
"output": 1.98,
|
|
3337
|
+
"cache_read": 0.044,
|
|
3338
|
+
"cache_write": 0.55,
|
|
3339
|
+
"tier": {
|
|
3340
|
+
"type": "context",
|
|
3341
|
+
"size": 272000
|
|
3342
|
+
}
|
|
3343
|
+
}
|
|
3344
|
+
],
|
|
3345
|
+
"context_over_200k": {
|
|
3346
|
+
"input": 0.44,
|
|
3347
|
+
"output": 1.98,
|
|
3348
|
+
"cache_read": 0.044,
|
|
3349
|
+
"cache_write": 0.55
|
|
3350
|
+
}
|
|
3351
|
+
}
|
|
3352
|
+
},
|
|
3170
3353
|
"zai.glm-4.7": {
|
|
3171
3354
|
"id": "zai.glm-4.7",
|
|
3172
3355
|
"name": "GLM-4.7",
|
|
@@ -3867,6 +4050,73 @@
|
|
|
3867
4050
|
"cache_write": 3.75
|
|
3868
4051
|
}
|
|
3869
4052
|
},
|
|
4053
|
+
"global.openai.gpt-5.6-terra": {
|
|
4054
|
+
"id": "global.openai.gpt-5.6-terra",
|
|
4055
|
+
"name": "GPT-5.6 Terra (Global)",
|
|
4056
|
+
"description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work",
|
|
4057
|
+
"family": "gpt-terra",
|
|
4058
|
+
"attachment": true,
|
|
4059
|
+
"reasoning": true,
|
|
4060
|
+
"reasoning_options": [
|
|
4061
|
+
{
|
|
4062
|
+
"type": "effort",
|
|
4063
|
+
"values": [
|
|
4064
|
+
"none",
|
|
4065
|
+
"low",
|
|
4066
|
+
"medium",
|
|
4067
|
+
"high",
|
|
4068
|
+
"xhigh",
|
|
4069
|
+
"max"
|
|
4070
|
+
]
|
|
4071
|
+
}
|
|
4072
|
+
],
|
|
4073
|
+
"tool_call": true,
|
|
4074
|
+
"structured_output": true,
|
|
4075
|
+
"temperature": false,
|
|
4076
|
+
"knowledge": "2026-02-16",
|
|
4077
|
+
"release_date": "2026-07-09",
|
|
4078
|
+
"last_updated": "2026-07-09",
|
|
4079
|
+
"modalities": {
|
|
4080
|
+
"input": [
|
|
4081
|
+
"text",
|
|
4082
|
+
"image",
|
|
4083
|
+
"pdf"
|
|
4084
|
+
],
|
|
4085
|
+
"output": [
|
|
4086
|
+
"text"
|
|
4087
|
+
]
|
|
4088
|
+
},
|
|
4089
|
+
"open_weights": false,
|
|
4090
|
+
"limit": {
|
|
4091
|
+
"context": 1050000,
|
|
4092
|
+
"input": 922000,
|
|
4093
|
+
"output": 128000
|
|
4094
|
+
},
|
|
4095
|
+
"cost": {
|
|
4096
|
+
"input": 2.2,
|
|
4097
|
+
"output": 13.2,
|
|
4098
|
+
"cache_read": 0.22,
|
|
4099
|
+
"cache_write": 2.75,
|
|
4100
|
+
"tiers": [
|
|
4101
|
+
{
|
|
4102
|
+
"input": 4.4,
|
|
4103
|
+
"output": 19.8,
|
|
4104
|
+
"cache_read": 0.44,
|
|
4105
|
+
"cache_write": 5.5,
|
|
4106
|
+
"tier": {
|
|
4107
|
+
"type": "context",
|
|
4108
|
+
"size": 272000
|
|
4109
|
+
}
|
|
4110
|
+
}
|
|
4111
|
+
],
|
|
4112
|
+
"context_over_200k": {
|
|
4113
|
+
"input": 4.4,
|
|
4114
|
+
"output": 19.8,
|
|
4115
|
+
"cache_read": 0.44,
|
|
4116
|
+
"cache_write": 5.5
|
|
4117
|
+
}
|
|
4118
|
+
}
|
|
4119
|
+
},
|
|
3870
4120
|
"openai.gpt-oss-20b-1:0": {
|
|
3871
4121
|
"id": "openai.gpt-oss-20b-1:0",
|
|
3872
4122
|
"name": "gpt-oss-20b",
|
data/data/deepinfra.json
CHANGED
|
@@ -1053,6 +1053,49 @@
|
|
|
1053
1053
|
"cache_read": 0.135
|
|
1054
1054
|
}
|
|
1055
1055
|
},
|
|
1056
|
+
"deepseek-ai/DeepSeek-V4-Pro-0813": {
|
|
1057
|
+
"id": "deepseek-ai/DeepSeek-V4-Pro-0813",
|
|
1058
|
+
"name": "DeepSeek V4 Pro 0813",
|
|
1059
|
+
"description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes",
|
|
1060
|
+
"family": "deepseek-thinking",
|
|
1061
|
+
"attachment": false,
|
|
1062
|
+
"reasoning": true,
|
|
1063
|
+
"reasoning_options": [
|
|
1064
|
+
{
|
|
1065
|
+
"type": "toggle"
|
|
1066
|
+
},
|
|
1067
|
+
{
|
|
1068
|
+
"type": "effort",
|
|
1069
|
+
"values": [
|
|
1070
|
+
"high",
|
|
1071
|
+
"max"
|
|
1072
|
+
]
|
|
1073
|
+
}
|
|
1074
|
+
],
|
|
1075
|
+
"tool_call": true,
|
|
1076
|
+
"structured_output": true,
|
|
1077
|
+
"temperature": true,
|
|
1078
|
+
"release_date": "2026-08-12",
|
|
1079
|
+
"last_updated": "2026-08-12",
|
|
1080
|
+
"modalities": {
|
|
1081
|
+
"input": [
|
|
1082
|
+
"text"
|
|
1083
|
+
],
|
|
1084
|
+
"output": [
|
|
1085
|
+
"text"
|
|
1086
|
+
]
|
|
1087
|
+
},
|
|
1088
|
+
"open_weights": false,
|
|
1089
|
+
"limit": {
|
|
1090
|
+
"context": 1048576,
|
|
1091
|
+
"output": 384000
|
|
1092
|
+
},
|
|
1093
|
+
"cost": {
|
|
1094
|
+
"input": 1.3,
|
|
1095
|
+
"output": 2.6,
|
|
1096
|
+
"cache_read": 0.1
|
|
1097
|
+
}
|
|
1098
|
+
},
|
|
1056
1099
|
"deepseek-ai/DeepSeek-V3": {
|
|
1057
1100
|
"id": "deepseek-ai/DeepSeek-V3",
|
|
1058
1101
|
"name": "DeepSeek-V3",
|
|
@@ -1195,6 +1238,39 @@
|
|
|
1195
1238
|
"output": 0.55
|
|
1196
1239
|
}
|
|
1197
1240
|
},
|
|
1241
|
+
"Qwen/Qwen3-VL-235B-A22B-Instruct": {
|
|
1242
|
+
"id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
|
|
1243
|
+
"name": "Qwen3 VL 235B A22B Instruct",
|
|
1244
|
+
"description": "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks",
|
|
1245
|
+
"family": "qwen",
|
|
1246
|
+
"attachment": true,
|
|
1247
|
+
"reasoning": false,
|
|
1248
|
+
"tool_call": true,
|
|
1249
|
+
"structured_output": true,
|
|
1250
|
+
"temperature": true,
|
|
1251
|
+
"knowledge": "2025-03-31",
|
|
1252
|
+
"release_date": "2025-09-23",
|
|
1253
|
+
"last_updated": "2025-09-23",
|
|
1254
|
+
"modalities": {
|
|
1255
|
+
"input": [
|
|
1256
|
+
"text",
|
|
1257
|
+
"image"
|
|
1258
|
+
],
|
|
1259
|
+
"output": [
|
|
1260
|
+
"text"
|
|
1261
|
+
]
|
|
1262
|
+
},
|
|
1263
|
+
"open_weights": true,
|
|
1264
|
+
"limit": {
|
|
1265
|
+
"context": 262144,
|
|
1266
|
+
"output": 32768
|
|
1267
|
+
},
|
|
1268
|
+
"cost": {
|
|
1269
|
+
"input": 0.2,
|
|
1270
|
+
"output": 0.88,
|
|
1271
|
+
"cache_read": 0.11
|
|
1272
|
+
}
|
|
1273
|
+
},
|
|
1198
1274
|
"Qwen/Qwen3.6-27B": {
|
|
1199
1275
|
"id": "Qwen/Qwen3.6-27B",
|
|
1200
1276
|
"name": "Qwen3.6 27B",
|
|
@@ -1665,6 +1741,92 @@
|
|
|
1665
1741
|
"output": 0.95
|
|
1666
1742
|
}
|
|
1667
1743
|
},
|
|
1744
|
+
"Qwen/Qwen3.8-2.4T-A95B": {
|
|
1745
|
+
"id": "Qwen/Qwen3.8-2.4T-A95B",
|
|
1746
|
+
"name": "Qwen3.8 2.4T A95B",
|
|
1747
|
+
"description": "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows",
|
|
1748
|
+
"family": "qwen",
|
|
1749
|
+
"attachment": false,
|
|
1750
|
+
"reasoning": true,
|
|
1751
|
+
"reasoning_options": [
|
|
1752
|
+
{
|
|
1753
|
+
"type": "effort",
|
|
1754
|
+
"values": [
|
|
1755
|
+
"low",
|
|
1756
|
+
"medium",
|
|
1757
|
+
"xhigh"
|
|
1758
|
+
]
|
|
1759
|
+
}
|
|
1760
|
+
],
|
|
1761
|
+
"tool_call": true,
|
|
1762
|
+
"structured_output": true,
|
|
1763
|
+
"temperature": true,
|
|
1764
|
+
"release_date": "2026-08-12",
|
|
1765
|
+
"last_updated": "2026-08-12",
|
|
1766
|
+
"modalities": {
|
|
1767
|
+
"input": [
|
|
1768
|
+
"text"
|
|
1769
|
+
],
|
|
1770
|
+
"output": [
|
|
1771
|
+
"text"
|
|
1772
|
+
]
|
|
1773
|
+
},
|
|
1774
|
+
"open_weights": true,
|
|
1775
|
+
"limit": {
|
|
1776
|
+
"context": 262144,
|
|
1777
|
+
"output": 131072
|
|
1778
|
+
},
|
|
1779
|
+
"cost": {
|
|
1780
|
+
"input": 2,
|
|
1781
|
+
"output": 6,
|
|
1782
|
+
"cache_read": 0.2
|
|
1783
|
+
}
|
|
1784
|
+
},
|
|
1785
|
+
"Qwen/Qwen3.8-27B": {
|
|
1786
|
+
"id": "Qwen/Qwen3.8-27B",
|
|
1787
|
+
"name": "Qwen3.8 27B",
|
|
1788
|
+
"description": "Dense 27B vision-language model for coding, agent tasks, and image and video understanding",
|
|
1789
|
+
"family": "qwen",
|
|
1790
|
+
"attachment": true,
|
|
1791
|
+
"reasoning": true,
|
|
1792
|
+
"reasoning_options": [
|
|
1793
|
+
{
|
|
1794
|
+
"type": "toggle"
|
|
1795
|
+
},
|
|
1796
|
+
{
|
|
1797
|
+
"type": "effort",
|
|
1798
|
+
"values": [
|
|
1799
|
+
"low",
|
|
1800
|
+
"medium",
|
|
1801
|
+
"xhigh"
|
|
1802
|
+
]
|
|
1803
|
+
}
|
|
1804
|
+
],
|
|
1805
|
+
"tool_call": true,
|
|
1806
|
+
"structured_output": true,
|
|
1807
|
+
"temperature": true,
|
|
1808
|
+
"release_date": "2026-08-14",
|
|
1809
|
+
"last_updated": "2026-08-14",
|
|
1810
|
+
"modalities": {
|
|
1811
|
+
"input": [
|
|
1812
|
+
"text",
|
|
1813
|
+
"image"
|
|
1814
|
+
],
|
|
1815
|
+
"output": [
|
|
1816
|
+
"text"
|
|
1817
|
+
]
|
|
1818
|
+
},
|
|
1819
|
+
"open_weights": true,
|
|
1820
|
+
"limit": {
|
|
1821
|
+
"context": 262144,
|
|
1822
|
+
"output": 32768
|
|
1823
|
+
},
|
|
1824
|
+
"cost": {
|
|
1825
|
+
"input": 0.4,
|
|
1826
|
+
"output": 3,
|
|
1827
|
+
"cache_read": 0.04
|
|
1828
|
+
}
|
|
1829
|
+
},
|
|
1668
1830
|
"Qwen/Qwen3-Next-80B-A3B-Instruct": {
|
|
1669
1831
|
"id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
|
|
1670
1832
|
"name": "Qwen3-Next 80B-A3B Instruct",
|
data/data/deepseek.json
CHANGED
|
@@ -170,6 +170,56 @@
|
|
|
170
170
|
"reasoning": 0.28,
|
|
171
171
|
"cache_read": 0.0028
|
|
172
172
|
}
|
|
173
|
+
},
|
|
174
|
+
"deepseek-v4-flash-vision-exp": {
|
|
175
|
+
"id": "deepseek-v4-flash-vision-exp",
|
|
176
|
+
"name": "DeepSeek V4 Flash Vision Exp",
|
|
177
|
+
"description": "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work",
|
|
178
|
+
"family": "deepseek-flash",
|
|
179
|
+
"attachment": true,
|
|
180
|
+
"reasoning": true,
|
|
181
|
+
"reasoning_options": [
|
|
182
|
+
{
|
|
183
|
+
"type": "toggle"
|
|
184
|
+
},
|
|
185
|
+
{
|
|
186
|
+
"type": "effort",
|
|
187
|
+
"values": [
|
|
188
|
+
"low",
|
|
189
|
+
"high",
|
|
190
|
+
"max"
|
|
191
|
+
]
|
|
192
|
+
}
|
|
193
|
+
],
|
|
194
|
+
"tool_call": true,
|
|
195
|
+
"interleaved": {
|
|
196
|
+
"field": "reasoning_content"
|
|
197
|
+
},
|
|
198
|
+
"structured_output": true,
|
|
199
|
+
"temperature": true,
|
|
200
|
+
"release_date": "2026-08-21",
|
|
201
|
+
"last_updated": "2026-08-21",
|
|
202
|
+
"modalities": {
|
|
203
|
+
"input": [
|
|
204
|
+
"text",
|
|
205
|
+
"image"
|
|
206
|
+
],
|
|
207
|
+
"output": [
|
|
208
|
+
"text"
|
|
209
|
+
]
|
|
210
|
+
},
|
|
211
|
+
"open_weights": false,
|
|
212
|
+
"limit": {
|
|
213
|
+
"context": 1000000,
|
|
214
|
+
"output": 384000
|
|
215
|
+
},
|
|
216
|
+
"status": "beta",
|
|
217
|
+
"cost": {
|
|
218
|
+
"input": 0.14,
|
|
219
|
+
"output": 0.28,
|
|
220
|
+
"reasoning": 0.28,
|
|
221
|
+
"cache_read": 0.0028
|
|
222
|
+
}
|
|
173
223
|
}
|
|
174
224
|
}
|
|
175
225
|
}
|
data/data/google.json
CHANGED
|
@@ -149,7 +149,7 @@
|
|
|
149
149
|
"gemini-flash-lite-latest": {
|
|
150
150
|
"id": "gemini-flash-lite-latest",
|
|
151
151
|
"name": "Gemini Flash-Lite Latest",
|
|
152
|
-
"description": "
|
|
152
|
+
"description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
|
|
153
153
|
"family": "gemini-flash-lite",
|
|
154
154
|
"attachment": true,
|
|
155
155
|
"reasoning": true,
|
|
@@ -167,9 +167,9 @@
|
|
|
167
167
|
"tool_call": true,
|
|
168
168
|
"structured_output": true,
|
|
169
169
|
"temperature": true,
|
|
170
|
-
"knowledge": "
|
|
171
|
-
"release_date": "2026-
|
|
172
|
-
"last_updated": "2026-
|
|
170
|
+
"knowledge": "2026-03",
|
|
171
|
+
"release_date": "2026-07-21",
|
|
172
|
+
"last_updated": "2026-07-21",
|
|
173
173
|
"modalities": {
|
|
174
174
|
"input": [
|
|
175
175
|
"text",
|
|
@@ -188,10 +188,9 @@
|
|
|
188
188
|
"output": 65536
|
|
189
189
|
},
|
|
190
190
|
"cost": {
|
|
191
|
-
"input": 0.
|
|
192
|
-
"output":
|
|
193
|
-
"cache_read": 0.
|
|
194
|
-
"input_audio": 0.5
|
|
191
|
+
"input": 0.3,
|
|
192
|
+
"output": 2.5,
|
|
193
|
+
"cache_read": 0.03
|
|
195
194
|
}
|
|
196
195
|
},
|
|
197
196
|
"gemini-3.5-flash-lite": {
|
|
@@ -751,7 +750,7 @@
|
|
|
751
750
|
"gemini-flash-latest": {
|
|
752
751
|
"id": "gemini-flash-latest",
|
|
753
752
|
"name": "Gemini Flash Latest",
|
|
754
|
-
"description": "
|
|
753
|
+
"description": "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning",
|
|
755
754
|
"family": "gemini-flash",
|
|
756
755
|
"attachment": true,
|
|
757
756
|
"reasoning": true,
|
|
@@ -759,7 +758,6 @@
|
|
|
759
758
|
{
|
|
760
759
|
"type": "effort",
|
|
761
760
|
"values": [
|
|
762
|
-
"minimal",
|
|
763
761
|
"low",
|
|
764
762
|
"medium",
|
|
765
763
|
"high"
|
|
@@ -769,9 +767,9 @@
|
|
|
769
767
|
"tool_call": true,
|
|
770
768
|
"structured_output": true,
|
|
771
769
|
"temperature": true,
|
|
772
|
-
"knowledge": "
|
|
773
|
-
"release_date": "2026-
|
|
774
|
-
"last_updated": "2026-
|
|
770
|
+
"knowledge": "2026-03",
|
|
771
|
+
"release_date": "2026-08-13",
|
|
772
|
+
"last_updated": "2026-08-13",
|
|
775
773
|
"modalities": {
|
|
776
774
|
"input": [
|
|
777
775
|
"text",
|
|
@@ -790,10 +788,10 @@
|
|
|
790
788
|
"output": 65536
|
|
791
789
|
},
|
|
792
790
|
"cost": {
|
|
793
|
-
"input":
|
|
794
|
-
"output":
|
|
795
|
-
"cache_read": 0.
|
|
796
|
-
"input_audio":
|
|
791
|
+
"input": 0.75,
|
|
792
|
+
"output": 3.75,
|
|
793
|
+
"cache_read": 0.075,
|
|
794
|
+
"input_audio": 0.75
|
|
797
795
|
}
|
|
798
796
|
},
|
|
799
797
|
"gemini-3.5-flash": {
|
|
@@ -1164,10 +1162,10 @@
|
|
|
1164
1162
|
"output": 65536
|
|
1165
1163
|
},
|
|
1166
1164
|
"cost": {
|
|
1167
|
-
"input":
|
|
1168
|
-
"output":
|
|
1169
|
-
"cache_read": 0.
|
|
1170
|
-
"input_audio":
|
|
1165
|
+
"input": 0.75,
|
|
1166
|
+
"output": 3.75,
|
|
1167
|
+
"cache_read": 0.075,
|
|
1168
|
+
"input_audio": 0.75
|
|
1171
1169
|
}
|
|
1172
1170
|
},
|
|
1173
1171
|
"gemini-3.1-flash-image": {
|