llm.rb 15.0.2 → 15.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +161 -1
  3. data/README.md +134 -85
  4. data/bin/llm.rb +38 -3
  5. data/data/bedrock.json +250 -0
  6. data/data/deepinfra.json +162 -0
  7. data/data/deepseek.json +50 -0
  8. data/data/google.json +19 -21
  9. data/data/openrouter.json +14280 -0
  10. data/data/xai.json +28 -16
  11. data/docs/deepdive/advanced/context.md +8 -6
  12. data/docs/deepdive/fundamentals/agents.md +11 -10
  13. data/docs/deepdive/fundamentals/stream.md +4 -4
  14. data/lib/llm/active_record.rb +1 -1
  15. data/lib/llm/agent.rb +6 -0
  16. data/lib/llm/context.rb +28 -13
  17. data/lib/llm/cost.rb +13 -0
  18. data/lib/llm/function/fork/task.rb +14 -10
  19. data/lib/llm/provider.rb +29 -8
  20. data/lib/llm/providers/anthropic.rb +1 -1
  21. data/lib/llm/providers/bedrock/models.rb +2 -2
  22. data/lib/llm/providers/bedrock.rb +1 -1
  23. data/lib/llm/providers/google.rb +1 -1
  24. data/lib/llm/providers/ollama.rb +1 -1
  25. data/lib/llm/providers/openai/responses.rb +2 -1
  26. data/lib/llm/providers/openai.rb +1 -1
  27. data/lib/llm/providers/openrouter.rb +87 -0
  28. data/lib/llm/repl/buffer.rb +20 -5
  29. data/lib/llm/repl/input.rb +5 -4
  30. data/lib/llm/repl/markdown/table.rb +5 -2
  31. data/lib/llm/repl/node.rb +26 -1
  32. data/lib/llm/repl/stream.rb +30 -3
  33. data/lib/llm/repl/window.rb +7 -7
  34. data/lib/llm/repl.rb +1 -0
  35. data/lib/llm/skill.rb +7 -1
  36. data/lib/llm/stream.rb +8 -3
  37. data/lib/llm/tools/git.rb +2 -2
  38. data/lib/llm/tools/mkdir.rb +2 -2
  39. data/lib/llm/tools/rg.rb +2 -2
  40. data/lib/llm/tools/ruby.rb +2 -2
  41. data/lib/llm/tools/shell.rb +2 -2
  42. data/lib/llm/tools/utils.rb +1 -1
  43. data/lib/llm/transport/curb.rb +5 -3
  44. data/lib/llm/transport/http.rb +5 -2
  45. data/lib/llm/transport/persistent_http.rb +6 -4
  46. data/lib/llm/transport/utils.rb +8 -6
  47. data/lib/llm/version.rb +1 -1
  48. data/lib/llm.rb +16 -1
  49. data/llm.gemspec +2 -1
  50. metadata +19 -3
data/bin/llm.rb CHANGED
@@ -49,16 +49,20 @@ def help
49
49
  warn ""
50
50
  warn "Options:"
51
51
  warn " -p PROVIDER Choose a provider"
52
+ warn " -m MODEL Choose a model"
52
53
  warn " -c STRATEGY Concurrency strategy for tool calls (eg thread, async, fork)"
53
54
  warn " -n TRANSPORT HTTP transports - net-http (default), net-http-persistent, and curb"
55
+ warn " -x TIMEOUT The default read timeout (in seconds)"
54
56
  warn " -t Temporary session that doesn't persist to disk"
55
57
  warn " -h Show this help"
56
58
  warn ""
57
59
  warn "Examples:"
58
60
  warn " #{prog} # auto-detect provider from $PROVIDER_API_KEY"
59
61
  warn " #{prog} -p openai # use OpenAI"
62
+ warn " #{prog} -m gpt-5.6 # use a model other than the provider default"
60
63
  warn " #{prog} -n curb # use libcurl"
61
64
  warn " #{prog} -c thread # run tool calls on a separate thread"
65
+ warn " #{prog} -x 900 # read timeout of 15mins"
62
66
  warn " #{prog} -h # this help"
63
67
  warn ""
64
68
  end
@@ -94,6 +98,11 @@ def fatal(ex)
94
98
  wrapped detail, " "
95
99
  end
96
100
  warn ""
101
+ warn " Backtrace:"
102
+ ex.backtrace.drop(1).first(3).each do |line|
103
+ wrapped line, " "
104
+ end
105
+ warn ""
97
106
  warn " This is an unexpected error. If it keeps happening,"
98
107
  warn " consider opening an issue at"
99
108
  warn " https://github.com/r-uby-dev/llm/issues"
@@ -153,6 +162,22 @@ def main(argv)
153
162
  help
154
163
  exit 1
155
164
  end
165
+ when '-m'
166
+ model = argv.shift
167
+ if model.nil?
168
+ warn "llm.rb: -m switch requires an argument"
169
+ help
170
+ exit 1
171
+ end
172
+ when '-x'
173
+ timeout = argv.shift
174
+ if timeout.nil?
175
+ warn "llm.rb: -x switch requires an argument"
176
+ help
177
+ exit 1
178
+ else
179
+ timeout = Integer(timeout)
180
+ end
156
181
  else
157
182
  warn "llm.rb: unknown option #{option}"
158
183
  help
@@ -164,9 +189,10 @@ def main(argv)
164
189
  # No provider has been given.
165
190
  # Try to infer one.
166
191
  transport ||= :net_http
192
+ options = timeout ? {timeout:, transport:} : {transport:}
167
193
  if provider.nil?
168
194
  llm = providers.filter_map do
169
- LLM.method(_1).call(transport:)
195
+ LLM.method(_1).call(**options)
170
196
  rescue ArgumentError
171
197
  end.first
172
198
  if llm.nil?
@@ -175,7 +201,7 @@ def main(argv)
175
201
  end
176
202
  else
177
203
  begin
178
- llm = LLM.method(provider).call(transport:)
204
+ llm = LLM.method(provider).call(**options)
179
205
  rescue ArgumentError
180
206
  warn "llm.rb: set credentials for #{provider}"
181
207
  exit 1
@@ -209,11 +235,20 @@ def main(argv)
209
235
  File.binwrite file, JSON.pretty_generate(data)
210
236
  end
211
237
 
238
+ if model
239
+ if not llm.registry.keys.include?(model)
240
+ warn "llm.rb: #{model} is not a valid #{llm.name} model"
241
+ exit 1
242
+ end
243
+ else
244
+ model = llm.default_model
245
+ end
246
+
212
247
  ##
213
248
  # Let's go!
214
249
  concurrency ||= :sequential
215
250
  path = temp ? nil : data[Dir.getwd]
216
- agent = LLM::Agent.new(llm, path:, concurrency:, tools: LLM::Tool.subclasses)
251
+ agent = LLM::Agent.new(llm, model:, path:, concurrency:, tools: LLM::Tool.subclasses)
217
252
  agent.repl
218
253
  rescue Interrupt
219
254
  warn "llm.rb: Bye!"
data/data/bedrock.json CHANGED
@@ -435,6 +435,73 @@
435
435
  "cache_write": 4.125
436
436
  }
437
437
  },
438
+ "global.openai.gpt-5.6-sol": {
439
+ "id": "global.openai.gpt-5.6-sol",
440
+ "name": "GPT-5.6 Sol (Global)",
441
+ "description": "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows",
442
+ "family": "gpt-sol",
443
+ "attachment": true,
444
+ "reasoning": true,
445
+ "reasoning_options": [
446
+ {
447
+ "type": "effort",
448
+ "values": [
449
+ "none",
450
+ "low",
451
+ "medium",
452
+ "high",
453
+ "xhigh",
454
+ "max"
455
+ ]
456
+ }
457
+ ],
458
+ "tool_call": true,
459
+ "structured_output": true,
460
+ "temperature": false,
461
+ "knowledge": "2026-02-16",
462
+ "release_date": "2026-07-09",
463
+ "last_updated": "2026-07-09",
464
+ "modalities": {
465
+ "input": [
466
+ "text",
467
+ "image",
468
+ "pdf"
469
+ ],
470
+ "output": [
471
+ "text"
472
+ ]
473
+ },
474
+ "open_weights": false,
475
+ "limit": {
476
+ "context": 1050000,
477
+ "input": 922000,
478
+ "output": 128000
479
+ },
480
+ "cost": {
481
+ "input": 5.5,
482
+ "output": 33,
483
+ "cache_read": 0.55,
484
+ "cache_write": 6.875,
485
+ "tiers": [
486
+ {
487
+ "input": 11,
488
+ "output": 49.5,
489
+ "cache_read": 1.1,
490
+ "cache_write": 13.75,
491
+ "tier": {
492
+ "type": "context",
493
+ "size": 272000
494
+ }
495
+ }
496
+ ],
497
+ "context_over_200k": {
498
+ "input": 11,
499
+ "output": 49.5,
500
+ "cache_read": 1.1,
501
+ "cache_write": 13.75
502
+ }
503
+ }
504
+ },
438
505
  "jp.anthropic.claude-sonnet-4-6": {
439
506
  "id": "jp.anthropic.claude-sonnet-4-6",
440
507
  "name": "Claude Sonnet 4.6 (JP)",
@@ -1907,6 +1974,55 @@
1907
1974
  "cache_write": 6.25
1908
1975
  }
1909
1976
  },
1977
+ "xai.grok-4.6": {
1978
+ "id": "xai.grok-4.6",
1979
+ "name": "Grok 4.6",
1980
+ "description": "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects",
1981
+ "family": "grok",
1982
+ "attachment": true,
1983
+ "reasoning": true,
1984
+ "reasoning_options": [
1985
+ {
1986
+ "type": "effort",
1987
+ "values": [
1988
+ "low",
1989
+ "medium",
1990
+ "high",
1991
+ "xhigh"
1992
+ ]
1993
+ }
1994
+ ],
1995
+ "tool_call": true,
1996
+ "structured_output": true,
1997
+ "temperature": true,
1998
+ "knowledge": "2026-02-01",
1999
+ "release_date": "2026-08-12",
2000
+ "last_updated": "2026-08-18",
2001
+ "modalities": {
2002
+ "input": [
2003
+ "text",
2004
+ "image"
2005
+ ],
2006
+ "output": [
2007
+ "text"
2008
+ ]
2009
+ },
2010
+ "open_weights": false,
2011
+ "limit": {
2012
+ "context": 500000,
2013
+ "output": 500000
2014
+ },
2015
+ "provider": {
2016
+ "npm": "@ai-sdk/amazon-bedrock/mantle",
2017
+ "api": "https://bedrock-mantle.${AWS_REGION}.api.aws/openai/v1",
2018
+ "shape": "responses"
2019
+ },
2020
+ "cost": {
2021
+ "input": 2.2,
2022
+ "output": 6.6,
2023
+ "cache_read": 0.55
2024
+ }
2025
+ },
1910
2026
  "anthropic.claude-sonnet-4-6": {
1911
2027
  "id": "anthropic.claude-sonnet-4-6",
1912
2028
  "name": "Claude Sonnet 4.6",
@@ -3167,6 +3283,73 @@
3167
3283
  "cache_read": 0.2
3168
3284
  }
3169
3285
  },
3286
+ "global.openai.gpt-5.6-luna": {
3287
+ "id": "global.openai.gpt-5.6-luna",
3288
+ "name": "GPT-5.6 Luna (Global)",
3289
+ "description": "Cost-efficient GPT-5.6 model for fast, high-volume workloads",
3290
+ "family": "gpt-luna",
3291
+ "attachment": true,
3292
+ "reasoning": true,
3293
+ "reasoning_options": [
3294
+ {
3295
+ "type": "effort",
3296
+ "values": [
3297
+ "none",
3298
+ "low",
3299
+ "medium",
3300
+ "high",
3301
+ "xhigh",
3302
+ "max"
3303
+ ]
3304
+ }
3305
+ ],
3306
+ "tool_call": true,
3307
+ "structured_output": true,
3308
+ "temperature": false,
3309
+ "knowledge": "2026-02-16",
3310
+ "release_date": "2026-07-09",
3311
+ "last_updated": "2026-07-09",
3312
+ "modalities": {
3313
+ "input": [
3314
+ "text",
3315
+ "image",
3316
+ "pdf"
3317
+ ],
3318
+ "output": [
3319
+ "text"
3320
+ ]
3321
+ },
3322
+ "open_weights": false,
3323
+ "limit": {
3324
+ "context": 1050000,
3325
+ "input": 922000,
3326
+ "output": 128000
3327
+ },
3328
+ "cost": {
3329
+ "input": 0.22,
3330
+ "output": 1.32,
3331
+ "cache_read": 0.022,
3332
+ "cache_write": 0.275,
3333
+ "tiers": [
3334
+ {
3335
+ "input": 0.44,
3336
+ "output": 1.98,
3337
+ "cache_read": 0.044,
3338
+ "cache_write": 0.55,
3339
+ "tier": {
3340
+ "type": "context",
3341
+ "size": 272000
3342
+ }
3343
+ }
3344
+ ],
3345
+ "context_over_200k": {
3346
+ "input": 0.44,
3347
+ "output": 1.98,
3348
+ "cache_read": 0.044,
3349
+ "cache_write": 0.55
3350
+ }
3351
+ }
3352
+ },
3170
3353
  "zai.glm-4.7": {
3171
3354
  "id": "zai.glm-4.7",
3172
3355
  "name": "GLM-4.7",
@@ -3867,6 +4050,73 @@
3867
4050
  "cache_write": 3.75
3868
4051
  }
3869
4052
  },
4053
+ "global.openai.gpt-5.6-terra": {
4054
+ "id": "global.openai.gpt-5.6-terra",
4055
+ "name": "GPT-5.6 Terra (Global)",
4056
+ "description": "Balanced GPT-5.6 model for capable, cost-efficient everyday work",
4057
+ "family": "gpt-terra",
4058
+ "attachment": true,
4059
+ "reasoning": true,
4060
+ "reasoning_options": [
4061
+ {
4062
+ "type": "effort",
4063
+ "values": [
4064
+ "none",
4065
+ "low",
4066
+ "medium",
4067
+ "high",
4068
+ "xhigh",
4069
+ "max"
4070
+ ]
4071
+ }
4072
+ ],
4073
+ "tool_call": true,
4074
+ "structured_output": true,
4075
+ "temperature": false,
4076
+ "knowledge": "2026-02-16",
4077
+ "release_date": "2026-07-09",
4078
+ "last_updated": "2026-07-09",
4079
+ "modalities": {
4080
+ "input": [
4081
+ "text",
4082
+ "image",
4083
+ "pdf"
4084
+ ],
4085
+ "output": [
4086
+ "text"
4087
+ ]
4088
+ },
4089
+ "open_weights": false,
4090
+ "limit": {
4091
+ "context": 1050000,
4092
+ "input": 922000,
4093
+ "output": 128000
4094
+ },
4095
+ "cost": {
4096
+ "input": 2.2,
4097
+ "output": 13.2,
4098
+ "cache_read": 0.22,
4099
+ "cache_write": 2.75,
4100
+ "tiers": [
4101
+ {
4102
+ "input": 4.4,
4103
+ "output": 19.8,
4104
+ "cache_read": 0.44,
4105
+ "cache_write": 5.5,
4106
+ "tier": {
4107
+ "type": "context",
4108
+ "size": 272000
4109
+ }
4110
+ }
4111
+ ],
4112
+ "context_over_200k": {
4113
+ "input": 4.4,
4114
+ "output": 19.8,
4115
+ "cache_read": 0.44,
4116
+ "cache_write": 5.5
4117
+ }
4118
+ }
4119
+ },
3870
4120
  "openai.gpt-oss-20b-1:0": {
3871
4121
  "id": "openai.gpt-oss-20b-1:0",
3872
4122
  "name": "gpt-oss-20b",
data/data/deepinfra.json CHANGED
@@ -1053,6 +1053,49 @@
1053
1053
  "cache_read": 0.135
1054
1054
  }
1055
1055
  },
1056
+ "deepseek-ai/DeepSeek-V4-Pro-0813": {
1057
+ "id": "deepseek-ai/DeepSeek-V4-Pro-0813",
1058
+ "name": "DeepSeek V4 Pro 0813",
1059
+ "description": "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes",
1060
+ "family": "deepseek-thinking",
1061
+ "attachment": false,
1062
+ "reasoning": true,
1063
+ "reasoning_options": [
1064
+ {
1065
+ "type": "toggle"
1066
+ },
1067
+ {
1068
+ "type": "effort",
1069
+ "values": [
1070
+ "high",
1071
+ "max"
1072
+ ]
1073
+ }
1074
+ ],
1075
+ "tool_call": true,
1076
+ "structured_output": true,
1077
+ "temperature": true,
1078
+ "release_date": "2026-08-12",
1079
+ "last_updated": "2026-08-12",
1080
+ "modalities": {
1081
+ "input": [
1082
+ "text"
1083
+ ],
1084
+ "output": [
1085
+ "text"
1086
+ ]
1087
+ },
1088
+ "open_weights": false,
1089
+ "limit": {
1090
+ "context": 1048576,
1091
+ "output": 384000
1092
+ },
1093
+ "cost": {
1094
+ "input": 1.3,
1095
+ "output": 2.6,
1096
+ "cache_read": 0.1
1097
+ }
1098
+ },
1056
1099
  "deepseek-ai/DeepSeek-V3": {
1057
1100
  "id": "deepseek-ai/DeepSeek-V3",
1058
1101
  "name": "DeepSeek-V3",
@@ -1195,6 +1238,39 @@
1195
1238
  "output": 0.55
1196
1239
  }
1197
1240
  },
1241
+ "Qwen/Qwen3-VL-235B-A22B-Instruct": {
1242
+ "id": "Qwen/Qwen3-VL-235B-A22B-Instruct",
1243
+ "name": "Qwen3 VL 235B A22B Instruct",
1244
+ "description": "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks",
1245
+ "family": "qwen",
1246
+ "attachment": true,
1247
+ "reasoning": false,
1248
+ "tool_call": true,
1249
+ "structured_output": true,
1250
+ "temperature": true,
1251
+ "knowledge": "2025-03-31",
1252
+ "release_date": "2025-09-23",
1253
+ "last_updated": "2025-09-23",
1254
+ "modalities": {
1255
+ "input": [
1256
+ "text",
1257
+ "image"
1258
+ ],
1259
+ "output": [
1260
+ "text"
1261
+ ]
1262
+ },
1263
+ "open_weights": true,
1264
+ "limit": {
1265
+ "context": 262144,
1266
+ "output": 32768
1267
+ },
1268
+ "cost": {
1269
+ "input": 0.2,
1270
+ "output": 0.88,
1271
+ "cache_read": 0.11
1272
+ }
1273
+ },
1198
1274
  "Qwen/Qwen3.6-27B": {
1199
1275
  "id": "Qwen/Qwen3.6-27B",
1200
1276
  "name": "Qwen3.6 27B",
@@ -1665,6 +1741,92 @@
1665
1741
  "output": 0.95
1666
1742
  }
1667
1743
  },
1744
+ "Qwen/Qwen3.8-2.4T-A95B": {
1745
+ "id": "Qwen/Qwen3.8-2.4T-A95B",
1746
+ "name": "Qwen3.8 2.4T A95B",
1747
+ "description": "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows",
1748
+ "family": "qwen",
1749
+ "attachment": false,
1750
+ "reasoning": true,
1751
+ "reasoning_options": [
1752
+ {
1753
+ "type": "effort",
1754
+ "values": [
1755
+ "low",
1756
+ "medium",
1757
+ "xhigh"
1758
+ ]
1759
+ }
1760
+ ],
1761
+ "tool_call": true,
1762
+ "structured_output": true,
1763
+ "temperature": true,
1764
+ "release_date": "2026-08-12",
1765
+ "last_updated": "2026-08-12",
1766
+ "modalities": {
1767
+ "input": [
1768
+ "text"
1769
+ ],
1770
+ "output": [
1771
+ "text"
1772
+ ]
1773
+ },
1774
+ "open_weights": true,
1775
+ "limit": {
1776
+ "context": 262144,
1777
+ "output": 131072
1778
+ },
1779
+ "cost": {
1780
+ "input": 2,
1781
+ "output": 6,
1782
+ "cache_read": 0.2
1783
+ }
1784
+ },
1785
+ "Qwen/Qwen3.8-27B": {
1786
+ "id": "Qwen/Qwen3.8-27B",
1787
+ "name": "Qwen3.8 27B",
1788
+ "description": "Dense 27B vision-language model for coding, agent tasks, and image and video understanding",
1789
+ "family": "qwen",
1790
+ "attachment": true,
1791
+ "reasoning": true,
1792
+ "reasoning_options": [
1793
+ {
1794
+ "type": "toggle"
1795
+ },
1796
+ {
1797
+ "type": "effort",
1798
+ "values": [
1799
+ "low",
1800
+ "medium",
1801
+ "xhigh"
1802
+ ]
1803
+ }
1804
+ ],
1805
+ "tool_call": true,
1806
+ "structured_output": true,
1807
+ "temperature": true,
1808
+ "release_date": "2026-08-14",
1809
+ "last_updated": "2026-08-14",
1810
+ "modalities": {
1811
+ "input": [
1812
+ "text",
1813
+ "image"
1814
+ ],
1815
+ "output": [
1816
+ "text"
1817
+ ]
1818
+ },
1819
+ "open_weights": true,
1820
+ "limit": {
1821
+ "context": 262144,
1822
+ "output": 32768
1823
+ },
1824
+ "cost": {
1825
+ "input": 0.4,
1826
+ "output": 3,
1827
+ "cache_read": 0.04
1828
+ }
1829
+ },
1668
1830
  "Qwen/Qwen3-Next-80B-A3B-Instruct": {
1669
1831
  "id": "Qwen/Qwen3-Next-80B-A3B-Instruct",
1670
1832
  "name": "Qwen3-Next 80B-A3B Instruct",
data/data/deepseek.json CHANGED
@@ -170,6 +170,56 @@
170
170
  "reasoning": 0.28,
171
171
  "cache_read": 0.0028
172
172
  }
173
+ },
174
+ "deepseek-v4-flash-vision-exp": {
175
+ "id": "deepseek-v4-flash-vision-exp",
176
+ "name": "DeepSeek V4 Flash Vision Exp",
177
+ "description": "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work",
178
+ "family": "deepseek-flash",
179
+ "attachment": true,
180
+ "reasoning": true,
181
+ "reasoning_options": [
182
+ {
183
+ "type": "toggle"
184
+ },
185
+ {
186
+ "type": "effort",
187
+ "values": [
188
+ "low",
189
+ "high",
190
+ "max"
191
+ ]
192
+ }
193
+ ],
194
+ "tool_call": true,
195
+ "interleaved": {
196
+ "field": "reasoning_content"
197
+ },
198
+ "structured_output": true,
199
+ "temperature": true,
200
+ "release_date": "2026-08-21",
201
+ "last_updated": "2026-08-21",
202
+ "modalities": {
203
+ "input": [
204
+ "text",
205
+ "image"
206
+ ],
207
+ "output": [
208
+ "text"
209
+ ]
210
+ },
211
+ "open_weights": false,
212
+ "limit": {
213
+ "context": 1000000,
214
+ "output": 384000
215
+ },
216
+ "status": "beta",
217
+ "cost": {
218
+ "input": 0.14,
219
+ "output": 0.28,
220
+ "reasoning": 0.28,
221
+ "cache_read": 0.0028
222
+ }
173
223
  }
174
224
  }
175
225
  }
data/data/google.json CHANGED
@@ -149,7 +149,7 @@
149
149
  "gemini-flash-lite-latest": {
150
150
  "id": "gemini-flash-lite-latest",
151
151
  "name": "Gemini Flash-Lite Latest",
152
- "description": "Low-latency Gemini model for high-volume multimodal and agent workloads",
152
+ "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
153
153
  "family": "gemini-flash-lite",
154
154
  "attachment": true,
155
155
  "reasoning": true,
@@ -167,9 +167,9 @@
167
167
  "tool_call": true,
168
168
  "structured_output": true,
169
169
  "temperature": true,
170
- "knowledge": "2025-01",
171
- "release_date": "2026-05-07",
172
- "last_updated": "2026-05-07",
170
+ "knowledge": "2026-03",
171
+ "release_date": "2026-07-21",
172
+ "last_updated": "2026-07-21",
173
173
  "modalities": {
174
174
  "input": [
175
175
  "text",
@@ -188,10 +188,9 @@
188
188
  "output": 65536
189
189
  },
190
190
  "cost": {
191
- "input": 0.25,
192
- "output": 1.5,
193
- "cache_read": 0.025,
194
- "input_audio": 0.5
191
+ "input": 0.3,
192
+ "output": 2.5,
193
+ "cache_read": 0.03
195
194
  }
196
195
  },
197
196
  "gemini-3.5-flash-lite": {
@@ -751,7 +750,7 @@
751
750
  "gemini-flash-latest": {
752
751
  "id": "gemini-flash-latest",
753
752
  "name": "Gemini Flash Latest",
754
- "description": "Fast Gemini model balancing multimodal reasoning, tool use, and cost",
753
+ "description": "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning",
755
754
  "family": "gemini-flash",
756
755
  "attachment": true,
757
756
  "reasoning": true,
@@ -759,7 +758,6 @@
759
758
  {
760
759
  "type": "effort",
761
760
  "values": [
762
- "minimal",
763
761
  "low",
764
762
  "medium",
765
763
  "high"
@@ -769,9 +767,9 @@
769
767
  "tool_call": true,
770
768
  "structured_output": true,
771
769
  "temperature": true,
772
- "knowledge": "2025-01",
773
- "release_date": "2026-05-19",
774
- "last_updated": "2026-05-19",
770
+ "knowledge": "2026-03",
771
+ "release_date": "2026-08-13",
772
+ "last_updated": "2026-08-13",
775
773
  "modalities": {
776
774
  "input": [
777
775
  "text",
@@ -790,10 +788,10 @@
790
788
  "output": 65536
791
789
  },
792
790
  "cost": {
793
- "input": 1.5,
794
- "output": 9,
795
- "cache_read": 0.15,
796
- "input_audio": 1.5
791
+ "input": 0.75,
792
+ "output": 3.75,
793
+ "cache_read": 0.075,
794
+ "input_audio": 0.75
797
795
  }
798
796
  },
799
797
  "gemini-3.5-flash": {
@@ -1164,10 +1162,10 @@
1164
1162
  "output": 65536
1165
1163
  },
1166
1164
  "cost": {
1167
- "input": 1.5,
1168
- "output": 7.5,
1169
- "cache_read": 0.15,
1170
- "input_audio": 1.5
1165
+ "input": 0.75,
1166
+ "output": 3.75,
1167
+ "cache_read": 0.075,
1168
+ "input_audio": 0.75
1171
1169
  }
1172
1170
  },
1173
1171
  "gemini-3.1-flash-image": {