llm.rb 15.4.0 → 15.4.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: de1f4eeee4762548e75eac2b5d4112de49cf446944b30c545b76e62750aad6c3
4
- data.tar.gz: 4abc1d2d707717fde266b48bfa626f06d553514761653db8c3cd190cfcffb864
3
+ metadata.gz: 4647447cce1ccfc7e4f062cfa46b6dd0363ffd1a86537200a7877daff983cda3
4
+ data.tar.gz: 31502899d4a6443b893a48baf18c75d02311920166a6d791d671fa33600e93b9
5
5
  SHA512:
6
- metadata.gz: c28cadd2a412e2623b472d3be7814aa51a462076782fc847bc73cdc57a601a630d909bb9b40c29d31dd7107b487bf169bdeb245e9803581df0d90d642694bc16
7
- data.tar.gz: 2fd3c1905d2db4212958e98e537db76d37f0f30f5cee4dda3cbf351c7f0fffe1ce6eaa5b38553e80b9441b029e2edf78c4e9ea75c200b731638dbbe49bd947df
6
+ metadata.gz: ee244a4b0d7d317632e1ee2f3dc44b82926086a71f2ded6ba7e3189eff9636dba2eed5f97d4ed6c83140584d3807b4872303a20b369a123be2336c32671758a1
7
+ data.tar.gz: 7ed45ea9770afabb6522b6939c9d333c28a5892e40e32de271d03a8c5fb3f9fa15b5c2ace19e74ac80342bc65e04a7b5edb48b6799a612f5754b8ccccd33bdaf
data/CHANGELOG.md CHANGED
@@ -17,6 +17,33 @@
17
17
 
18
18
  *No unreleased changes yet. Check back after the next release.*
19
19
 
20
+ ## v15.4.1
21
+
22
+ Changes since `v15.4.0`.
23
+
24
+ This release removes the gemspec's post-install message, so installing the gem
25
+ no longer prints the r.uby.dev notice. It also refreshes the model registry
26
+ with current model listings, limits, and pricing.
27
+
28
+ ### Core
29
+
30
+ * **remove the gemspec post install message** <br>
31
+ The gemspec no longer sets `post_install_message`, so installing the
32
+ gem no longer prints the r.uby.dev website notice.
33
+
34
+ ### Registry
35
+
36
+ * **refresh model metadata** <br>
37
+ Update `data/` with current model listings, limits, and pricing for the
38
+ Alibaba, Bedrock, DeepSeek, Google, Moonshot, and OpenRouter registries.
39
+ Bedrock adds the Kimi K3 and GPT-6 Sol and GPT-6 Luna families in both
40
+ the global and US regions and raises the context limit to 1M tokens for
41
+ two models, while Moonshot raises the Kimi K3 output limit to 1M tokens.
42
+ OpenRouter adds `qwen/qwen3.8-max-prime`, `z-ai/glm-5.3-prime`,
43
+ `upstage/solar-mini4`, the Aion 3.5 models, and `stealth/space-bunny-alpha`,
44
+ drops `mistralai/devstral-2512` and a free Ling 3.0 Flash VL entry, and
45
+ reprices several DeepSeek and Mistral models.
46
+
20
47
  ## v15.4.0
21
48
 
22
49
  Changes since `v15.3.0`.
data/README.md CHANGED
@@ -19,10 +19,13 @@ on CRuby. It has zero runtime dependencies by default, supports
19
19
  concurrent and parallel tool execution and has a single coherent API
20
20
  that spans 14+ providers.
21
21
 
22
- The most effective way to learn about llm.rb is to ask [the r.uby.dev chatbot](https://r.uby.dev)
23
- a question. It is connected to the llm.rb GitHub repository, backed by
24
- ActiveRecord and uses the builtin MCP feature to connect to GitHub. The chatbot
25
- is an llm.rb agent that is deployed with [roda-llm](https://github.com/r-uby-dev/roda-llm#readme).
22
+ It is possible to see llm.rb in action on the
23
+ [the r.uby.dev website](https://r.uby.dev) where
24
+ I am working on building an agentic platform that
25
+ users can use to manage multiple agents that are
26
+ specialized in different areas, and have access to
27
+ different services (eg GitHub, etc). Check it out if
28
+ curious. Still in early development.
26
29
 
27
30
  ## Install
28
31
 
@@ -362,9 +365,6 @@ for both Rack-based / Rails-based applications. On databases
362
365
  where it is supported, such as PostgreSQL, the column can be optimized by using
363
366
  the `jsonb` type.
364
367
 
365
- The following example is based on the agent used to power the
366
- [r.uby.dev chatbot](https://r.uby.dev).
367
-
368
368
  ```ruby
369
369
  require "active_record"
370
370
  require "llm"
@@ -1030,10 +1030,8 @@ IO.copy_stream res.images[0], "rocket-with-dog.svg"
1030
1030
  <summary>Where can I see llm.rb in action?</summary>
1031
1031
  <br>
1032
1032
 
1033
- The [r.uby.dev](https://r.uby.dev) website deploys
1034
- multiple llm.rb agents that guests can interact with
1035
- and there is even an agent that is connected to this
1036
- GitHub repository.
1033
+ The [r.uby.dev](https://r.uby.dev) website.
1034
+
1037
1035
  </details>
1038
1036
  <details>
1039
1037
  <summary>What about local LLM support?</summary>
data/data/alibaba.json CHANGED
@@ -38,7 +38,7 @@
38
38
  "open_weights": false,
39
39
  "limit": {
40
40
  "context": 1000000,
41
- "output": 65536
41
+ "output": 131072
42
42
  },
43
43
  "cost": {
44
44
  "input": 2.5,
@@ -1848,7 +1848,7 @@
1848
1848
  "open_weights": false,
1849
1849
  "limit": {
1850
1850
  "context": 1000000,
1851
- "output": 65536
1851
+ "output": 131072
1852
1852
  },
1853
1853
  "cost": {
1854
1854
  "input": 0.5,
data/data/bedrock.json CHANGED
@@ -566,7 +566,7 @@
566
566
  },
567
567
  "open_weights": false,
568
568
  "limit": {
569
- "context": 272000,
569
+ "context": 1000000,
570
570
  "output": 128000
571
571
  },
572
572
  "provider": {
@@ -886,7 +886,7 @@
886
886
  },
887
887
  "open_weights": false,
888
888
  "limit": {
889
- "context": 272000,
889
+ "context": 1000000,
890
890
  "output": 128000
891
891
  },
892
892
  "provider": {
@@ -1587,6 +1587,41 @@
1587
1587
  "cache_write": 6.875
1588
1588
  }
1589
1589
  },
1590
+ "global.moonshotai.kimi-k3": {
1591
+ "id": "global.moonshotai.kimi-k3",
1592
+ "name": "Kimi K3 (Global)",
1593
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
1594
+ "family": "kimi-k3",
1595
+ "attachment": true,
1596
+ "reasoning": true,
1597
+ "reasoning_options": [],
1598
+ "tool_call": true,
1599
+ "interleaved": true,
1600
+ "structured_output": true,
1601
+ "temperature": false,
1602
+ "release_date": "2026-07-16",
1603
+ "last_updated": "2026-07-16",
1604
+ "modalities": {
1605
+ "input": [
1606
+ "text",
1607
+ "image"
1608
+ ],
1609
+ "output": [
1610
+ "text"
1611
+ ]
1612
+ },
1613
+ "open_weights": true,
1614
+ "limit": {
1615
+ "context": 1048576,
1616
+ "output": 131072
1617
+ },
1618
+ "cost": {
1619
+ "input": 3,
1620
+ "output": 15,
1621
+ "cache_read": 0.3,
1622
+ "cache_write": 3.75
1623
+ }
1624
+ },
1590
1625
  "apac.amazon.nova-pro-v1:0": {
1591
1626
  "id": "apac.amazon.nova-pro-v1:0",
1592
1627
  "name": "Nova Pro (APAC)",
@@ -1675,6 +1710,41 @@
1675
1710
  "cache_write": 4.125
1676
1711
  }
1677
1712
  },
1713
+ "us.moonshotai.kimi-k3": {
1714
+ "id": "us.moonshotai.kimi-k3",
1715
+ "name": "Kimi K3 (US)",
1716
+ "description": "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work",
1717
+ "family": "kimi-k3",
1718
+ "attachment": true,
1719
+ "reasoning": true,
1720
+ "reasoning_options": [],
1721
+ "tool_call": true,
1722
+ "interleaved": true,
1723
+ "structured_output": true,
1724
+ "temperature": false,
1725
+ "release_date": "2026-07-16",
1726
+ "last_updated": "2026-07-16",
1727
+ "modalities": {
1728
+ "input": [
1729
+ "text",
1730
+ "image"
1731
+ ],
1732
+ "output": [
1733
+ "text"
1734
+ ]
1735
+ },
1736
+ "open_weights": true,
1737
+ "limit": {
1738
+ "context": 1048576,
1739
+ "output": 131072
1740
+ },
1741
+ "cost": {
1742
+ "input": 3.3,
1743
+ "output": 16.5,
1744
+ "cache_read": 0.33,
1745
+ "cache_write": 4.125
1746
+ }
1747
+ },
1678
1748
  "apac.amazon.nova-lite-v1:0": {
1679
1749
  "id": "apac.amazon.nova-lite-v1:0",
1680
1750
  "name": "Nova Lite (APAC)",
@@ -1933,6 +2003,72 @@
1933
2003
  "output": 1.2
1934
2004
  }
1935
2005
  },
2006
+ "us.openai.gpt-6-luna": {
2007
+ "id": "us.openai.gpt-6-luna",
2008
+ "name": "GPT-6 Luna (US)",
2009
+ "description": "OpenAI's most efficient model for focused, high-volume tasks",
2010
+ "family": "gpt-luna",
2011
+ "attachment": true,
2012
+ "reasoning": true,
2013
+ "reasoning_options": [
2014
+ {
2015
+ "type": "effort",
2016
+ "values": [
2017
+ "none",
2018
+ "low",
2019
+ "medium",
2020
+ "high",
2021
+ "xhigh",
2022
+ "max"
2023
+ ]
2024
+ }
2025
+ ],
2026
+ "tool_call": true,
2027
+ "structured_output": true,
2028
+ "temperature": false,
2029
+ "knowledge": "2026-05-18",
2030
+ "release_date": "2026-09-22",
2031
+ "last_updated": "2026-09-22",
2032
+ "modalities": {
2033
+ "input": [
2034
+ "text",
2035
+ "image"
2036
+ ],
2037
+ "output": [
2038
+ "text"
2039
+ ]
2040
+ },
2041
+ "open_weights": false,
2042
+ "limit": {
2043
+ "context": 1050000,
2044
+ "input": 922000,
2045
+ "output": 128000
2046
+ },
2047
+ "cost": {
2048
+ "input": 0.11,
2049
+ "output": 0.55,
2050
+ "cache_read": 0.011,
2051
+ "cache_write": 0.1375,
2052
+ "tiers": [
2053
+ {
2054
+ "input": 0.22,
2055
+ "output": 0.825,
2056
+ "cache_read": 0.022,
2057
+ "cache_write": 0.275,
2058
+ "tier": {
2059
+ "type": "context",
2060
+ "size": 272000
2061
+ }
2062
+ }
2063
+ ],
2064
+ "context_over_200k": {
2065
+ "input": 0.22,
2066
+ "output": 0.825,
2067
+ "cache_read": 0.022,
2068
+ "cache_write": 0.275
2069
+ }
2070
+ }
2071
+ },
1936
2072
  "meta.llama3-3-70b-instruct-v1:0": {
1937
2073
  "id": "meta.llama3-3-70b-instruct-v1:0",
1938
2074
  "name": "Llama 3.3 70B Instruct",
@@ -3258,6 +3394,72 @@
3258
3394
  "cache_write": 0.33
3259
3395
  }
3260
3396
  },
3397
+ "global.openai.gpt-6-sol": {
3398
+ "id": "global.openai.gpt-6-sol",
3399
+ "name": "GPT-6 Sol (Global)",
3400
+ "description": "OpenAI model for complex coding and agentic workflows",
3401
+ "family": "gpt-sol",
3402
+ "attachment": true,
3403
+ "reasoning": true,
3404
+ "reasoning_options": [
3405
+ {
3406
+ "type": "effort",
3407
+ "values": [
3408
+ "none",
3409
+ "low",
3410
+ "medium",
3411
+ "high",
3412
+ "xhigh",
3413
+ "max"
3414
+ ]
3415
+ }
3416
+ ],
3417
+ "tool_call": true,
3418
+ "structured_output": true,
3419
+ "temperature": false,
3420
+ "knowledge": "2026-04-20",
3421
+ "release_date": "2026-09-22",
3422
+ "last_updated": "2026-09-22",
3423
+ "modalities": {
3424
+ "input": [
3425
+ "text",
3426
+ "image"
3427
+ ],
3428
+ "output": [
3429
+ "text"
3430
+ ]
3431
+ },
3432
+ "open_weights": false,
3433
+ "limit": {
3434
+ "context": 1050000,
3435
+ "input": 922000,
3436
+ "output": 128000
3437
+ },
3438
+ "cost": {
3439
+ "input": 2,
3440
+ "output": 10,
3441
+ "cache_read": 0.2,
3442
+ "cache_write": 2.5,
3443
+ "tiers": [
3444
+ {
3445
+ "input": 4,
3446
+ "output": 15,
3447
+ "cache_read": 0.4,
3448
+ "cache_write": 5,
3449
+ "tier": {
3450
+ "type": "context",
3451
+ "size": 272000
3452
+ }
3453
+ }
3454
+ ],
3455
+ "context_over_200k": {
3456
+ "input": 4,
3457
+ "output": 15,
3458
+ "cache_read": 0.4,
3459
+ "cache_write": 5
3460
+ }
3461
+ }
3462
+ },
3261
3463
  "global.xai.grok-4.6": {
3262
3464
  "id": "global.xai.grok-4.6",
3263
3465
  "name": "Grok 4.6 (Global)",
@@ -4572,6 +4774,72 @@
4572
4774
  }
4573
4775
  }
4574
4776
  },
4777
+ "global.openai.gpt-6-luna": {
4778
+ "id": "global.openai.gpt-6-luna",
4779
+ "name": "GPT-6 Luna (Global)",
4780
+ "description": "OpenAI's most efficient model for focused, high-volume tasks",
4781
+ "family": "gpt-luna",
4782
+ "attachment": true,
4783
+ "reasoning": true,
4784
+ "reasoning_options": [
4785
+ {
4786
+ "type": "effort",
4787
+ "values": [
4788
+ "none",
4789
+ "low",
4790
+ "medium",
4791
+ "high",
4792
+ "xhigh",
4793
+ "max"
4794
+ ]
4795
+ }
4796
+ ],
4797
+ "tool_call": true,
4798
+ "structured_output": true,
4799
+ "temperature": false,
4800
+ "knowledge": "2026-05-18",
4801
+ "release_date": "2026-09-22",
4802
+ "last_updated": "2026-09-22",
4803
+ "modalities": {
4804
+ "input": [
4805
+ "text",
4806
+ "image"
4807
+ ],
4808
+ "output": [
4809
+ "text"
4810
+ ]
4811
+ },
4812
+ "open_weights": false,
4813
+ "limit": {
4814
+ "context": 1050000,
4815
+ "input": 922000,
4816
+ "output": 128000
4817
+ },
4818
+ "cost": {
4819
+ "input": 0.1,
4820
+ "output": 0.5,
4821
+ "cache_read": 0.01,
4822
+ "cache_write": 0.125,
4823
+ "tiers": [
4824
+ {
4825
+ "input": 0.2,
4826
+ "output": 0.75,
4827
+ "cache_read": 0.02,
4828
+ "cache_write": 0.25,
4829
+ "tier": {
4830
+ "type": "context",
4831
+ "size": 272000
4832
+ }
4833
+ }
4834
+ ],
4835
+ "context_over_200k": {
4836
+ "input": 0.2,
4837
+ "output": 0.75,
4838
+ "cache_read": 0.02,
4839
+ "cache_write": 0.25
4840
+ }
4841
+ }
4842
+ },
4575
4843
  "us.anthropic.claude-fable-5-1": {
4576
4844
  "id": "us.anthropic.claude-fable-5-1",
4577
4845
  "name": "Claude Fable 5.1 (US)",
@@ -5349,6 +5617,72 @@
5349
5617
  "cache_write": 0.06
5350
5618
  }
5351
5619
  },
5620
+ "us.openai.gpt-6-sol": {
5621
+ "id": "us.openai.gpt-6-sol",
5622
+ "name": "GPT-6 Sol (US)",
5623
+ "description": "OpenAI model for complex coding and agentic workflows",
5624
+ "family": "gpt-sol",
5625
+ "attachment": true,
5626
+ "reasoning": true,
5627
+ "reasoning_options": [
5628
+ {
5629
+ "type": "effort",
5630
+ "values": [
5631
+ "none",
5632
+ "low",
5633
+ "medium",
5634
+ "high",
5635
+ "xhigh",
5636
+ "max"
5637
+ ]
5638
+ }
5639
+ ],
5640
+ "tool_call": true,
5641
+ "structured_output": true,
5642
+ "temperature": false,
5643
+ "knowledge": "2026-04-20",
5644
+ "release_date": "2026-09-22",
5645
+ "last_updated": "2026-09-22",
5646
+ "modalities": {
5647
+ "input": [
5648
+ "text",
5649
+ "image"
5650
+ ],
5651
+ "output": [
5652
+ "text"
5653
+ ]
5654
+ },
5655
+ "open_weights": false,
5656
+ "limit": {
5657
+ "context": 1050000,
5658
+ "input": 922000,
5659
+ "output": 128000
5660
+ },
5661
+ "cost": {
5662
+ "input": 2.2,
5663
+ "output": 11,
5664
+ "cache_read": 0.22,
5665
+ "cache_write": 2.75,
5666
+ "tiers": [
5667
+ {
5668
+ "input": 4.4,
5669
+ "output": 16.5,
5670
+ "cache_read": 0.44,
5671
+ "cache_write": 5.5,
5672
+ "tier": {
5673
+ "type": "context",
5674
+ "size": 272000
5675
+ }
5676
+ }
5677
+ ],
5678
+ "context_over_200k": {
5679
+ "input": 4.4,
5680
+ "output": 16.5,
5681
+ "cache_read": 0.44,
5682
+ "cache_write": 5.5
5683
+ }
5684
+ }
5685
+ },
5352
5686
  "global.anthropic.claude-opus-4-7": {
5353
5687
  "id": "global.anthropic.claude-opus-4-7",
5354
5688
  "name": "Claude Opus 4.7 (Global)",
data/data/deepseek.json CHANGED
@@ -49,7 +49,7 @@
49
49
  "open_weights": true,
50
50
  "limit": {
51
51
  "context": 1000000,
52
- "output": 384000
52
+ "output": 393216
53
53
  },
54
54
  "status": "deprecated",
55
55
  "cost": {
@@ -100,7 +100,7 @@
100
100
  "open_weights": true,
101
101
  "limit": {
102
102
  "context": 1000000,
103
- "output": 384000
103
+ "output": 393216
104
104
  },
105
105
  "status": "deprecated",
106
106
  "cost": {
@@ -149,7 +149,7 @@
149
149
  "open_weights": true,
150
150
  "limit": {
151
151
  "context": 1000000,
152
- "output": 384000
152
+ "output": 393216
153
153
  },
154
154
  "cost": {
155
155
  "input": 0.435,
@@ -199,7 +199,7 @@
199
199
  "open_weights": true,
200
200
  "limit": {
201
201
  "context": 1000000,
202
- "output": 384000
202
+ "output": 393216
203
203
  },
204
204
  "cost": {
205
205
  "input": 0.15,
data/data/google.json CHANGED
@@ -138,7 +138,7 @@
138
138
  "open_weights": false,
139
139
  "limit": {
140
140
  "context": 65536,
141
- "output": 65536
141
+ "output": 4096
142
142
  },
143
143
  "cost": {
144
144
  "input": 0.25,
@@ -933,8 +933,8 @@
933
933
  },
934
934
  "open_weights": false,
935
935
  "limit": {
936
- "context": 65536,
937
- "output": 65536
936
+ "context": 131072,
937
+ "output": 32768
938
938
  },
939
939
  "cost": {
940
940
  "input": 0.5,
data/data/moonshot.json CHANGED
@@ -167,7 +167,7 @@
167
167
  "open_weights": true,
168
168
  "limit": {
169
169
  "context": 1048576,
170
- "output": 131072
170
+ "output": 1048576
171
171
  },
172
172
  "cost": {
173
173
  "input": 3,
data/data/openrouter.json CHANGED
@@ -1675,6 +1675,51 @@
1675
1675
  }
1676
1676
  }
1677
1677
  },
1678
+ "qwen/qwen3.8-max-prime": {
1679
+ "id": "qwen/qwen3.8-max-prime",
1680
+ "name": "Qwen3.8 Max Prime",
1681
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1682
+ "family": "qwen3.8-max",
1683
+ "attachment": true,
1684
+ "reasoning": true,
1685
+ "reasoning_options": [
1686
+ {
1687
+ "type": "effort",
1688
+ "values": [
1689
+ "minimal",
1690
+ "low",
1691
+ "medium",
1692
+ "high",
1693
+ "xhigh"
1694
+ ]
1695
+ }
1696
+ ],
1697
+ "tool_call": true,
1698
+ "structured_output": true,
1699
+ "temperature": true,
1700
+ "release_date": "2026-09-23",
1701
+ "last_updated": "2026-09-23",
1702
+ "modalities": {
1703
+ "input": [
1704
+ "text",
1705
+ "image",
1706
+ "video"
1707
+ ],
1708
+ "output": [
1709
+ "text"
1710
+ ]
1711
+ },
1712
+ "open_weights": false,
1713
+ "limit": {
1714
+ "context": 1000000,
1715
+ "output": 131072
1716
+ },
1717
+ "cost": {
1718
+ "input": 4,
1719
+ "output": 12,
1720
+ "cache_read": 0.5
1721
+ }
1722
+ },
1678
1723
  "qwen/qwen3-30b-a3b": {
1679
1724
  "id": "qwen/qwen3-30b-a3b",
1680
1725
  "name": "Qwen3 30B A3B",
@@ -2189,6 +2234,46 @@
2189
2234
  "cache_read": 0.25
2190
2235
  }
2191
2236
  },
2237
+ "aion-labs/aion-3.5": {
2238
+ "id": "aion-labs/aion-3.5",
2239
+ "name": "Aion 3.5",
2240
+ "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
2241
+ "attachment": false,
2242
+ "reasoning": true,
2243
+ "reasoning_options": [
2244
+ {
2245
+ "type": "effort",
2246
+ "values": [
2247
+ "low",
2248
+ "high",
2249
+ "max"
2250
+ ]
2251
+ }
2252
+ ],
2253
+ "tool_call": true,
2254
+ "structured_output": false,
2255
+ "temperature": true,
2256
+ "release_date": "2026-09-23",
2257
+ "last_updated": "2026-09-23",
2258
+ "modalities": {
2259
+ "input": [
2260
+ "text"
2261
+ ],
2262
+ "output": [
2263
+ "text"
2264
+ ]
2265
+ },
2266
+ "open_weights": false,
2267
+ "limit": {
2268
+ "context": 262144,
2269
+ "output": 32768
2270
+ },
2271
+ "cost": {
2272
+ "input": 3,
2273
+ "output": 6,
2274
+ "cache_read": 0.75
2275
+ }
2276
+ },
2192
2277
  "aion-labs/aion-2.0": {
2193
2278
  "id": "aion-labs/aion-2.0",
2194
2279
  "name": "Aion-2.0",
@@ -2282,6 +2367,46 @@
2282
2367
  "cache_read": 0.75
2283
2368
  }
2284
2369
  },
2370
+ "aion-labs/aion-3.5-mini": {
2371
+ "id": "aion-labs/aion-3.5-mini",
2372
+ "name": "Aion 3.5 Mini",
2373
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
2374
+ "attachment": false,
2375
+ "reasoning": true,
2376
+ "reasoning_options": [
2377
+ {
2378
+ "type": "effort",
2379
+ "values": [
2380
+ "low",
2381
+ "high",
2382
+ "max"
2383
+ ]
2384
+ }
2385
+ ],
2386
+ "tool_call": true,
2387
+ "structured_output": false,
2388
+ "temperature": true,
2389
+ "release_date": "2026-09-23",
2390
+ "last_updated": "2026-09-23",
2391
+ "modalities": {
2392
+ "input": [
2393
+ "text"
2394
+ ],
2395
+ "output": [
2396
+ "text"
2397
+ ]
2398
+ },
2399
+ "open_weights": false,
2400
+ "limit": {
2401
+ "context": 262144,
2402
+ "output": 32768
2403
+ },
2404
+ "cost": {
2405
+ "input": 0.7,
2406
+ "output": 1.4,
2407
+ "cache_read": 0.18
2408
+ }
2409
+ },
2285
2410
  "aion-labs/aion-3.0-mini": {
2286
2411
  "id": "aion-labs/aion-3.0-mini",
2287
2412
  "name": "Aion-3.0-Mini",
@@ -2623,9 +2748,9 @@
2623
2748
  "output": 943718
2624
2749
  },
2625
2750
  "cost": {
2626
- "input": 0.03,
2627
- "output": 0.8,
2628
- "cache_read": 0.008
2751
+ "input": 0.038,
2752
+ "output": 0.55,
2753
+ "cache_read": 0.0228
2629
2754
  }
2630
2755
  },
2631
2756
  "~deepseek/deepseek-pro-latest": {
@@ -2664,12 +2789,12 @@
2664
2789
  "open_weights": false,
2665
2790
  "limit": {
2666
2791
  "context": 1048576,
2667
- "output": 384000
2792
+ "output": 393216
2668
2793
  },
2669
2794
  "cost": {
2670
- "input": 0.4,
2671
- "output": 4.3,
2672
- "cache_read": 0.033
2795
+ "input": 0.38544,
2796
+ "output": 1.15632,
2797
+ "cache_read": 0.012264
2673
2798
  }
2674
2799
  },
2675
2800
  "~deepseek/deepseek-flash-latest": {
@@ -2712,9 +2837,9 @@
2712
2837
  "output": 943718
2713
2838
  },
2714
2839
  "cost": {
2715
- "input": 0.1,
2716
- "output": 0.5,
2717
- "cache_read": 0.01
2840
+ "input": 0.099,
2841
+ "output": 0.6,
2842
+ "cache_read": 0.06
2718
2843
  }
2719
2844
  },
2720
2845
  "dots-studio/dots-3-note-preview:free": {
@@ -3280,39 +3405,6 @@
3280
3405
  "output": 7.5
3281
3406
  }
3282
3407
  },
3283
- "mistralai/devstral-2512": {
3284
- "id": "mistralai/devstral-2512",
3285
- "name": "Devstral 2",
3286
- "description": "Mistral coding agent model for repository tasks and software engineering workflows",
3287
- "family": "devstral",
3288
- "attachment": true,
3289
- "reasoning": false,
3290
- "tool_call": true,
3291
- "structured_output": true,
3292
- "temperature": true,
3293
- "knowledge": "2025-12",
3294
- "release_date": "2025-12-09",
3295
- "last_updated": "2025-12-09",
3296
- "modalities": {
3297
- "input": [
3298
- "text",
3299
- "pdf"
3300
- ],
3301
- "output": [
3302
- "text"
3303
- ]
3304
- },
3305
- "open_weights": true,
3306
- "limit": {
3307
- "context": 262144,
3308
- "output": 209715
3309
- },
3310
- "cost": {
3311
- "input": 0.4,
3312
- "output": 2,
3313
- "cache_read": 0.04
3314
- }
3315
- },
3316
3408
  "mistralai/mistral-large-2407": {
3317
3409
  "id": "mistralai/mistral-large-2407",
3318
3410
  "name": "Mistral Large 2407",
@@ -3760,21 +3852,20 @@
3760
3852
  "tool_call": true,
3761
3853
  "structured_output": true,
3762
3854
  "temperature": true,
3763
- "knowledge": "2026-09-22",
3764
3855
  "release_date": "2026-09-22",
3765
3856
  "last_updated": "2026-09-22",
3766
3857
  "modalities": {
3767
3858
  "input": [
3768
3859
  "text",
3769
3860
  "image",
3770
- "video",
3771
- "audio"
3861
+ "audio",
3862
+ "video"
3772
3863
  ],
3773
3864
  "output": [
3774
3865
  "text"
3775
3866
  ]
3776
3867
  },
3777
- "open_weights": false,
3868
+ "open_weights": true,
3778
3869
  "limit": {
3779
3870
  "context": 1048576,
3780
3871
  "output": 131072
@@ -3806,14 +3897,14 @@
3806
3897
  "input": [
3807
3898
  "text",
3808
3899
  "image",
3809
- "video",
3810
- "audio"
3900
+ "audio",
3901
+ "video"
3811
3902
  ],
3812
3903
  "output": [
3813
3904
  "text"
3814
3905
  ]
3815
3906
  },
3816
- "open_weights": false,
3907
+ "open_weights": true,
3817
3908
  "limit": {
3818
3909
  "context": 1048576,
3819
3910
  "output": 131072
@@ -3839,21 +3930,20 @@
3839
3930
  "tool_call": true,
3840
3931
  "structured_output": true,
3841
3932
  "temperature": true,
3842
- "knowledge": "2026-09-22",
3843
3933
  "release_date": "2026-09-22",
3844
3934
  "last_updated": "2026-09-22",
3845
3935
  "modalities": {
3846
3936
  "input": [
3847
3937
  "text",
3848
3938
  "image",
3849
- "video",
3850
- "audio"
3939
+ "audio",
3940
+ "video"
3851
3941
  ],
3852
3942
  "output": [
3853
3943
  "text"
3854
3944
  ]
3855
3945
  },
3856
- "open_weights": false,
3946
+ "open_weights": true,
3857
3947
  "limit": {
3858
3948
  "context": 1048576,
3859
3949
  "output": 131072
@@ -4319,10 +4409,10 @@
4319
4409
  "open_weights": true,
4320
4410
  "limit": {
4321
4411
  "context": 262144,
4322
- "output": 235929
4412
+ "output": 131072
4323
4413
  },
4324
4414
  "cost": {
4325
- "input": 0.07,
4415
+ "input": 0.08,
4326
4416
  "output": 0.2,
4327
4417
  "cache_read": 0.04
4328
4418
  }
@@ -8421,8 +8511,8 @@
8421
8511
  "output": 943718
8422
8512
  },
8423
8513
  "cost": {
8424
- "input": 1.05,
8425
- "output": 13,
8514
+ "input": 1.4989,
8515
+ "output": 10.758,
8426
8516
  "cache_read": 0.3
8427
8517
  }
8428
8518
  },
@@ -8618,9 +8708,9 @@
8618
8708
  "output": 384000
8619
8709
  },
8620
8710
  "cost": {
8621
- "input": 1.32,
8622
- "output": 3.96,
8623
- "cache_read": 0.044
8711
+ "input": 0.462,
8712
+ "output": 1.386,
8713
+ "cache_read": 0.0154
8624
8714
  }
8625
8715
  },
8626
8716
  "deepseek/deepseek-v4-flash-0731": {
@@ -8710,9 +8800,9 @@
8710
8800
  "output": 384000
8711
8801
  },
8712
8802
  "cost": {
8713
- "input": 0.088606,
8714
- "output": 0.177212,
8715
- "cache_read": 0.017721
8803
+ "input": 0.07168,
8804
+ "output": 0.14336,
8805
+ "cache_read": 0.014336
8716
8806
  }
8717
8807
  },
8718
8808
  "deepseek/deepseek-v4.1-flash": {
@@ -8753,12 +8843,12 @@
8753
8843
  "open_weights": true,
8754
8844
  "limit": {
8755
8845
  "context": 1048576,
8756
- "output": 943718
8846
+ "output": 393216
8757
8847
  },
8758
8848
  "cost": {
8759
- "input": 0.1,
8760
- "output": 0.5,
8761
- "cache_read": 0.01
8849
+ "input": 0.15,
8850
+ "output": 0.6,
8851
+ "cache_read": 0.003
8762
8852
  }
8763
8853
  },
8764
8854
  "deepseek/deepseek-r1": {
@@ -9045,9 +9135,9 @@
9045
9135
  "output": 384000
9046
9136
  },
9047
9137
  "cost": {
9048
- "input": 0.95526,
9049
- "output": 1.91052,
9050
- "cache_read": 0.079605
9138
+ "input": 0.946386,
9139
+ "output": 1.892772,
9140
+ "cache_read": 0.078866
9051
9141
  }
9052
9142
  },
9053
9143
  "deepseek/deepseek-chat-v3-0324": {
@@ -9696,43 +9786,6 @@
9696
9786
  "output": 0
9697
9787
  }
9698
9788
  },
9699
- "inclusionai/ling-3.0-flash-vl:free": {
9700
- "id": "inclusionai/ling-3.0-flash-vl:free",
9701
- "name": "Ling 3.0 Flash VL (free)",
9702
- "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
9703
- "family": "ling",
9704
- "attachment": true,
9705
- "reasoning": true,
9706
- "reasoning_options": [
9707
- {
9708
- "type": "toggle"
9709
- }
9710
- ],
9711
- "tool_call": true,
9712
- "structured_output": false,
9713
- "temperature": true,
9714
- "release_date": "2026-09-10",
9715
- "last_updated": "2026-09-10",
9716
- "modalities": {
9717
- "input": [
9718
- "text",
9719
- "image",
9720
- "video"
9721
- ],
9722
- "output": [
9723
- "text"
9724
- ]
9725
- },
9726
- "open_weights": true,
9727
- "limit": {
9728
- "context": 262144,
9729
- "output": 32768
9730
- },
9731
- "cost": {
9732
- "input": 0,
9733
- "output": 0
9734
- }
9735
- },
9736
9789
  "inclusionai/ling-3.0-flash-vl": {
9737
9790
  "id": "inclusionai/ling-3.0-flash-vl",
9738
9791
  "name": "Ling 3.0 Flash VL",
@@ -9762,7 +9815,7 @@
9762
9815
  },
9763
9816
  "open_weights": true,
9764
9817
  "limit": {
9765
- "context": 131072,
9818
+ "context": 262144,
9766
9819
  "output": 32768
9767
9820
  },
9768
9821
  "cost": {
@@ -13753,9 +13806,9 @@
13753
13806
  "output": 131072
13754
13807
  },
13755
13808
  "cost": {
13756
- "input": 0.5625,
13757
- "output": 2.5,
13758
- "cache_read": 0.125
13809
+ "input": 0.5614,
13810
+ "output": 1.7644,
13811
+ "cache_read": 0.10426
13759
13812
  }
13760
13813
  },
13761
13814
  "moonshotai/kimi-k2-0905": {
@@ -14362,6 +14415,51 @@
14362
14415
  "cache_read": 0.018
14363
14416
  }
14364
14417
  },
14418
+ "upstage/solar-mini4": {
14419
+ "id": "upstage/solar-mini4",
14420
+ "name": "Solar Mini 4",
14421
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
14422
+ "family": "solar",
14423
+ "attachment": false,
14424
+ "reasoning": true,
14425
+ "reasoning_options": [
14426
+ {
14427
+ "type": "effort",
14428
+ "values": [
14429
+ "none",
14430
+ "minimal",
14431
+ "low",
14432
+ "medium",
14433
+ "high",
14434
+ "xhigh",
14435
+ "max"
14436
+ ]
14437
+ }
14438
+ ],
14439
+ "tool_call": true,
14440
+ "structured_output": true,
14441
+ "temperature": true,
14442
+ "release_date": "2026-09-23",
14443
+ "last_updated": "2026-09-23",
14444
+ "modalities": {
14445
+ "input": [
14446
+ "text"
14447
+ ],
14448
+ "output": [
14449
+ "text"
14450
+ ]
14451
+ },
14452
+ "open_weights": false,
14453
+ "limit": {
14454
+ "context": 524288,
14455
+ "output": 131072
14456
+ },
14457
+ "cost": {
14458
+ "input": 0.05,
14459
+ "output": 0.2,
14460
+ "cache_read": 0.005
14461
+ }
14462
+ },
14365
14463
  "arcee-ai/trinity-large-thinking": {
14366
14464
  "id": "arcee-ai/trinity-large-thinking",
14367
14465
  "name": "Trinity Large Thinking",
@@ -14431,9 +14529,9 @@
14431
14529
  "output": 128000
14432
14530
  },
14433
14531
  "cost": {
14434
- "input": 0.132,
14435
- "output": 0.528,
14436
- "cache_read": 0.033
14532
+ "input": 0.0825,
14533
+ "output": 0.33,
14534
+ "cache_read": 0.020625
14437
14535
  }
14438
14536
  },
14439
14537
  "tencent/hy4-preview": {
@@ -14914,7 +15012,7 @@
14914
15012
  "cost": {
14915
15013
  "input": 0.37,
14916
15014
  "output": 1.25,
14917
- "cache_read": 0.075
15015
+ "cache_read": 0.09
14918
15016
  }
14919
15017
  },
14920
15018
  "z-ai/glm-5.3-flash": {
@@ -15155,6 +15253,47 @@
15155
15253
  "cache_read": 0.1794
15156
15254
  }
15157
15255
  },
15256
+ "z-ai/glm-5.3-prime": {
15257
+ "id": "z-ai/glm-5.3-prime",
15258
+ "name": "GLM 5.3 Prime",
15259
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
15260
+ "family": "glm",
15261
+ "attachment": false,
15262
+ "reasoning": true,
15263
+ "reasoning_options": [
15264
+ {
15265
+ "type": "effort",
15266
+ "values": [
15267
+ "low",
15268
+ "high",
15269
+ "max"
15270
+ ]
15271
+ }
15272
+ ],
15273
+ "tool_call": true,
15274
+ "structured_output": false,
15275
+ "temperature": true,
15276
+ "release_date": "2026-09-23",
15277
+ "last_updated": "2026-09-23",
15278
+ "modalities": {
15279
+ "input": [
15280
+ "text"
15281
+ ],
15282
+ "output": [
15283
+ "text"
15284
+ ]
15285
+ },
15286
+ "open_weights": false,
15287
+ "limit": {
15288
+ "context": 1000000,
15289
+ "output": 131072
15290
+ },
15291
+ "cost": {
15292
+ "input": 2.8,
15293
+ "output": 8.8,
15294
+ "cache_read": 0.56
15295
+ }
15296
+ },
15158
15297
  "z-ai/glm-5-turbo": {
15159
15298
  "id": "z-ai/glm-5-turbo",
15160
15299
  "name": "GLM-5-Turbo",
@@ -15572,6 +15711,50 @@
15572
15711
  "input": 0.1,
15573
15712
  "output": 0.2
15574
15713
  }
15714
+ },
15715
+ "stealth/space-bunny-alpha": {
15716
+ "id": "stealth/space-bunny-alpha",
15717
+ "name": "Space Bunny Alpha",
15718
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
15719
+ "family": "alpha",
15720
+ "attachment": true,
15721
+ "reasoning": true,
15722
+ "reasoning_options": [
15723
+ {
15724
+ "type": "effort",
15725
+ "values": [
15726
+ "low",
15727
+ "medium",
15728
+ "high",
15729
+ "xhigh",
15730
+ "max"
15731
+ ]
15732
+ }
15733
+ ],
15734
+ "tool_call": true,
15735
+ "structured_output": false,
15736
+ "temperature": true,
15737
+ "release_date": "2026-09-23",
15738
+ "last_updated": "2026-09-23",
15739
+ "modalities": {
15740
+ "input": [
15741
+ "text",
15742
+ "image",
15743
+ "video"
15744
+ ],
15745
+ "output": [
15746
+ "text"
15747
+ ]
15748
+ },
15749
+ "open_weights": false,
15750
+ "limit": {
15751
+ "context": 1000000,
15752
+ "output": 524288
15753
+ },
15754
+ "cost": {
15755
+ "input": 0,
15756
+ "output": 0
15757
+ }
15575
15758
  }
15576
15759
  }
15577
15760
  }
data/lib/llm/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module LLM
4
- VERSION = "15.4.0"
4
+ VERSION = "15.4.1"
5
5
  end
data/llm.gemspec CHANGED
@@ -34,12 +34,6 @@ DESCRIPTION
34
34
  ]
35
35
  spec.executables = ["llm.rb"]
36
36
  spec.require_paths = ["lib"]
37
- spec.post_install_message = "\n" \
38
- "Want to see what I'm building with llm.rb? " \
39
- "\n" \
40
- "Checkout the https://r.uby.dev website." \
41
- "\n\n"
42
-
43
37
  spec.add_development_dependency "webmock", "~> 3.24.0"
44
38
  spec.add_development_dependency "yard", "~> 0.9.37"
45
39
  spec.add_development_dependency "redcarpet", "~> 3.6"
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: llm.rb
3
3
  version: !ruby/object:Gem::Version
4
- version: 15.4.0
4
+ version: 15.4.1
5
5
  platform: ruby
6
6
  authors:
7
7
  - Robert Gleeson
@@ -695,8 +695,6 @@ metadata:
695
695
  source_code_uri: https://github.com/r-uby-dev/llm
696
696
  documentation_uri: https://r.uby.dev
697
697
  changelog_uri: https://github.com/r-uby-dev/llm/blob/main/CHANGELOG.md
698
- post_install_message: "\nWant to see what I'm building with llm.rb? \nCheckout the
699
- https://r.uby.dev website.\n\n"
700
698
  rdoc_options: []
701
699
  require_paths:
702
700
  - lib