llm.rb 15.4.0 → 15.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/data/openrouter.json CHANGED
@@ -1675,6 +1675,51 @@
1675
1675
  }
1676
1676
  }
1677
1677
  },
1678
+ "qwen/qwen3.8-max-prime": {
1679
+ "id": "qwen/qwen3.8-max-prime",
1680
+ "name": "Qwen3.8 Max Prime",
1681
+ "description": "Qwen vision-language model for visual reasoning, documents, and agent tasks",
1682
+ "family": "qwen3.8-max",
1683
+ "attachment": true,
1684
+ "reasoning": true,
1685
+ "reasoning_options": [
1686
+ {
1687
+ "type": "effort",
1688
+ "values": [
1689
+ "minimal",
1690
+ "low",
1691
+ "medium",
1692
+ "high",
1693
+ "xhigh"
1694
+ ]
1695
+ }
1696
+ ],
1697
+ "tool_call": true,
1698
+ "structured_output": true,
1699
+ "temperature": true,
1700
+ "release_date": "2026-09-23",
1701
+ "last_updated": "2026-09-23",
1702
+ "modalities": {
1703
+ "input": [
1704
+ "text",
1705
+ "image",
1706
+ "video"
1707
+ ],
1708
+ "output": [
1709
+ "text"
1710
+ ]
1711
+ },
1712
+ "open_weights": false,
1713
+ "limit": {
1714
+ "context": 1000000,
1715
+ "output": 131072
1716
+ },
1717
+ "cost": {
1718
+ "input": 4,
1719
+ "output": 12,
1720
+ "cache_read": 0.5
1721
+ }
1722
+ },
1678
1723
  "qwen/qwen3-30b-a3b": {
1679
1724
  "id": "qwen/qwen3-30b-a3b",
1680
1725
  "name": "Qwen3 30B A3B",
@@ -2189,6 +2234,46 @@
2189
2234
  "cache_read": 0.25
2190
2235
  }
2191
2236
  },
2237
+ "aion-labs/aion-3.5": {
2238
+ "id": "aion-labs/aion-3.5",
2239
+ "name": "Aion 3.5",
2240
+ "description": "Reasoning model for deliberate analysis, multi-step problem solving, and tool use",
2241
+ "attachment": false,
2242
+ "reasoning": true,
2243
+ "reasoning_options": [
2244
+ {
2245
+ "type": "effort",
2246
+ "values": [
2247
+ "low",
2248
+ "high",
2249
+ "max"
2250
+ ]
2251
+ }
2252
+ ],
2253
+ "tool_call": true,
2254
+ "structured_output": false,
2255
+ "temperature": true,
2256
+ "release_date": "2026-09-23",
2257
+ "last_updated": "2026-09-23",
2258
+ "modalities": {
2259
+ "input": [
2260
+ "text"
2261
+ ],
2262
+ "output": [
2263
+ "text"
2264
+ ]
2265
+ },
2266
+ "open_weights": false,
2267
+ "limit": {
2268
+ "context": 262144,
2269
+ "output": 32768
2270
+ },
2271
+ "cost": {
2272
+ "input": 3,
2273
+ "output": 6,
2274
+ "cache_read": 0.75
2275
+ }
2276
+ },
2192
2277
  "aion-labs/aion-2.0": {
2193
2278
  "id": "aion-labs/aion-2.0",
2194
2279
  "name": "Aion-2.0",
@@ -2282,6 +2367,46 @@
2282
2367
  "cache_read": 0.75
2283
2368
  }
2284
2369
  },
2370
+ "aion-labs/aion-3.5-mini": {
2371
+ "id": "aion-labs/aion-3.5-mini",
2372
+ "name": "Aion 3.5 Mini",
2373
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
2374
+ "attachment": false,
2375
+ "reasoning": true,
2376
+ "reasoning_options": [
2377
+ {
2378
+ "type": "effort",
2379
+ "values": [
2380
+ "low",
2381
+ "high",
2382
+ "max"
2383
+ ]
2384
+ }
2385
+ ],
2386
+ "tool_call": true,
2387
+ "structured_output": false,
2388
+ "temperature": true,
2389
+ "release_date": "2026-09-23",
2390
+ "last_updated": "2026-09-23",
2391
+ "modalities": {
2392
+ "input": [
2393
+ "text"
2394
+ ],
2395
+ "output": [
2396
+ "text"
2397
+ ]
2398
+ },
2399
+ "open_weights": false,
2400
+ "limit": {
2401
+ "context": 262144,
2402
+ "output": 32768
2403
+ },
2404
+ "cost": {
2405
+ "input": 0.7,
2406
+ "output": 1.4,
2407
+ "cache_read": 0.18
2408
+ }
2409
+ },
2285
2410
  "aion-labs/aion-3.0-mini": {
2286
2411
  "id": "aion-labs/aion-3.0-mini",
2287
2412
  "name": "Aion-3.0-Mini",
@@ -2623,9 +2748,9 @@
2623
2748
  "output": 943718
2624
2749
  },
2625
2750
  "cost": {
2626
- "input": 0.03,
2627
- "output": 0.8,
2628
- "cache_read": 0.008
2751
+ "input": 0.038,
2752
+ "output": 0.55,
2753
+ "cache_read": 0.0228
2629
2754
  }
2630
2755
  },
2631
2756
  "~deepseek/deepseek-pro-latest": {
@@ -2664,12 +2789,12 @@
2664
2789
  "open_weights": false,
2665
2790
  "limit": {
2666
2791
  "context": 1048576,
2667
- "output": 384000
2792
+ "output": 393216
2668
2793
  },
2669
2794
  "cost": {
2670
- "input": 0.4,
2671
- "output": 4.3,
2672
- "cache_read": 0.033
2795
+ "input": 0.38544,
2796
+ "output": 1.15632,
2797
+ "cache_read": 0.012264
2673
2798
  }
2674
2799
  },
2675
2800
  "~deepseek/deepseek-flash-latest": {
@@ -2712,9 +2837,9 @@
2712
2837
  "output": 943718
2713
2838
  },
2714
2839
  "cost": {
2715
- "input": 0.1,
2716
- "output": 0.5,
2717
- "cache_read": 0.01
2840
+ "input": 0.099,
2841
+ "output": 0.6,
2842
+ "cache_read": 0.06
2718
2843
  }
2719
2844
  },
2720
2845
  "dots-studio/dots-3-note-preview:free": {
@@ -3280,39 +3405,6 @@
3280
3405
  "output": 7.5
3281
3406
  }
3282
3407
  },
3283
- "mistralai/devstral-2512": {
3284
- "id": "mistralai/devstral-2512",
3285
- "name": "Devstral 2",
3286
- "description": "Mistral coding agent model for repository tasks and software engineering workflows",
3287
- "family": "devstral",
3288
- "attachment": true,
3289
- "reasoning": false,
3290
- "tool_call": true,
3291
- "structured_output": true,
3292
- "temperature": true,
3293
- "knowledge": "2025-12",
3294
- "release_date": "2025-12-09",
3295
- "last_updated": "2025-12-09",
3296
- "modalities": {
3297
- "input": [
3298
- "text",
3299
- "pdf"
3300
- ],
3301
- "output": [
3302
- "text"
3303
- ]
3304
- },
3305
- "open_weights": true,
3306
- "limit": {
3307
- "context": 262144,
3308
- "output": 209715
3309
- },
3310
- "cost": {
3311
- "input": 0.4,
3312
- "output": 2,
3313
- "cache_read": 0.04
3314
- }
3315
- },
3316
3408
  "mistralai/mistral-large-2407": {
3317
3409
  "id": "mistralai/mistral-large-2407",
3318
3410
  "name": "Mistral Large 2407",
@@ -3760,21 +3852,20 @@
3760
3852
  "tool_call": true,
3761
3853
  "structured_output": true,
3762
3854
  "temperature": true,
3763
- "knowledge": "2026-09-22",
3764
3855
  "release_date": "2026-09-22",
3765
3856
  "last_updated": "2026-09-22",
3766
3857
  "modalities": {
3767
3858
  "input": [
3768
3859
  "text",
3769
3860
  "image",
3770
- "video",
3771
- "audio"
3861
+ "audio",
3862
+ "video"
3772
3863
  ],
3773
3864
  "output": [
3774
3865
  "text"
3775
3866
  ]
3776
3867
  },
3777
- "open_weights": false,
3868
+ "open_weights": true,
3778
3869
  "limit": {
3779
3870
  "context": 1048576,
3780
3871
  "output": 131072
@@ -3806,14 +3897,14 @@
3806
3897
  "input": [
3807
3898
  "text",
3808
3899
  "image",
3809
- "video",
3810
- "audio"
3900
+ "audio",
3901
+ "video"
3811
3902
  ],
3812
3903
  "output": [
3813
3904
  "text"
3814
3905
  ]
3815
3906
  },
3816
- "open_weights": false,
3907
+ "open_weights": true,
3817
3908
  "limit": {
3818
3909
  "context": 1048576,
3819
3910
  "output": 131072
@@ -3839,21 +3930,20 @@
3839
3930
  "tool_call": true,
3840
3931
  "structured_output": true,
3841
3932
  "temperature": true,
3842
- "knowledge": "2026-09-22",
3843
3933
  "release_date": "2026-09-22",
3844
3934
  "last_updated": "2026-09-22",
3845
3935
  "modalities": {
3846
3936
  "input": [
3847
3937
  "text",
3848
3938
  "image",
3849
- "video",
3850
- "audio"
3939
+ "audio",
3940
+ "video"
3851
3941
  ],
3852
3942
  "output": [
3853
3943
  "text"
3854
3944
  ]
3855
3945
  },
3856
- "open_weights": false,
3946
+ "open_weights": true,
3857
3947
  "limit": {
3858
3948
  "context": 1048576,
3859
3949
  "output": 131072
@@ -4319,10 +4409,10 @@
4319
4409
  "open_weights": true,
4320
4410
  "limit": {
4321
4411
  "context": 262144,
4322
- "output": 235929
4412
+ "output": 131072
4323
4413
  },
4324
4414
  "cost": {
4325
- "input": 0.07,
4415
+ "input": 0.08,
4326
4416
  "output": 0.2,
4327
4417
  "cache_read": 0.04
4328
4418
  }
@@ -8421,8 +8511,8 @@
8421
8511
  "output": 943718
8422
8512
  },
8423
8513
  "cost": {
8424
- "input": 1.05,
8425
- "output": 13,
8514
+ "input": 1.4989,
8515
+ "output": 10.758,
8426
8516
  "cache_read": 0.3
8427
8517
  }
8428
8518
  },
@@ -8618,9 +8708,9 @@
8618
8708
  "output": 384000
8619
8709
  },
8620
8710
  "cost": {
8621
- "input": 1.32,
8622
- "output": 3.96,
8623
- "cache_read": 0.044
8711
+ "input": 0.462,
8712
+ "output": 1.386,
8713
+ "cache_read": 0.0154
8624
8714
  }
8625
8715
  },
8626
8716
  "deepseek/deepseek-v4-flash-0731": {
@@ -8710,9 +8800,9 @@
8710
8800
  "output": 384000
8711
8801
  },
8712
8802
  "cost": {
8713
- "input": 0.088606,
8714
- "output": 0.177212,
8715
- "cache_read": 0.017721
8803
+ "input": 0.07168,
8804
+ "output": 0.14336,
8805
+ "cache_read": 0.014336
8716
8806
  }
8717
8807
  },
8718
8808
  "deepseek/deepseek-v4.1-flash": {
@@ -8753,12 +8843,12 @@
8753
8843
  "open_weights": true,
8754
8844
  "limit": {
8755
8845
  "context": 1048576,
8756
- "output": 943718
8846
+ "output": 393216
8757
8847
  },
8758
8848
  "cost": {
8759
- "input": 0.1,
8760
- "output": 0.5,
8761
- "cache_read": 0.01
8849
+ "input": 0.15,
8850
+ "output": 0.6,
8851
+ "cache_read": 0.003
8762
8852
  }
8763
8853
  },
8764
8854
  "deepseek/deepseek-r1": {
@@ -9045,9 +9135,9 @@
9045
9135
  "output": 384000
9046
9136
  },
9047
9137
  "cost": {
9048
- "input": 0.95526,
9049
- "output": 1.91052,
9050
- "cache_read": 0.079605
9138
+ "input": 0.946386,
9139
+ "output": 1.892772,
9140
+ "cache_read": 0.078866
9051
9141
  }
9052
9142
  },
9053
9143
  "deepseek/deepseek-chat-v3-0324": {
@@ -9696,43 +9786,6 @@
9696
9786
  "output": 0
9697
9787
  }
9698
9788
  },
9699
- "inclusionai/ling-3.0-flash-vl:free": {
9700
- "id": "inclusionai/ling-3.0-flash-vl:free",
9701
- "name": "Ling 3.0 Flash VL (free)",
9702
- "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
9703
- "family": "ling",
9704
- "attachment": true,
9705
- "reasoning": true,
9706
- "reasoning_options": [
9707
- {
9708
- "type": "toggle"
9709
- }
9710
- ],
9711
- "tool_call": true,
9712
- "structured_output": false,
9713
- "temperature": true,
9714
- "release_date": "2026-09-10",
9715
- "last_updated": "2026-09-10",
9716
- "modalities": {
9717
- "input": [
9718
- "text",
9719
- "image",
9720
- "video"
9721
- ],
9722
- "output": [
9723
- "text"
9724
- ]
9725
- },
9726
- "open_weights": true,
9727
- "limit": {
9728
- "context": 262144,
9729
- "output": 32768
9730
- },
9731
- "cost": {
9732
- "input": 0,
9733
- "output": 0
9734
- }
9735
- },
9736
9789
  "inclusionai/ling-3.0-flash-vl": {
9737
9790
  "id": "inclusionai/ling-3.0-flash-vl",
9738
9791
  "name": "Ling 3.0 Flash VL",
@@ -9762,7 +9815,7 @@
9762
9815
  },
9763
9816
  "open_weights": true,
9764
9817
  "limit": {
9765
- "context": 131072,
9818
+ "context": 262144,
9766
9819
  "output": 32768
9767
9820
  },
9768
9821
  "cost": {
@@ -13753,9 +13806,9 @@
13753
13806
  "output": 131072
13754
13807
  },
13755
13808
  "cost": {
13756
- "input": 0.5625,
13757
- "output": 2.5,
13758
- "cache_read": 0.125
13809
+ "input": 0.5614,
13810
+ "output": 1.7644,
13811
+ "cache_read": 0.10426
13759
13812
  }
13760
13813
  },
13761
13814
  "moonshotai/kimi-k2-0905": {
@@ -14362,6 +14415,51 @@
14362
14415
  "cache_read": 0.018
14363
14416
  }
14364
14417
  },
14418
+ "upstage/solar-mini4": {
14419
+ "id": "upstage/solar-mini4",
14420
+ "name": "Solar Mini 4",
14421
+ "description": "Efficient model for low-latency assistance, extraction, and routine automation",
14422
+ "family": "solar",
14423
+ "attachment": false,
14424
+ "reasoning": true,
14425
+ "reasoning_options": [
14426
+ {
14427
+ "type": "effort",
14428
+ "values": [
14429
+ "none",
14430
+ "minimal",
14431
+ "low",
14432
+ "medium",
14433
+ "high",
14434
+ "xhigh",
14435
+ "max"
14436
+ ]
14437
+ }
14438
+ ],
14439
+ "tool_call": true,
14440
+ "structured_output": true,
14441
+ "temperature": true,
14442
+ "release_date": "2026-09-23",
14443
+ "last_updated": "2026-09-23",
14444
+ "modalities": {
14445
+ "input": [
14446
+ "text"
14447
+ ],
14448
+ "output": [
14449
+ "text"
14450
+ ]
14451
+ },
14452
+ "open_weights": false,
14453
+ "limit": {
14454
+ "context": 524288,
14455
+ "output": 131072
14456
+ },
14457
+ "cost": {
14458
+ "input": 0.05,
14459
+ "output": 0.2,
14460
+ "cache_read": 0.005
14461
+ }
14462
+ },
14365
14463
  "arcee-ai/trinity-large-thinking": {
14366
14464
  "id": "arcee-ai/trinity-large-thinking",
14367
14465
  "name": "Trinity Large Thinking",
@@ -14431,9 +14529,9 @@
14431
14529
  "output": 128000
14432
14530
  },
14433
14531
  "cost": {
14434
- "input": 0.132,
14435
- "output": 0.528,
14436
- "cache_read": 0.033
14532
+ "input": 0.0825,
14533
+ "output": 0.33,
14534
+ "cache_read": 0.020625
14437
14535
  }
14438
14536
  },
14439
14537
  "tencent/hy4-preview": {
@@ -14914,7 +15012,7 @@
14914
15012
  "cost": {
14915
15013
  "input": 0.37,
14916
15014
  "output": 1.25,
14917
- "cache_read": 0.075
15015
+ "cache_read": 0.09
14918
15016
  }
14919
15017
  },
14920
15018
  "z-ai/glm-5.3-flash": {
@@ -15155,6 +15253,47 @@
15155
15253
  "cache_read": 0.1794
15156
15254
  }
15157
15255
  },
15256
+ "z-ai/glm-5.3-prime": {
15257
+ "id": "z-ai/glm-5.3-prime",
15258
+ "name": "GLM 5.3 Prime",
15259
+ "description": "Flagship GLM model for hybrid reasoning, coding, and agentic engineering",
15260
+ "family": "glm",
15261
+ "attachment": false,
15262
+ "reasoning": true,
15263
+ "reasoning_options": [
15264
+ {
15265
+ "type": "effort",
15266
+ "values": [
15267
+ "low",
15268
+ "high",
15269
+ "max"
15270
+ ]
15271
+ }
15272
+ ],
15273
+ "tool_call": true,
15274
+ "structured_output": false,
15275
+ "temperature": true,
15276
+ "release_date": "2026-09-23",
15277
+ "last_updated": "2026-09-23",
15278
+ "modalities": {
15279
+ "input": [
15280
+ "text"
15281
+ ],
15282
+ "output": [
15283
+ "text"
15284
+ ]
15285
+ },
15286
+ "open_weights": false,
15287
+ "limit": {
15288
+ "context": 1000000,
15289
+ "output": 131072
15290
+ },
15291
+ "cost": {
15292
+ "input": 2.8,
15293
+ "output": 8.8,
15294
+ "cache_read": 0.56
15295
+ }
15296
+ },
15158
15297
  "z-ai/glm-5-turbo": {
15159
15298
  "id": "z-ai/glm-5-turbo",
15160
15299
  "name": "GLM-5-Turbo",
@@ -15572,6 +15711,50 @@
15572
15711
  "input": 0.1,
15573
15712
  "output": 0.2
15574
15713
  }
15714
+ },
15715
+ "stealth/space-bunny-alpha": {
15716
+ "id": "stealth/space-bunny-alpha",
15717
+ "name": "Space Bunny Alpha",
15718
+ "description": "Multimodal reasoning model for visual analysis, planning, and tool use",
15719
+ "family": "alpha",
15720
+ "attachment": true,
15721
+ "reasoning": true,
15722
+ "reasoning_options": [
15723
+ {
15724
+ "type": "effort",
15725
+ "values": [
15726
+ "low",
15727
+ "medium",
15728
+ "high",
15729
+ "xhigh",
15730
+ "max"
15731
+ ]
15732
+ }
15733
+ ],
15734
+ "tool_call": true,
15735
+ "structured_output": false,
15736
+ "temperature": true,
15737
+ "release_date": "2026-09-23",
15738
+ "last_updated": "2026-09-23",
15739
+ "modalities": {
15740
+ "input": [
15741
+ "text",
15742
+ "image",
15743
+ "video"
15744
+ ],
15745
+ "output": [
15746
+ "text"
15747
+ ]
15748
+ },
15749
+ "open_weights": false,
15750
+ "limit": {
15751
+ "context": 1000000,
15752
+ "output": 524288
15753
+ },
15754
+ "cost": {
15755
+ "input": 0,
15756
+ "output": 0
15757
+ }
15575
15758
  }
15576
15759
  }
15577
15760
  }
@@ -156,6 +156,16 @@ Three more hooks cover a local tool call. `on_tool_start` fires before
156
156
  the tool runs and returns the span that `on_tool_finish` and
157
157
  `on_tool_error` receive.
158
158
 
159
+ A tracer's own lifetime is bracketed as well. `on_exit` fires once,
160
+ when the last scope that is open for that tracer ends. That scope can
161
+ belong to a different thread than the one that opened the first: a tool
162
+ runs on a thread of its own and scopes the turn's tracer while it does.
163
+ `on_exit` is called after the scoped lookup has been restored, so a
164
+ tracer that asks its provider for the current tracer from inside it
165
+ sees the tracer the next request will see. A tracer can be scoped again
166
+ afterwards, so it has to remain usable after `on_exit`, and `on_exit`
167
+ may be called more than once over its life.
168
+
159
169
  A turn is additionally bracketed with `start_trace` and `stop_trace`.
160
170
  The runtime calls them around every agent turn with a `trace_group_id`,
161
171
  and a tracer that supports it (such as
data/lib/llm/function.rb CHANGED
@@ -199,6 +199,12 @@ class LLM::Function
199
199
 
200
200
  ##
201
201
  # Set (or get) the function parameters
202
+ #
203
+ # @note
204
+ # A parameter type that is given as a proc is resolved when the
205
+ # parameters are rendered rather than here, so a tool can say
206
+ # `parameter :fruit, proc { Enum[Fruit.keys] }` and ask the model
207
+ # for what is available at the time of the call.
202
208
  # @yieldparam [LLM::Schema] schema The schema object
203
209
  # @return [LLM::Schema::Leaf, nil]
204
210
  def params
data/lib/llm/provider.rb CHANGED
@@ -396,9 +396,18 @@ class LLM::Provider
396
396
  # Set the provider's default tracer
397
397
  # This tracer is shared by the provider instance and becomes the fallback
398
398
  # whenever no scoped override is active.
399
+ #
400
+ # A tracer assigned this way is not scoped, so nothing declares when it
401
+ # is finished and {LLM::Tracer#on_exit} never fires for it. Release what
402
+ # it holds when the provider is done with, or scope it with
403
+ # {#with_tracer} instead.
404
+ #
399
405
  # @example
400
406
  # llm = LLM.openai(key: ENV["KEY"])
401
407
  # llm.tracer = LLM::Tracer.logger(llm, path: "/path/to/log.txt")
408
+ # llm.with_tracer(llm.tracer) do
409
+ # # ...
410
+ # end
402
411
  # @param [LLM::Tracer] tracer
403
412
  # A tracer
404
413
  # @return [void]
@@ -418,19 +427,25 @@ class LLM::Provider
418
427
  # @yield
419
428
  # @return [Object]
420
429
  def with_tracer(tracer)
430
+ scoped = tracer || LLM::Tracer::Null.new(self)
421
431
  wm = weakmaps.tracer
422
432
  had_override = wm.key?(self)
423
433
  previous = wm[self]
424
- wm[self] = tracer || LLM::Tracer::Null.new(self)
434
+ wm[self] = scoped
435
+ entered = LLM::Tracer.enter(scoped)
425
436
  yield
426
437
  ensure
427
438
  if had_override
428
439
  wm[self] = previous
429
- elsif wm.respond_to?(:delete)
430
- wm.delete(self)
431
440
  else
432
- wm[self] = nil
441
+ wm.respond_to?(:delete) ? wm.delete(self) : wm[self] = nil
433
442
  end
443
+ ##
444
+ # The scope this tracer was opened for is over, and the tracer is
445
+ # told so when the last one that is open for it ends - on whichever
446
+ # thread that happens, because a tool opens a scope of its own for
447
+ # the turn's tracer.
448
+ LLM::Tracer.exit(scoped) if entered
434
449
  end
435
450
 
436
451
  ##