llm.rb 15.0.2 → 15.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +161 -1
- data/README.md +134 -85
- data/bin/llm.rb +38 -3
- data/data/bedrock.json +250 -0
- data/data/deepinfra.json +162 -0
- data/data/deepseek.json +50 -0
- data/data/google.json +19 -21
- data/data/openrouter.json +14280 -0
- data/data/xai.json +28 -16
- data/docs/deepdive/advanced/context.md +8 -6
- data/docs/deepdive/fundamentals/agents.md +11 -10
- data/docs/deepdive/fundamentals/stream.md +4 -4
- data/lib/llm/active_record.rb +1 -1
- data/lib/llm/agent.rb +6 -0
- data/lib/llm/context.rb +28 -13
- data/lib/llm/cost.rb +13 -0
- data/lib/llm/function/fork/task.rb +14 -10
- data/lib/llm/provider.rb +29 -8
- data/lib/llm/providers/anthropic.rb +1 -1
- data/lib/llm/providers/bedrock/models.rb +2 -2
- data/lib/llm/providers/bedrock.rb +1 -1
- data/lib/llm/providers/google.rb +1 -1
- data/lib/llm/providers/ollama.rb +1 -1
- data/lib/llm/providers/openai/responses.rb +2 -1
- data/lib/llm/providers/openai.rb +1 -1
- data/lib/llm/providers/openrouter.rb +87 -0
- data/lib/llm/repl/buffer.rb +20 -5
- data/lib/llm/repl/input.rb +5 -4
- data/lib/llm/repl/markdown/table.rb +5 -2
- data/lib/llm/repl/node.rb +26 -1
- data/lib/llm/repl/stream.rb +30 -3
- data/lib/llm/repl/window.rb +7 -7
- data/lib/llm/repl.rb +1 -0
- data/lib/llm/skill.rb +7 -1
- data/lib/llm/stream.rb +8 -3
- data/lib/llm/tools/git.rb +2 -2
- data/lib/llm/tools/mkdir.rb +2 -2
- data/lib/llm/tools/rg.rb +2 -2
- data/lib/llm/tools/ruby.rb +2 -2
- data/lib/llm/tools/shell.rb +2 -2
- data/lib/llm/tools/utils.rb +1 -1
- data/lib/llm/transport/curb.rb +5 -3
- data/lib/llm/transport/http.rb +5 -2
- data/lib/llm/transport/persistent_http.rb +6 -4
- data/lib/llm/transport/utils.rb +8 -6
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +16 -1
- data/llm.gemspec +2 -1
- metadata +19 -3
data/data/xai.json
CHANGED
|
@@ -74,22 +74,6 @@
|
|
|
74
74
|
"context": 500000,
|
|
75
75
|
"output": 500000
|
|
76
76
|
},
|
|
77
|
-
"experimental": {
|
|
78
|
-
"modes": {
|
|
79
|
-
"fast": {
|
|
80
|
-
"cost": {
|
|
81
|
-
"input": 4,
|
|
82
|
-
"output": 12,
|
|
83
|
-
"cache_read": 1
|
|
84
|
-
},
|
|
85
|
-
"provider": {
|
|
86
|
-
"body": {
|
|
87
|
-
"service_tier": "priority"
|
|
88
|
-
}
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
}
|
|
92
|
-
},
|
|
93
77
|
"cost": {
|
|
94
78
|
"input": 2,
|
|
95
79
|
"output": 6,
|
|
@@ -386,6 +370,34 @@
|
|
|
386
370
|
}
|
|
387
371
|
}
|
|
388
372
|
},
|
|
373
|
+
"grok-imagine-image-2.0": {
|
|
374
|
+
"id": "grok-imagine-image-2.0",
|
|
375
|
+
"name": "Grok Imagine Image 2.0",
|
|
376
|
+
"description": "Image model for prompt-driven generation, editing, and visual design workflows",
|
|
377
|
+
"family": "grok",
|
|
378
|
+
"attachment": true,
|
|
379
|
+
"reasoning": false,
|
|
380
|
+
"tool_call": false,
|
|
381
|
+
"temperature": false,
|
|
382
|
+
"release_date": "2026-08-07",
|
|
383
|
+
"last_updated": "2026-08-07",
|
|
384
|
+
"modalities": {
|
|
385
|
+
"input": [
|
|
386
|
+
"text",
|
|
387
|
+
"image",
|
|
388
|
+
"pdf"
|
|
389
|
+
],
|
|
390
|
+
"output": [
|
|
391
|
+
"image",
|
|
392
|
+
"pdf"
|
|
393
|
+
]
|
|
394
|
+
},
|
|
395
|
+
"open_weights": false,
|
|
396
|
+
"limit": {
|
|
397
|
+
"context": 8000,
|
|
398
|
+
"output": 0
|
|
399
|
+
}
|
|
400
|
+
},
|
|
389
401
|
"grok-imagine-image": {
|
|
390
402
|
"id": "grok-imagine-image",
|
|
391
403
|
"name": "Grok Imagine Image",
|
|
@@ -74,12 +74,14 @@ with `store: false`, so no conversation state is kept server-side.
|
|
|
74
74
|
Pass `mode: :completions` to use the legacy Chat Completions API
|
|
75
75
|
instead. Every other provider defaults to `mode: :completions`.
|
|
76
76
|
|
|
77
|
-
A raw context disables
|
|
78
|
-
Pass `retry_budget:` to retry a
|
|
79
|
-
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
77
|
+
A raw context disables retries by default (`retry_budget: 0`).
|
|
78
|
+
Pass `retry_budget:` to retry a request that was rate limited
|
|
79
|
+
(`LLM::RateLimitError`) or timed out (`Timeout::Error`, covering
|
|
80
|
+
`Net::OpenTimeout` and `Net::ReadTimeout`), up to that many times.
|
|
81
|
+
Each retry sleeps a growing interval (2s, 4s, 6s, ...) and notifies
|
|
82
|
+
the stream through
|
|
83
|
+
[`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry-instance_method)
|
|
84
|
+
before trying again. An `LLM::Agent` enables a budget of 5 by
|
|
83
85
|
default, so most users never touch this directly.
|
|
84
86
|
|
|
85
87
|
### Manual loop
|
|
@@ -206,22 +206,22 @@ longer used.
|
|
|
206
206
|
#### Overview
|
|
207
207
|
|
|
208
208
|
[`LLM::Agent.retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#retry_budget-class_method)
|
|
209
|
-
is the maximum number of times an agent retries a
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
configuration. Only a raw
|
|
209
|
+
is the maximum number of times an agent retries a failed request
|
|
210
|
+
before giving up. It is enabled by default at five retries, so
|
|
211
|
+
most agents survive a transient 429 or a dropped connection without
|
|
212
|
+
any configuration. Only a raw
|
|
213
213
|
[`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
|
|
214
214
|
disables it by default.
|
|
215
215
|
|
|
216
216
|
#### How it works
|
|
217
217
|
|
|
218
|
-
When you want to control how many times a
|
|
218
|
+
When you want to control how many times a failed request is
|
|
219
219
|
retried, set the budget with
|
|
220
220
|
[`LLM::Agent.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#set-class_method)
|
|
221
221
|
or per-instance with the `retry_budget:` keyword argument. Each
|
|
222
|
-
retry notifies your stream through `
|
|
222
|
+
retry notifies your stream through `on_retry` and sleeps a
|
|
223
223
|
growing interval (2s, 4s, 6s, ...). Once the budget is spent, the
|
|
224
|
-
agent re-raises the
|
|
224
|
+
agent re-raises the error instead of blocking forever:
|
|
225
225
|
|
|
226
226
|
```ruby
|
|
227
227
|
class Chat < LLM::Agent
|
|
@@ -244,9 +244,10 @@ of hanging.
|
|
|
244
244
|
|
|
245
245
|
#### Notes
|
|
246
246
|
|
|
247
|
-
The retry budget applies to rate-limited requests
|
|
248
|
-
|
|
249
|
-
|
|
247
|
+
The retry budget applies to rate-limited requests
|
|
248
|
+
(`LLM::RateLimitError`) and timeouts (`Timeout::Error`, covering
|
|
249
|
+
`Net::OpenTimeout` and `Net::ReadTimeout`), other errors are never
|
|
250
|
+
retried. The budget defaults to five for agents, while a raw
|
|
250
251
|
[`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
|
|
251
252
|
defaults to zero (`retry_budget: 0`). A 429 is refused before any
|
|
252
253
|
content streams, so retrying the same request loses nothing. Pass
|
|
@@ -67,8 +67,8 @@ receives tokens as they arrive.
|
|
|
67
67
|
fires when the model requests a tool.
|
|
68
68
|
[`LLM::Stream#on_tool_return`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_tool_return)
|
|
69
69
|
fires when the tool completes.
|
|
70
|
-
[`LLM::Stream#
|
|
71
|
-
fires each time a
|
|
70
|
+
[`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry)
|
|
71
|
+
fires each time a failed request is retried. Compaction hooks
|
|
72
72
|
let you show progress or log what was trimmed. Skill hooks bracket a
|
|
73
73
|
skill's subagent execution:
|
|
74
74
|
[`LLM::Stream#on_skill_call`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_skill_call)
|
|
@@ -120,8 +120,8 @@ class MyStream < LLM::Stream
|
|
|
120
120
|
def on_skill_return(agent, skill, result)
|
|
121
121
|
end
|
|
122
122
|
|
|
123
|
-
# A request was rate limited and will be retried.
|
|
124
|
-
def
|
|
123
|
+
# A request was rate limited or timed out and will be retried.
|
|
124
|
+
def on_retry(error, attempt)
|
|
125
125
|
end
|
|
126
126
|
end
|
|
127
127
|
|
data/lib/llm/active_record.rb
CHANGED
|
@@ -39,7 +39,7 @@ module LLM::ActiveRecord
|
|
|
39
39
|
def self.serialize_context(ctx, format)
|
|
40
40
|
case format
|
|
41
41
|
when :string then ctx.to_json
|
|
42
|
-
when :json, :jsonb then ctx.
|
|
42
|
+
when :json, :jsonb then LLM.json.load(ctx.to_json)
|
|
43
43
|
else raise ArgumentError, "Unknown format: #{format.inspect}"
|
|
44
44
|
end
|
|
45
45
|
end
|
data/lib/llm/agent.rb
CHANGED
data/lib/llm/context.rb
CHANGED
|
@@ -39,6 +39,14 @@ module LLM
|
|
|
39
39
|
include Serializer
|
|
40
40
|
include Deserializer
|
|
41
41
|
|
|
42
|
+
TRY_ERRORS = [
|
|
43
|
+
"LLM::InsufficientQuotaError",
|
|
44
|
+
"LLM::RateLimitError",
|
|
45
|
+
"Net::ReadTimeout",
|
|
46
|
+
"Net::OpenTimeout"
|
|
47
|
+
]
|
|
48
|
+
private_constant :TRY_ERRORS
|
|
49
|
+
|
|
42
50
|
##
|
|
43
51
|
# Returns the set of runtime parameters that
|
|
44
52
|
# configure this context and must never be forwarded
|
|
@@ -476,7 +484,7 @@ module LLM
|
|
|
476
484
|
# Returns the model a Context is actively using
|
|
477
485
|
# @return [String]
|
|
478
486
|
def model
|
|
479
|
-
messages.find(&:assistant?)&.model || @params[:model]
|
|
487
|
+
messages.find(&:assistant?)&.model || @params[:model] || @llm.default_model
|
|
480
488
|
end
|
|
481
489
|
|
|
482
490
|
##
|
|
@@ -559,23 +567,30 @@ module LLM
|
|
|
559
567
|
|
|
560
568
|
##
|
|
561
569
|
##
|
|
562
|
-
# Runs a network call, retrying it
|
|
563
|
-
#
|
|
564
|
-
#
|
|
565
|
-
#
|
|
566
|
-
#
|
|
567
|
-
#
|
|
570
|
+
# Runs a network call, retrying it when the request is rate limited
|
|
571
|
+
# ({LLM::RateLimitError}) or times out (`Timeout::Error`, which covers
|
|
572
|
+
# `Net::OpenTimeout` and `Net::ReadTimeout`), up to the retry budget.
|
|
573
|
+
# Each retry notifies the stream and sleeps a growing interval
|
|
574
|
+
# (2s, 4s, 6s, ...) rather than the server's `retry_after`. A 429 is
|
|
575
|
+
# refused before any content streams, so retrying the same request
|
|
576
|
+
# loses nothing. The bare `retry` below re-runs the method body while
|
|
577
|
+
# `attempts ||= 0` keeps the count across attempts.
|
|
568
578
|
# @api private
|
|
569
579
|
# @return [Object]
|
|
570
580
|
def try
|
|
571
581
|
attempts ||= 0
|
|
572
582
|
yield
|
|
573
|
-
rescue
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
583
|
+
rescue => ex
|
|
584
|
+
case ex.class.to_s
|
|
585
|
+
when *TRY_ERRORS
|
|
586
|
+
raise if attempts >= retry_budget
|
|
587
|
+
attempts += 1
|
|
588
|
+
stream.on_retry(ex, attempts)
|
|
589
|
+
sleep 2.0 * attempts
|
|
590
|
+
retry
|
|
591
|
+
else
|
|
592
|
+
raise(ex)
|
|
593
|
+
end
|
|
579
594
|
end
|
|
580
595
|
|
|
581
596
|
# Executes a turn through the Responses API.
|
data/lib/llm/cost.rb
CHANGED
|
@@ -6,6 +6,14 @@
|
|
|
6
6
|
# output, input audio, output audio, input image, cache read, cache write,
|
|
7
7
|
# and reasoning costs separately and can return the total.
|
|
8
8
|
class LLM::Cost
|
|
9
|
+
##
|
|
10
|
+
# Build a zero-valued cost breakdown. Every component
|
|
11
|
+
# is nil (treated as no cost), so the total is 0.
|
|
12
|
+
# @return [LLM::Cost]
|
|
13
|
+
def self.zero
|
|
14
|
+
new
|
|
15
|
+
end
|
|
16
|
+
|
|
9
17
|
##
|
|
10
18
|
# Build a cost breakdown from token usage and model pricing.
|
|
11
19
|
# @param [LLM::Context] ctx
|
|
@@ -13,6 +21,11 @@ class LLM::Cost
|
|
|
13
21
|
# @return [LLM::Cost]
|
|
14
22
|
def self.from(ctx)
|
|
15
23
|
pricing = LLM.registry_for(ctx.llm).cost(model: ctx.model)
|
|
24
|
+
##
|
|
25
|
+
# A model may have no known pricing (eg OpenRouter's
|
|
26
|
+
# `openrouter/auto` auto-router). Fall back to a zero
|
|
27
|
+
# cost rather than crashing on a nil pricing.
|
|
28
|
+
return zero if pricing.nil?
|
|
16
29
|
usage = ctx.usage
|
|
17
30
|
output = usage.output_tokens - usage.reasoning_tokens
|
|
18
31
|
input = usage.input_tokens - usage.cache_read_tokens
|
|
@@ -32,13 +32,15 @@ class LLM::Function
|
|
|
32
32
|
@pid = Kernel.fork do
|
|
33
33
|
##
|
|
34
34
|
# The child inherits the parent's terminal. When
|
|
35
|
-
# the runtime runs under a curses REPL,
|
|
36
|
-
# tool writing
|
|
37
|
-
#
|
|
38
|
-
#
|
|
39
|
-
#
|
|
40
|
-
#
|
|
41
|
-
#
|
|
35
|
+
# the runtime runs under a curses REPL, a forked
|
|
36
|
+
# tool reading or writing the tty would steal the
|
|
37
|
+
# user's input or clobber the parent's display.
|
|
38
|
+
# Point all three standard streams at null so the
|
|
39
|
+
# child keeps off the user's terminal entirely. A
|
|
40
|
+
# tool that genuinely needs the terminal can reopen
|
|
41
|
+
# it via /dev/tty; the tty fd stays available to
|
|
42
|
+
# the child.
|
|
43
|
+
$stdin.reopen(File::NULL)
|
|
42
44
|
$stdout.reopen(File::NULL)
|
|
43
45
|
$stderr.reopen(File::NULL)
|
|
44
46
|
Fork::Job.new(@function, @ch).call
|
|
@@ -83,8 +85,10 @@ class LLM::Function
|
|
|
83
85
|
@tracer&.on_tool_finish(result:, span: @span)
|
|
84
86
|
result
|
|
85
87
|
ensure
|
|
86
|
-
|
|
87
|
-
|
|
88
|
+
if @guarded.nil?
|
|
89
|
+
reap
|
|
90
|
+
[@ch.control, @ch.result].each { _1.close unless _1.closed? } if @ch
|
|
91
|
+
end
|
|
88
92
|
end
|
|
89
93
|
alias_method :value, :wait
|
|
90
94
|
|
|
@@ -97,7 +101,7 @@ class LLM::Function
|
|
|
97
101
|
private
|
|
98
102
|
|
|
99
103
|
def reap
|
|
100
|
-
return if @waited
|
|
104
|
+
return if @waited || @guarded || !@pid
|
|
101
105
|
::Process.waitpid(@pid)
|
|
102
106
|
@waited = true
|
|
103
107
|
rescue Errno::ECHILD
|
data/lib/llm/provider.rb
CHANGED
|
@@ -17,7 +17,12 @@ class LLM::Provider
|
|
|
17
17
|
# @param [Integer] port
|
|
18
18
|
# The port number
|
|
19
19
|
# @param [Integer] timeout
|
|
20
|
-
# The number of seconds to wait for a response
|
|
20
|
+
# The number of seconds to wait for a response. Also serves as the
|
|
21
|
+
# default read timeout.
|
|
22
|
+
# @param [Integer, nil] read_timeout
|
|
23
|
+
# The number of seconds to wait for a response. Defaults to `timeout`.
|
|
24
|
+
# @param [Integer] connect_timeout
|
|
25
|
+
# The number of seconds to wait for a TCP connection to open.
|
|
21
26
|
# @param [Boolean] ssl
|
|
22
27
|
# Whether to use SSL for the connection
|
|
23
28
|
# @param [String] base_path
|
|
@@ -27,17 +32,27 @@ class LLM::Provider
|
|
|
27
32
|
# Requires the net-http-persistent gem.
|
|
28
33
|
# @param [LLM::Transport, Class, nil] transport
|
|
29
34
|
# Optional override with any {LLM::Transport} instance or subclass.
|
|
30
|
-
def initialize(key:, host:, port: 443, timeout:
|
|
35
|
+
def initialize(key:, host:, port: 443, timeout: 600, read_timeout: nil, connect_timeout: 5, ssl: true, base_path: "", persistent: false, transport: nil)
|
|
31
36
|
@key = key
|
|
32
37
|
@host = host
|
|
33
38
|
@port = port
|
|
34
|
-
@
|
|
39
|
+
@read_timeout = read_timeout || timeout
|
|
40
|
+
@timeout = @read_timeout
|
|
41
|
+
@connect_timeout = connect_timeout
|
|
35
42
|
@ssl = ssl
|
|
36
43
|
@base_path = LLM::Utils.normalize_base_path(base_path)
|
|
37
44
|
@base_uri = URI("#{ssl ? "https" : "http"}://#{host}:#{port}/")
|
|
38
45
|
@headers = {"User-Agent" => "llm.rb v#{LLM::VERSION}"}
|
|
39
|
-
@transport = LLM::Transport::Utils.resolve_transport(host:, port:, timeout:, ssl:, transport:, persistent:)
|
|
40
46
|
@monitor = Monitor.new
|
|
47
|
+
@transport = LLM::Transport::Utils.resolve_transport(
|
|
48
|
+
host:,
|
|
49
|
+
port:,
|
|
50
|
+
timeout: @read_timeout,
|
|
51
|
+
connect_timeout: @connect_timeout,
|
|
52
|
+
ssl:,
|
|
53
|
+
transport:,
|
|
54
|
+
persistent:
|
|
55
|
+
)
|
|
41
56
|
end
|
|
42
57
|
|
|
43
58
|
##
|
|
@@ -251,13 +266,19 @@ class LLM::Provider
|
|
|
251
266
|
# Add one or more headers to all requests
|
|
252
267
|
# @example
|
|
253
268
|
# llm = LLM.openai(key: ENV["KEY"])
|
|
254
|
-
# llm.with(
|
|
255
|
-
# llm.with(
|
|
269
|
+
# llm.with("OpenAI-Organization" => ENV["ORG"])
|
|
270
|
+
# llm.with("OpenAI-Project" => ENV["PROJECT"])
|
|
256
271
|
# @param [Hash<String,String>] headers
|
|
257
272
|
# One or more headers
|
|
273
|
+
# @note
|
|
274
|
+
# For backwards compatibility, headers can be
|
|
275
|
+
# provided via the `headers:` keyword argument,
|
|
276
|
+
# or provided directly as a Hash without the
|
|
277
|
+
# `headers:` key namespace.
|
|
258
278
|
# @return [LLM::Provider]
|
|
259
279
|
# Returns self
|
|
260
|
-
def with(headers
|
|
280
|
+
def with(**headers)
|
|
281
|
+
headers = headers.merge(headers.delete(:headers) || {})
|
|
261
282
|
lock do
|
|
262
283
|
tap { @headers.merge!(headers) }
|
|
263
284
|
end
|
|
@@ -412,7 +433,7 @@ class LLM::Provider
|
|
|
412
433
|
"#{@base_path}#{suffix}"
|
|
413
434
|
end
|
|
414
435
|
|
|
415
|
-
attr_reader :base_uri, :host, :port, :timeout, :ssl, :transport
|
|
436
|
+
attr_reader :base_uri, :host, :port, :timeout, :read_timeout, :connect_timeout, :ssl, :transport
|
|
416
437
|
|
|
417
438
|
##
|
|
418
439
|
# The headers to include with a request
|
|
@@ -159,7 +159,7 @@ module LLM
|
|
|
159
159
|
end
|
|
160
160
|
|
|
161
161
|
def normalize_complete_params(params)
|
|
162
|
-
params = {role: :user, model: default_model, max_tokens: 1024}.merge!(params)
|
|
162
|
+
params = {role: :user, model: params.delete(:model) || default_model, max_tokens: 1024}.merge!(params)
|
|
163
163
|
tools = resolve_tools(params.delete(:tools))
|
|
164
164
|
params = [params, adapt_tools(tools)].inject({}, &:merge!).compact
|
|
165
165
|
role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
|
|
@@ -51,7 +51,7 @@ class LLM::Bedrock
|
|
|
51
51
|
# @param [String] host
|
|
52
52
|
# @return [LLM::Transport]
|
|
53
53
|
def build_transport(host)
|
|
54
|
-
transport.class.new(host:, port: 443, timeout:, ssl: true)
|
|
54
|
+
transport.class.new(host:, port: 443, timeout:, connect_timeout:, ssl: true)
|
|
55
55
|
end
|
|
56
56
|
|
|
57
57
|
##
|
|
@@ -102,7 +102,7 @@ class LLM::Bedrock
|
|
|
102
102
|
end
|
|
103
103
|
end
|
|
104
104
|
|
|
105
|
-
[:timeout, :tracer, :transport].each do |m|
|
|
105
|
+
[:timeout, :connect_timeout, :tracer, :transport].each do |m|
|
|
106
106
|
define_method(m) { @provider.send(m) }
|
|
107
107
|
end
|
|
108
108
|
end
|
|
@@ -210,7 +210,7 @@ module LLM
|
|
|
210
210
|
end
|
|
211
211
|
|
|
212
212
|
def normalize_complete_params(params)
|
|
213
|
-
params = {role: :user, model: default_model, max_tokens: 2048}.merge!(params)
|
|
213
|
+
params = {role: :user, model: params.delete(:model) || default_model, max_tokens: 2048}.merge!(params)
|
|
214
214
|
tools = resolve_tools(params.delete(:tools))
|
|
215
215
|
params = [params, adapt_schema(params), adapt_tools(tools)].inject({}, &:merge!).compact
|
|
216
216
|
role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
|
data/lib/llm/providers/google.rb
CHANGED
|
@@ -202,7 +202,7 @@ module LLM
|
|
|
202
202
|
|
|
203
203
|
def normalize_complete_params(params)
|
|
204
204
|
except = %i[role model messages stream]
|
|
205
|
-
params = {role: :user, model: default_model}.merge!(params)
|
|
205
|
+
params = {role: :user, model: params.delete(:model) || default_model}.merge!(params)
|
|
206
206
|
tools = resolve_tools(params.delete(:tools))
|
|
207
207
|
config = adapt_generation_config(params.except(*except))
|
|
208
208
|
params = [params.except(:schema), config, adapt_tools(tools)].inject({}, &:merge!).compact
|
data/lib/llm/providers/ollama.rb
CHANGED
|
@@ -129,7 +129,7 @@ module LLM
|
|
|
129
129
|
end
|
|
130
130
|
|
|
131
131
|
def normalize_complete_params(params)
|
|
132
|
-
params = {role: :user, model: default_model, stream: true}.merge!(params)
|
|
132
|
+
params = {role: :user, model: params.delete(:model) || default_model, stream: true}.merge!(params)
|
|
133
133
|
tools = resolve_tools(params.delete(:tools))
|
|
134
134
|
params = [params, {format: params[:schema]}, adapt_tools(tools)].inject({}, &:merge!).compact
|
|
135
135
|
role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
|
|
@@ -35,7 +35,8 @@ class LLM::OpenAI
|
|
|
35
35
|
# When given an object a provider does not understand
|
|
36
36
|
# @return [LLM::Response]
|
|
37
37
|
def create(prompt, params = {})
|
|
38
|
-
params = {
|
|
38
|
+
params = {}.merge!(params)
|
|
39
|
+
params = {role: :user, model: params.delete(:model) || @provider.default_model}.merge!(params)
|
|
39
40
|
role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
|
|
40
41
|
tools = resolve_tools(params.delete(:tools))
|
|
41
42
|
params = [
|
data/lib/llm/providers/openai.rb
CHANGED
|
@@ -219,7 +219,7 @@ module LLM
|
|
|
219
219
|
end
|
|
220
220
|
|
|
221
221
|
def normalize_complete_params(params)
|
|
222
|
-
params = {role: :user, model: default_model}.merge!(params)
|
|
222
|
+
params = {role: :user, model: params.delete(:model) || default_model}.merge!(params)
|
|
223
223
|
tools = resolve_tools(params.delete(:tools))
|
|
224
224
|
params = [params, adapt_schema(params), adapt_tools(tools)].inject({}, &:merge!).compact
|
|
225
225
|
role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "openai" unless defined?(LLM::OpenAI)
|
|
4
|
+
|
|
5
|
+
module LLM
|
|
6
|
+
##
|
|
7
|
+
# The OpenRouter class implements a provider for
|
|
8
|
+
# [OpenRouter](https://openrouter.ai) through its OpenAI-compatible API.
|
|
9
|
+
#
|
|
10
|
+
# @example
|
|
11
|
+
# #!/usr/bin/env ruby
|
|
12
|
+
# require "llm"
|
|
13
|
+
#
|
|
14
|
+
# llm = LLM.openrouter(key: ENV["KEY"])
|
|
15
|
+
# ctx = LLM::Context.new(llm)
|
|
16
|
+
# ctx.talk "Hello"
|
|
17
|
+
class OpenRouter < OpenAI
|
|
18
|
+
HOST = "openrouter.ai"
|
|
19
|
+
BASE_PATH = "/api/v1"
|
|
20
|
+
|
|
21
|
+
##
|
|
22
|
+
# @param key (see LLM::Provider#initialize)
|
|
23
|
+
# @param host (see LLM::Provider#initialize)
|
|
24
|
+
# @param base_path (see LLM::Provider#initialize)
|
|
25
|
+
# @return [LLM::OpenRouter]
|
|
26
|
+
def initialize(host: HOST, base_path: BASE_PATH, **)
|
|
27
|
+
super
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
##
|
|
31
|
+
# @return [Symbol]
|
|
32
|
+
# Returns the provider's name
|
|
33
|
+
def name
|
|
34
|
+
:openrouter
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
##
|
|
38
|
+
# Provides an embedding.
|
|
39
|
+
# @see https://openrouter.ai/docs/api/api-reference/embeddings/create-embeddings OpenRouter docs
|
|
40
|
+
# @param input (see LLM::Provider#embed)
|
|
41
|
+
# @param model (see LLM::Provider#embed)
|
|
42
|
+
# @param params (see LLM::Provider#embed)
|
|
43
|
+
# @raise (see LLM::Provider#request)
|
|
44
|
+
# @return (see LLM::Provider#embed)
|
|
45
|
+
def embed(input, model: "openai/text-embedding-3-small", **params)
|
|
46
|
+
super
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
##
|
|
50
|
+
# @raise [NotImplementedError]
|
|
51
|
+
def files
|
|
52
|
+
raise NotImplementedError
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
##
|
|
56
|
+
# @raise [NotImplementedError]
|
|
57
|
+
def images
|
|
58
|
+
raise NotImplementedError
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
##
|
|
62
|
+
# @raise [NotImplementedError]
|
|
63
|
+
def audio
|
|
64
|
+
raise NotImplementedError
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
##
|
|
68
|
+
# @raise [NotImplementedError]
|
|
69
|
+
def moderations
|
|
70
|
+
raise NotImplementedError
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
##
|
|
74
|
+
# @raise [NotImplementedError]
|
|
75
|
+
def vector_stores
|
|
76
|
+
raise NotImplementedError
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
##
|
|
80
|
+
# Returns the default model for chat completions
|
|
81
|
+
# @see https://openrouter.ai/docs/guides/routing/routers/auto-router OpenRouter Auto Router
|
|
82
|
+
# @return [String]
|
|
83
|
+
def default_model
|
|
84
|
+
"openrouter/auto"
|
|
85
|
+
end
|
|
86
|
+
end
|
|
87
|
+
end
|
data/lib/llm/repl/buffer.rb
CHANGED
|
@@ -147,15 +147,15 @@ class LLM::Repl
|
|
|
147
147
|
if token == "\n"
|
|
148
148
|
rows << []
|
|
149
149
|
elsif token.match?(/\s/)
|
|
150
|
-
if sum(rows.last) + token
|
|
150
|
+
if sum(rows.last) + Node.width(token) <= width
|
|
151
151
|
rows.last << Node.new(token, attrs)
|
|
152
152
|
end
|
|
153
|
-
elsif sum(rows.last) + token
|
|
153
|
+
elsif sum(rows.last) + Node.width(token) > width
|
|
154
154
|
if sum(rows.last) == 0
|
|
155
155
|
##
|
|
156
156
|
# A single word wider than the row: break it
|
|
157
|
-
# every width
|
|
158
|
-
token
|
|
157
|
+
# every width columns.
|
|
158
|
+
slices(token, width).each do |piece|
|
|
159
159
|
rows.last << Node.new(piece, attrs)
|
|
160
160
|
if sum(rows.last) >= width
|
|
161
161
|
rows << []
|
|
@@ -190,7 +190,22 @@ class LLM::Repl
|
|
|
190
190
|
##
|
|
191
191
|
# @api private
|
|
192
192
|
def sum(row)
|
|
193
|
-
row.sum { _1
|
|
193
|
+
row.sum { _1.size }
|
|
194
|
+
end
|
|
195
|
+
|
|
196
|
+
##
|
|
197
|
+
# Splits a single word into pieces that each fit within the
|
|
198
|
+
# buffer's width, measured in display columns.
|
|
199
|
+
# @api private
|
|
200
|
+
def slices(token, width)
|
|
201
|
+
pieces = []
|
|
202
|
+
remaining = token
|
|
203
|
+
until remaining.empty?
|
|
204
|
+
piece = Node.slice(remaining, width)
|
|
205
|
+
pieces << piece
|
|
206
|
+
remaining = remaining[piece.length..] || +""
|
|
207
|
+
end
|
|
208
|
+
pieces
|
|
194
209
|
end
|
|
195
210
|
end
|
|
196
211
|
end
|
data/lib/llm/repl/input.rb
CHANGED
|
@@ -207,7 +207,8 @@ class LLM::Repl
|
|
|
207
207
|
def cursor_pos
|
|
208
208
|
scroll!
|
|
209
209
|
row, col = @cursor
|
|
210
|
-
col
|
|
210
|
+
col = Node.width(@rows[row].chars[0...col].map(&:to_s).join)
|
|
211
|
+
col += Node.width(prompt) if row.zero?
|
|
211
212
|
[row - @scroll, col]
|
|
212
213
|
end
|
|
213
214
|
|
|
@@ -435,7 +436,7 @@ class LLM::Repl
|
|
|
435
436
|
if char == "\n"
|
|
436
437
|
@rows.insert(row + 1, Row.new(:newline))
|
|
437
438
|
@cursor = [row + 1, 0]
|
|
438
|
-
elsif current_line
|
|
439
|
+
elsif Node.width(current_line) >= Curses.cols
|
|
439
440
|
if char == " "
|
|
440
441
|
@rows.insert(row + 1, Row.new(:space))
|
|
441
442
|
@cursor = [row + 1, 0]
|
|
@@ -510,8 +511,8 @@ class LLM::Repl
|
|
|
510
511
|
##
|
|
511
512
|
# Only the first row shares its columns with the prompt,
|
|
512
513
|
# so it has fewer columns left for text.
|
|
513
|
-
width = Curses.cols - (row.equal?(@rows.first) ? prompt
|
|
514
|
-
if row.chars.any? and row.to_s
|
|
514
|
+
width = Curses.cols - (row.equal?(@rows.first) ? Node.width(prompt) : 0)
|
|
515
|
+
if row.chars.any? and Node.width(row.to_s) + 1 > width
|
|
515
516
|
wrap(row)
|
|
516
517
|
end
|
|
517
518
|
@rows.last.chars << Char.new(char)
|
|
@@ -23,7 +23,10 @@ class LLM::Repl::Markdown
|
|
|
23
23
|
emit("| ", attrs)
|
|
24
24
|
row.each_with_index do |chunks, i|
|
|
25
25
|
width = widths[i]
|
|
26
|
-
chunks.each
|
|
26
|
+
chunks.each do |c|
|
|
27
|
+
text = c[:text].to_s
|
|
28
|
+
emit(text + (" " * (width - Node.width(text))), c[:attrs])
|
|
29
|
+
end
|
|
27
30
|
emit(" | ", attrs) unless i == row.size - 1
|
|
28
31
|
end
|
|
29
32
|
emit(" |", attrs)
|
|
@@ -77,7 +80,7 @@ class LLM::Repl::Markdown
|
|
|
77
80
|
return [] if rows.empty?
|
|
78
81
|
cols = rows.first.size
|
|
79
82
|
(0...cols).map do |i|
|
|
80
|
-
rows.map { |r| r[i].map { _1[:text] }.join
|
|
83
|
+
rows.map { |r| Node.width(r[i].map { _1[:text] }.join) }.max
|
|
81
84
|
end
|
|
82
85
|
end
|
|
83
86
|
end
|