llm.rb 15.0.2 → 15.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +161 -1
  3. data/README.md +134 -85
  4. data/bin/llm.rb +38 -3
  5. data/data/bedrock.json +250 -0
  6. data/data/deepinfra.json +162 -0
  7. data/data/deepseek.json +50 -0
  8. data/data/google.json +19 -21
  9. data/data/openrouter.json +14280 -0
  10. data/data/xai.json +28 -16
  11. data/docs/deepdive/advanced/context.md +8 -6
  12. data/docs/deepdive/fundamentals/agents.md +11 -10
  13. data/docs/deepdive/fundamentals/stream.md +4 -4
  14. data/lib/llm/active_record.rb +1 -1
  15. data/lib/llm/agent.rb +6 -0
  16. data/lib/llm/context.rb +28 -13
  17. data/lib/llm/cost.rb +13 -0
  18. data/lib/llm/function/fork/task.rb +14 -10
  19. data/lib/llm/provider.rb +29 -8
  20. data/lib/llm/providers/anthropic.rb +1 -1
  21. data/lib/llm/providers/bedrock/models.rb +2 -2
  22. data/lib/llm/providers/bedrock.rb +1 -1
  23. data/lib/llm/providers/google.rb +1 -1
  24. data/lib/llm/providers/ollama.rb +1 -1
  25. data/lib/llm/providers/openai/responses.rb +2 -1
  26. data/lib/llm/providers/openai.rb +1 -1
  27. data/lib/llm/providers/openrouter.rb +87 -0
  28. data/lib/llm/repl/buffer.rb +20 -5
  29. data/lib/llm/repl/input.rb +5 -4
  30. data/lib/llm/repl/markdown/table.rb +5 -2
  31. data/lib/llm/repl/node.rb +26 -1
  32. data/lib/llm/repl/stream.rb +30 -3
  33. data/lib/llm/repl/window.rb +7 -7
  34. data/lib/llm/repl.rb +1 -0
  35. data/lib/llm/skill.rb +7 -1
  36. data/lib/llm/stream.rb +8 -3
  37. data/lib/llm/tools/git.rb +2 -2
  38. data/lib/llm/tools/mkdir.rb +2 -2
  39. data/lib/llm/tools/rg.rb +2 -2
  40. data/lib/llm/tools/ruby.rb +2 -2
  41. data/lib/llm/tools/shell.rb +2 -2
  42. data/lib/llm/tools/utils.rb +1 -1
  43. data/lib/llm/transport/curb.rb +5 -3
  44. data/lib/llm/transport/http.rb +5 -2
  45. data/lib/llm/transport/persistent_http.rb +6 -4
  46. data/lib/llm/transport/utils.rb +8 -6
  47. data/lib/llm/version.rb +1 -1
  48. data/lib/llm.rb +16 -1
  49. data/llm.gemspec +2 -1
  50. metadata +19 -3
data/data/xai.json CHANGED
@@ -74,22 +74,6 @@
74
74
  "context": 500000,
75
75
  "output": 500000
76
76
  },
77
- "experimental": {
78
- "modes": {
79
- "fast": {
80
- "cost": {
81
- "input": 4,
82
- "output": 12,
83
- "cache_read": 1
84
- },
85
- "provider": {
86
- "body": {
87
- "service_tier": "priority"
88
- }
89
- }
90
- }
91
- }
92
- },
93
77
  "cost": {
94
78
  "input": 2,
95
79
  "output": 6,
@@ -386,6 +370,34 @@
386
370
  }
387
371
  }
388
372
  },
373
+ "grok-imagine-image-2.0": {
374
+ "id": "grok-imagine-image-2.0",
375
+ "name": "Grok Imagine Image 2.0",
376
+ "description": "Image model for prompt-driven generation, editing, and visual design workflows",
377
+ "family": "grok",
378
+ "attachment": true,
379
+ "reasoning": false,
380
+ "tool_call": false,
381
+ "temperature": false,
382
+ "release_date": "2026-08-07",
383
+ "last_updated": "2026-08-07",
384
+ "modalities": {
385
+ "input": [
386
+ "text",
387
+ "image",
388
+ "pdf"
389
+ ],
390
+ "output": [
391
+ "image",
392
+ "pdf"
393
+ ]
394
+ },
395
+ "open_weights": false,
396
+ "limit": {
397
+ "context": 8000,
398
+ "output": 0
399
+ }
400
+ },
389
401
  "grok-imagine-image": {
390
402
  "id": "grok-imagine-image",
391
403
  "name": "Grok Imagine Image",
@@ -74,12 +74,14 @@ with `store: false`, so no conversation state is kept server-side.
74
74
  Pass `mode: :completions` to use the legacy Chat Completions API
75
75
  instead. Every other provider defaults to `mode: :completions`.
76
76
 
77
- A raw context disables rate-limit retries by default (`retry_budget: 0`).
78
- Pass `retry_budget:` to retry a rate-limited request up to that many
79
- times. Each retry sleeps a growing interval (2s, 4s, 6s, ...) and
80
- notifies the stream through
81
- [`LLM::Stream#on_rate_limit`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_rate_limit-instance_method)
82
- before trying again. An `LLM::Agent` enables a budget of 3 by
77
+ A raw context disables retries by default (`retry_budget: 0`).
78
+ Pass `retry_budget:` to retry a request that was rate limited
79
+ (`LLM::RateLimitError`) or timed out (`Timeout::Error`, covering
80
+ `Net::OpenTimeout` and `Net::ReadTimeout`), up to that many times.
81
+ Each retry sleeps a growing interval (2s, 4s, 6s, ...) and notifies
82
+ the stream through
83
+ [`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry-instance_method)
84
+ before trying again. An `LLM::Agent` enables a budget of 5 by
83
85
  default, so most users never touch this directly.
84
86
 
85
87
  ### Manual loop
@@ -206,22 +206,22 @@ longer used.
206
206
  #### Overview
207
207
 
208
208
  [`LLM::Agent.retry_budget`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#retry_budget-class_method)
209
- is the maximum number of times an agent retries a rate-limited
210
- request before giving up. It is enabled by default at three
211
- retries, so most agents survive a transient 429 without any
212
- configuration. Only a raw
209
+ is the maximum number of times an agent retries a failed request
210
+ before giving up. It is enabled by default at five retries, so
211
+ most agents survive a transient 429 or a dropped connection without
212
+ any configuration. Only a raw
213
213
  [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
214
214
  disables it by default.
215
215
 
216
216
  #### How it works
217
217
 
218
- When you want to control how many times a rate-limited request is
218
+ When you want to control how many times a failed request is
219
219
  retried, set the budget with
220
220
  [`LLM::Agent.set`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#set-class_method)
221
221
  or per-instance with the `retry_budget:` keyword argument. Each
222
- retry notifies your stream through `on_rate_limit` and sleeps a
222
+ retry notifies your stream through `on_retry` and sleeps a
223
223
  growing interval (2s, 4s, 6s, ...). Once the budget is spent, the
224
- agent re-raises the rate-limit error instead of blocking forever:
224
+ agent re-raises the error instead of blocking forever:
225
225
 
226
226
  ```ruby
227
227
  class Chat < LLM::Agent
@@ -244,9 +244,10 @@ of hanging.
244
244
 
245
245
  #### Notes
246
246
 
247
- The retry budget applies to rate-limited requests only, other
248
- errors are never retried. The budget defaults to five for agents,
249
- while a raw
247
+ The retry budget applies to rate-limited requests
248
+ (`LLM::RateLimitError`) and timeouts (`Timeout::Error`, covering
249
+ `Net::OpenTimeout` and `Net::ReadTimeout`), other errors are never
250
+ retried. The budget defaults to five for agents, while a raw
250
251
  [`LLM::Context`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html)
251
252
  defaults to zero (`retry_budget: 0`). A 429 is refused before any
252
253
  content streams, so retrying the same request loses nothing. Pass
@@ -67,8 +67,8 @@ receives tokens as they arrive.
67
67
  fires when the model requests a tool.
68
68
  [`LLM::Stream#on_tool_return`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_tool_return)
69
69
  fires when the tool completes.
70
- [`LLM::Stream#on_rate_limit`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_rate_limit)
71
- fires each time a rate-limited request is retried. Compaction hooks
70
+ [`LLM::Stream#on_retry`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_retry)
71
+ fires each time a failed request is retried. Compaction hooks
72
72
  let you show progress or log what was trimmed. Skill hooks bracket a
73
73
  skill's subagent execution:
74
74
  [`LLM::Stream#on_skill_call`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_skill_call)
@@ -120,8 +120,8 @@ class MyStream < LLM::Stream
120
120
  def on_skill_return(agent, skill, result)
121
121
  end
122
122
 
123
- # A request was rate limited and will be retried.
124
- def on_rate_limit(error)
123
+ # A request was rate limited or timed out and will be retried.
124
+ def on_retry(error, attempt)
125
125
  end
126
126
  end
127
127
 
@@ -39,7 +39,7 @@ module LLM::ActiveRecord
39
39
  def self.serialize_context(ctx, format)
40
40
  case format
41
41
  when :string then ctx.to_json
42
- when :json, :jsonb then ctx.to_h
42
+ when :json, :jsonb then LLM.json.load(ctx.to_json)
43
43
  else raise ArgumentError, "Unknown format: #{format.inspect}"
44
44
  end
45
45
  end
data/lib/llm/agent.rb CHANGED
@@ -486,6 +486,12 @@ module LLM
486
486
  @ctx.messages
487
487
  end
488
488
 
489
+ ##
490
+ # @return [Integer]
491
+ def retry_budget
492
+ @ctx.retry_budget
493
+ end
494
+
489
495
  ##
490
496
  # @return [Array<LLM::Function>]
491
497
  def pending_functions
data/lib/llm/context.rb CHANGED
@@ -39,6 +39,14 @@ module LLM
39
39
  include Serializer
40
40
  include Deserializer
41
41
 
42
+ TRY_ERRORS = [
43
+ "LLM::InsufficientQuotaError",
44
+ "LLM::RateLimitError",
45
+ "Net::ReadTimeout",
46
+ "Net::OpenTimeout"
47
+ ]
48
+ private_constant :TRY_ERRORS
49
+
42
50
  ##
43
51
  # Returns the set of runtime parameters that
44
52
  # configure this context and must never be forwarded
@@ -476,7 +484,7 @@ module LLM
476
484
  # Returns the model a Context is actively using
477
485
  # @return [String]
478
486
  def model
479
- messages.find(&:assistant?)&.model || @params[:model]
487
+ messages.find(&:assistant?)&.model || @params[:model] || @llm.default_model
480
488
  end
481
489
 
482
490
  ##
@@ -559,23 +567,30 @@ module LLM
559
567
 
560
568
  ##
561
569
  ##
562
- # Runs a network call, retrying it on {LLM::RateLimitError} up to the
563
- # retry budget. Each retry notifies the stream and sleeps a growing
564
- # interval (2s, 4s, 6s, ...) rather than the server's `retry_after`.
565
- # A 429 is refused before any content streams, so retrying the same
566
- # request loses nothing. The bare `retry` below re-runs the method
567
- # body while `attempts ||= 0` keeps the count across attempts.
570
+ # Runs a network call, retrying it when the request is rate limited
571
+ # ({LLM::RateLimitError}) or times out (`Timeout::Error`, which covers
572
+ # `Net::OpenTimeout` and `Net::ReadTimeout`), up to the retry budget.
573
+ # Each retry notifies the stream and sleeps a growing interval
574
+ # (2s, 4s, 6s, ...) rather than the server's `retry_after`. A 429 is
575
+ # refused before any content streams, so retrying the same request
576
+ # loses nothing. The bare `retry` below re-runs the method body while
577
+ # `attempts ||= 0` keeps the count across attempts.
568
578
  # @api private
569
579
  # @return [Object]
570
580
  def try
571
581
  attempts ||= 0
572
582
  yield
573
- rescue LLM::RateLimitError => error
574
- raise if attempts >= retry_budget
575
- attempts += 1
576
- stream.on_rate_limit(error)
577
- sleep 2.0 * attempts
578
- retry
583
+ rescue => ex
584
+ case ex.class.to_s
585
+ when *TRY_ERRORS
586
+ raise if attempts >= retry_budget
587
+ attempts += 1
588
+ stream.on_retry(ex, attempts)
589
+ sleep 2.0 * attempts
590
+ retry
591
+ else
592
+ raise(ex)
593
+ end
579
594
  end
580
595
 
581
596
  # Executes a turn through the Responses API.
data/lib/llm/cost.rb CHANGED
@@ -6,6 +6,14 @@
6
6
  # output, input audio, output audio, input image, cache read, cache write,
7
7
  # and reasoning costs separately and can return the total.
8
8
  class LLM::Cost
9
+ ##
10
+ # Build a zero-valued cost breakdown. Every component
11
+ # is nil (treated as no cost), so the total is 0.
12
+ # @return [LLM::Cost]
13
+ def self.zero
14
+ new
15
+ end
16
+
9
17
  ##
10
18
  # Build a cost breakdown from token usage and model pricing.
11
19
  # @param [LLM::Context] ctx
@@ -13,6 +21,11 @@ class LLM::Cost
13
21
  # @return [LLM::Cost]
14
22
  def self.from(ctx)
15
23
  pricing = LLM.registry_for(ctx.llm).cost(model: ctx.model)
24
+ ##
25
+ # A model may have no known pricing (eg OpenRouter's
26
+ # `openrouter/auto` auto-router). Fall back to a zero
27
+ # cost rather than crashing on a nil pricing.
28
+ return zero if pricing.nil?
16
29
  usage = ctx.usage
17
30
  output = usage.output_tokens - usage.reasoning_tokens
18
31
  input = usage.input_tokens - usage.cache_read_tokens
@@ -32,13 +32,15 @@ class LLM::Function
32
32
  @pid = Kernel.fork do
33
33
  ##
34
34
  # The child inherits the parent's terminal. When
35
- # the runtime runs under a curses REPL, the forked
36
- # tool writing to the tty would clobber the parent's
37
- # display (a blank screen). Redirect the child's
38
- # stdout/stderr to null so it keeps off the user's
39
- # terminal. A tool that genuinely needs the terminal
40
- # can reopen it via /dev/tty; the tty fd stays
41
- # available to the child.
35
+ # the runtime runs under a curses REPL, a forked
36
+ # tool reading or writing the tty would steal the
37
+ # user's input or clobber the parent's display.
38
+ # Point all three standard streams at null so the
39
+ # child keeps off the user's terminal entirely. A
40
+ # tool that genuinely needs the terminal can reopen
41
+ # it via /dev/tty; the tty fd stays available to
42
+ # the child.
43
+ $stdin.reopen(File::NULL)
42
44
  $stdout.reopen(File::NULL)
43
45
  $stderr.reopen(File::NULL)
44
46
  Fork::Job.new(@function, @ch).call
@@ -83,8 +85,10 @@ class LLM::Function
83
85
  @tracer&.on_tool_finish(result:, span: @span)
84
86
  result
85
87
  ensure
86
- reap
87
- [@ch.control, @ch.result].each { _1.close unless _1.closed? }
88
+ if @guarded.nil?
89
+ reap
90
+ [@ch.control, @ch.result].each { _1.close unless _1.closed? } if @ch
91
+ end
88
92
  end
89
93
  alias_method :value, :wait
90
94
 
@@ -97,7 +101,7 @@ class LLM::Function
97
101
  private
98
102
 
99
103
  def reap
100
- return if @waited
104
+ return if @waited || @guarded || !@pid
101
105
  ::Process.waitpid(@pid)
102
106
  @waited = true
103
107
  rescue Errno::ECHILD
data/lib/llm/provider.rb CHANGED
@@ -17,7 +17,12 @@ class LLM::Provider
17
17
  # @param [Integer] port
18
18
  # The port number
19
19
  # @param [Integer] timeout
20
- # The number of seconds to wait for a response
20
+ # The number of seconds to wait for a response. Also serves as the
21
+ # default read timeout.
22
+ # @param [Integer, nil] read_timeout
23
+ # The number of seconds to wait for a response. Defaults to `timeout`.
24
+ # @param [Integer] connect_timeout
25
+ # The number of seconds to wait for a TCP connection to open.
21
26
  # @param [Boolean] ssl
22
27
  # Whether to use SSL for the connection
23
28
  # @param [String] base_path
@@ -27,17 +32,27 @@ class LLM::Provider
27
32
  # Requires the net-http-persistent gem.
28
33
  # @param [LLM::Transport, Class, nil] transport
29
34
  # Optional override with any {LLM::Transport} instance or subclass.
30
- def initialize(key:, host:, port: 443, timeout: 900, ssl: true, base_path: "", persistent: false, transport: nil)
35
+ def initialize(key:, host:, port: 443, timeout: 600, read_timeout: nil, connect_timeout: 5, ssl: true, base_path: "", persistent: false, transport: nil)
31
36
  @key = key
32
37
  @host = host
33
38
  @port = port
34
- @timeout = timeout
39
+ @read_timeout = read_timeout || timeout
40
+ @timeout = @read_timeout
41
+ @connect_timeout = connect_timeout
35
42
  @ssl = ssl
36
43
  @base_path = LLM::Utils.normalize_base_path(base_path)
37
44
  @base_uri = URI("#{ssl ? "https" : "http"}://#{host}:#{port}/")
38
45
  @headers = {"User-Agent" => "llm.rb v#{LLM::VERSION}"}
39
- @transport = LLM::Transport::Utils.resolve_transport(host:, port:, timeout:, ssl:, transport:, persistent:)
40
46
  @monitor = Monitor.new
47
+ @transport = LLM::Transport::Utils.resolve_transport(
48
+ host:,
49
+ port:,
50
+ timeout: @read_timeout,
51
+ connect_timeout: @connect_timeout,
52
+ ssl:,
53
+ transport:,
54
+ persistent:
55
+ )
41
56
  end
42
57
 
43
58
  ##
@@ -251,13 +266,19 @@ class LLM::Provider
251
266
  # Add one or more headers to all requests
252
267
  # @example
253
268
  # llm = LLM.openai(key: ENV["KEY"])
254
- # llm.with(headers: {"OpenAI-Organization" => ENV["ORG"]})
255
- # llm.with(headers: {"OpenAI-Project" => ENV["PROJECT"]})
269
+ # llm.with("OpenAI-Organization" => ENV["ORG"])
270
+ # llm.with("OpenAI-Project" => ENV["PROJECT"])
256
271
  # @param [Hash<String,String>] headers
257
272
  # One or more headers
273
+ # @note
274
+ # For backwards compatibility, headers can be
275
+ # provided via the `headers:` keyword argument,
276
+ # or provided directly as a Hash without the
277
+ # `headers:` key namespace.
258
278
  # @return [LLM::Provider]
259
279
  # Returns self
260
- def with(headers:)
280
+ def with(**headers)
281
+ headers = headers.merge(headers.delete(:headers) || {})
261
282
  lock do
262
283
  tap { @headers.merge!(headers) }
263
284
  end
@@ -412,7 +433,7 @@ class LLM::Provider
412
433
  "#{@base_path}#{suffix}"
413
434
  end
414
435
 
415
- attr_reader :base_uri, :host, :port, :timeout, :ssl, :transport
436
+ attr_reader :base_uri, :host, :port, :timeout, :read_timeout, :connect_timeout, :ssl, :transport
416
437
 
417
438
  ##
418
439
  # The headers to include with a request
@@ -159,7 +159,7 @@ module LLM
159
159
  end
160
160
 
161
161
  def normalize_complete_params(params)
162
- params = {role: :user, model: default_model, max_tokens: 1024}.merge!(params)
162
+ params = {role: :user, model: params.delete(:model) || default_model, max_tokens: 1024}.merge!(params)
163
163
  tools = resolve_tools(params.delete(:tools))
164
164
  params = [params, adapt_tools(tools)].inject({}, &:merge!).compact
165
165
  role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
@@ -51,7 +51,7 @@ class LLM::Bedrock
51
51
  # @param [String] host
52
52
  # @return [LLM::Transport]
53
53
  def build_transport(host)
54
- transport.class.new(host:, port: 443, timeout:, ssl: true)
54
+ transport.class.new(host:, port: 443, timeout:, connect_timeout:, ssl: true)
55
55
  end
56
56
 
57
57
  ##
@@ -102,7 +102,7 @@ class LLM::Bedrock
102
102
  end
103
103
  end
104
104
 
105
- [:timeout, :tracer, :transport].each do |m|
105
+ [:timeout, :connect_timeout, :tracer, :transport].each do |m|
106
106
  define_method(m) { @provider.send(m) }
107
107
  end
108
108
  end
@@ -210,7 +210,7 @@ module LLM
210
210
  end
211
211
 
212
212
  def normalize_complete_params(params)
213
- params = {role: :user, model: default_model, max_tokens: 2048}.merge!(params)
213
+ params = {role: :user, model: params.delete(:model) || default_model, max_tokens: 2048}.merge!(params)
214
214
  tools = resolve_tools(params.delete(:tools))
215
215
  params = [params, adapt_schema(params), adapt_tools(tools)].inject({}, &:merge!).compact
216
216
  role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
@@ -202,7 +202,7 @@ module LLM
202
202
 
203
203
  def normalize_complete_params(params)
204
204
  except = %i[role model messages stream]
205
- params = {role: :user, model: default_model}.merge!(params)
205
+ params = {role: :user, model: params.delete(:model) || default_model}.merge!(params)
206
206
  tools = resolve_tools(params.delete(:tools))
207
207
  config = adapt_generation_config(params.except(*except))
208
208
  params = [params.except(:schema), config, adapt_tools(tools)].inject({}, &:merge!).compact
@@ -129,7 +129,7 @@ module LLM
129
129
  end
130
130
 
131
131
  def normalize_complete_params(params)
132
- params = {role: :user, model: default_model, stream: true}.merge!(params)
132
+ params = {role: :user, model: params.delete(:model) || default_model, stream: true}.merge!(params)
133
133
  tools = resolve_tools(params.delete(:tools))
134
134
  params = [params, {format: params[:schema]}, adapt_tools(tools)].inject({}, &:merge!).compact
135
135
  role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
@@ -35,7 +35,8 @@ class LLM::OpenAI
35
35
  # When given an object a provider does not understand
36
36
  # @return [LLM::Response]
37
37
  def create(prompt, params = {})
38
- params = {role: :user, model: @provider.default_model}.merge!(params)
38
+ params = {}.merge!(params)
39
+ params = {role: :user, model: params.delete(:model) || @provider.default_model}.merge!(params)
39
40
  role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
40
41
  tools = resolve_tools(params.delete(:tools))
41
42
  params = [
@@ -219,7 +219,7 @@ module LLM
219
219
  end
220
220
 
221
221
  def normalize_complete_params(params)
222
- params = {role: :user, model: default_model}.merge!(params)
222
+ params = {role: :user, model: params.delete(:model) || default_model}.merge!(params)
223
223
  tools = resolve_tools(params.delete(:tools))
224
224
  params = [params, adapt_schema(params), adapt_tools(tools)].inject({}, &:merge!).compact
225
225
  role, stream = params.delete(:role), LLM::Stream.try(params.delete(:stream))
@@ -0,0 +1,87 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "openai" unless defined?(LLM::OpenAI)
4
+
5
+ module LLM
6
+ ##
7
+ # The OpenRouter class implements a provider for
8
+ # [OpenRouter](https://openrouter.ai) through its OpenAI-compatible API.
9
+ #
10
+ # @example
11
+ # #!/usr/bin/env ruby
12
+ # require "llm"
13
+ #
14
+ # llm = LLM.openrouter(key: ENV["KEY"])
15
+ # ctx = LLM::Context.new(llm)
16
+ # ctx.talk "Hello"
17
+ class OpenRouter < OpenAI
18
+ HOST = "openrouter.ai"
19
+ BASE_PATH = "/api/v1"
20
+
21
+ ##
22
+ # @param key (see LLM::Provider#initialize)
23
+ # @param host (see LLM::Provider#initialize)
24
+ # @param base_path (see LLM::Provider#initialize)
25
+ # @return [LLM::OpenRouter]
26
+ def initialize(host: HOST, base_path: BASE_PATH, **)
27
+ super
28
+ end
29
+
30
+ ##
31
+ # @return [Symbol]
32
+ # Returns the provider's name
33
+ def name
34
+ :openrouter
35
+ end
36
+
37
+ ##
38
+ # Provides an embedding.
39
+ # @see https://openrouter.ai/docs/api/api-reference/embeddings/create-embeddings OpenRouter docs
40
+ # @param input (see LLM::Provider#embed)
41
+ # @param model (see LLM::Provider#embed)
42
+ # @param params (see LLM::Provider#embed)
43
+ # @raise (see LLM::Provider#request)
44
+ # @return (see LLM::Provider#embed)
45
+ def embed(input, model: "openai/text-embedding-3-small", **params)
46
+ super
47
+ end
48
+
49
+ ##
50
+ # @raise [NotImplementedError]
51
+ def files
52
+ raise NotImplementedError
53
+ end
54
+
55
+ ##
56
+ # @raise [NotImplementedError]
57
+ def images
58
+ raise NotImplementedError
59
+ end
60
+
61
+ ##
62
+ # @raise [NotImplementedError]
63
+ def audio
64
+ raise NotImplementedError
65
+ end
66
+
67
+ ##
68
+ # @raise [NotImplementedError]
69
+ def moderations
70
+ raise NotImplementedError
71
+ end
72
+
73
+ ##
74
+ # @raise [NotImplementedError]
75
+ def vector_stores
76
+ raise NotImplementedError
77
+ end
78
+
79
+ ##
80
+ # Returns the default model for chat completions
81
+ # @see https://openrouter.ai/docs/guides/routing/routers/auto-router OpenRouter Auto Router
82
+ # @return [String]
83
+ def default_model
84
+ "openrouter/auto"
85
+ end
86
+ end
87
+ end
@@ -147,15 +147,15 @@ class LLM::Repl
147
147
  if token == "\n"
148
148
  rows << []
149
149
  elsif token.match?(/\s/)
150
- if sum(rows.last) + token.length <= width
150
+ if sum(rows.last) + Node.width(token) <= width
151
151
  rows.last << Node.new(token, attrs)
152
152
  end
153
- elsif sum(rows.last) + token.length > width
153
+ elsif sum(rows.last) + Node.width(token) > width
154
154
  if sum(rows.last) == 0
155
155
  ##
156
156
  # A single word wider than the row: break it
157
- # every width characters.
158
- token.scan(/.{1,#{width}}/).each do |piece|
157
+ # every width columns.
158
+ slices(token, width).each do |piece|
159
159
  rows.last << Node.new(piece, attrs)
160
160
  if sum(rows.last) >= width
161
161
  rows << []
@@ -190,7 +190,22 @@ class LLM::Repl
190
190
  ##
191
191
  # @api private
192
192
  def sum(row)
193
- row.sum { _1[:text].to_s.length }
193
+ row.sum { _1.size }
194
+ end
195
+
196
+ ##
197
+ # Splits a single word into pieces that each fit within the
198
+ # buffer's width, measured in display columns.
199
+ # @api private
200
+ def slices(token, width)
201
+ pieces = []
202
+ remaining = token
203
+ until remaining.empty?
204
+ piece = Node.slice(remaining, width)
205
+ pieces << piece
206
+ remaining = remaining[piece.length..] || +""
207
+ end
208
+ pieces
194
209
  end
195
210
  end
196
211
  end
@@ -207,7 +207,8 @@ class LLM::Repl
207
207
  def cursor_pos
208
208
  scroll!
209
209
  row, col = @cursor
210
- col += prompt.size if row.zero?
210
+ col = Node.width(@rows[row].chars[0...col].map(&:to_s).join)
211
+ col += Node.width(prompt) if row.zero?
211
212
  [row - @scroll, col]
212
213
  end
213
214
 
@@ -435,7 +436,7 @@ class LLM::Repl
435
436
  if char == "\n"
436
437
  @rows.insert(row + 1, Row.new(:newline))
437
438
  @cursor = [row + 1, 0]
438
- elsif current_line.size >= Curses.cols
439
+ elsif Node.width(current_line) >= Curses.cols
439
440
  if char == " "
440
441
  @rows.insert(row + 1, Row.new(:space))
441
442
  @cursor = [row + 1, 0]
@@ -510,8 +511,8 @@ class LLM::Repl
510
511
  ##
511
512
  # Only the first row shares its columns with the prompt,
512
513
  # so it has fewer columns left for text.
513
- width = Curses.cols - (row.equal?(@rows.first) ? prompt.size : 0)
514
- if row.chars.any? and row.to_s.length + 1 > width
514
+ width = Curses.cols - (row.equal?(@rows.first) ? Node.width(prompt) : 0)
515
+ if row.chars.any? and Node.width(row.to_s) + 1 > width
515
516
  wrap(row)
516
517
  end
517
518
  @rows.last.chars << Char.new(char)
@@ -23,7 +23,10 @@ class LLM::Repl::Markdown
23
23
  emit("| ", attrs)
24
24
  row.each_with_index do |chunks, i|
25
25
  width = widths[i]
26
- chunks.each { |c| emit(c[:text].ljust(width), c[:attrs]) }
26
+ chunks.each do |c|
27
+ text = c[:text].to_s
28
+ emit(text + (" " * (width - Node.width(text))), c[:attrs])
29
+ end
27
30
  emit(" | ", attrs) unless i == row.size - 1
28
31
  end
29
32
  emit(" |", attrs)
@@ -77,7 +80,7 @@ class LLM::Repl::Markdown
77
80
  return [] if rows.empty?
78
81
  cols = rows.first.size
79
82
  (0...cols).map do |i|
80
- rows.map { |r| r[i].map { _1[:text] }.join.length }.max
83
+ rows.map { |r| Node.width(r[i].map { _1[:text] }.join) }.max
81
84
  end
82
85
  end
83
86
  end