raix 2.0.6 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -71,6 +71,17 @@ module Raix
71
71
  # @option max_tool_calls [Integer] :max_tool_calls Maximum number of tool calls before forcing a text response. Defaults to the configured value.
72
72
  # @return [String|Hash] The completed chat response.
73
73
  def chat_completion(params: {}, loop: false, json: false, raw: false, openai: nil, save_response: true, messages: nil, available_tools: nil, max_tool_calls: nil)
74
+ complete_conversation(params:, loop:, json:, raw:, openai:, save_response:, messages:, available_tools:, max_tool_calls:)
75
+ end
76
+
77
+ # The body of chat_completion. Continuation rounds after a tool call recurse
78
+ # here directly rather than through the public method, which a subclass may
79
+ # override with a different signature (PromptDeclarations does).
80
+ def complete_conversation(params: {}, loop: false, json: false, raw: false, openai: nil, save_response: true, messages: nil, available_tools: nil, max_tool_calls: nil)
81
+ # Work on a copy: defaults are filled in and tool_choice is dropped between
82
+ # rounds, and none of that should leak into a Hash the caller may reuse.
83
+ params = params.dup
84
+
74
85
  # set params to default values if not provided
75
86
  params[:cache_at] ||= cache_at.presence
76
87
  params[:frequency_penalty] ||= frequency_penalty.presence
@@ -79,7 +90,11 @@ module Raix
79
90
  params[:max_completion_tokens] ||= max_completion_tokens.presence || configuration.max_completion_tokens
80
91
  params[:max_tokens] ||= max_tokens.presence || configuration.max_tokens
81
92
  params[:min_p] ||= min_p.presence
82
- params[:prediction] = { type: "content", content: params[:prediction] || prediction } if params[:prediction] || prediction.present?
93
+ if (predicted = params[:prediction] || prediction.presence)
94
+ # Continuation rounds pass an already-wrapped prediction back through
95
+ # here, so only wrap a bare value.
96
+ params[:prediction] = predicted.is_a?(Hash) && predicted.with_indifferent_access[:type] == "content" ? predicted : { type: "content", content: predicted }
97
+ end
83
98
  params[:presence_penalty] ||= presence_penalty.presence
84
99
  params[:provider] ||= provider.presence
85
100
  params[:repetition_penalty] ||= repetition_penalty.presence
@@ -87,7 +102,10 @@ module Raix
87
102
  params[:seed] ||= seed.presence
88
103
  params[:stop] ||= stop.presence
89
104
  params[:temperature] ||= temperature.presence || configuration.temperature
90
- params[:tool_choice] ||= tool_choice.presence
105
+ # A forced tool_choice applies to the first round only. Continuation
106
+ # rounds (depth > 0) leave it unset so the model can answer in text once
107
+ # its tool results are in.
108
+ params[:tool_choice] ||= tool_choice.presence if @tool_loop_depth.to_i.zero?
91
109
  params[:tools] = if available_tools == false
92
110
  nil
93
111
  elsif available_tools.is_a?(Array)
@@ -103,14 +121,10 @@ module Raix
103
121
  json = true if params[:response_format].is_a?(Raix::ResponseFormat)
104
122
 
105
123
  if json
106
- unless openai
107
- params[:provider] ||= {}
108
- params[:provider][:require_parameters] = true
109
- end
110
- if params[:response_format].blank?
111
- params[:response_format] ||= {}
112
- params[:response_format][:type] = "json_object"
113
- end
124
+ # Build fresh nested hashes rather than writing into ones the caller
125
+ # handed us; the dup above is shallow.
126
+ params[:provider] = (params[:provider] || {}).merge(require_parameters: true) unless openai
127
+ params[:response_format] = { type: "json_object" } if params[:response_format].blank?
114
128
  end
115
129
 
116
130
  # Deprecation warning for loop parameter
@@ -121,17 +135,9 @@ module Raix
121
135
  # Set max_tool_calls from parameter or configuration default
122
136
  self.max_tool_calls = max_tool_calls || configuration.max_tool_calls
123
137
 
124
- # Reset stop_tool_calls_and_respond flag
125
- @stop_tool_calls_and_respond = false
126
-
127
138
  # Track tool call count
128
139
  tool_call_count = 0
129
140
 
130
- # Reset the per-completion tool-call counter that FunctionToolAdapter's
131
- # generated wrappers use to enforce max_tool_calls under the RubyLLM
132
- # backend (see #increment_tool_call_count).
133
- @tool_call_count = 0
134
-
135
141
  # set the model to the default if not provided
136
142
  self.model ||= configuration.model
137
143
 
@@ -148,98 +154,118 @@ module Raix
148
154
  # Hooks can modify params and messages for logging, filtering, PII redaction, etc.
149
155
  run_before_completion_hooks(params, messages)
150
156
 
151
- begin
152
- # Snapshot the transcript length before the request. If a generated tool
153
- # halts RubyLLM's loop, force_final_response_after_halt uses this to
154
- # recover exactly the assistant/tool messages appended while executing
155
- # tools and replay them after the original messages — which may not live
156
- # in the transcript at all when `messages:` was passed explicitly. Only
157
- # tools can trigger a halt, so skip the transcript read entirely when
158
- # none are registered.
159
- transcript_size_before = params[:tools].present? ? transcript.flatten.compact.size : 0
157
+ # Each continuation after a tool round recurses with the allowance that is
158
+ # left, so remember the budget the caller actually asked for. That is the
159
+ # number the limit message has to quote.
160
+ @tool_loop_depth = @tool_loop_depth.to_i + 1
161
+ @max_tool_calls_budget = self.max_tool_calls if @tool_loop_depth == 1
162
+
163
+ # Start this loop with a clear stop flag and hand back whatever was set
164
+ # on entry when it returns. A nested chat_completion inside a tool body
165
+ # must neither erase a stop the enclosing loop's tool already requested
166
+ # nor leak its own stop into that loop.
167
+ stop_on_entry = @stop_tool_calls_and_respond
168
+ @stop_tool_calls_and_respond = false
160
169
 
161
- response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
170
+ # True only while parsing the model's final JSON response. The blank-JSON
171
+ # retry below must never fire for a JSON::ParserError raised by a tool
172
+ # body (in this frame or a continuation round), which would re-issue the
173
+ # request and run tools again.
174
+ parsing_response = false
162
175
 
163
- # RubyLLM runs the entire tool-call loop inside chat.complete/#ask. When
164
- # a generated tool halts that loop — because max_tool_calls was exceeded
165
- # or stop_tool_calls_and_respond! was called — the return value is a
166
- # RubyLLM::Tool::Halt rather than a Message. Re-issue one final
167
- # completion with no tools; the shared handling below then returns its
168
- # text/JSON as usual.
169
- if response.is_a?(RubyLLM::Tool::Halt)
170
- response = force_final_response_after_halt(response, params:, openai:, messages:, transcript_size_before:)
171
- end
176
+ begin
177
+ response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
172
178
 
173
179
  retry_count = 0
174
180
  content = nil
175
181
 
176
- # no need for additional processing if streaming
177
- return if stream && response.blank?
182
+ # Nothing came back to process (a streamed request that produced no message).
183
+ return if response.blank?
178
184
 
179
185
  # tuck the full response into a thread local in case needed
180
- Thread.current[:chat_completion_response] = response.is_a?(Hash) ? response.with_indifferent_access : response
186
+ Thread.current[:chat_completion_response] = response.with_indifferent_access
181
187
 
182
188
  # TODO: add a standardized callback hook for usage events
183
189
  # broadcast(:usage_event, usage_subject, self.class.name.to_s, response, premium?)
184
190
 
185
- tool_calls = response.dig("choices", 0, "message", "tool_calls") || []
191
+ # The model's own turn, kept intact (tool calls with their ids and
192
+ # signatures, reasoning details) so it can be replayed verbatim.
193
+ assistant_turn = (response.dig("choices", 0, "message") || {}).with_indifferent_access
194
+ tool_calls = assistant_turn[:tool_calls] || []
186
195
  if tool_calls.any?
187
- tool_call_count += tool_calls.size
188
-
189
- # Check if we've exceeded max_tool_calls
190
- if tool_call_count > self.max_tool_calls
191
- # Add system message about hitting the limit
192
- messages << { role: "system", content: "Maximum tool calls (#{self.max_tool_calls}) exceeded. Please provide a final response to the user without calling any more tools." }
193
-
194
- # Force a final response without tools
195
- params[:tools] = nil
196
- response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
197
-
198
- # Process the final response
199
- content = response.dig("choices", 0, "message", "content")
200
- transcript << { assistant: content } if save_response
201
- return raw ? response : content.to_s.strip
202
- end
203
-
204
- # Dispatch tool calls
205
- tool_calls.each do |tool_call| # TODO: parallelize this?
206
- # dispatch the called function
207
- function_name = tool_call["function"]["name"]
208
- arguments = JSON.parse(tool_call["function"]["arguments"].presence || "{}")
209
- raise "Unauthorized function call: #{function_name}" unless self.class.functions.map { |f| f[:name].to_sym }.include?(function_name.to_sym)
196
+ # Enforce the budget per call rather than per round: a single model
197
+ # response can pack several parallel tool calls, and the ones that
198
+ # still fit under the cap should run.
199
+ allowance = [self.max_tool_calls - tool_call_count, 0].max
200
+ cap_exceeded = tool_calls.size > allowance
201
+ tool_call_count += [tool_calls.size, allowance].min
202
+
203
+ # A call is authorized only if the function is declared on this class
204
+ # AND was offered on this request (the available_tools-filtered set).
205
+ # Declared-only would let a hidden tool through; offered-only would
206
+ # trust a hook-supplied name.
207
+ declared = self.class.respond_to?(:functions) ? Array(self.class.functions).map { |function| function[:name].to_s } : []
208
+ offered = tool_names_from(params[:tools]) & declared
209
+
210
+ # Every call the model made gets a result message, refused ones
211
+ # included, so the exchange replayed to the provider stays well-formed.
212
+ # Results accumulate as tools run: if one raises, the exchange is
213
+ # still recorded (failure and unexecuted calls spelled out) before the
214
+ # error propagates, so a retry sees what already happened.
215
+ tool_results = []
216
+ begin
217
+ tool_calls.each_with_index do |tool_call, index| # TODO: parallelize this?
218
+ result = if index < allowance
219
+ execute_tool_call(tool_call, offered:)
220
+ else
221
+ "Tool call refused: maximum tool calls (#{@max_tool_calls_budget}) exceeded."
222
+ end
210
223
 
211
- dispatch_tool_function(function_name, arguments.with_indifferent_access)
224
+ tool_results << tool_result_message(tool_call, result)
225
+ end
226
+ rescue StandardError => e
227
+ # The exception class is recorded, not its message: error text can
228
+ # carry response bodies, SQL, or credentials the model must not see.
229
+ failed, *unexecuted = tool_calls.drop(tool_results.size)
230
+ tool_results << tool_result_message(failed, "Tool call failed (#{e.class}).") if failed
231
+ unexecuted.each { |tool_call| tool_results << tool_result_message(tool_call, "Not executed: an earlier tool call in this batch failed.") }
232
+ transcript << [assistant_turn, *tool_results] if save_response
233
+ raise
212
234
  end
213
235
 
214
- # After executing tool calls, we need to continue the conversation
215
- # to let the AI process the results and provide a text response.
216
- # We continue until the AI responds with a regular assistant message
217
- # (not another tool call request), unless stop_tool_calls_and_respond! was called.
218
-
219
- # Use the updated transcript for the next call, not the original messages
220
- updated_messages = transcript.flatten.compact
221
- last_message = updated_messages.last
236
+ # Record the authoritative exchange (the model's ids and signatures,
237
+ # not synthetic ones) so a later chat_completion on this transcript can
238
+ # replay it faithfully, then continue from it. `save_response: false`
239
+ # keeps the exchange out of the transcript along with the final answer,
240
+ # which is how a nested chat_completion inside a tool body keeps its
241
+ # internal rounds out of the outer conversation's history.
242
+ transcript << [assistant_turn, *tool_results] if save_response
243
+ messages += [assistant_turn, *tool_results]
244
+
245
+ # A cap breach or stop_tool_calls_and_respond! ends the conversation
246
+ # with one final, tool-less completion that goes through the same
247
+ # response handling below. Otherwise let the AI process the tool
248
+ # results and either answer or call more tools.
249
+ if cap_exceeded || @stop_tool_calls_and_respond
250
+ response = force_final_response(params:, openai:, messages:, cap_exceeded:)
251
+ Thread.current[:chat_completion_response] = response.with_indifferent_access
252
+ else
253
+ # Drop a forced tool_choice before continuing, the way
254
+ # force_final_response does: re-sending "required" (or a named
255
+ # function) on every round would keep the model calling tools until
256
+ # the budget ran out instead of letting it answer.
257
+ params.delete(:tool_choice)
222
258
 
223
- if !@stop_tool_calls_and_respond && (last_message[:role] != "assistant" || last_message[:tool_calls].present?)
224
- # Send the updated transcript back to the AI
225
- return chat_completion(
259
+ return complete_conversation(
226
260
  params:,
227
261
  json:,
228
262
  raw:,
229
263
  openai:,
230
264
  save_response:,
231
- messages: nil, # Use transcript instead
265
+ messages:,
232
266
  available_tools:,
233
267
  max_tool_calls: self.max_tool_calls - tool_call_count
234
268
  )
235
- elsif @stop_tool_calls_and_respond
236
- # If stop_tool_calls_and_respond was set, force a final response without tools
237
- params[:tools] = nil
238
- response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
239
-
240
- content = response.dig("choices", 0, "message", "content")
241
- transcript << { assistant: content } if save_response
242
- return raw ? response : content.to_s.strip
243
269
  end
244
270
  end
245
271
 
@@ -253,12 +279,16 @@ module Raix
253
279
  # Make automatic JSON parsing available to non-OpenAI providers that don't support the response_format parameter
254
280
  content = content.match(%r{<json>(.*?)</json>}m)[1] if content.include?("<json>")
255
281
 
282
+ parsing_response = true
256
283
  return JSON.parse(content)
257
284
  end
258
285
 
259
286
  return content unless raw
260
287
  end
261
288
  rescue JSON::ParserError => e
289
+ # Only a parse failure of the model's own response is worth a retry.
290
+ raise e unless parsing_response
291
+
262
292
  if e.message.include?("not a valid") # blank JSON
263
293
  warn "Retrying blank JSON response... (#{retry_count} attempts) #{e.message}"
264
294
  retry_count += 1
@@ -274,8 +304,13 @@ module Raix
274
304
  # make sure we see the actual error message on console or Honeybadger
275
305
  warn "Chat completion failed!!!!!!!!!!!!!!!!: #{e.response[:body]}"
276
306
  raise e
307
+ ensure
308
+ @tool_loop_depth -= 1
309
+ @max_tool_calls_budget = nil if @tool_loop_depth.zero?
310
+ @stop_tool_calls_and_respond = stop_on_entry
277
311
  end
278
312
  end
313
+ private :complete_conversation
279
314
 
280
315
  # This method returns the transcript array.
281
316
  # Manually add your messages to it in the following abbreviated format
@@ -308,7 +343,7 @@ module Raix
308
343
  :openrouter
309
344
  end
310
345
 
311
- RubyLLM.chat(model: model_id, provider:, assume_model_exists: true)
346
+ RubyLLM.chat(model: model_id, provider:, protocol: :chat_completions, assume_model_exists: true)
312
347
  end
313
348
  end
314
349
 
@@ -323,35 +358,63 @@ module Raix
323
358
  public_send(function_name, arguments, cache)
324
359
  end
325
360
 
326
- # Increments and returns the per-completion tool-call counter. Called by the
327
- # generated FunctionToolAdapter wrappers to enforce max_tool_calls: RubyLLM
328
- # runs the whole tool loop inside a single chat.complete, so Raix never sees
329
- # the individual rounds and has to count from inside the tool. Internal API.
330
- def increment_tool_call_count
331
- @tool_call_count = @tool_call_count.to_i + 1
361
+ private
362
+
363
+ # Runs one tool call with this loop's bookkeeping shielded from
364
+ # re-entrancy. FunctionDispatch executes tool bodies on this same instance,
365
+ # so a tool that calls chat_completion again (a sub-agent pattern) would
366
+ # otherwise overwrite the outer loop's max_tool_calls, depth and budget,
367
+ # and reset a stop flag an earlier tool in this batch raised. The nested
368
+ # call starts from a clean slate and the outer values come back afterwards.
369
+ def dispatch_preserving_loop_state(function_name, arguments)
370
+ saved = [max_tool_calls, @tool_loop_depth, @max_tool_calls_budget, @stop_tool_calls_and_respond]
371
+
372
+ @tool_loop_depth = 0
373
+ @max_tool_calls_budget = nil
374
+ @stop_tool_calls_and_respond = false
375
+ dispatch_tool_function(function_name, arguments)
376
+ ensure
377
+ # A stop this tool requested belongs to this loop. A nested
378
+ # chat_completion puts back the flag it found on entry, so a stop raised
379
+ # inside it never shows up here.
380
+ stop_requested_here = @stop_tool_calls_and_respond
381
+ self.max_tool_calls, @tool_loop_depth, @max_tool_calls_budget, @stop_tool_calls_and_respond = saved
382
+ @stop_tool_calls_and_respond ||= stop_requested_here
332
383
  end
333
384
 
334
- private
385
+ def tool_result_message(tool_call, content)
386
+ { role: "tool", tool_call_id: tool_call[:id], name: tool_call.dig(:function, :name), content: content.to_s }
387
+ end
335
388
 
336
- # Issues the single final completion after RubyLLM's tool loop was halted,
337
- # then completes once with no tools and returns the OpenAI-compatible hash.
338
- #
339
- # The final request is rebuilt from the original `messages` (the transcript
340
- # or the caller-supplied `messages:` argument that drove the first request)
341
- # plus the assistant/tool exchange FunctionDispatch appended while executing
342
- # tools before the halt. `transcript_size_before` marks where that exchange
343
- # begins, so the caller's system/user prompt is preserved even when it was
344
- # never written into the transcript — otherwise the forced final call would
345
- # answer from tool results (or stale transcript) alone.
346
- def force_final_response_after_halt(halt, params:, openai:, messages:, transcript_size_before:)
347
- adapter = MessageAdapters::Base.new(self)
348
- tool_exchange = transcript.flatten.compact.drop(transcript_size_before).map { |msg| adapter.transform(msg) }
349
- messages += tool_exchange
389
+ # Authorizes and runs one tool call from the model, returning the value the
390
+ # tool result message carries back. A call for a tool that was not offered,
391
+ # or with a malformed argument payload, is reported to the model as a tool
392
+ # error rather than raised: earlier calls in the same batch may already
393
+ # have done their work, and the model can recover from a result.
394
+ def execute_tool_call(tool_call, offered:)
395
+ function_name = tool_call.dig(:function, :name).to_s
396
+ return "Tool call refused: #{function_name} is not available on this request." unless offered.include?(function_name)
397
+
398
+ # Only a missing or empty payload means "no arguments"; whitespace or any
399
+ # other unparseable text is malformed and must not run the tool.
400
+ raw_arguments = tool_call.dig(:function, :arguments).to_s
401
+ begin
402
+ arguments = raw_arguments.empty? ? {} : JSON.parse(raw_arguments)
403
+ rescue JSON::ParserError
404
+ return "Invalid arguments for #{function_name}: malformed JSON"
405
+ end
406
+ return "Invalid arguments for #{function_name}: expected a JSON object, got #{arguments.class}" unless arguments.is_a?(Hash)
407
+
408
+ dispatch_preserving_loop_state(function_name, arguments.with_indifferent_access)
409
+ end
350
410
 
351
- reason = halt.content
352
- if reason.is_a?(FunctionToolAdapter::ToolCallsCapReached)
353
- messages << { role: "system",
354
- content: "Maximum tool calls (#{reason.max_tool_calls}) exceeded. Please provide a final response to the user without calling any more tools." }
411
+ # Issues the single final completion that ends a conversation cut short by
412
+ # the max_tool_calls budget or by stop_tool_calls_and_respond!, and returns
413
+ # the OpenAI-compatible hash.
414
+ def force_final_response(params:, openai:, messages:, cap_exceeded:)
415
+ if cap_exceeded
416
+ messages += [{ role: "system",
417
+ content: "Maximum tool calls (#{@max_tool_calls_budget}) exceeded. Please provide a final response to the user without calling any more tools." }]
355
418
  end
356
419
 
357
420
  # Force a final response without tools. Drop tool_choice as well: a
@@ -364,6 +427,13 @@ module Raix
364
427
  ruby_llm_request(params: final_params, model: openai || model, messages:, openai_override: openai)
365
428
  end
366
429
 
430
+ # Function names declared by an OpenAI-shaped tools array. Tolerates string
431
+ # keys, since a before_completion hook may hand back tools that went
432
+ # through JSON.
433
+ def tool_names_from(tools)
434
+ Array(tools).map { |tool| tool.with_indifferent_access.dig(:function, :name).to_s }
435
+ end
436
+
367
437
  def filtered_tools(tool_names)
368
438
  return nil if tool_names.blank?
369
439
 
@@ -404,116 +474,189 @@ module Raix
404
474
  def ruby_llm_request(params:, model:, messages:, openai_override: nil)
405
475
  # Create a temporary chat instance for this request
406
476
  provider = determine_provider(model, openai_override)
407
- chat = RubyLLM.chat(model:, provider:, assume_model_exists: true)
477
+ chat = RubyLLM.chat(model:, provider:, protocol: :chat_completions, assume_model_exists: true)
408
478
 
409
- # Apply messages to the chat
410
- # Track if we have a user message to determine how to call ask
411
- has_user_message = false
479
+ # Apply messages to the chat. Structured content arrays (multipart text,
480
+ # images, Anthropic-style cache_control) are taken apart first, because
481
+ # RubyLLM messages only carry String content.
482
+ caching = false
483
+ cache_ttl = nil
412
484
 
413
485
  messages.each do |msg|
414
486
  role = msg[:role] || msg["role"]
415
- content = msg[:content] || msg["content"]
487
+ part = MultimodalContentAdapter.translate(msg[:content] || msg["content"])
488
+ content = part.content
489
+ caching ||= part.cache_boundary?
490
+ cache_ttl ||= part.cache_ttl
416
491
 
417
492
  case role.to_s
418
493
  when "system"
419
- chat.with_instructions(content)
494
+ chat.with_instructions(content, append: true, cache_until_here: part.cache_boundary?)
420
495
  when "user"
421
- has_user_message = true
422
- chat.add_message(role: :user, content: MultimodalContentAdapter.translate(content))
496
+ # A user turn must carry content on the wire. With attachments RubyLLM
497
+ # builds the parts itself; without them nil would be dropped from the
498
+ # payload entirely, which providers reject.
499
+ content = "" if content.nil? && part.attachments.empty?
500
+ added = chat.add_message(role: :user, content:, attachments: part.attachments)
501
+ added.cache_until_here if part.cache_boundary?
423
502
  when "assistant"
424
- if (tool_calls = msg[:tool_calls] || msg["tool_calls"])
425
- chat.add_message(role: :assistant, content:, tool_calls: normalize_tool_calls_for_ruby_llm(tool_calls))
426
- else
427
- chat.add_message(role: :assistant, content:)
428
- end
503
+ tool_calls = msg[:tool_calls] || msg["tool_calls"]
504
+ # RubyLLM requires the :content key even when nil (a tool-call turn);
505
+ # an assistant turn with neither content nor tool calls needs "" so
506
+ # the provider still receives a content field.
507
+ content = "" if content.nil? && tool_calls.blank?
508
+ attrs = { role: :assistant, content: }
509
+ attrs[:tool_calls] = normalize_tool_calls_for_ruby_llm(tool_calls) if tool_calls
510
+ # Signed reasoning has to make the round trip for multi-round tool use
511
+ # on models that require it (OpenRouter reasoning_details, Gemini
512
+ # thought signatures).
513
+ reasoning = {
514
+ raw_reasoning: msg[:raw_reasoning] || msg["raw_reasoning"],
515
+ thinking: msg[:thinking] || msg["thinking"],
516
+ thinking_signature: msg[:thinking_signature] || msg["thinking_signature"]
517
+ }.compact
518
+ added = chat.add_message(attrs.merge(reasoning))
519
+ added.cache_until_here if part.cache_boundary?
429
520
  when "tool"
430
521
  chat.add_message(
431
522
  role: :tool,
432
523
  content:,
433
524
  tool_call_id: msg[:tool_call_id] || msg["tool_call_id"]
434
525
  )
526
+ else
527
+ # Anything else (including the legacy "function" role) has no
528
+ # RubyLLM equivalent. Say so rather than dropping it silently.
529
+ warn "Raix: skipping message with unsupported role #{role.inspect}; RubyLLM accepts system, user, assistant, and tool"
435
530
  end
436
531
  end
437
532
 
533
+ # Render the cache boundaries marked above; without this RubyLLM sends no
534
+ # cache controls at all. A ttl from the content's cache_control rides along.
535
+ chat.with_caching({ ttl: cache_ttl }.compact) if caching
536
+
438
537
  # Apply configuration parameters
439
538
  chat.with_temperature(params[:temperature]) if params[:temperature]
539
+ if (max_output_tokens = params[:max_completion_tokens] || params[:max_tokens])
540
+ chat.with_max_output_tokens(max_output_tokens)
541
+ end
440
542
 
441
- # Apply additional params (RubyLLM with_params expects keyword args)
543
+ # Apply additional params. RubyLLM sends provider options into the
544
+ # request payload verbatim, which is what these OpenAI/OpenRouter-shaped
545
+ # knobs (top_p, seed, response_format, provider routing, ...) expect.
442
546
  additional_params = params.compact.except(:temperature, :tools, :max_tokens, :max_completion_tokens)
443
- chat.with_params(**additional_params) if additional_params.any?
547
+ chat.with_provider_options(additional_params) if additional_params.any?
444
548
 
445
- # Handle tools - convert Raix function declarations to RubyLLM tools
446
- if params[:tools].present? && respond_to?(:class) && self.class.respond_to?(:functions)
447
- ruby_llm_tools = FunctionToolAdapter.convert_tools_for_ruby_llm(self)
448
- ruby_llm_tools.each { |tool| chat.with_tool(tool) }
549
+ # Handle tools - convert Raix function declarations to RubyLLM tools.
550
+ # params[:tools] already reflects `available_tools`, so only the
551
+ # functions it names are registered with the chat.
552
+ if params[:tools].present? && self.class.respond_to?(:functions)
553
+ chat.with_tools(*FunctionToolAdapter.convert_tools_for_ruby_llm(self, only: tool_names_from(params[:tools])))
449
554
  end
450
555
 
451
- # Execute the completion
452
- if stream.present?
453
- # Streaming mode
454
- if has_user_message
455
- chat.complete(&stream)
456
- else
457
- chat.ask(&stream)
458
- end
459
- nil # Return nil for streaming as per original behavior
460
- else
461
- # Non-streaming mode - return OpenAI-compatible response format
462
- response_message = has_user_message ? chat.complete : chat.ask
463
-
464
- # A generated tool can halt RubyLLM's internal tool loop (max_tool_calls
465
- # exceeded or stop_tool_calls_and_respond!). When that happens
466
- # chat.complete/#ask returns a RubyLLM::Tool::Halt, not a Message, so it
467
- # has no #raw / #input_tokens. Hand it straight back to chat_completion,
468
- # which forces a final tool-less completion.
469
- return response_message if response_message.is_a?(RubyLLM::Tool::Halt)
470
-
471
- # Pull through the raw provider payload when available. OpenRouter's
472
- # `id` is the only handle we have to look up authoritative billing
473
- # cost via /api/v1/generation, and callers that watch the response
474
- # snapshot for `model` / cached-token counts shouldn't have to break
475
- # out of the OpenAI-compatible shape to get them.
476
- raw_body = response_message.raw.respond_to?(:body) ? response_message.raw.body : nil
477
- raw_body = {} unless raw_body.is_a?(Hash)
478
-
479
- usage_payload = {
480
- "prompt_tokens" => response_message.input_tokens,
481
- "completion_tokens" => response_message.output_tokens,
482
- "total_tokens" => (response_message.input_tokens || 0) + (response_message.output_tokens || 0)
483
- }
484
-
485
- # Merge prompt_tokens_details / completion_tokens_details (cached tokens,
486
- # reasoning tokens) when the provider supplied them.
487
- if (upstream_usage = raw_body["usage"]).is_a?(Hash)
488
- upstream_usage.each do |key, value|
489
- next if usage_payload.key?(key)
490
-
491
- usage_payload[key] = value
492
- end
493
- end
556
+ # Execute the completion. Raix drives the tool loop itself (see
557
+ # #chat_completion), so this asks RubyLLM for exactly one completion and
558
+ # leaves any tool calls in the response unexecuted. A streaming request
559
+ # yields chunks to the block and still returns the assembled message, so
560
+ # tool calls made mid-stream get dispatched like any other.
561
+ response_message = stream.present? ? chat.generate(&stream) : chat.generate
562
+ return nil if response_message.nil?
563
+
564
+ # Pull through the raw provider payload when available. OpenRouter's
565
+ # `id` is the only handle we have to look up authoritative billing
566
+ # cost via /api/v1/generation, and callers that watch the response
567
+ # snapshot for `model` / cached-token counts shouldn't have to break
568
+ # out of the OpenAI-compatible shape to get them.
569
+ raw_body = response_message.raw.respond_to?(:body) ? response_message.raw.body : nil
570
+ raw_body = {} unless raw_body.is_a?(Hash)
571
+ upstream_usage = raw_body["usage"].is_a?(Hash) ? raw_body["usage"] : {}
572
+
573
+ # Prefer the provider's own counts. RubyLLM's input figure excludes cached
574
+ # tokens, and only the provider knows the authoritative totals; its
575
+ # prompt_tokens_details / completion_tokens_details ride along untouched.
576
+ tokens = response_message.tokens
577
+ prompt_tokens = upstream_usage["prompt_tokens"] || [tokens.input, tokens.cache_read, tokens.cache_write].compact.sum
578
+ completion_tokens = upstream_usage["completion_tokens"] || tokens.output
579
+ usage_payload = upstream_usage.merge(
580
+ "prompt_tokens" => prompt_tokens,
581
+ "completion_tokens" => completion_tokens,
582
+ "total_tokens" => upstream_usage["total_tokens"] || (prompt_tokens.to_i + completion_tokens.to_i)
583
+ )
494
584
 
495
- {
496
- "id" => raw_body["id"],
497
- "model" => raw_body["model"] || response_message.model_id,
498
- "provider" => raw_body["provider"],
499
- "choices" => [
500
- {
501
- "message" => {
502
- "role" => "assistant",
503
- "content" => response_message.content,
504
- "tool_calls" => response_message.tool_calls
505
- },
506
- "finish_reason" => response_message.tool_call? ? "tool_calls" : "stop"
507
- }
508
- ],
509
- "usage" => usage_payload
510
- }
585
+ # The assistant turn carries what a continuation round has to replay:
586
+ # tool calls with their ids and signatures, plus any signed reasoning.
587
+ message = {
588
+ "role" => "assistant",
589
+ "content" => response_message.content,
590
+ "tool_calls" => serialize_tool_calls(response_message.tool_calls)
591
+ }
592
+ message["raw_reasoning"] = response_message.raw_reasoning if response_message.raw_reasoning
593
+ if (thinking = response_message.thinking)
594
+ message["thinking"] = thinking.text
595
+ message["thinking_signature"] = thinking.signature
511
596
  end
597
+
598
+ {
599
+ "id" => raw_body["id"],
600
+ "model" => raw_body["model"] || response_message.model,
601
+ "provider" => raw_body["provider"],
602
+ "choices" => [
603
+ {
604
+ "message" => message,
605
+ "finish_reason" => response_message.tool_call? ? "tool_calls" : "stop"
606
+ }
607
+ ],
608
+ "usage" => usage_payload
609
+ }
610
+ rescue RubyLLM::ToolCallParseError => e
611
+ # RubyLLM rejects a response whose tool arguments are not valid JSON before
612
+ # Raix ever sees a Message. Rebuild the assistant turn from the raw payload
613
+ # so the loop can answer each call with a tool error instead of aborting.
614
+ response = assistant_turn_from_raw_response(e.response)
615
+ raise e unless response
616
+
617
+ response
512
618
  rescue StandardError => e
513
619
  warn "RubyLLM request failed: #{e.message}"
514
620
  raise e
515
621
  end
516
622
 
623
+ # The OpenAI-compatible response hash for a provider payload RubyLLM could
624
+ # not parse into a Message, or nil when the payload has no tool calls to
625
+ # recover. Arguments stay as the provider sent them.
626
+ def assistant_turn_from_raw_response(raw)
627
+ body = raw.respond_to?(:body) ? raw.body : nil
628
+ return unless body.is_a?(Hash)
629
+
630
+ raw_message = body.dig("choices", 0, "message")
631
+ return unless raw_message.is_a?(Hash) && raw_message["tool_calls"].is_a?(Array)
632
+
633
+ # Replay keys tool calls by id, so a payload with missing or duplicate
634
+ # ids cannot be recovered into a valid exchange.
635
+ ids = raw_message["tool_calls"].map { |tool_call| tool_call["id"].to_s }
636
+ return if ids.any?(&:empty?) || ids.uniq.size != ids.size
637
+
638
+ # Keep the signed state RubyLLM would have extracted: OpenRouter's
639
+ # reasoning_details array and Gemini's per-call thought signature, which
640
+ # the wire nests under extra_content.google.
641
+ message = {
642
+ "role" => "assistant",
643
+ "content" => raw_message["content"],
644
+ "tool_calls" => raw_message["tool_calls"].map do |tool_call|
645
+ signature = tool_call.dig("extra_content", "google", "thought_signature")
646
+ signature ? tool_call.merge("thought_signature" => signature) : tool_call
647
+ end
648
+ }
649
+ message["raw_reasoning"] = raw_message["reasoning_details"] if raw_message["reasoning_details"].is_a?(Array)
650
+
651
+ {
652
+ "id" => body["id"],
653
+ "model" => body["model"],
654
+ "provider" => body["provider"],
655
+ "choices" => [{ "message" => message, "finish_reason" => "tool_calls" }],
656
+ "usage" => body["usage"].is_a?(Hash) ? body["usage"] : {}
657
+ }
658
+ end
659
+
517
660
  def determine_provider(model, openai_override)
518
661
  return :openai if openai_override
519
662
  return :openai if model.to_s.match?(/^gpt-/) || model.to_s.match?(/^o\d/)
@@ -522,13 +665,46 @@ module Raix
522
665
  :openrouter
523
666
  end
524
667
 
668
+ # Renders RubyLLM's tool calls (a Hash keyed by call id whose values are
669
+ # RubyLLM::ToolCall) as OpenAI's array-of-hashes shape, which is what the
670
+ # response hash Raix hands back to callers — and its own tool loop — reads.
671
+ def serialize_tool_calls(tool_calls)
672
+ return nil if tool_calls.blank?
673
+
674
+ tool_calls.values.map do |tool_call|
675
+ serialized = {
676
+ "id" => tool_call.id,
677
+ "type" => "function",
678
+ "function" => {
679
+ "name" => tool_call.name,
680
+ "arguments" => tool_call.arguments.to_json
681
+ }
682
+ }
683
+ serialized["thought_signature"] = tool_call.thought_signature if tool_call.thought_signature
684
+ serialized
685
+ end
686
+ end
687
+
688
+ # Arguments replayed from a recorded tool call. RubyLLM re-serializes them
689
+ # as JSON, so a payload the provider sent malformed (already answered with
690
+ # a tool error) is replayed as an empty object rather than raising again.
691
+ def parse_replayed_arguments(arguments)
692
+ return {} if arguments.blank?
693
+
694
+ parsed = JSON.parse(arguments)
695
+ parsed.is_a?(Hash) ? parsed : {}
696
+ rescue JSON::ParserError
697
+ {}
698
+ end
699
+
525
700
  # Raix's transcript stores assistant tool calls in OpenAI's array-of-hashes
526
701
  # shape (`[{ id:, type:, function: { name:, arguments: } }]`), but RubyLLM's
527
702
  # providers format tool calls from a Hash keyed by call id whose values
528
703
  # respond to #id/#name/#arguments (RubyLLM::ToolCall). Translate so a
529
704
  # transcript that already contains tool exchanges can be replayed back into
530
- # a fresh RubyLLM chat — notably for the forced final completion after a
531
- # max_tool_calls / stop_tool_calls_and_respond! halt.
705
+ # a fresh RubyLLM chat on every continuation round, including the forced
706
+ # final completion after a max_tool_calls cap breach or
707
+ # stop_tool_calls_and_respond!.
532
708
  def normalize_tool_calls_for_ruby_llm(tool_calls)
533
709
  return tool_calls if tool_calls.is_a?(Hash) && tool_calls.values.all?(RubyLLM::ToolCall)
534
710
 
@@ -536,8 +712,8 @@ module Raix
536
712
  tc = raw.respond_to?(:with_indifferent_access) ? raw.with_indifferent_access : raw
537
713
  function = tc[:function] || {}
538
714
  arguments = function[:arguments]
539
- arguments = JSON.parse(arguments) if arguments.is_a?(String) && arguments.present?
540
- acc[tc[:id]] = RubyLLM::ToolCall.new(id: tc[:id], name: function[:name], arguments: arguments || {})
715
+ arguments = parse_replayed_arguments(arguments) if arguments.is_a?(String)
716
+ acc[tc[:id]] = RubyLLM::ToolCall.new(id: tc[:id], name: function[:name], arguments: arguments || {}, thought_signature: tc[:thought_signature])
541
717
  end
542
718
  end
543
719
  end