raix 2.0.5 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -71,6 +71,17 @@ module Raix
71
71
  # @option max_tool_calls [Integer] :max_tool_calls Maximum number of tool calls before forcing a text response. Defaults to the configured value.
72
72
  # @return [String|Hash] The completed chat response.
73
73
  def chat_completion(params: {}, loop: false, json: false, raw: false, openai: nil, save_response: true, messages: nil, available_tools: nil, max_tool_calls: nil)
74
+ complete_conversation(params:, loop:, json:, raw:, openai:, save_response:, messages:, available_tools:, max_tool_calls:)
75
+ end
76
+
77
+ # The body of chat_completion. Continuation rounds after a tool call recurse
78
+ # here directly rather than through the public method, which a subclass may
79
+ # override with a different signature (PromptDeclarations does).
80
+ def complete_conversation(params: {}, loop: false, json: false, raw: false, openai: nil, save_response: true, messages: nil, available_tools: nil, max_tool_calls: nil)
81
+ # Work on a copy: defaults are filled in and tool_choice is dropped between
82
+ # rounds, and none of that should leak into a Hash the caller may reuse.
83
+ params = params.dup
84
+
74
85
  # set params to default values if not provided
75
86
  params[:cache_at] ||= cache_at.presence
76
87
  params[:frequency_penalty] ||= frequency_penalty.presence
@@ -79,7 +90,11 @@ module Raix
79
90
  params[:max_completion_tokens] ||= max_completion_tokens.presence || configuration.max_completion_tokens
80
91
  params[:max_tokens] ||= max_tokens.presence || configuration.max_tokens
81
92
  params[:min_p] ||= min_p.presence
82
- params[:prediction] = { type: "content", content: params[:prediction] || prediction } if params[:prediction] || prediction.present?
93
+ if (predicted = params[:prediction] || prediction.presence)
94
+ # Continuation rounds pass an already-wrapped prediction back through
95
+ # here, so only wrap a bare value.
96
+ params[:prediction] = predicted.is_a?(Hash) && predicted.with_indifferent_access[:type] == "content" ? predicted : { type: "content", content: predicted }
97
+ end
83
98
  params[:presence_penalty] ||= presence_penalty.presence
84
99
  params[:provider] ||= provider.presence
85
100
  params[:repetition_penalty] ||= repetition_penalty.presence
@@ -87,7 +102,10 @@ module Raix
87
102
  params[:seed] ||= seed.presence
88
103
  params[:stop] ||= stop.presence
89
104
  params[:temperature] ||= temperature.presence || configuration.temperature
90
- params[:tool_choice] ||= tool_choice.presence
105
+ # A forced tool_choice applies to the first round only. Continuation
106
+ # rounds (depth > 0) leave it unset so the model can answer in text once
107
+ # its tool results are in.
108
+ params[:tool_choice] ||= tool_choice.presence if @tool_loop_depth.to_i.zero?
91
109
  params[:tools] = if available_tools == false
92
110
  nil
93
111
  elsif available_tools.is_a?(Array)
@@ -103,14 +121,10 @@ module Raix
103
121
  json = true if params[:response_format].is_a?(Raix::ResponseFormat)
104
122
 
105
123
  if json
106
- unless openai
107
- params[:provider] ||= {}
108
- params[:provider][:require_parameters] = true
109
- end
110
- if params[:response_format].blank?
111
- params[:response_format] ||= {}
112
- params[:response_format][:type] = "json_object"
113
- end
124
+ # Build fresh nested hashes rather than writing into ones the caller
125
+ # handed us; the dup above is shallow.
126
+ params[:provider] = (params[:provider] || {}).merge(require_parameters: true) unless openai
127
+ params[:response_format] = { type: "json_object" } if params[:response_format].blank?
114
128
  end
115
129
 
116
130
  # Deprecation warning for loop parameter
@@ -121,9 +135,6 @@ module Raix
121
135
  # Set max_tool_calls from parameter or configuration default
122
136
  self.max_tool_calls = max_tool_calls || configuration.max_tool_calls
123
137
 
124
- # Reset stop_tool_calls_and_respond flag
125
- @stop_tool_calls_and_respond = false
126
-
127
138
  # Track tool call count
128
139
  tool_call_count = 0
129
140
 
@@ -143,78 +154,118 @@ module Raix
143
154
  # Hooks can modify params and messages for logging, filtering, PII redaction, etc.
144
155
  run_before_completion_hooks(params, messages)
145
156
 
157
+ # Each continuation after a tool round recurses with the allowance that is
158
+ # left, so remember the budget the caller actually asked for. That is the
159
+ # number the limit message has to quote.
160
+ @tool_loop_depth = @tool_loop_depth.to_i + 1
161
+ @max_tool_calls_budget = self.max_tool_calls if @tool_loop_depth == 1
162
+
163
+ # Start this loop with a clear stop flag and hand back whatever was set
164
+ # on entry when it returns. A nested chat_completion inside a tool body
165
+ # must neither erase a stop the enclosing loop's tool already requested
166
+ # nor leak its own stop into that loop.
167
+ stop_on_entry = @stop_tool_calls_and_respond
168
+ @stop_tool_calls_and_respond = false
169
+
170
+ # True only while parsing the model's final JSON response. The blank-JSON
171
+ # retry below must never fire for a JSON::ParserError raised by a tool
172
+ # body (in this frame or a continuation round), which would re-issue the
173
+ # request and run tools again.
174
+ parsing_response = false
175
+
146
176
  begin
147
177
  response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
178
+
148
179
  retry_count = 0
149
180
  content = nil
150
181
 
151
- # no need for additional processing if streaming
152
- return if stream && response.blank?
182
+ # Nothing came back to process (a streamed request that produced no message).
183
+ return if response.blank?
153
184
 
154
185
  # tuck the full response into a thread local in case needed
155
- Thread.current[:chat_completion_response] = response.is_a?(Hash) ? response.with_indifferent_access : response
186
+ Thread.current[:chat_completion_response] = response.with_indifferent_access
156
187
 
157
188
  # TODO: add a standardized callback hook for usage events
158
189
  # broadcast(:usage_event, usage_subject, self.class.name.to_s, response, premium?)
159
190
 
160
- tool_calls = response.dig("choices", 0, "message", "tool_calls") || []
191
+ # The model's own turn, kept intact (tool calls with their ids and
192
+ # signatures, reasoning details) so it can be replayed verbatim.
193
+ assistant_turn = (response.dig("choices", 0, "message") || {}).with_indifferent_access
194
+ tool_calls = assistant_turn[:tool_calls] || []
161
195
  if tool_calls.any?
162
- tool_call_count += tool_calls.size
163
-
164
- # Check if we've exceeded max_tool_calls
165
- if tool_call_count > self.max_tool_calls
166
- # Add system message about hitting the limit
167
- messages << { role: "system", content: "Maximum tool calls (#{self.max_tool_calls}) exceeded. Please provide a final response to the user without calling any more tools." }
168
-
169
- # Force a final response without tools
170
- params[:tools] = nil
171
- response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
172
-
173
- # Process the final response
174
- content = response.dig("choices", 0, "message", "content")
175
- transcript << { assistant: content } if save_response
176
- return raw ? response : content.to_s.strip
177
- end
178
-
179
- # Dispatch tool calls
180
- tool_calls.each do |tool_call| # TODO: parallelize this?
181
- # dispatch the called function
182
- function_name = tool_call["function"]["name"]
183
- arguments = JSON.parse(tool_call["function"]["arguments"].presence || "{}")
184
- raise "Unauthorized function call: #{function_name}" unless self.class.functions.map { |f| f[:name].to_sym }.include?(function_name.to_sym)
196
+ # Enforce the budget per call rather than per round: a single model
197
+ # response can pack several parallel tool calls, and the ones that
198
+ # still fit under the cap should run.
199
+ allowance = [self.max_tool_calls - tool_call_count, 0].max
200
+ cap_exceeded = tool_calls.size > allowance
201
+ tool_call_count += [tool_calls.size, allowance].min
202
+
203
+ # A call is authorized only if the function is declared on this class
204
+ # AND was offered on this request (the available_tools-filtered set).
205
+ # Declared-only would let a hidden tool through; offered-only would
206
+ # trust a hook-supplied name.
207
+ declared = self.class.respond_to?(:functions) ? Array(self.class.functions).map { |function| function[:name].to_s } : []
208
+ offered = tool_names_from(params[:tools]) & declared
209
+
210
+ # Every call the model made gets a result message, refused ones
211
+ # included, so the exchange replayed to the provider stays well-formed.
212
+ # Results accumulate as tools run: if one raises, the exchange is
213
+ # still recorded (failure and unexecuted calls spelled out) before the
214
+ # error propagates, so a retry sees what already happened.
215
+ tool_results = []
216
+ begin
217
+ tool_calls.each_with_index do |tool_call, index| # TODO: parallelize this?
218
+ result = if index < allowance
219
+ execute_tool_call(tool_call, offered:)
220
+ else
221
+ "Tool call refused: maximum tool calls (#{@max_tool_calls_budget}) exceeded."
222
+ end
185
223
 
186
- dispatch_tool_function(function_name, arguments.with_indifferent_access)
224
+ tool_results << tool_result_message(tool_call, result)
225
+ end
226
+ rescue StandardError => e
227
+ # The exception class is recorded, not its message: error text can
228
+ # carry response bodies, SQL, or credentials the model must not see.
229
+ failed, *unexecuted = tool_calls.drop(tool_results.size)
230
+ tool_results << tool_result_message(failed, "Tool call failed (#{e.class}).") if failed
231
+ unexecuted.each { |tool_call| tool_results << tool_result_message(tool_call, "Not executed: an earlier tool call in this batch failed.") }
232
+ transcript << [assistant_turn, *tool_results] if save_response
233
+ raise
187
234
  end
188
235
 
189
- # After executing tool calls, we need to continue the conversation
190
- # to let the AI process the results and provide a text response.
191
- # We continue until the AI responds with a regular assistant message
192
- # (not another tool call request), unless stop_tool_calls_and_respond! was called.
193
-
194
- # Use the updated transcript for the next call, not the original messages
195
- updated_messages = transcript.flatten.compact
196
- last_message = updated_messages.last
236
+ # Record the authoritative exchange (the model's ids and signatures,
237
+ # not synthetic ones) so a later chat_completion on this transcript can
238
+ # replay it faithfully, then continue from it. `save_response: false`
239
+ # keeps the exchange out of the transcript along with the final answer,
240
+ # which is how a nested chat_completion inside a tool body keeps its
241
+ # internal rounds out of the outer conversation's history.
242
+ transcript << [assistant_turn, *tool_results] if save_response
243
+ messages += [assistant_turn, *tool_results]
244
+
245
+ # A cap breach or stop_tool_calls_and_respond! ends the conversation
246
+ # with one final, tool-less completion that goes through the same
247
+ # response handling below. Otherwise let the AI process the tool
248
+ # results and either answer or call more tools.
249
+ if cap_exceeded || @stop_tool_calls_and_respond
250
+ response = force_final_response(params:, openai:, messages:, cap_exceeded:)
251
+ Thread.current[:chat_completion_response] = response.with_indifferent_access
252
+ else
253
+ # Drop a forced tool_choice before continuing, the way
254
+ # force_final_response does: re-sending "required" (or a named
255
+ # function) on every round would keep the model calling tools until
256
+ # the budget ran out instead of letting it answer.
257
+ params.delete(:tool_choice)
197
258
 
198
- if !@stop_tool_calls_and_respond && (last_message[:role] != "assistant" || last_message[:tool_calls].present?)
199
- # Send the updated transcript back to the AI
200
- return chat_completion(
259
+ return complete_conversation(
201
260
  params:,
202
261
  json:,
203
262
  raw:,
204
263
  openai:,
205
264
  save_response:,
206
- messages: nil, # Use transcript instead
265
+ messages:,
207
266
  available_tools:,
208
267
  max_tool_calls: self.max_tool_calls - tool_call_count
209
268
  )
210
- elsif @stop_tool_calls_and_respond
211
- # If stop_tool_calls_and_respond was set, force a final response without tools
212
- params[:tools] = nil
213
- response = ruby_llm_request(params:, model: openai || model, messages:, openai_override: openai)
214
-
215
- content = response.dig("choices", 0, "message", "content")
216
- transcript << { assistant: content } if save_response
217
- return raw ? response : content.to_s.strip
218
269
  end
219
270
  end
220
271
 
@@ -228,12 +279,16 @@ module Raix
228
279
  # Make automatic JSON parsing available to non-OpenAI providers that don't support the response_format parameter
229
280
  content = content.match(%r{<json>(.*?)</json>}m)[1] if content.include?("<json>")
230
281
 
282
+ parsing_response = true
231
283
  return JSON.parse(content)
232
284
  end
233
285
 
234
286
  return content unless raw
235
287
  end
236
288
  rescue JSON::ParserError => e
289
+ # Only a parse failure of the model's own response is worth a retry.
290
+ raise e unless parsing_response
291
+
237
292
  if e.message.include?("not a valid") # blank JSON
238
293
  warn "Retrying blank JSON response... (#{retry_count} attempts) #{e.message}"
239
294
  retry_count += 1
@@ -249,8 +304,13 @@ module Raix
249
304
  # make sure we see the actual error message on console or Honeybadger
250
305
  warn "Chat completion failed!!!!!!!!!!!!!!!!: #{e.response[:body]}"
251
306
  raise e
307
+ ensure
308
+ @tool_loop_depth -= 1
309
+ @max_tool_calls_budget = nil if @tool_loop_depth.zero?
310
+ @stop_tool_calls_and_respond = stop_on_entry
252
311
  end
253
312
  end
313
+ private :complete_conversation
254
314
 
255
315
  # This method returns the transcript array.
256
316
  # Manually add your messages to it in the following abbreviated format
@@ -283,7 +343,7 @@ module Raix
283
343
  :openrouter
284
344
  end
285
345
 
286
- RubyLLM.chat(model: model_id, provider:, assume_model_exists: true)
346
+ RubyLLM.chat(model: model_id, provider:, protocol: :chat_completions, assume_model_exists: true)
287
347
  end
288
348
  end
289
349
 
@@ -300,6 +360,80 @@ module Raix
300
360
 
301
361
  private
302
362
 
363
+ # Runs one tool call with this loop's bookkeeping shielded from
364
+ # re-entrancy. FunctionDispatch executes tool bodies on this same instance,
365
+ # so a tool that calls chat_completion again (a sub-agent pattern) would
366
+ # otherwise overwrite the outer loop's max_tool_calls, depth and budget,
367
+ # and reset a stop flag an earlier tool in this batch raised. The nested
368
+ # call starts from a clean slate and the outer values come back afterwards.
369
+ def dispatch_preserving_loop_state(function_name, arguments)
370
+ saved = [max_tool_calls, @tool_loop_depth, @max_tool_calls_budget, @stop_tool_calls_and_respond]
371
+
372
+ @tool_loop_depth = 0
373
+ @max_tool_calls_budget = nil
374
+ @stop_tool_calls_and_respond = false
375
+ dispatch_tool_function(function_name, arguments)
376
+ ensure
377
+ # A stop this tool requested belongs to this loop. A nested
378
+ # chat_completion puts back the flag it found on entry, so a stop raised
379
+ # inside it never shows up here.
380
+ stop_requested_here = @stop_tool_calls_and_respond
381
+ self.max_tool_calls, @tool_loop_depth, @max_tool_calls_budget, @stop_tool_calls_and_respond = saved
382
+ @stop_tool_calls_and_respond ||= stop_requested_here
383
+ end
384
+
385
+ def tool_result_message(tool_call, content)
386
+ { role: "tool", tool_call_id: tool_call[:id], name: tool_call.dig(:function, :name), content: content.to_s }
387
+ end
388
+
389
+ # Authorizes and runs one tool call from the model, returning the value the
390
+ # tool result message carries back. A call for a tool that was not offered,
391
+ # or with a malformed argument payload, is reported to the model as a tool
392
+ # error rather than raised: earlier calls in the same batch may already
393
+ # have done their work, and the model can recover from a result.
394
+ def execute_tool_call(tool_call, offered:)
395
+ function_name = tool_call.dig(:function, :name).to_s
396
+ return "Tool call refused: #{function_name} is not available on this request." unless offered.include?(function_name)
397
+
398
+ # Only a missing or empty payload means "no arguments"; whitespace or any
399
+ # other unparseable text is malformed and must not run the tool.
400
+ raw_arguments = tool_call.dig(:function, :arguments).to_s
401
+ begin
402
+ arguments = raw_arguments.empty? ? {} : JSON.parse(raw_arguments)
403
+ rescue JSON::ParserError
404
+ return "Invalid arguments for #{function_name}: malformed JSON"
405
+ end
406
+ return "Invalid arguments for #{function_name}: expected a JSON object, got #{arguments.class}" unless arguments.is_a?(Hash)
407
+
408
+ dispatch_preserving_loop_state(function_name, arguments.with_indifferent_access)
409
+ end
410
+
411
+ # Issues the single final completion that ends a conversation cut short by
412
+ # the max_tool_calls budget or by stop_tool_calls_and_respond!, and returns
413
+ # the OpenAI-compatible hash.
414
+ def force_final_response(params:, openai:, messages:, cap_exceeded:)
415
+ if cap_exceeded
416
+ messages += [{ role: "system",
417
+ content: "Maximum tool calls (#{@max_tool_calls_budget}) exceeded. Please provide a final response to the user without calling any more tools." }]
418
+ end
419
+
420
+ # Force a final response without tools. Drop tool_choice as well: a
421
+ # lingering tool_choice that forces tool use with no tools registered is a
422
+ # provider error.
423
+ final_params = params.dup
424
+ final_params[:tools] = nil
425
+ final_params.delete(:tool_choice)
426
+
427
+ ruby_llm_request(params: final_params, model: openai || model, messages:, openai_override: openai)
428
+ end
429
+
430
+ # Function names declared by an OpenAI-shaped tools array. Tolerates string
431
+ # keys, since a before_completion hook may hand back tools that went
432
+ # through JSON.
433
+ def tool_names_from(tools)
434
+ Array(tools).map { |tool| tool.with_indifferent_access.dig(:function, :name).to_s }
435
+ end
436
+
303
437
  def filtered_tools(tool_names)
304
438
  return nil if tool_names.blank?
305
439
 
@@ -340,109 +474,189 @@ module Raix
340
474
  def ruby_llm_request(params:, model:, messages:, openai_override: nil)
341
475
  # Create a temporary chat instance for this request
342
476
  provider = determine_provider(model, openai_override)
343
- chat = RubyLLM.chat(model:, provider:, assume_model_exists: true)
477
+ chat = RubyLLM.chat(model:, provider:, protocol: :chat_completions, assume_model_exists: true)
344
478
 
345
- # Apply messages to the chat
346
- # Track if we have a user message to determine how to call ask
347
- has_user_message = false
479
+ # Apply messages to the chat. Structured content arrays (multipart text,
480
+ # images, Anthropic-style cache_control) are taken apart first, because
481
+ # RubyLLM messages only carry String content.
482
+ caching = false
483
+ cache_ttl = nil
348
484
 
349
485
  messages.each do |msg|
350
486
  role = msg[:role] || msg["role"]
351
- content = msg[:content] || msg["content"]
487
+ part = MultimodalContentAdapter.translate(msg[:content] || msg["content"])
488
+ content = part.content
489
+ caching ||= part.cache_boundary?
490
+ cache_ttl ||= part.cache_ttl
352
491
 
353
492
  case role.to_s
354
493
  when "system"
355
- chat.with_instructions(content)
494
+ chat.with_instructions(content, append: true, cache_until_here: part.cache_boundary?)
356
495
  when "user"
357
- has_user_message = true
358
- chat.add_message(role: :user, content: MultimodalContentAdapter.translate(content))
496
+ # A user turn must carry content on the wire. With attachments RubyLLM
497
+ # builds the parts itself; without them nil would be dropped from the
498
+ # payload entirely, which providers reject.
499
+ content = "" if content.nil? && part.attachments.empty?
500
+ added = chat.add_message(role: :user, content:, attachments: part.attachments)
501
+ added.cache_until_here if part.cache_boundary?
359
502
  when "assistant"
360
- if msg[:tool_calls] || msg["tool_calls"]
361
- chat.add_message(role: :assistant, content:, tool_calls: msg[:tool_calls] || msg["tool_calls"])
362
- else
363
- chat.add_message(role: :assistant, content:)
364
- end
503
+ tool_calls = msg[:tool_calls] || msg["tool_calls"]
504
+ # RubyLLM requires the :content key even when nil (a tool-call turn);
505
+ # an assistant turn with neither content nor tool calls needs "" so
506
+ # the provider still receives a content field.
507
+ content = "" if content.nil? && tool_calls.blank?
508
+ attrs = { role: :assistant, content: }
509
+ attrs[:tool_calls] = normalize_tool_calls_for_ruby_llm(tool_calls) if tool_calls
510
+ # Signed reasoning has to make the round trip for multi-round tool use
511
+ # on models that require it (OpenRouter reasoning_details, Gemini
512
+ # thought signatures).
513
+ reasoning = {
514
+ raw_reasoning: msg[:raw_reasoning] || msg["raw_reasoning"],
515
+ thinking: msg[:thinking] || msg["thinking"],
516
+ thinking_signature: msg[:thinking_signature] || msg["thinking_signature"]
517
+ }.compact
518
+ added = chat.add_message(attrs.merge(reasoning))
519
+ added.cache_until_here if part.cache_boundary?
365
520
  when "tool"
366
521
  chat.add_message(
367
522
  role: :tool,
368
523
  content:,
369
524
  tool_call_id: msg[:tool_call_id] || msg["tool_call_id"]
370
525
  )
526
+ else
527
+ # Anything else (including the legacy "function" role) has no
528
+ # RubyLLM equivalent. Say so rather than dropping it silently.
529
+ warn "Raix: skipping message with unsupported role #{role.inspect}; RubyLLM accepts system, user, assistant, and tool"
371
530
  end
372
531
  end
373
532
 
533
+ # Render the cache boundaries marked above; without this RubyLLM sends no
534
+ # cache controls at all. A ttl from the content's cache_control rides along.
535
+ chat.with_caching({ ttl: cache_ttl }.compact) if caching
536
+
374
537
  # Apply configuration parameters
375
538
  chat.with_temperature(params[:temperature]) if params[:temperature]
539
+ if (max_output_tokens = params[:max_completion_tokens] || params[:max_tokens])
540
+ chat.with_max_output_tokens(max_output_tokens)
541
+ end
376
542
 
377
- # Apply additional params (RubyLLM with_params expects keyword args)
543
+ # Apply additional params. RubyLLM sends provider options into the
544
+ # request payload verbatim, which is what these OpenAI/OpenRouter-shaped
545
+ # knobs (top_p, seed, response_format, provider routing, ...) expect.
378
546
  additional_params = params.compact.except(:temperature, :tools, :max_tokens, :max_completion_tokens)
379
- chat.with_params(**additional_params) if additional_params.any?
547
+ chat.with_provider_options(additional_params) if additional_params.any?
380
548
 
381
- # Handle tools - convert Raix function declarations to RubyLLM tools
382
- if params[:tools].present? && respond_to?(:class) && self.class.respond_to?(:functions)
383
- ruby_llm_tools = FunctionToolAdapter.convert_tools_for_ruby_llm(self)
384
- ruby_llm_tools.each { |tool| chat.with_tool(tool) }
549
+ # Handle tools - convert Raix function declarations to RubyLLM tools.
550
+ # params[:tools] already reflects `available_tools`, so only the
551
+ # functions it names are registered with the chat.
552
+ if params[:tools].present? && self.class.respond_to?(:functions)
553
+ chat.with_tools(*FunctionToolAdapter.convert_tools_for_ruby_llm(self, only: tool_names_from(params[:tools])))
385
554
  end
386
555
 
387
- # Execute the completion
388
- if stream.present?
389
- # Streaming mode
390
- if has_user_message
391
- chat.complete(&stream)
392
- else
393
- chat.ask(&stream)
394
- end
395
- nil # Return nil for streaming as per original behavior
396
- else
397
- # Non-streaming mode - return OpenAI-compatible response format
398
- response_message = has_user_message ? chat.complete : chat.ask
399
-
400
- # Pull through the raw provider payload when available. OpenRouter's
401
- # `id` is the only handle we have to look up authoritative billing
402
- # cost via /api/v1/generation, and callers that watch the response
403
- # snapshot for `model` / cached-token counts shouldn't have to break
404
- # out of the OpenAI-compatible shape to get them.
405
- raw_body = response_message.raw.respond_to?(:body) ? response_message.raw.body : nil
406
- raw_body = {} unless raw_body.is_a?(Hash)
407
-
408
- usage_payload = {
409
- "prompt_tokens" => response_message.input_tokens,
410
- "completion_tokens" => response_message.output_tokens,
411
- "total_tokens" => (response_message.input_tokens || 0) + (response_message.output_tokens || 0)
412
- }
413
-
414
- # Merge prompt_tokens_details / completion_tokens_details (cached tokens,
415
- # reasoning tokens) when the provider supplied them.
416
- if (upstream_usage = raw_body["usage"]).is_a?(Hash)
417
- upstream_usage.each do |key, value|
418
- next if usage_payload.key?(key)
419
-
420
- usage_payload[key] = value
421
- end
422
- end
556
+ # Execute the completion. Raix drives the tool loop itself (see
557
+ # #chat_completion), so this asks RubyLLM for exactly one completion and
558
+ # leaves any tool calls in the response unexecuted. A streaming request
559
+ # yields chunks to the block and still returns the assembled message, so
560
+ # tool calls made mid-stream get dispatched like any other.
561
+ response_message = stream.present? ? chat.generate(&stream) : chat.generate
562
+ return nil if response_message.nil?
563
+
564
+ # Pull through the raw provider payload when available. OpenRouter's
565
+ # `id` is the only handle we have to look up authoritative billing
566
+ # cost via /api/v1/generation, and callers that watch the response
567
+ # snapshot for `model` / cached-token counts shouldn't have to break
568
+ # out of the OpenAI-compatible shape to get them.
569
+ raw_body = response_message.raw.respond_to?(:body) ? response_message.raw.body : nil
570
+ raw_body = {} unless raw_body.is_a?(Hash)
571
+ upstream_usage = raw_body["usage"].is_a?(Hash) ? raw_body["usage"] : {}
572
+
573
+ # Prefer the provider's own counts. RubyLLM's input figure excludes cached
574
+ # tokens, and only the provider knows the authoritative totals; its
575
+ # prompt_tokens_details / completion_tokens_details ride along untouched.
576
+ tokens = response_message.tokens
577
+ prompt_tokens = upstream_usage["prompt_tokens"] || [tokens.input, tokens.cache_read, tokens.cache_write].compact.sum
578
+ completion_tokens = upstream_usage["completion_tokens"] || tokens.output
579
+ usage_payload = upstream_usage.merge(
580
+ "prompt_tokens" => prompt_tokens,
581
+ "completion_tokens" => completion_tokens,
582
+ "total_tokens" => upstream_usage["total_tokens"] || (prompt_tokens.to_i + completion_tokens.to_i)
583
+ )
423
584
 
424
- {
425
- "id" => raw_body["id"],
426
- "model" => raw_body["model"] || response_message.model_id,
427
- "provider" => raw_body["provider"],
428
- "choices" => [
429
- {
430
- "message" => {
431
- "role" => "assistant",
432
- "content" => response_message.content,
433
- "tool_calls" => response_message.tool_calls
434
- },
435
- "finish_reason" => response_message.tool_call? ? "tool_calls" : "stop"
436
- }
437
- ],
438
- "usage" => usage_payload
439
- }
585
+ # The assistant turn carries what a continuation round has to replay:
586
+ # tool calls with their ids and signatures, plus any signed reasoning.
587
+ message = {
588
+ "role" => "assistant",
589
+ "content" => response_message.content,
590
+ "tool_calls" => serialize_tool_calls(response_message.tool_calls)
591
+ }
592
+ message["raw_reasoning"] = response_message.raw_reasoning if response_message.raw_reasoning
593
+ if (thinking = response_message.thinking)
594
+ message["thinking"] = thinking.text
595
+ message["thinking_signature"] = thinking.signature
440
596
  end
597
+
598
+ {
599
+ "id" => raw_body["id"],
600
+ "model" => raw_body["model"] || response_message.model,
601
+ "provider" => raw_body["provider"],
602
+ "choices" => [
603
+ {
604
+ "message" => message,
605
+ "finish_reason" => response_message.tool_call? ? "tool_calls" : "stop"
606
+ }
607
+ ],
608
+ "usage" => usage_payload
609
+ }
610
+ rescue RubyLLM::ToolCallParseError => e
611
+ # RubyLLM rejects a response whose tool arguments are not valid JSON before
612
+ # Raix ever sees a Message. Rebuild the assistant turn from the raw payload
613
+ # so the loop can answer each call with a tool error instead of aborting.
614
+ response = assistant_turn_from_raw_response(e.response)
615
+ raise e unless response
616
+
617
+ response
441
618
  rescue StandardError => e
442
619
  warn "RubyLLM request failed: #{e.message}"
443
620
  raise e
444
621
  end
445
622
 
623
+ # The OpenAI-compatible response hash for a provider payload RubyLLM could
624
+ # not parse into a Message, or nil when the payload has no tool calls to
625
+ # recover. Arguments stay as the provider sent them.
626
+ def assistant_turn_from_raw_response(raw)
627
+ body = raw.respond_to?(:body) ? raw.body : nil
628
+ return unless body.is_a?(Hash)
629
+
630
+ raw_message = body.dig("choices", 0, "message")
631
+ return unless raw_message.is_a?(Hash) && raw_message["tool_calls"].is_a?(Array)
632
+
633
+ # Replay keys tool calls by id, so a payload with missing or duplicate
634
+ # ids cannot be recovered into a valid exchange.
635
+ ids = raw_message["tool_calls"].map { |tool_call| tool_call["id"].to_s }
636
+ return if ids.any?(&:empty?) || ids.uniq.size != ids.size
637
+
638
+ # Keep the signed state RubyLLM would have extracted: OpenRouter's
639
+ # reasoning_details array and Gemini's per-call thought signature, which
640
+ # the wire nests under extra_content.google.
641
+ message = {
642
+ "role" => "assistant",
643
+ "content" => raw_message["content"],
644
+ "tool_calls" => raw_message["tool_calls"].map do |tool_call|
645
+ signature = tool_call.dig("extra_content", "google", "thought_signature")
646
+ signature ? tool_call.merge("thought_signature" => signature) : tool_call
647
+ end
648
+ }
649
+ message["raw_reasoning"] = raw_message["reasoning_details"] if raw_message["reasoning_details"].is_a?(Array)
650
+
651
+ {
652
+ "id" => body["id"],
653
+ "model" => body["model"],
654
+ "provider" => body["provider"],
655
+ "choices" => [{ "message" => message, "finish_reason" => "tool_calls" }],
656
+ "usage" => body["usage"].is_a?(Hash) ? body["usage"] : {}
657
+ }
658
+ end
659
+
446
660
  def determine_provider(model, openai_override)
447
661
  return :openai if openai_override
448
662
  return :openai if model.to_s.match?(/^gpt-/) || model.to_s.match?(/^o\d/)
@@ -450,5 +664,57 @@ module Raix
450
664
  # Default to openrouter for model IDs with provider prefix
451
665
  :openrouter
452
666
  end
667
+
668
+ # Renders RubyLLM's tool calls (a Hash keyed by call id whose values are
669
+ # RubyLLM::ToolCall) as OpenAI's array-of-hashes shape, which is what the
670
+ # response hash Raix hands back to callers — and its own tool loop — reads.
671
+ def serialize_tool_calls(tool_calls)
672
+ return nil if tool_calls.blank?
673
+
674
+ tool_calls.values.map do |tool_call|
675
+ serialized = {
676
+ "id" => tool_call.id,
677
+ "type" => "function",
678
+ "function" => {
679
+ "name" => tool_call.name,
680
+ "arguments" => tool_call.arguments.to_json
681
+ }
682
+ }
683
+ serialized["thought_signature"] = tool_call.thought_signature if tool_call.thought_signature
684
+ serialized
685
+ end
686
+ end
687
+
688
+ # Arguments replayed from a recorded tool call. RubyLLM re-serializes them
689
+ # as JSON, so a payload the provider sent malformed (already answered with
690
+ # a tool error) is replayed as an empty object rather than raising again.
691
+ def parse_replayed_arguments(arguments)
692
+ return {} if arguments.blank?
693
+
694
+ parsed = JSON.parse(arguments)
695
+ parsed.is_a?(Hash) ? parsed : {}
696
+ rescue JSON::ParserError
697
+ {}
698
+ end
699
+
700
+ # Raix's transcript stores assistant tool calls in OpenAI's array-of-hashes
701
+ # shape (`[{ id:, type:, function: { name:, arguments: } }]`), but RubyLLM's
702
+ # providers format tool calls from a Hash keyed by call id whose values
703
+ # respond to #id/#name/#arguments (RubyLLM::ToolCall). Translate so a
704
+ # transcript that already contains tool exchanges can be replayed back into
705
+ # a fresh RubyLLM chat on every continuation round, including the forced
706
+ # final completion after a max_tool_calls cap breach or
707
+ # stop_tool_calls_and_respond!.
708
+ def normalize_tool_calls_for_ruby_llm(tool_calls)
709
+ return tool_calls if tool_calls.is_a?(Hash) && tool_calls.values.all?(RubyLLM::ToolCall)
710
+
711
+ Array(tool_calls).each_with_object({}) do |raw, acc|
712
+ tc = raw.respond_to?(:with_indifferent_access) ? raw.with_indifferent_access : raw
713
+ function = tc[:function] || {}
714
+ arguments = function[:arguments]
715
+ arguments = parse_replayed_arguments(arguments) if arguments.is_a?(String)
716
+ acc[tc[:id]] = RubyLLM::ToolCall.new(id: tc[:id], name: function[:name], arguments: arguments || {}, thought_signature: tc[:thought_signature])
717
+ end
718
+ end
453
719
  end
454
720
  end