lex-llm 0.7.3 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +106 -0
- data/RULES.md +97 -0
- data/lib/legion/extensions/llm/auto_registration.rb +4 -13
- data/lib/legion/extensions/llm/canonical/chunk.rb +106 -92
- data/lib/legion/extensions/llm/canonical/content_block.rb +66 -75
- data/lib/legion/extensions/llm/canonical/message.rb +52 -67
- data/lib/legion/extensions/llm/canonical/params.rb +59 -33
- data/lib/legion/extensions/llm/canonical/request.rb +67 -54
- data/lib/legion/extensions/llm/canonical/response.rb +61 -76
- data/lib/legion/extensions/llm/canonical/strict.rb +105 -0
- data/lib/legion/extensions/llm/canonical/thinking.rb +99 -51
- data/lib/legion/extensions/llm/canonical/tool_call.rb +42 -51
- data/lib/legion/extensions/llm/canonical/tool_definition.rb +56 -46
- data/lib/legion/extensions/llm/canonical/tool_schema.rb +15 -22
- data/lib/legion/extensions/llm/canonical/usage.rb +47 -40
- data/lib/legion/extensions/llm/canonical.rb +6 -5
- data/lib/legion/extensions/llm/configuration.rb +40 -10
- data/lib/legion/extensions/llm/connection.rb +8 -30
- data/lib/legion/extensions/llm/credential_sources.rb +32 -49
- data/lib/legion/extensions/llm/discovery/actor.rb +92 -0
- data/lib/legion/extensions/llm/discovery/pipeline.rb +604 -0
- data/lib/legion/extensions/llm/error.rb +0 -14
- data/lib/legion/extensions/llm/fleet/contract_error.rb +15 -0
- data/lib/legion/extensions/llm/fleet/envelope_validation.rb +7 -6
- data/lib/legion/extensions/llm/fleet/fleet_envelope.rb +66 -0
- data/lib/legion/extensions/llm/fleet/protocol.rb +18 -5
- data/lib/legion/extensions/llm/fleet/provider_responder.rb +58 -153
- data/lib/legion/extensions/llm/fleet/token_validator.rb +15 -21
- data/lib/legion/extensions/llm/fleet/worker_execution.rb +107 -150
- data/lib/legion/extensions/llm/inventory/errors.rb +0 -1
- data/lib/legion/extensions/llm/inventory/evidence.rb +1 -1
- data/lib/legion/extensions/llm/inventory/identity.rb +55 -39
- data/lib/legion/extensions/llm/inventory/probe_token.rb +9 -12
- data/lib/legion/extensions/llm/inventory/publisher.rb +15 -53
- data/lib/legion/extensions/llm/inventory/records.rb +82 -99
- data/lib/legion/extensions/llm/inventory/registry.rb +54 -46
- data/lib/legion/extensions/llm/inventory/snapshot.rb +7 -21
- data/lib/legion/extensions/llm/inventory/weight_reconciler.rb +249 -0
- data/lib/legion/extensions/llm/inventory/weight_schema.rb +146 -0
- data/lib/legion/extensions/llm/provider/open_ai_compatible.rb +172 -182
- data/lib/legion/extensions/llm/provider.rb +192 -315
- data/lib/legion/extensions/llm/provider_contract.rb +25 -8
- data/lib/legion/extensions/llm/provider_settings.rb +5 -26
- data/lib/legion/extensions/llm/responses/thinking_extractor.rb +8 -1
- data/lib/legion/extensions/llm/responses/tool_arguments.rb +34 -0
- data/lib/legion/extensions/llm/routing/provider_outcome.rb +25 -0
- data/lib/legion/extensions/llm/routing/records.rb +34 -17
- data/lib/legion/extensions/llm/stream_accumulator.rb +186 -270
- data/lib/legion/extensions/llm/streaming.rb +50 -35
- data/lib/legion/extensions/llm/taxonomies.rb +29 -17
- data/lib/legion/extensions/llm/transport/fleet_lane.rb +8 -10
- data/lib/legion/extensions/llm/transport/messages/fleet_error.rb +3 -2
- data/lib/legion/extensions/llm/transport/messages/fleet_request.rb +6 -8
- data/lib/legion/extensions/llm/transport/messages/fleet_response.rb +10 -9
- data/lib/legion/extensions/llm/utils.rb +23 -5
- data/lib/legion/extensions/llm/version.rb +1 -1
- data/lib/legion/extensions/llm.rb +10 -98
- data/spec/legion/extensions/llm/auto_registration_spec.rb +4 -9
- data/spec/legion/extensions/llm/canonical/chunk_spec.rb +66 -252
- data/spec/legion/extensions/llm/canonical/content_block_spec.rb +52 -197
- data/spec/legion/extensions/llm/canonical/message_spec.rb +89 -204
- data/spec/legion/extensions/llm/canonical/params_spec.rb +57 -136
- data/spec/legion/extensions/llm/canonical/request_spec.rb +81 -143
- data/spec/legion/extensions/llm/canonical/response_spec.rb +68 -204
- data/spec/legion/extensions/llm/canonical/thinking_spec.rb +68 -148
- data/spec/legion/extensions/llm/canonical/tool_call_spec.rb +59 -162
- data/spec/legion/extensions/llm/canonical/tool_definition_spec.rb +55 -191
- data/spec/legion/extensions/llm/canonical/tool_schema_spec.rb +26 -67
- data/spec/legion/extensions/llm/canonical/usage_spec.rb +46 -155
- data/spec/legion/extensions/llm/configuration_spec.rb +31 -5
- data/spec/legion/extensions/llm/conformance/canonical_type_examples.rb +106 -0
- data/spec/legion/extensions/llm/conformance/conformance.rb +10 -2
- data/spec/legion/extensions/llm/conformance/provider_translator_examples.rb +1 -1
- data/spec/legion/extensions/llm/conformance/ssot_contract_conformance_spec.rb +130 -0
- data/spec/legion/extensions/llm/conformance/ssot_contract_examples.rb +507 -0
- data/spec/legion/extensions/llm/conformance/ssot_provider_examples.rb +11 -10
- data/spec/legion/extensions/llm/credential_sources_spec.rb +12 -13
- data/spec/legion/extensions/llm/error_spec.rb +2 -12
- data/spec/legion/extensions/llm/fleet/exact_offering_spec.rb +59 -45
- data/spec/legion/extensions/llm/fleet/provider_responder_spec.rb +178 -67
- data/spec/legion/extensions/llm/fleet/token_validator_spec.rb +7 -2
- data/spec/legion/extensions/llm/fleet/worker_execution_spec.rb +101 -62
- data/spec/legion/extensions/llm/fleet_messages_spec.rb +118 -123
- data/spec/legion/extensions/llm/inventory/boot_spec.rb +4 -4
- data/spec/legion/extensions/llm/inventory/identity_spec.rb +127 -110
- data/spec/legion/extensions/llm/inventory/probe_token_spec.rb +4 -4
- data/spec/legion/extensions/llm/inventory/publisher_spec.rb +8 -61
- data/spec/legion/extensions/llm/inventory/records_spec.rb +129 -41
- data/spec/legion/extensions/llm/inventory/registry_activation_spec.rb +55 -4
- data/spec/legion/extensions/llm/inventory/registry_replacement_spec.rb +5 -4
- data/spec/legion/extensions/llm/inventory/snapshot_spec.rb +11 -5
- data/spec/legion/extensions/llm/inventory/weight_reconciler_spec.rb +312 -0
- data/spec/legion/extensions/llm/inventory/weight_schema_spec.rb +141 -0
- data/spec/legion/extensions/llm/provider/open_ai_compatible_spec.rb +105 -68
- data/spec/legion/extensions/llm/provider/open_ai_compatible_tool_calls_array_spec.rb +7 -31
- data/spec/legion/extensions/llm/provider_contract_spec.rb +10 -15
- data/spec/legion/extensions/llm/provider_spec.rb +98 -78
- data/spec/legion/extensions/llm/routing/records_spec.rb +38 -5
- data/spec/legion/extensions/llm/stream_accumulator_spec.rb +174 -144
- data/spec/legion/extensions/llm/streaming_spec.rb +27 -0
- data/spec/legion/extensions/llm/taxonomies_spec.rb +43 -16
- data/spec/legion/extensions/llm/transport/fleet_lane_spec.rb +1 -1
- data/spec/legion/extensions/llm/utils_spec.rb +26 -7
- data/spec/legion/extensions/llm_base_contract_spec.rb +55 -90
- data/spec/legion/extensions/llm_extension_spec.rb +5 -5
- data/spec/support/fake_llm_provider.rb +45 -39
- data/spec/support/fake_ssot_harness.rb +7 -2
- data/spec/support/ssot_registry_helpers.rb +3 -2
- metadata +15 -54
- data/lib/legion/extensions/llm/agent.rb +0 -366
- data/lib/legion/extensions/llm/aliases.json +0 -436
- data/lib/legion/extensions/llm/aliases.rb +0 -67
- data/lib/legion/extensions/llm/attachment.rb +0 -229
- data/lib/legion/extensions/llm/chat.rb +0 -354
- data/lib/legion/extensions/llm/chunk.rb +0 -10
- data/lib/legion/extensions/llm/content.rb +0 -81
- data/lib/legion/extensions/llm/context.rb +0 -33
- data/lib/legion/extensions/llm/embedding.rb +0 -33
- data/lib/legion/extensions/llm/image.rb +0 -109
- data/lib/legion/extensions/llm/inventory/capabilities.rb +0 -40
- data/lib/legion/extensions/llm/inventory/scoped_refresher.rb +0 -311
- data/lib/legion/extensions/llm/message.rb +0 -118
- data/lib/legion/extensions/llm/mime_type.rb +0 -75
- data/lib/legion/extensions/llm/model/info.rb +0 -286
- data/lib/legion/extensions/llm/model/modalities.rb +0 -26
- data/lib/legion/extensions/llm/model/pricing.rb +0 -52
- data/lib/legion/extensions/llm/model/pricing_category.rb +0 -50
- data/lib/legion/extensions/llm/model/pricing_tier.rb +0 -37
- data/lib/legion/extensions/llm/model.rb +0 -11
- data/lib/legion/extensions/llm/models.json +0 -57313
- data/lib/legion/extensions/llm/models.rb +0 -530
- data/lib/legion/extensions/llm/models_schema.json +0 -168
- data/lib/legion/extensions/llm/moderation.rb +0 -60
- data/lib/legion/extensions/llm/registry_event_builder.rb +0 -141
- data/lib/legion/extensions/llm/registry_publisher.rb +0 -107
- data/lib/legion/extensions/llm/responses/chat_response.rb +0 -43
- data/lib/legion/extensions/llm/responses/embedding_response.rb +0 -38
- data/lib/legion/extensions/llm/responses/stream_chunk.rb +0 -43
- data/lib/legion/extensions/llm/routing/lane_key.rb +0 -66
- data/lib/legion/extensions/llm/routing/model_offering.rb +0 -241
- data/lib/legion/extensions/llm/routing/offering_registry.rb +0 -101
- data/lib/legion/extensions/llm/routing/registry_event.rb +0 -167
- data/lib/legion/extensions/llm/thinking.rb +0 -53
- data/lib/legion/extensions/llm/tokens.rb +0 -51
- data/lib/legion/extensions/llm/tool_call.rb +0 -34
- data/lib/legion/extensions/llm/transcription.rb +0 -39
- data/lib/legion/extensions/llm/transport/messages/registry_event.rb +0 -44
- data/spec/legion/extensions/llm/agent_spec.rb +0 -179
- data/spec/legion/extensions/llm/attachment_spec.rb +0 -25
- data/spec/legion/extensions/llm/conformance/fixtures/ssot_identity_vectors.json +0 -84
- data/spec/legion/extensions/llm/context_spec.rb +0 -127
- data/spec/legion/extensions/llm/inventory/capabilities_spec.rb +0 -43
- data/spec/legion/extensions/llm/inventory/scoped_refresher_spec.rb +0 -337
- data/spec/legion/extensions/llm/message_spec.rb +0 -64
- data/spec/legion/extensions/llm/model/info_spec.rb +0 -222
- data/spec/legion/extensions/llm/models_spec.rb +0 -104
- data/spec/legion/extensions/llm/registry_event_builder_spec.rb +0 -68
- data/spec/legion/extensions/llm/registry_publisher_spec.rb +0 -22
- data/spec/legion/extensions/llm/responses/response_objects_spec.rb +0 -75
- data/spec/legion/extensions/llm/routing/model_offering_spec.rb +0 -281
- data/spec/legion/extensions/llm/routing/offering_registry_spec.rb +0 -50
- data/spec/legion/extensions/llm/routing/registry_event_spec.rb +0 -120
|
@@ -4,7 +4,12 @@ module Legion
|
|
|
4
4
|
module Extensions
|
|
5
5
|
module Llm
|
|
6
6
|
class Provider
|
|
7
|
-
# Shared OpenAI-compatible HTTP payload and response adapter
|
|
7
|
+
# Shared OpenAI-compatible HTTP payload and response adapter — the
|
|
8
|
+
# reference implementation for the OpenAI wire dialect (08 R3).
|
|
9
|
+
# Renders FROM Canonical (render_payload) and parses TO Canonical
|
|
10
|
+
# (parse_completion_response / build_chunk). Provider-dialect
|
|
11
|
+
# translation (usage spellings, JSON-string arguments, reasoning
|
|
12
|
+
# fields) lives here, at the edge (03 O03a).
|
|
8
13
|
module OpenAICompatible
|
|
9
14
|
def stream_usage_supported? = false
|
|
10
15
|
def completion_url = '/v1/chat/completions'
|
|
@@ -20,68 +25,94 @@ module Legion
|
|
|
20
25
|
|
|
21
26
|
private
|
|
22
27
|
|
|
23
|
-
def render_payload(messages, tools:,
|
|
28
|
+
def render_payload(messages, tools:, model:, stream:, schema:, thinking:, params:, tool_prefs:) # rubocop:disable Metrics/ParameterLists
|
|
24
29
|
payload = {
|
|
25
|
-
model: model
|
|
30
|
+
model: model,
|
|
26
31
|
messages: format_openai_messages(messages),
|
|
27
|
-
temperature:
|
|
32
|
+
temperature: maybe_normalize_temperature(params),
|
|
28
33
|
stream: stream,
|
|
29
34
|
tools: format_openai_tools(tools),
|
|
30
35
|
tool_choice: openai_tool_choice(tool_prefs),
|
|
31
36
|
response_format: openai_response_format(schema),
|
|
32
37
|
reasoning_effort: openai_reasoning_effort(thinking)
|
|
33
38
|
}.compact
|
|
39
|
+
payload.merge!(openai_payload_params(params))
|
|
34
40
|
payload[:stream_options] = { include_usage: true } if stream && stream_usage_supported?
|
|
35
41
|
payload
|
|
36
42
|
end
|
|
37
43
|
|
|
44
|
+
# Canonical Params → OpenAI wire keys (edge translation, O03a).
|
|
45
|
+
def openai_payload_params(params)
|
|
46
|
+
return {} unless params
|
|
47
|
+
|
|
48
|
+
{
|
|
49
|
+
max_tokens: params.max_tokens,
|
|
50
|
+
top_p: params.top_p,
|
|
51
|
+
top_k: params.top_k,
|
|
52
|
+
stop: params.stop_sequences,
|
|
53
|
+
seed: params.seed,
|
|
54
|
+
frequency_penalty: params.frequency_penalty,
|
|
55
|
+
presence_penalty: params.presence_penalty,
|
|
56
|
+
response_format: openai_response_format_value(params.response_format)
|
|
57
|
+
}.compact
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
# response_format travels in its provider wire shape (client
|
|
61
|
+
# translator concern): a wire Hash passes through, a mode String is
|
|
62
|
+
# wrapped.
|
|
63
|
+
def openai_response_format_value(response_format)
|
|
64
|
+
return nil if response_format.nil?
|
|
65
|
+
return response_format if response_format.is_a?(::Hash)
|
|
66
|
+
|
|
67
|
+
{ type: response_format.to_s }
|
|
68
|
+
end
|
|
69
|
+
|
|
38
70
|
def format_openai_messages(messages)
|
|
39
71
|
messages.map do |message|
|
|
40
72
|
{
|
|
41
73
|
role: message.role.to_s,
|
|
42
|
-
content: openai_content(message.content
|
|
74
|
+
content: openai_content(message.content),
|
|
43
75
|
tool_call_id: message.tool_call_id,
|
|
44
76
|
tool_calls: format_openai_tool_calls(message.tool_calls)
|
|
45
77
|
}.compact
|
|
46
78
|
end
|
|
47
79
|
end
|
|
48
80
|
|
|
49
|
-
|
|
50
|
-
|
|
51
|
-
|
|
52
|
-
|
|
81
|
+
# L10: the role-parameterized sanitizer is deleted — text content
|
|
82
|
+
# passes through verbatim in every role (the decision comment
|
|
83
|
+
# below documents why).
|
|
84
|
+
def openai_content(content)
|
|
85
|
+
return content.map { |block| openai_content(block) } if content.is_a?(::Array)
|
|
86
|
+
return content.to_s if content.nil? || content.is_a?(::String)
|
|
53
87
|
|
|
54
|
-
|
|
55
|
-
end
|
|
88
|
+
return content.text.to_s if content.text?
|
|
56
89
|
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
90
|
+
{
|
|
91
|
+
type: content.type.to_s,
|
|
92
|
+
text: content.text,
|
|
93
|
+
data: content.data,
|
|
94
|
+
media_type: content.media_type,
|
|
95
|
+
source_type: content.source_type,
|
|
96
|
+
name: content.name,
|
|
97
|
+
id: content.id,
|
|
98
|
+
input: content.input,
|
|
99
|
+
tool_use_id: content.tool_use_id,
|
|
100
|
+
is_error: content.is_error
|
|
101
|
+
}.compact
|
|
64
102
|
end
|
|
65
103
|
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
|
|
69
|
-
|
|
70
|
-
|
|
71
|
-
|
|
72
|
-
# for client-facing responses; the OpenAI compat layer passes them through.
|
|
73
|
-
text
|
|
74
|
-
end
|
|
104
|
+
# Thinking-tag pass-through decision (L10: the vestigial sanitizer
|
|
105
|
+
# is deleted, the decision stays): qwen3.6 outputs thinking in
|
|
106
|
+
# tags and expects to see its own reasoning on subsequent rounds.
|
|
107
|
+
# The Anthropic API layer separates thinking into distinct content
|
|
108
|
+
# blocks for client-facing responses; the OpenAI compat layer passes
|
|
109
|
+
# them through untouched.
|
|
75
110
|
|
|
76
111
|
def format_openai_tool_calls(tool_calls)
|
|
77
112
|
return nil unless tool_calls&.any?
|
|
78
113
|
|
|
79
|
-
# Array
|
|
80
|
-
|
|
81
|
-
# this renderer depending on caller.
|
|
82
|
-
calls = tool_calls.is_a?(Hash) ? tool_calls.values : Array(tool_calls)
|
|
83
|
-
|
|
84
|
-
calls.map do |tool_call|
|
|
114
|
+
# Array<Canonical::ToolCall> only (the legacy Hash shape is deleted).
|
|
115
|
+
tool_calls.map do |tool_call|
|
|
85
116
|
{
|
|
86
117
|
id: tool_call.id,
|
|
87
118
|
type: 'function',
|
|
@@ -94,7 +125,7 @@ module Legion
|
|
|
94
125
|
end
|
|
95
126
|
|
|
96
127
|
def format_openai_tools(tools)
|
|
97
|
-
return nil if tools.empty?
|
|
128
|
+
return nil if tools.nil? || tools.empty?
|
|
98
129
|
|
|
99
130
|
tools.values.map do |tool|
|
|
100
131
|
{
|
|
@@ -124,175 +155,121 @@ module Legion
|
|
|
124
155
|
end
|
|
125
156
|
|
|
126
157
|
def openai_reasoning_effort(thinking)
|
|
127
|
-
return nil unless thinking.is_a?(
|
|
158
|
+
return nil unless thinking.is_a?(Canonical::Thinking::Config)
|
|
128
159
|
|
|
129
|
-
thinking
|
|
160
|
+
thinking.effort
|
|
130
161
|
end
|
|
131
162
|
|
|
163
|
+
# One response-parse boundary (08 R2): returns Canonical::Response.
|
|
132
164
|
def parse_completion_response(response)
|
|
133
165
|
body = response.body
|
|
134
166
|
choice = Array(body['choices']).first || {}
|
|
135
167
|
message = choice['message'] || {}
|
|
136
|
-
usage = body['usage'] || {}
|
|
137
|
-
content, thinking = extract_thinking_from_completion(message)
|
|
138
|
-
|
|
139
|
-
Legion::Extensions::Llm::Message.new(
|
|
140
|
-
role: :assistant,
|
|
141
|
-
content: content,
|
|
142
|
-
model_id: body['model'],
|
|
143
|
-
tool_calls: parse_tool_calls(message['tool_calls']),
|
|
144
|
-
thinking: thinking,
|
|
145
|
-
input_tokens: usage['prompt_tokens'],
|
|
146
|
-
output_tokens: usage['completion_tokens'],
|
|
147
|
-
reasoning_tokens: usage.dig('completion_tokens_details', 'reasoning_tokens'),
|
|
148
|
-
raw: body
|
|
149
|
-
)
|
|
150
|
-
end
|
|
151
|
-
|
|
152
|
-
def extract_thinking_from_completion(message)
|
|
153
168
|
extraction = Responses::ThinkingExtractor.extract(
|
|
154
169
|
message['content'],
|
|
155
170
|
metadata: thinking_metadata(message)
|
|
156
171
|
)
|
|
157
172
|
|
|
158
|
-
|
|
159
|
-
extraction.content,
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
)
|
|
164
|
-
|
|
173
|
+
Canonical::Response.build(
|
|
174
|
+
text: extraction.content,
|
|
175
|
+
thinking: thinking_value(extraction),
|
|
176
|
+
tool_calls: parse_tool_calls(message['tool_calls']),
|
|
177
|
+
usage: openai_usage_to_canonical(body['usage']),
|
|
178
|
+
stop_reason: stop_reason_lookup(body.dig('choices', 0, 'finish_reason')),
|
|
179
|
+
model: body['model']
|
|
180
|
+
)
|
|
165
181
|
end
|
|
166
182
|
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
reasoning_signature: message['reasoning_signature']
|
|
175
|
-
}.compact
|
|
183
|
+
# One thinking-metadata key list (10 U3): both the sync and chunk
|
|
184
|
+
# paths extract through the shared ThinkingExtractor core.
|
|
185
|
+
def thinking_metadata(wire_message)
|
|
186
|
+
Responses::ThinkingExtractor::THINKING_METADATA_KEYS.each_with_object({}) do |key, hash|
|
|
187
|
+
value = wire_message[key.to_s]
|
|
188
|
+
hash[key] = value unless value.nil?
|
|
189
|
+
end
|
|
176
190
|
end
|
|
177
191
|
|
|
178
|
-
def
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
thinking: thinking,
|
|
192
|
+
def thinking_value(extraction)
|
|
193
|
+
return nil if extraction.thinking.nil? && extraction.signature.nil?
|
|
194
|
+
|
|
195
|
+
Canonical::Thinking.build(content: extraction.thinking, signature: extraction.signature)
|
|
196
|
+
end
|
|
197
|
+
|
|
198
|
+
# OpenAI wire usage → canonical keys (edge translation, O03a).
|
|
199
|
+
def openai_usage_to_canonical(usage)
|
|
200
|
+
return nil if usage.to_h.empty?
|
|
201
|
+
|
|
202
|
+
Canonical::Usage.build(
|
|
190
203
|
input_tokens: usage['prompt_tokens'],
|
|
191
204
|
output_tokens: usage['completion_tokens'],
|
|
192
|
-
|
|
205
|
+
cache_read_tokens: usage.dig('prompt_tokens_details', 'cached_tokens') || usage.dig('input_tokens_details', 'cached_tokens'),
|
|
206
|
+
thinking_tokens: usage.dig('completion_tokens_details', 'reasoning_tokens') || usage.dig('output_tokens_details', 'reasoning_tokens')
|
|
193
207
|
)
|
|
194
208
|
end
|
|
195
209
|
|
|
196
|
-
|
|
197
|
-
|
|
210
|
+
# One chunk-parse boundary (08 R2): returns a Canonical::Chunk or an
|
|
211
|
+
# Array of them. In-band think tags are NOT split here — tags split
|
|
212
|
+
# across chunks are the accumulator's stateful job (U1); per-chunk
|
|
213
|
+
# metadata (e.g. reasoning_content) is separated here, statelessly.
|
|
214
|
+
def build_chunk(data)
|
|
215
|
+
choice = Array(data['choices']).first || {}
|
|
216
|
+
delta = choice['delta'] || {}
|
|
217
|
+
usage = data['usage'] || {}
|
|
198
218
|
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
219
|
+
chunks = []
|
|
220
|
+
metadata = thinking_metadata(delta)
|
|
221
|
+
if metadata.any?
|
|
222
|
+
extraction = Responses::ThinkingExtractor.extract(nil, metadata:)
|
|
223
|
+
if extraction.thinking
|
|
224
|
+
chunks << Canonical::Chunk.thinking_delta(
|
|
225
|
+
delta: extraction.thinking, request_id: nil, signature: extraction.signature
|
|
226
|
+
)
|
|
227
|
+
end
|
|
208
228
|
end
|
|
209
|
-
end
|
|
210
|
-
|
|
211
|
-
def extract_thinking_from_chunk(delta)
|
|
212
|
-
reasoning = delta['reasoning_content'] || delta['reasoning']
|
|
213
229
|
content = delta['content']
|
|
230
|
+
chunks << Canonical::Chunk.text_delta(delta: content.to_s, request_id: nil) if content
|
|
231
|
+
chunks.concat(parse_streaming_tool_calls(delta['tool_calls']))
|
|
232
|
+
canonical_usage = openai_usage_to_canonical(usage)
|
|
233
|
+
chunks << Canonical::Chunk.usage_chunk(usage: canonical_usage, request_id: nil) if canonical_usage
|
|
214
234
|
|
|
215
|
-
if
|
|
216
|
-
|
|
235
|
+
if chunks.empty?
|
|
236
|
+
nil
|
|
217
237
|
else
|
|
218
|
-
|
|
238
|
+
(chunks.size == 1 ? chunks.first : chunks)
|
|
219
239
|
end
|
|
220
240
|
end
|
|
221
241
|
|
|
222
|
-
def
|
|
223
|
-
return
|
|
242
|
+
def parse_streaming_tool_calls(tool_calls)
|
|
243
|
+
return [] unless tool_calls&.any?
|
|
224
244
|
|
|
225
|
-
tool_calls.
|
|
245
|
+
tool_calls.filter_map do |call|
|
|
226
246
|
function = call.fetch('function', {})
|
|
227
247
|
name = function['name']
|
|
228
|
-
id = call['id']
|
|
229
|
-
|
|
230
|
-
[
|
|
231
|
-
|
|
232
|
-
Legion::Extensions::Llm::ToolCall.new(
|
|
233
|
-
id: id&.to_s,
|
|
234
|
-
name: name,
|
|
235
|
-
arguments: parse_tool_arguments(function['arguments'])
|
|
236
|
-
)
|
|
237
|
-
]
|
|
238
|
-
end
|
|
239
|
-
end
|
|
240
|
-
|
|
241
|
-
def parse_tool_arguments(arguments)
|
|
242
|
-
return {} if arguments.nil? || arguments == ''
|
|
243
|
-
return arguments if arguments.is_a?(Hash)
|
|
244
|
-
|
|
245
|
-
Legion::JSON.parse(arguments, symbolize_names: false)
|
|
246
|
-
rescue Legion::JSON::ParseError => e
|
|
247
|
-
handle_exception(e, level: :warn, handled: true, operation: 'llm.provider.parse_tool_arguments')
|
|
248
|
-
{}
|
|
249
|
-
end
|
|
248
|
+
id = call['id']
|
|
249
|
+
index = call['index']
|
|
250
|
+
arguments_fragment = function['arguments'].to_s
|
|
251
|
+
next nil if id.nil? && name.nil? && index.nil? && arguments_fragment.empty?
|
|
250
252
|
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
Legion::Extensions::Llm::Model::Info.from_hash(
|
|
255
|
-
id: model.fetch('id'),
|
|
256
|
-
name: model['id'],
|
|
257
|
-
provider: provider,
|
|
258
|
-
created_at: model_created_at(model['created']),
|
|
259
|
-
capabilities: critical_capabilities,
|
|
260
|
-
modalities: modalities_for_capabilities(critical_capabilities),
|
|
261
|
-
metadata: model
|
|
253
|
+
Canonical::Chunk.tool_call_delta(
|
|
254
|
+
tool_call: { id: id, name: name, arguments: arguments_fragment, index: index },
|
|
255
|
+
request_id: nil
|
|
262
256
|
)
|
|
263
257
|
end
|
|
264
258
|
end
|
|
265
259
|
|
|
266
|
-
|
|
267
|
-
|
|
268
|
-
|
|
269
|
-
|
|
270
|
-
def critical_capabilities_for(capabilities, model)
|
|
271
|
-
return [] unless capabilities
|
|
272
|
-
return capabilities.critical_capabilities_for(model) if capabilities.respond_to?(:critical_capabilities_for)
|
|
273
|
-
|
|
274
|
-
{
|
|
275
|
-
'streaming' => :streaming?,
|
|
276
|
-
'function_calling' => :functions?,
|
|
277
|
-
'vision' => :vision?,
|
|
278
|
-
'embeddings' => :embeddings?,
|
|
279
|
-
'moderation' => :moderation?,
|
|
280
|
-
'image' => :images?,
|
|
281
|
-
'audio_transcription' => :audio_transcription?
|
|
282
|
-
}.filter_map do |capability, predicate|
|
|
283
|
-
capability if capabilities.respond_to?(predicate) && capabilities.public_send(predicate, model)
|
|
284
|
-
end
|
|
285
|
-
end
|
|
260
|
+
# Sync tool calls: the ONE strict arguments parser (10 U2).
|
|
261
|
+
def parse_tool_calls(tool_calls)
|
|
262
|
+
return [] unless tool_calls&.any?
|
|
286
263
|
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
264
|
+
tool_calls.map do |call|
|
|
265
|
+
function = call.fetch('function', {})
|
|
266
|
+
name = function['name']
|
|
267
|
+
id = call['id'] || name || call['index']
|
|
268
|
+
Canonical::ToolCall.build(
|
|
269
|
+
name: name.to_s,
|
|
270
|
+
id: id&.to_s,
|
|
271
|
+
arguments: Responses::ToolArguments.parse!(function['arguments'])
|
|
272
|
+
)
|
|
296
273
|
end
|
|
297
274
|
end
|
|
298
275
|
|
|
@@ -300,51 +277,64 @@ module Legion
|
|
|
300
277
|
{ model: model.respond_to?(:id) ? model.id : model, input: text, dimensions: dimensions }.compact
|
|
301
278
|
end
|
|
302
279
|
|
|
280
|
+
# 05 §3 documented artifact: { text:, model:, embedding: Array<Float>,
|
|
281
|
+
# usage: Canonical::Usage }.
|
|
303
282
|
def parse_embedding_response(response, model:, text:)
|
|
304
283
|
vectors = response.body.fetch('data', []).map { |item| item['embedding'] }
|
|
305
284
|
vectors = vectors.first unless text.is_a?(Array)
|
|
306
285
|
usage = response.body['usage'] || {}
|
|
307
286
|
|
|
308
|
-
|
|
309
|
-
|
|
287
|
+
{
|
|
288
|
+
text: text,
|
|
289
|
+
model: model.respond_to?(:id) ? model.id : model,
|
|
290
|
+
embedding: vectors,
|
|
291
|
+
usage: openai_usage_to_canonical(usage)
|
|
292
|
+
}.compact
|
|
310
293
|
end
|
|
311
294
|
|
|
312
295
|
def render_moderation_payload(input, model:)
|
|
313
|
-
|
|
296
|
+
input_text = input.is_a?(::Array) ? input.map(&:text).join("\n") : input
|
|
297
|
+
{ model: model, input: input_text }.compact
|
|
314
298
|
end
|
|
315
299
|
|
|
300
|
+
# 05 §3 documented artifact: { model:, result: { flagged:, categories: } }.
|
|
316
301
|
def parse_moderation_response(response, model:)
|
|
317
|
-
|
|
318
|
-
|
|
302
|
+
result = Array(response.body['results']).first || {}
|
|
303
|
+
{
|
|
304
|
+
model: response.body['model'] || model,
|
|
305
|
+
result: {
|
|
306
|
+
flagged: result['flagged'] == true,
|
|
307
|
+
categories: result['categories'] || {}
|
|
308
|
+
}
|
|
309
|
+
}
|
|
319
310
|
end
|
|
320
311
|
|
|
321
312
|
def render_image_payload(prompt, model:, size:, with:, mask:, params:) # rubocop:disable Metrics/ParameterLists
|
|
322
313
|
{ model: model, prompt: prompt, size: size, image: with, mask: mask }.merge(params).compact
|
|
323
314
|
end
|
|
324
315
|
|
|
316
|
+
# 05 §3 documented artifact: { model:, image:, size: }.
|
|
325
317
|
def parse_image_response(response, model:)
|
|
326
318
|
image = response.body.fetch('data', []).first || {}
|
|
327
|
-
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
usage: response.body['usage'] || {}
|
|
333
|
-
)
|
|
319
|
+
{
|
|
320
|
+
model: model,
|
|
321
|
+
image: image['url'] || image['b64_json'],
|
|
322
|
+
size: image['size']
|
|
323
|
+
}.compact
|
|
334
324
|
end
|
|
335
325
|
|
|
336
326
|
def render_transcription_payload(file_part, model:, language:, **options)
|
|
337
327
|
{ model: model, file: file_part, language: language }.merge(options).compact
|
|
338
328
|
end
|
|
339
329
|
|
|
330
|
+
# 05 §3 documented artifact: { model:, text:, language:, duration: }.
|
|
340
331
|
def parse_transcription_response(response, model:)
|
|
341
|
-
|
|
342
|
-
text: response.body['text'],
|
|
332
|
+
{
|
|
343
333
|
model: model,
|
|
334
|
+
text: response.body['text'],
|
|
344
335
|
language: response.body['language'],
|
|
345
|
-
duration: response.body['duration']
|
|
346
|
-
|
|
347
|
-
)
|
|
336
|
+
duration: response.body['duration']
|
|
337
|
+
}.compact
|
|
348
338
|
end
|
|
349
339
|
end
|
|
350
340
|
end
|