lex-llm-mlx 0.3.14 → 0.5.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,9 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  require 'digest'
4
+ require 'time'
5
+ require 'uri'
6
+ require 'faraday'
4
7
 
5
8
  begin
6
9
  require 'legion/extensions/actors/every'
@@ -8,159 +11,1002 @@ rescue LoadError => e
8
11
  warn(e.message) if $VERBOSE
9
12
  end
10
13
 
11
- begin
12
- require 'legion/extensions/llm/inventory/scoped_refresher'
13
- rescue LoadError => e
14
- warn(e.message) if $VERBOSE
14
+ unless defined?(Legion::Extensions::Actors::Every)
15
+ raise LoadError, 'LegionIO actor runtime is required for MLX discovery refresh'
15
16
  end
16
17
 
17
- return unless defined?(Legion::Extensions::Actors::Every)
18
+ require 'legion/extensions/llm/inventory/publisher'
19
+ require 'legion/extensions/llm/inventory/scoped_refresher'
20
+ require 'legion/extensions/llm/inventory/identity'
21
+ require 'legion/extensions/llm/inventory/records'
22
+ require 'legion/extensions/llm/inventory/evidence'
23
+ require 'legion/extensions/llm/inventory/probe_coordinator'
24
+ require 'legion/extensions/llm/inventory/weight_reconciler'
25
+ require 'legion/extensions/llm/routing/provider_outcome'
26
+ require 'legion/extensions/llm/taxonomies'
27
+ require 'legion/extensions/llm/capabilities'
28
+ require 'legion/extensions/llm/mlx/provider'
18
29
 
19
30
  module Legion
20
31
  module Extensions
21
32
  module Llm
22
33
  module Mlx
23
34
  module Actor
24
- class DiscoveryRefresh < Legion::Extensions::Actors::Every # rubocop:disable Style/Documentation, Metrics/ClassLength
25
- include Legion::Logging::Helper
35
+ # Evidence and offering-draft construction — included by DiscoveryRefresh.
36
+ module EvidenceBuilding
37
+ EMBEDDING_PATTERN = /embed|bge|e5|nomic/i
38
+ # Protocol-required evidence source: the default_false taxonomy member.
39
+ UNKNOWN_EVIDENCE_SRC = :default_false
26
40
 
27
- if defined?(Legion::Extensions::Llm::Inventory::ScopedRefresher)
28
- include Legion::Extensions::Llm::Inventory::ScopedRefresher
41
+ private
42
+
43
+ def embedding_model?(model_id:)
44
+ model_id.to_s.match?(EMBEDDING_PATTERN)
29
45
  end
30
46
 
31
- def self.every_seconds = 60
47
+ def build_offering_draft(model_id:, model_data:, instance_cfg:, instance_key:)
48
+ tier = instance_cfg[:tier] || :local
49
+ embed_supported = embedding_model?(model_id: model_id)
50
+ weight_inputs = Legion::Extensions::Llm::Inventory::WeightSchema.weight_inputs(
51
+ settings: Legion::Settings,
52
+ instance_key: instance_key,
53
+ provider_native_key: model_id,
54
+ model: model_id,
55
+ tier: tier
56
+ )
32
57
 
33
- def runner_class = self.class
34
- def runner_function = 'manual'
35
- def run_now? = true
36
- def use_runner? = false
37
- def check_subtask? = false
38
- def generate_task? = false
58
+ Legion::Extensions::Llm::Inventory::OfferingDraft.new(
59
+ provider_native_key: model_id,
60
+ model: model_id,
61
+ tier: tier,
62
+ operation_evidence: build_operation_evidence(embed_supported: embed_supported, model_id: model_id),
63
+ capability_evidence: build_capability_evidence(model_id: model_id),
64
+ context_evidence: build_context_evidence(model_data: model_data),
65
+ max_output_evidence: build_max_output_evidence(model_data: model_data),
66
+ embedding_dimensions_evidence: build_embedding_dimensions_evidence(
67
+ model_data: model_data, embed_supported: embed_supported
68
+ ),
69
+ model_revision_evidence: absent_value_evidence,
70
+ tokenizer_evidence: absent_value_evidence,
71
+ quota_domains: {},
72
+ metadata: build_offering_metadata(model_data: model_data, instance_key: instance_key),
73
+ publication_source: :provider_catalog,
74
+ weight_inputs: weight_inputs,
75
+ base_weight: Legion::Extensions::Llm::Inventory::WeightSchema.base_weight(weight_inputs)
76
+ )
77
+ end
39
78
 
40
- def time
41
- return self.class.every_seconds unless defined?(Legion::Settings)
79
+ def build_operation_evidence(embed_supported:, **)
80
+ now = Time.now.freeze
81
+ is_embedding = embed_supported
82
+ {
83
+ chat: op_evidence(operation: :chat, status: is_embedding ? :unsupported : :supported, observed_at: now),
84
+ stream_chat: op_evidence(operation: :stream_chat, status: is_embedding ? :unsupported : :supported,
85
+ observed_at: now),
86
+ embed: op_evidence(operation: :embed, status: is_embedding ? :supported : :unsupported,
87
+ observed_at: now),
88
+ image: op_evidence(operation: :image, status: :unsupported, observed_at: now),
89
+ transcribe: op_evidence(operation: :transcribe, status: :unsupported, observed_at: now),
90
+ translate: op_evidence(operation: :translate, status: :unsupported, observed_at: now),
91
+ speak: op_evidence(operation: :speak, status: :unsupported, observed_at: now),
92
+ moderate: op_evidence(operation: :moderate, status: :unsupported, observed_at: now),
93
+ count_tokens: op_evidence(operation: :count_tokens, status: :unknown, observed_at: now)
94
+ }
95
+ end
42
96
 
43
- Legion::Settings.dig(:extensions, :llm, :mlx, :discovery_interval) || self.class.every_seconds
97
+ def op_evidence(operation:, status:, observed_at:)
98
+ source = status == :unknown ? UNKNOWN_EVIDENCE_SRC : :provider_implementation
99
+ Legion::Extensions::Llm::Inventory::OperationEvidence.new(
100
+ operation: operation, status: status, source: source, observed_at: observed_at
101
+ )
44
102
  end
45
103
 
46
- def scope_key(**)
47
- { provider: :mlx }
104
+ def build_capability_evidence(model_id:)
105
+ is_embedding = embedding_model?(model_id: model_id)
106
+ caps = {
107
+ completion: cap_evidence(capability: :completion,
108
+ status: is_embedding ? :unsupported : :supported,
109
+ source: :provider_implementation),
110
+ streaming: cap_evidence(capability: :streaming,
111
+ status: is_embedding ? :unsupported : :supported,
112
+ source: :provider_implementation),
113
+ tools: cap_evidence(capability: :tools, status: :unknown, source: UNKNOWN_EVIDENCE_SRC),
114
+ thinking: cap_evidence(capability: :thinking, status: :unknown, source: UNKNOWN_EVIDENCE_SRC)
115
+ }
116
+
117
+ if is_embedding
118
+ caps[:embedding] = cap_evidence(
119
+ capability: :embedding, status: :supported, source: :provider_implementation
120
+ )
121
+ end
122
+
123
+ caps
48
124
  end
49
125
 
50
- def compute_lanes_for_scope(**)
51
- return [] unless defined?(Legion::LLM::Call::Registry)
126
+ def cap_evidence(capability:, status:, source:)
127
+ Legion::Extensions::Llm::Inventory::CapabilityEvidence.new(
128
+ capability: capability, status: status, source: source, observed_at: Time.now.freeze
129
+ )
130
+ end
131
+ end
52
132
 
53
- mlx_instances.flat_map { |entry| lanes_for_instance(entry) }
54
- rescue StandardError => e
55
- handle_exception(e, level: :warn, handled: true, operation: 'mlx.discovery_refresh.compute_lanes')
133
+ # Value-level evidence builders (context, output, dimensions, metadata) — included by DiscoveryRefresh.
134
+ module ValueEvidenceBuilding
135
+ private
136
+
137
+ def build_context_evidence(model_data:)
138
+ ctx = model_data[:max_model_len] || model_data[:context_length]
139
+ if ctx.is_a?(Integer) && ctx.positive?
140
+ Legion::Extensions::Llm::Inventory::ValueEvidence.new(
141
+ status: :known, value: ctx, source: :provider_catalog
142
+ )
143
+ else
144
+ absent_value_evidence
145
+ end
146
+ end
147
+
148
+ def build_max_output_evidence(model_data:)
149
+ max_out = model_data[:max_output_tokens] || model_data[:max_completion_tokens]
150
+ if max_out.is_a?(Integer) && max_out.positive?
151
+ Legion::Extensions::Llm::Inventory::ValueEvidence.new(
152
+ status: :known, value: max_out, source: :provider_catalog
153
+ )
154
+ else
155
+ absent_value_evidence
156
+ end
157
+ end
158
+
159
+ def build_embedding_dimensions_evidence(model_data:, embed_supported:)
160
+ unless embed_supported
161
+ return Legion::Extensions::Llm::Inventory::ValueEvidence.new(status: :unknown, source: :absent)
162
+ end
163
+
164
+ dims = model_data[:embedding_dimensions]
165
+ if dims.is_a?(Array) && !dims.empty? && dims.all? { |d| d.is_a?(Integer) && d.positive? }
166
+ Legion::Extensions::Llm::Inventory::ValueEvidence.new(
167
+ status: :known, value: dims.uniq.sort, source: :provider_catalog
168
+ )
169
+ else
170
+ absent_value_evidence
171
+ end
172
+ end
173
+
174
+ def absent_value_evidence
175
+ Legion::Extensions::Llm::Inventory::ValueEvidence.new(status: :unknown, source: :absent)
176
+ end
177
+
178
+ def build_offering_metadata(model_data:, instance_key:)
179
+ meta = { raw_model: model_data[:id].to_s }
180
+ meta[:parameter_count] = model_data[:parameter_count] if model_data[:parameter_count]
181
+ meta[:quantization] = model_data[:quantization].to_s if model_data[:quantization]
182
+ meta[:instance_id] = instance_key.instance_id
183
+ meta
184
+ end
185
+ end
186
+
187
+ # Model-discovery and offering-assembly — included by DiscoveryRefresh.
188
+ #
189
+ # Rescue discipline (D16): only network and response-parse errors
190
+ # are runtime conditions that may yield no offerings — an
191
+ # unreachable /v1/models is a probe outcome, not a bug. Programming
192
+ # errors (NameError/NoMethodError/ArgumentError) are NOT rescued
193
+ # here: converting them to [] would publish zero offerings and make
194
+ # a healthy instance invisible. They propagate to the per-instance
195
+ # isolation in the tick, which logs and retries next tick.
196
+ module OfferingAssembly
197
+ private
198
+
199
+ def discover_offerings_for_instance(instance_cfg:, instance_key:)
200
+ fetch_models(instance_cfg: instance_cfg).filter_map do |model_data|
201
+ next unless model_data.is_a?(Hash)
202
+
203
+ model_id = model_data[:id].to_s
204
+ next if model_id.empty?
205
+
206
+ build_offering_draft(
207
+ model_id: model_id, model_data: model_data,
208
+ instance_cfg: instance_cfg, instance_key: instance_key
209
+ )
210
+ end
211
+ end
212
+
213
+ def fetch_models(instance_cfg:)
214
+ base_url = normalize_api_base(instance_cfg[:mlx_api_base] || instance_cfg[:endpoint])
215
+ conn = build_api_connection(base_url: base_url, instance_cfg: instance_cfg)
216
+ parsed = Legion::JSON.load(conn.get('/v1/models').body)
217
+ data = parsed.is_a?(Hash) ? parsed[:data] : nil
218
+ data.is_a?(Array) ? data : []
219
+ rescue Faraday::Error, Legion::JSON::ParseError => e
220
+ handle_exception(e, level: :warn, handled: true, operation: 'mlx.actor.fetch_models')
56
221
  []
57
222
  end
223
+ end
58
224
 
59
- def credential_hash(**)
60
- mlx_settings = Legion::Settings.dig(:extensions, :llm, :mlx) || {}
61
- Digest::SHA256.hexdigest(mlx_settings[:api_key].to_s + mlx_settings[:instances].to_s)[0, 16]
225
+ # Health checking and readiness probe lifecycle — included by DiscoveryRefresh.
226
+ module HealthProbing
227
+ private
228
+
229
+ def check_health(instance_cfg:)
230
+ base_url = normalize_api_base(instance_cfg[:mlx_api_base] || instance_cfg[:endpoint])
231
+ conn = build_health_connection(base_url: base_url, instance_cfg: instance_cfg)
232
+ response = conn.get('/health')
233
+ build_readiness_from_response(response: response, base_url: base_url)
234
+ rescue Faraday::ConnectionFailed => e
235
+ handle_exception(e, level: :warn, handled: true, operation: 'mlx.actor.check_health')
236
+ readiness_failure(reason: "MLX /health connection failed: #{e.message}", error: e)
237
+ rescue StandardError => e
238
+ handle_exception(e, level: :warn, handled: true, operation: 'mlx.actor.check_health')
239
+ readiness_failure(reason: "MLX /health error: #{e.message}", error: e)
240
+ end
241
+
242
+ def build_readiness_from_response(response:, base_url:)
243
+ Legion::Extensions::Llm::Inventory::ReadinessResult.new(
244
+ ready: response.status == 200,
245
+ reason: "MLX /health returned #{response.status}",
246
+ metadata: { status: response.status, base_url: base_url }
247
+ )
248
+ end
249
+
250
+ def readiness_failure(reason:, error:)
251
+ Legion::Extensions::Llm::Inventory::ReadinessResult.new(
252
+ ready: false, reason:, metadata: { error_class: error.class.name }
253
+ )
254
+ end
255
+
256
+ def run_cadence_probe(instance_id:, state:)
257
+ coordinator = state[:probe_coordinator]
258
+ return unless coordinator.begin_probe
259
+
260
+ probe_token = publisher.readiness_probe_started(
261
+ instance_id: instance_id, physical_id: state[:physical_id],
262
+ publisher_token: state[:publisher_token]
263
+ )
264
+ readiness = check_health(instance_cfg: state[:instance_cfg])
265
+ coordinator.finish_probe
266
+ commit_probe_result(
267
+ instance_id: instance_id, physical_id: state[:physical_id],
268
+ probe_token: probe_token, readiness: readiness, state: state
269
+ )
270
+ rescue StandardError => e
271
+ begin
272
+ coordinator&.finish_probe
273
+ rescue StandardError => finish_err
274
+ handle_exception(finish_err, level: :warn, operation: 'mlx.actor.finish_probe')
275
+ end
276
+ handle_exception(e, level: :warn, operation: 'mlx.actor.cadence_probe', instance_id: instance_id)
62
277
  end
63
278
 
64
- def manual(**)
65
- tick if respond_to?(:tick)
279
+ def handle_reactive_probe(instance_id:, request:)
280
+ state = state_mutex.synchronize { @instance_states[instance_id] }
281
+ return unless state
282
+
283
+ coordinator = state[:probe_coordinator]
284
+ return unless coordinator.begin_probe(request: request)
285
+
286
+ perform_reactive_probe(
287
+ instance_id: instance_id, request: request,
288
+ state: state, coordinator: coordinator
289
+ )
66
290
  rescue StandardError => e
67
- handle_exception(e, level: :warn, handled: true, operation: 'mlx.actor.discovery_refresh')
291
+ begin
292
+ coordinator&.finish_probe(request: request)
293
+ rescue StandardError => finish_err
294
+ handle_exception(finish_err, level: :warn, operation: 'mlx.actor.finish_probe')
295
+ end
296
+ handle_exception(e, level: :warn, operation: 'mlx.actor.reactive_probe', instance_id: instance_id)
297
+ end
298
+
299
+ def perform_reactive_probe(instance_id:, request:, state:, coordinator:)
300
+ probe_token = publisher.readiness_probe_started(
301
+ instance_id: instance_id, physical_id: state[:physical_id],
302
+ publisher_token: state[:publisher_token]
303
+ )
304
+ readiness = check_health(instance_cfg: state[:instance_cfg])
305
+ coordinator.finish_probe(request: request)
306
+ commit_probe_result(
307
+ instance_id: instance_id, physical_id: state[:physical_id],
308
+ probe_token: probe_token, readiness: readiness, state: state
309
+ )
68
310
  end
69
311
 
312
+ def commit_probe_result(instance_id:, physical_id:, probe_token:, readiness:, state:)
313
+ state_mutex.synchronize do
314
+ return unless @instance_states[instance_id].equal?(state)
315
+
316
+ if readiness.ready?
317
+ publisher.readiness_succeeded(
318
+ instance_id: instance_id, physical_id: physical_id, probe_token: probe_token
319
+ )
320
+ else
321
+ publisher.readiness_failed(
322
+ instance_id: instance_id, physical_id: physical_id,
323
+ probe_token: probe_token, reason: readiness.reason
324
+ )
325
+ end
326
+ end
327
+ end
328
+
329
+ def build_probe_enqueue(instance_id:)
330
+ proc do |request:|
331
+ handle_reactive_probe(instance_id: instance_id, request: request)
332
+ true
333
+ rescue StandardError => e
334
+ handle_exception(e, level: :warn, operation: 'mlx.actor.probe_enqueue', instance_id: instance_id)
335
+ false
336
+ end
337
+ end
338
+ end
339
+
340
+ # MLX offering comparison excludes only evidence observation telemetry.
341
+ module OfferingComparison
342
+ OFFERING_SCALAR_EVIDENCE_FIELDS = %i[
343
+ context_evidence max_output_evidence embedding_dimensions_evidence
344
+ model_revision_evidence tokenizer_evidence
345
+ ].freeze
346
+
70
347
  private
71
348
 
72
- def mlx_instances
73
- Legion::LLM::Call::Registry.all_instances.select do |e|
74
- (e[:provider] || '').to_sym == :mlx
349
+ # Inventory evidence timestamps are telemetry only: Evidence explicitly
350
+ # excludes observed_at from authority, ordering, freshness, recovery, and
351
+ # selection. Compare every OfferingDraft field (including the stored weight
352
+ # pair) while removing only those volatile evidence timestamps.
353
+ def offerings_equivalent?(previous, current)
354
+ Array(previous).map { |draft| offering_comparison_state(draft) }.tally ==
355
+ Array(current).map { |draft| offering_comparison_state(draft) }.tally
356
+ end
357
+
358
+ def offering_comparison_state(draft)
359
+ state = draft.to_h
360
+ state[:operation_evidence] = comparison_evidence_map(draft.operation_evidence)
361
+ state[:capability_evidence] = comparison_evidence_map(draft.capability_evidence)
362
+ OFFERING_SCALAR_EVIDENCE_FIELDS.each do |field|
363
+ state[field] = comparison_evidence(draft.public_send(field))
75
364
  end
365
+ state
76
366
  end
77
367
 
78
- def offerings_for(adapter, instance_id)
79
- Array(adapter.discover_offerings(live: true))
80
- rescue StandardError => e
81
- handle_exception(e, level: :warn, handled: true,
82
- operation: 'mlx.discovery_refresh.discover_offerings',
83
- instance: instance_id)
84
- []
368
+ def comparison_evidence_map(evidence)
369
+ evidence.transform_values { |entry| comparison_evidence(entry) }
85
370
  end
86
371
 
87
- def lanes_for_instance(entry)
88
- adapter = entry[:adapter]
89
- instance_id = entry[:instance] || entry[:instance_id] || entry[:id]
90
- return [] unless adapter.respond_to?(:discover_offerings)
372
+ def comparison_evidence(evidence)
373
+ evidence.to_h.except(:observed_at)
374
+ end
375
+ end
91
376
 
92
- offerings_for(adapter, instance_id).filter_map do |raw_offering|
93
- offering = offering_to_hash(raw_offering)
94
- next unless offering
377
+ # MLX bindings for the shared writer reconciler. This module owns only
378
+ # actor-local publication synchronization and dormant tracking used by
379
+ # the existing discovery cadence.
380
+ module WeightPublication
381
+ private
95
382
 
96
- build_lanes(offering, instance_id)
97
- end.flatten
383
+ def replace_offerings_if_changed(instance_id:, state:)
384
+ new_offerings = discover_offerings_for_instance(
385
+ instance_cfg: state[:instance_cfg], instance_key: state[:instance_key]
386
+ )
387
+ Legion::Extensions::Llm::Inventory::WeightReconciler.commit_if_changed!(
388
+ settings: Legion::Settings,
389
+ instance_id: instance_id,
390
+ state: state,
391
+ discovered_offerings: new_offerings,
392
+ mutex: state_mutex,
393
+ equivalent: method(:offerings_equivalent?),
394
+ replace: method(:replace_weight_snapshot)
395
+ )
396
+ end
397
+
398
+ def replace_weight_snapshot(instance_id:, state:, offerings:, sequence:)
399
+ publisher.replace_instance_snapshot(
400
+ instance_id: instance_id,
401
+ publisher_token: state.fetch(:publisher_token),
402
+ offerings: offerings,
403
+ sequence: sequence,
404
+ physical_id: state.fetch(:physical_id)
405
+ )
98
406
  end
99
407
 
100
- def offering_to_hash(offering)
101
- return nil if offering.nil?
102
- return offering if offering.is_a?(Hash)
408
+ def commit_readiness(instance_id:, probe_token:, readiness:, state:)
409
+ if readiness.ready?
410
+ return Legion::Extensions::Llm::Inventory::WeightReconciler.activate_tracked!(
411
+ settings: Legion::Settings,
412
+ instance_id: instance_id,
413
+ state_key: instance_id,
414
+ state: state,
415
+ states: @instance_states,
416
+ mutex: state_mutex,
417
+ probe_token: probe_token,
418
+ activate: method(:activate_weight_snapshot),
419
+ activation_sequence: ->(tracked) { tracked.fetch(:sequence) }
420
+ )
421
+ end
422
+
423
+ state_mutex.synchronize do
424
+ return false unless @instance_states[instance_id].equal?(state)
103
425
 
104
- hash = offering.to_h
105
- hash[:type] ||= hash[:usage_type]
106
- hash[:enabled] = offering.respond_to?(:enabled?) ? offering.enabled? : true
107
- hash
426
+ publisher.readiness_failed(
427
+ instance_id: instance_id, physical_id: state[:physical_id],
428
+ probe_token: probe_token, reason: readiness.reason
429
+ )
430
+ end
431
+ true
108
432
  end
109
433
 
110
- def build_lanes(offering, instance_id)
111
- type = offering_type(offering[:type])
112
- tier = offering[:tier] || :local
113
- lane = build_lane(offering, instance_id, type, tier)
114
- lanes = [lane]
115
- lanes << fleet_lane(lane, instance_id, type) if fleet_enabled? && type == :inference
116
- lanes
434
+ def activate_weight_snapshot(instance_id:, state:, offerings:, sequence:, probe_token:)
435
+ publisher.activate_instance_snapshot(
436
+ instance_id: instance_id,
437
+ publisher_token: state.fetch(:publisher_token),
438
+ offerings: offerings,
439
+ sequence: sequence,
440
+ probe_token: probe_token,
441
+ physical_id: state.fetch(:physical_id)
442
+ )
117
443
  end
118
444
 
119
- def build_lane(offering, instance_id, type, tier)
120
- {
121
- id: compose_lane_id(tier: tier, instance_id: instance_id,
122
- type: type, model: offering[:model]),
123
- tier: tier,
445
+ def observe_dormant_weights
446
+ Legion::Extensions::Llm::Inventory::WeightReconciler.observe_dormant!(
447
+ settings: Legion::Settings,
124
448
  provider_family: :mlx,
449
+ states: @instance_states,
450
+ mutex: state_mutex,
451
+ tracker: dormant_weight_tracker,
452
+ dormant_logger: lambda do |key|
453
+ log.info(
454
+ "[llm][mlx] action=dormant_weight weight_key=#{key.inspect} no_lane_published=true"
455
+ )
456
+ end
457
+ )
458
+ end
459
+
460
+ def state_mutex
461
+ @state_mutex ||= Mutex.new
462
+ end
463
+
464
+ def dormant_weight_tracker
465
+ @dormant_weight_tracker ||= Legion::Extensions::Llm::Inventory::DormantWeightTracker.new
466
+ end
467
+ end
468
+
469
+ # Periodic refresh cycle — included by DiscoveryRefresh. Each tick
470
+ # re-scans the configured instances (late configuration appears
471
+ # without a restart; removed instances are reconciled out),
472
+ # re-activates instances still initializing after an initial
473
+ # readiness failure, and refreshes offerings + cadence probes for
474
+ # activated instances.
475
+ module TickCycle
476
+ private
477
+
478
+ def tick_refresh
479
+ instance_states
480
+ configured_ids = {}
481
+ configured_instances.each do |name, instance_cfg|
482
+ reconcile_configured_instance(
483
+ name: name, instance_cfg: instance_cfg, configured_ids: configured_ids
484
+ )
485
+ end
486
+
487
+ remove_unconfigured_instances(configured_ids: configured_ids)
488
+ observe_dormant_weights
489
+ end
490
+
491
+ def instance_states
492
+ state_mutex.synchronize { @instance_states ||= {} }
493
+ end
494
+
495
+ # Identity is the operator's config name — the key the frozen config
496
+ # and router use. Two names at one endpoint remain distinct instances.
497
+ def reconcile_configured_instance(name:, instance_cfg:, configured_ids:)
498
+ instance_id = name.to_s
499
+ configured_ids[instance_id] = true
500
+ state = state_mutex.synchronize { @instance_states[instance_id] }
501
+ if state.nil?
502
+ claim_and_activate_instance(name: name, instance_cfg: instance_cfg)
503
+ else
504
+ refresh_instance(instance_id: instance_id, name: name, state: state)
505
+ end
506
+ rescue StandardError => e
507
+ handle_exception(e, level: :warn, operation: 'mlx.actor.tick_refresh', instance_name: name.to_s)
508
+ end
509
+
510
+ def remove_unconfigured_instances(configured_ids:)
511
+ states = state_mutex.synchronize { @instance_states.to_a }
512
+ states.each do |instance_id, state|
513
+ next if configured_ids.key?(instance_id)
514
+
515
+ remove_instance_state(instance_id: instance_id, state: state)
516
+ rescue StandardError => e
517
+ handle_exception(e, level: :warn, operation: 'mlx.actor.remove_instance_state',
518
+ instance_id: instance_id)
519
+ end
520
+ end
521
+
522
+ def remove_instance_state(instance_id:, state:)
523
+ removed = state_mutex.synchronize do
524
+ next false unless @instance_states[instance_id].equal?(state)
525
+
526
+ publisher.remove_instance(
527
+ instance_id: instance_id, physical_id: state[:physical_id],
528
+ publisher_token: state[:publisher_token]
529
+ )
530
+ @instance_states.delete(instance_id)
531
+ true
532
+ end
533
+ clear_instance_health(config_name: state[:name]) if removed
534
+ removed
535
+ end
536
+
537
+ def refresh_instance(instance_id:, name:, state:)
538
+ status = publisher.snapshot.publication_status(instance_key: state[:instance_key])
539
+ if status.state == :initializing
540
+ reactivate_if_ready(instance_id: instance_id, name: name, state: state)
541
+ return
542
+ end
543
+
544
+ replace_offerings_if_changed(instance_id: instance_id, state: state)
545
+ run_cadence_probe(instance_id: instance_id, state: state)
546
+ write_instance_health(config_name: name, state: state)
547
+ end
548
+
549
+ # Initial-failure recovery: an instance stuck at :initializing
550
+ # (readiness failed at boot, e.g. transient outage) re-activates
551
+ # on the first healthy probe. While :initializing,
552
+ # replace_instance_snapshot and readiness_succeeded are invalid
553
+ # transitions — activate_instance_snapshot is the only legal
554
+ # commit, so the cadence probe path is not usable here.
555
+ def reactivate_if_ready(instance_id:, name:, state:)
556
+ offerings = discover_offerings_for_instance(
557
+ instance_cfg: state[:instance_cfg], instance_key: state[:instance_key]
558
+ )
559
+ Legion::Extensions::Llm::Inventory::WeightReconciler.commit_if_changed!(
560
+ settings: Legion::Settings,
125
561
  instance_id: instance_id,
126
- model: offering[:model],
127
- canonical_model_alias: offering[:canonical_model_alias],
128
- type: type,
129
- capabilities: normalize_capabilities(offering[:capabilities]),
130
- limits: offering[:limits] || {},
131
- enabled: offering.fetch(:enabled, true),
132
- cost: offering[:cost] || {}
562
+ state: state,
563
+ discovered_offerings: offerings,
564
+ mutex: state_mutex,
565
+ equivalent: method(:offerings_equivalent?),
566
+ replace: method(:replace_weight_snapshot)
567
+ )
568
+ probe_token = publisher.readiness_probe_started(
569
+ instance_id: instance_id, publisher_token: state[:publisher_token]
570
+ )
571
+ readiness = check_health(instance_cfg: state[:instance_cfg])
572
+ committed = commit_readiness(
573
+ instance_id: instance_id, probe_token: probe_token,
574
+ readiness: readiness, state: state
575
+ )
576
+ write_instance_health(config_name: name, state: state) if committed
577
+ end
578
+ end
579
+
580
+ # Instance configuration, ID derivation and settings — included by DiscoveryRefresh.
581
+ module InstanceConfig
582
+ private
583
+
584
+ def settings
585
+ Legion::Settings.dig(:extensions, :llm, :mlx) || {}
586
+ end
587
+
588
+ def configured_instances
589
+ Legion::Extensions::Llm::Mlx.configured_instances
590
+ end
591
+
592
+ # The SECONDARY physical id (host:port, or host:port/ak:<fp>
593
+ # when the instance is keyed). It is carried by InstanceKey
594
+ # for dedup and diagnostics only — never identity. Identity
595
+ # is the operator's config name (see tick_refresh).
596
+ def derive_physical_id(instance_cfg:)
597
+ base_url = instance_cfg[:mlx_api_base] || instance_cfg[:endpoint] || 'http://localhost:8000'
598
+ host_port = extract_host_port(url: base_url)
599
+ api_key = instance_cfg[:mlx_api_key] || instance_cfg.dig(:credentials, :api_key)
600
+
601
+ if api_key.is_a?(String) && !api_key.strip.empty?
602
+ fingerprint = ::Digest::SHA256.hexdigest(api_key)[0, 6]
603
+ "#{host_port}/ak:#{fingerprint}"
604
+ else
605
+ host_port
606
+ end
607
+ end
608
+
609
+ def extract_host_port(url:)
610
+ uri = URI.parse(url.to_s)
611
+ host = uri.host || 'localhost'
612
+ port = uri.port
613
+ "#{host}:#{port}"
614
+ rescue URI::InvalidURIError => e
615
+ handle_exception(e, level: :warn, operation: 'mlx.actor.extract_host_port', url: url.to_s)
616
+ raise
617
+ end
618
+
619
+ def build_instance_key(instance_id:, physical_id:)
620
+ Legion::Extensions::Llm::Inventory::Identity::InstanceKey.new(
621
+ provider_family: :mlx, instance_id: instance_id, physical_id: physical_id
622
+ )
623
+ end
624
+
625
+ def build_probe_coordinator(instance_id:, instance_key:)
626
+ Legion::Extensions::Llm::Inventory::ProbeCoordinator.new(
627
+ instance_key: instance_key,
628
+ enqueue: build_probe_enqueue(instance_id: instance_id)
629
+ )
630
+ end
631
+
632
+ def build_instance_state(**attrs)
633
+ attrs.merge(sequence: 0, published: false)
634
+ end
635
+ end
636
+
637
+ # Display-only health/capabilities written into the settings tree
638
+ # after each registry commit. Routing authority stays in the
639
+ # in-memory Registry; this hash exists so the status API
640
+ # (legion-llm /api/llm/providers) renders per-instance health.
641
+ # Keyed by the operator's config name, not the derived instance_id.
642
+ module HealthDisplay
643
+ HEALTH_SOURCE = :provider_probe
644
+
645
+ private
646
+
647
+ def write_instance_health(config_name:, state:)
648
+ instance_settings = settings[:instances]
649
+ return unless instance_settings.is_a?(Hash) && instance_settings[config_name].is_a?(Hash)
650
+
651
+ instance_settings[config_name][:health] = build_health_hash(state: state)
652
+ instance_settings[config_name][:capabilities] = build_display_capabilities(state: state)
653
+ rescue StandardError => e
654
+ handle_exception(e, level: :warn, operation: 'mlx.actor.write_instance_health',
655
+ instance_name: config_name.to_s)
656
+ end
657
+
658
+ def clear_instance_health(config_name:)
659
+ instance_settings = settings[:instances]
660
+ return unless instance_settings.is_a?(Hash) && instance_settings[config_name].is_a?(Hash)
661
+
662
+ instance_settings[config_name].delete(:health)
663
+ instance_settings[config_name].delete(:capabilities)
664
+ rescue StandardError => e
665
+ handle_exception(e, level: :warn, operation: 'mlx.actor.clear_instance_health',
666
+ instance_name: config_name.to_s)
667
+ end
668
+
669
+ def build_health_hash(state:)
670
+ instance_key = state[:instance_key]
671
+ status = publisher.snapshot.publication_status(instance_key: instance_key)
672
+ availability = publisher.snapshot.instance(instance_key: instance_key)&.availability
673
+ {
674
+ circuit_state: health_circuit_state(availability),
675
+ denied: false,
676
+ available: health_available?(availability),
677
+ adjustment: health_adjustment(availability),
678
+ reason: health_display_reason(status: status, availability: availability),
679
+ observed_at: health_observed_at(status: status, availability: availability),
680
+ last_probe_outcome: status.last_probe_outcome,
681
+ source: HEALTH_SOURCE
133
682
  }
134
683
  end
135
684
 
136
- def fleet_lane(lane, instance_id, type)
137
- lane.merge(
138
- tier: :fleet,
139
- id: compose_lane_id(tier: :fleet, instance_id: instance_id,
140
- type: type, model: lane[:model])
685
+ def health_available?(availability)
686
+ !availability.nil? && availability.state == :available
687
+ end
688
+
689
+ def health_circuit_state(availability)
690
+ health_available?(availability) ? :closed : :open
691
+ end
692
+
693
+ def health_adjustment(availability)
694
+ health_available?(availability) ? 0 : -50
695
+ end
696
+
697
+ def health_observed_at(status:, availability:)
698
+ # getutc (not utc): the registry freezes its Time objects, and
699
+ # Time#utc mutates the receiver in place.
700
+ (availability&.observed_at || status.last_probe_completed_at || Time.now).getutc.iso8601
701
+ end
702
+
703
+ def health_display_reason(status:, availability:)
704
+ return availability.reason if availability&.reason
705
+ return status.last_error if status.last_error
706
+
707
+ 'awaiting initial readiness'
708
+ end
709
+
710
+ def build_display_capabilities(state:)
711
+ state[:offerings].each_with_object(Hash.new(false)) do |draft, supported|
712
+ draft.capability_evidence.each do |capability, evidence|
713
+ supported[capability] = true if evidence.supported?
714
+ end
715
+ end.keys.sort
716
+ end
717
+ end
718
+
719
+ # HTTP connection builders — included by DiscoveryRefresh.
720
+ module HttpConnections
721
+ private
722
+
723
+ def normalize_api_base(url)
724
+ (url || 'http://localhost:8000').to_s.sub(%r{/v1/?\z}, '')
725
+ end
726
+
727
+ def build_health_connection(base_url:, instance_cfg:)
728
+ Faraday.new(url: base_url) do |f|
729
+ f.options.timeout = 5
730
+ f.options.open_timeout = 3
731
+ apply_auth_header(faraday: f, instance_cfg: instance_cfg)
732
+ f.adapter Faraday.default_adapter
733
+ end
734
+ end
735
+
736
+ def build_api_connection(base_url:, instance_cfg:)
737
+ Faraday.new(url: base_url) do |f|
738
+ f.options.timeout = 15
739
+ f.options.open_timeout = 5
740
+ f.headers['Accept'] = 'application/json'
741
+ apply_auth_header(faraday: f, instance_cfg: instance_cfg)
742
+ f.adapter Faraday.default_adapter
743
+ end
744
+ end
745
+
746
+ def apply_auth_header(faraday:, instance_cfg:)
747
+ api_key = instance_cfg[:mlx_api_key] || instance_cfg.dig(:credentials, :api_key)
748
+ return unless api_key.is_a?(String) && !api_key.strip.empty?
749
+
750
+ faraday.headers['Authorization'] = "Bearer #{api_key}"
751
+ end
752
+ end
753
+
754
+ # SSOT v3 periodic discovery actor for MLX provider instances.
755
+ # Claims configured instances, discovers models via /v1/models,
756
+ # probes health via /health, and publishes complete OfferingDraft
757
+ # snapshots through the Inventory::Publisher. Supports coalesced
758
+ # reactive probes after dispatch-triggered instance_unavailable
759
+ # transitions.
760
+ class DiscoveryRefresh < Legion::Extensions::Actors::Every
761
+ include Legion::Logging::Helper
762
+ include EvidenceBuilding
763
+ include ValueEvidenceBuilding
764
+ include OfferingAssembly
765
+ include HealthProbing
766
+ include OfferingComparison
767
+ include WeightPublication
768
+ include TickCycle
769
+ include InstanceConfig
770
+ include HealthDisplay
771
+ include HttpConnections
772
+
773
+ # Mirrors the registered lex-llm default
774
+ # (discovery.interval_seconds); used only when the settings tree
775
+ # has no discovery section. time must never return nil — a
776
+ # TimerTask with a nil interval fires exactly once and stops.
777
+ DEFAULT_DISCOVERY_INTERVAL_SECONDS = 300
778
+
779
+ def runner_class = self.class
780
+ def runner_function = 'manual'
781
+ def run_now? = true
782
+ def use_runner? = false
783
+ def check_subtask? = false
784
+ def generate_task? = false
785
+
786
+ def time
787
+ interval = settings.dig(:discovery, :interval_seconds)
788
+ interval.is_a?(Integer) && interval.positive? ? interval : DEFAULT_DISCOVERY_INTERVAL_SECONDS
789
+ end
790
+
791
+ def manual
792
+ tick_refresh
793
+ rescue StandardError => e
794
+ handle_exception(e, level: :warn, operation: 'mlx.actor.discovery_refresh')
795
+ end
796
+
797
+ def shutdown
798
+ remove_all_instances
799
+ rescue StandardError => e
800
+ handle_exception(e, level: :warn, operation: 'mlx.actor.discovery_refresh.shutdown')
801
+ end
802
+
803
+ private
804
+
805
+ def publisher
806
+ @publisher ||= Legion::Extensions::Llm::Inventory::Publisher.new(
807
+ provider_family: :mlx,
808
+ compatibility_adapter: Legion::Extensions::Llm::Inventory::ScopedRefresher::LegacyCoordinatorAdapter.new(
809
+ provider_family: :mlx
810
+ )
141
811
  )
142
812
  end
143
813
 
144
- def offering_type(raw)
145
- %i[embed embedding].include?(raw.to_s.to_sym) ? :embedding : :inference
814
+ def claim_and_activate_instance(name:, instance_cfg:)
815
+ instance_id = name.to_s
816
+ physical_id = derive_physical_id(instance_cfg: instance_cfg)
817
+ instance_key = build_instance_key(instance_id: instance_id, physical_id: physical_id)
818
+ offerings = discover_offerings_for_instance(instance_cfg: instance_cfg, instance_key: instance_key)
819
+ callable = Legion::Extensions::Llm::Mlx::Actor::MlxCallable.new(instance_cfg: instance_cfg, logger: log)
820
+ probe_coordinator = build_probe_coordinator(instance_id: instance_id, instance_key: instance_key)
821
+ publisher_token = publisher.claim_instance(
822
+ instance_id: instance_id, physical_id: physical_id, callable: callable,
823
+ probe_request_handle: probe_coordinator
824
+ )
825
+ run_activation(
826
+ instance_id: instance_id, publisher_token: publisher_token,
827
+ offerings: offerings,
828
+ instance_desc: { name: name, instance_id: instance_id, physical_id: physical_id,
829
+ instance_key: instance_key, instance_cfg: instance_cfg,
830
+ callable: callable, probe_coordinator: probe_coordinator }
831
+ )
832
+ end
833
+
834
+ def run_activation(instance_id:, publisher_token:, offerings:, instance_desc:)
835
+ instance_cfg = instance_desc[:instance_cfg]
836
+ state = build_instance_state(
837
+ **instance_desc, publisher_token: publisher_token, offerings: offerings
838
+ )
839
+ Legion::Extensions::Llm::Inventory::WeightReconciler.track_initializing!(
840
+ states: @instance_states,
841
+ state_key: instance_id,
842
+ state: state,
843
+ mutex: state_mutex
844
+ )
845
+ probe_token = publisher.readiness_probe_started(instance_id: instance_id,
846
+ publisher_token: publisher_token)
847
+ readiness = check_health(instance_cfg: instance_cfg)
848
+ committed = commit_readiness(
849
+ instance_id: instance_id, probe_token: probe_token,
850
+ readiness: readiness, state: state
851
+ )
852
+ write_instance_health(config_name: instance_desc[:name], state: state) if committed
853
+ end
854
+
855
+ def remove_all_instances
856
+ states = state_mutex.synchronize do
857
+ return if @instance_states.nil?
858
+
859
+ @instance_states.to_a
860
+ end
861
+ states.each do |instance_id, state|
862
+ remove_instance_state(instance_id: instance_id, state: state)
863
+ rescue StandardError => e
864
+ handle_exception(e, level: :warn, operation: 'mlx.actor.remove_instance',
865
+ instance_id: instance_id)
866
+ end
867
+ state_mutex.synchronize do
868
+ @instance_states.clear
869
+ dormant_weight_tracker.clear!
870
+ end
871
+ end
872
+ end
873
+
874
+ # Callable wrapper for an MLX provider instance. It is the
875
+ # exact-execution dispatch target: it implements the fleet dispatch
876
+ # operations (chat, stream_chat, embed, count_tokens) by delegating
877
+ # to a per-instance Mlx::Provider built from the instance config,
878
+ # plus the `disconnect` and `normalize_dispatch_error(error:)`
879
+ # contracts required by Inventory::CallableHandle and
880
+ # Routing::ProviderOutcome. Provider and Faraday errors are NOT
881
+ # rescued here so the coordinator's normalize_dispatch_error can
882
+ # classify them.
883
+ class MlxCallable
884
+ # Keys the base Provider exposes as named kwargs for the
885
+ # completion operations. Anything else the fleet passes is folded
886
+ # into the payload `params` hash.
887
+ COMPLETION_NAMED_KEYS = %i[tools temperature schema thinking tool_prefs headers].freeze
888
+ EMBED_NAMED_KEYS = %i[dimensions headers].freeze
889
+
890
+ def initialize(instance_cfg:, logger:)
891
+ @instance_cfg = instance_cfg
892
+ @logger = logger
893
+ @disconnected = false
894
+ @inference_calls = 0
895
+ end
896
+
897
+ def call_count
898
+ @inference_calls
899
+ end
900
+
901
+ def disconnected?
902
+ @disconnected
903
+ end
904
+
905
+ def disconnect
906
+ @disconnected = true
907
+ @provider&.disconnect
908
+ @logger.debug { '[mlx][callable] disconnected' }
146
909
  end
147
910
 
148
- def normalize_capabilities(caps)
149
- return [] unless defined?(Legion::LLM::Inventory::Capabilities)
911
+ # ── Fleet dispatch operations ───────────────────────────────────
150
912
 
151
- Legion::LLM::Inventory::Capabilities.normalize(caps)
913
+ def chat(messages:, model:, **rest)
914
+ record_inference
915
+ named, params = split_fleet_kwargs(rest, COMPLETION_NAMED_KEYS)
916
+ provider.chat(messages: messages, model: model_info(model), params: params, **named)
152
917
  end
153
918
 
154
- def compose_lane_id(tier:, instance_id:, type:, model:)
155
- Legion::Extensions::Llm::Inventory::ScopedRefresher.compose_id(
156
- tier: tier, provider_family: :mlx, instance_id: instance_id,
157
- type: type, model: model
919
+ def stream_chat(messages:, model:, **rest, &)
920
+ record_inference
921
+ named, params = split_fleet_kwargs(rest, COMPLETION_NAMED_KEYS)
922
+ provider.stream_chat(messages: messages, model: model_info(model), params: params, **named, &)
923
+ end
924
+
925
+ def embed(text:, model:, **rest)
926
+ record_inference
927
+ named, params = split_fleet_kwargs(rest, EMBED_NAMED_KEYS)
928
+ provider.embed(text: text, model: model_info(model), params: params, **named)
929
+ end
930
+
931
+ def count_tokens(messages:, model:, **rest)
932
+ record_inference
933
+ _named, params = split_fleet_kwargs(rest, [])
934
+ provider.count_tokens(messages: messages, model: model, params: params)
935
+ end
936
+
937
+ def normalize_dispatch_error(error:)
938
+ reason = error.message.to_s[0, 512]
939
+ kind = classify_error_kind(error: error)
940
+
941
+ Legion::Extensions::Llm::Routing::ProviderOutcome.new(
942
+ kind: kind,
943
+ reason: reason.empty? ? 'unknown dispatch error' : reason
158
944
  )
159
945
  end
160
946
 
161
- def fleet_enabled?
162
- mlx_settings = Legion::Settings.dig(:extensions, :llm, :mlx) || {}
163
- mlx_settings.dig(:fleet, :dispatch, :enabled)
947
+ private
948
+
949
+ def record_inference
950
+ @inference_calls += 1
951
+ end
952
+
953
+ def provider
954
+ @provider ||= Legion::Extensions::Llm::Mlx::Provider.new(@instance_cfg)
955
+ end
956
+
957
+ # The fleet passes the model as a bare string; the base Provider's
958
+ # payload renderer needs a Model::Info (model.id). Wrap strings
959
+ # only — pass through anything already carrying model identity.
960
+ def model_info(model)
961
+ return model if model.respond_to?(:id)
962
+
963
+ Legion::Extensions::Llm::Model::Info.new(
964
+ id: model.to_s, provider: Legion::Extensions::Llm::Mlx::PROVIDER_FAMILY
965
+ )
966
+ end
967
+
968
+ # Split the fleet's **rest into the base Provider's named kwargs
969
+ # and a payload params hash (any passed :params merged with
970
+ # unknown keys).
971
+ def split_fleet_kwargs(rest, named_keys)
972
+ named = rest.slice(*named_keys)
973
+ extra = rest.reject { |key, _| named.key?(key) }
974
+ params = (extra.delete(:params) || {}).to_h.merge(extra)
975
+ [named, params]
976
+ end
977
+
978
+ def classify_error_kind(error:)
979
+ case error
980
+ when Faraday::ConnectionFailed then :connection_failure
981
+ when Faraday::TimeoutError then :timeout
982
+ when Faraday::ClientError then classify_client_error(error: error)
983
+ when Faraday::ServerError then classify_server_error(error: error)
984
+ when Legion::Extensions::Llm::OverloadedError then :overloaded
985
+ else :provider_error
986
+ end
987
+ end
988
+
989
+ def classify_client_error(error:)
990
+ status = error.respond_to?(:response_status) ? error.response_status : nil
991
+ case status
992
+ when 401 then :authentication
993
+ when 403 then :authorization
994
+ when 404 then :model_missing
995
+ when 429 then :rate_limited
996
+ else :invalid_request
997
+ end
998
+ end
999
+
1000
+ def classify_server_error(error:)
1001
+ # NEVER classify raw 503/529/5xx as instance_unavailable by status alone.
1002
+ # Only an explicit flat MLX service/instance-unavailable signal (which MLX
1003
+ # does not produce) would justify instance_unavailable. For MLX, connection
1004
+ # failure (port unreachable) is the signal the instance is down.
1005
+ status = error.respond_to?(:response_status) ? error.response_status : nil
1006
+ case status
1007
+ when 503, 529 then :overloaded
1008
+ else :provider_error
1009
+ end
164
1010
  end
165
1011
  end
166
1012
  end