lex-llm 0.6.16 → 0.7.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (71) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +29 -0
  3. data/lex-llm.gemspec +1 -1
  4. data/lib/legion/extensions/llm/auto_registration.rb +3 -0
  5. data/lib/legion/extensions/llm/canonical/message.rb +5 -1
  6. data/lib/legion/extensions/llm/fleet/protocol.rb +12 -0
  7. data/lib/legion/extensions/llm/fleet/provider_responder.rb +40 -5
  8. data/lib/legion/extensions/llm/fleet/token_validator.rb +15 -0
  9. data/lib/legion/extensions/llm/fleet/worker_execution.rb +166 -2
  10. data/lib/legion/extensions/llm/inventory/callable_handle.rb +144 -0
  11. data/lib/legion/extensions/llm/inventory/errors.rb +34 -0
  12. data/lib/legion/extensions/llm/inventory/evidence.rb +134 -0
  13. data/lib/legion/extensions/llm/inventory/identity.rb +141 -0
  14. data/lib/legion/extensions/llm/inventory/immutable_value.rb +96 -0
  15. data/lib/legion/extensions/llm/inventory/probe_coordinator.rb +117 -0
  16. data/lib/legion/extensions/llm/inventory/probe_token.rb +146 -0
  17. data/lib/legion/extensions/llm/inventory/publisher.rb +116 -0
  18. data/lib/legion/extensions/llm/inventory/records.rb +599 -0
  19. data/lib/legion/extensions/llm/inventory/registry.rb +636 -0
  20. data/lib/legion/extensions/llm/inventory/scoped_refresher.rb +206 -0
  21. data/lib/legion/extensions/llm/inventory/snapshot.rb +91 -0
  22. data/lib/legion/extensions/llm/provider.rb +55 -29
  23. data/lib/legion/extensions/llm/routing/lane_key.rb +4 -0
  24. data/lib/legion/extensions/llm/routing/model_offering.rb +3 -0
  25. data/lib/legion/extensions/llm/routing/offering_registry.rb +2 -0
  26. data/lib/legion/extensions/llm/routing/provider_outcome.rb +54 -0
  27. data/lib/legion/extensions/llm/routing/records.rb +288 -0
  28. data/lib/legion/extensions/llm/settings_cascade.rb +120 -0
  29. data/lib/legion/extensions/llm/taxonomies.rb +90 -0
  30. data/lib/legion/extensions/llm/transport/messages/fleet_error.rb +3 -1
  31. data/lib/legion/extensions/llm/transport/messages/fleet_response.rb +3 -1
  32. data/lib/legion/extensions/llm/version.rb +1 -1
  33. data/lib/legion/extensions/llm.rb +18 -0
  34. data/spec/legion/extensions/llm/canonical/message_spec.rb +24 -0
  35. data/spec/legion/extensions/llm/conformance/fixtures/canonical_params_mapping_request.json +1 -0
  36. data/spec/legion/extensions/llm/conformance/fixtures/canonical_simple_text_request.json +1 -0
  37. data/spec/legion/extensions/llm/conformance/fixtures/canonical_system_prompt_request.json +1 -0
  38. data/spec/legion/extensions/llm/conformance/fixtures/canonical_thinking_request.json +1 -0
  39. data/spec/legion/extensions/llm/conformance/fixtures/canonical_tool_results_continuation_request.json +1 -0
  40. data/spec/legion/extensions/llm/conformance/fixtures/canonical_tools_request.json +1 -0
  41. data/spec/legion/extensions/llm/conformance/fixtures/ssot_identity_vectors.json +84 -0
  42. data/spec/legion/extensions/llm/conformance/ssot_provider_conformance_spec.rb +13 -0
  43. data/spec/legion/extensions/llm/conformance/ssot_provider_examples.rb +196 -0
  44. data/spec/legion/extensions/llm/fleet/exact_offering_spec.rb +179 -0
  45. data/spec/legion/extensions/llm/fleet/provider_responder_spec.rb +26 -0
  46. data/spec/legion/extensions/llm/fleet_messages_spec.rb +14 -0
  47. data/spec/legion/extensions/llm/inventory/boot_spec.rb +80 -0
  48. data/spec/legion/extensions/llm/inventory/callable_handle_spec.rb +127 -0
  49. data/spec/legion/extensions/llm/inventory/evidence_spec.rb +138 -0
  50. data/spec/legion/extensions/llm/inventory/identity_spec.rb +280 -0
  51. data/spec/legion/extensions/llm/inventory/immutable_value_spec.rb +132 -0
  52. data/spec/legion/extensions/llm/inventory/probe_coordinator_spec.rb +107 -0
  53. data/spec/legion/extensions/llm/inventory/probe_token_spec.rb +107 -0
  54. data/spec/legion/extensions/llm/inventory/publisher_spec.rb +144 -0
  55. data/spec/legion/extensions/llm/inventory/records_spec.rb +300 -0
  56. data/spec/legion/extensions/llm/inventory/registry_activation_spec.rb +140 -0
  57. data/spec/legion/extensions/llm/inventory/registry_availability_spec.rb +152 -0
  58. data/spec/legion/extensions/llm/inventory/registry_callable_spec.rb +63 -0
  59. data/spec/legion/extensions/llm/inventory/registry_replacement_spec.rb +69 -0
  60. data/spec/legion/extensions/llm/inventory/scoped_refresher_spec.rb +128 -0
  61. data/spec/legion/extensions/llm/inventory/snapshot_spec.rb +109 -0
  62. data/spec/legion/extensions/llm/provider_contract_spec.rb +20 -0
  63. data/spec/legion/extensions/llm/provider_spec.rb +83 -0
  64. data/spec/legion/extensions/llm/routing/provider_outcome_spec.rb +73 -0
  65. data/spec/legion/extensions/llm/routing/records_spec.rb +173 -0
  66. data/spec/legion/extensions/llm/settings_cascade_spec.rb +208 -0
  67. data/spec/legion/extensions/llm/taxonomies_spec.rb +51 -0
  68. data/spec/support/fake_llm_provider.rb +11 -0
  69. data/spec/support/fake_ssot_harness.rb +137 -0
  70. data/spec/support/ssot_registry_helpers.rb +75 -0
  71. metadata +40 -3
@@ -0,0 +1,84 @@
1
+ {
2
+ "description": "Binding SSOT v3 identity input/byte/digest fixtures. Recompute every field and digest; do not merely check a prefix. No Ruby object hash, hash-iteration order, or delimiter splitting is used anywhere.",
3
+ "algorithm": {
4
+ "frame": "binary bytes: decimal(UTF-8 byte length) + ASCII 0x3a colon + exact UTF-8 bytes",
5
+ "offering_id": "off:v1: + lowercase_hex(SHA256(frame(provider_family) + frame(instance_id) + frame(provider_native_key)))",
6
+ "lane_id": "lane:v1: + lowercase_hex(SHA256(ASCII(\"lane-v1\\u0000\") + frame(provider_family) + frame(instance_id) + frame(operation) + frame(model) + frame(offering_id)))",
7
+ "lane_domain_prefix_hex": "6c616e652d763100",
8
+ "notes": "Tier is never an identity input. provider_native_key, model, and offering_id preserve declared case."
9
+ },
10
+ "vectors": [
11
+ {
12
+ "provider_family": "vllm",
13
+ "instance_id": "h200",
14
+ "provider_native_key": "gemma4",
15
+ "model": "gemma4",
16
+ "operation": "chat",
17
+ "framed_hex": {
18
+ "provider_family": "343a766c6c6d",
19
+ "instance_id": "343a68323030",
20
+ "provider_native_key": "363a67656d6d6134",
21
+ "operation": "343a63686174",
22
+ "model": "363a67656d6d6134",
23
+ "offering_id": "37313a6f66663a76313a63393636373732643766386634323863323461373762653936663934653131316261313035326266303933303938663034313362353632633130353964636431"
24
+ },
25
+ "offering_hash_input_hex": "343a766c6c6d343a68323030363a67656d6d6134",
26
+ "expected_offering_id": "off:v1:c966772d7f8f428c24a77be96f94e111ba1052bf093098f0413b562c1059dcd1",
27
+ "lane_hash_input_hex": "6c616e652d763100343a766c6c6d343a68323030343a63686174363a67656d6d613437313a6f66663a76313a63393636373732643766386634323863323461373762653936663934653131316261313035326266303933303938663034313362353632633130353964636431",
28
+ "expected_lane_id": "lane:v1:430444bb58975ab56329c7c3a6c483a1b9cb49d1c089a56f3c93c6aa27d5f9b3"
29
+ },
30
+ {
31
+ "provider_family": "vllm",
32
+ "instance_id": "helios1",
33
+ "provider_native_key": "gemma4",
34
+ "model": "gemma4",
35
+ "operation": "chat",
36
+ "framed_hex": {
37
+ "provider_family": "343a766c6c6d",
38
+ "instance_id": "373a68656c696f7331",
39
+ "provider_native_key": "363a67656d6d6134",
40
+ "operation": "343a63686174",
41
+ "model": "363a67656d6d6134",
42
+ "offering_id": "37313a6f66663a76313a66336333316366313830313739363436343163386166656463656261316136366135326661626630313463666632623365316438383165333639376339656334"
43
+ },
44
+ "offering_hash_input_hex": "343a766c6c6d373a68656c696f7331363a67656d6d6134",
45
+ "expected_offering_id": "off:v1:f3c31cf18017964641c8afedceba1a66a52fabf014cff2b3e1d881e3697c9ec4",
46
+ "lane_hash_input_hex": "6c616e652d763100343a766c6c6d373a68656c696f7331343a63686174363a67656d6d613437313a6f66663a76313a66336333316366313830313739363436343163386166656463656261316136366135326661626630313463666632623365316438383165333639376339656334",
47
+ "expected_lane_id": "lane:v1:e4a99f7884ea1e9c36b0a0e62da5038832857f0e4b69234133faa81b0e30e880"
48
+ },
49
+ {
50
+ "provider_family": "azure_foundry",
51
+ "instance_id": "prod:eastus",
52
+ "provider_native_key": "deploy:gpt-4o",
53
+ "model": "gpt-4o",
54
+ "operation": "chat",
55
+ "framed_hex": {
56
+ "provider_family": "31333a617a7572655f666f756e647279",
57
+ "instance_id": "31313a70726f643a656173747573",
58
+ "provider_native_key": "31333a6465706c6f793a6770742d346f",
59
+ "operation": "343a63686174",
60
+ "model": "363a6770742d346f",
61
+ "offering_id": "37313a6f66663a76313a37646431346639386164346266656335316666623136396561323064633137616564316562653033613363396664666365383966383638343534633035646137"
62
+ },
63
+ "offering_hash_input_hex": "31333a617a7572655f666f756e64727931313a70726f643a65617374757331333a6465706c6f793a6770742d346f",
64
+ "expected_offering_id": "off:v1:7dd14f98ad4bfec51ffb169ea20dc17aed1ebe03a3c9fdfce89f868454c05da7",
65
+ "lane_hash_input_hex": "6c616e652d76310031333a617a7572655f666f756e64727931313a70726f643a656173747573343a63686174363a6770742d346f37313a6f66663a76313a37646431346639386164346266656335316666623136396561323064633137616564316562653033613363396664666365383966383638343534633035646137",
66
+ "expected_lane_id": "lane:v1:2334bc87506ae6312a4f64b53798d723d07279294ee6a3181e23db298cf910aa"
67
+ }
68
+ ],
69
+ "tier_invariance": {
70
+ "note": "The vllm+h200+gemma4+chat vector reproduces identical offering_id/lane_id under every tier; tier is not framed.",
71
+ "base_vector_index": 0,
72
+ "tiers": [
73
+ "local",
74
+ "frontier"
75
+ ]
76
+ },
77
+ "independence": {
78
+ "note": "vectors[0] and vectors[1] share provider_family+model but differ by instance_id, proving vllm+h200+gemma4 and vllm+helios1+gemma4 are independent identities.",
79
+ "vector_indexes": [
80
+ 0,
81
+ 1
82
+ ]
83
+ }
84
+ }
@@ -0,0 +1,13 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'spec_helper'
4
+ require_relative 'ssot_provider_examples'
5
+ require_relative '../../../../support/fake_ssot_harness'
6
+
7
+ # Self-test: the Phase 1 fake harness must satisfy the shared provider examples,
8
+ # proving the examples are correct and consumable by every provider PR.
9
+ RSpec.describe SpecSupport::FakeSsotHarness do
10
+ let(:ssot_harness) { described_class.new }
11
+
12
+ it_behaves_like 'an SSOT v3 provider adapter'
13
+ end
@@ -0,0 +1,196 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'legion/extensions/llm/inventory/publisher'
4
+ require 'legion/extensions/llm/inventory/probe_coordinator'
5
+ require 'legion/extensions/llm/fleet/worker_execution'
6
+ require 'legion/extensions/llm/fleet/protocol'
7
+
8
+ # Shared conformance examples required unchanged by every SSOT v3 provider PR.
9
+ # The including provider spec supplies one `let(:ssot_harness)` implementing the
10
+ # harness interface documented in phase-1-lex-llm-additive.md Task 12. These
11
+ # examples drive only the public Publisher/Registry API and never call a provider
12
+ # actor implementation directly.
13
+ RSpec.shared_examples 'an SSOT v3 provider adapter' do
14
+ harness_methods = %i[
15
+ provider_family instance_configs instance_id build_callable build_offering_drafts safe_readiness
16
+ inference_call_count normalize_dispatch_error instance_unavailable_error overloaded_error model_not_ready_error
17
+ ]
18
+
19
+ let(:registry) { Legion::Extensions::Llm::Inventory::Registry }
20
+ let(:configs) { ssot_harness.instance_configs }
21
+
22
+ before { registry.reset! }
23
+
24
+ def build_key(config)
25
+ Legion::Extensions::Llm::Inventory::Identity::InstanceKey.new(
26
+ provider_family: ssot_harness.provider_family, instance_id: ssot_harness.instance_id(instance_config: config)
27
+ )
28
+ end
29
+
30
+ def publisher_for
31
+ Legion::Extensions::Llm::Inventory::Publisher.new(provider_family: ssot_harness.provider_family)
32
+ end
33
+
34
+ def coordinator_for(key)
35
+ Legion::Extensions::Llm::Inventory::ProbeCoordinator.new(instance_key: key, enqueue: ->(**) { true })
36
+ end
37
+
38
+ # Claims and (when readiness succeeds) activates one instance through the public
39
+ # Publisher API. Returns the context needed for later assertions.
40
+ def bring_up(config, tier: :local)
41
+ publisher = publisher_for
42
+ key = build_key(config)
43
+ callable = ssot_harness.build_callable(instance_config: config)
44
+ token = publisher.claim_instance(instance_id: key.instance_id, callable: callable, probe_request_handle: coordinator_for(key))
45
+ probe = publisher.readiness_probe_started(instance_id: key.instance_id, publisher_token: token)
46
+ readiness = ssot_harness.safe_readiness(instance_config: config, callable: callable)
47
+ drafts = ssot_harness.build_offering_drafts(instance_config: config, callable: callable, tier: tier)
48
+ if readiness.ready?
49
+ publisher.activate_instance_snapshot(instance_id: key.instance_id, publisher_token: token, offerings: drafts, sequence: 0, probe_token: probe)
50
+ else
51
+ publisher.readiness_failed(instance_id: key.instance_id, probe_token: probe, reason: readiness.reason)
52
+ end
53
+ { publisher: publisher, key: key, callable: callable, token: token, drafts: drafts }
54
+ end
55
+
56
+ it 'implements the full harness interface' do
57
+ harness_methods.each do |method_name|
58
+ expect(ssot_harness).to respond_to(method_name), "ssot_harness must implement ##{method_name}"
59
+ end
60
+ end
61
+
62
+ it 'supplies two independent instances with distinct ids and distinct callables' do
63
+ expect(configs.size).to eq(2)
64
+ ids = configs.map { |config| ssot_harness.instance_id(instance_config: config) }
65
+ expect(ids.uniq.size).to eq(2)
66
+ callables = configs.map { |config| ssot_harness.build_callable(instance_config: config) }
67
+ expect(callables.first).not_to equal(callables.last)
68
+ expect(ssot_harness.safe_readiness(instance_config: configs.first, callable: callables.first))
69
+ .to be_a(Legion::Extensions::Llm::Inventory::ReadinessResult)
70
+ end
71
+
72
+ it 'publishes distinct identities and independently available lanes for the same model on two instances' do
73
+ a = bring_up(configs[0])
74
+ b = bring_up(configs[1])
75
+ snapshot = registry.snapshot
76
+ expect(snapshot.instance(instance_key: a[:key]).availability.state).to eq(:available)
77
+ expect(snapshot.instance(instance_key: b[:key]).availability.state).to eq(:available)
78
+ lane_a = snapshot.lanes_for(instance_key: a[:key]).first
79
+ lane_b = snapshot.lanes_for(instance_key: b[:key]).first
80
+ expect(lane_a.lane_id).not_to eq(lane_b.lane_id)
81
+ expect(lane_a.offering_id).not_to eq(lane_b.offering_id)
82
+ end
83
+
84
+ it 'exposes no selector-visible model or callable before startup readiness' do
85
+ publisher = publisher_for
86
+ key = build_key(configs[0])
87
+ publisher.claim_instance(instance_id: key.instance_id, callable: ssot_harness.build_callable(instance_config: configs[0]), probe_request_handle: coordinator_for(key))
88
+ expect(registry.snapshot.instance(instance_key: key)).to be_nil
89
+ expect(registry.snapshot.publication_status(instance_key: key).state).to eq(:initializing)
90
+ end
91
+
92
+ it 'leaves the instance initializing after an initial readiness failure' do
93
+ publisher = publisher_for
94
+ key = build_key(configs[0])
95
+ token = publisher.claim_instance(instance_id: key.instance_id, callable: ssot_harness.build_callable(instance_config: configs[0]), probe_request_handle: coordinator_for(key))
96
+ probe = publisher.readiness_probe_started(instance_id: key.instance_id, publisher_token: token)
97
+ publisher.readiness_failed(instance_id: key.instance_id, probe_token: probe, reason: 'probe failed')
98
+ expect(registry.snapshot.instance(instance_key: key)).to be_nil
99
+ expect(registry.snapshot.publication_status(instance_key: key).state).to eq(:initializing)
100
+ end
101
+
102
+ it 'performs no provider inference during readiness' do
103
+ context = bring_up(configs[0])
104
+ expect(ssot_harness.inference_call_count(callable: context[:callable])).to eq(0)
105
+ end
106
+
107
+ it 'supports complete refresh, complete empty, and failed-refresh retention' do
108
+ context = bring_up(configs[0])
109
+ context[:publisher].replace_instance_snapshot(instance_id: context[:key].instance_id, publisher_token: context[:token], offerings: context[:drafts], sequence: 1)
110
+ expect(registry.snapshot.offerings_for(instance_key: context[:key]).size).to eq(1)
111
+ context[:publisher].replace_instance_snapshot(instance_id: context[:key].instance_id, publisher_token: context[:token], offerings: [], sequence: 2)
112
+ expect(registry.snapshot.offerings_for(instance_key: context[:key])).to be_empty
113
+ expect do
114
+ context[:publisher].replace_instance_snapshot(instance_id: context[:key].instance_id, publisher_token: context[:token], offerings: [], sequence: 2)
115
+ end.to raise_error(Legion::Extensions::Llm::Inventory::Errors::StaleSequenceError)
116
+ end
117
+
118
+ it 'preserves offering and lane identity across a tier-only republication' do
119
+ context = bring_up(configs[0], tier: :local)
120
+ before_offering = registry.snapshot.offerings_for(instance_key: context[:key]).first.offering_id
121
+ before_lane = registry.snapshot.lanes_for(instance_key: context[:key]).first.lane_id
122
+ frontier_drafts = ssot_harness.build_offering_drafts(instance_config: configs[0], callable: context[:callable], tier: :frontier)
123
+ context[:publisher].replace_instance_snapshot(instance_id: context[:key].instance_id, publisher_token: context[:token], offerings: frontier_drafts, sequence: 1)
124
+ expect(registry.snapshot.offerings_for(instance_key: context[:key]).first.offering_id).to eq(before_offering)
125
+ expect(registry.snapshot.lanes_for(instance_key: context[:key]).first.lane_id).to eq(before_lane)
126
+ end
127
+
128
+ it 'refuses recovery from a stale probe started before the failure' do
129
+ context = bring_up(configs[0])
130
+ stale = context[:publisher].readiness_probe_started(instance_id: context[:key].instance_id, publisher_token: context[:token])
131
+ fresh = context[:publisher].readiness_probe_started(instance_id: context[:key].instance_id, publisher_token: context[:token])
132
+ context[:publisher].readiness_failed(instance_id: context[:key].instance_id, probe_token: fresh, reason: 'down')
133
+ result = context[:publisher].readiness_succeeded(instance_id: context[:key].instance_id, probe_token: stale)
134
+ expect(result.applied).to be(false)
135
+ expect(result.reason).to eq(:stale_probe)
136
+ end
137
+
138
+ it 'marks only one exact instance unavailable on a normalized instance_unavailable outcome' do
139
+ a = bring_up(configs[0])
140
+ b = bring_up(configs[1])
141
+ outcome = ssot_harness.normalize_dispatch_error(error: ssot_harness.instance_unavailable_error)
142
+ expect(outcome.kind).to eq(:instance_unavailable)
143
+ registry.dispatch_instance_unavailable(instance_key: a[:key], publisher_token_id: a[:token].publisher_token_id, reason: outcome.reason)
144
+ expect(registry.snapshot.instance(instance_key: a[:key]).availability.state).to eq(:unavailable)
145
+ expect(registry.snapshot.instance(instance_key: b[:key]).availability.state).to eq(:available)
146
+ end
147
+
148
+ it 'keeps overload and model-not-ready request-local regardless of 503/529 transport status' do
149
+ expect(ssot_harness.normalize_dispatch_error(error: ssot_harness.overloaded_error).kind).to eq(:overloaded)
150
+ expect(ssot_harness.normalize_dispatch_error(error: ssot_harness.model_not_ready_error).kind).to eq(:model_not_ready)
151
+ expect(ssot_harness.normalize_dispatch_error(error: ssot_harness.overloaded_error).kind).not_to eq(:instance_unavailable)
152
+ end
153
+
154
+ it 'executes an exact fleet request against only the captured callable' do
155
+ allow(Legion::Extensions::Llm::Fleet::WorkerExecution).to receive_messages(validate_identity!: true, validate_idempotency!: nil)
156
+ context = bring_up(configs[0])
157
+ offering = registry.snapshot.offerings_for(instance_key: context[:key]).first
158
+ envelope = {
159
+ execution_contract: Legion::Extensions::Llm::Fleet::Protocol::EXACT_EXECUTION_CONTRACT,
160
+ offering_id: offering.offering_id, provider: ssot_harness.provider_family.to_s,
161
+ provider_instance: context[:key].instance_id, model: offering.model, operation: 'chat', params: { messages: [] }
162
+ }
163
+ Legion::Extensions::Llm::Fleet::WorkerExecution.call(envelope: envelope, registry: registry)
164
+ expect(ssot_harness.inference_call_count(callable: context[:callable])).to eq(1)
165
+ end
166
+
167
+ it 'rejects provider/model defaults and a nil instance' do
168
+ identity = Legion::Extensions::Llm::Inventory::Identity
169
+ expect { identity::InstanceKey.new(provider_family: ssot_harness.provider_family, instance_id: 'default') }
170
+ .to raise_error(Legion::Extensions::Llm::Inventory::Errors::ValidationError)
171
+ expect { identity::InstanceKey.new(provider_family: ssot_harness.provider_family, instance_id: nil) }
172
+ .to raise_error(Legion::Extensions::Llm::Inventory::Errors::ValidationError)
173
+ end
174
+
175
+ it 'exposes snapshot each_instance and each_publication_status as ordered blockless enumerators' do
176
+ bring_up(configs[0])
177
+ bring_up(configs[1])
178
+ snapshot = registry.snapshot
179
+ expect(snapshot.each_instance).to be_a(Enumerator)
180
+ expect(snapshot.each_publication_status).to be_a(Enumerator)
181
+ ordered_ids = snapshot.each_instance.map { |record| [record.instance_key.provider_family.to_s, record.instance_key.instance_id] }
182
+ expect(ordered_ids).to eq(ordered_ids.sort)
183
+ expect(snapshot.each_publication_status.map(&:state)).to all(eq(:complete))
184
+ end
185
+
186
+ it 'normalizes a dispatch error into a ProviderOutcome on the captured callable itself' do
187
+ callable = ssot_harness.build_callable(instance_config: configs[0])
188
+ expect(callable).to respond_to(:normalize_dispatch_error)
189
+ unavailable = callable.normalize_dispatch_error(error: Legion::Extensions::Llm::ServiceUnavailableError.new('503'))
190
+ expect(unavailable).to be_a(Legion::Extensions::Llm::Routing::ProviderOutcome)
191
+ expect(unavailable.kind).to eq(:provider_error)
192
+ expect(unavailable.kind).not_to eq(:instance_unavailable)
193
+ overloaded = callable.normalize_dispatch_error(error: Legion::Extensions::Llm::OverloadedError.new('529'))
194
+ expect(overloaded.kind).to eq(:overloaded)
195
+ end
196
+ end
@@ -0,0 +1,179 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'spec_helper'
4
+ require 'legion/extensions/llm/fleet/worker_execution'
5
+ require 'legion/extensions/llm/fleet/token_validator'
6
+ require 'legion/extensions/llm/fleet/protocol'
7
+ require_relative '../../../../support/ssot_registry_helpers'
8
+
9
+ module Legion
10
+ module Extensions
11
+ module Llm
12
+ module Fleet
13
+ # Test-only anchor aligning this spec's path with its describe target.
14
+ module ExactOffering; end
15
+ end
16
+ end
17
+ end
18
+ end
19
+
20
+ RSpec.describe Legion::Extensions::Llm::Fleet::ExactOffering do
21
+ include SsotRegistryHelpers
22
+
23
+ inventory = Legion::Extensions::Llm::Inventory
24
+ errors = inventory::Errors
25
+ worker = Legion::Extensions::Llm::Fleet::WorkerExecution
26
+ protocol = Legion::Extensions::Llm::Fleet::Protocol
27
+ token_validator = Legion::Extensions::Llm::Fleet::TokenValidator
28
+
29
+ let(:recording_callable) do
30
+ Class.new do
31
+ attr_reader :calls
32
+
33
+ def initialize
34
+ @calls = []
35
+ end
36
+
37
+ def disconnect; end
38
+
39
+ def chat(messages:, model:, **rest)
40
+ @calls << { op: :chat, messages: messages, model: model, rest: rest }
41
+ { content: 'exact-ok' }
42
+ end
43
+
44
+ def embed(text:, model:, **rest)
45
+ @calls << { op: :embed, text: text, model: model, rest: rest }
46
+ { content: 'embed-ok' }
47
+ end
48
+ end.new
49
+ end
50
+
51
+ let(:key) { instance_key(family: 'vllm', instance: 'h200') }
52
+
53
+ before do
54
+ inventory::Registry.reset!
55
+ claim_and_activate(key: key, callable: recording_callable, coordinator: probe_coordinator(key))
56
+ allow(worker).to receive_messages(validate_identity!: true, validate_idempotency!: nil)
57
+ end
58
+
59
+ def offering_id
60
+ Legion::Extensions::Llm::Inventory::Identity.offering_id(instance_key: key, provider_native_key: 'gemma4')
61
+ end
62
+
63
+ def exact_envelope(**overrides)
64
+ {
65
+ execution_contract: Legion::Extensions::Llm::Fleet::Protocol::EXACT_EXECUTION_CONTRACT, offering_id: offering_id,
66
+ provider: 'vllm', provider_instance: 'h200', model: 'gemma4', operation: 'chat',
67
+ params: { messages: [] }
68
+ }.merge(overrides)
69
+ end
70
+
71
+ def handle_reference_count
72
+ Legion::Extensions::Llm::Inventory::Registry.snapshot.instance(instance_key: key).callable_handle.reference_count
73
+ end
74
+
75
+ describe 'exact dispatch' do
76
+ it 'invokes only the captured callable and releases the lease' do
77
+ worker.call(envelope: exact_envelope, registry: inventory::Registry)
78
+ expect(recording_callable.calls.map { |c| c[:op] }).to eq([:chat])
79
+ expect(handle_reference_count).to eq(0)
80
+ end
81
+
82
+ it 'requires exactly one of registry or provider' do
83
+ expect { worker.call(envelope: exact_envelope, registry: inventory::Registry, provider: Object.new) }
84
+ .to raise_error(worker::PolicyError)
85
+ expect { worker.call(envelope: exact_envelope) }.to raise_error(worker::PolicyError)
86
+ end
87
+
88
+ it 'releases the lease even when the callable raises' do
89
+ allow(recording_callable).to receive(:chat).and_raise(StandardError, 'boom')
90
+ expect { worker.call(envelope: exact_envelope, registry: inventory::Registry) }.to raise_error(StandardError)
91
+ expect(handle_reference_count).to eq(0)
92
+ end
93
+
94
+ it 'rejects a mismatched offering_id, model, operation, and absent instance before invocation' do
95
+ expect { worker.call(envelope: exact_envelope(offering_id: "off:v1:#{'0' * 64}"), registry: inventory::Registry) }
96
+ .to raise_error(errors::ExactOfferingMismatchError)
97
+ expect { worker.call(envelope: exact_envelope(model: 'other'), registry: inventory::Registry) }
98
+ .to raise_error(errors::ExactOfferingMismatchError)
99
+ expect { worker.call(envelope: exact_envelope(operation: 'embed'), registry: inventory::Registry) }
100
+ .to raise_error(errors::ExactOfferingMismatchError)
101
+ expect { worker.call(envelope: exact_envelope(provider_instance: 'absent'), registry: inventory::Registry) }
102
+ .to raise_error(errors::ExactOfferingMismatchError)
103
+ expect(recording_callable.calls).to be_empty
104
+ end
105
+
106
+ it 'rejects duplicate String/Symbol param spellings' do
107
+ expect { worker.call(envelope: exact_envelope(params: { 'messages' => [], :messages => [] }), registry: inventory::Registry) }
108
+ .to raise_error(errors::ExactOfferingMismatchError)
109
+ end
110
+
111
+ it 'rejects a params-supplied model (envelope model is the only model)' do
112
+ expect { worker.call(envelope: exact_envelope(params: { messages: [], model: 'x' }), registry: inventory::Registry) }
113
+ .to raise_error(errors::ExactOfferingMismatchError)
114
+ end
115
+
116
+ it 'rejects a missing required operation param' do
117
+ expect { worker.call(envelope: exact_envelope(params: {}), registry: inventory::Registry) }
118
+ .to raise_error(errors::ExactOfferingMismatchError)
119
+ end
120
+
121
+ it 'passes an explicit empty messages array and forwards unknown kwargs' do
122
+ worker.call(envelope: exact_envelope(params: { messages: [], temperature: 0.2 }), registry: inventory::Registry)
123
+ call = recording_callable.calls.first
124
+ expect(call[:messages]).to eq([])
125
+ expect(call[:rest]).to eq(temperature: 0.2)
126
+ end
127
+ end
128
+
129
+ describe 'registry-backed legacy v2 resolution' do
130
+ def activate_two_offerings
131
+ registry = Legion::Extensions::Llm::Inventory::Registry
132
+ registry.reset!
133
+ token = registry.claim_instance(instance_key: key, callable: recording_callable, probe_request_handle: probe_coordinator(key))
134
+ probe = registry.readiness_probe_started(instance_key: key, publisher_token: token)
135
+ two = drafts(native: 'a', model: 'gemma4') + drafts(native: 'b', model: 'gemma4')
136
+ registry.activate_instance_snapshot(publisher_token: token, instance_key: key, offerings: two, sequence: 0, probe_token: probe)
137
+ end
138
+
139
+ it 'refuses zero matches' do
140
+ envelope = { provider: 'vllm', provider_instance: 'h200', model: 'nope', operation: 'chat', params: { messages: [] } }
141
+ expect { worker.call(envelope: envelope, registry: inventory::Registry) }.to raise_error(errors::ExactOfferingMismatchError)
142
+ end
143
+
144
+ it 'refuses multiple matches' do
145
+ activate_two_offerings
146
+ envelope = { provider: 'vllm', provider_instance: 'h200', model: 'gemma4', operation: 'chat', params: { messages: [] } }
147
+ expect { worker.call(envelope: envelope, registry: inventory::Registry) }.to raise_error(errors::AmbiguousLegacyOfferingError)
148
+ end
149
+ end
150
+
151
+ describe 'legacy provider path still operates' do
152
+ it 'dispatches through the provider object without a registry' do
153
+ provider = recording_callable
154
+ envelope = { operation: 'chat', model: 'gemma4', params: { messages: [] } }
155
+ worker.call(envelope: envelope, provider: provider)
156
+ expect(recording_callable.calls.map { |c| c[:op] }).to eq([:chat])
157
+ end
158
+ end
159
+
160
+ describe 'TokenValidator exact signed claims' do
161
+ it 'requires both exact scalar claims to be signed and to match when the marker is present' do
162
+ envelope = { execution_contract: protocol::EXACT_EXECUTION_CONTRACT, offering_id: offering_id }
163
+ good = { execution_contract: protocol::EXACT_EXECUTION_CONTRACT, offering_id: offering_id }
164
+ expect { token_validator.validate_exact_execution_claims!(good, envelope) }.not_to raise_error
165
+
166
+ missing = { execution_contract: protocol::EXACT_EXECUTION_CONTRACT }
167
+ expect { token_validator.validate_exact_execution_claims!(missing, envelope) }
168
+ .to raise_error(Legion::Extensions::Llm::Fleet::TokenError)
169
+
170
+ mismatched = { execution_contract: protocol::EXACT_EXECUTION_CONTRACT, offering_id: "off:v1:#{'0' * 64}" }
171
+ expect { token_validator.validate_exact_execution_claims!(mismatched, envelope) }
172
+ .to raise_error(Legion::Extensions::Llm::Fleet::TokenError)
173
+ end
174
+
175
+ it 'is a no-op for a legacy v2 envelope without the marker' do
176
+ expect { token_validator.validate_exact_execution_claims!({}, { operation: 'chat' }) }.not_to raise_error
177
+ end
178
+ end
179
+ end
@@ -117,4 +117,30 @@ RSpec.describe Legion::Extensions::Llm::Fleet::ProviderResponder do
117
117
  described_class.transport_message_class(:FleetResponse)
118
118
  end.to raise_error(described_class::ConfigurationError, /fleet responder transport unavailable/)
119
119
  end
120
+
121
+ describe 'execution_contract marker validation' do
122
+ it 'accepts a legacy v2 envelope with no marker' do
123
+ envelope = described_class.parse_payload(payload)
124
+ expect { described_class.check_envelope!(envelope, provider_family: :ollama) }.not_to raise_error
125
+ end
126
+
127
+ it 'rejects an unknown nonempty marker' do
128
+ envelope = described_class.parse_payload(payload.merge(execution_contract: 'made_up'))
129
+ expect { described_class.check_envelope!(envelope, provider_family: :ollama) }
130
+ .to raise_error(ArgumentError, /unknown execution_contract/)
131
+ end
132
+
133
+ it 'requires offering_id for the exact marker' do
134
+ envelope = described_class.parse_payload(payload.merge(execution_contract: 'exact_offering_v1'))
135
+ expect { described_class.check_envelope!(envelope, provider_family: :ollama) }
136
+ .to raise_error(ArgumentError, /offering_id is required/)
137
+ end
138
+
139
+ it 'accepts a complete exact envelope' do
140
+ exact = payload.merge(execution_contract: 'exact_offering_v1', offering_id: 'off:v1:abc')
141
+ envelope = described_class.parse_payload(exact)
142
+ expect { described_class.check_envelope!(envelope, provider_family: :ollama) }.not_to raise_error
143
+ expect(described_class.exact?(envelope)).to be(true)
144
+ end
145
+ end
120
146
  end
@@ -189,6 +189,20 @@ RSpec.describe 'LLM fleet message envelopes' do
189
189
  expect(message.message).not_to include(:schema_version)
190
190
  end
191
191
 
192
+ it 'includes execution_contract and offering_id only for an exact request' do
193
+ exact = described_class.new(
194
+ request_id: 'req-1', correlation_id: 'corr-1', reply_to: 'llm.fleet.reply.node', content: 'hi',
195
+ execution_contract: 'exact_offering_v1', offering_id: 'off:v1:abc'
196
+ )
197
+ expect(exact.message).to include(execution_contract: 'exact_offering_v1', offering_id: 'off:v1:abc')
198
+
199
+ legacy = described_class.new(
200
+ request_id: 'req-1', correlation_id: 'corr-1', reply_to: 'llm.fleet.reply.node', content: 'hi'
201
+ )
202
+ expect(legacy.message).not_to have_key(:execution_contract)
203
+ expect(legacy.message).not_to have_key(:offering_id)
204
+ end
205
+
192
206
  it 'rejects response protocol versions other than v2' do
193
207
  expect do
194
208
  described_class.new(
@@ -0,0 +1,80 @@
1
+ # frozen_string_literal: true
2
+
3
+ require 'spec_helper'
4
+
5
+ module Legion
6
+ module Extensions
7
+ module Llm
8
+ module Inventory
9
+ # Test-only namespace anchor so this boot spec's file path aligns with its
10
+ # describe target under RSpec/SpecFilePathFormat. It adds no runtime
11
+ # behavior and exists only in the test load path.
12
+ module Boot; end
13
+ end
14
+ end
15
+ end
16
+ end
17
+
18
+ # Proves the SSOT v3 contract is wired by `require "legion/extensions/llm"`
19
+ # alone, with no legion-llm, LegionIO, transport, database, or timer dependency.
20
+ RSpec.describe Legion::Extensions::Llm::Inventory::Boot do
21
+ inventory = Legion::Extensions::Llm::Inventory
22
+
23
+ def build_callable
24
+ Class.new do
25
+ def disconnect; end
26
+ end.new
27
+ end
28
+
29
+ def probe_handle(key)
30
+ Legion::Extensions::Llm::Inventory::ProbeCoordinator.new(instance_key: key, enqueue: ->(**) { true })
31
+ end
32
+
33
+ def full_lifecycle(key)
34
+ registry = Legion::Extensions::Llm::Inventory::Registry
35
+ token = registry.claim_instance(instance_key: key, callable: build_callable, probe_request_handle: probe_handle(key))
36
+ probe = registry.readiness_probe_started(instance_key: key, publisher_token: token)
37
+ registry.activate_instance_snapshot(publisher_token: token, instance_key: key, offerings: [], sequence: 0, probe_token: probe)
38
+ end
39
+
40
+ before { inventory::Registry.reset! }
41
+
42
+ it 'loads without Legion::LLM or LegionIO present' do
43
+ expect(defined?(Legion::LLM)).to be_nil
44
+ expect(defined?(LegionIO)).to be_nil
45
+ end
46
+
47
+ it 'exposes every require-order constant' do
48
+ %i[
49
+ Errors ImmutableValue Identity Evidence CallableHandle DispatchLease ProbeToken ProbeRequest
50
+ ProbeCoordinator PublisherToken OfferingDraft OfferingRecord LaneRecord AvailabilityFact
51
+ ReadinessResult InstanceRecord PublicationStatus MutationResult Snapshot Registry Publisher
52
+ ].each { |const| expect(inventory.const_defined?(const)).to be(true), "missing Inventory::#{const}" }
53
+ %i[AttemptTargetKey QuotaDomainKey Exclusion Selection Rejection BodyModelHintDecision ProviderOutcome].each do |const|
54
+ expect(Legion::Extensions::Llm::Routing.const_defined?(const)).to be(true), "missing Routing::#{const}"
55
+ end
56
+ end
57
+
58
+ it 'runs claim -> probe -> activate -> snapshot after only requiring the extension' do
59
+ key = inventory::Identity::InstanceKey.new(provider_family: 'vllm', instance_id: 'h200')
60
+ result = full_lifecycle(key)
61
+ expect(result.applied).to be(true)
62
+ expect(inventory::Registry.snapshot.instance(instance_key: key).availability.state).to eq(:available)
63
+ end
64
+
65
+ it 'creates no new thread during boot or a mutation' do
66
+ baseline = Thread.list.size
67
+ key = inventory::Identity::InstanceKey.new(provider_family: 'vllm', instance_id: 'thread-check')
68
+ full_lifecycle(key)
69
+ expect(Thread.list.size).to eq(baseline)
70
+ end
71
+
72
+ it 'retains legacy Types aliases and adds no new SSOT alias' do
73
+ expect(Legion::Extensions::Llm::Types.const_defined?(:ModelOffering, false)).to be(true)
74
+ expect(Legion::Extensions::Llm::Types.const_defined?(:InstanceKey, false)).to be(false)
75
+ end
76
+
77
+ it 'preserves the optional transport message registration (autoload or loaded)' do
78
+ expect(Legion::Extensions::Llm::Transport::Messages.const_defined?(:FleetResponse, false)).to be(true)
79
+ end
80
+ end