lex-llm 0.7.6 → 0.8.4
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +139 -0
- data/RULES.md +97 -0
- data/lib/legion/extensions/llm/auto_registration.rb +4 -13
- data/lib/legion/extensions/llm/canonical/chunk.rb +107 -94
- data/lib/legion/extensions/llm/canonical/content_block.rb +65 -75
- data/lib/legion/extensions/llm/canonical/message.rb +53 -69
- data/lib/legion/extensions/llm/canonical/params.rb +58 -34
- data/lib/legion/extensions/llm/canonical/request.rb +68 -56
- data/lib/legion/extensions/llm/canonical/response.rb +62 -78
- data/lib/legion/extensions/llm/canonical/strict.rb +105 -0
- data/lib/legion/extensions/llm/canonical/thinking.rb +36 -84
- data/lib/legion/extensions/llm/canonical/thinking_config.rb +149 -0
- data/lib/legion/extensions/llm/canonical/tool_call.rb +43 -53
- data/lib/legion/extensions/llm/canonical/tool_definition.rb +56 -46
- data/lib/legion/extensions/llm/canonical/tool_schema.rb +15 -22
- data/lib/legion/extensions/llm/canonical/usage.rb +47 -40
- data/lib/legion/extensions/llm/canonical.rb +7 -5
- data/lib/legion/extensions/llm/configuration.rb +40 -10
- data/lib/legion/extensions/llm/connection.rb +8 -30
- data/lib/legion/extensions/llm/credential_sources.rb +32 -49
- data/lib/legion/extensions/llm/discovery/actor.rb +92 -0
- data/lib/legion/extensions/llm/discovery/pipeline.rb +604 -0
- data/lib/legion/extensions/llm/error.rb +0 -14
- data/lib/legion/extensions/llm/fleet/contract_error.rb +15 -0
- data/lib/legion/extensions/llm/fleet/envelope_validation.rb +7 -6
- data/lib/legion/extensions/llm/fleet/fleet_envelope.rb +66 -0
- data/lib/legion/extensions/llm/fleet/protocol.rb +18 -5
- data/lib/legion/extensions/llm/fleet/provider_responder.rb +58 -156
- data/lib/legion/extensions/llm/fleet/token_validator.rb +15 -21
- data/lib/legion/extensions/llm/fleet/worker_execution.rb +107 -154
- data/lib/legion/extensions/llm/inventory/errors.rb +0 -1
- data/lib/legion/extensions/llm/inventory/evidence.rb +1 -1
- data/lib/legion/extensions/llm/inventory/identity.rb +55 -39
- data/lib/legion/extensions/llm/inventory/probe_token.rb +9 -12
- data/lib/legion/extensions/llm/inventory/publisher.rb +15 -53
- data/lib/legion/extensions/llm/inventory/records.rb +36 -100
- data/lib/legion/extensions/llm/inventory/registry.rb +54 -48
- data/lib/legion/extensions/llm/inventory/snapshot.rb +7 -21
- data/lib/legion/extensions/llm/inventory/weight_reconciler.rb +13 -6
- data/lib/legion/extensions/llm/inventory/weight_schema.rb +47 -8
- data/lib/legion/extensions/llm/provider/open_ai_compatible.rb +178 -182
- data/lib/legion/extensions/llm/provider.rb +192 -315
- data/lib/legion/extensions/llm/provider_contract.rb +25 -8
- data/lib/legion/extensions/llm/provider_settings.rb +5 -26
- data/lib/legion/extensions/llm/responses/thinking_extractor.rb +8 -1
- data/lib/legion/extensions/llm/responses/tool_arguments.rb +48 -0
- data/lib/legion/extensions/llm/routing/provider_outcome.rb +25 -0
- data/lib/legion/extensions/llm/routing/records.rb +34 -17
- data/lib/legion/extensions/llm/stream_accumulator.rb +186 -270
- data/lib/legion/extensions/llm/streaming.rb +50 -35
- data/lib/legion/extensions/llm/taxonomies.rb +14 -26
- data/lib/legion/extensions/llm/transport/fleet_lane.rb +8 -10
- data/lib/legion/extensions/llm/transport/messages/fleet_error.rb +3 -2
- data/lib/legion/extensions/llm/transport/messages/fleet_request.rb +6 -8
- data/lib/legion/extensions/llm/transport/messages/fleet_response.rb +10 -9
- data/lib/legion/extensions/llm/utils.rb +23 -5
- data/lib/legion/extensions/llm/version.rb +1 -1
- data/lib/legion/extensions/llm.rb +8 -98
- data/spec/legion/extensions/llm/auto_registration_spec.rb +4 -9
- data/spec/legion/extensions/llm/canonical/chunk_spec.rb +66 -252
- data/spec/legion/extensions/llm/canonical/content_block_spec.rb +52 -197
- data/spec/legion/extensions/llm/canonical/message_spec.rb +89 -204
- data/spec/legion/extensions/llm/canonical/params_spec.rb +55 -136
- data/spec/legion/extensions/llm/canonical/request_spec.rb +81 -143
- data/spec/legion/extensions/llm/canonical/response_spec.rb +68 -204
- data/spec/legion/extensions/llm/canonical/thinking/config_spec.rb +222 -0
- data/spec/legion/extensions/llm/canonical/thinking_spec.rb +23 -171
- data/spec/legion/extensions/llm/canonical/tool_call_spec.rb +59 -162
- data/spec/legion/extensions/llm/canonical/tool_definition_spec.rb +55 -191
- data/spec/legion/extensions/llm/canonical/tool_schema_spec.rb +26 -67
- data/spec/legion/extensions/llm/canonical/usage_spec.rb +46 -155
- data/spec/legion/extensions/llm/configuration_spec.rb +31 -5
- data/spec/legion/extensions/llm/conformance/canonical_type_examples.rb +106 -0
- data/spec/legion/extensions/llm/conformance/client_translator_examples.rb +1 -2
- data/spec/legion/extensions/llm/conformance/conformance.rb +10 -2
- data/spec/legion/extensions/llm/conformance/fixtures/canonical_fleet_round_trip.json +1 -1
- data/spec/legion/extensions/llm/conformance/fixtures/canonical_thinking_request.json +2 -2
- data/spec/legion/extensions/llm/conformance/provider_translator_examples.rb +1 -1
- data/spec/legion/extensions/llm/conformance/ssot_contract_conformance_spec.rb +128 -0
- data/spec/legion/extensions/llm/conformance/ssot_contract_examples.rb +505 -0
- data/spec/legion/extensions/llm/conformance/ssot_provider_examples.rb +11 -10
- data/spec/legion/extensions/llm/credential_sources_spec.rb +12 -13
- data/spec/legion/extensions/llm/error_spec.rb +2 -12
- data/spec/legion/extensions/llm/fleet/exact_offering_spec.rb +59 -45
- data/spec/legion/extensions/llm/fleet/provider_responder_spec.rb +173 -95
- data/spec/legion/extensions/llm/fleet/token_validator_spec.rb +7 -2
- data/spec/legion/extensions/llm/fleet/worker_execution_spec.rb +93 -74
- data/spec/legion/extensions/llm/fleet_messages_spec.rb +119 -125
- data/spec/legion/extensions/llm/gemspec_spec.rb +1 -2
- data/spec/legion/extensions/llm/inventory/boot_spec.rb +4 -4
- data/spec/legion/extensions/llm/inventory/identity_spec.rb +127 -110
- data/spec/legion/extensions/llm/inventory/probe_token_spec.rb +4 -4
- data/spec/legion/extensions/llm/inventory/publisher_spec.rb +8 -61
- data/spec/legion/extensions/llm/inventory/records_spec.rb +82 -47
- data/spec/legion/extensions/llm/inventory/registry_activation_spec.rb +49 -16
- data/spec/legion/extensions/llm/inventory/registry_replacement_spec.rb +5 -4
- data/spec/legion/extensions/llm/inventory/snapshot_spec.rb +11 -5
- data/spec/legion/extensions/llm/inventory/weight_reconciler_spec.rb +7 -3
- data/spec/legion/extensions/llm/inventory/weight_schema_spec.rb +60 -10
- data/spec/legion/extensions/llm/provider/open_ai_compatible_spec.rb +128 -68
- data/spec/legion/extensions/llm/provider/open_ai_compatible_tool_calls_array_spec.rb +7 -31
- data/spec/legion/extensions/llm/provider_contract_spec.rb +10 -15
- data/spec/legion/extensions/llm/provider_spec.rb +97 -78
- data/spec/legion/extensions/llm/routing/records_spec.rb +38 -5
- data/spec/legion/extensions/llm/stream_accumulator_spec.rb +174 -144
- data/spec/legion/extensions/llm/streaming_spec.rb +27 -0
- data/spec/legion/extensions/llm/taxonomies_spec.rb +43 -43
- data/spec/legion/extensions/llm/transport/fleet_lane_spec.rb +1 -1
- data/spec/legion/extensions/llm/utils_spec.rb +26 -7
- data/spec/legion/extensions/llm_base_contract_spec.rb +55 -90
- data/spec/legion/extensions/llm_extension_spec.rb +5 -5
- data/spec/support/fake_llm_provider.rb +45 -39
- data/spec/support/fake_ssot_harness.rb +7 -2
- metadata +13 -54
- data/lib/legion/extensions/llm/agent.rb +0 -366
- data/lib/legion/extensions/llm/aliases.json +0 -436
- data/lib/legion/extensions/llm/aliases.rb +0 -67
- data/lib/legion/extensions/llm/attachment.rb +0 -229
- data/lib/legion/extensions/llm/chat.rb +0 -354
- data/lib/legion/extensions/llm/chunk.rb +0 -10
- data/lib/legion/extensions/llm/content.rb +0 -81
- data/lib/legion/extensions/llm/context.rb +0 -33
- data/lib/legion/extensions/llm/embedding.rb +0 -33
- data/lib/legion/extensions/llm/image.rb +0 -109
- data/lib/legion/extensions/llm/inventory/capabilities.rb +0 -40
- data/lib/legion/extensions/llm/inventory/scoped_refresher.rb +0 -310
- data/lib/legion/extensions/llm/message.rb +0 -118
- data/lib/legion/extensions/llm/mime_type.rb +0 -75
- data/lib/legion/extensions/llm/model/info.rb +0 -286
- data/lib/legion/extensions/llm/model/modalities.rb +0 -26
- data/lib/legion/extensions/llm/model/pricing.rb +0 -52
- data/lib/legion/extensions/llm/model/pricing_category.rb +0 -50
- data/lib/legion/extensions/llm/model/pricing_tier.rb +0 -37
- data/lib/legion/extensions/llm/model.rb +0 -11
- data/lib/legion/extensions/llm/models.json +0 -57313
- data/lib/legion/extensions/llm/models.rb +0 -530
- data/lib/legion/extensions/llm/models_schema.json +0 -168
- data/lib/legion/extensions/llm/moderation.rb +0 -60
- data/lib/legion/extensions/llm/registry_event_builder.rb +0 -141
- data/lib/legion/extensions/llm/registry_publisher.rb +0 -107
- data/lib/legion/extensions/llm/responses/chat_response.rb +0 -43
- data/lib/legion/extensions/llm/responses/embedding_response.rb +0 -38
- data/lib/legion/extensions/llm/responses/stream_chunk.rb +0 -43
- data/lib/legion/extensions/llm/routing/lane_key.rb +0 -66
- data/lib/legion/extensions/llm/routing/model_offering.rb +0 -241
- data/lib/legion/extensions/llm/routing/offering_registry.rb +0 -101
- data/lib/legion/extensions/llm/routing/registry_event.rb +0 -167
- data/lib/legion/extensions/llm/thinking.rb +0 -53
- data/lib/legion/extensions/llm/tokens.rb +0 -51
- data/lib/legion/extensions/llm/tool_call.rb +0 -34
- data/lib/legion/extensions/llm/transcription.rb +0 -39
- data/lib/legion/extensions/llm/transport/messages/registry_event.rb +0 -44
- data/spec/legion/extensions/llm/agent_spec.rb +0 -179
- data/spec/legion/extensions/llm/attachment_spec.rb +0 -25
- data/spec/legion/extensions/llm/conformance/fixtures/ssot_identity_vectors.json +0 -84
- data/spec/legion/extensions/llm/context_spec.rb +0 -127
- data/spec/legion/extensions/llm/inventory/capabilities_spec.rb +0 -43
- data/spec/legion/extensions/llm/inventory/scoped_refresher_spec.rb +0 -340
- data/spec/legion/extensions/llm/message_spec.rb +0 -64
- data/spec/legion/extensions/llm/model/info_spec.rb +0 -222
- data/spec/legion/extensions/llm/models_spec.rb +0 -104
- data/spec/legion/extensions/llm/registry_event_builder_spec.rb +0 -68
- data/spec/legion/extensions/llm/registry_publisher_spec.rb +0 -22
- data/spec/legion/extensions/llm/responses/response_objects_spec.rb +0 -75
- data/spec/legion/extensions/llm/routing/model_offering_spec.rb +0 -281
- data/spec/legion/extensions/llm/routing/offering_registry_spec.rb +0 -50
- data/spec/legion/extensions/llm/routing/registry_event_spec.rb +0 -120
|
@@ -1,104 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
require 'tempfile'
|
|
5
|
-
|
|
6
|
-
RSpec.describe Legion::Extensions::Llm::Models do
|
|
7
|
-
subject(:models) { described_class.new([chat_model, embedding_model]) }
|
|
8
|
-
|
|
9
|
-
include_context 'with fake llm provider'
|
|
10
|
-
|
|
11
|
-
let(:chat_model) do
|
|
12
|
-
Legion::Extensions::Llm::Model::Info.from_hash(
|
|
13
|
-
id: 'fake-chat-model',
|
|
14
|
-
name: 'Fake Chat Model',
|
|
15
|
-
provider: 'fake_llm',
|
|
16
|
-
family: 'fake',
|
|
17
|
-
modalities: { input: ['text'], output: ['text'] },
|
|
18
|
-
capabilities: %w[function_calling structured_output streaming]
|
|
19
|
-
)
|
|
20
|
-
end
|
|
21
|
-
|
|
22
|
-
let(:embedding_model) do
|
|
23
|
-
Legion::Extensions::Llm::Model::Info.from_hash(
|
|
24
|
-
id: 'fake-embedding-model',
|
|
25
|
-
name: 'Fake Embedding Model',
|
|
26
|
-
provider: 'fake_llm',
|
|
27
|
-
family: 'fake',
|
|
28
|
-
modalities: { input: ['text'], output: ['embeddings'] },
|
|
29
|
-
capabilities: ['embedding']
|
|
30
|
-
)
|
|
31
|
-
end
|
|
32
|
-
|
|
33
|
-
after do
|
|
34
|
-
described_class.instance_variable_set(:@instance, nil)
|
|
35
|
-
end
|
|
36
|
-
|
|
37
|
-
it 'filters models by provider and usage type' do
|
|
38
|
-
expect(models.by_provider(:fake_llm).map(&:id)).to eq(%w[fake-chat-model fake-embedding-model])
|
|
39
|
-
expect(models.chat_models.map(&:id)).to eq(['fake-chat-model'])
|
|
40
|
-
expect(models.embedding_models.map(&:id)).to eq(['fake-embedding-model'])
|
|
41
|
-
end
|
|
42
|
-
|
|
43
|
-
it 'resolves models through the provider registry' do
|
|
44
|
-
allow(described_class).to receive(:instance).and_return(models)
|
|
45
|
-
|
|
46
|
-
model, provider = described_class.resolve('fake-chat-model')
|
|
47
|
-
|
|
48
|
-
expect(model.id).to eq('fake-chat-model')
|
|
49
|
-
expect(provider).to be_a(SpecSupport::FakeLLMProvider)
|
|
50
|
-
end
|
|
51
|
-
|
|
52
|
-
it 'supports provider-specific model id normalization through provider hooks' do
|
|
53
|
-
normalizing_class = Class.new(SpecSupport::FakeLLMProvider) do
|
|
54
|
-
def self.resolve_model_id(model_id, config: nil) # rubocop:disable Lint/UnusedMethodArgument
|
|
55
|
-
model_id == 'alias-model' ? 'fake-chat-model' : model_id
|
|
56
|
-
end
|
|
57
|
-
end
|
|
58
|
-
|
|
59
|
-
# Temporarily define a namespace module so scan_provider_classes can find it
|
|
60
|
-
mod = Module.new do
|
|
61
|
-
const_set(:PROVIDER_FAMILY, :normalizing_fake)
|
|
62
|
-
end
|
|
63
|
-
mod.define_singleton_method(:provider_class) { normalizing_class }
|
|
64
|
-
stub_const('Legion::Extensions::Llm::NormalizingFakeProvider', mod)
|
|
65
|
-
|
|
66
|
-
normalized = Legion::Extensions::Llm::Model::Info.from_hash(
|
|
67
|
-
id: 'fake-chat-model',
|
|
68
|
-
name: 'Normalized',
|
|
69
|
-
provider: 'normalizing_fake',
|
|
70
|
-
modalities: { output: ['text'] }
|
|
71
|
-
)
|
|
72
|
-
registry = described_class.new([normalized])
|
|
73
|
-
|
|
74
|
-
expect(registry.find('alias-model', :normalizing_fake).id).to eq('fake-chat-model')
|
|
75
|
-
end
|
|
76
|
-
|
|
77
|
-
it 'raises when a model exists but its provider extension is not registered' do
|
|
78
|
-
registry = described_class.new([
|
|
79
|
-
Legion::Extensions::Llm::Model::Info.from_hash(
|
|
80
|
-
id: 'orphan-model',
|
|
81
|
-
name: 'Orphan',
|
|
82
|
-
provider: 'missing_provider'
|
|
83
|
-
)
|
|
84
|
-
])
|
|
85
|
-
allow(described_class).to receive(:instance).and_return(registry)
|
|
86
|
-
|
|
87
|
-
expect do
|
|
88
|
-
described_class.resolve('orphan-model')
|
|
89
|
-
end.to raise_error(Legion::Extensions::Llm::Error, /Unknown provider/)
|
|
90
|
-
end
|
|
91
|
-
|
|
92
|
-
it 'saves and loads model metadata as Legion JSON' do
|
|
93
|
-
temp_file = Tempfile.new(['models', '.json'])
|
|
94
|
-
models.save_to_json(temp_file.path)
|
|
95
|
-
|
|
96
|
-
parsed = Legion::JSON.parse(File.read(temp_file.path), symbolize_names: false)
|
|
97
|
-
expect(parsed.map { |model| model['id'] }).to eq(%w[fake-chat-model fake-embedding-model])
|
|
98
|
-
|
|
99
|
-
loaded = described_class.read_from_json(temp_file.path)
|
|
100
|
-
expect(loaded.map(&:id)).to eq(%w[fake-chat-model fake-embedding-model])
|
|
101
|
-
ensure
|
|
102
|
-
temp_file&.close!
|
|
103
|
-
end
|
|
104
|
-
end
|
|
@@ -1,68 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
|
|
5
|
-
RSpec.describe Legion::Extensions::Llm::RegistryEventBuilder do
|
|
6
|
-
subject(:builder) { described_class.new(provider_family: :ollama) }
|
|
7
|
-
|
|
8
|
-
describe '#provider_family' do
|
|
9
|
-
it 'normalizes to a downcased symbol' do
|
|
10
|
-
b = described_class.new(provider_family: 'Anthropic')
|
|
11
|
-
expect(b.provider_family).to eq(:anthropic)
|
|
12
|
-
end
|
|
13
|
-
end
|
|
14
|
-
|
|
15
|
-
describe '#model_available' do
|
|
16
|
-
let(:model) do
|
|
17
|
-
Legion::Extensions::Llm::Model::Info.from_hash(
|
|
18
|
-
id: 'llama-3.1-8b',
|
|
19
|
-
name: 'Llama 3.1 8B',
|
|
20
|
-
provider: 'ollama',
|
|
21
|
-
capabilities: %w[completion streaming],
|
|
22
|
-
modalities: { input: %w[text], output: %w[text] },
|
|
23
|
-
context_window: 128_000,
|
|
24
|
-
max_output_tokens: 8192
|
|
25
|
-
)
|
|
26
|
-
end
|
|
27
|
-
|
|
28
|
-
let(:readiness) { { ready: true, configured: true } }
|
|
29
|
-
|
|
30
|
-
it 'builds a RegistryEvent with offering data' do
|
|
31
|
-
event = builder.model_available(model, readiness: readiness)
|
|
32
|
-
expect(event).to be_a(Legion::Extensions::Llm::Routing::RegistryEvent)
|
|
33
|
-
expect(event.event_type).to eq(:offering_available)
|
|
34
|
-
expect(event.offering.model).to eq('llama-3.1-8b')
|
|
35
|
-
expect(event.offering.provider_family).to eq(:ollama)
|
|
36
|
-
end
|
|
37
|
-
|
|
38
|
-
it 'includes model health from readiness' do
|
|
39
|
-
event = builder.model_available(model, readiness: readiness)
|
|
40
|
-
expect(event.health[:ready]).to be true
|
|
41
|
-
expect(event.health[:status]).to eq(:available)
|
|
42
|
-
end
|
|
43
|
-
|
|
44
|
-
it 'includes extension metadata' do
|
|
45
|
-
event = builder.model_available(model, readiness: readiness)
|
|
46
|
-
expect(event.metadata[:extension]).to eq(:llm_ollama)
|
|
47
|
-
expect(event.metadata[:provider]).to eq(:ollama)
|
|
48
|
-
end
|
|
49
|
-
end
|
|
50
|
-
|
|
51
|
-
describe '#readiness' do
|
|
52
|
-
it 'builds an available event when ready' do
|
|
53
|
-
event = builder.readiness({ ready: true, configured: true })
|
|
54
|
-
expect(event.event_type).to eq(:offering_available)
|
|
55
|
-
end
|
|
56
|
-
|
|
57
|
-
it 'builds an unavailable event when not ready' do
|
|
58
|
-
event = builder.readiness({ ready: false, configured: true })
|
|
59
|
-
expect(event.event_type).to eq(:offering_unavailable)
|
|
60
|
-
end
|
|
61
|
-
|
|
62
|
-
it 'preserves error details from health' do
|
|
63
|
-
event = builder.readiness({ ready: false, health: { error: 'ConnectionRefused', message: 'refused' } })
|
|
64
|
-
expect(event.health[:error_class]).to eq('ConnectionRefused')
|
|
65
|
-
expect(event.health[:error]).to eq('refused')
|
|
66
|
-
end
|
|
67
|
-
end
|
|
68
|
-
end
|
|
@@ -1,22 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
|
|
5
|
-
RSpec.describe Legion::Extensions::Llm::RegistryPublisher do
|
|
6
|
-
subject(:publisher) { described_class.new(provider_family: :ollama, builder: builder) }
|
|
7
|
-
|
|
8
|
-
let(:builder) { instance_double(Legion::Extensions::Llm::RegistryEventBuilder) }
|
|
9
|
-
|
|
10
|
-
describe '#app_id' do
|
|
11
|
-
it 'includes the provider family' do
|
|
12
|
-
expect(publisher.app_id).to eq('lex-llm-ollama')
|
|
13
|
-
end
|
|
14
|
-
end
|
|
15
|
-
|
|
16
|
-
describe '#provider_family' do
|
|
17
|
-
it 'normalizes to a downcased symbol' do
|
|
18
|
-
pub = described_class.new(provider_family: 'Anthropic', builder: builder)
|
|
19
|
-
expect(pub.provider_family).to eq(:anthropic)
|
|
20
|
-
end
|
|
21
|
-
end
|
|
22
|
-
end
|
|
@@ -1,75 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
|
|
5
|
-
# rubocop:disable RSpec/DescribeClass
|
|
6
|
-
RSpec.describe 'LLM normalized response objects' do
|
|
7
|
-
let(:unsafe_metadata) do
|
|
8
|
-
{
|
|
9
|
-
reasoning_content: 'metadata secret',
|
|
10
|
-
reasoning: 'metadata reasoning',
|
|
11
|
-
thinking_text: 'metadata thinking',
|
|
12
|
-
raw: { reasoning_content: 'nested raw secret' },
|
|
13
|
-
'raw-response' => { reasoning_content: 'hyphen raw secret' },
|
|
14
|
-
'provider-body' => { thinking_text: 'hyphen body secret' },
|
|
15
|
-
vendor: 'vllm'
|
|
16
|
-
}
|
|
17
|
-
end
|
|
18
|
-
let(:unsafe_raw) { { 'choices' => [{ 'message' => { 'content' => '<think>raw secret</think>visible' } }] } }
|
|
19
|
-
|
|
20
|
-
it 'serializes chat responses without raw provider thinking fields' do
|
|
21
|
-
response = Legion::Extensions::Llm::Responses::ChatResponse.new(
|
|
22
|
-
content: "<think>tag secret</think>\nvisible",
|
|
23
|
-
metadata: unsafe_metadata,
|
|
24
|
-
raw: unsafe_raw
|
|
25
|
-
)
|
|
26
|
-
|
|
27
|
-
payload = response.to_h
|
|
28
|
-
encoded = Legion::JSON.dump(payload)
|
|
29
|
-
|
|
30
|
-
expect(payload).to eq(content: 'visible', metadata: { vendor: 'vllm' })
|
|
31
|
-
expect(encoded).not_to include('reasoning', 'reasoning_content', 'thinking_text', '<think>', 'raw secret',
|
|
32
|
-
'nested raw secret', 'hyphen raw secret', 'hyphen body secret')
|
|
33
|
-
expect(response.to_internal_h).to include(
|
|
34
|
-
thinking: 'metadata secretmetadata reasoningmetadata thinkingtag secret',
|
|
35
|
-
metadata: unsafe_metadata,
|
|
36
|
-
raw: unsafe_raw
|
|
37
|
-
)
|
|
38
|
-
end
|
|
39
|
-
|
|
40
|
-
it 'serializes stream chunks without raw provider thinking fields' do
|
|
41
|
-
chunk = Legion::Extensions::Llm::Responses::StreamChunk.new(
|
|
42
|
-
content: "<think>tag secret</think>\nvisible",
|
|
43
|
-
metadata: unsafe_metadata,
|
|
44
|
-
raw: unsafe_raw
|
|
45
|
-
)
|
|
46
|
-
|
|
47
|
-
payload = chunk.to_h
|
|
48
|
-
encoded = Legion::JSON.dump(payload)
|
|
49
|
-
|
|
50
|
-
expect(payload).to eq(content: 'visible', metadata: { vendor: 'vllm' })
|
|
51
|
-
expect(encoded).not_to include('reasoning', 'reasoning_content', 'thinking_text', '<think>', 'raw secret',
|
|
52
|
-
'nested raw secret', 'hyphen raw secret', 'hyphen body secret')
|
|
53
|
-
expect(chunk.to_internal_h).to include(
|
|
54
|
-
thinking: 'metadata secretmetadata reasoningmetadata thinkingtag secret',
|
|
55
|
-
metadata: unsafe_metadata,
|
|
56
|
-
raw: unsafe_raw
|
|
57
|
-
)
|
|
58
|
-
end
|
|
59
|
-
|
|
60
|
-
it 'serializes embedding responses without raw provider payloads' do
|
|
61
|
-
response = Legion::Extensions::Llm::Responses::EmbeddingResponse.new(
|
|
62
|
-
vectors: [[0.1]],
|
|
63
|
-
model: 'embed',
|
|
64
|
-
metadata: unsafe_metadata,
|
|
65
|
-
raw: unsafe_raw
|
|
66
|
-
)
|
|
67
|
-
|
|
68
|
-
payload = response.to_h
|
|
69
|
-
|
|
70
|
-
expect(payload).to eq(vectors: [[0.1]], model: 'embed', metadata: { vendor: 'vllm' })
|
|
71
|
-
expect(Legion::JSON.dump(payload)).not_to include('raw secret')
|
|
72
|
-
expect(response.to_internal_h).to include(metadata: unsafe_metadata, raw: unsafe_raw)
|
|
73
|
-
end
|
|
74
|
-
end
|
|
75
|
-
# rubocop:enable RSpec/DescribeClass
|
|
@@ -1,281 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
|
|
5
|
-
RSpec.describe Legion::Extensions::Llm::Routing::ModelOffering do
|
|
6
|
-
subject(:offering) do
|
|
7
|
-
described_class.new(
|
|
8
|
-
provider_family: :ollama,
|
|
9
|
-
instance_id: :'macbook-m4-max',
|
|
10
|
-
transport: :rabbitmq,
|
|
11
|
-
model: 'qwen3.6:27b-q4_K_M',
|
|
12
|
-
capabilities: %i[chat tools thinking],
|
|
13
|
-
limits: { context_window: 32_768, max_output_tokens: 8192 },
|
|
14
|
-
policy_tags: %i[phi_allowed internal_only]
|
|
15
|
-
)
|
|
16
|
-
end
|
|
17
|
-
|
|
18
|
-
it 'normalizes provider-neutral offering metadata' do
|
|
19
|
-
expect(offering).to have_attributes(
|
|
20
|
-
offering_id: 'ollama:macbook-m4-max:inference:qwen3-6-27b-q4-k-m',
|
|
21
|
-
provider_family: :ollama,
|
|
22
|
-
provider_instance: :'macbook-m4-max',
|
|
23
|
-
instance_id: :'macbook-m4-max',
|
|
24
|
-
transport: :rabbitmq,
|
|
25
|
-
tier: :fleet,
|
|
26
|
-
model: 'qwen3.6:27b-q4_K_M',
|
|
27
|
-
canonical_model_alias: 'qwen3.6:27b-q4_K_M',
|
|
28
|
-
usage_type: :inference
|
|
29
|
-
)
|
|
30
|
-
expect(offering.capabilities).to eq(%i[chat tools thinking])
|
|
31
|
-
expect(offering.context_window).to eq(32_768)
|
|
32
|
-
end
|
|
33
|
-
|
|
34
|
-
it 'accepts expanded contract fields while preserving instance_id compatibility' do
|
|
35
|
-
expanded = described_class.new(
|
|
36
|
-
offering_id: 'azure:gpt4o-prod',
|
|
37
|
-
provider_family: :azure_foundry,
|
|
38
|
-
model_family: :openai,
|
|
39
|
-
provider_instance: :eastus,
|
|
40
|
-
model: 'gpt4o-prod',
|
|
41
|
-
canonical_model_alias: 'gpt-4o',
|
|
42
|
-
routing_metadata: { region: 'eastus', deployment: 'gpt4o-prod' },
|
|
43
|
-
capabilities: %i[chat tools]
|
|
44
|
-
)
|
|
45
|
-
|
|
46
|
-
expect(expanded).to have_attributes(
|
|
47
|
-
offering_id: 'azure:gpt4o-prod',
|
|
48
|
-
provider_family: :azure_foundry,
|
|
49
|
-
model_family: :openai,
|
|
50
|
-
provider_instance: :eastus,
|
|
51
|
-
instance_id: :eastus,
|
|
52
|
-
model: 'gpt4o-prod',
|
|
53
|
-
canonical_model_alias: 'gpt-4o',
|
|
54
|
-
routing_metadata: { region: 'eastus', deployment: 'gpt4o-prod' }
|
|
55
|
-
)
|
|
56
|
-
expect(expanded.to_h).to include(
|
|
57
|
-
provider_instance: :eastus,
|
|
58
|
-
instance_id: :eastus,
|
|
59
|
-
canonical_model_alias: 'gpt-4o',
|
|
60
|
-
routing_metadata: { region: 'eastus', deployment: 'gpt4o-prod' }
|
|
61
|
-
)
|
|
62
|
-
end
|
|
63
|
-
|
|
64
|
-
it 'lifts model family and aliases from legacy metadata' do
|
|
65
|
-
legacy = described_class.new(
|
|
66
|
-
provider_family: :bedrock,
|
|
67
|
-
instance_id: :'us-east-1',
|
|
68
|
-
model: 'anthropic.claude-3-haiku-20240307-v1:0',
|
|
69
|
-
metadata: { model_family: :anthropic, alias: 'claude-3-haiku' }
|
|
70
|
-
)
|
|
71
|
-
|
|
72
|
-
expect(legacy.model_family).to eq(:anthropic)
|
|
73
|
-
expect(legacy.canonical_model_alias).to eq('claude-3-haiku')
|
|
74
|
-
expect(legacy.model_alias?('claude-3-haiku')).to be true
|
|
75
|
-
expect(legacy.model_alias?('anthropic.claude-3-haiku-20240307-v1:0')).to be true
|
|
76
|
-
end
|
|
77
|
-
|
|
78
|
-
it 'checks route eligibility without provider-specific code' do
|
|
79
|
-
expect(
|
|
80
|
-
offering.eligible_for?(
|
|
81
|
-
usage_type: :inference,
|
|
82
|
-
required_capabilities: %i[tools thinking],
|
|
83
|
-
min_context_window: 32_000,
|
|
84
|
-
policy_tags: [:phi_allowed]
|
|
85
|
-
)
|
|
86
|
-
).to be true
|
|
87
|
-
|
|
88
|
-
expect(offering.eligible_for?(min_context_window: 65_536)).to be false
|
|
89
|
-
expect(offering.eligible_for?(required_capabilities: [:vision])).to be false
|
|
90
|
-
end
|
|
91
|
-
|
|
92
|
-
it 'treats legacy function-calling capability names as tools support' do
|
|
93
|
-
legacy_tools = described_class.new(
|
|
94
|
-
provider_family: :vllm,
|
|
95
|
-
model: 'qwen-tools',
|
|
96
|
-
capabilities: %i[chat function_calling]
|
|
97
|
-
)
|
|
98
|
-
|
|
99
|
-
expect(legacy_tools.capabilities).to include(:tools)
|
|
100
|
-
expect(legacy_tools.capabilities).not_to include(:function_calling)
|
|
101
|
-
expect(legacy_tools.eligible_for?(required_capabilities: [:tools])).to be true
|
|
102
|
-
end
|
|
103
|
-
|
|
104
|
-
it 'treats disabled offerings as ineligible' do
|
|
105
|
-
disabled = described_class.new(
|
|
106
|
-
provider_family: :ollama,
|
|
107
|
-
instance_id: :local,
|
|
108
|
-
model: 'qwen',
|
|
109
|
-
metadata: { enabled: false }
|
|
110
|
-
)
|
|
111
|
-
|
|
112
|
-
expect(disabled).not_to be_enabled
|
|
113
|
-
expect(disabled.eligible_for?).to be false
|
|
114
|
-
end
|
|
115
|
-
|
|
116
|
-
it 'generates clean fleet inference lane keys with context windows' do
|
|
117
|
-
expect(offering.lane_key).to eq('llm.fleet.inference.qwen3-6-27b-q4-k-m.ctx32768')
|
|
118
|
-
end
|
|
119
|
-
|
|
120
|
-
it 'uses canonical model aliases for fleet lanes when provider deployments hide the base model' do
|
|
121
|
-
deployment = described_class.new(
|
|
122
|
-
provider_family: :azure_foundry,
|
|
123
|
-
provider_instance: :default,
|
|
124
|
-
model: 'gpt4o-prod',
|
|
125
|
-
canonical_model_alias: 'gpt-4o',
|
|
126
|
-
limits: { context_window: 128_000 }
|
|
127
|
-
)
|
|
128
|
-
|
|
129
|
-
expect(deployment.lane_key).to eq('llm.fleet.inference.gpt-4o.ctx128000')
|
|
130
|
-
end
|
|
131
|
-
|
|
132
|
-
it 'generates embedding lanes without context suffixes' do
|
|
133
|
-
embedding = described_class.new(
|
|
134
|
-
provider_family: :ollama,
|
|
135
|
-
instance_id: :'gpu-01',
|
|
136
|
-
transport: :rabbitmq,
|
|
137
|
-
model: 'nomic-embed-text:latest',
|
|
138
|
-
usage_type: :embed,
|
|
139
|
-
capabilities: [:embedding]
|
|
140
|
-
)
|
|
141
|
-
|
|
142
|
-
expect(embedding).to be_embedding
|
|
143
|
-
expect(embedding.lane_key).to eq('llm.fleet.embed.nomic-embed-text-latest')
|
|
144
|
-
end
|
|
145
|
-
|
|
146
|
-
it 'can include an eligibility fingerprint when lanes need stricter matching' do
|
|
147
|
-
key = offering.lane_key(include_fingerprint: true)
|
|
148
|
-
|
|
149
|
-
expect(key).to match(/\Allm\.fleet\.inference\.qwen3-6-27b-q4-k-m\.ctx32768\.elig\.[0-9a-f]{10}\z/)
|
|
150
|
-
expect(offering.eligibility_fingerprint).to eq(key.split('.').last)
|
|
151
|
-
end
|
|
152
|
-
|
|
153
|
-
it 'normalizes string numeric limits from JSON-backed settings' do
|
|
154
|
-
string_limits = described_class.new(
|
|
155
|
-
provider_family: :ollama,
|
|
156
|
-
model: 'qwen',
|
|
157
|
-
limits: { context_window: '32768', max_output_tokens: '8192' }
|
|
158
|
-
)
|
|
159
|
-
|
|
160
|
-
expect(string_limits.context_window).to eq(32_768)
|
|
161
|
-
expect(string_limits.max_output_tokens).to eq(8192)
|
|
162
|
-
expect(string_limits.eligible_for?(min_context_window: 32_000)).to be true
|
|
163
|
-
end
|
|
164
|
-
|
|
165
|
-
it 'normalizes string-keyed JSON-backed offering fields' do
|
|
166
|
-
json_offering = described_class.new(
|
|
167
|
-
'provider_family' => 'ollama',
|
|
168
|
-
'instance_id' => 'macbook-m4-max',
|
|
169
|
-
'transport' => 'rabbitmq',
|
|
170
|
-
'model' => 'nomic-embed-text',
|
|
171
|
-
'type' => 'embed',
|
|
172
|
-
'capabilities' => %w[embedding],
|
|
173
|
-
'limits' => { 'context_window' => '8192' },
|
|
174
|
-
'metadata' => { 'enabled' => true }
|
|
175
|
-
)
|
|
176
|
-
|
|
177
|
-
expect(json_offering.provider_family).to eq(:ollama)
|
|
178
|
-
expect(json_offering.instance_id).to eq(:'macbook-m4-max')
|
|
179
|
-
expect(json_offering.transport).to eq(:rabbitmq)
|
|
180
|
-
expect(json_offering.usage_type).to eq(:embedding)
|
|
181
|
-
expect(json_offering.context_window).to eq(8192)
|
|
182
|
-
expect(json_offering).to be_enabled
|
|
183
|
-
end
|
|
184
|
-
|
|
185
|
-
it 'treats string-keyed disabled metadata as ineligible' do
|
|
186
|
-
disabled = described_class.new(
|
|
187
|
-
'provider_family' => 'ollama',
|
|
188
|
-
'model' => 'qwen',
|
|
189
|
-
'metadata' => { 'enabled' => false }
|
|
190
|
-
)
|
|
191
|
-
|
|
192
|
-
expect(disabled).not_to be_enabled
|
|
193
|
-
expect(disabled.eligible_for?).to be false
|
|
194
|
-
end
|
|
195
|
-
|
|
196
|
-
it 'keeps sensitive metadata out of eligibility fingerprints' do
|
|
197
|
-
safe = described_class.new(
|
|
198
|
-
provider_family: :ollama,
|
|
199
|
-
model: 'qwen',
|
|
200
|
-
metadata: { eligibility: { endpoint_url: 'http://gpu.internal', network_boundary: :corp_lan } }
|
|
201
|
-
)
|
|
202
|
-
changed_secret = described_class.new(
|
|
203
|
-
provider_family: :ollama,
|
|
204
|
-
model: 'qwen',
|
|
205
|
-
metadata: { eligibility: { endpoint_url: 'http://other.internal', network_boundary: :corp_lan } }
|
|
206
|
-
)
|
|
207
|
-
|
|
208
|
-
expect(safe.eligibility_fingerprint).to eq(changed_secret.eligibility_fingerprint)
|
|
209
|
-
end
|
|
210
|
-
|
|
211
|
-
it 'serializes the normalized shape used by routers and registries' do
|
|
212
|
-
expect(offering.to_h).to include(
|
|
213
|
-
offering_id: 'ollama:macbook-m4-max:inference:qwen3-6-27b-q4-k-m',
|
|
214
|
-
provider_family: :ollama,
|
|
215
|
-
provider_instance: :'macbook-m4-max',
|
|
216
|
-
instance_id: :'macbook-m4-max',
|
|
217
|
-
tier: :fleet,
|
|
218
|
-
canonical_model_alias: 'qwen3.6:27b-q4_K_M',
|
|
219
|
-
usage_type: :inference,
|
|
220
|
-
limits: { context_window: 32_768, max_output_tokens: 8192 }
|
|
221
|
-
)
|
|
222
|
-
end
|
|
223
|
-
|
|
224
|
-
describe 'capability_sources' do
|
|
225
|
-
it 'accepts and exposes capability source metadata' do
|
|
226
|
-
sourced = described_class.new(
|
|
227
|
-
provider_family: :vllm,
|
|
228
|
-
provider_instance: :apollo,
|
|
229
|
-
model: 'gemma-4-12b-it',
|
|
230
|
-
capabilities: %i[streaming tools],
|
|
231
|
-
capability_sources: {
|
|
232
|
-
streaming: { value: true, source: :provider_envelope },
|
|
233
|
-
tools: { value: true, source: :instance_override },
|
|
234
|
-
embeddings: { value: false, source: :default_false }
|
|
235
|
-
}
|
|
236
|
-
)
|
|
237
|
-
|
|
238
|
-
expect(sourced.capabilities).to include(:streaming, :tools)
|
|
239
|
-
expect(sourced.capability_sources[:tools]).to eq(value: true, source: :instance_override)
|
|
240
|
-
expect(sourced.capability_sources[:embedding]).to eq(value: false, source: :default_false)
|
|
241
|
-
end
|
|
242
|
-
|
|
243
|
-
it 'includes capability_sources in to_h' do
|
|
244
|
-
sourced = described_class.new(
|
|
245
|
-
provider_family: :vllm,
|
|
246
|
-
provider_instance: :apollo,
|
|
247
|
-
model: 'gemma-4-12b-it',
|
|
248
|
-
capabilities: %i[streaming],
|
|
249
|
-
capability_sources: {
|
|
250
|
-
streaming: { value: true, source: :provider_envelope }
|
|
251
|
-
}
|
|
252
|
-
)
|
|
253
|
-
|
|
254
|
-
expect(sourced.to_h[:capability_sources]).to eq(
|
|
255
|
-
streaming: { value: true, source: :provider_envelope }
|
|
256
|
-
)
|
|
257
|
-
end
|
|
258
|
-
|
|
259
|
-
it 'normalizes string-keyed capability sources' do
|
|
260
|
-
sourced = described_class.new(
|
|
261
|
-
provider_family: :vllm,
|
|
262
|
-
model: 'test',
|
|
263
|
-
capability_sources: {
|
|
264
|
-
'streaming' => { 'value' => true, 'source' => 'provider_envelope' }
|
|
265
|
-
}
|
|
266
|
-
)
|
|
267
|
-
|
|
268
|
-
expect(sourced.capability_sources[:streaming]).to eq(value: true, source: :provider_envelope)
|
|
269
|
-
end
|
|
270
|
-
|
|
271
|
-
it 'defaults to empty hash when not provided' do
|
|
272
|
-
plain = described_class.new(
|
|
273
|
-
provider_family: :ollama,
|
|
274
|
-
model: 'test',
|
|
275
|
-
capabilities: %i[streaming]
|
|
276
|
-
)
|
|
277
|
-
|
|
278
|
-
expect(plain.capability_sources).to eq({})
|
|
279
|
-
end
|
|
280
|
-
end
|
|
281
|
-
end
|
|
@@ -1,50 +0,0 @@
|
|
|
1
|
-
# frozen_string_literal: true
|
|
2
|
-
|
|
3
|
-
require 'spec_helper'
|
|
4
|
-
|
|
5
|
-
RSpec.describe Legion::Extensions::Llm::Routing::OfferingRegistry do
|
|
6
|
-
subject(:registry) { described_class.new([chat, embedding]) }
|
|
7
|
-
|
|
8
|
-
let(:chat) do
|
|
9
|
-
Legion::Extensions::Llm::Routing::ModelOffering.new(
|
|
10
|
-
provider_family: :azure_foundry,
|
|
11
|
-
model_family: :openai,
|
|
12
|
-
provider_instance: :eastus,
|
|
13
|
-
model: 'gpt4o-prod',
|
|
14
|
-
canonical_model_alias: 'gpt-4o',
|
|
15
|
-
capabilities: %i[chat tools],
|
|
16
|
-
limits: { context_window: 128_000 }
|
|
17
|
-
)
|
|
18
|
-
end
|
|
19
|
-
|
|
20
|
-
let(:embedding) do
|
|
21
|
-
Legion::Extensions::Llm::Routing::ModelOffering.new(
|
|
22
|
-
provider_family: :bedrock,
|
|
23
|
-
model_family: :amazon,
|
|
24
|
-
instance_id: :'us-east-1',
|
|
25
|
-
model: 'amazon.titan-embed-text-v2:0',
|
|
26
|
-
canonical_model_alias: 'titan-embed-text-v2',
|
|
27
|
-
usage_type: :embedding,
|
|
28
|
-
capabilities: [:embedding]
|
|
29
|
-
)
|
|
30
|
-
end
|
|
31
|
-
|
|
32
|
-
it 'registers hashes and offerings by normalized offering_id' do
|
|
33
|
-
replacement = registry.register(
|
|
34
|
-
chat.to_h.merge(capabilities: %i[chat vision])
|
|
35
|
-
)
|
|
36
|
-
|
|
37
|
-
expect(registry.find(chat.offering_id)).to eq(replacement)
|
|
38
|
-
expect(registry.find(chat.offering_id).capabilities).to eq(%i[chat vision])
|
|
39
|
-
expect(registry.count).to eq(2)
|
|
40
|
-
end
|
|
41
|
-
|
|
42
|
-
it 'finds and filters offerings by the expanded routing contract' do
|
|
43
|
-
expect(registry.find_by_model_alias('gpt-4o')).to eq(chat)
|
|
44
|
-
expect(registry.filter(provider_family: :azure_foundry)).to eq([chat])
|
|
45
|
-
expect(registry.filter(model_family: :openai)).to eq([chat])
|
|
46
|
-
expect(registry.filter(provider_instance: :eastus)).to eq([chat])
|
|
47
|
-
expect(registry.filter(capability: :embedding)).to eq([embedding])
|
|
48
|
-
expect(registry.filter(model_alias: 'titan-embed-text-v2', usage_type: :embedding)).to eq([embedding])
|
|
49
|
-
end
|
|
50
|
-
end
|