little_ghost 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +50 -47
- data/docs/guides/{Core Concepts.md → core_concepts.md} +40 -20
- data/docs/guides/getting_started.md +164 -0
- data/lib/little_ghost/ag_ui/adapter.rb +1 -1
- data/lib/little_ghost/agent/context_management.rb +1 -4
- data/lib/little_ghost/agent.rb +94 -26
- data/lib/little_ghost/agent_builder.rb +1 -1
- data/lib/little_ghost/configuration.rb +210 -29
- data/lib/little_ghost/data/model_catalog.json +19178 -0
- data/lib/little_ghost/errors.rb +6 -0
- data/lib/little_ghost/invocation.rb +2 -17
- data/lib/little_ghost/model.rb +53 -33
- data/lib/little_ghost/model_capabilities.rb +6 -5
- data/lib/little_ghost/model_resolver.rb +316 -0
- data/lib/little_ghost/models/catalog/models_dev_source.rb +62 -0
- data/lib/little_ghost/models/catalog/source.rb +30 -0
- data/lib/little_ghost/models/catalog.rb +154 -0
- data/lib/little_ghost/models/catalog_snapshot.rb +44 -0
- data/lib/little_ghost/models/configuration.rb +43 -0
- data/lib/little_ghost/models/details.rb +48 -0
- data/lib/little_ghost/models/target.rb +29 -0
- data/lib/little_ghost/provider_registry.rb +75 -0
- data/lib/little_ghost/providers/anthropic/catalog_source.rb +31 -0
- data/lib/little_ghost/providers/anthropic.rb +222 -0
- data/lib/little_ghost/providers/base.rb +46 -0
- data/lib/little_ghost/providers/bedrock/aws_protocol.rb +132 -0
- data/lib/little_ghost/providers/bedrock/catalog_source.rb +200 -0
- data/lib/little_ghost/providers/bedrock/credential_resolver.rb +123 -0
- data/lib/little_ghost/providers/bedrock/http_client.rb +78 -0
- data/lib/little_ghost/providers/bedrock.rb +23 -15
- data/lib/little_ghost/providers/configuration.rb +79 -0
- data/lib/little_ghost/providers/gemini/catalog_source.rb +35 -0
- data/lib/little_ghost/providers/gemini.rb +204 -0
- data/lib/little_ghost/providers/open_router/catalog_source.rb +42 -0
- data/lib/little_ghost/providers/open_router.rb +6 -2
- data/lib/little_ghost/providers/openai_compatible.rb +19 -23
- data/lib/little_ghost/providers/vertex_ai/credential_resolver.rb +90 -0
- data/lib/little_ghost/providers/vertex_ai.rb +38 -0
- data/lib/little_ghost/run.rb +7 -7
- data/lib/little_ghost/runtime.rb +10 -7
- data/lib/little_ghost/sandbox.rb +5 -5
- data/lib/little_ghost/session_store.rb +3 -3
- data/lib/little_ghost/structured_output.rb +2 -8
- data/lib/little_ghost/support/http_client.rb +186 -0
- data/lib/little_ghost/{providers → support}/sse_parser.rb +1 -1
- data/lib/little_ghost/tool.rb +5 -1
- data/lib/little_ghost/tool_registry.rb +2 -3
- data/lib/little_ghost/version.rb +1 -1
- data/lib/little_ghost/workflow.rb +1 -1
- data/lib/little_ghost.rb +21 -15
- metadata +28 -7
- data/docs/guides/Getting Started.md +0 -187
- data/lib/little_ghost/default_model_registry.rb +0 -71
- data/lib/little_ghost/model_registry.rb +0 -173
- data/lib/little_ghost/providers/http_transport.rb +0 -149
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LittleGhost
|
|
4
|
+
module Providers
|
|
5
|
+
# Holds trusted provider connection settings independently from model
|
|
6
|
+
# profiles. Connection names and option keys are normalized to strings, and
|
|
7
|
+
# the resulting mapping is immutable.
|
|
8
|
+
class Configuration
|
|
9
|
+
# Normalized provider connections keyed by application-defined name.
|
|
10
|
+
attr_reader :connections
|
|
11
|
+
|
|
12
|
+
# Copies and freezes +connections+ so callers may safely reuse their input.
|
|
13
|
+
def initialize(connections = {})
|
|
14
|
+
unless connections.is_a?(Hash)
|
|
15
|
+
raise ConfigurationError, "providers must be a mapping"
|
|
16
|
+
end
|
|
17
|
+
|
|
18
|
+
@connections = normalize(connections).freeze
|
|
19
|
+
validate!
|
|
20
|
+
freeze
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
# Returns credentials merged into +configuration+ when +provider+ is
|
|
24
|
+
# constructed with +adapter+. Subclasses may resolve secrets lazily here.
|
|
25
|
+
# The base implementation adds no credentials.
|
|
26
|
+
def credentials(provider:, adapter:, configuration:)
|
|
27
|
+
{}
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
private
|
|
31
|
+
|
|
32
|
+
def normalize(connections)
|
|
33
|
+
connections.to_h do |name, options|
|
|
34
|
+
unless options.is_a?(Hash)
|
|
35
|
+
raise ConfigurationError, "providers.#{name} must be a mapping"
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
normalized = options.to_h { |key, value| [key.to_s, deep_copy(value)] }
|
|
39
|
+
normalized["adapter"] = normalized["adapter"].to_s if normalized.key?("adapter")
|
|
40
|
+
[name.to_s, deep_freeze(normalized)]
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
def deep_copy(value)
|
|
45
|
+
case value
|
|
46
|
+
when Hash
|
|
47
|
+
value.to_h { |key, child| [key.to_s, deep_copy(child)] }
|
|
48
|
+
when Array
|
|
49
|
+
value.map { |child| deep_copy(child) }
|
|
50
|
+
when String
|
|
51
|
+
value.dup
|
|
52
|
+
else
|
|
53
|
+
value
|
|
54
|
+
end
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
def deep_freeze(value)
|
|
58
|
+
case value
|
|
59
|
+
when Hash
|
|
60
|
+
value.each_value { |child| deep_freeze(child) }
|
|
61
|
+
when Array
|
|
62
|
+
value.each { |child| deep_freeze(child) }
|
|
63
|
+
when String
|
|
64
|
+
value.freeze
|
|
65
|
+
end
|
|
66
|
+
value.freeze if value.is_a?(Hash) || value.is_a?(Array)
|
|
67
|
+
value
|
|
68
|
+
end
|
|
69
|
+
|
|
70
|
+
def validate!
|
|
71
|
+
connections.each do |name, options|
|
|
72
|
+
if options["adapter"].to_s.empty?
|
|
73
|
+
raise ConfigurationError, "providers.#{name}.adapter is required"
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
end
|
|
79
|
+
end
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LittleGhost
|
|
4
|
+
module Providers
|
|
5
|
+
class Gemini < Base
|
|
6
|
+
# Enriches Gemini availability and limits from the Developer API.
|
|
7
|
+
class CatalogSource < Models::Catalog::Source
|
|
8
|
+
URL = URI("https://generativelanguage.googleapis.com/v1beta/models") # :nodoc:
|
|
9
|
+
|
|
10
|
+
# Creates a source for the named provider connection.
|
|
11
|
+
def initialize(provider:, credential_resolver:)
|
|
12
|
+
super(name: "gemini")
|
|
13
|
+
@provider = provider
|
|
14
|
+
@credential_resolver = credential_resolver
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def refresh(target: nil)
|
|
18
|
+
api_key = @credential_resolver.call.fetch("api_key")
|
|
19
|
+
values = JSON.parse(
|
|
20
|
+
Support::HTTPClient.new(open_timeout: 5, read_timeout: 30, max_response_bytes: 25 * 1024 * 1024)
|
|
21
|
+
.request(uri: URL, headers: {"x-goog-api-key" => api_key})
|
|
22
|
+
).fetch("models")
|
|
23
|
+
values.select! { |value| value["name"].to_s.delete_prefix("models/") == target.model_id } if target
|
|
24
|
+
values.to_h do |value|
|
|
25
|
+
id = value.fetch("name").delete_prefix("models/")
|
|
26
|
+
["#{@provider}:#{id}", {available: true, context_window: value["inputTokenLimit"],
|
|
27
|
+
max_output_tokens: value["outputTokenLimit"]}.compact]
|
|
28
|
+
end
|
|
29
|
+
rescue JSON::ParserError, KeyError => error
|
|
30
|
+
raise ProviderError, "Gemini returned an invalid catalog: #{error.message}"
|
|
31
|
+
end
|
|
32
|
+
end
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "base64"
|
|
4
|
+
require "json"
|
|
5
|
+
require "uri"
|
|
6
|
+
require_relative "../support/http_client"
|
|
7
|
+
require_relative "../support/sse_parser"
|
|
8
|
+
require_relative "gemini/catalog_source"
|
|
9
|
+
|
|
10
|
+
module LittleGhost
|
|
11
|
+
module Providers
|
|
12
|
+
# Zero-dependency Gemini generateContent adapter.
|
|
13
|
+
class Gemini < Base
|
|
14
|
+
# Request policy supported by Gemini and Vertex AI HTTP clients.
|
|
15
|
+
def self.request_options = %i[max_response_bytes open_timeout read_timeout].freeze
|
|
16
|
+
|
|
17
|
+
DEFAULT_BASE_URL = "https://generativelanguage.googleapis.com/v1beta/" # :nodoc:
|
|
18
|
+
|
|
19
|
+
# Provider-owned model identifier.
|
|
20
|
+
attr_reader :model
|
|
21
|
+
|
|
22
|
+
# Creates a Gemini generateContent client for +model+.
|
|
23
|
+
def initialize(api_key:, model:, base_url: DEFAULT_BASE_URL, open_timeout: 10, read_timeout: 120,
|
|
24
|
+
max_response_bytes: Support::HTTPClient::DEFAULT_MAX_RESPONSE_BYTES, transport: nil, **)
|
|
25
|
+
raise CredentialError, "Gemini api_key is required" if api_key.to_s.empty?
|
|
26
|
+
|
|
27
|
+
@api_key = api_key
|
|
28
|
+
@model = model
|
|
29
|
+
@transport = transport || Support::HTTPClient.new(base_url:, open_timeout:, read_timeout:, max_response_bytes:)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
def stream(request)
|
|
33
|
+
return enum_for(__method__, request) unless block_given?
|
|
34
|
+
|
|
35
|
+
parser = Support::SSEParser.new
|
|
36
|
+
normalizer = Normalizer.new(model:)
|
|
37
|
+
@transport.stream(
|
|
38
|
+
path: endpoint,
|
|
39
|
+
headers: request_headers(request),
|
|
40
|
+
body: JSON.generate(request_body(request)),
|
|
41
|
+
cancellation_token: request.cancellation_token,
|
|
42
|
+
deadline: request.deadline
|
|
43
|
+
) do |chunk|
|
|
44
|
+
parser.<<(chunk).each { |data| normalizer.consume(JSON.parse(data)).each { |event| yield event } }
|
|
45
|
+
end
|
|
46
|
+
parser.finish.each { |data| normalizer.consume(JSON.parse(data)).each { |event| yield event } }
|
|
47
|
+
normalizer.finish.each { |event| yield event }
|
|
48
|
+
rescue JSON::ParserError => error
|
|
49
|
+
raise ProtocolError, "Google returned invalid JSON: #{error.message}"
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def capabilities(metadata: {})
|
|
53
|
+
ModelCapabilities.new(native_structured_output: true, tools: true, tool_choice: true,
|
|
54
|
+
supported_parameters: metadata[:supported_parameters])
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
protected
|
|
58
|
+
|
|
59
|
+
def endpoint
|
|
60
|
+
"models/#{URI.encode_www_form_component(model)}:streamGenerateContent?alt=sse&key=#{URI.encode_www_form_component(@api_key)}"
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def request_headers(_request) = {"content-type" => "application/json", "accept" => "text/event-stream"}
|
|
64
|
+
|
|
65
|
+
private
|
|
66
|
+
|
|
67
|
+
def request_body(request)
|
|
68
|
+
system, messages = request.messages.partition { |message| message.role == :system }
|
|
69
|
+
body = {contents: messages.map { |message| google_message(message) }}
|
|
70
|
+
body[:systemInstruction] = {parts: system.flat_map { |message| message.content.grep(Content::Text).map { |block| {text: block.text} } }} unless system.empty?
|
|
71
|
+
body[:tools] = [{functionDeclarations: request.tools.map { |tool| google_tool(tool) }}] unless request.tools.empty?
|
|
72
|
+
body[:toolConfig] = google_tool_choice(request.tool_choice) if request.tool_choice
|
|
73
|
+
generation = request.settings.to_h.transform_keys { |key| google_setting(key) }
|
|
74
|
+
if request.output_schema
|
|
75
|
+
generation[:responseMimeType] = "application/json"
|
|
76
|
+
generation[:responseJsonSchema] = request.output_schema.fetch(:schema)
|
|
77
|
+
end
|
|
78
|
+
body[:generationConfig] = generation unless generation.empty?
|
|
79
|
+
body
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
def google_setting(key)
|
|
83
|
+
{max_tokens: :maxOutputTokens, top_p: :topP, top_k: :topK, stop_sequences: :stopSequences}[key.to_sym] || key.to_sym
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def google_message(message)
|
|
87
|
+
{role: (message.role == :assistant) ? "model" : "user", parts: message.content.map { |block| google_content(block) }}
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
def google_content(block)
|
|
91
|
+
case block
|
|
92
|
+
when Content::Text then {text: block.text}
|
|
93
|
+
when Content::Image, Content::Document
|
|
94
|
+
{inlineData: {mimeType: block.media_type, data: Base64.strict_encode64(block.data)}}
|
|
95
|
+
when Content::ToolUse then {functionCall: {id: block.id, name: block.name, args: block.input}}
|
|
96
|
+
when Content::ToolResult
|
|
97
|
+
{functionResponse: {id: block.tool_use_id, name: block.tool_use_id, response: {output: Array(block.content).join("\n")}}}
|
|
98
|
+
when Content::Reasoning then {text: block.text, thought: true}
|
|
99
|
+
else raise ConfigurationError, "Unsupported Google content block: #{block.class}"
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def google_tool(tool)
|
|
104
|
+
value = tool.is_a?(Hash) ? tool.transform_keys(&:to_sym) : {name: tool.name, description: tool.description, input_schema: tool.input_schema}
|
|
105
|
+
{name: value.fetch(:name), description: value[:description], parametersJsonSchema: value[:input_schema] || {}}
|
|
106
|
+
end
|
|
107
|
+
|
|
108
|
+
def google_tool_choice(choice)
|
|
109
|
+
return {functionCallingConfig: {mode: "ANY"}} if choice == :required
|
|
110
|
+
|
|
111
|
+
{functionCallingConfig: {mode: "ANY", allowedFunctionNames: [choice.fetch(:name).to_s]}}
|
|
112
|
+
end
|
|
113
|
+
|
|
114
|
+
class Normalizer # :nodoc:
|
|
115
|
+
def initialize(model:)
|
|
116
|
+
@model = model
|
|
117
|
+
@text = +""
|
|
118
|
+
@reasoning = +""
|
|
119
|
+
@tools = []
|
|
120
|
+
@usage = Usage.new
|
|
121
|
+
@started = false
|
|
122
|
+
@terminal = false
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
def consume(event)
|
|
126
|
+
raise ProviderError, "Google request failed: #{event.dig("error", "message")}" if event["error"]
|
|
127
|
+
|
|
128
|
+
events = []
|
|
129
|
+
unless @started
|
|
130
|
+
@started = true
|
|
131
|
+
events << StreamEvent.build(:message_start, id: nil, model: event["modelVersion"] || @model)
|
|
132
|
+
end
|
|
133
|
+
candidate = event.fetch("candidates", []).first
|
|
134
|
+
if candidate
|
|
135
|
+
candidate.dig("content", "parts")&.each { |part| events.concat(part_events(part)) }
|
|
136
|
+
if candidate["finishReason"]
|
|
137
|
+
@stop_reason = normalize_stop(candidate["finishReason"])
|
|
138
|
+
@terminal = true
|
|
139
|
+
end
|
|
140
|
+
end
|
|
141
|
+
if event["usageMetadata"]
|
|
142
|
+
@usage = usage(event.fetch("usageMetadata"))
|
|
143
|
+
events << StreamEvent.build(:usage, usage: @usage)
|
|
144
|
+
end
|
|
145
|
+
events
|
|
146
|
+
end
|
|
147
|
+
|
|
148
|
+
def finish
|
|
149
|
+
raise ProtocolError, "Google stream ended before a finish reason" unless @terminal
|
|
150
|
+
|
|
151
|
+
blocks = []
|
|
152
|
+
blocks << Content::Reasoning.new(text: @reasoning) unless @reasoning.empty?
|
|
153
|
+
blocks << Content::Text.new(text: @text) unless @text.empty?
|
|
154
|
+
blocks.concat(@tools)
|
|
155
|
+
response = ModelResponse.new(message: Message.new(role: :assistant, content: blocks),
|
|
156
|
+
stop_reason: @stop_reason, usage: @usage, metadata: {model: @model})
|
|
157
|
+
[StreamEvent.build(:message_stop, response:)]
|
|
158
|
+
end
|
|
159
|
+
|
|
160
|
+
private
|
|
161
|
+
|
|
162
|
+
def part_events(part)
|
|
163
|
+
if part["functionCall"]
|
|
164
|
+
call = part.fetch("functionCall")
|
|
165
|
+
index = @tools.length
|
|
166
|
+
tool = Content::ToolUse.new(id: call["id"] || "call-#{index}", name: call.fetch("name"), input: call["args"] || {})
|
|
167
|
+
@tools << tool
|
|
168
|
+
[
|
|
169
|
+
StreamEvent.build(:tool_call_start, index:, id: tool.id, name: tool.name),
|
|
170
|
+
StreamEvent.build(:tool_call_stop, index:, tool_use: tool)
|
|
171
|
+
]
|
|
172
|
+
elsif part["text"] && part["thought"]
|
|
173
|
+
@reasoning << part["text"]
|
|
174
|
+
[StreamEvent.build(:reasoning_delta, text: part["text"])]
|
|
175
|
+
elsif part["text"]
|
|
176
|
+
@text << part["text"]
|
|
177
|
+
[StreamEvent.build(:text_delta, text: part["text"])]
|
|
178
|
+
else
|
|
179
|
+
[]
|
|
180
|
+
end
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def usage(value)
|
|
184
|
+
cached = value["cachedContentTokenCount"] || 0
|
|
185
|
+
reasoning = value["thoughtsTokenCount"] || 0
|
|
186
|
+
Usage.new(
|
|
187
|
+
input_tokens: [Integer(value["promptTokenCount"] || 0) - Integer(cached), 0].max,
|
|
188
|
+
output_tokens: [Integer(value["candidatesTokenCount"] || 0) - Integer(reasoning), 0].max,
|
|
189
|
+
cache_read_tokens: cached,
|
|
190
|
+
reasoning_tokens: reasoning
|
|
191
|
+
)
|
|
192
|
+
end
|
|
193
|
+
|
|
194
|
+
def normalize_stop(value)
|
|
195
|
+
case value
|
|
196
|
+
when "MAX_TOKENS" then :max_tokens
|
|
197
|
+
when "SAFETY", "BLOCKLIST", "PROHIBITED_CONTENT", "SPII" then :content_filter
|
|
198
|
+
else @tools.empty? ? :end_turn : :tool_use
|
|
199
|
+
end
|
|
200
|
+
end
|
|
201
|
+
end
|
|
202
|
+
end
|
|
203
|
+
end
|
|
204
|
+
end
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LittleGhost
|
|
4
|
+
module Providers
|
|
5
|
+
class OpenRouter < OpenAICompatible
|
|
6
|
+
# Adds richer routing metadata and pricing from OpenRouter's live catalog.
|
|
7
|
+
class CatalogSource < Models::Catalog::Source
|
|
8
|
+
URL = URI("https://openrouter.ai/api/v1/models") # :nodoc:
|
|
9
|
+
|
|
10
|
+
# Creates a source for the named provider connection.
|
|
11
|
+
def initialize(provider:, credential_resolver:)
|
|
12
|
+
super(name: "openrouter")
|
|
13
|
+
@provider = provider
|
|
14
|
+
@credential_resolver = credential_resolver
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def refresh(target: nil)
|
|
18
|
+
api_key = @credential_resolver.call.fetch("api_key")
|
|
19
|
+
values = JSON.parse(
|
|
20
|
+
Support::HTTPClient.new(open_timeout: 5, read_timeout: 30, max_response_bytes: 25 * 1024 * 1024)
|
|
21
|
+
.request(uri: URL, headers: {"Authorization" => "Bearer #{api_key}"})
|
|
22
|
+
).fetch("data")
|
|
23
|
+
values.select! { |value| value["id"] == target.model_id } if target
|
|
24
|
+
values.to_h do |value|
|
|
25
|
+
pricing = value.fetch("pricing", {}).each_with_object({}) do |(key, amount), result|
|
|
26
|
+
normalized = {"prompt" => :input, "completion" => :output, "input_cache_read" => :cache_read,
|
|
27
|
+
"input_cache_write" => :cache_write}[key]
|
|
28
|
+
result[normalized] = Float(amount) * 1_000_000 if normalized
|
|
29
|
+
end
|
|
30
|
+
["#{@provider}:#{value.fetch("id")}", {
|
|
31
|
+
context_window: value["context_length"], max_output_tokens: value.dig("top_provider", "max_completion_tokens"),
|
|
32
|
+
supported_parameters: value["supported_parameters"], input_modalities: value.dig("architecture", "input_modalities"),
|
|
33
|
+
output_modalities: value.dig("architecture", "output_modalities"), pricing:
|
|
34
|
+
}.compact]
|
|
35
|
+
end
|
|
36
|
+
rescue JSON::ParserError, KeyError, ArgumentError => error
|
|
37
|
+
raise ProviderError, "OpenRouter returned an invalid catalog: #{error.message}"
|
|
38
|
+
end
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|
|
@@ -1,12 +1,13 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require_relative "openai_compatible"
|
|
4
|
+
require_relative "open_router/catalog_source"
|
|
4
5
|
|
|
5
6
|
module LittleGhost
|
|
6
7
|
module Providers
|
|
7
8
|
# OpenRouter gives one LittleGhost provider access to models routed through
|
|
8
|
-
# OpenRouter. Agents
|
|
9
|
-
#
|
|
9
|
+
# OpenRouter. Agents may select an OpenRouter target directly or keep a
|
|
10
|
+
# logical role while shared configuration selects its physical model.
|
|
10
11
|
#
|
|
11
12
|
# provider = LittleGhost::Providers::OpenRouter.new(
|
|
12
13
|
# api_key: ENV.fetch("OPENROUTER_API_KEY"),
|
|
@@ -19,6 +20,9 @@ module LittleGhost
|
|
|
19
20
|
# Requests that need a capability ask OpenRouter to route only to providers
|
|
20
21
|
# that advertise it.
|
|
21
22
|
class OpenRouter < OpenAICompatible
|
|
23
|
+
# Adds OpenRouter attribution to the shared OpenAI-compatible policy.
|
|
24
|
+
def self.request_options = (super + [:app_name]).freeze
|
|
25
|
+
|
|
22
26
|
# The OpenRouter API endpoint used when +base_url+ is omitted.
|
|
23
27
|
DEFAULT_BASE_URL = "https://openrouter.ai/api/v1/"
|
|
24
28
|
TOP_LEVEL_CACHE_MODELS = ["anthropic/", "~anthropic/"].freeze # :nodoc:
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
# frozen_string_literal: true
|
|
2
2
|
|
|
3
3
|
require "json"
|
|
4
|
-
require_relative "sse_parser"
|
|
5
|
-
require_relative "
|
|
4
|
+
require_relative "../support/sse_parser"
|
|
5
|
+
require_relative "../support/http_client"
|
|
6
6
|
|
|
7
7
|
module LittleGhost
|
|
8
8
|
module Providers
|
|
@@ -25,7 +25,12 @@ module LittleGhost
|
|
|
25
25
|
# A +:model_retry+ event reports each retry and whether text had already been
|
|
26
26
|
# emitted. Partial text may repeat after a retry, so consumers that assemble
|
|
27
27
|
# streams must use that event to discard or replace superseded output.
|
|
28
|
-
class OpenAICompatible
|
|
28
|
+
class OpenAICompatible < Base
|
|
29
|
+
# Request policy supported by OpenAI-compatible HTTP clients.
|
|
30
|
+
def self.request_options
|
|
31
|
+
%i[max_response_bytes max_retries max_retry_delay open_timeout read_timeout].freeze
|
|
32
|
+
end
|
|
33
|
+
|
|
29
34
|
# The OpenAI API endpoint used when +base_url+ is omitted.
|
|
30
35
|
DEFAULT_BASE_URL = "https://api.openai.com/v1/"
|
|
31
36
|
INITIAL_RETRY_DELAY = 1 # :nodoc:
|
|
@@ -89,7 +94,7 @@ module LittleGhost
|
|
|
89
94
|
open_timeout: 10,
|
|
90
95
|
read_timeout: 120,
|
|
91
96
|
allow_insecure_http: false,
|
|
92
|
-
max_response_bytes:
|
|
97
|
+
max_response_bytes: Support::HTTPClient::DEFAULT_MAX_RESPONSE_BYTES,
|
|
93
98
|
max_retries: 2,
|
|
94
99
|
max_retry_delay: MAX_RETRY_DELAY,
|
|
95
100
|
transport: nil,
|
|
@@ -104,7 +109,7 @@ module LittleGhost
|
|
|
104
109
|
@headers = headers.transform_keys(&:to_s).freeze
|
|
105
110
|
@max_retries = Integer(max_retries)
|
|
106
111
|
@max_retry_delay = Integer(max_retry_delay)
|
|
107
|
-
@transport = transport ||
|
|
112
|
+
@transport = transport || Support::HTTPClient.new(
|
|
108
113
|
base_url:,
|
|
109
114
|
open_timeout:,
|
|
110
115
|
read_timeout:,
|
|
@@ -155,10 +160,10 @@ module LittleGhost
|
|
|
155
160
|
end
|
|
156
161
|
end
|
|
157
162
|
|
|
158
|
-
# Returns the
|
|
163
|
+
# Returns the permissive capability contract expected from compatible APIs.
|
|
159
164
|
# Subclasses can override this when the endpoint advertises precise support.
|
|
160
165
|
def capabilities(metadata: {})
|
|
161
|
-
ModelCapabilities.
|
|
166
|
+
ModelCapabilities.permissive
|
|
162
167
|
end
|
|
163
168
|
|
|
164
169
|
private
|
|
@@ -186,14 +191,14 @@ module LittleGhost
|
|
|
186
191
|
|
|
187
192
|
def context_window_overflow?(error)
|
|
188
193
|
values = [error.message]
|
|
189
|
-
values << error.body if error.
|
|
194
|
+
values << error.body if error.is_a?(HTTPError)
|
|
190
195
|
values << error.error_type << error.code if error.is_a?(StreamError)
|
|
191
196
|
text = values.compact.join(" ").downcase
|
|
192
197
|
CONTEXT_OVERFLOW_MARKERS.any? { |marker| text.include?(marker) }
|
|
193
198
|
end
|
|
194
199
|
|
|
195
200
|
def stream_once(request)
|
|
196
|
-
parser = SSEParser.new
|
|
201
|
+
parser = Support::SSEParser.new
|
|
197
202
|
normalizer = normalizer_for(request)
|
|
198
203
|
|
|
199
204
|
@transport.stream(
|
|
@@ -382,21 +387,12 @@ module LittleGhost
|
|
|
382
387
|
end
|
|
383
388
|
|
|
384
389
|
def tool_definition(tool)
|
|
385
|
-
|
|
386
|
-
definition = tool.transform_keys(&:to_sym)
|
|
387
|
-
return {
|
|
388
|
-
name: definition.fetch(:name),
|
|
389
|
-
description: definition[:description],
|
|
390
|
-
input_schema: definition[:input_schema] || {},
|
|
391
|
-
strict: definition[:strict]
|
|
392
|
-
}
|
|
393
|
-
end
|
|
394
|
-
|
|
390
|
+
definition = tool.to_h.transform_keys(&:to_sym)
|
|
395
391
|
{
|
|
396
|
-
name:
|
|
397
|
-
description:
|
|
398
|
-
input_schema:
|
|
399
|
-
strict:
|
|
392
|
+
name: definition.fetch(:name),
|
|
393
|
+
description: definition[:description],
|
|
394
|
+
input_schema: definition[:input_schema] || {},
|
|
395
|
+
strict: definition[:strict]
|
|
400
396
|
}
|
|
401
397
|
end
|
|
402
398
|
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "base64"
|
|
4
|
+
require "json"
|
|
5
|
+
require "openssl"
|
|
6
|
+
require "uri"
|
|
7
|
+
|
|
8
|
+
module LittleGhost
|
|
9
|
+
module Providers
|
|
10
|
+
class VertexAI < Gemini
|
|
11
|
+
# Resolves Vertex access tokens from explicit values, service-account ADC,
|
|
12
|
+
# or the Google metadata server.
|
|
13
|
+
class CredentialResolver
|
|
14
|
+
TOKEN_URI = URI("https://oauth2.googleapis.com/token") # :nodoc:
|
|
15
|
+
SCOPE = "https://www.googleapis.com/auth/cloud-platform" # :nodoc:
|
|
16
|
+
|
|
17
|
+
# Uses an explicit token or Google application credentials in +environment+.
|
|
18
|
+
def initialize(environment: ENV, access_token: nil, clock: -> { Time.now.to_i })
|
|
19
|
+
@environment = environment
|
|
20
|
+
@access_token = access_token
|
|
21
|
+
@clock = clock
|
|
22
|
+
@mutex = Mutex.new
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
# Returns a current access token, refreshing cached credentials as needed.
|
|
26
|
+
def call(cancellation_token: nil, deadline: nil)
|
|
27
|
+
return @access_token unless @access_token.to_s.empty?
|
|
28
|
+
|
|
29
|
+
@mutex.synchronize do
|
|
30
|
+
return @cached_token if @cached_token && @expires_at && @expires_at > @clock.call + 60
|
|
31
|
+
|
|
32
|
+
file = @environment["GOOGLE_APPLICATION_CREDENTIALS"]
|
|
33
|
+
token, expires_in = if file
|
|
34
|
+
service_account_token(file, cancellation_token:, deadline:)
|
|
35
|
+
else
|
|
36
|
+
metadata_token(cancellation_token:, deadline:)
|
|
37
|
+
end
|
|
38
|
+
@cached_token = token
|
|
39
|
+
@expires_at = @clock.call + Integer(expires_in || 3600)
|
|
40
|
+
token
|
|
41
|
+
end
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
private
|
|
45
|
+
|
|
46
|
+
def service_account_token(path, cancellation_token:, deadline:)
|
|
47
|
+
document = JSON.parse(File.read(path))
|
|
48
|
+
raise CredentialError, "Google application credentials must be a service_account" unless document["type"] == "service_account"
|
|
49
|
+
|
|
50
|
+
now = @clock.call
|
|
51
|
+
header = urlsafe(JSON.generate(alg: "RS256", typ: "JWT"))
|
|
52
|
+
claim = urlsafe(JSON.generate(iss: document.fetch("client_email"), scope: SCOPE,
|
|
53
|
+
aud: document["token_uri"] || TOKEN_URI.to_s, iat: now, exp: now + 3600))
|
|
54
|
+
signature = OpenSSL::PKey::RSA.new(document.fetch("private_key")).sign(OpenSSL::Digest.new("SHA256"), "#{header}.#{claim}")
|
|
55
|
+
assertion = "#{header}.#{claim}.#{urlsafe(signature)}"
|
|
56
|
+
response = http_client.request(uri: URI(document["token_uri"] || TOKEN_URI.to_s), method: :post,
|
|
57
|
+
headers: {"content-type" => "application/x-www-form-urlencoded"},
|
|
58
|
+
body: URI.encode_www_form(grant_type: "urn:ietf:params:oauth:grant-type:jwt-bearer", assertion:),
|
|
59
|
+
cancellation_token:, deadline:)
|
|
60
|
+
token_json(response)
|
|
61
|
+
rescue Errno::ENOENT, JSON::ParserError, KeyError, OpenSSL::PKey::PKeyError => error
|
|
62
|
+
raise CredentialError, "Google service-account credentials are invalid: #{error.message}"
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
def metadata_token(cancellation_token:, deadline:)
|
|
66
|
+
response = http_client.request(
|
|
67
|
+
uri: URI("http://metadata.google.internal/computeMetadata/v1/instance/service-accounts/default/token"),
|
|
68
|
+
method: :get,
|
|
69
|
+
headers: {"Metadata-Flavor" => "Google"}, body: nil, cancellation_token:, deadline:,
|
|
70
|
+
allow_insecure_http: true
|
|
71
|
+
)
|
|
72
|
+
token_json(response)
|
|
73
|
+
rescue HTTPError
|
|
74
|
+
raise CredentialError, "No Vertex AI access token, service-account credentials, or metadata credentials were found"
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
def token_json(body)
|
|
78
|
+
value = JSON.parse(body)
|
|
79
|
+
[value.fetch("access_token"), value["expires_in"]]
|
|
80
|
+
rescue JSON::ParserError, KeyError => error
|
|
81
|
+
raise CredentialError, "Google token endpoint returned an invalid response: #{error.message}"
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def urlsafe(value) = Base64.urlsafe_encode64(value, padding: false)
|
|
85
|
+
|
|
86
|
+
def http_client = Support::HTTPClient.new(open_timeout: 2, read_timeout: 5, max_response_bytes: 1024 * 1024)
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "gemini"
|
|
4
|
+
require_relative "vertex_ai/credential_resolver"
|
|
5
|
+
|
|
6
|
+
module LittleGhost
|
|
7
|
+
module Providers
|
|
8
|
+
# Google Vertex AI variant of the Gemini wire protocol.
|
|
9
|
+
class VertexAI < Gemini
|
|
10
|
+
# Creates a Vertex AI client for a Google Cloud +project+ and +location+.
|
|
11
|
+
def initialize(model:, project:, location: "global", access_token: nil, credential_resolver: nil,
|
|
12
|
+
base_url: nil, **arguments)
|
|
13
|
+
@project = project.to_s
|
|
14
|
+
@location = location.to_s
|
|
15
|
+
@credential_resolver = credential_resolver || CredentialResolver.new(access_token:)
|
|
16
|
+
raise ConfigurationError, "Vertex AI project is required" if @project.empty?
|
|
17
|
+
|
|
18
|
+
base_url ||= "https://#{@location}-aiplatform.googleapis.com/v1/"
|
|
19
|
+
super(api_key: "unused", model:, base_url:, **arguments)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
protected
|
|
23
|
+
|
|
24
|
+
def endpoint
|
|
25
|
+
"projects/#{URI.encode_www_form_component(@project)}/locations/#{URI.encode_www_form_component(@location)}/publishers/google/models/#{URI.encode_www_form_component(model)}:streamGenerateContent?alt=sse"
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
def request_headers(request)
|
|
29
|
+
super.merge(
|
|
30
|
+
"authorization" => "Bearer #{@credential_resolver.call(
|
|
31
|
+
cancellation_token: request.cancellation_token,
|
|
32
|
+
deadline: request.deadline
|
|
33
|
+
)}"
|
|
34
|
+
)
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
end
|
data/lib/little_ghost/run.rb
CHANGED
|
@@ -12,10 +12,11 @@ module LittleGhost
|
|
|
12
12
|
# run.outcome # => "completed"
|
|
13
13
|
# run.response # => "Transfer 481 is waiting for the receiving bank."
|
|
14
14
|
#
|
|
15
|
-
#
|
|
16
|
-
# ask[rdoc-ref:LittleGhost::Agent#ask] consumes the event stream and
|
|
17
|
-
# the Run. For a live interface,
|
|
18
|
-
#
|
|
15
|
+
# The class-level ask helper[rdoc-ref:LittleGhost::Agent.ask] or standalone
|
|
16
|
+
# ask method[rdoc-ref:LittleGhost::Agent#ask] consumes the event stream and
|
|
17
|
+
# returns the Run. For a live interface, the class-level streaming
|
|
18
|
+
# helper[rdoc-ref:LittleGhost::Agent.stream_ask] or standalone streaming
|
|
19
|
+
# method[rdoc-ref:LittleGhost::Agent#stream_ask] yields StreamEvent objects
|
|
19
20
|
# and returns the same run after enumeration. A run can execute only once.
|
|
20
21
|
#
|
|
21
22
|
# Completion, failure, deadline, and cancellation become the +completed+,
|
|
@@ -119,7 +120,7 @@ module LittleGhost
|
|
|
119
120
|
|
|
120
121
|
@entrypoint
|
|
121
122
|
end
|
|
122
|
-
unless entrypoint.
|
|
123
|
+
unless entrypoint.is_a?(Agent)
|
|
123
124
|
raise AgentInterruptError, "Run entrypoint does not support interruptions"
|
|
124
125
|
end
|
|
125
126
|
|
|
@@ -417,7 +418,7 @@ module LittleGhost
|
|
|
417
418
|
def entrypoint_name
|
|
418
419
|
return entrypoint_class.name.to_s if workflow_run?
|
|
419
420
|
|
|
420
|
-
return @entrypoint.entrypoint_name if @entrypoint
|
|
421
|
+
return @entrypoint.entrypoint_name if @entrypoint.is_a?(Agent)
|
|
421
422
|
|
|
422
423
|
entrypoint_class.agent_id
|
|
423
424
|
end
|
|
@@ -563,7 +564,6 @@ module LittleGhost
|
|
|
563
564
|
|
|
564
565
|
def diagnostic_invocation_message
|
|
565
566
|
message = invocation.message
|
|
566
|
-
return message unless message.respond_to?(:text)
|
|
567
567
|
return message.text unless message.text.empty?
|
|
568
568
|
|
|
569
569
|
{
|