llm.rb 13.0.0 → 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +505 -14
- data/README.md +484 -50
- data/bin/llm.rb +148 -0
- data/data/anthropic.json +206 -263
- data/data/bedrock.json +2138 -1860
- data/data/deepinfra.json +1003 -624
- data/data/deepseek.json +38 -34
- data/data/google.json +1079 -371
- data/data/mistral.json +448 -368
- data/data/moonshot.json +384 -0
- data/data/openai.json +974 -1343
- data/data/xai.json +154 -126
- data/data/zai.json +191 -191
- data/lib/llm/agent.rb +123 -20
- data/lib/llm/context.rb +71 -88
- data/lib/llm/cost.rb +23 -17
- data/lib/llm/error.rb +0 -8
- data/lib/llm/function/array.rb +3 -3
- data/lib/llm/function/async/task.rb +2 -0
- data/lib/llm/function/fiber/task.rb +2 -0
- data/lib/llm/function/fork/task.rb +2 -0
- data/lib/llm/function/ractor/task.rb +2 -0
- data/lib/llm/function/sequential/group.rb +4 -1
- data/lib/llm/function/sequential/task.rb +1 -1
- data/lib/llm/function/task.rb +4 -0
- data/lib/llm/function/thread/task.rb +2 -0
- data/lib/llm/function.rb +33 -6
- data/lib/llm/guard/loop.rb +89 -0
- data/lib/llm/guard/null.rb +19 -0
- data/lib/llm/guard.rb +61 -0
- data/lib/llm/provider.rb +36 -0
- data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
- data/lib/llm/providers/anthropic.rb +2 -9
- data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
- data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
- data/lib/llm/providers/bedrock.rb +1 -8
- data/lib/llm/providers/google/stream_parser.rb +1 -0
- data/lib/llm/providers/google.rb +1 -8
- data/lib/llm/providers/mistral.rb +1 -1
- data/lib/llm/providers/moonshot.rb +76 -0
- data/lib/llm/providers/ollama.rb +2 -9
- data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
- data/lib/llm/providers/openai/responses.rb +7 -9
- data/lib/llm/providers/openai/stream_parser.rb +1 -0
- data/lib/llm/providers/openai.rb +4 -11
- data/lib/llm/repl/bar.rb +4 -3
- data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
- data/lib/llm/repl/color.rb +78 -0
- data/lib/llm/repl/command.rb +12 -5
- data/lib/llm/repl/commands/compact.rb +2 -2
- data/lib/llm/repl/commands/help.rb +3 -5
- data/lib/llm/repl/input/char.rb +46 -0
- data/lib/llm/repl/input/row.rb +39 -0
- data/lib/llm/repl/input.rb +251 -66
- data/lib/llm/repl/markdown/table.rb +11 -3
- data/lib/llm/repl/markdown.rb +34 -8
- data/lib/llm/repl/node.rb +37 -0
- data/lib/llm/repl/status.rb +42 -7
- data/lib/llm/repl/stream.rb +18 -6
- data/lib/llm/repl/walker.rb +3 -2
- data/lib/llm/repl/window.rb +54 -35
- data/lib/llm/repl.rb +74 -32
- data/lib/llm/skill.rb +20 -4
- data/lib/llm/stream.rb +8 -7
- data/lib/llm/tool.rb +29 -0
- data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
- data/lib/llm/tools/git.rb +3 -0
- data/lib/llm/tools/mkdir.rb +3 -0
- data/lib/llm/tools/rg.rb +3 -0
- data/lib/llm/tools/ruby.rb +46 -0
- data/lib/llm/tools/shell.rb +3 -0
- data/lib/llm/tracer/pretty_logger.rb +127 -0
- data/lib/llm/tracer.rb +1 -0
- data/lib/llm/transformer/null.rb +21 -0
- data/lib/llm/transformer.rb +55 -0
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +12 -2
- data/llm.gemspec +9 -2
- data/resources/deepdive/advanced/cancellation.md +74 -0
- data/resources/deepdive/advanced/compaction.md +83 -0
- data/resources/deepdive/advanced/context.md +267 -0
- data/resources/deepdive/advanced/guard.md +371 -0
- data/resources/deepdive/advanced/tracer.md +180 -0
- data/resources/deepdive/advanced/transformer.md +67 -0
- data/resources/deepdive/advanced/transports.md +45 -0
- data/resources/deepdive/everything_else/audio.md +122 -0
- data/resources/deepdive/everything_else/cost.md +99 -0
- data/resources/deepdive/everything_else/images.md +89 -0
- data/resources/deepdive/everything_else/object.md +108 -0
- data/resources/deepdive/everything_else/ocr.md +48 -0
- data/resources/deepdive/fundamentals/agents.md +202 -0
- data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
- data/resources/deepdive/fundamentals/concurrency.md +104 -0
- data/resources/deepdive/fundamentals/database.md +449 -0
- data/resources/deepdive/fundamentals/embeddings.md +157 -0
- data/resources/deepdive/fundamentals/repl.md +87 -0
- data/resources/deepdive/fundamentals/schema.md +61 -0
- data/resources/deepdive/fundamentals/skills.md +106 -0
- data/resources/deepdive/fundamentals/stream.md +110 -0
- data/resources/deepdive/fundamentals/tools.md +265 -0
- data/resources/deepdive/protocols/a2a.md +106 -0
- data/resources/deepdive/protocols/mcp.md +111 -0
- data/resources/deepdive.md +58 -1792
- metadata +51 -7
- data/lib/llm/loop_guard.rb +0 -107
data/lib/llm/agent.rb
CHANGED
|
@@ -11,20 +11,22 @@ module LLM
|
|
|
11
11
|
# {LLM::Context LLM::Context}: message history, usage, persistence,
|
|
12
12
|
# streaming parameters, and provider-backed requests still flow through
|
|
13
13
|
# an underlying context. The defining behavior of an agent is that it
|
|
14
|
-
# automatically resolves pending tool calls for you during `talk
|
|
15
|
-
#
|
|
14
|
+
# automatically resolves pending tool calls for you during `talk`,
|
|
15
|
+
# instead of leaving tool loops to the caller.
|
|
16
16
|
#
|
|
17
17
|
# **Notes:**
|
|
18
18
|
# * Instructions are injected once unless a system message is already present.
|
|
19
19
|
# * An agent automatically executes tool loops (unlike {LLM::Context LLM::Context}).
|
|
20
20
|
# * The automatic tool loop enables the wrapped context's `guard` by default.
|
|
21
|
-
# The built-in {LLM::
|
|
22
|
-
# patterns and blocks stuck execution before more tool work is
|
|
23
|
-
#
|
|
24
|
-
#
|
|
25
|
-
#
|
|
21
|
+
# The built-in {LLM::Guard::Loop LLM::Guard::Loop} detects repeated
|
|
22
|
+
# tool-call patterns and blocks stuck execution before more tool work is
|
|
23
|
+
# queued.
|
|
24
|
+
# * The tool loop can be bounded with `tool_budget`. Once the budget is
|
|
25
|
+
# spent, the agent sends an in-band advisory message back through the
|
|
26
|
+
# model and keeps the loop in-band. By default no budget is set
|
|
27
|
+
# (`nil`), so the feature is disabled.
|
|
26
28
|
# * Tool loop execution can be configured with `concurrency :sequential`,
|
|
27
|
-
# `:thread`, `:async`, `:fiber`, or `:ractor`.
|
|
29
|
+
# `:thread`, `:async`, `:fiber`, `:fork`, or `:ractor`.
|
|
28
30
|
#
|
|
29
31
|
# @example Subclass with defaults
|
|
30
32
|
# class SystemAdmin < LLM::Agent
|
|
@@ -52,6 +54,16 @@ module LLM
|
|
|
52
54
|
UNDEFINED = Object.new
|
|
53
55
|
private_constant :UNDEFINED
|
|
54
56
|
|
|
57
|
+
##
|
|
58
|
+
# @api private
|
|
59
|
+
CASE_PATTERN = /(?<=[a-z])(?=[A-Z])|(?<=[A-Z])(?=[A-Z][a-z])/
|
|
60
|
+
private_constant :CASE_PATTERN
|
|
61
|
+
|
|
62
|
+
##
|
|
63
|
+
# @api private
|
|
64
|
+
File = ::File
|
|
65
|
+
private_constant :File
|
|
66
|
+
|
|
55
67
|
##
|
|
56
68
|
# Returns a provider
|
|
57
69
|
# @return [LLM::Provider]
|
|
@@ -96,18 +108,44 @@ module LLM
|
|
|
96
108
|
|
|
97
109
|
##
|
|
98
110
|
# Set or get an agent's name
|
|
111
|
+
# @note
|
|
112
|
+
# This method serves as a self-documenting string
|
|
113
|
+
# and it is used by {LLM::Repl LLM::Repl}. It is
|
|
114
|
+
# optional but recommended.
|
|
99
115
|
# @param [String] name
|
|
100
116
|
# The agent name
|
|
101
117
|
# @return [String]
|
|
102
118
|
# Return's the agents name
|
|
103
119
|
def self.name(name = UNDEFINED, &block)
|
|
104
120
|
if name.equal?(UNDEFINED)
|
|
105
|
-
@name
|
|
121
|
+
if @name.nil?
|
|
122
|
+
name = to_s.split("::").last
|
|
123
|
+
@name = name.gsub(CASE_PATTERN, "-").downcase
|
|
124
|
+
else
|
|
125
|
+
@name
|
|
126
|
+
end
|
|
106
127
|
else
|
|
107
128
|
@name = block || name
|
|
108
129
|
end
|
|
109
130
|
end
|
|
110
131
|
|
|
132
|
+
##
|
|
133
|
+
# Set or get an agent's description
|
|
134
|
+
# @note
|
|
135
|
+
# This method serves as a self-documenting string.
|
|
136
|
+
# It is optional but recommended.
|
|
137
|
+
# @param [String] desc
|
|
138
|
+
# The agent's description
|
|
139
|
+
# @return [String, nil]
|
|
140
|
+
# Returns the agent's description
|
|
141
|
+
def self.description(desc = UNDEFINED, &block)
|
|
142
|
+
if desc.equal?(UNDEFINED)
|
|
143
|
+
@desc
|
|
144
|
+
else
|
|
145
|
+
@desc = block || desc
|
|
146
|
+
end
|
|
147
|
+
end
|
|
148
|
+
|
|
111
149
|
##
|
|
112
150
|
# Set or get the default model
|
|
113
151
|
# @param [String, nil] model
|
|
@@ -261,6 +299,42 @@ module LLM
|
|
|
261
299
|
end
|
|
262
300
|
end
|
|
263
301
|
|
|
302
|
+
##
|
|
303
|
+
# Set the file path where an agent's memory
|
|
304
|
+
# can be restored from, and written to.
|
|
305
|
+
# @param [String] path
|
|
306
|
+
# The path to a file
|
|
307
|
+
# @return [String, nil]
|
|
308
|
+
def self.path(path = UNDEFINED, &block)
|
|
309
|
+
if path.equal?(UNDEFINED)
|
|
310
|
+
@path
|
|
311
|
+
else
|
|
312
|
+
@path = path || block
|
|
313
|
+
end
|
|
314
|
+
end
|
|
315
|
+
|
|
316
|
+
##
|
|
317
|
+
# Set or get the maximum number of tool calls
|
|
318
|
+
# that are allowed in a single turn. Once the
|
|
319
|
+
# budget is spent, we will return an in-band
|
|
320
|
+
# message that informs the model it has spent
|
|
321
|
+
# its tool call budget - and usually a model
|
|
322
|
+
# will change course afterwards.
|
|
323
|
+
# @note
|
|
324
|
+
# By default this feature is disabled
|
|
325
|
+
# (set to `nil`).
|
|
326
|
+
# @param [Integer] budget
|
|
327
|
+
# The maximum number of tool calls to allow in
|
|
328
|
+
# a single turn.
|
|
329
|
+
# @return [Integer, nil]
|
|
330
|
+
def self.tool_budget(budget = UNDEFINED, &block)
|
|
331
|
+
if budget.equal?(UNDEFINED)
|
|
332
|
+
@tool_budget
|
|
333
|
+
else
|
|
334
|
+
@tool_budget = budget || block
|
|
335
|
+
end
|
|
336
|
+
end
|
|
337
|
+
|
|
264
338
|
##
|
|
265
339
|
# @param [LLM::Provider] llm
|
|
266
340
|
# A provider
|
|
@@ -276,9 +350,10 @@ module LLM
|
|
|
276
350
|
# @option params [LLM::Tracer, Proc, nil] :tracer Optional tracer override for this agent instance
|
|
277
351
|
# @option params [Symbol, Array<Symbol>, nil] :concurrency Defaults to the agent class concurrency
|
|
278
352
|
def initialize(llm, params = {})
|
|
353
|
+
params = {}.merge!(params)
|
|
279
354
|
@llm = llm
|
|
280
|
-
fields = %i[name model skills schema tracer stream tools concurrency instructions confirm]
|
|
281
|
-
fields_ivar = %i[name tracer concurrency instructions confirm]
|
|
355
|
+
fields = %i[name description path tool_budget model skills schema tracer stream tools concurrency instructions confirm]
|
|
356
|
+
fields_ivar = %i[name description path tool_budget tracer concurrency instructions confirm]
|
|
282
357
|
fields.each do |field|
|
|
283
358
|
resolvable = params.key?(field) ? params.delete(field) : self.class.public_send(field)
|
|
284
359
|
resolve_symbol = !%i[concurrency].include?(field)
|
|
@@ -292,7 +367,8 @@ module LLM
|
|
|
292
367
|
instance_variable_set(:"@#{field}", resolved)
|
|
293
368
|
end
|
|
294
369
|
end
|
|
295
|
-
@ctx = LLM::Context.new(llm, {guard:
|
|
370
|
+
@ctx = LLM::Context.new(llm, {guard: LLM::Guard::Loop}.merge(params))
|
|
371
|
+
@path and File.readable?(@path) ? @ctx.restore(path:) : nil
|
|
296
372
|
end
|
|
297
373
|
|
|
298
374
|
##
|
|
@@ -302,16 +378,32 @@ module LLM
|
|
|
302
378
|
@name
|
|
303
379
|
end
|
|
304
380
|
|
|
381
|
+
##
|
|
382
|
+
# Returns a file path where an agent's memory is
|
|
383
|
+
# restored from, and written to after each turn.
|
|
384
|
+
# @return [String, nil]
|
|
385
|
+
def path
|
|
386
|
+
@path
|
|
387
|
+
end
|
|
388
|
+
|
|
389
|
+
##
|
|
390
|
+
# Returns the agent's description
|
|
391
|
+
# @return [String, nil]
|
|
392
|
+
def description
|
|
393
|
+
@description
|
|
394
|
+
end
|
|
395
|
+
|
|
305
396
|
##
|
|
306
397
|
# Maintain a conversation via the chat completions API.
|
|
307
398
|
# This method immediately sends a request to the LLM and returns the response.
|
|
308
399
|
#
|
|
309
400
|
# @param prompt (see LLM::Provider#complete)
|
|
310
401
|
# @param [Hash] params The params passed to the provider, including optional :stream, :tools, :schema etc.
|
|
311
|
-
# @option params [Integer] :
|
|
312
|
-
# The
|
|
313
|
-
# in-band advisory
|
|
314
|
-
#
|
|
402
|
+
# @option params [Integer] :tool_budget
|
|
403
|
+
# The maximum number of tool calls that can be made in a single turn
|
|
404
|
+
# before the agent sends an in-band advisory message that tells the model
|
|
405
|
+
# it has spent its tool call budget - and usually the model will change
|
|
406
|
+
# course after that. By default this feature is disabled (set to `nil`).
|
|
315
407
|
# @return [LLM::Response] Returns the LLM's response for this turn.
|
|
316
408
|
# @example
|
|
317
409
|
# llm = LLM.openai(key: ENV["KEY"])
|
|
@@ -319,13 +411,17 @@ module LLM
|
|
|
319
411
|
# response = agent.talk("Hello, what is your name?")
|
|
320
412
|
# puts response.choices[0].content
|
|
321
413
|
def talk(prompt, params = {})
|
|
322
|
-
run_loop(prompt, params, :talk)
|
|
414
|
+
res = run_loop(prompt, params, :talk)
|
|
415
|
+
path ? @ctx.save(path:) : nil
|
|
416
|
+
res
|
|
323
417
|
end
|
|
324
418
|
|
|
325
419
|
##
|
|
326
420
|
# @see LLM::Context#ask
|
|
327
421
|
def ask(prompt, params = {})
|
|
328
|
-
run_loop(prompt, params, :ask)
|
|
422
|
+
res = run_loop(prompt, params, :ask)
|
|
423
|
+
path ? @ctx.save(path:) : nil
|
|
424
|
+
res
|
|
329
425
|
end
|
|
330
426
|
|
|
331
427
|
##
|
|
@@ -461,6 +557,13 @@ module LLM
|
|
|
461
557
|
@ctx.context_window
|
|
462
558
|
end
|
|
463
559
|
|
|
560
|
+
##
|
|
561
|
+
# @see LLM::Context#compacted?
|
|
562
|
+
# @return [Boolean]
|
|
563
|
+
def compacted?
|
|
564
|
+
@ctx.compacted?
|
|
565
|
+
end
|
|
566
|
+
|
|
464
567
|
##
|
|
465
568
|
# Start a minimalist repl that can interact
|
|
466
569
|
# with the agent and its current state. This
|
|
@@ -611,7 +714,7 @@ module LLM
|
|
|
611
714
|
def run_loop(prompt, params, target)
|
|
612
715
|
run = proc do
|
|
613
716
|
talk = @ctx.method(target)
|
|
614
|
-
max = params.key?(:
|
|
717
|
+
max = params.key?(:tool_budget) ? params.delete(:tool_budget) : @tool_budget
|
|
615
718
|
max = Integer(max) if max
|
|
616
719
|
stream = params[:stream] || @ctx.params[:stream]
|
|
617
720
|
params[:stream] = LLM::Stream.try(stream, extra: {concurrency:})
|
|
@@ -622,7 +725,7 @@ module LLM
|
|
|
622
725
|
break unless @ctx.pending_functions?
|
|
623
726
|
res = talk.call(call_functions, params)
|
|
624
727
|
end
|
|
625
|
-
res = talk.call(@ctx.pending_functions.map(&:
|
|
728
|
+
res = talk.call(@ctx.pending_functions.map(&:budget_spent), params) if @ctx.pending_functions?
|
|
626
729
|
else
|
|
627
730
|
res = talk.call(call_functions, params)
|
|
628
731
|
end
|
data/lib/llm/context.rb
CHANGED
|
@@ -83,13 +83,22 @@ module LLM
|
|
|
83
83
|
# {LLM::Compactor::Null}.
|
|
84
84
|
# @option params [Hash] :compactor_options
|
|
85
85
|
# Options passed to the compactor's `call` method. Defaults to `{}`.
|
|
86
|
+
# @option params [Class<LLM::Transformer>, nil] :transformer
|
|
87
|
+
# A transformer class to use for message transformation. Defaults to
|
|
88
|
+
# {LLM::Transformer::Null}.
|
|
89
|
+
# @option params [Hash] :transformer_options
|
|
90
|
+
# Options passed to the transformer's `call` method. Defaults to `{}`.
|
|
91
|
+
# @option params [Class<LLM::Guard>, nil] :guard
|
|
92
|
+
# A guard class to supervise agentic tool execution. Defaults to
|
|
93
|
+
# {LLM::Guard::Null}.
|
|
94
|
+
# @option params [Hash] :guard_options
|
|
95
|
+
# Options passed to the guard's `call` method. Defaults to `{}`.
|
|
86
96
|
# @option params [Array<LLM::Function>, nil] :tools Defaults to nil
|
|
87
97
|
# @option params [Array<String>, nil] :skills Defaults to nil
|
|
88
98
|
def initialize(llm, params = {})
|
|
99
|
+
params = {}.merge!(params)
|
|
89
100
|
@llm = llm
|
|
90
101
|
@mode = params.delete(:mode) || (llm.name == :openai ? :responses : :completions)
|
|
91
|
-
@guard = params.delete(:guard)
|
|
92
|
-
@transformer = params.delete(:transformer)
|
|
93
102
|
tools = [*params.delete(:tools), *load_skills(params.delete(:skills))]
|
|
94
103
|
@params = {model: llm.default_model, schema: nil}.compact.merge!(params)
|
|
95
104
|
@params[:tools] = tools unless tools.empty?
|
|
@@ -101,6 +110,14 @@ module LLM
|
|
|
101
110
|
klass: params.delete(:compactor) || LLM::Compactor::Null,
|
|
102
111
|
options: params.delete(:compactor_options) || {}
|
|
103
112
|
}
|
|
113
|
+
@transformer = {
|
|
114
|
+
klass: params.delete(:transformer) || LLM::Transformer::Null,
|
|
115
|
+
options: params.delete(:transformer_options) || {}
|
|
116
|
+
}
|
|
117
|
+
@guard = {
|
|
118
|
+
klass: params.delete(:guard) || LLM::Guard::Null,
|
|
119
|
+
options: params.delete(:guard_options) || {}
|
|
120
|
+
}
|
|
104
121
|
end
|
|
105
122
|
|
|
106
123
|
##
|
|
@@ -126,58 +143,35 @@ module LLM
|
|
|
126
143
|
alias_method :compacted?, :compacted
|
|
127
144
|
|
|
128
145
|
##
|
|
129
|
-
# Returns
|
|
146
|
+
# Returns the configured guard class.
|
|
130
147
|
#
|
|
131
148
|
# Guards are context-level supervisors for agentic execution. A guard can
|
|
132
149
|
# inspect the runtime state and decide whether pending tool work should be
|
|
133
150
|
# blocked before the context keeps looping.
|
|
134
151
|
#
|
|
135
|
-
# The
|
|
152
|
+
# The guard is stamped onto the functions the context binds, so it runs
|
|
153
|
+
# whenever a task is spawned — including tool calls queued from a stream
|
|
154
|
+
# via {LLM::Stream#on_tool_call}. A blocked call yields its in-band
|
|
155
|
+
# `guard_error` return without executing.
|
|
156
|
+
#
|
|
157
|
+
# The built-in implementation is {LLM::Guard::Loop LLM::Guard::Loop}, which
|
|
136
158
|
# detects repeated tool-call patterns and turns them into in-band
|
|
137
|
-
#
|
|
159
|
+
# `guard_error` tool returns.
|
|
138
160
|
#
|
|
139
|
-
# @return [
|
|
161
|
+
# @return [Class<LLM::Guard>]
|
|
140
162
|
def guard
|
|
141
|
-
|
|
142
|
-
@guard = LLM::LoopGuard.new if @guard == true
|
|
143
|
-
@guard = LLM::LoopGuard.new(@guard) if Hash === @guard
|
|
144
|
-
@guard
|
|
145
|
-
end
|
|
146
|
-
|
|
147
|
-
##
|
|
148
|
-
# Sets a guard or guard config.
|
|
149
|
-
#
|
|
150
|
-
# Guards must implement `call(ctx)` and return either `nil` or a warning
|
|
151
|
-
# string. Returning a warning tells the context to block pending tool work
|
|
152
|
-
# with guarded tool errors instead of continuing the loop.
|
|
153
|
-
#
|
|
154
|
-
# @param [#call, Hash, Boolean, nil] guard
|
|
155
|
-
# @return [#call, Hash, Boolean, nil]
|
|
156
|
-
def guard=(guard)
|
|
157
|
-
@guard = guard
|
|
163
|
+
@guard[:klass]
|
|
158
164
|
end
|
|
159
165
|
|
|
160
166
|
##
|
|
161
|
-
# Returns
|
|
167
|
+
# Returns the configured transformer class.
|
|
162
168
|
#
|
|
163
|
-
# Transformers
|
|
164
|
-
#
|
|
169
|
+
# Transformers rewrite the most recent message before it is sent to the
|
|
170
|
+
# provider.
|
|
165
171
|
#
|
|
166
|
-
# @return [
|
|
172
|
+
# @return [Class<LLM::Transformer>]
|
|
167
173
|
def transformer
|
|
168
|
-
@transformer
|
|
169
|
-
end
|
|
170
|
-
|
|
171
|
-
##
|
|
172
|
-
# Sets a transformer.
|
|
173
|
-
#
|
|
174
|
-
# Transformers must implement `call(ctx, prompt, params)` and return a
|
|
175
|
-
# two-element array of `[prompt, params]`.
|
|
176
|
-
#
|
|
177
|
-
# @param [#call, nil] transformer
|
|
178
|
-
# @return [#call, nil]
|
|
179
|
-
def transformer=(transformer)
|
|
180
|
-
@transformer = transformer
|
|
174
|
+
@transformer[:klass]
|
|
181
175
|
end
|
|
182
176
|
|
|
183
177
|
# Interact with the context via the chat completions API.
|
|
@@ -197,10 +191,12 @@ module LLM
|
|
|
197
191
|
repair!(@messages, prompt)
|
|
198
192
|
prompt, params, res = mode == :responses ? respond(prompt, params) : complete(prompt, params)
|
|
199
193
|
self.compacted = false
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
194
|
+
if prompt.all?(&:tool_return?)
|
|
195
|
+
@messages.concat prompt.map { LLM::Message.new(@llm.tool_role, _1.content, _1.extra) }
|
|
196
|
+
else
|
|
197
|
+
@messages.concat(prompt)
|
|
198
|
+
end
|
|
199
|
+
@messages.concat([res.choices[-1]].compact)
|
|
204
200
|
res
|
|
205
201
|
ensure
|
|
206
202
|
@owner = nil
|
|
@@ -245,6 +241,7 @@ module LLM
|
|
|
245
241
|
# @return [Array<LLM::Function>]
|
|
246
242
|
def pending_functions
|
|
247
243
|
return_ids = returns.map(&:id)
|
|
244
|
+
guard = @guard[:klass].new(self)
|
|
248
245
|
@messages
|
|
249
246
|
.select(&:assistant?)
|
|
250
247
|
.flat_map do |msg|
|
|
@@ -252,9 +249,11 @@ module LLM
|
|
|
252
249
|
fns.each do |fn|
|
|
253
250
|
fn.tracer = tracer
|
|
254
251
|
fn.model = msg.model
|
|
252
|
+
fn.guard = guard
|
|
255
253
|
end
|
|
256
254
|
end.extend(LLM::Function::Array)
|
|
257
255
|
end
|
|
256
|
+
|
|
258
257
|
##
|
|
259
258
|
# Returns whether there is pending tool work in this context.
|
|
260
259
|
# This prefers queued streamed tool work when present, and otherwise
|
|
@@ -268,15 +267,10 @@ module LLM
|
|
|
268
267
|
##
|
|
269
268
|
# Spawns a function through the context.
|
|
270
269
|
#
|
|
271
|
-
# When a guard is configured, this method can return an in-band guarded
|
|
272
|
-
# tool error instead of spawning work.
|
|
273
|
-
#
|
|
274
270
|
# @param [LLM::Function] function
|
|
275
271
|
# @param [Symbol] strategy
|
|
276
|
-
# @return [LLM::Function::
|
|
272
|
+
# @return [LLM::Function::Task]
|
|
277
273
|
def spawn(function, strategy)
|
|
278
|
-
warning = guard&.call(self)
|
|
279
|
-
return guarded_return_for(function, warning) if warning
|
|
280
274
|
function.task(strategy)
|
|
281
275
|
end
|
|
282
276
|
|
|
@@ -310,9 +304,11 @@ module LLM
|
|
|
310
304
|
# @return [Array<LLM::Function::Return>]
|
|
311
305
|
def wait(strategy, except: [])
|
|
312
306
|
if stream.queue.empty?
|
|
313
|
-
|
|
314
|
-
|
|
315
|
-
|
|
307
|
+
##
|
|
308
|
+
# Every pending function is spawned as a task that checks its own
|
|
309
|
+
# guard (stamped on the function) before running. Blocked tasks
|
|
310
|
+
# yield their guard's return, so all pending calls still close.
|
|
311
|
+
tools = except.empty? ? pending_functions : pending_functions - except
|
|
316
312
|
@queue = tools.task(strategy)
|
|
317
313
|
returns = @queue.wait
|
|
318
314
|
emit_tool_returns(tools, returns)
|
|
@@ -513,65 +509,52 @@ module LLM
|
|
|
513
509
|
[*skills].map { LLM::Skill.load(_1).to_tool(self) }
|
|
514
510
|
end
|
|
515
511
|
|
|
516
|
-
##
|
|
517
|
-
# Builds in-band guarded returns when the guard blocks tool work.
|
|
518
|
-
# @api private
|
|
519
|
-
def guarded_returns(tools:)
|
|
520
|
-
warning = guard&.call(self)
|
|
521
|
-
return unless warning
|
|
522
|
-
tools.map { guarded_return_for(_1, warning) }
|
|
523
|
-
end
|
|
524
|
-
|
|
525
512
|
##
|
|
526
513
|
# Rewrites a prompt and params through the configured transformer.
|
|
527
514
|
# @api private
|
|
528
|
-
def transform(prompt, params)
|
|
529
|
-
transformer = self
|
|
530
|
-
return [prompt, params] unless transformer
|
|
515
|
+
def transform(prompt, params, key: :messages)
|
|
516
|
+
transformer = @transformer[:klass].new(self)
|
|
531
517
|
stream = params[:stream]
|
|
532
|
-
stream.on_transform(
|
|
533
|
-
|
|
518
|
+
stream.on_transform(transformer)
|
|
519
|
+
role = params[:role] || @llm.user_role
|
|
520
|
+
messages = @llm.build_messages(prompt, params, role, key:)
|
|
521
|
+
messages[-1] = transformer.call(message: messages[-1], **@transformer[:options])
|
|
522
|
+
messages
|
|
534
523
|
ensure
|
|
535
|
-
stream.on_transform_finish(
|
|
524
|
+
stream.on_transform_finish(transformer)
|
|
536
525
|
end
|
|
537
526
|
|
|
538
527
|
##
|
|
539
528
|
# Executes a turn through the Responses API.
|
|
540
529
|
# @api private
|
|
541
530
|
def respond(prompt, params)
|
|
531
|
+
history = @messages.to_a
|
|
542
532
|
params = @params.merge(params)
|
|
543
|
-
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:)
|
|
533
|
+
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:, guard: @guard[:klass].new(self))
|
|
544
534
|
params[:stream] = LLM::Stream.try(params[:stream], extra:)
|
|
545
|
-
prompt, params = transform(prompt, params)
|
|
546
|
-
@stream = params[:stream]
|
|
547
535
|
res_id = params[:store] == false ? nil : @messages.find(&:assistant?)&.response&.response_id
|
|
548
|
-
input = res_id ? [] :
|
|
536
|
+
input = res_id ? [] : history
|
|
537
|
+
params[:input] = input
|
|
538
|
+
messages = transform(prompt, params, key: :input)
|
|
539
|
+
@stream = params[:stream]
|
|
540
|
+
new_messages = messages[input.size..]
|
|
549
541
|
params = params.merge(previous_response_id: res_id, input:).compact
|
|
550
|
-
[
|
|
542
|
+
[new_messages, params, @llm.responses.create(messages, params)]
|
|
551
543
|
end
|
|
552
544
|
|
|
553
545
|
##
|
|
554
546
|
# Executes a turn through the chat completions API.
|
|
555
547
|
# @api private
|
|
556
548
|
def complete(prompt, params)
|
|
557
|
-
|
|
549
|
+
history = @messages.to_a
|
|
550
|
+
params = params.merge(messages: history)
|
|
558
551
|
params = @params.merge(params)
|
|
559
|
-
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:)
|
|
552
|
+
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:, guard: @guard[:klass].new(self))
|
|
560
553
|
params[:stream] = LLM::Stream.try(params[:stream], extra:)
|
|
561
|
-
|
|
554
|
+
messages = transform(prompt, params)
|
|
562
555
|
@stream = params[:stream]
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
##
|
|
567
|
-
# Builds one guarded tool return for a blocked function call.
|
|
568
|
-
# @api private
|
|
569
|
-
def guarded_return_for(function, warning)
|
|
570
|
-
LLM::Function::Return.new(function.id, function.name, {
|
|
571
|
-
error: true,
|
|
572
|
-
type: LLM::GuardError.name,
|
|
573
|
-
message: warning
|
|
574
|
-
})
|
|
556
|
+
new_messages = messages[history.size..]
|
|
557
|
+
[new_messages, params, @llm.complete(messages, params)]
|
|
575
558
|
end
|
|
576
559
|
|
|
577
560
|
##
|
data/lib/llm/cost.rb
CHANGED
|
@@ -7,27 +7,27 @@
|
|
|
7
7
|
# and reasoning costs separately and can return the total.
|
|
8
8
|
#
|
|
9
9
|
# @attr [Float] input_costs
|
|
10
|
-
# Returns the input cost
|
|
10
|
+
# Returns the input cost, aliased as `input`
|
|
11
11
|
# @attr [Float] output_costs
|
|
12
|
-
# Returns the output cost
|
|
12
|
+
# Returns the output cost, aliased as `output`
|
|
13
13
|
# @attr [Float, nil] input_audio_costs
|
|
14
14
|
# Returns the input audio cost, or nil when no input audio tokens
|
|
15
|
-
# were used
|
|
15
|
+
# were used, aliased as `input_audio`
|
|
16
16
|
# @attr [Float, nil] output_audio_costs
|
|
17
17
|
# Returns the output audio cost, or nil when no output audio tokens
|
|
18
|
-
# were used
|
|
18
|
+
# were used, aliased as `output_audio`
|
|
19
19
|
# @attr [Float, nil] input_image_costs
|
|
20
20
|
# Returns the input image cost, or nil when no input image tokens
|
|
21
|
-
# were used
|
|
21
|
+
# were used, aliased as `input_image`
|
|
22
22
|
# @attr [Float, nil] cache_read_costs
|
|
23
23
|
# Returns the cache read cost, or nil when no cache tokens
|
|
24
|
-
# were used
|
|
24
|
+
# were used, aliased as `cache_read`
|
|
25
25
|
# @attr [Float, nil] cache_write_costs
|
|
26
26
|
# Returns the cache write cost, or nil when no cache creation
|
|
27
|
-
# tokens were used
|
|
27
|
+
# tokens were used, aliased as `cache_write`
|
|
28
28
|
# @attr [Float, nil] reasoning_costs
|
|
29
29
|
# Returns the reasoning cost, or nil when no reasoning tokens
|
|
30
|
-
# were used
|
|
30
|
+
# were used, aliased as `reasoning`
|
|
31
31
|
class LLM::Cost < Struct.new(
|
|
32
32
|
:input_costs, :output_costs,
|
|
33
33
|
:input_audio_costs, :output_audio_costs,
|
|
@@ -82,15 +82,10 @@ class LLM::Cost < Struct.new(
|
|
|
82
82
|
# Returns a hash with the non-nil cost components and the total
|
|
83
83
|
def to_h
|
|
84
84
|
{
|
|
85
|
-
input
|
|
86
|
-
|
|
87
|
-
input_audio
|
|
88
|
-
|
|
89
|
-
input_image: input_image_costs,
|
|
90
|
-
cache_read: cache_read_costs,
|
|
91
|
-
cache_write: cache_write_costs,
|
|
92
|
-
reasoning: reasoning_costs,
|
|
93
|
-
total: total
|
|
85
|
+
input:, output:,
|
|
86
|
+
cache_read:, cache_write:,
|
|
87
|
+
input_audio:, output_audio:, input_image:,
|
|
88
|
+
reasoning:, total:
|
|
94
89
|
}.compact
|
|
95
90
|
end
|
|
96
91
|
|
|
@@ -100,4 +95,15 @@ class LLM::Cost < Struct.new(
|
|
|
100
95
|
def to_s
|
|
101
96
|
format("%.12f", total).sub(/\.?0+$/, "")
|
|
102
97
|
end
|
|
98
|
+
|
|
99
|
+
##
|
|
100
|
+
# Aliases
|
|
101
|
+
alias_method :input, :input_costs
|
|
102
|
+
alias_method :output, :output_costs
|
|
103
|
+
alias_method :input_audio, :input_audio_costs
|
|
104
|
+
alias_method :output_audio, :output_audio_costs
|
|
105
|
+
alias_method :cache_read, :cache_read_costs
|
|
106
|
+
alias_method :cache_write, :cache_write_costs
|
|
107
|
+
alias_method :input_image, :input_image_costs
|
|
108
|
+
alias_method :reasoning, :reasoning_costs
|
|
103
109
|
end
|
data/lib/llm/error.rb
CHANGED
|
@@ -55,14 +55,6 @@ module LLM
|
|
|
55
55
|
# When the context window is exceeded
|
|
56
56
|
ContextWindowError = Class.new(InvalidRequestError)
|
|
57
57
|
|
|
58
|
-
##
|
|
59
|
-
# When stuck in a tool call loop
|
|
60
|
-
ToolLoopError = Class.new(Error)
|
|
61
|
-
|
|
62
|
-
##
|
|
63
|
-
# When a guard blocks pending tool execution
|
|
64
|
-
GuardError = Class.new(Error)
|
|
65
|
-
|
|
66
58
|
##
|
|
67
59
|
# When a request is interrupted
|
|
68
60
|
Interrupt = Class.new(Error)
|
data/lib/llm/function/array.rb
CHANGED
|
@@ -3,7 +3,7 @@
|
|
|
3
3
|
class LLM::Function
|
|
4
4
|
##
|
|
5
5
|
# The {LLM::Function::Array} module extends the array
|
|
6
|
-
# returned by {LLM::Context#
|
|
6
|
+
# returned by {LLM::Context#pending_functions} with methods
|
|
7
7
|
# that can call all pending functions sequentially or
|
|
8
8
|
# concurrently. The return values can be reported back
|
|
9
9
|
# to the LLM on the next turn.
|
|
@@ -57,9 +57,9 @@ class LLM::Function
|
|
|
57
57
|
#
|
|
58
58
|
# @param [Symbol] strategy
|
|
59
59
|
# Controls concurrency strategy:
|
|
60
|
-
# - `:
|
|
60
|
+
# - `:sequential`: Call functions sequentially without spawning
|
|
61
61
|
# - `:thread`: Use threads
|
|
62
|
-
# - `:
|
|
62
|
+
# - `:async`: Use async tasks (requires async gem)
|
|
63
63
|
# - `:fiber`: Use scheduler-backed fibers (requires Fiber.scheduler)
|
|
64
64
|
# - `:fork`: Use forked child processes
|
|
65
65
|
# - `:ractor`: Use Ruby ractors (class-based tools only; MCP tools are not supported)
|
|
@@ -32,6 +32,7 @@ module LLM::Function::Async
|
|
|
32
32
|
# pushed to a queue that {#wait} consumes.
|
|
33
33
|
# @return [nil]
|
|
34
34
|
def spawn
|
|
35
|
+
return if @guarded
|
|
35
36
|
@queue = Queue.new
|
|
36
37
|
@alive = true
|
|
37
38
|
@reactor.submit do
|
|
@@ -66,6 +67,7 @@ module LLM::Function::Async
|
|
|
66
67
|
# Wait for the result queue to contain a value.
|
|
67
68
|
# @return [LLM::Function::Return]
|
|
68
69
|
def wait
|
|
70
|
+
return @guarded if @guarded
|
|
69
71
|
spawn unless @queue
|
|
70
72
|
result = @queue.pop
|
|
71
73
|
@alive = false
|
|
@@ -22,6 +22,7 @@ module LLM::Function::Fiber
|
|
|
22
22
|
##
|
|
23
23
|
# @return [nil]
|
|
24
24
|
def spawn
|
|
25
|
+
return if @guarded
|
|
25
26
|
if Fiber.scheduler.nil?
|
|
26
27
|
raise ArgumentError, "Fiber concurrency requires Fiber.scheduler"
|
|
27
28
|
else
|
|
@@ -48,6 +49,7 @@ module LLM::Function::Fiber
|
|
|
48
49
|
##
|
|
49
50
|
# @return [LLM::Function::Return]
|
|
50
51
|
def wait
|
|
52
|
+
return @guarded if @guarded
|
|
51
53
|
spawn unless @fiber
|
|
52
54
|
@result ||= @fiber.value
|
|
53
55
|
end
|
|
@@ -20,6 +20,7 @@ class LLM::Function
|
|
|
20
20
|
##
|
|
21
21
|
# @return [LLM::Function::Fork::Task]
|
|
22
22
|
def spawn
|
|
23
|
+
return if @guarded
|
|
23
24
|
@span = @tracer&.on_tool_start(
|
|
24
25
|
id: @function.id, name: @function.name,
|
|
25
26
|
arguments: @function.arguments, model: @function.model
|
|
@@ -59,6 +60,7 @@ class LLM::Function
|
|
|
59
60
|
##
|
|
60
61
|
# @return [LLM::Function::Return]
|
|
61
62
|
def wait
|
|
63
|
+
return @guarded if @guarded
|
|
62
64
|
spawn unless @spawned
|
|
63
65
|
kind, data = @ch.result.recv
|
|
64
66
|
raise LLM::Interrupt if kind == :interrupt
|