llm.rb 15.2.2 → 15.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +197 -3
- data/README.md +180 -64
- data/bin/llm.rb +9 -2
- data/data/alibaba.json +45 -0
- data/data/anthropic.json +67 -0
- data/data/bedrock.json +1466 -397
- data/data/deepinfra.json +148 -16
- data/data/deepseek.json +3 -0
- data/data/mistral.json +42 -0
- data/data/openai.json +186 -0
- data/data/openrouter.json +1538 -378
- data/data/xai.json +53 -20
- data/data/zai.json +90 -4
- data/docs/deepdive/advanced/compaction.md +1 -2
- data/docs/deepdive/advanced/context.md +214 -1
- data/docs/deepdive/advanced/guard.md +9 -57
- data/docs/deepdive/features/builtin_tools.md +14 -16
- data/docs/deepdive/features/console.md +5 -0
- data/docs/deepdive/features/database.md +85 -10
- data/docs/deepdive/fundamentals/agents.md +7 -8
- data/docs/deepdive/fundamentals/providers.md +45 -5
- data/docs/deepdive/fundamentals/schema.md +73 -0
- data/docs/deepdive/fundamentals/tools.md +80 -27
- data/docs/deepdive/media/audio.md +8 -19
- data/docs/deepdive/media/images.md +8 -10
- data/docs/deepdive/media/ocr.md +1 -3
- data/docs/deepdive/reference/cost.md +48 -0
- data/docs/deepdive/reference/tracer.md +76 -0
- data/docs/deepdive.md +1 -1
- data/lib/llm/active_record/message.rb +113 -0
- data/lib/llm/active_record.rb +1 -0
- data/lib/llm/agent.rb +44 -19
- data/lib/llm/console/buffer.rb +9 -1
- data/lib/llm/console.rb +6 -1
- data/lib/llm/context/deserializer.rb +10 -3
- data/lib/llm/context.rb +62 -26
- data/lib/llm/guard.rb +2 -8
- data/lib/llm/message.rb +18 -7
- data/lib/llm/provider.rb +74 -16
- data/lib/llm/providers/alibaba.rb +15 -0
- data/lib/llm/providers/anthropic/error_handler.rb +5 -2
- data/lib/llm/providers/anthropic/files.rb +12 -12
- data/lib/llm/providers/anthropic/models.rb +2 -2
- data/lib/llm/providers/anthropic.rb +5 -3
- data/lib/llm/providers/bedrock/error_handler.rb +3 -2
- data/lib/llm/providers/bedrock/models.rb +5 -3
- data/lib/llm/providers/bedrock.rb +5 -3
- data/lib/llm/providers/deepinfra/audio.rb +4 -4
- data/lib/llm/providers/deepinfra/images.rb +4 -4
- data/lib/llm/providers/google/error_handler.rb +5 -2
- data/lib/llm/providers/google/files.rb +10 -10
- data/lib/llm/providers/google/images.rb +2 -2
- data/lib/llm/providers/google/models.rb +2 -2
- data/lib/llm/providers/google.rb +7 -7
- data/lib/llm/providers/mistral.rb +3 -1
- data/lib/llm/providers/ollama/error_handler.rb +5 -2
- data/lib/llm/providers/ollama/models.rb +2 -2
- data/lib/llm/providers/ollama.rb +7 -5
- data/lib/llm/providers/openai/audio.rb +6 -6
- data/lib/llm/providers/openai/error_handler.rb +5 -2
- data/lib/llm/providers/openai/files.rb +10 -10
- data/lib/llm/providers/openai/images.rb +4 -4
- data/lib/llm/providers/openai/models.rb +2 -2
- data/lib/llm/providers/openai/moderations.rb +2 -2
- data/lib/llm/providers/openai/request_adapter.rb +1 -1
- data/lib/llm/providers/openai/responses.rb +8 -8
- data/lib/llm/providers/openai/vector_stores.rb +22 -22
- data/lib/llm/providers/openai.rb +10 -8
- data/lib/llm/providers/xai/images.rb +4 -4
- data/lib/llm/schema.rb +24 -0
- data/lib/llm/tracer/telemetry.rb +4 -4
- data/lib/llm/tracer.rb +11 -3
- data/lib/llm/transport/execution.rb +8 -4
- data/lib/llm/utils.rb +13 -0
- data/lib/llm/version.rb +1 -1
- data/llm.gemspec +2 -2
- metadata +5 -5
- data/lib/llm/guard/loop.rb +0 -89
|
@@ -134,6 +134,82 @@ OpenTelemetry.
|
|
|
134
134
|
The tracer can also write to a file with the `path:` option to
|
|
135
135
|
[`LLM::Tracer::Logger.new`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Logger.html#initialize-instance_method).
|
|
136
136
|
|
|
137
|
+
### Hooks
|
|
138
|
+
|
|
139
|
+
#### Overview
|
|
140
|
+
|
|
141
|
+
[`LLM::Tracer`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer.html)
|
|
142
|
+
exposes one method per event in a request's lifecycle. A subclass
|
|
143
|
+
implements the events it cares about and routes them anywhere: a
|
|
144
|
+
logger, a metrics counter, or a database table.
|
|
145
|
+
|
|
146
|
+
#### How it works
|
|
147
|
+
|
|
148
|
+
Three hooks cover a provider request. `on_request_start` fires before
|
|
149
|
+
the request is sent and returns the span that `on_request_finish` and
|
|
150
|
+
`on_request_error` receive. Every request carries a `request_id`, a
|
|
151
|
+
UUIDv7 minted when the request begins and passed to all three hooks
|
|
152
|
+
for that request, so a tracer can correlate its events even when a
|
|
153
|
+
turn makes several requests.
|
|
154
|
+
|
|
155
|
+
Three more hooks cover a local tool call. `on_tool_start` fires before
|
|
156
|
+
the tool runs and returns the span that `on_tool_finish` and
|
|
157
|
+
`on_tool_error` receive.
|
|
158
|
+
|
|
159
|
+
A turn is additionally bracketed with `start_trace` and `stop_trace`.
|
|
160
|
+
The runtime calls them around every agent turn with a `trace_group_id`,
|
|
161
|
+
and a tracer that supports it (such as
|
|
162
|
+
[`LLM::Tracer::Telemetry`](https://r.uby.dev/api-docs/llm.rb/LLM/Tracer/Telemetry.html))
|
|
163
|
+
uses that id to give every span of the turn the same trace id:
|
|
164
|
+
|
|
165
|
+
```ruby
|
|
166
|
+
class MyTracer < LLM::Tracer
|
|
167
|
+
def on_request_start(operation:, model: nil, **)
|
|
168
|
+
warn "start #{operation} #{model}"
|
|
169
|
+
end
|
|
170
|
+
|
|
171
|
+
def on_request_finish(operation:, res:, **)
|
|
172
|
+
warn "finish #{operation}"
|
|
173
|
+
end
|
|
174
|
+
|
|
175
|
+
def on_request_error(ex:, **)
|
|
176
|
+
warn "error #{ex.class}"
|
|
177
|
+
end
|
|
178
|
+
|
|
179
|
+
def on_tool_start(id:, name:, arguments:, model:, **)
|
|
180
|
+
warn "tool #{name}"
|
|
181
|
+
end
|
|
182
|
+
|
|
183
|
+
def on_tool_finish(result:, **)
|
|
184
|
+
warn "tool #{result.name} done"
|
|
185
|
+
end
|
|
186
|
+
|
|
187
|
+
def on_tool_error(ex:, **)
|
|
188
|
+
warn "tool error #{ex.class}"
|
|
189
|
+
end
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
llm = LLM.deepseek(key: ENV["KEY"])
|
|
193
|
+
llm.tracer = MyTracer.new(llm)
|
|
194
|
+
agent = LLM::Agent.new(llm)
|
|
195
|
+
agent.talk "Hello"
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
#### Why would I use it?
|
|
199
|
+
|
|
200
|
+
The built-in tracers cover logging and OpenTelemetry. A hook lets you
|
|
201
|
+
send the same events somewhere else, and because the built-in tracers
|
|
202
|
+
accept the keywords they do not use, existing tracer code keeps working
|
|
203
|
+
as hooks gain parameters.
|
|
204
|
+
|
|
205
|
+
#### Notes
|
|
206
|
+
|
|
207
|
+
The base class raises `NotImplementedError` for any hook it does not
|
|
208
|
+
implement, so a tracer must cover every hook the runtime calls: the six
|
|
209
|
+
request and tool hooks above. Accept `**` to absorb keywords you do not
|
|
210
|
+
read, as the built-in tracers do, so a hook that gains a parameter does
|
|
211
|
+
not break your subclass.
|
|
212
|
+
|
|
137
213
|
### PrettyLogger
|
|
138
214
|
|
|
139
215
|
#### Overview
|
data/docs/deepdive.md
CHANGED
|
@@ -44,7 +44,7 @@ useful when you need to go beyond the basics.
|
|
|
44
44
|
#### Notes
|
|
45
45
|
|
|
46
46
|
The deepdive is a living document. Sections are added as new
|
|
47
|
-
features land. The [README.md](https://github.com/r-uby-dev/llm#readme)
|
|
47
|
+
features land. The [README.md](https://github.com/r-uby-dev/llm.rb#readme)
|
|
48
48
|
is the best place to start if you are new to llm.rb.
|
|
49
49
|
|
|
50
50
|
---
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LLM::ActiveRecord
|
|
4
|
+
##
|
|
5
|
+
# Represents a message in an agent's memory.
|
|
6
|
+
#
|
|
7
|
+
# This class is virtual and never materializes
|
|
8
|
+
# in the database as a real table. It provides
|
|
9
|
+
# a SQL view into the messages array stored in
|
|
10
|
+
# the JSONB column that carries an agent's
|
|
11
|
+
# runtime state, and it returns relations, so
|
|
12
|
+
# messages can be filtered and ordered in the
|
|
13
|
+
# database instead of in memory.
|
|
14
|
+
#
|
|
15
|
+
# It expects the agent and this class to share a
|
|
16
|
+
# connection, which they do when both live on the
|
|
17
|
+
# same database - the default in Rails.
|
|
18
|
+
#
|
|
19
|
+
# @example
|
|
20
|
+
# LLM::ActiveRecord::Message.for(agent:)
|
|
21
|
+
# .where(role: "assistant")
|
|
22
|
+
# .count
|
|
23
|
+
class Message < ActiveRecord::Base
|
|
24
|
+
##
|
|
25
|
+
# The name the derived table is given. It is "created"
|
|
26
|
+
# on-demand and it is here so that ActiveRecord has
|
|
27
|
+
# something to select from.
|
|
28
|
+
self.table_name = "llm_agent_messages"
|
|
29
|
+
|
|
30
|
+
##
|
|
31
|
+
# The derived table does not exist, so there is no schema
|
|
32
|
+
# to load. Without this, ActiveRecord asks the database
|
|
33
|
+
# for the columns of a table that was never created.
|
|
34
|
+
# @return [void]
|
|
35
|
+
def self.load_schema!
|
|
36
|
+
@columns_hash = {}.freeze
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
##
|
|
40
|
+
# Build a relation over one agent's messages.
|
|
41
|
+
#
|
|
42
|
+
# The table, the column and its type come from the
|
|
43
|
+
# agent's own class, so this works for any model that
|
|
44
|
+
# keeps its conversation in a column.
|
|
45
|
+
#
|
|
46
|
+
# The base fields the query guarantees are columns -
|
|
47
|
+
# id, role, content, tools - so they are what a
|
|
48
|
+
# caller's `where` and `order` are written against.
|
|
49
|
+
# The whole message is carried along as `data`, for
|
|
50
|
+
# fields the query does not name yet.
|
|
51
|
+
#
|
|
52
|
+
# The derived table is aliased as `llm_agent_messages`
|
|
53
|
+
# because ActiveRecord qualifies its SELECT with the
|
|
54
|
+
# class' table name.
|
|
55
|
+
#
|
|
56
|
+
# @param [ActiveRecord::Base] agent
|
|
57
|
+
# An instance of an ActiveRecord model.
|
|
58
|
+
# @return [ActiveRecord::Relation]
|
|
59
|
+
def self.for(agent:)
|
|
60
|
+
klass = agent.class
|
|
61
|
+
connection = klass.connection
|
|
62
|
+
table = connection.quote_table_name(klass.table_name)
|
|
63
|
+
options = klass.llm_plugin_options
|
|
64
|
+
column = connection.quote_column_name(options.fetch(:data_column))
|
|
65
|
+
from(<<~SQL).where(agent_id: agent&.id)
|
|
66
|
+
(SELECT #{table}.id AS agent_id,
|
|
67
|
+
(message ->> 'id') AS id,
|
|
68
|
+
(message ->> 'role') AS role,
|
|
69
|
+
(message ->> 'content') AS content,
|
|
70
|
+
(message -> 'tools') AS tools,
|
|
71
|
+
ordinality AS position,
|
|
72
|
+
message AS data
|
|
73
|
+
FROM #{table},
|
|
74
|
+
jsonb_array_elements(#{table}.#{column} -> 'messages')
|
|
75
|
+
WITH ORDINALITY AS each(message, ordinality))
|
|
76
|
+
AS llm_agent_messages
|
|
77
|
+
SQL
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
##
|
|
81
|
+
# @return (see LLM::Message#tool_call?)
|
|
82
|
+
def tool_call?
|
|
83
|
+
unwrap!.tool_call?
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
##
|
|
87
|
+
# @return (see LLM::Message#tool_return?)
|
|
88
|
+
def tool_return?
|
|
89
|
+
unwrap!.tool_return?
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
##
|
|
93
|
+
# The message, as the runtime would hand it back.
|
|
94
|
+
#
|
|
95
|
+
# The whole serialized message is used, so fields the
|
|
96
|
+
# query does not name: usage, reasoning content,
|
|
97
|
+
# and compaction survive the round trip through the
|
|
98
|
+
# database.
|
|
99
|
+
#
|
|
100
|
+
# @return [LLM::Message]
|
|
101
|
+
def unwrap!
|
|
102
|
+
@message ||= begin
|
|
103
|
+
stored = data.is_a?(Hash) ? data : {}
|
|
104
|
+
extra = stored.each_with_object({}) { |(key, value), acc| acc[key.to_sym] = value }
|
|
105
|
+
extra[:id] = stored["id"] || id
|
|
106
|
+
extra[:role] ||= role
|
|
107
|
+
extra[:content] ||= content
|
|
108
|
+
extra[:tool_calls] = stored["tools"] || tools
|
|
109
|
+
LLM::Message.new(extra[:role], extra[:content], extra)
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
end
|
|
113
|
+
end
|
data/lib/llm/active_record.rb
CHANGED
data/lib/llm/agent.rb
CHANGED
|
@@ -17,10 +17,6 @@ module LLM
|
|
|
17
17
|
# **Notes:**
|
|
18
18
|
# * Instructions are injected once unless a system message is already present.
|
|
19
19
|
# * An agent automatically executes tool loops (unlike {LLM::Context LLM::Context}).
|
|
20
|
-
# * The automatic tool loop enables the wrapped context's `guard` by default.
|
|
21
|
-
# The built-in {LLM::Guard::Loop LLM::Guard::Loop} detects repeated
|
|
22
|
-
# tool-call patterns and blocks stuck execution before more tool work is
|
|
23
|
-
# queued.
|
|
24
20
|
# * The tool loop can be bounded with `tool_budget`. Once the budget is
|
|
25
21
|
# spent, no further tool calls are run for that turn: the agent sends an
|
|
26
22
|
# in-band advisory message back through the model instead, and keeps
|
|
@@ -106,6 +102,8 @@ module LLM
|
|
|
106
102
|
# end
|
|
107
103
|
#
|
|
108
104
|
# @param [Hash] properties
|
|
105
|
+
# @option properties [String, Symbol, Proc] :name
|
|
106
|
+
# @option properties [String, Symbol, Proc] :description
|
|
109
107
|
# @option properties [String] :instructions
|
|
110
108
|
# @option properties [String] :model
|
|
111
109
|
# @option properties [Array<LLM::Function>] :tools
|
|
@@ -138,7 +136,7 @@ module LLM
|
|
|
138
136
|
# @return [String]
|
|
139
137
|
# Return's the agents name
|
|
140
138
|
def self.name(name = UNDEFINED, &block)
|
|
141
|
-
if name.equal?(UNDEFINED)
|
|
139
|
+
if name.equal?(UNDEFINED) and block.nil?
|
|
142
140
|
if @name.nil?
|
|
143
141
|
name = to_s.split("::").last
|
|
144
142
|
@name = name.gsub(CASE_PATTERN, "-").downcase
|
|
@@ -158,6 +156,11 @@ module LLM
|
|
|
158
156
|
##
|
|
159
157
|
# Set or get an agent's description
|
|
160
158
|
# @note
|
|
159
|
+
# Reading this on the class returns what was configured - a
|
|
160
|
+
# Symbol or Proc included - because an instance is what
|
|
161
|
+
# resolves those. Read it on an instance for the description
|
|
162
|
+
# itself.
|
|
163
|
+
# @note
|
|
161
164
|
# This method serves as a self-documenting string.
|
|
162
165
|
# It is optional but recommended.
|
|
163
166
|
# @param [String] desc
|
|
@@ -165,7 +168,7 @@ module LLM
|
|
|
165
168
|
# @return [String, nil]
|
|
166
169
|
# Returns the agent's description
|
|
167
170
|
def self.description(desc = UNDEFINED, &block)
|
|
168
|
-
if desc.equal?(UNDEFINED)
|
|
171
|
+
if desc.equal?(UNDEFINED) and block.nil?
|
|
169
172
|
@desc
|
|
170
173
|
else
|
|
171
174
|
@desc = block || desc
|
|
@@ -338,10 +341,10 @@ module LLM
|
|
|
338
341
|
# The path to a file
|
|
339
342
|
# @return [String, nil]
|
|
340
343
|
def self.path(path = UNDEFINED, &block)
|
|
341
|
-
if path.equal?(UNDEFINED)
|
|
344
|
+
if path.equal?(UNDEFINED) and block.nil?
|
|
342
345
|
@path
|
|
343
346
|
else
|
|
344
|
-
@path =
|
|
347
|
+
@path = block || path
|
|
345
348
|
end
|
|
346
349
|
end
|
|
347
350
|
|
|
@@ -365,10 +368,10 @@ module LLM
|
|
|
365
368
|
# a single turn.
|
|
366
369
|
# @return [Integer, nil]
|
|
367
370
|
def self.tool_budget(budget = UNDEFINED, &block)
|
|
368
|
-
if budget.equal?(UNDEFINED)
|
|
371
|
+
if budget.equal?(UNDEFINED) and block.nil?
|
|
369
372
|
@tool_budget
|
|
370
373
|
else
|
|
371
|
-
@tool_budget =
|
|
374
|
+
@tool_budget = block || budget
|
|
372
375
|
end
|
|
373
376
|
end
|
|
374
377
|
|
|
@@ -434,13 +437,11 @@ module LLM
|
|
|
434
437
|
end
|
|
435
438
|
end
|
|
436
439
|
##
|
|
437
|
-
#
|
|
438
|
-
#
|
|
439
|
-
#
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
params[:retry_budget] = retry_budget if params[:retry_budget].equal?(UNDEFINED)
|
|
443
|
-
@ctx = LLM::Context.new(llm, {guard: LLM::Guard::Loop}.merge(params))
|
|
440
|
+
# The provider decides how patient a turn should be: it
|
|
441
|
+
# knows its own API, and how often it recovers from a
|
|
442
|
+
# rate limit rather than failing outright.
|
|
443
|
+
params[:retry_budget] = llm.retry_budget if params[:retry_budget].equal?(UNDEFINED)
|
|
444
|
+
@ctx = LLM::Context.new(llm, params)
|
|
444
445
|
@path and File.readable?(@path) ? @ctx.restore(path:) : nil
|
|
445
446
|
end
|
|
446
447
|
|
|
@@ -510,6 +511,22 @@ module LLM
|
|
|
510
511
|
@ctx.messages
|
|
511
512
|
end
|
|
512
513
|
|
|
514
|
+
##
|
|
515
|
+
# Returns a stable id for this agent. The id is borrowed from
|
|
516
|
+
# the context it wraps, so it survives save and restore.
|
|
517
|
+
# @return [String]
|
|
518
|
+
def id
|
|
519
|
+
@ctx.id
|
|
520
|
+
end
|
|
521
|
+
|
|
522
|
+
##
|
|
523
|
+
# Returns the time this agent was created. It is derived
|
|
524
|
+
# from the context id, so it is not persisted separately.
|
|
525
|
+
# @return [Time, nil]
|
|
526
|
+
def created_at
|
|
527
|
+
@ctx.created_at
|
|
528
|
+
end
|
|
529
|
+
|
|
513
530
|
##
|
|
514
531
|
# @return [Integer]
|
|
515
532
|
def retry_budget
|
|
@@ -851,8 +868,16 @@ module LLM
|
|
|
851
868
|
end
|
|
852
869
|
res
|
|
853
870
|
end
|
|
854
|
-
|
|
855
|
-
|
|
871
|
+
##
|
|
872
|
+
# One turn, one trace group. A tracer
|
|
873
|
+
# is told where the turn begins and
|
|
874
|
+
# where it ends. The trace group ID
|
|
875
|
+
# identifies the turn.
|
|
876
|
+
tracer = @tracer || @llm.tracer
|
|
877
|
+
tracer.start_trace(name: "llm.turn", trace_group_id: SecureRandom.uuid_v7)
|
|
878
|
+
@llm.with_tracer(tracer, &run)
|
|
879
|
+
ensure
|
|
880
|
+
tracer&.stop_trace
|
|
856
881
|
end
|
|
857
882
|
|
|
858
883
|
##
|
data/lib/llm/console/buffer.rb
CHANGED
|
@@ -72,6 +72,14 @@ class LLM::Console
|
|
|
72
72
|
@snapshot = nil
|
|
73
73
|
end
|
|
74
74
|
|
|
75
|
+
##
|
|
76
|
+
# Returns true while the buffer has an active row
|
|
77
|
+
# for the streaming path to replace.
|
|
78
|
+
# @return [Boolean]
|
|
79
|
+
def open?
|
|
80
|
+
!@snapshot.nil?
|
|
81
|
+
end
|
|
82
|
+
|
|
75
83
|
##
|
|
76
84
|
# @return [void]
|
|
77
85
|
def scroll_up(height)
|
|
@@ -131,7 +139,7 @@ class LLM::Console
|
|
|
131
139
|
# One or more chunks.
|
|
132
140
|
# @return [void]
|
|
133
141
|
def replace(chunks)
|
|
134
|
-
@rows = @snapshot.map(&:dup)
|
|
142
|
+
@rows = @snapshot ? @snapshot.map(&:dup) : @rows
|
|
135
143
|
chunks.each { wrap(_1, @rows) }
|
|
136
144
|
end
|
|
137
145
|
|
data/lib/llm/console.rb
CHANGED
|
@@ -284,6 +284,11 @@ module LLM
|
|
|
284
284
|
# UI can repaint. Cap the stream chunks drained per call so
|
|
285
285
|
# read! gives control back to the key loop: leftover chunks
|
|
286
286
|
# stay queued and are drained by the next read!.
|
|
287
|
+
#
|
|
288
|
+
# A turn that was cancelled can still have chunks queued, and
|
|
289
|
+
# they arrive after the buffer is closed. There is no active
|
|
290
|
+
# row to replace at that point, so they are dropped.
|
|
291
|
+
next unless buffer.open?
|
|
287
292
|
status.text = think_text if stream.tools.empty?
|
|
288
293
|
write_message name, markdown(value), method: :replace
|
|
289
294
|
burst += 1
|
|
@@ -292,7 +297,7 @@ module LLM
|
|
|
292
297
|
self.status = value
|
|
293
298
|
when :done
|
|
294
299
|
status.text = "idle"
|
|
295
|
-
write_message
|
|
300
|
+
write_message(name, markdown(value), method: :replace) if buffer.open?
|
|
296
301
|
buffer.close
|
|
297
302
|
@thread = nil
|
|
298
303
|
when :cancel
|
|
@@ -5,7 +5,12 @@ class LLM::Context
|
|
|
5
5
|
# @api private
|
|
6
6
|
module Deserializer
|
|
7
7
|
##
|
|
8
|
-
# Restore a saved context state
|
|
8
|
+
# Restore a saved context state.
|
|
9
|
+
#
|
|
10
|
+
# The `context_used` and `context_window` keys carried by a payload
|
|
11
|
+
# are projections for queryability. They are deliberately ignored
|
|
12
|
+
# here: the runtime derives both from the messages and the
|
|
13
|
+
# registry, so a payload can never seed them.
|
|
9
14
|
# @param [String, nil] path
|
|
10
15
|
# The path to a JSON file
|
|
11
16
|
# @param [String, nil] string
|
|
@@ -25,6 +30,8 @@ class LLM::Context
|
|
|
25
30
|
else
|
|
26
31
|
LLM.json.load(string)
|
|
27
32
|
end
|
|
33
|
+
@id = ctx["id"] || @id
|
|
34
|
+
@created_at = nil
|
|
28
35
|
@messages.concat [*ctx["messages"]].map { deserialize_message(_1) }
|
|
29
36
|
@compacted = !!ctx["compacted"]
|
|
30
37
|
self
|
|
@@ -40,8 +47,8 @@ class LLM::Context
|
|
|
40
47
|
usage = payload["usage"]
|
|
41
48
|
reasoning_content = payload["reasoning_content"]
|
|
42
49
|
compaction = payload["compaction"]
|
|
43
|
-
|
|
44
|
-
extra = {tool_calls:, original_tool_calls:, tools: @params[:tools], usage:, reasoning_content:, compaction:,
|
|
50
|
+
id = payload["id"]
|
|
51
|
+
extra = {tool_calls:, original_tool_calls:, tools: @params[:tools], usage:, reasoning_content:, compaction:, id:}.compact
|
|
45
52
|
content = returns.nil? ? deserialize_content(payload["content"]) : returns
|
|
46
53
|
LLM::Message.new(payload["role"], content, extra)
|
|
47
54
|
end
|
data/lib/llm/context.rb
CHANGED
|
@@ -55,7 +55,7 @@ module LLM
|
|
|
55
55
|
# @api private
|
|
56
56
|
# @return [Array<Symbol>]
|
|
57
57
|
def self.params
|
|
58
|
-
%w[guard retry_budget concurrency transformer compactor record]
|
|
58
|
+
%w[guard retry_budget concurrency transformer compactor record id]
|
|
59
59
|
end
|
|
60
60
|
|
|
61
61
|
##
|
|
@@ -78,6 +78,12 @@ module LLM
|
|
|
78
78
|
# @return [Object, nil]
|
|
79
79
|
attr_reader :record
|
|
80
80
|
|
|
81
|
+
##
|
|
82
|
+
# Returns a stable id for this context. It is generated once
|
|
83
|
+
# on creation and restored with the runtime state on load.
|
|
84
|
+
# @return [String]
|
|
85
|
+
attr_reader :id
|
|
86
|
+
|
|
81
87
|
##
|
|
82
88
|
# @param [LLM::Provider] llm
|
|
83
89
|
# A provider
|
|
@@ -89,6 +95,8 @@ module LLM
|
|
|
89
95
|
# Defaults to `:responses` for OpenAI, otherwise it defaults
|
|
90
96
|
# to `:completions`.
|
|
91
97
|
# @option params [String] :model Defaults to the provider's default model
|
|
98
|
+
# @option params [String] :id
|
|
99
|
+
# A stable id for the context. Defaults to a UUID.
|
|
92
100
|
# @option params [Class<LLM::Compactor>, nil] :compactor
|
|
93
101
|
# A compactor class to use for context compaction. Defaults to
|
|
94
102
|
# {LLM::Compactor::Null}.
|
|
@@ -109,6 +117,7 @@ module LLM
|
|
|
109
117
|
def initialize(llm, params = {})
|
|
110
118
|
params = {}.merge!(params)
|
|
111
119
|
@llm = llm
|
|
120
|
+
@id = params.delete(:id) || SecureRandom.uuid_v7
|
|
112
121
|
@record = params.delete(:record)
|
|
113
122
|
@mode = params.delete(:mode) || (llm.name == :openai ? :responses : :completions)
|
|
114
123
|
tools = [*params.delete(:tools), *load_skills(params.delete(:skills))]
|
|
@@ -174,10 +183,6 @@ module LLM
|
|
|
174
183
|
# via {LLM::Stream#on_tool_call}. A blocked call yields its in-band
|
|
175
184
|
# `guard_error` return without executing.
|
|
176
185
|
#
|
|
177
|
-
# The built-in implementation is {LLM::Guard::Loop LLM::Guard::Loop}, which
|
|
178
|
-
# detects repeated tool-call patterns and turns them into in-band
|
|
179
|
-
# `guard_error` tool returns.
|
|
180
|
-
#
|
|
181
186
|
# @return [Class<LLM::Guard>]
|
|
182
187
|
def guard
|
|
183
188
|
@guard[:klass]
|
|
@@ -489,12 +494,29 @@ module LLM
|
|
|
489
494
|
end
|
|
490
495
|
|
|
491
496
|
##
|
|
497
|
+
# Returns the time this context was created, derived from the
|
|
498
|
+
# timestamp embedded in its UUIDv7 id, or nil when it is not.
|
|
499
|
+
# @return [Time, nil]
|
|
500
|
+
def created_at
|
|
501
|
+
@created_at ||= LLM::Utils.timestamp(@id)
|
|
502
|
+
end
|
|
503
|
+
|
|
504
|
+
##
|
|
505
|
+
# Returns the runtime state as a Hash.
|
|
506
|
+
#
|
|
507
|
+
# `context_used` and `context_window` are projections written for
|
|
508
|
+
# queryability, so that saved state can be inspected at rest
|
|
509
|
+
# without loading it. They are never read back; the runtime
|
|
510
|
+
# always derives them from the messages and the registry.
|
|
492
511
|
# @return [Hash]
|
|
493
512
|
def to_h
|
|
494
513
|
{
|
|
495
514
|
schema_version: 1,
|
|
515
|
+
id: @id,
|
|
496
516
|
model:,
|
|
497
517
|
compacted:,
|
|
518
|
+
context_used:,
|
|
519
|
+
context_window:,
|
|
498
520
|
messages: @messages.map { serialize_message(_1) }
|
|
499
521
|
}
|
|
500
522
|
end
|
|
@@ -597,33 +619,37 @@ module LLM
|
|
|
597
619
|
# Executes a turn through the Responses API.
|
|
598
620
|
# @api private
|
|
599
621
|
def respond(prompt, params)
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
607
|
-
|
|
608
|
-
|
|
609
|
-
|
|
610
|
-
|
|
611
|
-
|
|
622
|
+
@llm.with(**headers) do
|
|
623
|
+
history = @messages.to_a
|
|
624
|
+
params = @params.merge(params).reject { self.class.params.include?(_1.to_s) }
|
|
625
|
+
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:, guard: @guard[:klass].new(self))
|
|
626
|
+
params[:stream] = LLM::Stream.try(params[:stream], extra:)
|
|
627
|
+
res_id = params[:store] == false ? nil : @messages.find(&:assistant?)&.response&.response_id
|
|
628
|
+
input = res_id ? [] : history
|
|
629
|
+
params[:input] = input
|
|
630
|
+
messages = transform(prompt, params, key: :input)
|
|
631
|
+
@stream = params[:stream]
|
|
632
|
+
new_messages = messages[input.size..]
|
|
633
|
+
params = params.merge(previous_response_id: res_id, input:).compact
|
|
634
|
+
[new_messages, params, @llm.responses.create(messages, params)]
|
|
635
|
+
end
|
|
612
636
|
end
|
|
613
637
|
|
|
614
638
|
##
|
|
615
639
|
# Executes a turn through the chat completions API.
|
|
616
640
|
# @api private
|
|
617
641
|
def complete(prompt, params)
|
|
618
|
-
|
|
619
|
-
|
|
620
|
-
|
|
621
|
-
|
|
622
|
-
|
|
623
|
-
|
|
624
|
-
|
|
625
|
-
|
|
626
|
-
|
|
642
|
+
@llm.with(**headers) do
|
|
643
|
+
history = @messages.to_a
|
|
644
|
+
params = params.merge(messages: history)
|
|
645
|
+
params = @params.merge(params).reject { self.class.params.include?(_1.to_s) }
|
|
646
|
+
extra = params.slice(:model, :tools).merge!(ctx: self, tracer:, guard: @guard[:klass].new(self))
|
|
647
|
+
params[:stream] = LLM::Stream.try(params[:stream], extra:)
|
|
648
|
+
messages = transform(prompt, params)
|
|
649
|
+
@stream = params[:stream]
|
|
650
|
+
new_messages = messages[history.size..]
|
|
651
|
+
[new_messages, params, @llm.complete(messages, params)]
|
|
652
|
+
end
|
|
627
653
|
end
|
|
628
654
|
|
|
629
655
|
##
|
|
@@ -652,5 +678,15 @@ module LLM
|
|
|
652
678
|
end
|
|
653
679
|
messages << LLM::Message.new(@llm.tool_role, cancelled) unless cancelled.empty?
|
|
654
680
|
end
|
|
681
|
+
|
|
682
|
+
##
|
|
683
|
+
# @return [Hash]
|
|
684
|
+
def headers
|
|
685
|
+
if @llm.name == :openrouter
|
|
686
|
+
{"x-session-id" => @id}
|
|
687
|
+
else
|
|
688
|
+
{}
|
|
689
|
+
end
|
|
690
|
+
end
|
|
655
691
|
end
|
|
656
692
|
end
|
data/lib/llm/guard.rb
CHANGED
|
@@ -10,16 +10,10 @@ module LLM
|
|
|
10
10
|
# an {LLM::Function::Return LLM::Function::Return} that closes the
|
|
11
11
|
# pending tool call, or `nil` when execution should continue. The guard
|
|
12
12
|
# is stamped onto the functions the context binds, so it runs whenever a
|
|
13
|
-
# task is spawned — including tool calls queued from a stream.
|
|
14
|
-
#
|
|
15
|
-
# detects repeated tool-call patterns. {LLM::Guard::Null LLM::Guard::Null}
|
|
16
|
-
# is a no-op and the default.
|
|
17
|
-
#
|
|
18
|
-
# {LLM::Agent LLM::Agent} enables {LLM::Guard::Loop LLM::Guard::Loop}
|
|
19
|
-
# by default through its wrapped context.
|
|
13
|
+
# task is spawned — including tool calls queued from a stream.
|
|
14
|
+
# {LLM::Guard::Null LLM::Guard::Null} is a no-op and the default.
|
|
20
15
|
class Guard
|
|
21
16
|
require_relative "guard/null"
|
|
22
|
-
require_relative "guard/loop"
|
|
23
17
|
|
|
24
18
|
##
|
|
25
19
|
# @return [LLM::Context]
|
data/lib/llm/message.rb
CHANGED
|
@@ -18,9 +18,10 @@ module LLM
|
|
|
18
18
|
attr_reader :extra
|
|
19
19
|
|
|
20
20
|
##
|
|
21
|
-
# Returns
|
|
22
|
-
#
|
|
23
|
-
|
|
21
|
+
# Returns a stable id for this message. It is generated once
|
|
22
|
+
# on creation and restored with the runtime state on load.
|
|
23
|
+
# @return [String]
|
|
24
|
+
attr_reader :id
|
|
24
25
|
|
|
25
26
|
##
|
|
26
27
|
# Returns a new message
|
|
@@ -32,7 +33,15 @@ module LLM
|
|
|
32
33
|
@role = role.to_s
|
|
33
34
|
@content = content
|
|
34
35
|
@extra = LLM::Object.from(extra)
|
|
35
|
-
@
|
|
36
|
+
@id = extra[:id] || SecureRandom.uuid_v7
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
##
|
|
40
|
+
# Returns the time this message was created, derived from the
|
|
41
|
+
# timestamp embedded in its UUIDv7 id, or nil when it is not.
|
|
42
|
+
# @return [Time, nil]
|
|
43
|
+
def created_at
|
|
44
|
+
@created_at ||= LLM::Utils.timestamp(@id)
|
|
36
45
|
end
|
|
37
46
|
|
|
38
47
|
##
|
|
@@ -40,10 +49,10 @@ module LLM
|
|
|
40
49
|
# @return [Hash]
|
|
41
50
|
def to_h
|
|
42
51
|
{
|
|
52
|
+
id: @id,
|
|
43
53
|
role:,
|
|
44
54
|
content:,
|
|
45
55
|
reasoning_content:,
|
|
46
|
-
created_at: created_at.utc.iso8601,
|
|
47
56
|
compaction: extra.compaction,
|
|
48
57
|
tools: extra.tool_calls&.map { LLM::Object === _1 ? _1.to_h : _1 },
|
|
49
58
|
usage:,
|
|
@@ -58,13 +67,15 @@ module LLM
|
|
|
58
67
|
end
|
|
59
68
|
|
|
60
69
|
##
|
|
61
|
-
# Returns true when two objects have the same role and content
|
|
70
|
+
# Returns true when two objects have the same role and content.
|
|
71
|
+
# The id is ignored, because it identifies a message rather than
|
|
72
|
+
# describing it.
|
|
62
73
|
# @param [Object] other
|
|
63
74
|
# The other object to compare
|
|
64
75
|
# @return [Boolean]
|
|
65
76
|
def ==(other)
|
|
66
77
|
if other.respond_to?(:to_h)
|
|
67
|
-
to_h == other.to_h
|
|
78
|
+
to_h.except(:id) == other.to_h.except(:id)
|
|
68
79
|
else
|
|
69
80
|
false
|
|
70
81
|
end
|