llm.rb 13.0.0 → 14.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +505 -14
- data/README.md +484 -50
- data/bin/llm.rb +148 -0
- data/data/anthropic.json +206 -263
- data/data/bedrock.json +2138 -1860
- data/data/deepinfra.json +1003 -624
- data/data/deepseek.json +38 -34
- data/data/google.json +1079 -371
- data/data/mistral.json +448 -368
- data/data/moonshot.json +384 -0
- data/data/openai.json +974 -1343
- data/data/xai.json +154 -126
- data/data/zai.json +191 -191
- data/lib/llm/agent.rb +123 -20
- data/lib/llm/context.rb +71 -88
- data/lib/llm/cost.rb +23 -17
- data/lib/llm/error.rb +0 -8
- data/lib/llm/function/array.rb +3 -3
- data/lib/llm/function/async/task.rb +2 -0
- data/lib/llm/function/fiber/task.rb +2 -0
- data/lib/llm/function/fork/task.rb +2 -0
- data/lib/llm/function/ractor/task.rb +2 -0
- data/lib/llm/function/sequential/group.rb +4 -1
- data/lib/llm/function/sequential/task.rb +1 -1
- data/lib/llm/function/task.rb +4 -0
- data/lib/llm/function/thread/task.rb +2 -0
- data/lib/llm/function.rb +33 -6
- data/lib/llm/guard/loop.rb +89 -0
- data/lib/llm/guard/null.rb +19 -0
- data/lib/llm/guard.rb +61 -0
- data/lib/llm/provider.rb +36 -0
- data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
- data/lib/llm/providers/anthropic.rb +2 -9
- data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
- data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
- data/lib/llm/providers/bedrock.rb +1 -8
- data/lib/llm/providers/google/stream_parser.rb +1 -0
- data/lib/llm/providers/google.rb +1 -8
- data/lib/llm/providers/mistral.rb +1 -1
- data/lib/llm/providers/moonshot.rb +76 -0
- data/lib/llm/providers/ollama.rb +2 -9
- data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
- data/lib/llm/providers/openai/responses.rb +7 -9
- data/lib/llm/providers/openai/stream_parser.rb +1 -0
- data/lib/llm/providers/openai.rb +4 -11
- data/lib/llm/repl/bar.rb +4 -3
- data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
- data/lib/llm/repl/color.rb +78 -0
- data/lib/llm/repl/command.rb +12 -5
- data/lib/llm/repl/commands/compact.rb +2 -2
- data/lib/llm/repl/commands/help.rb +3 -5
- data/lib/llm/repl/input/char.rb +46 -0
- data/lib/llm/repl/input/row.rb +39 -0
- data/lib/llm/repl/input.rb +251 -66
- data/lib/llm/repl/markdown/table.rb +11 -3
- data/lib/llm/repl/markdown.rb +34 -8
- data/lib/llm/repl/node.rb +37 -0
- data/lib/llm/repl/status.rb +42 -7
- data/lib/llm/repl/stream.rb +18 -6
- data/lib/llm/repl/walker.rb +3 -2
- data/lib/llm/repl/window.rb +54 -35
- data/lib/llm/repl.rb +74 -32
- data/lib/llm/skill.rb +20 -4
- data/lib/llm/stream.rb +8 -7
- data/lib/llm/tool.rb +29 -0
- data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
- data/lib/llm/tools/git.rb +3 -0
- data/lib/llm/tools/mkdir.rb +3 -0
- data/lib/llm/tools/rg.rb +3 -0
- data/lib/llm/tools/ruby.rb +46 -0
- data/lib/llm/tools/shell.rb +3 -0
- data/lib/llm/tracer/pretty_logger.rb +127 -0
- data/lib/llm/tracer.rb +1 -0
- data/lib/llm/transformer/null.rb +21 -0
- data/lib/llm/transformer.rb +55 -0
- data/lib/llm/version.rb +1 -1
- data/lib/llm.rb +12 -2
- data/llm.gemspec +9 -2
- data/resources/deepdive/advanced/cancellation.md +74 -0
- data/resources/deepdive/advanced/compaction.md +83 -0
- data/resources/deepdive/advanced/context.md +267 -0
- data/resources/deepdive/advanced/guard.md +371 -0
- data/resources/deepdive/advanced/tracer.md +180 -0
- data/resources/deepdive/advanced/transformer.md +67 -0
- data/resources/deepdive/advanced/transports.md +45 -0
- data/resources/deepdive/everything_else/audio.md +122 -0
- data/resources/deepdive/everything_else/cost.md +99 -0
- data/resources/deepdive/everything_else/images.md +89 -0
- data/resources/deepdive/everything_else/object.md +108 -0
- data/resources/deepdive/everything_else/ocr.md +48 -0
- data/resources/deepdive/fundamentals/agents.md +202 -0
- data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
- data/resources/deepdive/fundamentals/concurrency.md +104 -0
- data/resources/deepdive/fundamentals/database.md +449 -0
- data/resources/deepdive/fundamentals/embeddings.md +157 -0
- data/resources/deepdive/fundamentals/repl.md +87 -0
- data/resources/deepdive/fundamentals/schema.md +61 -0
- data/resources/deepdive/fundamentals/skills.md +106 -0
- data/resources/deepdive/fundamentals/stream.md +110 -0
- data/resources/deepdive/fundamentals/tools.md +265 -0
- data/resources/deepdive/protocols/a2a.md +106 -0
- data/resources/deepdive/protocols/mcp.md +111 -0
- data/resources/deepdive.md +58 -1792
- metadata +51 -7
- data/lib/llm/loop_guard.rb +0 -107
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
class LLM::Tool
|
|
4
|
+
##
|
|
5
|
+
# The {LLM::Tool::Ruby LLM::Tool::Ruby} class implements
|
|
6
|
+
# a tool that can execute an arbitrary string of Ruby code
|
|
7
|
+
# and return the result. It uses `test-cmd.rb` under the
|
|
8
|
+
# hood for process management.
|
|
9
|
+
class Ruby < self
|
|
10
|
+
require_relative "utils"
|
|
11
|
+
include Utils
|
|
12
|
+
|
|
13
|
+
name "ruby"
|
|
14
|
+
description "runs a string of ruby code"
|
|
15
|
+
parameter :code, String, "a string of ruby code"
|
|
16
|
+
parameter :timeout, Integer, "maximum runtime before timeout"
|
|
17
|
+
required %i[code]
|
|
18
|
+
defaults timeout: 15
|
|
19
|
+
|
|
20
|
+
##
|
|
21
|
+
# @param [String] code
|
|
22
|
+
# Ruby code
|
|
23
|
+
# @param [Integer] timeout
|
|
24
|
+
# Runtime timeout
|
|
25
|
+
def call(code:, timeout: 15)
|
|
26
|
+
command = spawn(code:)
|
|
27
|
+
wait(command:, timeout:)
|
|
28
|
+
{ok: command.success?, stdout: command.stdout, stderr: command.stderr}
|
|
29
|
+
rescue LLM::Interrupt
|
|
30
|
+
command.kill! if command&.running?
|
|
31
|
+
raise
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
private
|
|
35
|
+
|
|
36
|
+
def spawn(code:)
|
|
37
|
+
Command
|
|
38
|
+
.new(RbConfig.ruby)
|
|
39
|
+
.argv("-e", code)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
require "rbconfig"
|
|
43
|
+
LLM.require "test-cmd.rb", "~> 1.1"
|
|
44
|
+
Command = Test::Cmd
|
|
45
|
+
end
|
|
46
|
+
end
|
data/lib/llm/tools/shell.rb
CHANGED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LLM
|
|
4
|
+
##
|
|
5
|
+
# The {LLM::Tracer::PrettyLogger LLM::Tracer::PrettyLogger} class
|
|
6
|
+
# writes human-readable request and tool call logs to a console
|
|
7
|
+
# or file. Each event is a single line with the relevant context
|
|
8
|
+
# inline, no structured JSON.
|
|
9
|
+
#
|
|
10
|
+
# @example
|
|
11
|
+
# llm = LLM.openai(key: ENV["KEY"])
|
|
12
|
+
# llm.tracer = LLM::Tracer::PrettyLogger.new(llm)
|
|
13
|
+
#
|
|
14
|
+
# @example Writing to a file
|
|
15
|
+
# llm.tracer = LLM::Tracer::PrettyLogger.new(llm, io: File.open("log.txt", "a"))
|
|
16
|
+
class Tracer::PrettyLogger < Tracer
|
|
17
|
+
##
|
|
18
|
+
# @param (see LLM::Tracer#initialize)
|
|
19
|
+
def initialize(provider, options = {})
|
|
20
|
+
super
|
|
21
|
+
setup!(**options)
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
##
|
|
25
|
+
# @param (see LLM::Tracer#on_request_start)
|
|
26
|
+
# @return [void]
|
|
27
|
+
def on_request_start(operation:, model: nil, **)
|
|
28
|
+
@start = Process.clock_gettime(Process::CLOCK_MONOTONIC)
|
|
29
|
+
name = operation == "chat" ? "chat" : operation
|
|
30
|
+
@io.puts "#{timestamp} #{provider_name} #{name} (#{model || "default"})"
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
##
|
|
34
|
+
# @param (see LLM::Tracer#on_request_finish)
|
|
35
|
+
# @return [void]
|
|
36
|
+
def on_request_finish(operation:, res:, model: nil, **)
|
|
37
|
+
elapsed = @start ? (Process.clock_gettime(Process::CLOCK_MONOTONIC) - @start).round(2) : nil
|
|
38
|
+
tokens = format_tokens(res)
|
|
39
|
+
name = operation == "chat" ? "chat" : operation
|
|
40
|
+
parts = ["#{timestamp} #{provider_name} #{name} done"]
|
|
41
|
+
parts << tokens if tokens
|
|
42
|
+
parts << "#{elapsed}s" if elapsed
|
|
43
|
+
@io.puts parts.join(", ")
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
##
|
|
47
|
+
# @param (see LLM::Tracer#on_request_error)
|
|
48
|
+
# @return [void]
|
|
49
|
+
def on_request_error(ex:, **)
|
|
50
|
+
@io.puts "#{timestamp} #{provider_name} error #{ex.class}: #{ex.message}"
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
##
|
|
54
|
+
# @param (see LLM::Tracer#on_tool_start)
|
|
55
|
+
# @return [void]
|
|
56
|
+
def on_tool_start(id:, name:, arguments:, **)
|
|
57
|
+
@io.puts "#{timestamp} #{name}(#{format_arguments(arguments)})"
|
|
58
|
+
end
|
|
59
|
+
|
|
60
|
+
##
|
|
61
|
+
# @param (see LLM::Tracer#on_tool_finish)
|
|
62
|
+
# @return [void]
|
|
63
|
+
def on_tool_finish(result:, **)
|
|
64
|
+
@io.puts "#{timestamp} #{result.name} -> #{format_value(result.value)}"
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
##
|
|
68
|
+
# @param (see LLM::Tracer#on_tool_error)
|
|
69
|
+
# @return [void]
|
|
70
|
+
def on_tool_error(ex:, **)
|
|
71
|
+
@io.puts "#{timestamp} tool error #{ex.class}: #{ex.message}"
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
private
|
|
75
|
+
|
|
76
|
+
def setup!(io: $stderr)
|
|
77
|
+
@io = io
|
|
78
|
+
@start = nil
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
def timestamp
|
|
82
|
+
Time.now.strftime("%H:%M:%S")
|
|
83
|
+
end
|
|
84
|
+
|
|
85
|
+
def format_tokens(res)
|
|
86
|
+
usage = res.usage
|
|
87
|
+
if usage.input_tokens and usage.output_tokens
|
|
88
|
+
"in=#{usage.input_tokens} out=#{usage.output_tokens}"
|
|
89
|
+
elsif usage.input_tokens
|
|
90
|
+
"in=#{usage.input_tokens}"
|
|
91
|
+
elsif usage.output_tokens
|
|
92
|
+
"out=#{usage.output_tokens}"
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def format_arguments(args, max: 50)
|
|
97
|
+
return "" unless args
|
|
98
|
+
case args
|
|
99
|
+
when Hash, LLM::Object
|
|
100
|
+
result = args.map { |k, v| "#{k}: #{format_value(v)}" }.join(", ")
|
|
101
|
+
when Array
|
|
102
|
+
result = args.map { |v| format_value(v) }.join(", ")
|
|
103
|
+
else
|
|
104
|
+
result = args.inspect
|
|
105
|
+
end
|
|
106
|
+
result.size > max ? "#{result[0...max - 1]}..." : result
|
|
107
|
+
end
|
|
108
|
+
|
|
109
|
+
def format_value(value, max: 18)
|
|
110
|
+
case value
|
|
111
|
+
when String
|
|
112
|
+
value.size > max ? "#{value[0...max]}...".inspect : value.inspect
|
|
113
|
+
when Array
|
|
114
|
+
items = value.take(2).map { format_value(_1, max: 10) }
|
|
115
|
+
items << "..." if value.size > 2
|
|
116
|
+
"[#{items.join(", ")}]"
|
|
117
|
+
when Hash
|
|
118
|
+
"{...}"
|
|
119
|
+
when nil
|
|
120
|
+
"nil"
|
|
121
|
+
else
|
|
122
|
+
str = value.inspect
|
|
123
|
+
str.size > max ? "#{str[0...max]}..." : str
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
end
|
|
127
|
+
end
|
data/lib/llm/tracer.rb
CHANGED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
class LLM::Transformer
|
|
4
|
+
##
|
|
5
|
+
# An {LLM::Transformer::Null LLM::Transformer::Null} is a
|
|
6
|
+
# transformer that does nothing. It is used as the default when
|
|
7
|
+
# no transformer strategy is configured.
|
|
8
|
+
#
|
|
9
|
+
# It returns the given message unchanged.
|
|
10
|
+
class Null < self
|
|
11
|
+
##
|
|
12
|
+
# @param [LLM::Message] message
|
|
13
|
+
# The message to transform
|
|
14
|
+
# @param [Hash] opts
|
|
15
|
+
# Ignored
|
|
16
|
+
# @return [LLM::Message]
|
|
17
|
+
def call(message:, **opts)
|
|
18
|
+
message
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module LLM
|
|
4
|
+
##
|
|
5
|
+
# {LLM::Transformer LLM::Transformer} is the superclass for
|
|
6
|
+
# message transformers in llm.rb.
|
|
7
|
+
#
|
|
8
|
+
# A transformer is bound to a context and rewrites a single
|
|
9
|
+
# message before it is sent to the provider. Each subclass
|
|
10
|
+
# implements a different transformation: it takes a message in
|
|
11
|
+
# {#call} and returns a message. {LLM::Transformer::Null} is a
|
|
12
|
+
# no-op (the default).
|
|
13
|
+
#
|
|
14
|
+
# A transformer may mutate the message in place or return a new
|
|
15
|
+
# one. Either way, the returned message is what gets sent.
|
|
16
|
+
class Transformer
|
|
17
|
+
require_relative "transformer/null"
|
|
18
|
+
|
|
19
|
+
##
|
|
20
|
+
# @return [LLM::Context]
|
|
21
|
+
attr_reader :ctx
|
|
22
|
+
|
|
23
|
+
##
|
|
24
|
+
# @param ctx [LLM::Context, LLM::Agent]
|
|
25
|
+
# @return [LLM::Transformer]
|
|
26
|
+
def initialize(ctx)
|
|
27
|
+
@ctx = LLM::Agent === ctx ? ctx.instance_variable_get(:@ctx) : ctx
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
##
|
|
31
|
+
# @abstract
|
|
32
|
+
# @param [LLM::Message] message
|
|
33
|
+
# The message to transform
|
|
34
|
+
# @param [Hash] opts
|
|
35
|
+
# Per-call options
|
|
36
|
+
# @return [LLM::Message]
|
|
37
|
+
def call(message:, **opts)
|
|
38
|
+
raise NotImplementedError
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
private
|
|
42
|
+
|
|
43
|
+
##
|
|
44
|
+
# @return [LLM::Stream]
|
|
45
|
+
def stream
|
|
46
|
+
@ctx.params[:stream]
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
##
|
|
50
|
+
# @return [LLM::Buffer]
|
|
51
|
+
def messages
|
|
52
|
+
@ctx.messages
|
|
53
|
+
end
|
|
54
|
+
end
|
|
55
|
+
end
|
data/lib/llm/version.rb
CHANGED
data/lib/llm.rb
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
#
|
|
8
8
|
# @example The three-step workflow
|
|
9
9
|
# require "llm"
|
|
10
|
-
# llm = LLM.deepseek(key: ENV["KEY"])
|
|
10
|
+
# llm = LLM.deepseek(key: ENV["KEY"]) # 1. pick a provider
|
|
11
11
|
# agent = LLM::Agent.new(llm, stream: $stdout) # 2. create an agent
|
|
12
12
|
# agent.talk "Hello world" # 3. talk to it
|
|
13
13
|
#
|
|
@@ -17,6 +17,7 @@ module LLM
|
|
|
17
17
|
require "stringio"
|
|
18
18
|
require "securerandom"
|
|
19
19
|
require_relative "llm/compactor"
|
|
20
|
+
require_relative "llm/transformer"
|
|
20
21
|
require_relative "llm/json_adapter"
|
|
21
22
|
require_relative "llm/tracer"
|
|
22
23
|
require_relative "llm/error"
|
|
@@ -40,7 +41,7 @@ module LLM
|
|
|
40
41
|
require_relative "llm/stream"
|
|
41
42
|
require_relative "llm/provider"
|
|
42
43
|
require_relative "llm/context"
|
|
43
|
-
require_relative "llm/
|
|
44
|
+
require_relative "llm/guard"
|
|
44
45
|
require_relative "llm/agent"
|
|
45
46
|
require_relative "llm/buffer"
|
|
46
47
|
require_relative "llm/function"
|
|
@@ -218,6 +219,15 @@ module LLM
|
|
|
218
219
|
LLM::ZAI.new(**)
|
|
219
220
|
end
|
|
220
221
|
|
|
222
|
+
##
|
|
223
|
+
# @param key (see LLM::Moonshot#initialize)
|
|
224
|
+
# @param host (see LLM::Moonshot#initialize)
|
|
225
|
+
# @return (see LLM::Moonshot#initialize)
|
|
226
|
+
def moonshot(**)
|
|
227
|
+
lock(:require) { require_relative "llm/providers/moonshot" unless defined?(LLM::Moonshot) }
|
|
228
|
+
LLM::Moonshot.new(**)
|
|
229
|
+
end
|
|
230
|
+
|
|
221
231
|
##
|
|
222
232
|
# @param [Hash] opts
|
|
223
233
|
# MCP client options
|
data/llm.gemspec
CHANGED
|
@@ -12,7 +12,7 @@ Gem::Specification.new do |spec|
|
|
|
12
12
|
spec.description = <<~DESCRIPTION
|
|
13
13
|
llm.rb is an advanced runtime for building capable AI applications
|
|
14
14
|
on CRuby. By default it has zero runtime dependencies although certain
|
|
15
|
-
functionality
|
|
15
|
+
functionality (such as ActiveRecord support) require
|
|
16
16
|
optional dependencies that are opt-in.
|
|
17
17
|
DESCRIPTION
|
|
18
18
|
|
|
@@ -30,9 +30,16 @@ DESCRIPTION
|
|
|
30
30
|
"lib/*.rb", "lib/**/*.rb",
|
|
31
31
|
"data/*.json", "CHANGELOG.md",
|
|
32
32
|
"resources/deepdive.md",
|
|
33
|
-
"
|
|
33
|
+
"resources/deepdive/*/*.md",
|
|
34
|
+
"llm.gemspec", "bin/llm.rb"
|
|
34
35
|
]
|
|
36
|
+
spec.executables = ["llm.rb"]
|
|
35
37
|
spec.require_paths = ["lib"]
|
|
38
|
+
spec.post_install_message = "\n" \
|
|
39
|
+
"Learn more about llm.rb by reading the deepdive:" \
|
|
40
|
+
"\n" \
|
|
41
|
+
"https://r.uby.dev/llm/deepdive" \
|
|
42
|
+
"\n\n"
|
|
36
43
|
|
|
37
44
|
spec.add_development_dependency "webmock", "~> 3.24.0"
|
|
38
45
|
spec.add_development_dependency "yard", "~> 0.9.37"
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
|
|
2
|
+
## Cancellation
|
|
3
|
+
|
|
4
|
+
### Introduction
|
|
5
|
+
|
|
6
|
+
#### Overview
|
|
7
|
+
|
|
8
|
+
Cancellation lets you abort a model request mid-stream and interrupt
|
|
9
|
+
any tools that are currently executing. The user changes their mind.
|
|
10
|
+
The model goes off course. A tool hangs. In all three cases,
|
|
11
|
+
cancellation stops the work and reclaims the tokens.
|
|
12
|
+
|
|
13
|
+
#### How it works
|
|
14
|
+
|
|
15
|
+
When you want to cancel an active request or tool call, call
|
|
16
|
+
[`LLM::Agent#cancel!`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#cancel!)
|
|
17
|
+
or
|
|
18
|
+
[`LLM::Context#cancel!`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#cancel!)
|
|
19
|
+
from any thread. Two things
|
|
20
|
+
happen at once:
|
|
21
|
+
|
|
22
|
+
[`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
|
|
23
|
+
is raised on the thread where
|
|
24
|
+
[`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
|
|
25
|
+
or
|
|
26
|
+
[`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk)
|
|
27
|
+
is running, so the caller
|
|
28
|
+
can rescue it and know the request was cancelled.
|
|
29
|
+
|
|
30
|
+
At the same time,
|
|
31
|
+
[`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
|
|
32
|
+
is raised on every active tool.
|
|
33
|
+
A tool running in a thread gets it on that thread. A tool in a
|
|
34
|
+
fiber gets it on that fiber. A tool in a forked process gets it
|
|
35
|
+
via a message over the control channel. Pending tools (not yet
|
|
36
|
+
started) are cancelled through
|
|
37
|
+
[`LLM::Function#cancel`](https://r.uby.dev/api-docs/llm.rb/LLM/Function.html#cancel)
|
|
38
|
+
without ever being executed.
|
|
39
|
+
|
|
40
|
+
The transport layer also cancels the in-flight HTTP request.
|
|
41
|
+
|
|
42
|
+
```ruby
|
|
43
|
+
require "llm"
|
|
44
|
+
|
|
45
|
+
llm = LLM.deepseek(key: ENV["DEEPSEEK_SECRET"])
|
|
46
|
+
agent = LLM::Agent.new(llm)
|
|
47
|
+
queue = Queue.new
|
|
48
|
+
|
|
49
|
+
Thread.new do
|
|
50
|
+
queue.push(nil)
|
|
51
|
+
sleep(2)
|
|
52
|
+
agent.cancel!
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
begin
|
|
56
|
+
queue.pop
|
|
57
|
+
agent.talk "write me a very long poem", stream: $stdout
|
|
58
|
+
rescue LLM::Interrupt
|
|
59
|
+
puts "request cancelled!"
|
|
60
|
+
end
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
#### Why would I use it?
|
|
64
|
+
|
|
65
|
+
Cancellation prevents wasted time and tokens when the model goes
|
|
66
|
+
off course, the user changes their mind, or a tool hangs. A forked
|
|
67
|
+
tool that enters an infinite loop would run forever without it.
|
|
68
|
+
|
|
69
|
+
#### Notes
|
|
70
|
+
|
|
71
|
+
The `:ractor` strategy delivers the interrupt through ractor
|
|
72
|
+
message passing. The `:fork` strategy delivers it via a message
|
|
73
|
+
over the xchan control channel. All other strategies raise the
|
|
74
|
+
exception directly on the executing thread or fiber.
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
|
|
2
|
+
## Compaction
|
|
3
|
+
|
|
4
|
+
### Introduction
|
|
5
|
+
|
|
6
|
+
#### Overview
|
|
7
|
+
|
|
8
|
+
Long-running conversations consume tokens. Without intervention,
|
|
9
|
+
every turn pushes toward the model's context window limit. Compaction
|
|
10
|
+
drops old messages to keep the conversation alive. The runtime runs
|
|
11
|
+
a compactor automatically before each
|
|
12
|
+
[`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
|
|
13
|
+
call, trimming the oldest messages when the conversation exceeds a configured size.
|
|
14
|
+
This keeps the context window healthy without manual intervention.
|
|
15
|
+
|
|
16
|
+
#### How it works
|
|
17
|
+
|
|
18
|
+
Compactors run automatically before each
|
|
19
|
+
[`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
|
|
20
|
+
call.
|
|
21
|
+
[`LLM::Compactor::Truncate`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Truncate.html)
|
|
22
|
+
strategy drops the oldest messages, keeping only the N most recent.
|
|
23
|
+
It preserves tool call/return pairs so the conversation never
|
|
24
|
+
contains an orphaned result.
|
|
25
|
+
|
|
26
|
+
The `keep:` parameter accepts an integer count or a percentage
|
|
27
|
+
string like `"80%"`. The default compactor is
|
|
28
|
+
[`LLM::Compactor::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Null.html),
|
|
29
|
+
which does nothing. A compactor can also be used standalone:
|
|
30
|
+
|
|
31
|
+
```ruby
|
|
32
|
+
ctx = LLM::Context.new(
|
|
33
|
+
llm,
|
|
34
|
+
compactor: LLM::Compactor::Truncate,
|
|
35
|
+
compactor_options: {keep: 64}
|
|
36
|
+
)
|
|
37
|
+
|
|
38
|
+
agent = LLM::Agent.new(
|
|
39
|
+
llm,
|
|
40
|
+
compactor: LLM::Compactor::Truncate,
|
|
41
|
+
compactor_options: {keep: 128}
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
compactor = LLM::Compactor::Truncate.new(agent)
|
|
45
|
+
compactor.call(keep: 200)
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
##### The `/compact` command
|
|
49
|
+
|
|
50
|
+
The REPL provides a `/compact` command that accepts a count or
|
|
51
|
+
percentage:
|
|
52
|
+
|
|
53
|
+
```ruby
|
|
54
|
+
# /compact # keep last 128 messages
|
|
55
|
+
# /compact 50 # keep last 50 messages
|
|
56
|
+
# /compact 75% # keep approximately 75% of messages
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
#### Why would I use it?
|
|
60
|
+
|
|
61
|
+
Without compaction, the provider eventually rejects requests because
|
|
62
|
+
the context window is full. Compaction keeps the conversation alive
|
|
63
|
+
by discarding old messages before they cause a problem.
|
|
64
|
+
|
|
65
|
+
Long-running agents that span hundreds of turns need this. A bug
|
|
66
|
+
investigation that bounces back and forth between diagnosis and
|
|
67
|
+
fix cannot fit every exchange in memory. Compaction trims the
|
|
68
|
+
unimportant parts and keeps the conversation alive.
|
|
69
|
+
|
|
70
|
+
#### Notes
|
|
71
|
+
|
|
72
|
+
The Truncate strategy is fast and has no dependencies. It operates
|
|
73
|
+
entirely in memory with a single pass over the message list. The
|
|
74
|
+
trade-off is that dropped messages are gone, so information may be
|
|
75
|
+
lost. Set `keep` to a higher number to retain more context.
|
|
76
|
+
|
|
77
|
+
Both strategies emit
|
|
78
|
+
[`LLM::Stream#on_compaction`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_compaction)
|
|
79
|
+
and
|
|
80
|
+
[`LLM::Stream#on_compaction_finish`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_compaction_finish)
|
|
81
|
+
stream callbacks so the UI can show progress. The context's
|
|
82
|
+
[`LLM::Context#compacted?`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#compacted?)
|
|
83
|
+
flag is `true` between compaction and the next model response.
|