llm.rb 13.0.0 → 14.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (106) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +505 -14
  3. data/README.md +484 -50
  4. data/bin/llm.rb +148 -0
  5. data/data/anthropic.json +206 -263
  6. data/data/bedrock.json +2138 -1860
  7. data/data/deepinfra.json +1003 -624
  8. data/data/deepseek.json +38 -34
  9. data/data/google.json +1079 -371
  10. data/data/mistral.json +448 -368
  11. data/data/moonshot.json +384 -0
  12. data/data/openai.json +974 -1343
  13. data/data/xai.json +154 -126
  14. data/data/zai.json +191 -191
  15. data/lib/llm/agent.rb +123 -20
  16. data/lib/llm/context.rb +71 -88
  17. data/lib/llm/cost.rb +23 -17
  18. data/lib/llm/error.rb +0 -8
  19. data/lib/llm/function/array.rb +3 -3
  20. data/lib/llm/function/async/task.rb +2 -0
  21. data/lib/llm/function/fiber/task.rb +2 -0
  22. data/lib/llm/function/fork/task.rb +2 -0
  23. data/lib/llm/function/ractor/task.rb +2 -0
  24. data/lib/llm/function/sequential/group.rb +4 -1
  25. data/lib/llm/function/sequential/task.rb +1 -1
  26. data/lib/llm/function/task.rb +4 -0
  27. data/lib/llm/function/thread/task.rb +2 -0
  28. data/lib/llm/function.rb +33 -6
  29. data/lib/llm/guard/loop.rb +89 -0
  30. data/lib/llm/guard/null.rb +19 -0
  31. data/lib/llm/guard.rb +61 -0
  32. data/lib/llm/provider.rb +36 -0
  33. data/lib/llm/providers/anthropic/stream_parser.rb +1 -0
  34. data/lib/llm/providers/anthropic.rb +2 -9
  35. data/lib/llm/providers/bedrock/request_adapter.rb +1 -1
  36. data/lib/llm/providers/bedrock/stream_parser.rb +1 -0
  37. data/lib/llm/providers/bedrock.rb +1 -8
  38. data/lib/llm/providers/google/stream_parser.rb +1 -0
  39. data/lib/llm/providers/google.rb +1 -8
  40. data/lib/llm/providers/mistral.rb +1 -1
  41. data/lib/llm/providers/moonshot.rb +76 -0
  42. data/lib/llm/providers/ollama.rb +2 -9
  43. data/lib/llm/providers/openai/responses/stream_parser.rb +1 -0
  44. data/lib/llm/providers/openai/responses.rb +7 -9
  45. data/lib/llm/providers/openai/stream_parser.rb +1 -0
  46. data/lib/llm/providers/openai.rb +4 -11
  47. data/lib/llm/repl/bar.rb +4 -3
  48. data/lib/llm/repl/{transcript.rb → buffer.rb} +69 -29
  49. data/lib/llm/repl/color.rb +78 -0
  50. data/lib/llm/repl/command.rb +12 -5
  51. data/lib/llm/repl/commands/compact.rb +2 -2
  52. data/lib/llm/repl/commands/help.rb +3 -5
  53. data/lib/llm/repl/input/char.rb +46 -0
  54. data/lib/llm/repl/input/row.rb +39 -0
  55. data/lib/llm/repl/input.rb +251 -66
  56. data/lib/llm/repl/markdown/table.rb +11 -3
  57. data/lib/llm/repl/markdown.rb +34 -8
  58. data/lib/llm/repl/node.rb +37 -0
  59. data/lib/llm/repl/status.rb +42 -7
  60. data/lib/llm/repl/stream.rb +18 -6
  61. data/lib/llm/repl/walker.rb +3 -2
  62. data/lib/llm/repl/window.rb +54 -35
  63. data/lib/llm/repl.rb +74 -32
  64. data/lib/llm/skill.rb +20 -4
  65. data/lib/llm/stream.rb +8 -7
  66. data/lib/llm/tool.rb +29 -0
  67. data/lib/llm/tools/{swap_text.rb → edit-file.rb} +3 -3
  68. data/lib/llm/tools/git.rb +3 -0
  69. data/lib/llm/tools/mkdir.rb +3 -0
  70. data/lib/llm/tools/rg.rb +3 -0
  71. data/lib/llm/tools/ruby.rb +46 -0
  72. data/lib/llm/tools/shell.rb +3 -0
  73. data/lib/llm/tracer/pretty_logger.rb +127 -0
  74. data/lib/llm/tracer.rb +1 -0
  75. data/lib/llm/transformer/null.rb +21 -0
  76. data/lib/llm/transformer.rb +55 -0
  77. data/lib/llm/version.rb +1 -1
  78. data/lib/llm.rb +12 -2
  79. data/llm.gemspec +9 -2
  80. data/resources/deepdive/advanced/cancellation.md +74 -0
  81. data/resources/deepdive/advanced/compaction.md +83 -0
  82. data/resources/deepdive/advanced/context.md +267 -0
  83. data/resources/deepdive/advanced/guard.md +371 -0
  84. data/resources/deepdive/advanced/tracer.md +180 -0
  85. data/resources/deepdive/advanced/transformer.md +67 -0
  86. data/resources/deepdive/advanced/transports.md +45 -0
  87. data/resources/deepdive/everything_else/audio.md +122 -0
  88. data/resources/deepdive/everything_else/cost.md +99 -0
  89. data/resources/deepdive/everything_else/images.md +89 -0
  90. data/resources/deepdive/everything_else/object.md +108 -0
  91. data/resources/deepdive/everything_else/ocr.md +48 -0
  92. data/resources/deepdive/fundamentals/agents.md +202 -0
  93. data/resources/deepdive/fundamentals/builtin_tools.md +191 -0
  94. data/resources/deepdive/fundamentals/concurrency.md +104 -0
  95. data/resources/deepdive/fundamentals/database.md +449 -0
  96. data/resources/deepdive/fundamentals/embeddings.md +157 -0
  97. data/resources/deepdive/fundamentals/repl.md +87 -0
  98. data/resources/deepdive/fundamentals/schema.md +61 -0
  99. data/resources/deepdive/fundamentals/skills.md +106 -0
  100. data/resources/deepdive/fundamentals/stream.md +110 -0
  101. data/resources/deepdive/fundamentals/tools.md +265 -0
  102. data/resources/deepdive/protocols/a2a.md +106 -0
  103. data/resources/deepdive/protocols/mcp.md +111 -0
  104. data/resources/deepdive.md +58 -1792
  105. metadata +51 -7
  106. data/lib/llm/loop_guard.rb +0 -107
@@ -0,0 +1,46 @@
1
+ # frozen_string_literal: true
2
+
3
+ class LLM::Tool
4
+ ##
5
+ # The {LLM::Tool::Ruby LLM::Tool::Ruby} class implements
6
+ # a tool that can execute an arbitrary string of Ruby code
7
+ # and return the result. It uses `test-cmd.rb` under the
8
+ # hood for process management.
9
+ class Ruby < self
10
+ require_relative "utils"
11
+ include Utils
12
+
13
+ name "ruby"
14
+ description "runs a string of ruby code"
15
+ parameter :code, String, "a string of ruby code"
16
+ parameter :timeout, Integer, "maximum runtime before timeout"
17
+ required %i[code]
18
+ defaults timeout: 15
19
+
20
+ ##
21
+ # @param [String] code
22
+ # Ruby code
23
+ # @param [Integer] timeout
24
+ # Runtime timeout
25
+ def call(code:, timeout: 15)
26
+ command = spawn(code:)
27
+ wait(command:, timeout:)
28
+ {ok: command.success?, stdout: command.stdout, stderr: command.stderr}
29
+ rescue LLM::Interrupt
30
+ command.kill! if command&.running?
31
+ raise
32
+ end
33
+
34
+ private
35
+
36
+ def spawn(code:)
37
+ Command
38
+ .new(RbConfig.ruby)
39
+ .argv("-e", code)
40
+ end
41
+
42
+ require "rbconfig"
43
+ LLM.require "test-cmd.rb", "~> 1.1"
44
+ Command = Test::Cmd
45
+ end
46
+ end
@@ -31,6 +31,9 @@ class LLM::Tool
31
31
  command = spawn(name:, arguments:)
32
32
  wait(command:, timeout:)
33
33
  {ok: command.success?, stdout: command.stdout, stderr: command.stderr}
34
+ rescue LLM::Interrupt
35
+ command.kill! if command&.running?
36
+ raise
34
37
  end
35
38
 
36
39
  private
@@ -0,0 +1,127 @@
1
+ # frozen_string_literal: true
2
+
3
+ module LLM
4
+ ##
5
+ # The {LLM::Tracer::PrettyLogger LLM::Tracer::PrettyLogger} class
6
+ # writes human-readable request and tool call logs to a console
7
+ # or file. Each event is a single line with the relevant context
8
+ # inline, no structured JSON.
9
+ #
10
+ # @example
11
+ # llm = LLM.openai(key: ENV["KEY"])
12
+ # llm.tracer = LLM::Tracer::PrettyLogger.new(llm)
13
+ #
14
+ # @example Writing to a file
15
+ # llm.tracer = LLM::Tracer::PrettyLogger.new(llm, io: File.open("log.txt", "a"))
16
+ class Tracer::PrettyLogger < Tracer
17
+ ##
18
+ # @param (see LLM::Tracer#initialize)
19
+ def initialize(provider, options = {})
20
+ super
21
+ setup!(**options)
22
+ end
23
+
24
+ ##
25
+ # @param (see LLM::Tracer#on_request_start)
26
+ # @return [void]
27
+ def on_request_start(operation:, model: nil, **)
28
+ @start = Process.clock_gettime(Process::CLOCK_MONOTONIC)
29
+ name = operation == "chat" ? "chat" : operation
30
+ @io.puts "#{timestamp} #{provider_name} #{name} (#{model || "default"})"
31
+ end
32
+
33
+ ##
34
+ # @param (see LLM::Tracer#on_request_finish)
35
+ # @return [void]
36
+ def on_request_finish(operation:, res:, model: nil, **)
37
+ elapsed = @start ? (Process.clock_gettime(Process::CLOCK_MONOTONIC) - @start).round(2) : nil
38
+ tokens = format_tokens(res)
39
+ name = operation == "chat" ? "chat" : operation
40
+ parts = ["#{timestamp} #{provider_name} #{name} done"]
41
+ parts << tokens if tokens
42
+ parts << "#{elapsed}s" if elapsed
43
+ @io.puts parts.join(", ")
44
+ end
45
+
46
+ ##
47
+ # @param (see LLM::Tracer#on_request_error)
48
+ # @return [void]
49
+ def on_request_error(ex:, **)
50
+ @io.puts "#{timestamp} #{provider_name} error #{ex.class}: #{ex.message}"
51
+ end
52
+
53
+ ##
54
+ # @param (see LLM::Tracer#on_tool_start)
55
+ # @return [void]
56
+ def on_tool_start(id:, name:, arguments:, **)
57
+ @io.puts "#{timestamp} #{name}(#{format_arguments(arguments)})"
58
+ end
59
+
60
+ ##
61
+ # @param (see LLM::Tracer#on_tool_finish)
62
+ # @return [void]
63
+ def on_tool_finish(result:, **)
64
+ @io.puts "#{timestamp} #{result.name} -> #{format_value(result.value)}"
65
+ end
66
+
67
+ ##
68
+ # @param (see LLM::Tracer#on_tool_error)
69
+ # @return [void]
70
+ def on_tool_error(ex:, **)
71
+ @io.puts "#{timestamp} tool error #{ex.class}: #{ex.message}"
72
+ end
73
+
74
+ private
75
+
76
+ def setup!(io: $stderr)
77
+ @io = io
78
+ @start = nil
79
+ end
80
+
81
+ def timestamp
82
+ Time.now.strftime("%H:%M:%S")
83
+ end
84
+
85
+ def format_tokens(res)
86
+ usage = res.usage
87
+ if usage.input_tokens and usage.output_tokens
88
+ "in=#{usage.input_tokens} out=#{usage.output_tokens}"
89
+ elsif usage.input_tokens
90
+ "in=#{usage.input_tokens}"
91
+ elsif usage.output_tokens
92
+ "out=#{usage.output_tokens}"
93
+ end
94
+ end
95
+
96
+ def format_arguments(args, max: 50)
97
+ return "" unless args
98
+ case args
99
+ when Hash, LLM::Object
100
+ result = args.map { |k, v| "#{k}: #{format_value(v)}" }.join(", ")
101
+ when Array
102
+ result = args.map { |v| format_value(v) }.join(", ")
103
+ else
104
+ result = args.inspect
105
+ end
106
+ result.size > max ? "#{result[0...max - 1]}..." : result
107
+ end
108
+
109
+ def format_value(value, max: 18)
110
+ case value
111
+ when String
112
+ value.size > max ? "#{value[0...max]}...".inspect : value.inspect
113
+ when Array
114
+ items = value.take(2).map { format_value(_1, max: 10) }
115
+ items << "..." if value.size > 2
116
+ "[#{items.join(", ")}]"
117
+ when Hash
118
+ "{...}"
119
+ when nil
120
+ "nil"
121
+ else
122
+ str = value.inspect
123
+ str.size > max ? "#{str[0...max]}..." : str
124
+ end
125
+ end
126
+ end
127
+ end
data/lib/llm/tracer.rb CHANGED
@@ -12,6 +12,7 @@ module LLM
12
12
  require_relative "tracer/logger"
13
13
  require_relative "tracer/telemetry"
14
14
  require_relative "tracer/null"
15
+ require_relative "tracer/pretty_logger"
15
16
 
16
17
  ##
17
18
  # @return [LLM::Provider]
@@ -0,0 +1,21 @@
1
+ # frozen_string_literal: true
2
+
3
+ class LLM::Transformer
4
+ ##
5
+ # An {LLM::Transformer::Null LLM::Transformer::Null} is a
6
+ # transformer that does nothing. It is used as the default when
7
+ # no transformer strategy is configured.
8
+ #
9
+ # It returns the given message unchanged.
10
+ class Null < self
11
+ ##
12
+ # @param [LLM::Message] message
13
+ # The message to transform
14
+ # @param [Hash] opts
15
+ # Ignored
16
+ # @return [LLM::Message]
17
+ def call(message:, **opts)
18
+ message
19
+ end
20
+ end
21
+ end
@@ -0,0 +1,55 @@
1
+ # frozen_string_literal: true
2
+
3
+ module LLM
4
+ ##
5
+ # {LLM::Transformer LLM::Transformer} is the superclass for
6
+ # message transformers in llm.rb.
7
+ #
8
+ # A transformer is bound to a context and rewrites a single
9
+ # message before it is sent to the provider. Each subclass
10
+ # implements a different transformation: it takes a message in
11
+ # {#call} and returns a message. {LLM::Transformer::Null} is a
12
+ # no-op (the default).
13
+ #
14
+ # A transformer may mutate the message in place or return a new
15
+ # one. Either way, the returned message is what gets sent.
16
+ class Transformer
17
+ require_relative "transformer/null"
18
+
19
+ ##
20
+ # @return [LLM::Context]
21
+ attr_reader :ctx
22
+
23
+ ##
24
+ # @param ctx [LLM::Context, LLM::Agent]
25
+ # @return [LLM::Transformer]
26
+ def initialize(ctx)
27
+ @ctx = LLM::Agent === ctx ? ctx.instance_variable_get(:@ctx) : ctx
28
+ end
29
+
30
+ ##
31
+ # @abstract
32
+ # @param [LLM::Message] message
33
+ # The message to transform
34
+ # @param [Hash] opts
35
+ # Per-call options
36
+ # @return [LLM::Message]
37
+ def call(message:, **opts)
38
+ raise NotImplementedError
39
+ end
40
+
41
+ private
42
+
43
+ ##
44
+ # @return [LLM::Stream]
45
+ def stream
46
+ @ctx.params[:stream]
47
+ end
48
+
49
+ ##
50
+ # @return [LLM::Buffer]
51
+ def messages
52
+ @ctx.messages
53
+ end
54
+ end
55
+ end
data/lib/llm/version.rb CHANGED
@@ -1,5 +1,5 @@
1
1
  # frozen_string_literal: true
2
2
 
3
3
  module LLM
4
- VERSION = "13.0.0"
4
+ VERSION = "14.0.0"
5
5
  end
data/lib/llm.rb CHANGED
@@ -7,7 +7,7 @@
7
7
  #
8
8
  # @example The three-step workflow
9
9
  # require "llm"
10
- # llm = LLM.deepseek(key: ENV["KEY"]) # 1. pick a provider
10
+ # llm = LLM.deepseek(key: ENV["KEY"]) # 1. pick a provider
11
11
  # agent = LLM::Agent.new(llm, stream: $stdout) # 2. create an agent
12
12
  # agent.talk "Hello world" # 3. talk to it
13
13
  #
@@ -17,6 +17,7 @@ module LLM
17
17
  require "stringio"
18
18
  require "securerandom"
19
19
  require_relative "llm/compactor"
20
+ require_relative "llm/transformer"
20
21
  require_relative "llm/json_adapter"
21
22
  require_relative "llm/tracer"
22
23
  require_relative "llm/error"
@@ -40,7 +41,7 @@ module LLM
40
41
  require_relative "llm/stream"
41
42
  require_relative "llm/provider"
42
43
  require_relative "llm/context"
43
- require_relative "llm/loop_guard"
44
+ require_relative "llm/guard"
44
45
  require_relative "llm/agent"
45
46
  require_relative "llm/buffer"
46
47
  require_relative "llm/function"
@@ -218,6 +219,15 @@ module LLM
218
219
  LLM::ZAI.new(**)
219
220
  end
220
221
 
222
+ ##
223
+ # @param key (see LLM::Moonshot#initialize)
224
+ # @param host (see LLM::Moonshot#initialize)
225
+ # @return (see LLM::Moonshot#initialize)
226
+ def moonshot(**)
227
+ lock(:require) { require_relative "llm/providers/moonshot" unless defined?(LLM::Moonshot) }
228
+ LLM::Moonshot.new(**)
229
+ end
230
+
221
231
  ##
222
232
  # @param [Hash] opts
223
233
  # MCP client options
data/llm.gemspec CHANGED
@@ -12,7 +12,7 @@ Gem::Specification.new do |spec|
12
12
  spec.description = <<~DESCRIPTION
13
13
  llm.rb is an advanced runtime for building capable AI applications
14
14
  on CRuby. By default it has zero runtime dependencies although certain
15
- functionality &ndash; such as ActiveRecord support &ndash; require
15
+ functionality (such as ActiveRecord support) require
16
16
  optional dependencies that are opt-in.
17
17
  DESCRIPTION
18
18
 
@@ -30,9 +30,16 @@ DESCRIPTION
30
30
  "lib/*.rb", "lib/**/*.rb",
31
31
  "data/*.json", "CHANGELOG.md",
32
32
  "resources/deepdive.md",
33
- "llm.gemspec"
33
+ "resources/deepdive/*/*.md",
34
+ "llm.gemspec", "bin/llm.rb"
34
35
  ]
36
+ spec.executables = ["llm.rb"]
35
37
  spec.require_paths = ["lib"]
38
+ spec.post_install_message = "\n" \
39
+ "Learn more about llm.rb by reading the deepdive:" \
40
+ "\n" \
41
+ "https://r.uby.dev/llm/deepdive" \
42
+ "\n\n"
36
43
 
37
44
  spec.add_development_dependency "webmock", "~> 3.24.0"
38
45
  spec.add_development_dependency "yard", "~> 0.9.37"
@@ -0,0 +1,74 @@
1
+
2
+ ## Cancellation
3
+
4
+ ### Introduction
5
+
6
+ #### Overview
7
+
8
+ Cancellation lets you abort a model request mid-stream and interrupt
9
+ any tools that are currently executing. The user changes their mind.
10
+ The model goes off course. A tool hangs. In all three cases,
11
+ cancellation stops the work and reclaims the tokens.
12
+
13
+ #### How it works
14
+
15
+ When you want to cancel an active request or tool call, call
16
+ [`LLM::Agent#cancel!`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#cancel!)
17
+ or
18
+ [`LLM::Context#cancel!`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#cancel!)
19
+ from any thread. Two things
20
+ happen at once:
21
+
22
+ [`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
23
+ is raised on the thread where
24
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
25
+ or
26
+ [`LLM::Context#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#talk)
27
+ is running, so the caller
28
+ can rescue it and know the request was cancelled.
29
+
30
+ At the same time,
31
+ [`LLM::Interrupt`](https://r.uby.dev/api-docs/llm.rb/LLM/Interrupt.html)
32
+ is raised on every active tool.
33
+ A tool running in a thread gets it on that thread. A tool in a
34
+ fiber gets it on that fiber. A tool in a forked process gets it
35
+ via a message over the control channel. Pending tools (not yet
36
+ started) are cancelled through
37
+ [`LLM::Function#cancel`](https://r.uby.dev/api-docs/llm.rb/LLM/Function.html#cancel)
38
+ without ever being executed.
39
+
40
+ The transport layer also cancels the in-flight HTTP request.
41
+
42
+ ```ruby
43
+ require "llm"
44
+
45
+ llm = LLM.deepseek(key: ENV["DEEPSEEK_SECRET"])
46
+ agent = LLM::Agent.new(llm)
47
+ queue = Queue.new
48
+
49
+ Thread.new do
50
+ queue.push(nil)
51
+ sleep(2)
52
+ agent.cancel!
53
+ end
54
+
55
+ begin
56
+ queue.pop
57
+ agent.talk "write me a very long poem", stream: $stdout
58
+ rescue LLM::Interrupt
59
+ puts "request cancelled!"
60
+ end
61
+ ```
62
+
63
+ #### Why would I use it?
64
+
65
+ Cancellation prevents wasted time and tokens when the model goes
66
+ off course, the user changes their mind, or a tool hangs. A forked
67
+ tool that enters an infinite loop would run forever without it.
68
+
69
+ #### Notes
70
+
71
+ The `:ractor` strategy delivers the interrupt through ractor
72
+ message passing. The `:fork` strategy delivers it via a message
73
+ over the xchan control channel. All other strategies raise the
74
+ exception directly on the executing thread or fiber.
@@ -0,0 +1,83 @@
1
+
2
+ ## Compaction
3
+
4
+ ### Introduction
5
+
6
+ #### Overview
7
+
8
+ Long-running conversations consume tokens. Without intervention,
9
+ every turn pushes toward the model's context window limit. Compaction
10
+ drops old messages to keep the conversation alive. The runtime runs
11
+ a compactor automatically before each
12
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
13
+ call, trimming the oldest messages when the conversation exceeds a configured size.
14
+ This keeps the context window healthy without manual intervention.
15
+
16
+ #### How it works
17
+
18
+ Compactors run automatically before each
19
+ [`LLM::Agent#talk`](https://r.uby.dev/api-docs/llm.rb/LLM/Agent.html#talk)
20
+ call.
21
+ [`LLM::Compactor::Truncate`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Truncate.html)
22
+ strategy drops the oldest messages, keeping only the N most recent.
23
+ It preserves tool call/return pairs so the conversation never
24
+ contains an orphaned result.
25
+
26
+ The `keep:` parameter accepts an integer count or a percentage
27
+ string like `"80%"`. The default compactor is
28
+ [`LLM::Compactor::Null`](https://r.uby.dev/api-docs/llm.rb/LLM/Compactor/Null.html),
29
+ which does nothing. A compactor can also be used standalone:
30
+
31
+ ```ruby
32
+ ctx = LLM::Context.new(
33
+ llm,
34
+ compactor: LLM::Compactor::Truncate,
35
+ compactor_options: {keep: 64}
36
+ )
37
+
38
+ agent = LLM::Agent.new(
39
+ llm,
40
+ compactor: LLM::Compactor::Truncate,
41
+ compactor_options: {keep: 128}
42
+ )
43
+
44
+ compactor = LLM::Compactor::Truncate.new(agent)
45
+ compactor.call(keep: 200)
46
+ ```
47
+
48
+ ##### The `/compact` command
49
+
50
+ The REPL provides a `/compact` command that accepts a count or
51
+ percentage:
52
+
53
+ ```ruby
54
+ # /compact # keep last 128 messages
55
+ # /compact 50 # keep last 50 messages
56
+ # /compact 75% # keep approximately 75% of messages
57
+ ```
58
+
59
+ #### Why would I use it?
60
+
61
+ Without compaction, the provider eventually rejects requests because
62
+ the context window is full. Compaction keeps the conversation alive
63
+ by discarding old messages before they cause a problem.
64
+
65
+ Long-running agents that span hundreds of turns need this. A bug
66
+ investigation that bounces back and forth between diagnosis and
67
+ fix cannot fit every exchange in memory. Compaction trims the
68
+ unimportant parts and keeps the conversation alive.
69
+
70
+ #### Notes
71
+
72
+ The Truncate strategy is fast and has no dependencies. It operates
73
+ entirely in memory with a single pass over the message list. The
74
+ trade-off is that dropped messages are gone, so information may be
75
+ lost. Set `keep` to a higher number to retain more context.
76
+
77
+ Both strategies emit
78
+ [`LLM::Stream#on_compaction`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_compaction)
79
+ and
80
+ [`LLM::Stream#on_compaction_finish`](https://r.uby.dev/api-docs/llm.rb/LLM/Stream.html#on_compaction_finish)
81
+ stream callbacks so the UI can show progress. The context's
82
+ [`LLM::Context#compacted?`](https://r.uby.dev/api-docs/llm.rb/LLM/Context.html#compacted?)
83
+ flag is `true` between compaction and the next model response.