omniai-anthropic 3.5.0 → 3.7.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
checksums.yaml CHANGED
@@ -1,7 +1,7 @@
1
1
  ---
2
2
  SHA256:
3
- metadata.gz: fcf24d90210fce6c02f7be928c435515a00e75163590946e3222cf36d424c7c9
4
- data.tar.gz: 0477ad1fdb3454dec2e84db5bcdf044b97eee780892f013ef85348460ac0fb70
3
+ metadata.gz: 20cd61e7e110c59273cb86062c34630d07d66b14a5da7be33cfb11f3a3f29514
4
+ data.tar.gz: 44dfedfa87d0cb07ca36fc155efd5ddb8507eef146202a41e98be1c1bbd0010c
5
5
  SHA512:
6
- metadata.gz: f91432693771d415d5b7375e00781c817e960711a74c66961808b9a2ebf68cc09a3c7ae455ed631666ef8b15643e9afa79962cef23e1334410eb8a352a233aa7
7
- data.tar.gz: b02acec6c970f8e651f465e5fdc07512db9dca9600b787bb42eb2beab866236ec12ea3af626c268ef1734aec67e294892d8b9c2b5048f9c90d516ee7c28633ad
6
+ metadata.gz: 815f8ed94a51cdbe222754218ab865a17227971e6d58d0c39aa6a87425919127af507871bd25263f355d22c5e46c5849c2bc34d6f3a955b6cebc6a06ce944c49
7
+ data.tar.gz: c4303f54ed395240b18c054f1776ac680323c89ba609c9c0c8dba57f5ad0c641d71eaa34a4713d9d6cc3356fdda59d4f020a186bc7b0d6aca5ec66f9bf2fe34d
data/README.md CHANGED
@@ -170,3 +170,21 @@ client.chat("What are the prime factors of 1234567?", model: "claude-sonnet-4-20
170
170
  The thinking content will stream first, followed by the response.
171
171
 
172
172
  [Anthropic API Reference `thinking`](https://docs.anthropic.com/en/docs/build-with-claude/thinking)
173
+
174
+ ### Prompt Caching
175
+
176
+ Prompt caching is opt-in. When enabled, the system prompt (or the last tool, when there is no system prompt) and the last block of the last message are marked with `cache_control`, so each round of a tool-call loop reads the history the previous round wrote.
177
+
178
+ ```ruby
179
+ client.chat(prompt, tools:, cache: true) # 5-minute TTL
180
+ client.chat(prompt, tools:, cache: { ttl: "1h" }) # 1-hour TTL
181
+ ```
182
+
183
+ `usage.input_tokens` reports the whole prompt, including cached tokens. The cache breakdown is on each response's raw usage:
184
+
185
+ ```ruby
186
+ response.response_chain.sum { |r| r.data.dig("usage", "cache_read_input_tokens").to_i }
187
+ response.response_chain.sum { |r| r.data.dig("usage", "cache_creation_input_tokens").to_i }
188
+ ```
189
+
190
+ [Anthropic API Reference `prompt caching`](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching)
@@ -81,10 +81,15 @@ module OmniAI
81
81
 
82
82
  input_tokens = data.dig("usage", "input_tokens")
83
83
  output_tokens = data.dig("usage", "output_tokens")
84
+ # The thinking breakdown arrives only on the final `message_delta`. Carrying it through keeps streamed
85
+ # responses reporting the same `thinking_tokens` as non-streamed ones; without it the key is dropped and
86
+ # every streamed response silently reports no reasoning.
87
+ output_tokens_details = data.dig("usage", "output_tokens_details")
84
88
 
85
89
  @data["usage"] ||= {}
86
90
  @data["usage"]["input_tokens"] = input_tokens if input_tokens
87
91
  @data["usage"]["output_tokens"] = output_tokens if output_tokens
92
+ @data["usage"]["output_tokens_details"] = output_tokens_details if output_tokens_details
88
93
  end
89
94
 
90
95
  # Handler for Type::MESSAGE_STOP
@@ -0,0 +1,37 @@
1
+ # frozen_string_literal: true
2
+
3
+ module OmniAI
4
+ module Anthropic
5
+ class Chat
6
+ # Overrides usage deserialize to read Anthropic's reasoning breakdown.
7
+ module UsageSerializer
8
+ # Anthropic already counts reasoning inside `output_tokens` and reports the breakdown alongside it at
9
+ # `output_tokens_details.thinking_tokens`, so the breakdown is read into `thinking_tokens` and the output
10
+ # count is left exactly as reported. Adding the two together would double count.
11
+ #
12
+ # When streaming, the breakdown arrives only on the final `message_delta`.
13
+ #
14
+ # @param data [Hash]
15
+ # @return [OmniAI::Chat::Usage]
16
+ def self.deserialize(data, *)
17
+ # Deserialize without a context so the generic flat parse runs rather than recursing into this method.
18
+ usage = OmniAI::Chat::Usage.deserialize(data)
19
+
20
+ # Only overwrite when the vendor container is actually present. A payload produced by `Usage#serialize`
21
+ # carries base's own `thinking_tokens` key and no `output_tokens_details`, so assigning unconditionally
22
+ # would clobber a correctly-parsed value with nil and break the round-trip. `unless nil?` rather than
23
+ # `||=`, so a reported zero from the wire still wins over base's nil.
24
+ thinking_tokens = data.dig("output_tokens_details", "thinking_tokens")
25
+ usage.thinking_tokens = thinking_tokens unless thinking_tokens.nil?
26
+
27
+ # Anthropic's `input_tokens` excludes cache reads and writes; adding them back keeps it the whole prompt, as
28
+ # for every other provider. A base-serialized payload has neither key and is already whole.
29
+ cache_tokens = data.values_at("cache_creation_input_tokens", "cache_read_input_tokens").compact
30
+ usage.input_tokens = (usage.input_tokens || 0) + cache_tokens.sum if cache_tokens.any?
31
+
32
+ usage
33
+ end
34
+ end
35
+ end
36
+ end
37
+ end
@@ -87,6 +87,8 @@ module OmniAI
87
87
 
88
88
  context.serializers[:thinking] = ThinkingSerializer.method(:serialize)
89
89
  context.deserializers[:thinking] = ThinkingSerializer.method(:deserialize)
90
+
91
+ context.deserializers[:usage] = UsageSerializer.method(:deserialize)
90
92
  end
91
93
 
92
94
  # @return [Hash]
@@ -163,19 +165,40 @@ module OmniAI
163
165
  end
164
166
  end
165
167
 
168
+ # When caching, the last block of the last message is marked so each tool-loop round reads the history the
169
+ # previous round wrote.
170
+ #
166
171
  # @return [Array<Hash>]
167
172
  def messages
168
- messages = @prompt.messages.reject(&:system?)
169
- messages.map { |message| message.serialize(context:) }
173
+ messages = @prompt.messages.reject(&:system?).map { |message| message.serialize(context:) }
174
+ return messages unless cache_control && messages.any?
175
+
176
+ *history, last = messages
177
+ history + [last.merge(content: with_cache_control(last[:content]))]
170
178
  end
171
179
 
172
- # @return [String, nil]
180
+ # When caching, a single marked text block. Tools render before system, so this breakpoint covers both.
181
+ #
182
+ # @return [String, Array<Hash>, nil]
173
183
  def system
174
184
  parts = @prompt.messages.filter(&:system?).filter(&:text?).map(&:text)
175
185
  parts << formatting if formatting?
176
186
  return if parts.empty?
177
187
 
178
- parts.join("\n\n")
188
+ text = parts.join("\n\n")
189
+ cache_control ? [{ type: "text", text:, cache_control: }] : text
190
+ end
191
+
192
+ # Translates the opt-in `cache` option to Anthropic's `cache_control`.
193
+ # Example: `cache: true` becomes `{ type: "ephemeral" }` (5-minute TTL)
194
+ # Example: `cache: { ttl: "1h" }` becomes `{ type: "ephemeral", ttl: "1h" }`
195
+ #
196
+ # @return [Hash, nil]
197
+ def cache_control
198
+ case @options[:cache]
199
+ when true then { type: "ephemeral" }
200
+ when Hash then { type: "ephemeral" }.merge(@options[:cache])
201
+ end
179
202
  end
180
203
 
181
204
  # @return [String]
@@ -215,9 +238,25 @@ module OmniAI
215
238
  end
216
239
  end
217
240
 
241
+ # When caching without a system prompt, the last tool carries the breakpoint instead.
242
+ #
218
243
  # @return [Array<Hash>, nil]
219
244
  def tools_payload
220
- @tools.map { |tool| tool.serialize(context:) } if @tools&.any?
245
+ return unless @tools&.any?
246
+
247
+ tools = @tools.map { |tool| tool.serialize(context:) }
248
+ cache_control && system.nil? ? with_cache_control(tools) : tools
249
+ end
250
+
251
+ # Relies on `MessageSerializer` normalizing every content part into a block Hash; a bare String would raise.
252
+ #
253
+ # @param blocks [Array<Hash>]
254
+ # @return [Array<Hash>]
255
+ def with_cache_control(blocks)
256
+ return blocks if blocks.empty?
257
+
258
+ *head, last = blocks
259
+ head + [last.merge(cache_control:)]
221
260
  end
222
261
 
223
262
  # @return [Boolean]
@@ -2,6 +2,6 @@
2
2
 
3
3
  module OmniAI
4
4
  module Anthropic
5
- VERSION = "3.5.0"
5
+ VERSION = "3.7.0"
6
6
  end
7
7
  end
metadata CHANGED
@@ -1,7 +1,7 @@
1
1
  --- !ruby/object:Gem::Specification
2
2
  name: omniai-anthropic
3
3
  version: !ruby/object:Gem::Version
4
- version: 3.5.0
4
+ version: 3.7.0
5
5
  platform: ruby
6
6
  authors:
7
7
  - Kevin Sylvestre
@@ -29,14 +29,14 @@ dependencies:
29
29
  requirements:
30
30
  - - "~>"
31
31
  - !ruby/object:Gem::Version
32
- version: '3.7'
32
+ version: '3.8'
33
33
  type: :runtime
34
34
  prerelease: false
35
35
  version_requirements: !ruby/object:Gem::Requirement
36
36
  requirements:
37
37
  - - "~>"
38
38
  - !ruby/object:Gem::Version
39
- version: '3.7'
39
+ version: '3.8'
40
40
  - !ruby/object:Gem::Dependency
41
41
  name: openssl
42
42
  requirement: !ruby/object:Gem::Requirement
@@ -89,6 +89,7 @@ files:
89
89
  - lib/omniai/anthropic/chat/tool_call_serializer.rb
90
90
  - lib/omniai/anthropic/chat/tool_serializer.rb
91
91
  - lib/omniai/anthropic/chat/url_serializer.rb
92
+ - lib/omniai/anthropic/chat/usage_serializer.rb
92
93
  - lib/omniai/anthropic/client.rb
93
94
  - lib/omniai/anthropic/computer.rb
94
95
  - lib/omniai/anthropic/config.rb