omniai-anthropic 3.5.0 → 3.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +18 -0
- data/lib/omniai/anthropic/chat/stream.rb +5 -0
- data/lib/omniai/anthropic/chat/usage_serializer.rb +37 -0
- data/lib/omniai/anthropic/chat.rb +44 -5
- data/lib/omniai/anthropic/version.rb +1 -1
- metadata +4 -3
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 20cd61e7e110c59273cb86062c34630d07d66b14a5da7be33cfb11f3a3f29514
|
|
4
|
+
data.tar.gz: 44dfedfa87d0cb07ca36fc155efd5ddb8507eef146202a41e98be1c1bbd0010c
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 815f8ed94a51cdbe222754218ab865a17227971e6d58d0c39aa6a87425919127af507871bd25263f355d22c5e46c5849c2bc34d6f3a955b6cebc6a06ce944c49
|
|
7
|
+
data.tar.gz: c4303f54ed395240b18c054f1776ac680323c89ba609c9c0c8dba57f5ad0c641d71eaa34a4713d9d6cc3356fdda59d4f020a186bc7b0d6aca5ec66f9bf2fe34d
|
data/README.md
CHANGED
|
@@ -170,3 +170,21 @@ client.chat("What are the prime factors of 1234567?", model: "claude-sonnet-4-20
|
|
|
170
170
|
The thinking content will stream first, followed by the response.
|
|
171
171
|
|
|
172
172
|
[Anthropic API Reference `thinking`](https://docs.anthropic.com/en/docs/build-with-claude/thinking)
|
|
173
|
+
|
|
174
|
+
### Prompt Caching
|
|
175
|
+
|
|
176
|
+
Prompt caching is opt-in. When enabled, the system prompt (or the last tool, when there is no system prompt) and the last block of the last message are marked with `cache_control`, so each round of a tool-call loop reads the history the previous round wrote.
|
|
177
|
+
|
|
178
|
+
```ruby
|
|
179
|
+
client.chat(prompt, tools:, cache: true) # 5-minute TTL
|
|
180
|
+
client.chat(prompt, tools:, cache: { ttl: "1h" }) # 1-hour TTL
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
`usage.input_tokens` reports the whole prompt, including cached tokens. The cache breakdown is on each response's raw usage:
|
|
184
|
+
|
|
185
|
+
```ruby
|
|
186
|
+
response.response_chain.sum { |r| r.data.dig("usage", "cache_read_input_tokens").to_i }
|
|
187
|
+
response.response_chain.sum { |r| r.data.dig("usage", "cache_creation_input_tokens").to_i }
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
[Anthropic API Reference `prompt caching`](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching)
|
|
@@ -81,10 +81,15 @@ module OmniAI
|
|
|
81
81
|
|
|
82
82
|
input_tokens = data.dig("usage", "input_tokens")
|
|
83
83
|
output_tokens = data.dig("usage", "output_tokens")
|
|
84
|
+
# The thinking breakdown arrives only on the final `message_delta`. Carrying it through keeps streamed
|
|
85
|
+
# responses reporting the same `thinking_tokens` as non-streamed ones; without it the key is dropped and
|
|
86
|
+
# every streamed response silently reports no reasoning.
|
|
87
|
+
output_tokens_details = data.dig("usage", "output_tokens_details")
|
|
84
88
|
|
|
85
89
|
@data["usage"] ||= {}
|
|
86
90
|
@data["usage"]["input_tokens"] = input_tokens if input_tokens
|
|
87
91
|
@data["usage"]["output_tokens"] = output_tokens if output_tokens
|
|
92
|
+
@data["usage"]["output_tokens_details"] = output_tokens_details if output_tokens_details
|
|
88
93
|
end
|
|
89
94
|
|
|
90
95
|
# Handler for Type::MESSAGE_STOP
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
module OmniAI
|
|
4
|
+
module Anthropic
|
|
5
|
+
class Chat
|
|
6
|
+
# Overrides usage deserialize to read Anthropic's reasoning breakdown.
|
|
7
|
+
module UsageSerializer
|
|
8
|
+
# Anthropic already counts reasoning inside `output_tokens` and reports the breakdown alongside it at
|
|
9
|
+
# `output_tokens_details.thinking_tokens`, so the breakdown is read into `thinking_tokens` and the output
|
|
10
|
+
# count is left exactly as reported. Adding the two together would double count.
|
|
11
|
+
#
|
|
12
|
+
# When streaming, the breakdown arrives only on the final `message_delta`.
|
|
13
|
+
#
|
|
14
|
+
# @param data [Hash]
|
|
15
|
+
# @return [OmniAI::Chat::Usage]
|
|
16
|
+
def self.deserialize(data, *)
|
|
17
|
+
# Deserialize without a context so the generic flat parse runs rather than recursing into this method.
|
|
18
|
+
usage = OmniAI::Chat::Usage.deserialize(data)
|
|
19
|
+
|
|
20
|
+
# Only overwrite when the vendor container is actually present. A payload produced by `Usage#serialize`
|
|
21
|
+
# carries base's own `thinking_tokens` key and no `output_tokens_details`, so assigning unconditionally
|
|
22
|
+
# would clobber a correctly-parsed value with nil and break the round-trip. `unless nil?` rather than
|
|
23
|
+
# `||=`, so a reported zero from the wire still wins over base's nil.
|
|
24
|
+
thinking_tokens = data.dig("output_tokens_details", "thinking_tokens")
|
|
25
|
+
usage.thinking_tokens = thinking_tokens unless thinking_tokens.nil?
|
|
26
|
+
|
|
27
|
+
# Anthropic's `input_tokens` excludes cache reads and writes; adding them back keeps it the whole prompt, as
|
|
28
|
+
# for every other provider. A base-serialized payload has neither key and is already whole.
|
|
29
|
+
cache_tokens = data.values_at("cache_creation_input_tokens", "cache_read_input_tokens").compact
|
|
30
|
+
usage.input_tokens = (usage.input_tokens || 0) + cache_tokens.sum if cache_tokens.any?
|
|
31
|
+
|
|
32
|
+
usage
|
|
33
|
+
end
|
|
34
|
+
end
|
|
35
|
+
end
|
|
36
|
+
end
|
|
37
|
+
end
|
|
@@ -87,6 +87,8 @@ module OmniAI
|
|
|
87
87
|
|
|
88
88
|
context.serializers[:thinking] = ThinkingSerializer.method(:serialize)
|
|
89
89
|
context.deserializers[:thinking] = ThinkingSerializer.method(:deserialize)
|
|
90
|
+
|
|
91
|
+
context.deserializers[:usage] = UsageSerializer.method(:deserialize)
|
|
90
92
|
end
|
|
91
93
|
|
|
92
94
|
# @return [Hash]
|
|
@@ -163,19 +165,40 @@ module OmniAI
|
|
|
163
165
|
end
|
|
164
166
|
end
|
|
165
167
|
|
|
168
|
+
# When caching, the last block of the last message is marked so each tool-loop round reads the history the
|
|
169
|
+
# previous round wrote.
|
|
170
|
+
#
|
|
166
171
|
# @return [Array<Hash>]
|
|
167
172
|
def messages
|
|
168
|
-
messages = @prompt.messages.reject(&:system?)
|
|
169
|
-
messages
|
|
173
|
+
messages = @prompt.messages.reject(&:system?).map { |message| message.serialize(context:) }
|
|
174
|
+
return messages unless cache_control && messages.any?
|
|
175
|
+
|
|
176
|
+
*history, last = messages
|
|
177
|
+
history + [last.merge(content: with_cache_control(last[:content]))]
|
|
170
178
|
end
|
|
171
179
|
|
|
172
|
-
#
|
|
180
|
+
# When caching, a single marked text block. Tools render before system, so this breakpoint covers both.
|
|
181
|
+
#
|
|
182
|
+
# @return [String, Array<Hash>, nil]
|
|
173
183
|
def system
|
|
174
184
|
parts = @prompt.messages.filter(&:system?).filter(&:text?).map(&:text)
|
|
175
185
|
parts << formatting if formatting?
|
|
176
186
|
return if parts.empty?
|
|
177
187
|
|
|
178
|
-
parts.join("\n\n")
|
|
188
|
+
text = parts.join("\n\n")
|
|
189
|
+
cache_control ? [{ type: "text", text:, cache_control: }] : text
|
|
190
|
+
end
|
|
191
|
+
|
|
192
|
+
# Translates the opt-in `cache` option to Anthropic's `cache_control`.
|
|
193
|
+
# Example: `cache: true` becomes `{ type: "ephemeral" }` (5-minute TTL)
|
|
194
|
+
# Example: `cache: { ttl: "1h" }` becomes `{ type: "ephemeral", ttl: "1h" }`
|
|
195
|
+
#
|
|
196
|
+
# @return [Hash, nil]
|
|
197
|
+
def cache_control
|
|
198
|
+
case @options[:cache]
|
|
199
|
+
when true then { type: "ephemeral" }
|
|
200
|
+
when Hash then { type: "ephemeral" }.merge(@options[:cache])
|
|
201
|
+
end
|
|
179
202
|
end
|
|
180
203
|
|
|
181
204
|
# @return [String]
|
|
@@ -215,9 +238,25 @@ module OmniAI
|
|
|
215
238
|
end
|
|
216
239
|
end
|
|
217
240
|
|
|
241
|
+
# When caching without a system prompt, the last tool carries the breakpoint instead.
|
|
242
|
+
#
|
|
218
243
|
# @return [Array<Hash>, nil]
|
|
219
244
|
def tools_payload
|
|
220
|
-
|
|
245
|
+
return unless @tools&.any?
|
|
246
|
+
|
|
247
|
+
tools = @tools.map { |tool| tool.serialize(context:) }
|
|
248
|
+
cache_control && system.nil? ? with_cache_control(tools) : tools
|
|
249
|
+
end
|
|
250
|
+
|
|
251
|
+
# Relies on `MessageSerializer` normalizing every content part into a block Hash; a bare String would raise.
|
|
252
|
+
#
|
|
253
|
+
# @param blocks [Array<Hash>]
|
|
254
|
+
# @return [Array<Hash>]
|
|
255
|
+
def with_cache_control(blocks)
|
|
256
|
+
return blocks if blocks.empty?
|
|
257
|
+
|
|
258
|
+
*head, last = blocks
|
|
259
|
+
head + [last.merge(cache_control:)]
|
|
221
260
|
end
|
|
222
261
|
|
|
223
262
|
# @return [Boolean]
|
metadata
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: omniai-anthropic
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 3.
|
|
4
|
+
version: 3.7.0
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- Kevin Sylvestre
|
|
@@ -29,14 +29,14 @@ dependencies:
|
|
|
29
29
|
requirements:
|
|
30
30
|
- - "~>"
|
|
31
31
|
- !ruby/object:Gem::Version
|
|
32
|
-
version: '3.
|
|
32
|
+
version: '3.8'
|
|
33
33
|
type: :runtime
|
|
34
34
|
prerelease: false
|
|
35
35
|
version_requirements: !ruby/object:Gem::Requirement
|
|
36
36
|
requirements:
|
|
37
37
|
- - "~>"
|
|
38
38
|
- !ruby/object:Gem::Version
|
|
39
|
-
version: '3.
|
|
39
|
+
version: '3.8'
|
|
40
40
|
- !ruby/object:Gem::Dependency
|
|
41
41
|
name: openssl
|
|
42
42
|
requirement: !ruby/object:Gem::Requirement
|
|
@@ -89,6 +89,7 @@ files:
|
|
|
89
89
|
- lib/omniai/anthropic/chat/tool_call_serializer.rb
|
|
90
90
|
- lib/omniai/anthropic/chat/tool_serializer.rb
|
|
91
91
|
- lib/omniai/anthropic/chat/url_serializer.rb
|
|
92
|
+
- lib/omniai/anthropic/chat/usage_serializer.rb
|
|
92
93
|
- lib/omniai/anthropic/client.rb
|
|
93
94
|
- lib/omniai/anthropic/computer.rb
|
|
94
95
|
- lib/omniai/anthropic/config.rb
|