omniai-anthropic 3.6.0 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +18 -0
- data/lib/omniai/anthropic/chat/usage_serializer.rb +5 -0
- data/lib/omniai/anthropic/chat.rb +44 -6
- data/lib/omniai/anthropic/version.rb +1 -1
- metadata +1 -1
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 1689783da704c9474e0c87a65b63a11589918a1c85d2cdbb3c936bb367903cf9
|
|
4
|
+
data.tar.gz: 581b32358a1c0464e7f3550fe0d0caac455955703b98dde80f1fb755e4f8d10d
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: a89aa7e9b61dd98d9b893ea147da74f35fec059fdc05a9ebecaf80651496f62aabfbc3d43ede744527bf5e811a26caa9dc928e6c25ec433b1faeef734063fffe
|
|
7
|
+
data.tar.gz: 81ff3b93a6f55df6848cc86f65984f1083e29fc9c8ddd03661c04126df3ace29df62837238cea61d69011bb1037ad1c1a83bc832750acbf680e98f28ce60aad2
|
data/README.md
CHANGED
|
@@ -170,3 +170,21 @@ client.chat("What are the prime factors of 1234567?", model: "claude-sonnet-4-20
|
|
|
170
170
|
The thinking content will stream first, followed by the response.
|
|
171
171
|
|
|
172
172
|
[Anthropic API Reference `thinking`](https://docs.anthropic.com/en/docs/build-with-claude/thinking)
|
|
173
|
+
|
|
174
|
+
### Prompt Caching
|
|
175
|
+
|
|
176
|
+
Prompt caching is opt-in. When enabled, the system prompt (or the last tool, when there is no system prompt) and the last block of the last message are marked with `cache_control`, so each round of a tool-call loop reads the history the previous round wrote.
|
|
177
|
+
|
|
178
|
+
```ruby
|
|
179
|
+
client.chat(prompt, tools:, cache: true) # 5-minute TTL
|
|
180
|
+
client.chat(prompt, tools:, cache: { ttl: "1h" }) # 1-hour TTL
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
`usage.input_tokens` reports the whole prompt, including cached tokens. The cache breakdown is on each response's raw usage:
|
|
184
|
+
|
|
185
|
+
```ruby
|
|
186
|
+
response.response_chain.sum { |r| r.data.dig("usage", "cache_read_input_tokens").to_i }
|
|
187
|
+
response.response_chain.sum { |r| r.data.dig("usage", "cache_creation_input_tokens").to_i }
|
|
188
|
+
```
|
|
189
|
+
|
|
190
|
+
[Anthropic API Reference `prompt caching`](https://docs.anthropic.com/en/docs/build-with-claude/prompt-caching)
|
|
@@ -24,6 +24,11 @@ module OmniAI
|
|
|
24
24
|
thinking_tokens = data.dig("output_tokens_details", "thinking_tokens")
|
|
25
25
|
usage.thinking_tokens = thinking_tokens unless thinking_tokens.nil?
|
|
26
26
|
|
|
27
|
+
# Anthropic's `input_tokens` excludes cache reads and writes; adding them back keeps it the whole prompt, as
|
|
28
|
+
# for every other provider. A base-serialized payload has neither key and is already whole.
|
|
29
|
+
cache_tokens = data.values_at("cache_creation_input_tokens", "cache_read_input_tokens").compact
|
|
30
|
+
usage.input_tokens = (usage.input_tokens || 0) + cache_tokens.sum if cache_tokens.any?
|
|
31
|
+
|
|
27
32
|
usage
|
|
28
33
|
end
|
|
29
34
|
end
|
|
@@ -47,6 +47,7 @@ module OmniAI
|
|
|
47
47
|
CLAUDE_OPUS_4_6 = "claude-opus-4-6"
|
|
48
48
|
CLAUDE_OPUS_4_7 = "claude-opus-4-7"
|
|
49
49
|
CLAUDE_OPUS_5 = "claude-opus-5"
|
|
50
|
+
CLAUDE_OPUS_5_5 = "claude-opus-5-5"
|
|
50
51
|
CLAUDE_SONNET_4_0 = "claude-sonnet-4-0"
|
|
51
52
|
CLAUDE_SONNET_4_5 = "claude-sonnet-4-5"
|
|
52
53
|
CLAUDE_SONNET_4_6 = "claude-sonnet-4-6"
|
|
@@ -54,7 +55,7 @@ module OmniAI
|
|
|
54
55
|
CLAUDE_FABLE_5 = "claude-fable-5"
|
|
55
56
|
|
|
56
57
|
CLAUDE_HAIKU = CLAUDE_HAIKU_4_5
|
|
57
|
-
CLAUDE_OPUS =
|
|
58
|
+
CLAUDE_OPUS = CLAUDE_OPUS_5_5
|
|
58
59
|
CLAUDE_SONNET = CLAUDE_SONNET_5
|
|
59
60
|
end
|
|
60
61
|
|
|
@@ -165,19 +166,40 @@ module OmniAI
|
|
|
165
166
|
end
|
|
166
167
|
end
|
|
167
168
|
|
|
169
|
+
# When caching, the last block of the last message is marked so each tool-loop round reads the history the
|
|
170
|
+
# previous round wrote.
|
|
171
|
+
#
|
|
168
172
|
# @return [Array<Hash>]
|
|
169
173
|
def messages
|
|
170
|
-
messages = @prompt.messages.reject(&:system?)
|
|
171
|
-
messages
|
|
174
|
+
messages = @prompt.messages.reject(&:system?).map { |message| message.serialize(context:) }
|
|
175
|
+
return messages unless cache_control && messages.any?
|
|
176
|
+
|
|
177
|
+
*history, last = messages
|
|
178
|
+
history + [last.merge(content: with_cache_control(last[:content]))]
|
|
172
179
|
end
|
|
173
180
|
|
|
174
|
-
#
|
|
181
|
+
# When caching, a single marked text block. Tools render before system, so this breakpoint covers both.
|
|
182
|
+
#
|
|
183
|
+
# @return [String, Array<Hash>, nil]
|
|
175
184
|
def system
|
|
176
185
|
parts = @prompt.messages.filter(&:system?).filter(&:text?).map(&:text)
|
|
177
186
|
parts << formatting if formatting?
|
|
178
187
|
return if parts.empty?
|
|
179
188
|
|
|
180
|
-
parts.join("\n\n")
|
|
189
|
+
text = parts.join("\n\n")
|
|
190
|
+
cache_control ? [{ type: "text", text:, cache_control: }] : text
|
|
191
|
+
end
|
|
192
|
+
|
|
193
|
+
# Translates the opt-in `cache` option to Anthropic's `cache_control`.
|
|
194
|
+
# Example: `cache: true` becomes `{ type: "ephemeral" }` (5-minute TTL)
|
|
195
|
+
# Example: `cache: { ttl: "1h" }` becomes `{ type: "ephemeral", ttl: "1h" }`
|
|
196
|
+
#
|
|
197
|
+
# @return [Hash, nil]
|
|
198
|
+
def cache_control
|
|
199
|
+
case @options[:cache]
|
|
200
|
+
when true then { type: "ephemeral" }
|
|
201
|
+
when Hash then { type: "ephemeral" }.merge(@options[:cache])
|
|
202
|
+
end
|
|
181
203
|
end
|
|
182
204
|
|
|
183
205
|
# @return [String]
|
|
@@ -217,9 +239,25 @@ module OmniAI
|
|
|
217
239
|
end
|
|
218
240
|
end
|
|
219
241
|
|
|
242
|
+
# When caching without a system prompt, the last tool carries the breakpoint instead.
|
|
243
|
+
#
|
|
220
244
|
# @return [Array<Hash>, nil]
|
|
221
245
|
def tools_payload
|
|
222
|
-
|
|
246
|
+
return unless @tools&.any?
|
|
247
|
+
|
|
248
|
+
tools = @tools.map { |tool| tool.serialize(context:) }
|
|
249
|
+
cache_control && system.nil? ? with_cache_control(tools) : tools
|
|
250
|
+
end
|
|
251
|
+
|
|
252
|
+
# Relies on `MessageSerializer` normalizing every content part into a block Hash; a bare String would raise.
|
|
253
|
+
#
|
|
254
|
+
# @param blocks [Array<Hash>]
|
|
255
|
+
# @return [Array<Hash>]
|
|
256
|
+
def with_cache_control(blocks)
|
|
257
|
+
return blocks if blocks.empty?
|
|
258
|
+
|
|
259
|
+
*head, last = blocks
|
|
260
|
+
head + [last.merge(cache_control:)]
|
|
223
261
|
end
|
|
224
262
|
|
|
225
263
|
# @return [Boolean]
|