scout-ai 1.2.5 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.vimproject +111 -31
- data/README.md +171 -320
- data/Rakefile +17 -1
- data/VERSION +1 -1
- data/doc/Improvements.md +325 -0
- data/doc/StartHere.md +110 -0
- data/doc/developer/Architecture.md +126 -0
- data/doc/developer/Backends.md +199 -0
- data/doc/developer/ChatLifecycle.md +183 -0
- data/doc/developer/DelegationInternals.md +295 -0
- data/doc/developer/DesignPrinciples.md +245 -0
- data/doc/developer/PromptProcessing.md +292 -0
- data/doc/developer/Provenance.md +317 -0
- data/doc/user/BuildingAgents.md +345 -0
- data/doc/user/Cookbook.md +333 -0
- data/doc/user/CoreConcepts.md +181 -0
- data/doc/user/Delegation.md +191 -0
- data/doc/user/GettingStarted.md +159 -0
- data/doc/user/ManagingContext.md +163 -0
- data/doc/user/MultiAgentWorkflows.md +256 -0
- data/doc/user/Python.md +159 -0
- data/doc/user/RunningInference.md +200 -0
- data/doc/user/ToolCalling.md +193 -0
- data/doc/user/WritingChats.md +197 -0
- data/lib/scout/llm/agent/chat.rb +62 -12
- data/lib/scout/llm/agent/delegate.rb +274 -65
- data/lib/scout/llm/agent/iterate.rb +2 -2
- data/lib/scout/llm/agent/save.rb +273 -0
- data/lib/scout/llm/agent/workflow.rb +164 -0
- data/lib/scout/llm/agent.rb +86 -61
- data/lib/scout/llm/ask.rb +36 -6
- data/lib/scout/llm/backends/anthropic.rb +9 -1
- data/lib/scout/llm/backends/bedrock.rb +15 -3
- data/lib/scout/llm/backends/default.rb +129 -89
- data/lib/scout/llm/backends/glm.rb +58 -0
- data/lib/scout/llm/backends/huggingface.rb +13 -1
- data/lib/scout/llm/backends/ollama.rb +13 -0
- data/lib/scout/llm/backends/openwebui.rb +8 -3
- data/lib/scout/llm/chat/agent_meta.rb +264 -0
- data/lib/scout/llm/chat/annotation.rb +37 -14
- data/lib/scout/llm/chat/parse.rb +28 -6
- data/lib/scout/llm/chat/persist.rb +25 -0
- data/lib/scout/llm/chat/process/clear.rb +19 -1
- data/lib/scout/llm/chat/process/files.rb +16 -1
- data/lib/scout/llm/chat/process/meta.rb +422 -32
- data/lib/scout/llm/chat/process/options.rb +21 -1
- data/lib/scout/llm/chat/process/tools.rb +49 -12
- data/lib/scout/llm/chat/process.rb +4 -0
- data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
- data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
- data/lib/scout/llm/chat/prompt.rb +48 -0
- data/lib/scout/llm/chat/provenance.rb +775 -0
- data/lib/scout/llm/chat/tool_calls.rb +76 -0
- data/lib/scout/llm/chat.rb +18 -2
- data/lib/scout/llm/embed.rb +8 -3
- data/lib/scout/llm/image.rb +86 -0
- data/lib/scout/llm/rag.rb +3 -3
- data/lib/scout/llm/tools/call.rb +159 -10
- data/lib/scout/llm/tools/knowledge_base.rb +1 -1
- data/lib/scout/llm/tools/workflow.rb +13 -5
- data/lib/scout-ai.rb +1 -0
- data/research/ChatAnalyst-required-changes.md +167 -0
- data/research/agent-delegation-analysis.md +810 -0
- data/research/agent-meta-provenance-integration-plan.md +622 -0
- data/research/agent-workflow-analysis.md +1120 -0
- data/research/backends-analysis.md +836 -0
- data/research/chat-core-analysis.md +946 -0
- data/research/chatanalyst-provenance/00-baseline.md +30 -0
- data/research/chatanalyst-provenance/01-repo-map.md +60 -0
- data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
- data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
- data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
- data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
- data/research/chatanalyst-provenance/07-critic-review.md +25 -0
- data/research/chatanalyst-provenance/final-report.md +45 -0
- data/research/chatanalyst-provenance/resumption.md +37 -0
- data/research/coding-philosophy-analysis.md +928 -0
- data/research/commands-analysis.md +947 -0
- data/research/multi-agent-patterns-analysis.md +853 -0
- data/research/prompt-strategies-analysis.md +630 -0
- data/research/prov-verbosity-fix-notes.md +77 -0
- data/research/provenance-analysis.md +469 -0
- data/research/provenance-navigation-design.md +640 -0
- data/research/synthesis-report.md +487 -0
- data/research/tools-system-analysis.md +779 -0
- data/scout-ai.gemspec +97 -13
- data/scout_commands/agent/ask +13 -3
- data/scout_commands/agent/kb +2 -0
- data/scout_commands/llm/ask +11 -4
- data/scout_commands/llm/md +76 -0
- data/scout_commands/llm/prov +602 -0
- data/scout_commands/llm/word +71 -0
- data/share/word/reference.docx +0 -0
- data/test/etc/AI/mock.yaml +11 -0
- data/test/fixtures/backends/anthropic.json +19 -0
- data/test/fixtures/backends/anthropic_tool_use.json +24 -0
- data/test/fixtures/backends/bedrock.json +8 -0
- data/test/fixtures/backends/bedrock_embedding.json +3 -0
- data/test/fixtures/backends/bedrock_tool_use.json +17 -0
- data/test/fixtures/backends/ollama.json +16 -0
- data/test/fixtures/backends/ollama_tool_call.json +27 -0
- data/test/fixtures/backends/openai_chat.json +21 -0
- data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
- data/test/fixtures/backends/responses.json +33 -0
- data/test/fixtures/backends/responses_tool_call.json +28 -0
- data/test/integration/README.md +32 -0
- data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
- data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
- data/test/integration/scout/llm/backends/test_relay.rb +52 -0
- data/test/integration/scout/llm/test_infrastructure.rb +74 -0
- data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
- data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
- data/test/integration/scout/model/test_base.rb +91 -0
- data/test/scout/llm/agent/test_chat.rb +8 -2
- data/test/scout/llm/agent/test_save.rb +413 -0
- data/test/scout/llm/agent/test_workflow.rb +110 -0
- data/test/scout/llm/backends/test_anthropic.rb +93 -10
- data/test/scout/llm/backends/test_bedrock.rb +118 -2
- data/test/scout/llm/backends/test_ollama.rb +70 -20
- data/test/scout/llm/backends/test_openwebui.rb +42 -40
- data/test/scout/llm/backends/test_relay.rb +4 -2
- data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
- data/test/scout/llm/chat/process/test_meta.rb +518 -0
- data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
- data/test/scout/llm/chat/test_agent_meta.rb +357 -0
- data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
- data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
- data/test/scout/llm/chat/test_parse.rb +70 -15
- data/test/scout/llm/chat/test_prov_cli.rb +274 -0
- data/test/scout/llm/chat/test_provenance.rb +240 -0
- data/test/scout/llm/chat/test_tool_calls.rb +38 -0
- data/test/scout/llm/test_agent.rb +13 -36
- data/test/scout/llm/test_ask.rb +75 -52
- data/test/scout/llm/test_chat.rb +107 -13
- data/test/scout/llm/test_embed.rb +48 -0
- data/test/scout/llm/test_rag.rb +23 -16
- data/test/scout/llm/test_tools.rb +12 -1
- data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
- data/test/scout/llm/tools/test_mcp.rb +5 -3
- data/test/scout/llm/tools/test_workflow.rb +23 -2
- data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
- data/test/scout/model/python/huggingface/test_causal.rb +9 -3
- data/test/scout/model/python/huggingface/test_classification.rb +11 -2
- data/test/scout/model/python/test_torch.rb +2 -0
- data/test/scout/model/python/torch/test_helpers.rb +4 -0
- data/test/scout/model/test_base.rb +4 -2
- data/test/support/availability.rb +231 -0
- data/test/support/fake_clients.rb +138 -0
- data/test/support/fixtures.rb +21 -0
- data/test/support/infrastructure_probes.rb +136 -0
- data/test/support/mock_backend.rb +215 -0
- data/test/test_helper.rb +32 -2
- metadata +96 -12
- data/doc/Agent.md +0 -354
- data/doc/Chat.md +0 -481
- data/doc/LLM.md +0 -356
- data/doc/PythonAgentTasks.md +0 -333
- data/doc/RAG.md +0 -129
- data/doc/USER_GUIDE.md +0 -572
- data/scout_commands/documenter +0 -148
- data/test/scout/llm/backends/test_openai.rb +0 -192
- data/test/scout/llm/backends/test_responses.rb +0 -238
- data/test/scout/llm/test_parse.rb +0 -98
|
@@ -0,0 +1,518 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
|
+
|
|
4
|
+
require 'scout/llm/chat'
|
|
5
|
+
require 'scout/llm/backends/responses'
|
|
6
|
+
|
|
7
|
+
class TestLLMUsageMeta < Test::Unit::TestCase
|
|
8
|
+
def setup
|
|
9
|
+
super
|
|
10
|
+
Chat::TOKEN_KEYS.each { |name| Thread.current["#{name}_s"] = 0 }
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def response(prompt: nil, completion: nil, total: nil,
|
|
14
|
+
cached: nil, cache_write: nil, reasoning: nil)
|
|
15
|
+
usage = {}
|
|
16
|
+
usage['prompt_tokens'] = prompt unless prompt.nil?
|
|
17
|
+
usage['completion_tokens'] = completion unless completion.nil?
|
|
18
|
+
usage['total_tokens'] = total unless total.nil?
|
|
19
|
+
usage['prompt_tokens_details'] = {}
|
|
20
|
+
usage['prompt_tokens_details']['cached_tokens'] = cached unless cached.nil?
|
|
21
|
+
usage['input_tokens_details'] = {} if cache_write || cached
|
|
22
|
+
usage['input_tokens_details'] ||= {}
|
|
23
|
+
usage['input_tokens_details']['cache_write_tokens'] = cache_write unless cache_write.nil?
|
|
24
|
+
usage['completion_tokens_details'] = {}
|
|
25
|
+
usage['completion_tokens_details']['reasoning_tokens'] = reasoning unless reasoning.nil?
|
|
26
|
+
{ 'usage' => usage }
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def chat(text)
|
|
30
|
+
Chat.setup(LLM.messages(text))
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def test_backend_records_direct_and_running_token_counts
|
|
34
|
+
first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
35
|
+
second = LLM::Responses.update_meta(response(prompt: 7, total: 7), first)
|
|
36
|
+
|
|
37
|
+
assert_equal 7, second['pt']
|
|
38
|
+
assert_nil second['ct']
|
|
39
|
+
assert_equal 9, second['pt_s']
|
|
40
|
+
assert_equal 12, second['tt_s']
|
|
41
|
+
assert_equal 9, second['pt_c']
|
|
42
|
+
assert_equal 3, second['ct_c']
|
|
43
|
+
assert_equal 12, second['tt_c']
|
|
44
|
+
assert_nil second['usage_id']
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def test_jobs_returns_all_projecting_jobs
|
|
48
|
+
conversation = chat <<-EOF
|
|
49
|
+
user: First
|
|
50
|
+
meta: job=WF/ask/first.chat
|
|
51
|
+
assistant: First answer
|
|
52
|
+
user: Second
|
|
53
|
+
meta: job=WF/ask/second.chat
|
|
54
|
+
assistant: Second answer
|
|
55
|
+
EOF
|
|
56
|
+
|
|
57
|
+
assert_equal %w[WF/ask/first.chat WF/ask/second.chat], conversation.jobs
|
|
58
|
+
assert_equal 'WF/ask/second.chat', conversation.meta[:job]
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def test_message_identity_includes_non_meta_history
|
|
62
|
+
first = chat <<-EOF
|
|
63
|
+
user: Question
|
|
64
|
+
meta: tt=5
|
|
65
|
+
assistant: Answer
|
|
66
|
+
EOF
|
|
67
|
+
same = chat <<-EOF
|
|
68
|
+
user: Question
|
|
69
|
+
assistant: Answer
|
|
70
|
+
EOF
|
|
71
|
+
different = chat <<-EOF
|
|
72
|
+
user: Different question
|
|
73
|
+
assistant: Answer
|
|
74
|
+
EOF
|
|
75
|
+
|
|
76
|
+
assert_equal first.message_index.last[:id], same.message_index.last[:id]
|
|
77
|
+
assert_not_equal first.message_index.last[:id], different.message_index.last[:id]
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def test_consecutive_meta_leaves_the_first_segment_orphaned
|
|
81
|
+
conversation = chat <<-EOF
|
|
82
|
+
user: Work
|
|
83
|
+
meta: tt=2
|
|
84
|
+
meta: job=WF/ask/work.chat
|
|
85
|
+
assistant: Done
|
|
86
|
+
EOF
|
|
87
|
+
|
|
88
|
+
trace = Chat.trace_chats([conversation])
|
|
89
|
+
assert_equal 2, trace.length
|
|
90
|
+
assert trace.first[:orphan]
|
|
91
|
+
assert_equal 2, trace.first[:meta][:tt]
|
|
92
|
+
assert_equal 'WF/ask/work.chat', trace.last[:meta][:job]
|
|
93
|
+
assert_equal 1, trace.last[:messages].length
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def test_final_meta_is_an_orphan_segment
|
|
97
|
+
conversation = chat <<-EOF
|
|
98
|
+
user: Work
|
|
99
|
+
meta: tt=2
|
|
100
|
+
assistant: Tool call removed
|
|
101
|
+
meta: tt=7
|
|
102
|
+
EOF
|
|
103
|
+
|
|
104
|
+
trace = Chat.trace_chats([conversation])
|
|
105
|
+
assert_equal 2, trace.length
|
|
106
|
+
assert_equal 7, trace.last[:meta][:tt]
|
|
107
|
+
assert trace.last[:orphan]
|
|
108
|
+
assert_empty trace.last[:messages]
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def test_meta_covers_a_multi_tool_response_segment
|
|
112
|
+
conversation = chat <<-EOF
|
|
113
|
+
user: Write two files
|
|
114
|
+
meta: tt=1000
|
|
115
|
+
function_call: {"name":"write","id":"one"}
|
|
116
|
+
function_call_output: {"id":"one","content":"done one"}
|
|
117
|
+
function_call: {"name":"write","id":"two"}
|
|
118
|
+
function_call_output: {"id":"two","content":"done two"}
|
|
119
|
+
assistant: Done
|
|
120
|
+
user: Next request
|
|
121
|
+
EOF
|
|
122
|
+
|
|
123
|
+
trace = Chat.trace_chats([conversation])
|
|
124
|
+
assert_equal 1, trace.length
|
|
125
|
+
assert_equal 1000, trace.first[:meta][:tt]
|
|
126
|
+
assert_equal 5, trace.first[:messages].length
|
|
127
|
+
assert !trace.first[:orphan]
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def test_project_keeps_inference_meta_inline_with_one_job_marker
|
|
131
|
+
response = [
|
|
132
|
+
{ role: :meta, content: 'tt=2' },
|
|
133
|
+
{ role: :function_call, content: '{"name":"write"}' },
|
|
134
|
+
{ role: :function_call_output, content: '{"content":"done"}' },
|
|
135
|
+
{ role: :meta, content: 'tt=7' },
|
|
136
|
+
{ role: :assistant, content: 'Done' }
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
projected = Chat.project('WF/ask/work.chat', response)
|
|
140
|
+
assert_equal %i[meta meta function_call function_call_output meta assistant], projected.collect { |m| m[:role] }
|
|
141
|
+
|
|
142
|
+
marker = Chat.parse_meta(projected.first[:content])
|
|
143
|
+
assert_equal 'WF/ask/work.chat', marker[:job]
|
|
144
|
+
assert Chat::TOKEN_KEYS.none? { |key| marker.include?(key) }
|
|
145
|
+
|
|
146
|
+
assert_equal 2, Chat.parse_meta(projected[1][:content])[:tt]
|
|
147
|
+
assert_equal :function_call, projected[2][:role], 'first inference meta stays adjacent to the call it produced'
|
|
148
|
+
assert_equal 7, Chat.parse_meta(projected[4][:content])[:tt]
|
|
149
|
+
|
|
150
|
+
trace = Chat.trace_chats([Chat.setup(projected)])
|
|
151
|
+
assert_equal 3, trace.length
|
|
152
|
+
assert trace.first[:orphan]
|
|
153
|
+
assert_equal [2, 7], trace[1..-1].collect { |entry| entry[:meta][:tt] }
|
|
154
|
+
assert_equal [2, 1], trace[1..-1].collect { |entry| entry[:messages].length }
|
|
155
|
+
assert trace[1..-1].none? { |entry| entry[:orphan] }
|
|
156
|
+
|
|
157
|
+
assert_equal 2, Chat.direct_entries([Chat.setup(projected)]).length
|
|
158
|
+
totals = Chat.token_totals([Chat.setup(projected)])
|
|
159
|
+
assert_equal 9, totals[:tt]
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
# Legacy chats carry no inference_id, so Chat.trace_indices falls back to the
|
|
163
|
+
# digest-based lineage id. The lineage id is computed from the preceding
|
|
164
|
+
# messages, and a projected copy sits behind a leading `job=` marker, so the
|
|
165
|
+
# projected and original copies of the same legacy inference get DIFFERENT
|
|
166
|
+
# lineage ids and are both counted. This is the documented legacy behaviour:
|
|
167
|
+
# precise deduplication requires inference_id, which every new inference has.
|
|
168
|
+
def test_project_legacy_meta_without_inference_id_is_not_deduplicated_across_chats
|
|
169
|
+
original = chat <<-EOF
|
|
170
|
+
user: Work
|
|
171
|
+
meta: pt=2 ct=1 tt=3
|
|
172
|
+
assistant: Done
|
|
173
|
+
EOF
|
|
174
|
+
|
|
175
|
+
projected = Chat.project('WF/ask/work.chat', [
|
|
176
|
+
{ role: :meta, content: 'pt=2 ct=1 tt=3' },
|
|
177
|
+
{ role: :assistant, content: 'Done' }
|
|
178
|
+
])
|
|
179
|
+
|
|
180
|
+
trace = Chat.trace_chats([Chat.setup(projected), original])
|
|
181
|
+
assert_equal 3, trace.length, 'job marker + two non-merged legacy lineages'
|
|
182
|
+
assert trace.all? { |entry| entry[:deduplication] == :legacy_lineage }
|
|
183
|
+
assert_not_equal trace.first[:lineage_id], trace.last[:lineage_id]
|
|
184
|
+
|
|
185
|
+
projected_totals = Chat.token_totals([Chat.setup(projected)])
|
|
186
|
+
assert_equal 3, projected_totals[:tt]
|
|
187
|
+
assert_equal 3, Chat.token_totals([original])[:tt]
|
|
188
|
+
assert_equal 6, Chat.token_totals([Chat.setup(projected), original])[:tt]
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def test_project_consumption_path_does_not_double_count_a_saved_projection
|
|
192
|
+
TmpFile.with_file(nil, false, :persistent => true) do |file|
|
|
193
|
+
original = chat <<-EOF
|
|
194
|
+
user: Work
|
|
195
|
+
meta: inference_id=request-one pt=10 ct=5 tt=15
|
|
196
|
+
assistant: Done
|
|
197
|
+
EOF
|
|
198
|
+
|
|
199
|
+
# chat_task: the job result chat is the projection; the log keeps the
|
|
200
|
+
# original metas.
|
|
201
|
+
projected = Chat.project('WF/ask/work.chat', [
|
|
202
|
+
{ role: :meta, content: 'inference_id=request-one pt=10 ct=5 tt=15' },
|
|
203
|
+
{ role: :assistant, content: 'Done' }
|
|
204
|
+
])
|
|
205
|
+
Open.write(file, Chat.print(Chat.setup(projected)))
|
|
206
|
+
|
|
207
|
+
# LLM::Agent#ask consumption path: load the persisted job chat and
|
|
208
|
+
# re-project it before counting.
|
|
209
|
+
loaded = Chat.load(file)
|
|
210
|
+
reprojected = Chat.project('WF/ask/work.chat', loaded)
|
|
211
|
+
|
|
212
|
+
assert_equal Chat.token_totals([original]), Chat.token_totals([Chat.setup(reprojected), original])
|
|
213
|
+
end
|
|
214
|
+
end
|
|
215
|
+
def test_trace_keeps_distinct_segments_for_direct_and_projected_metadata
|
|
216
|
+
direct = chat <<-EOF
|
|
217
|
+
user: Work
|
|
218
|
+
meta: tt=7
|
|
219
|
+
assistant: Done
|
|
220
|
+
EOF
|
|
221
|
+
projected = chat <<-EOF
|
|
222
|
+
user: Work
|
|
223
|
+
meta: job=WF/ask/work.chat
|
|
224
|
+
assistant: Done
|
|
225
|
+
EOF
|
|
226
|
+
|
|
227
|
+
trace = Chat.trace_chats([projected, direct])
|
|
228
|
+
assert_equal 2, trace.length
|
|
229
|
+
assert_equal ['WF/ask/work.chat', nil], trace.collect { |entry| entry[:meta][:job] }
|
|
230
|
+
assert_equal [nil, 7], trace.collect { |entry| entry[:meta][:tt] }
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
def test_job_meta_does_not_reset_the_last_direct_chat_total
|
|
234
|
+
messages = LLM.messages <<-EOF
|
|
235
|
+
user: Plan
|
|
236
|
+
meta: pt=10 ct=2 tt=12 pt_c=10 ct_c=2 tt_c=12
|
|
237
|
+
assistant: Plan complete
|
|
238
|
+
meta: job=WF/ask/work.chat
|
|
239
|
+
assistant: Work complete
|
|
240
|
+
EOF
|
|
241
|
+
|
|
242
|
+
current = Chat.meta(messages)
|
|
243
|
+
assert_equal 'WF/ask/work.chat', current[:job]
|
|
244
|
+
assert_equal 10, current[:pt_c]
|
|
245
|
+
assert_equal 2, current[:ct_c]
|
|
246
|
+
assert_equal 12, current[:tt_c]
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
# === Cache token accounting tests ===
|
|
250
|
+
|
|
251
|
+
def test_openai_responses_api_cache_tokens
|
|
252
|
+
resp = { 'usage' => {
|
|
253
|
+
'input_tokens' => 9,
|
|
254
|
+
'input_tokens_details' => { 'cache_write_tokens' => 5, 'cached_tokens' => 3 },
|
|
255
|
+
'output_tokens' => 174,
|
|
256
|
+
'output_tokens_details' => { 'reasoning_tokens' => 128 },
|
|
257
|
+
'total_tokens' => 183
|
|
258
|
+
} }
|
|
259
|
+
meta = LLM::Responses.update_meta(resp)
|
|
260
|
+
|
|
261
|
+
assert_equal 9, meta['pt']
|
|
262
|
+
assert_equal 174, meta['ct']
|
|
263
|
+
assert_equal 183, meta['tt']
|
|
264
|
+
assert_equal 3, meta['cct']
|
|
265
|
+
assert_equal 5, meta['cwt']
|
|
266
|
+
assert_equal 128, meta['rt']
|
|
267
|
+
# cumulative variants
|
|
268
|
+
assert_equal 9, meta['pt_c']
|
|
269
|
+
assert_equal 3, meta['cct_c']
|
|
270
|
+
assert_equal 5, meta['cwt_c']
|
|
271
|
+
assert_equal 128, meta['rt_c']
|
|
272
|
+
# session variants
|
|
273
|
+
assert_equal 9, meta['pt_s']
|
|
274
|
+
assert_equal 3, meta['cct_s']
|
|
275
|
+
assert_equal 5, meta['cwt_s']
|
|
276
|
+
assert_equal 128, meta['rt_s']
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
def test_glm_cache_tokens
|
|
280
|
+
resp = { 'usage' => {
|
|
281
|
+
'completion_tokens' => 97,
|
|
282
|
+
'completion_tokens_details' => { 'reasoning_tokens' => 92 },
|
|
283
|
+
'prompt_tokens' => 8,
|
|
284
|
+
'prompt_tokens_details' => { 'cached_tokens' => 4 },
|
|
285
|
+
'total_tokens' => 105
|
|
286
|
+
} }
|
|
287
|
+
meta = LLM::Responses.update_meta(resp)
|
|
288
|
+
|
|
289
|
+
assert_equal 8, meta['pt']
|
|
290
|
+
assert_equal 97, meta['ct']
|
|
291
|
+
assert_equal 105, meta['tt']
|
|
292
|
+
assert_equal 4, meta['cct']
|
|
293
|
+
assert_nil meta['cwt']
|
|
294
|
+
assert_equal 92, meta['rt']
|
|
295
|
+
# cumulative variants
|
|
296
|
+
assert_equal 4, meta['cct_c']
|
|
297
|
+
assert_equal 92, meta['rt_c']
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def test_anthropic_flat_cache_fields
|
|
301
|
+
resp = { 'usage' => {
|
|
302
|
+
'prompt_tokens' => 100,
|
|
303
|
+
'completion_tokens' => 50,
|
|
304
|
+
'cache_read_input_tokens' => 80,
|
|
305
|
+
'cache_creation_input_tokens' => 20
|
|
306
|
+
} }
|
|
307
|
+
meta = LLM::Responses.update_meta(resp)
|
|
308
|
+
|
|
309
|
+
assert_equal 100, meta['pt']
|
|
310
|
+
assert_equal 50, meta['ct']
|
|
311
|
+
assert_equal 150, meta['tt'] # computed
|
|
312
|
+
assert_equal 80, meta['cct']
|
|
313
|
+
assert_equal 20, meta['cwt']
|
|
314
|
+
assert_nil meta['rt']
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
def test_cumulative_cache_tokens_across_requests
|
|
318
|
+
first = LLM::Responses.update_meta(
|
|
319
|
+
response(prompt: 10, completion: 5, total: 15, cached: 3, reasoning: 2)
|
|
320
|
+
)
|
|
321
|
+
second = LLM::Responses.update_meta(
|
|
322
|
+
response(prompt: 8, completion: 4, total: 12, cached: 6, reasoning: 1),
|
|
323
|
+
first
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
assert_equal 3, first['cct_c']
|
|
327
|
+
assert_equal 2, first['rt_c']
|
|
328
|
+
assert_equal 9, second['cct_c'] # 3 + 6
|
|
329
|
+
assert_equal 3, second['rt_c'] # 2 + 1
|
|
330
|
+
assert_equal 18, second['pt_c'] # 10 + 8
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def test_normalize_usage_constants
|
|
334
|
+
assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
|
|
335
|
+
assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
|
|
336
|
+
end
|
|
337
|
+
|
|
338
|
+
def test_normalize_usage_openai_chat_api
|
|
339
|
+
usage = { 'prompt_tokens' => 9, 'completion_tokens' => 174, 'total_tokens' => 183 }
|
|
340
|
+
result = Chat.normalize_usage(usage)
|
|
341
|
+
assert_equal 9, result['pt']
|
|
342
|
+
assert_equal 174, result['ct']
|
|
343
|
+
assert_equal 183, result['tt']
|
|
344
|
+
assert_nil result['cct']
|
|
345
|
+
assert_nil result['cwt']
|
|
346
|
+
assert_nil result['rt']
|
|
347
|
+
end
|
|
348
|
+
|
|
349
|
+
def test_normalize_usage_computes_total_when_missing
|
|
350
|
+
usage = { 'prompt_tokens' => 10, 'completion_tokens' => 20 }
|
|
351
|
+
result = Chat.normalize_usage(usage)
|
|
352
|
+
assert_equal 30, result['tt']
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
def test_normalize_usage_handles_nil
|
|
356
|
+
assert_equal({}, Chat.normalize_usage(nil))
|
|
357
|
+
assert_equal({}, Chat.normalize_usage({}))
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
def test_direct_entries_includes_cache_tokens
|
|
361
|
+
conversation = chat <<-EOF
|
|
362
|
+
user: Work
|
|
363
|
+
meta: pt=10 ct=5 tt=15 cct=3 rt=2
|
|
364
|
+
assistant: Done
|
|
365
|
+
EOF
|
|
366
|
+
|
|
367
|
+
entries = Chat.direct_entries([conversation])
|
|
368
|
+
assert_equal 1, entries.length
|
|
369
|
+
assert_equal 3, entries.first[:meta][:cct]
|
|
370
|
+
assert_equal 2, entries.first[:meta][:rt]
|
|
371
|
+
end
|
|
372
|
+
|
|
373
|
+
def test_token_totals_aggregates_cache_fields
|
|
374
|
+
c1 = chat <<-EOF
|
|
375
|
+
user: Work
|
|
376
|
+
meta: pt=10 ct=5 tt=15 cct=3 rt=2
|
|
377
|
+
assistant: Done
|
|
378
|
+
EOF
|
|
379
|
+
c2 = chat <<-EOF
|
|
380
|
+
user: More work
|
|
381
|
+
meta: pt=20 ct=10 tt=30 cct=7 rt=8
|
|
382
|
+
assistant: Done again
|
|
383
|
+
EOF
|
|
384
|
+
|
|
385
|
+
totals = Chat.token_totals([c1, c2])
|
|
386
|
+
assert_equal 30, totals[:pt]
|
|
387
|
+
assert_equal 15, totals[:ct]
|
|
388
|
+
assert_equal 45, totals[:tt]
|
|
389
|
+
assert_equal 10, totals[:cct] # 3 + 7
|
|
390
|
+
assert_equal 10, totals[:rt] # 2 + 8
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
def test_print_tokens_shows_cache_fields
|
|
394
|
+
totals = { pt: 100, ct: 50, tt: 150, cct: 30, cwt: 10, rt: 20 }
|
|
395
|
+
output = Chat.print_tokens(totals)
|
|
396
|
+
assert output.include?('cached=30')
|
|
397
|
+
assert output.include?('cache_write=10')
|
|
398
|
+
assert output.include?('reasoning=20')
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
# === Quoted-value serialization/parsing tests ===
|
|
402
|
+
|
|
403
|
+
def test_serialize_meta_quotes_value_containing_equals
|
|
404
|
+
serialized = Chat.serialize_meta('reas' => 'thinking about a=b')
|
|
405
|
+
assert_equal 'reas="thinking about a=b"', serialized
|
|
406
|
+
end
|
|
407
|
+
|
|
408
|
+
def test_serialize_meta_does_not_quote_simple_values
|
|
409
|
+
serialized = Chat.serialize_meta('pt' => 10, 'job' => 'WF/ask/test.chat')
|
|
410
|
+
assert_equal 'pt=10 job=WF/ask/test.chat', serialized
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
def test_parse_meta_handles_quoted_value_with_equals
|
|
414
|
+
parsed = Chat.parse_meta('pt=10 reas="thinking about a=b"')
|
|
415
|
+
assert_equal 10, parsed[:pt]
|
|
416
|
+
assert_equal 'thinking about a=b', parsed[:reas]
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
def test_roundtrip_value_with_equals
|
|
420
|
+
meta = { 'pt' => 10, 'reas' => 'thinking about a=b and c=d', 'job' => 'WF/ask/test.chat' }
|
|
421
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
422
|
+
assert_equal meta, parsed
|
|
423
|
+
end
|
|
424
|
+
|
|
425
|
+
def test_roundtrip_value_with_quotes_and_backslashes
|
|
426
|
+
meta = { 'reas' => 'He said "hello" and a=b\c', 'pt' => 5 }
|
|
427
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
428
|
+
assert_equal meta, parsed
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
def test_backward_compat_unquoted_value_with_spaces
|
|
432
|
+
parsed = Chat.parse_meta('pt=10 ct=5 tt=15 reas=some text here')
|
|
433
|
+
assert_equal 10, parsed[:pt]
|
|
434
|
+
assert_equal 5, parsed[:ct]
|
|
435
|
+
assert_equal 15, parsed[:tt]
|
|
436
|
+
assert_equal 'some text here', parsed[:reas]
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
def test_backward_compat_job_and_reas_without_equals
|
|
440
|
+
parsed = Chat.parse_meta('job=WF/ask/test.chat reas=thinking about stuff')
|
|
441
|
+
assert_equal 'WF/ask/test.chat', parsed[:job]
|
|
442
|
+
assert_equal 'thinking about stuff', parsed[:reas]
|
|
443
|
+
end
|
|
444
|
+
|
|
445
|
+
def test_multiple_quoted_values_roundtrip
|
|
446
|
+
meta = { 'reas' => 'a=b', 'note' => 'c=d', 'pt' => 5 }
|
|
447
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
448
|
+
assert_equal meta, parsed
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
def test_realistic_roundtrip
|
|
452
|
+
meta = { 'pt' => 100, 'ct' => 50, 'tt' => 150,
|
|
453
|
+
'reas' => 'The user asked for x=y, so I computed z=2',
|
|
454
|
+
'job' => 'WF/ask/test.chat' }
|
|
455
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
456
|
+
assert_equal meta, parsed
|
|
457
|
+
end
|
|
458
|
+
|
|
459
|
+
# === Unescaped inner quotes in quoted values ===
|
|
460
|
+
|
|
461
|
+
def test_parse_meta_unescaped_quotes_inside_quoted_value
|
|
462
|
+
parsed = Chat.parse_meta(%q{pt=100 reas="Some text with ["/path/to/file"] inside."})
|
|
463
|
+
assert_equal 100, parsed[:pt]
|
|
464
|
+
assert_equal 'Some text with ["/path/to/file"] inside.', parsed[:reas]
|
|
465
|
+
end
|
|
466
|
+
|
|
467
|
+
def test_parse_meta_unescaped_quotes_and_keyval_patterns_inside_quoted
|
|
468
|
+
parsed = Chat.parse_meta(%q{pt=100 rt_c=10 reas="text total=48M prompt=47M end. accurate."})
|
|
469
|
+
assert_equal 100, parsed[:pt]
|
|
470
|
+
assert_equal 10, parsed[:rt_c]
|
|
471
|
+
assert_nil parsed[:total]
|
|
472
|
+
assert_nil parsed[:prompt]
|
|
473
|
+
assert_equal 'text total=48M prompt=47M end. accurate.', parsed[:reas]
|
|
474
|
+
end
|
|
475
|
+
|
|
476
|
+
def test_parse_meta_quoted_value_then_following_keys
|
|
477
|
+
parsed = Chat.parse_meta(%q{pt=100 reas="a=b" ct=200})
|
|
478
|
+
assert_equal 100, parsed[:pt]
|
|
479
|
+
assert_equal 'a=b', parsed[:reas]
|
|
480
|
+
assert_equal 200, parsed[:ct]
|
|
481
|
+
end
|
|
482
|
+
def test_backend_assigns_distinct_inference_ids
|
|
483
|
+
first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
484
|
+
second = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
485
|
+
assert_not_nil first['inference_id']
|
|
486
|
+
assert_not_equal first['inference_id'], second['inference_id']
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
def test_inference_id_controls_trace_deduplication
|
|
490
|
+
copied = chat <<-EOF
|
|
491
|
+
user: Work
|
|
492
|
+
meta: inference_id=request-one pt=2 ct=3 tt=5
|
|
493
|
+
assistant: Done
|
|
494
|
+
EOF
|
|
495
|
+
repeated = chat <<-EOF
|
|
496
|
+
user: Work
|
|
497
|
+
meta: inference_id=request-two pt=2 ct=3 tt=5
|
|
498
|
+
assistant: Done
|
|
499
|
+
EOF
|
|
500
|
+
|
|
501
|
+
assert_equal 1, Chat.trace_chats([copied, copied]).length
|
|
502
|
+
trace = Chat.trace_chats([copied, repeated])
|
|
503
|
+
assert_equal 2, trace.length
|
|
504
|
+
assert_equal [:inference_id, :inference_id], trace.collect { |entry| entry[:deduplication] }
|
|
505
|
+
end
|
|
506
|
+
|
|
507
|
+
def test_source_aware_trace_preserves_addresses
|
|
508
|
+
conversation = chat <<-EOF
|
|
509
|
+
user: Work
|
|
510
|
+
meta: inference_id=request-one tt=5
|
|
511
|
+
assistant: Done
|
|
512
|
+
EOF
|
|
513
|
+
entry = Chat.trace_chat_sources('/tmp/work.chat' => conversation).first
|
|
514
|
+
assert_equal ['/tmp/work.chat', 1], entry[:meta_address]
|
|
515
|
+
assert_equal [['/tmp/work.chat', 2]], entry[:message_addresses]
|
|
516
|
+
end
|
|
517
|
+
|
|
518
|
+
end
|
|
@@ -0,0 +1,183 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/chat'
|
|
4
|
+
require 'scout/llm/backends/responses'
|
|
5
|
+
|
|
6
|
+
class TestNormalizeUsage < Test::Unit::TestCase
|
|
7
|
+
def test_openai_chat_api
|
|
8
|
+
usage = {
|
|
9
|
+
"prompt_tokens" => 100,
|
|
10
|
+
"completion_tokens" => 50,
|
|
11
|
+
"total_tokens" => 150,
|
|
12
|
+
"prompt_tokens_details" => {"cached_tokens" => 20},
|
|
13
|
+
"completion_tokens_details" => {"reasoning_tokens" => 30}
|
|
14
|
+
}
|
|
15
|
+
result = Chat.normalize_usage(usage)
|
|
16
|
+
assert_equal 100, result['pt']
|
|
17
|
+
assert_equal 50, result['ct']
|
|
18
|
+
assert_equal 150, result['tt']
|
|
19
|
+
assert_equal 20, result['cct']
|
|
20
|
+
assert_nil result['cwt']
|
|
21
|
+
assert_equal 30, result['rt']
|
|
22
|
+
end
|
|
23
|
+
|
|
24
|
+
def test_openai_responses_api
|
|
25
|
+
usage = {
|
|
26
|
+
"input_tokens" => 9,
|
|
27
|
+
"input_tokens_details" => {"cache_write_tokens" => 0, "cached_tokens" => 0},
|
|
28
|
+
"output_tokens" => 174,
|
|
29
|
+
"output_tokens_details" => {"reasoning_tokens" => 128},
|
|
30
|
+
"total_tokens" => 183
|
|
31
|
+
}
|
|
32
|
+
result = Chat.normalize_usage(usage)
|
|
33
|
+
assert_equal 9, result['pt']
|
|
34
|
+
assert_equal 174, result['ct']
|
|
35
|
+
assert_equal 183, result['tt']
|
|
36
|
+
assert_equal 0, result['cct']
|
|
37
|
+
assert_equal 0, result['cwt']
|
|
38
|
+
assert_equal 128, result['rt']
|
|
39
|
+
end
|
|
40
|
+
|
|
41
|
+
def test_glm
|
|
42
|
+
usage = {
|
|
43
|
+
"completion_tokens" => 27,
|
|
44
|
+
"completion_tokens_details" => {"reasoning_tokens" => 92},
|
|
45
|
+
"prompt_tokens" => 8,
|
|
46
|
+
"prompt_tokens_details" => {"cached_tokens" => 0},
|
|
47
|
+
"total_tokens" => 105
|
|
48
|
+
}
|
|
49
|
+
result = Chat.normalize_usage(usage)
|
|
50
|
+
assert_equal 8, result['pt']
|
|
51
|
+
assert_equal 27, result['ct']
|
|
52
|
+
assert_equal 105, result['tt']
|
|
53
|
+
assert_equal 0, result['cct']
|
|
54
|
+
assert_nil result['cwt']
|
|
55
|
+
assert_equal 92, result['rt']
|
|
56
|
+
end
|
|
57
|
+
|
|
58
|
+
def test_anthropic_flat_fields
|
|
59
|
+
usage = {
|
|
60
|
+
"input_tokens" => 500,
|
|
61
|
+
"output_tokens" => 200,
|
|
62
|
+
"cache_read_input_tokens" => 150,
|
|
63
|
+
"cache_creation_input_tokens" => 50
|
|
64
|
+
}
|
|
65
|
+
result = Chat.normalize_usage(usage)
|
|
66
|
+
assert_equal 500, result['pt']
|
|
67
|
+
assert_equal 200, result['ct']
|
|
68
|
+
assert_equal 700, result['tt']
|
|
69
|
+
assert_equal 150, result['cct']
|
|
70
|
+
assert_equal 50, result['cwt']
|
|
71
|
+
assert_nil result['rt']
|
|
72
|
+
end
|
|
73
|
+
|
|
74
|
+
def test_simple_usage_without_cache
|
|
75
|
+
usage = {"prompt_tokens" => 10, "completion_tokens" => 5, "total_tokens" => 15}
|
|
76
|
+
result = Chat.normalize_usage(usage)
|
|
77
|
+
assert_equal 10, result['pt']
|
|
78
|
+
assert_equal 5, result['ct']
|
|
79
|
+
assert_equal 15, result['tt']
|
|
80
|
+
extras = result.select { |k,_v| !%w[pt ct tt].include?(k) }
|
|
81
|
+
assert_empty extras
|
|
82
|
+
end
|
|
83
|
+
|
|
84
|
+
def test_nil_usage
|
|
85
|
+
result = Chat.normalize_usage(nil)
|
|
86
|
+
assert_equal({}, result)
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
def test_empty_usage
|
|
90
|
+
result = Chat.normalize_usage({})
|
|
91
|
+
assert_equal({}, result)
|
|
92
|
+
end
|
|
93
|
+
|
|
94
|
+
def test_token_totals_with_cache
|
|
95
|
+
usage1 = {
|
|
96
|
+
"prompt_tokens" => 100,
|
|
97
|
+
"completion_tokens" => 50,
|
|
98
|
+
"total_tokens" => 150,
|
|
99
|
+
"prompt_tokens_details" => {"cached_tokens" => 20},
|
|
100
|
+
"completion_tokens_details" => {"reasoning_tokens" => 30}
|
|
101
|
+
}
|
|
102
|
+
meta_str1 = Chat.serialize_meta(Chat.normalize_usage(usage1))
|
|
103
|
+
chat1 = Chat.setup([
|
|
104
|
+
{role: :user, content: "hi"},
|
|
105
|
+
{role: :assistant, content: "hello"},
|
|
106
|
+
{role: :meta, content: meta_str1}
|
|
107
|
+
])
|
|
108
|
+
|
|
109
|
+
totals = Chat.token_totals([chat1])
|
|
110
|
+
assert_equal 100, totals[:pt]
|
|
111
|
+
assert_equal 50, totals[:ct]
|
|
112
|
+
assert_equal 150, totals[:tt]
|
|
113
|
+
assert_equal 20, totals[:cct]
|
|
114
|
+
assert_equal 0, totals[:cwt]
|
|
115
|
+
assert_equal 30, totals[:rt]
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
def test_print_tokens_with_cache
|
|
119
|
+
tokens = {pt: 100, ct: 50, tt: 150, cct: 20, cwt: 0, rt: 30}
|
|
120
|
+
str = Chat.print_tokens(tokens)
|
|
121
|
+
assert str.include?("prompt=100")
|
|
122
|
+
assert str.include?("completion=50")
|
|
123
|
+
assert str.include?("total=150")
|
|
124
|
+
assert str.include?("cached=20")
|
|
125
|
+
assert !str.include?("cache_write")
|
|
126
|
+
assert str.include?("reasoning=30")
|
|
127
|
+
end
|
|
128
|
+
|
|
129
|
+
def test_token_keys_constant
|
|
130
|
+
assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
|
|
131
|
+
assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def test_update_meta_with_cache_fields
|
|
135
|
+
%w(pt_s ct_s tt_s cct_s cwt_s rt_s).each { |name| Thread.current[name] = 0 }
|
|
136
|
+
|
|
137
|
+
response = {
|
|
138
|
+
'usage' => {
|
|
139
|
+
"prompt_tokens" => 100,
|
|
140
|
+
"completion_tokens" => 50,
|
|
141
|
+
"total_tokens" => 150,
|
|
142
|
+
"prompt_tokens_details" => {"cached_tokens" => 20},
|
|
143
|
+
"completion_tokens_details" => {"reasoning_tokens" => 30}
|
|
144
|
+
}
|
|
145
|
+
}
|
|
146
|
+
meta = LLM::Responses.update_meta(response)
|
|
147
|
+
assert_equal 100, meta['pt']
|
|
148
|
+
assert_equal 50, meta['ct']
|
|
149
|
+
assert_equal 150, meta['tt']
|
|
150
|
+
assert_equal 20, meta['cct']
|
|
151
|
+
assert_equal 30, meta['rt']
|
|
152
|
+
assert_equal 20, meta['cct_s']
|
|
153
|
+
assert_equal 30, meta['rt_s']
|
|
154
|
+
assert_equal 20, meta['cct_c']
|
|
155
|
+
assert_equal 30, meta['rt_c']
|
|
156
|
+
end
|
|
157
|
+
|
|
158
|
+
def test_update_meta_accumulates_cache_fields_across_requests
|
|
159
|
+
%w(pt_s ct_s tt_s cct_s cwt_s rt_s).each { |name| Thread.current[name] = 0 }
|
|
160
|
+
|
|
161
|
+
resp1 = { 'usage' => {
|
|
162
|
+
"prompt_tokens" => 10, "completion_tokens" => 5, "total_tokens" => 15,
|
|
163
|
+
"prompt_tokens_details" => {"cached_tokens" => 4},
|
|
164
|
+
"completion_tokens_details" => {"reasoning_tokens" => 3}
|
|
165
|
+
}}
|
|
166
|
+
meta1 = LLM::Responses.update_meta(resp1)
|
|
167
|
+
assert_equal 4, meta1['cct_s']
|
|
168
|
+
assert_equal 3, meta1['rt_s']
|
|
169
|
+
assert_equal 4, meta1['cct_c']
|
|
170
|
+
assert_equal 3, meta1['rt_c']
|
|
171
|
+
|
|
172
|
+
resp2 = { 'usage' => {
|
|
173
|
+
"prompt_tokens" => 20, "completion_tokens" => 10, "total_tokens" => 30,
|
|
174
|
+
"prompt_tokens_details" => {"cached_tokens" => 8},
|
|
175
|
+
"completion_tokens_details" => {"reasoning_tokens" => 6}
|
|
176
|
+
}}
|
|
177
|
+
meta2 = LLM::Responses.update_meta(resp2, meta1)
|
|
178
|
+
assert_equal 12, meta2['cct_s'] # 4 + 8
|
|
179
|
+
assert_equal 9, meta2['rt_s'] # 3 + 6
|
|
180
|
+
assert_equal 12, meta2['cct_c'] # 4 + 8
|
|
181
|
+
assert_equal 9, meta2['rt_c'] # 3 + 6
|
|
182
|
+
end
|
|
183
|
+
end
|