scout-ai 1.2.5 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +111 -31
  3. data/README.md +171 -320
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +62 -12
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +36 -6
  33. data/lib/scout/llm/backends/anthropic.rb +9 -1
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +129 -89
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +13 -1
  38. data/lib/scout/llm/backends/ollama.rb +13 -0
  39. data/lib/scout/llm/backends/openwebui.rb +8 -3
  40. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  41. data/lib/scout/llm/chat/annotation.rb +37 -14
  42. data/lib/scout/llm/chat/parse.rb +28 -6
  43. data/lib/scout/llm/chat/persist.rb +25 -0
  44. data/lib/scout/llm/chat/process/clear.rb +19 -1
  45. data/lib/scout/llm/chat/process/files.rb +16 -1
  46. data/lib/scout/llm/chat/process/meta.rb +422 -32
  47. data/lib/scout/llm/chat/process/options.rb +21 -1
  48. data/lib/scout/llm/chat/process/tools.rb +49 -12
  49. data/lib/scout/llm/chat/process.rb +4 -0
  50. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  51. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  52. data/lib/scout/llm/chat/prompt.rb +48 -0
  53. data/lib/scout/llm/chat/provenance.rb +775 -0
  54. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  55. data/lib/scout/llm/chat.rb +18 -2
  56. data/lib/scout/llm/embed.rb +8 -3
  57. data/lib/scout/llm/image.rb +86 -0
  58. data/lib/scout/llm/rag.rb +3 -3
  59. data/lib/scout/llm/tools/call.rb +159 -10
  60. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  61. data/lib/scout/llm/tools/workflow.rb +13 -5
  62. data/lib/scout-ai.rb +1 -0
  63. data/research/ChatAnalyst-required-changes.md +167 -0
  64. data/research/agent-delegation-analysis.md +810 -0
  65. data/research/agent-meta-provenance-integration-plan.md +622 -0
  66. data/research/agent-workflow-analysis.md +1120 -0
  67. data/research/backends-analysis.md +836 -0
  68. data/research/chat-core-analysis.md +946 -0
  69. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  70. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  71. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  72. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  73. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  74. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  75. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  76. data/research/chatanalyst-provenance/final-report.md +45 -0
  77. data/research/chatanalyst-provenance/resumption.md +37 -0
  78. data/research/coding-philosophy-analysis.md +928 -0
  79. data/research/commands-analysis.md +947 -0
  80. data/research/multi-agent-patterns-analysis.md +853 -0
  81. data/research/prompt-strategies-analysis.md +630 -0
  82. data/research/prov-verbosity-fix-notes.md +77 -0
  83. data/research/provenance-analysis.md +469 -0
  84. data/research/provenance-navigation-design.md +640 -0
  85. data/research/synthesis-report.md +487 -0
  86. data/research/tools-system-analysis.md +779 -0
  87. data/scout-ai.gemspec +97 -13
  88. data/scout_commands/agent/ask +13 -3
  89. data/scout_commands/agent/kb +2 -0
  90. data/scout_commands/llm/ask +11 -4
  91. data/scout_commands/llm/md +76 -0
  92. data/scout_commands/llm/prov +602 -0
  93. data/scout_commands/llm/word +71 -0
  94. data/share/word/reference.docx +0 -0
  95. data/test/etc/AI/mock.yaml +11 -0
  96. data/test/fixtures/backends/anthropic.json +19 -0
  97. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  98. data/test/fixtures/backends/bedrock.json +8 -0
  99. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  100. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  101. data/test/fixtures/backends/ollama.json +16 -0
  102. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  103. data/test/fixtures/backends/openai_chat.json +21 -0
  104. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  105. data/test/fixtures/backends/responses.json +33 -0
  106. data/test/fixtures/backends/responses_tool_call.json +28 -0
  107. data/test/integration/README.md +32 -0
  108. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  109. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  110. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  111. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  112. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  113. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  114. data/test/integration/scout/model/test_base.rb +91 -0
  115. data/test/scout/llm/agent/test_chat.rb +8 -2
  116. data/test/scout/llm/agent/test_save.rb +413 -0
  117. data/test/scout/llm/agent/test_workflow.rb +110 -0
  118. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  119. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  120. data/test/scout/llm/backends/test_ollama.rb +70 -20
  121. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  122. data/test/scout/llm/backends/test_relay.rb +4 -2
  123. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  124. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  125. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  126. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  127. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  128. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  129. data/test/scout/llm/chat/test_parse.rb +70 -15
  130. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  131. data/test/scout/llm/chat/test_provenance.rb +240 -0
  132. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  133. data/test/scout/llm/test_agent.rb +13 -36
  134. data/test/scout/llm/test_ask.rb +75 -52
  135. data/test/scout/llm/test_chat.rb +107 -13
  136. data/test/scout/llm/test_embed.rb +48 -0
  137. data/test/scout/llm/test_rag.rb +23 -16
  138. data/test/scout/llm/test_tools.rb +12 -1
  139. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  140. data/test/scout/llm/tools/test_mcp.rb +5 -3
  141. data/test/scout/llm/tools/test_workflow.rb +23 -2
  142. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  143. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  144. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  145. data/test/scout/model/python/test_torch.rb +2 -0
  146. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  147. data/test/scout/model/test_base.rb +4 -2
  148. data/test/support/availability.rb +231 -0
  149. data/test/support/fake_clients.rb +138 -0
  150. data/test/support/fixtures.rb +21 -0
  151. data/test/support/infrastructure_probes.rb +136 -0
  152. data/test/support/mock_backend.rb +215 -0
  153. data/test/test_helper.rb +32 -2
  154. metadata +96 -12
  155. data/doc/Agent.md +0 -354
  156. data/doc/Chat.md +0 -481
  157. data/doc/LLM.md +0 -356
  158. data/doc/PythonAgentTasks.md +0 -333
  159. data/doc/RAG.md +0 -129
  160. data/doc/USER_GUIDE.md +0 -572
  161. data/scout_commands/documenter +0 -148
  162. data/test/scout/llm/backends/test_openai.rb +0 -192
  163. data/test/scout/llm/backends/test_responses.rb +0 -238
  164. data/test/scout/llm/test_parse.rb +0 -98
@@ -0,0 +1,518 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
+
4
+ require 'scout/llm/chat'
5
+ require 'scout/llm/backends/responses'
6
+
7
+ class TestLLMUsageMeta < Test::Unit::TestCase
8
+ def setup
9
+ super
10
+ Chat::TOKEN_KEYS.each { |name| Thread.current["#{name}_s"] = 0 }
11
+ end
12
+
13
+ def response(prompt: nil, completion: nil, total: nil,
14
+ cached: nil, cache_write: nil, reasoning: nil)
15
+ usage = {}
16
+ usage['prompt_tokens'] = prompt unless prompt.nil?
17
+ usage['completion_tokens'] = completion unless completion.nil?
18
+ usage['total_tokens'] = total unless total.nil?
19
+ usage['prompt_tokens_details'] = {}
20
+ usage['prompt_tokens_details']['cached_tokens'] = cached unless cached.nil?
21
+ usage['input_tokens_details'] = {} if cache_write || cached
22
+ usage['input_tokens_details'] ||= {}
23
+ usage['input_tokens_details']['cache_write_tokens'] = cache_write unless cache_write.nil?
24
+ usage['completion_tokens_details'] = {}
25
+ usage['completion_tokens_details']['reasoning_tokens'] = reasoning unless reasoning.nil?
26
+ { 'usage' => usage }
27
+ end
28
+
29
+ def chat(text)
30
+ Chat.setup(LLM.messages(text))
31
+ end
32
+
33
+ def test_backend_records_direct_and_running_token_counts
34
+ first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
35
+ second = LLM::Responses.update_meta(response(prompt: 7, total: 7), first)
36
+
37
+ assert_equal 7, second['pt']
38
+ assert_nil second['ct']
39
+ assert_equal 9, second['pt_s']
40
+ assert_equal 12, second['tt_s']
41
+ assert_equal 9, second['pt_c']
42
+ assert_equal 3, second['ct_c']
43
+ assert_equal 12, second['tt_c']
44
+ assert_nil second['usage_id']
45
+ end
46
+
47
+ def test_jobs_returns_all_projecting_jobs
48
+ conversation = chat <<-EOF
49
+ user: First
50
+ meta: job=WF/ask/first.chat
51
+ assistant: First answer
52
+ user: Second
53
+ meta: job=WF/ask/second.chat
54
+ assistant: Second answer
55
+ EOF
56
+
57
+ assert_equal %w[WF/ask/first.chat WF/ask/second.chat], conversation.jobs
58
+ assert_equal 'WF/ask/second.chat', conversation.meta[:job]
59
+ end
60
+
61
+ def test_message_identity_includes_non_meta_history
62
+ first = chat <<-EOF
63
+ user: Question
64
+ meta: tt=5
65
+ assistant: Answer
66
+ EOF
67
+ same = chat <<-EOF
68
+ user: Question
69
+ assistant: Answer
70
+ EOF
71
+ different = chat <<-EOF
72
+ user: Different question
73
+ assistant: Answer
74
+ EOF
75
+
76
+ assert_equal first.message_index.last[:id], same.message_index.last[:id]
77
+ assert_not_equal first.message_index.last[:id], different.message_index.last[:id]
78
+ end
79
+
80
+ def test_consecutive_meta_leaves_the_first_segment_orphaned
81
+ conversation = chat <<-EOF
82
+ user: Work
83
+ meta: tt=2
84
+ meta: job=WF/ask/work.chat
85
+ assistant: Done
86
+ EOF
87
+
88
+ trace = Chat.trace_chats([conversation])
89
+ assert_equal 2, trace.length
90
+ assert trace.first[:orphan]
91
+ assert_equal 2, trace.first[:meta][:tt]
92
+ assert_equal 'WF/ask/work.chat', trace.last[:meta][:job]
93
+ assert_equal 1, trace.last[:messages].length
94
+ end
95
+
96
+ def test_final_meta_is_an_orphan_segment
97
+ conversation = chat <<-EOF
98
+ user: Work
99
+ meta: tt=2
100
+ assistant: Tool call removed
101
+ meta: tt=7
102
+ EOF
103
+
104
+ trace = Chat.trace_chats([conversation])
105
+ assert_equal 2, trace.length
106
+ assert_equal 7, trace.last[:meta][:tt]
107
+ assert trace.last[:orphan]
108
+ assert_empty trace.last[:messages]
109
+ end
110
+
111
+ def test_meta_covers_a_multi_tool_response_segment
112
+ conversation = chat <<-EOF
113
+ user: Write two files
114
+ meta: tt=1000
115
+ function_call: {"name":"write","id":"one"}
116
+ function_call_output: {"id":"one","content":"done one"}
117
+ function_call: {"name":"write","id":"two"}
118
+ function_call_output: {"id":"two","content":"done two"}
119
+ assistant: Done
120
+ user: Next request
121
+ EOF
122
+
123
+ trace = Chat.trace_chats([conversation])
124
+ assert_equal 1, trace.length
125
+ assert_equal 1000, trace.first[:meta][:tt]
126
+ assert_equal 5, trace.first[:messages].length
127
+ assert !trace.first[:orphan]
128
+ end
129
+
130
+ def test_project_keeps_inference_meta_inline_with_one_job_marker
131
+ response = [
132
+ { role: :meta, content: 'tt=2' },
133
+ { role: :function_call, content: '{"name":"write"}' },
134
+ { role: :function_call_output, content: '{"content":"done"}' },
135
+ { role: :meta, content: 'tt=7' },
136
+ { role: :assistant, content: 'Done' }
137
+ ]
138
+
139
+ projected = Chat.project('WF/ask/work.chat', response)
140
+ assert_equal %i[meta meta function_call function_call_output meta assistant], projected.collect { |m| m[:role] }
141
+
142
+ marker = Chat.parse_meta(projected.first[:content])
143
+ assert_equal 'WF/ask/work.chat', marker[:job]
144
+ assert Chat::TOKEN_KEYS.none? { |key| marker.include?(key) }
145
+
146
+ assert_equal 2, Chat.parse_meta(projected[1][:content])[:tt]
147
+ assert_equal :function_call, projected[2][:role], 'first inference meta stays adjacent to the call it produced'
148
+ assert_equal 7, Chat.parse_meta(projected[4][:content])[:tt]
149
+
150
+ trace = Chat.trace_chats([Chat.setup(projected)])
151
+ assert_equal 3, trace.length
152
+ assert trace.first[:orphan]
153
+ assert_equal [2, 7], trace[1..-1].collect { |entry| entry[:meta][:tt] }
154
+ assert_equal [2, 1], trace[1..-1].collect { |entry| entry[:messages].length }
155
+ assert trace[1..-1].none? { |entry| entry[:orphan] }
156
+
157
+ assert_equal 2, Chat.direct_entries([Chat.setup(projected)]).length
158
+ totals = Chat.token_totals([Chat.setup(projected)])
159
+ assert_equal 9, totals[:tt]
160
+ end
161
+
162
+ # Legacy chats carry no inference_id, so Chat.trace_indices falls back to the
163
+ # digest-based lineage id. The lineage id is computed from the preceding
164
+ # messages, and a projected copy sits behind a leading `job=` marker, so the
165
+ # projected and original copies of the same legacy inference get DIFFERENT
166
+ # lineage ids and are both counted. This is the documented legacy behaviour:
167
+ # precise deduplication requires inference_id, which every new inference has.
168
+ def test_project_legacy_meta_without_inference_id_is_not_deduplicated_across_chats
169
+ original = chat <<-EOF
170
+ user: Work
171
+ meta: pt=2 ct=1 tt=3
172
+ assistant: Done
173
+ EOF
174
+
175
+ projected = Chat.project('WF/ask/work.chat', [
176
+ { role: :meta, content: 'pt=2 ct=1 tt=3' },
177
+ { role: :assistant, content: 'Done' }
178
+ ])
179
+
180
+ trace = Chat.trace_chats([Chat.setup(projected), original])
181
+ assert_equal 3, trace.length, 'job marker + two non-merged legacy lineages'
182
+ assert trace.all? { |entry| entry[:deduplication] == :legacy_lineage }
183
+ assert_not_equal trace.first[:lineage_id], trace.last[:lineage_id]
184
+
185
+ projected_totals = Chat.token_totals([Chat.setup(projected)])
186
+ assert_equal 3, projected_totals[:tt]
187
+ assert_equal 3, Chat.token_totals([original])[:tt]
188
+ assert_equal 6, Chat.token_totals([Chat.setup(projected), original])[:tt]
189
+ end
190
+
191
+ def test_project_consumption_path_does_not_double_count_a_saved_projection
192
+ TmpFile.with_file(nil, false, :persistent => true) do |file|
193
+ original = chat <<-EOF
194
+ user: Work
195
+ meta: inference_id=request-one pt=10 ct=5 tt=15
196
+ assistant: Done
197
+ EOF
198
+
199
+ # chat_task: the job result chat is the projection; the log keeps the
200
+ # original metas.
201
+ projected = Chat.project('WF/ask/work.chat', [
202
+ { role: :meta, content: 'inference_id=request-one pt=10 ct=5 tt=15' },
203
+ { role: :assistant, content: 'Done' }
204
+ ])
205
+ Open.write(file, Chat.print(Chat.setup(projected)))
206
+
207
+ # LLM::Agent#ask consumption path: load the persisted job chat and
208
+ # re-project it before counting.
209
+ loaded = Chat.load(file)
210
+ reprojected = Chat.project('WF/ask/work.chat', loaded)
211
+
212
+ assert_equal Chat.token_totals([original]), Chat.token_totals([Chat.setup(reprojected), original])
213
+ end
214
+ end
215
+ def test_trace_keeps_distinct_segments_for_direct_and_projected_metadata
216
+ direct = chat <<-EOF
217
+ user: Work
218
+ meta: tt=7
219
+ assistant: Done
220
+ EOF
221
+ projected = chat <<-EOF
222
+ user: Work
223
+ meta: job=WF/ask/work.chat
224
+ assistant: Done
225
+ EOF
226
+
227
+ trace = Chat.trace_chats([projected, direct])
228
+ assert_equal 2, trace.length
229
+ assert_equal ['WF/ask/work.chat', nil], trace.collect { |entry| entry[:meta][:job] }
230
+ assert_equal [nil, 7], trace.collect { |entry| entry[:meta][:tt] }
231
+ end
232
+
233
+ def test_job_meta_does_not_reset_the_last_direct_chat_total
234
+ messages = LLM.messages <<-EOF
235
+ user: Plan
236
+ meta: pt=10 ct=2 tt=12 pt_c=10 ct_c=2 tt_c=12
237
+ assistant: Plan complete
238
+ meta: job=WF/ask/work.chat
239
+ assistant: Work complete
240
+ EOF
241
+
242
+ current = Chat.meta(messages)
243
+ assert_equal 'WF/ask/work.chat', current[:job]
244
+ assert_equal 10, current[:pt_c]
245
+ assert_equal 2, current[:ct_c]
246
+ assert_equal 12, current[:tt_c]
247
+ end
248
+
249
+ # === Cache token accounting tests ===
250
+
251
+ def test_openai_responses_api_cache_tokens
252
+ resp = { 'usage' => {
253
+ 'input_tokens' => 9,
254
+ 'input_tokens_details' => { 'cache_write_tokens' => 5, 'cached_tokens' => 3 },
255
+ 'output_tokens' => 174,
256
+ 'output_tokens_details' => { 'reasoning_tokens' => 128 },
257
+ 'total_tokens' => 183
258
+ } }
259
+ meta = LLM::Responses.update_meta(resp)
260
+
261
+ assert_equal 9, meta['pt']
262
+ assert_equal 174, meta['ct']
263
+ assert_equal 183, meta['tt']
264
+ assert_equal 3, meta['cct']
265
+ assert_equal 5, meta['cwt']
266
+ assert_equal 128, meta['rt']
267
+ # cumulative variants
268
+ assert_equal 9, meta['pt_c']
269
+ assert_equal 3, meta['cct_c']
270
+ assert_equal 5, meta['cwt_c']
271
+ assert_equal 128, meta['rt_c']
272
+ # session variants
273
+ assert_equal 9, meta['pt_s']
274
+ assert_equal 3, meta['cct_s']
275
+ assert_equal 5, meta['cwt_s']
276
+ assert_equal 128, meta['rt_s']
277
+ end
278
+
279
+ def test_glm_cache_tokens
280
+ resp = { 'usage' => {
281
+ 'completion_tokens' => 97,
282
+ 'completion_tokens_details' => { 'reasoning_tokens' => 92 },
283
+ 'prompt_tokens' => 8,
284
+ 'prompt_tokens_details' => { 'cached_tokens' => 4 },
285
+ 'total_tokens' => 105
286
+ } }
287
+ meta = LLM::Responses.update_meta(resp)
288
+
289
+ assert_equal 8, meta['pt']
290
+ assert_equal 97, meta['ct']
291
+ assert_equal 105, meta['tt']
292
+ assert_equal 4, meta['cct']
293
+ assert_nil meta['cwt']
294
+ assert_equal 92, meta['rt']
295
+ # cumulative variants
296
+ assert_equal 4, meta['cct_c']
297
+ assert_equal 92, meta['rt_c']
298
+ end
299
+
300
+ def test_anthropic_flat_cache_fields
301
+ resp = { 'usage' => {
302
+ 'prompt_tokens' => 100,
303
+ 'completion_tokens' => 50,
304
+ 'cache_read_input_tokens' => 80,
305
+ 'cache_creation_input_tokens' => 20
306
+ } }
307
+ meta = LLM::Responses.update_meta(resp)
308
+
309
+ assert_equal 100, meta['pt']
310
+ assert_equal 50, meta['ct']
311
+ assert_equal 150, meta['tt'] # computed
312
+ assert_equal 80, meta['cct']
313
+ assert_equal 20, meta['cwt']
314
+ assert_nil meta['rt']
315
+ end
316
+
317
+ def test_cumulative_cache_tokens_across_requests
318
+ first = LLM::Responses.update_meta(
319
+ response(prompt: 10, completion: 5, total: 15, cached: 3, reasoning: 2)
320
+ )
321
+ second = LLM::Responses.update_meta(
322
+ response(prompt: 8, completion: 4, total: 12, cached: 6, reasoning: 1),
323
+ first
324
+ )
325
+
326
+ assert_equal 3, first['cct_c']
327
+ assert_equal 2, first['rt_c']
328
+ assert_equal 9, second['cct_c'] # 3 + 6
329
+ assert_equal 3, second['rt_c'] # 2 + 1
330
+ assert_equal 18, second['pt_c'] # 10 + 8
331
+ end
332
+
333
+ def test_normalize_usage_constants
334
+ assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
335
+ assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
336
+ end
337
+
338
+ def test_normalize_usage_openai_chat_api
339
+ usage = { 'prompt_tokens' => 9, 'completion_tokens' => 174, 'total_tokens' => 183 }
340
+ result = Chat.normalize_usage(usage)
341
+ assert_equal 9, result['pt']
342
+ assert_equal 174, result['ct']
343
+ assert_equal 183, result['tt']
344
+ assert_nil result['cct']
345
+ assert_nil result['cwt']
346
+ assert_nil result['rt']
347
+ end
348
+
349
+ def test_normalize_usage_computes_total_when_missing
350
+ usage = { 'prompt_tokens' => 10, 'completion_tokens' => 20 }
351
+ result = Chat.normalize_usage(usage)
352
+ assert_equal 30, result['tt']
353
+ end
354
+
355
+ def test_normalize_usage_handles_nil
356
+ assert_equal({}, Chat.normalize_usage(nil))
357
+ assert_equal({}, Chat.normalize_usage({}))
358
+ end
359
+
360
+ def test_direct_entries_includes_cache_tokens
361
+ conversation = chat <<-EOF
362
+ user: Work
363
+ meta: pt=10 ct=5 tt=15 cct=3 rt=2
364
+ assistant: Done
365
+ EOF
366
+
367
+ entries = Chat.direct_entries([conversation])
368
+ assert_equal 1, entries.length
369
+ assert_equal 3, entries.first[:meta][:cct]
370
+ assert_equal 2, entries.first[:meta][:rt]
371
+ end
372
+
373
+ def test_token_totals_aggregates_cache_fields
374
+ c1 = chat <<-EOF
375
+ user: Work
376
+ meta: pt=10 ct=5 tt=15 cct=3 rt=2
377
+ assistant: Done
378
+ EOF
379
+ c2 = chat <<-EOF
380
+ user: More work
381
+ meta: pt=20 ct=10 tt=30 cct=7 rt=8
382
+ assistant: Done again
383
+ EOF
384
+
385
+ totals = Chat.token_totals([c1, c2])
386
+ assert_equal 30, totals[:pt]
387
+ assert_equal 15, totals[:ct]
388
+ assert_equal 45, totals[:tt]
389
+ assert_equal 10, totals[:cct] # 3 + 7
390
+ assert_equal 10, totals[:rt] # 2 + 8
391
+ end
392
+
393
+ def test_print_tokens_shows_cache_fields
394
+ totals = { pt: 100, ct: 50, tt: 150, cct: 30, cwt: 10, rt: 20 }
395
+ output = Chat.print_tokens(totals)
396
+ assert output.include?('cached=30')
397
+ assert output.include?('cache_write=10')
398
+ assert output.include?('reasoning=20')
399
+ end
400
+
401
+ # === Quoted-value serialization/parsing tests ===
402
+
403
+ def test_serialize_meta_quotes_value_containing_equals
404
+ serialized = Chat.serialize_meta('reas' => 'thinking about a=b')
405
+ assert_equal 'reas="thinking about a=b"', serialized
406
+ end
407
+
408
+ def test_serialize_meta_does_not_quote_simple_values
409
+ serialized = Chat.serialize_meta('pt' => 10, 'job' => 'WF/ask/test.chat')
410
+ assert_equal 'pt=10 job=WF/ask/test.chat', serialized
411
+ end
412
+
413
+ def test_parse_meta_handles_quoted_value_with_equals
414
+ parsed = Chat.parse_meta('pt=10 reas="thinking about a=b"')
415
+ assert_equal 10, parsed[:pt]
416
+ assert_equal 'thinking about a=b', parsed[:reas]
417
+ end
418
+
419
+ def test_roundtrip_value_with_equals
420
+ meta = { 'pt' => 10, 'reas' => 'thinking about a=b and c=d', 'job' => 'WF/ask/test.chat' }
421
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
422
+ assert_equal meta, parsed
423
+ end
424
+
425
+ def test_roundtrip_value_with_quotes_and_backslashes
426
+ meta = { 'reas' => 'He said "hello" and a=b\c', 'pt' => 5 }
427
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
428
+ assert_equal meta, parsed
429
+ end
430
+
431
+ def test_backward_compat_unquoted_value_with_spaces
432
+ parsed = Chat.parse_meta('pt=10 ct=5 tt=15 reas=some text here')
433
+ assert_equal 10, parsed[:pt]
434
+ assert_equal 5, parsed[:ct]
435
+ assert_equal 15, parsed[:tt]
436
+ assert_equal 'some text here', parsed[:reas]
437
+ end
438
+
439
+ def test_backward_compat_job_and_reas_without_equals
440
+ parsed = Chat.parse_meta('job=WF/ask/test.chat reas=thinking about stuff')
441
+ assert_equal 'WF/ask/test.chat', parsed[:job]
442
+ assert_equal 'thinking about stuff', parsed[:reas]
443
+ end
444
+
445
+ def test_multiple_quoted_values_roundtrip
446
+ meta = { 'reas' => 'a=b', 'note' => 'c=d', 'pt' => 5 }
447
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
448
+ assert_equal meta, parsed
449
+ end
450
+
451
+ def test_realistic_roundtrip
452
+ meta = { 'pt' => 100, 'ct' => 50, 'tt' => 150,
453
+ 'reas' => 'The user asked for x=y, so I computed z=2',
454
+ 'job' => 'WF/ask/test.chat' }
455
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
456
+ assert_equal meta, parsed
457
+ end
458
+
459
+ # === Unescaped inner quotes in quoted values ===
460
+
461
+ def test_parse_meta_unescaped_quotes_inside_quoted_value
462
+ parsed = Chat.parse_meta(%q{pt=100 reas="Some text with ["/path/to/file"] inside."})
463
+ assert_equal 100, parsed[:pt]
464
+ assert_equal 'Some text with ["/path/to/file"] inside.', parsed[:reas]
465
+ end
466
+
467
+ def test_parse_meta_unescaped_quotes_and_keyval_patterns_inside_quoted
468
+ parsed = Chat.parse_meta(%q{pt=100 rt_c=10 reas="text total=48M prompt=47M end. accurate."})
469
+ assert_equal 100, parsed[:pt]
470
+ assert_equal 10, parsed[:rt_c]
471
+ assert_nil parsed[:total]
472
+ assert_nil parsed[:prompt]
473
+ assert_equal 'text total=48M prompt=47M end. accurate.', parsed[:reas]
474
+ end
475
+
476
+ def test_parse_meta_quoted_value_then_following_keys
477
+ parsed = Chat.parse_meta(%q{pt=100 reas="a=b" ct=200})
478
+ assert_equal 100, parsed[:pt]
479
+ assert_equal 'a=b', parsed[:reas]
480
+ assert_equal 200, parsed[:ct]
481
+ end
482
+ def test_backend_assigns_distinct_inference_ids
483
+ first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
484
+ second = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
485
+ assert_not_nil first['inference_id']
486
+ assert_not_equal first['inference_id'], second['inference_id']
487
+ end
488
+
489
+ def test_inference_id_controls_trace_deduplication
490
+ copied = chat <<-EOF
491
+ user: Work
492
+ meta: inference_id=request-one pt=2 ct=3 tt=5
493
+ assistant: Done
494
+ EOF
495
+ repeated = chat <<-EOF
496
+ user: Work
497
+ meta: inference_id=request-two pt=2 ct=3 tt=5
498
+ assistant: Done
499
+ EOF
500
+
501
+ assert_equal 1, Chat.trace_chats([copied, copied]).length
502
+ trace = Chat.trace_chats([copied, repeated])
503
+ assert_equal 2, trace.length
504
+ assert_equal [:inference_id, :inference_id], trace.collect { |entry| entry[:deduplication] }
505
+ end
506
+
507
+ def test_source_aware_trace_preserves_addresses
508
+ conversation = chat <<-EOF
509
+ user: Work
510
+ meta: inference_id=request-one tt=5
511
+ assistant: Done
512
+ EOF
513
+ entry = Chat.trace_chat_sources('/tmp/work.chat' => conversation).first
514
+ assert_equal ['/tmp/work.chat', 1], entry[:meta_address]
515
+ assert_equal [['/tmp/work.chat', 2]], entry[:message_addresses]
516
+ end
517
+
518
+ end
@@ -0,0 +1,183 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/chat'
4
+ require 'scout/llm/backends/responses'
5
+
6
+ class TestNormalizeUsage < Test::Unit::TestCase
7
+ def test_openai_chat_api
8
+ usage = {
9
+ "prompt_tokens" => 100,
10
+ "completion_tokens" => 50,
11
+ "total_tokens" => 150,
12
+ "prompt_tokens_details" => {"cached_tokens" => 20},
13
+ "completion_tokens_details" => {"reasoning_tokens" => 30}
14
+ }
15
+ result = Chat.normalize_usage(usage)
16
+ assert_equal 100, result['pt']
17
+ assert_equal 50, result['ct']
18
+ assert_equal 150, result['tt']
19
+ assert_equal 20, result['cct']
20
+ assert_nil result['cwt']
21
+ assert_equal 30, result['rt']
22
+ end
23
+
24
+ def test_openai_responses_api
25
+ usage = {
26
+ "input_tokens" => 9,
27
+ "input_tokens_details" => {"cache_write_tokens" => 0, "cached_tokens" => 0},
28
+ "output_tokens" => 174,
29
+ "output_tokens_details" => {"reasoning_tokens" => 128},
30
+ "total_tokens" => 183
31
+ }
32
+ result = Chat.normalize_usage(usage)
33
+ assert_equal 9, result['pt']
34
+ assert_equal 174, result['ct']
35
+ assert_equal 183, result['tt']
36
+ assert_equal 0, result['cct']
37
+ assert_equal 0, result['cwt']
38
+ assert_equal 128, result['rt']
39
+ end
40
+
41
+ def test_glm
42
+ usage = {
43
+ "completion_tokens" => 27,
44
+ "completion_tokens_details" => {"reasoning_tokens" => 92},
45
+ "prompt_tokens" => 8,
46
+ "prompt_tokens_details" => {"cached_tokens" => 0},
47
+ "total_tokens" => 105
48
+ }
49
+ result = Chat.normalize_usage(usage)
50
+ assert_equal 8, result['pt']
51
+ assert_equal 27, result['ct']
52
+ assert_equal 105, result['tt']
53
+ assert_equal 0, result['cct']
54
+ assert_nil result['cwt']
55
+ assert_equal 92, result['rt']
56
+ end
57
+
58
+ def test_anthropic_flat_fields
59
+ usage = {
60
+ "input_tokens" => 500,
61
+ "output_tokens" => 200,
62
+ "cache_read_input_tokens" => 150,
63
+ "cache_creation_input_tokens" => 50
64
+ }
65
+ result = Chat.normalize_usage(usage)
66
+ assert_equal 500, result['pt']
67
+ assert_equal 200, result['ct']
68
+ assert_equal 700, result['tt']
69
+ assert_equal 150, result['cct']
70
+ assert_equal 50, result['cwt']
71
+ assert_nil result['rt']
72
+ end
73
+
74
+ def test_simple_usage_without_cache
75
+ usage = {"prompt_tokens" => 10, "completion_tokens" => 5, "total_tokens" => 15}
76
+ result = Chat.normalize_usage(usage)
77
+ assert_equal 10, result['pt']
78
+ assert_equal 5, result['ct']
79
+ assert_equal 15, result['tt']
80
+ extras = result.select { |k,_v| !%w[pt ct tt].include?(k) }
81
+ assert_empty extras
82
+ end
83
+
84
+ def test_nil_usage
85
+ result = Chat.normalize_usage(nil)
86
+ assert_equal({}, result)
87
+ end
88
+
89
+ def test_empty_usage
90
+ result = Chat.normalize_usage({})
91
+ assert_equal({}, result)
92
+ end
93
+
94
+ def test_token_totals_with_cache
95
+ usage1 = {
96
+ "prompt_tokens" => 100,
97
+ "completion_tokens" => 50,
98
+ "total_tokens" => 150,
99
+ "prompt_tokens_details" => {"cached_tokens" => 20},
100
+ "completion_tokens_details" => {"reasoning_tokens" => 30}
101
+ }
102
+ meta_str1 = Chat.serialize_meta(Chat.normalize_usage(usage1))
103
+ chat1 = Chat.setup([
104
+ {role: :user, content: "hi"},
105
+ {role: :assistant, content: "hello"},
106
+ {role: :meta, content: meta_str1}
107
+ ])
108
+
109
+ totals = Chat.token_totals([chat1])
110
+ assert_equal 100, totals[:pt]
111
+ assert_equal 50, totals[:ct]
112
+ assert_equal 150, totals[:tt]
113
+ assert_equal 20, totals[:cct]
114
+ assert_equal 0, totals[:cwt]
115
+ assert_equal 30, totals[:rt]
116
+ end
117
+
118
+ def test_print_tokens_with_cache
119
+ tokens = {pt: 100, ct: 50, tt: 150, cct: 20, cwt: 0, rt: 30}
120
+ str = Chat.print_tokens(tokens)
121
+ assert str.include?("prompt=100")
122
+ assert str.include?("completion=50")
123
+ assert str.include?("total=150")
124
+ assert str.include?("cached=20")
125
+ assert !str.include?("cache_write")
126
+ assert str.include?("reasoning=30")
127
+ end
128
+
129
+ def test_token_keys_constant
130
+ assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
131
+ assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
132
+ end
133
+
134
+ def test_update_meta_with_cache_fields
135
+ %w(pt_s ct_s tt_s cct_s cwt_s rt_s).each { |name| Thread.current[name] = 0 }
136
+
137
+ response = {
138
+ 'usage' => {
139
+ "prompt_tokens" => 100,
140
+ "completion_tokens" => 50,
141
+ "total_tokens" => 150,
142
+ "prompt_tokens_details" => {"cached_tokens" => 20},
143
+ "completion_tokens_details" => {"reasoning_tokens" => 30}
144
+ }
145
+ }
146
+ meta = LLM::Responses.update_meta(response)
147
+ assert_equal 100, meta['pt']
148
+ assert_equal 50, meta['ct']
149
+ assert_equal 150, meta['tt']
150
+ assert_equal 20, meta['cct']
151
+ assert_equal 30, meta['rt']
152
+ assert_equal 20, meta['cct_s']
153
+ assert_equal 30, meta['rt_s']
154
+ assert_equal 20, meta['cct_c']
155
+ assert_equal 30, meta['rt_c']
156
+ end
157
+
158
+ def test_update_meta_accumulates_cache_fields_across_requests
159
+ %w(pt_s ct_s tt_s cct_s cwt_s rt_s).each { |name| Thread.current[name] = 0 }
160
+
161
+ resp1 = { 'usage' => {
162
+ "prompt_tokens" => 10, "completion_tokens" => 5, "total_tokens" => 15,
163
+ "prompt_tokens_details" => {"cached_tokens" => 4},
164
+ "completion_tokens_details" => {"reasoning_tokens" => 3}
165
+ }}
166
+ meta1 = LLM::Responses.update_meta(resp1)
167
+ assert_equal 4, meta1['cct_s']
168
+ assert_equal 3, meta1['rt_s']
169
+ assert_equal 4, meta1['cct_c']
170
+ assert_equal 3, meta1['rt_c']
171
+
172
+ resp2 = { 'usage' => {
173
+ "prompt_tokens" => 20, "completion_tokens" => 10, "total_tokens" => 30,
174
+ "prompt_tokens_details" => {"cached_tokens" => 8},
175
+ "completion_tokens_details" => {"reasoning_tokens" => 6}
176
+ }}
177
+ meta2 = LLM::Responses.update_meta(resp2, meta1)
178
+ assert_equal 12, meta2['cct_s'] # 4 + 8
179
+ assert_equal 9, meta2['rt_s'] # 3 + 6
180
+ assert_equal 12, meta2['cct_c'] # 4 + 8
181
+ assert_equal 9, meta2['rt_c'] # 3 + 6
182
+ end
183
+ end