scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -2,9 +2,11 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestRelay < Test::Unit::TestCase
5
- def test_ask
5
+ # Real scp/ssh relay version moved to
6
+ # test/integration/scout/llm/backends/test_relay.rb: LLM::Relay shells out
7
+ # to `scp` and has no client seam, so the unit copy stays disabled.
8
+ def _test_ask
6
9
  Scout::Config.set(:server, 'localhost', :relay)
7
10
  ppp LLM::Relay.ask 'Say hi', model: 'gemma2'
8
11
  end
9
12
  end
10
-
@@ -0,0 +1,131 @@
1
+ # Reusable offline fixtures for agent_meta provenance tests (plan fixtures C,
2
+ # D, E; also intended for the later token-collector rounds). Everything runs
3
+ # on TmpFile.with_dir directories plus persisted-style chat text; no providers.
4
+ #
5
+ # Helpers provided:
6
+ # write_chat(dir, name, text) -> chat file path
7
+ # meta_receipt(content) -> agent_meta entry Hash
8
+ # receipt_output(call_id, agent_meta, ...) -> JSON envelope String
9
+ # receipt_chat_text(receipts, extra: nil) -> persisted chat text
10
+ # plain_delegation_chat(job_path) -> persisted chat text
11
+ # make_job(dir, ref, dependencies:, logs:) -> job path
12
+ # visit_signature(visits) -> comparable Array
13
+ # fixture_c(dir) -> C/D layout paths
14
+ # truncated_receipt_chat(call_id, agent_meta) -> fixture H text
15
+ require 'fileutils'
16
+ require 'json'
17
+
18
+ module AgentMetaFixtures
19
+ # Write persisted-style chat text and return its absolute path.
20
+ def write_chat(dir, name, text)
21
+ path = File.expand_path(File.join(dir, name))
22
+ FileUtils.mkdir_p(File.dirname(path))
23
+ File.write(path, text)
24
+ path
25
+ end
26
+
27
+ # One agent_meta receipt entry with direct meta content, e.g.
28
+ # meta_receipt('pt=10 ct=4 tt=14 inference_id=w1') or
29
+ # meta_receipt('job=Worker/ask/Default_w').
30
+ def meta_receipt(content)
31
+ {role: 'meta', content: content}
32
+ end
33
+
34
+ # JSON payload of a function_call_output envelope carrying agent_meta.
35
+ def receipt_output(call_id, agent_meta, name: 'ask', content: 'child answer')
36
+ {name: name, content: content, id: call_id, agent_meta: agent_meta}.to_json
37
+ end
38
+
39
+ # Persisted chat text with one paired ask call per receipt entry. Hash keys
40
+ # are call ids, values are the agent_meta payloads (Arrays, Strings, ...).
41
+ # `extra` lines are appended after the receipts (e.g. local meta lines).
42
+ #
43
+ # Message indexes produced by Chat.parse (single user turn, no leading
44
+ # empty user message since 49c0d20):
45
+ # 0 user, then per receipt: function_call, function_call_output.
46
+ def receipt_chat_text(receipts, extra: nil)
47
+ lines = ['user: Run the worker']
48
+ receipts.each do |call_id, agent_meta|
49
+ lines << 'function_call: ' + %({"name":"ask","arguments":{},"id":"#{call_id}"})
50
+ lines << 'function_call_output: ' + receipt_output(call_id, agent_meta)
51
+ end
52
+ lines.concat(Array(extra)) if extra
53
+ lines << 'assistant: done'
54
+ lines * "\n" + "\n"
55
+ end
56
+
57
+ # Persisted chat text delegating to `job_path` through an ordinary local
58
+ # `meta: job=` message (no receipts at all).
59
+ def plain_delegation_chat(job_path)
60
+ "user: Run the worker\nmeta: job=#{job_path}\nassistant: done\n"
61
+ end
62
+
63
+ # Create a job layout under `dir` for the relative reference `ref`
64
+ # (e.g. 'Worker/ask/Default_w'): the result file, a .info sidecar with
65
+ # `dependencies` (always written, matching scout-gear Step), and log chats
66
+ # written under `<job>.files/log/<name>` (Hash name -> chat text). Returns
67
+ # the job path.
68
+ def make_job(dir, ref, result: 'answer', dependencies: [], logs: {})
69
+ path = File.expand_path(File.join(dir, ref))
70
+ FileUtils.mkdir_p(File.dirname(path))
71
+ File.write(path, result)
72
+ File.write(path + '.info', {dependencies: dependencies}.to_json)
73
+ logs.each do |name, text|
74
+ log_path = File.join(path + '.files', 'log', name)
75
+ FileUtils.mkdir_p(File.dirname(log_path))
76
+ File.write(log_path, text)
77
+ end
78
+ path
79
+ end
80
+
81
+ # Comparable signature of traverse_provenance visits:
82
+ # [kind, path, relation, first_visit].
83
+ def visit_signature(visits)
84
+ visits.collect do |kind, object, _parent_kind, _parent, relation, first|
85
+ [kind, object.respond_to?(:path) ? object.path.to_s : object.to_s, relation, first]
86
+ end
87
+ end
88
+
89
+ # Fixture C layout (plan fixtures C/D): a parent receipt points at a Worker
90
+ # job whose log chat has two direct metas (w1, w2), a nested receipt for a
91
+ # Critic job (with one direct meta c1), and a dependency job. The parent
92
+ # chat also carries an ordinary local `meta: job=` reference to the Critic so
93
+ # both discovery paths coexist, plus one receipt-only direct meta (shadow).
94
+ # Returns [parent_chat_path, worker_job_path, critic_job_path, dep_job_path].
95
+ def fixture_c(dir)
96
+ critic = make_job(dir, 'Critic/ask/Default_c')
97
+ dep = make_job(dir, 'Dep/load/Default_1')
98
+
99
+ worker_log = receipt_chat_text(
100
+ {'wc1' => [meta_receipt('pt=30 ct=10 tt=40 inference_id=c1'),
101
+ meta_receipt("job=#{critic}")]},
102
+ extra: ['meta: pt=100 ct=50 tt=150 inference_id=w1',
103
+ 'meta: pt=20 ct=10 tt=30 inference_id=w2']
104
+ )
105
+ worker = make_job(dir, 'Worker/ask/Default_w',
106
+ dependencies: [dep], logs: {'agent.chat' => worker_log})
107
+
108
+ parent = write_chat(dir, 'parent.chat',
109
+ receipt_chat_text(
110
+ {'p1' => [meta_receipt('pt=10 tt=12 inference_id=shadow'),
111
+ meta_receipt("job=#{worker}")]},
112
+ extra: ["meta: job=#{critic}"]
113
+ ))
114
+
115
+ [parent, worker, critic, dep]
116
+ end
117
+
118
+ # Fixture H: the output content is the standard truncation exception JSON
119
+ # (error: :truncated) exactly as LLM.process_calls serializes it, while the
120
+ # agent_meta receipts survive in the same envelope.
121
+ def truncated_receipt_chat(call_id, agent_meta, name: 'ask', characters: 90_000)
122
+ exception_msg = "Function #{name} #{call_id} was executed successfully, but it returned #{characters} characters, which is more than the maximum of 30000. To protect the model context window this result was not returned."
123
+ content = {exception: exception_msg, stack: ['a', 'b']}.to_json
124
+ payload = {name: name, content: content, id: call_id, error: :truncated,
125
+ agent_meta: agent_meta}.to_json
126
+ "user: Run\n" +
127
+ %({"name":"#{name}","arguments":{},"id":"#{call_id}"}).sub(/^/, 'function_call: ') + "\n" +
128
+ "function_call_output: #{payload}\n" +
129
+ "assistant: done\n"
130
+ end
131
+ end
@@ -0,0 +1,518 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
+
4
+ require 'scout/llm/chat'
5
+ require 'scout/llm/backends/responses'
6
+
7
+ class TestLLMUsageMeta < Test::Unit::TestCase
8
+ def setup
9
+ super
10
+ Chat::TOKEN_KEYS.each { |name| Thread.current["#{name}_s"] = 0 }
11
+ end
12
+
13
+ def response(prompt: nil, completion: nil, total: nil,
14
+ cached: nil, cache_write: nil, reasoning: nil)
15
+ usage = {}
16
+ usage['prompt_tokens'] = prompt unless prompt.nil?
17
+ usage['completion_tokens'] = completion unless completion.nil?
18
+ usage['total_tokens'] = total unless total.nil?
19
+ usage['prompt_tokens_details'] = {}
20
+ usage['prompt_tokens_details']['cached_tokens'] = cached unless cached.nil?
21
+ usage['input_tokens_details'] = {} if cache_write || cached
22
+ usage['input_tokens_details'] ||= {}
23
+ usage['input_tokens_details']['cache_write_tokens'] = cache_write unless cache_write.nil?
24
+ usage['completion_tokens_details'] = {}
25
+ usage['completion_tokens_details']['reasoning_tokens'] = reasoning unless reasoning.nil?
26
+ { 'usage' => usage }
27
+ end
28
+
29
+ def chat(text)
30
+ Chat.setup(LLM.messages(text))
31
+ end
32
+
33
+ def test_backend_records_direct_and_running_token_counts
34
+ first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
35
+ second = LLM::Responses.update_meta(response(prompt: 7, total: 7), first)
36
+
37
+ assert_equal 7, second['pt']
38
+ assert_nil second['ct']
39
+ assert_equal 9, second['pt_s']
40
+ assert_equal 12, second['tt_s']
41
+ assert_equal 9, second['pt_c']
42
+ assert_equal 3, second['ct_c']
43
+ assert_equal 12, second['tt_c']
44
+ assert_nil second['usage_id']
45
+ end
46
+
47
+ def test_jobs_returns_all_projecting_jobs
48
+ conversation = chat <<-EOF
49
+ user: First
50
+ meta: job=WF/ask/first.chat
51
+ assistant: First answer
52
+ user: Second
53
+ meta: job=WF/ask/second.chat
54
+ assistant: Second answer
55
+ EOF
56
+
57
+ assert_equal %w[WF/ask/first.chat WF/ask/second.chat], conversation.jobs
58
+ assert_equal 'WF/ask/second.chat', conversation.meta[:job]
59
+ end
60
+
61
+ def test_message_identity_includes_non_meta_history
62
+ first = chat <<-EOF
63
+ user: Question
64
+ meta: tt=5
65
+ assistant: Answer
66
+ EOF
67
+ same = chat <<-EOF
68
+ user: Question
69
+ assistant: Answer
70
+ EOF
71
+ different = chat <<-EOF
72
+ user: Different question
73
+ assistant: Answer
74
+ EOF
75
+
76
+ assert_equal first.message_index.last[:id], same.message_index.last[:id]
77
+ assert_not_equal first.message_index.last[:id], different.message_index.last[:id]
78
+ end
79
+
80
+ def test_consecutive_meta_leaves_the_first_segment_orphaned
81
+ conversation = chat <<-EOF
82
+ user: Work
83
+ meta: tt=2
84
+ meta: job=WF/ask/work.chat
85
+ assistant: Done
86
+ EOF
87
+
88
+ trace = Chat.trace_chats([conversation])
89
+ assert_equal 2, trace.length
90
+ assert trace.first[:orphan]
91
+ assert_equal 2, trace.first[:meta][:tt]
92
+ assert_equal 'WF/ask/work.chat', trace.last[:meta][:job]
93
+ assert_equal 1, trace.last[:messages].length
94
+ end
95
+
96
+ def test_final_meta_is_an_orphan_segment
97
+ conversation = chat <<-EOF
98
+ user: Work
99
+ meta: tt=2
100
+ assistant: Tool call removed
101
+ meta: tt=7
102
+ EOF
103
+
104
+ trace = Chat.trace_chats([conversation])
105
+ assert_equal 2, trace.length
106
+ assert_equal 7, trace.last[:meta][:tt]
107
+ assert trace.last[:orphan]
108
+ assert_empty trace.last[:messages]
109
+ end
110
+
111
+ def test_meta_covers_a_multi_tool_response_segment
112
+ conversation = chat <<-EOF
113
+ user: Write two files
114
+ meta: tt=1000
115
+ function_call: {"name":"write","id":"one"}
116
+ function_call_output: {"id":"one","content":"done one"}
117
+ function_call: {"name":"write","id":"two"}
118
+ function_call_output: {"id":"two","content":"done two"}
119
+ assistant: Done
120
+ user: Next request
121
+ EOF
122
+
123
+ trace = Chat.trace_chats([conversation])
124
+ assert_equal 1, trace.length
125
+ assert_equal 1000, trace.first[:meta][:tt]
126
+ assert_equal 5, trace.first[:messages].length
127
+ assert !trace.first[:orphan]
128
+ end
129
+
130
+ def test_project_keeps_inference_meta_inline_with_one_job_marker
131
+ response = [
132
+ { role: :meta, content: 'tt=2' },
133
+ { role: :function_call, content: '{"name":"write"}' },
134
+ { role: :function_call_output, content: '{"content":"done"}' },
135
+ { role: :meta, content: 'tt=7' },
136
+ { role: :assistant, content: 'Done' }
137
+ ]
138
+
139
+ projected = Chat.project('WF/ask/work.chat', response)
140
+ assert_equal %i[meta meta function_call function_call_output meta assistant], projected.collect { |m| m[:role] }
141
+
142
+ marker = Chat.parse_meta(projected.first[:content])
143
+ assert_equal 'WF/ask/work.chat', marker[:job]
144
+ assert Chat::TOKEN_KEYS.none? { |key| marker.include?(key) }
145
+
146
+ assert_equal 2, Chat.parse_meta(projected[1][:content])[:tt]
147
+ assert_equal :function_call, projected[2][:role], 'first inference meta stays adjacent to the call it produced'
148
+ assert_equal 7, Chat.parse_meta(projected[4][:content])[:tt]
149
+
150
+ trace = Chat.trace_chats([Chat.setup(projected)])
151
+ assert_equal 3, trace.length
152
+ assert trace.first[:orphan]
153
+ assert_equal [2, 7], trace[1..-1].collect { |entry| entry[:meta][:tt] }
154
+ assert_equal [2, 1], trace[1..-1].collect { |entry| entry[:messages].length }
155
+ assert trace[1..-1].none? { |entry| entry[:orphan] }
156
+
157
+ assert_equal 2, Chat.direct_entries([Chat.setup(projected)]).length
158
+ totals = Chat.token_totals([Chat.setup(projected)])
159
+ assert_equal 9, totals[:tt]
160
+ end
161
+
162
+ # Legacy chats carry no inference_id, so Chat.trace_indices falls back to the
163
+ # digest-based lineage id. The lineage id is computed from the preceding
164
+ # messages, and a projected copy sits behind a leading `job=` marker, so the
165
+ # projected and original copies of the same legacy inference get DIFFERENT
166
+ # lineage ids and are both counted. This is the documented legacy behaviour:
167
+ # precise deduplication requires inference_id, which every new inference has.
168
+ def test_project_legacy_meta_without_inference_id_is_not_deduplicated_across_chats
169
+ original = chat <<-EOF
170
+ user: Work
171
+ meta: pt=2 ct=1 tt=3
172
+ assistant: Done
173
+ EOF
174
+
175
+ projected = Chat.project('WF/ask/work.chat', [
176
+ { role: :meta, content: 'pt=2 ct=1 tt=3' },
177
+ { role: :assistant, content: 'Done' }
178
+ ])
179
+
180
+ trace = Chat.trace_chats([Chat.setup(projected), original])
181
+ assert_equal 3, trace.length, 'job marker + two non-merged legacy lineages'
182
+ assert trace.all? { |entry| entry[:deduplication] == :legacy_lineage }
183
+ assert_not_equal trace.first[:lineage_id], trace.last[:lineage_id]
184
+
185
+ projected_totals = Chat.token_totals([Chat.setup(projected)])
186
+ assert_equal 3, projected_totals[:tt]
187
+ assert_equal 3, Chat.token_totals([original])[:tt]
188
+ assert_equal 6, Chat.token_totals([Chat.setup(projected), original])[:tt]
189
+ end
190
+
191
+ def test_project_consumption_path_does_not_double_count_a_saved_projection
192
+ TmpFile.with_file(nil, false, :persistent => true) do |file|
193
+ original = chat <<-EOF
194
+ user: Work
195
+ meta: inference_id=request-one pt=10 ct=5 tt=15
196
+ assistant: Done
197
+ EOF
198
+
199
+ # chat_task: the job result chat is the projection; the log keeps the
200
+ # original metas.
201
+ projected = Chat.project('WF/ask/work.chat', [
202
+ { role: :meta, content: 'inference_id=request-one pt=10 ct=5 tt=15' },
203
+ { role: :assistant, content: 'Done' }
204
+ ])
205
+ Open.write(file, Chat.print(Chat.setup(projected)))
206
+
207
+ # LLM::Agent#ask consumption path: load the persisted job chat and
208
+ # re-project it before counting.
209
+ loaded = Chat.load(file)
210
+ reprojected = Chat.project('WF/ask/work.chat', loaded)
211
+
212
+ assert_equal Chat.token_totals([original]), Chat.token_totals([Chat.setup(reprojected), original])
213
+ end
214
+ end
215
+ def test_trace_keeps_distinct_segments_for_direct_and_projected_metadata
216
+ direct = chat <<-EOF
217
+ user: Work
218
+ meta: tt=7
219
+ assistant: Done
220
+ EOF
221
+ projected = chat <<-EOF
222
+ user: Work
223
+ meta: job=WF/ask/work.chat
224
+ assistant: Done
225
+ EOF
226
+
227
+ trace = Chat.trace_chats([projected, direct])
228
+ assert_equal 2, trace.length
229
+ assert_equal ['WF/ask/work.chat', nil], trace.collect { |entry| entry[:meta][:job] }
230
+ assert_equal [nil, 7], trace.collect { |entry| entry[:meta][:tt] }
231
+ end
232
+
233
+ def test_job_meta_does_not_reset_the_last_direct_chat_total
234
+ messages = LLM.messages <<-EOF
235
+ user: Plan
236
+ meta: pt=10 ct=2 tt=12 pt_c=10 ct_c=2 tt_c=12
237
+ assistant: Plan complete
238
+ meta: job=WF/ask/work.chat
239
+ assistant: Work complete
240
+ EOF
241
+
242
+ current = Chat.meta(messages)
243
+ assert_equal 'WF/ask/work.chat', current[:job]
244
+ assert_equal 10, current[:pt_c]
245
+ assert_equal 2, current[:ct_c]
246
+ assert_equal 12, current[:tt_c]
247
+ end
248
+
249
+ # === Cache token accounting tests ===
250
+
251
+ def test_openai_responses_api_cache_tokens
252
+ resp = { 'usage' => {
253
+ 'input_tokens' => 9,
254
+ 'input_tokens_details' => { 'cache_write_tokens' => 5, 'cached_tokens' => 3 },
255
+ 'output_tokens' => 174,
256
+ 'output_tokens_details' => { 'reasoning_tokens' => 128 },
257
+ 'total_tokens' => 183
258
+ } }
259
+ meta = LLM::Responses.update_meta(resp)
260
+
261
+ assert_equal 9, meta['pt']
262
+ assert_equal 174, meta['ct']
263
+ assert_equal 183, meta['tt']
264
+ assert_equal 3, meta['cct']
265
+ assert_equal 5, meta['cwt']
266
+ assert_equal 128, meta['rt']
267
+ # cumulative variants
268
+ assert_equal 9, meta['pt_c']
269
+ assert_equal 3, meta['cct_c']
270
+ assert_equal 5, meta['cwt_c']
271
+ assert_equal 128, meta['rt_c']
272
+ # session variants
273
+ assert_equal 9, meta['pt_s']
274
+ assert_equal 3, meta['cct_s']
275
+ assert_equal 5, meta['cwt_s']
276
+ assert_equal 128, meta['rt_s']
277
+ end
278
+
279
+ def test_glm_cache_tokens
280
+ resp = { 'usage' => {
281
+ 'completion_tokens' => 97,
282
+ 'completion_tokens_details' => { 'reasoning_tokens' => 92 },
283
+ 'prompt_tokens' => 8,
284
+ 'prompt_tokens_details' => { 'cached_tokens' => 4 },
285
+ 'total_tokens' => 105
286
+ } }
287
+ meta = LLM::Responses.update_meta(resp)
288
+
289
+ assert_equal 8, meta['pt']
290
+ assert_equal 97, meta['ct']
291
+ assert_equal 105, meta['tt']
292
+ assert_equal 4, meta['cct']
293
+ assert_nil meta['cwt']
294
+ assert_equal 92, meta['rt']
295
+ # cumulative variants
296
+ assert_equal 4, meta['cct_c']
297
+ assert_equal 92, meta['rt_c']
298
+ end
299
+
300
+ def test_anthropic_flat_cache_fields
301
+ resp = { 'usage' => {
302
+ 'prompt_tokens' => 100,
303
+ 'completion_tokens' => 50,
304
+ 'cache_read_input_tokens' => 80,
305
+ 'cache_creation_input_tokens' => 20
306
+ } }
307
+ meta = LLM::Responses.update_meta(resp)
308
+
309
+ assert_equal 100, meta['pt']
310
+ assert_equal 50, meta['ct']
311
+ assert_equal 150, meta['tt'] # computed
312
+ assert_equal 80, meta['cct']
313
+ assert_equal 20, meta['cwt']
314
+ assert_nil meta['rt']
315
+ end
316
+
317
+ def test_cumulative_cache_tokens_across_requests
318
+ first = LLM::Responses.update_meta(
319
+ response(prompt: 10, completion: 5, total: 15, cached: 3, reasoning: 2)
320
+ )
321
+ second = LLM::Responses.update_meta(
322
+ response(prompt: 8, completion: 4, total: 12, cached: 6, reasoning: 1),
323
+ first
324
+ )
325
+
326
+ assert_equal 3, first['cct_c']
327
+ assert_equal 2, first['rt_c']
328
+ assert_equal 9, second['cct_c'] # 3 + 6
329
+ assert_equal 3, second['rt_c'] # 2 + 1
330
+ assert_equal 18, second['pt_c'] # 10 + 8
331
+ end
332
+
333
+ def test_normalize_usage_constants
334
+ assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
335
+ assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
336
+ end
337
+
338
+ def test_normalize_usage_openai_chat_api
339
+ usage = { 'prompt_tokens' => 9, 'completion_tokens' => 174, 'total_tokens' => 183 }
340
+ result = Chat.normalize_usage(usage)
341
+ assert_equal 9, result['pt']
342
+ assert_equal 174, result['ct']
343
+ assert_equal 183, result['tt']
344
+ assert_nil result['cct']
345
+ assert_nil result['cwt']
346
+ assert_nil result['rt']
347
+ end
348
+
349
+ def test_normalize_usage_computes_total_when_missing
350
+ usage = { 'prompt_tokens' => 10, 'completion_tokens' => 20 }
351
+ result = Chat.normalize_usage(usage)
352
+ assert_equal 30, result['tt']
353
+ end
354
+
355
+ def test_normalize_usage_handles_nil
356
+ assert_equal({}, Chat.normalize_usage(nil))
357
+ assert_equal({}, Chat.normalize_usage({}))
358
+ end
359
+
360
+ def test_direct_entries_includes_cache_tokens
361
+ conversation = chat <<-EOF
362
+ user: Work
363
+ meta: pt=10 ct=5 tt=15 cct=3 rt=2
364
+ assistant: Done
365
+ EOF
366
+
367
+ entries = Chat.direct_entries([conversation])
368
+ assert_equal 1, entries.length
369
+ assert_equal 3, entries.first[:meta][:cct]
370
+ assert_equal 2, entries.first[:meta][:rt]
371
+ end
372
+
373
+ def test_token_totals_aggregates_cache_fields
374
+ c1 = chat <<-EOF
375
+ user: Work
376
+ meta: pt=10 ct=5 tt=15 cct=3 rt=2
377
+ assistant: Done
378
+ EOF
379
+ c2 = chat <<-EOF
380
+ user: More work
381
+ meta: pt=20 ct=10 tt=30 cct=7 rt=8
382
+ assistant: Done again
383
+ EOF
384
+
385
+ totals = Chat.token_totals([c1, c2])
386
+ assert_equal 30, totals[:pt]
387
+ assert_equal 15, totals[:ct]
388
+ assert_equal 45, totals[:tt]
389
+ assert_equal 10, totals[:cct] # 3 + 7
390
+ assert_equal 10, totals[:rt] # 2 + 8
391
+ end
392
+
393
+ def test_print_tokens_shows_cache_fields
394
+ totals = { pt: 100, ct: 50, tt: 150, cct: 30, cwt: 10, rt: 20 }
395
+ output = Chat.print_tokens(totals)
396
+ assert output.include?('cached=30')
397
+ assert output.include?('cache_write=10')
398
+ assert output.include?('reasoning=20')
399
+ end
400
+
401
+ # === Quoted-value serialization/parsing tests ===
402
+
403
+ def test_serialize_meta_quotes_value_containing_equals
404
+ serialized = Chat.serialize_meta('reas' => 'thinking about a=b')
405
+ assert_equal 'reas="thinking about a=b"', serialized
406
+ end
407
+
408
+ def test_serialize_meta_does_not_quote_simple_values
409
+ serialized = Chat.serialize_meta('pt' => 10, 'job' => 'WF/ask/test.chat')
410
+ assert_equal 'pt=10 job=WF/ask/test.chat', serialized
411
+ end
412
+
413
+ def test_parse_meta_handles_quoted_value_with_equals
414
+ parsed = Chat.parse_meta('pt=10 reas="thinking about a=b"')
415
+ assert_equal 10, parsed[:pt]
416
+ assert_equal 'thinking about a=b', parsed[:reas]
417
+ end
418
+
419
+ def test_roundtrip_value_with_equals
420
+ meta = { 'pt' => 10, 'reas' => 'thinking about a=b and c=d', 'job' => 'WF/ask/test.chat' }
421
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
422
+ assert_equal meta, parsed
423
+ end
424
+
425
+ def test_roundtrip_value_with_quotes_and_backslashes
426
+ meta = { 'reas' => 'He said "hello" and a=b\c', 'pt' => 5 }
427
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
428
+ assert_equal meta, parsed
429
+ end
430
+
431
+ def test_backward_compat_unquoted_value_with_spaces
432
+ parsed = Chat.parse_meta('pt=10 ct=5 tt=15 reas=some text here')
433
+ assert_equal 10, parsed[:pt]
434
+ assert_equal 5, parsed[:ct]
435
+ assert_equal 15, parsed[:tt]
436
+ assert_equal 'some text here', parsed[:reas]
437
+ end
438
+
439
+ def test_backward_compat_job_and_reas_without_equals
440
+ parsed = Chat.parse_meta('job=WF/ask/test.chat reas=thinking about stuff')
441
+ assert_equal 'WF/ask/test.chat', parsed[:job]
442
+ assert_equal 'thinking about stuff', parsed[:reas]
443
+ end
444
+
445
+ def test_multiple_quoted_values_roundtrip
446
+ meta = { 'reas' => 'a=b', 'note' => 'c=d', 'pt' => 5 }
447
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
448
+ assert_equal meta, parsed
449
+ end
450
+
451
+ def test_realistic_roundtrip
452
+ meta = { 'pt' => 100, 'ct' => 50, 'tt' => 150,
453
+ 'reas' => 'The user asked for x=y, so I computed z=2',
454
+ 'job' => 'WF/ask/test.chat' }
455
+ parsed = Chat.parse_meta(Chat.serialize_meta(meta))
456
+ assert_equal meta, parsed
457
+ end
458
+
459
+ # === Unescaped inner quotes in quoted values ===
460
+
461
+ def test_parse_meta_unescaped_quotes_inside_quoted_value
462
+ parsed = Chat.parse_meta(%q{pt=100 reas="Some text with ["/path/to/file"] inside."})
463
+ assert_equal 100, parsed[:pt]
464
+ assert_equal 'Some text with ["/path/to/file"] inside.', parsed[:reas]
465
+ end
466
+
467
+ def test_parse_meta_unescaped_quotes_and_keyval_patterns_inside_quoted
468
+ parsed = Chat.parse_meta(%q{pt=100 rt_c=10 reas="text total=48M prompt=47M end. accurate."})
469
+ assert_equal 100, parsed[:pt]
470
+ assert_equal 10, parsed[:rt_c]
471
+ assert_nil parsed[:total]
472
+ assert_nil parsed[:prompt]
473
+ assert_equal 'text total=48M prompt=47M end. accurate.', parsed[:reas]
474
+ end
475
+
476
+ def test_parse_meta_quoted_value_then_following_keys
477
+ parsed = Chat.parse_meta(%q{pt=100 reas="a=b" ct=200})
478
+ assert_equal 100, parsed[:pt]
479
+ assert_equal 'a=b', parsed[:reas]
480
+ assert_equal 200, parsed[:ct]
481
+ end
482
+ def test_backend_assigns_distinct_inference_ids
483
+ first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
484
+ second = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
485
+ assert_not_nil first['inference_id']
486
+ assert_not_equal first['inference_id'], second['inference_id']
487
+ end
488
+
489
+ def test_inference_id_controls_trace_deduplication
490
+ copied = chat <<-EOF
491
+ user: Work
492
+ meta: inference_id=request-one pt=2 ct=3 tt=5
493
+ assistant: Done
494
+ EOF
495
+ repeated = chat <<-EOF
496
+ user: Work
497
+ meta: inference_id=request-two pt=2 ct=3 tt=5
498
+ assistant: Done
499
+ EOF
500
+
501
+ assert_equal 1, Chat.trace_chats([copied, copied]).length
502
+ trace = Chat.trace_chats([copied, repeated])
503
+ assert_equal 2, trace.length
504
+ assert_equal [:inference_id, :inference_id], trace.collect { |entry| entry[:deduplication] }
505
+ end
506
+
507
+ def test_source_aware_trace_preserves_addresses
508
+ conversation = chat <<-EOF
509
+ user: Work
510
+ meta: inference_id=request-one tt=5
511
+ assistant: Done
512
+ EOF
513
+ entry = Chat.trace_chat_sources('/tmp/work.chat' => conversation).first
514
+ assert_equal ['/tmp/work.chat', 1], entry[:meta_address]
515
+ assert_equal [['/tmp/work.chat', 2]], entry[:message_addresses]
516
+ end
517
+
518
+ end