scout-ai 1.2.3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.vimproject +138 -50
- data/README.md +171 -290
- data/Rakefile +17 -1
- data/VERSION +1 -1
- data/doc/Improvements.md +325 -0
- data/doc/StartHere.md +110 -0
- data/doc/developer/Architecture.md +126 -0
- data/doc/developer/Backends.md +199 -0
- data/doc/developer/ChatLifecycle.md +183 -0
- data/doc/developer/DelegationInternals.md +295 -0
- data/doc/developer/DesignPrinciples.md +245 -0
- data/doc/developer/PromptProcessing.md +292 -0
- data/doc/developer/Provenance.md +317 -0
- data/doc/user/BuildingAgents.md +345 -0
- data/doc/user/Cookbook.md +333 -0
- data/doc/user/CoreConcepts.md +181 -0
- data/doc/user/Delegation.md +191 -0
- data/doc/user/GettingStarted.md +159 -0
- data/doc/user/ManagingContext.md +163 -0
- data/doc/user/MultiAgentWorkflows.md +256 -0
- data/doc/user/Python.md +159 -0
- data/doc/user/RunningInference.md +200 -0
- data/doc/user/ToolCalling.md +193 -0
- data/doc/user/WritingChats.md +197 -0
- data/lib/scout/llm/agent/chat.rb +61 -11
- data/lib/scout/llm/agent/delegate.rb +274 -65
- data/lib/scout/llm/agent/iterate.rb +2 -2
- data/lib/scout/llm/agent/save.rb +273 -0
- data/lib/scout/llm/agent/workflow.rb +164 -0
- data/lib/scout/llm/agent.rb +86 -61
- data/lib/scout/llm/ask.rb +62 -17
- data/lib/scout/llm/backends/anthropic.rb +9 -2
- data/lib/scout/llm/backends/bedrock.rb +15 -3
- data/lib/scout/llm/backends/default.rb +183 -99
- data/lib/scout/llm/backends/glm.rb +58 -0
- data/lib/scout/llm/backends/huggingface.rb +196 -26
- data/lib/scout/llm/backends/ollama.rb +13 -1
- data/lib/scout/llm/backends/openai.rb +0 -2
- data/lib/scout/llm/backends/openwebui.rb +20 -13
- data/lib/scout/llm/backends/relay.rb +22 -22
- data/lib/scout/llm/backends/responses.rb +1 -1
- data/lib/scout/llm/chat/agent_meta.rb +264 -0
- data/lib/scout/llm/chat/annotation.rb +39 -10
- data/lib/scout/llm/chat/parse.rb +28 -6
- data/lib/scout/llm/chat/persist.rb +25 -0
- data/lib/scout/llm/chat/process/clear.rb +41 -6
- data/lib/scout/llm/chat/process/files.rb +21 -6
- data/lib/scout/llm/chat/process/meta.rb +421 -34
- data/lib/scout/llm/chat/process/options.rb +21 -1
- data/lib/scout/llm/chat/process/tools.rb +56 -15
- data/lib/scout/llm/chat/process.rb +4 -0
- data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
- data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
- data/lib/scout/llm/chat/prompt.rb +48 -0
- data/lib/scout/llm/chat/provenance.rb +775 -0
- data/lib/scout/llm/chat/tool_calls.rb +76 -0
- data/lib/scout/llm/chat.rb +18 -2
- data/lib/scout/llm/embed.rb +11 -3
- data/lib/scout/llm/image.rb +86 -0
- data/lib/scout/llm/mcp.rb +10 -2
- data/lib/scout/llm/rag.rb +3 -3
- data/lib/scout/llm/tools/call.rb +160 -11
- data/lib/scout/llm/tools/knowledge_base.rb +1 -1
- data/lib/scout/llm/tools/workflow.rb +32 -16
- data/lib/scout/model/python/huggingface/causal.rb +23 -5
- data/lib/scout/model/python/huggingface.rb +2 -1
- data/lib/scout-ai.rb +1 -0
- data/python/README.md +197 -14
- data/python/scout_ai/huggingface/eval.py +245 -34
- data/python/tests/test_huggingface_eval.py +58 -0
- data/research/ChatAnalyst-required-changes.md +167 -0
- data/research/agent-delegation-analysis.md +810 -0
- data/research/agent-meta-provenance-integration-plan.md +622 -0
- data/research/agent-workflow-analysis.md +1120 -0
- data/research/backends-analysis.md +836 -0
- data/research/chat-core-analysis.md +946 -0
- data/research/chatanalyst-provenance/00-baseline.md +30 -0
- data/research/chatanalyst-provenance/01-repo-map.md +60 -0
- data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
- data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
- data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
- data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
- data/research/chatanalyst-provenance/07-critic-review.md +25 -0
- data/research/chatanalyst-provenance/final-report.md +45 -0
- data/research/chatanalyst-provenance/resumption.md +37 -0
- data/research/coding-philosophy-analysis.md +928 -0
- data/research/commands-analysis.md +947 -0
- data/research/multi-agent-patterns-analysis.md +853 -0
- data/research/prompt-strategies-analysis.md +630 -0
- data/research/prov-verbosity-fix-notes.md +77 -0
- data/research/provenance-analysis.md +469 -0
- data/research/provenance-navigation-design.md +640 -0
- data/research/synthesis-report.md +487 -0
- data/research/tools-system-analysis.md +779 -0
- data/scout-ai.gemspec +100 -11
- data/scout_commands/agent/ask +13 -3
- data/scout_commands/agent/kb +2 -0
- data/scout_commands/llm/ask +11 -4
- data/scout_commands/llm/md +76 -0
- data/scout_commands/llm/process_queries +48 -0
- data/scout_commands/llm/prov +602 -0
- data/scout_commands/llm/word +71 -0
- data/scout_commands/workflow/mcp +43 -0
- data/share/word/reference.docx +0 -0
- data/test/etc/AI/mock.yaml +11 -0
- data/test/fixtures/backends/anthropic.json +19 -0
- data/test/fixtures/backends/anthropic_tool_use.json +24 -0
- data/test/fixtures/backends/bedrock.json +8 -0
- data/test/fixtures/backends/bedrock_embedding.json +3 -0
- data/test/fixtures/backends/bedrock_tool_use.json +17 -0
- data/test/fixtures/backends/ollama.json +16 -0
- data/test/fixtures/backends/ollama_tool_call.json +27 -0
- data/test/fixtures/backends/openai_chat.json +21 -0
- data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
- data/test/fixtures/backends/responses.json +33 -0
- data/test/fixtures/backends/responses_tool_call.json +28 -0
- data/test/integration/README.md +32 -0
- data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
- data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
- data/test/integration/scout/llm/backends/test_relay.rb +52 -0
- data/test/integration/scout/llm/test_infrastructure.rb +74 -0
- data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
- data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
- data/test/integration/scout/model/test_base.rb +91 -0
- data/test/scout/llm/agent/test_chat.rb +8 -2
- data/test/scout/llm/agent/test_save.rb +413 -0
- data/test/scout/llm/agent/test_workflow.rb +110 -0
- data/test/scout/llm/backends/test_anthropic.rb +93 -10
- data/test/scout/llm/backends/test_bedrock.rb +118 -2
- data/test/scout/llm/backends/test_huggingface.rb +137 -42
- data/test/scout/llm/backends/test_ollama.rb +70 -20
- data/test/scout/llm/backends/test_openwebui.rb +42 -40
- data/test/scout/llm/backends/test_relay.rb +4 -2
- data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
- data/test/scout/llm/chat/process/test_meta.rb +518 -0
- data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
- data/test/scout/llm/chat/test_agent_meta.rb +357 -0
- data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
- data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
- data/test/scout/llm/chat/test_parse.rb +70 -15
- data/test/scout/llm/chat/test_prov_cli.rb +274 -0
- data/test/scout/llm/chat/test_provenance.rb +240 -0
- data/test/scout/llm/chat/test_tool_calls.rb +38 -0
- data/test/scout/llm/test_agent.rb +13 -36
- data/test/scout/llm/test_ask.rb +75 -52
- data/test/scout/llm/test_chat.rb +107 -13
- data/test/scout/llm/test_embed.rb +48 -0
- data/test/scout/llm/test_rag.rb +23 -16
- data/test/scout/llm/test_tools.rb +12 -1
- data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
- data/test/scout/llm/tools/test_mcp.rb +5 -3
- data/test/scout/llm/tools/test_workflow.rb +23 -2
- data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
- data/test/scout/model/python/huggingface/test_causal.rb +9 -3
- data/test/scout/model/python/huggingface/test_classification.rb +11 -2
- data/test/scout/model/python/test_torch.rb +2 -0
- data/test/scout/model/python/torch/test_helpers.rb +4 -0
- data/test/scout/model/test_base.rb +4 -2
- data/test/support/availability.rb +231 -0
- data/test/support/fake_clients.rb +138 -0
- data/test/support/fixtures.rb +21 -0
- data/test/support/infrastructure_probes.rb +136 -0
- data/test/support/mock_backend.rb +215 -0
- data/test/test_helper.rb +32 -2
- metadata +99 -10
- data/doc/Agent.md +0 -327
- data/doc/Chat.md +0 -458
- data/doc/LLM.md +0 -340
- data/doc/RAG.md +0 -129
- data/scout_commands/documenter +0 -148
- data/test/scout/llm/backends/test_openai.rb +0 -192
- data/test/scout/llm/backends/test_responses.rb +0 -238
- data/test/scout/llm/test_parse.rb +0 -98
|
@@ -2,9 +2,11 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
4
|
class TestRelay < Test::Unit::TestCase
|
|
5
|
-
|
|
5
|
+
# Real scp/ssh relay version moved to
|
|
6
|
+
# test/integration/scout/llm/backends/test_relay.rb: LLM::Relay shells out
|
|
7
|
+
# to `scp` and has no client seam, so the unit copy stays disabled.
|
|
8
|
+
def _test_ask
|
|
6
9
|
Scout::Config.set(:server, 'localhost', :relay)
|
|
7
10
|
ppp LLM::Relay.ask 'Say hi', model: 'gemma2'
|
|
8
11
|
end
|
|
9
12
|
end
|
|
10
|
-
|
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
# Reusable offline fixtures for agent_meta provenance tests (plan fixtures C,
|
|
2
|
+
# D, E; also intended for the later token-collector rounds). Everything runs
|
|
3
|
+
# on TmpFile.with_dir directories plus persisted-style chat text; no providers.
|
|
4
|
+
#
|
|
5
|
+
# Helpers provided:
|
|
6
|
+
# write_chat(dir, name, text) -> chat file path
|
|
7
|
+
# meta_receipt(content) -> agent_meta entry Hash
|
|
8
|
+
# receipt_output(call_id, agent_meta, ...) -> JSON envelope String
|
|
9
|
+
# receipt_chat_text(receipts, extra: nil) -> persisted chat text
|
|
10
|
+
# plain_delegation_chat(job_path) -> persisted chat text
|
|
11
|
+
# make_job(dir, ref, dependencies:, logs:) -> job path
|
|
12
|
+
# visit_signature(visits) -> comparable Array
|
|
13
|
+
# fixture_c(dir) -> C/D layout paths
|
|
14
|
+
# truncated_receipt_chat(call_id, agent_meta) -> fixture H text
|
|
15
|
+
require 'fileutils'
|
|
16
|
+
require 'json'
|
|
17
|
+
|
|
18
|
+
module AgentMetaFixtures
|
|
19
|
+
# Write persisted-style chat text and return its absolute path.
|
|
20
|
+
def write_chat(dir, name, text)
|
|
21
|
+
path = File.expand_path(File.join(dir, name))
|
|
22
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
23
|
+
File.write(path, text)
|
|
24
|
+
path
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
# One agent_meta receipt entry with direct meta content, e.g.
|
|
28
|
+
# meta_receipt('pt=10 ct=4 tt=14 inference_id=w1') or
|
|
29
|
+
# meta_receipt('job=Worker/ask/Default_w').
|
|
30
|
+
def meta_receipt(content)
|
|
31
|
+
{role: 'meta', content: content}
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
# JSON payload of a function_call_output envelope carrying agent_meta.
|
|
35
|
+
def receipt_output(call_id, agent_meta, name: 'ask', content: 'child answer')
|
|
36
|
+
{name: name, content: content, id: call_id, agent_meta: agent_meta}.to_json
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# Persisted chat text with one paired ask call per receipt entry. Hash keys
|
|
40
|
+
# are call ids, values are the agent_meta payloads (Arrays, Strings, ...).
|
|
41
|
+
# `extra` lines are appended after the receipts (e.g. local meta lines).
|
|
42
|
+
#
|
|
43
|
+
# Message indexes produced by Chat.parse (single user turn, no leading
|
|
44
|
+
# empty user message since 49c0d20):
|
|
45
|
+
# 0 user, then per receipt: function_call, function_call_output.
|
|
46
|
+
def receipt_chat_text(receipts, extra: nil)
|
|
47
|
+
lines = ['user: Run the worker']
|
|
48
|
+
receipts.each do |call_id, agent_meta|
|
|
49
|
+
lines << 'function_call: ' + %({"name":"ask","arguments":{},"id":"#{call_id}"})
|
|
50
|
+
lines << 'function_call_output: ' + receipt_output(call_id, agent_meta)
|
|
51
|
+
end
|
|
52
|
+
lines.concat(Array(extra)) if extra
|
|
53
|
+
lines << 'assistant: done'
|
|
54
|
+
lines * "\n" + "\n"
|
|
55
|
+
end
|
|
56
|
+
|
|
57
|
+
# Persisted chat text delegating to `job_path` through an ordinary local
|
|
58
|
+
# `meta: job=` message (no receipts at all).
|
|
59
|
+
def plain_delegation_chat(job_path)
|
|
60
|
+
"user: Run the worker\nmeta: job=#{job_path}\nassistant: done\n"
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
# Create a job layout under `dir` for the relative reference `ref`
|
|
64
|
+
# (e.g. 'Worker/ask/Default_w'): the result file, a .info sidecar with
|
|
65
|
+
# `dependencies` (always written, matching scout-gear Step), and log chats
|
|
66
|
+
# written under `<job>.files/log/<name>` (Hash name -> chat text). Returns
|
|
67
|
+
# the job path.
|
|
68
|
+
def make_job(dir, ref, result: 'answer', dependencies: [], logs: {})
|
|
69
|
+
path = File.expand_path(File.join(dir, ref))
|
|
70
|
+
FileUtils.mkdir_p(File.dirname(path))
|
|
71
|
+
File.write(path, result)
|
|
72
|
+
File.write(path + '.info', {dependencies: dependencies}.to_json)
|
|
73
|
+
logs.each do |name, text|
|
|
74
|
+
log_path = File.join(path + '.files', 'log', name)
|
|
75
|
+
FileUtils.mkdir_p(File.dirname(log_path))
|
|
76
|
+
File.write(log_path, text)
|
|
77
|
+
end
|
|
78
|
+
path
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# Comparable signature of traverse_provenance visits:
|
|
82
|
+
# [kind, path, relation, first_visit].
|
|
83
|
+
def visit_signature(visits)
|
|
84
|
+
visits.collect do |kind, object, _parent_kind, _parent, relation, first|
|
|
85
|
+
[kind, object.respond_to?(:path) ? object.path.to_s : object.to_s, relation, first]
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# Fixture C layout (plan fixtures C/D): a parent receipt points at a Worker
|
|
90
|
+
# job whose log chat has two direct metas (w1, w2), a nested receipt for a
|
|
91
|
+
# Critic job (with one direct meta c1), and a dependency job. The parent
|
|
92
|
+
# chat also carries an ordinary local `meta: job=` reference to the Critic so
|
|
93
|
+
# both discovery paths coexist, plus one receipt-only direct meta (shadow).
|
|
94
|
+
# Returns [parent_chat_path, worker_job_path, critic_job_path, dep_job_path].
|
|
95
|
+
def fixture_c(dir)
|
|
96
|
+
critic = make_job(dir, 'Critic/ask/Default_c')
|
|
97
|
+
dep = make_job(dir, 'Dep/load/Default_1')
|
|
98
|
+
|
|
99
|
+
worker_log = receipt_chat_text(
|
|
100
|
+
{'wc1' => [meta_receipt('pt=30 ct=10 tt=40 inference_id=c1'),
|
|
101
|
+
meta_receipt("job=#{critic}")]},
|
|
102
|
+
extra: ['meta: pt=100 ct=50 tt=150 inference_id=w1',
|
|
103
|
+
'meta: pt=20 ct=10 tt=30 inference_id=w2']
|
|
104
|
+
)
|
|
105
|
+
worker = make_job(dir, 'Worker/ask/Default_w',
|
|
106
|
+
dependencies: [dep], logs: {'agent.chat' => worker_log})
|
|
107
|
+
|
|
108
|
+
parent = write_chat(dir, 'parent.chat',
|
|
109
|
+
receipt_chat_text(
|
|
110
|
+
{'p1' => [meta_receipt('pt=10 tt=12 inference_id=shadow'),
|
|
111
|
+
meta_receipt("job=#{worker}")]},
|
|
112
|
+
extra: ["meta: job=#{critic}"]
|
|
113
|
+
))
|
|
114
|
+
|
|
115
|
+
[parent, worker, critic, dep]
|
|
116
|
+
end
|
|
117
|
+
|
|
118
|
+
# Fixture H: the output content is the standard truncation exception JSON
|
|
119
|
+
# (error: :truncated) exactly as LLM.process_calls serializes it, while the
|
|
120
|
+
# agent_meta receipts survive in the same envelope.
|
|
121
|
+
def truncated_receipt_chat(call_id, agent_meta, name: 'ask', characters: 90_000)
|
|
122
|
+
exception_msg = "Function #{name} #{call_id} was executed successfully, but it returned #{characters} characters, which is more than the maximum of 30000. To protect the model context window this result was not returned."
|
|
123
|
+
content = {exception: exception_msg, stack: ['a', 'b']}.to_json
|
|
124
|
+
payload = {name: name, content: content, id: call_id, error: :truncated,
|
|
125
|
+
agent_meta: agent_meta}.to_json
|
|
126
|
+
"user: Run\n" +
|
|
127
|
+
%({"name":"#{name}","arguments":{},"id":"#{call_id}"}).sub(/^/, 'function_call: ') + "\n" +
|
|
128
|
+
"function_call_output: #{payload}\n" +
|
|
129
|
+
"assistant: done\n"
|
|
130
|
+
end
|
|
131
|
+
end
|
|
@@ -0,0 +1,518 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
|
+
|
|
4
|
+
require 'scout/llm/chat'
|
|
5
|
+
require 'scout/llm/backends/responses'
|
|
6
|
+
|
|
7
|
+
class TestLLMUsageMeta < Test::Unit::TestCase
|
|
8
|
+
def setup
|
|
9
|
+
super
|
|
10
|
+
Chat::TOKEN_KEYS.each { |name| Thread.current["#{name}_s"] = 0 }
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def response(prompt: nil, completion: nil, total: nil,
|
|
14
|
+
cached: nil, cache_write: nil, reasoning: nil)
|
|
15
|
+
usage = {}
|
|
16
|
+
usage['prompt_tokens'] = prompt unless prompt.nil?
|
|
17
|
+
usage['completion_tokens'] = completion unless completion.nil?
|
|
18
|
+
usage['total_tokens'] = total unless total.nil?
|
|
19
|
+
usage['prompt_tokens_details'] = {}
|
|
20
|
+
usage['prompt_tokens_details']['cached_tokens'] = cached unless cached.nil?
|
|
21
|
+
usage['input_tokens_details'] = {} if cache_write || cached
|
|
22
|
+
usage['input_tokens_details'] ||= {}
|
|
23
|
+
usage['input_tokens_details']['cache_write_tokens'] = cache_write unless cache_write.nil?
|
|
24
|
+
usage['completion_tokens_details'] = {}
|
|
25
|
+
usage['completion_tokens_details']['reasoning_tokens'] = reasoning unless reasoning.nil?
|
|
26
|
+
{ 'usage' => usage }
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
def chat(text)
|
|
30
|
+
Chat.setup(LLM.messages(text))
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def test_backend_records_direct_and_running_token_counts
|
|
34
|
+
first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
35
|
+
second = LLM::Responses.update_meta(response(prompt: 7, total: 7), first)
|
|
36
|
+
|
|
37
|
+
assert_equal 7, second['pt']
|
|
38
|
+
assert_nil second['ct']
|
|
39
|
+
assert_equal 9, second['pt_s']
|
|
40
|
+
assert_equal 12, second['tt_s']
|
|
41
|
+
assert_equal 9, second['pt_c']
|
|
42
|
+
assert_equal 3, second['ct_c']
|
|
43
|
+
assert_equal 12, second['tt_c']
|
|
44
|
+
assert_nil second['usage_id']
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def test_jobs_returns_all_projecting_jobs
|
|
48
|
+
conversation = chat <<-EOF
|
|
49
|
+
user: First
|
|
50
|
+
meta: job=WF/ask/first.chat
|
|
51
|
+
assistant: First answer
|
|
52
|
+
user: Second
|
|
53
|
+
meta: job=WF/ask/second.chat
|
|
54
|
+
assistant: Second answer
|
|
55
|
+
EOF
|
|
56
|
+
|
|
57
|
+
assert_equal %w[WF/ask/first.chat WF/ask/second.chat], conversation.jobs
|
|
58
|
+
assert_equal 'WF/ask/second.chat', conversation.meta[:job]
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def test_message_identity_includes_non_meta_history
|
|
62
|
+
first = chat <<-EOF
|
|
63
|
+
user: Question
|
|
64
|
+
meta: tt=5
|
|
65
|
+
assistant: Answer
|
|
66
|
+
EOF
|
|
67
|
+
same = chat <<-EOF
|
|
68
|
+
user: Question
|
|
69
|
+
assistant: Answer
|
|
70
|
+
EOF
|
|
71
|
+
different = chat <<-EOF
|
|
72
|
+
user: Different question
|
|
73
|
+
assistant: Answer
|
|
74
|
+
EOF
|
|
75
|
+
|
|
76
|
+
assert_equal first.message_index.last[:id], same.message_index.last[:id]
|
|
77
|
+
assert_not_equal first.message_index.last[:id], different.message_index.last[:id]
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
def test_consecutive_meta_leaves_the_first_segment_orphaned
|
|
81
|
+
conversation = chat <<-EOF
|
|
82
|
+
user: Work
|
|
83
|
+
meta: tt=2
|
|
84
|
+
meta: job=WF/ask/work.chat
|
|
85
|
+
assistant: Done
|
|
86
|
+
EOF
|
|
87
|
+
|
|
88
|
+
trace = Chat.trace_chats([conversation])
|
|
89
|
+
assert_equal 2, trace.length
|
|
90
|
+
assert trace.first[:orphan]
|
|
91
|
+
assert_equal 2, trace.first[:meta][:tt]
|
|
92
|
+
assert_equal 'WF/ask/work.chat', trace.last[:meta][:job]
|
|
93
|
+
assert_equal 1, trace.last[:messages].length
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
def test_final_meta_is_an_orphan_segment
|
|
97
|
+
conversation = chat <<-EOF
|
|
98
|
+
user: Work
|
|
99
|
+
meta: tt=2
|
|
100
|
+
assistant: Tool call removed
|
|
101
|
+
meta: tt=7
|
|
102
|
+
EOF
|
|
103
|
+
|
|
104
|
+
trace = Chat.trace_chats([conversation])
|
|
105
|
+
assert_equal 2, trace.length
|
|
106
|
+
assert_equal 7, trace.last[:meta][:tt]
|
|
107
|
+
assert trace.last[:orphan]
|
|
108
|
+
assert_empty trace.last[:messages]
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def test_meta_covers_a_multi_tool_response_segment
|
|
112
|
+
conversation = chat <<-EOF
|
|
113
|
+
user: Write two files
|
|
114
|
+
meta: tt=1000
|
|
115
|
+
function_call: {"name":"write","id":"one"}
|
|
116
|
+
function_call_output: {"id":"one","content":"done one"}
|
|
117
|
+
function_call: {"name":"write","id":"two"}
|
|
118
|
+
function_call_output: {"id":"two","content":"done two"}
|
|
119
|
+
assistant: Done
|
|
120
|
+
user: Next request
|
|
121
|
+
EOF
|
|
122
|
+
|
|
123
|
+
trace = Chat.trace_chats([conversation])
|
|
124
|
+
assert_equal 1, trace.length
|
|
125
|
+
assert_equal 1000, trace.first[:meta][:tt]
|
|
126
|
+
assert_equal 5, trace.first[:messages].length
|
|
127
|
+
assert !trace.first[:orphan]
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def test_project_keeps_inference_meta_inline_with_one_job_marker
|
|
131
|
+
response = [
|
|
132
|
+
{ role: :meta, content: 'tt=2' },
|
|
133
|
+
{ role: :function_call, content: '{"name":"write"}' },
|
|
134
|
+
{ role: :function_call_output, content: '{"content":"done"}' },
|
|
135
|
+
{ role: :meta, content: 'tt=7' },
|
|
136
|
+
{ role: :assistant, content: 'Done' }
|
|
137
|
+
]
|
|
138
|
+
|
|
139
|
+
projected = Chat.project('WF/ask/work.chat', response)
|
|
140
|
+
assert_equal %i[meta meta function_call function_call_output meta assistant], projected.collect { |m| m[:role] }
|
|
141
|
+
|
|
142
|
+
marker = Chat.parse_meta(projected.first[:content])
|
|
143
|
+
assert_equal 'WF/ask/work.chat', marker[:job]
|
|
144
|
+
assert Chat::TOKEN_KEYS.none? { |key| marker.include?(key) }
|
|
145
|
+
|
|
146
|
+
assert_equal 2, Chat.parse_meta(projected[1][:content])[:tt]
|
|
147
|
+
assert_equal :function_call, projected[2][:role], 'first inference meta stays adjacent to the call it produced'
|
|
148
|
+
assert_equal 7, Chat.parse_meta(projected[4][:content])[:tt]
|
|
149
|
+
|
|
150
|
+
trace = Chat.trace_chats([Chat.setup(projected)])
|
|
151
|
+
assert_equal 3, trace.length
|
|
152
|
+
assert trace.first[:orphan]
|
|
153
|
+
assert_equal [2, 7], trace[1..-1].collect { |entry| entry[:meta][:tt] }
|
|
154
|
+
assert_equal [2, 1], trace[1..-1].collect { |entry| entry[:messages].length }
|
|
155
|
+
assert trace[1..-1].none? { |entry| entry[:orphan] }
|
|
156
|
+
|
|
157
|
+
assert_equal 2, Chat.direct_entries([Chat.setup(projected)]).length
|
|
158
|
+
totals = Chat.token_totals([Chat.setup(projected)])
|
|
159
|
+
assert_equal 9, totals[:tt]
|
|
160
|
+
end
|
|
161
|
+
|
|
162
|
+
# Legacy chats carry no inference_id, so Chat.trace_indices falls back to the
|
|
163
|
+
# digest-based lineage id. The lineage id is computed from the preceding
|
|
164
|
+
# messages, and a projected copy sits behind a leading `job=` marker, so the
|
|
165
|
+
# projected and original copies of the same legacy inference get DIFFERENT
|
|
166
|
+
# lineage ids and are both counted. This is the documented legacy behaviour:
|
|
167
|
+
# precise deduplication requires inference_id, which every new inference has.
|
|
168
|
+
def test_project_legacy_meta_without_inference_id_is_not_deduplicated_across_chats
|
|
169
|
+
original = chat <<-EOF
|
|
170
|
+
user: Work
|
|
171
|
+
meta: pt=2 ct=1 tt=3
|
|
172
|
+
assistant: Done
|
|
173
|
+
EOF
|
|
174
|
+
|
|
175
|
+
projected = Chat.project('WF/ask/work.chat', [
|
|
176
|
+
{ role: :meta, content: 'pt=2 ct=1 tt=3' },
|
|
177
|
+
{ role: :assistant, content: 'Done' }
|
|
178
|
+
])
|
|
179
|
+
|
|
180
|
+
trace = Chat.trace_chats([Chat.setup(projected), original])
|
|
181
|
+
assert_equal 3, trace.length, 'job marker + two non-merged legacy lineages'
|
|
182
|
+
assert trace.all? { |entry| entry[:deduplication] == :legacy_lineage }
|
|
183
|
+
assert_not_equal trace.first[:lineage_id], trace.last[:lineage_id]
|
|
184
|
+
|
|
185
|
+
projected_totals = Chat.token_totals([Chat.setup(projected)])
|
|
186
|
+
assert_equal 3, projected_totals[:tt]
|
|
187
|
+
assert_equal 3, Chat.token_totals([original])[:tt]
|
|
188
|
+
assert_equal 6, Chat.token_totals([Chat.setup(projected), original])[:tt]
|
|
189
|
+
end
|
|
190
|
+
|
|
191
|
+
def test_project_consumption_path_does_not_double_count_a_saved_projection
|
|
192
|
+
TmpFile.with_file(nil, false, :persistent => true) do |file|
|
|
193
|
+
original = chat <<-EOF
|
|
194
|
+
user: Work
|
|
195
|
+
meta: inference_id=request-one pt=10 ct=5 tt=15
|
|
196
|
+
assistant: Done
|
|
197
|
+
EOF
|
|
198
|
+
|
|
199
|
+
# chat_task: the job result chat is the projection; the log keeps the
|
|
200
|
+
# original metas.
|
|
201
|
+
projected = Chat.project('WF/ask/work.chat', [
|
|
202
|
+
{ role: :meta, content: 'inference_id=request-one pt=10 ct=5 tt=15' },
|
|
203
|
+
{ role: :assistant, content: 'Done' }
|
|
204
|
+
])
|
|
205
|
+
Open.write(file, Chat.print(Chat.setup(projected)))
|
|
206
|
+
|
|
207
|
+
# LLM::Agent#ask consumption path: load the persisted job chat and
|
|
208
|
+
# re-project it before counting.
|
|
209
|
+
loaded = Chat.load(file)
|
|
210
|
+
reprojected = Chat.project('WF/ask/work.chat', loaded)
|
|
211
|
+
|
|
212
|
+
assert_equal Chat.token_totals([original]), Chat.token_totals([Chat.setup(reprojected), original])
|
|
213
|
+
end
|
|
214
|
+
end
|
|
215
|
+
def test_trace_keeps_distinct_segments_for_direct_and_projected_metadata
|
|
216
|
+
direct = chat <<-EOF
|
|
217
|
+
user: Work
|
|
218
|
+
meta: tt=7
|
|
219
|
+
assistant: Done
|
|
220
|
+
EOF
|
|
221
|
+
projected = chat <<-EOF
|
|
222
|
+
user: Work
|
|
223
|
+
meta: job=WF/ask/work.chat
|
|
224
|
+
assistant: Done
|
|
225
|
+
EOF
|
|
226
|
+
|
|
227
|
+
trace = Chat.trace_chats([projected, direct])
|
|
228
|
+
assert_equal 2, trace.length
|
|
229
|
+
assert_equal ['WF/ask/work.chat', nil], trace.collect { |entry| entry[:meta][:job] }
|
|
230
|
+
assert_equal [nil, 7], trace.collect { |entry| entry[:meta][:tt] }
|
|
231
|
+
end
|
|
232
|
+
|
|
233
|
+
def test_job_meta_does_not_reset_the_last_direct_chat_total
|
|
234
|
+
messages = LLM.messages <<-EOF
|
|
235
|
+
user: Plan
|
|
236
|
+
meta: pt=10 ct=2 tt=12 pt_c=10 ct_c=2 tt_c=12
|
|
237
|
+
assistant: Plan complete
|
|
238
|
+
meta: job=WF/ask/work.chat
|
|
239
|
+
assistant: Work complete
|
|
240
|
+
EOF
|
|
241
|
+
|
|
242
|
+
current = Chat.meta(messages)
|
|
243
|
+
assert_equal 'WF/ask/work.chat', current[:job]
|
|
244
|
+
assert_equal 10, current[:pt_c]
|
|
245
|
+
assert_equal 2, current[:ct_c]
|
|
246
|
+
assert_equal 12, current[:tt_c]
|
|
247
|
+
end
|
|
248
|
+
|
|
249
|
+
# === Cache token accounting tests ===
|
|
250
|
+
|
|
251
|
+
def test_openai_responses_api_cache_tokens
|
|
252
|
+
resp = { 'usage' => {
|
|
253
|
+
'input_tokens' => 9,
|
|
254
|
+
'input_tokens_details' => { 'cache_write_tokens' => 5, 'cached_tokens' => 3 },
|
|
255
|
+
'output_tokens' => 174,
|
|
256
|
+
'output_tokens_details' => { 'reasoning_tokens' => 128 },
|
|
257
|
+
'total_tokens' => 183
|
|
258
|
+
} }
|
|
259
|
+
meta = LLM::Responses.update_meta(resp)
|
|
260
|
+
|
|
261
|
+
assert_equal 9, meta['pt']
|
|
262
|
+
assert_equal 174, meta['ct']
|
|
263
|
+
assert_equal 183, meta['tt']
|
|
264
|
+
assert_equal 3, meta['cct']
|
|
265
|
+
assert_equal 5, meta['cwt']
|
|
266
|
+
assert_equal 128, meta['rt']
|
|
267
|
+
# cumulative variants
|
|
268
|
+
assert_equal 9, meta['pt_c']
|
|
269
|
+
assert_equal 3, meta['cct_c']
|
|
270
|
+
assert_equal 5, meta['cwt_c']
|
|
271
|
+
assert_equal 128, meta['rt_c']
|
|
272
|
+
# session variants
|
|
273
|
+
assert_equal 9, meta['pt_s']
|
|
274
|
+
assert_equal 3, meta['cct_s']
|
|
275
|
+
assert_equal 5, meta['cwt_s']
|
|
276
|
+
assert_equal 128, meta['rt_s']
|
|
277
|
+
end
|
|
278
|
+
|
|
279
|
+
def test_glm_cache_tokens
|
|
280
|
+
resp = { 'usage' => {
|
|
281
|
+
'completion_tokens' => 97,
|
|
282
|
+
'completion_tokens_details' => { 'reasoning_tokens' => 92 },
|
|
283
|
+
'prompt_tokens' => 8,
|
|
284
|
+
'prompt_tokens_details' => { 'cached_tokens' => 4 },
|
|
285
|
+
'total_tokens' => 105
|
|
286
|
+
} }
|
|
287
|
+
meta = LLM::Responses.update_meta(resp)
|
|
288
|
+
|
|
289
|
+
assert_equal 8, meta['pt']
|
|
290
|
+
assert_equal 97, meta['ct']
|
|
291
|
+
assert_equal 105, meta['tt']
|
|
292
|
+
assert_equal 4, meta['cct']
|
|
293
|
+
assert_nil meta['cwt']
|
|
294
|
+
assert_equal 92, meta['rt']
|
|
295
|
+
# cumulative variants
|
|
296
|
+
assert_equal 4, meta['cct_c']
|
|
297
|
+
assert_equal 92, meta['rt_c']
|
|
298
|
+
end
|
|
299
|
+
|
|
300
|
+
def test_anthropic_flat_cache_fields
|
|
301
|
+
resp = { 'usage' => {
|
|
302
|
+
'prompt_tokens' => 100,
|
|
303
|
+
'completion_tokens' => 50,
|
|
304
|
+
'cache_read_input_tokens' => 80,
|
|
305
|
+
'cache_creation_input_tokens' => 20
|
|
306
|
+
} }
|
|
307
|
+
meta = LLM::Responses.update_meta(resp)
|
|
308
|
+
|
|
309
|
+
assert_equal 100, meta['pt']
|
|
310
|
+
assert_equal 50, meta['ct']
|
|
311
|
+
assert_equal 150, meta['tt'] # computed
|
|
312
|
+
assert_equal 80, meta['cct']
|
|
313
|
+
assert_equal 20, meta['cwt']
|
|
314
|
+
assert_nil meta['rt']
|
|
315
|
+
end
|
|
316
|
+
|
|
317
|
+
def test_cumulative_cache_tokens_across_requests
|
|
318
|
+
first = LLM::Responses.update_meta(
|
|
319
|
+
response(prompt: 10, completion: 5, total: 15, cached: 3, reasoning: 2)
|
|
320
|
+
)
|
|
321
|
+
second = LLM::Responses.update_meta(
|
|
322
|
+
response(prompt: 8, completion: 4, total: 12, cached: 6, reasoning: 1),
|
|
323
|
+
first
|
|
324
|
+
)
|
|
325
|
+
|
|
326
|
+
assert_equal 3, first['cct_c']
|
|
327
|
+
assert_equal 2, first['rt_c']
|
|
328
|
+
assert_equal 9, second['cct_c'] # 3 + 6
|
|
329
|
+
assert_equal 3, second['rt_c'] # 2 + 1
|
|
330
|
+
assert_equal 18, second['pt_c'] # 10 + 8
|
|
331
|
+
end
|
|
332
|
+
|
|
333
|
+
def test_normalize_usage_constants
|
|
334
|
+
assert_equal %w[pt ct tt cct cwt rt], Chat::TOKEN_KEYS
|
|
335
|
+
assert_equal %w[pt_c ct_c tt_c cct_c cwt_c rt_c], Chat::CUMULATIVE_KEYS
|
|
336
|
+
end
|
|
337
|
+
|
|
338
|
+
def test_normalize_usage_openai_chat_api
|
|
339
|
+
usage = { 'prompt_tokens' => 9, 'completion_tokens' => 174, 'total_tokens' => 183 }
|
|
340
|
+
result = Chat.normalize_usage(usage)
|
|
341
|
+
assert_equal 9, result['pt']
|
|
342
|
+
assert_equal 174, result['ct']
|
|
343
|
+
assert_equal 183, result['tt']
|
|
344
|
+
assert_nil result['cct']
|
|
345
|
+
assert_nil result['cwt']
|
|
346
|
+
assert_nil result['rt']
|
|
347
|
+
end
|
|
348
|
+
|
|
349
|
+
def test_normalize_usage_computes_total_when_missing
|
|
350
|
+
usage = { 'prompt_tokens' => 10, 'completion_tokens' => 20 }
|
|
351
|
+
result = Chat.normalize_usage(usage)
|
|
352
|
+
assert_equal 30, result['tt']
|
|
353
|
+
end
|
|
354
|
+
|
|
355
|
+
def test_normalize_usage_handles_nil
|
|
356
|
+
assert_equal({}, Chat.normalize_usage(nil))
|
|
357
|
+
assert_equal({}, Chat.normalize_usage({}))
|
|
358
|
+
end
|
|
359
|
+
|
|
360
|
+
def test_direct_entries_includes_cache_tokens
|
|
361
|
+
conversation = chat <<-EOF
|
|
362
|
+
user: Work
|
|
363
|
+
meta: pt=10 ct=5 tt=15 cct=3 rt=2
|
|
364
|
+
assistant: Done
|
|
365
|
+
EOF
|
|
366
|
+
|
|
367
|
+
entries = Chat.direct_entries([conversation])
|
|
368
|
+
assert_equal 1, entries.length
|
|
369
|
+
assert_equal 3, entries.first[:meta][:cct]
|
|
370
|
+
assert_equal 2, entries.first[:meta][:rt]
|
|
371
|
+
end
|
|
372
|
+
|
|
373
|
+
def test_token_totals_aggregates_cache_fields
|
|
374
|
+
c1 = chat <<-EOF
|
|
375
|
+
user: Work
|
|
376
|
+
meta: pt=10 ct=5 tt=15 cct=3 rt=2
|
|
377
|
+
assistant: Done
|
|
378
|
+
EOF
|
|
379
|
+
c2 = chat <<-EOF
|
|
380
|
+
user: More work
|
|
381
|
+
meta: pt=20 ct=10 tt=30 cct=7 rt=8
|
|
382
|
+
assistant: Done again
|
|
383
|
+
EOF
|
|
384
|
+
|
|
385
|
+
totals = Chat.token_totals([c1, c2])
|
|
386
|
+
assert_equal 30, totals[:pt]
|
|
387
|
+
assert_equal 15, totals[:ct]
|
|
388
|
+
assert_equal 45, totals[:tt]
|
|
389
|
+
assert_equal 10, totals[:cct] # 3 + 7
|
|
390
|
+
assert_equal 10, totals[:rt] # 2 + 8
|
|
391
|
+
end
|
|
392
|
+
|
|
393
|
+
def test_print_tokens_shows_cache_fields
|
|
394
|
+
totals = { pt: 100, ct: 50, tt: 150, cct: 30, cwt: 10, rt: 20 }
|
|
395
|
+
output = Chat.print_tokens(totals)
|
|
396
|
+
assert output.include?('cached=30')
|
|
397
|
+
assert output.include?('cache_write=10')
|
|
398
|
+
assert output.include?('reasoning=20')
|
|
399
|
+
end
|
|
400
|
+
|
|
401
|
+
# === Quoted-value serialization/parsing tests ===
|
|
402
|
+
|
|
403
|
+
def test_serialize_meta_quotes_value_containing_equals
|
|
404
|
+
serialized = Chat.serialize_meta('reas' => 'thinking about a=b')
|
|
405
|
+
assert_equal 'reas="thinking about a=b"', serialized
|
|
406
|
+
end
|
|
407
|
+
|
|
408
|
+
def test_serialize_meta_does_not_quote_simple_values
|
|
409
|
+
serialized = Chat.serialize_meta('pt' => 10, 'job' => 'WF/ask/test.chat')
|
|
410
|
+
assert_equal 'pt=10 job=WF/ask/test.chat', serialized
|
|
411
|
+
end
|
|
412
|
+
|
|
413
|
+
def test_parse_meta_handles_quoted_value_with_equals
|
|
414
|
+
parsed = Chat.parse_meta('pt=10 reas="thinking about a=b"')
|
|
415
|
+
assert_equal 10, parsed[:pt]
|
|
416
|
+
assert_equal 'thinking about a=b', parsed[:reas]
|
|
417
|
+
end
|
|
418
|
+
|
|
419
|
+
def test_roundtrip_value_with_equals
|
|
420
|
+
meta = { 'pt' => 10, 'reas' => 'thinking about a=b and c=d', 'job' => 'WF/ask/test.chat' }
|
|
421
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
422
|
+
assert_equal meta, parsed
|
|
423
|
+
end
|
|
424
|
+
|
|
425
|
+
def test_roundtrip_value_with_quotes_and_backslashes
|
|
426
|
+
meta = { 'reas' => 'He said "hello" and a=b\c', 'pt' => 5 }
|
|
427
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
428
|
+
assert_equal meta, parsed
|
|
429
|
+
end
|
|
430
|
+
|
|
431
|
+
def test_backward_compat_unquoted_value_with_spaces
|
|
432
|
+
parsed = Chat.parse_meta('pt=10 ct=5 tt=15 reas=some text here')
|
|
433
|
+
assert_equal 10, parsed[:pt]
|
|
434
|
+
assert_equal 5, parsed[:ct]
|
|
435
|
+
assert_equal 15, parsed[:tt]
|
|
436
|
+
assert_equal 'some text here', parsed[:reas]
|
|
437
|
+
end
|
|
438
|
+
|
|
439
|
+
def test_backward_compat_job_and_reas_without_equals
|
|
440
|
+
parsed = Chat.parse_meta('job=WF/ask/test.chat reas=thinking about stuff')
|
|
441
|
+
assert_equal 'WF/ask/test.chat', parsed[:job]
|
|
442
|
+
assert_equal 'thinking about stuff', parsed[:reas]
|
|
443
|
+
end
|
|
444
|
+
|
|
445
|
+
def test_multiple_quoted_values_roundtrip
|
|
446
|
+
meta = { 'reas' => 'a=b', 'note' => 'c=d', 'pt' => 5 }
|
|
447
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
448
|
+
assert_equal meta, parsed
|
|
449
|
+
end
|
|
450
|
+
|
|
451
|
+
def test_realistic_roundtrip
|
|
452
|
+
meta = { 'pt' => 100, 'ct' => 50, 'tt' => 150,
|
|
453
|
+
'reas' => 'The user asked for x=y, so I computed z=2',
|
|
454
|
+
'job' => 'WF/ask/test.chat' }
|
|
455
|
+
parsed = Chat.parse_meta(Chat.serialize_meta(meta))
|
|
456
|
+
assert_equal meta, parsed
|
|
457
|
+
end
|
|
458
|
+
|
|
459
|
+
# === Unescaped inner quotes in quoted values ===
|
|
460
|
+
|
|
461
|
+
def test_parse_meta_unescaped_quotes_inside_quoted_value
|
|
462
|
+
parsed = Chat.parse_meta(%q{pt=100 reas="Some text with ["/path/to/file"] inside."})
|
|
463
|
+
assert_equal 100, parsed[:pt]
|
|
464
|
+
assert_equal 'Some text with ["/path/to/file"] inside.', parsed[:reas]
|
|
465
|
+
end
|
|
466
|
+
|
|
467
|
+
def test_parse_meta_unescaped_quotes_and_keyval_patterns_inside_quoted
|
|
468
|
+
parsed = Chat.parse_meta(%q{pt=100 rt_c=10 reas="text total=48M prompt=47M end. accurate."})
|
|
469
|
+
assert_equal 100, parsed[:pt]
|
|
470
|
+
assert_equal 10, parsed[:rt_c]
|
|
471
|
+
assert_nil parsed[:total]
|
|
472
|
+
assert_nil parsed[:prompt]
|
|
473
|
+
assert_equal 'text total=48M prompt=47M end. accurate.', parsed[:reas]
|
|
474
|
+
end
|
|
475
|
+
|
|
476
|
+
def test_parse_meta_quoted_value_then_following_keys
|
|
477
|
+
parsed = Chat.parse_meta(%q{pt=100 reas="a=b" ct=200})
|
|
478
|
+
assert_equal 100, parsed[:pt]
|
|
479
|
+
assert_equal 'a=b', parsed[:reas]
|
|
480
|
+
assert_equal 200, parsed[:ct]
|
|
481
|
+
end
|
|
482
|
+
def test_backend_assigns_distinct_inference_ids
|
|
483
|
+
first = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
484
|
+
second = LLM::Responses.update_meta(response(prompt: 2, completion: 3, total: 5))
|
|
485
|
+
assert_not_nil first['inference_id']
|
|
486
|
+
assert_not_equal first['inference_id'], second['inference_id']
|
|
487
|
+
end
|
|
488
|
+
|
|
489
|
+
def test_inference_id_controls_trace_deduplication
|
|
490
|
+
copied = chat <<-EOF
|
|
491
|
+
user: Work
|
|
492
|
+
meta: inference_id=request-one pt=2 ct=3 tt=5
|
|
493
|
+
assistant: Done
|
|
494
|
+
EOF
|
|
495
|
+
repeated = chat <<-EOF
|
|
496
|
+
user: Work
|
|
497
|
+
meta: inference_id=request-two pt=2 ct=3 tt=5
|
|
498
|
+
assistant: Done
|
|
499
|
+
EOF
|
|
500
|
+
|
|
501
|
+
assert_equal 1, Chat.trace_chats([copied, copied]).length
|
|
502
|
+
trace = Chat.trace_chats([copied, repeated])
|
|
503
|
+
assert_equal 2, trace.length
|
|
504
|
+
assert_equal [:inference_id, :inference_id], trace.collect { |entry| entry[:deduplication] }
|
|
505
|
+
end
|
|
506
|
+
|
|
507
|
+
def test_source_aware_trace_preserves_addresses
|
|
508
|
+
conversation = chat <<-EOF
|
|
509
|
+
user: Work
|
|
510
|
+
meta: inference_id=request-one tt=5
|
|
511
|
+
assistant: Done
|
|
512
|
+
EOF
|
|
513
|
+
entry = Chat.trace_chat_sources('/tmp/work.chat' => conversation).first
|
|
514
|
+
assert_equal ['/tmp/work.chat', 1], entry[:meta_address]
|
|
515
|
+
assert_equal [['/tmp/work.chat', 2]], entry[:message_addresses]
|
|
516
|
+
end
|
|
517
|
+
|
|
518
|
+
end
|