scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -10,22 +10,105 @@ user: say hi
10
10
  ppp LLM::Anthropic.ask prompt
11
11
  end
12
12
 
13
- def _test_ask
13
+ # ScoutCoder: Anthropic has no embeddings endpoint; the backend raises
14
+ # from embed_query, so the offline contract is the exception, not a vector.
15
+ def test_embeddings
16
+ assert_raise(RuntimeError) { LLM::Anthropic.embed 'Some text', log_errors: false, model: 'embedding-model' }
17
+ end
18
+
19
+ def test_ask
20
+ client = TestFixtures.anthropic_client('backends/anthropic')
21
+ res = LLM::Anthropic.ask 'user: write a script that sorts files in a directory',
22
+ client: client, model: 'claude-sonnet-4-5', persist: false
23
+
24
+ assert_equal 'Mock answer from Anthropic messages', res
25
+ assert_equal 1, client.calls.length
26
+ assert_equal 'claude-sonnet-4-5', client.calls.first[:model]
27
+ assert client.calls.first[:messages].any? { |m| m[:role].to_s == 'user' }
28
+ end
29
+
30
+ def test_tool_call_output_weather
31
+ Log.severity = 0
14
32
  prompt =<<-EOF
15
- user: write a script that sorts files in a directory
33
+ function_call:
34
+
35
+ {"name":"get_current_temperature", "arguments":{"location":"London","unit":"Celsius"},"id":"tNTnsQq2s6jGh0npOh43AwDD"}
36
+
37
+ function_call_output:
38
+
39
+ {"id":"tNTnsQq2s6jGh0npOh43AwDD", "content":"It's 15 degrees and raining."}
40
+
41
+ user:
42
+
43
+ should i take an umbrella?
16
44
  EOF
17
- sss 0
18
- ppp LLM::Anthropic.ask prompt
45
+ client = TestFixtures.anthropic_client('backends/anthropic')
46
+ res = LLM::Anthropic.ask prompt, client: client, persist: false
47
+
48
+ assert_equal 'Mock answer from Anthropic messages', res
49
+ # ScoutCoder: the Anthropic backend rewrites the function_call /
50
+ # function_call_output pair into content array items (tool_use /
51
+ # tool_result) instead of separate messages.
52
+ sent = client.calls.first[:messages]
53
+ assert sent.any? { |m| m[:role].to_s == 'user' }
54
+ assert sent.inspect.include?('tool_result')
19
55
  end
20
56
 
21
- def test_embeddings
22
- Log.severity = 0
23
- text =<<-EOF
24
- Some text
57
+ def test_tool
58
+ prompt =<<-EOF
59
+ user:
60
+ What is the weather in London. Should I take my umbrella?
25
61
  EOF
26
- emb = LLM::Anthropic.embed text, log_errors: true, model: 'embedding-model'
27
62
 
28
- assert(Float === emb.first)
63
+ tools = [
64
+ {
65
+ "type": "custom",
66
+ "name": "get_current_temperature",
67
+ "description": "Get the current temperature and raining conditions for a specific location",
68
+ "parameters": {
69
+ "type": "object",
70
+ "properties": {
71
+ "location": {
72
+ "type": "string",
73
+ "description": "The city and state, e.g., San Francisco, CA"
74
+ },
75
+ "unit": {
76
+ "type": "string",
77
+ "enum": ["Celsius", "Fahrenheit"],
78
+ "description": "The temperature unit to use. Infer this from the user's location."
79
+ }
80
+ },
81
+ "required": ["location", "unit"]
82
+ }
83
+ },
84
+ ]
85
+
86
+ client = TestFixtures.anthropic_client('backends/anthropic_tool_use', 'backends/anthropic')
87
+ respose = LLM::Anthropic.ask prompt, tools: tools,
88
+ client: client, log_errors: true, persist: false do |name,arguments|
89
+ "It's 15 degrees and raining."
90
+ end
91
+
92
+ assert_equal 'Mock answer from Anthropic messages', respose
93
+ assert_equal 2, client.calls.length
94
+
95
+ # ScoutCoder: LLM::Anthropic#format_tool_definitions renames
96
+ # `parameters` -> `input_schema` and forces type 'custom'.
97
+ sent_tools = client.calls.first[:tools]
98
+ assert sent_tools.any? { |t| (t[:name] || t['name']) == 'get_current_temperature' }
99
+ assert sent_tools.all? { |t| (t[:input_schema] || t['input_schema']) }
100
+ assert sent_tools.all? { |t| (t[:type] || t['type']).to_s == 'custom' }
101
+ end
102
+
103
+ def test_json_output
104
+ client = TestFixtures.anthropic_client('backends/anthropic')
105
+ res = LLM::Anthropic.ask 'user: What other movies have the protagonists of the original gost busters played on, just the top.',
106
+ format: :json, client: client, persist: false
107
+
108
+ assert_equal 'Mock answer from Anthropic messages', res
109
+ # ScoutCoder: Anthropic takes a response_format hash, not the string key
110
+ # the OpenAI backends use.
111
+ assert client.calls.first.inspect.include?('json_object')
29
112
  end
30
113
 
31
114
  def _test_tool_call_output_2
@@ -2,15 +2,132 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestLLMBedrock < Test::Unit::TestCase
5
+ # Offline Bedrock coverage: LLM::Bedrock.ask/embed accept options[:client]
6
+ # (an object with invoke_model(model_id:, content_type:, body:) returning
7
+ # something whose .body.string is JSON), so FakeBedrockClient exercises the
8
+ # full request construction + response parsing path without AWS credentials.
9
+ #
10
+ # The old real-service versions are kept disabled below (_test_*).
11
+
12
+ # ScoutCoder: LLM::Bedrock is not required by lib/scout-ai.rb; each test has
13
+ # to require 'scout/llm/backends/bedrock' itself (same as LLM.ask does when
14
+ # dispatching to the :bedrock backend).
15
+ def test_ask
16
+ require 'scout/llm/backends/bedrock'
17
+ client = TestFixtures.bedrock_client('backends/bedrock')
18
+
19
+ response = LLM::Bedrock.ask 'user: say hi to bedrock',
20
+ client: client,
21
+ model: 'anthropic.claude-3-sonnet-20240229-v1:0',
22
+ model_max_tokens: 100
23
+
24
+ assert_equal 'Mock answer from Bedrock', response
25
+
26
+ assert_equal 1, client.calls.length
27
+ call = client.calls.first
28
+ assert_equal 'anthropic.claude-3-sonnet-20240229-v1:0', call[:model_id]
29
+ assert_equal 'application/json', call[:content_type]
30
+ # ScoutCoder: IndiferentHash#pretty_print takes 0 args, so assert on
31
+ # individual keys instead of pp-ing the recorded parameters.
32
+ assert_equal 100, call[:body]['max_tokens']
33
+ assert call[:body]['messages'].any? { |m| m['role'] == 'user' }
34
+ assert call[:body]['messages'].any? { |m| m['content'].to_s.include?('say hi to bedrock') }
35
+ end
36
+
37
+ def test_ask_prompt_type
38
+ require 'scout/llm/backends/bedrock'
39
+ client = TestFixtures.bedrock_client('backends/bedrock')
40
+
41
+ response = LLM::Bedrock.ask 'user: say hi through the prompt endpoint',
42
+ client: client, type: :prompt,
43
+ model: 'meta.llama3-8b-instruct-v1:0',
44
+ model_max_tokens: 100
45
+
46
+ assert_equal 'Mock answer from Bedrock', response
47
+
48
+ body = client.calls.first[:body]
49
+ assert body.include?('prompt')
50
+ assert body['prompt'].to_s.include?('say hi through the prompt endpoint')
51
+ end
52
+
53
+ def test_embeddings
54
+ require 'scout/llm/backends/bedrock'
55
+ client = TestFixtures.bedrock_client('backends/bedrock_embedding')
56
+
57
+ emb = LLM::Bedrock.embed 'Some text', client: client,
58
+ model: 'amazon.titan-embed-text-v1'
59
+
60
+ assert(Float === emb.first)
61
+ assert_equal [0.1, 0.2, 0.3], emb
62
+ assert_equal 1, client.calls.length
63
+
64
+ call = client.calls.first
65
+ assert_equal 'amazon.titan-embed-text-v1', call[:model_id]
66
+ assert_equal 'Some text', call[:body]['inputText']
67
+ end
68
+
69
+ # Tool loop: the first payload carries a tool_call content entry, the block
70
+ # answers it, and the second payload is the final text answer.
71
+ def test_tool_loop
72
+ require 'scout/llm/backends/bedrock'
73
+ tools = [
74
+ {
75
+ "type": "function",
76
+ "function": {
77
+ "name": "get_current_temperature",
78
+ "description": "Get the current temperature for a specific location",
79
+ "parameters": {
80
+ "type": "object",
81
+ "properties": {
82
+ "location": { "type": "string",
83
+ "description": "The city and state, e.g., San Francisco, CA" },
84
+ "unit": { "type": "string", "enum": ["Celsius", "Fahrenheit"],
85
+ "description": "The temperature unit to use." }
86
+ },
87
+ "required": ["location", "unit"]
88
+ }
89
+ }
90
+ }
91
+ ]
92
+
93
+ client = TestFixtures.bedrock_client('backends/bedrock_tool_use', 'backends/bedrock')
94
+
95
+ calls_seen = []
96
+ response = LLM::Bedrock.ask 'user: What is the weather in London? Should I take an umbrella? Use the tool.',
97
+ tools: tools, client: client,
98
+ model: 'anthropic.claude-3-sonnet-20240229-v1:0',
99
+ model_max_tokens: 100 do |name, arguments|
100
+ calls_seen << [name, arguments]
101
+ "It's 15 degrees and raining."
102
+ end
103
+
104
+ assert_equal 'Mock answer from Bedrock', response
105
+
106
+ # the block was reached with the unwrapped tool call
107
+ assert_equal 1, calls_seen.length
108
+ assert_equal 'get_current_temperature', calls_seen.first.first
109
+ assert_equal 'London', calls_seen.first.last['location']
110
+
111
+ # two invoke_model rounds, the second one carrying the tool response
112
+ assert_equal 2, client.calls.length
113
+ sent = client.calls.last[:body]['messages']
114
+ assert sent.any? { |m| m['role'] == 'tool' }
115
+ tool_messages = sent.select { |m| m['role'] == 'tool' }
116
+ assert_equal 'It\'s 15 degrees and raining.', tool_messages.last['content']
117
+ assert_equal 'call_1', tool_messages.last['id']
118
+ end
119
+
5
120
  def _test_ask
121
+ # Real-service version: see the offline tests above. Requires AWS
122
+ # credentials and network.
6
123
  prompt =<<-EOF
7
124
  say hi
8
125
  EOF
9
126
  ppp LLM::Bedrock.ask prompt, model: "anthropic.claude-3-sonnet-20240229-v1:0", model_max_tokens: 100, model_anthropic_version: 'bedrock-2023-05-31'
10
127
  end
11
128
 
12
-
13
129
  def _test_embeddings
130
+ # Real-service version: see test_embeddings above.
14
131
  Log.severity = 0
15
132
  text =<<-EOF
16
133
  Some text
@@ -57,4 +174,3 @@ What is the weather in London. Should I take my umbrella? Use the provided tool
57
174
  ppp response
58
175
  end
59
176
  end
60
-
@@ -2,8 +2,24 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestLLMHF < Test::Unit::TestCase
5
+ class FakeHFClient
6
+ attr_reader :messages, :tools, :calls
5
7
 
6
- def test_ask
8
+ def initialize(*responses)
9
+ @responses = responses
10
+ @calls = 0
11
+ end
12
+
13
+ def chat(messages, tools, parameters = {})
14
+ @messages = messages
15
+ @tools = tools
16
+ response = @responses[@calls] || @responses.last
17
+ @calls += 1
18
+ response
19
+ end
20
+ end
21
+
22
+ def _test_ask
7
23
  Log.severity = 0
8
24
  prompt =<<-EOF
9
25
  system: you are a coding helper that only write code and inline comments. No extra explanations or comentary
@@ -13,61 +29,140 @@ user: write a script that sorts files in a directory
13
29
  ppp LLM::Huggingface.ask prompt, model: 'HuggingFaceTB/SmolLM2-135M-Instruct'
14
30
  end
15
31
 
16
- def _test_embeddings
17
- Log.severity = 0
18
- text =<<-EOF
19
- Some text
20
- EOF
21
- emb = LLM::Huggingface.embed text, model: 'distilbert-base-uncased-finetuned-sst-2-english'
22
- assert(Float === emb.first)
32
+ def test_format_tool_call
33
+ message = {
34
+ role: 'function_call',
35
+ content: {
36
+ name: 'get_current_temperature',
37
+ arguments: { location: 'London', unit: 'Celsius' },
38
+ id: 'call_123'
39
+ }.to_json
40
+ }
41
+
42
+ formatted = LLM::Huggingface.format_tool_call(message)
43
+
44
+ assert_equal 'assistant', formatted[:role]
45
+ assert_equal 'function', formatted.dig(:tool_calls, 0, :type)
46
+ assert_equal 'call_123', formatted.dig(:tool_calls, 0, :id)
47
+ assert_equal 'get_current_temperature', formatted.dig(:tool_calls, 0, :function, :name)
48
+ assert_equal 'London', formatted.dig(:tool_calls, 0, :function, :arguments, :location)
23
49
  end
24
50
 
25
- def _test_embedding_array
26
- Log.severity = 0
27
- text =<<-EOF
28
- Some text
29
- EOF
30
- emb = LLM::Huggingface.embed [text], model: 'distilbert-base-uncased-finetuned-sst-2-english'
31
- assert(Float === emb.first.first)
51
+ def test_format_tool_output
52
+ message = {
53
+ role: 'function_call_output',
54
+ content: {
55
+ id: 'call_123',
56
+ name: 'get_current_temperature',
57
+ content: "It's 15 degrees and raining."
58
+ }.to_json
59
+ }
60
+
61
+ formatted = LLM::Huggingface.format_tool_output(message)
62
+
63
+ assert_equal 'tool', formatted[:role]
64
+ assert_equal 'get_current_temperature', formatted[:name]
65
+ assert_equal 'call_123', formatted[:tool_call_id]
66
+ assert_equal "It's 15 degrees and raining.", formatted[:content]
32
67
  end
33
68
 
34
- def _test_tool
35
- prompt =<<-EOF
36
- What is the weather in London. Should I take an umbrella?
37
- EOF
69
+ def test_parse_tool_call
70
+ tool_call = {
71
+ id: 'call_123',
72
+ type: 'function',
73
+ function: {
74
+ name: 'get_current_temperature',
75
+ arguments: { location: 'London', unit: 'Celsius' }
76
+ }
77
+ }
78
+
79
+ parsed = LLM::Huggingface.parse_tool_call(tool_call)
80
+
81
+ assert_equal 'call_123', parsed[:id]
82
+ assert_equal 'get_current_temperature', parsed[:name]
83
+ assert_equal 'London', parsed.dig(:arguments, :location)
84
+ end
85
+
86
+ def test_ask_with_fake_client
87
+ client = FakeHFClient.new({ role: 'assistant', content: 'Hello from Huggingface' })
88
+
89
+ response = LLM::Huggingface.ask("user: say hi", client: client, log_response: false)
90
+
91
+ assert_equal 'Hello from Huggingface', response
92
+ assert_equal 1, client.calls
93
+ end
94
+
95
+ def test_ask_tool_loop_with_fake_client
96
+ client = FakeHFClient.new(
97
+ {
98
+ role: 'assistant',
99
+ content: '',
100
+ tool_calls: [
101
+ {
102
+ id: 'call_123',
103
+ type: 'function',
104
+ function: {
105
+ name: 'get_current_temperature',
106
+ arguments: { location: 'London', unit: 'Celsius' }
107
+ }
108
+ }
109
+ ]
110
+ },
111
+ {
112
+ role: 'assistant',
113
+ content: 'Take an umbrella.'
114
+ }
115
+ )
38
116
 
39
117
  tools = [
40
118
  {
41
- "type": "function",
42
- "function": {
43
- "name": "get_current_temperature",
44
- "description": "Get the current temperature for a specific location",
45
- "parameters": {
46
- "type": "object",
47
- "properties": {
48
- "location": {
49
- "type": "string",
50
- "description": "The city and state, e.g., San Francisco, CA"
51
- },
52
- "unit": {
53
- "type": "string",
54
- "enum": ["Celsius", "Fahrenheit"],
55
- "description": "The temperature unit to use. Infer this from the user's location."
56
- }
119
+ type: 'function',
120
+ function: {
121
+ name: 'get_current_temperature',
122
+ description: 'Get the current temperature',
123
+ parameters: {
124
+ type: 'object',
125
+ properties: {
126
+ location: { type: 'string' },
127
+ unit: { type: 'string' }
57
128
  },
58
- "required": ["location", "unit"]
129
+ required: %w(location unit)
59
130
  }
60
131
  }
61
- },
132
+ }
62
133
  ]
63
134
 
64
- sss 0
65
- respose = LLM::Huggingface.ask prompt, model: 'HuggingFaceTB/SmolLM2-135M-Instruct', tool_choice: 'required', tools: tools do |name,arguments|
66
- "It's raining cats and dogs"
135
+ response = LLM::Huggingface.ask("user: What is the weather in London?", client: client, tools: tools, log_response: false) do |_name, _arguments|
136
+ "It's 15 degrees and raining."
67
137
  end
68
138
 
69
- ppp respose
139
+ assert_equal 'Take an umbrella.', response
140
+ assert_equal 2, client.calls
70
141
  end
71
142
 
72
- end
143
+ def test_format_tool_definitions
144
+ tools = {
145
+ 'get_current_temperature' => [
146
+ nil,
147
+ {
148
+ name: 'get_current_temperature',
149
+ description: 'Get the current temperature',
150
+ parameters: {
151
+ type: 'object',
152
+ properties: {
153
+ location: { type: 'string' }
154
+ },
155
+ required: ['location'],
156
+ defaults: { unit: 'Celsius' }
157
+ }
158
+ }
159
+ ]
160
+ }
161
+
162
+ formatted = LLM::Huggingface.format_tool_definitions(tools)
73
163
 
164
+ assert_equal 'function', formatted.first[:type].to_s
165
+ assert_equal 'get_current_temperature', formatted.first.dig(:function, :name)
166
+ assert_nil formatted.first.dig(:function, :parameters, :defaults)
167
+ end
168
+ end
@@ -3,14 +3,72 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
3
3
 
4
4
  class TestLLMOllama < Test::Unit::TestCase
5
5
 
6
- def _test_ask
7
- Log.severity = 0
6
+ def test_ask
7
+ client = TestFixtures.ollama_client('backends/ollama')
8
+ res = LLM::OLlama.ask 'user: write a script that sorts files in a directory',
9
+ client: client, model: 'mistral', mode: 'chat', persist: false
10
+
11
+ assert_equal 'Mock answer from Ollama', res
12
+ assert_equal 1, client.calls.length
13
+ assert_equal 'mistral', client.calls.first[:model]
14
+ assert client.calls.first[:messages].any? { |m| m[:role].to_s == 'user' }
15
+ end
16
+
17
+ def test_tool
8
18
  prompt =<<-EOF
9
- system: you are a coding helper that only write code and inline comments. No extra explanations or comentary
10
- system: Avoid using backticks ``` to format code.
11
- user: write a script that sorts files in a directory
19
+ What is the weather in London. Should I take an umbrella?
12
20
  EOF
13
- ppp LLM::OLlama.ask prompt, model: 'mistral', mode: 'chat'
21
+
22
+ tools = [
23
+ {
24
+ "type": "function",
25
+ "function": {
26
+ "name": "get_current_temperature",
27
+ "description": "Get the current temperature for a specific location",
28
+ "parameters": {
29
+ "type": "object",
30
+ "properties": {
31
+ "location": {
32
+ "type": "string",
33
+ "description": "The city and state, e.g., San Francisco, CA"
34
+ },
35
+ "unit": {
36
+ "type": "string",
37
+ "enum": ["Celsius", "Fahrenheit"],
38
+ "description": "The temperature unit to use. Infer this from the user's location."
39
+ }
40
+ },
41
+ "required": ["location", "unit"]
42
+ }
43
+ }
44
+ },
45
+ ]
46
+
47
+ client = TestFixtures.ollama_client('backends/ollama_tool_call', 'backends/ollama')
48
+ respose = LLM::OLlama.ask prompt, model: 'gpt-oss',
49
+ client: client, tool_choice: 'required',
50
+ tools: tools, persist: false do |name,arguments|
51
+ "It's raining cats and dogs"
52
+ end
53
+
54
+ assert_equal 'Mock answer from Ollama', respose
55
+ assert_equal 2, client.calls.length
56
+
57
+ # ScoutCoder: LLM::OLlama#query calls client.chat(parameters) positionally
58
+ # (no `parameters:` kwarg) and the API returns an Array of chunk hashes.
59
+ sent_tools = client.calls.first[:tools]
60
+ assert sent_tools.any? { |t| (t.dig(:function, :name) || t.dig('function', 'name')) == 'get_current_temperature' }
61
+ end
62
+
63
+ def test_embeddings
64
+ payload = [{ 'embeddings' => [[0.1, 0.2, 0.3]] }]
65
+ client = FakeOllamaClient.new(payload)
66
+
67
+ emb = LLM::OLlama.embed 'Some text', client: client, model: 'mxbai-embed-large'
68
+
69
+ assert(Float === emb.first)
70
+ assert_equal [0.1, 0.2, 0.3], emb
71
+ assert_equal 'Some text', client.calls.first[:input]
14
72
  end
15
73
 
16
74
  def _test_tool_call_output
@@ -89,22 +147,14 @@ What is the weather in London. Should I take an umbrella?
89
147
  ppp respose
90
148
  end
91
149
 
92
- def _test_embeddings
93
- Log.severity = 0
94
- text =<<-EOF
95
- Some text
96
- EOF
97
- emb = LLM::OLlama.embed text, model: 'mxbai-embed-large', url: 'localhost:3331'
98
- assert(Float === emb.first)
99
- end
100
-
101
150
  def test_embedding_array
102
- Log.severity = 0
103
- text =<<-EOF
104
- Some text
105
- EOF
106
- emb = LLM::OLlama.embed [text], model: 'mxbai-embed-large', url: 'localhost:3331'
151
+ payload = [{ 'embeddings' => [[0.1, 0.2, 0.3], [0.4, 0.5, 0.6]] }]
152
+ client = FakeOllamaClient.new(payload)
153
+
154
+ emb = LLM::OLlama.embed ['Some text', 'More text'], client: client, model: 'mxbai-embed-large'
155
+
107
156
  assert(Float === emb.first.first)
157
+ assert_equal [[0.1, 0.2, 0.3], [0.4, 0.5, 0.6]], emb
108
158
  end
109
159
  end
110
160
 
@@ -2,13 +2,12 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestOpenWebUI < Test::Unit::TestCase
5
- def test_gepeto
5
+ # Real https://gepeto.bsc.es/api version moved to
6
+ # test/integration/scout/llm/backends/test_openwebui.rb (no client seam:
7
+ # LLM::OpenWebUIMethods#query posts through RestClient.post with a plain
8
+ # Hash "client").
9
+ def _test_gepeto
6
10
  Log.severity = 0
7
- prompt =<<-EOF
8
- system: you are a coding helper that only write code and comments without formatting so that it can work directly, avoid the initial and end commas ```.
9
- user: write a script that sorts files in a directory
10
- EOF
11
-
12
11
  prompt =<<-EOF
13
12
  user: write a script that sorts files in a directory
14
13
  EOF
@@ -16,42 +15,45 @@ user: write a script that sorts files in a directory
16
15
  ppp LLM::OpenWebUI.ask prompt, model: 'qwen3-vl:30b', url: "https://gepeto.bsc.es/api"
17
16
  end
18
17
 
19
- def _test_tool
20
- prompt =<<-EOF
21
- What is the weather in London. Should I take an umbrella?
22
- EOF
18
+ # Offline: stub RestClient.post so only request construction and response
19
+ # parsing are exercised. OpenWebUI is OpenAI-compatible, so the openai_chat
20
+ # fixture is replayed as the response body.
21
+ def test_ask_request_construction
22
+ fixture = TestFixtures.fixture('backends/openai_chat')
23
+ expected_answer = fixture.dig('choices', 0, 'message', 'content')
24
+
25
+ recorded = []
26
+ original = RestClient.method(:post)
23
27
 
24
- tools = [
25
- {
26
- "type": "function",
27
- "function": {
28
- "name": "get_current_temperature",
29
- "description": "Get the current temperature for a specific location",
30
- "parameters": {
31
- "type": "object",
32
- "properties": {
33
- "location": {
34
- "type": "string",
35
- "description": "The city and state, e.g., San Francisco, CA"
36
- },
37
- "unit": {
38
- "type": "string",
39
- "enum": ["Celsius", "Fahrenheit"],
40
- "description": "The temperature unit to use. Infer this from the user's location."
41
- }
42
- },
43
- "required": ["location", "unit"]
44
- }
45
- }
46
- },
47
- ]
48
-
49
- sss 0
50
- respose = LLM::OpenWebUI.ask prompt, model: 'gemma2:latest', tool_choice: 'required', tools: tools do |name,arguments|
51
- "It's raining cats and dogs"
28
+ RestClient.define_singleton_method(:post) do |url, payload, headers|
29
+ recorded << {url: url, payload: JSON.parse(payload), headers: headers}
30
+ body = Struct.new(:body).new(fixture.to_json)
31
+ body
52
32
  end
53
33
 
54
- ppp respose
34
+ begin
35
+ answer = LLM::OpenWebUI.ask 'user: write a script that sorts files in a directory offline',
36
+ model: 'qwen3-vl:30b', url: 'https://openwebui.example/api',
37
+ key: 'test-key', persist: false
38
+ ensure
39
+ RestClient.singleton_class.send(:define_method, :post, original)
40
+ end
41
+
42
+ assert_equal expected_answer, answer
43
+
44
+ assert_equal 1, recorded.length
45
+ call = recorded.first
46
+
47
+ assert call[:url].end_with?('chat/completions')
48
+ assert call[:url].start_with?('https://openwebui.example/api')
49
+
50
+ payload = call[:payload]
51
+ assert_equal 'qwen3-vl:30b', payload['model']
52
+ assert payload['messages'].any? { |m| m['role'] == 'user' }
53
+
54
+ headers = call[:headers]
55
+ assert headers['Authorization'] || headers[:Authorization]
56
+ assert_equal 'Bearer test-key', (headers['Authorization'] || headers[:Authorization]).to_s
57
+ assert((headers['Content-Type'] || headers[:Content_Type]).to_s.include?('application/json'))
55
58
  end
56
59
  end
57
-