scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -0,0 +1,38 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+ require 'scout/llm/chat'
3
+
4
+ class TestChatToolCalls < Test::Unit::TestCase
5
+ def chat(text)
6
+ Chat.setup(LLM.messages(text))
7
+ end
8
+
9
+ def test_pairs_calls_and_outputs_and_preserves_addresses
10
+ conversation = chat <<-EOF
11
+ function_call: {"name":"bash","arguments":{"cmd":"false"},"id":"call-1"}
12
+ function_call_output: {"id":"call-1","content":"{\\"exit_status\\":1}"}
13
+ EOF
14
+
15
+ call = Chat.tool_calls(conversation, source: '/tmp/example.chat').first
16
+ assert_equal 'bash', call[:name]
17
+ assert_equal ['/tmp/example.chat', 0], call[:call_address]
18
+ assert_equal ['/tmp/example.chat', 1], call[:output_address]
19
+ assert_equal({ success: false, reason: :exit_status, exit_status: 1 }, Chat.tool_call_status(call))
20
+ end
21
+
22
+ def test_missing_output_is_unknown
23
+ conversation = chat('function_call: {"name":"write","id":"call-2"}')
24
+ status = Chat.tool_call_status(Chat.tool_calls(conversation).first)
25
+ assert_nil status[:success]
26
+ assert_equal :missing_output, status[:reason]
27
+ end
28
+
29
+ def test_exception_output_fails
30
+ conversation = chat <<-EOF
31
+ function_call: {"name":"read","id":"call-3"}
32
+ function_call_output: {"id":"call-3","content":"{\\"exception\\":\\"denied\\"}"}
33
+ EOF
34
+ status = Chat.tool_call_status(Chat.tool_calls(conversation).first)
35
+ assert_equal false, status[:success]
36
+ assert_equal 'denied', status[:exception]
37
+ end
38
+ end
@@ -1,42 +1,11 @@
1
1
  require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
- require 'scout/knowledge_base'
5
- class TestLLMAgent < Test::Unit::TestCase
6
- def _test_system
7
- TmpFile.with_dir do |dir|
8
- kb = KnowledgeBase.new dir
9
- kb.format = {"Person" => "Alias"}
10
- kb.register :brothers, datafile_test(:person).brothers, undirected: true
11
- kb.register :marriages, datafile_test(:person).marriages, undirected: true, source: "=>Alias", target: "=>Alias"
12
- kb.register :parents, datafile_test(:person).parents
13
-
14
- agent = LLM::Agent.new knowledge_base: kb
15
-
16
- ppp agent.ask "Who is Miguel's brother-in-law. Brother in law is your spouses sibling or your sibling's spouse"
17
- end
18
- end
19
-
20
- def _test_workflow_eval
21
- agent = LLM::Agent.new
22
- agent.workflow do
23
- input :c_degrees, :float, "Degrees Celsius"
24
-
25
- task :c_to_f => :float do |c_degrees|
26
- (c_degrees * 9.0 / 5.0) + 32.0
27
- end
28
-
29
- export :c_to_f
30
- end
31
-
32
- agent.user "Convert 30 celsius into faranheit"
33
- res = agent.json_format({conversion: {type: :number}})
34
- assert_equal 86.0, res['conversion']
35
- end
4
+ require 'scout/llm/agent'
36
5
 
6
+ class TestLLMAgent < Test::Unit::TestCase
37
7
  def test_prompt
38
8
  agent = self.agent
39
-
40
9
  agent.start_chat.user <<-EOF
41
10
  My name is Miguel
42
11
  EOF
@@ -47,9 +16,17 @@ user:
47
16
  What is my name?
48
17
  EOF
49
18
 
19
+ # ScoutCoder: agent.prompt goes through LLM.ask, which persists its result;
20
+ # persist: false keeps unit tests away from Scout.var.cache.ask, and the
21
+ # mock backend records what the agent actually sent.
22
+ LLM::Mock.script('Your name is Miguel')
50
23
 
51
- iii chat
52
- ppp agent.prompt chat
24
+ res = agent.prompt chat, persist: false, endpoint: 'mock'
25
+
26
+ assert_equal 'Your name is Miguel', res
27
+
28
+ # the agent pipeline carried its start_chat messages into the request
29
+ messages, _options = LLM::Mock.calls.first
30
+ assert_include messages.collect { |m| m[:role].to_s }, 'user'
53
31
  end
54
32
  end
55
-
@@ -1,66 +1,89 @@
1
1
  require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
- require 'scout/workflow'
4
+ require 'scout/llm/ask'
5
5
  require 'scout/knowledge_base'
6
6
 
7
- class TestLLM < Test::Unit::TestCase
8
- def _test_ask
9
- Log.severity = 0
10
- prompt =<<-EOF
11
- system: you are a coding helper that only write code and comments without formatting so that it can work directly, avoid the initial and end commas ```.
12
- user: write a script that sorts files in a directory
13
- EOF
14
- ppp LLM.ask prompt
15
- ppp LLM.ask prompt
16
- end
17
-
18
- def _test_workflow_ask
19
- m = Module.new do
20
- extend Workflow
21
- self.name = "RecipeWorkflow"
22
-
23
- desc "List the steps to cook a recipe"
24
- input :recipe, :string, "Recipe for which to extract steps"
25
- task :recipe_steps => :array do |recipe|
26
- ["prepare batter", "bake"]
27
- end
28
-
29
- desc "Calculate time spent in each step of the recipe"
30
- input :step, :string, "Cooking step"
31
- task :step_time => :string do |step|
32
- case step
33
- when "prepare batter"
34
- "2 hours"
35
- when "bake"
36
- "30 minutes"
37
- else
38
- "1 minute"
39
- end
40
- end
41
- export :recipe_steps, :step_time
42
- end
43
-
44
- sss 0
45
- ppp LLM.workflow_ask(m, "How much time does it take to prepare a 'vanilla' cake recipe, use the tools provided to find out")
46
- end
7
+ class TestLLMAsk < Test::Unit::TestCase
47
8
 
48
- def test_knowledbase
9
+ # Two-hop tool chain against the real test KnowledgeBase, with only the
10
+ # inference mocked:
11
+ # marriages(Miki) -> Miki~Clei (Clei is Miki's wife)
12
+ # brothers(Clei) -> Clei~Guille (Guille is Clei's brother)
13
+ # final answer: Guille is Miki's brother in law
14
+ # The scripted tool calls are dispatched through LLM.process_calls, so the
15
+ # KnowledgeBase lookups, argument handling and function_call_output
16
+ # messages are all real; the mock only decides what the model "says" next.
17
+ def test_knowledge_base_ask_tool_chain
18
+ Log.severity = 0
19
+ # ScoutCoder: KnowledgeBase.new needs a directory it can write its indices
20
+ # into; registering the databases against TmpFile.with_dir keeps everything
21
+ # inside the test tmpdir (and offline).
49
22
  TmpFile.with_dir do |dir|
50
23
  kb = KnowledgeBase.new dir
51
- kb.format = {"Person" => "Alias"}
52
24
  kb.register :brothers, datafile_test(:person).brothers, undirected: true
53
25
  kb.register :marriages, datafile_test(:person).marriages, undirected: true, source: "=>Alias", target: "=>Alias"
54
- kb.register :parents, datafile_test(:person).parents
55
-
56
- Scout::Config.set(:backend, :openai, :llm)
57
- ppp LLM.knowledge_base_ask(kb, "Who is Miki's brother in law?", log_errors: true, model: 'gpt-4o')
58
- ppp LLM.knowledge_base_ask(kb, "Who is Miki's father in law?", log_errors: true, model: 'gpt-4o')
59
- Scout::Config.set(:backend, :ollama, :llm)
60
- ppp LLM.knowledge_base_ask(kb, "Who is Miki's brother in law?")
61
- ppp LLM.knowledge_base_ask(kb, "Who is Miki's father in law?")
26
+
27
+ LLM::Mock.script(
28
+ {tool_calls: [{name: 'marriages', arguments: {entities: ['Miki'], database: 'marriages'}}]},
29
+ {tool_calls: [{name: 'brothers', arguments: {entities: ['Clei'], database: 'brothers'}}]},
30
+ 'Guille is Miki\'s brother in law'
31
+ )
32
+
33
+ res = LLM.knowledge_base_ask(kb, "Who is Miki's brother in law? call the tool marriages and then brothers, ignore the tolls that return association_details", persist: false, endpoint: :mock)
34
+
35
+ assert_include res, 'Guille'
36
+
37
+ # the tool definitions for both databases reached the (mock) backend
38
+ assert_include LLM::Mock.tool_definitions.keys.collect(&:to_s), 'brothers'
39
+ assert_include LLM::Mock.tool_definitions.keys.collect(&:to_s), 'marriages'
40
+
41
+ # three rounds: marriages, brothers, final answer
42
+ assert_equal 3, LLM::Mock.calls.length
43
+
44
+ # hop 1: the marriages query for Miki really ran and its output was fed
45
+ # back to the model as a function_call_output
46
+ first_input, _first_options = LLM::Mock.calls[1]
47
+ hop1_output = first_input.find { |m| m[:role].to_s == 'function_call_output' }
48
+ assert_include hop1_output[:content].to_s, 'Miki~Clei'
49
+
50
+ # hop 2: the brothers query for Clei (the marriage partner) ran next
51
+ second_input, _second_options = LLM::Mock.calls[2]
52
+ hop2_output = second_input.select { |m| m[:role].to_s == 'function_call_output' }.last
53
+ assert_include hop2_output[:content].to_s, 'Clei~Guille'
62
54
  end
63
55
  end
64
56
 
65
- end
57
+ # Single-ask chain through the association: chat directive (no explicit kb
58
+ # object), so the KnowledgeBase is built from the directive itself.
59
+ def test_knowledge_base_association_tool_chain
60
+ question =<<-EOF
61
+ user:
62
+
63
+ Who is Miki's brother in law?
64
+
65
+ association: brothers #{datafile_test(:person).brothers} undirected=true
66
+ association: marriages #{datafile_test(:person).marriages} undirected=true source="=>Alias" target="=>Alias"
67
+ EOF
66
68
 
69
+ LLM::Mock.script(
70
+ {tool_calls: [{name: 'marriages', arguments: {entities: ['Miki']}}]},
71
+ {tool_calls: [{name: 'brothers', arguments: {entities: ['Clei']}}]},
72
+ 'Guille is Miki\'s brother in law'
73
+ )
74
+
75
+ res = LLM.ask question, persist: false, endpoint: :mock
76
+
77
+ assert_include res, 'Guille'
78
+
79
+ tool_names = LLM::Mock.tool_definitions.keys
80
+ assert_include tool_names, 'brothers'
81
+ assert_include tool_names, 'marriages'
82
+
83
+ # both hops produced their function_call_output messages
84
+ hop1_input, _ = LLM::Mock.calls[1]
85
+ hop2_input, _ = LLM::Mock.calls[2]
86
+ assert_include hop1_input.find { |m| m[:role].to_s == 'function_call_output' }[:content].to_s, 'Miki~Clei'
87
+ assert_include hop2_input.select { |m| m[:role].to_s == 'function_call_output' }.last[:content].to_s, 'Clei~Guille'
88
+ end
89
+ end
@@ -1,14 +1,39 @@
1
1
  require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
- class TestMessages < Test::Unit::TestCase
4
+ require 'rbbt/workflow'
5
+ require 'scout/knowledge_base'
6
+
7
+ # Offline replacement for the remote `Baking` workflow used by the original
8
+ # task:/tool: tests. Chat directives resolve workflow names with
9
+ # Kernel.const_get first (see Chat.load_workflow), so a named module is
10
+ # reachable without any git clone.
11
+ TestBaking = Module.new do
12
+ extend Workflow
13
+ self.name = "TestBaking"
14
+
15
+ desc "Bake a tray of muffins"
16
+ input :blueberries, :boolean, "Add blueberries", true
17
+ input :title, :string, "Recipe title"
18
+ input :list, :array, "Ingredient list"
19
+ task :bake_muffin_tray => :string do |blueberries, title, list|
20
+ "Baking muffins: #{title} (#{list.to_a * ', '}) blueberries=#{blueberries}"
21
+ end
5
22
 
23
+ desc "List the steps to cook a recipe"
24
+ input :recipe, :string, "Recipe for which to extract steps"
25
+ task :recipe_steps => :array do |recipe|
26
+ ["prepare batter", "bake"]
27
+ end
28
+ end
29
+
30
+ class TestMessages < Test::Unit::TestCase
6
31
 
7
32
  def test_task
8
33
  question =<<-EOF
9
34
  user:
10
35
 
11
- task: Baking bake_muffin_tray blueberries=true title="This is a title" list=one,two,"and three"
36
+ task: TestBaking bake_muffin_tray blueberries=true title="This is a title" list=one,two,"and three"
12
37
 
13
38
  How are muffins made?
14
39
 
@@ -17,48 +42,76 @@ How are muffins made?
17
42
  TmpFile.with_file question do |file|
18
43
  messages = LLM.chat file
19
44
  assert_include messages.collect{|m| m[:role] }, 'function_call'
20
- assert_include messages.find{|m| m[:role] == 'function_call_output' }[:content], 'Baking'
45
+ assert_include messages.find{|m| m[:role] == 'function_call' }[:content], 'TestBaking'
46
+ assert_include messages.find{|m| m[:role] == 'function_call_output' }[:content], 'Baking muffins'
21
47
  end
22
48
  end
23
49
 
24
50
  def test_tool
25
51
  require 'scout/llm/ask'
26
52
 
27
- sss 0
28
53
  question =<<-EOF
29
54
  user:
30
55
 
31
56
  Use the provided tool to learn the instructions of baking a tray of muffins. Don't
32
57
  give me your own recipe, return the one provided by the tool
33
58
 
34
- tool: Baking
59
+ tool: TestBaking bake_muffin_tray
35
60
  EOF
36
61
 
62
+ LLM::Mock.script 'The instructions say: bake the muffin tray'
63
+
37
64
  TmpFile.with_file question do |file|
38
- ppp LLM.ask file, endpoint: :nano
65
+ res = LLM.ask file, persist: false, endpoint: :mock
66
+ assert_equal 'The instructions say: bake the muffin tray', res
67
+
68
+ # the tool:/directive reached the backend as a workflow tool definition
69
+ assert_include LLM::Mock.tool_definitions.keys, 'bake_muffin_tray'
70
+ obj, definition = LLM::Mock.tool_definitions['bake_muffin_tray']
71
+ assert_equal TestBaking, obj
39
72
  end
40
73
  end
41
74
 
42
75
  def test_tools_with_task
43
76
  require 'scout/llm/ask'
44
77
 
78
+ # ScoutCoder: LLM.ask wraps the backend call in Persist.persist even when
79
+ # persist: false is passed, so two asks with the exact same question text
80
+ # collide on the cache key and the second one is served from cache without
81
+ # ever reaching the backend. Give each test a distinct question so every
82
+ # ask really exercises the (mock) backend.
45
83
  question =<<-EOF
46
84
  user:
47
85
 
48
- Use the provided tool to learn the instructions of baking a tray of muffins. Don't
86
+ Use the provided tool and then give me the answer you obtained from it. Don't
49
87
  give me your own recipe, return the one provided by the tool
50
88
 
51
- tool: Baking bake_muffin_tray
89
+ tool: TestBaking bake_muffin_tray
52
90
  EOF
53
91
 
92
+ LLM::Mock.script(
93
+ {tool_calls: [{name: 'bake_muffin_tray', arguments: {}}]},
94
+ 'The instructions say: bake the muffin tray'
95
+ )
96
+
54
97
  TmpFile.with_file question do |file|
55
- ppp LLM.ask file
98
+ res = LLM.ask file, persist: false, endpoint: :mock
99
+ assert_equal 'The instructions say: bake the muffin tray', res
100
+
101
+ # the scripted tool call actually ran the workflow task: the mock
102
+ # backend saw the tool definition and returned the final answer
103
+ assert_include LLM::Mock.tool_definitions.keys, 'bake_muffin_tray'
104
+
105
+ # the task output was really executed and fed back to the model
106
+ tool_input, _ = LLM::Mock.calls[1]
107
+ output = tool_input.find { |m| m[:role].to_s == 'function_call_output' }
108
+ assert_include output[:content].to_s, 'Baking muffins'
56
109
  end
57
110
  end
58
111
 
59
112
  def test_knowledge_base
60
113
  require 'scout/llm/ask'
61
- sss 0
114
+
62
115
  question =<<-EOF
63
116
  system:
64
117
 
@@ -66,15 +119,56 @@ Query the knowledge base of familiar relationships to answer the question
66
119
 
67
120
  user:
68
121
 
69
- Who is Miki's brother in law?
122
+ Who is Miki's brother in law? Use the associations.
70
123
 
71
124
  association: brothers #{datafile_test(:person).brothers} undirected=true
72
125
  association: marriages #{datafile_test(:person).marriages} undirected=true source="=>Alias" target="=>Alias"
73
126
  EOF
74
127
 
128
+ LLM::Mock.script 'Guille is Miki\'s brother in law'
129
+
75
130
  TmpFile.with_file question do |file|
76
- ppp LLM.ask file
131
+ res = LLM.ask file, persist: false, endpoint: :mock
132
+ assert_equal "Guille is Miki's brother in law", res
133
+
134
+ # the association: directives produced knowledge base tool definitions
135
+ tool_names = LLM::Mock.tool_definitions.keys
136
+ assert_include tool_names, 'brothers'
137
+ assert_include tool_names, 'marriages'
77
138
  end
78
139
  end
79
- end
80
140
 
141
+ # Two-hop knowledge base tool chain over the association: directives: only
142
+ # the inference is mocked (marriages Miki -> Clei, brothers Clei -> Guille),
143
+ # the database queries run for real. See test/scout/llm/test_ask.rb for the
144
+ # knowledge_base_ask variant.
145
+ def test_knowledge_base_tool_execution
146
+ require 'scout/llm/ask'
147
+
148
+ question =<<-EOF
149
+ user:
150
+
151
+ Who is Miki's brother in law? Use the associations.
152
+
153
+ association: brothers #{datafile_test(:person).brothers} undirected=true
154
+ association: marriages #{datafile_test(:person).marriages} undirected=true source="=>Alias" target="=>Alias"
155
+ EOF
156
+
157
+ # entities must be a JSON array of entity identifiers, not a bare string
158
+ # (call_knowledge_base JSON.parses it when it is a String)
159
+ LLM::Mock.script(
160
+ {tool_calls: [{name: 'marriages', arguments: {'entities' => '["Miki"]'}}]},
161
+ {tool_calls: [{name: 'brothers', arguments: {'entities' => '["Clei"]'}}]},
162
+ "Guille is Miki's brother in law"
163
+ )
164
+
165
+ res = LLM.ask question, persist: false, endpoint: :mock
166
+ assert_include res, 'Guille'
167
+ assert_include LLM::Mock.tool_definitions.keys, 'brothers'
168
+
169
+ hop1_input, _ = LLM::Mock.calls[1]
170
+ hop2_input, _ = LLM::Mock.calls[2]
171
+ assert_include hop1_input.find { |m| m[:role].to_s == 'function_call_output' }[:content].to_s, 'Miki~Clei'
172
+ assert_include hop2_input.select { |m| m[:role].to_s == 'function_call_output' }.last[:content].to_s, 'Clei~Guille'
173
+ end
174
+ end
@@ -0,0 +1,48 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
+
4
+ require 'scout/llm/embed'
5
+
6
+ class TestLLMEmbed < Test::Unit::TestCase
7
+ def test_embed_deterministic_fixed_dimension
8
+ # ScoutCoder: the default embed backend for tests is the registered
9
+ # LLM::Mock backend (Scout::Config.set({backend: :mock}, :embed, :llm) in
10
+ # test_helper), so LLM.embed resolves through the LLM::BACKENDS registry
11
+ # fallback and never touches the network.
12
+ v1 = LLM.embed('a text', endpoint: :mock)
13
+ v2 = LLM.embed('a text', endpoint: :mock)
14
+
15
+ assert_instance_of Array, v1
16
+ assert_equal LLM::Mock::DIMENSIONS, v1.length
17
+ assert v1.all? { |e| Float === e }
18
+ assert_equal v1, v2
19
+ end
20
+
21
+ def test_embed_array_input
22
+ vectors = LLM.embed(['one two', 'two three'], endpoint: :mock)
23
+
24
+ assert_equal 2, vectors.length
25
+ assert_equal [LLM::Mock::DIMENSIONS, LLM::Mock::DIMENSIONS], vectors.collect(&:length)
26
+ assert_equal LLM.embed('one two', endpoint: :mock), vectors.first
27
+ end
28
+
29
+ def test_embed_shared_words_closer
30
+ # cosine similarity: texts sharing words must be closer than disjoint texts
31
+ def cos(a, b)
32
+ dot = a.zip(b).inject(0.0) { |acc, (x, y)| acc + x * y }
33
+ na = Math.sqrt(a.inject(0.0) { |acc, x| acc + x * x })
34
+ nb = Math.sqrt(b.inject(0.0) { |acc, x| acc + x * x })
35
+ dot / (na * nb)
36
+ end
37
+
38
+ shared = LLM.embed('crime and theft', endpoint: :mock)
39
+ similar = LLM.embed('crime theft violence', endpoint: :mock)
40
+ other = LLM.embed('puppies and flowers', endpoint: :mock)
41
+
42
+ assert cos(shared, similar) > cos(shared, other)
43
+ end
44
+
45
+ def test_embed_explicit_backend
46
+ assert_equal LLM::Mock.embed('a text', endpoint: :mock), LLM.embed('a text', backend: :mock)
47
+ end
48
+ end
@@ -4,7 +4,7 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
4
4
  require 'scout/llm/embed'
5
5
 
6
6
  class TestLLMRAG < Test::Unit::TestCase
7
- def test_rag
7
+ def _test_rag
8
8
  text1 =<<-EOF
9
9
  Crime, Killing and Theft.
10
10
  EOF
@@ -15,19 +15,27 @@ Murder, felony and violence
15
15
  Puppies, cats and flowers
16
16
  EOF
17
17
 
18
- data = [ LLM.embed(text1),
19
- LLM.embed(text2),
20
- LLM.embed(text3)]
18
+ data = [ LLM.embed(text1, endpoint: :mock),
19
+ LLM.embed(text2, endpoint: :mock),
20
+ LLM.embed(text3, endpoint: :mock)]
21
21
 
22
22
  i = LLM::RAG.index(data)
23
- nodes, scores = i.search_knn LLM.embed('I love the zoo'), 1
23
+
24
+ # ScoutCoder: the mock embedding is a bag-of-words hash, so nearest
25
+ # neighbour assertions have to be built from literal word overlap
26
+ # ('violence' ties crime/murder texts, 'flowers' is unique to pets).
27
+ nodes, scores = i.search_knn LLM.embed('Puppies, cats and flowers', endpoint: :mock), 1
24
28
  assert_equal 2, nodes.first
25
29
 
26
- nodes, scores = i.search_knn LLM.embed('The victim got stabbed'), 2
27
- assert_equal [0, 1], nodes.sort
30
+ nodes, scores = i.search_knn LLM.embed('Murder and violence', endpoint: :mock), 2
31
+ assert_include nodes.sort, 1
32
+ assert_false nodes.sort.first == 2
33
+
34
+ # deterministic: same text always the same vector
35
+ assert_equal data.first, LLM.embed(text1, endpoint: :mock)
28
36
  end
29
37
 
30
- def test_rag_insity
38
+ def test_rag_insitu
31
39
  text1 =<<-EOF
32
40
  Crime, Killing and Theft.
33
41
  EOF
@@ -38,18 +46,17 @@ Murder, felony and violence
38
46
  Puppies, cats and flowers
39
47
  EOF
40
48
 
41
- LLM::RAG.top([text1, text2, text3], 2)
42
-
43
- data = [ LLM.embed(text1),
44
- LLM.embed(text2),
45
- LLM.embed(text3)]
49
+ data = [ LLM.embed(text1, endpoint: :mock),
50
+ LLM.embed(text2, endpoint: :mock),
51
+ LLM.embed(text3, endpoint: :mock)]
46
52
 
47
53
  i = LLM::RAG.index(data)
48
- nodes, scores = i.search_knn LLM.embed('I love the zoo'), 1
54
+ nodes, scores = i.search_knn LLM.embed('Puppies, cats and flowers', endpoint: :mock), 1
49
55
  assert_equal 2, nodes.first
50
56
 
51
- nodes, scores = i.search_knn LLM.embed('The victim got stabbed'), 2
52
- assert_equal [0, 1], nodes.sort
57
+ nodes, scores = i.search_knn LLM.embed('Murder and violence', endpoint: :mock), 2
58
+ assert_include nodes.sort, 1
59
+ assert_false nodes.sort.first == 2
53
60
  end
54
61
  end
55
62
 
@@ -33,7 +33,18 @@ class TestLLMTools < Test::Unit::TestCase
33
33
  LLM.task_tool_definition(m, :step_time)
34
34
 
35
35
  tool_definitions = LLM.workflow_tools(m)
36
- ppp JSON.pretty_generate tool_definitions
36
+
37
+ # workflow_tools returns {task => [workflow, definition]}; full end-to-end
38
+ # coverage (including call_workflow) lives in
39
+ # test/scout/llm/tools/test_workflow.rb
40
+ assert_equal %i(recipe_steps step_time).sort, tool_definitions.keys.sort
41
+ assert_equal m, tool_definitions[:recipe_steps].first
42
+
43
+ definition = tool_definitions[:recipe_steps].last
44
+ assert_equal :recipe_steps, definition[:name]
45
+ assert_equal 'List the steps to cook a recipe', definition[:description]
46
+ assert_equal :string, definition[:parameters][:properties][:recipe][:type]
47
+ assert_equal 'Recipe for which to extract steps', definition[:parameters][:properties][:recipe][:description]
37
48
  end
38
49
 
39
50
  def test_knowledbase_definition
@@ -13,7 +13,6 @@ class TestLLMToolKB < Test::Unit::TestCase
13
13
  assert_equal Person, kb.target_type(:parents)
14
14
 
15
15
  knowledge_base_definition = LLM.knowledge_base_tool_definition(kb)
16
- ppp JSON.pretty_generate knowledge_base_definition
17
16
 
18
17
  assert_equal ['Isa~Miki', 'Miki~Isa', 'Guille~Clei'], LLM.call_knowledge_base(kb, :brothers, entities: %w(Isa Miki Guille))
19
18
  end
@@ -2,10 +2,12 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestClass < Test::Unit::TestCase
5
- def test_client
5
+ # Real remote MCP server version moved to
6
+ # test/integration/scout/llm/tools/test_mcp.rb: LLM.mcp_tools needs a live
7
+ # MCP server (no seam to fake). Stdio coverage stays unit-side in
8
+ # test/scout/llm/test_mcp.rb.
9
+ def _test_client
6
10
  c = LLM.mcp_tools("https://api.githubcopilot.com/mcp/")
7
11
  assert_include c.keys, "get_me"
8
12
  end
9
13
  end
10
-
11
-
@@ -31,9 +31,30 @@ class TestLLMToolWorkflow < Test::Unit::TestCase
31
31
  LLM.task_tool_definition(m, :step_time)
32
32
 
33
33
  tool_definitions = LLM.workflow_tools(m)
34
- ppp JSON.pretty_generate tool_definitions
35
34
 
36
- assert_equal ["prepare batter", "bake"], LLM.call_workflow(m, :recipe_steps)
35
+ # workflow_tools returns {task => [workflow, definition]}
36
+ assert_equal %i(recipe_steps step_time).sort, tool_definitions.keys.sort
37
+ assert_equal m, tool_definitions[:recipe_steps].first
38
+
39
+ definition = tool_definitions[:recipe_steps].last
40
+ assert_equal :recipe_steps, definition[:name]
41
+ assert definition[:parameters][:properties].include?('recipe') || definition[:parameters][:properties].include?(:recipe)
42
+
43
+ # ScoutCoder: call_workflow returns a Step (the job) for regular tasks; the
44
+ # caller is expected to produce/read it. Only exec exports (or
45
+ # exec_type: 'exec') return the literal job result inline.
46
+ job = LLM.call_workflow(m, :recipe_steps)
47
+ assert(Step === job)
48
+ job.produce
49
+ assert_equal ["prepare batter", "bake"], job.load
50
+
51
+ exec_result = LLM.call_workflow(m, :recipe_steps, exec_type: 'exec')
52
+ assert_equal ["prepare batter", "bake"], exec_result
53
+
54
+ path = LLM.call_workflow(m, :recipe_steps, return_path: true)
55
+ assert_equal ["prepare batter", "bake"], Open.read(path).split("\n")
56
+
57
+ assert_equal "30 minutes", LLM.call_workflow(m, :step_time, step: 'bake', exec_type: 'exec')
37
58
  end
38
59
  end
39
60
 
@@ -1,25 +1,31 @@
1
1
  require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
+ MODEL = 'distilgpt2'
5
+
4
6
  require 'scout-ai'
5
7
  class TestClass < Test::Unit::TestCase
8
+ # Conditional omission: the local-cache probe never downloads; training
9
+ # 1000 epochs also needs torch, which is probed the same bounded way.
6
10
  def test_main
7
- model = NextTokenModel.new
11
+ omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
12
+ reason = Availability.python_modules_reason('transformers')
13
+ omit "python infrastructure missing: #{reason}" if reason
14
+
15
+ model = NextTokenModel.new
8
16
  train_texts = [
9
17
  "say hi, no!",
10
18
  "say hi, no no no",
11
19
  "say hi, hi ",
12
20
  "say hi, hi how are you ",
13
21
  "say hi, hi are you good",
14
- ]
15
-
16
- model_name = "distilgpt2" # Replace with your local/other HF Llama checkpoint as needed
22
+ ]
17
23
 
18
24
  TmpFile.with_path do |tmp_dir|
19
25
  iii tmp_dir
20
26
 
21
27
  sss 0
22
- model = NextTokenModel.new model_name, tmp_dir, training_num_train_epochs: 1000, training_learning_rate: 0.1
28
+ model = NextTokenModel.new MODEL, tmp_dir, training_num_train_epochs: 1000, training_learning_rate: 0.1
23
29
 
24
30
  iii :new
25
31
  chat = Chat.setup []