scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -2,9 +2,15 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
4
  class TestClass < Test::Unit::TestCase
5
+ MODEL = 'mistralai/Mistral-7B-Instruct-v0.3'
6
+
7
+ # Conditional omission: runs only when the model is already in the local
8
+ # huggingface cache. The probe never downloads (see
9
+ # test/support/availability.rb).
5
10
  def test_eval_chat
11
+ omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
6
12
  #model = CausalModel.new 'BSC-LT/salamandra-2b-instruct'
7
- model = CausalModel.new 'mistralai/Mistral-7B-Instruct-v0.3'
13
+ model = CausalModel.new MODEL
8
14
 
9
15
  model.init
10
16
 
@@ -17,8 +23,9 @@ class TestClass < Test::Unit::TestCase
17
23
  end
18
24
 
19
25
  def test_eval_train
26
+ omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
20
27
  #model = CausalModel.new 'BSC-LT/salamandra-2b-instruct'
21
- model = CausalModel.new 'mistralai/Mistral-7B-Instruct-v0.3'
28
+ model = CausalModel.new MODEL
22
29
 
23
30
  model.init
24
31
 
@@ -30,4 +37,3 @@ class TestClass < Test::Unit::TestCase
30
37
  ])
31
38
  end
32
39
  end
33
-
@@ -1,17 +1,26 @@
1
1
  require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
2
  require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
3
3
 
4
+ MODEL = 'bert-base-uncased'
5
+
4
6
  class TestSequenceClassification < Test::Unit::TestCase
5
7
  def _test_eval_sequence_classification
6
- model = SequenceClassificationModel.new 'bert-base-uncased', nil,
8
+ model = SequenceClassificationModel.new MODEL, nil,
7
9
  class_labels: %w(Bad Good)
8
10
 
9
11
  assert_include ["Bad", "Good"], model.eval("This is dog")
10
12
  assert_include ["Bad", "Good"], model.eval_list(["This is dog", "This is cat"]).first
11
13
  end
12
14
 
15
+ # Conditional omission: the probe only checks the local huggingface cache
16
+ # (never downloads) and that transformers/datasets are importable, so this
17
+ # runs whenever the environment can actually serve it.
13
18
  def test_train_sequence_classification
14
- model = SequenceClassificationModel.new 'bert-base-uncased', nil,
19
+ reason = Availability.python_modules_reason('transformers', 'datasets')
20
+ omit "python infrastructure missing: #{reason}" if reason
21
+ omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
22
+
23
+ model = SequenceClassificationModel.new MODEL, nil,
15
24
  class_labels: %w(Bad Good)
16
25
 
17
26
  model.init
@@ -3,6 +3,8 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
3
3
 
4
4
  class TestTorch < Test::Unit::TestCase
5
5
  def test_linear
6
+ omit "No python environment" unless Availability.python?
7
+ omit "Torch not installed" unless Availability.python_modules?(:torch)
6
8
  model = nil
7
9
 
8
10
  TmpFile.with_dir do |dir|
@@ -3,7 +3,11 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
3
3
 
4
4
  require 'scout/model/python/base'
5
5
  class TestTorchHelpers < Test::Unit::TestCase
6
+ # Conditional omission: the CUDA probe imports torch in a bounded
7
+ # subprocess and checks torch.cuda.is_available(), so this runs whenever a
8
+ # GPU-backed torch is actually present.
6
9
  def test_del
10
+ omit 'torch CUDA unavailable (probe: torch.cuda.is_available)' unless Availability.cuda?
7
11
  ScoutPython.init_scout
8
12
  ScoutPython.pyimport :torch
9
13
  batch = [[100.0]]
@@ -34,7 +34,10 @@ class TestClass < Test::Unit::TestCase
34
34
  assert_equal [4, 8], model.eval_list([1, 2])
35
35
  end
36
36
 
37
- def test_R_model
37
+ # Real R + e1071 version moved to test/integration/scout/model/test_base.rb
38
+ # (installs the R package online); the trivial ScoutModel tests above stay
39
+ # unit-side.
40
+ def _test_R_model
38
41
  require 'rbbt-util'
39
42
  require 'rbbt/util/R'
40
43
 
@@ -114,4 +117,3 @@ cat(label, file="#{results}");
114
117
  end
115
118
  end
116
119
  end
117
-
@@ -0,0 +1,231 @@
1
+ # Test support: bounded availability probes for conditional test omissions.
2
+ #
3
+ # Not named test_*.rb on purpose: the Rakefile test pattern
4
+ # ('test/**/test_*.rb') must not collect this file.
5
+ #
6
+ # Every probe returns true/false (never raises, never downloads, never hangs).
7
+ # Intended usage in a test:
8
+ #
9
+ # omit 'torch CUDA unavailable' unless Availability.cuda?
10
+ #
11
+ # Guidelines honoured here:
12
+ # * Huggingface model presence is checked against the local HF cache only
13
+ # (HF_HOME/hub or ~/.cache/huggingface/hub), NEVER through the
14
+ # transformers API: importing transformers or building a tokenizer would
15
+ # start a multi-GB download for anything that is not already cached.
16
+ # * Python module probes run `python -c "import ..."` in a fresh process
17
+ # with a hard timeout, so a hanging import cannot block the suite.
18
+ # * Results are memoized per process: probes are cheap but not free.
19
+ module Availability
20
+ DEFAULT_TIMEOUT = 60
21
+
22
+ class << self
23
+ def memo(name)
24
+ cache = (@cache ||= {})
25
+ return cache[name] if cache.key?(name)
26
+ cache[name] = begin
27
+ yield
28
+ rescue Exception
29
+ false
30
+ end
31
+ end
32
+
33
+ def python_executable
34
+ ENV['SCOUT_TEST_PYTHON'] || 'python'
35
+ end
36
+
37
+ # python itself is usable (python executable present and runnable)
38
+ def python?
39
+ memo(:python) do
40
+ run_bounded([python_executable, '-c', 'import sys; sys.exit(0)'], 30)
41
+ true
42
+ end
43
+ end
44
+
45
+ # all listed python modules import cleanly (fresh process, bounded)
46
+ def python_modules?(*mods)
47
+ memo(:"python_modules_#{mods.join(',')}") do
48
+ raise Errno::ENOENT unless python?
49
+ code = mods.collect { |m| "import #{m}" } * ';'
50
+ _out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
51
+ status == 0
52
+ end
53
+ end
54
+
55
+ # informative reason string when a python module is missing, nil otherwise
56
+ def python_modules_reason(*mods)
57
+ return 'python unavailable' unless python?
58
+ code = mods.collect { |m| "import #{m}" } * ';'
59
+ out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
60
+ return nil if status == 0
61
+ "python modules missing (#{mods * ','}): #{out.split("\n").last.to_s.strip}"
62
+ end
63
+
64
+ # torch is importable (does NOT imply CUDA)
65
+ def torch?
66
+ python_modules?('torch')
67
+ end
68
+
69
+ # torch is importable AND reports a usable CUDA device
70
+ def cuda?
71
+ memo(:cuda) do
72
+ return false unless torch?
73
+ code = 'import sys, torch; sys.exit(0 if torch.cuda.is_available() else 1)'
74
+ _out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
75
+ status == 0
76
+ end
77
+ end
78
+
79
+ # A Huggingface model is already in the local cache; this NEVER triggers a
80
+ # download (see the note at the top of the file).
81
+ def hf_model_cached?(repo)
82
+ memo(:"hf_model_#{repo}") do
83
+ hub = ENV['HF_HOME'] ? File.join(ENV['HF_HOME'], 'hub') : File.join(Dir.home, '.cache', 'huggingface', 'hub')
84
+ repo_dir = 'models--' + repo.to_s.tr('/', '--')
85
+ path = File.join(hub, repo_dir)
86
+ Dir.exist?(path) && Dir.glob(File.join(path, 'snapshots', '*')).any?
87
+ end
88
+ end
89
+
90
+ # Reason for a missing huggingface model
91
+ def hf_model_reason(repo)
92
+ return "huggingface model #{repo} not in local cache (HF_HOME=#{ENV['HF_HOME'] || '~/.cache/huggingface'}); probing would download it" unless hf_model_cached?(repo)
93
+ nil
94
+ end
95
+
96
+ # Rscript is present
97
+ def rscript?
98
+ memo(:rscript) { which?('Rscript') }
99
+ end
100
+
101
+ # R is present and the listed packages are installed (bounded Rscript -e)
102
+ def r_packages?(*pkgs)
103
+ memo(:"r_packages_#{pkgs.join(',')}") do
104
+ return false unless rscript?
105
+ code = "cat(if (all(c(#{pkgs.collect { |p| "'#{p}'" } * ', '}) %in% rownames(installed.packages()))) 'yes' else 'no')"
106
+ out, _err, status = run_bounded(['Rscript', '-e', code], DEFAULT_TIMEOUT, capture: true)
107
+ status == 0 && out.strip == 'yes'
108
+ end
109
+ end
110
+
111
+ def r_packages_reason(*pkgs)
112
+ return 'Rscript not found' unless rscript?
113
+ return nil if r_packages?(*pkgs)
114
+ "R packages missing: #{pkgs * ','}"
115
+ end
116
+
117
+ #{{{ endpoints
118
+
119
+ # An endpoint yaml exists in the Scout configuration paths (test/etc/AI,
120
+ # ~/.scout/etc/AI, ...). Purely local: no network access.
121
+ def endpoint_configured?(endpoint)
122
+ memo(:"endpoint_configured_#{endpoint}") do
123
+ Scout.etc.AI[endpoint.to_s].find_with_extension(:yaml).exists?
124
+ end
125
+ end
126
+
127
+ # The endpoint points to a host:port that accepts a TCP connection within
128
+ # the timeout. Only probed for endpoints whose config carries a url;
129
+ # endpoints without a url (mock, relay through scp, ...) count as
130
+ # reachable once configured.
131
+ def endpoint_reachable?(endpoint, timeout = 5)
132
+ memo(:"endpoint_reachable_#{endpoint}") do
133
+ path = Scout.etc.AI[endpoint.to_s].find_with_extension(:yaml)
134
+ return false unless path.exists?
135
+ url = path.yaml[:url] || path.yaml['url']
136
+ return true if url.nil?
137
+
138
+ require 'uri'
139
+ require 'socket'
140
+ uri = URI.parse(url.to_s)
141
+ port = uri.port || (uri.scheme == 'https' ? 443 : 80)
142
+ begin
143
+ Timeout.timeout(timeout) do
144
+ TCPSocket.open(uri.host, port) { |s| s.close }
145
+ end
146
+ true
147
+ rescue Exception
148
+ false
149
+ end
150
+ end
151
+ end
152
+
153
+ # Combined: configured AND (no url OR reachable)
154
+ def endpoint_available?(endpoint)
155
+ return false unless endpoint_configured?(endpoint)
156
+ endpoint_reachable?(endpoint)
157
+ end
158
+
159
+ def endpoint_reason(endpoint)
160
+ return "endpoint #{endpoint} not configured (no yaml in the Scout AI paths)" unless endpoint_configured?(endpoint)
161
+ return "endpoint #{endpoint} configured but not reachable" unless endpoint_reachable?(endpoint)
162
+ nil
163
+ end
164
+
165
+ # All endpoint names defined in Scout.etc.AI across the search paths
166
+ # (test/etc/AI plus the user configuration). Local filesystem only.
167
+ def configured_endpoints
168
+ memo(:configured_endpoints) do
169
+ Scout.etc.AI.glob('*.yaml').collect { |f| File.basename(f, '.yaml') }.uniq
170
+ end
171
+ end
172
+
173
+ # Endpoints that are real inference services (anything but the offline
174
+ # mock one defined by this test suite).
175
+ def real_endpoints
176
+ memo(:real_endpoints) do
177
+ configured_endpoints.reject { |e| e == 'mock' }
178
+ end
179
+ end
180
+
181
+ #{{{ internals
182
+
183
+ def which?(cmd)
184
+ ENV['PATH'].to_s.split(File::PATH_SEPARATOR).any? do |dir|
185
+ path = File.join(dir, cmd.to_s)
186
+ File.executable?(path)
187
+ end
188
+ end
189
+
190
+ # Runs cmd with a hard timeout. Returns [stdout, stderr, exit_status].
191
+ # ScoutCoder: Open3.capture3 threads cannot be killed, so Timeout.timeout
192
+ # only breaks the wait; the child is detached with Process.kill(:KILL) to
193
+ # guarantee the probe returns even when the command hangs (e.g. a python
194
+ # import that blocks on a broken environment).
195
+ def run_bounded(cmd, timeout = DEFAULT_TIMEOUT, capture: true)
196
+ out_r, out_w = IO.pipe
197
+ err_r, err_w = IO.pipe
198
+
199
+ pid = Process.spawn(*cmd, out: out_w, err: err_w)
200
+ out_w.close
201
+ err_w.close
202
+
203
+ status = nil
204
+ begin
205
+ Timeout.timeout(timeout) do
206
+ out = capture ? out_r.read : nil
207
+ err = capture ? err_r.read : nil
208
+ _pid, status = Process.wait2(pid)
209
+ return [out.to_s, err.to_s, status.exitstatus]
210
+ end
211
+ rescue Timeout::Error
212
+ begin
213
+ Process.kill(:KILL, pid)
214
+ rescue Errno::ESRCH
215
+ end
216
+ Process.detach(pid)
217
+ return ['', 'timeout', -1]
218
+ rescue Errno::ENOENT
219
+ begin
220
+ Process.kill(:KILL, pid)
221
+ rescue Errno::ESRCH, Errno::EPERM
222
+ end
223
+ return ['', 'not found', -1]
224
+ ensure
225
+ out_r.close rescue nil
226
+ err_r.close rescue nil
227
+ end
228
+ ['', 'unknown', -1]
229
+ end
230
+ end
231
+ end
@@ -0,0 +1,138 @@
1
+ # Test support: offline stand-ins for the provider HTTP clients.
2
+ #
3
+ # Not named test_*.rb on purpose: the Rakefile test pattern
4
+ # ('test/**/test_*.rb') must not collect this file.
5
+ #
6
+ # Each fake wraps one provider SDK client class. They all share:
7
+ # * a response list replayed in order (last entry repeats when exhausted)
8
+ # * `.calls` recording every parameters hash for assertions
9
+ # * optional failure injection: an Exception instance entry is raised
10
+ #
11
+ # Recorded parameters are plain Hash/IndiferentHash values so assertions can
12
+ # deep-compare them against expected payloads without type surprises.
13
+ class FakeResponseClientBase
14
+ attr_reader :calls, :responses
15
+
16
+ # NOTE: entries are NOT flattened. A single entry may itself be an Array
17
+ # (e.g. the Ollama API returns an array of chunk hashes per call, and
18
+ # OLlama#process_response iterates the returned value).
19
+ def initialize(*responses)
20
+ @responses = responses
21
+ @calls = []
22
+ @index = 0
23
+ end
24
+
25
+ # Replays the next scripted response. Exception entries are raised so tests
26
+ # can exercise provider error paths.
27
+ def next_response
28
+ entry = @responses[@index] || @responses.last
29
+ @index += 1
30
+ raise entry if Exception === entry
31
+ entry
32
+ end
33
+
34
+ def record(parameters)
35
+ @calls << IndiferentHash.setup(parameters.dup)
36
+ end
37
+ end
38
+
39
+ # ruby-openai Chat Completions + embeddings:
40
+ # client.chat(parameters:), client.embeddings(parameters:)
41
+ class FakeOpenAIChatClient < FakeResponseClientBase
42
+ def chat(parameters: {})
43
+ record(parameters)
44
+ next_response
45
+ end
46
+
47
+ def embeddings(parameters: {})
48
+ record(parameters)
49
+ next_response
50
+ end
51
+ end
52
+
53
+ # ruby-openai Responses API: client.responses.create(parameters:)
54
+ class FakeResponsesClient < FakeResponseClientBase
55
+ ResponseStub = Struct.new(:create)
56
+
57
+ def responses
58
+ self
59
+ end
60
+
61
+ def create(parameters: {})
62
+ record(parameters)
63
+ next_response
64
+ end
65
+ end
66
+
67
+ # anthropic gem: client.messages(parameters:)
68
+ class FakeAnthropicClient < FakeResponseClientBase
69
+ def messages(parameters: {})
70
+ record(parameters)
71
+ next_response
72
+ end
73
+ end
74
+
75
+ # ollama-ai gem: client.chat(parameters) (positional),
76
+ # client.request(path, parameters)
77
+ class FakeOllamaClient < FakeResponseClientBase
78
+ def chat(parameters = {})
79
+ record(parameters)
80
+ next_response
81
+ end
82
+
83
+ def request(path, parameters = {})
84
+ record(parameters.merge(path: path))
85
+ next_response
86
+ end
87
+ end
88
+
89
+ # aws-sdk-bedrockruntime: client.invoke_model(model_id:, content_type:, body:)
90
+ # returning an object whose .body.string is the JSON payload.
91
+ class FakeBedrockClient
92
+ BodyStub = Struct.new(:string)
93
+ ResponseStub = Struct.new(:body)
94
+
95
+ attr_reader :calls
96
+
97
+ def initialize(*payloads)
98
+ @payloads = payloads
99
+ @calls = []
100
+ @index = 0
101
+ end
102
+
103
+ # Records the request (body parsed back to a Hash for assertions) and returns
104
+ # the next scripted payload wrapped as invoke_model does.
105
+ def invoke_model(model_id:, content_type:, body:)
106
+ payload = @payloads[@index] || @payloads.last
107
+ @index += 1
108
+ @calls << IndiferentHash.setup({model_id: model_id, content_type: content_type, body: JSON.parse(body)})
109
+ ResponseStub.new(BodyStub.new(payload.to_json))
110
+ end
111
+ end
112
+
113
+ module TestFixtures
114
+ class << self
115
+ # Build a fake client pre-loaded with the named fixtures, e.g.
116
+ # TestFixtures.openai_client('backends/openai_chat_tool_call',
117
+ # 'backends/openai_chat')
118
+ def openai_client(*names)
119
+ FakeOpenAIChatClient.new(*names.collect { |n| fixture(n) })
120
+ end
121
+
122
+ def responses_client(*names)
123
+ FakeResponsesClient.new(*names.collect { |n| fixture(n) })
124
+ end
125
+
126
+ def anthropic_client(*names)
127
+ FakeAnthropicClient.new(*names.collect { |n| fixture(n) })
128
+ end
129
+
130
+ def ollama_client(*names)
131
+ FakeOllamaClient.new(*names.collect { |n| fixture(n) })
132
+ end
133
+
134
+ def bedrock_client(*names)
135
+ FakeBedrockClient.new(*names.collect { |n| fixture(n) })
136
+ end
137
+ end
138
+ end
@@ -0,0 +1,21 @@
1
+ # Test support: fixture loading for recorded LLM provider payloads.
2
+ #
3
+ # Not named test_*.rb on purpose: the Rakefile test pattern
4
+ # ('test/**/test_*.rb') must not collect this file.
5
+ module TestFixtures
6
+ FIXTURES_DIR = File.expand_path('../fixtures', __dir__)
7
+
8
+ class << self
9
+ # TestFixtures.fixture('backends/openai_chat')
10
+ # -> parsed JSON from test/fixtures/backends/openai_chat.json
11
+ def fixture(name)
12
+ path = fixture_path(name)
13
+ content = Open.read(path)
14
+ JSON.parse(content)
15
+ end
16
+
17
+ def fixture_path(name)
18
+ Path.setup(File.join(FIXTURES_DIR, name.to_s + '.json')).find
19
+ end
20
+ end
21
+ end
@@ -0,0 +1,136 @@
1
+ # Test support: infrastructure probe collector shared by the endpoint suites.
2
+ #
3
+ # Not named test_*.rb on purpose: the Rakefile test pattern must not collect
4
+ # this file.
5
+ #
6
+ # Runs a fixed set of probes against one inference target and reports the
7
+ # outcome (including latency and token usage) as a summary instead of raising:
8
+ #
9
+ # trivial : answer a trivial question
10
+ # tool_call: one weather-style tool call round (a block answers the tool)
11
+ # kb_query : the Miki brother-in-law knowledge base question, a two-hop
12
+ # tool chain (marriages then brothers)
13
+ #
14
+ # Failures are recorded as :fail rows, never raised, so one broken endpoint
15
+ # cannot break the run; the summary table is printed and written to
16
+ # results/infrastructure_summary.md by whichever suite runs last.
17
+ module InfrastructureProbes
18
+ TIMEOUT = 300
19
+
20
+ WEATHER_TOOL = {
21
+ "type": "function",
22
+ "function": {
23
+ "name": "get_current_temperature",
24
+ "description": "Get the current temperature and raining conditions for a specific location",
25
+ "parameters": {
26
+ "type": "object",
27
+ "properties": {
28
+ "location": { "type": "string", "description": "The city and state, e.g., San Francisco, CA" },
29
+ "unit": { "type": "string", "enum": ["Celsius", "Fahrenheit"],
30
+ "description": "The temperature unit to use. Infer this from the user's location." }
31
+ },
32
+ "required": ["location", "unit"]
33
+ }
34
+ }
35
+ }
36
+
37
+ class << self
38
+ attr_accessor :results
39
+ end
40
+ self.results ||= []
41
+
42
+ module_function
43
+
44
+ def record(target, probe, status, reason: nil, answer: nil, duration: nil, tokens: nil)
45
+ InfrastructureProbes.results << {target: target, probe: probe, status: status, reason: reason,
46
+ answer: answer, duration: duration, tokens: tokens}
47
+ end
48
+
49
+ # Run the three probes against `target` (a display name), using
50
+ # `ask_options` (already carrying :endpoint or nothing, i.e. the default).
51
+ # Nothing raises: each probe is rescued and recorded.
52
+ def run_probes(target, ask_options = {})
53
+ nonce = Time.now.to_i
54
+
55
+ run_probe(target, :trivial, ask_options) do
56
+ answer = LLM.ask "user: Reply with the single word OK and nothing else (nonce #{nonce})",
57
+ ask_options.merge(persist: false)
58
+ raise 'no answer' if answer.to_s.strip.empty?
59
+ answer
60
+ end
61
+
62
+ run_probe(target, :tool_call, ask_options) do
63
+ answer = LLM.ask "user: What is the weather in London? Use the provided tool and then answer in one sentence (nonce #{nonce}).",
64
+ ask_options.merge(tools: [WEATHER_TOOL], tool_choice: 'required', persist: false) do |_name, _arguments|
65
+ "It's 15 degrees and raining."
66
+ end
67
+ raise 'no answer' if answer.to_s.strip.empty?
68
+ answer
69
+ end
70
+
71
+ run_probe(target, :kb_query, ask_options) do
72
+ TmpFile.with_dir do |dir|
73
+ kb = KnowledgeBase.new dir
74
+ kb.register :brothers, Test::Unit::TestCase.datafile_test(:person).brothers, undirected: true
75
+ kb.register :marriages, Test::Unit::TestCase.datafile_test(:person).marriages,
76
+ undirected: true, source: "=>Alias", target: "=>Alias"
77
+
78
+ text = LLM.knowledge_base_ask(kb,
79
+ "Who is Miki's brother in law? The brother in law is your spouse's sibling. Use the marriages and brothers tools to find out (nonce #{nonce}).",
80
+ ask_options.merge(persist: false))
81
+ raise "answer did not mention Guille: #{text.to_s[0, 200]}" unless text.to_s.include?('Guille')
82
+ text
83
+ end
84
+ end
85
+ end
86
+
87
+ def run_probe(target, probe, ask_options)
88
+ start = Time.now
89
+ answer = Timeout.timeout(TIMEOUT) { yield }
90
+ record(target, probe, :ok, answer: answer, duration: Time.now - start, tokens: nil)
91
+ rescue Exception => e
92
+ record(target, probe, :fail,
93
+ reason: "#{e.class}: #{e.message.to_s.lines.first}",
94
+ duration: Time.now - start)
95
+ end
96
+
97
+ def summary_lines
98
+ lines = []
99
+ lines << 'Infrastructure endpoint summary'
100
+ lines << '=' * 96
101
+ InfrastructureProbes.results.each do |r|
102
+ line = [r[:target].to_s.ljust(14), r[:probe].to_s.ljust(11)]
103
+ case r[:status]
104
+ when :ok
105
+ line << 'OK'.ljust(5)
106
+ line << (r[:duration] ? "#{r[:duration].round(1)}s" : '').ljust(8)
107
+ line << " #{r[:answer].to_s.gsub(/\s+/, ' ')[0, 100]}"
108
+ when :fail
109
+ line << 'FAIL'.ljust(5)
110
+ line << (r[:duration] ? "#{r[:duration].round(1)}s" : '').ljust(8)
111
+ line << " #{r[:reason].to_s.gsub(/\s+/, ' ')[0, 140]}"
112
+ else
113
+ line << 'OMIT'.ljust(5)
114
+ line << " #{r[:reason].to_s.gsub(/\s+/, ' ')[0, 140]}"
115
+ end
116
+ lines << line * ' | '
117
+ end
118
+ lines << '=' * 96
119
+ lines << "#{InfrastructureProbes.results.count { |r| r[:status] == :ok }} ok, " +
120
+ "#{InfrastructureProbes.results.count { |r| r[:status] == :fail }} failed, " +
121
+ "#{InfrastructureProbes.results.count { |r| r[:status] == :omit }} omitted"
122
+ lines
123
+ end
124
+
125
+ def write_summary
126
+ return if InfrastructureProbes.results.empty?
127
+ content = summary_lines * "\n"
128
+ $stdout.puts "\n" + content
129
+ begin
130
+ Open.mkdir 'results'
131
+ Open.write 'results/infrastructure_summary.md', content + "\n"
132
+ rescue Exception
133
+ $stderr.puts "Could not write results/infrastructure_summary.md: #{$!.message}"
134
+ end
135
+ end
136
+ end