scout-ai 1.2.3 → 2.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. checksums.yaml +4 -4
  2. data/.vimproject +138 -50
  3. data/README.md +171 -290
  4. data/Rakefile +17 -1
  5. data/VERSION +1 -1
  6. data/doc/Improvements.md +325 -0
  7. data/doc/StartHere.md +110 -0
  8. data/doc/developer/Architecture.md +126 -0
  9. data/doc/developer/Backends.md +199 -0
  10. data/doc/developer/ChatLifecycle.md +183 -0
  11. data/doc/developer/DelegationInternals.md +295 -0
  12. data/doc/developer/DesignPrinciples.md +245 -0
  13. data/doc/developer/PromptProcessing.md +292 -0
  14. data/doc/developer/Provenance.md +317 -0
  15. data/doc/user/BuildingAgents.md +345 -0
  16. data/doc/user/Cookbook.md +333 -0
  17. data/doc/user/CoreConcepts.md +181 -0
  18. data/doc/user/Delegation.md +191 -0
  19. data/doc/user/GettingStarted.md +159 -0
  20. data/doc/user/ManagingContext.md +163 -0
  21. data/doc/user/MultiAgentWorkflows.md +256 -0
  22. data/doc/user/Python.md +159 -0
  23. data/doc/user/RunningInference.md +200 -0
  24. data/doc/user/ToolCalling.md +193 -0
  25. data/doc/user/WritingChats.md +197 -0
  26. data/lib/scout/llm/agent/chat.rb +61 -11
  27. data/lib/scout/llm/agent/delegate.rb +274 -65
  28. data/lib/scout/llm/agent/iterate.rb +2 -2
  29. data/lib/scout/llm/agent/save.rb +273 -0
  30. data/lib/scout/llm/agent/workflow.rb +164 -0
  31. data/lib/scout/llm/agent.rb +86 -61
  32. data/lib/scout/llm/ask.rb +62 -17
  33. data/lib/scout/llm/backends/anthropic.rb +9 -2
  34. data/lib/scout/llm/backends/bedrock.rb +15 -3
  35. data/lib/scout/llm/backends/default.rb +183 -99
  36. data/lib/scout/llm/backends/glm.rb +58 -0
  37. data/lib/scout/llm/backends/huggingface.rb +196 -26
  38. data/lib/scout/llm/backends/ollama.rb +13 -1
  39. data/lib/scout/llm/backends/openai.rb +0 -2
  40. data/lib/scout/llm/backends/openwebui.rb +20 -13
  41. data/lib/scout/llm/backends/relay.rb +22 -22
  42. data/lib/scout/llm/backends/responses.rb +1 -1
  43. data/lib/scout/llm/chat/agent_meta.rb +264 -0
  44. data/lib/scout/llm/chat/annotation.rb +39 -10
  45. data/lib/scout/llm/chat/parse.rb +28 -6
  46. data/lib/scout/llm/chat/persist.rb +25 -0
  47. data/lib/scout/llm/chat/process/clear.rb +41 -6
  48. data/lib/scout/llm/chat/process/files.rb +21 -6
  49. data/lib/scout/llm/chat/process/meta.rb +421 -34
  50. data/lib/scout/llm/chat/process/options.rb +21 -1
  51. data/lib/scout/llm/chat/process/tools.rb +56 -15
  52. data/lib/scout/llm/chat/process.rb +4 -0
  53. data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
  54. data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
  55. data/lib/scout/llm/chat/prompt.rb +48 -0
  56. data/lib/scout/llm/chat/provenance.rb +775 -0
  57. data/lib/scout/llm/chat/tool_calls.rb +76 -0
  58. data/lib/scout/llm/chat.rb +18 -2
  59. data/lib/scout/llm/embed.rb +11 -3
  60. data/lib/scout/llm/image.rb +86 -0
  61. data/lib/scout/llm/mcp.rb +10 -2
  62. data/lib/scout/llm/rag.rb +3 -3
  63. data/lib/scout/llm/tools/call.rb +160 -11
  64. data/lib/scout/llm/tools/knowledge_base.rb +1 -1
  65. data/lib/scout/llm/tools/workflow.rb +32 -16
  66. data/lib/scout/model/python/huggingface/causal.rb +23 -5
  67. data/lib/scout/model/python/huggingface.rb +2 -1
  68. data/lib/scout-ai.rb +1 -0
  69. data/python/README.md +197 -14
  70. data/python/scout_ai/huggingface/eval.py +245 -34
  71. data/python/tests/test_huggingface_eval.py +58 -0
  72. data/research/ChatAnalyst-required-changes.md +167 -0
  73. data/research/agent-delegation-analysis.md +810 -0
  74. data/research/agent-meta-provenance-integration-plan.md +622 -0
  75. data/research/agent-workflow-analysis.md +1120 -0
  76. data/research/backends-analysis.md +836 -0
  77. data/research/chat-core-analysis.md +946 -0
  78. data/research/chatanalyst-provenance/00-baseline.md +30 -0
  79. data/research/chatanalyst-provenance/01-repo-map.md +60 -0
  80. data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
  81. data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
  82. data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
  83. data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
  84. data/research/chatanalyst-provenance/07-critic-review.md +25 -0
  85. data/research/chatanalyst-provenance/final-report.md +45 -0
  86. data/research/chatanalyst-provenance/resumption.md +37 -0
  87. data/research/coding-philosophy-analysis.md +928 -0
  88. data/research/commands-analysis.md +947 -0
  89. data/research/multi-agent-patterns-analysis.md +853 -0
  90. data/research/prompt-strategies-analysis.md +630 -0
  91. data/research/prov-verbosity-fix-notes.md +77 -0
  92. data/research/provenance-analysis.md +469 -0
  93. data/research/provenance-navigation-design.md +640 -0
  94. data/research/synthesis-report.md +487 -0
  95. data/research/tools-system-analysis.md +779 -0
  96. data/scout-ai.gemspec +100 -11
  97. data/scout_commands/agent/ask +13 -3
  98. data/scout_commands/agent/kb +2 -0
  99. data/scout_commands/llm/ask +11 -4
  100. data/scout_commands/llm/md +76 -0
  101. data/scout_commands/llm/process_queries +48 -0
  102. data/scout_commands/llm/prov +602 -0
  103. data/scout_commands/llm/word +71 -0
  104. data/scout_commands/workflow/mcp +43 -0
  105. data/share/word/reference.docx +0 -0
  106. data/test/etc/AI/mock.yaml +11 -0
  107. data/test/fixtures/backends/anthropic.json +19 -0
  108. data/test/fixtures/backends/anthropic_tool_use.json +24 -0
  109. data/test/fixtures/backends/bedrock.json +8 -0
  110. data/test/fixtures/backends/bedrock_embedding.json +3 -0
  111. data/test/fixtures/backends/bedrock_tool_use.json +17 -0
  112. data/test/fixtures/backends/ollama.json +16 -0
  113. data/test/fixtures/backends/ollama_tool_call.json +27 -0
  114. data/test/fixtures/backends/openai_chat.json +21 -0
  115. data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
  116. data/test/fixtures/backends/responses.json +33 -0
  117. data/test/fixtures/backends/responses_tool_call.json +28 -0
  118. data/test/integration/README.md +32 -0
  119. data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
  120. data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
  121. data/test/integration/scout/llm/backends/test_relay.rb +52 -0
  122. data/test/integration/scout/llm/test_infrastructure.rb +74 -0
  123. data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
  124. data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
  125. data/test/integration/scout/model/test_base.rb +91 -0
  126. data/test/scout/llm/agent/test_chat.rb +8 -2
  127. data/test/scout/llm/agent/test_save.rb +413 -0
  128. data/test/scout/llm/agent/test_workflow.rb +110 -0
  129. data/test/scout/llm/backends/test_anthropic.rb +93 -10
  130. data/test/scout/llm/backends/test_bedrock.rb +118 -2
  131. data/test/scout/llm/backends/test_huggingface.rb +137 -42
  132. data/test/scout/llm/backends/test_ollama.rb +70 -20
  133. data/test/scout/llm/backends/test_openwebui.rb +42 -40
  134. data/test/scout/llm/backends/test_relay.rb +4 -2
  135. data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
  136. data/test/scout/llm/chat/process/test_meta.rb +518 -0
  137. data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
  138. data/test/scout/llm/chat/test_agent_meta.rb +357 -0
  139. data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
  140. data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
  141. data/test/scout/llm/chat/test_parse.rb +70 -15
  142. data/test/scout/llm/chat/test_prov_cli.rb +274 -0
  143. data/test/scout/llm/chat/test_provenance.rb +240 -0
  144. data/test/scout/llm/chat/test_tool_calls.rb +38 -0
  145. data/test/scout/llm/test_agent.rb +13 -36
  146. data/test/scout/llm/test_ask.rb +75 -52
  147. data/test/scout/llm/test_chat.rb +107 -13
  148. data/test/scout/llm/test_embed.rb +48 -0
  149. data/test/scout/llm/test_rag.rb +23 -16
  150. data/test/scout/llm/test_tools.rb +12 -1
  151. data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
  152. data/test/scout/llm/tools/test_mcp.rb +5 -3
  153. data/test/scout/llm/tools/test_workflow.rb +23 -2
  154. data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
  155. data/test/scout/model/python/huggingface/test_causal.rb +9 -3
  156. data/test/scout/model/python/huggingface/test_classification.rb +11 -2
  157. data/test/scout/model/python/test_torch.rb +2 -0
  158. data/test/scout/model/python/torch/test_helpers.rb +4 -0
  159. data/test/scout/model/test_base.rb +4 -2
  160. data/test/support/availability.rb +231 -0
  161. data/test/support/fake_clients.rb +138 -0
  162. data/test/support/fixtures.rb +21 -0
  163. data/test/support/infrastructure_probes.rb +136 -0
  164. data/test/support/mock_backend.rb +215 -0
  165. data/test/test_helper.rb +32 -2
  166. metadata +99 -10
  167. data/doc/Agent.md +0 -327
  168. data/doc/Chat.md +0 -458
  169. data/doc/LLM.md +0 -340
  170. data/doc/RAG.md +0 -129
  171. data/scout_commands/documenter +0 -148
  172. data/test/scout/llm/backends/test_openai.rb +0 -192
  173. data/test/scout/llm/backends/test_responses.rb +0 -238
  174. data/test/scout/llm/test_parse.rb +0 -98
@@ -0,0 +1,43 @@
1
+ #!/usr/bin/env ruby
2
+ # frozen_string_literal: true
3
+
4
+ require 'scout'
5
+
6
+ cmd = $previous_commands ?
7
+ "scout #{$previous_commands.any? ? "#{$previous_commands * ' '} " : ''}#{File.basename(__FILE__)}" :
8
+ $PROGRAM_NAME
9
+
10
+ options = SOPT.setup <<~EOF
11
+
12
+ Run a workflow as an MCP service
13
+
14
+ $ #{cmd} [<options>] <workflow> [<task_name>]*
15
+
16
+ You can name the tasks to export. If no tasks are named then it export all
17
+ those that were explicitedly exported by the workflow. If no tasks are defined
18
+ either way then all workflow tasks will be exported.
19
+
20
+ -h--help Print this help
21
+ EOF
22
+ if options[:help]
23
+ if defined? scout_usage
24
+ scout_usage
25
+ else
26
+ puts SOPT.doc
27
+ end
28
+ exit 0
29
+ end
30
+
31
+ help = IndiferentHash.process_options options, :help
32
+
33
+ workflow, *task_names = ARGV
34
+
35
+ require 'scout/llm/mcp'
36
+
37
+ raise MissingParameterException, :workflow if workflow.nil?
38
+
39
+
40
+ workflow = Workflow.require_workflow workflow
41
+
42
+ workflow.mcp_stdio(*task_names)
43
+
Binary file
@@ -0,0 +1,11 @@
1
+ # Offline endpoint used by the unit tests when an endpoint name is convenient:
2
+ # everything resolves to the registered LLM::Mock backend (see
3
+ # test/support/mock_backend.rb), so `rake test` never reaches a real
4
+ # inference service.
5
+ #
6
+ # ScoutCoder: this is the ONLY endpoint defined inside the repo. The 'test'
7
+ # endpoint is expected to be provided at installation level
8
+ # (~/.scout/etc/AI/test.yaml) and is consumed exclusively by the
9
+ # infrastructure suite (`rake test_infrastructure`).
10
+ backend: mock
11
+ model: mock-model
@@ -0,0 +1,19 @@
1
+ {
2
+ "id": "msg_mock_anthropic_001",
3
+ "type": "message",
4
+ "role": "assistant",
5
+ "model": "claude-sonnet-4-5",
6
+ "stop_reason": "end_turn",
7
+ "content": [
8
+ {
9
+ "type": "text",
10
+ "text": "Mock answer from Anthropic messages"
11
+ }
12
+ ],
13
+ "usage": {
14
+ "input_tokens": 15,
15
+ "output_tokens": 9,
16
+ "cache_read_input_tokens": 0,
17
+ "cache_creation_input_tokens": 0
18
+ }
19
+ }
@@ -0,0 +1,24 @@
1
+ {
2
+ "id": "msg_mock_anthropic_002",
3
+ "type": "message",
4
+ "role": "assistant",
5
+ "model": "claude-sonnet-4-5",
6
+ "stop_reason": "tool_use",
7
+ "content": [
8
+ {
9
+ "type": "tool_use",
10
+ "id": "toolu_mock_1",
11
+ "name": "get_current_temperature",
12
+ "input": {
13
+ "location": "London",
14
+ "unit": "Celsius"
15
+ }
16
+ }
17
+ ],
18
+ "usage": {
19
+ "input_tokens": 30,
20
+ "output_tokens": 14,
21
+ "cache_read_input_tokens": 2,
22
+ "cache_creation_input_tokens": 1
23
+ }
24
+ }
@@ -0,0 +1,8 @@
1
+ {
2
+ "content": [
3
+ {
4
+ "type": "text",
5
+ "text": "Mock answer from Bedrock"
6
+ }
7
+ ]
8
+ }
@@ -0,0 +1,3 @@
1
+ {
2
+ "embedding": [0.1, 0.2, 0.3]
3
+ }
@@ -0,0 +1,17 @@
1
+ {
2
+ "content": [
3
+ {
4
+ "type": "tool_use",
5
+ "tool_calls": [
6
+ {
7
+ "id": "call_1",
8
+ "type": "function",
9
+ "function": {
10
+ "name": "get_current_temperature",
11
+ "arguments": {"location": "London", "unit": "Celsius"}
12
+ }
13
+ }
14
+ ]
15
+ }
16
+ ]
17
+ }
@@ -0,0 +1,16 @@
1
+ [
2
+ {
3
+ "model": "llama3.1",
4
+ "created_at": "2025-01-01T00:00:00Z",
5
+ "message": {
6
+ "role": "assistant",
7
+ "content": "Mock answer from Ollama"
8
+ },
9
+ "done_reason": "stop",
10
+ "done": true,
11
+ "usage": {
12
+ "prompt_tokens": 12,
13
+ "completion_tokens": 7
14
+ }
15
+ }
16
+ ]
@@ -0,0 +1,27 @@
1
+ [
2
+ {
3
+ "model": "llama3.1",
4
+ "created_at": "2025-01-01T00:00:01Z",
5
+ "message": {
6
+ "role": "assistant",
7
+ "content": "",
8
+ "tool_calls": [
9
+ {
10
+ "function": {
11
+ "name": "get_current_temperature",
12
+ "arguments": {
13
+ "location": "London",
14
+ "unit": "Celsius"
15
+ }
16
+ }
17
+ }
18
+ ]
19
+ },
20
+ "done_reason": "stop",
21
+ "done": true,
22
+ "usage": {
23
+ "prompt_tokens": 14,
24
+ "completion_tokens": 9
25
+ }
26
+ }
27
+ ]
@@ -0,0 +1,21 @@
1
+ {
2
+ "id": "chatcmpl-mock-001",
3
+ "object": "chat.completion",
4
+ "created": 1700000000,
5
+ "model": "gpt-5-nano",
6
+ "choices": [
7
+ {
8
+ "index": 0,
9
+ "message": {
10
+ "role": "assistant",
11
+ "content": "Mock answer from Chat Completions"
12
+ },
13
+ "finish_reason": "stop"
14
+ }
15
+ ],
16
+ "usage": {
17
+ "prompt_tokens": 11,
18
+ "completion_tokens": 7,
19
+ "total_tokens": 18
20
+ }
21
+ }
@@ -0,0 +1,31 @@
1
+ {
2
+ "id": "chatcmpl-mock-002",
3
+ "object": "chat.completion",
4
+ "created": 1700000001,
5
+ "model": "gpt-5-nano",
6
+ "choices": [
7
+ {
8
+ "index": 0,
9
+ "message": {
10
+ "role": "assistant",
11
+ "content": null,
12
+ "tool_calls": [
13
+ {
14
+ "id": "call_mock_1",
15
+ "type": "function",
16
+ "function": {
17
+ "name": "get_current_temperature",
18
+ "arguments": "{\"location\":\"London\",\"unit\":\"Celsius\"}"
19
+ }
20
+ }
21
+ ]
22
+ },
23
+ "finish_reason": "tool_calls"
24
+ }
25
+ ],
26
+ "usage": {
27
+ "prompt_tokens": 20,
28
+ "completion_tokens": 13,
29
+ "total_tokens": 33
30
+ }
31
+ }
@@ -0,0 +1,33 @@
1
+ {
2
+ "id": "resp_mock_001",
3
+ "object": "response",
4
+ "created_at": 1700000000,
5
+ "model": "gpt-5-nano",
6
+ "status": "completed",
7
+ "output": [
8
+ {
9
+ "type": "message",
10
+ "id": "msg_mock_001",
11
+ "role": "assistant",
12
+ "status": "completed",
13
+ "content": [
14
+ {
15
+ "type": "output_text",
16
+ "text": "Mock answer from the Responses API",
17
+ "annotations": []
18
+ }
19
+ ]
20
+ }
21
+ ],
22
+ "usage": {
23
+ "input_tokens": 12,
24
+ "output_tokens": 8,
25
+ "total_tokens": 20,
26
+ "input_tokens_details": {
27
+ "cached_tokens": 4
28
+ },
29
+ "output_tokens_details": {
30
+ "reasoning_tokens": 0
31
+ }
32
+ }
33
+ }
@@ -0,0 +1,28 @@
1
+ {
2
+ "id": "resp_mock_002",
3
+ "object": "response",
4
+ "created_at": 1700000001,
5
+ "model": "gpt-5-nano",
6
+ "status": "completed",
7
+ "output": [
8
+ {
9
+ "type": "function_call",
10
+ "id": "fc_mock_1",
11
+ "call_id": "call_mock_1",
12
+ "name": "get_current_temperature",
13
+ "arguments": "{\"location\":\"London\",\"unit\":\"Celsius\"}",
14
+ "status": "completed"
15
+ }
16
+ ],
17
+ "usage": {
18
+ "input_tokens": 24,
19
+ "output_tokens": 16,
20
+ "total_tokens": 40,
21
+ "input_tokens_details": {
22
+ "cached_tokens": 6
23
+ },
24
+ "output_tokens_details": {
25
+ "reasoning_tokens": 2
26
+ }
27
+ }
28
+ }
@@ -0,0 +1,32 @@
1
+ # Infrastructure (real infrastructure) tests
2
+
3
+ Collected by `rake test_infrastructure` only; the default `rake test` task
4
+ explicitly excludes test/integration/**/test_*.rb and must never attempt any
5
+ inference.
6
+
7
+ ## Endpoint targets
8
+
9
+ - `test/scout/llm/test_infrastructure.rb` - general suite. Uses the endpoint
10
+ named `test` when it is defined at installation level
11
+ (~/.scout/etc/AI/test.yaml and friends); when it is not defined it
12
+ specifies no endpoint at all and the installation default applies. Omits
13
+ with a detected reason when neither can serve an ask.
14
+
15
+ - `test/scout/llm/backends/test_endpoints.rb` - per-backend suite. For each
16
+ backend (openai, responses, anthropic, ollama, bedrock, openwebui, relay)
17
+ it looks for an endpoint with the same name; missing endpoints omit with
18
+ the detected reason, configured ones run the three probes and report
19
+ latency/tokens in the shared summary.
20
+
21
+ The repo deliberately ships only the offline `mock` endpoint
22
+ (test/etc/AI/mock.yaml) so that `mock` is always available offline while the
23
+ `test` and backend endpoints remain the user's own, installation-level
24
+ configuration. Neither `rake test` nor `rake test_infrastructure` defines
25
+ them repo-side.
26
+
27
+ ## Probe reporting
28
+
29
+ Probes never raise: outcomes are collected by
30
+ `test/support/infrastructure_probes.rb`, printed at the end of the run, and
31
+ written to `results/infrastructure_summary.md` as a table of
32
+ target | probe | OK/FAIL/OMIT | latency | answer/reason.
@@ -0,0 +1,34 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/ask'
4
+ require 'scout/knowledge_base'
5
+ require File.expand_path(File.join(File.dirname(__FILE__), '..', '..', '..', '..', 'support', 'infrastructure_probes'))
6
+
7
+ # Per-backend infrastructure suites. Each backend looks for an endpoint with
8
+ # the same name as itself (openai, responses, anthropic, ollama, bedrock,
9
+ # openwebui, relay). If the endpoint is not defined in the Scout AI paths,
10
+ # the suite omits with the detected reason; if it is defined, it runs the
11
+ # three probes (trivial ask, weather tool call, KnowledgeBase query) and
12
+ # reports latency/token usage in the shared summary instead of raising.
13
+ class TestLLMEndpointsByBackend < Test::Unit::TestCase
14
+ include InfrastructureProbes
15
+
16
+ BACKENDS = %w[openai responses anthropic ollama bedrock openwebui relay].freeze
17
+
18
+ def test_backend_endpoints
19
+ BACKENDS.each do |backend|
20
+ reason = Availability.endpoint_reason(backend)
21
+ if reason
22
+ InfrastructureProbes.record(backend, :all, :omit, reason: reason)
23
+ next
24
+ end
25
+
26
+ # The backend's own implementation must match the endpoint; when the
27
+ # endpoint yaml points elsewhere the probes still use the endpoint as
28
+ # configured, the backend name is only the target label.
29
+ run_probes(backend, endpoint: backend)
30
+ end
31
+
32
+ InfrastructureProbes.write_summary
33
+ end
34
+ end
@@ -0,0 +1,61 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/backends/openwebui'
4
+
5
+ # Infrastructure suite for the OpenWebUI backend.
6
+ #
7
+ # Per-backend endpoint probes (same-named endpoint) live in test_endpoints.rb;
8
+ # this file keeps the historically separate direct-URL check against the
9
+ # gepeto service, which is not configured through an endpoint yaml.
10
+ #
11
+ # Conditional omission: only runs when a key is configured for the service
12
+ # (config/env) AND a trivial authenticated request succeeds; otherwise the
13
+ # test omits with the detected reason.
14
+
15
+ class TestOpenWebUI < Test::Unit::TestCase
16
+ URL = 'https://gepeto.bsc.es/api'
17
+
18
+ def test_gepeto
19
+ key = key_configured
20
+ omit "no key configured for #{URL} (set the openwebui key in your Scout config)" if key.nil?
21
+
22
+ reason = unavailable_reason
23
+ omit reason if reason
24
+
25
+ Log.severity = 0
26
+ prompt =<<-EOF
27
+ user: write a script that sorts files in a directory
28
+ EOF
29
+
30
+ res = LLM::OpenWebUI.ask prompt, model: 'qwen3-vl:30b', url: URL, key: key
31
+
32
+ assert res.is_a?(String) && !res.strip.empty?, 'OpenWebUI returned no answer'
33
+ end
34
+
35
+ private
36
+
37
+ def key_configured
38
+ key = Scout::Config.get(:key, :openwebui, :llm, env: 'OPENWEBUI_API_KEY', default: nil)
39
+ key.to_s.strip.empty? ? nil : key
40
+ end
41
+
42
+ # A minimal real request (not just TCP) so 401/404/timeout are detected as
43
+ # unavailability instead of raising as errors mid-test.
44
+ def unavailable_reason
45
+ key = key_configured
46
+ headers = { 'Authorization' => "Bearer #{key}", 'Content-Type' => 'application/json' }
47
+ payload = { model: 'qwen3-vl:30b', messages: [{role: 'user', content: 'Reply OK'}] }.to_json
48
+ Timeout.timeout(30) { RestClient.post(File.join(URL, 'chat/completions'), payload, headers) }
49
+ nil
50
+ rescue Timeout::Error
51
+ "request to #{URL} timed out"
52
+ rescue RestClient::Unauthorized
53
+ "key for #{URL} rejected (401)"
54
+ rescue RestClient::NotFound
55
+ nil # endpoint reachable and authenticated; model listing is not required
56
+ rescue RestClient::Exception => e
57
+ "#{URL} not usable: #{e.class}: #{e.message}"
58
+ rescue Errno::ECONNREFUSED, SocketError, Errno::EHOSTUNREACH => e
59
+ "#{URL} unreachable: #{e.class}"
60
+ end
61
+ end
@@ -0,0 +1,52 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/backends/relay'
4
+
5
+ # Integration (real infrastructure) copy of test/scout/llm/backends/test_relay.rb:
6
+ # LLM::Relay shells out to `scp` against a reachable server, so there is no
7
+ # client seam to fake. Kept out of the default unit suite; run with
8
+ # `rake test_integration`.
9
+ #
10
+ # Conditional omission: only runs when a relay server is configured,
11
+ # SSH-reachable AND a non-interactive `ssh true` succeeds (port 22 being open
12
+ # is not enough: without working keys the scp fails mid-test with an error
13
+ # instead of a clean omission).
14
+
15
+ class TestRelay < Test::Unit::TestCase
16
+ SERVER = ENV['SCOUT_TEST_RELAY_SERVER'] || Scout::Config.get(:server, :relay, default: 'localhost')
17
+
18
+ def test_ask
19
+ reason = unavailable_reason
20
+ omit "relay server #{SERVER} not usable (#{reason}; set SCOUT_TEST_RELAY_SERVER or the relay server config)" if reason
21
+
22
+ Scout::Config.set(:server, SERVER, :relay)
23
+ res = LLM::Relay.ask 'Say hi', model: 'gemma2'
24
+
25
+ assert res.is_a?(String) && !res.strip.empty?, 'relay returned no answer'
26
+ end
27
+
28
+ private
29
+
30
+ def unavailable_reason
31
+ server = SERVER.to_s
32
+ return 'no server configured' if server.empty?
33
+
34
+ host = server.split(':').first
35
+ return 'no host in server setting' if host.nil? || host.empty?
36
+
37
+ require 'socket'
38
+ begin
39
+ Timeout.timeout(5) { TCPSocket.open(host, 22) { |s| s.close } }
40
+ rescue Exception => e
41
+ return "ssh port unreachable (#{e.class})"
42
+ end
43
+
44
+ # ScoutCoder: use backticks (not Open.run, which does not exist) for the
45
+ # non-interactive ssh probe; BatchMode fails fast instead of hanging on a
46
+ # password prompt, so the probe itself cannot block the test run.
47
+ `ssh -o BatchMode=yes -o ConnectTimeout=5 -o StrictHostKeyChecking=accept-new #{host} true > /dev/null 2>&1`
48
+ return nil if $?.success?
49
+
50
+ "ssh login failed (exit #{$?.exitstatus})"
51
+ end
52
+ end
@@ -0,0 +1,74 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/ask'
4
+ require 'scout/knowledge_base'
5
+ require File.expand_path(File.join(File.dirname(__FILE__), '..', '..', '..', 'support', 'infrastructure_probes'))
6
+
7
+ # General infrastructure suite.
8
+ #
9
+ # Target resolution (in order):
10
+ # 1. an endpoint named 'test' defined at installation level
11
+ # (~/.scout/etc/AI/test.yaml and friends). The repo deliberately does
12
+ # not define one (only the offline 'mock' lives in test/etc/AI), so
13
+ # this one is always the user's own.
14
+ # 2. if 'test' is not defined, no endpoint is specified at all and the
15
+ # installation default (config default backend/endpoint) is used.
16
+ #
17
+ # The suite omits with a detected reason when neither can work, and reports
18
+ # probe outcomes (latency, tokens) in the summary instead of raising; see
19
+ # test/support/infrastructure_probes.rb.
20
+ class TestInfrastructureGeneral < Test::Unit::TestCase
21
+
22
+ # ScoutCoder: test_helper configures backend :mock for the unit suite; the
23
+ # infrastructure suite must NOT inherit it (it would silently run the mock
24
+ # as if it were real infrastructure). Scout::Config::CACHE is a plain hash
25
+ # of set() entries, so dropping the unit entry at load time restores
26
+ # whatever the installation actually configured.
27
+ UNIT_MOCK_CACHE_KEYS = Scout::Config::CACHE.keys.select { |k| k.to_s == 'backend' }.dup
28
+ UNIT_MOCK_CACHE_KEYS.each { |k| Scout::Config::CACHE.delete(k) }
29
+
30
+ TARGET = if Availability.endpoint_configured?('test')
31
+ :test_endpoint
32
+ else
33
+ :default
34
+ end
35
+
36
+ def test_general_trivial_question
37
+ reason = target_reason
38
+ InfrastructureProbes.record(target_name, :all, :omit, reason: reason) if reason
39
+ omit reason if reason
40
+
41
+ # the :test_endpoint case uses the installation-level endpoint named
42
+ # 'test'; the :default case specifies nothing and lets the installation
43
+ # default backend/endpoint apply
44
+ ask_options = TARGET == :test_endpoint ? {endpoint: 'test'} : {}
45
+ InfrastructureProbes.run_probes(target_name, ask_options)
46
+
47
+ # ScoutCoder: test-unit's class-level `shutdown` hook does not fire under
48
+ # this setup (neither direct file runs nor rake's test loader), and an
49
+ # at_exit hook registered at load time runs BEFORE the autorunner. Writing
50
+ # the summary at the end of this test method is the reliable spot; the
51
+ # per-backend suite appends its rows afterwards.
52
+ InfrastructureProbes.write_summary
53
+ end
54
+
55
+ private
56
+
57
+ def target_name
58
+ TARGET == :test_endpoint ? 'test' : '(default)'
59
+ end
60
+
61
+ # The default target is usable only when the installation actually
62
+ # configured something to talk to; an empty configuration cannot serve any
63
+ # ask, so we omit with the reason instead of failing.
64
+ def target_reason
65
+ if TARGET == :test_endpoint
66
+ Availability.endpoint_reason('test')
67
+ else
68
+ backend = Scout::Config.get(:backend, :ask, :llm, default: nil)
69
+ endpoint = Scout::Config.get(:endpoint, :ask, :llm, default: nil)
70
+ return "no 'test' endpoint defined at installation level and no default backend/endpoint configured" if backend.nil? && endpoint.nil?
71
+ nil
72
+ end
73
+ end
74
+ end
@@ -3,7 +3,7 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
3
3
 
4
4
  require "scout-ai"
5
5
  class TestMCP < Test::Unit::TestCase
6
- def test_workflow_stdio
6
+ def _test_workflow_stdio
7
7
  require "mcp/server/transports/stdio_transport"
8
8
  wf = Module.new do
9
9
  extend Workflow
@@ -0,0 +1,42 @@
1
+ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
2
+
3
+ require 'scout/llm/tools/mcp'
4
+
5
+ # Integration (real infrastructure) copy of test/scout/llm/tools/test_mcp.rb:
6
+ # test_client connects to https://api.githubcopilot.com/mcp/, which requires
7
+ # a GitHub token; the stdio MCP server coverage stays in the unit suite
8
+ # (test/scout/llm/test_mcp.rb). Run with `rake test_integration`.
9
+ #
10
+ # Conditional omission: only runs when a GitHub Copilot token is configured
11
+ # and the MCP endpoint accepts the connection; the host itself is reachable
12
+ # for anyone but answers 400 without credentials, which used to surface as
13
+ # an error instead of a skip.
14
+
15
+ class TestClass < Test::Unit::TestCase
16
+ URL = "https://api.githubcopilot.com/mcp/"
17
+
18
+ def test_client
19
+ key = ENV['GITHUB_COPILOT_TOKEN'] || ENV['GITHUB_TOKEN'] ||
20
+ Scout::Config.get(:key, :github, :copilot, default: nil)
21
+ omit "no GitHub Copilot token configured (set GITHUB_COPILOT_TOKEN)" if key.to_s.strip.empty?
22
+
23
+ omit "#{URL} unreachable" unless host_reachable?
24
+
25
+ c = LLM.mcp_tools(URL, key: key)
26
+ assert_include c.keys, "get_me"
27
+ end
28
+
29
+ private
30
+
31
+ def host_reachable?
32
+ require 'uri'
33
+ require 'socket'
34
+ uri = URI.parse(URL)
35
+ begin
36
+ Timeout.timeout(5) { TCPSocket.open(uri.host, uri.port) { |s| s.close } }
37
+ true
38
+ rescue Exception
39
+ false
40
+ end
41
+ end
42
+ end