scout-ai 1.2.3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.vimproject +138 -50
- data/README.md +171 -290
- data/Rakefile +17 -1
- data/VERSION +1 -1
- data/doc/Improvements.md +325 -0
- data/doc/StartHere.md +110 -0
- data/doc/developer/Architecture.md +126 -0
- data/doc/developer/Backends.md +199 -0
- data/doc/developer/ChatLifecycle.md +183 -0
- data/doc/developer/DelegationInternals.md +295 -0
- data/doc/developer/DesignPrinciples.md +245 -0
- data/doc/developer/PromptProcessing.md +292 -0
- data/doc/developer/Provenance.md +317 -0
- data/doc/user/BuildingAgents.md +345 -0
- data/doc/user/Cookbook.md +333 -0
- data/doc/user/CoreConcepts.md +181 -0
- data/doc/user/Delegation.md +191 -0
- data/doc/user/GettingStarted.md +159 -0
- data/doc/user/ManagingContext.md +163 -0
- data/doc/user/MultiAgentWorkflows.md +256 -0
- data/doc/user/Python.md +159 -0
- data/doc/user/RunningInference.md +200 -0
- data/doc/user/ToolCalling.md +193 -0
- data/doc/user/WritingChats.md +197 -0
- data/lib/scout/llm/agent/chat.rb +61 -11
- data/lib/scout/llm/agent/delegate.rb +274 -65
- data/lib/scout/llm/agent/iterate.rb +2 -2
- data/lib/scout/llm/agent/save.rb +273 -0
- data/lib/scout/llm/agent/workflow.rb +164 -0
- data/lib/scout/llm/agent.rb +86 -61
- data/lib/scout/llm/ask.rb +62 -17
- data/lib/scout/llm/backends/anthropic.rb +9 -2
- data/lib/scout/llm/backends/bedrock.rb +15 -3
- data/lib/scout/llm/backends/default.rb +183 -99
- data/lib/scout/llm/backends/glm.rb +58 -0
- data/lib/scout/llm/backends/huggingface.rb +196 -26
- data/lib/scout/llm/backends/ollama.rb +13 -1
- data/lib/scout/llm/backends/openai.rb +0 -2
- data/lib/scout/llm/backends/openwebui.rb +20 -13
- data/lib/scout/llm/backends/relay.rb +22 -22
- data/lib/scout/llm/backends/responses.rb +1 -1
- data/lib/scout/llm/chat/agent_meta.rb +264 -0
- data/lib/scout/llm/chat/annotation.rb +39 -10
- data/lib/scout/llm/chat/parse.rb +28 -6
- data/lib/scout/llm/chat/persist.rb +25 -0
- data/lib/scout/llm/chat/process/clear.rb +41 -6
- data/lib/scout/llm/chat/process/files.rb +21 -6
- data/lib/scout/llm/chat/process/meta.rb +421 -34
- data/lib/scout/llm/chat/process/options.rb +21 -1
- data/lib/scout/llm/chat/process/tools.rb +56 -15
- data/lib/scout/llm/chat/process.rb +4 -0
- data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
- data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
- data/lib/scout/llm/chat/prompt.rb +48 -0
- data/lib/scout/llm/chat/provenance.rb +775 -0
- data/lib/scout/llm/chat/tool_calls.rb +76 -0
- data/lib/scout/llm/chat.rb +18 -2
- data/lib/scout/llm/embed.rb +11 -3
- data/lib/scout/llm/image.rb +86 -0
- data/lib/scout/llm/mcp.rb +10 -2
- data/lib/scout/llm/rag.rb +3 -3
- data/lib/scout/llm/tools/call.rb +160 -11
- data/lib/scout/llm/tools/knowledge_base.rb +1 -1
- data/lib/scout/llm/tools/workflow.rb +32 -16
- data/lib/scout/model/python/huggingface/causal.rb +23 -5
- data/lib/scout/model/python/huggingface.rb +2 -1
- data/lib/scout-ai.rb +1 -0
- data/python/README.md +197 -14
- data/python/scout_ai/huggingface/eval.py +245 -34
- data/python/tests/test_huggingface_eval.py +58 -0
- data/research/ChatAnalyst-required-changes.md +167 -0
- data/research/agent-delegation-analysis.md +810 -0
- data/research/agent-meta-provenance-integration-plan.md +622 -0
- data/research/agent-workflow-analysis.md +1120 -0
- data/research/backends-analysis.md +836 -0
- data/research/chat-core-analysis.md +946 -0
- data/research/chatanalyst-provenance/00-baseline.md +30 -0
- data/research/chatanalyst-provenance/01-repo-map.md +60 -0
- data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
- data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
- data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
- data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
- data/research/chatanalyst-provenance/07-critic-review.md +25 -0
- data/research/chatanalyst-provenance/final-report.md +45 -0
- data/research/chatanalyst-provenance/resumption.md +37 -0
- data/research/coding-philosophy-analysis.md +928 -0
- data/research/commands-analysis.md +947 -0
- data/research/multi-agent-patterns-analysis.md +853 -0
- data/research/prompt-strategies-analysis.md +630 -0
- data/research/prov-verbosity-fix-notes.md +77 -0
- data/research/provenance-analysis.md +469 -0
- data/research/provenance-navigation-design.md +640 -0
- data/research/synthesis-report.md +487 -0
- data/research/tools-system-analysis.md +779 -0
- data/scout-ai.gemspec +100 -11
- data/scout_commands/agent/ask +13 -3
- data/scout_commands/agent/kb +2 -0
- data/scout_commands/llm/ask +11 -4
- data/scout_commands/llm/md +76 -0
- data/scout_commands/llm/process_queries +48 -0
- data/scout_commands/llm/prov +602 -0
- data/scout_commands/llm/word +71 -0
- data/scout_commands/workflow/mcp +43 -0
- data/share/word/reference.docx +0 -0
- data/test/etc/AI/mock.yaml +11 -0
- data/test/fixtures/backends/anthropic.json +19 -0
- data/test/fixtures/backends/anthropic_tool_use.json +24 -0
- data/test/fixtures/backends/bedrock.json +8 -0
- data/test/fixtures/backends/bedrock_embedding.json +3 -0
- data/test/fixtures/backends/bedrock_tool_use.json +17 -0
- data/test/fixtures/backends/ollama.json +16 -0
- data/test/fixtures/backends/ollama_tool_call.json +27 -0
- data/test/fixtures/backends/openai_chat.json +21 -0
- data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
- data/test/fixtures/backends/responses.json +33 -0
- data/test/fixtures/backends/responses_tool_call.json +28 -0
- data/test/integration/README.md +32 -0
- data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
- data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
- data/test/integration/scout/llm/backends/test_relay.rb +52 -0
- data/test/integration/scout/llm/test_infrastructure.rb +74 -0
- data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
- data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
- data/test/integration/scout/model/test_base.rb +91 -0
- data/test/scout/llm/agent/test_chat.rb +8 -2
- data/test/scout/llm/agent/test_save.rb +413 -0
- data/test/scout/llm/agent/test_workflow.rb +110 -0
- data/test/scout/llm/backends/test_anthropic.rb +93 -10
- data/test/scout/llm/backends/test_bedrock.rb +118 -2
- data/test/scout/llm/backends/test_huggingface.rb +137 -42
- data/test/scout/llm/backends/test_ollama.rb +70 -20
- data/test/scout/llm/backends/test_openwebui.rb +42 -40
- data/test/scout/llm/backends/test_relay.rb +4 -2
- data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
- data/test/scout/llm/chat/process/test_meta.rb +518 -0
- data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
- data/test/scout/llm/chat/test_agent_meta.rb +357 -0
- data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
- data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
- data/test/scout/llm/chat/test_parse.rb +70 -15
- data/test/scout/llm/chat/test_prov_cli.rb +274 -0
- data/test/scout/llm/chat/test_provenance.rb +240 -0
- data/test/scout/llm/chat/test_tool_calls.rb +38 -0
- data/test/scout/llm/test_agent.rb +13 -36
- data/test/scout/llm/test_ask.rb +75 -52
- data/test/scout/llm/test_chat.rb +107 -13
- data/test/scout/llm/test_embed.rb +48 -0
- data/test/scout/llm/test_rag.rb +23 -16
- data/test/scout/llm/test_tools.rb +12 -1
- data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
- data/test/scout/llm/tools/test_mcp.rb +5 -3
- data/test/scout/llm/tools/test_workflow.rb +23 -2
- data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
- data/test/scout/model/python/huggingface/test_causal.rb +9 -3
- data/test/scout/model/python/huggingface/test_classification.rb +11 -2
- data/test/scout/model/python/test_torch.rb +2 -0
- data/test/scout/model/python/torch/test_helpers.rb +4 -0
- data/test/scout/model/test_base.rb +4 -2
- data/test/support/availability.rb +231 -0
- data/test/support/fake_clients.rb +138 -0
- data/test/support/fixtures.rb +21 -0
- data/test/support/infrastructure_probes.rb +136 -0
- data/test/support/mock_backend.rb +215 -0
- data/test/test_helper.rb +32 -2
- metadata +99 -10
- data/doc/Agent.md +0 -327
- data/doc/Chat.md +0 -458
- data/doc/LLM.md +0 -340
- data/doc/RAG.md +0 -129
- data/scout_commands/documenter +0 -148
- data/test/scout/llm/backends/test_openai.rb +0 -192
- data/test/scout/llm/backends/test_responses.rb +0 -238
- data/test/scout/llm/test_parse.rb +0 -98
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
#!/usr/bin/env ruby
|
|
2
|
+
# frozen_string_literal: true
|
|
3
|
+
|
|
4
|
+
require 'scout'
|
|
5
|
+
|
|
6
|
+
cmd = $previous_commands ?
|
|
7
|
+
"scout #{$previous_commands.any? ? "#{$previous_commands * ' '} " : ''}#{File.basename(__FILE__)}" :
|
|
8
|
+
$PROGRAM_NAME
|
|
9
|
+
|
|
10
|
+
options = SOPT.setup <<~EOF
|
|
11
|
+
|
|
12
|
+
Run a workflow as an MCP service
|
|
13
|
+
|
|
14
|
+
$ #{cmd} [<options>] <workflow> [<task_name>]*
|
|
15
|
+
|
|
16
|
+
You can name the tasks to export. If no tasks are named then it export all
|
|
17
|
+
those that were explicitedly exported by the workflow. If no tasks are defined
|
|
18
|
+
either way then all workflow tasks will be exported.
|
|
19
|
+
|
|
20
|
+
-h--help Print this help
|
|
21
|
+
EOF
|
|
22
|
+
if options[:help]
|
|
23
|
+
if defined? scout_usage
|
|
24
|
+
scout_usage
|
|
25
|
+
else
|
|
26
|
+
puts SOPT.doc
|
|
27
|
+
end
|
|
28
|
+
exit 0
|
|
29
|
+
end
|
|
30
|
+
|
|
31
|
+
help = IndiferentHash.process_options options, :help
|
|
32
|
+
|
|
33
|
+
workflow, *task_names = ARGV
|
|
34
|
+
|
|
35
|
+
require 'scout/llm/mcp'
|
|
36
|
+
|
|
37
|
+
raise MissingParameterException, :workflow if workflow.nil?
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
workflow = Workflow.require_workflow workflow
|
|
41
|
+
|
|
42
|
+
workflow.mcp_stdio(*task_names)
|
|
43
|
+
|
|
Binary file
|
|
@@ -0,0 +1,11 @@
|
|
|
1
|
+
# Offline endpoint used by the unit tests when an endpoint name is convenient:
|
|
2
|
+
# everything resolves to the registered LLM::Mock backend (see
|
|
3
|
+
# test/support/mock_backend.rb), so `rake test` never reaches a real
|
|
4
|
+
# inference service.
|
|
5
|
+
#
|
|
6
|
+
# ScoutCoder: this is the ONLY endpoint defined inside the repo. The 'test'
|
|
7
|
+
# endpoint is expected to be provided at installation level
|
|
8
|
+
# (~/.scout/etc/AI/test.yaml) and is consumed exclusively by the
|
|
9
|
+
# infrastructure suite (`rake test_infrastructure`).
|
|
10
|
+
backend: mock
|
|
11
|
+
model: mock-model
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "msg_mock_anthropic_001",
|
|
3
|
+
"type": "message",
|
|
4
|
+
"role": "assistant",
|
|
5
|
+
"model": "claude-sonnet-4-5",
|
|
6
|
+
"stop_reason": "end_turn",
|
|
7
|
+
"content": [
|
|
8
|
+
{
|
|
9
|
+
"type": "text",
|
|
10
|
+
"text": "Mock answer from Anthropic messages"
|
|
11
|
+
}
|
|
12
|
+
],
|
|
13
|
+
"usage": {
|
|
14
|
+
"input_tokens": 15,
|
|
15
|
+
"output_tokens": 9,
|
|
16
|
+
"cache_read_input_tokens": 0,
|
|
17
|
+
"cache_creation_input_tokens": 0
|
|
18
|
+
}
|
|
19
|
+
}
|
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "msg_mock_anthropic_002",
|
|
3
|
+
"type": "message",
|
|
4
|
+
"role": "assistant",
|
|
5
|
+
"model": "claude-sonnet-4-5",
|
|
6
|
+
"stop_reason": "tool_use",
|
|
7
|
+
"content": [
|
|
8
|
+
{
|
|
9
|
+
"type": "tool_use",
|
|
10
|
+
"id": "toolu_mock_1",
|
|
11
|
+
"name": "get_current_temperature",
|
|
12
|
+
"input": {
|
|
13
|
+
"location": "London",
|
|
14
|
+
"unit": "Celsius"
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
],
|
|
18
|
+
"usage": {
|
|
19
|
+
"input_tokens": 30,
|
|
20
|
+
"output_tokens": 14,
|
|
21
|
+
"cache_read_input_tokens": 2,
|
|
22
|
+
"cache_creation_input_tokens": 1
|
|
23
|
+
}
|
|
24
|
+
}
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"model": "llama3.1",
|
|
4
|
+
"created_at": "2025-01-01T00:00:00Z",
|
|
5
|
+
"message": {
|
|
6
|
+
"role": "assistant",
|
|
7
|
+
"content": "Mock answer from Ollama"
|
|
8
|
+
},
|
|
9
|
+
"done_reason": "stop",
|
|
10
|
+
"done": true,
|
|
11
|
+
"usage": {
|
|
12
|
+
"prompt_tokens": 12,
|
|
13
|
+
"completion_tokens": 7
|
|
14
|
+
}
|
|
15
|
+
}
|
|
16
|
+
]
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
[
|
|
2
|
+
{
|
|
3
|
+
"model": "llama3.1",
|
|
4
|
+
"created_at": "2025-01-01T00:00:01Z",
|
|
5
|
+
"message": {
|
|
6
|
+
"role": "assistant",
|
|
7
|
+
"content": "",
|
|
8
|
+
"tool_calls": [
|
|
9
|
+
{
|
|
10
|
+
"function": {
|
|
11
|
+
"name": "get_current_temperature",
|
|
12
|
+
"arguments": {
|
|
13
|
+
"location": "London",
|
|
14
|
+
"unit": "Celsius"
|
|
15
|
+
}
|
|
16
|
+
}
|
|
17
|
+
}
|
|
18
|
+
]
|
|
19
|
+
},
|
|
20
|
+
"done_reason": "stop",
|
|
21
|
+
"done": true,
|
|
22
|
+
"usage": {
|
|
23
|
+
"prompt_tokens": 14,
|
|
24
|
+
"completion_tokens": 9
|
|
25
|
+
}
|
|
26
|
+
}
|
|
27
|
+
]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "chatcmpl-mock-001",
|
|
3
|
+
"object": "chat.completion",
|
|
4
|
+
"created": 1700000000,
|
|
5
|
+
"model": "gpt-5-nano",
|
|
6
|
+
"choices": [
|
|
7
|
+
{
|
|
8
|
+
"index": 0,
|
|
9
|
+
"message": {
|
|
10
|
+
"role": "assistant",
|
|
11
|
+
"content": "Mock answer from Chat Completions"
|
|
12
|
+
},
|
|
13
|
+
"finish_reason": "stop"
|
|
14
|
+
}
|
|
15
|
+
],
|
|
16
|
+
"usage": {
|
|
17
|
+
"prompt_tokens": 11,
|
|
18
|
+
"completion_tokens": 7,
|
|
19
|
+
"total_tokens": 18
|
|
20
|
+
}
|
|
21
|
+
}
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "chatcmpl-mock-002",
|
|
3
|
+
"object": "chat.completion",
|
|
4
|
+
"created": 1700000001,
|
|
5
|
+
"model": "gpt-5-nano",
|
|
6
|
+
"choices": [
|
|
7
|
+
{
|
|
8
|
+
"index": 0,
|
|
9
|
+
"message": {
|
|
10
|
+
"role": "assistant",
|
|
11
|
+
"content": null,
|
|
12
|
+
"tool_calls": [
|
|
13
|
+
{
|
|
14
|
+
"id": "call_mock_1",
|
|
15
|
+
"type": "function",
|
|
16
|
+
"function": {
|
|
17
|
+
"name": "get_current_temperature",
|
|
18
|
+
"arguments": "{\"location\":\"London\",\"unit\":\"Celsius\"}"
|
|
19
|
+
}
|
|
20
|
+
}
|
|
21
|
+
]
|
|
22
|
+
},
|
|
23
|
+
"finish_reason": "tool_calls"
|
|
24
|
+
}
|
|
25
|
+
],
|
|
26
|
+
"usage": {
|
|
27
|
+
"prompt_tokens": 20,
|
|
28
|
+
"completion_tokens": 13,
|
|
29
|
+
"total_tokens": 33
|
|
30
|
+
}
|
|
31
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "resp_mock_001",
|
|
3
|
+
"object": "response",
|
|
4
|
+
"created_at": 1700000000,
|
|
5
|
+
"model": "gpt-5-nano",
|
|
6
|
+
"status": "completed",
|
|
7
|
+
"output": [
|
|
8
|
+
{
|
|
9
|
+
"type": "message",
|
|
10
|
+
"id": "msg_mock_001",
|
|
11
|
+
"role": "assistant",
|
|
12
|
+
"status": "completed",
|
|
13
|
+
"content": [
|
|
14
|
+
{
|
|
15
|
+
"type": "output_text",
|
|
16
|
+
"text": "Mock answer from the Responses API",
|
|
17
|
+
"annotations": []
|
|
18
|
+
}
|
|
19
|
+
]
|
|
20
|
+
}
|
|
21
|
+
],
|
|
22
|
+
"usage": {
|
|
23
|
+
"input_tokens": 12,
|
|
24
|
+
"output_tokens": 8,
|
|
25
|
+
"total_tokens": 20,
|
|
26
|
+
"input_tokens_details": {
|
|
27
|
+
"cached_tokens": 4
|
|
28
|
+
},
|
|
29
|
+
"output_tokens_details": {
|
|
30
|
+
"reasoning_tokens": 0
|
|
31
|
+
}
|
|
32
|
+
}
|
|
33
|
+
}
|
|
@@ -0,0 +1,28 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "resp_mock_002",
|
|
3
|
+
"object": "response",
|
|
4
|
+
"created_at": 1700000001,
|
|
5
|
+
"model": "gpt-5-nano",
|
|
6
|
+
"status": "completed",
|
|
7
|
+
"output": [
|
|
8
|
+
{
|
|
9
|
+
"type": "function_call",
|
|
10
|
+
"id": "fc_mock_1",
|
|
11
|
+
"call_id": "call_mock_1",
|
|
12
|
+
"name": "get_current_temperature",
|
|
13
|
+
"arguments": "{\"location\":\"London\",\"unit\":\"Celsius\"}",
|
|
14
|
+
"status": "completed"
|
|
15
|
+
}
|
|
16
|
+
],
|
|
17
|
+
"usage": {
|
|
18
|
+
"input_tokens": 24,
|
|
19
|
+
"output_tokens": 16,
|
|
20
|
+
"total_tokens": 40,
|
|
21
|
+
"input_tokens_details": {
|
|
22
|
+
"cached_tokens": 6
|
|
23
|
+
},
|
|
24
|
+
"output_tokens_details": {
|
|
25
|
+
"reasoning_tokens": 2
|
|
26
|
+
}
|
|
27
|
+
}
|
|
28
|
+
}
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
# Infrastructure (real infrastructure) tests
|
|
2
|
+
|
|
3
|
+
Collected by `rake test_infrastructure` only; the default `rake test` task
|
|
4
|
+
explicitly excludes test/integration/**/test_*.rb and must never attempt any
|
|
5
|
+
inference.
|
|
6
|
+
|
|
7
|
+
## Endpoint targets
|
|
8
|
+
|
|
9
|
+
- `test/scout/llm/test_infrastructure.rb` - general suite. Uses the endpoint
|
|
10
|
+
named `test` when it is defined at installation level
|
|
11
|
+
(~/.scout/etc/AI/test.yaml and friends); when it is not defined it
|
|
12
|
+
specifies no endpoint at all and the installation default applies. Omits
|
|
13
|
+
with a detected reason when neither can serve an ask.
|
|
14
|
+
|
|
15
|
+
- `test/scout/llm/backends/test_endpoints.rb` - per-backend suite. For each
|
|
16
|
+
backend (openai, responses, anthropic, ollama, bedrock, openwebui, relay)
|
|
17
|
+
it looks for an endpoint with the same name; missing endpoints omit with
|
|
18
|
+
the detected reason, configured ones run the three probes and report
|
|
19
|
+
latency/tokens in the shared summary.
|
|
20
|
+
|
|
21
|
+
The repo deliberately ships only the offline `mock` endpoint
|
|
22
|
+
(test/etc/AI/mock.yaml) so that `mock` is always available offline while the
|
|
23
|
+
`test` and backend endpoints remain the user's own, installation-level
|
|
24
|
+
configuration. Neither `rake test` nor `rake test_infrastructure` defines
|
|
25
|
+
them repo-side.
|
|
26
|
+
|
|
27
|
+
## Probe reporting
|
|
28
|
+
|
|
29
|
+
Probes never raise: outcomes are collected by
|
|
30
|
+
`test/support/infrastructure_probes.rb`, printed at the end of the run, and
|
|
31
|
+
written to `results/infrastructure_summary.md` as a table of
|
|
32
|
+
target | probe | OK/FAIL/OMIT | latency | answer/reason.
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/ask'
|
|
4
|
+
require 'scout/knowledge_base'
|
|
5
|
+
require File.expand_path(File.join(File.dirname(__FILE__), '..', '..', '..', '..', 'support', 'infrastructure_probes'))
|
|
6
|
+
|
|
7
|
+
# Per-backend infrastructure suites. Each backend looks for an endpoint with
|
|
8
|
+
# the same name as itself (openai, responses, anthropic, ollama, bedrock,
|
|
9
|
+
# openwebui, relay). If the endpoint is not defined in the Scout AI paths,
|
|
10
|
+
# the suite omits with the detected reason; if it is defined, it runs the
|
|
11
|
+
# three probes (trivial ask, weather tool call, KnowledgeBase query) and
|
|
12
|
+
# reports latency/token usage in the shared summary instead of raising.
|
|
13
|
+
class TestLLMEndpointsByBackend < Test::Unit::TestCase
|
|
14
|
+
include InfrastructureProbes
|
|
15
|
+
|
|
16
|
+
BACKENDS = %w[openai responses anthropic ollama bedrock openwebui relay].freeze
|
|
17
|
+
|
|
18
|
+
def test_backend_endpoints
|
|
19
|
+
BACKENDS.each do |backend|
|
|
20
|
+
reason = Availability.endpoint_reason(backend)
|
|
21
|
+
if reason
|
|
22
|
+
InfrastructureProbes.record(backend, :all, :omit, reason: reason)
|
|
23
|
+
next
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
# The backend's own implementation must match the endpoint; when the
|
|
27
|
+
# endpoint yaml points elsewhere the probes still use the endpoint as
|
|
28
|
+
# configured, the backend name is only the target label.
|
|
29
|
+
run_probes(backend, endpoint: backend)
|
|
30
|
+
end
|
|
31
|
+
|
|
32
|
+
InfrastructureProbes.write_summary
|
|
33
|
+
end
|
|
34
|
+
end
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/backends/openwebui'
|
|
4
|
+
|
|
5
|
+
# Infrastructure suite for the OpenWebUI backend.
|
|
6
|
+
#
|
|
7
|
+
# Per-backend endpoint probes (same-named endpoint) live in test_endpoints.rb;
|
|
8
|
+
# this file keeps the historically separate direct-URL check against the
|
|
9
|
+
# gepeto service, which is not configured through an endpoint yaml.
|
|
10
|
+
#
|
|
11
|
+
# Conditional omission: only runs when a key is configured for the service
|
|
12
|
+
# (config/env) AND a trivial authenticated request succeeds; otherwise the
|
|
13
|
+
# test omits with the detected reason.
|
|
14
|
+
|
|
15
|
+
class TestOpenWebUI < Test::Unit::TestCase
|
|
16
|
+
URL = 'https://gepeto.bsc.es/api'
|
|
17
|
+
|
|
18
|
+
def test_gepeto
|
|
19
|
+
key = key_configured
|
|
20
|
+
omit "no key configured for #{URL} (set the openwebui key in your Scout config)" if key.nil?
|
|
21
|
+
|
|
22
|
+
reason = unavailable_reason
|
|
23
|
+
omit reason if reason
|
|
24
|
+
|
|
25
|
+
Log.severity = 0
|
|
26
|
+
prompt =<<-EOF
|
|
27
|
+
user: write a script that sorts files in a directory
|
|
28
|
+
EOF
|
|
29
|
+
|
|
30
|
+
res = LLM::OpenWebUI.ask prompt, model: 'qwen3-vl:30b', url: URL, key: key
|
|
31
|
+
|
|
32
|
+
assert res.is_a?(String) && !res.strip.empty?, 'OpenWebUI returned no answer'
|
|
33
|
+
end
|
|
34
|
+
|
|
35
|
+
private
|
|
36
|
+
|
|
37
|
+
def key_configured
|
|
38
|
+
key = Scout::Config.get(:key, :openwebui, :llm, env: 'OPENWEBUI_API_KEY', default: nil)
|
|
39
|
+
key.to_s.strip.empty? ? nil : key
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# A minimal real request (not just TCP) so 401/404/timeout are detected as
|
|
43
|
+
# unavailability instead of raising as errors mid-test.
|
|
44
|
+
def unavailable_reason
|
|
45
|
+
key = key_configured
|
|
46
|
+
headers = { 'Authorization' => "Bearer #{key}", 'Content-Type' => 'application/json' }
|
|
47
|
+
payload = { model: 'qwen3-vl:30b', messages: [{role: 'user', content: 'Reply OK'}] }.to_json
|
|
48
|
+
Timeout.timeout(30) { RestClient.post(File.join(URL, 'chat/completions'), payload, headers) }
|
|
49
|
+
nil
|
|
50
|
+
rescue Timeout::Error
|
|
51
|
+
"request to #{URL} timed out"
|
|
52
|
+
rescue RestClient::Unauthorized
|
|
53
|
+
"key for #{URL} rejected (401)"
|
|
54
|
+
rescue RestClient::NotFound
|
|
55
|
+
nil # endpoint reachable and authenticated; model listing is not required
|
|
56
|
+
rescue RestClient::Exception => e
|
|
57
|
+
"#{URL} not usable: #{e.class}: #{e.message}"
|
|
58
|
+
rescue Errno::ECONNREFUSED, SocketError, Errno::EHOSTUNREACH => e
|
|
59
|
+
"#{URL} unreachable: #{e.class}"
|
|
60
|
+
end
|
|
61
|
+
end
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/backends/relay'
|
|
4
|
+
|
|
5
|
+
# Integration (real infrastructure) copy of test/scout/llm/backends/test_relay.rb:
|
|
6
|
+
# LLM::Relay shells out to `scp` against a reachable server, so there is no
|
|
7
|
+
# client seam to fake. Kept out of the default unit suite; run with
|
|
8
|
+
# `rake test_integration`.
|
|
9
|
+
#
|
|
10
|
+
# Conditional omission: only runs when a relay server is configured,
|
|
11
|
+
# SSH-reachable AND a non-interactive `ssh true` succeeds (port 22 being open
|
|
12
|
+
# is not enough: without working keys the scp fails mid-test with an error
|
|
13
|
+
# instead of a clean omission).
|
|
14
|
+
|
|
15
|
+
class TestRelay < Test::Unit::TestCase
|
|
16
|
+
SERVER = ENV['SCOUT_TEST_RELAY_SERVER'] || Scout::Config.get(:server, :relay, default: 'localhost')
|
|
17
|
+
|
|
18
|
+
def test_ask
|
|
19
|
+
reason = unavailable_reason
|
|
20
|
+
omit "relay server #{SERVER} not usable (#{reason}; set SCOUT_TEST_RELAY_SERVER or the relay server config)" if reason
|
|
21
|
+
|
|
22
|
+
Scout::Config.set(:server, SERVER, :relay)
|
|
23
|
+
res = LLM::Relay.ask 'Say hi', model: 'gemma2'
|
|
24
|
+
|
|
25
|
+
assert res.is_a?(String) && !res.strip.empty?, 'relay returned no answer'
|
|
26
|
+
end
|
|
27
|
+
|
|
28
|
+
private
|
|
29
|
+
|
|
30
|
+
def unavailable_reason
|
|
31
|
+
server = SERVER.to_s
|
|
32
|
+
return 'no server configured' if server.empty?
|
|
33
|
+
|
|
34
|
+
host = server.split(':').first
|
|
35
|
+
return 'no host in server setting' if host.nil? || host.empty?
|
|
36
|
+
|
|
37
|
+
require 'socket'
|
|
38
|
+
begin
|
|
39
|
+
Timeout.timeout(5) { TCPSocket.open(host, 22) { |s| s.close } }
|
|
40
|
+
rescue Exception => e
|
|
41
|
+
return "ssh port unreachable (#{e.class})"
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
# ScoutCoder: use backticks (not Open.run, which does not exist) for the
|
|
45
|
+
# non-interactive ssh probe; BatchMode fails fast instead of hanging on a
|
|
46
|
+
# password prompt, so the probe itself cannot block the test run.
|
|
47
|
+
`ssh -o BatchMode=yes -o ConnectTimeout=5 -o StrictHostKeyChecking=accept-new #{host} true > /dev/null 2>&1`
|
|
48
|
+
return nil if $?.success?
|
|
49
|
+
|
|
50
|
+
"ssh login failed (exit #{$?.exitstatus})"
|
|
51
|
+
end
|
|
52
|
+
end
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/ask'
|
|
4
|
+
require 'scout/knowledge_base'
|
|
5
|
+
require File.expand_path(File.join(File.dirname(__FILE__), '..', '..', '..', 'support', 'infrastructure_probes'))
|
|
6
|
+
|
|
7
|
+
# General infrastructure suite.
|
|
8
|
+
#
|
|
9
|
+
# Target resolution (in order):
|
|
10
|
+
# 1. an endpoint named 'test' defined at installation level
|
|
11
|
+
# (~/.scout/etc/AI/test.yaml and friends). The repo deliberately does
|
|
12
|
+
# not define one (only the offline 'mock' lives in test/etc/AI), so
|
|
13
|
+
# this one is always the user's own.
|
|
14
|
+
# 2. if 'test' is not defined, no endpoint is specified at all and the
|
|
15
|
+
# installation default (config default backend/endpoint) is used.
|
|
16
|
+
#
|
|
17
|
+
# The suite omits with a detected reason when neither can work, and reports
|
|
18
|
+
# probe outcomes (latency, tokens) in the summary instead of raising; see
|
|
19
|
+
# test/support/infrastructure_probes.rb.
|
|
20
|
+
class TestInfrastructureGeneral < Test::Unit::TestCase
|
|
21
|
+
|
|
22
|
+
# ScoutCoder: test_helper configures backend :mock for the unit suite; the
|
|
23
|
+
# infrastructure suite must NOT inherit it (it would silently run the mock
|
|
24
|
+
# as if it were real infrastructure). Scout::Config::CACHE is a plain hash
|
|
25
|
+
# of set() entries, so dropping the unit entry at load time restores
|
|
26
|
+
# whatever the installation actually configured.
|
|
27
|
+
UNIT_MOCK_CACHE_KEYS = Scout::Config::CACHE.keys.select { |k| k.to_s == 'backend' }.dup
|
|
28
|
+
UNIT_MOCK_CACHE_KEYS.each { |k| Scout::Config::CACHE.delete(k) }
|
|
29
|
+
|
|
30
|
+
TARGET = if Availability.endpoint_configured?('test')
|
|
31
|
+
:test_endpoint
|
|
32
|
+
else
|
|
33
|
+
:default
|
|
34
|
+
end
|
|
35
|
+
|
|
36
|
+
def test_general_trivial_question
|
|
37
|
+
reason = target_reason
|
|
38
|
+
InfrastructureProbes.record(target_name, :all, :omit, reason: reason) if reason
|
|
39
|
+
omit reason if reason
|
|
40
|
+
|
|
41
|
+
# the :test_endpoint case uses the installation-level endpoint named
|
|
42
|
+
# 'test'; the :default case specifies nothing and lets the installation
|
|
43
|
+
# default backend/endpoint apply
|
|
44
|
+
ask_options = TARGET == :test_endpoint ? {endpoint: 'test'} : {}
|
|
45
|
+
InfrastructureProbes.run_probes(target_name, ask_options)
|
|
46
|
+
|
|
47
|
+
# ScoutCoder: test-unit's class-level `shutdown` hook does not fire under
|
|
48
|
+
# this setup (neither direct file runs nor rake's test loader), and an
|
|
49
|
+
# at_exit hook registered at load time runs BEFORE the autorunner. Writing
|
|
50
|
+
# the summary at the end of this test method is the reliable spot; the
|
|
51
|
+
# per-backend suite appends its rows afterwards.
|
|
52
|
+
InfrastructureProbes.write_summary
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
private
|
|
56
|
+
|
|
57
|
+
def target_name
|
|
58
|
+
TARGET == :test_endpoint ? 'test' : '(default)'
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
# The default target is usable only when the installation actually
|
|
62
|
+
# configured something to talk to; an empty configuration cannot serve any
|
|
63
|
+
# ask, so we omit with the reason instead of failing.
|
|
64
|
+
def target_reason
|
|
65
|
+
if TARGET == :test_endpoint
|
|
66
|
+
Availability.endpoint_reason('test')
|
|
67
|
+
else
|
|
68
|
+
backend = Scout::Config.get(:backend, :ask, :llm, default: nil)
|
|
69
|
+
endpoint = Scout::Config.get(:endpoint, :ask, :llm, default: nil)
|
|
70
|
+
return "no 'test' endpoint defined at installation level and no default backend/endpoint configured" if backend.nil? && endpoint.nil?
|
|
71
|
+
nil
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
end
|
|
@@ -3,7 +3,7 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
|
|
|
3
3
|
|
|
4
4
|
require "scout-ai"
|
|
5
5
|
class TestMCP < Test::Unit::TestCase
|
|
6
|
-
def
|
|
6
|
+
def _test_workflow_stdio
|
|
7
7
|
require "mcp/server/transports/stdio_transport"
|
|
8
8
|
wf = Module.new do
|
|
9
9
|
extend Workflow
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
|
+
|
|
3
|
+
require 'scout/llm/tools/mcp'
|
|
4
|
+
|
|
5
|
+
# Integration (real infrastructure) copy of test/scout/llm/tools/test_mcp.rb:
|
|
6
|
+
# test_client connects to https://api.githubcopilot.com/mcp/, which requires
|
|
7
|
+
# a GitHub token; the stdio MCP server coverage stays in the unit suite
|
|
8
|
+
# (test/scout/llm/test_mcp.rb). Run with `rake test_integration`.
|
|
9
|
+
#
|
|
10
|
+
# Conditional omission: only runs when a GitHub Copilot token is configured
|
|
11
|
+
# and the MCP endpoint accepts the connection; the host itself is reachable
|
|
12
|
+
# for anyone but answers 400 without credentials, which used to surface as
|
|
13
|
+
# an error instead of a skip.
|
|
14
|
+
|
|
15
|
+
class TestClass < Test::Unit::TestCase
|
|
16
|
+
URL = "https://api.githubcopilot.com/mcp/"
|
|
17
|
+
|
|
18
|
+
def test_client
|
|
19
|
+
key = ENV['GITHUB_COPILOT_TOKEN'] || ENV['GITHUB_TOKEN'] ||
|
|
20
|
+
Scout::Config.get(:key, :github, :copilot, default: nil)
|
|
21
|
+
omit "no GitHub Copilot token configured (set GITHUB_COPILOT_TOKEN)" if key.to_s.strip.empty?
|
|
22
|
+
|
|
23
|
+
omit "#{URL} unreachable" unless host_reachable?
|
|
24
|
+
|
|
25
|
+
c = LLM.mcp_tools(URL, key: key)
|
|
26
|
+
assert_include c.keys, "get_me"
|
|
27
|
+
end
|
|
28
|
+
|
|
29
|
+
private
|
|
30
|
+
|
|
31
|
+
def host_reachable?
|
|
32
|
+
require 'uri'
|
|
33
|
+
require 'socket'
|
|
34
|
+
uri = URI.parse(URL)
|
|
35
|
+
begin
|
|
36
|
+
Timeout.timeout(5) { TCPSocket.open(uri.host, uri.port) { |s| s.close } }
|
|
37
|
+
true
|
|
38
|
+
rescue Exception
|
|
39
|
+
false
|
|
40
|
+
end
|
|
41
|
+
end
|
|
42
|
+
end
|