scout-ai 1.2.3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.vimproject +138 -50
- data/README.md +171 -290
- data/Rakefile +17 -1
- data/VERSION +1 -1
- data/doc/Improvements.md +325 -0
- data/doc/StartHere.md +110 -0
- data/doc/developer/Architecture.md +126 -0
- data/doc/developer/Backends.md +199 -0
- data/doc/developer/ChatLifecycle.md +183 -0
- data/doc/developer/DelegationInternals.md +295 -0
- data/doc/developer/DesignPrinciples.md +245 -0
- data/doc/developer/PromptProcessing.md +292 -0
- data/doc/developer/Provenance.md +317 -0
- data/doc/user/BuildingAgents.md +345 -0
- data/doc/user/Cookbook.md +333 -0
- data/doc/user/CoreConcepts.md +181 -0
- data/doc/user/Delegation.md +191 -0
- data/doc/user/GettingStarted.md +159 -0
- data/doc/user/ManagingContext.md +163 -0
- data/doc/user/MultiAgentWorkflows.md +256 -0
- data/doc/user/Python.md +159 -0
- data/doc/user/RunningInference.md +200 -0
- data/doc/user/ToolCalling.md +193 -0
- data/doc/user/WritingChats.md +197 -0
- data/lib/scout/llm/agent/chat.rb +61 -11
- data/lib/scout/llm/agent/delegate.rb +274 -65
- data/lib/scout/llm/agent/iterate.rb +2 -2
- data/lib/scout/llm/agent/save.rb +273 -0
- data/lib/scout/llm/agent/workflow.rb +164 -0
- data/lib/scout/llm/agent.rb +86 -61
- data/lib/scout/llm/ask.rb +62 -17
- data/lib/scout/llm/backends/anthropic.rb +9 -2
- data/lib/scout/llm/backends/bedrock.rb +15 -3
- data/lib/scout/llm/backends/default.rb +183 -99
- data/lib/scout/llm/backends/glm.rb +58 -0
- data/lib/scout/llm/backends/huggingface.rb +196 -26
- data/lib/scout/llm/backends/ollama.rb +13 -1
- data/lib/scout/llm/backends/openai.rb +0 -2
- data/lib/scout/llm/backends/openwebui.rb +20 -13
- data/lib/scout/llm/backends/relay.rb +22 -22
- data/lib/scout/llm/backends/responses.rb +1 -1
- data/lib/scout/llm/chat/agent_meta.rb +264 -0
- data/lib/scout/llm/chat/annotation.rb +39 -10
- data/lib/scout/llm/chat/parse.rb +28 -6
- data/lib/scout/llm/chat/persist.rb +25 -0
- data/lib/scout/llm/chat/process/clear.rb +41 -6
- data/lib/scout/llm/chat/process/files.rb +21 -6
- data/lib/scout/llm/chat/process/meta.rb +421 -34
- data/lib/scout/llm/chat/process/options.rb +21 -1
- data/lib/scout/llm/chat/process/tools.rb +56 -15
- data/lib/scout/llm/chat/process.rb +4 -0
- data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
- data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
- data/lib/scout/llm/chat/prompt.rb +48 -0
- data/lib/scout/llm/chat/provenance.rb +775 -0
- data/lib/scout/llm/chat/tool_calls.rb +76 -0
- data/lib/scout/llm/chat.rb +18 -2
- data/lib/scout/llm/embed.rb +11 -3
- data/lib/scout/llm/image.rb +86 -0
- data/lib/scout/llm/mcp.rb +10 -2
- data/lib/scout/llm/rag.rb +3 -3
- data/lib/scout/llm/tools/call.rb +160 -11
- data/lib/scout/llm/tools/knowledge_base.rb +1 -1
- data/lib/scout/llm/tools/workflow.rb +32 -16
- data/lib/scout/model/python/huggingface/causal.rb +23 -5
- data/lib/scout/model/python/huggingface.rb +2 -1
- data/lib/scout-ai.rb +1 -0
- data/python/README.md +197 -14
- data/python/scout_ai/huggingface/eval.py +245 -34
- data/python/tests/test_huggingface_eval.py +58 -0
- data/research/ChatAnalyst-required-changes.md +167 -0
- data/research/agent-delegation-analysis.md +810 -0
- data/research/agent-meta-provenance-integration-plan.md +622 -0
- data/research/agent-workflow-analysis.md +1120 -0
- data/research/backends-analysis.md +836 -0
- data/research/chat-core-analysis.md +946 -0
- data/research/chatanalyst-provenance/00-baseline.md +30 -0
- data/research/chatanalyst-provenance/01-repo-map.md +60 -0
- data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
- data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
- data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
- data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
- data/research/chatanalyst-provenance/07-critic-review.md +25 -0
- data/research/chatanalyst-provenance/final-report.md +45 -0
- data/research/chatanalyst-provenance/resumption.md +37 -0
- data/research/coding-philosophy-analysis.md +928 -0
- data/research/commands-analysis.md +947 -0
- data/research/multi-agent-patterns-analysis.md +853 -0
- data/research/prompt-strategies-analysis.md +630 -0
- data/research/prov-verbosity-fix-notes.md +77 -0
- data/research/provenance-analysis.md +469 -0
- data/research/provenance-navigation-design.md +640 -0
- data/research/synthesis-report.md +487 -0
- data/research/tools-system-analysis.md +779 -0
- data/scout-ai.gemspec +100 -11
- data/scout_commands/agent/ask +13 -3
- data/scout_commands/agent/kb +2 -0
- data/scout_commands/llm/ask +11 -4
- data/scout_commands/llm/md +76 -0
- data/scout_commands/llm/process_queries +48 -0
- data/scout_commands/llm/prov +602 -0
- data/scout_commands/llm/word +71 -0
- data/scout_commands/workflow/mcp +43 -0
- data/share/word/reference.docx +0 -0
- data/test/etc/AI/mock.yaml +11 -0
- data/test/fixtures/backends/anthropic.json +19 -0
- data/test/fixtures/backends/anthropic_tool_use.json +24 -0
- data/test/fixtures/backends/bedrock.json +8 -0
- data/test/fixtures/backends/bedrock_embedding.json +3 -0
- data/test/fixtures/backends/bedrock_tool_use.json +17 -0
- data/test/fixtures/backends/ollama.json +16 -0
- data/test/fixtures/backends/ollama_tool_call.json +27 -0
- data/test/fixtures/backends/openai_chat.json +21 -0
- data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
- data/test/fixtures/backends/responses.json +33 -0
- data/test/fixtures/backends/responses_tool_call.json +28 -0
- data/test/integration/README.md +32 -0
- data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
- data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
- data/test/integration/scout/llm/backends/test_relay.rb +52 -0
- data/test/integration/scout/llm/test_infrastructure.rb +74 -0
- data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
- data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
- data/test/integration/scout/model/test_base.rb +91 -0
- data/test/scout/llm/agent/test_chat.rb +8 -2
- data/test/scout/llm/agent/test_save.rb +413 -0
- data/test/scout/llm/agent/test_workflow.rb +110 -0
- data/test/scout/llm/backends/test_anthropic.rb +93 -10
- data/test/scout/llm/backends/test_bedrock.rb +118 -2
- data/test/scout/llm/backends/test_huggingface.rb +137 -42
- data/test/scout/llm/backends/test_ollama.rb +70 -20
- data/test/scout/llm/backends/test_openwebui.rb +42 -40
- data/test/scout/llm/backends/test_relay.rb +4 -2
- data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
- data/test/scout/llm/chat/process/test_meta.rb +518 -0
- data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
- data/test/scout/llm/chat/test_agent_meta.rb +357 -0
- data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
- data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
- data/test/scout/llm/chat/test_parse.rb +70 -15
- data/test/scout/llm/chat/test_prov_cli.rb +274 -0
- data/test/scout/llm/chat/test_provenance.rb +240 -0
- data/test/scout/llm/chat/test_tool_calls.rb +38 -0
- data/test/scout/llm/test_agent.rb +13 -36
- data/test/scout/llm/test_ask.rb +75 -52
- data/test/scout/llm/test_chat.rb +107 -13
- data/test/scout/llm/test_embed.rb +48 -0
- data/test/scout/llm/test_rag.rb +23 -16
- data/test/scout/llm/test_tools.rb +12 -1
- data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
- data/test/scout/llm/tools/test_mcp.rb +5 -3
- data/test/scout/llm/tools/test_workflow.rb +23 -2
- data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
- data/test/scout/model/python/huggingface/test_causal.rb +9 -3
- data/test/scout/model/python/huggingface/test_classification.rb +11 -2
- data/test/scout/model/python/test_torch.rb +2 -0
- data/test/scout/model/python/torch/test_helpers.rb +4 -0
- data/test/scout/model/test_base.rb +4 -2
- data/test/support/availability.rb +231 -0
- data/test/support/fake_clients.rb +138 -0
- data/test/support/fixtures.rb +21 -0
- data/test/support/infrastructure_probes.rb +136 -0
- data/test/support/mock_backend.rb +215 -0
- data/test/test_helper.rb +32 -2
- metadata +99 -10
- data/doc/Agent.md +0 -327
- data/doc/Chat.md +0 -458
- data/doc/LLM.md +0 -340
- data/doc/RAG.md +0 -129
- data/scout_commands/documenter +0 -148
- data/test/scout/llm/backends/test_openai.rb +0 -192
- data/test/scout/llm/backends/test_responses.rb +0 -238
- data/test/scout/llm/test_parse.rb +0 -98
|
@@ -2,9 +2,15 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
4
|
class TestClass < Test::Unit::TestCase
|
|
5
|
+
MODEL = 'mistralai/Mistral-7B-Instruct-v0.3'
|
|
6
|
+
|
|
7
|
+
# Conditional omission: runs only when the model is already in the local
|
|
8
|
+
# huggingface cache. The probe never downloads (see
|
|
9
|
+
# test/support/availability.rb).
|
|
5
10
|
def test_eval_chat
|
|
11
|
+
omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
|
|
6
12
|
#model = CausalModel.new 'BSC-LT/salamandra-2b-instruct'
|
|
7
|
-
model = CausalModel.new
|
|
13
|
+
model = CausalModel.new MODEL
|
|
8
14
|
|
|
9
15
|
model.init
|
|
10
16
|
|
|
@@ -17,8 +23,9 @@ class TestClass < Test::Unit::TestCase
|
|
|
17
23
|
end
|
|
18
24
|
|
|
19
25
|
def test_eval_train
|
|
26
|
+
omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
|
|
20
27
|
#model = CausalModel.new 'BSC-LT/salamandra-2b-instruct'
|
|
21
|
-
model = CausalModel.new
|
|
28
|
+
model = CausalModel.new MODEL
|
|
22
29
|
|
|
23
30
|
model.init
|
|
24
31
|
|
|
@@ -30,4 +37,3 @@ class TestClass < Test::Unit::TestCase
|
|
|
30
37
|
])
|
|
31
38
|
end
|
|
32
39
|
end
|
|
33
|
-
|
|
@@ -1,17 +1,26 @@
|
|
|
1
1
|
require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
|
+
MODEL = 'bert-base-uncased'
|
|
5
|
+
|
|
4
6
|
class TestSequenceClassification < Test::Unit::TestCase
|
|
5
7
|
def _test_eval_sequence_classification
|
|
6
|
-
model = SequenceClassificationModel.new
|
|
8
|
+
model = SequenceClassificationModel.new MODEL, nil,
|
|
7
9
|
class_labels: %w(Bad Good)
|
|
8
10
|
|
|
9
11
|
assert_include ["Bad", "Good"], model.eval("This is dog")
|
|
10
12
|
assert_include ["Bad", "Good"], model.eval_list(["This is dog", "This is cat"]).first
|
|
11
13
|
end
|
|
12
14
|
|
|
15
|
+
# Conditional omission: the probe only checks the local huggingface cache
|
|
16
|
+
# (never downloads) and that transformers/datasets are importable, so this
|
|
17
|
+
# runs whenever the environment can actually serve it.
|
|
13
18
|
def test_train_sequence_classification
|
|
14
|
-
|
|
19
|
+
reason = Availability.python_modules_reason('transformers', 'datasets')
|
|
20
|
+
omit "python infrastructure missing: #{reason}" if reason
|
|
21
|
+
omit "huggingface model #{MODEL}: #{Availability.hf_model_reason(MODEL)}" unless Availability.hf_model_cached?(MODEL)
|
|
22
|
+
|
|
23
|
+
model = SequenceClassificationModel.new MODEL, nil,
|
|
15
24
|
class_labels: %w(Bad Good)
|
|
16
25
|
|
|
17
26
|
model.init
|
|
@@ -3,6 +3,8 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
|
|
|
3
3
|
|
|
4
4
|
class TestTorch < Test::Unit::TestCase
|
|
5
5
|
def test_linear
|
|
6
|
+
omit "No python environment" unless Availability.python?
|
|
7
|
+
omit "Torch not installed" unless Availability.python_modules?(:torch)
|
|
6
8
|
model = nil
|
|
7
9
|
|
|
8
10
|
TmpFile.with_dir do |dir|
|
|
@@ -3,7 +3,11 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
|
|
|
3
3
|
|
|
4
4
|
require 'scout/model/python/base'
|
|
5
5
|
class TestTorchHelpers < Test::Unit::TestCase
|
|
6
|
+
# Conditional omission: the CUDA probe imports torch in a bounded
|
|
7
|
+
# subprocess and checks torch.cuda.is_available(), so this runs whenever a
|
|
8
|
+
# GPU-backed torch is actually present.
|
|
6
9
|
def test_del
|
|
10
|
+
omit 'torch CUDA unavailable (probe: torch.cuda.is_available)' unless Availability.cuda?
|
|
7
11
|
ScoutPython.init_scout
|
|
8
12
|
ScoutPython.pyimport :torch
|
|
9
13
|
batch = [[100.0]]
|
|
@@ -34,7 +34,10 @@ class TestClass < Test::Unit::TestCase
|
|
|
34
34
|
assert_equal [4, 8], model.eval_list([1, 2])
|
|
35
35
|
end
|
|
36
36
|
|
|
37
|
-
|
|
37
|
+
# Real R + e1071 version moved to test/integration/scout/model/test_base.rb
|
|
38
|
+
# (installs the R package online); the trivial ScoutModel tests above stay
|
|
39
|
+
# unit-side.
|
|
40
|
+
def _test_R_model
|
|
38
41
|
require 'rbbt-util'
|
|
39
42
|
require 'rbbt/util/R'
|
|
40
43
|
|
|
@@ -114,4 +117,3 @@ cat(label, file="#{results}");
|
|
|
114
117
|
end
|
|
115
118
|
end
|
|
116
119
|
end
|
|
117
|
-
|
|
@@ -0,0 +1,231 @@
|
|
|
1
|
+
# Test support: bounded availability probes for conditional test omissions.
|
|
2
|
+
#
|
|
3
|
+
# Not named test_*.rb on purpose: the Rakefile test pattern
|
|
4
|
+
# ('test/**/test_*.rb') must not collect this file.
|
|
5
|
+
#
|
|
6
|
+
# Every probe returns true/false (never raises, never downloads, never hangs).
|
|
7
|
+
# Intended usage in a test:
|
|
8
|
+
#
|
|
9
|
+
# omit 'torch CUDA unavailable' unless Availability.cuda?
|
|
10
|
+
#
|
|
11
|
+
# Guidelines honoured here:
|
|
12
|
+
# * Huggingface model presence is checked against the local HF cache only
|
|
13
|
+
# (HF_HOME/hub or ~/.cache/huggingface/hub), NEVER through the
|
|
14
|
+
# transformers API: importing transformers or building a tokenizer would
|
|
15
|
+
# start a multi-GB download for anything that is not already cached.
|
|
16
|
+
# * Python module probes run `python -c "import ..."` in a fresh process
|
|
17
|
+
# with a hard timeout, so a hanging import cannot block the suite.
|
|
18
|
+
# * Results are memoized per process: probes are cheap but not free.
|
|
19
|
+
module Availability
|
|
20
|
+
DEFAULT_TIMEOUT = 60
|
|
21
|
+
|
|
22
|
+
class << self
|
|
23
|
+
def memo(name)
|
|
24
|
+
cache = (@cache ||= {})
|
|
25
|
+
return cache[name] if cache.key?(name)
|
|
26
|
+
cache[name] = begin
|
|
27
|
+
yield
|
|
28
|
+
rescue Exception
|
|
29
|
+
false
|
|
30
|
+
end
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
def python_executable
|
|
34
|
+
ENV['SCOUT_TEST_PYTHON'] || 'python'
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
# python itself is usable (python executable present and runnable)
|
|
38
|
+
def python?
|
|
39
|
+
memo(:python) do
|
|
40
|
+
run_bounded([python_executable, '-c', 'import sys; sys.exit(0)'], 30)
|
|
41
|
+
true
|
|
42
|
+
end
|
|
43
|
+
end
|
|
44
|
+
|
|
45
|
+
# all listed python modules import cleanly (fresh process, bounded)
|
|
46
|
+
def python_modules?(*mods)
|
|
47
|
+
memo(:"python_modules_#{mods.join(',')}") do
|
|
48
|
+
raise Errno::ENOENT unless python?
|
|
49
|
+
code = mods.collect { |m| "import #{m}" } * ';'
|
|
50
|
+
_out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
|
|
51
|
+
status == 0
|
|
52
|
+
end
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
# informative reason string when a python module is missing, nil otherwise
|
|
56
|
+
def python_modules_reason(*mods)
|
|
57
|
+
return 'python unavailable' unless python?
|
|
58
|
+
code = mods.collect { |m| "import #{m}" } * ';'
|
|
59
|
+
out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
|
|
60
|
+
return nil if status == 0
|
|
61
|
+
"python modules missing (#{mods * ','}): #{out.split("\n").last.to_s.strip}"
|
|
62
|
+
end
|
|
63
|
+
|
|
64
|
+
# torch is importable (does NOT imply CUDA)
|
|
65
|
+
def torch?
|
|
66
|
+
python_modules?('torch')
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# torch is importable AND reports a usable CUDA device
|
|
70
|
+
def cuda?
|
|
71
|
+
memo(:cuda) do
|
|
72
|
+
return false unless torch?
|
|
73
|
+
code = 'import sys, torch; sys.exit(0 if torch.cuda.is_available() else 1)'
|
|
74
|
+
_out, _err, status = run_bounded([python_executable, '-c', code], DEFAULT_TIMEOUT, capture: true)
|
|
75
|
+
status == 0
|
|
76
|
+
end
|
|
77
|
+
end
|
|
78
|
+
|
|
79
|
+
# A Huggingface model is already in the local cache; this NEVER triggers a
|
|
80
|
+
# download (see the note at the top of the file).
|
|
81
|
+
def hf_model_cached?(repo)
|
|
82
|
+
memo(:"hf_model_#{repo}") do
|
|
83
|
+
hub = ENV['HF_HOME'] ? File.join(ENV['HF_HOME'], 'hub') : File.join(Dir.home, '.cache', 'huggingface', 'hub')
|
|
84
|
+
repo_dir = 'models--' + repo.to_s.tr('/', '--')
|
|
85
|
+
path = File.join(hub, repo_dir)
|
|
86
|
+
Dir.exist?(path) && Dir.glob(File.join(path, 'snapshots', '*')).any?
|
|
87
|
+
end
|
|
88
|
+
end
|
|
89
|
+
|
|
90
|
+
# Reason for a missing huggingface model
|
|
91
|
+
def hf_model_reason(repo)
|
|
92
|
+
return "huggingface model #{repo} not in local cache (HF_HOME=#{ENV['HF_HOME'] || '~/.cache/huggingface'}); probing would download it" unless hf_model_cached?(repo)
|
|
93
|
+
nil
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
# Rscript is present
|
|
97
|
+
def rscript?
|
|
98
|
+
memo(:rscript) { which?('Rscript') }
|
|
99
|
+
end
|
|
100
|
+
|
|
101
|
+
# R is present and the listed packages are installed (bounded Rscript -e)
|
|
102
|
+
def r_packages?(*pkgs)
|
|
103
|
+
memo(:"r_packages_#{pkgs.join(',')}") do
|
|
104
|
+
return false unless rscript?
|
|
105
|
+
code = "cat(if (all(c(#{pkgs.collect { |p| "'#{p}'" } * ', '}) %in% rownames(installed.packages()))) 'yes' else 'no')"
|
|
106
|
+
out, _err, status = run_bounded(['Rscript', '-e', code], DEFAULT_TIMEOUT, capture: true)
|
|
107
|
+
status == 0 && out.strip == 'yes'
|
|
108
|
+
end
|
|
109
|
+
end
|
|
110
|
+
|
|
111
|
+
def r_packages_reason(*pkgs)
|
|
112
|
+
return 'Rscript not found' unless rscript?
|
|
113
|
+
return nil if r_packages?(*pkgs)
|
|
114
|
+
"R packages missing: #{pkgs * ','}"
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
#{{{ endpoints
|
|
118
|
+
|
|
119
|
+
# An endpoint yaml exists in the Scout configuration paths (test/etc/AI,
|
|
120
|
+
# ~/.scout/etc/AI, ...). Purely local: no network access.
|
|
121
|
+
def endpoint_configured?(endpoint)
|
|
122
|
+
memo(:"endpoint_configured_#{endpoint}") do
|
|
123
|
+
Scout.etc.AI[endpoint.to_s].find_with_extension(:yaml).exists?
|
|
124
|
+
end
|
|
125
|
+
end
|
|
126
|
+
|
|
127
|
+
# The endpoint points to a host:port that accepts a TCP connection within
|
|
128
|
+
# the timeout. Only probed for endpoints whose config carries a url;
|
|
129
|
+
# endpoints without a url (mock, relay through scp, ...) count as
|
|
130
|
+
# reachable once configured.
|
|
131
|
+
def endpoint_reachable?(endpoint, timeout = 5)
|
|
132
|
+
memo(:"endpoint_reachable_#{endpoint}") do
|
|
133
|
+
path = Scout.etc.AI[endpoint.to_s].find_with_extension(:yaml)
|
|
134
|
+
return false unless path.exists?
|
|
135
|
+
url = path.yaml[:url] || path.yaml['url']
|
|
136
|
+
return true if url.nil?
|
|
137
|
+
|
|
138
|
+
require 'uri'
|
|
139
|
+
require 'socket'
|
|
140
|
+
uri = URI.parse(url.to_s)
|
|
141
|
+
port = uri.port || (uri.scheme == 'https' ? 443 : 80)
|
|
142
|
+
begin
|
|
143
|
+
Timeout.timeout(timeout) do
|
|
144
|
+
TCPSocket.open(uri.host, port) { |s| s.close }
|
|
145
|
+
end
|
|
146
|
+
true
|
|
147
|
+
rescue Exception
|
|
148
|
+
false
|
|
149
|
+
end
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
|
|
153
|
+
# Combined: configured AND (no url OR reachable)
|
|
154
|
+
def endpoint_available?(endpoint)
|
|
155
|
+
return false unless endpoint_configured?(endpoint)
|
|
156
|
+
endpoint_reachable?(endpoint)
|
|
157
|
+
end
|
|
158
|
+
|
|
159
|
+
def endpoint_reason(endpoint)
|
|
160
|
+
return "endpoint #{endpoint} not configured (no yaml in the Scout AI paths)" unless endpoint_configured?(endpoint)
|
|
161
|
+
return "endpoint #{endpoint} configured but not reachable" unless endpoint_reachable?(endpoint)
|
|
162
|
+
nil
|
|
163
|
+
end
|
|
164
|
+
|
|
165
|
+
# All endpoint names defined in Scout.etc.AI across the search paths
|
|
166
|
+
# (test/etc/AI plus the user configuration). Local filesystem only.
|
|
167
|
+
def configured_endpoints
|
|
168
|
+
memo(:configured_endpoints) do
|
|
169
|
+
Scout.etc.AI.glob('*.yaml').collect { |f| File.basename(f, '.yaml') }.uniq
|
|
170
|
+
end
|
|
171
|
+
end
|
|
172
|
+
|
|
173
|
+
# Endpoints that are real inference services (anything but the offline
|
|
174
|
+
# mock one defined by this test suite).
|
|
175
|
+
def real_endpoints
|
|
176
|
+
memo(:real_endpoints) do
|
|
177
|
+
configured_endpoints.reject { |e| e == 'mock' }
|
|
178
|
+
end
|
|
179
|
+
end
|
|
180
|
+
|
|
181
|
+
#{{{ internals
|
|
182
|
+
|
|
183
|
+
def which?(cmd)
|
|
184
|
+
ENV['PATH'].to_s.split(File::PATH_SEPARATOR).any? do |dir|
|
|
185
|
+
path = File.join(dir, cmd.to_s)
|
|
186
|
+
File.executable?(path)
|
|
187
|
+
end
|
|
188
|
+
end
|
|
189
|
+
|
|
190
|
+
# Runs cmd with a hard timeout. Returns [stdout, stderr, exit_status].
|
|
191
|
+
# ScoutCoder: Open3.capture3 threads cannot be killed, so Timeout.timeout
|
|
192
|
+
# only breaks the wait; the child is detached with Process.kill(:KILL) to
|
|
193
|
+
# guarantee the probe returns even when the command hangs (e.g. a python
|
|
194
|
+
# import that blocks on a broken environment).
|
|
195
|
+
def run_bounded(cmd, timeout = DEFAULT_TIMEOUT, capture: true)
|
|
196
|
+
out_r, out_w = IO.pipe
|
|
197
|
+
err_r, err_w = IO.pipe
|
|
198
|
+
|
|
199
|
+
pid = Process.spawn(*cmd, out: out_w, err: err_w)
|
|
200
|
+
out_w.close
|
|
201
|
+
err_w.close
|
|
202
|
+
|
|
203
|
+
status = nil
|
|
204
|
+
begin
|
|
205
|
+
Timeout.timeout(timeout) do
|
|
206
|
+
out = capture ? out_r.read : nil
|
|
207
|
+
err = capture ? err_r.read : nil
|
|
208
|
+
_pid, status = Process.wait2(pid)
|
|
209
|
+
return [out.to_s, err.to_s, status.exitstatus]
|
|
210
|
+
end
|
|
211
|
+
rescue Timeout::Error
|
|
212
|
+
begin
|
|
213
|
+
Process.kill(:KILL, pid)
|
|
214
|
+
rescue Errno::ESRCH
|
|
215
|
+
end
|
|
216
|
+
Process.detach(pid)
|
|
217
|
+
return ['', 'timeout', -1]
|
|
218
|
+
rescue Errno::ENOENT
|
|
219
|
+
begin
|
|
220
|
+
Process.kill(:KILL, pid)
|
|
221
|
+
rescue Errno::ESRCH, Errno::EPERM
|
|
222
|
+
end
|
|
223
|
+
return ['', 'not found', -1]
|
|
224
|
+
ensure
|
|
225
|
+
out_r.close rescue nil
|
|
226
|
+
err_r.close rescue nil
|
|
227
|
+
end
|
|
228
|
+
['', 'unknown', -1]
|
|
229
|
+
end
|
|
230
|
+
end
|
|
231
|
+
end
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# Test support: offline stand-ins for the provider HTTP clients.
|
|
2
|
+
#
|
|
3
|
+
# Not named test_*.rb on purpose: the Rakefile test pattern
|
|
4
|
+
# ('test/**/test_*.rb') must not collect this file.
|
|
5
|
+
#
|
|
6
|
+
# Each fake wraps one provider SDK client class. They all share:
|
|
7
|
+
# * a response list replayed in order (last entry repeats when exhausted)
|
|
8
|
+
# * `.calls` recording every parameters hash for assertions
|
|
9
|
+
# * optional failure injection: an Exception instance entry is raised
|
|
10
|
+
#
|
|
11
|
+
# Recorded parameters are plain Hash/IndiferentHash values so assertions can
|
|
12
|
+
# deep-compare them against expected payloads without type surprises.
|
|
13
|
+
class FakeResponseClientBase
|
|
14
|
+
attr_reader :calls, :responses
|
|
15
|
+
|
|
16
|
+
# NOTE: entries are NOT flattened. A single entry may itself be an Array
|
|
17
|
+
# (e.g. the Ollama API returns an array of chunk hashes per call, and
|
|
18
|
+
# OLlama#process_response iterates the returned value).
|
|
19
|
+
def initialize(*responses)
|
|
20
|
+
@responses = responses
|
|
21
|
+
@calls = []
|
|
22
|
+
@index = 0
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
# Replays the next scripted response. Exception entries are raised so tests
|
|
26
|
+
# can exercise provider error paths.
|
|
27
|
+
def next_response
|
|
28
|
+
entry = @responses[@index] || @responses.last
|
|
29
|
+
@index += 1
|
|
30
|
+
raise entry if Exception === entry
|
|
31
|
+
entry
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def record(parameters)
|
|
35
|
+
@calls << IndiferentHash.setup(parameters.dup)
|
|
36
|
+
end
|
|
37
|
+
end
|
|
38
|
+
|
|
39
|
+
# ruby-openai Chat Completions + embeddings:
|
|
40
|
+
# client.chat(parameters:), client.embeddings(parameters:)
|
|
41
|
+
class FakeOpenAIChatClient < FakeResponseClientBase
|
|
42
|
+
def chat(parameters: {})
|
|
43
|
+
record(parameters)
|
|
44
|
+
next_response
|
|
45
|
+
end
|
|
46
|
+
|
|
47
|
+
def embeddings(parameters: {})
|
|
48
|
+
record(parameters)
|
|
49
|
+
next_response
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
# ruby-openai Responses API: client.responses.create(parameters:)
|
|
54
|
+
class FakeResponsesClient < FakeResponseClientBase
|
|
55
|
+
ResponseStub = Struct.new(:create)
|
|
56
|
+
|
|
57
|
+
def responses
|
|
58
|
+
self
|
|
59
|
+
end
|
|
60
|
+
|
|
61
|
+
def create(parameters: {})
|
|
62
|
+
record(parameters)
|
|
63
|
+
next_response
|
|
64
|
+
end
|
|
65
|
+
end
|
|
66
|
+
|
|
67
|
+
# anthropic gem: client.messages(parameters:)
|
|
68
|
+
class FakeAnthropicClient < FakeResponseClientBase
|
|
69
|
+
def messages(parameters: {})
|
|
70
|
+
record(parameters)
|
|
71
|
+
next_response
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
# ollama-ai gem: client.chat(parameters) (positional),
|
|
76
|
+
# client.request(path, parameters)
|
|
77
|
+
class FakeOllamaClient < FakeResponseClientBase
|
|
78
|
+
def chat(parameters = {})
|
|
79
|
+
record(parameters)
|
|
80
|
+
next_response
|
|
81
|
+
end
|
|
82
|
+
|
|
83
|
+
def request(path, parameters = {})
|
|
84
|
+
record(parameters.merge(path: path))
|
|
85
|
+
next_response
|
|
86
|
+
end
|
|
87
|
+
end
|
|
88
|
+
|
|
89
|
+
# aws-sdk-bedrockruntime: client.invoke_model(model_id:, content_type:, body:)
|
|
90
|
+
# returning an object whose .body.string is the JSON payload.
|
|
91
|
+
class FakeBedrockClient
|
|
92
|
+
BodyStub = Struct.new(:string)
|
|
93
|
+
ResponseStub = Struct.new(:body)
|
|
94
|
+
|
|
95
|
+
attr_reader :calls
|
|
96
|
+
|
|
97
|
+
def initialize(*payloads)
|
|
98
|
+
@payloads = payloads
|
|
99
|
+
@calls = []
|
|
100
|
+
@index = 0
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
# Records the request (body parsed back to a Hash for assertions) and returns
|
|
104
|
+
# the next scripted payload wrapped as invoke_model does.
|
|
105
|
+
def invoke_model(model_id:, content_type:, body:)
|
|
106
|
+
payload = @payloads[@index] || @payloads.last
|
|
107
|
+
@index += 1
|
|
108
|
+
@calls << IndiferentHash.setup({model_id: model_id, content_type: content_type, body: JSON.parse(body)})
|
|
109
|
+
ResponseStub.new(BodyStub.new(payload.to_json))
|
|
110
|
+
end
|
|
111
|
+
end
|
|
112
|
+
|
|
113
|
+
module TestFixtures
|
|
114
|
+
class << self
|
|
115
|
+
# Build a fake client pre-loaded with the named fixtures, e.g.
|
|
116
|
+
# TestFixtures.openai_client('backends/openai_chat_tool_call',
|
|
117
|
+
# 'backends/openai_chat')
|
|
118
|
+
def openai_client(*names)
|
|
119
|
+
FakeOpenAIChatClient.new(*names.collect { |n| fixture(n) })
|
|
120
|
+
end
|
|
121
|
+
|
|
122
|
+
def responses_client(*names)
|
|
123
|
+
FakeResponsesClient.new(*names.collect { |n| fixture(n) })
|
|
124
|
+
end
|
|
125
|
+
|
|
126
|
+
def anthropic_client(*names)
|
|
127
|
+
FakeAnthropicClient.new(*names.collect { |n| fixture(n) })
|
|
128
|
+
end
|
|
129
|
+
|
|
130
|
+
def ollama_client(*names)
|
|
131
|
+
FakeOllamaClient.new(*names.collect { |n| fixture(n) })
|
|
132
|
+
end
|
|
133
|
+
|
|
134
|
+
def bedrock_client(*names)
|
|
135
|
+
FakeBedrockClient.new(*names.collect { |n| fixture(n) })
|
|
136
|
+
end
|
|
137
|
+
end
|
|
138
|
+
end
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
# Test support: fixture loading for recorded LLM provider payloads.
|
|
2
|
+
#
|
|
3
|
+
# Not named test_*.rb on purpose: the Rakefile test pattern
|
|
4
|
+
# ('test/**/test_*.rb') must not collect this file.
|
|
5
|
+
module TestFixtures
|
|
6
|
+
FIXTURES_DIR = File.expand_path('../fixtures', __dir__)
|
|
7
|
+
|
|
8
|
+
class << self
|
|
9
|
+
# TestFixtures.fixture('backends/openai_chat')
|
|
10
|
+
# -> parsed JSON from test/fixtures/backends/openai_chat.json
|
|
11
|
+
def fixture(name)
|
|
12
|
+
path = fixture_path(name)
|
|
13
|
+
content = Open.read(path)
|
|
14
|
+
JSON.parse(content)
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def fixture_path(name)
|
|
18
|
+
Path.setup(File.join(FIXTURES_DIR, name.to_s + '.json')).find
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
end
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# Test support: infrastructure probe collector shared by the endpoint suites.
|
|
2
|
+
#
|
|
3
|
+
# Not named test_*.rb on purpose: the Rakefile test pattern must not collect
|
|
4
|
+
# this file.
|
|
5
|
+
#
|
|
6
|
+
# Runs a fixed set of probes against one inference target and reports the
|
|
7
|
+
# outcome (including latency and token usage) as a summary instead of raising:
|
|
8
|
+
#
|
|
9
|
+
# trivial : answer a trivial question
|
|
10
|
+
# tool_call: one weather-style tool call round (a block answers the tool)
|
|
11
|
+
# kb_query : the Miki brother-in-law knowledge base question, a two-hop
|
|
12
|
+
# tool chain (marriages then brothers)
|
|
13
|
+
#
|
|
14
|
+
# Failures are recorded as :fail rows, never raised, so one broken endpoint
|
|
15
|
+
# cannot break the run; the summary table is printed and written to
|
|
16
|
+
# results/infrastructure_summary.md by whichever suite runs last.
|
|
17
|
+
module InfrastructureProbes
|
|
18
|
+
TIMEOUT = 300
|
|
19
|
+
|
|
20
|
+
WEATHER_TOOL = {
|
|
21
|
+
"type": "function",
|
|
22
|
+
"function": {
|
|
23
|
+
"name": "get_current_temperature",
|
|
24
|
+
"description": "Get the current temperature and raining conditions for a specific location",
|
|
25
|
+
"parameters": {
|
|
26
|
+
"type": "object",
|
|
27
|
+
"properties": {
|
|
28
|
+
"location": { "type": "string", "description": "The city and state, e.g., San Francisco, CA" },
|
|
29
|
+
"unit": { "type": "string", "enum": ["Celsius", "Fahrenheit"],
|
|
30
|
+
"description": "The temperature unit to use. Infer this from the user's location." }
|
|
31
|
+
},
|
|
32
|
+
"required": ["location", "unit"]
|
|
33
|
+
}
|
|
34
|
+
}
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
class << self
|
|
38
|
+
attr_accessor :results
|
|
39
|
+
end
|
|
40
|
+
self.results ||= []
|
|
41
|
+
|
|
42
|
+
module_function
|
|
43
|
+
|
|
44
|
+
def record(target, probe, status, reason: nil, answer: nil, duration: nil, tokens: nil)
|
|
45
|
+
InfrastructureProbes.results << {target: target, probe: probe, status: status, reason: reason,
|
|
46
|
+
answer: answer, duration: duration, tokens: tokens}
|
|
47
|
+
end
|
|
48
|
+
|
|
49
|
+
# Run the three probes against `target` (a display name), using
|
|
50
|
+
# `ask_options` (already carrying :endpoint or nothing, i.e. the default).
|
|
51
|
+
# Nothing raises: each probe is rescued and recorded.
|
|
52
|
+
def run_probes(target, ask_options = {})
|
|
53
|
+
nonce = Time.now.to_i
|
|
54
|
+
|
|
55
|
+
run_probe(target, :trivial, ask_options) do
|
|
56
|
+
answer = LLM.ask "user: Reply with the single word OK and nothing else (nonce #{nonce})",
|
|
57
|
+
ask_options.merge(persist: false)
|
|
58
|
+
raise 'no answer' if answer.to_s.strip.empty?
|
|
59
|
+
answer
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
run_probe(target, :tool_call, ask_options) do
|
|
63
|
+
answer = LLM.ask "user: What is the weather in London? Use the provided tool and then answer in one sentence (nonce #{nonce}).",
|
|
64
|
+
ask_options.merge(tools: [WEATHER_TOOL], tool_choice: 'required', persist: false) do |_name, _arguments|
|
|
65
|
+
"It's 15 degrees and raining."
|
|
66
|
+
end
|
|
67
|
+
raise 'no answer' if answer.to_s.strip.empty?
|
|
68
|
+
answer
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
run_probe(target, :kb_query, ask_options) do
|
|
72
|
+
TmpFile.with_dir do |dir|
|
|
73
|
+
kb = KnowledgeBase.new dir
|
|
74
|
+
kb.register :brothers, Test::Unit::TestCase.datafile_test(:person).brothers, undirected: true
|
|
75
|
+
kb.register :marriages, Test::Unit::TestCase.datafile_test(:person).marriages,
|
|
76
|
+
undirected: true, source: "=>Alias", target: "=>Alias"
|
|
77
|
+
|
|
78
|
+
text = LLM.knowledge_base_ask(kb,
|
|
79
|
+
"Who is Miki's brother in law? The brother in law is your spouse's sibling. Use the marriages and brothers tools to find out (nonce #{nonce}).",
|
|
80
|
+
ask_options.merge(persist: false))
|
|
81
|
+
raise "answer did not mention Guille: #{text.to_s[0, 200]}" unless text.to_s.include?('Guille')
|
|
82
|
+
text
|
|
83
|
+
end
|
|
84
|
+
end
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def run_probe(target, probe, ask_options)
|
|
88
|
+
start = Time.now
|
|
89
|
+
answer = Timeout.timeout(TIMEOUT) { yield }
|
|
90
|
+
record(target, probe, :ok, answer: answer, duration: Time.now - start, tokens: nil)
|
|
91
|
+
rescue Exception => e
|
|
92
|
+
record(target, probe, :fail,
|
|
93
|
+
reason: "#{e.class}: #{e.message.to_s.lines.first}",
|
|
94
|
+
duration: Time.now - start)
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
def summary_lines
|
|
98
|
+
lines = []
|
|
99
|
+
lines << 'Infrastructure endpoint summary'
|
|
100
|
+
lines << '=' * 96
|
|
101
|
+
InfrastructureProbes.results.each do |r|
|
|
102
|
+
line = [r[:target].to_s.ljust(14), r[:probe].to_s.ljust(11)]
|
|
103
|
+
case r[:status]
|
|
104
|
+
when :ok
|
|
105
|
+
line << 'OK'.ljust(5)
|
|
106
|
+
line << (r[:duration] ? "#{r[:duration].round(1)}s" : '').ljust(8)
|
|
107
|
+
line << " #{r[:answer].to_s.gsub(/\s+/, ' ')[0, 100]}"
|
|
108
|
+
when :fail
|
|
109
|
+
line << 'FAIL'.ljust(5)
|
|
110
|
+
line << (r[:duration] ? "#{r[:duration].round(1)}s" : '').ljust(8)
|
|
111
|
+
line << " #{r[:reason].to_s.gsub(/\s+/, ' ')[0, 140]}"
|
|
112
|
+
else
|
|
113
|
+
line << 'OMIT'.ljust(5)
|
|
114
|
+
line << " #{r[:reason].to_s.gsub(/\s+/, ' ')[0, 140]}"
|
|
115
|
+
end
|
|
116
|
+
lines << line * ' | '
|
|
117
|
+
end
|
|
118
|
+
lines << '=' * 96
|
|
119
|
+
lines << "#{InfrastructureProbes.results.count { |r| r[:status] == :ok }} ok, " +
|
|
120
|
+
"#{InfrastructureProbes.results.count { |r| r[:status] == :fail }} failed, " +
|
|
121
|
+
"#{InfrastructureProbes.results.count { |r| r[:status] == :omit }} omitted"
|
|
122
|
+
lines
|
|
123
|
+
end
|
|
124
|
+
|
|
125
|
+
def write_summary
|
|
126
|
+
return if InfrastructureProbes.results.empty?
|
|
127
|
+
content = summary_lines * "\n"
|
|
128
|
+
$stdout.puts "\n" + content
|
|
129
|
+
begin
|
|
130
|
+
Open.mkdir 'results'
|
|
131
|
+
Open.write 'results/infrastructure_summary.md', content + "\n"
|
|
132
|
+
rescue Exception
|
|
133
|
+
$stderr.puts "Could not write results/infrastructure_summary.md: #{$!.message}"
|
|
134
|
+
end
|
|
135
|
+
end
|
|
136
|
+
end
|