scout-ai 1.2.3 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.vimproject +138 -50
- data/README.md +171 -290
- data/Rakefile +17 -1
- data/VERSION +1 -1
- data/doc/Improvements.md +325 -0
- data/doc/StartHere.md +110 -0
- data/doc/developer/Architecture.md +126 -0
- data/doc/developer/Backends.md +199 -0
- data/doc/developer/ChatLifecycle.md +183 -0
- data/doc/developer/DelegationInternals.md +295 -0
- data/doc/developer/DesignPrinciples.md +245 -0
- data/doc/developer/PromptProcessing.md +292 -0
- data/doc/developer/Provenance.md +317 -0
- data/doc/user/BuildingAgents.md +345 -0
- data/doc/user/Cookbook.md +333 -0
- data/doc/user/CoreConcepts.md +181 -0
- data/doc/user/Delegation.md +191 -0
- data/doc/user/GettingStarted.md +159 -0
- data/doc/user/ManagingContext.md +163 -0
- data/doc/user/MultiAgentWorkflows.md +256 -0
- data/doc/user/Python.md +159 -0
- data/doc/user/RunningInference.md +200 -0
- data/doc/user/ToolCalling.md +193 -0
- data/doc/user/WritingChats.md +197 -0
- data/lib/scout/llm/agent/chat.rb +61 -11
- data/lib/scout/llm/agent/delegate.rb +274 -65
- data/lib/scout/llm/agent/iterate.rb +2 -2
- data/lib/scout/llm/agent/save.rb +273 -0
- data/lib/scout/llm/agent/workflow.rb +164 -0
- data/lib/scout/llm/agent.rb +86 -61
- data/lib/scout/llm/ask.rb +62 -17
- data/lib/scout/llm/backends/anthropic.rb +9 -2
- data/lib/scout/llm/backends/bedrock.rb +15 -3
- data/lib/scout/llm/backends/default.rb +183 -99
- data/lib/scout/llm/backends/glm.rb +58 -0
- data/lib/scout/llm/backends/huggingface.rb +196 -26
- data/lib/scout/llm/backends/ollama.rb +13 -1
- data/lib/scout/llm/backends/openai.rb +0 -2
- data/lib/scout/llm/backends/openwebui.rb +20 -13
- data/lib/scout/llm/backends/relay.rb +22 -22
- data/lib/scout/llm/backends/responses.rb +1 -1
- data/lib/scout/llm/chat/agent_meta.rb +264 -0
- data/lib/scout/llm/chat/annotation.rb +39 -10
- data/lib/scout/llm/chat/parse.rb +28 -6
- data/lib/scout/llm/chat/persist.rb +25 -0
- data/lib/scout/llm/chat/process/clear.rb +41 -6
- data/lib/scout/llm/chat/process/files.rb +21 -6
- data/lib/scout/llm/chat/process/meta.rb +421 -34
- data/lib/scout/llm/chat/process/options.rb +21 -1
- data/lib/scout/llm/chat/process/tools.rb +56 -15
- data/lib/scout/llm/chat/process.rb +4 -0
- data/lib/scout/llm/chat/prompt/shorten_tools.rb +125 -0
- data/lib/scout/llm/chat/prompt/shorten_tools_epoch.rb +365 -0
- data/lib/scout/llm/chat/prompt.rb +48 -0
- data/lib/scout/llm/chat/provenance.rb +775 -0
- data/lib/scout/llm/chat/tool_calls.rb +76 -0
- data/lib/scout/llm/chat.rb +18 -2
- data/lib/scout/llm/embed.rb +11 -3
- data/lib/scout/llm/image.rb +86 -0
- data/lib/scout/llm/mcp.rb +10 -2
- data/lib/scout/llm/rag.rb +3 -3
- data/lib/scout/llm/tools/call.rb +160 -11
- data/lib/scout/llm/tools/knowledge_base.rb +1 -1
- data/lib/scout/llm/tools/workflow.rb +32 -16
- data/lib/scout/model/python/huggingface/causal.rb +23 -5
- data/lib/scout/model/python/huggingface.rb +2 -1
- data/lib/scout-ai.rb +1 -0
- data/python/README.md +197 -14
- data/python/scout_ai/huggingface/eval.py +245 -34
- data/python/tests/test_huggingface_eval.py +58 -0
- data/research/ChatAnalyst-required-changes.md +167 -0
- data/research/agent-delegation-analysis.md +810 -0
- data/research/agent-meta-provenance-integration-plan.md +622 -0
- data/research/agent-workflow-analysis.md +1120 -0
- data/research/backends-analysis.md +836 -0
- data/research/chat-core-analysis.md +946 -0
- data/research/chatanalyst-provenance/00-baseline.md +30 -0
- data/research/chatanalyst-provenance/01-repo-map.md +60 -0
- data/research/chatanalyst-provenance/02-event-reconstruction.md +55 -0
- data/research/chatanalyst-provenance/03-duplication-evidence.md +45 -0
- data/research/chatanalyst-provenance/04-tooling-root-cause.md +57 -0
- data/research/chatanalyst-provenance/05-fix-plan.md +46 -0
- data/research/chatanalyst-provenance/07-critic-review.md +25 -0
- data/research/chatanalyst-provenance/final-report.md +45 -0
- data/research/chatanalyst-provenance/resumption.md +37 -0
- data/research/coding-philosophy-analysis.md +928 -0
- data/research/commands-analysis.md +947 -0
- data/research/multi-agent-patterns-analysis.md +853 -0
- data/research/prompt-strategies-analysis.md +630 -0
- data/research/prov-verbosity-fix-notes.md +77 -0
- data/research/provenance-analysis.md +469 -0
- data/research/provenance-navigation-design.md +640 -0
- data/research/synthesis-report.md +487 -0
- data/research/tools-system-analysis.md +779 -0
- data/scout-ai.gemspec +100 -11
- data/scout_commands/agent/ask +13 -3
- data/scout_commands/agent/kb +2 -0
- data/scout_commands/llm/ask +11 -4
- data/scout_commands/llm/md +76 -0
- data/scout_commands/llm/process_queries +48 -0
- data/scout_commands/llm/prov +602 -0
- data/scout_commands/llm/word +71 -0
- data/scout_commands/workflow/mcp +43 -0
- data/share/word/reference.docx +0 -0
- data/test/etc/AI/mock.yaml +11 -0
- data/test/fixtures/backends/anthropic.json +19 -0
- data/test/fixtures/backends/anthropic_tool_use.json +24 -0
- data/test/fixtures/backends/bedrock.json +8 -0
- data/test/fixtures/backends/bedrock_embedding.json +3 -0
- data/test/fixtures/backends/bedrock_tool_use.json +17 -0
- data/test/fixtures/backends/ollama.json +16 -0
- data/test/fixtures/backends/ollama_tool_call.json +27 -0
- data/test/fixtures/backends/openai_chat.json +21 -0
- data/test/fixtures/backends/openai_chat_tool_call.json +31 -0
- data/test/fixtures/backends/responses.json +33 -0
- data/test/fixtures/backends/responses_tool_call.json +28 -0
- data/test/integration/README.md +32 -0
- data/test/integration/scout/llm/backends/test_endpoints.rb +34 -0
- data/test/integration/scout/llm/backends/test_openwebui.rb +61 -0
- data/test/integration/scout/llm/backends/test_relay.rb +52 -0
- data/test/integration/scout/llm/test_infrastructure.rb +74 -0
- data/test/{scout → integration/scout}/llm/test_mcp.rb +1 -1
- data/test/integration/scout/llm/tools/test_mcp.rb +42 -0
- data/test/integration/scout/model/test_base.rb +91 -0
- data/test/scout/llm/agent/test_chat.rb +8 -2
- data/test/scout/llm/agent/test_save.rb +413 -0
- data/test/scout/llm/agent/test_workflow.rb +110 -0
- data/test/scout/llm/backends/test_anthropic.rb +93 -10
- data/test/scout/llm/backends/test_bedrock.rb +118 -2
- data/test/scout/llm/backends/test_huggingface.rb +137 -42
- data/test/scout/llm/backends/test_ollama.rb +70 -20
- data/test/scout/llm/backends/test_openwebui.rb +42 -40
- data/test/scout/llm/backends/test_relay.rb +4 -2
- data/test/scout/llm/chat/agent_meta_fixtures.rb +131 -0
- data/test/scout/llm/chat/process/test_meta.rb +518 -0
- data/test/scout/llm/chat/process/test_normalize_usage.rb +183 -0
- data/test/scout/llm/chat/test_agent_meta.rb +357 -0
- data/test/scout/llm/chat/test_agent_meta_provenance.rb +467 -0
- data/test/scout/llm/chat/test_agent_meta_tokens.rb +594 -0
- data/test/scout/llm/chat/test_parse.rb +70 -15
- data/test/scout/llm/chat/test_prov_cli.rb +274 -0
- data/test/scout/llm/chat/test_provenance.rb +240 -0
- data/test/scout/llm/chat/test_tool_calls.rb +38 -0
- data/test/scout/llm/test_agent.rb +13 -36
- data/test/scout/llm/test_ask.rb +75 -52
- data/test/scout/llm/test_chat.rb +107 -13
- data/test/scout/llm/test_embed.rb +48 -0
- data/test/scout/llm/test_rag.rb +23 -16
- data/test/scout/llm/test_tools.rb +12 -1
- data/test/scout/llm/tools/test_knowledge_base.rb +0 -1
- data/test/scout/llm/tools/test_mcp.rb +5 -3
- data/test/scout/llm/tools/test_workflow.rb +23 -2
- data/test/scout/model/python/huggingface/causal/test_next_token.rb +11 -5
- data/test/scout/model/python/huggingface/test_causal.rb +9 -3
- data/test/scout/model/python/huggingface/test_classification.rb +11 -2
- data/test/scout/model/python/test_torch.rb +2 -0
- data/test/scout/model/python/torch/test_helpers.rb +4 -0
- data/test/scout/model/test_base.rb +4 -2
- data/test/support/availability.rb +231 -0
- data/test/support/fake_clients.rb +138 -0
- data/test/support/fixtures.rb +21 -0
- data/test/support/infrastructure_probes.rb +136 -0
- data/test/support/mock_backend.rb +215 -0
- data/test/test_helper.rb +32 -2
- metadata +99 -10
- data/doc/Agent.md +0 -327
- data/doc/Chat.md +0 -458
- data/doc/LLM.md +0 -340
- data/doc/RAG.md +0 -129
- data/scout_commands/documenter +0 -148
- data/test/scout/llm/backends/test_openai.rb +0 -192
- data/test/scout/llm/backends/test_responses.rb +0 -238
- data/test/scout/llm/test_parse.rb +0 -98
|
@@ -10,22 +10,105 @@ user: say hi
|
|
|
10
10
|
ppp LLM::Anthropic.ask prompt
|
|
11
11
|
end
|
|
12
12
|
|
|
13
|
-
|
|
13
|
+
# ScoutCoder: Anthropic has no embeddings endpoint; the backend raises
|
|
14
|
+
# from embed_query, so the offline contract is the exception, not a vector.
|
|
15
|
+
def test_embeddings
|
|
16
|
+
assert_raise(RuntimeError) { LLM::Anthropic.embed 'Some text', log_errors: false, model: 'embedding-model' }
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def test_ask
|
|
20
|
+
client = TestFixtures.anthropic_client('backends/anthropic')
|
|
21
|
+
res = LLM::Anthropic.ask 'user: write a script that sorts files in a directory',
|
|
22
|
+
client: client, model: 'claude-sonnet-4-5', persist: false
|
|
23
|
+
|
|
24
|
+
assert_equal 'Mock answer from Anthropic messages', res
|
|
25
|
+
assert_equal 1, client.calls.length
|
|
26
|
+
assert_equal 'claude-sonnet-4-5', client.calls.first[:model]
|
|
27
|
+
assert client.calls.first[:messages].any? { |m| m[:role].to_s == 'user' }
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def test_tool_call_output_weather
|
|
31
|
+
Log.severity = 0
|
|
14
32
|
prompt =<<-EOF
|
|
15
|
-
|
|
33
|
+
function_call:
|
|
34
|
+
|
|
35
|
+
{"name":"get_current_temperature", "arguments":{"location":"London","unit":"Celsius"},"id":"tNTnsQq2s6jGh0npOh43AwDD"}
|
|
36
|
+
|
|
37
|
+
function_call_output:
|
|
38
|
+
|
|
39
|
+
{"id":"tNTnsQq2s6jGh0npOh43AwDD", "content":"It's 15 degrees and raining."}
|
|
40
|
+
|
|
41
|
+
user:
|
|
42
|
+
|
|
43
|
+
should i take an umbrella?
|
|
16
44
|
EOF
|
|
17
|
-
|
|
18
|
-
|
|
45
|
+
client = TestFixtures.anthropic_client('backends/anthropic')
|
|
46
|
+
res = LLM::Anthropic.ask prompt, client: client, persist: false
|
|
47
|
+
|
|
48
|
+
assert_equal 'Mock answer from Anthropic messages', res
|
|
49
|
+
# ScoutCoder: the Anthropic backend rewrites the function_call /
|
|
50
|
+
# function_call_output pair into content array items (tool_use /
|
|
51
|
+
# tool_result) instead of separate messages.
|
|
52
|
+
sent = client.calls.first[:messages]
|
|
53
|
+
assert sent.any? { |m| m[:role].to_s == 'user' }
|
|
54
|
+
assert sent.inspect.include?('tool_result')
|
|
19
55
|
end
|
|
20
56
|
|
|
21
|
-
def
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
57
|
+
def test_tool
|
|
58
|
+
prompt =<<-EOF
|
|
59
|
+
user:
|
|
60
|
+
What is the weather in London. Should I take my umbrella?
|
|
25
61
|
EOF
|
|
26
|
-
emb = LLM::Anthropic.embed text, log_errors: true, model: 'embedding-model'
|
|
27
62
|
|
|
28
|
-
|
|
63
|
+
tools = [
|
|
64
|
+
{
|
|
65
|
+
"type": "custom",
|
|
66
|
+
"name": "get_current_temperature",
|
|
67
|
+
"description": "Get the current temperature and raining conditions for a specific location",
|
|
68
|
+
"parameters": {
|
|
69
|
+
"type": "object",
|
|
70
|
+
"properties": {
|
|
71
|
+
"location": {
|
|
72
|
+
"type": "string",
|
|
73
|
+
"description": "The city and state, e.g., San Francisco, CA"
|
|
74
|
+
},
|
|
75
|
+
"unit": {
|
|
76
|
+
"type": "string",
|
|
77
|
+
"enum": ["Celsius", "Fahrenheit"],
|
|
78
|
+
"description": "The temperature unit to use. Infer this from the user's location."
|
|
79
|
+
}
|
|
80
|
+
},
|
|
81
|
+
"required": ["location", "unit"]
|
|
82
|
+
}
|
|
83
|
+
},
|
|
84
|
+
]
|
|
85
|
+
|
|
86
|
+
client = TestFixtures.anthropic_client('backends/anthropic_tool_use', 'backends/anthropic')
|
|
87
|
+
respose = LLM::Anthropic.ask prompt, tools: tools,
|
|
88
|
+
client: client, log_errors: true, persist: false do |name,arguments|
|
|
89
|
+
"It's 15 degrees and raining."
|
|
90
|
+
end
|
|
91
|
+
|
|
92
|
+
assert_equal 'Mock answer from Anthropic messages', respose
|
|
93
|
+
assert_equal 2, client.calls.length
|
|
94
|
+
|
|
95
|
+
# ScoutCoder: LLM::Anthropic#format_tool_definitions renames
|
|
96
|
+
# `parameters` -> `input_schema` and forces type 'custom'.
|
|
97
|
+
sent_tools = client.calls.first[:tools]
|
|
98
|
+
assert sent_tools.any? { |t| (t[:name] || t['name']) == 'get_current_temperature' }
|
|
99
|
+
assert sent_tools.all? { |t| (t[:input_schema] || t['input_schema']) }
|
|
100
|
+
assert sent_tools.all? { |t| (t[:type] || t['type']).to_s == 'custom' }
|
|
101
|
+
end
|
|
102
|
+
|
|
103
|
+
def test_json_output
|
|
104
|
+
client = TestFixtures.anthropic_client('backends/anthropic')
|
|
105
|
+
res = LLM::Anthropic.ask 'user: What other movies have the protagonists of the original gost busters played on, just the top.',
|
|
106
|
+
format: :json, client: client, persist: false
|
|
107
|
+
|
|
108
|
+
assert_equal 'Mock answer from Anthropic messages', res
|
|
109
|
+
# ScoutCoder: Anthropic takes a response_format hash, not the string key
|
|
110
|
+
# the OpenAI backends use.
|
|
111
|
+
assert client.calls.first.inspect.include?('json_object')
|
|
29
112
|
end
|
|
30
113
|
|
|
31
114
|
def _test_tool_call_output_2
|
|
@@ -2,15 +2,132 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
4
|
class TestLLMBedrock < Test::Unit::TestCase
|
|
5
|
+
# Offline Bedrock coverage: LLM::Bedrock.ask/embed accept options[:client]
|
|
6
|
+
# (an object with invoke_model(model_id:, content_type:, body:) returning
|
|
7
|
+
# something whose .body.string is JSON), so FakeBedrockClient exercises the
|
|
8
|
+
# full request construction + response parsing path without AWS credentials.
|
|
9
|
+
#
|
|
10
|
+
# The old real-service versions are kept disabled below (_test_*).
|
|
11
|
+
|
|
12
|
+
# ScoutCoder: LLM::Bedrock is not required by lib/scout-ai.rb; each test has
|
|
13
|
+
# to require 'scout/llm/backends/bedrock' itself (same as LLM.ask does when
|
|
14
|
+
# dispatching to the :bedrock backend).
|
|
15
|
+
def test_ask
|
|
16
|
+
require 'scout/llm/backends/bedrock'
|
|
17
|
+
client = TestFixtures.bedrock_client('backends/bedrock')
|
|
18
|
+
|
|
19
|
+
response = LLM::Bedrock.ask 'user: say hi to bedrock',
|
|
20
|
+
client: client,
|
|
21
|
+
model: 'anthropic.claude-3-sonnet-20240229-v1:0',
|
|
22
|
+
model_max_tokens: 100
|
|
23
|
+
|
|
24
|
+
assert_equal 'Mock answer from Bedrock', response
|
|
25
|
+
|
|
26
|
+
assert_equal 1, client.calls.length
|
|
27
|
+
call = client.calls.first
|
|
28
|
+
assert_equal 'anthropic.claude-3-sonnet-20240229-v1:0', call[:model_id]
|
|
29
|
+
assert_equal 'application/json', call[:content_type]
|
|
30
|
+
# ScoutCoder: IndiferentHash#pretty_print takes 0 args, so assert on
|
|
31
|
+
# individual keys instead of pp-ing the recorded parameters.
|
|
32
|
+
assert_equal 100, call[:body]['max_tokens']
|
|
33
|
+
assert call[:body]['messages'].any? { |m| m['role'] == 'user' }
|
|
34
|
+
assert call[:body]['messages'].any? { |m| m['content'].to_s.include?('say hi to bedrock') }
|
|
35
|
+
end
|
|
36
|
+
|
|
37
|
+
def test_ask_prompt_type
|
|
38
|
+
require 'scout/llm/backends/bedrock'
|
|
39
|
+
client = TestFixtures.bedrock_client('backends/bedrock')
|
|
40
|
+
|
|
41
|
+
response = LLM::Bedrock.ask 'user: say hi through the prompt endpoint',
|
|
42
|
+
client: client, type: :prompt,
|
|
43
|
+
model: 'meta.llama3-8b-instruct-v1:0',
|
|
44
|
+
model_max_tokens: 100
|
|
45
|
+
|
|
46
|
+
assert_equal 'Mock answer from Bedrock', response
|
|
47
|
+
|
|
48
|
+
body = client.calls.first[:body]
|
|
49
|
+
assert body.include?('prompt')
|
|
50
|
+
assert body['prompt'].to_s.include?('say hi through the prompt endpoint')
|
|
51
|
+
end
|
|
52
|
+
|
|
53
|
+
def test_embeddings
|
|
54
|
+
require 'scout/llm/backends/bedrock'
|
|
55
|
+
client = TestFixtures.bedrock_client('backends/bedrock_embedding')
|
|
56
|
+
|
|
57
|
+
emb = LLM::Bedrock.embed 'Some text', client: client,
|
|
58
|
+
model: 'amazon.titan-embed-text-v1'
|
|
59
|
+
|
|
60
|
+
assert(Float === emb.first)
|
|
61
|
+
assert_equal [0.1, 0.2, 0.3], emb
|
|
62
|
+
assert_equal 1, client.calls.length
|
|
63
|
+
|
|
64
|
+
call = client.calls.first
|
|
65
|
+
assert_equal 'amazon.titan-embed-text-v1', call[:model_id]
|
|
66
|
+
assert_equal 'Some text', call[:body]['inputText']
|
|
67
|
+
end
|
|
68
|
+
|
|
69
|
+
# Tool loop: the first payload carries a tool_call content entry, the block
|
|
70
|
+
# answers it, and the second payload is the final text answer.
|
|
71
|
+
def test_tool_loop
|
|
72
|
+
require 'scout/llm/backends/bedrock'
|
|
73
|
+
tools = [
|
|
74
|
+
{
|
|
75
|
+
"type": "function",
|
|
76
|
+
"function": {
|
|
77
|
+
"name": "get_current_temperature",
|
|
78
|
+
"description": "Get the current temperature for a specific location",
|
|
79
|
+
"parameters": {
|
|
80
|
+
"type": "object",
|
|
81
|
+
"properties": {
|
|
82
|
+
"location": { "type": "string",
|
|
83
|
+
"description": "The city and state, e.g., San Francisco, CA" },
|
|
84
|
+
"unit": { "type": "string", "enum": ["Celsius", "Fahrenheit"],
|
|
85
|
+
"description": "The temperature unit to use." }
|
|
86
|
+
},
|
|
87
|
+
"required": ["location", "unit"]
|
|
88
|
+
}
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
]
|
|
92
|
+
|
|
93
|
+
client = TestFixtures.bedrock_client('backends/bedrock_tool_use', 'backends/bedrock')
|
|
94
|
+
|
|
95
|
+
calls_seen = []
|
|
96
|
+
response = LLM::Bedrock.ask 'user: What is the weather in London? Should I take an umbrella? Use the tool.',
|
|
97
|
+
tools: tools, client: client,
|
|
98
|
+
model: 'anthropic.claude-3-sonnet-20240229-v1:0',
|
|
99
|
+
model_max_tokens: 100 do |name, arguments|
|
|
100
|
+
calls_seen << [name, arguments]
|
|
101
|
+
"It's 15 degrees and raining."
|
|
102
|
+
end
|
|
103
|
+
|
|
104
|
+
assert_equal 'Mock answer from Bedrock', response
|
|
105
|
+
|
|
106
|
+
# the block was reached with the unwrapped tool call
|
|
107
|
+
assert_equal 1, calls_seen.length
|
|
108
|
+
assert_equal 'get_current_temperature', calls_seen.first.first
|
|
109
|
+
assert_equal 'London', calls_seen.first.last['location']
|
|
110
|
+
|
|
111
|
+
# two invoke_model rounds, the second one carrying the tool response
|
|
112
|
+
assert_equal 2, client.calls.length
|
|
113
|
+
sent = client.calls.last[:body]['messages']
|
|
114
|
+
assert sent.any? { |m| m['role'] == 'tool' }
|
|
115
|
+
tool_messages = sent.select { |m| m['role'] == 'tool' }
|
|
116
|
+
assert_equal 'It\'s 15 degrees and raining.', tool_messages.last['content']
|
|
117
|
+
assert_equal 'call_1', tool_messages.last['id']
|
|
118
|
+
end
|
|
119
|
+
|
|
5
120
|
def _test_ask
|
|
121
|
+
# Real-service version: see the offline tests above. Requires AWS
|
|
122
|
+
# credentials and network.
|
|
6
123
|
prompt =<<-EOF
|
|
7
124
|
say hi
|
|
8
125
|
EOF
|
|
9
126
|
ppp LLM::Bedrock.ask prompt, model: "anthropic.claude-3-sonnet-20240229-v1:0", model_max_tokens: 100, model_anthropic_version: 'bedrock-2023-05-31'
|
|
10
127
|
end
|
|
11
128
|
|
|
12
|
-
|
|
13
129
|
def _test_embeddings
|
|
130
|
+
# Real-service version: see test_embeddings above.
|
|
14
131
|
Log.severity = 0
|
|
15
132
|
text =<<-EOF
|
|
16
133
|
Some text
|
|
@@ -57,4 +174,3 @@ What is the weather in London. Should I take my umbrella? Use the provided tool
|
|
|
57
174
|
ppp response
|
|
58
175
|
end
|
|
59
176
|
end
|
|
60
|
-
|
|
@@ -2,8 +2,24 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
4
|
class TestLLMHF < Test::Unit::TestCase
|
|
5
|
+
class FakeHFClient
|
|
6
|
+
attr_reader :messages, :tools, :calls
|
|
5
7
|
|
|
6
|
-
|
|
8
|
+
def initialize(*responses)
|
|
9
|
+
@responses = responses
|
|
10
|
+
@calls = 0
|
|
11
|
+
end
|
|
12
|
+
|
|
13
|
+
def chat(messages, tools, parameters = {})
|
|
14
|
+
@messages = messages
|
|
15
|
+
@tools = tools
|
|
16
|
+
response = @responses[@calls] || @responses.last
|
|
17
|
+
@calls += 1
|
|
18
|
+
response
|
|
19
|
+
end
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
def _test_ask
|
|
7
23
|
Log.severity = 0
|
|
8
24
|
prompt =<<-EOF
|
|
9
25
|
system: you are a coding helper that only write code and inline comments. No extra explanations or comentary
|
|
@@ -13,61 +29,140 @@ user: write a script that sorts files in a directory
|
|
|
13
29
|
ppp LLM::Huggingface.ask prompt, model: 'HuggingFaceTB/SmolLM2-135M-Instruct'
|
|
14
30
|
end
|
|
15
31
|
|
|
16
|
-
def
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
32
|
+
def test_format_tool_call
|
|
33
|
+
message = {
|
|
34
|
+
role: 'function_call',
|
|
35
|
+
content: {
|
|
36
|
+
name: 'get_current_temperature',
|
|
37
|
+
arguments: { location: 'London', unit: 'Celsius' },
|
|
38
|
+
id: 'call_123'
|
|
39
|
+
}.to_json
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
formatted = LLM::Huggingface.format_tool_call(message)
|
|
43
|
+
|
|
44
|
+
assert_equal 'assistant', formatted[:role]
|
|
45
|
+
assert_equal 'function', formatted.dig(:tool_calls, 0, :type)
|
|
46
|
+
assert_equal 'call_123', formatted.dig(:tool_calls, 0, :id)
|
|
47
|
+
assert_equal 'get_current_temperature', formatted.dig(:tool_calls, 0, :function, :name)
|
|
48
|
+
assert_equal 'London', formatted.dig(:tool_calls, 0, :function, :arguments, :location)
|
|
23
49
|
end
|
|
24
50
|
|
|
25
|
-
def
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
51
|
+
def test_format_tool_output
|
|
52
|
+
message = {
|
|
53
|
+
role: 'function_call_output',
|
|
54
|
+
content: {
|
|
55
|
+
id: 'call_123',
|
|
56
|
+
name: 'get_current_temperature',
|
|
57
|
+
content: "It's 15 degrees and raining."
|
|
58
|
+
}.to_json
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
formatted = LLM::Huggingface.format_tool_output(message)
|
|
62
|
+
|
|
63
|
+
assert_equal 'tool', formatted[:role]
|
|
64
|
+
assert_equal 'get_current_temperature', formatted[:name]
|
|
65
|
+
assert_equal 'call_123', formatted[:tool_call_id]
|
|
66
|
+
assert_equal "It's 15 degrees and raining.", formatted[:content]
|
|
32
67
|
end
|
|
33
68
|
|
|
34
|
-
def
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
69
|
+
def test_parse_tool_call
|
|
70
|
+
tool_call = {
|
|
71
|
+
id: 'call_123',
|
|
72
|
+
type: 'function',
|
|
73
|
+
function: {
|
|
74
|
+
name: 'get_current_temperature',
|
|
75
|
+
arguments: { location: 'London', unit: 'Celsius' }
|
|
76
|
+
}
|
|
77
|
+
}
|
|
78
|
+
|
|
79
|
+
parsed = LLM::Huggingface.parse_tool_call(tool_call)
|
|
80
|
+
|
|
81
|
+
assert_equal 'call_123', parsed[:id]
|
|
82
|
+
assert_equal 'get_current_temperature', parsed[:name]
|
|
83
|
+
assert_equal 'London', parsed.dig(:arguments, :location)
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
def test_ask_with_fake_client
|
|
87
|
+
client = FakeHFClient.new({ role: 'assistant', content: 'Hello from Huggingface' })
|
|
88
|
+
|
|
89
|
+
response = LLM::Huggingface.ask("user: say hi", client: client, log_response: false)
|
|
90
|
+
|
|
91
|
+
assert_equal 'Hello from Huggingface', response
|
|
92
|
+
assert_equal 1, client.calls
|
|
93
|
+
end
|
|
94
|
+
|
|
95
|
+
def test_ask_tool_loop_with_fake_client
|
|
96
|
+
client = FakeHFClient.new(
|
|
97
|
+
{
|
|
98
|
+
role: 'assistant',
|
|
99
|
+
content: '',
|
|
100
|
+
tool_calls: [
|
|
101
|
+
{
|
|
102
|
+
id: 'call_123',
|
|
103
|
+
type: 'function',
|
|
104
|
+
function: {
|
|
105
|
+
name: 'get_current_temperature',
|
|
106
|
+
arguments: { location: 'London', unit: 'Celsius' }
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
]
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
role: 'assistant',
|
|
113
|
+
content: 'Take an umbrella.'
|
|
114
|
+
}
|
|
115
|
+
)
|
|
38
116
|
|
|
39
117
|
tools = [
|
|
40
118
|
{
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
|
|
49
|
-
|
|
50
|
-
"description": "The city and state, e.g., San Francisco, CA"
|
|
51
|
-
},
|
|
52
|
-
"unit": {
|
|
53
|
-
"type": "string",
|
|
54
|
-
"enum": ["Celsius", "Fahrenheit"],
|
|
55
|
-
"description": "The temperature unit to use. Infer this from the user's location."
|
|
56
|
-
}
|
|
119
|
+
type: 'function',
|
|
120
|
+
function: {
|
|
121
|
+
name: 'get_current_temperature',
|
|
122
|
+
description: 'Get the current temperature',
|
|
123
|
+
parameters: {
|
|
124
|
+
type: 'object',
|
|
125
|
+
properties: {
|
|
126
|
+
location: { type: 'string' },
|
|
127
|
+
unit: { type: 'string' }
|
|
57
128
|
},
|
|
58
|
-
|
|
129
|
+
required: %w(location unit)
|
|
59
130
|
}
|
|
60
131
|
}
|
|
61
|
-
}
|
|
132
|
+
}
|
|
62
133
|
]
|
|
63
134
|
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
"It's raining cats and dogs"
|
|
135
|
+
response = LLM::Huggingface.ask("user: What is the weather in London?", client: client, tools: tools, log_response: false) do |_name, _arguments|
|
|
136
|
+
"It's 15 degrees and raining."
|
|
67
137
|
end
|
|
68
138
|
|
|
69
|
-
|
|
139
|
+
assert_equal 'Take an umbrella.', response
|
|
140
|
+
assert_equal 2, client.calls
|
|
70
141
|
end
|
|
71
142
|
|
|
72
|
-
|
|
143
|
+
def test_format_tool_definitions
|
|
144
|
+
tools = {
|
|
145
|
+
'get_current_temperature' => [
|
|
146
|
+
nil,
|
|
147
|
+
{
|
|
148
|
+
name: 'get_current_temperature',
|
|
149
|
+
description: 'Get the current temperature',
|
|
150
|
+
parameters: {
|
|
151
|
+
type: 'object',
|
|
152
|
+
properties: {
|
|
153
|
+
location: { type: 'string' }
|
|
154
|
+
},
|
|
155
|
+
required: ['location'],
|
|
156
|
+
defaults: { unit: 'Celsius' }
|
|
157
|
+
}
|
|
158
|
+
}
|
|
159
|
+
]
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
formatted = LLM::Huggingface.format_tool_definitions(tools)
|
|
73
163
|
|
|
164
|
+
assert_equal 'function', formatted.first[:type].to_s
|
|
165
|
+
assert_equal 'get_current_temperature', formatted.first.dig(:function, :name)
|
|
166
|
+
assert_nil formatted.first.dig(:function, :parameters, :defaults)
|
|
167
|
+
end
|
|
168
|
+
end
|
|
@@ -3,14 +3,72 @@ require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1
|
|
|
3
3
|
|
|
4
4
|
class TestLLMOllama < Test::Unit::TestCase
|
|
5
5
|
|
|
6
|
-
def
|
|
7
|
-
|
|
6
|
+
def test_ask
|
|
7
|
+
client = TestFixtures.ollama_client('backends/ollama')
|
|
8
|
+
res = LLM::OLlama.ask 'user: write a script that sorts files in a directory',
|
|
9
|
+
client: client, model: 'mistral', mode: 'chat', persist: false
|
|
10
|
+
|
|
11
|
+
assert_equal 'Mock answer from Ollama', res
|
|
12
|
+
assert_equal 1, client.calls.length
|
|
13
|
+
assert_equal 'mistral', client.calls.first[:model]
|
|
14
|
+
assert client.calls.first[:messages].any? { |m| m[:role].to_s == 'user' }
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def test_tool
|
|
8
18
|
prompt =<<-EOF
|
|
9
|
-
|
|
10
|
-
system: Avoid using backticks ``` to format code.
|
|
11
|
-
user: write a script that sorts files in a directory
|
|
19
|
+
What is the weather in London. Should I take an umbrella?
|
|
12
20
|
EOF
|
|
13
|
-
|
|
21
|
+
|
|
22
|
+
tools = [
|
|
23
|
+
{
|
|
24
|
+
"type": "function",
|
|
25
|
+
"function": {
|
|
26
|
+
"name": "get_current_temperature",
|
|
27
|
+
"description": "Get the current temperature for a specific location",
|
|
28
|
+
"parameters": {
|
|
29
|
+
"type": "object",
|
|
30
|
+
"properties": {
|
|
31
|
+
"location": {
|
|
32
|
+
"type": "string",
|
|
33
|
+
"description": "The city and state, e.g., San Francisco, CA"
|
|
34
|
+
},
|
|
35
|
+
"unit": {
|
|
36
|
+
"type": "string",
|
|
37
|
+
"enum": ["Celsius", "Fahrenheit"],
|
|
38
|
+
"description": "The temperature unit to use. Infer this from the user's location."
|
|
39
|
+
}
|
|
40
|
+
},
|
|
41
|
+
"required": ["location", "unit"]
|
|
42
|
+
}
|
|
43
|
+
}
|
|
44
|
+
},
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
client = TestFixtures.ollama_client('backends/ollama_tool_call', 'backends/ollama')
|
|
48
|
+
respose = LLM::OLlama.ask prompt, model: 'gpt-oss',
|
|
49
|
+
client: client, tool_choice: 'required',
|
|
50
|
+
tools: tools, persist: false do |name,arguments|
|
|
51
|
+
"It's raining cats and dogs"
|
|
52
|
+
end
|
|
53
|
+
|
|
54
|
+
assert_equal 'Mock answer from Ollama', respose
|
|
55
|
+
assert_equal 2, client.calls.length
|
|
56
|
+
|
|
57
|
+
# ScoutCoder: LLM::OLlama#query calls client.chat(parameters) positionally
|
|
58
|
+
# (no `parameters:` kwarg) and the API returns an Array of chunk hashes.
|
|
59
|
+
sent_tools = client.calls.first[:tools]
|
|
60
|
+
assert sent_tools.any? { |t| (t.dig(:function, :name) || t.dig('function', 'name')) == 'get_current_temperature' }
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def test_embeddings
|
|
64
|
+
payload = [{ 'embeddings' => [[0.1, 0.2, 0.3]] }]
|
|
65
|
+
client = FakeOllamaClient.new(payload)
|
|
66
|
+
|
|
67
|
+
emb = LLM::OLlama.embed 'Some text', client: client, model: 'mxbai-embed-large'
|
|
68
|
+
|
|
69
|
+
assert(Float === emb.first)
|
|
70
|
+
assert_equal [0.1, 0.2, 0.3], emb
|
|
71
|
+
assert_equal 'Some text', client.calls.first[:input]
|
|
14
72
|
end
|
|
15
73
|
|
|
16
74
|
def _test_tool_call_output
|
|
@@ -89,22 +147,14 @@ What is the weather in London. Should I take an umbrella?
|
|
|
89
147
|
ppp respose
|
|
90
148
|
end
|
|
91
149
|
|
|
92
|
-
def _test_embeddings
|
|
93
|
-
Log.severity = 0
|
|
94
|
-
text =<<-EOF
|
|
95
|
-
Some text
|
|
96
|
-
EOF
|
|
97
|
-
emb = LLM::OLlama.embed text, model: 'mxbai-embed-large', url: 'localhost:3331'
|
|
98
|
-
assert(Float === emb.first)
|
|
99
|
-
end
|
|
100
|
-
|
|
101
150
|
def test_embedding_array
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
151
|
+
payload = [{ 'embeddings' => [[0.1, 0.2, 0.3], [0.4, 0.5, 0.6]] }]
|
|
152
|
+
client = FakeOllamaClient.new(payload)
|
|
153
|
+
|
|
154
|
+
emb = LLM::OLlama.embed ['Some text', 'More text'], client: client, model: 'mxbai-embed-large'
|
|
155
|
+
|
|
107
156
|
assert(Float === emb.first.first)
|
|
157
|
+
assert_equal [[0.1, 0.2, 0.3], [0.4, 0.5, 0.6]], emb
|
|
108
158
|
end
|
|
109
159
|
end
|
|
110
160
|
|
|
@@ -2,13 +2,12 @@ require File.expand_path(__FILE__).sub(%r(/test/.*), '/test/test_helper.rb')
|
|
|
2
2
|
require File.expand_path(__FILE__).sub(%r(.*/test/), '').sub(/test_(.*)\.rb/,'\1')
|
|
3
3
|
|
|
4
4
|
class TestOpenWebUI < Test::Unit::TestCase
|
|
5
|
-
|
|
5
|
+
# Real https://gepeto.bsc.es/api version moved to
|
|
6
|
+
# test/integration/scout/llm/backends/test_openwebui.rb (no client seam:
|
|
7
|
+
# LLM::OpenWebUIMethods#query posts through RestClient.post with a plain
|
|
8
|
+
# Hash "client").
|
|
9
|
+
def _test_gepeto
|
|
6
10
|
Log.severity = 0
|
|
7
|
-
prompt =<<-EOF
|
|
8
|
-
system: you are a coding helper that only write code and comments without formatting so that it can work directly, avoid the initial and end commas ```.
|
|
9
|
-
user: write a script that sorts files in a directory
|
|
10
|
-
EOF
|
|
11
|
-
|
|
12
11
|
prompt =<<-EOF
|
|
13
12
|
user: write a script that sorts files in a directory
|
|
14
13
|
EOF
|
|
@@ -16,42 +15,45 @@ user: write a script that sorts files in a directory
|
|
|
16
15
|
ppp LLM::OpenWebUI.ask prompt, model: 'qwen3-vl:30b', url: "https://gepeto.bsc.es/api"
|
|
17
16
|
end
|
|
18
17
|
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
18
|
+
# Offline: stub RestClient.post so only request construction and response
|
|
19
|
+
# parsing are exercised. OpenWebUI is OpenAI-compatible, so the openai_chat
|
|
20
|
+
# fixture is replayed as the response body.
|
|
21
|
+
def test_ask_request_construction
|
|
22
|
+
fixture = TestFixtures.fixture('backends/openai_chat')
|
|
23
|
+
expected_answer = fixture.dig('choices', 0, 'message', 'content')
|
|
24
|
+
|
|
25
|
+
recorded = []
|
|
26
|
+
original = RestClient.method(:post)
|
|
23
27
|
|
|
24
|
-
|
|
25
|
-
{
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
"name": "get_current_temperature",
|
|
29
|
-
"description": "Get the current temperature for a specific location",
|
|
30
|
-
"parameters": {
|
|
31
|
-
"type": "object",
|
|
32
|
-
"properties": {
|
|
33
|
-
"location": {
|
|
34
|
-
"type": "string",
|
|
35
|
-
"description": "The city and state, e.g., San Francisco, CA"
|
|
36
|
-
},
|
|
37
|
-
"unit": {
|
|
38
|
-
"type": "string",
|
|
39
|
-
"enum": ["Celsius", "Fahrenheit"],
|
|
40
|
-
"description": "The temperature unit to use. Infer this from the user's location."
|
|
41
|
-
}
|
|
42
|
-
},
|
|
43
|
-
"required": ["location", "unit"]
|
|
44
|
-
}
|
|
45
|
-
}
|
|
46
|
-
},
|
|
47
|
-
]
|
|
48
|
-
|
|
49
|
-
sss 0
|
|
50
|
-
respose = LLM::OpenWebUI.ask prompt, model: 'gemma2:latest', tool_choice: 'required', tools: tools do |name,arguments|
|
|
51
|
-
"It's raining cats and dogs"
|
|
28
|
+
RestClient.define_singleton_method(:post) do |url, payload, headers|
|
|
29
|
+
recorded << {url: url, payload: JSON.parse(payload), headers: headers}
|
|
30
|
+
body = Struct.new(:body).new(fixture.to_json)
|
|
31
|
+
body
|
|
52
32
|
end
|
|
53
33
|
|
|
54
|
-
|
|
34
|
+
begin
|
|
35
|
+
answer = LLM::OpenWebUI.ask 'user: write a script that sorts files in a directory offline',
|
|
36
|
+
model: 'qwen3-vl:30b', url: 'https://openwebui.example/api',
|
|
37
|
+
key: 'test-key', persist: false
|
|
38
|
+
ensure
|
|
39
|
+
RestClient.singleton_class.send(:define_method, :post, original)
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
assert_equal expected_answer, answer
|
|
43
|
+
|
|
44
|
+
assert_equal 1, recorded.length
|
|
45
|
+
call = recorded.first
|
|
46
|
+
|
|
47
|
+
assert call[:url].end_with?('chat/completions')
|
|
48
|
+
assert call[:url].start_with?('https://openwebui.example/api')
|
|
49
|
+
|
|
50
|
+
payload = call[:payload]
|
|
51
|
+
assert_equal 'qwen3-vl:30b', payload['model']
|
|
52
|
+
assert payload['messages'].any? { |m| m['role'] == 'user' }
|
|
53
|
+
|
|
54
|
+
headers = call[:headers]
|
|
55
|
+
assert headers['Authorization'] || headers[:Authorization]
|
|
56
|
+
assert_equal 'Bearer test-key', (headers['Authorization'] || headers[:Authorization]).to_s
|
|
57
|
+
assert((headers['Content-Type'] || headers[:Content_Type]).to_s.include?('application/json'))
|
|
55
58
|
end
|
|
56
59
|
end
|
|
57
|
-
|