ollama-client 1.3.0 → 1.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/.env.example +11 -0
- data/API_CONTRACT.md +79 -8
- data/CHANGELOG.md +45 -0
- data/CONTRIBUTING.md +8 -0
- data/README.md +205 -273
- data/ROADMAP.md +23 -0
- data/docs/API_GAPS.md +18 -141
- data/docs/ARCHITECTURE.md +141 -0
- data/docs/AREAS_FOR_CONSIDERATION.md +13 -1
- data/docs/CLOUD.md +19 -0
- data/docs/CONSOLE_IMPROVEMENTS.md +20 -1
- data/docs/ECOSYSTEM_GEMS.md +41 -0
- data/docs/ECOSYSTEM_STRATEGY.md +151 -0
- data/docs/GETTING_STARTED.md +16 -8
- data/docs/INTEGRATION_TESTING.md +23 -5
- data/docs/PRODUCTION_FIXES.md +15 -3
- data/docs/QUICK_START.md +1 -1
- data/docs/README.md +1 -0
- data/docs/RUBYLLM_ADOPTION_MATRIX.md +273 -0
- data/docs/adr/001-openai-boundary.md +12 -0
- data/docs/adr/002-transport-abstraction.md +12 -0
- data/docs/adr/003-response-normalization.md +12 -0
- data/docs/adr/004-mock-transport.md +16 -0
- data/docs/adr/005-error-taxonomy.md +16 -0
- data/docs/adr/006-stream-runtime.md +16 -0
- data/docs/ecosystem/BOUNDARIES.md +19 -0
- data/docs/ecosystem/DEPENDENCY_GRAPH.md +14 -0
- data/docs/ecosystem/DESIGN_PRINCIPLES.md +9 -0
- data/docs/ecosystem/EXISTING_REPOS.md +31 -0
- data/docs/ecosystem/EXPERIMENTAL_LABS.md +24 -0
- data/docs/ecosystem/OVERVIEW.md +28 -0
- data/docs/ecosystem/RELEASE_ORDER.md +17 -0
- data/docs/observability/README.md +8 -0
- data/docs/rails/README.md +7 -0
- data/docs/rfcs/0001-stream-runtime.md +6 -0
- data/docs/rfcs/0002-schema-system.md +6 -0
- data/docs/rfcs/0003-observability-hooks.md +6 -0
- data/docs/rfcs/0004-async-runtime.md +6 -0
- data/docs/rfcs/README.md +16 -0
- data/docs/runtime/ERROR_CONTRACT.md +14 -0
- data/docs/runtime/SCHEMA_CONTRACT.md +28 -0
- data/docs/runtime/STREAM_CONTRACT.md +17 -0
- data/docs/runtime/STREAM_RUNTIME.md +20 -0
- data/docs/runtime/TRANSPORT_CONTRACT.md +26 -0
- data/docs/schema/README.md +8 -0
- data/docs/schema/STRUCTURED_OUTPUTS.md +22 -0
- data/docs/streaming/README.md +8 -0
- data/docs/testing/README.md +7 -0
- data/docs/testing/REPLAY_SYSTEM.md +17 -0
- data/docs/transport/README.md +7 -0
- data/examples/README.md +44 -0
- data/examples/basic_chat.rb +9 -0
- data/examples/cloud_models.rb +162 -0
- data/examples/embeddings.rb +10 -0
- data/examples/free_catalog.json +290 -0
- data/examples/generate.rb +9 -0
- data/examples/streaming.rb +12 -0
- data/examples/structured_tools.rb +90 -0
- data/examples/tool_calling_direct.rb +101 -0
- data/examples/tool_dto_example.rb +94 -0
- data/exe/ollama-client +5 -1
- data/lib/ollama/agent/executor.rb +243 -0
- data/lib/ollama/agent/messages.rb +31 -0
- data/lib/ollama/agent/planner.rb +45 -0
- data/lib/ollama/api_key_pool.rb +61 -0
- data/lib/ollama/attachment.rb +74 -0
- data/lib/ollama/capabilities.rb +1 -1
- data/lib/ollama/chat_response.rb +33 -0
- data/lib/ollama/client/chat/request_preparer.rb +82 -0
- data/lib/ollama/client/chat.rb +58 -68
- data/lib/ollama/client/chat_stream_processor.rb +85 -25
- data/lib/ollama/client/generate/request_preparer.rb +118 -0
- data/lib/ollama/client/generate/response_formatter.rb +95 -0
- data/lib/ollama/client/generate.rb +83 -165
- data/lib/ollama/client/model_management.rb +182 -84
- data/lib/ollama/client/openai_compat.rb +185 -0
- data/lib/ollama/client/raw.rb +66 -0
- data/lib/ollama/client/tool_intent.rb +38 -0
- data/lib/ollama/client/web_search.rb +39 -0
- data/lib/ollama/client.rb +83 -35
- data/lib/ollama/config.rb +122 -17
- data/lib/ollama/embeddings.rb +67 -29
- data/lib/ollama/errors.rb +55 -1
- data/lib/ollama/events.rb +55 -0
- data/lib/ollama/generate_stream_handler.rb +23 -4
- data/lib/ollama/http_error_handler.rb +42 -0
- data/lib/ollama/messages.rb +109 -0
- data/lib/ollama/middleware/cache.rb +73 -0
- data/lib/ollama/middleware/logger.rb +74 -0
- data/lib/ollama/middleware/metrics.rb +84 -0
- data/lib/ollama/middleware/tracing.rb +101 -0
- data/lib/ollama/middleware.rb +39 -0
- data/lib/ollama/model_profile.rb +1 -1
- data/lib/ollama/openai.rb +16 -0
- data/lib/ollama/options.rb +63 -1
- data/lib/ollama/params.rb +139 -0
- data/lib/ollama/parsers/base.rb +24 -0
- data/lib/ollama/parsers/chat.rb +22 -0
- data/lib/ollama/parsers/embeddings.rb +23 -0
- data/lib/ollama/parsers/generate.rb +38 -0
- data/lib/ollama/parsers/list_running.rb +15 -0
- data/lib/ollama/parsers/show_model.rb +14 -0
- data/lib/ollama/parsers/version.rb +15 -0
- data/lib/ollama/pipeline.rb +169 -0
- data/lib/ollama/plugins.rb +83 -0
- data/lib/ollama/policies/auto_pull.rb +69 -0
- data/lib/ollama/policies/base.rb +70 -0
- data/lib/ollama/policies/capability_validation.rb +88 -0
- data/lib/ollama/policies/fallback.rb +57 -0
- data/lib/ollama/policies/rate_limit.rb +115 -0
- data/lib/ollama/policies/repair_json.rb +137 -0
- data/lib/ollama/policies/retry/strategies/exponential.rb +27 -0
- data/lib/ollama/policies/retry/strategies/fixed.rb +27 -0
- data/lib/ollama/policies/retry/strategies/jitter.rb +30 -0
- data/lib/ollama/policies/retry/strategies/linear.rb +27 -0
- data/lib/ollama/policies/retry.rb +152 -0
- data/lib/ollama/policies/schema_repair.rb +180 -0
- data/lib/ollama/policies/timeout.rb +43 -0
- data/lib/ollama/policies.rb +24 -0
- data/lib/ollama/prompt.rb +86 -0
- data/lib/ollama/prompt_adapters/base.rb +3 -2
- data/lib/ollama/prompt_adapters/gemma4.rb +22 -14
- data/lib/ollama/prompts/tool_planner.rb +35 -0
- data/lib/ollama/providers/base.rb +70 -0
- data/lib/ollama/providers/llama_cpp.rb +131 -0
- data/lib/ollama/providers/ollama.rb +54 -0
- data/lib/ollama/providers/openai.rb +134 -0
- data/lib/ollama/providers.rb +28 -0
- data/lib/ollama/rate_limit_handler.rb +48 -0
- data/lib/ollama/request.rb +169 -0
- data/lib/ollama/response.rb +3 -2
- data/lib/ollama/responses/base.rb +11 -0
- data/lib/ollama/responses/chat.rb +11 -0
- data/lib/ollama/responses/embeddings.rb +14 -0
- data/lib/ollama/responses/generate.rb +22 -0
- data/lib/ollama/schema_dsl.rb +97 -0
- data/lib/ollama/schema_validator.rb +90 -53
- data/lib/ollama/schemas/tool_intent.json +15 -0
- data/lib/ollama/schemas/tool_intent.rb +16 -0
- data/lib/ollama/serializers/base.rb +44 -0
- data/lib/ollama/serializers/chat.rb +42 -0
- data/lib/ollama/serializers/embeddings.rb +36 -0
- data/lib/ollama/serializers/generate.rb +47 -0
- data/lib/ollama/streaming_observer.rb +22 -0
- data/lib/ollama/testing.rb +103 -0
- data/lib/ollama/tool/function/parameters/property.rb +72 -0
- data/lib/ollama/tool/function/parameters.rb +101 -0
- data/lib/ollama/tool/function.rb +78 -0
- data/lib/ollama/tool.rb +60 -0
- data/lib/ollama/tool_dsl.rb +93 -0
- data/lib/ollama/tool_intent.rb +19 -0
- data/lib/ollama/transport/base.rb +42 -0
- data/lib/ollama/transport/mock.rb +48 -0
- data/lib/ollama/transport/net_http.rb +76 -0
- data/lib/ollama/transport/request.rb +20 -0
- data/lib/ollama/transport/response.rb +41 -0
- data/lib/ollama/transport.rb +26 -0
- data/lib/ollama/version.rb +1 -1
- data/lib/ollama_client.rb +35 -0
- data/script/live_branch_smoke/chat_exercises.rb +232 -0
- data/script/live_branch_smoke/generation_exercises.rb +128 -0
- data/script/live_branch_smoke/model_exercises.rb +105 -0
- data/script/live_branch_smoke/utility_exercises.rb +375 -0
- data/script/live_branch_smoke.rb +159 -174
- data/test_all_features.rb +315 -0
- metadata +158 -24
- data/.cursor/.gitignore +0 -1
- data/RELEASE_NOTES_v0.2.6.md +0 -41
- data/devagent_proper.rb +0 -430
- data/docs/TESTING.md +0 -508
- data/examples/agent_loop.rb +0 -120
- data/examples/failure_modes/invalid_json_repair.rb +0 -42
- data/examples/production/rails_agent.rb +0 -62
- data/market.jpg +0 -0
- data/print_capabilities.rb +0 -20
- data/schema.json +0 -1
- data/test_tool.rb +0 -26
data/README.md
CHANGED
|
@@ -5,371 +5,303 @@
|
|
|
5
5
|
[](https://www.ruby-lang.org)
|
|
6
6
|
[](LICENSE.txt)
|
|
7
7
|
|
|
8
|
-
> **
|
|
8
|
+
> **The production-safe Ruby AI SDK for Ollama.**
|
|
9
9
|
|
|
10
|
-
|
|
11
|
-
A failure-aware, contract-driven client that covers **all 12 Ollama API endpoints** with production guarantees.
|
|
10
|
+
A failure-aware, contract-driven client that wraps the Ollama API in clean, idiomatic Ruby. Built for correctness, determinism, and zero-magic reliability.
|
|
12
11
|
|
|
13
|
-
|
|
12
|
+
---
|
|
14
13
|
|
|
15
|
-
## Why
|
|
14
|
+
## Why ollama-client?
|
|
16
15
|
|
|
17
|
-
Other Ollama clients give you raw HTTP
|
|
16
|
+
Other Ollama clients give you raw HTTP hashes. This SDK gives you **production guarantees** and a native Ruby developer experience.
|
|
17
|
+
|
|
18
|
+
### 1. Failure-Aware By Design
|
|
18
19
|
|
|
19
20
|
| What goes wrong | What other gems do | What `ollama-client` does |
|
|
20
21
|
|---|---|---|
|
|
21
22
|
| Model isn't downloaded | Raise error | Auto-pull → retry |
|
|
22
23
|
| Ollama server is down | Hang for 60s | Fast-fail instantly |
|
|
23
|
-
| LLM returns broken JSON | Crash your parser | Repair prompt → retry |
|
|
24
|
+
| LLM returns broken JSON | Crash your JSON parser | Repair prompt → retry |
|
|
24
25
|
| Request times out | Raise immediately | Exponential backoff |
|
|
25
|
-
| Schema violation | You find out in
|
|
26
|
+
| Schema violation | You find out in production | `SchemaViolationError` before it reaches your code |
|
|
26
27
|
|
|
27
|
-
|
|
28
|
+
### 2. Ruby Objects Everywhere
|
|
29
|
+
|
|
30
|
+
Stop parsing raw JSON strings or dig-digging through nested string hashes.
|
|
28
31
|
|
|
29
32
|
```ruby
|
|
30
|
-
|
|
33
|
+
# Other gems
|
|
34
|
+
response["message"]["content"]
|
|
35
|
+
response["total_duration"]
|
|
36
|
+
|
|
37
|
+
# ollama-client
|
|
38
|
+
response.message.content
|
|
39
|
+
response.total_duration
|
|
31
40
|
```
|
|
32
41
|
|
|
33
|
-
|
|
42
|
+
---
|
|
34
43
|
|
|
35
|
-
|
|
44
|
+
## Installation
|
|
36
45
|
|
|
37
46
|
```ruby
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
client = Ollama::Client.new
|
|
41
|
-
# model: "llama3.2:3b", timeout: 30, retries: 2, strict_json: true
|
|
47
|
+
bundle add ollama-client
|
|
42
48
|
```
|
|
43
49
|
|
|
44
|
-
|
|
50
|
+
---
|
|
45
51
|
|
|
46
|
-
|
|
52
|
+
## Configuration
|
|
47
53
|
|
|
48
|
-
|
|
49
|
-
response = client.chat(
|
|
50
|
-
messages: [
|
|
51
|
-
{ role: "system", content: "You are a helpful assistant." },
|
|
52
|
-
{ role: "user", content: "What is Ruby?" }
|
|
53
|
-
]
|
|
54
|
-
)
|
|
54
|
+
Set up your client globally (e.g. in a Rails initializer) or construct thread-safe local configs for concurrent background jobs.
|
|
55
55
|
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
|
|
60
|
-
|
|
56
|
+
```ruby
|
|
57
|
+
# config/initializers/ollama.rb
|
|
58
|
+
Ollama.configure do |config|
|
|
59
|
+
config.default_model = "qwen2.5-coder:7b"
|
|
60
|
+
config.base_url = ENV["OLLAMA_URL"] || "http://localhost:11434"
|
|
61
|
+
config.timeout = 30
|
|
62
|
+
config.retries = 2
|
|
63
|
+
end
|
|
61
64
|
```
|
|
62
65
|
|
|
63
|
-
|
|
66
|
+
---
|
|
64
67
|
|
|
65
|
-
|
|
66
|
-
messages = [{ role: "user", content: "What is the weather in London?" }]
|
|
67
|
-
|
|
68
|
-
tools = [
|
|
69
|
-
{
|
|
70
|
-
type: "function",
|
|
71
|
-
function: {
|
|
72
|
-
name: "get_weather",
|
|
73
|
-
description: "Get weather for a city",
|
|
74
|
-
parameters: {
|
|
75
|
-
type: "object",
|
|
76
|
-
properties: { city: { type: "string" } },
|
|
77
|
-
required: ["city"]
|
|
78
|
-
}
|
|
79
|
-
}
|
|
80
|
-
}
|
|
81
|
-
]
|
|
82
|
-
|
|
83
|
-
response = client.chat(messages: messages, tools: tools)
|
|
84
|
-
response.message.tool_calls.first.name # => "get_weather"
|
|
85
|
-
response.message.tool_calls.first.arguments # => { "city" => "London" }
|
|
86
|
-
```
|
|
68
|
+
## Testing Without a Live Server
|
|
87
69
|
|
|
88
|
-
|
|
70
|
+
Don't let your test suite depend on a running Ollama server. `ollama-client` ships with zero-dependency mocking helpers:
|
|
89
71
|
|
|
90
72
|
```ruby
|
|
91
|
-
|
|
92
|
-
|
|
93
|
-
|
|
94
|
-
|
|
95
|
-
|
|
73
|
+
# spec/spec_helper.rb or rails_helper.rb
|
|
74
|
+
require "ollama/testing"
|
|
75
|
+
|
|
76
|
+
RSpec.configure do |config|
|
|
77
|
+
config.include Ollama::Testing
|
|
78
|
+
config.before(:each) { clear_ollama_stubs }
|
|
79
|
+
end
|
|
80
|
+
|
|
81
|
+
# In your spec file
|
|
82
|
+
it "generates a summary" do
|
|
83
|
+
stub_ollama_chat(content: "This is a mocked summary.")
|
|
84
|
+
|
|
85
|
+
client = Ollama::Client.new
|
|
86
|
+
response = client.chat(messages: [{ role: "user", content: "..." }])
|
|
87
|
+
expect(response.content).to eq("This is a mocked summary.")
|
|
88
|
+
end
|
|
96
89
|
```
|
|
97
90
|
|
|
98
|
-
|
|
91
|
+
---
|
|
99
92
|
|
|
100
|
-
|
|
93
|
+
## Core DSLs: Writing Clean Ruby
|
|
101
94
|
|
|
102
|
-
|
|
103
|
-
|
|
95
|
+
### 1. Prompt DSL
|
|
96
|
+
Encapsulate your prompt templates into reusable class objects.
|
|
104
97
|
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
98
|
+
```ruby
|
|
99
|
+
class ExplainCode < Ollama::Prompt
|
|
100
|
+
input :code, :language
|
|
101
|
+
system "You are a senior Ruby engineer."
|
|
102
|
+
user { "Explain this #{language} implementation:\n#{code}" }
|
|
103
|
+
end
|
|
104
|
+
|
|
105
|
+
# Usage:
|
|
106
|
+
prompt = ExplainCode.new(code: "def foo; end", language: "Ruby")
|
|
107
|
+
client.chat(messages: prompt.to_h)
|
|
108
108
|
```
|
|
109
109
|
|
|
110
|
-
|
|
110
|
+
### 2. Tool DSL
|
|
111
|
+
Define type-safe tools that serialize automatically into standard JSON schemas.
|
|
111
112
|
|
|
112
113
|
```ruby
|
|
113
|
-
|
|
114
|
+
class WeatherTool < Ollama::ToolDSL
|
|
115
|
+
description "Get current weather for a city"
|
|
116
|
+
tool_name "get_weather"
|
|
114
117
|
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
keep_alive: "10m", # Keep model loaded
|
|
120
|
-
logprobs: true, # Return log probabilities
|
|
121
|
-
top_logprobs: 5
|
|
122
|
-
)
|
|
123
|
-
```
|
|
118
|
+
input do
|
|
119
|
+
string :city, description: "The name of the city"
|
|
120
|
+
string :unit, optional: true, default: "celsius"
|
|
121
|
+
end
|
|
124
122
|
|
|
125
|
-
|
|
123
|
+
call do
|
|
124
|
+
# Define tool action or handle execution
|
|
125
|
+
end
|
|
126
|
+
end
|
|
126
127
|
|
|
127
|
-
|
|
128
|
-
client.
|
|
129
|
-
|
|
128
|
+
# Usage:
|
|
129
|
+
client.chat(
|
|
130
|
+
messages: [{ role: "user", content: "What is the weather in London?" }],
|
|
131
|
+
tools: [WeatherTool.to_tool_hash]
|
|
132
|
+
)
|
|
130
133
|
```
|
|
131
134
|
|
|
132
|
-
|
|
135
|
+
### 3. Structured Outputs (Schema DSL)
|
|
136
|
+
Force the model to output valid JSON matching your schema structure.
|
|
133
137
|
|
|
134
138
|
```ruby
|
|
135
|
-
|
|
136
|
-
"
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
"action" => { "type" => "string", "enum" => ["search", "calculate", "finish"] },
|
|
140
|
-
"confidence" => { "type" => "number" }
|
|
141
|
-
}
|
|
142
|
-
}
|
|
139
|
+
class TradeSignal < Ollama::SchemaDSL
|
|
140
|
+
string :action, enum: ["BUY", "SELL", "HOLD"]
|
|
141
|
+
number :confidence, description: "Confidence score from 0 to 1"
|
|
142
|
+
end
|
|
143
143
|
|
|
144
|
-
|
|
145
|
-
result
|
|
144
|
+
# Usage:
|
|
145
|
+
result = client.generate(
|
|
146
|
+
prompt: "Analyze the current AAPL price trend.",
|
|
147
|
+
schema: TradeSignal.to_tool_hash # Generates valid JSON schema
|
|
148
|
+
)
|
|
149
|
+
result["action"] # => "BUY"
|
|
146
150
|
result["confidence"] # => 0.95
|
|
147
151
|
```
|
|
148
152
|
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
#### Structured Thinking (Zero-Magic CoT extraction)
|
|
153
|
+
---
|
|
152
154
|
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
> **Note:** Requires a thinking model. Supported defaults: `/deepseek/i`, `/qwen/i`, `/r1/i`.
|
|
155
|
+
## Real-World Rails Patterns
|
|
156
156
|
|
|
157
|
+
### In a Controller (Streaming Chat)
|
|
157
158
|
```ruby
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
"
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
result["final"]["decision"] # => "WAIT"
|
|
159
|
+
class ChatsController < ApplicationController
|
|
160
|
+
def create
|
|
161
|
+
client = Ollama::Client.new
|
|
162
|
+
|
|
163
|
+
response.headers["Content-Type"] = "text/event-stream"
|
|
164
|
+
response.headers["Last-Modified"] = Time.now.httpdate
|
|
165
|
+
|
|
166
|
+
client.chat(
|
|
167
|
+
messages: [{ role: "user", content: params[:message] }],
|
|
168
|
+
hooks: {
|
|
169
|
+
on_token: ->(token) { response.stream.write(token) }
|
|
170
|
+
}
|
|
171
|
+
)
|
|
172
|
+
ensure
|
|
173
|
+
response.stream.close
|
|
174
|
+
end
|
|
175
|
+
end
|
|
176
176
|
```
|
|
177
177
|
|
|
178
|
-
|
|
179
|
-
|
|
178
|
+
### In a Service Object (RAG Embeddings)
|
|
180
179
|
```ruby
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
180
|
+
class DocumentIndexer
|
|
181
|
+
def initialize(document)
|
|
182
|
+
@document = document
|
|
183
|
+
@client = Ollama::Client.new
|
|
184
|
+
end
|
|
185
|
+
|
|
186
|
+
def index!
|
|
187
|
+
vectors = @client.embeddings.embed(
|
|
188
|
+
model: "nomic-embed-text",
|
|
189
|
+
input: @document.content
|
|
190
|
+
)
|
|
191
|
+
|
|
192
|
+
@document.update!(embedding: vectors)
|
|
193
|
+
end
|
|
194
|
+
end
|
|
189
195
|
```
|
|
190
196
|
|
|
191
|
-
|
|
197
|
+
---
|
|
192
198
|
|
|
193
|
-
|
|
199
|
+
## Core API & Endpoint Coverage
|
|
194
200
|
|
|
201
|
+
### Chat (Multi-turn)
|
|
195
202
|
```ruby
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
on_error: ->(err) { warn err.message },
|
|
202
|
-
on_complete: -> { puts "\nDone" }
|
|
203
|
-
}
|
|
204
|
-
)
|
|
205
|
-
|
|
206
|
-
# Stream chat tokens with log probabilities
|
|
207
|
-
client.chat(
|
|
208
|
-
messages: [{ role: "user", content: "Tell me a story" }],
|
|
209
|
-
logprobs: true,
|
|
210
|
-
hooks: {
|
|
211
|
-
# If your block takes 2 args, it receives the logprobs array for that token
|
|
212
|
-
on_token: ->(token, logprobs) {
|
|
213
|
-
print token
|
|
214
|
-
# logprobs is an Array of Hashes, e.g. [{"token"=>"Once", "logprob"=>-0.12}, ...]
|
|
215
|
-
},
|
|
216
|
-
on_complete: -> { puts }
|
|
217
|
-
}
|
|
203
|
+
response = client.chat(
|
|
204
|
+
messages: [
|
|
205
|
+
{ role: "system", content: "You are a helpful assistant." },
|
|
206
|
+
{ role: "user", content: "What is Ruby?" }
|
|
207
|
+
]
|
|
218
208
|
)
|
|
209
|
+
response.message.content # => "Ruby is a dynamic..."
|
|
210
|
+
response.done? # => true
|
|
219
211
|
```
|
|
220
212
|
|
|
221
|
-
###
|
|
213
|
+
### Generate (Prompt -> Completion)
|
|
214
|
+
```ruby
|
|
215
|
+
client.generate(prompt: "Explain blocks in Ruby.")
|
|
216
|
+
# => "Blocks are anonymous closures..."
|
|
217
|
+
```
|
|
222
218
|
|
|
219
|
+
### Thinking Mode (DeepSeek-R1 / Qwen Reasoning)
|
|
223
220
|
```ruby
|
|
224
|
-
client.
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
client.embeddings.embed(model: "nomic-embed-text:latest", input: ["text1", "text2"])
|
|
229
|
-
|
|
230
|
-
# With options
|
|
231
|
-
client.embeddings.embed(
|
|
232
|
-
model: "nomic-embed-text:latest",
|
|
233
|
-
input: "text",
|
|
234
|
-
truncate: true, # Truncate long inputs
|
|
235
|
-
dimensions: 256, # Embedding dimensions
|
|
236
|
-
keep_alive: "5m" # Keep model loaded
|
|
221
|
+
response = client.chat(
|
|
222
|
+
model: "deepseek-r1",
|
|
223
|
+
messages: [{ role: "user", content: "Solve: 2x + 5 = 15" }],
|
|
224
|
+
think: true
|
|
237
225
|
)
|
|
226
|
+
response.message.thinking # => "Subtract 5 from both sides... divide by 2..."
|
|
227
|
+
response.message.content # => "Therefore, x = 5."
|
|
238
228
|
```
|
|
239
229
|
|
|
240
230
|
### Model Management
|
|
241
|
-
|
|
242
231
|
```ruby
|
|
243
|
-
client.list_models
|
|
244
|
-
|
|
245
|
-
client.
|
|
246
|
-
client.list_running # Currently loaded models (aliased as `ps`)
|
|
247
|
-
client.show_model(model: "qwen2.5-coder:7b") # Model details, capabilities
|
|
248
|
-
client.show_model(model: "qwen2.5-coder:7b", verbose: true) # Include model_info
|
|
249
|
-
client.pull("llama3.2:3b") # Download a model
|
|
250
|
-
client.delete_model(model: "old-model") # Remove a model
|
|
251
|
-
client.copy_model(source: "qwen2.5-coder:7b", destination: "qwen2.5-coder:7b-backup")
|
|
252
|
-
client.create_model(model: "my-model", from: "qwen2.5-coder:7b", system: "You are Alpaca")
|
|
253
|
-
client.push_model(model: "user/my-model") # Push to registry
|
|
254
|
-
client.version # => "0.12.6"
|
|
232
|
+
client.list_models # Returns models with capability profiles
|
|
233
|
+
client.pull("qwen3.5:4b") # Pull new model
|
|
234
|
+
client.delete_model(model: "old-model")
|
|
255
235
|
```
|
|
256
236
|
|
|
257
|
-
###
|
|
258
|
-
|
|
259
|
-
Pass via `options:` on `chat` or `generate`:
|
|
260
|
-
|
|
237
|
+
### Web Search & Fetch (Ollama Cloud)
|
|
261
238
|
```ruby
|
|
262
|
-
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
num_predict: 256,
|
|
267
|
-
stop: ["END"],
|
|
268
|
-
presence_penalty: 0.5,
|
|
269
|
-
frequency_penalty: -0.3
|
|
270
|
-
)
|
|
239
|
+
client = Ollama::Client.new(config: Ollama::Config.new.tap do |c|
|
|
240
|
+
c.base_url = "https://ollama.com"
|
|
241
|
+
c.api_key = ENV.fetch("OLLAMA_API_KEY")
|
|
242
|
+
end)
|
|
271
243
|
|
|
272
|
-
client.
|
|
244
|
+
client.web_search(query: "what is ollama?") # => [{ "title" => ..., "url" => ..., "content" => ... }]
|
|
245
|
+
client.web_fetch(url: "https://ollama.com") # => { "title" => ..., "content" => ..., "links" => [...] }
|
|
273
246
|
```
|
|
274
247
|
|
|
275
|
-
|
|
276
|
-
<summary>All supported options</summary>
|
|
248
|
+
---
|
|
277
249
|
|
|
278
|
-
|
|
279
|
-
|---|---|---|
|
|
280
|
-
| `temperature` | Float (0–2) | Sampling temperature |
|
|
281
|
-
| `top_p` | Float (0–1) | Nucleus sampling |
|
|
282
|
-
| `top_k` | Integer | Top-K sampling |
|
|
283
|
-
| `num_ctx` | Integer | Context window size |
|
|
284
|
-
| `num_predict` | Integer | Max tokens to generate |
|
|
285
|
-
| `repeat_penalty` | Float (0–2) | Repeat penalty |
|
|
286
|
-
| `seed` | Integer | Random seed |
|
|
287
|
-
| `stop` | Array | Stop sequences |
|
|
288
|
-
| `tfs_z` | Float | Tail-free sampling |
|
|
289
|
-
| `mirostat` | 0/1/2 | Mirostat sampling mode |
|
|
290
|
-
| `mirostat_tau` | Float | Mirostat target entropy |
|
|
291
|
-
| `mirostat_eta` | Float | Mirostat learning rate |
|
|
292
|
-
| `typical_p` | Float (0–1) | Typical-p sampling |
|
|
293
|
-
| `presence_penalty` | Float (-2–2) | Presence penalty |
|
|
294
|
-
| `frequency_penalty` | Float (-2–2) | Frequency penalty |
|
|
295
|
-
| `num_gpu` | Integer | GPU layers |
|
|
296
|
-
| `num_thread` | Integer | CPU threads |
|
|
297
|
-
| `num_keep` | Integer | Tokens to keep for context |
|
|
298
|
-
|
|
299
|
-
</details>
|
|
300
|
-
|
|
301
|
-
## CLI
|
|
302
|
-
|
|
303
|
-
A strict, JSON-first CLI ships with the gem:
|
|
304
|
-
|
|
305
|
-
```bash
|
|
306
|
-
# Generate text
|
|
307
|
-
ollama-client generate --prompt "Explain Ruby blocks"
|
|
308
|
-
|
|
309
|
-
# Structured output with schema
|
|
310
|
-
echo '{"type":"object","properties":{"category":{"type":"string"}}}' > schema.json
|
|
311
|
-
ollama-client generate --prompt "Classify this" --schema schema.json --json
|
|
312
|
-
|
|
313
|
-
# Stream tokens
|
|
314
|
-
ollama-client generate --prompt "Write a poem" --stream
|
|
315
|
-
|
|
316
|
-
# Embeddings
|
|
317
|
-
ollama-client embed --input "What is Ruby?" --model nomic-embed-text:latest
|
|
318
|
-
|
|
319
|
-
# List models
|
|
320
|
-
ollama-client models
|
|
321
|
-
|
|
322
|
-
# Pull a model
|
|
323
|
-
ollama-client pull llama3.2:3b
|
|
324
|
-
```
|
|
325
|
-
|
|
326
|
-
All errors output as structured JSON to stderr. No hidden behavior.
|
|
250
|
+
## Advanced
|
|
327
251
|
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
252
|
+
### Composable Middleware Pipeline
|
|
253
|
+
```ruby
|
|
254
|
+
client = Ollama::Client.new
|
|
255
|
+
client.use Ollama::Middleware::Logger # Logs requests/responses
|
|
256
|
+
client.use MyCustomMiddleware
|
|
332
257
|
```
|
|
333
258
|
|
|
259
|
+
### Production Policies
|
|
334
260
|
```ruby
|
|
335
|
-
verbose! # Enable HTTP request/response logging
|
|
336
|
-
quiet! # Disable it
|
|
337
|
-
|
|
338
261
|
client = Ollama::Client.new
|
|
339
|
-
client.version # Prints full HTTP request/response to STDERR
|
|
340
|
-
```
|
|
341
262
|
|
|
342
|
-
|
|
263
|
+
# Retry network failures and HTTP 429/5xx with backoff
|
|
264
|
+
client.use Ollama::Policies::Retry, max_attempts: 3
|
|
343
265
|
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
| **Model missing (404)** | Auto-pull → retry your request |
|
|
347
|
-
| **Server unreachable** | Instant `Ollama::Error` — no waiting |
|
|
348
|
-
| **Timeout** | Exponential backoff (`2^attempt` seconds) |
|
|
349
|
-
| **Invalid JSON** | Repair prompt → retry → `InvalidJSONError` if exhausted |
|
|
350
|
-
| **Schema violation** | Repair prompt → retry → `SchemaViolationError` if exhausted |
|
|
351
|
-
| **Streaming error** | `StreamError` raised with Ollama's error message |
|
|
266
|
+
# Auto-pull missing models on 404, then retry
|
|
267
|
+
client.use Ollama::Policies::AutoPull
|
|
352
268
|
|
|
353
|
-
|
|
269
|
+
# Fall back to alternative models on failure
|
|
270
|
+
client.use Ollama::Policies::Fallback, models: %w[gemma4:31b llama3.1:8b]
|
|
271
|
+
```
|
|
354
272
|
|
|
355
|
-
|
|
273
|
+
See `API_CONTRACT.md` → "Policy Middleware" for the full catalog (Timeout, RateLimit,
|
|
274
|
+
CapabilityValidation, RepairJson, SchemaRepair).
|
|
356
275
|
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
4. No silent coercion of malformed JSON — ever
|
|
361
|
-
5. Typed errors over generic exceptions — always
|
|
276
|
+
### Agent Executor (Tool-Calling Loop)
|
|
277
|
+
```ruby
|
|
278
|
+
client = Ollama::Client.new
|
|
362
279
|
|
|
363
|
-
|
|
280
|
+
tools = {
|
|
281
|
+
"get_price" => ->(symbol:) { { symbol: symbol, price: 24500.50 } }
|
|
282
|
+
}
|
|
364
283
|
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
284
|
+
executor = Ollama::Agent::Executor.new(client, tools: tools)
|
|
285
|
+
answer = executor.run(
|
|
286
|
+
system: "You are a helpful trading assistant.",
|
|
287
|
+
user: "What is the price of NIFTY?"
|
|
288
|
+
)
|
|
289
|
+
puts answer
|
|
290
|
+
```
|
|
368
291
|
|
|
369
|
-
|
|
370
|
-
|
|
292
|
+
### OpenAI Compatibility
|
|
293
|
+
```ruby
|
|
294
|
+
require "ollama/openai"
|
|
295
|
+
|
|
296
|
+
client = Ollama::Client.new
|
|
297
|
+
client.openai.chat.completions.create(
|
|
298
|
+
model: "qwen2.5-coder:7b",
|
|
299
|
+
messages: [{ role: "user", content: "hello" }]
|
|
300
|
+
)
|
|
371
301
|
```
|
|
372
302
|
|
|
303
|
+
---
|
|
304
|
+
|
|
373
305
|
## License
|
|
374
306
|
|
|
375
307
|
MIT. See [LICENSE.txt](LICENSE.txt).
|
data/ROADMAP.md
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
# Roadmap
|
|
2
|
+
|
|
3
|
+
## Runtime stabilization
|
|
4
|
+
|
|
5
|
+
1. Stream runtime object + SSE parser
|
|
6
|
+
2. Full HTTP decoupling
|
|
7
|
+
3. Retry policy engine
|
|
8
|
+
4. Connection pooling
|
|
9
|
+
|
|
10
|
+
## Ecosystem expansion
|
|
11
|
+
|
|
12
|
+
1. `ollama-openai`
|
|
13
|
+
2. `ollama-testing`
|
|
14
|
+
3. `ollama-observability`
|
|
15
|
+
4. `ollama-stream`
|
|
16
|
+
5. `ollama-schema`
|
|
17
|
+
6. `ollama-rails`
|
|
18
|
+
|
|
19
|
+
## Enterprise/runtime intelligence
|
|
20
|
+
|
|
21
|
+
- Router/runtime scheduling
|
|
22
|
+
- Cluster routing
|
|
23
|
+
- Health/circuit layers
|