ai-lite 0.6.0 → 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +18 -0
- data/README.md +386 -12
- data/lib/ai_lite/gemini/chat.rb +78 -0
- data/lib/ai_lite/gemini/configuration.rb +19 -0
- data/lib/ai_lite/gemini/models.rb +102 -0
- data/lib/ai_lite/gemini/streaming.rb +148 -0
- data/lib/ai_lite/gemini/youtube.rb +36 -0
- data/lib/ai_lite/gemini.rb +92 -0
- data/lib/ai_lite/openai/configuration.rb +51 -0
- data/lib/ai_lite/openai/models.rb +96 -0
- data/lib/ai_lite/openai/streaming.rb +130 -0
- data/lib/ai_lite/openai.rb +607 -0
- data/lib/ai_lite/version.rb +1 -1
- data/lib/ai_lite.rb +32 -623
- data/test/ai_lite_test.rb +242 -73
- data/test/gemini_models_test.rb +112 -0
- data/test/gemini_streaming_test.rb +166 -0
- data/test/gemini_test.rb +200 -0
- data/test/gemini_youtube_test.rb +57 -0
- data/test/openai_models_test.rb +138 -0
- data/test/openai_streaming_test.rb +189 -0
- data/test/run.rb +3 -0
- metadata +27 -6
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: d320744aa3ea523ee76415327cad74343186881e21e28910b6b13a556d46eef3
|
|
4
|
+
data.tar.gz: e93d793fdeeefb1c8a2498906b48826934eee3981cadeb4f7073005e0140bd41
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 4fa921ce3f81bdaed04e0facab46c8efdb863b98299d9452ba9f6bdac90d3a9f302e5850e067529b7d66f9552ff41945bb75f38257d6515ba761673c0d8f71d5
|
|
7
|
+
data.tar.gz: 0575ce9fb53e2f419940c08a6626a4d528a5661d7cef2e5cbca7c3b9b22587c0288ea4db37b69d6bf03fd8d470c66de68df6515fb226ceb77483b2beb91a90ab
|
data/CHANGELOG.md
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 1.0.0 (unreleased)
|
|
4
|
+
|
|
5
|
+
- Introduce independent `AiLite::OpenAI` and `AiLite::Gemini` providers.
|
|
6
|
+
- Preserve documented legacy OpenAI entry points through `AiLite`, with deprecation
|
|
7
|
+
warnings once per entry point per process. Removal is planned for v2.0.
|
|
8
|
+
- Add OpenAI and Gemini text streaming through `chat_stream`.
|
|
9
|
+
- Add `available_models`, `model_available?`, `configured_models`, and
|
|
10
|
+
`check_configured_models` to both providers; Gemini discovery follows all pages.
|
|
11
|
+
- Add Gemini chat with native structured media input and interaction continuation,
|
|
12
|
+
plus a YouTube URL helper. Existing OpenAI utility methods remain available.
|
|
13
|
+
- Document model-list limitations, provider-specific debug output, and migration.
|
|
14
|
+
- Treat failed/incomplete OpenAI generations as errors even when HTTP status is 200.
|
|
15
|
+
- Handle fragmented UTF-8 and mixed SSE line endings in both stream parsers.
|
|
16
|
+
|
|
17
|
+
Gemini embeddings, image generation, speech, file transcription, additional
|
|
18
|
+
providers, and async convenience methods are deferred beyond v1.
|
data/README.md
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
# AI Lite
|
|
2
2
|
|
|
3
|
-
AI Lite is a pure Ruby, dependency-light client
|
|
3
|
+
AI Lite is a pure Ruby, dependency-light client with separate OpenAI and Gemini providers.
|
|
4
|
+
OpenAI supports the full utility methods below; Gemini supports chat and streaming with text and structured media input.
|
|
4
5
|
|
|
5
6
|
It is built for Rails apps and plain Ruby projects where installing a full OpenAI SDK is too heavy, incompatible, or unnecessary. It is especially useful in legacy Rails apps where Rails, ActiveSupport, Ruby, or HTTP-client dependency constraints make larger client libraries difficult to add.
|
|
6
7
|
|
|
@@ -24,12 +25,229 @@ ai.speak("Read this aloud")
|
|
|
24
25
|
ai.transcribe("tmp/meeting.mp3")
|
|
25
26
|
```
|
|
26
27
|
|
|
28
|
+
## Provider support
|
|
29
|
+
|
|
30
|
+
| Method | OpenAI | Gemini |
|
|
31
|
+
| --- | --- | --- |
|
|
32
|
+
| `chat`, `chat_stream` | Yes | Yes |
|
|
33
|
+
| Four model helpers | Yes | Yes |
|
|
34
|
+
| `youtube` | No | Yes |
|
|
35
|
+
| `moderate`, `embed`, `image`, `speak`, `transcribe` | Yes | Not implemented |
|
|
36
|
+
|
|
37
|
+
Unsupported provider methods are absent and raise Ruby's `NoMethodError` if called.
|
|
38
|
+
Configure providers during application startup; do not mutate shared configuration
|
|
39
|
+
or client headers while requests are running. The gem does not provide automatic
|
|
40
|
+
retries, worker threads, or a background-job queue.
|
|
41
|
+
|
|
42
|
+
## Gemini
|
|
43
|
+
|
|
44
|
+
```ruby
|
|
45
|
+
require "ai_lite"
|
|
46
|
+
|
|
47
|
+
gemini = AiLite::Gemini.new # Reads GEMINI_API_KEY
|
|
48
|
+
result = gemini.chat("Tell me a joke")
|
|
49
|
+
puts result["content"]
|
|
50
|
+
warn result["error"] if result["error"]
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
The lookup order is an explicit `api_key:`, Gemini configuration, then
|
|
54
|
+
`ENV["GEMINI_API_KEY"]`. OpenAI credentials are never used for Gemini. The gem
|
|
55
|
+
does not load `.env` automatically.
|
|
56
|
+
|
|
57
|
+
```ruby
|
|
58
|
+
AiLite::Gemini.configure do |config|
|
|
59
|
+
config.api_key = ENV["GEMINI_API_KEY"]
|
|
60
|
+
config.model = "gemini-3.8-flash"
|
|
61
|
+
config.timeout = 120
|
|
62
|
+
config.max_output_tokens = 2000
|
|
63
|
+
end
|
|
64
|
+
|
|
65
|
+
AiLite::Gemini.chat("Tell me a joke")
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Each provider has independent configuration and a cached client. Constructor
|
|
69
|
+
keywords override configured defaults. Configuration changes reset that provider's
|
|
70
|
+
class-level client; existing instances keep their settings. Gemini uses the
|
|
71
|
+
Interactions API at `https://generativelanguage.googleapis.com/v1beta/interactions`
|
|
72
|
+
with `x-goog-api-key` authentication. Its API version and defaults belong to Gemini.
|
|
73
|
+
|
|
74
|
+
`chat` accepts a string, a native Gemini content hash, or an array of native
|
|
75
|
+
content blocks or interaction steps. Structured inputs pass through unchanged:
|
|
76
|
+
|
|
77
|
+
```ruby
|
|
78
|
+
result = gemini.chat([
|
|
79
|
+
{ type: "video", uri: "https://www.youtube.com/watch?v=YOUR_VIDEO_ID" },
|
|
80
|
+
{ type: "text", text: "Summarize the key points." }
|
|
81
|
+
])
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Google validates media access and model support. This does not download videos,
|
|
85
|
+
upload local files, or make arbitrary video webpages accessible. Message history
|
|
86
|
+
uses Gemini's native `user_input` / `model_output` steps with `content` arrays;
|
|
87
|
+
OpenAI-style role/message arrays are not automatically translated.
|
|
88
|
+
|
|
89
|
+
Multi-turn continuation uses the familiar keyword:
|
|
90
|
+
|
|
91
|
+
```ruby
|
|
92
|
+
first = gemini.chat("My name is William.")
|
|
93
|
+
second = gemini.chat("What is my name?", previous_response_id: first["response_id"])
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
`instructions:` maps to `system_instruction`; `previous_response_id:` maps to
|
|
97
|
+
`previous_interaction_id`; `max_output_tokens:` maps into `generation_config`.
|
|
98
|
+
Gemini-specific settings remain available through `options:`:
|
|
99
|
+
|
|
100
|
+
```ruby
|
|
101
|
+
gemini.chat("Say hello", instructions: "Be concise.", options: {
|
|
102
|
+
store: false,
|
|
103
|
+
generation_config: { temperature: 0.4 }
|
|
104
|
+
})
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
Interactions are stored by default by the provider; `store: false` opts out and
|
|
108
|
+
prevents later server-side continuation from that interaction. See Google's
|
|
109
|
+
[Interactions guide](https://ai.google.dev/gemini-api/docs/interactions-overview)
|
|
110
|
+
and [API reference](https://ai.google.dev/api/interactions-api).
|
|
111
|
+
|
|
112
|
+
Results use the standard `content`, `response_id`, `status`, `error`, and `raw`
|
|
113
|
+
keys. Text is extracted only from `model_output` steps; JSON-looking text is
|
|
114
|
+
parsed. `debug: true` retains the complete response, including non-text output,
|
|
115
|
+
tool calls, and usage. Non-completed interactions (including `requires_action`)
|
|
116
|
+
return an error rather than pretending the answer is complete; no tools are
|
|
117
|
+
executed automatically. Input/API/transport errors return an error envelope;
|
|
118
|
+
constructing a client without a key raises `ArgumentError`.
|
|
119
|
+
|
|
120
|
+
Gemini embeddings and image/speech methods are not implemented yet. Use `chat_stream` for streaming; `chat` rejects
|
|
121
|
+
`stream: true`. Both methods reject `background: true` before sending a request.
|
|
122
|
+
Legacy `AiLite` entry points still use OpenAI.
|
|
123
|
+
|
|
124
|
+
### Gemini model checks
|
|
125
|
+
|
|
126
|
+
```ruby
|
|
127
|
+
gemini = AiLite::Gemini.new
|
|
128
|
+
result = gemini.available_models
|
|
129
|
+
puts result["content"] unless result["error"]
|
|
130
|
+
|
|
131
|
+
gemini.model_available?("gemini-3.8-flash") # true / false / nil on lookup failure
|
|
132
|
+
gemini.configured_models # { "chat" => "gemini-3.8-flash" }
|
|
133
|
+
gemini.check_configured_models # Standard envelope containing a report
|
|
134
|
+
|
|
135
|
+
# All four also work at class level:
|
|
136
|
+
AiLite::Gemini.available_models(debug: true)
|
|
137
|
+
AiLite::Gemini.model_available?("gemini-3.8-flash")
|
|
138
|
+
AiLite::Gemini.configured_models # No key or network required
|
|
139
|
+
AiLite::Gemini.check_configured_models
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
`available_models` follows every page of Gemini's `/v1beta/models` endpoint and
|
|
143
|
+
returns unique model IDs in `content`, with the `models/` resource prefix removed.
|
|
144
|
+
`model_available?` accepts IDs with or without that prefix, uses exact matching
|
|
145
|
+
(no alias resolution), and fetches a fresh complete list. Invalid empty/non-string
|
|
146
|
+
IDs raise `ArgumentError`. Lookup errors return `nil`; inspect `available_models`
|
|
147
|
+
or `check_configured_models` for error details.
|
|
148
|
+
|
|
149
|
+
`configured_models` returns a local snapshot of the instance's settings; its
|
|
150
|
+
class method reports current Gemini configuration. Currently only `chat` is
|
|
151
|
+
configured: `chat_stream` and `youtube` use that same model.
|
|
152
|
+
|
|
153
|
+
`check_configured_models` fetches the list once across its pages, then returns:
|
|
154
|
+
|
|
155
|
+
```ruby
|
|
156
|
+
result["content"]["chat"]
|
|
157
|
+
# { "model" => "gemini-3.8-flash", "usage" => "chat",
|
|
158
|
+
# "available" => true, "error" => nil } # if listed
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
Any failed page makes the entire lookup fail; partial lists are not treated as
|
|
162
|
+
complete. Reports use `available: nil` on failure, with both entry-level and
|
|
163
|
+
outer errors. `debug: true` includes original page responses in
|
|
164
|
+
`raw["pages"]`, preserving model metadata and any received error response.
|
|
165
|
+
These methods make only model-list GET requests, never generation requests.
|
|
166
|
+
They do not verify free-tier eligibility, quota, endpoint compatibility, or
|
|
167
|
+
whether a model is temporarily overloaded. See Google's
|
|
168
|
+
[Models API documentation](https://ai.google.dev/api/models).
|
|
169
|
+
|
|
170
|
+
### YouTube helper
|
|
171
|
+
|
|
172
|
+
```ruby
|
|
173
|
+
result = gemini.youtube(
|
|
174
|
+
"https://www.youtube.com/watch?v=DV69qh0BzoA",
|
|
175
|
+
prompt: "Summarize this video in five bullet points.",
|
|
176
|
+
model: "gemini-3.5-flash-lite"
|
|
177
|
+
)
|
|
178
|
+
puts result["content"]
|
|
179
|
+
warn result["error"] if result["error"]
|
|
180
|
+
```
|
|
181
|
+
|
|
182
|
+
`AiLite::Gemini.youtube(url, prompt: ..., **kwargs)` also works. The helper
|
|
183
|
+
validates one HTTPS YouTube video URL and a nonempty prompt, builds native video
|
|
184
|
+
and text input, and delegates to `chat`. It accepts the same chat keywords,
|
|
185
|
+
including `model:`, `instructions:`, `previous_response_id:`, `max_output_tokens:`,
|
|
186
|
+
`debug:`, and `options:`, and returns the same envelope.
|
|
187
|
+
|
|
188
|
+
Supported URL forms are `youtube.com/watch?v=...` (including `www` and `m`),
|
|
189
|
+
`youtu.be/...`, and YouTube `/shorts/`, `/embed/`, and `/live/` video paths.
|
|
190
|
+
URLs, including timestamp parameters, pass through unchanged. Validation checks
|
|
191
|
+
URL shape and video ID, not whether the video is public or accessible to Gemini.
|
|
192
|
+
The helper builds the same request as this explicit input:
|
|
193
|
+
|
|
194
|
+
```ruby
|
|
195
|
+
gemini.chat([
|
|
196
|
+
{ type: "video", uri: url },
|
|
197
|
+
{ type: "text", text: "Summarize this video." }
|
|
198
|
+
])
|
|
199
|
+
# Equivalent convenience call:
|
|
200
|
+
gemini.youtube(url, prompt: "Summarize this video.")
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Invalid URLs/prompts return an error envelope without a network request. No video
|
|
204
|
+
is downloaded locally. Gemini controls video support, limits, and processing.
|
|
205
|
+
|
|
206
|
+
For streaming, pass the video and text blocks directly to `chat_stream`; the
|
|
207
|
+
`youtube` helper itself uses non-streaming `chat`.
|
|
208
|
+
|
|
209
|
+
### Gemini streaming
|
|
210
|
+
|
|
211
|
+
```ruby
|
|
212
|
+
result = gemini.chat_stream("Tell me a short story", debug: true) do |text|
|
|
213
|
+
print text
|
|
214
|
+
$stdout.flush
|
|
215
|
+
end
|
|
216
|
+
puts
|
|
217
|
+
warn result["error"] if result["error"]
|
|
218
|
+
|
|
219
|
+
# Class-level calls also work:
|
|
220
|
+
AiLite::Gemini.chat_stream("Hello") { |text| print text }
|
|
221
|
+
```
|
|
222
|
+
|
|
223
|
+
`chat_stream` accepts the same input and keywords as `chat` and requires a block.
|
|
224
|
+
It yields model-output text fragments as they arrive, excluding reasoning and
|
|
225
|
+
other non-text events, and returns the standard envelope with the accumulated
|
|
226
|
+
answer and interaction ID. JSON-looking final text is parsed just like `chat`.
|
|
227
|
+
It runs in the calling thread without creating background threads.
|
|
228
|
+
|
|
229
|
+
Gemini's final streaming event may contain only partial interaction metadata.
|
|
230
|
+
With `debug: true`, `raw` therefore contains `{ "interaction" => ..., "events" => [...] }`:
|
|
231
|
+
the latest lifecycle interaction object and every received event, including tool,
|
|
232
|
+
reasoning, and usage events when provided. This differs from non-streaming `chat`,
|
|
233
|
+
whose `raw` is the complete HTTP response. No extra retrieval request is made;
|
|
234
|
+
retaining all events uses additional memory. Without debugging, `raw` is `nil`.
|
|
235
|
+
|
|
236
|
+
HTTP errors, malformed/disconnected streams, and incomplete or failed interactions
|
|
237
|
+
return an error envelope, retaining partial text when available. Always check
|
|
238
|
+
`error`, even with HTTP status `200`. Missing blocks raise `ArgumentError`, and
|
|
239
|
+
exceptions from your block propagate. The configured timeout applies to each
|
|
240
|
+
network read, not the total stream duration. See Google's
|
|
241
|
+
[streaming event reference](https://ai.google.dev/api/interactions-api#InteractionSseStreamEvent).
|
|
242
|
+
|
|
243
|
+
The remaining usage sections below describe the OpenAI provider.
|
|
244
|
+
|
|
27
245
|
## Usage
|
|
28
246
|
|
|
29
247
|
```ruby
|
|
30
248
|
require "ai_lite"
|
|
31
249
|
|
|
32
|
-
ai = AiLite.new
|
|
250
|
+
ai = AiLite::OpenAI.new
|
|
33
251
|
result = ai.chat("Say hello")
|
|
34
252
|
|
|
35
253
|
puts result["content"]
|
|
@@ -40,7 +258,7 @@ By default, the client looks for an API key in `OPENAI_API_KEY`, then falls back
|
|
|
40
258
|
You can also pass the key directly:
|
|
41
259
|
|
|
42
260
|
```ruby
|
|
43
|
-
ai = AiLite.new(api_key: "sk-...")
|
|
261
|
+
ai = AiLite::OpenAI.new(api_key: "sk-...")
|
|
44
262
|
```
|
|
45
263
|
|
|
46
264
|
## Configuration
|
|
@@ -49,7 +267,7 @@ In Rails, configure the default client from an initializer:
|
|
|
49
267
|
|
|
50
268
|
```ruby
|
|
51
269
|
# config/initializers/ai_lite.rb
|
|
52
|
-
AiLite.configure do |config|
|
|
270
|
+
AiLite::OpenAI.configure do |config|
|
|
53
271
|
config.api_key = ENV["OPENAI_API_KEY"]
|
|
54
272
|
config.model = "gpt-5.5"
|
|
55
273
|
config.moderation_model = "omni-moderation-latest"
|
|
@@ -80,16 +298,47 @@ Any value left unset falls back to AI Lite's default:
|
|
|
80
298
|
Then use the configured singleton-style client:
|
|
81
299
|
|
|
82
300
|
```ruby
|
|
83
|
-
result = AiLite.chat("Say hello")
|
|
301
|
+
result = AiLite::OpenAI.chat("Say hello")
|
|
84
302
|
```
|
|
85
303
|
|
|
86
304
|
You can still instantiate a separate client for another token:
|
|
87
305
|
|
|
88
306
|
```ruby
|
|
89
|
-
client = AiLite.new(api_key: "sk-other-token")
|
|
307
|
+
client = AiLite::OpenAI.new(api_key: "sk-other-token")
|
|
90
308
|
result = client.chat("Say hello")
|
|
91
309
|
```
|
|
92
310
|
|
|
311
|
+
## Migrating to provider namespaces
|
|
312
|
+
|
|
313
|
+
Use `AiLite::OpenAI` for OpenAI configuration, class methods, and instances.
|
|
314
|
+
OpenAI owns its API URLs, defaults, configuration, HTTP requests, and response handling.
|
|
315
|
+
Gemini has its own configuration, requests, and response handling; provider capabilities are implemented separately.
|
|
316
|
+
|
|
317
|
+
Existing `AiLite.new`, `AiLite.chat`, and the other top-level methods continue to
|
|
318
|
+
forward to OpenAI, sharing the same configuration and cached client. They emit a
|
|
319
|
+
deprecation warning to stderr once per legacy entry point per process. They will
|
|
320
|
+
remain available throughout v1.x and will be removed in v2.0.
|
|
321
|
+
|
|
322
|
+
```ruby
|
|
323
|
+
# Before
|
|
324
|
+
AiLite.configure { |config| config.api_key = ENV["OPENAI_API_KEY"] }
|
|
325
|
+
ai = AiLite.new
|
|
326
|
+
|
|
327
|
+
# After
|
|
328
|
+
AiLite::OpenAI.configure { |config| config.api_key = ENV["OPENAI_API_KEY"] }
|
|
329
|
+
ai = AiLite::OpenAI.new
|
|
330
|
+
# Class methods work too: AiLite::OpenAI.chat("Hello")
|
|
331
|
+
```
|
|
332
|
+
|
|
333
|
+
`AiLite.new` returns an `AiLite::OpenAI` instance. Compatibility covers documented
|
|
334
|
+
construction, configuration, utility methods, and constants; subclassing the legacy
|
|
335
|
+
`AiLite` class or relying on its exact instance class is not supported. Migrate such
|
|
336
|
+
code to `AiLite::OpenAI`.
|
|
337
|
+
|
|
338
|
+
Existing utility calls keep their methods and result envelopes. Legacy
|
|
339
|
+
constants such as `AiLite::DEFAULT_MODEL` and `AiLite::Configuration` remain
|
|
340
|
+
aliases for their OpenAI counterparts. The gem version stays at `AiLite::VERSION`.
|
|
341
|
+
|
|
93
342
|
## Chat
|
|
94
343
|
|
|
95
344
|
```ruby
|
|
@@ -109,6 +358,7 @@ result = ai.chat(
|
|
|
109
358
|
- `input`
|
|
110
359
|
- `max_output_tokens`
|
|
111
360
|
- optional `instructions`
|
|
361
|
+
- optional `previous_response_id`
|
|
112
362
|
- optional `debug`
|
|
113
363
|
- optional extra `options`
|
|
114
364
|
|
|
@@ -116,23 +366,81 @@ The default model is `gpt-5.5`.
|
|
|
116
366
|
|
|
117
367
|
The OpenAI API URL is fixed to `https://api.openai.com/v1/responses`.
|
|
118
368
|
|
|
369
|
+
### Message-Array Input
|
|
370
|
+
|
|
371
|
+
For a conversation with multiple input messages, pass an array using the OpenAI Responses API message shape:
|
|
372
|
+
|
|
373
|
+
```ruby
|
|
374
|
+
result = ai.chat([
|
|
375
|
+
{ role: "developer", content: "Be concise." },
|
|
376
|
+
{ role: "user", content: "Explain dependency injection." }
|
|
377
|
+
])
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
AI Lite sends the array unchanged as `input`, so you can also use richer Responses API content items when needed.
|
|
381
|
+
|
|
119
382
|
### Multi-Turn Chat
|
|
120
383
|
|
|
121
|
-
Responses include a `response_id
|
|
384
|
+
Responses include a `response_id`. Pass it to the next call as `previous_response_id` to continue the conversation:
|
|
122
385
|
|
|
123
386
|
```ruby
|
|
124
387
|
first = ai.chat("Tell me a short joke.")
|
|
125
388
|
|
|
126
389
|
follow_up = ai.chat(
|
|
127
390
|
"Explain why that is funny.",
|
|
128
|
-
|
|
129
|
-
previous_response_id: first["response_id"]
|
|
130
|
-
}
|
|
391
|
+
previous_response_id: first["response_id"]
|
|
131
392
|
)
|
|
132
393
|
|
|
133
394
|
puts follow_up["content"]
|
|
134
395
|
```
|
|
135
396
|
|
|
397
|
+
Passing `previous_response_id` through `options` remains supported for backward compatibility.
|
|
398
|
+
|
|
399
|
+
### Streaming Chat
|
|
400
|
+
|
|
401
|
+
Use `chat_stream` to receive text as it arrives. It accepts the same keywords as
|
|
402
|
+
`chat`, requires a block, and returns the usual result envelope when the stream
|
|
403
|
+
finishes.
|
|
404
|
+
|
|
405
|
+
```ruby
|
|
406
|
+
ai = AiLite::OpenAI.new
|
|
407
|
+
|
|
408
|
+
result = ai.chat_stream("Explain Ruby blocks in three sentences.") do |text|
|
|
409
|
+
print text
|
|
410
|
+
$stdout.flush
|
|
411
|
+
end
|
|
412
|
+
puts
|
|
413
|
+
|
|
414
|
+
warn result["error"] if result["error"]
|
|
415
|
+
# result["content"] contains the final answer.
|
|
416
|
+
# result["response_id"] can be passed as previous_response_id on the next call.
|
|
417
|
+
```
|
|
418
|
+
|
|
419
|
+
Class-level usage is also supported:
|
|
420
|
+
|
|
421
|
+
```ruby
|
|
422
|
+
AiLite::OpenAI.chat_stream("Say hello") { |text| print text }
|
|
423
|
+
```
|
|
424
|
+
|
|
425
|
+
The block receives text strings, which are not necessarily whole words or tokens.
|
|
426
|
+
JSON-looking final output is parsed in the returned `content`, just like `chat`;
|
|
427
|
+
the block still receives the original text fragments. With `debug: true`, `raw`
|
|
428
|
+
contains the terminal response object (or the available error/last event on failure),
|
|
429
|
+
not a history of every streaming event. Tool arguments, reasoning, and refusal
|
|
430
|
+
events are not yielded as text; the terminal response retains these outputs in
|
|
431
|
+
`raw` when debugging. See the [OpenAI streaming event reference](https://developers.openai.com/api/reference/resources/responses/streaming-events).
|
|
432
|
+
|
|
433
|
+
Always check `result["error"]`: failed, incomplete, malformed, or disconnected
|
|
434
|
+
streams return an error, retaining text already received when available. An HTTP
|
|
435
|
+
status of `200` alone does not mean generation completed successfully. Exceptions
|
|
436
|
+
raised by your block propagate to the caller; calling without a block raises
|
|
437
|
+
`ArgumentError` before making a request.
|
|
438
|
+
|
|
439
|
+
Streaming runs in the calling thread and creates no worker threads. In Rails,
|
|
440
|
+
forwarding chunks to a browser still requires an appropriate streaming response
|
|
441
|
+
or messaging mechanism in the application. The configured read timeout applies
|
|
442
|
+
to individual network reads, not to the total generation duration.
|
|
443
|
+
|
|
136
444
|
## Moderation
|
|
137
445
|
|
|
138
446
|
Use `moderate` to classify user-submitted text or images for potentially harmful content before saving, publishing, or sending it into another AI call.
|
|
@@ -494,9 +802,75 @@ result["content"] # transcript text
|
|
|
494
802
|
result["raw"] # full response body when available
|
|
495
803
|
```
|
|
496
804
|
|
|
805
|
+
## Model Availability
|
|
806
|
+
|
|
807
|
+
These helpers are implemented inside the OpenAI provider. Instance and class-level
|
|
808
|
+
calls are supported; class-level network calls use the configured OpenAI client.
|
|
809
|
+
|
|
810
|
+
```ruby
|
|
811
|
+
ai = AiLite::OpenAI.new
|
|
812
|
+
|
|
813
|
+
result = ai.available_models
|
|
814
|
+
if result["error"]
|
|
815
|
+
warn result["error"]
|
|
816
|
+
else
|
|
817
|
+
puts result["content"] # Array of exact model IDs
|
|
818
|
+
end
|
|
819
|
+
|
|
820
|
+
ai.model_available?("gpt-5.5") # true, false, or nil if the lookup failed
|
|
821
|
+
AiLite::OpenAI.available_models(debug: true) # raw includes the full model list
|
|
822
|
+
```
|
|
823
|
+
|
|
824
|
+
`available_models` makes one authenticated `GET /v1/models` request and returns
|
|
825
|
+
the standard envelope. Empty model lists are valid. `model_available?` makes a
|
|
826
|
+
fresh list request and compares the exact ID, without prefix or alias matching.
|
|
827
|
+
It returns `nil` on a lookup error; use `available_models` or
|
|
828
|
+
`check_configured_models` when you need the error details. Empty or non-string
|
|
829
|
+
IDs raise `ArgumentError`. As with other network methods, constructing a client
|
|
830
|
+
without an API key raises `ArgumentError`.
|
|
831
|
+
|
|
832
|
+
Inspect configured models locally:
|
|
833
|
+
|
|
834
|
+
```ruby
|
|
835
|
+
ai.configured_models
|
|
836
|
+
# => { "chat" => "gpt-5.5", "moderation" => "omni-moderation-latest",
|
|
837
|
+
# "embedding" => "text-embedding-3-small", "image" => "gpt-image-2",
|
|
838
|
+
# "speech" => "gpt-4o-mini-tts", "transcription" => "gpt-transcribe" }
|
|
839
|
+
|
|
840
|
+
AiLite::OpenAI.configured_models # No API key or network request required
|
|
841
|
+
```
|
|
842
|
+
|
|
843
|
+
`configured_models` returns a plain hash of usage names to model IDs. An instance
|
|
844
|
+
reports its own settings, including constructor overrides; the class method
|
|
845
|
+
reports the current provider configuration. Neither includes credentials.
|
|
846
|
+
|
|
847
|
+
Check all configured models with a single list request:
|
|
848
|
+
|
|
849
|
+
```ruby
|
|
850
|
+
result = ai.check_configured_models
|
|
851
|
+
result["content"]["chat"]
|
|
852
|
+
# => { "model" => "gpt-5.5", "usage" => "chat",
|
|
853
|
+
# "available" => true, "error" => nil } # if listed for this key
|
|
854
|
+
|
|
855
|
+
warn result["error"] if result["error"]
|
|
856
|
+
```
|
|
857
|
+
|
|
858
|
+
Each usage has a report entry. A successful lookup returns `available: true` or
|
|
859
|
+
`false`; lookup failures return `available: nil`, an entry-level error, and the
|
|
860
|
+
same error in the outer envelope. `debug: true` includes the original list or
|
|
861
|
+
error response in `raw`. Network checks use the client's configured timeout and
|
|
862
|
+
do not cache results or run automatically during initialization.
|
|
863
|
+
|
|
864
|
+
**These are model-list availability checks only.** They do not run chat, image,
|
|
865
|
+
speech, embedding, moderation, or transcription generation requests and create
|
|
866
|
+
no generated output. A listed model does not guarantee endpoint compatibility,
|
|
867
|
+
sufficient quota, or permission for every operation. See the
|
|
868
|
+
[OpenAI List models reference](https://developers.openai.com/api/reference/resources/models/methods/list).
|
|
869
|
+
|
|
497
870
|
## Return Shape
|
|
498
871
|
|
|
499
|
-
|
|
872
|
+
Generation methods, `available_models`, and `check_configured_models` return a hash envelope.
|
|
873
|
+
`configured_models` returns a plain hash; `model_available?` returns `true`, `false`, or `nil`.
|
|
500
874
|
|
|
501
875
|
Text output:
|
|
502
876
|
|
|
@@ -553,5 +927,5 @@ result = ai.chat("Say hello", debug: true)
|
|
|
553
927
|
Run the test suite:
|
|
554
928
|
|
|
555
929
|
```sh
|
|
556
|
-
ruby -Ilib:test test/
|
|
930
|
+
ruby -Ilib:test test/run.rb
|
|
557
931
|
```
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
class AiLite
|
|
2
|
+
class Gemini
|
|
3
|
+
module Chat
|
|
4
|
+
def chat(message, model: nil, instructions: nil, previous_response_id: nil, max_output_tokens: nil, debug: false, options: {})
|
|
5
|
+
response = nil
|
|
6
|
+
raw = nil
|
|
7
|
+
response_id = nil
|
|
8
|
+
payload = chat_payload(message, model: model, instructions: instructions,
|
|
9
|
+
previous_response_id: previous_response_id,
|
|
10
|
+
max_output_tokens: max_output_tokens, options: options)
|
|
11
|
+
response = post_interaction(payload)
|
|
12
|
+
raw = response.body
|
|
13
|
+
raw = JSON.parse(raw)
|
|
14
|
+
raise "Invalid Gemini response: expected an object" unless raw.is_a?(Hash)
|
|
15
|
+
|
|
16
|
+
response_id = raw["id"]
|
|
17
|
+
unless response.code.to_i.between?(200, 299)
|
|
18
|
+
return result_envelope(status: response.code.to_i, error: error_message(raw),
|
|
19
|
+
response_id: response_id, raw: raw, debug: debug)
|
|
20
|
+
end
|
|
21
|
+
|
|
22
|
+
error = if raw["error"]
|
|
23
|
+
error_message(raw)
|
|
24
|
+
elsif raw["status"] != "completed"
|
|
25
|
+
"Gemini interaction #{raw['status'] || 'has no status'}; inspect raw with debug: true"
|
|
26
|
+
end
|
|
27
|
+
steps = raw["steps"]
|
|
28
|
+
raise "Invalid Gemini response: expected steps" unless steps.is_a?(Array) || error
|
|
29
|
+
|
|
30
|
+
text = Array(steps).flat_map do |step|
|
|
31
|
+
next [] unless step.is_a?(Hash) && step["type"] == "model_output"
|
|
32
|
+
|
|
33
|
+
Array(step["content"]).map do |part|
|
|
34
|
+
part["text"] if part.is_a?(Hash) && part["type"] == "text" && part["text"].is_a?(String)
|
|
35
|
+
end.compact
|
|
36
|
+
end.join.strip
|
|
37
|
+
content = if text.empty?
|
|
38
|
+
nil
|
|
39
|
+
else
|
|
40
|
+
begin
|
|
41
|
+
JSON.parse(text)
|
|
42
|
+
rescue JSON::ParserError
|
|
43
|
+
text
|
|
44
|
+
end
|
|
45
|
+
end
|
|
46
|
+
result_envelope(status: response.code.to_i, content: content, error: error,
|
|
47
|
+
response_id: response_id, raw: raw, debug: debug)
|
|
48
|
+
rescue StandardError => e
|
|
49
|
+
result_envelope(status: response ? response.code.to_i : "unknown", error: e.message,
|
|
50
|
+
response_id: response_id, raw: raw, debug: debug)
|
|
51
|
+
end
|
|
52
|
+
private
|
|
53
|
+
|
|
54
|
+
def chat_payload(message, model:, instructions:, previous_response_id:, max_output_tokens:, options:, stream: false)
|
|
55
|
+
payload = options.each_with_object({}) { |(key, value), result| result[key.to_s] = value }
|
|
56
|
+
if (!stream && payload["stream"]) || payload["background"]
|
|
57
|
+
raise ArgumentError, "Use chat_stream for streaming; background execution is not supported"
|
|
58
|
+
end
|
|
59
|
+
unless message.is_a?(String) || message.is_a?(Hash) || message.is_a?(Array)
|
|
60
|
+
raise ArgumentError, "message must be a string, a Gemini content hash, or an array of Gemini content/steps"
|
|
61
|
+
end
|
|
62
|
+
generation = payload.fetch("generation_config", {})
|
|
63
|
+
raise ArgumentError, "generation_config must be a hash" unless generation.is_a?(Hash)
|
|
64
|
+
|
|
65
|
+
generation = generation.each_with_object({}) { |(key, value), result| result[key.to_s] = value }
|
|
66
|
+
generation["max_output_tokens"] = max_output_tokens || generation["max_output_tokens"] || self.max_output_tokens
|
|
67
|
+
payload.merge!("model" => model || self.model, "input" => message, "generation_config" => generation)
|
|
68
|
+
payload["system_instruction"] = instructions if instructions
|
|
69
|
+
payload["previous_interaction_id"] = previous_response_id if previous_response_id
|
|
70
|
+
payload["stream"] = true if stream
|
|
71
|
+
payload
|
|
72
|
+
end
|
|
73
|
+
end
|
|
74
|
+
|
|
75
|
+
include Chat
|
|
76
|
+
private_constant :Chat
|
|
77
|
+
end
|
|
78
|
+
end
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
class AiLite
|
|
2
|
+
class Gemini
|
|
3
|
+
API_BASE_URL = "https://generativelanguage.googleapis.com/v1beta".freeze
|
|
4
|
+
DEFAULT_MODEL = "gemini-3.8-flash".freeze
|
|
5
|
+
DEFAULT_TIMEOUT = 120
|
|
6
|
+
DEFAULT_MAX_OUTPUT_TOKENS = 2000
|
|
7
|
+
|
|
8
|
+
class Configuration
|
|
9
|
+
attr_accessor :api_key, :model, :timeout, :max_output_tokens
|
|
10
|
+
|
|
11
|
+
def initialize
|
|
12
|
+
@api_key = nil
|
|
13
|
+
@model = DEFAULT_MODEL
|
|
14
|
+
@timeout = DEFAULT_TIMEOUT
|
|
15
|
+
@max_output_tokens = DEFAULT_MAX_OUTPUT_TOKENS
|
|
16
|
+
end
|
|
17
|
+
end
|
|
18
|
+
end
|
|
19
|
+
end
|