ai-lite 0.5.0 → 0.6.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/README.md +100 -5
- data/lib/ai_lite/version.rb +1 -1
- data/lib/ai_lite.rb +153 -4
- data/test/ai_lite_test.rb +236 -4
- metadata +5 -5
checksums.yaml
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
SHA256:
|
|
3
|
-
metadata.gz:
|
|
4
|
-
data.tar.gz:
|
|
3
|
+
metadata.gz: 6f7c5581a41c1371b086b33502200da2ad06cab07cc202efd0f943616effb738
|
|
4
|
+
data.tar.gz: 3a1c897735f7aa945e4cca0fb3bea1a86c3beefc7e823e8a149b78e93ac3b7c8
|
|
5
5
|
SHA512:
|
|
6
|
-
metadata.gz:
|
|
7
|
-
data.tar.gz:
|
|
6
|
+
metadata.gz: 6c5f349cdf43703435f4ab0c08216ce59c73737042ad179ed675339a37238b175dffdc96d5a0014ac0a89dca554e4ed4d4279c417bd13a25fd9dc80d4d9dadc7
|
|
7
|
+
data.tar.gz: 0475ffdba930c5a5dc6c463f78c6abfc7c14306b233b1d05c17e6e6aa5f20bb706271c994e4aec0f2f350f656bcda3c681716751e280968ec8962222cfdea432
|
data/README.md
CHANGED
|
@@ -11,7 +11,7 @@ This gem is intentionally small:
|
|
|
11
11
|
- No Rails dependency
|
|
12
12
|
- No official OpenAI gem dependency
|
|
13
13
|
- No Faraday, HTTParty, ActiveSupport, or connection pool dependency
|
|
14
|
-
- Uses only Ruby stdlib: `Net::HTTP`, `URI`, `JSON`, and `
|
|
14
|
+
- Uses only Ruby stdlib: `Net::HTTP`, `URI`, `JSON`, `Base64`, and `SecureRandom`
|
|
15
15
|
|
|
16
16
|
It is not meant to replace the official OpenAI SDK. It is a small wrapper for projects that only need a few clean interfaces:
|
|
17
17
|
|
|
@@ -21,6 +21,7 @@ ai.moderate("User submitted text")
|
|
|
21
21
|
ai.embed("Text to vectorize")
|
|
22
22
|
ai.image("A simple app icon")
|
|
23
23
|
ai.speak("Read this aloud")
|
|
24
|
+
ai.transcribe("tmp/meeting.mp3")
|
|
24
25
|
```
|
|
25
26
|
|
|
26
27
|
## Usage
|
|
@@ -56,11 +57,26 @@ AiLite.configure do |config|
|
|
|
56
57
|
config.image_model = "gpt-image-2"
|
|
57
58
|
config.speech_model = "gpt-4o-mini-tts"
|
|
58
59
|
config.speech_voice = "alloy"
|
|
60
|
+
config.transcription_model = "gpt-transcribe"
|
|
59
61
|
config.timeout = 120
|
|
60
62
|
config.max_output_tokens = 2000
|
|
61
63
|
end
|
|
62
64
|
```
|
|
63
65
|
|
|
66
|
+
Any value left unset falls back to AI Lite's default:
|
|
67
|
+
|
|
68
|
+
- `model`: `gpt-5.5`
|
|
69
|
+
- `moderation_model`: `omni-moderation-latest`
|
|
70
|
+
- `embedding_model`: `text-embedding-3-small`
|
|
71
|
+
- `image_model`: `gpt-image-2`
|
|
72
|
+
- `speech_model`: `gpt-4o-mini-tts`
|
|
73
|
+
- `speech_voice`: `alloy`
|
|
74
|
+
- `transcription_model`: `gpt-transcribe`
|
|
75
|
+
- `timeout`: `120`
|
|
76
|
+
- `max_output_tokens`: `2000`
|
|
77
|
+
|
|
78
|
+
`timeout` is applied to both the HTTP connection timeout and the HTTP read timeout.
|
|
79
|
+
|
|
64
80
|
Then use the configured singleton-style client:
|
|
65
81
|
|
|
66
82
|
```ruby
|
|
@@ -93,6 +109,7 @@ result = ai.chat(
|
|
|
93
109
|
- `input`
|
|
94
110
|
- `max_output_tokens`
|
|
95
111
|
- optional `instructions`
|
|
112
|
+
- optional `previous_response_id`
|
|
96
113
|
- optional `debug`
|
|
97
114
|
- optional extra `options`
|
|
98
115
|
|
|
@@ -100,23 +117,36 @@ The default model is `gpt-5.5`.
|
|
|
100
117
|
|
|
101
118
|
The OpenAI API URL is fixed to `https://api.openai.com/v1/responses`.
|
|
102
119
|
|
|
120
|
+
### Message-Array Input
|
|
121
|
+
|
|
122
|
+
For a conversation with multiple input messages, pass an array using the OpenAI Responses API message shape:
|
|
123
|
+
|
|
124
|
+
```ruby
|
|
125
|
+
result = ai.chat([
|
|
126
|
+
{ role: "developer", content: "Be concise." },
|
|
127
|
+
{ role: "user", content: "Explain dependency injection." }
|
|
128
|
+
])
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
AI Lite sends the array unchanged as `input`, so you can also use richer Responses API content items when needed.
|
|
132
|
+
|
|
103
133
|
### Multi-Turn Chat
|
|
104
134
|
|
|
105
|
-
Responses include a `response_id
|
|
135
|
+
Responses include a `response_id`. Pass it to the next call as `previous_response_id` to continue the conversation:
|
|
106
136
|
|
|
107
137
|
```ruby
|
|
108
138
|
first = ai.chat("Tell me a short joke.")
|
|
109
139
|
|
|
110
140
|
follow_up = ai.chat(
|
|
111
141
|
"Explain why that is funny.",
|
|
112
|
-
|
|
113
|
-
previous_response_id: first["response_id"]
|
|
114
|
-
}
|
|
142
|
+
previous_response_id: first["response_id"]
|
|
115
143
|
)
|
|
116
144
|
|
|
117
145
|
puts follow_up["content"]
|
|
118
146
|
```
|
|
119
147
|
|
|
148
|
+
Passing `previous_response_id` through `options` remains supported for backward compatibility.
|
|
149
|
+
|
|
120
150
|
## Moderation
|
|
121
151
|
|
|
122
152
|
Use `moderate` to classify user-submitted text or images for potentially harmful content before saving, publishing, or sending it into another AI call.
|
|
@@ -413,6 +443,71 @@ result = ai.speak(
|
|
|
413
443
|
)
|
|
414
444
|
```
|
|
415
445
|
|
|
446
|
+
## Transcription
|
|
447
|
+
|
|
448
|
+
Use `transcribe` to turn an audio file into text.
|
|
449
|
+
|
|
450
|
+
```ruby
|
|
451
|
+
result = ai.transcribe("tmp/meeting.mp3")
|
|
452
|
+
|
|
453
|
+
puts result["content"]
|
|
454
|
+
```
|
|
455
|
+
|
|
456
|
+
By default, `content` is the transcript text:
|
|
457
|
+
|
|
458
|
+
```ruby
|
|
459
|
+
{
|
|
460
|
+
"content" => "Welcome everyone, let's get started.",
|
|
461
|
+
"response_id" => nil,
|
|
462
|
+
"status" => 200,
|
|
463
|
+
"error" => nil,
|
|
464
|
+
"raw" => nil
|
|
465
|
+
}
|
|
466
|
+
```
|
|
467
|
+
|
|
468
|
+
`transcribe` sends a multipart `POST` request to `/v1/audio/transcriptions` with:
|
|
469
|
+
|
|
470
|
+
- `file`
|
|
471
|
+
- `model`
|
|
472
|
+
- optional `language`
|
|
473
|
+
- optional `prompt`
|
|
474
|
+
- optional `response_format`
|
|
475
|
+
- optional `temperature`
|
|
476
|
+
- optional `timestamp_granularities`
|
|
477
|
+
- optional `debug`
|
|
478
|
+
- optional extra `options`
|
|
479
|
+
|
|
480
|
+
The default transcription model is `gpt-transcribe`.
|
|
481
|
+
|
|
482
|
+
Supported local audio extensions are `.flac`, `.m4a`, `.mp3`, `.mp4`, `.mpeg`, `.mpga`, `.ogg`, `.wav`, and `.webm`.
|
|
483
|
+
|
|
484
|
+
Pass `language` in ISO-639-1 format when you know the input language:
|
|
485
|
+
|
|
486
|
+
```ruby
|
|
487
|
+
result = ai.transcribe(
|
|
488
|
+
"tmp/meeting.mp3",
|
|
489
|
+
language: "en"
|
|
490
|
+
)
|
|
491
|
+
```
|
|
492
|
+
|
|
493
|
+
Use `prompt` to provide words, names, or style context that may help the transcription:
|
|
494
|
+
|
|
495
|
+
```ruby
|
|
496
|
+
result = ai.transcribe(
|
|
497
|
+
"tmp/support-call.mp3",
|
|
498
|
+
prompt: "The speakers may mention AI Lite, RubyGems, and Net::HTTP."
|
|
499
|
+
)
|
|
500
|
+
```
|
|
501
|
+
|
|
502
|
+
Pass `debug: true` to include the raw OpenAI response:
|
|
503
|
+
|
|
504
|
+
```ruby
|
|
505
|
+
result = ai.transcribe("tmp/meeting.mp3", debug: true)
|
|
506
|
+
|
|
507
|
+
result["content"] # transcript text
|
|
508
|
+
result["raw"] # full response body when available
|
|
509
|
+
```
|
|
510
|
+
|
|
416
511
|
## Return Shape
|
|
417
512
|
|
|
418
513
|
Methods return a hash envelope.
|
data/lib/ai_lite/version.rb
CHANGED
data/lib/ai_lite.rb
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
require "base64"
|
|
2
2
|
require "json"
|
|
3
3
|
require "net/http"
|
|
4
|
+
require "securerandom"
|
|
4
5
|
require "uri"
|
|
5
6
|
require_relative "ai_lite/version"
|
|
6
7
|
|
|
@@ -13,6 +14,7 @@ class AiLite
|
|
|
13
14
|
DEFAULT_SPEECH_MODEL = "gpt-4o-mini-tts".freeze
|
|
14
15
|
DEFAULT_SPEECH_VOICE = "alloy".freeze
|
|
15
16
|
DEFAULT_SPEECH_FORMAT = "mp3".freeze
|
|
17
|
+
DEFAULT_TRANSCRIPTION_MODEL = "gpt-transcribe".freeze
|
|
16
18
|
DEFAULT_TIMEOUT = 120
|
|
17
19
|
DEFAULT_MAX_OUTPUT_TOKENS = 2000
|
|
18
20
|
IMAGE_MIME_TYPES = {
|
|
@@ -22,9 +24,20 @@ class AiLite
|
|
|
22
24
|
".png" => "image/png",
|
|
23
25
|
".webp" => "image/webp"
|
|
24
26
|
}.freeze
|
|
27
|
+
AUDIO_MIME_TYPES = {
|
|
28
|
+
".flac" => "audio/flac",
|
|
29
|
+
".m4a" => "audio/mp4",
|
|
30
|
+
".mp3" => "audio/mpeg",
|
|
31
|
+
".mp4" => "audio/mp4",
|
|
32
|
+
".mpeg" => "audio/mpeg",
|
|
33
|
+
".mpga" => "audio/mpeg",
|
|
34
|
+
".ogg" => "audio/ogg",
|
|
35
|
+
".wav" => "audio/wav",
|
|
36
|
+
".webm" => "audio/webm"
|
|
37
|
+
}.freeze
|
|
25
38
|
|
|
26
39
|
class Configuration
|
|
27
|
-
attr_accessor :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :timeout, :max_output_tokens
|
|
40
|
+
attr_accessor :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :transcription_model, :timeout, :max_output_tokens
|
|
28
41
|
|
|
29
42
|
def initialize
|
|
30
43
|
@api_key = nil
|
|
@@ -34,6 +47,7 @@ class AiLite
|
|
|
34
47
|
@image_model = DEFAULT_IMAGE_MODEL
|
|
35
48
|
@speech_model = DEFAULT_SPEECH_MODEL
|
|
36
49
|
@speech_voice = DEFAULT_SPEECH_VOICE
|
|
50
|
+
@transcription_model = DEFAULT_TRANSCRIPTION_MODEL
|
|
37
51
|
@timeout = DEFAULT_TIMEOUT
|
|
38
52
|
@max_output_tokens = DEFAULT_MAX_OUTPUT_TOKENS
|
|
39
53
|
end
|
|
@@ -80,14 +94,18 @@ class AiLite
|
|
|
80
94
|
client.speak(text, **kwargs)
|
|
81
95
|
end
|
|
82
96
|
|
|
97
|
+
def transcribe(file_path, **kwargs)
|
|
98
|
+
client.transcribe(file_path, **kwargs)
|
|
99
|
+
end
|
|
100
|
+
|
|
83
101
|
def reset_client!
|
|
84
102
|
@client = nil
|
|
85
103
|
end
|
|
86
104
|
end
|
|
87
105
|
|
|
88
|
-
attr_reader :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :timeout, :max_output_tokens, :headers
|
|
106
|
+
attr_reader :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :transcription_model, :timeout, :max_output_tokens, :headers
|
|
89
107
|
|
|
90
|
-
def initialize(api_key: nil, model: nil, moderation_model: nil, embedding_model: nil, image_model: nil, speech_model: nil, speech_voice: nil, timeout: nil, max_output_tokens: nil)
|
|
108
|
+
def initialize(api_key: nil, model: nil, moderation_model: nil, embedding_model: nil, image_model: nil, speech_model: nil, speech_voice: nil, transcription_model: nil, timeout: nil, max_output_tokens: nil)
|
|
91
109
|
@api_key = api_key || self.class.configuration.api_key || ENV["OPENAI_API_KEY"] || ENV["OPEN_AI_TOKEN"]
|
|
92
110
|
raise ArgumentError, "Missing OpenAI API key" if @api_key.to_s.strip.empty?
|
|
93
111
|
|
|
@@ -97,6 +115,7 @@ class AiLite
|
|
|
97
115
|
@image_model = image_model || self.class.configuration.image_model
|
|
98
116
|
@speech_model = speech_model || self.class.configuration.speech_model
|
|
99
117
|
@speech_voice = speech_voice || self.class.configuration.speech_voice
|
|
118
|
+
@transcription_model = transcription_model || self.class.configuration.transcription_model
|
|
100
119
|
@timeout = timeout || self.class.configuration.timeout
|
|
101
120
|
@max_output_tokens = max_output_tokens || self.class.configuration.max_output_tokens
|
|
102
121
|
@headers = {
|
|
@@ -105,13 +124,14 @@ class AiLite
|
|
|
105
124
|
}
|
|
106
125
|
end
|
|
107
126
|
|
|
108
|
-
def chat(message, model: nil, instructions: nil, max_output_tokens: nil, debug: false, options: {})
|
|
127
|
+
def chat(message, model: nil, instructions: nil, previous_response_id: nil, max_output_tokens: nil, debug: false, options: {})
|
|
109
128
|
payload = options.merge(
|
|
110
129
|
model: model || self.model,
|
|
111
130
|
input: message,
|
|
112
131
|
max_output_tokens: max_output_tokens || self.max_output_tokens
|
|
113
132
|
)
|
|
114
133
|
payload[:instructions] = instructions if instructions
|
|
134
|
+
payload[:previous_response_id] = previous_response_id if previous_response_id
|
|
115
135
|
|
|
116
136
|
extract_content(post(payload), debug: debug)
|
|
117
137
|
rescue => e
|
|
@@ -178,6 +198,24 @@ class AiLite
|
|
|
178
198
|
prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
|
|
179
199
|
end
|
|
180
200
|
|
|
201
|
+
def transcribe(file_path, model: nil, language: nil, prompt: nil, response_format: nil, temperature: nil, timestamp_granularities: nil, debug: false, options: {})
|
|
202
|
+
fields = options.merge(
|
|
203
|
+
model: model || transcription_model
|
|
204
|
+
)
|
|
205
|
+
fields[:language] = language if language
|
|
206
|
+
fields[:prompt] = prompt if prompt
|
|
207
|
+
fields[:response_format] = response_format if response_format
|
|
208
|
+
fields[:temperature] = temperature unless temperature.nil?
|
|
209
|
+
fields[:timestamp_granularities] = timestamp_granularities if timestamp_granularities
|
|
210
|
+
|
|
211
|
+
extract_transcription(
|
|
212
|
+
post_multipart(fields, file_field: audio_file_field(file_path), endpoint: transcription_endpoint),
|
|
213
|
+
debug: debug
|
|
214
|
+
)
|
|
215
|
+
rescue => e
|
|
216
|
+
prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
|
|
217
|
+
end
|
|
218
|
+
|
|
181
219
|
private
|
|
182
220
|
|
|
183
221
|
def post(payload, endpoint: response_endpoint)
|
|
@@ -197,6 +235,21 @@ class AiLite
|
|
|
197
235
|
end
|
|
198
236
|
end
|
|
199
237
|
|
|
238
|
+
def post_multipart(fields, file_field:, endpoint:)
|
|
239
|
+
uri = URI.parse(endpoint)
|
|
240
|
+
boundary = "----AiLiteBoundary#{SecureRandom.hex(16)}"
|
|
241
|
+
request = Net::HTTP::Post.new(uri)
|
|
242
|
+
request["Authorization"] = headers["Authorization"]
|
|
243
|
+
request["Content-Type"] = "multipart/form-data; boundary=#{boundary}"
|
|
244
|
+
request.body = multipart_body(fields, file_field: file_field, boundary: boundary)
|
|
245
|
+
|
|
246
|
+
Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == "https") do |http|
|
|
247
|
+
http.open_timeout = timeout if http.respond_to?(:open_timeout=)
|
|
248
|
+
http.read_timeout = timeout if http.respond_to?(:read_timeout=)
|
|
249
|
+
http.request(request)
|
|
250
|
+
end
|
|
251
|
+
end
|
|
252
|
+
|
|
200
253
|
def response_endpoint
|
|
201
254
|
"#{API_BASE_URL}/responses"
|
|
202
255
|
end
|
|
@@ -217,6 +270,10 @@ class AiLite
|
|
|
217
270
|
"#{API_BASE_URL}/audio/speech"
|
|
218
271
|
end
|
|
219
272
|
|
|
273
|
+
def transcription_endpoint
|
|
274
|
+
"#{API_BASE_URL}/audio/transcriptions"
|
|
275
|
+
end
|
|
276
|
+
|
|
220
277
|
def extract_content(response, debug: false)
|
|
221
278
|
status = response.code.to_i
|
|
222
279
|
parsed_response = JSON.parse(response.body)
|
|
@@ -359,6 +416,31 @@ class AiLite
|
|
|
359
416
|
prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
|
|
360
417
|
end
|
|
361
418
|
|
|
419
|
+
def extract_transcription(response, debug: false)
|
|
420
|
+
status = response.code.to_i
|
|
421
|
+
parsed_response = parse_error_response(response.body)
|
|
422
|
+
|
|
423
|
+
unless success_status?(status)
|
|
424
|
+
return prettify_data(
|
|
425
|
+
status: status,
|
|
426
|
+
error: error_message(parsed_response),
|
|
427
|
+
response_id: parsed_response.is_a?(Hash) ? parsed_response["id"] : nil,
|
|
428
|
+
raw: parsed_response,
|
|
429
|
+
debug: debug
|
|
430
|
+
)
|
|
431
|
+
end
|
|
432
|
+
|
|
433
|
+
prettify_data(
|
|
434
|
+
status: status,
|
|
435
|
+
content: transcription_content(parsed_response),
|
|
436
|
+
response_id: parsed_response.is_a?(Hash) ? parsed_response["id"] : nil,
|
|
437
|
+
raw: parsed_response,
|
|
438
|
+
debug: debug
|
|
439
|
+
)
|
|
440
|
+
rescue => e
|
|
441
|
+
prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
|
|
442
|
+
end
|
|
443
|
+
|
|
362
444
|
def extract_output_text(raw)
|
|
363
445
|
Array(raw["output"]).flat_map do |item|
|
|
364
446
|
next [] unless item.is_a?(Hash) && item["type"] == "message"
|
|
@@ -418,6 +500,24 @@ class AiLite
|
|
|
418
500
|
end
|
|
419
501
|
end
|
|
420
502
|
|
|
503
|
+
def audio_file_field(path)
|
|
504
|
+
raise ArgumentError, "Audio file not found: #{path}" unless File.file?(path)
|
|
505
|
+
|
|
506
|
+
{
|
|
507
|
+
name: "file",
|
|
508
|
+
path: path,
|
|
509
|
+
filename: File.basename(path),
|
|
510
|
+
content_type: audio_mime_type(path)
|
|
511
|
+
}
|
|
512
|
+
end
|
|
513
|
+
|
|
514
|
+
def audio_mime_type(path)
|
|
515
|
+
extension = File.extname(path).downcase
|
|
516
|
+
AUDIO_MIME_TYPES.fetch(extension) do
|
|
517
|
+
raise ArgumentError, "Unsupported audio type for transcription: #{extension}"
|
|
518
|
+
end
|
|
519
|
+
end
|
|
520
|
+
|
|
421
521
|
def moderation_content(raw)
|
|
422
522
|
results = raw["results"]
|
|
423
523
|
return nil unless results.is_a?(Array)
|
|
@@ -455,6 +555,55 @@ class AiLite
|
|
|
455
555
|
}
|
|
456
556
|
end
|
|
457
557
|
|
|
558
|
+
def transcription_content(raw)
|
|
559
|
+
return raw["text"] if raw.is_a?(Hash) && raw.key?("text")
|
|
560
|
+
|
|
561
|
+
raw
|
|
562
|
+
end
|
|
563
|
+
|
|
564
|
+
def multipart_body(fields, file_field:, boundary:)
|
|
565
|
+
body = String.new(encoding: Encoding::BINARY)
|
|
566
|
+
|
|
567
|
+
fields.each do |name, value|
|
|
568
|
+
multipart_field_parts(name, value).each do |field_name, field_value|
|
|
569
|
+
body << "--#{boundary}\r\n".b
|
|
570
|
+
body << "Content-Disposition: form-data; name=\"#{multipart_quote(field_name)}\"\r\n\r\n".b
|
|
571
|
+
body << field_value.to_s.b
|
|
572
|
+
body << "\r\n".b
|
|
573
|
+
end
|
|
574
|
+
end
|
|
575
|
+
|
|
576
|
+
body << "--#{boundary}\r\n".b
|
|
577
|
+
body << "Content-Disposition: form-data; name=\"#{multipart_quote(file_field[:name])}\"; filename=\"#{multipart_quote(file_field[:filename])}\"\r\n".b
|
|
578
|
+
body << "Content-Type: #{file_field[:content_type]}\r\n\r\n".b
|
|
579
|
+
body << File.binread(file_field[:path])
|
|
580
|
+
body << "\r\n--#{boundary}--\r\n".b
|
|
581
|
+
body
|
|
582
|
+
end
|
|
583
|
+
|
|
584
|
+
def multipart_field_parts(name, value)
|
|
585
|
+
return [] if value.nil?
|
|
586
|
+
|
|
587
|
+
if value.is_a?(Array)
|
|
588
|
+
value.map { |item| ["#{name}[]", multipart_value(item)] }
|
|
589
|
+
else
|
|
590
|
+
[[name.to_s, multipart_value(value)]]
|
|
591
|
+
end
|
|
592
|
+
end
|
|
593
|
+
|
|
594
|
+
def multipart_value(value)
|
|
595
|
+
case value
|
|
596
|
+
when Hash
|
|
597
|
+
JSON.generate(value)
|
|
598
|
+
else
|
|
599
|
+
value
|
|
600
|
+
end
|
|
601
|
+
end
|
|
602
|
+
|
|
603
|
+
def multipart_quote(value)
|
|
604
|
+
value.to_s.gsub("\\", "\\\\").gsub("\"", "\\\"").delete("\r\n")
|
|
605
|
+
end
|
|
606
|
+
|
|
458
607
|
def parse_error_response(body)
|
|
459
608
|
JSON.parse(body)
|
|
460
609
|
rescue JSON::ParserError
|
data/test/ai_lite_test.rb
CHANGED
|
@@ -36,6 +36,7 @@ class AiLiteTest < Minitest::Test
|
|
|
36
36
|
image_model: "gpt-image-test",
|
|
37
37
|
speech_model: "gpt-speech-test",
|
|
38
38
|
speech_voice: "verse",
|
|
39
|
+
transcription_model: "gpt-transcribe-test",
|
|
39
40
|
timeout: 10
|
|
40
41
|
)
|
|
41
42
|
|
|
@@ -46,6 +47,7 @@ class AiLiteTest < Minitest::Test
|
|
|
46
47
|
assert_equal "gpt-image-test", client.image_model
|
|
47
48
|
assert_equal "gpt-speech-test", client.speech_model
|
|
48
49
|
assert_equal "verse", client.speech_voice
|
|
50
|
+
assert_equal "gpt-transcribe-test", client.transcription_model
|
|
49
51
|
assert_equal 10, client.timeout
|
|
50
52
|
assert_equal 2000, client.max_output_tokens
|
|
51
53
|
assert_equal "Bearer explicit-key", client.headers["Authorization"]
|
|
@@ -63,6 +65,7 @@ class AiLiteTest < Minitest::Test
|
|
|
63
65
|
config.image_model = "gpt-image-test"
|
|
64
66
|
config.speech_model = "gpt-speech-test"
|
|
65
67
|
config.speech_voice = "verse"
|
|
68
|
+
config.transcription_model = "gpt-transcribe-test"
|
|
66
69
|
config.timeout = 15
|
|
67
70
|
config.max_output_tokens = 750
|
|
68
71
|
end
|
|
@@ -76,6 +79,7 @@ class AiLiteTest < Minitest::Test
|
|
|
76
79
|
assert_equal "gpt-image-test", client.image_model
|
|
77
80
|
assert_equal "gpt-speech-test", client.speech_model
|
|
78
81
|
assert_equal "verse", client.speech_voice
|
|
82
|
+
assert_equal "gpt-transcribe-test", client.transcription_model
|
|
79
83
|
assert_equal 15, client.timeout
|
|
80
84
|
assert_equal 750, client.max_output_tokens
|
|
81
85
|
assert_same client, AiLite.client
|
|
@@ -102,6 +106,7 @@ class AiLiteTest < Minitest::Test
|
|
|
102
106
|
config.image_model = "gpt-image-config"
|
|
103
107
|
config.speech_model = "gpt-speech-config"
|
|
104
108
|
config.speech_voice = "sage"
|
|
109
|
+
config.transcription_model = "gpt-transcribe-config"
|
|
105
110
|
config.timeout = 15
|
|
106
111
|
config.max_output_tokens = 750
|
|
107
112
|
end
|
|
@@ -114,6 +119,7 @@ class AiLiteTest < Minitest::Test
|
|
|
114
119
|
image_model: "gpt-image-explicit",
|
|
115
120
|
speech_model: "gpt-speech-explicit",
|
|
116
121
|
speech_voice: "coral",
|
|
122
|
+
transcription_model: "gpt-transcribe-explicit",
|
|
117
123
|
timeout: 5,
|
|
118
124
|
max_output_tokens: 300
|
|
119
125
|
)
|
|
@@ -125,6 +131,7 @@ class AiLiteTest < Minitest::Test
|
|
|
125
131
|
assert_equal "gpt-image-explicit", client.image_model
|
|
126
132
|
assert_equal "gpt-speech-explicit", client.speech_model
|
|
127
133
|
assert_equal "coral", client.speech_voice
|
|
134
|
+
assert_equal "gpt-transcribe-explicit", client.transcription_model
|
|
128
135
|
assert_equal 5, client.timeout
|
|
129
136
|
assert_equal 300, client.max_output_tokens
|
|
130
137
|
end
|
|
@@ -231,15 +238,13 @@ class AiLiteTest < Minitest::Test
|
|
|
231
238
|
end
|
|
232
239
|
end
|
|
233
240
|
|
|
234
|
-
def
|
|
241
|
+
def test_chat_supports_previous_response_id
|
|
235
242
|
client = AiLite.new(api_key: "token-abc")
|
|
236
243
|
|
|
237
244
|
with_stubbed_http(success_response("Follow-up")) do |captured, _response|
|
|
238
245
|
result = client.chat(
|
|
239
246
|
"Explain why that is funny",
|
|
240
|
-
|
|
241
|
-
previous_response_id: "resp_previous_123"
|
|
242
|
-
}
|
|
247
|
+
previous_response_id: "resp_previous_123"
|
|
243
248
|
)
|
|
244
249
|
payload = JSON.parse(captured[:http].last_request.body)
|
|
245
250
|
|
|
@@ -249,6 +254,53 @@ class AiLiteTest < Minitest::Test
|
|
|
249
254
|
end
|
|
250
255
|
end
|
|
251
256
|
|
|
257
|
+
def test_chat_supports_message_array_input
|
|
258
|
+
client = AiLite.new(api_key: "token-abc")
|
|
259
|
+
messages = [
|
|
260
|
+
{ role: "developer", content: "Be concise." },
|
|
261
|
+
{ role: "user", content: "Say hello." }
|
|
262
|
+
]
|
|
263
|
+
|
|
264
|
+
with_stubbed_http(success_response("Hello!")) do |captured, _response|
|
|
265
|
+
client.chat(messages)
|
|
266
|
+
payload = JSON.parse(captured[:http].last_request.body)
|
|
267
|
+
|
|
268
|
+
assert_equal(
|
|
269
|
+
[
|
|
270
|
+
{ "role" => "developer", "content" => "Be concise." },
|
|
271
|
+
{ "role" => "user", "content" => "Say hello." }
|
|
272
|
+
],
|
|
273
|
+
payload["input"]
|
|
274
|
+
)
|
|
275
|
+
end
|
|
276
|
+
end
|
|
277
|
+
|
|
278
|
+
def test_chat_first_class_previous_response_id_overrides_options
|
|
279
|
+
client = AiLite.new(api_key: "token-abc")
|
|
280
|
+
|
|
281
|
+
with_stubbed_http(success_response("Follow-up")) do |captured, _response|
|
|
282
|
+
client.chat(
|
|
283
|
+
"Continue",
|
|
284
|
+
previous_response_id: "resp_keyword",
|
|
285
|
+
options: { previous_response_id: "resp_options" }
|
|
286
|
+
)
|
|
287
|
+
payload = JSON.parse(captured[:http].last_request.body)
|
|
288
|
+
|
|
289
|
+
assert_equal "resp_keyword", payload["previous_response_id"]
|
|
290
|
+
end
|
|
291
|
+
end
|
|
292
|
+
|
|
293
|
+
def test_chat_keeps_supporting_previous_response_id_through_options
|
|
294
|
+
client = AiLite.new(api_key: "token-abc")
|
|
295
|
+
|
|
296
|
+
with_stubbed_http(success_response("Follow-up")) do |captured, _response|
|
|
297
|
+
client.chat("Continue", options: { previous_response_id: "resp_options" })
|
|
298
|
+
payload = JSON.parse(captured[:http].last_request.body)
|
|
299
|
+
|
|
300
|
+
assert_equal "resp_options", payload["previous_response_id"]
|
|
301
|
+
end
|
|
302
|
+
end
|
|
303
|
+
|
|
252
304
|
def test_extracts_text_from_nested_output_text_items
|
|
253
305
|
client = AiLite.new(api_key: "token-abc")
|
|
254
306
|
body = JSON.generate(
|
|
@@ -785,6 +837,171 @@ class AiLiteTest < Minitest::Test
|
|
|
785
837
|
end
|
|
786
838
|
end
|
|
787
839
|
|
|
840
|
+
def test_transcribe_sends_post_to_audio_transcriptions_with_default_multipart_payload
|
|
841
|
+
client = AiLite.new(api_key: "token-abc")
|
|
842
|
+
|
|
843
|
+
Tempfile.create(["meeting", ".mp3"]) do |file|
|
|
844
|
+
file.binmode
|
|
845
|
+
file.write("fake audio")
|
|
846
|
+
file.flush
|
|
847
|
+
|
|
848
|
+
with_stubbed_http(transcription_response("Hello from the file")) do |captured, _response|
|
|
849
|
+
result = client.transcribe(file.path)
|
|
850
|
+
request = captured[:http].last_request
|
|
851
|
+
|
|
852
|
+
assert_equal "Hello from the file", result["content"]
|
|
853
|
+
assert_nil result["response_id"]
|
|
854
|
+
assert_equal 200, result["status"]
|
|
855
|
+
assert_nil result["error"]
|
|
856
|
+
assert_nil result["raw"]
|
|
857
|
+
assert_equal "api.openai.com", captured[:host]
|
|
858
|
+
assert_equal 443, captured[:port]
|
|
859
|
+
assert_equal true, captured[:use_ssl]
|
|
860
|
+
assert_instance_of Net::HTTP::Post, request
|
|
861
|
+
assert_equal "/v1/audio/transcriptions", request.path
|
|
862
|
+
assert_equal "Bearer token-abc", request["Authorization"]
|
|
863
|
+
assert_match(/\Amultipart\/form-data; boundary=----AiLiteBoundary/, request["Content-Type"])
|
|
864
|
+
assert_multipart_field request.body, "model", "gpt-transcribe"
|
|
865
|
+
assert_multipart_file request.body, "file", File.basename(file.path), "audio/mpeg", "fake audio"
|
|
866
|
+
end
|
|
867
|
+
end
|
|
868
|
+
end
|
|
869
|
+
|
|
870
|
+
def test_transcribe_includes_options_and_optional_fields
|
|
871
|
+
client = AiLite.new(api_key: "token-abc")
|
|
872
|
+
|
|
873
|
+
Tempfile.create(["meeting", ".wav"]) do |file|
|
|
874
|
+
file.binmode
|
|
875
|
+
file.write("fake wav")
|
|
876
|
+
file.flush
|
|
877
|
+
|
|
878
|
+
with_stubbed_http(transcription_response("Detailed transcript")) do |captured, _response|
|
|
879
|
+
client.transcribe(
|
|
880
|
+
file.path,
|
|
881
|
+
model: "gpt-4o-transcribe",
|
|
882
|
+
language: "en",
|
|
883
|
+
prompt: "The speaker may say AI Lite.",
|
|
884
|
+
response_format: "verbose_json",
|
|
885
|
+
temperature: 0,
|
|
886
|
+
timestamp_granularities: ["word", "segment"],
|
|
887
|
+
options: {
|
|
888
|
+
chunking_strategy: "auto",
|
|
889
|
+
include: ["logprobs"],
|
|
890
|
+
metadata: { source: "test" }
|
|
891
|
+
}
|
|
892
|
+
)
|
|
893
|
+
body = captured[:http].last_request.body
|
|
894
|
+
|
|
895
|
+
assert_multipart_field body, "model", "gpt-4o-transcribe"
|
|
896
|
+
assert_multipart_field body, "language", "en"
|
|
897
|
+
assert_multipart_field body, "prompt", "The speaker may say AI Lite."
|
|
898
|
+
assert_multipart_field body, "response_format", "verbose_json"
|
|
899
|
+
assert_multipart_field body, "temperature", "0"
|
|
900
|
+
assert_multipart_field body, "timestamp_granularities[]", "word"
|
|
901
|
+
assert_multipart_field body, "timestamp_granularities[]", "segment"
|
|
902
|
+
assert_multipart_field body, "chunking_strategy", "auto"
|
|
903
|
+
assert_multipart_field body, "include[]", "logprobs"
|
|
904
|
+
assert_multipart_field body, "metadata", JSON.generate("source" => "test")
|
|
905
|
+
assert_multipart_file body, "file", File.basename(file.path), "audio/wav", "fake wav"
|
|
906
|
+
end
|
|
907
|
+
end
|
|
908
|
+
end
|
|
909
|
+
|
|
910
|
+
def test_transcribe_uses_class_level_configured_client
|
|
911
|
+
AiLite.configure do |config|
|
|
912
|
+
config.api_key = "configured-key"
|
|
913
|
+
config.transcription_model = "gpt-transcription-config"
|
|
914
|
+
end
|
|
915
|
+
|
|
916
|
+
Tempfile.create(["meeting", ".m4a"]) do |file|
|
|
917
|
+
file.binmode
|
|
918
|
+
file.write("fake m4a")
|
|
919
|
+
file.flush
|
|
920
|
+
|
|
921
|
+
with_stubbed_http(transcription_response) do |captured, _response|
|
|
922
|
+
AiLite.transcribe(file.path)
|
|
923
|
+
request = captured[:http].last_request
|
|
924
|
+
|
|
925
|
+
assert_multipart_field request.body, "model", "gpt-transcription-config"
|
|
926
|
+
assert_equal "Bearer configured-key", request["Authorization"]
|
|
927
|
+
end
|
|
928
|
+
end
|
|
929
|
+
end
|
|
930
|
+
|
|
931
|
+
def test_transcribe_returns_plain_text_response_formats
|
|
932
|
+
client = AiLite.new(api_key: "token-abc")
|
|
933
|
+
|
|
934
|
+
Tempfile.create(["meeting", ".webm"]) do |file|
|
|
935
|
+
file.binmode
|
|
936
|
+
file.write("fake webm")
|
|
937
|
+
file.flush
|
|
938
|
+
|
|
939
|
+
with_stubbed_http(FakeResponse.new("200", "plain transcript")) do |_captured, _response|
|
|
940
|
+
result = client.transcribe(file.path, response_format: "text")
|
|
941
|
+
|
|
942
|
+
assert_equal "plain transcript", result["content"]
|
|
943
|
+
assert_equal 200, result["status"]
|
|
944
|
+
assert_nil result["error"]
|
|
945
|
+
assert_nil result["raw"]
|
|
946
|
+
end
|
|
947
|
+
end
|
|
948
|
+
end
|
|
949
|
+
|
|
950
|
+
def test_transcribe_debug_true_returns_raw_response
|
|
951
|
+
client = AiLite.new(api_key: "token-abc")
|
|
952
|
+
|
|
953
|
+
Tempfile.create(["meeting", ".ogg"]) do |file|
|
|
954
|
+
file.binmode
|
|
955
|
+
file.write("fake ogg")
|
|
956
|
+
file.flush
|
|
957
|
+
|
|
958
|
+
with_stubbed_http(transcription_response("Debug transcript", usage: { "input_tokens" => 4 })) do |_captured, _response|
|
|
959
|
+
result = client.transcribe(file.path, debug: true)
|
|
960
|
+
|
|
961
|
+
assert_equal "Debug transcript", result["content"]
|
|
962
|
+
assert_equal({ "input_tokens" => 4 }, result["raw"]["usage"])
|
|
963
|
+
end
|
|
964
|
+
end
|
|
965
|
+
end
|
|
966
|
+
|
|
967
|
+
def test_transcribe_http_errors_return_standard_envelope
|
|
968
|
+
client = AiLite.new(api_key: "token-abc")
|
|
969
|
+
body = JSON.generate("error" => { "message" => "Unsupported audio format" })
|
|
970
|
+
|
|
971
|
+
Tempfile.create(["meeting", ".mp3"]) do |file|
|
|
972
|
+
file.binmode
|
|
973
|
+
file.write("fake audio")
|
|
974
|
+
file.flush
|
|
975
|
+
|
|
976
|
+
with_stubbed_http(FakeResponse.new("400", body)) do |_captured, _response|
|
|
977
|
+
result = client.transcribe(file.path)
|
|
978
|
+
|
|
979
|
+
assert_nil result["content"]
|
|
980
|
+
assert_nil result["response_id"]
|
|
981
|
+
assert_equal 400, result["status"]
|
|
982
|
+
assert_equal "Unsupported audio format", result["error"]
|
|
983
|
+
assert_nil result["raw"]
|
|
984
|
+
end
|
|
985
|
+
end
|
|
986
|
+
end
|
|
987
|
+
|
|
988
|
+
def test_transcribe_local_file_errors_return_standard_envelope
|
|
989
|
+
client = AiLite.new(api_key: "token-abc")
|
|
990
|
+
|
|
991
|
+
missing_result = client.transcribe("tmp/missing-audio.mp3")
|
|
992
|
+
assert_nil missing_result["content"]
|
|
993
|
+
assert_equal "unknown", missing_result["status"]
|
|
994
|
+
assert_equal "Audio file not found: tmp/missing-audio.mp3", missing_result["error"]
|
|
995
|
+
|
|
996
|
+
Tempfile.create(["meeting", ".txt"]) do |file|
|
|
997
|
+
result = client.transcribe(file.path)
|
|
998
|
+
|
|
999
|
+
assert_nil result["content"]
|
|
1000
|
+
assert_equal "unknown", result["status"]
|
|
1001
|
+
assert_equal "Unsupported audio type for transcription: .txt", result["error"]
|
|
1002
|
+
end
|
|
1003
|
+
end
|
|
1004
|
+
|
|
788
1005
|
def test_http_errors_return_standard_envelope
|
|
789
1006
|
client = AiLite.new(api_key: "token-abc")
|
|
790
1007
|
body = JSON.generate("error" => { "message" => "Invalid API key" })
|
|
@@ -943,6 +1160,21 @@ class AiLiteTest < Minitest::Test
|
|
|
943
1160
|
FakeResponse.new("200", audio)
|
|
944
1161
|
end
|
|
945
1162
|
|
|
1163
|
+
def transcription_response(text = "Transcribed text", **extra)
|
|
1164
|
+
FakeResponse.new("200", JSON.generate({ "text" => text }.merge(extra)))
|
|
1165
|
+
end
|
|
1166
|
+
|
|
1167
|
+
def assert_multipart_field(body, name, value)
|
|
1168
|
+
assert_includes body, "Content-Disposition: form-data; name=\"#{name}\""
|
|
1169
|
+
assert_includes body, "\r\n\r\n#{value}\r\n"
|
|
1170
|
+
end
|
|
1171
|
+
|
|
1172
|
+
def assert_multipart_file(body, name, filename, content_type, content)
|
|
1173
|
+
assert_includes body, "Content-Disposition: form-data; name=\"#{name}\"; filename=\"#{filename}\""
|
|
1174
|
+
assert_includes body, "Content-Type: #{content_type}"
|
|
1175
|
+
assert_includes body, "\r\n\r\n#{content}\r\n"
|
|
1176
|
+
end
|
|
1177
|
+
|
|
946
1178
|
def with_env(values)
|
|
947
1179
|
originals = {}
|
|
948
1180
|
|
metadata
CHANGED
|
@@ -1,17 +1,18 @@
|
|
|
1
1
|
--- !ruby/object:Gem::Specification
|
|
2
2
|
name: ai-lite
|
|
3
3
|
version: !ruby/object:Gem::Version
|
|
4
|
-
version: 0.
|
|
4
|
+
version: 0.6.1
|
|
5
5
|
platform: ruby
|
|
6
6
|
authors:
|
|
7
7
|
- William Basmayor
|
|
8
8
|
autorequire:
|
|
9
9
|
bindir: bin
|
|
10
10
|
cert_chain: []
|
|
11
|
-
date: 2026-
|
|
11
|
+
date: 2026-10-05 00:00:00.000000000 Z
|
|
12
12
|
dependencies: []
|
|
13
13
|
description: AI Lite is a dependency-light Ruby client for Rails apps and plain Ruby
|
|
14
|
-
projects that need simple OpenAI
|
|
14
|
+
projects that need simple OpenAI chat, moderation, embedding, image, speech, and
|
|
15
|
+
transcription calls.
|
|
15
16
|
email:
|
|
16
17
|
executables: []
|
|
17
18
|
extensions: []
|
|
@@ -45,6 +46,5 @@ requirements: []
|
|
|
45
46
|
rubygems_version: 3.5.22
|
|
46
47
|
signing_key:
|
|
47
48
|
specification_version: 4
|
|
48
|
-
summary: Minimal Ruby client for simple
|
|
49
|
-
API.
|
|
49
|
+
summary: Minimal Ruby client for simple OpenAI API calls.
|
|
50
50
|
test_files: []
|