ai-lite 0.6.1 → 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
data/lib/ai_lite.rb CHANGED
@@ -1,638 +1,46 @@
1
- require "base64"
2
- require "json"
3
- require "net/http"
4
- require "securerandom"
5
- require "uri"
6
1
  require_relative "ai_lite/version"
2
+ require_relative "ai_lite/openai"
3
+ require_relative "ai_lite/gemini"
7
4
 
8
5
  class AiLite
9
- API_BASE_URL = "https://api.openai.com/v1".freeze
10
- DEFAULT_MODEL = "gpt-5.5".freeze
11
- DEFAULT_MODERATION_MODEL = "omni-moderation-latest".freeze
12
- DEFAULT_EMBEDDING_MODEL = "text-embedding-3-small".freeze
13
- DEFAULT_IMAGE_MODEL = "gpt-image-2".freeze
14
- DEFAULT_SPEECH_MODEL = "gpt-4o-mini-tts".freeze
15
- DEFAULT_SPEECH_VOICE = "alloy".freeze
16
- DEFAULT_SPEECH_FORMAT = "mp3".freeze
17
- DEFAULT_TRANSCRIPTION_MODEL = "gpt-transcribe".freeze
18
- DEFAULT_TIMEOUT = 120
19
- DEFAULT_MAX_OUTPUT_TOKENS = 2000
20
- IMAGE_MIME_TYPES = {
21
- ".gif" => "image/gif",
22
- ".jpeg" => "image/jpeg",
23
- ".jpg" => "image/jpeg",
24
- ".png" => "image/png",
25
- ".webp" => "image/webp"
26
- }.freeze
27
- AUDIO_MIME_TYPES = {
28
- ".flac" => "audio/flac",
29
- ".m4a" => "audio/mp4",
30
- ".mp3" => "audio/mpeg",
31
- ".mp4" => "audio/mp4",
32
- ".mpeg" => "audio/mpeg",
33
- ".mpga" => "audio/mpeg",
34
- ".ogg" => "audio/ogg",
35
- ".wav" => "audio/wav",
36
- ".webm" => "audio/webm"
37
- }.freeze
38
-
39
- class Configuration
40
- attr_accessor :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :transcription_model, :timeout, :max_output_tokens
41
-
42
- def initialize
43
- @api_key = nil
44
- @model = DEFAULT_MODEL
45
- @moderation_model = DEFAULT_MODERATION_MODEL
46
- @embedding_model = DEFAULT_EMBEDDING_MODEL
47
- @image_model = DEFAULT_IMAGE_MODEL
48
- @speech_model = DEFAULT_SPEECH_MODEL
49
- @speech_voice = DEFAULT_SPEECH_VOICE
50
- @transcription_model = DEFAULT_TRANSCRIPTION_MODEL
51
- @timeout = DEFAULT_TIMEOUT
52
- @max_output_tokens = DEFAULT_MAX_OUTPUT_TOKENS
53
- end
54
- end
55
-
56
- class << self
57
- def configuration
58
- @configuration ||= Configuration.new
59
- end
60
-
61
- def configure
62
- yield(configuration)
63
- reset_client!
64
- configuration
65
- end
66
-
67
- def reset_configuration!
68
- @configuration = Configuration.new
69
- reset_client!
70
- configuration
71
- end
72
-
73
- def client
74
- @client ||= new
75
- end
76
-
77
- def chat(message, **kwargs)
78
- client.chat(message, **kwargs)
79
- end
80
-
81
- def moderate(input = nil, **kwargs)
82
- client.moderate(input, **kwargs)
83
- end
84
-
85
- def embed(input, **kwargs)
86
- client.embed(input, **kwargs)
87
- end
88
-
89
- def image(prompt, **kwargs)
90
- client.image(prompt, **kwargs)
91
- end
92
-
93
- def speak(text, **kwargs)
94
- client.speak(text, **kwargs)
95
- end
96
-
97
- def transcribe(file_path, **kwargs)
98
- client.transcribe(file_path, **kwargs)
99
- end
100
-
101
- def reset_client!
102
- @client = nil
103
- end
104
- end
105
-
106
- attr_reader :api_key, :model, :moderation_model, :embedding_model, :image_model, :speech_model, :speech_voice, :transcription_model, :timeout, :max_output_tokens, :headers
107
-
108
- def initialize(api_key: nil, model: nil, moderation_model: nil, embedding_model: nil, image_model: nil, speech_model: nil, speech_voice: nil, transcription_model: nil, timeout: nil, max_output_tokens: nil)
109
- @api_key = api_key || self.class.configuration.api_key || ENV["OPENAI_API_KEY"] || ENV["OPEN_AI_TOKEN"]
110
- raise ArgumentError, "Missing OpenAI API key" if @api_key.to_s.strip.empty?
111
-
112
- @model = model || self.class.configuration.model
113
- @moderation_model = moderation_model || self.class.configuration.moderation_model
114
- @embedding_model = embedding_model || self.class.configuration.embedding_model
115
- @image_model = image_model || self.class.configuration.image_model
116
- @speech_model = speech_model || self.class.configuration.speech_model
117
- @speech_voice = speech_voice || self.class.configuration.speech_voice
118
- @transcription_model = transcription_model || self.class.configuration.transcription_model
119
- @timeout = timeout || self.class.configuration.timeout
120
- @max_output_tokens = max_output_tokens || self.class.configuration.max_output_tokens
121
- @headers = {
122
- "Authorization" => "Bearer #{@api_key}",
123
- "Content-Type" => "application/json"
124
- }
125
- end
126
-
127
- def chat(message, model: nil, instructions: nil, previous_response_id: nil, max_output_tokens: nil, debug: false, options: {})
128
- payload = options.merge(
129
- model: model || self.model,
130
- input: message,
131
- max_output_tokens: max_output_tokens || self.max_output_tokens
132
- )
133
- payload[:instructions] = instructions if instructions
134
- payload[:previous_response_id] = previous_response_id if previous_response_id
135
-
136
- extract_content(post(payload), debug: debug)
137
- rescue => e
138
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
139
- end
140
-
141
- def moderate(input = nil, model: nil, text: nil, image_url: nil, image_path: nil, debug: false, options: {})
142
- payload = options.merge(
143
- model: model || moderation_model,
144
- input: moderation_input(input, text: text, image_url: image_url, image_path: image_path)
145
- )
146
-
147
- extract_moderation(post(payload, endpoint: moderation_endpoint), debug: debug)
148
- rescue => e
149
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
150
- end
151
-
152
- def embed(input, model: nil, dimensions: nil, encoding_format: nil, debug: false, options: {})
153
- payload = options.merge(
154
- model: model || embedding_model,
155
- input: input
156
- )
157
- payload[:dimensions] = dimensions if dimensions
158
- payload[:encoding_format] = encoding_format if encoding_format
159
-
160
- extract_embedding(post(payload, endpoint: embedding_endpoint), multiple: input.is_a?(Array), debug: debug)
161
- rescue => e
162
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
163
- end
164
-
165
- def image(prompt, model: nil, size: nil, quality: nil, background: nil, output_format: nil, output_path: nil, debug: false, options: {})
166
- payload = options.merge(
167
- model: model || image_model,
168
- prompt: prompt
169
- )
170
- payload[:size] = size if size
171
- payload[:quality] = quality if quality
172
- payload[:background] = background if background
173
- payload[:output_format] = output_format if output_format
174
-
175
- extract_image(post(payload, endpoint: image_endpoint), output_path: output_path, debug: debug)
176
- rescue => e
177
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
6
+ # Preserve previously public constants while ownership stays with OpenAI.
7
+ OpenAI.constants(false).each do |name|
8
+ const_set(name, OpenAI.const_get(name))
178
9
  end
179
10
 
180
- def speak(text, model: nil, voice: nil, response_format: nil, speed: nil, instructions: nil, output_path: nil, base64: false, debug: false, options: {})
181
- payload = options.merge(
182
- model: model || speech_model,
183
- input: text,
184
- voice: voice || speech_voice
185
- )
186
- payload[:response_format] = response_format if response_format
187
- payload[:speed] = speed if speed
188
- payload[:instructions] = instructions if instructions
189
-
190
- extract_speech(
191
- post(payload, endpoint: speech_endpoint),
192
- output_path: output_path,
193
- base64: base64,
194
- response_format: payload[:response_format] || payload["response_format"] || DEFAULT_SPEECH_FORMAT,
195
- debug: debug
196
- )
197
- rescue => e
198
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
199
- end
200
-
201
- def transcribe(file_path, model: nil, language: nil, prompt: nil, response_format: nil, temperature: nil, timestamp_granularities: nil, debug: false, options: {})
202
- fields = options.merge(
203
- model: model || transcription_model
204
- )
205
- fields[:language] = language if language
206
- fields[:prompt] = prompt if prompt
207
- fields[:response_format] = response_format if response_format
208
- fields[:temperature] = temperature unless temperature.nil?
209
- fields[:timestamp_granularities] = timestamp_granularities if timestamp_granularities
210
-
211
- extract_transcription(
212
- post_multipart(fields, file_field: audio_file_field(file_path), endpoint: transcription_endpoint),
213
- debug: debug
214
- )
215
- rescue => e
216
- prettify_data(status: "unknown", error: e.message, raw: nil, debug: debug)
217
- end
218
-
219
- private
220
-
221
- def post(payload, endpoint: response_endpoint)
222
- uri = URI.parse(endpoint)
223
- request = Net::HTTP::Post.new(uri)
224
-
225
- headers.each do |key, value|
226
- request[key] = value
227
- end
228
-
229
- request.body = JSON.generate(payload)
230
-
231
- Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == "https") do |http|
232
- http.open_timeout = timeout if http.respond_to?(:open_timeout=)
233
- http.read_timeout = timeout if http.respond_to?(:read_timeout=)
234
- http.request(request)
235
- end
236
- end
237
-
238
- def post_multipart(fields, file_field:, endpoint:)
239
- uri = URI.parse(endpoint)
240
- boundary = "----AiLiteBoundary#{SecureRandom.hex(16)}"
241
- request = Net::HTTP::Post.new(uri)
242
- request["Authorization"] = headers["Authorization"]
243
- request["Content-Type"] = "multipart/form-data; boundary=#{boundary}"
244
- request.body = multipart_body(fields, file_field: file_field, boundary: boundary)
245
-
246
- Net::HTTP.start(uri.host, uri.port, use_ssl: uri.scheme == "https") do |http|
247
- http.open_timeout = timeout if http.respond_to?(:open_timeout=)
248
- http.read_timeout = timeout if http.respond_to?(:read_timeout=)
249
- http.request(request)
250
- end
251
- end
252
-
253
- def response_endpoint
254
- "#{API_BASE_URL}/responses"
255
- end
256
-
257
- def moderation_endpoint
258
- "#{API_BASE_URL}/moderations"
259
- end
260
-
261
- def embedding_endpoint
262
- "#{API_BASE_URL}/embeddings"
263
- end
264
-
265
- def image_endpoint
266
- "#{API_BASE_URL}/images/generations"
267
- end
268
-
269
- def speech_endpoint
270
- "#{API_BASE_URL}/audio/speech"
271
- end
272
-
273
- def transcription_endpoint
274
- "#{API_BASE_URL}/audio/transcriptions"
275
- end
276
-
277
- def extract_content(response, debug: false)
278
- status = response.code.to_i
279
- parsed_response = JSON.parse(response.body)
280
-
281
- unless success_status?(status)
282
- return prettify_data(
283
- status: status,
284
- error: error_message(parsed_response),
285
- response_id: parsed_response["id"],
286
- raw: parsed_response,
287
- debug: debug
288
- )
289
- end
290
-
291
- raw_content = extract_output_text(parsed_response)
292
- content = parse_content(raw_content)
293
- prettify_data(
294
- status: status,
295
- content: content,
296
- response_id: parsed_response["id"],
297
- raw: parsed_response,
298
- debug: debug
299
- )
300
- rescue JSON::ParserError => e
301
- prettify_data(status: response_status(response), error: e.message, raw: response&.body, debug: debug)
302
- rescue => e
303
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
304
- end
305
-
306
- def extract_moderation(response, debug: false)
307
- status = response.code.to_i
308
- parsed_response = JSON.parse(response.body)
309
-
310
- unless success_status?(status)
311
- return prettify_data(
312
- status: status,
313
- error: error_message(parsed_response),
314
- response_id: parsed_response["id"],
315
- raw: parsed_response,
316
- debug: debug
317
- )
318
- end
319
-
320
- prettify_data(
321
- status: status,
322
- content: moderation_content(parsed_response),
323
- response_id: parsed_response["id"],
324
- raw: parsed_response,
325
- debug: debug
326
- )
327
- rescue JSON::ParserError => e
328
- prettify_data(status: response_status(response), error: e.message, raw: response&.body, debug: debug)
329
- rescue => e
330
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
331
- end
332
-
333
- def extract_embedding(response, multiple:, debug: false)
334
- status = response.code.to_i
335
- parsed_response = JSON.parse(response.body)
336
-
337
- unless success_status?(status)
338
- return prettify_data(
339
- status: status,
340
- error: error_message(parsed_response),
341
- response_id: parsed_response["id"],
342
- raw: parsed_response,
343
- debug: debug
344
- )
345
- end
11
+ @deprecation_mutex = Mutex.new
12
+ @deprecated_entry_points = {}
346
13
 
347
- prettify_data(
348
- status: status,
349
- content: embedding_content(parsed_response, multiple: multiple),
350
- response_id: parsed_response["id"],
351
- raw: parsed_response,
352
- debug: debug
353
- )
354
- rescue JSON::ParserError => e
355
- prettify_data(status: response_status(response), error: e.message, raw: response&.body, debug: debug)
356
- rescue => e
357
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
358
- end
359
-
360
- def extract_image(response, output_path:, debug: false)
361
- status = response.code.to_i
362
- parsed_response = JSON.parse(response.body)
363
-
364
- unless success_status?(status)
365
- return prettify_data(
366
- status: status,
367
- error: error_message(parsed_response),
368
- response_id: parsed_response["id"],
369
- raw: parsed_response,
370
- debug: debug
371
- )
372
- end
373
-
374
- content = image_content(parsed_response)
375
- write_image_output(output_path, content) if output_path
376
-
377
- prettify_data(
378
- status: status,
379
- content: content,
380
- response_id: parsed_response["id"],
381
- raw: parsed_response,
382
- debug: debug
383
- )
384
- rescue JSON::ParserError => e
385
- prettify_data(status: response_status(response), error: e.message, raw: response&.body, debug: debug)
386
- rescue => e
387
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
388
- end
389
-
390
- def extract_speech(response, output_path:, base64:, response_format:, debug: false)
391
- status = response.code.to_i
392
-
393
- unless success_status?(status)
394
- parsed_response = parse_error_response(response.body)
395
-
396
- return prettify_data(
397
- status: status,
398
- error: error_message(parsed_response),
399
- response_id: parsed_response.is_a?(Hash) ? parsed_response["id"] : nil,
400
- raw: parsed_response,
401
- debug: debug
402
- )
403
- end
404
-
405
- audio = response.body
406
- File.binwrite(output_path, audio) if output_path
407
-
408
- prettify_data(
409
- status: status,
410
- content: speech_content(audio, output_path: output_path, base64: base64, response_format: response_format),
411
- response_id: nil,
412
- raw: audio,
413
- debug: debug
414
- )
415
- rescue => e
416
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
417
- end
418
-
419
- def extract_transcription(response, debug: false)
420
- status = response.code.to_i
421
- parsed_response = parse_error_response(response.body)
422
-
423
- unless success_status?(status)
424
- return prettify_data(
425
- status: status,
426
- error: error_message(parsed_response),
427
- response_id: parsed_response.is_a?(Hash) ? parsed_response["id"] : nil,
428
- raw: parsed_response,
429
- debug: debug
430
- )
431
- end
432
-
433
- prettify_data(
434
- status: status,
435
- content: transcription_content(parsed_response),
436
- response_id: parsed_response.is_a?(Hash) ? parsed_response["id"] : nil,
437
- raw: parsed_response,
438
- debug: debug
439
- )
440
- rescue => e
441
- prettify_data(status: response_status(response), error: e.message, raw: nil, debug: debug)
442
- end
443
-
444
- def extract_output_text(raw)
445
- Array(raw["output"]).flat_map do |item|
446
- next [] unless item.is_a?(Hash) && item["type"] == "message"
447
-
448
- Array(item["content"]).map do |content|
449
- next unless content.is_a?(Hash) && content["type"] == "output_text"
450
-
451
- content["text"]
452
- end.compact
453
- end.join.strip
454
- end
455
-
456
- def parse_content(raw_text)
457
- return nil if raw_text.nil? || raw_text.empty?
458
-
459
- JSON.parse(raw_text)
460
- rescue JSON::ParserError
461
- raw_text
462
- end
463
-
464
- def moderation_input(input, text:, image_url:, image_path:)
465
- unless text || image_url || image_path
466
- raise ArgumentError, "Missing moderation input" if input.nil?
467
-
468
- return input
469
- end
470
-
471
- items = []
472
- items.concat(Array(input).map { |value| moderation_input_item(value) }) unless input.nil?
473
- items << { type: "text", text: text } if text
474
- items << { type: "image_url", image_url: { url: image_url } } if image_url
475
- items << { type: "image_url", image_url: { url: image_data_url(image_path) } } if image_path
476
- raise ArgumentError, "Missing moderation input" if items.empty?
477
-
478
- items
479
- end
480
-
481
- def moderation_input_item(value)
482
- case value
483
- when String
484
- { type: "text", text: value }
485
- when Hash
486
- value
487
- else
488
- raise ArgumentError, "Unsupported moderation input item: #{value.class}"
489
- end
490
- end
491
-
492
- def image_data_url(path)
493
- mime_type = image_mime_type(path)
494
- "data:#{mime_type};base64,#{Base64.strict_encode64(File.binread(path))}"
495
- end
496
-
497
- def image_mime_type(path)
498
- IMAGE_MIME_TYPES.fetch(File.extname(path).downcase) do
499
- raise ArgumentError, "Unsupported image type for moderation: #{File.extname(path)}"
500
- end
501
- end
502
-
503
- def audio_file_field(path)
504
- raise ArgumentError, "Audio file not found: #{path}" unless File.file?(path)
505
-
506
- {
507
- name: "file",
508
- path: path,
509
- filename: File.basename(path),
510
- content_type: audio_mime_type(path)
511
- }
512
- end
513
-
514
- def audio_mime_type(path)
515
- extension = File.extname(path).downcase
516
- AUDIO_MIME_TYPES.fetch(extension) do
517
- raise ArgumentError, "Unsupported audio type for transcription: #{extension}"
518
- end
519
- end
520
-
521
- def moderation_content(raw)
522
- results = raw["results"]
523
- return nil unless results.is_a?(Array)
524
-
525
- results.length == 1 ? results.first : results
526
- end
527
-
528
- def embedding_content(raw, multiple:)
529
- embeddings = Array(raw["data"]).map do |item|
530
- item["embedding"] if item.is_a?(Hash)
531
- end.compact
532
-
533
- multiple ? embeddings : embeddings.first
534
- end
535
-
536
- def image_content(raw)
537
- image = Array(raw["data"]).find { |item| item.is_a?(Hash) && item["b64_json"] }
538
- image && image["b64_json"]
539
- end
540
-
541
- def write_image_output(path, content)
542
- raise "No image data returned" if content.to_s.empty?
543
-
544
- File.binwrite(path, Base64.decode64(content))
545
- end
546
-
547
- def speech_content(audio, output_path:, base64:, response_format:)
548
- return Base64.strict_encode64(audio) if base64
549
- return audio unless output_path
550
-
551
- {
552
- "path" => output_path,
553
- "bytes" => audio.bytesize,
554
- "format" => response_format
555
- }
556
- end
557
-
558
- def transcription_content(raw)
559
- return raw["text"] if raw.is_a?(Hash) && raw.key?("text")
560
-
561
- raw
562
- end
563
-
564
- def multipart_body(fields, file_field:, boundary:)
565
- body = String.new(encoding: Encoding::BINARY)
566
-
567
- fields.each do |name, value|
568
- multipart_field_parts(name, value).each do |field_name, field_value|
569
- body << "--#{boundary}\r\n".b
570
- body << "Content-Disposition: form-data; name=\"#{multipart_quote(field_name)}\"\r\n\r\n".b
571
- body << field_value.to_s.b
572
- body << "\r\n".b
14
+ class << self
15
+ def new(**kwargs)
16
+ warn_legacy_entry_point(:new)
17
+ OpenAI.new(**kwargs)
18
+ end
19
+
20
+ %i[configuration configure reset_configuration! client reset_client!
21
+ chat chat_stream moderate embed image speak transcribe
22
+ available_models model_available? configured_models check_configured_models].each do |method_name|
23
+ define_method(method_name) do |*args, **kwargs, &block|
24
+ warn_legacy_entry_point(method_name)
25
+ if kwargs.empty?
26
+ OpenAI.public_send(method_name, *args, &block)
27
+ else
28
+ OpenAI.public_send(method_name, *args, **kwargs, &block)
29
+ end
573
30
  end
574
31
  end
575
32
 
576
- body << "--#{boundary}\r\n".b
577
- body << "Content-Disposition: form-data; name=\"#{multipart_quote(file_field[:name])}\"; filename=\"#{multipart_quote(file_field[:filename])}\"\r\n".b
578
- body << "Content-Type: #{file_field[:content_type]}\r\n\r\n".b
579
- body << File.binread(file_field[:path])
580
- body << "\r\n--#{boundary}--\r\n".b
581
- body
582
- end
33
+ private
583
34
 
584
- def multipart_field_parts(name, value)
585
- return [] if value.nil?
35
+ def warn_legacy_entry_point(method_name)
36
+ @deprecation_mutex.synchronize do
37
+ return if @deprecated_entry_points[method_name]
586
38
 
587
- if value.is_a?(Array)
588
- value.map { |item| ["#{name}[]", multipart_value(item)] }
589
- else
590
- [[name.to_s, multipart_value(value)]]
591
- end
592
- end
593
-
594
- def multipart_value(value)
595
- case value
596
- when Hash
597
- JSON.generate(value)
598
- else
599
- value
600
- end
601
- end
602
-
603
- def multipart_quote(value)
604
- value.to_s.gsub("\\", "\\\\").gsub("\"", "\\\"").delete("\r\n")
605
- end
606
-
607
- def parse_error_response(body)
608
- JSON.parse(body)
609
- rescue JSON::ParserError
610
- body
611
- end
612
-
613
- def success_status?(status)
614
- status >= 200 && status < 300
615
- end
616
-
617
- def error_message(raw)
618
- if raw.is_a?(Hash)
619
- raw.dig("error", "message") || raw["error"] || raw["message"] || raw.to_s
620
- else
621
- raw.to_s
39
+ warn "[ai-lite] AiLite.#{method_name} is deprecated. " \
40
+ "Use AiLite::OpenAI.#{method_name} instead. " \
41
+ "The legacy entry point will be removed in v2.0."
42
+ @deprecated_entry_points[method_name] = true
43
+ end
622
44
  end
623
45
  end
624
-
625
- def response_status(response)
626
- response&.code&.to_i || "unknown"
627
- end
628
-
629
- def prettify_data(status:, content: nil, error: nil, response_id: nil, raw:, debug: false)
630
- {
631
- "content" => content,
632
- "response_id" => response_id,
633
- "status" => status,
634
- "error" => error,
635
- "raw" => debug ? raw : nil
636
- }
637
- end
638
46
  end