zer0-image-generator 0.6.0 → 0.8.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. checksums.yaml +4 -4
  2. data/CHANGELOG.md +20 -0
  3. data/README.md +106 -2
  4. data/lib/zer0_image_generator/abc/style_pack.rb +145 -0
  5. data/lib/zer0_image_generator/abc.rb +75 -0
  6. data/lib/zer0_image_generator/all.rb +57 -0
  7. data/lib/zer0_image_generator/claude/client.rb +291 -0
  8. data/lib/zer0_image_generator/claude/orchestration.rb +128 -0
  9. data/lib/zer0_image_generator/cli.rb +318 -0
  10. data/lib/zer0_image_generator/config.rb +192 -0
  11. data/lib/zer0_image_generator/constants.rb +177 -0
  12. data/lib/zer0_image_generator/content.rb +379 -0
  13. data/lib/zer0_image_generator/engine.rb +21 -0
  14. data/lib/zer0_image_generator/freesvg/cache.rb +167 -0
  15. data/lib/zer0_image_generator/freesvg/client.rb +467 -0
  16. data/lib/zer0_image_generator/http.rb +264 -0
  17. data/lib/zer0_image_generator/library.rb +201 -0
  18. data/lib/zer0_image_generator/logging.rb +236 -0
  19. data/lib/zer0_image_generator/preview_generator.py +1072 -100
  20. data/lib/zer0_image_generator/prompt.rb +48 -0
  21. data/lib/zer0_image_generator/providers/base.rb +152 -0
  22. data/lib/zer0_image_generator/providers/gemini.rb +60 -0
  23. data/lib/zer0_image_generator/providers/local.rb +90 -0
  24. data/lib/zer0_image_generator/providers/openai.rb +102 -0
  25. data/lib/zer0_image_generator/providers/stability.rb +64 -0
  26. data/lib/zer0_image_generator/providers/xai.rb +81 -0
  27. data/lib/zer0_image_generator/providers/xai_auth.rb +270 -0
  28. data/lib/zer0_image_generator/providers.rb +22 -0
  29. data/lib/zer0_image_generator/runner.rb +573 -0
  30. data/lib/zer0_image_generator/settings.rb +386 -0
  31. data/lib/zer0_image_generator/stats.rb +61 -0
  32. data/lib/zer0_image_generator/support/py_random.rb +189 -0
  33. data/lib/zer0_image_generator/svg/banner_seed.rb +241 -0
  34. data/lib/zer0_image_generator/svg/generators/flowfield.rb +154 -0
  35. data/lib/zer0_image_generator/svg/generators/invaders.rb +125 -0
  36. data/lib/zer0_image_generator/svg/generators/lowpoly.rb +254 -0
  37. data/lib/zer0_image_generator/svg/generators/lsystem.rb +228 -0
  38. data/lib/zer0_image_generator/svg/generators/mandala.rb +155 -0
  39. data/lib/zer0_image_generator/svg/generators/pixelquest.rb +144 -0
  40. data/lib/zer0_image_generator/svg/generators/starmap.rb +179 -0
  41. data/lib/zer0_image_generator/svg/lint.rb +400 -0
  42. data/lib/zer0_image_generator/svg/local_renderer.rb +606 -0
  43. data/lib/zer0_image_generator/svg/pixel_kit.rb +167 -0
  44. data/lib/zer0_image_generator/svg/rasterizer.rb +196 -0
  45. data/lib/zer0_image_generator/svg/sanitizer.rb +159 -0
  46. data/lib/zer0_image_generator/version.rb +1 -1
  47. metadata +43 -2
@@ -0,0 +1,48 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "content"
4
+
5
+ module Zer0ImageGenerator
6
+ # Prompt layer: turn a ContentFile plus resolved Settings into the text prompt
7
+ # the image model receives. Ported from the oracle
8
+ # (lib/zer0_image_generator/preview_generator.py). Wording is carried over
9
+ # verbatim from the original engine and must not drift — the committed showcase
10
+ # briefs were produced from these exact strings.
11
+ module Prompt
12
+ # Fallback used by build_enhance_prompt when settings.enhance_prompt is
13
+ # empty. Kept byte-identical to the oracle's DEFAULT_ENHANCE_PROMPT.
14
+ DEFAULT_ENHANCE_PROMPT =
15
+ "Improve this preview banner image: fix any misspelled, garbled, or " \
16
+ "incorrect text so it reads clearly and accurately. Sharpen visual details " \
17
+ "and improve composition while preserving the original art style, color " \
18
+ "palette, and theme. Ensure the image is clean and professional."
19
+
20
+ # Template prompt — wording carried over from the original engine.
21
+ def self.build_prompt(cf, settings)
22
+ parts = ["Create a blog preview banner image for an article titled '#{cf.title}'."]
23
+ parts << "The article is about: #{cf.description}." unless cf.description.empty?
24
+ parts << "Categories: #{cf.categories}." unless cf.categories.empty?
25
+ # First 500 characters, whitespace-collapsed — Content.collapse_whitespace
26
+ # reproduces CPython's `re.sub(r"\s+", " ", ...).strip()` exactly.
27
+ excerpt = Content.collapse_whitespace(cf.content[0, 500] || "")
28
+ parts << "Key themes from content: #{excerpt}" unless excerpt.empty?
29
+ parts << "Art style: #{settings.style}."
30
+ unless settings.style_modifiers.empty?
31
+ parts << "Additional style: #{settings.style_modifiers}."
32
+ end
33
+ parts << "The image should be suitable as a wide blog header/banner image with " \
34
+ "clean composition. No text or words in the image."
35
+ parts.join(" ")
36
+ end
37
+
38
+ def self.build_enhance_prompt(cf, settings)
39
+ prompt = settings.enhance_prompt.empty? ? DEFAULT_ENHANCE_PROMPT : settings.enhance_prompt
40
+ unless cf.title.empty?
41
+ prompt += " Context: This is a preview banner for an article titled '#{cf.title}'."
42
+ end
43
+ prompt += " Article topic: #{cf.description}." unless cf.description.empty?
44
+ prompt += " Maintain the #{settings.style} artistic style."
45
+ prompt
46
+ end
47
+ end
48
+ end
@@ -0,0 +1,152 @@
1
+ # frozen_string_literal: true
2
+
3
+ require "base64"
4
+ require "pathname"
5
+
6
+ # Port of the provider-layer scaffolding from preview_generator.py
7
+ # (ImageResult, EditUnsupported, Provider, RunContext, adapt_openai_size_quality,
8
+ # and the shared _write_image_payload helper). The concrete providers live in
9
+ # sibling files; this one carries only what they all share.
10
+ #
11
+ # Cross-slice dependencies are referenced by their natural names and resolved at
12
+ # call time (never at load time), so this file loads even while the http/svg/
13
+ # settings/logging slices are still landing:
14
+ # Zer0ImageGenerator::Http — .json/.multipart/.download_to/.with_retries
15
+ # Zer0ImageGenerator::HttpStatusError#error_message
16
+ # Zer0ImageGenerator::Svg::Sanitizer.sanitize_svg / ::LocalRenderer.{seed_for,
17
+ # render_local_svg} / ::Rasterizer.rasterize_svg + Svg::SvgError
18
+ # Zer0ImageGenerator.effective_model(settings, provider)
19
+ # Zer0ImageGenerator::Logging.warn / .info / .debug
20
+ module Zer0ImageGenerator
21
+ # Historical engine behavior: adapt shared size/quality settings per model
22
+ # family. The mapping MUST stay exact — gpt-image-* and dall-e-3 accept
23
+ # disjoint size/quality vocabularies, and a wrong value 400s the request.
24
+ def self.adapt_openai_size_quality(model, size, quality)
25
+ if model.start_with?("gpt-image-") && size == "1792x1024"
26
+ size = "1536x1024"
27
+ elsif model.start_with?("dall-e-") && quality == "auto"
28
+ quality = "standard"
29
+ end
30
+ [size, quality]
31
+ end
32
+
33
+ # Outcome of one generate/edit call. Positional (ok, kind, path) mirrors the
34
+ # Python dataclass construction order; `error` stays a keyword so the many
35
+ # `ImageResult(False, error=...)` call sites read the same in Ruby.
36
+ class ImageResult
37
+ attr_reader :ok, :kind, :path, :error
38
+
39
+ def initialize(ok, kind = "png", path = nil, error: nil)
40
+ @ok = ok
41
+ @kind = kind # "png" | "svg"
42
+ @path = path
43
+ @error = error
44
+ end
45
+
46
+ # Runner (another slice) reads `.ok`; `.ok?` is the idiomatic Ruby alias.
47
+ def ok?
48
+ @ok
49
+ end
50
+ end
51
+
52
+ # Raised by Provider#edit for renderers with no image-edit capability; the
53
+ # runner catches it and falls back to OpenAI (the historical --enhance path).
54
+ class EditUnsupported < StandardError
55
+ attr_reader :provider
56
+
57
+ def initialize(provider)
58
+ @provider = provider
59
+ super("provider '#{provider}' has no image-edit capability")
60
+ end
61
+ end
62
+
63
+ class Provider
64
+ def name
65
+ "base"
66
+ end
67
+
68
+ def is_configured(_env)
69
+ raise NotImplementedError
70
+ end
71
+
72
+ def missing_hint(_env)
73
+ raise NotImplementedError
74
+ end
75
+
76
+ def default_model
77
+ raise NotImplementedError
78
+ end
79
+
80
+ # One line naming the credential this run will authenticate with, for the
81
+ # config banner. nil (the default) means "there is nothing interesting to
82
+ # say" — a provider with exactly one possible credential.
83
+ def auth_description(_env)
84
+ nil
85
+ end
86
+
87
+ def generate(_prompt, _settings, _out_base, _ctx)
88
+ raise NotImplementedError
89
+ end
90
+
91
+ # Enhance an existing image. Providers without an edit capability raise
92
+ # EditUnsupported; the runner then falls back to OpenAI.
93
+ def edit(_image_path, _prompt, _settings, _ctx, _out_path = nil)
94
+ raise EditUnsupported, name
95
+ end
96
+
97
+ private
98
+
99
+ # Persist one OpenAI-style data[0] entry (b64_json preferred, else url).
100
+ # nil/empty strings count as "absent" to match Python's truthiness on
101
+ # `entry.get(...)` — a present-but-empty value must not be treated as data.
102
+ def write_image_payload(entry, out_path)
103
+ b64 = entry["b64_json"]
104
+ if b64 && !b64.empty?
105
+ File.binwrite(out_path.to_s, Base64.decode64(b64))
106
+ return true
107
+ end
108
+ url = entry["url"]
109
+ if url && !url.empty?
110
+ Http.download_to(url, out_path)
111
+ return true
112
+ end
113
+ false
114
+ end
115
+ end
116
+
117
+ # Run-scoped services handed to providers. The per-file `slug` is why the
118
+ # runner hands each worker its OWN copy (see #copy_with): the local provider
119
+ # derives its deterministic seed from the slug, so a shared mutable slug would
120
+ # cross-contaminate concurrent renders.
121
+ class RunContext
122
+ attr_reader :project_root, :env
123
+ attr_accessor :anthropic, :slug, :article
124
+
125
+ def initialize(project_root:, env:, anthropic: nil, slug: "", article: nil)
126
+ @project_root = project_root
127
+ @env = env
128
+ @anthropic = anthropic
129
+ @slug = slug
130
+ # The page's own context (slug/section/title/categories/description/body),
131
+ # set per file by the runner. The local renderer derives BOTH composition
132
+ # and palette from it, which is what makes a banner about its article
133
+ # instead of about its filename.
134
+ @article = article
135
+ end
136
+
137
+ def claude
138
+ # Lazy, memoized — matches the Python `if self.anthropic is None`.
139
+ @anthropic ||= AnthropicClient.new(@env)
140
+ end
141
+
142
+ # Per-file copy carrying a fresh slug (the runner calls `ctx.with_slug(slug)`
143
+ # once per worker). project_root/env/anthropic are shared by reference on
144
+ # purpose: env is read-only and the AnthropicClient is stateless after init,
145
+ # so sharing them across workers is thread-safe; only the slug is per-file
146
+ # and therefore must not be shared.
147
+ def with_slug(slug, article: nil)
148
+ RunContext.new(project_root: @project_root, env: @env,
149
+ anthropic: @anthropic, slug: slug, article: article)
150
+ end
151
+ end
152
+ end
@@ -0,0 +1,60 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "base"
4
+
5
+ module Zer0ImageGenerator
6
+ # Google Gemini generateContent. The image comes back as inline base64 nested
7
+ # under candidates[].content.parts[], keyed either `inlineData` (REST/JSON) or
8
+ # `inline_data` (proto) — accept both.
9
+ class GeminiProvider < Provider
10
+ def name
11
+ "gemini"
12
+ end
13
+
14
+ def is_configured(env)
15
+ key = env["GEMINI_API_KEY"]
16
+ !key.nil? && !key.empty?
17
+ end
18
+
19
+ def missing_hint(_env)
20
+ "GEMINI_API_KEY environment variable is required for the Gemini provider"
21
+ end
22
+
23
+ def default_model
24
+ "gemini-2.5-flash-image"
25
+ end
26
+
27
+ def generate(prompt, settings, out_base, ctx)
28
+ model = Zer0ImageGenerator.effective_model(settings, self)
29
+ out_path = out_base.sub_ext(".png")
30
+ url = "https://generativelanguage.googleapis.com/v1beta/models/" \
31
+ "#{model}:generateContent"
32
+ payload = { "contents" => [{ "parts" => [{ "text" => prompt }] }] }
33
+ begin
34
+ data = Http.with_retries("Gemini API") do
35
+ Http.json(url, payload,
36
+ { "x-goog-api-key" => ctx.env.fetch("GEMINI_API_KEY") },
37
+ timeout: 900)
38
+ end
39
+ (data["candidates"] || []).each do |candidate|
40
+ parts = ((candidate["content"] || {})["parts"]) || []
41
+ parts.each do |part|
42
+ inline = part["inlineData"] || part["inline_data"]
43
+ next unless inline
44
+
45
+ payload_b64 = inline["data"]
46
+ next if payload_b64.nil? || payload_b64.empty?
47
+
48
+ File.binwrite(out_path.to_s, Base64.decode64(payload_b64))
49
+ return ImageResult.new(true, "png", out_path)
50
+ end
51
+ end
52
+ ImageResult.new(false, error: "No inline image data in Gemini response")
53
+ rescue HttpStatusError => exc
54
+ ImageResult.new(false, error: "Gemini API error: #{exc.error_message}")
55
+ rescue StandardError => exc
56
+ ImageResult.new(false, error: exc.message)
57
+ end
58
+ end
59
+ end
60
+ end
@@ -0,0 +1,90 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "base"
4
+
5
+ module Zer0ImageGenerator
6
+ # Shared sanitize -> write -> rasterize tail for SVG-producing providers
7
+ # (local now; claude when that renderer lands). Mixed into a Provider.
8
+ module SvgProviderMixin
9
+ def finish_svg(svg_text, out_base, settings, ctx)
10
+ begin
11
+ clean, notes = Svg::Sanitizer.sanitize_svg(svg_text)
12
+ rescue Svg::SvgError => exc
13
+ return ImageResult.new(false, error: exc.message)
14
+ end
15
+ notes.each { |note| Logging.warn("SVG sanitizer: #{note}") }
16
+ svg_path = out_base.sub_ext(".svg")
17
+ png_path = out_base.sub_ext(".png")
18
+ File.write(svg_path.to_s, clean)
19
+ tool = Svg::Rasterizer.rasterize_svg(svg_path, png_path, ctx.project_root,
20
+ preference: settings.rasterizer)
21
+ if tool
22
+ # missing_ok: the rasterizer already consumed the SVG on some backends.
23
+ File.delete(svg_path.to_s) if File.exist?(svg_path.to_s)
24
+ return ImageResult.new(true, "png", png_path)
25
+ end
26
+ # `rasterizer: none` is a deliberate site policy (SVG-only), not a missing
27
+ # dependency — say so instead of nagging about librsvg.
28
+ if settings.rasterizer == "none"
29
+ Logging.info("SVG-only mode (rasterizer: none) — wrote #{File.basename(svg_path.to_s)}")
30
+ else
31
+ Logging.warn(
32
+ "No SVG rasterizer available — keeping the .svg preview. Social " \
33
+ "og:image works best as PNG: install librsvg (`brew install librsvg`) " \
34
+ "or Playwright (`npx playwright install chromium`), or set " \
35
+ "`preview_images.rasterizer: none` to make SVG-only the intent."
36
+ )
37
+ end
38
+ ImageResult.new(true, "svg", svg_path)
39
+ end
40
+ end
41
+
42
+ # Network-free, deterministic renderer: it renders the local SVG, sanitizes
43
+ # it, and rasterizes it. This is the path CI's integration job exercises.
44
+ class LocalProvider < Provider
45
+ include SvgProviderMixin
46
+
47
+ def name
48
+ "local"
49
+ end
50
+
51
+ def is_configured(_env)
52
+ true
53
+ end
54
+
55
+ def missing_hint(_env)
56
+ ""
57
+ end
58
+
59
+ def default_model
60
+ "template-svg"
61
+ end
62
+
63
+ def generate(_prompt, settings, out_base, ctx)
64
+ # `ctx.slug or out_base.stem`: an empty slug (Python-falsy) falls back to
65
+ # the output stem for the seed, but the raw slug — empty or not — is still
66
+ # what render_local_svg titles the banner with.
67
+ seed_key = ctx.slug.to_s.empty? ? stem_of(out_base) : ctx.slug
68
+ seed = Svg::LocalRenderer.seed_for(seed_key)
69
+ svg_text = Svg::LocalRenderer.render_local_svg(ctx.slug, seed, ctx.article)
70
+ finish_svg(svg_text, out_base, settings, ctx)
71
+ end
72
+
73
+ def edit(image_path, prompt, _settings, _ctx, _out_path = nil)
74
+ # Historical behavior: the local provider "enhances" by doing nothing (no
75
+ # API), so dry testing of --enhance needs no credentials.
76
+ Logging.warn("Local provider: No actual enhancement. Logging prompt...")
77
+ Logging.debug("Enhancement prompt: #{prompt[0, 400]}...")
78
+ Logging.info("Placeholder: would enhance #{image_path}")
79
+ ImageResult.new(true, "png", image_path)
80
+ end
81
+
82
+ private
83
+
84
+ # Pathname has no #stem; Path.stem is the basename minus its final suffix.
85
+ def stem_of(path)
86
+ s = path.to_s
87
+ File.basename(s, File.extname(s))
88
+ end
89
+ end
90
+ end
@@ -0,0 +1,102 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "base"
4
+
5
+ module Zer0ImageGenerator
6
+ # OpenAI images/generations + images/edits (the --enhance path).
7
+ class OpenAIProvider < Provider
8
+ def name
9
+ "openai"
10
+ end
11
+
12
+ def is_configured(env)
13
+ key = env["OPENAI_API_KEY"]
14
+ !key.nil? && !key.empty?
15
+ end
16
+
17
+ def missing_hint(_env)
18
+ "OPENAI_API_KEY environment variable is required for the OpenAI provider"
19
+ end
20
+
21
+ def default_model
22
+ "gpt-image-2"
23
+ end
24
+
25
+ def generate(prompt, settings, out_base, ctx)
26
+ model = Zer0ImageGenerator.effective_model(settings, self)
27
+ size, quality = Zer0ImageGenerator.adapt_openai_size_quality(
28
+ model, settings.size, settings.quality
29
+ )
30
+ out_path = out_base.sub_ext(".png")
31
+ payload = { "model" => model, "prompt" => prompt, "n" => 1,
32
+ "size" => size, "quality" => quality }
33
+ Logging.debug("OpenAI generate: model=#{model} size=#{size} quality=#{quality}")
34
+ begin
35
+ data = Http.with_retries("OpenAI API") do
36
+ Http.json("https://api.openai.com/v1/images/generations",
37
+ payload, headers(ctx.env), timeout: 900)
38
+ end
39
+ entries = data["data"] || []
40
+ if entries.empty? || !write_image_payload(entries[0], out_path)
41
+ return ImageResult.new(false, error: "No image data in OpenAI response")
42
+ end
43
+
44
+ ImageResult.new(true, "png", out_path)
45
+ rescue HttpStatusError => exc
46
+ ImageResult.new(false, error: "OpenAI API error: #{exc.error_message}")
47
+ rescue StandardError => exc
48
+ ImageResult.new(false, error: exc.message)
49
+ end
50
+ end
51
+
52
+ def edit(image_path, prompt, settings, ctx, out_path = nil)
53
+ model = settings.enhance_model
54
+ fields = {
55
+ "prompt" => prompt,
56
+ "model" => model,
57
+ "n" => "1",
58
+ "size" => "auto",
59
+ "quality" => settings.enhance_quality,
60
+ "output_format" => settings.enhance_format,
61
+ }
62
+ # gpt-image-2 does not accept input_fidelity (historical behavior).
63
+ fields["input_fidelity"] = settings.enhance_fidelity if model != "gpt-image-2"
64
+ files = [["image[]", image_path.basename.to_s,
65
+ File.binread(image_path.to_s), "image/png"]]
66
+ out_path ||= image_path
67
+ Logging.debug(
68
+ "OpenAI edit: model=#{model} fidelity=" \
69
+ "#{fields.fetch('input_fidelity', '(omitted)')} format=#{settings.enhance_format}"
70
+ )
71
+ begin
72
+ data = Http.with_retries("OpenAI edits API") do
73
+ Http.multipart("https://api.openai.com/v1/images/edits",
74
+ fields, files, headers(ctx.env), timeout: 900)
75
+ end
76
+ usage = data["usage"] || {}
77
+ # Python `if usage.get("total_tokens")` treats 0 as absent — mirror that
78
+ # so Ruby's truthy 0 doesn't emit a bogus "0 total" line.
79
+ total_tokens = usage["total_tokens"]
80
+ Logging.debug("Token usage: #{total_tokens} total") if total_tokens && total_tokens != 0
81
+ entries = data["data"] || []
82
+ if entries.empty? || !write_image_payload(entries[0], out_path)
83
+ return ImageResult.new(false, error: "No image data in enhance response")
84
+ end
85
+
86
+ revised = entries[0]["revised_prompt"]
87
+ Logging.debug("Revised prompt: #{revised[0, 200]}...") if revised && !revised.empty?
88
+ ImageResult.new(true, "png", out_path)
89
+ rescue HttpStatusError => exc
90
+ ImageResult.new(false, error: "OpenAI enhance API error: #{exc.error_message}")
91
+ rescue StandardError => exc
92
+ ImageResult.new(false, error: exc.message)
93
+ end
94
+ end
95
+
96
+ private
97
+
98
+ def headers(env)
99
+ { "Authorization" => "Bearer #{env.fetch('OPENAI_API_KEY')}" }
100
+ end
101
+ end
102
+ end
@@ -0,0 +1,64 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "base"
4
+
5
+ module Zer0ImageGenerator
6
+ # Stability AI SDXL v1 text-to-image. Returns base64 under `artifacts`, not the
7
+ # OpenAI-style `data` envelope, so it decodes inline rather than via
8
+ # write_image_payload.
9
+ class StabilityProvider < Provider
10
+ def name
11
+ "stability"
12
+ end
13
+
14
+ def is_configured(env)
15
+ key = env["STABILITY_API_KEY"]
16
+ !key.nil? && !key.empty?
17
+ end
18
+
19
+ def missing_hint(_env)
20
+ "STABILITY_API_KEY environment variable is required for the Stability AI provider"
21
+ end
22
+
23
+ def default_model
24
+ "stable-diffusion-xl-1024-v1-0"
25
+ end
26
+
27
+ def generate(prompt, _settings, out_base, ctx)
28
+ out_path = out_base.sub_ext(".png")
29
+ payload = {
30
+ "text_prompts" => [{ "text" => prompt }],
31
+ "cfg_scale" => 7,
32
+ # SDXL v1 endpoint accepts fixed dimension sets; 1024x1024 preserved
33
+ # from the original engine.
34
+ "height" => 1024,
35
+ "width" => 1024,
36
+ "samples" => 1,
37
+ "steps" => 30,
38
+ }
39
+ begin
40
+ data = Http.with_retries("Stability API") do
41
+ Http.json(
42
+ "https://api.stability.ai/v1/generation/" \
43
+ "stable-diffusion-xl-1024-v1-0/text-to-image",
44
+ payload,
45
+ { "Authorization" => "Bearer #{ctx.env.fetch('STABILITY_API_KEY')}" },
46
+ timeout: 900
47
+ )
48
+ end
49
+ artifacts = data["artifacts"] || []
50
+ first_b64 = artifacts.empty? ? nil : artifacts[0]["base64"]
51
+ if artifacts.empty? || first_b64.nil? || first_b64.empty?
52
+ return ImageResult.new(false, error: "No image data in Stability response")
53
+ end
54
+
55
+ File.binwrite(out_path.to_s, Base64.decode64(first_b64))
56
+ ImageResult.new(true, "png", out_path)
57
+ rescue HttpStatusError => exc
58
+ ImageResult.new(false, error: "Stability API error: #{exc.error_message}")
59
+ rescue StandardError => exc
60
+ ImageResult.new(false, error: exc.message)
61
+ end
62
+ end
63
+ end
64
+ end
@@ -0,0 +1,81 @@
1
+ # frozen_string_literal: true
2
+
3
+ require_relative "base"
4
+ require_relative "xai_auth"
5
+
6
+ module Zer0ImageGenerator
7
+ # xAI (Grok) images/generations.
8
+ #
9
+ # Two ways in, tried in order (XAIAuth owns the resolution): a Grok OAuth
10
+ # access token, then the XAI_API_KEY. Both are Bearer credentials on the same
11
+ # endpoint, so the fallback is genuinely a retry with a different token — an
12
+ # OAuth token that has gone stale between runs costs one rejected request and
13
+ # then rides the API key instead of failing the page.
14
+ class XAIProvider < Provider
15
+ # Statuses that mean "this credential is no good" rather than "this request
16
+ # is no good" — the only ones worth re-trying on the next rung.
17
+ AUTH_FAILURE_STATUSES = [401, 403].freeze
18
+
19
+ # xAI caps the prompt; the original engine slices the first 1000 chars.
20
+ PROMPT_LIMIT = 1000
21
+
22
+ def name
23
+ "xai"
24
+ end
25
+
26
+ def is_configured(env)
27
+ !XAIAuth.chain(env).empty?
28
+ end
29
+
30
+ def missing_hint(_env)
31
+ XAIAuth::MISSING_HINT
32
+ end
33
+
34
+ def auth_description(env)
35
+ XAIAuth.describe(env)
36
+ end
37
+
38
+ def default_model
39
+ "grok-2-image"
40
+ end
41
+
42
+ def generate(prompt, settings, out_base, ctx)
43
+ credentials = XAIAuth.chain(ctx.env)
44
+ return ImageResult.new(false, error: XAIAuth::MISSING_HINT) if credentials.empty?
45
+
46
+ model = Zer0ImageGenerator.effective_model(settings, self)
47
+ out_path = out_base.sub_ext(".png")
48
+ payload = { "model" => model, "prompt" => prompt[0, PROMPT_LIMIT], "n" => 1 }
49
+ url = "#{XAIAuth.base_url(ctx.env)}/images/generations"
50
+
51
+ credentials.each_with_index do |credential, index|
52
+ Logging.debug("xAI generate: model=#{model} auth=#{credential.source}")
53
+ begin
54
+ data = Http.with_retries("xAI API") do
55
+ Http.json(url, payload, credential.headers, timeout: 900)
56
+ end
57
+ entries = data["data"] || []
58
+ if entries.empty? || !write_image_payload(entries[0], out_path)
59
+ return ImageResult.new(false, error: "No image data in xAI response")
60
+ end
61
+
62
+ return ImageResult.new(true, "png", out_path)
63
+ rescue HttpStatusError => exc
64
+ fallback = credentials[index + 1]
65
+ if AUTH_FAILURE_STATUSES.include?(exc.status) && fallback
66
+ Logging.warn("xAI rejected the #{credential.source} (HTTP #{exc.status}) — " \
67
+ "retrying with #{fallback.source}")
68
+ next
69
+ end
70
+
71
+ return ImageResult.new(false, error: "xAI API error: #{exc.error_message}")
72
+ rescue StandardError => exc
73
+ return ImageResult.new(false, error: exc.message)
74
+ end
75
+ end
76
+
77
+ # Unreachable: the last rung has no fallback and always returns above.
78
+ ImageResult.new(false, error: "No usable xAI credential")
79
+ end
80
+ end
81
+ end