zer0-image-generator 0.6.0 → 0.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- checksums.yaml +4 -4
- data/CHANGELOG.md +20 -0
- data/README.md +106 -2
- data/lib/zer0_image_generator/abc/style_pack.rb +145 -0
- data/lib/zer0_image_generator/abc.rb +75 -0
- data/lib/zer0_image_generator/all.rb +57 -0
- data/lib/zer0_image_generator/claude/client.rb +291 -0
- data/lib/zer0_image_generator/claude/orchestration.rb +128 -0
- data/lib/zer0_image_generator/cli.rb +318 -0
- data/lib/zer0_image_generator/config.rb +192 -0
- data/lib/zer0_image_generator/constants.rb +177 -0
- data/lib/zer0_image_generator/content.rb +379 -0
- data/lib/zer0_image_generator/engine.rb +21 -0
- data/lib/zer0_image_generator/freesvg/cache.rb +167 -0
- data/lib/zer0_image_generator/freesvg/client.rb +467 -0
- data/lib/zer0_image_generator/http.rb +264 -0
- data/lib/zer0_image_generator/library.rb +201 -0
- data/lib/zer0_image_generator/logging.rb +236 -0
- data/lib/zer0_image_generator/preview_generator.py +1072 -100
- data/lib/zer0_image_generator/prompt.rb +48 -0
- data/lib/zer0_image_generator/providers/base.rb +152 -0
- data/lib/zer0_image_generator/providers/gemini.rb +60 -0
- data/lib/zer0_image_generator/providers/local.rb +90 -0
- data/lib/zer0_image_generator/providers/openai.rb +102 -0
- data/lib/zer0_image_generator/providers/stability.rb +64 -0
- data/lib/zer0_image_generator/providers/xai.rb +81 -0
- data/lib/zer0_image_generator/providers/xai_auth.rb +270 -0
- data/lib/zer0_image_generator/providers.rb +22 -0
- data/lib/zer0_image_generator/runner.rb +573 -0
- data/lib/zer0_image_generator/settings.rb +386 -0
- data/lib/zer0_image_generator/stats.rb +61 -0
- data/lib/zer0_image_generator/support/py_random.rb +189 -0
- data/lib/zer0_image_generator/svg/banner_seed.rb +241 -0
- data/lib/zer0_image_generator/svg/generators/flowfield.rb +154 -0
- data/lib/zer0_image_generator/svg/generators/invaders.rb +125 -0
- data/lib/zer0_image_generator/svg/generators/lowpoly.rb +254 -0
- data/lib/zer0_image_generator/svg/generators/lsystem.rb +228 -0
- data/lib/zer0_image_generator/svg/generators/mandala.rb +155 -0
- data/lib/zer0_image_generator/svg/generators/pixelquest.rb +144 -0
- data/lib/zer0_image_generator/svg/generators/starmap.rb +179 -0
- data/lib/zer0_image_generator/svg/lint.rb +400 -0
- data/lib/zer0_image_generator/svg/local_renderer.rb +606 -0
- data/lib/zer0_image_generator/svg/pixel_kit.rb +167 -0
- data/lib/zer0_image_generator/svg/rasterizer.rb +196 -0
- data/lib/zer0_image_generator/svg/sanitizer.rb +159 -0
- data/lib/zer0_image_generator/version.rb +1 -1
- metadata +43 -2
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "content"
|
|
4
|
+
|
|
5
|
+
module Zer0ImageGenerator
|
|
6
|
+
# Prompt layer: turn a ContentFile plus resolved Settings into the text prompt
|
|
7
|
+
# the image model receives. Ported from the oracle
|
|
8
|
+
# (lib/zer0_image_generator/preview_generator.py). Wording is carried over
|
|
9
|
+
# verbatim from the original engine and must not drift — the committed showcase
|
|
10
|
+
# briefs were produced from these exact strings.
|
|
11
|
+
module Prompt
|
|
12
|
+
# Fallback used by build_enhance_prompt when settings.enhance_prompt is
|
|
13
|
+
# empty. Kept byte-identical to the oracle's DEFAULT_ENHANCE_PROMPT.
|
|
14
|
+
DEFAULT_ENHANCE_PROMPT =
|
|
15
|
+
"Improve this preview banner image: fix any misspelled, garbled, or " \
|
|
16
|
+
"incorrect text so it reads clearly and accurately. Sharpen visual details " \
|
|
17
|
+
"and improve composition while preserving the original art style, color " \
|
|
18
|
+
"palette, and theme. Ensure the image is clean and professional."
|
|
19
|
+
|
|
20
|
+
# Template prompt — wording carried over from the original engine.
|
|
21
|
+
def self.build_prompt(cf, settings)
|
|
22
|
+
parts = ["Create a blog preview banner image for an article titled '#{cf.title}'."]
|
|
23
|
+
parts << "The article is about: #{cf.description}." unless cf.description.empty?
|
|
24
|
+
parts << "Categories: #{cf.categories}." unless cf.categories.empty?
|
|
25
|
+
# First 500 characters, whitespace-collapsed — Content.collapse_whitespace
|
|
26
|
+
# reproduces CPython's `re.sub(r"\s+", " ", ...).strip()` exactly.
|
|
27
|
+
excerpt = Content.collapse_whitespace(cf.content[0, 500] || "")
|
|
28
|
+
parts << "Key themes from content: #{excerpt}" unless excerpt.empty?
|
|
29
|
+
parts << "Art style: #{settings.style}."
|
|
30
|
+
unless settings.style_modifiers.empty?
|
|
31
|
+
parts << "Additional style: #{settings.style_modifiers}."
|
|
32
|
+
end
|
|
33
|
+
parts << "The image should be suitable as a wide blog header/banner image with " \
|
|
34
|
+
"clean composition. No text or words in the image."
|
|
35
|
+
parts.join(" ")
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def self.build_enhance_prompt(cf, settings)
|
|
39
|
+
prompt = settings.enhance_prompt.empty? ? DEFAULT_ENHANCE_PROMPT : settings.enhance_prompt
|
|
40
|
+
unless cf.title.empty?
|
|
41
|
+
prompt += " Context: This is a preview banner for an article titled '#{cf.title}'."
|
|
42
|
+
end
|
|
43
|
+
prompt += " Article topic: #{cf.description}." unless cf.description.empty?
|
|
44
|
+
prompt += " Maintain the #{settings.style} artistic style."
|
|
45
|
+
prompt
|
|
46
|
+
end
|
|
47
|
+
end
|
|
48
|
+
end
|
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require "base64"
|
|
4
|
+
require "pathname"
|
|
5
|
+
|
|
6
|
+
# Port of the provider-layer scaffolding from preview_generator.py
|
|
7
|
+
# (ImageResult, EditUnsupported, Provider, RunContext, adapt_openai_size_quality,
|
|
8
|
+
# and the shared _write_image_payload helper). The concrete providers live in
|
|
9
|
+
# sibling files; this one carries only what they all share.
|
|
10
|
+
#
|
|
11
|
+
# Cross-slice dependencies are referenced by their natural names and resolved at
|
|
12
|
+
# call time (never at load time), so this file loads even while the http/svg/
|
|
13
|
+
# settings/logging slices are still landing:
|
|
14
|
+
# Zer0ImageGenerator::Http — .json/.multipart/.download_to/.with_retries
|
|
15
|
+
# Zer0ImageGenerator::HttpStatusError#error_message
|
|
16
|
+
# Zer0ImageGenerator::Svg::Sanitizer.sanitize_svg / ::LocalRenderer.{seed_for,
|
|
17
|
+
# render_local_svg} / ::Rasterizer.rasterize_svg + Svg::SvgError
|
|
18
|
+
# Zer0ImageGenerator.effective_model(settings, provider)
|
|
19
|
+
# Zer0ImageGenerator::Logging.warn / .info / .debug
|
|
20
|
+
module Zer0ImageGenerator
|
|
21
|
+
# Historical engine behavior: adapt shared size/quality settings per model
|
|
22
|
+
# family. The mapping MUST stay exact — gpt-image-* and dall-e-3 accept
|
|
23
|
+
# disjoint size/quality vocabularies, and a wrong value 400s the request.
|
|
24
|
+
def self.adapt_openai_size_quality(model, size, quality)
|
|
25
|
+
if model.start_with?("gpt-image-") && size == "1792x1024"
|
|
26
|
+
size = "1536x1024"
|
|
27
|
+
elsif model.start_with?("dall-e-") && quality == "auto"
|
|
28
|
+
quality = "standard"
|
|
29
|
+
end
|
|
30
|
+
[size, quality]
|
|
31
|
+
end
|
|
32
|
+
|
|
33
|
+
# Outcome of one generate/edit call. Positional (ok, kind, path) mirrors the
|
|
34
|
+
# Python dataclass construction order; `error` stays a keyword so the many
|
|
35
|
+
# `ImageResult(False, error=...)` call sites read the same in Ruby.
|
|
36
|
+
class ImageResult
|
|
37
|
+
attr_reader :ok, :kind, :path, :error
|
|
38
|
+
|
|
39
|
+
def initialize(ok, kind = "png", path = nil, error: nil)
|
|
40
|
+
@ok = ok
|
|
41
|
+
@kind = kind # "png" | "svg"
|
|
42
|
+
@path = path
|
|
43
|
+
@error = error
|
|
44
|
+
end
|
|
45
|
+
|
|
46
|
+
# Runner (another slice) reads `.ok`; `.ok?` is the idiomatic Ruby alias.
|
|
47
|
+
def ok?
|
|
48
|
+
@ok
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
# Raised by Provider#edit for renderers with no image-edit capability; the
|
|
53
|
+
# runner catches it and falls back to OpenAI (the historical --enhance path).
|
|
54
|
+
class EditUnsupported < StandardError
|
|
55
|
+
attr_reader :provider
|
|
56
|
+
|
|
57
|
+
def initialize(provider)
|
|
58
|
+
@provider = provider
|
|
59
|
+
super("provider '#{provider}' has no image-edit capability")
|
|
60
|
+
end
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
class Provider
|
|
64
|
+
def name
|
|
65
|
+
"base"
|
|
66
|
+
end
|
|
67
|
+
|
|
68
|
+
def is_configured(_env)
|
|
69
|
+
raise NotImplementedError
|
|
70
|
+
end
|
|
71
|
+
|
|
72
|
+
def missing_hint(_env)
|
|
73
|
+
raise NotImplementedError
|
|
74
|
+
end
|
|
75
|
+
|
|
76
|
+
def default_model
|
|
77
|
+
raise NotImplementedError
|
|
78
|
+
end
|
|
79
|
+
|
|
80
|
+
# One line naming the credential this run will authenticate with, for the
|
|
81
|
+
# config banner. nil (the default) means "there is nothing interesting to
|
|
82
|
+
# say" — a provider with exactly one possible credential.
|
|
83
|
+
def auth_description(_env)
|
|
84
|
+
nil
|
|
85
|
+
end
|
|
86
|
+
|
|
87
|
+
def generate(_prompt, _settings, _out_base, _ctx)
|
|
88
|
+
raise NotImplementedError
|
|
89
|
+
end
|
|
90
|
+
|
|
91
|
+
# Enhance an existing image. Providers without an edit capability raise
|
|
92
|
+
# EditUnsupported; the runner then falls back to OpenAI.
|
|
93
|
+
def edit(_image_path, _prompt, _settings, _ctx, _out_path = nil)
|
|
94
|
+
raise EditUnsupported, name
|
|
95
|
+
end
|
|
96
|
+
|
|
97
|
+
private
|
|
98
|
+
|
|
99
|
+
# Persist one OpenAI-style data[0] entry (b64_json preferred, else url).
|
|
100
|
+
# nil/empty strings count as "absent" to match Python's truthiness on
|
|
101
|
+
# `entry.get(...)` — a present-but-empty value must not be treated as data.
|
|
102
|
+
def write_image_payload(entry, out_path)
|
|
103
|
+
b64 = entry["b64_json"]
|
|
104
|
+
if b64 && !b64.empty?
|
|
105
|
+
File.binwrite(out_path.to_s, Base64.decode64(b64))
|
|
106
|
+
return true
|
|
107
|
+
end
|
|
108
|
+
url = entry["url"]
|
|
109
|
+
if url && !url.empty?
|
|
110
|
+
Http.download_to(url, out_path)
|
|
111
|
+
return true
|
|
112
|
+
end
|
|
113
|
+
false
|
|
114
|
+
end
|
|
115
|
+
end
|
|
116
|
+
|
|
117
|
+
# Run-scoped services handed to providers. The per-file `slug` is why the
|
|
118
|
+
# runner hands each worker its OWN copy (see #copy_with): the local provider
|
|
119
|
+
# derives its deterministic seed from the slug, so a shared mutable slug would
|
|
120
|
+
# cross-contaminate concurrent renders.
|
|
121
|
+
class RunContext
|
|
122
|
+
attr_reader :project_root, :env
|
|
123
|
+
attr_accessor :anthropic, :slug, :article
|
|
124
|
+
|
|
125
|
+
def initialize(project_root:, env:, anthropic: nil, slug: "", article: nil)
|
|
126
|
+
@project_root = project_root
|
|
127
|
+
@env = env
|
|
128
|
+
@anthropic = anthropic
|
|
129
|
+
@slug = slug
|
|
130
|
+
# The page's own context (slug/section/title/categories/description/body),
|
|
131
|
+
# set per file by the runner. The local renderer derives BOTH composition
|
|
132
|
+
# and palette from it, which is what makes a banner about its article
|
|
133
|
+
# instead of about its filename.
|
|
134
|
+
@article = article
|
|
135
|
+
end
|
|
136
|
+
|
|
137
|
+
def claude
|
|
138
|
+
# Lazy, memoized — matches the Python `if self.anthropic is None`.
|
|
139
|
+
@anthropic ||= AnthropicClient.new(@env)
|
|
140
|
+
end
|
|
141
|
+
|
|
142
|
+
# Per-file copy carrying a fresh slug (the runner calls `ctx.with_slug(slug)`
|
|
143
|
+
# once per worker). project_root/env/anthropic are shared by reference on
|
|
144
|
+
# purpose: env is read-only and the AnthropicClient is stateless after init,
|
|
145
|
+
# so sharing them across workers is thread-safe; only the slug is per-file
|
|
146
|
+
# and therefore must not be shared.
|
|
147
|
+
def with_slug(slug, article: nil)
|
|
148
|
+
RunContext.new(project_root: @project_root, env: @env,
|
|
149
|
+
anthropic: @anthropic, slug: slug, article: article)
|
|
150
|
+
end
|
|
151
|
+
end
|
|
152
|
+
end
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "base"
|
|
4
|
+
|
|
5
|
+
module Zer0ImageGenerator
|
|
6
|
+
# Google Gemini generateContent. The image comes back as inline base64 nested
|
|
7
|
+
# under candidates[].content.parts[], keyed either `inlineData` (REST/JSON) or
|
|
8
|
+
# `inline_data` (proto) — accept both.
|
|
9
|
+
class GeminiProvider < Provider
|
|
10
|
+
def name
|
|
11
|
+
"gemini"
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def is_configured(env)
|
|
15
|
+
key = env["GEMINI_API_KEY"]
|
|
16
|
+
!key.nil? && !key.empty?
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def missing_hint(_env)
|
|
20
|
+
"GEMINI_API_KEY environment variable is required for the Gemini provider"
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def default_model
|
|
24
|
+
"gemini-2.5-flash-image"
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def generate(prompt, settings, out_base, ctx)
|
|
28
|
+
model = Zer0ImageGenerator.effective_model(settings, self)
|
|
29
|
+
out_path = out_base.sub_ext(".png")
|
|
30
|
+
url = "https://generativelanguage.googleapis.com/v1beta/models/" \
|
|
31
|
+
"#{model}:generateContent"
|
|
32
|
+
payload = { "contents" => [{ "parts" => [{ "text" => prompt }] }] }
|
|
33
|
+
begin
|
|
34
|
+
data = Http.with_retries("Gemini API") do
|
|
35
|
+
Http.json(url, payload,
|
|
36
|
+
{ "x-goog-api-key" => ctx.env.fetch("GEMINI_API_KEY") },
|
|
37
|
+
timeout: 900)
|
|
38
|
+
end
|
|
39
|
+
(data["candidates"] || []).each do |candidate|
|
|
40
|
+
parts = ((candidate["content"] || {})["parts"]) || []
|
|
41
|
+
parts.each do |part|
|
|
42
|
+
inline = part["inlineData"] || part["inline_data"]
|
|
43
|
+
next unless inline
|
|
44
|
+
|
|
45
|
+
payload_b64 = inline["data"]
|
|
46
|
+
next if payload_b64.nil? || payload_b64.empty?
|
|
47
|
+
|
|
48
|
+
File.binwrite(out_path.to_s, Base64.decode64(payload_b64))
|
|
49
|
+
return ImageResult.new(true, "png", out_path)
|
|
50
|
+
end
|
|
51
|
+
end
|
|
52
|
+
ImageResult.new(false, error: "No inline image data in Gemini response")
|
|
53
|
+
rescue HttpStatusError => exc
|
|
54
|
+
ImageResult.new(false, error: "Gemini API error: #{exc.error_message}")
|
|
55
|
+
rescue StandardError => exc
|
|
56
|
+
ImageResult.new(false, error: exc.message)
|
|
57
|
+
end
|
|
58
|
+
end
|
|
59
|
+
end
|
|
60
|
+
end
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "base"
|
|
4
|
+
|
|
5
|
+
module Zer0ImageGenerator
|
|
6
|
+
# Shared sanitize -> write -> rasterize tail for SVG-producing providers
|
|
7
|
+
# (local now; claude when that renderer lands). Mixed into a Provider.
|
|
8
|
+
module SvgProviderMixin
|
|
9
|
+
def finish_svg(svg_text, out_base, settings, ctx)
|
|
10
|
+
begin
|
|
11
|
+
clean, notes = Svg::Sanitizer.sanitize_svg(svg_text)
|
|
12
|
+
rescue Svg::SvgError => exc
|
|
13
|
+
return ImageResult.new(false, error: exc.message)
|
|
14
|
+
end
|
|
15
|
+
notes.each { |note| Logging.warn("SVG sanitizer: #{note}") }
|
|
16
|
+
svg_path = out_base.sub_ext(".svg")
|
|
17
|
+
png_path = out_base.sub_ext(".png")
|
|
18
|
+
File.write(svg_path.to_s, clean)
|
|
19
|
+
tool = Svg::Rasterizer.rasterize_svg(svg_path, png_path, ctx.project_root,
|
|
20
|
+
preference: settings.rasterizer)
|
|
21
|
+
if tool
|
|
22
|
+
# missing_ok: the rasterizer already consumed the SVG on some backends.
|
|
23
|
+
File.delete(svg_path.to_s) if File.exist?(svg_path.to_s)
|
|
24
|
+
return ImageResult.new(true, "png", png_path)
|
|
25
|
+
end
|
|
26
|
+
# `rasterizer: none` is a deliberate site policy (SVG-only), not a missing
|
|
27
|
+
# dependency — say so instead of nagging about librsvg.
|
|
28
|
+
if settings.rasterizer == "none"
|
|
29
|
+
Logging.info("SVG-only mode (rasterizer: none) — wrote #{File.basename(svg_path.to_s)}")
|
|
30
|
+
else
|
|
31
|
+
Logging.warn(
|
|
32
|
+
"No SVG rasterizer available — keeping the .svg preview. Social " \
|
|
33
|
+
"og:image works best as PNG: install librsvg (`brew install librsvg`) " \
|
|
34
|
+
"or Playwright (`npx playwright install chromium`), or set " \
|
|
35
|
+
"`preview_images.rasterizer: none` to make SVG-only the intent."
|
|
36
|
+
)
|
|
37
|
+
end
|
|
38
|
+
ImageResult.new(true, "svg", svg_path)
|
|
39
|
+
end
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
# Network-free, deterministic renderer: it renders the local SVG, sanitizes
|
|
43
|
+
# it, and rasterizes it. This is the path CI's integration job exercises.
|
|
44
|
+
class LocalProvider < Provider
|
|
45
|
+
include SvgProviderMixin
|
|
46
|
+
|
|
47
|
+
def name
|
|
48
|
+
"local"
|
|
49
|
+
end
|
|
50
|
+
|
|
51
|
+
def is_configured(_env)
|
|
52
|
+
true
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
def missing_hint(_env)
|
|
56
|
+
""
|
|
57
|
+
end
|
|
58
|
+
|
|
59
|
+
def default_model
|
|
60
|
+
"template-svg"
|
|
61
|
+
end
|
|
62
|
+
|
|
63
|
+
def generate(_prompt, settings, out_base, ctx)
|
|
64
|
+
# `ctx.slug or out_base.stem`: an empty slug (Python-falsy) falls back to
|
|
65
|
+
# the output stem for the seed, but the raw slug — empty or not — is still
|
|
66
|
+
# what render_local_svg titles the banner with.
|
|
67
|
+
seed_key = ctx.slug.to_s.empty? ? stem_of(out_base) : ctx.slug
|
|
68
|
+
seed = Svg::LocalRenderer.seed_for(seed_key)
|
|
69
|
+
svg_text = Svg::LocalRenderer.render_local_svg(ctx.slug, seed, ctx.article)
|
|
70
|
+
finish_svg(svg_text, out_base, settings, ctx)
|
|
71
|
+
end
|
|
72
|
+
|
|
73
|
+
def edit(image_path, prompt, _settings, _ctx, _out_path = nil)
|
|
74
|
+
# Historical behavior: the local provider "enhances" by doing nothing (no
|
|
75
|
+
# API), so dry testing of --enhance needs no credentials.
|
|
76
|
+
Logging.warn("Local provider: No actual enhancement. Logging prompt...")
|
|
77
|
+
Logging.debug("Enhancement prompt: #{prompt[0, 400]}...")
|
|
78
|
+
Logging.info("Placeholder: would enhance #{image_path}")
|
|
79
|
+
ImageResult.new(true, "png", image_path)
|
|
80
|
+
end
|
|
81
|
+
|
|
82
|
+
private
|
|
83
|
+
|
|
84
|
+
# Pathname has no #stem; Path.stem is the basename minus its final suffix.
|
|
85
|
+
def stem_of(path)
|
|
86
|
+
s = path.to_s
|
|
87
|
+
File.basename(s, File.extname(s))
|
|
88
|
+
end
|
|
89
|
+
end
|
|
90
|
+
end
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "base"
|
|
4
|
+
|
|
5
|
+
module Zer0ImageGenerator
|
|
6
|
+
# OpenAI images/generations + images/edits (the --enhance path).
|
|
7
|
+
class OpenAIProvider < Provider
|
|
8
|
+
def name
|
|
9
|
+
"openai"
|
|
10
|
+
end
|
|
11
|
+
|
|
12
|
+
def is_configured(env)
|
|
13
|
+
key = env["OPENAI_API_KEY"]
|
|
14
|
+
!key.nil? && !key.empty?
|
|
15
|
+
end
|
|
16
|
+
|
|
17
|
+
def missing_hint(_env)
|
|
18
|
+
"OPENAI_API_KEY environment variable is required for the OpenAI provider"
|
|
19
|
+
end
|
|
20
|
+
|
|
21
|
+
def default_model
|
|
22
|
+
"gpt-image-2"
|
|
23
|
+
end
|
|
24
|
+
|
|
25
|
+
def generate(prompt, settings, out_base, ctx)
|
|
26
|
+
model = Zer0ImageGenerator.effective_model(settings, self)
|
|
27
|
+
size, quality = Zer0ImageGenerator.adapt_openai_size_quality(
|
|
28
|
+
model, settings.size, settings.quality
|
|
29
|
+
)
|
|
30
|
+
out_path = out_base.sub_ext(".png")
|
|
31
|
+
payload = { "model" => model, "prompt" => prompt, "n" => 1,
|
|
32
|
+
"size" => size, "quality" => quality }
|
|
33
|
+
Logging.debug("OpenAI generate: model=#{model} size=#{size} quality=#{quality}")
|
|
34
|
+
begin
|
|
35
|
+
data = Http.with_retries("OpenAI API") do
|
|
36
|
+
Http.json("https://api.openai.com/v1/images/generations",
|
|
37
|
+
payload, headers(ctx.env), timeout: 900)
|
|
38
|
+
end
|
|
39
|
+
entries = data["data"] || []
|
|
40
|
+
if entries.empty? || !write_image_payload(entries[0], out_path)
|
|
41
|
+
return ImageResult.new(false, error: "No image data in OpenAI response")
|
|
42
|
+
end
|
|
43
|
+
|
|
44
|
+
ImageResult.new(true, "png", out_path)
|
|
45
|
+
rescue HttpStatusError => exc
|
|
46
|
+
ImageResult.new(false, error: "OpenAI API error: #{exc.error_message}")
|
|
47
|
+
rescue StandardError => exc
|
|
48
|
+
ImageResult.new(false, error: exc.message)
|
|
49
|
+
end
|
|
50
|
+
end
|
|
51
|
+
|
|
52
|
+
def edit(image_path, prompt, settings, ctx, out_path = nil)
|
|
53
|
+
model = settings.enhance_model
|
|
54
|
+
fields = {
|
|
55
|
+
"prompt" => prompt,
|
|
56
|
+
"model" => model,
|
|
57
|
+
"n" => "1",
|
|
58
|
+
"size" => "auto",
|
|
59
|
+
"quality" => settings.enhance_quality,
|
|
60
|
+
"output_format" => settings.enhance_format,
|
|
61
|
+
}
|
|
62
|
+
# gpt-image-2 does not accept input_fidelity (historical behavior).
|
|
63
|
+
fields["input_fidelity"] = settings.enhance_fidelity if model != "gpt-image-2"
|
|
64
|
+
files = [["image[]", image_path.basename.to_s,
|
|
65
|
+
File.binread(image_path.to_s), "image/png"]]
|
|
66
|
+
out_path ||= image_path
|
|
67
|
+
Logging.debug(
|
|
68
|
+
"OpenAI edit: model=#{model} fidelity=" \
|
|
69
|
+
"#{fields.fetch('input_fidelity', '(omitted)')} format=#{settings.enhance_format}"
|
|
70
|
+
)
|
|
71
|
+
begin
|
|
72
|
+
data = Http.with_retries("OpenAI edits API") do
|
|
73
|
+
Http.multipart("https://api.openai.com/v1/images/edits",
|
|
74
|
+
fields, files, headers(ctx.env), timeout: 900)
|
|
75
|
+
end
|
|
76
|
+
usage = data["usage"] || {}
|
|
77
|
+
# Python `if usage.get("total_tokens")` treats 0 as absent — mirror that
|
|
78
|
+
# so Ruby's truthy 0 doesn't emit a bogus "0 total" line.
|
|
79
|
+
total_tokens = usage["total_tokens"]
|
|
80
|
+
Logging.debug("Token usage: #{total_tokens} total") if total_tokens && total_tokens != 0
|
|
81
|
+
entries = data["data"] || []
|
|
82
|
+
if entries.empty? || !write_image_payload(entries[0], out_path)
|
|
83
|
+
return ImageResult.new(false, error: "No image data in enhance response")
|
|
84
|
+
end
|
|
85
|
+
|
|
86
|
+
revised = entries[0]["revised_prompt"]
|
|
87
|
+
Logging.debug("Revised prompt: #{revised[0, 200]}...") if revised && !revised.empty?
|
|
88
|
+
ImageResult.new(true, "png", out_path)
|
|
89
|
+
rescue HttpStatusError => exc
|
|
90
|
+
ImageResult.new(false, error: "OpenAI enhance API error: #{exc.error_message}")
|
|
91
|
+
rescue StandardError => exc
|
|
92
|
+
ImageResult.new(false, error: exc.message)
|
|
93
|
+
end
|
|
94
|
+
end
|
|
95
|
+
|
|
96
|
+
private
|
|
97
|
+
|
|
98
|
+
def headers(env)
|
|
99
|
+
{ "Authorization" => "Bearer #{env.fetch('OPENAI_API_KEY')}" }
|
|
100
|
+
end
|
|
101
|
+
end
|
|
102
|
+
end
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "base"
|
|
4
|
+
|
|
5
|
+
module Zer0ImageGenerator
|
|
6
|
+
# Stability AI SDXL v1 text-to-image. Returns base64 under `artifacts`, not the
|
|
7
|
+
# OpenAI-style `data` envelope, so it decodes inline rather than via
|
|
8
|
+
# write_image_payload.
|
|
9
|
+
class StabilityProvider < Provider
|
|
10
|
+
def name
|
|
11
|
+
"stability"
|
|
12
|
+
end
|
|
13
|
+
|
|
14
|
+
def is_configured(env)
|
|
15
|
+
key = env["STABILITY_API_KEY"]
|
|
16
|
+
!key.nil? && !key.empty?
|
|
17
|
+
end
|
|
18
|
+
|
|
19
|
+
def missing_hint(_env)
|
|
20
|
+
"STABILITY_API_KEY environment variable is required for the Stability AI provider"
|
|
21
|
+
end
|
|
22
|
+
|
|
23
|
+
def default_model
|
|
24
|
+
"stable-diffusion-xl-1024-v1-0"
|
|
25
|
+
end
|
|
26
|
+
|
|
27
|
+
def generate(prompt, _settings, out_base, ctx)
|
|
28
|
+
out_path = out_base.sub_ext(".png")
|
|
29
|
+
payload = {
|
|
30
|
+
"text_prompts" => [{ "text" => prompt }],
|
|
31
|
+
"cfg_scale" => 7,
|
|
32
|
+
# SDXL v1 endpoint accepts fixed dimension sets; 1024x1024 preserved
|
|
33
|
+
# from the original engine.
|
|
34
|
+
"height" => 1024,
|
|
35
|
+
"width" => 1024,
|
|
36
|
+
"samples" => 1,
|
|
37
|
+
"steps" => 30,
|
|
38
|
+
}
|
|
39
|
+
begin
|
|
40
|
+
data = Http.with_retries("Stability API") do
|
|
41
|
+
Http.json(
|
|
42
|
+
"https://api.stability.ai/v1/generation/" \
|
|
43
|
+
"stable-diffusion-xl-1024-v1-0/text-to-image",
|
|
44
|
+
payload,
|
|
45
|
+
{ "Authorization" => "Bearer #{ctx.env.fetch('STABILITY_API_KEY')}" },
|
|
46
|
+
timeout: 900
|
|
47
|
+
)
|
|
48
|
+
end
|
|
49
|
+
artifacts = data["artifacts"] || []
|
|
50
|
+
first_b64 = artifacts.empty? ? nil : artifacts[0]["base64"]
|
|
51
|
+
if artifacts.empty? || first_b64.nil? || first_b64.empty?
|
|
52
|
+
return ImageResult.new(false, error: "No image data in Stability response")
|
|
53
|
+
end
|
|
54
|
+
|
|
55
|
+
File.binwrite(out_path.to_s, Base64.decode64(first_b64))
|
|
56
|
+
ImageResult.new(true, "png", out_path)
|
|
57
|
+
rescue HttpStatusError => exc
|
|
58
|
+
ImageResult.new(false, error: "Stability API error: #{exc.error_message}")
|
|
59
|
+
rescue StandardError => exc
|
|
60
|
+
ImageResult.new(false, error: exc.message)
|
|
61
|
+
end
|
|
62
|
+
end
|
|
63
|
+
end
|
|
64
|
+
end
|
|
@@ -0,0 +1,81 @@
|
|
|
1
|
+
# frozen_string_literal: true
|
|
2
|
+
|
|
3
|
+
require_relative "base"
|
|
4
|
+
require_relative "xai_auth"
|
|
5
|
+
|
|
6
|
+
module Zer0ImageGenerator
|
|
7
|
+
# xAI (Grok) images/generations.
|
|
8
|
+
#
|
|
9
|
+
# Two ways in, tried in order (XAIAuth owns the resolution): a Grok OAuth
|
|
10
|
+
# access token, then the XAI_API_KEY. Both are Bearer credentials on the same
|
|
11
|
+
# endpoint, so the fallback is genuinely a retry with a different token — an
|
|
12
|
+
# OAuth token that has gone stale between runs costs one rejected request and
|
|
13
|
+
# then rides the API key instead of failing the page.
|
|
14
|
+
class XAIProvider < Provider
|
|
15
|
+
# Statuses that mean "this credential is no good" rather than "this request
|
|
16
|
+
# is no good" — the only ones worth re-trying on the next rung.
|
|
17
|
+
AUTH_FAILURE_STATUSES = [401, 403].freeze
|
|
18
|
+
|
|
19
|
+
# xAI caps the prompt; the original engine slices the first 1000 chars.
|
|
20
|
+
PROMPT_LIMIT = 1000
|
|
21
|
+
|
|
22
|
+
def name
|
|
23
|
+
"xai"
|
|
24
|
+
end
|
|
25
|
+
|
|
26
|
+
def is_configured(env)
|
|
27
|
+
!XAIAuth.chain(env).empty?
|
|
28
|
+
end
|
|
29
|
+
|
|
30
|
+
def missing_hint(_env)
|
|
31
|
+
XAIAuth::MISSING_HINT
|
|
32
|
+
end
|
|
33
|
+
|
|
34
|
+
def auth_description(env)
|
|
35
|
+
XAIAuth.describe(env)
|
|
36
|
+
end
|
|
37
|
+
|
|
38
|
+
def default_model
|
|
39
|
+
"grok-2-image"
|
|
40
|
+
end
|
|
41
|
+
|
|
42
|
+
def generate(prompt, settings, out_base, ctx)
|
|
43
|
+
credentials = XAIAuth.chain(ctx.env)
|
|
44
|
+
return ImageResult.new(false, error: XAIAuth::MISSING_HINT) if credentials.empty?
|
|
45
|
+
|
|
46
|
+
model = Zer0ImageGenerator.effective_model(settings, self)
|
|
47
|
+
out_path = out_base.sub_ext(".png")
|
|
48
|
+
payload = { "model" => model, "prompt" => prompt[0, PROMPT_LIMIT], "n" => 1 }
|
|
49
|
+
url = "#{XAIAuth.base_url(ctx.env)}/images/generations"
|
|
50
|
+
|
|
51
|
+
credentials.each_with_index do |credential, index|
|
|
52
|
+
Logging.debug("xAI generate: model=#{model} auth=#{credential.source}")
|
|
53
|
+
begin
|
|
54
|
+
data = Http.with_retries("xAI API") do
|
|
55
|
+
Http.json(url, payload, credential.headers, timeout: 900)
|
|
56
|
+
end
|
|
57
|
+
entries = data["data"] || []
|
|
58
|
+
if entries.empty? || !write_image_payload(entries[0], out_path)
|
|
59
|
+
return ImageResult.new(false, error: "No image data in xAI response")
|
|
60
|
+
end
|
|
61
|
+
|
|
62
|
+
return ImageResult.new(true, "png", out_path)
|
|
63
|
+
rescue HttpStatusError => exc
|
|
64
|
+
fallback = credentials[index + 1]
|
|
65
|
+
if AUTH_FAILURE_STATUSES.include?(exc.status) && fallback
|
|
66
|
+
Logging.warn("xAI rejected the #{credential.source} (HTTP #{exc.status}) — " \
|
|
67
|
+
"retrying with #{fallback.source}")
|
|
68
|
+
next
|
|
69
|
+
end
|
|
70
|
+
|
|
71
|
+
return ImageResult.new(false, error: "xAI API error: #{exc.error_message}")
|
|
72
|
+
rescue StandardError => exc
|
|
73
|
+
return ImageResult.new(false, error: exc.message)
|
|
74
|
+
end
|
|
75
|
+
end
|
|
76
|
+
|
|
77
|
+
# Unreachable: the last rung has no fallback and always returns above.
|
|
78
|
+
ImageResult.new(false, error: "No usable xAI credential")
|
|
79
|
+
end
|
|
80
|
+
end
|
|
81
|
+
end
|