codex-ai 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {codex_ai-0.2.2 → codex_ai-0.2.4}/CHANGELOG.md +9 -0
  2. {codex_ai-0.2.2 → codex_ai-0.2.4}/PKG-INFO +22 -15
  3. {codex_ai-0.2.2 → codex_ai-0.2.4}/README.md +17 -10
  4. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/core/README.md +19 -15
  5. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/core/data_flow.md +5 -3
  6. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/providers/README.md +5 -4
  7. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/providers/data_flow.md +2 -1
  8. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/core/README.md +18 -15
  9. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/core/data_flow.md +5 -3
  10. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/providers/README.md +5 -4
  11. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/providers/data_flow.md +2 -1
  12. {codex_ai-0.2.2 → codex_ai-0.2.4}/pyproject.toml +2 -2
  13. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/__init__.py +1 -1
  14. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/dispatcher.py +8 -5
  15. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/protocol.py +13 -9
  16. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/__init__.py +1 -1
  17. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/gemini.py +63 -15
  18. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/openai.py +4 -3
  19. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_dispatcher.py +6 -2
  20. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_protocol.py +1 -0
  21. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/test_gemini_provider.py +30 -1
  22. {codex_ai-0.2.2 → codex_ai-0.2.4}/uv.lock +6 -6
  23. {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/ci.yml +0 -0
  24. {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/docs.yml +0 -0
  25. {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/publish.yml +0 -0
  26. {codex_ai-0.2.2 → codex_ai-0.2.4}/.gitignore +0 -0
  27. {codex_ai-0.2.2 → codex_ai-0.2.4}/.nojekyll +0 -0
  28. {codex_ai-0.2.2 → codex_ai-0.2.4}/.pre-commit-config.yaml +0 -0
  29. {codex_ai-0.2.2 → codex_ai-0.2.4}/.python-version +0 -0
  30. {codex_ai-0.2.2 → codex_ai-0.2.4}/.secrets.baseline +0 -0
  31. {codex_ai-0.2.2 → codex_ai-0.2.4}/LICENSE +0 -0
  32. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/changelog.md +0 -0
  33. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/dispatcher.md +0 -0
  34. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/exceptions.md +0 -0
  35. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/protocol.md +0 -0
  36. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/router.md +0 -0
  37. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/sync.md +0 -0
  38. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/index.md +0 -0
  39. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/providers/gemini.md +0 -0
  40. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/providers/openai.md +0 -0
  41. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/index.md +0 -0
  42. {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/stylesheets/extra.css +0 -0
  43. {codex_ai-0.2.2 → codex_ai-0.2.4}/mkdocs.yml +0 -0
  44. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/__init__.py +0 -0
  45. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/exceptions.py +0 -0
  46. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/router.py +0 -0
  47. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/sync.py +0 -0
  48. {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/py.typed +0 -0
  49. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/conftest.py +0 -0
  50. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/__init__.py +0 -0
  51. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/conftest.py +0 -0
  52. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/test_providers_integration.py +0 -0
  53. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/__init__.py +0 -0
  54. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/conftest.py +0 -0
  55. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/__init__.py +0 -0
  56. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_exceptions.py +0 -0
  57. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_router.py +0 -0
  58. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_sync.py +0 -0
  59. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/__init__.py +0 -0
  60. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/test_openai_provider.py +0 -0
  61. {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/test_public_api.py +0 -0
  62. {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/__init__.py +0 -0
  63. {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/README.md +0 -0
  64. {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/__init__.py +0 -0
  65. {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/check.py +0 -0
  66. {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/generate_project_tree.py +0 -0
@@ -4,6 +4,15 @@ All notable changes to this project will be documented in this file.
4
4
 
5
5
  ## [Unreleased]
6
6
 
7
+ ## [0.2.3] - 2026-05-17
8
+
9
+ ### Added
10
+ - Added explicit Gemini image `image_config` support for aspect ratio and image size controls.
11
+ - Added a Gemini image retry from `image_size="4K"` to `image_size="2K"` when the initial 4K request is rejected.
12
+
13
+ ### Changed
14
+ - Updated the Gemini extra to `google-genai==2.3.0` for the SDK image configuration contract.
15
+
7
16
  ## [0.2.2] - 2026-05-15
8
17
 
9
18
  ### Fixed
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codex-ai
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: Gemini-first and OpenAI provider helpers for Codex
5
5
  Project-URL: Homepage, https://github.com/codexdlc/codex-ai
6
6
  Project-URL: Documentation, https://codexdlc.github.io/codex-ai/
@@ -16,15 +16,15 @@ Classifier: License :: OSI Approved :: Apache Software License
16
16
  Classifier: Programming Language :: Python :: 3.12
17
17
  Classifier: Programming Language :: Python :: 3.13
18
18
  Requires-Python: >=3.12
19
- Requires-Dist: codex-core<0.4.0,>=0.2.2
19
+ Requires-Dist: codex-core<0.5.0,>=0.2.2
20
20
  Requires-Dist: pydantic<3.0,>=2.0
21
21
  Provides-Extra: all
22
- Requires-Dist: google-genai==1.68.0; extra == 'all'
22
+ Requires-Dist: google-genai==2.3.0; extra == 'all'
23
23
  Requires-Dist: openai<2.0,>=1.0; extra == 'all'
24
24
  Provides-Extra: dev
25
25
  Requires-Dist: bandit>=1.7; extra == 'dev'
26
26
  Requires-Dist: detect-secrets>=1.5; extra == 'dev'
27
- Requires-Dist: google-genai==1.68.0; extra == 'dev'
27
+ Requires-Dist: google-genai==2.3.0; extra == 'dev'
28
28
  Requires-Dist: mypy>=1.10; extra == 'dev'
29
29
  Requires-Dist: openai<2.0,>=1.0; extra == 'dev'
30
30
  Requires-Dist: pip-audit>=2.7; extra == 'dev'
@@ -40,7 +40,7 @@ Requires-Dist: mkdocs-material>=9.0; extra == 'docs'
40
40
  Requires-Dist: mkdocs>=1.5; extra == 'docs'
41
41
  Requires-Dist: mkdocstrings[python]>=0.24; extra == 'docs'
42
42
  Provides-Extra: gemini
43
- Requires-Dist: google-genai==1.68.0; extra == 'gemini'
43
+ Requires-Dist: google-genai==2.3.0; extra == 'gemini'
44
44
  Provides-Extra: openai
45
45
  Requires-Dist: openai<2.0,>=1.0; extra == 'openai'
46
46
  Description-Content-Type: text/markdown
@@ -49,10 +49,10 @@ Description-Content-Type: text/markdown
49
49
 
50
50
  [![PyPI version](https://img.shields.io/pypi/v/codex-ai.svg)](https://pypi.org/project/codex-ai/)
51
51
  [![Python](https://img.shields.io/pypi/pyversions/codex-ai.svg)](https://pypi.org/project/codex-ai/)
52
- [![CI](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
52
+ [![CI](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml/badge.svg?branch=develop)](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
53
53
  [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](https://github.com/codexdlc/codex-ai/blob/main/LICENSE)
54
54
 
55
- Gemini-first and OpenAI provider helpers for the Codex ecosystem. The library keeps the legacy prompt router for text generation, and exposes direct provider methods for practical Gemini workflows.
55
+ Gemini-first API helpers for the Codex ecosystem. The active surface is direct Gemini text, JSON, Gemini image, and Imagen generation. OpenAI remains as a small text-only adapter, and the router/dispatcher layer is kept for legacy text workflows.
56
56
 
57
57
  ## Install
58
58
 
@@ -83,9 +83,10 @@ gemini = GeminiProvider(api_key="AIza...")
83
83
  text = await gemini.generate_text("Write one short tavern rumor.")
84
84
  loot = await gemini.generate_json("Create one loot item.", schema=LootItem)
85
85
  image_bytes, content_type = await gemini.generate_image_bytes(
86
- "A fantasy clan banner, game icon style.",
87
- model="gemini-2.5-flash-image",
88
- response_mime_type="image/webp",
86
+ "Square tactical dark fantasy ruined capital city map, no labels.",
87
+ model="gemini-3-pro-image-preview",
88
+ response_mime_type="image/png",
89
+ image_config={"aspect_ratio": "1:1", "image_size": "4K"},
89
90
  )
90
91
 
91
92
  imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
@@ -98,11 +99,13 @@ imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
98
99
 
99
100
  `generate_image_bytes()` targets Gemini image models through `generate_content` and treats
100
101
  `response_mime_type` as a preferred/fallback MIME type. It does not pass image MIME values
101
- to Gemini's text `response_mime_type` config field. Use `generate_imagen_bytes()` for
102
+ to Gemini's text `response_mime_type` config field. Pass Gemini image controls such as
103
+ `aspect_ratio` and `image_size` with `image_config`; if a `4K` request is rejected,
104
+ the Gemini provider retries once with `2K`. Use `generate_imagen_bytes()` for
102
105
  Imagen models; that path uses `generate_images` and passes the requested MIME as
103
106
  `output_mime_type`.
104
107
 
105
- ## Router Pipeline
108
+ ## Legacy Text Router
106
109
 
107
110
  ```python
108
111
  from codex_ai import GeminiProvider, LLMDispatcher, LLMMessage, LLMRouter, PromptResult
@@ -124,13 +127,17 @@ dispatcher.include_router(router)
124
127
  response = await dispatcher.process("chat", text="Hello!")
125
128
  ```
126
129
 
130
+ Use this path only when you already have prompt builders registered through `LLMRouter`.
131
+ New Gemini integrations should call `generate_text()`, `generate_json()`,
132
+ `generate_image_bytes()`, or `generate_imagen_bytes()` directly.
133
+
127
134
  ## Modules
128
135
 
129
136
  | Module | Extra | Description |
130
137
  | :--- | :--- | :--- |
131
- | `codex_ai.core` | - | Dispatcher, router, protocol types, sync wrapper, and shared exception contract |
132
- | `codex_ai.providers.gemini` | `[gemini]` | Google Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
133
- | `codex_ai.providers.openai` | `[openai]` | OpenAI Chat Completions text provider |
138
+ | `codex_ai.providers.gemini` | `[gemini]` | Primary API: Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
139
+ | `codex_ai.providers.openai` | `[openai]` | Text-only OpenAI Chat Completions adapter |
140
+ | `codex_ai.core` | - | Legacy text router/dispatcher contracts and shared provider exceptions |
134
141
 
135
142
  ## Development
136
143
 
@@ -2,10 +2,10 @@
2
2
 
3
3
  [![PyPI version](https://img.shields.io/pypi/v/codex-ai.svg)](https://pypi.org/project/codex-ai/)
4
4
  [![Python](https://img.shields.io/pypi/pyversions/codex-ai.svg)](https://pypi.org/project/codex-ai/)
5
- [![CI](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml/badge.svg?branch=main)](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
5
+ [![CI](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml/badge.svg?branch=develop)](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
6
6
  [![License](https://img.shields.io/badge/license-Apache%202.0-blue.svg)](https://github.com/codexdlc/codex-ai/blob/main/LICENSE)
7
7
 
8
- Gemini-first and OpenAI provider helpers for the Codex ecosystem. The library keeps the legacy prompt router for text generation, and exposes direct provider methods for practical Gemini workflows.
8
+ Gemini-first API helpers for the Codex ecosystem. The active surface is direct Gemini text, JSON, Gemini image, and Imagen generation. OpenAI remains as a small text-only adapter, and the router/dispatcher layer is kept for legacy text workflows.
9
9
 
10
10
  ## Install
11
11
 
@@ -36,9 +36,10 @@ gemini = GeminiProvider(api_key="AIza...")
36
36
  text = await gemini.generate_text("Write one short tavern rumor.")
37
37
  loot = await gemini.generate_json("Create one loot item.", schema=LootItem)
38
38
  image_bytes, content_type = await gemini.generate_image_bytes(
39
- "A fantasy clan banner, game icon style.",
40
- model="gemini-2.5-flash-image",
41
- response_mime_type="image/webp",
39
+ "Square tactical dark fantasy ruined capital city map, no labels.",
40
+ model="gemini-3-pro-image-preview",
41
+ response_mime_type="image/png",
42
+ image_config={"aspect_ratio": "1:1", "image_size": "4K"},
42
43
  )
43
44
 
44
45
  imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
@@ -51,11 +52,13 @@ imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
51
52
 
52
53
  `generate_image_bytes()` targets Gemini image models through `generate_content` and treats
53
54
  `response_mime_type` as a preferred/fallback MIME type. It does not pass image MIME values
54
- to Gemini's text `response_mime_type` config field. Use `generate_imagen_bytes()` for
55
+ to Gemini's text `response_mime_type` config field. Pass Gemini image controls such as
56
+ `aspect_ratio` and `image_size` with `image_config`; if a `4K` request is rejected,
57
+ the Gemini provider retries once with `2K`. Use `generate_imagen_bytes()` for
55
58
  Imagen models; that path uses `generate_images` and passes the requested MIME as
56
59
  `output_mime_type`.
57
60
 
58
- ## Router Pipeline
61
+ ## Legacy Text Router
59
62
 
60
63
  ```python
61
64
  from codex_ai import GeminiProvider, LLMDispatcher, LLMMessage, LLMRouter, PromptResult
@@ -77,13 +80,17 @@ dispatcher.include_router(router)
77
80
  response = await dispatcher.process("chat", text="Hello!")
78
81
  ```
79
82
 
83
+ Use this path only when you already have prompt builders registered through `LLMRouter`.
84
+ New Gemini integrations should call `generate_text()`, `generate_json()`,
85
+ `generate_image_bytes()`, or `generate_imagen_bytes()` directly.
86
+
80
87
  ## Modules
81
88
 
82
89
  | Module | Extra | Description |
83
90
  | :--- | :--- | :--- |
84
- | `codex_ai.core` | - | Dispatcher, router, protocol types, sync wrapper, and shared exception contract |
85
- | `codex_ai.providers.gemini` | `[gemini]` | Google Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
86
- | `codex_ai.providers.openai` | `[openai]` | OpenAI Chat Completions text provider |
91
+ | `codex_ai.providers.gemini` | `[gemini]` | Primary API: Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
92
+ | `codex_ai.providers.openai` | `[openai]` | Text-only OpenAI Chat Completions adapter |
93
+ | `codex_ai.core` | - | Legacy text router/dispatcher contracts and shared provider exceptions |
87
94
 
88
95
  ## Development
89
96
 
@@ -2,19 +2,22 @@
2
2
 
3
3
  ## Purpose
4
4
 
5
- `codex_ai.core` is the orchestration layer for LLM interactions. It decouples prompt construction from provider selection — prompt logic is defined once and can be routed to any backend without changes.
5
+ `codex_ai.core` is the legacy text orchestration layer. It keeps existing `LLMRouter`/`LLMDispatcher` prompt-builder workflows working while the active API surface moves to direct provider methods.
6
6
 
7
7
  ## Why It's a Module
8
8
 
9
- Working directly with LLM SDKs across a codebase creates three recurring problems:
9
+ Older Codex integrations use mode-based prompt builders. This module keeps that shape stable without making it the primary abstraction for new work:
10
10
 
11
- | Problem | What breaks |
12
- |---------|-------------|
13
- | Prompt logic scattered across call sites | Hard to test, review, or reuse |
14
- | Provider SDK calls coupled to business code | Switching providers requires touching every call |
15
- | No unified contract for async/sync contexts | Different wiring for Django views, ARQ workers, bots |
11
+ | Need | Current role |
12
+ |------|--------------|
13
+ | Keep registered prompt builders working | `LLMRouter` maps modes to builders |
14
+ | Run existing text flows without rewriting callers | `LLMDispatcher.process()` still calls `provider.answer()` |
15
+ | Bridge sync-only contexts | `SyncLLMDispatcher` remains available for CLI/WSGI code |
16
16
 
17
- `core` solves these by introducing a clean pipeline:
17
+ For new Gemini work, prefer `GeminiProvider.generate_text()`, `generate_json()`,
18
+ `generate_image_bytes()`, and `generate_imagen_bytes()` directly.
19
+
20
+ The retained text pipeline is:
18
21
 
19
22
  ```
20
23
  @router.prompt("mode") → PromptResult (frozen DTO) → LLMProviderProtocol.answer()
@@ -32,7 +35,7 @@ Working directly with LLM SDKs across a codebase creates three recurring problem
32
35
  │ include_router(router)
33
36
  ▼
34
37
  ┌──────────────────────┐
35
- │ LLMDispatcher │ orchestrates builder → provider
38
+ │ LLMDispatcher │ legacy text builder → provider
36
39
  │ │
37
40
  │ .process(mode, **kw)│
38
41
  └──────┬───────────────┘
@@ -52,19 +55,20 @@ Working directly with LLM SDKs across a codebase creates three recurring problem
52
55
 
53
56
  | Component | Class | Role |
54
57
  |-----------|-------|------|
55
- | `protocol.py` | `PromptResult` | Frozen DTO — immutable prompt passed to providers |
56
- | `protocol.py` | `LLMProviderProtocol` | Structural protocol — any class with `async answer()` qualifies |
58
+ | `protocol.py` | `PromptResult` | Frozen DTO used by legacy text prompt builders |
59
+ | `protocol.py` | `LLMProviderProtocol` | Structural text compatibility protocol for `answer()` |
57
60
  | `protocol.py` | `PromptBuilder` | Type alias for `async def (...) -> PromptResult` |
58
61
  | `router.py` | `LLMRouter` | Registry: maps `mode` strings to builder functions via decorator |
59
- | `dispatcher.py` | `LLMDispatcher` | Wires router + provider; single entry point for prompt execution |
62
+ | `dispatcher.py` | `LLMDispatcher` | Runs legacy text prompts and delegates direct provider convenience methods |
60
63
  | `sync.py` | `SyncLLMDispatcher` | Wraps `LLMDispatcher` with `asyncio.run()` for WSGI/CLI contexts |
61
64
  | `exceptions.py` | `LLMProviderError` | Base exception raised by all provider implementations |
62
65
 
63
66
  ## Key Design Decisions
64
67
 
65
- - **Frozen DTO (`PromptResult`)** — built once by the builder, cannot be mutated downstream. Prevents accidental state sharing between requests.
66
- - **`@runtime_checkable` Protocol** — `isinstance(obj, LLMProviderProtocol)` works at runtime. No inheritance required from any base class.
67
- - **Mode-based dispatch** — `dispatcher.process("chat", ...)` maps to a registered builder. Adding a new prompt type never touches existing code.
68
+ - **Direct provider APIs first** — Gemini capabilities are exposed as explicit methods instead of being forced through a universal provider interface.
69
+ - **Frozen DTO (`PromptResult`)** — kept for legacy builders and cannot be mutated downstream.
70
+ - **`@runtime_checkable` Protocol** — retained for runtime checks at compatibility boundaries.
71
+ - **Mode-based dispatch is legacy text infrastructure** — `dispatcher.process("chat", ...)` maps to a registered builder for existing flows.
68
72
  - **All logs at `DEBUG`** — dispatcher emits only debug-level messages. Production log level controls visibility without code changes.
69
73
  - **`SyncLLMDispatcher` for Django only** — uses `asyncio.run()` which creates a new event loop. Never call from inside an async context (ARQ, async views, bots) — use `LLMDispatcher` directly.
70
74
 
@@ -1,6 +1,8 @@
1
1
  # Core — Data Flow
2
2
 
3
- ## Request Lifecycle
3
+ ## Legacy Text Request Lifecycle
4
+
5
+ This flow is retained for existing `LLMRouter` integrations. New Gemini work should use direct provider methods.
4
6
 
5
7
  ```
6
8
  1. Application calls:
@@ -13,14 +15,14 @@
13
15
  prompt = await builder(text="Hello!")
14
16
  # → PromptResult(messages=[LLMMessage(role="user", content="Hello!")])
15
17
 
16
- 4. Dispatcher calls provider:
18
+ 4. Dispatcher calls the text compatibility method:
17
19
  response = await self._provider.answer(prompt, **kw)
18
20
 
19
21
  5. Provider returns text:
20
22
  "Hi there! How can I help?"
21
23
  ```
22
24
 
23
- ## Component Interactions
25
+ ## Legacy Component Interactions
24
26
 
25
27
  ```
26
28
  Application
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## Purpose
4
4
 
5
- `codex_ai.providers` contains concrete provider adapters for the APIs currently supported by the library: Gemini and OpenAI.
5
+ `codex_ai.providers` contains concrete SDK adapters, not a broad interchangeable provider framework. Gemini is the product focus; OpenAI is retained only for text generation.
6
6
 
7
7
  Gemini is the primary target and exposes direct methods for text, JSON, and image generation:
8
8
 
@@ -13,7 +13,7 @@ await gemini.generate_image_bytes(...)
13
13
  await gemini.generate_imagen_bytes(...)
14
14
  ```
15
15
 
16
- OpenAI is kept as a text provider with the same `generate_text(...)` convenience.
16
+ OpenAI is kept as a text-only adapter with the same `generate_text(...)` convenience.
17
17
 
18
18
  ## Architecture
19
19
 
@@ -47,6 +47,7 @@ LLMRouter builder -> PromptResult -> LLMDispatcher.process() -> provider.answer(
47
47
  - Gemini-specific capabilities are represented directly instead of being hidden behind a broad universal abstraction.
48
48
  - JSON generation uses provider-native JSON configuration and still validates locally with `json.loads` and optional Pydantic models.
49
49
  - Gemini image generation and Imagen generation are separate explicit methods because they use different SDK calls.
50
- - `generate_image_bytes()` uses Gemini `generate_content` with image modality. Its `response_mime_type` is only a preferred/fallback MIME type.
50
+ - `generate_image_bytes()` uses Gemini `generate_content` with image modality. Its `response_mime_type` is only a preferred/fallback MIME type, and Gemini image controls are passed through `image_config`.
51
+ - `generate_image_bytes()` retries a rejected `image_config={"image_size": "4K"}` request once as `2K`.
51
52
  - `generate_imagen_bytes()` uses Imagen `generate_images` and passes the requested MIME as `output_mime_type`.
52
- - Anthropic, OpenRouter, and multi-provider failover are not active APIs in this alpha line.
53
+ - Anthropic, OpenRouter, and multi-provider failover are not active APIs in this line.
@@ -28,12 +28,13 @@ PromptResult | str
28
28
  ```
29
29
  prompt: str
30
30
  -> GeminiProvider.generate_image_bytes()
31
- -> GenerateContentConfig(response_modalities=[IMAGE])
31
+ -> GenerateContentConfig(response_modalities=[IMAGE], image_config=...)
32
32
  -> first inline_data image part
33
33
  -> (bytes, actual_mime_type)
34
34
  ```
35
35
 
36
36
  `response_mime_type` is not passed to `GenerateContentConfig.response_mime_type` on this path; it is only a fallback content type when Gemini omits `inline_data.mime_type`.
37
+ When `image_config.image_size` is `4K` and Gemini rejects the request, the provider retries once with `image_size` changed to `2K`.
37
38
 
38
39
  ## Imagen Images
39
40
 
@@ -2,19 +2,21 @@
2
2
 
3
3
  ## Назначение
4
4
 
5
- `codex_ai.core` — оркестрационный слой для работы с LLM. Разделяет построение промпта и выбор провайдера: логика промптов описывается один раз и может быть направлена к любому бэкенду без изменений.
5
+ `codex_ai.core` — legacy-слой текстовой оркестрации. Он сохраняет существующие workflow на `LLMRouter`/`LLMDispatcher`, пока активная API-поверхность переехала в прямые методы провайдеров.
6
6
 
7
7
  ## Зачем это модуль
8
8
 
9
- Прямая работа с LLM SDK по всей кодовой базе создаёт три повторяющихся проблемы:
9
+ Старые Codex-интеграции используют mode-based prompt builders. Этот модуль сохраняет такую форму, но больше не является основной абстракцией для новой работы:
10
10
 
11
- | Проблема | Что ломается |
12
- |----------|-------------|
13
- | Логика промптов разбросана по call site-ам | Трудно тестировать, ревьювить и переиспользовать |
14
- | Вызовы SDK провайдера связаны с бизнес-кодом | Смена провайдера требует правок в каждом вызове |
15
- | Нет единого контракта для async/sync контекстов | Разная обвязка для Django views, ARQ workers, ботов |
11
+ | Потребность | Текущая роль |
12
+ |-------------|--------------|
13
+ | Сохранить зарегистрированные prompt builders | `LLMRouter` связывает modes с билдерами |
14
+ | Не переписывать существующие текстовые вызовы | `LLMDispatcher.process()` продолжает вызывать `provider.answer()` |
15
+ | Поддержать sync-only контексты | `SyncLLMDispatcher` остается для CLI/WSGI кода |
16
16
 
17
- `core` решает это через единый pipeline:
17
+ Для новой Gemini-интеграции лучше использовать прямые методы `GeminiProvider.generate_text()`, `generate_json()`, `generate_image_bytes()` и `generate_imagen_bytes()`.
18
+
19
+ Сохраненный текстовый pipeline:
18
20
 
19
21
  ```
20
22
  @router.prompt("mode") → PromptResult (frozen DTO) → LLMProviderProtocol.answer()
@@ -32,7 +34,7 @@
32
34
  │ include_router(router)
33
35
  ▼
34
36
  ┌──────────────────────┐
35
- │ LLMDispatcher │ оркестрирует builder → provider
37
+ │ LLMDispatcher │ legacy text builder → provider
36
38
  │ │
37
39
  │ .process(mode, **kw)│
38
40
  └──────┬───────────────┘
@@ -52,19 +54,20 @@
52
54
 
53
55
  | Компонент | Класс | Роль |
54
56
  |-----------|-------|------|
55
- | `protocol.py` | `PromptResult` | Frozen DTO — иммутабельный промпт, передаваемый провайдерам |
56
- | `protocol.py` | `LLMProviderProtocol` | Структурный протокол — любой класс с `async answer()` подходит |
57
+ | `protocol.py` | `PromptResult` | Frozen DTO для legacy text prompt builders |
58
+ | `protocol.py` | `LLMProviderProtocol` | Структурный текстовый compatibility-протокол для `answer()` |
57
59
  | `protocol.py` | `PromptBuilder` | Type alias для `async def (...) -> PromptResult` |
58
60
  | `router.py` | `LLMRouter` | Реестр: связывает строки `mode` с функциями-билдерами через декоратор |
59
- | `dispatcher.py` | `LLMDispatcher` | Связывает router + provider; единая точка входа для выполнения промптов |
61
+ | `dispatcher.py` | `LLMDispatcher` | Выполняет legacy text prompts и делегирует прямые provider convenience methods |
60
62
  | `sync.py` | `SyncLLMDispatcher` | Оборачивает `LLMDispatcher` через `asyncio.run()` для WSGI/CLI |
61
63
  | `exceptions.py` | `LLMProviderError` | Базовое исключение, поднимаемое всеми реализациями провайдеров |
62
64
 
63
65
  ## Ключевые решения
64
66
 
65
- - **Frozen DTO (`PromptResult`)** — строится один раз в билдере, не может быть изменён downstream. Исключает случайный shared state между запросами.
66
- - **`@runtime_checkable` Protocol** — `isinstance(obj, LLMProviderProtocol)` работает в runtime. Наследование от базового класса не требуется.
67
- - **Mode-based dispatch** — `dispatcher.process("chat", ...)` находит зарегистрированный билдер. Добавление нового типа промпта не затрагивает существующий код.
67
+ - **Direct provider APIs first** — возможности Gemini раскрываются явными методами, а не проталкиваются через универсальный provider interface.
68
+ - **Frozen DTO (`PromptResult`)** — сохранен для legacy builders и не может быть изменён downstream.
69
+ - **`@runtime_checkable` Protocol** — оставлен для runtime-проверок на compatibility boundaries.
70
+ - **Mode-based dispatch — legacy text infrastructure** — `dispatcher.process("chat", ...)` находит зарегистрированный билдер для существующих flows.
68
71
  - **Все логи на `DEBUG`** — диспетчер пишет только debug-сообщения. Уровень логирования в продакшне управляет видимостью без изменений кода.
69
72
  - **`SyncLLMDispatcher` только для Django** — использует `asyncio.run()`, создающий новый event loop. Никогда не вызывать из async-контекста (ARQ, async views, боты) — используйте `LLMDispatcher` напрямую.
70
73
 
@@ -1,6 +1,8 @@
1
1
  # Core — Поток данных
2
2
 
3
- ## Жизненный цикл запроса
3
+ ## Жизненный цикл legacy text запроса
4
+
5
+ Этот flow сохранен для существующих интеграций на `LLMRouter`. Новую Gemini-интеграцию лучше писать через прямые методы провайдера.
4
6
 
5
7
  ```
6
8
  1. Приложение вызывает:
@@ -13,14 +15,14 @@
13
15
  prompt = await builder(text="Привет!")
14
16
  # → PromptResult(messages=[LLMMessage(role="user", content="Привет!")])
15
17
 
16
- 4. Диспетчер вызывает провайдер:
18
+ 4. Диспетчер вызывает текстовый compatibility method:
17
19
  response = await self._provider.answer(prompt, **kw)
18
20
 
19
21
  5. Провайдер возвращает текст:
20
22
  "Привет! Чем могу помочь?"
21
23
  ```
22
24
 
23
- ## Взаимодействие компонентов
25
+ ## Взаимодействие legacy-компонентов
24
26
 
25
27
  ```
26
28
  Приложение
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## Назначение
4
4
 
5
- `codex_ai.providers` содержит адаптеры для API, которые сейчас реально поддерживаются библиотекой: Gemini и OpenAI.
5
+ `codex_ai.providers` содержит конкретные SDK-адаптеры, а не широкий interchangeable provider framework. Фокус продукта — Gemini; OpenAI сохранен только для текстовой генерации.
6
6
 
7
7
  Gemini является основным направлением и дает прямые методы для текста, JSON и картинок:
8
8
 
@@ -13,7 +13,7 @@ await gemini.generate_image_bytes(...)
13
13
  await gemini.generate_imagen_bytes(...)
14
14
  ```
15
15
 
16
- OpenAI оставлен как текстовый провайдер с `generate_text(...)`.
16
+ OpenAI оставлен как text-only адаптер с `generate_text(...)`.
17
17
 
18
18
  ## Архитектура
19
19
 
@@ -47,6 +47,7 @@ LLMRouter builder -> PromptResult -> LLMDispatcher.process() -> provider.answer(
47
47
  - Возможности Gemini представлены напрямую, без широкой универсальной абстракции.
48
48
  - JSON generation использует native JSON config провайдера и локальную проверку через `json.loads` и optional Pydantic schema.
49
49
  - Gemini image generation и Imagen generation разведены в отдельные явные методы, потому что они используют разные SDK calls.
50
- - `generate_image_bytes()` использует Gemini `generate_content` с image modality. Его `response_mime_type` является только preferred/fallback MIME type.
50
+ - `generate_image_bytes()` использует Gemini `generate_content` с image modality. Его `response_mime_type` является только preferred/fallback MIME type, а Gemini image controls передаются через `image_config`.
51
+ - `generate_image_bytes()` один раз повторяет отклоненный `image_config={"image_size": "4K"}` запрос как `2K`.
51
52
  - `generate_imagen_bytes()` использует Imagen `generate_images` и передает requested MIME как `output_mime_type`.
52
- - Anthropic, OpenRouter и multi-provider failover не являются активными API в этой alpha-линейке.
53
+ - Anthropic, OpenRouter и multi-provider failover не являются активными API в этой линейке.
@@ -28,12 +28,13 @@ PromptResult | str
28
28
  ```
29
29
  prompt: str
30
30
  -> GeminiProvider.generate_image_bytes()
31
- -> GenerateContentConfig(response_modalities=[IMAGE])
31
+ -> GenerateContentConfig(response_modalities=[IMAGE], image_config=...)
32
32
  -> first inline_data image part
33
33
  -> (bytes, actual_mime_type)
34
34
  ```
35
35
 
36
36
  `response_mime_type` в этом пути не передается в `GenerateContentConfig.response_mime_type`; он используется только как fallback content type, если Gemini не вернул `inline_data.mime_type`.
37
+ Если `image_config.image_size` равен `4K` и Gemini отклоняет запрос, провайдер один раз повторяет его с `image_size="2K"`.
37
38
 
38
39
  ## Imagen Images
39
40
 
@@ -20,7 +20,7 @@ classifiers = [
20
20
  ]
21
21
  dependencies = [
22
22
  "pydantic>=2.0,<3.0",
23
- "codex-core>=0.2.2,<0.4.0",
23
+ "codex-core>=0.2.2,<0.5.0",
24
24
  ]
25
25
 
26
26
  [project.urls]
@@ -31,7 +31,7 @@ Issues = "https://github.com/codexdlc/codex-ai/issues"
31
31
 
32
32
  [project.optional-dependencies]
33
33
  openai = ["openai>=1.0,<2.0"]
34
- gemini = ["google-genai==1.68.0"]
34
+ gemini = ["google-genai==2.3.0"]
35
35
  all = [
36
36
  "codex-ai[openai,gemini]",
37
37
  ]
@@ -1,7 +1,7 @@
1
1
  """
2
2
  codex_ai.core
3
3
  =============
4
- Core types, contracts, and dispatching logic for the LLM abstraction layer.
4
+ Core legacy text router, dispatcher, and provider compatibility contracts.
5
5
  """
6
6
 
7
7
  from .dispatcher import LLMDispatcher
@@ -1,10 +1,10 @@
1
1
  """
2
2
  codex_ai.core.dispatcher
3
3
  =========================
4
- LLMDispatcher — orchestrates prompt building and LLM provider calls.
4
+ LLMDispatcher — legacy text prompt dispatcher plus direct provider delegation.
5
5
 
6
6
  Registers routers, selects the correct builder by mode,
7
- and delegates response generation to the provider.
7
+ and delegates text response generation to the provider's ``answer()`` method.
8
8
  """
9
9
 
10
10
  from __future__ import annotations
@@ -27,13 +27,14 @@ log = logging.getLogger(__name__)
27
27
 
28
28
  class LLMDispatcher:
29
29
  """
30
- Orchestrates prompt building and LLM response generation.
30
+ Orchestrates legacy text prompt building and provider calls.
31
31
 
32
32
  Connects one or more LLMRouters, selects the builder by mode,
33
- calls it with provided kwargs, then passes the result to the provider.
33
+ calls it with provided kwargs, then passes the result to the provider's
34
+ text compatibility ``answer()`` method.
34
35
 
35
36
  Args:
36
- provider: LLM backend implementing LLMProviderProtocol.
37
+ provider: Text-compatible provider implementing LLMProviderProtocol.
37
38
 
38
39
  Example:
39
40
  ```python
@@ -96,6 +97,7 @@ class LLMDispatcher:
96
97
  *,
97
98
  model: str | None = None,
98
99
  response_mime_type: str = "image/webp",
100
+ image_config: dict[str, Any] | None = None,
99
101
  **kwargs: Any,
100
102
  ) -> tuple[bytes, str]:
101
103
  """
@@ -114,6 +116,7 @@ class LLMDispatcher:
114
116
  prompt,
115
117
  model=model,
116
118
  response_mime_type=response_mime_type,
119
+ image_config=image_config,
117
120
  **kwargs,
118
121
  )
119
122
 
@@ -1,10 +1,10 @@
1
1
  """
2
2
  codex_ai.core.protocol
3
3
  =======================
4
- Core types and contracts for the LLM abstraction layer.
4
+ Core types and contracts for legacy text routing and direct provider helpers.
5
5
 
6
6
  PromptResult — frozen DTO returned by every prompt builder.
7
- LLMProviderProtocol — adapter contract for LLM backends (OpenAI, Gemini, etc.).
7
+ LLMProviderProtocol — legacy text compatibility contract for ``answer()``.
8
8
  TextGenerationProvider — optional adapter contract for direct text generation.
9
9
  JsonGenerationProvider — optional adapter contract for direct JSON generation.
10
10
  ImageGenerationProvider — optional adapter contract for binary image generation.
@@ -39,11 +39,10 @@ class PromptResult(BaseDTO):
39
39
  """
40
40
  Frozen DTO produced by a prompt builder.
41
41
 
42
- Passed directly to the LLM provider's ``answer()`` method.
42
+ Passed directly to the provider's legacy text ``answer()`` method.
43
43
 
44
44
  Attributes:
45
- messages: Provider-specific message list (OpenAI ChatCompletionMessageParam format
46
- or Gemini contents). Each provider interprets this field as needed.
45
+ messages: Ordered text message list consumed by compatibility adapters.
47
46
  system: Optional system/developer instruction (top-level string, used by Gemini
48
47
  and OpenAI o-series models that accept a dedicated system field).
49
48
 
@@ -68,9 +67,11 @@ class PromptResult(BaseDTO):
68
67
  @runtime_checkable
69
68
  class LLMProviderProtocol(Protocol):
70
69
  """
71
- Adapter contract for LLM backends.
70
+ Legacy text adapter contract.
72
71
 
73
- Implement this protocol to add a new provider (OpenAI, Gemini, Anthropic, etc.).
72
+ New Gemini integrations should prefer direct provider methods such as
73
+ ``generate_text()``, ``generate_json()``, and ``generate_image_bytes()``.
74
+ This protocol remains for router/dispatcher text compatibility.
74
75
 
75
76
  Example:
76
77
  ```python
@@ -82,14 +83,14 @@ class LLMProviderProtocol(Protocol):
82
83
 
83
84
  async def answer(self, prompt: PromptResult, **kw: Any) -> str:
84
85
  """
85
- Send prompt to the LLM and return the response text.
86
+ Send a legacy text prompt and return the response text.
86
87
 
87
88
  Args:
88
89
  prompt: Frozen DTO with messages and system instruction.
89
90
  **kw: Extra provider-specific kwargs (temperature, max_tokens, etc.).
90
91
 
91
92
  Returns:
92
- Response text from the LLM.
93
+ Response text from the provider.
93
94
  """
94
95
  ...
95
96
 
@@ -148,6 +149,7 @@ class ImageGenerationProvider(Protocol):
148
149
  *,
149
150
  model: str | None = None,
150
151
  response_mime_type: str = "image/webp",
152
+ image_config: dict[str, Any] | None = None,
151
153
  **kwargs: Any,
152
154
  ) -> tuple[bytes, str]:
153
155
  """
@@ -157,6 +159,8 @@ class ImageGenerationProvider(Protocol):
157
159
  prompt: Plain image-generation prompt.
158
160
  model: Optional image model override.
159
161
  response_mime_type: Requested/preferred image MIME type.
162
+ image_config: Optional image generation controls such as
163
+ ``{"aspect_ratio": "1:1", "image_size": "4K"}``.
160
164
  **kwargs: Extra provider-specific kwargs.
161
165
 
162
166
  Returns:
@@ -1,7 +1,7 @@
1
1
  """
2
2
  codex_ai.providers
3
3
  ==================
4
- Provider implementations (Gemini and OpenAI).
4
+ Provider adapters. Gemini is the primary API; OpenAI is text-only.
5
5
 
6
6
  Providers are lazy-loaded to avoid mandatory dependency on all SDK packages.
7
7
  """
@@ -1,7 +1,7 @@
1
1
  """
2
2
  codex_ai.providers.gemini
3
3
  ==========================
4
- GeminiProvider — LLM provider backed by Google Gemini (google-genai).
4
+ GeminiProvider — Gemini-first direct API backed by google-genai.
5
5
 
6
6
  Requires: ``pip install codex-ai[gemini]``
7
7
  """
@@ -34,9 +34,10 @@ _DEFAULT_IMAGEN_MODEL = "imagen-3.0-generate-002"
34
34
 
35
35
  class GeminiProvider:
36
36
  """
37
- LLM provider using Google Gemini via the google-genai SDK.
37
+ Direct Gemini adapter using the google-genai SDK.
38
38
 
39
- Implements LLMProviderProtocol.
39
+ Implements legacy text compatibility through ``answer()`` and exposes
40
+ direct Gemini methods for text, JSON, Gemini image, and Imagen generation.
40
41
 
41
42
  Args:
42
43
  api_key: Google AI API key.
@@ -186,6 +187,7 @@ class GeminiProvider:
186
187
  *,
187
188
  model: str | None = None,
188
189
  response_mime_type: str = "image/webp",
190
+ image_config: dict[str, Any] | None = None,
189
191
  **kwargs: Any,
190
192
  ) -> tuple[bytes, str]:
191
193
  """
@@ -206,29 +208,75 @@ class GeminiProvider:
206
208
  runtime_kw = kwargs.copy()
207
209
  runtime_kw.pop("model", None)
208
210
  runtime_kw.pop("response_mime_type", None)
211
+ runtime_kw.pop("image_config", None)
209
212
 
210
- config = genai_types.GenerateContentConfig(
211
- response_modalities=[genai_types.Modality.IMAGE],
213
+ config_kwargs: dict[str, Any] = {
214
+ "response_modalities": [genai_types.Modality.IMAGE],
212
215
  **runtime_kw,
213
- )
216
+ }
217
+ if image_config is not None:
218
+ config_kwargs["image_config"] = image_config
214
219
 
215
220
  try:
216
- response = await self._client.aio.models.generate_content(
221
+ return await self._generate_image_once(
217
222
  model=selected_model,
218
- contents=prompt,
219
- config=config,
223
+ prompt=prompt,
224
+ config_kwargs=config_kwargs,
225
+ fallback_mime=requested_mime,
220
226
  )
221
- image = self._extract_first_inline_image(response, fallback_mime=requested_mime)
222
- if image is not None:
223
- return image
224
-
225
- detail = self._describe_non_image_response(response)
226
- raise LLMProviderError(f"Gemini image generation did not return image data{detail}")
227
227
  except LLMProviderError:
228
228
  raise
229
229
  except Exception as exc:
230
+ fallback_config_kwargs = self._fallback_4k_image_config_to_2k(config_kwargs)
231
+ if fallback_config_kwargs is not None:
232
+ try:
233
+ return await self._generate_image_once(
234
+ model=selected_model,
235
+ prompt=prompt,
236
+ config_kwargs=fallback_config_kwargs,
237
+ fallback_mime=requested_mime,
238
+ )
239
+ except LLMProviderError:
240
+ raise
241
+ except Exception as fallback_exc:
242
+ raise LLMProviderError(f"Gemini image generation error: {fallback_exc}") from fallback_exc
243
+
230
244
  raise LLMProviderError(f"Gemini image generation error: {exc}") from exc
231
245
 
246
+ async def _generate_image_once(
247
+ self,
248
+ *,
249
+ model: str,
250
+ prompt: str,
251
+ config_kwargs: dict[str, Any],
252
+ fallback_mime: str,
253
+ ) -> tuple[bytes, str]:
254
+ config = genai_types.GenerateContentConfig(**config_kwargs)
255
+ response = await self._client.aio.models.generate_content(
256
+ model=model,
257
+ contents=prompt,
258
+ config=config,
259
+ )
260
+ image = self._extract_first_inline_image(response, fallback_mime=fallback_mime)
261
+ if image is not None:
262
+ return image
263
+
264
+ detail = self._describe_non_image_response(response)
265
+ raise LLMProviderError(f"Gemini image generation did not return image data{detail}")
266
+
267
+ @staticmethod
268
+ def _fallback_4k_image_config_to_2k(config_kwargs: dict[str, Any]) -> dict[str, Any] | None:
269
+ image_config = config_kwargs.get("image_config")
270
+ if not isinstance(image_config, dict) or image_config.get("image_size") != "4K":
271
+ return None
272
+
273
+ fallback_image_config = image_config.copy()
274
+ fallback_image_config["image_size"] = "2K"
275
+
276
+ fallback_config_kwargs = config_kwargs.copy()
277
+ fallback_config_kwargs["image_config"] = fallback_image_config
278
+ return fallback_config_kwargs
279
+
232
280
  async def generate_imagen_bytes(
233
281
  self,
234
282
  prompt: str,
@@ -1,7 +1,7 @@
1
1
  """
2
2
  codex_ai.providers.openai
3
3
  ==========================
4
- OpenAIProvider — LLM provider backed by OpenAI's Chat Completions API.
4
+ OpenAIProvider — text-only adapter backed by OpenAI's Chat Completions API.
5
5
 
6
6
  Requires: ``pip install codex-ai[openai]``
7
7
  """
@@ -28,9 +28,10 @@ _DEFAULT_MODEL = "gpt-4o-mini"
28
28
 
29
29
  class OpenAIProvider:
30
30
  """
31
- LLM provider using OpenAI Chat Completions.
31
+ Text-only adapter using OpenAI Chat Completions.
32
32
 
33
- Implements LLMProviderProtocol.
33
+ Implements legacy text compatibility through ``answer()`` and direct
34
+ ``generate_text()`` convenience.
34
35
 
35
36
  Args:
36
37
  api_key: OpenAI API key.
@@ -126,9 +126,10 @@ class ImageProvider:
126
126
  *,
127
127
  model: str | None = None,
128
128
  response_mime_type: str = "image/webp",
129
+ image_config: dict | None = None,
129
130
  **kwargs,
130
131
  ) -> tuple[bytes, str]:
131
- self.calls.append((prompt, model, response_mime_type, kwargs))
132
+ self.calls.append((prompt, model, response_mime_type, image_config, kwargs))
132
133
  return b"image-bytes", "image/png"
133
134
 
134
135
  async def generate_imagen_bytes(
@@ -159,11 +160,14 @@ async def test_dispatcher_generate_image_bytes_delegates_to_image_provider():
159
160
  "draw a castle",
160
161
  model="gemini-image",
161
162
  response_mime_type="image/webp",
163
+ image_config={"aspect_ratio": "1:1", "image_size": "4K"},
162
164
  seed=123,
163
165
  )
164
166
 
165
167
  assert result == (b"image-bytes", "image/png")
166
- assert provider.calls == [("draw a castle", "gemini-image", "image/webp", {"seed": 123})]
168
+ assert provider.calls == [
169
+ ("draw a castle", "gemini-image", "image/webp", {"aspect_ratio": "1:1", "image_size": "4K"}, {"seed": 123})
170
+ ]
167
171
 
168
172
 
169
173
  async def test_dispatcher_generate_image_bytes_raises_for_unsupported_provider(mock_provider):
@@ -106,6 +106,7 @@ def test_image_generation_provider_structural_check():
106
106
  *,
107
107
  model: str | None = None,
108
108
  response_mime_type: str = "image/webp",
109
+ image_config: dict | None = None,
109
110
  **kwargs,
110
111
  ) -> tuple[bytes, str]:
111
112
  return b"image", response_mime_type
@@ -418,14 +418,43 @@ async def test_gemini_generate_image_bytes_config_requests_image_modality_not_te
418
418
  with patch("codex_ai.providers.gemini.genai_types") as mock_types:
419
419
  mock_types.Modality.IMAGE = "IMAGE"
420
420
  mock_types.GenerateContentConfig.return_value = MagicMock()
421
- await provider.generate_image_bytes("draw a castle", response_mime_type="image/webp", seed=7)
421
+ await provider.generate_image_bytes(
422
+ "draw a castle",
423
+ response_mime_type="image/png",
424
+ image_config={"aspect_ratio": "1:1", "image_size": "4K"},
425
+ seed=7,
426
+ )
422
427
 
423
428
  config_kwargs = mock_types.GenerateContentConfig.call_args.kwargs
424
429
  assert config_kwargs["response_modalities"] == ["IMAGE"]
425
430
  assert "response_mime_type" not in config_kwargs
431
+ assert config_kwargs["image_config"] == {"aspect_ratio": "1:1", "image_size": "4K"}
426
432
  assert config_kwargs["seed"] == 7
427
433
 
428
434
 
435
+ async def test_gemini_generate_image_bytes_falls_back_from_4k_to_2k_when_config_rejected():
436
+ provider, mock_generate, _ = _make_provider()
437
+ mock_generate.side_effect = [
438
+ ValueError("unsupported image_size"),
439
+ _image_response(data=b"png", mime_type="image/png"),
440
+ ]
441
+
442
+ with patch("codex_ai.providers.gemini.genai_types") as mock_types:
443
+ mock_types.Modality.IMAGE = "IMAGE"
444
+ mock_types.GenerateContentConfig.side_effect = lambda **kwargs: kwargs
445
+ result = await provider.generate_image_bytes(
446
+ "draw a castle",
447
+ response_mime_type="image/png",
448
+ image_config={"aspect_ratio": "1:1", "image_size": "4K"},
449
+ )
450
+
451
+ assert result == (b"png", "image/png")
452
+ first_config = mock_generate.call_args_list[0].kwargs["config"]
453
+ second_config = mock_generate.call_args_list[1].kwargs["config"]
454
+ assert first_config["image_config"] == {"aspect_ratio": "1:1", "image_size": "4K"}
455
+ assert second_config["image_config"] == {"aspect_ratio": "1:1", "image_size": "2K"}
456
+
457
+
429
458
  async def test_gemini_generate_image_bytes_returns_inline_image_bytes_and_actual_mime():
430
459
  provider, mock_generate, _ = _make_provider()
431
460
  mock_generate.return_value = _image_response(data=b"png-bytes", mime_type="image/png")
@@ -344,9 +344,9 @@ requires-dist = [
344
344
  { name = "bandit", marker = "extra == 'dev'", specifier = ">=1.7" },
345
345
  { name = "codex-core", specifier = ">=0.2.2,<0.4.0" },
346
346
  { name = "detect-secrets", marker = "extra == 'dev'", specifier = ">=1.5" },
347
- { name = "google-genai", marker = "extra == 'all'", specifier = "==1.68.0" },
348
- { name = "google-genai", marker = "extra == 'dev'", specifier = "==1.68.0" },
349
- { name = "google-genai", marker = "extra == 'gemini'", specifier = "==1.68.0" },
347
+ { name = "google-genai", marker = "extra == 'all'", specifier = "==2.3.0" },
348
+ { name = "google-genai", marker = "extra == 'dev'", specifier = "==2.3.0" },
349
+ { name = "google-genai", marker = "extra == 'gemini'", specifier = "==2.3.0" },
350
350
  { name = "mike", marker = "extra == 'docs'", specifier = ">=2.0" },
351
351
  { name = "mkdocs", marker = "extra == 'docs'", specifier = ">=1.5" },
352
352
  { name = "mkdocs-include-markdown-plugin", marker = "extra == 'docs'" },
@@ -622,7 +622,7 @@ requests = [
622
622
 
623
623
  [[package]]
624
624
  name = "google-genai"
625
- version = "1.68.0"
625
+ version = "2.3.0"
626
626
  source = { registry = "https://pypi.org/simple" }
627
627
  dependencies = [
628
628
  { name = "anyio" },
@@ -636,9 +636,9 @@ dependencies = [
636
636
  { name = "typing-extensions" },
637
637
  { name = "websockets" },
638
638
  ]
639
- sdist = { url = "https://files.pythonhosted.org/packages/9c/2c/f059982dbcb658cc535c81bbcbe7e2c040d675f4b563b03cdb01018a4bc3/google_genai-1.68.0.tar.gz", hash = "sha256:ac30c0b8bc630f9372993a97e4a11dae0e36f2e10d7c55eacdca95a9fa14ca96", size = 511285, upload-time = "2026-03-18T01:03:18.243Z" }
639
+ sdist = { url = "https://files.pythonhosted.org/packages/02/8e/dfa4b34dd4c0baffccf6466fc68d6d35011662d43e7d79accb902320db74/google_genai-2.3.0.tar.gz", hash = "sha256:e877c750a4ccacdd9928fc3aa8ca8820ce85cade0ca51bd83feceacf5959b579", size = 546930, upload-time = "2026-05-15T06:22:36.264Z" }
640
640
  wheels = [
641
- { url = "https://files.pythonhosted.org/packages/84/de/7d3ee9c94b74c3578ea4f88d45e8de9405902f857932334d81e89bce3dfa/google_genai-1.68.0-py3-none-any.whl", hash = "sha256:a1bc9919c0e2ea2907d1e319b65471d3d6d58c54822039a249fe1323e4178d15", size = 750912, upload-time = "2026-03-18T01:03:15.983Z" },
641
+ { url = "https://files.pythonhosted.org/packages/b4/6e/aa6b30b09f58b946750fc4089c5248fbd3576f746e0e818d88633559dc84/google_genai-2.3.0-py3-none-any.whl", hash = "sha256:89d3c71c9f5f5b931b405b88a5837aea2bd4d27ed90323b9599f5760bbb91d92", size = 805484, upload-time = "2026-05-15T06:22:34.247Z" },
642
642
  ]
643
643
 
644
644
  [[package]]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes