codex-ai 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {codex_ai-0.2.2 → codex_ai-0.2.4}/CHANGELOG.md +9 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/PKG-INFO +22 -15
- {codex_ai-0.2.2 → codex_ai-0.2.4}/README.md +17 -10
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/core/README.md +19 -15
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/core/data_flow.md +5 -3
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/providers/README.md +5 -4
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/architecture/providers/data_flow.md +2 -1
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/core/README.md +18 -15
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/core/data_flow.md +5 -3
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/providers/README.md +5 -4
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/ru/architecture/providers/data_flow.md +2 -1
- {codex_ai-0.2.2 → codex_ai-0.2.4}/pyproject.toml +2 -2
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/__init__.py +1 -1
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/dispatcher.py +8 -5
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/protocol.py +13 -9
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/__init__.py +1 -1
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/gemini.py +63 -15
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/providers/openai.py +4 -3
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_dispatcher.py +6 -2
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_protocol.py +1 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/test_gemini_provider.py +30 -1
- {codex_ai-0.2.2 → codex_ai-0.2.4}/uv.lock +6 -6
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/ci.yml +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/docs.yml +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.github/workflows/publish.yml +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.gitignore +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.nojekyll +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.pre-commit-config.yaml +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.python-version +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/.secrets.baseline +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/LICENSE +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/changelog.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/dispatcher.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/exceptions.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/protocol.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/router.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/core/sync.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/index.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/providers/gemini.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/en/api/providers/openai.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/index.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/docs/stylesheets/extra.css +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/mkdocs.yml +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/exceptions.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/router.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/core/sync.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/src/codex_ai/py.typed +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/conftest.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/conftest.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/integration/test_providers_integration.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/conftest.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_exceptions.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_router.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/core/test_sync.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/providers/test_openai_provider.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tests/unit/test_public_api.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/README.md +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/__init__.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/check.py +0 -0
- {codex_ai-0.2.2 → codex_ai-0.2.4}/tools/dev/generate_project_tree.py +0 -0
|
@@ -4,6 +4,15 @@ All notable changes to this project will be documented in this file.
|
|
|
4
4
|
|
|
5
5
|
## [Unreleased]
|
|
6
6
|
|
|
7
|
+
## [0.2.3] - 2026-05-17
|
|
8
|
+
|
|
9
|
+
### Added
|
|
10
|
+
- Added explicit Gemini image `image_config` support for aspect ratio and image size controls.
|
|
11
|
+
- Added a Gemini image retry from `image_size="4K"` to `image_size="2K"` when the initial 4K request is rejected.
|
|
12
|
+
|
|
13
|
+
### Changed
|
|
14
|
+
- Updated the Gemini extra to `google-genai==2.3.0` for the SDK image configuration contract.
|
|
15
|
+
|
|
7
16
|
## [0.2.2] - 2026-05-15
|
|
8
17
|
|
|
9
18
|
### Fixed
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: codex-ai
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: Gemini-first and OpenAI provider helpers for Codex
|
|
5
5
|
Project-URL: Homepage, https://github.com/codexdlc/codex-ai
|
|
6
6
|
Project-URL: Documentation, https://codexdlc.github.io/codex-ai/
|
|
@@ -16,15 +16,15 @@ Classifier: License :: OSI Approved :: Apache Software License
|
|
|
16
16
|
Classifier: Programming Language :: Python :: 3.12
|
|
17
17
|
Classifier: Programming Language :: Python :: 3.13
|
|
18
18
|
Requires-Python: >=3.12
|
|
19
|
-
Requires-Dist: codex-core<0.
|
|
19
|
+
Requires-Dist: codex-core<0.5.0,>=0.2.2
|
|
20
20
|
Requires-Dist: pydantic<3.0,>=2.0
|
|
21
21
|
Provides-Extra: all
|
|
22
|
-
Requires-Dist: google-genai==
|
|
22
|
+
Requires-Dist: google-genai==2.3.0; extra == 'all'
|
|
23
23
|
Requires-Dist: openai<2.0,>=1.0; extra == 'all'
|
|
24
24
|
Provides-Extra: dev
|
|
25
25
|
Requires-Dist: bandit>=1.7; extra == 'dev'
|
|
26
26
|
Requires-Dist: detect-secrets>=1.5; extra == 'dev'
|
|
27
|
-
Requires-Dist: google-genai==
|
|
27
|
+
Requires-Dist: google-genai==2.3.0; extra == 'dev'
|
|
28
28
|
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
29
29
|
Requires-Dist: openai<2.0,>=1.0; extra == 'dev'
|
|
30
30
|
Requires-Dist: pip-audit>=2.7; extra == 'dev'
|
|
@@ -40,7 +40,7 @@ Requires-Dist: mkdocs-material>=9.0; extra == 'docs'
|
|
|
40
40
|
Requires-Dist: mkdocs>=1.5; extra == 'docs'
|
|
41
41
|
Requires-Dist: mkdocstrings[python]>=0.24; extra == 'docs'
|
|
42
42
|
Provides-Extra: gemini
|
|
43
|
-
Requires-Dist: google-genai==
|
|
43
|
+
Requires-Dist: google-genai==2.3.0; extra == 'gemini'
|
|
44
44
|
Provides-Extra: openai
|
|
45
45
|
Requires-Dist: openai<2.0,>=1.0; extra == 'openai'
|
|
46
46
|
Description-Content-Type: text/markdown
|
|
@@ -49,10 +49,10 @@ Description-Content-Type: text/markdown
|
|
|
49
49
|
|
|
50
50
|
[](https://pypi.org/project/codex-ai/)
|
|
51
51
|
[](https://pypi.org/project/codex-ai/)
|
|
52
|
-
[](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
|
|
53
53
|
[](https://github.com/codexdlc/codex-ai/blob/main/LICENSE)
|
|
54
54
|
|
|
55
|
-
Gemini-first
|
|
55
|
+
Gemini-first API helpers for the Codex ecosystem. The active surface is direct Gemini text, JSON, Gemini image, and Imagen generation. OpenAI remains as a small text-only adapter, and the router/dispatcher layer is kept for legacy text workflows.
|
|
56
56
|
|
|
57
57
|
## Install
|
|
58
58
|
|
|
@@ -83,9 +83,10 @@ gemini = GeminiProvider(api_key="AIza...")
|
|
|
83
83
|
text = await gemini.generate_text("Write one short tavern rumor.")
|
|
84
84
|
loot = await gemini.generate_json("Create one loot item.", schema=LootItem)
|
|
85
85
|
image_bytes, content_type = await gemini.generate_image_bytes(
|
|
86
|
-
"
|
|
87
|
-
model="gemini-
|
|
88
|
-
response_mime_type="image/
|
|
86
|
+
"Square tactical dark fantasy ruined capital city map, no labels.",
|
|
87
|
+
model="gemini-3-pro-image-preview",
|
|
88
|
+
response_mime_type="image/png",
|
|
89
|
+
image_config={"aspect_ratio": "1:1", "image_size": "4K"},
|
|
89
90
|
)
|
|
90
91
|
|
|
91
92
|
imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
|
|
@@ -98,11 +99,13 @@ imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
|
|
|
98
99
|
|
|
99
100
|
`generate_image_bytes()` targets Gemini image models through `generate_content` and treats
|
|
100
101
|
`response_mime_type` as a preferred/fallback MIME type. It does not pass image MIME values
|
|
101
|
-
to Gemini's text `response_mime_type` config field.
|
|
102
|
+
to Gemini's text `response_mime_type` config field. Pass Gemini image controls such as
|
|
103
|
+
`aspect_ratio` and `image_size` with `image_config`; if a `4K` request is rejected,
|
|
104
|
+
the Gemini provider retries once with `2K`. Use `generate_imagen_bytes()` for
|
|
102
105
|
Imagen models; that path uses `generate_images` and passes the requested MIME as
|
|
103
106
|
`output_mime_type`.
|
|
104
107
|
|
|
105
|
-
## Router
|
|
108
|
+
## Legacy Text Router
|
|
106
109
|
|
|
107
110
|
```python
|
|
108
111
|
from codex_ai import GeminiProvider, LLMDispatcher, LLMMessage, LLMRouter, PromptResult
|
|
@@ -124,13 +127,17 @@ dispatcher.include_router(router)
|
|
|
124
127
|
response = await dispatcher.process("chat", text="Hello!")
|
|
125
128
|
```
|
|
126
129
|
|
|
130
|
+
Use this path only when you already have prompt builders registered through `LLMRouter`.
|
|
131
|
+
New Gemini integrations should call `generate_text()`, `generate_json()`,
|
|
132
|
+
`generate_image_bytes()`, or `generate_imagen_bytes()` directly.
|
|
133
|
+
|
|
127
134
|
## Modules
|
|
128
135
|
|
|
129
136
|
| Module | Extra | Description |
|
|
130
137
|
| :--- | :--- | :--- |
|
|
131
|
-
| `codex_ai.
|
|
132
|
-
| `codex_ai.providers.
|
|
133
|
-
| `codex_ai.
|
|
138
|
+
| `codex_ai.providers.gemini` | `[gemini]` | Primary API: Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
|
|
139
|
+
| `codex_ai.providers.openai` | `[openai]` | Text-only OpenAI Chat Completions adapter |
|
|
140
|
+
| `codex_ai.core` | - | Legacy text router/dispatcher contracts and shared provider exceptions |
|
|
134
141
|
|
|
135
142
|
## Development
|
|
136
143
|
|
|
@@ -2,10 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
[](https://pypi.org/project/codex-ai/)
|
|
4
4
|
[](https://pypi.org/project/codex-ai/)
|
|
5
|
-
[](https://github.com/codexdlc/codex-ai/actions/workflows/ci.yml)
|
|
6
6
|
[](https://github.com/codexdlc/codex-ai/blob/main/LICENSE)
|
|
7
7
|
|
|
8
|
-
Gemini-first
|
|
8
|
+
Gemini-first API helpers for the Codex ecosystem. The active surface is direct Gemini text, JSON, Gemini image, and Imagen generation. OpenAI remains as a small text-only adapter, and the router/dispatcher layer is kept for legacy text workflows.
|
|
9
9
|
|
|
10
10
|
## Install
|
|
11
11
|
|
|
@@ -36,9 +36,10 @@ gemini = GeminiProvider(api_key="AIza...")
|
|
|
36
36
|
text = await gemini.generate_text("Write one short tavern rumor.")
|
|
37
37
|
loot = await gemini.generate_json("Create one loot item.", schema=LootItem)
|
|
38
38
|
image_bytes, content_type = await gemini.generate_image_bytes(
|
|
39
|
-
"
|
|
40
|
-
model="gemini-
|
|
41
|
-
response_mime_type="image/
|
|
39
|
+
"Square tactical dark fantasy ruined capital city map, no labels.",
|
|
40
|
+
model="gemini-3-pro-image-preview",
|
|
41
|
+
response_mime_type="image/png",
|
|
42
|
+
image_config={"aspect_ratio": "1:1", "image_size": "4K"},
|
|
42
43
|
)
|
|
43
44
|
|
|
44
45
|
imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
|
|
@@ -51,11 +52,13 @@ imagen_bytes, imagen_content_type = await gemini.generate_imagen_bytes(
|
|
|
51
52
|
|
|
52
53
|
`generate_image_bytes()` targets Gemini image models through `generate_content` and treats
|
|
53
54
|
`response_mime_type` as a preferred/fallback MIME type. It does not pass image MIME values
|
|
54
|
-
to Gemini's text `response_mime_type` config field.
|
|
55
|
+
to Gemini's text `response_mime_type` config field. Pass Gemini image controls such as
|
|
56
|
+
`aspect_ratio` and `image_size` with `image_config`; if a `4K` request is rejected,
|
|
57
|
+
the Gemini provider retries once with `2K`. Use `generate_imagen_bytes()` for
|
|
55
58
|
Imagen models; that path uses `generate_images` and passes the requested MIME as
|
|
56
59
|
`output_mime_type`.
|
|
57
60
|
|
|
58
|
-
## Router
|
|
61
|
+
## Legacy Text Router
|
|
59
62
|
|
|
60
63
|
```python
|
|
61
64
|
from codex_ai import GeminiProvider, LLMDispatcher, LLMMessage, LLMRouter, PromptResult
|
|
@@ -77,13 +80,17 @@ dispatcher.include_router(router)
|
|
|
77
80
|
response = await dispatcher.process("chat", text="Hello!")
|
|
78
81
|
```
|
|
79
82
|
|
|
83
|
+
Use this path only when you already have prompt builders registered through `LLMRouter`.
|
|
84
|
+
New Gemini integrations should call `generate_text()`, `generate_json()`,
|
|
85
|
+
`generate_image_bytes()`, or `generate_imagen_bytes()` directly.
|
|
86
|
+
|
|
80
87
|
## Modules
|
|
81
88
|
|
|
82
89
|
| Module | Extra | Description |
|
|
83
90
|
| :--- | :--- | :--- |
|
|
84
|
-
| `codex_ai.
|
|
85
|
-
| `codex_ai.providers.
|
|
86
|
-
| `codex_ai.
|
|
91
|
+
| `codex_ai.providers.gemini` | `[gemini]` | Primary API: Gemini text, JSON, Gemini image, and Imagen generation via pinned `google-genai` |
|
|
92
|
+
| `codex_ai.providers.openai` | `[openai]` | Text-only OpenAI Chat Completions adapter |
|
|
93
|
+
| `codex_ai.core` | - | Legacy text router/dispatcher contracts and shared provider exceptions |
|
|
87
94
|
|
|
88
95
|
## Development
|
|
89
96
|
|
|
@@ -2,19 +2,22 @@
|
|
|
2
2
|
|
|
3
3
|
## Purpose
|
|
4
4
|
|
|
5
|
-
`codex_ai.core` is the orchestration layer
|
|
5
|
+
`codex_ai.core` is the legacy text orchestration layer. It keeps existing `LLMRouter`/`LLMDispatcher` prompt-builder workflows working while the active API surface moves to direct provider methods.
|
|
6
6
|
|
|
7
7
|
## Why It's a Module
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Older Codex integrations use mode-based prompt builders. This module keeps that shape stable without making it the primary abstraction for new work:
|
|
10
10
|
|
|
11
|
-
|
|
|
12
|
-
|
|
13
|
-
|
|
|
14
|
-
|
|
|
15
|
-
|
|
|
11
|
+
| Need | Current role |
|
|
12
|
+
|------|--------------|
|
|
13
|
+
| Keep registered prompt builders working | `LLMRouter` maps modes to builders |
|
|
14
|
+
| Run existing text flows without rewriting callers | `LLMDispatcher.process()` still calls `provider.answer()` |
|
|
15
|
+
| Bridge sync-only contexts | `SyncLLMDispatcher` remains available for CLI/WSGI code |
|
|
16
16
|
|
|
17
|
-
|
|
17
|
+
For new Gemini work, prefer `GeminiProvider.generate_text()`, `generate_json()`,
|
|
18
|
+
`generate_image_bytes()`, and `generate_imagen_bytes()` directly.
|
|
19
|
+
|
|
20
|
+
The retained text pipeline is:
|
|
18
21
|
|
|
19
22
|
```
|
|
20
23
|
@router.prompt("mode") → PromptResult (frozen DTO) → LLMProviderProtocol.answer()
|
|
@@ -32,7 +35,7 @@ Working directly with LLM SDKs across a codebase creates three recurring problem
|
|
|
32
35
|
│ include_router(router)
|
|
33
36
|
▼
|
|
34
37
|
┌──────────────────────┐
|
|
35
|
-
│ LLMDispatcher │
|
|
38
|
+
│ LLMDispatcher │ legacy text builder → provider
|
|
36
39
|
│ │
|
|
37
40
|
│ .process(mode, **kw)│
|
|
38
41
|
└──────┬───────────────┘
|
|
@@ -52,19 +55,20 @@ Working directly with LLM SDKs across a codebase creates three recurring problem
|
|
|
52
55
|
|
|
53
56
|
| Component | Class | Role |
|
|
54
57
|
|-----------|-------|------|
|
|
55
|
-
| `protocol.py` | `PromptResult` | Frozen DTO
|
|
56
|
-
| `protocol.py` | `LLMProviderProtocol` | Structural
|
|
58
|
+
| `protocol.py` | `PromptResult` | Frozen DTO used by legacy text prompt builders |
|
|
59
|
+
| `protocol.py` | `LLMProviderProtocol` | Structural text compatibility protocol for `answer()` |
|
|
57
60
|
| `protocol.py` | `PromptBuilder` | Type alias for `async def (...) -> PromptResult` |
|
|
58
61
|
| `router.py` | `LLMRouter` | Registry: maps `mode` strings to builder functions via decorator |
|
|
59
|
-
| `dispatcher.py` | `LLMDispatcher` |
|
|
62
|
+
| `dispatcher.py` | `LLMDispatcher` | Runs legacy text prompts and delegates direct provider convenience methods |
|
|
60
63
|
| `sync.py` | `SyncLLMDispatcher` | Wraps `LLMDispatcher` with `asyncio.run()` for WSGI/CLI contexts |
|
|
61
64
|
| `exceptions.py` | `LLMProviderError` | Base exception raised by all provider implementations |
|
|
62
65
|
|
|
63
66
|
## Key Design Decisions
|
|
64
67
|
|
|
65
|
-
- **
|
|
66
|
-
-
|
|
67
|
-
-
|
|
68
|
+
- **Direct provider APIs first** — Gemini capabilities are exposed as explicit methods instead of being forced through a universal provider interface.
|
|
69
|
+
- **Frozen DTO (`PromptResult`)** — kept for legacy builders and cannot be mutated downstream.
|
|
70
|
+
- **`@runtime_checkable` Protocol** — retained for runtime checks at compatibility boundaries.
|
|
71
|
+
- **Mode-based dispatch is legacy text infrastructure** — `dispatcher.process("chat", ...)` maps to a registered builder for existing flows.
|
|
68
72
|
- **All logs at `DEBUG`** — dispatcher emits only debug-level messages. Production log level controls visibility without code changes.
|
|
69
73
|
- **`SyncLLMDispatcher` for Django only** — uses `asyncio.run()` which creates a new event loop. Never call from inside an async context (ARQ, async views, bots) — use `LLMDispatcher` directly.
|
|
70
74
|
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# Core — Data Flow
|
|
2
2
|
|
|
3
|
-
## Request Lifecycle
|
|
3
|
+
## Legacy Text Request Lifecycle
|
|
4
|
+
|
|
5
|
+
This flow is retained for existing `LLMRouter` integrations. New Gemini work should use direct provider methods.
|
|
4
6
|
|
|
5
7
|
```
|
|
6
8
|
1. Application calls:
|
|
@@ -13,14 +15,14 @@
|
|
|
13
15
|
prompt = await builder(text="Hello!")
|
|
14
16
|
# → PromptResult(messages=[LLMMessage(role="user", content="Hello!")])
|
|
15
17
|
|
|
16
|
-
4. Dispatcher calls
|
|
18
|
+
4. Dispatcher calls the text compatibility method:
|
|
17
19
|
response = await self._provider.answer(prompt, **kw)
|
|
18
20
|
|
|
19
21
|
5. Provider returns text:
|
|
20
22
|
"Hi there! How can I help?"
|
|
21
23
|
```
|
|
22
24
|
|
|
23
|
-
## Component Interactions
|
|
25
|
+
## Legacy Component Interactions
|
|
24
26
|
|
|
25
27
|
```
|
|
26
28
|
Application
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Purpose
|
|
4
4
|
|
|
5
|
-
`codex_ai.providers` contains concrete
|
|
5
|
+
`codex_ai.providers` contains concrete SDK adapters, not a broad interchangeable provider framework. Gemini is the product focus; OpenAI is retained only for text generation.
|
|
6
6
|
|
|
7
7
|
Gemini is the primary target and exposes direct methods for text, JSON, and image generation:
|
|
8
8
|
|
|
@@ -13,7 +13,7 @@ await gemini.generate_image_bytes(...)
|
|
|
13
13
|
await gemini.generate_imagen_bytes(...)
|
|
14
14
|
```
|
|
15
15
|
|
|
16
|
-
OpenAI is kept as a text
|
|
16
|
+
OpenAI is kept as a text-only adapter with the same `generate_text(...)` convenience.
|
|
17
17
|
|
|
18
18
|
## Architecture
|
|
19
19
|
|
|
@@ -47,6 +47,7 @@ LLMRouter builder -> PromptResult -> LLMDispatcher.process() -> provider.answer(
|
|
|
47
47
|
- Gemini-specific capabilities are represented directly instead of being hidden behind a broad universal abstraction.
|
|
48
48
|
- JSON generation uses provider-native JSON configuration and still validates locally with `json.loads` and optional Pydantic models.
|
|
49
49
|
- Gemini image generation and Imagen generation are separate explicit methods because they use different SDK calls.
|
|
50
|
-
- `generate_image_bytes()` uses Gemini `generate_content` with image modality. Its `response_mime_type` is only a preferred/fallback MIME type
|
|
50
|
+
- `generate_image_bytes()` uses Gemini `generate_content` with image modality. Its `response_mime_type` is only a preferred/fallback MIME type, and Gemini image controls are passed through `image_config`.
|
|
51
|
+
- `generate_image_bytes()` retries a rejected `image_config={"image_size": "4K"}` request once as `2K`.
|
|
51
52
|
- `generate_imagen_bytes()` uses Imagen `generate_images` and passes the requested MIME as `output_mime_type`.
|
|
52
|
-
- Anthropic, OpenRouter, and multi-provider failover are not active APIs in this
|
|
53
|
+
- Anthropic, OpenRouter, and multi-provider failover are not active APIs in this line.
|
|
@@ -28,12 +28,13 @@ PromptResult | str
|
|
|
28
28
|
```
|
|
29
29
|
prompt: str
|
|
30
30
|
-> GeminiProvider.generate_image_bytes()
|
|
31
|
-
-> GenerateContentConfig(response_modalities=[IMAGE])
|
|
31
|
+
-> GenerateContentConfig(response_modalities=[IMAGE], image_config=...)
|
|
32
32
|
-> first inline_data image part
|
|
33
33
|
-> (bytes, actual_mime_type)
|
|
34
34
|
```
|
|
35
35
|
|
|
36
36
|
`response_mime_type` is not passed to `GenerateContentConfig.response_mime_type` on this path; it is only a fallback content type when Gemini omits `inline_data.mime_type`.
|
|
37
|
+
When `image_config.image_size` is `4K` and Gemini rejects the request, the provider retries once with `image_size` changed to `2K`.
|
|
37
38
|
|
|
38
39
|
## Imagen Images
|
|
39
40
|
|
|
@@ -2,19 +2,21 @@
|
|
|
2
2
|
|
|
3
3
|
## Назначение
|
|
4
4
|
|
|
5
|
-
`codex_ai.core` —
|
|
5
|
+
`codex_ai.core` — legacy-слой текстовой оркестрации. Он сохраняет существующие workflow на `LLMRouter`/`LLMDispatcher`, пока активная API-поверхность переехала в прямые методы провайдеров.
|
|
6
6
|
|
|
7
7
|
## Зачем это модуль
|
|
8
8
|
|
|
9
|
-
|
|
9
|
+
Старые Codex-интеграции используют mode-based prompt builders. Этот модуль сохраняет такую форму, но больше не является основной абстракцией для новой работы:
|
|
10
10
|
|
|
11
|
-
|
|
|
12
|
-
|
|
13
|
-
|
|
|
14
|
-
|
|
|
15
|
-
|
|
|
11
|
+
| Потребность | Текущая роль |
|
|
12
|
+
|-------------|--------------|
|
|
13
|
+
| Сохранить зарегистрированные prompt builders | `LLMRouter` связывает modes с билдерами |
|
|
14
|
+
| Не переписывать существующие текстовые вызовы | `LLMDispatcher.process()` продолжает вызывать `provider.answer()` |
|
|
15
|
+
| Поддержать sync-only контексты | `SyncLLMDispatcher` остается для CLI/WSGI кода |
|
|
16
16
|
|
|
17
|
-
`
|
|
17
|
+
Для новой Gemini-интеграции лучше использовать прямые методы `GeminiProvider.generate_text()`, `generate_json()`, `generate_image_bytes()` и `generate_imagen_bytes()`.
|
|
18
|
+
|
|
19
|
+
Сохраненный текстовый pipeline:
|
|
18
20
|
|
|
19
21
|
```
|
|
20
22
|
@router.prompt("mode") → PromptResult (frozen DTO) → LLMProviderProtocol.answer()
|
|
@@ -32,7 +34,7 @@
|
|
|
32
34
|
│ include_router(router)
|
|
33
35
|
▼
|
|
34
36
|
┌──────────────────────┐
|
|
35
|
-
│ LLMDispatcher │
|
|
37
|
+
│ LLMDispatcher │ legacy text builder → provider
|
|
36
38
|
│ │
|
|
37
39
|
│ .process(mode, **kw)│
|
|
38
40
|
└──────┬───────────────┘
|
|
@@ -52,19 +54,20 @@
|
|
|
52
54
|
|
|
53
55
|
| Компонент | Класс | Роль |
|
|
54
56
|
|-----------|-------|------|
|
|
55
|
-
| `protocol.py` | `PromptResult` | Frozen DTO
|
|
56
|
-
| `protocol.py` | `LLMProviderProtocol` | Структурный
|
|
57
|
+
| `protocol.py` | `PromptResult` | Frozen DTO для legacy text prompt builders |
|
|
58
|
+
| `protocol.py` | `LLMProviderProtocol` | Структурный текстовый compatibility-протокол для `answer()` |
|
|
57
59
|
| `protocol.py` | `PromptBuilder` | Type alias для `async def (...) -> PromptResult` |
|
|
58
60
|
| `router.py` | `LLMRouter` | Реестр: связывает строки `mode` с функциями-билдерами через декоратор |
|
|
59
|
-
| `dispatcher.py` | `LLMDispatcher` |
|
|
61
|
+
| `dispatcher.py` | `LLMDispatcher` | Выполняет legacy text prompts и делегирует прямые provider convenience methods |
|
|
60
62
|
| `sync.py` | `SyncLLMDispatcher` | Оборачивает `LLMDispatcher` через `asyncio.run()` для WSGI/CLI |
|
|
61
63
|
| `exceptions.py` | `LLMProviderError` | Базовое исключение, поднимаемое всеми реализациями провайдеров |
|
|
62
64
|
|
|
63
65
|
## Ключевые решения
|
|
64
66
|
|
|
65
|
-
- **
|
|
66
|
-
-
|
|
67
|
-
-
|
|
67
|
+
- **Direct provider APIs first** — возможности Gemini раскрываются явными методами, а не проталкиваются через универсальный provider interface.
|
|
68
|
+
- **Frozen DTO (`PromptResult`)** — сохранен для legacy builders и не может быть изменён downstream.
|
|
69
|
+
- **`@runtime_checkable` Protocol** — оставлен для runtime-проверок на compatibility boundaries.
|
|
70
|
+
- **Mode-based dispatch — legacy text infrastructure** — `dispatcher.process("chat", ...)` находит зарегистрированный билдер для существующих flows.
|
|
68
71
|
- **Все логи на `DEBUG`** — диспетчер пишет только debug-сообщения. Уровень логирования в продакшне управляет видимостью без изменений кода.
|
|
69
72
|
- **`SyncLLMDispatcher` только для Django** — использует `asyncio.run()`, создающий новый event loop. Никогда не вызывать из async-контекста (ARQ, async views, боты) — используйте `LLMDispatcher` напрямую.
|
|
70
73
|
|
|
@@ -1,6 +1,8 @@
|
|
|
1
1
|
# Core — Поток данных
|
|
2
2
|
|
|
3
|
-
## Жизненный цикл запроса
|
|
3
|
+
## Жизненный цикл legacy text запроса
|
|
4
|
+
|
|
5
|
+
Этот flow сохранен для существующих интеграций на `LLMRouter`. Новую Gemini-интеграцию лучше писать через прямые методы провайдера.
|
|
4
6
|
|
|
5
7
|
```
|
|
6
8
|
1. Приложение вызывает:
|
|
@@ -13,14 +15,14 @@
|
|
|
13
15
|
prompt = await builder(text="Привет!")
|
|
14
16
|
# → PromptResult(messages=[LLMMessage(role="user", content="Привет!")])
|
|
15
17
|
|
|
16
|
-
4. Диспетчер вызывает
|
|
18
|
+
4. Диспетчер вызывает текстовый compatibility method:
|
|
17
19
|
response = await self._provider.answer(prompt, **kw)
|
|
18
20
|
|
|
19
21
|
5. Провайдер возвращает текст:
|
|
20
22
|
"Привет! Чем могу помочь?"
|
|
21
23
|
```
|
|
22
24
|
|
|
23
|
-
## Взаимодействие
|
|
25
|
+
## Взаимодействие legacy-компонентов
|
|
24
26
|
|
|
25
27
|
```
|
|
26
28
|
Приложение
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
## Назначение
|
|
4
4
|
|
|
5
|
-
`codex_ai.providers` содержит
|
|
5
|
+
`codex_ai.providers` содержит конкретные SDK-адаптеры, а не широкий interchangeable provider framework. Фокус продукта — Gemini; OpenAI сохранен только для текстовой генерации.
|
|
6
6
|
|
|
7
7
|
Gemini является основным направлением и дает прямые методы для текста, JSON и картинок:
|
|
8
8
|
|
|
@@ -13,7 +13,7 @@ await gemini.generate_image_bytes(...)
|
|
|
13
13
|
await gemini.generate_imagen_bytes(...)
|
|
14
14
|
```
|
|
15
15
|
|
|
16
|
-
OpenAI оставлен как
|
|
16
|
+
OpenAI оставлен как text-only адаптер с `generate_text(...)`.
|
|
17
17
|
|
|
18
18
|
## Архитектура
|
|
19
19
|
|
|
@@ -47,6 +47,7 @@ LLMRouter builder -> PromptResult -> LLMDispatcher.process() -> provider.answer(
|
|
|
47
47
|
- Возможности Gemini представлены напрямую, без широкой универсальной абстракции.
|
|
48
48
|
- JSON generation использует native JSON config провайдера и локальную проверку через `json.loads` и optional Pydantic schema.
|
|
49
49
|
- Gemini image generation и Imagen generation разведены в отдельные явные методы, потому что они используют разные SDK calls.
|
|
50
|
-
- `generate_image_bytes()` использует Gemini `generate_content` с image modality. Его `response_mime_type` является только preferred/fallback MIME type
|
|
50
|
+
- `generate_image_bytes()` использует Gemini `generate_content` с image modality. Его `response_mime_type` является только preferred/fallback MIME type, а Gemini image controls передаются через `image_config`.
|
|
51
|
+
- `generate_image_bytes()` один раз повторяет отклоненный `image_config={"image_size": "4K"}` запрос как `2K`.
|
|
51
52
|
- `generate_imagen_bytes()` использует Imagen `generate_images` и передает requested MIME как `output_mime_type`.
|
|
52
|
-
- Anthropic, OpenRouter и multi-provider failover не являются активными API в этой
|
|
53
|
+
- Anthropic, OpenRouter и multi-provider failover не являются активными API в этой линейке.
|
|
@@ -28,12 +28,13 @@ PromptResult | str
|
|
|
28
28
|
```
|
|
29
29
|
prompt: str
|
|
30
30
|
-> GeminiProvider.generate_image_bytes()
|
|
31
|
-
-> GenerateContentConfig(response_modalities=[IMAGE])
|
|
31
|
+
-> GenerateContentConfig(response_modalities=[IMAGE], image_config=...)
|
|
32
32
|
-> first inline_data image part
|
|
33
33
|
-> (bytes, actual_mime_type)
|
|
34
34
|
```
|
|
35
35
|
|
|
36
36
|
`response_mime_type` в этом пути не передается в `GenerateContentConfig.response_mime_type`; он используется только как fallback content type, если Gemini не вернул `inline_data.mime_type`.
|
|
37
|
+
Если `image_config.image_size` равен `4K` и Gemini отклоняет запрос, провайдер один раз повторяет его с `image_size="2K"`.
|
|
37
38
|
|
|
38
39
|
## Imagen Images
|
|
39
40
|
|
|
@@ -20,7 +20,7 @@ classifiers = [
|
|
|
20
20
|
]
|
|
21
21
|
dependencies = [
|
|
22
22
|
"pydantic>=2.0,<3.0",
|
|
23
|
-
"codex-core>=0.2.2,<0.
|
|
23
|
+
"codex-core>=0.2.2,<0.5.0",
|
|
24
24
|
]
|
|
25
25
|
|
|
26
26
|
[project.urls]
|
|
@@ -31,7 +31,7 @@ Issues = "https://github.com/codexdlc/codex-ai/issues"
|
|
|
31
31
|
|
|
32
32
|
[project.optional-dependencies]
|
|
33
33
|
openai = ["openai>=1.0,<2.0"]
|
|
34
|
-
gemini = ["google-genai==
|
|
34
|
+
gemini = ["google-genai==2.3.0"]
|
|
35
35
|
all = [
|
|
36
36
|
"codex-ai[openai,gemini]",
|
|
37
37
|
]
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""
|
|
2
2
|
codex_ai.core.dispatcher
|
|
3
3
|
=========================
|
|
4
|
-
LLMDispatcher —
|
|
4
|
+
LLMDispatcher — legacy text prompt dispatcher plus direct provider delegation.
|
|
5
5
|
|
|
6
6
|
Registers routers, selects the correct builder by mode,
|
|
7
|
-
and delegates response generation to the provider.
|
|
7
|
+
and delegates text response generation to the provider's ``answer()`` method.
|
|
8
8
|
"""
|
|
9
9
|
|
|
10
10
|
from __future__ import annotations
|
|
@@ -27,13 +27,14 @@ log = logging.getLogger(__name__)
|
|
|
27
27
|
|
|
28
28
|
class LLMDispatcher:
|
|
29
29
|
"""
|
|
30
|
-
Orchestrates prompt building and
|
|
30
|
+
Orchestrates legacy text prompt building and provider calls.
|
|
31
31
|
|
|
32
32
|
Connects one or more LLMRouters, selects the builder by mode,
|
|
33
|
-
calls it with provided kwargs, then passes the result to the provider
|
|
33
|
+
calls it with provided kwargs, then passes the result to the provider's
|
|
34
|
+
text compatibility ``answer()`` method.
|
|
34
35
|
|
|
35
36
|
Args:
|
|
36
|
-
provider:
|
|
37
|
+
provider: Text-compatible provider implementing LLMProviderProtocol.
|
|
37
38
|
|
|
38
39
|
Example:
|
|
39
40
|
```python
|
|
@@ -96,6 +97,7 @@ class LLMDispatcher:
|
|
|
96
97
|
*,
|
|
97
98
|
model: str | None = None,
|
|
98
99
|
response_mime_type: str = "image/webp",
|
|
100
|
+
image_config: dict[str, Any] | None = None,
|
|
99
101
|
**kwargs: Any,
|
|
100
102
|
) -> tuple[bytes, str]:
|
|
101
103
|
"""
|
|
@@ -114,6 +116,7 @@ class LLMDispatcher:
|
|
|
114
116
|
prompt,
|
|
115
117
|
model=model,
|
|
116
118
|
response_mime_type=response_mime_type,
|
|
119
|
+
image_config=image_config,
|
|
117
120
|
**kwargs,
|
|
118
121
|
)
|
|
119
122
|
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
"""
|
|
2
2
|
codex_ai.core.protocol
|
|
3
3
|
=======================
|
|
4
|
-
Core types and contracts for
|
|
4
|
+
Core types and contracts for legacy text routing and direct provider helpers.
|
|
5
5
|
|
|
6
6
|
PromptResult — frozen DTO returned by every prompt builder.
|
|
7
|
-
LLMProviderProtocol —
|
|
7
|
+
LLMProviderProtocol — legacy text compatibility contract for ``answer()``.
|
|
8
8
|
TextGenerationProvider — optional adapter contract for direct text generation.
|
|
9
9
|
JsonGenerationProvider — optional adapter contract for direct JSON generation.
|
|
10
10
|
ImageGenerationProvider — optional adapter contract for binary image generation.
|
|
@@ -39,11 +39,10 @@ class PromptResult(BaseDTO):
|
|
|
39
39
|
"""
|
|
40
40
|
Frozen DTO produced by a prompt builder.
|
|
41
41
|
|
|
42
|
-
Passed directly to the
|
|
42
|
+
Passed directly to the provider's legacy text ``answer()`` method.
|
|
43
43
|
|
|
44
44
|
Attributes:
|
|
45
|
-
messages:
|
|
46
|
-
or Gemini contents). Each provider interprets this field as needed.
|
|
45
|
+
messages: Ordered text message list consumed by compatibility adapters.
|
|
47
46
|
system: Optional system/developer instruction (top-level string, used by Gemini
|
|
48
47
|
and OpenAI o-series models that accept a dedicated system field).
|
|
49
48
|
|
|
@@ -68,9 +67,11 @@ class PromptResult(BaseDTO):
|
|
|
68
67
|
@runtime_checkable
|
|
69
68
|
class LLMProviderProtocol(Protocol):
|
|
70
69
|
"""
|
|
71
|
-
|
|
70
|
+
Legacy text adapter contract.
|
|
72
71
|
|
|
73
|
-
|
|
72
|
+
New Gemini integrations should prefer direct provider methods such as
|
|
73
|
+
``generate_text()``, ``generate_json()``, and ``generate_image_bytes()``.
|
|
74
|
+
This protocol remains for router/dispatcher text compatibility.
|
|
74
75
|
|
|
75
76
|
Example:
|
|
76
77
|
```python
|
|
@@ -82,14 +83,14 @@ class LLMProviderProtocol(Protocol):
|
|
|
82
83
|
|
|
83
84
|
async def answer(self, prompt: PromptResult, **kw: Any) -> str:
|
|
84
85
|
"""
|
|
85
|
-
Send
|
|
86
|
+
Send a legacy text prompt and return the response text.
|
|
86
87
|
|
|
87
88
|
Args:
|
|
88
89
|
prompt: Frozen DTO with messages and system instruction.
|
|
89
90
|
**kw: Extra provider-specific kwargs (temperature, max_tokens, etc.).
|
|
90
91
|
|
|
91
92
|
Returns:
|
|
92
|
-
Response text from the
|
|
93
|
+
Response text from the provider.
|
|
93
94
|
"""
|
|
94
95
|
...
|
|
95
96
|
|
|
@@ -148,6 +149,7 @@ class ImageGenerationProvider(Protocol):
|
|
|
148
149
|
*,
|
|
149
150
|
model: str | None = None,
|
|
150
151
|
response_mime_type: str = "image/webp",
|
|
152
|
+
image_config: dict[str, Any] | None = None,
|
|
151
153
|
**kwargs: Any,
|
|
152
154
|
) -> tuple[bytes, str]:
|
|
153
155
|
"""
|
|
@@ -157,6 +159,8 @@ class ImageGenerationProvider(Protocol):
|
|
|
157
159
|
prompt: Plain image-generation prompt.
|
|
158
160
|
model: Optional image model override.
|
|
159
161
|
response_mime_type: Requested/preferred image MIME type.
|
|
162
|
+
image_config: Optional image generation controls such as
|
|
163
|
+
``{"aspect_ratio": "1:1", "image_size": "4K"}``.
|
|
160
164
|
**kwargs: Extra provider-specific kwargs.
|
|
161
165
|
|
|
162
166
|
Returns:
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""
|
|
2
2
|
codex_ai.providers.gemini
|
|
3
3
|
==========================
|
|
4
|
-
GeminiProvider —
|
|
4
|
+
GeminiProvider — Gemini-first direct API backed by google-genai.
|
|
5
5
|
|
|
6
6
|
Requires: ``pip install codex-ai[gemini]``
|
|
7
7
|
"""
|
|
@@ -34,9 +34,10 @@ _DEFAULT_IMAGEN_MODEL = "imagen-3.0-generate-002"
|
|
|
34
34
|
|
|
35
35
|
class GeminiProvider:
|
|
36
36
|
"""
|
|
37
|
-
|
|
37
|
+
Direct Gemini adapter using the google-genai SDK.
|
|
38
38
|
|
|
39
|
-
Implements
|
|
39
|
+
Implements legacy text compatibility through ``answer()`` and exposes
|
|
40
|
+
direct Gemini methods for text, JSON, Gemini image, and Imagen generation.
|
|
40
41
|
|
|
41
42
|
Args:
|
|
42
43
|
api_key: Google AI API key.
|
|
@@ -186,6 +187,7 @@ class GeminiProvider:
|
|
|
186
187
|
*,
|
|
187
188
|
model: str | None = None,
|
|
188
189
|
response_mime_type: str = "image/webp",
|
|
190
|
+
image_config: dict[str, Any] | None = None,
|
|
189
191
|
**kwargs: Any,
|
|
190
192
|
) -> tuple[bytes, str]:
|
|
191
193
|
"""
|
|
@@ -206,29 +208,75 @@ class GeminiProvider:
|
|
|
206
208
|
runtime_kw = kwargs.copy()
|
|
207
209
|
runtime_kw.pop("model", None)
|
|
208
210
|
runtime_kw.pop("response_mime_type", None)
|
|
211
|
+
runtime_kw.pop("image_config", None)
|
|
209
212
|
|
|
210
|
-
|
|
211
|
-
response_modalities
|
|
213
|
+
config_kwargs: dict[str, Any] = {
|
|
214
|
+
"response_modalities": [genai_types.Modality.IMAGE],
|
|
212
215
|
**runtime_kw,
|
|
213
|
-
|
|
216
|
+
}
|
|
217
|
+
if image_config is not None:
|
|
218
|
+
config_kwargs["image_config"] = image_config
|
|
214
219
|
|
|
215
220
|
try:
|
|
216
|
-
|
|
221
|
+
return await self._generate_image_once(
|
|
217
222
|
model=selected_model,
|
|
218
|
-
|
|
219
|
-
|
|
223
|
+
prompt=prompt,
|
|
224
|
+
config_kwargs=config_kwargs,
|
|
225
|
+
fallback_mime=requested_mime,
|
|
220
226
|
)
|
|
221
|
-
image = self._extract_first_inline_image(response, fallback_mime=requested_mime)
|
|
222
|
-
if image is not None:
|
|
223
|
-
return image
|
|
224
|
-
|
|
225
|
-
detail = self._describe_non_image_response(response)
|
|
226
|
-
raise LLMProviderError(f"Gemini image generation did not return image data{detail}")
|
|
227
227
|
except LLMProviderError:
|
|
228
228
|
raise
|
|
229
229
|
except Exception as exc:
|
|
230
|
+
fallback_config_kwargs = self._fallback_4k_image_config_to_2k(config_kwargs)
|
|
231
|
+
if fallback_config_kwargs is not None:
|
|
232
|
+
try:
|
|
233
|
+
return await self._generate_image_once(
|
|
234
|
+
model=selected_model,
|
|
235
|
+
prompt=prompt,
|
|
236
|
+
config_kwargs=fallback_config_kwargs,
|
|
237
|
+
fallback_mime=requested_mime,
|
|
238
|
+
)
|
|
239
|
+
except LLMProviderError:
|
|
240
|
+
raise
|
|
241
|
+
except Exception as fallback_exc:
|
|
242
|
+
raise LLMProviderError(f"Gemini image generation error: {fallback_exc}") from fallback_exc
|
|
243
|
+
|
|
230
244
|
raise LLMProviderError(f"Gemini image generation error: {exc}") from exc
|
|
231
245
|
|
|
246
|
+
async def _generate_image_once(
|
|
247
|
+
self,
|
|
248
|
+
*,
|
|
249
|
+
model: str,
|
|
250
|
+
prompt: str,
|
|
251
|
+
config_kwargs: dict[str, Any],
|
|
252
|
+
fallback_mime: str,
|
|
253
|
+
) -> tuple[bytes, str]:
|
|
254
|
+
config = genai_types.GenerateContentConfig(**config_kwargs)
|
|
255
|
+
response = await self._client.aio.models.generate_content(
|
|
256
|
+
model=model,
|
|
257
|
+
contents=prompt,
|
|
258
|
+
config=config,
|
|
259
|
+
)
|
|
260
|
+
image = self._extract_first_inline_image(response, fallback_mime=fallback_mime)
|
|
261
|
+
if image is not None:
|
|
262
|
+
return image
|
|
263
|
+
|
|
264
|
+
detail = self._describe_non_image_response(response)
|
|
265
|
+
raise LLMProviderError(f"Gemini image generation did not return image data{detail}")
|
|
266
|
+
|
|
267
|
+
@staticmethod
|
|
268
|
+
def _fallback_4k_image_config_to_2k(config_kwargs: dict[str, Any]) -> dict[str, Any] | None:
|
|
269
|
+
image_config = config_kwargs.get("image_config")
|
|
270
|
+
if not isinstance(image_config, dict) or image_config.get("image_size") != "4K":
|
|
271
|
+
return None
|
|
272
|
+
|
|
273
|
+
fallback_image_config = image_config.copy()
|
|
274
|
+
fallback_image_config["image_size"] = "2K"
|
|
275
|
+
|
|
276
|
+
fallback_config_kwargs = config_kwargs.copy()
|
|
277
|
+
fallback_config_kwargs["image_config"] = fallback_image_config
|
|
278
|
+
return fallback_config_kwargs
|
|
279
|
+
|
|
232
280
|
async def generate_imagen_bytes(
|
|
233
281
|
self,
|
|
234
282
|
prompt: str,
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""
|
|
2
2
|
codex_ai.providers.openai
|
|
3
3
|
==========================
|
|
4
|
-
OpenAIProvider —
|
|
4
|
+
OpenAIProvider — text-only adapter backed by OpenAI's Chat Completions API.
|
|
5
5
|
|
|
6
6
|
Requires: ``pip install codex-ai[openai]``
|
|
7
7
|
"""
|
|
@@ -28,9 +28,10 @@ _DEFAULT_MODEL = "gpt-4o-mini"
|
|
|
28
28
|
|
|
29
29
|
class OpenAIProvider:
|
|
30
30
|
"""
|
|
31
|
-
|
|
31
|
+
Text-only adapter using OpenAI Chat Completions.
|
|
32
32
|
|
|
33
|
-
Implements
|
|
33
|
+
Implements legacy text compatibility through ``answer()`` and direct
|
|
34
|
+
``generate_text()`` convenience.
|
|
34
35
|
|
|
35
36
|
Args:
|
|
36
37
|
api_key: OpenAI API key.
|
|
@@ -126,9 +126,10 @@ class ImageProvider:
|
|
|
126
126
|
*,
|
|
127
127
|
model: str | None = None,
|
|
128
128
|
response_mime_type: str = "image/webp",
|
|
129
|
+
image_config: dict | None = None,
|
|
129
130
|
**kwargs,
|
|
130
131
|
) -> tuple[bytes, str]:
|
|
131
|
-
self.calls.append((prompt, model, response_mime_type, kwargs))
|
|
132
|
+
self.calls.append((prompt, model, response_mime_type, image_config, kwargs))
|
|
132
133
|
return b"image-bytes", "image/png"
|
|
133
134
|
|
|
134
135
|
async def generate_imagen_bytes(
|
|
@@ -159,11 +160,14 @@ async def test_dispatcher_generate_image_bytes_delegates_to_image_provider():
|
|
|
159
160
|
"draw a castle",
|
|
160
161
|
model="gemini-image",
|
|
161
162
|
response_mime_type="image/webp",
|
|
163
|
+
image_config={"aspect_ratio": "1:1", "image_size": "4K"},
|
|
162
164
|
seed=123,
|
|
163
165
|
)
|
|
164
166
|
|
|
165
167
|
assert result == (b"image-bytes", "image/png")
|
|
166
|
-
assert provider.calls == [
|
|
168
|
+
assert provider.calls == [
|
|
169
|
+
("draw a castle", "gemini-image", "image/webp", {"aspect_ratio": "1:1", "image_size": "4K"}, {"seed": 123})
|
|
170
|
+
]
|
|
167
171
|
|
|
168
172
|
|
|
169
173
|
async def test_dispatcher_generate_image_bytes_raises_for_unsupported_provider(mock_provider):
|
|
@@ -106,6 +106,7 @@ def test_image_generation_provider_structural_check():
|
|
|
106
106
|
*,
|
|
107
107
|
model: str | None = None,
|
|
108
108
|
response_mime_type: str = "image/webp",
|
|
109
|
+
image_config: dict | None = None,
|
|
109
110
|
**kwargs,
|
|
110
111
|
) -> tuple[bytes, str]:
|
|
111
112
|
return b"image", response_mime_type
|
|
@@ -418,14 +418,43 @@ async def test_gemini_generate_image_bytes_config_requests_image_modality_not_te
|
|
|
418
418
|
with patch("codex_ai.providers.gemini.genai_types") as mock_types:
|
|
419
419
|
mock_types.Modality.IMAGE = "IMAGE"
|
|
420
420
|
mock_types.GenerateContentConfig.return_value = MagicMock()
|
|
421
|
-
await provider.generate_image_bytes(
|
|
421
|
+
await provider.generate_image_bytes(
|
|
422
|
+
"draw a castle",
|
|
423
|
+
response_mime_type="image/png",
|
|
424
|
+
image_config={"aspect_ratio": "1:1", "image_size": "4K"},
|
|
425
|
+
seed=7,
|
|
426
|
+
)
|
|
422
427
|
|
|
423
428
|
config_kwargs = mock_types.GenerateContentConfig.call_args.kwargs
|
|
424
429
|
assert config_kwargs["response_modalities"] == ["IMAGE"]
|
|
425
430
|
assert "response_mime_type" not in config_kwargs
|
|
431
|
+
assert config_kwargs["image_config"] == {"aspect_ratio": "1:1", "image_size": "4K"}
|
|
426
432
|
assert config_kwargs["seed"] == 7
|
|
427
433
|
|
|
428
434
|
|
|
435
|
+
async def test_gemini_generate_image_bytes_falls_back_from_4k_to_2k_when_config_rejected():
|
|
436
|
+
provider, mock_generate, _ = _make_provider()
|
|
437
|
+
mock_generate.side_effect = [
|
|
438
|
+
ValueError("unsupported image_size"),
|
|
439
|
+
_image_response(data=b"png", mime_type="image/png"),
|
|
440
|
+
]
|
|
441
|
+
|
|
442
|
+
with patch("codex_ai.providers.gemini.genai_types") as mock_types:
|
|
443
|
+
mock_types.Modality.IMAGE = "IMAGE"
|
|
444
|
+
mock_types.GenerateContentConfig.side_effect = lambda **kwargs: kwargs
|
|
445
|
+
result = await provider.generate_image_bytes(
|
|
446
|
+
"draw a castle",
|
|
447
|
+
response_mime_type="image/png",
|
|
448
|
+
image_config={"aspect_ratio": "1:1", "image_size": "4K"},
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
assert result == (b"png", "image/png")
|
|
452
|
+
first_config = mock_generate.call_args_list[0].kwargs["config"]
|
|
453
|
+
second_config = mock_generate.call_args_list[1].kwargs["config"]
|
|
454
|
+
assert first_config["image_config"] == {"aspect_ratio": "1:1", "image_size": "4K"}
|
|
455
|
+
assert second_config["image_config"] == {"aspect_ratio": "1:1", "image_size": "2K"}
|
|
456
|
+
|
|
457
|
+
|
|
429
458
|
async def test_gemini_generate_image_bytes_returns_inline_image_bytes_and_actual_mime():
|
|
430
459
|
provider, mock_generate, _ = _make_provider()
|
|
431
460
|
mock_generate.return_value = _image_response(data=b"png-bytes", mime_type="image/png")
|
|
@@ -344,9 +344,9 @@ requires-dist = [
|
|
|
344
344
|
{ name = "bandit", marker = "extra == 'dev'", specifier = ">=1.7" },
|
|
345
345
|
{ name = "codex-core", specifier = ">=0.2.2,<0.4.0" },
|
|
346
346
|
{ name = "detect-secrets", marker = "extra == 'dev'", specifier = ">=1.5" },
|
|
347
|
-
{ name = "google-genai", marker = "extra == 'all'", specifier = "==
|
|
348
|
-
{ name = "google-genai", marker = "extra == 'dev'", specifier = "==
|
|
349
|
-
{ name = "google-genai", marker = "extra == 'gemini'", specifier = "==
|
|
347
|
+
{ name = "google-genai", marker = "extra == 'all'", specifier = "==2.3.0" },
|
|
348
|
+
{ name = "google-genai", marker = "extra == 'dev'", specifier = "==2.3.0" },
|
|
349
|
+
{ name = "google-genai", marker = "extra == 'gemini'", specifier = "==2.3.0" },
|
|
350
350
|
{ name = "mike", marker = "extra == 'docs'", specifier = ">=2.0" },
|
|
351
351
|
{ name = "mkdocs", marker = "extra == 'docs'", specifier = ">=1.5" },
|
|
352
352
|
{ name = "mkdocs-include-markdown-plugin", marker = "extra == 'docs'" },
|
|
@@ -622,7 +622,7 @@ requests = [
|
|
|
622
622
|
|
|
623
623
|
[[package]]
|
|
624
624
|
name = "google-genai"
|
|
625
|
-
version = "
|
|
625
|
+
version = "2.3.0"
|
|
626
626
|
source = { registry = "https://pypi.org/simple" }
|
|
627
627
|
dependencies = [
|
|
628
628
|
{ name = "anyio" },
|
|
@@ -636,9 +636,9 @@ dependencies = [
|
|
|
636
636
|
{ name = "typing-extensions" },
|
|
637
637
|
{ name = "websockets" },
|
|
638
638
|
]
|
|
639
|
-
sdist = { url = "https://files.pythonhosted.org/packages/
|
|
639
|
+
sdist = { url = "https://files.pythonhosted.org/packages/02/8e/dfa4b34dd4c0baffccf6466fc68d6d35011662d43e7d79accb902320db74/google_genai-2.3.0.tar.gz", hash = "sha256:e877c750a4ccacdd9928fc3aa8ca8820ce85cade0ca51bd83feceacf5959b579", size = 546930, upload-time = "2026-05-15T06:22:36.264Z" }
|
|
640
640
|
wheels = [
|
|
641
|
-
{ url = "https://files.pythonhosted.org/packages/
|
|
641
|
+
{ url = "https://files.pythonhosted.org/packages/b4/6e/aa6b30b09f58b946750fc4089c5248fbd3576f746e0e818d88633559dc84/google_genai-2.3.0-py3-none-any.whl", hash = "sha256:89d3c71c9f5f5b931b405b88a5837aea2bd4d27ed90323b9599f5760bbb91d92", size = 805484, upload-time = "2026-05-15T06:22:34.247Z" },
|
|
642
642
|
]
|
|
643
643
|
|
|
644
644
|
[[package]]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|