nl2data-openai 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nl2data_openai-0.1.0/PKG-INFO +160 -0
- nl2data_openai-0.1.0/README.md +136 -0
- nl2data_openai-0.1.0/pyproject.toml +56 -0
- nl2data_openai-0.1.0/setup.cfg +4 -0
- nl2data_openai-0.1.0/src/nl2data_openai/__init__.py +15 -0
- nl2data_openai-0.1.0/src/nl2data_openai/client.py +88 -0
- nl2data_openai-0.1.0/src/nl2data_openai/config.py +63 -0
- nl2data_openai-0.1.0/src/nl2data_openai/live_evaluation.py +286 -0
- nl2data_openai-0.1.0/src/nl2data_openai/mapping.py +413 -0
- nl2data_openai-0.1.0/src/nl2data_openai/provider.py +261 -0
- nl2data_openai-0.1.0/src/nl2data_openai.egg-info/PKG-INFO +160 -0
- nl2data_openai-0.1.0/src/nl2data_openai.egg-info/SOURCES.txt +13 -0
- nl2data_openai-0.1.0/src/nl2data_openai.egg-info/dependency_links.txt +1 -0
- nl2data_openai-0.1.0/src/nl2data_openai.egg-info/requires.txt +8 -0
- nl2data_openai-0.1.0/src/nl2data_openai.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nl2data-openai
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: OpenAI structured-output provider for the nl2data-core model provider boundary.
|
|
5
|
+
Author: NL2Data Contributors
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Keywords: nl2data,openai,structured output,model provider
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
14
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
15
|
+
Requires-Python: >=3.11
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Requires-Dist: nl2data-core>=0.1.0
|
|
18
|
+
Requires-Dist: openai<3,>=1.40
|
|
19
|
+
Provides-Extra: dev
|
|
20
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
21
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == "dev"
|
|
22
|
+
Requires-Dist: mypy>=1.10; extra == "dev"
|
|
23
|
+
Requires-Dist: ruff>=0.5; extra == "dev"
|
|
24
|
+
|
|
25
|
+
# nl2data-openai
|
|
26
|
+
|
|
27
|
+
An optional OpenAI structured-output provider for
|
|
28
|
+
[nl2data-core](https://github.com/emmansun/nl2data-core). It implements the
|
|
29
|
+
provider-neutral asynchronous `ModelProvider` contract: bounded
|
|
30
|
+
`ModelInvocationRequest` in, typed `ModelResponse`/normalized
|
|
31
|
+
`ModelInvocationError` out, with the OpenAI SDK isolated to this package.
|
|
32
|
+
|
|
33
|
+
The core import boundary never loads the OpenAI SDK; this package imports
|
|
34
|
+
it **lazily** at client build time — never at import, construction, or
|
|
35
|
+
capability inspection.
|
|
36
|
+
|
|
37
|
+
## Install
|
|
38
|
+
|
|
39
|
+
```bash
|
|
40
|
+
pip install nl2data-openai
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Requires Python 3.11+, `nl2data-core>=0.1.0`, and `openai>=1.40,<3`.
|
|
44
|
+
|
|
45
|
+
From a source checkout:
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
pip install -e ".[dev]" # from the repository root (core)
|
|
49
|
+
pip install -e packages/nl2data-openai # this package (editable)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
## Public surface
|
|
53
|
+
|
|
54
|
+
```python
|
|
55
|
+
from nl2data_openai import OpenAIProviderConfig, OpenAIModelProvider
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
- `OpenAIProviderConfig` — vendor `model_name` plus bounded invocation
|
|
59
|
+
settings: `max_input_chars`, `max_output_tokens`, `temperature`,
|
|
60
|
+
`timeout_seconds`, optional `base_url` and `organization`. Capabilities are derived from this
|
|
61
|
+
configuration **without any network call**.
|
|
62
|
+
- `OpenAIModelProvider` — the `ModelProvider` port implementation.
|
|
63
|
+
`close()` is idempotent and never leaks native clients or exceptions.
|
|
64
|
+
|
|
65
|
+
## Credential injection
|
|
66
|
+
|
|
67
|
+
API keys never enter core models, configuration fingerprints, request
|
|
68
|
+
metadata, workflow state, telemetry, or errors. Inject them through one
|
|
69
|
+
of:
|
|
70
|
+
|
|
71
|
+
1. An `api_key_resolver` callable at provider construction, or
|
|
72
|
+
2. A `client_factory` at provider construction, or
|
|
73
|
+
3. The `OPENAI_API_KEY` environment variable — read **only when the
|
|
74
|
+
client is first built**.
|
|
75
|
+
|
|
76
|
+
```python
|
|
77
|
+
import os
|
|
78
|
+
|
|
79
|
+
from nl2data_openai import OpenAIProviderConfig, OpenAIModelProvider
|
|
80
|
+
|
|
81
|
+
provider = OpenAIModelProvider(
|
|
82
|
+
config=OpenAIProviderConfig(model_name="gpt-4o-mini"),
|
|
83
|
+
api_key_resolver=lambda: os.environ["OPENAI_API_KEY"], # host-owned
|
|
84
|
+
)
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
See [Secrets and live testing](../../docs/operations/secrets.md) for the
|
|
88
|
+
full credential-handling contract.
|
|
89
|
+
|
|
90
|
+
## Gateway compatibility
|
|
91
|
+
|
|
92
|
+
Set `base_url` to point at any OpenAI-compatible gateway (OpenAI,
|
|
93
|
+
Azure OpenAI-compatible endpoints, or a self-hosted proxy):
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
OpenAIProviderConfig(
|
|
97
|
+
model_name="deployed-model",
|
|
98
|
+
base_url="https://your-gateway.example/v1", # host-owned endpoint
|
|
99
|
+
timeout_seconds=60,
|
|
100
|
+
)
|
|
101
|
+
```
|
|
102
|
+
|
|
103
|
+
`base_url` and `organization` are bounded configuration fields; the
|
|
104
|
+
endpoint itself is a host-owned setting and never part of core
|
|
105
|
+
configuration or evidence.
|
|
106
|
+
|
|
107
|
+
## Model selection and limits
|
|
108
|
+
|
|
109
|
+
`OpenAIProviderConfig` carries `model_name` plus bounded invocation
|
|
110
|
+
settings (`max_input_chars`, `max_output_tokens`, `temperature`,
|
|
111
|
+
`timeout_seconds`, optional `base_url`, `organization`). Provider calls
|
|
112
|
+
are bounded by the resolver's attempt budget; the provider performs
|
|
113
|
+
**exactly one vendor request per `generate()` call** — timeout, retry,
|
|
114
|
+
and attempt-budget policy belong to `IntentResolver`.
|
|
115
|
+
|
|
116
|
+
## Failure classification
|
|
117
|
+
|
|
118
|
+
| Condition | Normalized result |
|
|
119
|
+
| --- | --- |
|
|
120
|
+
| Authentication/configuration failure | Non-retryable `INVALID_REQUEST` |
|
|
121
|
+
| Timeout, connection, rate-limit, transient service error | Retryable `MODEL_TIMEOUT` / `PROVIDER_UNAVAILABLE` |
|
|
122
|
+
|
|
123
|
+
## Live testing
|
|
124
|
+
|
|
125
|
+
An opt-in live evaluation profile (`run_live_openai_evaluation` in
|
|
126
|
+
`nl2data_openai.live_evaluation`) runs the deterministic AI dataset
|
|
127
|
+
against the real provider and classifies every case as `verified`,
|
|
128
|
+
`unavailable`, or `skipped`. Without injected credentials/factory or
|
|
129
|
+
`OPENAI_API_KEY`, every case is `skipped` — default CI needs no
|
|
130
|
+
credentials and makes no network access.
|
|
131
|
+
|
|
132
|
+
Local run from the repository root (credentials from the environment
|
|
133
|
+
only — the script never writes them to disk or includes them in output):
|
|
134
|
+
|
|
135
|
+
```bash
|
|
136
|
+
$env:OPENAI_API_KEY = "..." # host secret injection, never committed
|
|
137
|
+
$env:OPENAI_BASE_URL = "https://api.openai.com/v1"
|
|
138
|
+
$env:OPENAI_MODEL = "gpt-4o-mini"
|
|
139
|
+
$env:OPENAI_TIMEOUT_SECONDS = "60" # optional
|
|
140
|
+
$env:OPENAI_LIVE_CASES = "normal-intent" # optional, comma-separated
|
|
141
|
+
python scripts/run_openai_live.py
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Exit code is 0 only when every selected case is `verified`. Evidence
|
|
145
|
+
carries only protected fingerprints and normalized codes.
|
|
146
|
+
|
|
147
|
+
## Rollback
|
|
148
|
+
|
|
149
|
+
Swap the provider back to the core's deterministic `FakeModelProvider`
|
|
150
|
+
(`nl2data_core.ai.fake` — contributor-only) at composition time to remove
|
|
151
|
+
the SDK dependency and network access while keeping the same resolver,
|
|
152
|
+
governance, and evaluation gates. No runtime migration is involved: the
|
|
153
|
+
provider is a composition input, so rollback is a deployment decision.
|
|
154
|
+
|
|
155
|
+
## More documentation
|
|
156
|
+
|
|
157
|
+
- [Documentation index](../../docs/README.md)
|
|
158
|
+
- [Adding a model provider](../../docs/development/adding-adapter-or-provider.md)
|
|
159
|
+
- [Secrets and live testing](../../docs/operations/secrets.md)
|
|
160
|
+
- [Capabilities and support](../../docs/reference/capabilities.md)
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# nl2data-openai
|
|
2
|
+
|
|
3
|
+
An optional OpenAI structured-output provider for
|
|
4
|
+
[nl2data-core](https://github.com/emmansun/nl2data-core). It implements the
|
|
5
|
+
provider-neutral asynchronous `ModelProvider` contract: bounded
|
|
6
|
+
`ModelInvocationRequest` in, typed `ModelResponse`/normalized
|
|
7
|
+
`ModelInvocationError` out, with the OpenAI SDK isolated to this package.
|
|
8
|
+
|
|
9
|
+
The core import boundary never loads the OpenAI SDK; this package imports
|
|
10
|
+
it **lazily** at client build time — never at import, construction, or
|
|
11
|
+
capability inspection.
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install nl2data-openai
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
Requires Python 3.11+, `nl2data-core>=0.1.0`, and `openai>=1.40,<3`.
|
|
20
|
+
|
|
21
|
+
From a source checkout:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install -e ".[dev]" # from the repository root (core)
|
|
25
|
+
pip install -e packages/nl2data-openai # this package (editable)
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Public surface
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
from nl2data_openai import OpenAIProviderConfig, OpenAIModelProvider
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
- `OpenAIProviderConfig` — vendor `model_name` plus bounded invocation
|
|
35
|
+
settings: `max_input_chars`, `max_output_tokens`, `temperature`,
|
|
36
|
+
`timeout_seconds`, optional `base_url` and `organization`. Capabilities are derived from this
|
|
37
|
+
configuration **without any network call**.
|
|
38
|
+
- `OpenAIModelProvider` — the `ModelProvider` port implementation.
|
|
39
|
+
`close()` is idempotent and never leaks native clients or exceptions.
|
|
40
|
+
|
|
41
|
+
## Credential injection
|
|
42
|
+
|
|
43
|
+
API keys never enter core models, configuration fingerprints, request
|
|
44
|
+
metadata, workflow state, telemetry, or errors. Inject them through one
|
|
45
|
+
of:
|
|
46
|
+
|
|
47
|
+
1. An `api_key_resolver` callable at provider construction, or
|
|
48
|
+
2. A `client_factory` at provider construction, or
|
|
49
|
+
3. The `OPENAI_API_KEY` environment variable — read **only when the
|
|
50
|
+
client is first built**.
|
|
51
|
+
|
|
52
|
+
```python
|
|
53
|
+
import os
|
|
54
|
+
|
|
55
|
+
from nl2data_openai import OpenAIProviderConfig, OpenAIModelProvider
|
|
56
|
+
|
|
57
|
+
provider = OpenAIModelProvider(
|
|
58
|
+
config=OpenAIProviderConfig(model_name="gpt-4o-mini"),
|
|
59
|
+
api_key_resolver=lambda: os.environ["OPENAI_API_KEY"], # host-owned
|
|
60
|
+
)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
See [Secrets and live testing](../../docs/operations/secrets.md) for the
|
|
64
|
+
full credential-handling contract.
|
|
65
|
+
|
|
66
|
+
## Gateway compatibility
|
|
67
|
+
|
|
68
|
+
Set `base_url` to point at any OpenAI-compatible gateway (OpenAI,
|
|
69
|
+
Azure OpenAI-compatible endpoints, or a self-hosted proxy):
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
OpenAIProviderConfig(
|
|
73
|
+
model_name="deployed-model",
|
|
74
|
+
base_url="https://your-gateway.example/v1", # host-owned endpoint
|
|
75
|
+
timeout_seconds=60,
|
|
76
|
+
)
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
`base_url` and `organization` are bounded configuration fields; the
|
|
80
|
+
endpoint itself is a host-owned setting and never part of core
|
|
81
|
+
configuration or evidence.
|
|
82
|
+
|
|
83
|
+
## Model selection and limits
|
|
84
|
+
|
|
85
|
+
`OpenAIProviderConfig` carries `model_name` plus bounded invocation
|
|
86
|
+
settings (`max_input_chars`, `max_output_tokens`, `temperature`,
|
|
87
|
+
`timeout_seconds`, optional `base_url`, `organization`). Provider calls
|
|
88
|
+
are bounded by the resolver's attempt budget; the provider performs
|
|
89
|
+
**exactly one vendor request per `generate()` call** — timeout, retry,
|
|
90
|
+
and attempt-budget policy belong to `IntentResolver`.
|
|
91
|
+
|
|
92
|
+
## Failure classification
|
|
93
|
+
|
|
94
|
+
| Condition | Normalized result |
|
|
95
|
+
| --- | --- |
|
|
96
|
+
| Authentication/configuration failure | Non-retryable `INVALID_REQUEST` |
|
|
97
|
+
| Timeout, connection, rate-limit, transient service error | Retryable `MODEL_TIMEOUT` / `PROVIDER_UNAVAILABLE` |
|
|
98
|
+
|
|
99
|
+
## Live testing
|
|
100
|
+
|
|
101
|
+
An opt-in live evaluation profile (`run_live_openai_evaluation` in
|
|
102
|
+
`nl2data_openai.live_evaluation`) runs the deterministic AI dataset
|
|
103
|
+
against the real provider and classifies every case as `verified`,
|
|
104
|
+
`unavailable`, or `skipped`. Without injected credentials/factory or
|
|
105
|
+
`OPENAI_API_KEY`, every case is `skipped` — default CI needs no
|
|
106
|
+
credentials and makes no network access.
|
|
107
|
+
|
|
108
|
+
Local run from the repository root (credentials from the environment
|
|
109
|
+
only — the script never writes them to disk or includes them in output):
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
$env:OPENAI_API_KEY = "..." # host secret injection, never committed
|
|
113
|
+
$env:OPENAI_BASE_URL = "https://api.openai.com/v1"
|
|
114
|
+
$env:OPENAI_MODEL = "gpt-4o-mini"
|
|
115
|
+
$env:OPENAI_TIMEOUT_SECONDS = "60" # optional
|
|
116
|
+
$env:OPENAI_LIVE_CASES = "normal-intent" # optional, comma-separated
|
|
117
|
+
python scripts/run_openai_live.py
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
Exit code is 0 only when every selected case is `verified`. Evidence
|
|
121
|
+
carries only protected fingerprints and normalized codes.
|
|
122
|
+
|
|
123
|
+
## Rollback
|
|
124
|
+
|
|
125
|
+
Swap the provider back to the core's deterministic `FakeModelProvider`
|
|
126
|
+
(`nl2data_core.ai.fake` — contributor-only) at composition time to remove
|
|
127
|
+
the SDK dependency and network access while keeping the same resolver,
|
|
128
|
+
governance, and evaluation gates. No runtime migration is involved: the
|
|
129
|
+
provider is a composition input, so rollback is a deployment decision.
|
|
130
|
+
|
|
131
|
+
## More documentation
|
|
132
|
+
|
|
133
|
+
- [Documentation index](../../docs/README.md)
|
|
134
|
+
- [Adding a model provider](../../docs/development/adding-adapter-or-provider.md)
|
|
135
|
+
- [Secrets and live testing](../../docs/operations/secrets.md)
|
|
136
|
+
- [Capabilities and support](../../docs/reference/capabilities.md)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "nl2data-openai"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "OpenAI structured-output provider for the nl2data-core model provider boundary."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = { text = "Apache-2.0" }
|
|
12
|
+
authors = [{ name = "NL2Data Contributors" }]
|
|
13
|
+
keywords = ["nl2data", "openai", "structured output", "model provider"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"Programming Language :: Python :: 3",
|
|
18
|
+
"Programming Language :: Python :: 3.11",
|
|
19
|
+
"Programming Language :: Python :: 3.12",
|
|
20
|
+
"Programming Language :: Python :: 3.13",
|
|
21
|
+
"Topic :: Software Development :: Libraries",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"nl2data-core>=0.1.0",
|
|
25
|
+
"openai>=1.40,<3",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
dev = [
|
|
30
|
+
"pytest>=8.0",
|
|
31
|
+
"pytest-asyncio>=0.23",
|
|
32
|
+
"mypy>=1.10",
|
|
33
|
+
"ruff>=0.5",
|
|
34
|
+
]
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[tool.mypy]
|
|
40
|
+
python_version = "3.11"
|
|
41
|
+
files = ["src"]
|
|
42
|
+
mypy_path = ["../../src"]
|
|
43
|
+
check_untyped_defs = true
|
|
44
|
+
disallow_untyped_defs = true
|
|
45
|
+
no_implicit_optional = true
|
|
46
|
+
warn_redundant_casts = true
|
|
47
|
+
warn_unused_ignores = true
|
|
48
|
+
warn_return_any = true
|
|
49
|
+
|
|
50
|
+
[tool.ruff]
|
|
51
|
+
line-length = 100
|
|
52
|
+
target-version = "py311"
|
|
53
|
+
src = ["src"]
|
|
54
|
+
|
|
55
|
+
[tool.ruff.lint]
|
|
56
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""OpenAI structured-output provider for ``nl2data-core``.
|
|
2
|
+
|
|
3
|
+
An independent optional distribution implementing the provider-neutral
|
|
4
|
+
``ModelProvider`` contract. The OpenAI SDK is never imported at package
|
|
5
|
+
import time; clients are constructed lazily on first generation from
|
|
6
|
+
injected credentials or a client factory, so core imports and capability
|
|
7
|
+
inspection stay fully offline.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from .config import OpenAIProviderConfig
|
|
13
|
+
from .provider import OpenAIModelProvider
|
|
14
|
+
|
|
15
|
+
__all__ = ["OpenAIProviderConfig", "OpenAIModelProvider"]
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Lazy optional OpenAI SDK boundary for the provider package.
|
|
2
|
+
|
|
3
|
+
The ``openai`` package is loaded only inside this module through
|
|
4
|
+
:func:`importlib.import_module`, so importing ``nl2data_openai``, the core,
|
|
5
|
+
or the provider never imports the SDK. Client construction happens lazily
|
|
6
|
+
on first generation; no import-time or capability-time network access
|
|
7
|
+
exists. Error predicates duck-type by class name so injected fake clients
|
|
8
|
+
raise structurally identical errors without the SDK installed.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
from importlib import import_module
|
|
14
|
+
from importlib.util import find_spec
|
|
15
|
+
from typing import Any
|
|
16
|
+
|
|
17
|
+
from nl2data_core.ai.errors import ModelErrorCode, ModelInvocationError
|
|
18
|
+
|
|
19
|
+
from .config import OpenAIProviderConfig
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def driver_available() -> bool:
|
|
23
|
+
"""Whether the optional ``openai`` SDK is installed."""
|
|
24
|
+
return find_spec("openai") is not None
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def build_openai_client(config: OpenAIProviderConfig, *, api_key: str) -> Any:
|
|
28
|
+
"""Lazily import the SDK and build a bounded ``AsyncOpenAI`` client.
|
|
29
|
+
|
|
30
|
+
Raises a normalized ``PROVIDER_UNAVAILABLE`` error when the SDK is
|
|
31
|
+
missing or the client cannot be constructed; the key and any driver
|
|
32
|
+
exception text never enter the error.
|
|
33
|
+
"""
|
|
34
|
+
if not driver_available():
|
|
35
|
+
raise ModelInvocationError(
|
|
36
|
+
ModelErrorCode.PROVIDER_UNAVAILABLE,
|
|
37
|
+
"the openai SDK is not installed; install the 'nl2data-openai' package",
|
|
38
|
+
details={"cause_type": "ImportError"},
|
|
39
|
+
)
|
|
40
|
+
try:
|
|
41
|
+
openai = import_module("openai")
|
|
42
|
+
kwargs: dict[str, Any] = {"api_key": api_key, "timeout": config.timeout_seconds}
|
|
43
|
+
if config.base_url is not None:
|
|
44
|
+
kwargs["base_url"] = config.base_url
|
|
45
|
+
if config.organization is not None:
|
|
46
|
+
kwargs["organization"] = config.organization
|
|
47
|
+
return openai.AsyncOpenAI(**kwargs)
|
|
48
|
+
except ModelInvocationError:
|
|
49
|
+
raise
|
|
50
|
+
except Exception as error:
|
|
51
|
+
raise ModelInvocationError(
|
|
52
|
+
ModelErrorCode.PROVIDER_UNAVAILABLE,
|
|
53
|
+
"the openai client could not be constructed",
|
|
54
|
+
details={"cause_type": type(error).__name__},
|
|
55
|
+
) from error
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _class_name(error: BaseException) -> str:
|
|
59
|
+
return error.__class__.__name__
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def is_timeout_error(error: BaseException) -> bool:
|
|
63
|
+
"""SDK or builtin timeout signals (duck-typed by class name)."""
|
|
64
|
+
if isinstance(error, TimeoutError):
|
|
65
|
+
return True
|
|
66
|
+
return _class_name(error) == "APITimeoutError"
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def is_connection_error(error: BaseException) -> bool:
|
|
70
|
+
"""Connection failure signals (duck-typed by class name)."""
|
|
71
|
+
if isinstance(error, ConnectionError):
|
|
72
|
+
return True
|
|
73
|
+
return _class_name(error) in {"APIConnectionError", "APIConnectionPoolTimeoutError"}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def is_rate_limit_error(error: BaseException) -> bool:
|
|
77
|
+
"""Rate-limit signals (duck-typed by class name)."""
|
|
78
|
+
return _class_name(error) == "RateLimitError"
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def is_authentication_error(error: BaseException) -> bool:
|
|
82
|
+
"""Credential-rejection signals (duck-typed by class name)."""
|
|
83
|
+
return _class_name(error) in {"AuthenticationError", "PermissionDeniedError"}
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def is_status_error(error: BaseException) -> bool:
|
|
87
|
+
"""Any SDK status error exposing an HTTP status code (duck-typed)."""
|
|
88
|
+
return _class_name(error) == "APIStatusError" or hasattr(error, "status_code")
|
|
@@ -0,0 +1,63 @@
|
|
|
1
|
+
"""Immutable credential-free configuration for the OpenAI provider.
|
|
2
|
+
|
|
3
|
+
The configuration carries model selection and bounded invocation settings
|
|
4
|
+
only. API keys never enter this model: hosts inject credentials through an
|
|
5
|
+
``api_key_resolver`` callable or a ``client_factory`` at provider
|
|
6
|
+
construction, so keys cannot appear in configuration fingerprints, request
|
|
7
|
+
metadata, workflow state, telemetry, or error records.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any
|
|
13
|
+
|
|
14
|
+
from nl2data_core.canonical import strict_sha256_fingerprint
|
|
15
|
+
from pydantic import BaseModel, ConfigDict, Field, model_validator
|
|
16
|
+
|
|
17
|
+
_FINGERPRINT_PATTERN = r"^sha256:[0-9a-f]{64}$"
|
|
18
|
+
_MAX_OUTPUT_TOKENS = 131_072
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
class OpenAIProviderConfig(BaseModel):
|
|
22
|
+
"""Immutable bounded OpenAI invocation settings.
|
|
23
|
+
|
|
24
|
+
``model_name`` selects the vendor model; all other fields bound the
|
|
25
|
+
invocation. ``base_url`` and ``organization`` are optional host-owned
|
|
26
|
+
endpoint overrides that never appear in normalized errors.
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
model_config = ConfigDict(frozen=True, extra="forbid")
|
|
30
|
+
|
|
31
|
+
model_name: str = Field(min_length=1, max_length=128)
|
|
32
|
+
max_input_chars: int = Field(default=100_000, ge=1_000, le=1_000_000)
|
|
33
|
+
max_output_tokens: int = Field(default=4096, ge=1, le=_MAX_OUTPUT_TOKENS)
|
|
34
|
+
temperature: float | None = Field(default=None, ge=0.0, le=2.0)
|
|
35
|
+
timeout_seconds: float = Field(default=30.0, gt=0.0, le=3600.0)
|
|
36
|
+
base_url: str | None = Field(default=None, max_length=512)
|
|
37
|
+
organization: str | None = Field(default=None, max_length=256)
|
|
38
|
+
merge_developer_into_system: bool = False
|
|
39
|
+
fingerprint: str = Field(default="", pattern=_FINGERPRINT_PATTERN)
|
|
40
|
+
|
|
41
|
+
@model_validator(mode="after")
|
|
42
|
+
def _compute_fingerprint(self) -> OpenAIProviderConfig:
|
|
43
|
+
object.__setattr__(self, "fingerprint", strict_sha256_fingerprint(self.safe_payload()))
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
def safe_payload(self) -> dict[str, Any]:
|
|
47
|
+
"""Serializable payload with no credential-bearing fields."""
|
|
48
|
+
return {
|
|
49
|
+
"model_name": self.model_name,
|
|
50
|
+
"max_input_chars": self.max_input_chars,
|
|
51
|
+
"max_output_tokens": self.max_output_tokens,
|
|
52
|
+
"temperature": self.temperature,
|
|
53
|
+
"timeout_seconds": self.timeout_seconds,
|
|
54
|
+
"base_url": self.base_url,
|
|
55
|
+
"organization": self.organization,
|
|
56
|
+
"merge_developer_into_system": self.merge_developer_into_system,
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
def safe_dump(self) -> dict[str, Any]:
|
|
60
|
+
"""Diagnostics-safe serialization; contains no secrets."""
|
|
61
|
+
payload = self.safe_payload()
|
|
62
|
+
payload["fingerprint"] = self.fingerprint
|
|
63
|
+
return payload
|