ai-registry 0.3.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ai_registry-0.3.1/LICENSE +21 -0
- ai_registry-0.3.1/PKG-INFO +263 -0
- ai_registry-0.3.1/README.md +227 -0
- ai_registry-0.3.1/pyproject.toml +138 -0
- ai_registry-0.3.1/src/ai_registry/__init__.py +145 -0
- ai_registry-0.3.1/src/ai_registry/adapters/__init__.py +23 -0
- ai_registry-0.3.1/src/ai_registry/adapters/langchain.py +113 -0
- ai_registry-0.3.1/src/ai_registry/adapters/openai.py +62 -0
- ai_registry-0.3.1/src/ai_registry/adapters/pydantic_ai.py +103 -0
- ai_registry-0.3.1/src/ai_registry/catalog.py +609 -0
- ai_registry-0.3.1/src/ai_registry/cli.py +352 -0
- ai_registry-0.3.1/src/ai_registry/config.py +469 -0
- ai_registry-0.3.1/src/ai_registry/data/models_dev/README.md +38 -0
- ai_registry-0.3.1/src/ai_registry/data/models_dev/api.json +8537 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/alibaba.yml +19 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/anthropic.yml +19 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/deepseek.yml +33 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/google.yml +24 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/minimax.yml +43 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/moonshotai.yml +19 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/openai.yml +23 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/uniapi.yml +352 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/xai.yml +19 -0
- ai_registry-0.3.1/src/ai_registry/data/providers/zhipuai.yml +19 -0
- ai_registry-0.3.1/src/ai_registry/errors.py +39 -0
- ai_registry-0.3.1/src/ai_registry/factory.py +99 -0
- ai_registry-0.3.1/src/ai_registry/py.typed +0 -0
- ai_registry-0.3.1/src/ai_registry/registry.py +685 -0
- ai_registry-0.3.1/src/ai_registry/routing.py +258 -0
- ai_registry-0.3.1/src/ai_registry/schemas.py +561 -0
- ai_registry-0.3.1/src/ai_registry/secrets.py +70 -0
- ai_registry-0.3.1/src/ai_registry/testing.py +190 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ai-registry contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,263 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ai-registry
|
|
3
|
+
Version: 0.3.1
|
|
4
|
+
Summary: Resolve a stable alias to a concrete AI provider deployment. No requests, no gateway, no literal secrets.
|
|
5
|
+
Keywords: ai,llm,model-routing,provider-registry,configuration,secrets
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
License-File: LICENSE
|
|
8
|
+
Classifier: Development Status :: 3 - Alpha
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
12
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
17
|
+
Classifier: Typing :: Typed
|
|
18
|
+
Requires-Dist: pydantic>=2
|
|
19
|
+
Requires-Dist: pyyaml
|
|
20
|
+
Requires-Dist: typer>=0.12 ; extra == 'cli'
|
|
21
|
+
Requires-Dist: langchain-openai>=0.2 ; extra == 'langchain'
|
|
22
|
+
Requires-Dist: langchain-anthropic>=0.2 ; extra == 'langchain'
|
|
23
|
+
Requires-Dist: openai>=1.40 ; extra == 'openai'
|
|
24
|
+
Requires-Dist: pydantic-ai-slim[openai]>=1.0 ; extra == 'pydantic-ai'
|
|
25
|
+
Requires-Python: >=3.11
|
|
26
|
+
Project-URL: Changelog, https://github.com/limoiie/ai-registry/blob/main/CHANGELOG.md
|
|
27
|
+
Project-URL: Documentation, https://github.com/limoiie/ai-registry#documentation
|
|
28
|
+
Project-URL: Homepage, https://github.com/limoiie/ai-registry
|
|
29
|
+
Project-URL: Issues, https://github.com/limoiie/ai-registry/issues
|
|
30
|
+
Project-URL: Source, https://github.com/limoiie/ai-registry
|
|
31
|
+
Provides-Extra: cli
|
|
32
|
+
Provides-Extra: langchain
|
|
33
|
+
Provides-Extra: openai
|
|
34
|
+
Provides-Extra: pydantic-ai
|
|
35
|
+
Description-Content-Type: text/markdown
|
|
36
|
+
|
|
37
|
+
# ai-registry
|
|
38
|
+
|
|
39
|
+
[](https://pypi.org/project/ai-registry/)
|
|
40
|
+
[](https://pypi.org/project/ai-registry/)
|
|
41
|
+
[](https://github.com/limoiie/ai-registry/actions/workflows/ci.yml)
|
|
42
|
+
[](https://github.com/astral-sh/ruff)
|
|
43
|
+
[](https://mypy-lang.org/)
|
|
44
|
+
[](LICENSE)
|
|
45
|
+
|
|
46
|
+
**Resolve a stable alias to a concrete AI deployment — provider, model,
|
|
47
|
+
account, endpoint — and stop there.**
|
|
48
|
+
|
|
49
|
+
`ai-registry` is the configuration layer under your AI stack, not another client.
|
|
50
|
+
It answers one question — *"which provider, which model, whose key, what URL?"* —
|
|
51
|
+
and hands the answer to whatever SDK you already use.
|
|
52
|
+
|
|
53
|
+
It makes **no network requests**, defines **no chat interface**, ships **no
|
|
54
|
+
gateway**, and **cannot hold a literal API key**. The runtime dependency closure
|
|
55
|
+
is six packages.
|
|
56
|
+
|
|
57
|
+
```python
|
|
58
|
+
from ai_registry import Registry
|
|
59
|
+
|
|
60
|
+
registry = Registry.from_project() # reads ./.ai-registry/registry.yaml
|
|
61
|
+
resolution = registry.resolve("chat-default")
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
| What you get | Example | Why it matters |
|
|
65
|
+
| --- | --- | --- |
|
|
66
|
+
| `resolution.provider_id` | `'deepseek'` | Which provider was chosen. |
|
|
67
|
+
| `resolution.model_id` | `'deepseek-chat'` | The id to send, and to log. |
|
|
68
|
+
| `resolution.account_id` | `'deepseek-primary'` | **Which key** was chosen. |
|
|
69
|
+
| `resolution.endpoint` | `'https://api.deepseek.com/anthropic'` | Where to send it. |
|
|
70
|
+
| `resolution.model.capabilities` | `tools=True, reasoning=False, …` | What the model can do. |
|
|
71
|
+
| `resolution.resolve_credential()` | `'sk-…'` | Resolved lazily, only at the adapter boundary. |
|
|
72
|
+
|
|
73
|
+
The model, provider, account, or endpoint behind `"chat-default"` can change in
|
|
74
|
+
configuration without touching application code.
|
|
75
|
+
|
|
76
|
+
## Why a resolver instead of a client
|
|
77
|
+
|
|
78
|
+
Most tools in this space fuse two separate concerns: *deciding what to call* and
|
|
79
|
+
*calling it*. Fusing them means you adopt someone's request translation, error
|
|
80
|
+
types, retry semantics, and dependency tree in order to get their configuration
|
|
81
|
+
model.
|
|
82
|
+
|
|
83
|
+
That fusion has a well-known failure mode. From
|
|
84
|
+
[openai/codex#1370](https://github.com/openai/codex/issues/1370), on Azure
|
|
85
|
+
deployment names:
|
|
86
|
+
|
|
87
|
+
> No logic should inspect the user's deployment name when making assumptions
|
|
88
|
+
> about model capabilities.
|
|
89
|
+
|
|
90
|
+
That is exactly the split this library enforces. A **deployment** is where you
|
|
91
|
+
send the request. A **model** is what the model can do. They are different
|
|
92
|
+
objects, and `Resolution` gives you both.
|
|
93
|
+
|
|
94
|
+
## Install
|
|
95
|
+
|
|
96
|
+
```bash
|
|
97
|
+
pip install ai-registry # core: pydantic + pyyaml only
|
|
98
|
+
pip install "ai-registry[cli]" # + ai-registry command
|
|
99
|
+
pip install "ai-registry[openai]" # + OpenAI SDK adapter
|
|
100
|
+
pip install "ai-registry[pydantic-ai]" # + Pydantic AI adapter
|
|
101
|
+
pip install "ai-registry[langchain]" # + LangChain chat/embeddings adapter
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Requires Python 3.11+.
|
|
105
|
+
|
|
106
|
+
## Configure
|
|
107
|
+
|
|
108
|
+
Projects keep configuration in `.ai-registry/` at the repository root, resolved
|
|
109
|
+
from any subdirectory:
|
|
110
|
+
|
|
111
|
+
```text
|
|
112
|
+
<project-root>/
|
|
113
|
+
└── .ai-registry/
|
|
114
|
+
├── providers/ # OPTIONAL — override or add providers
|
|
115
|
+
├── registry.yaml # default
|
|
116
|
+
├── registry.dev.yaml # OPTIONAL — profile "dev"
|
|
117
|
+
└── registry.prod.yaml # OPTIONAL — profile "prod"
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
```yaml
|
|
121
|
+
# .ai-registry/registry.yaml
|
|
122
|
+
accounts:
|
|
123
|
+
- id: deepseek-primary
|
|
124
|
+
provider_id: deepseek
|
|
125
|
+
credential: env:DEEPSEEK_API_KEY # a reference, never a key
|
|
126
|
+
enabled_models: [deepseek-chat, deepseek-reasoner]
|
|
127
|
+
- id: uniapi-chat
|
|
128
|
+
provider_id: uniapi
|
|
129
|
+
credential: env:UNIAPI_API_KEY
|
|
130
|
+
enabled_models: [deepseek-v4-flash]
|
|
131
|
+
|
|
132
|
+
aliases:
|
|
133
|
+
- id: chat-default
|
|
134
|
+
deployments:
|
|
135
|
+
- account_id: deepseek-primary # preferred
|
|
136
|
+
model_id: deepseek-chat
|
|
137
|
+
- account_id: uniapi-chat # used when the first is cooling down
|
|
138
|
+
model_id: deepseek-v4-flash
|
|
139
|
+
routing: {priority: 1}
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Select a profile with `Registry.from_project(profile="prod")` or
|
|
143
|
+
`AI_REGISTRY_PROFILE=prod`.
|
|
144
|
+
|
|
145
|
+
## Use the result
|
|
146
|
+
|
|
147
|
+
Build a client yourself, or use an optional adapter:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
from ai_registry import ClientFactory
|
|
151
|
+
|
|
152
|
+
factory = ClientFactory.from_project()
|
|
153
|
+
|
|
154
|
+
# An openai.OpenAI client, pointed at the resolved endpoint and key:
|
|
155
|
+
client = factory.openai_client("chat-default")
|
|
156
|
+
|
|
157
|
+
# A Pydantic AI model — a framework with no config-file registry of its own:
|
|
158
|
+
model = factory.pydantic_ai_model("chat-default")
|
|
159
|
+
|
|
160
|
+
# Registry routing becomes Pydantic AI request-time failover:
|
|
161
|
+
model = factory.pydantic_ai_fallback_model("chat-default")
|
|
162
|
+
|
|
163
|
+
# A LangChain model — LangGraph states take it directly as a node model:
|
|
164
|
+
model = factory.langchain_chat_model("chat-default")
|
|
165
|
+
```
|
|
166
|
+
|
|
167
|
+
Or drive failover yourself — `candidates()` returns every healthy deployment in
|
|
168
|
+
preference order:
|
|
169
|
+
|
|
170
|
+
```python
|
|
171
|
+
for resolution in factory.registry.candidates("chat-default"):
|
|
172
|
+
try:
|
|
173
|
+
response = call(resolution)
|
|
174
|
+
except TransientError:
|
|
175
|
+
factory.registry.record_failure(resolution) # cools this target down everywhere
|
|
176
|
+
continue
|
|
177
|
+
factory.registry.record_success(resolution)
|
|
178
|
+
break
|
|
179
|
+
```
|
|
180
|
+
|
|
181
|
+
## Inspect it
|
|
182
|
+
|
|
183
|
+
```console
|
|
184
|
+
$ ai-registry explain chat-default
|
|
185
|
+
1. deepseek-primary/deepseek-chat -> https://api.deepseek.com/anthropic (credential: env (ok))
|
|
186
|
+
2. uniapi-chat/deepseek-v4-flash -> https://uni-api.cstcloud.cn/v1 (credential: env (unresolved))
|
|
187
|
+
|
|
188
|
+
$ ai-registry models --type chat --tools --min-context 500000
|
|
189
|
+
deepseek deepseek-v4-flash chat 1000000 reasoning,structured_output,tools
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
`resolve` and `explain` report whether a credential resolves, never its name or
|
|
193
|
+
value.
|
|
194
|
+
|
|
195
|
+
## Features
|
|
196
|
+
|
|
197
|
+
- **Alias-first resolution**, with an explicit `account-id/model-id` escape hatch.
|
|
198
|
+
- **Multiple accounts and keys** per provider, as reusable first-class objects.
|
|
199
|
+
- **Deterministic deployment selection**: priority, weight, disabled, plus a
|
|
200
|
+
health store with time-based cooldown and exponential backoff.
|
|
201
|
+
- **Capability-aware catalog queries** — find a chat model with tools and a
|
|
202
|
+
200k window without hardcoding a model id.
|
|
203
|
+
- **Secret references, not secrets**: `env:` / `secret_ref:`, resolved lazily,
|
|
204
|
+
with a pluggable `SecretResolver` for Vault, AWS, or Kubernetes.
|
|
205
|
+
- **Secret-safe by construction**: literal keys rejected; `repr`, dumps, and
|
|
206
|
+
errors redacted; credential-bearing URLs refused.
|
|
207
|
+
- **Bring your own catalog**: `ai_registry.catalog.load_models_dev()` imports a
|
|
208
|
+
models.dev `api.json`, and the built-in catalog takes its whole model set from
|
|
209
|
+
a vendored snapshot — provider files carry only endpoints, settings, and
|
|
210
|
+
deltas. Maintainers refresh it with `ai-registry catalog diff|sync`.
|
|
211
|
+
- **Observability built in**: `Resolution.otel_attributes()` emits OpenTelemetry
|
|
212
|
+
GenAI attributes plus which alias and account were selected.
|
|
213
|
+
- **Testable**: `ai_registry.testing` ships supported fakes.
|
|
214
|
+
|
|
215
|
+
## Why not X?
|
|
216
|
+
|
|
217
|
+
| | Alias layer | Resolution separable from invocation | Multiple keys per provider | In-process | Rejects literal secrets |
|
|
218
|
+
|---|---|---|---|---|---|
|
|
219
|
+
| **ai-registry** | config file | yes, that's all it does | yes, as `Account` | yes (6 pkgs) | **yes** |
|
|
220
|
+
| **LiteLLM Router** | `model_name` groups | yes, via `get_available_deployment()` | yes, by repeating entries | yes (49 pkgs) | no |
|
|
221
|
+
| **`llm` (simonw)** | `aliases.json` | yes, `llm.get_model()` | yes, `api_key_name` | yes | no (plaintext `keys.json`) |
|
|
222
|
+
| **Pydantic AI** | `'openai:gpt-4o'` string | no — returns a live model | no | yes | n/a |
|
|
223
|
+
| **Vercel AI SDK registry** | `customProvider` (code) | no — returns a live model | no | yes (TS) | n/a |
|
|
224
|
+
| **Portkey / Helicone / Bifrost / Kong** | gateway config | no | yes | no — network hop | varies |
|
|
225
|
+
|
|
226
|
+
**Use LiteLLM instead** if you want one `completion()` call across providers,
|
|
227
|
+
request translation, retries, spend tracking, or a proxy. It does far more than
|
|
228
|
+
this library and does it well.
|
|
229
|
+
|
|
230
|
+
**Use ai-registry** if you already have SDKs you like and only want the
|
|
231
|
+
configuration layer — especially if the dependency footprint of the component
|
|
232
|
+
that touches every API key you own is something you care about. For scale: the
|
|
233
|
+
resolved closure is 49 packages for `litellm`, 107 for `litellm[proxy]`, and 6
|
|
234
|
+
here. In March 2026 two `litellm` releases were
|
|
235
|
+
[compromised on PyPI](https://docs.litellm.ai/blog/security-update-march-2026)
|
|
236
|
+
with a `.pth` file that harvested cloud credentials on every Python start. Fewer
|
|
237
|
+
moving parts near your keys is a real property, not an aesthetic one.
|
|
238
|
+
|
|
239
|
+
The two are not exclusive: point an `Account` at a LiteLLM proxy and use this to
|
|
240
|
+
decide which one.
|
|
241
|
+
|
|
242
|
+
## Documentation
|
|
243
|
+
|
|
244
|
+
| Guide | Contents |
|
|
245
|
+
| --- | --- |
|
|
246
|
+
| [Overview](docs/overview.md) | The idea, concepts, architecture, goals and non-goals. |
|
|
247
|
+
| [Getting started](docs/getting-started.md) | Install, first resolution, loading files, built-in data. |
|
|
248
|
+
| [Configuration](docs/configuration.md) | Complete file format, field reference, validation. |
|
|
249
|
+
| [Use cases](docs/use-cases.md) | Failover, weighted routing, embeddings, custom endpoints, secret resolvers. |
|
|
250
|
+
| [API reference](docs/api-reference.md) | Every public class, function, field, and behavior. |
|
|
251
|
+
| [Security](docs/security.md) | Secret-handling guarantees, redaction rules, limitations. |
|
|
252
|
+
| [Development](docs/development.md) | Environment, checks, conventions, testing. |
|
|
253
|
+
|
|
254
|
+
## Non-goals
|
|
255
|
+
|
|
256
|
+
This package does not make network requests, define a unified AI response
|
|
257
|
+
interface, translate or retry requests, or ship a gateway. Deployment selection
|
|
258
|
+
is deterministic and based on configuration plus recorded health; request-time
|
|
259
|
+
concerns belong to the caller or the adapter.
|
|
260
|
+
|
|
261
|
+
## License
|
|
262
|
+
|
|
263
|
+
[MIT](LICENSE).
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
# ai-registry
|
|
2
|
+
|
|
3
|
+
[](https://pypi.org/project/ai-registry/)
|
|
4
|
+
[](https://pypi.org/project/ai-registry/)
|
|
5
|
+
[](https://github.com/limoiie/ai-registry/actions/workflows/ci.yml)
|
|
6
|
+
[](https://github.com/astral-sh/ruff)
|
|
7
|
+
[](https://mypy-lang.org/)
|
|
8
|
+
[](LICENSE)
|
|
9
|
+
|
|
10
|
+
**Resolve a stable alias to a concrete AI deployment — provider, model,
|
|
11
|
+
account, endpoint — and stop there.**
|
|
12
|
+
|
|
13
|
+
`ai-registry` is the configuration layer under your AI stack, not another client.
|
|
14
|
+
It answers one question — *"which provider, which model, whose key, what URL?"* —
|
|
15
|
+
and hands the answer to whatever SDK you already use.
|
|
16
|
+
|
|
17
|
+
It makes **no network requests**, defines **no chat interface**, ships **no
|
|
18
|
+
gateway**, and **cannot hold a literal API key**. The runtime dependency closure
|
|
19
|
+
is six packages.
|
|
20
|
+
|
|
21
|
+
```python
|
|
22
|
+
from ai_registry import Registry
|
|
23
|
+
|
|
24
|
+
registry = Registry.from_project() # reads ./.ai-registry/registry.yaml
|
|
25
|
+
resolution = registry.resolve("chat-default")
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
| What you get | Example | Why it matters |
|
|
29
|
+
| --- | --- | --- |
|
|
30
|
+
| `resolution.provider_id` | `'deepseek'` | Which provider was chosen. |
|
|
31
|
+
| `resolution.model_id` | `'deepseek-chat'` | The id to send, and to log. |
|
|
32
|
+
| `resolution.account_id` | `'deepseek-primary'` | **Which key** was chosen. |
|
|
33
|
+
| `resolution.endpoint` | `'https://api.deepseek.com/anthropic'` | Where to send it. |
|
|
34
|
+
| `resolution.model.capabilities` | `tools=True, reasoning=False, …` | What the model can do. |
|
|
35
|
+
| `resolution.resolve_credential()` | `'sk-…'` | Resolved lazily, only at the adapter boundary. |
|
|
36
|
+
|
|
37
|
+
The model, provider, account, or endpoint behind `"chat-default"` can change in
|
|
38
|
+
configuration without touching application code.
|
|
39
|
+
|
|
40
|
+
## Why a resolver instead of a client
|
|
41
|
+
|
|
42
|
+
Most tools in this space fuse two separate concerns: *deciding what to call* and
|
|
43
|
+
*calling it*. Fusing them means you adopt someone's request translation, error
|
|
44
|
+
types, retry semantics, and dependency tree in order to get their configuration
|
|
45
|
+
model.
|
|
46
|
+
|
|
47
|
+
That fusion has a well-known failure mode. From
|
|
48
|
+
[openai/codex#1370](https://github.com/openai/codex/issues/1370), on Azure
|
|
49
|
+
deployment names:
|
|
50
|
+
|
|
51
|
+
> No logic should inspect the user's deployment name when making assumptions
|
|
52
|
+
> about model capabilities.
|
|
53
|
+
|
|
54
|
+
That is exactly the split this library enforces. A **deployment** is where you
|
|
55
|
+
send the request. A **model** is what the model can do. They are different
|
|
56
|
+
objects, and `Resolution` gives you both.
|
|
57
|
+
|
|
58
|
+
## Install
|
|
59
|
+
|
|
60
|
+
```bash
|
|
61
|
+
pip install ai-registry # core: pydantic + pyyaml only
|
|
62
|
+
pip install "ai-registry[cli]" # + ai-registry command
|
|
63
|
+
pip install "ai-registry[openai]" # + OpenAI SDK adapter
|
|
64
|
+
pip install "ai-registry[pydantic-ai]" # + Pydantic AI adapter
|
|
65
|
+
pip install "ai-registry[langchain]" # + LangChain chat/embeddings adapter
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Requires Python 3.11+.
|
|
69
|
+
|
|
70
|
+
## Configure
|
|
71
|
+
|
|
72
|
+
Projects keep configuration in `.ai-registry/` at the repository root, resolved
|
|
73
|
+
from any subdirectory:
|
|
74
|
+
|
|
75
|
+
```text
|
|
76
|
+
<project-root>/
|
|
77
|
+
└── .ai-registry/
|
|
78
|
+
├── providers/ # OPTIONAL — override or add providers
|
|
79
|
+
├── registry.yaml # default
|
|
80
|
+
├── registry.dev.yaml # OPTIONAL — profile "dev"
|
|
81
|
+
└── registry.prod.yaml # OPTIONAL — profile "prod"
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
```yaml
|
|
85
|
+
# .ai-registry/registry.yaml
|
|
86
|
+
accounts:
|
|
87
|
+
- id: deepseek-primary
|
|
88
|
+
provider_id: deepseek
|
|
89
|
+
credential: env:DEEPSEEK_API_KEY # a reference, never a key
|
|
90
|
+
enabled_models: [deepseek-chat, deepseek-reasoner]
|
|
91
|
+
- id: uniapi-chat
|
|
92
|
+
provider_id: uniapi
|
|
93
|
+
credential: env:UNIAPI_API_KEY
|
|
94
|
+
enabled_models: [deepseek-v4-flash]
|
|
95
|
+
|
|
96
|
+
aliases:
|
|
97
|
+
- id: chat-default
|
|
98
|
+
deployments:
|
|
99
|
+
- account_id: deepseek-primary # preferred
|
|
100
|
+
model_id: deepseek-chat
|
|
101
|
+
- account_id: uniapi-chat # used when the first is cooling down
|
|
102
|
+
model_id: deepseek-v4-flash
|
|
103
|
+
routing: {priority: 1}
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
Select a profile with `Registry.from_project(profile="prod")` or
|
|
107
|
+
`AI_REGISTRY_PROFILE=prod`.
|
|
108
|
+
|
|
109
|
+
## Use the result
|
|
110
|
+
|
|
111
|
+
Build a client yourself, or use an optional adapter:
|
|
112
|
+
|
|
113
|
+
```python
|
|
114
|
+
from ai_registry import ClientFactory
|
|
115
|
+
|
|
116
|
+
factory = ClientFactory.from_project()
|
|
117
|
+
|
|
118
|
+
# An openai.OpenAI client, pointed at the resolved endpoint and key:
|
|
119
|
+
client = factory.openai_client("chat-default")
|
|
120
|
+
|
|
121
|
+
# A Pydantic AI model — a framework with no config-file registry of its own:
|
|
122
|
+
model = factory.pydantic_ai_model("chat-default")
|
|
123
|
+
|
|
124
|
+
# Registry routing becomes Pydantic AI request-time failover:
|
|
125
|
+
model = factory.pydantic_ai_fallback_model("chat-default")
|
|
126
|
+
|
|
127
|
+
# A LangChain model — LangGraph states take it directly as a node model:
|
|
128
|
+
model = factory.langchain_chat_model("chat-default")
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Or drive failover yourself — `candidates()` returns every healthy deployment in
|
|
132
|
+
preference order:
|
|
133
|
+
|
|
134
|
+
```python
|
|
135
|
+
for resolution in factory.registry.candidates("chat-default"):
|
|
136
|
+
try:
|
|
137
|
+
response = call(resolution)
|
|
138
|
+
except TransientError:
|
|
139
|
+
factory.registry.record_failure(resolution) # cools this target down everywhere
|
|
140
|
+
continue
|
|
141
|
+
factory.registry.record_success(resolution)
|
|
142
|
+
break
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
## Inspect it
|
|
146
|
+
|
|
147
|
+
```console
|
|
148
|
+
$ ai-registry explain chat-default
|
|
149
|
+
1. deepseek-primary/deepseek-chat -> https://api.deepseek.com/anthropic (credential: env (ok))
|
|
150
|
+
2. uniapi-chat/deepseek-v4-flash -> https://uni-api.cstcloud.cn/v1 (credential: env (unresolved))
|
|
151
|
+
|
|
152
|
+
$ ai-registry models --type chat --tools --min-context 500000
|
|
153
|
+
deepseek deepseek-v4-flash chat 1000000 reasoning,structured_output,tools
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
`resolve` and `explain` report whether a credential resolves, never its name or
|
|
157
|
+
value.
|
|
158
|
+
|
|
159
|
+
## Features
|
|
160
|
+
|
|
161
|
+
- **Alias-first resolution**, with an explicit `account-id/model-id` escape hatch.
|
|
162
|
+
- **Multiple accounts and keys** per provider, as reusable first-class objects.
|
|
163
|
+
- **Deterministic deployment selection**: priority, weight, disabled, plus a
|
|
164
|
+
health store with time-based cooldown and exponential backoff.
|
|
165
|
+
- **Capability-aware catalog queries** — find a chat model with tools and a
|
|
166
|
+
200k window without hardcoding a model id.
|
|
167
|
+
- **Secret references, not secrets**: `env:` / `secret_ref:`, resolved lazily,
|
|
168
|
+
with a pluggable `SecretResolver` for Vault, AWS, or Kubernetes.
|
|
169
|
+
- **Secret-safe by construction**: literal keys rejected; `repr`, dumps, and
|
|
170
|
+
errors redacted; credential-bearing URLs refused.
|
|
171
|
+
- **Bring your own catalog**: `ai_registry.catalog.load_models_dev()` imports a
|
|
172
|
+
models.dev `api.json`, and the built-in catalog takes its whole model set from
|
|
173
|
+
a vendored snapshot — provider files carry only endpoints, settings, and
|
|
174
|
+
deltas. Maintainers refresh it with `ai-registry catalog diff|sync`.
|
|
175
|
+
- **Observability built in**: `Resolution.otel_attributes()` emits OpenTelemetry
|
|
176
|
+
GenAI attributes plus which alias and account were selected.
|
|
177
|
+
- **Testable**: `ai_registry.testing` ships supported fakes.
|
|
178
|
+
|
|
179
|
+
## Why not X?
|
|
180
|
+
|
|
181
|
+
| | Alias layer | Resolution separable from invocation | Multiple keys per provider | In-process | Rejects literal secrets |
|
|
182
|
+
|---|---|---|---|---|---|
|
|
183
|
+
| **ai-registry** | config file | yes, that's all it does | yes, as `Account` | yes (6 pkgs) | **yes** |
|
|
184
|
+
| **LiteLLM Router** | `model_name` groups | yes, via `get_available_deployment()` | yes, by repeating entries | yes (49 pkgs) | no |
|
|
185
|
+
| **`llm` (simonw)** | `aliases.json` | yes, `llm.get_model()` | yes, `api_key_name` | yes | no (plaintext `keys.json`) |
|
|
186
|
+
| **Pydantic AI** | `'openai:gpt-4o'` string | no — returns a live model | no | yes | n/a |
|
|
187
|
+
| **Vercel AI SDK registry** | `customProvider` (code) | no — returns a live model | no | yes (TS) | n/a |
|
|
188
|
+
| **Portkey / Helicone / Bifrost / Kong** | gateway config | no | yes | no — network hop | varies |
|
|
189
|
+
|
|
190
|
+
**Use LiteLLM instead** if you want one `completion()` call across providers,
|
|
191
|
+
request translation, retries, spend tracking, or a proxy. It does far more than
|
|
192
|
+
this library and does it well.
|
|
193
|
+
|
|
194
|
+
**Use ai-registry** if you already have SDKs you like and only want the
|
|
195
|
+
configuration layer — especially if the dependency footprint of the component
|
|
196
|
+
that touches every API key you own is something you care about. For scale: the
|
|
197
|
+
resolved closure is 49 packages for `litellm`, 107 for `litellm[proxy]`, and 6
|
|
198
|
+
here. In March 2026 two `litellm` releases were
|
|
199
|
+
[compromised on PyPI](https://docs.litellm.ai/blog/security-update-march-2026)
|
|
200
|
+
with a `.pth` file that harvested cloud credentials on every Python start. Fewer
|
|
201
|
+
moving parts near your keys is a real property, not an aesthetic one.
|
|
202
|
+
|
|
203
|
+
The two are not exclusive: point an `Account` at a LiteLLM proxy and use this to
|
|
204
|
+
decide which one.
|
|
205
|
+
|
|
206
|
+
## Documentation
|
|
207
|
+
|
|
208
|
+
| Guide | Contents |
|
|
209
|
+
| --- | --- |
|
|
210
|
+
| [Overview](docs/overview.md) | The idea, concepts, architecture, goals and non-goals. |
|
|
211
|
+
| [Getting started](docs/getting-started.md) | Install, first resolution, loading files, built-in data. |
|
|
212
|
+
| [Configuration](docs/configuration.md) | Complete file format, field reference, validation. |
|
|
213
|
+
| [Use cases](docs/use-cases.md) | Failover, weighted routing, embeddings, custom endpoints, secret resolvers. |
|
|
214
|
+
| [API reference](docs/api-reference.md) | Every public class, function, field, and behavior. |
|
|
215
|
+
| [Security](docs/security.md) | Secret-handling guarantees, redaction rules, limitations. |
|
|
216
|
+
| [Development](docs/development.md) | Environment, checks, conventions, testing. |
|
|
217
|
+
|
|
218
|
+
## Non-goals
|
|
219
|
+
|
|
220
|
+
This package does not make network requests, define a unified AI response
|
|
221
|
+
interface, translate or retry requests, or ship a gateway. Deployment selection
|
|
222
|
+
is deterministic and based on configuration plus recorded health; request-time
|
|
223
|
+
concerns belong to the caller or the adapter.
|
|
224
|
+
|
|
225
|
+
## License
|
|
226
|
+
|
|
227
|
+
[MIT](LICENSE).
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["uv-build>=0.8.0,<0.9.0"]
|
|
3
|
+
build-backend = "uv_build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ai-registry"
|
|
7
|
+
version = "0.3.1"
|
|
8
|
+
description = "Resolve a stable alias to a concrete AI provider deployment. No requests, no gateway, no literal secrets."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "MIT"
|
|
12
|
+
license-files = ["LICENSE"]
|
|
13
|
+
keywords = [
|
|
14
|
+
"ai",
|
|
15
|
+
"llm",
|
|
16
|
+
"model-routing",
|
|
17
|
+
"provider-registry",
|
|
18
|
+
"configuration",
|
|
19
|
+
"secrets",
|
|
20
|
+
]
|
|
21
|
+
classifiers = [
|
|
22
|
+
"Development Status :: 3 - Alpha",
|
|
23
|
+
"Intended Audience :: Developers",
|
|
24
|
+
"License :: OSI Approved :: MIT License",
|
|
25
|
+
"Programming Language :: Python :: 3 :: Only",
|
|
26
|
+
"Programming Language :: Python :: 3.11",
|
|
27
|
+
"Programming Language :: Python :: 3.12",
|
|
28
|
+
"Programming Language :: Python :: 3.13",
|
|
29
|
+
"Programming Language :: Python :: 3.14",
|
|
30
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
31
|
+
"Typing :: Typed",
|
|
32
|
+
]
|
|
33
|
+
dependencies = [
|
|
34
|
+
"pydantic>=2",
|
|
35
|
+
"pyyaml",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.optional-dependencies]
|
|
39
|
+
cli = ["typer>=0.12"]
|
|
40
|
+
openai = ["openai>=1.40"]
|
|
41
|
+
# OpenAIChatModel (formerly OpenAIModel) and pydantic_ai.providers.* are
|
|
42
|
+
# required by the adapter, so the floor is 1.0 rather than the oldest release.
|
|
43
|
+
pydantic-ai = ["pydantic-ai-slim[openai]>=1.0"]
|
|
44
|
+
# `api_key`/`base_url`/`model`/`timeout` constructor kwargs are required by the
|
|
45
|
+
# adapter; langchain-openai and langchain-anthropic provide them from 0.2 on.
|
|
46
|
+
langchain = ["langchain-openai>=0.2", "langchain-anthropic>=0.2"]
|
|
47
|
+
|
|
48
|
+
[project.scripts]
|
|
49
|
+
ai-registry = "ai_registry.cli:main"
|
|
50
|
+
|
|
51
|
+
[project.urls]
|
|
52
|
+
Homepage = "https://github.com/limoiie/ai-registry"
|
|
53
|
+
Documentation = "https://github.com/limoiie/ai-registry#documentation"
|
|
54
|
+
Source = "https://github.com/limoiie/ai-registry"
|
|
55
|
+
Changelog = "https://github.com/limoiie/ai-registry/blob/main/CHANGELOG.md"
|
|
56
|
+
Issues = "https://github.com/limoiie/ai-registry/issues"
|
|
57
|
+
|
|
58
|
+
[dependency-groups]
|
|
59
|
+
dev = [
|
|
60
|
+
"mypy>=1.13",
|
|
61
|
+
"pytest>=8.3",
|
|
62
|
+
"pytest-cov>=5.0",
|
|
63
|
+
"ruff>=0.8",
|
|
64
|
+
"types-pyyaml>=6.0",
|
|
65
|
+
# The CLI extra is exercised by the test suite; the runtime core does not
|
|
66
|
+
# depend on it.
|
|
67
|
+
"typer>=0.12",
|
|
68
|
+
]
|
|
69
|
+
docs = [
|
|
70
|
+
"mkdocs-material>=9.5",
|
|
71
|
+
"mkdocstrings[python]>=0.26",
|
|
72
|
+
]
|
|
73
|
+
|
|
74
|
+
[tool.uv]
|
|
75
|
+
default-groups = ["dev"]
|
|
76
|
+
|
|
77
|
+
[tool.pytest.ini_options]
|
|
78
|
+
testpaths = ["tests"]
|
|
79
|
+
addopts = "--strict-markers --strict-config"
|
|
80
|
+
|
|
81
|
+
[tool.coverage.run]
|
|
82
|
+
source = ["ai_registry"]
|
|
83
|
+
branch = true
|
|
84
|
+
|
|
85
|
+
[tool.coverage.report]
|
|
86
|
+
exclude_also = [
|
|
87
|
+
"if TYPE_CHECKING:",
|
|
88
|
+
"raise NotImplementedError",
|
|
89
|
+
"\\.\\.\\.",
|
|
90
|
+
]
|
|
91
|
+
|
|
92
|
+
[tool.ruff]
|
|
93
|
+
line-length = 88
|
|
94
|
+
target-version = "py311"
|
|
95
|
+
# `docs/plans` is a keep-verbatim record: the formatter would rewrite the
|
|
96
|
+
# illustrative code fences (quotes, wrapping) and alter the historical notes.
|
|
97
|
+
extend-exclude = ["docs/plans"]
|
|
98
|
+
|
|
99
|
+
[tool.ruff.lint]
|
|
100
|
+
select = [
|
|
101
|
+
"E", # pycodestyle errors
|
|
102
|
+
"F", # pyflakes
|
|
103
|
+
"I", # import sorting
|
|
104
|
+
"UP", # pyupgrade
|
|
105
|
+
"B", # bugbear
|
|
106
|
+
"SIM", # simplify
|
|
107
|
+
"RUF", # ruff-specific
|
|
108
|
+
]
|
|
109
|
+
|
|
110
|
+
[tool.ruff.lint.per-file-ignores]
|
|
111
|
+
# Typer expresses CLI arguments as function defaults.
|
|
112
|
+
"src/ai_registry/cli.py" = ["B008"]
|
|
113
|
+
|
|
114
|
+
[tool.mypy]
|
|
115
|
+
python_version = "3.11"
|
|
116
|
+
strict = true
|
|
117
|
+
|
|
118
|
+
[[tool.mypy.overrides]]
|
|
119
|
+
module = ["ai_registry.cli"]
|
|
120
|
+
# Typer's decorators are untyped in the optional-extra import path.
|
|
121
|
+
disallow_untyped_decorators = false
|
|
122
|
+
|
|
123
|
+
[[tool.mypy.overrides]]
|
|
124
|
+
# Optional extras are not installed in the default dev environment; the adapters
|
|
125
|
+
# import them defensively and are covered by tests using stub modules.
|
|
126
|
+
module = [
|
|
127
|
+
"openai",
|
|
128
|
+
"openai.*",
|
|
129
|
+
"pydantic_ai",
|
|
130
|
+
"pydantic_ai.*",
|
|
131
|
+
"langchain_core",
|
|
132
|
+
"langchain_core.*",
|
|
133
|
+
"langchain_openai",
|
|
134
|
+
"langchain_openai.*",
|
|
135
|
+
"langchain_anthropic",
|
|
136
|
+
"langchain_anthropic.*",
|
|
137
|
+
]
|
|
138
|
+
ignore_missing_imports = true
|