infimal 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- infimal-0.2.1/.gitignore +49 -0
- infimal-0.2.1/PKG-INFO +152 -0
- infimal-0.2.1/README.md +138 -0
- infimal-0.2.1/pyproject.toml +37 -0
- infimal-0.2.1/src/infimal/__init__.py +154 -0
- infimal-0.2.1/src/infimal/_synth.py +347 -0
- infimal-0.2.1/src/infimal/app.py +253 -0
- infimal-0.2.1/src/infimal/client.py +697 -0
- infimal-0.2.1/src/infimal/frame.py +76 -0
- infimal-0.2.1/src/infimal/gateway.py +686 -0
- infimal-0.2.1/src/infimal/harness.py +499 -0
- infimal-0.2.1/src/infimal/models.py +202 -0
- infimal-0.2.1/src/infimal/plan.py +244 -0
- infimal-0.2.1/src/infimal/request.py +766 -0
- infimal-0.2.1/src/infimal/ring.py +294 -0
- infimal-0.2.1/src/infimal/runtime.py +100 -0
- infimal-0.2.1/tests/conftest.py +59 -0
- infimal-0.2.1/tests/fixtures/README.md +18 -0
- infimal-0.2.1/tests/fixtures/chat_stream.sse +12 -0
- infimal-0.2.1/tests/fixtures/deploy_request.json +27 -0
- infimal-0.2.1/tests/test_app.py +141 -0
- infimal-0.2.1/tests/test_client_contract.py +231 -0
- infimal-0.2.1/tests/test_endpoint.py +98 -0
- infimal-0.2.1/tests/test_env.py +77 -0
- infimal-0.2.1/tests/test_exports.py +46 -0
- infimal-0.2.1/tests/test_frame.py +100 -0
- infimal-0.2.1/tests/test_gateway.py +350 -0
- infimal-0.2.1/tests/test_harness.py +431 -0
- infimal-0.2.1/tests/test_models.py +85 -0
- infimal-0.2.1/tests/test_plan.py +177 -0
- infimal-0.2.1/tests/test_request.py +364 -0
- infimal-0.2.1/tests/test_ring.py +191 -0
- infimal-0.2.1/tests/test_runtime.py +52 -0
- infimal-0.2.1/tests/test_synth.py +339 -0
- infimal-0.2.1/uv.lock +343 -0
infimal-0.2.1/.gitignore
ADDED
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
.env
|
|
2
|
+
.env.*
|
|
3
|
+
bench/results/
|
|
4
|
+
bench/sample_10s.wav
|
|
5
|
+
__pycache__/
|
|
6
|
+
*.pyc
|
|
7
|
+
.venv/
|
|
8
|
+
!.env.example
|
|
9
|
+
# The website's two API origins, which are `PUBLIC_` — compiled into the bundle and readable in the
|
|
10
|
+
# page source of the deployed site — so there is nothing to keep out of the repository, and without
|
|
11
|
+
# them `cd web && npm run build` fails on a fresh clone (D314). Nothing secret goes in that file.
|
|
12
|
+
!web/.env
|
|
13
|
+
target/
|
|
14
|
+
.sqlx/
|
|
15
|
+
sdk/dist/
|
|
16
|
+
runner/dist/
|
|
17
|
+
studio/dist/
|
|
18
|
+
|
|
19
|
+
# The Railway IaC SDK. `.railway/railway.ts` imports `railway/iac`, and `railway config
|
|
20
|
+
# plan` cannot evaluate it without the package; `package.json` pins the version so the
|
|
21
|
+
# plan is the same one everywhere. `npm install` at the repository root is the setup step.
|
|
22
|
+
node_modules/
|
|
23
|
+
sdk/.venv/
|
|
24
|
+
.pytest_cache/
|
|
25
|
+
.ruff_cache/
|
|
26
|
+
|
|
27
|
+
# Claude Code session state and agent worktrees
|
|
28
|
+
.claude/
|
|
29
|
+
|
|
30
|
+
# per-user Claude Code permissions
|
|
31
|
+
.claude/settings.local.json
|
|
32
|
+
|
|
33
|
+
# Native CLI release payloads are built in CI and staged into the static site, never committed.
|
|
34
|
+
web/static/cli/latest
|
|
35
|
+
web/static/cli/latest.json
|
|
36
|
+
web/static/cli/*/
|
|
37
|
+
|
|
38
|
+
# Evidence (run logs, renders, probes) goes to the bucket, never git (D361). Some of it has held
|
|
39
|
+
# live credentials: output/h100-deployment-2026-09-26/bootstrap-h100.sh carries a Tailscale key.
|
|
40
|
+
output/
|
|
41
|
+
|
|
42
|
+
# macOS Finder metadata.
|
|
43
|
+
.DS_Store
|
|
44
|
+
**/.DS_Store
|
|
45
|
+
|
|
46
|
+
# Customer gateway declarations (`examples/<tenant>/gateway.py`): a tenant's own model names,
|
|
47
|
+
# provider endpoints and the *names* of its credential variables. No value is in them, but they
|
|
48
|
+
# are one customer's configuration and not the platform's, so they stay out of the tree.
|
|
49
|
+
examples/
|
infimal-0.2.1/PKG-INFO
ADDED
|
@@ -0,0 +1,152 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: infimal
|
|
3
|
+
Version: 0.2.1
|
|
4
|
+
Summary: The infimal SDK, the in-container harness, and the module behind the native CLI's apps plan, apply and deploy.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Requires-Python: >=3.11
|
|
7
|
+
Requires-Dist: httpx>=0.27
|
|
8
|
+
Requires-Dist: pydantic>=2.7
|
|
9
|
+
Provides-Extra: dev
|
|
10
|
+
Requires-Dist: pytest-asyncio>=0.23; extra == 'dev'
|
|
11
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
12
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
|
|
15
|
+
# infimal
|
|
16
|
+
|
|
17
|
+
The infimal Python SDK and in-container harness.
|
|
18
|
+
|
|
19
|
+
```python
|
|
20
|
+
from infimal import Client
|
|
21
|
+
|
|
22
|
+
client = Client.from_env() # INFIMAL_API_KEY; INFIMAL_ENDPOINT is optional
|
|
23
|
+
print(client.model("deepseek/deepseek-v4-flash").chat("Explain this diff.").max_tokens(256).run().text)
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
`client.model(id)` is a request handle: `.chat(...)`, `.embeddings(...)`, `.speech(...)`,
|
|
27
|
+
`.images(...)` and `.video(...)` build a request, `run()` or `stream()` sends it once, and
|
|
28
|
+
`.capabilities()` says what the listing accepts and whether it is serving. Images and video are
|
|
29
|
+
jobs: `submit()` returns a `Job` you can `wait()`, `refresh()` or `cancel()`, and `client.job(id)`
|
|
30
|
+
adopts one after a restart.
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from infimal import App, Latency, Scale
|
|
34
|
+
|
|
35
|
+
app = App("voice", perf=Latency(p95_ms=300), scale=Scale(max_replicas=4))
|
|
36
|
+
|
|
37
|
+
@app.setup
|
|
38
|
+
def load():
|
|
39
|
+
# Runs once, on a builder GPU. Weights, CUDA context, warm caches, CUDA graphs — everything
|
|
40
|
+
# expensive belongs here, because what happens next is that the whole live process is frozen.
|
|
41
|
+
return load_model()
|
|
42
|
+
|
|
43
|
+
@app.handler
|
|
44
|
+
def transcribe(model, request):
|
|
45
|
+
return model(request["audio"])
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
## The environment
|
|
49
|
+
|
|
50
|
+
| Variable | Meaning |
|
|
51
|
+
|---|---|
|
|
52
|
+
| `INFIMAL_API_KEY` | the API key; required |
|
|
53
|
+
| `INFIMAL_ENDPOINT` | the API origin; defaults to `https://api.infimal.ai`; the host from before the rename remains supported |
|
|
54
|
+
|
|
55
|
+
`GLOW_API_KEY` and `GLOW_ENDPOINT`, the names before the rename, are still read **in 0.2 only**,
|
|
56
|
+
when the new ones are unset, and each read raises a one-line `DeprecationWarning` naming the
|
|
57
|
+
variable to set instead. 0.3 stops reading them. `GlowError` and `GlowUnreachable` resolve to
|
|
58
|
+
`InfimalError` and `InfimalUnreachable` for the same one release.
|
|
59
|
+
|
|
60
|
+
## The CLI
|
|
61
|
+
|
|
62
|
+
The customer CLI is the native `infimal` binary (`curl -fsSL https://infimal.ai/install.sh | sh`;
|
|
63
|
+
see `crates/infimal/README.md`), and it is the only one. This package installs **no console script
|
|
64
|
+
and has no verbs**; for the three commands that execute your module, the binary runs this package
|
|
65
|
+
(`python -m infimal._synth`, a private JSON-in, JSON-out subprocess described in `docs/sdk.md`), so
|
|
66
|
+
install the SDK where your module's dependencies are (`pip install infimal`), or point `INFIMAL_PYTHON`
|
|
67
|
+
at that interpreter.
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
infimal apps plan [module] # execute your module and show what deploying would change
|
|
71
|
+
infimal apps apply [module] # apply it: rebuild, or reconfigure a live placement in place
|
|
72
|
+
infimal apps deploy [module] # build a snapshot and place it
|
|
73
|
+
infimal apps snapshots <app> # published artifacts and how fast they restore
|
|
74
|
+
infimal apps status <app> # where the endpoint is on the residency gradient right now
|
|
75
|
+
infimal apps logs <app> # build logs
|
|
76
|
+
|
|
77
|
+
infimal models list # the price list: what each model costs you per million tokens
|
|
78
|
+
infimal billing balance # credit available, credit held against in-flight requests
|
|
79
|
+
infimal billing fund 25 --wait # buy credit; --wait blocks until the payment has landed
|
|
80
|
+
infimal usage show --since 30d [--by model]
|
|
81
|
+
infimal keys create|list|revoke
|
|
82
|
+
|
|
83
|
+
infimal jobs submit --surface images|video --model M --preset I1 --wait
|
|
84
|
+
infimal jobs get <id> # state, measured timings, presigned artifact URLs
|
|
85
|
+
infimal jobs list
|
|
86
|
+
infimal jobs cancel <id> # only before it starts; a running generation has been paid for
|
|
87
|
+
infimal jobs download <id> <dir>
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
Images and video are **jobs**, not requests: a clip is minutes of GPU time, so the submit reserves
|
|
91
|
+
the credit, answers `202` with an id, and you poll. `--wait` is that loop; the exit code is 1 if the
|
|
92
|
+
job failed.
|
|
93
|
+
|
|
94
|
+
A key is minted `full` — the whole account — unless you ask for the narrow role:
|
|
95
|
+
|
|
96
|
+
```
|
|
97
|
+
infimal keys create client-app --role inference # the OpenAI surface, the model list, your usage
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
Administrators have a second set, which needs a tenant on the operator's admin list. There is no
|
|
101
|
+
administrative key: administration belongs to the account, so no key can grant it or carry it.
|
|
102
|
+
|
|
103
|
+
```
|
|
104
|
+
infimal admin models list
|
|
105
|
+
infimal admin models set-price <id> --input 0.15 --output 0.60 [--cache-read 0.015]
|
|
106
|
+
infimal admin models set-price <id> --flat 0.04 | --unpriced [--unpin]
|
|
107
|
+
infimal admin sources add --model <id> --name <name> --base-url <url> \
|
|
108
|
+
--upstream-model <id> --key-env PROVIDER_KEY \
|
|
109
|
+
--cost-input 0.10 --cost-output 0.30 [--wire-config @wire.json]
|
|
110
|
+
infimal admin sources list|rekey|delete
|
|
111
|
+
infimal admin grants set <tenant> <model> --granted true|false
|
|
112
|
+
infimal admin tenants show <tenant>
|
|
113
|
+
infimal admin tenants set-discount <tenant> 2000 # basis points off the sale price
|
|
114
|
+
infimal admin topup <tenant> 25 --reference <ref>
|
|
115
|
+
infimal admin audit
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
**Every command takes `--json`**, and that is the output a script should hold on to: the human text
|
|
119
|
+
is for a human and is allowed to change. **Every mutation takes `--plan`**, a local, redacted preview
|
|
120
|
+
of the request it would send. Two habits the admin commands enforce — an upstream credential is read
|
|
121
|
+
from an environment variable named by `--key-env`, never from an argument that would sit in the
|
|
122
|
+
process table and the shell history; and `--reference` on a top-up is the idempotency key, so
|
|
123
|
+
re-running it credits once rather than twice.
|
|
124
|
+
|
|
125
|
+
## Two things this package is opinionated about
|
|
126
|
+
|
|
127
|
+
**You never name a GPU.** There is no `gpu=` parameter. You declare `perf=Latency(p95_ms=300)` or
|
|
128
|
+
`perf=Throughput(rps=50)`, or just a `tier`, and the platform picks the silicon and reports back
|
|
129
|
+
which one it used. Pricing follows the same rule: `infimal models list` quotes per token, never
|
|
130
|
+
per GPU-hour, and it is the same number you are billed — whether the tokens came off our own GPUs
|
|
131
|
+
or were bought from a provider.
|
|
132
|
+
|
|
133
|
+
(There used to be an `estimate` verb that quoted a monthly band from a declared traffic envelope. It
|
|
134
|
+
priced GPU-hours against a rate card nothing charged against, so the band could never be checked
|
|
135
|
+
against a bill, and it is gone rather than left returning a guess.)
|
|
136
|
+
|
|
137
|
+
**`@app.setup` is a snapshot boundary, not just an init hook.** After it runs, the process — VRAM
|
|
138
|
+
included — is checkpointed, and every later request starts from that image instead of an import. The
|
|
139
|
+
corollary is the one rule it imposes: nothing that cannot be checkpointed may survive `setup`. No
|
|
140
|
+
open sockets, no pipes. The SDK checks the obvious cases and tells you in a sentence, because the
|
|
141
|
+
alternative is an opaque CRIU failure minutes into a build.
|
|
142
|
+
|
|
143
|
+
## Development
|
|
144
|
+
|
|
145
|
+
```
|
|
146
|
+
uv run --extra dev pytest
|
|
147
|
+
uv run --extra dev ruff check .
|
|
148
|
+
```
|
|
149
|
+
|
|
150
|
+
The ring tests also check this implementation against the Rust gateway's, byte for byte, whenever
|
|
151
|
+
`cargo` is available: both map the same file, and a layout drift would silently corrupt every
|
|
152
|
+
request.
|
infimal-0.2.1/README.md
ADDED
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# infimal
|
|
2
|
+
|
|
3
|
+
The infimal Python SDK and in-container harness.
|
|
4
|
+
|
|
5
|
+
```python
|
|
6
|
+
from infimal import Client
|
|
7
|
+
|
|
8
|
+
client = Client.from_env() # INFIMAL_API_KEY; INFIMAL_ENDPOINT is optional
|
|
9
|
+
print(client.model("deepseek/deepseek-v4-flash").chat("Explain this diff.").max_tokens(256).run().text)
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
`client.model(id)` is a request handle: `.chat(...)`, `.embeddings(...)`, `.speech(...)`,
|
|
13
|
+
`.images(...)` and `.video(...)` build a request, `run()` or `stream()` sends it once, and
|
|
14
|
+
`.capabilities()` says what the listing accepts and whether it is serving. Images and video are
|
|
15
|
+
jobs: `submit()` returns a `Job` you can `wait()`, `refresh()` or `cancel()`, and `client.job(id)`
|
|
16
|
+
adopts one after a restart.
|
|
17
|
+
|
|
18
|
+
```python
|
|
19
|
+
from infimal import App, Latency, Scale
|
|
20
|
+
|
|
21
|
+
app = App("voice", perf=Latency(p95_ms=300), scale=Scale(max_replicas=4))
|
|
22
|
+
|
|
23
|
+
@app.setup
|
|
24
|
+
def load():
|
|
25
|
+
# Runs once, on a builder GPU. Weights, CUDA context, warm caches, CUDA graphs — everything
|
|
26
|
+
# expensive belongs here, because what happens next is that the whole live process is frozen.
|
|
27
|
+
return load_model()
|
|
28
|
+
|
|
29
|
+
@app.handler
|
|
30
|
+
def transcribe(model, request):
|
|
31
|
+
return model(request["audio"])
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## The environment
|
|
35
|
+
|
|
36
|
+
| Variable | Meaning |
|
|
37
|
+
|---|---|
|
|
38
|
+
| `INFIMAL_API_KEY` | the API key; required |
|
|
39
|
+
| `INFIMAL_ENDPOINT` | the API origin; defaults to `https://api.infimal.ai`; the host from before the rename remains supported |
|
|
40
|
+
|
|
41
|
+
`GLOW_API_KEY` and `GLOW_ENDPOINT`, the names before the rename, are still read **in 0.2 only**,
|
|
42
|
+
when the new ones are unset, and each read raises a one-line `DeprecationWarning` naming the
|
|
43
|
+
variable to set instead. 0.3 stops reading them. `GlowError` and `GlowUnreachable` resolve to
|
|
44
|
+
`InfimalError` and `InfimalUnreachable` for the same one release.
|
|
45
|
+
|
|
46
|
+
## The CLI
|
|
47
|
+
|
|
48
|
+
The customer CLI is the native `infimal` binary (`curl -fsSL https://infimal.ai/install.sh | sh`;
|
|
49
|
+
see `crates/infimal/README.md`), and it is the only one. This package installs **no console script
|
|
50
|
+
and has no verbs**; for the three commands that execute your module, the binary runs this package
|
|
51
|
+
(`python -m infimal._synth`, a private JSON-in, JSON-out subprocess described in `docs/sdk.md`), so
|
|
52
|
+
install the SDK where your module's dependencies are (`pip install infimal`), or point `INFIMAL_PYTHON`
|
|
53
|
+
at that interpreter.
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
infimal apps plan [module] # execute your module and show what deploying would change
|
|
57
|
+
infimal apps apply [module] # apply it: rebuild, or reconfigure a live placement in place
|
|
58
|
+
infimal apps deploy [module] # build a snapshot and place it
|
|
59
|
+
infimal apps snapshots <app> # published artifacts and how fast they restore
|
|
60
|
+
infimal apps status <app> # where the endpoint is on the residency gradient right now
|
|
61
|
+
infimal apps logs <app> # build logs
|
|
62
|
+
|
|
63
|
+
infimal models list # the price list: what each model costs you per million tokens
|
|
64
|
+
infimal billing balance # credit available, credit held against in-flight requests
|
|
65
|
+
infimal billing fund 25 --wait # buy credit; --wait blocks until the payment has landed
|
|
66
|
+
infimal usage show --since 30d [--by model]
|
|
67
|
+
infimal keys create|list|revoke
|
|
68
|
+
|
|
69
|
+
infimal jobs submit --surface images|video --model M --preset I1 --wait
|
|
70
|
+
infimal jobs get <id> # state, measured timings, presigned artifact URLs
|
|
71
|
+
infimal jobs list
|
|
72
|
+
infimal jobs cancel <id> # only before it starts; a running generation has been paid for
|
|
73
|
+
infimal jobs download <id> <dir>
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Images and video are **jobs**, not requests: a clip is minutes of GPU time, so the submit reserves
|
|
77
|
+
the credit, answers `202` with an id, and you poll. `--wait` is that loop; the exit code is 1 if the
|
|
78
|
+
job failed.
|
|
79
|
+
|
|
80
|
+
A key is minted `full` — the whole account — unless you ask for the narrow role:
|
|
81
|
+
|
|
82
|
+
```
|
|
83
|
+
infimal keys create client-app --role inference # the OpenAI surface, the model list, your usage
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
Administrators have a second set, which needs a tenant on the operator's admin list. There is no
|
|
87
|
+
administrative key: administration belongs to the account, so no key can grant it or carry it.
|
|
88
|
+
|
|
89
|
+
```
|
|
90
|
+
infimal admin models list
|
|
91
|
+
infimal admin models set-price <id> --input 0.15 --output 0.60 [--cache-read 0.015]
|
|
92
|
+
infimal admin models set-price <id> --flat 0.04 | --unpriced [--unpin]
|
|
93
|
+
infimal admin sources add --model <id> --name <name> --base-url <url> \
|
|
94
|
+
--upstream-model <id> --key-env PROVIDER_KEY \
|
|
95
|
+
--cost-input 0.10 --cost-output 0.30 [--wire-config @wire.json]
|
|
96
|
+
infimal admin sources list|rekey|delete
|
|
97
|
+
infimal admin grants set <tenant> <model> --granted true|false
|
|
98
|
+
infimal admin tenants show <tenant>
|
|
99
|
+
infimal admin tenants set-discount <tenant> 2000 # basis points off the sale price
|
|
100
|
+
infimal admin topup <tenant> 25 --reference <ref>
|
|
101
|
+
infimal admin audit
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
**Every command takes `--json`**, and that is the output a script should hold on to: the human text
|
|
105
|
+
is for a human and is allowed to change. **Every mutation takes `--plan`**, a local, redacted preview
|
|
106
|
+
of the request it would send. Two habits the admin commands enforce — an upstream credential is read
|
|
107
|
+
from an environment variable named by `--key-env`, never from an argument that would sit in the
|
|
108
|
+
process table and the shell history; and `--reference` on a top-up is the idempotency key, so
|
|
109
|
+
re-running it credits once rather than twice.
|
|
110
|
+
|
|
111
|
+
## Two things this package is opinionated about
|
|
112
|
+
|
|
113
|
+
**You never name a GPU.** There is no `gpu=` parameter. You declare `perf=Latency(p95_ms=300)` or
|
|
114
|
+
`perf=Throughput(rps=50)`, or just a `tier`, and the platform picks the silicon and reports back
|
|
115
|
+
which one it used. Pricing follows the same rule: `infimal models list` quotes per token, never
|
|
116
|
+
per GPU-hour, and it is the same number you are billed — whether the tokens came off our own GPUs
|
|
117
|
+
or were bought from a provider.
|
|
118
|
+
|
|
119
|
+
(There used to be an `estimate` verb that quoted a monthly band from a declared traffic envelope. It
|
|
120
|
+
priced GPU-hours against a rate card nothing charged against, so the band could never be checked
|
|
121
|
+
against a bill, and it is gone rather than left returning a guess.)
|
|
122
|
+
|
|
123
|
+
**`@app.setup` is a snapshot boundary, not just an init hook.** After it runs, the process — VRAM
|
|
124
|
+
included — is checkpointed, and every later request starts from that image instead of an import. The
|
|
125
|
+
corollary is the one rule it imposes: nothing that cannot be checkpointed may survive `setup`. No
|
|
126
|
+
open sockets, no pipes. The SDK checks the obvious cases and tells you in a sentence, because the
|
|
127
|
+
alternative is an opaque CRIU failure minutes into a build.
|
|
128
|
+
|
|
129
|
+
## Development
|
|
130
|
+
|
|
131
|
+
```
|
|
132
|
+
uv run --extra dev pytest
|
|
133
|
+
uv run --extra dev ruff check .
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
The ring tests also check this implementation against the Rust gateway's, byte for byte, whenever
|
|
137
|
+
`cargo` is available: both map the same file, and a layout drift would silently corrupt every
|
|
138
|
+
request.
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "infimal"
|
|
3
|
+
version = "0.2.1"
|
|
4
|
+
description = "The infimal SDK, the in-container harness, and the module behind the native CLI's apps plan, apply and deploy."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.11"
|
|
7
|
+
license = { text = "Apache-2.0" }
|
|
8
|
+
dependencies = [
|
|
9
|
+
"pydantic>=2.7",
|
|
10
|
+
"httpx>=0.27",
|
|
11
|
+
]
|
|
12
|
+
|
|
13
|
+
[project.optional-dependencies]
|
|
14
|
+
dev = [
|
|
15
|
+
"pytest>=8",
|
|
16
|
+
"pytest-asyncio>=0.23",
|
|
17
|
+
"ruff>=0.5",
|
|
18
|
+
]
|
|
19
|
+
|
|
20
|
+
[build-system]
|
|
21
|
+
requires = ["hatchling"]
|
|
22
|
+
build-backend = "hatchling.build"
|
|
23
|
+
|
|
24
|
+
[tool.hatch.build.targets.wheel]
|
|
25
|
+
packages = ["src/infimal"]
|
|
26
|
+
|
|
27
|
+
[tool.ruff]
|
|
28
|
+
line-length = 100
|
|
29
|
+
target-version = "py311"
|
|
30
|
+
|
|
31
|
+
[tool.ruff.lint]
|
|
32
|
+
select = ["E", "F", "I", "UP", "B", "SIM"]
|
|
33
|
+
ignore = ["E501"]
|
|
34
|
+
|
|
35
|
+
[tool.pytest.ini_options]
|
|
36
|
+
testpaths = ["tests"]
|
|
37
|
+
asyncio_mode = "auto"
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
"""infimal: snapshot-native serverless GPU endpoints.
|
|
2
|
+
|
|
3
|
+
```python
|
|
4
|
+
from infimal import App, Latency, Scale
|
|
5
|
+
|
|
6
|
+
app = App("voice", perf=Latency(p95_ms=300), scale=Scale(max_replicas=4))
|
|
7
|
+
|
|
8
|
+
@app.setup
|
|
9
|
+
def load():
|
|
10
|
+
return load_model() # runs once, on a builder GPU, then frozen into a snapshot
|
|
11
|
+
|
|
12
|
+
@app.handler
|
|
13
|
+
def transcribe(model, request):
|
|
14
|
+
return model(request["audio"])
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
`infimal apps deploy` gives you an HTTPS endpoint that scales to zero and comes back from that
|
|
18
|
+
snapshot instead of a cold load. You never name a GPU: declare a latency or throughput target, or a
|
|
19
|
+
tier, and the platform picks the silicon.
|
|
20
|
+
|
|
21
|
+
Calling a model that is already served is the client:
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
from infimal import Client
|
|
25
|
+
|
|
26
|
+
client = Client.from_env()
|
|
27
|
+
print(client.model("qwen3-8b").chat("Explain this code.").max_tokens(128).run().text)
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The client names (`Client`, `InfimalError`, `Job`, `Request`, `ChatResult` and the rest) are
|
|
31
|
+
imported on first use, not by `import infimal`: an App module is imported inside the serving
|
|
32
|
+
container by the harness, and that import must not pull in the HTTP stack.
|
|
33
|
+
"""
|
|
34
|
+
|
|
35
|
+
from typing import TYPE_CHECKING, Any
|
|
36
|
+
|
|
37
|
+
from .app import App, SnapshotBoundaryError, discover
|
|
38
|
+
from .gateway import Credential, Flat, Gateway, Model, PerToken, Provider, Source, Wire
|
|
39
|
+
from .models import (
|
|
40
|
+
Batch,
|
|
41
|
+
Deployment,
|
|
42
|
+
Endpoint,
|
|
43
|
+
Engine,
|
|
44
|
+
Latency,
|
|
45
|
+
Reservation,
|
|
46
|
+
Scale,
|
|
47
|
+
Throughput,
|
|
48
|
+
Tier,
|
|
49
|
+
TrafficEnvelope,
|
|
50
|
+
)
|
|
51
|
+
from .ring import Frame, FrameKind, Ring, RingError, RingFull
|
|
52
|
+
from .runtime import (
|
|
53
|
+
is_build,
|
|
54
|
+
production_vram_bytes,
|
|
55
|
+
snapshot_vram_bytes,
|
|
56
|
+
snapshot_vram_fraction,
|
|
57
|
+
target_vram_bytes,
|
|
58
|
+
)
|
|
59
|
+
|
|
60
|
+
if TYPE_CHECKING:
|
|
61
|
+
from .client import Client, InfimalError, InfimalUnreachable
|
|
62
|
+
from .request import (
|
|
63
|
+
Artifact,
|
|
64
|
+
Capabilities,
|
|
65
|
+
ChatResult,
|
|
66
|
+
Job,
|
|
67
|
+
JobResult,
|
|
68
|
+
LoraSpec,
|
|
69
|
+
ModelHandle,
|
|
70
|
+
Request,
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
__version__ = "0.2.1"
|
|
74
|
+
|
|
75
|
+
#: The client surface, by the module that defines it. Loaded lazily (PEP 562) so that
|
|
76
|
+
#: `import infimal` inside a container never imports httpx. `GlowError` and `GlowUnreachable` are
|
|
77
|
+
#: the pre-rename names of the two errors, resolvable in 0.2 only and left out of `__all__`.
|
|
78
|
+
_LAZY = {
|
|
79
|
+
"Client": "client",
|
|
80
|
+
"InfimalError": "client",
|
|
81
|
+
"InfimalUnreachable": "client",
|
|
82
|
+
"GlowError": "client",
|
|
83
|
+
"GlowUnreachable": "client",
|
|
84
|
+
"ModelHandle": "request",
|
|
85
|
+
"Request": "request",
|
|
86
|
+
"ChatResult": "request",
|
|
87
|
+
"JobResult": "request",
|
|
88
|
+
"Job": "request",
|
|
89
|
+
"Artifact": "request",
|
|
90
|
+
"Capabilities": "request",
|
|
91
|
+
"LoraSpec": "request",
|
|
92
|
+
}
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def __getattr__(name: str) -> Any:
|
|
96
|
+
module = _LAZY.get(name)
|
|
97
|
+
if module is None:
|
|
98
|
+
raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
|
|
99
|
+
from importlib import import_module
|
|
100
|
+
|
|
101
|
+
value = getattr(import_module(f".{module}", __name__), name)
|
|
102
|
+
globals()[name] = value
|
|
103
|
+
return value
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def __dir__() -> list[str]:
|
|
107
|
+
return sorted(set(globals()) | set(_LAZY))
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
__all__ = [
|
|
111
|
+
"App",
|
|
112
|
+
"Artifact",
|
|
113
|
+
"Batch",
|
|
114
|
+
"Capabilities",
|
|
115
|
+
"ChatResult",
|
|
116
|
+
"Client",
|
|
117
|
+
"Credential",
|
|
118
|
+
"Deployment",
|
|
119
|
+
"Endpoint",
|
|
120
|
+
"Engine",
|
|
121
|
+
"Flat",
|
|
122
|
+
"Frame",
|
|
123
|
+
"FrameKind",
|
|
124
|
+
"Gateway",
|
|
125
|
+
"InfimalError",
|
|
126
|
+
"InfimalUnreachable",
|
|
127
|
+
"Job",
|
|
128
|
+
"JobResult",
|
|
129
|
+
"Latency",
|
|
130
|
+
"LoraSpec",
|
|
131
|
+
"Model",
|
|
132
|
+
"ModelHandle",
|
|
133
|
+
"PerToken",
|
|
134
|
+
"Provider",
|
|
135
|
+
"Request",
|
|
136
|
+
"Reservation",
|
|
137
|
+
"Ring",
|
|
138
|
+
"RingError",
|
|
139
|
+
"RingFull",
|
|
140
|
+
"Scale",
|
|
141
|
+
"SnapshotBoundaryError",
|
|
142
|
+
"Source",
|
|
143
|
+
"Throughput",
|
|
144
|
+
"Tier",
|
|
145
|
+
"TrafficEnvelope",
|
|
146
|
+
"Wire",
|
|
147
|
+
"__version__",
|
|
148
|
+
"discover",
|
|
149
|
+
"is_build",
|
|
150
|
+
"production_vram_bytes",
|
|
151
|
+
"snapshot_vram_bytes",
|
|
152
|
+
"snapshot_vram_fraction",
|
|
153
|
+
"target_vram_bytes",
|
|
154
|
+
]
|