hugpy-wrapper 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hugpy_wrapper-0.1.0/PKG-INFO +101 -0
- hugpy_wrapper-0.1.0/README.md +67 -0
- hugpy_wrapper-0.1.0/fitevict/__init__.py +34 -0
- hugpy_wrapper-0.1.0/fitevict/adapters/__init__.py +6 -0
- hugpy_wrapper-0.1.0/fitevict/adapters/db.py +487 -0
- hugpy_wrapper-0.1.0/fitevict/adapters/llama_cpp.py +1298 -0
- hugpy_wrapper-0.1.0/fitevict/adapters/transformers.py +149 -0
- hugpy_wrapper-0.1.0/fitevict/calls_view.py +225 -0
- hugpy_wrapper-0.1.0/fitevict/evict.py +167 -0
- hugpy_wrapper-0.1.0/fitevict/flex.py +174 -0
- hugpy_wrapper-0.1.0/fitevict/front_door.py +782 -0
- hugpy_wrapper-0.1.0/fitevict/gpu_state.py +173 -0
- hugpy_wrapper-0.1.0/fitevict/host.py +84 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/__init__.py +8 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/app_dirs.py +177 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/binaries.py +47 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/constants.py +91 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/gguf_election.py +263 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/gguf_inspect.py +985 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/gguf_need.py +142 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/hardware.py +204 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/model_resolve.py +227 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/native_resolve.py +115 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/no_think.py +30 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/nvml.py +128 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/platform_facade.py +64 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/procutil.py +78 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/spill_reserve.py +227 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/tf_child.py +556 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/tf_facts.py +144 -0
- hugpy_wrapper-0.1.0/fitevict/lifted/timings.py +148 -0
- hugpy_wrapper-0.1.0/fitevict/plan.py +312 -0
- hugpy_wrapper-0.1.0/fitevict/predict.py +107 -0
- hugpy_wrapper-0.1.0/fitevict/run_engine.py +223 -0
- hugpy_wrapper-0.1.0/fitevict/sizing.py +585 -0
- hugpy_wrapper-0.1.0/fitevict/types.py +345 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/PKG-INFO +101 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/SOURCES.txt +45 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/dependency_links.txt +1 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/entry_points.txt +3 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/requires.txt +20 -0
- hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/top_level.txt +1 -0
- hugpy_wrapper-0.1.0/pyproject.toml +52 -0
- hugpy_wrapper-0.1.0/setup.cfg +4 -0
- hugpy_wrapper-0.1.0/tests/test_evict.py +88 -0
- hugpy_wrapper-0.1.0/tests/test_from_call_log.py +242 -0
- hugpy_wrapper-0.1.0/tests/test_plan.py +192 -0
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hugpy-wrapper
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: The hugpy wrapper (import name: fitevict): a per-box evict-to-fit front door. An OpenAI /v1 server that resolves a called model to its weights (GGUF via llama.cpp, or transformers), places it on the GPU by evicting what doesn't fit, serves it, and logs every call.
|
|
5
|
+
Author-email: putkoff <support@hugpy.ai>
|
|
6
|
+
License: Proprietary
|
|
7
|
+
Project-URL: Homepage, https://hugpy.ai
|
|
8
|
+
Keywords: llama.cpp,inference,evict-to-fit,vram,openai,wrapper,allocator
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Operating System :: POSIX :: Linux
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
14
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
Provides-Extra: test
|
|
18
|
+
Requires-Dist: pytest; extra == "test"
|
|
19
|
+
Provides-Extra: db
|
|
20
|
+
Requires-Dist: psycopg[binary]>=3; extra == "db"
|
|
21
|
+
Provides-Extra: transformers
|
|
22
|
+
Requires-Dist: torch==2.14.0; extra == "transformers"
|
|
23
|
+
Requires-Dist: transformers==5.17.0; extra == "transformers"
|
|
24
|
+
Requires-Dist: accelerate==1.15.0; extra == "transformers"
|
|
25
|
+
Requires-Dist: bitsandbytes==0.50.2; extra == "transformers"
|
|
26
|
+
Requires-Dist: peft==0.20.0; extra == "transformers"
|
|
27
|
+
Requires-Dist: compressed-tensors>=0.15.0; extra == "transformers"
|
|
28
|
+
Requires-Dist: safetensors==0.8.0; extra == "transformers"
|
|
29
|
+
Requires-Dist: tokenizers==0.23.2; extra == "transformers"
|
|
30
|
+
Requires-Dist: sentencepiece==0.2.2; extra == "transformers"
|
|
31
|
+
Requires-Dist: tiktoken==0.14.0; extra == "transformers"
|
|
32
|
+
Requires-Dist: protobuf==7.36.1; extra == "transformers"
|
|
33
|
+
Requires-Dist: huggingface_hub==1.31.0; extra == "transformers"
|
|
34
|
+
|
|
35
|
+
# hugpy-wrapper (`fitevict`)
|
|
36
|
+
|
|
37
|
+
pip install hugpy-wrapper # import fitevict; fitevict-serve
|
|
38
|
+
pip install "hugpy-wrapper[transformers]" # the transformers engine (own venv)
|
|
39
|
+
|
|
40
|
+
hugpy's per-box wrapper: an **evict-to-fit front door over llama.cpp** (and transformers). Point any OpenAI
|
|
41
|
+
client's base URL at it; it turns one address into a self-managing model server.
|
|
42
|
+
|
|
43
|
+
A `/v1` call carries only a model **name** — never a location. llama has no idea
|
|
44
|
+
where a gguf lives. `fitevict` is the layer that does what the call can't: for
|
|
45
|
+
each request it resolves the name → gguf path, runs an **evict-to-fit** plan
|
|
46
|
+
against what's resident and what's being called, places the model into llama
|
|
47
|
+
(launch/evict), forwards the call, splits the model's `<think>` out of the
|
|
48
|
+
answer, and records the call — raw stamps + engine geometry — to a call log that
|
|
49
|
+
every metric derives from.
|
|
50
|
+
|
|
51
|
+
The package depends on nothing else in hugpy. The decision core is pure stdlib; hardware
|
|
52
|
+
facts come from `nvidia-smi`; model/gguf facts from the files on disk.
|
|
53
|
+
|
|
54
|
+
## Layout
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
fitevict/
|
|
58
|
+
types.py frozen data contract (DeviceBudget, Resident, LoadRequest, FitPlan, ...)
|
|
59
|
+
evict.py plan_eviction / sort_key — the victim selector (pure)
|
|
60
|
+
plan.py plan_fit — staged decision + quant ladder (pure)
|
|
61
|
+
flex.py ctx-band compress + layers-that-fit offload (pure)
|
|
62
|
+
host.py EngineHost protocol + drive() loop (measure→plan→evict→load)
|
|
63
|
+
front_door.py the WRAPPER: persistent OpenAI /v1 server (owns the address)
|
|
64
|
+
adapters/
|
|
65
|
+
llama_cpp.py concrete EngineHost: measure GPU/RAM, price GGUFs, launch/evict llama-server
|
|
66
|
+
db.py call_log writer (Postgres inference_engine.call_log)
|
|
67
|
+
lifted/ measure/act helpers lifted clean from hugpy (imports stripped):
|
|
68
|
+
gguf_inspect, gguf_need, hardware, spill_reserve, model_resolve,
|
|
69
|
+
native_resolve, supervisor/procutil, timings, no_think, app_dirs, ...
|
|
70
|
+
run_engine.py a juggling harness: discover models, fire a sequence that exceeds the card
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
## Install
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
pip install . # core + wrapper (stdlib only)
|
|
77
|
+
pip install .[db] # + psycopg for direct call_log DSN writes (else shells to psql)
|
|
78
|
+
pip install .[test] # + pytest
|
|
79
|
+
```
|
|
80
|
+
|
|
81
|
+
## Run
|
|
82
|
+
|
|
83
|
+
```
|
|
84
|
+
fitevict-serve --host 127.0.0.1 --port 8080 # the front door (owns /v1)
|
|
85
|
+
fitevict-run # the juggle harness on this box
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Then point any OpenAI client at `http://127.0.0.1:8080/v1`:
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
curl -s http://127.0.0.1:8080/v1/chat/completions -H 'Content-Type: application/json' \
|
|
92
|
+
-d '{"model":"<name>","messages":[{"role":"user","content":"hi"}],"max_tokens":128}'
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
The model is named, not located; the wrapper finds it, fits it, serves it, logs it.
|
|
96
|
+
|
|
97
|
+
## Test
|
|
98
|
+
|
|
99
|
+
```
|
|
100
|
+
pytest # pure unit tests + a call_log replay + the VL-flood fixture
|
|
101
|
+
```
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
# hugpy-wrapper (`fitevict`)
|
|
2
|
+
|
|
3
|
+
pip install hugpy-wrapper # import fitevict; fitevict-serve
|
|
4
|
+
pip install "hugpy-wrapper[transformers]" # the transformers engine (own venv)
|
|
5
|
+
|
|
6
|
+
hugpy's per-box wrapper: an **evict-to-fit front door over llama.cpp** (and transformers). Point any OpenAI
|
|
7
|
+
client's base URL at it; it turns one address into a self-managing model server.
|
|
8
|
+
|
|
9
|
+
A `/v1` call carries only a model **name** — never a location. llama has no idea
|
|
10
|
+
where a gguf lives. `fitevict` is the layer that does what the call can't: for
|
|
11
|
+
each request it resolves the name → gguf path, runs an **evict-to-fit** plan
|
|
12
|
+
against what's resident and what's being called, places the model into llama
|
|
13
|
+
(launch/evict), forwards the call, splits the model's `<think>` out of the
|
|
14
|
+
answer, and records the call — raw stamps + engine geometry — to a call log that
|
|
15
|
+
every metric derives from.
|
|
16
|
+
|
|
17
|
+
The package depends on nothing else in hugpy. The decision core is pure stdlib; hardware
|
|
18
|
+
facts come from `nvidia-smi`; model/gguf facts from the files on disk.
|
|
19
|
+
|
|
20
|
+
## Layout
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
fitevict/
|
|
24
|
+
types.py frozen data contract (DeviceBudget, Resident, LoadRequest, FitPlan, ...)
|
|
25
|
+
evict.py plan_eviction / sort_key — the victim selector (pure)
|
|
26
|
+
plan.py plan_fit — staged decision + quant ladder (pure)
|
|
27
|
+
flex.py ctx-band compress + layers-that-fit offload (pure)
|
|
28
|
+
host.py EngineHost protocol + drive() loop (measure→plan→evict→load)
|
|
29
|
+
front_door.py the WRAPPER: persistent OpenAI /v1 server (owns the address)
|
|
30
|
+
adapters/
|
|
31
|
+
llama_cpp.py concrete EngineHost: measure GPU/RAM, price GGUFs, launch/evict llama-server
|
|
32
|
+
db.py call_log writer (Postgres inference_engine.call_log)
|
|
33
|
+
lifted/ measure/act helpers lifted clean from hugpy (imports stripped):
|
|
34
|
+
gguf_inspect, gguf_need, hardware, spill_reserve, model_resolve,
|
|
35
|
+
native_resolve, supervisor/procutil, timings, no_think, app_dirs, ...
|
|
36
|
+
run_engine.py a juggling harness: discover models, fire a sequence that exceeds the card
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
## Install
|
|
40
|
+
|
|
41
|
+
```
|
|
42
|
+
pip install . # core + wrapper (stdlib only)
|
|
43
|
+
pip install .[db] # + psycopg for direct call_log DSN writes (else shells to psql)
|
|
44
|
+
pip install .[test] # + pytest
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Run
|
|
48
|
+
|
|
49
|
+
```
|
|
50
|
+
fitevict-serve --host 127.0.0.1 --port 8080 # the front door (owns /v1)
|
|
51
|
+
fitevict-run # the juggle harness on this box
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Then point any OpenAI client at `http://127.0.0.1:8080/v1`:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
curl -s http://127.0.0.1:8080/v1/chat/completions -H 'Content-Type: application/json' \
|
|
58
|
+
-d '{"model":"<name>","messages":[{"role":"user","content":"hi"}],"max_tokens":128}'
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
The model is named, not located; the wrapper finds it, fits it, serves it, logs it.
|
|
62
|
+
|
|
63
|
+
## Test
|
|
64
|
+
|
|
65
|
+
```
|
|
66
|
+
pytest # pure unit tests + a call_log replay + the VL-flood fixture
|
|
67
|
+
```
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""fitevict — a pure, deterministic evict-to-fit / placement allocator.
|
|
2
|
+
|
|
3
|
+
Clean-room: NO hugpy import. The core (`types`, `evict`, `plan`) is stdlib-only
|
|
4
|
+
and reads nothing live — every fact arrives in a frozen record captured once by
|
|
5
|
+
the caller, so identical records produce byte-identical plans. The `host` seam
|
|
6
|
+
(`EngineHost` / `drive`) and the `adapters` package hold the impure edges that
|
|
7
|
+
wrap a real inference engine (llama.cpp first).
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from .evict import evict_order, plan_eviction, sort_key
|
|
12
|
+
from .host import DriveResult, EngineHost, drive
|
|
13
|
+
from .plan import plan_fit
|
|
14
|
+
from .types import (
|
|
15
|
+
ACTIONS, FAILURE_KINDS, RAM, VRAM, DeviceBudget, Eviction, EvictionPlan,
|
|
16
|
+
FitFailure, FitPlan, HostBudget, LoadRequest, ModelFootprint, Offload,
|
|
17
|
+
Policy, Resident, canonical_key, key_equivalent,
|
|
18
|
+
)
|
|
19
|
+
|
|
20
|
+
__version__ = "0.1.0"
|
|
21
|
+
|
|
22
|
+
__all__ = [
|
|
23
|
+
"__version__",
|
|
24
|
+
# types
|
|
25
|
+
"VRAM", "RAM", "ACTIONS", "FAILURE_KINDS", "canonical_key", "key_equivalent",
|
|
26
|
+
"DeviceBudget", "HostBudget", "ModelFootprint", "Resident", "LoadRequest",
|
|
27
|
+
"Policy", "Eviction", "EvictionPlan", "FitFailure", "Offload", "FitPlan",
|
|
28
|
+
# evict
|
|
29
|
+
"sort_key", "plan_eviction", "evict_order",
|
|
30
|
+
# plan
|
|
31
|
+
"plan_fit",
|
|
32
|
+
# host
|
|
33
|
+
"EngineHost", "DriveResult", "drive",
|
|
34
|
+
]
|