hugpy-wrapper 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. hugpy_wrapper-0.1.0/PKG-INFO +101 -0
  2. hugpy_wrapper-0.1.0/README.md +67 -0
  3. hugpy_wrapper-0.1.0/fitevict/__init__.py +34 -0
  4. hugpy_wrapper-0.1.0/fitevict/adapters/__init__.py +6 -0
  5. hugpy_wrapper-0.1.0/fitevict/adapters/db.py +487 -0
  6. hugpy_wrapper-0.1.0/fitevict/adapters/llama_cpp.py +1298 -0
  7. hugpy_wrapper-0.1.0/fitevict/adapters/transformers.py +149 -0
  8. hugpy_wrapper-0.1.0/fitevict/calls_view.py +225 -0
  9. hugpy_wrapper-0.1.0/fitevict/evict.py +167 -0
  10. hugpy_wrapper-0.1.0/fitevict/flex.py +174 -0
  11. hugpy_wrapper-0.1.0/fitevict/front_door.py +782 -0
  12. hugpy_wrapper-0.1.0/fitevict/gpu_state.py +173 -0
  13. hugpy_wrapper-0.1.0/fitevict/host.py +84 -0
  14. hugpy_wrapper-0.1.0/fitevict/lifted/__init__.py +8 -0
  15. hugpy_wrapper-0.1.0/fitevict/lifted/app_dirs.py +177 -0
  16. hugpy_wrapper-0.1.0/fitevict/lifted/binaries.py +47 -0
  17. hugpy_wrapper-0.1.0/fitevict/lifted/constants.py +91 -0
  18. hugpy_wrapper-0.1.0/fitevict/lifted/gguf_election.py +263 -0
  19. hugpy_wrapper-0.1.0/fitevict/lifted/gguf_inspect.py +985 -0
  20. hugpy_wrapper-0.1.0/fitevict/lifted/gguf_need.py +142 -0
  21. hugpy_wrapper-0.1.0/fitevict/lifted/hardware.py +204 -0
  22. hugpy_wrapper-0.1.0/fitevict/lifted/model_resolve.py +227 -0
  23. hugpy_wrapper-0.1.0/fitevict/lifted/native_resolve.py +115 -0
  24. hugpy_wrapper-0.1.0/fitevict/lifted/no_think.py +30 -0
  25. hugpy_wrapper-0.1.0/fitevict/lifted/nvml.py +128 -0
  26. hugpy_wrapper-0.1.0/fitevict/lifted/platform_facade.py +64 -0
  27. hugpy_wrapper-0.1.0/fitevict/lifted/procutil.py +78 -0
  28. hugpy_wrapper-0.1.0/fitevict/lifted/spill_reserve.py +227 -0
  29. hugpy_wrapper-0.1.0/fitevict/lifted/tf_child.py +556 -0
  30. hugpy_wrapper-0.1.0/fitevict/lifted/tf_facts.py +144 -0
  31. hugpy_wrapper-0.1.0/fitevict/lifted/timings.py +148 -0
  32. hugpy_wrapper-0.1.0/fitevict/plan.py +312 -0
  33. hugpy_wrapper-0.1.0/fitevict/predict.py +107 -0
  34. hugpy_wrapper-0.1.0/fitevict/run_engine.py +223 -0
  35. hugpy_wrapper-0.1.0/fitevict/sizing.py +585 -0
  36. hugpy_wrapper-0.1.0/fitevict/types.py +345 -0
  37. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/PKG-INFO +101 -0
  38. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/SOURCES.txt +45 -0
  39. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/dependency_links.txt +1 -0
  40. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/entry_points.txt +3 -0
  41. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/requires.txt +20 -0
  42. hugpy_wrapper-0.1.0/hugpy_wrapper.egg-info/top_level.txt +1 -0
  43. hugpy_wrapper-0.1.0/pyproject.toml +52 -0
  44. hugpy_wrapper-0.1.0/setup.cfg +4 -0
  45. hugpy_wrapper-0.1.0/tests/test_evict.py +88 -0
  46. hugpy_wrapper-0.1.0/tests/test_from_call_log.py +242 -0
  47. hugpy_wrapper-0.1.0/tests/test_plan.py +192 -0
@@ -0,0 +1,101 @@
1
+ Metadata-Version: 2.4
2
+ Name: hugpy-wrapper
3
+ Version: 0.1.0
4
+ Summary: The hugpy wrapper (import name: fitevict): a per-box evict-to-fit front door. An OpenAI /v1 server that resolves a called model to its weights (GGUF via llama.cpp, or transformers), places it on the GPU by evicting what doesn't fit, serves it, and logs every call.
5
+ Author-email: putkoff <support@hugpy.ai>
6
+ License: Proprietary
7
+ Project-URL: Homepage, https://hugpy.ai
8
+ Keywords: llama.cpp,inference,evict-to-fit,vram,openai,wrapper,allocator
9
+ Classifier: Development Status :: 3 - Alpha
10
+ Classifier: Intended Audience :: Developers
11
+ Classifier: Operating System :: POSIX :: Linux
12
+ Classifier: Programming Language :: Python :: 3
13
+ Classifier: Programming Language :: Python :: 3 :: Only
14
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
15
+ Requires-Python: >=3.9
16
+ Description-Content-Type: text/markdown
17
+ Provides-Extra: test
18
+ Requires-Dist: pytest; extra == "test"
19
+ Provides-Extra: db
20
+ Requires-Dist: psycopg[binary]>=3; extra == "db"
21
+ Provides-Extra: transformers
22
+ Requires-Dist: torch==2.14.0; extra == "transformers"
23
+ Requires-Dist: transformers==5.17.0; extra == "transformers"
24
+ Requires-Dist: accelerate==1.15.0; extra == "transformers"
25
+ Requires-Dist: bitsandbytes==0.50.2; extra == "transformers"
26
+ Requires-Dist: peft==0.20.0; extra == "transformers"
27
+ Requires-Dist: compressed-tensors>=0.15.0; extra == "transformers"
28
+ Requires-Dist: safetensors==0.8.0; extra == "transformers"
29
+ Requires-Dist: tokenizers==0.23.2; extra == "transformers"
30
+ Requires-Dist: sentencepiece==0.2.2; extra == "transformers"
31
+ Requires-Dist: tiktoken==0.14.0; extra == "transformers"
32
+ Requires-Dist: protobuf==7.36.1; extra == "transformers"
33
+ Requires-Dist: huggingface_hub==1.31.0; extra == "transformers"
34
+
35
+ # hugpy-wrapper (`fitevict`)
36
+
37
+ pip install hugpy-wrapper # import fitevict; fitevict-serve
38
+ pip install "hugpy-wrapper[transformers]" # the transformers engine (own venv)
39
+
40
+ hugpy's per-box wrapper: an **evict-to-fit front door over llama.cpp** (and transformers). Point any OpenAI
41
+ client's base URL at it; it turns one address into a self-managing model server.
42
+
43
+ A `/v1` call carries only a model **name** — never a location. llama has no idea
44
+ where a gguf lives. `fitevict` is the layer that does what the call can't: for
45
+ each request it resolves the name → gguf path, runs an **evict-to-fit** plan
46
+ against what's resident and what's being called, places the model into llama
47
+ (launch/evict), forwards the call, splits the model's `<think>` out of the
48
+ answer, and records the call — raw stamps + engine geometry — to a call log that
49
+ every metric derives from.
50
+
51
+ The package depends on nothing else in hugpy. The decision core is pure stdlib; hardware
52
+ facts come from `nvidia-smi`; model/gguf facts from the files on disk.
53
+
54
+ ## Layout
55
+
56
+ ```
57
+ fitevict/
58
+ types.py frozen data contract (DeviceBudget, Resident, LoadRequest, FitPlan, ...)
59
+ evict.py plan_eviction / sort_key — the victim selector (pure)
60
+ plan.py plan_fit — staged decision + quant ladder (pure)
61
+ flex.py ctx-band compress + layers-that-fit offload (pure)
62
+ host.py EngineHost protocol + drive() loop (measure→plan→evict→load)
63
+ front_door.py the WRAPPER: persistent OpenAI /v1 server (owns the address)
64
+ adapters/
65
+ llama_cpp.py concrete EngineHost: measure GPU/RAM, price GGUFs, launch/evict llama-server
66
+ db.py call_log writer (Postgres inference_engine.call_log)
67
+ lifted/ measure/act helpers lifted clean from hugpy (imports stripped):
68
+ gguf_inspect, gguf_need, hardware, spill_reserve, model_resolve,
69
+ native_resolve, supervisor/procutil, timings, no_think, app_dirs, ...
70
+ run_engine.py a juggling harness: discover models, fire a sequence that exceeds the card
71
+ ```
72
+
73
+ ## Install
74
+
75
+ ```
76
+ pip install . # core + wrapper (stdlib only)
77
+ pip install .[db] # + psycopg for direct call_log DSN writes (else shells to psql)
78
+ pip install .[test] # + pytest
79
+ ```
80
+
81
+ ## Run
82
+
83
+ ```
84
+ fitevict-serve --host 127.0.0.1 --port 8080 # the front door (owns /v1)
85
+ fitevict-run # the juggle harness on this box
86
+ ```
87
+
88
+ Then point any OpenAI client at `http://127.0.0.1:8080/v1`:
89
+
90
+ ```
91
+ curl -s http://127.0.0.1:8080/v1/chat/completions -H 'Content-Type: application/json' \
92
+ -d '{"model":"<name>","messages":[{"role":"user","content":"hi"}],"max_tokens":128}'
93
+ ```
94
+
95
+ The model is named, not located; the wrapper finds it, fits it, serves it, logs it.
96
+
97
+ ## Test
98
+
99
+ ```
100
+ pytest # pure unit tests + a call_log replay + the VL-flood fixture
101
+ ```
@@ -0,0 +1,67 @@
1
+ # hugpy-wrapper (`fitevict`)
2
+
3
+ pip install hugpy-wrapper # import fitevict; fitevict-serve
4
+ pip install "hugpy-wrapper[transformers]" # the transformers engine (own venv)
5
+
6
+ hugpy's per-box wrapper: an **evict-to-fit front door over llama.cpp** (and transformers). Point any OpenAI
7
+ client's base URL at it; it turns one address into a self-managing model server.
8
+
9
+ A `/v1` call carries only a model **name** — never a location. llama has no idea
10
+ where a gguf lives. `fitevict` is the layer that does what the call can't: for
11
+ each request it resolves the name → gguf path, runs an **evict-to-fit** plan
12
+ against what's resident and what's being called, places the model into llama
13
+ (launch/evict), forwards the call, splits the model's `<think>` out of the
14
+ answer, and records the call — raw stamps + engine geometry — to a call log that
15
+ every metric derives from.
16
+
17
+ The package depends on nothing else in hugpy. The decision core is pure stdlib; hardware
18
+ facts come from `nvidia-smi`; model/gguf facts from the files on disk.
19
+
20
+ ## Layout
21
+
22
+ ```
23
+ fitevict/
24
+ types.py frozen data contract (DeviceBudget, Resident, LoadRequest, FitPlan, ...)
25
+ evict.py plan_eviction / sort_key — the victim selector (pure)
26
+ plan.py plan_fit — staged decision + quant ladder (pure)
27
+ flex.py ctx-band compress + layers-that-fit offload (pure)
28
+ host.py EngineHost protocol + drive() loop (measure→plan→evict→load)
29
+ front_door.py the WRAPPER: persistent OpenAI /v1 server (owns the address)
30
+ adapters/
31
+ llama_cpp.py concrete EngineHost: measure GPU/RAM, price GGUFs, launch/evict llama-server
32
+ db.py call_log writer (Postgres inference_engine.call_log)
33
+ lifted/ measure/act helpers lifted clean from hugpy (imports stripped):
34
+ gguf_inspect, gguf_need, hardware, spill_reserve, model_resolve,
35
+ native_resolve, supervisor/procutil, timings, no_think, app_dirs, ...
36
+ run_engine.py a juggling harness: discover models, fire a sequence that exceeds the card
37
+ ```
38
+
39
+ ## Install
40
+
41
+ ```
42
+ pip install . # core + wrapper (stdlib only)
43
+ pip install .[db] # + psycopg for direct call_log DSN writes (else shells to psql)
44
+ pip install .[test] # + pytest
45
+ ```
46
+
47
+ ## Run
48
+
49
+ ```
50
+ fitevict-serve --host 127.0.0.1 --port 8080 # the front door (owns /v1)
51
+ fitevict-run # the juggle harness on this box
52
+ ```
53
+
54
+ Then point any OpenAI client at `http://127.0.0.1:8080/v1`:
55
+
56
+ ```
57
+ curl -s http://127.0.0.1:8080/v1/chat/completions -H 'Content-Type: application/json' \
58
+ -d '{"model":"<name>","messages":[{"role":"user","content":"hi"}],"max_tokens":128}'
59
+ ```
60
+
61
+ The model is named, not located; the wrapper finds it, fits it, serves it, logs it.
62
+
63
+ ## Test
64
+
65
+ ```
66
+ pytest # pure unit tests + a call_log replay + the VL-flood fixture
67
+ ```
@@ -0,0 +1,34 @@
1
+ """fitevict — a pure, deterministic evict-to-fit / placement allocator.
2
+
3
+ Clean-room: NO hugpy import. The core (`types`, `evict`, `plan`) is stdlib-only
4
+ and reads nothing live — every fact arrives in a frozen record captured once by
5
+ the caller, so identical records produce byte-identical plans. The `host` seam
6
+ (`EngineHost` / `drive`) and the `adapters` package hold the impure edges that
7
+ wrap a real inference engine (llama.cpp first).
8
+ """
9
+ from __future__ import annotations
10
+
11
+ from .evict import evict_order, plan_eviction, sort_key
12
+ from .host import DriveResult, EngineHost, drive
13
+ from .plan import plan_fit
14
+ from .types import (
15
+ ACTIONS, FAILURE_KINDS, RAM, VRAM, DeviceBudget, Eviction, EvictionPlan,
16
+ FitFailure, FitPlan, HostBudget, LoadRequest, ModelFootprint, Offload,
17
+ Policy, Resident, canonical_key, key_equivalent,
18
+ )
19
+
20
+ __version__ = "0.1.0"
21
+
22
+ __all__ = [
23
+ "__version__",
24
+ # types
25
+ "VRAM", "RAM", "ACTIONS", "FAILURE_KINDS", "canonical_key", "key_equivalent",
26
+ "DeviceBudget", "HostBudget", "ModelFootprint", "Resident", "LoadRequest",
27
+ "Policy", "Eviction", "EvictionPlan", "FitFailure", "Offload", "FitPlan",
28
+ # evict
29
+ "sort_key", "plan_eviction", "evict_order",
30
+ # plan
31
+ "plan_fit",
32
+ # host
33
+ "EngineHost", "DriveResult", "drive",
34
+ ]
@@ -0,0 +1,6 @@
1
+ """Engine adapters — the impure geometry/pricing edges. llama.cpp first."""
2
+ from __future__ import annotations
3
+
4
+ from .llama_cpp import gguf_footprint
5
+
6
+ __all__ = ["gguf_footprint"]