offlinedemo 0.1.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,106 @@
1
+ Metadata-Version: 2.3
2
+ Name: offlinedemo
3
+ Version: 0.3.0
4
+ Summary: Easy demonstration of models by offlineisbetter.
5
+ Author: offlineisbetter
6
+ Author-email: offlineisbetter <dev@offlineisbetter.com>
7
+ Requires-Dist: annotated-doc==0.0.5
8
+ Requires-Dist: anyio==4.15.1
9
+ Requires-Dist: certifi==2026.7.22
10
+ Requires-Dist: click==8.5.0
11
+ Requires-Dist: cuda-bindings==13.4.3
12
+ Requires-Dist: cuda-pathfinder==1.8.2
13
+ Requires-Dist: cuda-toolkit==13.0.3.0
14
+ Requires-Dist: filelock==4.0.6
15
+ Requires-Dist: flatbuffers==25.12.19
16
+ Requires-Dist: fsspec==2026.9.0
17
+ Requires-Dist: h11==0.16.0
18
+ Requires-Dist: hf-xet==1.6.0
19
+ Requires-Dist: httpcore==1.0.9
20
+ Requires-Dist: httpx==0.28.1
21
+ Requires-Dist: huggingface-hub==1.33.0
22
+ Requires-Dist: idna==3.20
23
+ Requires-Dist: jinja2==3.1.6
24
+ Requires-Dist: markdown-it-py==4.2.0
25
+ Requires-Dist: markupsafe==3.0.3
26
+ Requires-Dist: mdurl==0.1.2
27
+ Requires-Dist: ml-dtypes==0.6.0
28
+ Requires-Dist: mpmath==1.3.0
29
+ Requires-Dist: networkx==3.7
30
+ Requires-Dist: numpy==2.5.3
31
+ Requires-Dist: nvidia-cublas==13.1.1.3
32
+ Requires-Dist: nvidia-cuda-cupti==13.0.85
33
+ Requires-Dist: nvidia-cuda-nvrtc==13.0.88
34
+ Requires-Dist: nvidia-cuda-runtime==13.0.96
35
+ Requires-Dist: nvidia-cudnn-cu13==9.24.0.43
36
+ Requires-Dist: nvidia-cufft==12.0.0.61
37
+ Requires-Dist: nvidia-cufile==1.15.1.6
38
+ Requires-Dist: nvidia-curand==10.4.0.35
39
+ Requires-Dist: nvidia-cusolver==12.0.4.66
40
+ Requires-Dist: nvidia-cusparse==12.6.3.3
41
+ Requires-Dist: nvidia-cusparselt-cu13==0.8.1
42
+ Requires-Dist: nvidia-nccl-cu13==2.30.7
43
+ Requires-Dist: nvidia-nvjitlink==13.4.92
44
+ Requires-Dist: nvidia-nvshmem-cu13==3.4.5
45
+ Requires-Dist: nvidia-nvtx==13.0.85
46
+ Requires-Dist: onnx==1.23.1
47
+ Requires-Dist: onnxruntime==1.30.0
48
+ Requires-Dist: packaging==26.3
49
+ Requires-Dist: protobuf==7.36.2
50
+ Requires-Dist: pygments==2.21.0
51
+ Requires-Dist: pyyaml==6.0.3
52
+ Requires-Dist: regex==2026.9.29
53
+ Requires-Dist: rich==15.0.0
54
+ Requires-Dist: safetensors==0.8.0
55
+ Requires-Dist: setuptools==84.0.0
56
+ Requires-Dist: shellingham==1.5.4
57
+ Requires-Dist: sympy==1.14.0
58
+ Requires-Dist: tokenizers==0.23.2
59
+ Requires-Dist: torch==2.14.0
60
+ Requires-Dist: tqdm==4.70.1
61
+ Requires-Dist: transformers==5.17.0
62
+ Requires-Dist: triton==3.8.0
63
+ Requires-Dist: typer==0.27.2
64
+ Requires-Dist: typing-extensions==4.16.0
65
+ Requires-Python: >=3.13
66
+ Description-Content-Type: text/markdown
67
+
68
+ # _offlineisbetter_
69
+
70
+ inference for models by _offlineisbetter_.
71
+
72
+ we believe that you shouldn't give your data to faceless companies, that you deserve to run text models locally, and that you shouldn't need to buy expensive hardware. so we're building _offlineisbetter_.
73
+
74
+ _offlineisbetter_ models will be for _encoding_ tasks: sentiment analysis, text tagging, document retrieval, etc., rather than for _decoding_ tasks like autoregressive generation. we believe that it's wasteful and dangerous to depend on cloud apis for frontier language models to do these simple tasks, and it should be almost mindless to download a model to _use_ it without dealing with runtimes or quantization formats.
75
+
76
+ ## try it yourself
77
+
78
+ try our first model yourself. our first model is a small (230m) text model for sentiment analysis called `offline-sentiment-small`.
79
+
80
+ ```bash
81
+ pip install offlinedemo
82
+ ```
83
+
84
+ download the model archive from the website and unpack the model.
85
+
86
+ ```bash
87
+ tar -xvf offline-sentiment-small.tar
88
+ ```
89
+
90
+ run the demo, passing the inflated directory containing the model checkpoint.
91
+
92
+ ```bash
93
+ offlinedemo offline-sentiment-small
94
+ ```
95
+
96
+ ## benchmarks
97
+
98
+ | model | parameters | p95 (ms) | f1 (validation) |
99
+ |:---:|:---:|:---:|:---:|
100
+ | `offline-sentiment-small` | 230m | 80.32 | 0.9489 |
101
+ | `distilbert-base` | 67m | 66.15 | 0.9321 |
102
+ | `roberta-base` |
103
+
104
+ ## philosophy
105
+
106
+ succinctly, the core philosophy of _offlineisbetter_ is that parameter-efficient and low-latency models should be easily accessible to everybody. of course hugging face and `transformers.pipeline` allows you to run sentiment analysis in three lines of python, but for more parameter-efficient models, already quantized and with optimized computation graphs.
@@ -0,0 +1,39 @@
1
+ # _offlineisbetter_
2
+
3
+ inference for models by _offlineisbetter_.
4
+
5
+ we believe that you shouldn't give your data to faceless companies, that you deserve to run text models locally, and that you shouldn't need to buy expensive hardware. so we're building _offlineisbetter_.
6
+
7
+ _offlineisbetter_ models will be for _encoding_ tasks: sentiment analysis, text tagging, document retrieval, etc., rather than for _decoding_ tasks like autoregressive generation. we believe that it's wasteful and dangerous to depend on cloud apis for frontier language models to do these simple tasks, and it should be almost mindless to download a model to _use_ it without dealing with runtimes or quantization formats.
8
+
9
+ ## try it yourself
10
+
11
+ try our first model yourself. our first model is a small (230m) text model for sentiment analysis called `offline-sentiment-small`.
12
+
13
+ ```bash
14
+ pip install offlinedemo
15
+ ```
16
+
17
+ download the model archive from the website and unpack the model.
18
+
19
+ ```bash
20
+ tar -xvf offline-sentiment-small.tar
21
+ ```
22
+
23
+ run the demo, passing the inflated directory containing the model checkpoint.
24
+
25
+ ```bash
26
+ offlinedemo offline-sentiment-small
27
+ ```
28
+
29
+ ## benchmarks
30
+
31
+ | model | parameters | p95 (ms) | f1 (validation) |
32
+ |:---:|:---:|:---:|:---:|
33
+ | `offline-sentiment-small` | 230m | 80.32 | 0.9489 |
34
+ | `distilbert-base` | 67m | 66.15 | 0.9321 |
35
+ | `roberta-base` |
36
+
37
+ ## philosophy
38
+
39
+ succinctly, the core philosophy of _offlineisbetter_ is that parameter-efficient and low-latency models should be easily accessible to everybody. of course hugging face and `transformers.pipeline` allows you to run sentiment analysis in three lines of python, but for more parameter-efficient models, already quantized and with optimized computation graphs.
@@ -0,0 +1,77 @@
1
+ [project]
2
+ name = "offlinedemo"
3
+ version = "0.3.0"
4
+ description = "Easy demonstration of models by offlineisbetter."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ dependencies = [
8
+ "annotated-doc==0.0.5",
9
+ "anyio==4.15.1",
10
+ "certifi==2026.7.22",
11
+ "click==8.5.0",
12
+ "cuda-bindings==13.4.3",
13
+ "cuda-pathfinder==1.8.2",
14
+ "cuda-toolkit==13.0.3.0",
15
+ "filelock==4.0.6",
16
+ "flatbuffers==25.12.19",
17
+ "fsspec==2026.9.0",
18
+ "h11==0.16.0",
19
+ "hf-xet==1.6.0",
20
+ "httpcore==1.0.9",
21
+ "httpx==0.28.1",
22
+ "huggingface-hub==1.33.0",
23
+ "idna==3.20",
24
+ "jinja2==3.1.6",
25
+ "markdown-it-py==4.2.0",
26
+ "markupsafe==3.0.3",
27
+ "mdurl==0.1.2",
28
+ "ml-dtypes==0.6.0",
29
+ "mpmath==1.3.0",
30
+ "networkx==3.7",
31
+ "numpy==2.5.3",
32
+ "nvidia-cublas==13.1.1.3",
33
+ "nvidia-cuda-cupti==13.0.85",
34
+ "nvidia-cuda-nvrtc==13.0.88",
35
+ "nvidia-cuda-runtime==13.0.96",
36
+ "nvidia-cudnn-cu13==9.24.0.43",
37
+ "nvidia-cufft==12.0.0.61",
38
+ "nvidia-cufile==1.15.1.6",
39
+ "nvidia-curand==10.4.0.35",
40
+ "nvidia-cusolver==12.0.4.66",
41
+ "nvidia-cusparse==12.6.3.3",
42
+ "nvidia-cusparselt-cu13==0.8.1",
43
+ "nvidia-nccl-cu13==2.30.7",
44
+ "nvidia-nvjitlink==13.4.92",
45
+ "nvidia-nvshmem-cu13==3.4.5",
46
+ "nvidia-nvtx==13.0.85",
47
+ "onnx==1.23.1",
48
+ "onnxruntime==1.30.0",
49
+ "packaging==26.3",
50
+ "protobuf==7.36.2",
51
+ "pygments==2.21.0",
52
+ "pyyaml==6.0.3",
53
+ "regex==2026.9.29",
54
+ "rich==15.0.0",
55
+ "safetensors==0.8.0",
56
+ "setuptools==84.0.0",
57
+ "shellingham==1.5.4",
58
+ "sympy==1.14.0",
59
+ "tokenizers==0.23.2",
60
+ "torch==2.14.0",
61
+ "tqdm==4.70.1",
62
+ "transformers==5.17.0",
63
+ "triton==3.8.0",
64
+ "typer==0.27.2",
65
+ "typing-extensions==4.16.0",
66
+ ]
67
+
68
+ [[project.authors]]
69
+ name = "offlineisbetter"
70
+ email = "dev@offlineisbetter.com"
71
+
72
+ [project.scripts]
73
+ offlinedemo = "offlinedemo:main"
74
+
75
+ [build-system]
76
+ requires = ["uv_build>=0.12.19,<0.13.0"]
77
+ build-backend = "uv_build"
@@ -0,0 +1,76 @@
1
+ [project]
2
+ name = "offlinedemo"
3
+ version = "0.3.0"
4
+ description = "Easy demonstration of models by offlineisbetter."
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "offlineisbetter", email = "dev@offlineisbetter.com" }
8
+ ]
9
+ requires-python = ">=3.13"
10
+ dependencies = [
11
+ "annotated-doc==0.0.5",
12
+ "anyio==4.15.1",
13
+ "certifi==2026.7.22",
14
+ "click==8.5.0",
15
+ "cuda-bindings==13.4.3",
16
+ "cuda-pathfinder==1.8.2",
17
+ "cuda-toolkit==13.0.3.0",
18
+ "filelock==4.0.6",
19
+ "flatbuffers==25.12.19",
20
+ "fsspec==2026.9.0",
21
+ "h11==0.16.0",
22
+ "hf-xet==1.6.0",
23
+ "httpcore==1.0.9",
24
+ "httpx==0.28.1",
25
+ "huggingface-hub==1.33.0",
26
+ "idna==3.20",
27
+ "jinja2==3.1.6",
28
+ "markdown-it-py==4.2.0",
29
+ "markupsafe==3.0.3",
30
+ "mdurl==0.1.2",
31
+ "ml-dtypes==0.6.0",
32
+ "mpmath==1.3.0",
33
+ "networkx==3.7",
34
+ "numpy==2.5.3",
35
+ "nvidia-cublas==13.1.1.3",
36
+ "nvidia-cuda-cupti==13.0.85",
37
+ "nvidia-cuda-nvrtc==13.0.88",
38
+ "nvidia-cuda-runtime==13.0.96",
39
+ "nvidia-cudnn-cu13==9.24.0.43",
40
+ "nvidia-cufft==12.0.0.61",
41
+ "nvidia-cufile==1.15.1.6",
42
+ "nvidia-curand==10.4.0.35",
43
+ "nvidia-cusolver==12.0.4.66",
44
+ "nvidia-cusparse==12.6.3.3",
45
+ "nvidia-cusparselt-cu13==0.8.1",
46
+ "nvidia-nccl-cu13==2.30.7",
47
+ "nvidia-nvjitlink==13.4.92",
48
+ "nvidia-nvshmem-cu13==3.4.5",
49
+ "nvidia-nvtx==13.0.85",
50
+ "onnx==1.23.1",
51
+ "onnxruntime==1.30.0",
52
+ "packaging==26.3",
53
+ "protobuf==7.36.2",
54
+ "pygments==2.21.0",
55
+ "pyyaml==6.0.3",
56
+ "regex==2026.9.29",
57
+ "rich==15.0.0",
58
+ "safetensors==0.8.0",
59
+ "setuptools==84.0.0",
60
+ "shellingham==1.5.4",
61
+ "sympy==1.14.0",
62
+ "tokenizers==0.23.2",
63
+ "torch==2.14.0",
64
+ "tqdm==4.70.1",
65
+ "transformers==5.17.0",
66
+ "triton==3.8.0",
67
+ "typer==0.27.2",
68
+ "typing-extensions==4.16.0",
69
+ ]
70
+
71
+ [project.scripts]
72
+ offlinedemo = "offlinedemo:main"
73
+
74
+ [build-system]
75
+ requires = ["uv_build>=0.12.19,<0.13.0"]
76
+ build-backend = "uv_build"
@@ -2,16 +2,18 @@
2
2
  # Copyright (c) 2026- offlineisbetter
3
3
 
4
4
  from pathlib import Path
5
+ import json
5
6
  import sys
6
7
  import time
7
8
 
9
+ import numpy as np
8
10
  from transformers import AutoTokenizer
9
11
  import onnxruntime as ort
10
12
 
11
13
  def main():
12
14
  # Get checkpoint name
13
15
  if len(sys.argv) < 2:
14
- print("offlinedemo [checkpoint]")
16
+ print("offlinedemo [checkpoint-dir]")
15
17
  return
16
18
  checkpoint = Path(sys.argv[1])
17
19
 
@@ -21,13 +23,19 @@ def main():
21
23
  providers=["CPUExecutionProvider"],
22
24
  )
23
25
  print("Model loaded!")
24
- print()
26
+
27
+ # Load classes
28
+ with open(checkpoint / "offlineisbetter.json", "r") as f:
29
+ classes = json.load(f)
30
+ classes = {v: k for k, v in classes.items()}
25
31
 
26
32
  # Load tokenizer
27
33
  tokenizer = AutoTokenizer.from_pretrained(
28
34
  checkpoint,
29
35
  local_files_only = True,
30
36
  )
37
+ print("Tokenizer loaded!")
38
+ print()
31
39
 
32
40
  # Ask user for input
33
41
  user_input = input("offlineisbetter >> ")
@@ -35,9 +43,12 @@ def main():
35
43
  # Tokenize and run inference
36
44
  start = time.perf_counter()
37
45
  tokens = tokenizer(user_input)
38
- result = session.run(None, tokens)[0]
46
+ result = session.run(None, tokens)[0][0]
47
+ idx = int(np.argmax(result))
48
+ c = classes[idx]
39
49
  duration = time.perf_counter() - start
40
50
 
41
51
  # Report to user
42
52
  print(f"LOGITS: {result}")
53
+ print(f"CLASS: {c}")
43
54
  print(f"LATENCY: {duration*1000:.3f} ms")
@@ -1,12 +0,0 @@
1
- Metadata-Version: 2.3
2
- Name: offlinedemo
3
- Version: 0.1.0
4
- Summary: Easy demonstration of models by offlineisbetter.
5
- Author: offlineisbetter
6
- Author-email: offlineisbetter <dev@offlineisbetter.com>
7
- Requires-Python: >=3.13
8
- Description-Content-Type: text/markdown
9
-
10
- # _offlineisbetter_
11
-
12
- Inference for _offlineisbetter_ models.
@@ -1,3 +0,0 @@
1
- # _offlineisbetter_
2
-
3
- Inference for _offlineisbetter_ models.
@@ -1,18 +0,0 @@
1
- [project]
2
- name = "offlinedemo"
3
- version = "0.1.0"
4
- description = "Easy demonstration of models by offlineisbetter."
5
- readme = "README.md"
6
- requires-python = ">=3.13"
7
- dependencies = []
8
-
9
- [[project.authors]]
10
- name = "offlineisbetter"
11
- email = "dev@offlineisbetter.com"
12
-
13
- [project.scripts]
14
- offlinedemo = "offlinedemo:main"
15
-
16
- [build-system]
17
- requires = ["uv_build>=0.12.19,<0.13.0"]
18
- build-backend = "uv_build"
@@ -1,17 +0,0 @@
1
- [project]
2
- name = "offlinedemo"
3
- version = "0.1.0"
4
- description = "Easy demonstration of models by offlineisbetter."
5
- readme = "README.md"
6
- authors = [
7
- { name = "offlineisbetter", email = "dev@offlineisbetter.com" }
8
- ]
9
- requires-python = ">=3.13"
10
- dependencies = []
11
-
12
- [project.scripts]
13
- offlinedemo = "offlinedemo:main"
14
-
15
- [build-system]
16
- requires = ["uv_build>=0.12.19,<0.13.0"]
17
- build-backend = "uv_build"