offlinedemo 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,67 +1,12 @@
1
1
  Metadata-Version: 2.3
2
2
  Name: offlinedemo
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Easy demonstration of models by offlineisbetter.
5
5
  Author: offlineisbetter
6
6
  Author-email: offlineisbetter <dev@offlineisbetter.com>
7
- Requires-Dist: annotated-doc==0.0.5
8
- Requires-Dist: anyio==4.15.1
9
- Requires-Dist: certifi==2026.7.22
10
- Requires-Dist: click==8.5.0
11
- Requires-Dist: cuda-bindings==13.4.3
12
- Requires-Dist: cuda-pathfinder==1.8.2
13
- Requires-Dist: cuda-toolkit==13.0.3.0
14
- Requires-Dist: filelock==4.0.6
15
- Requires-Dist: flatbuffers==25.12.19
16
- Requires-Dist: fsspec==2026.9.0
17
- Requires-Dist: h11==0.16.0
18
- Requires-Dist: hf-xet==1.6.0
19
- Requires-Dist: httpcore==1.0.9
20
- Requires-Dist: httpx==0.28.1
21
- Requires-Dist: huggingface-hub==1.33.0
22
- Requires-Dist: idna==3.20
23
- Requires-Dist: jinja2==3.1.6
24
- Requires-Dist: markdown-it-py==4.2.0
25
- Requires-Dist: markupsafe==3.0.3
26
- Requires-Dist: mdurl==0.1.2
27
- Requires-Dist: ml-dtypes==0.6.0
28
- Requires-Dist: mpmath==1.3.0
29
- Requires-Dist: networkx==3.7
30
- Requires-Dist: numpy==2.5.3
31
- Requires-Dist: nvidia-cublas==13.1.1.3
32
- Requires-Dist: nvidia-cuda-cupti==13.0.85
33
- Requires-Dist: nvidia-cuda-nvrtc==13.0.88
34
- Requires-Dist: nvidia-cuda-runtime==13.0.96
35
- Requires-Dist: nvidia-cudnn-cu13==9.24.0.43
36
- Requires-Dist: nvidia-cufft==12.0.0.61
37
- Requires-Dist: nvidia-cufile==1.15.1.6
38
- Requires-Dist: nvidia-curand==10.4.0.35
39
- Requires-Dist: nvidia-cusolver==12.0.4.66
40
- Requires-Dist: nvidia-cusparse==12.6.3.3
41
- Requires-Dist: nvidia-cusparselt-cu13==0.8.1
42
- Requires-Dist: nvidia-nccl-cu13==2.30.7
43
- Requires-Dist: nvidia-nvjitlink==13.4.92
44
- Requires-Dist: nvidia-nvshmem-cu13==3.4.5
45
- Requires-Dist: nvidia-nvtx==13.0.85
46
- Requires-Dist: onnx==1.23.1
47
- Requires-Dist: onnxruntime==1.30.0
48
- Requires-Dist: packaging==26.3
49
- Requires-Dist: protobuf==7.36.2
50
- Requires-Dist: pygments==2.21.0
51
- Requires-Dist: pyyaml==6.0.3
52
- Requires-Dist: regex==2026.9.29
53
- Requires-Dist: rich==15.0.0
54
- Requires-Dist: safetensors==0.8.0
55
- Requires-Dist: setuptools==84.0.0
56
- Requires-Dist: shellingham==1.5.4
57
- Requires-Dist: sympy==1.14.0
58
- Requires-Dist: tokenizers==0.23.2
59
- Requires-Dist: torch==2.14.0
60
- Requires-Dist: tqdm==4.70.1
61
- Requires-Dist: transformers==5.17.0
62
- Requires-Dist: triton==3.8.0
63
- Requires-Dist: typer==0.27.2
64
- Requires-Dist: typing-extensions==4.16.0
7
+ Requires-Dist: numpy>=2.5.3
8
+ Requires-Dist: onnxruntime>=1.30.0
9
+ Requires-Dist: tokenizers>=0.23.2
65
10
  Requires-Python: >=3.13
66
11
  Description-Content-Type: text/markdown
67
12
 
@@ -81,7 +26,7 @@ try our first model yourself. our first model is a small (230m) text model for s
81
26
  pip install offlinedemo
82
27
  ```
83
28
 
84
- download the model archive from the website and unpack the model.
29
+ download the model archive from the releases page of this repo and unpack the model.
85
30
 
86
31
  ```bash
87
32
  tar -xvf offline-sentiment-small.tar
@@ -97,6 +42,8 @@ offlinedemo offline-sentiment-small
97
42
 
98
43
  below are benchmarks for `offline-sentiment-small` on binary sentiment classification using the [stanfordnlp/sst2](https://huggingface.co/datasets/stanfordnlp/sst2) validation set. all benchmarks were completed on the ryzen 9950x3d cpu on one thread.
99
44
 
45
+ critically: _we only allowed 5 minutes of setup time_ for each model. our focus is on improving the developer's experience and reducing headache! that means that `*bert` models were used in their original form (`torch`), as downloaded from hugging face, as was ours (`onnx`). this is _not an apples-to-apples comparison_ in the same runtime, and that's intentional.
46
+
100
47
  | model | parameters | p95 (ms) | f1 (validation) |
101
48
  |:---:|:---:|:---:|:---:|
102
49
  | `offline-sentiment-small` | 230m | 80.32 | 0.9489 |
@@ -14,7 +14,7 @@ try our first model yourself. our first model is a small (230m) text model for s
14
14
  pip install offlinedemo
15
15
  ```
16
16
 
17
- download the model archive from the website and unpack the model.
17
+ download the model archive from the releases page of this repo and unpack the model.
18
18
 
19
19
  ```bash
20
20
  tar -xvf offline-sentiment-small.tar
@@ -30,6 +30,8 @@ offlinedemo offline-sentiment-small
30
30
 
31
31
  below are benchmarks for `offline-sentiment-small` on binary sentiment classification using the [stanfordnlp/sst2](https://huggingface.co/datasets/stanfordnlp/sst2) validation set. all benchmarks were completed on the ryzen 9950x3d cpu on one thread.
32
32
 
33
+ critically: _we only allowed 5 minutes of setup time_ for each model. our focus is on improving the developer's experience and reducing headache! that means that `*bert` models were used in their original form (`torch`), as downloaded from hugging face, as was ours (`onnx`). this is _not an apples-to-apples comparison_ in the same runtime, and that's intentional.
34
+
33
35
  | model | parameters | p95 (ms) | f1 (validation) |
34
36
  |:---:|:---:|:---:|:---:|
35
37
  | `offline-sentiment-small` | 230m | 80.32 | 0.9489 |
@@ -0,0 +1,22 @@
1
+ [project]
2
+ name = "offlinedemo"
3
+ version = "0.5.0"
4
+ description = "Easy demonstration of models by offlineisbetter."
5
+ readme = "README.md"
6
+ requires-python = ">=3.13"
7
+ dependencies = [
8
+ "numpy>=2.5.3",
9
+ "onnxruntime>=1.30.0",
10
+ "tokenizers>=0.23.2",
11
+ ]
12
+
13
+ [[project.authors]]
14
+ name = "offlineisbetter"
15
+ email = "dev@offlineisbetter.com"
16
+
17
+ [project.scripts]
18
+ offlinedemo = "offlinedemo:main"
19
+
20
+ [build-system]
21
+ requires = ["uv_build>=0.12.19,<0.13.0"]
22
+ build-backend = "uv_build"
@@ -0,0 +1,21 @@
1
+ [project]
2
+ name = "offlinedemo"
3
+ version = "0.5.0"
4
+ description = "Easy demonstration of models by offlineisbetter."
5
+ readme = "README.md"
6
+ authors = [
7
+ { name = "offlineisbetter", email = "dev@offlineisbetter.com" }
8
+ ]
9
+ requires-python = ">=3.13"
10
+ dependencies = [
11
+ "numpy>=2.5.3",
12
+ "onnxruntime>=1.30.0",
13
+ "tokenizers>=0.23.2",
14
+ ]
15
+
16
+ [project.scripts]
17
+ offlinedemo = "offlinedemo:main"
18
+
19
+ [build-system]
20
+ requires = ["uv_build>=0.12.19,<0.13.0"]
21
+ build-backend = "uv_build"
@@ -3,11 +3,12 @@
3
3
 
4
4
  from pathlib import Path
5
5
  import json
6
+ import os
6
7
  import sys
7
8
  import time
8
9
 
9
10
  import numpy as np
10
- from transformers import AutoTokenizer
11
+ from tokenizers import Tokenizer
11
12
  import onnxruntime as ort
12
13
 
13
14
  def main():
@@ -30,10 +31,7 @@ def main():
30
31
  classes = {v: k for k, v in classes.items()}
31
32
 
32
33
  # Load tokenizer
33
- tokenizer = AutoTokenizer.from_pretrained(
34
- checkpoint,
35
- local_files_only = True,
36
- )
34
+ tokenizer = Tokenizer.from_file(str(checkpoint / "tokenizer.json"))
37
35
  print("Tokenizer loaded!")
38
36
  print()
39
37
 
@@ -42,7 +40,11 @@ def main():
42
40
 
43
41
  # Tokenize and run inference
44
42
  start = time.perf_counter()
45
- tokens = tokenizer(user_input)
43
+ encoded = tokenizer.encode(user_input)
44
+ tokens = {
45
+ "input_ids": np.array(encoded.ids),
46
+ "attention_mask": np.array(encoded.attention_mask),
47
+ }
46
48
  result = session.run(None, tokens)[0][0]
47
49
  idx = int(np.argmax(result))
48
50
  c = classes[idx]
@@ -1,77 +0,0 @@
1
- [project]
2
- name = "offlinedemo"
3
- version = "0.4.0"
4
- description = "Easy demonstration of models by offlineisbetter."
5
- readme = "README.md"
6
- requires-python = ">=3.13"
7
- dependencies = [
8
- "annotated-doc==0.0.5",
9
- "anyio==4.15.1",
10
- "certifi==2026.7.22",
11
- "click==8.5.0",
12
- "cuda-bindings==13.4.3",
13
- "cuda-pathfinder==1.8.2",
14
- "cuda-toolkit==13.0.3.0",
15
- "filelock==4.0.6",
16
- "flatbuffers==25.12.19",
17
- "fsspec==2026.9.0",
18
- "h11==0.16.0",
19
- "hf-xet==1.6.0",
20
- "httpcore==1.0.9",
21
- "httpx==0.28.1",
22
- "huggingface-hub==1.33.0",
23
- "idna==3.20",
24
- "jinja2==3.1.6",
25
- "markdown-it-py==4.2.0",
26
- "markupsafe==3.0.3",
27
- "mdurl==0.1.2",
28
- "ml-dtypes==0.6.0",
29
- "mpmath==1.3.0",
30
- "networkx==3.7",
31
- "numpy==2.5.3",
32
- "nvidia-cublas==13.1.1.3",
33
- "nvidia-cuda-cupti==13.0.85",
34
- "nvidia-cuda-nvrtc==13.0.88",
35
- "nvidia-cuda-runtime==13.0.96",
36
- "nvidia-cudnn-cu13==9.24.0.43",
37
- "nvidia-cufft==12.0.0.61",
38
- "nvidia-cufile==1.15.1.6",
39
- "nvidia-curand==10.4.0.35",
40
- "nvidia-cusolver==12.0.4.66",
41
- "nvidia-cusparse==12.6.3.3",
42
- "nvidia-cusparselt-cu13==0.8.1",
43
- "nvidia-nccl-cu13==2.30.7",
44
- "nvidia-nvjitlink==13.4.92",
45
- "nvidia-nvshmem-cu13==3.4.5",
46
- "nvidia-nvtx==13.0.85",
47
- "onnx==1.23.1",
48
- "onnxruntime==1.30.0",
49
- "packaging==26.3",
50
- "protobuf==7.36.2",
51
- "pygments==2.21.0",
52
- "pyyaml==6.0.3",
53
- "regex==2026.9.29",
54
- "rich==15.0.0",
55
- "safetensors==0.8.0",
56
- "setuptools==84.0.0",
57
- "shellingham==1.5.4",
58
- "sympy==1.14.0",
59
- "tokenizers==0.23.2",
60
- "torch==2.14.0",
61
- "tqdm==4.70.1",
62
- "transformers==5.17.0",
63
- "triton==3.8.0",
64
- "typer==0.27.2",
65
- "typing-extensions==4.16.0",
66
- ]
67
-
68
- [[project.authors]]
69
- name = "offlineisbetter"
70
- email = "dev@offlineisbetter.com"
71
-
72
- [project.scripts]
73
- offlinedemo = "offlinedemo:main"
74
-
75
- [build-system]
76
- requires = ["uv_build>=0.12.19,<0.13.0"]
77
- build-backend = "uv_build"
@@ -1,76 +0,0 @@
1
- [project]
2
- name = "offlinedemo"
3
- version = "0.4.0"
4
- description = "Easy demonstration of models by offlineisbetter."
5
- readme = "README.md"
6
- authors = [
7
- { name = "offlineisbetter", email = "dev@offlineisbetter.com" }
8
- ]
9
- requires-python = ">=3.13"
10
- dependencies = [
11
- "annotated-doc==0.0.5",
12
- "anyio==4.15.1",
13
- "certifi==2026.7.22",
14
- "click==8.5.0",
15
- "cuda-bindings==13.4.3",
16
- "cuda-pathfinder==1.8.2",
17
- "cuda-toolkit==13.0.3.0",
18
- "filelock==4.0.6",
19
- "flatbuffers==25.12.19",
20
- "fsspec==2026.9.0",
21
- "h11==0.16.0",
22
- "hf-xet==1.6.0",
23
- "httpcore==1.0.9",
24
- "httpx==0.28.1",
25
- "huggingface-hub==1.33.0",
26
- "idna==3.20",
27
- "jinja2==3.1.6",
28
- "markdown-it-py==4.2.0",
29
- "markupsafe==3.0.3",
30
- "mdurl==0.1.2",
31
- "ml-dtypes==0.6.0",
32
- "mpmath==1.3.0",
33
- "networkx==3.7",
34
- "numpy==2.5.3",
35
- "nvidia-cublas==13.1.1.3",
36
- "nvidia-cuda-cupti==13.0.85",
37
- "nvidia-cuda-nvrtc==13.0.88",
38
- "nvidia-cuda-runtime==13.0.96",
39
- "nvidia-cudnn-cu13==9.24.0.43",
40
- "nvidia-cufft==12.0.0.61",
41
- "nvidia-cufile==1.15.1.6",
42
- "nvidia-curand==10.4.0.35",
43
- "nvidia-cusolver==12.0.4.66",
44
- "nvidia-cusparse==12.6.3.3",
45
- "nvidia-cusparselt-cu13==0.8.1",
46
- "nvidia-nccl-cu13==2.30.7",
47
- "nvidia-nvjitlink==13.4.92",
48
- "nvidia-nvshmem-cu13==3.4.5",
49
- "nvidia-nvtx==13.0.85",
50
- "onnx==1.23.1",
51
- "onnxruntime==1.30.0",
52
- "packaging==26.3",
53
- "protobuf==7.36.2",
54
- "pygments==2.21.0",
55
- "pyyaml==6.0.3",
56
- "regex==2026.9.29",
57
- "rich==15.0.0",
58
- "safetensors==0.8.0",
59
- "setuptools==84.0.0",
60
- "shellingham==1.5.4",
61
- "sympy==1.14.0",
62
- "tokenizers==0.23.2",
63
- "torch==2.14.0",
64
- "tqdm==4.70.1",
65
- "transformers==5.17.0",
66
- "triton==3.8.0",
67
- "typer==0.27.2",
68
- "typing-extensions==4.16.0",
69
- ]
70
-
71
- [project.scripts]
72
- offlinedemo = "offlinedemo:main"
73
-
74
- [build-system]
75
- requires = ["uv_build>=0.12.19,<0.13.0"]
76
- build-backend = "uv_build"