jeffy-classify 0.1.0a8__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- jeffy_classify-0.1.0a8/.gitignore +56 -0
- jeffy_classify-0.1.0a8/ATTRIBUTION.md +76 -0
- jeffy_classify-0.1.0a8/LICENSE +21 -0
- jeffy_classify-0.1.0a8/PKG-INFO +325 -0
- jeffy_classify-0.1.0a8/README.md +291 -0
- jeffy_classify-0.1.0a8/data/eval_results/benchmark.html +293 -0
- jeffy_classify-0.1.0a8/data/eval_results/benchmark.json +727 -0
- jeffy_classify-0.1.0a8/data/eval_results/jeff_comparison.json +162 -0
- jeffy_classify-0.1.0a8/examples/demo.gif +0 -0
- jeffy_classify-0.1.0a8/examples/demo_output.txt +43 -0
- jeffy_classify-0.1.0a8/examples/demo_recording.sh +84 -0
- jeffy_classify-0.1.0a8/examples/doom/fire_exact_head.npz +0 -0
- jeffy_classify-0.1.0a8/examples/doom/fire_s_classifier.npz +0 -0
- jeffy_classify-0.1.0a8/examples/doom/fire_s_schema.json +19 -0
- jeffy_classify-0.1.0a8/examples/doom_battle.gif +0 -0
- jeffy_classify-0.1.0a8/examples/doom_defend.gif +0 -0
- jeffy_classify-0.1.0a8/examples/inbox_demo.gif +0 -0
- jeffy_classify-0.1.0a8/examples/inbox_demo.py +146 -0
- jeffy_classify-0.1.0a8/examples/inbox_demo_source.html +599 -0
- jeffy_classify-0.1.0a8/examples/playground.png +0 -0
- jeffy_classify-0.1.0a8/pyproject.toml +61 -0
- jeffy_classify-0.1.0a8/src/jeffy/__init__.py +7 -0
- jeffy_classify-0.1.0a8/src/jeffy/benchmark_page.py +200 -0
- jeffy_classify-0.1.0a8/src/jeffy/build_pack.py +355 -0
- jeffy_classify-0.1.0a8/src/jeffy/catalog.py +148 -0
- jeffy_classify-0.1.0a8/src/jeffy/compare_jeff.py +454 -0
- jeffy_classify-0.1.0a8/src/jeffy/engine.py +194 -0
- jeffy_classify-0.1.0a8/src/jeffy/evaluate.py +401 -0
- jeffy_classify-0.1.0a8/src/jeffy/examples/__init__.py +15 -0
- jeffy_classify-0.1.0a8/src/jeffy/examples/reviews.csv +25 -0
- jeffy_classify-0.1.0a8/src/jeffy/model_pack.py +145 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/ag_news/manifest.json +31 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/ag_news/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/banking77/manifest.json +177 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/banking77/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/clinc_oos/manifest.json +325 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/clinc_oos/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/dbpedia/manifest.json +51 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/dbpedia/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/emotion/manifest.json +35 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/emotion/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/imdb/manifest.json +27 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/imdb/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/massive_intent/manifest.json +143 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/massive_intent/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/pack_manifest.json +152 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/sms_spam/manifest.json +27 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/sms_spam/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/snli/manifest.json +29 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/snli/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/sst2/manifest.json +27 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/sst2/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_emotion/manifest.json +31 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_emotion/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_offensive/manifest.json +27 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_offensive/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_sentiment/manifest.json +29 -0
- jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_sentiment/model.npz +0 -0
- jeffy_classify-0.1.0a8/src/jeffy/server.py +532 -0
- jeffy_classify-0.1.0a8/src/jeffy/train.py +281 -0
- jeffy_classify-0.1.0a8/tests/test_api.py +266 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.egg-info/
|
|
5
|
+
dist/
|
|
6
|
+
build/
|
|
7
|
+
*.egg
|
|
8
|
+
.eggs/
|
|
9
|
+
|
|
10
|
+
# Virtual environments
|
|
11
|
+
venv/
|
|
12
|
+
.venv/
|
|
13
|
+
env/
|
|
14
|
+
|
|
15
|
+
# IDE
|
|
16
|
+
.idea/
|
|
17
|
+
.vscode/
|
|
18
|
+
*.swp
|
|
19
|
+
*.swo
|
|
20
|
+
*~
|
|
21
|
+
|
|
22
|
+
# OS
|
|
23
|
+
.DS_Store
|
|
24
|
+
Thumbs.db
|
|
25
|
+
|
|
26
|
+
# Large model files (encoder weights - not tracked)
|
|
27
|
+
models/encoder/
|
|
28
|
+
*.safetensors
|
|
29
|
+
*.bin
|
|
30
|
+
*.onnx
|
|
31
|
+
|
|
32
|
+
# Dataset downloads and caches
|
|
33
|
+
data/downloads/
|
|
34
|
+
data/cache/
|
|
35
|
+
.cache/
|
|
36
|
+
|
|
37
|
+
# Sensitive
|
|
38
|
+
.env
|
|
39
|
+
*.credentials
|
|
40
|
+
*.token
|
|
41
|
+
|
|
42
|
+
# ViZDoom artifacts (not part of Jeffy)
|
|
43
|
+
_vizdoom.ini
|
|
44
|
+
*.wad
|
|
45
|
+
|
|
46
|
+
# Large per-example prediction files (reproduce with scripts)
|
|
47
|
+
# Keep summaries and comparison JSON in git
|
|
48
|
+
data/eval_results/*_predictions.jsonl
|
|
49
|
+
|
|
50
|
+
# Build output (regenerate with jeffy-build)
|
|
51
|
+
data/model_pack/
|
|
52
|
+
|
|
53
|
+
# Drafts and conversation artifacts (private)
|
|
54
|
+
SHOW_HN_DRAFT.md
|
|
55
|
+
JEFFY_HANDOFF.md
|
|
56
|
+
snapshot.json
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
# Attribution and Upstream Licenses
|
|
2
|
+
|
|
3
|
+
## Jeffy Code
|
|
4
|
+
|
|
5
|
+
MIT License. See LICENSE.
|
|
6
|
+
|
|
7
|
+
## Shared Encoder
|
|
8
|
+
|
|
9
|
+
**BAAI/bge-large-en-v1.5**
|
|
10
|
+
- Source: https://huggingface.co/BAAI/bge-large-en-v1.5
|
|
11
|
+
- License: MIT
|
|
12
|
+
- Downloaded at runtime by sentence-transformers; not included in this repository.
|
|
13
|
+
|
|
14
|
+
## Inspiration
|
|
15
|
+
|
|
16
|
+
**Jeff** by Mathias Strasser
|
|
17
|
+
- Source: https://github.com/firelex/jeff
|
|
18
|
+
- License: MIT (code), Apache 2.0 (model weights)
|
|
19
|
+
- Jeffy's /v1/systemone endpoint and playground design are inspired by Jeff.
|
|
20
|
+
No Jeff code is included in this repository.
|
|
21
|
+
|
|
22
|
+
## Pretrained Head Artifacts
|
|
23
|
+
|
|
24
|
+
The model pack contains logistic regression coefficients and scaler statistics
|
|
25
|
+
trained on public datasets. These are derived model parameters — numerical
|
|
26
|
+
arrays that do not contain any training text.
|
|
27
|
+
|
|
28
|
+
Each artifact's `manifest.json` records the source dataset. Source terms were
|
|
29
|
+
checked on HuggingFace and original project sites (Sep 2026).
|
|
30
|
+
|
|
31
|
+
**What we distribute:** Numerical classifier parameters (coefficients, scaler means/scales).
|
|
32
|
+
**What we do NOT distribute:** Training text, dataset copies, or dataset downloads.
|
|
33
|
+
**Reproduction:** Each head can be retrained from its source dataset via `jeffy-build`.
|
|
34
|
+
|
|
35
|
+
### Verified: explicit license permitting derivative works
|
|
36
|
+
|
|
37
|
+
| Dataset | License | Source | Checked |
|
|
38
|
+
|---------|---------|--------|---------|
|
|
39
|
+
| banking77 | CC BY 4.0 | [HF](https://huggingface.co/datasets/legacy-datasets/banking77) | HF metadata |
|
|
40
|
+
| clinc_oos | CC BY 3.0 | [HF](https://huggingface.co/datasets/clinc_oos) | HF metadata |
|
|
41
|
+
| massive_intent | CC BY 4.0 | [HF](https://huggingface.co/datasets/mteb/amazon_massive_intent) | HF metadata |
|
|
42
|
+
| sms_spam | CC BY 4.0 | [HF](https://huggingface.co/datasets/ucirvine/sms_spam) | HF metadata |
|
|
43
|
+
| snli | CC BY-SA 4.0 | [HF](https://huggingface.co/datasets/stanfordnlp/snli) | HF metadata |
|
|
44
|
+
| dbpedia | CC BY-SA 3.0 | [HF](https://huggingface.co/datasets/fancyzhx/dbpedia_14) | HF metadata |
|
|
45
|
+
| tweet_eval_sentiment | CC BY 3.0 | [HF](https://huggingface.co/datasets/cardiffnlp/tweet_eval) | Listed per-subset on HF card |
|
|
46
|
+
|
|
47
|
+
### No explicit license on HF; no restriction on trained weights found
|
|
48
|
+
|
|
49
|
+
| Dataset | HF License Field | Original Source | Finding |
|
|
50
|
+
|---------|-----------------|-----------------|---------|
|
|
51
|
+
| ag_news | "unknown" | AG corpus (original site unreachable) | No terms found. Academic paper origin, widely redistributed. |
|
|
52
|
+
| imdb | "other" | [Stanford](https://ai.stanford.edu/~amaas/data/sentiment/) | Original site requests citation only. No use restrictions stated. |
|
|
53
|
+
| sst2 | "unknown" | Stanford NLP | No terms on HF or original page beyond citation. |
|
|
54
|
+
| emotion | "other" | [HF](https://huggingface.co/datasets/dair-ai/emotion) | HF card states "educational and research purposes only." |
|
|
55
|
+
|
|
56
|
+
### Unresolved: source terms require further review
|
|
57
|
+
|
|
58
|
+
| Dataset | Issue | Detail |
|
|
59
|
+
|---------|-------|--------|
|
|
60
|
+
| emotion | HF card says "educational and research purposes only" | This may restrict commercial use of the dataset. Whether it applies to derived model weights (which contain no text) is not established. |
|
|
61
|
+
| tweet_eval_emotion | Twitter API TOS required; per-subset license undefined | All tweet_eval subsets require Twitter TOS compliance. Our weights contain no tweet text. The sentiment subset is CC BY 3.0; the emotion and offensive subsets have undefined per-subset licenses. |
|
|
62
|
+
| tweet_eval_offensive | Twitter API TOS required; per-subset license undefined | Same as above. |
|
|
63
|
+
|
|
64
|
+
No dataset in this inventory explicitly prohibits distributing trained model
|
|
65
|
+
weights. The three unresolved cases involve ambiguous or restrictive dataset-use
|
|
66
|
+
terms whose applicability to derived numerical parameters has not been reviewed.
|
|
67
|
+
|
|
68
|
+
## scikit-learn
|
|
69
|
+
|
|
70
|
+
Bundled pretrained artifacts use numpy's .npz format for portability.
|
|
71
|
+
Custom-trained models also save a joblib/pickle backup.
|
|
72
|
+
scikit-learn (BSD-3-Clause) is required at runtime to reconstruct classifiers.
|
|
73
|
+
|
|
74
|
+
## sentence-transformers
|
|
75
|
+
|
|
76
|
+
The encoder is loaded via sentence-transformers, which is Apache 2.0 licensed.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nicolas Brenner
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,325 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: jeffy-classify
|
|
3
|
+
Version: 0.1.0a8
|
|
4
|
+
Summary: Pretrained text classifiers: 13 ready-to-use heads, train your own in seconds
|
|
5
|
+
Project-URL: Homepage, https://github.com/nicobrenner/jeffy
|
|
6
|
+
Project-URL: Repository, https://github.com/nicobrenner/jeffy
|
|
7
|
+
Project-URL: Issues, https://github.com/nicobrenner/jeffy/issues
|
|
8
|
+
Author: Nico Brenner
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: embeddings,machine-learning,nlp,pretrained,text-classification
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
21
|
+
Requires-Python: >=3.11
|
|
22
|
+
Requires-Dist: fastapi>=0.100.0
|
|
23
|
+
Requires-Dist: joblib>=1.3.0
|
|
24
|
+
Requires-Dist: numpy>=1.24.0
|
|
25
|
+
Requires-Dist: scikit-learn>=1.3.0
|
|
26
|
+
Requires-Dist: scipy>=1.10.0
|
|
27
|
+
Requires-Dist: sentence-transformers>=2.2.0
|
|
28
|
+
Requires-Dist: uvicorn>=0.20.0
|
|
29
|
+
Provides-Extra: build
|
|
30
|
+
Requires-Dist: datasets>=2.14.0; extra == 'build'
|
|
31
|
+
Provides-Extra: dev
|
|
32
|
+
Requires-Dist: httpx>=0.24.0; extra == 'dev'
|
|
33
|
+
Description-Content-Type: text/markdown
|
|
34
|
+
|
|
35
|
+
# Jeffy
|
|
36
|
+
|
|
37
|
+
Pretrained text classifiers you can run and retrain on CPU.
|
|
38
|
+
|
|
39
|
+
<p align="center">
|
|
40
|
+
<img src="examples/inbox_demo.gif" width="100%" alt="Inbox Router">
|
|
41
|
+
</p>
|
|
42
|
+
<p align="center">
|
|
43
|
+
<img src="examples/doom_battle.gif" width="49%" alt="Doom Battle">
|
|
44
|
+
<img src="examples/doom_defend.gif" width="49%" alt="Defend the Center">
|
|
45
|
+
</p>
|
|
46
|
+
|
|
47
|
+
## Try it
|
|
48
|
+
|
|
49
|
+
[Install uv](https://docs.astral.sh/uv/getting-started/installation/), then:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
uvx --python 3.12 \
|
|
53
|
+
--from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
|
|
54
|
+
jeffy-serve
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Open http://localhost:8400, pick a classifier, and paste one of these:
|
|
58
|
+
|
|
59
|
+
| Classifier | Try this text |
|
|
60
|
+
|------------|---------------|
|
|
61
|
+
| banking77 | I was charged twice for the same transaction |
|
|
62
|
+
| sms_spam | WINNER! You have been selected for a free cruise. Reply YES to claim. |
|
|
63
|
+
| ag_news | The Federal Reserve raised interest rates by 25 basis points on Wednesday |
|
|
64
|
+
|
|
65
|
+
<p align="center">
|
|
66
|
+
<img src="examples/playground.png" width="100%" alt="Jeffy Playground">
|
|
67
|
+
</p>
|
|
68
|
+
|
|
69
|
+
## Pretrained capabilities
|
|
70
|
+
|
|
71
|
+
13 classifiers ship with the package. Weights are logistic regression coefficients (derived model parameters, not copies of training data). Source datasets and licenses are documented in `ATTRIBUTION.md`.
|
|
72
|
+
|
|
73
|
+
| Task | What it does | Classes | Test Acc | Test F1 |
|
|
74
|
+
|------|-------------|---------|----------|---------|
|
|
75
|
+
| sms_spam | SMS spam detection | 2 | 99.1% | 98.0% |
|
|
76
|
+
| dbpedia | Wikipedia article category | 14 | 96.0% | 95.9% |
|
|
77
|
+
| imdb | Movie review sentiment (long text) | 2 | 94.8% | 94.8% |
|
|
78
|
+
| banking77 | Banking customer intent | 77 | 94.3% | 94.3% |
|
|
79
|
+
| ag_news | News topic (world/sports/business/tech) | 4 | 90.5% | 90.5% |
|
|
80
|
+
| sst2 | Movie review sentiment | 2 | 90.1% | 90.1% |
|
|
81
|
+
| clinc_oos | Voice assistant intent + out-of-scope | 151 | 88.4% | 92.1% |
|
|
82
|
+
| massive_intent | Smart home voice commands | 60 | 88.1% | 86.4% |
|
|
83
|
+
| tweet_eval_offensive | Offensive language | 2 | 81.0% | 74.8% |
|
|
84
|
+
| tweet_eval_emotion | Tweet emotion | 4 | 78.1% | 74.7% |
|
|
85
|
+
| emotion | Text emotion (6 emotions) | 6 | 75.5% | 67.8% |
|
|
86
|
+
| tweet_eval_sentiment | Tweet sentiment (3-way) | 3 | 66.2% | 65.7% |
|
|
87
|
+
| snli | Natural language inference | 3 | 65.6% | 65.2% |
|
|
88
|
+
|
|
89
|
+
Test accuracy on held-out splits. Details in `data/eval_results/benchmark.json`.
|
|
90
|
+
|
|
91
|
+
**Weaknesses:** SNLI (65.6%) and tweet_eval_sentiment (66.2%) are below what task-specific models achieve. Emotion (75.5%) has limited class coverage. Probabilities are uncalibrated.
|
|
92
|
+
|
|
93
|
+
## Train a custom classifier
|
|
94
|
+
|
|
95
|
+
```bash
|
|
96
|
+
uvx --python 3.12 \
|
|
97
|
+
--from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
|
|
98
|
+
jeffy-train --example --save-dir my_models
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
```
|
|
102
|
+
Loaded 24 examples from reviews.csv
|
|
103
|
+
Training 'reviews': 24 examples, 2 classes
|
|
104
|
+
Split: 19 train, 5 test
|
|
105
|
+
Test accuracy: 100.0%
|
|
106
|
+
Saved to my_models/reviews/
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
`--example` uses a bundled 24-row product review CSV. To bring your own:
|
|
110
|
+
|
|
111
|
+
```bash
|
|
112
|
+
uvx --python 3.12 \
|
|
113
|
+
--from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
|
|
114
|
+
jeffy-train --input your_data.csv --text-col text --label-col label \
|
|
115
|
+
--task-id your_task --save-dir my_models
|
|
116
|
+
```
|
|
117
|
+
|
|
118
|
+
Supports `.csv`, `.tsv`, and `.jsonl`.
|
|
119
|
+
|
|
120
|
+
## Serve a custom model
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
JEFFY_PACK_DIR=my_models uvx --python 3.12 \
|
|
124
|
+
--from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
|
|
125
|
+
jeffy-serve
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
```bash
|
|
129
|
+
curl -s -X POST http://localhost:8400/v1/predict \
|
|
130
|
+
-H "Content-Type: application/json" \
|
|
131
|
+
-d '{"text": "The battery life is amazing", "task": "reviews"}'
|
|
132
|
+
# → {"label": "positive", "confidence": 0.87, ...}
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
## Install from source
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
git clone https://github.com/nicobrenner/jeffy.git
|
|
139
|
+
cd jeffy
|
|
140
|
+
|
|
141
|
+
# With uv (recommended)
|
|
142
|
+
uv venv && uv pip install -e .
|
|
143
|
+
|
|
144
|
+
# Or with pip
|
|
145
|
+
python -m venv .venv && source .venv/bin/activate
|
|
146
|
+
pip install -e .
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
The first prediction downloads the shared encoder ([bge-large-en-v1.5](https://huggingface.co/BAAI/bge-large-en-v1.5), ~1.2 GB, cached afterward).
|
|
150
|
+
|
|
151
|
+
## API
|
|
152
|
+
|
|
153
|
+
### Python
|
|
154
|
+
|
|
155
|
+
```python
|
|
156
|
+
from jeffy.engine import Engine
|
|
157
|
+
|
|
158
|
+
engine = Engine()
|
|
159
|
+
engine.load()
|
|
160
|
+
|
|
161
|
+
# List available classifiers
|
|
162
|
+
for name, cap in engine.capabilities.items():
|
|
163
|
+
print(f"{name}: {cap.description} ({cap.n_classes} classes)")
|
|
164
|
+
|
|
165
|
+
# Classify text
|
|
166
|
+
result = engine.predict("banking77", "I was charged twice for the same transaction")
|
|
167
|
+
print(result["label"]) # "transaction_charged_twice"
|
|
168
|
+
print(result["confidence"]) # 0.999
|
|
169
|
+
print(result["probabilities"]) # {"transaction_charged_twice": 0.999, ...}
|
|
170
|
+
```
|
|
171
|
+
|
|
172
|
+
### SDK walkthrough
|
|
173
|
+
|
|
174
|
+

|
|
175
|
+
|
|
176
|
+
### HTTP
|
|
177
|
+
|
|
178
|
+
```bash
|
|
179
|
+
# Banking intent
|
|
180
|
+
curl -s -X POST http://localhost:8400/v1/predict \
|
|
181
|
+
-H "Content-Type: application/json" \
|
|
182
|
+
-d '{"text": "I was charged twice for the same transaction", "task": "banking77"}'
|
|
183
|
+
# → {"label": "transaction_charged_twice", "confidence": 0.999, ...}
|
|
184
|
+
|
|
185
|
+
# Spam detection
|
|
186
|
+
curl -s -X POST http://localhost:8400/v1/predict \
|
|
187
|
+
-H "Content-Type: application/json" \
|
|
188
|
+
-d '{"text": "WINNER! You have been selected for a free cruise. Reply YES to claim.", "task": "sms_spam"}'
|
|
189
|
+
# → {"label": "spam", "confidence": 0.91, ...}
|
|
190
|
+
|
|
191
|
+
# News topic
|
|
192
|
+
curl -s -X POST http://localhost:8400/v1/predict \
|
|
193
|
+
-H "Content-Type: application/json" \
|
|
194
|
+
-d '{"text": "The Federal Reserve raised interest rates by 25 basis points on Wednesday", "task": "ag_news"}'
|
|
195
|
+
# → {"label": "Business", "confidence": 0.86, ...}
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
### List capabilities
|
|
199
|
+
|
|
200
|
+
```bash
|
|
201
|
+
curl -s http://localhost:8400/v1/capabilities | python3 -c "
|
|
202
|
+
import json, sys
|
|
203
|
+
for c in json.load(sys.stdin)['capabilities']:
|
|
204
|
+
print(f\"{c['task_id']:25s} {c['n_classes']:3d} classes {c['test_accuracy']:.1%} {c['name']}\")"
|
|
205
|
+
```
|
|
206
|
+
|
|
207
|
+
### Per-capability metadata
|
|
208
|
+
|
|
209
|
+
Each shipped classifier has a `manifest.json` with label names, source dataset, HuggingFace path, stated license, encoder identity, training/test counts, and integrity hashes.
|
|
210
|
+
|
|
211
|
+
```bash
|
|
212
|
+
curl -s http://localhost:8400/v1/capabilities/banking77 | python3 -m json.tool
|
|
213
|
+
```
|
|
214
|
+
|
|
215
|
+
## Train from Python
|
|
216
|
+
|
|
217
|
+
```python
|
|
218
|
+
from jeffy.train import train_classifier
|
|
219
|
+
|
|
220
|
+
clf = train_classifier(
|
|
221
|
+
texts=["great product!", "terrible service", "fast shipping", "broken on arrival"],
|
|
222
|
+
labels=["positive", "negative", "positive", "negative"],
|
|
223
|
+
task_id="my_reviews",
|
|
224
|
+
)
|
|
225
|
+
|
|
226
|
+
result = clf.predict("the quality exceeded my expectations")
|
|
227
|
+
print(result["label"]) # "positive"
|
|
228
|
+
|
|
229
|
+
clf.save("my_models")
|
|
230
|
+
```
|
|
231
|
+
|
|
232
|
+
### Tuning options
|
|
233
|
+
|
|
234
|
+
| Parameter | Default | Description |
|
|
235
|
+
|-----------|---------|-------------|
|
|
236
|
+
| `C` | 0.01 | How aggressively the model fits your data. Low (0.001) = conservative, keeps predictions closer to "I'm not sure." High (1.0) = trusts individual training examples more. If the model is great on training data but bad on new data (overfitting), lower C. |
|
|
237
|
+
| `test_size` | 0.2 | What fraction of your data to hold back for testing. With 100 examples at 0.2, it trains on 80 and tests on 20. Set to 0 to train on everything (useful when you have very little data and will test manually). |
|
|
238
|
+
| `cv_folds` | 3 | Cross-validation: splits your training data into 3 parts, trains on 2 and tests on 1, rotates three times, averages the scores. Gives a more reliable accuracy estimate than a single split. Set to 0 to skip (faster, less reliable estimate). |
|
|
239
|
+
|
|
240
|
+
Start with the defaults. With <50 examples per class, expect noisy estimates.
|
|
241
|
+
|
|
242
|
+
## Reproduce the evaluation
|
|
243
|
+
|
|
244
|
+
```bash
|
|
245
|
+
# Install with build dependencies
|
|
246
|
+
uv pip install -e ".[build]"
|
|
247
|
+
# or: pip install -e ".[build]"
|
|
248
|
+
|
|
249
|
+
# Retrain all 13 heads from source datasets (~40 min, downloads ~5 GB)
|
|
250
|
+
jeffy-build --out data/model_pack
|
|
251
|
+
|
|
252
|
+
# Evaluate on held-out test sets with tuned baselines
|
|
253
|
+
jeffy-evaluate --baselines --latency --device cpu
|
|
254
|
+
```
|
|
255
|
+
|
|
256
|
+
## Evaluation notes
|
|
257
|
+
|
|
258
|
+
- **SST-2**: Evaluated on `validation` split (official test labels are not public).
|
|
259
|
+
- **SMS Spam**: Random split (test_size=0.2, seed=42); no standard benchmark split.
|
|
260
|
+
- **SNLI**: Input encoded as `premise [SEP] hypothesis`. Label -1 filtered.
|
|
261
|
+
- **CLINC-OOS**: 151 classes including out-of-scope. In-scope accuracy 96.5%, OOS detection 51.7%.
|
|
262
|
+
- **MASSIVE**: English only (config `en`).
|
|
263
|
+
|
|
264
|
+
## Deployment
|
|
265
|
+
|
|
266
|
+
| Component | Size | Required for |
|
|
267
|
+
|-----------|------|-------------|
|
|
268
|
+
| Jeffy package (wheel) | 1.5 MB | Always (includes all 13 heads) |
|
|
269
|
+
| Encoder (bge-large-en-v1.5) | ~1.2 GB | Inference (downloaded on first use) |
|
|
270
|
+
| `datasets` package | ~100 MB | Retraining from HuggingFace only |
|
|
271
|
+
|
|
272
|
+
Runtime memory: ~2 GB (encoder loaded once, shared across all heads).
|
|
273
|
+
|
|
274
|
+
**Latency** (CPU, single example, Linux aarch64):
|
|
275
|
+
|
|
276
|
+
| Stage | p50 | Notes |
|
|
277
|
+
|-------|-----|-------|
|
|
278
|
+
| Embedding | 50–80 ms | Dominates; varies with input length |
|
|
279
|
+
| Classifier | <1 ms | Negligible |
|
|
280
|
+
| Total | 50–80 ms | End-to-end |
|
|
281
|
+
|
|
282
|
+
## Model security
|
|
283
|
+
|
|
284
|
+
Bundled pretrained artifacts use numpy `.npz` format (portable, no pickle). Custom-trained models also save a joblib pickle backup. **Only load custom pickle artifacts from trusted sources.** Each artifact's `manifest.json` includes integrity hashes verified on load.
|
|
285
|
+
|
|
286
|
+
## Provenance and licensing
|
|
287
|
+
|
|
288
|
+
Jeffy code is MIT-licensed. Head artifacts are derived from public datasets; redistribution permissions have **not been independently verified** for all sources. See `ATTRIBUTION.md` for per-dataset license status.
|
|
289
|
+
|
|
290
|
+
| Dataset | Stated license |
|
|
291
|
+
|---------|---------------|
|
|
292
|
+
| banking77, massive_intent, sms_spam | CC BY 4.0 |
|
|
293
|
+
| clinc_oos | CC BY 3.0 |
|
|
294
|
+
| dbpedia | CC BY-SA 3.0 |
|
|
295
|
+
| snli | CC BY-SA 4.0 |
|
|
296
|
+
| ag_news, imdb | Academic / non-commercial |
|
|
297
|
+
| sst2 | Stanford academic license |
|
|
298
|
+
| emotion | Academic |
|
|
299
|
+
| tweet_eval_* | Twitter TOS / academic |
|
|
300
|
+
|
|
301
|
+
The encoder ([bge-large-en-v1.5](https://huggingface.co/BAAI/bge-large-en-v1.5)) is MIT-licensed.
|
|
302
|
+
|
|
303
|
+
## Tested
|
|
304
|
+
|
|
305
|
+
Verified with clean-environment wheel and sdist install on Linux aarch64, Python 3.12, scikit-learn 1.9+, sentence-transformers 6.1+, numpy 2.5+. Pretrained artifacts use numpy `.npz` format, avoiding sklearn version coupling.
|
|
306
|
+
|
|
307
|
+
## What's not included
|
|
308
|
+
|
|
309
|
+
- **No zero-shot / general classification.** Each task needs a trained head. Unknown tasks return an error.
|
|
310
|
+
- **No LLM fallback.** This release is pure embedding + classifier.
|
|
311
|
+
- **No automatic task routing.** You must specify which classifier to use.
|
|
312
|
+
|
|
313
|
+
## Roadmap
|
|
314
|
+
|
|
315
|
+
| Status | Milestone |
|
|
316
|
+
|--------|-----------|
|
|
317
|
+
| **Available** | Pretrained classifier library, SDK/API, playground, custom training from CSV/JSONL |
|
|
318
|
+
| **Next** | Landing page, PyPI release |
|
|
319
|
+
| **Planned** | Broader classifier catalog, released in verified batches |
|
|
320
|
+
| **Planned** | Automatic routing among supported classifiers |
|
|
321
|
+
| **Planned** | Optional local/API LLM fallback for unsupported tasks |
|
|
322
|
+
| **Planned** | Non-text classifiers (game state, sensor data, structured features) |
|
|
323
|
+
| **Exploring** | Assisted labeling, retraining from corrections, classifier sharing |
|
|
324
|
+
|
|
325
|
+
Suggestions for datasets, capabilities, or workflows are welcome as issues.
|