jeffy-classify 0.1.0a8__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (61) hide show
  1. jeffy_classify-0.1.0a8/.gitignore +56 -0
  2. jeffy_classify-0.1.0a8/ATTRIBUTION.md +76 -0
  3. jeffy_classify-0.1.0a8/LICENSE +21 -0
  4. jeffy_classify-0.1.0a8/PKG-INFO +325 -0
  5. jeffy_classify-0.1.0a8/README.md +291 -0
  6. jeffy_classify-0.1.0a8/data/eval_results/benchmark.html +293 -0
  7. jeffy_classify-0.1.0a8/data/eval_results/benchmark.json +727 -0
  8. jeffy_classify-0.1.0a8/data/eval_results/jeff_comparison.json +162 -0
  9. jeffy_classify-0.1.0a8/examples/demo.gif +0 -0
  10. jeffy_classify-0.1.0a8/examples/demo_output.txt +43 -0
  11. jeffy_classify-0.1.0a8/examples/demo_recording.sh +84 -0
  12. jeffy_classify-0.1.0a8/examples/doom/fire_exact_head.npz +0 -0
  13. jeffy_classify-0.1.0a8/examples/doom/fire_s_classifier.npz +0 -0
  14. jeffy_classify-0.1.0a8/examples/doom/fire_s_schema.json +19 -0
  15. jeffy_classify-0.1.0a8/examples/doom_battle.gif +0 -0
  16. jeffy_classify-0.1.0a8/examples/doom_defend.gif +0 -0
  17. jeffy_classify-0.1.0a8/examples/inbox_demo.gif +0 -0
  18. jeffy_classify-0.1.0a8/examples/inbox_demo.py +146 -0
  19. jeffy_classify-0.1.0a8/examples/inbox_demo_source.html +599 -0
  20. jeffy_classify-0.1.0a8/examples/playground.png +0 -0
  21. jeffy_classify-0.1.0a8/pyproject.toml +61 -0
  22. jeffy_classify-0.1.0a8/src/jeffy/__init__.py +7 -0
  23. jeffy_classify-0.1.0a8/src/jeffy/benchmark_page.py +200 -0
  24. jeffy_classify-0.1.0a8/src/jeffy/build_pack.py +355 -0
  25. jeffy_classify-0.1.0a8/src/jeffy/catalog.py +148 -0
  26. jeffy_classify-0.1.0a8/src/jeffy/compare_jeff.py +454 -0
  27. jeffy_classify-0.1.0a8/src/jeffy/engine.py +194 -0
  28. jeffy_classify-0.1.0a8/src/jeffy/evaluate.py +401 -0
  29. jeffy_classify-0.1.0a8/src/jeffy/examples/__init__.py +15 -0
  30. jeffy_classify-0.1.0a8/src/jeffy/examples/reviews.csv +25 -0
  31. jeffy_classify-0.1.0a8/src/jeffy/model_pack.py +145 -0
  32. jeffy_classify-0.1.0a8/src/jeffy/pack/ag_news/manifest.json +31 -0
  33. jeffy_classify-0.1.0a8/src/jeffy/pack/ag_news/model.npz +0 -0
  34. jeffy_classify-0.1.0a8/src/jeffy/pack/banking77/manifest.json +177 -0
  35. jeffy_classify-0.1.0a8/src/jeffy/pack/banking77/model.npz +0 -0
  36. jeffy_classify-0.1.0a8/src/jeffy/pack/clinc_oos/manifest.json +325 -0
  37. jeffy_classify-0.1.0a8/src/jeffy/pack/clinc_oos/model.npz +0 -0
  38. jeffy_classify-0.1.0a8/src/jeffy/pack/dbpedia/manifest.json +51 -0
  39. jeffy_classify-0.1.0a8/src/jeffy/pack/dbpedia/model.npz +0 -0
  40. jeffy_classify-0.1.0a8/src/jeffy/pack/emotion/manifest.json +35 -0
  41. jeffy_classify-0.1.0a8/src/jeffy/pack/emotion/model.npz +0 -0
  42. jeffy_classify-0.1.0a8/src/jeffy/pack/imdb/manifest.json +27 -0
  43. jeffy_classify-0.1.0a8/src/jeffy/pack/imdb/model.npz +0 -0
  44. jeffy_classify-0.1.0a8/src/jeffy/pack/massive_intent/manifest.json +143 -0
  45. jeffy_classify-0.1.0a8/src/jeffy/pack/massive_intent/model.npz +0 -0
  46. jeffy_classify-0.1.0a8/src/jeffy/pack/pack_manifest.json +152 -0
  47. jeffy_classify-0.1.0a8/src/jeffy/pack/sms_spam/manifest.json +27 -0
  48. jeffy_classify-0.1.0a8/src/jeffy/pack/sms_spam/model.npz +0 -0
  49. jeffy_classify-0.1.0a8/src/jeffy/pack/snli/manifest.json +29 -0
  50. jeffy_classify-0.1.0a8/src/jeffy/pack/snli/model.npz +0 -0
  51. jeffy_classify-0.1.0a8/src/jeffy/pack/sst2/manifest.json +27 -0
  52. jeffy_classify-0.1.0a8/src/jeffy/pack/sst2/model.npz +0 -0
  53. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_emotion/manifest.json +31 -0
  54. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_emotion/model.npz +0 -0
  55. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_offensive/manifest.json +27 -0
  56. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_offensive/model.npz +0 -0
  57. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_sentiment/manifest.json +29 -0
  58. jeffy_classify-0.1.0a8/src/jeffy/pack/tweet_eval_sentiment/model.npz +0 -0
  59. jeffy_classify-0.1.0a8/src/jeffy/server.py +532 -0
  60. jeffy_classify-0.1.0a8/src/jeffy/train.py +281 -0
  61. jeffy_classify-0.1.0a8/tests/test_api.py +266 -0
@@ -0,0 +1,56 @@
1
+ # Python
2
+ __pycache__/
3
+ *.py[cod]
4
+ *.egg-info/
5
+ dist/
6
+ build/
7
+ *.egg
8
+ .eggs/
9
+
10
+ # Virtual environments
11
+ venv/
12
+ .venv/
13
+ env/
14
+
15
+ # IDE
16
+ .idea/
17
+ .vscode/
18
+ *.swp
19
+ *.swo
20
+ *~
21
+
22
+ # OS
23
+ .DS_Store
24
+ Thumbs.db
25
+
26
+ # Large model files (encoder weights - not tracked)
27
+ models/encoder/
28
+ *.safetensors
29
+ *.bin
30
+ *.onnx
31
+
32
+ # Dataset downloads and caches
33
+ data/downloads/
34
+ data/cache/
35
+ .cache/
36
+
37
+ # Sensitive
38
+ .env
39
+ *.credentials
40
+ *.token
41
+
42
+ # ViZDoom artifacts (not part of Jeffy)
43
+ _vizdoom.ini
44
+ *.wad
45
+
46
+ # Large per-example prediction files (reproduce with scripts)
47
+ # Keep summaries and comparison JSON in git
48
+ data/eval_results/*_predictions.jsonl
49
+
50
+ # Build output (regenerate with jeffy-build)
51
+ data/model_pack/
52
+
53
+ # Drafts and conversation artifacts (private)
54
+ SHOW_HN_DRAFT.md
55
+ JEFFY_HANDOFF.md
56
+ snapshot.json
@@ -0,0 +1,76 @@
1
+ # Attribution and Upstream Licenses
2
+
3
+ ## Jeffy Code
4
+
5
+ MIT License. See LICENSE.
6
+
7
+ ## Shared Encoder
8
+
9
+ **BAAI/bge-large-en-v1.5**
10
+ - Source: https://huggingface.co/BAAI/bge-large-en-v1.5
11
+ - License: MIT
12
+ - Downloaded at runtime by sentence-transformers; not included in this repository.
13
+
14
+ ## Inspiration
15
+
16
+ **Jeff** by Mathias Strasser
17
+ - Source: https://github.com/firelex/jeff
18
+ - License: MIT (code), Apache 2.0 (model weights)
19
+ - Jeffy's /v1/systemone endpoint and playground design are inspired by Jeff.
20
+ No Jeff code is included in this repository.
21
+
22
+ ## Pretrained Head Artifacts
23
+
24
+ The model pack contains logistic regression coefficients and scaler statistics
25
+ trained on public datasets. These are derived model parameters — numerical
26
+ arrays that do not contain any training text.
27
+
28
+ Each artifact's `manifest.json` records the source dataset. Source terms were
29
+ checked on HuggingFace and original project sites (Sep 2026).
30
+
31
+ **What we distribute:** Numerical classifier parameters (coefficients, scaler means/scales).
32
+ **What we do NOT distribute:** Training text, dataset copies, or dataset downloads.
33
+ **Reproduction:** Each head can be retrained from its source dataset via `jeffy-build`.
34
+
35
+ ### Verified: explicit license permitting derivative works
36
+
37
+ | Dataset | License | Source | Checked |
38
+ |---------|---------|--------|---------|
39
+ | banking77 | CC BY 4.0 | [HF](https://huggingface.co/datasets/legacy-datasets/banking77) | HF metadata |
40
+ | clinc_oos | CC BY 3.0 | [HF](https://huggingface.co/datasets/clinc_oos) | HF metadata |
41
+ | massive_intent | CC BY 4.0 | [HF](https://huggingface.co/datasets/mteb/amazon_massive_intent) | HF metadata |
42
+ | sms_spam | CC BY 4.0 | [HF](https://huggingface.co/datasets/ucirvine/sms_spam) | HF metadata |
43
+ | snli | CC BY-SA 4.0 | [HF](https://huggingface.co/datasets/stanfordnlp/snli) | HF metadata |
44
+ | dbpedia | CC BY-SA 3.0 | [HF](https://huggingface.co/datasets/fancyzhx/dbpedia_14) | HF metadata |
45
+ | tweet_eval_sentiment | CC BY 3.0 | [HF](https://huggingface.co/datasets/cardiffnlp/tweet_eval) | Listed per-subset on HF card |
46
+
47
+ ### No explicit license on HF; no restriction on trained weights found
48
+
49
+ | Dataset | HF License Field | Original Source | Finding |
50
+ |---------|-----------------|-----------------|---------|
51
+ | ag_news | "unknown" | AG corpus (original site unreachable) | No terms found. Academic paper origin, widely redistributed. |
52
+ | imdb | "other" | [Stanford](https://ai.stanford.edu/~amaas/data/sentiment/) | Original site requests citation only. No use restrictions stated. |
53
+ | sst2 | "unknown" | Stanford NLP | No terms on HF or original page beyond citation. |
54
+ | emotion | "other" | [HF](https://huggingface.co/datasets/dair-ai/emotion) | HF card states "educational and research purposes only." |
55
+
56
+ ### Unresolved: source terms require further review
57
+
58
+ | Dataset | Issue | Detail |
59
+ |---------|-------|--------|
60
+ | emotion | HF card says "educational and research purposes only" | This may restrict commercial use of the dataset. Whether it applies to derived model weights (which contain no text) is not established. |
61
+ | tweet_eval_emotion | Twitter API TOS required; per-subset license undefined | All tweet_eval subsets require Twitter TOS compliance. Our weights contain no tweet text. The sentiment subset is CC BY 3.0; the emotion and offensive subsets have undefined per-subset licenses. |
62
+ | tweet_eval_offensive | Twitter API TOS required; per-subset license undefined | Same as above. |
63
+
64
+ No dataset in this inventory explicitly prohibits distributing trained model
65
+ weights. The three unresolved cases involve ambiguous or restrictive dataset-use
66
+ terms whose applicability to derived numerical parameters has not been reviewed.
67
+
68
+ ## scikit-learn
69
+
70
+ Bundled pretrained artifacts use numpy's .npz format for portability.
71
+ Custom-trained models also save a joblib/pickle backup.
72
+ scikit-learn (BSD-3-Clause) is required at runtime to reconstruct classifiers.
73
+
74
+ ## sentence-transformers
75
+
76
+ The encoder is loaded via sentence-transformers, which is Apache 2.0 licensed.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Nicolas Brenner
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,325 @@
1
+ Metadata-Version: 2.5
2
+ Name: jeffy-classify
3
+ Version: 0.1.0a8
4
+ Summary: Pretrained text classifiers: 13 ready-to-use heads, train your own in seconds
5
+ Project-URL: Homepage, https://github.com/nicobrenner/jeffy
6
+ Project-URL: Repository, https://github.com/nicobrenner/jeffy
7
+ Project-URL: Issues, https://github.com/nicobrenner/jeffy/issues
8
+ Author: Nico Brenner
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: embeddings,machine-learning,nlp,pretrained,text-classification
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
20
+ Classifier: Topic :: Text Processing :: Linguistic
21
+ Requires-Python: >=3.11
22
+ Requires-Dist: fastapi>=0.100.0
23
+ Requires-Dist: joblib>=1.3.0
24
+ Requires-Dist: numpy>=1.24.0
25
+ Requires-Dist: scikit-learn>=1.3.0
26
+ Requires-Dist: scipy>=1.10.0
27
+ Requires-Dist: sentence-transformers>=2.2.0
28
+ Requires-Dist: uvicorn>=0.20.0
29
+ Provides-Extra: build
30
+ Requires-Dist: datasets>=2.14.0; extra == 'build'
31
+ Provides-Extra: dev
32
+ Requires-Dist: httpx>=0.24.0; extra == 'dev'
33
+ Description-Content-Type: text/markdown
34
+
35
+ # Jeffy
36
+
37
+ Pretrained text classifiers you can run and retrain on CPU.
38
+
39
+ <p align="center">
40
+ <img src="examples/inbox_demo.gif" width="100%" alt="Inbox Router">
41
+ </p>
42
+ <p align="center">
43
+ <img src="examples/doom_battle.gif" width="49%" alt="Doom Battle">
44
+ <img src="examples/doom_defend.gif" width="49%" alt="Defend the Center">
45
+ </p>
46
+
47
+ ## Try it
48
+
49
+ [Install uv](https://docs.astral.sh/uv/getting-started/installation/), then:
50
+
51
+ ```bash
52
+ uvx --python 3.12 \
53
+ --from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
54
+ jeffy-serve
55
+ ```
56
+
57
+ Open http://localhost:8400, pick a classifier, and paste one of these:
58
+
59
+ | Classifier | Try this text |
60
+ |------------|---------------|
61
+ | banking77 | I was charged twice for the same transaction |
62
+ | sms_spam | WINNER! You have been selected for a free cruise. Reply YES to claim. |
63
+ | ag_news | The Federal Reserve raised interest rates by 25 basis points on Wednesday |
64
+
65
+ <p align="center">
66
+ <img src="examples/playground.png" width="100%" alt="Jeffy Playground">
67
+ </p>
68
+
69
+ ## Pretrained capabilities
70
+
71
+ 13 classifiers ship with the package. Weights are logistic regression coefficients (derived model parameters, not copies of training data). Source datasets and licenses are documented in `ATTRIBUTION.md`.
72
+
73
+ | Task | What it does | Classes | Test Acc | Test F1 |
74
+ |------|-------------|---------|----------|---------|
75
+ | sms_spam | SMS spam detection | 2 | 99.1% | 98.0% |
76
+ | dbpedia | Wikipedia article category | 14 | 96.0% | 95.9% |
77
+ | imdb | Movie review sentiment (long text) | 2 | 94.8% | 94.8% |
78
+ | banking77 | Banking customer intent | 77 | 94.3% | 94.3% |
79
+ | ag_news | News topic (world/sports/business/tech) | 4 | 90.5% | 90.5% |
80
+ | sst2 | Movie review sentiment | 2 | 90.1% | 90.1% |
81
+ | clinc_oos | Voice assistant intent + out-of-scope | 151 | 88.4% | 92.1% |
82
+ | massive_intent | Smart home voice commands | 60 | 88.1% | 86.4% |
83
+ | tweet_eval_offensive | Offensive language | 2 | 81.0% | 74.8% |
84
+ | tweet_eval_emotion | Tweet emotion | 4 | 78.1% | 74.7% |
85
+ | emotion | Text emotion (6 emotions) | 6 | 75.5% | 67.8% |
86
+ | tweet_eval_sentiment | Tweet sentiment (3-way) | 3 | 66.2% | 65.7% |
87
+ | snli | Natural language inference | 3 | 65.6% | 65.2% |
88
+
89
+ Test accuracy on held-out splits. Details in `data/eval_results/benchmark.json`.
90
+
91
+ **Weaknesses:** SNLI (65.6%) and tweet_eval_sentiment (66.2%) are below what task-specific models achieve. Emotion (75.5%) has limited class coverage. Probabilities are uncalibrated.
92
+
93
+ ## Train a custom classifier
94
+
95
+ ```bash
96
+ uvx --python 3.12 \
97
+ --from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
98
+ jeffy-train --example --save-dir my_models
99
+ ```
100
+
101
+ ```
102
+ Loaded 24 examples from reviews.csv
103
+ Training 'reviews': 24 examples, 2 classes
104
+ Split: 19 train, 5 test
105
+ Test accuracy: 100.0%
106
+ Saved to my_models/reviews/
107
+ ```
108
+
109
+ `--example` uses a bundled 24-row product review CSV. To bring your own:
110
+
111
+ ```bash
112
+ uvx --python 3.12 \
113
+ --from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
114
+ jeffy-train --input your_data.csv --text-col text --label-col label \
115
+ --task-id your_task --save-dir my_models
116
+ ```
117
+
118
+ Supports `.csv`, `.tsv`, and `.jsonl`.
119
+
120
+ ## Serve a custom model
121
+
122
+ ```bash
123
+ JEFFY_PACK_DIR=my_models uvx --python 3.12 \
124
+ --from "jeffy-classify @ git+https://github.com/nicobrenner/jeffy.git@v0.1.0-alpha.7" \
125
+ jeffy-serve
126
+ ```
127
+
128
+ ```bash
129
+ curl -s -X POST http://localhost:8400/v1/predict \
130
+ -H "Content-Type: application/json" \
131
+ -d '{"text": "The battery life is amazing", "task": "reviews"}'
132
+ # → {"label": "positive", "confidence": 0.87, ...}
133
+ ```
134
+
135
+ ## Install from source
136
+
137
+ ```bash
138
+ git clone https://github.com/nicobrenner/jeffy.git
139
+ cd jeffy
140
+
141
+ # With uv (recommended)
142
+ uv venv && uv pip install -e .
143
+
144
+ # Or with pip
145
+ python -m venv .venv && source .venv/bin/activate
146
+ pip install -e .
147
+ ```
148
+
149
+ The first prediction downloads the shared encoder ([bge-large-en-v1.5](https://huggingface.co/BAAI/bge-large-en-v1.5), ~1.2 GB, cached afterward).
150
+
151
+ ## API
152
+
153
+ ### Python
154
+
155
+ ```python
156
+ from jeffy.engine import Engine
157
+
158
+ engine = Engine()
159
+ engine.load()
160
+
161
+ # List available classifiers
162
+ for name, cap in engine.capabilities.items():
163
+ print(f"{name}: {cap.description} ({cap.n_classes} classes)")
164
+
165
+ # Classify text
166
+ result = engine.predict("banking77", "I was charged twice for the same transaction")
167
+ print(result["label"]) # "transaction_charged_twice"
168
+ print(result["confidence"]) # 0.999
169
+ print(result["probabilities"]) # {"transaction_charged_twice": 0.999, ...}
170
+ ```
171
+
172
+ ### SDK walkthrough
173
+
174
+ ![SDK walkthrough](examples/demo.gif)
175
+
176
+ ### HTTP
177
+
178
+ ```bash
179
+ # Banking intent
180
+ curl -s -X POST http://localhost:8400/v1/predict \
181
+ -H "Content-Type: application/json" \
182
+ -d '{"text": "I was charged twice for the same transaction", "task": "banking77"}'
183
+ # → {"label": "transaction_charged_twice", "confidence": 0.999, ...}
184
+
185
+ # Spam detection
186
+ curl -s -X POST http://localhost:8400/v1/predict \
187
+ -H "Content-Type: application/json" \
188
+ -d '{"text": "WINNER! You have been selected for a free cruise. Reply YES to claim.", "task": "sms_spam"}'
189
+ # → {"label": "spam", "confidence": 0.91, ...}
190
+
191
+ # News topic
192
+ curl -s -X POST http://localhost:8400/v1/predict \
193
+ -H "Content-Type: application/json" \
194
+ -d '{"text": "The Federal Reserve raised interest rates by 25 basis points on Wednesday", "task": "ag_news"}'
195
+ # → {"label": "Business", "confidence": 0.86, ...}
196
+ ```
197
+
198
+ ### List capabilities
199
+
200
+ ```bash
201
+ curl -s http://localhost:8400/v1/capabilities | python3 -c "
202
+ import json, sys
203
+ for c in json.load(sys.stdin)['capabilities']:
204
+ print(f\"{c['task_id']:25s} {c['n_classes']:3d} classes {c['test_accuracy']:.1%} {c['name']}\")"
205
+ ```
206
+
207
+ ### Per-capability metadata
208
+
209
+ Each shipped classifier has a `manifest.json` with label names, source dataset, HuggingFace path, stated license, encoder identity, training/test counts, and integrity hashes.
210
+
211
+ ```bash
212
+ curl -s http://localhost:8400/v1/capabilities/banking77 | python3 -m json.tool
213
+ ```
214
+
215
+ ## Train from Python
216
+
217
+ ```python
218
+ from jeffy.train import train_classifier
219
+
220
+ clf = train_classifier(
221
+ texts=["great product!", "terrible service", "fast shipping", "broken on arrival"],
222
+ labels=["positive", "negative", "positive", "negative"],
223
+ task_id="my_reviews",
224
+ )
225
+
226
+ result = clf.predict("the quality exceeded my expectations")
227
+ print(result["label"]) # "positive"
228
+
229
+ clf.save("my_models")
230
+ ```
231
+
232
+ ### Tuning options
233
+
234
+ | Parameter | Default | Description |
235
+ |-----------|---------|-------------|
236
+ | `C` | 0.01 | How aggressively the model fits your data. Low (0.001) = conservative, keeps predictions closer to "I'm not sure." High (1.0) = trusts individual training examples more. If the model is great on training data but bad on new data (overfitting), lower C. |
237
+ | `test_size` | 0.2 | What fraction of your data to hold back for testing. With 100 examples at 0.2, it trains on 80 and tests on 20. Set to 0 to train on everything (useful when you have very little data and will test manually). |
238
+ | `cv_folds` | 3 | Cross-validation: splits your training data into 3 parts, trains on 2 and tests on 1, rotates three times, averages the scores. Gives a more reliable accuracy estimate than a single split. Set to 0 to skip (faster, less reliable estimate). |
239
+
240
+ Start with the defaults. With <50 examples per class, expect noisy estimates.
241
+
242
+ ## Reproduce the evaluation
243
+
244
+ ```bash
245
+ # Install with build dependencies
246
+ uv pip install -e ".[build]"
247
+ # or: pip install -e ".[build]"
248
+
249
+ # Retrain all 13 heads from source datasets (~40 min, downloads ~5 GB)
250
+ jeffy-build --out data/model_pack
251
+
252
+ # Evaluate on held-out test sets with tuned baselines
253
+ jeffy-evaluate --baselines --latency --device cpu
254
+ ```
255
+
256
+ ## Evaluation notes
257
+
258
+ - **SST-2**: Evaluated on `validation` split (official test labels are not public).
259
+ - **SMS Spam**: Random split (test_size=0.2, seed=42); no standard benchmark split.
260
+ - **SNLI**: Input encoded as `premise [SEP] hypothesis`. Label -1 filtered.
261
+ - **CLINC-OOS**: 151 classes including out-of-scope. In-scope accuracy 96.5%, OOS detection 51.7%.
262
+ - **MASSIVE**: English only (config `en`).
263
+
264
+ ## Deployment
265
+
266
+ | Component | Size | Required for |
267
+ |-----------|------|-------------|
268
+ | Jeffy package (wheel) | 1.5 MB | Always (includes all 13 heads) |
269
+ | Encoder (bge-large-en-v1.5) | ~1.2 GB | Inference (downloaded on first use) |
270
+ | `datasets` package | ~100 MB | Retraining from HuggingFace only |
271
+
272
+ Runtime memory: ~2 GB (encoder loaded once, shared across all heads).
273
+
274
+ **Latency** (CPU, single example, Linux aarch64):
275
+
276
+ | Stage | p50 | Notes |
277
+ |-------|-----|-------|
278
+ | Embedding | 50–80 ms | Dominates; varies with input length |
279
+ | Classifier | <1 ms | Negligible |
280
+ | Total | 50–80 ms | End-to-end |
281
+
282
+ ## Model security
283
+
284
+ Bundled pretrained artifacts use numpy `.npz` format (portable, no pickle). Custom-trained models also save a joblib pickle backup. **Only load custom pickle artifacts from trusted sources.** Each artifact's `manifest.json` includes integrity hashes verified on load.
285
+
286
+ ## Provenance and licensing
287
+
288
+ Jeffy code is MIT-licensed. Head artifacts are derived from public datasets; redistribution permissions have **not been independently verified** for all sources. See `ATTRIBUTION.md` for per-dataset license status.
289
+
290
+ | Dataset | Stated license |
291
+ |---------|---------------|
292
+ | banking77, massive_intent, sms_spam | CC BY 4.0 |
293
+ | clinc_oos | CC BY 3.0 |
294
+ | dbpedia | CC BY-SA 3.0 |
295
+ | snli | CC BY-SA 4.0 |
296
+ | ag_news, imdb | Academic / non-commercial |
297
+ | sst2 | Stanford academic license |
298
+ | emotion | Academic |
299
+ | tweet_eval_* | Twitter TOS / academic |
300
+
301
+ The encoder ([bge-large-en-v1.5](https://huggingface.co/BAAI/bge-large-en-v1.5)) is MIT-licensed.
302
+
303
+ ## Tested
304
+
305
+ Verified with clean-environment wheel and sdist install on Linux aarch64, Python 3.12, scikit-learn 1.9+, sentence-transformers 6.1+, numpy 2.5+. Pretrained artifacts use numpy `.npz` format, avoiding sklearn version coupling.
306
+
307
+ ## What's not included
308
+
309
+ - **No zero-shot / general classification.** Each task needs a trained head. Unknown tasks return an error.
310
+ - **No LLM fallback.** This release is pure embedding + classifier.
311
+ - **No automatic task routing.** You must specify which classifier to use.
312
+
313
+ ## Roadmap
314
+
315
+ | Status | Milestone |
316
+ |--------|-----------|
317
+ | **Available** | Pretrained classifier library, SDK/API, playground, custom training from CSV/JSONL |
318
+ | **Next** | Landing page, PyPI release |
319
+ | **Planned** | Broader classifier catalog, released in verified batches |
320
+ | **Planned** | Automatic routing among supported classifiers |
321
+ | **Planned** | Optional local/API LLM fallback for unsupported tasks |
322
+ | **Planned** | Non-text classifiers (game state, sensor data, structured features) |
323
+ | **Exploring** | Assisted labeling, retraining from corrections, classifier sharing |
324
+
325
+ Suggestions for datasets, capabilities, or workflows are welcome as issues.