anthracite-ai 1.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- anthracite_ai-1.5.0/LICENSE +15 -0
- anthracite_ai-1.5.0/MANIFEST.in +3 -0
- anthracite_ai-1.5.0/PKG-INFO +465 -0
- anthracite_ai-1.5.0/README.md +412 -0
- anthracite_ai-1.5.0/anthracite/__init__.py +292 -0
- anthracite_ai-1.5.0/anthracite/architectures/__init__.py +9 -0
- anthracite_ai-1.5.0/anthracite/architectures/anthracite1_embedding.py +231 -0
- anthracite_ai-1.5.0/anthracite/architectures/anthracite1_text.py +309 -0
- anthracite_ai-1.5.0/anthracite/cli.py +161 -0
- anthracite_ai-1.5.0/anthracite/core/__init__.py +4 -0
- anthracite_ai-1.5.0/anthracite/core/config.py +216 -0
- anthracite_ai-1.5.0/anthracite/core/exceptions.py +62 -0
- anthracite_ai-1.5.0/anthracite/core/finetuner.py +292 -0
- anthracite_ai-1.5.0/anthracite/core/registry.py +98 -0
- anthracite_ai-1.5.0/anthracite/core/trainer.py +487 -0
- anthracite_ai-1.5.0/anthracite/datasets/__init__.py +9 -0
- anthracite_ai-1.5.0/anthracite/datasets/hf.py +67 -0
- anthracite_ai-1.5.0/anthracite/datasets/loader.py +139 -0
- anthracite_ai-1.5.0/anthracite/datasets/local.py +235 -0
- anthracite_ai-1.5.0/anthracite/datasets/pairs.py +125 -0
- anthracite_ai-1.5.0/anthracite/datasets/sft.py +307 -0
- anthracite_ai-1.5.0/anthracite/datasets/text.py +142 -0
- anthracite_ai-1.5.0/anthracite/devices/__init__.py +3 -0
- anthracite_ai-1.5.0/anthracite/devices/auto.py +147 -0
- anthracite_ai-1.5.0/anthracite/devices/cpu.py +54 -0
- anthracite_ai-1.5.0/anthracite/devices/cuda.py +69 -0
- anthracite_ai-1.5.0/anthracite/devices/tpu.py +68 -0
- anthracite_ai-1.5.0/anthracite/inference/__init__.py +13 -0
- anthracite_ai-1.5.0/anthracite/inference/embedding.py +90 -0
- anthracite_ai-1.5.0/anthracite/inference/loader.py +86 -0
- anthracite_ai-1.5.0/anthracite/inference/text.py +64 -0
- anthracite_ai-1.5.0/anthracite/interface/__init__.py +3 -0
- anthracite_ai-1.5.0/anthracite/interface/generator.py +215 -0
- anthracite_ai-1.5.0/anthracite/io/__init__.py +4 -0
- anthracite_ai-1.5.0/anthracite/io/config.py +65 -0
- anthracite_ai-1.5.0/anthracite/io/metadata.py +157 -0
- anthracite_ai-1.5.0/anthracite/io/safetensors.py +147 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/__init__.py +5 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/builder.py +218 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/bundled.py +64 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/external.py +265 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/tokenizer.py +230 -0
- anthracite_ai-1.5.0/anthracite/tokenizer/vocabulary.py +91 -0
- anthracite_ai-1.5.0/anthracite/training/__init__.py +18 -0
- anthracite_ai-1.5.0/anthracite/training/checkpoint.py +145 -0
- anthracite_ai-1.5.0/anthracite/training/loop.py +227 -0
- anthracite_ai-1.5.0/anthracite/training/memory.py +165 -0
- anthracite_ai-1.5.0/anthracite/training/optimizer.py +38 -0
- anthracite_ai-1.5.0/anthracite/training/precision.py +94 -0
- anthracite_ai-1.5.0/anthracite/training/scheduler.py +64 -0
- anthracite_ai-1.5.0/anthracite/utils/logging.py +105 -0
- anthracite_ai-1.5.0/anthracite/utils/parameters.py +187 -0
- anthracite_ai-1.5.0/anthracite/utils/progress.py +66 -0
- anthracite_ai-1.5.0/anthracite/utils/seed.py +76 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/PKG-INFO +465 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/SOURCES.txt +61 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/dependency_links.txt +1 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/entry_points.txt +2 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/requires.txt +16 -0
- anthracite_ai-1.5.0/anthracite_ai.egg-info/top_level.txt +1 -0
- anthracite_ai-1.5.0/pyproject.toml +60 -0
- anthracite_ai-1.5.0/setup.cfg +4 -0
- anthracite_ai-1.5.0/tests/test_smoke.py +170 -0
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
Apache License 2.0
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Nebulix Labs
|
|
4
|
+
|
|
5
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
6
|
+
you may not use this file except in compliance with the License.
|
|
7
|
+
You may obtain a copy of the License at
|
|
8
|
+
|
|
9
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
10
|
+
|
|
11
|
+
Unless required by applicable law or agreed to in writing, software
|
|
12
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
13
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
14
|
+
See the License for the specific language governing permissions and
|
|
15
|
+
limitations under the License.
|
|
@@ -0,0 +1,465 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: anthracite-ai
|
|
3
|
+
Version: 1.5.0
|
|
4
|
+
Summary: Universal AI training & fine-tuning framework with smart SFT, external tokenizer support, and HF streaming.
|
|
5
|
+
License: Apache License 2.0
|
|
6
|
+
|
|
7
|
+
Copyright (c) 2026 Nebulix Labs
|
|
8
|
+
|
|
9
|
+
Licensed under the Apache License, Version 2.0 (the "License");
|
|
10
|
+
you may not use this file except in compliance with the License.
|
|
11
|
+
You may obtain a copy of the License at
|
|
12
|
+
|
|
13
|
+
http://www.apache.org/licenses/LICENSE-2.0
|
|
14
|
+
|
|
15
|
+
Unless required by applicable law or agreed to in writing, software
|
|
16
|
+
distributed under the License is distributed on an "AS IS" BASIS,
|
|
17
|
+
WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
|
|
18
|
+
See the License for the specific language governing permissions and
|
|
19
|
+
limitations under the License.
|
|
20
|
+
|
|
21
|
+
Project-URL: Homepage, https://github.com/your-org/anthracite
|
|
22
|
+
Project-URL: Documentation, https://github.com/your-org/anthracite#readme
|
|
23
|
+
Project-URL: Bug Tracker, https://github.com/your-org/anthracite/issues
|
|
24
|
+
Keywords: ai,deep-learning,llm,training,fine-tuning,tokenizer,nlp
|
|
25
|
+
Classifier: Development Status :: 4 - Beta
|
|
26
|
+
Classifier: Intended Audience :: Science/Research
|
|
27
|
+
Classifier: Intended Audience :: Developers
|
|
28
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
29
|
+
Classifier: Programming Language :: Python :: 3
|
|
30
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
31
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
32
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
33
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
34
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
35
|
+
Requires-Python: >=3.9
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
License-File: LICENSE
|
|
38
|
+
Requires-Dist: torch>=2.0
|
|
39
|
+
Requires-Dist: numpy>=1.24
|
|
40
|
+
Requires-Dist: safetensors>=0.4
|
|
41
|
+
Requires-Dist: tqdm>=4.65
|
|
42
|
+
Provides-Extra: hf
|
|
43
|
+
Requires-Dist: datasets>=2.14; extra == "hf"
|
|
44
|
+
Requires-Dist: huggingface_hub>=0.20; extra == "hf"
|
|
45
|
+
Requires-Dist: tokenizers>=0.15; extra == "hf"
|
|
46
|
+
Requires-Dist: transformers>=4.36; extra == "hf"
|
|
47
|
+
Provides-Extra: all
|
|
48
|
+
Requires-Dist: datasets>=2.14; extra == "all"
|
|
49
|
+
Requires-Dist: huggingface_hub>=0.20; extra == "all"
|
|
50
|
+
Requires-Dist: tokenizers>=0.15; extra == "all"
|
|
51
|
+
Requires-Dist: transformers>=4.36; extra == "all"
|
|
52
|
+
Dynamic: license-file
|
|
53
|
+
|
|
54
|
+
# Anthracite v1.5
|
|
55
|
+
|
|
56
|
+
**Universal AI training & fine-tuning framework** — train text generation and embedding models from scratch or fine-tune them with one Python call.
|
|
57
|
+
|
|
58
|
+
---
|
|
59
|
+
|
|
60
|
+
## Table of Contents
|
|
61
|
+
|
|
62
|
+
1. [Installation](#installation)
|
|
63
|
+
2. [Quick Start](#quick-start)
|
|
64
|
+
3. [Tokenizer Options](#tokenizer-options-new-in-v15)
|
|
65
|
+
4. [Supervised Fine-Tuning (SFT)](#supervised-fine-tuning-sft)
|
|
66
|
+
5. [Dataset Formats](#dataset-formats--smart-key-detection)
|
|
67
|
+
6. [HF Dataset Streaming](#hf-dataset-streaming-new-in-v15)
|
|
68
|
+
7. [Embedding Models](#embedding-models)
|
|
69
|
+
8. [Generation & Inference](#generation--inference)
|
|
70
|
+
9. [Fine-Tuning an Existing Model](#fine-tuning-an-existing-model)
|
|
71
|
+
10. [Configuration Reference](#configuration-reference)
|
|
72
|
+
11. [PyPI Upload Guide](#pypi-upload-guide)
|
|
73
|
+
12. [What Changed in v1.5](#whats-new-in-v15)
|
|
74
|
+
|
|
75
|
+
---
|
|
76
|
+
|
|
77
|
+
## Installation
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install anthracite-ai # core (PyTorch required separately)
|
|
81
|
+
pip install anthracite-ai[hf] # + HuggingFace tokenizers, datasets, hub
|
|
82
|
+
pip install anthracite-ai[all] # everything above
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
> **PyTorch** is not bundled (too large). Install it from https://pytorch.org first.
|
|
86
|
+
|
|
87
|
+
---
|
|
88
|
+
|
|
89
|
+
## Quick Start
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from anthracite import train, generate
|
|
93
|
+
|
|
94
|
+
# 1. Train a 20 M parameter text model
|
|
95
|
+
train(
|
|
96
|
+
model_name = "MyGPT-20M",
|
|
97
|
+
dataset = "my_data.jsonl", # local file, HF dataset id, or directory
|
|
98
|
+
tokens = 100_000_000, # training token budget
|
|
99
|
+
params = "20M", # target parameter count
|
|
100
|
+
context_length = 512,
|
|
101
|
+
tokenizer = "bundled", # ← use built-in GPT-2 tokenizer (new v1.5)
|
|
102
|
+
device = "auto",
|
|
103
|
+
precision = "auto",
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
# 2. Generate text
|
|
107
|
+
text = generate("./models/MyGPT-20M", "Once upon a time")
|
|
108
|
+
print(text)
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
---
|
|
112
|
+
|
|
113
|
+
## Tokenizer Options (new in v1.5)
|
|
114
|
+
|
|
115
|
+
Anthracite v1.5 gives you **full control** over the tokenizer. Four modes:
|
|
116
|
+
|
|
117
|
+
### 1. Train from scratch (default, v1.0 behaviour)
|
|
118
|
+
```python
|
|
119
|
+
train(..., tokenizer=None) # trains a byte-level BPE tokenizer on your dataset
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
### 2. Built-in bundled tokenizer
|
|
123
|
+
A GPT-2 style 50 k BPE tokenizer ships with Anthracite. Use it to skip the tokenizer training step:
|
|
124
|
+
```python
|
|
125
|
+
train(..., tokenizer="bundled")
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
### 3. HuggingFace tokenizer by repo id
|
|
129
|
+
```python
|
|
130
|
+
# Load by HF Hub repo id
|
|
131
|
+
train(..., tokenizer="gpt2")
|
|
132
|
+
train(..., tokenizer="mistralai/Mistral-7B-v0.1")
|
|
133
|
+
train(..., tokenizer="meta-llama/Llama-2-7b-hf")
|
|
134
|
+
|
|
135
|
+
# Load a specific file: repo/model/filename
|
|
136
|
+
train(..., tokenizer="bert-base-uncased/tokenizer.json")
|
|
137
|
+
```
|
|
138
|
+
Requires: `pip install transformers` (or `tokenizers` + `huggingface_hub`)
|
|
139
|
+
|
|
140
|
+
### 4. Local tokenizer directory / file
|
|
141
|
+
```python
|
|
142
|
+
train(..., tokenizer="./my_tokenizer/") # directory with tokenizer.json
|
|
143
|
+
train(..., tokenizer="./tokenizer.json") # exact file path
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
### 5. Live tokenizer object (any library)
|
|
147
|
+
```python
|
|
148
|
+
from transformers import AutoTokenizer
|
|
149
|
+
tok = AutoTokenizer.from_pretrained("gpt2")
|
|
150
|
+
train(..., tokenizer=tok)
|
|
151
|
+
|
|
152
|
+
# SentencePiece, tiktoken, tokenizers, etc. all work via duck-typing:
|
|
153
|
+
import tiktoken
|
|
154
|
+
enc = tiktoken.get_encoding("cl100k_base")
|
|
155
|
+
train(..., tokenizer=enc)
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
### Standalone tokenizer loading
|
|
159
|
+
```python
|
|
160
|
+
from anthracite import load_tokenizer
|
|
161
|
+
|
|
162
|
+
tok = load_tokenizer("bundled") # built-in
|
|
163
|
+
tok = load_tokenizer("gpt2") # HF Hub
|
|
164
|
+
tok = load_tokenizer("./my_tokenizer/") # local path
|
|
165
|
+
|
|
166
|
+
ids = tok.encode("Hello world!")
|
|
167
|
+
text = tok.decode(ids)
|
|
168
|
+
print(text) # "Hello world!"
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
---
|
|
172
|
+
|
|
173
|
+
## Supervised Fine-Tuning (SFT)
|
|
174
|
+
|
|
175
|
+
SFT mode trains the model to generate the *output* given the *input*, and masks input tokens from the loss so the model doesn't just learn to repeat the prompt.
|
|
176
|
+
|
|
177
|
+
```python
|
|
178
|
+
from anthracite import train
|
|
179
|
+
|
|
180
|
+
train(
|
|
181
|
+
model_name = "MyChat",
|
|
182
|
+
dataset = "instructions.jsonl",
|
|
183
|
+
tokens = 50_000_000,
|
|
184
|
+
params = "20M",
|
|
185
|
+
objective = "sft", # ← enable SFT mode
|
|
186
|
+
sft_mask_input = True, # mask prompt tokens (default True)
|
|
187
|
+
sft_format = "auto", # auto-detect format (default)
|
|
188
|
+
tokenizer = "bundled",
|
|
189
|
+
)
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
#### Explicit format
|
|
193
|
+
```python
|
|
194
|
+
train(..., sft_format="alpaca") # Alpaca instruction format
|
|
195
|
+
train(..., sft_format="sharegpt") # ShareGPT multi-turn
|
|
196
|
+
train(..., sft_format="chatML") # ChatML messages format
|
|
197
|
+
train(..., sft_format="openai") # prompt / completion
|
|
198
|
+
train(..., sft_format="qa") # question / answer
|
|
199
|
+
train(..., sft_format="text") # raw text, no masking
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
---
|
|
203
|
+
|
|
204
|
+
## Dataset Formats & Smart Key Detection
|
|
205
|
+
|
|
206
|
+
Anthracite v1.5 **automatically detects** your dataset format. You don't need to specify column names — the `SmartKeyExtractor` peeks at a sample of records, scores every known format, and picks the best match.
|
|
207
|
+
|
|
208
|
+
### Supported formats
|
|
209
|
+
|
|
210
|
+
| Format | Keys detected | Input | Output |
|
|
211
|
+
|---|---|---|---|
|
|
212
|
+
| **Alpaca** | `instruction`, `input`, `output` | instruction + input | output |
|
|
213
|
+
| **Alpaca-short** | `instruction`, `output` | instruction | output |
|
|
214
|
+
| **OpenAI** | `prompt`, `completion` | prompt | completion |
|
|
215
|
+
| **Prompt-Response** | `prompt`, `response` | prompt | response |
|
|
216
|
+
| **QA** | `question`, `answer` | question | answer |
|
|
217
|
+
| **Generic IO** | `input`, `output` | input | output |
|
|
218
|
+
| **AI-User** | `user`, `ai` | user | ai |
|
|
219
|
+
| **ShareGPT** | `conversations` list | full conversation | (no mask) |
|
|
220
|
+
| **ChatML** | `messages` list | full conversation | (no mask) |
|
|
221
|
+
| **Raw text** | `text`, `content`, `body` | full text | (no mask) |
|
|
222
|
+
| **2-column generic** | any 2 string keys | first key | second key |
|
|
223
|
+
|
|
224
|
+
### Alpaca format example
|
|
225
|
+
```jsonl
|
|
226
|
+
{"instruction": "Translate to French.", "input": "Hello world", "output": "Bonjour le monde"}
|
|
227
|
+
{"instruction": "Summarise this.", "input": "Long text...", "output": "Short summary."}
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
### ShareGPT / ChatML format example
|
|
231
|
+
```jsonl
|
|
232
|
+
{"conversations": [{"from": "human", "value": "Hi"}, {"from": "gpt", "value": "Hello!"}]}
|
|
233
|
+
{"messages": [{"role": "user", "content": "Hi"}, {"role": "assistant", "content": "Hello!"}]}
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
### OpenAI / prompt-completion format
|
|
237
|
+
```jsonl
|
|
238
|
+
{"prompt": "Tell me a joke.", "completion": "Why did the chicken..."}
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
### Raw text format
|
|
242
|
+
```jsonl
|
|
243
|
+
{"text": "Once upon a time in a land far away..."}
|
|
244
|
+
```
|
|
245
|
+
|
|
246
|
+
---
|
|
247
|
+
|
|
248
|
+
## HF Dataset Streaming (new in v1.5)
|
|
249
|
+
|
|
250
|
+
When you pass a Hugging Face dataset id, Anthracite now **streams** the data instead of downloading the whole dataset first. This means:
|
|
251
|
+
|
|
252
|
+
- Training starts **immediately** (no multi-GB download wait)
|
|
253
|
+
- Large datasets (100 GB+) fit on machines with little disk space
|
|
254
|
+
- Memory usage stays constant regardless of dataset size
|
|
255
|
+
|
|
256
|
+
```python
|
|
257
|
+
train(
|
|
258
|
+
model_name = "MyGPT",
|
|
259
|
+
dataset = "HuggingFaceFW/fineweb", # 15 TB — streams fine!
|
|
260
|
+
tokens = 1_000_000_000,
|
|
261
|
+
params = "100M",
|
|
262
|
+
hf_streaming = True, # default in v1.5
|
|
263
|
+
)
|
|
264
|
+
```
|
|
265
|
+
|
|
266
|
+
To download fully (v1.0 behaviour):
|
|
267
|
+
```python
|
|
268
|
+
train(..., hf_streaming=False)
|
|
269
|
+
```
|
|
270
|
+
|
|
271
|
+
---
|
|
272
|
+
|
|
273
|
+
## Embedding Models
|
|
274
|
+
|
|
275
|
+
```python
|
|
276
|
+
from anthracite import train, embed, similarity
|
|
277
|
+
|
|
278
|
+
# Train a sentence embedding model
|
|
279
|
+
train(
|
|
280
|
+
model_name = "MyEmbedder",
|
|
281
|
+
model_type = "embedding",
|
|
282
|
+
dataset = "corpus.txt",
|
|
283
|
+
tokens = 20_000_000,
|
|
284
|
+
params = "30M",
|
|
285
|
+
context_length = 256,
|
|
286
|
+
objective = "contrastive", # or "mlm" for pre-training stage
|
|
287
|
+
tokenizer = "bundled",
|
|
288
|
+
)
|
|
289
|
+
|
|
290
|
+
# Use it
|
|
291
|
+
vecs = embed("./models/MyEmbedder", ["sentence A", "sentence B"])
|
|
292
|
+
score = similarity(vecs[0], vecs[1])
|
|
293
|
+
print(f"Similarity: {score:.3f}")
|
|
294
|
+
```
|
|
295
|
+
|
|
296
|
+
---
|
|
297
|
+
|
|
298
|
+
## Generation & Inference
|
|
299
|
+
|
|
300
|
+
```python
|
|
301
|
+
from anthracite import generate
|
|
302
|
+
|
|
303
|
+
# Simple generation
|
|
304
|
+
text = generate("./models/MyGPT-20M", "The quick brown fox")
|
|
305
|
+
print(text)
|
|
306
|
+
|
|
307
|
+
# With options
|
|
308
|
+
text = generate(
|
|
309
|
+
"./models/MyGPT-20M",
|
|
310
|
+
"Tell me about space",
|
|
311
|
+
max_new_tokens = 300,
|
|
312
|
+
temperature = 0.8,
|
|
313
|
+
top_p = 0.9,
|
|
314
|
+
)
|
|
315
|
+
```
|
|
316
|
+
|
|
317
|
+
---
|
|
318
|
+
|
|
319
|
+
## Fine-Tuning an Existing Model
|
|
320
|
+
|
|
321
|
+
```python
|
|
322
|
+
from anthracite import finetune
|
|
323
|
+
|
|
324
|
+
finetune(
|
|
325
|
+
model = "./models/MyGPT-20M", # base model path or HF repo id
|
|
326
|
+
dataset = "instructions.jsonl",
|
|
327
|
+
tokens = 10_000_000,
|
|
328
|
+
objective = "sft",
|
|
329
|
+
tokenizer = None, # None = use base model's tokenizer (default)
|
|
330
|
+
output_dir = "./models/MyGPT-20M-SFT",
|
|
331
|
+
)
|
|
332
|
+
```
|
|
333
|
+
|
|
334
|
+
You can also fine-tune from a HuggingFace Hub model:
|
|
335
|
+
```python
|
|
336
|
+
finetune(
|
|
337
|
+
model = "your-username/MyGPT-20M", # HF Hub repo id
|
|
338
|
+
dataset = "instructions.jsonl",
|
|
339
|
+
tokens = 5_000_000,
|
|
340
|
+
)
|
|
341
|
+
```
|
|
342
|
+
|
|
343
|
+
---
|
|
344
|
+
|
|
345
|
+
## Configuration Reference
|
|
346
|
+
|
|
347
|
+
All parameters for `train()` and `finetune()`:
|
|
348
|
+
|
|
349
|
+
| Parameter | Default | Description |
|
|
350
|
+
|---|---|---|
|
|
351
|
+
| `model_name` | `"anthracite-model"` | Output model name / directory |
|
|
352
|
+
| `model_type` | `"text_gen"` | `"text_gen"` or `"embedding"` |
|
|
353
|
+
| `dataset` | — | File path, HF dataset id, or list of strings |
|
|
354
|
+
| `tokens` | `10_000_000` | Training token budget (int or `"100M"`) |
|
|
355
|
+
| `params` | `"20M"` | Target parameters (int or `"20M"`) |
|
|
356
|
+
| `context_length` | `512` | Sequence length |
|
|
357
|
+
| `vocab_size` | `32768` | Vocabulary size (when training tokenizer from scratch) |
|
|
358
|
+
| `tokenizer` | `None` | `None` / `"bundled"` / HF id / path / object |
|
|
359
|
+
| `device` | `"auto"` | `"auto"` / `"cuda"` / `"mps"` / `"cpu"` |
|
|
360
|
+
| `precision` | `"auto"` | `"auto"` / `"fp32"` / `"bf16"` / `"fp16"` |
|
|
361
|
+
| `batch_size` | `"auto"` | Effective batch size or `"auto"` |
|
|
362
|
+
| `learning_rate` | `3e-4` | Peak learning rate |
|
|
363
|
+
| `output_dir` | `./models/<name>` | Where to save the model |
|
|
364
|
+
| `objective` | `"auto"` | `"auto"` / `"causal_lm"` / `"mlm"` / `"contrastive"` / `"sft"` |
|
|
365
|
+
| `sft_mask_input` | `True` | Mask prompt tokens from loss (SFT only) |
|
|
366
|
+
| `sft_format` | `"auto"` | Dataset format hint (SFT only) |
|
|
367
|
+
| `hf_streaming` | `True` | Stream HF datasets (don't download fully) |
|
|
368
|
+
| `checkpoint_interval` | `5000` | Save checkpoint every N steps |
|
|
369
|
+
| `keep_last_checkpoints` | `3` | Number of recent checkpoints to keep |
|
|
370
|
+
| `seed` | `42` | Random seed |
|
|
371
|
+
| `val_split` | `0.01` | Fraction of data held out for validation |
|
|
372
|
+
| `gradient_accumulation` | `None` | Accumulation steps (auto if `None`) |
|
|
373
|
+
| `weight_decay` | `0.1` | AdamW weight decay |
|
|
374
|
+
| `grad_clip` | `1.0` | Gradient clipping norm |
|
|
375
|
+
| `warmup_ratio` | `0.02` | Fraction of steps used for LR warm-up |
|
|
376
|
+
| `num_workers` | `0` | DataLoader worker processes |
|
|
377
|
+
|
|
378
|
+
---
|
|
379
|
+
|
|
380
|
+
## PyPI Upload Guide
|
|
381
|
+
|
|
382
|
+
Follow these steps to publish your own fork or a new release to PyPI.
|
|
383
|
+
|
|
384
|
+
### 1. Install build tools
|
|
385
|
+
```bash
|
|
386
|
+
pip install build twine
|
|
387
|
+
```
|
|
388
|
+
|
|
389
|
+
### 2. Build the package
|
|
390
|
+
```bash
|
|
391
|
+
cd anthracite-pkg # the directory containing pyproject.toml
|
|
392
|
+
python -m build # creates dist/anthracite_ai-1.5.0.tar.gz and .whl
|
|
393
|
+
```
|
|
394
|
+
|
|
395
|
+
### 3. Test on TestPyPI first (recommended)
|
|
396
|
+
```bash
|
|
397
|
+
# Upload to TestPyPI
|
|
398
|
+
twine upload --repository testpypi dist/*
|
|
399
|
+
|
|
400
|
+
# Install from TestPyPI to verify
|
|
401
|
+
pip install --index-url https://test.pypi.org/simple/ anthracite-ai
|
|
402
|
+
```
|
|
403
|
+
|
|
404
|
+
### 4. Upload to PyPI
|
|
405
|
+
```bash
|
|
406
|
+
twine upload dist/*
|
|
407
|
+
# Enter your PyPI username and password (or API token)
|
|
408
|
+
```
|
|
409
|
+
|
|
410
|
+
### 5. Using an API token (safer than password)
|
|
411
|
+
```bash
|
|
412
|
+
# Create a token at https://pypi.org/manage/account/token/
|
|
413
|
+
twine upload dist/* -u __token__ -p pypi-<your-token-here>
|
|
414
|
+
```
|
|
415
|
+
|
|
416
|
+
### 6. Automate with GitHub Actions
|
|
417
|
+
Create `.github/workflows/publish.yml`:
|
|
418
|
+
```yaml
|
|
419
|
+
name: Publish to PyPI
|
|
420
|
+
on:
|
|
421
|
+
release:
|
|
422
|
+
types: [published]
|
|
423
|
+
jobs:
|
|
424
|
+
deploy:
|
|
425
|
+
runs-on: ubuntu-latest
|
|
426
|
+
steps:
|
|
427
|
+
- uses: actions/checkout@v4
|
|
428
|
+
- uses: actions/setup-python@v5
|
|
429
|
+
with: { python-version: "3.11" }
|
|
430
|
+
- run: pip install build twine
|
|
431
|
+
- run: python -m build
|
|
432
|
+
- run: twine upload dist/*
|
|
433
|
+
env:
|
|
434
|
+
TWINE_USERNAME: __token__
|
|
435
|
+
TWINE_PASSWORD: ${{ secrets.PYPI_API_TOKEN }}
|
|
436
|
+
```
|
|
437
|
+
|
|
438
|
+
---
|
|
439
|
+
|
|
440
|
+
## What's New in v1.5
|
|
441
|
+
|
|
442
|
+
### Tokenizer overhaul
|
|
443
|
+
- **`tokenizer="bundled"`** — GPT-2 tokenizer ships with Anthracite, no training needed.
|
|
444
|
+
- **External tokenizer support** — load from HF Hub, local path, or pass any live tokenizer object (HuggingFace, tiktoken, SentencePiece, etc.).
|
|
445
|
+
- **Bug fix** — `encode()` and `decode()` no longer require `self` to be passed manually; the tokenizer works correctly as a standalone object.
|
|
446
|
+
|
|
447
|
+
### Smart SFT format detection (`SmartKeyExtractor`)
|
|
448
|
+
- Automatically detects Alpaca, ShareGPT, ChatML, OpenAI, QA, and 10+ other formats.
|
|
449
|
+
- Correctly identifies `input` (prompt) vs `output` (answer) fields.
|
|
450
|
+
- Masks input tokens from the training loss so the model learns to generate answers, not repeat prompts.
|
|
451
|
+
|
|
452
|
+
### HF Dataset streaming
|
|
453
|
+
- Hugging Face datasets stream by default (`hf_streaming=True`).
|
|
454
|
+
- Training starts immediately without downloading multi-GB datasets.
|
|
455
|
+
- Works with any HF dataset, including terabyte-scale ones.
|
|
456
|
+
|
|
457
|
+
### PyPI-ready packaging
|
|
458
|
+
- `pyproject.toml` with proper metadata, optional dependencies, and entry points.
|
|
459
|
+
- `pip install anthracite-ai` and `pip install anthracite-ai[hf]` work out of the box.
|
|
460
|
+
|
|
461
|
+
---
|
|
462
|
+
|
|
463
|
+
## License
|
|
464
|
+
|
|
465
|
+
MIT License — see `LICENSE` for details.
|