anthracite-ai 1.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (63) hide show
  1. anthracite_ai-1.5.0/LICENSE +15 -0
  2. anthracite_ai-1.5.0/MANIFEST.in +3 -0
  3. anthracite_ai-1.5.0/PKG-INFO +465 -0
  4. anthracite_ai-1.5.0/README.md +412 -0
  5. anthracite_ai-1.5.0/anthracite/__init__.py +292 -0
  6. anthracite_ai-1.5.0/anthracite/architectures/__init__.py +9 -0
  7. anthracite_ai-1.5.0/anthracite/architectures/anthracite1_embedding.py +231 -0
  8. anthracite_ai-1.5.0/anthracite/architectures/anthracite1_text.py +309 -0
  9. anthracite_ai-1.5.0/anthracite/cli.py +161 -0
  10. anthracite_ai-1.5.0/anthracite/core/__init__.py +4 -0
  11. anthracite_ai-1.5.0/anthracite/core/config.py +216 -0
  12. anthracite_ai-1.5.0/anthracite/core/exceptions.py +62 -0
  13. anthracite_ai-1.5.0/anthracite/core/finetuner.py +292 -0
  14. anthracite_ai-1.5.0/anthracite/core/registry.py +98 -0
  15. anthracite_ai-1.5.0/anthracite/core/trainer.py +487 -0
  16. anthracite_ai-1.5.0/anthracite/datasets/__init__.py +9 -0
  17. anthracite_ai-1.5.0/anthracite/datasets/hf.py +67 -0
  18. anthracite_ai-1.5.0/anthracite/datasets/loader.py +139 -0
  19. anthracite_ai-1.5.0/anthracite/datasets/local.py +235 -0
  20. anthracite_ai-1.5.0/anthracite/datasets/pairs.py +125 -0
  21. anthracite_ai-1.5.0/anthracite/datasets/sft.py +307 -0
  22. anthracite_ai-1.5.0/anthracite/datasets/text.py +142 -0
  23. anthracite_ai-1.5.0/anthracite/devices/__init__.py +3 -0
  24. anthracite_ai-1.5.0/anthracite/devices/auto.py +147 -0
  25. anthracite_ai-1.5.0/anthracite/devices/cpu.py +54 -0
  26. anthracite_ai-1.5.0/anthracite/devices/cuda.py +69 -0
  27. anthracite_ai-1.5.0/anthracite/devices/tpu.py +68 -0
  28. anthracite_ai-1.5.0/anthracite/inference/__init__.py +13 -0
  29. anthracite_ai-1.5.0/anthracite/inference/embedding.py +90 -0
  30. anthracite_ai-1.5.0/anthracite/inference/loader.py +86 -0
  31. anthracite_ai-1.5.0/anthracite/inference/text.py +64 -0
  32. anthracite_ai-1.5.0/anthracite/interface/__init__.py +3 -0
  33. anthracite_ai-1.5.0/anthracite/interface/generator.py +215 -0
  34. anthracite_ai-1.5.0/anthracite/io/__init__.py +4 -0
  35. anthracite_ai-1.5.0/anthracite/io/config.py +65 -0
  36. anthracite_ai-1.5.0/anthracite/io/metadata.py +157 -0
  37. anthracite_ai-1.5.0/anthracite/io/safetensors.py +147 -0
  38. anthracite_ai-1.5.0/anthracite/tokenizer/__init__.py +5 -0
  39. anthracite_ai-1.5.0/anthracite/tokenizer/builder.py +218 -0
  40. anthracite_ai-1.5.0/anthracite/tokenizer/bundled.py +64 -0
  41. anthracite_ai-1.5.0/anthracite/tokenizer/external.py +265 -0
  42. anthracite_ai-1.5.0/anthracite/tokenizer/tokenizer.py +230 -0
  43. anthracite_ai-1.5.0/anthracite/tokenizer/vocabulary.py +91 -0
  44. anthracite_ai-1.5.0/anthracite/training/__init__.py +18 -0
  45. anthracite_ai-1.5.0/anthracite/training/checkpoint.py +145 -0
  46. anthracite_ai-1.5.0/anthracite/training/loop.py +227 -0
  47. anthracite_ai-1.5.0/anthracite/training/memory.py +165 -0
  48. anthracite_ai-1.5.0/anthracite/training/optimizer.py +38 -0
  49. anthracite_ai-1.5.0/anthracite/training/precision.py +94 -0
  50. anthracite_ai-1.5.0/anthracite/training/scheduler.py +64 -0
  51. anthracite_ai-1.5.0/anthracite/utils/logging.py +105 -0
  52. anthracite_ai-1.5.0/anthracite/utils/parameters.py +187 -0
  53. anthracite_ai-1.5.0/anthracite/utils/progress.py +66 -0
  54. anthracite_ai-1.5.0/anthracite/utils/seed.py +76 -0
  55. anthracite_ai-1.5.0/anthracite_ai.egg-info/PKG-INFO +465 -0
  56. anthracite_ai-1.5.0/anthracite_ai.egg-info/SOURCES.txt +61 -0
  57. anthracite_ai-1.5.0/anthracite_ai.egg-info/dependency_links.txt +1 -0
  58. anthracite_ai-1.5.0/anthracite_ai.egg-info/entry_points.txt +2 -0
  59. anthracite_ai-1.5.0/anthracite_ai.egg-info/requires.txt +16 -0
  60. anthracite_ai-1.5.0/anthracite_ai.egg-info/top_level.txt +1 -0
  61. anthracite_ai-1.5.0/pyproject.toml +60 -0
  62. anthracite_ai-1.5.0/setup.cfg +4 -0
  63. anthracite_ai-1.5.0/tests/test_smoke.py +170 -0
@@ -0,0 +1,15 @@
1
+ Apache License 2.0
2
+
3
+ Copyright (c) 2026 Nebulix Labs
4
+
5
+ Licensed under the Apache License, Version 2.0 (the "License");
6
+ you may not use this file except in compliance with the License.
7
+ You may obtain a copy of the License at
8
+
9
+ http://www.apache.org/licenses/LICENSE-2.0
10
+
11
+ Unless required by applicable law or agreed to in writing, software
12
+ distributed under the License is distributed on an "AS IS" BASIS,
13
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
+ See the License for the specific language governing permissions and
15
+ limitations under the License.
@@ -0,0 +1,3 @@
1
+ include README.md
2
+ include LICENSE
3
+ recursive-include anthracite *.py *.typed
@@ -0,0 +1,465 @@
1
+ Metadata-Version: 2.4
2
+ Name: anthracite-ai
3
+ Version: 1.5.0
4
+ Summary: Universal AI training & fine-tuning framework with smart SFT, external tokenizer support, and HF streaming.
5
+ License: Apache License 2.0
6
+
7
+ Copyright (c) 2026 Nebulix Labs
8
+
9
+ Licensed under the Apache License, Version 2.0 (the "License");
10
+ you may not use this file except in compliance with the License.
11
+ You may obtain a copy of the License at
12
+
13
+ http://www.apache.org/licenses/LICENSE-2.0
14
+
15
+ Unless required by applicable law or agreed to in writing, software
16
+ distributed under the License is distributed on an "AS IS" BASIS,
17
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
18
+ See the License for the specific language governing permissions and
19
+ limitations under the License.
20
+
21
+ Project-URL: Homepage, https://github.com/your-org/anthracite
22
+ Project-URL: Documentation, https://github.com/your-org/anthracite#readme
23
+ Project-URL: Bug Tracker, https://github.com/your-org/anthracite/issues
24
+ Keywords: ai,deep-learning,llm,training,fine-tuning,tokenizer,nlp
25
+ Classifier: Development Status :: 4 - Beta
26
+ Classifier: Intended Audience :: Science/Research
27
+ Classifier: Intended Audience :: Developers
28
+ Classifier: License :: OSI Approved :: MIT License
29
+ Classifier: Programming Language :: Python :: 3
30
+ Classifier: Programming Language :: Python :: 3.9
31
+ Classifier: Programming Language :: Python :: 3.10
32
+ Classifier: Programming Language :: Python :: 3.11
33
+ Classifier: Programming Language :: Python :: 3.12
34
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
35
+ Requires-Python: >=3.9
36
+ Description-Content-Type: text/markdown
37
+ License-File: LICENSE
38
+ Requires-Dist: torch>=2.0
39
+ Requires-Dist: numpy>=1.24
40
+ Requires-Dist: safetensors>=0.4
41
+ Requires-Dist: tqdm>=4.65
42
+ Provides-Extra: hf
43
+ Requires-Dist: datasets>=2.14; extra == "hf"
44
+ Requires-Dist: huggingface_hub>=0.20; extra == "hf"
45
+ Requires-Dist: tokenizers>=0.15; extra == "hf"
46
+ Requires-Dist: transformers>=4.36; extra == "hf"
47
+ Provides-Extra: all
48
+ Requires-Dist: datasets>=2.14; extra == "all"
49
+ Requires-Dist: huggingface_hub>=0.20; extra == "all"
50
+ Requires-Dist: tokenizers>=0.15; extra == "all"
51
+ Requires-Dist: transformers>=4.36; extra == "all"
52
+ Dynamic: license-file
53
+
54
+ # Anthracite v1.5
55
+
56
+ **Universal AI training & fine-tuning framework** — train text generation and embedding models from scratch or fine-tune them with one Python call.
57
+
58
+ ---
59
+
60
+ ## Table of Contents
61
+
62
+ 1. [Installation](#installation)
63
+ 2. [Quick Start](#quick-start)
64
+ 3. [Tokenizer Options](#tokenizer-options-new-in-v15)
65
+ 4. [Supervised Fine-Tuning (SFT)](#supervised-fine-tuning-sft)
66
+ 5. [Dataset Formats](#dataset-formats--smart-key-detection)
67
+ 6. [HF Dataset Streaming](#hf-dataset-streaming-new-in-v15)
68
+ 7. [Embedding Models](#embedding-models)
69
+ 8. [Generation & Inference](#generation--inference)
70
+ 9. [Fine-Tuning an Existing Model](#fine-tuning-an-existing-model)
71
+ 10. [Configuration Reference](#configuration-reference)
72
+ 11. [PyPI Upload Guide](#pypi-upload-guide)
73
+ 12. [What Changed in v1.5](#whats-new-in-v15)
74
+
75
+ ---
76
+
77
+ ## Installation
78
+
79
+ ```bash
80
+ pip install anthracite-ai # core (PyTorch required separately)
81
+ pip install anthracite-ai[hf] # + HuggingFace tokenizers, datasets, hub
82
+ pip install anthracite-ai[all] # everything above
83
+ ```
84
+
85
+ > **PyTorch** is not bundled (too large). Install it from https://pytorch.org first.
86
+
87
+ ---
88
+
89
+ ## Quick Start
90
+
91
+ ```python
92
+ from anthracite import train, generate
93
+
94
+ # 1. Train a 20 M parameter text model
95
+ train(
96
+ model_name = "MyGPT-20M",
97
+ dataset = "my_data.jsonl", # local file, HF dataset id, or directory
98
+ tokens = 100_000_000, # training token budget
99
+ params = "20M", # target parameter count
100
+ context_length = 512,
101
+ tokenizer = "bundled", # ← use built-in GPT-2 tokenizer (new v1.5)
102
+ device = "auto",
103
+ precision = "auto",
104
+ )
105
+
106
+ # 2. Generate text
107
+ text = generate("./models/MyGPT-20M", "Once upon a time")
108
+ print(text)
109
+ ```
110
+
111
+ ---
112
+
113
+ ## Tokenizer Options (new in v1.5)
114
+
115
+ Anthracite v1.5 gives you **full control** over the tokenizer. Four modes:
116
+
117
+ ### 1. Train from scratch (default, v1.0 behaviour)
118
+ ```python
119
+ train(..., tokenizer=None) # trains a byte-level BPE tokenizer on your dataset
120
+ ```
121
+
122
+ ### 2. Built-in bundled tokenizer
123
+ A GPT-2 style 50 k BPE tokenizer ships with Anthracite. Use it to skip the tokenizer training step:
124
+ ```python
125
+ train(..., tokenizer="bundled")
126
+ ```
127
+
128
+ ### 3. HuggingFace tokenizer by repo id
129
+ ```python
130
+ # Load by HF Hub repo id
131
+ train(..., tokenizer="gpt2")
132
+ train(..., tokenizer="mistralai/Mistral-7B-v0.1")
133
+ train(..., tokenizer="meta-llama/Llama-2-7b-hf")
134
+
135
+ # Load a specific file: repo/model/filename
136
+ train(..., tokenizer="bert-base-uncased/tokenizer.json")
137
+ ```
138
+ Requires: `pip install transformers` (or `tokenizers` + `huggingface_hub`)
139
+
140
+ ### 4. Local tokenizer directory / file
141
+ ```python
142
+ train(..., tokenizer="./my_tokenizer/") # directory with tokenizer.json
143
+ train(..., tokenizer="./tokenizer.json") # exact file path
144
+ ```
145
+
146
+ ### 5. Live tokenizer object (any library)
147
+ ```python
148
+ from transformers import AutoTokenizer
149
+ tok = AutoTokenizer.from_pretrained("gpt2")
150
+ train(..., tokenizer=tok)
151
+
152
+ # SentencePiece, tiktoken, tokenizers, etc. all work via duck-typing:
153
+ import tiktoken
154
+ enc = tiktoken.get_encoding("cl100k_base")
155
+ train(..., tokenizer=enc)
156
+ ```
157
+
158
+ ### Standalone tokenizer loading
159
+ ```python
160
+ from anthracite import load_tokenizer
161
+
162
+ tok = load_tokenizer("bundled") # built-in
163
+ tok = load_tokenizer("gpt2") # HF Hub
164
+ tok = load_tokenizer("./my_tokenizer/") # local path
165
+
166
+ ids = tok.encode("Hello world!")
167
+ text = tok.decode(ids)
168
+ print(text) # "Hello world!"
169
+ ```
170
+
171
+ ---
172
+
173
+ ## Supervised Fine-Tuning (SFT)
174
+
175
+ SFT mode trains the model to generate the *output* given the *input*, and masks input tokens from the loss so the model doesn't just learn to repeat the prompt.
176
+
177
+ ```python
178
+ from anthracite import train
179
+
180
+ train(
181
+ model_name = "MyChat",
182
+ dataset = "instructions.jsonl",
183
+ tokens = 50_000_000,
184
+ params = "20M",
185
+ objective = "sft", # ← enable SFT mode
186
+ sft_mask_input = True, # mask prompt tokens (default True)
187
+ sft_format = "auto", # auto-detect format (default)
188
+ tokenizer = "bundled",
189
+ )
190
+ ```
191
+
192
+ #### Explicit format
193
+ ```python
194
+ train(..., sft_format="alpaca") # Alpaca instruction format
195
+ train(..., sft_format="sharegpt") # ShareGPT multi-turn
196
+ train(..., sft_format="chatML") # ChatML messages format
197
+ train(..., sft_format="openai") # prompt / completion
198
+ train(..., sft_format="qa") # question / answer
199
+ train(..., sft_format="text") # raw text, no masking
200
+ ```
201
+
202
+ ---
203
+
204
+ ## Dataset Formats & Smart Key Detection
205
+
206
+ Anthracite v1.5 **automatically detects** your dataset format. You don't need to specify column names — the `SmartKeyExtractor` peeks at a sample of records, scores every known format, and picks the best match.
207
+
208
+ ### Supported formats
209
+
210
+ | Format | Keys detected | Input | Output |
211
+ |---|---|---|---|
212
+ | **Alpaca** | `instruction`, `input`, `output` | instruction + input | output |
213
+ | **Alpaca-short** | `instruction`, `output` | instruction | output |
214
+ | **OpenAI** | `prompt`, `completion` | prompt | completion |
215
+ | **Prompt-Response** | `prompt`, `response` | prompt | response |
216
+ | **QA** | `question`, `answer` | question | answer |
217
+ | **Generic IO** | `input`, `output` | input | output |
218
+ | **AI-User** | `user`, `ai` | user | ai |
219
+ | **ShareGPT** | `conversations` list | full conversation | (no mask) |
220
+ | **ChatML** | `messages` list | full conversation | (no mask) |
221
+ | **Raw text** | `text`, `content`, `body` | full text | (no mask) |
222
+ | **2-column generic** | any 2 string keys | first key | second key |
223
+
224
+ ### Alpaca format example
225
+ ```jsonl
226
+ {"instruction": "Translate to French.", "input": "Hello world", "output": "Bonjour le monde"}
227
+ {"instruction": "Summarise this.", "input": "Long text...", "output": "Short summary."}
228
+ ```
229
+
230
+ ### ShareGPT / ChatML format example
231
+ ```jsonl
232
+ {"conversations": [{"from": "human", "value": "Hi"}, {"from": "gpt", "value": "Hello!"}]}
233
+ {"messages": [{"role": "user", "content": "Hi"}, {"role": "assistant", "content": "Hello!"}]}
234
+ ```
235
+
236
+ ### OpenAI / prompt-completion format
237
+ ```jsonl
238
+ {"prompt": "Tell me a joke.", "completion": "Why did the chicken..."}
239
+ ```
240
+
241
+ ### Raw text format
242
+ ```jsonl
243
+ {"text": "Once upon a time in a land far away..."}
244
+ ```
245
+
246
+ ---
247
+
248
+ ## HF Dataset Streaming (new in v1.5)
249
+
250
+ When you pass a Hugging Face dataset id, Anthracite now **streams** the data instead of downloading the whole dataset first. This means:
251
+
252
+ - Training starts **immediately** (no multi-GB download wait)
253
+ - Large datasets (100 GB+) fit on machines with little disk space
254
+ - Memory usage stays constant regardless of dataset size
255
+
256
+ ```python
257
+ train(
258
+ model_name = "MyGPT",
259
+ dataset = "HuggingFaceFW/fineweb", # 15 TB — streams fine!
260
+ tokens = 1_000_000_000,
261
+ params = "100M",
262
+ hf_streaming = True, # default in v1.5
263
+ )
264
+ ```
265
+
266
+ To download fully (v1.0 behaviour):
267
+ ```python
268
+ train(..., hf_streaming=False)
269
+ ```
270
+
271
+ ---
272
+
273
+ ## Embedding Models
274
+
275
+ ```python
276
+ from anthracite import train, embed, similarity
277
+
278
+ # Train a sentence embedding model
279
+ train(
280
+ model_name = "MyEmbedder",
281
+ model_type = "embedding",
282
+ dataset = "corpus.txt",
283
+ tokens = 20_000_000,
284
+ params = "30M",
285
+ context_length = 256,
286
+ objective = "contrastive", # or "mlm" for pre-training stage
287
+ tokenizer = "bundled",
288
+ )
289
+
290
+ # Use it
291
+ vecs = embed("./models/MyEmbedder", ["sentence A", "sentence B"])
292
+ score = similarity(vecs[0], vecs[1])
293
+ print(f"Similarity: {score:.3f}")
294
+ ```
295
+
296
+ ---
297
+
298
+ ## Generation & Inference
299
+
300
+ ```python
301
+ from anthracite import generate
302
+
303
+ # Simple generation
304
+ text = generate("./models/MyGPT-20M", "The quick brown fox")
305
+ print(text)
306
+
307
+ # With options
308
+ text = generate(
309
+ "./models/MyGPT-20M",
310
+ "Tell me about space",
311
+ max_new_tokens = 300,
312
+ temperature = 0.8,
313
+ top_p = 0.9,
314
+ )
315
+ ```
316
+
317
+ ---
318
+
319
+ ## Fine-Tuning an Existing Model
320
+
321
+ ```python
322
+ from anthracite import finetune
323
+
324
+ finetune(
325
+ model = "./models/MyGPT-20M", # base model path or HF repo id
326
+ dataset = "instructions.jsonl",
327
+ tokens = 10_000_000,
328
+ objective = "sft",
329
+ tokenizer = None, # None = use base model's tokenizer (default)
330
+ output_dir = "./models/MyGPT-20M-SFT",
331
+ )
332
+ ```
333
+
334
+ You can also fine-tune from a HuggingFace Hub model:
335
+ ```python
336
+ finetune(
337
+ model = "your-username/MyGPT-20M", # HF Hub repo id
338
+ dataset = "instructions.jsonl",
339
+ tokens = 5_000_000,
340
+ )
341
+ ```
342
+
343
+ ---
344
+
345
+ ## Configuration Reference
346
+
347
+ All parameters for `train()` and `finetune()`:
348
+
349
+ | Parameter | Default | Description |
350
+ |---|---|---|
351
+ | `model_name` | `"anthracite-model"` | Output model name / directory |
352
+ | `model_type` | `"text_gen"` | `"text_gen"` or `"embedding"` |
353
+ | `dataset` | — | File path, HF dataset id, or list of strings |
354
+ | `tokens` | `10_000_000` | Training token budget (int or `"100M"`) |
355
+ | `params` | `"20M"` | Target parameters (int or `"20M"`) |
356
+ | `context_length` | `512` | Sequence length |
357
+ | `vocab_size` | `32768` | Vocabulary size (when training tokenizer from scratch) |
358
+ | `tokenizer` | `None` | `None` / `"bundled"` / HF id / path / object |
359
+ | `device` | `"auto"` | `"auto"` / `"cuda"` / `"mps"` / `"cpu"` |
360
+ | `precision` | `"auto"` | `"auto"` / `"fp32"` / `"bf16"` / `"fp16"` |
361
+ | `batch_size` | `"auto"` | Effective batch size or `"auto"` |
362
+ | `learning_rate` | `3e-4` | Peak learning rate |
363
+ | `output_dir` | `./models/<name>` | Where to save the model |
364
+ | `objective` | `"auto"` | `"auto"` / `"causal_lm"` / `"mlm"` / `"contrastive"` / `"sft"` |
365
+ | `sft_mask_input` | `True` | Mask prompt tokens from loss (SFT only) |
366
+ | `sft_format` | `"auto"` | Dataset format hint (SFT only) |
367
+ | `hf_streaming` | `True` | Stream HF datasets (don't download fully) |
368
+ | `checkpoint_interval` | `5000` | Save checkpoint every N steps |
369
+ | `keep_last_checkpoints` | `3` | Number of recent checkpoints to keep |
370
+ | `seed` | `42` | Random seed |
371
+ | `val_split` | `0.01` | Fraction of data held out for validation |
372
+ | `gradient_accumulation` | `None` | Accumulation steps (auto if `None`) |
373
+ | `weight_decay` | `0.1` | AdamW weight decay |
374
+ | `grad_clip` | `1.0` | Gradient clipping norm |
375
+ | `warmup_ratio` | `0.02` | Fraction of steps used for LR warm-up |
376
+ | `num_workers` | `0` | DataLoader worker processes |
377
+
378
+ ---
379
+
380
+ ## PyPI Upload Guide
381
+
382
+ Follow these steps to publish your own fork or a new release to PyPI.
383
+
384
+ ### 1. Install build tools
385
+ ```bash
386
+ pip install build twine
387
+ ```
388
+
389
+ ### 2. Build the package
390
+ ```bash
391
+ cd anthracite-pkg # the directory containing pyproject.toml
392
+ python -m build # creates dist/anthracite_ai-1.5.0.tar.gz and .whl
393
+ ```
394
+
395
+ ### 3. Test on TestPyPI first (recommended)
396
+ ```bash
397
+ # Upload to TestPyPI
398
+ twine upload --repository testpypi dist/*
399
+
400
+ # Install from TestPyPI to verify
401
+ pip install --index-url https://test.pypi.org/simple/ anthracite-ai
402
+ ```
403
+
404
+ ### 4. Upload to PyPI
405
+ ```bash
406
+ twine upload dist/*
407
+ # Enter your PyPI username and password (or API token)
408
+ ```
409
+
410
+ ### 5. Using an API token (safer than password)
411
+ ```bash
412
+ # Create a token at https://pypi.org/manage/account/token/
413
+ twine upload dist/* -u __token__ -p pypi-<your-token-here>
414
+ ```
415
+
416
+ ### 6. Automate with GitHub Actions
417
+ Create `.github/workflows/publish.yml`:
418
+ ```yaml
419
+ name: Publish to PyPI
420
+ on:
421
+ release:
422
+ types: [published]
423
+ jobs:
424
+ deploy:
425
+ runs-on: ubuntu-latest
426
+ steps:
427
+ - uses: actions/checkout@v4
428
+ - uses: actions/setup-python@v5
429
+ with: { python-version: "3.11" }
430
+ - run: pip install build twine
431
+ - run: python -m build
432
+ - run: twine upload dist/*
433
+ env:
434
+ TWINE_USERNAME: __token__
435
+ TWINE_PASSWORD: ${{ secrets.PYPI_API_TOKEN }}
436
+ ```
437
+
438
+ ---
439
+
440
+ ## What's New in v1.5
441
+
442
+ ### Tokenizer overhaul
443
+ - **`tokenizer="bundled"`** — GPT-2 tokenizer ships with Anthracite, no training needed.
444
+ - **External tokenizer support** — load from HF Hub, local path, or pass any live tokenizer object (HuggingFace, tiktoken, SentencePiece, etc.).
445
+ - **Bug fix** — `encode()` and `decode()` no longer require `self` to be passed manually; the tokenizer works correctly as a standalone object.
446
+
447
+ ### Smart SFT format detection (`SmartKeyExtractor`)
448
+ - Automatically detects Alpaca, ShareGPT, ChatML, OpenAI, QA, and 10+ other formats.
449
+ - Correctly identifies `input` (prompt) vs `output` (answer) fields.
450
+ - Masks input tokens from the training loss so the model learns to generate answers, not repeat prompts.
451
+
452
+ ### HF Dataset streaming
453
+ - Hugging Face datasets stream by default (`hf_streaming=True`).
454
+ - Training starts immediately without downloading multi-GB datasets.
455
+ - Works with any HF dataset, including terabyte-scale ones.
456
+
457
+ ### PyPI-ready packaging
458
+ - `pyproject.toml` with proper metadata, optional dependencies, and entry points.
459
+ - `pip install anthracite-ai` and `pip install anthracite-ai[hf]` work out of the box.
460
+
461
+ ---
462
+
463
+ ## License
464
+
465
+ MIT License — see `LICENSE` for details.