bertuner 0.2.2__tar.gz → 0.2.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {bertuner-0.2.2 → bertuner-0.2.3}/PKG-INFO +15 -1
- {bertuner-0.2.2 → bertuner-0.2.3}/README.md +14 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/BERTuner.py +24 -13
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/CustomTrainer.py +3 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/__init__.py +1 -1
- bertuner-0.2.3/bertuner/compat.py +15 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/PKG-INFO +15 -1
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/SOURCES.txt +2 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/pyproject.toml +1 -1
- {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_bertuner.py +3 -2
- bertuner-0.2.3/tests/test_transformers_compat.py +88 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/LICENSE +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/Predictor.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/TensorBoardCallback.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/constants.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/exceptions.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/utils.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/dependency_links.txt +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/requires.txt +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/top_level.txt +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/setup.cfg +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_numerical_stability.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_predictor.py +0 -0
- {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_utils.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bertuner
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support
|
|
5
5
|
Author-email: elemets <alafunnell@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -58,6 +58,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
|
|
|
58
58
|
pip install -r requirements.txt
|
|
59
59
|
```
|
|
60
60
|
|
|
61
|
+
BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
|
|
62
|
+
4.48 baseline and the latest available 4.x and 5.x releases, including final
|
|
63
|
+
training, TensorBoard logging, warmup scheduling, and gradient accumulation.
|
|
64
|
+
Future releases are checked by a weekly CI run rather than assumed compatible.
|
|
65
|
+
|
|
66
|
+
When initializing a classifier from a base pretrained model, a Transformers 5
|
|
67
|
+
load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
|
|
68
|
+
as `MISSING`. This is expected when replacing the language-model head with a
|
|
69
|
+
classification head; the new head is learned during fine-tuning. Unexpected
|
|
70
|
+
encoder weights or shape mismatches should still be investigated.
|
|
71
|
+
|
|
72
|
+
Run the test suite with `python -m pytest tests -q`. Training regression tests
|
|
73
|
+
use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
|
|
74
|
+
|
|
61
75
|
MLflow tracking works in two modes:
|
|
62
76
|
|
|
63
77
|
```bash
|
|
@@ -17,6 +17,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
|
|
|
17
17
|
pip install -r requirements.txt
|
|
18
18
|
```
|
|
19
19
|
|
|
20
|
+
BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
|
|
21
|
+
4.48 baseline and the latest available 4.x and 5.x releases, including final
|
|
22
|
+
training, TensorBoard logging, warmup scheduling, and gradient accumulation.
|
|
23
|
+
Future releases are checked by a weekly CI run rather than assumed compatible.
|
|
24
|
+
|
|
25
|
+
When initializing a classifier from a base pretrained model, a Transformers 5
|
|
26
|
+
load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
|
|
27
|
+
as `MISSING`. This is expected when replacing the language-model head with a
|
|
28
|
+
classification head; the new head is learned during fine-tuning. Unexpected
|
|
29
|
+
encoder weights or shape mismatches should still be investigated.
|
|
30
|
+
|
|
31
|
+
Run the test suite with `python -m pytest tests -q`. Training regression tests
|
|
32
|
+
use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
|
|
33
|
+
|
|
20
34
|
MLflow tracking works in two modes:
|
|
21
35
|
|
|
22
36
|
```bash
|
|
@@ -15,7 +15,6 @@ from transformers import (
|
|
|
15
15
|
AutoTokenizer,
|
|
16
16
|
AutoConfig,
|
|
17
17
|
AutoModelForSequenceClassification,
|
|
18
|
-
TrainingArguments,
|
|
19
18
|
DataCollatorWithPadding,
|
|
20
19
|
EarlyStoppingCallback,
|
|
21
20
|
set_seed,
|
|
@@ -38,6 +37,7 @@ from sklearn.preprocessing import label_binarize
|
|
|
38
37
|
from mlflow.tracking import MlflowClient
|
|
39
38
|
|
|
40
39
|
from bertuner.CustomTrainer import CustomTrainer
|
|
40
|
+
from bertuner.compat import TrainingArguments
|
|
41
41
|
from bertuner.exceptions import NonFiniteTrainingError, NoStableTrialError
|
|
42
42
|
from bertuner.TensorBoardCallback import (
|
|
43
43
|
TensorBoardSyncCallback,
|
|
@@ -56,6 +56,7 @@ from bertuner.constants import (
|
|
|
56
56
|
MODEL_DROPOUT_ATTRS,
|
|
57
57
|
SEED,
|
|
58
58
|
)
|
|
59
|
+
import inspect
|
|
59
60
|
|
|
60
61
|
|
|
61
62
|
class BERTuneClassifier:
|
|
@@ -746,7 +747,6 @@ class BERTuneClassifier:
|
|
|
746
747
|
"gradient_checkpointing": self._use_gradient_checkpointing(max_length),
|
|
747
748
|
"gradient_checkpointing_kwargs": {"use_reentrant": False},
|
|
748
749
|
"weight_decay": params["weight_decay"],
|
|
749
|
-
"warmup_ratio": params["warmup_ratio"],
|
|
750
750
|
"metric_for_best_model": f"eval_{self.optimize_metric}",
|
|
751
751
|
"greater_is_better": self.greater_is_better,
|
|
752
752
|
"eval_strategy": "epoch",
|
|
@@ -760,17 +760,20 @@ class BERTuneClassifier:
|
|
|
760
760
|
"seed": self.seed,
|
|
761
761
|
**self._precision_flags(precision),
|
|
762
762
|
}
|
|
763
|
-
|
|
764
|
-
|
|
765
|
-
|
|
766
|
-
|
|
767
|
-
else
|
|
768
|
-
|
|
769
|
-
|
|
770
|
-
|
|
771
|
-
|
|
772
|
-
|
|
773
|
-
|
|
763
|
+
|
|
764
|
+
_TA_PARAMS = inspect.signature(TrainingArguments.__init__).parameters
|
|
765
|
+
|
|
766
|
+
# v5 removed warmup_ratio and accepts fractions in warmup_steps.
|
|
767
|
+
warmup_key = "warmup_ratio" if "warmup_ratio" in _TA_PARAMS else "warmup_steps"
|
|
768
|
+
kwargs[warmup_key] = params["warmup_ratio"]
|
|
769
|
+
kwargs["lr_scheduler_type"] = params["scheduler"]
|
|
770
|
+
# Final training uses an explicit TensorBoard writer in _build_trainer;
|
|
771
|
+
# newer Transformers versions removed TrainingArguments.logging_dir.
|
|
772
|
+
kwargs.update(remove_unused_columns=True, report_to=["none"])
|
|
773
|
+
args = TrainingArguments(**kwargs)
|
|
774
|
+
if warmup_key == "warmup_steps":
|
|
775
|
+
args._bertuner_warmup_ratio = params["warmup_ratio"]
|
|
776
|
+
return args
|
|
774
777
|
|
|
775
778
|
def _build_trainer(
|
|
776
779
|
self,
|
|
@@ -792,8 +795,12 @@ class BERTuneClassifier:
|
|
|
792
795
|
)
|
|
793
796
|
]
|
|
794
797
|
if final:
|
|
798
|
+
from transformers.integrations import TensorBoardCallback
|
|
799
|
+
from torch.utils.tensorboard import SummaryWriter
|
|
800
|
+
|
|
795
801
|
callbacks.extend(
|
|
796
802
|
[
|
|
803
|
+
TensorBoardCallback(tb_writer=SummaryWriter(logging_dir)),
|
|
797
804
|
TensorBoardSyncCallback(logging_dir),
|
|
798
805
|
CleanupCheckpointsCallback,
|
|
799
806
|
]
|
|
@@ -857,11 +864,15 @@ class BERTuneClassifier:
|
|
|
857
864
|
try:
|
|
858
865
|
trainer.train()
|
|
859
866
|
except Exception:
|
|
867
|
+
from transformers.integrations import TensorBoardCallback
|
|
868
|
+
|
|
860
869
|
# Trainer does not emit on_train_end after an exception. Close only
|
|
861
870
|
# BERTuner-owned writers before the caller decides whether to retry.
|
|
862
871
|
for callback in trainer.callback_handler.callbacks:
|
|
863
872
|
if isinstance(callback, TensorBoardSyncCallback):
|
|
864
873
|
callback.writer.close()
|
|
874
|
+
elif isinstance(callback, TensorBoardCallback) and callback.tb_writer is not None:
|
|
875
|
+
callback.tb_writer.close()
|
|
865
876
|
raise
|
|
866
877
|
return trainer, model
|
|
867
878
|
|
|
@@ -69,6 +69,9 @@ class CustomTrainer(Trainer):
|
|
|
69
69
|
callbacks = list(kwargs.pop("callbacks", None) or [])
|
|
70
70
|
callbacks.append(NonFiniteGradientCallback(training_precision))
|
|
71
71
|
super().__init__(callbacks=callbacks, **kwargs)
|
|
72
|
+
# compute_loss returns a microbatch mean and does not consume
|
|
73
|
+
# num_items_in_batch, even when the model's forward accepts **kwargs.
|
|
74
|
+
self.model_accepts_loss_kwargs = False
|
|
72
75
|
self.loss_type = loss_type
|
|
73
76
|
self.class_weights = class_weights
|
|
74
77
|
self.training_precision = training_precision
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
"""Small adapters for Transformers APIs shared by supported 4.x and 5.x."""
|
|
2
|
+
|
|
3
|
+
import math
|
|
4
|
+
|
|
5
|
+
from transformers import TrainingArguments as HFTrainingArguments
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class TrainingArguments(HFTrainingArguments):
|
|
9
|
+
def get_warmup_steps(self, num_training_steps):
|
|
10
|
+
ratio = getattr(self, "_bertuner_warmup_ratio", None)
|
|
11
|
+
if ratio is not None:
|
|
12
|
+
# New warmup_steps treats 1.0 as one step, whereas the old
|
|
13
|
+
# warmup_ratio treats it as the entire training run.
|
|
14
|
+
return math.ceil(num_training_steps * ratio)
|
|
15
|
+
return super().get_warmup_steps(num_training_steps)
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: bertuner
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.3
|
|
4
4
|
Summary: Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support
|
|
5
5
|
Author-email: elemets <alafunnell@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -58,6 +58,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
|
|
|
58
58
|
pip install -r requirements.txt
|
|
59
59
|
```
|
|
60
60
|
|
|
61
|
+
BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
|
|
62
|
+
4.48 baseline and the latest available 4.x and 5.x releases, including final
|
|
63
|
+
training, TensorBoard logging, warmup scheduling, and gradient accumulation.
|
|
64
|
+
Future releases are checked by a weekly CI run rather than assumed compatible.
|
|
65
|
+
|
|
66
|
+
When initializing a classifier from a base pretrained model, a Transformers 5
|
|
67
|
+
load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
|
|
68
|
+
as `MISSING`. This is expected when replacing the language-model head with a
|
|
69
|
+
classification head; the new head is learned during fine-tuning. Unexpected
|
|
70
|
+
encoder weights or shape mismatches should still be investigated.
|
|
71
|
+
|
|
72
|
+
Run the test suite with `python -m pytest tests -q`. Training regression tests
|
|
73
|
+
use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
|
|
74
|
+
|
|
61
75
|
MLflow tracking works in two modes:
|
|
62
76
|
|
|
63
77
|
```bash
|
|
@@ -6,6 +6,7 @@ bertuner/CustomTrainer.py
|
|
|
6
6
|
bertuner/Predictor.py
|
|
7
7
|
bertuner/TensorBoardCallback.py
|
|
8
8
|
bertuner/__init__.py
|
|
9
|
+
bertuner/compat.py
|
|
9
10
|
bertuner/constants.py
|
|
10
11
|
bertuner/exceptions.py
|
|
11
12
|
bertuner/utils.py
|
|
@@ -17,4 +18,5 @@ bertuner.egg-info/top_level.txt
|
|
|
17
18
|
tests/test_bertuner.py
|
|
18
19
|
tests/test_numerical_stability.py
|
|
19
20
|
tests/test_predictor.py
|
|
21
|
+
tests/test_transformers_compat.py
|
|
20
22
|
tests/test_utils.py
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "bertuner"
|
|
7
|
-
version = "0.2.
|
|
7
|
+
version = "0.2.3"
|
|
8
8
|
description = "Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -483,6 +483,7 @@ class TestLossCurveLogging:
|
|
|
483
483
|
]
|
|
484
484
|
)
|
|
485
485
|
|
|
486
|
+
(tmp_path / "downloaded-artifacts").mkdir()
|
|
486
487
|
artifact_path = MlflowClient().download_artifacts(
|
|
487
488
|
run.info.run_id,
|
|
488
489
|
"plots/training_vs_evaluation_loss.png",
|
|
@@ -785,7 +786,7 @@ class TestMulticlassWeightedLoss:
|
|
|
785
786
|
loss = trainer._singlelabel_loss(logits, labels, torch.device("cpu"))
|
|
786
787
|
assert torch.isfinite(loss)
|
|
787
788
|
|
|
788
|
-
def test_end_to_end_training_three_classes(self, tmp_path):
|
|
789
|
+
def test_end_to_end_training_three_classes(self, tmp_path, tiny_model_path):
|
|
789
790
|
import torch
|
|
790
791
|
from transformers import (
|
|
791
792
|
AutoModelForSequenceClassification,
|
|
@@ -795,7 +796,7 @@ class TestMulticlassWeightedLoss:
|
|
|
795
796
|
)
|
|
796
797
|
from bertuner.CustomTrainer import CustomTrainer
|
|
797
798
|
|
|
798
|
-
model_path =
|
|
799
|
+
model_path = tiny_model_path
|
|
799
800
|
clf = make_classifier(tmp_path, dataframe=make_df(n=60, num_classes=3))
|
|
800
801
|
assert clf.num_labels == 3
|
|
801
802
|
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""Run unchanged against the oldest supported Transformers and current releases."""
|
|
2
|
+
import copy
|
|
3
|
+
from types import SimpleNamespace
|
|
4
|
+
|
|
5
|
+
import numpy as np
|
|
6
|
+
import pytest
|
|
7
|
+
import torch
|
|
8
|
+
from datasets import Dataset
|
|
9
|
+
from tensorboard.backend.event_processing.event_accumulator import EventAccumulator
|
|
10
|
+
from transformers import AutoTokenizer, TrainingArguments
|
|
11
|
+
|
|
12
|
+
from bertuner.CustomTrainer import CustomTrainer
|
|
13
|
+
from test_bertuner import make_classifier, make_df
|
|
14
|
+
from test_numerical_stability import sampled_params
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@pytest.mark.parametrize("ratio", [0.0, 0.1, 1.0])
|
|
18
|
+
@pytest.mark.parametrize("final", [False, True])
|
|
19
|
+
def test_warmup_and_scheduler(tmp_path, ratio, final):
|
|
20
|
+
clf = make_classifier(tmp_path)
|
|
21
|
+
clf.optimize_metric = "avg_precision"
|
|
22
|
+
clf.greater_is_better = True
|
|
23
|
+
params = dict(sampled_params(), warmup_ratio=ratio, scheduler="cosine")
|
|
24
|
+
args = clf._build_training_arguments(
|
|
25
|
+
params, str(tmp_path / "output"), 32, "fp32", final=final,
|
|
26
|
+
logging_dir=str(tmp_path / "logs"),
|
|
27
|
+
)
|
|
28
|
+
assert args.get_warmup_steps(100) == int(100 * ratio)
|
|
29
|
+
assert args.lr_scheduler_type == "cosine"
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_final_training_logs_and_reloads(tmp_path, tiny_model_path):
|
|
33
|
+
clf = make_classifier(tmp_path, dataframe=make_df(n=30, num_classes=3))
|
|
34
|
+
clf.optimize_metric = "f1"
|
|
35
|
+
clf.greater_is_better = True
|
|
36
|
+
tokenizer = AutoTokenizer.from_pretrained(tiny_model_path)
|
|
37
|
+
train, val, _ = clf._prepare_datasets(tokenizer, None, max_length=32)
|
|
38
|
+
params = sampled_params()
|
|
39
|
+
log_dir = str(tmp_path / "logs")
|
|
40
|
+
args = clf._build_training_arguments(
|
|
41
|
+
params, str(tmp_path / "output"), 32, "fp32", final=True, logging_dir=log_dir,
|
|
42
|
+
)
|
|
43
|
+
args.num_train_epochs = 1
|
|
44
|
+
args.use_cpu = True
|
|
45
|
+
trainer = clf._build_trainer(
|
|
46
|
+
clf._load_model(tiny_model_path, 0.0), args, train, val, tokenizer,
|
|
47
|
+
params, clf._compute_class_weights(train), "fp32", final=True,
|
|
48
|
+
logging_dir=log_dir,
|
|
49
|
+
)
|
|
50
|
+
assert np.isfinite(trainer.train().training_loss)
|
|
51
|
+
assert trainer.state.best_model_checkpoint is not None
|
|
52
|
+
assert np.isfinite(trainer.evaluate()["eval_loss"])
|
|
53
|
+
events = EventAccumulator(log_dir).Reload()
|
|
54
|
+
assert "train/loss" in events.Tags()["scalars"]
|
|
55
|
+
assert "eval/loss" in events.Tags()["scalars"]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
class KwargsClassifier(torch.nn.Module):
|
|
59
|
+
def __init__(self):
|
|
60
|
+
super().__init__()
|
|
61
|
+
self.linear = torch.nn.Linear(2, 2)
|
|
62
|
+
self.accepts_loss_kwargs = True
|
|
63
|
+
|
|
64
|
+
def forward(self, features, labels=None, **kwargs):
|
|
65
|
+
return SimpleNamespace(logits=self.linear(features))
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_accumulation_matches_full_batch_update(tmp_path):
|
|
69
|
+
torch.manual_seed(42)
|
|
70
|
+
initial = KwargsClassifier()
|
|
71
|
+
dataset = Dataset.from_dict({"features": [[1., 2.]] * 4, "labels": [1] * 4})
|
|
72
|
+
models = []
|
|
73
|
+
for batch, accumulation in [(4, 1), (2, 2)]:
|
|
74
|
+
model = copy.deepcopy(initial)
|
|
75
|
+
args = TrainingArguments(
|
|
76
|
+
output_dir=str(tmp_path / str(batch)), use_cpu=True,
|
|
77
|
+
per_device_train_batch_size=batch, gradient_accumulation_steps=accumulation,
|
|
78
|
+
max_steps=1, max_grad_norm=0., report_to="none", save_strategy="no",
|
|
79
|
+
lr_scheduler_type="constant", disable_tqdm=True,
|
|
80
|
+
)
|
|
81
|
+
trainer = CustomTrainer(
|
|
82
|
+
model=model, args=args, train_dataset=dataset, loss_type="plain",
|
|
83
|
+
optimizers=(torch.optim.SGD(model.parameters(), lr=0.1), None),
|
|
84
|
+
)
|
|
85
|
+
trainer.train()
|
|
86
|
+
models.append(model)
|
|
87
|
+
for full, accumulated in zip(models[0].parameters(), models[1].parameters()):
|
|
88
|
+
torch.testing.assert_close(full, accumulated)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|