bertuner 0.2.2__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (24) hide show
  1. {bertuner-0.2.2 → bertuner-0.2.3}/PKG-INFO +15 -1
  2. {bertuner-0.2.2 → bertuner-0.2.3}/README.md +14 -0
  3. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/BERTuner.py +24 -13
  4. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/CustomTrainer.py +3 -0
  5. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/__init__.py +1 -1
  6. bertuner-0.2.3/bertuner/compat.py +15 -0
  7. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/PKG-INFO +15 -1
  8. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/SOURCES.txt +2 -0
  9. {bertuner-0.2.2 → bertuner-0.2.3}/pyproject.toml +1 -1
  10. {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_bertuner.py +3 -2
  11. bertuner-0.2.3/tests/test_transformers_compat.py +88 -0
  12. {bertuner-0.2.2 → bertuner-0.2.3}/LICENSE +0 -0
  13. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/Predictor.py +0 -0
  14. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/TensorBoardCallback.py +0 -0
  15. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/constants.py +0 -0
  16. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/exceptions.py +0 -0
  17. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner/utils.py +0 -0
  18. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/dependency_links.txt +0 -0
  19. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/requires.txt +0 -0
  20. {bertuner-0.2.2 → bertuner-0.2.3}/bertuner.egg-info/top_level.txt +0 -0
  21. {bertuner-0.2.2 → bertuner-0.2.3}/setup.cfg +0 -0
  22. {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_numerical_stability.py +0 -0
  23. {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_predictor.py +0 -0
  24. {bertuner-0.2.2 → bertuner-0.2.3}/tests/test_utils.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bertuner
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support
5
5
  Author-email: elemets <alafunnell@gmail.com>
6
6
  License: MIT
@@ -58,6 +58,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
58
58
  pip install -r requirements.txt
59
59
  ```
60
60
 
61
+ BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
62
+ 4.48 baseline and the latest available 4.x and 5.x releases, including final
63
+ training, TensorBoard logging, warmup scheduling, and gradient accumulation.
64
+ Future releases are checked by a weekly CI run rather than assumed compatible.
65
+
66
+ When initializing a classifier from a base pretrained model, a Transformers 5
67
+ load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
68
+ as `MISSING`. This is expected when replacing the language-model head with a
69
+ classification head; the new head is learned during fine-tuning. Unexpected
70
+ encoder weights or shape mismatches should still be investigated.
71
+
72
+ Run the test suite with `python -m pytest tests -q`. Training regression tests
73
+ use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
74
+
61
75
  MLflow tracking works in two modes:
62
76
 
63
77
  ```bash
@@ -17,6 +17,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
17
17
  pip install -r requirements.txt
18
18
  ```
19
19
 
20
+ BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
21
+ 4.48 baseline and the latest available 4.x and 5.x releases, including final
22
+ training, TensorBoard logging, warmup scheduling, and gradient accumulation.
23
+ Future releases are checked by a weekly CI run rather than assumed compatible.
24
+
25
+ When initializing a classifier from a base pretrained model, a Transformers 5
26
+ load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
27
+ as `MISSING`. This is expected when replacing the language-model head with a
28
+ classification head; the new head is learned during fine-tuning. Unexpected
29
+ encoder weights or shape mismatches should still be investigated.
30
+
31
+ Run the test suite with `python -m pytest tests -q`. Training regression tests
32
+ use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
33
+
20
34
  MLflow tracking works in two modes:
21
35
 
22
36
  ```bash
@@ -15,7 +15,6 @@ from transformers import (
15
15
  AutoTokenizer,
16
16
  AutoConfig,
17
17
  AutoModelForSequenceClassification,
18
- TrainingArguments,
19
18
  DataCollatorWithPadding,
20
19
  EarlyStoppingCallback,
21
20
  set_seed,
@@ -38,6 +37,7 @@ from sklearn.preprocessing import label_binarize
38
37
  from mlflow.tracking import MlflowClient
39
38
 
40
39
  from bertuner.CustomTrainer import CustomTrainer
40
+ from bertuner.compat import TrainingArguments
41
41
  from bertuner.exceptions import NonFiniteTrainingError, NoStableTrialError
42
42
  from bertuner.TensorBoardCallback import (
43
43
  TensorBoardSyncCallback,
@@ -56,6 +56,7 @@ from bertuner.constants import (
56
56
  MODEL_DROPOUT_ATTRS,
57
57
  SEED,
58
58
  )
59
+ import inspect
59
60
 
60
61
 
61
62
  class BERTuneClassifier:
@@ -746,7 +747,6 @@ class BERTuneClassifier:
746
747
  "gradient_checkpointing": self._use_gradient_checkpointing(max_length),
747
748
  "gradient_checkpointing_kwargs": {"use_reentrant": False},
748
749
  "weight_decay": params["weight_decay"],
749
- "warmup_ratio": params["warmup_ratio"],
750
750
  "metric_for_best_model": f"eval_{self.optimize_metric}",
751
751
  "greater_is_better": self.greater_is_better,
752
752
  "eval_strategy": "epoch",
@@ -760,17 +760,20 @@ class BERTuneClassifier:
760
760
  "seed": self.seed,
761
761
  **self._precision_flags(precision),
762
762
  }
763
- if final:
764
- # Canonical final metrics are logged explicitly after restoring the
765
- # best checkpoint; Trainer only sends loss curves to TensorBoard.
766
- kwargs.update(report_to=["tensorboard"], logging_dir=logging_dir)
767
- else:
768
- kwargs.update(
769
- lr_scheduler_type=params["scheduler"],
770
- remove_unused_columns=True,
771
- report_to=["none"],
772
- )
773
- return TrainingArguments(**kwargs)
763
+
764
+ _TA_PARAMS = inspect.signature(TrainingArguments.__init__).parameters
765
+
766
+ # v5 removed warmup_ratio and accepts fractions in warmup_steps.
767
+ warmup_key = "warmup_ratio" if "warmup_ratio" in _TA_PARAMS else "warmup_steps"
768
+ kwargs[warmup_key] = params["warmup_ratio"]
769
+ kwargs["lr_scheduler_type"] = params["scheduler"]
770
+ # Final training uses an explicit TensorBoard writer in _build_trainer;
771
+ # newer Transformers versions removed TrainingArguments.logging_dir.
772
+ kwargs.update(remove_unused_columns=True, report_to=["none"])
773
+ args = TrainingArguments(**kwargs)
774
+ if warmup_key == "warmup_steps":
775
+ args._bertuner_warmup_ratio = params["warmup_ratio"]
776
+ return args
774
777
 
775
778
  def _build_trainer(
776
779
  self,
@@ -792,8 +795,12 @@ class BERTuneClassifier:
792
795
  )
793
796
  ]
794
797
  if final:
798
+ from transformers.integrations import TensorBoardCallback
799
+ from torch.utils.tensorboard import SummaryWriter
800
+
795
801
  callbacks.extend(
796
802
  [
803
+ TensorBoardCallback(tb_writer=SummaryWriter(logging_dir)),
797
804
  TensorBoardSyncCallback(logging_dir),
798
805
  CleanupCheckpointsCallback,
799
806
  ]
@@ -857,11 +864,15 @@ class BERTuneClassifier:
857
864
  try:
858
865
  trainer.train()
859
866
  except Exception:
867
+ from transformers.integrations import TensorBoardCallback
868
+
860
869
  # Trainer does not emit on_train_end after an exception. Close only
861
870
  # BERTuner-owned writers before the caller decides whether to retry.
862
871
  for callback in trainer.callback_handler.callbacks:
863
872
  if isinstance(callback, TensorBoardSyncCallback):
864
873
  callback.writer.close()
874
+ elif isinstance(callback, TensorBoardCallback) and callback.tb_writer is not None:
875
+ callback.tb_writer.close()
865
876
  raise
866
877
  return trainer, model
867
878
 
@@ -69,6 +69,9 @@ class CustomTrainer(Trainer):
69
69
  callbacks = list(kwargs.pop("callbacks", None) or [])
70
70
  callbacks.append(NonFiniteGradientCallback(training_precision))
71
71
  super().__init__(callbacks=callbacks, **kwargs)
72
+ # compute_loss returns a microbatch mean and does not consume
73
+ # num_items_in_batch, even when the model's forward accepts **kwargs.
74
+ self.model_accepts_loss_kwargs = False
72
75
  self.loss_type = loss_type
73
76
  self.class_weights = class_weights
74
77
  self.training_precision = training_precision
@@ -1,6 +1,6 @@
1
1
  """bertuner: hyperparameter optimization and fine-tuning for BERT-style text classifiers."""
2
2
 
3
- __version__ = "0.2.2"
3
+ __version__ = "0.2.3"
4
4
 
5
5
  from bertuner.BERTuner import BERTuneClassifier
6
6
  from bertuner.Predictor import BERTunePredictor
@@ -0,0 +1,15 @@
1
+ """Small adapters for Transformers APIs shared by supported 4.x and 5.x."""
2
+
3
+ import math
4
+
5
+ from transformers import TrainingArguments as HFTrainingArguments
6
+
7
+
8
+ class TrainingArguments(HFTrainingArguments):
9
+ def get_warmup_steps(self, num_training_steps):
10
+ ratio = getattr(self, "_bertuner_warmup_ratio", None)
11
+ if ratio is not None:
12
+ # New warmup_steps treats 1.0 as one step, whereas the old
13
+ # warmup_ratio treats it as the entire training run.
14
+ return math.ceil(num_training_steps * ratio)
15
+ return super().get_warmup_steps(num_training_steps)
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: bertuner
3
- Version: 0.2.2
3
+ Version: 0.2.3
4
4
  Summary: Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support
5
5
  Author-email: elemets <alafunnell@gmail.com>
6
6
  License: MIT
@@ -58,6 +58,20 @@ git clone https://github.com/elemets/bertuner && cd bertuner
58
58
  pip install -r requirements.txt
59
59
  ```
60
60
 
61
+ BERTuner supports Transformers 4.48+ and 5.x. Compatibility tests cover the
62
+ 4.48 baseline and the latest available 4.x and 5.x releases, including final
63
+ training, TensorBoard logging, warmup scheduling, and gradient accumulation.
64
+ Future releases are checked by a weekly CI run rather than assumed compatible.
65
+
66
+ When initializing a classifier from a base pretrained model, a Transformers 5
67
+ load report may list `lm_head.*` weights as `UNEXPECTED` and classifier weights
68
+ as `MISSING`. This is expected when replacing the language-model head with a
69
+ classification head; the new head is learned during fine-tuning. Unexpected
70
+ encoder weights or shape mismatches should still be investigated.
71
+
72
+ Run the test suite with `python -m pytest tests -q`. Training regression tests
73
+ use a tiny local checkpoint; cached-model predictor tests skip if unavailable.
74
+
61
75
  MLflow tracking works in two modes:
62
76
 
63
77
  ```bash
@@ -6,6 +6,7 @@ bertuner/CustomTrainer.py
6
6
  bertuner/Predictor.py
7
7
  bertuner/TensorBoardCallback.py
8
8
  bertuner/__init__.py
9
+ bertuner/compat.py
9
10
  bertuner/constants.py
10
11
  bertuner/exceptions.py
11
12
  bertuner/utils.py
@@ -17,4 +18,5 @@ bertuner.egg-info/top_level.txt
17
18
  tests/test_bertuner.py
18
19
  tests/test_numerical_stability.py
19
20
  tests/test_predictor.py
21
+ tests/test_transformers_compat.py
20
22
  tests/test_utils.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "bertuner"
7
- version = "0.2.2"
7
+ version = "0.2.3"
8
8
  description = "Hyperparameter optimization and fine-tuning for BERT-style text classifiers (Optuna + MLflow), with long-context ModernBERT support"
9
9
  readme = "README.md"
10
10
  license = { text = "MIT" }
@@ -483,6 +483,7 @@ class TestLossCurveLogging:
483
483
  ]
484
484
  )
485
485
 
486
+ (tmp_path / "downloaded-artifacts").mkdir()
486
487
  artifact_path = MlflowClient().download_artifacts(
487
488
  run.info.run_id,
488
489
  "plots/training_vs_evaluation_loss.png",
@@ -785,7 +786,7 @@ class TestMulticlassWeightedLoss:
785
786
  loss = trainer._singlelabel_loss(logits, labels, torch.device("cpu"))
786
787
  assert torch.isfinite(loss)
787
788
 
788
- def test_end_to_end_training_three_classes(self, tmp_path):
789
+ def test_end_to_end_training_three_classes(self, tmp_path, tiny_model_path):
789
790
  import torch
790
791
  from transformers import (
791
792
  AutoModelForSequenceClassification,
@@ -795,7 +796,7 @@ class TestMulticlassWeightedLoss:
795
796
  )
796
797
  from bertuner.CustomTrainer import CustomTrainer
797
798
 
798
- model_path = "prajjwal1/bert-tiny"
799
+ model_path = tiny_model_path
799
800
  clf = make_classifier(tmp_path, dataframe=make_df(n=60, num_classes=3))
800
801
  assert clf.num_labels == 3
801
802
 
@@ -0,0 +1,88 @@
1
+ """Run unchanged against the oldest supported Transformers and current releases."""
2
+ import copy
3
+ from types import SimpleNamespace
4
+
5
+ import numpy as np
6
+ import pytest
7
+ import torch
8
+ from datasets import Dataset
9
+ from tensorboard.backend.event_processing.event_accumulator import EventAccumulator
10
+ from transformers import AutoTokenizer, TrainingArguments
11
+
12
+ from bertuner.CustomTrainer import CustomTrainer
13
+ from test_bertuner import make_classifier, make_df
14
+ from test_numerical_stability import sampled_params
15
+
16
+
17
+ @pytest.mark.parametrize("ratio", [0.0, 0.1, 1.0])
18
+ @pytest.mark.parametrize("final", [False, True])
19
+ def test_warmup_and_scheduler(tmp_path, ratio, final):
20
+ clf = make_classifier(tmp_path)
21
+ clf.optimize_metric = "avg_precision"
22
+ clf.greater_is_better = True
23
+ params = dict(sampled_params(), warmup_ratio=ratio, scheduler="cosine")
24
+ args = clf._build_training_arguments(
25
+ params, str(tmp_path / "output"), 32, "fp32", final=final,
26
+ logging_dir=str(tmp_path / "logs"),
27
+ )
28
+ assert args.get_warmup_steps(100) == int(100 * ratio)
29
+ assert args.lr_scheduler_type == "cosine"
30
+
31
+
32
+ def test_final_training_logs_and_reloads(tmp_path, tiny_model_path):
33
+ clf = make_classifier(tmp_path, dataframe=make_df(n=30, num_classes=3))
34
+ clf.optimize_metric = "f1"
35
+ clf.greater_is_better = True
36
+ tokenizer = AutoTokenizer.from_pretrained(tiny_model_path)
37
+ train, val, _ = clf._prepare_datasets(tokenizer, None, max_length=32)
38
+ params = sampled_params()
39
+ log_dir = str(tmp_path / "logs")
40
+ args = clf._build_training_arguments(
41
+ params, str(tmp_path / "output"), 32, "fp32", final=True, logging_dir=log_dir,
42
+ )
43
+ args.num_train_epochs = 1
44
+ args.use_cpu = True
45
+ trainer = clf._build_trainer(
46
+ clf._load_model(tiny_model_path, 0.0), args, train, val, tokenizer,
47
+ params, clf._compute_class_weights(train), "fp32", final=True,
48
+ logging_dir=log_dir,
49
+ )
50
+ assert np.isfinite(trainer.train().training_loss)
51
+ assert trainer.state.best_model_checkpoint is not None
52
+ assert np.isfinite(trainer.evaluate()["eval_loss"])
53
+ events = EventAccumulator(log_dir).Reload()
54
+ assert "train/loss" in events.Tags()["scalars"]
55
+ assert "eval/loss" in events.Tags()["scalars"]
56
+
57
+
58
+ class KwargsClassifier(torch.nn.Module):
59
+ def __init__(self):
60
+ super().__init__()
61
+ self.linear = torch.nn.Linear(2, 2)
62
+ self.accepts_loss_kwargs = True
63
+
64
+ def forward(self, features, labels=None, **kwargs):
65
+ return SimpleNamespace(logits=self.linear(features))
66
+
67
+
68
+ def test_accumulation_matches_full_batch_update(tmp_path):
69
+ torch.manual_seed(42)
70
+ initial = KwargsClassifier()
71
+ dataset = Dataset.from_dict({"features": [[1., 2.]] * 4, "labels": [1] * 4})
72
+ models = []
73
+ for batch, accumulation in [(4, 1), (2, 2)]:
74
+ model = copy.deepcopy(initial)
75
+ args = TrainingArguments(
76
+ output_dir=str(tmp_path / str(batch)), use_cpu=True,
77
+ per_device_train_batch_size=batch, gradient_accumulation_steps=accumulation,
78
+ max_steps=1, max_grad_norm=0., report_to="none", save_strategy="no",
79
+ lr_scheduler_type="constant", disable_tqdm=True,
80
+ )
81
+ trainer = CustomTrainer(
82
+ model=model, args=args, train_dataset=dataset, loss_type="plain",
83
+ optimizers=(torch.optim.SGD(model.parameters(), lr=0.1), None),
84
+ )
85
+ trainer.train()
86
+ models.append(model)
87
+ for full, accumulated in zip(models[0].parameters(), models[1].parameters()):
88
+ torch.testing.assert_close(full, accumulated)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes