mostlyai-engine 2.5.0__tar.gz → 2.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/PKG-INFO +25 -18
  2. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/README.md +15 -7
  3. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/__init__.py +1 -1
  4. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/common.py +22 -4
  5. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/hf_engine.py +1 -1
  6. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/vllm_engine.py +1 -6
  7. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/generation.py +0 -1
  8. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/interface.py +3 -0
  9. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/lstm.py +33 -1
  10. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/training.py +14 -4
  11. mostlyai_engine-2.6.0/mostlyai/engine/_language/xgrammar_hf_logits.py +69 -0
  12. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/generation.py +5 -3
  13. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/training.py +4 -1
  14. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/pyproject.toml +10 -10
  15. mostlyai_engine-2.5.0/LICENSE_HEADER +0 -13
  16. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/.gitignore +0 -0
  17. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/LICENSE +0 -0
  18. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_common.py +0 -0
  19. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_dtypes.py +0 -0
  20. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/__init__.py +0 -0
  21. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
  22. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
  23. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
  24. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
  25. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/text.py +0 -0
  26. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
  27. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
  28. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
  29. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
  30. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
  31. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
  32. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/numeric.py +0 -0
  33. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/__init__.py +0 -0
  34. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/encoding.py +0 -0
  35. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/__init__.py +0 -0
  36. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/base.py +0 -0
  37. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
  38. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/xgrammar_utils.py +0 -0
  39. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_memory.py +0 -0
  40. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/__init__.py +0 -0
  41. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/argn.py +0 -0
  42. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/common.py +0 -0
  43. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/encoding.py +0 -0
  44. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/fairness.py +0 -0
  45. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/interface.py +0 -0
  46. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/probability.py +0 -0
  47. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/training.py +0 -0
  48. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_training_utils.py +0 -0
  49. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_workspace.py +0 -0
  50. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/analysis.py +0 -0
  51. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/domain.py +0 -0
  52. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/encoding.py +0 -0
  53. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/generation.py +0 -0
  54. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/logging.py +0 -0
  55. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/random_state.py +0 -0
  56. {mostlyai_engine-2.5.0 → mostlyai_engine-2.6.0}/mostlyai/engine/splitting.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyai-engine
3
- Version: 2.5.0
3
+ Version: 2.6.0
4
4
  Summary: Synthetic Data Engine
5
5
  Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
6
6
  Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
@@ -8,7 +8,6 @@ Project-URL: documentation, https://mostly-ai.github.io/mostlyai-engine/
8
8
  Author-email: MOSTLY AI <dev@mostly.ai>
9
9
  License-Expression: Apache-2.0
10
10
  License-File: LICENSE
11
- License-File: LICENSE_HEADER
12
11
  Classifier: Development Status :: 5 - Production/Stable
13
12
  Classifier: Intended Audience :: Developers
14
13
  Classifier: Intended Audience :: Financial and Insurance Industry
@@ -32,21 +31,21 @@ Requires-Dist: json-repair>=0.47.0
32
31
  Requires-Dist: numpy>=2.0.0
33
32
  Requires-Dist: opacus>=1.5.4
34
33
  Requires-Dist: pandas>=2.2.0
35
- Requires-Dist: peft>=0.12.0
34
+ Requires-Dist: peft>=0.18.2
36
35
  Requires-Dist: psutil<6,>=5.9.5
37
36
  Requires-Dist: pyarrow>=16.0.0
38
37
  Requires-Dist: scikit-learn>=1.4.0
39
- Requires-Dist: setuptools>=77.0.3
40
- Requires-Dist: tokenizers>=0.21.0
41
- Requires-Dist: torch<2.10.0,>=2.9.0
42
- Requires-Dist: torchaudio<2.10.0,>=2.9.0
43
- Requires-Dist: torchvision<0.25.0,>=0.24.0
44
- Requires-Dist: transformers<5,>=4.55.0
45
- Requires-Dist: xgrammar>=0.1.21
38
+ Requires-Dist: setuptools<81.0.0,>=77.0.3
39
+ Requires-Dist: tokenizers>=0.21.1
40
+ Requires-Dist: torch<2.12.0,>=2.11.0
41
+ Requires-Dist: torchaudio<2.12.0,>=2.11.0
42
+ Requires-Dist: torchvision<0.27.0,>=0.26.0
43
+ Requires-Dist: transformers>=5.5.1
44
+ Requires-Dist: xgrammar<1.0.0,>=0.1.32
46
45
  Provides-Extra: gpu
47
46
  Requires-Dist: bitsandbytes==0.42.0; (sys_platform == 'darwin') and extra == 'gpu'
48
47
  Requires-Dist: bitsandbytes>=0.45.5; (sys_platform == 'linux') and extra == 'gpu'
49
- Requires-Dist: vllm==0.12.0; (sys_platform == 'linux' or sys_platform == 'darwin') and extra == 'gpu'
48
+ Requires-Dist: vllm==0.20.0; (sys_platform == 'linux' or sys_platform == 'darwin') and extra == 'gpu'
50
49
  Description-Content-Type: text/markdown
51
50
 
52
51
  # Synthetic Data Engine 💎
@@ -113,10 +112,13 @@ or alternatively for a GPU setup (needed for LLM finetuning and inference):
113
112
  uv pip install -U 'mostlyai-engine[gpu]'
114
113
  ```
115
114
 
116
- On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
115
+ On Linux, one can explicitly install the CPU-only variant of PyTorch together with `mostlyai-engine`:
117
116
 
118
117
  ```bash
119
- uv pip install -U torch==2.9.1+cpu torchvision==0.24.1+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
118
+ uv pip install --index-strategy unsafe-first-match -U \
119
+ torch==2.11.0+cpu torchvision==0.26.0+cpu torchaudio==2.11.0+cpu \
120
+ mostlyai-engine \
121
+ --extra-index-url https://download.pytorch.org/whl/cpu
120
122
  ```
121
123
 
122
124
  ## TabularARGN for Flat Data
@@ -191,10 +193,15 @@ from sklearn.metrics import accuracy_score, roc_auc_score
191
193
 
192
194
  # predict class labels for a categorical
193
195
  predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
196
+ # model-conditional class probabilities (same inputs as predict; target column dropped from seed)
197
+ probabilities = argn.predict_proba(data_test, target="income")
194
198
 
195
199
  # evaluate performance
196
- accuracy = accuracy_score(data_test["income"], predictions)
197
- auc = roc_auc_score(data_test["income"], probabilities[:, 1])
200
+ accuracy = accuracy_score(data_test["income"], predictions["income"])
201
+ # AUC: sklearn needs binary 0/1 targets and scores for the "positive" class (here: second category)
202
+ pos_label = probabilities.columns[1]
203
+ y_true_bin = (data_test["income"] == pos_label).astype(int)
204
+ auc = roc_auc_score(y_true_bin, probabilities[pos_label])
198
205
  print(f"Accuracy: {accuracy:.3f}, AUC: {auc:.3f}")
199
206
  ```
200
207
 
@@ -305,7 +312,7 @@ argn.sample(ctx_data=ctx_data)
305
312
 
306
313
  The `LanguageModel` class provides a scikit-learn-compatible interface for working with semi-structured textual data. It leverages pre-trained language models or trains lightweight LSTM models from scratch to generate synthetic text data.
307
314
 
308
- **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pre-trained HuggingFace models by setting model to e.g. `microsoft/phi-1.5` (GPU required).
315
+ **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pretrained Hugging Face models (`model="<hub/repo>"`; GPU required). Verified checkpoints include `HuggingFaceTB/SmolLM2-135M`, `HuggingFaceTB/SmolLM3-3B`, `Qwen/Qwen3-0.6B`, and `microsoft/phi-4`.
309
316
 
310
317
  ### Model Training
311
318
 
@@ -360,5 +367,5 @@ lm.sample(
360
367
 
361
368
  Example notebooks demonstrating various use cases are available in the `examples` directory:
362
369
  - TabularARGN for flat tabular data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/flat.ipynb)
363
- - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
364
- - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
370
+ - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
371
+ - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
@@ -62,10 +62,13 @@ or alternatively for a GPU setup (needed for LLM finetuning and inference):
62
62
  uv pip install -U 'mostlyai-engine[gpu]'
63
63
  ```
64
64
 
65
- On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
65
+ On Linux, one can explicitly install the CPU-only variant of PyTorch together with `mostlyai-engine`:
66
66
 
67
67
  ```bash
68
- uv pip install -U torch==2.9.1+cpu torchvision==0.24.1+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
68
+ uv pip install --index-strategy unsafe-first-match -U \
69
+ torch==2.11.0+cpu torchvision==0.26.0+cpu torchaudio==2.11.0+cpu \
70
+ mostlyai-engine \
71
+ --extra-index-url https://download.pytorch.org/whl/cpu
69
72
  ```
70
73
 
71
74
  ## TabularARGN for Flat Data
@@ -140,10 +143,15 @@ from sklearn.metrics import accuracy_score, roc_auc_score
140
143
 
141
144
  # predict class labels for a categorical
142
145
  predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
146
+ # model-conditional class probabilities (same inputs as predict; target column dropped from seed)
147
+ probabilities = argn.predict_proba(data_test, target="income")
143
148
 
144
149
  # evaluate performance
145
- accuracy = accuracy_score(data_test["income"], predictions)
146
- auc = roc_auc_score(data_test["income"], probabilities[:, 1])
150
+ accuracy = accuracy_score(data_test["income"], predictions["income"])
151
+ # AUC: sklearn needs binary 0/1 targets and scores for the "positive" class (here: second category)
152
+ pos_label = probabilities.columns[1]
153
+ y_true_bin = (data_test["income"] == pos_label).astype(int)
154
+ auc = roc_auc_score(y_true_bin, probabilities[pos_label])
147
155
  print(f"Accuracy: {accuracy:.3f}, AUC: {auc:.3f}")
148
156
  ```
149
157
 
@@ -254,7 +262,7 @@ argn.sample(ctx_data=ctx_data)
254
262
 
255
263
  The `LanguageModel` class provides a scikit-learn-compatible interface for working with semi-structured textual data. It leverages pre-trained language models or trains lightweight LSTM models from scratch to generate synthetic text data.
256
264
 
257
- **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pre-trained HuggingFace models by setting model to e.g. `microsoft/phi-1.5` (GPU required).
265
+ **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pretrained Hugging Face models (`model="<hub/repo>"`; GPU required). Verified checkpoints include `HuggingFaceTB/SmolLM2-135M`, `HuggingFaceTB/SmolLM3-3B`, `Qwen/Qwen3-0.6B`, and `microsoft/phi-4`.
258
266
 
259
267
  ### Model Training
260
268
 
@@ -309,5 +317,5 @@ lm.sample(
309
317
 
310
318
  Example notebooks demonstrating various use cases are available in the `examples` directory:
311
319
  - TabularARGN for flat tabular data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/flat.ipynb)
312
- - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
313
- - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
320
+ - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
321
+ - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
@@ -34,7 +34,7 @@ __all__ = [
34
34
  "TabularARGN",
35
35
  "LanguageModel",
36
36
  ]
37
- __version__ = "2.5.0"
37
+ __version__ = "2.6.0"
38
38
 
39
39
  # suppress specific warning related to os.fork() in multi-threaded processes
40
40
  warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
@@ -52,8 +52,19 @@ def get_attention_implementation(config: PretrainedConfig) -> str | None:
52
52
 
53
53
 
54
54
  def load_base_model_and_config(
55
- model_id_or_path: str | Path, device: torch.device, is_peft_adapter: bool, is_training: bool
55
+ model_id_or_path: str | Path,
56
+ device: torch.device,
57
+ is_peft_adapter: bool,
58
+ is_training: bool,
59
+ *,
60
+ differential_privacy: bool = False,
56
61
  ) -> tuple[PreTrainedModel, PretrainedConfig]:
62
+ """Load a HF base model (and config) for language training or inference.
63
+
64
+ When ``differential_privacy`` is True (Opacus DP training), the loader prefers
65
+ settings that keep per-sample gradients well-defined: float32 weights, no int4
66
+ training path, eager attention (not fused SDPA), and no gradient checkpointing.
67
+ """
57
68
  # opacus DP does not support parallel/sharded training
58
69
  model_id_or_path = str(model_id_or_path)
59
70
  if is_peft_adapter:
@@ -79,7 +90,13 @@ def load_base_model_and_config(
79
90
  "CUDA device was found but bitsandbytes is not available. Please use extra [gpu] to install bitsandbytes for quantization."
80
91
  )
81
92
  bf16_supported = is_bf16_supported(device)
82
- if bf16_supported:
93
+ # Opacus needs reliable per-sample grads; bfloat16 params + grad_sample hooks are a poor match on many setups.
94
+ use_int4_training = is_gpu_training and is_bitsandbytes_available and not differential_privacy
95
+ if differential_privacy:
96
+ torch_dtype = torch.float32
97
+ # Eager attention keeps standard backward paths; fused SDPA can break Opacus grad_sample hooks.
98
+ attn_implementation = "eager"
99
+ elif bf16_supported:
83
100
  attn_implementation = get_attention_implementation(config)
84
101
  torch_dtype = torch.bfloat16
85
102
  else:
@@ -87,7 +104,7 @@ def load_base_model_and_config(
87
104
  torch_dtype = torch.float32
88
105
  if hasattr(config, "quantization_config"):
89
106
  quantization_config = AutoQuantizationConfig.from_dict(config.quantization_config)
90
- elif is_gpu_training and is_bitsandbytes_available:
107
+ elif use_int4_training:
91
108
  quantization_config = BitsAndBytesConfig(
92
109
  load_in_4bit=True,
93
110
  bnb_4bit_quant_type="nf4",
@@ -123,8 +140,9 @@ def load_base_model_and_config(
123
140
  if isinstance(quantization_config, BitsAndBytesConfig):
124
141
  # convert all non-kbit layers to float32
125
142
  model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=False)
126
- if is_gpu_training and model.supports_gradient_checkpointing:
143
+ if is_gpu_training and model.supports_gradient_checkpointing and not differential_privacy:
127
144
  # pay 50% time penalty for _large_ memory savings
145
+ # gradient checkpointing breaks Opacus per-sample gradient hooks
128
146
  _LOG.info("enable gradient checkpointing")
129
147
  model.gradient_checkpointing_enable()
130
148
  model.enable_input_require_grads()
@@ -22,11 +22,11 @@ import torch
22
22
  from peft import PeftModel
23
23
  from pydantic import BaseModel
24
24
  from transformers import AutoTokenizer
25
- from xgrammar.contrib.hf import LogitsProcessor
26
25
 
27
26
  from mostlyai.engine._language.common import load_base_model_and_config
28
27
  from mostlyai.engine._language.engine.base import EngineMetrics, LanguageEngine
29
28
  from mostlyai.engine._language.tokenizer_utils import tokenize_fn
29
+ from mostlyai.engine._language.xgrammar_hf_logits import LogitsProcessor
30
30
  from mostlyai.engine._language.xgrammar_utils import create_compiled_grammars
31
31
 
32
32
 
@@ -14,11 +14,6 @@
14
14
 
15
15
  from __future__ import annotations
16
16
 
17
- import os
18
-
19
- os.environ["VLLM_USE_V1"] = "1"
20
-
21
-
22
17
  import time
23
18
  from os import PathLike
24
19
 
@@ -28,7 +23,7 @@ from pydantic import BaseModel
28
23
  from transformers import AutoConfig, AutoTokenizer
29
24
  from vllm import LLM, SamplingParams
30
25
  from vllm.distributed import cleanup_dist_env_and_memory
31
- from vllm.inputs.data import TokensPrompt
26
+ from vllm.inputs.llm import TokensPrompt
32
27
  from vllm.lora.request import LoRARequest
33
28
  from vllm.sampling_params import StructuredOutputsParams
34
29
 
@@ -150,7 +150,6 @@ def generate(
150
150
  _LOG.info("GENERATE_LANGUAGE started")
151
151
  t0_ = time.time()
152
152
  os.environ["VLLM_LOGGING_LEVEL"] = "WARNING"
153
- os.environ["VLLM_NO_DEPRECATION_WARNING"] = "1"
154
153
 
155
154
  @contextlib.contextmanager
156
155
  def tqdm_disabled():
@@ -55,6 +55,9 @@ class LanguageModel(BaseEstimator):
55
55
  tgt_encoding_types: Dictionary mapping column names to encoding types.
56
56
  Example: {'category': 'LANGUAGE_CATEGORICAL', 'headline': 'LANGUAGE_TEXT'}
57
57
  model: The identifier of the language model to train. Defaults to MOSTLY_AI/LSTMFromScratch-3m.
58
+ Pretrained Hugging Face checkpoints are supported; verified examples include
59
+ HuggingFaceTB/SmolLM2-135M, HuggingFaceTB/SmolLM3-3B, Qwen/Qwen3-0.6B, and microsoft/phi-4
60
+ (GPU strongly recommended).
58
61
  max_training_time: Maximum training time in minutes. Defaults to 14400 (10 days).
59
62
  max_epochs: Maximum number of training epochs. Defaults to 100.
60
63
  batch_size: Per-device batch size for training and validation. If None, determined automatically.
@@ -22,6 +22,17 @@ from transformers.modeling_outputs import CausalLMOutput
22
22
  _LOG = logging.getLogger(__name__)
23
23
 
24
24
 
25
+ def _dplstm_state_dict_aliases(num_layers: int) -> dict[str, str]:
26
+ """Nested DPLSTM names -> flat ``nn.LSTM`` names (same storage; needed for checkpoint save)."""
27
+ aliases: dict[str, str] = {}
28
+ for i in range(num_layers):
29
+ aliases[f"lstm.l{i}.ih.weight"] = f"lstm.weight_ih_l{i}"
30
+ aliases[f"lstm.l{i}.ih.bias"] = f"lstm.bias_ih_l{i}"
31
+ aliases[f"lstm.l{i}.hh.weight"] = f"lstm.weight_hh_l{i}"
32
+ aliases[f"lstm.l{i}.hh.bias"] = f"lstm.bias_hh_l{i}"
33
+ return aliases
34
+
35
+
25
36
  class LSTMFromScratchConfig(PretrainedConfig):
26
37
  model_type = model_id = "MOSTLY_AI/LSTMFromScratch-3m"
27
38
 
@@ -54,7 +65,6 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
54
65
 
55
66
  def __init__(self, config: LSTMFromScratchConfig):
56
67
  super().__init__(config)
57
- self.config = config
58
68
 
59
69
  self.embedding = nn.Embedding(self.config.vocab_size, self.config.embedding_size)
60
70
  self.dropout = nn.Dropout(self.config.dropout)
@@ -77,6 +87,28 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
77
87
  # this will be filled by left_to_right_padding() during the generation
78
88
  self.pad_token_id = None
79
89
 
90
+ # `_tied_weights_keys` is always a dict: empty unless DP (see `remove_tied_weights_from_state_dict`).
91
+ self._tied_weights_keys = _dplstm_state_dict_aliases(self.config.num_layers) if self.config.with_dp else {}
92
+
93
+ self.post_init()
94
+
95
+ def _init_weights(self, module: nn.Module) -> None:
96
+ # Keep PyTorch defaults for our main modules (historical behavior). HF post_init()
97
+ # still runs init_weights on the rest (e.g. any submodules inside DPLSTM).
98
+ if module in (self.embedding, self.lm_head, self.lstm):
99
+ return
100
+ super()._init_weights(module)
101
+
102
+ def get_expanded_tied_weights_keys(self, all_submodels: bool = False) -> dict[str, str]:
103
+ """
104
+ Transformers >= 5 sets `all_tied_weights_keys` from this. `self._tied_weights_keys` is also read when
105
+ saving (see `remove_tied_weights_from_state_dict`). Keep both in sync for DPLSTM aliases.
106
+ """
107
+ expanded = getattr(super(), "get_expanded_tied_weights_keys", None)
108
+ out: dict[str, str] = dict(expanded(all_submodels=all_submodels)) if expanded is not None else {}
109
+ out.update(self._tied_weights_keys)
110
+ return out
111
+
80
112
  def forward(
81
113
  self,
82
114
  input_ids: torch.Tensor,
@@ -233,7 +233,8 @@ def _gpu_estimate_max_batch_size(
233
233
  model: PreTrainedModel | GradSampleModule, device: torch.device, max_tokens: int, initial_batch_size: int
234
234
  ) -> int:
235
235
  batch_size = 2 ** int(np.log2(initial_batch_size))
236
- optimizer = torch.optim.AdamW(params=model.parameters())
236
+ # Match training optimizer: only trainable params (e.g. LoRA), for consistent memory probe.
237
+ optimizer = torch.optim.AdamW(params=[p for p in model.parameters() if p.requires_grad])
237
238
 
238
239
  # create test batch of zeros with estimated max sequence length
239
240
  def create_test_batch(batch_size: int):
@@ -339,7 +340,8 @@ def train(
339
340
  _LOG.info(f"{torch.cuda.device_count()=}")
340
341
  bf16_supported = is_bf16_supported(device)
341
342
  _LOG.info(f"{bf16_supported=}")
342
- use_mixed_precision = bf16_supported and model != LSTMFromScratchConfig.model_id
343
+ use_mixed_precision = bf16_supported and model != LSTMFromScratchConfig.model_id and not with_dp
344
+ # DP uses float32 + no autocast (see load_base_model_and_config); bf16 autocast breaks Opacus grad_sample.
343
345
  _LOG.info(f"{use_mixed_precision=}")
344
346
 
345
347
  ctx_stats = workspace.ctx_stats.read()
@@ -450,7 +452,11 @@ def train(
450
452
  if resume_from_last_checkpoint:
451
453
  tokenizer = AutoTokenizer.from_pretrained(model_id_or_path, **tokenizer_args)
452
454
  model, _ = load_base_model_and_config(
453
- model_id_or_path, device=device, is_peft_adapter=False, is_training=True
455
+ model_id_or_path,
456
+ device=device,
457
+ is_peft_adapter=False,
458
+ is_training=True,
459
+ differential_privacy=with_dp,
454
460
  )
455
461
  else:
456
462
  # fresh initialization of the custom tokenizer and LSTM model
@@ -468,6 +474,7 @@ def train(
468
474
  device=device,
469
475
  is_peft_adapter=resume_from_last_checkpoint,
470
476
  is_training=True,
477
+ differential_privacy=with_dp,
471
478
  )
472
479
  tokenizer = AutoTokenizer.from_pretrained(model_id_or_path, **tokenizer_args)
473
480
  if tokenizer.eos_token is None:
@@ -584,7 +591,10 @@ def train(
584
591
  batch_size=val_batch_size,
585
592
  collate_fn=data_collator,
586
593
  )
587
- optimizer = torch.optim.AdamW(params=model.parameters(), lr=initial_lr)
594
+ optimizer = torch.optim.AdamW(
595
+ params=[p for p in model.parameters() if p.requires_grad],
596
+ lr=initial_lr,
597
+ ) # frozen PEFT base weights must not be in Opacus optimizer (no grad_sample on unused params)
588
598
  early_stopper = EarlyStopper(val_loss_patience=4)
589
599
  lr_scheduler: LRScheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(
590
600
  optimizer=optimizer,
@@ -0,0 +1,69 @@
1
+ # Copyright 2025 MOSTLY AI
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """xgrammar Hugging Face LogitsProcessor with fixes for our dependency stack."""
16
+
17
+ from __future__ import annotations
18
+
19
+ import torch
20
+ import xgrammar as xgr
21
+ from xgrammar.contrib.hf import LogitsProcessor as _UpstreamLogitsProcessor
22
+
23
+
24
+ class LogitsProcessor(_UpstreamLogitsProcessor):
25
+ """
26
+ Same as ``xgrammar.contrib.hf.LogitsProcessor``, but passes a Python ``int`` to
27
+ ``GrammarMatcher.accept_token``. Upstream uses ``input_ids[i][-1]`` (a 0-dim tensor);
28
+ newer PyTorch / Transformers stacks leave that as ``torch.Tensor``, while xgrammar's
29
+ TVM FFI expects ``int``.
30
+ """
31
+
32
+ def __call__(self, input_ids: torch.LongTensor, scores: torch.FloatTensor) -> torch.FloatTensor:
33
+ if len(self.matchers) == 0:
34
+ self.batch_size = input_ids.shape[0]
35
+ self.compiled_grammars = (
36
+ self.compiled_grammars if len(self.compiled_grammars) > 1 else self.compiled_grammars * self.batch_size
37
+ )
38
+ assert len(self.compiled_grammars) == self.batch_size, (
39
+ "The number of compiled grammars must be equal to the batch size."
40
+ )
41
+ self.matchers = [xgr.GrammarMatcher(self.compiled_grammars[i]) for i in range(self.batch_size)]
42
+ self.token_bitmask = xgr.allocate_token_bitmask(self.batch_size, self.full_vocab_size)
43
+
44
+ if input_ids.shape[0] != self.batch_size:
45
+ raise RuntimeError(
46
+ "Expect input_ids.shape[0] to be LogitsProcessor.batch_size."
47
+ + f"Got {input_ids.shape[0]} for the former, and {self.batch_size} for the latter."
48
+ )
49
+
50
+ if not self.prefilled:
51
+ self.prefilled = True
52
+ else:
53
+ for i in range(self.batch_size):
54
+ if not self.matchers[i].is_terminated():
55
+ sampled_token = int(input_ids[i, -1].item())
56
+ assert self.matchers[i].accept_token(sampled_token)
57
+
58
+ for i in range(self.batch_size):
59
+ if not self.matchers[i].is_terminated():
60
+ self.matchers[i].fill_next_token_bitmask(self.token_bitmask, i)
61
+
62
+ device_type = scores.device.type
63
+ if device_type != "cuda":
64
+ scores = scores.to("cpu")
65
+ xgr.apply_token_bitmask_inplace(scores, self.token_bitmask.to(scores.device))
66
+ if device_type != "cuda":
67
+ scores = scores.to(device_type)
68
+
69
+ return scores
@@ -1066,7 +1066,7 @@ def generate(
1066
1066
  sidx_df = encode_positional_column(sidx, max_seq_len=seq_steps, prefix=SIDX_SUB_COLUMN_PREFIX)
1067
1067
  sidx_vals = {
1068
1068
  c: torch.unsqueeze(
1069
- torch.as_tensor(sidx_df[c].to_numpy(), device=model.device).type(torch.int),
1069
+ torch.as_tensor(sidx_df[c].to_numpy(copy=True), device=model.device).type(torch.int),
1070
1070
  dim=-1,
1071
1071
  )
1072
1072
  for c in sidx_df
@@ -1110,7 +1110,7 @@ def generate(
1110
1110
  sdec = pd.Series([0] * step_size) # initial sequence index decile
1111
1111
  sdec_vals = {
1112
1112
  f"{SDEC_SUB_COLUMN_PREFIX}cat": torch.unsqueeze(
1113
- torch.as_tensor(sdec.to_numpy(), device=model.device).type(torch.int), dim=-1
1113
+ torch.as_tensor(sdec.to_numpy(copy=True), device=model.device).type(torch.int), dim=-1
1114
1114
  )
1115
1115
  }
1116
1116
 
@@ -1119,7 +1119,9 @@ def generate(
1119
1119
  if len(seed_step_encoded) > 0:
1120
1120
  seed_vals = {
1121
1121
  col: torch.unsqueeze(
1122
- torch.as_tensor(seed_step_encoded[col].to_numpy(), device=model.device).type(torch.int),
1122
+ torch.as_tensor(seed_step_encoded[col].to_numpy(copy=True), device=model.device).type(
1123
+ torch.int
1124
+ ),
1123
1125
  dim=-1,
1124
1126
  )
1125
1127
  for col in seed_step_encoded.columns
@@ -47,7 +47,10 @@ def train(
47
47
  - `ModelStore`: Trained model checkpoints and logs.
48
48
 
49
49
  Args:
50
- model: The identifier of the model to train. If tabular, defaults to MOSTLY_AI/Medium. If language, defaults to MOSTLY_AI/LSTMFromScratch-3m.
50
+ model: The identifier of the model to train. If tabular, defaults to MOSTLY_AI/Medium. If language,
51
+ defaults to MOSTLY_AI/LSTMFromScratch-3m. For language models, Hugging Face hub ids are supported;
52
+ verified pretrained checkpoints include HuggingFaceTB/SmolLM2-135M, HuggingFaceTB/SmolLM3-3B,
53
+ Qwen/Qwen3-0.6B, and microsoft/phi-4.
51
54
  max_training_time: Maximum training time in minutes. If None, defaults to 10 days.
52
55
  max_epochs: Maximum number of training epochs. If None, defaults to 100 epochs.
53
56
  batch_size: Per-device batch size for training and validation. If None, determined automatically.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "mostlyai-engine"
3
- version = "2.5.0"
3
+ version = "2.6.0"
4
4
  description = "Synthetic Data Engine"
5
5
  authors = [{ name = "MOSTLY AI", email = "dev@mostly.ai" }]
6
6
  requires-python = ">=3.11,<3.14"
@@ -24,32 +24,32 @@ classifiers = [
24
24
  ]
25
25
 
26
26
  dependencies = [
27
- "setuptools>=77.0.3",
27
+ "setuptools>=77.0.3,<81.0.0", # vllm 0.20 caps setuptools for Python > 3.11
28
28
  "numpy>=2.0.0",
29
29
  "pandas>=2.2.0",
30
30
  "pyarrow>=16.0.0",
31
31
  "joblib>=1.4.2",
32
32
  "scikit-learn>=1.4.0",
33
33
  "psutil>=5.9.5,<6", # upgrade when colab psutil is updated
34
- "tokenizers>=0.21.0",
35
- "transformers>=4.55.0,<5", # keep <5 for vllm==0.12 compatibility
34
+ "tokenizers>=0.21.1", # vllm 0.20 lower bound
35
+ "transformers>=5.5.1", # vllm 0.20 forbids 5.0–5.4 and 5.5.0; >=5.5.1 needed for newer HF model types (e.g. gemma4)
36
36
  "datasets>=3.0.0",
37
37
  "accelerate>=1.5.0",
38
- "peft>=0.12.0",
38
+ "peft>=0.18.2", # transformers 5.7+ checks min PEFT in model.add_adapter (integrations/peft.py)
39
39
  "huggingface-hub[hf-xet]>=0.30.2",
40
40
  "opacus>=1.5.4",
41
- "xgrammar>=0.1.21",
41
+ "xgrammar>=0.1.32,<1.0.0", # aligned with vllm 0.20
42
42
  "json-repair>=0.47.0",
43
- "torch>=2.9.0,<2.10.0",
44
- "torchaudio>=2.9.0,<2.10.0",
45
- "torchvision>=0.24.0,<0.25.0"
43
+ "torch>=2.11.0,<2.12.0",
44
+ "torchaudio>=2.11.0,<2.12.0",
45
+ "torchvision>=0.26.0,<0.27.0"
46
46
  ]
47
47
 
48
48
  [project.optional-dependencies]
49
49
  gpu = [
50
50
  "bitsandbytes==0.42.0; sys_platform == 'darwin'",
51
51
  "bitsandbytes>=0.45.5; sys_platform == 'linux'",
52
- "vllm==0.12.0; sys_platform == 'linux' or sys_platform == 'darwin'",
52
+ "vllm==0.20.0; sys_platform == 'linux' or sys_platform == 'darwin'",
53
53
  ]
54
54
 
55
55
  [dependency-groups]
@@ -1,13 +0,0 @@
1
- Copyright 2025 MOSTLY AI
2
-
3
- Licensed under the Apache License, Version 2.0 (the "License");
4
- you may not use this file except in compliance with the License.
5
- You may obtain a copy of the License at
6
-
7
- http://www.apache.org/licenses/LICENSE-2.0
8
-
9
- Unless required by applicable law or agreed to in writing, software
10
- distributed under the License is distributed on an "AS IS" BASIS,
11
- WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
- See the License for the specific language governing permissions and
13
- limitations under the License.
File without changes