mostlyai-engine 2.4.0__tar.gz → 2.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/PKG-INFO +26 -19
  2. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/README.md +15 -7
  3. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/__init__.py +1 -1
  4. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_common.py +6 -3
  5. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/character.py +3 -1
  6. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/datetime.py +6 -5
  7. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/common.py +22 -4
  8. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/hf_engine.py +1 -1
  9. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/vllm_engine.py +1 -6
  10. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/generation.py +0 -1
  11. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/interface.py +3 -0
  12. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/lstm.py +33 -1
  13. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/training.py +14 -4
  14. mostlyai_engine-2.6.0/mostlyai/engine/_language/xgrammar_hf_logits.py +69 -0
  15. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/xgrammar_utils.py +11 -2
  16. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/common.py +1 -1
  17. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/generation.py +11 -8
  18. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/training.py +4 -1
  19. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/pyproject.toml +13 -13
  20. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/.gitignore +0 -0
  21. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/LICENSE +0 -0
  22. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_dtypes.py +0 -0
  23. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/__init__.py +0 -0
  24. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
  25. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
  26. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
  27. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
  28. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/language/text.py +0 -0
  29. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
  30. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
  31. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
  32. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
  33. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_encoding_types/tabular/numeric.py +0 -0
  34. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/__init__.py +0 -0
  35. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/encoding.py +0 -0
  36. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/__init__.py +0 -0
  37. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/engine/base.py +0 -0
  38. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
  39. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_memory.py +0 -0
  40. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/__init__.py +0 -0
  41. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/argn.py +0 -0
  42. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/encoding.py +0 -0
  43. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/fairness.py +0 -0
  44. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/interface.py +0 -0
  45. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/probability.py +0 -0
  46. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_tabular/training.py +0 -0
  47. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_training_utils.py +0 -0
  48. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/_workspace.py +0 -0
  49. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/analysis.py +0 -0
  50. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/domain.py +0 -0
  51. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/encoding.py +0 -0
  52. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/generation.py +0 -0
  53. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/logging.py +0 -0
  54. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/random_state.py +0 -0
  55. {mostlyai_engine-2.4.0 → mostlyai_engine-2.6.0}/mostlyai/engine/splitting.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyai-engine
3
- Version: 2.4.0
3
+ Version: 2.6.0
4
4
  Summary: Synthetic Data Engine
5
5
  Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
6
6
  Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
@@ -17,13 +17,12 @@ Classifier: Intended Audience :: Science/Research
17
17
  Classifier: Intended Audience :: Telecommunications Industry
18
18
  Classifier: License :: OSI Approved :: Apache Software License
19
19
  Classifier: Operating System :: OS Independent
20
- Classifier: Programming Language :: Python :: 3.10
21
20
  Classifier: Programming Language :: Python :: 3.11
22
21
  Classifier: Programming Language :: Python :: 3.12
23
22
  Classifier: Programming Language :: Python :: 3.13
24
23
  Classifier: Topic :: Software Development :: Libraries
25
24
  Classifier: Typing :: Typed
26
- Requires-Python: >=3.10
25
+ Requires-Python: <3.14,>=3.11
27
26
  Requires-Dist: accelerate>=1.5.0
28
27
  Requires-Dist: datasets>=3.0.0
29
28
  Requires-Dist: huggingface-hub[hf-xet]>=0.30.2
@@ -32,21 +31,21 @@ Requires-Dist: json-repair>=0.47.0
32
31
  Requires-Dist: numpy>=2.0.0
33
32
  Requires-Dist: opacus>=1.5.4
34
33
  Requires-Dist: pandas>=2.2.0
35
- Requires-Dist: peft>=0.12.0
34
+ Requires-Dist: peft>=0.18.2
36
35
  Requires-Dist: psutil<6,>=5.9.5
37
36
  Requires-Dist: pyarrow>=16.0.0
38
37
  Requires-Dist: scikit-learn>=1.4.0
39
- Requires-Dist: setuptools>=77.0.3
40
- Requires-Dist: tokenizers>=0.21.0
41
- Requires-Dist: torch<2.10.0,>=2.9.0
42
- Requires-Dist: torchaudio<2.10.0,>=2.9.0
43
- Requires-Dist: torchvision<0.25.0,>=0.24.0
44
- Requires-Dist: transformers>=4.55.0
45
- Requires-Dist: xgrammar>=0.1.21
38
+ Requires-Dist: setuptools<81.0.0,>=77.0.3
39
+ Requires-Dist: tokenizers>=0.21.1
40
+ Requires-Dist: torch<2.12.0,>=2.11.0
41
+ Requires-Dist: torchaudio<2.12.0,>=2.11.0
42
+ Requires-Dist: torchvision<0.27.0,>=0.26.0
43
+ Requires-Dist: transformers>=5.5.1
44
+ Requires-Dist: xgrammar<1.0.0,>=0.1.32
46
45
  Provides-Extra: gpu
47
46
  Requires-Dist: bitsandbytes==0.42.0; (sys_platform == 'darwin') and extra == 'gpu'
48
47
  Requires-Dist: bitsandbytes>=0.45.5; (sys_platform == 'linux') and extra == 'gpu'
49
- Requires-Dist: vllm==0.12.0; (sys_platform == 'linux' or sys_platform == 'darwin') and extra == 'gpu'
48
+ Requires-Dist: vllm==0.20.0; (sys_platform == 'linux' or sys_platform == 'darwin') and extra == 'gpu'
50
49
  Description-Content-Type: text/markdown
51
50
 
52
51
  # Synthetic Data Engine 💎
@@ -113,10 +112,13 @@ or alternatively for a GPU setup (needed for LLM finetuning and inference):
113
112
  uv pip install -U 'mostlyai-engine[gpu]'
114
113
  ```
115
114
 
116
- On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
115
+ On Linux, one can explicitly install the CPU-only variant of PyTorch together with `mostlyai-engine`:
117
116
 
118
117
  ```bash
119
- uv pip install -U torch==2.9.1+cpu torchvision==0.24.1+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
118
+ uv pip install --index-strategy unsafe-first-match -U \
119
+ torch==2.11.0+cpu torchvision==0.26.0+cpu torchaudio==2.11.0+cpu \
120
+ mostlyai-engine \
121
+ --extra-index-url https://download.pytorch.org/whl/cpu
120
122
  ```
121
123
 
122
124
  ## TabularARGN for Flat Data
@@ -191,10 +193,15 @@ from sklearn.metrics import accuracy_score, roc_auc_score
191
193
 
192
194
  # predict class labels for a categorical
193
195
  predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
196
+ # model-conditional class probabilities (same inputs as predict; target column dropped from seed)
197
+ probabilities = argn.predict_proba(data_test, target="income")
194
198
 
195
199
  # evaluate performance
196
- accuracy = accuracy_score(data_test["income"], predictions)
197
- auc = roc_auc_score(data_test["income"], probabilities[:, 1])
200
+ accuracy = accuracy_score(data_test["income"], predictions["income"])
201
+ # AUC: sklearn needs binary 0/1 targets and scores for the "positive" class (here: second category)
202
+ pos_label = probabilities.columns[1]
203
+ y_true_bin = (data_test["income"] == pos_label).astype(int)
204
+ auc = roc_auc_score(y_true_bin, probabilities[pos_label])
198
205
  print(f"Accuracy: {accuracy:.3f}, AUC: {auc:.3f}")
199
206
  ```
200
207
 
@@ -305,7 +312,7 @@ argn.sample(ctx_data=ctx_data)
305
312
 
306
313
  The `LanguageModel` class provides a scikit-learn-compatible interface for working with semi-structured textual data. It leverages pre-trained language models or trains lightweight LSTM models from scratch to generate synthetic text data.
307
314
 
308
- **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pre-trained HuggingFace models by setting model to e.g. `microsoft/phi-1.5` (GPU required).
315
+ **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pretrained Hugging Face models (`model="<hub/repo>"`; GPU required). Verified checkpoints include `HuggingFaceTB/SmolLM2-135M`, `HuggingFaceTB/SmolLM3-3B`, `Qwen/Qwen3-0.6B`, and `microsoft/phi-4`.
309
316
 
310
317
  ### Model Training
311
318
 
@@ -360,5 +367,5 @@ lm.sample(
360
367
 
361
368
  Example notebooks demonstrating various use cases are available in the `examples` directory:
362
369
  - TabularARGN for flat tabular data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/flat.ipynb)
363
- - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
364
- - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
370
+ - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
371
+ - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
@@ -62,10 +62,13 @@ or alternatively for a GPU setup (needed for LLM finetuning and inference):
62
62
  uv pip install -U 'mostlyai-engine[gpu]'
63
63
  ```
64
64
 
65
- On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
65
+ On Linux, one can explicitly install the CPU-only variant of PyTorch together with `mostlyai-engine`:
66
66
 
67
67
  ```bash
68
- uv pip install -U torch==2.9.1+cpu torchvision==0.24.1+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
68
+ uv pip install --index-strategy unsafe-first-match -U \
69
+ torch==2.11.0+cpu torchvision==0.26.0+cpu torchaudio==2.11.0+cpu \
70
+ mostlyai-engine \
71
+ --extra-index-url https://download.pytorch.org/whl/cpu
69
72
  ```
70
73
 
71
74
  ## TabularARGN for Flat Data
@@ -140,10 +143,15 @@ from sklearn.metrics import accuracy_score, roc_auc_score
140
143
 
141
144
  # predict class labels for a categorical
142
145
  predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
146
+ # model-conditional class probabilities (same inputs as predict; target column dropped from seed)
147
+ probabilities = argn.predict_proba(data_test, target="income")
143
148
 
144
149
  # evaluate performance
145
- accuracy = accuracy_score(data_test["income"], predictions)
146
- auc = roc_auc_score(data_test["income"], probabilities[:, 1])
150
+ accuracy = accuracy_score(data_test["income"], predictions["income"])
151
+ # AUC: sklearn needs binary 0/1 targets and scores for the "positive" class (here: second category)
152
+ pos_label = probabilities.columns[1]
153
+ y_true_bin = (data_test["income"] == pos_label).astype(int)
154
+ auc = roc_auc_score(y_true_bin, probabilities[pos_label])
147
155
  print(f"Accuracy: {accuracy:.3f}, AUC: {auc:.3f}")
148
156
  ```
149
157
 
@@ -254,7 +262,7 @@ argn.sample(ctx_data=ctx_data)
254
262
 
255
263
  The `LanguageModel` class provides a scikit-learn-compatible interface for working with semi-structured textual data. It leverages pre-trained language models or trains lightweight LSTM models from scratch to generate synthetic text data.
256
264
 
257
- **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pre-trained HuggingFace models by setting model to e.g. `microsoft/phi-1.5` (GPU required).
265
+ **Note**: The default model is `MOSTLY_AI/LSTMFromScratch-3m`, a lightweight LSTM model trained from scratch (GPU strongly recommended). You can also use pretrained Hugging Face models (`model="<hub/repo>"`; GPU required). Verified checkpoints include `HuggingFaceTB/SmolLM2-135M`, `HuggingFaceTB/SmolLM3-3B`, `Qwen/Qwen3-0.6B`, and `microsoft/phi-4`.
258
266
 
259
267
  ### Model Training
260
268
 
@@ -309,5 +317,5 @@ lm.sample(
309
317
 
310
318
  Example notebooks demonstrating various use cases are available in the `examples` directory:
311
319
  - TabularARGN for flat tabular data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/flat.ipynb)
312
- - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
313
- - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
320
+ - TabularARGN for sequential data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/sequential.ipynb)
321
+ - LanguageModel for textual data [![Run on Colab](https://img.shields.io/badge/Open%20in-Colab-blue?logo=google-colab)](https://colab.research.google.com/github/mostly-ai/mostlyai-engine/blob/main/examples/language.ipynb)
@@ -34,7 +34,7 @@ __all__ = [
34
34
  "TabularARGN",
35
35
  "LanguageModel",
36
36
  ]
37
- __version__ = "2.4.0"
37
+ __version__ = "2.6.0"
38
38
 
39
39
  # suppress specific warning related to os.fork() in multi-threaded processes
40
40
  warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
@@ -273,8 +273,9 @@ def safe_convert_datetime(values: pd.Series, date_only: bool = False) -> pd.Seri
273
273
  utc=True,
274
274
  dayfirst=False, # assume 1/3/2020 is Jan 3
275
275
  )
276
- # check whether firstday=True yields less non-NA, and if so, switch to using that flag
277
- if values_parsed_fixed.isna().sum() > values.isna().sum():
276
+ has_slash_dates = values.astype("string").str.contains(r"^\s*\d{1,2}/\d{1,2}/\d{4}(?:\s|$)", regex=True).any()
277
+ # some mixed-format slash dates are interpreted more reliably with dayfirst=True
278
+ if has_slash_dates and values_parsed_fixed.isna().sum() > values.isna().sum():
278
279
  values_parsed_fixed_dayfirst = pd.to_datetime(
279
280
  values,
280
281
  errors="coerce", # silently map invalid dates to NA
@@ -937,6 +938,7 @@ def impute_from_non_nan_distribution(values: pd.Series, column_stats: dict) -> t
937
938
  Returns:
938
939
  tuple[pd.Series, pd.Series]: The series with imputed values and the mask of NaNs.
939
940
  """
941
+ values = values.copy()
940
942
  nan_mask = values.isna()
941
943
  vc = values.value_counts(normalize=True)
942
944
  if vc.empty:
@@ -944,7 +946,8 @@ def impute_from_non_nan_distribution(values: pd.Series, column_stats: dict) -> t
944
946
  probs = vc.values
945
947
  categories = vc.index
946
948
  # NOTE: an alternative will be to use the largest remainder method
947
- values[nan_mask] = np.random.choice(categories, size=nan_mask.sum(), p=probs)
949
+ if nan_mask.any():
950
+ values.loc[nan_mask] = np.random.choice(categories, size=nan_mask.sum(), p=probs)
948
951
  return values, nan_mask.astype(int)
949
952
 
950
953
 
@@ -102,7 +102,9 @@ def encode_character(values: pd.Series, stats: dict, _: pd.Series | None = None)
102
102
  df_split = split_sub_columns_character(values, max_string_length)
103
103
  for idx in range(max_string_length):
104
104
  sub_col = f"P{idx}"
105
- np_codes = np.array(pd.Categorical(df_split[sub_col], categories=stats["codes"][sub_col]).codes)
105
+ categories = list(stats["codes"][sub_col].keys())
106
+ values_at_pos = df_split[sub_col].where(df_split[sub_col].isin(categories), UNKNOWN_TOKEN)
107
+ np_codes = np.array(pd.Categorical(values_at_pos, categories=categories).codes)
106
108
  np.place(np_codes, np_codes == -1, 0)
107
109
  df_split[sub_col] = np_codes
108
110
  if stats["has_nan"]:
@@ -234,9 +234,10 @@ def decode_datetime(df_encoded: pd.DataFrame, stats: dict):
234
234
  d = df_encoded["day"] + stats["min_values"]["day"]
235
235
  # fix invalid dates by setting these to last day of month
236
236
  is_leap = y.apply(lambda x: calendar.isleap(x))
237
- d[is_leap & (m == 2) & (d > 29)] = 29
238
- d[~is_leap & (m == 2) & (d > 28)] = 28
239
- d[((m == 4) | (m == 6) | (m == 9) | (m == 11)) & (d > 30)] = 30
237
+ d = d.copy()
238
+ d.loc[is_leap & (m == 2) & (d > 29)] = 29
239
+ d.loc[~is_leap & (m == 2) & (d > 28)] = 28
240
+ d.loc[((m == 4) | (m == 6) | (m == 9) | (m == 11)) & (d > 30)] = 30
240
241
  # concatenate to datetime string
241
242
  y = y.astype(str)
242
243
  m = m.astype(str).str.zfill(2)
@@ -271,7 +272,7 @@ def decode_datetime(df_encoded: pd.DataFrame, stats: dict):
271
272
  # set all values to NaN if no valid values were present
272
273
  values[df_encoded["nan"] == 0] = pd.NA
273
274
  # convert from string to datetime
274
- values = pd.to_datetime(values)
275
+ values = pd.to_datetime(values).astype("datetime64[ns]")
275
276
  if not stats["has_time"]:
276
- values = pd.to_datetime(values.dt.date)
277
+ values = pd.to_datetime(values.dt.date).astype("datetime64[ns]")
277
278
  return values
@@ -52,8 +52,19 @@ def get_attention_implementation(config: PretrainedConfig) -> str | None:
52
52
 
53
53
 
54
54
  def load_base_model_and_config(
55
- model_id_or_path: str | Path, device: torch.device, is_peft_adapter: bool, is_training: bool
55
+ model_id_or_path: str | Path,
56
+ device: torch.device,
57
+ is_peft_adapter: bool,
58
+ is_training: bool,
59
+ *,
60
+ differential_privacy: bool = False,
56
61
  ) -> tuple[PreTrainedModel, PretrainedConfig]:
62
+ """Load a HF base model (and config) for language training or inference.
63
+
64
+ When ``differential_privacy`` is True (Opacus DP training), the loader prefers
65
+ settings that keep per-sample gradients well-defined: float32 weights, no int4
66
+ training path, eager attention (not fused SDPA), and no gradient checkpointing.
67
+ """
57
68
  # opacus DP does not support parallel/sharded training
58
69
  model_id_or_path = str(model_id_or_path)
59
70
  if is_peft_adapter:
@@ -79,7 +90,13 @@ def load_base_model_and_config(
79
90
  "CUDA device was found but bitsandbytes is not available. Please use extra [gpu] to install bitsandbytes for quantization."
80
91
  )
81
92
  bf16_supported = is_bf16_supported(device)
82
- if bf16_supported:
93
+ # Opacus needs reliable per-sample grads; bfloat16 params + grad_sample hooks are a poor match on many setups.
94
+ use_int4_training = is_gpu_training and is_bitsandbytes_available and not differential_privacy
95
+ if differential_privacy:
96
+ torch_dtype = torch.float32
97
+ # Eager attention keeps standard backward paths; fused SDPA can break Opacus grad_sample hooks.
98
+ attn_implementation = "eager"
99
+ elif bf16_supported:
83
100
  attn_implementation = get_attention_implementation(config)
84
101
  torch_dtype = torch.bfloat16
85
102
  else:
@@ -87,7 +104,7 @@ def load_base_model_and_config(
87
104
  torch_dtype = torch.float32
88
105
  if hasattr(config, "quantization_config"):
89
106
  quantization_config = AutoQuantizationConfig.from_dict(config.quantization_config)
90
- elif is_gpu_training and is_bitsandbytes_available:
107
+ elif use_int4_training:
91
108
  quantization_config = BitsAndBytesConfig(
92
109
  load_in_4bit=True,
93
110
  bnb_4bit_quant_type="nf4",
@@ -123,8 +140,9 @@ def load_base_model_and_config(
123
140
  if isinstance(quantization_config, BitsAndBytesConfig):
124
141
  # convert all non-kbit layers to float32
125
142
  model = prepare_model_for_kbit_training(model, use_gradient_checkpointing=False)
126
- if is_gpu_training and model.supports_gradient_checkpointing:
143
+ if is_gpu_training and model.supports_gradient_checkpointing and not differential_privacy:
127
144
  # pay 50% time penalty for _large_ memory savings
145
+ # gradient checkpointing breaks Opacus per-sample gradient hooks
128
146
  _LOG.info("enable gradient checkpointing")
129
147
  model.gradient_checkpointing_enable()
130
148
  model.enable_input_require_grads()
@@ -22,11 +22,11 @@ import torch
22
22
  from peft import PeftModel
23
23
  from pydantic import BaseModel
24
24
  from transformers import AutoTokenizer
25
- from xgrammar.contrib.hf import LogitsProcessor
26
25
 
27
26
  from mostlyai.engine._language.common import load_base_model_and_config
28
27
  from mostlyai.engine._language.engine.base import EngineMetrics, LanguageEngine
29
28
  from mostlyai.engine._language.tokenizer_utils import tokenize_fn
29
+ from mostlyai.engine._language.xgrammar_hf_logits import LogitsProcessor
30
30
  from mostlyai.engine._language.xgrammar_utils import create_compiled_grammars
31
31
 
32
32
 
@@ -14,11 +14,6 @@
14
14
 
15
15
  from __future__ import annotations
16
16
 
17
- import os
18
-
19
- os.environ["VLLM_USE_V1"] = "1"
20
-
21
-
22
17
  import time
23
18
  from os import PathLike
24
19
 
@@ -28,7 +23,7 @@ from pydantic import BaseModel
28
23
  from transformers import AutoConfig, AutoTokenizer
29
24
  from vllm import LLM, SamplingParams
30
25
  from vllm.distributed import cleanup_dist_env_and_memory
31
- from vllm.inputs.data import TokensPrompt
26
+ from vllm.inputs.llm import TokensPrompt
32
27
  from vllm.lora.request import LoRARequest
33
28
  from vllm.sampling_params import StructuredOutputsParams
34
29
 
@@ -150,7 +150,6 @@ def generate(
150
150
  _LOG.info("GENERATE_LANGUAGE started")
151
151
  t0_ = time.time()
152
152
  os.environ["VLLM_LOGGING_LEVEL"] = "WARNING"
153
- os.environ["VLLM_NO_DEPRECATION_WARNING"] = "1"
154
153
 
155
154
  @contextlib.contextmanager
156
155
  def tqdm_disabled():
@@ -55,6 +55,9 @@ class LanguageModel(BaseEstimator):
55
55
  tgt_encoding_types: Dictionary mapping column names to encoding types.
56
56
  Example: {'category': 'LANGUAGE_CATEGORICAL', 'headline': 'LANGUAGE_TEXT'}
57
57
  model: The identifier of the language model to train. Defaults to MOSTLY_AI/LSTMFromScratch-3m.
58
+ Pretrained Hugging Face checkpoints are supported; verified examples include
59
+ HuggingFaceTB/SmolLM2-135M, HuggingFaceTB/SmolLM3-3B, Qwen/Qwen3-0.6B, and microsoft/phi-4
60
+ (GPU strongly recommended).
58
61
  max_training_time: Maximum training time in minutes. Defaults to 14400 (10 days).
59
62
  max_epochs: Maximum number of training epochs. Defaults to 100.
60
63
  batch_size: Per-device batch size for training and validation. If None, determined automatically.
@@ -22,6 +22,17 @@ from transformers.modeling_outputs import CausalLMOutput
22
22
  _LOG = logging.getLogger(__name__)
23
23
 
24
24
 
25
+ def _dplstm_state_dict_aliases(num_layers: int) -> dict[str, str]:
26
+ """Nested DPLSTM names -> flat ``nn.LSTM`` names (same storage; needed for checkpoint save)."""
27
+ aliases: dict[str, str] = {}
28
+ for i in range(num_layers):
29
+ aliases[f"lstm.l{i}.ih.weight"] = f"lstm.weight_ih_l{i}"
30
+ aliases[f"lstm.l{i}.ih.bias"] = f"lstm.bias_ih_l{i}"
31
+ aliases[f"lstm.l{i}.hh.weight"] = f"lstm.weight_hh_l{i}"
32
+ aliases[f"lstm.l{i}.hh.bias"] = f"lstm.bias_hh_l{i}"
33
+ return aliases
34
+
35
+
25
36
  class LSTMFromScratchConfig(PretrainedConfig):
26
37
  model_type = model_id = "MOSTLY_AI/LSTMFromScratch-3m"
27
38
 
@@ -54,7 +65,6 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
54
65
 
55
66
  def __init__(self, config: LSTMFromScratchConfig):
56
67
  super().__init__(config)
57
- self.config = config
58
68
 
59
69
  self.embedding = nn.Embedding(self.config.vocab_size, self.config.embedding_size)
60
70
  self.dropout = nn.Dropout(self.config.dropout)
@@ -77,6 +87,28 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
77
87
  # this will be filled by left_to_right_padding() during the generation
78
88
  self.pad_token_id = None
79
89
 
90
+ # `_tied_weights_keys` is always a dict: empty unless DP (see `remove_tied_weights_from_state_dict`).
91
+ self._tied_weights_keys = _dplstm_state_dict_aliases(self.config.num_layers) if self.config.with_dp else {}
92
+
93
+ self.post_init()
94
+
95
+ def _init_weights(self, module: nn.Module) -> None:
96
+ # Keep PyTorch defaults for our main modules (historical behavior). HF post_init()
97
+ # still runs init_weights on the rest (e.g. any submodules inside DPLSTM).
98
+ if module in (self.embedding, self.lm_head, self.lstm):
99
+ return
100
+ super()._init_weights(module)
101
+
102
+ def get_expanded_tied_weights_keys(self, all_submodels: bool = False) -> dict[str, str]:
103
+ """
104
+ Transformers >= 5 sets `all_tied_weights_keys` from this. `self._tied_weights_keys` is also read when
105
+ saving (see `remove_tied_weights_from_state_dict`). Keep both in sync for DPLSTM aliases.
106
+ """
107
+ expanded = getattr(super(), "get_expanded_tied_weights_keys", None)
108
+ out: dict[str, str] = dict(expanded(all_submodels=all_submodels)) if expanded is not None else {}
109
+ out.update(self._tied_weights_keys)
110
+ return out
111
+
80
112
  def forward(
81
113
  self,
82
114
  input_ids: torch.Tensor,
@@ -233,7 +233,8 @@ def _gpu_estimate_max_batch_size(
233
233
  model: PreTrainedModel | GradSampleModule, device: torch.device, max_tokens: int, initial_batch_size: int
234
234
  ) -> int:
235
235
  batch_size = 2 ** int(np.log2(initial_batch_size))
236
- optimizer = torch.optim.AdamW(params=model.parameters())
236
+ # Match training optimizer: only trainable params (e.g. LoRA), for consistent memory probe.
237
+ optimizer = torch.optim.AdamW(params=[p for p in model.parameters() if p.requires_grad])
237
238
 
238
239
  # create test batch of zeros with estimated max sequence length
239
240
  def create_test_batch(batch_size: int):
@@ -339,7 +340,8 @@ def train(
339
340
  _LOG.info(f"{torch.cuda.device_count()=}")
340
341
  bf16_supported = is_bf16_supported(device)
341
342
  _LOG.info(f"{bf16_supported=}")
342
- use_mixed_precision = bf16_supported and model != LSTMFromScratchConfig.model_id
343
+ use_mixed_precision = bf16_supported and model != LSTMFromScratchConfig.model_id and not with_dp
344
+ # DP uses float32 + no autocast (see load_base_model_and_config); bf16 autocast breaks Opacus grad_sample.
343
345
  _LOG.info(f"{use_mixed_precision=}")
344
346
 
345
347
  ctx_stats = workspace.ctx_stats.read()
@@ -450,7 +452,11 @@ def train(
450
452
  if resume_from_last_checkpoint:
451
453
  tokenizer = AutoTokenizer.from_pretrained(model_id_or_path, **tokenizer_args)
452
454
  model, _ = load_base_model_and_config(
453
- model_id_or_path, device=device, is_peft_adapter=False, is_training=True
455
+ model_id_or_path,
456
+ device=device,
457
+ is_peft_adapter=False,
458
+ is_training=True,
459
+ differential_privacy=with_dp,
454
460
  )
455
461
  else:
456
462
  # fresh initialization of the custom tokenizer and LSTM model
@@ -468,6 +474,7 @@ def train(
468
474
  device=device,
469
475
  is_peft_adapter=resume_from_last_checkpoint,
470
476
  is_training=True,
477
+ differential_privacy=with_dp,
471
478
  )
472
479
  tokenizer = AutoTokenizer.from_pretrained(model_id_or_path, **tokenizer_args)
473
480
  if tokenizer.eos_token is None:
@@ -584,7 +591,10 @@ def train(
584
591
  batch_size=val_batch_size,
585
592
  collate_fn=data_collator,
586
593
  )
587
- optimizer = torch.optim.AdamW(params=model.parameters(), lr=initial_lr)
594
+ optimizer = torch.optim.AdamW(
595
+ params=[p for p in model.parameters() if p.requires_grad],
596
+ lr=initial_lr,
597
+ ) # frozen PEFT base weights must not be in Opacus optimizer (no grad_sample on unused params)
588
598
  early_stopper = EarlyStopper(val_loss_patience=4)
589
599
  lr_scheduler: LRScheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(
590
600
  optimizer=optimizer,
@@ -0,0 +1,69 @@
1
+ # Copyright 2025 MOSTLY AI
2
+ #
3
+ # Licensed under the Apache License, Version 2.0 (the "License");
4
+ # you may not use this file except in compliance with the License.
5
+ # You may obtain a copy of the License at
6
+ #
7
+ # http://www.apache.org/licenses/LICENSE-2.0
8
+ #
9
+ # Unless required by applicable law or agreed to in writing, software
10
+ # distributed under the License is distributed on an "AS IS" BASIS,
11
+ # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ # See the License for the specific language governing permissions and
13
+ # limitations under the License.
14
+
15
+ """xgrammar Hugging Face LogitsProcessor with fixes for our dependency stack."""
16
+
17
+ from __future__ import annotations
18
+
19
+ import torch
20
+ import xgrammar as xgr
21
+ from xgrammar.contrib.hf import LogitsProcessor as _UpstreamLogitsProcessor
22
+
23
+
24
+ class LogitsProcessor(_UpstreamLogitsProcessor):
25
+ """
26
+ Same as ``xgrammar.contrib.hf.LogitsProcessor``, but passes a Python ``int`` to
27
+ ``GrammarMatcher.accept_token``. Upstream uses ``input_ids[i][-1]`` (a 0-dim tensor);
28
+ newer PyTorch / Transformers stacks leave that as ``torch.Tensor``, while xgrammar's
29
+ TVM FFI expects ``int``.
30
+ """
31
+
32
+ def __call__(self, input_ids: torch.LongTensor, scores: torch.FloatTensor) -> torch.FloatTensor:
33
+ if len(self.matchers) == 0:
34
+ self.batch_size = input_ids.shape[0]
35
+ self.compiled_grammars = (
36
+ self.compiled_grammars if len(self.compiled_grammars) > 1 else self.compiled_grammars * self.batch_size
37
+ )
38
+ assert len(self.compiled_grammars) == self.batch_size, (
39
+ "The number of compiled grammars must be equal to the batch size."
40
+ )
41
+ self.matchers = [xgr.GrammarMatcher(self.compiled_grammars[i]) for i in range(self.batch_size)]
42
+ self.token_bitmask = xgr.allocate_token_bitmask(self.batch_size, self.full_vocab_size)
43
+
44
+ if input_ids.shape[0] != self.batch_size:
45
+ raise RuntimeError(
46
+ "Expect input_ids.shape[0] to be LogitsProcessor.batch_size."
47
+ + f"Got {input_ids.shape[0]} for the former, and {self.batch_size} for the latter."
48
+ )
49
+
50
+ if not self.prefilled:
51
+ self.prefilled = True
52
+ else:
53
+ for i in range(self.batch_size):
54
+ if not self.matchers[i].is_terminated():
55
+ sampled_token = int(input_ids[i, -1].item())
56
+ assert self.matchers[i].accept_token(sampled_token)
57
+
58
+ for i in range(self.batch_size):
59
+ if not self.matchers[i].is_terminated():
60
+ self.matchers[i].fill_next_token_bitmask(self.token_bitmask, i)
61
+
62
+ device_type = scores.device.type
63
+ if device_type != "cuda":
64
+ scores = scores.to("cpu")
65
+ xgr.apply_token_bitmask_inplace(scores, self.token_bitmask.to(scores.device))
66
+ if device_type != "cuda":
67
+ scores = scores.to(device_type)
68
+
69
+ return scores
@@ -75,14 +75,23 @@ def create_schemas(
75
75
  numeric_fields = field_types.get(ModelEncodingType.language_numeric, [])
76
76
  datetime_fields = field_types.get(ModelEncodingType.language_datetime, [])
77
77
  cache = {}
78
+
79
+ def _normalize_seed_value(seed_value):
80
+ return None if pd.isna(seed_value) else seed_value
81
+
78
82
  for _, seed_row in seed_df.iterrows():
79
- cache_key = hash(tuple(sorted([(field_name, str(seed_value)) for field_name, seed_value in seed_row.items()])))
83
+ normalized_seed_items = [
84
+ (field_name, _normalize_seed_value(seed_value)) for field_name, seed_value in seed_row.items()
85
+ ]
86
+ cache_key = hash(
87
+ tuple(sorted([(field_name, str(seed_value)) for field_name, seed_value in normalized_seed_items]))
88
+ )
80
89
  if cache_key in cache:
81
90
  yield cache[cache_key]
82
91
  continue
83
92
  model_dict = {}
84
93
  if not seed_row.empty:
85
- model_dict |= {field_name: (Literal[seed_value], ...) for field_name, seed_value in seed_row.items()} # type: ignore[valid-type]
94
+ model_dict |= {field_name: (Literal[seed_value], ...) for field_name, seed_value in normalized_seed_items} # type: ignore[valid-type]
86
95
  for field_name in unseeded_fields:
87
96
  if field_name in categorical_fields:
88
97
  categories = stats["columns"][field_name]["categories"]
@@ -197,7 +197,7 @@ def prepare_context_inputs(
197
197
  # Build flat context inputs (CTXFLT/*)
198
198
  ctxflt_inputs = {
199
199
  col: torch.unsqueeze(
200
- torch.as_tensor(ctx_encoded[col].to_numpy(), device=device).type(torch.int),
200
+ torch.as_tensor(ctx_encoded[col].to_numpy(copy=True), device=device).type(torch.int),
201
201
  dim=-1,
202
202
  )
203
203
  for col in ctx_encoded.columns
@@ -640,10 +640,9 @@ def decode_buffered_samples(
640
640
  keys=keys,
641
641
  key_name=tgt_context_key,
642
642
  )
643
- df_syn = df_syn.drop(
644
- columns=[c for c in df_syn.columns if c.startswith(POSITIONAL_COLUMN)],
645
- axis=1,
646
- ).reset_index(drop=True)
643
+ df_syn = df_syn.drop(columns=[c for c in df_syn.columns if c.startswith(POSITIONAL_COLUMN)]).reset_index(
644
+ drop=True
645
+ )
647
646
  else:
648
647
  data, seed_data = zip(*buffer.buffer)
649
648
  df_syn = pd.concat(data, axis=0).reset_index(drop=True)
@@ -1067,7 +1066,7 @@ def generate(
1067
1066
  sidx_df = encode_positional_column(sidx, max_seq_len=seq_steps, prefix=SIDX_SUB_COLUMN_PREFIX)
1068
1067
  sidx_vals = {
1069
1068
  c: torch.unsqueeze(
1070
- torch.as_tensor(sidx_df[c].to_numpy(), device=model.device).type(torch.int),
1069
+ torch.as_tensor(sidx_df[c].to_numpy(copy=True), device=model.device).type(torch.int),
1071
1070
  dim=-1,
1072
1071
  )
1073
1072
  for c in sidx_df
@@ -1111,7 +1110,7 @@ def generate(
1111
1110
  sdec = pd.Series([0] * step_size) # initial sequence index decile
1112
1111
  sdec_vals = {
1113
1112
  f"{SDEC_SUB_COLUMN_PREFIX}cat": torch.unsqueeze(
1114
- torch.as_tensor(sdec.to_numpy(), device=model.device).type(torch.int), dim=-1
1113
+ torch.as_tensor(sdec.to_numpy(copy=True), device=model.device).type(torch.int), dim=-1
1115
1114
  )
1116
1115
  }
1117
1116
 
@@ -1120,7 +1119,9 @@ def generate(
1120
1119
  if len(seed_step_encoded) > 0:
1121
1120
  seed_vals = {
1122
1121
  col: torch.unsqueeze(
1123
- torch.as_tensor(seed_step_encoded[col].to_numpy(), device=model.device).type(torch.int),
1122
+ torch.as_tensor(seed_step_encoded[col].to_numpy(copy=True), device=model.device).type(
1123
+ torch.int
1124
+ ),
1124
1125
  dim=-1,
1125
1126
  )
1126
1127
  for col in seed_step_encoded.columns
@@ -1224,7 +1225,9 @@ def generate(
1224
1225
  # Use context inputs prepared earlier
1225
1226
  x = ctx_inputs
1226
1227
  fixed_values = {
1227
- col: torch.as_tensor(seed_batch_encoded[col].to_numpy(), device=model.device).type(torch.int)
1228
+ col: torch.as_tensor(seed_batch_encoded[col].to_numpy(copy=True), device=model.device).type(
1229
+ torch.int
1230
+ )
1228
1231
  for col in seed_batch_encoded.columns
1229
1232
  if col in tgt_sub_columns
1230
1233
  }
@@ -47,7 +47,10 @@ def train(
47
47
  - `ModelStore`: Trained model checkpoints and logs.
48
48
 
49
49
  Args:
50
- model: The identifier of the model to train. If tabular, defaults to MOSTLY_AI/Medium. If language, defaults to MOSTLY_AI/LSTMFromScratch-3m.
50
+ model: The identifier of the model to train. If tabular, defaults to MOSTLY_AI/Medium. If language,
51
+ defaults to MOSTLY_AI/LSTMFromScratch-3m. For language models, Hugging Face hub ids are supported;
52
+ verified pretrained checkpoints include HuggingFaceTB/SmolLM2-135M, HuggingFaceTB/SmolLM3-3B,
53
+ Qwen/Qwen3-0.6B, and microsoft/phi-4.
51
54
  max_training_time: Maximum training time in minutes. If None, defaults to 10 days.
52
55
  max_epochs: Maximum number of training epochs. If None, defaults to 100 epochs.
53
56
  batch_size: Per-device batch size for training and validation. If None, determined automatically.
@@ -1,9 +1,9 @@
1
1
  [project]
2
2
  name = "mostlyai-engine"
3
- version = "2.4.0"
3
+ version = "2.6.0"
4
4
  description = "Synthetic Data Engine"
5
5
  authors = [{ name = "MOSTLY AI", email = "dev@mostly.ai" }]
6
- requires-python = ">=3.10"
6
+ requires-python = ">=3.11,<3.14"
7
7
  readme = "README.md"
8
8
  license = "Apache-2.0"
9
9
  classifiers = [
@@ -14,7 +14,6 @@ classifiers = [
14
14
  "Intended Audience :: Financial and Insurance Industry",
15
15
  "Intended Audience :: Healthcare Industry",
16
16
  "Intended Audience :: Telecommunications Industry",
17
- "Programming Language :: Python :: 3.10",
18
17
  "Programming Language :: Python :: 3.11",
19
18
  "Programming Language :: Python :: 3.12",
20
19
  "Programming Language :: Python :: 3.13",
@@ -25,32 +24,32 @@ classifiers = [
25
24
  ]
26
25
 
27
26
  dependencies = [
28
- "setuptools>=77.0.3",
27
+ "setuptools>=77.0.3,<81.0.0", # vllm 0.20 caps setuptools for Python > 3.11
29
28
  "numpy>=2.0.0",
30
29
  "pandas>=2.2.0",
31
30
  "pyarrow>=16.0.0",
32
31
  "joblib>=1.4.2",
33
32
  "scikit-learn>=1.4.0",
34
33
  "psutil>=5.9.5,<6", # upgrade when colab psutil is updated
35
- "tokenizers>=0.21.0",
36
- "transformers>=4.55.0",
34
+ "tokenizers>=0.21.1", # vllm 0.20 lower bound
35
+ "transformers>=5.5.1", # vllm 0.20 forbids 5.0–5.4 and 5.5.0; >=5.5.1 needed for newer HF model types (e.g. gemma4)
37
36
  "datasets>=3.0.0",
38
37
  "accelerate>=1.5.0",
39
- "peft>=0.12.0",
38
+ "peft>=0.18.2", # transformers 5.7+ checks min PEFT in model.add_adapter (integrations/peft.py)
40
39
  "huggingface-hub[hf-xet]>=0.30.2",
41
40
  "opacus>=1.5.4",
42
- "xgrammar>=0.1.21",
41
+ "xgrammar>=0.1.32,<1.0.0", # aligned with vllm 0.20
43
42
  "json-repair>=0.47.0",
44
- "torch>=2.9.0,<2.10.0",
45
- "torchaudio>=2.9.0,<2.10.0",
46
- "torchvision>=0.24.0,<0.25.0"
43
+ "torch>=2.11.0,<2.12.0",
44
+ "torchaudio>=2.11.0,<2.12.0",
45
+ "torchvision>=0.26.0,<0.27.0"
47
46
  ]
48
47
 
49
48
  [project.optional-dependencies]
50
49
  gpu = [
51
50
  "bitsandbytes==0.42.0; sys_platform == 'darwin'",
52
51
  "bitsandbytes>=0.45.5; sys_platform == 'linux'",
53
- "vllm==0.12.0; sys_platform == 'linux' or sys_platform == 'darwin'",
52
+ "vllm==0.20.0; sys_platform == 'linux' or sys_platform == 'darwin'",
54
53
  ]
55
54
 
56
55
  [dependency-groups]
@@ -95,8 +94,9 @@ requires = ["hatchling", "hatch-vcs"]
95
94
  build-backend = "hatchling.build"
96
95
 
97
96
  [tool.ruff]
98
- target-version = "py310"
97
+ target-version = "py311"
99
98
  line-length = 120
99
+ extend-exclude = ["*.ipynb"]
100
100
  [tool.ruff.format]
101
101
  exclude = ["examples/*.ipynb"]
102
102
  [tool.ruff.lint]
File without changes