mostlyai-engine 1.4.5__tar.gz → 1.4.7__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/PKG-INFO +9 -9
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/README.md +1 -1
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/__init__.py +1 -1
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/common.py +32 -7
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/vllm_engine.py +1 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/training.py +51 -7
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/generation.py +1 -9
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/pyproject.toml +9 -9
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/.gitignore +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/LICENSE +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_common.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_dtypes.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/text.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/numeric.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/encoding.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/base.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/generation.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/lstm.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/xgrammar_utils.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_memory.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/__init__.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/argn.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/common.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/encoding.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/fairness.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_tabular/training.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_training_utils.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_workspace.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/analysis.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/domain.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/encoding.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/generation.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/logging.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/random_state.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/splitting.py +0 -0
- {mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/training.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mostlyai-engine
|
|
3
|
-
Version: 1.4.
|
|
3
|
+
Version: 1.4.7
|
|
4
4
|
Summary: Synthetic Data Engine
|
|
5
5
|
Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
|
|
6
6
|
Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
|
|
@@ -29,23 +29,23 @@ Requires-Dist: datasets>=3.0.0
|
|
|
29
29
|
Requires-Dist: huggingface-hub[hf-xet]>=0.30.2
|
|
30
30
|
Requires-Dist: joblib>=1.4.2
|
|
31
31
|
Requires-Dist: json-repair<0.47.0,>=0.30.0
|
|
32
|
-
Requires-Dist: numpy>=
|
|
33
|
-
Requires-Dist: opacus>=1.5.
|
|
32
|
+
Requires-Dist: numpy>=2.0.0
|
|
33
|
+
Requires-Dist: opacus>=1.5.4
|
|
34
34
|
Requires-Dist: pandas~=2.2.0
|
|
35
35
|
Requires-Dist: peft>=0.12.0
|
|
36
36
|
Requires-Dist: psutil<6,>=5.9.5
|
|
37
37
|
Requires-Dist: pyarrow>=16.0.0
|
|
38
38
|
Requires-Dist: setuptools>=77.0.3
|
|
39
39
|
Requires-Dist: tokenizers>=0.21.0
|
|
40
|
-
Requires-Dist: torch<2.
|
|
41
|
-
Requires-Dist: torchaudio<2.
|
|
42
|
-
Requires-Dist: torchvision<0.
|
|
40
|
+
Requires-Dist: torch<2.7.1,>=2.7.0
|
|
41
|
+
Requires-Dist: torchaudio<2.7.1,>=2.7.0
|
|
42
|
+
Requires-Dist: torchvision<0.22.1,>=0.22.0
|
|
43
43
|
Requires-Dist: transformers>=4.51.0
|
|
44
|
-
Requires-Dist: xgrammar>=0.1.
|
|
44
|
+
Requires-Dist: xgrammar>=0.1.19
|
|
45
45
|
Provides-Extra: gpu
|
|
46
46
|
Requires-Dist: bitsandbytes==0.42.0; (sys_platform == 'darwin') and extra == 'gpu'
|
|
47
47
|
Requires-Dist: bitsandbytes>=0.45.5; (sys_platform == 'linux') and extra == 'gpu'
|
|
48
|
-
Requires-Dist: vllm==0.
|
|
48
|
+
Requires-Dist: vllm==0.9.1; (sys_platform == 'linux' or sys_platform == 'darwin') and extra == 'gpu'
|
|
49
49
|
Description-Content-Type: text/markdown
|
|
50
50
|
|
|
51
51
|
# Synthetic Data Engine 💎
|
|
@@ -91,7 +91,7 @@ pip install -U 'mostlyai-engine[gpu]'
|
|
|
91
91
|
On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
|
|
92
92
|
|
|
93
93
|
```bash
|
|
94
|
-
pip install -U torch==2.
|
|
94
|
+
pip install -U torch==2.7.0+cpu torchvision==0.22.0+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
|
|
95
95
|
```
|
|
96
96
|
|
|
97
97
|
## Quick start
|
|
@@ -41,7 +41,7 @@ pip install -U 'mostlyai-engine[gpu]'
|
|
|
41
41
|
On Linux, one can explicitly install the CPU-only variant of torch together with `mostlyai-engine`:
|
|
42
42
|
|
|
43
43
|
```bash
|
|
44
|
-
pip install -U torch==2.
|
|
44
|
+
pip install -U torch==2.7.0+cpu torchvision==0.22.0+cpu mostlyai-engine --extra-index-url https://download.pytorch.org/whl/cpu
|
|
45
45
|
```
|
|
46
46
|
|
|
47
47
|
## Quick start
|
|
@@ -22,7 +22,7 @@ from mostlyai.engine.splitting import split
|
|
|
22
22
|
from mostlyai.engine.training import train
|
|
23
23
|
|
|
24
24
|
__all__ = ["split", "analyze", "encode", "train", "generate", "init_logging", "set_random_state"]
|
|
25
|
-
__version__ = "1.4.
|
|
25
|
+
__version__ = "1.4.7"
|
|
26
26
|
|
|
27
27
|
# suppress specific warning related to os.fork() in multi-threaded processes
|
|
28
28
|
warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
|
|
@@ -18,7 +18,15 @@ from pathlib import Path
|
|
|
18
18
|
|
|
19
19
|
import torch
|
|
20
20
|
from peft import PeftConfig, prepare_model_for_kbit_training
|
|
21
|
-
from transformers import
|
|
21
|
+
from transformers import (
|
|
22
|
+
AutoConfig,
|
|
23
|
+
AutoModel,
|
|
24
|
+
AutoModelForCausalLM,
|
|
25
|
+
AutoModelForImageTextToText,
|
|
26
|
+
BitsAndBytesConfig,
|
|
27
|
+
PretrainedConfig,
|
|
28
|
+
PreTrainedModel,
|
|
29
|
+
)
|
|
22
30
|
|
|
23
31
|
from mostlyai.engine._language.lstm import LSTMFromScratchConfig
|
|
24
32
|
|
|
@@ -35,7 +43,7 @@ def is_bf16_supported(device: torch.device) -> bool:
|
|
|
35
43
|
|
|
36
44
|
|
|
37
45
|
def get_attention_implementation(config: PretrainedConfig) -> str | None:
|
|
38
|
-
model_cls =
|
|
46
|
+
model_cls = AutoModel._model_mapping[type(config)]
|
|
39
47
|
attn_implementation = None
|
|
40
48
|
if getattr(model_cls, "_supports_sdpa", False):
|
|
41
49
|
attn_implementation = "sdpa"
|
|
@@ -45,6 +53,7 @@ def get_attention_implementation(config: PretrainedConfig) -> str | None:
|
|
|
45
53
|
def load_base_model_and_config(
|
|
46
54
|
model_id_or_path: str | Path, device: torch.device, is_peft_adapter: bool, is_training: bool
|
|
47
55
|
) -> tuple[PreTrainedModel, PretrainedConfig]:
|
|
56
|
+
# opacus DP does not support parallel/sharded training
|
|
48
57
|
model_id_or_path = str(model_id_or_path)
|
|
49
58
|
if is_peft_adapter:
|
|
50
59
|
# get the base model name from adapter_config.json
|
|
@@ -84,13 +93,29 @@ def load_base_model_and_config(
|
|
|
84
93
|
)
|
|
85
94
|
else:
|
|
86
95
|
quantization_config = None
|
|
87
|
-
|
|
96
|
+
|
|
97
|
+
if device.type == "cuda" and device.index is not None:
|
|
98
|
+
device_map = str(device)
|
|
99
|
+
else:
|
|
100
|
+
device_map = "auto"
|
|
101
|
+
|
|
102
|
+
if hasattr(config, "text_config") and hasattr(config, "vision_config"):
|
|
103
|
+
config.text_config.use_cache = use_cache
|
|
104
|
+
config.text_config.attn_implementation = attn_implementation
|
|
105
|
+
auto_model_cls = AutoModelForImageTextToText
|
|
106
|
+
elif hasattr(config, "use_cache"):
|
|
107
|
+
config.use_cache = use_cache
|
|
108
|
+
config.attn_implementation = attn_implementation
|
|
109
|
+
auto_model_cls = AutoModelForCausalLM
|
|
110
|
+
else:
|
|
111
|
+
raise ValueError("Unsupported model")
|
|
112
|
+
|
|
113
|
+
model = auto_model_cls.from_pretrained(
|
|
88
114
|
model_id_or_path,
|
|
89
|
-
|
|
90
|
-
|
|
91
|
-
use_cache=use_cache,
|
|
92
|
-
device_map=device,
|
|
115
|
+
config=config,
|
|
116
|
+
device_map=device_map,
|
|
93
117
|
quantization_config=quantization_config,
|
|
118
|
+
torch_dtype=torch_dtype,
|
|
94
119
|
)
|
|
95
120
|
if quantization_config:
|
|
96
121
|
# convert all non-kbit layers to float32
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/vllm_engine.py
RENAMED
|
@@ -152,6 +152,7 @@ class VLLMEngine(LanguageEngine):
|
|
|
152
152
|
# enforce_eager=True, # results in big slowdown, but is needed when running pytest locally
|
|
153
153
|
swap_space=0,
|
|
154
154
|
disable_log_stats=True,
|
|
155
|
+
tensor_parallel_size=torch.cuda.device_count(),
|
|
155
156
|
)
|
|
156
157
|
self.tokenizer = AutoTokenizer.from_pretrained(
|
|
157
158
|
model_path,
|
|
@@ -28,13 +28,16 @@ from pathlib import Path
|
|
|
28
28
|
import numpy as np
|
|
29
29
|
import pandas as pd
|
|
30
30
|
import torch
|
|
31
|
+
from accelerate import Accelerator, FullyShardedDataParallelPlugin
|
|
31
32
|
from datasets import Dataset, DatasetDict, disable_progress_bar, load_dataset
|
|
33
|
+
from huggingface_hub import get_safetensors_metadata
|
|
32
34
|
from opacus import GradSampleModule, PrivacyEngine
|
|
33
35
|
from opacus.accountants import GaussianAccountant, PRVAccountant, RDPAccountant
|
|
34
36
|
from opacus.grad_sample import register_grad_sampler
|
|
35
37
|
from opacus.utils.batch_memory_manager import wrap_data_loader
|
|
36
38
|
from peft import LoraConfig, PeftModel
|
|
37
39
|
from torch import nn
|
|
40
|
+
from torch.distributed.fsdp.fully_sharded_data_parallel import FullOptimStateDictConfig, FullStateDictConfig
|
|
38
41
|
from torch.nn import CrossEntropyLoss
|
|
39
42
|
from torch.optim.lr_scheduler import LRScheduler
|
|
40
43
|
from torch.utils.data import DataLoader
|
|
@@ -212,6 +215,12 @@ def _calculate_val_loss(model: PreTrainedModel | GradSampleModule, val_dataloade
|
|
|
212
215
|
return val_loss_avg.item()
|
|
213
216
|
|
|
214
217
|
|
|
218
|
+
def get_num_model_params(model: str) -> int:
|
|
219
|
+
metadata = get_safetensors_metadata(model)
|
|
220
|
+
no_of_model_params = next(iter(metadata.parameter_count.values()))
|
|
221
|
+
return no_of_model_params
|
|
222
|
+
|
|
223
|
+
|
|
215
224
|
def _calculate_max_tokens(tokenized_trn_dataset: Dataset) -> int:
|
|
216
225
|
max_tokens = 0
|
|
217
226
|
for example in tokenized_trn_dataset:
|
|
@@ -294,13 +303,37 @@ def train(
|
|
|
294
303
|
) as progress:
|
|
295
304
|
_LOG.info(f"numpy={version('numpy')}, pandas={version('pandas')}")
|
|
296
305
|
_LOG.info(f"torch={version('torch')}, opacus={version('opacus')}")
|
|
297
|
-
_LOG.info(f"transformers={version('transformers')}, peft={version('peft')}")
|
|
306
|
+
_LOG.info(f"transformers={version('transformers')}, accelerate={version('accelerate')}, peft={version('peft')}")
|
|
307
|
+
with_dp = differential_privacy is not None
|
|
298
308
|
device = (
|
|
299
309
|
torch.device(device)
|
|
300
310
|
if device is not None
|
|
301
311
|
else (torch.device("cuda") if torch.cuda.is_available() else torch.device("cpu"))
|
|
302
312
|
)
|
|
313
|
+
|
|
314
|
+
single_gpu_threshold = 7_000_000_000
|
|
315
|
+
if (
|
|
316
|
+
device.type == "cuda"
|
|
317
|
+
and device.index is None
|
|
318
|
+
and (
|
|
319
|
+
with_dp or model == LSTMFromScratchConfig.model_id or get_num_model_params(model) < single_gpu_threshold
|
|
320
|
+
)
|
|
321
|
+
):
|
|
322
|
+
device = torch.device("cuda:0")
|
|
323
|
+
_LOG.info("device set to single gpu (cuda:0) because model is too small or differential privacy is enabled")
|
|
324
|
+
|
|
325
|
+
if not with_dp:
|
|
326
|
+
if device.type == "cuda":
|
|
327
|
+
fsdp_plugin = FullyShardedDataParallelPlugin(
|
|
328
|
+
state_dict_config=FullStateDictConfig(offload_to_cpu=True, rank0_only=False),
|
|
329
|
+
optim_state_dict_config=FullOptimStateDictConfig(offload_to_cpu=True, rank0_only=False),
|
|
330
|
+
)
|
|
331
|
+
else:
|
|
332
|
+
fsdp_plugin = None
|
|
333
|
+
accelerator = Accelerator(fsdp_plugin=fsdp_plugin, cpu=device.type == "cpu")
|
|
334
|
+
|
|
303
335
|
_LOG.info(f"{device=}")
|
|
336
|
+
_LOG.info(f"{torch.cuda.device_count()=}")
|
|
304
337
|
bf16_supported = is_bf16_supported(device)
|
|
305
338
|
_LOG.info(f"{bf16_supported=}")
|
|
306
339
|
use_mixed_precision = bf16_supported and model != LSTMFromScratchConfig.model_id
|
|
@@ -324,7 +357,7 @@ def train(
|
|
|
324
357
|
max_epochs = max_epochs_cap
|
|
325
358
|
else:
|
|
326
359
|
_LOG.info(f"{max_epochs=}")
|
|
327
|
-
|
|
360
|
+
|
|
328
361
|
_LOG.info(f"{with_dp=}")
|
|
329
362
|
_LOG.info(f"{model_state_strategy=}")
|
|
330
363
|
|
|
@@ -566,6 +599,11 @@ def train(
|
|
|
566
599
|
min_lr=0.1 * initial_lr,
|
|
567
600
|
# threshold=0, # if we prefer to completely mimic the behavior of previous implementation
|
|
568
601
|
)
|
|
602
|
+
is_reduce_lr_on_plateau = isinstance(lr_scheduler, torch.optim.lr_scheduler.ReduceLROnPlateau)
|
|
603
|
+
|
|
604
|
+
if not with_dp:
|
|
605
|
+
model, optimizer, lr_scheduler = accelerator.prepare(model, optimizer, lr_scheduler)
|
|
606
|
+
|
|
569
607
|
if (
|
|
570
608
|
model_state_strategy == ModelStateStrategy.resume
|
|
571
609
|
and model_checkpoint.optimizer_and_lr_scheduler_paths_exist()
|
|
@@ -626,6 +664,7 @@ def train(
|
|
|
626
664
|
else:
|
|
627
665
|
privacy_engine = None
|
|
628
666
|
dp_config, dp_total_delta, dp_accountant = None, None, None
|
|
667
|
+
trn_dataloader = accelerator.prepare(trn_dataloader)
|
|
629
668
|
|
|
630
669
|
progress_message = None
|
|
631
670
|
start_trn_time = time.time()
|
|
@@ -667,8 +706,12 @@ def train(
|
|
|
667
706
|
outputs = model(**step_data)
|
|
668
707
|
# FIXME approximation, should be divided by total sum of number of tokens in the batch
|
|
669
708
|
# as in _calculate_per_label_losses, also the final sample may be smaller than the batch size.
|
|
670
|
-
|
|
671
|
-
|
|
709
|
+
if with_dp:
|
|
710
|
+
step_loss = outputs.loss
|
|
711
|
+
step_loss.backward()
|
|
712
|
+
else:
|
|
713
|
+
step_loss = outputs.loss / gradient_accumulation_steps
|
|
714
|
+
accelerator.backward(step_loss)
|
|
672
715
|
accumulated_steps += 1
|
|
673
716
|
# explicitly count the number of processed samples as the actual batch size can vary when DP is on
|
|
674
717
|
samples += step_data["input_ids"].shape[0]
|
|
@@ -686,8 +729,9 @@ def train(
|
|
|
686
729
|
current_lr = optimizer.param_groups[0][
|
|
687
730
|
"lr"
|
|
688
731
|
] # currently assume that we have the same lr for all param groups
|
|
732
|
+
|
|
689
733
|
# only the scheduling for ReduceLROnPlateau is postponed until the metric becomes available
|
|
690
|
-
if not
|
|
734
|
+
if not is_reduce_lr_on_plateau:
|
|
691
735
|
lr_scheduler.step()
|
|
692
736
|
|
|
693
737
|
# do validation
|
|
@@ -727,7 +771,7 @@ def train(
|
|
|
727
771
|
# check for early stopping
|
|
728
772
|
do_stop = early_stopper(val_loss=val_loss) or has_exceeded_dp_max_epsilon
|
|
729
773
|
# scheduling for ReduceLROnPlateau
|
|
730
|
-
if
|
|
774
|
+
if is_reduce_lr_on_plateau:
|
|
731
775
|
lr_scheduler.step(metrics=val_loss)
|
|
732
776
|
|
|
733
777
|
# log progress, either by time or by steps, whatever is shorter
|
|
@@ -788,7 +832,7 @@ def train(
|
|
|
788
832
|
if not model_checkpoint.has_saved_once():
|
|
789
833
|
_LOG.info("saving model weights, as none were saved so far")
|
|
790
834
|
model_checkpoint.save_checkpoint(
|
|
791
|
-
model=model,
|
|
835
|
+
model=model if with_dp else accelerator.unwrap_model(model),
|
|
792
836
|
optimizer=optimizer,
|
|
793
837
|
lr_scheduler=lr_scheduler,
|
|
794
838
|
dp_accountant=privacy_engine.accountant if with_dp else None,
|
|
@@ -758,15 +758,7 @@ def generate(
|
|
|
758
758
|
_LOG.info(f"{gen_column_order=}")
|
|
759
759
|
if not enable_flexible_generation:
|
|
760
760
|
# check if resolved column order is the same as the one from training
|
|
761
|
-
trn_column_order =
|
|
762
|
-
trn_column_order += [
|
|
763
|
-
get_argn_name(
|
|
764
|
-
argn_processor=tgt_stats["columns"][col][ARGN_PROCESSOR],
|
|
765
|
-
argn_table=tgt_stats["columns"][col][ARGN_TABLE],
|
|
766
|
-
argn_column=tgt_stats["columns"][col][ARGN_COLUMN],
|
|
767
|
-
)
|
|
768
|
-
for col in tgt_stats["columns"].keys()
|
|
769
|
-
]
|
|
761
|
+
trn_column_order = get_columns_from_cardinalities(tgt_cardinalities)
|
|
770
762
|
_LOG.info(f"{trn_column_order=}")
|
|
771
763
|
if gen_column_order != trn_column_order:
|
|
772
764
|
raise ValueError(
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "mostlyai-engine"
|
|
3
|
-
version = "1.4.
|
|
3
|
+
version = "1.4.7"
|
|
4
4
|
description = "Synthetic Data Engine"
|
|
5
5
|
authors = [{ name = "MOSTLY AI", email = "dev@mostly.ai" }]
|
|
6
6
|
requires-python = ">=3.10"
|
|
@@ -25,8 +25,8 @@ classifiers = [
|
|
|
25
25
|
]
|
|
26
26
|
|
|
27
27
|
dependencies = [
|
|
28
|
-
"setuptools>=77.0.3", # similar to vllm 0.
|
|
29
|
-
"numpy>=
|
|
28
|
+
"setuptools>=77.0.3", # similar to vllm 0.9.1
|
|
29
|
+
"numpy>=2.0.0",
|
|
30
30
|
"pandas~=2.2.0",
|
|
31
31
|
"pyarrow>=16.0.0",
|
|
32
32
|
"joblib>=1.4.2",
|
|
@@ -37,19 +37,19 @@ dependencies = [
|
|
|
37
37
|
"accelerate>=1.5.0",
|
|
38
38
|
"peft>=0.12.0",
|
|
39
39
|
"huggingface-hub[hf-xet]>=0.30.2",
|
|
40
|
-
"opacus>=1.5.
|
|
41
|
-
"xgrammar>=0.1.
|
|
40
|
+
"opacus>=1.5.4",
|
|
41
|
+
"xgrammar>=0.1.19", # for vllm 0.9.1 compatibility
|
|
42
42
|
"json-repair>=0.30.0, <0.47.0", # fixes errors in tests when building engine 1.4.4
|
|
43
|
-
"torch>=2.
|
|
44
|
-
"torchaudio>=2.
|
|
45
|
-
"torchvision>=0.
|
|
43
|
+
"torch>=2.7.0,<2.7.1", # for vllm 0.9.1 compatibility
|
|
44
|
+
"torchaudio>=2.7.0,<2.7.1", # for vllm 0.9.1 compatibility
|
|
45
|
+
"torchvision>=0.22.0,<0.22.1" # for vllm 0.9.1 compatibility
|
|
46
46
|
]
|
|
47
47
|
|
|
48
48
|
[project.optional-dependencies]
|
|
49
49
|
gpu = [
|
|
50
50
|
"bitsandbytes==0.42.0; sys_platform == 'darwin'",
|
|
51
51
|
"bitsandbytes>=0.45.5; sys_platform == 'linux'",
|
|
52
|
-
"vllm==0.
|
|
52
|
+
"vllm==0.9.1; sys_platform == 'linux' or sys_platform == 'darwin'",
|
|
53
53
|
]
|
|
54
54
|
|
|
55
55
|
[dependency-groups]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/numeric.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/language/text.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/character.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/itt.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/lat_long.py
RENAMED
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_encoding_types/tabular/numeric.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/engine/hf_engine.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-1.4.5 → mostlyai_engine-1.4.7}/mostlyai/engine/_language/tokenizer_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|