mostlyai-engine 2.6.1__tar.gz → 2.6.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/PKG-INFO +3 -3
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/__init__.py +1 -1
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/training.py +13 -6
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/argn.py +2 -1
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/training.py +35 -11
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/pyproject.toml +3 -3
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/.gitignore +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/LICENSE +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/README.md +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_common.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_dtypes.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/text.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/numeric.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/common.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/encoding.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/base.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/vllm_engine.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/generation.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/interface.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/lstm.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/xgrammar_hf_logits.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/xgrammar_utils.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_memory.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/__init__.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/common.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/encoding.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/fairness.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/generation.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/interface.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_tabular/probability.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_training_utils.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_workspace.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/analysis.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/domain.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/encoding.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/generation.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/logging.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/random_state.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/splitting.py +0 -0
- {mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/training.py +0 -0
|
@@ -1,11 +1,11 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mostlyai-engine
|
|
3
|
-
Version: 2.6.
|
|
3
|
+
Version: 2.6.2
|
|
4
4
|
Summary: Synthetic Data Engine
|
|
5
5
|
Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
|
|
6
6
|
Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
|
|
7
|
+
Project-URL: issues, https://github.com/mostly-ai/mostlyai-engine/issues
|
|
7
8
|
Project-URL: documentation, https://mostly-ai.github.io/mostlyai-engine/
|
|
8
|
-
Author-email: MOSTLY AI <dev@mostly.ai>
|
|
9
9
|
License-Expression: Apache-2.0
|
|
10
10
|
License-File: LICENSE
|
|
11
11
|
Classifier: Development Status :: 5 - Production/Stable
|
|
@@ -29,7 +29,7 @@ Requires-Dist: huggingface-hub[hf-xet]>=0.30.2
|
|
|
29
29
|
Requires-Dist: joblib>=1.4.2
|
|
30
30
|
Requires-Dist: json-repair>=0.47.0
|
|
31
31
|
Requires-Dist: numpy>=2.0.0
|
|
32
|
-
Requires-Dist: opacus>=1.
|
|
32
|
+
Requires-Dist: opacus>=1.6.0
|
|
33
33
|
Requires-Dist: pandas>=2.2.0
|
|
34
34
|
Requires-Dist: peft>=0.18.2
|
|
35
35
|
Requires-Dist: psutil<6,>=5.9.5
|
|
@@ -34,7 +34,7 @@ __all__ = [
|
|
|
34
34
|
"TabularARGN",
|
|
35
35
|
"LanguageModel",
|
|
36
36
|
]
|
|
37
|
-
__version__ = "2.6.
|
|
37
|
+
__version__ = "2.6.2"
|
|
38
38
|
|
|
39
39
|
# suppress specific warning related to os.fork() in multi-threaded processes
|
|
40
40
|
warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
|
|
@@ -31,7 +31,7 @@ from datasets import Dataset, DatasetDict, disable_progress_bar, load_dataset
|
|
|
31
31
|
from huggingface_hub import get_safetensors_metadata
|
|
32
32
|
from opacus import GradSampleModule, PrivacyEngine
|
|
33
33
|
from opacus.accountants import GaussianAccountant, PRVAccountant, RDPAccountant
|
|
34
|
-
from opacus.grad_sample import register_grad_sampler
|
|
34
|
+
from opacus.grad_sample import GradSampleHooks, register_grad_sampler
|
|
35
35
|
from opacus.utils.batch_memory_manager import wrap_data_loader
|
|
36
36
|
from peft import LoraConfig, PeftModel
|
|
37
37
|
from torch import nn
|
|
@@ -626,6 +626,7 @@ def train(
|
|
|
626
626
|
# this can help accelerate GPU compute
|
|
627
627
|
torch.backends.cudnn.benchmark = True
|
|
628
628
|
|
|
629
|
+
dp_grad_sample_hooks: GradSampleHooks | None = None
|
|
629
630
|
if with_dp:
|
|
630
631
|
if isinstance(differential_privacy, DifferentialPrivacyConfig):
|
|
631
632
|
dp_config = differential_privacy.model_dump()
|
|
@@ -650,18 +651,21 @@ def train(
|
|
|
650
651
|
privacy_engine.accountant.load_state_dict(
|
|
651
652
|
torch.load(workspace.model_dp_accountant_path, map_location=device, weights_only=True),
|
|
652
653
|
)
|
|
653
|
-
# Opacus
|
|
654
|
-
#
|
|
655
|
-
# -
|
|
656
|
-
# -
|
|
657
|
-
|
|
654
|
+
# Opacus returns GradSampleHooks when wrap_model=False: hooks attach to the original module so HF /
|
|
655
|
+
# Transformers sees an unwrapped PreTrainedModel (requires Opacus >= 1.6).
|
|
656
|
+
# - dp_grad_sample_hooks: must call .cleanup() after training to remove backward hooks and param attrs
|
|
657
|
+
# - optimizer: wrapped in DPOptimizer (virtual vs logical steps)
|
|
658
|
+
# - dataloader: UniformWithReplacementSampler when poisson_sampling=True
|
|
659
|
+
dp_grad_sample_hooks, optimizer, trn_dataloader = privacy_engine.make_private(
|
|
658
660
|
module=model,
|
|
659
661
|
optimizer=optimizer,
|
|
660
662
|
data_loader=trn_dataloader,
|
|
661
663
|
noise_multiplier=dp_config.get("noise_multiplier"),
|
|
662
664
|
max_grad_norm=dp_config.get("max_grad_norm"),
|
|
663
665
|
poisson_sampling=True,
|
|
666
|
+
wrap_model=False,
|
|
664
667
|
)
|
|
668
|
+
model = dp_grad_sample_hooks._module
|
|
665
669
|
# this further wraps the dataloader with batch_sampler=BatchSplittingSampler to achieve gradient accumulation
|
|
666
670
|
# it will split the sampled logical batches into smaller sub-batches with batch_size
|
|
667
671
|
trn_dataloader = wrap_data_loader(
|
|
@@ -835,6 +839,9 @@ def train(
|
|
|
835
839
|
if total_training_time > max_training_time:
|
|
836
840
|
do_stop = True
|
|
837
841
|
|
|
842
|
+
if dp_grad_sample_hooks is not None:
|
|
843
|
+
dp_grad_sample_hooks.cleanup()
|
|
844
|
+
|
|
838
845
|
# no checkpoint is saved yet because the training stopped before the first epoch ended
|
|
839
846
|
if not model_checkpoint.has_saved_once():
|
|
840
847
|
_LOG.info("saving model weights, as none were saved so far")
|
|
@@ -289,7 +289,8 @@ class SequentialContextEmbedders(Embedders):
|
|
|
289
289
|
mask = None
|
|
290
290
|
for sub_col in self.cardinalities:
|
|
291
291
|
xs = torch.as_tensor(x[sub_col], device=self.device)
|
|
292
|
-
xs
|
|
292
|
+
if xs.is_nested:
|
|
293
|
+
xs = torch.nested.to_padded_tensor(xs, padding=-1)
|
|
293
294
|
mask = (xs != -1).squeeze(-1)
|
|
294
295
|
xs = torch.where(xs == -1, torch.tensor(0), xs)
|
|
295
296
|
xs = self.get(sub_col)(xs)
|
|
@@ -138,10 +138,20 @@ class BatchCollator:
|
|
|
138
138
|
For sequence data, it will sample subsequences with lengths up to max_sequence_window.
|
|
139
139
|
"""
|
|
140
140
|
|
|
141
|
-
def __init__(
|
|
141
|
+
def __init__(
|
|
142
|
+
self,
|
|
143
|
+
is_sequential: bool,
|
|
144
|
+
max_sequence_window: int | None,
|
|
145
|
+
device: torch.device,
|
|
146
|
+
*,
|
|
147
|
+
use_nested_ctxseq: bool = True,
|
|
148
|
+
):
|
|
142
149
|
self.is_sequential = is_sequential
|
|
143
150
|
self.max_sequence_window = max_sequence_window
|
|
144
151
|
self.device = device
|
|
152
|
+
# Opacus per-sample gradients do not support NestedTensor on CPU/CUDA; use padded
|
|
153
|
+
# dense tensors for CTXSEQ when training with DP (see test_tabular_sequential DP path).
|
|
154
|
+
self.use_nested_ctxseq = use_nested_ctxseq
|
|
145
155
|
|
|
146
156
|
def __call__(self, batch: list[dict]) -> dict[str, torch.Tensor]:
|
|
147
157
|
batch = pd.DataFrame(batch)
|
|
@@ -177,15 +187,26 @@ class BatchCollator:
|
|
|
177
187
|
dim=-1,
|
|
178
188
|
)
|
|
179
189
|
elif column.startswith(CTXSEQ):
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
torch.
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
190
|
+
if self.use_nested_ctxseq:
|
|
191
|
+
# construct row tensors and convert the list to nested column tensor
|
|
192
|
+
tensors[column] = torch.unsqueeze(
|
|
193
|
+
torch.nested.as_nested_tensor(
|
|
194
|
+
[torch.tensor(row, dtype=torch.int64, device=self.device) for row in batch[column]],
|
|
195
|
+
dtype=torch.int64,
|
|
196
|
+
device=self.device,
|
|
197
|
+
),
|
|
198
|
+
dim=-1,
|
|
199
|
+
)
|
|
200
|
+
else:
|
|
201
|
+
# padded batch (variable-length rows); -1 marks padding (matches SequentialContextEmbedders)
|
|
202
|
+
tensors[column] = torch.unsqueeze(
|
|
203
|
+
torch.tensor(
|
|
204
|
+
np.array(list(zip_longest(*batch[column], fillvalue=-1))).T,
|
|
205
|
+
dtype=torch.int64,
|
|
206
|
+
device=self.device,
|
|
207
|
+
),
|
|
208
|
+
dim=-1,
|
|
209
|
+
)
|
|
189
210
|
return tensors
|
|
190
211
|
|
|
191
212
|
@staticmethod
|
|
@@ -544,7 +565,10 @@ def train(
|
|
|
544
565
|
|
|
545
566
|
# and see if it's possible to make it compatible with DP
|
|
546
567
|
batch_collator = BatchCollator(
|
|
547
|
-
is_sequential=is_sequential,
|
|
568
|
+
is_sequential=is_sequential,
|
|
569
|
+
max_sequence_window=max_sequence_window,
|
|
570
|
+
device=device,
|
|
571
|
+
use_nested_ctxseq=not with_dp,
|
|
548
572
|
)
|
|
549
573
|
disable_progress_bar()
|
|
550
574
|
trn_dataset = load_dataset("parquet", data_files=[str(p) for p in workspace.encoded_data_trn.fetch_all()])[
|
|
@@ -1,8 +1,7 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "mostlyai-engine"
|
|
3
|
-
version = "2.6.
|
|
3
|
+
version = "2.6.2"
|
|
4
4
|
description = "Synthetic Data Engine"
|
|
5
|
-
authors = [{ name = "MOSTLY AI", email = "dev@mostly.ai" }]
|
|
6
5
|
requires-python = ">=3.11,<3.14"
|
|
7
6
|
readme = "README.md"
|
|
8
7
|
license = "Apache-2.0"
|
|
@@ -37,7 +36,7 @@ dependencies = [
|
|
|
37
36
|
"accelerate>=1.5.0",
|
|
38
37
|
"peft>=0.18.2", # transformers 5.7+ checks min PEFT in model.add_adapter (integrations/peft.py)
|
|
39
38
|
"huggingface-hub[hf-xet]>=0.30.2",
|
|
40
|
-
"opacus>=1.
|
|
39
|
+
"opacus>=1.6.0",
|
|
41
40
|
"xgrammar>=0.1.32,<1.0.0", # aligned with vllm 0.20
|
|
42
41
|
"json-repair>=0.47.0",
|
|
43
42
|
"torch>=2.11.0,<2.12.0",
|
|
@@ -75,6 +74,7 @@ docs = [
|
|
|
75
74
|
[project.urls]
|
|
76
75
|
homepage = "https://github.com/mostly-ai/mostlyai-engine"
|
|
77
76
|
repository = "https://github.com/mostly-ai/mostlyai-engine"
|
|
77
|
+
issues = "https://github.com/mostly-ai/mostlyai-engine/issues"
|
|
78
78
|
documentation = "https://mostly-ai.github.io/mostlyai-engine/"
|
|
79
79
|
|
|
80
80
|
[tool.uv]
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/numeric.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/language/text.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/character.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/itt.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/lat_long.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_encoding_types/tabular/numeric.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/hf_engine.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/engine/vllm_engine.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/tokenizer_utils.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.6.1 → mostlyai_engine-2.6.2}/mostlyai/engine/_language/xgrammar_hf_logits.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|