mostlyai-engine 2.7.1__tar.gz → 2.7.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/PKG-INFO +1 -1
  2. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/__init__.py +1 -1
  3. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/numeric.py +6 -4
  4. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/numeric.py +16 -5
  5. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/lstm.py +4 -1
  6. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/encoding.py +2 -4
  7. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/training.py +28 -14
  8. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/analysis.py +28 -23
  9. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/pyproject.toml +1 -1
  10. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/.gitignore +0 -0
  11. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/LICENSE +0 -0
  12. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/README.md +0 -0
  13. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_common.py +0 -0
  14. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_dtypes.py +0 -0
  15. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/__init__.py +0 -0
  16. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
  17. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
  18. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
  19. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/text.py +0 -0
  20. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
  21. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
  22. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
  23. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
  24. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
  25. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
  26. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/__init__.py +0 -0
  27. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/common.py +0 -0
  28. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/encoding.py +0 -0
  29. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/__init__.py +0 -0
  30. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/base.py +0 -0
  31. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
  32. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/vllm_engine.py +0 -0
  33. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/generation.py +0 -0
  34. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/interface.py +0 -0
  35. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
  36. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/training.py +0 -0
  37. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/xgrammar_hf_logits.py +0 -0
  38. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/xgrammar_utils.py +0 -0
  39. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_memory.py +0 -0
  40. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/__init__.py +0 -0
  41. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/argn.py +0 -0
  42. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/common.py +0 -0
  43. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/fairness.py +0 -0
  44. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/generation.py +0 -0
  45. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/interface.py +0 -0
  46. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/probability.py +0 -0
  47. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_training_utils.py +0 -0
  48. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/_workspace.py +0 -0
  49. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/domain.py +0 -0
  50. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/encoding.py +0 -0
  51. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/generation.py +0 -0
  52. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/logging.py +0 -0
  53. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/random_state.py +0 -0
  54. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/splitting.py +0 -0
  55. {mostlyai_engine-2.7.1 → mostlyai_engine-2.7.2}/mostlyai/engine/training.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: mostlyai-engine
3
- Version: 2.7.1
3
+ Version: 2.7.2
4
4
  Summary: Synthetic Data Engine
5
5
  Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
6
6
  Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
@@ -34,7 +34,7 @@ __all__ = [
34
34
  "TabularARGN",
35
35
  "LanguageModel",
36
36
  ]
37
- __version__ = "2.7.1"
37
+ __version__ = "2.7.2"
38
38
 
39
39
  # suppress specific warning related to os.fork() in multi-threaded processes
40
40
  warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
@@ -43,8 +43,8 @@ def analyze_language_numeric(values: pd.Series, root_keys: pd.Series, _: pd.Seri
43
43
 
44
44
  # determine max scale
45
45
  def count_scale(num: float) -> int:
46
- # represent number as fixed point string, remove trailing zeros and decimal point
47
- num = format(num, "f").rstrip("0").rstrip(".")
46
+ # preserve fractional precision without scientific notation or trailing zeros
47
+ num = np.format_float_positional(num, unique=True, trim="-")
48
48
  if "." in num:
49
49
  # in case of decimal, return number of digits after decimal point
50
50
  return len(num.split(".")[1])
@@ -133,11 +133,13 @@ def encode_language_numeric(values: pd.Series, stats: dict, _: pd.Series | None
133
133
  def decode_language_numeric(x: pd.Series, stats: dict[str, str]) -> pd.Series:
134
134
  x = pd.to_numeric(x, errors="coerce")
135
135
  x = x.round(stats["max_scale"])
136
+ # Nullable pandas numeric dtypes expose their corresponding NumPy dtype.
137
+ np_dtype = np.dtype(getattr(x.dtype, "numpy_dtype", x.dtype))
136
138
  if stats["min"] is not None:
137
- reduced_min = np.dtype(x.dtype).type(stats["min"])
139
+ reduced_min = np_dtype.type(stats["min"])
138
140
  x.loc[x < reduced_min] = reduced_min
139
141
  if stats["max"] is not None:
140
- reduced_max = np.dtype(x.dtype).type(stats["max"])
142
+ reduced_max = np_dtype.type(stats["max"])
141
143
  x.loc[x > reduced_max] = reduced_max
142
144
  dtype = "Int64" if stats["max_scale"] == 0 else float
143
145
  return x.astype(dtype)
@@ -77,6 +77,7 @@ NUMERIC_DISCRETE_NULL_TOKEN = CATEGORICAL_NULL_TOKEN
77
77
  # maximum and minimum precision that is being considered
78
78
  NUMERIC_DIGIT_MAX_DECIMAL = 18
79
79
  NUMERIC_DIGIT_MIN_DECIMAL = -8
80
+ NUMERIC_DIGIT_MAX_SCALE = 20
80
81
 
81
82
 
82
83
  def _type_safe_numeric_series(numeric_array: np.ndarray | list, pd_dtype: str) -> pd.Series:
@@ -136,7 +137,11 @@ def split_sub_columns_digit(
136
137
  # rely on `np.format_float_positional` to determine string representation of absolute values
137
138
  values_str = (
138
139
  values.abs()
139
- .apply(lambda x: np.format_float_positional(x, unique=True, pad_left=50, pad_right=20, precision=20))
140
+ .apply(
141
+ lambda x: np.format_float_positional(
142
+ x, unique=True, pad_left=50, pad_right=NUMERIC_DIGIT_MAX_SCALE, precision=NUMERIC_DIGIT_MAX_SCALE
143
+ )
144
+ )
140
145
  # convert to string[pyarrow] for faster processing
141
146
  .astype("string[pyarrow]")
142
147
  # replace nan with pd.NA for faster processing
@@ -199,7 +204,13 @@ def analyze_numeric(
199
204
  max_n = max_values.sort_values(ascending=False).head(ANALYZE_MIN_MAX_TOP_N).astype("float").tolist()
200
205
 
201
206
  # split values into digits; used for digit numeric encoding, plus to determine precision
202
- df_split = split_sub_columns_digit(values)
207
+ # Extend the default digit window for fine fractions, up to the formatter's precision.
208
+ max_scale = max(
209
+ (len(np.format_float_positional(v, unique=True, trim="-").partition(".")[2]) for v in non_na_values),
210
+ default=0,
211
+ )
212
+ min_decimal = min(NUMERIC_DIGIT_MIN_DECIMAL, -min(max_scale, NUMERIC_DIGIT_MAX_SCALE))
213
+ df_split = split_sub_columns_digit(values, min_decimal=min_decimal)
203
214
  is_not_nan = df_split["nan"] == 0
204
215
  has_nan = sum(df_split["nan"]) > 0
205
216
  has_neg = sum(df_split["neg"]) > 0
@@ -238,9 +249,9 @@ def analyze_reduce_numeric(
238
249
  # check if there are negative values
239
250
  has_neg = any([j["has_neg"] for j in stats_list])
240
251
  # determine precision to apply rounding of sampled values during generation
241
- keys = stats_list[0]["max_digits"].keys()
242
- min_digits = {k: min([j["min_digits"][k] for j in stats_list]) for k in keys}
243
- max_digits = {k: max([j["max_digits"][k] for j in stats_list]) for k in keys}
252
+ keys = sorted({k for stats in stats_list for k in stats["max_digits"]}, key=lambda k: int(k[1:]), reverse=True)
253
+ min_digits = {k: min([j["min_digits"].get(k, 0) for j in stats_list]) for k in keys}
254
+ max_digits = {k: max([j["max_digits"].get(k, 0) for j in stats_list]) for k in keys}
244
255
  non_zero_prec = [k for k in keys if max_digits[k] > 0 and k.startswith("E")]
245
256
  min_decimal = min([int(k[1:]) for k in non_zero_prec]) if len(non_zero_prec) > 0 else 0
246
257
 
@@ -140,13 +140,16 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
140
140
  )
141
141
 
142
142
  def prepare_inputs_for_generation(
143
- self, input_ids: torch.Tensor, attention_mask: torch.Tensor, **kwargs
143
+ self, input_ids: torch.Tensor, attention_mask: torch.Tensor | None = None, **kwargs
144
144
  ) -> dict[str, torch.Tensor]:
145
145
  """
146
146
  This function is mandatory so that the model is able to use the Hugging Face `.generate()` method.
147
147
  Since `.generate()` works with left-padded sequences but the model is trained with right-padded sequences,
148
148
  we need to convert the padding side here to make it work properly.
149
149
  """
150
+ # Transformers may omit an all-ones mask for an unpadded batch.
151
+ if attention_mask is None:
152
+ attention_mask = torch.ones_like(input_ids)
150
153
  lengths = attention_mask.sum(dim=1)
151
154
  return {
152
155
  "input_ids": self.left_to_right_padding(input_ids, lengths),
@@ -19,7 +19,7 @@ from pathlib import Path
19
19
 
20
20
  import numpy as np
21
21
  import pandas as pd
22
- from joblib import Parallel, cpu_count, delayed, parallel_config
22
+ from joblib import Parallel, cpu_count, delayed
23
23
 
24
24
  from mostlyai.engine._common import (
25
25
  ARGN_COLUMN,
@@ -238,9 +238,7 @@ def encode_df(
238
238
  )
239
239
  )
240
240
  if delayed_encodes:
241
- with parallel_config("loky", n_jobs=n_jobs):
242
- df_columns.extend(Parallel()(delayed_encodes))
243
-
241
+ df_columns.extend(Parallel(n_jobs=n_jobs)(delayed_encodes))
244
242
  df = pd.concat(df_columns, axis=1) if df_columns else pd.DataFrame()
245
243
  return df, ctx_primary_key, tgt_context_key
246
244
 
@@ -154,6 +154,8 @@ class BatchCollator:
154
154
  self.use_nested_ctxseq = use_nested_ctxseq
155
155
 
156
156
  def __call__(self, batch: list[dict]) -> dict[str, torch.Tensor]:
157
+ if not batch:
158
+ return {}
157
159
  batch = pd.DataFrame(batch)
158
160
  if self.is_sequential and self.max_sequence_window:
159
161
  batch = self._slice_sequences(batch, self.max_sequence_window)
@@ -678,6 +680,9 @@ def train(
678
680
  max_grad_norm=dp_config.get("max_grad_norm"),
679
681
  poisson_sampling=True,
680
682
  )
683
+ # Our dictionary collator represents empty batches as {}, including the first
684
+ # batch, which Opacus cannot infer from the raw dictionary dataset.
685
+ trn_dataloader.collate_fn = batch_collator
681
686
  # this further wraps the dataloader with batch_sampler=BatchSplittingSampler to achieve gradient accumulation
682
687
  # it will split the sampled logical batches into smaller sub-batches with batch_size
683
688
  trn_dataloader = wrap_data_loader(
@@ -712,22 +717,31 @@ def train(
712
717
  except StopIteration:
713
718
  trn_data_iter = iter(trn_dataloader)
714
719
  step_data = next(trn_data_iter)
715
- # forward pass + calculate sample losses
716
- step_losses = _calculate_sample_losses(argn, step_data)
717
- # FIXME in sequential case, this is an approximation, it should be divided by total sum of masks in the
718
- # entire batch to get the average loss per sample. Less importantly the final sample may be smaller
719
- # than the batch size in both flat and sequential case.
720
- # calculate total step loss
721
- step_loss = torch.mean(step_losses) / (1 if with_dp else gradient_accumulation_steps)
722
720
  if with_dp:
723
- # opacus handles the gradient accumulation internally
724
721
  optimizer.zero_grad(set_to_none=True)
725
- # backward pass
726
- with warnings.catch_warnings():
727
- warnings.filterwarnings("ignore", category=FutureWarning, message="Using a non-full backward hook*")
728
- if with_dp:
729
- warnings.filterwarnings("ignore", category=UserWarning, message="Full backward hook is firing*")
730
- step_loss.backward()
722
+ if with_dp and not step_data:
723
+ # Poisson sampling can produce empty logical batches. Bypass the
724
+ # RNN forward pass, but retain Opacus noise and accounting below.
725
+ for parameter in optimizer.params:
726
+ parameter.grad_sample = parameter.new_empty((0, *parameter.shape))
727
+ step_losses = torch.empty(0, device=device)
728
+ else:
729
+ # forward pass + calculate sample losses
730
+ step_losses = _calculate_sample_losses(argn, step_data)
731
+ # FIXME in sequential case, this is an approximation, it should be divided by total sum of masks in the
732
+ # entire batch to get the average loss per sample. Less importantly the final sample may be smaller
733
+ # than the batch size in both flat and sequential case.
734
+ step_loss = torch.mean(step_losses) / (1 if with_dp else gradient_accumulation_steps)
735
+ # backward pass
736
+ with warnings.catch_warnings():
737
+ warnings.filterwarnings(
738
+ "ignore", category=FutureWarning, message="Using a non-full backward hook*"
739
+ )
740
+ if with_dp:
741
+ warnings.filterwarnings(
742
+ "ignore", category=UserWarning, message="Full backward hook is firing*"
743
+ )
744
+ step_loss.backward()
731
745
  accumulated_steps += 1
732
746
  # explicitly count the number of processed samples as the actual batch size can vary when DP is on
733
747
  samples += step_losses.shape[0]
@@ -17,6 +17,7 @@ Provides analysis functionality of the engine
17
17
  """
18
18
 
19
19
  import logging
20
+ import os
20
21
  import time
21
22
  from collections.abc import Iterable
22
23
  from pathlib import Path
@@ -24,7 +25,7 @@ from typing import Any, Literal
24
25
 
25
26
  import numpy as np
26
27
  import pandas as pd
27
- from joblib import Parallel, cpu_count, delayed, parallel_config
28
+ from joblib import Parallel, cpu_count, delayed
28
29
 
29
30
  from mostlyai.engine._common import (
30
31
  ANALYZE_REDUCE_MIN_MAX_N,
@@ -251,18 +252,21 @@ def _analyze_partition(
251
252
  else:
252
253
  ctx_root_keys = ctx_primary_keys.rename("__rkey")
253
254
 
255
+ # Unique row IDs suffice for root-level counts; reuse them across columns.
256
+ tgt_root_keys = pd.Series(np.arange(len(tgt_df)), index=tgt_df.index, name="root_keys")
257
+
254
258
  # analyze all target columns
255
- with parallel_config("loky", n_jobs=n_jobs):
256
- results = Parallel()(
257
- delayed(_analyze_col)(
258
- values=tgt_df[column],
259
- encoding_type=encoding_type,
260
- context_keys=tgt_context_keys,
261
- )
262
- for column, encoding_type in tgt_encoding_types.items()
259
+ results = Parallel(n_jobs=n_jobs)(
260
+ delayed(_analyze_col)(
261
+ values=tgt_df[column],
262
+ root_keys=tgt_root_keys,
263
+ parent_pid=os.getpid(),
264
+ encoding_type=encoding_type,
265
+ context_keys=tgt_context_keys,
263
266
  )
264
- tgt_column_stats = {column: stats for column, stats in zip(tgt_encoding_types.keys(), results)}
265
-
267
+ for column, encoding_type in tgt_encoding_types.items()
268
+ )
269
+ tgt_column_stats = {column: stats for column, stats in zip(tgt_encoding_types.keys(), results)}
266
270
  # collect target sequence length stats
267
271
  tgt_seq_len = _analyze_seq_len(
268
272
  tgt_context_keys=tgt_context_keys,
@@ -293,17 +297,16 @@ def _analyze_partition(
293
297
 
294
298
  # analyze all context columns
295
299
  assert isinstance(ctx_encoding_types, dict)
296
- with parallel_config("loky", n_jobs=n_jobs):
297
- results = Parallel()(
298
- delayed(_analyze_col)(
299
- values=ctx_df[column],
300
- encoding_type=encoding_type,
301
- root_keys=ctx_root_keys,
302
- )
303
- for column, encoding_type in ctx_encoding_types.items()
300
+ results = Parallel(n_jobs=n_jobs)(
301
+ delayed(_analyze_col)(
302
+ values=ctx_df[column],
303
+ parent_pid=os.getpid(),
304
+ encoding_type=encoding_type,
305
+ root_keys=ctx_root_keys,
304
306
  )
305
- ctx_column_stats = {column: stats for column, stats in zip(ctx_encoding_types.keys(), results)}
306
-
307
+ for column, encoding_type in ctx_encoding_types.items()
308
+ )
309
+ ctx_column_stats = {column: stats for column, stats in zip(ctx_encoding_types.keys(), results)}
307
310
  # persist context stats
308
311
  assert isinstance(ctx_stats_path, Path) and ctx_stats_path.exists()
309
312
  ctx_stats_file = ctx_stats_path / f"part.{partition_id}.json"
@@ -516,8 +519,10 @@ def _analyze_col(
516
519
  encoding_type: ModelEncodingType,
517
520
  root_keys: pd.Series | None = None,
518
521
  context_keys: pd.Series | None = None,
522
+ parent_pid: int | None = None,
519
523
  ) -> dict:
520
- set_random_state(worker=True)
524
+ if os.getpid() != parent_pid:
525
+ set_random_state(worker=True)
521
526
 
522
527
  stats: dict = {"encoding_type": encoding_type}
523
528
 
@@ -526,7 +531,7 @@ def _analyze_col(
526
531
  return stats
527
532
 
528
533
  if root_keys is None:
529
- root_keys = pd.Series([str(i) for i in range(len(values))], name="root_keys")
534
+ root_keys = pd.Series(np.arange(len(values)), index=values.index, name="root_keys")
530
535
 
531
536
  if is_sequential(values):
532
537
  # analyze sequential column
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "mostlyai-engine"
3
- version = "2.7.1"
3
+ version = "2.7.2"
4
4
  description = "Synthetic Data Engine"
5
5
  requires-python = ">=3.11,<3.14"
6
6
  readme = "README.md"
File without changes