mostlyai-engine 2.4.0__tar.gz → 2.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (55) hide show
  1. mostlyai_engine-2.5.0/LICENSE_HEADER +13 -0
  2. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/PKG-INFO +4 -4
  3. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/__init__.py +1 -1
  4. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_common.py +6 -3
  5. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/character.py +3 -1
  6. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/datetime.py +6 -5
  7. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/xgrammar_utils.py +11 -2
  8. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/common.py +1 -1
  9. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/generation.py +6 -5
  10. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/pyproject.toml +5 -5
  11. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/.gitignore +0 -0
  12. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/LICENSE +0 -0
  13. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/README.md +0 -0
  14. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_dtypes.py +0 -0
  15. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/__init__.py +0 -0
  16. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
  17. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
  18. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
  19. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
  20. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/language/text.py +0 -0
  21. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
  22. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
  23. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
  24. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
  25. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_encoding_types/tabular/numeric.py +0 -0
  26. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/__init__.py +0 -0
  27. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/common.py +0 -0
  28. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/encoding.py +0 -0
  29. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/engine/__init__.py +0 -0
  30. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/engine/base.py +0 -0
  31. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
  32. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/engine/vllm_engine.py +0 -0
  33. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/generation.py +0 -0
  34. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/interface.py +0 -0
  35. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/lstm.py +0 -0
  36. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
  37. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_language/training.py +0 -0
  38. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_memory.py +0 -0
  39. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/__init__.py +0 -0
  40. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/argn.py +0 -0
  41. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/encoding.py +0 -0
  42. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/fairness.py +0 -0
  43. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/interface.py +0 -0
  44. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/probability.py +0 -0
  45. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_tabular/training.py +0 -0
  46. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_training_utils.py +0 -0
  47. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/_workspace.py +0 -0
  48. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/analysis.py +0 -0
  49. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/domain.py +0 -0
  50. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/encoding.py +0 -0
  51. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/generation.py +0 -0
  52. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/logging.py +0 -0
  53. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/random_state.py +0 -0
  54. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/splitting.py +0 -0
  55. {mostlyai_engine-2.4.0 → mostlyai_engine-2.5.0}/mostlyai/engine/training.py +0 -0
@@ -0,0 +1,13 @@
1
+ Copyright 2025 MOSTLY AI
2
+
3
+ Licensed under the Apache License, Version 2.0 (the "License");
4
+ you may not use this file except in compliance with the License.
5
+ You may obtain a copy of the License at
6
+
7
+ http://www.apache.org/licenses/LICENSE-2.0
8
+
9
+ Unless required by applicable law or agreed to in writing, software
10
+ distributed under the License is distributed on an "AS IS" BASIS,
11
+ WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
12
+ See the License for the specific language governing permissions and
13
+ limitations under the License.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: mostlyai-engine
3
- Version: 2.4.0
3
+ Version: 2.5.0
4
4
  Summary: Synthetic Data Engine
5
5
  Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
6
6
  Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
@@ -8,6 +8,7 @@ Project-URL: documentation, https://mostly-ai.github.io/mostlyai-engine/
8
8
  Author-email: MOSTLY AI <dev@mostly.ai>
9
9
  License-Expression: Apache-2.0
10
10
  License-File: LICENSE
11
+ License-File: LICENSE_HEADER
11
12
  Classifier: Development Status :: 5 - Production/Stable
12
13
  Classifier: Intended Audience :: Developers
13
14
  Classifier: Intended Audience :: Financial and Insurance Industry
@@ -17,13 +18,12 @@ Classifier: Intended Audience :: Science/Research
17
18
  Classifier: Intended Audience :: Telecommunications Industry
18
19
  Classifier: License :: OSI Approved :: Apache Software License
19
20
  Classifier: Operating System :: OS Independent
20
- Classifier: Programming Language :: Python :: 3.10
21
21
  Classifier: Programming Language :: Python :: 3.11
22
22
  Classifier: Programming Language :: Python :: 3.12
23
23
  Classifier: Programming Language :: Python :: 3.13
24
24
  Classifier: Topic :: Software Development :: Libraries
25
25
  Classifier: Typing :: Typed
26
- Requires-Python: >=3.10
26
+ Requires-Python: <3.14,>=3.11
27
27
  Requires-Dist: accelerate>=1.5.0
28
28
  Requires-Dist: datasets>=3.0.0
29
29
  Requires-Dist: huggingface-hub[hf-xet]>=0.30.2
@@ -41,7 +41,7 @@ Requires-Dist: tokenizers>=0.21.0
41
41
  Requires-Dist: torch<2.10.0,>=2.9.0
42
42
  Requires-Dist: torchaudio<2.10.0,>=2.9.0
43
43
  Requires-Dist: torchvision<0.25.0,>=0.24.0
44
- Requires-Dist: transformers>=4.55.0
44
+ Requires-Dist: transformers<5,>=4.55.0
45
45
  Requires-Dist: xgrammar>=0.1.21
46
46
  Provides-Extra: gpu
47
47
  Requires-Dist: bitsandbytes==0.42.0; (sys_platform == 'darwin') and extra == 'gpu'
@@ -34,7 +34,7 @@ __all__ = [
34
34
  "TabularARGN",
35
35
  "LanguageModel",
36
36
  ]
37
- __version__ = "2.4.0"
37
+ __version__ = "2.5.0"
38
38
 
39
39
  # suppress specific warning related to os.fork() in multi-threaded processes
40
40
  warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
@@ -273,8 +273,9 @@ def safe_convert_datetime(values: pd.Series, date_only: bool = False) -> pd.Seri
273
273
  utc=True,
274
274
  dayfirst=False, # assume 1/3/2020 is Jan 3
275
275
  )
276
- # check whether firstday=True yields less non-NA, and if so, switch to using that flag
277
- if values_parsed_fixed.isna().sum() > values.isna().sum():
276
+ has_slash_dates = values.astype("string").str.contains(r"^\s*\d{1,2}/\d{1,2}/\d{4}(?:\s|$)", regex=True).any()
277
+ # some mixed-format slash dates are interpreted more reliably with dayfirst=True
278
+ if has_slash_dates and values_parsed_fixed.isna().sum() > values.isna().sum():
278
279
  values_parsed_fixed_dayfirst = pd.to_datetime(
279
280
  values,
280
281
  errors="coerce", # silently map invalid dates to NA
@@ -937,6 +938,7 @@ def impute_from_non_nan_distribution(values: pd.Series, column_stats: dict) -> t
937
938
  Returns:
938
939
  tuple[pd.Series, pd.Series]: The series with imputed values and the mask of NaNs.
939
940
  """
941
+ values = values.copy()
940
942
  nan_mask = values.isna()
941
943
  vc = values.value_counts(normalize=True)
942
944
  if vc.empty:
@@ -944,7 +946,8 @@ def impute_from_non_nan_distribution(values: pd.Series, column_stats: dict) -> t
944
946
  probs = vc.values
945
947
  categories = vc.index
946
948
  # NOTE: an alternative will be to use the largest remainder method
947
- values[nan_mask] = np.random.choice(categories, size=nan_mask.sum(), p=probs)
949
+ if nan_mask.any():
950
+ values.loc[nan_mask] = np.random.choice(categories, size=nan_mask.sum(), p=probs)
948
951
  return values, nan_mask.astype(int)
949
952
 
950
953
 
@@ -102,7 +102,9 @@ def encode_character(values: pd.Series, stats: dict, _: pd.Series | None = None)
102
102
  df_split = split_sub_columns_character(values, max_string_length)
103
103
  for idx in range(max_string_length):
104
104
  sub_col = f"P{idx}"
105
- np_codes = np.array(pd.Categorical(df_split[sub_col], categories=stats["codes"][sub_col]).codes)
105
+ categories = list(stats["codes"][sub_col].keys())
106
+ values_at_pos = df_split[sub_col].where(df_split[sub_col].isin(categories), UNKNOWN_TOKEN)
107
+ np_codes = np.array(pd.Categorical(values_at_pos, categories=categories).codes)
106
108
  np.place(np_codes, np_codes == -1, 0)
107
109
  df_split[sub_col] = np_codes
108
110
  if stats["has_nan"]:
@@ -234,9 +234,10 @@ def decode_datetime(df_encoded: pd.DataFrame, stats: dict):
234
234
  d = df_encoded["day"] + stats["min_values"]["day"]
235
235
  # fix invalid dates by setting these to last day of month
236
236
  is_leap = y.apply(lambda x: calendar.isleap(x))
237
- d[is_leap & (m == 2) & (d > 29)] = 29
238
- d[~is_leap & (m == 2) & (d > 28)] = 28
239
- d[((m == 4) | (m == 6) | (m == 9) | (m == 11)) & (d > 30)] = 30
237
+ d = d.copy()
238
+ d.loc[is_leap & (m == 2) & (d > 29)] = 29
239
+ d.loc[~is_leap & (m == 2) & (d > 28)] = 28
240
+ d.loc[((m == 4) | (m == 6) | (m == 9) | (m == 11)) & (d > 30)] = 30
240
241
  # concatenate to datetime string
241
242
  y = y.astype(str)
242
243
  m = m.astype(str).str.zfill(2)
@@ -271,7 +272,7 @@ def decode_datetime(df_encoded: pd.DataFrame, stats: dict):
271
272
  # set all values to NaN if no valid values were present
272
273
  values[df_encoded["nan"] == 0] = pd.NA
273
274
  # convert from string to datetime
274
- values = pd.to_datetime(values)
275
+ values = pd.to_datetime(values).astype("datetime64[ns]")
275
276
  if not stats["has_time"]:
276
- values = pd.to_datetime(values.dt.date)
277
+ values = pd.to_datetime(values.dt.date).astype("datetime64[ns]")
277
278
  return values
@@ -75,14 +75,23 @@ def create_schemas(
75
75
  numeric_fields = field_types.get(ModelEncodingType.language_numeric, [])
76
76
  datetime_fields = field_types.get(ModelEncodingType.language_datetime, [])
77
77
  cache = {}
78
+
79
+ def _normalize_seed_value(seed_value):
80
+ return None if pd.isna(seed_value) else seed_value
81
+
78
82
  for _, seed_row in seed_df.iterrows():
79
- cache_key = hash(tuple(sorted([(field_name, str(seed_value)) for field_name, seed_value in seed_row.items()])))
83
+ normalized_seed_items = [
84
+ (field_name, _normalize_seed_value(seed_value)) for field_name, seed_value in seed_row.items()
85
+ ]
86
+ cache_key = hash(
87
+ tuple(sorted([(field_name, str(seed_value)) for field_name, seed_value in normalized_seed_items]))
88
+ )
80
89
  if cache_key in cache:
81
90
  yield cache[cache_key]
82
91
  continue
83
92
  model_dict = {}
84
93
  if not seed_row.empty:
85
- model_dict |= {field_name: (Literal[seed_value], ...) for field_name, seed_value in seed_row.items()} # type: ignore[valid-type]
94
+ model_dict |= {field_name: (Literal[seed_value], ...) for field_name, seed_value in normalized_seed_items} # type: ignore[valid-type]
86
95
  for field_name in unseeded_fields:
87
96
  if field_name in categorical_fields:
88
97
  categories = stats["columns"][field_name]["categories"]
@@ -197,7 +197,7 @@ def prepare_context_inputs(
197
197
  # Build flat context inputs (CTXFLT/*)
198
198
  ctxflt_inputs = {
199
199
  col: torch.unsqueeze(
200
- torch.as_tensor(ctx_encoded[col].to_numpy(), device=device).type(torch.int),
200
+ torch.as_tensor(ctx_encoded[col].to_numpy(copy=True), device=device).type(torch.int),
201
201
  dim=-1,
202
202
  )
203
203
  for col in ctx_encoded.columns
@@ -640,10 +640,9 @@ def decode_buffered_samples(
640
640
  keys=keys,
641
641
  key_name=tgt_context_key,
642
642
  )
643
- df_syn = df_syn.drop(
644
- columns=[c for c in df_syn.columns if c.startswith(POSITIONAL_COLUMN)],
645
- axis=1,
646
- ).reset_index(drop=True)
643
+ df_syn = df_syn.drop(columns=[c for c in df_syn.columns if c.startswith(POSITIONAL_COLUMN)]).reset_index(
644
+ drop=True
645
+ )
647
646
  else:
648
647
  data, seed_data = zip(*buffer.buffer)
649
648
  df_syn = pd.concat(data, axis=0).reset_index(drop=True)
@@ -1224,7 +1223,9 @@ def generate(
1224
1223
  # Use context inputs prepared earlier
1225
1224
  x = ctx_inputs
1226
1225
  fixed_values = {
1227
- col: torch.as_tensor(seed_batch_encoded[col].to_numpy(), device=model.device).type(torch.int)
1226
+ col: torch.as_tensor(seed_batch_encoded[col].to_numpy(copy=True), device=model.device).type(
1227
+ torch.int
1228
+ )
1228
1229
  for col in seed_batch_encoded.columns
1229
1230
  if col in tgt_sub_columns
1230
1231
  }
@@ -1,9 +1,9 @@
1
1
  [project]
2
2
  name = "mostlyai-engine"
3
- version = "2.4.0"
3
+ version = "2.5.0"
4
4
  description = "Synthetic Data Engine"
5
5
  authors = [{ name = "MOSTLY AI", email = "dev@mostly.ai" }]
6
- requires-python = ">=3.10"
6
+ requires-python = ">=3.11,<3.14"
7
7
  readme = "README.md"
8
8
  license = "Apache-2.0"
9
9
  classifiers = [
@@ -14,7 +14,6 @@ classifiers = [
14
14
  "Intended Audience :: Financial and Insurance Industry",
15
15
  "Intended Audience :: Healthcare Industry",
16
16
  "Intended Audience :: Telecommunications Industry",
17
- "Programming Language :: Python :: 3.10",
18
17
  "Programming Language :: Python :: 3.11",
19
18
  "Programming Language :: Python :: 3.12",
20
19
  "Programming Language :: Python :: 3.13",
@@ -33,7 +32,7 @@ dependencies = [
33
32
  "scikit-learn>=1.4.0",
34
33
  "psutil>=5.9.5,<6", # upgrade when colab psutil is updated
35
34
  "tokenizers>=0.21.0",
36
- "transformers>=4.55.0",
35
+ "transformers>=4.55.0,<5", # keep <5 for vllm==0.12 compatibility
37
36
  "datasets>=3.0.0",
38
37
  "accelerate>=1.5.0",
39
38
  "peft>=0.12.0",
@@ -95,8 +94,9 @@ requires = ["hatchling", "hatch-vcs"]
95
94
  build-backend = "hatchling.build"
96
95
 
97
96
  [tool.ruff]
98
- target-version = "py310"
97
+ target-version = "py311"
99
98
  line-length = 120
99
+ extend-exclude = ["*.ipynb"]
100
100
  [tool.ruff.format]
101
101
  exclude = ["examples/*.ipynb"]
102
102
  [tool.ruff.lint]
File without changes