mostlyai-engine 2.0.0__tar.gz → 2.0.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/PKG-INFO +9 -5
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/README.md +8 -4
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/__init__.py +1 -1
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/numeric.py +22 -15
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/encoding.py +5 -1
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/pyproject.toml +1 -1
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/.gitignore +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/LICENSE +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_common.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_dtypes.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/numeric.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/text.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/common.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/encoding.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/base.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/vllm_engine.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/generation.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/interface.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/lstm.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/training.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/xgrammar_utils.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_memory.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/__init__.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/argn.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/common.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/fairness.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/generation.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/interface.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_tabular/training.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_training_utils.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_workspace.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/analysis.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/domain.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/encoding.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/generation.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/logging.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/random_state.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/splitting.py +0 -0
- {mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/training.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: mostlyai-engine
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.1
|
|
4
4
|
Summary: Synthetic Data Engine
|
|
5
5
|
Project-URL: homepage, https://github.com/mostly-ai/mostlyai-engine
|
|
6
6
|
Project-URL: repository, https://github.com/mostly-ai/mostlyai-engine
|
|
@@ -173,8 +173,12 @@ data_with_missings.loc[0:299, "age"] = pd.NA
|
|
|
173
173
|
data_with_missings.loc[0:199, "race"] = pd.NA
|
|
174
174
|
data_with_missings.loc[100:299, "income"] = pd.NA
|
|
175
175
|
|
|
176
|
-
# impute missing values
|
|
177
|
-
argn.impute(data_with_missings)
|
|
176
|
+
# impute missing values each with a random sample
|
|
177
|
+
data_imputed = argn.impute(data_with_missings)
|
|
178
|
+
|
|
179
|
+
# impute missing values each with their point estimates
|
|
180
|
+
data_imputed = argn.impute(data_with_missings, n_draws=100)
|
|
181
|
+
|
|
178
182
|
```
|
|
179
183
|
|
|
180
184
|
### Predictions / Classification
|
|
@@ -185,10 +189,10 @@ Predict any categorical target column:
|
|
|
185
189
|
from sklearn.metrics import accuracy_score, roc_auc_score
|
|
186
190
|
|
|
187
191
|
# predict class labels for a categorical
|
|
188
|
-
predictions = argn.predict(data_test, target="income", n_draws=
|
|
192
|
+
predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
|
|
189
193
|
|
|
190
194
|
# predict class probabilities for a categorical
|
|
191
|
-
probabilities = argn.predict_proba(data_test, target="income", n_draws=
|
|
195
|
+
probabilities = argn.predict_proba(data_test, target="income", n_draws=100)
|
|
192
196
|
|
|
193
197
|
# evaluate performance
|
|
194
198
|
accuracy = accuracy_score(data_test["income"], predictions)
|
|
@@ -122,8 +122,12 @@ data_with_missings.loc[0:299, "age"] = pd.NA
|
|
|
122
122
|
data_with_missings.loc[0:199, "race"] = pd.NA
|
|
123
123
|
data_with_missings.loc[100:299, "income"] = pd.NA
|
|
124
124
|
|
|
125
|
-
# impute missing values
|
|
126
|
-
argn.impute(data_with_missings)
|
|
125
|
+
# impute missing values each with a random sample
|
|
126
|
+
data_imputed = argn.impute(data_with_missings)
|
|
127
|
+
|
|
128
|
+
# impute missing values each with their point estimates
|
|
129
|
+
data_imputed = argn.impute(data_with_missings, n_draws=100)
|
|
130
|
+
|
|
127
131
|
```
|
|
128
132
|
|
|
129
133
|
### Predictions / Classification
|
|
@@ -134,10 +138,10 @@ Predict any categorical target column:
|
|
|
134
138
|
from sklearn.metrics import accuracy_score, roc_auc_score
|
|
135
139
|
|
|
136
140
|
# predict class labels for a categorical
|
|
137
|
-
predictions = argn.predict(data_test, target="income", n_draws=
|
|
141
|
+
predictions = argn.predict(data_test, target="income", n_draws=100, agg_fn="mode")
|
|
138
142
|
|
|
139
143
|
# predict class probabilities for a categorical
|
|
140
|
-
probabilities = argn.predict_proba(data_test, target="income", n_draws=
|
|
144
|
+
probabilities = argn.predict_proba(data_test, target="income", n_draws=100)
|
|
141
145
|
|
|
142
146
|
# evaluate performance
|
|
143
147
|
accuracy = accuracy_score(data_test["income"], predictions)
|
|
@@ -34,7 +34,7 @@ __all__ = [
|
|
|
34
34
|
"TabularARGN",
|
|
35
35
|
"LanguageModel",
|
|
36
36
|
]
|
|
37
|
-
__version__ = "2.0.
|
|
37
|
+
__version__ = "2.0.1"
|
|
38
38
|
|
|
39
39
|
# suppress specific warning related to os.fork() in multi-threaded processes
|
|
40
40
|
warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/numeric.py
RENAMED
|
@@ -105,6 +105,19 @@ def _type_safe_numeric_series(numeric_array: np.ndarray | list, pd_dtype: str) -
|
|
|
105
105
|
return pd.Series(np.array([v for v in numeric_array]).astype(np_dtype), dtype=pd_dtype)
|
|
106
106
|
|
|
107
107
|
|
|
108
|
+
def _cast_based_on_min_decimal(values: pd.Series, min_decimal: int) -> pd.Series:
|
|
109
|
+
# try to convert to int when min_decimal is 0, if possible
|
|
110
|
+
dtype = "Float64" if min_decimal < 0 else "Int64"
|
|
111
|
+
if dtype == "Int64":
|
|
112
|
+
values = values.round()
|
|
113
|
+
try:
|
|
114
|
+
values = values.astype(dtype)
|
|
115
|
+
except TypeError:
|
|
116
|
+
if dtype == "Int64":
|
|
117
|
+
values = values.astype("Float64") # if couldn't safely convert to int, stick to float
|
|
118
|
+
return values
|
|
119
|
+
|
|
120
|
+
|
|
108
121
|
def split_sub_columns_digit(
|
|
109
122
|
values: pd.Series,
|
|
110
123
|
max_decimal=NUMERIC_DIGIT_MAX_DECIMAL,
|
|
@@ -295,6 +308,9 @@ def analyze_reduce_numeric(
|
|
|
295
308
|
encoding_type = ModelEncodingType.tabular_numeric_binned
|
|
296
309
|
|
|
297
310
|
if encoding_type == ModelEncodingType.tabular_numeric_discrete:
|
|
311
|
+
if min_decimal >= 0:
|
|
312
|
+
# remove decimal part from categories
|
|
313
|
+
categories = [str(cat).split(".")[0] for cat in categories]
|
|
298
314
|
# add NULL token if NaN values exist
|
|
299
315
|
if has_nan:
|
|
300
316
|
categories = [NUMERIC_DISCRETE_NULL_TOKEN] + categories
|
|
@@ -368,6 +384,8 @@ def analyze_reduce_numeric(
|
|
|
368
384
|
|
|
369
385
|
|
|
370
386
|
def encode_numeric(values: pd.Series, stats: dict, _: pd.Series | None = None) -> pd.DataFrame:
|
|
387
|
+
values = safe_convert_numeric(values)
|
|
388
|
+
|
|
371
389
|
if stats["encoding_type"] == ModelEncodingType.tabular_numeric_discrete:
|
|
372
390
|
df = _encode_numeric_discrete(values, stats)
|
|
373
391
|
elif stats["encoding_type"] == ModelEncodingType.tabular_numeric_digit:
|
|
@@ -380,31 +398,21 @@ def encode_numeric(values: pd.Series, stats: dict, _: pd.Series | None = None) -
|
|
|
380
398
|
|
|
381
399
|
|
|
382
400
|
def _encode_numeric_discrete(values: pd.Series, stats: dict, _: pd.Series | None = None) -> pd.DataFrame:
|
|
383
|
-
values =
|
|
401
|
+
values = _cast_based_on_min_decimal(values, stats["min_decimal"])
|
|
384
402
|
df = encode_categorical(values, stats)
|
|
385
403
|
return df
|
|
386
404
|
|
|
387
405
|
|
|
388
406
|
def _encode_numeric_digit(values: pd.Series, stats: dict, _: pd.Series | None = None) -> pd.DataFrame:
|
|
389
|
-
values =
|
|
390
|
-
# try to convert to int, if possible
|
|
391
|
-
dtype = "Int64" if stats["min_decimal"] == 0 else "Float64"
|
|
392
|
-
if dtype == "Int64":
|
|
393
|
-
values = values.round()
|
|
394
|
-
try:
|
|
395
|
-
values = values.astype(dtype)
|
|
396
|
-
except TypeError:
|
|
397
|
-
if dtype == "Int64": # if couldn't safely convert to int, stick to float
|
|
398
|
-
dtype = "Float64"
|
|
399
|
-
values = values.astype(dtype)
|
|
407
|
+
values = _cast_based_on_min_decimal(values, stats["min_decimal"])
|
|
400
408
|
# reset index, as `values.mask` can throw errors for misaligned indices
|
|
401
409
|
values.reset_index(drop=True, inplace=True)
|
|
402
410
|
# replace extreme values with min/max
|
|
403
411
|
if stats["min"] is not None:
|
|
404
|
-
reduced_min = _type_safe_numeric_series([stats["min"]], dtype).iloc[0]
|
|
412
|
+
reduced_min = _type_safe_numeric_series([stats["min"]], values.dtype).iloc[0]
|
|
405
413
|
values = values.where((values.isna()) | (values >= reduced_min), reduced_min)
|
|
406
414
|
if stats["max"] is not None:
|
|
407
|
-
reduced_max = _type_safe_numeric_series([stats["max"]], dtype).iloc[0]
|
|
415
|
+
reduced_max = _type_safe_numeric_series([stats["max"]], values.dtype).iloc[0]
|
|
408
416
|
values = values.where((values.isna()) | (values <= reduced_max), reduced_max)
|
|
409
417
|
values, nan_mask = impute_from_non_nan_distribution(values, stats)
|
|
410
418
|
# split to sub_columns
|
|
@@ -431,7 +439,6 @@ def _encode_numeric_digit(values: pd.Series, stats: dict, _: pd.Series | None =
|
|
|
431
439
|
|
|
432
440
|
|
|
433
441
|
def _encode_numeric_binned(values: pd.Series, stats: dict, _: pd.Series | None = None) -> pd.DataFrame:
|
|
434
|
-
values = safe_convert_numeric(values)
|
|
435
442
|
bins = stats["bins"].copy()
|
|
436
443
|
min_value = bins[0]
|
|
437
444
|
max_value = bins[-1]
|
|
@@ -13,6 +13,7 @@
|
|
|
13
13
|
# limitations under the License.
|
|
14
14
|
|
|
15
15
|
import logging
|
|
16
|
+
import os
|
|
16
17
|
import time
|
|
17
18
|
from pathlib import Path
|
|
18
19
|
|
|
@@ -233,6 +234,7 @@ def encode_df(
|
|
|
233
234
|
values=df[column],
|
|
234
235
|
column_stats=column_stats,
|
|
235
236
|
context_keys=context_keys,
|
|
237
|
+
parent_pid=os.getpid(),
|
|
236
238
|
)
|
|
237
239
|
)
|
|
238
240
|
if delayed_encodes:
|
|
@@ -247,8 +249,10 @@ def _encode_col(
|
|
|
247
249
|
values: pd.Series,
|
|
248
250
|
column_stats: dict,
|
|
249
251
|
context_keys: pd.Series | None = None,
|
|
252
|
+
parent_pid: int | None = None,
|
|
250
253
|
) -> pd.DataFrame:
|
|
251
|
-
|
|
254
|
+
if os.getpid() != parent_pid:
|
|
255
|
+
set_random_state(worker=True)
|
|
252
256
|
is_sequential_column = is_sequential(values)
|
|
253
257
|
if is_sequential_column:
|
|
254
258
|
# explode nested columns and encode the same way as flat columns
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/numeric.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/language/text.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/character.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/itt.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_encoding_types/tabular/lat_long.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/hf_engine.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/engine/vllm_engine.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.0.0 → mostlyai_engine-2.0.1}/mostlyai/engine/_language/tokenizer_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|