mostlyai-engine 2.7.0__tar.gz → 2.7.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/PKG-INFO +1 -1
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/__init__.py +1 -1
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/numeric.py +6 -4
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/numeric.py +16 -5
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/lstm.py +4 -1
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/xgrammar_utils.py +11 -8
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/encoding.py +2 -4
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/training.py +28 -14
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/analysis.py +28 -23
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/pyproject.toml +1 -1
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/.gitignore +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/LICENSE +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/README.md +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_common.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_dtypes.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/categorical.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/datetime.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/text.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/categorical.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/character.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/datetime.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/itt.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/lat_long.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/common.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/encoding.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/base.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/hf_engine.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/vllm_engine.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/generation.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/interface.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/tokenizer_utils.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/training.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/xgrammar_hf_logits.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_memory.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/__init__.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/argn.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/common.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/fairness.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/generation.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/interface.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_tabular/probability.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_training_utils.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_workspace.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/domain.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/encoding.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/generation.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/logging.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/random_state.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/splitting.py +0 -0
- {mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/training.py +0 -0
|
@@ -34,7 +34,7 @@ __all__ = [
|
|
|
34
34
|
"TabularARGN",
|
|
35
35
|
"LanguageModel",
|
|
36
36
|
]
|
|
37
|
-
__version__ = "2.7.
|
|
37
|
+
__version__ = "2.7.2"
|
|
38
38
|
|
|
39
39
|
# suppress specific warning related to os.fork() in multi-threaded processes
|
|
40
40
|
warnings.filterwarnings("ignore", category=DeprecationWarning, message=".*multi-threaded.*fork.*")
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/numeric.py
RENAMED
|
@@ -43,8 +43,8 @@ def analyze_language_numeric(values: pd.Series, root_keys: pd.Series, _: pd.Seri
|
|
|
43
43
|
|
|
44
44
|
# determine max scale
|
|
45
45
|
def count_scale(num: float) -> int:
|
|
46
|
-
#
|
|
47
|
-
num =
|
|
46
|
+
# preserve fractional precision without scientific notation or trailing zeros
|
|
47
|
+
num = np.format_float_positional(num, unique=True, trim="-")
|
|
48
48
|
if "." in num:
|
|
49
49
|
# in case of decimal, return number of digits after decimal point
|
|
50
50
|
return len(num.split(".")[1])
|
|
@@ -133,11 +133,13 @@ def encode_language_numeric(values: pd.Series, stats: dict, _: pd.Series | None
|
|
|
133
133
|
def decode_language_numeric(x: pd.Series, stats: dict[str, str]) -> pd.Series:
|
|
134
134
|
x = pd.to_numeric(x, errors="coerce")
|
|
135
135
|
x = x.round(stats["max_scale"])
|
|
136
|
+
# Nullable pandas numeric dtypes expose their corresponding NumPy dtype.
|
|
137
|
+
np_dtype = np.dtype(getattr(x.dtype, "numpy_dtype", x.dtype))
|
|
136
138
|
if stats["min"] is not None:
|
|
137
|
-
reduced_min =
|
|
139
|
+
reduced_min = np_dtype.type(stats["min"])
|
|
138
140
|
x.loc[x < reduced_min] = reduced_min
|
|
139
141
|
if stats["max"] is not None:
|
|
140
|
-
reduced_max =
|
|
142
|
+
reduced_max = np_dtype.type(stats["max"])
|
|
141
143
|
x.loc[x > reduced_max] = reduced_max
|
|
142
144
|
dtype = "Int64" if stats["max_scale"] == 0 else float
|
|
143
145
|
return x.astype(dtype)
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/numeric.py
RENAMED
|
@@ -77,6 +77,7 @@ NUMERIC_DISCRETE_NULL_TOKEN = CATEGORICAL_NULL_TOKEN
|
|
|
77
77
|
# maximum and minimum precision that is being considered
|
|
78
78
|
NUMERIC_DIGIT_MAX_DECIMAL = 18
|
|
79
79
|
NUMERIC_DIGIT_MIN_DECIMAL = -8
|
|
80
|
+
NUMERIC_DIGIT_MAX_SCALE = 20
|
|
80
81
|
|
|
81
82
|
|
|
82
83
|
def _type_safe_numeric_series(numeric_array: np.ndarray | list, pd_dtype: str) -> pd.Series:
|
|
@@ -136,7 +137,11 @@ def split_sub_columns_digit(
|
|
|
136
137
|
# rely on `np.format_float_positional` to determine string representation of absolute values
|
|
137
138
|
values_str = (
|
|
138
139
|
values.abs()
|
|
139
|
-
.apply(
|
|
140
|
+
.apply(
|
|
141
|
+
lambda x: np.format_float_positional(
|
|
142
|
+
x, unique=True, pad_left=50, pad_right=NUMERIC_DIGIT_MAX_SCALE, precision=NUMERIC_DIGIT_MAX_SCALE
|
|
143
|
+
)
|
|
144
|
+
)
|
|
140
145
|
# convert to string[pyarrow] for faster processing
|
|
141
146
|
.astype("string[pyarrow]")
|
|
142
147
|
# replace nan with pd.NA for faster processing
|
|
@@ -199,7 +204,13 @@ def analyze_numeric(
|
|
|
199
204
|
max_n = max_values.sort_values(ascending=False).head(ANALYZE_MIN_MAX_TOP_N).astype("float").tolist()
|
|
200
205
|
|
|
201
206
|
# split values into digits; used for digit numeric encoding, plus to determine precision
|
|
202
|
-
|
|
207
|
+
# Extend the default digit window for fine fractions, up to the formatter's precision.
|
|
208
|
+
max_scale = max(
|
|
209
|
+
(len(np.format_float_positional(v, unique=True, trim="-").partition(".")[2]) for v in non_na_values),
|
|
210
|
+
default=0,
|
|
211
|
+
)
|
|
212
|
+
min_decimal = min(NUMERIC_DIGIT_MIN_DECIMAL, -min(max_scale, NUMERIC_DIGIT_MAX_SCALE))
|
|
213
|
+
df_split = split_sub_columns_digit(values, min_decimal=min_decimal)
|
|
203
214
|
is_not_nan = df_split["nan"] == 0
|
|
204
215
|
has_nan = sum(df_split["nan"]) > 0
|
|
205
216
|
has_neg = sum(df_split["neg"]) > 0
|
|
@@ -238,9 +249,9 @@ def analyze_reduce_numeric(
|
|
|
238
249
|
# check if there are negative values
|
|
239
250
|
has_neg = any([j["has_neg"] for j in stats_list])
|
|
240
251
|
# determine precision to apply rounding of sampled values during generation
|
|
241
|
-
keys = stats_list[
|
|
242
|
-
min_digits = {k: min([j["min_digits"]
|
|
243
|
-
max_digits = {k: max([j["max_digits"]
|
|
252
|
+
keys = sorted({k for stats in stats_list for k in stats["max_digits"]}, key=lambda k: int(k[1:]), reverse=True)
|
|
253
|
+
min_digits = {k: min([j["min_digits"].get(k, 0) for j in stats_list]) for k in keys}
|
|
254
|
+
max_digits = {k: max([j["max_digits"].get(k, 0) for j in stats_list]) for k in keys}
|
|
244
255
|
non_zero_prec = [k for k in keys if max_digits[k] > 0 and k.startswith("E")]
|
|
245
256
|
min_decimal = min([int(k[1:]) for k in non_zero_prec]) if len(non_zero_prec) > 0 else 0
|
|
246
257
|
|
|
@@ -140,13 +140,16 @@ class LSTMFromScratchLMHeadModel(PreTrainedModel, GenerationMixin):
|
|
|
140
140
|
)
|
|
141
141
|
|
|
142
142
|
def prepare_inputs_for_generation(
|
|
143
|
-
self, input_ids: torch.Tensor, attention_mask: torch.Tensor, **kwargs
|
|
143
|
+
self, input_ids: torch.Tensor, attention_mask: torch.Tensor | None = None, **kwargs
|
|
144
144
|
) -> dict[str, torch.Tensor]:
|
|
145
145
|
"""
|
|
146
146
|
This function is mandatory so that the model is able to use the Hugging Face `.generate()` method.
|
|
147
147
|
Since `.generate()` works with left-padded sequences but the model is trained with right-padded sequences,
|
|
148
148
|
we need to convert the padding side here to make it work properly.
|
|
149
149
|
"""
|
|
150
|
+
# Transformers may omit an all-ones mask for an unpadded batch.
|
|
151
|
+
if attention_mask is None:
|
|
152
|
+
attention_mask = torch.ones_like(input_ids)
|
|
150
153
|
lengths = attention_mask.sum(dim=1)
|
|
151
154
|
return {
|
|
152
155
|
"input_ids": self.left_to_right_padding(input_ids, lengths),
|
|
@@ -32,14 +32,17 @@ JSON_NULL = "null"
|
|
|
32
32
|
|
|
33
33
|
|
|
34
34
|
def prepend_grammar_root_with_space(grammar: str) -> str:
|
|
35
|
-
#
|
|
36
|
-
#
|
|
37
|
-
#
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
35
|
+
# JSON-schema grammars start at "{". Training strings look like ` {...} {...}`
|
|
36
|
+
# and generation prompts look like ` {...}`, so the first output token must be a space.
|
|
37
|
+
# xgrammar <=0.2.3 emits `root ::= "{"`. xgrammar 0.2.7 emits `root ::= (("{"`.
|
|
38
|
+
replacements = (
|
|
39
|
+
('root ::= "{"', 'root ::= " {"'),
|
|
40
|
+
('root ::= (("{"', 'root ::= ((" {"'),
|
|
41
|
+
)
|
|
42
|
+
for start_of_grammar, start_of_grammar_with_space in replacements:
|
|
43
|
+
if start_of_grammar in grammar:
|
|
44
|
+
return grammar.replace(start_of_grammar, start_of_grammar_with_space)
|
|
45
|
+
raise AssertionError("unrecognized xgrammar root rule")
|
|
43
46
|
|
|
44
47
|
|
|
45
48
|
def ensure_seed_can_be_tokenized(seed_data: pd.DataFrame, tokenizer: PreTrainedTokenizerBase) -> pd.DataFrame:
|
|
@@ -19,7 +19,7 @@ from pathlib import Path
|
|
|
19
19
|
|
|
20
20
|
import numpy as np
|
|
21
21
|
import pandas as pd
|
|
22
|
-
from joblib import Parallel, cpu_count, delayed
|
|
22
|
+
from joblib import Parallel, cpu_count, delayed
|
|
23
23
|
|
|
24
24
|
from mostlyai.engine._common import (
|
|
25
25
|
ARGN_COLUMN,
|
|
@@ -238,9 +238,7 @@ def encode_df(
|
|
|
238
238
|
)
|
|
239
239
|
)
|
|
240
240
|
if delayed_encodes:
|
|
241
|
-
|
|
242
|
-
df_columns.extend(Parallel()(delayed_encodes))
|
|
243
|
-
|
|
241
|
+
df_columns.extend(Parallel(n_jobs=n_jobs)(delayed_encodes))
|
|
244
242
|
df = pd.concat(df_columns, axis=1) if df_columns else pd.DataFrame()
|
|
245
243
|
return df, ctx_primary_key, tgt_context_key
|
|
246
244
|
|
|
@@ -154,6 +154,8 @@ class BatchCollator:
|
|
|
154
154
|
self.use_nested_ctxseq = use_nested_ctxseq
|
|
155
155
|
|
|
156
156
|
def __call__(self, batch: list[dict]) -> dict[str, torch.Tensor]:
|
|
157
|
+
if not batch:
|
|
158
|
+
return {}
|
|
157
159
|
batch = pd.DataFrame(batch)
|
|
158
160
|
if self.is_sequential and self.max_sequence_window:
|
|
159
161
|
batch = self._slice_sequences(batch, self.max_sequence_window)
|
|
@@ -678,6 +680,9 @@ def train(
|
|
|
678
680
|
max_grad_norm=dp_config.get("max_grad_norm"),
|
|
679
681
|
poisson_sampling=True,
|
|
680
682
|
)
|
|
683
|
+
# Our dictionary collator represents empty batches as {}, including the first
|
|
684
|
+
# batch, which Opacus cannot infer from the raw dictionary dataset.
|
|
685
|
+
trn_dataloader.collate_fn = batch_collator
|
|
681
686
|
# this further wraps the dataloader with batch_sampler=BatchSplittingSampler to achieve gradient accumulation
|
|
682
687
|
# it will split the sampled logical batches into smaller sub-batches with batch_size
|
|
683
688
|
trn_dataloader = wrap_data_loader(
|
|
@@ -712,22 +717,31 @@ def train(
|
|
|
712
717
|
except StopIteration:
|
|
713
718
|
trn_data_iter = iter(trn_dataloader)
|
|
714
719
|
step_data = next(trn_data_iter)
|
|
715
|
-
# forward pass + calculate sample losses
|
|
716
|
-
step_losses = _calculate_sample_losses(argn, step_data)
|
|
717
|
-
# FIXME in sequential case, this is an approximation, it should be divided by total sum of masks in the
|
|
718
|
-
# entire batch to get the average loss per sample. Less importantly the final sample may be smaller
|
|
719
|
-
# than the batch size in both flat and sequential case.
|
|
720
|
-
# calculate total step loss
|
|
721
|
-
step_loss = torch.mean(step_losses) / (1 if with_dp else gradient_accumulation_steps)
|
|
722
720
|
if with_dp:
|
|
723
|
-
# opacus handles the gradient accumulation internally
|
|
724
721
|
optimizer.zero_grad(set_to_none=True)
|
|
725
|
-
|
|
726
|
-
|
|
727
|
-
|
|
728
|
-
|
|
729
|
-
|
|
730
|
-
|
|
722
|
+
if with_dp and not step_data:
|
|
723
|
+
# Poisson sampling can produce empty logical batches. Bypass the
|
|
724
|
+
# RNN forward pass, but retain Opacus noise and accounting below.
|
|
725
|
+
for parameter in optimizer.params:
|
|
726
|
+
parameter.grad_sample = parameter.new_empty((0, *parameter.shape))
|
|
727
|
+
step_losses = torch.empty(0, device=device)
|
|
728
|
+
else:
|
|
729
|
+
# forward pass + calculate sample losses
|
|
730
|
+
step_losses = _calculate_sample_losses(argn, step_data)
|
|
731
|
+
# FIXME in sequential case, this is an approximation, it should be divided by total sum of masks in the
|
|
732
|
+
# entire batch to get the average loss per sample. Less importantly the final sample may be smaller
|
|
733
|
+
# than the batch size in both flat and sequential case.
|
|
734
|
+
step_loss = torch.mean(step_losses) / (1 if with_dp else gradient_accumulation_steps)
|
|
735
|
+
# backward pass
|
|
736
|
+
with warnings.catch_warnings():
|
|
737
|
+
warnings.filterwarnings(
|
|
738
|
+
"ignore", category=FutureWarning, message="Using a non-full backward hook*"
|
|
739
|
+
)
|
|
740
|
+
if with_dp:
|
|
741
|
+
warnings.filterwarnings(
|
|
742
|
+
"ignore", category=UserWarning, message="Full backward hook is firing*"
|
|
743
|
+
)
|
|
744
|
+
step_loss.backward()
|
|
731
745
|
accumulated_steps += 1
|
|
732
746
|
# explicitly count the number of processed samples as the actual batch size can vary when DP is on
|
|
733
747
|
samples += step_losses.shape[0]
|
|
@@ -17,6 +17,7 @@ Provides analysis functionality of the engine
|
|
|
17
17
|
"""
|
|
18
18
|
|
|
19
19
|
import logging
|
|
20
|
+
import os
|
|
20
21
|
import time
|
|
21
22
|
from collections.abc import Iterable
|
|
22
23
|
from pathlib import Path
|
|
@@ -24,7 +25,7 @@ from typing import Any, Literal
|
|
|
24
25
|
|
|
25
26
|
import numpy as np
|
|
26
27
|
import pandas as pd
|
|
27
|
-
from joblib import Parallel, cpu_count, delayed
|
|
28
|
+
from joblib import Parallel, cpu_count, delayed
|
|
28
29
|
|
|
29
30
|
from mostlyai.engine._common import (
|
|
30
31
|
ANALYZE_REDUCE_MIN_MAX_N,
|
|
@@ -251,18 +252,21 @@ def _analyze_partition(
|
|
|
251
252
|
else:
|
|
252
253
|
ctx_root_keys = ctx_primary_keys.rename("__rkey")
|
|
253
254
|
|
|
255
|
+
# Unique row IDs suffice for root-level counts; reuse them across columns.
|
|
256
|
+
tgt_root_keys = pd.Series(np.arange(len(tgt_df)), index=tgt_df.index, name="root_keys")
|
|
257
|
+
|
|
254
258
|
# analyze all target columns
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
262
|
-
for column, encoding_type in tgt_encoding_types.items()
|
|
259
|
+
results = Parallel(n_jobs=n_jobs)(
|
|
260
|
+
delayed(_analyze_col)(
|
|
261
|
+
values=tgt_df[column],
|
|
262
|
+
root_keys=tgt_root_keys,
|
|
263
|
+
parent_pid=os.getpid(),
|
|
264
|
+
encoding_type=encoding_type,
|
|
265
|
+
context_keys=tgt_context_keys,
|
|
263
266
|
)
|
|
264
|
-
|
|
265
|
-
|
|
267
|
+
for column, encoding_type in tgt_encoding_types.items()
|
|
268
|
+
)
|
|
269
|
+
tgt_column_stats = {column: stats for column, stats in zip(tgt_encoding_types.keys(), results)}
|
|
266
270
|
# collect target sequence length stats
|
|
267
271
|
tgt_seq_len = _analyze_seq_len(
|
|
268
272
|
tgt_context_keys=tgt_context_keys,
|
|
@@ -293,17 +297,16 @@ def _analyze_partition(
|
|
|
293
297
|
|
|
294
298
|
# analyze all context columns
|
|
295
299
|
assert isinstance(ctx_encoding_types, dict)
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
|
|
302
|
-
)
|
|
303
|
-
for column, encoding_type in ctx_encoding_types.items()
|
|
300
|
+
results = Parallel(n_jobs=n_jobs)(
|
|
301
|
+
delayed(_analyze_col)(
|
|
302
|
+
values=ctx_df[column],
|
|
303
|
+
parent_pid=os.getpid(),
|
|
304
|
+
encoding_type=encoding_type,
|
|
305
|
+
root_keys=ctx_root_keys,
|
|
304
306
|
)
|
|
305
|
-
|
|
306
|
-
|
|
307
|
+
for column, encoding_type in ctx_encoding_types.items()
|
|
308
|
+
)
|
|
309
|
+
ctx_column_stats = {column: stats for column, stats in zip(ctx_encoding_types.keys(), results)}
|
|
307
310
|
# persist context stats
|
|
308
311
|
assert isinstance(ctx_stats_path, Path) and ctx_stats_path.exists()
|
|
309
312
|
ctx_stats_file = ctx_stats_path / f"part.{partition_id}.json"
|
|
@@ -516,8 +519,10 @@ def _analyze_col(
|
|
|
516
519
|
encoding_type: ModelEncodingType,
|
|
517
520
|
root_keys: pd.Series | None = None,
|
|
518
521
|
context_keys: pd.Series | None = None,
|
|
522
|
+
parent_pid: int | None = None,
|
|
519
523
|
) -> dict:
|
|
520
|
-
|
|
524
|
+
if os.getpid() != parent_pid:
|
|
525
|
+
set_random_state(worker=True)
|
|
521
526
|
|
|
522
527
|
stats: dict = {"encoding_type": encoding_type}
|
|
523
528
|
|
|
@@ -526,7 +531,7 @@ def _analyze_col(
|
|
|
526
531
|
return stats
|
|
527
532
|
|
|
528
533
|
if root_keys is None:
|
|
529
|
-
root_keys = pd.Series(
|
|
534
|
+
root_keys = pd.Series(np.arange(len(values)), index=values.index, name="root_keys")
|
|
530
535
|
|
|
531
536
|
if is_sequential(values):
|
|
532
537
|
# analyze sequential column
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/language/text.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/character.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/datetime.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/itt.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_encoding_types/tabular/lat_long.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/__init__.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/hf_engine.py
RENAMED
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/engine/vllm_engine.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/tokenizer_utils.py
RENAMED
|
File without changes
|
|
File without changes
|
{mostlyai_engine-2.7.0 → mostlyai_engine-2.7.2}/mostlyai/engine/_language/xgrammar_hf_logits.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|