mloda-community-string 0.3.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,136 @@
1
+ """Base class for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from mloda.core.abstract_plugins.components.feature import Feature
8
+ from mloda.core.abstract_plugins.components.feature_chainer.feature_chain_parser import FeatureChainParser
9
+ from mloda.core.abstract_plugins.components.feature_chainer.feature_chain_parser_mixin import FeatureChainParserMixin
10
+ from mloda.core.abstract_plugins.components.feature_set import FeatureSet
11
+ from mloda.provider import DefaultOptionKeys, FeatureGroup
12
+
13
+ STRING_OPS = {
14
+ "upper": "Convert string to uppercase",
15
+ "lower": "Convert string to lowercase",
16
+ "trim": "Strip leading and trailing whitespace",
17
+ "length": "Return the length of the string (integer)",
18
+ "reverse": "Reverse the string",
19
+ }
20
+
21
+
22
+ class StringFeatureGroup(FeatureChainParserMixin, FeatureGroup):
23
+ """Base class for element-wise string operations that preserve row count.
24
+
25
+ String operations transform a single string column element by element.
26
+ The output always has the same number of rows as the input.
27
+
28
+ Supported operations: upper, lower, trim, length, reverse.
29
+
30
+ Not every compute framework supports every operation. For example,
31
+ SQLite has no native ``REVERSE`` function, so ``SqliteStringOps``
32
+ restricts its supported set via ``_validate_string_match``.
33
+ SQLite's ``UPPER``/``LOWER`` only handle ASCII characters;
34
+ non-ASCII accented characters are not transformed.
35
+
36
+ Feature Creation Methods
37
+ ------------------------
38
+
39
+ 1. String-based (pattern):
40
+ Features follow the naming pattern ``<source_column>__<operation>``.
41
+
42
+ Examples::
43
+
44
+ Feature("name__upper")
45
+ Feature("title__trim")
46
+ Feature("description__length")
47
+
48
+ 2. Configuration-based:
49
+ Uses Options with context parameters::
50
+
51
+ Feature(
52
+ "uppercased_name",
53
+ options=Options(context={
54
+ "string_op": "upper",
55
+ "in_features": "name",
56
+ }),
57
+ )
58
+ """
59
+
60
+ PREFIX_PATTERN = r".+__(upper|lower|trim|length|reverse)$"
61
+
62
+ MIN_IN_FEATURES = 1
63
+ MAX_IN_FEATURES = 1
64
+
65
+ STRING_OP = "string_op"
66
+
67
+ PROPERTY_MAPPING = {
68
+ STRING_OP: {
69
+ **STRING_OPS,
70
+ DefaultOptionKeys.context: True,
71
+ DefaultOptionKeys.strict_validation: True,
72
+ },
73
+ DefaultOptionKeys.in_features: {
74
+ "explanation": "Source string column",
75
+ DefaultOptionKeys.context: True,
76
+ DefaultOptionKeys.strict_validation: False,
77
+ },
78
+ }
79
+
80
+ @classmethod
81
+ def _validate_string_match(cls, feature_name: str, operation_config: str, source_feature: str) -> bool:
82
+ """Validate that the parsed operation is a known string operation.
83
+
84
+ Subclasses (e.g. SqliteStringOps) can override to further restrict
85
+ the set of supported operations.
86
+ """
87
+ return operation_config in STRING_OPS
88
+
89
+ @classmethod
90
+ def get_string_op(cls, feature_name: str) -> str:
91
+ """Extract the string operation from a pattern-based feature name."""
92
+ prefix_patterns = cls._get_prefix_patterns()
93
+ operation_config, _ = FeatureChainParser.parse_feature_name(feature_name, prefix_patterns)
94
+ if operation_config is not None:
95
+ return operation_config
96
+ raise ValueError(f"Could not extract string operation from feature name: {feature_name}")
97
+
98
+ @classmethod
99
+ def _extract_string_op(cls, feature: Feature) -> str:
100
+ """Extract string operation from feature (string-based or config-based)."""
101
+ feature_name = feature.name
102
+ prefix_patterns = cls._get_prefix_patterns()
103
+ operation_config, _ = FeatureChainParser.parse_feature_name(feature_name, prefix_patterns)
104
+ if operation_config is not None:
105
+ return operation_config
106
+ op = feature.options.get(cls.STRING_OP)
107
+ if op is None:
108
+ raise ValueError(f"Could not extract string operation for {feature_name}")
109
+ return str(op)
110
+
111
+ @classmethod
112
+ def calculate_feature(cls, data: Any, features: FeatureSet) -> Any:
113
+ """Shared loop: extract params from each feature, delegate to _compute_string."""
114
+ table = data
115
+
116
+ for feature in features.features:
117
+ feature_name = feature.name
118
+
119
+ source_features = cls._extract_source_features(feature)
120
+ source_col = source_features[0]
121
+ op = cls._extract_string_op(feature)
122
+
123
+ table = cls._compute_string(table, feature_name, source_col, op)
124
+
125
+ return table
126
+
127
+ @classmethod
128
+ def _compute_string(
129
+ cls,
130
+ data: Any,
131
+ feature_name: str,
132
+ source_col: str,
133
+ op: str,
134
+ ) -> Any:
135
+ """Subclasses must implement the actual string computation."""
136
+ raise NotImplementedError
@@ -0,0 +1,49 @@
1
+ """DuckDB implementation for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from mloda.provider import ComputeFramework
8
+ from mloda_plugins.compute_framework.base_implementations.duckdb.duckdb_framework import DuckDBFramework
9
+ from mloda_plugins.compute_framework.base_implementations.duckdb.duckdb_relation import DuckdbRelation
10
+ from mloda_plugins.compute_framework.base_implementations.sql.sql_utils import quote_ident
11
+
12
+ from mloda.community.feature_groups.data_operations.string.base import (
13
+ StringFeatureGroup,
14
+ )
15
+
16
+ # DuckDB native string function expressions.
17
+ _DUCKDB_STRING_EXPRS: dict[str, str] = {
18
+ "upper": "UPPER({col})",
19
+ "lower": "LOWER({col})",
20
+ "trim": "TRIM({col})",
21
+ "length": "LENGTH({col})",
22
+ "reverse": "REVERSE({col})",
23
+ }
24
+
25
+
26
+ class DuckdbStringOps(StringFeatureGroup):
27
+ @classmethod
28
+ def compute_framework_rule(cls) -> set[type[ComputeFramework]] | None:
29
+ return {DuckDBFramework}
30
+
31
+ @classmethod
32
+ def _compute_string(
33
+ cls,
34
+ data: Any,
35
+ feature_name: str,
36
+ source_col: str,
37
+ op: str,
38
+ ) -> DuckdbRelation:
39
+ expr_template = _DUCKDB_STRING_EXPRS.get(op)
40
+ if expr_template is None:
41
+ raise ValueError(f"Unsupported string operation for DuckDB: {op}")
42
+
43
+ quoted_source = quote_ident(source_col)
44
+ expr = expr_template.format(col=quoted_source)
45
+ quoted_feature = quote_ident(feature_name)
46
+
47
+ raw_sql = f"*, {expr} AS {quoted_feature}"
48
+ result: DuckdbRelation = data.select(_raw_sql=raw_sql)
49
+ return result
@@ -0,0 +1,44 @@
1
+ """Pandas implementation for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import pandas as pd
6
+
7
+ from mloda.provider import ComputeFramework
8
+ from mloda_plugins.compute_framework.base_implementations.pandas.dataframe import PandasDataFrame
9
+
10
+ from mloda.community.feature_groups.data_operations.string.base import (
11
+ StringFeatureGroup,
12
+ )
13
+
14
+
15
+ class PandasStringOps(StringFeatureGroup):
16
+ @classmethod
17
+ def compute_framework_rule(cls) -> set[type[ComputeFramework]] | None:
18
+ return {PandasDataFrame}
19
+
20
+ @classmethod
21
+ def _compute_string(
22
+ cls,
23
+ data: pd.DataFrame,
24
+ feature_name: str,
25
+ source_col: str,
26
+ op: str,
27
+ ) -> pd.DataFrame:
28
+ data = data.copy()
29
+ col = data[source_col]
30
+
31
+ if op == "upper":
32
+ data[feature_name] = col.str.upper()
33
+ elif op == "lower":
34
+ data[feature_name] = col.str.lower()
35
+ elif op == "trim":
36
+ data[feature_name] = col.str.strip()
37
+ elif op == "length":
38
+ data[feature_name] = col.str.len()
39
+ elif op == "reverse":
40
+ data[feature_name] = col.str[::-1]
41
+ else:
42
+ raise ValueError(f"Unsupported string operation: {op}")
43
+
44
+ return data
@@ -0,0 +1,43 @@
1
+ """Polars Lazy implementation for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import polars as pl
6
+
7
+ from mloda.provider import ComputeFramework
8
+ from mloda_plugins.compute_framework.base_implementations.polars.lazy_dataframe import PolarsLazyDataFrame
9
+
10
+ from mloda.community.feature_groups.data_operations.string.base import (
11
+ StringFeatureGroup,
12
+ )
13
+
14
+
15
+ class PolarsLazyStringOps(StringFeatureGroup):
16
+ @classmethod
17
+ def compute_framework_rule(cls) -> set[type[ComputeFramework]] | None:
18
+ return {PolarsLazyDataFrame}
19
+
20
+ @classmethod
21
+ def _compute_string(
22
+ cls,
23
+ data: pl.LazyFrame,
24
+ feature_name: str,
25
+ source_col: str,
26
+ op: str,
27
+ ) -> pl.LazyFrame:
28
+ col = pl.col(source_col)
29
+
30
+ if op == "upper":
31
+ expr = col.str.to_uppercase()
32
+ elif op == "lower":
33
+ expr = col.str.to_lowercase()
34
+ elif op == "trim":
35
+ expr = col.str.strip_chars()
36
+ elif op == "length":
37
+ expr = col.str.len_chars()
38
+ elif op == "reverse":
39
+ expr = col.str.reverse()
40
+ else:
41
+ raise ValueError(f"Unsupported string operation: {op}")
42
+
43
+ return data.with_columns(expr.alias(feature_name))
@@ -0,0 +1,43 @@
1
+ """PyArrow implementation for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import pyarrow as pa
6
+ import pyarrow.compute as pc
7
+
8
+ from mloda.provider import ComputeFramework
9
+ from mloda_plugins.compute_framework.base_implementations.pyarrow.table import PyArrowTable
10
+
11
+ from mloda.community.feature_groups.data_operations.string.base import (
12
+ StringFeatureGroup,
13
+ )
14
+
15
+ _PYARROW_STRING_FUNCS: dict[str, str] = {
16
+ "upper": "utf8_upper",
17
+ "lower": "utf8_lower",
18
+ "trim": "utf8_trim_whitespace",
19
+ "length": "utf8_length",
20
+ "reverse": "utf8_reverse",
21
+ }
22
+
23
+
24
+ class PyArrowStringOps(StringFeatureGroup):
25
+ @classmethod
26
+ def compute_framework_rule(cls) -> set[type[ComputeFramework]] | None:
27
+ return {PyArrowTable}
28
+
29
+ @classmethod
30
+ def _compute_string(
31
+ cls,
32
+ table: pa.Table,
33
+ feature_name: str,
34
+ source_col: str,
35
+ op: str,
36
+ ) -> pa.Table:
37
+ func_name = _PYARROW_STRING_FUNCS.get(op)
38
+ if func_name is None:
39
+ raise ValueError(f"Unsupported string operation: {op}")
40
+
41
+ col = table.column(source_col)
42
+ new_col = getattr(pc, func_name)(col)
43
+ return table.append_column(feature_name, new_col)
@@ -0,0 +1,72 @@
1
+ """SQLite implementation for string operation feature groups."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from mloda.provider import ComputeFramework
6
+ from mloda_plugins.compute_framework.base_implementations.sql.sql_utils import quote_ident
7
+ from mloda_plugins.compute_framework.base_implementations.sqlite.sqlite_framework import SqliteFramework
8
+ from mloda_plugins.compute_framework.base_implementations.sqlite.sqlite_relation import SqliteRelation
9
+
10
+ from mloda.community.feature_groups.data_operations.reserved_columns import assert_no_reserved_columns
11
+ from mloda.community.feature_groups.data_operations.string.base import (
12
+ StringFeatureGroup,
13
+ )
14
+
15
+ # SQLite's native UPPER/LOWER are ASCII-only: UPPER('héllo') returns 'HéLLO'
16
+ # instead of 'HÉLLO'. Rather than emulate unicode-aware semantics in Python
17
+ # and risk silent divergence from the PyArrow reference, SQLite refuses to
18
+ # match upper/lower and lets the resolver fall back to another framework.
19
+ # The same pattern is used for 'reverse', which SQLite has no native function for.
20
+ _SQLITE_STRING_EXPRS: dict[str, str] = {
21
+ "trim": "TRIM({col})",
22
+ "length": "LENGTH({col})",
23
+ }
24
+
25
+
26
+ class SqliteStringOps(StringFeatureGroup):
27
+ @classmethod
28
+ def compute_framework_rule(cls) -> set[type[ComputeFramework]] | None:
29
+ return {SqliteFramework}
30
+
31
+ @classmethod
32
+ def _validate_string_match(cls, feature_name: str, operation_config: str, source_feature: str) -> bool:
33
+ """SQLite only supports trim and length. upper/lower are ASCII-only
34
+ in SQLite so they diverge from the PyArrow reference; reverse has
35
+ no native SQLite function. All three are refused at match time."""
36
+ return operation_config in _SQLITE_STRING_EXPRS
37
+
38
+ @classmethod
39
+ def _compute_string(
40
+ cls,
41
+ data: SqliteRelation,
42
+ feature_name: str,
43
+ source_col: str,
44
+ op: str,
45
+ ) -> SqliteRelation:
46
+ assert_no_reserved_columns(data.columns, framework="SQLite", operation="string")
47
+
48
+ expr_template = _SQLITE_STRING_EXPRS.get(op)
49
+ if expr_template is None:
50
+ raise ValueError(f"Unsupported string operation for SQLite: {op}")
51
+
52
+ quoted_source = quote_ident(source_col)
53
+ expr = expr_template.format(col=quoted_source)
54
+ quoted_feature = quote_ident(feature_name)
55
+ qrn = quote_ident("__mloda_rn__")
56
+
57
+ sql = " ".join(
58
+ [
59
+ "SELECT",
60
+ f"{expr} AS {quoted_feature},",
61
+ f"ROW_NUMBER() OVER (ORDER BY rowid) AS {qrn}",
62
+ "FROM",
63
+ f"{quote_ident(data.table_name)}",
64
+ "ORDER BY",
65
+ qrn,
66
+ ]
67
+ )
68
+ cursor = data.connection.execute(sql)
69
+ rows = cursor.fetchall()
70
+
71
+ result_values = [row[0] for row in rows]
72
+ return data.append_column(feature_name, result_values)
@@ -0,0 +1,13 @@
1
+ Metadata-Version: 2.4
2
+ Name: mloda-community-string
3
+ Version: 0.3.1
4
+ Summary: String operation feature group (element-wise, row-preserving)
5
+ Author-email: Tom Kaltofen <info@mloda.ai>
6
+ License-Expression: Apache-2.0
7
+ Project-URL: Homepage, https://mloda.ai
8
+ Project-URL: Repository, https://github.com/mloda-ai/mloda-registry
9
+ Requires-Python: >=3.10
10
+ Requires-Dist: mloda-community-data-operations>=0.2.12
11
+ Provides-Extra: dev
12
+ Requires-Dist: mloda-testing; extra == "dev"
13
+ Requires-Dist: pytest>=9.0.3; extra == "dev"
@@ -0,0 +1,11 @@
1
+ mloda/community/feature_groups/data_operations/string/__init__.py,sha256=AbpHGcgLb-kRsJGnwFEktk7uzpZOCcBY74-YBdrKVGs,1
2
+ mloda/community/feature_groups/data_operations/string/base.py,sha256=oUMEzaROR852lTzZoxqMiQQypMfd99qZMhKko4H_c7E,4876
3
+ mloda/community/feature_groups/data_operations/string/duckdb_string.py,sha256=tWTgD9xojOek60VXWMiFiJVFQB7oRly5oHRlrd2D1nU,1604
4
+ mloda/community/feature_groups/data_operations/string/pandas_string.py,sha256=QH1RU9ejmInh1q9FP0hfpuqY3hND2F0Ox6SVARcLVwc,1254
5
+ mloda/community/feature_groups/data_operations/string/polars_lazy_string.py,sha256=GOU47bvdWh2Z0yOD0-w0IOt66hTjSbKcga8PT9nTvSM,1250
6
+ mloda/community/feature_groups/data_operations/string/pyarrow_string.py,sha256=L5I9Aupf1upkRLwGwXzUU8vYgUqoUO0PMpMeAjV1Wwg,1214
7
+ mloda/community/feature_groups/data_operations/string/sqlite_string.py,sha256=LPGy_T6iWTC3imbsSCf4noRSSUEa2q7ZPhAaBC7j8go,2865
8
+ mloda_community_string-0.3.1.dist-info/METADATA,sha256=V2AYfE0OsZO0mZyeDtVTwwuvb-I3eqIhGVdpKFybvtc,508
9
+ mloda_community_string-0.3.1.dist-info/WHEEL,sha256=SmOxYU7pzNKBqASvQJ7DjX3XGUF92lrGhMb3R6_iiqI,91
10
+ mloda_community_string-0.3.1.dist-info/top_level.txt,sha256=srwNjqzXpP1WfLRhbrVc9JLt8TdfsaQe-Th7cGcAvYY,6
11
+ mloda_community_string-0.3.1.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (79.0.1)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+