layeredcompmodel 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,4 @@
1
+ from .model import LayeredCompModel, calculate_wilson_mean
2
+
3
+ __all__ = ["LayeredCompModel", "calculate_wilson_mean"]
4
+ __version__ = "0.1.0"
@@ -0,0 +1,639 @@
1
+ import numpy as np
2
+ import pandas as pd
3
+ import json
4
+ from sklearn.base import BaseEstimator, RegressorMixin
5
+
6
+ from sklearn.utils.validation import check_is_fitted
7
+ from joblib import Parallel, delayed
8
+ from collections import deque
9
+
10
+ from typing import Any, Callable, cast, Dict, List, Optional, Sequence, Tuple, Union
11
+
12
+ from pandas import DataFrame, Series
13
+
14
+
15
+ def calculate_wilson_mean(y: Sequence[float]) -> float:
16
+ """
17
+ Calculates the Wilson mean: trim the top 2.5% and the bottom 2.5% (the middle 95%)
18
+ and return the mean of the remaining data.
19
+ """
20
+ y_array = np.asarray(y, dtype=float)
21
+ if y_array.size == 0:
22
+ return float(np.nan)
23
+ low, high = np.percentile(y_array, [2.5, 97.5])
24
+ trimmed_y = y_array[(y_array >= low) & (y_array <= high)]
25
+ if trimmed_y.size == 0:
26
+ return float(np.mean(y_array))
27
+ return float(np.mean(trimmed_y))
28
+
29
+
30
+ class CompNode:
31
+ def __init__(self, depth: int, wilson_mean: float, count: int, filter_col: Optional[str] = None, filter_val: Optional[Union[str, float]] = None, is_numeric: bool = False, variant: Optional[str] = None) -> None:
32
+ self.depth = depth
33
+ self.wilson_mean = wilson_mean
34
+ self.count = count
35
+ self.filter_col = filter_col
36
+ self.filter_val = filter_val
37
+ self.is_numeric = is_numeric
38
+ self.variant = variant
39
+ self.children: List[CompNode] = []
40
+
41
+
42
+ class LayeredCompModel(RegressorMixin, BaseEstimator):
43
+ def __init__(self, weight_falloff: float = 0.5, split_metric: str = 'mae', n_jobs: int = 1) -> None:
44
+ self.weight_falloff: float = weight_falloff
45
+ self._split_metric_name: str = split_metric
46
+ self.n_jobs: int = n_jobs
47
+
48
+ def get_params(self, deep: bool = True) -> Dict[str, Any]:
49
+ return {
50
+ "weight_falloff": self.weight_falloff,
51
+ "split_metric": self._split_metric_name,
52
+ "n_jobs": self.n_jobs,
53
+ }
54
+
55
+ def set_params(self, **params: Any) -> "LayeredCompModel":
56
+ for key, value in params.items():
57
+ if key == "split_metric":
58
+ self._split_metric_name = value
59
+ else:
60
+ setattr(self, key, value)
61
+ return self
62
+
63
+ def _get_mae(self, y_subset: np.ndarray) -> float:
64
+ if len(y_subset) < 2:
65
+ return np.inf
66
+
67
+ # y_subset is now a numpy array
68
+ mean = np.mean(y_subset)
69
+
70
+ mae = np.mean(np.abs(y_subset - mean))
71
+ return mae
72
+
73
+ def _get_mse(self, y_subset: np.ndarray) -> float:
74
+ if len(y_subset) < 2:
75
+ return np.inf
76
+
77
+ # y_subset is now a numpy array
78
+ mean = np.mean(y_subset)
79
+
80
+ mse = np.mean((y_subset - mean) ** 2)
81
+ return mse
82
+
83
+ def _find_best_split(self, X_full: DataFrame, y_full: Series, indices: np.ndarray, columns: List[str], pre_sorted_indices: Optional[Dict[str, np.ndarray]] = None) -> Optional[Tuple[str, Union[str, float], bool]]:
84
+ # We want to MINIMIZE the weighted MAE / base MAE ratio
85
+ # Initializing best_score with 1.0 (no improvement)
86
+ best_score: float = 1.0
87
+ best_split: Optional[Tuple[str, Union[str, float], bool]] = None
88
+
89
+ y = y_full.iloc[indices]
90
+ # X = X_full.iloc[indices] # Removed since we use indices for filtering
91
+
92
+ total_count = len(y)
93
+ y_values = y.values
94
+ base_metric = self._split_metric(y_values)
95
+ if base_metric == 0 or base_metric == np.inf:
96
+ # Cannot improve or not enough data
97
+ return None
98
+
99
+ for col in columns:
100
+ is_numeric = pd.api.types.is_numeric_dtype(X_full[col])
101
+
102
+ if is_numeric:
103
+ # Optimized Numeric split logic using pre-sorted maps
104
+ if pre_sorted_indices and col in pre_sorted_indices:
105
+ # Filter pre-sorted indices to keep only those present in the current node
106
+ # Optimized filtering using np.isin
107
+ col_pre_sorted = pre_sorted_indices[col]
108
+ mask_in_node = np.isin(col_pre_sorted, indices)
109
+ col_sorted_indices = col_pre_sorted[mask_in_node]
110
+
111
+ if len(col_sorted_indices) < 8:
112
+ continue
113
+
114
+ y_values_all = y_full.values
115
+ X_col_values_all = X_full[col].values
116
+
117
+ y_col_values = y_values_all[col_sorted_indices]
118
+ X_col_values = X_col_values_all[col_sorted_indices]
119
+
120
+ # Pre-calculate prefix sums/counts for faster metric updates if possible
121
+ # But self.split_metric (MAE/MSE) depends on the MEAN of the subset, which changes.
122
+ # So we still need to call the metric function.
123
+
124
+ best_col_score = 1.0
125
+ best_col_midpoint = None
126
+
127
+ # Binary search on indices of the sorted array
128
+ num_iterations = min(10, int(np.log2(len(col_sorted_indices))))
129
+ low_idx = 0
130
+ high_idx = len(col_sorted_indices) - 1
131
+
132
+ for _ in range(num_iterations):
133
+ mid_idx = (low_idx + high_idx) // 2
134
+
135
+ # Ensure we split at a point where the value changes to avoid redundant checks
136
+ # and ensure consistency with <= logic.
137
+ # If X_col_values[mid_idx] == X_col_values[mid_idx+1], we should move mid_idx
138
+ curr_mid_idx = mid_idx
139
+ while curr_mid_idx < high_idx and X_col_values[curr_mid_idx] == X_col_values[curr_mid_idx + 1]:
140
+ curr_mid_idx += 1
141
+
142
+ if curr_mid_idx == high_idx:
143
+ # Try searching backwards
144
+ curr_mid_idx = mid_idx
145
+ while curr_mid_idx > low_idx and X_col_values[curr_mid_idx] == X_col_values[
146
+ curr_mid_idx - 1]:
147
+ curr_mid_idx -= 1
148
+ if curr_mid_idx == low_idx:
149
+ # All values in this range are the same
150
+ high_idx = mid_idx - 1
151
+ continue
152
+ else:
153
+ curr_mid_idx -= 1 # Split point is BEFORE this value
154
+
155
+ midpoint = X_col_values[curr_mid_idx]
156
+
157
+ y_low = y_col_values[:curr_mid_idx + 1]
158
+ y_high = y_col_values[curr_mid_idx + 1:]
159
+ metric_low = self._split_metric(y_low)
160
+ metric_high = self._split_metric(y_high)
161
+
162
+ weighted_metric = (metric_low * len(y_low) + metric_high * len(y_high)) / len(y_col_values)
163
+ score = weighted_metric / base_metric
164
+
165
+ if score < best_col_score:
166
+ best_col_score = score
167
+ best_col_midpoint = midpoint
168
+ elif score == best_col_score:
169
+ # Tie break: split most evenly
170
+ if abs(len(y_low) - len(y_col_values) / 2) < abs(
171
+ (len(y_col_values) - len(y_low)) - len(y_col_values) / 2):
172
+ best_col_midpoint = midpoint
173
+
174
+ if metric_low < metric_high:
175
+ high_idx = mid_idx - 1
176
+ else:
177
+ low_idx = mid_idx + 1
178
+
179
+ if best_col_score < best_score:
180
+ best_score = best_col_score
181
+ best_split = (col, cast(Union[str, float], best_col_midpoint), True)
182
+ else:
183
+ # Fallback to old numeric split logic (if no pre_sorted_indices)
184
+ X_col_full = X_full[col].iloc[indices]
185
+ mask = X_col_full.notna()
186
+ y_col = y.iloc[mask.values].values
187
+ X_col_values_ser = X_col_full.iloc[mask.values]
188
+
189
+ if len(y_col) < 8:
190
+ continue
191
+
192
+ num_iterations = min(10, int(np.log2(len(y_col))))
193
+ best_col_score = 1.0
194
+ best_col_midpoint = None
195
+
196
+ feature_values = X_col_values_ser.sort_values()
197
+ low_idx = 0
198
+ high_idx = len(feature_values) - 1
199
+
200
+ for _ in range(num_iterations):
201
+ mid_idx = (low_idx + high_idx) // 2
202
+ midpoint = feature_values.iloc[mid_idx]
203
+
204
+ lower_mask = X_col_values_ser <= midpoint
205
+
206
+ y_low = y_col[lower_mask.values]
207
+ y_high = y_col[~lower_mask.values]
208
+
209
+ if len(y_low) == 0 or len(y_high) == 0:
210
+ if len(y_low) == 0:
211
+ low_idx = mid_idx + 1
212
+ else:
213
+ high_idx = mid_idx - 1
214
+ continue
215
+
216
+ metric_low = self._split_metric(y_low)
217
+ metric_high = self._split_metric(y_high)
218
+ weighted_metric = (metric_low * len(y_low) + metric_high * len(y_high)) / len(y_col)
219
+ score = weighted_metric / base_metric
220
+
221
+ if score < best_col_score:
222
+ best_col_score = score
223
+ best_col_midpoint = midpoint
224
+ elif score == best_col_score:
225
+ current_best_low_count = (
226
+ X_col_values_ser <= best_col_midpoint).sum() if best_col_midpoint is not None else 0
227
+ if abs(len(y_low) - len(y_col) / 2) < abs(current_best_low_count - len(y_col) / 2):
228
+ best_col_midpoint = midpoint
229
+
230
+ if metric_low < metric_high:
231
+ high_idx = mid_idx - 1
232
+ else:
233
+ low_idx = mid_idx + 1
234
+
235
+ if best_col_score < best_score:
236
+ best_score = best_col_score
237
+ best_split = (col, cast(Union[str, float], best_col_midpoint), True)
238
+ else:
239
+ # Categorical split logic (one-vs-rest)
240
+ # Treat NaNs as a distinct category
241
+ X_col_filled = X_full[col].iloc[indices].fillna("NaN")
242
+ try:
243
+ variants = X_col_filled.unique()
244
+ except TypeError:
245
+ raise TypeError("argument must be a string or a number") from None
246
+ X_col_filled_values = X_col_filled.values
247
+
248
+ for var in variants:
249
+ mask = X_col_filled_values == var
250
+ y_v = y_values[mask]
251
+ y_inv = y_values[~mask]
252
+
253
+ if len(y_v) == 0 or len(y_inv) == 0:
254
+ continue
255
+
256
+ metric_v = self._split_metric(y_v)
257
+ metric_inv = self._split_metric(y_inv)
258
+ weighted_metric = (metric_v * len(y_v) + metric_inv * len(y_inv)) / total_count
259
+ score = weighted_metric / base_metric
260
+ if score < best_score:
261
+ best_score = score
262
+ best_split = (col, cast(Union[str, float], var), False)
263
+ elif score == best_score:
264
+ # Tie break
265
+ current_best_v_count = (X_col_filled_values == best_split[1]).sum() if best_split and not \
266
+ best_split[2] else 0
267
+ if abs(len(y_v) - total_count / 2) < abs(current_best_v_count - total_count / 2):
268
+ best_split = (col, cast(Union[str, float], var), False)
269
+
270
+ return best_split
271
+
272
+ def fit(self, X: DataFrame, y: Series, verbose: bool = False) -> "LayeredCompModel":
273
+ # Convert to pandas for easier manipulation
274
+ if isinstance(X, np.ndarray):
275
+ X = pd.DataFrame(X)
276
+ if isinstance(y, np.ndarray):
277
+ y = pd.Series(y)
278
+
279
+ # Ensure it's a dataframe if it was already something else (like a list)
280
+ if not isinstance(X, pd.DataFrame):
281
+ X = pd.DataFrame(X)
282
+ if not isinstance(y, pd.Series):
283
+ y = pd.Series(y)
284
+
285
+ if np.iscomplexobj(X.values) or np.iscomplexobj(y.values):
286
+ raise ValueError("Complex data not supported")
287
+
288
+ if X.shape[1] == 0:
289
+ raise ValueError(f"0 feature(s) (shape={X.shape}) while a minimum of 1 is required.")
290
+ if len(X) == 0:
291
+ raise ValueError(f"Found array with 0 sample(s) (shape={X.shape}) while a minimum of 1 is required.")
292
+ if len(y) == 0:
293
+ raise ValueError(f"Found array with 0 sample(s) (shape={y.shape}) while a minimum of 1 is required.")
294
+
295
+ # Sklearn compliance: NaN/inf checks for y (X numeric NaNs excluded per spec)
296
+ if pd.isna(y).any():
297
+ raise ValueError("Input y contains NaN.")
298
+ if pd.api.types.is_numeric_dtype(y) and np.isinf(y.values).any():
299
+ raise ValueError("Input y contains infinity.")
300
+
301
+ self.columns_: List[str] = X.columns.tolist()
302
+ self.n_features_in_: int = X.shape[1]
303
+
304
+ if self._split_metric_name not in ('mae', 'mse'):
305
+ raise ValueError(f"Invalid split_metric: {self._split_metric_name}. Supported: 'mae', 'mse'.")
306
+ self._split_metric: Callable[[np.ndarray], float] = self._get_mae if self._split_metric_name == 'mae' else self._get_mse
307
+
308
+ # Pre-calculate sorted index maps for numeric columns
309
+ self.pre_sorted_indices_: Dict[str, np.ndarray] = {}
310
+ for col in self.columns_:
311
+ if pd.api.types.is_numeric_dtype(X[col]):
312
+ # Drop NaNs and sort
313
+ valid_mask = X[col].notna()
314
+ # Use integer positions (0..N-1) for indexing into .values later
315
+ valid_indices = np.where(valid_mask)[0]
316
+ sorted_local_idx = np.argsort(X[col].values[valid_mask])
317
+ self.pre_sorted_indices_[col] = valid_indices[sorted_local_idx]
318
+
319
+ # Use integer positions for internal processing
320
+ indices = np.arange(len(X))
321
+ self.tree_ = self._build_tree(X, y, indices, depth=0, verbose=verbose)
322
+ return self
323
+
324
+ def _build_tree(self, X_full: DataFrame, y_full: Series, indices: np.ndarray, depth: int, variant: Optional[str] = None, verbose: bool = False) -> CompNode:
325
+ y_initial = y_full.iloc[indices]
326
+ root_node_mean = calculate_wilson_mean(y_initial)
327
+ root_node = CompNode(depth=depth, wilson_mean=root_node_mean, count=len(y_initial), variant=variant)
328
+
329
+ # queue stores (node, indices)
330
+ queue = deque([(root_node, indices)])
331
+
332
+ while queue:
333
+ # If n_jobs > 1, we could process the current level of the queue in parallel.
334
+ # However, a simple BFS with joblib.Parallel over the current queue is more efficient.
335
+
336
+ nodes_to_process = []
337
+ while queue:
338
+ nodes_to_process.append(queue.popleft())
339
+
340
+ if not nodes_to_process:
341
+ break
342
+
343
+ # Prepare tasks for find_best_split
344
+ # We only process nodes that can be split
345
+ tasks = []
346
+ valid_nodes_indices = []
347
+ for i, (node, node_indices) in enumerate(nodes_to_process):
348
+ if len(node_indices) >= 2:
349
+ tasks.append(delayed(self._find_best_split)(X_full, y_full, node_indices, self.columns_,
350
+ self.pre_sorted_indices_))
351
+ valid_nodes_indices.append(i)
352
+
353
+ if tasks:
354
+ results = Parallel(n_jobs=self.n_jobs)(tasks)
355
+
356
+ for idx, split in zip(valid_nodes_indices, results):
357
+ if split:
358
+ node, node_indices = nodes_to_process[idx]
359
+ col, val, is_numeric = split
360
+ node.filter_col = col
361
+ node.filter_val = val
362
+ node.is_numeric = is_numeric
363
+
364
+ X_s = X_full.iloc[node_indices]
365
+
366
+ if is_numeric:
367
+ mask_notna = X_s[col].notna()
368
+ # indices are already integer positions
369
+ valid_sub_indices = node_indices[mask_notna.values]
370
+ X_clean_values = X_full[col].values[valid_sub_indices]
371
+
372
+ mask_low = X_clean_values <= val
373
+ indices_low = valid_sub_indices[mask_low]
374
+ indices_high = valid_sub_indices[~mask_low]
375
+
376
+ if len(indices_low) > 0:
377
+ y_low = y_full.iloc[indices_low]
378
+ child_low = CompNode(depth=node.depth + 1,
379
+ wilson_mean=calculate_wilson_mean(y_low),
380
+ count=len(y_low),
381
+ variant='<=')
382
+ node.children.append(child_low)
383
+ queue.append((child_low, indices_low))
384
+ if len(indices_high) > 0:
385
+ y_high = y_full.iloc[indices_high]
386
+ child_high = CompNode(depth=node.depth + 1,
387
+ wilson_mean=calculate_wilson_mean(y_high),
388
+ count=len(y_high),
389
+ variant='>')
390
+ node.children.append(child_high)
391
+ queue.append((child_high, indices_high))
392
+ else:
393
+ X_col = X_s[col].fillna("NaN")
394
+ mask = (X_col == val).values
395
+ indices_v = node_indices[mask]
396
+ indices_rest = node_indices[~mask]
397
+
398
+ if len(indices_v) > 0:
399
+ y_v = y_full.iloc[indices_v]
400
+ child_v = CompNode(depth=node.depth + 1,
401
+ wilson_mean=calculate_wilson_mean(y_v),
402
+ count=len(y_v),
403
+ variant='=')
404
+ node.children.append(child_v)
405
+ queue.append((child_v, indices_v))
406
+ if len(indices_rest) > 0:
407
+ y_rest = y_full.iloc[indices_rest]
408
+ child_rest = CompNode(depth=node.depth + 1,
409
+ wilson_mean=calculate_wilson_mean(y_rest),
410
+ count=len(y_rest),
411
+ variant='!=')
412
+ node.children.append(child_rest)
413
+ queue.append((child_rest, indices_rest))
414
+
415
+ if verbose:
416
+ for idx in valid_nodes_indices:
417
+ node, _ = nodes_to_process[idx]
418
+ print(f"Depth: {node.depth}, Split: {node.filter_col} {node.filter_val}")
419
+
420
+ return root_node
421
+
422
+ def predict(self, X: DataFrame) -> np.ndarray:
423
+ check_is_fitted(self)
424
+ if hasattr(self, 'n_features_in_') and X.shape[1] != self.n_features_in_:
425
+ raise ValueError(f"X has {X.shape[1]} features, but {type(self).__name__} is expecting {self.n_features_in_} features as input.")
426
+ if isinstance(X, np.ndarray):
427
+ X = pd.DataFrame(X, columns=self.columns_)
428
+
429
+ predictions = X.apply(self._predict_row, axis=1)
430
+ return predictions.values
431
+
432
+ def _predict_row(self, row: pd.Series) -> float:
433
+ path = []
434
+ curr = self.tree_
435
+
436
+ while curr:
437
+ path.append(curr)
438
+ if not curr.children or curr.filter_col is None:
439
+ break
440
+
441
+ # Find which child matches
442
+ col = curr.filter_col
443
+ val = curr.filter_val
444
+ row_val = row[col]
445
+
446
+ matched_child = None
447
+ if curr.is_numeric:
448
+ if pd.isna(row_val):
449
+ # Spec doesn't explicitly say what to do if NaN at prediction time for numeric.
450
+ # "the parcel will still slot into a node slightly higher up the tree"
451
+ break
452
+
453
+ row_num = pd.to_numeric(row_val, errors='coerce')
454
+ val_num = pd.to_numeric(val, errors='coerce')
455
+ if pd.isna(row_num) or pd.isna(val_num):
456
+ break
457
+ is_less_equal = row_num <= val_num
458
+
459
+ if is_less_equal:
460
+ matched_child = curr.children[0]
461
+ else:
462
+ if len(curr.children) > 1:
463
+ matched_child = curr.children[1]
464
+ else:
465
+ # Categorical
466
+ if pd.isna(row_val):
467
+ row_val = "NaN"
468
+ else:
469
+ row_val = str(row_val)
470
+
471
+ if row_val == str(val):
472
+ matched_child = curr.children[0]
473
+ else:
474
+ if len(curr.children) > 1:
475
+ matched_child = curr.children[1]
476
+
477
+ if matched_child:
478
+ curr = matched_child
479
+ else:
480
+ break
481
+
482
+ # Calculate weights
483
+ # path is from root to leaf
484
+ # x = 0 for leaf, x = 1 for root
485
+ # n nodes in path. index i from 0 (root) to n-1 (leaf).
486
+ # x_i = (n - 1 - i) / (n - 1) if n > 1 else 0
487
+ n = len(path)
488
+ if n == 1:
489
+ return path[0].wilson_mean
490
+
491
+ total_w: float = 0.0
492
+ weighted_sum: float = 0.0
493
+ for i, node in enumerate(path):
494
+ x: float = (n - 1 - i) / (n - 1)
495
+ w: float = (1 - x) ** self.weight_falloff
496
+ weighted_sum += node.wilson_mean * w
497
+ total_w += w
498
+
499
+ return weighted_sum / total_w
500
+
501
+ def to_dict(self) -> Dict[str, Any]:
502
+ """
503
+ Exports the trained tree structure as a dictionary.
504
+ """
505
+ check_is_fitted(self)
506
+ assert self.tree_ is not None
507
+
508
+ def _node_to_dict(node: Optional[CompNode]) -> Optional[Dict[str, Any]]:
509
+ if not node:
510
+ return None
511
+
512
+ d = {
513
+ "wilson_mean": float(node.wilson_mean),
514
+ "count": int(node.count),
515
+ "depth": int(node.depth),
516
+ "filter_col": node.filter_col,
517
+ "filter_val": node.filter_val,
518
+ "is_numeric": bool(node.is_numeric),
519
+ "children": [_node_to_dict(child) for child in node.children]
520
+ }
521
+ if node.variant is not None:
522
+ d["variant"] = node.variant
523
+
524
+ # Handle non-serializable filter_val (like numpy types)
525
+ if d["filter_val"] is not None:
526
+ if isinstance(d["filter_val"], (np.int64, np.float64, np.int32, np.float32)):
527
+ d["filter_val"] = d["filter_val"].item()
528
+ elif not isinstance(d["filter_val"], (str, int, float, bool)):
529
+ d["filter_val"] = str(d["filter_val"])
530
+
531
+ return d
532
+
533
+ return cast(Dict[str, Any], _node_to_dict(self.tree_))
534
+
535
+ def to_json(self, indent: int = 4) -> str:
536
+ """
537
+ Exports the trained tree structure as a JSON string.
538
+ """
539
+ return json.dumps(self.to_dict(), indent=indent)
540
+
541
+ def explain_value(self, row: Union[DataFrame, pd.Series, Dict[str, Any]]) -> Dict[str, Any]:
542
+ """
543
+ Audits and traces the path that a row takes through the tree.
544
+ """
545
+ check_is_fitted(self)
546
+ if isinstance(row, pd.Series):
547
+ pass
548
+ elif isinstance(row, dict):
549
+ row = pd.Series(row)
550
+ else:
551
+ # Assume it's a single-row DataFrame or array
552
+ if hasattr(row, 'iloc'):
553
+ row = row.iloc[0]
554
+ else:
555
+ row = pd.Series(row, index=self.columns_)
556
+
557
+ path_nodes = []
558
+ curr = self.tree_
559
+
560
+ while curr:
561
+ node_info = {
562
+ "depth": curr.depth,
563
+ "wilson_mean": float(curr.wilson_mean),
564
+ "count": int(curr.count),
565
+ "variant": curr.variant,
566
+ "filter_col": curr.filter_col,
567
+ "filter_val": curr.filter_val,
568
+ "is_numeric": curr.is_numeric,
569
+ "actual_value": row[curr.filter_col] if curr.filter_col is not None else None
570
+ }
571
+ path_nodes.append(node_info)
572
+
573
+ if not curr.children or curr.filter_col is None:
574
+ break
575
+
576
+ col = curr.filter_col
577
+ val = curr.filter_val
578
+ row_val = row[col]
579
+
580
+ matched_child = None
581
+ if curr.is_numeric:
582
+ if pd.isna(row_val):
583
+ break
584
+ row_num = pd.to_numeric(row_val, errors='coerce')
585
+ val_num = pd.to_numeric(val, errors='coerce')
586
+ if pd.isna(row_num) or pd.isna(val_num):
587
+ break
588
+ is_less_equal = row_num <= val_num
589
+
590
+ if is_less_equal:
591
+ matched_child = curr.children[0]
592
+ else:
593
+ if len(curr.children) > 1:
594
+ matched_child = curr.children[1]
595
+ else:
596
+ if pd.isna(row_val):
597
+ row_val = "NaN"
598
+ else:
599
+ row_val = str(row_val)
600
+
601
+ if row_val == str(val):
602
+ matched_child = curr.children[0]
603
+ else:
604
+ if len(curr.children) > 1:
605
+ matched_child = curr.children[1]
606
+
607
+ if matched_child:
608
+ curr = matched_child
609
+ else:
610
+ break
611
+
612
+ # Calculate weights and final prediction
613
+ n = len(path_nodes)
614
+ total_w: float = 0.0
615
+ if n == 1:
616
+ final_pred = cast(float, path_nodes[0]["wilson_mean"])
617
+ total_w = 1.0
618
+ for node in path_nodes:
619
+ node["weight"] = 1.0
620
+ else:
621
+ weighted_sum: float = 0.0
622
+ for i, node in enumerate(path_nodes):
623
+ x: float = (n - 1 - i) / (n - 1)
624
+ w: float = (1 - x) ** self.weight_falloff
625
+ node["weight"] = w
626
+ weighted_sum += cast(float, node["wilson_mean"]) * w
627
+ total_w += w
628
+ final_pred = weighted_sum / total_w
629
+
630
+ # Build calculation string
631
+ calc_parts = [f"{n['wilson_mean']:.0f}*{cast(float, n['weight']):.3f}" for n in path_nodes]
632
+ calculation_str = f"({' + '.join(calc_parts)}) / {total_w:.4f} = {final_pred:.2f}"
633
+
634
+ return {
635
+ "final_prediction": float(final_pred),
636
+ "weight_falloff": float(self.weight_falloff),
637
+ "path": path_nodes,
638
+ "calculation": calculation_str
639
+ }
File without changes
@@ -0,0 +1,176 @@
1
+ Metadata-Version: 2.4
2
+ Name: layeredcompmodel
3
+ Version: 0.1.0
4
+ Summary: Hierarchical tree-based model for robust parcel sale price predictions using weighted Wilson means.
5
+ Project-URL: Homepage, https://github.com/JohnKossa/layeredcompmodel
6
+ Project-URL: Repository, https://github.com/JohnKossa/layeredcompmodel.git
7
+ Project-URL: Bug Tracker, https://github.com/JohnKossa/layeredcompmodel/issues
8
+ Author: John Kossa
9
+ License: MIT License
10
+
11
+ Copyright (c) 2026 Your Name
12
+
13
+ Permission is hereby granted, free of charge, to any person obtaining a copy
14
+ of this software and associated documentation files (the "Software"), to deal
15
+ in the Software without restriction, including without limitation the rights
16
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
17
+ copies of the Software, and to permit persons to whom the Software is
18
+ furnished to do so, subject to the following conditions:
19
+
20
+ The above copyright notice and this permission notice shall be included in all
21
+ copies or substantial portions of the Software.
22
+
23
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
24
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
25
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
26
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
27
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
28
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
29
+ SOFTWARE.
30
+ License-File: LICENSE
31
+ Keywords: hierarchical-model,real-estate,regression,scikit-learn
32
+ Classifier: Development Status :: 4 - Beta
33
+ Classifier: Intended Audience :: Developers
34
+ Classifier: License :: OSI Approved :: MIT License
35
+ Classifier: Operating System :: OS Independent
36
+ Classifier: Programming Language :: Python :: 3
37
+ Classifier: Programming Language :: Python :: 3.10
38
+ Classifier: Programming Language :: Python :: 3.11
39
+ Classifier: Programming Language :: Python :: 3.12
40
+ Classifier: Programming Language :: Python :: 3.13
41
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
42
+ Requires-Python: >=3.10
43
+ Requires-Dist: numpy>=1.24.0
44
+ Requires-Dist: pandas>=2.0.0
45
+ Requires-Dist: scikit-learn>=1.3.0
46
+ Requires-Dist: scipy>=1.10.0
47
+ Provides-Extra: dev
48
+ Requires-Dist: black; extra == 'dev'
49
+ Requires-Dist: build; extra == 'dev'
50
+ Requires-Dist: mypy; extra == 'dev'
51
+ Requires-Dist: pytest-cov; extra == 'dev'
52
+ Requires-Dist: pytest>=7.0; extra == 'dev'
53
+ Requires-Dist: ruff; extra == 'dev'
54
+ Requires-Dist: scikit-learn[dev]; extra == 'dev'
55
+ Description-Content-Type: text/markdown
56
+
57
+ # LayeredCompModel
58
+
59
+ [![PyPI version](https://badge.fury.io/py/layeredcompmodel.svg)](https://pypi.org/project/layeredcompmodel/)
60
+ [![Documentation Status](https://readthedocs.org/projects/layeredcompmodel/badge/?version=latest)](https://layeredcompmodel.readthedocs.io/en/latest/?badge=latest)
61
+ [![Tests](https://github.com/JohnKossa/layeredcompmodel/actions/workflows/ci.yml/badge.svg)](https://github.com/JohnKossa/layeredcompmodel/actions)
62
+ [![License](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
63
+
64
+ Hierarchical tree-based regressor for robust predictions (e.g., parcel sale prices) using path-weighted Wilson means (95% trimmed means for outlier resistance).
65
+
66
+ * [MODEL_SPEC.md](MODEL_SPEC.md): High-level method.
67
+ * [SPEC.md](SPEC.md): Detailed implementation specs.
68
+
69
+ ## Features
70
+
71
+ - **Scikit-learn compatible**: Inherits `BaseEstimator`/`RegressorMixin`; works with `Pipeline`, `GridSearchCV`, `cross_val_score`, pickling.
72
+ - **Automatic feature handling**: Categorical (one-vs-rest splits), numeric (binary search breakpoints), NaNs/missing values.
73
+ - **Robust statistics**: Wilson means prevent outlier swings.
74
+ - **Configurable weighting**: `weight_falloff` balances local accuracy vs. market normativity.
75
+ - **Explainable**: `explain_value(row)` shows path, weights, means.
76
+ - **Serializable**: `to_json()`, `to_dict()`.
77
+ - **Parallel**: `n_jobs` support.
78
+
79
+ ### NaN Handling
80
+ - **Categorical**: Treated as distinct "NaN" category.
81
+ - **Numeric**: Excluded from splits (robust; per SPEC.md).
82
+ - **Target `y`**: Must be finite (raises `ValueError`).
83
+ - Strict checks: Use `Pipeline([('imputer', SimpleImputer()), ('model', LayeredCompModel())])`.
84
+
85
+ ## Installation
86
+
87
+ ```bash
88
+ pip install layeredcompmodel
89
+ ```
90
+
91
+ For development:
92
+
93
+ ```bash
94
+ git clone https://github.com/JohnKossa/layeredcompmodel.git
95
+ cd layeredcompmodel
96
+ pip install -e .[dev]
97
+ ```
98
+
99
+ ## Quickstart
100
+
101
+ ```python
102
+ import pandas as pd
103
+ import numpy as np
104
+ from layeredcompmodel import LayeredCompModel
105
+
106
+ # Synthetic real-estate-like data
107
+ rng = np.random.default_rng(42)
108
+ n_samples = 100
109
+ data = {
110
+ 'neighborhood': rng.choice(['North', 'South', 'East'], n_samples),
111
+ 'size_sqft': rng.normal(2000, 500, n_samples),
112
+ 'price': rng.normal(500000, 100000, n_samples) + 100 * rng.normal(0, 1, n_samples) * (rng.normal(0, 1, n_samples) * 2000)
113
+ }
114
+ df = pd.DataFrame(data)
115
+ X = df[['neighborhood', 'size_sqft']]
116
+ y = df['price']
117
+
118
+ # Train
119
+ model = LayeredCompModel(weight_falloff=0.8, n_jobs=1)
120
+ model.fit(X, y)
121
+
122
+ # Predict
123
+ predictions = model.predict(X)
124
+ print(f&quot;Predictions shape: {predictions.shape}&quot;)
125
+ print(f&quot;MAE: {np.mean(np.abs(predictions - y)):.0f}&quot;)
126
+
127
+ # Explain single prediction
128
+ explanation = model.explain_value(X.iloc[0:1].squeeze())
129
+ print(explanation)
130
+ ```
131
+
132
+ ## API Reference
133
+
134
+ ### LayeredCompModel(weight_falloff=0.5, split_metric='mae', n_jobs=1)
135
+
136
+ - `fit(X, y)`: Build tree from features `X` (DataFrame), target `y` (Series).
137
+ - `predict(X)`: Predict using path-weighted means.
138
+ - `explain_value(row)`: Dict with path nodes, depths, weights, wilson_means.
139
+ - `to_json(indent=4)`: JSON tree dump.
140
+ - `tree_`: Root `CompNode` (filter_col, filter_val, wilson_mean, children).
141
+
142
+ See [docs](https://layeredcompmodel.readthedocs.io) (TBD).
143
+
144
+ ## Examples
145
+
146
+ See [`examples/quickstart.py`](examples/quickstart.py) for a runnable example (code matches Quickstart above).
147
+
148
+ **Run it:**
149
+ ```bash
150
+ python examples/quickstart.py
151
+ ```
152
+
153
+ **Expected output:**
154
+ ```
155
+ Predictions shape: (100,)
156
+ MAE: 126914
157
+ {'final_prediction': 530354.0426294187, 'weight_falloff': 0.8, 'path': [{'depth': 0, 'wilson_mean': 476353.91361128056, 'count': 100, 'is_leaf': False, 'filter_col': 'size_sqft', 'filter_val': 2101.366485546922}, {'depth': 1, 'wilson_mean': 553953.0606894617, 'count': 42, 'is_leaf': False, 'filter_col': 'neighborhood', 'filter_val': 'North'}, {'depth': 2, 'wilson_mean': 525096.3185716979, 'count': 13, 'is_leaf': True}], 'calculation': '0.199*476354 + 0.512*553953 + 0.289*525096 = 530354'}
158
+ ```
159
+
160
+ ## Development & Testing
161
+
162
+ ```bash
163
+ pytest tests/ --cov=layeredcompmodel
164
+ black src/
165
+ mypy src/
166
+ ```
167
+
168
+ CI/CD, Sphinx docs: planned.
169
+
170
+ ## Citing
171
+
172
+ Kossa, J. (2026). LayeredCompModel. GitHub. https://github.com/JohnKossa/layeredcompmodel
173
+
174
+ ## License
175
+
176
+ [MIT](LICENSE)
@@ -0,0 +1,7 @@
1
+ layeredcompmodel/__init__.py,sha256=cUJjF69GcTVPWdUyxBfl5CkzibKoErCg3LnGnzqAt58,140
2
+ layeredcompmodel/model.py,sha256=QkpuguxZsPHZTYdcFVdcZH6ePsicKq0rB2aJTndpGCk,28108
3
+ layeredcompmodel/py.typed,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
4
+ layeredcompmodel-0.1.0.dist-info/METADATA,sha256=G-wRcYXyX_onSE0ewJzpGjLVdXUN5kmNegHwVxliGm4,6959
5
+ layeredcompmodel-0.1.0.dist-info/WHEEL,sha256=QccIxa26bgl1E6uMy58deGWi-0aeIkkangHcxk2kWfw,87
6
+ layeredcompmodel-0.1.0.dist-info/licenses/LICENSE,sha256=S_fbX1OLfL7GYfxafKhn4E6-9cDMpbxy3-OgQisZr5A,1087
7
+ layeredcompmodel-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.29.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Your Name
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.