gitlabds 2.1.5__tar.gz → 2.1.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (26) hide show
  1. {gitlabds-2.1.5 → gitlabds-2.1.6}/PKG-INFO +1 -1
  2. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/dummy.py +97 -20
  3. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/model_evaluator.py +1 -1
  4. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds.egg-info/PKG-INFO +1 -1
  5. {gitlabds-2.1.5 → gitlabds-2.1.6}/LICENSE +0 -0
  6. {gitlabds-2.1.5 → gitlabds-2.1.6}/README.md +0 -0
  7. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/__init__.py +0 -0
  8. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/baselines.py +0 -0
  9. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/config_generator.py +0 -0
  10. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/feature_reduction.py +0 -0
  11. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/insights.py +0 -0
  12. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/memory_optimization.py +0 -0
  13. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/missing.py +0 -0
  14. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/missing_check.py +0 -0
  15. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/missing_fill.py +0 -0
  16. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/monitoring_metrics.py +0 -0
  17. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/outliers.py +0 -0
  18. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/split_data.py +0 -0
  19. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds/trends.py +0 -0
  20. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds.egg-info/SOURCES.txt +0 -0
  21. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds.egg-info/dependency_links.txt +0 -0
  22. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds.egg-info/requires.txt +0 -0
  23. {gitlabds-2.1.5 → gitlabds-2.1.6}/gitlabds.egg-info/top_level.txt +0 -0
  24. {gitlabds-2.1.5 → gitlabds-2.1.6}/setup.cfg +0 -0
  25. {gitlabds-2.1.5 → gitlabds-2.1.6}/setup.py +0 -0
  26. {gitlabds-2.1.5 → gitlabds-2.1.6}/tests/__init__.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gitlabds
3
- Version: 2.1.5
3
+ Version: 2.1.6
4
4
  Summary: Gitlab Data Science and Modeling Tools
5
5
  Home-page: https://gitlab.com/gitlab-data/gitlabds
6
6
  Author: Kevin Dietz
@@ -50,6 +50,10 @@ def dummy_code(
50
50
  - The transformed DataFrame with dummy-coded columns
51
51
  - Dictionary of dummy coding information that can be used with apply_dummy()
52
52
  """
53
+ import pandas as pd
54
+ import warnings
55
+ from typing import Dict, List, Any, Union, Optional, Tuple
56
+
53
57
  # Parameter validation
54
58
  if not isinstance(df, pd.DataFrame):
55
59
  raise TypeError("df must be a pandas DataFrame")
@@ -103,7 +107,6 @@ def dummy_code(
103
107
  # Determine number of levels for each field
104
108
  cat_levels = df[var_list].select_dtypes(include="object").nunique(dropna=True, axis=0)
105
109
 
106
-
107
110
  # Filter columns by level count criteria
108
111
  eligible_cat_cols = cat_levels[(cat_levels <= categorical_max_levels) & (cat_levels > 1)]
109
112
 
@@ -156,7 +159,17 @@ def dummy_code(
156
159
  for col in num_columns:
157
160
  # Store unique values for later use
158
161
  values = sorted(df[col].dropna().unique().tolist())
159
- dummy_info[col] = values
162
+
163
+ # create column names with float format for numeric data
164
+ formatted_values = []
165
+ for v in values:
166
+ if isinstance(v, (int, float)):
167
+ # Store as float to match get_dummies column naming
168
+ formatted_values.append(float(v))
169
+ else:
170
+ formatted_values.append(v)
171
+
172
+ dummy_info[col] = formatted_values
160
173
 
161
174
  # Perform dummy coding if we have columns to encode
162
175
  if num_columns:
@@ -222,6 +235,10 @@ def dummy_top(
222
235
  - The transformed DataFrame with dummy-coded columns
223
236
  - Dictionary of dummy coding information that can be used with apply_dummy()
224
237
  """
238
+ import pandas as pd
239
+ import warnings
240
+ from typing import Dict, List, Any, Union, Optional, Tuple
241
+
225
242
  # Parameter validation
226
243
  if not isinstance(df, pd.DataFrame):
227
244
  raise TypeError("df must be a pandas DataFrame")
@@ -279,6 +296,10 @@ def dummy_top(
279
296
  else:
280
297
  print("No categorical columns found")
281
298
 
299
+ # Collect all dummy columns to create at once
300
+ all_dummy_cols = {}
301
+ columns_to_drop = []
302
+
282
303
  # Create dummy codes for categorical fields that exceed min_threshold
283
304
  for col in cat_columns:
284
305
  # Calculate value frequencies
@@ -300,18 +321,32 @@ def dummy_top(
300
321
  print(f"No levels exceed threshold for column '{col}'")
301
322
  continue
302
323
 
303
- # Store information for this column
304
- dummy_info[col] = high_freq_list
324
+ # For numeric columns, ensure float format to match what get_dummies creates
325
+ if pd.api.types.is_numeric_dtype(df[col]):
326
+ formatted_values = [float(v) if isinstance(v, (int, float)) else v for v in high_freq_list]
327
+ else:
328
+ formatted_values = high_freq_list
329
+
330
+ dummy_info[col] = formatted_values
305
331
 
306
332
  # Create dummies for values that exceed threshold
307
333
  for value in high_freq_list:
308
334
  dummy_name = f"{col}{prefix_sep}{value}"
309
335
  # Create dummy column (1 if equal to value, 0 otherwise)
310
- work_df[dummy_name] = (work_df[col] == value).astype(int)
336
+ all_dummy_cols[dummy_name] = (work_df[col] == value).astype(int)
311
337
 
312
- # Drop categorical field if requested
338
+ # Mark for dropping if requested
313
339
  if drop_categorical:
314
- work_df = work_df.drop(columns=[col])
340
+ columns_to_drop.append(col)
341
+
342
+ # Create all dummy columns at once using concat (prevents fragmentation)
343
+ if all_dummy_cols:
344
+ dummy_df = pd.DataFrame(all_dummy_cols, index=work_df.index)
345
+ work_df = pd.concat([work_df, dummy_df], axis=1)
346
+
347
+ # Drop categorical fields if requested
348
+ if columns_to_drop:
349
+ work_df = work_df.drop(columns=columns_to_drop)
315
350
 
316
351
  # Store the prefix_sep in the dummy_info to use with apply_dummy
317
352
  dummy_info["_config"] = {"prefix_sep": prefix_sep}
@@ -319,11 +354,12 @@ def dummy_top(
319
354
  return work_df, dummy_info
320
355
 
321
356
 
357
+
322
358
  def apply_dummy(
323
359
  df: pd.DataFrame,
324
360
  dummy_info: Dict[str, List[Any]],
325
361
  drop_original: bool = False,
326
-
362
+ dummy_na: bool = False,
327
363
  ) -> pd.DataFrame:
328
364
  """
329
365
  Apply dummy coding to a new dataframe using the information from a previous
@@ -338,12 +374,18 @@ def apply_dummy(
338
374
  dummy_code() or dummy_top().
339
375
  drop_original : bool, default=False
340
376
  Whether to drop the original columns after dummy coding.
377
+ dummy_na : bool, default=False
378
+ Whether to create dummy variables for NA values.
341
379
 
342
380
  Returns
343
381
  -------
344
382
  pd.DataFrame
345
383
  Transformed dataframe with dummy-coded columns.
346
384
  """
385
+ import pandas as pd
386
+ import warnings
387
+ from typing import Dict, List, Any
388
+
347
389
  # Validate input
348
390
  if not isinstance(df, pd.DataFrame):
349
391
  raise TypeError("df must be a pandas DataFrame")
@@ -353,43 +395,78 @@ def apply_dummy(
353
395
 
354
396
  if df.empty:
355
397
  raise ValueError("DataFrame is empty")
356
-
398
+
357
399
  # Get the prefix_sep from dummy_info or use default
358
400
  prefix_sep = "_dummy_"
401
+ processing_dummy_info = dummy_info.copy()
402
+
359
403
  if "_config" in dummy_info and "prefix_sep" in dummy_info["_config"]:
360
404
  prefix_sep = dummy_info["_config"]["prefix_sep"]
361
405
  # Remove _config from processing
362
- dummy_info = {k: v for k, v in dummy_info.items() if k != "_config"}
406
+ processing_dummy_info = {k: v for k, v in dummy_info.items() if k != "_config"}
363
407
 
364
408
  # Create a copy to avoid modifying the original
365
409
  work_df = df.copy(deep=True)
366
410
 
411
+ # Collect all new dummy columns in a dictionary
412
+ new_dummy_cols = {}
413
+ columns_to_drop = []
414
+
367
415
  # Process each column in the dummy_info dictionary
368
- for col, values in dummy_info.items():
416
+ for col, values in processing_dummy_info.items():
369
417
  if col not in work_df.columns:
370
418
  warnings.warn(f"Column '{col}' not found in dataframe and will be skipped")
371
419
  continue
372
-
420
+
373
421
  # Create dummies for each value
374
422
  for value in values:
375
423
  dummy_name = f"{col}{prefix_sep}{value}"
376
424
 
377
- # Handle numeric vs string values appropriately
378
- if pd.api.types.is_numeric_dtype(type(value)):
379
- work_df[dummy_name] = (work_df[col] == value).astype(int)
425
+ # Handle type mismatches between config and actual data
426
+ # Try both the original value and converted versions
427
+ if pd.api.types.is_numeric_dtype(work_df[col]):
428
+ # For numeric columns, try both int and float versions
429
+ try:
430
+ # First try exact match
431
+ mask = (work_df[col] == value)
432
+
433
+ # If no matches and value types differ, try conversion
434
+ if not mask.any():
435
+ if isinstance(value, float) and work_df[col].dtype in ['int64', 'int32', 'int8']:
436
+ # Config expects float (1.0) but data is int (1)
437
+ mask = (work_df[col] == int(value))
438
+ elif isinstance(value, int) and work_df[col].dtype in ['float64', 'float32']:
439
+ # Config expects int (1) but data is float (1.0)
440
+ mask = (work_df[col] == float(value))
441
+
442
+ new_dummy_cols[dummy_name] = mask.astype(int)
443
+
444
+ except (ValueError, TypeError):
445
+ # Fallback to string comparison if numeric conversion fails
446
+ new_dummy_cols[dummy_name] = (work_df[col].astype(str) == str(value)).astype(int)
380
447
  else:
381
- work_df[dummy_name] = (work_df[col] == value).astype(int)
448
+ # For non-numeric columns, use direct comparison
449
+ new_dummy_cols[dummy_name] = (work_df[col] == value).astype(int)
382
450
 
383
451
  # Create dummy for NA values if requested
384
452
  if dummy_na:
385
- dummy_name = f"{col}_dummy_nan"
386
- work_df[dummy_name] = work_df[col].isna().astype(int)
453
+ dummy_name = f"{col}{prefix_sep}nan"
454
+ new_dummy_cols[dummy_name] = work_df[col].isna().astype(int)
387
455
 
388
- # Drop the original column if requested
456
+ # Mark column for dropping if requested
389
457
  if drop_original:
390
458
  # For categorical columns (from dummy_code) or any column (from dummy_top)
391
459
  # if its object type or explicitly requested
392
460
  if pd.api.types.is_object_dtype(work_df[col]) or drop_original:
393
- work_df = work_df.drop(columns=[col])
461
+ columns_to_drop.append(col)
462
+
463
+ # Create all dummy columns at once using concat
464
+ if new_dummy_cols:
465
+ dummy_df = pd.DataFrame(new_dummy_cols, index=work_df.index)
466
+ work_df = pd.concat([work_df, dummy_df], axis=1)
467
+
468
+ # Drop original columns if requested
469
+ if columns_to_drop:
470
+ work_df = work_df.drop(columns=columns_to_drop)
394
471
 
395
472
  return work_df
@@ -965,7 +965,7 @@ class ModelEvaluator:
965
965
  # Calculate metrics efficiently
966
966
  class_lift[f"class_{class_label}_total"] = grouped.size()
967
967
  class_lift[f"class_{class_label}_count"] = grouped.apply(
968
- lambda x: sum(x["actual"] == class_label)
968
+ lambda x: sum(x["actual"] == class_label), include_groups=False
969
969
  )
970
970
  class_lift[f"class_{class_label}_accuracy"] = (
971
971
  class_lift[f"class_{class_label}_count"] /
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: gitlabds
3
- Version: 2.1.5
3
+ Version: 2.1.6
4
4
  Summary: Gitlab Data Science and Modeling Tools
5
5
  Home-page: https://gitlab.com/gitlab-data/gitlabds
6
6
  Author: Kevin Dietz
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes