c50py 0.2.2__tar.gz → 0.2.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
- Metadata-Version: 2.1
1
+ Metadata-Version: 2.4
2
2
  Name: c50py
3
- Version: 0.2.2
3
+ Version: 0.2.4
4
4
  Summary: C5.0-like Decision Trees in Python (scikit-learn style)
5
5
  Author-email: David Díaz <daviddiazsolis@gmail.com>
6
6
  License: MIT
@@ -18,10 +18,11 @@ Requires-Dist: numpy>=1.21
18
18
  Requires-Dist: scikit-learn>=1.2
19
19
  Provides-Extra: graphviz
20
20
  Requires-Dist: graphviz>=0.20; extra == "graphviz"
21
+ Dynamic: license-file
21
22
 
22
- # c5py — C5.0‑style Decision Trees for Python (clean 0.2.0)
23
+ # c50py — C5.0‑style Decision Trees for Python (clean 0.2.0)
23
24
 
24
- `c5py` provides transparent, easily inspectable decision trees modelled on
25
+ `c50py` provides transparent, easily inspectable decision trees modelled on
25
26
  Quinlan’s C5.0 algorithm. Both classification and regression trees are
26
27
  supported and expose a scikit‑learn‑like API. The implementation is written
27
28
  from scratch in pure Python/Numpy and includes support for numeric and
@@ -86,7 +87,7 @@ Fit a regression tree to the diabetes dataset and obtain a visualisation:
86
87
  ```python
87
88
  import pandas as pd
88
89
  from time import perf_counter
89
- from c5py import C5Regressor
90
+ from c50py import C5Regressor
90
91
 
91
92
  df = pd.read_csv("diabetes.csv")
92
93
  y = df["target"].values
@@ -1,114 +1,114 @@
1
- # c5py — C5.0‑style Decision Trees for Python (clean 0.2.0)
2
-
3
- `c5py` provides transparent, easily inspectable decision trees modelled on
4
- Quinlan’s C5.0 algorithm. Both classification and regression trees are
5
- supported and expose a scikit‑learn‑like API. The implementation is written
6
- from scratch in pure Python/Numpy and includes support for numeric and
7
- categorical variables, missing values, pre‑ and post‑pruning, boosting,
8
- rule tracing/export and Graphviz visualisation.
9
-
10
- ## Features
11
-
12
- - **Scikit-learn API:** `fit(X, y)`, `predict(X)`, `score(X, y)`.
13
- - **Categorical support:** Pass `categorical_features=[0, 2]` to handle categories natively.
14
- - **Sample weights:** Supports `sample_weight` in `fit` for weighted splitting and pruning.
15
- - **Missing values:** Handles missing values using C5.0's fractional propagation strategy.
16
- - **Boosting:** Set `trials=10` to train a boosted ensemble.
17
- - **Rule export:** call `export_rules()` to get a list of human-readable rules.
18
- - **Graphviz export:** call `export_graphviz()` to visualize the tree.
19
- - **Pretty printing:** call `print_tree` to display the learned splits in a
20
- readable nested `if`/`else` format (single trees only).
21
-
22
- ## Documentation
23
-
24
- For a comprehensive guide on how to use `c50py`, including advanced features and examples, please see the [Usage Guide](USAGE_GUIDE.md).
25
-
26
- ## Installation (development mode)
27
-
28
- Install the package into your environment in editable mode:
29
-
30
- ```bash
31
- pip install -e .
32
- ```
33
-
34
- ## Quickstart (Classification)
35
-
36
- ```python
37
- import pandas as pd
38
- from time import perf_counter
39
- from c50py import C5Classifier
40
-
41
- df = pd.read_csv("titanic.csv")
42
- t0 = perf_counter(); clf.fit(X, y); print(f"fit: {perf_counter()-t0:.3f}s")
43
-
44
- # Inspect the tree
45
- clf.print_tree(feature_names=features, class_names=["No", "Yes"])
46
-
47
- # Extract rules for each sample
48
- rules = clf.predict_rule(X, feature_names=features)
49
- print(rules[:5])
50
-
51
- # Export as Graphviz
52
- path = clf.export_graphviz(
53
- "titanic_tree",
54
- feature_names=features,
55
- class_names=["No", "Yes"],
56
- format="dot" # save a .dot file directly
57
- )
58
- print(f"DOT file written to {path}")
59
- ```
60
-
61
- ## Quickstart – Regression (Diabetes)
62
-
63
- Fit a regression tree to the diabetes dataset and obtain a visualisation:
64
-
65
- ```python
66
- import pandas as pd
67
- from time import perf_counter
68
- from c5py import C5Regressor
69
-
70
- df = pd.read_csv("diabetes.csv")
71
- y = df["target"].values
72
- X_df = df.drop(columns=["target"])
73
- X = X_df.values.astype(object)
74
- features = list(X_df.columns)
75
-
76
- reg = C5Regressor(
77
- min_samples_split=30,
78
- min_samples_leaf=10,
79
- pruning=True, cf=0.25, global_pruning=True,
80
- feature_names=features,
81
- random_state=42,
82
- infer_categorical=False, int_as_categorical=False,
83
- numeric_threshold_strategy="quantile", max_numeric_thresholds=64
84
- )
85
-
86
- start = perf_counter(); reg.fit(X, y); print(f"fit: {perf_counter()-start:.3f}s")
87
-
88
- # Export to DOT (Graphviz installed optional)
89
- dot_path = reg.export_graphviz("diabetes_tree", feature_names=features, format="dot")
90
- print(f"Tree saved to {dot_path}")
91
-
92
- # Export human‑readable rules (single trees only)
93
- rules = reg.export_rules(feature_names=features)
94
- print(rules[:3])
95
- ```
96
-
97
- ## Performance tuning
98
-
99
- Several hyperparameters influence model complexity and performance:
100
-
101
- - **`numeric_threshold_strategy`** (`'quantile'` | `'all'`): subsample candidate numeric
102
- thresholds. With `'quantile'` the number of splits considered is limited to
103
- `max_numeric_thresholds` per feature per node. `'all'` evaluates every unique
104
- midpoint (slower on large datasets).
105
- - **`max_numeric_thresholds`**: number of candidate thresholds when using
106
- `'quantile'` (typically 32–64).
107
- - **`categorical_features`**: list of names or indices marking categorical columns.
108
- - **`max_categories_exhaustive`**: maximum cardinality for exhaustive subset search on
109
- categorical features; beyond this a simpler one‑vs‑rest strategy is used.
110
- - **`infer_categorical`/`int_as_categorical`**: enable automatic detection of
111
- categorical/boolean/integer columns when dtype information is not explicit.
112
- - **`max_depth`**: optional depth limit for extremely noisy or deep trees.
113
-
114
- When boosting (`trials > 1`) the same hyperparameters apply to each base tree.
1
+ # c50py — C5.0‑style Decision Trees for Python (clean 0.2.0)
2
+
3
+ `c50py` provides transparent, easily inspectable decision trees modelled on
4
+ Quinlan’s C5.0 algorithm. Both classification and regression trees are
5
+ supported and expose a scikit‑learn‑like API. The implementation is written
6
+ from scratch in pure Python/Numpy and includes support for numeric and
7
+ categorical variables, missing values, pre‑ and post‑pruning, boosting,
8
+ rule tracing/export and Graphviz visualisation.
9
+
10
+ ## Features
11
+
12
+ - **Scikit-learn API:** `fit(X, y)`, `predict(X)`, `score(X, y)`.
13
+ - **Categorical support:** Pass `categorical_features=[0, 2]` to handle categories natively.
14
+ - **Sample weights:** Supports `sample_weight` in `fit` for weighted splitting and pruning.
15
+ - **Missing values:** Handles missing values using C5.0's fractional propagation strategy.
16
+ - **Boosting:** Set `trials=10` to train a boosted ensemble.
17
+ - **Rule export:** call `export_rules()` to get a list of human-readable rules.
18
+ - **Graphviz export:** call `export_graphviz()` to visualize the tree.
19
+ - **Pretty printing:** call `print_tree` to display the learned splits in a
20
+ readable nested `if`/`else` format (single trees only).
21
+
22
+ ## Documentation
23
+
24
+ For a comprehensive guide on how to use `c50py`, including advanced features and examples, please see the [Usage Guide](USAGE_GUIDE.md).
25
+
26
+ ## Installation (development mode)
27
+
28
+ Install the package into your environment in editable mode:
29
+
30
+ ```bash
31
+ pip install -e .
32
+ ```
33
+
34
+ ## Quickstart (Classification)
35
+
36
+ ```python
37
+ import pandas as pd
38
+ from time import perf_counter
39
+ from c50py import C5Classifier
40
+
41
+ df = pd.read_csv("titanic.csv")
42
+ t0 = perf_counter(); clf.fit(X, y); print(f"fit: {perf_counter()-t0:.3f}s")
43
+
44
+ # Inspect the tree
45
+ clf.print_tree(feature_names=features, class_names=["No", "Yes"])
46
+
47
+ # Extract rules for each sample
48
+ rules = clf.predict_rule(X, feature_names=features)
49
+ print(rules[:5])
50
+
51
+ # Export as Graphviz
52
+ path = clf.export_graphviz(
53
+ "titanic_tree",
54
+ feature_names=features,
55
+ class_names=["No", "Yes"],
56
+ format="dot" # save a .dot file directly
57
+ )
58
+ print(f"DOT file written to {path}")
59
+ ```
60
+
61
+ ## Quickstart – Regression (Diabetes)
62
+
63
+ Fit a regression tree to the diabetes dataset and obtain a visualisation:
64
+
65
+ ```python
66
+ import pandas as pd
67
+ from time import perf_counter
68
+ from c50py import C5Regressor
69
+
70
+ df = pd.read_csv("diabetes.csv")
71
+ y = df["target"].values
72
+ X_df = df.drop(columns=["target"])
73
+ X = X_df.values.astype(object)
74
+ features = list(X_df.columns)
75
+
76
+ reg = C5Regressor(
77
+ min_samples_split=30,
78
+ min_samples_leaf=10,
79
+ pruning=True, cf=0.25, global_pruning=True,
80
+ feature_names=features,
81
+ random_state=42,
82
+ infer_categorical=False, int_as_categorical=False,
83
+ numeric_threshold_strategy="quantile", max_numeric_thresholds=64
84
+ )
85
+
86
+ start = perf_counter(); reg.fit(X, y); print(f"fit: {perf_counter()-start:.3f}s")
87
+
88
+ # Export to DOT (Graphviz installed optional)
89
+ dot_path = reg.export_graphviz("diabetes_tree", feature_names=features, format="dot")
90
+ print(f"Tree saved to {dot_path}")
91
+
92
+ # Export human‑readable rules (single trees only)
93
+ rules = reg.export_rules(feature_names=features)
94
+ print(rules[:3])
95
+ ```
96
+
97
+ ## Performance tuning
98
+
99
+ Several hyperparameters influence model complexity and performance:
100
+
101
+ - **`numeric_threshold_strategy`** (`'quantile'` | `'all'`): subsample candidate numeric
102
+ thresholds. With `'quantile'` the number of splits considered is limited to
103
+ `max_numeric_thresholds` per feature per node. `'all'` evaluates every unique
104
+ midpoint (slower on large datasets).
105
+ - **`max_numeric_thresholds`**: number of candidate thresholds when using
106
+ `'quantile'` (typically 32–64).
107
+ - **`categorical_features`**: list of names or indices marking categorical columns.
108
+ - **`max_categories_exhaustive`**: maximum cardinality for exhaustive subset search on
109
+ categorical features; beyond this a simpler one‑vs‑rest strategy is used.
110
+ - **`infer_categorical`/`int_as_categorical`**: enable automatic detection of
111
+ categorical/boolean/integer columns when dtype information is not explicit.
112
+ - **`max_depth`**: optional depth limit for extremely noisy or deep trees.
113
+
114
+ When boosting (`trials > 1`) the same hyperparameters apply to each base tree.
@@ -1,37 +1,37 @@
1
- [build-system]
2
- requires = ["setuptools>=64", "wheel"]
3
- build-backend = "setuptools.build_meta"
4
-
5
- [project]
6
- name = "c50py"
7
- version = "0.2.2"
8
- description = "C5.0-like Decision Trees in Python (scikit-learn style)"
9
- readme = "README.md"
10
- license = { text = "MIT" }
11
- authors = [{ name = "David Díaz", email = "daviddiazsolis@gmail.com" }]
12
- requires-python = ">=3.9"
13
- dependencies = [
14
- "numpy>=1.21",
15
- "scikit-learn>=1.2"
16
- ]
17
-
18
- classifiers = [
19
- "Programming Language :: Python :: 3",
20
- "License :: OSI Approved :: MIT License",
21
- "Operating System :: OS Independent",
22
- "Intended Audience :: Science/Research",
23
- "Topic :: Scientific/Engineering :: Artificial Intelligence"
24
- ]
25
-
26
- [project.optional-dependencies]
27
- graphviz = ["graphviz>=0.20"]
28
-
29
- [project.urls]
30
- Homepage = "https://github.com/daviddiazsolis/c50py"
31
- Issues = "https://github.com/daviddiazsolis/c50py/issues"
32
-
33
- [tool.setuptools.packages.find]
34
- where = ["src"]
35
-
36
- [tool.setuptools]
37
- include-package-data = true
1
+ [build-system]
2
+ requires = ["setuptools>=64", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "c50py"
7
+ version = "0.2.4"
8
+ description = "C5.0-like Decision Trees in Python (scikit-learn style)"
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ authors = [{ name = "David Díaz", email = "daviddiazsolis@gmail.com" }]
12
+ requires-python = ">=3.9"
13
+ dependencies = [
14
+ "numpy>=1.21",
15
+ "scikit-learn>=1.2"
16
+ ]
17
+
18
+ classifiers = [
19
+ "Programming Language :: Python :: 3",
20
+ "License :: OSI Approved :: MIT License",
21
+ "Operating System :: OS Independent",
22
+ "Intended Audience :: Science/Research",
23
+ "Topic :: Scientific/Engineering :: Artificial Intelligence"
24
+ ]
25
+
26
+ [project.optional-dependencies]
27
+ graphviz = ["graphviz>=0.20"]
28
+
29
+ [project.urls]
30
+ Homepage = "https://github.com/daviddiazsolis/c50py"
31
+ Issues = "https://github.com/daviddiazsolis/c50py/issues"
32
+
33
+ [tool.setuptools.packages.find]
34
+ where = ["src"]
35
+
36
+ [tool.setuptools]
37
+ include-package-data = true