c50py 0.2.2__tar.gz → 0.2.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {c50py-0.2.2/src/c50py.egg-info → c50py-0.2.4}/PKG-INFO +6 -5
- {c50py-0.2.2 → c50py-0.2.4}/README.md +114 -114
- {c50py-0.2.2 → c50py-0.2.4}/pyproject.toml +37 -37
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py/regressor.py +905 -903
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py/tree.py +1089 -1089
- {c50py-0.2.2 → c50py-0.2.4/src/c50py.egg-info}/PKG-INFO +6 -5
- {c50py-0.2.2 → c50py-0.2.4}/LICENSE +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/setup.cfg +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py/__init__.py +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py.egg-info/SOURCES.txt +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py.egg-info/dependency_links.txt +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py.egg-info/requires.txt +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/src/c50py.egg-info/top_level.txt +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/tests/test_classifier_extended.py +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/tests/test_features.py +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/tests/test_import.py +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/tests/test_regressor_extended.py +0 -0
- {c50py-0.2.2 → c50py-0.2.4}/tests/test_smoke_fit.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: c50py
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.4
|
|
4
4
|
Summary: C5.0-like Decision Trees in Python (scikit-learn style)
|
|
5
5
|
Author-email: David Díaz <daviddiazsolis@gmail.com>
|
|
6
6
|
License: MIT
|
|
@@ -18,10 +18,11 @@ Requires-Dist: numpy>=1.21
|
|
|
18
18
|
Requires-Dist: scikit-learn>=1.2
|
|
19
19
|
Provides-Extra: graphviz
|
|
20
20
|
Requires-Dist: graphviz>=0.20; extra == "graphviz"
|
|
21
|
+
Dynamic: license-file
|
|
21
22
|
|
|
22
|
-
#
|
|
23
|
+
# c50py — C5.0‑style Decision Trees for Python (clean 0.2.0)
|
|
23
24
|
|
|
24
|
-
`
|
|
25
|
+
`c50py` provides transparent, easily inspectable decision trees modelled on
|
|
25
26
|
Quinlan’s C5.0 algorithm. Both classification and regression trees are
|
|
26
27
|
supported and expose a scikit‑learn‑like API. The implementation is written
|
|
27
28
|
from scratch in pure Python/Numpy and includes support for numeric and
|
|
@@ -86,7 +87,7 @@ Fit a regression tree to the diabetes dataset and obtain a visualisation:
|
|
|
86
87
|
```python
|
|
87
88
|
import pandas as pd
|
|
88
89
|
from time import perf_counter
|
|
89
|
-
from
|
|
90
|
+
from c50py import C5Regressor
|
|
90
91
|
|
|
91
92
|
df = pd.read_csv("diabetes.csv")
|
|
92
93
|
y = df["target"].values
|
|
@@ -1,114 +1,114 @@
|
|
|
1
|
-
#
|
|
2
|
-
|
|
3
|
-
`
|
|
4
|
-
Quinlan’s C5.0 algorithm. Both classification and regression trees are
|
|
5
|
-
supported and expose a scikit‑learn‑like API. The implementation is written
|
|
6
|
-
from scratch in pure Python/Numpy and includes support for numeric and
|
|
7
|
-
categorical variables, missing values, pre‑ and post‑pruning, boosting,
|
|
8
|
-
rule tracing/export and Graphviz visualisation.
|
|
9
|
-
|
|
10
|
-
## Features
|
|
11
|
-
|
|
12
|
-
- **Scikit-learn API:** `fit(X, y)`, `predict(X)`, `score(X, y)`.
|
|
13
|
-
- **Categorical support:** Pass `categorical_features=[0, 2]` to handle categories natively.
|
|
14
|
-
- **Sample weights:** Supports `sample_weight` in `fit` for weighted splitting and pruning.
|
|
15
|
-
- **Missing values:** Handles missing values using C5.0's fractional propagation strategy.
|
|
16
|
-
- **Boosting:** Set `trials=10` to train a boosted ensemble.
|
|
17
|
-
- **Rule export:** call `export_rules()` to get a list of human-readable rules.
|
|
18
|
-
- **Graphviz export:** call `export_graphviz()` to visualize the tree.
|
|
19
|
-
- **Pretty printing:** call `print_tree` to display the learned splits in a
|
|
20
|
-
readable nested `if`/`else` format (single trees only).
|
|
21
|
-
|
|
22
|
-
## Documentation
|
|
23
|
-
|
|
24
|
-
For a comprehensive guide on how to use `c50py`, including advanced features and examples, please see the [Usage Guide](USAGE_GUIDE.md).
|
|
25
|
-
|
|
26
|
-
## Installation (development mode)
|
|
27
|
-
|
|
28
|
-
Install the package into your environment in editable mode:
|
|
29
|
-
|
|
30
|
-
```bash
|
|
31
|
-
pip install -e .
|
|
32
|
-
```
|
|
33
|
-
|
|
34
|
-
## Quickstart (Classification)
|
|
35
|
-
|
|
36
|
-
```python
|
|
37
|
-
import pandas as pd
|
|
38
|
-
from time import perf_counter
|
|
39
|
-
from c50py import C5Classifier
|
|
40
|
-
|
|
41
|
-
df = pd.read_csv("titanic.csv")
|
|
42
|
-
t0 = perf_counter(); clf.fit(X, y); print(f"fit: {perf_counter()-t0:.3f}s")
|
|
43
|
-
|
|
44
|
-
# Inspect the tree
|
|
45
|
-
clf.print_tree(feature_names=features, class_names=["No", "Yes"])
|
|
46
|
-
|
|
47
|
-
# Extract rules for each sample
|
|
48
|
-
rules = clf.predict_rule(X, feature_names=features)
|
|
49
|
-
print(rules[:5])
|
|
50
|
-
|
|
51
|
-
# Export as Graphviz
|
|
52
|
-
path = clf.export_graphviz(
|
|
53
|
-
"titanic_tree",
|
|
54
|
-
feature_names=features,
|
|
55
|
-
class_names=["No", "Yes"],
|
|
56
|
-
format="dot" # save a .dot file directly
|
|
57
|
-
)
|
|
58
|
-
print(f"DOT file written to {path}")
|
|
59
|
-
```
|
|
60
|
-
|
|
61
|
-
## Quickstart – Regression (Diabetes)
|
|
62
|
-
|
|
63
|
-
Fit a regression tree to the diabetes dataset and obtain a visualisation:
|
|
64
|
-
|
|
65
|
-
```python
|
|
66
|
-
import pandas as pd
|
|
67
|
-
from time import perf_counter
|
|
68
|
-
from
|
|
69
|
-
|
|
70
|
-
df = pd.read_csv("diabetes.csv")
|
|
71
|
-
y = df["target"].values
|
|
72
|
-
X_df = df.drop(columns=["target"])
|
|
73
|
-
X = X_df.values.astype(object)
|
|
74
|
-
features = list(X_df.columns)
|
|
75
|
-
|
|
76
|
-
reg = C5Regressor(
|
|
77
|
-
min_samples_split=30,
|
|
78
|
-
min_samples_leaf=10,
|
|
79
|
-
pruning=True, cf=0.25, global_pruning=True,
|
|
80
|
-
feature_names=features,
|
|
81
|
-
random_state=42,
|
|
82
|
-
infer_categorical=False, int_as_categorical=False,
|
|
83
|
-
numeric_threshold_strategy="quantile", max_numeric_thresholds=64
|
|
84
|
-
)
|
|
85
|
-
|
|
86
|
-
start = perf_counter(); reg.fit(X, y); print(f"fit: {perf_counter()-start:.3f}s")
|
|
87
|
-
|
|
88
|
-
# Export to DOT (Graphviz installed optional)
|
|
89
|
-
dot_path = reg.export_graphviz("diabetes_tree", feature_names=features, format="dot")
|
|
90
|
-
print(f"Tree saved to {dot_path}")
|
|
91
|
-
|
|
92
|
-
# Export human‑readable rules (single trees only)
|
|
93
|
-
rules = reg.export_rules(feature_names=features)
|
|
94
|
-
print(rules[:3])
|
|
95
|
-
```
|
|
96
|
-
|
|
97
|
-
## Performance tuning
|
|
98
|
-
|
|
99
|
-
Several hyperparameters influence model complexity and performance:
|
|
100
|
-
|
|
101
|
-
- **`numeric_threshold_strategy`** (`'quantile'` | `'all'`): subsample candidate numeric
|
|
102
|
-
thresholds. With `'quantile'` the number of splits considered is limited to
|
|
103
|
-
`max_numeric_thresholds` per feature per node. `'all'` evaluates every unique
|
|
104
|
-
midpoint (slower on large datasets).
|
|
105
|
-
- **`max_numeric_thresholds`**: number of candidate thresholds when using
|
|
106
|
-
`'quantile'` (typically 32–64).
|
|
107
|
-
- **`categorical_features`**: list of names or indices marking categorical columns.
|
|
108
|
-
- **`max_categories_exhaustive`**: maximum cardinality for exhaustive subset search on
|
|
109
|
-
categorical features; beyond this a simpler one‑vs‑rest strategy is used.
|
|
110
|
-
- **`infer_categorical`/`int_as_categorical`**: enable automatic detection of
|
|
111
|
-
categorical/boolean/integer columns when dtype information is not explicit.
|
|
112
|
-
- **`max_depth`**: optional depth limit for extremely noisy or deep trees.
|
|
113
|
-
|
|
114
|
-
When boosting (`trials > 1`) the same hyperparameters apply to each base tree.
|
|
1
|
+
# c50py — C5.0‑style Decision Trees for Python (clean 0.2.0)
|
|
2
|
+
|
|
3
|
+
`c50py` provides transparent, easily inspectable decision trees modelled on
|
|
4
|
+
Quinlan’s C5.0 algorithm. Both classification and regression trees are
|
|
5
|
+
supported and expose a scikit‑learn‑like API. The implementation is written
|
|
6
|
+
from scratch in pure Python/Numpy and includes support for numeric and
|
|
7
|
+
categorical variables, missing values, pre‑ and post‑pruning, boosting,
|
|
8
|
+
rule tracing/export and Graphviz visualisation.
|
|
9
|
+
|
|
10
|
+
## Features
|
|
11
|
+
|
|
12
|
+
- **Scikit-learn API:** `fit(X, y)`, `predict(X)`, `score(X, y)`.
|
|
13
|
+
- **Categorical support:** Pass `categorical_features=[0, 2]` to handle categories natively.
|
|
14
|
+
- **Sample weights:** Supports `sample_weight` in `fit` for weighted splitting and pruning.
|
|
15
|
+
- **Missing values:** Handles missing values using C5.0's fractional propagation strategy.
|
|
16
|
+
- **Boosting:** Set `trials=10` to train a boosted ensemble.
|
|
17
|
+
- **Rule export:** call `export_rules()` to get a list of human-readable rules.
|
|
18
|
+
- **Graphviz export:** call `export_graphviz()` to visualize the tree.
|
|
19
|
+
- **Pretty printing:** call `print_tree` to display the learned splits in a
|
|
20
|
+
readable nested `if`/`else` format (single trees only).
|
|
21
|
+
|
|
22
|
+
## Documentation
|
|
23
|
+
|
|
24
|
+
For a comprehensive guide on how to use `c50py`, including advanced features and examples, please see the [Usage Guide](USAGE_GUIDE.md).
|
|
25
|
+
|
|
26
|
+
## Installation (development mode)
|
|
27
|
+
|
|
28
|
+
Install the package into your environment in editable mode:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install -e .
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Quickstart (Classification)
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
import pandas as pd
|
|
38
|
+
from time import perf_counter
|
|
39
|
+
from c50py import C5Classifier
|
|
40
|
+
|
|
41
|
+
df = pd.read_csv("titanic.csv")
|
|
42
|
+
t0 = perf_counter(); clf.fit(X, y); print(f"fit: {perf_counter()-t0:.3f}s")
|
|
43
|
+
|
|
44
|
+
# Inspect the tree
|
|
45
|
+
clf.print_tree(feature_names=features, class_names=["No", "Yes"])
|
|
46
|
+
|
|
47
|
+
# Extract rules for each sample
|
|
48
|
+
rules = clf.predict_rule(X, feature_names=features)
|
|
49
|
+
print(rules[:5])
|
|
50
|
+
|
|
51
|
+
# Export as Graphviz
|
|
52
|
+
path = clf.export_graphviz(
|
|
53
|
+
"titanic_tree",
|
|
54
|
+
feature_names=features,
|
|
55
|
+
class_names=["No", "Yes"],
|
|
56
|
+
format="dot" # save a .dot file directly
|
|
57
|
+
)
|
|
58
|
+
print(f"DOT file written to {path}")
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
## Quickstart – Regression (Diabetes)
|
|
62
|
+
|
|
63
|
+
Fit a regression tree to the diabetes dataset and obtain a visualisation:
|
|
64
|
+
|
|
65
|
+
```python
|
|
66
|
+
import pandas as pd
|
|
67
|
+
from time import perf_counter
|
|
68
|
+
from c50py import C5Regressor
|
|
69
|
+
|
|
70
|
+
df = pd.read_csv("diabetes.csv")
|
|
71
|
+
y = df["target"].values
|
|
72
|
+
X_df = df.drop(columns=["target"])
|
|
73
|
+
X = X_df.values.astype(object)
|
|
74
|
+
features = list(X_df.columns)
|
|
75
|
+
|
|
76
|
+
reg = C5Regressor(
|
|
77
|
+
min_samples_split=30,
|
|
78
|
+
min_samples_leaf=10,
|
|
79
|
+
pruning=True, cf=0.25, global_pruning=True,
|
|
80
|
+
feature_names=features,
|
|
81
|
+
random_state=42,
|
|
82
|
+
infer_categorical=False, int_as_categorical=False,
|
|
83
|
+
numeric_threshold_strategy="quantile", max_numeric_thresholds=64
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
start = perf_counter(); reg.fit(X, y); print(f"fit: {perf_counter()-start:.3f}s")
|
|
87
|
+
|
|
88
|
+
# Export to DOT (Graphviz installed optional)
|
|
89
|
+
dot_path = reg.export_graphviz("diabetes_tree", feature_names=features, format="dot")
|
|
90
|
+
print(f"Tree saved to {dot_path}")
|
|
91
|
+
|
|
92
|
+
# Export human‑readable rules (single trees only)
|
|
93
|
+
rules = reg.export_rules(feature_names=features)
|
|
94
|
+
print(rules[:3])
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
## Performance tuning
|
|
98
|
+
|
|
99
|
+
Several hyperparameters influence model complexity and performance:
|
|
100
|
+
|
|
101
|
+
- **`numeric_threshold_strategy`** (`'quantile'` | `'all'`): subsample candidate numeric
|
|
102
|
+
thresholds. With `'quantile'` the number of splits considered is limited to
|
|
103
|
+
`max_numeric_thresholds` per feature per node. `'all'` evaluates every unique
|
|
104
|
+
midpoint (slower on large datasets).
|
|
105
|
+
- **`max_numeric_thresholds`**: number of candidate thresholds when using
|
|
106
|
+
`'quantile'` (typically 32–64).
|
|
107
|
+
- **`categorical_features`**: list of names or indices marking categorical columns.
|
|
108
|
+
- **`max_categories_exhaustive`**: maximum cardinality for exhaustive subset search on
|
|
109
|
+
categorical features; beyond this a simpler one‑vs‑rest strategy is used.
|
|
110
|
+
- **`infer_categorical`/`int_as_categorical`**: enable automatic detection of
|
|
111
|
+
categorical/boolean/integer columns when dtype information is not explicit.
|
|
112
|
+
- **`max_depth`**: optional depth limit for extremely noisy or deep trees.
|
|
113
|
+
|
|
114
|
+
When boosting (`trials > 1`) the same hyperparameters apply to each base tree.
|
|
@@ -1,37 +1,37 @@
|
|
|
1
|
-
[build-system]
|
|
2
|
-
requires = ["setuptools>=64", "wheel"]
|
|
3
|
-
build-backend = "setuptools.build_meta"
|
|
4
|
-
|
|
5
|
-
[project]
|
|
6
|
-
name = "c50py"
|
|
7
|
-
version = "0.2.
|
|
8
|
-
description = "C5.0-like Decision Trees in Python (scikit-learn style)"
|
|
9
|
-
readme = "README.md"
|
|
10
|
-
license = { text = "MIT" }
|
|
11
|
-
authors = [{ name = "David Díaz", email = "daviddiazsolis@gmail.com" }]
|
|
12
|
-
requires-python = ">=3.9"
|
|
13
|
-
dependencies = [
|
|
14
|
-
"numpy>=1.21",
|
|
15
|
-
"scikit-learn>=1.2"
|
|
16
|
-
]
|
|
17
|
-
|
|
18
|
-
classifiers = [
|
|
19
|
-
"Programming Language :: Python :: 3",
|
|
20
|
-
"License :: OSI Approved :: MIT License",
|
|
21
|
-
"Operating System :: OS Independent",
|
|
22
|
-
"Intended Audience :: Science/Research",
|
|
23
|
-
"Topic :: Scientific/Engineering :: Artificial Intelligence"
|
|
24
|
-
]
|
|
25
|
-
|
|
26
|
-
[project.optional-dependencies]
|
|
27
|
-
graphviz = ["graphviz>=0.20"]
|
|
28
|
-
|
|
29
|
-
[project.urls]
|
|
30
|
-
Homepage = "https://github.com/daviddiazsolis/c50py"
|
|
31
|
-
Issues = "https://github.com/daviddiazsolis/c50py/issues"
|
|
32
|
-
|
|
33
|
-
[tool.setuptools.packages.find]
|
|
34
|
-
where = ["src"]
|
|
35
|
-
|
|
36
|
-
[tool.setuptools]
|
|
37
|
-
include-package-data = true
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=64", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "c50py"
|
|
7
|
+
version = "0.2.4"
|
|
8
|
+
description = "C5.0-like Decision Trees in Python (scikit-learn style)"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
authors = [{ name = "David Díaz", email = "daviddiazsolis@gmail.com" }]
|
|
12
|
+
requires-python = ">=3.9"
|
|
13
|
+
dependencies = [
|
|
14
|
+
"numpy>=1.21",
|
|
15
|
+
"scikit-learn>=1.2"
|
|
16
|
+
]
|
|
17
|
+
|
|
18
|
+
classifiers = [
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"License :: OSI Approved :: MIT License",
|
|
21
|
+
"Operating System :: OS Independent",
|
|
22
|
+
"Intended Audience :: Science/Research",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence"
|
|
24
|
+
]
|
|
25
|
+
|
|
26
|
+
[project.optional-dependencies]
|
|
27
|
+
graphviz = ["graphviz>=0.20"]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/daviddiazsolis/c50py"
|
|
31
|
+
Issues = "https://github.com/daviddiazsolis/c50py/issues"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools.packages.find]
|
|
34
|
+
where = ["src"]
|
|
35
|
+
|
|
36
|
+
[tool.setuptools]
|
|
37
|
+
include-package-data = true
|