benchmark-reliability 0.2.0__tar.gz → 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/PKG-INFO +12 -12
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/README.md +8 -8
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/pyproject.toml +8 -4
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/setup.py +4 -3
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/PKG-INFO +12 -12
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/SOURCES.txt +30 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/__init__.py +3 -1
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/analyzer.py +91 -70
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/cli.py +22 -65
- benchmark_reliability-0.3.0/src/brf/registry/manifest.yaml +66 -0
- benchmark_reliability-0.3.0/src/brf/registry/registry_v2.1.json +1634 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/__init__.py +155 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/abalone.py +44 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/air_quality_uci.py +73 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/assistments.py +3 -8
- benchmark_reliability-0.3.0/src/brf/registry/sources/auto_mpg.py +47 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/boston_housing.py +42 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/climate_weather.py +49 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/college_scorecard.py +1 -11
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/colleges_aaup.py +0 -1
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/colleges_usnews.py +9 -16
- benchmark_reliability-0.3.0/src/brf/registry/sources/cpu_act.py +41 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/credit_card.py +45 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/cross_domain_batch.py +154 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/customer_churn.py +47 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/electricity.py +44 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/energy_building.py +44 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/energy_efficiency.py +42 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/entrance_exam.py +65 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/external_validation.py +180 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/german_credit.py +46 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/higher_ed.py +62 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/kaggle_students_performance.py +46 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/kdd_cup_2010.py +85 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/law_school.py +51 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/mathe.py +0 -1
- benchmark_reliability-0.3.0/src/brf/registry/sources/mm_tba.py +161 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/nursery.py +56 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/oli.py +2 -15
- benchmark_reliability-0.3.0/src/brf/registry/sources/olympics.py +55 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/oulad.py +75 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/pisa2015.py +62 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/pollution.py +42 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/real_estate.py +48 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/seoul_bike.py +43 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/student_absences.py +50 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/student_depression.py +6 -17
- benchmark_reliability-0.3.0/src/brf/registry/sources/student_dropout.py +51 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/student_health.py +52 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/students_exam_scores.py +55 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/tae.py +0 -1
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/sources/turkiye.py +0 -3
- benchmark_reliability-0.3.0/src/brf/registry/sources/uci_student.py +52 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/uci_student_math.py +58 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/wine_quality.py +54 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/xapi_edu.py +70 -0
- benchmark_reliability-0.3.0/src/brf/registry/sources/yacht.py +42 -0
- benchmark_reliability-0.3.0/src/brf/registry/verify.py +114 -0
- benchmark_reliability-0.2.0/src/brf/registry/manifest.yaml +0 -47
- benchmark_reliability-0.2.0/src/brf/registry/sources/__init__.py +0 -216
- benchmark_reliability-0.2.0/src/brf/registry/sources/entrance_exam.py +0 -55
- benchmark_reliability-0.2.0/src/brf/registry/sources/higher_ed.py +0 -53
- benchmark_reliability-0.2.0/src/brf/registry/sources/mm_tba.py +0 -49
- benchmark_reliability-0.2.0/src/brf/registry/sources/oulad.py +0 -52
- benchmark_reliability-0.2.0/src/brf/registry/sources/student_dropout.py +0 -75
- benchmark_reliability-0.2.0/src/brf/registry/sources/uci_student.py +0 -53
- benchmark_reliability-0.2.0/src/brf/registry/sources/xapi_edu.py +0 -51
- benchmark_reliability-0.2.0/src/brf/registry/verify.py +0 -73
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/setup.cfg +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/dependency_links.txt +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/entry_points.txt +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/requires.txt +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/benchmark_reliability.egg-info/top_level.txt +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/cli.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/metrics/__init__.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/metrics/baseline_gap.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/metrics/instability.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/metrics/metadata.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/metrics/null_test.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/phase/__init__.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/phase/classifier.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/phase/embedding.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/phase/visualization.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/registry/__init__.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/report/__init__.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/report/json_export.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/src/brf/report/latex_export.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/tests/test_analyzer.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/tests/test_metrics.py +0 -0
- {benchmark_reliability-0.2.0 → benchmark_reliability-0.3.0}/tests/test_phase.py +0 -0
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: benchmark-reliability
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Benchmark Reliability Framework (BRF) - dataset-level reliability auditing with built-in benchmark registry
|
|
5
5
|
Author-email: zhanglizhuo <zhanglizhuo@gmail.com>
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/zhanglizhuo/
|
|
8
|
-
Project-URL: Repository, https://github.com/zhanglizhuo/
|
|
7
|
+
Project-URL: Homepage, https://github.com/zhanglizhuo/BRFPackage
|
|
8
|
+
Project-URL: Repository, https://github.com/zhanglizhuo/BRFPackage
|
|
9
9
|
Keywords: benchmark reliability,dataset auditing,educational AI,machine learning,registry
|
|
10
10
|
Classifier: Development Status :: 3 - Alpha
|
|
11
11
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -65,9 +65,9 @@ print(analyzer.diagnose()["summary"])
|
|
|
65
65
|
for dim, issue in analyzer.diagnose()["details"].items():
|
|
66
66
|
print(f" {dim}: {issue}")
|
|
67
67
|
|
|
68
|
-
# Percentile rank against
|
|
68
|
+
# Percentile rank against the 51 benchmarks in the BRF Registry (v2.1 reference)
|
|
69
69
|
print(analyzer.rank())
|
|
70
|
-
# {'S_percentile':
|
|
70
|
+
# {'S_percentile': 17.6, 'E_percentile': 27.5, 'reference': 'BRF Registry v2.1 (51 benchmarks)'}
|
|
71
71
|
|
|
72
72
|
# One-paragraph recommendation
|
|
73
73
|
print(analyzer.recommend())
|
|
@@ -77,7 +77,7 @@ print(analyzer.recommend())
|
|
|
77
77
|
|
|
78
78
|
```bash
|
|
79
79
|
$ brf registry list
|
|
80
|
-
BRF Registry
|
|
80
|
+
BRF Registry
|
|
81
81
|
|
|
82
82
|
assistments ASSISTments 2009-2010 N= 3729 G= 124
|
|
83
83
|
college_scorecard US College Scorecard N= 7804 G= 59
|
|
@@ -95,7 +95,7 @@ BRF Audit: Teaching Assistant Evaluation (tae)
|
|
|
95
95
|
### Download and verify all datasets
|
|
96
96
|
|
|
97
97
|
```bash
|
|
98
|
-
$ brf registry sync # download + SHA-256 verify all
|
|
98
|
+
$ brf registry sync # download + SHA-256 verify all
|
|
99
99
|
$ brf registry info oulad
|
|
100
100
|
name: oulad
|
|
101
101
|
display_name: Open University Learning Analytics Dataset
|
|
@@ -122,7 +122,7 @@ communication shorthand only --- **the signal is in the continuous values**.
|
|
|
122
122
|
|
|
123
123
|
Use `analyzer.diagnose()` for per-dimension explanations and actionable
|
|
124
124
|
recommendations, or `analyzer.rank()` to see percentile scores against
|
|
125
|
-
the
|
|
125
|
+
the 51 benchmarks in the BRF Registry (v2.1 reference).
|
|
126
126
|
|
|
127
127
|
## CLI Reference
|
|
128
128
|
|
|
@@ -152,8 +152,8 @@ To cite the BRF framework and package (JOSS paper forthcoming):
|
|
|
152
152
|
@software{zhang2026brf,
|
|
153
153
|
author = {Lizhuo Zhang},
|
|
154
154
|
title = {benchmark-reliability: Benchmark Reliability Framework},
|
|
155
|
-
url = {https://github.com/zhanglizhuo/
|
|
156
|
-
version = {0.
|
|
155
|
+
url = {https://github.com/zhanglizhuo/BRFPackage},
|
|
156
|
+
version = {0.3.0},
|
|
157
157
|
year = {2026},
|
|
158
158
|
}
|
|
159
159
|
```
|
|
@@ -162,7 +162,7 @@ The behavior audit protocol is described in:
|
|
|
162
162
|
|
|
163
163
|
> Zhang, L. *BehaviorAudit: a four-dimension protocol for auditing
|
|
164
164
|
> benchmark reliability under group-aware evaluation.*
|
|
165
|
-
> Scientific Reports (
|
|
165
|
+
> Scientific Reports (2026). https://doi.org/10.1038/s41598-026-69629-6
|
|
166
166
|
|
|
167
167
|
## Related Work
|
|
168
168
|
|
|
@@ -38,9 +38,9 @@ print(analyzer.diagnose()["summary"])
|
|
|
38
38
|
for dim, issue in analyzer.diagnose()["details"].items():
|
|
39
39
|
print(f" {dim}: {issue}")
|
|
40
40
|
|
|
41
|
-
# Percentile rank against
|
|
41
|
+
# Percentile rank against the 51 benchmarks in the BRF Registry (v2.1 reference)
|
|
42
42
|
print(analyzer.rank())
|
|
43
|
-
# {'S_percentile':
|
|
43
|
+
# {'S_percentile': 17.6, 'E_percentile': 27.5, 'reference': 'BRF Registry v2.1 (51 benchmarks)'}
|
|
44
44
|
|
|
45
45
|
# One-paragraph recommendation
|
|
46
46
|
print(analyzer.recommend())
|
|
@@ -50,7 +50,7 @@ print(analyzer.recommend())
|
|
|
50
50
|
|
|
51
51
|
```bash
|
|
52
52
|
$ brf registry list
|
|
53
|
-
BRF Registry
|
|
53
|
+
BRF Registry
|
|
54
54
|
|
|
55
55
|
assistments ASSISTments 2009-2010 N= 3729 G= 124
|
|
56
56
|
college_scorecard US College Scorecard N= 7804 G= 59
|
|
@@ -68,7 +68,7 @@ BRF Audit: Teaching Assistant Evaluation (tae)
|
|
|
68
68
|
### Download and verify all datasets
|
|
69
69
|
|
|
70
70
|
```bash
|
|
71
|
-
$ brf registry sync # download + SHA-256 verify all
|
|
71
|
+
$ brf registry sync # download + SHA-256 verify all
|
|
72
72
|
$ brf registry info oulad
|
|
73
73
|
name: oulad
|
|
74
74
|
display_name: Open University Learning Analytics Dataset
|
|
@@ -95,7 +95,7 @@ communication shorthand only --- **the signal is in the continuous values**.
|
|
|
95
95
|
|
|
96
96
|
Use `analyzer.diagnose()` for per-dimension explanations and actionable
|
|
97
97
|
recommendations, or `analyzer.rank()` to see percentile scores against
|
|
98
|
-
the
|
|
98
|
+
the 51 benchmarks in the BRF Registry (v2.1 reference).
|
|
99
99
|
|
|
100
100
|
## CLI Reference
|
|
101
101
|
|
|
@@ -125,8 +125,8 @@ To cite the BRF framework and package (JOSS paper forthcoming):
|
|
|
125
125
|
@software{zhang2026brf,
|
|
126
126
|
author = {Lizhuo Zhang},
|
|
127
127
|
title = {benchmark-reliability: Benchmark Reliability Framework},
|
|
128
|
-
url = {https://github.com/zhanglizhuo/
|
|
129
|
-
version = {0.
|
|
128
|
+
url = {https://github.com/zhanglizhuo/BRFPackage},
|
|
129
|
+
version = {0.3.0},
|
|
130
130
|
year = {2026},
|
|
131
131
|
}
|
|
132
132
|
```
|
|
@@ -135,7 +135,7 @@ The behavior audit protocol is described in:
|
|
|
135
135
|
|
|
136
136
|
> Zhang, L. *BehaviorAudit: a four-dimension protocol for auditing
|
|
137
137
|
> benchmark reliability under group-aware evaluation.*
|
|
138
|
-
> Scientific Reports (
|
|
138
|
+
> Scientific Reports (2026). https://doi.org/10.1038/s41598-026-69629-6
|
|
139
139
|
|
|
140
140
|
## Related Work
|
|
141
141
|
|
|
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
|
|
|
4
4
|
|
|
5
5
|
[project]
|
|
6
6
|
name = "benchmark-reliability"
|
|
7
|
-
version = "0.
|
|
7
|
+
version = "0.3.0"
|
|
8
8
|
description = "Benchmark Reliability Framework (BRF) - dataset-level reliability auditing with built-in benchmark registry"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = { text = "MIT" }
|
|
@@ -37,14 +37,18 @@ dependencies = [
|
|
|
37
37
|
brf = "brf.cli:main"
|
|
38
38
|
|
|
39
39
|
[project.urls]
|
|
40
|
-
Homepage = "https://github.com/zhanglizhuo/
|
|
41
|
-
Repository = "https://github.com/zhanglizhuo/
|
|
40
|
+
Homepage = "https://github.com/zhanglizhuo/BRFPackage"
|
|
41
|
+
Repository = "https://github.com/zhanglizhuo/BRFPackage"
|
|
42
42
|
|
|
43
43
|
[tool.setuptools]
|
|
44
44
|
license-files = []
|
|
45
45
|
|
|
46
46
|
[tool.setuptools.packages.find]
|
|
47
47
|
where = ["src"]
|
|
48
|
+
include = ["brf", "brf.metrics", "brf.phase", "brf.registry", "brf.registry.sources", "brf.report"]
|
|
48
49
|
|
|
49
50
|
[tool.setuptools.package-data]
|
|
50
|
-
"brf.registry" = ["manifest.yaml"]
|
|
51
|
+
"brf.registry" = ["manifest.yaml", "registry_v2.1.json"]
|
|
52
|
+
|
|
53
|
+
[tool.setuptools.exclude-package-data]
|
|
54
|
+
"brf.registry" = ["cache/*", "cache/**/*"]
|
|
@@ -2,15 +2,16 @@ from setuptools import setup, find_packages
|
|
|
2
2
|
|
|
3
3
|
setup(
|
|
4
4
|
name="benchmark-reliability",
|
|
5
|
-
version="0.
|
|
5
|
+
version="0.3.0",
|
|
6
6
|
packages=find_packages(where="src"),
|
|
7
7
|
package_dir={"": "src"},
|
|
8
8
|
package_data={
|
|
9
9
|
"brf.registry": ["manifest.yaml"],
|
|
10
10
|
},
|
|
11
11
|
install_requires=[
|
|
12
|
-
"numpy>=1.
|
|
13
|
-
"scikit-learn>=0
|
|
12
|
+
"numpy>=1.21",
|
|
13
|
+
"scikit-learn>=1.0",
|
|
14
|
+
"matplotlib>=3.5",
|
|
14
15
|
"pandas>=1.1",
|
|
15
16
|
"scipy>=1.5",
|
|
16
17
|
"openml>=0.12",
|
|
@@ -1,11 +1,11 @@
|
|
|
1
|
-
Metadata-Version: 2.
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
2
|
Name: benchmark-reliability
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.3.0
|
|
4
4
|
Summary: Benchmark Reliability Framework (BRF) - dataset-level reliability auditing with built-in benchmark registry
|
|
5
5
|
Author-email: zhanglizhuo <zhanglizhuo@gmail.com>
|
|
6
6
|
License: MIT
|
|
7
|
-
Project-URL: Homepage, https://github.com/zhanglizhuo/
|
|
8
|
-
Project-URL: Repository, https://github.com/zhanglizhuo/
|
|
7
|
+
Project-URL: Homepage, https://github.com/zhanglizhuo/BRFPackage
|
|
8
|
+
Project-URL: Repository, https://github.com/zhanglizhuo/BRFPackage
|
|
9
9
|
Keywords: benchmark reliability,dataset auditing,educational AI,machine learning,registry
|
|
10
10
|
Classifier: Development Status :: 3 - Alpha
|
|
11
11
|
Classifier: License :: OSI Approved :: MIT License
|
|
@@ -65,9 +65,9 @@ print(analyzer.diagnose()["summary"])
|
|
|
65
65
|
for dim, issue in analyzer.diagnose()["details"].items():
|
|
66
66
|
print(f" {dim}: {issue}")
|
|
67
67
|
|
|
68
|
-
# Percentile rank against
|
|
68
|
+
# Percentile rank against the 51 benchmarks in the BRF Registry (v2.1 reference)
|
|
69
69
|
print(analyzer.rank())
|
|
70
|
-
# {'S_percentile':
|
|
70
|
+
# {'S_percentile': 17.6, 'E_percentile': 27.5, 'reference': 'BRF Registry v2.1 (51 benchmarks)'}
|
|
71
71
|
|
|
72
72
|
# One-paragraph recommendation
|
|
73
73
|
print(analyzer.recommend())
|
|
@@ -77,7 +77,7 @@ print(analyzer.recommend())
|
|
|
77
77
|
|
|
78
78
|
```bash
|
|
79
79
|
$ brf registry list
|
|
80
|
-
BRF Registry
|
|
80
|
+
BRF Registry
|
|
81
81
|
|
|
82
82
|
assistments ASSISTments 2009-2010 N= 3729 G= 124
|
|
83
83
|
college_scorecard US College Scorecard N= 7804 G= 59
|
|
@@ -95,7 +95,7 @@ BRF Audit: Teaching Assistant Evaluation (tae)
|
|
|
95
95
|
### Download and verify all datasets
|
|
96
96
|
|
|
97
97
|
```bash
|
|
98
|
-
$ brf registry sync # download + SHA-256 verify all
|
|
98
|
+
$ brf registry sync # download + SHA-256 verify all
|
|
99
99
|
$ brf registry info oulad
|
|
100
100
|
name: oulad
|
|
101
101
|
display_name: Open University Learning Analytics Dataset
|
|
@@ -122,7 +122,7 @@ communication shorthand only --- **the signal is in the continuous values**.
|
|
|
122
122
|
|
|
123
123
|
Use `analyzer.diagnose()` for per-dimension explanations and actionable
|
|
124
124
|
recommendations, or `analyzer.rank()` to see percentile scores against
|
|
125
|
-
the
|
|
125
|
+
the 51 benchmarks in the BRF Registry (v2.1 reference).
|
|
126
126
|
|
|
127
127
|
## CLI Reference
|
|
128
128
|
|
|
@@ -152,8 +152,8 @@ To cite the BRF framework and package (JOSS paper forthcoming):
|
|
|
152
152
|
@software{zhang2026brf,
|
|
153
153
|
author = {Lizhuo Zhang},
|
|
154
154
|
title = {benchmark-reliability: Benchmark Reliability Framework},
|
|
155
|
-
url = {https://github.com/zhanglizhuo/
|
|
156
|
-
version = {0.
|
|
155
|
+
url = {https://github.com/zhanglizhuo/BRFPackage},
|
|
156
|
+
version = {0.3.0},
|
|
157
157
|
year = {2026},
|
|
158
158
|
}
|
|
159
159
|
```
|
|
@@ -162,7 +162,7 @@ The behavior audit protocol is described in:
|
|
|
162
162
|
|
|
163
163
|
> Zhang, L. *BehaviorAudit: a four-dimension protocol for auditing
|
|
164
164
|
> benchmark reliability under group-aware evaluation.*
|
|
165
|
-
> Scientific Reports (
|
|
165
|
+
> Scientific Reports (2026). https://doi.org/10.1038/s41598-026-69629-6
|
|
166
166
|
|
|
167
167
|
## Related Work
|
|
168
168
|
|
|
@@ -22,24 +22,54 @@ src/brf/phase/visualization.py
|
|
|
22
22
|
src/brf/registry/__init__.py
|
|
23
23
|
src/brf/registry/cli.py
|
|
24
24
|
src/brf/registry/manifest.yaml
|
|
25
|
+
src/brf/registry/registry_v2.1.json
|
|
25
26
|
src/brf/registry/verify.py
|
|
26
27
|
src/brf/registry/sources/__init__.py
|
|
28
|
+
src/brf/registry/sources/abalone.py
|
|
29
|
+
src/brf/registry/sources/air_quality_uci.py
|
|
27
30
|
src/brf/registry/sources/assistments.py
|
|
31
|
+
src/brf/registry/sources/auto_mpg.py
|
|
32
|
+
src/brf/registry/sources/boston_housing.py
|
|
33
|
+
src/brf/registry/sources/climate_weather.py
|
|
28
34
|
src/brf/registry/sources/college_scorecard.py
|
|
29
35
|
src/brf/registry/sources/colleges_aaup.py
|
|
30
36
|
src/brf/registry/sources/colleges_usnews.py
|
|
37
|
+
src/brf/registry/sources/cpu_act.py
|
|
38
|
+
src/brf/registry/sources/credit_card.py
|
|
39
|
+
src/brf/registry/sources/cross_domain_batch.py
|
|
40
|
+
src/brf/registry/sources/customer_churn.py
|
|
41
|
+
src/brf/registry/sources/electricity.py
|
|
42
|
+
src/brf/registry/sources/energy_building.py
|
|
43
|
+
src/brf/registry/sources/energy_efficiency.py
|
|
31
44
|
src/brf/registry/sources/entrance_exam.py
|
|
45
|
+
src/brf/registry/sources/external_validation.py
|
|
46
|
+
src/brf/registry/sources/german_credit.py
|
|
32
47
|
src/brf/registry/sources/higher_ed.py
|
|
48
|
+
src/brf/registry/sources/kaggle_students_performance.py
|
|
49
|
+
src/brf/registry/sources/kdd_cup_2010.py
|
|
50
|
+
src/brf/registry/sources/law_school.py
|
|
33
51
|
src/brf/registry/sources/mathe.py
|
|
34
52
|
src/brf/registry/sources/mm_tba.py
|
|
53
|
+
src/brf/registry/sources/nursery.py
|
|
35
54
|
src/brf/registry/sources/oli.py
|
|
55
|
+
src/brf/registry/sources/olympics.py
|
|
36
56
|
src/brf/registry/sources/oulad.py
|
|
57
|
+
src/brf/registry/sources/pisa2015.py
|
|
58
|
+
src/brf/registry/sources/pollution.py
|
|
59
|
+
src/brf/registry/sources/real_estate.py
|
|
60
|
+
src/brf/registry/sources/seoul_bike.py
|
|
61
|
+
src/brf/registry/sources/student_absences.py
|
|
37
62
|
src/brf/registry/sources/student_depression.py
|
|
38
63
|
src/brf/registry/sources/student_dropout.py
|
|
64
|
+
src/brf/registry/sources/student_health.py
|
|
65
|
+
src/brf/registry/sources/students_exam_scores.py
|
|
39
66
|
src/brf/registry/sources/tae.py
|
|
40
67
|
src/brf/registry/sources/turkiye.py
|
|
41
68
|
src/brf/registry/sources/uci_student.py
|
|
69
|
+
src/brf/registry/sources/uci_student_math.py
|
|
70
|
+
src/brf/registry/sources/wine_quality.py
|
|
42
71
|
src/brf/registry/sources/xapi_edu.py
|
|
72
|
+
src/brf/registry/sources/yacht.py
|
|
43
73
|
src/brf/report/__init__.py
|
|
44
74
|
src/brf/report/json_export.py
|
|
45
75
|
src/brf/report/latex_export.py
|
|
@@ -27,7 +27,7 @@ class BRFAnalyzer:
|
|
|
27
27
|
raise ValueError("n_splits must be >= 2")
|
|
28
28
|
self.n_splits = n_splits
|
|
29
29
|
self.n_permutations = n_permutations
|
|
30
|
-
self.model = model
|
|
30
|
+
self.model = model if model is not None else Ridge(alpha=1.0)
|
|
31
31
|
self.seed = seed
|
|
32
32
|
self.scale = scale
|
|
33
33
|
|
|
@@ -137,11 +137,15 @@ class BRFAnalyzer:
|
|
|
137
137
|
|
|
138
138
|
# ---- improved reporting (v0.2) ----
|
|
139
139
|
|
|
140
|
-
def diagnose(self
|
|
140
|
+
def diagnose(self, n_samples: Optional[int] = None,
|
|
141
|
+
n_features: Optional[int] = None,
|
|
142
|
+
n_groups: Optional[int] = None) -> Dict[str, str]:
|
|
141
143
|
"""Return structured diagnosis explaining *why* the dataset is in its current state.
|
|
142
144
|
|
|
143
|
-
|
|
144
|
-
|
|
145
|
+
Args:
|
|
146
|
+
n_samples: Total sample count (for context-aware suggestions).
|
|
147
|
+
n_features: Feature count.
|
|
148
|
+
n_groups: Group count.
|
|
145
149
|
"""
|
|
146
150
|
if not self._fitted:
|
|
147
151
|
raise RuntimeError("call fit() before accessing diagnose()")
|
|
@@ -149,71 +153,81 @@ class BRFAnalyzer:
|
|
|
149
153
|
issues = {}
|
|
150
154
|
suggestions = {}
|
|
151
155
|
|
|
156
|
+
n = n_samples or 0
|
|
157
|
+
p = n_features or 0
|
|
158
|
+
g = n_groups or 0
|
|
159
|
+
|
|
152
160
|
# --- Predictive signal (B) ---
|
|
153
161
|
if self.B < 0:
|
|
154
162
|
issues["B"] = (f"Model performs WORSE than the mean baseline "
|
|
155
|
-
f"(B={self.B:.3f}).
|
|
156
|
-
|
|
157
|
-
|
|
163
|
+
f"(B={self.B:.3f}). Features carry no useful signal.")
|
|
164
|
+
suggestions["B"] = ("Reconsider feature engineering and target definition. "
|
|
165
|
+
"The chosen features cannot predict this target.")
|
|
158
166
|
elif self.B < 0.05:
|
|
159
|
-
issues["B"] = (f"Marginal
|
|
160
|
-
f"
|
|
161
|
-
suggestions["B"] = "Add more informative features or reframe the task."
|
|
167
|
+
issues["B"] = (f"Marginal signal (B={self.B:.3f}). "
|
|
168
|
+
f"Features explain very little variance.")
|
|
169
|
+
suggestions["B"] = "Add more informative features or reframe the prediction task."
|
|
162
170
|
elif self.B < 0.2:
|
|
163
|
-
issues["B"] = (f"Moderate
|
|
171
|
+
issues["B"] = (f"Moderate signal (B={self.B:.3f}).")
|
|
164
172
|
suggestions["B"] = None
|
|
165
173
|
else:
|
|
166
|
-
issues["B"] = (f"Strong
|
|
174
|
+
issues["B"] = (f"Strong signal (B={self.B:.3f}).")
|
|
167
175
|
suggestions["B"] = None
|
|
168
176
|
|
|
169
177
|
# --- Instability (I) ---
|
|
170
178
|
if self.I > 1.0:
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
179
|
+
per_group = f"~{n//g} per group" if g > 0 and n > 0 else ""
|
|
180
|
+
n_feat_ratio = f" (N/p={n//p})" if p > 0 and n > 0 else ""
|
|
181
|
+
issues["I"] = (f"High instability (I={self.I:.3f}). "
|
|
182
|
+
f"R^2 varies dramatically across data splits.")
|
|
183
|
+
if n > 0 and n < 200:
|
|
184
|
+
suggestions["I"] = (f"Only N={n} samples{per_group}. "
|
|
185
|
+
f"Increase to 300+ for stable estimates.")
|
|
186
|
+
elif p > 0 and n > 0 and n / p < 10:
|
|
187
|
+
suggestions["I"] = (f"N/p={n//p} is low{n_feat_ratio}. "
|
|
188
|
+
f"Increase N or reduce features (currently {p}).")
|
|
189
|
+
else:
|
|
190
|
+
suggestions["I"] = "Increase N, reduce p, or use stronger regularization."
|
|
176
191
|
elif self.I > 0.3:
|
|
177
192
|
issues["I"] = (f"Moderate instability (I={self.I:.3f}).")
|
|
178
|
-
suggestions["I"] = "Consider larger N
|
|
193
|
+
suggestions["I"] = "Consider larger N for more stable estimates."
|
|
179
194
|
else:
|
|
180
|
-
issues["I"] = (f"Low instability (I={self.I:.3f}). "
|
|
181
|
-
f"Model is robust to data split variation.")
|
|
195
|
+
issues["I"] = (f"Low instability (I={self.I:.3f}). Stable across splits.")
|
|
182
196
|
suggestions["I"] = None
|
|
183
197
|
|
|
184
198
|
# --- Null separation (N) ---
|
|
185
199
|
if self.N < 0.5:
|
|
186
|
-
issues["N"] = (f"
|
|
187
|
-
f"(N={self.N:.3f}).
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
200
|
+
issues["N"] = (f"Signal indistinguishable from noise "
|
|
201
|
+
f"(N={self.N:.3f}). Model rarely beats permutation.")
|
|
202
|
+
if self.B <= 0:
|
|
203
|
+
suggestions["N"] = "No predictive relationship detected. Reconsider features/target."
|
|
204
|
+
else:
|
|
205
|
+
suggestions["N"] = "Weak signal. Increase N or simplify the feature set."
|
|
191
206
|
elif self.N < 0.8:
|
|
192
|
-
issues["N"] = (f"
|
|
193
|
-
|
|
194
|
-
suggestions["N"] = "Increase sample size or feature quality for more reliable separation."
|
|
207
|
+
issues["N"] = (f"Inconsistent signal separation (N={self.N:.3f}).")
|
|
208
|
+
suggestions["N"] = "Increase N or improve feature quality."
|
|
195
209
|
else:
|
|
196
|
-
issues["N"] = (f"
|
|
197
|
-
f"(N={self.N:.3f}). Clear signal above noise.")
|
|
210
|
+
issues["N"] = (f"Clear signal above noise (N={self.N:.3f}).")
|
|
198
211
|
suggestions["N"] = None
|
|
199
212
|
|
|
200
213
|
# --- Metadata adequacy (M) ---
|
|
201
214
|
if self.M < 0.1:
|
|
202
|
-
issues["M"] = (f"Insufficient group
|
|
203
|
-
f"Groups are too few,
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
215
|
+
issues["M"] = (f"Insufficient group structure (M={self.M:.3f}). "
|
|
216
|
+
f"Groups are too few, absent, or severely imbalanced.")
|
|
217
|
+
if g < 5:
|
|
218
|
+
suggestions["M"] = (f"Only {g} group(s). Add group annotations "
|
|
219
|
+
f"with >=5 categories for meaningful cross-group evaluation.")
|
|
220
|
+
else:
|
|
221
|
+
suggestions["M"] = (f"{g} groups but highly imbalanced. "
|
|
222
|
+
f"Use a more balanced grouping variable.")
|
|
207
223
|
elif self.M < 0.3:
|
|
208
|
-
issues["M"] = (f"Weak group
|
|
209
|
-
f"Group structure exists but is sparse or imbalanced.")
|
|
224
|
+
issues["M"] = (f"Weak group structure (M={self.M:.3f}).")
|
|
210
225
|
suggestions["M"] = "Use a finer-grained grouping variable if available."
|
|
211
226
|
elif self.M < 0.5:
|
|
212
|
-
issues["M"] = (f"Moderate group
|
|
227
|
+
issues["M"] = (f"Moderate group structure (M={self.M:.3f}).")
|
|
213
228
|
suggestions["M"] = None
|
|
214
229
|
else:
|
|
215
|
-
issues["M"] = (f"Strong group
|
|
216
|
-
f"Group structure is well-defined and balanced.")
|
|
230
|
+
issues["M"] = (f"Strong group structure (M={self.M:.3f}).")
|
|
217
231
|
suggestions["M"] = None
|
|
218
232
|
|
|
219
233
|
# --- Synthesis ---
|
|
@@ -237,10 +251,10 @@ class BRFAnalyzer:
|
|
|
237
251
|
}
|
|
238
252
|
|
|
239
253
|
def rank(self) -> Dict[str, float]:
|
|
240
|
-
"""Percentile rank of S and E against the BRF Registry
|
|
254
|
+
"""Percentile rank of S and E against the BRF Registry v2.1 benchmarks.
|
|
241
255
|
|
|
242
256
|
Returns percentiles (0-100) indicating where this dataset's S and E
|
|
243
|
-
fall relative to the
|
|
257
|
+
fall relative to the audited benchmarks in the bundled Registry reference (v2.1). Requires
|
|
244
258
|
data to be accessible.
|
|
245
259
|
"""
|
|
246
260
|
if not self._fitted:
|
|
@@ -259,45 +273,52 @@ class BRFAnalyzer:
|
|
|
259
273
|
return {
|
|
260
274
|
"S_percentile": round(pctile(s_vals, self.S), 1),
|
|
261
275
|
"E_percentile": round(pctile(e_vals, self.E), 1),
|
|
262
|
-
"reference": f"BRF Registry
|
|
276
|
+
"reference": f"BRF Registry v2.1 ({len(s_vals)} benchmarks)",
|
|
263
277
|
}
|
|
264
278
|
|
|
265
|
-
def recommend(self
|
|
266
|
-
|
|
267
|
-
|
|
279
|
+
def recommend(self, n_samples: Optional[int] = None,
|
|
280
|
+
n_features: Optional[int] = None,
|
|
281
|
+
n_groups: Optional[int] = None) -> str:
|
|
282
|
+
"""Actionable recommendations for benchmark improvement.
|
|
283
|
+
|
|
284
|
+
Args:
|
|
285
|
+
n_samples, n_features, n_groups: Optional context for concrete suggestions
|
|
286
|
+
(e.g., "Only N=151 samples. Increase to 300+").
|
|
287
|
+
"""
|
|
288
|
+
d = self.diagnose(n_samples, n_features, n_groups)
|
|
268
289
|
recs = d["recommendations"]
|
|
269
290
|
if not recs:
|
|
270
|
-
return
|
|
271
|
-
"No specific action recommended.")
|
|
272
|
-
# Prioritize: B < 0 is most critical, then N < 0.5, then I > 1, then M < 0.1
|
|
273
|
-
priority = []
|
|
274
|
-
if self.B < 0:
|
|
275
|
-
priority.append("B")
|
|
276
|
-
if self.N < 0.5:
|
|
277
|
-
priority.append("N")
|
|
278
|
-
if self.I > 1.0:
|
|
279
|
-
priority.append("I")
|
|
280
|
-
if self.M < 0.1:
|
|
281
|
-
priority.append("M")
|
|
282
|
-
if not priority:
|
|
283
|
-
priority = [k for k in recs]
|
|
291
|
+
return "No issues found. Your benchmark metrics are within normal ranges."
|
|
284
292
|
|
|
285
|
-
lines = [
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
return "
|
|
293
|
+
lines = []
|
|
294
|
+
for dim in ["B", "N", "I", "M"]:
|
|
295
|
+
if dim in recs:
|
|
296
|
+
lines.append(f"[{dim}] {recs[dim]}")
|
|
297
|
+
return "\n".join(lines)
|
|
298
|
+
|
|
299
|
+
def recommend_dict(self) -> Dict:
|
|
300
|
+
"""Structured actionable recommendations as a dict.
|
|
301
|
+
|
|
302
|
+
Returns {dimension: {"issue": ..., "action": ..., "value": ...}}.
|
|
303
|
+
"""
|
|
304
|
+
d = self.diagnose()
|
|
305
|
+
out = {}
|
|
306
|
+
for dim in ["B", "N", "I", "M"]:
|
|
307
|
+
if dim in d["details"] and dim in d["recommendations"]:
|
|
308
|
+
out[dim] = {
|
|
309
|
+
"issue": d["details"][dim],
|
|
310
|
+
"action": d["recommendations"][dim],
|
|
311
|
+
"value": getattr(self, dim),
|
|
312
|
+
}
|
|
313
|
+
return out
|
|
290
314
|
|
|
291
315
|
def _load_registry_ref(self) -> Optional[List[Dict]]:
|
|
292
316
|
"""Load Registry reference data for percentile ranking."""
|
|
293
317
|
if self._registry_ref is not None:
|
|
294
318
|
return self._registry_ref
|
|
295
|
-
#
|
|
319
|
+
# Bundled registry reference (ships with the package at brf/registry/)
|
|
296
320
|
candidates = [
|
|
297
|
-
Path(__file__).parent.
|
|
298
|
-
/ "BRFRegistry" / "results" / "registry_v1.5.json",
|
|
299
|
-
Path(__file__).parent.parent.parent
|
|
300
|
-
/ "BRFRegistry" / "results" / "registry_v1.5.json",
|
|
321
|
+
Path(__file__).parent / "registry" / "registry_v2.1.json",
|
|
301
322
|
]
|
|
302
323
|
for p in candidates:
|
|
303
324
|
if p.exists():
|