FeatureRank 0.1.0__tar.gz → 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- featurerank-0.1.2/FeatureRank/GUI.py +589 -0
- featurerank-0.1.2/FeatureRank/__init__.py +80 -0
- featurerank-0.1.2/FeatureRank/__main__.py +7 -0
- featurerank-0.1.2/FeatureRank.egg-info/PKG-INFO +435 -0
- featurerank-0.1.2/FeatureRank.egg-info/SOURCES.txt +48 -0
- featurerank-0.1.2/FeatureRank.egg-info/dependency_links.txt +1 -0
- featurerank-0.1.2/FeatureRank.egg-info/entry_points.txt +2 -0
- featurerank-0.1.2/FeatureRank.egg-info/requires.txt +6 -0
- featurerank-0.1.2/FeatureRank.egg-info/top_level.txt +3 -0
- featurerank-0.1.2/PKG-INFO +435 -0
- featurerank-0.1.2/README.md +420 -0
- featurerank-0.1.2/pyproject.toml +26 -0
- featurerank-0.1.2/scripts/EvaluatePaperDimensionReduction.py +383 -0
- featurerank-0.1.2/scripts/FeatureBlockDatasetTools.py +628 -0
- featurerank-0.1.2/scripts/FeatureRank.py +175 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateAccuracyBoxplotFromList.py +220 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateClassificationFigure.py +487 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateClusterClassificationSummaryFigure.py +649 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateClusterFigure1.py +513 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateConfusionMatrixPanel.py +118 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreatePrecisionRecallPanel.py +101 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateRegressionFigure2.py +766 -0
- featurerank-0.1.2/scripts/Figure_Scripts/CreateRocPanel.py +95 -0
- featurerank-0.1.2/scripts/GeneratePaperDimensionReduction.py +333 -0
- featurerank-0.1.2/scripts/PaperDimensionReductionCommon.py +336 -0
- featurerank-0.1.2/scripts/RunAutoencoder.py +19 -0
- featurerank-0.1.2/scripts/RunBlockFeatureSelection.py +463 -0
- featurerank-0.1.2/scripts/RunDimensionReduction.py +145 -0
- featurerank-0.1.2/scripts/RunFeatureRankCV.py +689 -0
- featurerank-0.1.2/scripts/Simple.py +112 -0
- featurerank-0.1.2/scripts/__init__.py +1 -0
- featurerank-0.1.2/setup.cfg +4 -0
- featurerank-0.1.2/src/AutoencoderFeatureSelection.py +161 -0
- featurerank-0.1.2/src/Classification.py +1374 -0
- featurerank-0.1.2/src/Clustering.py +475 -0
- featurerank-0.1.2/src/Config.py +102 -0
- featurerank-0.1.2/src/DataLoader.py +262 -0
- featurerank-0.1.2/src/DivideCombine.py +161 -0
- featurerank-0.1.2/src/Experiment.py +492 -0
- featurerank-0.1.2/src/Models.py +108 -0
- featurerank-0.1.2/src/OutputPaths.py +133 -0
- featurerank-0.1.2/src/Preprocessing.py +305 -0
- featurerank-0.1.2/src/Regression.py +583 -0
- featurerank-0.1.2/src/Reporting.py +204 -0
- featurerank-0.1.2/src/Runtime.py +46 -0
- featurerank-0.1.2/src/Utils.py +55 -0
- featurerank-0.1.2/src/Workflow.py +105 -0
- featurerank-0.1.2/src/__init__.py +5 -0
- featurerank-0.1.2/tests/test_feature_rank_refactor.py +104 -0
- featurerank-0.1.0/.gitignore +0 -63
- featurerank-0.1.0/CHANGELOG.md +0 -25
- featurerank-0.1.0/LICENSE +0 -21
- featurerank-0.1.0/PKG-INFO +0 -115
- featurerank-0.1.0/PUBLISHING.md +0 -54
- featurerank-0.1.0/README.md +0 -90
- featurerank-0.1.0/pyproject.toml +0 -42
- featurerank-0.1.0/src/FeatureRank/__init__.py +0 -3
- featurerank-0.1.0/src/FeatureRank/basic_operator.py +0 -16
- featurerank-0.1.0/test.py +0 -6
- featurerank-0.1.0/tests/test_basic_operator.py +0 -26
|
@@ -0,0 +1,589 @@
|
|
|
1
|
+
"""Small desktop interface for the existing FeatureRank workflows.
|
|
2
|
+
|
|
3
|
+
The window only collects user choices. Global and Divide & Combine work is
|
|
4
|
+
still performed by ``scripts.FeatureRank`` and the existing ``src`` modules.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import os
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
import queue
|
|
12
|
+
import subprocess
|
|
13
|
+
import sys
|
|
14
|
+
import threading
|
|
15
|
+
import time
|
|
16
|
+
import tkinter as tk
|
|
17
|
+
from tkinter import filedialog, messagebox, ttk
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
PERCENT_VALUES = tuple(str(value) for value in range(10, 101, 10))
|
|
21
|
+
BLOCK_VALUES = ("5", "10", "20")
|
|
22
|
+
MODE_LABELS = {
|
|
23
|
+
"GL": "GL — Global Feature Ranking",
|
|
24
|
+
"DC": "DC — Divide & Combine Feature Ranking",
|
|
25
|
+
}
|
|
26
|
+
TASK_LABELS = {
|
|
27
|
+
"Classification": "classification",
|
|
28
|
+
"Regression": "regression",
|
|
29
|
+
"Clustering": "clustering",
|
|
30
|
+
}
|
|
31
|
+
_WINDOW_OPENED = False
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _dataset_choices() -> list[str]:
|
|
35
|
+
"""Find the repository's paired raw datasets for the dropdown."""
|
|
36
|
+
project_root = Path(__file__).resolve().parent.parent
|
|
37
|
+
directories = (Path.cwd() / "data" / "raw", project_root / "data" / "raw")
|
|
38
|
+
choices: list[str] = []
|
|
39
|
+
for directory in directories:
|
|
40
|
+
if not directory.exists():
|
|
41
|
+
continue
|
|
42
|
+
for data_file in sorted(directory.glob("*_data.csv")):
|
|
43
|
+
if data_file.is_file() and data_file.name not in choices:
|
|
44
|
+
choices.append(data_file.name)
|
|
45
|
+
return choices or ["breast_cancer_data.csv"]
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _open_folder(path: Path) -> None:
|
|
49
|
+
"""Open a result directory using the current operating system."""
|
|
50
|
+
path.mkdir(parents=True, exist_ok=True)
|
|
51
|
+
if sys.platform == "darwin":
|
|
52
|
+
subprocess.Popen(["open", str(path)])
|
|
53
|
+
elif os.name == "nt":
|
|
54
|
+
os.startfile(str(path)) # type: ignore[attr-defined]
|
|
55
|
+
else:
|
|
56
|
+
subprocess.Popen(["xdg-open", str(path)])
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _tk_window_is_safe() -> bool:
|
|
60
|
+
"""Check Tk in a child process so a native Tk crash cannot kill the app."""
|
|
61
|
+
probe = "import tkinter as tk; " "root = tk.Tk(); " "root.withdraw(); " "root.destroy()"
|
|
62
|
+
try:
|
|
63
|
+
result = subprocess.run(
|
|
64
|
+
[sys.executable, "-c", probe],
|
|
65
|
+
stdin=subprocess.DEVNULL,
|
|
66
|
+
stdout=subprocess.DEVNULL,
|
|
67
|
+
stderr=subprocess.DEVNULL,
|
|
68
|
+
timeout=5,
|
|
69
|
+
check=False,
|
|
70
|
+
)
|
|
71
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
72
|
+
return False
|
|
73
|
+
return result.returncode == 0
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
class FeatureRankGUI:
|
|
77
|
+
"""Tkinter view and controller for one FeatureRank experiment."""
|
|
78
|
+
|
|
79
|
+
def __init__(self, root: tk.Tk) -> None:
|
|
80
|
+
self.root = root
|
|
81
|
+
self.root.title("FeatureRank")
|
|
82
|
+
self.root.geometry("780x720")
|
|
83
|
+
self.root.minsize(700, 620)
|
|
84
|
+
|
|
85
|
+
self.messages: queue.Queue[tuple[str, object]] = queue.Queue()
|
|
86
|
+
self.worker: threading.Thread | None = None
|
|
87
|
+
self.last_output_dir: Path | None = None
|
|
88
|
+
self.block_count = 10
|
|
89
|
+
self.blocks_seen = 0
|
|
90
|
+
|
|
91
|
+
self.dataset_var = tk.StringVar(value=_dataset_choices()[0])
|
|
92
|
+
self.percent_var = tk.StringVar(value="20")
|
|
93
|
+
self.task_var = tk.StringVar(value="Classification")
|
|
94
|
+
self.mode_var = tk.StringVar(value=MODE_LABELS["GL"])
|
|
95
|
+
self.block_var = tk.StringVar(value="10")
|
|
96
|
+
self.stage_var = tk.StringVar(value="Ready")
|
|
97
|
+
self.status_var = tk.StringVar(value="Select a dataset and start an experiment.")
|
|
98
|
+
self.summary_vars = {
|
|
99
|
+
key: tk.StringVar(value="N/A")
|
|
100
|
+
for key in (
|
|
101
|
+
"Dataset",
|
|
102
|
+
"Mode",
|
|
103
|
+
"Feature Percentage",
|
|
104
|
+
"Block Count",
|
|
105
|
+
"Original Feature Count",
|
|
106
|
+
"Selected Feature Count",
|
|
107
|
+
"Accuracy",
|
|
108
|
+
"RMSE",
|
|
109
|
+
"Cluster RMSE",
|
|
110
|
+
"Execution Time",
|
|
111
|
+
"Output Directory",
|
|
112
|
+
)
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
self._configure_style()
|
|
116
|
+
self._build_window()
|
|
117
|
+
|
|
118
|
+
def _configure_style(self) -> None:
|
|
119
|
+
style = ttk.Style(self.root)
|
|
120
|
+
try:
|
|
121
|
+
style.theme_use("clam")
|
|
122
|
+
except tk.TclError:
|
|
123
|
+
pass
|
|
124
|
+
style.configure("Title.TLabel", font=("Helvetica", 20, "bold"), foreground="#1f2937")
|
|
125
|
+
style.configure("Subtitle.TLabel", font=("Helvetica", 10), foreground="#4b5563")
|
|
126
|
+
style.configure("Stage.TLabel", font=("Helvetica", 11, "bold"), foreground="#1f4e79")
|
|
127
|
+
style.configure("Run.TButton", font=("Helvetica", 11, "bold"), padding=(18, 8))
|
|
128
|
+
|
|
129
|
+
def _build_window(self) -> None:
|
|
130
|
+
self.root.columnconfigure(0, weight=1)
|
|
131
|
+
self.root.rowconfigure(3, weight=1)
|
|
132
|
+
|
|
133
|
+
header = ttk.Frame(self.root, padding=(28, 24, 28, 12))
|
|
134
|
+
header.grid(row=0, column=0, sticky="ew")
|
|
135
|
+
ttk.Label(header, text="FeatureRank", style="Title.TLabel").pack(anchor="w")
|
|
136
|
+
ttk.Label(
|
|
137
|
+
header,
|
|
138
|
+
text="Autoencoder-based feature ranking and selection",
|
|
139
|
+
style="Subtitle.TLabel",
|
|
140
|
+
).pack(anchor="w", pady=(4, 0))
|
|
141
|
+
|
|
142
|
+
options = ttk.LabelFrame(self.root, text="Experiment", padding=18)
|
|
143
|
+
options.grid(row=1, column=0, padx=28, pady=8, sticky="ew")
|
|
144
|
+
options.columnconfigure(1, weight=1)
|
|
145
|
+
|
|
146
|
+
ttk.Label(options, text="Dataset").grid(row=0, column=0, sticky="w", pady=7)
|
|
147
|
+
self.dataset_combo = ttk.Combobox(
|
|
148
|
+
options, textvariable=self.dataset_var, values=_dataset_choices(), state="readonly"
|
|
149
|
+
)
|
|
150
|
+
self.dataset_combo.grid(row=0, column=1, sticky="ew", padx=(14, 8), pady=7)
|
|
151
|
+
ttk.Button(options, text="Browse…", command=self._browse_dataset).grid(
|
|
152
|
+
row=0, column=2, pady=7
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
ttk.Label(options, text="Task").grid(row=1, column=0, sticky="w", pady=7)
|
|
156
|
+
ttk.Combobox(
|
|
157
|
+
options,
|
|
158
|
+
textvariable=self.task_var,
|
|
159
|
+
values=tuple(TASK_LABELS),
|
|
160
|
+
state="readonly",
|
|
161
|
+
).grid(row=1, column=1, sticky="ew", padx=(14, 8), pady=7)
|
|
162
|
+
|
|
163
|
+
ttk.Label(options, text="Feature Percent").grid(row=2, column=0, sticky="w", pady=7)
|
|
164
|
+
ttk.Combobox(
|
|
165
|
+
options,
|
|
166
|
+
textvariable=self.percent_var,
|
|
167
|
+
values=PERCENT_VALUES,
|
|
168
|
+
state="readonly",
|
|
169
|
+
width=12,
|
|
170
|
+
).grid(row=2, column=1, sticky="w", padx=(14, 8), pady=7)
|
|
171
|
+
|
|
172
|
+
ttk.Label(options, text="Mode").grid(row=3, column=0, sticky="w", pady=7)
|
|
173
|
+
self.mode_combo = ttk.Combobox(
|
|
174
|
+
options,
|
|
175
|
+
textvariable=self.mode_var,
|
|
176
|
+
values=tuple(MODE_LABELS.values()),
|
|
177
|
+
state="readonly",
|
|
178
|
+
)
|
|
179
|
+
self.mode_combo.grid(row=3, column=1, sticky="ew", padx=(14, 8), pady=7)
|
|
180
|
+
self.mode_combo.bind("<<ComboboxSelected>>", self._mode_changed)
|
|
181
|
+
|
|
182
|
+
ttk.Label(options, text="Block Count").grid(row=4, column=0, sticky="w", pady=7)
|
|
183
|
+
self.block_combo = ttk.Combobox(
|
|
184
|
+
options, textvariable=self.block_var, values=BLOCK_VALUES, state="disabled", width=12
|
|
185
|
+
)
|
|
186
|
+
self.block_combo.grid(row=4, column=1, sticky="w", padx=(14, 8), pady=7)
|
|
187
|
+
|
|
188
|
+
action_row = ttk.Frame(self.root, padding=(28, 8, 28, 8))
|
|
189
|
+
action_row.grid(row=2, column=0, sticky="ew")
|
|
190
|
+
action_row.columnconfigure(1, weight=1)
|
|
191
|
+
self.start_button = ttk.Button(
|
|
192
|
+
action_row, text="START", style="Run.TButton", command=self._start
|
|
193
|
+
)
|
|
194
|
+
self.start_button.grid(row=0, column=0, sticky="w")
|
|
195
|
+
ttk.Label(action_row, textvariable=self.stage_var, style="Stage.TLabel").grid(
|
|
196
|
+
row=0, column=1, sticky="e", padx=(12, 0)
|
|
197
|
+
)
|
|
198
|
+
|
|
199
|
+
progress_frame = ttk.Frame(self.root, padding=(28, 0, 28, 12))
|
|
200
|
+
progress_frame.grid(row=3, column=0, sticky="nsew")
|
|
201
|
+
progress_frame.columnconfigure(0, weight=1)
|
|
202
|
+
progress_frame.rowconfigure(2, weight=1)
|
|
203
|
+
self.progress = ttk.Progressbar(progress_frame, mode="determinate", maximum=100)
|
|
204
|
+
self.progress.grid(row=0, column=0, sticky="ew")
|
|
205
|
+
ttk.Label(progress_frame, textvariable=self.status_var, style="Subtitle.TLabel").grid(
|
|
206
|
+
row=1, column=0, sticky="w", pady=(6, 8)
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
log_frame = ttk.LabelFrame(progress_frame, text="Log / Output", padding=8)
|
|
210
|
+
log_frame.grid(row=2, column=0, sticky="nsew")
|
|
211
|
+
log_frame.columnconfigure(0, weight=1)
|
|
212
|
+
log_frame.rowconfigure(0, weight=1)
|
|
213
|
+
self.log_text = tk.Text(
|
|
214
|
+
log_frame,
|
|
215
|
+
height=13,
|
|
216
|
+
wrap="word",
|
|
217
|
+
state="disabled",
|
|
218
|
+
background="#f8fafc",
|
|
219
|
+
foreground="#1f2937",
|
|
220
|
+
relief="flat",
|
|
221
|
+
padx=8,
|
|
222
|
+
pady=8,
|
|
223
|
+
)
|
|
224
|
+
self.log_text.grid(row=0, column=0, sticky="nsew")
|
|
225
|
+
scrollbar = ttk.Scrollbar(log_frame, orient="vertical", command=self.log_text.yview)
|
|
226
|
+
scrollbar.grid(row=0, column=1, sticky="ns")
|
|
227
|
+
self.log_text.configure(yscrollcommand=scrollbar.set)
|
|
228
|
+
|
|
229
|
+
summary = ttk.LabelFrame(self.root, text="Summary", padding=14)
|
|
230
|
+
summary.grid(row=4, column=0, padx=28, pady=(0, 10), sticky="ew")
|
|
231
|
+
summary.columnconfigure(1, weight=1)
|
|
232
|
+
for row, (key, value) in enumerate(self.summary_vars.items()):
|
|
233
|
+
ttk.Label(summary, text=key).grid(row=row // 2, column=(row % 2) * 2, sticky="w")
|
|
234
|
+
ttk.Label(summary, textvariable=value, foreground="#1f4e79").grid(
|
|
235
|
+
row=row // 2, column=(row % 2) * 2 + 1, sticky="w", padx=(8, 26), pady=2
|
|
236
|
+
)
|
|
237
|
+
|
|
238
|
+
self.open_button = ttk.Button(
|
|
239
|
+
self.root, text="Open Results Folder", command=self._open_results, state="disabled"
|
|
240
|
+
)
|
|
241
|
+
self.open_button.grid(row=5, column=0, padx=28, pady=(0, 18), sticky="w")
|
|
242
|
+
|
|
243
|
+
def _browse_dataset(self) -> None:
|
|
244
|
+
selected = filedialog.askopenfilename(
|
|
245
|
+
title="Select dataset",
|
|
246
|
+
filetypes=[("CSV files", "*.csv"), ("Text files", "*.txt"), ("All files", "*.*")],
|
|
247
|
+
)
|
|
248
|
+
if selected:
|
|
249
|
+
self.dataset_var.set(selected)
|
|
250
|
+
|
|
251
|
+
def _mode_changed(self, _event: object = None) -> None:
|
|
252
|
+
is_dc = self.mode_var.get().startswith("DC")
|
|
253
|
+
self.block_combo.configure(state="readonly" if is_dc else "disabled")
|
|
254
|
+
|
|
255
|
+
def _start(self) -> None:
|
|
256
|
+
dataset = self.dataset_var.get().strip()
|
|
257
|
+
if not dataset:
|
|
258
|
+
messagebox.showerror("FeatureRank", "Please select a dataset.")
|
|
259
|
+
return
|
|
260
|
+
try:
|
|
261
|
+
percent = float(self.percent_var.get())
|
|
262
|
+
if not 0 < percent <= 100:
|
|
263
|
+
raise ValueError
|
|
264
|
+
except ValueError:
|
|
265
|
+
messagebox.showerror("FeatureRank", "Feature Percent must be between 10 and 100.")
|
|
266
|
+
return
|
|
267
|
+
|
|
268
|
+
mode = "dc" if self.mode_var.get().startswith("DC") else "global"
|
|
269
|
+
task = TASK_LABELS[self.task_var.get()]
|
|
270
|
+
block_count = int(self.block_var.get()) if mode == "dc" else 10
|
|
271
|
+
self.block_count = block_count
|
|
272
|
+
self.blocks_seen = 0
|
|
273
|
+
self.last_output_dir = None
|
|
274
|
+
self.open_button.configure(state="disabled")
|
|
275
|
+
self.start_button.configure(state="disabled")
|
|
276
|
+
self.progress.configure(value=2)
|
|
277
|
+
self.stage_var.set("Loading Dataset")
|
|
278
|
+
self.status_var.set("The experiment is running. You can follow progress in the log.")
|
|
279
|
+
self._clear_log()
|
|
280
|
+
|
|
281
|
+
self.worker = threading.Thread(
|
|
282
|
+
target=self._run_experiment,
|
|
283
|
+
args=(dataset, percent, task, mode, block_count),
|
|
284
|
+
daemon=True,
|
|
285
|
+
)
|
|
286
|
+
self.worker.start()
|
|
287
|
+
self.root.after(100, self._poll_messages)
|
|
288
|
+
|
|
289
|
+
def _run_experiment(
|
|
290
|
+
self, dataset: str, percent: float, task: str, mode: str, block_count: int
|
|
291
|
+
) -> None:
|
|
292
|
+
started_at = time.perf_counter()
|
|
293
|
+
try:
|
|
294
|
+
# Keep TensorFlow out of the Tk process. On macOS, loading a model
|
|
295
|
+
# from a GUI thread can terminate the native Python process. The
|
|
296
|
+
# existing FeatureRank CLI runs in a separate worker instead; its
|
|
297
|
+
# stdout is streamed back into this window.
|
|
298
|
+
from src.Config import ExperimentConfig
|
|
299
|
+
|
|
300
|
+
config = ExperimentConfig(
|
|
301
|
+
dataset_name=dataset,
|
|
302
|
+
task=task,
|
|
303
|
+
feature_percent=percent,
|
|
304
|
+
random_state=42,
|
|
305
|
+
encoding_dim=8,
|
|
306
|
+
target_column="target",
|
|
307
|
+
id_column="ID",
|
|
308
|
+
cluster_k=None,
|
|
309
|
+
save_details=False,
|
|
310
|
+
)
|
|
311
|
+
python_executable = Path(sys.executable).with_name("python")
|
|
312
|
+
if not python_executable.exists():
|
|
313
|
+
python_executable = Path(sys.executable)
|
|
314
|
+
command = [
|
|
315
|
+
str(python_executable),
|
|
316
|
+
"-m",
|
|
317
|
+
"scripts.FeatureRank",
|
|
318
|
+
"--dataset-name",
|
|
319
|
+
dataset,
|
|
320
|
+
"--task",
|
|
321
|
+
task,
|
|
322
|
+
"--feature-percent",
|
|
323
|
+
str(percent),
|
|
324
|
+
"--random-state",
|
|
325
|
+
"42",
|
|
326
|
+
"--dc" if mode == "dc" else "--global",
|
|
327
|
+
]
|
|
328
|
+
if mode == "dc":
|
|
329
|
+
command.extend(["--block-count", str(block_count)])
|
|
330
|
+
|
|
331
|
+
process = subprocess.Popen(
|
|
332
|
+
command,
|
|
333
|
+
cwd=Path.cwd(),
|
|
334
|
+
stdout=subprocess.PIPE,
|
|
335
|
+
stderr=subprocess.STDOUT,
|
|
336
|
+
text=True,
|
|
337
|
+
bufsize=1,
|
|
338
|
+
)
|
|
339
|
+
assert process.stdout is not None
|
|
340
|
+
for line in process.stdout:
|
|
341
|
+
self.messages.put(("log", line))
|
|
342
|
+
return_code = process.wait()
|
|
343
|
+
if return_code != 0:
|
|
344
|
+
if return_code in {139, -11}:
|
|
345
|
+
raise RuntimeError(
|
|
346
|
+
"TensorFlow Anaconda ortaminda baslatilamadi (native crash). "
|
|
347
|
+
"Anaconda TensorFlow paketini guncelleyin veya uyumlu bir "
|
|
348
|
+
"Python yorumlayicisi kullanin."
|
|
349
|
+
)
|
|
350
|
+
raise RuntimeError(f"FeatureRank worker stopped with code {return_code}.")
|
|
351
|
+
|
|
352
|
+
average = 0.0
|
|
353
|
+
values: list[float] = []
|
|
354
|
+
summary = self._read_summary(
|
|
355
|
+
config,
|
|
356
|
+
mode,
|
|
357
|
+
block_count,
|
|
358
|
+
average,
|
|
359
|
+
values,
|
|
360
|
+
time.perf_counter() - started_at,
|
|
361
|
+
)
|
|
362
|
+
self.messages.put(("done", summary))
|
|
363
|
+
except Exception as error: # show a useful GUI message instead of a traceback window
|
|
364
|
+
self.messages.put(("error", str(error)))
|
|
365
|
+
|
|
366
|
+
def _read_summary(
|
|
367
|
+
self,
|
|
368
|
+
config: object,
|
|
369
|
+
mode: str,
|
|
370
|
+
block_count: int,
|
|
371
|
+
average: float,
|
|
372
|
+
values: list[float],
|
|
373
|
+
elapsed: float,
|
|
374
|
+
) -> dict[str, str]:
|
|
375
|
+
import json
|
|
376
|
+
|
|
377
|
+
from src.OutputPaths import format_feature_percent_tag, task_output_dir
|
|
378
|
+
|
|
379
|
+
dataset_name = str(config.dataset_name)
|
|
380
|
+
task = str(config.task)
|
|
381
|
+
tag = format_feature_percent_tag(float(config.feature_percent))
|
|
382
|
+
folder = Path(dataset_name).stem
|
|
383
|
+
if mode == "dc":
|
|
384
|
+
from scripts.FeatureBlockDatasetTools import dataset_base_name
|
|
385
|
+
|
|
386
|
+
folder = f"{dataset_base_name(dataset_name)}_selected_features_combined_data"
|
|
387
|
+
metrics_dir = task_output_dir(task, folder) / "metrics"
|
|
388
|
+
metric_filename = (
|
|
389
|
+
f"top_{tag}_cluster_metrics.json"
|
|
390
|
+
if task == "clustering"
|
|
391
|
+
else f"top_{tag}_test_metrics.json"
|
|
392
|
+
)
|
|
393
|
+
metric_path = metrics_dir / metric_filename
|
|
394
|
+
if not metric_path.exists():
|
|
395
|
+
candidates = sorted(
|
|
396
|
+
(
|
|
397
|
+
metrics_dir.parent.parent.rglob(f"top_{tag}*_metrics.json")
|
|
398
|
+
if metrics_dir.parent.parent.exists()
|
|
399
|
+
else []
|
|
400
|
+
),
|
|
401
|
+
key=lambda item: item.stat().st_mtime,
|
|
402
|
+
)
|
|
403
|
+
metric_path = candidates[-1] if candidates else metric_path
|
|
404
|
+
|
|
405
|
+
metrics: dict[str, object] = {}
|
|
406
|
+
if metric_path.exists():
|
|
407
|
+
metrics = json.loads(metric_path.read_text(encoding="utf-8"))
|
|
408
|
+
selected_count = metrics.get("selected_feature_count", "N/A")
|
|
409
|
+
original_count = metrics.get("original_feature_count", "N/A")
|
|
410
|
+
if original_count == "N/A":
|
|
411
|
+
original_count = self._count_features(config)
|
|
412
|
+
accuracy = metrics.get("test_accuracy", average) if task == "classification" else "N/A"
|
|
413
|
+
rmse = metrics.get("regression_rmse", average) if task == "regression" else "N/A"
|
|
414
|
+
cluster_rmse = metrics.get("cluster_rmse", average) if task == "clustering" else "N/A"
|
|
415
|
+
|
|
416
|
+
return {
|
|
417
|
+
"Dataset": Path(dataset_name).name,
|
|
418
|
+
"Mode": "DC" if mode == "dc" else "GL",
|
|
419
|
+
"Feature Percentage": f"{float(config.feature_percent):g}%",
|
|
420
|
+
"Block Count": str(block_count) if mode == "dc" else "N/A",
|
|
421
|
+
"Original Feature Count": str(original_count),
|
|
422
|
+
"Selected Feature Count": str(selected_count),
|
|
423
|
+
"Accuracy": f"{float(accuracy):.6f}" if accuracy != "N/A" else "N/A",
|
|
424
|
+
"RMSE": f"{float(rmse):.6f}" if rmse != "N/A" else "N/A",
|
|
425
|
+
"Cluster RMSE": (f"{float(cluster_rmse):.6f}" if cluster_rmse != "N/A" else "N/A"),
|
|
426
|
+
"Execution Time": f"{float(metrics.get('elapsed_seconds', elapsed)):.2f} s",
|
|
427
|
+
"Output Directory": str(metric_path.parent.parent) if metric_path.exists() else "N/A",
|
|
428
|
+
}
|
|
429
|
+
|
|
430
|
+
@staticmethod
|
|
431
|
+
def _count_features(config: object) -> str:
|
|
432
|
+
try:
|
|
433
|
+
from src.DataLoader import load_data
|
|
434
|
+
|
|
435
|
+
frame = load_data(config.dataset_name, target_column=config.target_column)
|
|
436
|
+
excluded = {config.target_column}
|
|
437
|
+
if config.id_column:
|
|
438
|
+
excluded.add(config.id_column)
|
|
439
|
+
return str(len([column for column in frame.columns if column not in excluded]))
|
|
440
|
+
except Exception:
|
|
441
|
+
return "N/A"
|
|
442
|
+
|
|
443
|
+
def _poll_messages(self) -> None:
|
|
444
|
+
while True:
|
|
445
|
+
try:
|
|
446
|
+
kind, payload = self.messages.get_nowait()
|
|
447
|
+
except queue.Empty:
|
|
448
|
+
break
|
|
449
|
+
if kind == "log":
|
|
450
|
+
self._handle_log(str(payload))
|
|
451
|
+
elif kind == "done":
|
|
452
|
+
self._completed(payload) # type: ignore[arg-type]
|
|
453
|
+
elif kind == "error":
|
|
454
|
+
self._failed(str(payload))
|
|
455
|
+
|
|
456
|
+
if self.worker and self.worker.is_alive():
|
|
457
|
+
self.root.after(100, self._poll_messages)
|
|
458
|
+
|
|
459
|
+
def _handle_log(self, text: str) -> None:
|
|
460
|
+
for line in text.splitlines():
|
|
461
|
+
line = line.strip()
|
|
462
|
+
if not line or self._is_debug_line(line):
|
|
463
|
+
continue
|
|
464
|
+
self._append_log(line)
|
|
465
|
+
self._update_progress(line)
|
|
466
|
+
|
|
467
|
+
@staticmethod
|
|
468
|
+
def _is_debug_line(line: str) -> bool:
|
|
469
|
+
lower = line.lower()
|
|
470
|
+
return any(
|
|
471
|
+
marker in lower
|
|
472
|
+
for marker in ("onednn", "cuda error", "cudnn", "cublas", "tensorflow/core")
|
|
473
|
+
)
|
|
474
|
+
|
|
475
|
+
def _update_progress(self, line: str) -> None:
|
|
476
|
+
lower = line.lower()
|
|
477
|
+
if "veri yukleniyor" in lower or "loading" in lower:
|
|
478
|
+
self._set_progress("Loading Dataset", 10)
|
|
479
|
+
elif "[dc] blok" in lower:
|
|
480
|
+
self.blocks_seen += 1
|
|
481
|
+
progress = 15 + int(50 * self.blocks_seen / max(self.block_count, 1))
|
|
482
|
+
self._set_progress(f"Processing Block {self.blocks_seen}/{self.block_count}", progress)
|
|
483
|
+
elif "autoencoder" in lower or "feature" in lower and "sec" in lower:
|
|
484
|
+
self._set_progress("Ranking Features", 55)
|
|
485
|
+
elif "birles" in lower or "combine" in lower:
|
|
486
|
+
self._set_progress("Combining Features", 75)
|
|
487
|
+
elif "egitim" in lower or "training" in lower or "classifier" in lower:
|
|
488
|
+
self._set_progress("Training Model", 85)
|
|
489
|
+
elif "metrik" in lower or "metric" in lower or "tamamlandi" in lower:
|
|
490
|
+
self._set_progress("Calculating Results", 95)
|
|
491
|
+
|
|
492
|
+
def _set_progress(self, stage: str, value: int) -> None:
|
|
493
|
+
self.stage_var.set(stage)
|
|
494
|
+
self.progress.configure(value=min(value, 99))
|
|
495
|
+
|
|
496
|
+
def _completed(self, summary: dict[str, str]) -> None:
|
|
497
|
+
for key, value in summary.items():
|
|
498
|
+
self.summary_vars[key].set(value)
|
|
499
|
+
output = summary.get("Output Directory", "N/A")
|
|
500
|
+
if output != "N/A":
|
|
501
|
+
self.last_output_dir = Path(output)
|
|
502
|
+
self.open_button.configure(state="normal")
|
|
503
|
+
self.progress.configure(value=100)
|
|
504
|
+
self.stage_var.set("Completed")
|
|
505
|
+
self.status_var.set("FeatureRank completed successfully.")
|
|
506
|
+
self.start_button.configure(state="normal")
|
|
507
|
+
|
|
508
|
+
def _failed(self, error: str) -> None:
|
|
509
|
+
self.progress.configure(value=0)
|
|
510
|
+
self.stage_var.set("Failed")
|
|
511
|
+
self.status_var.set("The experiment could not be completed.")
|
|
512
|
+
self._append_log(f"ERROR: {error}")
|
|
513
|
+
self.start_button.configure(state="normal")
|
|
514
|
+
messagebox.showerror("FeatureRank", error)
|
|
515
|
+
|
|
516
|
+
def _clear_log(self) -> None:
|
|
517
|
+
self.log_text.configure(state="normal")
|
|
518
|
+
self.log_text.delete("1.0", tk.END)
|
|
519
|
+
self.log_text.configure(state="disabled")
|
|
520
|
+
|
|
521
|
+
def _append_log(self, line: str) -> None:
|
|
522
|
+
self.log_text.configure(state="normal")
|
|
523
|
+
self.log_text.insert(tk.END, line + "\n")
|
|
524
|
+
self.log_text.see(tk.END)
|
|
525
|
+
self.log_text.configure(state="disabled")
|
|
526
|
+
|
|
527
|
+
def _open_results(self) -> None:
|
|
528
|
+
if self.last_output_dir is not None:
|
|
529
|
+
_open_folder(self.last_output_dir)
|
|
530
|
+
|
|
531
|
+
|
|
532
|
+
def Launch() -> FeatureRankGUI | None:
|
|
533
|
+
"""Open the FeatureRank desktop application.
|
|
534
|
+
|
|
535
|
+
``None`` is returned in headless environments so importing the package
|
|
536
|
+
does not crash notebooks, documentation builders, or automated tests.
|
|
537
|
+
"""
|
|
538
|
+
global _WINDOW_OPENED
|
|
539
|
+
if _WINDOW_OPENED:
|
|
540
|
+
return None
|
|
541
|
+
# Codex and other CI runners may expose Tkinter while lacking a window
|
|
542
|
+
# server. Calling ``Tk()`` on macOS in that state can abort the process,
|
|
543
|
+
# so leave imports safe in those runners.
|
|
544
|
+
if os.environ.get("CODEX_CI") == "1":
|
|
545
|
+
return None
|
|
546
|
+
|
|
547
|
+
# Anaconda's macOS interpreter can crash inside Cocoa/Tk when a window is
|
|
548
|
+
# created from an interactive ``python`` process. ``pythonw`` owns the
|
|
549
|
+
# application event loop, so launch the GUI there and keep the import
|
|
550
|
+
# process safe. The child sets a guard and enters this function directly.
|
|
551
|
+
if sys.platform == "darwin" and os.environ.get("FEATURERANK_GUI_CHILD") != "1":
|
|
552
|
+
pythonw = Path(sys.executable).with_name("pythonw")
|
|
553
|
+
if pythonw.exists() and _tk_window_is_safe():
|
|
554
|
+
child_environment = os.environ.copy()
|
|
555
|
+
child_environment["FEATURERANK_NO_AUTO_GUI"] = "1"
|
|
556
|
+
child_environment["FEATURERANK_GUI_CHILD"] = "1"
|
|
557
|
+
try:
|
|
558
|
+
subprocess.Popen(
|
|
559
|
+
[
|
|
560
|
+
str(pythonw),
|
|
561
|
+
"-c",
|
|
562
|
+
"from FeatureRank.GUI import Launch; Launch()",
|
|
563
|
+
],
|
|
564
|
+
env=child_environment,
|
|
565
|
+
start_new_session=True,
|
|
566
|
+
)
|
|
567
|
+
except OSError:
|
|
568
|
+
pass
|
|
569
|
+
else:
|
|
570
|
+
_WINDOW_OPENED = True
|
|
571
|
+
return None
|
|
572
|
+
|
|
573
|
+
if sys.platform not in {"darwin", "win32"} and not (
|
|
574
|
+
os.environ.get("DISPLAY") or os.environ.get("WAYLAND_DISPLAY")
|
|
575
|
+
):
|
|
576
|
+
return None
|
|
577
|
+
if not _tk_window_is_safe():
|
|
578
|
+
return None
|
|
579
|
+
try:
|
|
580
|
+
root = tk.Tk()
|
|
581
|
+
except tk.TclError:
|
|
582
|
+
return None
|
|
583
|
+
_WINDOW_OPENED = True
|
|
584
|
+
application = FeatureRankGUI(root)
|
|
585
|
+
root.mainloop()
|
|
586
|
+
return application
|
|
587
|
+
|
|
588
|
+
|
|
589
|
+
__all__ = ["FeatureRankGUI", "Launch"]
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""Public Python entry point for the FeatureRank project.
|
|
2
|
+
|
|
3
|
+
On a normal desktop import the GUI opens automatically. The explicit
|
|
4
|
+
``Launch`` function remains available for applications that want to control
|
|
5
|
+
when the window is shown.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
from .GUI import Launch
|
|
13
|
+
|
|
14
|
+
__version__ = "0.1.2"
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def _run_cli(arguments: list[str] | None = None) -> None:
|
|
18
|
+
"""Load the CLI only when a workflow is actually started."""
|
|
19
|
+
from scripts.FeatureRank import main as run_cli
|
|
20
|
+
|
|
21
|
+
run_cli(arguments)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def run(
|
|
25
|
+
dataset_name: str,
|
|
26
|
+
feature_percent: float,
|
|
27
|
+
task: str = "classification",
|
|
28
|
+
mode: str = "global",
|
|
29
|
+
block_count: int = 10,
|
|
30
|
+
random_state: int | None = 42,
|
|
31
|
+
) -> None:
|
|
32
|
+
"""Start one FeatureRank experiment from Python code.
|
|
33
|
+
|
|
34
|
+
``dataset_name`` and ``feature_percent`` are the only required values, so
|
|
35
|
+
a graphical interface can call this function directly. The remaining
|
|
36
|
+
arguments mirror the small set of choices exposed by the command line.
|
|
37
|
+
"""
|
|
38
|
+
mode_name = mode.lower().strip()
|
|
39
|
+
if mode_name not in {"global", "dc"}:
|
|
40
|
+
raise ValueError("mode 'global' veya 'dc' olmali.")
|
|
41
|
+
|
|
42
|
+
arguments = [
|
|
43
|
+
"--dataset-name",
|
|
44
|
+
str(dataset_name),
|
|
45
|
+
"--task",
|
|
46
|
+
str(task),
|
|
47
|
+
"--feature-percent",
|
|
48
|
+
str(feature_percent),
|
|
49
|
+
"--random-state",
|
|
50
|
+
"none" if random_state is None else str(random_state),
|
|
51
|
+
]
|
|
52
|
+
arguments.append("--dc" if mode_name == "dc" else "--global")
|
|
53
|
+
if mode_name == "dc":
|
|
54
|
+
arguments.extend(["--block-count", str(block_count)])
|
|
55
|
+
|
|
56
|
+
_run_cli(arguments)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def main() -> None:
|
|
60
|
+
"""Entry point used by the installed ``FeatureRank`` command."""
|
|
61
|
+
Launch()
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _should_auto_launch() -> bool:
|
|
65
|
+
"""Avoid surprise windows in tests and in ``python -m`` startup."""
|
|
66
|
+
if os.environ.get("FEATURERANK_NO_GUI", "").lower() in {"1", "true", "yes"}:
|
|
67
|
+
return False
|
|
68
|
+
if os.environ.get("FEATURERANK_NO_AUTO_GUI", "").lower() in {"1", "true", "yes"}:
|
|
69
|
+
return False
|
|
70
|
+
if "pytest" in sys.modules or os.environ.get("PYTEST_CURRENT_TEST"):
|
|
71
|
+
return False
|
|
72
|
+
command_name = Path(sys.argv[0]).name if sys.argv else ""
|
|
73
|
+
return command_name not in {"FeatureRank", "FeatureRank.exe", "__main__.py"}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
if _should_auto_launch():
|
|
77
|
+
Launch()
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
__all__ = ["Launch", "main", "run", "__version__"]
|