epokio 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- epokio/__init__.py +39 -0
- epokio/__main__.py +4 -0
- epokio/adapters.py +367 -0
- epokio/agent.py +530 -0
- epokio/analysis.py +161 -0
- epokio/auth.py +94 -0
- epokio/autostart.py +154 -0
- epokio/cli.py +59 -0
- epokio/config.py +59 -0
- epokio/diagnose.py +65 -0
- epokio/discover.py +69 -0
- epokio/envs.py +229 -0
- epokio/health.py +238 -0
- epokio/i18n.py +193 -0
- epokio/jobs.py +694 -0
- epokio/jsonfile.py +68 -0
- epokio/logger.py +249 -0
- epokio/mcp_server.py +209 -0
- epokio/monitor.py +94 -0
- epokio/msg.py +151 -0
- epokio/notify.py +77 -0
- epokio/onboard.py +378 -0
- epokio/report.py +141 -0
- epokio/rundetail.py +137 -0
- epokio/runmeta.py +137 -0
- epokio/scan.py +411 -0
- epokio/server.py +273 -0
- epokio/sources.py +28 -0
- epokio/sysinfo.py +272 -0
- epokio/textnorm.py +79 -0
- epokio/tfevents.py +199 -0
- epokio/tray.py +276 -0
- epokio/tui.py +283 -0
- epokio/versions.py +78 -0
- epokio/watcher.py +159 -0
- epokio/web/index.html +1344 -0
- epokio-0.3.0.dist-info/METADATA +73 -0
- epokio-0.3.0.dist-info/RECORD +42 -0
- epokio-0.3.0.dist-info/WHEEL +5 -0
- epokio-0.3.0.dist-info/entry_points.txt +6 -0
- epokio-0.3.0.dist-info/licenses/LICENSE +21 -0
- epokio-0.3.0.dist-info/top_level.txt +1 -0
epokio/__init__.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
"""Epokio agent. 예전 이름은 TrainBar였다."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
# 판 번호는 여기 한 곳(pyproject가 이것을 읽는다). ★설치 정보로 읽어, 맥 앱(PYTHONPATH로 소스를 돌린다)과
|
|
7
|
+
# exe에서는 'unknown'이거나 따로 깔린 다른 epokio의 판이 나왔다
|
|
8
|
+
__version__ = "0.3.0"
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _migrate_home() -> None:
|
|
12
|
+
"""옛 데이터 폴더 ~/.trainbar 를 ~/.epokio 로 한 번 옮긴다(새 폴더가 없을 때만).
|
|
13
|
+
기록 파일 안의 옛 경로도 새 경로로 고친다(대기열 결과 폴더, 알림함의 학습 위치)."""
|
|
14
|
+
old, new = Path.home() / ".trainbar", Path.home() / ".epokio"
|
|
15
|
+
if not old.is_dir() or new.exists():
|
|
16
|
+
return
|
|
17
|
+
try:
|
|
18
|
+
old.rename(new)
|
|
19
|
+
for f in new.glob("*.json"):
|
|
20
|
+
text = f.read_text(encoding="utf-8")
|
|
21
|
+
if "/.trainbar/" in text or "\\\\.trainbar\\\\" in text:
|
|
22
|
+
f.write_text(text.replace("/.trainbar/", "/.epokio/").replace("\\\\.trainbar\\\\", "\\\\.epokio\\\\"),
|
|
23
|
+
encoding="utf-8")
|
|
24
|
+
except OSError:
|
|
25
|
+
pass # 옮기지 못하면 새로 시작한다. 옛 폴더는 그대로 남는다
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
_migrate_home()
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def version() -> str:
|
|
32
|
+
"""이 코드의 판(맥 앱 판 build_app.sh와 같은 숫자로 맞춘다)"""
|
|
33
|
+
return __version__
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def start(folder, epochs=None, **params):
|
|
37
|
+
"""직접 짠 학습 코드용 기록기: `run = epokio.start("runs/exp", epochs=50, lr=1e-4)` 후 `run.log(val_loss=..., acc=...)`"""
|
|
38
|
+
from .logger import start as _start
|
|
39
|
+
return _start(folder, epochs, **params)
|
epokio/__main__.py
ADDED
epokio/adapters.py
ADDED
|
@@ -0,0 +1,367 @@
|
|
|
1
|
+
"""프레임워크별 학습 기록을 한 모양으로 바꾼다(어댑터).
|
|
2
|
+
|
|
3
|
+
Epokio의 나머지 부분(목록·상태·곡선·해설·비교)은 "에폭별 행" 하나만 본다:
|
|
4
|
+
[{"epoch": "1", "time": "12.3", "train/xxx_loss": "...", "val/xxx_loss": "...", "metrics/xxx": "..."}, ...]
|
|
5
|
+
값은 문자열이다(results.csv를 읽을 때와 같게). 열 이름 규칙:
|
|
6
|
+
손실 train/<이름>_loss · val/<이름>_loss 점수 metrics/<이름> (높을수록 좋은 것만)
|
|
7
|
+
|
|
8
|
+
새 프레임워크를 넣으려면 Adapter를 하나 더 만들어 ADAPTERS에 넣으면 된다.
|
|
9
|
+
★코드를 고치지 않는다는 원칙: 프레임워크가 원래 남기는 파일만 읽는다.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import csv
|
|
14
|
+
import json
|
|
15
|
+
import math
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import dataclass, field
|
|
18
|
+
from pathlib import Path
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Loaded:
|
|
23
|
+
framework: str
|
|
24
|
+
rows: list[dict]
|
|
25
|
+
source: Path # 갱신 시각을 잴 파일
|
|
26
|
+
total: int | None = None # 계획 에폭
|
|
27
|
+
args: dict = field(default_factory=dict)
|
|
28
|
+
epoch: int | None = None # 끝낸 에폭 수를 따로 아는 프레임워크(HF: 줄의 에폭은 평가 시점이라 다를 수 있다)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _num(v) -> str:
|
|
32
|
+
try:
|
|
33
|
+
x = float(v)
|
|
34
|
+
except (TypeError, ValueError):
|
|
35
|
+
return ""
|
|
36
|
+
if math.isnan(x):
|
|
37
|
+
return "nan"
|
|
38
|
+
return str(int(x)) if x.is_integer() and abs(x) < 1e15 else repr(x)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _read_csv(p: Path) -> list[dict]:
|
|
42
|
+
with p.open(encoding="utf-8", errors="ignore", newline="") as fh:
|
|
43
|
+
return [{(k or "").strip(): (v or "").strip() for k, v in r.items()} for r in csv.DictReader(fh)]
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def _yaml_value(text: str, key: str) -> str | None:
|
|
47
|
+
"""한 줄짜리 'key: value'만 읽는다(yaml 의존성 없이)"""
|
|
48
|
+
m = re.search(rf"^\s*{re.escape(key)}\s*:\s*([^#\n]+)", text, re.M)
|
|
49
|
+
return m.group(1).strip() if m else None
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
class Adapter:
|
|
53
|
+
name = ""
|
|
54
|
+
|
|
55
|
+
def detect(self, d: Path, names: set[str]) -> bool:
|
|
56
|
+
raise NotImplementedError
|
|
57
|
+
|
|
58
|
+
def load(self, d: Path) -> Loaded | None:
|
|
59
|
+
raise NotImplementedError
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
class Ultralytics(Adapter):
|
|
63
|
+
"""results.csv + args.yaml. 열 이름이 이미 규칙과 같다(train/box_loss, metrics/mAP50(B))."""
|
|
64
|
+
name = "ultralytics"
|
|
65
|
+
|
|
66
|
+
def detect(self, d, names):
|
|
67
|
+
return "results.csv" in names or "args.yaml" in names
|
|
68
|
+
|
|
69
|
+
def load(self, d):
|
|
70
|
+
p = d / "results.csv"
|
|
71
|
+
if not p.exists():
|
|
72
|
+
return None
|
|
73
|
+
rows = [r for r in _read_csv(p) if r.get("epoch")]
|
|
74
|
+
# epokio.start()로 기록한 직접 짠 학습은 'custom'. ★ultralytics로 보여서 "다시 학습" 등 YOLO 전용 기능이 붙었다
|
|
75
|
+
try:
|
|
76
|
+
custom = "task: custom" in (d / "args.yaml").read_text(encoding="utf-8", errors="ignore").splitlines()[:1]
|
|
77
|
+
except OSError:
|
|
78
|
+
custom = False
|
|
79
|
+
return Loaded("custom" if custom else self.name, rows, p)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class HuggingFace(Adapter):
|
|
83
|
+
"""transformers Trainer의 trainer_state.json (출력 폴더, 또는 가장 최근 checkpoint-N 안).
|
|
84
|
+
log_history에서 eval_* 기록이 나온 지점을 에폭 한 줄로 삼는다."""
|
|
85
|
+
name = "huggingface"
|
|
86
|
+
SKIP = {"eval_runtime", "eval_samples_per_second", "eval_steps_per_second", "epoch", "step", "eval_loss"}
|
|
87
|
+
# 점수가 아닌 것(시간·처리량·토큰 수). ★eval_model_preparation_time이 '대표 점수'로 뽑혔다
|
|
88
|
+
NOT_SCORE = re.compile(r"(_time|_runtime|_per_second|samples|num_tokens|^lr$|learning_rate)", re.IGNORECASE)
|
|
89
|
+
|
|
90
|
+
def _state(self, d: Path) -> Path | None:
|
|
91
|
+
if (d / "trainer_state.json").exists():
|
|
92
|
+
return d / "trainer_state.json"
|
|
93
|
+
cks = sorted(d.glob("checkpoint-*/trainer_state.json"), key=lambda p: int(re.sub(r"\D", "", p.parent.name) or 0))
|
|
94
|
+
return cks[-1] if cks else None
|
|
95
|
+
|
|
96
|
+
def detect(self, d, names):
|
|
97
|
+
return "trainer_state.json" in names or any(n.startswith("checkpoint-") for n in names if (d / n).is_dir()) \
|
|
98
|
+
and self._state(d) is not None
|
|
99
|
+
|
|
100
|
+
def load(self, d):
|
|
101
|
+
sp = self._state(d)
|
|
102
|
+
if not sp:
|
|
103
|
+
return None
|
|
104
|
+
try:
|
|
105
|
+
st = json.loads(sp.read_text(encoding="utf-8"))
|
|
106
|
+
except (OSError, ValueError):
|
|
107
|
+
return None
|
|
108
|
+
hist = st.get("log_history", [])
|
|
109
|
+
rows, last_train = [], None
|
|
110
|
+
for h in hist:
|
|
111
|
+
if "loss" in h:
|
|
112
|
+
last_train = h["loss"]
|
|
113
|
+
if "eval_loss" not in h and not any(k.startswith("eval_") for k in h):
|
|
114
|
+
continue
|
|
115
|
+
# 평가 시점의 에폭 그대로(0.25, 0.5…). ★올림(ceil)해서 첫 평가(0.1)에 1/1 '끝남'과 알림이 갔고,
|
|
116
|
+
# 한 에폭에 평가가 여러 번이면 마지막 것만 남아 곡선이 한 점이었다
|
|
117
|
+
ep = round(float(h.get("epoch") or len(rows) + 1), 3)
|
|
118
|
+
row = {"epoch": f"{ep:g}", "val/eval_loss": _num(h.get("eval_loss"))}
|
|
119
|
+
if last_train is not None:
|
|
120
|
+
row["train/train_loss"] = _num(last_train)
|
|
121
|
+
for k, v in h.items():
|
|
122
|
+
if k.startswith("eval_") and k not in self.SKIP and not self.NOT_SCORE.search(k[5:]) and isinstance(v, (int, float)):
|
|
123
|
+
row[metric_column(k[5:], True)] = _num(v)
|
|
124
|
+
if rows and rows[-1]["epoch"] == row["epoch"]:
|
|
125
|
+
rows[-1] = row # 한 에폭에 평가가 여러 번이면 마지막 것
|
|
126
|
+
else:
|
|
127
|
+
rows.append(row)
|
|
128
|
+
if not rows and last_train is not None: # 평가 없이 학습만: 손실만 보인다
|
|
129
|
+
rows = [{"epoch": f"{round(float(st.get('epoch') or 1), 3):g}", "train/train_loss": _num(last_train)}]
|
|
130
|
+
total = int(st["num_train_epochs"]) if st.get("num_train_epochs") else None
|
|
131
|
+
# 끝낸 에폭: 내림. 정해진 걸음(max_steps)을 다 갔으면 끝난 것
|
|
132
|
+
done = int(math.floor(float(st.get("epoch") or 0) + 1e-6))
|
|
133
|
+
if total and st.get("max_steps") and (st.get("global_step") or 0) >= st["max_steps"]:
|
|
134
|
+
done = max(done, total)
|
|
135
|
+
args = {k: str(st[k]) for k in ("num_train_epochs", "train_batch_size", "max_steps", "best_model_checkpoint") if st.get(k) is not None}
|
|
136
|
+
return Loaded(self.name, rows, sp, total, args, epoch=done)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
class Lightning(Adapter):
|
|
140
|
+
"""PyTorch Lightning CSVLogger: lightning_logs/version_N/metrics.csv (+ hparams.yaml).
|
|
141
|
+
학습·검증 값이 서로 다른 줄에 흩어져 있어 에폭별로 모은다."""
|
|
142
|
+
name = "lightning"
|
|
143
|
+
|
|
144
|
+
def detect(self, d, names):
|
|
145
|
+
return "metrics.csv" in names and ("hparams.yaml" in names or d.name.startswith("version_"))
|
|
146
|
+
|
|
147
|
+
def load(self, d):
|
|
148
|
+
p = d / "metrics.csv"
|
|
149
|
+
try:
|
|
150
|
+
raw = _read_csv(p)
|
|
151
|
+
except OSError:
|
|
152
|
+
return None
|
|
153
|
+
by: dict[int, dict] = {}
|
|
154
|
+
for r in raw:
|
|
155
|
+
try:
|
|
156
|
+
ep = int(float(r.get("epoch", "")))
|
|
157
|
+
except ValueError:
|
|
158
|
+
continue
|
|
159
|
+
row = by.setdefault(ep, {"epoch": str(ep + 1)}) # Lightning 에폭은 0부터
|
|
160
|
+
for k, v in r.items():
|
|
161
|
+
# lr-Adam(LearningRateMonitor)은 점수가 아니다. ★최고 학습률이 '최고 점수'로 보였다
|
|
162
|
+
if k in ("epoch", "step") or v == "" or k.lower().startswith(("lr-", "lr_")) or k.lower() in ("lr", "learning_rate"):
|
|
163
|
+
continue
|
|
164
|
+
row[self._name(k)] = v
|
|
165
|
+
rows = [by[k] for k in sorted(by)]
|
|
166
|
+
total = None
|
|
167
|
+
if (d / "hparams.yaml").exists():
|
|
168
|
+
t = (d / "hparams.yaml").read_text(encoding="utf-8", errors="ignore")
|
|
169
|
+
v = _yaml_value(t, "max_epochs") or _yaml_value(t, "epochs")
|
|
170
|
+
total = int(float(v)) if v and v.replace(".", "").isdigit() else None
|
|
171
|
+
return Loaded(self.name, rows, p, total)
|
|
172
|
+
|
|
173
|
+
@staticmethod
|
|
174
|
+
def _name(k: str) -> str:
|
|
175
|
+
k = k.replace("_epoch", "").replace("/", "_")
|
|
176
|
+
if "loss" in k:
|
|
177
|
+
side = "val" if k.startswith(("val", "valid")) else "train"
|
|
178
|
+
return f"{side}/{k if k.endswith('_loss') else k + '_loss'}"
|
|
179
|
+
return metric_column(k, k.startswith(("val", "valid")))
|
|
180
|
+
|
|
181
|
+
|
|
182
|
+
class Keras(Adapter):
|
|
183
|
+
"""Keras CSVLogger. 파일 이름은 사용자가 정하므로 흔한 이름만 본다(training.log, history.csv, *keras*.csv)."""
|
|
184
|
+
name = "keras"
|
|
185
|
+
FILES = ("training.log", "history.csv", "training.csv", "keras_log.csv")
|
|
186
|
+
|
|
187
|
+
def _file(self, d: Path, names: set[str]) -> Path | None:
|
|
188
|
+
for n in self.FILES:
|
|
189
|
+
if n in names:
|
|
190
|
+
return d / n
|
|
191
|
+
for n in names:
|
|
192
|
+
if n.endswith(".csv") and "keras" in n.lower():
|
|
193
|
+
return d / n
|
|
194
|
+
return None
|
|
195
|
+
|
|
196
|
+
def detect(self, d, names):
|
|
197
|
+
f = self._file(d, names)
|
|
198
|
+
if not f:
|
|
199
|
+
return False
|
|
200
|
+
try:
|
|
201
|
+
head = f.open(encoding="utf-8", errors="ignore").readline().lower()
|
|
202
|
+
except OSError:
|
|
203
|
+
return False
|
|
204
|
+
return head.startswith("epoch") and "loss" in head
|
|
205
|
+
|
|
206
|
+
def load(self, d):
|
|
207
|
+
f = self._file(d, {p.name for p in d.iterdir()})
|
|
208
|
+
if not f:
|
|
209
|
+
return None
|
|
210
|
+
rows = []
|
|
211
|
+
for r in _read_csv(f):
|
|
212
|
+
try:
|
|
213
|
+
ep = int(float(r["epoch"])) + 1 # Keras 에폭은 0부터
|
|
214
|
+
except (KeyError, ValueError):
|
|
215
|
+
continue
|
|
216
|
+
row = {"epoch": str(ep)}
|
|
217
|
+
for k, v in r.items():
|
|
218
|
+
if k == "epoch":
|
|
219
|
+
continue
|
|
220
|
+
if k in ("loss", "val_loss"):
|
|
221
|
+
row["val/val_loss" if k == "val_loss" else "train/train_loss"] = v
|
|
222
|
+
elif k.endswith("loss"):
|
|
223
|
+
row[("val/" if k.startswith("val_") else "train/") + k] = v
|
|
224
|
+
elif k in ("lr", "learning_rate"):
|
|
225
|
+
continue
|
|
226
|
+
else:
|
|
227
|
+
row[metric_column(k, k.startswith("val_"))] = v
|
|
228
|
+
rows.append(row)
|
|
229
|
+
return Loaded(self.name, rows, f)
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
class TensorBoard(Adapter):
|
|
233
|
+
"""TensorBoard 이벤트 파일. Lightning의 기본 기록기(lightning_logs/version_N)와 trainer_state.json을 남기지 않은
|
|
234
|
+
Hugging Face 학습(<출력>/runs/<날짜_기계>)이 이것만 남긴다. ★예전엔 이런 학습이 목록에 아예 안 보였다.
|
|
235
|
+
검증 값이 찍힌 걸음마다 한 줄을 만들고, 그때까지의 학습 값을 함께 싣는다."""
|
|
236
|
+
name = "tensorboard"
|
|
237
|
+
# 점수가 아닌 것. 에폭 태그 자체(epoch·train/epoch)는 따로 쓴다(★train_loss_epoch 같은 에폭 평균까지 빼면 안 된다)
|
|
238
|
+
NOT_SCORE = re.compile(r"(^|[/_])(lr|learning_rate|grad_norm|hp_metric)($|[/_-])|_time$|runtime|per_second|"
|
|
239
|
+
r"samples|num_tokens|flos", re.IGNORECASE)
|
|
240
|
+
|
|
241
|
+
@staticmethod
|
|
242
|
+
def _files(d: Path) -> list[Path]:
|
|
243
|
+
return sorted(p for p in d.iterdir() if p.name.startswith("events.out.tfevents."))
|
|
244
|
+
|
|
245
|
+
@classmethod
|
|
246
|
+
def _keras(cls, d: Path, names) -> bool:
|
|
247
|
+
"""Keras TensorBoard 콜백: <log_dir>/train 과 <log_dir>/validation 에 따로 쓴다.
|
|
248
|
+
★둘이 다른 학습 두 개로 보였고, 검증 손실이 train/epoch_loss로 들어갔다"""
|
|
249
|
+
try:
|
|
250
|
+
return {"train", "validation"} <= set(names) and all(cls._files(d / s) for s in ("train", "validation"))
|
|
251
|
+
except OSError:
|
|
252
|
+
return False
|
|
253
|
+
|
|
254
|
+
def detect(self, d, names):
|
|
255
|
+
if self._keras(d, names):
|
|
256
|
+
return True
|
|
257
|
+
if not any(n.startswith("events.out.tfevents.") for n in names):
|
|
258
|
+
return False
|
|
259
|
+
# HF 출력 폴더에 trainer_state.json이 있으면 그쪽(HuggingFace)이 읽는다. 같은 학습이 두 번 보이지 않게
|
|
260
|
+
out = d.parent.parent if d.parent.name == "runs" else None
|
|
261
|
+
return not (out and ((out / "trainer_state.json").exists() or any(out.glob("checkpoint-*/trainer_state.json"))))
|
|
262
|
+
|
|
263
|
+
@staticmethod
|
|
264
|
+
def _column(tag: str) -> str:
|
|
265
|
+
key = tag.replace("/", "_")
|
|
266
|
+
if key.lower().startswith("eval_"):
|
|
267
|
+
key = "val_" + key[5:]
|
|
268
|
+
key = re.sub(r"_epoch$", "", key)
|
|
269
|
+
val = key.lower().startswith(("val", "valid", "test"))
|
|
270
|
+
if "loss" in key.lower():
|
|
271
|
+
return f"{'val' if val else 'train'}/{key if key.endswith('_loss') else key + '_loss'}"
|
|
272
|
+
return metric_column(key, val)
|
|
273
|
+
|
|
274
|
+
def load(self, d):
|
|
275
|
+
from . import tfevents
|
|
276
|
+
keras = self._keras(d, {p.name for p in d.iterdir()})
|
|
277
|
+
# Keras는 폴더가 곧 쪽이다(train/validation). 태그 epoch_loss → train_loss·val_loss
|
|
278
|
+
parts = [(d / "train", "train_"), (d / "validation", "val_")] if keras else [(d, "")]
|
|
279
|
+
files = [(f, pre) for sub, pre in parts for f in self._files(sub)]
|
|
280
|
+
if not files:
|
|
281
|
+
return None
|
|
282
|
+
scalars: dict[int, dict[str, float]] = {}
|
|
283
|
+
walls: dict[int, float] = {}
|
|
284
|
+
for f, pre in files: # 다시 시작하면 파일이 하나 더 생긴다
|
|
285
|
+
sc, wl = tfevents.read(f)
|
|
286
|
+
for step, vals in sc.items():
|
|
287
|
+
if pre:
|
|
288
|
+
vals = {pre + re.sub(r"^epoch_", "", k): v for k, v in vals.items() if k.startswith("epoch_")}
|
|
289
|
+
scalars.setdefault(step, {}).update(vals)
|
|
290
|
+
for step, w in wl.items():
|
|
291
|
+
walls.setdefault(step, w)
|
|
292
|
+
if not scalars:
|
|
293
|
+
return None
|
|
294
|
+
tags = {t for v in scalars.values() for t in v}
|
|
295
|
+
hf = "train/epoch" in tags
|
|
296
|
+
etag = "train/epoch" if hf else "epoch" if "epoch" in tags else None
|
|
297
|
+
is_val = lambda t: t.lower().startswith(("val", "eval", "valid", "test"))
|
|
298
|
+
keep = {t: self._column(t) for t in tags
|
|
299
|
+
if t not in ("epoch", "train/epoch") and not self.NOT_SCORE.search(t.lower()) and not t.endswith("_step")}
|
|
300
|
+
steps = sorted(scalars)
|
|
301
|
+
marks = [s for s in steps if any(is_val(t) for t in scalars[s])] or steps
|
|
302
|
+
if len(marks) > 2000: # 평가가 없고 걸음이 아주 많으면 고르게 줄인다
|
|
303
|
+
k = len(marks) / 2000
|
|
304
|
+
marks = [marks[int(i * k)] for i in range(2000)] + [marks[-1]]
|
|
305
|
+
want, rows, latest, ep, t0, j = set(marks), [], {}, None, min(walls.values(), default=None), 0
|
|
306
|
+
for s in steps:
|
|
307
|
+
vals = scalars[s]
|
|
308
|
+
if etag in vals:
|
|
309
|
+
ep = vals[etag]
|
|
310
|
+
latest.update(vals)
|
|
311
|
+
if s not in want:
|
|
312
|
+
continue
|
|
313
|
+
j += 1
|
|
314
|
+
if keras:
|
|
315
|
+
e = s + 1 # Keras는 걸음이 곧 에폭(0부터)
|
|
316
|
+
else:
|
|
317
|
+
e = (round(ep, 3) if hf else int(ep) + 1) if ep is not None else j # Lightning 에폭은 0부터
|
|
318
|
+
row = {"epoch": f"{e:g}"}
|
|
319
|
+
if t0 is not None and s in walls:
|
|
320
|
+
row["time"] = f"{walls[s] - t0:.1f}"
|
|
321
|
+
for t, col in keep.items():
|
|
322
|
+
if t in latest:
|
|
323
|
+
row[col] = _num(latest[t])
|
|
324
|
+
if rows and rows[-1]["epoch"] == row["epoch"]:
|
|
325
|
+
rows[-1] = row
|
|
326
|
+
else:
|
|
327
|
+
rows.append(row)
|
|
328
|
+
total = None
|
|
329
|
+
if (d / "hparams.yaml").exists():
|
|
330
|
+
t = (d / "hparams.yaml").read_text(encoding="utf-8", errors="ignore")
|
|
331
|
+
v = _yaml_value(t, "max_epochs") or _yaml_value(t, "num_train_epochs") or _yaml_value(t, "epochs")
|
|
332
|
+
total = int(float(v)) if v and v.replace(".", "").isdigit() else None
|
|
333
|
+
done = int(math.floor(ep + 1e-6)) if hf and ep is not None else None
|
|
334
|
+
return Loaded(self.name, rows, max((f for f, _ in files), key=lambda f: f.stat().st_mtime), total, epoch=done)
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
ADAPTERS: list[Adapter] = [Ultralytics(), HuggingFace(), Lightning(), Keras(), TensorBoard()]
|
|
338
|
+
|
|
339
|
+
# 낮을수록 좋은 지표. metrics/ 는 '높을수록 좋은 것만'이라는 규칙이라 손실 쪽으로 보낸다.
|
|
340
|
+
# ★mae·rmse를 metrics/ 로 두었더니 최고점 고르기가 가장 나쁜 에폭을 'best'로 골랐다(Keras 회귀 모델)
|
|
341
|
+
# ★crossentropy 같은 손실을 metrics로 받은 Keras에서 가장 나쁜 에폭을 best로 골랐다
|
|
342
|
+
_LOWER = re.compile(r"(^|_)(mae|mse|rmse|msle|mape|error|err|perplexity|wer|cer|loss|crossentropy|hinge|kl|"
|
|
343
|
+
r"kld|divergence|poisson|logcosh|nll)($|_)", re.IGNORECASE)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def metric_column(key: str, val_side: bool) -> str:
|
|
347
|
+
return f"{'val' if val_side else 'train'}/{key}_loss" if _LOWER.search(key) else f"metrics/{key}"
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def detect(d: Path, names: set[str] | None = None) -> Adapter | None:
|
|
351
|
+
if names is None:
|
|
352
|
+
try:
|
|
353
|
+
names = {p.name for p in d.iterdir()}
|
|
354
|
+
except OSError:
|
|
355
|
+
return None
|
|
356
|
+
for a in ADAPTERS:
|
|
357
|
+
try:
|
|
358
|
+
if a.detect(d, names):
|
|
359
|
+
return a
|
|
360
|
+
except OSError:
|
|
361
|
+
continue
|
|
362
|
+
return None
|
|
363
|
+
|
|
364
|
+
|
|
365
|
+
def load(d: Path) -> Loaded | None:
|
|
366
|
+
a = detect(d)
|
|
367
|
+
return a.load(d) if a else None
|