@jenga-ai/agent 1.0.0 → 1.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +28 -10
- package/agents/scrum-master.md +75 -0
- package/mcp/router/embedder.js +1 -1
- package/mcp/training_runner/index.js +239 -0
- package/mcp/training_runner/package-lock.json +1065 -0
- package/mcp/training_runner/package.json +15 -0
- package/package.json +13 -11
- package/skills/close-story/SKILL.md +203 -0
- package/skills/close-story/scripts/check-story-closeable.sh +195 -0
- package/skills/close-story/scripts/compute-scope-divergence.sh +128 -0
- package/skills/close-story/scripts/extract-diff-stats.sh +48 -0
- package/skills/close-story/scripts/extract-task-diff-stats.sh +97 -0
- package/skills/close-story/scripts/update-task-frontmatter.sh +103 -0
- package/skills/commit/SKILL.md +18 -0
- package/skills/distribute/CONFIG_SCHEMA.md +90 -0
- package/skills/distribute/SKILL.md +173 -0
- package/skills/distribute/scripts/check-version.sh +74 -0
- package/skills/distribute/scripts/commit-version-bump.sh +108 -0
- package/skills/distribute/scripts/distribute-changes.sh +381 -0
- package/skills/do/SKILL.md +314 -0
- package/skills/do/assets/intent-vs-diff-prompt.md +69 -0
- package/skills/doc/assets/path-objectives.yaml +13 -0
- package/skills/init/SKILL.md +4 -3
- package/skills/init/assets/strategy_stub_template.md +38 -0
- package/skills/init/scripts/init.sh +6 -1
- package/skills/jenga/SKILL.md +51 -2
- package/skills/strategy/SKILL.md +312 -0
- package/templates/SCRUM_BOARD_SCHEMA.md +49 -0
- package/skills/train/SKILL.md +0 -116
- package/skills/train/assets/dashboard-templates/classifiers.html +0 -106
- package/skills/train/assets/dashboard-templates/nlp.html +0 -102
- package/skills/train/assets/dashboard-templates/transformers.html +0 -98
- package/skills/train/assets/results-parsers/__init__.py +0 -9
- package/skills/train/assets/results-parsers/classifiers.py +0 -84
- package/skills/train/assets/results-parsers/nlp.py +0 -88
- package/skills/train/assets/results-parsers/reporter.py +0 -154
- package/skills/train/assets/results-parsers/transformers.py +0 -120
- package/skills/train/train_cli.py +0 -786
|
@@ -1,102 +0,0 @@
|
|
|
1
|
-
<!DOCTYPE html>
|
|
2
|
-
<html lang="en">
|
|
3
|
-
<head>
|
|
4
|
-
<meta charset="UTF-8" />
|
|
5
|
-
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
-
<title>NLP Pipeline Dashboard — {{job_name}}</title>
|
|
7
|
-
<style>
|
|
8
|
-
*, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
|
|
9
|
-
body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
|
|
10
|
-
header { margin-bottom: 2rem; }
|
|
11
|
-
header h1 { font-size: 1.6rem; font-weight: 700; }
|
|
12
|
-
header p { color: #555; margin-top: .25rem; font-size: .9rem; }
|
|
13
|
-
.card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
|
|
14
|
-
.card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
|
|
15
|
-
.metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
|
|
16
|
-
.metric-box { background: #fff8e1; border-radius: 8px; padding: 1rem; text-align: center; }
|
|
17
|
-
.metric-box .value { font-size: 2rem; font-weight: 700; color: #e67700; }
|
|
18
|
-
.metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
|
|
19
|
-
.bar-chart { display: flex; flex-direction: column; gap: .6rem; }
|
|
20
|
-
.bar-row { display: flex; align-items: center; gap: .5rem; }
|
|
21
|
-
.bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
|
|
22
|
-
.bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
|
|
23
|
-
.bar-fill { height: 100%; border-radius: 4px; background: #e67700; transition: width .4s; }
|
|
24
|
-
.bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
|
|
25
|
-
table { width: 100%; border-collapse: collapse; }
|
|
26
|
-
th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
|
|
27
|
-
th { background: #f8f9fa; font-weight: 600; color: #555; }
|
|
28
|
-
footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
|
|
29
|
-
</style>
|
|
30
|
-
</head>
|
|
31
|
-
<body>
|
|
32
|
-
|
|
33
|
-
<header>
|
|
34
|
-
<h1>🔤 NLP Pipeline Dashboard</h1>
|
|
35
|
-
<p>Job: <strong>{{job_name}}</strong> | Model: <strong>{{model_name}}</strong> | Task: <strong>{{task}}</strong></p>
|
|
36
|
-
</header>
|
|
37
|
-
|
|
38
|
-
<div class="card">
|
|
39
|
-
<h2>Key Metrics</h2>
|
|
40
|
-
<div class="metrics-grid">
|
|
41
|
-
<div class="metric-box">
|
|
42
|
-
<div class="value">{{f1}}</div>
|
|
43
|
-
<div class="label">F1 Score</div>
|
|
44
|
-
</div>
|
|
45
|
-
<div class="metric-box">
|
|
46
|
-
<div class="value">{{precision}}</div>
|
|
47
|
-
<div class="label">Precision</div>
|
|
48
|
-
</div>
|
|
49
|
-
<div class="metric-box">
|
|
50
|
-
<div class="value">{{recall}}</div>
|
|
51
|
-
<div class="label">Recall</div>
|
|
52
|
-
</div>
|
|
53
|
-
<div class="metric-box">
|
|
54
|
-
<div class="value">{{iterations}}</div>
|
|
55
|
-
<div class="label">Iterations</div>
|
|
56
|
-
</div>
|
|
57
|
-
</div>
|
|
58
|
-
</div>
|
|
59
|
-
|
|
60
|
-
<div class="card">
|
|
61
|
-
<h2>Score Breakdown</h2>
|
|
62
|
-
<div class="bar-chart">
|
|
63
|
-
<div class="bar-row">
|
|
64
|
-
<span class="bar-label">F1 Score</span>
|
|
65
|
-
<div class="bar-track"><div class="bar-fill" style="width: calc({{f1}} * 100%)"></div></div>
|
|
66
|
-
<span class="bar-val">{{f1}}</span>
|
|
67
|
-
</div>
|
|
68
|
-
<div class="bar-row">
|
|
69
|
-
<span class="bar-label">Precision</span>
|
|
70
|
-
<div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
|
|
71
|
-
<span class="bar-val">{{precision}}</span>
|
|
72
|
-
</div>
|
|
73
|
-
<div class="bar-row">
|
|
74
|
-
<span class="bar-label">Recall</span>
|
|
75
|
-
<div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
|
|
76
|
-
<span class="bar-val">{{recall}}</span>
|
|
77
|
-
</div>
|
|
78
|
-
</div>
|
|
79
|
-
</div>
|
|
80
|
-
|
|
81
|
-
<div class="card">
|
|
82
|
-
<h2>Summary Table</h2>
|
|
83
|
-
<table>
|
|
84
|
-
<thead>
|
|
85
|
-
<tr><th>Metric</th><th>Value</th></tr>
|
|
86
|
-
</thead>
|
|
87
|
-
<tbody>
|
|
88
|
-
<tr><td>F1 Score</td><td>{{f1}}</td></tr>
|
|
89
|
-
<tr><td>Precision</td><td>{{precision}}</td></tr>
|
|
90
|
-
<tr><td>Recall</td><td>{{recall}}</td></tr>
|
|
91
|
-
<tr><td>Iterations</td><td>{{iterations}}</td></tr>
|
|
92
|
-
<tr><td>Model Name</td><td>{{model_name}}</td></tr>
|
|
93
|
-
<tr><td>Task</td><td>{{task}}</td></tr>
|
|
94
|
-
<tr><td>Job Name</td><td>{{job_name}}</td></tr>
|
|
95
|
-
</tbody>
|
|
96
|
-
</table>
|
|
97
|
-
</div>
|
|
98
|
-
|
|
99
|
-
<footer>Generated by JengaAgent /train skill | {{job_name}}</footer>
|
|
100
|
-
|
|
101
|
-
</body>
|
|
102
|
-
</html>
|
|
@@ -1,98 +0,0 @@
|
|
|
1
|
-
<!DOCTYPE html>
|
|
2
|
-
<html lang="en">
|
|
3
|
-
<head>
|
|
4
|
-
<meta charset="UTF-8" />
|
|
5
|
-
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
-
<title>Transformer Fine-Tuning Dashboard — {{job_name}}</title>
|
|
7
|
-
<style>
|
|
8
|
-
*, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
|
|
9
|
-
body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
|
|
10
|
-
header { margin-bottom: 2rem; }
|
|
11
|
-
header h1 { font-size: 1.6rem; font-weight: 700; }
|
|
12
|
-
header p { color: #555; margin-top: .25rem; font-size: .9rem; }
|
|
13
|
-
.card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
|
|
14
|
-
.card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
|
|
15
|
-
.metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); gap: 1rem; }
|
|
16
|
-
.metric-box { background: #f0fff4; border-radius: 8px; padding: 1rem; text-align: center; }
|
|
17
|
-
.metric-box .value { font-size: 2rem; font-weight: 700; color: #2f9e44; }
|
|
18
|
-
.metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
|
|
19
|
-
.loss-section { margin-top: 1rem; }
|
|
20
|
-
.loss-row { display: flex; justify-content: space-between; padding: .5rem 0; border-bottom: 1px solid #eee; font-size: .9rem; }
|
|
21
|
-
.loss-row .loss-label { color: #555; }
|
|
22
|
-
.loss-row .loss-val { font-weight: 600; color: #2f9e44; }
|
|
23
|
-
table { width: 100%; border-collapse: collapse; }
|
|
24
|
-
th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
|
|
25
|
-
th { background: #f8f9fa; font-weight: 600; color: #555; }
|
|
26
|
-
.perplexity-note { font-size: .8rem; color: #777; margin-top: .5rem; }
|
|
27
|
-
footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
|
|
28
|
-
</style>
|
|
29
|
-
</head>
|
|
30
|
-
<body>
|
|
31
|
-
|
|
32
|
-
<header>
|
|
33
|
-
<h1>🤗 Transformer Fine-Tuning Dashboard</h1>
|
|
34
|
-
<p>Job: <strong>{{job_name}}</strong> | Model: <strong>{{model_name}}</strong></p>
|
|
35
|
-
</header>
|
|
36
|
-
|
|
37
|
-
<div class="card">
|
|
38
|
-
<h2>Key Metrics</h2>
|
|
39
|
-
<div class="metrics-grid">
|
|
40
|
-
<div class="metric-box">
|
|
41
|
-
<div class="value">{{perplexity}}</div>
|
|
42
|
-
<div class="label">Perplexity</div>
|
|
43
|
-
</div>
|
|
44
|
-
<div class="metric-box">
|
|
45
|
-
<div class="value">{{eval_loss}}</div>
|
|
46
|
-
<div class="label">Eval Loss</div>
|
|
47
|
-
</div>
|
|
48
|
-
<div class="metric-box">
|
|
49
|
-
<div class="value">{{train_loss}}</div>
|
|
50
|
-
<div class="label">Train Loss</div>
|
|
51
|
-
</div>
|
|
52
|
-
<div class="metric-box">
|
|
53
|
-
<div class="value">{{epochs}}</div>
|
|
54
|
-
<div class="label">Epochs</div>
|
|
55
|
-
</div>
|
|
56
|
-
</div>
|
|
57
|
-
<p class="perplexity-note">ℹ️ Lower perplexity = better language model. Derived from eval_loss (e^loss).</p>
|
|
58
|
-
</div>
|
|
59
|
-
|
|
60
|
-
<div class="card">
|
|
61
|
-
<h2>Loss Breakdown</h2>
|
|
62
|
-
<div class="loss-section">
|
|
63
|
-
<div class="loss-row">
|
|
64
|
-
<span class="loss-label">Evaluation Loss</span>
|
|
65
|
-
<span class="loss-val">{{eval_loss}}</span>
|
|
66
|
-
</div>
|
|
67
|
-
<div class="loss-row">
|
|
68
|
-
<span class="loss-label">Training Loss</span>
|
|
69
|
-
<span class="loss-val">{{train_loss}}</span>
|
|
70
|
-
</div>
|
|
71
|
-
<div class="loss-row">
|
|
72
|
-
<span class="loss-label">Perplexity (e^eval_loss)</span>
|
|
73
|
-
<span class="loss-val">{{perplexity}}</span>
|
|
74
|
-
</div>
|
|
75
|
-
</div>
|
|
76
|
-
</div>
|
|
77
|
-
|
|
78
|
-
<div class="card">
|
|
79
|
-
<h2>Summary Table</h2>
|
|
80
|
-
<table>
|
|
81
|
-
<thead>
|
|
82
|
-
<tr><th>Metric</th><th>Value</th></tr>
|
|
83
|
-
</thead>
|
|
84
|
-
<tbody>
|
|
85
|
-
<tr><td>Perplexity</td><td>{{perplexity}}</td></tr>
|
|
86
|
-
<tr><td>Eval Loss</td><td>{{eval_loss}}</td></tr>
|
|
87
|
-
<tr><td>Train Loss</td><td>{{train_loss}}</td></tr>
|
|
88
|
-
<tr><td>Epochs</td><td>{{epochs}}</td></tr>
|
|
89
|
-
<tr><td>Model Name</td><td>{{model_name}}</td></tr>
|
|
90
|
-
<tr><td>Job Name</td><td>{{job_name}}</td></tr>
|
|
91
|
-
</tbody>
|
|
92
|
-
</table>
|
|
93
|
-
</div>
|
|
94
|
-
|
|
95
|
-
<footer>Generated by JengaAgent /train skill | {{job_name}}</footer>
|
|
96
|
-
|
|
97
|
-
</body>
|
|
98
|
-
</html>
|
|
@@ -1,9 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
results-parsers — ML training result extraction utilities.
|
|
3
|
-
|
|
4
|
-
Each submodule exposes: parse(job_dir: Path) -> dict
|
|
5
|
-
Each submodule extracts type-specific metrics from job output artifacts.
|
|
6
|
-
"""
|
|
7
|
-
from pathlib import Path
|
|
8
|
-
|
|
9
|
-
PARSERS = ["classifiers", "transformers", "nlp"]
|
|
@@ -1,84 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
classifiers.py — Results parser for sklearn-based classifier jobs.
|
|
3
|
-
|
|
4
|
-
Extracts: accuracy, precision, recall, f1_score, model_type.
|
|
5
|
-
Reads from results.json or scans training-results/ for JSON files.
|
|
6
|
-
"""
|
|
7
|
-
import json
|
|
8
|
-
from pathlib import Path
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
def parse(job_dir: Path) -> dict:
|
|
12
|
-
"""
|
|
13
|
-
Parse training results for a classifiers job.
|
|
14
|
-
|
|
15
|
-
Returns a dict with keys:
|
|
16
|
-
accuracy, precision, recall, f1_score, model_type
|
|
17
|
-
Returns empty dict if no results can be found; never raises.
|
|
18
|
-
"""
|
|
19
|
-
job_dir = Path(job_dir)
|
|
20
|
-
metrics = {}
|
|
21
|
-
|
|
22
|
-
# 1. Try results.json at root
|
|
23
|
-
results_json = job_dir / "results.json"
|
|
24
|
-
if results_json.exists():
|
|
25
|
-
try:
|
|
26
|
-
data = json.loads(results_json.read_text())
|
|
27
|
-
metrics = _extract_from_dict(data)
|
|
28
|
-
if metrics:
|
|
29
|
-
return metrics
|
|
30
|
-
except Exception:
|
|
31
|
-
pass
|
|
32
|
-
|
|
33
|
-
# 2. Scan training-results/ for any JSON files
|
|
34
|
-
training_results_dir = job_dir / "training-results"
|
|
35
|
-
if training_results_dir.exists():
|
|
36
|
-
for json_file in sorted(training_results_dir.rglob("*.json")):
|
|
37
|
-
try:
|
|
38
|
-
data = json.loads(json_file.read_text())
|
|
39
|
-
metrics = _extract_from_dict(data)
|
|
40
|
-
if metrics:
|
|
41
|
-
return metrics
|
|
42
|
-
except Exception:
|
|
43
|
-
continue
|
|
44
|
-
|
|
45
|
-
return metrics
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
def _extract_from_dict(data: dict) -> dict:
|
|
49
|
-
"""Extract classifier-relevant keys from a parsed dict."""
|
|
50
|
-
metrics = {}
|
|
51
|
-
if not isinstance(data, dict):
|
|
52
|
-
return metrics
|
|
53
|
-
|
|
54
|
-
# Common field names produced by sklearn classification_report / custom scripts
|
|
55
|
-
field_map = {
|
|
56
|
-
"accuracy": ["accuracy", "test_accuracy", "val_accuracy", "acc"],
|
|
57
|
-
"precision": ["precision", "weighted avg.precision", "macro avg.precision"],
|
|
58
|
-
"recall": ["recall", "weighted avg.recall", "macro avg.recall"],
|
|
59
|
-
"f1_score": ["f1_score", "f1", "weighted avg.f1-score", "macro avg.f1-score"],
|
|
60
|
-
"model_type": ["model_type", "model", "algorithm", "classifier"],
|
|
61
|
-
}
|
|
62
|
-
|
|
63
|
-
for target_key, candidates in field_map.items():
|
|
64
|
-
for candidate in candidates:
|
|
65
|
-
# Support dot-path lookup (e.g. "weighted avg.precision")
|
|
66
|
-
value = _deep_get(data, candidate)
|
|
67
|
-
if value is not None:
|
|
68
|
-
metrics[target_key] = value
|
|
69
|
-
break
|
|
70
|
-
|
|
71
|
-
return metrics
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
def _deep_get(data: dict, dotted_key: str):
|
|
75
|
-
"""Retrieve a value from a nested dict using a dot-separated key path."""
|
|
76
|
-
keys = dotted_key.split(".")
|
|
77
|
-
current = data
|
|
78
|
-
for k in keys:
|
|
79
|
-
if not isinstance(current, dict):
|
|
80
|
-
return None
|
|
81
|
-
current = current.get(k)
|
|
82
|
-
if current is None:
|
|
83
|
-
return None
|
|
84
|
-
return current
|
|
@@ -1,88 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
nlp.py — Results parser for spaCy / NLP pipeline training jobs.
|
|
3
|
-
|
|
4
|
-
Extracts: f1, precision, recall, model_name, task, iterations.
|
|
5
|
-
Reads from results.json or scans training-results/ for JSON files.
|
|
6
|
-
"""
|
|
7
|
-
import json
|
|
8
|
-
from pathlib import Path
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
def parse(job_dir: Path) -> dict:
|
|
12
|
-
"""
|
|
13
|
-
Parse training results for an nlp job.
|
|
14
|
-
|
|
15
|
-
Returns a dict with keys:
|
|
16
|
-
f1, precision, recall, model_name, task, iterations
|
|
17
|
-
Returns empty dict if no results can be found; never raises.
|
|
18
|
-
"""
|
|
19
|
-
job_dir = Path(job_dir)
|
|
20
|
-
metrics = {}
|
|
21
|
-
|
|
22
|
-
# 1. Try results.json at root
|
|
23
|
-
results_json = job_dir / "results.json"
|
|
24
|
-
if results_json.exists():
|
|
25
|
-
try:
|
|
26
|
-
data = json.loads(results_json.read_text())
|
|
27
|
-
metrics = _extract_from_dict(data)
|
|
28
|
-
if metrics:
|
|
29
|
-
return metrics
|
|
30
|
-
except Exception:
|
|
31
|
-
pass
|
|
32
|
-
|
|
33
|
-
# 2. Scan training-results/ for JSON files
|
|
34
|
-
training_results_dir = job_dir / "training-results"
|
|
35
|
-
if training_results_dir.exists():
|
|
36
|
-
for json_file in sorted(training_results_dir.rglob("*.json")):
|
|
37
|
-
try:
|
|
38
|
-
data = json.loads(json_file.read_text())
|
|
39
|
-
metrics = _extract_from_dict(data)
|
|
40
|
-
if metrics:
|
|
41
|
-
return metrics
|
|
42
|
-
except Exception:
|
|
43
|
-
continue
|
|
44
|
-
|
|
45
|
-
# 3. Try spaCy training output (scores.json or metrics.json)
|
|
46
|
-
for scores_file in sorted(job_dir.rglob("scores.json")) + sorted(job_dir.rglob("metrics.json")):
|
|
47
|
-
try:
|
|
48
|
-
data = json.loads(scores_file.read_text())
|
|
49
|
-
metrics = _extract_from_dict(data)
|
|
50
|
-
if metrics:
|
|
51
|
-
return metrics
|
|
52
|
-
except Exception:
|
|
53
|
-
continue
|
|
54
|
-
|
|
55
|
-
return metrics
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
def _extract_from_dict(data: dict) -> dict:
|
|
59
|
-
"""Extract NLP-relevant keys from a parsed dict."""
|
|
60
|
-
metrics = {}
|
|
61
|
-
if not isinstance(data, dict):
|
|
62
|
-
return metrics
|
|
63
|
-
|
|
64
|
-
field_map = {
|
|
65
|
-
"f1": ["f1", "f1_score", "ents_f", "token_f", "tag_f", "sents_f", "score"],
|
|
66
|
-
"precision": ["precision", "ents_p", "token_p", "tag_p"],
|
|
67
|
-
"recall": ["recall", "ents_r", "token_r", "tag_r"],
|
|
68
|
-
"model_name": ["model_name", "model", "base_model"],
|
|
69
|
-
"task": ["task", "pipeline_component", "component"],
|
|
70
|
-
"iterations": ["iterations", "n_iter", "steps", "batches_trained"],
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
for target_key, candidates in field_map.items():
|
|
74
|
-
for candidate in candidates:
|
|
75
|
-
value = data.get(candidate)
|
|
76
|
-
if value is None:
|
|
77
|
-
# Try nested under "scores" or "results" key
|
|
78
|
-
for wrapper in ("scores", "results", "metrics"):
|
|
79
|
-
nested = data.get(wrapper, {})
|
|
80
|
-
if isinstance(nested, dict):
|
|
81
|
-
value = nested.get(candidate)
|
|
82
|
-
if value is not None:
|
|
83
|
-
break
|
|
84
|
-
if value is not None:
|
|
85
|
-
metrics[target_key] = value
|
|
86
|
-
break
|
|
87
|
-
|
|
88
|
-
return metrics
|
|
@@ -1,154 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
reporter.py — Surface training results to terminal, summary.md, and results.json.
|
|
3
|
-
|
|
4
|
-
Usage:
|
|
5
|
-
from skills.train.assets.results_parsers.reporter import surface_results
|
|
6
|
-
surface_results(job_dir, job_type)
|
|
7
|
-
"""
|
|
8
|
-
import importlib
|
|
9
|
-
import json
|
|
10
|
-
import sys
|
|
11
|
-
from datetime import datetime, timezone
|
|
12
|
-
from pathlib import Path
|
|
13
|
-
|
|
14
|
-
_PARSER_PKG = Path(__file__).resolve().parent
|
|
15
|
-
sys.path.insert(0, str(_PARSER_PKG.parent.parent.parent)) # ensure skills/ is on path
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
def _load_parser(job_type: str):
|
|
19
|
-
"""Dynamically import the type-specific parser module."""
|
|
20
|
-
try:
|
|
21
|
-
spec_path = _PARSER_PKG / f"{job_type}.py"
|
|
22
|
-
import importlib.util
|
|
23
|
-
spec = importlib.util.spec_from_file_location(job_type, spec_path)
|
|
24
|
-
mod = importlib.util.module_from_spec(spec)
|
|
25
|
-
spec.loader.exec_module(mod)
|
|
26
|
-
return mod
|
|
27
|
-
except Exception as e:
|
|
28
|
-
return None
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
def _format_value(v) -> str:
|
|
32
|
-
if isinstance(v, float):
|
|
33
|
-
return f"{v:.4f}"
|
|
34
|
-
return str(v)
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
def _print_terminal_summary(job_dir: Path, job_type: str, metrics: dict):
|
|
38
|
-
"""Print a box-formatted summary to stdout."""
|
|
39
|
-
job_name = job_dir.name
|
|
40
|
-
width = 54
|
|
41
|
-
border = "─" * width
|
|
42
|
-
print(f"\n┌{border}┐")
|
|
43
|
-
print(f"│ 📊 Training Summary — {job_name:<{width - 25}}│")
|
|
44
|
-
print(f"│ Type: {job_type:<{width - 9}}│")
|
|
45
|
-
print(f"├{border}┤")
|
|
46
|
-
if metrics:
|
|
47
|
-
for k, v in metrics.items():
|
|
48
|
-
label = k.replace("_", " ").title()
|
|
49
|
-
value = _format_value(v)
|
|
50
|
-
line = f" {label}: {value}"
|
|
51
|
-
print(f"│{line:<{width + 1}}│")
|
|
52
|
-
else:
|
|
53
|
-
print(f"│ (no metrics found — check training-results/ or results.json) │")
|
|
54
|
-
print(f"└{border}┘\n")
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
def _write_summary_md(job_dir: Path, job_type: str, metrics: dict):
|
|
58
|
-
"""Write a Markdown summary file to the job directory."""
|
|
59
|
-
job_name = job_dir.name
|
|
60
|
-
timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
|
|
61
|
-
lines = [
|
|
62
|
-
f"# Training Summary — {job_name}",
|
|
63
|
-
f"",
|
|
64
|
-
f"**Type:** {job_type} ",
|
|
65
|
-
f"**Generated:** {timestamp} ",
|
|
66
|
-
f"",
|
|
67
|
-
f"## Metrics",
|
|
68
|
-
f"",
|
|
69
|
-
f"| Metric | Value |",
|
|
70
|
-
f"|--------|-------|",
|
|
71
|
-
]
|
|
72
|
-
if metrics:
|
|
73
|
-
for k, v in metrics.items():
|
|
74
|
-
label = k.replace("_", " ").title()
|
|
75
|
-
lines.append(f"| {label} | {_format_value(v)} |")
|
|
76
|
-
else:
|
|
77
|
-
lines.append("| — | No metrics found |")
|
|
78
|
-
|
|
79
|
-
summary_path = job_dir / "summary.md"
|
|
80
|
-
summary_path.write_text("\n".join(lines) + "\n")
|
|
81
|
-
return summary_path
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
def _write_results_json(job_dir: Path, job_type: str, metrics: dict):
|
|
85
|
-
"""Write or update results.json in the job directory."""
|
|
86
|
-
results_path = job_dir / "results.json"
|
|
87
|
-
|
|
88
|
-
# Merge with existing results.json if present
|
|
89
|
-
existing = {}
|
|
90
|
-
if results_path.exists():
|
|
91
|
-
try:
|
|
92
|
-
existing = json.loads(results_path.read_text())
|
|
93
|
-
except Exception:
|
|
94
|
-
pass
|
|
95
|
-
|
|
96
|
-
existing.update({
|
|
97
|
-
"job_name": job_dir.name,
|
|
98
|
-
"job_type": job_type,
|
|
99
|
-
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
100
|
-
"metrics": metrics,
|
|
101
|
-
})
|
|
102
|
-
|
|
103
|
-
results_path.write_text(json.dumps(existing, indent=2))
|
|
104
|
-
return results_path
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
def surface_results(job_dir, job_type: str):
|
|
108
|
-
"""
|
|
109
|
-
Parse training results for the given job and surface them:
|
|
110
|
-
1. Print terminal summary
|
|
111
|
-
2. Write summary.md to job directory
|
|
112
|
-
3. Write/update results.json in job directory
|
|
113
|
-
|
|
114
|
-
Args:
|
|
115
|
-
job_dir: Path (or str) to the job directory
|
|
116
|
-
job_type: One of 'classifiers', 'transformers', 'nlp'
|
|
117
|
-
"""
|
|
118
|
-
job_dir = Path(job_dir)
|
|
119
|
-
|
|
120
|
-
parser = _load_parser(job_type)
|
|
121
|
-
if parser is None:
|
|
122
|
-
print(f"⚠️ No parser found for type '{job_type}' — skipping result surfacing.")
|
|
123
|
-
return {}
|
|
124
|
-
|
|
125
|
-
try:
|
|
126
|
-
metrics = parser.parse(job_dir) or {}
|
|
127
|
-
except Exception as e:
|
|
128
|
-
print(f"⚠️ Parser error for '{job_type}': {e}")
|
|
129
|
-
metrics = {}
|
|
130
|
-
|
|
131
|
-
_print_terminal_summary(job_dir, job_type, metrics)
|
|
132
|
-
|
|
133
|
-
try:
|
|
134
|
-
summary_path = _write_summary_md(job_dir, job_type, metrics)
|
|
135
|
-
print(f"📄 summary.md written to {summary_path}")
|
|
136
|
-
except Exception as e:
|
|
137
|
-
print(f"⚠️ Could not write summary.md: {e}")
|
|
138
|
-
|
|
139
|
-
try:
|
|
140
|
-
results_path = _write_results_json(job_dir, job_type, metrics)
|
|
141
|
-
print(f"📄 results.json written to {results_path}")
|
|
142
|
-
except Exception as e:
|
|
143
|
-
print(f"⚠️ Could not write results.json: {e}")
|
|
144
|
-
|
|
145
|
-
return metrics
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
if __name__ == "__main__":
|
|
149
|
-
import argparse
|
|
150
|
-
p = argparse.ArgumentParser(description="Surface training results for a job directory.")
|
|
151
|
-
p.add_argument("job_dir", help="Path to the job directory")
|
|
152
|
-
p.add_argument("job_type", choices=["classifiers", "transformers", "nlp"])
|
|
153
|
-
ns = p.parse_args()
|
|
154
|
-
surface_results(ns.job_dir, ns.job_type)
|
|
@@ -1,120 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
transformers.py — Results parser for HuggingFace Transformers fine-tuning jobs.
|
|
3
|
-
|
|
4
|
-
Extracts: perplexity, eval_loss, train_loss, epochs, model_name.
|
|
5
|
-
Reads from results.json or scans training-results/ and trainer_state.json.
|
|
6
|
-
"""
|
|
7
|
-
import json
|
|
8
|
-
import math
|
|
9
|
-
from pathlib import Path
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
def parse(job_dir: Path) -> dict:
|
|
13
|
-
"""
|
|
14
|
-
Parse training results for a transformers job.
|
|
15
|
-
|
|
16
|
-
Returns a dict with keys:
|
|
17
|
-
perplexity, eval_loss, train_loss, epochs, model_name
|
|
18
|
-
Returns empty dict if no results can be found; never raises.
|
|
19
|
-
"""
|
|
20
|
-
job_dir = Path(job_dir)
|
|
21
|
-
metrics = {}
|
|
22
|
-
|
|
23
|
-
# 1. Try results.json at root
|
|
24
|
-
results_json = job_dir / "results.json"
|
|
25
|
-
if results_json.exists():
|
|
26
|
-
try:
|
|
27
|
-
data = json.loads(results_json.read_text())
|
|
28
|
-
metrics = _extract_from_dict(data)
|
|
29
|
-
if metrics:
|
|
30
|
-
return metrics
|
|
31
|
-
except Exception:
|
|
32
|
-
pass
|
|
33
|
-
|
|
34
|
-
# 2. Try trainer_state.json (HuggingFace Trainer output)
|
|
35
|
-
for trainer_state in sorted(job_dir.rglob("trainer_state.json")):
|
|
36
|
-
try:
|
|
37
|
-
data = json.loads(trainer_state.read_text())
|
|
38
|
-
metrics = _extract_from_trainer_state(data)
|
|
39
|
-
if metrics:
|
|
40
|
-
return metrics
|
|
41
|
-
except Exception:
|
|
42
|
-
continue
|
|
43
|
-
|
|
44
|
-
# 3. Scan training-results/ for JSON files
|
|
45
|
-
training_results_dir = job_dir / "training-results"
|
|
46
|
-
if training_results_dir.exists():
|
|
47
|
-
for json_file in sorted(training_results_dir.rglob("*.json")):
|
|
48
|
-
try:
|
|
49
|
-
data = json.loads(json_file.read_text())
|
|
50
|
-
metrics = _extract_from_dict(data)
|
|
51
|
-
if metrics:
|
|
52
|
-
return metrics
|
|
53
|
-
except Exception:
|
|
54
|
-
continue
|
|
55
|
-
|
|
56
|
-
return metrics
|
|
57
|
-
|
|
58
|
-
|
|
59
|
-
def _extract_from_dict(data: dict) -> dict:
|
|
60
|
-
"""Extract transformer-relevant keys from a parsed dict."""
|
|
61
|
-
metrics = {}
|
|
62
|
-
if not isinstance(data, dict):
|
|
63
|
-
return metrics
|
|
64
|
-
|
|
65
|
-
field_map = {
|
|
66
|
-
"eval_loss": ["eval_loss", "validation_loss", "val_loss"],
|
|
67
|
-
"train_loss": ["train_loss", "training_loss", "loss"],
|
|
68
|
-
"epochs": ["epochs", "num_train_epochs", "epoch"],
|
|
69
|
-
"model_name": ["model_name", "model", "base_model", "pretrained_model_name_or_path"],
|
|
70
|
-
"perplexity": ["perplexity", "eval_perplexity"],
|
|
71
|
-
}
|
|
72
|
-
|
|
73
|
-
for target_key, candidates in field_map.items():
|
|
74
|
-
for candidate in candidates:
|
|
75
|
-
value = data.get(candidate)
|
|
76
|
-
if value is not None:
|
|
77
|
-
metrics[target_key] = value
|
|
78
|
-
break
|
|
79
|
-
|
|
80
|
-
# Derive perplexity from eval_loss if not already present
|
|
81
|
-
if "perplexity" not in metrics and "eval_loss" in metrics:
|
|
82
|
-
try:
|
|
83
|
-
metrics["perplexity"] = round(math.exp(float(metrics["eval_loss"])), 4)
|
|
84
|
-
except (ValueError, OverflowError):
|
|
85
|
-
pass
|
|
86
|
-
|
|
87
|
-
return metrics
|
|
88
|
-
|
|
89
|
-
|
|
90
|
-
def _extract_from_trainer_state(data: dict) -> dict:
|
|
91
|
-
"""Extract metrics from HuggingFace Trainer's trainer_state.json."""
|
|
92
|
-
metrics = {}
|
|
93
|
-
if not isinstance(data, dict):
|
|
94
|
-
return metrics
|
|
95
|
-
|
|
96
|
-
# Best metrics
|
|
97
|
-
best_metric = data.get("best_metric")
|
|
98
|
-
if best_metric is not None:
|
|
99
|
-
metrics["eval_loss"] = best_metric
|
|
100
|
-
|
|
101
|
-
# Epoch count
|
|
102
|
-
epoch = data.get("epoch")
|
|
103
|
-
if epoch is not None:
|
|
104
|
-
metrics["epochs"] = epoch
|
|
105
|
-
|
|
106
|
-
# Last eval from log history
|
|
107
|
-
log_history = data.get("log_history", [])
|
|
108
|
-
for entry in reversed(log_history):
|
|
109
|
-
if "eval_loss" in entry:
|
|
110
|
-
metrics["eval_loss"] = entry["eval_loss"]
|
|
111
|
-
break
|
|
112
|
-
|
|
113
|
-
# Derive perplexity
|
|
114
|
-
if "perplexity" not in metrics and "eval_loss" in metrics:
|
|
115
|
-
try:
|
|
116
|
-
metrics["perplexity"] = round(math.exp(float(metrics["eval_loss"])), 4)
|
|
117
|
-
except (ValueError, OverflowError):
|
|
118
|
-
pass
|
|
119
|
-
|
|
120
|
-
return metrics
|