@jenga-ai/agent 1.0.0 → 1.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/README.md +28 -10
  2. package/agents/scrum-master.md +75 -0
  3. package/mcp/router/embedder.js +1 -1
  4. package/mcp/training_runner/index.js +239 -0
  5. package/mcp/training_runner/package-lock.json +1065 -0
  6. package/mcp/training_runner/package.json +15 -0
  7. package/package.json +13 -11
  8. package/skills/close-story/SKILL.md +203 -0
  9. package/skills/close-story/scripts/check-story-closeable.sh +195 -0
  10. package/skills/close-story/scripts/compute-scope-divergence.sh +128 -0
  11. package/skills/close-story/scripts/extract-diff-stats.sh +48 -0
  12. package/skills/close-story/scripts/extract-task-diff-stats.sh +97 -0
  13. package/skills/close-story/scripts/update-task-frontmatter.sh +103 -0
  14. package/skills/commit/SKILL.md +18 -0
  15. package/skills/distribute/CONFIG_SCHEMA.md +90 -0
  16. package/skills/distribute/SKILL.md +173 -0
  17. package/skills/distribute/scripts/check-version.sh +74 -0
  18. package/skills/distribute/scripts/commit-version-bump.sh +108 -0
  19. package/skills/distribute/scripts/distribute-changes.sh +381 -0
  20. package/skills/do/SKILL.md +314 -0
  21. package/skills/do/assets/intent-vs-diff-prompt.md +69 -0
  22. package/skills/doc/assets/path-objectives.yaml +13 -0
  23. package/skills/init/SKILL.md +4 -3
  24. package/skills/init/assets/strategy_stub_template.md +38 -0
  25. package/skills/init/scripts/init.sh +6 -1
  26. package/skills/jenga/SKILL.md +51 -2
  27. package/skills/strategy/SKILL.md +312 -0
  28. package/templates/SCRUM_BOARD_SCHEMA.md +49 -0
  29. package/skills/train/SKILL.md +0 -116
  30. package/skills/train/assets/dashboard-templates/classifiers.html +0 -106
  31. package/skills/train/assets/dashboard-templates/nlp.html +0 -102
  32. package/skills/train/assets/dashboard-templates/transformers.html +0 -98
  33. package/skills/train/assets/results-parsers/__init__.py +0 -9
  34. package/skills/train/assets/results-parsers/classifiers.py +0 -84
  35. package/skills/train/assets/results-parsers/nlp.py +0 -88
  36. package/skills/train/assets/results-parsers/reporter.py +0 -154
  37. package/skills/train/assets/results-parsers/transformers.py +0 -120
  38. package/skills/train/train_cli.py +0 -786
@@ -1,102 +0,0 @@
1
- <!DOCTYPE html>
2
- <html lang="en">
3
- <head>
4
- <meta charset="UTF-8" />
5
- <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
- <title>NLP Pipeline Dashboard — {{job_name}}</title>
7
- <style>
8
- *, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
9
- body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
10
- header { margin-bottom: 2rem; }
11
- header h1 { font-size: 1.6rem; font-weight: 700; }
12
- header p { color: #555; margin-top: .25rem; font-size: .9rem; }
13
- .card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
14
- .card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
15
- .metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
16
- .metric-box { background: #fff8e1; border-radius: 8px; padding: 1rem; text-align: center; }
17
- .metric-box .value { font-size: 2rem; font-weight: 700; color: #e67700; }
18
- .metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
19
- .bar-chart { display: flex; flex-direction: column; gap: .6rem; }
20
- .bar-row { display: flex; align-items: center; gap: .5rem; }
21
- .bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
22
- .bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
23
- .bar-fill { height: 100%; border-radius: 4px; background: #e67700; transition: width .4s; }
24
- .bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
25
- table { width: 100%; border-collapse: collapse; }
26
- th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
27
- th { background: #f8f9fa; font-weight: 600; color: #555; }
28
- footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
29
- </style>
30
- </head>
31
- <body>
32
-
33
- <header>
34
- <h1>🔤 NLP Pipeline Dashboard</h1>
35
- <p>Job: <strong>{{job_name}}</strong> &nbsp;|&nbsp; Model: <strong>{{model_name}}</strong> &nbsp;|&nbsp; Task: <strong>{{task}}</strong></p>
36
- </header>
37
-
38
- <div class="card">
39
- <h2>Key Metrics</h2>
40
- <div class="metrics-grid">
41
- <div class="metric-box">
42
- <div class="value">{{f1}}</div>
43
- <div class="label">F1 Score</div>
44
- </div>
45
- <div class="metric-box">
46
- <div class="value">{{precision}}</div>
47
- <div class="label">Precision</div>
48
- </div>
49
- <div class="metric-box">
50
- <div class="value">{{recall}}</div>
51
- <div class="label">Recall</div>
52
- </div>
53
- <div class="metric-box">
54
- <div class="value">{{iterations}}</div>
55
- <div class="label">Iterations</div>
56
- </div>
57
- </div>
58
- </div>
59
-
60
- <div class="card">
61
- <h2>Score Breakdown</h2>
62
- <div class="bar-chart">
63
- <div class="bar-row">
64
- <span class="bar-label">F1 Score</span>
65
- <div class="bar-track"><div class="bar-fill" style="width: calc({{f1}} * 100%)"></div></div>
66
- <span class="bar-val">{{f1}}</span>
67
- </div>
68
- <div class="bar-row">
69
- <span class="bar-label">Precision</span>
70
- <div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
71
- <span class="bar-val">{{precision}}</span>
72
- </div>
73
- <div class="bar-row">
74
- <span class="bar-label">Recall</span>
75
- <div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
76
- <span class="bar-val">{{recall}}</span>
77
- </div>
78
- </div>
79
- </div>
80
-
81
- <div class="card">
82
- <h2>Summary Table</h2>
83
- <table>
84
- <thead>
85
- <tr><th>Metric</th><th>Value</th></tr>
86
- </thead>
87
- <tbody>
88
- <tr><td>F1 Score</td><td>{{f1}}</td></tr>
89
- <tr><td>Precision</td><td>{{precision}}</td></tr>
90
- <tr><td>Recall</td><td>{{recall}}</td></tr>
91
- <tr><td>Iterations</td><td>{{iterations}}</td></tr>
92
- <tr><td>Model Name</td><td>{{model_name}}</td></tr>
93
- <tr><td>Task</td><td>{{task}}</td></tr>
94
- <tr><td>Job Name</td><td>{{job_name}}</td></tr>
95
- </tbody>
96
- </table>
97
- </div>
98
-
99
- <footer>Generated by JengaAgent /train skill &nbsp;|&nbsp; {{job_name}}</footer>
100
-
101
- </body>
102
- </html>
@@ -1,98 +0,0 @@
1
- <!DOCTYPE html>
2
- <html lang="en">
3
- <head>
4
- <meta charset="UTF-8" />
5
- <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
- <title>Transformer Fine-Tuning Dashboard — {{job_name}}</title>
7
- <style>
8
- *, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
9
- body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
10
- header { margin-bottom: 2rem; }
11
- header h1 { font-size: 1.6rem; font-weight: 700; }
12
- header p { color: #555; margin-top: .25rem; font-size: .9rem; }
13
- .card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
14
- .card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
15
- .metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); gap: 1rem; }
16
- .metric-box { background: #f0fff4; border-radius: 8px; padding: 1rem; text-align: center; }
17
- .metric-box .value { font-size: 2rem; font-weight: 700; color: #2f9e44; }
18
- .metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
19
- .loss-section { margin-top: 1rem; }
20
- .loss-row { display: flex; justify-content: space-between; padding: .5rem 0; border-bottom: 1px solid #eee; font-size: .9rem; }
21
- .loss-row .loss-label { color: #555; }
22
- .loss-row .loss-val { font-weight: 600; color: #2f9e44; }
23
- table { width: 100%; border-collapse: collapse; }
24
- th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
25
- th { background: #f8f9fa; font-weight: 600; color: #555; }
26
- .perplexity-note { font-size: .8rem; color: #777; margin-top: .5rem; }
27
- footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
28
- </style>
29
- </head>
30
- <body>
31
-
32
- <header>
33
- <h1>🤗 Transformer Fine-Tuning Dashboard</h1>
34
- <p>Job: <strong>{{job_name}}</strong> &nbsp;|&nbsp; Model: <strong>{{model_name}}</strong></p>
35
- </header>
36
-
37
- <div class="card">
38
- <h2>Key Metrics</h2>
39
- <div class="metrics-grid">
40
- <div class="metric-box">
41
- <div class="value">{{perplexity}}</div>
42
- <div class="label">Perplexity</div>
43
- </div>
44
- <div class="metric-box">
45
- <div class="value">{{eval_loss}}</div>
46
- <div class="label">Eval Loss</div>
47
- </div>
48
- <div class="metric-box">
49
- <div class="value">{{train_loss}}</div>
50
- <div class="label">Train Loss</div>
51
- </div>
52
- <div class="metric-box">
53
- <div class="value">{{epochs}}</div>
54
- <div class="label">Epochs</div>
55
- </div>
56
- </div>
57
- <p class="perplexity-note">ℹ️ Lower perplexity = better language model. Derived from eval_loss (e^loss).</p>
58
- </div>
59
-
60
- <div class="card">
61
- <h2>Loss Breakdown</h2>
62
- <div class="loss-section">
63
- <div class="loss-row">
64
- <span class="loss-label">Evaluation Loss</span>
65
- <span class="loss-val">{{eval_loss}}</span>
66
- </div>
67
- <div class="loss-row">
68
- <span class="loss-label">Training Loss</span>
69
- <span class="loss-val">{{train_loss}}</span>
70
- </div>
71
- <div class="loss-row">
72
- <span class="loss-label">Perplexity (e^eval_loss)</span>
73
- <span class="loss-val">{{perplexity}}</span>
74
- </div>
75
- </div>
76
- </div>
77
-
78
- <div class="card">
79
- <h2>Summary Table</h2>
80
- <table>
81
- <thead>
82
- <tr><th>Metric</th><th>Value</th></tr>
83
- </thead>
84
- <tbody>
85
- <tr><td>Perplexity</td><td>{{perplexity}}</td></tr>
86
- <tr><td>Eval Loss</td><td>{{eval_loss}}</td></tr>
87
- <tr><td>Train Loss</td><td>{{train_loss}}</td></tr>
88
- <tr><td>Epochs</td><td>{{epochs}}</td></tr>
89
- <tr><td>Model Name</td><td>{{model_name}}</td></tr>
90
- <tr><td>Job Name</td><td>{{job_name}}</td></tr>
91
- </tbody>
92
- </table>
93
- </div>
94
-
95
- <footer>Generated by JengaAgent /train skill &nbsp;|&nbsp; {{job_name}}</footer>
96
-
97
- </body>
98
- </html>
@@ -1,9 +0,0 @@
1
- """
2
- results-parsers — ML training result extraction utilities.
3
-
4
- Each submodule exposes: parse(job_dir: Path) -> dict
5
- Each submodule extracts type-specific metrics from job output artifacts.
6
- """
7
- from pathlib import Path
8
-
9
- PARSERS = ["classifiers", "transformers", "nlp"]
@@ -1,84 +0,0 @@
1
- """
2
- classifiers.py — Results parser for sklearn-based classifier jobs.
3
-
4
- Extracts: accuracy, precision, recall, f1_score, model_type.
5
- Reads from results.json or scans training-results/ for JSON files.
6
- """
7
- import json
8
- from pathlib import Path
9
-
10
-
11
- def parse(job_dir: Path) -> dict:
12
- """
13
- Parse training results for a classifiers job.
14
-
15
- Returns a dict with keys:
16
- accuracy, precision, recall, f1_score, model_type
17
- Returns empty dict if no results can be found; never raises.
18
- """
19
- job_dir = Path(job_dir)
20
- metrics = {}
21
-
22
- # 1. Try results.json at root
23
- results_json = job_dir / "results.json"
24
- if results_json.exists():
25
- try:
26
- data = json.loads(results_json.read_text())
27
- metrics = _extract_from_dict(data)
28
- if metrics:
29
- return metrics
30
- except Exception:
31
- pass
32
-
33
- # 2. Scan training-results/ for any JSON files
34
- training_results_dir = job_dir / "training-results"
35
- if training_results_dir.exists():
36
- for json_file in sorted(training_results_dir.rglob("*.json")):
37
- try:
38
- data = json.loads(json_file.read_text())
39
- metrics = _extract_from_dict(data)
40
- if metrics:
41
- return metrics
42
- except Exception:
43
- continue
44
-
45
- return metrics
46
-
47
-
48
- def _extract_from_dict(data: dict) -> dict:
49
- """Extract classifier-relevant keys from a parsed dict."""
50
- metrics = {}
51
- if not isinstance(data, dict):
52
- return metrics
53
-
54
- # Common field names produced by sklearn classification_report / custom scripts
55
- field_map = {
56
- "accuracy": ["accuracy", "test_accuracy", "val_accuracy", "acc"],
57
- "precision": ["precision", "weighted avg.precision", "macro avg.precision"],
58
- "recall": ["recall", "weighted avg.recall", "macro avg.recall"],
59
- "f1_score": ["f1_score", "f1", "weighted avg.f1-score", "macro avg.f1-score"],
60
- "model_type": ["model_type", "model", "algorithm", "classifier"],
61
- }
62
-
63
- for target_key, candidates in field_map.items():
64
- for candidate in candidates:
65
- # Support dot-path lookup (e.g. "weighted avg.precision")
66
- value = _deep_get(data, candidate)
67
- if value is not None:
68
- metrics[target_key] = value
69
- break
70
-
71
- return metrics
72
-
73
-
74
- def _deep_get(data: dict, dotted_key: str):
75
- """Retrieve a value from a nested dict using a dot-separated key path."""
76
- keys = dotted_key.split(".")
77
- current = data
78
- for k in keys:
79
- if not isinstance(current, dict):
80
- return None
81
- current = current.get(k)
82
- if current is None:
83
- return None
84
- return current
@@ -1,88 +0,0 @@
1
- """
2
- nlp.py — Results parser for spaCy / NLP pipeline training jobs.
3
-
4
- Extracts: f1, precision, recall, model_name, task, iterations.
5
- Reads from results.json or scans training-results/ for JSON files.
6
- """
7
- import json
8
- from pathlib import Path
9
-
10
-
11
- def parse(job_dir: Path) -> dict:
12
- """
13
- Parse training results for an nlp job.
14
-
15
- Returns a dict with keys:
16
- f1, precision, recall, model_name, task, iterations
17
- Returns empty dict if no results can be found; never raises.
18
- """
19
- job_dir = Path(job_dir)
20
- metrics = {}
21
-
22
- # 1. Try results.json at root
23
- results_json = job_dir / "results.json"
24
- if results_json.exists():
25
- try:
26
- data = json.loads(results_json.read_text())
27
- metrics = _extract_from_dict(data)
28
- if metrics:
29
- return metrics
30
- except Exception:
31
- pass
32
-
33
- # 2. Scan training-results/ for JSON files
34
- training_results_dir = job_dir / "training-results"
35
- if training_results_dir.exists():
36
- for json_file in sorted(training_results_dir.rglob("*.json")):
37
- try:
38
- data = json.loads(json_file.read_text())
39
- metrics = _extract_from_dict(data)
40
- if metrics:
41
- return metrics
42
- except Exception:
43
- continue
44
-
45
- # 3. Try spaCy training output (scores.json or metrics.json)
46
- for scores_file in sorted(job_dir.rglob("scores.json")) + sorted(job_dir.rglob("metrics.json")):
47
- try:
48
- data = json.loads(scores_file.read_text())
49
- metrics = _extract_from_dict(data)
50
- if metrics:
51
- return metrics
52
- except Exception:
53
- continue
54
-
55
- return metrics
56
-
57
-
58
- def _extract_from_dict(data: dict) -> dict:
59
- """Extract NLP-relevant keys from a parsed dict."""
60
- metrics = {}
61
- if not isinstance(data, dict):
62
- return metrics
63
-
64
- field_map = {
65
- "f1": ["f1", "f1_score", "ents_f", "token_f", "tag_f", "sents_f", "score"],
66
- "precision": ["precision", "ents_p", "token_p", "tag_p"],
67
- "recall": ["recall", "ents_r", "token_r", "tag_r"],
68
- "model_name": ["model_name", "model", "base_model"],
69
- "task": ["task", "pipeline_component", "component"],
70
- "iterations": ["iterations", "n_iter", "steps", "batches_trained"],
71
- }
72
-
73
- for target_key, candidates in field_map.items():
74
- for candidate in candidates:
75
- value = data.get(candidate)
76
- if value is None:
77
- # Try nested under "scores" or "results" key
78
- for wrapper in ("scores", "results", "metrics"):
79
- nested = data.get(wrapper, {})
80
- if isinstance(nested, dict):
81
- value = nested.get(candidate)
82
- if value is not None:
83
- break
84
- if value is not None:
85
- metrics[target_key] = value
86
- break
87
-
88
- return metrics
@@ -1,154 +0,0 @@
1
- """
2
- reporter.py — Surface training results to terminal, summary.md, and results.json.
3
-
4
- Usage:
5
- from skills.train.assets.results_parsers.reporter import surface_results
6
- surface_results(job_dir, job_type)
7
- """
8
- import importlib
9
- import json
10
- import sys
11
- from datetime import datetime, timezone
12
- from pathlib import Path
13
-
14
- _PARSER_PKG = Path(__file__).resolve().parent
15
- sys.path.insert(0, str(_PARSER_PKG.parent.parent.parent)) # ensure skills/ is on path
16
-
17
-
18
- def _load_parser(job_type: str):
19
- """Dynamically import the type-specific parser module."""
20
- try:
21
- spec_path = _PARSER_PKG / f"{job_type}.py"
22
- import importlib.util
23
- spec = importlib.util.spec_from_file_location(job_type, spec_path)
24
- mod = importlib.util.module_from_spec(spec)
25
- spec.loader.exec_module(mod)
26
- return mod
27
- except Exception as e:
28
- return None
29
-
30
-
31
- def _format_value(v) -> str:
32
- if isinstance(v, float):
33
- return f"{v:.4f}"
34
- return str(v)
35
-
36
-
37
- def _print_terminal_summary(job_dir: Path, job_type: str, metrics: dict):
38
- """Print a box-formatted summary to stdout."""
39
- job_name = job_dir.name
40
- width = 54
41
- border = "─" * width
42
- print(f"\n┌{border}┐")
43
- print(f"│ 📊 Training Summary — {job_name:<{width - 25}}│")
44
- print(f"│ Type: {job_type:<{width - 9}}│")
45
- print(f"├{border}┤")
46
- if metrics:
47
- for k, v in metrics.items():
48
- label = k.replace("_", " ").title()
49
- value = _format_value(v)
50
- line = f" {label}: {value}"
51
- print(f"│{line:<{width + 1}}│")
52
- else:
53
- print(f"│ (no metrics found — check training-results/ or results.json) │")
54
- print(f"└{border}┘\n")
55
-
56
-
57
- def _write_summary_md(job_dir: Path, job_type: str, metrics: dict):
58
- """Write a Markdown summary file to the job directory."""
59
- job_name = job_dir.name
60
- timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
61
- lines = [
62
- f"# Training Summary — {job_name}",
63
- f"",
64
- f"**Type:** {job_type} ",
65
- f"**Generated:** {timestamp} ",
66
- f"",
67
- f"## Metrics",
68
- f"",
69
- f"| Metric | Value |",
70
- f"|--------|-------|",
71
- ]
72
- if metrics:
73
- for k, v in metrics.items():
74
- label = k.replace("_", " ").title()
75
- lines.append(f"| {label} | {_format_value(v)} |")
76
- else:
77
- lines.append("| — | No metrics found |")
78
-
79
- summary_path = job_dir / "summary.md"
80
- summary_path.write_text("\n".join(lines) + "\n")
81
- return summary_path
82
-
83
-
84
- def _write_results_json(job_dir: Path, job_type: str, metrics: dict):
85
- """Write or update results.json in the job directory."""
86
- results_path = job_dir / "results.json"
87
-
88
- # Merge with existing results.json if present
89
- existing = {}
90
- if results_path.exists():
91
- try:
92
- existing = json.loads(results_path.read_text())
93
- except Exception:
94
- pass
95
-
96
- existing.update({
97
- "job_name": job_dir.name,
98
- "job_type": job_type,
99
- "generated_at": datetime.now(timezone.utc).isoformat(),
100
- "metrics": metrics,
101
- })
102
-
103
- results_path.write_text(json.dumps(existing, indent=2))
104
- return results_path
105
-
106
-
107
- def surface_results(job_dir, job_type: str):
108
- """
109
- Parse training results for the given job and surface them:
110
- 1. Print terminal summary
111
- 2. Write summary.md to job directory
112
- 3. Write/update results.json in job directory
113
-
114
- Args:
115
- job_dir: Path (or str) to the job directory
116
- job_type: One of 'classifiers', 'transformers', 'nlp'
117
- """
118
- job_dir = Path(job_dir)
119
-
120
- parser = _load_parser(job_type)
121
- if parser is None:
122
- print(f"⚠️ No parser found for type '{job_type}' — skipping result surfacing.")
123
- return {}
124
-
125
- try:
126
- metrics = parser.parse(job_dir) or {}
127
- except Exception as e:
128
- print(f"⚠️ Parser error for '{job_type}': {e}")
129
- metrics = {}
130
-
131
- _print_terminal_summary(job_dir, job_type, metrics)
132
-
133
- try:
134
- summary_path = _write_summary_md(job_dir, job_type, metrics)
135
- print(f"📄 summary.md written to {summary_path}")
136
- except Exception as e:
137
- print(f"⚠️ Could not write summary.md: {e}")
138
-
139
- try:
140
- results_path = _write_results_json(job_dir, job_type, metrics)
141
- print(f"📄 results.json written to {results_path}")
142
- except Exception as e:
143
- print(f"⚠️ Could not write results.json: {e}")
144
-
145
- return metrics
146
-
147
-
148
- if __name__ == "__main__":
149
- import argparse
150
- p = argparse.ArgumentParser(description="Surface training results for a job directory.")
151
- p.add_argument("job_dir", help="Path to the job directory")
152
- p.add_argument("job_type", choices=["classifiers", "transformers", "nlp"])
153
- ns = p.parse_args()
154
- surface_results(ns.job_dir, ns.job_type)
@@ -1,120 +0,0 @@
1
- """
2
- transformers.py — Results parser for HuggingFace Transformers fine-tuning jobs.
3
-
4
- Extracts: perplexity, eval_loss, train_loss, epochs, model_name.
5
- Reads from results.json or scans training-results/ and trainer_state.json.
6
- """
7
- import json
8
- import math
9
- from pathlib import Path
10
-
11
-
12
- def parse(job_dir: Path) -> dict:
13
- """
14
- Parse training results for a transformers job.
15
-
16
- Returns a dict with keys:
17
- perplexity, eval_loss, train_loss, epochs, model_name
18
- Returns empty dict if no results can be found; never raises.
19
- """
20
- job_dir = Path(job_dir)
21
- metrics = {}
22
-
23
- # 1. Try results.json at root
24
- results_json = job_dir / "results.json"
25
- if results_json.exists():
26
- try:
27
- data = json.loads(results_json.read_text())
28
- metrics = _extract_from_dict(data)
29
- if metrics:
30
- return metrics
31
- except Exception:
32
- pass
33
-
34
- # 2. Try trainer_state.json (HuggingFace Trainer output)
35
- for trainer_state in sorted(job_dir.rglob("trainer_state.json")):
36
- try:
37
- data = json.loads(trainer_state.read_text())
38
- metrics = _extract_from_trainer_state(data)
39
- if metrics:
40
- return metrics
41
- except Exception:
42
- continue
43
-
44
- # 3. Scan training-results/ for JSON files
45
- training_results_dir = job_dir / "training-results"
46
- if training_results_dir.exists():
47
- for json_file in sorted(training_results_dir.rglob("*.json")):
48
- try:
49
- data = json.loads(json_file.read_text())
50
- metrics = _extract_from_dict(data)
51
- if metrics:
52
- return metrics
53
- except Exception:
54
- continue
55
-
56
- return metrics
57
-
58
-
59
- def _extract_from_dict(data: dict) -> dict:
60
- """Extract transformer-relevant keys from a parsed dict."""
61
- metrics = {}
62
- if not isinstance(data, dict):
63
- return metrics
64
-
65
- field_map = {
66
- "eval_loss": ["eval_loss", "validation_loss", "val_loss"],
67
- "train_loss": ["train_loss", "training_loss", "loss"],
68
- "epochs": ["epochs", "num_train_epochs", "epoch"],
69
- "model_name": ["model_name", "model", "base_model", "pretrained_model_name_or_path"],
70
- "perplexity": ["perplexity", "eval_perplexity"],
71
- }
72
-
73
- for target_key, candidates in field_map.items():
74
- for candidate in candidates:
75
- value = data.get(candidate)
76
- if value is not None:
77
- metrics[target_key] = value
78
- break
79
-
80
- # Derive perplexity from eval_loss if not already present
81
- if "perplexity" not in metrics and "eval_loss" in metrics:
82
- try:
83
- metrics["perplexity"] = round(math.exp(float(metrics["eval_loss"])), 4)
84
- except (ValueError, OverflowError):
85
- pass
86
-
87
- return metrics
88
-
89
-
90
- def _extract_from_trainer_state(data: dict) -> dict:
91
- """Extract metrics from HuggingFace Trainer's trainer_state.json."""
92
- metrics = {}
93
- if not isinstance(data, dict):
94
- return metrics
95
-
96
- # Best metrics
97
- best_metric = data.get("best_metric")
98
- if best_metric is not None:
99
- metrics["eval_loss"] = best_metric
100
-
101
- # Epoch count
102
- epoch = data.get("epoch")
103
- if epoch is not None:
104
- metrics["epochs"] = epoch
105
-
106
- # Last eval from log history
107
- log_history = data.get("log_history", [])
108
- for entry in reversed(log_history):
109
- if "eval_loss" in entry:
110
- metrics["eval_loss"] = entry["eval_loss"]
111
- break
112
-
113
- # Derive perplexity
114
- if "perplexity" not in metrics and "eval_loss" in metrics:
115
- try:
116
- metrics["perplexity"] = round(math.exp(float(metrics["eval_loss"])), 4)
117
- except (ValueError, OverflowError):
118
- pass
119
-
120
- return metrics