@jenga-ai/agent 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +201 -0
- package/README.md +340 -0
- package/agents/ai_engineer.md +113 -0
- package/agents/developer.md +236 -0
- package/agents/scrum-master.md +349 -0
- package/agents/scrutiny-agent.md +137 -0
- package/agents/solution-assessor.md +185 -0
- package/agents/tester.md +339 -0
- package/bin/jenga.js +70 -0
- package/hooks/copilot_session_end.sh +29 -0
- package/hooks/on_session_end.sh +238 -0
- package/hooks/prompt_router.sh +11 -0
- package/hooks/prompt_router_helper.js +52 -0
- package/hooks/session_end_helper.js +29 -0
- package/hooks/session_end_watcher.sh +24 -0
- package/lib/commands/attach.js +47 -0
- package/lib/commands/init.js +207 -0
- package/lib/commands/start.js +16 -0
- package/lib/commands/status.js +53 -0
- package/lib/config-schema.js +72 -0
- package/lib/inject-settings.js +61 -0
- package/lib/mirror.js +244 -0
- package/lib/resolve-project-dir.sh +47 -0
- package/mcp/execute-ticket/index.js +10 -0
- package/mcp/execute-ticket/package.json +5 -0
- package/mcp/help/index.js +79 -0
- package/mcp/help/package.json +14 -0
- package/mcp/router/README.md +19 -0
- package/mcp/router/embedder.js +23 -0
- package/mcp/router/index.js +204 -0
- package/mcp/router/matcher.js +87 -0
- package/mcp/router/package-lock.json +1048 -0
- package/mcp/router/package.json +11 -0
- package/mcp/router/skill-index.js +104 -0
- package/package.json +47 -0
- package/scripts/board_resolver.sh +46 -0
- package/scripts/e25_s01_extract_board_graph.py +292 -0
- package/scripts/e25_s01_generate_synthetic_board.py +90 -0
- package/scripts/measurement-10x.json +50 -0
- package/scripts/measurement-10x.txt +4 -0
- package/scripts/measurement-real.json +50 -0
- package/scripts/measurement-real.txt +4 -0
- package/scripts/postinstall.js +165 -0
- package/scripts/todo_cleanup.sh +22 -0
- package/scripts/todo_manager.sh +86 -0
- package/scripts/validate-board.sh +190 -0
- package/scripts/validate-story-format.sh +53 -0
- package/skills/brainstorm/SKILL.md +47 -0
- package/skills/btw/SKILL.md +42 -0
- package/skills/commit/SKILL.md +29 -0
- package/skills/commit/assets/user_instructions_template.md +22 -0
- package/skills/continue/SKILL.md +29 -0
- package/skills/convert/SKILL.md +124 -0
- package/skills/convert/convert_cli.py +235 -0
- package/skills/convert/tests/sample.csv +4 -0
- package/skills/convert/tests/sample.json +5 -0
- package/skills/convert/tests/sample.jsonl +3 -0
- package/skills/convert/tests/sample.yaml +18 -0
- package/skills/convert/tests/sample_obj.csv +2 -0
- package/skills/convert/tests/sample_obj.json +9 -0
- package/skills/deep-dive/SKILL.md +167 -0
- package/skills/do/SKILL.md +88 -0
- package/skills/do/assets/sender_template.json +12 -0
- package/skills/doc/SKILL.md +314 -0
- package/skills/doc/assets/path-objectives.yaml +38 -0
- package/skills/doc-sync/SKILL.md +167 -0
- package/skills/doc-sync/assets/default_excludes.txt +21 -0
- package/skills/doc-sync/assets/doc_targets.md +14 -0
- package/skills/dooo/SKILL.md +60 -0
- package/skills/error/SKILL.md +29 -0
- package/skills/evaluate/SKILL.md +45 -0
- package/skills/evaluate/assets/evaluation_invokation_template.yml +3 -0
- package/skills/evaluate/assets/evaluation_rapport_template.md +24 -0
- package/skills/examplify/SKILL.md +42 -0
- package/skills/help/SKILL.md +36 -0
- package/skills/improve/SKILL.md +55 -0
- package/skills/index/scripts/board-index +4 -0
- package/skills/index/scripts/board_index.py +615 -0
- package/skills/index/scripts/smoke_test.sh +86 -0
- package/skills/init/SKILL.md +44 -0
- package/skills/init/assets/.gitignore_template +15 -0
- package/skills/init/assets/PROJECT_SUMMARY_template.md +13 -0
- package/skills/init/assets/directory_structure.txt +13 -0
- package/skills/init/assets/test-config_template.json +4 -0
- package/skills/init/assets/workflow_template.json +30 -0
- package/skills/init/scripts/init.sh +48 -0
- package/skills/jbp/SKILL.md +25 -0
- package/skills/jenga/SKILL.md +68 -0
- package/skills/lgtm/SKILL.md +21 -0
- package/skills/mirror-public/SKILL.md +237 -0
- package/skills/mirror-public/assets/config.json +5 -0
- package/skills/mirror-public/scripts/mirror.sh +374 -0
- package/skills/pi-plan/SKILL.md +62 -0
- package/skills/pi-plan/assets/epic.json +7 -0
- package/skills/pi-plan/assets/story_template.md +18 -0
- package/skills/proceed/SKILL.md +29 -0
- package/skills/publish/SKILL.md +351 -0
- package/skills/publish/adapters/droplet.md +200 -0
- package/skills/publish/adapters/mobile-ios.md +114 -0
- package/skills/publish/adapters/npm-ci.md +223 -0
- package/skills/publish/adapters/npm.md +121 -0
- package/skills/publish/assets/ExportOptions.plist.template +19 -0
- package/skills/publish/assets/ci-contract.md +111 -0
- package/skills/publish/assets/ownership-matrix.md +17 -0
- package/skills/publish/assets/publish.example.json +85 -0
- package/skills/publish/assets/publish.example.npm-ci.json +40 -0
- package/skills/publish/assets/publish.example.npm.json +41 -0
- package/skills/publish/assets/secrets-guide.md +104 -0
- package/skills/publish/schemas/fixtures/npm-ci-minimal.json +17 -0
- package/skills/publish/schemas/fixtures/npm-ci-with-empty-secrets.json +18 -0
- package/skills/publish/schemas/fixtures/npm-ci-with-workflow-path.json +18 -0
- package/skills/publish/schemas/publish.schema.json +428 -0
- package/skills/publish/scripts/check_target_config.sh +96 -0
- package/skills/publish/scripts/droplet_pipeline.sh +208 -0
- package/skills/publish/scripts/generate_release_notes.sh +200 -0
- package/skills/publish/scripts/ios_pipeline.sh +486 -0
- package/skills/publish/scripts/npm_ci_pipeline.sh +225 -0
- package/skills/publish/scripts/npm_pipeline.sh +249 -0
- package/skills/publish/scripts/publish_common.sh +253 -0
- package/skills/publish/scripts/publish_deploy.sh +538 -0
- package/skills/publish/scripts/reconcile_tags.sh +135 -0
- package/skills/publish/scripts/run_gates.sh +616 -0
- package/skills/publish/scripts/setup_wizard.sh +394 -0
- package/skills/publish/scripts/show_history.sh +95 -0
- package/skills/publish/scripts/suggest_semver_bump.sh +105 -0
- package/skills/publish/scripts/validate_config.sh +163 -0
- package/skills/publish/scripts/validate_droplet_env.sh +45 -0
- package/skills/publish/scripts/validate_ios_env.sh +68 -0
- package/skills/publish/scripts/validate_npm_ci_env.sh +71 -0
- package/skills/publish/scripts/validate_npm_env.sh +22 -0
- package/skills/publish/scripts/write_ledger_entry.sh +126 -0
- package/skills/publish/wizards/droplet.md +275 -0
- package/skills/publish/wizards/mobile-ios.md +157 -0
- package/skills/publish/wizards/npm-ci.md +240 -0
- package/skills/publish/wizards/npm.md +224 -0
- package/skills/reconcile/SKILL.md +93 -0
- package/skills/reconcile/assets/report_format.md +44 -0
- package/skills/reconcile-origin/SKILL.md +75 -0
- package/skills/reconcile-origin/scripts/reconcile-origin.sh +372 -0
- package/skills/redo/SKILL.md +70 -0
- package/skills/route/SKILL.md +180 -0
- package/skills/self-sync/SKILL.md +73 -0
- package/skills/self-sync/scripts/run.js +136 -0
- package/skills/skillify/SKILL.md +68 -0
- package/skills/skillify/assets/init-new/SKILL.md +35 -0
- package/skills/skillify/assets/init-new/assets/.gitignore_template +15 -0
- package/skills/skillify/assets/init-new/assets/PROJECT_SUMMARY_template.md +13 -0
- package/skills/skillify/assets/init-new/assets/directory_structure.txt +10 -0
- package/skills/skillify/assets/init-new/assets/test-config_template.json +4 -0
- package/skills/skillify/assets/init-new/assets/workflow_template.json +17 -0
- package/skills/skillify/assets/init-new/scripts/init.sh +48 -0
- package/skills/skillify/assets/init-old/SKILL.md +124 -0
- package/skills/spinoff/SKILL.md +48 -0
- package/skills/status/SKILL.md +33 -0
- package/skills/status/assets/output_format.md +41 -0
- package/skills/todo/SKILL.md +46 -0
- package/skills/todo/assets/todo_handoff_template.md +22 -0
- package/skills/todo/assets/todo_template.md +3 -0
- package/skills/train/SKILL.md +116 -0
- package/skills/train/assets/dashboard-templates/classifiers.html +106 -0
- package/skills/train/assets/dashboard-templates/nlp.html +102 -0
- package/skills/train/assets/dashboard-templates/transformers.html +98 -0
- package/skills/train/assets/results-parsers/__init__.py +9 -0
- package/skills/train/assets/results-parsers/classifiers.py +84 -0
- package/skills/train/assets/results-parsers/nlp.py +88 -0
- package/skills/train/assets/results-parsers/reporter.py +154 -0
- package/skills/train/assets/results-parsers/transformers.py +120 -0
- package/skills/train/train_cli.py +786 -0
- package/templates/EXECUTION_PLAN_TEMPLATE.md +43 -0
- package/templates/EXECUTION_SUMMARY_TEMPLATE.md +50 -0
- package/templates/JENGA_CONFIG_TEMPLATE.json +23 -0
- package/templates/PROBLEM_RAPPORT_TEMPLATE.md +88 -0
- package/templates/SCRUM_BOARD_SCHEMA.md +311 -0
- package/templates/SKILL.md +16 -0
- package/templates/SKILL_TEMPLATE.md +28 -0
- package/templates/USER_INSTRUCTIONS_TEMPLATE.md +22 -0
- package/templates/copilot-instructions.md.tpl +55 -0
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: train
|
|
3
|
+
description: Scaffold and run ML training jobs. Use `new <type> <job-name>` to scaffold a job from a template, or `run <job-dir>` to execute the two-phase validate → train pipeline on an existing job.
|
|
4
|
+
keywords:
|
|
5
|
+
- train
|
|
6
|
+
- ml training
|
|
7
|
+
- model training
|
|
8
|
+
- machine learning
|
|
9
|
+
- fine-tune
|
|
10
|
+
examples:
|
|
11
|
+
- "scaffold a new training job"
|
|
12
|
+
- "train a classifier model"
|
|
13
|
+
---
|
|
14
|
+
|
|
15
|
+
# Train — ML Training Job Orchestrator
|
|
16
|
+
|
|
17
|
+
## Usage
|
|
18
|
+
|
|
19
|
+
```
|
|
20
|
+
python skills/train/train_cli.py <subcommand> [args]
|
|
21
|
+
|
|
22
|
+
Subcommands:
|
|
23
|
+
new <type> <job-name> Scaffold a new training job from a template
|
|
24
|
+
run <job-dir> Execute validate → train pipeline on an existing job
|
|
25
|
+
|
|
26
|
+
Types: classifiers, transformers, nlp
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
If an unrecognised subcommand is given, argparse prints the usage message above and
|
|
30
|
+
exits with a non-zero code.
|
|
31
|
+
|
|
32
|
+
## Subcommands
|
|
33
|
+
|
|
34
|
+
### `/train new <type> <job-name>`
|
|
35
|
+
Scaffold a new training job from the appropriate template (positional / scriptable mode).
|
|
36
|
+
|
|
37
|
+
**Supported types:** `classifiers`, `transformers`, `nlp`
|
|
38
|
+
|
|
39
|
+
**Optional flags:**
|
|
40
|
+
- `--model <value>` — Override the model name/type in `config.yaml`
|
|
41
|
+
- `--epochs <n>` — Override the number of training epochs/iterations
|
|
42
|
+
- `--batch-size <n>` — Override the batch size
|
|
43
|
+
|
|
44
|
+
**Steps:**
|
|
45
|
+
1. Validate `<type>` is one of: `classifiers`, `transformers`, `nlp`
|
|
46
|
+
2. Copy the template directory from `.training/template/<type>/` to `jobs/<job-name>/`
|
|
47
|
+
- If `jobs/<job-name>/` already exists, auto-suffix: `jobs/<job-name>-1/`, `jobs/<job-name>-2/`, etc.
|
|
48
|
+
3. Apply any CLI flag overrides to the scaffolded `config.yaml`
|
|
49
|
+
4. Generate `start.sh` in the job directory (executable)
|
|
50
|
+
5. Read the `workflow:` block from `config.yaml` and print next-step instructions
|
|
51
|
+
6. Print: `✅ Scaffolded job '<job-name>' from template '<type>' at jobs/<job-name>/`
|
|
52
|
+
|
|
53
|
+
### `/train new --interactive [--full]`
|
|
54
|
+
Launch a guided wizard that scaffolds the job **and** configures `config.yaml` interactively.
|
|
55
|
+
|
|
56
|
+
**Flags:**
|
|
57
|
+
- `--interactive` / `-i` — Enable the wizard. Job name and type are collected via prompts.
|
|
58
|
+
- `--full` — Expand the wizard to prompt for **every** configurable field (not just the critical ones).
|
|
59
|
+
|
|
60
|
+
**Wizard flow:**
|
|
61
|
+
1. Prompt for job name (freeform, used as directory name under `jobs/`)
|
|
62
|
+
2. Prompt for job type (`classifiers`, `transformers`, or `nlp`)
|
|
63
|
+
3. Scaffold the job directory from the template
|
|
64
|
+
4. Prompt for critical config fields (2–3 key model params + data path per type):
|
|
65
|
+
- **classifiers**: train file, target column, model algorithm, n_estimators, test split
|
|
66
|
+
- **transformers**: model name, task, train file, epochs, learning rate
|
|
67
|
+
- **nlp**: model name, task, train file, iterations, learning rate
|
|
68
|
+
5. With `--full`: additionally prompt for every remaining field in `config.yaml`
|
|
69
|
+
6. Write user responses back into the scaffolded `config.yaml`
|
|
70
|
+
|
|
71
|
+
Each prompt shows the field name, its current default, and an inline description hint.
|
|
72
|
+
|
|
73
|
+
**Ctrl+C handling:**
|
|
74
|
+
Pressing `Ctrl+C` at any point during the wizard interrupts cleanly. If a scaffold directory has already been created, the user is asked:
|
|
75
|
+
```
|
|
76
|
+
Delete scaffolded directory '<path>'? [y/N]
|
|
77
|
+
```
|
|
78
|
+
Answering `y` removes the directory; `N` (or Enter) keeps it.
|
|
79
|
+
|
|
80
|
+
### `/train run <job-dir>`
|
|
81
|
+
Execute the two-phase pre-flight → smoke test pipeline on an existing job directory.
|
|
82
|
+
|
|
83
|
+
**Steps:**
|
|
84
|
+
|
|
85
|
+
#### Phase A — Pre-flight validation
|
|
86
|
+
1. Print: `[pre-flight] Running validate.py in <job-dir>...`
|
|
87
|
+
2. Run `validate.py` as a subprocess inside `<job-dir>`:
|
|
88
|
+
```bash
|
|
89
|
+
cd <job-dir> && python validate.py
|
|
90
|
+
```
|
|
91
|
+
3. Stream all stdout/stderr output to the terminal in real time.
|
|
92
|
+
4. If `validate.py` exits non-zero:
|
|
93
|
+
- Print: `❌ [pre-flight] FAILED — halting before smoke test.`
|
|
94
|
+
- Surface the full error output.
|
|
95
|
+
- **Stop here. Do not proceed to Phase B.**
|
|
96
|
+
5. If `validate.py` exits zero:
|
|
97
|
+
- Print: `✅ [pre-flight] Passed.`
|
|
98
|
+
|
|
99
|
+
#### Phase B — Smoke test
|
|
100
|
+
1. Print: `[smoke test] Running train.py --smoke in <job-dir>...`
|
|
101
|
+
2. Run `train.py --smoke` as a subprocess inside `<job-dir>`:
|
|
102
|
+
```bash
|
|
103
|
+
cd <job-dir> && python train.py --smoke
|
|
104
|
+
```
|
|
105
|
+
3. Stream all stdout/stderr output to the terminal in real time.
|
|
106
|
+
4. If `train.py` exits non-zero:
|
|
107
|
+
- Print: `❌ [smoke test] FAILED.`
|
|
108
|
+
- Surface the full error output.
|
|
109
|
+
5. If `train.py` exits zero:
|
|
110
|
+
- Print: `✅ [smoke test] Passed. Job '<job-dir>' completed successfully.`
|
|
111
|
+
|
|
112
|
+
## Notes
|
|
113
|
+
- `train.py` and `validate.py` must have no awareness of each other — the `/train` skill owns the gate.
|
|
114
|
+
- Both subprocess calls must stream output (not buffer) so the user sees progress in real time.
|
|
115
|
+
- Phase B is **never** invoked if Phase A fails.
|
|
116
|
+
- The `--interactive` wizard requires `pyyaml` (`pip install pyyaml`) to patch `config.yaml`.
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
<!DOCTYPE html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="UTF-8" />
|
|
5
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
+
<title>Classifier Training Dashboard — {{job_name}}</title>
|
|
7
|
+
<style>
|
|
8
|
+
*, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
|
|
9
|
+
body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
|
|
10
|
+
header { margin-bottom: 2rem; }
|
|
11
|
+
header h1 { font-size: 1.6rem; font-weight: 700; }
|
|
12
|
+
header p { color: #555; margin-top: .25rem; font-size: .9rem; }
|
|
13
|
+
.card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
|
|
14
|
+
.card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
|
|
15
|
+
.metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
|
|
16
|
+
.metric-box { background: #f0f4ff; border-radius: 8px; padding: 1rem; text-align: center; }
|
|
17
|
+
.metric-box .value { font-size: 2rem; font-weight: 700; color: #3b5bdb; }
|
|
18
|
+
.metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
|
|
19
|
+
.bar-chart { display: flex; flex-direction: column; gap: .6rem; }
|
|
20
|
+
.bar-row { display: flex; align-items: center; gap .5rem; }
|
|
21
|
+
.bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
|
|
22
|
+
.bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
|
|
23
|
+
.bar-fill { height: 100%; border-radius: 4px; background: #3b5bdb; transition: width .4s; }
|
|
24
|
+
.bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
|
|
25
|
+
table { width: 100%; border-collapse: collapse; }
|
|
26
|
+
th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
|
|
27
|
+
th { background: #f8f9fa; font-weight: 600; color: #555; }
|
|
28
|
+
footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
|
|
29
|
+
</style>
|
|
30
|
+
</head>
|
|
31
|
+
<body>
|
|
32
|
+
|
|
33
|
+
<header>
|
|
34
|
+
<h1>🤖 Classifier Training Dashboard</h1>
|
|
35
|
+
<p>Job: <strong>{{job_name}}</strong> | Model: <strong>{{model_type}}</strong></p>
|
|
36
|
+
</header>
|
|
37
|
+
|
|
38
|
+
<div class="card">
|
|
39
|
+
<h2>Key Metrics</h2>
|
|
40
|
+
<div class="metrics-grid">
|
|
41
|
+
<div class="metric-box">
|
|
42
|
+
<div class="value">{{accuracy}}</div>
|
|
43
|
+
<div class="label">Accuracy</div>
|
|
44
|
+
</div>
|
|
45
|
+
<div class="metric-box">
|
|
46
|
+
<div class="value">{{f1_score}}</div>
|
|
47
|
+
<div class="label">F1 Score</div>
|
|
48
|
+
</div>
|
|
49
|
+
<div class="metric-box">
|
|
50
|
+
<div class="value">{{precision}}</div>
|
|
51
|
+
<div class="label">Precision</div>
|
|
52
|
+
</div>
|
|
53
|
+
<div class="metric-box">
|
|
54
|
+
<div class="value">{{recall}}</div>
|
|
55
|
+
<div class="label">Recall</div>
|
|
56
|
+
</div>
|
|
57
|
+
</div>
|
|
58
|
+
</div>
|
|
59
|
+
|
|
60
|
+
<div class="card">
|
|
61
|
+
<h2>Metric Comparison</h2>
|
|
62
|
+
<div class="bar-chart">
|
|
63
|
+
<div class="bar-row">
|
|
64
|
+
<span class="bar-label">Accuracy</span>
|
|
65
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{accuracy}} * 100%)"></div></div>
|
|
66
|
+
<span class="bar-val">{{accuracy}}</span>
|
|
67
|
+
</div>
|
|
68
|
+
<div class="bar-row">
|
|
69
|
+
<span class="bar-label">F1 Score</span>
|
|
70
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{f1_score}} * 100%)"></div></div>
|
|
71
|
+
<span class="bar-val">{{f1_score}}</span>
|
|
72
|
+
</div>
|
|
73
|
+
<div class="bar-row">
|
|
74
|
+
<span class="bar-label">Precision</span>
|
|
75
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
|
|
76
|
+
<span class="bar-val">{{precision}}</span>
|
|
77
|
+
</div>
|
|
78
|
+
<div class="bar-row">
|
|
79
|
+
<span class="bar-label">Recall</span>
|
|
80
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
|
|
81
|
+
<span class="bar-val">{{recall}}</span>
|
|
82
|
+
</div>
|
|
83
|
+
</div>
|
|
84
|
+
</div>
|
|
85
|
+
|
|
86
|
+
<div class="card">
|
|
87
|
+
<h2>Summary Table</h2>
|
|
88
|
+
<table>
|
|
89
|
+
<thead>
|
|
90
|
+
<tr><th>Metric</th><th>Value</th></tr>
|
|
91
|
+
</thead>
|
|
92
|
+
<tbody>
|
|
93
|
+
<tr><td>Accuracy</td><td>{{accuracy}}</td></tr>
|
|
94
|
+
<tr><td>F1 Score</td><td>{{f1_score}}</td></tr>
|
|
95
|
+
<tr><td>Precision</td><td>{{precision}}</td></tr>
|
|
96
|
+
<tr><td>Recall</td><td>{{recall}}</td></tr>
|
|
97
|
+
<tr><td>Model Type</td><td>{{model_type}}</td></tr>
|
|
98
|
+
<tr><td>Job Name</td><td>{{job_name}}</td></tr>
|
|
99
|
+
</tbody>
|
|
100
|
+
</table>
|
|
101
|
+
</div>
|
|
102
|
+
|
|
103
|
+
<footer>Generated by JengaAgent /train skill | {{job_name}}</footer>
|
|
104
|
+
|
|
105
|
+
</body>
|
|
106
|
+
</html>
|
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
<!DOCTYPE html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="UTF-8" />
|
|
5
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
+
<title>NLP Pipeline Dashboard — {{job_name}}</title>
|
|
7
|
+
<style>
|
|
8
|
+
*, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
|
|
9
|
+
body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
|
|
10
|
+
header { margin-bottom: 2rem; }
|
|
11
|
+
header h1 { font-size: 1.6rem; font-weight: 700; }
|
|
12
|
+
header p { color: #555; margin-top: .25rem; font-size: .9rem; }
|
|
13
|
+
.card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
|
|
14
|
+
.card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
|
|
15
|
+
.metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
|
|
16
|
+
.metric-box { background: #fff8e1; border-radius: 8px; padding: 1rem; text-align: center; }
|
|
17
|
+
.metric-box .value { font-size: 2rem; font-weight: 700; color: #e67700; }
|
|
18
|
+
.metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
|
|
19
|
+
.bar-chart { display: flex; flex-direction: column; gap: .6rem; }
|
|
20
|
+
.bar-row { display: flex; align-items: center; gap: .5rem; }
|
|
21
|
+
.bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
|
|
22
|
+
.bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
|
|
23
|
+
.bar-fill { height: 100%; border-radius: 4px; background: #e67700; transition: width .4s; }
|
|
24
|
+
.bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
|
|
25
|
+
table { width: 100%; border-collapse: collapse; }
|
|
26
|
+
th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
|
|
27
|
+
th { background: #f8f9fa; font-weight: 600; color: #555; }
|
|
28
|
+
footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
|
|
29
|
+
</style>
|
|
30
|
+
</head>
|
|
31
|
+
<body>
|
|
32
|
+
|
|
33
|
+
<header>
|
|
34
|
+
<h1>🔤 NLP Pipeline Dashboard</h1>
|
|
35
|
+
<p>Job: <strong>{{job_name}}</strong> | Model: <strong>{{model_name}}</strong> | Task: <strong>{{task}}</strong></p>
|
|
36
|
+
</header>
|
|
37
|
+
|
|
38
|
+
<div class="card">
|
|
39
|
+
<h2>Key Metrics</h2>
|
|
40
|
+
<div class="metrics-grid">
|
|
41
|
+
<div class="metric-box">
|
|
42
|
+
<div class="value">{{f1}}</div>
|
|
43
|
+
<div class="label">F1 Score</div>
|
|
44
|
+
</div>
|
|
45
|
+
<div class="metric-box">
|
|
46
|
+
<div class="value">{{precision}}</div>
|
|
47
|
+
<div class="label">Precision</div>
|
|
48
|
+
</div>
|
|
49
|
+
<div class="metric-box">
|
|
50
|
+
<div class="value">{{recall}}</div>
|
|
51
|
+
<div class="label">Recall</div>
|
|
52
|
+
</div>
|
|
53
|
+
<div class="metric-box">
|
|
54
|
+
<div class="value">{{iterations}}</div>
|
|
55
|
+
<div class="label">Iterations</div>
|
|
56
|
+
</div>
|
|
57
|
+
</div>
|
|
58
|
+
</div>
|
|
59
|
+
|
|
60
|
+
<div class="card">
|
|
61
|
+
<h2>Score Breakdown</h2>
|
|
62
|
+
<div class="bar-chart">
|
|
63
|
+
<div class="bar-row">
|
|
64
|
+
<span class="bar-label">F1 Score</span>
|
|
65
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{f1}} * 100%)"></div></div>
|
|
66
|
+
<span class="bar-val">{{f1}}</span>
|
|
67
|
+
</div>
|
|
68
|
+
<div class="bar-row">
|
|
69
|
+
<span class="bar-label">Precision</span>
|
|
70
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
|
|
71
|
+
<span class="bar-val">{{precision}}</span>
|
|
72
|
+
</div>
|
|
73
|
+
<div class="bar-row">
|
|
74
|
+
<span class="bar-label">Recall</span>
|
|
75
|
+
<div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
|
|
76
|
+
<span class="bar-val">{{recall}}</span>
|
|
77
|
+
</div>
|
|
78
|
+
</div>
|
|
79
|
+
</div>
|
|
80
|
+
|
|
81
|
+
<div class="card">
|
|
82
|
+
<h2>Summary Table</h2>
|
|
83
|
+
<table>
|
|
84
|
+
<thead>
|
|
85
|
+
<tr><th>Metric</th><th>Value</th></tr>
|
|
86
|
+
</thead>
|
|
87
|
+
<tbody>
|
|
88
|
+
<tr><td>F1 Score</td><td>{{f1}}</td></tr>
|
|
89
|
+
<tr><td>Precision</td><td>{{precision}}</td></tr>
|
|
90
|
+
<tr><td>Recall</td><td>{{recall}}</td></tr>
|
|
91
|
+
<tr><td>Iterations</td><td>{{iterations}}</td></tr>
|
|
92
|
+
<tr><td>Model Name</td><td>{{model_name}}</td></tr>
|
|
93
|
+
<tr><td>Task</td><td>{{task}}</td></tr>
|
|
94
|
+
<tr><td>Job Name</td><td>{{job_name}}</td></tr>
|
|
95
|
+
</tbody>
|
|
96
|
+
</table>
|
|
97
|
+
</div>
|
|
98
|
+
|
|
99
|
+
<footer>Generated by JengaAgent /train skill | {{job_name}}</footer>
|
|
100
|
+
|
|
101
|
+
</body>
|
|
102
|
+
</html>
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
<!DOCTYPE html>
|
|
2
|
+
<html lang="en">
|
|
3
|
+
<head>
|
|
4
|
+
<meta charset="UTF-8" />
|
|
5
|
+
<meta name="viewport" content="width=device-width, initial-scale=1.0" />
|
|
6
|
+
<title>Transformer Fine-Tuning Dashboard — {{job_name}}</title>
|
|
7
|
+
<style>
|
|
8
|
+
*, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
|
|
9
|
+
body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
|
|
10
|
+
header { margin-bottom: 2rem; }
|
|
11
|
+
header h1 { font-size: 1.6rem; font-weight: 700; }
|
|
12
|
+
header p { color: #555; margin-top: .25rem; font-size: .9rem; }
|
|
13
|
+
.card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
|
|
14
|
+
.card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
|
|
15
|
+
.metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); gap: 1rem; }
|
|
16
|
+
.metric-box { background: #f0fff4; border-radius: 8px; padding: 1rem; text-align: center; }
|
|
17
|
+
.metric-box .value { font-size: 2rem; font-weight: 700; color: #2f9e44; }
|
|
18
|
+
.metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
|
|
19
|
+
.loss-section { margin-top: 1rem; }
|
|
20
|
+
.loss-row { display: flex; justify-content: space-between; padding: .5rem 0; border-bottom: 1px solid #eee; font-size: .9rem; }
|
|
21
|
+
.loss-row .loss-label { color: #555; }
|
|
22
|
+
.loss-row .loss-val { font-weight: 600; color: #2f9e44; }
|
|
23
|
+
table { width: 100%; border-collapse: collapse; }
|
|
24
|
+
th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
|
|
25
|
+
th { background: #f8f9fa; font-weight: 600; color: #555; }
|
|
26
|
+
.perplexity-note { font-size: .8rem; color: #777; margin-top: .5rem; }
|
|
27
|
+
footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
|
|
28
|
+
</style>
|
|
29
|
+
</head>
|
|
30
|
+
<body>
|
|
31
|
+
|
|
32
|
+
<header>
|
|
33
|
+
<h1>🤗 Transformer Fine-Tuning Dashboard</h1>
|
|
34
|
+
<p>Job: <strong>{{job_name}}</strong> | Model: <strong>{{model_name}}</strong></p>
|
|
35
|
+
</header>
|
|
36
|
+
|
|
37
|
+
<div class="card">
|
|
38
|
+
<h2>Key Metrics</h2>
|
|
39
|
+
<div class="metrics-grid">
|
|
40
|
+
<div class="metric-box">
|
|
41
|
+
<div class="value">{{perplexity}}</div>
|
|
42
|
+
<div class="label">Perplexity</div>
|
|
43
|
+
</div>
|
|
44
|
+
<div class="metric-box">
|
|
45
|
+
<div class="value">{{eval_loss}}</div>
|
|
46
|
+
<div class="label">Eval Loss</div>
|
|
47
|
+
</div>
|
|
48
|
+
<div class="metric-box">
|
|
49
|
+
<div class="value">{{train_loss}}</div>
|
|
50
|
+
<div class="label">Train Loss</div>
|
|
51
|
+
</div>
|
|
52
|
+
<div class="metric-box">
|
|
53
|
+
<div class="value">{{epochs}}</div>
|
|
54
|
+
<div class="label">Epochs</div>
|
|
55
|
+
</div>
|
|
56
|
+
</div>
|
|
57
|
+
<p class="perplexity-note">ℹ️ Lower perplexity = better language model. Derived from eval_loss (e^loss).</p>
|
|
58
|
+
</div>
|
|
59
|
+
|
|
60
|
+
<div class="card">
|
|
61
|
+
<h2>Loss Breakdown</h2>
|
|
62
|
+
<div class="loss-section">
|
|
63
|
+
<div class="loss-row">
|
|
64
|
+
<span class="loss-label">Evaluation Loss</span>
|
|
65
|
+
<span class="loss-val">{{eval_loss}}</span>
|
|
66
|
+
</div>
|
|
67
|
+
<div class="loss-row">
|
|
68
|
+
<span class="loss-label">Training Loss</span>
|
|
69
|
+
<span class="loss-val">{{train_loss}}</span>
|
|
70
|
+
</div>
|
|
71
|
+
<div class="loss-row">
|
|
72
|
+
<span class="loss-label">Perplexity (e^eval_loss)</span>
|
|
73
|
+
<span class="loss-val">{{perplexity}}</span>
|
|
74
|
+
</div>
|
|
75
|
+
</div>
|
|
76
|
+
</div>
|
|
77
|
+
|
|
78
|
+
<div class="card">
|
|
79
|
+
<h2>Summary Table</h2>
|
|
80
|
+
<table>
|
|
81
|
+
<thead>
|
|
82
|
+
<tr><th>Metric</th><th>Value</th></tr>
|
|
83
|
+
</thead>
|
|
84
|
+
<tbody>
|
|
85
|
+
<tr><td>Perplexity</td><td>{{perplexity}}</td></tr>
|
|
86
|
+
<tr><td>Eval Loss</td><td>{{eval_loss}}</td></tr>
|
|
87
|
+
<tr><td>Train Loss</td><td>{{train_loss}}</td></tr>
|
|
88
|
+
<tr><td>Epochs</td><td>{{epochs}}</td></tr>
|
|
89
|
+
<tr><td>Model Name</td><td>{{model_name}}</td></tr>
|
|
90
|
+
<tr><td>Job Name</td><td>{{job_name}}</td></tr>
|
|
91
|
+
</tbody>
|
|
92
|
+
</table>
|
|
93
|
+
</div>
|
|
94
|
+
|
|
95
|
+
<footer>Generated by JengaAgent /train skill | {{job_name}}</footer>
|
|
96
|
+
|
|
97
|
+
</body>
|
|
98
|
+
</html>
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
"""
|
|
2
|
+
results-parsers — ML training result extraction utilities.
|
|
3
|
+
|
|
4
|
+
Each submodule exposes: parse(job_dir: Path) -> dict
|
|
5
|
+
Each submodule extracts type-specific metrics from job output artifacts.
|
|
6
|
+
"""
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
PARSERS = ["classifiers", "transformers", "nlp"]
|
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""
|
|
2
|
+
classifiers.py — Results parser for sklearn-based classifier jobs.
|
|
3
|
+
|
|
4
|
+
Extracts: accuracy, precision, recall, f1_score, model_type.
|
|
5
|
+
Reads from results.json or scans training-results/ for JSON files.
|
|
6
|
+
"""
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def parse(job_dir: Path) -> dict:
|
|
12
|
+
"""
|
|
13
|
+
Parse training results for a classifiers job.
|
|
14
|
+
|
|
15
|
+
Returns a dict with keys:
|
|
16
|
+
accuracy, precision, recall, f1_score, model_type
|
|
17
|
+
Returns empty dict if no results can be found; never raises.
|
|
18
|
+
"""
|
|
19
|
+
job_dir = Path(job_dir)
|
|
20
|
+
metrics = {}
|
|
21
|
+
|
|
22
|
+
# 1. Try results.json at root
|
|
23
|
+
results_json = job_dir / "results.json"
|
|
24
|
+
if results_json.exists():
|
|
25
|
+
try:
|
|
26
|
+
data = json.loads(results_json.read_text())
|
|
27
|
+
metrics = _extract_from_dict(data)
|
|
28
|
+
if metrics:
|
|
29
|
+
return metrics
|
|
30
|
+
except Exception:
|
|
31
|
+
pass
|
|
32
|
+
|
|
33
|
+
# 2. Scan training-results/ for any JSON files
|
|
34
|
+
training_results_dir = job_dir / "training-results"
|
|
35
|
+
if training_results_dir.exists():
|
|
36
|
+
for json_file in sorted(training_results_dir.rglob("*.json")):
|
|
37
|
+
try:
|
|
38
|
+
data = json.loads(json_file.read_text())
|
|
39
|
+
metrics = _extract_from_dict(data)
|
|
40
|
+
if metrics:
|
|
41
|
+
return metrics
|
|
42
|
+
except Exception:
|
|
43
|
+
continue
|
|
44
|
+
|
|
45
|
+
return metrics
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _extract_from_dict(data: dict) -> dict:
|
|
49
|
+
"""Extract classifier-relevant keys from a parsed dict."""
|
|
50
|
+
metrics = {}
|
|
51
|
+
if not isinstance(data, dict):
|
|
52
|
+
return metrics
|
|
53
|
+
|
|
54
|
+
# Common field names produced by sklearn classification_report / custom scripts
|
|
55
|
+
field_map = {
|
|
56
|
+
"accuracy": ["accuracy", "test_accuracy", "val_accuracy", "acc"],
|
|
57
|
+
"precision": ["precision", "weighted avg.precision", "macro avg.precision"],
|
|
58
|
+
"recall": ["recall", "weighted avg.recall", "macro avg.recall"],
|
|
59
|
+
"f1_score": ["f1_score", "f1", "weighted avg.f1-score", "macro avg.f1-score"],
|
|
60
|
+
"model_type": ["model_type", "model", "algorithm", "classifier"],
|
|
61
|
+
}
|
|
62
|
+
|
|
63
|
+
for target_key, candidates in field_map.items():
|
|
64
|
+
for candidate in candidates:
|
|
65
|
+
# Support dot-path lookup (e.g. "weighted avg.precision")
|
|
66
|
+
value = _deep_get(data, candidate)
|
|
67
|
+
if value is not None:
|
|
68
|
+
metrics[target_key] = value
|
|
69
|
+
break
|
|
70
|
+
|
|
71
|
+
return metrics
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _deep_get(data: dict, dotted_key: str):
|
|
75
|
+
"""Retrieve a value from a nested dict using a dot-separated key path."""
|
|
76
|
+
keys = dotted_key.split(".")
|
|
77
|
+
current = data
|
|
78
|
+
for k in keys:
|
|
79
|
+
if not isinstance(current, dict):
|
|
80
|
+
return None
|
|
81
|
+
current = current.get(k)
|
|
82
|
+
if current is None:
|
|
83
|
+
return None
|
|
84
|
+
return current
|
|
@@ -0,0 +1,88 @@
|
|
|
1
|
+
"""
|
|
2
|
+
nlp.py — Results parser for spaCy / NLP pipeline training jobs.
|
|
3
|
+
|
|
4
|
+
Extracts: f1, precision, recall, model_name, task, iterations.
|
|
5
|
+
Reads from results.json or scans training-results/ for JSON files.
|
|
6
|
+
"""
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def parse(job_dir: Path) -> dict:
|
|
12
|
+
"""
|
|
13
|
+
Parse training results for an nlp job.
|
|
14
|
+
|
|
15
|
+
Returns a dict with keys:
|
|
16
|
+
f1, precision, recall, model_name, task, iterations
|
|
17
|
+
Returns empty dict if no results can be found; never raises.
|
|
18
|
+
"""
|
|
19
|
+
job_dir = Path(job_dir)
|
|
20
|
+
metrics = {}
|
|
21
|
+
|
|
22
|
+
# 1. Try results.json at root
|
|
23
|
+
results_json = job_dir / "results.json"
|
|
24
|
+
if results_json.exists():
|
|
25
|
+
try:
|
|
26
|
+
data = json.loads(results_json.read_text())
|
|
27
|
+
metrics = _extract_from_dict(data)
|
|
28
|
+
if metrics:
|
|
29
|
+
return metrics
|
|
30
|
+
except Exception:
|
|
31
|
+
pass
|
|
32
|
+
|
|
33
|
+
# 2. Scan training-results/ for JSON files
|
|
34
|
+
training_results_dir = job_dir / "training-results"
|
|
35
|
+
if training_results_dir.exists():
|
|
36
|
+
for json_file in sorted(training_results_dir.rglob("*.json")):
|
|
37
|
+
try:
|
|
38
|
+
data = json.loads(json_file.read_text())
|
|
39
|
+
metrics = _extract_from_dict(data)
|
|
40
|
+
if metrics:
|
|
41
|
+
return metrics
|
|
42
|
+
except Exception:
|
|
43
|
+
continue
|
|
44
|
+
|
|
45
|
+
# 3. Try spaCy training output (scores.json or metrics.json)
|
|
46
|
+
for scores_file in sorted(job_dir.rglob("scores.json")) + sorted(job_dir.rglob("metrics.json")):
|
|
47
|
+
try:
|
|
48
|
+
data = json.loads(scores_file.read_text())
|
|
49
|
+
metrics = _extract_from_dict(data)
|
|
50
|
+
if metrics:
|
|
51
|
+
return metrics
|
|
52
|
+
except Exception:
|
|
53
|
+
continue
|
|
54
|
+
|
|
55
|
+
return metrics
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _extract_from_dict(data: dict) -> dict:
|
|
59
|
+
"""Extract NLP-relevant keys from a parsed dict."""
|
|
60
|
+
metrics = {}
|
|
61
|
+
if not isinstance(data, dict):
|
|
62
|
+
return metrics
|
|
63
|
+
|
|
64
|
+
field_map = {
|
|
65
|
+
"f1": ["f1", "f1_score", "ents_f", "token_f", "tag_f", "sents_f", "score"],
|
|
66
|
+
"precision": ["precision", "ents_p", "token_p", "tag_p"],
|
|
67
|
+
"recall": ["recall", "ents_r", "token_r", "tag_r"],
|
|
68
|
+
"model_name": ["model_name", "model", "base_model"],
|
|
69
|
+
"task": ["task", "pipeline_component", "component"],
|
|
70
|
+
"iterations": ["iterations", "n_iter", "steps", "batches_trained"],
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
for target_key, candidates in field_map.items():
|
|
74
|
+
for candidate in candidates:
|
|
75
|
+
value = data.get(candidate)
|
|
76
|
+
if value is None:
|
|
77
|
+
# Try nested under "scores" or "results" key
|
|
78
|
+
for wrapper in ("scores", "results", "metrics"):
|
|
79
|
+
nested = data.get(wrapper, {})
|
|
80
|
+
if isinstance(nested, dict):
|
|
81
|
+
value = nested.get(candidate)
|
|
82
|
+
if value is not None:
|
|
83
|
+
break
|
|
84
|
+
if value is not None:
|
|
85
|
+
metrics[target_key] = value
|
|
86
|
+
break
|
|
87
|
+
|
|
88
|
+
return metrics
|