@jenga-ai/agent 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (177) hide show
  1. package/LICENSE +201 -0
  2. package/README.md +340 -0
  3. package/agents/ai_engineer.md +113 -0
  4. package/agents/developer.md +236 -0
  5. package/agents/scrum-master.md +349 -0
  6. package/agents/scrutiny-agent.md +137 -0
  7. package/agents/solution-assessor.md +185 -0
  8. package/agents/tester.md +339 -0
  9. package/bin/jenga.js +70 -0
  10. package/hooks/copilot_session_end.sh +29 -0
  11. package/hooks/on_session_end.sh +238 -0
  12. package/hooks/prompt_router.sh +11 -0
  13. package/hooks/prompt_router_helper.js +52 -0
  14. package/hooks/session_end_helper.js +29 -0
  15. package/hooks/session_end_watcher.sh +24 -0
  16. package/lib/commands/attach.js +47 -0
  17. package/lib/commands/init.js +207 -0
  18. package/lib/commands/start.js +16 -0
  19. package/lib/commands/status.js +53 -0
  20. package/lib/config-schema.js +72 -0
  21. package/lib/inject-settings.js +61 -0
  22. package/lib/mirror.js +244 -0
  23. package/lib/resolve-project-dir.sh +47 -0
  24. package/mcp/execute-ticket/index.js +10 -0
  25. package/mcp/execute-ticket/package.json +5 -0
  26. package/mcp/help/index.js +79 -0
  27. package/mcp/help/package.json +14 -0
  28. package/mcp/router/README.md +19 -0
  29. package/mcp/router/embedder.js +23 -0
  30. package/mcp/router/index.js +204 -0
  31. package/mcp/router/matcher.js +87 -0
  32. package/mcp/router/package-lock.json +1048 -0
  33. package/mcp/router/package.json +11 -0
  34. package/mcp/router/skill-index.js +104 -0
  35. package/package.json +47 -0
  36. package/scripts/board_resolver.sh +46 -0
  37. package/scripts/e25_s01_extract_board_graph.py +292 -0
  38. package/scripts/e25_s01_generate_synthetic_board.py +90 -0
  39. package/scripts/measurement-10x.json +50 -0
  40. package/scripts/measurement-10x.txt +4 -0
  41. package/scripts/measurement-real.json +50 -0
  42. package/scripts/measurement-real.txt +4 -0
  43. package/scripts/postinstall.js +165 -0
  44. package/scripts/todo_cleanup.sh +22 -0
  45. package/scripts/todo_manager.sh +86 -0
  46. package/scripts/validate-board.sh +190 -0
  47. package/scripts/validate-story-format.sh +53 -0
  48. package/skills/brainstorm/SKILL.md +47 -0
  49. package/skills/btw/SKILL.md +42 -0
  50. package/skills/commit/SKILL.md +29 -0
  51. package/skills/commit/assets/user_instructions_template.md +22 -0
  52. package/skills/continue/SKILL.md +29 -0
  53. package/skills/convert/SKILL.md +124 -0
  54. package/skills/convert/convert_cli.py +235 -0
  55. package/skills/convert/tests/sample.csv +4 -0
  56. package/skills/convert/tests/sample.json +5 -0
  57. package/skills/convert/tests/sample.jsonl +3 -0
  58. package/skills/convert/tests/sample.yaml +18 -0
  59. package/skills/convert/tests/sample_obj.csv +2 -0
  60. package/skills/convert/tests/sample_obj.json +9 -0
  61. package/skills/deep-dive/SKILL.md +167 -0
  62. package/skills/do/SKILL.md +88 -0
  63. package/skills/do/assets/sender_template.json +12 -0
  64. package/skills/doc/SKILL.md +314 -0
  65. package/skills/doc/assets/path-objectives.yaml +38 -0
  66. package/skills/doc-sync/SKILL.md +167 -0
  67. package/skills/doc-sync/assets/default_excludes.txt +21 -0
  68. package/skills/doc-sync/assets/doc_targets.md +14 -0
  69. package/skills/dooo/SKILL.md +60 -0
  70. package/skills/error/SKILL.md +29 -0
  71. package/skills/evaluate/SKILL.md +45 -0
  72. package/skills/evaluate/assets/evaluation_invokation_template.yml +3 -0
  73. package/skills/evaluate/assets/evaluation_rapport_template.md +24 -0
  74. package/skills/examplify/SKILL.md +42 -0
  75. package/skills/help/SKILL.md +36 -0
  76. package/skills/improve/SKILL.md +55 -0
  77. package/skills/index/scripts/board-index +4 -0
  78. package/skills/index/scripts/board_index.py +615 -0
  79. package/skills/index/scripts/smoke_test.sh +86 -0
  80. package/skills/init/SKILL.md +44 -0
  81. package/skills/init/assets/.gitignore_template +15 -0
  82. package/skills/init/assets/PROJECT_SUMMARY_template.md +13 -0
  83. package/skills/init/assets/directory_structure.txt +13 -0
  84. package/skills/init/assets/test-config_template.json +4 -0
  85. package/skills/init/assets/workflow_template.json +30 -0
  86. package/skills/init/scripts/init.sh +48 -0
  87. package/skills/jbp/SKILL.md +25 -0
  88. package/skills/jenga/SKILL.md +68 -0
  89. package/skills/lgtm/SKILL.md +21 -0
  90. package/skills/mirror-public/SKILL.md +237 -0
  91. package/skills/mirror-public/assets/config.json +5 -0
  92. package/skills/mirror-public/scripts/mirror.sh +374 -0
  93. package/skills/pi-plan/SKILL.md +62 -0
  94. package/skills/pi-plan/assets/epic.json +7 -0
  95. package/skills/pi-plan/assets/story_template.md +18 -0
  96. package/skills/proceed/SKILL.md +29 -0
  97. package/skills/publish/SKILL.md +351 -0
  98. package/skills/publish/adapters/droplet.md +200 -0
  99. package/skills/publish/adapters/mobile-ios.md +114 -0
  100. package/skills/publish/adapters/npm-ci.md +223 -0
  101. package/skills/publish/adapters/npm.md +121 -0
  102. package/skills/publish/assets/ExportOptions.plist.template +19 -0
  103. package/skills/publish/assets/ci-contract.md +111 -0
  104. package/skills/publish/assets/ownership-matrix.md +17 -0
  105. package/skills/publish/assets/publish.example.json +85 -0
  106. package/skills/publish/assets/publish.example.npm-ci.json +40 -0
  107. package/skills/publish/assets/publish.example.npm.json +41 -0
  108. package/skills/publish/assets/secrets-guide.md +104 -0
  109. package/skills/publish/schemas/fixtures/npm-ci-minimal.json +17 -0
  110. package/skills/publish/schemas/fixtures/npm-ci-with-empty-secrets.json +18 -0
  111. package/skills/publish/schemas/fixtures/npm-ci-with-workflow-path.json +18 -0
  112. package/skills/publish/schemas/publish.schema.json +428 -0
  113. package/skills/publish/scripts/check_target_config.sh +96 -0
  114. package/skills/publish/scripts/droplet_pipeline.sh +208 -0
  115. package/skills/publish/scripts/generate_release_notes.sh +200 -0
  116. package/skills/publish/scripts/ios_pipeline.sh +486 -0
  117. package/skills/publish/scripts/npm_ci_pipeline.sh +225 -0
  118. package/skills/publish/scripts/npm_pipeline.sh +249 -0
  119. package/skills/publish/scripts/publish_common.sh +253 -0
  120. package/skills/publish/scripts/publish_deploy.sh +538 -0
  121. package/skills/publish/scripts/reconcile_tags.sh +135 -0
  122. package/skills/publish/scripts/run_gates.sh +616 -0
  123. package/skills/publish/scripts/setup_wizard.sh +394 -0
  124. package/skills/publish/scripts/show_history.sh +95 -0
  125. package/skills/publish/scripts/suggest_semver_bump.sh +105 -0
  126. package/skills/publish/scripts/validate_config.sh +163 -0
  127. package/skills/publish/scripts/validate_droplet_env.sh +45 -0
  128. package/skills/publish/scripts/validate_ios_env.sh +68 -0
  129. package/skills/publish/scripts/validate_npm_ci_env.sh +71 -0
  130. package/skills/publish/scripts/validate_npm_env.sh +22 -0
  131. package/skills/publish/scripts/write_ledger_entry.sh +126 -0
  132. package/skills/publish/wizards/droplet.md +275 -0
  133. package/skills/publish/wizards/mobile-ios.md +157 -0
  134. package/skills/publish/wizards/npm-ci.md +240 -0
  135. package/skills/publish/wizards/npm.md +224 -0
  136. package/skills/reconcile/SKILL.md +93 -0
  137. package/skills/reconcile/assets/report_format.md +44 -0
  138. package/skills/reconcile-origin/SKILL.md +75 -0
  139. package/skills/reconcile-origin/scripts/reconcile-origin.sh +372 -0
  140. package/skills/redo/SKILL.md +70 -0
  141. package/skills/route/SKILL.md +180 -0
  142. package/skills/self-sync/SKILL.md +73 -0
  143. package/skills/self-sync/scripts/run.js +136 -0
  144. package/skills/skillify/SKILL.md +68 -0
  145. package/skills/skillify/assets/init-new/SKILL.md +35 -0
  146. package/skills/skillify/assets/init-new/assets/.gitignore_template +15 -0
  147. package/skills/skillify/assets/init-new/assets/PROJECT_SUMMARY_template.md +13 -0
  148. package/skills/skillify/assets/init-new/assets/directory_structure.txt +10 -0
  149. package/skills/skillify/assets/init-new/assets/test-config_template.json +4 -0
  150. package/skills/skillify/assets/init-new/assets/workflow_template.json +17 -0
  151. package/skills/skillify/assets/init-new/scripts/init.sh +48 -0
  152. package/skills/skillify/assets/init-old/SKILL.md +124 -0
  153. package/skills/spinoff/SKILL.md +48 -0
  154. package/skills/status/SKILL.md +33 -0
  155. package/skills/status/assets/output_format.md +41 -0
  156. package/skills/todo/SKILL.md +46 -0
  157. package/skills/todo/assets/todo_handoff_template.md +22 -0
  158. package/skills/todo/assets/todo_template.md +3 -0
  159. package/skills/train/SKILL.md +116 -0
  160. package/skills/train/assets/dashboard-templates/classifiers.html +106 -0
  161. package/skills/train/assets/dashboard-templates/nlp.html +102 -0
  162. package/skills/train/assets/dashboard-templates/transformers.html +98 -0
  163. package/skills/train/assets/results-parsers/__init__.py +9 -0
  164. package/skills/train/assets/results-parsers/classifiers.py +84 -0
  165. package/skills/train/assets/results-parsers/nlp.py +88 -0
  166. package/skills/train/assets/results-parsers/reporter.py +154 -0
  167. package/skills/train/assets/results-parsers/transformers.py +120 -0
  168. package/skills/train/train_cli.py +786 -0
  169. package/templates/EXECUTION_PLAN_TEMPLATE.md +43 -0
  170. package/templates/EXECUTION_SUMMARY_TEMPLATE.md +50 -0
  171. package/templates/JENGA_CONFIG_TEMPLATE.json +23 -0
  172. package/templates/PROBLEM_RAPPORT_TEMPLATE.md +88 -0
  173. package/templates/SCRUM_BOARD_SCHEMA.md +311 -0
  174. package/templates/SKILL.md +16 -0
  175. package/templates/SKILL_TEMPLATE.md +28 -0
  176. package/templates/USER_INSTRUCTIONS_TEMPLATE.md +22 -0
  177. package/templates/copilot-instructions.md.tpl +55 -0
@@ -0,0 +1,3 @@
1
+ # Todo
2
+
3
+ <!-- Format: <mission title>: <E##_S##> (epic/story ref optional) -->
@@ -0,0 +1,116 @@
1
+ ---
2
+ name: train
3
+ description: Scaffold and run ML training jobs. Use `new <type> <job-name>` to scaffold a job from a template, or `run <job-dir>` to execute the two-phase validate → train pipeline on an existing job.
4
+ keywords:
5
+ - train
6
+ - ml training
7
+ - model training
8
+ - machine learning
9
+ - fine-tune
10
+ examples:
11
+ - "scaffold a new training job"
12
+ - "train a classifier model"
13
+ ---
14
+
15
+ # Train — ML Training Job Orchestrator
16
+
17
+ ## Usage
18
+
19
+ ```
20
+ python skills/train/train_cli.py <subcommand> [args]
21
+
22
+ Subcommands:
23
+ new <type> <job-name> Scaffold a new training job from a template
24
+ run <job-dir> Execute validate → train pipeline on an existing job
25
+
26
+ Types: classifiers, transformers, nlp
27
+ ```
28
+
29
+ If an unrecognised subcommand is given, argparse prints the usage message above and
30
+ exits with a non-zero code.
31
+
32
+ ## Subcommands
33
+
34
+ ### `/train new <type> <job-name>`
35
+ Scaffold a new training job from the appropriate template (positional / scriptable mode).
36
+
37
+ **Supported types:** `classifiers`, `transformers`, `nlp`
38
+
39
+ **Optional flags:**
40
+ - `--model <value>` — Override the model name/type in `config.yaml`
41
+ - `--epochs <n>` — Override the number of training epochs/iterations
42
+ - `--batch-size <n>` — Override the batch size
43
+
44
+ **Steps:**
45
+ 1. Validate `<type>` is one of: `classifiers`, `transformers`, `nlp`
46
+ 2. Copy the template directory from `.training/template/<type>/` to `jobs/<job-name>/`
47
+ - If `jobs/<job-name>/` already exists, auto-suffix: `jobs/<job-name>-1/`, `jobs/<job-name>-2/`, etc.
48
+ 3. Apply any CLI flag overrides to the scaffolded `config.yaml`
49
+ 4. Generate `start.sh` in the job directory (executable)
50
+ 5. Read the `workflow:` block from `config.yaml` and print next-step instructions
51
+ 6. Print: `✅ Scaffolded job '<job-name>' from template '<type>' at jobs/<job-name>/`
52
+
53
+ ### `/train new --interactive [--full]`
54
+ Launch a guided wizard that scaffolds the job **and** configures `config.yaml` interactively.
55
+
56
+ **Flags:**
57
+ - `--interactive` / `-i` — Enable the wizard. Job name and type are collected via prompts.
58
+ - `--full` — Expand the wizard to prompt for **every** configurable field (not just the critical ones).
59
+
60
+ **Wizard flow:**
61
+ 1. Prompt for job name (freeform, used as directory name under `jobs/`)
62
+ 2. Prompt for job type (`classifiers`, `transformers`, or `nlp`)
63
+ 3. Scaffold the job directory from the template
64
+ 4. Prompt for critical config fields (2–3 key model params + data path per type):
65
+ - **classifiers**: train file, target column, model algorithm, n_estimators, test split
66
+ - **transformers**: model name, task, train file, epochs, learning rate
67
+ - **nlp**: model name, task, train file, iterations, learning rate
68
+ 5. With `--full`: additionally prompt for every remaining field in `config.yaml`
69
+ 6. Write user responses back into the scaffolded `config.yaml`
70
+
71
+ Each prompt shows the field name, its current default, and an inline description hint.
72
+
73
+ **Ctrl+C handling:**
74
+ Pressing `Ctrl+C` at any point during the wizard interrupts cleanly. If a scaffold directory has already been created, the user is asked:
75
+ ```
76
+ Delete scaffolded directory '<path>'? [y/N]
77
+ ```
78
+ Answering `y` removes the directory; `N` (or Enter) keeps it.
79
+
80
+ ### `/train run <job-dir>`
81
+ Execute the two-phase pre-flight → smoke test pipeline on an existing job directory.
82
+
83
+ **Steps:**
84
+
85
+ #### Phase A — Pre-flight validation
86
+ 1. Print: `[pre-flight] Running validate.py in <job-dir>...`
87
+ 2. Run `validate.py` as a subprocess inside `<job-dir>`:
88
+ ```bash
89
+ cd <job-dir> && python validate.py
90
+ ```
91
+ 3. Stream all stdout/stderr output to the terminal in real time.
92
+ 4. If `validate.py` exits non-zero:
93
+ - Print: `❌ [pre-flight] FAILED — halting before smoke test.`
94
+ - Surface the full error output.
95
+ - **Stop here. Do not proceed to Phase B.**
96
+ 5. If `validate.py` exits zero:
97
+ - Print: `✅ [pre-flight] Passed.`
98
+
99
+ #### Phase B — Smoke test
100
+ 1. Print: `[smoke test] Running train.py --smoke in <job-dir>...`
101
+ 2. Run `train.py --smoke` as a subprocess inside `<job-dir>`:
102
+ ```bash
103
+ cd <job-dir> && python train.py --smoke
104
+ ```
105
+ 3. Stream all stdout/stderr output to the terminal in real time.
106
+ 4. If `train.py` exits non-zero:
107
+ - Print: `❌ [smoke test] FAILED.`
108
+ - Surface the full error output.
109
+ 5. If `train.py` exits zero:
110
+ - Print: `✅ [smoke test] Passed. Job '<job-dir>' completed successfully.`
111
+
112
+ ## Notes
113
+ - `train.py` and `validate.py` must have no awareness of each other — the `/train` skill owns the gate.
114
+ - Both subprocess calls must stream output (not buffer) so the user sees progress in real time.
115
+ - Phase B is **never** invoked if Phase A fails.
116
+ - The `--interactive` wizard requires `pyyaml` (`pip install pyyaml`) to patch `config.yaml`.
@@ -0,0 +1,106 @@
1
+ <!DOCTYPE html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8" />
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
+ <title>Classifier Training Dashboard — {{job_name}}</title>
7
+ <style>
8
+ *, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
9
+ body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
10
+ header { margin-bottom: 2rem; }
11
+ header h1 { font-size: 1.6rem; font-weight: 700; }
12
+ header p { color: #555; margin-top: .25rem; font-size: .9rem; }
13
+ .card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
14
+ .card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
15
+ .metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
16
+ .metric-box { background: #f0f4ff; border-radius: 8px; padding: 1rem; text-align: center; }
17
+ .metric-box .value { font-size: 2rem; font-weight: 700; color: #3b5bdb; }
18
+ .metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
19
+ .bar-chart { display: flex; flex-direction: column; gap: .6rem; }
20
+ .bar-row { display: flex; align-items: center; gap .5rem; }
21
+ .bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
22
+ .bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
23
+ .bar-fill { height: 100%; border-radius: 4px; background: #3b5bdb; transition: width .4s; }
24
+ .bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
25
+ table { width: 100%; border-collapse: collapse; }
26
+ th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
27
+ th { background: #f8f9fa; font-weight: 600; color: #555; }
28
+ footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
29
+ </style>
30
+ </head>
31
+ <body>
32
+
33
+ <header>
34
+ <h1>🤖 Classifier Training Dashboard</h1>
35
+ <p>Job: <strong>{{job_name}}</strong> &nbsp;|&nbsp; Model: <strong>{{model_type}}</strong></p>
36
+ </header>
37
+
38
+ <div class="card">
39
+ <h2>Key Metrics</h2>
40
+ <div class="metrics-grid">
41
+ <div class="metric-box">
42
+ <div class="value">{{accuracy}}</div>
43
+ <div class="label">Accuracy</div>
44
+ </div>
45
+ <div class="metric-box">
46
+ <div class="value">{{f1_score}}</div>
47
+ <div class="label">F1 Score</div>
48
+ </div>
49
+ <div class="metric-box">
50
+ <div class="value">{{precision}}</div>
51
+ <div class="label">Precision</div>
52
+ </div>
53
+ <div class="metric-box">
54
+ <div class="value">{{recall}}</div>
55
+ <div class="label">Recall</div>
56
+ </div>
57
+ </div>
58
+ </div>
59
+
60
+ <div class="card">
61
+ <h2>Metric Comparison</h2>
62
+ <div class="bar-chart">
63
+ <div class="bar-row">
64
+ <span class="bar-label">Accuracy</span>
65
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{accuracy}} * 100%)"></div></div>
66
+ <span class="bar-val">{{accuracy}}</span>
67
+ </div>
68
+ <div class="bar-row">
69
+ <span class="bar-label">F1 Score</span>
70
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{f1_score}} * 100%)"></div></div>
71
+ <span class="bar-val">{{f1_score}}</span>
72
+ </div>
73
+ <div class="bar-row">
74
+ <span class="bar-label">Precision</span>
75
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
76
+ <span class="bar-val">{{precision}}</span>
77
+ </div>
78
+ <div class="bar-row">
79
+ <span class="bar-label">Recall</span>
80
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
81
+ <span class="bar-val">{{recall}}</span>
82
+ </div>
83
+ </div>
84
+ </div>
85
+
86
+ <div class="card">
87
+ <h2>Summary Table</h2>
88
+ <table>
89
+ <thead>
90
+ <tr><th>Metric</th><th>Value</th></tr>
91
+ </thead>
92
+ <tbody>
93
+ <tr><td>Accuracy</td><td>{{accuracy}}</td></tr>
94
+ <tr><td>F1 Score</td><td>{{f1_score}}</td></tr>
95
+ <tr><td>Precision</td><td>{{precision}}</td></tr>
96
+ <tr><td>Recall</td><td>{{recall}}</td></tr>
97
+ <tr><td>Model Type</td><td>{{model_type}}</td></tr>
98
+ <tr><td>Job Name</td><td>{{job_name}}</td></tr>
99
+ </tbody>
100
+ </table>
101
+ </div>
102
+
103
+ <footer>Generated by JengaAgent /train skill &nbsp;|&nbsp; {{job_name}}</footer>
104
+
105
+ </body>
106
+ </html>
@@ -0,0 +1,102 @@
1
+ <!DOCTYPE html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8" />
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
+ <title>NLP Pipeline Dashboard — {{job_name}}</title>
7
+ <style>
8
+ *, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
9
+ body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
10
+ header { margin-bottom: 2rem; }
11
+ header h1 { font-size: 1.6rem; font-weight: 700; }
12
+ header p { color: #555; margin-top: .25rem; font-size: .9rem; }
13
+ .card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
14
+ .card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
15
+ .metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(150px, 1fr)); gap: 1rem; }
16
+ .metric-box { background: #fff8e1; border-radius: 8px; padding: 1rem; text-align: center; }
17
+ .metric-box .value { font-size: 2rem; font-weight: 700; color: #e67700; }
18
+ .metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
19
+ .bar-chart { display: flex; flex-direction: column; gap: .6rem; }
20
+ .bar-row { display: flex; align-items: center; gap: .5rem; }
21
+ .bar-label { width: 100px; font-size: .85rem; color: #444; flex-shrink: 0; }
22
+ .bar-track { flex: 1; background: #e9ecef; border-radius: 4px; height: 20px; position: relative; }
23
+ .bar-fill { height: 100%; border-radius: 4px; background: #e67700; transition: width .4s; }
24
+ .bar-val { width: 50px; text-align: right; font-size: .8rem; color: #444; }
25
+ table { width: 100%; border-collapse: collapse; }
26
+ th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
27
+ th { background: #f8f9fa; font-weight: 600; color: #555; }
28
+ footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
29
+ </style>
30
+ </head>
31
+ <body>
32
+
33
+ <header>
34
+ <h1>🔤 NLP Pipeline Dashboard</h1>
35
+ <p>Job: <strong>{{job_name}}</strong> &nbsp;|&nbsp; Model: <strong>{{model_name}}</strong> &nbsp;|&nbsp; Task: <strong>{{task}}</strong></p>
36
+ </header>
37
+
38
+ <div class="card">
39
+ <h2>Key Metrics</h2>
40
+ <div class="metrics-grid">
41
+ <div class="metric-box">
42
+ <div class="value">{{f1}}</div>
43
+ <div class="label">F1 Score</div>
44
+ </div>
45
+ <div class="metric-box">
46
+ <div class="value">{{precision}}</div>
47
+ <div class="label">Precision</div>
48
+ </div>
49
+ <div class="metric-box">
50
+ <div class="value">{{recall}}</div>
51
+ <div class="label">Recall</div>
52
+ </div>
53
+ <div class="metric-box">
54
+ <div class="value">{{iterations}}</div>
55
+ <div class="label">Iterations</div>
56
+ </div>
57
+ </div>
58
+ </div>
59
+
60
+ <div class="card">
61
+ <h2>Score Breakdown</h2>
62
+ <div class="bar-chart">
63
+ <div class="bar-row">
64
+ <span class="bar-label">F1 Score</span>
65
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{f1}} * 100%)"></div></div>
66
+ <span class="bar-val">{{f1}}</span>
67
+ </div>
68
+ <div class="bar-row">
69
+ <span class="bar-label">Precision</span>
70
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{precision}} * 100%)"></div></div>
71
+ <span class="bar-val">{{precision}}</span>
72
+ </div>
73
+ <div class="bar-row">
74
+ <span class="bar-label">Recall</span>
75
+ <div class="bar-track"><div class="bar-fill" style="width: calc({{recall}} * 100%)"></div></div>
76
+ <span class="bar-val">{{recall}}</span>
77
+ </div>
78
+ </div>
79
+ </div>
80
+
81
+ <div class="card">
82
+ <h2>Summary Table</h2>
83
+ <table>
84
+ <thead>
85
+ <tr><th>Metric</th><th>Value</th></tr>
86
+ </thead>
87
+ <tbody>
88
+ <tr><td>F1 Score</td><td>{{f1}}</td></tr>
89
+ <tr><td>Precision</td><td>{{precision}}</td></tr>
90
+ <tr><td>Recall</td><td>{{recall}}</td></tr>
91
+ <tr><td>Iterations</td><td>{{iterations}}</td></tr>
92
+ <tr><td>Model Name</td><td>{{model_name}}</td></tr>
93
+ <tr><td>Task</td><td>{{task}}</td></tr>
94
+ <tr><td>Job Name</td><td>{{job_name}}</td></tr>
95
+ </tbody>
96
+ </table>
97
+ </div>
98
+
99
+ <footer>Generated by JengaAgent /train skill &nbsp;|&nbsp; {{job_name}}</footer>
100
+
101
+ </body>
102
+ </html>
@@ -0,0 +1,98 @@
1
+ <!DOCTYPE html>
2
+ <html lang="en">
3
+ <head>
4
+ <meta charset="UTF-8" />
5
+ <meta name="viewport" content="width=device-width, initial-scale=1.0" />
6
+ <title>Transformer Fine-Tuning Dashboard — {{job_name}}</title>
7
+ <style>
8
+ *, *::before, *::after { box-sizing: border-box; margin: 0; padding: 0; }
9
+ body { font-family: system-ui, -apple-system, sans-serif; background: #f5f7fa; color: #1a1a2e; padding: 2rem; }
10
+ header { margin-bottom: 2rem; }
11
+ header h1 { font-size: 1.6rem; font-weight: 700; }
12
+ header p { color: #555; margin-top: .25rem; font-size: .9rem; }
13
+ .card { background: #fff; border-radius: 10px; box-shadow: 0 2px 8px rgba(0,0,0,.08); padding: 1.5rem; margin-bottom: 1.5rem; }
14
+ .card h2 { font-size: 1rem; font-weight: 600; color: #333; margin-bottom: 1rem; text-transform: uppercase; letter-spacing: .05em; }
15
+ .metrics-grid { display: grid; grid-template-columns: repeat(auto-fit, minmax(160px, 1fr)); gap: 1rem; }
16
+ .metric-box { background: #f0fff4; border-radius: 8px; padding: 1rem; text-align: center; }
17
+ .metric-box .value { font-size: 2rem; font-weight: 700; color: #2f9e44; }
18
+ .metric-box .label { font-size: .8rem; color: #666; margin-top: .25rem; }
19
+ .loss-section { margin-top: 1rem; }
20
+ .loss-row { display: flex; justify-content: space-between; padding: .5rem 0; border-bottom: 1px solid #eee; font-size: .9rem; }
21
+ .loss-row .loss-label { color: #555; }
22
+ .loss-row .loss-val { font-weight: 600; color: #2f9e44; }
23
+ table { width: 100%; border-collapse: collapse; }
24
+ th, td { padding: .6rem 1rem; text-align: left; font-size: .9rem; border-bottom: 1px solid #eee; }
25
+ th { background: #f8f9fa; font-weight: 600; color: #555; }
26
+ .perplexity-note { font-size: .8rem; color: #777; margin-top: .5rem; }
27
+ footer { text-align: center; font-size: .75rem; color: #999; margin-top: 2rem; }
28
+ </style>
29
+ </head>
30
+ <body>
31
+
32
+ <header>
33
+ <h1>🤗 Transformer Fine-Tuning Dashboard</h1>
34
+ <p>Job: <strong>{{job_name}}</strong> &nbsp;|&nbsp; Model: <strong>{{model_name}}</strong></p>
35
+ </header>
36
+
37
+ <div class="card">
38
+ <h2>Key Metrics</h2>
39
+ <div class="metrics-grid">
40
+ <div class="metric-box">
41
+ <div class="value">{{perplexity}}</div>
42
+ <div class="label">Perplexity</div>
43
+ </div>
44
+ <div class="metric-box">
45
+ <div class="value">{{eval_loss}}</div>
46
+ <div class="label">Eval Loss</div>
47
+ </div>
48
+ <div class="metric-box">
49
+ <div class="value">{{train_loss}}</div>
50
+ <div class="label">Train Loss</div>
51
+ </div>
52
+ <div class="metric-box">
53
+ <div class="value">{{epochs}}</div>
54
+ <div class="label">Epochs</div>
55
+ </div>
56
+ </div>
57
+ <p class="perplexity-note">ℹ️ Lower perplexity = better language model. Derived from eval_loss (e^loss).</p>
58
+ </div>
59
+
60
+ <div class="card">
61
+ <h2>Loss Breakdown</h2>
62
+ <div class="loss-section">
63
+ <div class="loss-row">
64
+ <span class="loss-label">Evaluation Loss</span>
65
+ <span class="loss-val">{{eval_loss}}</span>
66
+ </div>
67
+ <div class="loss-row">
68
+ <span class="loss-label">Training Loss</span>
69
+ <span class="loss-val">{{train_loss}}</span>
70
+ </div>
71
+ <div class="loss-row">
72
+ <span class="loss-label">Perplexity (e^eval_loss)</span>
73
+ <span class="loss-val">{{perplexity}}</span>
74
+ </div>
75
+ </div>
76
+ </div>
77
+
78
+ <div class="card">
79
+ <h2>Summary Table</h2>
80
+ <table>
81
+ <thead>
82
+ <tr><th>Metric</th><th>Value</th></tr>
83
+ </thead>
84
+ <tbody>
85
+ <tr><td>Perplexity</td><td>{{perplexity}}</td></tr>
86
+ <tr><td>Eval Loss</td><td>{{eval_loss}}</td></tr>
87
+ <tr><td>Train Loss</td><td>{{train_loss}}</td></tr>
88
+ <tr><td>Epochs</td><td>{{epochs}}</td></tr>
89
+ <tr><td>Model Name</td><td>{{model_name}}</td></tr>
90
+ <tr><td>Job Name</td><td>{{job_name}}</td></tr>
91
+ </tbody>
92
+ </table>
93
+ </div>
94
+
95
+ <footer>Generated by JengaAgent /train skill &nbsp;|&nbsp; {{job_name}}</footer>
96
+
97
+ </body>
98
+ </html>
@@ -0,0 +1,9 @@
1
+ """
2
+ results-parsers — ML training result extraction utilities.
3
+
4
+ Each submodule exposes: parse(job_dir: Path) -> dict
5
+ Each submodule extracts type-specific metrics from job output artifacts.
6
+ """
7
+ from pathlib import Path
8
+
9
+ PARSERS = ["classifiers", "transformers", "nlp"]
@@ -0,0 +1,84 @@
1
+ """
2
+ classifiers.py — Results parser for sklearn-based classifier jobs.
3
+
4
+ Extracts: accuracy, precision, recall, f1_score, model_type.
5
+ Reads from results.json or scans training-results/ for JSON files.
6
+ """
7
+ import json
8
+ from pathlib import Path
9
+
10
+
11
+ def parse(job_dir: Path) -> dict:
12
+ """
13
+ Parse training results for a classifiers job.
14
+
15
+ Returns a dict with keys:
16
+ accuracy, precision, recall, f1_score, model_type
17
+ Returns empty dict if no results can be found; never raises.
18
+ """
19
+ job_dir = Path(job_dir)
20
+ metrics = {}
21
+
22
+ # 1. Try results.json at root
23
+ results_json = job_dir / "results.json"
24
+ if results_json.exists():
25
+ try:
26
+ data = json.loads(results_json.read_text())
27
+ metrics = _extract_from_dict(data)
28
+ if metrics:
29
+ return metrics
30
+ except Exception:
31
+ pass
32
+
33
+ # 2. Scan training-results/ for any JSON files
34
+ training_results_dir = job_dir / "training-results"
35
+ if training_results_dir.exists():
36
+ for json_file in sorted(training_results_dir.rglob("*.json")):
37
+ try:
38
+ data = json.loads(json_file.read_text())
39
+ metrics = _extract_from_dict(data)
40
+ if metrics:
41
+ return metrics
42
+ except Exception:
43
+ continue
44
+
45
+ return metrics
46
+
47
+
48
+ def _extract_from_dict(data: dict) -> dict:
49
+ """Extract classifier-relevant keys from a parsed dict."""
50
+ metrics = {}
51
+ if not isinstance(data, dict):
52
+ return metrics
53
+
54
+ # Common field names produced by sklearn classification_report / custom scripts
55
+ field_map = {
56
+ "accuracy": ["accuracy", "test_accuracy", "val_accuracy", "acc"],
57
+ "precision": ["precision", "weighted avg.precision", "macro avg.precision"],
58
+ "recall": ["recall", "weighted avg.recall", "macro avg.recall"],
59
+ "f1_score": ["f1_score", "f1", "weighted avg.f1-score", "macro avg.f1-score"],
60
+ "model_type": ["model_type", "model", "algorithm", "classifier"],
61
+ }
62
+
63
+ for target_key, candidates in field_map.items():
64
+ for candidate in candidates:
65
+ # Support dot-path lookup (e.g. "weighted avg.precision")
66
+ value = _deep_get(data, candidate)
67
+ if value is not None:
68
+ metrics[target_key] = value
69
+ break
70
+
71
+ return metrics
72
+
73
+
74
+ def _deep_get(data: dict, dotted_key: str):
75
+ """Retrieve a value from a nested dict using a dot-separated key path."""
76
+ keys = dotted_key.split(".")
77
+ current = data
78
+ for k in keys:
79
+ if not isinstance(current, dict):
80
+ return None
81
+ current = current.get(k)
82
+ if current is None:
83
+ return None
84
+ return current
@@ -0,0 +1,88 @@
1
+ """
2
+ nlp.py — Results parser for spaCy / NLP pipeline training jobs.
3
+
4
+ Extracts: f1, precision, recall, model_name, task, iterations.
5
+ Reads from results.json or scans training-results/ for JSON files.
6
+ """
7
+ import json
8
+ from pathlib import Path
9
+
10
+
11
+ def parse(job_dir: Path) -> dict:
12
+ """
13
+ Parse training results for an nlp job.
14
+
15
+ Returns a dict with keys:
16
+ f1, precision, recall, model_name, task, iterations
17
+ Returns empty dict if no results can be found; never raises.
18
+ """
19
+ job_dir = Path(job_dir)
20
+ metrics = {}
21
+
22
+ # 1. Try results.json at root
23
+ results_json = job_dir / "results.json"
24
+ if results_json.exists():
25
+ try:
26
+ data = json.loads(results_json.read_text())
27
+ metrics = _extract_from_dict(data)
28
+ if metrics:
29
+ return metrics
30
+ except Exception:
31
+ pass
32
+
33
+ # 2. Scan training-results/ for JSON files
34
+ training_results_dir = job_dir / "training-results"
35
+ if training_results_dir.exists():
36
+ for json_file in sorted(training_results_dir.rglob("*.json")):
37
+ try:
38
+ data = json.loads(json_file.read_text())
39
+ metrics = _extract_from_dict(data)
40
+ if metrics:
41
+ return metrics
42
+ except Exception:
43
+ continue
44
+
45
+ # 3. Try spaCy training output (scores.json or metrics.json)
46
+ for scores_file in sorted(job_dir.rglob("scores.json")) + sorted(job_dir.rglob("metrics.json")):
47
+ try:
48
+ data = json.loads(scores_file.read_text())
49
+ metrics = _extract_from_dict(data)
50
+ if metrics:
51
+ return metrics
52
+ except Exception:
53
+ continue
54
+
55
+ return metrics
56
+
57
+
58
+ def _extract_from_dict(data: dict) -> dict:
59
+ """Extract NLP-relevant keys from a parsed dict."""
60
+ metrics = {}
61
+ if not isinstance(data, dict):
62
+ return metrics
63
+
64
+ field_map = {
65
+ "f1": ["f1", "f1_score", "ents_f", "token_f", "tag_f", "sents_f", "score"],
66
+ "precision": ["precision", "ents_p", "token_p", "tag_p"],
67
+ "recall": ["recall", "ents_r", "token_r", "tag_r"],
68
+ "model_name": ["model_name", "model", "base_model"],
69
+ "task": ["task", "pipeline_component", "component"],
70
+ "iterations": ["iterations", "n_iter", "steps", "batches_trained"],
71
+ }
72
+
73
+ for target_key, candidates in field_map.items():
74
+ for candidate in candidates:
75
+ value = data.get(candidate)
76
+ if value is None:
77
+ # Try nested under "scores" or "results" key
78
+ for wrapper in ("scores", "results", "metrics"):
79
+ nested = data.get(wrapper, {})
80
+ if isinstance(nested, dict):
81
+ value = nested.get(candidate)
82
+ if value is not None:
83
+ break
84
+ if value is not None:
85
+ metrics[target_key] = value
86
+ break
87
+
88
+ return metrics