champollion-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +133 -0
- package/README.md +245 -0
- package/bin/server.js +19 -0
- package/instructions.md +234 -0
- package/package.json +50 -0
- package/src/index.js +1106 -0
- package/src/tools/forge.js +140 -0
- package/src/tools/harness.js +727 -0
- package/src/tools/languages.js +329 -0
- package/src/tools/queue.js +313 -0
- package/src/tools/reliability.js +190 -0
- package/src/tools/results.js +346 -0
- package/src/tools/training.js +349 -0
- package/src/tools/translate.js +385 -0
|
@@ -0,0 +1,349 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* get_training_guardrails — how to train an NMT model without fooling
|
|
3
|
+
* yourself, as a tool answer agents discover BEFORE re-inventing the
|
|
4
|
+
* catalogued mistakes.
|
|
5
|
+
*
|
|
6
|
+
* Every rule below is the mechanization of a real, measured failure from
|
|
7
|
+
* Champollion's Plains Cree development work (the 2026-07-12 mistake
|
|
8
|
+
* ledger). The programmatic enforcement lives in the monorepo's `forge/`
|
|
9
|
+
* package (nmt-forge: guards, synthesis engine, fenced training loop); this
|
|
10
|
+
* tool is the discovery surface, so the content is useful even where forge
|
|
11
|
+
* isn't installed.
|
|
12
|
+
*
|
|
13
|
+
* Static, self-contained, no I/O — the guidance IS the data.
|
|
14
|
+
*/
|
|
15
|
+
|
|
16
|
+
const GUARDRAILS = [
|
|
17
|
+
{
|
|
18
|
+
id: 'discovery',
|
|
19
|
+
name: 'Start from the language card (SSOT discovery)',
|
|
20
|
+
rule: 'Before designing anything, read the language\'s card: '
|
|
21
|
+
+ '`nmt-forge discover <iso639-3>` reports what EXISTS (scripts, '
|
|
22
|
+
+ 'analyzers, dictionaries, corpora, eval datasets with '
|
|
23
|
+
+ 'do_not_train/quarantine flags, LYSS referee plugin specs) and where '
|
|
24
|
+
+ 'the language sits on the asset ladder: (1) parallel text → guarded '
|
|
25
|
+
+ 'training; (2) +monolingual → tagged backtranslation; (3) '
|
|
26
|
+
+ '+dictionary+grammar → cited template pack; (4) +morphological '
|
|
27
|
+
+ 'analyzer → round-trip-VERIFIED synthesis; (5) +LYSS referee → the '
|
|
28
|
+
+ 'language\'s own metric in scoring and checkpoint selection. '
|
|
29
|
+
+ 'Absence on a card means UNKNOWN, never zero. `nmt-forge init <code>` '
|
|
30
|
+
+ 'scaffolds a project (workspace + starter config + NEXT_STEPS brief) '
|
|
31
|
+
+ 'from the card. The tool is fully general: a bare card still gets '
|
|
32
|
+
+ 'every guard and the full training loop — only the upper rungs wait '
|
|
33
|
+
+ 'on assets.',
|
|
34
|
+
mistake: 'Designing for one language\'s asset mix and silently assuming '
|
|
35
|
+
+ 'it generalizes — or reading a sparse card as "this language has '
|
|
36
|
+
+ 'nothing" and giving up, when the card is an index that may simply '
|
|
37
|
+
+ 'not record the resource yet.',
|
|
38
|
+
forge: 'nmt-forge discover <code> [--json] / nmt-forge init <code> '
|
|
39
|
+
+ '--pair SRC-TGT; library: nmt_forge.cards.discover().',
|
|
40
|
+
},
|
|
41
|
+
{
|
|
42
|
+
id: 'split',
|
|
43
|
+
name: 'Group-disjoint splits (split-guard)',
|
|
44
|
+
rule: 'Never row-level random-split a corpus with repeated sources or '
|
|
45
|
+
+ 'targets. Union-find every pair sharing a SOURCE or TARGET into one '
|
|
46
|
+
+ 'group; whole groups land on one side; verify zero overlap and '
|
|
47
|
+
+ 'hard-fail otherwise.',
|
|
48
|
+
mistake: 'A textbook mapped many English drills to one Cree word '
|
|
49
|
+
+ '("Feed him"/"Feed her" → asam). Random split put one copy in train, '
|
|
50
|
+
+ 'its twin in test: 17/54 test answers were literally in training '
|
|
51
|
+
+ '(leaked rows scored 83 chrF++ vs 44 clean).',
|
|
52
|
+
forge: 'nmt-forge split … / nmt_forge.guards.split_guard.group_split(); '
|
|
53
|
+
+ 'verify any existing split with verify_disjoint().',
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
id: 'dev-fence',
|
|
57
|
+
name: 'Checkpoint selection never sees the test set (dev-fence)',
|
|
58
|
+
rule: 'Early stopping / best-checkpoint selection runs on a DEV slice '
|
|
59
|
+
+ 'carved from the TRAIN side (also group-disjoint) — never on the '
|
|
60
|
+
+ 'test set. Refuse to train without a dev set.',
|
|
61
|
+
mistake: 'Every run before 2026-07-12 kept the checkpoint that scored '
|
|
62
|
+
+ 'best ON THE TEST SET — like peeking at the final exam after every '
|
|
63
|
+
+ 'study session. Consequence: there was NO valid baseline.',
|
|
64
|
+
forge: 'nmt_forge.guards.dev_fence.DevFence — content-checks dev rows '
|
|
65
|
+
+ 'against every registered test set (catches file-copies too).',
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
id: 'leak-audit',
|
|
69
|
+
name: 'Screen every corpus against every eval set (leak-audit)',
|
|
70
|
+
rule: 'Before training on ANY corpus or harvest (including monolingual '
|
|
71
|
+
+ 'text for backtranslation): exact canonical-key check on BOTH sides, '
|
|
72
|
+
+ 'near-duplicate screen (token Jaccard ≥ 0.6), whole-file identity '
|
|
73
|
+
+ 'check. Keep the audit manifest next to the corpus.',
|
|
74
|
+
mistake: 'A harvested "training" text WAS the gold textbook — the screen '
|
|
75
|
+
+ 'caught 489/489 lines. Two more harvests were partial eval sets '
|
|
76
|
+
+ '(27/27, 46/46). Same-domain documents share reworded lines exact '
|
|
77
|
+
+ 'matching misses.',
|
|
78
|
+
forge: 'nmt-forge leak-audit --strict / nmt_forge.guards.leak_audit.',
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
id: 'funnel',
|
|
82
|
+
name: 'Count every pipeline stage (funnel-audit)',
|
|
83
|
+
rule: 'Decompose data-pipeline yield stage by stage (dictionary → '
|
|
84
|
+
+ 'canonicalized → parsed → emitted) with drop reasons. Canonicalize '
|
|
85
|
+
+ 'orthography AT EVERY ADAPTER BOUNDARY, and check whether dropped '
|
|
86
|
+
+ 'items would survive canonicalized.',
|
|
87
|
+
mistake: 'A one-character orthography mismatch (dictionary ý vs analyzer '
|
|
88
|
+
+ 'y) silently deleted 1,375 verbs from generation for WEEKS. Nobody '
|
|
89
|
+
+ 'was counting.',
|
|
90
|
+
forge: 'nmt_forge.guards.funnel_audit.Funnel + assert_none_recoverable() '
|
|
91
|
+
+ '(the ý-bug detector).',
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
id: 'conventions',
|
|
95
|
+
name: 'One output orthography (convention-lint)',
|
|
96
|
+
rule: 'Canonicalize training targets ONCE at data-build time; train on a '
|
|
97
|
+
+ 'single spelling convention; normalize refs the same way at scoring; '
|
|
98
|
+
+ 'keep a mixed-convention output metric in the battery.',
|
|
99
|
+
mistake: 'Targets deliberately duplicated across four orthographies '
|
|
100
|
+
+ 'taught the model they were interchangeable — it MIXED conventions '
|
|
101
|
+
+ 'within single sentences.',
|
|
102
|
+
forge: 'nmt_forge.guards.convention_lint.assert_single_convention() / '
|
|
103
|
+
+ 'mixed_convention_rate().',
|
|
104
|
+
},
|
|
105
|
+
{
|
|
106
|
+
id: 'coverage',
|
|
107
|
+
name: 'Structural coverage vs a cited grammar checklist (coverage-map)',
|
|
108
|
+
rule: 'Account template kinds against a checklist transcribed from '
|
|
109
|
+
+ 'published grammars (work-level citations). Volume must not hide '
|
|
110
|
+
+ 'structural gaps; required phenomena with zero pairs refuse the build.',
|
|
111
|
+
mistake: 'One million manufactured pairs in ~23 shapes: no imperatives, '
|
|
112
|
+
+ 'no wh-questions, no possession, no inverse — core grammar the '
|
|
113
|
+
+ 'generator COULD produce; the templates never asked.',
|
|
114
|
+
forge: 'nmt_forge.guards.coverage_map; packs ship cited checklists.',
|
|
115
|
+
},
|
|
116
|
+
{
|
|
117
|
+
id: 'strata',
|
|
118
|
+
name: 'Cap per-kind dominance (sample-strata)',
|
|
119
|
+
rule: 'Sample synthetic corpora with a per-kind cap (default 15%) via '
|
|
120
|
+
+ 'capped reservoir; record per-kind seen/kept in a manifest.',
|
|
121
|
+
mistake: 'Two template kinds (conditionals) were 54% of the corpus and a '
|
|
122
|
+
+ 'uniform sample kept that ratio — half the training signal on two '
|
|
123
|
+
+ 'shapes.',
|
|
124
|
+
forge: 'nmt-forge sample / nmt_forge.guards.sample_strata.',
|
|
125
|
+
},
|
|
126
|
+
{
|
|
127
|
+
id: 'ci',
|
|
128
|
+
name: 'Error bars always, full referee stack (ci-scoring)',
|
|
129
|
+
rule: 'Never report a score without its 95% bootstrap CI; use paired '
|
|
130
|
+
+ 'approximate randomization for A/B claims. Delegate ALL metric math '
|
|
131
|
+
+ 'to mt-eval-harness — never re-implement scoring. forge speaks the '
|
|
132
|
+
+ 'harness\'s WHOLE stack: chrF++/BLEU/exact-match, the neural lanes '
|
|
133
|
+
+ '(comet, comet-qe, metricx — inference once, bootstrap over cached '
|
|
134
|
+
+ 'per-entry scores; MetricX is LOWER-is-better and the direction '
|
|
135
|
+
+ 'rides the score through selection and A/B winners; missing extras '
|
|
136
|
+
+ 'report an install fix, never a number), LYSS plugin lanes, and the '
|
|
137
|
+
+ 'harness\'s own plugin discovery (--card-plugins CODE). Before '
|
|
138
|
+
+ 'selecting checkpoints on ANY automatic metric, check metric trust '
|
|
139
|
+
+ '(`nmt-forge discover <code>` shows the WMT meta-eval correlations; '
|
|
140
|
+
+ 'get_metric_reliability serves the same data): for some families '
|
|
141
|
+
+ 'BLEU barely tracks humans while COMET works (Inuktitut r=0.16 vs '
|
|
142
|
+
+ '0.86), and for most low-resource families the honest answer is '
|
|
143
|
+
+ 'UNMEASURED.',
|
|
144
|
+
mistake: '"Oral-story improved 16.7 → 18.1" on 37 sentences was noise '
|
|
145
|
+
+ 'dressed as signal; a bespoke evaluator (battery.py) drifted from '
|
|
146
|
+
+ 'the maintained referee; and selecting on a metric no human judgment '
|
|
147
|
+
+ 'ever validated optimizes the wrong thing invisibly.',
|
|
148
|
+
forge: 'nmt_forge.guards.ci_scoring (wraps mt_eval_harness.confidence / '
|
|
149
|
+
+ '.significance / .metrics_comet / .metrics_metricx); '
|
|
150
|
+
+ '--metric comet --target-lang <code> on score/compare.',
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
id: 'ledger',
|
|
154
|
+
name: 'Adaptive use is visible (eval-ledger)',
|
|
155
|
+
rule: 'Log every read of every registered eval file (purpose-tagged, '
|
|
156
|
+
+ 'config-bound, append-only, hash-chained). Sealed sets are ONE-SHOT: '
|
|
157
|
+
+ 'spent once; a respend needs a loud, ledgered override.',
|
|
158
|
+
mistake: 'Eval pairs quietly guided development decisions, so they were '
|
|
159
|
+
+ 'partly spent as a test — and nothing recorded it.',
|
|
160
|
+
forge: 'nmt-forge ledger show --set <name> for the spend report.',
|
|
161
|
+
},
|
|
162
|
+
{
|
|
163
|
+
id: 'prereg',
|
|
164
|
+
name: 'Predictions before results (preregister)',
|
|
165
|
+
rule: 'Write falsifiable predictions (metric, direction, margin, '
|
|
166
|
+
+ 'rationale) BEFORE scoring a test set; the comparison table refuses '
|
|
167
|
+
+ 'to render without a preregistration that predates the first scoring '
|
|
168
|
+
+ 'read.',
|
|
169
|
+
mistake: 'The "champion" and the findings built on it were selected via '
|
|
170
|
+
+ 'test-set peeking; what made honest recovery possible was the '
|
|
171
|
+
+ 'pre-registration habit — including the predictions that failed.',
|
|
172
|
+
forge: 'nmt-forge prereg new … / nmt_forge.guards.preregister.',
|
|
173
|
+
},
|
|
174
|
+
{
|
|
175
|
+
id: 'synthesis',
|
|
176
|
+
name: 'Verified, cited, provenance-stamped synthetic data',
|
|
177
|
+
rule: 'Every generated word must round-trip through the language\'s '
|
|
178
|
+
+ 'morphological analyzer (generate → analyze → same analysis). Every '
|
|
179
|
+
+ 'template kind cites a published grammar. Plausibility filters are '
|
|
180
|
+
+ 'named and counted. Rows carry champollion-derived provenance and '
|
|
181
|
+
+ 'synthetic: true. Test sets are REAL DATA ONLY — synthetic rows in a '
|
|
182
|
+
+ 'test set are refused at registration.',
|
|
183
|
+
mistake: 'Unverified generation emits garbage silently; uncited '
|
|
184
|
+
+ 'templates drift from the grammar; and testing on synthetic data '
|
|
185
|
+
+ 'measures the generator, not translation.',
|
|
186
|
+
forge: 'nmt_forge.synthesis (engine enforces the emit law); '
|
|
187
|
+
+ 'nmt-forge synth <pack>.',
|
|
188
|
+
},
|
|
189
|
+
{
|
|
190
|
+
id: 'schedule',
|
|
191
|
+
name: 'Derived stopping schedule (schedule-sanity)',
|
|
192
|
+
rule: 'Never let raw early stopping govern a synthetic-heavy run. When '
|
|
193
|
+
+ 'the mix is dominated by tagged synthetic data and the dev set is '
|
|
194
|
+
+ 'real, dev loss bottoming early and drifting up is the model '
|
|
195
|
+
+ 'fitting the synthetic mass — EXPECTED, not convergence. The '
|
|
196
|
+
+ 'stopping floor must be DERIVED from the config (forge: max of one '
|
|
197
|
+
+ 'full pass over the mix and 30% of planned steps, capped at 60%; '
|
|
198
|
+
+ 'auto-activated in the synthetic-heavy regime), never a magic '
|
|
199
|
+
+ '--min-steps flag the user must know. Regimes are named presets '
|
|
200
|
+
+ '("synthetic-heavy", "balanced", or "auto"); every intervention '
|
|
201
|
+
+ 'prints the dev-loss trajectory and the reason in plain language. '
|
|
202
|
+
+ 'Prefer a dev GENERATION metric over loss for checkpoint selection '
|
|
203
|
+
+ 'in this regime.',
|
|
204
|
+
mistake: 'The first CLEAN-protocol crk run (2026-07-12) died at epoch '
|
|
205
|
+
+ '0.52 of 115k steps: 97.5% synthetic mix, honest 42-row real dev, '
|
|
206
|
+
+ 'dev loss bottomed at step ~8k, patience-6 declared convergence. '
|
|
207
|
+
+ 'Invisible in all prior runs because their dev was (illegitimately) '
|
|
208
|
+
+ 'the test set — the honest protocol surfaced the real bug.',
|
|
209
|
+
forge: 'automatic in nmt-forge run (config "regime": "auto"; '
|
|
210
|
+
+ 'nmt_forge.training.schedule.plan_schedule); the run manifest '
|
|
211
|
+
+ 'carries the derived floor, the reason, and any stop/suppress '
|
|
212
|
+
+ 'events with trajectories.',
|
|
213
|
+
},
|
|
214
|
+
{
|
|
215
|
+
id: 'training',
|
|
216
|
+
name: 'Training-loop defaults',
|
|
217
|
+
rule: 'Tag synthetic sources (<synth>, <bt> — Caswell et al. 2019, '
|
|
218
|
+
+ 'adapted); keep gold untagged so it anchors style. Write the gold '
|
|
219
|
+
+ 'exposure math down (upweight × detected augmentation multiplier). '
|
|
220
|
+
+ 'Give decode caps headroom (≥ 1.5× the longest reference in tokens). '
|
|
221
|
+
+ 'Never train on datasets the mt-eval registry marks do_not_train or '
|
|
222
|
+
+ 'quarantined (that is ALL registry benchmark sets). Select '
|
|
223
|
+
+ 'checkpoints on dev loss or a dev GENERATION metric — "eval-loss is '
|
|
224
|
+
+ 'enough" is an open question, not a fact.',
|
|
225
|
+
mistake: 'Silent ×20-is-really-×54 exposure differences broke A/B '
|
|
226
|
+
+ 'fairness; a 160-token cap sat next to a 107-token max reference; '
|
|
227
|
+
+ 'and benchmark sets are one download away from being training data.',
|
|
228
|
+
forge: 'nmt-forge run config.json — all of these are the defaults.',
|
|
229
|
+
},
|
|
230
|
+
];
|
|
231
|
+
|
|
232
|
+
/**
|
|
233
|
+
* The command order a driving agent follows, and the forge_* tool that runs
|
|
234
|
+
* each step. Pairs with the guardrails (the "why") — this is the "how", so
|
|
235
|
+
* the rulebook and the hands reference each other.
|
|
236
|
+
*/
|
|
237
|
+
const COMMAND_ORDER = [
|
|
238
|
+
{ step: 'orient', tool: 'forge_status',
|
|
239
|
+
what: 'WHERE AM I + THE next command. Call first and after every step.' },
|
|
240
|
+
{ step: 'discover', tool: 'forge_discover',
|
|
241
|
+
what: 'what the language has (asset ladder). Absence = unknown, not zero.' },
|
|
242
|
+
{ step: 'scaffold', tool: 'forge_init', what: 'workspace + starter config.' },
|
|
243
|
+
{ step: 'split', tool: 'forge_split',
|
|
244
|
+
what: 'group-disjoint train/dev/test carve.' },
|
|
245
|
+
{ step: 'register', tool: 'forge_register_eval',
|
|
246
|
+
what: 'name dev/test/sealed sets so guards can key off roles.' },
|
|
247
|
+
{ step: 'screen', tool: 'forge_leak_audit',
|
|
248
|
+
what: 'target-side leakage is fatal; source-only near-dupe is kept.' },
|
|
249
|
+
{ step: 'prereg', tool: 'forge_prereg',
|
|
250
|
+
what: 'predictions BEFORE results, or scoring is refused.' },
|
|
251
|
+
{ step: 'preflight', tool: 'forge_preflight',
|
|
252
|
+
what: 'every gate a command will hit, with fixes — before you run it.' },
|
|
253
|
+
{ step: 'train', tool: '(terminal) nmt-forge run config.json',
|
|
254
|
+
what: 'the GPU job — not an MCP tool. Run it in the BACKGROUND with '
|
|
255
|
+
+ 'output to a log; do NOT poll on a timer — watch the log only for '
|
|
256
|
+
+ 'refused|Error|wall-clock|continuity|RUN EXIT (everything else is '
|
|
257
|
+
+ 'progress noise that burns your user\'s budget). A live GUI panel '
|
|
258
|
+
+ 'auto-opens for the human (port 8377) — it is theirs, not yours. '
|
|
259
|
+
+ 'Never invent a time budget: set model.time_budget_hours from what '
|
|
260
|
+
+ 'the user says and let the wall-clock gate refuse what cannot fit.' },
|
|
261
|
+
{ step: 'evaluate', tool: 'forge_evaluate',
|
|
262
|
+
what: 'decode battery with the selected checkpoint + score + diagnose.' },
|
|
263
|
+
{ step: 'diagnose', tool: 'forge_lint',
|
|
264
|
+
what: 'weak registers → likeliest cause → the lever to pull next.' },
|
|
265
|
+
];
|
|
266
|
+
|
|
267
|
+
const TAXONOMY_POINTER =
|
|
268
|
+
'Full failure taxonomy (failure → symptom an agent sees → the guard/lint '
|
|
269
|
+
+ 'that catches it → fix, plus a ranked GAP list): forge/docs/'
|
|
270
|
+
+ 'FAILURE_TAXONOMY.md in the monorepo. The forge_* MCP tools drive each '
|
|
271
|
+
+ 'guard; get_training_guardrails (this tool) is the rulebook behind them.';
|
|
272
|
+
|
|
273
|
+
/**
|
|
274
|
+
* @param {string} [topic] Optional filter: a guardrail id or a substring of
|
|
275
|
+
* its name/rule (case-insensitive).
|
|
276
|
+
* @returns {{status: string, matched: number, guardrails: Array}}
|
|
277
|
+
*/
|
|
278
|
+
export function trainingGuardrails(topic) {
|
|
279
|
+
if (!topic || !String(topic).trim()) {
|
|
280
|
+
return {
|
|
281
|
+
status: 'ok',
|
|
282
|
+
matched: GUARDRAILS.length,
|
|
283
|
+
guardrails: GUARDRAILS,
|
|
284
|
+
command_order: COMMAND_ORDER,
|
|
285
|
+
taxonomy: TAXONOMY_POINTER,
|
|
286
|
+
};
|
|
287
|
+
}
|
|
288
|
+
const q = String(topic).trim().toLowerCase();
|
|
289
|
+
const hits = GUARDRAILS.filter((g) =>
|
|
290
|
+
g.id === q
|
|
291
|
+
|| g.name.toLowerCase().includes(q)
|
|
292
|
+
|| g.rule.toLowerCase().includes(q)
|
|
293
|
+
|| g.mistake.toLowerCase().includes(q));
|
|
294
|
+
return {
|
|
295
|
+
status: hits.length ? 'ok' : 'no-match',
|
|
296
|
+
matched: hits.length,
|
|
297
|
+
topic: q,
|
|
298
|
+
guardrails: hits,
|
|
299
|
+
known_topics: hits.length ? undefined : GUARDRAILS.map((g) => g.id),
|
|
300
|
+
};
|
|
301
|
+
}
|
|
302
|
+
|
|
303
|
+
/**
|
|
304
|
+
* Render a guardrails answer as agent-readable text.
|
|
305
|
+
*
|
|
306
|
+
* @param {{status: string, guardrails: Array, known_topics?: string[]}} answer
|
|
307
|
+
* @returns {string}
|
|
308
|
+
*/
|
|
309
|
+
export function formatTrainingGuardrails(answer) {
|
|
310
|
+
if (answer.status === 'no-match') {
|
|
311
|
+
return `No guardrail matches "${answer.topic}". Known topics: `
|
|
312
|
+
+ `${answer.known_topics.join(', ')}. Call without a topic for all.`;
|
|
313
|
+
}
|
|
314
|
+
const parts = [
|
|
315
|
+
'NMT training guardrails — each rule mechanizes a real, measured mistake',
|
|
316
|
+
'(crk-translate ledger, 2026-07-12). Enforcement: the forge/ package',
|
|
317
|
+
'(nmt-forge) in the Champollion monorepo.',
|
|
318
|
+
'',
|
|
319
|
+
];
|
|
320
|
+
for (const g of answer.guardrails) {
|
|
321
|
+
parts.push(`## ${g.name}`);
|
|
322
|
+
parts.push(`RULE: ${g.rule}`);
|
|
323
|
+
parts.push(`THE MISTAKE IT KILLS: ${g.mistake}`);
|
|
324
|
+
parts.push(`TOOLING: ${g.forge}`);
|
|
325
|
+
parts.push('');
|
|
326
|
+
}
|
|
327
|
+
if (answer.command_order) {
|
|
328
|
+
parts.push('# The loop — call forge_status first, then follow its '
|
|
329
|
+
+ 'next_command; the forge_* tools run each guard:');
|
|
330
|
+
for (const c of answer.command_order) {
|
|
331
|
+
parts.push(` ${c.step.padEnd(10)} ${c.tool} — ${c.what}`);
|
|
332
|
+
}
|
|
333
|
+
parts.push('');
|
|
334
|
+
}
|
|
335
|
+
if (answer.taxonomy) {
|
|
336
|
+
parts.push(answer.taxonomy);
|
|
337
|
+
}
|
|
338
|
+
parts.push('DOCS (human-readable — share these with the user driving you):');
|
|
339
|
+
parts.push(
|
|
340
|
+
'- Vocabulary, zero background: '
|
|
341
|
+
+ 'https://champollion.dev/docs/network/context/mt-training-concepts');
|
|
342
|
+
parts.push(
|
|
343
|
+
'- Step-by-step, agent-forward walkthrough: '
|
|
344
|
+
+ 'https://champollion.dev/docs/network/tutorials/train-your-own-model');
|
|
345
|
+
parts.push(
|
|
346
|
+
'- The guardrails on one page: '
|
|
347
|
+
+ 'https://champollion.dev/docs/network/getting-started/training-honestly');
|
|
348
|
+
return parts.join('\n').trim();
|
|
349
|
+
}
|