nmt-forge 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (115) hide show
  1. nmt_forge-0.2.0/LICENSE +133 -0
  2. nmt_forge-0.2.0/PKG-INFO +411 -0
  3. nmt_forge-0.2.0/README.md +376 -0
  4. nmt_forge-0.2.0/nmt_forge/__init__.py +16 -0
  5. nmt_forge-0.2.0/nmt_forge/_harness.py +255 -0
  6. nmt_forge-0.2.0/nmt_forge/advisor.py +2419 -0
  7. nmt_forge-0.2.0/nmt_forge/canonical.py +143 -0
  8. nmt_forge-0.2.0/nmt_forge/cards.py +1289 -0
  9. nmt_forge-0.2.0/nmt_forge/cli.py +3150 -0
  10. nmt_forge-0.2.0/nmt_forge/errors.py +179 -0
  11. nmt_forge-0.2.0/nmt_forge/export.py +2020 -0
  12. nmt_forge-0.2.0/nmt_forge/guards/__init__.py +39 -0
  13. nmt_forge-0.2.0/nmt_forge/guards/battery_lint.py +349 -0
  14. nmt_forge-0.2.0/nmt_forge/guards/ci_scoring.py +1730 -0
  15. nmt_forge-0.2.0/nmt_forge/guards/convention_lint.py +130 -0
  16. nmt_forge-0.2.0/nmt_forge/guards/coverage_map.py +141 -0
  17. nmt_forge-0.2.0/nmt_forge/guards/dev_fence.py +207 -0
  18. nmt_forge-0.2.0/nmt_forge/guards/funnel_audit.py +140 -0
  19. nmt_forge-0.2.0/nmt_forge/guards/leak_audit.py +1308 -0
  20. nmt_forge-0.2.0/nmt_forge/guards/preregister.py +883 -0
  21. nmt_forge-0.2.0/nmt_forge/guards/sample_strata.py +136 -0
  22. nmt_forge-0.2.0/nmt_forge/guards/split_guard.py +552 -0
  23. nmt_forge-0.2.0/nmt_forge/harness_bridge.py +645 -0
  24. nmt_forge-0.2.0/nmt_forge/harness_caveats.py +222 -0
  25. nmt_forge-0.2.0/nmt_forge/harness_data.py +78 -0
  26. nmt_forge-0.2.0/nmt_forge/ledger.py +144 -0
  27. nmt_forge-0.2.0/nmt_forge/monitor.py +530 -0
  28. nmt_forge-0.2.0/nmt_forge/plugins.py +180 -0
  29. nmt_forge-0.2.0/nmt_forge/privacy.py +269 -0
  30. nmt_forge-0.2.0/nmt_forge/registry.py +699 -0
  31. nmt_forge-0.2.0/nmt_forge/reporting.py +493 -0
  32. nmt_forge-0.2.0/nmt_forge/runlock.py +258 -0
  33. nmt_forge-0.2.0/nmt_forge/scaffold.py +633 -0
  34. nmt_forge-0.2.0/nmt_forge/scoring_standard.py +274 -0
  35. nmt_forge-0.2.0/nmt_forge/serve.py +529 -0
  36. nmt_forge-0.2.0/nmt_forge/synthesis/__init__.py +6 -0
  37. nmt_forge-0.2.0/nmt_forge/synthesis/analyzer.py +79 -0
  38. nmt_forge-0.2.0/nmt_forge/synthesis/engine.py +215 -0
  39. nmt_forge-0.2.0/nmt_forge/synthesis/filters.py +146 -0
  40. nmt_forge-0.2.0/nmt_forge/synthesis/packs.py +147 -0
  41. nmt_forge-0.2.0/nmt_forge/synthesis/probe.py +72 -0
  42. nmt_forge-0.2.0/nmt_forge/synthesis/run.py +16 -0
  43. nmt_forge-0.2.0/nmt_forge/synthesis/templates.py +110 -0
  44. nmt_forge-0.2.0/nmt_forge/textpipe.py +494 -0
  45. nmt_forge-0.2.0/nmt_forge/training/__init__.py +13 -0
  46. nmt_forge-0.2.0/nmt_forge/training/backends.py +925 -0
  47. nmt_forge-0.2.0/nmt_forge/training/backtranslation.py +93 -0
  48. nmt_forge-0.2.0/nmt_forge/training/config.py +217 -0
  49. nmt_forge-0.2.0/nmt_forge/training/evaluate.py +268 -0
  50. nmt_forge-0.2.0/nmt_forge/training/mix.py +275 -0
  51. nmt_forge-0.2.0/nmt_forge/training/presets.py +156 -0
  52. nmt_forge-0.2.0/nmt_forge/training/run.py +357 -0
  53. nmt_forge-0.2.0/nmt_forge/training/schedule.py +405 -0
  54. nmt_forge-0.2.0/nmt_forge/training/selection.py +172 -0
  55. nmt_forge-0.2.0/nmt_forge/workspace.py +36 -0
  56. nmt_forge-0.2.0/nmt_forge.egg-info/PKG-INFO +411 -0
  57. nmt_forge-0.2.0/nmt_forge.egg-info/SOURCES.txt +113 -0
  58. nmt_forge-0.2.0/nmt_forge.egg-info/dependency_links.txt +1 -0
  59. nmt_forge-0.2.0/nmt_forge.egg-info/entry_points.txt +2 -0
  60. nmt_forge-0.2.0/nmt_forge.egg-info/requires.txt +15 -0
  61. nmt_forge-0.2.0/nmt_forge.egg-info/top_level.txt +1 -0
  62. nmt_forge-0.2.0/pyproject.toml +80 -0
  63. nmt_forge-0.2.0/setup.cfg +4 -0
  64. nmt_forge-0.2.0/tests/test_advisor.py +243 -0
  65. nmt_forge-0.2.0/tests/test_agent_contract.py +534 -0
  66. nmt_forge-0.2.0/tests/test_backtranslation.py +65 -0
  67. nmt_forge-0.2.0/tests/test_battery.py +191 -0
  68. nmt_forge-0.2.0/tests/test_battery_lint.py +180 -0
  69. nmt_forge-0.2.0/tests/test_canonical.py +61 -0
  70. nmt_forge-0.2.0/tests/test_cards.py +625 -0
  71. nmt_forge-0.2.0/tests/test_ci_scoring.py +134 -0
  72. nmt_forge-0.2.0/tests/test_cli.py +340 -0
  73. nmt_forge-0.2.0/tests/test_convention_lint.py +67 -0
  74. nmt_forge-0.2.0/tests/test_coverage_map.py +73 -0
  75. nmt_forge-0.2.0/tests/test_decode_hook.py +79 -0
  76. nmt_forge-0.2.0/tests/test_dev_fence.py +68 -0
  77. nmt_forge-0.2.0/tests/test_evaluate.py +133 -0
  78. nmt_forge-0.2.0/tests/test_export_layout.py +758 -0
  79. nmt_forge-0.2.0/tests/test_funnel_audit.py +69 -0
  80. nmt_forge-0.2.0/tests/test_harness_parity.py +249 -0
  81. nmt_forge-0.2.0/tests/test_leak_audit.py +362 -0
  82. nmt_forge-0.2.0/tests/test_ledger.py +64 -0
  83. nmt_forge-0.2.0/tests/test_lyss_bridge.py +213 -0
  84. nmt_forge-0.2.0/tests/test_monitor.py +157 -0
  85. nmt_forge-0.2.0/tests/test_near_twin_forecast.py +319 -0
  86. nmt_forge-0.2.0/tests/test_novice_simulation.py +125 -0
  87. nmt_forge-0.2.0/tests/test_pack_loading.py +40 -0
  88. nmt_forge-0.2.0/tests/test_prereg_format.py +127 -0
  89. nmt_forge-0.2.0/tests/test_preregister.py +194 -0
  90. nmt_forge-0.2.0/tests/test_private_text.py +495 -0
  91. nmt_forge-0.2.0/tests/test_registry.py +112 -0
  92. nmt_forge-0.2.0/tests/test_registry_tsv.py +24 -0
  93. nmt_forge-0.2.0/tests/test_round10_forge.py +606 -0
  94. nmt_forge-0.2.0/tests/test_round11_forge.py +353 -0
  95. nmt_forge-0.2.0/tests/test_round12_forge.py +588 -0
  96. nmt_forge-0.2.0/tests/test_round13_forge.py +516 -0
  97. nmt_forge-0.2.0/tests/test_round6_forge.py +452 -0
  98. nmt_forge-0.2.0/tests/test_round7_forge.py +615 -0
  99. nmt_forge-0.2.0/tests/test_round8_forge.py +428 -0
  100. nmt_forge-0.2.0/tests/test_round9_forge.py +548 -0
  101. nmt_forge-0.2.0/tests/test_run_lock.py +131 -0
  102. nmt_forge-0.2.0/tests/test_sample_strata.py +82 -0
  103. nmt_forge-0.2.0/tests/test_scaffold.py +139 -0
  104. nmt_forge-0.2.0/tests/test_schedule.py +253 -0
  105. nmt_forge-0.2.0/tests/test_scoring_standard.py +199 -0
  106. nmt_forge-0.2.0/tests/test_serve_content.py +299 -0
  107. nmt_forge-0.2.0/tests/test_split_guard.py +201 -0
  108. nmt_forge-0.2.0/tests/test_synthesis_core.py +161 -0
  109. nmt_forge-0.2.0/tests/test_synthesis_engine.py +184 -0
  110. nmt_forge-0.2.0/tests/test_textpipe.py +232 -0
  111. nmt_forge-0.2.0/tests/test_tiny_cpu_training.py +235 -0
  112. nmt_forge-0.2.0/tests/test_training_config.py +51 -0
  113. nmt_forge-0.2.0/tests/test_training_mix.py +244 -0
  114. nmt_forge-0.2.0/tests/test_training_run.py +178 -0
  115. nmt_forge-0.2.0/tests/test_training_selection.py +72 -0
@@ -0,0 +1,133 @@
1
+ # PolyForm Noncommercial License 1.0.0
2
+
3
+ <https://polyformproject.org/licenses/noncommercial/1.0.0>
4
+
5
+ ## Acceptance
6
+
7
+ In order to get any license under these terms, you must agree
8
+ to them as both strict obligations and conditions to all
9
+ your licenses.
10
+
11
+ ## Copyright License
12
+
13
+ The licensor grants you a copyright license for the
14
+ software to do everything you might do with the software
15
+ that would otherwise infringe the licensor's copyright
16
+ in it for any permitted purpose. However, you may
17
+ only distribute the software according to [Distribution
18
+ License](#distribution-license) and make changes or new works
19
+ based on the software according to [Changes and New Works
20
+ License](#changes-and-new-works-license).
21
+
22
+ ## Distribution License
23
+
24
+ The licensor grants you an additional copyright license
25
+ to distribute copies of the software. Your license
26
+ to distribute covers distributing the software with
27
+ changes and new works permitted by [Changes and New Works
28
+ License](#changes-and-new-works-license).
29
+
30
+ ## Notices
31
+
32
+ You must ensure that anyone who gets a copy of any part of
33
+ the software from you also gets a copy of these terms or the
34
+ URL for them above, as well as copies of any plain-text lines
35
+ beginning with `Required Notice:` that the licensor provided
36
+ with the software. For example:
37
+
38
+ > Required Notice: Copyright Yoyodyne, Inc. (http://example.com)
39
+
40
+ ## Changes and New Works License
41
+
42
+ The licensor grants you an additional copyright license to
43
+ make changes and new works based on the software for any
44
+ permitted purpose.
45
+
46
+ ## Patent License
47
+
48
+ The licensor grants you a patent license for the software that
49
+ covers patent claims the licensor can license, or becomes able
50
+ to license, that you would infringe by using the software.
51
+
52
+ ## Noncommercial Purposes
53
+
54
+ Any noncommercial purpose is a permitted purpose.
55
+
56
+ ## Personal Uses
57
+
58
+ Personal use for research, experiment, and testing for
59
+ the benefit of public knowledge, personal study, private
60
+ entertainment, hobby projects, amateur pursuits, or religious
61
+ observance, without any anticipated commercial application,
62
+ is use for a permitted purpose.
63
+
64
+ ## Noncommercial Organizations
65
+
66
+ Use by any charitable organization, educational institution,
67
+ public research organization, public safety or health
68
+ organization, environmental protection organization,
69
+ or government institution is use for a permitted purpose
70
+ regardless of the source of funding or obligations resulting
71
+ from the funding.
72
+
73
+ ## Fair Use
74
+
75
+ You may have "fair use" rights for the software under the
76
+ law. These terms do not limit them.
77
+
78
+ ## No Other Rights
79
+
80
+ These terms do not allow you to sublicense or transfer any of
81
+ your licenses to anyone else, or prevent the licensor from
82
+ granting licenses to anyone else. These terms do not imply
83
+ any other licenses.
84
+
85
+ ## Patent Defense
86
+
87
+ If you make any written claim that the software infringes or
88
+ contributes to infringement of any patent, your patent license
89
+ for the software granted under these terms ends immediately. If
90
+ your company makes such a claim, your patent license ends
91
+ immediately for work on behalf of your company.
92
+
93
+ ## Violations
94
+
95
+ The first time you are notified in writing that you have
96
+ violated any of these terms, or done anything with the software
97
+ not covered by your licenses, your licenses can nonetheless
98
+ continue if you come into full compliance with these terms,
99
+ and take practical steps to correct past violations, within
100
+ 32 days of receiving notice. Otherwise, all your licenses
101
+ end immediately.
102
+
103
+ ## No Liability
104
+
105
+ ***As far as the law allows, the software comes as is, without
106
+ any warranty or condition, and the licensor will not be liable
107
+ to you for any damages arising out of these terms or the use
108
+ or nature of the software, under any kind of legal claim.***
109
+
110
+ ## Definitions
111
+
112
+ The **licensor** is the individual or entity offering these
113
+ terms, and the **software** is the software the licensor makes
114
+ available under these terms.
115
+
116
+ **You** refers to the individual or entity agreeing to these
117
+ terms.
118
+
119
+ **Your company** is any legal entity, sole proprietorship,
120
+ or other kind of organization that you work for, plus all
121
+ organizations that have control over, are under the control of,
122
+ or are under common control with that organization. **Control**
123
+ means ownership of substantially all the assets of an entity,
124
+ or the power to direct its management and policies by vote,
125
+ contract, or otherwise. Control can be direct or indirect.
126
+
127
+ **Your licenses** are all the licenses granted to you for the
128
+ software under these terms.
129
+
130
+ **Use** means anything you do with the software requiring one
131
+ of your licenses.
132
+
133
+ Required Notice: Copyright Curtis Forbes — Champollion (https://champollion.dev)
@@ -0,0 +1,411 @@
1
+ Metadata-Version: 2.4
2
+ Name: nmt-forge
3
+ Version: 0.2.0
4
+ Summary: NMT training suite that makes catalogued training/eval mistakes structurally hard — group-disjoint splits, dev fencing, leak audits, CIs by default, preregistration — from a CPU-sized first model to an exported, servable one.
5
+ Author: Curtis Forbes
6
+ License-Expression: PolyForm-Noncommercial-1.0.0
7
+ Project-URL: Homepage, https://champollion.dev
8
+ Project-URL: Repository, https://github.com/gamedaysuits/Champollion
9
+ Project-URL: Issues, https://github.com/gamedaysuits/Champollion/issues
10
+ Keywords: machine-translation,nmt,training,low-resource-languages,data-synthesis,evaluation-hygiene
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
18
+ Classifier: Topic :: Text Processing :: Linguistic
19
+ Requires-Python: >=3.11
20
+ Description-Content-Type: text/markdown
21
+ License-File: LICENSE
22
+ Requires-Dist: mt-eval-harness>=0.2.0
23
+ Provides-Extra: hf
24
+ Requires-Dist: torch>=2.0; extra == "hf"
25
+ Requires-Dist: transformers>=4.46; extra == "hf"
26
+ Requires-Dist: accelerate>=1.1.0; extra == "hf"
27
+ Requires-Dist: tokenizers>=0.19; extra == "hf"
28
+ Requires-Dist: sentencepiece>=0.1.99; extra == "hf"
29
+ Requires-Dist: peft>=0.10; extra == "hf"
30
+ Provides-Extra: fst
31
+ Requires-Dist: pyhfst>=1.4; extra == "fst"
32
+ Provides-Extra: dev
33
+ Requires-Dist: pytest>=8.0; extra == "dev"
34
+ Dynamic: license-file
35
+
36
+ # nmt-forge
37
+
38
+ **Train NMT models without fooling yourself.** nmt-forge is a general-purpose
39
+ training suite that makes the classic training/eval mistakes — leaked test
40
+ sets, test-driven checkpoint selection, scores without error bars, structural
41
+ gaps hidden by volume — **structurally hard to commit**. It doesn't warn; it
42
+ refuses, and every refusal says what happened, why it corrupts results, and
43
+ the exact fix.
44
+
45
+ Every guard mechanizes a real, measured failure from Champollion's Plains
46
+ Cree work (the 2026-07-12 mistake ledger). Scoring is delegated entirely to
47
+ [`mt-eval-harness`](../arena) — forge implements zero metrics.
48
+
49
+ ## Install
50
+
51
+ ```bash
52
+ python3 -m pip install nmt-forge # the guards, splits, audits, scoring (via mt-eval-harness)
53
+ python3 -m pip install 'nmt-forge[hf]' # + training, export and serving (torch, transformers,
54
+ # accelerate, sentencepiece, peft — CPU wheels are fine)
55
+ ```
56
+
57
+ `mt-eval-harness` comes in as a dependency: it is forge's scorer AND its
58
+ language-card resolver, so `discover`/`init` work from a plain pip install —
59
+ cards come from `--cards-dir`, `$MT_EVAL_CARDS_DIR`, a checkout or
60
+ `node_modules/champollion` above you, or the public card index (cached,
61
+ reused offline). Offline with no cache:
62
+ `npx champollion network card crk --json > cards/crk.json`, then `--cards-dir cards`.
63
+
64
+ ## From `pip install` to a served model — on a laptop
65
+
66
+ The north star: *"a Cree model for our school"* — ~1,600 parallel pairs and a
67
+ private, teacher-checked test set — or *"an Atya model for the hospital"* —
68
+ ~900 pairs and a sensitive test set. No GPU required:
69
+
70
+ ```bash
71
+ nmt-forge init crk --dir school-mt # card → workspace + config + NEXT_STEPS.md
72
+ cd school-mt # run everything from the project dir
73
+ nmt-forge registry add project-test ~/teacher-test.jsonl --role test # YOUR test set
74
+ nmt-forge leak-audit ~/pairs.jsonl --clean-to pairs.clean.jsonl # screen against it
75
+ # …says most test rows have a near-twin in your pairs? a second, twin-free model:
76
+ # --clean-to pairs.notwins.jsonl --drop-test-twins (its OWN file — it writes
77
+ # config-notwins.json too) — one prereg per model, named after it
78
+ nmt-forge prereg template --out predictions.json # write predictions BEFORE any score —
79
+ nmt-forge prereg new all-data --eval-set project-test --predictions predictions.json
80
+ # …named after the model it predicts (`prereg new notwins …` for the twin-free one),
81
+ # and before any benchmark (mt-eval run) on the test set: a read blocks a later prereg
82
+ # …then read the training guardrails, once, before the split: the "Train a Model
83
+ # Honestly" page (agents: the MCP tool get_training_guardrails)
84
+ nmt-forge split pairs.clean.jsonl --test 0 --dev 100 --seed 42 \
85
+ --out data/split --register project # train/dev only — the test set stays yours
86
+ nmt-forge preflight run --config config.json # the checks run makes, incl. the [hf] extra
87
+ nmt-forge run config.json # cpu-tiny: minutes on a CPU, no download
88
+ nmt-forge export .forge/runs/<run>/run-manifest.json --prereg all-data --out export/
89
+ # …the twin-free model: --prereg notwins --out export-<its run>/ (with two preregs
90
+ # on one test set, export refuses to guess which one predicted which model)
91
+ nmt-forge serve export/model # http://127.0.0.1:8378 — champollion can call it
92
+ ```
93
+
94
+ That order — register the test set, screen the corpus, one preregistration
95
+ per model, (benchmarks), read the guardrails, split, train — is the one `nmt-forge init`,
96
+ `NEXT_STEPS.md` and `nmt-forge status` give. `--prereg <id>` on export names
97
+ the prediction that judges each model. Pinning is the other way:
98
+ `prereg new <id> … --config-hash <hash>` binds a prediction to one config —
99
+ the full hash `nmt-forge preflight run --config <its config>` prints — but any
100
+ later edit of that config (a time budget, say) changes the hash and drops the
101
+ pin, so naming it on export is the simpler route.
102
+
103
+ `export` scores the test set once (prereg-gated, 95% CIs), writes an
104
+ **mt-eval RunLog + TestReport** (built by the harness itself with the metric
105
+ battery `mt-eval run` loads for your language, so `mt-eval compare` works on
106
+ it — anything it could not compute is listed with the command that computes
107
+ it) into `export/evaluation/`, and packages the model into `export/model/`:
108
+ weights, tokenizer, a champollion plugin manifest and a `DEPLOY.md` (with
109
+ the per-pair `fallback` config for strings the model cannot do). Every
110
+ caveat the harness writes on the score — a near-constant output (one of a
111
+ few sentences for many different inputs), length, copies of the source —
112
+ is passed on in its own words beside the score: in the export summary,
113
+ `forge-model.json`, `DEPLOY.md`, `status`, `report`, `compare` and `lint`.
114
+ Scores follow the harness's scoring standard (`standard/1`): the headline —
115
+ the number to quote — is corpus **chrF++ with its 95% bootstrap CI**, written
116
+ `chrF++ 47.5 [45.9, 49.0]`, with its sacreBLEU signature wherever a full record
117
+ is kept (`forge-model.json`, `DEPLOY.md`, the export summary). BLEU, spBLEU,
118
+ TER and COMET are shown beside it, never blended into it; exact match and the
119
+ referee lanes are labelled diagnostics. forge prints no composite and no
120
+ quality label: only a speaker's review certifies quality. `nmt-forge lint`
121
+ takes the battery manifest (`export/evaluation/battery-hyps-battery.json`); given
122
+ a run's `run-manifest.json` it lints every scored export of that run, or tells
123
+ you the `nmt-forge export` command to run first.
124
+ **`export/model/` is what you deploy — it holds no test sentence;
125
+ `export/evaluation/` holds your test set's text and never leaves your
126
+ machine** (marked with the test set's terms when it has any). `serve` speaks the champollion
127
+ **api-method** contract (`POST /translate` — `"method": "api"`) and an
128
+ **OpenAI-compatible** `/v1/chat/completions` (`champollion sync --method
129
+ local`) — app strings and Markdown content files alike. The model never sees
130
+ placeholders, ICU plural/select syntax, tags or Markdown markup: the server
131
+ copies them around the model's output and hands it only the text between
132
+ them, sentence by sentence. `export` prints the verdict on each
133
+ preregistered prediction and how many test sentences have a near-twin in
134
+ training (when most do, it says the score measures recall, not
135
+ translation) — and `split`, `leak-audit` and `preflight run` say the same
136
+ count BEFORE training, while the test set is still unspent, with the fix:
137
+ `split --near-dupe 0.6` holds out whole templates when forge carves the
138
+ test set; for a FIXED test set like the teacher's, `leak-audit <pairs>
139
+ --clean-to <pairs.notwins.jsonl> --drop-test-twins` drops the training rows
140
+ that are near-twins of it (same measure and threshold as the count), says
141
+ how many go and what the strict subset becomes, and refuses to leave
142
+ nothing to train on. It writes to its own file: `leak-audit` refuses to
143
+ replace a file a config, a run or a split reads, or another audit's output
144
+ (`--overwrite` only on purpose). At every step `nmt-forge
145
+ status` names the next command; while a run is training it says `training`
146
+ — wait — and `nmt-forge run` refuses to start a second run in the same
147
+ workspace. With several exported models, the user picks the one to deploy:
148
+ `nmt-forge choose <export>/model` records it (serving a model to try it is
149
+ recorded as served, never as the choice).
150
+
151
+ **Enter it in a sovereign contest.** A contest's declarative lane (Lane A,
152
+ `mt-eval contest submit-model`) takes a model as data: weights, config and
153
+ tokenizer. `export/model/DEPLOY.md` §6 names exactly those files (never
154
+ `forge-model.json` — this model's scores on your test set and local paths —
155
+ nor `DEPLOY.md` or `champollion-plugin/`; `submit-model` leaves all three
156
+ out by itself), the architecture, and the parameter
157
+ count the lane checks, read from the weights file's header, with the
158
+ command. The contest side: [Run a Sovereign
159
+ Contest](https://champollion.dev/docs/network/sovereignty/run-a-sovereign-contest#lane-a--declarative-model-preferred-for-standard-nmt).
160
+
161
+ **A private test set stays private.** Mark it the way a data steward does —
162
+ `champollion network register-corpus --tier local-only --role test`, or a sidecar next to the
163
+ file: `echo '{"transmission": "local-only"}' > teacher-test.tsv.champollion.json`
164
+ — and forge, like `mt-eval`, never prints its sentences. `leak-audit` names
165
+ the rows that matched it by line number (a row that matched a test answer
166
+ IS most of that answer), says once why, and shows the text only with
167
+ `--show-text` — for a person at the terminal; an AI agent passes whatever it
168
+ reads to its model provider. `--json` never carries sentences. The same goes
169
+ for sealed and consent-required corpora (the harness decides which), for a
170
+ training corpus that is itself marked, and for an error message that quotes
171
+ such a row (`[sentence withheld]`). Files forge writes into your folders keep
172
+ the text, and files it carves from a marked corpus (`split` sides, `leak-audit
173
+ --clean-to`, `sample`) get the same mark. When the sidecar names a registered
174
+ corpora card whose sha256 matches the file, the card's id is the set's
175
+ **dataset id** in forge's registry listing, reports, export and mt-eval
176
+ RunLog — the name `mt-eval` runs on the file use — with your `registry add`
177
+ name (`project-test`) kept beside it.
178
+
179
+ **Model presets** (`nmt-forge init --model …`, written out as explicit numbers
180
+ in `config.json`):
181
+
182
+ | preset | what it is | needs | honest expectation |
183
+ |---|---|---|---|
184
+ | `cpu-tiny` (default) | a ~6M-parameter Marian trained from scratch; BPE vocabulary fit on the TRAIN rows only | a CPU, no download — 1,600 pairs train in ~2–3 minutes on a laptop | weak: on 1–2k pairs, chrF++ roughly 5–30 (the top end only for highly templated data) — your data's phrases and templates, not general translation |
185
+ | `cpu-finetune --base <hf-id>` | fine-tune a small pretrained Marian/opus-mt (you name a RELATED pair) | a CPU, ~300 MB download | usually better than cpu-tiny when a related pair exists — measure it, don't assume |
186
+ | `nllb-600m` | NLLB-200 distilled 600M + LoRA | a GPU, ~2.5 GB download | the strongest start; the wall-clock gate refuses it on a CPU in minutes |
187
+
188
+ The point of `cpu-tiny` is not the score: it makes the WHOLE loop real — the
189
+ fence, the audits, the preregistered test, an exportable model the CLI can
190
+ call — so a better model later drops into the same project and is measured
191
+ the same way.
192
+
193
+ ## Sixty seconds, end to end
194
+
195
+ ```bash
196
+ # carve an honest split: pairs sharing a source OR target stay together
197
+ $ nmt-forge split corpus.jsonl --test 150 --dev 42 --seed 42 \
198
+ --out data/split --register textbook
199
+ split corpus.jsonl: 1240 rows in 1187 share-groups (largest 4)
200
+ train 1048 · dev 42 · test 150 → data/split/
201
+ verified: 0 shared canonical source/target keys across sides
202
+
203
+ # train: one command, config-hashed, dev-fenced
204
+ $ nmt-forge run config.json
205
+ dev report (95% CIs — there is no bare-score rendering):
206
+ n=42 · set=textbook-dev
207
+ chrf++ 44.31 [41.20, 47.15] 95% CI
208
+
209
+ # score the test set? not without predictions written down first
210
+ $ nmt-forge score --eval-set textbook-test --hyps decoded.txt
211
+ [preregister] no preregistration for eval set 'textbook-test' ...
212
+ why: results looked at without written-down expectations become post-hoc stories
213
+ fix: write one FIRST: ... — then score
214
+
215
+ $ nmt-forge prereg template --out preds.json # the ONE format; edit it
216
+ $ nmt-forge prereg new e2 --eval-set textbook-test --predictions preds.json
217
+ $ nmt-forge score --eval-set textbook-test --hyps decoded.txt
218
+ chrf++ 46.02 [43.11, 48.87] 95% CI
219
+ ```
220
+
221
+ That's the whole philosophy: the honest path is the easy path, and the
222
+ dishonest paths are closed with actionable messages.
223
+
224
+ **Agents:** every command takes `--json` — exactly one JSON document on
225
+ stdout, refusals as `{"error": {…, "why", "fix"}}` with exit 2. Shapes:
226
+ [docs/JSON_OUTPUT.md](docs/JSON_OUTPUT.md). Call `nmt-forge status --json`
227
+ first, always.
228
+
229
+ ## Any language: start from the card
230
+
231
+ forge is general-purpose across all ~7,900 SSOT language cards. `discover`
232
+ answers "what does this language actually have?" honestly (absence on a card
233
+ = **unknown**, never zero), and `init` scaffolds a project from it:
234
+
235
+ ```bash
236
+ $ nmt-forge discover nav
237
+ Navajo (nav) · ltr
238
+ WHAT THE CARD SAYS EXISTS (absence = unknown, not zero):
239
+ OPUS: 5 corpora, 36533 aligned pairs
240
+ unknown (card is silent): analyzers, dictionaries, eval datasets
241
+ THE ASSET LADDER — what this language can do TODAY:
242
+ ✓ rung 1: parallel text → train with every guard (no pack needed)
243
+ ? rung 2: monolingual text → the tagged backtranslation lane
244
+ ? rung 3: dictionary (+ grammar) → a cited template pack is worth building
245
+ ? rung 4: morphological analyzer → round-trip-VERIFIED synthesis
246
+ ? rung 5: LYSS referee → the language's own metric in selection
247
+ note: no analyzer on the card → synthesis is off the menu until one
248
+ exists; every guard and the training loop work regardless
249
+
250
+ $ nmt-forge init nav --dir my-navajo-mt
251
+ # → .forge/ workspace + starter config.json + NEXT_STEPS.md (the agent brief)
252
+
253
+ # a language the index doesn't have yet? still trainable — nothing invented:
254
+ $ nmt-forge init qzx --no-card --name "My Language" --dir my-mt
255
+ ```
256
+
257
+ A rich card wires more for free: Plains Cree's card carries its eval-dataset
258
+ ids (cross-checked against the mt-eval registry — all flagged NEVER TRAIN ON
259
+ THIS) and its LYSS referee, so `discover crk` emits ready-to-paste
260
+ `--plugin champollion_lyss...` lanes. The referee is an OPTIONAL add-on with
261
+ its own license, never a requirement: `init crk` wires its lanes into the
262
+ starter config only when the package is installed, and otherwise names it in
263
+ NEXT_STEPS.md — the config passes its own preflight on a plain install. Same
264
+ tool, no special cases — the card decides.
265
+
266
+ ## What's inside (start here, dig later)
267
+
268
+ | you want to… | use | it kills the mistake of… |
269
+ |---|---|---|
270
+ | split a corpus | `split` / `verify-split` | test answers hiding in training via shared sources/targets |
271
+ | pick checkpoints | the run's **dev-fence** | the test set choosing the model — and training on the dev set's own rows (a file written before the split, such as an early twin-free corpus, is refused with the re-audit that fixes it) |
272
+ | screen any corpus/harvest | `leak-audit` | training on eval text (exact, reworded, or whole-file) — while KEEPING template siblings ("I see the dog" vs "I see the cat") and similar-prompt/different-answer pairs (an *identical* prompt is dropped whatever its answer), and saying which is which, with examples (by line number only for a private — local-only, sealed, consent-required — set or corpus); `--drop-test-twins` also drops the near-twins of a FIXED test set when they would turn its score into recall |
273
+ | generate training data | `synth` + a language pack | unverified forms, uncited templates, invisible gaps |
274
+ | sample synthetic data | `sample` | two template kinds hogging half the signal |
275
+ | report numbers | `score` / `compare` / `evaluate` | scores with no error bars, no preregistration |
276
+ | hand the model on | `export` / `serve` | a result nobody else can compare (mt-eval TestReport) or deploy (champollion api / OpenAI-compatible endpoint) |
277
+ | see how spent an eval set is | `ledger show` | invisible adaptive use; sealed sets are one-shot |
278
+
279
+ Each guard is also a library call under `nmt_forge.guards.*`. The full
280
+ mistake→mechanism map, with the measured numbers behind each guard, is in
281
+ [DESIGN.md](DESIGN.md).
282
+
283
+ ## Language packs plug in — forge ships none
284
+
285
+ forge is general-purpose; language-specific code lives in the language's own
286
+ home and plugs in through the pack interface (analyzer + dictionary adapter +
287
+ orthography + **grammar-cited** templates + checklist):
288
+
289
+ ```bash
290
+ # from any checkout (no install needed):
291
+ nmt-forge synth nmt_forge_crk.pack:get_pack --out data/synth.jsonl
292
+ # or, once the pack's package is installed (entry point): nmt-forge synth crk
293
+ ```
294
+
295
+ The engine enforces the **emit law** on every pack: every generated word must
296
+ round-trip through the language's analyzer, every closed-class literal must
297
+ be cited, every filter is named and counted, and every row is stamped
298
+ `synthetic: true` — which is exactly why the registry refuses synthetic rows
299
+ in test sets (tests are real data only). The Plains Cree reference pack lives
300
+ in crk-translate (`nmt_forge_crk`); FST models and dictionaries stay
301
+ **separate, user-fetched tools** under their own licenses — never bundled.
302
+
303
+ ## The full harness referee stack — neural metrics included
304
+
305
+ forge speaks everything the eval harness speaks, by delegation: the
306
+ deterministic lanes (chrF++/BLEU/exact-match), the **neural lanes** —
307
+ COMET, COMET-QE, MetricX — and the harness's own plugin discovery (FST
308
+ word-validity, behavioral linters, card-declared metrics):
309
+
310
+ ```bash
311
+ nmt-forge score --eval-set project-test --hyps decoded.txt \
312
+ --metric chrf++ --metric comet --target-lang iku --card-plugins iku
313
+ chrf++ 32.10 [29.4, 34.9] 95% CI
314
+ comet 0.71 [ 0.66, 0.75] 95% CI
315
+ metricx 3.20 [ 2.9, 3.6] 95% CI (lower = better)
316
+ ```
317
+
318
+ Neural inference runs **once**; the bootstrap re-averages cached per-entry
319
+ scores (the harness's own CI pattern). Missing extras report an install fix
320
+ — never a fabricated number — and checkpoint selection *refuses* rather
321
+ than silently switching metrics. Direction is first-class: MetricX's
322
+ lower-is-better rides the score through rendering, selection, and A/B
323
+ winners. And `discover` tells you which lane to **believe**, from the WMT
324
+ meta-evaluations:
325
+
326
+ ```
327
+ $ nmt-forge discover iku
328
+ metric trust (Eskimo-Aleut, WMT meta-eval): comet_score r=0.86, ... bleu r=0.163
329
+ ```
330
+
331
+ — for Inuktitut, BLEU barely tracks human judgment while COMET does; for
332
+ other families it's the reverse; for most low-resource families the honest
333
+ answer is UNMEASURED. forge surfaces that before you select on anything.
334
+
335
+ ## LYSS referees plug in too
336
+
337
+ LYSS eval-standard linters (harness `MetricPlugin` protocol — Plains Cree
338
+ today, more languages later) drop into every scoring surface:
339
+
340
+ ```bash
341
+ nmt-forge score --eval-set textbook-test --hyps decoded.txt \
342
+ --plugin champollion_lyss.crk.metrics:CrkLinterMetric
343
+ chrf++ 46.02 [43.11, 48.87] 95% CI
344
+ crk_linter:equivalent_match_rate 0.31 [ 0.24, 0.38] 95% CI
345
+ crk_linter · variant_class_counts: LONG_VOWEL_MACRON=9, WORD_ORDER=3
346
+ ```
347
+
348
+ Every numeric aggregate a plugin reports gets a bootstrap CI; per-entry
349
+ computation runs once (a thousand resamples never re-run an FST); a plugin
350
+ that says `available: false` is shown unavailable, never fabricated. And
351
+ checkpoint selection can use the language's own referee:
352
+
353
+ ```jsonc
354
+ "selection": {"metric": "generation:crk_linter:equivalent_match_rate",
355
+ "plugins": ["champollion_lyss.crk.metrics:CrkLinterMetric"]}
356
+ ```
357
+
358
+ ## A worked example: the half-epoch death
359
+
360
+ The first CLEAN-protocol Cree run — honest group-disjoint dev, leak-audited
361
+ mix — died at epoch 0.52 of a 115,000-step plan. Mechanism: the mix was
362
+ 97.5% tagged synthetic; early in training the model fits the synthetic
363
+ mass, so dev loss on the 42 *real* dev sentences bottomed at step ~8k and
364
+ drifted upward, and patience-6 declared convergence at half an epoch. Every
365
+ earlier run had hidden this bug by (illegitimately) using the test set as
366
+ dev. **The honest setup is what surfaced it — that's the point of the
367
+ suite.**
368
+
369
+ forge makes the fix the default, not a flag (`schedule-sanity`):
370
+
371
+ ```
372
+ [schedule-sanity] train: regime=synthetic-heavy (auto-detected), floor=38,205 of 114,614 planned steps
373
+ [schedule-sanity] early stopping ASKED to stop at step 14,000 but the
374
+ schedule floor (38,205) held training, because: the mix is 97.5% synthetic
375
+ and the dev set is REAL: early in training the model fits the synthetic
376
+ mass, so dev loss on real sentences bottoms fast and drifts up — that
377
+ pattern is EXPECTED, not convergence …
378
+ ```
379
+
380
+ The floor is **derived** from the config (max of one full pass over the mix
381
+ and 30% of planned steps, capped at 60%) and activates only in the
382
+ `synthetic-heavy` regime — auto-detected from the mix, overridable with one
383
+ word (`"regime": "balanced"`), never ten flags. Every intervention prints
384
+ the dev-loss trajectory and the why; nobody should diagnose this from raw
385
+ logs again.
386
+
387
+ ## Training defaults that encode the ledger
388
+
389
+ Tagged synthetic lanes (Caswell et al. 2019) with gold untagged; gold
390
+ upweighting with the **exposure math written into the manifest**; per-kind
391
+ sampling caps; curriculum stages; a backtranslation lane that leak-audits
392
+ mono text *before* spending translation; generation-headroom checks; and a
393
+ do-not-train gate — datasets the mt-eval registry protects never enter a mix.
394
+ Details and the config schema: [DESIGN.md](DESIGN.md) §6.
395
+
396
+ ## What forge refuses to be
397
+
398
+ Not an evaluator (the harness scores), not a corpus host (manifests are
399
+ content-free — hashes and counts, never text), not a language-card writer,
400
+ not a leaderboard.
401
+
402
+ ## License
403
+
404
+ PolyForm Noncommercial 1.0.0 — source-available, free for noncommercial use;
405
+ using it for a commercial purpose is not covered by this license (relicensed
406
+ from AGPL-3.0-or-later on 2026-08-17, before any release shipped). Who is
407
+ covered, in plain words with examples: [Who may use this](https://champollion.dev/docs/getting-started/who-may-use-this) (a summary,
408
+ not legal advice; [LICENSE](LICENSE) governs). The harness it uses stays
409
+ open-source AGPL-3.0-or-later. Analyzer models and
410
+ dictionaries consumed by packs are upstream artifacts fetched by the user
411
+ under the upstream's terms.