nmt-forge 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nmt_forge-0.2.0/LICENSE +133 -0
- nmt_forge-0.2.0/PKG-INFO +411 -0
- nmt_forge-0.2.0/README.md +376 -0
- nmt_forge-0.2.0/nmt_forge/__init__.py +16 -0
- nmt_forge-0.2.0/nmt_forge/_harness.py +255 -0
- nmt_forge-0.2.0/nmt_forge/advisor.py +2419 -0
- nmt_forge-0.2.0/nmt_forge/canonical.py +143 -0
- nmt_forge-0.2.0/nmt_forge/cards.py +1289 -0
- nmt_forge-0.2.0/nmt_forge/cli.py +3150 -0
- nmt_forge-0.2.0/nmt_forge/errors.py +179 -0
- nmt_forge-0.2.0/nmt_forge/export.py +2020 -0
- nmt_forge-0.2.0/nmt_forge/guards/__init__.py +39 -0
- nmt_forge-0.2.0/nmt_forge/guards/battery_lint.py +349 -0
- nmt_forge-0.2.0/nmt_forge/guards/ci_scoring.py +1730 -0
- nmt_forge-0.2.0/nmt_forge/guards/convention_lint.py +130 -0
- nmt_forge-0.2.0/nmt_forge/guards/coverage_map.py +141 -0
- nmt_forge-0.2.0/nmt_forge/guards/dev_fence.py +207 -0
- nmt_forge-0.2.0/nmt_forge/guards/funnel_audit.py +140 -0
- nmt_forge-0.2.0/nmt_forge/guards/leak_audit.py +1308 -0
- nmt_forge-0.2.0/nmt_forge/guards/preregister.py +883 -0
- nmt_forge-0.2.0/nmt_forge/guards/sample_strata.py +136 -0
- nmt_forge-0.2.0/nmt_forge/guards/split_guard.py +552 -0
- nmt_forge-0.2.0/nmt_forge/harness_bridge.py +645 -0
- nmt_forge-0.2.0/nmt_forge/harness_caveats.py +222 -0
- nmt_forge-0.2.0/nmt_forge/harness_data.py +78 -0
- nmt_forge-0.2.0/nmt_forge/ledger.py +144 -0
- nmt_forge-0.2.0/nmt_forge/monitor.py +530 -0
- nmt_forge-0.2.0/nmt_forge/plugins.py +180 -0
- nmt_forge-0.2.0/nmt_forge/privacy.py +269 -0
- nmt_forge-0.2.0/nmt_forge/registry.py +699 -0
- nmt_forge-0.2.0/nmt_forge/reporting.py +493 -0
- nmt_forge-0.2.0/nmt_forge/runlock.py +258 -0
- nmt_forge-0.2.0/nmt_forge/scaffold.py +633 -0
- nmt_forge-0.2.0/nmt_forge/scoring_standard.py +274 -0
- nmt_forge-0.2.0/nmt_forge/serve.py +529 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/__init__.py +6 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/analyzer.py +79 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/engine.py +215 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/filters.py +146 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/packs.py +147 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/probe.py +72 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/run.py +16 -0
- nmt_forge-0.2.0/nmt_forge/synthesis/templates.py +110 -0
- nmt_forge-0.2.0/nmt_forge/textpipe.py +494 -0
- nmt_forge-0.2.0/nmt_forge/training/__init__.py +13 -0
- nmt_forge-0.2.0/nmt_forge/training/backends.py +925 -0
- nmt_forge-0.2.0/nmt_forge/training/backtranslation.py +93 -0
- nmt_forge-0.2.0/nmt_forge/training/config.py +217 -0
- nmt_forge-0.2.0/nmt_forge/training/evaluate.py +268 -0
- nmt_forge-0.2.0/nmt_forge/training/mix.py +275 -0
- nmt_forge-0.2.0/nmt_forge/training/presets.py +156 -0
- nmt_forge-0.2.0/nmt_forge/training/run.py +357 -0
- nmt_forge-0.2.0/nmt_forge/training/schedule.py +405 -0
- nmt_forge-0.2.0/nmt_forge/training/selection.py +172 -0
- nmt_forge-0.2.0/nmt_forge/workspace.py +36 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/PKG-INFO +411 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/SOURCES.txt +113 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/dependency_links.txt +1 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/entry_points.txt +2 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/requires.txt +15 -0
- nmt_forge-0.2.0/nmt_forge.egg-info/top_level.txt +1 -0
- nmt_forge-0.2.0/pyproject.toml +80 -0
- nmt_forge-0.2.0/setup.cfg +4 -0
- nmt_forge-0.2.0/tests/test_advisor.py +243 -0
- nmt_forge-0.2.0/tests/test_agent_contract.py +534 -0
- nmt_forge-0.2.0/tests/test_backtranslation.py +65 -0
- nmt_forge-0.2.0/tests/test_battery.py +191 -0
- nmt_forge-0.2.0/tests/test_battery_lint.py +180 -0
- nmt_forge-0.2.0/tests/test_canonical.py +61 -0
- nmt_forge-0.2.0/tests/test_cards.py +625 -0
- nmt_forge-0.2.0/tests/test_ci_scoring.py +134 -0
- nmt_forge-0.2.0/tests/test_cli.py +340 -0
- nmt_forge-0.2.0/tests/test_convention_lint.py +67 -0
- nmt_forge-0.2.0/tests/test_coverage_map.py +73 -0
- nmt_forge-0.2.0/tests/test_decode_hook.py +79 -0
- nmt_forge-0.2.0/tests/test_dev_fence.py +68 -0
- nmt_forge-0.2.0/tests/test_evaluate.py +133 -0
- nmt_forge-0.2.0/tests/test_export_layout.py +758 -0
- nmt_forge-0.2.0/tests/test_funnel_audit.py +69 -0
- nmt_forge-0.2.0/tests/test_harness_parity.py +249 -0
- nmt_forge-0.2.0/tests/test_leak_audit.py +362 -0
- nmt_forge-0.2.0/tests/test_ledger.py +64 -0
- nmt_forge-0.2.0/tests/test_lyss_bridge.py +213 -0
- nmt_forge-0.2.0/tests/test_monitor.py +157 -0
- nmt_forge-0.2.0/tests/test_near_twin_forecast.py +319 -0
- nmt_forge-0.2.0/tests/test_novice_simulation.py +125 -0
- nmt_forge-0.2.0/tests/test_pack_loading.py +40 -0
- nmt_forge-0.2.0/tests/test_prereg_format.py +127 -0
- nmt_forge-0.2.0/tests/test_preregister.py +194 -0
- nmt_forge-0.2.0/tests/test_private_text.py +495 -0
- nmt_forge-0.2.0/tests/test_registry.py +112 -0
- nmt_forge-0.2.0/tests/test_registry_tsv.py +24 -0
- nmt_forge-0.2.0/tests/test_round10_forge.py +606 -0
- nmt_forge-0.2.0/tests/test_round11_forge.py +353 -0
- nmt_forge-0.2.0/tests/test_round12_forge.py +588 -0
- nmt_forge-0.2.0/tests/test_round13_forge.py +516 -0
- nmt_forge-0.2.0/tests/test_round6_forge.py +452 -0
- nmt_forge-0.2.0/tests/test_round7_forge.py +615 -0
- nmt_forge-0.2.0/tests/test_round8_forge.py +428 -0
- nmt_forge-0.2.0/tests/test_round9_forge.py +548 -0
- nmt_forge-0.2.0/tests/test_run_lock.py +131 -0
- nmt_forge-0.2.0/tests/test_sample_strata.py +82 -0
- nmt_forge-0.2.0/tests/test_scaffold.py +139 -0
- nmt_forge-0.2.0/tests/test_schedule.py +253 -0
- nmt_forge-0.2.0/tests/test_scoring_standard.py +199 -0
- nmt_forge-0.2.0/tests/test_serve_content.py +299 -0
- nmt_forge-0.2.0/tests/test_split_guard.py +201 -0
- nmt_forge-0.2.0/tests/test_synthesis_core.py +161 -0
- nmt_forge-0.2.0/tests/test_synthesis_engine.py +184 -0
- nmt_forge-0.2.0/tests/test_textpipe.py +232 -0
- nmt_forge-0.2.0/tests/test_tiny_cpu_training.py +235 -0
- nmt_forge-0.2.0/tests/test_training_config.py +51 -0
- nmt_forge-0.2.0/tests/test_training_mix.py +244 -0
- nmt_forge-0.2.0/tests/test_training_run.py +178 -0
- nmt_forge-0.2.0/tests/test_training_selection.py +72 -0
nmt_forge-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
# PolyForm Noncommercial License 1.0.0
|
|
2
|
+
|
|
3
|
+
<https://polyformproject.org/licenses/noncommercial/1.0.0>
|
|
4
|
+
|
|
5
|
+
## Acceptance
|
|
6
|
+
|
|
7
|
+
In order to get any license under these terms, you must agree
|
|
8
|
+
to them as both strict obligations and conditions to all
|
|
9
|
+
your licenses.
|
|
10
|
+
|
|
11
|
+
## Copyright License
|
|
12
|
+
|
|
13
|
+
The licensor grants you a copyright license for the
|
|
14
|
+
software to do everything you might do with the software
|
|
15
|
+
that would otherwise infringe the licensor's copyright
|
|
16
|
+
in it for any permitted purpose. However, you may
|
|
17
|
+
only distribute the software according to [Distribution
|
|
18
|
+
License](#distribution-license) and make changes or new works
|
|
19
|
+
based on the software according to [Changes and New Works
|
|
20
|
+
License](#changes-and-new-works-license).
|
|
21
|
+
|
|
22
|
+
## Distribution License
|
|
23
|
+
|
|
24
|
+
The licensor grants you an additional copyright license
|
|
25
|
+
to distribute copies of the software. Your license
|
|
26
|
+
to distribute covers distributing the software with
|
|
27
|
+
changes and new works permitted by [Changes and New Works
|
|
28
|
+
License](#changes-and-new-works-license).
|
|
29
|
+
|
|
30
|
+
## Notices
|
|
31
|
+
|
|
32
|
+
You must ensure that anyone who gets a copy of any part of
|
|
33
|
+
the software from you also gets a copy of these terms or the
|
|
34
|
+
URL for them above, as well as copies of any plain-text lines
|
|
35
|
+
beginning with `Required Notice:` that the licensor provided
|
|
36
|
+
with the software. For example:
|
|
37
|
+
|
|
38
|
+
> Required Notice: Copyright Yoyodyne, Inc. (http://example.com)
|
|
39
|
+
|
|
40
|
+
## Changes and New Works License
|
|
41
|
+
|
|
42
|
+
The licensor grants you an additional copyright license to
|
|
43
|
+
make changes and new works based on the software for any
|
|
44
|
+
permitted purpose.
|
|
45
|
+
|
|
46
|
+
## Patent License
|
|
47
|
+
|
|
48
|
+
The licensor grants you a patent license for the software that
|
|
49
|
+
covers patent claims the licensor can license, or becomes able
|
|
50
|
+
to license, that you would infringe by using the software.
|
|
51
|
+
|
|
52
|
+
## Noncommercial Purposes
|
|
53
|
+
|
|
54
|
+
Any noncommercial purpose is a permitted purpose.
|
|
55
|
+
|
|
56
|
+
## Personal Uses
|
|
57
|
+
|
|
58
|
+
Personal use for research, experiment, and testing for
|
|
59
|
+
the benefit of public knowledge, personal study, private
|
|
60
|
+
entertainment, hobby projects, amateur pursuits, or religious
|
|
61
|
+
observance, without any anticipated commercial application,
|
|
62
|
+
is use for a permitted purpose.
|
|
63
|
+
|
|
64
|
+
## Noncommercial Organizations
|
|
65
|
+
|
|
66
|
+
Use by any charitable organization, educational institution,
|
|
67
|
+
public research organization, public safety or health
|
|
68
|
+
organization, environmental protection organization,
|
|
69
|
+
or government institution is use for a permitted purpose
|
|
70
|
+
regardless of the source of funding or obligations resulting
|
|
71
|
+
from the funding.
|
|
72
|
+
|
|
73
|
+
## Fair Use
|
|
74
|
+
|
|
75
|
+
You may have "fair use" rights for the software under the
|
|
76
|
+
law. These terms do not limit them.
|
|
77
|
+
|
|
78
|
+
## No Other Rights
|
|
79
|
+
|
|
80
|
+
These terms do not allow you to sublicense or transfer any of
|
|
81
|
+
your licenses to anyone else, or prevent the licensor from
|
|
82
|
+
granting licenses to anyone else. These terms do not imply
|
|
83
|
+
any other licenses.
|
|
84
|
+
|
|
85
|
+
## Patent Defense
|
|
86
|
+
|
|
87
|
+
If you make any written claim that the software infringes or
|
|
88
|
+
contributes to infringement of any patent, your patent license
|
|
89
|
+
for the software granted under these terms ends immediately. If
|
|
90
|
+
your company makes such a claim, your patent license ends
|
|
91
|
+
immediately for work on behalf of your company.
|
|
92
|
+
|
|
93
|
+
## Violations
|
|
94
|
+
|
|
95
|
+
The first time you are notified in writing that you have
|
|
96
|
+
violated any of these terms, or done anything with the software
|
|
97
|
+
not covered by your licenses, your licenses can nonetheless
|
|
98
|
+
continue if you come into full compliance with these terms,
|
|
99
|
+
and take practical steps to correct past violations, within
|
|
100
|
+
32 days of receiving notice. Otherwise, all your licenses
|
|
101
|
+
end immediately.
|
|
102
|
+
|
|
103
|
+
## No Liability
|
|
104
|
+
|
|
105
|
+
***As far as the law allows, the software comes as is, without
|
|
106
|
+
any warranty or condition, and the licensor will not be liable
|
|
107
|
+
to you for any damages arising out of these terms or the use
|
|
108
|
+
or nature of the software, under any kind of legal claim.***
|
|
109
|
+
|
|
110
|
+
## Definitions
|
|
111
|
+
|
|
112
|
+
The **licensor** is the individual or entity offering these
|
|
113
|
+
terms, and the **software** is the software the licensor makes
|
|
114
|
+
available under these terms.
|
|
115
|
+
|
|
116
|
+
**You** refers to the individual or entity agreeing to these
|
|
117
|
+
terms.
|
|
118
|
+
|
|
119
|
+
**Your company** is any legal entity, sole proprietorship,
|
|
120
|
+
or other kind of organization that you work for, plus all
|
|
121
|
+
organizations that have control over, are under the control of,
|
|
122
|
+
or are under common control with that organization. **Control**
|
|
123
|
+
means ownership of substantially all the assets of an entity,
|
|
124
|
+
or the power to direct its management and policies by vote,
|
|
125
|
+
contract, or otherwise. Control can be direct or indirect.
|
|
126
|
+
|
|
127
|
+
**Your licenses** are all the licenses granted to you for the
|
|
128
|
+
software under these terms.
|
|
129
|
+
|
|
130
|
+
**Use** means anything you do with the software requiring one
|
|
131
|
+
of your licenses.
|
|
132
|
+
|
|
133
|
+
Required Notice: Copyright Curtis Forbes — Champollion (https://champollion.dev)
|
nmt_forge-0.2.0/PKG-INFO
ADDED
|
@@ -0,0 +1,411 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: nmt-forge
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: NMT training suite that makes catalogued training/eval mistakes structurally hard — group-disjoint splits, dev fencing, leak audits, CIs by default, preregistration — from a CPU-sized first model to an exported, servable one.
|
|
5
|
+
Author: Curtis Forbes
|
|
6
|
+
License-Expression: PolyForm-Noncommercial-1.0.0
|
|
7
|
+
Project-URL: Homepage, https://champollion.dev
|
|
8
|
+
Project-URL: Repository, https://github.com/gamedaysuits/Champollion
|
|
9
|
+
Project-URL: Issues, https://github.com/gamedaysuits/Champollion/issues
|
|
10
|
+
Keywords: machine-translation,nmt,training,low-resource-languages,data-synthesis,evaluation-hygiene
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
License-File: LICENSE
|
|
22
|
+
Requires-Dist: mt-eval-harness>=0.2.0
|
|
23
|
+
Provides-Extra: hf
|
|
24
|
+
Requires-Dist: torch>=2.0; extra == "hf"
|
|
25
|
+
Requires-Dist: transformers>=4.46; extra == "hf"
|
|
26
|
+
Requires-Dist: accelerate>=1.1.0; extra == "hf"
|
|
27
|
+
Requires-Dist: tokenizers>=0.19; extra == "hf"
|
|
28
|
+
Requires-Dist: sentencepiece>=0.1.99; extra == "hf"
|
|
29
|
+
Requires-Dist: peft>=0.10; extra == "hf"
|
|
30
|
+
Provides-Extra: fst
|
|
31
|
+
Requires-Dist: pyhfst>=1.4; extra == "fst"
|
|
32
|
+
Provides-Extra: dev
|
|
33
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
34
|
+
Dynamic: license-file
|
|
35
|
+
|
|
36
|
+
# nmt-forge
|
|
37
|
+
|
|
38
|
+
**Train NMT models without fooling yourself.** nmt-forge is a general-purpose
|
|
39
|
+
training suite that makes the classic training/eval mistakes — leaked test
|
|
40
|
+
sets, test-driven checkpoint selection, scores without error bars, structural
|
|
41
|
+
gaps hidden by volume — **structurally hard to commit**. It doesn't warn; it
|
|
42
|
+
refuses, and every refusal says what happened, why it corrupts results, and
|
|
43
|
+
the exact fix.
|
|
44
|
+
|
|
45
|
+
Every guard mechanizes a real, measured failure from Champollion's Plains
|
|
46
|
+
Cree work (the 2026-07-12 mistake ledger). Scoring is delegated entirely to
|
|
47
|
+
[`mt-eval-harness`](../arena) — forge implements zero metrics.
|
|
48
|
+
|
|
49
|
+
## Install
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
python3 -m pip install nmt-forge # the guards, splits, audits, scoring (via mt-eval-harness)
|
|
53
|
+
python3 -m pip install 'nmt-forge[hf]' # + training, export and serving (torch, transformers,
|
|
54
|
+
# accelerate, sentencepiece, peft — CPU wheels are fine)
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
`mt-eval-harness` comes in as a dependency: it is forge's scorer AND its
|
|
58
|
+
language-card resolver, so `discover`/`init` work from a plain pip install —
|
|
59
|
+
cards come from `--cards-dir`, `$MT_EVAL_CARDS_DIR`, a checkout or
|
|
60
|
+
`node_modules/champollion` above you, or the public card index (cached,
|
|
61
|
+
reused offline). Offline with no cache:
|
|
62
|
+
`npx champollion network card crk --json > cards/crk.json`, then `--cards-dir cards`.
|
|
63
|
+
|
|
64
|
+
## From `pip install` to a served model — on a laptop
|
|
65
|
+
|
|
66
|
+
The north star: *"a Cree model for our school"* — ~1,600 parallel pairs and a
|
|
67
|
+
private, teacher-checked test set — or *"an Atya model for the hospital"* —
|
|
68
|
+
~900 pairs and a sensitive test set. No GPU required:
|
|
69
|
+
|
|
70
|
+
```bash
|
|
71
|
+
nmt-forge init crk --dir school-mt # card → workspace + config + NEXT_STEPS.md
|
|
72
|
+
cd school-mt # run everything from the project dir
|
|
73
|
+
nmt-forge registry add project-test ~/teacher-test.jsonl --role test # YOUR test set
|
|
74
|
+
nmt-forge leak-audit ~/pairs.jsonl --clean-to pairs.clean.jsonl # screen against it
|
|
75
|
+
# …says most test rows have a near-twin in your pairs? a second, twin-free model:
|
|
76
|
+
# --clean-to pairs.notwins.jsonl --drop-test-twins (its OWN file — it writes
|
|
77
|
+
# config-notwins.json too) — one prereg per model, named after it
|
|
78
|
+
nmt-forge prereg template --out predictions.json # write predictions BEFORE any score —
|
|
79
|
+
nmt-forge prereg new all-data --eval-set project-test --predictions predictions.json
|
|
80
|
+
# …named after the model it predicts (`prereg new notwins …` for the twin-free one),
|
|
81
|
+
# and before any benchmark (mt-eval run) on the test set: a read blocks a later prereg
|
|
82
|
+
# …then read the training guardrails, once, before the split: the "Train a Model
|
|
83
|
+
# Honestly" page (agents: the MCP tool get_training_guardrails)
|
|
84
|
+
nmt-forge split pairs.clean.jsonl --test 0 --dev 100 --seed 42 \
|
|
85
|
+
--out data/split --register project # train/dev only — the test set stays yours
|
|
86
|
+
nmt-forge preflight run --config config.json # the checks run makes, incl. the [hf] extra
|
|
87
|
+
nmt-forge run config.json # cpu-tiny: minutes on a CPU, no download
|
|
88
|
+
nmt-forge export .forge/runs/<run>/run-manifest.json --prereg all-data --out export/
|
|
89
|
+
# …the twin-free model: --prereg notwins --out export-<its run>/ (with two preregs
|
|
90
|
+
# on one test set, export refuses to guess which one predicted which model)
|
|
91
|
+
nmt-forge serve export/model # http://127.0.0.1:8378 — champollion can call it
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
That order — register the test set, screen the corpus, one preregistration
|
|
95
|
+
per model, (benchmarks), read the guardrails, split, train — is the one `nmt-forge init`,
|
|
96
|
+
`NEXT_STEPS.md` and `nmt-forge status` give. `--prereg <id>` on export names
|
|
97
|
+
the prediction that judges each model. Pinning is the other way:
|
|
98
|
+
`prereg new <id> … --config-hash <hash>` binds a prediction to one config —
|
|
99
|
+
the full hash `nmt-forge preflight run --config <its config>` prints — but any
|
|
100
|
+
later edit of that config (a time budget, say) changes the hash and drops the
|
|
101
|
+
pin, so naming it on export is the simpler route.
|
|
102
|
+
|
|
103
|
+
`export` scores the test set once (prereg-gated, 95% CIs), writes an
|
|
104
|
+
**mt-eval RunLog + TestReport** (built by the harness itself with the metric
|
|
105
|
+
battery `mt-eval run` loads for your language, so `mt-eval compare` works on
|
|
106
|
+
it — anything it could not compute is listed with the command that computes
|
|
107
|
+
it) into `export/evaluation/`, and packages the model into `export/model/`:
|
|
108
|
+
weights, tokenizer, a champollion plugin manifest and a `DEPLOY.md` (with
|
|
109
|
+
the per-pair `fallback` config for strings the model cannot do). Every
|
|
110
|
+
caveat the harness writes on the score — a near-constant output (one of a
|
|
111
|
+
few sentences for many different inputs), length, copies of the source —
|
|
112
|
+
is passed on in its own words beside the score: in the export summary,
|
|
113
|
+
`forge-model.json`, `DEPLOY.md`, `status`, `report`, `compare` and `lint`.
|
|
114
|
+
Scores follow the harness's scoring standard (`standard/1`): the headline —
|
|
115
|
+
the number to quote — is corpus **chrF++ with its 95% bootstrap CI**, written
|
|
116
|
+
`chrF++ 47.5 [45.9, 49.0]`, with its sacreBLEU signature wherever a full record
|
|
117
|
+
is kept (`forge-model.json`, `DEPLOY.md`, the export summary). BLEU, spBLEU,
|
|
118
|
+
TER and COMET are shown beside it, never blended into it; exact match and the
|
|
119
|
+
referee lanes are labelled diagnostics. forge prints no composite and no
|
|
120
|
+
quality label: only a speaker's review certifies quality. `nmt-forge lint`
|
|
121
|
+
takes the battery manifest (`export/evaluation/battery-hyps-battery.json`); given
|
|
122
|
+
a run's `run-manifest.json` it lints every scored export of that run, or tells
|
|
123
|
+
you the `nmt-forge export` command to run first.
|
|
124
|
+
**`export/model/` is what you deploy — it holds no test sentence;
|
|
125
|
+
`export/evaluation/` holds your test set's text and never leaves your
|
|
126
|
+
machine** (marked with the test set's terms when it has any). `serve` speaks the champollion
|
|
127
|
+
**api-method** contract (`POST /translate` — `"method": "api"`) and an
|
|
128
|
+
**OpenAI-compatible** `/v1/chat/completions` (`champollion sync --method
|
|
129
|
+
local`) — app strings and Markdown content files alike. The model never sees
|
|
130
|
+
placeholders, ICU plural/select syntax, tags or Markdown markup: the server
|
|
131
|
+
copies them around the model's output and hands it only the text between
|
|
132
|
+
them, sentence by sentence. `export` prints the verdict on each
|
|
133
|
+
preregistered prediction and how many test sentences have a near-twin in
|
|
134
|
+
training (when most do, it says the score measures recall, not
|
|
135
|
+
translation) — and `split`, `leak-audit` and `preflight run` say the same
|
|
136
|
+
count BEFORE training, while the test set is still unspent, with the fix:
|
|
137
|
+
`split --near-dupe 0.6` holds out whole templates when forge carves the
|
|
138
|
+
test set; for a FIXED test set like the teacher's, `leak-audit <pairs>
|
|
139
|
+
--clean-to <pairs.notwins.jsonl> --drop-test-twins` drops the training rows
|
|
140
|
+
that are near-twins of it (same measure and threshold as the count), says
|
|
141
|
+
how many go and what the strict subset becomes, and refuses to leave
|
|
142
|
+
nothing to train on. It writes to its own file: `leak-audit` refuses to
|
|
143
|
+
replace a file a config, a run or a split reads, or another audit's output
|
|
144
|
+
(`--overwrite` only on purpose). At every step `nmt-forge
|
|
145
|
+
status` names the next command; while a run is training it says `training`
|
|
146
|
+
— wait — and `nmt-forge run` refuses to start a second run in the same
|
|
147
|
+
workspace. With several exported models, the user picks the one to deploy:
|
|
148
|
+
`nmt-forge choose <export>/model` records it (serving a model to try it is
|
|
149
|
+
recorded as served, never as the choice).
|
|
150
|
+
|
|
151
|
+
**Enter it in a sovereign contest.** A contest's declarative lane (Lane A,
|
|
152
|
+
`mt-eval contest submit-model`) takes a model as data: weights, config and
|
|
153
|
+
tokenizer. `export/model/DEPLOY.md` §6 names exactly those files (never
|
|
154
|
+
`forge-model.json` — this model's scores on your test set and local paths —
|
|
155
|
+
nor `DEPLOY.md` or `champollion-plugin/`; `submit-model` leaves all three
|
|
156
|
+
out by itself), the architecture, and the parameter
|
|
157
|
+
count the lane checks, read from the weights file's header, with the
|
|
158
|
+
command. The contest side: [Run a Sovereign
|
|
159
|
+
Contest](https://champollion.dev/docs/network/sovereignty/run-a-sovereign-contest#lane-a--declarative-model-preferred-for-standard-nmt).
|
|
160
|
+
|
|
161
|
+
**A private test set stays private.** Mark it the way a data steward does —
|
|
162
|
+
`champollion network register-corpus --tier local-only --role test`, or a sidecar next to the
|
|
163
|
+
file: `echo '{"transmission": "local-only"}' > teacher-test.tsv.champollion.json`
|
|
164
|
+
— and forge, like `mt-eval`, never prints its sentences. `leak-audit` names
|
|
165
|
+
the rows that matched it by line number (a row that matched a test answer
|
|
166
|
+
IS most of that answer), says once why, and shows the text only with
|
|
167
|
+
`--show-text` — for a person at the terminal; an AI agent passes whatever it
|
|
168
|
+
reads to its model provider. `--json` never carries sentences. The same goes
|
|
169
|
+
for sealed and consent-required corpora (the harness decides which), for a
|
|
170
|
+
training corpus that is itself marked, and for an error message that quotes
|
|
171
|
+
such a row (`[sentence withheld]`). Files forge writes into your folders keep
|
|
172
|
+
the text, and files it carves from a marked corpus (`split` sides, `leak-audit
|
|
173
|
+
--clean-to`, `sample`) get the same mark. When the sidecar names a registered
|
|
174
|
+
corpora card whose sha256 matches the file, the card's id is the set's
|
|
175
|
+
**dataset id** in forge's registry listing, reports, export and mt-eval
|
|
176
|
+
RunLog — the name `mt-eval` runs on the file use — with your `registry add`
|
|
177
|
+
name (`project-test`) kept beside it.
|
|
178
|
+
|
|
179
|
+
**Model presets** (`nmt-forge init --model …`, written out as explicit numbers
|
|
180
|
+
in `config.json`):
|
|
181
|
+
|
|
182
|
+
| preset | what it is | needs | honest expectation |
|
|
183
|
+
|---|---|---|---|
|
|
184
|
+
| `cpu-tiny` (default) | a ~6M-parameter Marian trained from scratch; BPE vocabulary fit on the TRAIN rows only | a CPU, no download — 1,600 pairs train in ~2–3 minutes on a laptop | weak: on 1–2k pairs, chrF++ roughly 5–30 (the top end only for highly templated data) — your data's phrases and templates, not general translation |
|
|
185
|
+
| `cpu-finetune --base <hf-id>` | fine-tune a small pretrained Marian/opus-mt (you name a RELATED pair) | a CPU, ~300 MB download | usually better than cpu-tiny when a related pair exists — measure it, don't assume |
|
|
186
|
+
| `nllb-600m` | NLLB-200 distilled 600M + LoRA | a GPU, ~2.5 GB download | the strongest start; the wall-clock gate refuses it on a CPU in minutes |
|
|
187
|
+
|
|
188
|
+
The point of `cpu-tiny` is not the score: it makes the WHOLE loop real — the
|
|
189
|
+
fence, the audits, the preregistered test, an exportable model the CLI can
|
|
190
|
+
call — so a better model later drops into the same project and is measured
|
|
191
|
+
the same way.
|
|
192
|
+
|
|
193
|
+
## Sixty seconds, end to end
|
|
194
|
+
|
|
195
|
+
```bash
|
|
196
|
+
# carve an honest split: pairs sharing a source OR target stay together
|
|
197
|
+
$ nmt-forge split corpus.jsonl --test 150 --dev 42 --seed 42 \
|
|
198
|
+
--out data/split --register textbook
|
|
199
|
+
split corpus.jsonl: 1240 rows in 1187 share-groups (largest 4)
|
|
200
|
+
train 1048 · dev 42 · test 150 → data/split/
|
|
201
|
+
verified: 0 shared canonical source/target keys across sides
|
|
202
|
+
|
|
203
|
+
# train: one command, config-hashed, dev-fenced
|
|
204
|
+
$ nmt-forge run config.json
|
|
205
|
+
dev report (95% CIs — there is no bare-score rendering):
|
|
206
|
+
n=42 · set=textbook-dev
|
|
207
|
+
chrf++ 44.31 [41.20, 47.15] 95% CI
|
|
208
|
+
|
|
209
|
+
# score the test set? not without predictions written down first
|
|
210
|
+
$ nmt-forge score --eval-set textbook-test --hyps decoded.txt
|
|
211
|
+
[preregister] no preregistration for eval set 'textbook-test' ...
|
|
212
|
+
why: results looked at without written-down expectations become post-hoc stories
|
|
213
|
+
fix: write one FIRST: ... — then score
|
|
214
|
+
|
|
215
|
+
$ nmt-forge prereg template --out preds.json # the ONE format; edit it
|
|
216
|
+
$ nmt-forge prereg new e2 --eval-set textbook-test --predictions preds.json
|
|
217
|
+
$ nmt-forge score --eval-set textbook-test --hyps decoded.txt
|
|
218
|
+
chrf++ 46.02 [43.11, 48.87] 95% CI
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
That's the whole philosophy: the honest path is the easy path, and the
|
|
222
|
+
dishonest paths are closed with actionable messages.
|
|
223
|
+
|
|
224
|
+
**Agents:** every command takes `--json` — exactly one JSON document on
|
|
225
|
+
stdout, refusals as `{"error": {…, "why", "fix"}}` with exit 2. Shapes:
|
|
226
|
+
[docs/JSON_OUTPUT.md](docs/JSON_OUTPUT.md). Call `nmt-forge status --json`
|
|
227
|
+
first, always.
|
|
228
|
+
|
|
229
|
+
## Any language: start from the card
|
|
230
|
+
|
|
231
|
+
forge is general-purpose across all ~7,900 SSOT language cards. `discover`
|
|
232
|
+
answers "what does this language actually have?" honestly (absence on a card
|
|
233
|
+
= **unknown**, never zero), and `init` scaffolds a project from it:
|
|
234
|
+
|
|
235
|
+
```bash
|
|
236
|
+
$ nmt-forge discover nav
|
|
237
|
+
Navajo (nav) · ltr
|
|
238
|
+
WHAT THE CARD SAYS EXISTS (absence = unknown, not zero):
|
|
239
|
+
OPUS: 5 corpora, 36533 aligned pairs
|
|
240
|
+
unknown (card is silent): analyzers, dictionaries, eval datasets
|
|
241
|
+
THE ASSET LADDER — what this language can do TODAY:
|
|
242
|
+
✓ rung 1: parallel text → train with every guard (no pack needed)
|
|
243
|
+
? rung 2: monolingual text → the tagged backtranslation lane
|
|
244
|
+
? rung 3: dictionary (+ grammar) → a cited template pack is worth building
|
|
245
|
+
? rung 4: morphological analyzer → round-trip-VERIFIED synthesis
|
|
246
|
+
? rung 5: LYSS referee → the language's own metric in selection
|
|
247
|
+
note: no analyzer on the card → synthesis is off the menu until one
|
|
248
|
+
exists; every guard and the training loop work regardless
|
|
249
|
+
|
|
250
|
+
$ nmt-forge init nav --dir my-navajo-mt
|
|
251
|
+
# → .forge/ workspace + starter config.json + NEXT_STEPS.md (the agent brief)
|
|
252
|
+
|
|
253
|
+
# a language the index doesn't have yet? still trainable — nothing invented:
|
|
254
|
+
$ nmt-forge init qzx --no-card --name "My Language" --dir my-mt
|
|
255
|
+
```
|
|
256
|
+
|
|
257
|
+
A rich card wires more for free: Plains Cree's card carries its eval-dataset
|
|
258
|
+
ids (cross-checked against the mt-eval registry — all flagged NEVER TRAIN ON
|
|
259
|
+
THIS) and its LYSS referee, so `discover crk` emits ready-to-paste
|
|
260
|
+
`--plugin champollion_lyss...` lanes. The referee is an OPTIONAL add-on with
|
|
261
|
+
its own license, never a requirement: `init crk` wires its lanes into the
|
|
262
|
+
starter config only when the package is installed, and otherwise names it in
|
|
263
|
+
NEXT_STEPS.md — the config passes its own preflight on a plain install. Same
|
|
264
|
+
tool, no special cases — the card decides.
|
|
265
|
+
|
|
266
|
+
## What's inside (start here, dig later)
|
|
267
|
+
|
|
268
|
+
| you want to… | use | it kills the mistake of… |
|
|
269
|
+
|---|---|---|
|
|
270
|
+
| split a corpus | `split` / `verify-split` | test answers hiding in training via shared sources/targets |
|
|
271
|
+
| pick checkpoints | the run's **dev-fence** | the test set choosing the model — and training on the dev set's own rows (a file written before the split, such as an early twin-free corpus, is refused with the re-audit that fixes it) |
|
|
272
|
+
| screen any corpus/harvest | `leak-audit` | training on eval text (exact, reworded, or whole-file) — while KEEPING template siblings ("I see the dog" vs "I see the cat") and similar-prompt/different-answer pairs (an *identical* prompt is dropped whatever its answer), and saying which is which, with examples (by line number only for a private — local-only, sealed, consent-required — set or corpus); `--drop-test-twins` also drops the near-twins of a FIXED test set when they would turn its score into recall |
|
|
273
|
+
| generate training data | `synth` + a language pack | unverified forms, uncited templates, invisible gaps |
|
|
274
|
+
| sample synthetic data | `sample` | two template kinds hogging half the signal |
|
|
275
|
+
| report numbers | `score` / `compare` / `evaluate` | scores with no error bars, no preregistration |
|
|
276
|
+
| hand the model on | `export` / `serve` | a result nobody else can compare (mt-eval TestReport) or deploy (champollion api / OpenAI-compatible endpoint) |
|
|
277
|
+
| see how spent an eval set is | `ledger show` | invisible adaptive use; sealed sets are one-shot |
|
|
278
|
+
|
|
279
|
+
Each guard is also a library call under `nmt_forge.guards.*`. The full
|
|
280
|
+
mistake→mechanism map, with the measured numbers behind each guard, is in
|
|
281
|
+
[DESIGN.md](DESIGN.md).
|
|
282
|
+
|
|
283
|
+
## Language packs plug in — forge ships none
|
|
284
|
+
|
|
285
|
+
forge is general-purpose; language-specific code lives in the language's own
|
|
286
|
+
home and plugs in through the pack interface (analyzer + dictionary adapter +
|
|
287
|
+
orthography + **grammar-cited** templates + checklist):
|
|
288
|
+
|
|
289
|
+
```bash
|
|
290
|
+
# from any checkout (no install needed):
|
|
291
|
+
nmt-forge synth nmt_forge_crk.pack:get_pack --out data/synth.jsonl
|
|
292
|
+
# or, once the pack's package is installed (entry point): nmt-forge synth crk
|
|
293
|
+
```
|
|
294
|
+
|
|
295
|
+
The engine enforces the **emit law** on every pack: every generated word must
|
|
296
|
+
round-trip through the language's analyzer, every closed-class literal must
|
|
297
|
+
be cited, every filter is named and counted, and every row is stamped
|
|
298
|
+
`synthetic: true` — which is exactly why the registry refuses synthetic rows
|
|
299
|
+
in test sets (tests are real data only). The Plains Cree reference pack lives
|
|
300
|
+
in crk-translate (`nmt_forge_crk`); FST models and dictionaries stay
|
|
301
|
+
**separate, user-fetched tools** under their own licenses — never bundled.
|
|
302
|
+
|
|
303
|
+
## The full harness referee stack — neural metrics included
|
|
304
|
+
|
|
305
|
+
forge speaks everything the eval harness speaks, by delegation: the
|
|
306
|
+
deterministic lanes (chrF++/BLEU/exact-match), the **neural lanes** —
|
|
307
|
+
COMET, COMET-QE, MetricX — and the harness's own plugin discovery (FST
|
|
308
|
+
word-validity, behavioral linters, card-declared metrics):
|
|
309
|
+
|
|
310
|
+
```bash
|
|
311
|
+
nmt-forge score --eval-set project-test --hyps decoded.txt \
|
|
312
|
+
--metric chrf++ --metric comet --target-lang iku --card-plugins iku
|
|
313
|
+
chrf++ 32.10 [29.4, 34.9] 95% CI
|
|
314
|
+
comet 0.71 [ 0.66, 0.75] 95% CI
|
|
315
|
+
metricx 3.20 [ 2.9, 3.6] 95% CI (lower = better)
|
|
316
|
+
```
|
|
317
|
+
|
|
318
|
+
Neural inference runs **once**; the bootstrap re-averages cached per-entry
|
|
319
|
+
scores (the harness's own CI pattern). Missing extras report an install fix
|
|
320
|
+
— never a fabricated number — and checkpoint selection *refuses* rather
|
|
321
|
+
than silently switching metrics. Direction is first-class: MetricX's
|
|
322
|
+
lower-is-better rides the score through rendering, selection, and A/B
|
|
323
|
+
winners. And `discover` tells you which lane to **believe**, from the WMT
|
|
324
|
+
meta-evaluations:
|
|
325
|
+
|
|
326
|
+
```
|
|
327
|
+
$ nmt-forge discover iku
|
|
328
|
+
metric trust (Eskimo-Aleut, WMT meta-eval): comet_score r=0.86, ... bleu r=0.163
|
|
329
|
+
```
|
|
330
|
+
|
|
331
|
+
— for Inuktitut, BLEU barely tracks human judgment while COMET does; for
|
|
332
|
+
other families it's the reverse; for most low-resource families the honest
|
|
333
|
+
answer is UNMEASURED. forge surfaces that before you select on anything.
|
|
334
|
+
|
|
335
|
+
## LYSS referees plug in too
|
|
336
|
+
|
|
337
|
+
LYSS eval-standard linters (harness `MetricPlugin` protocol — Plains Cree
|
|
338
|
+
today, more languages later) drop into every scoring surface:
|
|
339
|
+
|
|
340
|
+
```bash
|
|
341
|
+
nmt-forge score --eval-set textbook-test --hyps decoded.txt \
|
|
342
|
+
--plugin champollion_lyss.crk.metrics:CrkLinterMetric
|
|
343
|
+
chrf++ 46.02 [43.11, 48.87] 95% CI
|
|
344
|
+
crk_linter:equivalent_match_rate 0.31 [ 0.24, 0.38] 95% CI
|
|
345
|
+
crk_linter · variant_class_counts: LONG_VOWEL_MACRON=9, WORD_ORDER=3
|
|
346
|
+
```
|
|
347
|
+
|
|
348
|
+
Every numeric aggregate a plugin reports gets a bootstrap CI; per-entry
|
|
349
|
+
computation runs once (a thousand resamples never re-run an FST); a plugin
|
|
350
|
+
that says `available: false` is shown unavailable, never fabricated. And
|
|
351
|
+
checkpoint selection can use the language's own referee:
|
|
352
|
+
|
|
353
|
+
```jsonc
|
|
354
|
+
"selection": {"metric": "generation:crk_linter:equivalent_match_rate",
|
|
355
|
+
"plugins": ["champollion_lyss.crk.metrics:CrkLinterMetric"]}
|
|
356
|
+
```
|
|
357
|
+
|
|
358
|
+
## A worked example: the half-epoch death
|
|
359
|
+
|
|
360
|
+
The first CLEAN-protocol Cree run — honest group-disjoint dev, leak-audited
|
|
361
|
+
mix — died at epoch 0.52 of a 115,000-step plan. Mechanism: the mix was
|
|
362
|
+
97.5% tagged synthetic; early in training the model fits the synthetic
|
|
363
|
+
mass, so dev loss on the 42 *real* dev sentences bottomed at step ~8k and
|
|
364
|
+
drifted upward, and patience-6 declared convergence at half an epoch. Every
|
|
365
|
+
earlier run had hidden this bug by (illegitimately) using the test set as
|
|
366
|
+
dev. **The honest setup is what surfaced it — that's the point of the
|
|
367
|
+
suite.**
|
|
368
|
+
|
|
369
|
+
forge makes the fix the default, not a flag (`schedule-sanity`):
|
|
370
|
+
|
|
371
|
+
```
|
|
372
|
+
[schedule-sanity] train: regime=synthetic-heavy (auto-detected), floor=38,205 of 114,614 planned steps
|
|
373
|
+
[schedule-sanity] early stopping ASKED to stop at step 14,000 but the
|
|
374
|
+
schedule floor (38,205) held training, because: the mix is 97.5% synthetic
|
|
375
|
+
and the dev set is REAL: early in training the model fits the synthetic
|
|
376
|
+
mass, so dev loss on real sentences bottoms fast and drifts up — that
|
|
377
|
+
pattern is EXPECTED, not convergence …
|
|
378
|
+
```
|
|
379
|
+
|
|
380
|
+
The floor is **derived** from the config (max of one full pass over the mix
|
|
381
|
+
and 30% of planned steps, capped at 60%) and activates only in the
|
|
382
|
+
`synthetic-heavy` regime — auto-detected from the mix, overridable with one
|
|
383
|
+
word (`"regime": "balanced"`), never ten flags. Every intervention prints
|
|
384
|
+
the dev-loss trajectory and the why; nobody should diagnose this from raw
|
|
385
|
+
logs again.
|
|
386
|
+
|
|
387
|
+
## Training defaults that encode the ledger
|
|
388
|
+
|
|
389
|
+
Tagged synthetic lanes (Caswell et al. 2019) with gold untagged; gold
|
|
390
|
+
upweighting with the **exposure math written into the manifest**; per-kind
|
|
391
|
+
sampling caps; curriculum stages; a backtranslation lane that leak-audits
|
|
392
|
+
mono text *before* spending translation; generation-headroom checks; and a
|
|
393
|
+
do-not-train gate — datasets the mt-eval registry protects never enter a mix.
|
|
394
|
+
Details and the config schema: [DESIGN.md](DESIGN.md) §6.
|
|
395
|
+
|
|
396
|
+
## What forge refuses to be
|
|
397
|
+
|
|
398
|
+
Not an evaluator (the harness scores), not a corpus host (manifests are
|
|
399
|
+
content-free — hashes and counts, never text), not a language-card writer,
|
|
400
|
+
not a leaderboard.
|
|
401
|
+
|
|
402
|
+
## License
|
|
403
|
+
|
|
404
|
+
PolyForm Noncommercial 1.0.0 — source-available, free for noncommercial use;
|
|
405
|
+
using it for a commercial purpose is not covered by this license (relicensed
|
|
406
|
+
from AGPL-3.0-or-later on 2026-08-17, before any release shipped). Who is
|
|
407
|
+
covered, in plain words with examples: [Who may use this](https://champollion.dev/docs/getting-started/who-may-use-this) (a summary,
|
|
408
|
+
not legal advice; [LICENSE](LICENSE) governs). The harness it uses stays
|
|
409
|
+
open-source AGPL-3.0-or-later. Analyzer models and
|
|
410
|
+
dictionaries consumed by packs are upstream artifacts fetched by the user
|
|
411
|
+
under the upstream's terms.
|