nmt-forge 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- nmt_forge/__init__.py +16 -0
- nmt_forge/_harness.py +255 -0
- nmt_forge/advisor.py +2419 -0
- nmt_forge/canonical.py +143 -0
- nmt_forge/cards.py +1289 -0
- nmt_forge/cli.py +3150 -0
- nmt_forge/errors.py +179 -0
- nmt_forge/export.py +2020 -0
- nmt_forge/guards/__init__.py +39 -0
- nmt_forge/guards/battery_lint.py +349 -0
- nmt_forge/guards/ci_scoring.py +1730 -0
- nmt_forge/guards/convention_lint.py +130 -0
- nmt_forge/guards/coverage_map.py +141 -0
- nmt_forge/guards/dev_fence.py +207 -0
- nmt_forge/guards/funnel_audit.py +140 -0
- nmt_forge/guards/leak_audit.py +1308 -0
- nmt_forge/guards/preregister.py +883 -0
- nmt_forge/guards/sample_strata.py +136 -0
- nmt_forge/guards/split_guard.py +552 -0
- nmt_forge/harness_bridge.py +645 -0
- nmt_forge/harness_caveats.py +222 -0
- nmt_forge/harness_data.py +78 -0
- nmt_forge/ledger.py +144 -0
- nmt_forge/monitor.py +530 -0
- nmt_forge/plugins.py +180 -0
- nmt_forge/privacy.py +269 -0
- nmt_forge/registry.py +699 -0
- nmt_forge/reporting.py +493 -0
- nmt_forge/runlock.py +258 -0
- nmt_forge/scaffold.py +633 -0
- nmt_forge/scoring_standard.py +274 -0
- nmt_forge/serve.py +529 -0
- nmt_forge/synthesis/__init__.py +6 -0
- nmt_forge/synthesis/analyzer.py +79 -0
- nmt_forge/synthesis/engine.py +215 -0
- nmt_forge/synthesis/filters.py +146 -0
- nmt_forge/synthesis/packs.py +147 -0
- nmt_forge/synthesis/probe.py +72 -0
- nmt_forge/synthesis/run.py +16 -0
- nmt_forge/synthesis/templates.py +110 -0
- nmt_forge/textpipe.py +494 -0
- nmt_forge/training/__init__.py +13 -0
- nmt_forge/training/backends.py +925 -0
- nmt_forge/training/backtranslation.py +93 -0
- nmt_forge/training/config.py +217 -0
- nmt_forge/training/evaluate.py +268 -0
- nmt_forge/training/mix.py +275 -0
- nmt_forge/training/presets.py +156 -0
- nmt_forge/training/run.py +357 -0
- nmt_forge/training/schedule.py +405 -0
- nmt_forge/training/selection.py +172 -0
- nmt_forge/workspace.py +36 -0
- nmt_forge-0.2.0.dist-info/METADATA +411 -0
- nmt_forge-0.2.0.dist-info/RECORD +58 -0
- nmt_forge-0.2.0.dist-info/WHEEL +5 -0
- nmt_forge-0.2.0.dist-info/entry_points.txt +2 -0
- nmt_forge-0.2.0.dist-info/licenses/LICENSE +133 -0
- nmt_forge-0.2.0.dist-info/top_level.txt +1 -0
nmt_forge/__init__.py
ADDED
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""nmt-forge — an NMT training suite that makes the catalogued mistakes hard.
|
|
2
|
+
|
|
3
|
+
Requirements document: the 2026-07-12 crk-translate mistake ledger (11
|
|
4
|
+
entries, each mistake → concrete example → guard). Failure taxonomy:
|
|
5
|
+
forge/docs/FAILURE_TAXONOMY.md.
|
|
6
|
+
|
|
7
|
+
The suite refuses bad practice with actionable messages (what / why / fix),
|
|
8
|
+
synthesizes training data only through round-trip-verified, grammar-cited
|
|
9
|
+
templates, and delegates ALL scoring to mt-eval.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
__version__ = "0.2.0"
|
|
13
|
+
|
|
14
|
+
from .canonical import canonical_key, config_hash, stable_hash # noqa: F401
|
|
15
|
+
from .errors import ForgeError, GuardrailViolation, ResourceMissing # noqa: F401
|
|
16
|
+
from .workspace import Workspace # noqa: F401
|
nmt_forge/_harness.py
ADDED
|
@@ -0,0 +1,255 @@
|
|
|
1
|
+
"""Import shim for the eval harness (``mt-eval-harness`` on PyPI).
|
|
2
|
+
|
|
3
|
+
forge implements ZERO metrics (mistake #10 in the requirements ledger: a
|
|
4
|
+
bespoke evaluator partially re-implemented scoring and drifted). All scoring
|
|
5
|
+
delegates to ``mt_eval_harness``:
|
|
6
|
+
|
|
7
|
+
- point scores + bootstrap CIs: ``mt_eval_harness.confidence``
|
|
8
|
+
- A/B significance: ``mt_eval_harness.significance``
|
|
9
|
+
- language cards (resolver + the ONE adapter): ``mt_eval_harness.language_cards``
|
|
10
|
+
- RunLog / TestReport (the bridge to ``mt-eval``): ``mt_eval_harness.pipeline``
|
|
11
|
+
and ``mt_eval_harness.tester``
|
|
12
|
+
|
|
13
|
+
``mt-eval-harness`` is a declared dependency of nmt-forge, so a normal
|
|
14
|
+
``python3 -m pip install nmt-forge`` brings it in. Resolution order:
|
|
15
|
+
1. the installed ``mt-eval-harness`` distribution (import ``mt_eval_harness``);
|
|
16
|
+
2. ``$MT_EVAL_HARNESS_PATH`` (a checkout containing ``mt_eval_harness/``);
|
|
17
|
+
3. monorepo fallback: the sibling ``arena/`` directory (forge lives at
|
|
18
|
+
``<repo>/forge/``, the harness at ``<repo>/arena/mt_eval_harness/``).
|
|
19
|
+
|
|
20
|
+
Fails loud with install instructions — never a silent no-scores path.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import contextlib
|
|
26
|
+
import functools
|
|
27
|
+
import importlib
|
|
28
|
+
import io
|
|
29
|
+
import json
|
|
30
|
+
import os
|
|
31
|
+
import sys
|
|
32
|
+
from pathlib import Path
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _candidate_paths() -> list[Path]:
|
|
36
|
+
cands = []
|
|
37
|
+
env = os.environ.get("MT_EVAL_HARNESS_PATH")
|
|
38
|
+
if env:
|
|
39
|
+
cands.append(Path(env))
|
|
40
|
+
# <repo>/forge/nmt_forge/_harness.py → <repo>/arena
|
|
41
|
+
cands.append(Path(__file__).resolve().parents[2] / "arena")
|
|
42
|
+
return cands
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def _load_from_checkout(cand: Path):
|
|
46
|
+
"""Import ``<cand>/mt_eval_harness`` as the package WITHOUT putting
|
|
47
|
+
``cand`` on sys.path.
|
|
48
|
+
|
|
49
|
+
Adding the whole ``arena/`` directory to sys.path (the pre-2026-10
|
|
50
|
+
fallback) also made every sibling directory importable as a namespace
|
|
51
|
+
package — ``arena/datasets/`` then shadowed Hugging Face ``datasets``,
|
|
52
|
+
transformers believed `datasets` was installed, and every HF training
|
|
53
|
+
run crashed inside the Trainer's dataloader. Loading the one package by
|
|
54
|
+
location avoids that whole class of collision."""
|
|
55
|
+
import importlib.util
|
|
56
|
+
|
|
57
|
+
pkg_dir = cand / "mt_eval_harness"
|
|
58
|
+
spec = importlib.util.spec_from_file_location(
|
|
59
|
+
"mt_eval_harness", pkg_dir / "__init__.py",
|
|
60
|
+
submodule_search_locations=[str(pkg_dir)])
|
|
61
|
+
if spec is None or spec.loader is None:
|
|
62
|
+
raise ImportError(f"cannot load mt_eval_harness from {pkg_dir}")
|
|
63
|
+
module = importlib.util.module_from_spec(spec)
|
|
64
|
+
sys.modules["mt_eval_harness"] = module
|
|
65
|
+
try:
|
|
66
|
+
spec.loader.exec_module(module)
|
|
67
|
+
except BaseException:
|
|
68
|
+
sys.modules.pop("mt_eval_harness", None)
|
|
69
|
+
raise
|
|
70
|
+
return module
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def load_harness():
|
|
74
|
+
"""Import and return the ``mt_eval_harness`` package, or raise ForgeError."""
|
|
75
|
+
try:
|
|
76
|
+
return importlib.import_module("mt_eval_harness")
|
|
77
|
+
except ImportError:
|
|
78
|
+
pass
|
|
79
|
+
for cand in _candidate_paths():
|
|
80
|
+
if (cand / "mt_eval_harness" / "__init__.py").is_file():
|
|
81
|
+
try:
|
|
82
|
+
return _load_from_checkout(cand)
|
|
83
|
+
except ImportError:
|
|
84
|
+
continue
|
|
85
|
+
from .errors import ForgeError
|
|
86
|
+
|
|
87
|
+
raise ForgeError(
|
|
88
|
+
"mt-eval-harness is not importable — forge delegates ALL scoring to "
|
|
89
|
+
"it and has no fallback scorer by design.\n"
|
|
90
|
+
" fix: `python3 -m pip install mt-eval-harness` (it is a declared dependency of "
|
|
91
|
+
"nmt-forge, so `python3 -m pip install nmt-forge` normally brings it in), or "
|
|
92
|
+
"point MT_EVAL_HARNESS_PATH at a checkout containing "
|
|
93
|
+
"mt_eval_harness/, or run from the Champollion monorepo where arena/ "
|
|
94
|
+
"is a sibling of forge/."
|
|
95
|
+
)
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def confidence():
|
|
99
|
+
load_harness()
|
|
100
|
+
return importlib.import_module("mt_eval_harness.confidence")
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def significance():
|
|
104
|
+
load_harness()
|
|
105
|
+
return importlib.import_module("mt_eval_harness.significance")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def language_cards_mod():
|
|
109
|
+
"""The harness's language_cards module (the ONE Python card adapter)."""
|
|
110
|
+
load_harness()
|
|
111
|
+
return importlib.import_module("mt_eval_harness.language_cards")
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def corpus_loader_mod():
|
|
115
|
+
"""The harness's corpus loader (TSV is read by the harness's own rules)."""
|
|
116
|
+
load_harness()
|
|
117
|
+
return importlib.import_module("mt_eval_harness.corpus_loader")
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def language_cards_remote_mod():
|
|
121
|
+
"""The harness's remote card source (its LanguageCardsUnavailable is the
|
|
122
|
+
fail-loud signal forge converts into an actionable ResourceMissing)."""
|
|
123
|
+
load_harness()
|
|
124
|
+
return importlib.import_module("mt_eval_harness.language_cards_remote")
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def harness_version() -> str:
|
|
128
|
+
"""The installed harness version (recorded in exports and run logs)."""
|
|
129
|
+
return getattr(load_harness(), "__version__", "unknown")
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def transmission_policy_mod():
|
|
133
|
+
"""The harness's transmission policy: which corpora may reach which
|
|
134
|
+
model — and, since 2026-10, which corpus sentences may be PRINTED
|
|
135
|
+
(``withheld_text_reason`` / ``withheld_note`` / ``scrub_corpus_text``).
|
|
136
|
+
forge decides nothing about a corpus's terms itself."""
|
|
137
|
+
load_harness()
|
|
138
|
+
return importlib.import_module("mt_eval_harness.transmission_policy")
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def corpus_terms(path) -> dict:
|
|
142
|
+
"""What the harness reads about a corpus file's terms and identity,
|
|
143
|
+
WITHOUT scoring anything: ``{"meta", "envelope", "notes"}``.
|
|
144
|
+
|
|
145
|
+
``envelope`` is a harness-JSON corpus's own ``dataset`` block (``{}`` for
|
|
146
|
+
TSV / JSONL / a bare list); ``meta`` is that block merged with the
|
|
147
|
+
steward's ``<file>.champollion.json`` sidecar and the corpora card it
|
|
148
|
+
registers (``corpus_loader.merge_steward_sidecar`` — the same merge an
|
|
149
|
+
``mt-eval run`` on the file does: ``transmission``, a sealed
|
|
150
|
+
``segment``, ``license``, ``id``, ``corpus_card``, ``contamination``);
|
|
151
|
+
``notes`` is what the harness printed while merging (e.g. why a card's
|
|
152
|
+
id was not applied). Raises ValueError for an unreadable sidecar or
|
|
153
|
+
envelope — a steward wrote it to restrict the data, so a typo must not
|
|
154
|
+
unlock it.
|
|
155
|
+
"""
|
|
156
|
+
cl = corpus_loader_mod()
|
|
157
|
+
p = Path(path)
|
|
158
|
+
envelope: dict = {}
|
|
159
|
+
if p.suffix.lower() == ".json" and p.is_file():
|
|
160
|
+
try:
|
|
161
|
+
_, envelope = cl._load_harness_json(p, None)
|
|
162
|
+
except SystemExit as exc: # the loader's own "not valid JSON" exit
|
|
163
|
+
raise ValueError(" ".join(str(exc).split())) from None
|
|
164
|
+
envelope = dict(envelope) if isinstance(envelope, dict) else {}
|
|
165
|
+
buf = io.StringIO()
|
|
166
|
+
with contextlib.redirect_stdout(buf):
|
|
167
|
+
meta = cl.merge_steward_sidecar(p, dict(envelope))
|
|
168
|
+
notes = [" ".join(ln.split()) for ln in buf.getvalue().splitlines()
|
|
169
|
+
if ln.strip()]
|
|
170
|
+
return {"meta": meta, "envelope": envelope, "notes": notes}
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def card_not_applied_reason(path) -> str:
|
|
174
|
+
"""Why the corpora card a sidecar names was NOT applied to this file
|
|
175
|
+
(moved, unreadable, no id, or the file changed since registration — the
|
|
176
|
+
harness's own words), or ``""`` (no card named, or it applies)."""
|
|
177
|
+
cl = corpus_loader_mod()
|
|
178
|
+
side = cl.read_steward_sidecar(path)
|
|
179
|
+
if not side.get("card"):
|
|
180
|
+
return ""
|
|
181
|
+
buf = io.StringIO()
|
|
182
|
+
with contextlib.redirect_stdout(buf):
|
|
183
|
+
card = cl.registered_card(path, side)
|
|
184
|
+
if card is not None:
|
|
185
|
+
return ""
|
|
186
|
+
text = " ".join(buf.getvalue().split())
|
|
187
|
+
return text.removeprefix("Steward: ").strip() or (
|
|
188
|
+
f"the card {side['card']} named in the sidecar was not applied")
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
@functools.lru_cache(maxsize=512)
|
|
192
|
+
def _registry_entry(dataset_id: str, corpus_path: str, meta_key: str):
|
|
193
|
+
"""``publish.registry_entry_for_run`` (id, then the path/basename
|
|
194
|
+
ladder), memoized per process: one lookup reads the 5,600-entry
|
|
195
|
+
registry, and a command may ask about the same file several times."""
|
|
196
|
+
load_harness()
|
|
197
|
+
from mt_eval_harness.publish import registry_entry_for_run
|
|
198
|
+
|
|
199
|
+
return registry_entry_for_run(dataset_id, corpus_path=corpus_path,
|
|
200
|
+
corpus_meta=json.loads(meta_key))
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def corpus_transmission_policy(path, *, dataset_id: str = ""):
|
|
204
|
+
"""The harness's TransmissionPolicy for a corpus file, resolved exactly
|
|
205
|
+
as a run on it would be: the file's own terms (``corpus_terms``) and
|
|
206
|
+
its mt-eval registry entry (by ``dataset_id``, then the path ladder).
|
|
207
|
+
Raises ValueError when the file's terms are unreadable."""
|
|
208
|
+
tp = transmission_policy_mod()
|
|
209
|
+
meta = corpus_terms(path)["meta"]
|
|
210
|
+
p = str(Path(path).resolve())
|
|
211
|
+
did, entry = _registry_entry(
|
|
212
|
+
str(dataset_id or ""), p,
|
|
213
|
+
json.dumps(meta, sort_keys=True, ensure_ascii=False, default=str))
|
|
214
|
+
return tp.resolve_transmission_policy(did, registry_entry=entry,
|
|
215
|
+
corpus_meta=meta)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def derived_mark(path, *, dataset_id: str = "") -> dict:
|
|
219
|
+
"""The terms a file DERIVED from this corpus must carry, as the harness
|
|
220
|
+
decides them (``corpus_loader.derived_mark``, the mark ``mt-eval`` puts
|
|
221
|
+
on the run logs and reports it writes), or ``{}``.
|
|
222
|
+
|
|
223
|
+
Read over the policy a run on the file would resolve — the file's own
|
|
224
|
+
terms AND its mt-eval registry entry — so a set the registry seals or
|
|
225
|
+
quarantines, or whose consent-required licence only the registry
|
|
226
|
+
records, is marked too. ``{}`` from a harness that predates the helper
|
|
227
|
+
(forge then carries the file's own terms alone). Raises ValueError when
|
|
228
|
+
the file's terms are unreadable."""
|
|
229
|
+
cl = corpus_loader_mod()
|
|
230
|
+
helper = getattr(cl, "derived_mark", None)
|
|
231
|
+
if helper is None:
|
|
232
|
+
return {}
|
|
233
|
+
policy = corpus_transmission_policy(path, dataset_id=dataset_id)
|
|
234
|
+
meta = corpus_terms(path)["meta"]
|
|
235
|
+
return dict(helper({"config": {"corpus_path": str(path),
|
|
236
|
+
"transmission_policy":
|
|
237
|
+
policy.as_provenance()},
|
|
238
|
+
"provenance": {"dataset_meta": meta}}))
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def withheld_text_reason(path, *, dataset_id: str = "") -> str:
|
|
242
|
+
"""Why forge must not PRINT this corpus file's sentences, or ``""``.
|
|
243
|
+
|
|
244
|
+
The harness decides (``transmission_policy.withheld_text_reason``): a
|
|
245
|
+
steward's local-only mark (sidecar or JSON envelope), or a policy that
|
|
246
|
+
refuses remote models (sealed segment, quarantined, consent-required
|
|
247
|
+
license). An unreadable sidecar/envelope counts as a reason."""
|
|
248
|
+
tp = transmission_policy_mod()
|
|
249
|
+
try:
|
|
250
|
+
policy = corpus_transmission_policy(path, dataset_id=dataset_id)
|
|
251
|
+
except ValueError as exc:
|
|
252
|
+
return f"its terms could not be read ({exc})"
|
|
253
|
+
return tp.withheld_text_reason({"config": {
|
|
254
|
+
"corpus_path": str(path),
|
|
255
|
+
"transmission_policy": policy.as_provenance()}})
|