nmt-forge 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. nmt_forge/__init__.py +16 -0
  2. nmt_forge/_harness.py +255 -0
  3. nmt_forge/advisor.py +2419 -0
  4. nmt_forge/canonical.py +143 -0
  5. nmt_forge/cards.py +1289 -0
  6. nmt_forge/cli.py +3150 -0
  7. nmt_forge/errors.py +179 -0
  8. nmt_forge/export.py +2020 -0
  9. nmt_forge/guards/__init__.py +39 -0
  10. nmt_forge/guards/battery_lint.py +349 -0
  11. nmt_forge/guards/ci_scoring.py +1730 -0
  12. nmt_forge/guards/convention_lint.py +130 -0
  13. nmt_forge/guards/coverage_map.py +141 -0
  14. nmt_forge/guards/dev_fence.py +207 -0
  15. nmt_forge/guards/funnel_audit.py +140 -0
  16. nmt_forge/guards/leak_audit.py +1308 -0
  17. nmt_forge/guards/preregister.py +883 -0
  18. nmt_forge/guards/sample_strata.py +136 -0
  19. nmt_forge/guards/split_guard.py +552 -0
  20. nmt_forge/harness_bridge.py +645 -0
  21. nmt_forge/harness_caveats.py +222 -0
  22. nmt_forge/harness_data.py +78 -0
  23. nmt_forge/ledger.py +144 -0
  24. nmt_forge/monitor.py +530 -0
  25. nmt_forge/plugins.py +180 -0
  26. nmt_forge/privacy.py +269 -0
  27. nmt_forge/registry.py +699 -0
  28. nmt_forge/reporting.py +493 -0
  29. nmt_forge/runlock.py +258 -0
  30. nmt_forge/scaffold.py +633 -0
  31. nmt_forge/scoring_standard.py +274 -0
  32. nmt_forge/serve.py +529 -0
  33. nmt_forge/synthesis/__init__.py +6 -0
  34. nmt_forge/synthesis/analyzer.py +79 -0
  35. nmt_forge/synthesis/engine.py +215 -0
  36. nmt_forge/synthesis/filters.py +146 -0
  37. nmt_forge/synthesis/packs.py +147 -0
  38. nmt_forge/synthesis/probe.py +72 -0
  39. nmt_forge/synthesis/run.py +16 -0
  40. nmt_forge/synthesis/templates.py +110 -0
  41. nmt_forge/textpipe.py +494 -0
  42. nmt_forge/training/__init__.py +13 -0
  43. nmt_forge/training/backends.py +925 -0
  44. nmt_forge/training/backtranslation.py +93 -0
  45. nmt_forge/training/config.py +217 -0
  46. nmt_forge/training/evaluate.py +268 -0
  47. nmt_forge/training/mix.py +275 -0
  48. nmt_forge/training/presets.py +156 -0
  49. nmt_forge/training/run.py +357 -0
  50. nmt_forge/training/schedule.py +405 -0
  51. nmt_forge/training/selection.py +172 -0
  52. nmt_forge/workspace.py +36 -0
  53. nmt_forge-0.2.0.dist-info/METADATA +411 -0
  54. nmt_forge-0.2.0.dist-info/RECORD +58 -0
  55. nmt_forge-0.2.0.dist-info/WHEEL +5 -0
  56. nmt_forge-0.2.0.dist-info/entry_points.txt +2 -0
  57. nmt_forge-0.2.0.dist-info/licenses/LICENSE +133 -0
  58. nmt_forge-0.2.0.dist-info/top_level.txt +1 -0
nmt_forge/__init__.py ADDED
@@ -0,0 +1,16 @@
1
+ """nmt-forge — an NMT training suite that makes the catalogued mistakes hard.
2
+
3
+ Requirements document: the 2026-07-12 crk-translate mistake ledger (11
4
+ entries, each mistake → concrete example → guard). Failure taxonomy:
5
+ forge/docs/FAILURE_TAXONOMY.md.
6
+
7
+ The suite refuses bad practice with actionable messages (what / why / fix),
8
+ synthesizes training data only through round-trip-verified, grammar-cited
9
+ templates, and delegates ALL scoring to mt-eval.
10
+ """
11
+
12
+ __version__ = "0.2.0"
13
+
14
+ from .canonical import canonical_key, config_hash, stable_hash # noqa: F401
15
+ from .errors import ForgeError, GuardrailViolation, ResourceMissing # noqa: F401
16
+ from .workspace import Workspace # noqa: F401
nmt_forge/_harness.py ADDED
@@ -0,0 +1,255 @@
1
+ """Import shim for the eval harness (``mt-eval-harness`` on PyPI).
2
+
3
+ forge implements ZERO metrics (mistake #10 in the requirements ledger: a
4
+ bespoke evaluator partially re-implemented scoring and drifted). All scoring
5
+ delegates to ``mt_eval_harness``:
6
+
7
+ - point scores + bootstrap CIs: ``mt_eval_harness.confidence``
8
+ - A/B significance: ``mt_eval_harness.significance``
9
+ - language cards (resolver + the ONE adapter): ``mt_eval_harness.language_cards``
10
+ - RunLog / TestReport (the bridge to ``mt-eval``): ``mt_eval_harness.pipeline``
11
+ and ``mt_eval_harness.tester``
12
+
13
+ ``mt-eval-harness`` is a declared dependency of nmt-forge, so a normal
14
+ ``python3 -m pip install nmt-forge`` brings it in. Resolution order:
15
+ 1. the installed ``mt-eval-harness`` distribution (import ``mt_eval_harness``);
16
+ 2. ``$MT_EVAL_HARNESS_PATH`` (a checkout containing ``mt_eval_harness/``);
17
+ 3. monorepo fallback: the sibling ``arena/`` directory (forge lives at
18
+ ``<repo>/forge/``, the harness at ``<repo>/arena/mt_eval_harness/``).
19
+
20
+ Fails loud with install instructions — never a silent no-scores path.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import contextlib
26
+ import functools
27
+ import importlib
28
+ import io
29
+ import json
30
+ import os
31
+ import sys
32
+ from pathlib import Path
33
+
34
+
35
+ def _candidate_paths() -> list[Path]:
36
+ cands = []
37
+ env = os.environ.get("MT_EVAL_HARNESS_PATH")
38
+ if env:
39
+ cands.append(Path(env))
40
+ # <repo>/forge/nmt_forge/_harness.py → <repo>/arena
41
+ cands.append(Path(__file__).resolve().parents[2] / "arena")
42
+ return cands
43
+
44
+
45
+ def _load_from_checkout(cand: Path):
46
+ """Import ``<cand>/mt_eval_harness`` as the package WITHOUT putting
47
+ ``cand`` on sys.path.
48
+
49
+ Adding the whole ``arena/`` directory to sys.path (the pre-2026-10
50
+ fallback) also made every sibling directory importable as a namespace
51
+ package — ``arena/datasets/`` then shadowed Hugging Face ``datasets``,
52
+ transformers believed `datasets` was installed, and every HF training
53
+ run crashed inside the Trainer's dataloader. Loading the one package by
54
+ location avoids that whole class of collision."""
55
+ import importlib.util
56
+
57
+ pkg_dir = cand / "mt_eval_harness"
58
+ spec = importlib.util.spec_from_file_location(
59
+ "mt_eval_harness", pkg_dir / "__init__.py",
60
+ submodule_search_locations=[str(pkg_dir)])
61
+ if spec is None or spec.loader is None:
62
+ raise ImportError(f"cannot load mt_eval_harness from {pkg_dir}")
63
+ module = importlib.util.module_from_spec(spec)
64
+ sys.modules["mt_eval_harness"] = module
65
+ try:
66
+ spec.loader.exec_module(module)
67
+ except BaseException:
68
+ sys.modules.pop("mt_eval_harness", None)
69
+ raise
70
+ return module
71
+
72
+
73
+ def load_harness():
74
+ """Import and return the ``mt_eval_harness`` package, or raise ForgeError."""
75
+ try:
76
+ return importlib.import_module("mt_eval_harness")
77
+ except ImportError:
78
+ pass
79
+ for cand in _candidate_paths():
80
+ if (cand / "mt_eval_harness" / "__init__.py").is_file():
81
+ try:
82
+ return _load_from_checkout(cand)
83
+ except ImportError:
84
+ continue
85
+ from .errors import ForgeError
86
+
87
+ raise ForgeError(
88
+ "mt-eval-harness is not importable — forge delegates ALL scoring to "
89
+ "it and has no fallback scorer by design.\n"
90
+ " fix: `python3 -m pip install mt-eval-harness` (it is a declared dependency of "
91
+ "nmt-forge, so `python3 -m pip install nmt-forge` normally brings it in), or "
92
+ "point MT_EVAL_HARNESS_PATH at a checkout containing "
93
+ "mt_eval_harness/, or run from the Champollion monorepo where arena/ "
94
+ "is a sibling of forge/."
95
+ )
96
+
97
+
98
+ def confidence():
99
+ load_harness()
100
+ return importlib.import_module("mt_eval_harness.confidence")
101
+
102
+
103
+ def significance():
104
+ load_harness()
105
+ return importlib.import_module("mt_eval_harness.significance")
106
+
107
+
108
+ def language_cards_mod():
109
+ """The harness's language_cards module (the ONE Python card adapter)."""
110
+ load_harness()
111
+ return importlib.import_module("mt_eval_harness.language_cards")
112
+
113
+
114
+ def corpus_loader_mod():
115
+ """The harness's corpus loader (TSV is read by the harness's own rules)."""
116
+ load_harness()
117
+ return importlib.import_module("mt_eval_harness.corpus_loader")
118
+
119
+
120
+ def language_cards_remote_mod():
121
+ """The harness's remote card source (its LanguageCardsUnavailable is the
122
+ fail-loud signal forge converts into an actionable ResourceMissing)."""
123
+ load_harness()
124
+ return importlib.import_module("mt_eval_harness.language_cards_remote")
125
+
126
+
127
+ def harness_version() -> str:
128
+ """The installed harness version (recorded in exports and run logs)."""
129
+ return getattr(load_harness(), "__version__", "unknown")
130
+
131
+
132
+ def transmission_policy_mod():
133
+ """The harness's transmission policy: which corpora may reach which
134
+ model — and, since 2026-10, which corpus sentences may be PRINTED
135
+ (``withheld_text_reason`` / ``withheld_note`` / ``scrub_corpus_text``).
136
+ forge decides nothing about a corpus's terms itself."""
137
+ load_harness()
138
+ return importlib.import_module("mt_eval_harness.transmission_policy")
139
+
140
+
141
+ def corpus_terms(path) -> dict:
142
+ """What the harness reads about a corpus file's terms and identity,
143
+ WITHOUT scoring anything: ``{"meta", "envelope", "notes"}``.
144
+
145
+ ``envelope`` is a harness-JSON corpus's own ``dataset`` block (``{}`` for
146
+ TSV / JSONL / a bare list); ``meta`` is that block merged with the
147
+ steward's ``<file>.champollion.json`` sidecar and the corpora card it
148
+ registers (``corpus_loader.merge_steward_sidecar`` — the same merge an
149
+ ``mt-eval run`` on the file does: ``transmission``, a sealed
150
+ ``segment``, ``license``, ``id``, ``corpus_card``, ``contamination``);
151
+ ``notes`` is what the harness printed while merging (e.g. why a card's
152
+ id was not applied). Raises ValueError for an unreadable sidecar or
153
+ envelope — a steward wrote it to restrict the data, so a typo must not
154
+ unlock it.
155
+ """
156
+ cl = corpus_loader_mod()
157
+ p = Path(path)
158
+ envelope: dict = {}
159
+ if p.suffix.lower() == ".json" and p.is_file():
160
+ try:
161
+ _, envelope = cl._load_harness_json(p, None)
162
+ except SystemExit as exc: # the loader's own "not valid JSON" exit
163
+ raise ValueError(" ".join(str(exc).split())) from None
164
+ envelope = dict(envelope) if isinstance(envelope, dict) else {}
165
+ buf = io.StringIO()
166
+ with contextlib.redirect_stdout(buf):
167
+ meta = cl.merge_steward_sidecar(p, dict(envelope))
168
+ notes = [" ".join(ln.split()) for ln in buf.getvalue().splitlines()
169
+ if ln.strip()]
170
+ return {"meta": meta, "envelope": envelope, "notes": notes}
171
+
172
+
173
+ def card_not_applied_reason(path) -> str:
174
+ """Why the corpora card a sidecar names was NOT applied to this file
175
+ (moved, unreadable, no id, or the file changed since registration — the
176
+ harness's own words), or ``""`` (no card named, or it applies)."""
177
+ cl = corpus_loader_mod()
178
+ side = cl.read_steward_sidecar(path)
179
+ if not side.get("card"):
180
+ return ""
181
+ buf = io.StringIO()
182
+ with contextlib.redirect_stdout(buf):
183
+ card = cl.registered_card(path, side)
184
+ if card is not None:
185
+ return ""
186
+ text = " ".join(buf.getvalue().split())
187
+ return text.removeprefix("Steward: ").strip() or (
188
+ f"the card {side['card']} named in the sidecar was not applied")
189
+
190
+
191
+ @functools.lru_cache(maxsize=512)
192
+ def _registry_entry(dataset_id: str, corpus_path: str, meta_key: str):
193
+ """``publish.registry_entry_for_run`` (id, then the path/basename
194
+ ladder), memoized per process: one lookup reads the 5,600-entry
195
+ registry, and a command may ask about the same file several times."""
196
+ load_harness()
197
+ from mt_eval_harness.publish import registry_entry_for_run
198
+
199
+ return registry_entry_for_run(dataset_id, corpus_path=corpus_path,
200
+ corpus_meta=json.loads(meta_key))
201
+
202
+
203
+ def corpus_transmission_policy(path, *, dataset_id: str = ""):
204
+ """The harness's TransmissionPolicy for a corpus file, resolved exactly
205
+ as a run on it would be: the file's own terms (``corpus_terms``) and
206
+ its mt-eval registry entry (by ``dataset_id``, then the path ladder).
207
+ Raises ValueError when the file's terms are unreadable."""
208
+ tp = transmission_policy_mod()
209
+ meta = corpus_terms(path)["meta"]
210
+ p = str(Path(path).resolve())
211
+ did, entry = _registry_entry(
212
+ str(dataset_id or ""), p,
213
+ json.dumps(meta, sort_keys=True, ensure_ascii=False, default=str))
214
+ return tp.resolve_transmission_policy(did, registry_entry=entry,
215
+ corpus_meta=meta)
216
+
217
+
218
+ def derived_mark(path, *, dataset_id: str = "") -> dict:
219
+ """The terms a file DERIVED from this corpus must carry, as the harness
220
+ decides them (``corpus_loader.derived_mark``, the mark ``mt-eval`` puts
221
+ on the run logs and reports it writes), or ``{}``.
222
+
223
+ Read over the policy a run on the file would resolve — the file's own
224
+ terms AND its mt-eval registry entry — so a set the registry seals or
225
+ quarantines, or whose consent-required licence only the registry
226
+ records, is marked too. ``{}`` from a harness that predates the helper
227
+ (forge then carries the file's own terms alone). Raises ValueError when
228
+ the file's terms are unreadable."""
229
+ cl = corpus_loader_mod()
230
+ helper = getattr(cl, "derived_mark", None)
231
+ if helper is None:
232
+ return {}
233
+ policy = corpus_transmission_policy(path, dataset_id=dataset_id)
234
+ meta = corpus_terms(path)["meta"]
235
+ return dict(helper({"config": {"corpus_path": str(path),
236
+ "transmission_policy":
237
+ policy.as_provenance()},
238
+ "provenance": {"dataset_meta": meta}}))
239
+
240
+
241
+ def withheld_text_reason(path, *, dataset_id: str = "") -> str:
242
+ """Why forge must not PRINT this corpus file's sentences, or ``""``.
243
+
244
+ The harness decides (``transmission_policy.withheld_text_reason``): a
245
+ steward's local-only mark (sidecar or JSON envelope), or a policy that
246
+ refuses remote models (sealed segment, quarantined, consent-required
247
+ license). An unreadable sidecar/envelope counts as a reason."""
248
+ tp = transmission_policy_mod()
249
+ try:
250
+ policy = corpus_transmission_policy(path, dataset_id=dataset_id)
251
+ except ValueError as exc:
252
+ return f"its terms could not be read ({exc})"
253
+ return tp.withheld_text_reason({"config": {
254
+ "corpus_path": str(path),
255
+ "transmission_policy": policy.as_provenance()}})