paratext-cli 0.2.0__tar.gz → 0.2.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/PKG-INFO +4 -2
  2. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/README.md +3 -1
  3. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/paratext.example.toml +5 -0
  4. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/pyproject.toml +1 -1
  5. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/__init__.py +1 -1
  6. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/cli.py +5 -4
  7. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/config.py +2 -1
  8. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/datasets.py +49 -0
  9. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/extract.py +3 -1
  10. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/server.py +9 -0
  11. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/static/app.js +98 -1
  12. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_config.py +8 -0
  13. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_rounds.py +62 -0
  14. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/uv.lock +1 -1
  15. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/.github/workflows/ci.yml +0 -0
  16. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/.github/workflows/publish.yml +0 -0
  17. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/.gitignore +0 -0
  18. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/AGENTS.md +0 -0
  19. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/LICENSE +0 -0
  20. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/NOTICE +0 -0
  21. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/assets/logo.png +0 -0
  22. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/assets/logo.svg +0 -0
  23. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/configuration.md +0 -0
  24. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/export.md +0 -0
  25. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/green-scheduling.md +0 -0
  26. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/hf-export-spec.md +0 -0
  27. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/publishing.md +0 -0
  28. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/docs/scanned-cards.md +0 -0
  29. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/AGENTS.md +0 -0
  30. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/carbon.py +0 -0
  31. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/cards.py +0 -0
  32. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/catalogue.py +0 -0
  33. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/hf_export.py +0 -0
  34. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/inspect.py +0 -0
  35. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/io.py +0 -0
  36. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/packaging.py +0 -0
  37. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/paratext.example.toml +0 -0
  38. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/projects/__init__.py +0 -0
  39. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/projects/card_template/__init__.py +0 -0
  40. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/projects/card_template/prompt.md +0 -0
  41. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/projects/card_template/schema.py +0 -0
  42. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/records.py +0 -0
  43. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/__init__.py +0 -0
  44. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/hf_oauth.py +0 -0
  45. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/static/favicon.svg +0 -0
  46. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/static/index.html +0 -0
  47. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/static/oat.min.css +0 -0
  48. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/review/static/oat.min.js +0 -0
  49. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/runner.py +0 -0
  50. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/scaffold.py +0 -0
  51. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/sources.py +0 -0
  52. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/src/paratext/store.py +0 -0
  53. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_carbon.py +0 -0
  54. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_catalogue.py +0 -0
  55. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_cli.py +0 -0
  56. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_extract.py +0 -0
  57. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_guide.py +0 -0
  58. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_hf_export.py +0 -0
  59. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_io.py +0 -0
  60. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_packaging.py +0 -0
  61. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_project_audit.py +0 -0
  62. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_records.py +0 -0
  63. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_review.py +0 -0
  64. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_runner.py +0 -0
  65. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_scaffold.py +0 -0
  66. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_show_through.py +0 -0
  67. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_sources.py +0 -0
  68. {paratext_cli-0.2.0 → paratext_cli-0.2.2}/tests/test_verso.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: paratext-cli
3
- Version: 0.2.0
3
+ Version: 0.2.2
4
4
  Summary: Modular, project-based metadata extraction from digitised library/archive collections with a multimodal model
5
5
  License-Expression: Apache-2.0
6
6
  License-File: LICENSE
@@ -159,7 +159,9 @@ prompt version, keyed on the prompt's hash:
159
159
  `-r3`, …). The UI shows the two most recent rounds side by side and highlights
160
160
  what changed. The flag is needed because a run resumes on sample id: without
161
161
  it the existing extractions are already there, so the model is never called.
162
- paratext stops and says so rather than resuming into a stale file.
162
+ paratext stops and says so rather than resuming into a stale file. On a small
163
+ collection, `re-extract = true` in `paratext.toml` makes it the default and
164
+ the loop needs no flag.
163
165
  - **Re-run the same prompt** (a resume, or a bigger `--limit`) → the current round
164
166
  is updated in place, keeping the annotations you've already made.
165
167
 
@@ -134,7 +134,9 @@ prompt version, keyed on the prompt's hash:
134
134
  `-r3`, …). The UI shows the two most recent rounds side by side and highlights
135
135
  what changed. The flag is needed because a run resumes on sample id: without
136
136
  it the existing extractions are already there, so the model is never called.
137
- paratext stops and says so rather than resuming into a stale file.
137
+ paratext stops and says so rather than resuming into a stale file. On a small
138
+ collection, `re-extract = true` in `paratext.toml` makes it the default and
139
+ the loop needs no flag.
138
140
  - **Re-run the same prompt** (a resume, or a bigger `--limit`) → the current round
139
141
  is updated in place, keeping the annotations you've already made.
140
142
 
@@ -19,6 +19,11 @@ output = "output/cards.jsonl"
19
19
  # prompt no longer opens a new round to compare against. Leave it unset unless
20
20
  # that's what you want.
21
21
  # review-out = "review/cards"
22
+ # Editing the prompt makes existing extractions stale, and paratext stops rather
23
+ # than resuming into them. On a small collection where a re-run is cheap, this
24
+ # discards them and calls the model again instead — the run/edit/re-run loop
25
+ # then needs no flag. Leave it off for anything big or metered.
26
+ # re-extract = true
22
27
 
23
28
  # Verbatim passthrough into the request body — for provider parameters paratext
24
29
  # doesn't model. Reasoning controls live here, and every provider spells them
@@ -3,7 +3,7 @@
3
3
  # the plain name on PyPI belongs to an unrelated project, so the distribution
4
4
  # carries the `-cli` suffix.
5
5
  name = "paratext-cli"
6
- version = "0.2.0"
6
+ version = "0.2.2"
7
7
  description = "Modular, project-based metadata extraction from digitised library/archive collections with a multimodal model"
8
8
  readme = "README.md"
9
9
  # Core is pure-Python + numpy/pillow/pydantic. The [detector] extra needs
@@ -1,3 +1,3 @@
1
1
  """paratext — metadata extraction from digitised collections with a multimodal model."""
2
2
 
3
- __version__ = "0.2.0"
3
+ __version__ = "0.2.2"
@@ -61,6 +61,7 @@ HARDCODED_DEFAULTS: dict = {
61
61
  "review_port": DEFAULT_PORT,
62
62
  "no_structured": False,
63
63
  "skip_preflight": False,
64
+ "re_extract": False,
64
65
  "max_tokens": None, # None → runner.DEFAULT_MAX_TOKENS, or the project's own
65
66
  }
66
67
 
@@ -214,7 +215,7 @@ def _do_extract(args: argparse.Namespace) -> None:
214
215
  energy=energy,
215
216
  max_tokens=getattr(args, "max_tokens", None),
216
217
  extra_body=config_extra_body(args.project),
217
- re_extract=getattr(args, "re_extract", False),
218
+ re_extract=args.re_extract,
218
219
  )
219
220
 
220
221
 
@@ -675,9 +676,9 @@ def _add_extract_args(p: argparse.ArgumentParser) -> None:
675
676
  p.add_argument("--base-url", default=None, help="model endpoint base URL (OpenAI-compatible)")
676
677
  p.add_argument("--api-key", default=None, help="API key for the endpoint (often unused)")
677
678
  p.add_argument("--limit", type=int, default=None, help="Process at most N inputs")
678
- p.add_argument("--re-extract", action="store_true",
679
- help="Discard existing extractions and call the model again "
680
- "(needed when the prompt or model has changed)")
679
+ p.add_argument("--re-extract", action="store_true", default=None,
680
+ help="Discard existing extractions and call the model again when the "
681
+ "prompt or model has changed (config re-extract, else stop and say so)")
681
682
  p.add_argument("--max-tokens", type=int, default=None, metavar="N",
682
683
  help="Output-token ceiling per call (default: 8192; raise it for "
683
684
  "models that reason before answering)")
@@ -62,6 +62,7 @@ RECOGNISED = (
62
62
  "limit",
63
63
  "no_structured",
64
64
  "skip_preflight",
65
+ "re_extract",
65
66
  "max_tokens",
66
67
  )
67
68
 
@@ -232,7 +233,7 @@ def coerce_paths(d: dict) -> dict:
232
233
  for key in ("limit", "review_port", "max_tokens"):
233
234
  if key in d and isinstance(d[key], str):
234
235
  d[key] = int(d[key])
235
- for key in ("no_structured", "skip_preflight"):
236
+ for key in ("no_structured", "skip_preflight", "re_extract"):
236
237
  if key in d and isinstance(d[key], str):
237
238
  d[key] = d[key].lower() in ("1", "true", "yes", "on")
238
239
  return d
@@ -176,6 +176,55 @@ def load_view(dataset: dict, samples: list[dict]) -> dict:
176
176
  return synthesise_view(dataset, samples)
177
177
 
178
178
 
179
+ def _model_fields(view: dict) -> list[dict]:
180
+ """The model-output field list from a view contract, key/label/type only."""
181
+ for panel in view.get("panels", []):
182
+ if panel.get("source") == "model_output":
183
+ return [
184
+ {k: f.get(k) for k in ("key", "label", "type")}
185
+ for f in panel.get("fields", [])
186
+ ]
187
+ return []
188
+
189
+
190
+ def diff_fields(before: list[dict], after: list[dict]) -> dict:
191
+ """What changed between two rounds' field lists, keyed on field name."""
192
+ b = {f["key"]: f for f in before}
193
+ a = {f["key"]: f for f in after}
194
+ return {
195
+ "added": [a[k] for k in a if k not in b],
196
+ "removed": [b[k] for k in b if k not in a],
197
+ "retyped": [
198
+ {"key": k, "label": a[k].get("label"),
199
+ "from": b[k].get("type"), "to": a[k].get("type")}
200
+ for k in a
201
+ if k in b and a[k].get("type") != b[k].get("type")
202
+ ],
203
+ }
204
+
205
+
206
+ def schema_history(datasets: list[dict]) -> list[dict]:
207
+ """Field list per round for one dataset family, oldest first, each carrying
208
+ what changed since the round before it (`changes` is None for the first)."""
209
+ rounds = []
210
+ for ds in sorted(datasets, key=lambda d: d.get("round") or 0):
211
+ vp = ds["dir"] / "view.json"
212
+ view = json.loads(vp.read_text()) if vp.is_file() else synthesise_view(
213
+ ds, load_samples(ds)
214
+ )
215
+ rounds.append({
216
+ "round": ds.get("round"),
217
+ "dataset": ds["name"],
218
+ "schema_version": view.get("schema_version"),
219
+ "fields": _model_fields(view),
220
+ })
221
+ prev = None
222
+ for r in rounds:
223
+ r["changes"] = diff_fields(prev["fields"], r["fields"]) if prev else None
224
+ prev = r
225
+ return rounds
226
+
227
+
179
228
  def review_stats(
180
229
  total: int, annotations: list[dict], gold_ids: set[str] | None = None
181
230
  ) -> dict:
@@ -78,7 +78,9 @@ def _guard_stale_output(output: Path, header: dict, project: str, re_extract: bo
78
78
  f"Those records answer a different question, and resume would skip every\n"
79
79
  f"sample and call the model zero times. Either:\n"
80
80
  f" paratext run -p {project} --re-extract # redo them\n"
81
- f" paratext run -p {project} --output <new>.jsonl # keep both"
81
+ f" paratext run -p {project} --output <new>.jsonl # keep both\n"
82
+ f"On a small, cheap collection you can make the first the default:\n"
83
+ f" re-extract = true under [project.{project}] in paratext.toml"
82
84
  )
83
85
 
84
86
 
@@ -29,6 +29,7 @@ from ..datasets import (
29
29
  load_view,
30
30
  resolve_dataset,
31
31
  review_stats,
32
+ schema_history,
32
33
  )
33
34
  from ..store import Store, default_db_path
34
35
 
@@ -126,6 +127,8 @@ class Handler(BaseHTTPRequestHandler):
126
127
  return self._api_projects()
127
128
  if path == "/api/prompts":
128
129
  return self._api_prompts(self._dataset(qs))
130
+ if path == "/api/schema":
131
+ return self._api_schema(self._dataset(qs))
129
132
  if path == "/api/export/fields":
130
133
  return self._api_export_fields(self._dataset(qs), qs)
131
134
  if path in ("/api/export/marc", "/api/export/dc"):
@@ -299,6 +302,12 @@ class Handler(BaseHTTPRequestHandler):
299
302
  )
300
303
  self._json(rows)
301
304
 
305
+ def _api_schema(self, ds):
306
+ siblings = [d for d in discover_datasets(self.data_dir) if d["base"] == ds["base"]]
307
+ self._json(
308
+ {"dataset": ds["name"], "base": ds["base"], "rounds": schema_history(siblings)}
309
+ )
310
+
302
311
  def _api_prompts(self, ds):
303
312
  siblings = [d for d in discover_datasets(self.data_dir) if d["base"] == ds["base"]]
304
313
  groups: dict[str, dict] = {}
@@ -1047,15 +1047,110 @@ function renderPromptsPanel(prompts) {
1047
1047
  `;
1048
1048
  }
1049
1049
 
1050
+
1051
+ // ── Fields panel ──────────────────────────────────────────────────────
1052
+ // The schema in the terms someone meeting one for the first time can read:
1053
+ // what the model is asked to fill in, and what changed since last round.
1054
+ // A table, not a +/- diff — the audience for this is learning what a field is.
1055
+
1056
+ const TYPE_WORDS = {
1057
+ string: "text",
1058
+ integer: "whole number",
1059
+ number: "number",
1060
+ boolean: "yes / no",
1061
+ array: "list",
1062
+ object: "group",
1063
+ };
1064
+
1065
+ const typeWord = (t) => TYPE_WORDS[t] ?? (t ?? "text");
1066
+
1067
+ function renderFieldsPanel(rounds) {
1068
+ if (!rounds || !rounds.length) return "";
1069
+ const latest = rounds[rounds.length - 1];
1070
+ if (!latest.fields.length) return "";
1071
+
1072
+ // Where each field entered, so an unchanged row can still say "since r1".
1073
+ const firstSeen = new Map();
1074
+ for (const r of rounds) {
1075
+ for (const f of r.fields) if (!firstSeen.has(f.key)) firstSeen.set(f.key, r.round);
1076
+ }
1077
+ const c = latest.changes ?? { added: [], removed: [], retyped: [] };
1078
+ const addedKeys = new Set(c.added.map((f) => f.key));
1079
+ const retyped = new Map(c.retyped.map((f) => [f.key, f]));
1080
+
1081
+ const multi = rounds.length > 1;
1082
+ const status = (f) => {
1083
+ if (!multi) return "";
1084
+ if (addedKeys.has(f.key))
1085
+ return `<span class="ok">new this round</span>`;
1086
+ const rt = retyped.get(f.key);
1087
+ if (rt)
1088
+ return `<span class="warn">was ${escapeHtml(typeWord(rt.from))}</span>`;
1089
+ const seen = firstSeen.get(f.key);
1090
+ return `<span class="text-light">since round ${seen ?? 1}</span>`;
1091
+ };
1092
+
1093
+ const rows = latest.fields
1094
+ .map(
1095
+ (f) => `<tr>
1096
+ <td><code>${escapeHtml(f.key)}</code></td>
1097
+ <td>${escapeHtml(f.label ?? f.key)}</td>
1098
+ <td>${escapeHtml(typeWord(f.type))}</td>
1099
+ <td>${status(f)}</td>
1100
+ </tr>`,
1101
+ )
1102
+ .join("");
1103
+
1104
+ const gone = (c.removed ?? [])
1105
+ .map(
1106
+ (f) =>
1107
+ `<tr class="text-light"><td><code>${escapeHtml(f.key)}</code></td>
1108
+ <td>${escapeHtml(f.label ?? f.key)}</td>
1109
+ <td>${escapeHtml(typeWord(f.type))}</td>
1110
+ <td class="bad">dropped this round</td></tr>`,
1111
+ )
1112
+ .join("");
1113
+
1114
+ const changed =
1115
+ (c.added?.length ?? 0) + (c.removed?.length ?? 0) + (c.retyped?.length ?? 0);
1116
+ const summary =
1117
+ rounds.length < 2
1118
+ ? `${latest.fields.length} field${latest.fields.length === 1 ? "" : "s"}`
1119
+ : changed === 0
1120
+ ? `${latest.fields.length} fields · unchanged since round ${
1121
+ rounds[rounds.length - 2].round
1122
+ }`
1123
+ : `${latest.fields.length} fields · ${changed} change${
1124
+ changed === 1 ? "" : "s"
1125
+ } this round`;
1126
+
1127
+ return `<details style="margin:1rem 0;" ${changed ? "open" : ""}>
1128
+ <summary><strong>Fields</strong> <small style="color:var(--muted-foreground);">(${escapeHtml(
1129
+ summary,
1130
+ )})</small></summary>
1131
+ <p class="text-light mt-2" style="font-size:.875rem;">
1132
+ One row per piece of metadata the model is asked to produce for every card.
1133
+ Together they are the <em>schema</em>. Change them in
1134
+ <code>schema.py</code>; describe them in <code>prompt.md</code>.
1135
+ </p>
1136
+ <div class="table"><table>
1137
+ <thead><tr><th>Field</th><th>Shown as</th><th>Holds</th><th></th></tr></thead>
1138
+ <tbody>${rows}${gone}</tbody>
1139
+ </table></div>
1140
+ </details>`;
1141
+ }
1142
+
1050
1143
  async function renderStats() {
1051
- const [statsRes, tableRes, promptsRes] = await Promise.all([
1144
+ const [statsRes, tableRes, promptsRes, schemaRes] = await Promise.all([
1052
1145
  fetch(api("api/stats")),
1053
1146
  fetch(api("api/table")),
1054
1147
  fetch(api("api/prompts")),
1148
+ fetch(api("api/schema")),
1055
1149
  ]);
1056
1150
  const s = await statsRes.json();
1057
1151
  const rows = await tableRes.json();
1058
1152
  const promptsData = await promptsRes.json();
1153
+ const schemaData = await schemaRes.json();
1059
1154
  document.getElementById("progress").textContent = "";
1060
1155
 
1061
1156
  const badge = (v) => {
@@ -1104,6 +1199,8 @@ async function renderStats() {
1104
1199
  <a href="#/eval">build →</a>)</small></dd></div>
1105
1200
  </dl>
1106
1201
 
1202
+ ${renderFieldsPanel(schemaData.rounds ?? [])}
1203
+
1107
1204
  ${renderPromptsPanel(promptsData.prompts ?? [])}
1108
1205
 
1109
1206
  <div class="controls">
@@ -162,3 +162,11 @@ def test_max_tokens_from_the_environment(tmp_path, monkeypatch):
162
162
  monkeypatch.chdir(tmp_path)
163
163
  monkeypatch.setenv("PARATEXT_MAX_TOKENS", "16384")
164
164
  assert cfg.coerce_paths(cfg.load_defaults(None))["max_tokens"] == 16384
165
+
166
+
167
+ def test_re_extract_is_recognised_and_coerced():
168
+ from paratext.config import RECOGNISED, coerce_paths
169
+
170
+ assert "re_extract" in RECOGNISED
171
+ assert coerce_paths({"re_extract": "true"})["re_extract"] is True
172
+ assert coerce_paths({"re_extract": "no"})["re_extract"] is False
@@ -130,3 +130,65 @@ def test_changed_prompt_reports_why(tmp_path, monkeypatch):
130
130
  _mk_round(review, "demo", 1, "hashA", model="m")
131
131
  _, _, reuse, reason = cli._resolve_round("demo", "hashB", None, model="m")
132
132
  assert reuse is False and reason == "the prompt changed"
133
+
134
+
135
+ def _view(fields, version="v1"):
136
+ return {
137
+ "contract_version": 1,
138
+ "schema": "cards",
139
+ "schema_version": version,
140
+ "panels": [{"source": "model_output", "title": "Model output", "fields": fields}],
141
+ }
142
+
143
+
144
+ def _round_dir(tmp_path, name, n, fields):
145
+ import json
146
+
147
+ d = tmp_path / name
148
+ d.mkdir()
149
+ (d / "view.json").write_text(json.dumps(_view(fields)))
150
+ (d / "samples.json").write_text("[]")
151
+ return {"name": name, "base": "cards", "round": n, "dir": d}
152
+
153
+
154
+ def test_diff_fields_reports_added_removed_and_retyped():
155
+ from paratext.datasets import diff_fields
156
+
157
+ before = [{"key": "a", "label": "A", "type": "string"},
158
+ {"key": "b", "label": "B", "type": "string"}]
159
+ after = [{"key": "a", "label": "A", "type": "integer"},
160
+ {"key": "c", "label": "C", "type": "string"}]
161
+ d = diff_fields(before, after)
162
+ assert [f["key"] for f in d["added"]] == ["c"]
163
+ assert [f["key"] for f in d["removed"]] == ["b"]
164
+ assert d["retyped"] == [{"key": "a", "label": "A", "from": "string", "to": "integer"}]
165
+
166
+
167
+ def test_diff_fields_is_empty_when_nothing_moved():
168
+ from paratext.datasets import diff_fields
169
+
170
+ fields = [{"key": "a", "label": "A", "type": "string"}]
171
+ assert diff_fields(fields, list(fields)) == {"added": [], "removed": [], "retyped": []}
172
+
173
+
174
+ def test_schema_history_orders_rounds_and_diffs_each_against_the_last(tmp_path):
175
+ from paratext.datasets import schema_history
176
+
177
+ r1 = _round_dir(tmp_path, "cards-r1", 1, [{"key": "a", "label": "A", "type": "string"}])
178
+ r2 = _round_dir(tmp_path, "cards-r2", 2, [
179
+ {"key": "a", "label": "A", "type": "string"},
180
+ {"key": "b", "label": "B", "type": "integer"},
181
+ ])
182
+ # Out of order in, oldest first out.
183
+ hist = schema_history([r2, r1])
184
+ assert [r["round"] for r in hist] == [1, 2]
185
+ assert hist[0]["changes"] is None # nothing to compare the first round to
186
+ assert [f["key"] for f in hist[1]["changes"]["added"]] == ["b"]
187
+
188
+
189
+ def test_schema_history_of_a_single_round(tmp_path):
190
+ from paratext.datasets import schema_history
191
+
192
+ r1 = _round_dir(tmp_path, "cards-r1", 1, [{"key": "a", "label": "A", "type": "string"}])
193
+ hist = schema_history([r1])
194
+ assert len(hist) == 1 and hist[0]["changes"] is None
@@ -803,7 +803,7 @@ wheels = [
803
803
 
804
804
  [[package]]
805
805
  name = "paratext-cli"
806
- version = "0.2.0"
806
+ version = "0.2.2"
807
807
  source = { editable = "." }
808
808
  dependencies = [
809
809
  { name = "huggingface-hub" },
File without changes
File without changes
File without changes
File without changes