cellar-extractor 2.0.2__tar.gz → 2.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {cellar_extractor-2.0.2/cellar_extractor.egg-info → cellar_extractor-2.0.3}/PKG-INFO +22 -1
  2. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/README.md +21 -0
  3. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/__init__.py +3 -0
  4. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/_version.py +3 -3
  5. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/eurlex_scraping.py +72 -14
  6. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3/cellar_extractor.egg-info}/PKG-INFO +22 -1
  7. cellar_extractor-2.0.3/cellar_extractor.egg-info/scm_version.json +8 -0
  8. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_metadata_hygiene_local.py +34 -0
  9. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector6_cellar_supplement.py +12 -18
  10. cellar_extractor-2.0.2/cellar_extractor.egg-info/scm_version.json +0 -8
  11. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.env.example +0 -0
  12. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.flake8 +0 -0
  13. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.github/workflows/ci.yml +0 -0
  14. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.github/workflows/github-actions.yml +0 -0
  15. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/FIELDS.md +0 -0
  16. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/LICENSE +0 -0
  17. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/MANIFEST.in +0 -0
  18. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/build_package.py +0 -0
  19. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar.py +0 -0
  20. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_extra_extract.py +0 -0
  21. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_queries.py +0 -0
  22. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_sparql_queries.py +0 -0
  23. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/citations_adder.py +0 -0
  24. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/csv_extractor.py +0 -0
  25. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/fulltext_saving.py +0 -0
  26. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/json_to_csv.py +0 -0
  27. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/nodes_and_edges.py +0 -0
  28. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/operative_extractions.py +0 -0
  29. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/persistence.py +0 -0
  30. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/schema.py +0 -0
  31. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/sparql.py +0 -0
  32. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/SOURCES.txt +0 -0
  33. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/dependency_links.txt +0 -0
  34. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/requires.txt +0 -0
  35. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/scm_file_list.json +0 -0
  36. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/top_level.txt +0 -0
  37. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/pyproject.toml +0 -0
  38. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/requirements.txt +0 -0
  39. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/setup.cfg +0 -0
  40. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/setup.py +0 -0
  41. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/61986CJ0062.ENG.txt +0 -0
  42. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar.py +0 -0
  43. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_integration.py +0 -0
  44. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_queries_local.py +0 -0
  45. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_sparql_queries.py +0 -0
  46. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_citation_graph_integration.py +0 -0
  47. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_citations_adder_local.py +0 -0
  48. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_corpus_2020_integration.py +0 -0
  49. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_extra_cellar_local.py +0 -0
  50. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_fulltext_saving_local.py +0 -0
  51. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_infocuria_adapter.py +0 -0
  52. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_infocuria_integration.py +0 -0
  53. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_multilang_fulltext.py +0 -0
  54. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_nodes_and_edges_local.py +0 -0
  55. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_operative_extractions_local.py +0 -0
  56. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_real_fetch_integration.py +0 -0
  57. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_retry_behavior.py +0 -0
  58. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_samples_dump_integration.py +0 -0
  59. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_schema_local.py +0 -0
  60. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector3_adapter.py +0 -0
  61. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector6_cellar_fallback.py +0 -0
  62. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector8_adapter.py +0 -0
  63. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector8_integration.py +0 -0
  64. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sparql_local.py +0 -0
  65. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_credentials_integration.py +0 -0
  66. {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_redundancy_integration.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cellar-extractor
3
- Version: 2.0.2
3
+ Version: 2.0.3
4
4
  Summary: Library for extracting CELLAR case law data from EUR-Lex
5
5
  Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
6
6
  License: Apache-2.0
@@ -127,6 +127,11 @@ df = cell.get_cellar(
127
127
 
128
128
  Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
129
129
 
130
+ For direct manifestation access, use
131
+ `get_cellar_manifestations_by_celex()`. It canonicalizes composite and
132
+ `_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
133
+ derived summaries or notices from being returned as judgment full text.
134
+
130
135
  You can also save explicitly to a custom path instead of the default `data/` location:
131
136
 
132
137
  ```python
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
153
158
  )
154
159
  ```
155
160
 
161
+ For targeted CELEX repair or supplementation, use the public CELLAR
162
+ manifestation API. It unions every CELLAR work sharing the CELEX before choosing
163
+ the best downloadable item per language:
164
+
165
+ ```python
166
+ import cellar_extractor as cell
167
+
168
+ _, manifestations = cell.get_cellar_manifestations_by_celex(
169
+ "62020CJ0414", sector="6"
170
+ )
171
+ english = [m for m in manifestations if m["language"] == "EN"]
172
+ fulltexts = cell.extract_cellar_fulltexts(english)
173
+ ```
174
+
156
175
  Returns:
157
176
 
158
177
  - `extra_df`: enriched dataframe
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
303
322
  | `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
304
323
  | `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
305
324
  | `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
325
+ | `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
326
+ | `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
306
327
  | `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
307
328
  | `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
308
329
  | `FetchOperativePart` | Extract operative part from a single case document |
@@ -92,6 +92,11 @@ df = cell.get_cellar(
92
92
 
93
93
  Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
94
94
 
95
+ For direct manifestation access, use
96
+ `get_cellar_manifestations_by_celex()`. It canonicalizes composite and
97
+ `_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
98
+ derived summaries or notices from being returned as judgment full text.
99
+
95
100
  You can also save explicitly to a custom path instead of the default `data/` location:
96
101
 
97
102
  ```python
@@ -118,6 +123,20 @@ extra_df, fulltext = cell.get_cellar_extra(
118
123
  )
119
124
  ```
120
125
 
126
+ For targeted CELEX repair or supplementation, use the public CELLAR
127
+ manifestation API. It unions every CELLAR work sharing the CELEX before choosing
128
+ the best downloadable item per language:
129
+
130
+ ```python
131
+ import cellar_extractor as cell
132
+
133
+ _, manifestations = cell.get_cellar_manifestations_by_celex(
134
+ "62020CJ0414", sector="6"
135
+ )
136
+ english = [m for m in manifestations if m["language"] == "EN"]
137
+ fulltexts = cell.extract_cellar_fulltexts(english)
138
+ ```
139
+
121
140
  Returns:
122
141
 
123
142
  - `extra_df`: enriched dataframe
@@ -268,6 +287,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
268
287
  | `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
269
288
  | `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
270
289
  | `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
290
+ | `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
291
+ | `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
271
292
  | `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
272
293
  | `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
273
294
  | `FetchOperativePart` | Extract operative part from a single case document |
@@ -9,6 +9,9 @@ from cellar_extractor.cellar import get_cellar_extra
9
9
  from cellar_extractor.cellar import get_nodes_and_edges_lists
10
10
  from cellar_extractor.cellar import filter_subject_matter
11
11
  from cellar_extractor.eurlex_scraping import get_legislation_by_celex_id
12
+ from cellar_extractor.eurlex_scraping import get_cellar_manifestations_by_celex
13
+ from cellar_extractor.eurlex_scraping import extract_cellar_fulltexts
14
+ from cellar_extractor.eurlex_scraping import normalize_celex
12
15
  from cellar_extractor.operative_extractions import FetchOperativePart
13
16
  from cellar_extractor.operative_extractions import Writing
14
17
  import logging
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '2.0.2'
22
- __version_tuple__ = version_tuple = (2, 0, 2)
21
+ __version__ = version = '2.0.3'
22
+ __version_tuple__ = version_tuple = (2, 0, 3)
23
23
 
24
- __commit_id__ = commit_id = 'gf354af594'
24
+ __commit_id__ = commit_id = 'g1aaeae809'
@@ -129,6 +129,17 @@ def _normalize_celex(celex):
129
129
  return value
130
130
 
131
131
 
132
+ def normalize_celex(celex):
133
+ """Return the canonical work CELEX used for full-text retrieval.
134
+
135
+ CELLAR metadata can bundle a primary document with derived information,
136
+ summary, and résumé works (for example ``62020CJ0414_SUM;62020CJ0414``).
137
+ Full-text callers must resolve the unsuffixed base work so a derived work
138
+ can never occupy the judgment's language slot.
139
+ """
140
+ return _normalize_celex(celex)
141
+
142
+
132
143
  def _sleep_with_backoff(attempt, base=0.2):
133
144
  time.sleep(base * (attempt + 1) + random.uniform(0.0, 0.1))
134
145
 
@@ -413,6 +424,25 @@ def _fetch_sector8_items_for_celex(celex, sector="8"):
413
424
  return work_uris, candidates
414
425
 
415
426
 
427
+ def get_cellar_manifestations_by_celex(celex, sector="8"):
428
+ """Return all CELLAR works and manifestation candidates for a CELEX.
429
+
430
+ A CELEX can resolve to multiple CELLAR works with different language
431
+ coverage. This public entry point deliberately returns the union across
432
+ every matching work so callers do not need to depend on the extractor's
433
+ private sector-8 helpers.
434
+
435
+ The return value is ``(work_uris, manifestations)``. Each manifestation
436
+ contains ``item_url``, ``format``, and ``language`` keys and is deduplicated
437
+ on that triple. ``sector`` defaults to ``"8"`` and may be set to ``"6"``
438
+ for CJEU documents.
439
+ """
440
+ canonical_celex = normalize_celex(celex)
441
+ if not canonical_celex:
442
+ return [], []
443
+ return _fetch_sector8_items_for_celex(canonical_celex, sector=sector)
444
+
445
+
416
446
  def _fetch_sector8_work_uri(celex, sector="8"):
417
447
  """Resolve a CELEX to its CELLAR work URI.
418
448
 
@@ -553,6 +583,17 @@ def _fanout_fulltexts_from_candidates(candidates, source_label):
553
583
  return out
554
584
 
555
585
 
586
+ def extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM"):
587
+ """Download the best manifestation per language as fulltext records.
588
+
589
+ ``manifestations`` is the candidate list returned by
590
+ :func:`get_cellar_manifestations_by_celex`. Empty bodies are omitted and
591
+ each returned dictionary contains ``text``, ``html``, ``text_source``,
592
+ ``text_language``, and ``text_format``.
593
+ """
594
+ return _fanout_fulltexts_from_candidates(manifestations, source_label)
595
+
596
+
556
597
  def _get_case_data_sector8(celex, language="EN"):
557
598
  work_uris, main_candidates = _fetch_sector8_items_for_celex(celex)
558
599
 
@@ -1104,15 +1145,14 @@ def _get_case_data_sector6(celex, language="EN"):
1104
1145
  }
1105
1146
  )
1106
1147
 
1107
- # Supplement InfoCuria fulltexts with any languages CELLAR has that
1108
- # InfoCuria didn't expose. InfoCuria's documents.searchHits typically
1109
- # carries only the procedural language plus EN; CELLAR's CDM model
1110
- # exposes all 23 EU-official manifestations via expression_uses_language.
1111
- # We keep InfoCuria's entries where languages overlap (the court's own
1112
- # publication is typically higher fidelity) and append CELLAR-sourced
1113
- # entries for any missing languages. Metadata fields (judge, advocate,
1114
- # directory_codes, etc.) remain InfoCuria-sourced — CELLAR cannot
1115
- # populate them.
1148
+ # Merge InfoCuria fulltexts with every language CELLAR has. CELLAR is
1149
+ # authoritative when both sources expose the same language because its
1150
+ # manifestation belongs to the canonical CELEX work. InfoCuria's
1151
+ # procedure search has returned a different document under the judgment
1152
+ # slot in production (AG opinions, procedural orders, and notices), so an
1153
+ # existing InfoCuria language must not block the canonical manifestation.
1154
+ # Metadata fields (judge, advocate, directory_codes, etc.) remain
1155
+ # InfoCuria-sourced — CELLAR cannot populate them.
1116
1156
  #
1117
1157
  # All-in-one try/except: a CELLAR-side failure must never kill the
1118
1158
  # InfoCuria-sourced row we already built.
@@ -1122,18 +1162,36 @@ def _get_case_data_sector6(celex, language="EN"):
1122
1162
  cellar_fulltexts = _fanout_fulltexts_from_candidates(
1123
1163
  cellar_candidates, source_label="CELLAR_ITEM"
1124
1164
  )
1165
+ by_language = {
1166
+ entry.get("text_language", "").upper(): entry
1167
+ for entry in fulltexts
1168
+ if entry.get("text_language", "")
1169
+ }
1125
1170
  for entry in cellar_fulltexts:
1126
1171
  entry_lang = entry.get("text_language", "").upper()
1127
- if not entry_lang or entry_lang in seen_langs:
1172
+ if not entry_lang:
1128
1173
  continue
1129
1174
  seen_langs.add(entry_lang)
1130
- fulltexts.append(entry)
1175
+ by_language[entry_lang] = entry
1176
+ fulltexts = list(by_language.values())
1131
1177
  except Exception:
1132
1178
  # CELLAR supplementation is best-effort. If anything goes wrong
1133
1179
  # (SPARQL timeout, network hiccup, schema drift on the manifestation
1134
1180
  # graph) the InfoCuria-only result is still returned.
1135
1181
  pass
1136
1182
 
1183
+ primary = next(
1184
+ (
1185
+ entry
1186
+ for entry in fulltexts
1187
+ if entry.get("text_language", "").upper() == str(language).upper()
1188
+ ),
1189
+ None,
1190
+ )
1191
+ if primary is not None:
1192
+ text = primary.get("text", "")
1193
+ html = primary.get("html", "")
1194
+
1137
1195
  return {
1138
1196
  "html": html,
1139
1197
  "text": text,
@@ -1146,9 +1204,9 @@ def _get_case_data_sector6(celex, language="EN"):
1146
1204
  "affecting_ids": affecting_ids,
1147
1205
  "affecting_string": affecting_string,
1148
1206
  "citations_extra": citations_extra,
1149
- "text_source": "INFOCURIA_BLOB_HTML" if text != "" else "",
1150
- "text_language": str(doc_lang).upper() if doc_lang else language.upper(),
1151
- "text_format": "html" if text != "" else "",
1207
+ "text_source": primary.get("text_source", "") if primary else "",
1208
+ "text_language": primary.get("text_language", "") if primary else "",
1209
+ "text_format": primary.get("text_format", "") if primary else "",
1152
1210
  "summary_source": summary_source,
1153
1211
  "summary_language": "EN" if summary != "" else "",
1154
1212
  "sector": "6",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cellar-extractor
3
- Version: 2.0.2
3
+ Version: 2.0.3
4
4
  Summary: Library for extracting CELLAR case law data from EUR-Lex
5
5
  Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
6
6
  License: Apache-2.0
@@ -127,6 +127,11 @@ df = cell.get_cellar(
127
127
 
128
128
  Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
129
129
 
130
+ For direct manifestation access, use
131
+ `get_cellar_manifestations_by_celex()`. It canonicalizes composite and
132
+ `_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
133
+ derived summaries or notices from being returned as judgment full text.
134
+
130
135
  You can also save explicitly to a custom path instead of the default `data/` location:
131
136
 
132
137
  ```python
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
153
158
  )
154
159
  ```
155
160
 
161
+ For targeted CELEX repair or supplementation, use the public CELLAR
162
+ manifestation API. It unions every CELLAR work sharing the CELEX before choosing
163
+ the best downloadable item per language:
164
+
165
+ ```python
166
+ import cellar_extractor as cell
167
+
168
+ _, manifestations = cell.get_cellar_manifestations_by_celex(
169
+ "62020CJ0414", sector="6"
170
+ )
171
+ english = [m for m in manifestations if m["language"] == "EN"]
172
+ fulltexts = cell.extract_cellar_fulltexts(english)
173
+ ```
174
+
156
175
  Returns:
157
176
 
158
177
  - `extra_df`: enriched dataframe
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
303
322
  | `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
304
323
  | `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
305
324
  | `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
325
+ | `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
326
+ | `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
306
327
  | `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
307
328
  | `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
308
329
  | `FetchOperativePart` | Extract operative part from a single case document |
@@ -0,0 +1,8 @@
1
+ {
2
+ "tag": "2.0.3",
3
+ "distance": 0,
4
+ "node": "g1aaeae8096046eeeb4e5ab1b7507f92cda10057c",
5
+ "dirty": false,
6
+ "branch": "HEAD",
7
+ "node_date": "2026-09-15"
8
+ }
@@ -6,6 +6,7 @@
6
6
  at the metadata assembly point, in both the celex- and ecli-keyed loops.
7
7
  """
8
8
 
9
+ import cellar_extractor as cell
9
10
  import cellar_extractor.cellar_queries as cq
10
11
  import cellar_extractor.eurlex_scraping as es
11
12
 
@@ -132,3 +133,36 @@ def test_items_for_celex_unions_across_all_works(monkeypatch):
132
133
  assert len(uris) == 2
133
134
  langs = sorted(c["language"] for c in cands)
134
135
  assert langs == ["DE", "EN", "FR", "NL"] # union, DE deduped
136
+
137
+
138
+ def test_public_manifestation_and_fulltext_api(monkeypatch):
139
+ manifestations = [
140
+ {"item_url": "u_en", "format": "xhtml", "language": "EN"},
141
+ {"item_url": "u_fr", "format": "xhtml", "language": "FR"},
142
+ ]
143
+ monkeypatch.setattr(
144
+ es,
145
+ "_fetch_sector8_items_for_celex",
146
+ lambda celex, sector="8": ([f"http://cellar/{sector}/{celex}"], manifestations),
147
+ )
148
+
149
+ works, found = cell.get_cellar_manifestations_by_celex(
150
+ "62024CJ0001_SUM;62024CJ0001", sector="6"
151
+ )
152
+ assert works == ["http://cellar/6/62024CJ0001"]
153
+ assert found == manifestations
154
+ assert cell.normalize_celex("62024CJ0001_SUM;62024CJ0001") == "62024CJ0001"
155
+
156
+ monkeypatch.setattr(
157
+ es,
158
+ "_fanout_fulltexts_from_candidates",
159
+ lambda candidates, source_label: [
160
+ {"text_language": candidate["language"], "text_source": source_label}
161
+ for candidate in candidates
162
+ ],
163
+ )
164
+ rows = cell.extract_cellar_fulltexts(found, source_label="CELLAR_ITEM")
165
+ assert rows == [
166
+ {"text_language": "EN", "text_source": "CELLAR_ITEM"},
167
+ {"text_language": "FR", "text_source": "CELLAR_ITEM"},
168
+ ]
@@ -7,8 +7,8 @@ languages for the same work.
7
7
 
8
8
  After this change, sector 6 always supplements InfoCuria's fulltexts with
9
9
  CELLAR's manifestation graph (unconditionally — not env-var gated), so the
10
- ``fulltexts`` list contains every language CELLAR has, with InfoCuria's
11
- entries kept verbatim where both sources have the same language.
10
+ ``fulltexts`` list contains every language CELLAR has, with canonical CELLAR
11
+ manifestations replacing InfoCuria entries where both sources overlap.
12
12
  """
13
13
 
14
14
  from __future__ import annotations
@@ -154,9 +154,8 @@ def test_infocuria_success_is_supplemented_with_cellar_languages(monkeypatch):
154
154
  assert langs == {"EN", "FR", "DE", "IT", "NL", "ES", "PT"}
155
155
 
156
156
 
157
- def test_infocuria_entries_are_preserved_when_languages_overlap(monkeypatch):
158
- """When both sources have the same language, keep InfoCuria's entry —
159
- it's the court's own publication and typically has higher fidelity."""
157
+ def test_cellar_entries_replace_infocuria_when_languages_overlap(monkeypatch):
158
+ """Canonical CELLAR works win overlaps to prevent wrong-doc captures."""
160
159
  eurlex_scraping._get_case_data_cached.cache_clear()
161
160
  _patch_infocuria(monkeypatch, doc_langs=["EN", "FR"])
162
161
  _patch_cellar(monkeypatch, languages=["EN", "FR", "DE"])
@@ -164,14 +163,11 @@ def test_infocuria_entries_are_preserved_when_languages_overlap(monkeypatch):
164
163
  data = eurlex_scraping._get_case_data_sector6("62024CJ0001", language="EN")
165
164
 
166
165
  by_lang = {entry["text_language"]: entry for entry in data["fulltexts"]}
167
- # The EN + FR entries should still be from InfoCuria.
168
- assert by_lang["EN"]["text_source"] == "INFOCURIA_BLOB_HTML"
169
- assert "infocuria body in EN" in by_lang["EN"]["text"]
170
- assert by_lang["FR"]["text_source"] == "INFOCURIA_BLOB_HTML"
171
- assert "infocuria body in FR" in by_lang["FR"]["text"]
172
- # DE only existed in CELLAR — comes through with CELLAR_ITEM source.
173
- assert by_lang["DE"]["text_source"] == "CELLAR_ITEM"
174
- assert "cellar body" in by_lang["DE"]["text"]
166
+ for language in ("EN", "FR", "DE"):
167
+ assert by_lang[language]["text_source"] == "CELLAR_ITEM"
168
+ assert "cellar body" in by_lang[language]["text"]
169
+ assert data["text_source"] == "CELLAR_ITEM"
170
+ assert "cellar body" in data["text"]
175
171
 
176
172
 
177
173
  def test_metadata_fields_remain_from_infocuria_even_when_cellar_supplements(monkeypatch):
@@ -250,9 +246,8 @@ def test_metadata_fields_remain_from_infocuria_even_when_cellar_supplements(monk
250
246
  assert "Environment" in data["keywords"]
251
247
 
252
248
 
253
- def test_supplementation_is_noop_when_cellar_has_nothing_extra(monkeypatch):
254
- """If CELLAR returns no new languages, the fulltexts list is unchanged
255
- and the metadata is untouched. Belt-and-braces idempotency."""
249
+ def test_cellar_replaces_all_overlapping_languages(monkeypatch):
250
+ """Even exact language overlap is replaced by the canonical work."""
256
251
  eurlex_scraping._get_case_data_cached.cache_clear()
257
252
  _patch_infocuria(monkeypatch, doc_langs=["EN", "FR", "DE"])
258
253
  _patch_cellar(monkeypatch, languages=["EN", "FR", "DE"]) # exact overlap
@@ -261,9 +256,8 @@ def test_supplementation_is_noop_when_cellar_has_nothing_extra(monkeypatch):
261
256
 
262
257
  langs = sorted(entry["text_language"] for entry in data["fulltexts"])
263
258
  assert langs == ["DE", "EN", "FR"]
264
- # No CELLAR_ITEM entries — every language already had an InfoCuria source.
265
259
  sources = {entry["text_source"] for entry in data["fulltexts"]}
266
- assert sources == {"INFOCURIA_BLOB_HTML"}
260
+ assert sources == {"CELLAR_ITEM"}
267
261
 
268
262
 
269
263
  def test_supplementation_skips_when_cellar_work_uri_unresolvable(monkeypatch):
@@ -1,8 +0,0 @@
1
- {
2
- "tag": "2.0.2",
3
- "distance": 0,
4
- "node": "gf354af5941cc140dd6de68aa3495f73831261e45",
5
- "dirty": false,
6
- "branch": "HEAD",
7
- "node_date": "2026-08-28"
8
- }