cellar-extractor 2.0.2__tar.gz → 2.0.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (66) hide show
  1. {cellar_extractor-2.0.2/cellar_extractor.egg-info → cellar_extractor-2.0.4}/PKG-INFO +22 -1
  2. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/README.md +21 -0
  3. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/__init__.py +3 -0
  4. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/_version.py +3 -3
  5. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar.py +31 -4
  6. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_queries.py +199 -0
  7. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/eurlex_scraping.py +128 -25
  8. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4/cellar_extractor.egg-info}/PKG-INFO +22 -1
  9. cellar_extractor-2.0.4/cellar_extractor.egg-info/scm_version.json +8 -0
  10. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar.py +72 -19
  11. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_queries_local.py +107 -1
  12. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_infocuria_adapter.py +53 -3
  13. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_metadata_hygiene_local.py +34 -0
  14. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_multilang_fulltext.py +14 -1
  15. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_supplement.py +12 -18
  16. cellar_extractor-2.0.2/cellar_extractor.egg-info/scm_version.json +0 -8
  17. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.env.example +0 -0
  18. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.flake8 +0 -0
  19. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.github/workflows/ci.yml +0 -0
  20. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.github/workflows/github-actions.yml +0 -0
  21. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/FIELDS.md +0 -0
  22. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/LICENSE +0 -0
  23. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/MANIFEST.in +0 -0
  24. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/build_package.py +0 -0
  25. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_extra_extract.py +0 -0
  26. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_sparql_queries.py +0 -0
  27. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/citations_adder.py +0 -0
  28. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/csv_extractor.py +0 -0
  29. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/fulltext_saving.py +0 -0
  30. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/json_to_csv.py +0 -0
  31. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/nodes_and_edges.py +0 -0
  32. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/operative_extractions.py +0 -0
  33. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/persistence.py +0 -0
  34. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/schema.py +0 -0
  35. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/sparql.py +0 -0
  36. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/SOURCES.txt +0 -0
  37. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/dependency_links.txt +0 -0
  38. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/requires.txt +0 -0
  39. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/scm_file_list.json +0 -0
  40. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/top_level.txt +0 -0
  41. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/pyproject.toml +0 -0
  42. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/requirements.txt +0 -0
  43. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/setup.cfg +0 -0
  44. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/setup.py +0 -0
  45. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/61986CJ0062.ENG.txt +0 -0
  46. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_integration.py +0 -0
  47. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_sparql_queries.py +0 -0
  48. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_citation_graph_integration.py +0 -0
  49. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_citations_adder_local.py +0 -0
  50. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_corpus_2020_integration.py +0 -0
  51. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_extra_cellar_local.py +0 -0
  52. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_fulltext_saving_local.py +0 -0
  53. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_infocuria_integration.py +0 -0
  54. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_nodes_and_edges_local.py +0 -0
  55. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_operative_extractions_local.py +0 -0
  56. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_real_fetch_integration.py +0 -0
  57. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_retry_behavior.py +0 -0
  58. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_samples_dump_integration.py +0 -0
  59. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_schema_local.py +0 -0
  60. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector3_adapter.py +0 -0
  61. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_fallback.py +0 -0
  62. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector8_adapter.py +0 -0
  63. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector8_integration.py +0 -0
  64. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sparql_local.py +0 -0
  65. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_webservice_credentials_integration.py +0 -0
  66. {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_webservice_redundancy_integration.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: cellar-extractor
3
- Version: 2.0.2
3
+ Version: 2.0.4
4
4
  Summary: Library for extracting CELLAR case law data from EUR-Lex
5
5
  Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
6
6
  License: Apache-2.0
@@ -127,6 +127,11 @@ df = cell.get_cellar(
127
127
 
128
128
  Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
129
129
 
130
+ For direct manifestation access, use
131
+ `get_cellar_manifestations_by_celex()`. It canonicalizes composite and
132
+ `_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
133
+ derived summaries or notices from being returned as judgment full text.
134
+
130
135
  You can also save explicitly to a custom path instead of the default `data/` location:
131
136
 
132
137
  ```python
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
153
158
  )
154
159
  ```
155
160
 
161
+ For targeted CELEX repair or supplementation, use the public CELLAR
162
+ manifestation API. It unions every CELLAR work sharing the CELEX before choosing
163
+ the best downloadable item per language:
164
+
165
+ ```python
166
+ import cellar_extractor as cell
167
+
168
+ _, manifestations = cell.get_cellar_manifestations_by_celex(
169
+ "62020CJ0414", sector="6"
170
+ )
171
+ english = [m for m in manifestations if m["language"] == "EN"]
172
+ fulltexts = cell.extract_cellar_fulltexts(english)
173
+ ```
174
+
156
175
  Returns:
157
176
 
158
177
  - `extra_df`: enriched dataframe
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
303
322
  | `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
304
323
  | `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
305
324
  | `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
325
+ | `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
326
+ | `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
306
327
  | `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
307
328
  | `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
308
329
  | `FetchOperativePart` | Extract operative part from a single case document |
@@ -92,6 +92,11 @@ df = cell.get_cellar(
92
92
 
93
93
  Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
94
94
 
95
+ For direct manifestation access, use
96
+ `get_cellar_manifestations_by_celex()`. It canonicalizes composite and
97
+ `_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
98
+ derived summaries or notices from being returned as judgment full text.
99
+
95
100
  You can also save explicitly to a custom path instead of the default `data/` location:
96
101
 
97
102
  ```python
@@ -118,6 +123,20 @@ extra_df, fulltext = cell.get_cellar_extra(
118
123
  )
119
124
  ```
120
125
 
126
+ For targeted CELEX repair or supplementation, use the public CELLAR
127
+ manifestation API. It unions every CELLAR work sharing the CELEX before choosing
128
+ the best downloadable item per language:
129
+
130
+ ```python
131
+ import cellar_extractor as cell
132
+
133
+ _, manifestations = cell.get_cellar_manifestations_by_celex(
134
+ "62020CJ0414", sector="6"
135
+ )
136
+ english = [m for m in manifestations if m["language"] == "EN"]
137
+ fulltexts = cell.extract_cellar_fulltexts(english)
138
+ ```
139
+
121
140
  Returns:
122
141
 
123
142
  - `extra_df`: enriched dataframe
@@ -268,6 +287,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
268
287
  | `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
269
288
  | `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
270
289
  | `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
290
+ | `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
291
+ | `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
271
292
  | `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
272
293
  | `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
273
294
  | `FetchOperativePart` | Extract operative part from a single case document |
@@ -9,6 +9,9 @@ from cellar_extractor.cellar import get_cellar_extra
9
9
  from cellar_extractor.cellar import get_nodes_and_edges_lists
10
10
  from cellar_extractor.cellar import filter_subject_matter
11
11
  from cellar_extractor.eurlex_scraping import get_legislation_by_celex_id
12
+ from cellar_extractor.eurlex_scraping import get_cellar_manifestations_by_celex
13
+ from cellar_extractor.eurlex_scraping import extract_cellar_fulltexts
14
+ from cellar_extractor.eurlex_scraping import normalize_celex
12
15
  from cellar_extractor.operative_extractions import FetchOperativePart
13
16
  from cellar_extractor.operative_extractions import Writing
14
17
  import logging
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
18
18
  commit_id: str | None
19
19
  __commit_id__: str | None
20
20
 
21
- __version__ = version = '2.0.2'
22
- __version_tuple__ = version_tuple = (2, 0, 2)
21
+ __version__ = version = '2.0.4'
22
+ __version_tuple__ = version_tuple = (2, 0, 4)
23
23
 
24
- __commit_id__ = commit_id = 'gf354af594'
24
+ __commit_id__ = commit_id = 'g792a41d4a'
@@ -4,7 +4,12 @@ from concurrent.futures import ThreadPoolExecutor
4
4
  from tqdm import tqdm
5
5
 
6
6
  from cellar_extractor.cellar_extra_extract import extra_cellar
7
- from cellar_extractor.cellar_queries import get_all_eclis, get_raw_cellar_metadata
7
+ from cellar_extractor.cellar_queries import (
8
+ get_all_eclis,
9
+ get_infocuria_document_metadata,
10
+ get_raw_cellar_metadata,
11
+ reconcile_document_metadata,
12
+ )
8
13
  from cellar_extractor.json_to_csv import json_to_csv_returning
9
14
  from cellar_extractor.nodes_and_edges import get_nodes_and_edges
10
15
  from cellar_extractor.persistence import (
@@ -70,6 +75,7 @@ def get_cellar(
70
75
  output_path=None,
71
76
  return_data=None,
72
77
  save=None,
78
+ reconcile_infocuria=False,
73
79
  ):
74
80
  """
75
81
  Fetch base CELLAR metadata.
@@ -91,10 +97,24 @@ def get_cellar(
91
97
  logging.info(f"Up until the specified end date {ed}")
92
98
  eclis = get_all_eclis(starting_date=sd, ending_date=ed, limit=max_ecli)
93
99
  logging.info(f"Found {len(eclis)} ECLIs")
94
- if len(eclis) == 0:
100
+ all_eclis = _fetch_metadata_batches(eclis)
101
+ if reconcile_infocuria:
102
+ infocuria_metadata = get_infocuria_document_metadata(
103
+ starting_date=sd,
104
+ ending_date=ed,
105
+ limit=max_ecli,
106
+ )
107
+ logging.info(
108
+ "Found %s ECLIs in the InfoCuria document catalogue",
109
+ len(infocuria_metadata),
110
+ )
111
+ all_eclis = reconcile_document_metadata(all_eclis, infocuria_metadata)
112
+ if max_ecli is not None and len(all_eclis) > max_ecli:
113
+ all_eclis = dict(list(all_eclis.items())[:max_ecli])
114
+
115
+ if len(all_eclis) == 0:
95
116
  logging.info(f"No data to download found between {sd} and {ed}")
96
117
  return False
97
- all_eclis = _fetch_metadata_batches(eclis)
98
118
 
99
119
  result = _materialize_cellar_output(all_eclis, file_format)
100
120
  if save_enabled:
@@ -136,7 +156,14 @@ def get_cellar_extra(
136
156
  if not ed:
137
157
  ed = datetime.now().isoformat(timespec="seconds")
138
158
  save_enabled = resolve_save_enabled(save=save, save_file=save_file, default=True)
139
- data = get_cellar(ed=ed, save=False, max_ecli=max_ecli, sd=sd, file_format="csv")
159
+ data = get_cellar(
160
+ ed=ed,
161
+ save=False,
162
+ max_ecli=max_ecli,
163
+ sd=sd,
164
+ file_format="csv",
165
+ reconcile_infocuria=True,
166
+ )
140
167
  if data is False:
141
168
  logging.warning("Cellar extraction unsuccessful")
142
169
  return False, False
@@ -1,6 +1,8 @@
1
+ import re
1
2
  import time
2
3
  from datetime import date, datetime, timedelta
3
4
 
5
+ import requests
4
6
  from SPARQLWrapper import SPARQLWrapper, JSON, POST
5
7
 
6
8
  # Literal placeholder CELLAR emits while a property is awaiting curation;
@@ -13,6 +15,17 @@ MAX_SORTED_TOP_LIMIT = 10000
13
15
  ECLI_WINDOW_DAYS = 366
14
16
  SPARQL_REQUEST_TIMEOUT_SECONDS = 30
15
17
  SPARQL_RETRY_BACKOFF_BASE_SECONDS = 0.5
18
+ INFOCURIA_SEARCH_ENDPOINT = "https://infocuriaws.curia.europa.eu/elastic-connector/search"
19
+ INFOCURIA_PAGE_SIZE = 100
20
+ INFOCURIA_REQUEST_TIMEOUT_SECONDS = 60
21
+ INFOCURIA_IDENTITY_FIELDS = {
22
+ "case-law_ecli",
23
+ "resource_legal_id_celex",
24
+ "work_date_document",
25
+ "resource_legal_type",
26
+ "resource_legal_id_sector",
27
+ "case-law_affaire_number",
28
+ }
16
29
 
17
30
 
18
31
  def _query_with_retries(sparql, retries, error_message):
@@ -157,6 +170,192 @@ def get_all_eclis(starting_date=None, ending_date=None, limit=None, max_retries=
157
170
  return eclis
158
171
 
159
172
 
173
+ def _normalize_infocuria_celex(value):
174
+ """Return a canonical primary-document CELEX from an InfoCuria hit.
175
+
176
+ InfoCuria writes numbered document variants as ``.01`` while EUR-Lex and
177
+ CELLAR use ``(01)``. Derived summary/information works are deliberately
178
+ rejected instead of being collapsed onto their base CELEX.
179
+ """
180
+ if value is None:
181
+ return ""
182
+ celex = str(value).replace(" ", "").strip()
183
+ if celex == "":
184
+ return ""
185
+ celex = celex.split(";", 1)[0]
186
+ if re.search(r"_(?:SUM|RES|INF)$", celex, flags=re.IGNORECASE):
187
+ return ""
188
+ celex = re.sub(r"\.(\d{2})$", r"(\1)", celex)
189
+ if not celex.startswith("6"):
190
+ return ""
191
+ return celex
192
+
193
+
194
+ def _build_infocuria_search_payload(starting_date, ending_date, page_number, page_size):
195
+ start = page_number * page_size + 1
196
+ return {
197
+ "multiSearchTerms": [],
198
+ "searchTerm": "",
199
+ "ecli": "",
200
+ "publishedId": "",
201
+ "usualName": "",
202
+ "logicDocId": "",
203
+ "repJurExpand": False,
204
+ "pagination": {
205
+ "pageNumber": page_number,
206
+ "pageSize": page_size,
207
+ "from": start,
208
+ "to": start + page_size - 1,
209
+ "origin": "jurisprudence",
210
+ },
211
+ "sortTermList": [
212
+ {
213
+ "sortDirection": "ASC",
214
+ "sortTerm": "DOC_DATE",
215
+ "sortSourceTab": "jurisprudence",
216
+ }
217
+ ],
218
+ "filtersValue": [{"field": "docDate", "values": [starting_date, ending_date]}],
219
+ "advancedFiltersValue": [],
220
+ "language": "EN",
221
+ "isSearchExact": False,
222
+ "searchSources": ["document", "metadata"],
223
+ "tabName": "jurisprudence",
224
+ "isAllTabsRequest": False,
225
+ }
226
+
227
+
228
+ def _query_infocuria_page(payload, max_retries):
229
+ last_error = None
230
+ for attempt in range(max_retries):
231
+ try:
232
+ response = requests.post(
233
+ INFOCURIA_SEARCH_ENDPOINT,
234
+ json=payload,
235
+ timeout=INFOCURIA_REQUEST_TIMEOUT_SECONDS,
236
+ )
237
+ response.raise_for_status()
238
+ result = response.json()
239
+ if not isinstance(result, dict):
240
+ raise ValueError("InfoCuria search response is not an object")
241
+ return result
242
+ except Exception as exc:
243
+ last_error = exc
244
+ if attempt < max_retries - 1:
245
+ time.sleep(SPARQL_RETRY_BACKOFF_BASE_SECONDS * (2**attempt))
246
+ raise RuntimeError(
247
+ "Failed to query InfoCuria document catalogue after retries"
248
+ ) from last_error
249
+
250
+
251
+ def _infocuria_metadata_from_hit(hit):
252
+ content = hit.get("content", {}) if isinstance(hit, dict) else {}
253
+ if not isinstance(content, dict):
254
+ return None
255
+
256
+ ecli = str(content.get("ecli") or "").strip()
257
+ celex = _normalize_infocuria_celex(content.get("celex"))
258
+ document_date = str(content.get("docDate") or "").strip()
259
+ if not ecli.startswith("ECLI:EU:") or celex == "" or document_date == "":
260
+ return None
261
+
262
+ resource_type = celex[5:7] if len(celex) >= 7 else ""
263
+ metadata = {
264
+ "case-law_ecli": [ecli],
265
+ "resource_legal_id_celex": [celex],
266
+ "work_date_document": [document_date],
267
+ "resource_legal_id_sector": [celex[0]],
268
+ "metadata_catalog_source": ["infocuria"],
269
+ }
270
+ if resource_type:
271
+ metadata["resource_legal_type"] = [resource_type]
272
+ published_id = str(content.get("idPublished") or "").strip()
273
+ if published_id:
274
+ metadata["case-law_affaire_number"] = [published_id]
275
+ return ecli, metadata
276
+
277
+
278
+ def get_infocuria_document_metadata(
279
+ starting_date=None,
280
+ ending_date=None,
281
+ limit=None,
282
+ max_retries=3,
283
+ ):
284
+ """Enumerate official InfoCuria documents for a date range.
285
+
286
+ CELLAR's SPARQL graph does not contain every document that is available
287
+ through EUR-Lex/InfoCuria, particularly procedural orders. InfoCuria is
288
+ therefore used as a second catalogue source. Results use the same
289
+ predicate-map shape as :func:`get_raw_cellar_metadata` so callers can
290
+ reconcile the sources before normal schema flattening.
291
+ """
292
+ metadata = {}
293
+ for window_start, window_end in _build_ecli_windows(
294
+ starting_date=starting_date,
295
+ ending_date=ending_date,
296
+ ):
297
+ page_number = 0
298
+ while True:
299
+ remaining = None if limit is None else limit - len(metadata)
300
+ if remaining is not None and remaining <= 0:
301
+ return metadata
302
+ page_size = INFOCURIA_PAGE_SIZE
303
+
304
+ payload = _build_infocuria_search_payload(
305
+ str(window_start)[:10],
306
+ str(window_end)[:10],
307
+ page_number,
308
+ page_size,
309
+ )
310
+ result = _query_infocuria_page(payload, max_retries=max_retries)
311
+ hits = result.get("searchHits", [])
312
+ if not isinstance(hits, list):
313
+ raise RuntimeError("InfoCuria catalogue returned invalid searchHits")
314
+
315
+ for hit in hits:
316
+ parsed = _infocuria_metadata_from_hit(hit)
317
+ if parsed is None:
318
+ continue
319
+ ecli, values = parsed
320
+ # Multiple logical documents can share an ECLI. Their CELEX
321
+ # identity is normally identical; retaining the first hit is
322
+ # deterministic because the endpoint is date-sorted.
323
+ metadata.setdefault(ecli, values)
324
+ if limit is not None and len(metadata) >= limit:
325
+ return metadata
326
+
327
+ total_hits = int(result.get("totalHits") or 0)
328
+ consumed = (page_number + 1) * page_size
329
+ if not hits or consumed >= total_hits:
330
+ break
331
+ page_number += 1
332
+ return metadata
333
+
334
+
335
+ def reconcile_document_metadata(cellar_metadata, infocuria_metadata):
336
+ """Merge both catalogues, preferring InfoCuria for document identity.
337
+
338
+ Rich CELLAR metadata is retained. InfoCuria supplies missing documents
339
+ and corrects identity fields when CELLAR associates an ECLI with a stale
340
+ or sibling CELEX work.
341
+ """
342
+ reconciled = {
343
+ ecli: {key: list(values) for key, values in values_by_key.items()}
344
+ for ecli, values_by_key in (cellar_metadata or {}).items()
345
+ }
346
+ for ecli, values_by_key in (infocuria_metadata or {}).items():
347
+ if ecli not in reconciled:
348
+ reconciled[ecli] = {
349
+ key: list(values) for key, values in values_by_key.items()
350
+ }
351
+ continue
352
+ target = reconciled[ecli]
353
+ for key, values in values_by_key.items():
354
+ if key in INFOCURIA_IDENTITY_FIELDS or not target.get(key):
355
+ target[key] = list(values)
356
+ return reconciled
357
+
358
+
160
359
  def get_raw_cellar_metadata_by_celex(
161
360
  celex_ids,
162
361
  get_labels=True,
@@ -126,9 +126,21 @@ def _normalize_celex(celex):
126
126
  value = non_inf[0] if non_inf else options[0]
127
127
  if "_" in value:
128
128
  value = value.split("_")[0]
129
+ value = re.sub(r"\.(\d{2})$", r"(\1)", value)
129
130
  return value
130
131
 
131
132
 
133
+ def normalize_celex(celex):
134
+ """Return the canonical work CELEX used for full-text retrieval.
135
+
136
+ CELLAR metadata can bundle a primary document with derived information,
137
+ summary, and résumé works (for example ``62020CJ0414_SUM;62020CJ0414``).
138
+ Full-text callers must resolve the unsuffixed base work so a derived work
139
+ can never occupy the judgment's language slot.
140
+ """
141
+ return _normalize_celex(celex)
142
+
143
+
132
144
  def _sleep_with_backoff(attempt, base=0.2):
133
145
  time.sleep(base * (attempt + 1) + random.uniform(0.0, 0.1))
134
146
 
@@ -413,6 +425,25 @@ def _fetch_sector8_items_for_celex(celex, sector="8"):
413
425
  return work_uris, candidates
414
426
 
415
427
 
428
+ def get_cellar_manifestations_by_celex(celex, sector="8"):
429
+ """Return all CELLAR works and manifestation candidates for a CELEX.
430
+
431
+ A CELEX can resolve to multiple CELLAR works with different language
432
+ coverage. This public entry point deliberately returns the union across
433
+ every matching work so callers do not need to depend on the extractor's
434
+ private sector-8 helpers.
435
+
436
+ The return value is ``(work_uris, manifestations)``. Each manifestation
437
+ contains ``item_url``, ``format``, and ``language`` keys and is deduplicated
438
+ on that triple. ``sector`` defaults to ``"8"`` and may be set to ``"6"``
439
+ for CJEU documents.
440
+ """
441
+ canonical_celex = normalize_celex(celex)
442
+ if not canonical_celex:
443
+ return [], []
444
+ return _fetch_sector8_items_for_celex(canonical_celex, sector=sector)
445
+
446
+
416
447
  def _fetch_sector8_work_uri(celex, sector="8"):
417
448
  """Resolve a CELEX to its CELLAR work URI.
418
449
 
@@ -553,6 +584,17 @@ def _fanout_fulltexts_from_candidates(candidates, source_label):
553
584
  return out
554
585
 
555
586
 
587
+ def extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM"):
588
+ """Download the best manifestation per language as fulltext records.
589
+
590
+ ``manifestations`` is the candidate list returned by
591
+ :func:`get_cellar_manifestations_by_celex`. Empty bodies are omitted and
592
+ each returned dictionary contains ``text``, ``html``, ``text_source``,
593
+ ``text_language``, and ``text_format``.
594
+ """
595
+ return _fanout_fulltexts_from_candidates(manifestations, source_label)
596
+
597
+
556
598
  def _get_case_data_sector8(celex, language="EN"):
557
599
  work_uris, main_candidates = _fetch_sector8_items_for_celex(celex)
558
600
 
@@ -773,7 +815,7 @@ def _collect_affecting_ids(content):
773
815
  return ids
774
816
 
775
817
 
776
- def _choose_best_document(doc_hits, language="EN"):
818
+ def _choose_best_document(doc_hits, language="EN", celex=None):
777
819
  candidates = []
778
820
  for hit in doc_hits or []:
779
821
  content = hit.get("content", {}) if isinstance(hit, dict) else {}
@@ -786,6 +828,23 @@ def _choose_best_document(doc_hits, language="EN"):
786
828
  if len(candidates) == 0:
787
829
  return None
788
830
 
831
+ normalized_target = _normalize_celex(celex) if celex else ""
832
+ if normalized_target:
833
+ candidates_with_celex = [
834
+ doc for doc in candidates if _normalize_celex(doc.get("celex", ""))
835
+ ]
836
+ exact_candidates = [
837
+ doc
838
+ for doc in candidates_with_celex
839
+ if _normalize_celex(doc.get("celex", "")) == normalized_target
840
+ ]
841
+ if exact_candidates:
842
+ candidates = exact_candidates
843
+ elif candidates_with_celex:
844
+ # Selecting a judgment merely because it is the highest-ranked
845
+ # document can attach a sibling judgment to an order CELEX.
846
+ return None
847
+
789
848
  type_priority = {
790
849
  "ARRET": 0,
791
850
  "ORDONNANCE": 1,
@@ -987,7 +1046,11 @@ def _get_case_data_sector6(celex, language="EN"):
987
1046
  if isinstance(root_hit, dict)
988
1047
  else []
989
1048
  )
990
- selected_doc = _choose_best_document(documents, language=language)
1049
+ selected_doc = _choose_best_document(
1050
+ documents,
1051
+ language=language,
1052
+ celex=normalized,
1053
+ )
991
1054
  if selected_doc is None:
992
1055
  return _get_case_data_sector6_cellar_fallback(normalized, language=language)
993
1056
 
@@ -1043,18 +1106,41 @@ def _get_case_data_sector6(celex, language="EN"):
1043
1106
  else:
1044
1107
  summary_source = ""
1045
1108
 
1046
- # Multi-language fanout: InfoCuria's `documents.searchHits` already
1047
- # carries every docLang variant of the procedure. Fetch each blob the
1048
- # same way the primary one was fetched and build a fulltexts list.
1109
+ # Multi-language fanout must remain within the selected logical document.
1110
+ # A procedure can contain judgments, orders, opinions, and notices. The
1111
+ # selected document's groupByLogicalId list is the authoritative language
1112
+ # family; older responses without that field fall back to documents with
1113
+ # the same logicDocId.
1049
1114
  fulltexts: list = []
1050
1115
  seen_langs: set = set()
1051
1116
  session = _get_http_session()
1052
- for doc in documents or []:
1053
- if not isinstance(doc, dict):
1054
- continue
1055
- content = doc.get("content", {}) if isinstance(doc, dict) else {}
1056
- if not isinstance(content, dict):
1057
- continue
1117
+ selected_logic_id = str(selected_doc.get("logicDocId", ""))
1118
+ selected_group = selected_doc.get("groupByLogicalId") or []
1119
+ variant_contents = []
1120
+ if isinstance(selected_group, list) and selected_group:
1121
+ for variant in selected_group:
1122
+ if not isinstance(variant, dict):
1123
+ continue
1124
+ variant_contents.append(
1125
+ {
1126
+ "docLang": variant.get("docLang"),
1127
+ "docFormats": variant.get("formats") or [],
1128
+ "logicDocId": selected_logic_id,
1129
+ "idProcedure": variant.get("idProcedure")
1130
+ or selected_doc.get("idProcedure"),
1131
+ }
1132
+ )
1133
+ else:
1134
+ for doc in documents or []:
1135
+ content = doc.get("content", {}) if isinstance(doc, dict) else {}
1136
+ if not isinstance(content, dict):
1137
+ continue
1138
+ if str(content.get("logicDocId", "")) == selected_logic_id:
1139
+ variant_contents.append(content)
1140
+ if not variant_contents:
1141
+ variant_contents = [selected_doc]
1142
+
1143
+ for content in variant_contents:
1058
1144
  variant_lang = content.get("docLang") or ""
1059
1145
  variant_lang_upper = str(variant_lang).upper()
1060
1146
  if variant_lang_upper == "" or variant_lang_upper in seen_langs:
@@ -1104,15 +1190,14 @@ def _get_case_data_sector6(celex, language="EN"):
1104
1190
  }
1105
1191
  )
1106
1192
 
1107
- # Supplement InfoCuria fulltexts with any languages CELLAR has that
1108
- # InfoCuria didn't expose. InfoCuria's documents.searchHits typically
1109
- # carries only the procedural language plus EN; CELLAR's CDM model
1110
- # exposes all 23 EU-official manifestations via expression_uses_language.
1111
- # We keep InfoCuria's entries where languages overlap (the court's own
1112
- # publication is typically higher fidelity) and append CELLAR-sourced
1113
- # entries for any missing languages. Metadata fields (judge, advocate,
1114
- # directory_codes, etc.) remain InfoCuria-sourced — CELLAR cannot
1115
- # populate them.
1193
+ # Merge InfoCuria fulltexts with every language CELLAR has. CELLAR is
1194
+ # authoritative when both sources expose the same language because its
1195
+ # manifestation belongs to the canonical CELEX work. InfoCuria's
1196
+ # procedure search has returned a different document under the judgment
1197
+ # slot in production (AG opinions, procedural orders, and notices), so an
1198
+ # existing InfoCuria language must not block the canonical manifestation.
1199
+ # Metadata fields (judge, advocate, directory_codes, etc.) remain
1200
+ # InfoCuria-sourced — CELLAR cannot populate them.
1116
1201
  #
1117
1202
  # All-in-one try/except: a CELLAR-side failure must never kill the
1118
1203
  # InfoCuria-sourced row we already built.
@@ -1122,18 +1207,36 @@ def _get_case_data_sector6(celex, language="EN"):
1122
1207
  cellar_fulltexts = _fanout_fulltexts_from_candidates(
1123
1208
  cellar_candidates, source_label="CELLAR_ITEM"
1124
1209
  )
1210
+ by_language = {
1211
+ entry.get("text_language", "").upper(): entry
1212
+ for entry in fulltexts
1213
+ if entry.get("text_language", "")
1214
+ }
1125
1215
  for entry in cellar_fulltexts:
1126
1216
  entry_lang = entry.get("text_language", "").upper()
1127
- if not entry_lang or entry_lang in seen_langs:
1217
+ if not entry_lang:
1128
1218
  continue
1129
1219
  seen_langs.add(entry_lang)
1130
- fulltexts.append(entry)
1220
+ by_language[entry_lang] = entry
1221
+ fulltexts = list(by_language.values())
1131
1222
  except Exception:
1132
1223
  # CELLAR supplementation is best-effort. If anything goes wrong
1133
1224
  # (SPARQL timeout, network hiccup, schema drift on the manifestation
1134
1225
  # graph) the InfoCuria-only result is still returned.
1135
1226
  pass
1136
1227
 
1228
+ primary = next(
1229
+ (
1230
+ entry
1231
+ for entry in fulltexts
1232
+ if entry.get("text_language", "").upper() == str(language).upper()
1233
+ ),
1234
+ None,
1235
+ )
1236
+ if primary is not None:
1237
+ text = primary.get("text", "")
1238
+ html = primary.get("html", "")
1239
+
1137
1240
  return {
1138
1241
  "html": html,
1139
1242
  "text": text,
@@ -1146,9 +1249,9 @@ def _get_case_data_sector6(celex, language="EN"):
1146
1249
  "affecting_ids": affecting_ids,
1147
1250
  "affecting_string": affecting_string,
1148
1251
  "citations_extra": citations_extra,
1149
- "text_source": "INFOCURIA_BLOB_HTML" if text != "" else "",
1150
- "text_language": str(doc_lang).upper() if doc_lang else language.upper(),
1151
- "text_format": "html" if text != "" else "",
1252
+ "text_source": primary.get("text_source", "") if primary else "",
1253
+ "text_language": primary.get("text_language", "") if primary else "",
1254
+ "text_format": primary.get("text_format", "") if primary else "",
1152
1255
  "summary_source": summary_source,
1153
1256
  "summary_language": "EN" if summary != "" else "",
1154
1257
  "sector": "6",