cellar-extractor 2.0.3__tar.gz → 2.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cellar_extractor-2.0.3/cellar_extractor.egg-info → cellar_extractor-2.0.4}/PKG-INFO +1 -1
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/_version.py +3 -3
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/cellar.py +31 -4
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/cellar_queries.py +199 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/eurlex_scraping.py +56 -11
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4/cellar_extractor.egg-info}/PKG-INFO +1 -1
- cellar_extractor-2.0.4/cellar_extractor.egg-info/scm_version.json +8 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_cellar.py +72 -19
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_cellar_queries_local.py +107 -1
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_infocuria_adapter.py +53 -3
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_multilang_fulltext.py +14 -1
- cellar_extractor-2.0.3/cellar_extractor.egg-info/scm_version.json +0 -8
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/.env.example +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/.flake8 +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/.github/workflows/ci.yml +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/.github/workflows/github-actions.yml +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/FIELDS.md +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/LICENSE +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/MANIFEST.in +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/README.md +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/build_package.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/__init__.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/cellar_extra_extract.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/citations_adder.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/csv_extractor.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/fulltext_saving.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/json_to_csv.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/nodes_and_edges.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/operative_extractions.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/persistence.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/schema.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor/sparql.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/SOURCES.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/dependency_links.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/requires.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/scm_file_list.json +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/top_level.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/pyproject.toml +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/requirements.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/setup.cfg +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/setup.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/61986CJ0062.ENG.txt +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_cellar_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_citation_graph_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_citations_adder_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_corpus_2020_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_extra_cellar_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_fulltext_saving_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_infocuria_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_metadata_hygiene_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_nodes_and_edges_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_operative_extractions_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_real_fetch_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_retry_behavior.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_samples_dump_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_schema_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sector3_adapter.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_fallback.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_supplement.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sector8_adapter.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sector8_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_sparql_local.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_webservice_credentials_integration.py +0 -0
- {cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_webservice_redundancy_integration.py +0 -0
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (2, 0,
|
|
21
|
+
__version__ = version = '2.0.4'
|
|
22
|
+
__version_tuple__ = version_tuple = (2, 0, 4)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g792a41d4a'
|
|
@@ -4,7 +4,12 @@ from concurrent.futures import ThreadPoolExecutor
|
|
|
4
4
|
from tqdm import tqdm
|
|
5
5
|
|
|
6
6
|
from cellar_extractor.cellar_extra_extract import extra_cellar
|
|
7
|
-
from cellar_extractor.cellar_queries import
|
|
7
|
+
from cellar_extractor.cellar_queries import (
|
|
8
|
+
get_all_eclis,
|
|
9
|
+
get_infocuria_document_metadata,
|
|
10
|
+
get_raw_cellar_metadata,
|
|
11
|
+
reconcile_document_metadata,
|
|
12
|
+
)
|
|
8
13
|
from cellar_extractor.json_to_csv import json_to_csv_returning
|
|
9
14
|
from cellar_extractor.nodes_and_edges import get_nodes_and_edges
|
|
10
15
|
from cellar_extractor.persistence import (
|
|
@@ -70,6 +75,7 @@ def get_cellar(
|
|
|
70
75
|
output_path=None,
|
|
71
76
|
return_data=None,
|
|
72
77
|
save=None,
|
|
78
|
+
reconcile_infocuria=False,
|
|
73
79
|
):
|
|
74
80
|
"""
|
|
75
81
|
Fetch base CELLAR metadata.
|
|
@@ -91,10 +97,24 @@ def get_cellar(
|
|
|
91
97
|
logging.info(f"Up until the specified end date {ed}")
|
|
92
98
|
eclis = get_all_eclis(starting_date=sd, ending_date=ed, limit=max_ecli)
|
|
93
99
|
logging.info(f"Found {len(eclis)} ECLIs")
|
|
94
|
-
|
|
100
|
+
all_eclis = _fetch_metadata_batches(eclis)
|
|
101
|
+
if reconcile_infocuria:
|
|
102
|
+
infocuria_metadata = get_infocuria_document_metadata(
|
|
103
|
+
starting_date=sd,
|
|
104
|
+
ending_date=ed,
|
|
105
|
+
limit=max_ecli,
|
|
106
|
+
)
|
|
107
|
+
logging.info(
|
|
108
|
+
"Found %s ECLIs in the InfoCuria document catalogue",
|
|
109
|
+
len(infocuria_metadata),
|
|
110
|
+
)
|
|
111
|
+
all_eclis = reconcile_document_metadata(all_eclis, infocuria_metadata)
|
|
112
|
+
if max_ecli is not None and len(all_eclis) > max_ecli:
|
|
113
|
+
all_eclis = dict(list(all_eclis.items())[:max_ecli])
|
|
114
|
+
|
|
115
|
+
if len(all_eclis) == 0:
|
|
95
116
|
logging.info(f"No data to download found between {sd} and {ed}")
|
|
96
117
|
return False
|
|
97
|
-
all_eclis = _fetch_metadata_batches(eclis)
|
|
98
118
|
|
|
99
119
|
result = _materialize_cellar_output(all_eclis, file_format)
|
|
100
120
|
if save_enabled:
|
|
@@ -136,7 +156,14 @@ def get_cellar_extra(
|
|
|
136
156
|
if not ed:
|
|
137
157
|
ed = datetime.now().isoformat(timespec="seconds")
|
|
138
158
|
save_enabled = resolve_save_enabled(save=save, save_file=save_file, default=True)
|
|
139
|
-
data = get_cellar(
|
|
159
|
+
data = get_cellar(
|
|
160
|
+
ed=ed,
|
|
161
|
+
save=False,
|
|
162
|
+
max_ecli=max_ecli,
|
|
163
|
+
sd=sd,
|
|
164
|
+
file_format="csv",
|
|
165
|
+
reconcile_infocuria=True,
|
|
166
|
+
)
|
|
140
167
|
if data is False:
|
|
141
168
|
logging.warning("Cellar extraction unsuccessful")
|
|
142
169
|
return False, False
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import re
|
|
1
2
|
import time
|
|
2
3
|
from datetime import date, datetime, timedelta
|
|
3
4
|
|
|
5
|
+
import requests
|
|
4
6
|
from SPARQLWrapper import SPARQLWrapper, JSON, POST
|
|
5
7
|
|
|
6
8
|
# Literal placeholder CELLAR emits while a property is awaiting curation;
|
|
@@ -13,6 +15,17 @@ MAX_SORTED_TOP_LIMIT = 10000
|
|
|
13
15
|
ECLI_WINDOW_DAYS = 366
|
|
14
16
|
SPARQL_REQUEST_TIMEOUT_SECONDS = 30
|
|
15
17
|
SPARQL_RETRY_BACKOFF_BASE_SECONDS = 0.5
|
|
18
|
+
INFOCURIA_SEARCH_ENDPOINT = "https://infocuriaws.curia.europa.eu/elastic-connector/search"
|
|
19
|
+
INFOCURIA_PAGE_SIZE = 100
|
|
20
|
+
INFOCURIA_REQUEST_TIMEOUT_SECONDS = 60
|
|
21
|
+
INFOCURIA_IDENTITY_FIELDS = {
|
|
22
|
+
"case-law_ecli",
|
|
23
|
+
"resource_legal_id_celex",
|
|
24
|
+
"work_date_document",
|
|
25
|
+
"resource_legal_type",
|
|
26
|
+
"resource_legal_id_sector",
|
|
27
|
+
"case-law_affaire_number",
|
|
28
|
+
}
|
|
16
29
|
|
|
17
30
|
|
|
18
31
|
def _query_with_retries(sparql, retries, error_message):
|
|
@@ -157,6 +170,192 @@ def get_all_eclis(starting_date=None, ending_date=None, limit=None, max_retries=
|
|
|
157
170
|
return eclis
|
|
158
171
|
|
|
159
172
|
|
|
173
|
+
def _normalize_infocuria_celex(value):
|
|
174
|
+
"""Return a canonical primary-document CELEX from an InfoCuria hit.
|
|
175
|
+
|
|
176
|
+
InfoCuria writes numbered document variants as ``.01`` while EUR-Lex and
|
|
177
|
+
CELLAR use ``(01)``. Derived summary/information works are deliberately
|
|
178
|
+
rejected instead of being collapsed onto their base CELEX.
|
|
179
|
+
"""
|
|
180
|
+
if value is None:
|
|
181
|
+
return ""
|
|
182
|
+
celex = str(value).replace(" ", "").strip()
|
|
183
|
+
if celex == "":
|
|
184
|
+
return ""
|
|
185
|
+
celex = celex.split(";", 1)[0]
|
|
186
|
+
if re.search(r"_(?:SUM|RES|INF)$", celex, flags=re.IGNORECASE):
|
|
187
|
+
return ""
|
|
188
|
+
celex = re.sub(r"\.(\d{2})$", r"(\1)", celex)
|
|
189
|
+
if not celex.startswith("6"):
|
|
190
|
+
return ""
|
|
191
|
+
return celex
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _build_infocuria_search_payload(starting_date, ending_date, page_number, page_size):
|
|
195
|
+
start = page_number * page_size + 1
|
|
196
|
+
return {
|
|
197
|
+
"multiSearchTerms": [],
|
|
198
|
+
"searchTerm": "",
|
|
199
|
+
"ecli": "",
|
|
200
|
+
"publishedId": "",
|
|
201
|
+
"usualName": "",
|
|
202
|
+
"logicDocId": "",
|
|
203
|
+
"repJurExpand": False,
|
|
204
|
+
"pagination": {
|
|
205
|
+
"pageNumber": page_number,
|
|
206
|
+
"pageSize": page_size,
|
|
207
|
+
"from": start,
|
|
208
|
+
"to": start + page_size - 1,
|
|
209
|
+
"origin": "jurisprudence",
|
|
210
|
+
},
|
|
211
|
+
"sortTermList": [
|
|
212
|
+
{
|
|
213
|
+
"sortDirection": "ASC",
|
|
214
|
+
"sortTerm": "DOC_DATE",
|
|
215
|
+
"sortSourceTab": "jurisprudence",
|
|
216
|
+
}
|
|
217
|
+
],
|
|
218
|
+
"filtersValue": [{"field": "docDate", "values": [starting_date, ending_date]}],
|
|
219
|
+
"advancedFiltersValue": [],
|
|
220
|
+
"language": "EN",
|
|
221
|
+
"isSearchExact": False,
|
|
222
|
+
"searchSources": ["document", "metadata"],
|
|
223
|
+
"tabName": "jurisprudence",
|
|
224
|
+
"isAllTabsRequest": False,
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _query_infocuria_page(payload, max_retries):
|
|
229
|
+
last_error = None
|
|
230
|
+
for attempt in range(max_retries):
|
|
231
|
+
try:
|
|
232
|
+
response = requests.post(
|
|
233
|
+
INFOCURIA_SEARCH_ENDPOINT,
|
|
234
|
+
json=payload,
|
|
235
|
+
timeout=INFOCURIA_REQUEST_TIMEOUT_SECONDS,
|
|
236
|
+
)
|
|
237
|
+
response.raise_for_status()
|
|
238
|
+
result = response.json()
|
|
239
|
+
if not isinstance(result, dict):
|
|
240
|
+
raise ValueError("InfoCuria search response is not an object")
|
|
241
|
+
return result
|
|
242
|
+
except Exception as exc:
|
|
243
|
+
last_error = exc
|
|
244
|
+
if attempt < max_retries - 1:
|
|
245
|
+
time.sleep(SPARQL_RETRY_BACKOFF_BASE_SECONDS * (2**attempt))
|
|
246
|
+
raise RuntimeError(
|
|
247
|
+
"Failed to query InfoCuria document catalogue after retries"
|
|
248
|
+
) from last_error
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _infocuria_metadata_from_hit(hit):
|
|
252
|
+
content = hit.get("content", {}) if isinstance(hit, dict) else {}
|
|
253
|
+
if not isinstance(content, dict):
|
|
254
|
+
return None
|
|
255
|
+
|
|
256
|
+
ecli = str(content.get("ecli") or "").strip()
|
|
257
|
+
celex = _normalize_infocuria_celex(content.get("celex"))
|
|
258
|
+
document_date = str(content.get("docDate") or "").strip()
|
|
259
|
+
if not ecli.startswith("ECLI:EU:") or celex == "" or document_date == "":
|
|
260
|
+
return None
|
|
261
|
+
|
|
262
|
+
resource_type = celex[5:7] if len(celex) >= 7 else ""
|
|
263
|
+
metadata = {
|
|
264
|
+
"case-law_ecli": [ecli],
|
|
265
|
+
"resource_legal_id_celex": [celex],
|
|
266
|
+
"work_date_document": [document_date],
|
|
267
|
+
"resource_legal_id_sector": [celex[0]],
|
|
268
|
+
"metadata_catalog_source": ["infocuria"],
|
|
269
|
+
}
|
|
270
|
+
if resource_type:
|
|
271
|
+
metadata["resource_legal_type"] = [resource_type]
|
|
272
|
+
published_id = str(content.get("idPublished") or "").strip()
|
|
273
|
+
if published_id:
|
|
274
|
+
metadata["case-law_affaire_number"] = [published_id]
|
|
275
|
+
return ecli, metadata
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def get_infocuria_document_metadata(
|
|
279
|
+
starting_date=None,
|
|
280
|
+
ending_date=None,
|
|
281
|
+
limit=None,
|
|
282
|
+
max_retries=3,
|
|
283
|
+
):
|
|
284
|
+
"""Enumerate official InfoCuria documents for a date range.
|
|
285
|
+
|
|
286
|
+
CELLAR's SPARQL graph does not contain every document that is available
|
|
287
|
+
through EUR-Lex/InfoCuria, particularly procedural orders. InfoCuria is
|
|
288
|
+
therefore used as a second catalogue source. Results use the same
|
|
289
|
+
predicate-map shape as :func:`get_raw_cellar_metadata` so callers can
|
|
290
|
+
reconcile the sources before normal schema flattening.
|
|
291
|
+
"""
|
|
292
|
+
metadata = {}
|
|
293
|
+
for window_start, window_end in _build_ecli_windows(
|
|
294
|
+
starting_date=starting_date,
|
|
295
|
+
ending_date=ending_date,
|
|
296
|
+
):
|
|
297
|
+
page_number = 0
|
|
298
|
+
while True:
|
|
299
|
+
remaining = None if limit is None else limit - len(metadata)
|
|
300
|
+
if remaining is not None and remaining <= 0:
|
|
301
|
+
return metadata
|
|
302
|
+
page_size = INFOCURIA_PAGE_SIZE
|
|
303
|
+
|
|
304
|
+
payload = _build_infocuria_search_payload(
|
|
305
|
+
str(window_start)[:10],
|
|
306
|
+
str(window_end)[:10],
|
|
307
|
+
page_number,
|
|
308
|
+
page_size,
|
|
309
|
+
)
|
|
310
|
+
result = _query_infocuria_page(payload, max_retries=max_retries)
|
|
311
|
+
hits = result.get("searchHits", [])
|
|
312
|
+
if not isinstance(hits, list):
|
|
313
|
+
raise RuntimeError("InfoCuria catalogue returned invalid searchHits")
|
|
314
|
+
|
|
315
|
+
for hit in hits:
|
|
316
|
+
parsed = _infocuria_metadata_from_hit(hit)
|
|
317
|
+
if parsed is None:
|
|
318
|
+
continue
|
|
319
|
+
ecli, values = parsed
|
|
320
|
+
# Multiple logical documents can share an ECLI. Their CELEX
|
|
321
|
+
# identity is normally identical; retaining the first hit is
|
|
322
|
+
# deterministic because the endpoint is date-sorted.
|
|
323
|
+
metadata.setdefault(ecli, values)
|
|
324
|
+
if limit is not None and len(metadata) >= limit:
|
|
325
|
+
return metadata
|
|
326
|
+
|
|
327
|
+
total_hits = int(result.get("totalHits") or 0)
|
|
328
|
+
consumed = (page_number + 1) * page_size
|
|
329
|
+
if not hits or consumed >= total_hits:
|
|
330
|
+
break
|
|
331
|
+
page_number += 1
|
|
332
|
+
return metadata
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def reconcile_document_metadata(cellar_metadata, infocuria_metadata):
|
|
336
|
+
"""Merge both catalogues, preferring InfoCuria for document identity.
|
|
337
|
+
|
|
338
|
+
Rich CELLAR metadata is retained. InfoCuria supplies missing documents
|
|
339
|
+
and corrects identity fields when CELLAR associates an ECLI with a stale
|
|
340
|
+
or sibling CELEX work.
|
|
341
|
+
"""
|
|
342
|
+
reconciled = {
|
|
343
|
+
ecli: {key: list(values) for key, values in values_by_key.items()}
|
|
344
|
+
for ecli, values_by_key in (cellar_metadata or {}).items()
|
|
345
|
+
}
|
|
346
|
+
for ecli, values_by_key in (infocuria_metadata or {}).items():
|
|
347
|
+
if ecli not in reconciled:
|
|
348
|
+
reconciled[ecli] = {
|
|
349
|
+
key: list(values) for key, values in values_by_key.items()
|
|
350
|
+
}
|
|
351
|
+
continue
|
|
352
|
+
target = reconciled[ecli]
|
|
353
|
+
for key, values in values_by_key.items():
|
|
354
|
+
if key in INFOCURIA_IDENTITY_FIELDS or not target.get(key):
|
|
355
|
+
target[key] = list(values)
|
|
356
|
+
return reconciled
|
|
357
|
+
|
|
358
|
+
|
|
160
359
|
def get_raw_cellar_metadata_by_celex(
|
|
161
360
|
celex_ids,
|
|
162
361
|
get_labels=True,
|
|
@@ -126,6 +126,7 @@ def _normalize_celex(celex):
|
|
|
126
126
|
value = non_inf[0] if non_inf else options[0]
|
|
127
127
|
if "_" in value:
|
|
128
128
|
value = value.split("_")[0]
|
|
129
|
+
value = re.sub(r"\.(\d{2})$", r"(\1)", value)
|
|
129
130
|
return value
|
|
130
131
|
|
|
131
132
|
|
|
@@ -814,7 +815,7 @@ def _collect_affecting_ids(content):
|
|
|
814
815
|
return ids
|
|
815
816
|
|
|
816
817
|
|
|
817
|
-
def _choose_best_document(doc_hits, language="EN"):
|
|
818
|
+
def _choose_best_document(doc_hits, language="EN", celex=None):
|
|
818
819
|
candidates = []
|
|
819
820
|
for hit in doc_hits or []:
|
|
820
821
|
content = hit.get("content", {}) if isinstance(hit, dict) else {}
|
|
@@ -827,6 +828,23 @@ def _choose_best_document(doc_hits, language="EN"):
|
|
|
827
828
|
if len(candidates) == 0:
|
|
828
829
|
return None
|
|
829
830
|
|
|
831
|
+
normalized_target = _normalize_celex(celex) if celex else ""
|
|
832
|
+
if normalized_target:
|
|
833
|
+
candidates_with_celex = [
|
|
834
|
+
doc for doc in candidates if _normalize_celex(doc.get("celex", ""))
|
|
835
|
+
]
|
|
836
|
+
exact_candidates = [
|
|
837
|
+
doc
|
|
838
|
+
for doc in candidates_with_celex
|
|
839
|
+
if _normalize_celex(doc.get("celex", "")) == normalized_target
|
|
840
|
+
]
|
|
841
|
+
if exact_candidates:
|
|
842
|
+
candidates = exact_candidates
|
|
843
|
+
elif candidates_with_celex:
|
|
844
|
+
# Selecting a judgment merely because it is the highest-ranked
|
|
845
|
+
# document can attach a sibling judgment to an order CELEX.
|
|
846
|
+
return None
|
|
847
|
+
|
|
830
848
|
type_priority = {
|
|
831
849
|
"ARRET": 0,
|
|
832
850
|
"ORDONNANCE": 1,
|
|
@@ -1028,7 +1046,11 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1028
1046
|
if isinstance(root_hit, dict)
|
|
1029
1047
|
else []
|
|
1030
1048
|
)
|
|
1031
|
-
selected_doc = _choose_best_document(
|
|
1049
|
+
selected_doc = _choose_best_document(
|
|
1050
|
+
documents,
|
|
1051
|
+
language=language,
|
|
1052
|
+
celex=normalized,
|
|
1053
|
+
)
|
|
1032
1054
|
if selected_doc is None:
|
|
1033
1055
|
return _get_case_data_sector6_cellar_fallback(normalized, language=language)
|
|
1034
1056
|
|
|
@@ -1084,18 +1106,41 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1084
1106
|
else:
|
|
1085
1107
|
summary_source = ""
|
|
1086
1108
|
|
|
1087
|
-
# Multi-language fanout
|
|
1088
|
-
#
|
|
1089
|
-
#
|
|
1109
|
+
# Multi-language fanout must remain within the selected logical document.
|
|
1110
|
+
# A procedure can contain judgments, orders, opinions, and notices. The
|
|
1111
|
+
# selected document's groupByLogicalId list is the authoritative language
|
|
1112
|
+
# family; older responses without that field fall back to documents with
|
|
1113
|
+
# the same logicDocId.
|
|
1090
1114
|
fulltexts: list = []
|
|
1091
1115
|
seen_langs: set = set()
|
|
1092
1116
|
session = _get_http_session()
|
|
1093
|
-
|
|
1094
|
-
|
|
1095
|
-
|
|
1096
|
-
|
|
1097
|
-
|
|
1098
|
-
|
|
1117
|
+
selected_logic_id = str(selected_doc.get("logicDocId", ""))
|
|
1118
|
+
selected_group = selected_doc.get("groupByLogicalId") or []
|
|
1119
|
+
variant_contents = []
|
|
1120
|
+
if isinstance(selected_group, list) and selected_group:
|
|
1121
|
+
for variant in selected_group:
|
|
1122
|
+
if not isinstance(variant, dict):
|
|
1123
|
+
continue
|
|
1124
|
+
variant_contents.append(
|
|
1125
|
+
{
|
|
1126
|
+
"docLang": variant.get("docLang"),
|
|
1127
|
+
"docFormats": variant.get("formats") or [],
|
|
1128
|
+
"logicDocId": selected_logic_id,
|
|
1129
|
+
"idProcedure": variant.get("idProcedure")
|
|
1130
|
+
or selected_doc.get("idProcedure"),
|
|
1131
|
+
}
|
|
1132
|
+
)
|
|
1133
|
+
else:
|
|
1134
|
+
for doc in documents or []:
|
|
1135
|
+
content = doc.get("content", {}) if isinstance(doc, dict) else {}
|
|
1136
|
+
if not isinstance(content, dict):
|
|
1137
|
+
continue
|
|
1138
|
+
if str(content.get("logicDocId", "")) == selected_logic_id:
|
|
1139
|
+
variant_contents.append(content)
|
|
1140
|
+
if not variant_contents:
|
|
1141
|
+
variant_contents = [selected_doc]
|
|
1142
|
+
|
|
1143
|
+
for content in variant_contents:
|
|
1099
1144
|
variant_lang = content.get("docLang") or ""
|
|
1100
1145
|
variant_lang_upper = str(variant_lang).upper()
|
|
1101
1146
|
if variant_lang_upper == "" or variant_lang_upper in seen_langs:
|
|
@@ -25,7 +25,9 @@ def test_get_cellar_csv_in_memory(monkeypatch):
|
|
|
25
25
|
monkeypatch.setattr(
|
|
26
26
|
cellar,
|
|
27
27
|
"get_all_eclis",
|
|
28
|
-
lambda starting_date, ending_date, limit=None: ["E1", "E2"][:limit]
|
|
28
|
+
lambda starting_date, ending_date, limit=None: ["E1", "E2"][:limit]
|
|
29
|
+
if limit
|
|
30
|
+
else ["E1", "E2"],
|
|
29
31
|
)
|
|
30
32
|
monkeypatch.setattr(
|
|
31
33
|
cellar,
|
|
@@ -50,7 +52,9 @@ def test_get_cellar_csv_in_memory(monkeypatch):
|
|
|
50
52
|
|
|
51
53
|
|
|
52
54
|
def test_get_cellar_json_in_memory(monkeypatch):
|
|
53
|
-
monkeypatch.setattr(
|
|
55
|
+
monkeypatch.setattr(
|
|
56
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
57
|
+
)
|
|
54
58
|
monkeypatch.setattr(
|
|
55
59
|
cellar,
|
|
56
60
|
"get_raw_cellar_metadata",
|
|
@@ -71,7 +75,9 @@ def test_get_cellar_json_in_memory(monkeypatch):
|
|
|
71
75
|
|
|
72
76
|
def test_get_cellar_json_save_file(monkeypatch, tmp_path):
|
|
73
77
|
monkeypatch.chdir(tmp_path)
|
|
74
|
-
monkeypatch.setattr(
|
|
78
|
+
monkeypatch.setattr(
|
|
79
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
80
|
+
)
|
|
75
81
|
monkeypatch.setattr(
|
|
76
82
|
cellar,
|
|
77
83
|
"get_raw_cellar_metadata",
|
|
@@ -94,7 +100,9 @@ def test_get_cellar_json_save_file(monkeypatch, tmp_path):
|
|
|
94
100
|
|
|
95
101
|
def test_get_cellar_in_memory_does_not_create_default_output_dir(monkeypatch, tmp_path):
|
|
96
102
|
monkeypatch.chdir(tmp_path)
|
|
97
|
-
monkeypatch.setattr(
|
|
103
|
+
monkeypatch.setattr(
|
|
104
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
105
|
+
)
|
|
98
106
|
monkeypatch.setattr(
|
|
99
107
|
cellar,
|
|
100
108
|
"get_raw_cellar_metadata",
|
|
@@ -113,7 +121,9 @@ def test_get_cellar_in_memory_does_not_create_default_output_dir(monkeypatch, tm
|
|
|
113
121
|
|
|
114
122
|
|
|
115
123
|
def test_get_cellar_save_file_supports_custom_output_path(monkeypatch, tmp_path):
|
|
116
|
-
monkeypatch.setattr(
|
|
124
|
+
monkeypatch.setattr(
|
|
125
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
126
|
+
)
|
|
117
127
|
monkeypatch.setattr(
|
|
118
128
|
cellar,
|
|
119
129
|
"get_raw_cellar_metadata",
|
|
@@ -136,8 +146,12 @@ def test_get_cellar_save_file_supports_custom_output_path(monkeypatch, tmp_path)
|
|
|
136
146
|
assert result.loc[0, "celex"] == "62025CJ0001"
|
|
137
147
|
|
|
138
148
|
|
|
139
|
-
def test_get_cellar_save_to_output_dir_creates_only_requested_parents(
|
|
140
|
-
monkeypatch
|
|
149
|
+
def test_get_cellar_save_to_output_dir_creates_only_requested_parents(
|
|
150
|
+
monkeypatch, tmp_path
|
|
151
|
+
):
|
|
152
|
+
monkeypatch.setattr(
|
|
153
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
154
|
+
)
|
|
141
155
|
monkeypatch.setattr(
|
|
142
156
|
cellar,
|
|
143
157
|
"get_raw_cellar_metadata",
|
|
@@ -189,7 +203,9 @@ def test_get_cellar_preserves_ecli_order_across_parallel_metadata_batches(monkey
|
|
|
189
203
|
monkeypatch.setattr(
|
|
190
204
|
cellar,
|
|
191
205
|
"get_all_eclis",
|
|
192
|
-
lambda starting_date, ending_date, limit=None: eclis[:limit]
|
|
206
|
+
lambda starting_date, ending_date, limit=None: eclis[:limit]
|
|
207
|
+
if limit
|
|
208
|
+
else eclis,
|
|
193
209
|
)
|
|
194
210
|
|
|
195
211
|
def _fake_get_raw_cellar_metadata(batch):
|
|
@@ -206,7 +222,9 @@ def test_get_cellar_preserves_ecli_order_across_parallel_metadata_batches(monkey
|
|
|
206
222
|
for index, ecli in enumerate(batch)
|
|
207
223
|
}
|
|
208
224
|
|
|
209
|
-
monkeypatch.setattr(
|
|
225
|
+
monkeypatch.setattr(
|
|
226
|
+
cellar, "get_raw_cellar_metadata", _fake_get_raw_cellar_metadata
|
|
227
|
+
)
|
|
210
228
|
|
|
211
229
|
df = cellar.get_cellar(
|
|
212
230
|
ed="2025-01-02T00:00:00",
|
|
@@ -221,10 +239,22 @@ def test_get_cellar_preserves_ecli_order_across_parallel_metadata_batches(monkey
|
|
|
221
239
|
|
|
222
240
|
def test_get_cellar_extra_in_memory_calls_extra(monkeypatch):
|
|
223
241
|
base_df = pd.DataFrame({"ecli": ["E1"], "celex": ["62025CJ0001"]})
|
|
224
|
-
monkeypatch.setattr(cellar, "get_cellar", lambda **kwargs: base_df)
|
|
225
242
|
called = {}
|
|
226
243
|
|
|
227
|
-
def
|
|
244
|
+
def _fake_get_cellar(**kwargs):
|
|
245
|
+
called["reconcile_infocuria"] = kwargs.get("reconcile_infocuria")
|
|
246
|
+
return base_df
|
|
247
|
+
|
|
248
|
+
monkeypatch.setattr(cellar, "get_cellar", _fake_get_cellar)
|
|
249
|
+
|
|
250
|
+
def _fake_extra(
|
|
251
|
+
data,
|
|
252
|
+
threads,
|
|
253
|
+
username,
|
|
254
|
+
password,
|
|
255
|
+
metadata_output_path=None,
|
|
256
|
+
fulltext_output_path=None,
|
|
257
|
+
):
|
|
228
258
|
called["threads"] = threads
|
|
229
259
|
called["username"] = username
|
|
230
260
|
called["password"] = password
|
|
@@ -247,6 +277,7 @@ def test_get_cellar_extra_in_memory_calls_extra(monkeypatch):
|
|
|
247
277
|
assert len(data) == 1
|
|
248
278
|
assert len(fulltext) == 1
|
|
249
279
|
assert called == {
|
|
280
|
+
"reconcile_infocuria": True,
|
|
250
281
|
"threads": 4,
|
|
251
282
|
"username": "user",
|
|
252
283
|
"password": "pass",
|
|
@@ -260,7 +291,14 @@ def test_get_cellar_extra_save_file_calls_extra_with_path(monkeypatch, tmp_path)
|
|
|
260
291
|
monkeypatch.setattr(cellar, "get_cellar", lambda **kwargs: base_df)
|
|
261
292
|
called = {}
|
|
262
293
|
|
|
263
|
-
def _fake_extra(
|
|
294
|
+
def _fake_extra(
|
|
295
|
+
data,
|
|
296
|
+
threads,
|
|
297
|
+
username,
|
|
298
|
+
password,
|
|
299
|
+
metadata_output_path=None,
|
|
300
|
+
fulltext_output_path=None,
|
|
301
|
+
):
|
|
264
302
|
called["metadata_output_path"] = metadata_output_path
|
|
265
303
|
called["fulltext_output_path"] = fulltext_output_path
|
|
266
304
|
called["threads"] = threads
|
|
@@ -278,11 +316,15 @@ def test_get_cellar_extra_save_file_calls_extra_with_path(monkeypatch, tmp_path)
|
|
|
278
316
|
threads=3,
|
|
279
317
|
)
|
|
280
318
|
|
|
281
|
-
assert
|
|
282
|
-
"
|
|
319
|
+
assert (
|
|
320
|
+
str(called["metadata_output_path"])
|
|
321
|
+
.replace("\\", "/")
|
|
322
|
+
.endswith("data/cellar_extra_2025-01-01_2025-01-02T00_00_00.csv")
|
|
283
323
|
)
|
|
284
|
-
assert
|
|
285
|
-
"
|
|
324
|
+
assert (
|
|
325
|
+
str(called["fulltext_output_path"])
|
|
326
|
+
.replace("\\", "/")
|
|
327
|
+
.endswith("data/cellar_extra_2025-01-01_2025-01-02T00_00_00_fulltext.json")
|
|
286
328
|
)
|
|
287
329
|
assert called["threads"] == 3
|
|
288
330
|
|
|
@@ -292,7 +334,14 @@ def test_get_cellar_extra_supports_independent_output_paths(monkeypatch, tmp_pat
|
|
|
292
334
|
monkeypatch.setattr(cellar, "get_cellar", lambda **kwargs: base_df)
|
|
293
335
|
called = {}
|
|
294
336
|
|
|
295
|
-
def _fake_extra(
|
|
337
|
+
def _fake_extra(
|
|
338
|
+
data,
|
|
339
|
+
threads,
|
|
340
|
+
username,
|
|
341
|
+
password,
|
|
342
|
+
metadata_output_path=None,
|
|
343
|
+
fulltext_output_path=None,
|
|
344
|
+
):
|
|
296
345
|
called["metadata_output_path"] = metadata_output_path
|
|
297
346
|
called["fulltext_output_path"] = fulltext_output_path
|
|
298
347
|
return data, [{"celex": "62025CJ0001"}]
|
|
@@ -315,7 +364,9 @@ def test_get_cellar_extra_supports_independent_output_paths(monkeypatch, tmp_pat
|
|
|
315
364
|
assert output[1] == [{"celex": "62025CJ0001"}]
|
|
316
365
|
|
|
317
366
|
|
|
318
|
-
def test_get_cellar_extra_in_memory_does_not_create_default_output_dir(
|
|
367
|
+
def test_get_cellar_extra_in_memory_does_not_create_default_output_dir(
|
|
368
|
+
monkeypatch, tmp_path
|
|
369
|
+
):
|
|
319
370
|
monkeypatch.chdir(tmp_path)
|
|
320
371
|
base_df = pd.DataFrame({"ecli": ["E1"], "celex": ["62025CJ0001"]})
|
|
321
372
|
monkeypatch.setattr(cellar, "get_cellar", lambda **kwargs: base_df)
|
|
@@ -337,7 +388,9 @@ def test_get_cellar_extra_in_memory_does_not_create_default_output_dir(monkeypat
|
|
|
337
388
|
|
|
338
389
|
|
|
339
390
|
def test_get_cellar_save_file_alias_still_works(monkeypatch):
|
|
340
|
-
monkeypatch.setattr(
|
|
391
|
+
monkeypatch.setattr(
|
|
392
|
+
cellar, "get_all_eclis", lambda starting_date, ending_date, limit=None: ["E1"]
|
|
393
|
+
)
|
|
341
394
|
monkeypatch.setattr(
|
|
342
395
|
cellar,
|
|
343
396
|
"get_raw_cellar_metadata",
|
|
@@ -103,7 +103,9 @@ def test_get_raw_cellar_metadata_filters_requested_eclis(monkeypatch):
|
|
|
103
103
|
"bindings": [
|
|
104
104
|
{
|
|
105
105
|
"ecli": {"value": "ECLI:EU:C:2025:1"},
|
|
106
|
-
"p": {
|
|
106
|
+
"p": {
|
|
107
|
+
"value": "http://publications.europa.eu/ontology/cdm#case-law_ecli"
|
|
108
|
+
},
|
|
107
109
|
"o": {"value": "ECLI:EU:C:2025:1"},
|
|
108
110
|
}
|
|
109
111
|
]
|
|
@@ -215,3 +217,107 @@ def test_get_all_eclis_applies_large_limit_locally_after_chunking(monkeypatch):
|
|
|
215
217
|
|
|
216
218
|
assert result == ["ECLI:EU:C:2025:1", "ECLI:EU:C:2025:2", "ECLI:EU:C:2025:3"]
|
|
217
219
|
assert len(fake.queries) == 2
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
class _FakeResponse:
|
|
223
|
+
def __init__(self, payload):
|
|
224
|
+
self.payload = payload
|
|
225
|
+
|
|
226
|
+
def raise_for_status(self):
|
|
227
|
+
return None
|
|
228
|
+
|
|
229
|
+
def json(self):
|
|
230
|
+
return self.payload
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def test_infocuria_catalog_paginates_and_normalizes_document_variants(monkeypatch):
|
|
234
|
+
calls = []
|
|
235
|
+
pages = [
|
|
236
|
+
{
|
|
237
|
+
"totalHits": 3,
|
|
238
|
+
"searchHits": [
|
|
239
|
+
{
|
|
240
|
+
"content": {
|
|
241
|
+
"ecli": "ECLI:EU:F:2011:62",
|
|
242
|
+
"celex": "62011FO0005.01",
|
|
243
|
+
"docDate": "2011-05-12",
|
|
244
|
+
"docTypeCode": "ORDONNANCE",
|
|
245
|
+
"idPublished": "F-5/11",
|
|
246
|
+
}
|
|
247
|
+
},
|
|
248
|
+
{
|
|
249
|
+
"content": {
|
|
250
|
+
"ecli": "ECLI:EU:C:2011:1",
|
|
251
|
+
"celex": "62011CJ0001_SUM",
|
|
252
|
+
"docDate": "2011-05-13",
|
|
253
|
+
}
|
|
254
|
+
},
|
|
255
|
+
],
|
|
256
|
+
},
|
|
257
|
+
{
|
|
258
|
+
"totalHits": 3,
|
|
259
|
+
"searchHits": [
|
|
260
|
+
{
|
|
261
|
+
"content": {
|
|
262
|
+
"ecli": "ECLI:EU:C:2011:2",
|
|
263
|
+
"celex": "62011CO0002",
|
|
264
|
+
"docDate": "2011-05-14",
|
|
265
|
+
"idPublished": "C-2/11",
|
|
266
|
+
}
|
|
267
|
+
}
|
|
268
|
+
],
|
|
269
|
+
},
|
|
270
|
+
]
|
|
271
|
+
|
|
272
|
+
def _post(url, json, timeout):
|
|
273
|
+
calls.append((url, json, timeout))
|
|
274
|
+
return _FakeResponse(pages[len(calls) - 1])
|
|
275
|
+
|
|
276
|
+
monkeypatch.setattr(cellar_queries.requests, "post", _post)
|
|
277
|
+
monkeypatch.setattr(cellar_queries, "INFOCURIA_PAGE_SIZE", 2)
|
|
278
|
+
|
|
279
|
+
result = cellar_queries.get_infocuria_document_metadata(
|
|
280
|
+
starting_date="2011-05-01",
|
|
281
|
+
ending_date="2011-05-31",
|
|
282
|
+
)
|
|
283
|
+
|
|
284
|
+
assert set(result) == {"ECLI:EU:F:2011:62", "ECLI:EU:C:2011:2"}
|
|
285
|
+
assert result["ECLI:EU:F:2011:62"]["resource_legal_id_celex"] == ["62011FO0005(01)"]
|
|
286
|
+
assert result["ECLI:EU:F:2011:62"]["metadata_catalog_source"] == ["infocuria"]
|
|
287
|
+
assert result["ECLI:EU:C:2011:2"]["resource_legal_type"] == ["CO"]
|
|
288
|
+
assert calls[0][1]["pagination"]["from"] == 1
|
|
289
|
+
assert calls[1][1]["pagination"]["from"] == 3
|
|
290
|
+
assert calls[0][1]["filtersValue"] == [
|
|
291
|
+
{"field": "docDate", "values": ["2011-05-01", "2011-05-31"]}
|
|
292
|
+
]
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def test_reconcile_document_metadata_adds_and_overrides_infocuria_identity():
|
|
296
|
+
cellar = {
|
|
297
|
+
"ECLI:EU:T:2014:1": {
|
|
298
|
+
"case-law_ecli": ["ECLI:EU:T:2014:1"],
|
|
299
|
+
"resource_legal_id_celex": ["62013TO0505(01)"],
|
|
300
|
+
"work_date_document": ["2014-01-10"],
|
|
301
|
+
"subject_matter": ["Staff cases"],
|
|
302
|
+
}
|
|
303
|
+
}
|
|
304
|
+
infocuria = {
|
|
305
|
+
"ECLI:EU:T:2014:1": {
|
|
306
|
+
"case-law_ecli": ["ECLI:EU:T:2014:1"],
|
|
307
|
+
"resource_legal_id_celex": ["62013TO0505(02)"],
|
|
308
|
+
"work_date_document": ["2014-01-10"],
|
|
309
|
+
},
|
|
310
|
+
"ECLI:EU:T:2014:166": {
|
|
311
|
+
"case-law_ecli": ["ECLI:EU:T:2014:166"],
|
|
312
|
+
"resource_legal_id_celex": ["62013TO0505(03)"],
|
|
313
|
+
"work_date_document": ["2014-04-02"],
|
|
314
|
+
},
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
result = cellar_queries.reconcile_document_metadata(cellar, infocuria)
|
|
318
|
+
|
|
319
|
+
assert result["ECLI:EU:T:2014:1"]["resource_legal_id_celex"] == ["62013TO0505(02)"]
|
|
320
|
+
assert result["ECLI:EU:T:2014:1"]["subject_matter"] == ["Staff cases"]
|
|
321
|
+
assert result["ECLI:EU:T:2014:166"]["resource_legal_id_celex"] == [
|
|
322
|
+
"62013TO0505(03)"
|
|
323
|
+
]
|
|
@@ -40,6 +40,51 @@ def test_choose_best_document_prefers_target_language():
|
|
|
40
40
|
assert selected["logicDocId"] == "id_2"
|
|
41
41
|
|
|
42
42
|
|
|
43
|
+
def test_choose_best_document_requires_requested_celex_before_type_priority():
|
|
44
|
+
docs = [
|
|
45
|
+
{
|
|
46
|
+
"content": {
|
|
47
|
+
"docLang": "EN",
|
|
48
|
+
"docFormats": ["HTML"],
|
|
49
|
+
"logicDocId": "id_judgment",
|
|
50
|
+
"idProcedure": "C/0444/11/00000000RP/01/P/01",
|
|
51
|
+
"docTypeCode": "ARRET",
|
|
52
|
+
"celex": "62011CJ0444",
|
|
53
|
+
}
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"content": {
|
|
57
|
+
"docLang": "EN",
|
|
58
|
+
"docFormats": ["HTML"],
|
|
59
|
+
"logicDocId": "id_order",
|
|
60
|
+
"idProcedure": "C/0444/11/00000000RP/01/P/01",
|
|
61
|
+
"docTypeCode": "ORDONNANCE",
|
|
62
|
+
"celex": "62011CO0444",
|
|
63
|
+
}
|
|
64
|
+
},
|
|
65
|
+
]
|
|
66
|
+
|
|
67
|
+
selected = eurlex_scraping._choose_best_document(
|
|
68
|
+
docs,
|
|
69
|
+
language="EN",
|
|
70
|
+
celex="62011CO0444",
|
|
71
|
+
)
|
|
72
|
+
|
|
73
|
+
assert selected["logicDocId"] == "id_order"
|
|
74
|
+
assert (
|
|
75
|
+
eurlex_scraping._choose_best_document(
|
|
76
|
+
docs,
|
|
77
|
+
language="EN",
|
|
78
|
+
celex="62011CC0444",
|
|
79
|
+
)
|
|
80
|
+
is None
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_normalize_celex_converts_infocuria_dot_variant():
|
|
85
|
+
assert eurlex_scraping.normalize_celex("62011FO0005.01") == "62011FO0005(01)"
|
|
86
|
+
|
|
87
|
+
|
|
43
88
|
def test_extract_summary_from_documents_prefers_summary_marker():
|
|
44
89
|
docs = [
|
|
45
90
|
{
|
|
@@ -103,9 +148,13 @@ def test_get_case_data_by_celex_id_builds_blob_request(monkeypatch):
|
|
|
103
148
|
"content": {
|
|
104
149
|
"matCodeML": [{"label": [{"en": "Environment"}]}],
|
|
105
150
|
"matCode": ["ENVI"],
|
|
106
|
-
"advocateML": [
|
|
151
|
+
"advocateML": [
|
|
152
|
+
{"code": "KOK", "label": [{"en": "Kokott"}]}
|
|
153
|
+
],
|
|
107
154
|
"avg": "KOK",
|
|
108
|
-
"reportingJudgeML": [
|
|
155
|
+
"reportingJudgeML": [
|
|
156
|
+
{"code": "SGE", "label": [{"en": "Spielmann"}]}
|
|
157
|
+
],
|
|
109
158
|
"reportingJudge": "SGE",
|
|
110
159
|
"joinAffairs": ["C-2/20"],
|
|
111
160
|
"procedureResultTypeML": [{"label": [{"en": "Judgment"}]}],
|
|
@@ -158,7 +207,8 @@ def test_get_case_data_by_celex_id_builds_blob_request(monkeypatch):
|
|
|
158
207
|
# noops — supplementation behaviour itself is covered in
|
|
159
208
|
# test_sector6_cellar_supplement.py.
|
|
160
209
|
monkeypatch.setattr(
|
|
161
|
-
eurlex_scraping,
|
|
210
|
+
eurlex_scraping,
|
|
211
|
+
"_fetch_sector8_work_uri",
|
|
162
212
|
lambda celex, sector="8": "",
|
|
163
213
|
)
|
|
164
214
|
|
|
@@ -130,6 +130,15 @@ def test_sector6_infocuria_returns_fulltexts_list_with_all_languages(monkeypatch
|
|
|
130
130
|
"docTypeCode": "ARRET",
|
|
131
131
|
}
|
|
132
132
|
},
|
|
133
|
+
{
|
|
134
|
+
"content": {
|
|
135
|
+
"docLang": "IT",
|
|
136
|
+
"docFormats": ["HTML"],
|
|
137
|
+
"logicDocId": "id_unrelated_opinion",
|
|
138
|
+
"idProcedure": "C/0131/24/00000000RP/01/P/01",
|
|
139
|
+
"docTypeCode": "CONCL",
|
|
140
|
+
}
|
|
141
|
+
},
|
|
133
142
|
]
|
|
134
143
|
}
|
|
135
144
|
},
|
|
@@ -249,7 +258,9 @@ def test_sector6_cellar_fallback_returns_all_languages(monkeypatch):
|
|
|
249
258
|
{"item_url": "u_it", "format": "xhtml", "language": "IT"},
|
|
250
259
|
{"item_url": "u_fr", "format": "html", "language": "FR"},
|
|
251
260
|
]
|
|
252
|
-
monkeypatch.setattr(
|
|
261
|
+
monkeypatch.setattr(
|
|
262
|
+
eurlex_scraping, "_post_json", lambda url, payload, retries=3: []
|
|
263
|
+
)
|
|
253
264
|
monkeypatch.setattr(
|
|
254
265
|
eurlex_scraping, "_fetch_sector8_work_uri", lambda celex, sector="8": work_uri
|
|
255
266
|
)
|
|
@@ -357,6 +368,8 @@ def test_build_fulltext_records_normalizes_composite_celex():
|
|
|
357
368
|
"missing_reasons": "",
|
|
358
369
|
}
|
|
359
370
|
]
|
|
371
|
+
|
|
372
|
+
|
|
360
373
|
def test_build_fulltext_records_falls_back_to_single_when_no_fulltexts_list():
|
|
361
374
|
"""Backwards-compat: legacy infocuria_data dicts that don't carry a
|
|
362
375
|
`fulltexts` list (e.g. from a plug-in or stub) still produce exactly
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/scm_file_list.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_webservice_credentials_integration.py
RENAMED
|
File without changes
|
{cellar_extractor-2.0.3 → cellar_extractor-2.0.4}/tests/test_webservice_redundancy_integration.py
RENAMED
|
File without changes
|