cellar-extractor 2.0.2__tar.gz → 2.0.4__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cellar_extractor-2.0.2/cellar_extractor.egg-info → cellar_extractor-2.0.4}/PKG-INFO +22 -1
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/README.md +21 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/__init__.py +3 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/_version.py +3 -3
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar.py +31 -4
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_queries.py +199 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/eurlex_scraping.py +128 -25
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4/cellar_extractor.egg-info}/PKG-INFO +22 -1
- cellar_extractor-2.0.4/cellar_extractor.egg-info/scm_version.json +8 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar.py +72 -19
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_queries_local.py +107 -1
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_infocuria_adapter.py +53 -3
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_metadata_hygiene_local.py +34 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_multilang_fulltext.py +14 -1
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_supplement.py +12 -18
- cellar_extractor-2.0.2/cellar_extractor.egg-info/scm_version.json +0 -8
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.env.example +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.flake8 +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.github/workflows/ci.yml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/.github/workflows/github-actions.yml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/FIELDS.md +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/LICENSE +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/MANIFEST.in +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/build_package.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_extra_extract.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/citations_adder.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/csv_extractor.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/fulltext_saving.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/json_to_csv.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/nodes_and_edges.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/operative_extractions.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/persistence.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/schema.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor/sparql.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/SOURCES.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/dependency_links.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/requires.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/scm_file_list.json +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/cellar_extractor.egg-info/top_level.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/pyproject.toml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/requirements.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/setup.cfg +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/setup.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/61986CJ0062.ENG.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_citation_graph_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_citations_adder_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_corpus_2020_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_extra_cellar_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_fulltext_saving_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_infocuria_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_nodes_and_edges_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_operative_extractions_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_real_fetch_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_retry_behavior.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_samples_dump_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_schema_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector3_adapter.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector6_cellar_fallback.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector8_adapter.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sector8_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_sparql_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_webservice_credentials_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.4}/tests/test_webservice_redundancy_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cellar-extractor
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.4
|
|
4
4
|
Summary: Library for extracting CELLAR case law data from EUR-Lex
|
|
5
5
|
Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
|
|
6
6
|
License: Apache-2.0
|
|
@@ -127,6 +127,11 @@ df = cell.get_cellar(
|
|
|
127
127
|
|
|
128
128
|
Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
|
|
129
129
|
|
|
130
|
+
For direct manifestation access, use
|
|
131
|
+
`get_cellar_manifestations_by_celex()`. It canonicalizes composite and
|
|
132
|
+
`_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
|
|
133
|
+
derived summaries or notices from being returned as judgment full text.
|
|
134
|
+
|
|
130
135
|
You can also save explicitly to a custom path instead of the default `data/` location:
|
|
131
136
|
|
|
132
137
|
```python
|
|
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
|
|
|
153
158
|
)
|
|
154
159
|
```
|
|
155
160
|
|
|
161
|
+
For targeted CELEX repair or supplementation, use the public CELLAR
|
|
162
|
+
manifestation API. It unions every CELLAR work sharing the CELEX before choosing
|
|
163
|
+
the best downloadable item per language:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
import cellar_extractor as cell
|
|
167
|
+
|
|
168
|
+
_, manifestations = cell.get_cellar_manifestations_by_celex(
|
|
169
|
+
"62020CJ0414", sector="6"
|
|
170
|
+
)
|
|
171
|
+
english = [m for m in manifestations if m["language"] == "EN"]
|
|
172
|
+
fulltexts = cell.extract_cellar_fulltexts(english)
|
|
173
|
+
```
|
|
174
|
+
|
|
156
175
|
Returns:
|
|
157
176
|
|
|
158
177
|
- `extra_df`: enriched dataframe
|
|
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
|
|
|
303
322
|
| `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
|
|
304
323
|
| `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
|
|
305
324
|
| `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
|
|
325
|
+
| `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
|
|
326
|
+
| `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
|
|
306
327
|
| `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
|
|
307
328
|
| `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
|
|
308
329
|
| `FetchOperativePart` | Extract operative part from a single case document |
|
|
@@ -92,6 +92,11 @@ df = cell.get_cellar(
|
|
|
92
92
|
|
|
93
93
|
Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
|
|
94
94
|
|
|
95
|
+
For direct manifestation access, use
|
|
96
|
+
`get_cellar_manifestations_by_celex()`. It canonicalizes composite and
|
|
97
|
+
`_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
|
|
98
|
+
derived summaries or notices from being returned as judgment full text.
|
|
99
|
+
|
|
95
100
|
You can also save explicitly to a custom path instead of the default `data/` location:
|
|
96
101
|
|
|
97
102
|
```python
|
|
@@ -118,6 +123,20 @@ extra_df, fulltext = cell.get_cellar_extra(
|
|
|
118
123
|
)
|
|
119
124
|
```
|
|
120
125
|
|
|
126
|
+
For targeted CELEX repair or supplementation, use the public CELLAR
|
|
127
|
+
manifestation API. It unions every CELLAR work sharing the CELEX before choosing
|
|
128
|
+
the best downloadable item per language:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
import cellar_extractor as cell
|
|
132
|
+
|
|
133
|
+
_, manifestations = cell.get_cellar_manifestations_by_celex(
|
|
134
|
+
"62020CJ0414", sector="6"
|
|
135
|
+
)
|
|
136
|
+
english = [m for m in manifestations if m["language"] == "EN"]
|
|
137
|
+
fulltexts = cell.extract_cellar_fulltexts(english)
|
|
138
|
+
```
|
|
139
|
+
|
|
121
140
|
Returns:
|
|
122
141
|
|
|
123
142
|
- `extra_df`: enriched dataframe
|
|
@@ -268,6 +287,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
|
|
|
268
287
|
| `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
|
|
269
288
|
| `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
|
|
270
289
|
| `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
|
|
290
|
+
| `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
|
|
291
|
+
| `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
|
|
271
292
|
| `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
|
|
272
293
|
| `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
|
|
273
294
|
| `FetchOperativePart` | Extract operative part from a single case document |
|
|
@@ -9,6 +9,9 @@ from cellar_extractor.cellar import get_cellar_extra
|
|
|
9
9
|
from cellar_extractor.cellar import get_nodes_and_edges_lists
|
|
10
10
|
from cellar_extractor.cellar import filter_subject_matter
|
|
11
11
|
from cellar_extractor.eurlex_scraping import get_legislation_by_celex_id
|
|
12
|
+
from cellar_extractor.eurlex_scraping import get_cellar_manifestations_by_celex
|
|
13
|
+
from cellar_extractor.eurlex_scraping import extract_cellar_fulltexts
|
|
14
|
+
from cellar_extractor.eurlex_scraping import normalize_celex
|
|
12
15
|
from cellar_extractor.operative_extractions import FetchOperativePart
|
|
13
16
|
from cellar_extractor.operative_extractions import Writing
|
|
14
17
|
import logging
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (2, 0,
|
|
21
|
+
__version__ = version = '2.0.4'
|
|
22
|
+
__version_tuple__ = version_tuple = (2, 0, 4)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g792a41d4a'
|
|
@@ -4,7 +4,12 @@ from concurrent.futures import ThreadPoolExecutor
|
|
|
4
4
|
from tqdm import tqdm
|
|
5
5
|
|
|
6
6
|
from cellar_extractor.cellar_extra_extract import extra_cellar
|
|
7
|
-
from cellar_extractor.cellar_queries import
|
|
7
|
+
from cellar_extractor.cellar_queries import (
|
|
8
|
+
get_all_eclis,
|
|
9
|
+
get_infocuria_document_metadata,
|
|
10
|
+
get_raw_cellar_metadata,
|
|
11
|
+
reconcile_document_metadata,
|
|
12
|
+
)
|
|
8
13
|
from cellar_extractor.json_to_csv import json_to_csv_returning
|
|
9
14
|
from cellar_extractor.nodes_and_edges import get_nodes_and_edges
|
|
10
15
|
from cellar_extractor.persistence import (
|
|
@@ -70,6 +75,7 @@ def get_cellar(
|
|
|
70
75
|
output_path=None,
|
|
71
76
|
return_data=None,
|
|
72
77
|
save=None,
|
|
78
|
+
reconcile_infocuria=False,
|
|
73
79
|
):
|
|
74
80
|
"""
|
|
75
81
|
Fetch base CELLAR metadata.
|
|
@@ -91,10 +97,24 @@ def get_cellar(
|
|
|
91
97
|
logging.info(f"Up until the specified end date {ed}")
|
|
92
98
|
eclis = get_all_eclis(starting_date=sd, ending_date=ed, limit=max_ecli)
|
|
93
99
|
logging.info(f"Found {len(eclis)} ECLIs")
|
|
94
|
-
|
|
100
|
+
all_eclis = _fetch_metadata_batches(eclis)
|
|
101
|
+
if reconcile_infocuria:
|
|
102
|
+
infocuria_metadata = get_infocuria_document_metadata(
|
|
103
|
+
starting_date=sd,
|
|
104
|
+
ending_date=ed,
|
|
105
|
+
limit=max_ecli,
|
|
106
|
+
)
|
|
107
|
+
logging.info(
|
|
108
|
+
"Found %s ECLIs in the InfoCuria document catalogue",
|
|
109
|
+
len(infocuria_metadata),
|
|
110
|
+
)
|
|
111
|
+
all_eclis = reconcile_document_metadata(all_eclis, infocuria_metadata)
|
|
112
|
+
if max_ecli is not None and len(all_eclis) > max_ecli:
|
|
113
|
+
all_eclis = dict(list(all_eclis.items())[:max_ecli])
|
|
114
|
+
|
|
115
|
+
if len(all_eclis) == 0:
|
|
95
116
|
logging.info(f"No data to download found between {sd} and {ed}")
|
|
96
117
|
return False
|
|
97
|
-
all_eclis = _fetch_metadata_batches(eclis)
|
|
98
118
|
|
|
99
119
|
result = _materialize_cellar_output(all_eclis, file_format)
|
|
100
120
|
if save_enabled:
|
|
@@ -136,7 +156,14 @@ def get_cellar_extra(
|
|
|
136
156
|
if not ed:
|
|
137
157
|
ed = datetime.now().isoformat(timespec="seconds")
|
|
138
158
|
save_enabled = resolve_save_enabled(save=save, save_file=save_file, default=True)
|
|
139
|
-
data = get_cellar(
|
|
159
|
+
data = get_cellar(
|
|
160
|
+
ed=ed,
|
|
161
|
+
save=False,
|
|
162
|
+
max_ecli=max_ecli,
|
|
163
|
+
sd=sd,
|
|
164
|
+
file_format="csv",
|
|
165
|
+
reconcile_infocuria=True,
|
|
166
|
+
)
|
|
140
167
|
if data is False:
|
|
141
168
|
logging.warning("Cellar extraction unsuccessful")
|
|
142
169
|
return False, False
|
|
@@ -1,6 +1,8 @@
|
|
|
1
|
+
import re
|
|
1
2
|
import time
|
|
2
3
|
from datetime import date, datetime, timedelta
|
|
3
4
|
|
|
5
|
+
import requests
|
|
4
6
|
from SPARQLWrapper import SPARQLWrapper, JSON, POST
|
|
5
7
|
|
|
6
8
|
# Literal placeholder CELLAR emits while a property is awaiting curation;
|
|
@@ -13,6 +15,17 @@ MAX_SORTED_TOP_LIMIT = 10000
|
|
|
13
15
|
ECLI_WINDOW_DAYS = 366
|
|
14
16
|
SPARQL_REQUEST_TIMEOUT_SECONDS = 30
|
|
15
17
|
SPARQL_RETRY_BACKOFF_BASE_SECONDS = 0.5
|
|
18
|
+
INFOCURIA_SEARCH_ENDPOINT = "https://infocuriaws.curia.europa.eu/elastic-connector/search"
|
|
19
|
+
INFOCURIA_PAGE_SIZE = 100
|
|
20
|
+
INFOCURIA_REQUEST_TIMEOUT_SECONDS = 60
|
|
21
|
+
INFOCURIA_IDENTITY_FIELDS = {
|
|
22
|
+
"case-law_ecli",
|
|
23
|
+
"resource_legal_id_celex",
|
|
24
|
+
"work_date_document",
|
|
25
|
+
"resource_legal_type",
|
|
26
|
+
"resource_legal_id_sector",
|
|
27
|
+
"case-law_affaire_number",
|
|
28
|
+
}
|
|
16
29
|
|
|
17
30
|
|
|
18
31
|
def _query_with_retries(sparql, retries, error_message):
|
|
@@ -157,6 +170,192 @@ def get_all_eclis(starting_date=None, ending_date=None, limit=None, max_retries=
|
|
|
157
170
|
return eclis
|
|
158
171
|
|
|
159
172
|
|
|
173
|
+
def _normalize_infocuria_celex(value):
|
|
174
|
+
"""Return a canonical primary-document CELEX from an InfoCuria hit.
|
|
175
|
+
|
|
176
|
+
InfoCuria writes numbered document variants as ``.01`` while EUR-Lex and
|
|
177
|
+
CELLAR use ``(01)``. Derived summary/information works are deliberately
|
|
178
|
+
rejected instead of being collapsed onto their base CELEX.
|
|
179
|
+
"""
|
|
180
|
+
if value is None:
|
|
181
|
+
return ""
|
|
182
|
+
celex = str(value).replace(" ", "").strip()
|
|
183
|
+
if celex == "":
|
|
184
|
+
return ""
|
|
185
|
+
celex = celex.split(";", 1)[0]
|
|
186
|
+
if re.search(r"_(?:SUM|RES|INF)$", celex, flags=re.IGNORECASE):
|
|
187
|
+
return ""
|
|
188
|
+
celex = re.sub(r"\.(\d{2})$", r"(\1)", celex)
|
|
189
|
+
if not celex.startswith("6"):
|
|
190
|
+
return ""
|
|
191
|
+
return celex
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def _build_infocuria_search_payload(starting_date, ending_date, page_number, page_size):
|
|
195
|
+
start = page_number * page_size + 1
|
|
196
|
+
return {
|
|
197
|
+
"multiSearchTerms": [],
|
|
198
|
+
"searchTerm": "",
|
|
199
|
+
"ecli": "",
|
|
200
|
+
"publishedId": "",
|
|
201
|
+
"usualName": "",
|
|
202
|
+
"logicDocId": "",
|
|
203
|
+
"repJurExpand": False,
|
|
204
|
+
"pagination": {
|
|
205
|
+
"pageNumber": page_number,
|
|
206
|
+
"pageSize": page_size,
|
|
207
|
+
"from": start,
|
|
208
|
+
"to": start + page_size - 1,
|
|
209
|
+
"origin": "jurisprudence",
|
|
210
|
+
},
|
|
211
|
+
"sortTermList": [
|
|
212
|
+
{
|
|
213
|
+
"sortDirection": "ASC",
|
|
214
|
+
"sortTerm": "DOC_DATE",
|
|
215
|
+
"sortSourceTab": "jurisprudence",
|
|
216
|
+
}
|
|
217
|
+
],
|
|
218
|
+
"filtersValue": [{"field": "docDate", "values": [starting_date, ending_date]}],
|
|
219
|
+
"advancedFiltersValue": [],
|
|
220
|
+
"language": "EN",
|
|
221
|
+
"isSearchExact": False,
|
|
222
|
+
"searchSources": ["document", "metadata"],
|
|
223
|
+
"tabName": "jurisprudence",
|
|
224
|
+
"isAllTabsRequest": False,
|
|
225
|
+
}
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _query_infocuria_page(payload, max_retries):
|
|
229
|
+
last_error = None
|
|
230
|
+
for attempt in range(max_retries):
|
|
231
|
+
try:
|
|
232
|
+
response = requests.post(
|
|
233
|
+
INFOCURIA_SEARCH_ENDPOINT,
|
|
234
|
+
json=payload,
|
|
235
|
+
timeout=INFOCURIA_REQUEST_TIMEOUT_SECONDS,
|
|
236
|
+
)
|
|
237
|
+
response.raise_for_status()
|
|
238
|
+
result = response.json()
|
|
239
|
+
if not isinstance(result, dict):
|
|
240
|
+
raise ValueError("InfoCuria search response is not an object")
|
|
241
|
+
return result
|
|
242
|
+
except Exception as exc:
|
|
243
|
+
last_error = exc
|
|
244
|
+
if attempt < max_retries - 1:
|
|
245
|
+
time.sleep(SPARQL_RETRY_BACKOFF_BASE_SECONDS * (2**attempt))
|
|
246
|
+
raise RuntimeError(
|
|
247
|
+
"Failed to query InfoCuria document catalogue after retries"
|
|
248
|
+
) from last_error
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _infocuria_metadata_from_hit(hit):
|
|
252
|
+
content = hit.get("content", {}) if isinstance(hit, dict) else {}
|
|
253
|
+
if not isinstance(content, dict):
|
|
254
|
+
return None
|
|
255
|
+
|
|
256
|
+
ecli = str(content.get("ecli") or "").strip()
|
|
257
|
+
celex = _normalize_infocuria_celex(content.get("celex"))
|
|
258
|
+
document_date = str(content.get("docDate") or "").strip()
|
|
259
|
+
if not ecli.startswith("ECLI:EU:") or celex == "" or document_date == "":
|
|
260
|
+
return None
|
|
261
|
+
|
|
262
|
+
resource_type = celex[5:7] if len(celex) >= 7 else ""
|
|
263
|
+
metadata = {
|
|
264
|
+
"case-law_ecli": [ecli],
|
|
265
|
+
"resource_legal_id_celex": [celex],
|
|
266
|
+
"work_date_document": [document_date],
|
|
267
|
+
"resource_legal_id_sector": [celex[0]],
|
|
268
|
+
"metadata_catalog_source": ["infocuria"],
|
|
269
|
+
}
|
|
270
|
+
if resource_type:
|
|
271
|
+
metadata["resource_legal_type"] = [resource_type]
|
|
272
|
+
published_id = str(content.get("idPublished") or "").strip()
|
|
273
|
+
if published_id:
|
|
274
|
+
metadata["case-law_affaire_number"] = [published_id]
|
|
275
|
+
return ecli, metadata
|
|
276
|
+
|
|
277
|
+
|
|
278
|
+
def get_infocuria_document_metadata(
|
|
279
|
+
starting_date=None,
|
|
280
|
+
ending_date=None,
|
|
281
|
+
limit=None,
|
|
282
|
+
max_retries=3,
|
|
283
|
+
):
|
|
284
|
+
"""Enumerate official InfoCuria documents for a date range.
|
|
285
|
+
|
|
286
|
+
CELLAR's SPARQL graph does not contain every document that is available
|
|
287
|
+
through EUR-Lex/InfoCuria, particularly procedural orders. InfoCuria is
|
|
288
|
+
therefore used as a second catalogue source. Results use the same
|
|
289
|
+
predicate-map shape as :func:`get_raw_cellar_metadata` so callers can
|
|
290
|
+
reconcile the sources before normal schema flattening.
|
|
291
|
+
"""
|
|
292
|
+
metadata = {}
|
|
293
|
+
for window_start, window_end in _build_ecli_windows(
|
|
294
|
+
starting_date=starting_date,
|
|
295
|
+
ending_date=ending_date,
|
|
296
|
+
):
|
|
297
|
+
page_number = 0
|
|
298
|
+
while True:
|
|
299
|
+
remaining = None if limit is None else limit - len(metadata)
|
|
300
|
+
if remaining is not None and remaining <= 0:
|
|
301
|
+
return metadata
|
|
302
|
+
page_size = INFOCURIA_PAGE_SIZE
|
|
303
|
+
|
|
304
|
+
payload = _build_infocuria_search_payload(
|
|
305
|
+
str(window_start)[:10],
|
|
306
|
+
str(window_end)[:10],
|
|
307
|
+
page_number,
|
|
308
|
+
page_size,
|
|
309
|
+
)
|
|
310
|
+
result = _query_infocuria_page(payload, max_retries=max_retries)
|
|
311
|
+
hits = result.get("searchHits", [])
|
|
312
|
+
if not isinstance(hits, list):
|
|
313
|
+
raise RuntimeError("InfoCuria catalogue returned invalid searchHits")
|
|
314
|
+
|
|
315
|
+
for hit in hits:
|
|
316
|
+
parsed = _infocuria_metadata_from_hit(hit)
|
|
317
|
+
if parsed is None:
|
|
318
|
+
continue
|
|
319
|
+
ecli, values = parsed
|
|
320
|
+
# Multiple logical documents can share an ECLI. Their CELEX
|
|
321
|
+
# identity is normally identical; retaining the first hit is
|
|
322
|
+
# deterministic because the endpoint is date-sorted.
|
|
323
|
+
metadata.setdefault(ecli, values)
|
|
324
|
+
if limit is not None and len(metadata) >= limit:
|
|
325
|
+
return metadata
|
|
326
|
+
|
|
327
|
+
total_hits = int(result.get("totalHits") or 0)
|
|
328
|
+
consumed = (page_number + 1) * page_size
|
|
329
|
+
if not hits or consumed >= total_hits:
|
|
330
|
+
break
|
|
331
|
+
page_number += 1
|
|
332
|
+
return metadata
|
|
333
|
+
|
|
334
|
+
|
|
335
|
+
def reconcile_document_metadata(cellar_metadata, infocuria_metadata):
|
|
336
|
+
"""Merge both catalogues, preferring InfoCuria for document identity.
|
|
337
|
+
|
|
338
|
+
Rich CELLAR metadata is retained. InfoCuria supplies missing documents
|
|
339
|
+
and corrects identity fields when CELLAR associates an ECLI with a stale
|
|
340
|
+
or sibling CELEX work.
|
|
341
|
+
"""
|
|
342
|
+
reconciled = {
|
|
343
|
+
ecli: {key: list(values) for key, values in values_by_key.items()}
|
|
344
|
+
for ecli, values_by_key in (cellar_metadata or {}).items()
|
|
345
|
+
}
|
|
346
|
+
for ecli, values_by_key in (infocuria_metadata or {}).items():
|
|
347
|
+
if ecli not in reconciled:
|
|
348
|
+
reconciled[ecli] = {
|
|
349
|
+
key: list(values) for key, values in values_by_key.items()
|
|
350
|
+
}
|
|
351
|
+
continue
|
|
352
|
+
target = reconciled[ecli]
|
|
353
|
+
for key, values in values_by_key.items():
|
|
354
|
+
if key in INFOCURIA_IDENTITY_FIELDS or not target.get(key):
|
|
355
|
+
target[key] = list(values)
|
|
356
|
+
return reconciled
|
|
357
|
+
|
|
358
|
+
|
|
160
359
|
def get_raw_cellar_metadata_by_celex(
|
|
161
360
|
celex_ids,
|
|
162
361
|
get_labels=True,
|
|
@@ -126,9 +126,21 @@ def _normalize_celex(celex):
|
|
|
126
126
|
value = non_inf[0] if non_inf else options[0]
|
|
127
127
|
if "_" in value:
|
|
128
128
|
value = value.split("_")[0]
|
|
129
|
+
value = re.sub(r"\.(\d{2})$", r"(\1)", value)
|
|
129
130
|
return value
|
|
130
131
|
|
|
131
132
|
|
|
133
|
+
def normalize_celex(celex):
|
|
134
|
+
"""Return the canonical work CELEX used for full-text retrieval.
|
|
135
|
+
|
|
136
|
+
CELLAR metadata can bundle a primary document with derived information,
|
|
137
|
+
summary, and résumé works (for example ``62020CJ0414_SUM;62020CJ0414``).
|
|
138
|
+
Full-text callers must resolve the unsuffixed base work so a derived work
|
|
139
|
+
can never occupy the judgment's language slot.
|
|
140
|
+
"""
|
|
141
|
+
return _normalize_celex(celex)
|
|
142
|
+
|
|
143
|
+
|
|
132
144
|
def _sleep_with_backoff(attempt, base=0.2):
|
|
133
145
|
time.sleep(base * (attempt + 1) + random.uniform(0.0, 0.1))
|
|
134
146
|
|
|
@@ -413,6 +425,25 @@ def _fetch_sector8_items_for_celex(celex, sector="8"):
|
|
|
413
425
|
return work_uris, candidates
|
|
414
426
|
|
|
415
427
|
|
|
428
|
+
def get_cellar_manifestations_by_celex(celex, sector="8"):
|
|
429
|
+
"""Return all CELLAR works and manifestation candidates for a CELEX.
|
|
430
|
+
|
|
431
|
+
A CELEX can resolve to multiple CELLAR works with different language
|
|
432
|
+
coverage. This public entry point deliberately returns the union across
|
|
433
|
+
every matching work so callers do not need to depend on the extractor's
|
|
434
|
+
private sector-8 helpers.
|
|
435
|
+
|
|
436
|
+
The return value is ``(work_uris, manifestations)``. Each manifestation
|
|
437
|
+
contains ``item_url``, ``format``, and ``language`` keys and is deduplicated
|
|
438
|
+
on that triple. ``sector`` defaults to ``"8"`` and may be set to ``"6"``
|
|
439
|
+
for CJEU documents.
|
|
440
|
+
"""
|
|
441
|
+
canonical_celex = normalize_celex(celex)
|
|
442
|
+
if not canonical_celex:
|
|
443
|
+
return [], []
|
|
444
|
+
return _fetch_sector8_items_for_celex(canonical_celex, sector=sector)
|
|
445
|
+
|
|
446
|
+
|
|
416
447
|
def _fetch_sector8_work_uri(celex, sector="8"):
|
|
417
448
|
"""Resolve a CELEX to its CELLAR work URI.
|
|
418
449
|
|
|
@@ -553,6 +584,17 @@ def _fanout_fulltexts_from_candidates(candidates, source_label):
|
|
|
553
584
|
return out
|
|
554
585
|
|
|
555
586
|
|
|
587
|
+
def extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM"):
|
|
588
|
+
"""Download the best manifestation per language as fulltext records.
|
|
589
|
+
|
|
590
|
+
``manifestations`` is the candidate list returned by
|
|
591
|
+
:func:`get_cellar_manifestations_by_celex`. Empty bodies are omitted and
|
|
592
|
+
each returned dictionary contains ``text``, ``html``, ``text_source``,
|
|
593
|
+
``text_language``, and ``text_format``.
|
|
594
|
+
"""
|
|
595
|
+
return _fanout_fulltexts_from_candidates(manifestations, source_label)
|
|
596
|
+
|
|
597
|
+
|
|
556
598
|
def _get_case_data_sector8(celex, language="EN"):
|
|
557
599
|
work_uris, main_candidates = _fetch_sector8_items_for_celex(celex)
|
|
558
600
|
|
|
@@ -773,7 +815,7 @@ def _collect_affecting_ids(content):
|
|
|
773
815
|
return ids
|
|
774
816
|
|
|
775
817
|
|
|
776
|
-
def _choose_best_document(doc_hits, language="EN"):
|
|
818
|
+
def _choose_best_document(doc_hits, language="EN", celex=None):
|
|
777
819
|
candidates = []
|
|
778
820
|
for hit in doc_hits or []:
|
|
779
821
|
content = hit.get("content", {}) if isinstance(hit, dict) else {}
|
|
@@ -786,6 +828,23 @@ def _choose_best_document(doc_hits, language="EN"):
|
|
|
786
828
|
if len(candidates) == 0:
|
|
787
829
|
return None
|
|
788
830
|
|
|
831
|
+
normalized_target = _normalize_celex(celex) if celex else ""
|
|
832
|
+
if normalized_target:
|
|
833
|
+
candidates_with_celex = [
|
|
834
|
+
doc for doc in candidates if _normalize_celex(doc.get("celex", ""))
|
|
835
|
+
]
|
|
836
|
+
exact_candidates = [
|
|
837
|
+
doc
|
|
838
|
+
for doc in candidates_with_celex
|
|
839
|
+
if _normalize_celex(doc.get("celex", "")) == normalized_target
|
|
840
|
+
]
|
|
841
|
+
if exact_candidates:
|
|
842
|
+
candidates = exact_candidates
|
|
843
|
+
elif candidates_with_celex:
|
|
844
|
+
# Selecting a judgment merely because it is the highest-ranked
|
|
845
|
+
# document can attach a sibling judgment to an order CELEX.
|
|
846
|
+
return None
|
|
847
|
+
|
|
789
848
|
type_priority = {
|
|
790
849
|
"ARRET": 0,
|
|
791
850
|
"ORDONNANCE": 1,
|
|
@@ -987,7 +1046,11 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
987
1046
|
if isinstance(root_hit, dict)
|
|
988
1047
|
else []
|
|
989
1048
|
)
|
|
990
|
-
selected_doc = _choose_best_document(
|
|
1049
|
+
selected_doc = _choose_best_document(
|
|
1050
|
+
documents,
|
|
1051
|
+
language=language,
|
|
1052
|
+
celex=normalized,
|
|
1053
|
+
)
|
|
991
1054
|
if selected_doc is None:
|
|
992
1055
|
return _get_case_data_sector6_cellar_fallback(normalized, language=language)
|
|
993
1056
|
|
|
@@ -1043,18 +1106,41 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1043
1106
|
else:
|
|
1044
1107
|
summary_source = ""
|
|
1045
1108
|
|
|
1046
|
-
# Multi-language fanout
|
|
1047
|
-
#
|
|
1048
|
-
#
|
|
1109
|
+
# Multi-language fanout must remain within the selected logical document.
|
|
1110
|
+
# A procedure can contain judgments, orders, opinions, and notices. The
|
|
1111
|
+
# selected document's groupByLogicalId list is the authoritative language
|
|
1112
|
+
# family; older responses without that field fall back to documents with
|
|
1113
|
+
# the same logicDocId.
|
|
1049
1114
|
fulltexts: list = []
|
|
1050
1115
|
seen_langs: set = set()
|
|
1051
1116
|
session = _get_http_session()
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
|
|
1055
|
-
|
|
1056
|
-
|
|
1057
|
-
|
|
1117
|
+
selected_logic_id = str(selected_doc.get("logicDocId", ""))
|
|
1118
|
+
selected_group = selected_doc.get("groupByLogicalId") or []
|
|
1119
|
+
variant_contents = []
|
|
1120
|
+
if isinstance(selected_group, list) and selected_group:
|
|
1121
|
+
for variant in selected_group:
|
|
1122
|
+
if not isinstance(variant, dict):
|
|
1123
|
+
continue
|
|
1124
|
+
variant_contents.append(
|
|
1125
|
+
{
|
|
1126
|
+
"docLang": variant.get("docLang"),
|
|
1127
|
+
"docFormats": variant.get("formats") or [],
|
|
1128
|
+
"logicDocId": selected_logic_id,
|
|
1129
|
+
"idProcedure": variant.get("idProcedure")
|
|
1130
|
+
or selected_doc.get("idProcedure"),
|
|
1131
|
+
}
|
|
1132
|
+
)
|
|
1133
|
+
else:
|
|
1134
|
+
for doc in documents or []:
|
|
1135
|
+
content = doc.get("content", {}) if isinstance(doc, dict) else {}
|
|
1136
|
+
if not isinstance(content, dict):
|
|
1137
|
+
continue
|
|
1138
|
+
if str(content.get("logicDocId", "")) == selected_logic_id:
|
|
1139
|
+
variant_contents.append(content)
|
|
1140
|
+
if not variant_contents:
|
|
1141
|
+
variant_contents = [selected_doc]
|
|
1142
|
+
|
|
1143
|
+
for content in variant_contents:
|
|
1058
1144
|
variant_lang = content.get("docLang") or ""
|
|
1059
1145
|
variant_lang_upper = str(variant_lang).upper()
|
|
1060
1146
|
if variant_lang_upper == "" or variant_lang_upper in seen_langs:
|
|
@@ -1104,15 +1190,14 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1104
1190
|
}
|
|
1105
1191
|
)
|
|
1106
1192
|
|
|
1107
|
-
#
|
|
1108
|
-
#
|
|
1109
|
-
#
|
|
1110
|
-
#
|
|
1111
|
-
#
|
|
1112
|
-
#
|
|
1113
|
-
#
|
|
1114
|
-
#
|
|
1115
|
-
# populate them.
|
|
1193
|
+
# Merge InfoCuria fulltexts with every language CELLAR has. CELLAR is
|
|
1194
|
+
# authoritative when both sources expose the same language because its
|
|
1195
|
+
# manifestation belongs to the canonical CELEX work. InfoCuria's
|
|
1196
|
+
# procedure search has returned a different document under the judgment
|
|
1197
|
+
# slot in production (AG opinions, procedural orders, and notices), so an
|
|
1198
|
+
# existing InfoCuria language must not block the canonical manifestation.
|
|
1199
|
+
# Metadata fields (judge, advocate, directory_codes, etc.) remain
|
|
1200
|
+
# InfoCuria-sourced — CELLAR cannot populate them.
|
|
1116
1201
|
#
|
|
1117
1202
|
# All-in-one try/except: a CELLAR-side failure must never kill the
|
|
1118
1203
|
# InfoCuria-sourced row we already built.
|
|
@@ -1122,18 +1207,36 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1122
1207
|
cellar_fulltexts = _fanout_fulltexts_from_candidates(
|
|
1123
1208
|
cellar_candidates, source_label="CELLAR_ITEM"
|
|
1124
1209
|
)
|
|
1210
|
+
by_language = {
|
|
1211
|
+
entry.get("text_language", "").upper(): entry
|
|
1212
|
+
for entry in fulltexts
|
|
1213
|
+
if entry.get("text_language", "")
|
|
1214
|
+
}
|
|
1125
1215
|
for entry in cellar_fulltexts:
|
|
1126
1216
|
entry_lang = entry.get("text_language", "").upper()
|
|
1127
|
-
if not entry_lang
|
|
1217
|
+
if not entry_lang:
|
|
1128
1218
|
continue
|
|
1129
1219
|
seen_langs.add(entry_lang)
|
|
1130
|
-
|
|
1220
|
+
by_language[entry_lang] = entry
|
|
1221
|
+
fulltexts = list(by_language.values())
|
|
1131
1222
|
except Exception:
|
|
1132
1223
|
# CELLAR supplementation is best-effort. If anything goes wrong
|
|
1133
1224
|
# (SPARQL timeout, network hiccup, schema drift on the manifestation
|
|
1134
1225
|
# graph) the InfoCuria-only result is still returned.
|
|
1135
1226
|
pass
|
|
1136
1227
|
|
|
1228
|
+
primary = next(
|
|
1229
|
+
(
|
|
1230
|
+
entry
|
|
1231
|
+
for entry in fulltexts
|
|
1232
|
+
if entry.get("text_language", "").upper() == str(language).upper()
|
|
1233
|
+
),
|
|
1234
|
+
None,
|
|
1235
|
+
)
|
|
1236
|
+
if primary is not None:
|
|
1237
|
+
text = primary.get("text", "")
|
|
1238
|
+
html = primary.get("html", "")
|
|
1239
|
+
|
|
1137
1240
|
return {
|
|
1138
1241
|
"html": html,
|
|
1139
1242
|
"text": text,
|
|
@@ -1146,9 +1249,9 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1146
1249
|
"affecting_ids": affecting_ids,
|
|
1147
1250
|
"affecting_string": affecting_string,
|
|
1148
1251
|
"citations_extra": citations_extra,
|
|
1149
|
-
"text_source": "
|
|
1150
|
-
"text_language":
|
|
1151
|
-
"text_format": "
|
|
1252
|
+
"text_source": primary.get("text_source", "") if primary else "",
|
|
1253
|
+
"text_language": primary.get("text_language", "") if primary else "",
|
|
1254
|
+
"text_format": primary.get("text_format", "") if primary else "",
|
|
1152
1255
|
"summary_source": summary_source,
|
|
1153
1256
|
"summary_language": "EN" if summary != "" else "",
|
|
1154
1257
|
"sector": "6",
|