cellar-extractor 2.0.2__tar.gz → 2.0.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {cellar_extractor-2.0.2/cellar_extractor.egg-info → cellar_extractor-2.0.3}/PKG-INFO +22 -1
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/README.md +21 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/__init__.py +3 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/_version.py +3 -3
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/eurlex_scraping.py +72 -14
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3/cellar_extractor.egg-info}/PKG-INFO +22 -1
- cellar_extractor-2.0.3/cellar_extractor.egg-info/scm_version.json +8 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_metadata_hygiene_local.py +34 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector6_cellar_supplement.py +12 -18
- cellar_extractor-2.0.2/cellar_extractor.egg-info/scm_version.json +0 -8
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.env.example +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.flake8 +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.github/workflows/ci.yml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/.github/workflows/github-actions.yml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/FIELDS.md +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/LICENSE +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/MANIFEST.in +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/build_package.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_extra_extract.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_queries.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/citations_adder.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/csv_extractor.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/fulltext_saving.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/json_to_csv.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/nodes_and_edges.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/operative_extractions.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/persistence.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/schema.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor/sparql.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/SOURCES.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/dependency_links.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/requires.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/scm_file_list.json +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/top_level.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/pyproject.toml +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/requirements.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/setup.cfg +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/setup.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/61986CJ0062.ENG.txt +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_queries_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_cellar_sparql_queries.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_citation_graph_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_citations_adder_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_corpus_2020_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_extra_cellar_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_fulltext_saving_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_infocuria_adapter.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_infocuria_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_multilang_fulltext.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_nodes_and_edges_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_operative_extractions_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_real_fetch_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_retry_behavior.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_samples_dump_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_schema_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector3_adapter.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector6_cellar_fallback.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector8_adapter.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sector8_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_sparql_local.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_credentials_integration.py +0 -0
- {cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_redundancy_integration.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cellar-extractor
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.3
|
|
4
4
|
Summary: Library for extracting CELLAR case law data from EUR-Lex
|
|
5
5
|
Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
|
|
6
6
|
License: Apache-2.0
|
|
@@ -127,6 +127,11 @@ df = cell.get_cellar(
|
|
|
127
127
|
|
|
128
128
|
Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
|
|
129
129
|
|
|
130
|
+
For direct manifestation access, use
|
|
131
|
+
`get_cellar_manifestations_by_celex()`. It canonicalizes composite and
|
|
132
|
+
`_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
|
|
133
|
+
derived summaries or notices from being returned as judgment full text.
|
|
134
|
+
|
|
130
135
|
You can also save explicitly to a custom path instead of the default `data/` location:
|
|
131
136
|
|
|
132
137
|
```python
|
|
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
|
|
|
153
158
|
)
|
|
154
159
|
```
|
|
155
160
|
|
|
161
|
+
For targeted CELEX repair or supplementation, use the public CELLAR
|
|
162
|
+
manifestation API. It unions every CELLAR work sharing the CELEX before choosing
|
|
163
|
+
the best downloadable item per language:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
import cellar_extractor as cell
|
|
167
|
+
|
|
168
|
+
_, manifestations = cell.get_cellar_manifestations_by_celex(
|
|
169
|
+
"62020CJ0414", sector="6"
|
|
170
|
+
)
|
|
171
|
+
english = [m for m in manifestations if m["language"] == "EN"]
|
|
172
|
+
fulltexts = cell.extract_cellar_fulltexts(english)
|
|
173
|
+
```
|
|
174
|
+
|
|
156
175
|
Returns:
|
|
157
176
|
|
|
158
177
|
- `extra_df`: enriched dataframe
|
|
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
|
|
|
303
322
|
| `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
|
|
304
323
|
| `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
|
|
305
324
|
| `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
|
|
325
|
+
| `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
|
|
326
|
+
| `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
|
|
306
327
|
| `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
|
|
307
328
|
| `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
|
|
308
329
|
| `FetchOperativePart` | Extract operative part from a single case document |
|
|
@@ -92,6 +92,11 @@ df = cell.get_cellar(
|
|
|
92
92
|
|
|
93
93
|
Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
|
|
94
94
|
|
|
95
|
+
For direct manifestation access, use
|
|
96
|
+
`get_cellar_manifestations_by_celex()`. It canonicalizes composite and
|
|
97
|
+
`_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
|
|
98
|
+
derived summaries or notices from being returned as judgment full text.
|
|
99
|
+
|
|
95
100
|
You can also save explicitly to a custom path instead of the default `data/` location:
|
|
96
101
|
|
|
97
102
|
```python
|
|
@@ -118,6 +123,20 @@ extra_df, fulltext = cell.get_cellar_extra(
|
|
|
118
123
|
)
|
|
119
124
|
```
|
|
120
125
|
|
|
126
|
+
For targeted CELEX repair or supplementation, use the public CELLAR
|
|
127
|
+
manifestation API. It unions every CELLAR work sharing the CELEX before choosing
|
|
128
|
+
the best downloadable item per language:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
import cellar_extractor as cell
|
|
132
|
+
|
|
133
|
+
_, manifestations = cell.get_cellar_manifestations_by_celex(
|
|
134
|
+
"62020CJ0414", sector="6"
|
|
135
|
+
)
|
|
136
|
+
english = [m for m in manifestations if m["language"] == "EN"]
|
|
137
|
+
fulltexts = cell.extract_cellar_fulltexts(english)
|
|
138
|
+
```
|
|
139
|
+
|
|
121
140
|
Returns:
|
|
122
141
|
|
|
123
142
|
- `extra_df`: enriched dataframe
|
|
@@ -268,6 +287,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
|
|
|
268
287
|
| `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
|
|
269
288
|
| `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
|
|
270
289
|
| `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
|
|
290
|
+
| `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
|
|
291
|
+
| `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
|
|
271
292
|
| `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
|
|
272
293
|
| `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
|
|
273
294
|
| `FetchOperativePart` | Extract operative part from a single case document |
|
|
@@ -9,6 +9,9 @@ from cellar_extractor.cellar import get_cellar_extra
|
|
|
9
9
|
from cellar_extractor.cellar import get_nodes_and_edges_lists
|
|
10
10
|
from cellar_extractor.cellar import filter_subject_matter
|
|
11
11
|
from cellar_extractor.eurlex_scraping import get_legislation_by_celex_id
|
|
12
|
+
from cellar_extractor.eurlex_scraping import get_cellar_manifestations_by_celex
|
|
13
|
+
from cellar_extractor.eurlex_scraping import extract_cellar_fulltexts
|
|
14
|
+
from cellar_extractor.eurlex_scraping import normalize_celex
|
|
12
15
|
from cellar_extractor.operative_extractions import FetchOperativePart
|
|
13
16
|
from cellar_extractor.operative_extractions import Writing
|
|
14
17
|
import logging
|
|
@@ -18,7 +18,7 @@ version_tuple: tuple[int | str, ...]
|
|
|
18
18
|
commit_id: str | None
|
|
19
19
|
__commit_id__: str | None
|
|
20
20
|
|
|
21
|
-
__version__ = version = '2.0.
|
|
22
|
-
__version_tuple__ = version_tuple = (2, 0,
|
|
21
|
+
__version__ = version = '2.0.3'
|
|
22
|
+
__version_tuple__ = version_tuple = (2, 0, 3)
|
|
23
23
|
|
|
24
|
-
__commit_id__ = commit_id = '
|
|
24
|
+
__commit_id__ = commit_id = 'g1aaeae809'
|
|
@@ -129,6 +129,17 @@ def _normalize_celex(celex):
|
|
|
129
129
|
return value
|
|
130
130
|
|
|
131
131
|
|
|
132
|
+
def normalize_celex(celex):
|
|
133
|
+
"""Return the canonical work CELEX used for full-text retrieval.
|
|
134
|
+
|
|
135
|
+
CELLAR metadata can bundle a primary document with derived information,
|
|
136
|
+
summary, and résumé works (for example ``62020CJ0414_SUM;62020CJ0414``).
|
|
137
|
+
Full-text callers must resolve the unsuffixed base work so a derived work
|
|
138
|
+
can never occupy the judgment's language slot.
|
|
139
|
+
"""
|
|
140
|
+
return _normalize_celex(celex)
|
|
141
|
+
|
|
142
|
+
|
|
132
143
|
def _sleep_with_backoff(attempt, base=0.2):
|
|
133
144
|
time.sleep(base * (attempt + 1) + random.uniform(0.0, 0.1))
|
|
134
145
|
|
|
@@ -413,6 +424,25 @@ def _fetch_sector8_items_for_celex(celex, sector="8"):
|
|
|
413
424
|
return work_uris, candidates
|
|
414
425
|
|
|
415
426
|
|
|
427
|
+
def get_cellar_manifestations_by_celex(celex, sector="8"):
|
|
428
|
+
"""Return all CELLAR works and manifestation candidates for a CELEX.
|
|
429
|
+
|
|
430
|
+
A CELEX can resolve to multiple CELLAR works with different language
|
|
431
|
+
coverage. This public entry point deliberately returns the union across
|
|
432
|
+
every matching work so callers do not need to depend on the extractor's
|
|
433
|
+
private sector-8 helpers.
|
|
434
|
+
|
|
435
|
+
The return value is ``(work_uris, manifestations)``. Each manifestation
|
|
436
|
+
contains ``item_url``, ``format``, and ``language`` keys and is deduplicated
|
|
437
|
+
on that triple. ``sector`` defaults to ``"8"`` and may be set to ``"6"``
|
|
438
|
+
for CJEU documents.
|
|
439
|
+
"""
|
|
440
|
+
canonical_celex = normalize_celex(celex)
|
|
441
|
+
if not canonical_celex:
|
|
442
|
+
return [], []
|
|
443
|
+
return _fetch_sector8_items_for_celex(canonical_celex, sector=sector)
|
|
444
|
+
|
|
445
|
+
|
|
416
446
|
def _fetch_sector8_work_uri(celex, sector="8"):
|
|
417
447
|
"""Resolve a CELEX to its CELLAR work URI.
|
|
418
448
|
|
|
@@ -553,6 +583,17 @@ def _fanout_fulltexts_from_candidates(candidates, source_label):
|
|
|
553
583
|
return out
|
|
554
584
|
|
|
555
585
|
|
|
586
|
+
def extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM"):
|
|
587
|
+
"""Download the best manifestation per language as fulltext records.
|
|
588
|
+
|
|
589
|
+
``manifestations`` is the candidate list returned by
|
|
590
|
+
:func:`get_cellar_manifestations_by_celex`. Empty bodies are omitted and
|
|
591
|
+
each returned dictionary contains ``text``, ``html``, ``text_source``,
|
|
592
|
+
``text_language``, and ``text_format``.
|
|
593
|
+
"""
|
|
594
|
+
return _fanout_fulltexts_from_candidates(manifestations, source_label)
|
|
595
|
+
|
|
596
|
+
|
|
556
597
|
def _get_case_data_sector8(celex, language="EN"):
|
|
557
598
|
work_uris, main_candidates = _fetch_sector8_items_for_celex(celex)
|
|
558
599
|
|
|
@@ -1104,15 +1145,14 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1104
1145
|
}
|
|
1105
1146
|
)
|
|
1106
1147
|
|
|
1107
|
-
#
|
|
1108
|
-
#
|
|
1109
|
-
#
|
|
1110
|
-
#
|
|
1111
|
-
#
|
|
1112
|
-
#
|
|
1113
|
-
#
|
|
1114
|
-
#
|
|
1115
|
-
# populate them.
|
|
1148
|
+
# Merge InfoCuria fulltexts with every language CELLAR has. CELLAR is
|
|
1149
|
+
# authoritative when both sources expose the same language because its
|
|
1150
|
+
# manifestation belongs to the canonical CELEX work. InfoCuria's
|
|
1151
|
+
# procedure search has returned a different document under the judgment
|
|
1152
|
+
# slot in production (AG opinions, procedural orders, and notices), so an
|
|
1153
|
+
# existing InfoCuria language must not block the canonical manifestation.
|
|
1154
|
+
# Metadata fields (judge, advocate, directory_codes, etc.) remain
|
|
1155
|
+
# InfoCuria-sourced — CELLAR cannot populate them.
|
|
1116
1156
|
#
|
|
1117
1157
|
# All-in-one try/except: a CELLAR-side failure must never kill the
|
|
1118
1158
|
# InfoCuria-sourced row we already built.
|
|
@@ -1122,18 +1162,36 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1122
1162
|
cellar_fulltexts = _fanout_fulltexts_from_candidates(
|
|
1123
1163
|
cellar_candidates, source_label="CELLAR_ITEM"
|
|
1124
1164
|
)
|
|
1165
|
+
by_language = {
|
|
1166
|
+
entry.get("text_language", "").upper(): entry
|
|
1167
|
+
for entry in fulltexts
|
|
1168
|
+
if entry.get("text_language", "")
|
|
1169
|
+
}
|
|
1125
1170
|
for entry in cellar_fulltexts:
|
|
1126
1171
|
entry_lang = entry.get("text_language", "").upper()
|
|
1127
|
-
if not entry_lang
|
|
1172
|
+
if not entry_lang:
|
|
1128
1173
|
continue
|
|
1129
1174
|
seen_langs.add(entry_lang)
|
|
1130
|
-
|
|
1175
|
+
by_language[entry_lang] = entry
|
|
1176
|
+
fulltexts = list(by_language.values())
|
|
1131
1177
|
except Exception:
|
|
1132
1178
|
# CELLAR supplementation is best-effort. If anything goes wrong
|
|
1133
1179
|
# (SPARQL timeout, network hiccup, schema drift on the manifestation
|
|
1134
1180
|
# graph) the InfoCuria-only result is still returned.
|
|
1135
1181
|
pass
|
|
1136
1182
|
|
|
1183
|
+
primary = next(
|
|
1184
|
+
(
|
|
1185
|
+
entry
|
|
1186
|
+
for entry in fulltexts
|
|
1187
|
+
if entry.get("text_language", "").upper() == str(language).upper()
|
|
1188
|
+
),
|
|
1189
|
+
None,
|
|
1190
|
+
)
|
|
1191
|
+
if primary is not None:
|
|
1192
|
+
text = primary.get("text", "")
|
|
1193
|
+
html = primary.get("html", "")
|
|
1194
|
+
|
|
1137
1195
|
return {
|
|
1138
1196
|
"html": html,
|
|
1139
1197
|
"text": text,
|
|
@@ -1146,9 +1204,9 @@ def _get_case_data_sector6(celex, language="EN"):
|
|
|
1146
1204
|
"affecting_ids": affecting_ids,
|
|
1147
1205
|
"affecting_string": affecting_string,
|
|
1148
1206
|
"citations_extra": citations_extra,
|
|
1149
|
-
"text_source": "
|
|
1150
|
-
"text_language":
|
|
1151
|
-
"text_format": "
|
|
1207
|
+
"text_source": primary.get("text_source", "") if primary else "",
|
|
1208
|
+
"text_language": primary.get("text_language", "") if primary else "",
|
|
1209
|
+
"text_format": primary.get("text_format", "") if primary else "",
|
|
1152
1210
|
"summary_source": summary_source,
|
|
1153
1211
|
"summary_language": "EN" if summary != "" else "",
|
|
1154
1212
|
"sector": "6",
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: cellar-extractor
|
|
3
|
-
Version: 2.0.
|
|
3
|
+
Version: 2.0.3
|
|
4
4
|
Summary: Library for extracting CELLAR case law data from EUR-Lex
|
|
5
5
|
Author-email: LawTech Lab <law-techlab@maastrichtuniversity.nl>
|
|
6
6
|
License: Apache-2.0
|
|
@@ -127,6 +127,11 @@ df = cell.get_cellar(
|
|
|
127
127
|
|
|
128
128
|
Returns a dataframe with base metadata such as CELEX, ECLI, type, dates, and subject-matter-related fields.
|
|
129
129
|
|
|
130
|
+
For direct manifestation access, use
|
|
131
|
+
`get_cellar_manifestations_by_celex()`. It canonicalizes composite and
|
|
132
|
+
`_SUM`/`_RES`/`_INF` identifiers to the base work before querying, preventing
|
|
133
|
+
derived summaries or notices from being returned as judgment full text.
|
|
134
|
+
|
|
130
135
|
You can also save explicitly to a custom path instead of the default `data/` location:
|
|
131
136
|
|
|
132
137
|
```python
|
|
@@ -153,6 +158,20 @@ extra_df, fulltext = cell.get_cellar_extra(
|
|
|
153
158
|
)
|
|
154
159
|
```
|
|
155
160
|
|
|
161
|
+
For targeted CELEX repair or supplementation, use the public CELLAR
|
|
162
|
+
manifestation API. It unions every CELLAR work sharing the CELEX before choosing
|
|
163
|
+
the best downloadable item per language:
|
|
164
|
+
|
|
165
|
+
```python
|
|
166
|
+
import cellar_extractor as cell
|
|
167
|
+
|
|
168
|
+
_, manifestations = cell.get_cellar_manifestations_by_celex(
|
|
169
|
+
"62020CJ0414", sector="6"
|
|
170
|
+
)
|
|
171
|
+
english = [m for m in manifestations if m["language"] == "EN"]
|
|
172
|
+
fulltexts = cell.extract_cellar_fulltexts(english)
|
|
173
|
+
```
|
|
174
|
+
|
|
156
175
|
Returns:
|
|
157
176
|
|
|
158
177
|
- `extra_df`: enriched dataframe
|
|
@@ -303,6 +322,8 @@ Imported from [`cellar_extractor/__init__.py`](/Users/davidwickerhf/Projects/wor
|
|
|
303
322
|
| `get_cellar(...)` | Fetch base CELLAR metadata (case law only) |
|
|
304
323
|
| `get_cellar_extra(...)` | Fetch enriched metadata + full text (case law only) |
|
|
305
324
|
| `get_legislation_by_celex_id(celex, language="EN")` | Fetch sector 3 / sector 0 legislation XHTML by CELEX |
|
|
325
|
+
| `get_cellar_manifestations_by_celex(celex, sector="8")` | Resolve every CELLAR work for a CELEX and return their deduplicated manifestation union |
|
|
326
|
+
| `extract_cellar_fulltexts(manifestations, source_label="CELLAR_ITEM")` | Download the best manifestation per language as fulltext records |
|
|
306
327
|
| `get_nodes_and_edges_lists(df, only_local=False)` | Build citation graph lists |
|
|
307
328
|
| `filter_subject_matter(df, phrase)` | Filter dataframe by subject phrase |
|
|
308
329
|
| `FetchOperativePart` | Extract operative part from a single case document |
|
|
@@ -6,6 +6,7 @@
|
|
|
6
6
|
at the metadata assembly point, in both the celex- and ecli-keyed loops.
|
|
7
7
|
"""
|
|
8
8
|
|
|
9
|
+
import cellar_extractor as cell
|
|
9
10
|
import cellar_extractor.cellar_queries as cq
|
|
10
11
|
import cellar_extractor.eurlex_scraping as es
|
|
11
12
|
|
|
@@ -132,3 +133,36 @@ def test_items_for_celex_unions_across_all_works(monkeypatch):
|
|
|
132
133
|
assert len(uris) == 2
|
|
133
134
|
langs = sorted(c["language"] for c in cands)
|
|
134
135
|
assert langs == ["DE", "EN", "FR", "NL"] # union, DE deduped
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def test_public_manifestation_and_fulltext_api(monkeypatch):
|
|
139
|
+
manifestations = [
|
|
140
|
+
{"item_url": "u_en", "format": "xhtml", "language": "EN"},
|
|
141
|
+
{"item_url": "u_fr", "format": "xhtml", "language": "FR"},
|
|
142
|
+
]
|
|
143
|
+
monkeypatch.setattr(
|
|
144
|
+
es,
|
|
145
|
+
"_fetch_sector8_items_for_celex",
|
|
146
|
+
lambda celex, sector="8": ([f"http://cellar/{sector}/{celex}"], manifestations),
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
works, found = cell.get_cellar_manifestations_by_celex(
|
|
150
|
+
"62024CJ0001_SUM;62024CJ0001", sector="6"
|
|
151
|
+
)
|
|
152
|
+
assert works == ["http://cellar/6/62024CJ0001"]
|
|
153
|
+
assert found == manifestations
|
|
154
|
+
assert cell.normalize_celex("62024CJ0001_SUM;62024CJ0001") == "62024CJ0001"
|
|
155
|
+
|
|
156
|
+
monkeypatch.setattr(
|
|
157
|
+
es,
|
|
158
|
+
"_fanout_fulltexts_from_candidates",
|
|
159
|
+
lambda candidates, source_label: [
|
|
160
|
+
{"text_language": candidate["language"], "text_source": source_label}
|
|
161
|
+
for candidate in candidates
|
|
162
|
+
],
|
|
163
|
+
)
|
|
164
|
+
rows = cell.extract_cellar_fulltexts(found, source_label="CELLAR_ITEM")
|
|
165
|
+
assert rows == [
|
|
166
|
+
{"text_language": "EN", "text_source": "CELLAR_ITEM"},
|
|
167
|
+
{"text_language": "FR", "text_source": "CELLAR_ITEM"},
|
|
168
|
+
]
|
|
@@ -7,8 +7,8 @@ languages for the same work.
|
|
|
7
7
|
|
|
8
8
|
After this change, sector 6 always supplements InfoCuria's fulltexts with
|
|
9
9
|
CELLAR's manifestation graph (unconditionally — not env-var gated), so the
|
|
10
|
-
``fulltexts`` list contains every language CELLAR has, with
|
|
11
|
-
|
|
10
|
+
``fulltexts`` list contains every language CELLAR has, with canonical CELLAR
|
|
11
|
+
manifestations replacing InfoCuria entries where both sources overlap.
|
|
12
12
|
"""
|
|
13
13
|
|
|
14
14
|
from __future__ import annotations
|
|
@@ -154,9 +154,8 @@ def test_infocuria_success_is_supplemented_with_cellar_languages(monkeypatch):
|
|
|
154
154
|
assert langs == {"EN", "FR", "DE", "IT", "NL", "ES", "PT"}
|
|
155
155
|
|
|
156
156
|
|
|
157
|
-
def
|
|
158
|
-
"""
|
|
159
|
-
it's the court's own publication and typically has higher fidelity."""
|
|
157
|
+
def test_cellar_entries_replace_infocuria_when_languages_overlap(monkeypatch):
|
|
158
|
+
"""Canonical CELLAR works win overlaps to prevent wrong-doc captures."""
|
|
160
159
|
eurlex_scraping._get_case_data_cached.cache_clear()
|
|
161
160
|
_patch_infocuria(monkeypatch, doc_langs=["EN", "FR"])
|
|
162
161
|
_patch_cellar(monkeypatch, languages=["EN", "FR", "DE"])
|
|
@@ -164,14 +163,11 @@ def test_infocuria_entries_are_preserved_when_languages_overlap(monkeypatch):
|
|
|
164
163
|
data = eurlex_scraping._get_case_data_sector6("62024CJ0001", language="EN")
|
|
165
164
|
|
|
166
165
|
by_lang = {entry["text_language"]: entry for entry in data["fulltexts"]}
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
assert
|
|
171
|
-
assert "
|
|
172
|
-
# DE only existed in CELLAR — comes through with CELLAR_ITEM source.
|
|
173
|
-
assert by_lang["DE"]["text_source"] == "CELLAR_ITEM"
|
|
174
|
-
assert "cellar body" in by_lang["DE"]["text"]
|
|
166
|
+
for language in ("EN", "FR", "DE"):
|
|
167
|
+
assert by_lang[language]["text_source"] == "CELLAR_ITEM"
|
|
168
|
+
assert "cellar body" in by_lang[language]["text"]
|
|
169
|
+
assert data["text_source"] == "CELLAR_ITEM"
|
|
170
|
+
assert "cellar body" in data["text"]
|
|
175
171
|
|
|
176
172
|
|
|
177
173
|
def test_metadata_fields_remain_from_infocuria_even_when_cellar_supplements(monkeypatch):
|
|
@@ -250,9 +246,8 @@ def test_metadata_fields_remain_from_infocuria_even_when_cellar_supplements(monk
|
|
|
250
246
|
assert "Environment" in data["keywords"]
|
|
251
247
|
|
|
252
248
|
|
|
253
|
-
def
|
|
254
|
-
"""
|
|
255
|
-
and the metadata is untouched. Belt-and-braces idempotency."""
|
|
249
|
+
def test_cellar_replaces_all_overlapping_languages(monkeypatch):
|
|
250
|
+
"""Even exact language overlap is replaced by the canonical work."""
|
|
256
251
|
eurlex_scraping._get_case_data_cached.cache_clear()
|
|
257
252
|
_patch_infocuria(monkeypatch, doc_langs=["EN", "FR", "DE"])
|
|
258
253
|
_patch_cellar(monkeypatch, languages=["EN", "FR", "DE"]) # exact overlap
|
|
@@ -261,9 +256,8 @@ def test_supplementation_is_noop_when_cellar_has_nothing_extra(monkeypatch):
|
|
|
261
256
|
|
|
262
257
|
langs = sorted(entry["text_language"] for entry in data["fulltexts"])
|
|
263
258
|
assert langs == ["DE", "EN", "FR"]
|
|
264
|
-
# No CELLAR_ITEM entries — every language already had an InfoCuria source.
|
|
265
259
|
sources = {entry["text_source"] for entry in data["fulltexts"]}
|
|
266
|
-
assert sources == {"
|
|
260
|
+
assert sources == {"CELLAR_ITEM"}
|
|
267
261
|
|
|
268
262
|
|
|
269
263
|
def test_supplementation_skips_when_cellar_work_uri_unresolvable(monkeypatch):
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/cellar_extractor.egg-info/scm_file_list.json
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_credentials_integration.py
RENAMED
|
File without changes
|
{cellar_extractor-2.0.2 → cellar_extractor-2.0.3}/tests/test_webservice_redundancy_integration.py
RENAMED
|
File without changes
|