echr-extractor 0.0.1.dev1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- echr_extractor/ECHR_html_downloader.py +82 -0
- echr_extractor/ECHR_metadata_harvester.py +575 -0
- echr_extractor/ECHR_nodes_edges_list_transform.py +308 -0
- echr_extractor/__init__.py +18 -0
- echr_extractor/_version.py +34 -0
- echr_extractor/clean_ref.py +5 -0
- echr_extractor/cli.py +125 -0
- echr_extractor/echr.py +266 -0
- echr_extractor/testing_file.py +20 -0
- echr_extractor-0.0.1.dev1.dist-info/METADATA +189 -0
- echr_extractor-0.0.1.dev1.dist-info/RECORD +15 -0
- echr_extractor-0.0.1.dev1.dist-info/WHEEL +5 -0
- echr_extractor-0.0.1.dev1.dist-info/entry_points.txt +2 -0
- echr_extractor-0.0.1.dev1.dist-info/licenses/LICENSE +201 -0
- echr_extractor-0.0.1.dev1.dist-info/top_level.txt +1 -0
echr_extractor/echr.py
ADDED
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import logging
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
|
|
6
|
+
from .ECHR_html_downloader import download_full_text_main
|
|
7
|
+
from .ECHR_metadata_harvester import get_echr_metadata
|
|
8
|
+
from .ECHR_nodes_edges_list_transform import echr_nodes_edges
|
|
9
|
+
|
|
10
|
+
"""
|
|
11
|
+
Enhanced ECHR data extraction with improved batching, error handling, and memory management.
|
|
12
|
+
|
|
13
|
+
Key improvements:
|
|
14
|
+
- Date range batching for large datasets to prevent timeouts
|
|
15
|
+
- Enhanced error handling with exponential backoff
|
|
16
|
+
- Progress tracking with tqdm progress bars
|
|
17
|
+
- Memory-efficient processing for large datasets
|
|
18
|
+
- Configurable batch sizes and retry parameters
|
|
19
|
+
- Better logging and status reporting
|
|
20
|
+
|
|
21
|
+
The original functionality is preserved for backward compatibility.
|
|
22
|
+
"""
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def get_echr(
|
|
26
|
+
start_id=0,
|
|
27
|
+
end_id=None,
|
|
28
|
+
start_date=None,
|
|
29
|
+
count=None,
|
|
30
|
+
end_date=None,
|
|
31
|
+
verbose=False,
|
|
32
|
+
save_file="y",
|
|
33
|
+
fields=None,
|
|
34
|
+
link=None,
|
|
35
|
+
language=None,
|
|
36
|
+
query_payload=None,
|
|
37
|
+
# New configuration parameters
|
|
38
|
+
batch_size=500,
|
|
39
|
+
timeout=60,
|
|
40
|
+
retry_attempts=3,
|
|
41
|
+
max_attempts=20,
|
|
42
|
+
days_per_batch=365,
|
|
43
|
+
progress_bar=True,
|
|
44
|
+
memory_efficient=True,
|
|
45
|
+
):
|
|
46
|
+
"""
|
|
47
|
+
Enhanced ECHR metadata extraction with improved reliability and performance.
|
|
48
|
+
|
|
49
|
+
This function provides a high-level interface for extracting ECHR metadata with
|
|
50
|
+
advanced features like date batching, progress tracking, and memory management.
|
|
51
|
+
|
|
52
|
+
:param int start_id: The index to start the search from (default: 0).
|
|
53
|
+
:param int end_id: The index to end search at, where None fetches all results.
|
|
54
|
+
:param str start_date: The point from which to save cases (YYYY-MM-DD format).
|
|
55
|
+
:param int count: Number of records to fetch (alternative to end_id).
|
|
56
|
+
:param str end_date: The point before which to save cases (YYYY-MM-DD format).
|
|
57
|
+
:param bool verbose: Whether or not to print extra information (default: False).
|
|
58
|
+
:param str save_file: Whether to save results to file ("y" or "n", default: "y").
|
|
59
|
+
:param list fields: List of fields to extract (default: None, uses all fields).
|
|
60
|
+
:param str link: Custom HUDOC link for advanced queries.
|
|
61
|
+
:param list language: List of language codes (default: ["ENG"]).
|
|
62
|
+
:param str query_payload: Custom query payload for advanced searches.
|
|
63
|
+
:param int batch_size: Number of records to fetch per batch, max 500 (default: 500).
|
|
64
|
+
:param float timeout: Request timeout in seconds (default: 60).
|
|
65
|
+
:param int retry_attempts: Number of retry attempts for failed requests (default: 3).
|
|
66
|
+
:param int max_attempts: Maximum total attempts before giving up (default: 20).
|
|
67
|
+
:param int days_per_batch: Number of days per date batch for large date ranges (default: 365).
|
|
68
|
+
:param bool progress_bar: Whether to show progress bar (default: True).
|
|
69
|
+
:param bool memory_efficient: Whether to use memory-efficient processing (default: True).
|
|
70
|
+
|
|
71
|
+
:return: pandas.DataFrame containing the extracted metadata, or False if extraction failed.
|
|
72
|
+
|
|
73
|
+
Example:
|
|
74
|
+
# Basic usage (backward compatible)
|
|
75
|
+
df = get_echr(start_id=0, end_id=1000, verbose=True)
|
|
76
|
+
|
|
77
|
+
# Advanced usage with date batching
|
|
78
|
+
df = get_echr(
|
|
79
|
+
start_date='2020-01-01',
|
|
80
|
+
end_date='2023-12-31',
|
|
81
|
+
batch_size=250,
|
|
82
|
+
days_per_batch=180,
|
|
83
|
+
progress_bar=True
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
# Memory-efficient processing for large datasets
|
|
87
|
+
df = get_echr(
|
|
88
|
+
start_id=0,
|
|
89
|
+
end_id=50000,
|
|
90
|
+
memory_efficient=True,
|
|
91
|
+
batch_size=200
|
|
92
|
+
)
|
|
93
|
+
"""
|
|
94
|
+
if language is None:
|
|
95
|
+
language = ["ENG"]
|
|
96
|
+
if count:
|
|
97
|
+
end_id = int(start_id) + count
|
|
98
|
+
if verbose:
|
|
99
|
+
logging.info(f"--- STARTING ECHR DOWNLOAD FOR {count} RECORDS ---")
|
|
100
|
+
else:
|
|
101
|
+
if verbose:
|
|
102
|
+
logging.info("--- STARTING ECHR DOWNLOAD ---")
|
|
103
|
+
|
|
104
|
+
df = get_echr_metadata(
|
|
105
|
+
start_id=start_id,
|
|
106
|
+
end_id=end_id,
|
|
107
|
+
start_date=start_date,
|
|
108
|
+
end_date=end_date,
|
|
109
|
+
verbose=verbose,
|
|
110
|
+
fields=fields,
|
|
111
|
+
link=link,
|
|
112
|
+
language=language,
|
|
113
|
+
query_payload=query_payload,
|
|
114
|
+
batch_size=batch_size,
|
|
115
|
+
timeout=timeout,
|
|
116
|
+
retry_attempts=retry_attempts,
|
|
117
|
+
max_attempts=max_attempts,
|
|
118
|
+
days_per_batch=days_per_batch,
|
|
119
|
+
progress_bar=progress_bar,
|
|
120
|
+
memory_efficient=memory_efficient,
|
|
121
|
+
)
|
|
122
|
+
if df is False:
|
|
123
|
+
return False
|
|
124
|
+
if save_file == "y":
|
|
125
|
+
filename = determine_filename(start_id, end_id, start_date, end_date)
|
|
126
|
+
Path("data").mkdir(parents=True, exist_ok=True)
|
|
127
|
+
file_path = os.path.join("data", filename + ".csv")
|
|
128
|
+
df.to_csv(file_path, index=False)
|
|
129
|
+
logging.info("\n--- DONE ---")
|
|
130
|
+
return df
|
|
131
|
+
else:
|
|
132
|
+
logging.info("\n--- DONE ---")
|
|
133
|
+
return df
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
def determine_filename(start_id, end_id, start_date, end_date):
|
|
137
|
+
if end_id:
|
|
138
|
+
if start_date and end_date:
|
|
139
|
+
filename = (
|
|
140
|
+
f"echr_metadata_index_{start_id}-{end_id}_dates_{start_date}-{end_date}"
|
|
141
|
+
)
|
|
142
|
+
elif start_date:
|
|
143
|
+
filename = f"echr_metadata_{start_id}-{end_id}_dates_{start_date}-END"
|
|
144
|
+
elif end_date:
|
|
145
|
+
filename = f"echr_metadata_{start_id}-{end_id}_datesSTART-{end_date}"
|
|
146
|
+
else:
|
|
147
|
+
filename = f"echr_metadata_{start_id}-{end_id}_dates_START-END"
|
|
148
|
+
else:
|
|
149
|
+
if start_date and end_date:
|
|
150
|
+
filename = (
|
|
151
|
+
f"echr_metadata_index_{start_id}-ALL_dates_{start_date}-{end_date}"
|
|
152
|
+
)
|
|
153
|
+
elif start_date:
|
|
154
|
+
filename = f"echr_metadata_{start_id}-ALL_dates_{start_date}-END"
|
|
155
|
+
elif end_date:
|
|
156
|
+
filename = f"echr_metadata_{start_id}-ALL_dates_START-{end_date}"
|
|
157
|
+
else:
|
|
158
|
+
filename = f"echr_metadata_{start_id}-ALL_dates_START-END"
|
|
159
|
+
return filename
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def get_echr_extra(
|
|
163
|
+
start_id=0,
|
|
164
|
+
end_id=None,
|
|
165
|
+
start_date=None,
|
|
166
|
+
count=None,
|
|
167
|
+
end_date=None,
|
|
168
|
+
verbose=False,
|
|
169
|
+
save_file="y",
|
|
170
|
+
threads=10,
|
|
171
|
+
fields=None,
|
|
172
|
+
link=None,
|
|
173
|
+
language=None,
|
|
174
|
+
query_payload=None,
|
|
175
|
+
# New configuration parameters
|
|
176
|
+
batch_size=500,
|
|
177
|
+
timeout=60,
|
|
178
|
+
retry_attempts=3,
|
|
179
|
+
max_attempts=20,
|
|
180
|
+
days_per_batch=365,
|
|
181
|
+
progress_bar=True,
|
|
182
|
+
memory_efficient=True,
|
|
183
|
+
):
|
|
184
|
+
"""
|
|
185
|
+
Enhanced ECHR metadata and full-text extraction with improved reliability and performance.
|
|
186
|
+
|
|
187
|
+
This function extracts both metadata and full-text content from ECHR cases with
|
|
188
|
+
advanced features like date batching, progress tracking, and memory management.
|
|
189
|
+
|
|
190
|
+
:param int start_id: The index to start the search from (default: 0).
|
|
191
|
+
:param int end_id: The index to end search at, where None fetches all results.
|
|
192
|
+
:param str start_date: The point from which to save cases (YYYY-MM-DD format).
|
|
193
|
+
:param int count: Number of records to fetch (alternative to end_id).
|
|
194
|
+
:param str end_date: The point before which to save cases (YYYY-MM-DD format).
|
|
195
|
+
:param bool verbose: Whether or not to print extra information (default: False).
|
|
196
|
+
:param str save_file: Whether to save results to file ("y" or "n", default: "y").
|
|
197
|
+
:param int threads: Number of threads for full-text download (default: 10).
|
|
198
|
+
:param list fields: List of fields to extract (default: None, uses all fields).
|
|
199
|
+
:param str link: Custom HUDOC link for advanced queries.
|
|
200
|
+
:param list language: List of language codes (default: ["ENG"]).
|
|
201
|
+
:param str query_payload: Custom query payload for advanced searches.
|
|
202
|
+
:param int batch_size: Number of records to fetch per batch, max 500 (default: 500).
|
|
203
|
+
:param float timeout: Request timeout in seconds (default: 60).
|
|
204
|
+
:param int retry_attempts: Number of retry attempts for failed requests (default: 3).
|
|
205
|
+
:param int max_attempts: Maximum total attempts before giving up (default: 20).
|
|
206
|
+
:param int days_per_batch: Number of days per date batch for large date ranges (default: 365).
|
|
207
|
+
:param bool progress_bar: Whether to show progress bar (default: True).
|
|
208
|
+
:param bool memory_efficient: Whether to use memory-efficient processing (default: True).
|
|
209
|
+
|
|
210
|
+
:return: tuple of (pandas.DataFrame, list) containing metadata and full-text data,
|
|
211
|
+
or (False, False) if extraction failed.
|
|
212
|
+
"""
|
|
213
|
+
df = get_echr(
|
|
214
|
+
start_id=start_id,
|
|
215
|
+
end_id=end_id,
|
|
216
|
+
start_date=start_date,
|
|
217
|
+
end_date=end_date,
|
|
218
|
+
verbose=verbose,
|
|
219
|
+
count=count,
|
|
220
|
+
save_file="n",
|
|
221
|
+
fields=fields,
|
|
222
|
+
link=link,
|
|
223
|
+
language=language,
|
|
224
|
+
query_payload=query_payload,
|
|
225
|
+
batch_size=batch_size,
|
|
226
|
+
timeout=timeout,
|
|
227
|
+
retry_attempts=retry_attempts,
|
|
228
|
+
max_attempts=max_attempts,
|
|
229
|
+
days_per_batch=days_per_batch,
|
|
230
|
+
progress_bar=progress_bar,
|
|
231
|
+
memory_efficient=memory_efficient,
|
|
232
|
+
)
|
|
233
|
+
logging.info("Full-text download will now begin")
|
|
234
|
+
if df is False:
|
|
235
|
+
return False, False
|
|
236
|
+
json_list = download_full_text_main(df, threads)
|
|
237
|
+
logging.info("Full-text download finished")
|
|
238
|
+
if save_file == "y":
|
|
239
|
+
filename = determine_filename(start_id, end_id, start_date, end_date)
|
|
240
|
+
filename_json = filename.replace("metadata", "full_text")
|
|
241
|
+
Path("data").mkdir(parents=True, exist_ok=True)
|
|
242
|
+
file_path = os.path.join("data", filename + ".csv")
|
|
243
|
+
df.to_csv(file_path, index=False)
|
|
244
|
+
file_path_json = os.path.join("data", filename_json + ".json")
|
|
245
|
+
with open(file_path_json, "w") as f:
|
|
246
|
+
json.dump(json_list, f)
|
|
247
|
+
return df, json_list
|
|
248
|
+
else:
|
|
249
|
+
return df, json_list
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def get_nodes_edges(metadata_path=None, df=None, save_file="y"):
|
|
253
|
+
nodes, edges = echr_nodes_edges(metadata_path=metadata_path, data=df)
|
|
254
|
+
if save_file == "y":
|
|
255
|
+
Path("data").mkdir(parents=True, exist_ok=True)
|
|
256
|
+
edges.to_csv(
|
|
257
|
+
os.path.join("data", "ECHR_edges.csv"), index=False, encoding="utf-8"
|
|
258
|
+
)
|
|
259
|
+
nodes.to_csv(
|
|
260
|
+
os.path.join("data", "ECHR_nodes.csv"), index=False, encoding="utf-8"
|
|
261
|
+
)
|
|
262
|
+
nodes.to_json(os.path.join("data", "ECHR_nodes.json"), orient="records")
|
|
263
|
+
edges.to_json(os.path.join("data", "ECHR_edges.json"), orient="records")
|
|
264
|
+
return nodes, edges
|
|
265
|
+
|
|
266
|
+
return nodes, edges
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
# import logging
|
|
2
|
+
import sys
|
|
3
|
+
from os.path import abspath
|
|
4
|
+
|
|
5
|
+
import echr_extractor
|
|
6
|
+
|
|
7
|
+
current_dir = abspath(__file__)
|
|
8
|
+
correct_dir = "\\".join(current_dir.replace("\\", "/").split("/")[:-2])
|
|
9
|
+
sys.path.append(correct_dir)
|
|
10
|
+
# print(sys.path)
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
if __name__ == "__main__":
|
|
14
|
+
payload = (
|
|
15
|
+
"contentsitename:ECHR AND (NOT (doctype=PR OR doctype=HFCOMOLD OR doctype=HECOMOLD)) "
|
|
16
|
+
'AND ((NOT "has been a violation of Article 6") AND ("has been no violation of Article 6")) '
|
|
17
|
+
'AND ((languageisocode="ENG")) AND ((documentcollectionid="GRANDCHAMBER") OR (documentcollectionid="CHAMBER"))'
|
|
18
|
+
)
|
|
19
|
+
df = echr_extractor.get_echr(query_payload=payload)
|
|
20
|
+
b = 2
|
|
@@ -0,0 +1,189 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: echr-extractor
|
|
3
|
+
Version: 0.0.1.dev1
|
|
4
|
+
Summary: Python library for extracting case law data from the European Court of Human Rights (ECHR) HUDOC database
|
|
5
|
+
Author-email: LawTech Lab <lawtech@maastrichtuniversity.nl>
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Project-URL: Homepage, https://github.com/maastrichtlawtech/echr-extractor
|
|
8
|
+
Project-URL: Repository, https://github.com/maastrichtlawtech/echr-extractor
|
|
9
|
+
Project-URL: Bug Reports, https://github.com/maastrichtlawtech/echr-extractor/issues
|
|
10
|
+
Project-URL: Documentation, https://github.com/maastrichtlawtech/echr-extractor
|
|
11
|
+
Keywords: echr,extractor,european,convention,human,rights,court,case-law,legal,hudoc,data-extraction
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Intended Audience :: Legal Industry
|
|
15
|
+
Classifier: Intended Audience :: Science/Research
|
|
16
|
+
Classifier: License :: OSI Approved :: Apache Software License
|
|
17
|
+
Classifier: Operating System :: OS Independent
|
|
18
|
+
Classifier: Programming Language :: Python :: 3
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
23
|
+
Classifier: Topic :: Scientific/Engineering :: Information Analysis
|
|
24
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
25
|
+
Classifier: Topic :: Text Processing :: Markup :: HTML
|
|
26
|
+
Requires-Python: >=3.8
|
|
27
|
+
Description-Content-Type: text/markdown
|
|
28
|
+
License-File: LICENSE
|
|
29
|
+
Requires-Dist: requests>=2.26.0
|
|
30
|
+
Requires-Dist: pandas>=1.3.0
|
|
31
|
+
Requires-Dist: beautifulsoup4>=4.9.3
|
|
32
|
+
Requires-Dist: dateparser>=1.0.0
|
|
33
|
+
Requires-Dist: tqdm>=4.60.0
|
|
34
|
+
Provides-Extra: dev
|
|
35
|
+
Requires-Dist: pytest>=6.0; extra == "dev"
|
|
36
|
+
Requires-Dist: pytest-cov>=2.10; extra == "dev"
|
|
37
|
+
Requires-Dist: black>=21.0.0; extra == "dev"
|
|
38
|
+
Requires-Dist: isort>=5.0.0; extra == "dev"
|
|
39
|
+
Requires-Dist: flake8>=3.8.0; extra == "dev"
|
|
40
|
+
Requires-Dist: mypy>=0.910; extra == "dev"
|
|
41
|
+
Provides-Extra: docs
|
|
42
|
+
Requires-Dist: sphinx>=4.0.0; extra == "docs"
|
|
43
|
+
Requires-Dist: sphinx-rtd-theme>=0.5.0; extra == "docs"
|
|
44
|
+
Dynamic: license-file
|
|
45
|
+
|
|
46
|
+
# ECHR Extractor
|
|
47
|
+
|
|
48
|
+
Python library for extracting case law data from the European Court of Human Rights (ECHR) HUDOC database.
|
|
49
|
+
|
|
50
|
+
## Features
|
|
51
|
+
|
|
52
|
+
- Extract metadata for ECHR cases from the HUDOC database
|
|
53
|
+
- Download full text content for cases
|
|
54
|
+
- Support for custom date ranges and case ID ranges
|
|
55
|
+
- Multiple language support
|
|
56
|
+
- Generate nodes and edges for network analysis
|
|
57
|
+
- Flexible output formats (CSV, JSON, in-memory DataFrames)
|
|
58
|
+
|
|
59
|
+
## Installation
|
|
60
|
+
|
|
61
|
+
```bash
|
|
62
|
+
pip install echr-extractor
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
## Quick Start
|
|
66
|
+
|
|
67
|
+
```python
|
|
68
|
+
from echr_extractor import get_echr, get_echr_extra, get_nodes_edges
|
|
69
|
+
|
|
70
|
+
# Get basic metadata for cases
|
|
71
|
+
df = get_echr(start_id=0, count=100, language=['ENG'])
|
|
72
|
+
|
|
73
|
+
# Get metadata + full text
|
|
74
|
+
df, full_texts = get_echr_extra(start_id=0, count=100, language=['ENG'])
|
|
75
|
+
|
|
76
|
+
# Generate network data
|
|
77
|
+
nodes, edges = get_nodes_edges(df=df)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Functions
|
|
81
|
+
|
|
82
|
+
### `get_echr`
|
|
83
|
+
|
|
84
|
+
Gets all available metadata for ECHR cases from the HUDOC database.
|
|
85
|
+
|
|
86
|
+
**Parameters:**
|
|
87
|
+
- `start_id` (int, optional): The ID of the first case to download (default: 0)
|
|
88
|
+
- `end_id` (int, optional): The ID of the last case to download (default: maximum available)
|
|
89
|
+
- `count` (int, optional): Number of cases per language to download (default: None)
|
|
90
|
+
- `start_date` (str, optional): Start publication date (yyyy-mm-dd) (default: None)
|
|
91
|
+
- `end_date` (str, optional): End publication date (yyyy-mm-dd) (default: current date)
|
|
92
|
+
- `verbose` (bool, optional): Show progress information (default: False)
|
|
93
|
+
- `fields` (list, optional): Limit metadata fields to download (default: all fields)
|
|
94
|
+
- `save_file` (str, optional): Save as CSV file ('y') or return DataFrame ('n') (default: 'y')
|
|
95
|
+
- `language` (list, optional): Languages to download (default: ['ENG'])
|
|
96
|
+
- `link` (str, optional): Direct HUDOC search URL (default: None)
|
|
97
|
+
- `query_payload` (str, optional): Direct API query payload (default: None)
|
|
98
|
+
|
|
99
|
+
### `get_echr_extra`
|
|
100
|
+
|
|
101
|
+
Gets metadata and downloads full text for each case.
|
|
102
|
+
|
|
103
|
+
**Parameters:** Same as `get_echr` plus:
|
|
104
|
+
- `threads` (int, optional): Number of threads for parallel download (default: 10)
|
|
105
|
+
|
|
106
|
+
### `get_nodes_edges`
|
|
107
|
+
|
|
108
|
+
Generates nodes and edges for network analysis from case metadata.
|
|
109
|
+
|
|
110
|
+
**Parameters:**
|
|
111
|
+
- `metadata_path` (str, optional): Path to metadata CSV file (default: None)
|
|
112
|
+
- `df` (DataFrame, optional): Metadata DataFrame (default: None)
|
|
113
|
+
- `save_file` (str, optional): Save as files ('y') or return objects ('n') (default: 'y')
|
|
114
|
+
|
|
115
|
+
## Advanced Usage
|
|
116
|
+
|
|
117
|
+
### Using Custom Search URLs
|
|
118
|
+
|
|
119
|
+
You can use direct HUDOC search URLs:
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
url = "https://hudoc.echr.coe.int/eng#{%22itemid%22:[%22001-57574%22]}"
|
|
123
|
+
df = get_echr(link=url)
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
### Using Query Payloads
|
|
127
|
+
|
|
128
|
+
For more robust searching, use simple field:value queries:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
payload = 'article:8'
|
|
132
|
+
df = get_echr(query_payload=payload)
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
### Date Range Filtering
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
df = get_echr(
|
|
139
|
+
start_date="2020-01-01",
|
|
140
|
+
end_date="2023-12-31",
|
|
141
|
+
language=['ENG', 'FRE']
|
|
142
|
+
)
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
### Specific Fields Only
|
|
146
|
+
|
|
147
|
+
```python
|
|
148
|
+
fields = ['itemid', 'doctypebranch', 'title', 'kpdate']
|
|
149
|
+
df = get_echr(count=100, fields=fields)
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
## Requirements
|
|
153
|
+
|
|
154
|
+
- Python 3.8+
|
|
155
|
+
- requests
|
|
156
|
+
- pandas
|
|
157
|
+
- beautifulsoup4
|
|
158
|
+
- dateparser
|
|
159
|
+
- tqdm
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
|
|
164
|
+
|
|
165
|
+
## Contributors
|
|
166
|
+
|
|
167
|
+
- Benjamin Rodrigues de Miranda
|
|
168
|
+
- Chloe Crombach
|
|
169
|
+
- Piotr Lewandowski
|
|
170
|
+
- Pranav Bapat
|
|
171
|
+
- Shashank MC
|
|
172
|
+
- Gijs van Dijck
|
|
173
|
+
|
|
174
|
+
## Citation
|
|
175
|
+
|
|
176
|
+
If you use this library in your research, please cite:
|
|
177
|
+
|
|
178
|
+
```bibtex
|
|
179
|
+
@software{echr_extractor,
|
|
180
|
+
title={ECHR Extractor: Python Library for European Court of Human Rights Data},
|
|
181
|
+
author={LawTech Lab, Maastricht University},
|
|
182
|
+
url={https://github.com/maastrichtlawtech/echr-extractor},
|
|
183
|
+
year={2024}
|
|
184
|
+
}
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
## Support
|
|
188
|
+
|
|
189
|
+
For bug reports and feature requests, please open an issue on GitHub.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
echr_extractor/ECHR_html_downloader.py,sha256=dj2C0XZMMiV4PuFETgL6L4vY7acW0Wm6qDUScd2tybE,2689
|
|
2
|
+
echr_extractor/ECHR_metadata_harvester.py,sha256=cXXejidQ0cRFQB6ryjWnzcYKqH4U9uS6K6nM7xyriuk,20106
|
|
3
|
+
echr_extractor/ECHR_nodes_edges_list_transform.py,sha256=L2VjFQA1qXTqLgD9ug-9HCcqvjBlhjhJCTjOATgeTLM,9248
|
|
4
|
+
echr_extractor/__init__.py,sha256=Ni6iuNF27KAfpLv6e7H9N66mL4nN0jEuW6soTtrAhkM,403
|
|
5
|
+
echr_extractor/_version.py,sha256=QeQ7AWx2KoSWKCEjOGq3ydCSILGUGbDRIZahOZvyHSQ,717
|
|
6
|
+
echr_extractor/clean_ref.py,sha256=GF6jx3tTALa_ztl2Y2gNRU2oztgZIq8XHnV8D6vj-d8,146
|
|
7
|
+
echr_extractor/cli.py,sha256=WTjbjPanDa2zz76UmgVI4db_xMFJ4VIsiLZHXfaGkAw,3993
|
|
8
|
+
echr_extractor/echr.py,sha256=j4E53I7u4Z7Cwx6HURPNxpjb24Ten3W-NFUjb5Tb35g,10136
|
|
9
|
+
echr_extractor/testing_file.py,sha256=K6P8uTzeosaihXe-6bd3srxxqbD1J9UjuU-5DXbwooI,665
|
|
10
|
+
echr_extractor-0.0.1.dev1.dist-info/licenses/LICENSE,sha256=QwcOLU5TJoTeUhuIXzhdCEEDDvorGiC6-3YTOl4TecE,11356
|
|
11
|
+
echr_extractor-0.0.1.dev1.dist-info/METADATA,sha256=_JDtKtCIFEMa1oL8GxzqQN3-oZhIvLAcdXwvRU-_HAg,5842
|
|
12
|
+
echr_extractor-0.0.1.dev1.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
|
|
13
|
+
echr_extractor-0.0.1.dev1.dist-info/entry_points.txt,sha256=TmNI9MFujDX0hsVO9kCCUX4fy6HfLSHZQjAQXlJNtrw,59
|
|
14
|
+
echr_extractor-0.0.1.dev1.dist-info/top_level.txt,sha256=_3q8klfvemMhS7KOxqjrEIJROrwhIM0bWKDAvldqfQY,15
|
|
15
|
+
echr_extractor-0.0.1.dev1.dist-info/RECORD,,
|