echr-extractor 0.0.1.dev1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
echr_extractor/echr.py ADDED
@@ -0,0 +1,266 @@
1
+ import json
2
+ import logging
3
+ import os
4
+ from pathlib import Path
5
+
6
+ from .ECHR_html_downloader import download_full_text_main
7
+ from .ECHR_metadata_harvester import get_echr_metadata
8
+ from .ECHR_nodes_edges_list_transform import echr_nodes_edges
9
+
10
+ """
11
+ Enhanced ECHR data extraction with improved batching, error handling, and memory management.
12
+
13
+ Key improvements:
14
+ - Date range batching for large datasets to prevent timeouts
15
+ - Enhanced error handling with exponential backoff
16
+ - Progress tracking with tqdm progress bars
17
+ - Memory-efficient processing for large datasets
18
+ - Configurable batch sizes and retry parameters
19
+ - Better logging and status reporting
20
+
21
+ The original functionality is preserved for backward compatibility.
22
+ """
23
+
24
+
25
+ def get_echr(
26
+ start_id=0,
27
+ end_id=None,
28
+ start_date=None,
29
+ count=None,
30
+ end_date=None,
31
+ verbose=False,
32
+ save_file="y",
33
+ fields=None,
34
+ link=None,
35
+ language=None,
36
+ query_payload=None,
37
+ # New configuration parameters
38
+ batch_size=500,
39
+ timeout=60,
40
+ retry_attempts=3,
41
+ max_attempts=20,
42
+ days_per_batch=365,
43
+ progress_bar=True,
44
+ memory_efficient=True,
45
+ ):
46
+ """
47
+ Enhanced ECHR metadata extraction with improved reliability and performance.
48
+
49
+ This function provides a high-level interface for extracting ECHR metadata with
50
+ advanced features like date batching, progress tracking, and memory management.
51
+
52
+ :param int start_id: The index to start the search from (default: 0).
53
+ :param int end_id: The index to end search at, where None fetches all results.
54
+ :param str start_date: The point from which to save cases (YYYY-MM-DD format).
55
+ :param int count: Number of records to fetch (alternative to end_id).
56
+ :param str end_date: The point before which to save cases (YYYY-MM-DD format).
57
+ :param bool verbose: Whether or not to print extra information (default: False).
58
+ :param str save_file: Whether to save results to file ("y" or "n", default: "y").
59
+ :param list fields: List of fields to extract (default: None, uses all fields).
60
+ :param str link: Custom HUDOC link for advanced queries.
61
+ :param list language: List of language codes (default: ["ENG"]).
62
+ :param str query_payload: Custom query payload for advanced searches.
63
+ :param int batch_size: Number of records to fetch per batch, max 500 (default: 500).
64
+ :param float timeout: Request timeout in seconds (default: 60).
65
+ :param int retry_attempts: Number of retry attempts for failed requests (default: 3).
66
+ :param int max_attempts: Maximum total attempts before giving up (default: 20).
67
+ :param int days_per_batch: Number of days per date batch for large date ranges (default: 365).
68
+ :param bool progress_bar: Whether to show progress bar (default: True).
69
+ :param bool memory_efficient: Whether to use memory-efficient processing (default: True).
70
+
71
+ :return: pandas.DataFrame containing the extracted metadata, or False if extraction failed.
72
+
73
+ Example:
74
+ # Basic usage (backward compatible)
75
+ df = get_echr(start_id=0, end_id=1000, verbose=True)
76
+
77
+ # Advanced usage with date batching
78
+ df = get_echr(
79
+ start_date='2020-01-01',
80
+ end_date='2023-12-31',
81
+ batch_size=250,
82
+ days_per_batch=180,
83
+ progress_bar=True
84
+ )
85
+
86
+ # Memory-efficient processing for large datasets
87
+ df = get_echr(
88
+ start_id=0,
89
+ end_id=50000,
90
+ memory_efficient=True,
91
+ batch_size=200
92
+ )
93
+ """
94
+ if language is None:
95
+ language = ["ENG"]
96
+ if count:
97
+ end_id = int(start_id) + count
98
+ if verbose:
99
+ logging.info(f"--- STARTING ECHR DOWNLOAD FOR {count} RECORDS ---")
100
+ else:
101
+ if verbose:
102
+ logging.info("--- STARTING ECHR DOWNLOAD ---")
103
+
104
+ df = get_echr_metadata(
105
+ start_id=start_id,
106
+ end_id=end_id,
107
+ start_date=start_date,
108
+ end_date=end_date,
109
+ verbose=verbose,
110
+ fields=fields,
111
+ link=link,
112
+ language=language,
113
+ query_payload=query_payload,
114
+ batch_size=batch_size,
115
+ timeout=timeout,
116
+ retry_attempts=retry_attempts,
117
+ max_attempts=max_attempts,
118
+ days_per_batch=days_per_batch,
119
+ progress_bar=progress_bar,
120
+ memory_efficient=memory_efficient,
121
+ )
122
+ if df is False:
123
+ return False
124
+ if save_file == "y":
125
+ filename = determine_filename(start_id, end_id, start_date, end_date)
126
+ Path("data").mkdir(parents=True, exist_ok=True)
127
+ file_path = os.path.join("data", filename + ".csv")
128
+ df.to_csv(file_path, index=False)
129
+ logging.info("\n--- DONE ---")
130
+ return df
131
+ else:
132
+ logging.info("\n--- DONE ---")
133
+ return df
134
+
135
+
136
+ def determine_filename(start_id, end_id, start_date, end_date):
137
+ if end_id:
138
+ if start_date and end_date:
139
+ filename = (
140
+ f"echr_metadata_index_{start_id}-{end_id}_dates_{start_date}-{end_date}"
141
+ )
142
+ elif start_date:
143
+ filename = f"echr_metadata_{start_id}-{end_id}_dates_{start_date}-END"
144
+ elif end_date:
145
+ filename = f"echr_metadata_{start_id}-{end_id}_datesSTART-{end_date}"
146
+ else:
147
+ filename = f"echr_metadata_{start_id}-{end_id}_dates_START-END"
148
+ else:
149
+ if start_date and end_date:
150
+ filename = (
151
+ f"echr_metadata_index_{start_id}-ALL_dates_{start_date}-{end_date}"
152
+ )
153
+ elif start_date:
154
+ filename = f"echr_metadata_{start_id}-ALL_dates_{start_date}-END"
155
+ elif end_date:
156
+ filename = f"echr_metadata_{start_id}-ALL_dates_START-{end_date}"
157
+ else:
158
+ filename = f"echr_metadata_{start_id}-ALL_dates_START-END"
159
+ return filename
160
+
161
+
162
+ def get_echr_extra(
163
+ start_id=0,
164
+ end_id=None,
165
+ start_date=None,
166
+ count=None,
167
+ end_date=None,
168
+ verbose=False,
169
+ save_file="y",
170
+ threads=10,
171
+ fields=None,
172
+ link=None,
173
+ language=None,
174
+ query_payload=None,
175
+ # New configuration parameters
176
+ batch_size=500,
177
+ timeout=60,
178
+ retry_attempts=3,
179
+ max_attempts=20,
180
+ days_per_batch=365,
181
+ progress_bar=True,
182
+ memory_efficient=True,
183
+ ):
184
+ """
185
+ Enhanced ECHR metadata and full-text extraction with improved reliability and performance.
186
+
187
+ This function extracts both metadata and full-text content from ECHR cases with
188
+ advanced features like date batching, progress tracking, and memory management.
189
+
190
+ :param int start_id: The index to start the search from (default: 0).
191
+ :param int end_id: The index to end search at, where None fetches all results.
192
+ :param str start_date: The point from which to save cases (YYYY-MM-DD format).
193
+ :param int count: Number of records to fetch (alternative to end_id).
194
+ :param str end_date: The point before which to save cases (YYYY-MM-DD format).
195
+ :param bool verbose: Whether or not to print extra information (default: False).
196
+ :param str save_file: Whether to save results to file ("y" or "n", default: "y").
197
+ :param int threads: Number of threads for full-text download (default: 10).
198
+ :param list fields: List of fields to extract (default: None, uses all fields).
199
+ :param str link: Custom HUDOC link for advanced queries.
200
+ :param list language: List of language codes (default: ["ENG"]).
201
+ :param str query_payload: Custom query payload for advanced searches.
202
+ :param int batch_size: Number of records to fetch per batch, max 500 (default: 500).
203
+ :param float timeout: Request timeout in seconds (default: 60).
204
+ :param int retry_attempts: Number of retry attempts for failed requests (default: 3).
205
+ :param int max_attempts: Maximum total attempts before giving up (default: 20).
206
+ :param int days_per_batch: Number of days per date batch for large date ranges (default: 365).
207
+ :param bool progress_bar: Whether to show progress bar (default: True).
208
+ :param bool memory_efficient: Whether to use memory-efficient processing (default: True).
209
+
210
+ :return: tuple of (pandas.DataFrame, list) containing metadata and full-text data,
211
+ or (False, False) if extraction failed.
212
+ """
213
+ df = get_echr(
214
+ start_id=start_id,
215
+ end_id=end_id,
216
+ start_date=start_date,
217
+ end_date=end_date,
218
+ verbose=verbose,
219
+ count=count,
220
+ save_file="n",
221
+ fields=fields,
222
+ link=link,
223
+ language=language,
224
+ query_payload=query_payload,
225
+ batch_size=batch_size,
226
+ timeout=timeout,
227
+ retry_attempts=retry_attempts,
228
+ max_attempts=max_attempts,
229
+ days_per_batch=days_per_batch,
230
+ progress_bar=progress_bar,
231
+ memory_efficient=memory_efficient,
232
+ )
233
+ logging.info("Full-text download will now begin")
234
+ if df is False:
235
+ return False, False
236
+ json_list = download_full_text_main(df, threads)
237
+ logging.info("Full-text download finished")
238
+ if save_file == "y":
239
+ filename = determine_filename(start_id, end_id, start_date, end_date)
240
+ filename_json = filename.replace("metadata", "full_text")
241
+ Path("data").mkdir(parents=True, exist_ok=True)
242
+ file_path = os.path.join("data", filename + ".csv")
243
+ df.to_csv(file_path, index=False)
244
+ file_path_json = os.path.join("data", filename_json + ".json")
245
+ with open(file_path_json, "w") as f:
246
+ json.dump(json_list, f)
247
+ return df, json_list
248
+ else:
249
+ return df, json_list
250
+
251
+
252
+ def get_nodes_edges(metadata_path=None, df=None, save_file="y"):
253
+ nodes, edges = echr_nodes_edges(metadata_path=metadata_path, data=df)
254
+ if save_file == "y":
255
+ Path("data").mkdir(parents=True, exist_ok=True)
256
+ edges.to_csv(
257
+ os.path.join("data", "ECHR_edges.csv"), index=False, encoding="utf-8"
258
+ )
259
+ nodes.to_csv(
260
+ os.path.join("data", "ECHR_nodes.csv"), index=False, encoding="utf-8"
261
+ )
262
+ nodes.to_json(os.path.join("data", "ECHR_nodes.json"), orient="records")
263
+ edges.to_json(os.path.join("data", "ECHR_edges.json"), orient="records")
264
+ return nodes, edges
265
+
266
+ return nodes, edges
@@ -0,0 +1,20 @@
1
+ # import logging
2
+ import sys
3
+ from os.path import abspath
4
+
5
+ import echr_extractor
6
+
7
+ current_dir = abspath(__file__)
8
+ correct_dir = "\\".join(current_dir.replace("\\", "/").split("/")[:-2])
9
+ sys.path.append(correct_dir)
10
+ # print(sys.path)
11
+
12
+
13
+ if __name__ == "__main__":
14
+ payload = (
15
+ "contentsitename:ECHR AND (NOT (doctype=PR OR doctype=HFCOMOLD OR doctype=HECOMOLD)) "
16
+ 'AND ((NOT "has been a violation of Article 6") AND ("has been no violation of Article 6")) '
17
+ 'AND ((languageisocode="ENG")) AND ((documentcollectionid="GRANDCHAMBER") OR (documentcollectionid="CHAMBER"))'
18
+ )
19
+ df = echr_extractor.get_echr(query_payload=payload)
20
+ b = 2
@@ -0,0 +1,189 @@
1
+ Metadata-Version: 2.4
2
+ Name: echr-extractor
3
+ Version: 0.0.1.dev1
4
+ Summary: Python library for extracting case law data from the European Court of Human Rights (ECHR) HUDOC database
5
+ Author-email: LawTech Lab <lawtech@maastrichtuniversity.nl>
6
+ License: Apache-2.0
7
+ Project-URL: Homepage, https://github.com/maastrichtlawtech/echr-extractor
8
+ Project-URL: Repository, https://github.com/maastrichtlawtech/echr-extractor
9
+ Project-URL: Bug Reports, https://github.com/maastrichtlawtech/echr-extractor/issues
10
+ Project-URL: Documentation, https://github.com/maastrichtlawtech/echr-extractor
11
+ Keywords: echr,extractor,european,convention,human,rights,court,case-law,legal,hudoc,data-extraction
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Intended Audience :: Legal Industry
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: License :: OSI Approved :: Apache Software License
17
+ Classifier: Operating System :: OS Independent
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.8
20
+ Classifier: Programming Language :: Python :: 3.9
21
+ Classifier: Programming Language :: Python :: 3.10
22
+ Classifier: Programming Language :: Python :: 3.11
23
+ Classifier: Topic :: Scientific/Engineering :: Information Analysis
24
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
25
+ Classifier: Topic :: Text Processing :: Markup :: HTML
26
+ Requires-Python: >=3.8
27
+ Description-Content-Type: text/markdown
28
+ License-File: LICENSE
29
+ Requires-Dist: requests>=2.26.0
30
+ Requires-Dist: pandas>=1.3.0
31
+ Requires-Dist: beautifulsoup4>=4.9.3
32
+ Requires-Dist: dateparser>=1.0.0
33
+ Requires-Dist: tqdm>=4.60.0
34
+ Provides-Extra: dev
35
+ Requires-Dist: pytest>=6.0; extra == "dev"
36
+ Requires-Dist: pytest-cov>=2.10; extra == "dev"
37
+ Requires-Dist: black>=21.0.0; extra == "dev"
38
+ Requires-Dist: isort>=5.0.0; extra == "dev"
39
+ Requires-Dist: flake8>=3.8.0; extra == "dev"
40
+ Requires-Dist: mypy>=0.910; extra == "dev"
41
+ Provides-Extra: docs
42
+ Requires-Dist: sphinx>=4.0.0; extra == "docs"
43
+ Requires-Dist: sphinx-rtd-theme>=0.5.0; extra == "docs"
44
+ Dynamic: license-file
45
+
46
+ # ECHR Extractor
47
+
48
+ Python library for extracting case law data from the European Court of Human Rights (ECHR) HUDOC database.
49
+
50
+ ## Features
51
+
52
+ - Extract metadata for ECHR cases from the HUDOC database
53
+ - Download full text content for cases
54
+ - Support for custom date ranges and case ID ranges
55
+ - Multiple language support
56
+ - Generate nodes and edges for network analysis
57
+ - Flexible output formats (CSV, JSON, in-memory DataFrames)
58
+
59
+ ## Installation
60
+
61
+ ```bash
62
+ pip install echr-extractor
63
+ ```
64
+
65
+ ## Quick Start
66
+
67
+ ```python
68
+ from echr_extractor import get_echr, get_echr_extra, get_nodes_edges
69
+
70
+ # Get basic metadata for cases
71
+ df = get_echr(start_id=0, count=100, language=['ENG'])
72
+
73
+ # Get metadata + full text
74
+ df, full_texts = get_echr_extra(start_id=0, count=100, language=['ENG'])
75
+
76
+ # Generate network data
77
+ nodes, edges = get_nodes_edges(df=df)
78
+ ```
79
+
80
+ ## Functions
81
+
82
+ ### `get_echr`
83
+
84
+ Gets all available metadata for ECHR cases from the HUDOC database.
85
+
86
+ **Parameters:**
87
+ - `start_id` (int, optional): The ID of the first case to download (default: 0)
88
+ - `end_id` (int, optional): The ID of the last case to download (default: maximum available)
89
+ - `count` (int, optional): Number of cases per language to download (default: None)
90
+ - `start_date` (str, optional): Start publication date (yyyy-mm-dd) (default: None)
91
+ - `end_date` (str, optional): End publication date (yyyy-mm-dd) (default: current date)
92
+ - `verbose` (bool, optional): Show progress information (default: False)
93
+ - `fields` (list, optional): Limit metadata fields to download (default: all fields)
94
+ - `save_file` (str, optional): Save as CSV file ('y') or return DataFrame ('n') (default: 'y')
95
+ - `language` (list, optional): Languages to download (default: ['ENG'])
96
+ - `link` (str, optional): Direct HUDOC search URL (default: None)
97
+ - `query_payload` (str, optional): Direct API query payload (default: None)
98
+
99
+ ### `get_echr_extra`
100
+
101
+ Gets metadata and downloads full text for each case.
102
+
103
+ **Parameters:** Same as `get_echr` plus:
104
+ - `threads` (int, optional): Number of threads for parallel download (default: 10)
105
+
106
+ ### `get_nodes_edges`
107
+
108
+ Generates nodes and edges for network analysis from case metadata.
109
+
110
+ **Parameters:**
111
+ - `metadata_path` (str, optional): Path to metadata CSV file (default: None)
112
+ - `df` (DataFrame, optional): Metadata DataFrame (default: None)
113
+ - `save_file` (str, optional): Save as files ('y') or return objects ('n') (default: 'y')
114
+
115
+ ## Advanced Usage
116
+
117
+ ### Using Custom Search URLs
118
+
119
+ You can use direct HUDOC search URLs:
120
+
121
+ ```python
122
+ url = "https://hudoc.echr.coe.int/eng#{%22itemid%22:[%22001-57574%22]}"
123
+ df = get_echr(link=url)
124
+ ```
125
+
126
+ ### Using Query Payloads
127
+
128
+ For more robust searching, use simple field:value queries:
129
+
130
+ ```python
131
+ payload = 'article:8'
132
+ df = get_echr(query_payload=payload)
133
+ ```
134
+
135
+ ### Date Range Filtering
136
+
137
+ ```python
138
+ df = get_echr(
139
+ start_date="2020-01-01",
140
+ end_date="2023-12-31",
141
+ language=['ENG', 'FRE']
142
+ )
143
+ ```
144
+
145
+ ### Specific Fields Only
146
+
147
+ ```python
148
+ fields = ['itemid', 'doctypebranch', 'title', 'kpdate']
149
+ df = get_echr(count=100, fields=fields)
150
+ ```
151
+
152
+ ## Requirements
153
+
154
+ - Python 3.8+
155
+ - requests
156
+ - pandas
157
+ - beautifulsoup4
158
+ - dateparser
159
+ - tqdm
160
+
161
+ ## License
162
+
163
+ This project is licensed under the Apache License 2.0 - see the LICENSE file for details.
164
+
165
+ ## Contributors
166
+
167
+ - Benjamin Rodrigues de Miranda
168
+ - Chloe Crombach
169
+ - Piotr Lewandowski
170
+ - Pranav Bapat
171
+ - Shashank MC
172
+ - Gijs van Dijck
173
+
174
+ ## Citation
175
+
176
+ If you use this library in your research, please cite:
177
+
178
+ ```bibtex
179
+ @software{echr_extractor,
180
+ title={ECHR Extractor: Python Library for European Court of Human Rights Data},
181
+ author={LawTech Lab, Maastricht University},
182
+ url={https://github.com/maastrichtlawtech/echr-extractor},
183
+ year={2024}
184
+ }
185
+ ```
186
+
187
+ ## Support
188
+
189
+ For bug reports and feature requests, please open an issue on GitHub.
@@ -0,0 +1,15 @@
1
+ echr_extractor/ECHR_html_downloader.py,sha256=dj2C0XZMMiV4PuFETgL6L4vY7acW0Wm6qDUScd2tybE,2689
2
+ echr_extractor/ECHR_metadata_harvester.py,sha256=cXXejidQ0cRFQB6ryjWnzcYKqH4U9uS6K6nM7xyriuk,20106
3
+ echr_extractor/ECHR_nodes_edges_list_transform.py,sha256=L2VjFQA1qXTqLgD9ug-9HCcqvjBlhjhJCTjOATgeTLM,9248
4
+ echr_extractor/__init__.py,sha256=Ni6iuNF27KAfpLv6e7H9N66mL4nN0jEuW6soTtrAhkM,403
5
+ echr_extractor/_version.py,sha256=QeQ7AWx2KoSWKCEjOGq3ydCSILGUGbDRIZahOZvyHSQ,717
6
+ echr_extractor/clean_ref.py,sha256=GF6jx3tTALa_ztl2Y2gNRU2oztgZIq8XHnV8D6vj-d8,146
7
+ echr_extractor/cli.py,sha256=WTjbjPanDa2zz76UmgVI4db_xMFJ4VIsiLZHXfaGkAw,3993
8
+ echr_extractor/echr.py,sha256=j4E53I7u4Z7Cwx6HURPNxpjb24Ten3W-NFUjb5Tb35g,10136
9
+ echr_extractor/testing_file.py,sha256=K6P8uTzeosaihXe-6bd3srxxqbD1J9UjuU-5DXbwooI,665
10
+ echr_extractor-0.0.1.dev1.dist-info/licenses/LICENSE,sha256=QwcOLU5TJoTeUhuIXzhdCEEDDvorGiC6-3YTOl4TecE,11356
11
+ echr_extractor-0.0.1.dev1.dist-info/METADATA,sha256=_JDtKtCIFEMa1oL8GxzqQN3-oZhIvLAcdXwvRU-_HAg,5842
12
+ echr_extractor-0.0.1.dev1.dist-info/WHEEL,sha256=_zCd3N1l69ArxyTb8rzEoP9TpbYXkqRFSNOD5OuxnTs,91
13
+ echr_extractor-0.0.1.dev1.dist-info/entry_points.txt,sha256=TmNI9MFujDX0hsVO9kCCUX4fy6HfLSHZQjAQXlJNtrw,59
14
+ echr_extractor-0.0.1.dev1.dist-info/top_level.txt,sha256=_3q8klfvemMhS7KOxqjrEIJROrwhIM0bWKDAvldqfQY,15
15
+ echr_extractor-0.0.1.dev1.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (80.9.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ echr-extractor = echr_extractor.cli:main