echr-extractor 0.0.1.dev1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,308 @@
1
+ import logging
2
+ import re
3
+
4
+ import dateparser
5
+ import numpy as np
6
+ import pandas as pd
7
+ from tqdm import tqdm
8
+
9
+ from .clean_ref import clean_pattern
10
+
11
+
12
+ def open_metadata(PATH_metadata):
13
+ """
14
+ Finds the ECHR metadata file and loads it into a dataframe
15
+
16
+ param filename_metadata: string with path to metadata
17
+ """
18
+ try:
19
+ df = pd.read_csv(PATH_metadata) # change hard coded path
20
+ return df
21
+ except FileNotFoundError:
22
+ logging.warning("File not found. Please check the path to the metadata file.")
23
+ return False
24
+
25
+
26
+ def concat_metadata(df):
27
+ agg_func = {
28
+ "itemid": "first",
29
+ "appno": "first",
30
+ "article": "first",
31
+ "conclusion": "first",
32
+ "docname": "first",
33
+ "doctype": "first",
34
+ "doctypebranch": "first",
35
+ "ecli": "first",
36
+ "importance": "first",
37
+ "judgementdate": "first",
38
+ "languageisocode": ", ".join,
39
+ "originatingbody": "first",
40
+ "violation": "first",
41
+ "nonviolation": "first",
42
+ "extractedappno": "first",
43
+ "scl": "first",
44
+ }
45
+ new_df = df.groupby("ecli").agg(agg_func)
46
+ # print(new_df)
47
+ return new_df
48
+
49
+
50
+ def get_language_from_metadata(df):
51
+ df = concat_metadata(df)
52
+ df.to_json("langisocode-nodes.json", orient="records")
53
+
54
+
55
+ def retrieve_edges_list(df):
56
+ """
57
+ Returns a dataframe consisting of 2 columns 'source' and 'target' which
58
+ indicate a reference link between cases.
59
+
60
+ params:
61
+ df -- the node list extracted from the metadata
62
+ df -- the complete dataframe from the metadata
63
+ """
64
+ edges = list()
65
+
66
+ count = 0
67
+ missing_cases = []
68
+ bar = tqdm(total=len(df.index), colour="GREEN", position=0, leave=True)
69
+ for index, item in df.iterrows():
70
+ bar.update(1)
71
+ eclis = []
72
+ extracted_appnos = []
73
+ if item.extractedappno is not np.nan:
74
+ extracted_appnos = item.extractedappno.split(";")
75
+ if item.scl is np.nan:
76
+ continue
77
+ """
78
+ Split the references from the scl column
79
+ into a list of references.
80
+
81
+ Example:
82
+ references in string: "Ali v. Switzerland, 5 August 1998, § 32,
83
+ Reports of Judgments and
84
+ Decisions 1998-V;Sevgi Erdogan v. Turkey (striking out),
85
+ no. 28492/95, 29 April 2003"
86
+
87
+ ["Ali v. Switzerland, 5 August 1998, § 32, Reports of Judgments and
88
+ Decisions 1998-V", "Sevgi Erdogan v. Turkey (striking out), no.
89
+ 28492/95, 29 April 2003"]
90
+ """
91
+ ref_list = item.scl.split(";")
92
+ new_ref_list = [i.replace("\n", "") for i in ref_list]
93
+
94
+ for ref in new_ref_list:
95
+ app_number = re.findall(r"\d{3,5}/\d{2}", ref)
96
+
97
+ app_number.extend(extracted_appnos)
98
+
99
+ app_number = set(app_number)
100
+
101
+ if len(app_number) > 0:
102
+ # get dataframe with all possible cases by application number
103
+ app_number = [";".join(app_number)]
104
+ case = lookup_app_number(app_number, df)
105
+ if len(case) == 0: # if failed try name?
106
+ case = lookup_casename(ref, df)
107
+ else: # if no application number in reference
108
+ # get dataframe with all possible cases by casename
109
+ case = lookup_casename(ref, df)
110
+
111
+ components = ref.split(",")
112
+ # get the year of case
113
+ year_from_ref = get_year_from_ref(components)
114
+
115
+ # remove cases in different language than reference
116
+ case = remove_cases_based_on_language(case, components)
117
+ case = remove_cases_based_on_year(case, year_from_ref)
118
+
119
+ if len(case) > 0:
120
+ for _, row in case.iterrows():
121
+ eclis.append(row.ecli)
122
+ else:
123
+ count = count + 1
124
+ missing_cases.append(ref)
125
+
126
+ eclis = set(eclis)
127
+ eclis = [i for i in eclis if (i and i != item.ecli)]
128
+ # add ecli to edges list
129
+ if len(eclis) == 0: # This should not have to happen at every iteration,
130
+ # concat might be slow
131
+ continue
132
+ for target in eclis:
133
+ edges.append({"source": item.ecli, "target": target})
134
+
135
+ edges = pd.DataFrame.from_records(edges)
136
+ return edges
137
+
138
+
139
+ def remove_cases_based_on_year(case, year_from_ref):
140
+ for id, i in case.iterrows():
141
+ if i.judgementdate is np.nan:
142
+ continue
143
+ try:
144
+ date = dateparser.parse(i.judgementdate)
145
+ except Exception:
146
+ date = False
147
+ if date:
148
+ year_from_case = date.year
149
+ if year_from_case - year_from_ref == 0:
150
+ case = case[
151
+ case["judgementdate"].str.contains(
152
+ str(year_from_ref), regex=False, flags=re.IGNORECASE, na=False
153
+ )
154
+ ]
155
+ return case
156
+
157
+
158
+ def remove_cases_based_on_language(cases, components):
159
+ for id, it in cases.iterrows():
160
+ if "v." in components[0]:
161
+ lang = "ENG"
162
+ else:
163
+ lang = "FRE"
164
+
165
+ if lang not in it.languageisocode:
166
+ cases = cases[
167
+ cases["languageisocode"].str.contains(
168
+ lang, regex=False, flags=re.IGNORECASE, na=False
169
+ )
170
+ ]
171
+ return cases
172
+
173
+
174
+ def lookup_app_number(pattern, df):
175
+ """
176
+ Returns a list with rows containing the cases linked
177
+ to the found app numbers.
178
+ """
179
+ row = df.loc[df["appno"].isin(pattern)]
180
+
181
+ if row.empty:
182
+ return pd.DataFrame()
183
+ elif row.shape[0] > 1:
184
+ return row
185
+ else:
186
+ return row
187
+
188
+
189
+ def lookup_casename(ref, df):
190
+ """
191
+ Process the reference for lookup in metadata.
192
+ Returns the rows corresponding to the cases.
193
+
194
+ - Example of the processing (2 variants) -
195
+
196
+ Original reference from scl:
197
+ - Hentrich v. France, 22 September 1994, § 42, Series A no. 296-A
198
+ - Eur. Court H.R. James and Others judgment of 21 February 1986,
199
+ Series A no. 98, p. 46, para. 81
200
+
201
+ Split on ',' and take first item:
202
+ Hentrich v. France
203
+ Eur. Court H.R. James and Others judgment of 21 February 1986
204
+
205
+ If certain pattern from CLEAN_REF in case name, then remove:
206
+ Eur. Court H.R. James and Others judgment of 21 February 1986 -->
207
+ James and Others
208
+
209
+ Change name to upper case and add additional text to match metadata:
210
+ Hentrich v. France --> CASE OF HENTRICH V. FRANCE
211
+ James and Others --> CASE OF JAMES AND OTHERS
212
+ """
213
+ name = get_casename(ref)
214
+
215
+ # DEV note: In case, add more patterns to clean_ref.py in future
216
+ patterns = clean_pattern
217
+
218
+ uptext = name.upper()
219
+
220
+ if "NO." in uptext:
221
+ uptext = uptext.replace("NO.", "No.")
222
+
223
+ if "BV" in uptext:
224
+ uptext = uptext.replace("BV", "B.V.")
225
+
226
+ if "V." in name:
227
+ uptext = uptext.replace("V.", "v.")
228
+ lang = "ENG"
229
+ else:
230
+ uptext = uptext.replace("C.", "c.")
231
+ lang = "FRE"
232
+
233
+ for pattern in patterns:
234
+ uptext = re.sub(pattern, "", uptext)
235
+
236
+ uptext = re.sub(r"\[.*", "", uptext)
237
+ uptext = uptext.strip()
238
+ row = df[df["docname"].str.contains(uptext, regex=False, flags=re.IGNORECASE)]
239
+
240
+ # if len(row) == 0:
241
+ # print("no cases matched: ", name)
242
+
243
+ return row
244
+
245
+
246
+ def get_casename(ref):
247
+ count = 0
248
+ if "v." in ref:
249
+ slice_at_versus = ref.split("v.") # skip if typo (count how many)
250
+ elif "c." in ref:
251
+ slice_at_versus = ref.split("c.")
252
+ else:
253
+ count = count + 1
254
+ name = ref.split(",")
255
+ return name[0]
256
+
257
+ num_commas = slice_at_versus[0].count(",")
258
+
259
+ if num_commas > 0:
260
+ num_commas = num_commas + 1
261
+ name = ",".join(ref.split(",", num_commas)[:num_commas])
262
+ else:
263
+ name = ref.split(",")
264
+ return name[0]
265
+ return name
266
+
267
+
268
+ def get_year_from_ref(ref):
269
+ for component in ref:
270
+ if "§" in component:
271
+ continue
272
+ component = re.sub("judgment of ", "", component)
273
+ if dateparser.parse(component) is not None:
274
+ date = dateparser.parse(component)
275
+ elif "ECHR" in component or "CEDH" in component:
276
+ if "ECHR" in component or "CEDH" in component:
277
+ date = re.sub("ECHR ", "", component)
278
+ date = re.sub("CEDH ", "", date)
279
+ date = date.strip()
280
+ date = re.sub("-.*", "", date)
281
+ date = re.sub(r"\s.*", "", date)
282
+ date = dateparser.parse(date)
283
+
284
+ try:
285
+ return date.year
286
+ except AttributeError:
287
+ return 0
288
+
289
+
290
+ def echr_nodes_edges(metadata_path=None, data=None):
291
+ """
292
+ Create nodes and edges list for the ECHR data.
293
+ """
294
+ logging.info("\n--- COLLECTING METADATA ---\n")
295
+ if metadata_path:
296
+ data = open_metadata(metadata_path)
297
+ elif data is None:
298
+ logging.warning("No dataframe data provided. Returning...")
299
+ return "", ""
300
+ logging.info("\n--- EXTRACTING NODES LIST ---\n")
301
+ # get_language_from_metadata(nodes)
302
+
303
+ logging.info("\n--- EXTRACTING EDGES LIST ---\n")
304
+ edges = retrieve_edges_list(data)
305
+
306
+ # nodes.to_json(JSON_ECHR_NODES, orient="records")
307
+ # edges.to_json(JSON_ECHR_EDGES, orient="records")
308
+ return data, edges
@@ -0,0 +1,18 @@
1
+ """ECHR Extractor - Python library for extracting ECHR case data."""
2
+
3
+ import logging
4
+
5
+ from .echr import get_echr, get_echr_extra, get_nodes_edges
6
+
7
+ __version__ = "1.0.44"
8
+ __author__ = "LawTech Lab, Maastricht University"
9
+ __email__ = "lawtech@maastrichtuniversity.nl"
10
+
11
+ # Configure logging
12
+ logging.basicConfig(level=logging.INFO)
13
+
14
+ __all__ = [
15
+ "get_echr",
16
+ "get_echr_extra",
17
+ "get_nodes_edges",
18
+ ]
@@ -0,0 +1,34 @@
1
+ # file generated by setuptools-scm
2
+ # don't change, don't track in version control
3
+
4
+ __all__ = [
5
+ "__version__",
6
+ "__version_tuple__",
7
+ "version",
8
+ "version_tuple",
9
+ "__commit_id__",
10
+ "commit_id",
11
+ ]
12
+
13
+ TYPE_CHECKING = False
14
+ if TYPE_CHECKING:
15
+ from typing import Tuple
16
+ from typing import Union
17
+
18
+ VERSION_TUPLE = Tuple[Union[int, str], ...]
19
+ COMMIT_ID = Union[str, None]
20
+ else:
21
+ VERSION_TUPLE = object
22
+ COMMIT_ID = object
23
+
24
+ version: str
25
+ __version__: str
26
+ __version_tuple__: VERSION_TUPLE
27
+ version_tuple: VERSION_TUPLE
28
+ commit_id: COMMIT_ID
29
+ __commit_id__: COMMIT_ID
30
+
31
+ __version__ = version = '0.0.1.dev1'
32
+ __version_tuple__ = version_tuple = (0, 0, 1, 'dev1')
33
+
34
+ __commit_id__ = commit_id = None
@@ -0,0 +1,5 @@
1
+ """
2
+ This module contains the list of patterns for reference lookup in metadata
3
+ """
4
+
5
+ clean_pattern = ["EUR. COURT H.R.", "JUDGMENT OF.*", " DU.*"]
echr_extractor/cli.py ADDED
@@ -0,0 +1,125 @@
1
+ """Command-line interface for ECHR Extractor."""
2
+
3
+ import argparse
4
+ import sys
5
+
6
+ from . import get_echr, get_echr_extra, get_nodes_edges
7
+
8
+
9
+ def main() -> None:
10
+ """Main CLI entry point."""
11
+ parser = argparse.ArgumentParser(
12
+ description="Extract case law data from ECHR HUDOC database"
13
+ )
14
+
15
+ subparsers = parser.add_subparsers(dest="command", help="Available commands")
16
+
17
+ # Basic extraction command
18
+ extract_parser = subparsers.add_parser("extract", help="Extract ECHR metadata")
19
+ add_common_args(extract_parser)
20
+
21
+ # Full extraction command
22
+ extract_full_parser = subparsers.add_parser(
23
+ "extract-full", help="Extract ECHR metadata and full text"
24
+ )
25
+ add_common_args(extract_full_parser)
26
+ extract_full_parser.add_argument(
27
+ "--threads",
28
+ type=int,
29
+ default=10,
30
+ help="Number of threads for parallel download (default: 10)",
31
+ )
32
+
33
+ # Network analysis command
34
+ network_parser = subparsers.add_parser(
35
+ "network", help="Generate nodes and edges for network analysis"
36
+ )
37
+ network_parser.add_argument(
38
+ "--metadata-path", type=str, help="Path to metadata CSV file"
39
+ )
40
+ network_parser.add_argument(
41
+ "--no-save", action="store_true", help="Don't save files, return objects only"
42
+ )
43
+
44
+ args = parser.parse_args()
45
+
46
+ if args.command is None:
47
+ parser.print_help()
48
+ sys.exit(1)
49
+
50
+ try:
51
+ if args.command == "extract":
52
+ result = get_echr(
53
+ start_id=args.start_id,
54
+ end_id=args.end_id,
55
+ count=args.count,
56
+ start_date=args.start_date,
57
+ end_date=args.end_date,
58
+ verbose=args.verbose,
59
+ save_file="n" if args.no_save else "y",
60
+ fields=args.fields,
61
+ language=args.language,
62
+ )
63
+ print(f"Extracted {len(result)} cases")
64
+
65
+ elif args.command == "extract-full":
66
+ df, texts = get_echr_extra(
67
+ start_id=args.start_id,
68
+ end_id=args.end_id,
69
+ count=args.count,
70
+ start_date=args.start_date,
71
+ end_date=args.end_date,
72
+ verbose=args.verbose,
73
+ save_file="n" if args.no_save else "y",
74
+ threads=args.threads,
75
+ fields=args.fields,
76
+ language=args.language,
77
+ )
78
+ print(f"Extracted {len(df)} cases with full text")
79
+
80
+ elif args.command == "network":
81
+ nodes, edges = get_nodes_edges(
82
+ metadata_path=args.metadata_path, save_file="n" if args.no_save else "y"
83
+ )
84
+ print(f"Generated {len(nodes)} nodes and {len(edges)} edges")
85
+
86
+ except Exception as e:
87
+ print(f"Error: {e}", file=sys.stderr)
88
+ sys.exit(1)
89
+
90
+
91
+ def add_common_args(parser: argparse.ArgumentParser) -> None:
92
+ """Add common arguments to a parser."""
93
+ parser.add_argument(
94
+ "--start-id",
95
+ type=int,
96
+ default=0,
97
+ help="ID of first case to download (default: 0)",
98
+ )
99
+ parser.add_argument("--end-id", type=int, help="ID of last case to download")
100
+ parser.add_argument(
101
+ "--count", type=int, help="Number of cases per language to download"
102
+ )
103
+ parser.add_argument(
104
+ "--start-date", type=str, help="Start publication date (yyyy-mm-dd)"
105
+ )
106
+ parser.add_argument(
107
+ "--end-date", type=str, help="End publication date (yyyy-mm-dd)"
108
+ )
109
+ parser.add_argument(
110
+ "--verbose", action="store_true", help="Show progress information"
111
+ )
112
+ parser.add_argument(
113
+ "--no-save", action="store_true", help="Don't save files, return objects only"
114
+ )
115
+ parser.add_argument("--fields", nargs="+", help="Limit metadata fields to download")
116
+ parser.add_argument(
117
+ "--language",
118
+ nargs="+",
119
+ default=["ENG"],
120
+ help="Languages to download (default: ENG)",
121
+ )
122
+
123
+
124
+ if __name__ == "__main__":
125
+ main()