pyformatjson 0.2.1__tar.gz → 0.2.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/PKG-INFO +1 -1
- pyformatjson-0.2.2/pyformatjson/core/update_json.py +204 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyproject.toml +1 -1
- pyformatjson-0.2.1/pyformatjson/core/update_json.py +0 -115
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/LICENSE +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/README.md +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyformatjson/__init__.py +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyformatjson/core/_base.py +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyformatjson/main.py +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyformatjson/tools/generate_dict.py +0 -0
- {pyformatjson-0.2.1 → pyformatjson-0.2.2}/pyformatjson/tools/write_dict.py +0 -0
|
@@ -0,0 +1,204 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
import re
|
|
6
|
+
from typing import Any, Dict, List, Tuple
|
|
7
|
+
|
|
8
|
+
from ._base import split_data_list, split_text_by_length
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def load_json_data(path_json: str, filename: str) -> Dict:
|
|
12
|
+
try:
|
|
13
|
+
file_path = os.path.join(path_json, filename)
|
|
14
|
+
if not os.path.exists(file_path):
|
|
15
|
+
return {}
|
|
16
|
+
|
|
17
|
+
with open(file_path, "r", encoding="utf-8") as file:
|
|
18
|
+
return json.load(file)
|
|
19
|
+
|
|
20
|
+
except Exception as e:
|
|
21
|
+
print(f"Error loading {filename}: {e}")
|
|
22
|
+
return {}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
|
|
26
|
+
"""Update and format JSON file containing conference/journal data.
|
|
27
|
+
|
|
28
|
+
This function loads JSON data, processes and formats text fields by splitting
|
|
29
|
+
long text into appropriate lengths, checks for duplicate abbreviations, and
|
|
30
|
+
saves the updated data back to the file.
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
full_json_cj (str): Full path to the conferences/journals JSON file
|
|
34
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
35
|
+
|
|
36
|
+
Returns:
|
|
37
|
+
Dict[str, Any]: Processed JSON data dictionary.
|
|
38
|
+
"""
|
|
39
|
+
# Load Json Data
|
|
40
|
+
json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
|
|
41
|
+
|
|
42
|
+
# Process and format text fields in JSON data.
|
|
43
|
+
for pub in json_dict:
|
|
44
|
+
for flag in ["txt_abouts", "txt_remarks"]:
|
|
45
|
+
data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
|
|
46
|
+
temps = []
|
|
47
|
+
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
48
|
+
temps.extend(split_text_by_length(line, 105))
|
|
49
|
+
if temps:
|
|
50
|
+
json_dict[pub].update({flag: temps})
|
|
51
|
+
|
|
52
|
+
for abbr in json_dict[pub][conferences_or_journals]:
|
|
53
|
+
for flag in ["txt_abouts", "txt_remarks"]:
|
|
54
|
+
data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
|
|
55
|
+
temps = []
|
|
56
|
+
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
57
|
+
temps.extend(split_text_by_length(line, 97))
|
|
58
|
+
if temps:
|
|
59
|
+
json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
|
|
60
|
+
|
|
61
|
+
# Generate standard form
|
|
62
|
+
abbr_dict = generate_standard_form(json_dict, conferences_or_journals)
|
|
63
|
+
|
|
64
|
+
_, flag = CheckAcronymAbbrAndFullDict().run(abbr_dict)
|
|
65
|
+
|
|
66
|
+
# Save updated JSON
|
|
67
|
+
if flag and json_dict:
|
|
68
|
+
with open(full_json_cj, "w", encoding="utf-8") as f:
|
|
69
|
+
f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
70
|
+
|
|
71
|
+
return json_dict
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def generate_standard_form(json_dict: Dict[str, Any], conferences_or_journals: str):
|
|
75
|
+
# Check for duplicate abbreviations in the data.
|
|
76
|
+
abbr_list = []
|
|
77
|
+
for pub in json_dict:
|
|
78
|
+
if conferences_or_journals in json_dict[pub]:
|
|
79
|
+
for abbr in json_dict[pub][conferences_or_journals]:
|
|
80
|
+
if abbr in abbr_list:
|
|
81
|
+
raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
|
|
82
|
+
abbr_list.append(abbr)
|
|
83
|
+
|
|
84
|
+
# Extract abbreviation and name data from all publications
|
|
85
|
+
abbr_dict: Dict[str, Dict[str, List[str]]] = {}
|
|
86
|
+
for pub in json_dict:
|
|
87
|
+
if conferences_or_journals in json_dict[pub]:
|
|
88
|
+
for abbr, v in json_dict[pub][conferences_or_journals].items():
|
|
89
|
+
# Store both abbreviated names and full names for each abbreviation
|
|
90
|
+
abbr_dict.update({abbr: {"names_abbr": v.get("names_abbr", []), "names_full": v.get("names_full", [])}})
|
|
91
|
+
|
|
92
|
+
return abbr_dict
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
class CheckAcronymAbbrAndFullDict:
|
|
96
|
+
def __init__(self, names_abbr="names_abbr", names_full="names_full"):
|
|
97
|
+
self.names_abbr = names_abbr
|
|
98
|
+
self.names_full = names_full
|
|
99
|
+
|
|
100
|
+
def run(self, dict_data: dict[str, dict[str, list[str]]]) -> Tuple[dict[str, dict[str, list[str]]], bool]:
|
|
101
|
+
# Check if each acronym has equal number of abbreviations and full forms
|
|
102
|
+
dict_data, length_check = self._validate_lengths(dict_data)
|
|
103
|
+
|
|
104
|
+
# Check for duplicate abbreviations or full forms across all acronyms
|
|
105
|
+
dict_data, duplicate_check = self._check_duplicates(dict_data)
|
|
106
|
+
|
|
107
|
+
# Check for matching patterns in both abbreviations and full forms
|
|
108
|
+
dict_data, abbr_match_check = self._check_matches(dict_data, self.names_abbr)
|
|
109
|
+
dict_data, full_match_check = self._check_matches(dict_data, self.names_full)
|
|
110
|
+
|
|
111
|
+
return dict_data, all([length_check, duplicate_check, abbr_match_check, full_match_check])
|
|
112
|
+
|
|
113
|
+
def _validate_lengths(self, dict_data):
|
|
114
|
+
"""Validate that each acronym has equal number of abbreviations and full forms."""
|
|
115
|
+
valid_data, all_valid = {}, True
|
|
116
|
+
for acronym, value_dict in dict_data.items():
|
|
117
|
+
names_abbr = value_dict.get(self.names_abbr, [])
|
|
118
|
+
names_full = value_dict.get(self.names_full, [])
|
|
119
|
+
|
|
120
|
+
if len(names_abbr) != len(names_full):
|
|
121
|
+
all_valid = False
|
|
122
|
+
print(
|
|
123
|
+
f"Length mismatch in '{acronym}': {len(names_abbr)} abbreviations vs {len(names_full)} full forms"
|
|
124
|
+
)
|
|
125
|
+
else:
|
|
126
|
+
valid_data[acronym] = value_dict
|
|
127
|
+
return valid_data, all_valid
|
|
128
|
+
|
|
129
|
+
def _check_duplicates(self, data):
|
|
130
|
+
"""Check for duplicate abbreviations or full forms across all acronyms."""
|
|
131
|
+
valid_data = {}
|
|
132
|
+
all_unique = True
|
|
133
|
+
seen_abbrs = set()
|
|
134
|
+
seen_fulls = set()
|
|
135
|
+
|
|
136
|
+
for acronym, values in data.items():
|
|
137
|
+
has_duplicate = False
|
|
138
|
+
|
|
139
|
+
# Check for duplicate abbreviations
|
|
140
|
+
abbrs_lower = set([abbr.lower() for abbr in values.get(self.names_abbr, [])])
|
|
141
|
+
for abbr in abbrs_lower:
|
|
142
|
+
if abbr in seen_abbrs:
|
|
143
|
+
print(f"Duplicate abbreviation '{abbr}' found in '{acronym}'")
|
|
144
|
+
has_duplicate = True
|
|
145
|
+
else:
|
|
146
|
+
seen_abbrs.add(abbr)
|
|
147
|
+
|
|
148
|
+
# Check for duplicate full forms
|
|
149
|
+
fulls_lower = set([full.lower() for full in values.get(self.names_full, [])])
|
|
150
|
+
for full in fulls_lower:
|
|
151
|
+
if full in seen_fulls:
|
|
152
|
+
print(f"Duplicate full form '{full}' found in '{acronym}'")
|
|
153
|
+
has_duplicate = True
|
|
154
|
+
else:
|
|
155
|
+
seen_fulls.add(full)
|
|
156
|
+
|
|
157
|
+
if not has_duplicate:
|
|
158
|
+
valid_data[acronym] = values
|
|
159
|
+
else:
|
|
160
|
+
all_unique = False
|
|
161
|
+
|
|
162
|
+
return valid_data, all_unique
|
|
163
|
+
|
|
164
|
+
def _check_matches(self, data, key_type: str):
|
|
165
|
+
"""Check for exact matches in abbreviations or full forms between different acronyms."""
|
|
166
|
+
valid_data = {}
|
|
167
|
+
no_matches = True
|
|
168
|
+
acronyms = sorted(list(data.keys()))
|
|
169
|
+
|
|
170
|
+
for i, main_acronym in enumerate(acronyms):
|
|
171
|
+
# Normalize items: lowercase and remove parentheses
|
|
172
|
+
main_items = [
|
|
173
|
+
item.lower().replace("(", "").replace(")", "")
|
|
174
|
+
for item in data[main_acronym].get(key_type, [])
|
|
175
|
+
]
|
|
176
|
+
|
|
177
|
+
# Create exact match patterns
|
|
178
|
+
patterns = [re.compile(f"^{item}$") for item in main_items]
|
|
179
|
+
|
|
180
|
+
matches_found = []
|
|
181
|
+
|
|
182
|
+
# Compare with other acronyms
|
|
183
|
+
for other_acronym in acronyms[i + 1:]:
|
|
184
|
+
other_items = [
|
|
185
|
+
item.lower().replace("(", "").replace(")", "")
|
|
186
|
+
for item in data[other_acronym].get(key_type, [])
|
|
187
|
+
]
|
|
188
|
+
|
|
189
|
+
# Find matching items
|
|
190
|
+
matching_items = [
|
|
191
|
+
item for item in other_items
|
|
192
|
+
if any(pattern.match(item) for pattern in patterns)
|
|
193
|
+
]
|
|
194
|
+
|
|
195
|
+
if matching_items:
|
|
196
|
+
matches_found.append([main_acronym, other_acronym, matching_items])
|
|
197
|
+
|
|
198
|
+
if matches_found:
|
|
199
|
+
no_matches = False
|
|
200
|
+
print(f"Found matches in {key_type}: {matches_found}")
|
|
201
|
+
else:
|
|
202
|
+
valid_data[main_acronym] = data[main_acronym]
|
|
203
|
+
|
|
204
|
+
return valid_data, no_matches
|
|
@@ -1,115 +0,0 @@
|
|
|
1
|
-
# coding=utf-8
|
|
2
|
-
|
|
3
|
-
import json
|
|
4
|
-
import os
|
|
5
|
-
from typing import Any, Dict
|
|
6
|
-
|
|
7
|
-
from ._base import split_data_list, split_text_by_length
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
def load_json_data(path_json: str, filename: str) -> Dict:
|
|
11
|
-
"""Load JSON data from a specified file.
|
|
12
|
-
|
|
13
|
-
This function attempts to load JSON data from a file located in the specified
|
|
14
|
-
directory. If the file doesn't exist or there's an error loading it, an empty
|
|
15
|
-
dictionary is returned.
|
|
16
|
-
|
|
17
|
-
Args:
|
|
18
|
-
path_json (str): Directory path containing the JSON file.
|
|
19
|
-
filename (str): Name of the JSON file (with .json extension).
|
|
20
|
-
|
|
21
|
-
Returns:
|
|
22
|
-
Dict: The loaded JSON data as a dictionary, or empty dict if file not found
|
|
23
|
-
or error occurs.
|
|
24
|
-
|
|
25
|
-
Example:
|
|
26
|
-
>>> load_json_data("/data", "conferences")
|
|
27
|
-
{"publisher1": {"conferences": {...}}}
|
|
28
|
-
"""
|
|
29
|
-
try:
|
|
30
|
-
file_path = os.path.join(path_json, filename)
|
|
31
|
-
if not os.path.exists(file_path):
|
|
32
|
-
return {}
|
|
33
|
-
|
|
34
|
-
with open(file_path, "r", encoding="utf-8") as file:
|
|
35
|
-
return json.load(file)
|
|
36
|
-
|
|
37
|
-
except Exception as e:
|
|
38
|
-
print(f"Error loading {filename}: {e}")
|
|
39
|
-
return {}
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
|
|
43
|
-
"""Update and format JSON file containing conference/journal data.
|
|
44
|
-
|
|
45
|
-
This function loads JSON data, processes and formats text fields by splitting
|
|
46
|
-
long text into appropriate lengths, checks for duplicate abbreviations, and
|
|
47
|
-
saves the updated data back to the file.
|
|
48
|
-
|
|
49
|
-
Args:
|
|
50
|
-
full_json_cj (str): Full path to the conferences/journals JSON file
|
|
51
|
-
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
52
|
-
|
|
53
|
-
Returns:
|
|
54
|
-
Dict[str, Any]: Processed JSON data dictionary.
|
|
55
|
-
"""
|
|
56
|
-
# Load Json Data
|
|
57
|
-
json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
|
|
58
|
-
|
|
59
|
-
# Process and format text fields in JSON data.
|
|
60
|
-
for pub in json_dict:
|
|
61
|
-
for flag in ["txt_abouts", "txt_remarks"]:
|
|
62
|
-
data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
|
|
63
|
-
temps = []
|
|
64
|
-
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
65
|
-
temps.extend(split_text_by_length(line, 105))
|
|
66
|
-
if temps:
|
|
67
|
-
json_dict[pub].update({flag: temps})
|
|
68
|
-
|
|
69
|
-
for abbr in json_dict[pub][conferences_or_journals]:
|
|
70
|
-
for flag in ["txt_abouts", "txt_remarks"]:
|
|
71
|
-
data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
|
|
72
|
-
temps = []
|
|
73
|
-
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
74
|
-
temps.extend(split_text_by_length(line, 97))
|
|
75
|
-
if temps:
|
|
76
|
-
json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
|
|
77
|
-
|
|
78
|
-
# Check for duplicate abbreviations
|
|
79
|
-
_check_duplicate_abbr(json_dict, conferences_or_journals)
|
|
80
|
-
|
|
81
|
-
# Save updated JSON
|
|
82
|
-
if json_dict:
|
|
83
|
-
with open(full_json_cj, "w", encoding="utf-8") as f:
|
|
84
|
-
f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
85
|
-
|
|
86
|
-
return json_dict
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
|
|
90
|
-
"""Check for duplicate abbreviations in the data.
|
|
91
|
-
|
|
92
|
-
This function validates that there are no duplicate abbreviations within
|
|
93
|
-
the same publication type across all publishers.
|
|
94
|
-
|
|
95
|
-
Args:
|
|
96
|
-
json_dict (Dict[str, Any]): JSON data dictionary containing publication information.
|
|
97
|
-
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
98
|
-
|
|
99
|
-
Raises:
|
|
100
|
-
ValueError: If duplicate abbreviations are found in the data.
|
|
101
|
-
|
|
102
|
-
Example:
|
|
103
|
-
>>> _check_duplicate_abbr(data, "conferences")
|
|
104
|
-
# Raises ValueError if "ICML" appears twice in conferences
|
|
105
|
-
"""
|
|
106
|
-
abbr_list = []
|
|
107
|
-
|
|
108
|
-
for pub in json_dict:
|
|
109
|
-
if conferences_or_journals in json_dict[pub]:
|
|
110
|
-
for abbr in json_dict[pub][conferences_or_journals]:
|
|
111
|
-
if abbr in abbr_list:
|
|
112
|
-
raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
|
|
113
|
-
abbr_list.append(abbr)
|
|
114
|
-
|
|
115
|
-
return None
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|