pyformatjson 0.2.1__tar.gz → 0.2.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyformatjson
3
- Version: 0.2.1
3
+ Version: 0.2.3
4
4
  Summary: pyformatjson
5
5
  License: GPL-3.0-or-later
6
6
  License-File: LICENSE
@@ -0,0 +1,204 @@
1
+ # coding=utf-8
2
+
3
+ import json
4
+ import os
5
+ import re
6
+ from typing import Any, Dict, List, Tuple
7
+
8
+ from ._base import split_data_list, split_text_by_length
9
+
10
+
11
+ def load_json_data(path_json: str, filename: str) -> Dict:
12
+ try:
13
+ file_path = os.path.join(path_json, filename)
14
+ if not os.path.exists(file_path):
15
+ return {}
16
+
17
+ with open(file_path, "r", encoding="utf-8") as file:
18
+ return json.load(file)
19
+
20
+ except Exception as e:
21
+ print(f"Error loading {filename}: {e}")
22
+ return {}
23
+
24
+
25
+ def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
26
+ """Update and format JSON file containing conference/journal data.
27
+
28
+ This function loads JSON data, processes and formats text fields by splitting
29
+ long text into appropriate lengths, checks for duplicate abbreviations, and
30
+ saves the updated data back to the file.
31
+
32
+ Args:
33
+ full_json_cj (str): Full path to the conferences/journals JSON file
34
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
35
+
36
+ Returns:
37
+ Dict[str, Any]: Processed JSON data dictionary.
38
+ """
39
+ # Load Json Data
40
+ json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
41
+
42
+ # Process and format text fields in JSON data.
43
+ for pub in json_dict:
44
+ for flag in ["txt_abouts", "txt_remarks"]:
45
+ data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
46
+ temps = []
47
+ for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
48
+ temps.extend(split_text_by_length(line, 105))
49
+ if temps:
50
+ json_dict[pub].update({flag: temps})
51
+
52
+ for abbr in json_dict[pub][conferences_or_journals]:
53
+ for flag in ["txt_abouts", "txt_remarks"]:
54
+ data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
55
+ temps = []
56
+ for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
57
+ temps.extend(split_text_by_length(line, 97))
58
+ if temps:
59
+ json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
60
+
61
+ # Generate standard form
62
+ abbr_dict = generate_standard_form(json_dict, conferences_or_journals)
63
+
64
+ _, flag = CheckAcronymAbbrAndFullDict().run(abbr_dict)
65
+
66
+ # Save updated JSON
67
+ if flag and json_dict:
68
+ with open(full_json_cj, "w", encoding="utf-8") as f:
69
+ f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
70
+
71
+ return json_dict
72
+
73
+
74
+ def generate_standard_form(json_dict: Dict[str, Any], conferences_or_journals: str):
75
+ # Check for duplicate abbreviations in the data.
76
+ abbr_list = []
77
+ for pub in json_dict:
78
+ if conferences_or_journals in json_dict[pub]:
79
+ for abbr in json_dict[pub][conferences_or_journals]:
80
+ if abbr in abbr_list:
81
+ raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
82
+ abbr_list.append(abbr)
83
+
84
+ # Extract abbreviation and name data from all publications
85
+ abbr_dict: Dict[str, Dict[str, List[str]]] = {}
86
+ for pub in json_dict:
87
+ if conferences_or_journals in json_dict[pub]:
88
+ for abbr, v in json_dict[pub][conferences_or_journals].items():
89
+ # Store both abbreviated names and full names for each abbreviation
90
+ abbr_dict.update({abbr: {"names_abbr": v.get("names_abbr", []), "names_full": v.get("names_full", [])}})
91
+
92
+ return abbr_dict
93
+
94
+
95
+ class CheckAcronymAbbrAndFullDict:
96
+ def __init__(self, names_abbr="names_abbr", names_full="names_full"):
97
+ self.names_abbr = names_abbr
98
+ self.names_full = names_full
99
+
100
+ def run(self, dict_data: dict[str, dict[str, list[str]]]) -> Tuple[dict[str, dict[str, list[str]]], bool]:
101
+ # Check if each acronym has equal number of abbreviations and full forms
102
+ dict_data, length_check = self._validate_lengths(dict_data)
103
+
104
+ # Check for duplicate abbreviations or full forms across all acronyms
105
+ dict_data, duplicate_check = self._check_duplicates(dict_data)
106
+
107
+ # Check for matching patterns in both abbreviations and full forms
108
+ dict_data, abbr_match_check = self._check_matches(dict_data, self.names_abbr)
109
+ dict_data, full_match_check = self._check_matches(dict_data, self.names_full)
110
+
111
+ return dict_data, all([length_check, duplicate_check, abbr_match_check, full_match_check])
112
+
113
+ def _validate_lengths(self, dict_data):
114
+ """Validate that each acronym has equal number of abbreviations and full forms."""
115
+ valid_data, all_valid = {}, True
116
+ for acronym, value_dict in dict_data.items():
117
+ names_abbr = value_dict.get(self.names_abbr, [])
118
+ names_full = value_dict.get(self.names_full, [])
119
+
120
+ if len(names_abbr) != len(names_full):
121
+ all_valid = False
122
+ print(
123
+ f"Length mismatch in '{acronym}': {len(names_abbr)} abbreviations vs {len(names_full)} full forms"
124
+ )
125
+ else:
126
+ valid_data[acronym] = value_dict
127
+ return valid_data, all_valid
128
+
129
+ def _check_duplicates(self, data):
130
+ """Check for duplicate abbreviations or full forms across all acronyms."""
131
+ valid_data = {}
132
+ all_unique = True
133
+ seen_abbrs = set()
134
+ seen_fulls = set()
135
+
136
+ for acronym, values in data.items():
137
+ has_duplicate = False
138
+
139
+ # Check for duplicate abbreviations
140
+ abbrs_lower = set([abbr.lower() for abbr in values.get(self.names_abbr, [])])
141
+ for abbr in abbrs_lower:
142
+ if abbr in seen_abbrs:
143
+ print(f"Duplicate abbreviation '{abbr}' found in '{acronym}'")
144
+ has_duplicate = True
145
+ else:
146
+ seen_abbrs.add(abbr)
147
+
148
+ # Check for duplicate full forms
149
+ fulls_lower = set([full.lower() for full in values.get(self.names_full, [])])
150
+ for full in fulls_lower:
151
+ if full in seen_fulls:
152
+ print(f"Duplicate full form '{full}' found in '{acronym}'")
153
+ has_duplicate = True
154
+ else:
155
+ seen_fulls.add(full)
156
+
157
+ if not has_duplicate:
158
+ valid_data[acronym] = values
159
+ else:
160
+ all_unique = False
161
+
162
+ return valid_data, all_unique
163
+
164
+ def _check_matches(self, data, key_type: str):
165
+ """Check for exact matches in abbreviations or full forms between different acronyms."""
166
+ valid_data = {}
167
+ no_matches = True
168
+ acronyms = sorted(list(data.keys()))
169
+
170
+ for i, main_acronym in enumerate(acronyms):
171
+ # Normalize items: lowercase and remove parentheses
172
+ main_items = [
173
+ item.lower().replace("(", "").replace(")", "")
174
+ for item in data[main_acronym].get(key_type, [])
175
+ ]
176
+
177
+ # Create exact match patterns
178
+ patterns = [re.compile(f"^{item}$") for item in main_items]
179
+
180
+ matches_found = []
181
+
182
+ # Compare with other acronyms
183
+ for other_acronym in acronyms[i + 1:]:
184
+ other_items = [
185
+ item.lower().replace("(", "").replace(")", "")
186
+ for item in data[other_acronym].get(key_type, [])
187
+ ]
188
+
189
+ # Find matching items
190
+ matching_items = [
191
+ item for item in other_items
192
+ if any(pattern.match(item) for pattern in patterns)
193
+ ]
194
+
195
+ if matching_items:
196
+ matches_found.append([main_acronym, other_acronym, matching_items])
197
+
198
+ if matches_found:
199
+ no_matches = False
200
+ print(f"Found matches in {key_type}: {matches_found}")
201
+ else:
202
+ valid_data[main_acronym] = data[main_acronym]
203
+
204
+ return valid_data, no_matches
@@ -0,0 +1,92 @@
1
+ [tool.poetry]
2
+ name = "pyformatjson"
3
+ version = "0.2.3"
4
+ description = "pyformatjson"
5
+ license = "GPL-3.0-or-later"
6
+ authors = ["NextAI <nextartifintell@gmail.com>"]
7
+ maintainers = ["NextAI <nextartifintell@gmail.com>"]
8
+ readme = ["README.md"]
9
+ homepage = "https://github.com/Easy-PhD/pyformatjson"
10
+ repository = "https://github.com/Easy-PhD/pyformatjson"
11
+ documentation = "https://github.com/Easy-PhD/pyformatjson"
12
+ keywords = ["Python", "Json"]
13
+ classifiers = ["Topic :: Software Development :: Libraries :: Python Modules"]
14
+
15
+ [tool.poetry.dependencies]
16
+ python = ">=3.12"
17
+
18
+ [tool.poetry.group.dev.dependencies]
19
+ sphinx = "^8.2.3"
20
+ sphinx-rtd-theme = "^3.0.2"
21
+ sphinx-copybutton = "^0.5.2"
22
+
23
+ [tool.pyright]
24
+ venvPath = "."
25
+ venv = ".venv"
26
+ reportOptionalSubscript = "none" # "none", "information", "error"
27
+ reportIncompatibleMethodOverride = "none" # "none", "information", "error"
28
+
29
+ [tool.ruff]
30
+ line-length = 120
31
+ indent-width = 4
32
+ exclude = [".venv"]
33
+ extend-exclude = ["tests"]
34
+
35
+ [tool.ruff.lint]
36
+ extend-select = [
37
+ "E", # pycodestyle errors
38
+ "W", # pycodestyle warnings
39
+ "F", # Pyflakes
40
+ "I", # isort
41
+ "D", # pydocstyle (docstring conventions)
42
+ "N", # pep8-naming
43
+ "UP", # pyupgrade
44
+ "YTT", # flake8-2020
45
+ "ANN", # flake8-annotations
46
+ "S", # flake8-bandit
47
+ "B", # flake8-bugbear
48
+ "A", # flake8-builtins
49
+ "C4", # flake8-comprehensions
50
+ "RUF", # Ruff-specific rules
51
+ ]
52
+
53
+ ignore = [
54
+ "ANN001", # Missing type annotation of function argument
55
+ "ANN002", # Missing type annotation for `*args`
56
+ "ANN003", # Missing type annotation for `**kwargs`
57
+ "ANN201", # Missing return type of public function
58
+ "ANN202", # Missing return type of private function
59
+ "ANN204", # Missing return type annotation for special method `__init__`
60
+ "ANN205", # Missing return type annotation for staticmethod
61
+ "D102", # Missing docstring in public method
62
+ "D103", # Missing docstring in public function
63
+ "D107", # Missing docstring in `__init__`
64
+ "D204", # 1 blank line required after class docstring
65
+ "D417", # Missing argument description in the docstring
66
+ "RUF002", # Docstring contains ambiguous `–` (EN-DASH)
67
+ ]
68
+
69
+ [tool.ruff.lint.pydocstyle]
70
+ convention = "google"
71
+
72
+ [tool.ruff.format]
73
+ quote-style = "double"
74
+ indent-style = "space"
75
+
76
+ [tool.mypy]
77
+ ignore_missing_imports = true
78
+
79
+ [tool.black]
80
+ line-length = 120
81
+ skip-magic-trailing-comma = true
82
+
83
+ [tool.isort]
84
+ line_length = 120
85
+ profile = "black"
86
+ include_trailing_comma = true
87
+
88
+ [tool.pytest]
89
+
90
+ [build-system]
91
+ requires = ["poetry-core>=1.0.0"]
92
+ build-backend = "poetry.core.masonry.api"
@@ -1,115 +0,0 @@
1
- # coding=utf-8
2
-
3
- import json
4
- import os
5
- from typing import Any, Dict
6
-
7
- from ._base import split_data_list, split_text_by_length
8
-
9
-
10
- def load_json_data(path_json: str, filename: str) -> Dict:
11
- """Load JSON data from a specified file.
12
-
13
- This function attempts to load JSON data from a file located in the specified
14
- directory. If the file doesn't exist or there's an error loading it, an empty
15
- dictionary is returned.
16
-
17
- Args:
18
- path_json (str): Directory path containing the JSON file.
19
- filename (str): Name of the JSON file (with .json extension).
20
-
21
- Returns:
22
- Dict: The loaded JSON data as a dictionary, or empty dict if file not found
23
- or error occurs.
24
-
25
- Example:
26
- >>> load_json_data("/data", "conferences")
27
- {"publisher1": {"conferences": {...}}}
28
- """
29
- try:
30
- file_path = os.path.join(path_json, filename)
31
- if not os.path.exists(file_path):
32
- return {}
33
-
34
- with open(file_path, "r", encoding="utf-8") as file:
35
- return json.load(file)
36
-
37
- except Exception as e:
38
- print(f"Error loading {filename}: {e}")
39
- return {}
40
-
41
-
42
- def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
43
- """Update and format JSON file containing conference/journal data.
44
-
45
- This function loads JSON data, processes and formats text fields by splitting
46
- long text into appropriate lengths, checks for duplicate abbreviations, and
47
- saves the updated data back to the file.
48
-
49
- Args:
50
- full_json_cj (str): Full path to the conferences/journals JSON file
51
- conferences_or_journals (str): Type of publication ('conferences' or 'journals').
52
-
53
- Returns:
54
- Dict[str, Any]: Processed JSON data dictionary.
55
- """
56
- # Load Json Data
57
- json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
58
-
59
- # Process and format text fields in JSON data.
60
- for pub in json_dict:
61
- for flag in ["txt_abouts", "txt_remarks"]:
62
- data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
63
- temps = []
64
- for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
65
- temps.extend(split_text_by_length(line, 105))
66
- if temps:
67
- json_dict[pub].update({flag: temps})
68
-
69
- for abbr in json_dict[pub][conferences_or_journals]:
70
- for flag in ["txt_abouts", "txt_remarks"]:
71
- data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
72
- temps = []
73
- for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
74
- temps.extend(split_text_by_length(line, 97))
75
- if temps:
76
- json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
77
-
78
- # Check for duplicate abbreviations
79
- _check_duplicate_abbr(json_dict, conferences_or_journals)
80
-
81
- # Save updated JSON
82
- if json_dict:
83
- with open(full_json_cj, "w", encoding="utf-8") as f:
84
- f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
85
-
86
- return json_dict
87
-
88
-
89
- def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
90
- """Check for duplicate abbreviations in the data.
91
-
92
- This function validates that there are no duplicate abbreviations within
93
- the same publication type across all publishers.
94
-
95
- Args:
96
- json_dict (Dict[str, Any]): JSON data dictionary containing publication information.
97
- conferences_or_journals (str): Type of publication ('conferences' or 'journals').
98
-
99
- Raises:
100
- ValueError: If duplicate abbreviations are found in the data.
101
-
102
- Example:
103
- >>> _check_duplicate_abbr(data, "conferences")
104
- # Raises ValueError if "ICML" appears twice in conferences
105
- """
106
- abbr_list = []
107
-
108
- for pub in json_dict:
109
- if conferences_or_journals in json_dict[pub]:
110
- for abbr in json_dict[pub][conferences_or_journals]:
111
- if abbr in abbr_list:
112
- raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
113
- abbr_list.append(abbr)
114
-
115
- return None
@@ -1,65 +0,0 @@
1
- [tool.poetry]
2
- name = "pyformatjson"
3
- version = "0.2.1"
4
- description = "pyformatjson"
5
- license = "GPL-3.0-or-later"
6
- authors = ["NextAI <nextartifintell@gmail.com>"]
7
- maintainers = ["NextAI <nextartifintell@gmail.com>"]
8
- readme = ["README.md"]
9
- homepage = "https://github.com/Easy-PhD/pyformatjson"
10
- repository = "https://github.com/Easy-PhD/pyformatjson"
11
- documentation = "https://github.com/Easy-PhD/pyformatjson"
12
- keywords = ["Python", "Json"]
13
- classifiers = ["Topic :: Software Development :: Libraries :: Python Modules"]
14
-
15
- [tool.poetry.dependencies]
16
- python = ">=3.12"
17
-
18
- [tool.poetry.group.dev.dependencies]
19
- mypy = "^1.18.2"
20
- ruff = "^0.13.1"
21
- pycodestyle = "^2.14.0"
22
- pydocstyle = "^6.3.0"
23
- flake8 = "^7.3.0"
24
- isort = "^6.0.1"
25
- black = "^25.9.0"
26
- pyright = "^1.1.405"
27
- sphinx = "^8.2.3"
28
- sphinx-rtd-theme = "^3.0.2"
29
- sphinx-copybutton = "^0.5.2"
30
-
31
- [tool.pyright]
32
- venvPath = "."
33
- venv = ".venv"
34
-
35
- [tool.ruff]
36
- line-length = 120
37
- indent-width = 4
38
-
39
- exclude = [".venv"]
40
- extend-exclude = ["tests"]
41
-
42
- [tool.ruff.lint]
43
- extend-select = ["I"]
44
-
45
- [tool.ruff.format]
46
- quote-style = "double"
47
- indent-style = "space"
48
-
49
- [tool.mypy]
50
- ignore_missing_imports = true
51
-
52
- [tool.black]
53
- line-length = 120
54
- skip-magic-trailing-comma = true
55
-
56
- [tool.isort]
57
- line_length = 120
58
- profile = "black"
59
- include_trailing_comma = true
60
-
61
- [tool.pytest]
62
-
63
- [build-system]
64
- requires = ["poetry-core>=1.0.0"]
65
- build-backend = "poetry.core.masonry.api"
File without changes
File without changes