pyformatjson 0.0.1__tar.gz → 0.0.2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyformatjson
3
- Version: 0.0.1
3
+ Version: 0.0.2
4
4
  Summary: pyformatjson
5
5
  License: GPL-3.0-or-later
6
6
  Keywords: Python,json
@@ -8,9 +8,14 @@ Author: NextAI
8
8
  Author-email: nextartifintell@gmail.com
9
9
  Maintainer: NextAI
10
10
  Maintainer-email: nextartifintell@gmail.com
11
- Requires-Python: >=3.13
11
+ Requires-Python: >=3.8
12
12
  Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
13
13
  Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.8
15
+ Classifier: Programming Language :: Python :: 3.9
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
14
19
  Classifier: Programming Language :: Python :: 3.13
15
20
  Classifier: Programming Language :: Python :: 3.14
16
21
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
@@ -0,0 +1,17 @@
1
+ """PyFormatJSON: A Python library for formatting and processing JSON data for conferences and journals.
2
+
3
+ This package provides utilities for processing conference and journal data from JSON files,
4
+ generating markdown documentation, and creating various formatted outputs for academic
5
+ publication management.
6
+
7
+ Modules:
8
+ core: Core functionality for data processing and JSON manipulation
9
+ tools: Utility tools for data generation and markdown writing
10
+
11
+ Classes:
12
+ GenerateDataDict: Generates data dictionaries from JSON input
13
+ WriteDataToMd: Writes publication data to Markdown files
14
+
15
+ Functions:
16
+ main_generate_md_files: Main function to generate markdown files for conferences and journals
17
+ """
@@ -0,0 +1,223 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ import re
5
+ from typing import List
6
+
7
+
8
+ def split_text_by_length(text, max_length=120) -> List[str]:
9
+ """Split text into lines of specified maximum length.
10
+
11
+ This function breaks long text into multiple lines, ensuring each line
12
+ does not exceed the specified maximum length. It attempts to break at
13
+ word boundaries when possible.
14
+
15
+ Args:
16
+ text (str): The input text to be split into lines.
17
+ max_length (int, optional): Maximum length for each line. Defaults to 120.
18
+
19
+ Returns:
20
+ List[str]: A list of text lines, each not exceeding max_length characters.
21
+
22
+ Example:
23
+ >>> split_text_by_length("This is a very long text that needs to be split", 20)
24
+ ['This is a very long', 'text that needs to be', 'split']
25
+ """
26
+ lines = []
27
+ while text:
28
+ if len(text) <= max_length:
29
+ lines.append(text)
30
+ break
31
+
32
+ split_pos = text.rfind(" ", 0, max_length + 1)
33
+ if split_pos == -1:
34
+ split_pos = max_length
35
+
36
+ line = text[:split_pos]
37
+ lines.append(line)
38
+
39
+ text = text[split_pos:]
40
+
41
+ new_lines = []
42
+ for line in lines:
43
+ new_lines.append(line)
44
+ return new_lines
45
+
46
+
47
+ def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "next") -> List[str]:
48
+ """Split data list according to the split pattern.
49
+
50
+ This function splits each string in the data list using the provided regex pattern
51
+ and reconstructs the data based on the last_next parameter. The pattern must use
52
+ capturing parentheses to define split points.
53
+
54
+ Args:
55
+ split_pattern (str): Regular expression pattern for splitting. Must use capturing
56
+ parentheses, e.g., r"(\n)" for newline splits.
57
+ data_list (List[str]): List of strings to be split and processed.
58
+ last_next (str, optional): Determines how to handle split parts. "next" places
59
+ the split character at the beginning of the next part, "last" places it at
60
+ the end of the current part. Defaults to "next".
61
+
62
+ Returns:
63
+ List[str]: New list of processed strings with empty strings filtered out.
64
+
65
+ Raises:
66
+ re.error: If the split_pattern is not a valid regular expression.
67
+
68
+ Example:
69
+ >>> split_data_list(r"(\n)", ["line1\nline2", "line3\nline4"], "next")
70
+ ['line1', 'line2', 'line3', 'line4']
71
+ """
72
+ new_data_list = []
73
+ for line in data_list:
74
+ split_list = re.split(split_pattern, line)
75
+ list_one = split_list[0 : len(split_list) : 2]
76
+ list_two = split_list[1 : len(split_list) : 2]
77
+
78
+ temp = []
79
+ if last_next == "next":
80
+ list_two.insert(0, "")
81
+ temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
82
+ if last_next == "last":
83
+ list_two.append("")
84
+ temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
85
+ new_data_list.extend(temp)
86
+ new_data_list = [line for line in new_data_list if line.strip()]
87
+ return new_data_list
88
+
89
+
90
+ def standardize_path(path_input: str) -> str:
91
+ """Standardize and ensure a directory path exists.
92
+
93
+ This function expands environment variables and user home directory references
94
+ in the path, then creates the directory if it doesn't exist.
95
+
96
+ Args:
97
+ path_input (str): The input path to be standardized and created.
98
+
99
+ Returns:
100
+ str: The standardized absolute path.
101
+
102
+ Example:
103
+ >>> standardize_path("~/Documents/data")
104
+ '/Users/username/Documents/data'
105
+ """
106
+ path_input = os.path.expandvars(os.path.expanduser(path_input))
107
+ if not os.path.exists(path_input):
108
+ os.makedirs(path_input)
109
+ return path_input
110
+
111
+
112
+ def sort_strings_with_embedded_numbers(s: str) -> List[str]:
113
+ """Split string into pieces for natural sorting with embedded numbers.
114
+
115
+ This function splits a string into pieces where numbers are converted to integers
116
+ for proper natural sorting (e.g., "item2" comes before "item10").
117
+
118
+ Args:
119
+ s (str): The string to be split into sortable pieces.
120
+
121
+ Returns:
122
+ List[str]: List of string pieces with numbers converted to integers.
123
+
124
+ Example:
125
+ >>> sort_strings_with_embedded_numbers("item10")
126
+ ['item', 10]
127
+ """
128
+ re_digits = re.compile(r"(\d+)")
129
+ pieces = re_digits.split(s)
130
+ pieces[1::2] = map(int, pieces[1::2])
131
+ return pieces
132
+
133
+
134
+ def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
135
+ """Sort list of strings with embedded numbers naturally.
136
+
137
+ This function sorts a list of strings using natural sorting that handles
138
+ embedded numbers correctly (e.g., "item2" comes before "item10").
139
+
140
+ Args:
141
+ str_int (List[str]): List of strings to be sorted.
142
+ reverse (bool, optional): If True, sorts in descending order. Defaults to False.
143
+
144
+ Returns:
145
+ List[str]: Sorted list of strings.
146
+
147
+ Example:
148
+ >>> sort_int_str(["item10", "item2", "item1"])
149
+ ['item1', 'item2', 'item10']
150
+ """
151
+ return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
152
+
153
+
154
+ class IterateSortDict(object):
155
+ """A class for recursively sorting dictionary keys with natural sorting.
156
+
157
+ This class provides methods to sort dictionary keys recursively, handling
158
+ nested dictionaries and using natural sorting for strings with embedded numbers.
159
+
160
+ Attributes:
161
+ reverse (bool): If True, sorts keys in descending order. Defaults to False.
162
+
163
+ Example:
164
+ >>> sorter = IterateSortDict(reverse=False)
165
+ >>> data = {"item10": {"sub2": 1, "sub1": 2}, "item2": 3}
166
+ >>> sorted_data = sorter.dict_update(data)
167
+ """
168
+
169
+ def __init__(self, reverse: bool = False) -> None:
170
+ """Initialize the IterateSortDict instance.
171
+
172
+ Args:
173
+ reverse (bool, optional): If True, sorts keys in descending order.
174
+ Defaults to False.
175
+ """
176
+ self.reverse = reverse
177
+
178
+ def dict_update(self, old):
179
+ """Update and sort a dictionary recursively.
180
+
181
+ This method sorts the dictionary keys and recursively processes
182
+ any nested dictionaries.
183
+
184
+ Args:
185
+ old (dict): The dictionary to be sorted and updated.
186
+
187
+ Returns:
188
+ dict: The updated dictionary with sorted keys at all levels.
189
+ """
190
+ old = self.dict_sort_iteration(old)
191
+ old = self.dict_sort(old)
192
+ return old
193
+
194
+ def dict_sort_iteration(self, old: dict):
195
+ """Recursively sort nested dictionaries.
196
+
197
+ This method iterates through the dictionary and recursively sorts
198
+ any nested dictionary values.
199
+
200
+ Args:
201
+ old (dict): The dictionary to be processed recursively.
202
+
203
+ Returns:
204
+ dict: The dictionary with nested dictionaries sorted.
205
+ """
206
+ for key in old:
207
+ if isinstance(old[key], dict):
208
+ old[key] = self.dict_update(old[key])
209
+ return old
210
+
211
+ def dict_sort(self, old: dict):
212
+ """Sort dictionary keys using natural sorting.
213
+
214
+ This method sorts the top-level keys of the dictionary using
215
+ natural sorting that handles embedded numbers.
216
+
217
+ Args:
218
+ old (dict): The dictionary whose keys are to be sorted.
219
+
220
+ Returns:
221
+ dict: A new dictionary with sorted keys.
222
+ """
223
+ return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
@@ -8,7 +8,24 @@ from ._base import split_data_list, split_text_by_length
8
8
 
9
9
 
10
10
  def load_json_data(path_json: str, filename: str) -> Dict:
11
- """Load JSON data from file."""
11
+ """Load JSON data from a specified file.
12
+
13
+ This function attempts to load JSON data from a file located in the specified
14
+ directory. If the file doesn't exist or there's an error loading it, an empty
15
+ dictionary is returned.
16
+
17
+ Args:
18
+ path_json (str): Directory path containing the JSON file.
19
+ filename (str): Name of the JSON file (without .json extension).
20
+
21
+ Returns:
22
+ Dict: The loaded JSON data as a dictionary, or empty dict if file not found
23
+ or error occurs.
24
+
25
+ Example:
26
+ >>> load_json_data("/data", "conferences")
27
+ {"publisher1": {"conferences": {...}}}
28
+ """
12
29
  try:
13
30
  file_path = os.path.join(path_json, f"{filename}.json")
14
31
  if not os.path.exists(file_path):
@@ -23,15 +40,25 @@ def load_json_data(path_json: str, filename: str) -> Dict:
23
40
 
24
41
 
25
42
  def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
26
- """
27
- Update and format JSON file containing conference/journal data.
43
+ """Update and format JSON file containing conference/journal data.
44
+
45
+ This function loads JSON data, processes and formats text fields by splitting
46
+ long text into appropriate lengths, checks for duplicate abbreviations, and
47
+ saves the updated data back to the file.
28
48
 
29
49
  Args:
30
- path_root (str): Root directory path
31
- conferences_or_journals (str): Type of publication ('conferences' or 'journals')
50
+ path_root (str): Root directory path containing the JSON file.
51
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
32
52
 
33
53
  Returns:
34
- Dict[str, Any]: Processed JSON data
54
+ Dict[str, Any]: Processed JSON data dictionary.
55
+
56
+ Raises:
57
+ ValueError: If duplicate abbreviations are found in the data.
58
+
59
+ Example:
60
+ >>> update_json_file("/data", "conferences")
61
+ {"publisher1": {"conferences": {"conf1": {...}}}}
35
62
  """
36
63
  # Load Json Data
37
64
  json_dict = load_json_data(path_root, conferences_or_journals)
@@ -68,15 +95,21 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
68
95
 
69
96
 
70
97
  def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
71
- """
72
- Check for duplicate abbreviations in the data.
98
+ """Check for duplicate abbreviations in the data.
99
+
100
+ This function validates that there are no duplicate abbreviations within
101
+ the same publication type across all publishers.
73
102
 
74
103
  Args:
75
- json_dict: JSON data dictionary
76
- conferences_or_journals (str): Type of publication ('conferences' or 'journals')
104
+ json_dict (Dict[str, Any]): JSON data dictionary containing publication information.
105
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
77
106
 
78
107
  Raises:
79
- ValueError: If duplicate abbreviations are found
108
+ ValueError: If duplicate abbreviations are found in the data.
109
+
110
+ Example:
111
+ >>> _check_duplicate_abbr(data, "conferences")
112
+ # Raises ValueError if "ICML" appears twice in conferences
80
113
  """
81
114
  abbr_list = []
82
115
 
@@ -0,0 +1,155 @@
1
+ # coding=utf-8
2
+
3
+ import json
4
+ import os
5
+ from typing import Optional
6
+
7
+ from .core._base import standardize_path
8
+ from .core.update_json import load_json_data, update_json_file
9
+ from .tools.generate_dict import GenerateDataDict
10
+ from .tools.write_dict import WriteDataToMd
11
+
12
+
13
+ def main_generate_md_files(
14
+ path_json: str,
15
+ path_output_md: str,
16
+ path_output_simplified_json: str,
17
+ path_spidered_bibs: Optional[str] = None,
18
+ for_vue: bool = True,
19
+ conferences_or_journals: Optional[str] = None,
20
+ keywords_category_name: str = "",
21
+ ) -> None:
22
+ """Generate markdown files for conferences and journals.
23
+
24
+ This function processes JSON data containing conference and journal information,
25
+ generates various markdown documentation files, and creates simplified JSON outputs.
26
+ It supports both conference and journal processing with customizable keyword categories.
27
+
28
+ Args:
29
+ path_json (str): Path to the input JSON data file containing publication information.
30
+ path_output_md (str): Output directory path where markdown files will be saved.
31
+ path_output_simplified_json (str): Output directory path for simplified JSON files.
32
+ path_spidered_bibs (Optional[str], optional): Directory containing crawled BibTeX files.
33
+ Defaults to None.
34
+ for_vue (bool, optional): Whether to generate Vue.js-compatible format for date calculations.
35
+ Defaults to True.
36
+ conferences_or_journals (Optional[str], optional): Specify 'conferences' or 'journals' to
37
+ process only one type, or None to process both. Defaults to None.
38
+ keywords_category_name (str, optional): The category name for keywords filtering.
39
+ Defaults to "".
40
+
41
+ Returns:
42
+ None: This function does not return a value.
43
+
44
+ Raises:
45
+ FileNotFoundError: If the input JSON file or required directories are not found.
46
+ ValueError: If there are issues with the data format or duplicate abbreviations.
47
+
48
+ Example:
49
+ >>> main_generate_md_files(
50
+ ... path_json="/data/publications.json",
51
+ ... path_output_md="/output/markdown",
52
+ ... path_output_simplified_json="/output/json",
53
+ ... for_vue=True,
54
+ ... conferences_or_journals="conferences"
55
+ ... )
56
+ """
57
+ # Standardize all paths
58
+ path_json = standardize_path(path_json)
59
+ path_output_md = standardize_path(path_output_md)
60
+ path_output_simplified_json = standardize_path(path_output_simplified_json)
61
+
62
+ path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
63
+
64
+ # Process keyword category name and load data
65
+ keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
66
+ category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
67
+ keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
68
+
69
+ # Validate data availability
70
+ if not keywords_list or not keywords_category_name:
71
+ keywords_list, keywords_category_name = [], ""
72
+
73
+ # Process both conferences and journals
74
+ for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
75
+ # Skip if specific type requested and doesn't match
76
+ if conferences_or_journals and conferences_or_journals.lower() != cj:
77
+ continue
78
+
79
+ # Update JSON data
80
+ json_dict = update_json_file(path_json, cj)
81
+ if not json_dict:
82
+ continue
83
+
84
+ # Simplify JSON data
85
+ simplify_json(json_dict, cj, path_output_simplified_json)
86
+
87
+ # Generate data dictionaries
88
+ path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
89
+ generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
90
+ publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
91
+ if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
92
+ continue
93
+
94
+ # Initialize writer and save all markdown files
95
+ _path_output = os.path.join(path_output_md, f"{cj.title()}")
96
+ save_data = WriteDataToMd(
97
+ cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
98
+ )
99
+ # Save various documentation files
100
+ save_data.save_introductions()
101
+ save_data.save_categories(keywords_category_name, keywords_list)
102
+ save_data.save_categories_separate_keywords()
103
+
104
+ save_data.save_publishers()
105
+ save_data.save_publishers_separate_abbrs()
106
+
107
+ save_data.save_statistics(keywords_category_name, keywords_list)
108
+ save_data.save_statistics_separate_abbrs()
109
+
110
+ return None
111
+
112
+
113
+ def simplify_json(json_dict, cj: str, output_dir: str) -> None:
114
+ """Simplify JSON dictionary by extracting only essential fields.
115
+
116
+ This function creates a simplified version of the JSON dictionary containing
117
+ only the names_abbr and names_full fields for each publisher and publication type.
118
+ The simplified data is saved to a new JSON file in the specified output directory.
119
+
120
+ Args:
121
+ json_dict (dict): The original JSON dictionary containing publication data.
122
+ cj (str): The type of publication, either 'conferences' or 'journals'.
123
+ output_dir (str): Directory path where the simplified JSON file will be saved.
124
+
125
+ Returns:
126
+ None: This function does not return a value.
127
+
128
+ Note:
129
+ The function creates a new JSON file named '{cj}.json' in the output directory.
130
+ Only the 'names_abbr' and 'names_full' fields are preserved in the simplified version.
131
+
132
+ Example:
133
+ >>> simplify_json(
134
+ ... json_dict=publication_data,
135
+ ... cj="conferences",
136
+ ... output_dir="/output/simplified"
137
+ ... )
138
+ """
139
+ new_json_dict = {}
140
+ for publisher in json_dict:
141
+ for abbr in json_dict[publisher][cj.lower()]:
142
+ names_abbr = json_dict[publisher][cj.lower()][abbr].get("names_abbr", [])
143
+ names_full = json_dict[publisher][cj.lower()][abbr].get("names_full", [])
144
+
145
+ new_json_dict.setdefault(publisher, {}).setdefault(cj.lower(), {}).setdefault(abbr, {}).update(
146
+ {"names_abbr": names_abbr, "names_full": names_full}
147
+ )
148
+
149
+ # Save updated JSON
150
+ if new_json_dict:
151
+ path_file = os.path.join(output_dir, f"{cj}.json")
152
+ with open(path_file, "w", encoding="utf-8") as f:
153
+ f.write(json.dumps(new_json_dict, indent=4, sort_keys=True, ensure_ascii=True))
154
+
155
+ return None