pyformatjson 0.0.1__tar.gz → 0.0.3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyformatjson
3
- Version: 0.0.1
3
+ Version: 0.0.3
4
4
  Summary: pyformatjson
5
5
  License: GPL-3.0-or-later
6
6
  Keywords: Python,json
@@ -0,0 +1,17 @@
1
+ """PyFormatJSON: A Python library for formatting and processing JSON data for conferences and journals.
2
+
3
+ This package provides utilities for processing conference and journal data from JSON files,
4
+ generating markdown documentation, and creating various formatted outputs for academic
5
+ publication management.
6
+
7
+ Modules:
8
+ core: Core functionality for data processing and JSON manipulation
9
+ tools: Utility tools for data generation and markdown writing
10
+
11
+ Classes:
12
+ GenerateDataDict: Generates data dictionaries from JSON input
13
+ WriteDataToMd: Writes publication data to Markdown files
14
+
15
+ Functions:
16
+ main_generate_md_files: Main function to generate markdown files for conferences and journals
17
+ """
@@ -0,0 +1,223 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ import re
5
+ from typing import List
6
+
7
+
8
+ def split_text_by_length(text, max_length=120) -> List[str]:
9
+ """Split text into lines of specified maximum length.
10
+
11
+ This function breaks long text into multiple lines, ensuring each line
12
+ does not exceed the specified maximum length. It attempts to break at
13
+ word boundaries when possible.
14
+
15
+ Args:
16
+ text (str): The input text to be split into lines.
17
+ max_length (int, optional): Maximum length for each line. Defaults to 120.
18
+
19
+ Returns:
20
+ List[str]: A list of text lines, each not exceeding max_length characters.
21
+
22
+ Example:
23
+ >>> split_text_by_length("This is a very long text that needs to be split", 20)
24
+ ['This is a very long', 'text that needs to be', 'split']
25
+ """
26
+ lines = []
27
+ while text:
28
+ if len(text) <= max_length:
29
+ lines.append(text)
30
+ break
31
+
32
+ split_pos = text.rfind(" ", 0, max_length + 1)
33
+ if split_pos == -1:
34
+ split_pos = max_length
35
+
36
+ line = text[:split_pos]
37
+ lines.append(line)
38
+
39
+ text = text[split_pos:]
40
+
41
+ new_lines = []
42
+ for line in lines:
43
+ new_lines.append(line)
44
+ return new_lines
45
+
46
+
47
+ def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "next") -> List[str]:
48
+ """Split data list according to the split pattern.
49
+
50
+ This function splits each string in the data list using the provided regex pattern
51
+ and reconstructs the data based on the last_next parameter. The pattern must use
52
+ capturing parentheses to define split points.
53
+
54
+ Args:
55
+ split_pattern (str): Regular expression pattern for splitting. Must use capturing
56
+ parentheses, e.g., r"(\n)" for newline splits.
57
+ data_list (List[str]): List of strings to be split and processed.
58
+ last_next (str, optional): Determines how to handle split parts. "next" places
59
+ the split character at the beginning of the next part, "last" places it at
60
+ the end of the current part. Defaults to "next".
61
+
62
+ Returns:
63
+ List[str]: New list of processed strings with empty strings filtered out.
64
+
65
+ Raises:
66
+ re.error: If the split_pattern is not a valid regular expression.
67
+
68
+ Example:
69
+ >>> split_data_list(r"(\n)", ["line1\nline2", "line3\nline4"], "next")
70
+ ['line1', 'line2', 'line3', 'line4']
71
+ """
72
+ new_data_list = []
73
+ for line in data_list:
74
+ split_list = re.split(split_pattern, line)
75
+ list_one = split_list[0 : len(split_list) : 2]
76
+ list_two = split_list[1 : len(split_list) : 2]
77
+
78
+ temp = []
79
+ if last_next == "next":
80
+ list_two.insert(0, "")
81
+ temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
82
+ if last_next == "last":
83
+ list_two.append("")
84
+ temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
85
+ new_data_list.extend(temp)
86
+ new_data_list = [line for line in new_data_list if line.strip()]
87
+ return new_data_list
88
+
89
+
90
+ def standardize_path(path_input: str) -> str:
91
+ """Standardize and ensure a directory path exists.
92
+
93
+ This function expands environment variables and user home directory references
94
+ in the path, then creates the directory if it doesn't exist.
95
+
96
+ Args:
97
+ path_input (str): The input path to be standardized and created.
98
+
99
+ Returns:
100
+ str: The standardized absolute path.
101
+
102
+ Example:
103
+ >>> standardize_path("~/Documents/data")
104
+ '/Users/username/Documents/data'
105
+ """
106
+ path_input = os.path.expandvars(os.path.expanduser(path_input))
107
+ if not os.path.exists(path_input):
108
+ os.makedirs(path_input)
109
+ return path_input
110
+
111
+
112
+ def sort_strings_with_embedded_numbers(s: str) -> List[str]:
113
+ """Split string into pieces for natural sorting with embedded numbers.
114
+
115
+ This function splits a string into pieces where numbers are converted to integers
116
+ for proper natural sorting (e.g., "item2" comes before "item10").
117
+
118
+ Args:
119
+ s (str): The string to be split into sortable pieces.
120
+
121
+ Returns:
122
+ List[str]: List of string pieces with numbers converted to integers.
123
+
124
+ Example:
125
+ >>> sort_strings_with_embedded_numbers("item10")
126
+ ['item', 10]
127
+ """
128
+ re_digits = re.compile(r"(\d+)")
129
+ pieces = re_digits.split(s)
130
+ pieces[1::2] = map(int, pieces[1::2])
131
+ return pieces
132
+
133
+
134
+ def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
135
+ """Sort list of strings with embedded numbers naturally.
136
+
137
+ This function sorts a list of strings using natural sorting that handles
138
+ embedded numbers correctly (e.g., "item2" comes before "item10").
139
+
140
+ Args:
141
+ str_int (List[str]): List of strings to be sorted.
142
+ reverse (bool, optional): If True, sorts in descending order. Defaults to False.
143
+
144
+ Returns:
145
+ List[str]: Sorted list of strings.
146
+
147
+ Example:
148
+ >>> sort_int_str(["item10", "item2", "item1"])
149
+ ['item1', 'item2', 'item10']
150
+ """
151
+ return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
152
+
153
+
154
+ class IterateSortDict(object):
155
+ """A class for recursively sorting dictionary keys with natural sorting.
156
+
157
+ This class provides methods to sort dictionary keys recursively, handling
158
+ nested dictionaries and using natural sorting for strings with embedded numbers.
159
+
160
+ Attributes:
161
+ reverse (bool): If True, sorts keys in descending order. Defaults to False.
162
+
163
+ Example:
164
+ >>> sorter = IterateSortDict(reverse=False)
165
+ >>> data = {"item10": {"sub2": 1, "sub1": 2}, "item2": 3}
166
+ >>> sorted_data = sorter.dict_update(data)
167
+ """
168
+
169
+ def __init__(self, reverse: bool = False) -> None:
170
+ """Initialize the IterateSortDict instance.
171
+
172
+ Args:
173
+ reverse (bool, optional): If True, sorts keys in descending order.
174
+ Defaults to False.
175
+ """
176
+ self.reverse = reverse
177
+
178
+ def dict_update(self, old):
179
+ """Update and sort a dictionary recursively.
180
+
181
+ This method sorts the dictionary keys and recursively processes
182
+ any nested dictionaries.
183
+
184
+ Args:
185
+ old (dict): The dictionary to be sorted and updated.
186
+
187
+ Returns:
188
+ dict: The updated dictionary with sorted keys at all levels.
189
+ """
190
+ old = self.dict_sort_iteration(old)
191
+ old = self.dict_sort(old)
192
+ return old
193
+
194
+ def dict_sort_iteration(self, old: dict):
195
+ """Recursively sort nested dictionaries.
196
+
197
+ This method iterates through the dictionary and recursively sorts
198
+ any nested dictionary values.
199
+
200
+ Args:
201
+ old (dict): The dictionary to be processed recursively.
202
+
203
+ Returns:
204
+ dict: The dictionary with nested dictionaries sorted.
205
+ """
206
+ for key in old:
207
+ if isinstance(old[key], dict):
208
+ old[key] = self.dict_update(old[key])
209
+ return old
210
+
211
+ def dict_sort(self, old: dict):
212
+ """Sort dictionary keys using natural sorting.
213
+
214
+ This method sorts the top-level keys of the dictionary using
215
+ natural sorting that handles embedded numbers.
216
+
217
+ Args:
218
+ old (dict): The dictionary whose keys are to be sorted.
219
+
220
+ Returns:
221
+ dict: A new dictionary with sorted keys.
222
+ """
223
+ return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
@@ -8,7 +8,24 @@ from ._base import split_data_list, split_text_by_length
8
8
 
9
9
 
10
10
  def load_json_data(path_json: str, filename: str) -> Dict:
11
- """Load JSON data from file."""
11
+ """Load JSON data from a specified file.
12
+
13
+ This function attempts to load JSON data from a file located in the specified
14
+ directory. If the file doesn't exist or there's an error loading it, an empty
15
+ dictionary is returned.
16
+
17
+ Args:
18
+ path_json (str): Directory path containing the JSON file.
19
+ filename (str): Name of the JSON file (without .json extension).
20
+
21
+ Returns:
22
+ Dict: The loaded JSON data as a dictionary, or empty dict if file not found
23
+ or error occurs.
24
+
25
+ Example:
26
+ >>> load_json_data("/data", "conferences")
27
+ {"publisher1": {"conferences": {...}}}
28
+ """
12
29
  try:
13
30
  file_path = os.path.join(path_json, f"{filename}.json")
14
31
  if not os.path.exists(file_path):
@@ -23,15 +40,25 @@ def load_json_data(path_json: str, filename: str) -> Dict:
23
40
 
24
41
 
25
42
  def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
26
- """
27
- Update and format JSON file containing conference/journal data.
43
+ """Update and format JSON file containing conference/journal data.
44
+
45
+ This function loads JSON data, processes and formats text fields by splitting
46
+ long text into appropriate lengths, checks for duplicate abbreviations, and
47
+ saves the updated data back to the file.
28
48
 
29
49
  Args:
30
- path_root (str): Root directory path
31
- conferences_or_journals (str): Type of publication ('conferences' or 'journals')
50
+ path_root (str): Root directory path containing the JSON file.
51
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
32
52
 
33
53
  Returns:
34
- Dict[str, Any]: Processed JSON data
54
+ Dict[str, Any]: Processed JSON data dictionary.
55
+
56
+ Raises:
57
+ ValueError: If duplicate abbreviations are found in the data.
58
+
59
+ Example:
60
+ >>> update_json_file("/data", "conferences")
61
+ {"publisher1": {"conferences": {"conf1": {...}}}}
35
62
  """
36
63
  # Load Json Data
37
64
  json_dict = load_json_data(path_root, conferences_or_journals)
@@ -68,15 +95,21 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
68
95
 
69
96
 
70
97
  def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
71
- """
72
- Check for duplicate abbreviations in the data.
98
+ """Check for duplicate abbreviations in the data.
99
+
100
+ This function validates that there are no duplicate abbreviations within
101
+ the same publication type across all publishers.
73
102
 
74
103
  Args:
75
- json_dict: JSON data dictionary
76
- conferences_or_journals (str): Type of publication ('conferences' or 'journals')
104
+ json_dict (Dict[str, Any]): JSON data dictionary containing publication information.
105
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
77
106
 
78
107
  Raises:
79
- ValueError: If duplicate abbreviations are found
108
+ ValueError: If duplicate abbreviations are found in the data.
109
+
110
+ Example:
111
+ >>> _check_duplicate_abbr(data, "conferences")
112
+ # Raises ValueError if "ICML" appears twice in conferences
80
113
  """
81
114
  abbr_list = []
82
115
 
@@ -0,0 +1,155 @@
1
+ # coding=utf-8
2
+
3
+ import json
4
+ import os
5
+ from typing import Optional
6
+
7
+ from .core._base import standardize_path
8
+ from .core.update_json import load_json_data, update_json_file
9
+ from .tools.generate_dict import GenerateDataDict
10
+ from .tools.write_dict import WriteDataToMd
11
+
12
+
13
+ def main_generate_md_files(
14
+ path_json: str,
15
+ path_output_md: str,
16
+ path_output_simplified_json: str,
17
+ path_spidered_bibs: Optional[str] = None,
18
+ for_vue: bool = True,
19
+ conferences_or_journals: Optional[str] = None,
20
+ keywords_category_name: str = "",
21
+ ) -> None:
22
+ """Generate markdown files for conferences and journals.
23
+
24
+ This function processes JSON data containing conference and journal information,
25
+ generates various markdown documentation files, and creates simplified JSON outputs.
26
+ It supports both conference and journal processing with customizable keyword categories.
27
+
28
+ Args:
29
+ path_json (str): Path to the input JSON data file containing publication information.
30
+ path_output_md (str): Output directory path where markdown files will be saved.
31
+ path_output_simplified_json (str): Output directory path for simplified JSON files.
32
+ path_spidered_bibs (Optional[str], optional): Directory containing crawled BibTeX files.
33
+ Defaults to None.
34
+ for_vue (bool, optional): Whether to generate Vue.js-compatible format for date calculations.
35
+ Defaults to True.
36
+ conferences_or_journals (Optional[str], optional): Specify 'conferences' or 'journals' to
37
+ process only one type, or None to process both. Defaults to None.
38
+ keywords_category_name (str, optional): The category name for keywords filtering.
39
+ Defaults to "".
40
+
41
+ Returns:
42
+ None: This function does not return a value.
43
+
44
+ Raises:
45
+ FileNotFoundError: If the input JSON file or required directories are not found.
46
+ ValueError: If there are issues with the data format or duplicate abbreviations.
47
+
48
+ Example:
49
+ >>> main_generate_md_files(
50
+ ... path_json="/data/publications.json",
51
+ ... path_output_md="/output/markdown",
52
+ ... path_output_simplified_json="/output/json",
53
+ ... for_vue=True,
54
+ ... conferences_or_journals="conferences"
55
+ ... )
56
+ """
57
+ # Standardize all paths
58
+ path_json = standardize_path(path_json)
59
+ path_output_md = standardize_path(path_output_md)
60
+ path_output_simplified_json = standardize_path(path_output_simplified_json)
61
+
62
+ path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
63
+
64
+ # Process keyword category name and load data
65
+ keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
66
+ category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
67
+ keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
68
+
69
+ # Validate data availability
70
+ if not keywords_list or not keywords_category_name:
71
+ keywords_list, keywords_category_name = [], ""
72
+
73
+ # Process both conferences and journals
74
+ for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
75
+ # Skip if specific type requested and doesn't match
76
+ if conferences_or_journals and conferences_or_journals.lower() != cj:
77
+ continue
78
+
79
+ # Update JSON data
80
+ json_dict = update_json_file(path_json, cj)
81
+ if not json_dict:
82
+ continue
83
+
84
+ # Simplify JSON data
85
+ simplify_json(json_dict, cj, path_output_simplified_json)
86
+
87
+ # Generate data dictionaries
88
+ path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
89
+ generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
90
+ publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
91
+ if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
92
+ continue
93
+
94
+ # Initialize writer and save all markdown files
95
+ _path_output = os.path.join(path_output_md, f"{cj.title()}")
96
+ save_data = WriteDataToMd(
97
+ cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
98
+ )
99
+ # Save various documentation files
100
+ save_data.save_introductions()
101
+ save_data.save_categories(keywords_category_name, keywords_list)
102
+ save_data.save_categories_separate_keywords()
103
+
104
+ save_data.save_publishers()
105
+ save_data.save_publishers_separate_abbrs()
106
+
107
+ save_data.save_statistics(keywords_category_name, keywords_list)
108
+ save_data.save_statistics_separate_abbrs()
109
+
110
+ return None
111
+
112
+
113
+ def simplify_json(json_dict, cj: str, output_dir: str) -> None:
114
+ """Simplify JSON dictionary by extracting only essential fields.
115
+
116
+ This function creates a simplified version of the JSON dictionary containing
117
+ only the names_abbr and names_full fields for each publisher and publication type.
118
+ The simplified data is saved to a new JSON file in the specified output directory.
119
+
120
+ Args:
121
+ json_dict (dict): The original JSON dictionary containing publication data.
122
+ cj (str): The type of publication, either 'conferences' or 'journals'.
123
+ output_dir (str): Directory path where the simplified JSON file will be saved.
124
+
125
+ Returns:
126
+ None: This function does not return a value.
127
+
128
+ Note:
129
+ The function creates a new JSON file named '{cj}.json' in the output directory.
130
+ Only the 'names_abbr' and 'names_full' fields are preserved in the simplified version.
131
+
132
+ Example:
133
+ >>> simplify_json(
134
+ ... json_dict=publication_data,
135
+ ... cj="conferences",
136
+ ... output_dir="/output/simplified"
137
+ ... )
138
+ """
139
+ new_json_dict = {}
140
+ for publisher in json_dict:
141
+ for abbr in json_dict[publisher][cj.lower()]:
142
+ names_abbr = json_dict[publisher][cj.lower()][abbr].get("names_abbr", [])
143
+ names_full = json_dict[publisher][cj.lower()][abbr].get("names_full", [])
144
+
145
+ new_json_dict.setdefault(publisher, {}).setdefault(cj.lower(), {}).setdefault(abbr, {}).update(
146
+ {"names_abbr": names_abbr, "names_full": names_full}
147
+ )
148
+
149
+ # Save updated JSON
150
+ if new_json_dict:
151
+ path_file = os.path.join(output_dir, f"{cj}.json")
152
+ with open(path_file, "w", encoding="utf-8") as f:
153
+ f.write(json.dumps(new_json_dict, indent=4, sort_keys=True, ensure_ascii=True))
154
+
155
+ return None
@@ -7,6 +7,21 @@ from typing import Optional
7
7
 
8
8
 
9
9
  def conference_journal_header():
10
+ """Generate markdown table headers for conferences and journals.
11
+
12
+ This function creates the appropriate markdown table headers for displaying
13
+ conference and journal information in tabular format.
14
+
15
+ Returns:
16
+ tuple: A tuple containing two lists:
17
+ - conference_header: Markdown table headers for conferences
18
+ - journal_header: Markdown table headers for journals
19
+
20
+ Example:
21
+ >>> conf_header, journal_header = conference_journal_header()
22
+ >>> print(conf_header[0])
23
+ |Publishers|Full/Homepage|Abbr/About|Acronym/Archive|Period/DBLP|...
24
+ """
10
25
  o = "|Publishers|Full/Homepage|Abbr/About|"
11
26
  t = "|- |- |- |"
12
27
  conference_header = [
@@ -21,6 +36,29 @@ def conference_journal_header():
21
36
 
22
37
 
23
38
  class GenerateDataDict(object):
39
+ """Generate data dictionaries from JSON input for conferences and journals.
40
+
41
+ This class processes JSON data containing conference or journal information
42
+ and generates structured dictionaries for markdown table generation, including
43
+ publisher metadata, keyword-based indexing, and Mermaid diagram data.
44
+
45
+ Attributes:
46
+ cj (str): Type of publication ('conferences' or 'journals').
47
+ ia (str): Publication type ('inproceedings' or 'article').
48
+ json_dict (dict): Input JSON data containing publication information.
49
+ path_spidered_cj (Optional[str]): Path to spidered conference/journal data.
50
+ for_vue (bool): Whether to generate Vue.js-compatible format.
51
+
52
+ Example:
53
+ >>> generator = GenerateDataDict(
54
+ ... conferences_or_journals="conferences",
55
+ ... inproceedings_or_article="inproceedings",
56
+ ... json_dict=publication_data,
57
+ ... for_vue=True
58
+ ... )
59
+ >>> publisher_meta, publisher_abbr, keyword_abbr = generator.generate()
60
+ """
61
+
24
62
  def __init__(
25
63
  self,
26
64
  conferences_or_journals: str,
@@ -29,6 +67,17 @@ class GenerateDataDict(object):
29
67
  for_vue: bool = True,
30
68
  path_spidered_conferences_or_journals: Optional[str] = None,
31
69
  ) -> None:
70
+ """Initialize the GenerateDataDict instance.
71
+
72
+ Args:
73
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
74
+ inproceedings_or_article (str): Publication type ('inproceedings' or 'article').
75
+ json_dict (dict): Input JSON data containing publication information.
76
+ for_vue (bool, optional): Whether to generate Vue.js-compatible format.
77
+ Defaults to True.
78
+ path_spidered_conferences_or_journals (Optional[str], optional): Path to
79
+ spidered conference/journal data. Defaults to None.
80
+ """
32
81
  self.cj = conferences_or_journals
33
82
  self.ia = inproceedings_or_article
34
83
  self.json_dict = json_dict
@@ -37,13 +86,22 @@ class GenerateDataDict(object):
37
86
  self.for_vue = for_vue
38
87
 
39
88
  def generate(self):
40
- """
41
- Generate publisher metadata and keyword-based publication information.
89
+ """Generate publisher metadata and keyword-based publication information.
90
+
91
+ This method processes the JSON data to create three main dictionaries:
92
+ 1. Publisher metadata with URLs and descriptions
93
+ 2. Publisher abbreviation metadata with detailed publication info
94
+ 3. Keyword-based metadata for easy searching and categorization
42
95
 
43
96
  Returns:
44
- tuple: Contains two dictionaries:
45
- - publisher_meta_abbr_dict: Publisher metadata and abbreviations
46
- - publication_keyword_row_info_dict: Keyword-indexed publication info
97
+ tuple: A tuple containing three dictionaries:
98
+ - publisher_meta_dict: Publisher metadata including URLs and descriptions
99
+ - publisher_abbr_meta_dict: Publication details indexed by publisher and abbreviation
100
+ - keyword_abbr_meta_dict: Publication details indexed by keywords
101
+
102
+ Example:
103
+ >>> generator = GenerateDataDict(...)
104
+ >>> pub_meta, pub_abbr, keyword_abbr = generator.generate()
47
105
  """
48
106
  publisher_meta_dict, keyword_abbr_meta_dict, publisher_abbr_meta_dict = {}, {}, {}
49
107
 
@@ -76,13 +134,15 @@ class GenerateDataDict(object):
76
134
  remarks = [p for p in self.json_dict[publisher].get("txt_remarks", []) if p.strip()]
77
135
 
78
136
  # Update publisher metadata
79
- publisher_meta_dict.setdefault(publisher, {}).update({
80
- "full_name_url": full_url,
81
- "txt_abouts": abouts,
82
- "txt_remarks": remarks,
83
- "urls_about": urls_about,
84
- "url_conferences_or_journals": f"[{self.cj.title()}]({urls_cj[0]})" if urls_cj else "",
85
- })
137
+ publisher_meta_dict.setdefault(publisher, {}).update(
138
+ {
139
+ "full_name_url": full_url,
140
+ "txt_abouts": abouts,
141
+ "txt_remarks": remarks,
142
+ "urls_about": urls_about,
143
+ "url_conferences_or_journals": f"[{self.cj.title()}]({urls_cj[0]})" if urls_cj else "",
144
+ }
145
+ )
86
146
 
87
147
  # Process each abbreviation (conference/journal)
88
148
  for abbr in self.json_dict[publisher][self.cj]:
@@ -105,17 +165,31 @@ class GenerateDataDict(object):
105
165
  return publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict
106
166
 
107
167
  def conference_or_journal(self, publisher_url: str, abbr: str, abbr_dict: dict):
108
- """
109
- Process conference or journal data and generate formatted information.
168
+ """Process conference or journal data and generate formatted information.
169
+
170
+ This method processes individual conference or journal data, validates
171
+ name lengths, extracts information, formats URLs, and generates table
172
+ row data for markdown output.
110
173
 
111
174
  Args:
112
- publisher_url: Publisher's URL
113
- abbr: Abbreviation identifier
114
- abbr_dict: Dictionary containing publication details
175
+ publisher_url (str): Publisher's URL for markdown linking.
176
+ abbr (str): Abbreviation identifier for the publication.
177
+ abbr_dict (dict): Dictionary containing publication details including
178
+ names, URLs, dates, scores, and keywords.
115
179
 
116
180
  Returns:
117
- dict: Contains formatted about text, remarks, and table row
118
- list: Sorted list of keywords
181
+ tuple: A tuple containing:
182
+ - dict: Contains formatted about text, remarks, and table row data
183
+ - list: Sorted list of keywords for the publication
184
+
185
+ Raises:
186
+ ValueError: If full and abbreviated names have mismatched lengths.
187
+
188
+ Example:
189
+ >>> result = generator.conference_or_journal(
190
+ ... "https://publisher.com", "ICML", conf_data
191
+ ... )
192
+ >>> abouts, keywords = result
119
193
  """
120
194
  # Validate full and abbreviated names match in length
121
195
  self._validate_name_lengths(abbr_dict)
@@ -138,23 +212,42 @@ class GenerateDataDict(object):
138
212
 
139
213
  # Generate appropriate table row based on type
140
214
  row_inf = self._generate_table_row(
141
- publisher_url, full_name, abbr_name, url_home,
142
- url_about, period, top, keywords_url, abbr, abbr_dict
215
+ publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords_url, abbr, abbr_dict
143
216
  )
144
217
 
145
218
  return {"txt_abouts": abouts, "txt_remarks": remarks, "row_inf": row_inf}, keywords
146
219
 
147
220
  def _validate_name_lengths(self, abbr_dict: dict):
148
- """Validate that full and abbreviated names arrays have equal length."""
221
+ """Validate that full and abbreviated names arrays have equal length.
222
+
223
+ This method ensures that the full names and abbreviated names arrays
224
+ have the same length, which is required for proper data processing.
225
+
226
+ Args:
227
+ abbr_dict (dict): Dictionary containing publication data.
228
+
229
+ Raises:
230
+ ValueError: If the lengths of names_full and names_abbr don't match.
231
+ """
149
232
  full_names = abbr_dict.get("names_full", [])
150
233
  abbr_names = abbr_dict.get("names_abbr", [])
151
234
  if len(full_names) != len(abbr_names):
152
- raise ValueError(f"Length mismatch: {len(full_names)} full names vs {len(abbr_names)} abbreviated names")
235
+ raise ValueError(f"Length mismatch: {len(full_names)} {full_names} vs {len(abbr_names)} abbreviated names")
153
236
 
154
237
  def _extract_full_abbr_names(self, abbr_dict: dict):
155
- """Extract full and abbreviated names from dictionary."""
238
+ """Extract full and abbreviated names from dictionary.
239
+
240
+ This method extracts the appropriate full and abbreviated names based on
241
+ the publication type (conferences vs journals).
242
+
243
+ Args:
244
+ abbr_dict (dict): Dictionary containing publication data.
245
+
246
+ Returns:
247
+ tuple: A tuple containing (full_name, abbr_name).
248
+ """
156
249
  # For journals: use first full name from list; for conferences: use single name
157
- full_name = (abbr_dict.get("names_full", [""])[0] if self.cj == "journals" else abbr_dict.get("name", ""))
250
+ full_name = abbr_dict.get("names_full", [""])[0] if self.cj == "journals" else abbr_dict.get("name", "")
158
251
  abbr_name = abbr_dict.get("names_abbr", [""])[0]
159
252
  return full_name, abbr_name
160
253
 
@@ -175,7 +268,7 @@ class GenerateDataDict(object):
175
268
  if acronym_dblp := abbr_dict.get("acronym_dblp", ""):
176
269
  journal_conf = "journals" if self.cj == "journals" else "conf"
177
270
  dblp_url = f"https://dblp.org/db/{journal_conf}/{acronym_dblp}/index.html"
178
- period = f'[{period}]({dblp_url})'
271
+ period = f"[{period}]({dblp_url})"
179
272
 
180
273
  return period
181
274
 
@@ -209,9 +302,7 @@ class GenerateDataDict(object):
209
302
  all_keywords = sorted(set(all_keywords))
210
303
  # Create Google search links for each keyword
211
304
  google_base = "https://www.google.com/search?q="
212
- keywords_url = [
213
- f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords
214
- ]
305
+ keywords_url = [f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords]
215
306
 
216
307
  # For category
217
308
  # Flatten keywords and remove duplicates
@@ -276,19 +367,28 @@ class GenerateDataDict(object):
276
367
  abstract_due, start_date, today = self._process_conference_dates(abbr_dict)
277
368
 
278
369
  # Format date indicators for Vue or standard display
279
- abstract_indicator, start_indicator = self._format_date_indicators(
280
- abstract_due, start_date, today
281
- )
370
+ abstract_indicator, start_indicator = self._format_date_indicators(abstract_due, start_date, today)
282
371
 
283
372
  # Get year URL for start date link
284
373
  year_url = abbr_dict.get("conf_url", "")
285
374
 
286
375
  # Build and return markdown table row
287
376
  return self._build_conference_row(
288
- publisher_url, full_name, abbr_name, url_home, url_about,
289
- archive_display, period, top, abbr_dict, abstract_due,
290
- abstract_indicator, start_date, start_indicator, year_url,
291
- keywords
377
+ publisher_url,
378
+ full_name,
379
+ abbr_name,
380
+ url_home,
381
+ url_about,
382
+ archive_display,
383
+ period,
384
+ top,
385
+ abbr_dict,
386
+ abstract_due,
387
+ abstract_indicator,
388
+ start_date,
389
+ start_indicator,
390
+ year_url,
391
+ keywords,
292
392
  )
293
393
 
294
394
  def _process_conference_dates(self, abbr_dict):
@@ -328,8 +428,22 @@ class GenerateDataDict(object):
328
428
  return abstract_indicator, start_indicator
329
429
 
330
430
  def _build_conference_row(
331
- self, publisher_url, full_name, abbr_name, url_home, url_about, archive_display, period, top, abbr_dict,
332
- abstract_due, abstract_indicator, start_date, start_indicator, year_url, keywords
431
+ self,
432
+ publisher_url,
433
+ full_name,
434
+ abbr_name,
435
+ url_home,
436
+ url_about,
437
+ archive_display,
438
+ period,
439
+ top,
440
+ abbr_dict,
441
+ abstract_due,
442
+ abstract_indicator,
443
+ start_date,
444
+ start_indicator,
445
+ year_url,
446
+ keywords,
333
447
  ):
334
448
  """Construct conference table row string."""
335
449
  # Format date strings
@@ -340,19 +454,21 @@ class GenerateDataDict(object):
340
454
  start_date_display = self._format_link(start_date_str, year_url) if start_date_str else ""
341
455
 
342
456
  # Build table row
343
- return f"|{publisher_url}|" \
344
- f"{self._format_link(full_name, url_home)}|" \
345
- f"{self._format_link(abbr_name, url_about)}|" \
346
- f"{archive_display}|" \
347
- f"{period}|" \
348
- f"{top}|" \
349
- f"{abbr_dict.get('score_ccf', '')}|" \
350
- f"{abstract_date_str}|" \
351
- f"{abstract_indicator}|" \
352
- f"{start_date_display}|" \
353
- f"{start_indicator}|" \
354
- f"{abbr_dict.get('conf_location', '').strip()}|" \
355
- f"{'; '.join(keywords)}|"
457
+ return (
458
+ f"|{publisher_url}|"
459
+ f"{self._format_link(full_name, url_home)}|"
460
+ f"{self._format_link(abbr_name, url_about)}|"
461
+ f"{archive_display}|"
462
+ f"{period}|"
463
+ f"{top}|"
464
+ f"{abbr_dict.get('score_ccf', '')}|"
465
+ f"{abstract_date_str}|"
466
+ f"{abstract_indicator}|"
467
+ f"{start_date_display}|"
468
+ f"{start_indicator}|"
469
+ f"{abbr_dict.get('conf_location', '').strip()}|"
470
+ f"{'; '.join(keywords)}|"
471
+ )
356
472
 
357
473
  # Journals
358
474
  def _generate_for_journal(
@@ -382,25 +498,26 @@ class GenerateDataDict(object):
382
498
 
383
499
  # Build and return markdown table row
384
500
  return self._build_journal_row(
385
- publisher_url, full_name, abbr_name, url_home, url_about,
386
- issues_display, period, top, abbr_dict, keywords
501
+ publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
387
502
  )
388
503
 
389
504
  def _build_journal_row(
390
505
  self, publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
391
506
  ):
392
507
  """Construct journal table row string."""
393
- return f"|{publisher_url}|" \
394
- f"{self._format_link(full_name, url_home)}|" \
395
- f"{self._format_link(abbr_name, url_about)}|" \
396
- f"{issues_display}|" \
397
- f"{period}|" \
398
- f"{top}|" \
399
- f"{abbr_dict.get('score_ccf', '')}|" \
400
- f"{abbr_dict.get('score_cas', '')}|" \
401
- f"{abbr_dict.get('score_jcr', '')}|" \
402
- f"{abbr_dict.get('score_if', '')}|" \
403
- f"{'; '.join(keywords)}|"
508
+ return (
509
+ f"|{publisher_url}|"
510
+ f"{self._format_link(full_name, url_home)}|"
511
+ f"{self._format_link(abbr_name, url_about)}|"
512
+ f"{issues_display}|"
513
+ f"{period}|"
514
+ f"{top}|"
515
+ f"{abbr_dict.get('score_ccf', '')}|"
516
+ f"{abbr_dict.get('score_cas', '')}|"
517
+ f"{abbr_dict.get('score_jcr', '')}|"
518
+ f"{abbr_dict.get('score_if', '')}|"
519
+ f"{'; '.join(keywords)}|"
520
+ )
404
521
 
405
522
  # Mermaid data
406
523
  def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
@@ -413,7 +530,7 @@ class GenerateDataDict(object):
413
530
  mermaid, data_dict = [], {}
414
531
  # |AAAI|1980|95|Proceedings of the First National Conference on Artificial Intelligence|
415
532
  regex = re.compile(r"\|.*\|([0-9]+)\|([0-9]+)\|.*\|")
416
- with open(full_readme, "r") as file:
533
+ with open(full_readme, "r", encoding="utf-8") as file:
417
534
  data_list = file.readlines()
418
535
  for line in data_list:
419
536
  if mch := regex.search(line):
@@ -425,15 +542,15 @@ class GenerateDataDict(object):
425
542
  mermaid = ["```mermaid\n"]
426
543
  mermaid.extend(
427
544
  [
428
- '---\n',
429
- 'config:\n',
430
- ' xyChart:\n',
431
- ' width: 1200\n',
432
- ' height: 600\n',
433
- ' themeVariables:\n',
434
- ' xyChart:\n',
545
+ "---\n",
546
+ "config:\n",
547
+ " xyChart:\n",
548
+ " width: 1200\n",
549
+ " height: 600\n",
550
+ " themeVariables:\n",
551
+ " xyChart:\n",
435
552
  ' titleColor: "#ff0000"\n',
436
- '---\n'
553
+ "---\n",
437
554
  ]
438
555
  )
439
556
  mermaid.extend(["xychart-beta\n", f' title "{abbr}"\n'])
@@ -8,16 +8,55 @@ from .generate_dict import conference_journal_header
8
8
 
9
9
 
10
10
  def conference_journal_informations():
11
+ """Generate informational content for conferences and journals.
12
+
13
+ This function provides additional informational content that can be
14
+ included in markdown documentation for conferences and journals.
15
+
16
+ Returns:
17
+ tuple: A tuple containing two lists:
18
+ - conference_inf: List of informational strings for conferences
19
+ - journal_inf: List of informational strings for journals
20
+
21
+ Example:
22
+ >>> conf_info, journal_info = conference_journal_informations()
23
+ >>> print(conf_info[0])
24
+ !> [List of Upcoming International Conferences](...)
25
+ """
11
26
  conference_inf = [
12
27
  "!> [List of Upcoming International Conferences](https://internationalconferencealerts.com/all-events.php)\n\n",
13
- "!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n"
28
+ "!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n",
14
29
  ]
15
30
  journal_inf = []
16
31
  return conference_inf, journal_inf
17
32
 
18
33
 
19
34
  class WriteDataToMd(object):
20
- """Class to write publication data to Markdown files."""
35
+ """Class to write publication data to Markdown files.
36
+
37
+ This class provides methods to generate various markdown documentation files
38
+ from processed publication data, including introduction files, categorized
39
+ listings, publisher information, and statistics.
40
+
41
+ Attributes:
42
+ cj (str): Type of publication ('conferences' or 'journals').
43
+ ia (str): Publication type ('inproceedings' or 'article').
44
+ publisher_meta_dict (dict): Publisher metadata dictionary.
45
+ publisher_abbr_meta_dict (dict): Publisher abbreviation metadata dictionary.
46
+ keyword_abbr_meta_dict (dict): Keyword-based metadata dictionary.
47
+ path_output (str): Output directory path for generated files.
48
+
49
+ Example:
50
+ >>> writer = WriteDataToMd(
51
+ ... conferences_or_journals="conferences",
52
+ ... inproceedings_or_article="inproceedings",
53
+ ... publisher_meta_dict=pub_meta,
54
+ ... publisher_abbr_meta_dict=pub_abbr,
55
+ ... keyword_abbr_meta_dict=keyword_abbr,
56
+ ... path_output="/output"
57
+ ... )
58
+ >>> writer.save_introductions()
59
+ """
21
60
 
22
61
  def __init__(
23
62
  self,
@@ -28,7 +67,16 @@ class WriteDataToMd(object):
28
67
  keyword_abbr_meta_dict: dict,
29
68
  path_output: str,
30
69
  ) -> None:
31
- """Initialize with publication data and output path."""
70
+ """Initialize with publication data and output path.
71
+
72
+ Args:
73
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals').
74
+ inproceedings_or_article (str): Publication type ('inproceedings' or 'article').
75
+ publisher_meta_dict (dict): Publisher metadata dictionary.
76
+ publisher_abbr_meta_dict (dict): Publisher abbreviation metadata dictionary.
77
+ keyword_abbr_meta_dict (dict): Keyword-based metadata dictionary.
78
+ path_output (str): Output directory path for generated files.
79
+ """
32
80
  self.cj = conferences_or_journals # "conferences" or "journals"
33
81
  self.ia = inproceedings_or_article # "inproceedings" or "article"
34
82
  self.publisher_meta_dict = publisher_meta_dict
@@ -41,7 +89,18 @@ class WriteDataToMd(object):
41
89
  ]
42
90
 
43
91
  def save_introductions(self) -> None:
44
- """Save introduction file with all conferences/journals list."""
92
+ """Save introduction file with all conferences/journals list.
93
+
94
+ This method generates a comprehensive markdown file containing all
95
+ conferences or journals in a tabular format with appropriate headers
96
+ and informational content.
97
+
98
+ Returns:
99
+ None: This method does not return a value.
100
+
101
+ Note:
102
+ The output file is saved as 'Introductions_{type}.md' in the output directory.
103
+ """
45
104
  conference_header, journal_header = conference_journal_header()
46
105
  conference_inf, journal_inf = conference_journal_informations()
47
106
 
@@ -62,13 +121,13 @@ class WriteDataToMd(object):
62
121
  idx = 1
63
122
  for publisher in self.publisher_abbr_meta_dict:
64
123
  for abbr in self.publisher_abbr_meta_dict[publisher]:
65
- row_info = self.publisher_abbr_meta_dict[publisher][abbr]['row_inf']
124
+ row_info = self.publisher_abbr_meta_dict[publisher][abbr]["row_inf"]
66
125
  data_list.append(f"|{idx}{row_info}\n")
67
126
  idx += 1
68
127
 
69
128
  # Write to file
70
129
  output_file = os.path.join(self.path_output, f"Introductions_{self.cj.title()}.md")
71
- with open(output_file, "w") as f:
130
+ with open(output_file, "w", encoding="utf-8") as f:
72
131
  f.writelines(data_list)
73
132
 
74
133
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
@@ -88,7 +147,21 @@ class WriteDataToMd(object):
88
147
 
89
148
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
90
149
  def save_categories(self, keywords_category_name: str, keywords_list: List[str]) -> None:
91
- """Save publications categorized by keywords."""
150
+ """Save publications categorized by keywords.
151
+
152
+ This method generates markdown files organizing publications by their
153
+ keywords, creating separate sections for each keyword category.
154
+
155
+ Args:
156
+ keywords_category_name (str): The category name for keywords filtering.
157
+ keywords_list (List[str]): List of keywords to include in the output.
158
+
159
+ Returns:
160
+ None: This method does not return a value.
161
+
162
+ Note:
163
+ The output file is saved as 'Categories_{type}_{category}.md' in the output directory.
164
+ """
92
165
  conference_header, journal_header = conference_journal_header()
93
166
  data_list = [f"# {self.cj.title()}\n\n"]
94
167
  data_list.extend(self._default_inf)
@@ -110,7 +183,9 @@ class WriteDataToMd(object):
110
183
 
111
184
  # Write to file
112
185
  category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
113
- with open(os.path.join(self.path_output, f"Categories_{self.cj.title()}{category_postfix}.md"), "w") as f:
186
+ with open(
187
+ os.path.join(self.path_output, f"Categories_{self.cj.title()}{category_postfix}.md"), "w", encoding="utf-8"
188
+ ) as f:
114
189
  f.writelines(data_list)
115
190
 
116
191
  return None
@@ -136,18 +211,31 @@ class WriteDataToMd(object):
136
211
 
137
212
  # Write keyword-specific file
138
213
  path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
139
- with open(os.path.join(path_key, f"{keyword.replace(' ', '_')}.md"), "w") as f:
214
+ # Create safe filename by replacing invalid characters
215
+ safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
216
+ with open(os.path.join(path_key, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
140
217
  f.writelines(data_list)
141
218
 
142
219
  return None
143
220
 
144
221
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
145
222
  def save_publishers(self) -> None:
146
- """Save publisher overview file with basic information."""
223
+ """Save publisher overview file with basic information.
224
+
225
+ This method generates a markdown file containing an overview of all
226
+ publishers with their basic information, about pages, and links to
227
+ detailed publisher-specific files.
228
+
229
+ Returns:
230
+ None: This method does not return a value.
231
+
232
+ Note:
233
+ The output file is saved as 'Publishers_{type}.md' in the output directory.
234
+ """
147
235
  data_list_pub = [
148
236
  f"# Introductions of Publishers and {self.cj.title()}\n\n",
149
237
  "| |Publishers|About US|Conferences/Journals|Separate Links|\n",
150
- "|-|- |- |- |- |\n"
238
+ "|-|- |- |- |- |\n",
151
239
  ]
152
240
  idx = 1
153
241
 
@@ -171,7 +259,7 @@ class WriteDataToMd(object):
171
259
  idx += 1
172
260
 
173
261
  # Write to file
174
- with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w") as f:
262
+ with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w", encoding="utf-8") as f:
175
263
  f.writelines(data_list_pub)
176
264
  return None
177
265
 
@@ -205,7 +293,7 @@ class WriteDataToMd(object):
205
293
 
206
294
  # Add row information
207
295
  row_info = self.publisher_abbr_meta_dict[pub][abbr]["row_inf"]
208
- data_list.append(f'{row_info}\n\n')
296
+ data_list.append(f"{row_info}\n\n")
209
297
 
210
298
  # Add remarks and about for this abbreviation
211
299
  for flag in ["txt_remarks", "txt_abouts"]:
@@ -222,7 +310,7 @@ class WriteDataToMd(object):
222
310
 
223
311
  # Write publisher-specific file
224
312
  path_pub = standardize_path(os.path.join(self.path_output, f"Publishers_{self.cj.title()}"))
225
- with open(os.path.join(path_pub, f"{pub}.md"), "w") as f:
313
+ with open(os.path.join(path_pub, f"{pub}.md"), "w", encoding="utf-8") as f:
226
314
  f.writelines(data_list)
227
315
 
228
316
  return None
@@ -232,13 +320,15 @@ class WriteDataToMd(object):
232
320
  data_list = [
233
321
  f"# Statistics of keywords in {self.cj.title()}\n\n",
234
322
  "| |keywords|Separate Links|\n",
235
- "|-|- |- |\n"
323
+ "|-|- |- |\n",
236
324
  ]
237
325
  idx = 1
238
326
 
239
327
  # Add publications for each category
240
328
  for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
241
- local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{keyword.replace(' ', '_')}.md)"
329
+ # Create safe filename for URL
330
+ safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
331
+ local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{safe_keyword}.md)"
242
332
 
243
333
  # Create table row
244
334
  row = f"| {idx} | {keyword} | {local_url} |\n"
@@ -247,7 +337,9 @@ class WriteDataToMd(object):
247
337
 
248
338
  # Write to file
249
339
  category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
250
- with open(os.path.join(self.path_output, f"Statistics_{self.cj.title()}{category_postfix}.md"), "w") as f:
340
+ with open(
341
+ os.path.join(self.path_output, f"Statistics_{self.cj.title()}{category_postfix}.md"), "w", encoding="utf-8"
342
+ ) as f:
251
343
  f.writelines(data_list)
252
344
  return None
253
345
 
@@ -269,7 +361,7 @@ class WriteDataToMd(object):
269
361
 
270
362
  # Add row information
271
363
  row_info = self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"]
272
- data_list.append(f'{row_info}\n\n')
364
+ data_list.append(f"{row_info}\n\n")
273
365
 
274
366
  # Add statistics if available
275
367
  if statistics := self.keyword_abbr_meta_dict[keyword][abbr].get("statistics", []):
@@ -278,7 +370,9 @@ class WriteDataToMd(object):
278
370
 
279
371
  # Write publisher-specific file
280
372
  path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
281
- with open(os.path.join(path_pub, f"{keyword.replace(' ', '_')}.md"), "w") as f:
373
+ # Create safe filename by replacing invalid characters
374
+ safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
375
+ with open(os.path.join(path_pub, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
282
376
  f.writelines(data_list)
283
377
 
284
378
  return None
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "pyformatjson"
3
- version = "0.0.1"
3
+ version = "0.0.3"
4
4
  description = "pyformatjson"
5
5
  license = "GPL-3.0-or-later"
6
6
  authors = ["NextAI <nextartifintell@gmail.com>"]
File without changes
@@ -1,103 +0,0 @@
1
- # coding=utf-8
2
-
3
- import os
4
- import re
5
- from typing import List
6
-
7
-
8
- def split_text_by_length(text, max_length=120) -> List[str]:
9
- lines = []
10
- while text:
11
- if len(text) <= max_length:
12
- lines.append(text)
13
- break
14
-
15
- split_pos = text.rfind(" ", 0, max_length + 1)
16
- if split_pos == -1:
17
- split_pos = max_length
18
-
19
- line = text[:split_pos]
20
- lines.append(line)
21
-
22
- text = text[split_pos:]
23
-
24
- new_lines = []
25
- for line in lines:
26
- new_lines.append(line)
27
- return new_lines
28
-
29
-
30
- def split_data_list(
31
- split_pattern: str, data_list: List[str], last_next: str = "next"
32
- ) -> List[str]:
33
- r"""Split data list according to the split pattern.
34
-
35
- The capturing parentheses must be used in the pattern, such as `(\n)`.
36
-
37
- Args:
38
- split_pattern (str): split pattern.
39
- data_list (List[str]): data list.
40
- last_next (str): "next" or "last".
41
-
42
- Returns:
43
- List[str]: new data list.
44
-
45
- Examples:
46
- split_pattern = r"(\n)", last_next = "next" or "last".
47
- """
48
- new_data_list = []
49
- for line in data_list:
50
- split_list = re.split(split_pattern, line)
51
- list_one = split_list[0: len(split_list): 2]
52
- list_two = split_list[1: len(split_list): 2]
53
-
54
- temp = []
55
- if last_next == "next":
56
- list_two.insert(0, "")
57
- temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
58
- if last_next == "last":
59
- list_two.append("")
60
- temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
61
- new_data_list.extend(temp)
62
- new_data_list = [line for line in new_data_list if line.strip()]
63
- return new_data_list
64
-
65
-
66
- def standardize_path(path_input: str) -> str:
67
- path_input = os.path.expandvars(os.path.expanduser(path_input))
68
- if not os.path.exists(path_input):
69
- os.makedirs(path_input)
70
- return path_input
71
-
72
-
73
- def sort_strings_with_embedded_numbers(s: str) -> List[str]:
74
- re_digits = re.compile(r"(\d+)")
75
- pieces = re_digits.split(s)
76
- pieces[1::2] = map(int, pieces[1::2])
77
- return pieces
78
-
79
-
80
- def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
81
- return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
82
-
83
-
84
- class IterateSortDict(object):
85
- def __init__(self, reverse: bool = False) -> None:
86
- self.reverse = reverse
87
-
88
- def dict_update(self, old):
89
- """Update."""
90
- old = self.dict_sort_iteration(old)
91
- old = self.dict_sort(old)
92
- return old
93
-
94
- def dict_sort_iteration(self, old: dict):
95
- """Sort."""
96
- for key in old:
97
- if isinstance(old[key], dict):
98
- old[key] = self.dict_update(old[key])
99
- return old
100
-
101
- def dict_sort(self, old: dict):
102
- """Sort."""
103
- return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
@@ -1,79 +0,0 @@
1
- # coding=utf-8
2
-
3
- import os
4
- from typing import Optional
5
-
6
- from .core._base import standardize_path
7
- from .core.update_json import load_json_data, update_json_file
8
- from .tools.generate_dict import GenerateDataDict
9
- from .tools.write_dict import WriteDataToMd
10
-
11
-
12
- def main_generate_md_files(
13
- path_json: str,
14
- path_output: str,
15
- path_spidered_bibs: Optional[str] = None,
16
- for_vue: bool = True,
17
- conferences_or_journals: Optional[str] = None,
18
- keywords_category_name: str = ""
19
- ) -> None:
20
- """
21
- Generate markdown files for conferences and journals.
22
-
23
- Args:
24
- path_json: Path to JSON data file
25
- path_output: Output directory for markdown files
26
- path_spidered_bibs: Directory containing crawled BibTeX files
27
- for_vue: Whether to generate Vue-compatible format
28
- conferences_or_journals: Specify 'conferences' or 'journals', None for both
29
- keywords_category_name: The category name of keywords
30
- """
31
- # Standardize all paths
32
- path_json = standardize_path(path_json)
33
- path_output = standardize_path(path_output)
34
- path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
35
-
36
- # Process keyword category name and load data
37
- keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
38
- category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
39
- keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
40
-
41
- # Validate data availability
42
- if not keywords_list or not keywords_category_name:
43
- keywords_list, keywords_category_name = [], ""
44
-
45
- # Process both conferences and journals
46
- for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
47
- # Skip if specific type requested and doesn't match
48
- if conferences_or_journals and conferences_or_journals.lower() != cj:
49
- continue
50
-
51
- # Update JSON data
52
- json_dict = update_json_file(path_json, cj)
53
- if not json_dict:
54
- continue
55
-
56
- # Generate data dictionaries
57
- path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
58
- generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
59
- publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
60
- if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
61
- continue
62
-
63
- # Initialize writer and save all markdown files
64
- _path_output = os.path.join(path_output, f"{cj.title()}")
65
- save_data = WriteDataToMd(
66
- cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
67
- )
68
- # Save various documentation files
69
- save_data.save_introductions()
70
- save_data.save_categories(keywords_category_name, keywords_list)
71
- save_data.save_categories_separate_keywords()
72
-
73
- save_data.save_publishers()
74
- save_data.save_publishers_separate_abbrs()
75
-
76
- save_data.save_statistics(keywords_category_name, keywords_list)
77
- save_data.save_statistics_separate_abbrs()
78
-
79
- return None
File without changes