pyformatjson 0.0.1__tar.gz → 0.0.3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/PKG-INFO +1 -1
- pyformatjson-0.0.3/pyformatjson/__init__.py +17 -0
- pyformatjson-0.0.3/pyformatjson/core/_base.py +223 -0
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/pyformatjson/core/update_json.py +44 -11
- pyformatjson-0.0.3/pyformatjson/main.py +155 -0
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/pyformatjson/tools/generate_dict.py +190 -73
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/pyformatjson/tools/write_dict.py +113 -19
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/pyproject.toml +1 -1
- pyformatjson-0.0.1/pyformatjson/__init__.py +0 -0
- pyformatjson-0.0.1/pyformatjson/core/_base.py +0 -103
- pyformatjson-0.0.1/pyformatjson/main.py +0 -79
- {pyformatjson-0.0.1 → pyformatjson-0.0.3}/README.md +0 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""PyFormatJSON: A Python library for formatting and processing JSON data for conferences and journals.
|
|
2
|
+
|
|
3
|
+
This package provides utilities for processing conference and journal data from JSON files,
|
|
4
|
+
generating markdown documentation, and creating various formatted outputs for academic
|
|
5
|
+
publication management.
|
|
6
|
+
|
|
7
|
+
Modules:
|
|
8
|
+
core: Core functionality for data processing and JSON manipulation
|
|
9
|
+
tools: Utility tools for data generation and markdown writing
|
|
10
|
+
|
|
11
|
+
Classes:
|
|
12
|
+
GenerateDataDict: Generates data dictionaries from JSON input
|
|
13
|
+
WriteDataToMd: Writes publication data to Markdown files
|
|
14
|
+
|
|
15
|
+
Functions:
|
|
16
|
+
main_generate_md_files: Main function to generate markdown files for conferences and journals
|
|
17
|
+
"""
|
|
@@ -0,0 +1,223 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
from typing import List
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def split_text_by_length(text, max_length=120) -> List[str]:
|
|
9
|
+
"""Split text into lines of specified maximum length.
|
|
10
|
+
|
|
11
|
+
This function breaks long text into multiple lines, ensuring each line
|
|
12
|
+
does not exceed the specified maximum length. It attempts to break at
|
|
13
|
+
word boundaries when possible.
|
|
14
|
+
|
|
15
|
+
Args:
|
|
16
|
+
text (str): The input text to be split into lines.
|
|
17
|
+
max_length (int, optional): Maximum length for each line. Defaults to 120.
|
|
18
|
+
|
|
19
|
+
Returns:
|
|
20
|
+
List[str]: A list of text lines, each not exceeding max_length characters.
|
|
21
|
+
|
|
22
|
+
Example:
|
|
23
|
+
>>> split_text_by_length("This is a very long text that needs to be split", 20)
|
|
24
|
+
['This is a very long', 'text that needs to be', 'split']
|
|
25
|
+
"""
|
|
26
|
+
lines = []
|
|
27
|
+
while text:
|
|
28
|
+
if len(text) <= max_length:
|
|
29
|
+
lines.append(text)
|
|
30
|
+
break
|
|
31
|
+
|
|
32
|
+
split_pos = text.rfind(" ", 0, max_length + 1)
|
|
33
|
+
if split_pos == -1:
|
|
34
|
+
split_pos = max_length
|
|
35
|
+
|
|
36
|
+
line = text[:split_pos]
|
|
37
|
+
lines.append(line)
|
|
38
|
+
|
|
39
|
+
text = text[split_pos:]
|
|
40
|
+
|
|
41
|
+
new_lines = []
|
|
42
|
+
for line in lines:
|
|
43
|
+
new_lines.append(line)
|
|
44
|
+
return new_lines
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "next") -> List[str]:
|
|
48
|
+
"""Split data list according to the split pattern.
|
|
49
|
+
|
|
50
|
+
This function splits each string in the data list using the provided regex pattern
|
|
51
|
+
and reconstructs the data based on the last_next parameter. The pattern must use
|
|
52
|
+
capturing parentheses to define split points.
|
|
53
|
+
|
|
54
|
+
Args:
|
|
55
|
+
split_pattern (str): Regular expression pattern for splitting. Must use capturing
|
|
56
|
+
parentheses, e.g., r"(\n)" for newline splits.
|
|
57
|
+
data_list (List[str]): List of strings to be split and processed.
|
|
58
|
+
last_next (str, optional): Determines how to handle split parts. "next" places
|
|
59
|
+
the split character at the beginning of the next part, "last" places it at
|
|
60
|
+
the end of the current part. Defaults to "next".
|
|
61
|
+
|
|
62
|
+
Returns:
|
|
63
|
+
List[str]: New list of processed strings with empty strings filtered out.
|
|
64
|
+
|
|
65
|
+
Raises:
|
|
66
|
+
re.error: If the split_pattern is not a valid regular expression.
|
|
67
|
+
|
|
68
|
+
Example:
|
|
69
|
+
>>> split_data_list(r"(\n)", ["line1\nline2", "line3\nline4"], "next")
|
|
70
|
+
['line1', 'line2', 'line3', 'line4']
|
|
71
|
+
"""
|
|
72
|
+
new_data_list = []
|
|
73
|
+
for line in data_list:
|
|
74
|
+
split_list = re.split(split_pattern, line)
|
|
75
|
+
list_one = split_list[0 : len(split_list) : 2]
|
|
76
|
+
list_two = split_list[1 : len(split_list) : 2]
|
|
77
|
+
|
|
78
|
+
temp = []
|
|
79
|
+
if last_next == "next":
|
|
80
|
+
list_two.insert(0, "")
|
|
81
|
+
temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
|
|
82
|
+
if last_next == "last":
|
|
83
|
+
list_two.append("")
|
|
84
|
+
temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
|
|
85
|
+
new_data_list.extend(temp)
|
|
86
|
+
new_data_list = [line for line in new_data_list if line.strip()]
|
|
87
|
+
return new_data_list
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def standardize_path(path_input: str) -> str:
|
|
91
|
+
"""Standardize and ensure a directory path exists.
|
|
92
|
+
|
|
93
|
+
This function expands environment variables and user home directory references
|
|
94
|
+
in the path, then creates the directory if it doesn't exist.
|
|
95
|
+
|
|
96
|
+
Args:
|
|
97
|
+
path_input (str): The input path to be standardized and created.
|
|
98
|
+
|
|
99
|
+
Returns:
|
|
100
|
+
str: The standardized absolute path.
|
|
101
|
+
|
|
102
|
+
Example:
|
|
103
|
+
>>> standardize_path("~/Documents/data")
|
|
104
|
+
'/Users/username/Documents/data'
|
|
105
|
+
"""
|
|
106
|
+
path_input = os.path.expandvars(os.path.expanduser(path_input))
|
|
107
|
+
if not os.path.exists(path_input):
|
|
108
|
+
os.makedirs(path_input)
|
|
109
|
+
return path_input
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def sort_strings_with_embedded_numbers(s: str) -> List[str]:
|
|
113
|
+
"""Split string into pieces for natural sorting with embedded numbers.
|
|
114
|
+
|
|
115
|
+
This function splits a string into pieces where numbers are converted to integers
|
|
116
|
+
for proper natural sorting (e.g., "item2" comes before "item10").
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
s (str): The string to be split into sortable pieces.
|
|
120
|
+
|
|
121
|
+
Returns:
|
|
122
|
+
List[str]: List of string pieces with numbers converted to integers.
|
|
123
|
+
|
|
124
|
+
Example:
|
|
125
|
+
>>> sort_strings_with_embedded_numbers("item10")
|
|
126
|
+
['item', 10]
|
|
127
|
+
"""
|
|
128
|
+
re_digits = re.compile(r"(\d+)")
|
|
129
|
+
pieces = re_digits.split(s)
|
|
130
|
+
pieces[1::2] = map(int, pieces[1::2])
|
|
131
|
+
return pieces
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
|
|
135
|
+
"""Sort list of strings with embedded numbers naturally.
|
|
136
|
+
|
|
137
|
+
This function sorts a list of strings using natural sorting that handles
|
|
138
|
+
embedded numbers correctly (e.g., "item2" comes before "item10").
|
|
139
|
+
|
|
140
|
+
Args:
|
|
141
|
+
str_int (List[str]): List of strings to be sorted.
|
|
142
|
+
reverse (bool, optional): If True, sorts in descending order. Defaults to False.
|
|
143
|
+
|
|
144
|
+
Returns:
|
|
145
|
+
List[str]: Sorted list of strings.
|
|
146
|
+
|
|
147
|
+
Example:
|
|
148
|
+
>>> sort_int_str(["item10", "item2", "item1"])
|
|
149
|
+
['item1', 'item2', 'item10']
|
|
150
|
+
"""
|
|
151
|
+
return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
class IterateSortDict(object):
|
|
155
|
+
"""A class for recursively sorting dictionary keys with natural sorting.
|
|
156
|
+
|
|
157
|
+
This class provides methods to sort dictionary keys recursively, handling
|
|
158
|
+
nested dictionaries and using natural sorting for strings with embedded numbers.
|
|
159
|
+
|
|
160
|
+
Attributes:
|
|
161
|
+
reverse (bool): If True, sorts keys in descending order. Defaults to False.
|
|
162
|
+
|
|
163
|
+
Example:
|
|
164
|
+
>>> sorter = IterateSortDict(reverse=False)
|
|
165
|
+
>>> data = {"item10": {"sub2": 1, "sub1": 2}, "item2": 3}
|
|
166
|
+
>>> sorted_data = sorter.dict_update(data)
|
|
167
|
+
"""
|
|
168
|
+
|
|
169
|
+
def __init__(self, reverse: bool = False) -> None:
|
|
170
|
+
"""Initialize the IterateSortDict instance.
|
|
171
|
+
|
|
172
|
+
Args:
|
|
173
|
+
reverse (bool, optional): If True, sorts keys in descending order.
|
|
174
|
+
Defaults to False.
|
|
175
|
+
"""
|
|
176
|
+
self.reverse = reverse
|
|
177
|
+
|
|
178
|
+
def dict_update(self, old):
|
|
179
|
+
"""Update and sort a dictionary recursively.
|
|
180
|
+
|
|
181
|
+
This method sorts the dictionary keys and recursively processes
|
|
182
|
+
any nested dictionaries.
|
|
183
|
+
|
|
184
|
+
Args:
|
|
185
|
+
old (dict): The dictionary to be sorted and updated.
|
|
186
|
+
|
|
187
|
+
Returns:
|
|
188
|
+
dict: The updated dictionary with sorted keys at all levels.
|
|
189
|
+
"""
|
|
190
|
+
old = self.dict_sort_iteration(old)
|
|
191
|
+
old = self.dict_sort(old)
|
|
192
|
+
return old
|
|
193
|
+
|
|
194
|
+
def dict_sort_iteration(self, old: dict):
|
|
195
|
+
"""Recursively sort nested dictionaries.
|
|
196
|
+
|
|
197
|
+
This method iterates through the dictionary and recursively sorts
|
|
198
|
+
any nested dictionary values.
|
|
199
|
+
|
|
200
|
+
Args:
|
|
201
|
+
old (dict): The dictionary to be processed recursively.
|
|
202
|
+
|
|
203
|
+
Returns:
|
|
204
|
+
dict: The dictionary with nested dictionaries sorted.
|
|
205
|
+
"""
|
|
206
|
+
for key in old:
|
|
207
|
+
if isinstance(old[key], dict):
|
|
208
|
+
old[key] = self.dict_update(old[key])
|
|
209
|
+
return old
|
|
210
|
+
|
|
211
|
+
def dict_sort(self, old: dict):
|
|
212
|
+
"""Sort dictionary keys using natural sorting.
|
|
213
|
+
|
|
214
|
+
This method sorts the top-level keys of the dictionary using
|
|
215
|
+
natural sorting that handles embedded numbers.
|
|
216
|
+
|
|
217
|
+
Args:
|
|
218
|
+
old (dict): The dictionary whose keys are to be sorted.
|
|
219
|
+
|
|
220
|
+
Returns:
|
|
221
|
+
dict: A new dictionary with sorted keys.
|
|
222
|
+
"""
|
|
223
|
+
return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
|
|
@@ -8,7 +8,24 @@ from ._base import split_data_list, split_text_by_length
|
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
def load_json_data(path_json: str, filename: str) -> Dict:
|
|
11
|
-
"""Load JSON data from file.
|
|
11
|
+
"""Load JSON data from a specified file.
|
|
12
|
+
|
|
13
|
+
This function attempts to load JSON data from a file located in the specified
|
|
14
|
+
directory. If the file doesn't exist or there's an error loading it, an empty
|
|
15
|
+
dictionary is returned.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
path_json (str): Directory path containing the JSON file.
|
|
19
|
+
filename (str): Name of the JSON file (without .json extension).
|
|
20
|
+
|
|
21
|
+
Returns:
|
|
22
|
+
Dict: The loaded JSON data as a dictionary, or empty dict if file not found
|
|
23
|
+
or error occurs.
|
|
24
|
+
|
|
25
|
+
Example:
|
|
26
|
+
>>> load_json_data("/data", "conferences")
|
|
27
|
+
{"publisher1": {"conferences": {...}}}
|
|
28
|
+
"""
|
|
12
29
|
try:
|
|
13
30
|
file_path = os.path.join(path_json, f"{filename}.json")
|
|
14
31
|
if not os.path.exists(file_path):
|
|
@@ -23,15 +40,25 @@ def load_json_data(path_json: str, filename: str) -> Dict:
|
|
|
23
40
|
|
|
24
41
|
|
|
25
42
|
def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
|
|
26
|
-
"""
|
|
27
|
-
|
|
43
|
+
"""Update and format JSON file containing conference/journal data.
|
|
44
|
+
|
|
45
|
+
This function loads JSON data, processes and formats text fields by splitting
|
|
46
|
+
long text into appropriate lengths, checks for duplicate abbreviations, and
|
|
47
|
+
saves the updated data back to the file.
|
|
28
48
|
|
|
29
49
|
Args:
|
|
30
|
-
path_root (str): Root directory path
|
|
31
|
-
conferences_or_journals (str): Type of publication ('conferences' or 'journals')
|
|
50
|
+
path_root (str): Root directory path containing the JSON file.
|
|
51
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
32
52
|
|
|
33
53
|
Returns:
|
|
34
|
-
Dict[str, Any]: Processed JSON data
|
|
54
|
+
Dict[str, Any]: Processed JSON data dictionary.
|
|
55
|
+
|
|
56
|
+
Raises:
|
|
57
|
+
ValueError: If duplicate abbreviations are found in the data.
|
|
58
|
+
|
|
59
|
+
Example:
|
|
60
|
+
>>> update_json_file("/data", "conferences")
|
|
61
|
+
{"publisher1": {"conferences": {"conf1": {...}}}}
|
|
35
62
|
"""
|
|
36
63
|
# Load Json Data
|
|
37
64
|
json_dict = load_json_data(path_root, conferences_or_journals)
|
|
@@ -68,15 +95,21 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
|
|
|
68
95
|
|
|
69
96
|
|
|
70
97
|
def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
|
|
71
|
-
"""
|
|
72
|
-
|
|
98
|
+
"""Check for duplicate abbreviations in the data.
|
|
99
|
+
|
|
100
|
+
This function validates that there are no duplicate abbreviations within
|
|
101
|
+
the same publication type across all publishers.
|
|
73
102
|
|
|
74
103
|
Args:
|
|
75
|
-
json_dict: JSON data dictionary
|
|
76
|
-
conferences_or_journals (str): Type of publication ('conferences' or 'journals')
|
|
104
|
+
json_dict (Dict[str, Any]): JSON data dictionary containing publication information.
|
|
105
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
77
106
|
|
|
78
107
|
Raises:
|
|
79
|
-
ValueError: If duplicate abbreviations are found
|
|
108
|
+
ValueError: If duplicate abbreviations are found in the data.
|
|
109
|
+
|
|
110
|
+
Example:
|
|
111
|
+
>>> _check_duplicate_abbr(data, "conferences")
|
|
112
|
+
# Raises ValueError if "ICML" appears twice in conferences
|
|
80
113
|
"""
|
|
81
114
|
abbr_list = []
|
|
82
115
|
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
from .core._base import standardize_path
|
|
8
|
+
from .core.update_json import load_json_data, update_json_file
|
|
9
|
+
from .tools.generate_dict import GenerateDataDict
|
|
10
|
+
from .tools.write_dict import WriteDataToMd
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
def main_generate_md_files(
|
|
14
|
+
path_json: str,
|
|
15
|
+
path_output_md: str,
|
|
16
|
+
path_output_simplified_json: str,
|
|
17
|
+
path_spidered_bibs: Optional[str] = None,
|
|
18
|
+
for_vue: bool = True,
|
|
19
|
+
conferences_or_journals: Optional[str] = None,
|
|
20
|
+
keywords_category_name: str = "",
|
|
21
|
+
) -> None:
|
|
22
|
+
"""Generate markdown files for conferences and journals.
|
|
23
|
+
|
|
24
|
+
This function processes JSON data containing conference and journal information,
|
|
25
|
+
generates various markdown documentation files, and creates simplified JSON outputs.
|
|
26
|
+
It supports both conference and journal processing with customizable keyword categories.
|
|
27
|
+
|
|
28
|
+
Args:
|
|
29
|
+
path_json (str): Path to the input JSON data file containing publication information.
|
|
30
|
+
path_output_md (str): Output directory path where markdown files will be saved.
|
|
31
|
+
path_output_simplified_json (str): Output directory path for simplified JSON files.
|
|
32
|
+
path_spidered_bibs (Optional[str], optional): Directory containing crawled BibTeX files.
|
|
33
|
+
Defaults to None.
|
|
34
|
+
for_vue (bool, optional): Whether to generate Vue.js-compatible format for date calculations.
|
|
35
|
+
Defaults to True.
|
|
36
|
+
conferences_or_journals (Optional[str], optional): Specify 'conferences' or 'journals' to
|
|
37
|
+
process only one type, or None to process both. Defaults to None.
|
|
38
|
+
keywords_category_name (str, optional): The category name for keywords filtering.
|
|
39
|
+
Defaults to "".
|
|
40
|
+
|
|
41
|
+
Returns:
|
|
42
|
+
None: This function does not return a value.
|
|
43
|
+
|
|
44
|
+
Raises:
|
|
45
|
+
FileNotFoundError: If the input JSON file or required directories are not found.
|
|
46
|
+
ValueError: If there are issues with the data format or duplicate abbreviations.
|
|
47
|
+
|
|
48
|
+
Example:
|
|
49
|
+
>>> main_generate_md_files(
|
|
50
|
+
... path_json="/data/publications.json",
|
|
51
|
+
... path_output_md="/output/markdown",
|
|
52
|
+
... path_output_simplified_json="/output/json",
|
|
53
|
+
... for_vue=True,
|
|
54
|
+
... conferences_or_journals="conferences"
|
|
55
|
+
... )
|
|
56
|
+
"""
|
|
57
|
+
# Standardize all paths
|
|
58
|
+
path_json = standardize_path(path_json)
|
|
59
|
+
path_output_md = standardize_path(path_output_md)
|
|
60
|
+
path_output_simplified_json = standardize_path(path_output_simplified_json)
|
|
61
|
+
|
|
62
|
+
path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
|
|
63
|
+
|
|
64
|
+
# Process keyword category name and load data
|
|
65
|
+
keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
|
|
66
|
+
category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
|
|
67
|
+
keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
|
|
68
|
+
|
|
69
|
+
# Validate data availability
|
|
70
|
+
if not keywords_list or not keywords_category_name:
|
|
71
|
+
keywords_list, keywords_category_name = [], ""
|
|
72
|
+
|
|
73
|
+
# Process both conferences and journals
|
|
74
|
+
for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
|
|
75
|
+
# Skip if specific type requested and doesn't match
|
|
76
|
+
if conferences_or_journals and conferences_or_journals.lower() != cj:
|
|
77
|
+
continue
|
|
78
|
+
|
|
79
|
+
# Update JSON data
|
|
80
|
+
json_dict = update_json_file(path_json, cj)
|
|
81
|
+
if not json_dict:
|
|
82
|
+
continue
|
|
83
|
+
|
|
84
|
+
# Simplify JSON data
|
|
85
|
+
simplify_json(json_dict, cj, path_output_simplified_json)
|
|
86
|
+
|
|
87
|
+
# Generate data dictionaries
|
|
88
|
+
path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
|
|
89
|
+
generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
|
|
90
|
+
publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
|
|
91
|
+
if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
|
|
92
|
+
continue
|
|
93
|
+
|
|
94
|
+
# Initialize writer and save all markdown files
|
|
95
|
+
_path_output = os.path.join(path_output_md, f"{cj.title()}")
|
|
96
|
+
save_data = WriteDataToMd(
|
|
97
|
+
cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
|
|
98
|
+
)
|
|
99
|
+
# Save various documentation files
|
|
100
|
+
save_data.save_introductions()
|
|
101
|
+
save_data.save_categories(keywords_category_name, keywords_list)
|
|
102
|
+
save_data.save_categories_separate_keywords()
|
|
103
|
+
|
|
104
|
+
save_data.save_publishers()
|
|
105
|
+
save_data.save_publishers_separate_abbrs()
|
|
106
|
+
|
|
107
|
+
save_data.save_statistics(keywords_category_name, keywords_list)
|
|
108
|
+
save_data.save_statistics_separate_abbrs()
|
|
109
|
+
|
|
110
|
+
return None
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def simplify_json(json_dict, cj: str, output_dir: str) -> None:
|
|
114
|
+
"""Simplify JSON dictionary by extracting only essential fields.
|
|
115
|
+
|
|
116
|
+
This function creates a simplified version of the JSON dictionary containing
|
|
117
|
+
only the names_abbr and names_full fields for each publisher and publication type.
|
|
118
|
+
The simplified data is saved to a new JSON file in the specified output directory.
|
|
119
|
+
|
|
120
|
+
Args:
|
|
121
|
+
json_dict (dict): The original JSON dictionary containing publication data.
|
|
122
|
+
cj (str): The type of publication, either 'conferences' or 'journals'.
|
|
123
|
+
output_dir (str): Directory path where the simplified JSON file will be saved.
|
|
124
|
+
|
|
125
|
+
Returns:
|
|
126
|
+
None: This function does not return a value.
|
|
127
|
+
|
|
128
|
+
Note:
|
|
129
|
+
The function creates a new JSON file named '{cj}.json' in the output directory.
|
|
130
|
+
Only the 'names_abbr' and 'names_full' fields are preserved in the simplified version.
|
|
131
|
+
|
|
132
|
+
Example:
|
|
133
|
+
>>> simplify_json(
|
|
134
|
+
... json_dict=publication_data,
|
|
135
|
+
... cj="conferences",
|
|
136
|
+
... output_dir="/output/simplified"
|
|
137
|
+
... )
|
|
138
|
+
"""
|
|
139
|
+
new_json_dict = {}
|
|
140
|
+
for publisher in json_dict:
|
|
141
|
+
for abbr in json_dict[publisher][cj.lower()]:
|
|
142
|
+
names_abbr = json_dict[publisher][cj.lower()][abbr].get("names_abbr", [])
|
|
143
|
+
names_full = json_dict[publisher][cj.lower()][abbr].get("names_full", [])
|
|
144
|
+
|
|
145
|
+
new_json_dict.setdefault(publisher, {}).setdefault(cj.lower(), {}).setdefault(abbr, {}).update(
|
|
146
|
+
{"names_abbr": names_abbr, "names_full": names_full}
|
|
147
|
+
)
|
|
148
|
+
|
|
149
|
+
# Save updated JSON
|
|
150
|
+
if new_json_dict:
|
|
151
|
+
path_file = os.path.join(output_dir, f"{cj}.json")
|
|
152
|
+
with open(path_file, "w", encoding="utf-8") as f:
|
|
153
|
+
f.write(json.dumps(new_json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
154
|
+
|
|
155
|
+
return None
|
|
@@ -7,6 +7,21 @@ from typing import Optional
|
|
|
7
7
|
|
|
8
8
|
|
|
9
9
|
def conference_journal_header():
|
|
10
|
+
"""Generate markdown table headers for conferences and journals.
|
|
11
|
+
|
|
12
|
+
This function creates the appropriate markdown table headers for displaying
|
|
13
|
+
conference and journal information in tabular format.
|
|
14
|
+
|
|
15
|
+
Returns:
|
|
16
|
+
tuple: A tuple containing two lists:
|
|
17
|
+
- conference_header: Markdown table headers for conferences
|
|
18
|
+
- journal_header: Markdown table headers for journals
|
|
19
|
+
|
|
20
|
+
Example:
|
|
21
|
+
>>> conf_header, journal_header = conference_journal_header()
|
|
22
|
+
>>> print(conf_header[0])
|
|
23
|
+
|Publishers|Full/Homepage|Abbr/About|Acronym/Archive|Period/DBLP|...
|
|
24
|
+
"""
|
|
10
25
|
o = "|Publishers|Full/Homepage|Abbr/About|"
|
|
11
26
|
t = "|- |- |- |"
|
|
12
27
|
conference_header = [
|
|
@@ -21,6 +36,29 @@ def conference_journal_header():
|
|
|
21
36
|
|
|
22
37
|
|
|
23
38
|
class GenerateDataDict(object):
|
|
39
|
+
"""Generate data dictionaries from JSON input for conferences and journals.
|
|
40
|
+
|
|
41
|
+
This class processes JSON data containing conference or journal information
|
|
42
|
+
and generates structured dictionaries for markdown table generation, including
|
|
43
|
+
publisher metadata, keyword-based indexing, and Mermaid diagram data.
|
|
44
|
+
|
|
45
|
+
Attributes:
|
|
46
|
+
cj (str): Type of publication ('conferences' or 'journals').
|
|
47
|
+
ia (str): Publication type ('inproceedings' or 'article').
|
|
48
|
+
json_dict (dict): Input JSON data containing publication information.
|
|
49
|
+
path_spidered_cj (Optional[str]): Path to spidered conference/journal data.
|
|
50
|
+
for_vue (bool): Whether to generate Vue.js-compatible format.
|
|
51
|
+
|
|
52
|
+
Example:
|
|
53
|
+
>>> generator = GenerateDataDict(
|
|
54
|
+
... conferences_or_journals="conferences",
|
|
55
|
+
... inproceedings_or_article="inproceedings",
|
|
56
|
+
... json_dict=publication_data,
|
|
57
|
+
... for_vue=True
|
|
58
|
+
... )
|
|
59
|
+
>>> publisher_meta, publisher_abbr, keyword_abbr = generator.generate()
|
|
60
|
+
"""
|
|
61
|
+
|
|
24
62
|
def __init__(
|
|
25
63
|
self,
|
|
26
64
|
conferences_or_journals: str,
|
|
@@ -29,6 +67,17 @@ class GenerateDataDict(object):
|
|
|
29
67
|
for_vue: bool = True,
|
|
30
68
|
path_spidered_conferences_or_journals: Optional[str] = None,
|
|
31
69
|
) -> None:
|
|
70
|
+
"""Initialize the GenerateDataDict instance.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
74
|
+
inproceedings_or_article (str): Publication type ('inproceedings' or 'article').
|
|
75
|
+
json_dict (dict): Input JSON data containing publication information.
|
|
76
|
+
for_vue (bool, optional): Whether to generate Vue.js-compatible format.
|
|
77
|
+
Defaults to True.
|
|
78
|
+
path_spidered_conferences_or_journals (Optional[str], optional): Path to
|
|
79
|
+
spidered conference/journal data. Defaults to None.
|
|
80
|
+
"""
|
|
32
81
|
self.cj = conferences_or_journals
|
|
33
82
|
self.ia = inproceedings_or_article
|
|
34
83
|
self.json_dict = json_dict
|
|
@@ -37,13 +86,22 @@ class GenerateDataDict(object):
|
|
|
37
86
|
self.for_vue = for_vue
|
|
38
87
|
|
|
39
88
|
def generate(self):
|
|
40
|
-
"""
|
|
41
|
-
|
|
89
|
+
"""Generate publisher metadata and keyword-based publication information.
|
|
90
|
+
|
|
91
|
+
This method processes the JSON data to create three main dictionaries:
|
|
92
|
+
1. Publisher metadata with URLs and descriptions
|
|
93
|
+
2. Publisher abbreviation metadata with detailed publication info
|
|
94
|
+
3. Keyword-based metadata for easy searching and categorization
|
|
42
95
|
|
|
43
96
|
Returns:
|
|
44
|
-
tuple:
|
|
45
|
-
-
|
|
46
|
-
-
|
|
97
|
+
tuple: A tuple containing three dictionaries:
|
|
98
|
+
- publisher_meta_dict: Publisher metadata including URLs and descriptions
|
|
99
|
+
- publisher_abbr_meta_dict: Publication details indexed by publisher and abbreviation
|
|
100
|
+
- keyword_abbr_meta_dict: Publication details indexed by keywords
|
|
101
|
+
|
|
102
|
+
Example:
|
|
103
|
+
>>> generator = GenerateDataDict(...)
|
|
104
|
+
>>> pub_meta, pub_abbr, keyword_abbr = generator.generate()
|
|
47
105
|
"""
|
|
48
106
|
publisher_meta_dict, keyword_abbr_meta_dict, publisher_abbr_meta_dict = {}, {}, {}
|
|
49
107
|
|
|
@@ -76,13 +134,15 @@ class GenerateDataDict(object):
|
|
|
76
134
|
remarks = [p for p in self.json_dict[publisher].get("txt_remarks", []) if p.strip()]
|
|
77
135
|
|
|
78
136
|
# Update publisher metadata
|
|
79
|
-
publisher_meta_dict.setdefault(publisher, {}).update(
|
|
80
|
-
|
|
81
|
-
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
|
|
85
|
-
|
|
137
|
+
publisher_meta_dict.setdefault(publisher, {}).update(
|
|
138
|
+
{
|
|
139
|
+
"full_name_url": full_url,
|
|
140
|
+
"txt_abouts": abouts,
|
|
141
|
+
"txt_remarks": remarks,
|
|
142
|
+
"urls_about": urls_about,
|
|
143
|
+
"url_conferences_or_journals": f"[{self.cj.title()}]({urls_cj[0]})" if urls_cj else "",
|
|
144
|
+
}
|
|
145
|
+
)
|
|
86
146
|
|
|
87
147
|
# Process each abbreviation (conference/journal)
|
|
88
148
|
for abbr in self.json_dict[publisher][self.cj]:
|
|
@@ -105,17 +165,31 @@ class GenerateDataDict(object):
|
|
|
105
165
|
return publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict
|
|
106
166
|
|
|
107
167
|
def conference_or_journal(self, publisher_url: str, abbr: str, abbr_dict: dict):
|
|
108
|
-
"""
|
|
109
|
-
|
|
168
|
+
"""Process conference or journal data and generate formatted information.
|
|
169
|
+
|
|
170
|
+
This method processes individual conference or journal data, validates
|
|
171
|
+
name lengths, extracts information, formats URLs, and generates table
|
|
172
|
+
row data for markdown output.
|
|
110
173
|
|
|
111
174
|
Args:
|
|
112
|
-
publisher_url: Publisher's URL
|
|
113
|
-
abbr: Abbreviation identifier
|
|
114
|
-
abbr_dict: Dictionary containing publication details
|
|
175
|
+
publisher_url (str): Publisher's URL for markdown linking.
|
|
176
|
+
abbr (str): Abbreviation identifier for the publication.
|
|
177
|
+
abbr_dict (dict): Dictionary containing publication details including
|
|
178
|
+
names, URLs, dates, scores, and keywords.
|
|
115
179
|
|
|
116
180
|
Returns:
|
|
117
|
-
|
|
118
|
-
|
|
181
|
+
tuple: A tuple containing:
|
|
182
|
+
- dict: Contains formatted about text, remarks, and table row data
|
|
183
|
+
- list: Sorted list of keywords for the publication
|
|
184
|
+
|
|
185
|
+
Raises:
|
|
186
|
+
ValueError: If full and abbreviated names have mismatched lengths.
|
|
187
|
+
|
|
188
|
+
Example:
|
|
189
|
+
>>> result = generator.conference_or_journal(
|
|
190
|
+
... "https://publisher.com", "ICML", conf_data
|
|
191
|
+
... )
|
|
192
|
+
>>> abouts, keywords = result
|
|
119
193
|
"""
|
|
120
194
|
# Validate full and abbreviated names match in length
|
|
121
195
|
self._validate_name_lengths(abbr_dict)
|
|
@@ -138,23 +212,42 @@ class GenerateDataDict(object):
|
|
|
138
212
|
|
|
139
213
|
# Generate appropriate table row based on type
|
|
140
214
|
row_inf = self._generate_table_row(
|
|
141
|
-
publisher_url, full_name, abbr_name, url_home,
|
|
142
|
-
url_about, period, top, keywords_url, abbr, abbr_dict
|
|
215
|
+
publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords_url, abbr, abbr_dict
|
|
143
216
|
)
|
|
144
217
|
|
|
145
218
|
return {"txt_abouts": abouts, "txt_remarks": remarks, "row_inf": row_inf}, keywords
|
|
146
219
|
|
|
147
220
|
def _validate_name_lengths(self, abbr_dict: dict):
|
|
148
|
-
"""Validate that full and abbreviated names arrays have equal length.
|
|
221
|
+
"""Validate that full and abbreviated names arrays have equal length.
|
|
222
|
+
|
|
223
|
+
This method ensures that the full names and abbreviated names arrays
|
|
224
|
+
have the same length, which is required for proper data processing.
|
|
225
|
+
|
|
226
|
+
Args:
|
|
227
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
228
|
+
|
|
229
|
+
Raises:
|
|
230
|
+
ValueError: If the lengths of names_full and names_abbr don't match.
|
|
231
|
+
"""
|
|
149
232
|
full_names = abbr_dict.get("names_full", [])
|
|
150
233
|
abbr_names = abbr_dict.get("names_abbr", [])
|
|
151
234
|
if len(full_names) != len(abbr_names):
|
|
152
|
-
raise ValueError(f"Length mismatch: {len(full_names)}
|
|
235
|
+
raise ValueError(f"Length mismatch: {len(full_names)} {full_names} vs {len(abbr_names)} abbreviated names")
|
|
153
236
|
|
|
154
237
|
def _extract_full_abbr_names(self, abbr_dict: dict):
|
|
155
|
-
"""Extract full and abbreviated names from dictionary.
|
|
238
|
+
"""Extract full and abbreviated names from dictionary.
|
|
239
|
+
|
|
240
|
+
This method extracts the appropriate full and abbreviated names based on
|
|
241
|
+
the publication type (conferences vs journals).
|
|
242
|
+
|
|
243
|
+
Args:
|
|
244
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
245
|
+
|
|
246
|
+
Returns:
|
|
247
|
+
tuple: A tuple containing (full_name, abbr_name).
|
|
248
|
+
"""
|
|
156
249
|
# For journals: use first full name from list; for conferences: use single name
|
|
157
|
-
full_name =
|
|
250
|
+
full_name = abbr_dict.get("names_full", [""])[0] if self.cj == "journals" else abbr_dict.get("name", "")
|
|
158
251
|
abbr_name = abbr_dict.get("names_abbr", [""])[0]
|
|
159
252
|
return full_name, abbr_name
|
|
160
253
|
|
|
@@ -175,7 +268,7 @@ class GenerateDataDict(object):
|
|
|
175
268
|
if acronym_dblp := abbr_dict.get("acronym_dblp", ""):
|
|
176
269
|
journal_conf = "journals" if self.cj == "journals" else "conf"
|
|
177
270
|
dblp_url = f"https://dblp.org/db/{journal_conf}/{acronym_dblp}/index.html"
|
|
178
|
-
period = f
|
|
271
|
+
period = f"[{period}]({dblp_url})"
|
|
179
272
|
|
|
180
273
|
return period
|
|
181
274
|
|
|
@@ -209,9 +302,7 @@ class GenerateDataDict(object):
|
|
|
209
302
|
all_keywords = sorted(set(all_keywords))
|
|
210
303
|
# Create Google search links for each keyword
|
|
211
304
|
google_base = "https://www.google.com/search?q="
|
|
212
|
-
keywords_url = [
|
|
213
|
-
f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords
|
|
214
|
-
]
|
|
305
|
+
keywords_url = [f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords]
|
|
215
306
|
|
|
216
307
|
# For category
|
|
217
308
|
# Flatten keywords and remove duplicates
|
|
@@ -276,19 +367,28 @@ class GenerateDataDict(object):
|
|
|
276
367
|
abstract_due, start_date, today = self._process_conference_dates(abbr_dict)
|
|
277
368
|
|
|
278
369
|
# Format date indicators for Vue or standard display
|
|
279
|
-
abstract_indicator, start_indicator = self._format_date_indicators(
|
|
280
|
-
abstract_due, start_date, today
|
|
281
|
-
)
|
|
370
|
+
abstract_indicator, start_indicator = self._format_date_indicators(abstract_due, start_date, today)
|
|
282
371
|
|
|
283
372
|
# Get year URL for start date link
|
|
284
373
|
year_url = abbr_dict.get("conf_url", "")
|
|
285
374
|
|
|
286
375
|
# Build and return markdown table row
|
|
287
376
|
return self._build_conference_row(
|
|
288
|
-
publisher_url,
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
377
|
+
publisher_url,
|
|
378
|
+
full_name,
|
|
379
|
+
abbr_name,
|
|
380
|
+
url_home,
|
|
381
|
+
url_about,
|
|
382
|
+
archive_display,
|
|
383
|
+
period,
|
|
384
|
+
top,
|
|
385
|
+
abbr_dict,
|
|
386
|
+
abstract_due,
|
|
387
|
+
abstract_indicator,
|
|
388
|
+
start_date,
|
|
389
|
+
start_indicator,
|
|
390
|
+
year_url,
|
|
391
|
+
keywords,
|
|
292
392
|
)
|
|
293
393
|
|
|
294
394
|
def _process_conference_dates(self, abbr_dict):
|
|
@@ -328,8 +428,22 @@ class GenerateDataDict(object):
|
|
|
328
428
|
return abstract_indicator, start_indicator
|
|
329
429
|
|
|
330
430
|
def _build_conference_row(
|
|
331
|
-
self,
|
|
332
|
-
|
|
431
|
+
self,
|
|
432
|
+
publisher_url,
|
|
433
|
+
full_name,
|
|
434
|
+
abbr_name,
|
|
435
|
+
url_home,
|
|
436
|
+
url_about,
|
|
437
|
+
archive_display,
|
|
438
|
+
period,
|
|
439
|
+
top,
|
|
440
|
+
abbr_dict,
|
|
441
|
+
abstract_due,
|
|
442
|
+
abstract_indicator,
|
|
443
|
+
start_date,
|
|
444
|
+
start_indicator,
|
|
445
|
+
year_url,
|
|
446
|
+
keywords,
|
|
333
447
|
):
|
|
334
448
|
"""Construct conference table row string."""
|
|
335
449
|
# Format date strings
|
|
@@ -340,19 +454,21 @@ class GenerateDataDict(object):
|
|
|
340
454
|
start_date_display = self._format_link(start_date_str, year_url) if start_date_str else ""
|
|
341
455
|
|
|
342
456
|
# Build table row
|
|
343
|
-
return
|
|
344
|
-
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
457
|
+
return (
|
|
458
|
+
f"|{publisher_url}|"
|
|
459
|
+
f"{self._format_link(full_name, url_home)}|"
|
|
460
|
+
f"{self._format_link(abbr_name, url_about)}|"
|
|
461
|
+
f"{archive_display}|"
|
|
462
|
+
f"{period}|"
|
|
463
|
+
f"{top}|"
|
|
464
|
+
f"{abbr_dict.get('score_ccf', '')}|"
|
|
465
|
+
f"{abstract_date_str}|"
|
|
466
|
+
f"{abstract_indicator}|"
|
|
467
|
+
f"{start_date_display}|"
|
|
468
|
+
f"{start_indicator}|"
|
|
469
|
+
f"{abbr_dict.get('conf_location', '').strip()}|"
|
|
470
|
+
f"{'; '.join(keywords)}|"
|
|
471
|
+
)
|
|
356
472
|
|
|
357
473
|
# Journals
|
|
358
474
|
def _generate_for_journal(
|
|
@@ -382,25 +498,26 @@ class GenerateDataDict(object):
|
|
|
382
498
|
|
|
383
499
|
# Build and return markdown table row
|
|
384
500
|
return self._build_journal_row(
|
|
385
|
-
publisher_url, full_name, abbr_name, url_home, url_about,
|
|
386
|
-
issues_display, period, top, abbr_dict, keywords
|
|
501
|
+
publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
|
|
387
502
|
)
|
|
388
503
|
|
|
389
504
|
def _build_journal_row(
|
|
390
505
|
self, publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
|
|
391
506
|
):
|
|
392
507
|
"""Construct journal table row string."""
|
|
393
|
-
return
|
|
394
|
-
|
|
395
|
-
|
|
396
|
-
|
|
397
|
-
|
|
398
|
-
|
|
399
|
-
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
508
|
+
return (
|
|
509
|
+
f"|{publisher_url}|"
|
|
510
|
+
f"{self._format_link(full_name, url_home)}|"
|
|
511
|
+
f"{self._format_link(abbr_name, url_about)}|"
|
|
512
|
+
f"{issues_display}|"
|
|
513
|
+
f"{period}|"
|
|
514
|
+
f"{top}|"
|
|
515
|
+
f"{abbr_dict.get('score_ccf', '')}|"
|
|
516
|
+
f"{abbr_dict.get('score_cas', '')}|"
|
|
517
|
+
f"{abbr_dict.get('score_jcr', '')}|"
|
|
518
|
+
f"{abbr_dict.get('score_if', '')}|"
|
|
519
|
+
f"{'; '.join(keywords)}|"
|
|
520
|
+
)
|
|
404
521
|
|
|
405
522
|
# Mermaid data
|
|
406
523
|
def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
|
|
@@ -413,7 +530,7 @@ class GenerateDataDict(object):
|
|
|
413
530
|
mermaid, data_dict = [], {}
|
|
414
531
|
# |AAAI|1980|95|Proceedings of the First National Conference on Artificial Intelligence|
|
|
415
532
|
regex = re.compile(r"\|.*\|([0-9]+)\|([0-9]+)\|.*\|")
|
|
416
|
-
with open(full_readme, "r") as file:
|
|
533
|
+
with open(full_readme, "r", encoding="utf-8") as file:
|
|
417
534
|
data_list = file.readlines()
|
|
418
535
|
for line in data_list:
|
|
419
536
|
if mch := regex.search(line):
|
|
@@ -425,15 +542,15 @@ class GenerateDataDict(object):
|
|
|
425
542
|
mermaid = ["```mermaid\n"]
|
|
426
543
|
mermaid.extend(
|
|
427
544
|
[
|
|
428
|
-
|
|
429
|
-
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
|
|
433
|
-
|
|
434
|
-
|
|
545
|
+
"---\n",
|
|
546
|
+
"config:\n",
|
|
547
|
+
" xyChart:\n",
|
|
548
|
+
" width: 1200\n",
|
|
549
|
+
" height: 600\n",
|
|
550
|
+
" themeVariables:\n",
|
|
551
|
+
" xyChart:\n",
|
|
435
552
|
' titleColor: "#ff0000"\n',
|
|
436
|
-
|
|
553
|
+
"---\n",
|
|
437
554
|
]
|
|
438
555
|
)
|
|
439
556
|
mermaid.extend(["xychart-beta\n", f' title "{abbr}"\n'])
|
|
@@ -8,16 +8,55 @@ from .generate_dict import conference_journal_header
|
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
def conference_journal_informations():
|
|
11
|
+
"""Generate informational content for conferences and journals.
|
|
12
|
+
|
|
13
|
+
This function provides additional informational content that can be
|
|
14
|
+
included in markdown documentation for conferences and journals.
|
|
15
|
+
|
|
16
|
+
Returns:
|
|
17
|
+
tuple: A tuple containing two lists:
|
|
18
|
+
- conference_inf: List of informational strings for conferences
|
|
19
|
+
- journal_inf: List of informational strings for journals
|
|
20
|
+
|
|
21
|
+
Example:
|
|
22
|
+
>>> conf_info, journal_info = conference_journal_informations()
|
|
23
|
+
>>> print(conf_info[0])
|
|
24
|
+
!> [List of Upcoming International Conferences](...)
|
|
25
|
+
"""
|
|
11
26
|
conference_inf = [
|
|
12
27
|
"!> [List of Upcoming International Conferences](https://internationalconferencealerts.com/all-events.php)\n\n",
|
|
13
|
-
"!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n"
|
|
28
|
+
"!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n",
|
|
14
29
|
]
|
|
15
30
|
journal_inf = []
|
|
16
31
|
return conference_inf, journal_inf
|
|
17
32
|
|
|
18
33
|
|
|
19
34
|
class WriteDataToMd(object):
|
|
20
|
-
"""Class to write publication data to Markdown files.
|
|
35
|
+
"""Class to write publication data to Markdown files.
|
|
36
|
+
|
|
37
|
+
This class provides methods to generate various markdown documentation files
|
|
38
|
+
from processed publication data, including introduction files, categorized
|
|
39
|
+
listings, publisher information, and statistics.
|
|
40
|
+
|
|
41
|
+
Attributes:
|
|
42
|
+
cj (str): Type of publication ('conferences' or 'journals').
|
|
43
|
+
ia (str): Publication type ('inproceedings' or 'article').
|
|
44
|
+
publisher_meta_dict (dict): Publisher metadata dictionary.
|
|
45
|
+
publisher_abbr_meta_dict (dict): Publisher abbreviation metadata dictionary.
|
|
46
|
+
keyword_abbr_meta_dict (dict): Keyword-based metadata dictionary.
|
|
47
|
+
path_output (str): Output directory path for generated files.
|
|
48
|
+
|
|
49
|
+
Example:
|
|
50
|
+
>>> writer = WriteDataToMd(
|
|
51
|
+
... conferences_or_journals="conferences",
|
|
52
|
+
... inproceedings_or_article="inproceedings",
|
|
53
|
+
... publisher_meta_dict=pub_meta,
|
|
54
|
+
... publisher_abbr_meta_dict=pub_abbr,
|
|
55
|
+
... keyword_abbr_meta_dict=keyword_abbr,
|
|
56
|
+
... path_output="/output"
|
|
57
|
+
... )
|
|
58
|
+
>>> writer.save_introductions()
|
|
59
|
+
"""
|
|
21
60
|
|
|
22
61
|
def __init__(
|
|
23
62
|
self,
|
|
@@ -28,7 +67,16 @@ class WriteDataToMd(object):
|
|
|
28
67
|
keyword_abbr_meta_dict: dict,
|
|
29
68
|
path_output: str,
|
|
30
69
|
) -> None:
|
|
31
|
-
"""Initialize with publication data and output path.
|
|
70
|
+
"""Initialize with publication data and output path.
|
|
71
|
+
|
|
72
|
+
Args:
|
|
73
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
74
|
+
inproceedings_or_article (str): Publication type ('inproceedings' or 'article').
|
|
75
|
+
publisher_meta_dict (dict): Publisher metadata dictionary.
|
|
76
|
+
publisher_abbr_meta_dict (dict): Publisher abbreviation metadata dictionary.
|
|
77
|
+
keyword_abbr_meta_dict (dict): Keyword-based metadata dictionary.
|
|
78
|
+
path_output (str): Output directory path for generated files.
|
|
79
|
+
"""
|
|
32
80
|
self.cj = conferences_or_journals # "conferences" or "journals"
|
|
33
81
|
self.ia = inproceedings_or_article # "inproceedings" or "article"
|
|
34
82
|
self.publisher_meta_dict = publisher_meta_dict
|
|
@@ -41,7 +89,18 @@ class WriteDataToMd(object):
|
|
|
41
89
|
]
|
|
42
90
|
|
|
43
91
|
def save_introductions(self) -> None:
|
|
44
|
-
"""Save introduction file with all conferences/journals list.
|
|
92
|
+
"""Save introduction file with all conferences/journals list.
|
|
93
|
+
|
|
94
|
+
This method generates a comprehensive markdown file containing all
|
|
95
|
+
conferences or journals in a tabular format with appropriate headers
|
|
96
|
+
and informational content.
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
None: This method does not return a value.
|
|
100
|
+
|
|
101
|
+
Note:
|
|
102
|
+
The output file is saved as 'Introductions_{type}.md' in the output directory.
|
|
103
|
+
"""
|
|
45
104
|
conference_header, journal_header = conference_journal_header()
|
|
46
105
|
conference_inf, journal_inf = conference_journal_informations()
|
|
47
106
|
|
|
@@ -62,13 +121,13 @@ class WriteDataToMd(object):
|
|
|
62
121
|
idx = 1
|
|
63
122
|
for publisher in self.publisher_abbr_meta_dict:
|
|
64
123
|
for abbr in self.publisher_abbr_meta_dict[publisher]:
|
|
65
|
-
row_info = self.publisher_abbr_meta_dict[publisher][abbr][
|
|
124
|
+
row_info = self.publisher_abbr_meta_dict[publisher][abbr]["row_inf"]
|
|
66
125
|
data_list.append(f"|{idx}{row_info}\n")
|
|
67
126
|
idx += 1
|
|
68
127
|
|
|
69
128
|
# Write to file
|
|
70
129
|
output_file = os.path.join(self.path_output, f"Introductions_{self.cj.title()}.md")
|
|
71
|
-
with open(output_file, "w") as f:
|
|
130
|
+
with open(output_file, "w", encoding="utf-8") as f:
|
|
72
131
|
f.writelines(data_list)
|
|
73
132
|
|
|
74
133
|
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
@@ -88,7 +147,21 @@ class WriteDataToMd(object):
|
|
|
88
147
|
|
|
89
148
|
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
90
149
|
def save_categories(self, keywords_category_name: str, keywords_list: List[str]) -> None:
|
|
91
|
-
"""Save publications categorized by keywords.
|
|
150
|
+
"""Save publications categorized by keywords.
|
|
151
|
+
|
|
152
|
+
This method generates markdown files organizing publications by their
|
|
153
|
+
keywords, creating separate sections for each keyword category.
|
|
154
|
+
|
|
155
|
+
Args:
|
|
156
|
+
keywords_category_name (str): The category name for keywords filtering.
|
|
157
|
+
keywords_list (List[str]): List of keywords to include in the output.
|
|
158
|
+
|
|
159
|
+
Returns:
|
|
160
|
+
None: This method does not return a value.
|
|
161
|
+
|
|
162
|
+
Note:
|
|
163
|
+
The output file is saved as 'Categories_{type}_{category}.md' in the output directory.
|
|
164
|
+
"""
|
|
92
165
|
conference_header, journal_header = conference_journal_header()
|
|
93
166
|
data_list = [f"# {self.cj.title()}\n\n"]
|
|
94
167
|
data_list.extend(self._default_inf)
|
|
@@ -110,7 +183,9 @@ class WriteDataToMd(object):
|
|
|
110
183
|
|
|
111
184
|
# Write to file
|
|
112
185
|
category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
|
|
113
|
-
with open(
|
|
186
|
+
with open(
|
|
187
|
+
os.path.join(self.path_output, f"Categories_{self.cj.title()}{category_postfix}.md"), "w", encoding="utf-8"
|
|
188
|
+
) as f:
|
|
114
189
|
f.writelines(data_list)
|
|
115
190
|
|
|
116
191
|
return None
|
|
@@ -136,18 +211,31 @@ class WriteDataToMd(object):
|
|
|
136
211
|
|
|
137
212
|
# Write keyword-specific file
|
|
138
213
|
path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
|
|
139
|
-
|
|
214
|
+
# Create safe filename by replacing invalid characters
|
|
215
|
+
safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
|
|
216
|
+
with open(os.path.join(path_key, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
|
|
140
217
|
f.writelines(data_list)
|
|
141
218
|
|
|
142
219
|
return None
|
|
143
220
|
|
|
144
221
|
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
145
222
|
def save_publishers(self) -> None:
|
|
146
|
-
"""Save publisher overview file with basic information.
|
|
223
|
+
"""Save publisher overview file with basic information.
|
|
224
|
+
|
|
225
|
+
This method generates a markdown file containing an overview of all
|
|
226
|
+
publishers with their basic information, about pages, and links to
|
|
227
|
+
detailed publisher-specific files.
|
|
228
|
+
|
|
229
|
+
Returns:
|
|
230
|
+
None: This method does not return a value.
|
|
231
|
+
|
|
232
|
+
Note:
|
|
233
|
+
The output file is saved as 'Publishers_{type}.md' in the output directory.
|
|
234
|
+
"""
|
|
147
235
|
data_list_pub = [
|
|
148
236
|
f"# Introductions of Publishers and {self.cj.title()}\n\n",
|
|
149
237
|
"| |Publishers|About US|Conferences/Journals|Separate Links|\n",
|
|
150
|
-
"|-|- |- |- |- |\n"
|
|
238
|
+
"|-|- |- |- |- |\n",
|
|
151
239
|
]
|
|
152
240
|
idx = 1
|
|
153
241
|
|
|
@@ -171,7 +259,7 @@ class WriteDataToMd(object):
|
|
|
171
259
|
idx += 1
|
|
172
260
|
|
|
173
261
|
# Write to file
|
|
174
|
-
with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w") as f:
|
|
262
|
+
with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w", encoding="utf-8") as f:
|
|
175
263
|
f.writelines(data_list_pub)
|
|
176
264
|
return None
|
|
177
265
|
|
|
@@ -205,7 +293,7 @@ class WriteDataToMd(object):
|
|
|
205
293
|
|
|
206
294
|
# Add row information
|
|
207
295
|
row_info = self.publisher_abbr_meta_dict[pub][abbr]["row_inf"]
|
|
208
|
-
data_list.append(f
|
|
296
|
+
data_list.append(f"{row_info}\n\n")
|
|
209
297
|
|
|
210
298
|
# Add remarks and about for this abbreviation
|
|
211
299
|
for flag in ["txt_remarks", "txt_abouts"]:
|
|
@@ -222,7 +310,7 @@ class WriteDataToMd(object):
|
|
|
222
310
|
|
|
223
311
|
# Write publisher-specific file
|
|
224
312
|
path_pub = standardize_path(os.path.join(self.path_output, f"Publishers_{self.cj.title()}"))
|
|
225
|
-
with open(os.path.join(path_pub, f"{pub}.md"), "w") as f:
|
|
313
|
+
with open(os.path.join(path_pub, f"{pub}.md"), "w", encoding="utf-8") as f:
|
|
226
314
|
f.writelines(data_list)
|
|
227
315
|
|
|
228
316
|
return None
|
|
@@ -232,13 +320,15 @@ class WriteDataToMd(object):
|
|
|
232
320
|
data_list = [
|
|
233
321
|
f"# Statistics of keywords in {self.cj.title()}\n\n",
|
|
234
322
|
"| |keywords|Separate Links|\n",
|
|
235
|
-
"|-|- |- |\n"
|
|
323
|
+
"|-|- |- |\n",
|
|
236
324
|
]
|
|
237
325
|
idx = 1
|
|
238
326
|
|
|
239
327
|
# Add publications for each category
|
|
240
328
|
for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
|
|
241
|
-
|
|
329
|
+
# Create safe filename for URL
|
|
330
|
+
safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
|
|
331
|
+
local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{safe_keyword}.md)"
|
|
242
332
|
|
|
243
333
|
# Create table row
|
|
244
334
|
row = f"| {idx} | {keyword} | {local_url} |\n"
|
|
@@ -247,7 +337,9 @@ class WriteDataToMd(object):
|
|
|
247
337
|
|
|
248
338
|
# Write to file
|
|
249
339
|
category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
|
|
250
|
-
with open(
|
|
340
|
+
with open(
|
|
341
|
+
os.path.join(self.path_output, f"Statistics_{self.cj.title()}{category_postfix}.md"), "w", encoding="utf-8"
|
|
342
|
+
) as f:
|
|
251
343
|
f.writelines(data_list)
|
|
252
344
|
return None
|
|
253
345
|
|
|
@@ -269,7 +361,7 @@ class WriteDataToMd(object):
|
|
|
269
361
|
|
|
270
362
|
# Add row information
|
|
271
363
|
row_info = self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"]
|
|
272
|
-
data_list.append(f
|
|
364
|
+
data_list.append(f"{row_info}\n\n")
|
|
273
365
|
|
|
274
366
|
# Add statistics if available
|
|
275
367
|
if statistics := self.keyword_abbr_meta_dict[keyword][abbr].get("statistics", []):
|
|
@@ -278,7 +370,9 @@ class WriteDataToMd(object):
|
|
|
278
370
|
|
|
279
371
|
# Write publisher-specific file
|
|
280
372
|
path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
|
|
281
|
-
|
|
373
|
+
# Create safe filename by replacing invalid characters
|
|
374
|
+
safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
|
|
375
|
+
with open(os.path.join(path_pub, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
|
|
282
376
|
f.writelines(data_list)
|
|
283
377
|
|
|
284
378
|
return None
|
|
File without changes
|
|
@@ -1,103 +0,0 @@
|
|
|
1
|
-
# coding=utf-8
|
|
2
|
-
|
|
3
|
-
import os
|
|
4
|
-
import re
|
|
5
|
-
from typing import List
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
def split_text_by_length(text, max_length=120) -> List[str]:
|
|
9
|
-
lines = []
|
|
10
|
-
while text:
|
|
11
|
-
if len(text) <= max_length:
|
|
12
|
-
lines.append(text)
|
|
13
|
-
break
|
|
14
|
-
|
|
15
|
-
split_pos = text.rfind(" ", 0, max_length + 1)
|
|
16
|
-
if split_pos == -1:
|
|
17
|
-
split_pos = max_length
|
|
18
|
-
|
|
19
|
-
line = text[:split_pos]
|
|
20
|
-
lines.append(line)
|
|
21
|
-
|
|
22
|
-
text = text[split_pos:]
|
|
23
|
-
|
|
24
|
-
new_lines = []
|
|
25
|
-
for line in lines:
|
|
26
|
-
new_lines.append(line)
|
|
27
|
-
return new_lines
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
def split_data_list(
|
|
31
|
-
split_pattern: str, data_list: List[str], last_next: str = "next"
|
|
32
|
-
) -> List[str]:
|
|
33
|
-
r"""Split data list according to the split pattern.
|
|
34
|
-
|
|
35
|
-
The capturing parentheses must be used in the pattern, such as `(\n)`.
|
|
36
|
-
|
|
37
|
-
Args:
|
|
38
|
-
split_pattern (str): split pattern.
|
|
39
|
-
data_list (List[str]): data list.
|
|
40
|
-
last_next (str): "next" or "last".
|
|
41
|
-
|
|
42
|
-
Returns:
|
|
43
|
-
List[str]: new data list.
|
|
44
|
-
|
|
45
|
-
Examples:
|
|
46
|
-
split_pattern = r"(\n)", last_next = "next" or "last".
|
|
47
|
-
"""
|
|
48
|
-
new_data_list = []
|
|
49
|
-
for line in data_list:
|
|
50
|
-
split_list = re.split(split_pattern, line)
|
|
51
|
-
list_one = split_list[0: len(split_list): 2]
|
|
52
|
-
list_two = split_list[1: len(split_list): 2]
|
|
53
|
-
|
|
54
|
-
temp = []
|
|
55
|
-
if last_next == "next":
|
|
56
|
-
list_two.insert(0, "")
|
|
57
|
-
temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
|
|
58
|
-
if last_next == "last":
|
|
59
|
-
list_two.append("")
|
|
60
|
-
temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
|
|
61
|
-
new_data_list.extend(temp)
|
|
62
|
-
new_data_list = [line for line in new_data_list if line.strip()]
|
|
63
|
-
return new_data_list
|
|
64
|
-
|
|
65
|
-
|
|
66
|
-
def standardize_path(path_input: str) -> str:
|
|
67
|
-
path_input = os.path.expandvars(os.path.expanduser(path_input))
|
|
68
|
-
if not os.path.exists(path_input):
|
|
69
|
-
os.makedirs(path_input)
|
|
70
|
-
return path_input
|
|
71
|
-
|
|
72
|
-
|
|
73
|
-
def sort_strings_with_embedded_numbers(s: str) -> List[str]:
|
|
74
|
-
re_digits = re.compile(r"(\d+)")
|
|
75
|
-
pieces = re_digits.split(s)
|
|
76
|
-
pieces[1::2] = map(int, pieces[1::2])
|
|
77
|
-
return pieces
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
|
|
81
|
-
return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
|
|
82
|
-
|
|
83
|
-
|
|
84
|
-
class IterateSortDict(object):
|
|
85
|
-
def __init__(self, reverse: bool = False) -> None:
|
|
86
|
-
self.reverse = reverse
|
|
87
|
-
|
|
88
|
-
def dict_update(self, old):
|
|
89
|
-
"""Update."""
|
|
90
|
-
old = self.dict_sort_iteration(old)
|
|
91
|
-
old = self.dict_sort(old)
|
|
92
|
-
return old
|
|
93
|
-
|
|
94
|
-
def dict_sort_iteration(self, old: dict):
|
|
95
|
-
"""Sort."""
|
|
96
|
-
for key in old:
|
|
97
|
-
if isinstance(old[key], dict):
|
|
98
|
-
old[key] = self.dict_update(old[key])
|
|
99
|
-
return old
|
|
100
|
-
|
|
101
|
-
def dict_sort(self, old: dict):
|
|
102
|
-
"""Sort."""
|
|
103
|
-
return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
|
|
@@ -1,79 +0,0 @@
|
|
|
1
|
-
# coding=utf-8
|
|
2
|
-
|
|
3
|
-
import os
|
|
4
|
-
from typing import Optional
|
|
5
|
-
|
|
6
|
-
from .core._base import standardize_path
|
|
7
|
-
from .core.update_json import load_json_data, update_json_file
|
|
8
|
-
from .tools.generate_dict import GenerateDataDict
|
|
9
|
-
from .tools.write_dict import WriteDataToMd
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
def main_generate_md_files(
|
|
13
|
-
path_json: str,
|
|
14
|
-
path_output: str,
|
|
15
|
-
path_spidered_bibs: Optional[str] = None,
|
|
16
|
-
for_vue: bool = True,
|
|
17
|
-
conferences_or_journals: Optional[str] = None,
|
|
18
|
-
keywords_category_name: str = ""
|
|
19
|
-
) -> None:
|
|
20
|
-
"""
|
|
21
|
-
Generate markdown files for conferences and journals.
|
|
22
|
-
|
|
23
|
-
Args:
|
|
24
|
-
path_json: Path to JSON data file
|
|
25
|
-
path_output: Output directory for markdown files
|
|
26
|
-
path_spidered_bibs: Directory containing crawled BibTeX files
|
|
27
|
-
for_vue: Whether to generate Vue-compatible format
|
|
28
|
-
conferences_or_journals: Specify 'conferences' or 'journals', None for both
|
|
29
|
-
keywords_category_name: The category name of keywords
|
|
30
|
-
"""
|
|
31
|
-
# Standardize all paths
|
|
32
|
-
path_json = standardize_path(path_json)
|
|
33
|
-
path_output = standardize_path(path_output)
|
|
34
|
-
path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
|
|
35
|
-
|
|
36
|
-
# Process keyword category name and load data
|
|
37
|
-
keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
|
|
38
|
-
category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
|
|
39
|
-
keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
|
|
40
|
-
|
|
41
|
-
# Validate data availability
|
|
42
|
-
if not keywords_list or not keywords_category_name:
|
|
43
|
-
keywords_list, keywords_category_name = [], ""
|
|
44
|
-
|
|
45
|
-
# Process both conferences and journals
|
|
46
|
-
for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
|
|
47
|
-
# Skip if specific type requested and doesn't match
|
|
48
|
-
if conferences_or_journals and conferences_or_journals.lower() != cj:
|
|
49
|
-
continue
|
|
50
|
-
|
|
51
|
-
# Update JSON data
|
|
52
|
-
json_dict = update_json_file(path_json, cj)
|
|
53
|
-
if not json_dict:
|
|
54
|
-
continue
|
|
55
|
-
|
|
56
|
-
# Generate data dictionaries
|
|
57
|
-
path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
|
|
58
|
-
generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
|
|
59
|
-
publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
|
|
60
|
-
if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
|
|
61
|
-
continue
|
|
62
|
-
|
|
63
|
-
# Initialize writer and save all markdown files
|
|
64
|
-
_path_output = os.path.join(path_output, f"{cj.title()}")
|
|
65
|
-
save_data = WriteDataToMd(
|
|
66
|
-
cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
|
|
67
|
-
)
|
|
68
|
-
# Save various documentation files
|
|
69
|
-
save_data.save_introductions()
|
|
70
|
-
save_data.save_categories(keywords_category_name, keywords_list)
|
|
71
|
-
save_data.save_categories_separate_keywords()
|
|
72
|
-
|
|
73
|
-
save_data.save_publishers()
|
|
74
|
-
save_data.save_publishers_separate_abbrs()
|
|
75
|
-
|
|
76
|
-
save_data.save_statistics(keywords_category_name, keywords_list)
|
|
77
|
-
save_data.save_statistics_separate_abbrs()
|
|
78
|
-
|
|
79
|
-
return None
|
|
File without changes
|