pyformatjson 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- pyformatjson/__init__.py +0 -0
- pyformatjson/core/_base.py +103 -0
- pyformatjson/core/update_json.py +90 -0
- pyformatjson/main.py +79 -0
- pyformatjson/tools/generate_dict.py +456 -0
- pyformatjson/tools/write_dict.py +284 -0
- pyformatjson-0.0.1.dist-info/METADATA +23 -0
- pyformatjson-0.0.1.dist-info/RECORD +9 -0
- pyformatjson-0.0.1.dist-info/WHEEL +4 -0
pyformatjson/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
from typing import List
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
def split_text_by_length(text, max_length=120) -> List[str]:
|
|
9
|
+
lines = []
|
|
10
|
+
while text:
|
|
11
|
+
if len(text) <= max_length:
|
|
12
|
+
lines.append(text)
|
|
13
|
+
break
|
|
14
|
+
|
|
15
|
+
split_pos = text.rfind(" ", 0, max_length + 1)
|
|
16
|
+
if split_pos == -1:
|
|
17
|
+
split_pos = max_length
|
|
18
|
+
|
|
19
|
+
line = text[:split_pos]
|
|
20
|
+
lines.append(line)
|
|
21
|
+
|
|
22
|
+
text = text[split_pos:]
|
|
23
|
+
|
|
24
|
+
new_lines = []
|
|
25
|
+
for line in lines:
|
|
26
|
+
new_lines.append(line)
|
|
27
|
+
return new_lines
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def split_data_list(
|
|
31
|
+
split_pattern: str, data_list: List[str], last_next: str = "next"
|
|
32
|
+
) -> List[str]:
|
|
33
|
+
r"""Split data list according to the split pattern.
|
|
34
|
+
|
|
35
|
+
The capturing parentheses must be used in the pattern, such as `(\n)`.
|
|
36
|
+
|
|
37
|
+
Args:
|
|
38
|
+
split_pattern (str): split pattern.
|
|
39
|
+
data_list (List[str]): data list.
|
|
40
|
+
last_next (str): "next" or "last".
|
|
41
|
+
|
|
42
|
+
Returns:
|
|
43
|
+
List[str]: new data list.
|
|
44
|
+
|
|
45
|
+
Examples:
|
|
46
|
+
split_pattern = r"(\n)", last_next = "next" or "last".
|
|
47
|
+
"""
|
|
48
|
+
new_data_list = []
|
|
49
|
+
for line in data_list:
|
|
50
|
+
split_list = re.split(split_pattern, line)
|
|
51
|
+
list_one = split_list[0: len(split_list): 2]
|
|
52
|
+
list_two = split_list[1: len(split_list): 2]
|
|
53
|
+
|
|
54
|
+
temp = []
|
|
55
|
+
if last_next == "next":
|
|
56
|
+
list_two.insert(0, "")
|
|
57
|
+
temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
|
|
58
|
+
if last_next == "last":
|
|
59
|
+
list_two.append("")
|
|
60
|
+
temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
|
|
61
|
+
new_data_list.extend(temp)
|
|
62
|
+
new_data_list = [line for line in new_data_list if line.strip()]
|
|
63
|
+
return new_data_list
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def standardize_path(path_input: str) -> str:
|
|
67
|
+
path_input = os.path.expandvars(os.path.expanduser(path_input))
|
|
68
|
+
if not os.path.exists(path_input):
|
|
69
|
+
os.makedirs(path_input)
|
|
70
|
+
return path_input
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def sort_strings_with_embedded_numbers(s: str) -> List[str]:
|
|
74
|
+
re_digits = re.compile(r"(\d+)")
|
|
75
|
+
pieces = re_digits.split(s)
|
|
76
|
+
pieces[1::2] = map(int, pieces[1::2])
|
|
77
|
+
return pieces
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
|
|
81
|
+
return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class IterateSortDict(object):
|
|
85
|
+
def __init__(self, reverse: bool = False) -> None:
|
|
86
|
+
self.reverse = reverse
|
|
87
|
+
|
|
88
|
+
def dict_update(self, old):
|
|
89
|
+
"""Update."""
|
|
90
|
+
old = self.dict_sort_iteration(old)
|
|
91
|
+
old = self.dict_sort(old)
|
|
92
|
+
return old
|
|
93
|
+
|
|
94
|
+
def dict_sort_iteration(self, old: dict):
|
|
95
|
+
"""Sort."""
|
|
96
|
+
for key in old:
|
|
97
|
+
if isinstance(old[key], dict):
|
|
98
|
+
old[key] = self.dict_update(old[key])
|
|
99
|
+
return old
|
|
100
|
+
|
|
101
|
+
def dict_sort(self, old: dict):
|
|
102
|
+
"""Sort."""
|
|
103
|
+
return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from typing import Any, Dict
|
|
6
|
+
|
|
7
|
+
from ._base import split_data_list, split_text_by_length
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def load_json_data(path_json: str, filename: str) -> Dict:
|
|
11
|
+
"""Load JSON data from file."""
|
|
12
|
+
try:
|
|
13
|
+
file_path = os.path.join(path_json, f"{filename}.json")
|
|
14
|
+
if not os.path.exists(file_path):
|
|
15
|
+
return {}
|
|
16
|
+
|
|
17
|
+
with open(file_path, "r", encoding="utf-8") as file:
|
|
18
|
+
return json.load(file)
|
|
19
|
+
|
|
20
|
+
except Exception as e:
|
|
21
|
+
print(f"Error loading {filename}.json: {e}")
|
|
22
|
+
return {}
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
|
|
26
|
+
"""
|
|
27
|
+
Update and format JSON file containing conference/journal data.
|
|
28
|
+
|
|
29
|
+
Args:
|
|
30
|
+
path_root (str): Root directory path
|
|
31
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals')
|
|
32
|
+
|
|
33
|
+
Returns:
|
|
34
|
+
Dict[str, Any]: Processed JSON data
|
|
35
|
+
"""
|
|
36
|
+
# Load Json Data
|
|
37
|
+
json_dict = load_json_data(path_root, conferences_or_journals)
|
|
38
|
+
|
|
39
|
+
# Process and format text fields in JSON data.
|
|
40
|
+
for pub in json_dict:
|
|
41
|
+
for flag in ["txt_abouts", "txt_remarks"]:
|
|
42
|
+
data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
|
|
43
|
+
temps = []
|
|
44
|
+
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
45
|
+
temps.extend(split_text_by_length(line, 105))
|
|
46
|
+
if temps:
|
|
47
|
+
json_dict[pub].update({flag: temps})
|
|
48
|
+
|
|
49
|
+
for abbr in json_dict[pub][conferences_or_journals]:
|
|
50
|
+
for flag in ["txt_abouts", "txt_remarks"]:
|
|
51
|
+
data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
|
|
52
|
+
temps = []
|
|
53
|
+
for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
|
|
54
|
+
temps.extend(split_text_by_length(line, 97))
|
|
55
|
+
if temps:
|
|
56
|
+
json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
|
|
57
|
+
|
|
58
|
+
# Check for duplicate abbreviations
|
|
59
|
+
_check_duplicate_abbr(json_dict, conferences_or_journals)
|
|
60
|
+
|
|
61
|
+
# Save updated JSON
|
|
62
|
+
if json_dict:
|
|
63
|
+
path_file = os.path.join(path_root, f"{conferences_or_journals}.json")
|
|
64
|
+
with open(path_file, "w", encoding="utf-8") as f:
|
|
65
|
+
f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
66
|
+
|
|
67
|
+
return json_dict
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
|
|
71
|
+
"""
|
|
72
|
+
Check for duplicate abbreviations in the data.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
json_dict: JSON data dictionary
|
|
76
|
+
conferences_or_journals (str): Type of publication ('conferences' or 'journals')
|
|
77
|
+
|
|
78
|
+
Raises:
|
|
79
|
+
ValueError: If duplicate abbreviations are found
|
|
80
|
+
"""
|
|
81
|
+
abbr_list = []
|
|
82
|
+
|
|
83
|
+
for pub in json_dict:
|
|
84
|
+
if conferences_or_journals in json_dict[pub]:
|
|
85
|
+
for abbr in json_dict[pub][conferences_or_journals]:
|
|
86
|
+
if abbr in abbr_list:
|
|
87
|
+
raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
|
|
88
|
+
abbr_list.append(abbr)
|
|
89
|
+
|
|
90
|
+
return None
|
pyformatjson/main.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
from .core._base import standardize_path
|
|
7
|
+
from .core.update_json import load_json_data, update_json_file
|
|
8
|
+
from .tools.generate_dict import GenerateDataDict
|
|
9
|
+
from .tools.write_dict import WriteDataToMd
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main_generate_md_files(
|
|
13
|
+
path_json: str,
|
|
14
|
+
path_output: str,
|
|
15
|
+
path_spidered_bibs: Optional[str] = None,
|
|
16
|
+
for_vue: bool = True,
|
|
17
|
+
conferences_or_journals: Optional[str] = None,
|
|
18
|
+
keywords_category_name: str = ""
|
|
19
|
+
) -> None:
|
|
20
|
+
"""
|
|
21
|
+
Generate markdown files for conferences and journals.
|
|
22
|
+
|
|
23
|
+
Args:
|
|
24
|
+
path_json: Path to JSON data file
|
|
25
|
+
path_output: Output directory for markdown files
|
|
26
|
+
path_spidered_bibs: Directory containing crawled BibTeX files
|
|
27
|
+
for_vue: Whether to generate Vue-compatible format
|
|
28
|
+
conferences_or_journals: Specify 'conferences' or 'journals', None for both
|
|
29
|
+
keywords_category_name: The category name of keywords
|
|
30
|
+
"""
|
|
31
|
+
# Standardize all paths
|
|
32
|
+
path_json = standardize_path(path_json)
|
|
33
|
+
path_output = standardize_path(path_output)
|
|
34
|
+
path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
|
|
35
|
+
|
|
36
|
+
# Process keyword category name and load data
|
|
37
|
+
keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
|
|
38
|
+
category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
|
|
39
|
+
keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
|
|
40
|
+
|
|
41
|
+
# Validate data availability
|
|
42
|
+
if not keywords_list or not keywords_category_name:
|
|
43
|
+
keywords_list, keywords_category_name = [], ""
|
|
44
|
+
|
|
45
|
+
# Process both conferences and journals
|
|
46
|
+
for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
|
|
47
|
+
# Skip if specific type requested and doesn't match
|
|
48
|
+
if conferences_or_journals and conferences_or_journals.lower() != cj:
|
|
49
|
+
continue
|
|
50
|
+
|
|
51
|
+
# Update JSON data
|
|
52
|
+
json_dict = update_json_file(path_json, cj)
|
|
53
|
+
if not json_dict:
|
|
54
|
+
continue
|
|
55
|
+
|
|
56
|
+
# Generate data dictionaries
|
|
57
|
+
path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
|
|
58
|
+
generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
|
|
59
|
+
publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
|
|
60
|
+
if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
|
|
61
|
+
continue
|
|
62
|
+
|
|
63
|
+
# Initialize writer and save all markdown files
|
|
64
|
+
_path_output = os.path.join(path_output, f"{cj.title()}")
|
|
65
|
+
save_data = WriteDataToMd(
|
|
66
|
+
cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
|
|
67
|
+
)
|
|
68
|
+
# Save various documentation files
|
|
69
|
+
save_data.save_introductions()
|
|
70
|
+
save_data.save_categories(keywords_category_name, keywords_list)
|
|
71
|
+
save_data.save_categories_separate_keywords()
|
|
72
|
+
|
|
73
|
+
save_data.save_publishers()
|
|
74
|
+
save_data.save_publishers_separate_abbrs()
|
|
75
|
+
|
|
76
|
+
save_data.save_statistics(keywords_category_name, keywords_list)
|
|
77
|
+
save_data.save_statistics_separate_abbrs()
|
|
78
|
+
|
|
79
|
+
return None
|
|
@@ -0,0 +1,456 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import re
|
|
5
|
+
from datetime import datetime
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def conference_journal_header():
|
|
10
|
+
o = "|Publishers|Full/Homepage|Abbr/About|"
|
|
11
|
+
t = "|- |- |- |"
|
|
12
|
+
conference_header = [
|
|
13
|
+
f"{o}Acronym/Archive|Period/DBLP|Top|CCF|Submission|Days Left|Main Conf.|Days Left|Location|Keywords/Google|\n",
|
|
14
|
+
f"{t}- |- |- |- |- |- | |- |- |- |\n",
|
|
15
|
+
]
|
|
16
|
+
journal_header = [
|
|
17
|
+
f"{o}Acronym/Issues|Period/DBLP|Top/Early|CCF|CAS|JCR|IF|Keywords/Google|\n",
|
|
18
|
+
f"{t}- |- |- |- |- |- |- |- |\n",
|
|
19
|
+
]
|
|
20
|
+
return conference_header, journal_header
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
class GenerateDataDict(object):
|
|
24
|
+
def __init__(
|
|
25
|
+
self,
|
|
26
|
+
conferences_or_journals: str,
|
|
27
|
+
inproceedings_or_article: str,
|
|
28
|
+
json_dict: dict,
|
|
29
|
+
for_vue: bool = True,
|
|
30
|
+
path_spidered_conferences_or_journals: Optional[str] = None,
|
|
31
|
+
) -> None:
|
|
32
|
+
self.cj = conferences_or_journals
|
|
33
|
+
self.ia = inproceedings_or_article
|
|
34
|
+
self.json_dict = json_dict
|
|
35
|
+
|
|
36
|
+
self.path_spidered_cj = path_spidered_conferences_or_journals
|
|
37
|
+
self.for_vue = for_vue
|
|
38
|
+
|
|
39
|
+
def generate(self):
|
|
40
|
+
"""
|
|
41
|
+
Generate publisher metadata and keyword-based publication information.
|
|
42
|
+
|
|
43
|
+
Returns:
|
|
44
|
+
tuple: Contains two dictionaries:
|
|
45
|
+
- publisher_meta_abbr_dict: Publisher metadata and abbreviations
|
|
46
|
+
- publication_keyword_row_info_dict: Keyword-indexed publication info
|
|
47
|
+
"""
|
|
48
|
+
publisher_meta_dict, keyword_abbr_meta_dict, publisher_abbr_meta_dict = {}, {}, {}
|
|
49
|
+
|
|
50
|
+
for publisher in self.json_dict:
|
|
51
|
+
# Extract and clean about texts
|
|
52
|
+
abouts = [p for p in self.json_dict[publisher].get("txt_abouts", []) if p.strip()]
|
|
53
|
+
|
|
54
|
+
# Extract and clean about URLs
|
|
55
|
+
urls_about = [p.strip() for p in self.json_dict[publisher].get("urls_about", []) if p.strip()]
|
|
56
|
+
|
|
57
|
+
# Get full names
|
|
58
|
+
names_full = self.json_dict[publisher].get("names_full", [])
|
|
59
|
+
|
|
60
|
+
# Get homepage URLs
|
|
61
|
+
urls_homepage = self.json_dict[publisher].get("urls_homepage", [])
|
|
62
|
+
|
|
63
|
+
# Extract and clean conference/journal URLs
|
|
64
|
+
urls_cj = [url.strip() for url in self.json_dict[publisher].get(f"urls_{self.cj}", []) if url.strip()]
|
|
65
|
+
|
|
66
|
+
# Create publisher URL with markdown formatting if homepage exists
|
|
67
|
+
publisher_url = f"[{publisher}]({urls_homepage[0]})" if urls_homepage else publisher
|
|
68
|
+
|
|
69
|
+
# Create full name URL with markdown formatting if available
|
|
70
|
+
if names_full:
|
|
71
|
+
full_url = f"[{names_full[0]}]({urls_homepage[0]})" if urls_homepage else names_full[0]
|
|
72
|
+
else:
|
|
73
|
+
full_url = publisher
|
|
74
|
+
|
|
75
|
+
# Extract and clean remarks
|
|
76
|
+
remarks = [p for p in self.json_dict[publisher].get("txt_remarks", []) if p.strip()]
|
|
77
|
+
|
|
78
|
+
# Update publisher metadata
|
|
79
|
+
publisher_meta_dict.setdefault(publisher, {}).update({
|
|
80
|
+
"full_name_url": full_url,
|
|
81
|
+
"txt_abouts": abouts,
|
|
82
|
+
"txt_remarks": remarks,
|
|
83
|
+
"urls_about": urls_about,
|
|
84
|
+
"url_conferences_or_journals": f"[{self.cj.title()}]({urls_cj[0]})" if urls_cj else "",
|
|
85
|
+
})
|
|
86
|
+
|
|
87
|
+
# Process each abbreviation (conference/journal)
|
|
88
|
+
for abbr in self.json_dict[publisher][self.cj]:
|
|
89
|
+
abbr_dict = self.json_dict[publisher][self.cj][abbr]
|
|
90
|
+
|
|
91
|
+
# Get conference/journal info and keywords
|
|
92
|
+
temp_dict, keywords = self.conference_or_journal(publisher_url, abbr, abbr_dict)
|
|
93
|
+
|
|
94
|
+
# Generate mermaid diagram data
|
|
95
|
+
mermaid = self.generate_mermaid_data(publisher, abbr, self.ia)
|
|
96
|
+
|
|
97
|
+
publisher_abbr_meta_dict.setdefault(publisher, {}).setdefault(abbr, {}).update(temp_dict)
|
|
98
|
+
publisher_abbr_meta_dict.setdefault(publisher, {}).setdefault(abbr, {}).update({"statistics": mermaid})
|
|
99
|
+
|
|
100
|
+
# Index by keywords for quick lookup
|
|
101
|
+
for keyword in keywords:
|
|
102
|
+
keyword_abbr_meta_dict.setdefault(keyword, {}).setdefault(abbr, {}).update(temp_dict)
|
|
103
|
+
keyword_abbr_meta_dict.setdefault(keyword, {}).setdefault(abbr, {}).update({"statistics": mermaid})
|
|
104
|
+
|
|
105
|
+
return publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict
|
|
106
|
+
|
|
107
|
+
def conference_or_journal(self, publisher_url: str, abbr: str, abbr_dict: dict):
|
|
108
|
+
"""
|
|
109
|
+
Process conference or journal data and generate formatted information.
|
|
110
|
+
|
|
111
|
+
Args:
|
|
112
|
+
publisher_url: Publisher's URL
|
|
113
|
+
abbr: Abbreviation identifier
|
|
114
|
+
abbr_dict: Dictionary containing publication details
|
|
115
|
+
|
|
116
|
+
Returns:
|
|
117
|
+
dict: Contains formatted about text, remarks, and table row
|
|
118
|
+
list: Sorted list of keywords
|
|
119
|
+
"""
|
|
120
|
+
# Validate full and abbreviated names match in length
|
|
121
|
+
self._validate_name_lengths(abbr_dict)
|
|
122
|
+
|
|
123
|
+
# Extract basic information
|
|
124
|
+
full_name, abbr_name = self._extract_full_abbr_names(abbr_dict)
|
|
125
|
+
url_home = self._extract_homepage_url(abbr_dict)
|
|
126
|
+
period = self._format_period_with_dblp(abbr_dict)
|
|
127
|
+
|
|
128
|
+
# Extract text content
|
|
129
|
+
abouts = self._extract_text_content(abbr_dict, "txt_abouts")
|
|
130
|
+
remarks = self._extract_text_content(abbr_dict, "txt_remarks")
|
|
131
|
+
url_about = self._extract_first_url(abbr_dict, "urls_about")
|
|
132
|
+
|
|
133
|
+
# Process keywords with Google search links
|
|
134
|
+
keywords, keywords_url = self._process_keywords(abbr_dict)
|
|
135
|
+
|
|
136
|
+
# Format top score with early access link if available
|
|
137
|
+
top = self._format_top_score(abbr_dict)
|
|
138
|
+
|
|
139
|
+
# Generate appropriate table row based on type
|
|
140
|
+
row_inf = self._generate_table_row(
|
|
141
|
+
publisher_url, full_name, abbr_name, url_home,
|
|
142
|
+
url_about, period, top, keywords_url, abbr, abbr_dict
|
|
143
|
+
)
|
|
144
|
+
|
|
145
|
+
return {"txt_abouts": abouts, "txt_remarks": remarks, "row_inf": row_inf}, keywords
|
|
146
|
+
|
|
147
|
+
def _validate_name_lengths(self, abbr_dict: dict):
|
|
148
|
+
"""Validate that full and abbreviated names arrays have equal length."""
|
|
149
|
+
full_names = abbr_dict.get("names_full", [])
|
|
150
|
+
abbr_names = abbr_dict.get("names_abbr", [])
|
|
151
|
+
if len(full_names) != len(abbr_names):
|
|
152
|
+
raise ValueError(f"Length mismatch: {len(full_names)} full names vs {len(abbr_names)} abbreviated names")
|
|
153
|
+
|
|
154
|
+
def _extract_full_abbr_names(self, abbr_dict: dict):
|
|
155
|
+
"""Extract full and abbreviated names from dictionary."""
|
|
156
|
+
# For journals: use first full name from list; for conferences: use single name
|
|
157
|
+
full_name = (abbr_dict.get("names_full", [""])[0] if self.cj == "journals" else abbr_dict.get("name", ""))
|
|
158
|
+
abbr_name = abbr_dict.get("names_abbr", [""])[0]
|
|
159
|
+
return full_name, abbr_name
|
|
160
|
+
|
|
161
|
+
def _extract_homepage_url(self, abbr_dict: dict):
|
|
162
|
+
"""Extract and clean homepage URL."""
|
|
163
|
+
urls = [u.strip() for u in abbr_dict.get("urls_homepage", []) if u.strip()]
|
|
164
|
+
return urls[0] if urls else ""
|
|
165
|
+
|
|
166
|
+
def _format_period_with_dblp(self, abbr_dict: dict):
|
|
167
|
+
"""Format publication period with DBLP link if available."""
|
|
168
|
+
start_year = abbr_dict.get("year_start", "")
|
|
169
|
+
end_year = abbr_dict.get("year_end", "")
|
|
170
|
+
|
|
171
|
+
# Create period string (e.g., "2020 - 2023" or "2020 -")
|
|
172
|
+
period = f"{start_year} - {end_year}" if start_year and end_year else f"{start_year} -" if start_year else ""
|
|
173
|
+
|
|
174
|
+
# Add DBLP link if acronym available
|
|
175
|
+
if acronym_dblp := abbr_dict.get("acronym_dblp", ""):
|
|
176
|
+
journal_conf = "journals" if self.cj == "journals" else "conf"
|
|
177
|
+
dblp_url = f"https://dblp.org/db/{journal_conf}/{acronym_dblp}/index.html"
|
|
178
|
+
period = f'[{period}]({dblp_url})'
|
|
179
|
+
|
|
180
|
+
return period
|
|
181
|
+
|
|
182
|
+
def _extract_text_content(self, abbr_dict: dict, key: str):
|
|
183
|
+
"""Extract and clean text content from dictionary."""
|
|
184
|
+
return [text for text in abbr_dict.get(key, []) if text.strip()]
|
|
185
|
+
|
|
186
|
+
def _extract_first_url(self, abbr_dict: dict, key: str):
|
|
187
|
+
"""Extract first URL from a list in dictionary."""
|
|
188
|
+
urls = [url.strip() for url in abbr_dict.get(key, []) if url.strip()]
|
|
189
|
+
return urls[0].split(",")[0] if urls else ""
|
|
190
|
+
|
|
191
|
+
def _process_keywords(self, abbr_dict: dict):
|
|
192
|
+
"""Process keywords and convert to Google search URLs."""
|
|
193
|
+
keywords_dict = abbr_dict.get("keywords_dict", {})
|
|
194
|
+
|
|
195
|
+
# Clean and sort keywords
|
|
196
|
+
cleaned_keywords = {}
|
|
197
|
+
for category, words in keywords_dict.items():
|
|
198
|
+
if category.strip():
|
|
199
|
+
sorted_words = sorted(set([word.strip() for word in words if word.strip()]))
|
|
200
|
+
cleaned_keywords[category.strip()] = sorted_words
|
|
201
|
+
|
|
202
|
+
# Flatten keywords and remove duplicates
|
|
203
|
+
all_keywords = []
|
|
204
|
+
for category, words in cleaned_keywords.items():
|
|
205
|
+
if words:
|
|
206
|
+
all_keywords.extend(words)
|
|
207
|
+
else:
|
|
208
|
+
all_keywords.append(category)
|
|
209
|
+
all_keywords = sorted(set(all_keywords))
|
|
210
|
+
# Create Google search links for each keyword
|
|
211
|
+
google_base = "https://www.google.com/search?q="
|
|
212
|
+
keywords_url = [
|
|
213
|
+
f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords
|
|
214
|
+
]
|
|
215
|
+
|
|
216
|
+
# For category
|
|
217
|
+
# Flatten keywords and remove duplicates
|
|
218
|
+
all_keywords = []
|
|
219
|
+
for category, words in cleaned_keywords.items():
|
|
220
|
+
all_keywords.extend(words)
|
|
221
|
+
all_keywords.append(category)
|
|
222
|
+
all_keywords = sorted(set(all_keywords))
|
|
223
|
+
|
|
224
|
+
return all_keywords, keywords_url
|
|
225
|
+
|
|
226
|
+
def _format_top_score(self, abbr_dict: dict):
|
|
227
|
+
"""Format top score with optional early access link."""
|
|
228
|
+
is_top = "True" if abbr_dict.get("score_top", False) else "False"
|
|
229
|
+
url_early_access = abbr_dict.get("url_early_access", "")
|
|
230
|
+
return self._format_link(is_top, url_early_access)
|
|
231
|
+
|
|
232
|
+
def _format_link(self, text, url):
|
|
233
|
+
"""Format text as markdown link if URL provided."""
|
|
234
|
+
return f"[{text}]({url})" if url else text
|
|
235
|
+
|
|
236
|
+
def _generate_table_row(
|
|
237
|
+
self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
|
|
238
|
+
):
|
|
239
|
+
"""Generate appropriate table row based on publication type."""
|
|
240
|
+
if self.cj == "conferences":
|
|
241
|
+
return self._generate_for_conference(
|
|
242
|
+
publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
|
|
243
|
+
)
|
|
244
|
+
else:
|
|
245
|
+
return self._generate_for_journal(
|
|
246
|
+
publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
# Conferences
|
|
250
|
+
def _generate_for_conference(
|
|
251
|
+
self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
|
|
252
|
+
):
|
|
253
|
+
"""
|
|
254
|
+
Generate a markdown table row for conference information.
|
|
255
|
+
|
|
256
|
+
Args:
|
|
257
|
+
publisher_url: URL of the publisher
|
|
258
|
+
full_name: Full name of the conference
|
|
259
|
+
abbr_name: Abbreviated name of the conference
|
|
260
|
+
url_home: Homepage URL
|
|
261
|
+
url_about: About page URL
|
|
262
|
+
period: Conference period/frequency
|
|
263
|
+
top: Conference ranking/tier
|
|
264
|
+
keywords: List of keywords
|
|
265
|
+
abbr: Conference abbreviation
|
|
266
|
+
abbr_dict: Dictionary containing conference details
|
|
267
|
+
|
|
268
|
+
Returns:
|
|
269
|
+
str: Formatted markdown table row
|
|
270
|
+
"""
|
|
271
|
+
# Get archive URL and format link
|
|
272
|
+
archive_url = self._extract_first_url(abbr_dict, "urls_archive")
|
|
273
|
+
archive_display = self._format_link(abbr, archive_url)
|
|
274
|
+
|
|
275
|
+
# Process conference dates
|
|
276
|
+
abstract_due, start_date, today = self._process_conference_dates(abbr_dict)
|
|
277
|
+
|
|
278
|
+
# Format date indicators for Vue or standard display
|
|
279
|
+
abstract_indicator, start_indicator = self._format_date_indicators(
|
|
280
|
+
abstract_due, start_date, today
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
# Get year URL for start date link
|
|
284
|
+
year_url = abbr_dict.get("conf_url", "")
|
|
285
|
+
|
|
286
|
+
# Build and return markdown table row
|
|
287
|
+
return self._build_conference_row(
|
|
288
|
+
publisher_url, full_name, abbr_name, url_home, url_about,
|
|
289
|
+
archive_display, period, top, abbr_dict, abstract_due,
|
|
290
|
+
abstract_indicator, start_date, start_indicator, year_url,
|
|
291
|
+
keywords
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
def _process_conference_dates(self, abbr_dict):
|
|
295
|
+
"""Parse and return conference dates."""
|
|
296
|
+
# Parse abstract due date
|
|
297
|
+
abstract_due = None
|
|
298
|
+
if due_str := abbr_dict.get("conf_abstract_due", "").strip():
|
|
299
|
+
abstract_due = datetime.strptime(due_str, "%d/%m/%Y").date()
|
|
300
|
+
|
|
301
|
+
# Parse conference start date
|
|
302
|
+
start_date = None
|
|
303
|
+
if start_str := abbr_dict.get("conf_date_start", "").strip():
|
|
304
|
+
start_date = datetime.strptime(start_str, "%d/%m/%Y").date()
|
|
305
|
+
|
|
306
|
+
# Get today's date
|
|
307
|
+
today = datetime.strptime(datetime.now().strftime("%d/%m/%Y"), "%d/%m/%Y").date()
|
|
308
|
+
|
|
309
|
+
return abstract_due, start_date, today
|
|
310
|
+
|
|
311
|
+
def _format_date_indicators(self, abstract_due, start_date, today):
|
|
312
|
+
"""Format date indicators for display."""
|
|
313
|
+
abstract_indicator, start_indicator = "", ""
|
|
314
|
+
|
|
315
|
+
if self.for_vue:
|
|
316
|
+
# Vue.js template format
|
|
317
|
+
if abstract_due:
|
|
318
|
+
abstract_indicator = f"**{{{{ diffDate('{abstract_due}') }}}}**"
|
|
319
|
+
if start_date:
|
|
320
|
+
start_indicator = f"**{{{{ diffDate('{start_date}') }}}}**"
|
|
321
|
+
else:
|
|
322
|
+
# Standard day count format
|
|
323
|
+
if abstract_due:
|
|
324
|
+
abstract_indicator = (abstract_due - today).days if today <= abstract_due else "Expired"
|
|
325
|
+
if start_date:
|
|
326
|
+
start_indicator = (start_date - today).days if today <= start_date else "Expired"
|
|
327
|
+
|
|
328
|
+
return abstract_indicator, start_indicator
|
|
329
|
+
|
|
330
|
+
def _build_conference_row(
|
|
331
|
+
self, publisher_url, full_name, abbr_name, url_home, url_about, archive_display, period, top, abbr_dict,
|
|
332
|
+
abstract_due, abstract_indicator, start_date, start_indicator, year_url, keywords
|
|
333
|
+
):
|
|
334
|
+
"""Construct conference table row string."""
|
|
335
|
+
# Format date strings
|
|
336
|
+
abstract_date_str = abstract_due.strftime("%d/%m/%Y") if abstract_due else ""
|
|
337
|
+
start_date_str = start_date.strftime("%d/%m/%Y") if start_date else ""
|
|
338
|
+
|
|
339
|
+
# Format start date with link if available
|
|
340
|
+
start_date_display = self._format_link(start_date_str, year_url) if start_date_str else ""
|
|
341
|
+
|
|
342
|
+
# Build table row
|
|
343
|
+
return f"|{publisher_url}|" \
|
|
344
|
+
f"{self._format_link(full_name, url_home)}|" \
|
|
345
|
+
f"{self._format_link(abbr_name, url_about)}|" \
|
|
346
|
+
f"{archive_display}|" \
|
|
347
|
+
f"{period}|" \
|
|
348
|
+
f"{top}|" \
|
|
349
|
+
f"{abbr_dict.get('score_ccf', '')}|" \
|
|
350
|
+
f"{abstract_date_str}|" \
|
|
351
|
+
f"{abstract_indicator}|" \
|
|
352
|
+
f"{start_date_display}|" \
|
|
353
|
+
f"{start_indicator}|" \
|
|
354
|
+
f"{abbr_dict.get('conf_location', '').strip()}|" \
|
|
355
|
+
f"{'; '.join(keywords)}|"
|
|
356
|
+
|
|
357
|
+
# Journals
|
|
358
|
+
def _generate_for_journal(
|
|
359
|
+
self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
|
|
360
|
+
):
|
|
361
|
+
"""
|
|
362
|
+
Generate a markdown table row for journal information.
|
|
363
|
+
|
|
364
|
+
Args:
|
|
365
|
+
publisher_url: URL of the publisher
|
|
366
|
+
full_name: Full name of the journal
|
|
367
|
+
abbr_name: Abbreviated name of the journal
|
|
368
|
+
url_home: Homepage URL
|
|
369
|
+
url_about: About page URL
|
|
370
|
+
period: Publication period/frequency
|
|
371
|
+
top: Journal ranking/tier
|
|
372
|
+
keywords: List of keywords
|
|
373
|
+
abbr: Journal abbreviation
|
|
374
|
+
abbr_dict: Dictionary containing journal details
|
|
375
|
+
|
|
376
|
+
Returns:
|
|
377
|
+
str: Formatted markdown table row
|
|
378
|
+
"""
|
|
379
|
+
# Get issues URL and format link
|
|
380
|
+
issues_url = self._extract_first_url(abbr_dict, "urls_issues")
|
|
381
|
+
issues_display = self._format_link(abbr, issues_url)
|
|
382
|
+
|
|
383
|
+
# Build and return markdown table row
|
|
384
|
+
return self._build_journal_row(
|
|
385
|
+
publisher_url, full_name, abbr_name, url_home, url_about,
|
|
386
|
+
issues_display, period, top, abbr_dict, keywords
|
|
387
|
+
)
|
|
388
|
+
|
|
389
|
+
def _build_journal_row(
|
|
390
|
+
self, publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
|
|
391
|
+
):
|
|
392
|
+
"""Construct journal table row string."""
|
|
393
|
+
return f"|{publisher_url}|" \
|
|
394
|
+
f"{self._format_link(full_name, url_home)}|" \
|
|
395
|
+
f"{self._format_link(abbr_name, url_about)}|" \
|
|
396
|
+
f"{issues_display}|" \
|
|
397
|
+
f"{period}|" \
|
|
398
|
+
f"{top}|" \
|
|
399
|
+
f"{abbr_dict.get('score_ccf', '')}|" \
|
|
400
|
+
f"{abbr_dict.get('score_cas', '')}|" \
|
|
401
|
+
f"{abbr_dict.get('score_jcr', '')}|" \
|
|
402
|
+
f"{abbr_dict.get('score_if', '')}|" \
|
|
403
|
+
f"{'; '.join(keywords)}|"
|
|
404
|
+
|
|
405
|
+
# Mermaid data
|
|
406
|
+
def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
|
|
407
|
+
path_spidered_cj = self.path_spidered_cj if self.path_spidered_cj else ""
|
|
408
|
+
path_readme = os.path.join(path_spidered_cj, publisher, abbr, inproceedings_or_article)
|
|
409
|
+
full_readme = os.path.expanduser(os.path.join(path_readme, "README.md"))
|
|
410
|
+
if not os.path.exists(full_readme):
|
|
411
|
+
return []
|
|
412
|
+
|
|
413
|
+
mermaid, data_dict = [], {}
|
|
414
|
+
# |AAAI|1980|95|Proceedings of the First National Conference on Artificial Intelligence|
|
|
415
|
+
regex = re.compile(r"\|.*\|([0-9]+)\|([0-9]+)\|.*\|")
|
|
416
|
+
with open(full_readme, "r") as file:
|
|
417
|
+
data_list = file.readlines()
|
|
418
|
+
for line in data_list:
|
|
419
|
+
if mch := regex.search(line):
|
|
420
|
+
data_dict.setdefault(mch.group(1), []).append(mch.group(2))
|
|
421
|
+
data_dict = {year: sum([int(n) for n in data_dict[year]]) for year in data_dict}
|
|
422
|
+
|
|
423
|
+
# Mermaid
|
|
424
|
+
if len(data_dict) != 0:
|
|
425
|
+
mermaid = ["```mermaid\n"]
|
|
426
|
+
mermaid.extend(
|
|
427
|
+
[
|
|
428
|
+
'---\n',
|
|
429
|
+
'config:\n',
|
|
430
|
+
' xyChart:\n',
|
|
431
|
+
' width: 1200\n',
|
|
432
|
+
' height: 600\n',
|
|
433
|
+
' themeVariables:\n',
|
|
434
|
+
' xyChart:\n',
|
|
435
|
+
' titleColor: "#ff0000"\n',
|
|
436
|
+
'---\n'
|
|
437
|
+
]
|
|
438
|
+
)
|
|
439
|
+
mermaid.extend(["xychart-beta\n", f' title "{abbr}"\n'])
|
|
440
|
+
|
|
441
|
+
x_axis, bar, line = [], [], []
|
|
442
|
+
for year in data_dict:
|
|
443
|
+
x_axis.append(int(year))
|
|
444
|
+
bar.append(data_dict[year])
|
|
445
|
+
line.append(data_dict[year])
|
|
446
|
+
|
|
447
|
+
idx = next((i for i, year in enumerate(x_axis) if year >= 2000), len(x_axis))
|
|
448
|
+
x_axis, bar, line = x_axis[idx:], bar[idx:], line[idx:]
|
|
449
|
+
|
|
450
|
+
mermaid.append(f" x-axis {x_axis}\n")
|
|
451
|
+
mermaid.append(' y-axis "Number of Papers"\n')
|
|
452
|
+
mermaid.append(f" bar {bar}\n")
|
|
453
|
+
mermaid.append(f" line {line}\n")
|
|
454
|
+
mermaid.append("```\n")
|
|
455
|
+
|
|
456
|
+
return mermaid
|
|
@@ -0,0 +1,284 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from typing import List
|
|
5
|
+
|
|
6
|
+
from ..core._base import standardize_path
|
|
7
|
+
from .generate_dict import conference_journal_header
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def conference_journal_informations():
|
|
11
|
+
conference_inf = [
|
|
12
|
+
"!> [List of Upcoming International Conferences](https://internationalconferencealerts.com/all-events.php)\n\n",
|
|
13
|
+
"!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n"
|
|
14
|
+
]
|
|
15
|
+
journal_inf = []
|
|
16
|
+
return conference_inf, journal_inf
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class WriteDataToMd(object):
|
|
20
|
+
"""Class to write publication data to Markdown files."""
|
|
21
|
+
|
|
22
|
+
def __init__(
|
|
23
|
+
self,
|
|
24
|
+
conferences_or_journals: str,
|
|
25
|
+
inproceedings_or_article: str,
|
|
26
|
+
publisher_meta_dict: dict,
|
|
27
|
+
publisher_abbr_meta_dict: dict,
|
|
28
|
+
keyword_abbr_meta_dict: dict,
|
|
29
|
+
path_output: str,
|
|
30
|
+
) -> None:
|
|
31
|
+
"""Initialize with publication data and output path."""
|
|
32
|
+
self.cj = conferences_or_journals # "conferences" or "journals"
|
|
33
|
+
self.ia = inproceedings_or_article # "inproceedings" or "article"
|
|
34
|
+
self.publisher_meta_dict = publisher_meta_dict
|
|
35
|
+
self.publisher_abbr_meta_dict = publisher_abbr_meta_dict
|
|
36
|
+
self.keyword_abbr_meta_dict = keyword_abbr_meta_dict
|
|
37
|
+
self.path_output = standardize_path(path_output)
|
|
38
|
+
|
|
39
|
+
self._default_inf = [
|
|
40
|
+
"- The data for TOP, CCF, CAS, JCR, and IF are sourced from [easyScholar](https://www.easyscholar.cc/).\n\n"
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
def save_introductions(self) -> None:
|
|
44
|
+
"""Save introduction file with all conferences/journals list."""
|
|
45
|
+
conference_header, journal_header = conference_journal_header()
|
|
46
|
+
conference_inf, journal_inf = conference_journal_informations()
|
|
47
|
+
|
|
48
|
+
data_list = [f"# {self.cj.title()}\n\n"]
|
|
49
|
+
data_list.extend(self._default_inf)
|
|
50
|
+
|
|
51
|
+
# Add appropriate headers based on type
|
|
52
|
+
if self.cj.lower() == "conferences":
|
|
53
|
+
data_list.extend(conference_inf)
|
|
54
|
+
data_list.append("| " + conference_header[0])
|
|
55
|
+
data_list.append("|- " + conference_header[1])
|
|
56
|
+
else:
|
|
57
|
+
data_list.extend(journal_inf)
|
|
58
|
+
data_list.append("| " + journal_header[0])
|
|
59
|
+
data_list.append("|- " + journal_header[1])
|
|
60
|
+
|
|
61
|
+
# Add all publications to table
|
|
62
|
+
idx = 1
|
|
63
|
+
for publisher in self.publisher_abbr_meta_dict:
|
|
64
|
+
for abbr in self.publisher_abbr_meta_dict[publisher]:
|
|
65
|
+
row_info = self.publisher_abbr_meta_dict[publisher][abbr]['row_inf']
|
|
66
|
+
data_list.append(f"|{idx}{row_info}\n")
|
|
67
|
+
idx += 1
|
|
68
|
+
|
|
69
|
+
# Write to file
|
|
70
|
+
output_file = os.path.join(self.path_output, f"Introductions_{self.cj.title()}.md")
|
|
71
|
+
with open(output_file, "w") as f:
|
|
72
|
+
f.writelines(data_list)
|
|
73
|
+
|
|
74
|
+
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
75
|
+
def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: List[str]):
|
|
76
|
+
keywords = list(self.keyword_abbr_meta_dict.keys())
|
|
77
|
+
|
|
78
|
+
# Get and sort publication types
|
|
79
|
+
if keywords_category_name and keywords_list:
|
|
80
|
+
_keywords = []
|
|
81
|
+
for keyword in keywords_list:
|
|
82
|
+
if keyword in keywords:
|
|
83
|
+
_keywords.append(keyword)
|
|
84
|
+
return _keywords
|
|
85
|
+
else:
|
|
86
|
+
# default
|
|
87
|
+
return sorted(keywords)
|
|
88
|
+
|
|
89
|
+
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
90
|
+
def save_categories(self, keywords_category_name: str, keywords_list: List[str]) -> None:
|
|
91
|
+
"""Save publications categorized by keywords."""
|
|
92
|
+
conference_header, journal_header = conference_journal_header()
|
|
93
|
+
data_list = [f"# {self.cj.title()}\n\n"]
|
|
94
|
+
data_list.extend(self._default_inf)
|
|
95
|
+
|
|
96
|
+
# Add publications for each category
|
|
97
|
+
for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
|
|
98
|
+
data_list.append(f"## {keyword}\n\n")
|
|
99
|
+
|
|
100
|
+
# Add appropriate header
|
|
101
|
+
if self.cj == "conferences":
|
|
102
|
+
data_list.extend(conference_header)
|
|
103
|
+
else:
|
|
104
|
+
data_list.extend(journal_header)
|
|
105
|
+
|
|
106
|
+
# Add all publications in this category
|
|
107
|
+
for abbr in self.keyword_abbr_meta_dict[keyword]:
|
|
108
|
+
data_list.append(self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"] + "\n")
|
|
109
|
+
data_list.append("\n")
|
|
110
|
+
|
|
111
|
+
# Write to file
|
|
112
|
+
category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
|
|
113
|
+
with open(os.path.join(self.path_output, f"Categories_{self.cj.title()}{category_postfix}.md"), "w") as f:
|
|
114
|
+
f.writelines(data_list)
|
|
115
|
+
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
def save_categories_separate_keywords(self) -> None:
|
|
119
|
+
conference_header, journal_header = conference_journal_header()
|
|
120
|
+
|
|
121
|
+
# Add publications for each category
|
|
122
|
+
for keyword in self.keyword_abbr_meta_dict:
|
|
123
|
+
data_list = [f"# {keyword}\n\n"]
|
|
124
|
+
data_list.extend(self._default_inf)
|
|
125
|
+
|
|
126
|
+
# Add appropriate header
|
|
127
|
+
if self.cj == "conferences":
|
|
128
|
+
data_list.extend(conference_header)
|
|
129
|
+
else:
|
|
130
|
+
data_list.extend(journal_header)
|
|
131
|
+
|
|
132
|
+
# Add all publications in this category
|
|
133
|
+
for abbr in self.keyword_abbr_meta_dict[keyword]:
|
|
134
|
+
data_list.append(self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"] + "\n")
|
|
135
|
+
data_list.append("\n")
|
|
136
|
+
|
|
137
|
+
# Write keyword-specific file
|
|
138
|
+
path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
|
|
139
|
+
with open(os.path.join(path_key, f"{keyword.replace(' ', '_')}.md"), "w") as f:
|
|
140
|
+
f.writelines(data_list)
|
|
141
|
+
|
|
142
|
+
return None
|
|
143
|
+
|
|
144
|
+
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
145
|
+
def save_publishers(self) -> None:
|
|
146
|
+
"""Save publisher overview file with basic information."""
|
|
147
|
+
data_list_pub = [
|
|
148
|
+
f"# Introductions of Publishers and {self.cj.title()}\n\n",
|
|
149
|
+
"| |Publishers|About US|Conferences/Journals|Separate Links|\n",
|
|
150
|
+
"|-|- |- |- |- |\n"
|
|
151
|
+
]
|
|
152
|
+
idx = 1
|
|
153
|
+
|
|
154
|
+
# Add each publisher to table
|
|
155
|
+
for pub in self.publisher_meta_dict:
|
|
156
|
+
meta = self.publisher_meta_dict[pub]
|
|
157
|
+
|
|
158
|
+
full_name_url, about_url, cj_url, local_url = "", "", "", ""
|
|
159
|
+
if x := meta.get("full_name_url", ""):
|
|
160
|
+
full_name_url = x
|
|
161
|
+
if x := meta.get("url_conferences_or_journals", ""):
|
|
162
|
+
cj_url = x
|
|
163
|
+
if pub_intr_urls := meta.get("urls_about", []):
|
|
164
|
+
about_url = f"[About US]({pub_intr_urls[0]})"
|
|
165
|
+
|
|
166
|
+
local_url = f"[{pub}](data/{self.cj.title()}/Publishers_{self.cj.title()}/{pub}.md)"
|
|
167
|
+
|
|
168
|
+
# Create table row
|
|
169
|
+
row = f"| {idx} | {full_name_url} | {about_url} | {cj_url} | {local_url} |\n"
|
|
170
|
+
data_list_pub.append(row)
|
|
171
|
+
idx += 1
|
|
172
|
+
|
|
173
|
+
# Write to file
|
|
174
|
+
with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w") as f:
|
|
175
|
+
f.writelines(data_list_pub)
|
|
176
|
+
return None
|
|
177
|
+
|
|
178
|
+
def save_publishers_separate_abbrs(self) -> None:
|
|
179
|
+
conference_header, journal_header = conference_journal_header()
|
|
180
|
+
for pub in self.publisher_meta_dict:
|
|
181
|
+
data_list = [f"# {pub}\n\n"]
|
|
182
|
+
data_list.extend(self._default_inf)
|
|
183
|
+
meta = self.publisher_meta_dict[pub]
|
|
184
|
+
|
|
185
|
+
# Add about and remarks sections
|
|
186
|
+
for flag in ["txt_remarks", "txt_abouts"]:
|
|
187
|
+
if temps := meta.get(flag, []):
|
|
188
|
+
temps[-1] = f"{temps[-1].rstrip()}\n\n"
|
|
189
|
+
if temps:
|
|
190
|
+
data_list.append(f"## {flag.title()}\n\n")
|
|
191
|
+
data_list.extend(temps)
|
|
192
|
+
|
|
193
|
+
# Add each conference/journal abbreviation
|
|
194
|
+
if pub not in self.publisher_abbr_meta_dict:
|
|
195
|
+
continue
|
|
196
|
+
|
|
197
|
+
for abbr in self.publisher_abbr_meta_dict[pub]:
|
|
198
|
+
data_list.append(f"## {abbr}\n\n")
|
|
199
|
+
|
|
200
|
+
# Add appropriate header
|
|
201
|
+
if self.cj == "conferences":
|
|
202
|
+
data_list.extend(conference_header)
|
|
203
|
+
else:
|
|
204
|
+
data_list.extend(journal_header)
|
|
205
|
+
|
|
206
|
+
# Add row information
|
|
207
|
+
row_info = self.publisher_abbr_meta_dict[pub][abbr]["row_inf"]
|
|
208
|
+
data_list.append(f'{row_info}\n\n')
|
|
209
|
+
|
|
210
|
+
# Add remarks and about for this abbreviation
|
|
211
|
+
for flag in ["txt_remarks", "txt_abouts"]:
|
|
212
|
+
if temps := self.publisher_abbr_meta_dict[pub][abbr].get(flag, []):
|
|
213
|
+
temps[-1] = f"{temps[-1].rstrip()}\n\n"
|
|
214
|
+
if temps:
|
|
215
|
+
data_list.append(f"### {flag.split('_')[-1].title()}\n\n")
|
|
216
|
+
data_list.extend(temps)
|
|
217
|
+
|
|
218
|
+
# Add statistics if available
|
|
219
|
+
if statistics := self.publisher_abbr_meta_dict[pub][abbr].get("statistics", []):
|
|
220
|
+
data_list.extend(statistics)
|
|
221
|
+
data_list.append("\n")
|
|
222
|
+
|
|
223
|
+
# Write publisher-specific file
|
|
224
|
+
path_pub = standardize_path(os.path.join(self.path_output, f"Publishers_{self.cj.title()}"))
|
|
225
|
+
with open(os.path.join(path_pub, f"{pub}.md"), "w") as f:
|
|
226
|
+
f.writelines(data_list)
|
|
227
|
+
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
231
|
+
def save_statistics(self, keywords_category_name: str, keywords_list: List[str]) -> None:
|
|
232
|
+
data_list = [
|
|
233
|
+
f"# Statistics of keywords in {self.cj.title()}\n\n",
|
|
234
|
+
"| |keywords|Separate Links|\n",
|
|
235
|
+
"|-|- |- |\n"
|
|
236
|
+
]
|
|
237
|
+
idx = 1
|
|
238
|
+
|
|
239
|
+
# Add publications for each category
|
|
240
|
+
for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
|
|
241
|
+
local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{keyword.replace(' ', '_')}.md)"
|
|
242
|
+
|
|
243
|
+
# Create table row
|
|
244
|
+
row = f"| {idx} | {keyword} | {local_url} |\n"
|
|
245
|
+
data_list.append(row)
|
|
246
|
+
idx += 1
|
|
247
|
+
|
|
248
|
+
# Write to file
|
|
249
|
+
category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
|
|
250
|
+
with open(os.path.join(self.path_output, f"Statistics_{self.cj.title()}{category_postfix}.md"), "w") as f:
|
|
251
|
+
f.writelines(data_list)
|
|
252
|
+
return None
|
|
253
|
+
|
|
254
|
+
def save_statistics_separate_abbrs(self) -> None:
|
|
255
|
+
conference_header, journal_header = conference_journal_header()
|
|
256
|
+
|
|
257
|
+
# Add publications for each category
|
|
258
|
+
for keyword in self.keyword_abbr_meta_dict:
|
|
259
|
+
data_list = [f"# {keyword}\n\n"]
|
|
260
|
+
|
|
261
|
+
for abbr in self.keyword_abbr_meta_dict[keyword]:
|
|
262
|
+
data_list.append(f"## {abbr}\n\n")
|
|
263
|
+
|
|
264
|
+
# Add appropriate header
|
|
265
|
+
if self.cj == "conferences":
|
|
266
|
+
data_list.extend(conference_header)
|
|
267
|
+
else:
|
|
268
|
+
data_list.extend(journal_header)
|
|
269
|
+
|
|
270
|
+
# Add row information
|
|
271
|
+
row_info = self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"]
|
|
272
|
+
data_list.append(f'{row_info}\n\n')
|
|
273
|
+
|
|
274
|
+
# Add statistics if available
|
|
275
|
+
if statistics := self.keyword_abbr_meta_dict[keyword][abbr].get("statistics", []):
|
|
276
|
+
data_list.extend(statistics)
|
|
277
|
+
data_list.append("\n")
|
|
278
|
+
|
|
279
|
+
# Write publisher-specific file
|
|
280
|
+
path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
|
|
281
|
+
with open(os.path.join(path_pub, f"{keyword.replace(' ', '_')}.md"), "w") as f:
|
|
282
|
+
f.writelines(data_list)
|
|
283
|
+
|
|
284
|
+
return None
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: pyformatjson
|
|
3
|
+
Version: 0.0.1
|
|
4
|
+
Summary: pyformatjson
|
|
5
|
+
License: GPL-3.0-or-later
|
|
6
|
+
Keywords: Python,json
|
|
7
|
+
Author: NextAI
|
|
8
|
+
Author-email: nextartifintell@gmail.com
|
|
9
|
+
Maintainer: NextAI
|
|
10
|
+
Maintainer-email: nextartifintell@gmail.com
|
|
11
|
+
Requires-Python: >=3.13
|
|
12
|
+
Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.14
|
|
16
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
17
|
+
Project-URL: Documentation, https://github.com/NextArtifIntell/pyformatjson
|
|
18
|
+
Project-URL: Homepage, https://github.com/NextArtifIntell/pyformatjson
|
|
19
|
+
Project-URL: Repository, https://github.com/NextArtifIntell/pyformatjson
|
|
20
|
+
Description-Content-Type: text/markdown
|
|
21
|
+
|
|
22
|
+
# PyFormatJson
|
|
23
|
+
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
pyformatjson/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
2
|
+
pyformatjson/core/_base.py,sha256=rLu503M9cgbIaByP5zTo7JZo-8HZoBfPyb1y7_Btkcs,2878
|
|
3
|
+
pyformatjson/core/update_json.py,sha256=AdvwIrlZoqE-36UxUumcUQ3DuSc1SSBJUGF2T550YZc,3119
|
|
4
|
+
pyformatjson/main.py,sha256=7nypq1inVyjI8tj_rL_BsnLGsUuWtSdhgBSPkkw5In8,3179
|
|
5
|
+
pyformatjson/tools/generate_dict.py,sha256=naQfKk9gqXZaTeC93wBCLmJ8a9wusRMMiOVqRCOIZ0o,19408
|
|
6
|
+
pyformatjson/tools/write_dict.py,sha256=HpkJYbxdxxAxS_3WMBw-QyKuZ0IXWus1-EUEcPkW8Jw,11930
|
|
7
|
+
pyformatjson-0.0.1.dist-info/METADATA,sha256=IEPUNlD8AXPHlK2l-eHX_4-8whZa2FVgn0VOGCSArRY,855
|
|
8
|
+
pyformatjson-0.0.1.dist-info/WHEEL,sha256=M5asmiAlL6HEcOq52Yi5mmk9KmTVjY2RDPtO4p9DMrc,88
|
|
9
|
+
pyformatjson-0.0.1.dist-info/RECORD,,
|