pyformatjson 0.0.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,23 @@
1
+ Metadata-Version: 2.4
2
+ Name: pyformatjson
3
+ Version: 0.0.1
4
+ Summary: pyformatjson
5
+ License: GPL-3.0-or-later
6
+ Keywords: Python,json
7
+ Author: NextAI
8
+ Author-email: nextartifintell@gmail.com
9
+ Maintainer: NextAI
10
+ Maintainer-email: nextartifintell@gmail.com
11
+ Requires-Python: >=3.13
12
+ Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Programming Language :: Python :: 3.14
16
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
17
+ Project-URL: Documentation, https://github.com/NextArtifIntell/pyformatjson
18
+ Project-URL: Homepage, https://github.com/NextArtifIntell/pyformatjson
19
+ Project-URL: Repository, https://github.com/NextArtifIntell/pyformatjson
20
+ Description-Content-Type: text/markdown
21
+
22
+ # PyFormatJson
23
+
@@ -0,0 +1 @@
1
+ # PyFormatJson
File without changes
@@ -0,0 +1,103 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ import re
5
+ from typing import List
6
+
7
+
8
+ def split_text_by_length(text, max_length=120) -> List[str]:
9
+ lines = []
10
+ while text:
11
+ if len(text) <= max_length:
12
+ lines.append(text)
13
+ break
14
+
15
+ split_pos = text.rfind(" ", 0, max_length + 1)
16
+ if split_pos == -1:
17
+ split_pos = max_length
18
+
19
+ line = text[:split_pos]
20
+ lines.append(line)
21
+
22
+ text = text[split_pos:]
23
+
24
+ new_lines = []
25
+ for line in lines:
26
+ new_lines.append(line)
27
+ return new_lines
28
+
29
+
30
+ def split_data_list(
31
+ split_pattern: str, data_list: List[str], last_next: str = "next"
32
+ ) -> List[str]:
33
+ r"""Split data list according to the split pattern.
34
+
35
+ The capturing parentheses must be used in the pattern, such as `(\n)`.
36
+
37
+ Args:
38
+ split_pattern (str): split pattern.
39
+ data_list (List[str]): data list.
40
+ last_next (str): "next" or "last".
41
+
42
+ Returns:
43
+ List[str]: new data list.
44
+
45
+ Examples:
46
+ split_pattern = r"(\n)", last_next = "next" or "last".
47
+ """
48
+ new_data_list = []
49
+ for line in data_list:
50
+ split_list = re.split(split_pattern, line)
51
+ list_one = split_list[0: len(split_list): 2]
52
+ list_two = split_list[1: len(split_list): 2]
53
+
54
+ temp = []
55
+ if last_next == "next":
56
+ list_two.insert(0, "")
57
+ temp = [list_two[i] + list_one[i] for i in range(len(list_one))]
58
+ if last_next == "last":
59
+ list_two.append("")
60
+ temp = [list_one[i] + list_two[i] for i in range(len(list_one))]
61
+ new_data_list.extend(temp)
62
+ new_data_list = [line for line in new_data_list if line.strip()]
63
+ return new_data_list
64
+
65
+
66
+ def standardize_path(path_input: str) -> str:
67
+ path_input = os.path.expandvars(os.path.expanduser(path_input))
68
+ if not os.path.exists(path_input):
69
+ os.makedirs(path_input)
70
+ return path_input
71
+
72
+
73
+ def sort_strings_with_embedded_numbers(s: str) -> List[str]:
74
+ re_digits = re.compile(r"(\d+)")
75
+ pieces = re_digits.split(s)
76
+ pieces[1::2] = map(int, pieces[1::2])
77
+ return pieces
78
+
79
+
80
+ def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
81
+ return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
82
+
83
+
84
+ class IterateSortDict(object):
85
+ def __init__(self, reverse: bool = False) -> None:
86
+ self.reverse = reverse
87
+
88
+ def dict_update(self, old):
89
+ """Update."""
90
+ old = self.dict_sort_iteration(old)
91
+ old = self.dict_sort(old)
92
+ return old
93
+
94
+ def dict_sort_iteration(self, old: dict):
95
+ """Sort."""
96
+ for key in old:
97
+ if isinstance(old[key], dict):
98
+ old[key] = self.dict_update(old[key])
99
+ return old
100
+
101
+ def dict_sort(self, old: dict):
102
+ """Sort."""
103
+ return {k: old[k] for k in sort_int_str(list(old.keys()), self.reverse)}
@@ -0,0 +1,90 @@
1
+ # coding=utf-8
2
+
3
+ import json
4
+ import os
5
+ from typing import Any, Dict
6
+
7
+ from ._base import split_data_list, split_text_by_length
8
+
9
+
10
+ def load_json_data(path_json: str, filename: str) -> Dict:
11
+ """Load JSON data from file."""
12
+ try:
13
+ file_path = os.path.join(path_json, f"{filename}.json")
14
+ if not os.path.exists(file_path):
15
+ return {}
16
+
17
+ with open(file_path, "r", encoding="utf-8") as file:
18
+ return json.load(file)
19
+
20
+ except Exception as e:
21
+ print(f"Error loading {filename}.json: {e}")
22
+ return {}
23
+
24
+
25
+ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
26
+ """
27
+ Update and format JSON file containing conference/journal data.
28
+
29
+ Args:
30
+ path_root (str): Root directory path
31
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals')
32
+
33
+ Returns:
34
+ Dict[str, Any]: Processed JSON data
35
+ """
36
+ # Load Json Data
37
+ json_dict = load_json_data(path_root, conferences_or_journals)
38
+
39
+ # Process and format text fields in JSON data.
40
+ for pub in json_dict:
41
+ for flag in ["txt_abouts", "txt_remarks"]:
42
+ data_list = [p for p in json_dict[pub].get(flag, []) if p.strip()]
43
+ temps = []
44
+ for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
45
+ temps.extend(split_text_by_length(line, 105))
46
+ if temps:
47
+ json_dict[pub].update({flag: temps})
48
+
49
+ for abbr in json_dict[pub][conferences_or_journals]:
50
+ for flag in ["txt_abouts", "txt_remarks"]:
51
+ data_list = [i for i in json_dict[pub][conferences_or_journals][abbr].get(flag, []) if i.strip()]
52
+ temps = []
53
+ for line in split_data_list(r"(\n+)", ["".join(data_list)], "next"):
54
+ temps.extend(split_text_by_length(line, 97))
55
+ if temps:
56
+ json_dict[pub][conferences_or_journals][abbr].update({flag: temps})
57
+
58
+ # Check for duplicate abbreviations
59
+ _check_duplicate_abbr(json_dict, conferences_or_journals)
60
+
61
+ # Save updated JSON
62
+ if json_dict:
63
+ path_file = os.path.join(path_root, f"{conferences_or_journals}.json")
64
+ with open(path_file, "w", encoding="utf-8") as f:
65
+ f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
66
+
67
+ return json_dict
68
+
69
+
70
+ def _check_duplicate_abbr(json_dict: Dict[str, Any], conferences_or_journals: str) -> None:
71
+ """
72
+ Check for duplicate abbreviations in the data.
73
+
74
+ Args:
75
+ json_dict: JSON data dictionary
76
+ conferences_or_journals (str): Type of publication ('conferences' or 'journals')
77
+
78
+ Raises:
79
+ ValueError: If duplicate abbreviations are found
80
+ """
81
+ abbr_list = []
82
+
83
+ for pub in json_dict:
84
+ if conferences_or_journals in json_dict[pub]:
85
+ for abbr in json_dict[pub][conferences_or_journals]:
86
+ if abbr in abbr_list:
87
+ raise ValueError(f"Duplicate abbreviation: {abbr} in {conferences_or_journals} {pub}")
88
+ abbr_list.append(abbr)
89
+
90
+ return None
@@ -0,0 +1,79 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ from typing import Optional
5
+
6
+ from .core._base import standardize_path
7
+ from .core.update_json import load_json_data, update_json_file
8
+ from .tools.generate_dict import GenerateDataDict
9
+ from .tools.write_dict import WriteDataToMd
10
+
11
+
12
+ def main_generate_md_files(
13
+ path_json: str,
14
+ path_output: str,
15
+ path_spidered_bibs: Optional[str] = None,
16
+ for_vue: bool = True,
17
+ conferences_or_journals: Optional[str] = None,
18
+ keywords_category_name: str = ""
19
+ ) -> None:
20
+ """
21
+ Generate markdown files for conferences and journals.
22
+
23
+ Args:
24
+ path_json: Path to JSON data file
25
+ path_output: Output directory for markdown files
26
+ path_spidered_bibs: Directory containing crawled BibTeX files
27
+ for_vue: Whether to generate Vue-compatible format
28
+ conferences_or_journals: Specify 'conferences' or 'journals', None for both
29
+ keywords_category_name: The category name of keywords
30
+ """
31
+ # Standardize all paths
32
+ path_json = standardize_path(path_json)
33
+ path_output = standardize_path(path_output)
34
+ path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
35
+
36
+ # Process keyword category name and load data
37
+ keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
38
+ category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
39
+ keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
40
+
41
+ # Validate data availability
42
+ if not keywords_list or not keywords_category_name:
43
+ keywords_list, keywords_category_name = [], ""
44
+
45
+ # Process both conferences and journals
46
+ for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
47
+ # Skip if specific type requested and doesn't match
48
+ if conferences_or_journals and conferences_or_journals.lower() != cj:
49
+ continue
50
+
51
+ # Update JSON data
52
+ json_dict = update_json_file(path_json, cj)
53
+ if not json_dict:
54
+ continue
55
+
56
+ # Generate data dictionaries
57
+ path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
58
+ generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
59
+ publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
60
+ if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
61
+ continue
62
+
63
+ # Initialize writer and save all markdown files
64
+ _path_output = os.path.join(path_output, f"{cj.title()}")
65
+ save_data = WriteDataToMd(
66
+ cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
67
+ )
68
+ # Save various documentation files
69
+ save_data.save_introductions()
70
+ save_data.save_categories(keywords_category_name, keywords_list)
71
+ save_data.save_categories_separate_keywords()
72
+
73
+ save_data.save_publishers()
74
+ save_data.save_publishers_separate_abbrs()
75
+
76
+ save_data.save_statistics(keywords_category_name, keywords_list)
77
+ save_data.save_statistics_separate_abbrs()
78
+
79
+ return None
@@ -0,0 +1,456 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ import re
5
+ from datetime import datetime
6
+ from typing import Optional
7
+
8
+
9
+ def conference_journal_header():
10
+ o = "|Publishers|Full/Homepage|Abbr/About|"
11
+ t = "|- |- |- |"
12
+ conference_header = [
13
+ f"{o}Acronym/Archive|Period/DBLP|Top|CCF|Submission|Days Left|Main Conf.|Days Left|Location|Keywords/Google|\n",
14
+ f"{t}- |- |- |- |- |- | |- |- |- |\n",
15
+ ]
16
+ journal_header = [
17
+ f"{o}Acronym/Issues|Period/DBLP|Top/Early|CCF|CAS|JCR|IF|Keywords/Google|\n",
18
+ f"{t}- |- |- |- |- |- |- |- |\n",
19
+ ]
20
+ return conference_header, journal_header
21
+
22
+
23
+ class GenerateDataDict(object):
24
+ def __init__(
25
+ self,
26
+ conferences_or_journals: str,
27
+ inproceedings_or_article: str,
28
+ json_dict: dict,
29
+ for_vue: bool = True,
30
+ path_spidered_conferences_or_journals: Optional[str] = None,
31
+ ) -> None:
32
+ self.cj = conferences_or_journals
33
+ self.ia = inproceedings_or_article
34
+ self.json_dict = json_dict
35
+
36
+ self.path_spidered_cj = path_spidered_conferences_or_journals
37
+ self.for_vue = for_vue
38
+
39
+ def generate(self):
40
+ """
41
+ Generate publisher metadata and keyword-based publication information.
42
+
43
+ Returns:
44
+ tuple: Contains two dictionaries:
45
+ - publisher_meta_abbr_dict: Publisher metadata and abbreviations
46
+ - publication_keyword_row_info_dict: Keyword-indexed publication info
47
+ """
48
+ publisher_meta_dict, keyword_abbr_meta_dict, publisher_abbr_meta_dict = {}, {}, {}
49
+
50
+ for publisher in self.json_dict:
51
+ # Extract and clean about texts
52
+ abouts = [p for p in self.json_dict[publisher].get("txt_abouts", []) if p.strip()]
53
+
54
+ # Extract and clean about URLs
55
+ urls_about = [p.strip() for p in self.json_dict[publisher].get("urls_about", []) if p.strip()]
56
+
57
+ # Get full names
58
+ names_full = self.json_dict[publisher].get("names_full", [])
59
+
60
+ # Get homepage URLs
61
+ urls_homepage = self.json_dict[publisher].get("urls_homepage", [])
62
+
63
+ # Extract and clean conference/journal URLs
64
+ urls_cj = [url.strip() for url in self.json_dict[publisher].get(f"urls_{self.cj}", []) if url.strip()]
65
+
66
+ # Create publisher URL with markdown formatting if homepage exists
67
+ publisher_url = f"[{publisher}]({urls_homepage[0]})" if urls_homepage else publisher
68
+
69
+ # Create full name URL with markdown formatting if available
70
+ if names_full:
71
+ full_url = f"[{names_full[0]}]({urls_homepage[0]})" if urls_homepage else names_full[0]
72
+ else:
73
+ full_url = publisher
74
+
75
+ # Extract and clean remarks
76
+ remarks = [p for p in self.json_dict[publisher].get("txt_remarks", []) if p.strip()]
77
+
78
+ # Update publisher metadata
79
+ publisher_meta_dict.setdefault(publisher, {}).update({
80
+ "full_name_url": full_url,
81
+ "txt_abouts": abouts,
82
+ "txt_remarks": remarks,
83
+ "urls_about": urls_about,
84
+ "url_conferences_or_journals": f"[{self.cj.title()}]({urls_cj[0]})" if urls_cj else "",
85
+ })
86
+
87
+ # Process each abbreviation (conference/journal)
88
+ for abbr in self.json_dict[publisher][self.cj]:
89
+ abbr_dict = self.json_dict[publisher][self.cj][abbr]
90
+
91
+ # Get conference/journal info and keywords
92
+ temp_dict, keywords = self.conference_or_journal(publisher_url, abbr, abbr_dict)
93
+
94
+ # Generate mermaid diagram data
95
+ mermaid = self.generate_mermaid_data(publisher, abbr, self.ia)
96
+
97
+ publisher_abbr_meta_dict.setdefault(publisher, {}).setdefault(abbr, {}).update(temp_dict)
98
+ publisher_abbr_meta_dict.setdefault(publisher, {}).setdefault(abbr, {}).update({"statistics": mermaid})
99
+
100
+ # Index by keywords for quick lookup
101
+ for keyword in keywords:
102
+ keyword_abbr_meta_dict.setdefault(keyword, {}).setdefault(abbr, {}).update(temp_dict)
103
+ keyword_abbr_meta_dict.setdefault(keyword, {}).setdefault(abbr, {}).update({"statistics": mermaid})
104
+
105
+ return publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict
106
+
107
+ def conference_or_journal(self, publisher_url: str, abbr: str, abbr_dict: dict):
108
+ """
109
+ Process conference or journal data and generate formatted information.
110
+
111
+ Args:
112
+ publisher_url: Publisher's URL
113
+ abbr: Abbreviation identifier
114
+ abbr_dict: Dictionary containing publication details
115
+
116
+ Returns:
117
+ dict: Contains formatted about text, remarks, and table row
118
+ list: Sorted list of keywords
119
+ """
120
+ # Validate full and abbreviated names match in length
121
+ self._validate_name_lengths(abbr_dict)
122
+
123
+ # Extract basic information
124
+ full_name, abbr_name = self._extract_full_abbr_names(abbr_dict)
125
+ url_home = self._extract_homepage_url(abbr_dict)
126
+ period = self._format_period_with_dblp(abbr_dict)
127
+
128
+ # Extract text content
129
+ abouts = self._extract_text_content(abbr_dict, "txt_abouts")
130
+ remarks = self._extract_text_content(abbr_dict, "txt_remarks")
131
+ url_about = self._extract_first_url(abbr_dict, "urls_about")
132
+
133
+ # Process keywords with Google search links
134
+ keywords, keywords_url = self._process_keywords(abbr_dict)
135
+
136
+ # Format top score with early access link if available
137
+ top = self._format_top_score(abbr_dict)
138
+
139
+ # Generate appropriate table row based on type
140
+ row_inf = self._generate_table_row(
141
+ publisher_url, full_name, abbr_name, url_home,
142
+ url_about, period, top, keywords_url, abbr, abbr_dict
143
+ )
144
+
145
+ return {"txt_abouts": abouts, "txt_remarks": remarks, "row_inf": row_inf}, keywords
146
+
147
+ def _validate_name_lengths(self, abbr_dict: dict):
148
+ """Validate that full and abbreviated names arrays have equal length."""
149
+ full_names = abbr_dict.get("names_full", [])
150
+ abbr_names = abbr_dict.get("names_abbr", [])
151
+ if len(full_names) != len(abbr_names):
152
+ raise ValueError(f"Length mismatch: {len(full_names)} full names vs {len(abbr_names)} abbreviated names")
153
+
154
+ def _extract_full_abbr_names(self, abbr_dict: dict):
155
+ """Extract full and abbreviated names from dictionary."""
156
+ # For journals: use first full name from list; for conferences: use single name
157
+ full_name = (abbr_dict.get("names_full", [""])[0] if self.cj == "journals" else abbr_dict.get("name", ""))
158
+ abbr_name = abbr_dict.get("names_abbr", [""])[0]
159
+ return full_name, abbr_name
160
+
161
+ def _extract_homepage_url(self, abbr_dict: dict):
162
+ """Extract and clean homepage URL."""
163
+ urls = [u.strip() for u in abbr_dict.get("urls_homepage", []) if u.strip()]
164
+ return urls[0] if urls else ""
165
+
166
+ def _format_period_with_dblp(self, abbr_dict: dict):
167
+ """Format publication period with DBLP link if available."""
168
+ start_year = abbr_dict.get("year_start", "")
169
+ end_year = abbr_dict.get("year_end", "")
170
+
171
+ # Create period string (e.g., "2020 - 2023" or "2020 -")
172
+ period = f"{start_year} - {end_year}" if start_year and end_year else f"{start_year} -" if start_year else ""
173
+
174
+ # Add DBLP link if acronym available
175
+ if acronym_dblp := abbr_dict.get("acronym_dblp", ""):
176
+ journal_conf = "journals" if self.cj == "journals" else "conf"
177
+ dblp_url = f"https://dblp.org/db/{journal_conf}/{acronym_dblp}/index.html"
178
+ period = f'[{period}]({dblp_url})'
179
+
180
+ return period
181
+
182
+ def _extract_text_content(self, abbr_dict: dict, key: str):
183
+ """Extract and clean text content from dictionary."""
184
+ return [text for text in abbr_dict.get(key, []) if text.strip()]
185
+
186
+ def _extract_first_url(self, abbr_dict: dict, key: str):
187
+ """Extract first URL from a list in dictionary."""
188
+ urls = [url.strip() for url in abbr_dict.get(key, []) if url.strip()]
189
+ return urls[0].split(",")[0] if urls else ""
190
+
191
+ def _process_keywords(self, abbr_dict: dict):
192
+ """Process keywords and convert to Google search URLs."""
193
+ keywords_dict = abbr_dict.get("keywords_dict", {})
194
+
195
+ # Clean and sort keywords
196
+ cleaned_keywords = {}
197
+ for category, words in keywords_dict.items():
198
+ if category.strip():
199
+ sorted_words = sorted(set([word.strip() for word in words if word.strip()]))
200
+ cleaned_keywords[category.strip()] = sorted_words
201
+
202
+ # Flatten keywords and remove duplicates
203
+ all_keywords = []
204
+ for category, words in cleaned_keywords.items():
205
+ if words:
206
+ all_keywords.extend(words)
207
+ else:
208
+ all_keywords.append(category)
209
+ all_keywords = sorted(set(all_keywords))
210
+ # Create Google search links for each keyword
211
+ google_base = "https://www.google.com/search?q="
212
+ keywords_url = [
213
+ f"[{keyword}]({google_base}" + re.sub(r"\s+", "+", keyword) + ")" for keyword in all_keywords
214
+ ]
215
+
216
+ # For category
217
+ # Flatten keywords and remove duplicates
218
+ all_keywords = []
219
+ for category, words in cleaned_keywords.items():
220
+ all_keywords.extend(words)
221
+ all_keywords.append(category)
222
+ all_keywords = sorted(set(all_keywords))
223
+
224
+ return all_keywords, keywords_url
225
+
226
+ def _format_top_score(self, abbr_dict: dict):
227
+ """Format top score with optional early access link."""
228
+ is_top = "True" if abbr_dict.get("score_top", False) else "False"
229
+ url_early_access = abbr_dict.get("url_early_access", "")
230
+ return self._format_link(is_top, url_early_access)
231
+
232
+ def _format_link(self, text, url):
233
+ """Format text as markdown link if URL provided."""
234
+ return f"[{text}]({url})" if url else text
235
+
236
+ def _generate_table_row(
237
+ self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
238
+ ):
239
+ """Generate appropriate table row based on publication type."""
240
+ if self.cj == "conferences":
241
+ return self._generate_for_conference(
242
+ publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
243
+ )
244
+ else:
245
+ return self._generate_for_journal(
246
+ publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
247
+ )
248
+
249
+ # Conferences
250
+ def _generate_for_conference(
251
+ self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
252
+ ):
253
+ """
254
+ Generate a markdown table row for conference information.
255
+
256
+ Args:
257
+ publisher_url: URL of the publisher
258
+ full_name: Full name of the conference
259
+ abbr_name: Abbreviated name of the conference
260
+ url_home: Homepage URL
261
+ url_about: About page URL
262
+ period: Conference period/frequency
263
+ top: Conference ranking/tier
264
+ keywords: List of keywords
265
+ abbr: Conference abbreviation
266
+ abbr_dict: Dictionary containing conference details
267
+
268
+ Returns:
269
+ str: Formatted markdown table row
270
+ """
271
+ # Get archive URL and format link
272
+ archive_url = self._extract_first_url(abbr_dict, "urls_archive")
273
+ archive_display = self._format_link(abbr, archive_url)
274
+
275
+ # Process conference dates
276
+ abstract_due, start_date, today = self._process_conference_dates(abbr_dict)
277
+
278
+ # Format date indicators for Vue or standard display
279
+ abstract_indicator, start_indicator = self._format_date_indicators(
280
+ abstract_due, start_date, today
281
+ )
282
+
283
+ # Get year URL for start date link
284
+ year_url = abbr_dict.get("conf_url", "")
285
+
286
+ # Build and return markdown table row
287
+ return self._build_conference_row(
288
+ publisher_url, full_name, abbr_name, url_home, url_about,
289
+ archive_display, period, top, abbr_dict, abstract_due,
290
+ abstract_indicator, start_date, start_indicator, year_url,
291
+ keywords
292
+ )
293
+
294
+ def _process_conference_dates(self, abbr_dict):
295
+ """Parse and return conference dates."""
296
+ # Parse abstract due date
297
+ abstract_due = None
298
+ if due_str := abbr_dict.get("conf_abstract_due", "").strip():
299
+ abstract_due = datetime.strptime(due_str, "%d/%m/%Y").date()
300
+
301
+ # Parse conference start date
302
+ start_date = None
303
+ if start_str := abbr_dict.get("conf_date_start", "").strip():
304
+ start_date = datetime.strptime(start_str, "%d/%m/%Y").date()
305
+
306
+ # Get today's date
307
+ today = datetime.strptime(datetime.now().strftime("%d/%m/%Y"), "%d/%m/%Y").date()
308
+
309
+ return abstract_due, start_date, today
310
+
311
+ def _format_date_indicators(self, abstract_due, start_date, today):
312
+ """Format date indicators for display."""
313
+ abstract_indicator, start_indicator = "", ""
314
+
315
+ if self.for_vue:
316
+ # Vue.js template format
317
+ if abstract_due:
318
+ abstract_indicator = f"**{{{{ diffDate('{abstract_due}') }}}}**"
319
+ if start_date:
320
+ start_indicator = f"**{{{{ diffDate('{start_date}') }}}}**"
321
+ else:
322
+ # Standard day count format
323
+ if abstract_due:
324
+ abstract_indicator = (abstract_due - today).days if today <= abstract_due else "Expired"
325
+ if start_date:
326
+ start_indicator = (start_date - today).days if today <= start_date else "Expired"
327
+
328
+ return abstract_indicator, start_indicator
329
+
330
+ def _build_conference_row(
331
+ self, publisher_url, full_name, abbr_name, url_home, url_about, archive_display, period, top, abbr_dict,
332
+ abstract_due, abstract_indicator, start_date, start_indicator, year_url, keywords
333
+ ):
334
+ """Construct conference table row string."""
335
+ # Format date strings
336
+ abstract_date_str = abstract_due.strftime("%d/%m/%Y") if abstract_due else ""
337
+ start_date_str = start_date.strftime("%d/%m/%Y") if start_date else ""
338
+
339
+ # Format start date with link if available
340
+ start_date_display = self._format_link(start_date_str, year_url) if start_date_str else ""
341
+
342
+ # Build table row
343
+ return f"|{publisher_url}|" \
344
+ f"{self._format_link(full_name, url_home)}|" \
345
+ f"{self._format_link(abbr_name, url_about)}|" \
346
+ f"{archive_display}|" \
347
+ f"{period}|" \
348
+ f"{top}|" \
349
+ f"{abbr_dict.get('score_ccf', '')}|" \
350
+ f"{abstract_date_str}|" \
351
+ f"{abstract_indicator}|" \
352
+ f"{start_date_display}|" \
353
+ f"{start_indicator}|" \
354
+ f"{abbr_dict.get('conf_location', '').strip()}|" \
355
+ f"{'; '.join(keywords)}|"
356
+
357
+ # Journals
358
+ def _generate_for_journal(
359
+ self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
360
+ ):
361
+ """
362
+ Generate a markdown table row for journal information.
363
+
364
+ Args:
365
+ publisher_url: URL of the publisher
366
+ full_name: Full name of the journal
367
+ abbr_name: Abbreviated name of the journal
368
+ url_home: Homepage URL
369
+ url_about: About page URL
370
+ period: Publication period/frequency
371
+ top: Journal ranking/tier
372
+ keywords: List of keywords
373
+ abbr: Journal abbreviation
374
+ abbr_dict: Dictionary containing journal details
375
+
376
+ Returns:
377
+ str: Formatted markdown table row
378
+ """
379
+ # Get issues URL and format link
380
+ issues_url = self._extract_first_url(abbr_dict, "urls_issues")
381
+ issues_display = self._format_link(abbr, issues_url)
382
+
383
+ # Build and return markdown table row
384
+ return self._build_journal_row(
385
+ publisher_url, full_name, abbr_name, url_home, url_about,
386
+ issues_display, period, top, abbr_dict, keywords
387
+ )
388
+
389
+ def _build_journal_row(
390
+ self, publisher_url, full_name, abbr_name, url_home, url_about, issues_display, period, top, abbr_dict, keywords
391
+ ):
392
+ """Construct journal table row string."""
393
+ return f"|{publisher_url}|" \
394
+ f"{self._format_link(full_name, url_home)}|" \
395
+ f"{self._format_link(abbr_name, url_about)}|" \
396
+ f"{issues_display}|" \
397
+ f"{period}|" \
398
+ f"{top}|" \
399
+ f"{abbr_dict.get('score_ccf', '')}|" \
400
+ f"{abbr_dict.get('score_cas', '')}|" \
401
+ f"{abbr_dict.get('score_jcr', '')}|" \
402
+ f"{abbr_dict.get('score_if', '')}|" \
403
+ f"{'; '.join(keywords)}|"
404
+
405
+ # Mermaid data
406
+ def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
407
+ path_spidered_cj = self.path_spidered_cj if self.path_spidered_cj else ""
408
+ path_readme = os.path.join(path_spidered_cj, publisher, abbr, inproceedings_or_article)
409
+ full_readme = os.path.expanduser(os.path.join(path_readme, "README.md"))
410
+ if not os.path.exists(full_readme):
411
+ return []
412
+
413
+ mermaid, data_dict = [], {}
414
+ # |AAAI|1980|95|Proceedings of the First National Conference on Artificial Intelligence|
415
+ regex = re.compile(r"\|.*\|([0-9]+)\|([0-9]+)\|.*\|")
416
+ with open(full_readme, "r") as file:
417
+ data_list = file.readlines()
418
+ for line in data_list:
419
+ if mch := regex.search(line):
420
+ data_dict.setdefault(mch.group(1), []).append(mch.group(2))
421
+ data_dict = {year: sum([int(n) for n in data_dict[year]]) for year in data_dict}
422
+
423
+ # Mermaid
424
+ if len(data_dict) != 0:
425
+ mermaid = ["```mermaid\n"]
426
+ mermaid.extend(
427
+ [
428
+ '---\n',
429
+ 'config:\n',
430
+ ' xyChart:\n',
431
+ ' width: 1200\n',
432
+ ' height: 600\n',
433
+ ' themeVariables:\n',
434
+ ' xyChart:\n',
435
+ ' titleColor: "#ff0000"\n',
436
+ '---\n'
437
+ ]
438
+ )
439
+ mermaid.extend(["xychart-beta\n", f' title "{abbr}"\n'])
440
+
441
+ x_axis, bar, line = [], [], []
442
+ for year in data_dict:
443
+ x_axis.append(int(year))
444
+ bar.append(data_dict[year])
445
+ line.append(data_dict[year])
446
+
447
+ idx = next((i for i, year in enumerate(x_axis) if year >= 2000), len(x_axis))
448
+ x_axis, bar, line = x_axis[idx:], bar[idx:], line[idx:]
449
+
450
+ mermaid.append(f" x-axis {x_axis}\n")
451
+ mermaid.append(' y-axis "Number of Papers"\n')
452
+ mermaid.append(f" bar {bar}\n")
453
+ mermaid.append(f" line {line}\n")
454
+ mermaid.append("```\n")
455
+
456
+ return mermaid
@@ -0,0 +1,284 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ from typing import List
5
+
6
+ from ..core._base import standardize_path
7
+ from .generate_dict import conference_journal_header
8
+
9
+
10
+ def conference_journal_informations():
11
+ conference_inf = [
12
+ "!> [List of Upcoming International Conferences](https://internationalconferencealerts.com/all-events.php)\n\n",
13
+ "!> [Conferences in Theoretical Computer Science](https://www.lix.polytechnique.fr/~hermann/conf.php)\n\n"
14
+ ]
15
+ journal_inf = []
16
+ return conference_inf, journal_inf
17
+
18
+
19
+ class WriteDataToMd(object):
20
+ """Class to write publication data to Markdown files."""
21
+
22
+ def __init__(
23
+ self,
24
+ conferences_or_journals: str,
25
+ inproceedings_or_article: str,
26
+ publisher_meta_dict: dict,
27
+ publisher_abbr_meta_dict: dict,
28
+ keyword_abbr_meta_dict: dict,
29
+ path_output: str,
30
+ ) -> None:
31
+ """Initialize with publication data and output path."""
32
+ self.cj = conferences_or_journals # "conferences" or "journals"
33
+ self.ia = inproceedings_or_article # "inproceedings" or "article"
34
+ self.publisher_meta_dict = publisher_meta_dict
35
+ self.publisher_abbr_meta_dict = publisher_abbr_meta_dict
36
+ self.keyword_abbr_meta_dict = keyword_abbr_meta_dict
37
+ self.path_output = standardize_path(path_output)
38
+
39
+ self._default_inf = [
40
+ "- The data for TOP, CCF, CAS, JCR, and IF are sourced from [easyScholar](https://www.easyscholar.cc/).\n\n"
41
+ ]
42
+
43
+ def save_introductions(self) -> None:
44
+ """Save introduction file with all conferences/journals list."""
45
+ conference_header, journal_header = conference_journal_header()
46
+ conference_inf, journal_inf = conference_journal_informations()
47
+
48
+ data_list = [f"# {self.cj.title()}\n\n"]
49
+ data_list.extend(self._default_inf)
50
+
51
+ # Add appropriate headers based on type
52
+ if self.cj.lower() == "conferences":
53
+ data_list.extend(conference_inf)
54
+ data_list.append("| " + conference_header[0])
55
+ data_list.append("|- " + conference_header[1])
56
+ else:
57
+ data_list.extend(journal_inf)
58
+ data_list.append("| " + journal_header[0])
59
+ data_list.append("|- " + journal_header[1])
60
+
61
+ # Add all publications to table
62
+ idx = 1
63
+ for publisher in self.publisher_abbr_meta_dict:
64
+ for abbr in self.publisher_abbr_meta_dict[publisher]:
65
+ row_info = self.publisher_abbr_meta_dict[publisher][abbr]['row_inf']
66
+ data_list.append(f"|{idx}{row_info}\n")
67
+ idx += 1
68
+
69
+ # Write to file
70
+ output_file = os.path.join(self.path_output, f"Introductions_{self.cj.title()}.md")
71
+ with open(output_file, "w") as f:
72
+ f.writelines(data_list)
73
+
74
+ # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
75
+ def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: List[str]):
76
+ keywords = list(self.keyword_abbr_meta_dict.keys())
77
+
78
+ # Get and sort publication types
79
+ if keywords_category_name and keywords_list:
80
+ _keywords = []
81
+ for keyword in keywords_list:
82
+ if keyword in keywords:
83
+ _keywords.append(keyword)
84
+ return _keywords
85
+ else:
86
+ # default
87
+ return sorted(keywords)
88
+
89
+ # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
90
+ def save_categories(self, keywords_category_name: str, keywords_list: List[str]) -> None:
91
+ """Save publications categorized by keywords."""
92
+ conference_header, journal_header = conference_journal_header()
93
+ data_list = [f"# {self.cj.title()}\n\n"]
94
+ data_list.extend(self._default_inf)
95
+
96
+ # Add publications for each category
97
+ for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
98
+ data_list.append(f"## {keyword}\n\n")
99
+
100
+ # Add appropriate header
101
+ if self.cj == "conferences":
102
+ data_list.extend(conference_header)
103
+ else:
104
+ data_list.extend(journal_header)
105
+
106
+ # Add all publications in this category
107
+ for abbr in self.keyword_abbr_meta_dict[keyword]:
108
+ data_list.append(self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"] + "\n")
109
+ data_list.append("\n")
110
+
111
+ # Write to file
112
+ category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
113
+ with open(os.path.join(self.path_output, f"Categories_{self.cj.title()}{category_postfix}.md"), "w") as f:
114
+ f.writelines(data_list)
115
+
116
+ return None
117
+
118
+ def save_categories_separate_keywords(self) -> None:
119
+ conference_header, journal_header = conference_journal_header()
120
+
121
+ # Add publications for each category
122
+ for keyword in self.keyword_abbr_meta_dict:
123
+ data_list = [f"# {keyword}\n\n"]
124
+ data_list.extend(self._default_inf)
125
+
126
+ # Add appropriate header
127
+ if self.cj == "conferences":
128
+ data_list.extend(conference_header)
129
+ else:
130
+ data_list.extend(journal_header)
131
+
132
+ # Add all publications in this category
133
+ for abbr in self.keyword_abbr_meta_dict[keyword]:
134
+ data_list.append(self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"] + "\n")
135
+ data_list.append("\n")
136
+
137
+ # Write keyword-specific file
138
+ path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
139
+ with open(os.path.join(path_key, f"{keyword.replace(' ', '_')}.md"), "w") as f:
140
+ f.writelines(data_list)
141
+
142
+ return None
143
+
144
+ # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
145
+ def save_publishers(self) -> None:
146
+ """Save publisher overview file with basic information."""
147
+ data_list_pub = [
148
+ f"# Introductions of Publishers and {self.cj.title()}\n\n",
149
+ "| |Publishers|About US|Conferences/Journals|Separate Links|\n",
150
+ "|-|- |- |- |- |\n"
151
+ ]
152
+ idx = 1
153
+
154
+ # Add each publisher to table
155
+ for pub in self.publisher_meta_dict:
156
+ meta = self.publisher_meta_dict[pub]
157
+
158
+ full_name_url, about_url, cj_url, local_url = "", "", "", ""
159
+ if x := meta.get("full_name_url", ""):
160
+ full_name_url = x
161
+ if x := meta.get("url_conferences_or_journals", ""):
162
+ cj_url = x
163
+ if pub_intr_urls := meta.get("urls_about", []):
164
+ about_url = f"[About US]({pub_intr_urls[0]})"
165
+
166
+ local_url = f"[{pub}](data/{self.cj.title()}/Publishers_{self.cj.title()}/{pub}.md)"
167
+
168
+ # Create table row
169
+ row = f"| {idx} | {full_name_url} | {about_url} | {cj_url} | {local_url} |\n"
170
+ data_list_pub.append(row)
171
+ idx += 1
172
+
173
+ # Write to file
174
+ with open(os.path.join(self.path_output, f"Publishers_{self.cj.title()}.md"), "w") as f:
175
+ f.writelines(data_list_pub)
176
+ return None
177
+
178
+ def save_publishers_separate_abbrs(self) -> None:
179
+ conference_header, journal_header = conference_journal_header()
180
+ for pub in self.publisher_meta_dict:
181
+ data_list = [f"# {pub}\n\n"]
182
+ data_list.extend(self._default_inf)
183
+ meta = self.publisher_meta_dict[pub]
184
+
185
+ # Add about and remarks sections
186
+ for flag in ["txt_remarks", "txt_abouts"]:
187
+ if temps := meta.get(flag, []):
188
+ temps[-1] = f"{temps[-1].rstrip()}\n\n"
189
+ if temps:
190
+ data_list.append(f"## {flag.title()}\n\n")
191
+ data_list.extend(temps)
192
+
193
+ # Add each conference/journal abbreviation
194
+ if pub not in self.publisher_abbr_meta_dict:
195
+ continue
196
+
197
+ for abbr in self.publisher_abbr_meta_dict[pub]:
198
+ data_list.append(f"## {abbr}\n\n")
199
+
200
+ # Add appropriate header
201
+ if self.cj == "conferences":
202
+ data_list.extend(conference_header)
203
+ else:
204
+ data_list.extend(journal_header)
205
+
206
+ # Add row information
207
+ row_info = self.publisher_abbr_meta_dict[pub][abbr]["row_inf"]
208
+ data_list.append(f'{row_info}\n\n')
209
+
210
+ # Add remarks and about for this abbreviation
211
+ for flag in ["txt_remarks", "txt_abouts"]:
212
+ if temps := self.publisher_abbr_meta_dict[pub][abbr].get(flag, []):
213
+ temps[-1] = f"{temps[-1].rstrip()}\n\n"
214
+ if temps:
215
+ data_list.append(f"### {flag.split('_')[-1].title()}\n\n")
216
+ data_list.extend(temps)
217
+
218
+ # Add statistics if available
219
+ if statistics := self.publisher_abbr_meta_dict[pub][abbr].get("statistics", []):
220
+ data_list.extend(statistics)
221
+ data_list.append("\n")
222
+
223
+ # Write publisher-specific file
224
+ path_pub = standardize_path(os.path.join(self.path_output, f"Publishers_{self.cj.title()}"))
225
+ with open(os.path.join(path_pub, f"{pub}.md"), "w") as f:
226
+ f.writelines(data_list)
227
+
228
+ return None
229
+
230
+ # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
231
+ def save_statistics(self, keywords_category_name: str, keywords_list: List[str]) -> None:
232
+ data_list = [
233
+ f"# Statistics of keywords in {self.cj.title()}\n\n",
234
+ "| |keywords|Separate Links|\n",
235
+ "|-|- |- |\n"
236
+ ]
237
+ idx = 1
238
+
239
+ # Add publications for each category
240
+ for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
241
+ local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{keyword.replace(' ', '_')}.md)"
242
+
243
+ # Create table row
244
+ row = f"| {idx} | {keyword} | {local_url} |\n"
245
+ data_list.append(row)
246
+ idx += 1
247
+
248
+ # Write to file
249
+ category_postfix = f"_{keywords_category_name.title()}" if keywords_category_name else ""
250
+ with open(os.path.join(self.path_output, f"Statistics_{self.cj.title()}{category_postfix}.md"), "w") as f:
251
+ f.writelines(data_list)
252
+ return None
253
+
254
+ def save_statistics_separate_abbrs(self) -> None:
255
+ conference_header, journal_header = conference_journal_header()
256
+
257
+ # Add publications for each category
258
+ for keyword in self.keyword_abbr_meta_dict:
259
+ data_list = [f"# {keyword}\n\n"]
260
+
261
+ for abbr in self.keyword_abbr_meta_dict[keyword]:
262
+ data_list.append(f"## {abbr}\n\n")
263
+
264
+ # Add appropriate header
265
+ if self.cj == "conferences":
266
+ data_list.extend(conference_header)
267
+ else:
268
+ data_list.extend(journal_header)
269
+
270
+ # Add row information
271
+ row_info = self.keyword_abbr_meta_dict[keyword][abbr]["row_inf"]
272
+ data_list.append(f'{row_info}\n\n')
273
+
274
+ # Add statistics if available
275
+ if statistics := self.keyword_abbr_meta_dict[keyword][abbr].get("statistics", []):
276
+ data_list.extend(statistics)
277
+ data_list.append("\n")
278
+
279
+ # Write publisher-specific file
280
+ path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
281
+ with open(os.path.join(path_pub, f"{keyword.replace(' ', '_')}.md"), "w") as f:
282
+ f.writelines(data_list)
283
+
284
+ return None
@@ -0,0 +1,62 @@
1
+ [tool.poetry]
2
+ name = "pyformatjson"
3
+ version = "0.0.1"
4
+ description = "pyformatjson"
5
+ license = "GPL-3.0-or-later"
6
+ authors = ["NextAI <nextartifintell@gmail.com>"]
7
+ maintainers = ["NextAI <nextartifintell@gmail.com>"]
8
+ readme = ["README.md"]
9
+ homepage = "https://github.com/NextArtifIntell/pyformatjson"
10
+ repository = "https://github.com/NextArtifIntell/pyformatjson"
11
+ documentation = "https://github.com/NextArtifIntell/pyformatjson"
12
+ keywords = ["Python", "json"]
13
+ classifiers = ["Topic :: Software Development :: Libraries :: Python Modules"]
14
+
15
+ [tool.poetry.dependencies]
16
+ python = ">=3.13"
17
+
18
+ [tool.poetry.group.dev.dependencies]
19
+ mypy = "^1.18.2"
20
+ ruff = "^0.13.1"
21
+ pycodestyle = "^2.14.0"
22
+ pydocstyle = "^6.3.0"
23
+ flake8 = "^7.3.0"
24
+ isort = "^6.0.1"
25
+ black = "^25.9.0"
26
+
27
+
28
+ [tool.pyright]
29
+ venvPath = "."
30
+ venv = ".venv"
31
+
32
+ [tool.ruff]
33
+ line-length = 120
34
+ indent-width = 4
35
+
36
+ exclude = [".venv"]
37
+ extend-exclude = ["tests"]
38
+
39
+ [tool.ruff.lint]
40
+ extend-select = ["I"]
41
+
42
+ [tool.ruff.format]
43
+ quote-style = "double"
44
+ indent-style = "space"
45
+
46
+ [tool.mypy]
47
+ ignore_missing_imports = true
48
+
49
+ [tool.black]
50
+ line-length = 120
51
+ skip-magic-trailing-comma = true
52
+
53
+ [tool.isort]
54
+ line_length = 120
55
+ profile = "black"
56
+ include_trailing_comma = true
57
+
58
+ [tool.pytest]
59
+
60
+ [build-system]
61
+ requires = ["poetry-core>=1.0.0"]
62
+ build-backend = "poetry.core.masonry.api"