pyformatjson 0.0.3__tar.gz → 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/PKG-INFO +4 -3
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyformatjson/core/update_json.py +7 -15
- pyformatjson-0.2.0/pyformatjson/main.py +106 -0
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyformatjson/tools/generate_dict.py +74 -7
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyformatjson/tools/write_dict.py +84 -4
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyproject.toml +4 -3
- pyformatjson-0.0.3/pyformatjson/main.py +0 -155
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/README.md +0 -0
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyformatjson/__init__.py +0 -0
- {pyformatjson-0.0.3 → pyformatjson-0.2.0}/pyformatjson/core/_base.py +0 -0
|
@@ -1,16 +1,17 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: pyformatjson
|
|
3
|
-
Version: 0.0
|
|
3
|
+
Version: 0.2.0
|
|
4
4
|
Summary: pyformatjson
|
|
5
5
|
License: GPL-3.0-or-later
|
|
6
|
-
Keywords: Python,
|
|
6
|
+
Keywords: Python,Json
|
|
7
7
|
Author: NextAI
|
|
8
8
|
Author-email: nextartifintell@gmail.com
|
|
9
9
|
Maintainer: NextAI
|
|
10
10
|
Maintainer-email: nextartifintell@gmail.com
|
|
11
|
-
Requires-Python: >=3.
|
|
11
|
+
Requires-Python: >=3.12
|
|
12
12
|
Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
|
|
13
13
|
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
14
15
|
Classifier: Programming Language :: Python :: 3.13
|
|
15
16
|
Classifier: Programming Language :: Python :: 3.14
|
|
16
17
|
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
@@ -16,7 +16,7 @@ def load_json_data(path_json: str, filename: str) -> Dict:
|
|
|
16
16
|
|
|
17
17
|
Args:
|
|
18
18
|
path_json (str): Directory path containing the JSON file.
|
|
19
|
-
filename (str): Name of the JSON file (
|
|
19
|
+
filename (str): Name of the JSON file (with .json extension).
|
|
20
20
|
|
|
21
21
|
Returns:
|
|
22
22
|
Dict: The loaded JSON data as a dictionary, or empty dict if file not found
|
|
@@ -27,7 +27,7 @@ def load_json_data(path_json: str, filename: str) -> Dict:
|
|
|
27
27
|
{"publisher1": {"conferences": {...}}}
|
|
28
28
|
"""
|
|
29
29
|
try:
|
|
30
|
-
file_path = os.path.join(path_json,
|
|
30
|
+
file_path = os.path.join(path_json, filename)
|
|
31
31
|
if not os.path.exists(file_path):
|
|
32
32
|
return {}
|
|
33
33
|
|
|
@@ -35,11 +35,11 @@ def load_json_data(path_json: str, filename: str) -> Dict:
|
|
|
35
35
|
return json.load(file)
|
|
36
36
|
|
|
37
37
|
except Exception as e:
|
|
38
|
-
print(f"Error loading {filename}
|
|
38
|
+
print(f"Error loading {filename}: {e}")
|
|
39
39
|
return {}
|
|
40
40
|
|
|
41
41
|
|
|
42
|
-
def update_json_file(
|
|
42
|
+
def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
|
|
43
43
|
"""Update and format JSON file containing conference/journal data.
|
|
44
44
|
|
|
45
45
|
This function loads JSON data, processes and formats text fields by splitting
|
|
@@ -47,21 +47,14 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
|
|
|
47
47
|
saves the updated data back to the file.
|
|
48
48
|
|
|
49
49
|
Args:
|
|
50
|
-
|
|
50
|
+
full_json_cj (str): Full path to the conferences/journals JSON file
|
|
51
51
|
conferences_or_journals (str): Type of publication ('conferences' or 'journals').
|
|
52
52
|
|
|
53
53
|
Returns:
|
|
54
54
|
Dict[str, Any]: Processed JSON data dictionary.
|
|
55
|
-
|
|
56
|
-
Raises:
|
|
57
|
-
ValueError: If duplicate abbreviations are found in the data.
|
|
58
|
-
|
|
59
|
-
Example:
|
|
60
|
-
>>> update_json_file("/data", "conferences")
|
|
61
|
-
{"publisher1": {"conferences": {"conf1": {...}}}}
|
|
62
55
|
"""
|
|
63
56
|
# Load Json Data
|
|
64
|
-
json_dict = load_json_data(
|
|
57
|
+
json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
|
|
65
58
|
|
|
66
59
|
# Process and format text fields in JSON data.
|
|
67
60
|
for pub in json_dict:
|
|
@@ -87,8 +80,7 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
|
|
|
87
80
|
|
|
88
81
|
# Save updated JSON
|
|
89
82
|
if json_dict:
|
|
90
|
-
|
|
91
|
-
with open(path_file, "w", encoding="utf-8") as f:
|
|
83
|
+
with open(full_json_cj, "w", encoding="utf-8") as f:
|
|
92
84
|
f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
93
85
|
|
|
94
86
|
return json_dict
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
# coding=utf-8
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from typing import Optional
|
|
5
|
+
|
|
6
|
+
from .core._base import standardize_path
|
|
7
|
+
from .core.update_json import load_json_data, update_json_file
|
|
8
|
+
from .tools.generate_dict import GenerateDataDict
|
|
9
|
+
from .tools.write_dict import WriteDataToMd
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def main_generate_md_files(
|
|
13
|
+
full_json_c: str,
|
|
14
|
+
full_json_j: str,
|
|
15
|
+
full_json_k: str,
|
|
16
|
+
path_output: str,
|
|
17
|
+
path_spidered_bibs: Optional[str] = None,
|
|
18
|
+
keywords_category_name: str = "",
|
|
19
|
+
for_vue: bool = True,
|
|
20
|
+
) -> None:
|
|
21
|
+
"""Generate comprehensive markdown documentation for academic publications.
|
|
22
|
+
|
|
23
|
+
This function serves as the main entry point for processing conference and journal
|
|
24
|
+
data from JSON files and generating various markdown documentation outputs. It
|
|
25
|
+
handles both conference and journal data processing, creates categorized outputs,
|
|
26
|
+
and generates statistics and publisher information files.
|
|
27
|
+
|
|
28
|
+
The function processes three types of JSON files:
|
|
29
|
+
- Conference data (full_json_c)
|
|
30
|
+
- Journal data (full_json_j)
|
|
31
|
+
- Keywords data (full_json_k)
|
|
32
|
+
|
|
33
|
+
Args:
|
|
34
|
+
full_json_c (str): Full path to the conferences JSON file containing
|
|
35
|
+
conference publication data.
|
|
36
|
+
full_json_j (str): Full path to the journals JSON file containing
|
|
37
|
+
journal publication data.
|
|
38
|
+
full_json_k (str): Full path to the keywords JSON file containing
|
|
39
|
+
keyword categorization data.
|
|
40
|
+
path_output (str): Output directory path where all generated markdown
|
|
41
|
+
files will be saved.
|
|
42
|
+
path_spidered_bibs (Optional[str], optional): Directory path containing
|
|
43
|
+
spidered BibTeX files for additional data processing. Defaults to None.
|
|
44
|
+
keywords_category_name (str, optional): Category name for filtering
|
|
45
|
+
keywords. If provided, only keywords from this category will be
|
|
46
|
+
processed. Defaults to "".
|
|
47
|
+
for_vue (bool, optional): Whether to generate Vue.js-compatible format
|
|
48
|
+
for date calculations and dynamic content. Defaults to True.
|
|
49
|
+
|
|
50
|
+
Returns:
|
|
51
|
+
None: This function does not return a value.
|
|
52
|
+
"""
|
|
53
|
+
# Standardize all paths
|
|
54
|
+
full_json_c = os.path.expanduser(full_json_c)
|
|
55
|
+
full_json_j = os.path.expanduser(full_json_j)
|
|
56
|
+
full_json_k = os.path.expanduser(full_json_k)
|
|
57
|
+
|
|
58
|
+
path_output = standardize_path(path_output)
|
|
59
|
+
|
|
60
|
+
path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
|
|
61
|
+
|
|
62
|
+
# Process keyword category name and load data
|
|
63
|
+
keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
|
|
64
|
+
category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
|
|
65
|
+
keywords_json = load_json_data(os.path.dirname(full_json_k), os.path.basename(full_json_k))
|
|
66
|
+
keywords_list = keywords_json.get(f"{category_prefix}keywords", [])
|
|
67
|
+
|
|
68
|
+
# Validate data availability
|
|
69
|
+
if not keywords_list or not keywords_category_name:
|
|
70
|
+
keywords_list, keywords_category_name = [], ""
|
|
71
|
+
|
|
72
|
+
# Process both conferences and journals
|
|
73
|
+
for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
|
|
74
|
+
# Update JSON data
|
|
75
|
+
json_dict = {}
|
|
76
|
+
if cj == "conferences":
|
|
77
|
+
json_dict = update_json_file(full_json_c, cj)
|
|
78
|
+
elif cj == "journals":
|
|
79
|
+
json_dict = update_json_file(full_json_j, cj)
|
|
80
|
+
if not json_dict:
|
|
81
|
+
continue
|
|
82
|
+
|
|
83
|
+
# Generate data dictionaries
|
|
84
|
+
path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
|
|
85
|
+
generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
|
|
86
|
+
publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
|
|
87
|
+
if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
|
|
88
|
+
continue
|
|
89
|
+
|
|
90
|
+
# Initialize writer and save all markdown files
|
|
91
|
+
_path_output = os.path.join(path_output, f"{cj.title()}")
|
|
92
|
+
save_data = WriteDataToMd(
|
|
93
|
+
cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
|
|
94
|
+
)
|
|
95
|
+
# Save various documentation files
|
|
96
|
+
save_data.save_introductions()
|
|
97
|
+
save_data.save_categories(keywords_category_name, keywords_list)
|
|
98
|
+
save_data.save_categories_separate_keywords()
|
|
99
|
+
|
|
100
|
+
save_data.save_publishers()
|
|
101
|
+
save_data.save_publishers_separate_abbrs()
|
|
102
|
+
|
|
103
|
+
save_data.save_statistics(keywords_category_name, keywords_list)
|
|
104
|
+
save_data.save_statistics_separate_abbrs()
|
|
105
|
+
|
|
106
|
+
return None
|
|
@@ -252,12 +252,26 @@ class GenerateDataDict(object):
|
|
|
252
252
|
return full_name, abbr_name
|
|
253
253
|
|
|
254
254
|
def _extract_homepage_url(self, abbr_dict: dict):
|
|
255
|
-
"""Extract and clean homepage URL.
|
|
255
|
+
"""Extract and clean homepage URL.
|
|
256
|
+
|
|
257
|
+
Args:
|
|
258
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
259
|
+
|
|
260
|
+
Returns:
|
|
261
|
+
str: The first valid homepage URL, or empty string if none found.
|
|
262
|
+
"""
|
|
256
263
|
urls = [u.strip() for u in abbr_dict.get("urls_homepage", []) if u.strip()]
|
|
257
264
|
return urls[0] if urls else ""
|
|
258
265
|
|
|
259
266
|
def _format_period_with_dblp(self, abbr_dict: dict):
|
|
260
|
-
"""Format publication period with DBLP link if available.
|
|
267
|
+
"""Format publication period with DBLP link if available.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
271
|
+
|
|
272
|
+
Returns:
|
|
273
|
+
str: Formatted period string with optional DBLP link.
|
|
274
|
+
"""
|
|
261
275
|
start_year = abbr_dict.get("year_start", "")
|
|
262
276
|
end_year = abbr_dict.get("year_end", "")
|
|
263
277
|
|
|
@@ -273,16 +287,41 @@ class GenerateDataDict(object):
|
|
|
273
287
|
return period
|
|
274
288
|
|
|
275
289
|
def _extract_text_content(self, abbr_dict: dict, key: str):
|
|
276
|
-
"""Extract and clean text content from dictionary.
|
|
290
|
+
"""Extract and clean text content from dictionary.
|
|
291
|
+
|
|
292
|
+
Args:
|
|
293
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
294
|
+
key (str): Key to extract text content from.
|
|
295
|
+
|
|
296
|
+
Returns:
|
|
297
|
+
List[str]: List of non-empty text content.
|
|
298
|
+
"""
|
|
277
299
|
return [text for text in abbr_dict.get(key, []) if text.strip()]
|
|
278
300
|
|
|
279
301
|
def _extract_first_url(self, abbr_dict: dict, key: str):
|
|
280
|
-
"""Extract first URL from a list in dictionary.
|
|
302
|
+
"""Extract first URL from a list in dictionary.
|
|
303
|
+
|
|
304
|
+
Args:
|
|
305
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
306
|
+
key (str): Key to extract URLs from.
|
|
307
|
+
|
|
308
|
+
Returns:
|
|
309
|
+
str: The first valid URL, or empty string if none found.
|
|
310
|
+
"""
|
|
281
311
|
urls = [url.strip() for url in abbr_dict.get(key, []) if url.strip()]
|
|
282
312
|
return urls[0].split(",")[0] if urls else ""
|
|
283
313
|
|
|
284
314
|
def _process_keywords(self, abbr_dict: dict):
|
|
285
|
-
"""Process keywords and convert to Google search URLs.
|
|
315
|
+
"""Process keywords and convert to Google search URLs.
|
|
316
|
+
|
|
317
|
+
Args:
|
|
318
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
319
|
+
|
|
320
|
+
Returns:
|
|
321
|
+
tuple: A tuple containing (keywords, keywords_url) where keywords
|
|
322
|
+
is a sorted list of unique keywords and keywords_url is a list
|
|
323
|
+
of markdown-formatted Google search links.
|
|
324
|
+
"""
|
|
286
325
|
keywords_dict = abbr_dict.get("keywords_dict", {})
|
|
287
326
|
|
|
288
327
|
# Clean and sort keywords
|
|
@@ -315,13 +354,28 @@ class GenerateDataDict(object):
|
|
|
315
354
|
return all_keywords, keywords_url
|
|
316
355
|
|
|
317
356
|
def _format_top_score(self, abbr_dict: dict):
|
|
318
|
-
"""Format top score with optional early access link.
|
|
357
|
+
"""Format top score with optional early access link.
|
|
358
|
+
|
|
359
|
+
Args:
|
|
360
|
+
abbr_dict (dict): Dictionary containing publication data.
|
|
361
|
+
|
|
362
|
+
Returns:
|
|
363
|
+
str: Formatted top score with optional early access link.
|
|
364
|
+
"""
|
|
319
365
|
is_top = "True" if abbr_dict.get("score_top", False) else "False"
|
|
320
366
|
url_early_access = abbr_dict.get("url_early_access", "")
|
|
321
367
|
return self._format_link(is_top, url_early_access)
|
|
322
368
|
|
|
323
369
|
def _format_link(self, text, url):
|
|
324
|
-
"""Format text as markdown link if URL provided.
|
|
370
|
+
"""Format text as markdown link if URL provided.
|
|
371
|
+
|
|
372
|
+
Args:
|
|
373
|
+
text (str): Text to display.
|
|
374
|
+
url (str): URL to link to.
|
|
375
|
+
|
|
376
|
+
Returns:
|
|
377
|
+
str: Markdown-formatted link or plain text if no URL provided.
|
|
378
|
+
"""
|
|
325
379
|
return f"[{text}]({url})" if url else text
|
|
326
380
|
|
|
327
381
|
def _generate_table_row(
|
|
@@ -521,6 +575,19 @@ class GenerateDataDict(object):
|
|
|
521
575
|
|
|
522
576
|
# Mermaid data
|
|
523
577
|
def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
|
|
578
|
+
"""Generate Mermaid diagram data from spidered README files.
|
|
579
|
+
|
|
580
|
+
This method reads spidered data from README files and generates
|
|
581
|
+
Mermaid chart configuration for visualizing publication statistics.
|
|
582
|
+
|
|
583
|
+
Args:
|
|
584
|
+
publisher (str): Publisher name.
|
|
585
|
+
abbr (str): Publication abbreviation.
|
|
586
|
+
inproceedings_or_article (str): Publication type.
|
|
587
|
+
|
|
588
|
+
Returns:
|
|
589
|
+
List[str]: Mermaid chart configuration lines, or empty list if no data found.
|
|
590
|
+
"""
|
|
524
591
|
path_spidered_cj = self.path_spidered_cj if self.path_spidered_cj else ""
|
|
525
592
|
path_readme = os.path.join(path_spidered_cj, publisher, abbr, inproceedings_or_article)
|
|
526
593
|
full_readme = os.path.expanduser(os.path.join(path_readme, "README.md"))
|
|
@@ -1,12 +1,23 @@
|
|
|
1
1
|
# coding=utf-8
|
|
2
2
|
|
|
3
3
|
import os
|
|
4
|
+
import re
|
|
4
5
|
from typing import List
|
|
5
6
|
|
|
6
7
|
from ..core._base import standardize_path
|
|
7
8
|
from .generate_dict import conference_journal_header
|
|
8
9
|
|
|
9
10
|
|
|
11
|
+
def create_safe_filename(text: str) -> str:
|
|
12
|
+
"""Create a safe filename across all platforms."""
|
|
13
|
+
# Remove or replace invalid characters
|
|
14
|
+
safe_text = re.sub(r'[<>:"/\\|?*]', '_', text)
|
|
15
|
+
# Remove leading/trailing spaces and dots
|
|
16
|
+
safe_text = safe_text.strip(' .')
|
|
17
|
+
# Ensure it's not empty
|
|
18
|
+
return safe_text if safe_text else "unnamed"
|
|
19
|
+
|
|
20
|
+
|
|
10
21
|
def conference_journal_informations():
|
|
11
22
|
"""Generate informational content for conferences and journals.
|
|
12
23
|
|
|
@@ -132,6 +143,22 @@ class WriteDataToMd(object):
|
|
|
132
143
|
|
|
133
144
|
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
134
145
|
def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: List[str]):
|
|
146
|
+
"""Get default or customized keywords based on category and provided list.
|
|
147
|
+
|
|
148
|
+
This method returns either a filtered list of keywords based on the provided
|
|
149
|
+
category and keyword list, or all available keywords sorted alphabetically.
|
|
150
|
+
|
|
151
|
+
Args:
|
|
152
|
+
keywords_category_name (str): The category name for keywords filtering.
|
|
153
|
+
keywords_list (List[str]): List of keywords to filter by.
|
|
154
|
+
|
|
155
|
+
Returns:
|
|
156
|
+
List[str]: List of keywords to use for processing.
|
|
157
|
+
|
|
158
|
+
Example:
|
|
159
|
+
>>> writer._default_or_customized_keywords("ai", ["machine learning", "deep learning"])
|
|
160
|
+
["deep learning", "machine learning"]
|
|
161
|
+
"""
|
|
135
162
|
keywords = list(self.keyword_abbr_meta_dict.keys())
|
|
136
163
|
|
|
137
164
|
# Get and sort publication types
|
|
@@ -191,6 +218,18 @@ class WriteDataToMd(object):
|
|
|
191
218
|
return None
|
|
192
219
|
|
|
193
220
|
def save_categories_separate_keywords(self) -> None:
|
|
221
|
+
"""Save publications categorized by keywords in separate files.
|
|
222
|
+
|
|
223
|
+
This method generates individual markdown files for each keyword category,
|
|
224
|
+
creating separate files for better organization and navigation.
|
|
225
|
+
|
|
226
|
+
Returns:
|
|
227
|
+
None: This method does not return a value.
|
|
228
|
+
|
|
229
|
+
Note:
|
|
230
|
+
Each keyword gets its own file saved as '{keyword}.md' in the
|
|
231
|
+
Categories_{type} subdirectory.
|
|
232
|
+
"""
|
|
194
233
|
conference_header, journal_header = conference_journal_header()
|
|
195
234
|
|
|
196
235
|
# Add publications for each category
|
|
@@ -212,7 +251,7 @@ class WriteDataToMd(object):
|
|
|
212
251
|
# Write keyword-specific file
|
|
213
252
|
path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
|
|
214
253
|
# Create safe filename by replacing invalid characters
|
|
215
|
-
safe_keyword =
|
|
254
|
+
safe_keyword = create_safe_filename(keyword).replace(" ", "_")
|
|
216
255
|
with open(os.path.join(path_key, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
|
|
217
256
|
f.writelines(data_list)
|
|
218
257
|
|
|
@@ -264,6 +303,19 @@ class WriteDataToMd(object):
|
|
|
264
303
|
return None
|
|
265
304
|
|
|
266
305
|
def save_publishers_separate_abbrs(self) -> None:
|
|
306
|
+
"""Save detailed publisher information in separate files.
|
|
307
|
+
|
|
308
|
+
This method generates individual markdown files for each publisher,
|
|
309
|
+
containing detailed information about their conferences/journals,
|
|
310
|
+
including about sections, remarks, and statistics.
|
|
311
|
+
|
|
312
|
+
Returns:
|
|
313
|
+
None: This method does not return a value.
|
|
314
|
+
|
|
315
|
+
Note:
|
|
316
|
+
Each publisher gets its own file saved as '{publisher}.md' in the
|
|
317
|
+
Publishers_{type} subdirectory.
|
|
318
|
+
"""
|
|
267
319
|
conference_header, journal_header = conference_journal_header()
|
|
268
320
|
for pub in self.publisher_meta_dict:
|
|
269
321
|
data_list = [f"# {pub}\n\n"]
|
|
@@ -317,6 +369,21 @@ class WriteDataToMd(object):
|
|
|
317
369
|
|
|
318
370
|
# --------- --------- --------- --------- --------- --------- --------- --------- --------- #
|
|
319
371
|
def save_statistics(self, keywords_category_name: str, keywords_list: List[str]) -> None:
|
|
372
|
+
"""Save statistics overview file for keywords.
|
|
373
|
+
|
|
374
|
+
This method generates a markdown file containing statistics overview
|
|
375
|
+
for all keywords with links to their detailed pages.
|
|
376
|
+
|
|
377
|
+
Args:
|
|
378
|
+
keywords_category_name (str): The category name for keywords filtering.
|
|
379
|
+
keywords_list (List[str]): List of keywords to include in the output.
|
|
380
|
+
|
|
381
|
+
Returns:
|
|
382
|
+
None: This method does not return a value.
|
|
383
|
+
|
|
384
|
+
Note:
|
|
385
|
+
The output file is saved as 'Statistics_{type}_{category}.md' in the output directory.
|
|
386
|
+
"""
|
|
320
387
|
data_list = [
|
|
321
388
|
f"# Statistics of keywords in {self.cj.title()}\n\n",
|
|
322
389
|
"| |keywords|Separate Links|\n",
|
|
@@ -327,8 +394,9 @@ class WriteDataToMd(object):
|
|
|
327
394
|
# Add publications for each category
|
|
328
395
|
for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
|
|
329
396
|
# Create safe filename for URL
|
|
330
|
-
safe_keyword =
|
|
331
|
-
|
|
397
|
+
safe_keyword = create_safe_filename(keyword).replace(" ", "_")
|
|
398
|
+
ll = os.path.join("data", self.cj.title(), f"Statistics_{self.cj.title()}", f"{safe_keyword}.md")
|
|
399
|
+
local_url = f"[Link]({ll})"
|
|
332
400
|
|
|
333
401
|
# Create table row
|
|
334
402
|
row = f"| {idx} | {keyword} | {local_url} |\n"
|
|
@@ -344,6 +412,18 @@ class WriteDataToMd(object):
|
|
|
344
412
|
return None
|
|
345
413
|
|
|
346
414
|
def save_statistics_separate_abbrs(self) -> None:
|
|
415
|
+
"""Save detailed statistics for each keyword in separate files.
|
|
416
|
+
|
|
417
|
+
This method generates individual markdown files for each keyword,
|
|
418
|
+
containing detailed statistics and publication information.
|
|
419
|
+
|
|
420
|
+
Returns:
|
|
421
|
+
None: This method does not return a value.
|
|
422
|
+
|
|
423
|
+
Note:
|
|
424
|
+
Each keyword gets its own file saved as '{keyword}.md' in the
|
|
425
|
+
Statistics_{type} subdirectory.
|
|
426
|
+
"""
|
|
347
427
|
conference_header, journal_header = conference_journal_header()
|
|
348
428
|
|
|
349
429
|
# Add publications for each category
|
|
@@ -371,7 +451,7 @@ class WriteDataToMd(object):
|
|
|
371
451
|
# Write publisher-specific file
|
|
372
452
|
path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
|
|
373
453
|
# Create safe filename by replacing invalid characters
|
|
374
|
-
safe_keyword =
|
|
454
|
+
safe_keyword = create_safe_filename(keyword).replace(" ", "_")
|
|
375
455
|
with open(os.path.join(path_pub, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
|
|
376
456
|
f.writelines(data_list)
|
|
377
457
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
[tool.poetry]
|
|
2
2
|
name = "pyformatjson"
|
|
3
|
-
version = "0.0
|
|
3
|
+
version = "0.2.0"
|
|
4
4
|
description = "pyformatjson"
|
|
5
5
|
license = "GPL-3.0-or-later"
|
|
6
6
|
authors = ["NextAI <nextartifintell@gmail.com>"]
|
|
@@ -9,11 +9,11 @@ readme = ["README.md"]
|
|
|
9
9
|
homepage = "https://github.com/NextArtifIntell/pyformatjson"
|
|
10
10
|
repository = "https://github.com/NextArtifIntell/pyformatjson"
|
|
11
11
|
documentation = "https://github.com/NextArtifIntell/pyformatjson"
|
|
12
|
-
keywords = ["Python", "
|
|
12
|
+
keywords = ["Python", "Json"]
|
|
13
13
|
classifiers = ["Topic :: Software Development :: Libraries :: Python Modules"]
|
|
14
14
|
|
|
15
15
|
[tool.poetry.dependencies]
|
|
16
|
-
python = ">=3.
|
|
16
|
+
python = ">=3.12"
|
|
17
17
|
|
|
18
18
|
[tool.poetry.group.dev.dependencies]
|
|
19
19
|
mypy = "^1.18.2"
|
|
@@ -23,6 +23,7 @@ pydocstyle = "^6.3.0"
|
|
|
23
23
|
flake8 = "^7.3.0"
|
|
24
24
|
isort = "^6.0.1"
|
|
25
25
|
black = "^25.9.0"
|
|
26
|
+
pyright = "^1.1.405"
|
|
26
27
|
|
|
27
28
|
|
|
28
29
|
[tool.pyright]
|
|
@@ -1,155 +0,0 @@
|
|
|
1
|
-
# coding=utf-8
|
|
2
|
-
|
|
3
|
-
import json
|
|
4
|
-
import os
|
|
5
|
-
from typing import Optional
|
|
6
|
-
|
|
7
|
-
from .core._base import standardize_path
|
|
8
|
-
from .core.update_json import load_json_data, update_json_file
|
|
9
|
-
from .tools.generate_dict import GenerateDataDict
|
|
10
|
-
from .tools.write_dict import WriteDataToMd
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
def main_generate_md_files(
|
|
14
|
-
path_json: str,
|
|
15
|
-
path_output_md: str,
|
|
16
|
-
path_output_simplified_json: str,
|
|
17
|
-
path_spidered_bibs: Optional[str] = None,
|
|
18
|
-
for_vue: bool = True,
|
|
19
|
-
conferences_or_journals: Optional[str] = None,
|
|
20
|
-
keywords_category_name: str = "",
|
|
21
|
-
) -> None:
|
|
22
|
-
"""Generate markdown files for conferences and journals.
|
|
23
|
-
|
|
24
|
-
This function processes JSON data containing conference and journal information,
|
|
25
|
-
generates various markdown documentation files, and creates simplified JSON outputs.
|
|
26
|
-
It supports both conference and journal processing with customizable keyword categories.
|
|
27
|
-
|
|
28
|
-
Args:
|
|
29
|
-
path_json (str): Path to the input JSON data file containing publication information.
|
|
30
|
-
path_output_md (str): Output directory path where markdown files will be saved.
|
|
31
|
-
path_output_simplified_json (str): Output directory path for simplified JSON files.
|
|
32
|
-
path_spidered_bibs (Optional[str], optional): Directory containing crawled BibTeX files.
|
|
33
|
-
Defaults to None.
|
|
34
|
-
for_vue (bool, optional): Whether to generate Vue.js-compatible format for date calculations.
|
|
35
|
-
Defaults to True.
|
|
36
|
-
conferences_or_journals (Optional[str], optional): Specify 'conferences' or 'journals' to
|
|
37
|
-
process only one type, or None to process both. Defaults to None.
|
|
38
|
-
keywords_category_name (str, optional): The category name for keywords filtering.
|
|
39
|
-
Defaults to "".
|
|
40
|
-
|
|
41
|
-
Returns:
|
|
42
|
-
None: This function does not return a value.
|
|
43
|
-
|
|
44
|
-
Raises:
|
|
45
|
-
FileNotFoundError: If the input JSON file or required directories are not found.
|
|
46
|
-
ValueError: If there are issues with the data format or duplicate abbreviations.
|
|
47
|
-
|
|
48
|
-
Example:
|
|
49
|
-
>>> main_generate_md_files(
|
|
50
|
-
... path_json="/data/publications.json",
|
|
51
|
-
... path_output_md="/output/markdown",
|
|
52
|
-
... path_output_simplified_json="/output/json",
|
|
53
|
-
... for_vue=True,
|
|
54
|
-
... conferences_or_journals="conferences"
|
|
55
|
-
... )
|
|
56
|
-
"""
|
|
57
|
-
# Standardize all paths
|
|
58
|
-
path_json = standardize_path(path_json)
|
|
59
|
-
path_output_md = standardize_path(path_output_md)
|
|
60
|
-
path_output_simplified_json = standardize_path(path_output_simplified_json)
|
|
61
|
-
|
|
62
|
-
path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
|
|
63
|
-
|
|
64
|
-
# Process keyword category name and load data
|
|
65
|
-
keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
|
|
66
|
-
category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
|
|
67
|
-
keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
|
|
68
|
-
|
|
69
|
-
# Validate data availability
|
|
70
|
-
if not keywords_list or not keywords_category_name:
|
|
71
|
-
keywords_list, keywords_category_name = [], ""
|
|
72
|
-
|
|
73
|
-
# Process both conferences and journals
|
|
74
|
-
for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
|
|
75
|
-
# Skip if specific type requested and doesn't match
|
|
76
|
-
if conferences_or_journals and conferences_or_journals.lower() != cj:
|
|
77
|
-
continue
|
|
78
|
-
|
|
79
|
-
# Update JSON data
|
|
80
|
-
json_dict = update_json_file(path_json, cj)
|
|
81
|
-
if not json_dict:
|
|
82
|
-
continue
|
|
83
|
-
|
|
84
|
-
# Simplify JSON data
|
|
85
|
-
simplify_json(json_dict, cj, path_output_simplified_json)
|
|
86
|
-
|
|
87
|
-
# Generate data dictionaries
|
|
88
|
-
path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
|
|
89
|
-
generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
|
|
90
|
-
publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
|
|
91
|
-
if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
|
|
92
|
-
continue
|
|
93
|
-
|
|
94
|
-
# Initialize writer and save all markdown files
|
|
95
|
-
_path_output = os.path.join(path_output_md, f"{cj.title()}")
|
|
96
|
-
save_data = WriteDataToMd(
|
|
97
|
-
cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
|
|
98
|
-
)
|
|
99
|
-
# Save various documentation files
|
|
100
|
-
save_data.save_introductions()
|
|
101
|
-
save_data.save_categories(keywords_category_name, keywords_list)
|
|
102
|
-
save_data.save_categories_separate_keywords()
|
|
103
|
-
|
|
104
|
-
save_data.save_publishers()
|
|
105
|
-
save_data.save_publishers_separate_abbrs()
|
|
106
|
-
|
|
107
|
-
save_data.save_statistics(keywords_category_name, keywords_list)
|
|
108
|
-
save_data.save_statistics_separate_abbrs()
|
|
109
|
-
|
|
110
|
-
return None
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
def simplify_json(json_dict, cj: str, output_dir: str) -> None:
|
|
114
|
-
"""Simplify JSON dictionary by extracting only essential fields.
|
|
115
|
-
|
|
116
|
-
This function creates a simplified version of the JSON dictionary containing
|
|
117
|
-
only the names_abbr and names_full fields for each publisher and publication type.
|
|
118
|
-
The simplified data is saved to a new JSON file in the specified output directory.
|
|
119
|
-
|
|
120
|
-
Args:
|
|
121
|
-
json_dict (dict): The original JSON dictionary containing publication data.
|
|
122
|
-
cj (str): The type of publication, either 'conferences' or 'journals'.
|
|
123
|
-
output_dir (str): Directory path where the simplified JSON file will be saved.
|
|
124
|
-
|
|
125
|
-
Returns:
|
|
126
|
-
None: This function does not return a value.
|
|
127
|
-
|
|
128
|
-
Note:
|
|
129
|
-
The function creates a new JSON file named '{cj}.json' in the output directory.
|
|
130
|
-
Only the 'names_abbr' and 'names_full' fields are preserved in the simplified version.
|
|
131
|
-
|
|
132
|
-
Example:
|
|
133
|
-
>>> simplify_json(
|
|
134
|
-
... json_dict=publication_data,
|
|
135
|
-
... cj="conferences",
|
|
136
|
-
... output_dir="/output/simplified"
|
|
137
|
-
... )
|
|
138
|
-
"""
|
|
139
|
-
new_json_dict = {}
|
|
140
|
-
for publisher in json_dict:
|
|
141
|
-
for abbr in json_dict[publisher][cj.lower()]:
|
|
142
|
-
names_abbr = json_dict[publisher][cj.lower()][abbr].get("names_abbr", [])
|
|
143
|
-
names_full = json_dict[publisher][cj.lower()][abbr].get("names_full", [])
|
|
144
|
-
|
|
145
|
-
new_json_dict.setdefault(publisher, {}).setdefault(cj.lower(), {}).setdefault(abbr, {}).update(
|
|
146
|
-
{"names_abbr": names_abbr, "names_full": names_full}
|
|
147
|
-
)
|
|
148
|
-
|
|
149
|
-
# Save updated JSON
|
|
150
|
-
if new_json_dict:
|
|
151
|
-
path_file = os.path.join(output_dir, f"{cj}.json")
|
|
152
|
-
with open(path_file, "w", encoding="utf-8") as f:
|
|
153
|
-
f.write(json.dumps(new_json_dict, indent=4, sort_keys=True, ensure_ascii=True))
|
|
154
|
-
|
|
155
|
-
return None
|
|
File without changes
|
|
File without changes
|
|
File without changes
|