pyformatjson 0.0.3__tar.gz → 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,16 +1,17 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyformatjson
3
- Version: 0.0.3
3
+ Version: 0.2.0
4
4
  Summary: pyformatjson
5
5
  License: GPL-3.0-or-later
6
- Keywords: Python,json
6
+ Keywords: Python,Json
7
7
  Author: NextAI
8
8
  Author-email: nextartifintell@gmail.com
9
9
  Maintainer: NextAI
10
10
  Maintainer-email: nextartifintell@gmail.com
11
- Requires-Python: >=3.13
11
+ Requires-Python: >=3.12
12
12
  Classifier: License :: OSI Approved :: GNU General Public License v3 or later (GPLv3+)
13
13
  Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.12
14
15
  Classifier: Programming Language :: Python :: 3.13
15
16
  Classifier: Programming Language :: Python :: 3.14
16
17
  Classifier: Topic :: Software Development :: Libraries :: Python Modules
@@ -16,7 +16,7 @@ def load_json_data(path_json: str, filename: str) -> Dict:
16
16
 
17
17
  Args:
18
18
  path_json (str): Directory path containing the JSON file.
19
- filename (str): Name of the JSON file (without .json extension).
19
+ filename (str): Name of the JSON file (with .json extension).
20
20
 
21
21
  Returns:
22
22
  Dict: The loaded JSON data as a dictionary, or empty dict if file not found
@@ -27,7 +27,7 @@ def load_json_data(path_json: str, filename: str) -> Dict:
27
27
  {"publisher1": {"conferences": {...}}}
28
28
  """
29
29
  try:
30
- file_path = os.path.join(path_json, f"{filename}.json")
30
+ file_path = os.path.join(path_json, filename)
31
31
  if not os.path.exists(file_path):
32
32
  return {}
33
33
 
@@ -35,11 +35,11 @@ def load_json_data(path_json: str, filename: str) -> Dict:
35
35
  return json.load(file)
36
36
 
37
37
  except Exception as e:
38
- print(f"Error loading {filename}.json: {e}")
38
+ print(f"Error loading {filename}: {e}")
39
39
  return {}
40
40
 
41
41
 
42
- def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str, Any]:
42
+ def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
43
43
  """Update and format JSON file containing conference/journal data.
44
44
 
45
45
  This function loads JSON data, processes and formats text fields by splitting
@@ -47,21 +47,14 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
47
47
  saves the updated data back to the file.
48
48
 
49
49
  Args:
50
- path_root (str): Root directory path containing the JSON file.
50
+ full_json_cj (str): Full path to the conferences/journals JSON file
51
51
  conferences_or_journals (str): Type of publication ('conferences' or 'journals').
52
52
 
53
53
  Returns:
54
54
  Dict[str, Any]: Processed JSON data dictionary.
55
-
56
- Raises:
57
- ValueError: If duplicate abbreviations are found in the data.
58
-
59
- Example:
60
- >>> update_json_file("/data", "conferences")
61
- {"publisher1": {"conferences": {"conf1": {...}}}}
62
55
  """
63
56
  # Load Json Data
64
- json_dict = load_json_data(path_root, conferences_or_journals)
57
+ json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
65
58
 
66
59
  # Process and format text fields in JSON data.
67
60
  for pub in json_dict:
@@ -87,8 +80,7 @@ def update_json_file(path_root: str, conferences_or_journals: str) -> Dict[str,
87
80
 
88
81
  # Save updated JSON
89
82
  if json_dict:
90
- path_file = os.path.join(path_root, f"{conferences_or_journals}.json")
91
- with open(path_file, "w", encoding="utf-8") as f:
83
+ with open(full_json_cj, "w", encoding="utf-8") as f:
92
84
  f.write(json.dumps(json_dict, indent=4, sort_keys=True, ensure_ascii=True))
93
85
 
94
86
  return json_dict
@@ -0,0 +1,106 @@
1
+ # coding=utf-8
2
+
3
+ import os
4
+ from typing import Optional
5
+
6
+ from .core._base import standardize_path
7
+ from .core.update_json import load_json_data, update_json_file
8
+ from .tools.generate_dict import GenerateDataDict
9
+ from .tools.write_dict import WriteDataToMd
10
+
11
+
12
+ def main_generate_md_files(
13
+ full_json_c: str,
14
+ full_json_j: str,
15
+ full_json_k: str,
16
+ path_output: str,
17
+ path_spidered_bibs: Optional[str] = None,
18
+ keywords_category_name: str = "",
19
+ for_vue: bool = True,
20
+ ) -> None:
21
+ """Generate comprehensive markdown documentation for academic publications.
22
+
23
+ This function serves as the main entry point for processing conference and journal
24
+ data from JSON files and generating various markdown documentation outputs. It
25
+ handles both conference and journal data processing, creates categorized outputs,
26
+ and generates statistics and publisher information files.
27
+
28
+ The function processes three types of JSON files:
29
+ - Conference data (full_json_c)
30
+ - Journal data (full_json_j)
31
+ - Keywords data (full_json_k)
32
+
33
+ Args:
34
+ full_json_c (str): Full path to the conferences JSON file containing
35
+ conference publication data.
36
+ full_json_j (str): Full path to the journals JSON file containing
37
+ journal publication data.
38
+ full_json_k (str): Full path to the keywords JSON file containing
39
+ keyword categorization data.
40
+ path_output (str): Output directory path where all generated markdown
41
+ files will be saved.
42
+ path_spidered_bibs (Optional[str], optional): Directory path containing
43
+ spidered BibTeX files for additional data processing. Defaults to None.
44
+ keywords_category_name (str, optional): Category name for filtering
45
+ keywords. If provided, only keywords from this category will be
46
+ processed. Defaults to "".
47
+ for_vue (bool, optional): Whether to generate Vue.js-compatible format
48
+ for date calculations and dynamic content. Defaults to True.
49
+
50
+ Returns:
51
+ None: This function does not return a value.
52
+ """
53
+ # Standardize all paths
54
+ full_json_c = os.path.expanduser(full_json_c)
55
+ full_json_j = os.path.expanduser(full_json_j)
56
+ full_json_k = os.path.expanduser(full_json_k)
57
+
58
+ path_output = standardize_path(path_output)
59
+
60
+ path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
61
+
62
+ # Process keyword category name and load data
63
+ keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
64
+ category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
65
+ keywords_json = load_json_data(os.path.dirname(full_json_k), os.path.basename(full_json_k))
66
+ keywords_list = keywords_json.get(f"{category_prefix}keywords", [])
67
+
68
+ # Validate data availability
69
+ if not keywords_list or not keywords_category_name:
70
+ keywords_list, keywords_category_name = [], ""
71
+
72
+ # Process both conferences and journals
73
+ for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
74
+ # Update JSON data
75
+ json_dict = {}
76
+ if cj == "conferences":
77
+ json_dict = update_json_file(full_json_c, cj)
78
+ elif cj == "journals":
79
+ json_dict = update_json_file(full_json_j, cj)
80
+ if not json_dict:
81
+ continue
82
+
83
+ # Generate data dictionaries
84
+ path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
85
+ generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
86
+ publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
87
+ if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
88
+ continue
89
+
90
+ # Initialize writer and save all markdown files
91
+ _path_output = os.path.join(path_output, f"{cj.title()}")
92
+ save_data = WriteDataToMd(
93
+ cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
94
+ )
95
+ # Save various documentation files
96
+ save_data.save_introductions()
97
+ save_data.save_categories(keywords_category_name, keywords_list)
98
+ save_data.save_categories_separate_keywords()
99
+
100
+ save_data.save_publishers()
101
+ save_data.save_publishers_separate_abbrs()
102
+
103
+ save_data.save_statistics(keywords_category_name, keywords_list)
104
+ save_data.save_statistics_separate_abbrs()
105
+
106
+ return None
@@ -252,12 +252,26 @@ class GenerateDataDict(object):
252
252
  return full_name, abbr_name
253
253
 
254
254
  def _extract_homepage_url(self, abbr_dict: dict):
255
- """Extract and clean homepage URL."""
255
+ """Extract and clean homepage URL.
256
+
257
+ Args:
258
+ abbr_dict (dict): Dictionary containing publication data.
259
+
260
+ Returns:
261
+ str: The first valid homepage URL, or empty string if none found.
262
+ """
256
263
  urls = [u.strip() for u in abbr_dict.get("urls_homepage", []) if u.strip()]
257
264
  return urls[0] if urls else ""
258
265
 
259
266
  def _format_period_with_dblp(self, abbr_dict: dict):
260
- """Format publication period with DBLP link if available."""
267
+ """Format publication period with DBLP link if available.
268
+
269
+ Args:
270
+ abbr_dict (dict): Dictionary containing publication data.
271
+
272
+ Returns:
273
+ str: Formatted period string with optional DBLP link.
274
+ """
261
275
  start_year = abbr_dict.get("year_start", "")
262
276
  end_year = abbr_dict.get("year_end", "")
263
277
 
@@ -273,16 +287,41 @@ class GenerateDataDict(object):
273
287
  return period
274
288
 
275
289
  def _extract_text_content(self, abbr_dict: dict, key: str):
276
- """Extract and clean text content from dictionary."""
290
+ """Extract and clean text content from dictionary.
291
+
292
+ Args:
293
+ abbr_dict (dict): Dictionary containing publication data.
294
+ key (str): Key to extract text content from.
295
+
296
+ Returns:
297
+ List[str]: List of non-empty text content.
298
+ """
277
299
  return [text for text in abbr_dict.get(key, []) if text.strip()]
278
300
 
279
301
  def _extract_first_url(self, abbr_dict: dict, key: str):
280
- """Extract first URL from a list in dictionary."""
302
+ """Extract first URL from a list in dictionary.
303
+
304
+ Args:
305
+ abbr_dict (dict): Dictionary containing publication data.
306
+ key (str): Key to extract URLs from.
307
+
308
+ Returns:
309
+ str: The first valid URL, or empty string if none found.
310
+ """
281
311
  urls = [url.strip() for url in abbr_dict.get(key, []) if url.strip()]
282
312
  return urls[0].split(",")[0] if urls else ""
283
313
 
284
314
  def _process_keywords(self, abbr_dict: dict):
285
- """Process keywords and convert to Google search URLs."""
315
+ """Process keywords and convert to Google search URLs.
316
+
317
+ Args:
318
+ abbr_dict (dict): Dictionary containing publication data.
319
+
320
+ Returns:
321
+ tuple: A tuple containing (keywords, keywords_url) where keywords
322
+ is a sorted list of unique keywords and keywords_url is a list
323
+ of markdown-formatted Google search links.
324
+ """
286
325
  keywords_dict = abbr_dict.get("keywords_dict", {})
287
326
 
288
327
  # Clean and sort keywords
@@ -315,13 +354,28 @@ class GenerateDataDict(object):
315
354
  return all_keywords, keywords_url
316
355
 
317
356
  def _format_top_score(self, abbr_dict: dict):
318
- """Format top score with optional early access link."""
357
+ """Format top score with optional early access link.
358
+
359
+ Args:
360
+ abbr_dict (dict): Dictionary containing publication data.
361
+
362
+ Returns:
363
+ str: Formatted top score with optional early access link.
364
+ """
319
365
  is_top = "True" if abbr_dict.get("score_top", False) else "False"
320
366
  url_early_access = abbr_dict.get("url_early_access", "")
321
367
  return self._format_link(is_top, url_early_access)
322
368
 
323
369
  def _format_link(self, text, url):
324
- """Format text as markdown link if URL provided."""
370
+ """Format text as markdown link if URL provided.
371
+
372
+ Args:
373
+ text (str): Text to display.
374
+ url (str): URL to link to.
375
+
376
+ Returns:
377
+ str: Markdown-formatted link or plain text if no URL provided.
378
+ """
325
379
  return f"[{text}]({url})" if url else text
326
380
 
327
381
  def _generate_table_row(
@@ -521,6 +575,19 @@ class GenerateDataDict(object):
521
575
 
522
576
  # Mermaid data
523
577
  def generate_mermaid_data(self, publisher: str, abbr: str, inproceedings_or_article: str):
578
+ """Generate Mermaid diagram data from spidered README files.
579
+
580
+ This method reads spidered data from README files and generates
581
+ Mermaid chart configuration for visualizing publication statistics.
582
+
583
+ Args:
584
+ publisher (str): Publisher name.
585
+ abbr (str): Publication abbreviation.
586
+ inproceedings_or_article (str): Publication type.
587
+
588
+ Returns:
589
+ List[str]: Mermaid chart configuration lines, or empty list if no data found.
590
+ """
524
591
  path_spidered_cj = self.path_spidered_cj if self.path_spidered_cj else ""
525
592
  path_readme = os.path.join(path_spidered_cj, publisher, abbr, inproceedings_or_article)
526
593
  full_readme = os.path.expanduser(os.path.join(path_readme, "README.md"))
@@ -1,12 +1,23 @@
1
1
  # coding=utf-8
2
2
 
3
3
  import os
4
+ import re
4
5
  from typing import List
5
6
 
6
7
  from ..core._base import standardize_path
7
8
  from .generate_dict import conference_journal_header
8
9
 
9
10
 
11
+ def create_safe_filename(text: str) -> str:
12
+ """Create a safe filename across all platforms."""
13
+ # Remove or replace invalid characters
14
+ safe_text = re.sub(r'[<>:"/\\|?*]', '_', text)
15
+ # Remove leading/trailing spaces and dots
16
+ safe_text = safe_text.strip(' .')
17
+ # Ensure it's not empty
18
+ return safe_text if safe_text else "unnamed"
19
+
20
+
10
21
  def conference_journal_informations():
11
22
  """Generate informational content for conferences and journals.
12
23
 
@@ -132,6 +143,22 @@ class WriteDataToMd(object):
132
143
 
133
144
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
134
145
  def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: List[str]):
146
+ """Get default or customized keywords based on category and provided list.
147
+
148
+ This method returns either a filtered list of keywords based on the provided
149
+ category and keyword list, or all available keywords sorted alphabetically.
150
+
151
+ Args:
152
+ keywords_category_name (str): The category name for keywords filtering.
153
+ keywords_list (List[str]): List of keywords to filter by.
154
+
155
+ Returns:
156
+ List[str]: List of keywords to use for processing.
157
+
158
+ Example:
159
+ >>> writer._default_or_customized_keywords("ai", ["machine learning", "deep learning"])
160
+ ["deep learning", "machine learning"]
161
+ """
135
162
  keywords = list(self.keyword_abbr_meta_dict.keys())
136
163
 
137
164
  # Get and sort publication types
@@ -191,6 +218,18 @@ class WriteDataToMd(object):
191
218
  return None
192
219
 
193
220
  def save_categories_separate_keywords(self) -> None:
221
+ """Save publications categorized by keywords in separate files.
222
+
223
+ This method generates individual markdown files for each keyword category,
224
+ creating separate files for better organization and navigation.
225
+
226
+ Returns:
227
+ None: This method does not return a value.
228
+
229
+ Note:
230
+ Each keyword gets its own file saved as '{keyword}.md' in the
231
+ Categories_{type} subdirectory.
232
+ """
194
233
  conference_header, journal_header = conference_journal_header()
195
234
 
196
235
  # Add publications for each category
@@ -212,7 +251,7 @@ class WriteDataToMd(object):
212
251
  # Write keyword-specific file
213
252
  path_key = standardize_path(os.path.join(self.path_output, f"Categories_{self.cj.title()}"))
214
253
  # Create safe filename by replacing invalid characters
215
- safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
254
+ safe_keyword = create_safe_filename(keyword).replace(" ", "_")
216
255
  with open(os.path.join(path_key, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
217
256
  f.writelines(data_list)
218
257
 
@@ -264,6 +303,19 @@ class WriteDataToMd(object):
264
303
  return None
265
304
 
266
305
  def save_publishers_separate_abbrs(self) -> None:
306
+ """Save detailed publisher information in separate files.
307
+
308
+ This method generates individual markdown files for each publisher,
309
+ containing detailed information about their conferences/journals,
310
+ including about sections, remarks, and statistics.
311
+
312
+ Returns:
313
+ None: This method does not return a value.
314
+
315
+ Note:
316
+ Each publisher gets its own file saved as '{publisher}.md' in the
317
+ Publishers_{type} subdirectory.
318
+ """
267
319
  conference_header, journal_header = conference_journal_header()
268
320
  for pub in self.publisher_meta_dict:
269
321
  data_list = [f"# {pub}\n\n"]
@@ -317,6 +369,21 @@ class WriteDataToMd(object):
317
369
 
318
370
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
319
371
  def save_statistics(self, keywords_category_name: str, keywords_list: List[str]) -> None:
372
+ """Save statistics overview file for keywords.
373
+
374
+ This method generates a markdown file containing statistics overview
375
+ for all keywords with links to their detailed pages.
376
+
377
+ Args:
378
+ keywords_category_name (str): The category name for keywords filtering.
379
+ keywords_list (List[str]): List of keywords to include in the output.
380
+
381
+ Returns:
382
+ None: This method does not return a value.
383
+
384
+ Note:
385
+ The output file is saved as 'Statistics_{type}_{category}.md' in the output directory.
386
+ """
320
387
  data_list = [
321
388
  f"# Statistics of keywords in {self.cj.title()}\n\n",
322
389
  "| |keywords|Separate Links|\n",
@@ -327,8 +394,9 @@ class WriteDataToMd(object):
327
394
  # Add publications for each category
328
395
  for keyword in self._default_or_customized_keywords(keywords_category_name, keywords_list):
329
396
  # Create safe filename for URL
330
- safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
331
- local_url = f"[Link](data/{self.cj.title()}/Statistics_{self.cj.title()}/{safe_keyword}.md)"
397
+ safe_keyword = create_safe_filename(keyword).replace(" ", "_")
398
+ ll = os.path.join("data", self.cj.title(), f"Statistics_{self.cj.title()}", f"{safe_keyword}.md")
399
+ local_url = f"[Link]({ll})"
332
400
 
333
401
  # Create table row
334
402
  row = f"| {idx} | {keyword} | {local_url} |\n"
@@ -344,6 +412,18 @@ class WriteDataToMd(object):
344
412
  return None
345
413
 
346
414
  def save_statistics_separate_abbrs(self) -> None:
415
+ """Save detailed statistics for each keyword in separate files.
416
+
417
+ This method generates individual markdown files for each keyword,
418
+ containing detailed statistics and publication information.
419
+
420
+ Returns:
421
+ None: This method does not return a value.
422
+
423
+ Note:
424
+ Each keyword gets its own file saved as '{keyword}.md' in the
425
+ Statistics_{type} subdirectory.
426
+ """
347
427
  conference_header, journal_header = conference_journal_header()
348
428
 
349
429
  # Add publications for each category
@@ -371,7 +451,7 @@ class WriteDataToMd(object):
371
451
  # Write publisher-specific file
372
452
  path_pub = standardize_path(os.path.join(self.path_output, f"Statistics_{self.cj.title()}"))
373
453
  # Create safe filename by replacing invalid characters
374
- safe_keyword = "".join(c if c.isalnum() or c in "-_" else "_" for c in keyword)
454
+ safe_keyword = create_safe_filename(keyword).replace(" ", "_")
375
455
  with open(os.path.join(path_pub, f"{safe_keyword}.md"), "w", encoding="utf-8") as f:
376
456
  f.writelines(data_list)
377
457
 
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "pyformatjson"
3
- version = "0.0.3"
3
+ version = "0.2.0"
4
4
  description = "pyformatjson"
5
5
  license = "GPL-3.0-or-later"
6
6
  authors = ["NextAI <nextartifintell@gmail.com>"]
@@ -9,11 +9,11 @@ readme = ["README.md"]
9
9
  homepage = "https://github.com/NextArtifIntell/pyformatjson"
10
10
  repository = "https://github.com/NextArtifIntell/pyformatjson"
11
11
  documentation = "https://github.com/NextArtifIntell/pyformatjson"
12
- keywords = ["Python", "json"]
12
+ keywords = ["Python", "Json"]
13
13
  classifiers = ["Topic :: Software Development :: Libraries :: Python Modules"]
14
14
 
15
15
  [tool.poetry.dependencies]
16
- python = ">=3.13"
16
+ python = ">=3.12"
17
17
 
18
18
  [tool.poetry.group.dev.dependencies]
19
19
  mypy = "^1.18.2"
@@ -23,6 +23,7 @@ pydocstyle = "^6.3.0"
23
23
  flake8 = "^7.3.0"
24
24
  isort = "^6.0.1"
25
25
  black = "^25.9.0"
26
+ pyright = "^1.1.405"
26
27
 
27
28
 
28
29
  [tool.pyright]
@@ -1,155 +0,0 @@
1
- # coding=utf-8
2
-
3
- import json
4
- import os
5
- from typing import Optional
6
-
7
- from .core._base import standardize_path
8
- from .core.update_json import load_json_data, update_json_file
9
- from .tools.generate_dict import GenerateDataDict
10
- from .tools.write_dict import WriteDataToMd
11
-
12
-
13
- def main_generate_md_files(
14
- path_json: str,
15
- path_output_md: str,
16
- path_output_simplified_json: str,
17
- path_spidered_bibs: Optional[str] = None,
18
- for_vue: bool = True,
19
- conferences_or_journals: Optional[str] = None,
20
- keywords_category_name: str = "",
21
- ) -> None:
22
- """Generate markdown files for conferences and journals.
23
-
24
- This function processes JSON data containing conference and journal information,
25
- generates various markdown documentation files, and creates simplified JSON outputs.
26
- It supports both conference and journal processing with customizable keyword categories.
27
-
28
- Args:
29
- path_json (str): Path to the input JSON data file containing publication information.
30
- path_output_md (str): Output directory path where markdown files will be saved.
31
- path_output_simplified_json (str): Output directory path for simplified JSON files.
32
- path_spidered_bibs (Optional[str], optional): Directory containing crawled BibTeX files.
33
- Defaults to None.
34
- for_vue (bool, optional): Whether to generate Vue.js-compatible format for date calculations.
35
- Defaults to True.
36
- conferences_or_journals (Optional[str], optional): Specify 'conferences' or 'journals' to
37
- process only one type, or None to process both. Defaults to None.
38
- keywords_category_name (str, optional): The category name for keywords filtering.
39
- Defaults to "".
40
-
41
- Returns:
42
- None: This function does not return a value.
43
-
44
- Raises:
45
- FileNotFoundError: If the input JSON file or required directories are not found.
46
- ValueError: If there are issues with the data format or duplicate abbreviations.
47
-
48
- Example:
49
- >>> main_generate_md_files(
50
- ... path_json="/data/publications.json",
51
- ... path_output_md="/output/markdown",
52
- ... path_output_simplified_json="/output/json",
53
- ... for_vue=True,
54
- ... conferences_or_journals="conferences"
55
- ... )
56
- """
57
- # Standardize all paths
58
- path_json = standardize_path(path_json)
59
- path_output_md = standardize_path(path_output_md)
60
- path_output_simplified_json = standardize_path(path_output_simplified_json)
61
-
62
- path_spidered_bibs = standardize_path(path_spidered_bibs) if path_spidered_bibs else ""
63
-
64
- # Process keyword category name and load data
65
- keywords_category_name = keywords_category_name.lower().strip() if keywords_category_name else ""
66
- category_prefix = f"{keywords_category_name}_" if keywords_category_name else ""
67
- keywords_list = load_json_data(path_json, "keywords").get(f"{category_prefix}keywords", [])
68
-
69
- # Validate data availability
70
- if not keywords_list or not keywords_category_name:
71
- keywords_list, keywords_category_name = [], ""
72
-
73
- # Process both conferences and journals
74
- for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
75
- # Skip if specific type requested and doesn't match
76
- if conferences_or_journals and conferences_or_journals.lower() != cj:
77
- continue
78
-
79
- # Update JSON data
80
- json_dict = update_json_file(path_json, cj)
81
- if not json_dict:
82
- continue
83
-
84
- # Simplify JSON data
85
- simplify_json(json_dict, cj, path_output_simplified_json)
86
-
87
- # Generate data dictionaries
88
- path_spidered_cj = os.path.join(path_spidered_bibs, cj.title())
89
- generater = GenerateDataDict(cj, ia, json_dict, for_vue, path_spidered_cj)
90
- publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict = generater.generate()
91
- if not (publisher_meta_dict and publisher_abbr_meta_dict and keyword_abbr_meta_dict):
92
- continue
93
-
94
- # Initialize writer and save all markdown files
95
- _path_output = os.path.join(path_output_md, f"{cj.title()}")
96
- save_data = WriteDataToMd(
97
- cj, ia, publisher_meta_dict, publisher_abbr_meta_dict, keyword_abbr_meta_dict, _path_output
98
- )
99
- # Save various documentation files
100
- save_data.save_introductions()
101
- save_data.save_categories(keywords_category_name, keywords_list)
102
- save_data.save_categories_separate_keywords()
103
-
104
- save_data.save_publishers()
105
- save_data.save_publishers_separate_abbrs()
106
-
107
- save_data.save_statistics(keywords_category_name, keywords_list)
108
- save_data.save_statistics_separate_abbrs()
109
-
110
- return None
111
-
112
-
113
- def simplify_json(json_dict, cj: str, output_dir: str) -> None:
114
- """Simplify JSON dictionary by extracting only essential fields.
115
-
116
- This function creates a simplified version of the JSON dictionary containing
117
- only the names_abbr and names_full fields for each publisher and publication type.
118
- The simplified data is saved to a new JSON file in the specified output directory.
119
-
120
- Args:
121
- json_dict (dict): The original JSON dictionary containing publication data.
122
- cj (str): The type of publication, either 'conferences' or 'journals'.
123
- output_dir (str): Directory path where the simplified JSON file will be saved.
124
-
125
- Returns:
126
- None: This function does not return a value.
127
-
128
- Note:
129
- The function creates a new JSON file named '{cj}.json' in the output directory.
130
- Only the 'names_abbr' and 'names_full' fields are preserved in the simplified version.
131
-
132
- Example:
133
- >>> simplify_json(
134
- ... json_dict=publication_data,
135
- ... cj="conferences",
136
- ... output_dir="/output/simplified"
137
- ... )
138
- """
139
- new_json_dict = {}
140
- for publisher in json_dict:
141
- for abbr in json_dict[publisher][cj.lower()]:
142
- names_abbr = json_dict[publisher][cj.lower()][abbr].get("names_abbr", [])
143
- names_full = json_dict[publisher][cj.lower()][abbr].get("names_full", [])
144
-
145
- new_json_dict.setdefault(publisher, {}).setdefault(cj.lower(), {}).setdefault(abbr, {}).update(
146
- {"names_abbr": names_abbr, "names_full": names_full}
147
- )
148
-
149
- # Save updated JSON
150
- if new_json_dict:
151
- path_file = os.path.join(output_dir, f"{cj}.json")
152
- with open(path_file, "w", encoding="utf-8") as f:
153
- f.write(json.dumps(new_json_dict, indent=4, sort_keys=True, ensure_ascii=True))
154
-
155
- return None
File without changes