pyformatjson 0.2.4__tar.gz → 0.2.5__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: pyformatjson
3
- Version: 0.2.4
3
+ Version: 0.2.5
4
4
  Summary: pyformatjson
5
5
  License: GPL-3.0-or-later
6
6
  License-File: LICENSE
File without changes
@@ -1,11 +1,8 @@
1
- # coding=utf-8
2
-
3
1
  import os
4
2
  import re
5
- from typing import List
6
3
 
7
4
 
8
- def split_text_by_length(text, max_length=120) -> List[str]:
5
+ def split_text_by_length(text, max_length=120) -> list[str]:
9
6
  """Split text into lines of specified maximum length.
10
7
 
11
8
  This function breaks long text into multiple lines, ensuring each line
@@ -17,7 +14,7 @@ def split_text_by_length(text, max_length=120) -> List[str]:
17
14
  max_length (int, optional): Maximum length for each line. Defaults to 120.
18
15
 
19
16
  Returns:
20
- List[str]: A list of text lines, each not exceeding max_length characters.
17
+ list[str]: A list of text lines, each not exceeding max_length characters.
21
18
 
22
19
  Example:
23
20
  >>> split_text_by_length("This is a very long text that needs to be split", 20)
@@ -44,8 +41,8 @@ def split_text_by_length(text, max_length=120) -> List[str]:
44
41
  return new_lines
45
42
 
46
43
 
47
- def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "next") -> List[str]:
48
- """Split data list according to the split pattern.
44
+ def split_data_list(split_pattern: str, data_list: list[str], last_next: str = "next") -> list[str]:
45
+ r"""Split data list according to the split pattern.
49
46
 
50
47
  This function splits each string in the data list using the provided regex pattern
51
48
  and reconstructs the data based on the last_next parameter. The pattern must use
@@ -54,13 +51,13 @@ def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "
54
51
  Args:
55
52
  split_pattern (str): Regular expression pattern for splitting. Must use capturing
56
53
  parentheses, e.g., r"(\n)" for newline splits.
57
- data_list (List[str]): List of strings to be split and processed.
54
+ data_list (list[str]): List of strings to be split and processed.
58
55
  last_next (str, optional): Determines how to handle split parts. "next" places
59
56
  the split character at the beginning of the next part, "last" places it at
60
57
  the end of the current part. Defaults to "next".
61
58
 
62
59
  Returns:
63
- List[str]: New list of processed strings with empty strings filtered out.
60
+ list[str]: New list of processed strings with empty strings filtered out.
64
61
 
65
62
  Raises:
66
63
  re.error: If the split_pattern is not a valid regular expression.
@@ -72,8 +69,8 @@ def split_data_list(split_pattern: str, data_list: List[str], last_next: str = "
72
69
  new_data_list = []
73
70
  for line in data_list:
74
71
  split_list = re.split(split_pattern, line)
75
- list_one = split_list[0 : len(split_list) : 2]
76
- list_two = split_list[1 : len(split_list) : 2]
72
+ list_one = split_list[0: len(split_list): 2]
73
+ list_two = split_list[1: len(split_list): 2]
77
74
 
78
75
  temp = []
79
76
  if last_next == "next":
@@ -109,7 +106,7 @@ def standardize_path(path_input: str) -> str:
109
106
  return path_input
110
107
 
111
108
 
112
- def sort_strings_with_embedded_numbers(s: str) -> List[str]:
109
+ def sort_strings_with_embedded_numbers(s: str) -> list[str]:
113
110
  """Split string into pieces for natural sorting with embedded numbers.
114
111
 
115
112
  This function splits a string into pieces where numbers are converted to integers
@@ -119,7 +116,7 @@ def sort_strings_with_embedded_numbers(s: str) -> List[str]:
119
116
  s (str): The string to be split into sortable pieces.
120
117
 
121
118
  Returns:
122
- List[str]: List of string pieces with numbers converted to integers.
119
+ list[str]: List of string pieces with numbers converted to integers.
123
120
 
124
121
  Example:
125
122
  >>> sort_strings_with_embedded_numbers("item10")
@@ -131,18 +128,18 @@ def sort_strings_with_embedded_numbers(s: str) -> List[str]:
131
128
  return pieces
132
129
 
133
130
 
134
- def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
131
+ def sort_int_str(str_int: list[str], reverse: bool = False) -> list[str]:
135
132
  """Sort list of strings with embedded numbers naturally.
136
133
 
137
134
  This function sorts a list of strings using natural sorting that handles
138
135
  embedded numbers correctly (e.g., "item2" comes before "item10").
139
136
 
140
137
  Args:
141
- str_int (List[str]): List of strings to be sorted.
138
+ str_int (list[str]): List of strings to be sorted.
142
139
  reverse (bool, optional): If True, sorts in descending order. Defaults to False.
143
140
 
144
141
  Returns:
145
- List[str]: Sorted list of strings.
142
+ list[str]: Sorted list of strings.
146
143
 
147
144
  Example:
148
145
  >>> sort_int_str(["item10", "item2", "item1"])
@@ -151,7 +148,7 @@ def sort_int_str(str_int: List[str], reverse: bool = False) -> List[str]:
151
148
  return sorted(str_int, key=sort_strings_with_embedded_numbers, reverse=reverse)
152
149
 
153
150
 
154
- class IterateSortDict(object):
151
+ class IterateSortDict:
155
152
  """A class for recursively sorting dictionary keys with natural sorting.
156
153
 
157
154
  This class provides methods to sort dictionary keys recursively, handling
@@ -1,20 +1,18 @@
1
- # coding=utf-8
2
-
3
1
  import json
4
2
  import os
5
3
  import re
6
- from typing import Any, Dict, List, Tuple
4
+ from typing import Any
7
5
 
8
6
  from ._base import split_data_list, split_text_by_length
9
7
 
10
8
 
11
- def load_json_data(path_json: str, filename: str) -> Dict:
9
+ def load_json_data(path_json: str, filename: str) -> dict:
12
10
  try:
13
11
  file_path = os.path.join(path_json, filename)
14
12
  if not os.path.exists(file_path):
15
13
  return {}
16
14
 
17
- with open(file_path, "r", encoding="utf-8") as file:
15
+ with open(file_path, encoding="utf-8") as file:
18
16
  return json.load(file)
19
17
 
20
18
  except Exception as e:
@@ -22,7 +20,7 @@ def load_json_data(path_json: str, filename: str) -> Dict:
22
20
  return {}
23
21
 
24
22
 
25
- def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[str, Any]:
23
+ def update_json_file(full_json_cj: str, conferences_or_journals: str) -> dict[str, Any]:
26
24
  """Update and format JSON file containing conference/journal data.
27
25
 
28
26
  This function loads JSON data, processes and formats text fields by splitting
@@ -34,7 +32,7 @@ def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[st
34
32
  conferences_or_journals (str): Type of publication ('conferences' or 'journals').
35
33
 
36
34
  Returns:
37
- Dict[str, Any]: Processed JSON data dictionary.
35
+ dict[str, Any]: Processed JSON data dictionary.
38
36
  """
39
37
  # Load Json Data
40
38
  json_dict = load_json_data(os.path.dirname(full_json_cj), os.path.basename(full_json_cj))
@@ -71,7 +69,7 @@ def update_json_file(full_json_cj: str, conferences_or_journals: str) -> Dict[st
71
69
  return json_dict
72
70
 
73
71
 
74
- def generate_standard_form(json_dict: Dict[str, Any], conferences_or_journals: str):
72
+ def generate_standard_form(json_dict: dict[str, Any], conferences_or_journals: str):
75
73
  # Check for duplicate abbreviations in the data.
76
74
  abbr_list = []
77
75
  for pub in json_dict:
@@ -82,7 +80,7 @@ def generate_standard_form(json_dict: Dict[str, Any], conferences_or_journals: s
82
80
  abbr_list.append(abbr)
83
81
 
84
82
  # Extract abbreviation and name data from all publications
85
- abbr_dict: Dict[str, Dict[str, List[str]]] = {}
83
+ abbr_dict: dict[str, dict[str, list[str]]] = {}
86
84
  for pub in json_dict:
87
85
  if conferences_or_journals in json_dict[pub]:
88
86
  for abbr, v in json_dict[pub][conferences_or_journals].items():
@@ -97,7 +95,7 @@ class CheckAcronymAbbrAndFullDict:
97
95
  self.names_abbr = names_abbr
98
96
  self.names_full = names_full
99
97
 
100
- def run(self, dict_data: dict[str, dict[str, list[str]]]) -> Tuple[dict[str, dict[str, list[str]]], bool]:
98
+ def run(self, dict_data: dict[str, dict[str, list[str]]]) -> tuple[dict[str, dict[str, list[str]]], bool]:
101
99
  # Check if each acronym has equal number of abbreviations and full forms
102
100
  dict_data, length_check = self._validate_lengths(dict_data)
103
101
 
@@ -137,7 +135,7 @@ class CheckAcronymAbbrAndFullDict:
137
135
  has_duplicate = False
138
136
 
139
137
  # Check for duplicate abbreviations
140
- abbrs_lower = set([abbr.lower() for abbr in values.get(self.names_abbr, [])])
138
+ abbrs_lower = {abbr.lower() for abbr in values.get(self.names_abbr, [])}
141
139
  for abbr in abbrs_lower:
142
140
  if abbr in seen_abbrs:
143
141
  print(f"Duplicate abbreviation '{abbr}' found in '{acronym}'")
@@ -146,7 +144,7 @@ class CheckAcronymAbbrAndFullDict:
146
144
  seen_abbrs.add(abbr)
147
145
 
148
146
  # Check for duplicate full forms
149
- fulls_lower = set([full.lower() for full in values.get(self.names_full, [])])
147
+ fulls_lower = {full.lower() for full in values.get(self.names_full, [])}
150
148
  for full in fulls_lower:
151
149
  if full in seen_fulls:
152
150
  print(f"Duplicate full form '{full}' found in '{acronym}'")
@@ -165,40 +163,41 @@ class CheckAcronymAbbrAndFullDict:
165
163
  """Check for exact matches in abbreviations or full forms between different acronyms."""
166
164
  valid_data = {}
167
165
  no_matches = True
168
- acronyms = sorted(list(data.keys()))
166
+ acronyms_bak = sorted(data.keys())
169
167
 
170
- for i, main_acronym in enumerate(acronyms):
171
- # Normalize items: lowercase and remove parentheses
172
- main_items = [
173
- item.lower().replace("(", "").replace(")", "")
174
- for item in data[main_acronym].get(key_type, [])
175
- ]
168
+ for acronyms in [acronyms_bak, acronyms_bak[::-1]]:
169
+ for i, main_acronym in enumerate(acronyms):
170
+ # Normalize items: lowercase and remove parentheses
171
+ main_items = [
172
+ item.lower().replace("(", "").replace(")", "")
173
+ for item in data[main_acronym].get(key_type, [])
174
+ ]
176
175
 
177
- # Create exact match patterns
178
- patterns = [re.compile(f"^{item}$") for item in main_items]
176
+ # Create exact match patterns
177
+ patterns = [re.compile(f"^{item}$") for item in main_items]
179
178
 
180
- matches_found = []
179
+ matches_found = []
181
180
 
182
- # Compare with other acronyms
183
- for other_acronym in acronyms[i + 1:]:
184
- other_items = [
185
- item.lower().replace("(", "").replace(")", "")
186
- for item in data[other_acronym].get(key_type, [])
187
- ]
181
+ # Compare with other acronyms
182
+ for other_acronym in acronyms[i + 1:]:
183
+ other_items = [
184
+ item.lower().replace("(", "").replace(")", "")
185
+ for item in data[other_acronym].get(key_type, [])
186
+ ]
188
187
 
189
- # Find matching items
190
- matching_items = [
191
- item for item in other_items
192
- if any(pattern.match(item) for pattern in patterns)
193
- ]
188
+ # Find matching items
189
+ matching_items = [
190
+ item for item in other_items
191
+ if any(pattern.match(item) for pattern in patterns)
192
+ ]
194
193
 
195
- if matching_items:
196
- matches_found.append([main_acronym, other_acronym, matching_items])
194
+ if matching_items:
195
+ matches_found.append([main_acronym, other_acronym, matching_items])
197
196
 
198
- if matches_found:
199
- no_matches = False
200
- print(f"Found matches in {key_type}: {matches_found}")
201
- else:
202
- valid_data[main_acronym] = data[main_acronym]
197
+ if matches_found:
198
+ no_matches = False
199
+ print(f"Found matches in {key_type}: {matches_found}")
200
+ else:
201
+ valid_data[main_acronym] = data[main_acronym]
203
202
 
204
203
  return valid_data, no_matches
@@ -1,7 +1,4 @@
1
- # coding=utf-8
2
-
3
1
  import os
4
- from typing import Optional
5
2
 
6
3
  from .core._base import standardize_path
7
4
  from .core.update_json import load_json_data, update_json_file
@@ -14,7 +11,7 @@ def main_generate_md_files(
14
11
  full_json_j: str,
15
12
  full_json_k: str,
16
13
  path_output: str,
17
- path_spidered_bibs: Optional[str] = None,
14
+ path_spidered_bibs: str | None = None,
18
15
  keywords_category_name: str = "",
19
16
  for_vue: bool = True,
20
17
  ) -> None:
@@ -70,7 +67,7 @@ def main_generate_md_files(
70
67
  keywords_list, keywords_category_name = [], ""
71
68
 
72
69
  # Process both conferences and journals
73
- for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"]):
70
+ for cj, ia in zip(["conferences", "journals"], ["inproceedings", "article"], strict=True):
74
71
  # Update JSON data
75
72
  json_dict = {}
76
73
  if cj == "conferences":
@@ -1,9 +1,6 @@
1
- # coding=utf-8
2
-
3
1
  import os
4
2
  import re
5
3
  from datetime import datetime
6
- from typing import Optional
7
4
 
8
5
 
9
6
  def conference_journal_header():
@@ -35,7 +32,7 @@ def conference_journal_header():
35
32
  return conference_header, journal_header
36
33
 
37
34
 
38
- class GenerateDataDict(object):
35
+ class GenerateDataDict:
39
36
  """Generate data dictionaries from JSON input for conferences and journals.
40
37
 
41
38
  This class processes JSON data containing conference or journal information
@@ -65,7 +62,7 @@ class GenerateDataDict(object):
65
62
  inproceedings_or_article: str,
66
63
  json_dict: dict,
67
64
  for_vue: bool = True,
68
- path_spidered_conferences_or_journals: Optional[str] = None,
65
+ path_spidered_conferences_or_journals: str | None = None,
69
66
  ) -> None:
70
67
  """Initialize the GenerateDataDict instance.
71
68
 
@@ -294,7 +291,7 @@ class GenerateDataDict(object):
294
291
  key (str): Key to extract text content from.
295
292
 
296
293
  Returns:
297
- List[str]: List of non-empty text content.
294
+ list[str]: List of non-empty text content.
298
295
  """
299
296
  return [text for text in abbr_dict.get(key, []) if text.strip()]
300
297
 
@@ -328,7 +325,7 @@ class GenerateDataDict(object):
328
325
  cleaned_keywords = {}
329
326
  for category, words in keywords_dict.items():
330
327
  if category.strip():
331
- sorted_words = sorted(set([word.strip() for word in words if word.strip()]))
328
+ sorted_words = sorted({word.strip() for word in words if word.strip()})
332
329
  cleaned_keywords[category.strip()] = sorted_words
333
330
 
334
331
  # Flatten keywords and remove duplicates
@@ -395,8 +392,7 @@ class GenerateDataDict(object):
395
392
  def _generate_for_conference(
396
393
  self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
397
394
  ):
398
- """
399
- Generate a markdown table row for conference information.
395
+ """Generate a markdown table row for conference information.
400
396
 
401
397
  Args:
402
398
  publisher_url: URL of the publisher
@@ -528,8 +524,7 @@ class GenerateDataDict(object):
528
524
  def _generate_for_journal(
529
525
  self, publisher_url, full_name, abbr_name, url_home, url_about, period, top, keywords, abbr, abbr_dict
530
526
  ):
531
- """
532
- Generate a markdown table row for journal information.
527
+ r"""Generate a markdown table row for journal information.
533
528
 
534
529
  Args:
535
530
  publisher_url: URL of the publisher
@@ -586,7 +581,7 @@ class GenerateDataDict(object):
586
581
  inproceedings_or_article (str): Publication type.
587
582
 
588
583
  Returns:
589
- List[str]: Mermaid chart configuration lines, or empty list if no data found.
584
+ list[str]: Mermaid chart configuration lines, or empty list if no data found.
590
585
  """
591
586
  path_spidered_cj = self.path_spidered_cj if self.path_spidered_cj else ""
592
587
  path_readme = os.path.join(path_spidered_cj, publisher, abbr, inproceedings_or_article)
@@ -597,7 +592,7 @@ class GenerateDataDict(object):
597
592
  mermaid, data_dict = [], {}
598
593
  # |AAAI|1980|95|Proceedings of the First National Conference on Artificial Intelligence|
599
594
  regex = re.compile(r"\|.*\|([0-9]+)\|([0-9]+)\|.*\|")
600
- with open(full_readme, "r", encoding="utf-8") as file:
595
+ with open(full_readme, encoding="utf-8") as file:
601
596
  data_list = file.readlines()
602
597
  for line in data_list:
603
598
  if mch := regex.search(line):
@@ -1,8 +1,5 @@
1
- # coding=utf-8
2
-
3
1
  import os
4
2
  import re
5
- from typing import List
6
3
 
7
4
  from ..core._base import standardize_path
8
5
  from .generate_dict import conference_journal_header
@@ -42,7 +39,7 @@ def conference_journal_informations():
42
39
  return conference_inf, journal_inf
43
40
 
44
41
 
45
- class WriteDataToMd(object):
42
+ class WriteDataToMd:
46
43
  """Class to write publication data to Markdown files.
47
44
 
48
45
  This class provides methods to generate various markdown documentation files
@@ -142,7 +139,7 @@ class WriteDataToMd(object):
142
139
  f.writelines(data_list)
143
140
 
144
141
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
145
- def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: List[str]):
142
+ def _default_or_customized_keywords(self, keywords_category_name: str, keywords_list: list[str]):
146
143
  """Get default or customized keywords based on category and provided list.
147
144
 
148
145
  This method returns either a filtered list of keywords based on the provided
@@ -150,10 +147,10 @@ class WriteDataToMd(object):
150
147
 
151
148
  Args:
152
149
  keywords_category_name (str): The category name for keywords filtering.
153
- keywords_list (List[str]): List of keywords to filter by.
150
+ keywords_list (list[str]): List of keywords to filter by.
154
151
 
155
152
  Returns:
156
- List[str]: List of keywords to use for processing.
153
+ list[str]: List of keywords to use for processing.
157
154
 
158
155
  Example:
159
156
  >>> writer._default_or_customized_keywords("ai", ["machine learning", "deep learning"])
@@ -173,7 +170,7 @@ class WriteDataToMd(object):
173
170
  return sorted(keywords)
174
171
 
175
172
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
176
- def save_categories(self, keywords_category_name: str, keywords_list: List[str]) -> None:
173
+ def save_categories(self, keywords_category_name: str, keywords_list: list[str]) -> None:
177
174
  """Save publications categorized by keywords.
178
175
 
179
176
  This method generates markdown files organizing publications by their
@@ -181,7 +178,7 @@ class WriteDataToMd(object):
181
178
 
182
179
  Args:
183
180
  keywords_category_name (str): The category name for keywords filtering.
184
- keywords_list (List[str]): List of keywords to include in the output.
181
+ keywords_list (list[str]): List of keywords to include in the output.
185
182
 
186
183
  Returns:
187
184
  None: This method does not return a value.
@@ -368,7 +365,7 @@ class WriteDataToMd(object):
368
365
  return None
369
366
 
370
367
  # --------- --------- --------- --------- --------- --------- --------- --------- --------- #
371
- def save_statistics(self, keywords_category_name: str, keywords_list: List[str]) -> None:
368
+ def save_statistics(self, keywords_category_name: str, keywords_list: list[str]) -> None:
372
369
  """Save statistics overview file for keywords.
373
370
 
374
371
  This method generates a markdown file containing statistics overview
@@ -376,7 +373,7 @@ class WriteDataToMd(object):
376
373
 
377
374
  Args:
378
375
  keywords_category_name (str): The category name for keywords filtering.
379
- keywords_list (List[str]): List of keywords to include in the output.
376
+ keywords_list (list[str]): List of keywords to include in the output.
380
377
 
381
378
  Returns:
382
379
  None: This method does not return a value.
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "pyformatjson"
3
- version = "0.2.4"
3
+ version = "0.2.5"
4
4
  description = "pyformatjson"
5
5
  license = "GPL-3.0-or-later"
6
6
  authors = ["NextAI <nextartifintell@gmail.com>"]
@@ -23,8 +23,6 @@ sphinx-copybutton = "^0.5.2"
23
23
  [tool.pyright]
24
24
  venvPath = "."
25
25
  venv = ".venv"
26
- reportOptionalSubscript = "none" # "none", "information", "error"
27
- reportIncompatibleMethodOverride = "none" # "none", "information", "error"
28
26
 
29
27
  [tool.ruff]
30
28
  line-length = 120
@@ -48,6 +46,7 @@ extend-select = [
48
46
  "A", # flake8-builtins
49
47
  "C4", # flake8-comprehensions
50
48
  "RUF", # Ruff-specific rules
49
+ "BLE", # flake8-blind-except
51
50
  ]
52
51
 
53
52
  ignore = [
@@ -58,12 +57,16 @@ ignore = [
58
57
  "ANN202", # Missing return type of private function
59
58
  "ANN204", # Missing return type annotation for special method `__init__`
60
59
  "ANN205", # Missing return type annotation for staticmethod
60
+ "D101", # Missing docstring in public class
61
61
  "D102", # Missing docstring in public method
62
62
  "D103", # Missing docstring in public function
63
+ "D104", # Missing docstring in public package
63
64
  "D107", # Missing docstring in `__init__`
65
+ "D100", # Missing docstring in public module
64
66
  "D204", # 1 blank line required after class docstring
65
67
  "D417", # Missing argument description in the docstring
66
68
  "RUF002", # Docstring contains ambiguous `–` (EN-DASH)
69
+ "RUF022", # `__all__` is not sorted
67
70
  ]
68
71
 
69
72
  [tool.ruff.lint.pydocstyle]
File without changes
File without changes