ckparser 0.2.1__tar.gz → 0.3.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ckparser
3
- Version: 0.2.1
3
+ Version: 0.3.1
4
4
  Summary: Parse Paradox Jomini data files to Python/JSON and revert them back experimentally.
5
5
  Author: Marc Debureaux (debnet)
6
6
  License-Expression: MIT
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: ckparser
3
- Version: 0.2.1
3
+ Version: 0.3.1
4
4
  Summary: Parse Paradox Jomini data files to Python/JSON and revert them back experimentally.
5
5
  Author: Marc Debureaux (debnet)
6
6
  License-Expression: MIT
@@ -41,7 +41,7 @@ import re
41
41
  import time
42
42
 
43
43
  # Script version
44
- __version__ = "0.2.1"
44
+ __version__ = "0.3.1"
45
45
 
46
46
  # Logger (because logging is awesome)
47
47
  logger = logging.getLogger(__name__)
@@ -82,8 +82,8 @@ def jomini_object_hook(obj):
82
82
 
83
83
  json_load = functools.partial(json.load, object_hook=jomini_object_hook)
84
84
  json_loads = functools.partial(json.loads, object_hook=jomini_object_hook)
85
- json_dump = functools.partial(json.dump, cls=JominiJSONEncoder, indent=4, ensure_ascii=False)
86
- json_dumps = functools.partial(json.dumps, cls=JominiJSONEncoder, indent=4, ensure_ascii=False)
85
+ json_dump = functools.partial(json.dump, cls=JominiJSONEncoder, indent=4)
86
+ json_dumps = functools.partial(json.dumps, cls=JominiJSONEncoder, indent=4)
87
87
 
88
88
  # Boolean transformation
89
89
  booleans = {"yes": True, "no": False}
@@ -102,17 +102,17 @@ regex_string_multiline = re.compile(r"\"[^\"]*\"", re.MULTILINE)
102
102
  # Regex for quoted strings inside quoted strings
103
103
  regex_inner_string = re.compile(r"\|(?P<index>\d+)\|")
104
104
  # Regex to remove comments in files
105
- regex_comment = re.compile(r"(?P<space>\s*)(?P<comment>#.*)$", re.MULTILINE)
105
+ regex_comment = re.compile(r"(?P<space>\s*)(?P<comment>#.*)", re.MULTILINE)
106
106
  # Regex to fix blocks with no equal sign
107
- regex_block = re.compile(r"^([^\s\{\=]+)\s*\{\s*$", re.MULTILINE)
107
+ regex_missing = re.compile(r"^\s*([\w\.]+)\s+([{])", re.MULTILINE)
108
108
  # Regex to remove "list" prefix
109
109
  regex_list = re.compile(r"\s*=\s*list\s+([\{\"\|])", re.MULTILINE)
110
110
  # Regex for color blocks (color = [rgb|hsv] { x y z })
111
- regex_color = re.compile(r"=\s*(?P<type>\w+)\s*{")
111
+ regex_color = re.compile(r"=\s*(?P<type>\w+)\s*{", re.MULTILINE)
112
112
  # Regex to parse items with format key=value
113
113
  regex_inline = re.compile(r"([^\s\"]+\s*[?!<=>]+\s*(([^@\"]\[?[^\s]+\]?)|(\"[^\"]+\")|(@\[[^\]]+\]))|(@\w+))")
114
- # Regex to parse blocks with bracket below the key
115
- regex_values = re.compile(r"(([?!<=>]+)\s*\n+)|(\n+\s*([?!<=>]+))")
114
+ # Regex to parse blocks with bracket below the key/operator
115
+ regex_block = re.compile(r"(\s*([?!<=>]+)\s+\{)|(\s+\s*([?!<=>])+\s*\{)", re.MULTILINE)
116
116
  # Regex to parse lines with format key=value
117
117
  regex_line = re.compile(r"\"?(?P<key>[^\s\"]+)\"?\s*(?P<operator>[?!<=>]+)\s*(list\s*)?(?P<value>.*)")
118
118
  # Regex to parse independent items in a list
@@ -125,6 +125,8 @@ regex_empty = re.compile(r"(\n\s*\n)+", re.MULTILINE)
125
125
  regex_locale = re.compile(r"^\s*(?P<key>[^\:#]+)\:(\d+)?\s\"(?P<value>.+)\".*$")
126
126
  # Regex for keywords
127
127
  regex_keyword = re.compile(r"(" + "|".join(map(re.escape, sorted(keywords, key=len, reverse=True))) + r") ")
128
+ # Regex for fixing line count when bracket is below the key/operator
129
+ regex_count = re.compile(r"([?!<=>]+)\s*([§\n]+)\s*\{")
128
130
  # Regex for string indexes
129
131
  regex_index = re.compile(r"\|(\d+)\|")
130
132
  # Regex for variables
@@ -182,7 +184,7 @@ def read_file(path, encoding="utf_8_sig"):
182
184
  encoding = result["encoding"]
183
185
  logger.debug(f"Detected encoding: {result['encoding']} ({result['confidence']:0.0%})")
184
186
  del raw_data
185
- with open(path, encoding=encoding) as file:
187
+ with open(path, "r", encoding=encoding) as file:
186
188
  return file.read()
187
189
 
188
190
 
@@ -199,19 +201,19 @@ def parse_text(text, return_text_on_error=False, comments=False, dates=False, fi
199
201
  """
200
202
 
201
203
  def replace(match):
202
- nonlocal strings, index
203
- index = len(strings)
204
- strings[str(index)] = match.group(0).replace("\n", "\\n")
205
- return f"|{index}|"
204
+ nonlocal strings, strings_index
205
+ strings_index = len(strings)
206
+ strings[str(strings_index)] = match.group(0).replace("\n", "\\n")
207
+ return f"|{strings_index}|"
206
208
 
207
209
  def replace_comment(match):
208
- nonlocal strings, index
209
- index = len(strings)
210
+ nonlocal strings, strings_index
211
+ strings_index = len(strings)
210
212
  value, space = match.group("comment").replace('"', "'").strip(), match.group("space")
211
- strings[str(index)] = f'"{value}"'
213
+ strings[str(strings_index)] = f'"{value}"'
212
214
  if not value.strip():
213
215
  return ""
214
- return f"\n{space}#{index}=|{index}|\n"
216
+ return f"\n{space}#{strings_index}=|{strings_index}|"
215
217
 
216
218
  def set_variable(key, value, is_global=is_global):
217
219
  global global_variables
@@ -223,29 +225,41 @@ def parse_text(text, return_text_on_error=False, comments=False, dates=False, fi
223
225
 
224
226
  root = {}
225
227
  nodes = [("", root)]
226
- strings, index = {}, 0
228
+ strings, strings_index = {}, 0
227
229
  variables = global_variables.copy()
228
230
  # Cleaning document
231
+ text = text.replace("\n", "\n§\n")
229
232
  text = regex_string.sub(replace, text)
230
233
  if comments:
231
234
  text = regex_comment.sub(replace_comment, text)
232
235
  else:
233
236
  text = regex_comment.sub("", text)
234
237
  text = regex_string_multiline.sub(replace, text)
238
+ if missings := regex_missing.findall(text):
239
+ if filename:
240
+ logger.warning(f"Filename: {filename}")
241
+ for key, val in missings:
242
+ logger.warning(f"Potential missing `=` operator between `{key}` and `{val}` needs to be fixed.")
243
+ text = regex_missing.sub(r"\g<1>=\g<2>", text)
244
+ text = regex_color.sub(r"={\n\g<1>", text)
235
245
  text = regex_list.sub(r"|list=\g<1>", text)
236
- text = regex_block.sub(r"\g<1>={", text)
237
246
  text = text.replace("{", "\n{\n").replace("}", "\n}\n")
238
- text = regex_color.sub(r"={\n\g<1>", text)
239
247
  text = regex_inline.sub(r"\g<1>\n", text)
240
- text = regex_values.sub(r"\g<2>\g<4>", text)
248
+ text = regex_block.sub(r"\g<2>\g<4>{", text)
241
249
  text = regex_empty.sub(r"\n", text)
242
250
  text = regex_keyword.sub(r"\1|", text)
251
+ text = regex_count.sub(r"\g<1>{\n\g<2>", text)
243
252
  text = regex_index.sub(lambda match: strings[match.group(1)], text)
244
253
 
245
254
  # Parsing document line by line
246
- for line_number, line_text in enumerate(text.splitlines(), start=1):
255
+ line_number = 1
256
+ for line_text in text.splitlines():
247
257
  try:
248
258
  line_text = line_text.strip()
259
+ # Line number
260
+ if count := line_text.count("§"):
261
+ line_number += count
262
+ line_text = line_text.rstrip("§")
249
263
  # Nothing to do if line is empty
250
264
  if not line_text:
251
265
  continue
@@ -511,32 +525,32 @@ def parse_file(
511
525
  start_time = time.monotonic()
512
526
  if base_dir:
513
527
  base_dir = os.sep.join(str(base_dir).rstrip(os.sep).split(os.sep)[:-1]) + os.sep
514
- base_dir = os.path.dirname(path.replace(base_dir.replace(os.sep, "/"), ""))
528
+ base_dir = os.path.dirname(path.replace(base_dir, ""))
515
529
  base_dir = base_dir or "."
516
530
  text = read_file(path, encoding)
517
531
  if not text or not text.strip():
518
532
  return None
519
533
  for pattern, replacement in patch or []:
520
534
  text = re.sub(pattern, replacement, text)
521
- filename = os.path.join(base_dir, os.path.basename(path)).replace(os.sep, "/")
535
+ filename = os.path.join(base_dir, os.path.basename(path))
522
536
  logger.debug(f"Parsing {filename}")
523
537
  data = parse_text(
524
538
  text, return_text_on_error=True, comments=comments, dates=dates, filename=filename, is_global=is_global
525
539
  )
526
540
  if save:
527
541
  filename, _ = os.path.splitext(os.path.basename(path))
528
- directory = os.path.join(output_dir or "output", *base_dir.split("/")).replace(os.sep, "/")
542
+ directory = os.path.join(output_dir or "output", *base_dir.split(os.sep))
529
543
  os.makedirs(directory, exist_ok=True)
530
544
  if not isinstance(data, dict):
531
545
  filename = os.path.join(directory, filename + ".error")
532
546
  try:
533
- with open(filename, "w") as file:
547
+ with open(filename, "w", encoding=encoding) as file:
534
548
  file.write(data)
535
549
  except UnicodeEncodeError as error:
536
550
  logger.error(f"Unable to write file {filename}: {error}")
537
551
  else:
538
552
  filename = os.path.join(directory, filename + ".json")
539
- with open(filename, "w") as file:
553
+ with open(filename, "w", encoding="utf-8") as file:
540
554
  json_dump(data, file)
541
555
  total_time = time.monotonic() - start_time
542
556
  logger.debug(f"Elapsed time: {total_time:0.3f}s!")
@@ -589,7 +603,7 @@ def parse_all_files(
589
603
  for filename in all_files:
590
604
  if not filename.lower().endswith(".txt"):
591
605
  continue
592
- filepath = os.path.join(current_path, filename).replace(os.sep, "/")
606
+ filepath = os.path.join(current_path, filename)
593
607
  data = parse_file(
594
608
  filepath,
595
609
  output_dir=output_dir,
@@ -603,7 +617,7 @@ def parse_all_files(
603
617
  if isinstance(data, str):
604
618
  errors.append(filepath)
605
619
  continue
606
- filepath = filepath.replace(str(path), "").lstrip("/")
620
+ filepath = filepath.replace(str(path), "").lstrip(os.sep)
607
621
  success[filepath] = data if keep_data else True
608
622
  total_time = time.monotonic() - start_time
609
623
  logger.info(f"{len(success)} parsed file(s) and {len(errors)} errors in {total_time:0.3f}s!")
@@ -647,7 +661,7 @@ def parse_all_locales(path, encoding="utf_8_sig", language="english", save=False
647
661
  key, _, value = match.groups()
648
662
  locales[key] = value
649
663
  if save:
650
- with open(output, "w") as file:
664
+ with open(output, "w", encoding="utf-8") as file:
651
665
  json_dump(locales, file, sort_keys=True)
652
666
  return locales
653
667
 
@@ -822,7 +836,7 @@ def revert_file(path, output_dir=None, encoding="utf_8_sig", base_dir=None, save
822
836
  base_dir = os.sep.join(str(base_dir).rstrip(os.sep).split(os.sep)[:-1]) + os.sep
823
837
  base_dir = os.path.dirname(path.replace(base_dir, ""))
824
838
  base_dir = base_dir or "."
825
- with open(path) as file:
839
+ with open(path, "r", encoding="utf-8") as file:
826
840
  data = json.load(file)
827
841
  filename = os.path.join(str(base_dir), os.path.basename(path))
828
842
  logger.debug(f"Reverting {filename}")
@@ -846,7 +860,7 @@ def load_variables(filepath="_variables.json"):
846
860
  global global_variables
847
861
  if not os.path.exists(filepath):
848
862
  return
849
- with open(filepath) as file:
863
+ with open(filepath, "r", encoding="utf-8") as file:
850
864
  global_variables = json.load(file)
851
865
 
852
866
 
@@ -856,7 +870,7 @@ def save_variables(filepath="_variables.json"):
856
870
  """
857
871
  if not global_variables:
858
872
  return
859
- with open(filepath, "w") as file:
873
+ with open(filepath, "w", encoding="utf-8") as file:
860
874
  json_dump(global_variables, file, sort_keys=True)
861
875
 
862
876
 
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "ckparser"
7
- version = "0.2.1"
7
+ version = "0.3.1"
8
8
  description = "Parse Paradox Jomini data files to Python/JSON and revert them back experimentally."
9
9
  readme = "README.md"
10
10
  requires-python = ">=3.9"
File without changes
File without changes
File without changes
File without changes