partialjson 1.0.0__tar.gz → 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: partialjson
3
- Version: 1.0.0
3
+ Version: 1.1.0
4
4
  Summary: Parse incomplete or partial json
5
5
  Home-page: https://github.com/iw4p/partialjson
6
6
  Author: Nima Akbarzadeh
@@ -8,6 +8,8 @@ Author-email: iw4p@protonmail.com
8
8
  License: MIT
9
9
  Description-Content-Type: text/markdown
10
10
  License-File: LICENSE
11
+ Provides-Extra: json5
12
+ Requires-Dist: json5; extra == "json5"
11
13
  Dynamic: author
12
14
  Dynamic: author-email
13
15
  Dynamic: description
@@ -15,6 +17,7 @@ Dynamic: description-content-type
15
17
  Dynamic: home-page
16
18
  Dynamic: license
17
19
  Dynamic: license-file
20
+ Dynamic: provides-extra
18
21
  Dynamic: summary
19
22
 
20
23
  # PartialJson
@@ -34,18 +37,18 @@ Dynamic: summary
34
37
  ## Example
35
38
 
36
39
  ```python
37
- from partialjson.json_parser import JSONParser
40
+ from partialjson import JSONParser
38
41
  parser = JSONParser()
39
42
 
40
43
  incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
41
44
  print(parser.parse(incomplete_json))
42
- # {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
45
+ # {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
43
46
  ```
44
47
 
45
- Problem with `\n`? strict mode is here
48
+ Problem with `\n`? Use `strict=False`:
46
49
 
47
50
  ```python
48
- from partialjson.json_parser import JSONParser
51
+ from partialjson import JSONParser
49
52
  parser = JSONParser(strict=False)
50
53
 
51
54
  incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
@@ -53,6 +56,21 @@ print(parser.parse(incomplete_json))
53
56
  # {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
54
57
  ```
55
58
 
59
+ ### JSON5 support
60
+
61
+ Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
62
+
63
+ ```python
64
+ from partialjson import create_json5_parser
65
+ parser = create_json5_parser()
66
+
67
+ incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
68
+ print(parser.parse(incomplete_json5))
69
+ # {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
70
+ ```
71
+
72
+ Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
73
+
56
74
  ### Installation
57
75
 
58
76
  ```sh
@@ -15,18 +15,18 @@
15
15
  ## Example
16
16
 
17
17
  ```python
18
- from partialjson.json_parser import JSONParser
18
+ from partialjson import JSONParser
19
19
  parser = JSONParser()
20
20
 
21
21
  incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
22
22
  print(parser.parse(incomplete_json))
23
- # {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
23
+ # {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
24
24
  ```
25
25
 
26
- Problem with `\n`? strict mode is here
26
+ Problem with `\n`? Use `strict=False`:
27
27
 
28
28
  ```python
29
- from partialjson.json_parser import JSONParser
29
+ from partialjson import JSONParser
30
30
  parser = JSONParser(strict=False)
31
31
 
32
32
  incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
@@ -34,6 +34,21 @@ print(parser.parse(incomplete_json))
34
34
  # {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
35
35
  ```
36
36
 
37
+ ### JSON5 support
38
+
39
+ Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
40
+
41
+ ```python
42
+ from partialjson import create_json5_parser
43
+ parser = create_json5_parser()
44
+
45
+ incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
46
+ print(parser.parse(incomplete_json5))
47
+ # {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
48
+ ```
49
+
50
+ Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
51
+
37
52
  ### Installation
38
53
 
39
54
  ```sh
@@ -4,9 +4,10 @@ Partial Json.
4
4
  Parsing ChatGPT JSON stream response — Partial and incomplete JSON parser python library for OpenAI
5
5
  """
6
6
 
7
- from .json_parser import JSONParser
7
+ from .json_parser import JSONParser, create_json_parser
8
+ from .json5_parser import create_json5_parser
8
9
 
9
- __version__ = "1.0.0"
10
+ __version__ = "1.1.0"
10
11
  __author__ = "Nima Akbarzadeh"
11
12
  __author_email__ = "iw4p@protonmail.com"
12
13
  __license__ = "MIT"
@@ -16,5 +17,7 @@ PYPI_SIMPLE_ENDPOINT: str = "https://pypi.org/project/partialjson"
16
17
 
17
18
  __all__ = [
18
19
  "JSONParser",
20
+ "create_json_parser",
21
+ "create_json5_parser",
19
22
  "PYPI_SIMPLE_ENDPOINT",
20
23
  ]
@@ -0,0 +1,346 @@
1
+ """JSON5 parser - extends JSON with comments, unquoted keys, single quotes, etc."""
2
+ import json
3
+ import re
4
+
5
+ try:
6
+ import json5
7
+ except ImportError:
8
+ json5 = None
9
+
10
+
11
+ def create_json5_parser(strict=True, on_extra_token=None):
12
+ """Create a JSON5 parser."""
13
+ return _JSON5Parser(strict=strict, on_extra_token=on_extra_token)
14
+
15
+
16
+ def _default_on_extra_token(text, data, reminding):
17
+ print("Parsed JSON with extra tokens:", {"text": text, "data": data, "reminding": reminding})
18
+
19
+
20
+ _INCOMPLETE_ESCAPE_REGEX = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3}|x[0-9a-fA-F]{0,1})?$")
21
+ _JSON5_WHITESPACE = "\v\f\u00A0\u2028\u2029\uFEFF"
22
+
23
+
24
+ class _JSON5Parser:
25
+ """JSON5 parser with comments, unquoted keys, single quotes, hex, Infinity, etc."""
26
+
27
+ def __init__(self, strict=True, on_extra_token=None):
28
+ self.strict = strict
29
+ self.on_extra_token = on_extra_token or _default_on_extra_token
30
+ self.last_parse_reminding = None
31
+ self._parsers = self._build_parsers()
32
+
33
+ def _build_parsers(self):
34
+ parsers = {
35
+ " ": self._parse_space,
36
+ "\r": self._parse_space,
37
+ "\n": self._parse_space,
38
+ "\t": self._parse_space,
39
+ "[": self._parse_array,
40
+ "{": self._parse_object,
41
+ '"': self._parse_string,
42
+ "'": self._parse_string,
43
+ "t": self._parse_true,
44
+ "f": self._parse_false,
45
+ "n": self._parse_null,
46
+ "/": self._parse_space,
47
+ "+": self._parse_number,
48
+ "I": self._parse_number,
49
+ "N": self._parse_n_literal,
50
+ "T": self._parse_true,
51
+ "F": self._parse_false,
52
+ }
53
+ for c in _JSON5_WHITESPACE:
54
+ parsers[c] = self._parse_space
55
+ for c in "0123456789.-":
56
+ parsers[c] = self._parse_number
57
+ return parsers
58
+
59
+ def parse(self, s):
60
+ if len(s) >= 1:
61
+ if json5:
62
+ try:
63
+ return json5.loads(s)
64
+ except (json.JSONDecodeError, ValueError) as e:
65
+ data, reminding = self.parse_any(s, e)
66
+ self.last_parse_reminding = reminding
67
+ if self.on_extra_token and reminding:
68
+ self.on_extra_token(s, data, reminding)
69
+ return data
70
+ data, reminding = self.parse_any(s, json.JSONDecodeError("", "", 0))
71
+ self.last_parse_reminding = reminding
72
+ if self.on_extra_token and reminding:
73
+ self.on_extra_token(s, data, reminding)
74
+ return data
75
+ return json.loads("{}")
76
+
77
+ def parse_any(self, s, e):
78
+ if not s:
79
+ raise e
80
+ while s and self._is_space_or_comment_start(s):
81
+ s = self._parse_space(s, e)
82
+ if not s:
83
+ return None, ""
84
+ parser = self._parsers.get(s[0])
85
+ if not parser:
86
+ raise e
87
+ return parser(s, e)
88
+
89
+ def _is_space_or_comment_start(self, s):
90
+ if not s:
91
+ return False
92
+ c = s[0]
93
+ if c.isspace() or c in _JSON5_WHITESPACE:
94
+ return True
95
+ if s.startswith("//") or s.startswith("/*"):
96
+ return True
97
+ return False
98
+
99
+ def _parse_space(self, s, e):
100
+ i = 0
101
+ while i < len(s):
102
+ if s[i].isspace() or s[i] in _JSON5_WHITESPACE:
103
+ i += 1
104
+ elif s[i : i + 2] == "//":
105
+ i += 2
106
+ while i < len(s) and s[i] not in "\n\r\u2028\u2029":
107
+ i += 1
108
+ elif s[i : i + 2] == "/*":
109
+ i += 2
110
+ end = s.find("*/", i)
111
+ if end == -1:
112
+ return ""
113
+ i = end + 2
114
+ else:
115
+ break
116
+ return s[i:]
117
+
118
+ def _parse_array(self, s, e):
119
+ s = s[1:]
120
+ acc = []
121
+ while True:
122
+ while s and self._is_space_or_comment_start(s):
123
+ s = self._parse_space(s, e)
124
+ if not s:
125
+ break
126
+ if s[0] == "]":
127
+ s = s[1:]
128
+ break
129
+ res, s = self.parse_any(s, e)
130
+ acc.append(res)
131
+ while s and self._is_space_or_comment_start(s):
132
+ s = self._parse_space(s, e)
133
+ if s and s.startswith(","):
134
+ s = s[1:]
135
+ return acc, s
136
+
137
+ def _parse_object(self, s, e):
138
+ s = s[1:]
139
+ acc = {}
140
+ while True:
141
+ while s and self._is_space_or_comment_start(s):
142
+ s = self._parse_space(s, e)
143
+ if not s:
144
+ break
145
+ if s[0] == "}":
146
+ s = s[1:]
147
+ break
148
+ if s[0] not in '"\'':
149
+ key, s = self._parse_identifier(s, e)
150
+ if not key:
151
+ while s and self._is_space_or_comment_start(s):
152
+ s = self._parse_space(s, e)
153
+ if s and s[0] == "}":
154
+ s = s[1:]
155
+ break
156
+ else:
157
+ key, s = self.parse_any(s, e)
158
+ while s and self._is_space_or_comment_start(s):
159
+ s = self._parse_space(s, e)
160
+ if not s or s[0] == "}":
161
+ if key is not None:
162
+ acc[key] = None
163
+ if s and s[0] == "}":
164
+ s = s[1:]
165
+ break
166
+ if s[0] != ":":
167
+ if key is not None:
168
+ acc[key] = None
169
+ break
170
+ s = s[1:]
171
+ while s and self._is_space_or_comment_start(s):
172
+ s = self._parse_space(s, e)
173
+ if not s or s[0] in ",}":
174
+ acc[key] = None
175
+ if s and s.startswith(","):
176
+ s = s[1:]
177
+ elif s and s.startswith("}"):
178
+ s = s[1:]
179
+ break
180
+ while s and self._is_space_or_comment_start(s):
181
+ s = self._parse_space(s, e)
182
+ if s and (
183
+ s[0] in self._parsers
184
+ or s[0] in "/+IN"
185
+ or s[0] in _JSON5_WHITESPACE
186
+ ):
187
+ value, s = self.parse_any(s, e)
188
+ acc[key] = value
189
+ else:
190
+ if key is not None:
191
+ acc[key] = None
192
+ break
193
+ while s and self._is_space_or_comment_start(s):
194
+ s = self._parse_space(s, e)
195
+ if s and s.startswith(","):
196
+ s = s[1:]
197
+ return acc, s
198
+
199
+ def _parse_identifier(self, s, e):
200
+ i = 0
201
+ while i < len(s) and (s[i].isalnum() or s[i] in "_$"):
202
+ i += 1
203
+ return s[:i], s[i:]
204
+
205
+ def _parse_string(self, s, e):
206
+ quote = s[0]
207
+ end = 1
208
+ while end < len(s):
209
+ if s[end] == "\\":
210
+ end += 2
211
+ continue
212
+ if s[end] == quote:
213
+ break
214
+ end += 1
215
+
216
+ if end >= len(s):
217
+ content = s[1:]
218
+ if not self.strict:
219
+ return content, ""
220
+ if _INCOMPLETE_ESCAPE_REGEX.match(content):
221
+ return "", ""
222
+ try:
223
+ if quote == "'":
224
+ return content, ""
225
+ return json.loads(f'"{content}"'), ""
226
+ except json.JSONDecodeError:
227
+ return "", ""
228
+
229
+ str_val = s[: end + 1]
230
+ remainder = s[end + 1 :]
231
+
232
+ if json5:
233
+ try:
234
+ return json5.loads(str_val), remainder
235
+ except Exception:
236
+ pass
237
+
238
+ decoded = str_val[1:-1]
239
+ decoded = re.sub(r"\\\n", "", decoded)
240
+ decoded = re.sub(r"\\\r\n", "", decoded)
241
+
242
+ def replace_hex(match):
243
+ return chr(int(match.group(1), 16))
244
+
245
+ decoded = re.sub(r"\\x([0-9a-fA-F]{2})", replace_hex, decoded)
246
+
247
+ if quote == "'":
248
+ decoded = decoded.replace('"', '\\"').replace("\\'", "'")
249
+ try:
250
+ return json.loads(f'"{decoded}"'), remainder
251
+ except Exception:
252
+ return decoded, remainder
253
+ if "\\x" in decoded or "\\\n" in str_val or "\\\r" in str_val:
254
+ return decoded, remainder
255
+ try:
256
+ return json.loads(str_val), remainder
257
+ except Exception:
258
+ return decoded, remainder
259
+
260
+ def _parse_number(self, s, e):
261
+ if s.startswith(("-0x", "-0X")):
262
+ i = 3
263
+ while i < len(s) and s[i] in "0123456789abcdefABCDEF":
264
+ i += 1
265
+ num_str = s[1:i]
266
+ remainder = s[i:]
267
+ if len(num_str) <= 2:
268
+ return s[:3], ""
269
+ return -int(num_str, 16), remainder
270
+ if s.startswith(("+0x", "+0X")):
271
+ i = 3
272
+ while i < len(s) and s[i] in "0123456789abcdefABCDEF":
273
+ i += 1
274
+ num_str = s[1:i]
275
+ remainder = s[i:]
276
+ if len(num_str) <= 2:
277
+ return s[:3], ""
278
+ return int(num_str, 16), remainder
279
+ if s.startswith(("0x", "0X")):
280
+ i = 2
281
+ while i < len(s) and s[i] in "0123456789abcdefABCDEF":
282
+ i += 1
283
+ num_str = s[:i]
284
+ remainder = s[i:]
285
+ if len(num_str) <= 2:
286
+ return num_str, ""
287
+ return int(num_str, 16), remainder
288
+
289
+ for literal, val in [("Infinity", float("inf")), ("NaN", float("nan"))]:
290
+ if s.startswith(literal):
291
+ return val, s[len(literal) :]
292
+ if s.startswith("+" + literal):
293
+ return val, s[len(literal) + 1 :]
294
+ if s.startswith("-" + literal):
295
+ return -val if literal == "Infinity" else val, s[len(literal) + 1 :]
296
+
297
+ if s.startswith(".") and len(s) > 1 and s[1].isdigit():
298
+ i = 1
299
+ while i < len(s) and s[i].isdigit():
300
+ i += 1
301
+ num_str = s[:i]
302
+ return float(num_str), s[i:]
303
+
304
+ if s.startswith("+"):
305
+ res, remainder = self._parse_number(s[1:], e)
306
+ return res, remainder
307
+
308
+ i = 0
309
+ while i < len(s) and s[i] in "0123456789.-":
310
+ i += 1
311
+ num_str = s[:i]
312
+ s = s[i:]
313
+ if not num_str or num_str == "-" or num_str == ".":
314
+ return num_str, ""
315
+ try:
316
+ if num_str.endswith("."):
317
+ num = int(num_str[:-1])
318
+ else:
319
+ num = (
320
+ float(num_str)
321
+ if "." in num_str or "e" in num_str or "E" in num_str
322
+ else int(num_str)
323
+ )
324
+ except ValueError:
325
+ raise e
326
+ return num, s
327
+
328
+ def _parse_n_literal(self, s, e):
329
+ if s.lower().startswith("nan"):
330
+ return self._parse_number(s, e)
331
+ return self._parse_null(s, e)
332
+
333
+ def _parse_true(self, s, e):
334
+ if s.lower().startswith("true"):
335
+ return True, s[4:]
336
+ raise e
337
+
338
+ def _parse_false(self, s, e):
339
+ if s.lower().startswith("false"):
340
+ return False, s[5:]
341
+ raise e
342
+
343
+ def _parse_null(self, s, e):
344
+ if s.lower().startswith("null"):
345
+ return None, s[4:]
346
+ raise e
@@ -0,0 +1,233 @@
1
+ """Pure JSON parser - no JSON5 extensions."""
2
+ import json
3
+ import re
4
+
5
+
6
+ def create_json_parser(strict=True, on_extra_token=None):
7
+ """Create a JSON parser (no JSON5 extensions)."""
8
+ return _JSONParser(strict=strict, on_extra_token=on_extra_token)
9
+
10
+
11
+ def _default_on_extra_token(text, data, reminding):
12
+ print("Parsed JSON with extra tokens:", {"text": text, "data": data, "reminding": reminding})
13
+
14
+
15
+ _INCOMPLETE_ESCAPE_REGEX = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3}|x[0-9a-fA-F]{0,1})?$")
16
+
17
+
18
+ class _JSONParser:
19
+ """Internal JSON-only parser implementation."""
20
+
21
+ def __init__(self, strict=True, on_extra_token=None):
22
+ self.strict = strict
23
+ self.on_extra_token = on_extra_token or _default_on_extra_token
24
+ self.last_parse_reminding = None
25
+ self._parsers = self._build_parsers()
26
+
27
+ def _build_parsers(self):
28
+ parsers = {
29
+ " ": self._parse_space,
30
+ "\r": self._parse_space,
31
+ "\n": self._parse_space,
32
+ "\t": self._parse_space,
33
+ "[": self._parse_array,
34
+ "{": self._parse_object,
35
+ '"': self._parse_string,
36
+ "t": self._parse_true,
37
+ "f": self._parse_false,
38
+ "n": self._parse_null,
39
+ }
40
+ for c in "0123456789.-":
41
+ parsers[c] = self._parse_number
42
+ return parsers
43
+
44
+ def parse(self, s):
45
+ if len(s) >= 1:
46
+ try:
47
+ return json.loads(s)
48
+ except (json.JSONDecodeError, ValueError) as e:
49
+ data, reminding = self.parse_any(s, e)
50
+ self.last_parse_reminding = reminding
51
+ if self.on_extra_token and reminding:
52
+ self.on_extra_token(s, data, reminding)
53
+ return data
54
+ return json.loads("{}")
55
+
56
+ def parse_any(self, s, e):
57
+ if not s:
58
+ raise e
59
+ while s and s[0].isspace():
60
+ s = self._parse_space(s, e)
61
+ if not s:
62
+ return None, ""
63
+ parser = self._parsers.get(s[0])
64
+ if not parser:
65
+ raise e
66
+ return parser(s, e)
67
+
68
+ def _parse_space(self, s, e):
69
+ i = 0
70
+ while i < len(s) and s[i].isspace():
71
+ i += 1
72
+ return s[i:]
73
+
74
+ def _parse_array(self, s, e):
75
+ s = s[1:]
76
+ acc = []
77
+ while True:
78
+ while s and s[0].isspace():
79
+ s = self._parse_space(s, e)
80
+ if not s:
81
+ break
82
+ if s[0] == "]":
83
+ s = s[1:]
84
+ break
85
+ res, s = self.parse_any(s, e)
86
+ acc.append(res)
87
+ while s and s[0].isspace():
88
+ s = self._parse_space(s, e)
89
+ if s and s.startswith(","):
90
+ s = s[1:]
91
+ return acc, s
92
+
93
+ def _parse_object(self, s, e):
94
+ s = s[1:]
95
+ acc = {}
96
+ while True:
97
+ while s and s[0].isspace():
98
+ s = self._parse_space(s, e)
99
+ if not s:
100
+ break
101
+ if s[0] == "}":
102
+ s = s[1:]
103
+ break
104
+ key, s = self.parse_any(s, e)
105
+ while s and s[0].isspace():
106
+ s = self._parse_space(s, e)
107
+ if not s or s[0] == "}":
108
+ if key is not None:
109
+ acc[key] = None
110
+ if s and s[0] == "}":
111
+ s = s[1:]
112
+ break
113
+ if s[0] != ":":
114
+ if key is not None:
115
+ acc[key] = None
116
+ break
117
+ s = s[1:]
118
+ while s and s[0].isspace():
119
+ s = self._parse_space(s, e)
120
+ if not s or s[0] in ",}":
121
+ acc[key] = None
122
+ if s and s.startswith(","):
123
+ s = s[1:]
124
+ elif s and s.startswith("}"):
125
+ s = s[1:]
126
+ break
127
+ if s and s[0] in self._parsers:
128
+ value, s = self.parse_any(s, e)
129
+ acc[key] = value
130
+ else:
131
+ if key is not None:
132
+ acc[key] = None
133
+ break
134
+ while s and s[0].isspace():
135
+ s = self._parse_space(s, e)
136
+ if s and s.startswith(","):
137
+ s = s[1:]
138
+ return acc, s
139
+
140
+ def _parse_string(self, s, e):
141
+ quote = s[0]
142
+ end = 1
143
+ while end < len(s):
144
+ if s[end] == "\\":
145
+ end += 2
146
+ continue
147
+ if s[end] == quote:
148
+ break
149
+ end += 1
150
+
151
+ if end >= len(s):
152
+ content = s[1:]
153
+ if not self.strict:
154
+ return content, ""
155
+ if _INCOMPLETE_ESCAPE_REGEX.match(content):
156
+ return "", ""
157
+ try:
158
+ return json.loads(f'"{content}"'), ""
159
+ except json.JSONDecodeError:
160
+ return "", ""
161
+
162
+ str_val = s[: end + 1]
163
+ remainder = s[end + 1 :]
164
+ if not self.strict:
165
+ return str_val[1:-1], remainder
166
+ return json.loads(str_val), remainder
167
+
168
+ def _parse_number(self, s, e):
169
+ i = 0
170
+ while i < len(s) and s[i] in "0123456789.-":
171
+ i += 1
172
+ num_str = s[:i]
173
+ s = s[i:]
174
+ if not num_str or num_str == "-" or num_str == ".":
175
+ return num_str, ""
176
+ try:
177
+ if num_str.endswith("."):
178
+ num = int(num_str[:-1])
179
+ else:
180
+ num = (
181
+ float(num_str)
182
+ if "." in num_str or "e" in num_str or "E" in num_str
183
+ else int(num_str)
184
+ )
185
+ except ValueError:
186
+ raise e
187
+ return num, s
188
+
189
+ def _parse_true(self, s, e):
190
+ if s.startswith("t") or s.startswith("T"):
191
+ return True, s[4:]
192
+ raise e
193
+
194
+ def _parse_false(self, s, e):
195
+ if s.startswith("f") or s.startswith("F"):
196
+ return False, s[5:]
197
+ raise e
198
+
199
+ def _parse_null(self, s, e):
200
+ if s.startswith("n"):
201
+ return None, s[4:]
202
+ raise e
203
+
204
+
205
+ # Backward compatibility
206
+ class JSONParser:
207
+ """JSON parser. Use create_json_parser() or create_json5_parser() for new code."""
208
+
209
+ def __init__(self, strict=True, json5_enabled=False, on_extra_token=None):
210
+ if json5_enabled:
211
+ from .json5_parser import create_json5_parser
212
+
213
+ self._impl = create_json5_parser(strict=strict, on_extra_token=on_extra_token)
214
+ else:
215
+ self._impl = create_json_parser(strict=strict, on_extra_token=on_extra_token)
216
+
217
+ def parse(self, s):
218
+ return self._impl.parse(s)
219
+
220
+ def parse_any(self, s, e):
221
+ return self._impl.parse_any(s, e)
222
+
223
+ @property
224
+ def last_parse_reminding(self):
225
+ return getattr(self._impl, "last_parse_reminding", None)
226
+
227
+ @property
228
+ def on_extra_token(self):
229
+ return getattr(self._impl, "on_extra_token", None)
230
+
231
+ @on_extra_token.setter
232
+ def on_extra_token(self, value):
233
+ self._impl.on_extra_token = value
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: partialjson
3
- Version: 1.0.0
3
+ Version: 1.1.0
4
4
  Summary: Parse incomplete or partial json
5
5
  Home-page: https://github.com/iw4p/partialjson
6
6
  Author: Nima Akbarzadeh
@@ -8,6 +8,8 @@ Author-email: iw4p@protonmail.com
8
8
  License: MIT
9
9
  Description-Content-Type: text/markdown
10
10
  License-File: LICENSE
11
+ Provides-Extra: json5
12
+ Requires-Dist: json5; extra == "json5"
11
13
  Dynamic: author
12
14
  Dynamic: author-email
13
15
  Dynamic: description
@@ -15,6 +17,7 @@ Dynamic: description-content-type
15
17
  Dynamic: home-page
16
18
  Dynamic: license
17
19
  Dynamic: license-file
20
+ Dynamic: provides-extra
18
21
  Dynamic: summary
19
22
 
20
23
  # PartialJson
@@ -34,18 +37,18 @@ Dynamic: summary
34
37
  ## Example
35
38
 
36
39
  ```python
37
- from partialjson.json_parser import JSONParser
40
+ from partialjson import JSONParser
38
41
  parser = JSONParser()
39
42
 
40
43
  incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
41
44
  print(parser.parse(incomplete_json))
42
- # {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
45
+ # {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
43
46
  ```
44
47
 
45
- Problem with `\n`? strict mode is here
48
+ Problem with `\n`? Use `strict=False`:
46
49
 
47
50
  ```python
48
- from partialjson.json_parser import JSONParser
51
+ from partialjson import JSONParser
49
52
  parser = JSONParser(strict=False)
50
53
 
51
54
  incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
@@ -53,6 +56,21 @@ print(parser.parse(incomplete_json))
53
56
  # {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
54
57
  ```
55
58
 
59
+ ### JSON5 support
60
+
61
+ Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
62
+
63
+ ```python
64
+ from partialjson import create_json5_parser
65
+ parser = create_json5_parser()
66
+
67
+ incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
68
+ print(parser.parse(incomplete_json5))
69
+ # {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
70
+ ```
71
+
72
+ Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
73
+
56
74
  ### Installation
57
75
 
58
76
  ```sh
@@ -2,9 +2,12 @@ LICENSE
2
2
  README.md
3
3
  setup.py
4
4
  partialjson/__init__.py
5
+ partialjson/json5_parser.py
5
6
  partialjson/json_parser.py
6
7
  partialjson.egg-info/PKG-INFO
7
8
  partialjson.egg-info/SOURCES.txt
8
9
  partialjson.egg-info/dependency_links.txt
10
+ partialjson.egg-info/requires.txt
9
11
  partialjson.egg-info/top_level.txt
12
+ tests/test_json5.py
10
13
  tests/test_parser.py
@@ -0,0 +1,3 @@
1
+
2
+ [json5]
3
+ json5
@@ -31,4 +31,5 @@ setup(
31
31
  author_email=get_property("__author_email__"),
32
32
  license=get_property("__license__"),
33
33
  packages=["partialjson"],
34
+ extras_require={"json5": ["json5"]},
34
35
  )
@@ -0,0 +1,57 @@
1
+ import pytest
2
+ import math
3
+ from partialjson.json_parser import JSONParser
4
+
5
+ def test_json5_comments():
6
+ parser = JSONParser(json5_enabled=True)
7
+ assert parser.parse("{// comment\n\"a\": 1}") == {"a": 1}
8
+ assert parser.parse("{/* multi-line\n comment */\"a\": 1}") == {"a": 1}
9
+ assert parser.parse("{/* incomplete comment") == {}
10
+
11
+ def test_json5_unquoted_keys():
12
+ parser = JSONParser(json5_enabled=True)
13
+ assert parser.parse("{a: 1, b: 2}") == {"a": 1, "b": 2}
14
+ assert parser.parse("{_foo: \"bar\", $baz: 3}") == {"_foo": "bar", "$baz": 3}
15
+ assert parser.parse("{a: 1, ") == {"a": 1}
16
+
17
+ def test_json5_single_quotes():
18
+ parser = JSONParser(json5_enabled=True)
19
+ assert parser.parse("'hello'") == "hello"
20
+ assert parser.parse("{'a': 'b'}") == {"a": "b"}
21
+ assert parser.parse("'it\\'s me'") == "it's me"
22
+
23
+ def test_json5_multi_line_strings():
24
+ parser = JSONParser(json5_enabled=True)
25
+ assert parser.parse("'line1\\\nline2'") == "line1line2"
26
+ assert parser.parse("\"line1\\\r\nline2\"") == "line1line2"
27
+
28
+ def test_json5_hex_numbers():
29
+ parser = JSONParser(json5_enabled=True)
30
+ assert parser.parse("0x1f") == 31
31
+ assert parser.parse("-0x10") == -16 # Note: JSON5 spec says hex can have optional sign
32
+ # Actually checking spec: "Hexadecimal numbers ... may be prefixed with an optional plus or minus sign"
33
+ assert parser.parse("0XFF") == 255
34
+
35
+ def test_json5_special_numbers():
36
+ parser = JSONParser(json5_enabled=True)
37
+ assert parser.parse("Infinity") == float("inf")
38
+ assert parser.parse("-Infinity") == float("-inf")
39
+ assert math.isnan(parser.parse("NaN"))
40
+ assert parser.parse(".5") == 0.5
41
+ assert parser.parse("+42") == 42
42
+
43
+ def test_json5_case_insensitive_literals():
44
+ parser = JSONParser(json5_enabled=True)
45
+ assert parser.parse("True") is True
46
+ assert parser.parse("FALSE") is False
47
+ assert parser.parse("Null") is None
48
+
49
+ def test_json5_trailing_commas():
50
+ # Already supported but good to verify with JSON5 enabled
51
+ parser = JSONParser(json5_enabled=True)
52
+ assert parser.parse("[1, 2, 3,]") == [1, 2, 3]
53
+ assert parser.parse("{a: 1, b: 2,}") == {"a": 1, "b": 2}
54
+
55
+ def test_json5_whitespace():
56
+ parser = JSONParser(json5_enabled=True)
57
+ assert parser.parse("{\v\"a\"\f: 1\u00A0}") == {"a": 1}
@@ -84,8 +84,8 @@ def test_unicode_escape_complete():
84
84
  assert parser.parse('{"a":"\\u20AC"}') == {"a": "€"}
85
85
 
86
86
 
87
- def test_unicode_escape_incomplete_strict():
87
+ def test_incomplete_string_without_closing_quote():
88
88
  parser = JSONParser(strict=True)
89
- assert parser.parse('{"a":"\\u123"') == {"a": "\u1234"}
90
- assert parser.parse('{"a":"\\u"') == {"a": ""}
91
- assert parser.parse('{"a":"\\""') == {"a": ""}
89
+ # When string has no closing quote, incomplete escapes are handled
90
+ assert parser.parse('{"a":"\\u') == {"a": ""}
91
+ assert parser.parse('{"a":"\\') == {"a": ""}
@@ -1,168 +0,0 @@
1
- import json
2
- import re
3
-
4
-
5
- class JSONParser:
6
- def __init__(self, strict=True):
7
- self.strict = strict
8
- self.parsers = {
9
- " ": self.parse_space,
10
- "\r": self.parse_space,
11
- "\n": self.parse_space,
12
- "\t": self.parse_space,
13
- "[": self.parse_array,
14
- "{": self.parse_object,
15
- '"': self.parse_string,
16
- "t": self.parse_true,
17
- "f": self.parse_false,
18
- "n": self.parse_null,
19
- }
20
- for c in "0123456789.-":
21
- self.parsers[c] = self.parse_number
22
-
23
- self.last_parse_reminding = None
24
- self.on_extra_token = self.default_on_extra_token
25
-
26
- def default_on_extra_token(self, text, data, reminding):
27
- print(
28
- "Parsed JSON with extra tokens:",
29
- {"text": text, "data": data, "reminding": reminding},
30
- )
31
-
32
- def parse(self, s):
33
- if len(s) >= 1:
34
- try:
35
- return json.loads(s)
36
- except json.JSONDecodeError as e:
37
- data, reminding = self.parse_any(s, e)
38
- self.last_parse_reminding = reminding
39
- if self.on_extra_token and reminding:
40
- self.on_extra_token(s, data, reminding)
41
- return data
42
- else:
43
- return json.loads("{}")
44
-
45
- def parse_any(self, s, e):
46
- if not s:
47
- raise e
48
- parser = self.parsers.get(s[0])
49
- if not parser:
50
- raise e
51
- return parser(s, e)
52
-
53
- def parse_space(self, s, e):
54
- return self.parse_any(s.strip(), e)
55
-
56
- def parse_array(self, s, e):
57
- s = s[1:] # skip starting '['
58
- acc = []
59
- s = s.strip()
60
- while s:
61
- if s[0] == "]":
62
- s = s[1:] # skip ending ']'
63
- break
64
- res, s = self.parse_any(s, e)
65
- acc.append(res)
66
- s = s.strip()
67
- if s.startswith(","):
68
- s = s[1:]
69
- s = s.strip()
70
- return acc, s
71
-
72
- def parse_object(self, s, e):
73
- s = s[1:] # skip starting '{'
74
- acc = {}
75
- s = s.strip()
76
- while s:
77
- if s[0] == "}":
78
- s = s[1:] # skip ending '}'
79
- break
80
- key, s = self.parse_any(s, e)
81
- s = s.strip()
82
-
83
- if not s or s[0] == "}":
84
- acc[key] = None
85
- break
86
-
87
- if s[0] != ":":
88
- raise e # or handle this scenario as per your requirement
89
-
90
- s = s[1:] # skip ':'
91
- s = s.strip()
92
-
93
- if not s or s[0] in ",}":
94
- acc[key] = None
95
- if s.startswith(","):
96
- s = s[1:]
97
- break
98
-
99
- value, s = self.parse_any(s, e)
100
- acc[key] = value
101
- s = s.strip()
102
- if s.startswith(","):
103
- s = s[1:]
104
- s = s.strip()
105
- return acc, s
106
-
107
- def parse_string(self, s, e):
108
- incomplete_escape_regex = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3})?$")
109
- end = s.find('"', 1)
110
- while end != -1 and s[end - 1] == "\\": # Handle escaped quotes
111
- end = s.find('"', end + 1)
112
- if end == -1:
113
- # Incomplete string: handle it based on strict mode
114
- if not self.strict:
115
- # Check for incomplete escape sequences
116
- if incomplete_escape_regex.match(s[1:]):
117
- return s[1:], ""
118
- return s[1:], ""
119
- else:
120
- # Handle incomplete escape sequences
121
- if incomplete_escape_regex.match(s[1:]):
122
- return "", ""
123
- # Attempt to parse the string without incomplete escape sequences
124
- try:
125
- return json.loads(f'"{s[1:]}"'), ""
126
- except json.JSONDecodeError:
127
- return "", ""
128
- str_val = s[: end + 1]
129
- s = s[end + 1 :]
130
- if not self.strict:
131
- return str_val[1:-1], s # Remove surrounding quotes for strict mode
132
- return json.loads(str_val), s
133
-
134
- def parse_number(self, s, e):
135
- i = 0
136
- while i < len(s) and s[i] in "0123456789.-":
137
- i += 1
138
- num_str = s[:i]
139
- s = s[i:]
140
- if not num_str or num_str == "-" or num_str == ".":
141
- return num_str, ""
142
- try:
143
- if num_str.endswith("."):
144
- num = int(num_str[:-1])
145
- else:
146
- num = (
147
- float(num_str)
148
- if "." in num_str or "e" in num_str or "E" in num_str
149
- else int(num_str)
150
- )
151
- except ValueError:
152
- raise e
153
- return num, s
154
-
155
- def parse_true(self, s, e):
156
- if s.startswith("t") or s.startswith("T"):
157
- return True, s[4:]
158
- raise e
159
-
160
- def parse_false(self, s, e):
161
- if s.startswith("f") or s.startswith("F"):
162
- return False, s[5:]
163
- raise e
164
-
165
- def parse_null(self, s, e):
166
- if s.startswith("n"):
167
- return None, s[4:]
168
- raise e
File without changes
File without changes