partialjson 1.0.0__tar.gz → 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {partialjson-1.0.0 → partialjson-1.1.0}/PKG-INFO +23 -5
- {partialjson-1.0.0 → partialjson-1.1.0}/README.md +19 -4
- {partialjson-1.0.0 → partialjson-1.1.0}/partialjson/__init__.py +5 -2
- partialjson-1.1.0/partialjson/json5_parser.py +346 -0
- partialjson-1.1.0/partialjson/json_parser.py +233 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/partialjson.egg-info/PKG-INFO +23 -5
- {partialjson-1.0.0 → partialjson-1.1.0}/partialjson.egg-info/SOURCES.txt +3 -0
- partialjson-1.1.0/partialjson.egg-info/requires.txt +3 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/setup.py +1 -0
- partialjson-1.1.0/tests/test_json5.py +57 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/tests/test_parser.py +4 -4
- partialjson-1.0.0/partialjson/json_parser.py +0 -168
- {partialjson-1.0.0 → partialjson-1.1.0}/LICENSE +0 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/partialjson.egg-info/dependency_links.txt +0 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/partialjson.egg-info/top_level.txt +0 -0
- {partialjson-1.0.0 → partialjson-1.1.0}/setup.cfg +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: partialjson
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.0
|
|
4
4
|
Summary: Parse incomplete or partial json
|
|
5
5
|
Home-page: https://github.com/iw4p/partialjson
|
|
6
6
|
Author: Nima Akbarzadeh
|
|
@@ -8,6 +8,8 @@ Author-email: iw4p@protonmail.com
|
|
|
8
8
|
License: MIT
|
|
9
9
|
Description-Content-Type: text/markdown
|
|
10
10
|
License-File: LICENSE
|
|
11
|
+
Provides-Extra: json5
|
|
12
|
+
Requires-Dist: json5; extra == "json5"
|
|
11
13
|
Dynamic: author
|
|
12
14
|
Dynamic: author-email
|
|
13
15
|
Dynamic: description
|
|
@@ -15,6 +17,7 @@ Dynamic: description-content-type
|
|
|
15
17
|
Dynamic: home-page
|
|
16
18
|
Dynamic: license
|
|
17
19
|
Dynamic: license-file
|
|
20
|
+
Dynamic: provides-extra
|
|
18
21
|
Dynamic: summary
|
|
19
22
|
|
|
20
23
|
# PartialJson
|
|
@@ -34,18 +37,18 @@ Dynamic: summary
|
|
|
34
37
|
## Example
|
|
35
38
|
|
|
36
39
|
```python
|
|
37
|
-
from partialjson
|
|
40
|
+
from partialjson import JSONParser
|
|
38
41
|
parser = JSONParser()
|
|
39
42
|
|
|
40
43
|
incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
41
44
|
print(parser.parse(incomplete_json))
|
|
42
|
-
# {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
45
|
+
# {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
43
46
|
```
|
|
44
47
|
|
|
45
|
-
Problem with `\n`? strict
|
|
48
|
+
Problem with `\n`? Use `strict=False`:
|
|
46
49
|
|
|
47
50
|
```python
|
|
48
|
-
from partialjson
|
|
51
|
+
from partialjson import JSONParser
|
|
49
52
|
parser = JSONParser(strict=False)
|
|
50
53
|
|
|
51
54
|
incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
@@ -53,6 +56,21 @@ print(parser.parse(incomplete_json))
|
|
|
53
56
|
# {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
54
57
|
```
|
|
55
58
|
|
|
59
|
+
### JSON5 support
|
|
60
|
+
|
|
61
|
+
Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from partialjson import create_json5_parser
|
|
65
|
+
parser = create_json5_parser()
|
|
66
|
+
|
|
67
|
+
incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
|
|
68
|
+
print(parser.parse(incomplete_json5))
|
|
69
|
+
# {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
|
|
73
|
+
|
|
56
74
|
### Installation
|
|
57
75
|
|
|
58
76
|
```sh
|
|
@@ -15,18 +15,18 @@
|
|
|
15
15
|
## Example
|
|
16
16
|
|
|
17
17
|
```python
|
|
18
|
-
from partialjson
|
|
18
|
+
from partialjson import JSONParser
|
|
19
19
|
parser = JSONParser()
|
|
20
20
|
|
|
21
21
|
incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
22
22
|
print(parser.parse(incomplete_json))
|
|
23
|
-
# {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
23
|
+
# {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
24
24
|
```
|
|
25
25
|
|
|
26
|
-
Problem with `\n`? strict
|
|
26
|
+
Problem with `\n`? Use `strict=False`:
|
|
27
27
|
|
|
28
28
|
```python
|
|
29
|
-
from partialjson
|
|
29
|
+
from partialjson import JSONParser
|
|
30
30
|
parser = JSONParser(strict=False)
|
|
31
31
|
|
|
32
32
|
incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
@@ -34,6 +34,21 @@ print(parser.parse(incomplete_json))
|
|
|
34
34
|
# {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
35
35
|
```
|
|
36
36
|
|
|
37
|
+
### JSON5 support
|
|
38
|
+
|
|
39
|
+
Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
|
|
40
|
+
|
|
41
|
+
```python
|
|
42
|
+
from partialjson import create_json5_parser
|
|
43
|
+
parser = create_json5_parser()
|
|
44
|
+
|
|
45
|
+
incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
|
|
46
|
+
print(parser.parse(incomplete_json5))
|
|
47
|
+
# {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
|
|
51
|
+
|
|
37
52
|
### Installation
|
|
38
53
|
|
|
39
54
|
```sh
|
|
@@ -4,9 +4,10 @@ Partial Json.
|
|
|
4
4
|
Parsing ChatGPT JSON stream response — Partial and incomplete JSON parser python library for OpenAI
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
from .json_parser import JSONParser
|
|
7
|
+
from .json_parser import JSONParser, create_json_parser
|
|
8
|
+
from .json5_parser import create_json5_parser
|
|
8
9
|
|
|
9
|
-
__version__ = "1.
|
|
10
|
+
__version__ = "1.1.0"
|
|
10
11
|
__author__ = "Nima Akbarzadeh"
|
|
11
12
|
__author_email__ = "iw4p@protonmail.com"
|
|
12
13
|
__license__ = "MIT"
|
|
@@ -16,5 +17,7 @@ PYPI_SIMPLE_ENDPOINT: str = "https://pypi.org/project/partialjson"
|
|
|
16
17
|
|
|
17
18
|
__all__ = [
|
|
18
19
|
"JSONParser",
|
|
20
|
+
"create_json_parser",
|
|
21
|
+
"create_json5_parser",
|
|
19
22
|
"PYPI_SIMPLE_ENDPOINT",
|
|
20
23
|
]
|
|
@@ -0,0 +1,346 @@
|
|
|
1
|
+
"""JSON5 parser - extends JSON with comments, unquoted keys, single quotes, etc."""
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
try:
|
|
6
|
+
import json5
|
|
7
|
+
except ImportError:
|
|
8
|
+
json5 = None
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def create_json5_parser(strict=True, on_extra_token=None):
|
|
12
|
+
"""Create a JSON5 parser."""
|
|
13
|
+
return _JSON5Parser(strict=strict, on_extra_token=on_extra_token)
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _default_on_extra_token(text, data, reminding):
|
|
17
|
+
print("Parsed JSON with extra tokens:", {"text": text, "data": data, "reminding": reminding})
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
_INCOMPLETE_ESCAPE_REGEX = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3}|x[0-9a-fA-F]{0,1})?$")
|
|
21
|
+
_JSON5_WHITESPACE = "\v\f\u00A0\u2028\u2029\uFEFF"
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class _JSON5Parser:
|
|
25
|
+
"""JSON5 parser with comments, unquoted keys, single quotes, hex, Infinity, etc."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, strict=True, on_extra_token=None):
|
|
28
|
+
self.strict = strict
|
|
29
|
+
self.on_extra_token = on_extra_token or _default_on_extra_token
|
|
30
|
+
self.last_parse_reminding = None
|
|
31
|
+
self._parsers = self._build_parsers()
|
|
32
|
+
|
|
33
|
+
def _build_parsers(self):
|
|
34
|
+
parsers = {
|
|
35
|
+
" ": self._parse_space,
|
|
36
|
+
"\r": self._parse_space,
|
|
37
|
+
"\n": self._parse_space,
|
|
38
|
+
"\t": self._parse_space,
|
|
39
|
+
"[": self._parse_array,
|
|
40
|
+
"{": self._parse_object,
|
|
41
|
+
'"': self._parse_string,
|
|
42
|
+
"'": self._parse_string,
|
|
43
|
+
"t": self._parse_true,
|
|
44
|
+
"f": self._parse_false,
|
|
45
|
+
"n": self._parse_null,
|
|
46
|
+
"/": self._parse_space,
|
|
47
|
+
"+": self._parse_number,
|
|
48
|
+
"I": self._parse_number,
|
|
49
|
+
"N": self._parse_n_literal,
|
|
50
|
+
"T": self._parse_true,
|
|
51
|
+
"F": self._parse_false,
|
|
52
|
+
}
|
|
53
|
+
for c in _JSON5_WHITESPACE:
|
|
54
|
+
parsers[c] = self._parse_space
|
|
55
|
+
for c in "0123456789.-":
|
|
56
|
+
parsers[c] = self._parse_number
|
|
57
|
+
return parsers
|
|
58
|
+
|
|
59
|
+
def parse(self, s):
|
|
60
|
+
if len(s) >= 1:
|
|
61
|
+
if json5:
|
|
62
|
+
try:
|
|
63
|
+
return json5.loads(s)
|
|
64
|
+
except (json.JSONDecodeError, ValueError) as e:
|
|
65
|
+
data, reminding = self.parse_any(s, e)
|
|
66
|
+
self.last_parse_reminding = reminding
|
|
67
|
+
if self.on_extra_token and reminding:
|
|
68
|
+
self.on_extra_token(s, data, reminding)
|
|
69
|
+
return data
|
|
70
|
+
data, reminding = self.parse_any(s, json.JSONDecodeError("", "", 0))
|
|
71
|
+
self.last_parse_reminding = reminding
|
|
72
|
+
if self.on_extra_token and reminding:
|
|
73
|
+
self.on_extra_token(s, data, reminding)
|
|
74
|
+
return data
|
|
75
|
+
return json.loads("{}")
|
|
76
|
+
|
|
77
|
+
def parse_any(self, s, e):
|
|
78
|
+
if not s:
|
|
79
|
+
raise e
|
|
80
|
+
while s and self._is_space_or_comment_start(s):
|
|
81
|
+
s = self._parse_space(s, e)
|
|
82
|
+
if not s:
|
|
83
|
+
return None, ""
|
|
84
|
+
parser = self._parsers.get(s[0])
|
|
85
|
+
if not parser:
|
|
86
|
+
raise e
|
|
87
|
+
return parser(s, e)
|
|
88
|
+
|
|
89
|
+
def _is_space_or_comment_start(self, s):
|
|
90
|
+
if not s:
|
|
91
|
+
return False
|
|
92
|
+
c = s[0]
|
|
93
|
+
if c.isspace() or c in _JSON5_WHITESPACE:
|
|
94
|
+
return True
|
|
95
|
+
if s.startswith("//") or s.startswith("/*"):
|
|
96
|
+
return True
|
|
97
|
+
return False
|
|
98
|
+
|
|
99
|
+
def _parse_space(self, s, e):
|
|
100
|
+
i = 0
|
|
101
|
+
while i < len(s):
|
|
102
|
+
if s[i].isspace() or s[i] in _JSON5_WHITESPACE:
|
|
103
|
+
i += 1
|
|
104
|
+
elif s[i : i + 2] == "//":
|
|
105
|
+
i += 2
|
|
106
|
+
while i < len(s) and s[i] not in "\n\r\u2028\u2029":
|
|
107
|
+
i += 1
|
|
108
|
+
elif s[i : i + 2] == "/*":
|
|
109
|
+
i += 2
|
|
110
|
+
end = s.find("*/", i)
|
|
111
|
+
if end == -1:
|
|
112
|
+
return ""
|
|
113
|
+
i = end + 2
|
|
114
|
+
else:
|
|
115
|
+
break
|
|
116
|
+
return s[i:]
|
|
117
|
+
|
|
118
|
+
def _parse_array(self, s, e):
|
|
119
|
+
s = s[1:]
|
|
120
|
+
acc = []
|
|
121
|
+
while True:
|
|
122
|
+
while s and self._is_space_or_comment_start(s):
|
|
123
|
+
s = self._parse_space(s, e)
|
|
124
|
+
if not s:
|
|
125
|
+
break
|
|
126
|
+
if s[0] == "]":
|
|
127
|
+
s = s[1:]
|
|
128
|
+
break
|
|
129
|
+
res, s = self.parse_any(s, e)
|
|
130
|
+
acc.append(res)
|
|
131
|
+
while s and self._is_space_or_comment_start(s):
|
|
132
|
+
s = self._parse_space(s, e)
|
|
133
|
+
if s and s.startswith(","):
|
|
134
|
+
s = s[1:]
|
|
135
|
+
return acc, s
|
|
136
|
+
|
|
137
|
+
def _parse_object(self, s, e):
|
|
138
|
+
s = s[1:]
|
|
139
|
+
acc = {}
|
|
140
|
+
while True:
|
|
141
|
+
while s and self._is_space_or_comment_start(s):
|
|
142
|
+
s = self._parse_space(s, e)
|
|
143
|
+
if not s:
|
|
144
|
+
break
|
|
145
|
+
if s[0] == "}":
|
|
146
|
+
s = s[1:]
|
|
147
|
+
break
|
|
148
|
+
if s[0] not in '"\'':
|
|
149
|
+
key, s = self._parse_identifier(s, e)
|
|
150
|
+
if not key:
|
|
151
|
+
while s and self._is_space_or_comment_start(s):
|
|
152
|
+
s = self._parse_space(s, e)
|
|
153
|
+
if s and s[0] == "}":
|
|
154
|
+
s = s[1:]
|
|
155
|
+
break
|
|
156
|
+
else:
|
|
157
|
+
key, s = self.parse_any(s, e)
|
|
158
|
+
while s and self._is_space_or_comment_start(s):
|
|
159
|
+
s = self._parse_space(s, e)
|
|
160
|
+
if not s or s[0] == "}":
|
|
161
|
+
if key is not None:
|
|
162
|
+
acc[key] = None
|
|
163
|
+
if s and s[0] == "}":
|
|
164
|
+
s = s[1:]
|
|
165
|
+
break
|
|
166
|
+
if s[0] != ":":
|
|
167
|
+
if key is not None:
|
|
168
|
+
acc[key] = None
|
|
169
|
+
break
|
|
170
|
+
s = s[1:]
|
|
171
|
+
while s and self._is_space_or_comment_start(s):
|
|
172
|
+
s = self._parse_space(s, e)
|
|
173
|
+
if not s or s[0] in ",}":
|
|
174
|
+
acc[key] = None
|
|
175
|
+
if s and s.startswith(","):
|
|
176
|
+
s = s[1:]
|
|
177
|
+
elif s and s.startswith("}"):
|
|
178
|
+
s = s[1:]
|
|
179
|
+
break
|
|
180
|
+
while s and self._is_space_or_comment_start(s):
|
|
181
|
+
s = self._parse_space(s, e)
|
|
182
|
+
if s and (
|
|
183
|
+
s[0] in self._parsers
|
|
184
|
+
or s[0] in "/+IN"
|
|
185
|
+
or s[0] in _JSON5_WHITESPACE
|
|
186
|
+
):
|
|
187
|
+
value, s = self.parse_any(s, e)
|
|
188
|
+
acc[key] = value
|
|
189
|
+
else:
|
|
190
|
+
if key is not None:
|
|
191
|
+
acc[key] = None
|
|
192
|
+
break
|
|
193
|
+
while s and self._is_space_or_comment_start(s):
|
|
194
|
+
s = self._parse_space(s, e)
|
|
195
|
+
if s and s.startswith(","):
|
|
196
|
+
s = s[1:]
|
|
197
|
+
return acc, s
|
|
198
|
+
|
|
199
|
+
def _parse_identifier(self, s, e):
|
|
200
|
+
i = 0
|
|
201
|
+
while i < len(s) and (s[i].isalnum() or s[i] in "_$"):
|
|
202
|
+
i += 1
|
|
203
|
+
return s[:i], s[i:]
|
|
204
|
+
|
|
205
|
+
def _parse_string(self, s, e):
|
|
206
|
+
quote = s[0]
|
|
207
|
+
end = 1
|
|
208
|
+
while end < len(s):
|
|
209
|
+
if s[end] == "\\":
|
|
210
|
+
end += 2
|
|
211
|
+
continue
|
|
212
|
+
if s[end] == quote:
|
|
213
|
+
break
|
|
214
|
+
end += 1
|
|
215
|
+
|
|
216
|
+
if end >= len(s):
|
|
217
|
+
content = s[1:]
|
|
218
|
+
if not self.strict:
|
|
219
|
+
return content, ""
|
|
220
|
+
if _INCOMPLETE_ESCAPE_REGEX.match(content):
|
|
221
|
+
return "", ""
|
|
222
|
+
try:
|
|
223
|
+
if quote == "'":
|
|
224
|
+
return content, ""
|
|
225
|
+
return json.loads(f'"{content}"'), ""
|
|
226
|
+
except json.JSONDecodeError:
|
|
227
|
+
return "", ""
|
|
228
|
+
|
|
229
|
+
str_val = s[: end + 1]
|
|
230
|
+
remainder = s[end + 1 :]
|
|
231
|
+
|
|
232
|
+
if json5:
|
|
233
|
+
try:
|
|
234
|
+
return json5.loads(str_val), remainder
|
|
235
|
+
except Exception:
|
|
236
|
+
pass
|
|
237
|
+
|
|
238
|
+
decoded = str_val[1:-1]
|
|
239
|
+
decoded = re.sub(r"\\\n", "", decoded)
|
|
240
|
+
decoded = re.sub(r"\\\r\n", "", decoded)
|
|
241
|
+
|
|
242
|
+
def replace_hex(match):
|
|
243
|
+
return chr(int(match.group(1), 16))
|
|
244
|
+
|
|
245
|
+
decoded = re.sub(r"\\x([0-9a-fA-F]{2})", replace_hex, decoded)
|
|
246
|
+
|
|
247
|
+
if quote == "'":
|
|
248
|
+
decoded = decoded.replace('"', '\\"').replace("\\'", "'")
|
|
249
|
+
try:
|
|
250
|
+
return json.loads(f'"{decoded}"'), remainder
|
|
251
|
+
except Exception:
|
|
252
|
+
return decoded, remainder
|
|
253
|
+
if "\\x" in decoded or "\\\n" in str_val or "\\\r" in str_val:
|
|
254
|
+
return decoded, remainder
|
|
255
|
+
try:
|
|
256
|
+
return json.loads(str_val), remainder
|
|
257
|
+
except Exception:
|
|
258
|
+
return decoded, remainder
|
|
259
|
+
|
|
260
|
+
def _parse_number(self, s, e):
|
|
261
|
+
if s.startswith(("-0x", "-0X")):
|
|
262
|
+
i = 3
|
|
263
|
+
while i < len(s) and s[i] in "0123456789abcdefABCDEF":
|
|
264
|
+
i += 1
|
|
265
|
+
num_str = s[1:i]
|
|
266
|
+
remainder = s[i:]
|
|
267
|
+
if len(num_str) <= 2:
|
|
268
|
+
return s[:3], ""
|
|
269
|
+
return -int(num_str, 16), remainder
|
|
270
|
+
if s.startswith(("+0x", "+0X")):
|
|
271
|
+
i = 3
|
|
272
|
+
while i < len(s) and s[i] in "0123456789abcdefABCDEF":
|
|
273
|
+
i += 1
|
|
274
|
+
num_str = s[1:i]
|
|
275
|
+
remainder = s[i:]
|
|
276
|
+
if len(num_str) <= 2:
|
|
277
|
+
return s[:3], ""
|
|
278
|
+
return int(num_str, 16), remainder
|
|
279
|
+
if s.startswith(("0x", "0X")):
|
|
280
|
+
i = 2
|
|
281
|
+
while i < len(s) and s[i] in "0123456789abcdefABCDEF":
|
|
282
|
+
i += 1
|
|
283
|
+
num_str = s[:i]
|
|
284
|
+
remainder = s[i:]
|
|
285
|
+
if len(num_str) <= 2:
|
|
286
|
+
return num_str, ""
|
|
287
|
+
return int(num_str, 16), remainder
|
|
288
|
+
|
|
289
|
+
for literal, val in [("Infinity", float("inf")), ("NaN", float("nan"))]:
|
|
290
|
+
if s.startswith(literal):
|
|
291
|
+
return val, s[len(literal) :]
|
|
292
|
+
if s.startswith("+" + literal):
|
|
293
|
+
return val, s[len(literal) + 1 :]
|
|
294
|
+
if s.startswith("-" + literal):
|
|
295
|
+
return -val if literal == "Infinity" else val, s[len(literal) + 1 :]
|
|
296
|
+
|
|
297
|
+
if s.startswith(".") and len(s) > 1 and s[1].isdigit():
|
|
298
|
+
i = 1
|
|
299
|
+
while i < len(s) and s[i].isdigit():
|
|
300
|
+
i += 1
|
|
301
|
+
num_str = s[:i]
|
|
302
|
+
return float(num_str), s[i:]
|
|
303
|
+
|
|
304
|
+
if s.startswith("+"):
|
|
305
|
+
res, remainder = self._parse_number(s[1:], e)
|
|
306
|
+
return res, remainder
|
|
307
|
+
|
|
308
|
+
i = 0
|
|
309
|
+
while i < len(s) and s[i] in "0123456789.-":
|
|
310
|
+
i += 1
|
|
311
|
+
num_str = s[:i]
|
|
312
|
+
s = s[i:]
|
|
313
|
+
if not num_str or num_str == "-" or num_str == ".":
|
|
314
|
+
return num_str, ""
|
|
315
|
+
try:
|
|
316
|
+
if num_str.endswith("."):
|
|
317
|
+
num = int(num_str[:-1])
|
|
318
|
+
else:
|
|
319
|
+
num = (
|
|
320
|
+
float(num_str)
|
|
321
|
+
if "." in num_str or "e" in num_str or "E" in num_str
|
|
322
|
+
else int(num_str)
|
|
323
|
+
)
|
|
324
|
+
except ValueError:
|
|
325
|
+
raise e
|
|
326
|
+
return num, s
|
|
327
|
+
|
|
328
|
+
def _parse_n_literal(self, s, e):
|
|
329
|
+
if s.lower().startswith("nan"):
|
|
330
|
+
return self._parse_number(s, e)
|
|
331
|
+
return self._parse_null(s, e)
|
|
332
|
+
|
|
333
|
+
def _parse_true(self, s, e):
|
|
334
|
+
if s.lower().startswith("true"):
|
|
335
|
+
return True, s[4:]
|
|
336
|
+
raise e
|
|
337
|
+
|
|
338
|
+
def _parse_false(self, s, e):
|
|
339
|
+
if s.lower().startswith("false"):
|
|
340
|
+
return False, s[5:]
|
|
341
|
+
raise e
|
|
342
|
+
|
|
343
|
+
def _parse_null(self, s, e):
|
|
344
|
+
if s.lower().startswith("null"):
|
|
345
|
+
return None, s[4:]
|
|
346
|
+
raise e
|
|
@@ -0,0 +1,233 @@
|
|
|
1
|
+
"""Pure JSON parser - no JSON5 extensions."""
|
|
2
|
+
import json
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def create_json_parser(strict=True, on_extra_token=None):
|
|
7
|
+
"""Create a JSON parser (no JSON5 extensions)."""
|
|
8
|
+
return _JSONParser(strict=strict, on_extra_token=on_extra_token)
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _default_on_extra_token(text, data, reminding):
|
|
12
|
+
print("Parsed JSON with extra tokens:", {"text": text, "data": data, "reminding": reminding})
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
_INCOMPLETE_ESCAPE_REGEX = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3}|x[0-9a-fA-F]{0,1})?$")
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class _JSONParser:
|
|
19
|
+
"""Internal JSON-only parser implementation."""
|
|
20
|
+
|
|
21
|
+
def __init__(self, strict=True, on_extra_token=None):
|
|
22
|
+
self.strict = strict
|
|
23
|
+
self.on_extra_token = on_extra_token or _default_on_extra_token
|
|
24
|
+
self.last_parse_reminding = None
|
|
25
|
+
self._parsers = self._build_parsers()
|
|
26
|
+
|
|
27
|
+
def _build_parsers(self):
|
|
28
|
+
parsers = {
|
|
29
|
+
" ": self._parse_space,
|
|
30
|
+
"\r": self._parse_space,
|
|
31
|
+
"\n": self._parse_space,
|
|
32
|
+
"\t": self._parse_space,
|
|
33
|
+
"[": self._parse_array,
|
|
34
|
+
"{": self._parse_object,
|
|
35
|
+
'"': self._parse_string,
|
|
36
|
+
"t": self._parse_true,
|
|
37
|
+
"f": self._parse_false,
|
|
38
|
+
"n": self._parse_null,
|
|
39
|
+
}
|
|
40
|
+
for c in "0123456789.-":
|
|
41
|
+
parsers[c] = self._parse_number
|
|
42
|
+
return parsers
|
|
43
|
+
|
|
44
|
+
def parse(self, s):
|
|
45
|
+
if len(s) >= 1:
|
|
46
|
+
try:
|
|
47
|
+
return json.loads(s)
|
|
48
|
+
except (json.JSONDecodeError, ValueError) as e:
|
|
49
|
+
data, reminding = self.parse_any(s, e)
|
|
50
|
+
self.last_parse_reminding = reminding
|
|
51
|
+
if self.on_extra_token and reminding:
|
|
52
|
+
self.on_extra_token(s, data, reminding)
|
|
53
|
+
return data
|
|
54
|
+
return json.loads("{}")
|
|
55
|
+
|
|
56
|
+
def parse_any(self, s, e):
|
|
57
|
+
if not s:
|
|
58
|
+
raise e
|
|
59
|
+
while s and s[0].isspace():
|
|
60
|
+
s = self._parse_space(s, e)
|
|
61
|
+
if not s:
|
|
62
|
+
return None, ""
|
|
63
|
+
parser = self._parsers.get(s[0])
|
|
64
|
+
if not parser:
|
|
65
|
+
raise e
|
|
66
|
+
return parser(s, e)
|
|
67
|
+
|
|
68
|
+
def _parse_space(self, s, e):
|
|
69
|
+
i = 0
|
|
70
|
+
while i < len(s) and s[i].isspace():
|
|
71
|
+
i += 1
|
|
72
|
+
return s[i:]
|
|
73
|
+
|
|
74
|
+
def _parse_array(self, s, e):
|
|
75
|
+
s = s[1:]
|
|
76
|
+
acc = []
|
|
77
|
+
while True:
|
|
78
|
+
while s and s[0].isspace():
|
|
79
|
+
s = self._parse_space(s, e)
|
|
80
|
+
if not s:
|
|
81
|
+
break
|
|
82
|
+
if s[0] == "]":
|
|
83
|
+
s = s[1:]
|
|
84
|
+
break
|
|
85
|
+
res, s = self.parse_any(s, e)
|
|
86
|
+
acc.append(res)
|
|
87
|
+
while s and s[0].isspace():
|
|
88
|
+
s = self._parse_space(s, e)
|
|
89
|
+
if s and s.startswith(","):
|
|
90
|
+
s = s[1:]
|
|
91
|
+
return acc, s
|
|
92
|
+
|
|
93
|
+
def _parse_object(self, s, e):
|
|
94
|
+
s = s[1:]
|
|
95
|
+
acc = {}
|
|
96
|
+
while True:
|
|
97
|
+
while s and s[0].isspace():
|
|
98
|
+
s = self._parse_space(s, e)
|
|
99
|
+
if not s:
|
|
100
|
+
break
|
|
101
|
+
if s[0] == "}":
|
|
102
|
+
s = s[1:]
|
|
103
|
+
break
|
|
104
|
+
key, s = self.parse_any(s, e)
|
|
105
|
+
while s and s[0].isspace():
|
|
106
|
+
s = self._parse_space(s, e)
|
|
107
|
+
if not s or s[0] == "}":
|
|
108
|
+
if key is not None:
|
|
109
|
+
acc[key] = None
|
|
110
|
+
if s and s[0] == "}":
|
|
111
|
+
s = s[1:]
|
|
112
|
+
break
|
|
113
|
+
if s[0] != ":":
|
|
114
|
+
if key is not None:
|
|
115
|
+
acc[key] = None
|
|
116
|
+
break
|
|
117
|
+
s = s[1:]
|
|
118
|
+
while s and s[0].isspace():
|
|
119
|
+
s = self._parse_space(s, e)
|
|
120
|
+
if not s or s[0] in ",}":
|
|
121
|
+
acc[key] = None
|
|
122
|
+
if s and s.startswith(","):
|
|
123
|
+
s = s[1:]
|
|
124
|
+
elif s and s.startswith("}"):
|
|
125
|
+
s = s[1:]
|
|
126
|
+
break
|
|
127
|
+
if s and s[0] in self._parsers:
|
|
128
|
+
value, s = self.parse_any(s, e)
|
|
129
|
+
acc[key] = value
|
|
130
|
+
else:
|
|
131
|
+
if key is not None:
|
|
132
|
+
acc[key] = None
|
|
133
|
+
break
|
|
134
|
+
while s and s[0].isspace():
|
|
135
|
+
s = self._parse_space(s, e)
|
|
136
|
+
if s and s.startswith(","):
|
|
137
|
+
s = s[1:]
|
|
138
|
+
return acc, s
|
|
139
|
+
|
|
140
|
+
def _parse_string(self, s, e):
|
|
141
|
+
quote = s[0]
|
|
142
|
+
end = 1
|
|
143
|
+
while end < len(s):
|
|
144
|
+
if s[end] == "\\":
|
|
145
|
+
end += 2
|
|
146
|
+
continue
|
|
147
|
+
if s[end] == quote:
|
|
148
|
+
break
|
|
149
|
+
end += 1
|
|
150
|
+
|
|
151
|
+
if end >= len(s):
|
|
152
|
+
content = s[1:]
|
|
153
|
+
if not self.strict:
|
|
154
|
+
return content, ""
|
|
155
|
+
if _INCOMPLETE_ESCAPE_REGEX.match(content):
|
|
156
|
+
return "", ""
|
|
157
|
+
try:
|
|
158
|
+
return json.loads(f'"{content}"'), ""
|
|
159
|
+
except json.JSONDecodeError:
|
|
160
|
+
return "", ""
|
|
161
|
+
|
|
162
|
+
str_val = s[: end + 1]
|
|
163
|
+
remainder = s[end + 1 :]
|
|
164
|
+
if not self.strict:
|
|
165
|
+
return str_val[1:-1], remainder
|
|
166
|
+
return json.loads(str_val), remainder
|
|
167
|
+
|
|
168
|
+
def _parse_number(self, s, e):
|
|
169
|
+
i = 0
|
|
170
|
+
while i < len(s) and s[i] in "0123456789.-":
|
|
171
|
+
i += 1
|
|
172
|
+
num_str = s[:i]
|
|
173
|
+
s = s[i:]
|
|
174
|
+
if not num_str or num_str == "-" or num_str == ".":
|
|
175
|
+
return num_str, ""
|
|
176
|
+
try:
|
|
177
|
+
if num_str.endswith("."):
|
|
178
|
+
num = int(num_str[:-1])
|
|
179
|
+
else:
|
|
180
|
+
num = (
|
|
181
|
+
float(num_str)
|
|
182
|
+
if "." in num_str or "e" in num_str or "E" in num_str
|
|
183
|
+
else int(num_str)
|
|
184
|
+
)
|
|
185
|
+
except ValueError:
|
|
186
|
+
raise e
|
|
187
|
+
return num, s
|
|
188
|
+
|
|
189
|
+
def _parse_true(self, s, e):
|
|
190
|
+
if s.startswith("t") or s.startswith("T"):
|
|
191
|
+
return True, s[4:]
|
|
192
|
+
raise e
|
|
193
|
+
|
|
194
|
+
def _parse_false(self, s, e):
|
|
195
|
+
if s.startswith("f") or s.startswith("F"):
|
|
196
|
+
return False, s[5:]
|
|
197
|
+
raise e
|
|
198
|
+
|
|
199
|
+
def _parse_null(self, s, e):
|
|
200
|
+
if s.startswith("n"):
|
|
201
|
+
return None, s[4:]
|
|
202
|
+
raise e
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
# Backward compatibility
|
|
206
|
+
class JSONParser:
|
|
207
|
+
"""JSON parser. Use create_json_parser() or create_json5_parser() for new code."""
|
|
208
|
+
|
|
209
|
+
def __init__(self, strict=True, json5_enabled=False, on_extra_token=None):
|
|
210
|
+
if json5_enabled:
|
|
211
|
+
from .json5_parser import create_json5_parser
|
|
212
|
+
|
|
213
|
+
self._impl = create_json5_parser(strict=strict, on_extra_token=on_extra_token)
|
|
214
|
+
else:
|
|
215
|
+
self._impl = create_json_parser(strict=strict, on_extra_token=on_extra_token)
|
|
216
|
+
|
|
217
|
+
def parse(self, s):
|
|
218
|
+
return self._impl.parse(s)
|
|
219
|
+
|
|
220
|
+
def parse_any(self, s, e):
|
|
221
|
+
return self._impl.parse_any(s, e)
|
|
222
|
+
|
|
223
|
+
@property
|
|
224
|
+
def last_parse_reminding(self):
|
|
225
|
+
return getattr(self._impl, "last_parse_reminding", None)
|
|
226
|
+
|
|
227
|
+
@property
|
|
228
|
+
def on_extra_token(self):
|
|
229
|
+
return getattr(self._impl, "on_extra_token", None)
|
|
230
|
+
|
|
231
|
+
@on_extra_token.setter
|
|
232
|
+
def on_extra_token(self, value):
|
|
233
|
+
self._impl.on_extra_token = value
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: partialjson
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.1.0
|
|
4
4
|
Summary: Parse incomplete or partial json
|
|
5
5
|
Home-page: https://github.com/iw4p/partialjson
|
|
6
6
|
Author: Nima Akbarzadeh
|
|
@@ -8,6 +8,8 @@ Author-email: iw4p@protonmail.com
|
|
|
8
8
|
License: MIT
|
|
9
9
|
Description-Content-Type: text/markdown
|
|
10
10
|
License-File: LICENSE
|
|
11
|
+
Provides-Extra: json5
|
|
12
|
+
Requires-Dist: json5; extra == "json5"
|
|
11
13
|
Dynamic: author
|
|
12
14
|
Dynamic: author-email
|
|
13
15
|
Dynamic: description
|
|
@@ -15,6 +17,7 @@ Dynamic: description-content-type
|
|
|
15
17
|
Dynamic: home-page
|
|
16
18
|
Dynamic: license
|
|
17
19
|
Dynamic: license-file
|
|
20
|
+
Dynamic: provides-extra
|
|
18
21
|
Dynamic: summary
|
|
19
22
|
|
|
20
23
|
# PartialJson
|
|
@@ -34,18 +37,18 @@ Dynamic: summary
|
|
|
34
37
|
## Example
|
|
35
38
|
|
|
36
39
|
```python
|
|
37
|
-
from partialjson
|
|
40
|
+
from partialjson import JSONParser
|
|
38
41
|
parser = JSONParser()
|
|
39
42
|
|
|
40
43
|
incomplete_json = '{"name": "John Doe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
41
44
|
print(parser.parse(incomplete_json))
|
|
42
|
-
# {'name': 'John', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
45
|
+
# {'name': 'John Doe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
43
46
|
```
|
|
44
47
|
|
|
45
|
-
Problem with `\n`? strict
|
|
48
|
+
Problem with `\n`? Use `strict=False`:
|
|
46
49
|
|
|
47
50
|
```python
|
|
48
|
-
from partialjson
|
|
51
|
+
from partialjson import JSONParser
|
|
49
52
|
parser = JSONParser(strict=False)
|
|
50
53
|
|
|
51
54
|
incomplete_json = '{"name": "John\nDoe", "age": 30, "is_student": false, "courses": ["Math", "Science"'
|
|
@@ -53,6 +56,21 @@ print(parser.parse(incomplete_json))
|
|
|
53
56
|
# {'name': 'John\nDoe', 'age': 30, 'is_student': False, 'courses': ['Math', 'Science']}
|
|
54
57
|
```
|
|
55
58
|
|
|
59
|
+
### JSON5 support
|
|
60
|
+
|
|
61
|
+
Use `create_json5_parser` or `JSONParser(json5_enabled=True)` for JSON5 (comments, unquoted keys, single quotes, etc.):
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
from partialjson import create_json5_parser
|
|
65
|
+
parser = create_json5_parser()
|
|
66
|
+
|
|
67
|
+
incomplete_json5 = '{name: "Demo", version: 1.0, items: [1, 2, 3,]'
|
|
68
|
+
print(parser.parse(incomplete_json5))
|
|
69
|
+
# {'name': 'Demo', 'version': 1.0, 'items': [1, 2, 3]}
|
|
70
|
+
```
|
|
71
|
+
|
|
72
|
+
Install the optional `json5` dependency for full JSON5 support: `pip install partialjson[json5]`
|
|
73
|
+
|
|
56
74
|
### Installation
|
|
57
75
|
|
|
58
76
|
```sh
|
|
@@ -2,9 +2,12 @@ LICENSE
|
|
|
2
2
|
README.md
|
|
3
3
|
setup.py
|
|
4
4
|
partialjson/__init__.py
|
|
5
|
+
partialjson/json5_parser.py
|
|
5
6
|
partialjson/json_parser.py
|
|
6
7
|
partialjson.egg-info/PKG-INFO
|
|
7
8
|
partialjson.egg-info/SOURCES.txt
|
|
8
9
|
partialjson.egg-info/dependency_links.txt
|
|
10
|
+
partialjson.egg-info/requires.txt
|
|
9
11
|
partialjson.egg-info/top_level.txt
|
|
12
|
+
tests/test_json5.py
|
|
10
13
|
tests/test_parser.py
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
import pytest
|
|
2
|
+
import math
|
|
3
|
+
from partialjson.json_parser import JSONParser
|
|
4
|
+
|
|
5
|
+
def test_json5_comments():
|
|
6
|
+
parser = JSONParser(json5_enabled=True)
|
|
7
|
+
assert parser.parse("{// comment\n\"a\": 1}") == {"a": 1}
|
|
8
|
+
assert parser.parse("{/* multi-line\n comment */\"a\": 1}") == {"a": 1}
|
|
9
|
+
assert parser.parse("{/* incomplete comment") == {}
|
|
10
|
+
|
|
11
|
+
def test_json5_unquoted_keys():
|
|
12
|
+
parser = JSONParser(json5_enabled=True)
|
|
13
|
+
assert parser.parse("{a: 1, b: 2}") == {"a": 1, "b": 2}
|
|
14
|
+
assert parser.parse("{_foo: \"bar\", $baz: 3}") == {"_foo": "bar", "$baz": 3}
|
|
15
|
+
assert parser.parse("{a: 1, ") == {"a": 1}
|
|
16
|
+
|
|
17
|
+
def test_json5_single_quotes():
|
|
18
|
+
parser = JSONParser(json5_enabled=True)
|
|
19
|
+
assert parser.parse("'hello'") == "hello"
|
|
20
|
+
assert parser.parse("{'a': 'b'}") == {"a": "b"}
|
|
21
|
+
assert parser.parse("'it\\'s me'") == "it's me"
|
|
22
|
+
|
|
23
|
+
def test_json5_multi_line_strings():
|
|
24
|
+
parser = JSONParser(json5_enabled=True)
|
|
25
|
+
assert parser.parse("'line1\\\nline2'") == "line1line2"
|
|
26
|
+
assert parser.parse("\"line1\\\r\nline2\"") == "line1line2"
|
|
27
|
+
|
|
28
|
+
def test_json5_hex_numbers():
|
|
29
|
+
parser = JSONParser(json5_enabled=True)
|
|
30
|
+
assert parser.parse("0x1f") == 31
|
|
31
|
+
assert parser.parse("-0x10") == -16 # Note: JSON5 spec says hex can have optional sign
|
|
32
|
+
# Actually checking spec: "Hexadecimal numbers ... may be prefixed with an optional plus or minus sign"
|
|
33
|
+
assert parser.parse("0XFF") == 255
|
|
34
|
+
|
|
35
|
+
def test_json5_special_numbers():
|
|
36
|
+
parser = JSONParser(json5_enabled=True)
|
|
37
|
+
assert parser.parse("Infinity") == float("inf")
|
|
38
|
+
assert parser.parse("-Infinity") == float("-inf")
|
|
39
|
+
assert math.isnan(parser.parse("NaN"))
|
|
40
|
+
assert parser.parse(".5") == 0.5
|
|
41
|
+
assert parser.parse("+42") == 42
|
|
42
|
+
|
|
43
|
+
def test_json5_case_insensitive_literals():
|
|
44
|
+
parser = JSONParser(json5_enabled=True)
|
|
45
|
+
assert parser.parse("True") is True
|
|
46
|
+
assert parser.parse("FALSE") is False
|
|
47
|
+
assert parser.parse("Null") is None
|
|
48
|
+
|
|
49
|
+
def test_json5_trailing_commas():
|
|
50
|
+
# Already supported but good to verify with JSON5 enabled
|
|
51
|
+
parser = JSONParser(json5_enabled=True)
|
|
52
|
+
assert parser.parse("[1, 2, 3,]") == [1, 2, 3]
|
|
53
|
+
assert parser.parse("{a: 1, b: 2,}") == {"a": 1, "b": 2}
|
|
54
|
+
|
|
55
|
+
def test_json5_whitespace():
|
|
56
|
+
parser = JSONParser(json5_enabled=True)
|
|
57
|
+
assert parser.parse("{\v\"a\"\f: 1\u00A0}") == {"a": 1}
|
|
@@ -84,8 +84,8 @@ def test_unicode_escape_complete():
|
|
|
84
84
|
assert parser.parse('{"a":"\\u20AC"}') == {"a": "€"}
|
|
85
85
|
|
|
86
86
|
|
|
87
|
-
def
|
|
87
|
+
def test_incomplete_string_without_closing_quote():
|
|
88
88
|
parser = JSONParser(strict=True)
|
|
89
|
-
|
|
90
|
-
assert parser.parse('{"a":"\\u
|
|
91
|
-
assert parser.parse('{"a":"\\
|
|
89
|
+
# When string has no closing quote, incomplete escapes are handled
|
|
90
|
+
assert parser.parse('{"a":"\\u') == {"a": ""}
|
|
91
|
+
assert parser.parse('{"a":"\\') == {"a": ""}
|
|
@@ -1,168 +0,0 @@
|
|
|
1
|
-
import json
|
|
2
|
-
import re
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
class JSONParser:
|
|
6
|
-
def __init__(self, strict=True):
|
|
7
|
-
self.strict = strict
|
|
8
|
-
self.parsers = {
|
|
9
|
-
" ": self.parse_space,
|
|
10
|
-
"\r": self.parse_space,
|
|
11
|
-
"\n": self.parse_space,
|
|
12
|
-
"\t": self.parse_space,
|
|
13
|
-
"[": self.parse_array,
|
|
14
|
-
"{": self.parse_object,
|
|
15
|
-
'"': self.parse_string,
|
|
16
|
-
"t": self.parse_true,
|
|
17
|
-
"f": self.parse_false,
|
|
18
|
-
"n": self.parse_null,
|
|
19
|
-
}
|
|
20
|
-
for c in "0123456789.-":
|
|
21
|
-
self.parsers[c] = self.parse_number
|
|
22
|
-
|
|
23
|
-
self.last_parse_reminding = None
|
|
24
|
-
self.on_extra_token = self.default_on_extra_token
|
|
25
|
-
|
|
26
|
-
def default_on_extra_token(self, text, data, reminding):
|
|
27
|
-
print(
|
|
28
|
-
"Parsed JSON with extra tokens:",
|
|
29
|
-
{"text": text, "data": data, "reminding": reminding},
|
|
30
|
-
)
|
|
31
|
-
|
|
32
|
-
def parse(self, s):
|
|
33
|
-
if len(s) >= 1:
|
|
34
|
-
try:
|
|
35
|
-
return json.loads(s)
|
|
36
|
-
except json.JSONDecodeError as e:
|
|
37
|
-
data, reminding = self.parse_any(s, e)
|
|
38
|
-
self.last_parse_reminding = reminding
|
|
39
|
-
if self.on_extra_token and reminding:
|
|
40
|
-
self.on_extra_token(s, data, reminding)
|
|
41
|
-
return data
|
|
42
|
-
else:
|
|
43
|
-
return json.loads("{}")
|
|
44
|
-
|
|
45
|
-
def parse_any(self, s, e):
|
|
46
|
-
if not s:
|
|
47
|
-
raise e
|
|
48
|
-
parser = self.parsers.get(s[0])
|
|
49
|
-
if not parser:
|
|
50
|
-
raise e
|
|
51
|
-
return parser(s, e)
|
|
52
|
-
|
|
53
|
-
def parse_space(self, s, e):
|
|
54
|
-
return self.parse_any(s.strip(), e)
|
|
55
|
-
|
|
56
|
-
def parse_array(self, s, e):
|
|
57
|
-
s = s[1:] # skip starting '['
|
|
58
|
-
acc = []
|
|
59
|
-
s = s.strip()
|
|
60
|
-
while s:
|
|
61
|
-
if s[0] == "]":
|
|
62
|
-
s = s[1:] # skip ending ']'
|
|
63
|
-
break
|
|
64
|
-
res, s = self.parse_any(s, e)
|
|
65
|
-
acc.append(res)
|
|
66
|
-
s = s.strip()
|
|
67
|
-
if s.startswith(","):
|
|
68
|
-
s = s[1:]
|
|
69
|
-
s = s.strip()
|
|
70
|
-
return acc, s
|
|
71
|
-
|
|
72
|
-
def parse_object(self, s, e):
|
|
73
|
-
s = s[1:] # skip starting '{'
|
|
74
|
-
acc = {}
|
|
75
|
-
s = s.strip()
|
|
76
|
-
while s:
|
|
77
|
-
if s[0] == "}":
|
|
78
|
-
s = s[1:] # skip ending '}'
|
|
79
|
-
break
|
|
80
|
-
key, s = self.parse_any(s, e)
|
|
81
|
-
s = s.strip()
|
|
82
|
-
|
|
83
|
-
if not s or s[0] == "}":
|
|
84
|
-
acc[key] = None
|
|
85
|
-
break
|
|
86
|
-
|
|
87
|
-
if s[0] != ":":
|
|
88
|
-
raise e # or handle this scenario as per your requirement
|
|
89
|
-
|
|
90
|
-
s = s[1:] # skip ':'
|
|
91
|
-
s = s.strip()
|
|
92
|
-
|
|
93
|
-
if not s or s[0] in ",}":
|
|
94
|
-
acc[key] = None
|
|
95
|
-
if s.startswith(","):
|
|
96
|
-
s = s[1:]
|
|
97
|
-
break
|
|
98
|
-
|
|
99
|
-
value, s = self.parse_any(s, e)
|
|
100
|
-
acc[key] = value
|
|
101
|
-
s = s.strip()
|
|
102
|
-
if s.startswith(","):
|
|
103
|
-
s = s[1:]
|
|
104
|
-
s = s.strip()
|
|
105
|
-
return acc, s
|
|
106
|
-
|
|
107
|
-
def parse_string(self, s, e):
|
|
108
|
-
incomplete_escape_regex = re.compile(r"^\\(?:u[0-9a-fA-F]{0,3})?$")
|
|
109
|
-
end = s.find('"', 1)
|
|
110
|
-
while end != -1 and s[end - 1] == "\\": # Handle escaped quotes
|
|
111
|
-
end = s.find('"', end + 1)
|
|
112
|
-
if end == -1:
|
|
113
|
-
# Incomplete string: handle it based on strict mode
|
|
114
|
-
if not self.strict:
|
|
115
|
-
# Check for incomplete escape sequences
|
|
116
|
-
if incomplete_escape_regex.match(s[1:]):
|
|
117
|
-
return s[1:], ""
|
|
118
|
-
return s[1:], ""
|
|
119
|
-
else:
|
|
120
|
-
# Handle incomplete escape sequences
|
|
121
|
-
if incomplete_escape_regex.match(s[1:]):
|
|
122
|
-
return "", ""
|
|
123
|
-
# Attempt to parse the string without incomplete escape sequences
|
|
124
|
-
try:
|
|
125
|
-
return json.loads(f'"{s[1:]}"'), ""
|
|
126
|
-
except json.JSONDecodeError:
|
|
127
|
-
return "", ""
|
|
128
|
-
str_val = s[: end + 1]
|
|
129
|
-
s = s[end + 1 :]
|
|
130
|
-
if not self.strict:
|
|
131
|
-
return str_val[1:-1], s # Remove surrounding quotes for strict mode
|
|
132
|
-
return json.loads(str_val), s
|
|
133
|
-
|
|
134
|
-
def parse_number(self, s, e):
|
|
135
|
-
i = 0
|
|
136
|
-
while i < len(s) and s[i] in "0123456789.-":
|
|
137
|
-
i += 1
|
|
138
|
-
num_str = s[:i]
|
|
139
|
-
s = s[i:]
|
|
140
|
-
if not num_str or num_str == "-" or num_str == ".":
|
|
141
|
-
return num_str, ""
|
|
142
|
-
try:
|
|
143
|
-
if num_str.endswith("."):
|
|
144
|
-
num = int(num_str[:-1])
|
|
145
|
-
else:
|
|
146
|
-
num = (
|
|
147
|
-
float(num_str)
|
|
148
|
-
if "." in num_str or "e" in num_str or "E" in num_str
|
|
149
|
-
else int(num_str)
|
|
150
|
-
)
|
|
151
|
-
except ValueError:
|
|
152
|
-
raise e
|
|
153
|
-
return num, s
|
|
154
|
-
|
|
155
|
-
def parse_true(self, s, e):
|
|
156
|
-
if s.startswith("t") or s.startswith("T"):
|
|
157
|
-
return True, s[4:]
|
|
158
|
-
raise e
|
|
159
|
-
|
|
160
|
-
def parse_false(self, s, e):
|
|
161
|
-
if s.startswith("f") or s.startswith("F"):
|
|
162
|
-
return False, s[5:]
|
|
163
|
-
raise e
|
|
164
|
-
|
|
165
|
-
def parse_null(self, s, e):
|
|
166
|
-
if s.startswith("n"):
|
|
167
|
-
return None, s[4:]
|
|
168
|
-
raise e
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|