links-notation 0.21.2__tar.gz → 0.22.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {links_notation-0.21.2/links_notation.egg-info → links_notation-0.22.0}/PKG-INFO +37 -4
- {links_notation-0.21.2 → links_notation-0.22.0}/README.md +36 -3
- {links_notation-0.21.2 → links_notation-0.22.0}/README.ru.md +37 -3
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/__init__.py +3 -1
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/parser.py +286 -102
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/stream_parser.py +26 -8
- {links_notation-0.21.2 → links_notation-0.22.0/links_notation.egg-info}/PKG-INFO +37 -4
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/SOURCES.txt +1 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/pyproject.toml +1 -1
- links_notation-0.22.0/tests/test_nesting_limit.py +137 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/LICENSE +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/MANIFEST.in +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/comments.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/format_config.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/formatter.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/link.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/quotes.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/dependency_links.txt +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/requires.txt +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/top_level.txt +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/setup.cfg +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_api.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_comments.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_edge_case_parser.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_empty_reference.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_format_config.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indentation_consistency.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indented_id_nested_values.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indented_id_syntax.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_link.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_links_group.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_mixed_indentation_modes.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multi_quote_parser.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multiline_parser.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multiline_quoted_string.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_indentation.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_parser.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_self_reference.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_single_line_parser.py +0 -0
- {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_stream_parser.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: links-notation
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.22.0
|
|
4
4
|
Summary: Python implementation of the Links Notation parser
|
|
5
5
|
Author-email: "Link.Foundation" <drakonard@gmail.com>
|
|
6
6
|
License-Expression: Unlicense
|
|
@@ -140,9 +140,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
|
|
|
140
140
|
|
|
141
141
|
The main parser class for Links Notation.
|
|
142
142
|
|
|
143
|
-
- `__init__(
|
|
144
|
-
a `#` is an ordinary character instead
|
|
145
|
-
|
|
143
|
+
- `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
|
|
144
|
+
Create a parser; with `comments=False` a `#` is an ordinary character instead
|
|
145
|
+
of the start of a comment
|
|
146
|
+
- `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
|
|
147
|
+
objects; raises `ParseError` when the text does not parse or nests links
|
|
148
|
+
deeper than `max_depth`
|
|
149
|
+
- `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
|
|
150
|
+
implementation
|
|
151
|
+
|
|
152
|
+
`max_depth` is how deep links may nest (default: 64). Every parenthesized group
|
|
153
|
+
and every indentation level is one level, and the lines of a document start at
|
|
154
|
+
level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
|
|
155
|
+
group on an indented line are refused. A document nested deeper is refused with
|
|
156
|
+
a `ParseError` rather than recursed into until Python's recursion limit is hit.
|
|
157
|
+
|
|
158
|
+
### ParseError
|
|
159
|
+
|
|
160
|
+
Raised when parsing fails. When a document nests links deeper than
|
|
161
|
+
`max_depth`, it points at the group or the line that is one level too deep:
|
|
162
|
+
|
|
163
|
+
```text
|
|
164
|
+
Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
|
|
165
|
+
1 | ((((a))))
|
|
166
|
+
| ^
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
- `max_depth`: The deepest nesting the parser accepts, when the document nests
|
|
170
|
+
deeper; `None` for any other error
|
|
171
|
+
- `line`, `column`: Where the offending group or line starts, counted from 1
|
|
172
|
+
- `offset`: The same position as a character offset from the start of the
|
|
173
|
+
document
|
|
174
|
+
- `line_text`: The offending line, as written
|
|
175
|
+
|
|
176
|
+
`StreamParser` reports the same error as a `StreamParseError` whose `error` is
|
|
177
|
+
the `ParseError` and whose `line`, `column` and `offset` are counted from the
|
|
178
|
+
start of the stream.
|
|
146
179
|
|
|
147
180
|
### Link
|
|
148
181
|
|
|
@@ -102,9 +102,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
|
|
|
102
102
|
|
|
103
103
|
The main parser class for Links Notation.
|
|
104
104
|
|
|
105
|
-
- `__init__(
|
|
106
|
-
a `#` is an ordinary character instead
|
|
107
|
-
|
|
105
|
+
- `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
|
|
106
|
+
Create a parser; with `comments=False` a `#` is an ordinary character instead
|
|
107
|
+
of the start of a comment
|
|
108
|
+
- `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
|
|
109
|
+
objects; raises `ParseError` when the text does not parse or nests links
|
|
110
|
+
deeper than `max_depth`
|
|
111
|
+
- `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
|
|
112
|
+
implementation
|
|
113
|
+
|
|
114
|
+
`max_depth` is how deep links may nest (default: 64). Every parenthesized group
|
|
115
|
+
and every indentation level is one level, and the lines of a document start at
|
|
116
|
+
level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
|
|
117
|
+
group on an indented line are refused. A document nested deeper is refused with
|
|
118
|
+
a `ParseError` rather than recursed into until Python's recursion limit is hit.
|
|
119
|
+
|
|
120
|
+
### ParseError
|
|
121
|
+
|
|
122
|
+
Raised when parsing fails. When a document nests links deeper than
|
|
123
|
+
`max_depth`, it points at the group or the line that is one level too deep:
|
|
124
|
+
|
|
125
|
+
```text
|
|
126
|
+
Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
|
|
127
|
+
1 | ((((a))))
|
|
128
|
+
| ^
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
- `max_depth`: The deepest nesting the parser accepts, when the document nests
|
|
132
|
+
deeper; `None` for any other error
|
|
133
|
+
- `line`, `column`: Where the offending group or line starts, counted from 1
|
|
134
|
+
- `offset`: The same position as a character offset from the start of the
|
|
135
|
+
document
|
|
136
|
+
- `line_text`: The offending line, as written
|
|
137
|
+
|
|
138
|
+
`StreamParser` reports the same error as a `StreamParseError` whose `error` is
|
|
139
|
+
the `ParseError` and whose `line`, `column` and `offset` are counted from the
|
|
140
|
+
start of the stream.
|
|
108
141
|
|
|
109
142
|
### Link
|
|
110
143
|
|
|
@@ -103,9 +103,43 @@ for link in parse_chunks(["одна свя", "зь\nвторая связь"]):
|
|
|
103
103
|
|
|
104
104
|
Основной класс парсера для Links Notation.
|
|
105
105
|
|
|
106
|
-
- `__init__(
|
|
107
|
-
`#` — обычный символ, а не начало
|
|
108
|
-
|
|
106
|
+
- `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
|
|
107
|
+
создать парсер; при `comments=False` `#` — обычный символ, а не начало
|
|
108
|
+
комментария
|
|
109
|
+
- `parse(input_text: str) -> List[Link]`: Парсинг текста Links Notation в
|
|
110
|
+
объекты Link; выбрасывает `ParseError`, если текст не удалось разобрать или
|
|
111
|
+
связи в нем вложены глубже `max_depth`
|
|
112
|
+
- `DEFAULT_MAX_DEPTH`: значение `max_depth` по умолчанию, 64, одинаковое во всех
|
|
113
|
+
реализациях
|
|
114
|
+
|
|
115
|
+
`max_depth` — насколько глубоко могут вкладываться связи (по умолчанию 64).
|
|
116
|
+
Каждая группа в скобках и каждый уровень отступа — это один уровень, а строки
|
|
117
|
+
документа находятся на уровне 0, поэтому при `max_depth=1` `(a)` принимается,
|
|
118
|
+
а `((a))`, `(a (b))` и группа на строке с отступом отклоняются. Документ с более
|
|
119
|
+
глубокой вложенностью отклоняется с `ParseError`, а не разбирается рекурсивно,
|
|
120
|
+
пока не будет достигнут предел рекурсии Python.
|
|
121
|
+
|
|
122
|
+
### ParseError
|
|
123
|
+
|
|
124
|
+
Выбрасывается при ошибке разбора. Когда связи в документе вложены глубже
|
|
125
|
+
`max_depth`, указывает на группу или строку, которая на уровень глубже
|
|
126
|
+
допустимого:
|
|
127
|
+
|
|
128
|
+
```text
|
|
129
|
+
Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
|
|
130
|
+
1 | ((((a))))
|
|
131
|
+
| ^
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
- `max_depth`: наибольшая допустимая глубина вложенности, если документ вложен
|
|
135
|
+
глубже; `None` для любой другой ошибки
|
|
136
|
+
- `line`, `column`: где начинается группа или строка, считая с 1
|
|
137
|
+
- `offset`: та же позиция как смещение в символах от начала документа
|
|
138
|
+
- `line_text`: строка с ошибкой в том виде, в каком она написана
|
|
139
|
+
|
|
140
|
+
`StreamParser` сообщает о той же ошибке через `StreamParseError`, у которого
|
|
141
|
+
`error` — это `ParseError`, а `line`, `column` и `offset` отсчитываются от
|
|
142
|
+
начала потока.
|
|
109
143
|
|
|
110
144
|
### Link
|
|
111
145
|
|
|
@@ -14,7 +14,7 @@ from .comments import strip_comments
|
|
|
14
14
|
from .format_config import FormatConfig
|
|
15
15
|
from .formatter import format_links
|
|
16
16
|
from .link import Link
|
|
17
|
-
from .parser import Parser
|
|
17
|
+
from .parser import DEFAULT_MAX_DEPTH, ParseError, Parser
|
|
18
18
|
from .stream_parser import StreamParseError, StreamParser, StreamPosition, parse_async_chunks, parse_chunks
|
|
19
19
|
|
|
20
20
|
|
|
@@ -47,7 +47,9 @@ def _read_version() -> str:
|
|
|
47
47
|
__version__ = _read_version()
|
|
48
48
|
|
|
49
49
|
__all__ = [
|
|
50
|
+
"DEFAULT_MAX_DEPTH",
|
|
50
51
|
"Link",
|
|
52
|
+
"ParseError",
|
|
51
53
|
"Parser",
|
|
52
54
|
"StreamParseError",
|
|
53
55
|
"StreamParser",
|
|
@@ -5,15 +5,135 @@ This module provides parsing functionality for Links Notation (Lino),
|
|
|
5
5
|
converting text into structured Link objects.
|
|
6
6
|
"""
|
|
7
7
|
|
|
8
|
-
|
|
8
|
+
import re
|
|
9
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
9
10
|
|
|
10
11
|
from .comments import strip_comments
|
|
11
12
|
from .link import Link
|
|
12
13
|
from .quotes import _parse_quoted_string_at
|
|
13
14
|
|
|
15
|
+
#: How deep links may nest unless a parser is told otherwise: every
|
|
16
|
+
#: parenthesized group and every indentation level is one level, and the lines
|
|
17
|
+
#: of a document are at level 0. Every implementation shares this default.
|
|
18
|
+
DEFAULT_MAX_DEPTH = 64
|
|
19
|
+
|
|
20
|
+
#: The number of characters a quoted line is cut down to.
|
|
21
|
+
_QUOTED_LINE_WIDTH = 80
|
|
22
|
+
|
|
23
|
+
#: What a message writes in place of the part of a long line it left out.
|
|
24
|
+
_ELLIPSIS = "..."
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
#: The characters that can end a line or open a quoted string.
|
|
28
|
+
_LINE_BREAK_OR_QUOTE = re.compile("[\n\"'`]")
|
|
29
|
+
|
|
30
|
+
#: The characters that can separate an id from its values or open a quoted string.
|
|
31
|
+
_COLON_OR_QUOTE = re.compile("[:\"'`]")
|
|
32
|
+
|
|
33
|
+
#: A run of opening parentheses, a run of closing ones, or a character that can
|
|
34
|
+
#: open a quoted string.
|
|
35
|
+
_PAREN_RUN_OR_QUOTE = re.compile("\\(+|\\)+|[\"'`]")
|
|
36
|
+
|
|
14
37
|
|
|
15
38
|
class ParseError(Exception):
|
|
16
|
-
"""
|
|
39
|
+
"""
|
|
40
|
+
Exception raised when parsing fails.
|
|
41
|
+
|
|
42
|
+
An error that points at a place in the document carries that place:
|
|
43
|
+
``offset`` (characters from the start of the document), ``line`` and
|
|
44
|
+
``column`` (both counted from 1), and ``line_text`` (the offending line as
|
|
45
|
+
written, without its line ending); for any other error they are None.
|
|
46
|
+
|
|
47
|
+
A document nested deeper than the parser's ``max_depth`` is refused with
|
|
48
|
+
this error too; then ``max_depth`` says how deep the nesting may go. It is
|
|
49
|
+
None for any other error.
|
|
50
|
+
"""
|
|
51
|
+
|
|
52
|
+
def __init__(
|
|
53
|
+
self,
|
|
54
|
+
message: str = "",
|
|
55
|
+
*,
|
|
56
|
+
offset: Optional[int] = None,
|
|
57
|
+
line: Optional[int] = None,
|
|
58
|
+
column: Optional[int] = None,
|
|
59
|
+
line_text: Optional[str] = None,
|
|
60
|
+
max_depth: Optional[int] = None,
|
|
61
|
+
):
|
|
62
|
+
super().__init__(message)
|
|
63
|
+
self.offset = offset
|
|
64
|
+
self.line = line
|
|
65
|
+
self.column = column
|
|
66
|
+
self.line_text = line_text
|
|
67
|
+
self.max_depth = max_depth
|
|
68
|
+
|
|
69
|
+
@classmethod
|
|
70
|
+
def nesting_too_deep(cls, document: str, offset: int, max_depth: int) -> "ParseError":
|
|
71
|
+
"""The error for a document nested deeper than ``max_depth`` at ``offset``."""
|
|
72
|
+
line, column, line_text = _locate(document, offset)
|
|
73
|
+
summary = f"line {line}, column {column}: nesting depth exceeds the maximum of {max_depth}"
|
|
74
|
+
return cls(
|
|
75
|
+
f"Nesting too deep at {summary}\n{_quote(line, line_text, column)}",
|
|
76
|
+
offset=offset,
|
|
77
|
+
line=line,
|
|
78
|
+
column=column,
|
|
79
|
+
line_text=line_text,
|
|
80
|
+
max_depth=max_depth,
|
|
81
|
+
)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def _locate(document: str, offset: int) -> Tuple[int, int, str]:
|
|
85
|
+
"""
|
|
86
|
+
The line and column ``offset`` falls on, both counted from 1, and that line
|
|
87
|
+
without its line ending. CR, LF and CRLF all end a line.
|
|
88
|
+
"""
|
|
89
|
+
offset = max(0, min(offset, len(document)))
|
|
90
|
+
line = 1
|
|
91
|
+
line_start = 0
|
|
92
|
+
cursor = 0
|
|
93
|
+
while cursor < offset:
|
|
94
|
+
character = document[cursor]
|
|
95
|
+
cursor += 1
|
|
96
|
+
if character == "\r":
|
|
97
|
+
line += 1
|
|
98
|
+
if cursor < offset and document[cursor] == "\n":
|
|
99
|
+
cursor += 1
|
|
100
|
+
line_start = cursor
|
|
101
|
+
elif character == "\n":
|
|
102
|
+
line += 1
|
|
103
|
+
line_start = cursor
|
|
104
|
+
line_end = len(document)
|
|
105
|
+
for ending in ("\r", "\n"):
|
|
106
|
+
found = document.find(ending, line_start)
|
|
107
|
+
if 0 <= found < line_end:
|
|
108
|
+
line_end = found
|
|
109
|
+
return line, offset - line_start + 1, document[line_start:line_end]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _quote(number: int, line_text: str, column: int) -> str:
|
|
113
|
+
"""
|
|
114
|
+
The offending line with a caret under the offending column, quoted the way
|
|
115
|
+
a compiler quotes source. A long line is shown as a window around the caret.
|
|
116
|
+
"""
|
|
117
|
+
quoted, column = _window_around(line_text, column)
|
|
118
|
+
gutter = " " * len(str(number))
|
|
119
|
+
return f"{number} | {quoted}\n{gutter} | {' ' * (column - 1)}^"
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _window_around(line_text: str, column: int) -> Tuple[str, int]:
|
|
123
|
+
"""
|
|
124
|
+
Cut a line down to a window around ``column``, and say which column the
|
|
125
|
+
offending character sits at in that window. Both columns count from 1.
|
|
126
|
+
"""
|
|
127
|
+
if len(line_text) <= _QUOTED_LINE_WIDTH:
|
|
128
|
+
return line_text, column
|
|
129
|
+
|
|
130
|
+
target = column - 1
|
|
131
|
+
last_start = len(line_text) - _QUOTED_LINE_WIDTH
|
|
132
|
+
start = min(max(target - _QUOTED_LINE_WIDTH // 2, 0), last_start)
|
|
133
|
+
end = start + _QUOTED_LINE_WIDTH
|
|
134
|
+
quoted = (_ELLIPSIS if start > 0 else "") + line_text[start:end] + (_ELLIPSIS if end < len(line_text) else "")
|
|
135
|
+
shift = len(_ELLIPSIS) if start > 0 else 0
|
|
136
|
+
return quoted, target - start + shift + 1
|
|
17
137
|
|
|
18
138
|
|
|
19
139
|
class Parser:
|
|
@@ -26,7 +146,7 @@ class Parser:
|
|
|
26
146
|
def __init__(
|
|
27
147
|
self,
|
|
28
148
|
max_input_size: int = 10 * 1024 * 1024,
|
|
29
|
-
max_depth: int =
|
|
149
|
+
max_depth: int = DEFAULT_MAX_DEPTH,
|
|
30
150
|
comments: bool = True,
|
|
31
151
|
):
|
|
32
152
|
"""
|
|
@@ -34,7 +154,10 @@ class Parser:
|
|
|
34
154
|
|
|
35
155
|
Args:
|
|
36
156
|
max_input_size: Maximum input size in bytes (default: 10MB)
|
|
37
|
-
max_depth:
|
|
157
|
+
max_depth: How deep links may nest (default: 64). Every
|
|
158
|
+
parenthesized group and every indentation level is one level,
|
|
159
|
+
and the lines of a document are at level 0; a document nested
|
|
160
|
+
deeper is refused with a ParseError whose ``max_depth`` is set
|
|
38
161
|
comments: Whether ``#`` starts a comment that runs to the end of
|
|
39
162
|
its line; when False it is an ordinary character (default: True)
|
|
40
163
|
"""
|
|
@@ -42,7 +165,15 @@ class Parser:
|
|
|
42
165
|
self.pos = 0
|
|
43
166
|
self.text = ""
|
|
44
167
|
self.lines = []
|
|
168
|
+
self.line_offsets: List[int] = []
|
|
45
169
|
self.base_indentation = None
|
|
170
|
+
# The document as written, for quoting the offending line in an error
|
|
171
|
+
self.source = ""
|
|
172
|
+
# Depth of the lines at level 0 of the context being parsed: the number
|
|
173
|
+
# of parenthesized groups around it
|
|
174
|
+
self.context_depth = 0
|
|
175
|
+
# Depth of the line being parsed
|
|
176
|
+
self.depth = 0
|
|
46
177
|
self.max_input_size = max_input_size
|
|
47
178
|
self.max_depth = max_depth
|
|
48
179
|
self.comments = comments
|
|
@@ -58,7 +189,8 @@ class Parser:
|
|
|
58
189
|
List of parsed Link objects
|
|
59
190
|
|
|
60
191
|
Raises:
|
|
61
|
-
ParseError: If parsing fails
|
|
192
|
+
ParseError: If parsing fails, including when the document nests
|
|
193
|
+
links deeper than ``max_depth``
|
|
62
194
|
TypeError: If input is not a string
|
|
63
195
|
ValueError: If input exceeds maximum size
|
|
64
196
|
"""
|
|
@@ -79,11 +211,14 @@ class Parser:
|
|
|
79
211
|
prepared = strip_comments(input_text) if self.comments else input_text
|
|
80
212
|
|
|
81
213
|
self.text = prepared
|
|
214
|
+
self.source = input_text
|
|
82
215
|
# Use smart line splitting that respects quoted strings
|
|
83
|
-
self.lines = self._split_lines_respecting_quotes(prepared)
|
|
216
|
+
self.lines, self.line_offsets = self._split_lines_respecting_quotes(prepared, 0)
|
|
84
217
|
self.pos = 0
|
|
85
218
|
self.indentation_stack = [0]
|
|
86
219
|
self.base_indentation = None
|
|
220
|
+
self.context_depth = 0
|
|
221
|
+
self.depth = 0
|
|
87
222
|
|
|
88
223
|
raw_result = self._parse_document()
|
|
89
224
|
return self._transform_result(raw_result)
|
|
@@ -96,6 +231,13 @@ class Parser:
|
|
|
96
231
|
except (KeyError, IndexError, AttributeError) as e:
|
|
97
232
|
# Catch specific parsing-related exceptions
|
|
98
233
|
raise ParseError(f"Parse error: {str(e)}") from e
|
|
234
|
+
except RecursionError:
|
|
235
|
+
# Only reachable with a max_depth set higher than Python's recursion
|
|
236
|
+
# limit allows; the default is far below it.
|
|
237
|
+
raise ParseError(
|
|
238
|
+
"Parse error: the document is nested too deeply for Python's recursion limit; "
|
|
239
|
+
f"lower max_depth (currently {self.max_depth}) to refuse it with a located error"
|
|
240
|
+
) from None
|
|
99
241
|
|
|
100
242
|
def _skip_quoted_string(self, text: str, start: int) -> int:
|
|
101
243
|
"""
|
|
@@ -107,7 +249,7 @@ class Parser:
|
|
|
107
249
|
parsed = _parse_quoted_string_at(text, start)
|
|
108
250
|
return -1 if parsed is None else parsed[1]
|
|
109
251
|
|
|
110
|
-
def _split_lines_respecting_quotes(self, text: str) -> List[str]:
|
|
252
|
+
def _split_lines_respecting_quotes(self, text: str, base: int) -> Tuple[List[str], List[int]]:
|
|
111
253
|
"""
|
|
112
254
|
Split text into lines, but preserve newlines inside quoted strings
|
|
113
255
|
and handle multiline parenthesized expressions.
|
|
@@ -115,47 +257,47 @@ class Parser:
|
|
|
115
257
|
Quoted strings can span multiple lines, and newlines within them
|
|
116
258
|
should be preserved as part of the string value. Also, parenthesized
|
|
117
259
|
expressions that span multiple lines are kept together.
|
|
260
|
+
|
|
261
|
+
Returns the lines and where each of them starts in the document, given
|
|
262
|
+
that text starts at ``base``.
|
|
118
263
|
"""
|
|
119
264
|
lines = []
|
|
120
|
-
|
|
265
|
+
offsets = []
|
|
266
|
+
line_start = 0
|
|
267
|
+
# Parentheses are counted in bulk between the characters that matter
|
|
268
|
+
# here, so a long run of them costs one pass in C rather than a Python
|
|
269
|
+
# step per character.
|
|
121
270
|
paren_depth = 0
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
while i < len(text):
|
|
125
|
-
char = text[i]
|
|
271
|
+
counted = 0
|
|
272
|
+
position = 0
|
|
126
273
|
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
|
|
141
|
-
|
|
142
|
-
if paren_depth > 0:
|
|
143
|
-
# Inside unclosed parens: preserve the newline
|
|
144
|
-
current_line += char
|
|
145
|
-
else:
|
|
146
|
-
# Parentheses balanced: this is a line break
|
|
147
|
-
lines.append(current_line)
|
|
148
|
-
current_line = ""
|
|
274
|
+
while True:
|
|
275
|
+
found = _LINE_BREAK_OR_QUOTE.search(text, position)
|
|
276
|
+
if found is None:
|
|
277
|
+
break
|
|
278
|
+
i = found.start()
|
|
279
|
+
paren_depth += text.count("(", counted, i) - text.count(")", counted, i)
|
|
280
|
+
|
|
281
|
+
if text[i] == "\n":
|
|
282
|
+
# Inside unclosed parens the newline is preserved; with the
|
|
283
|
+
# parentheses balanced it is a line break
|
|
284
|
+
if paren_depth <= 0:
|
|
285
|
+
lines.append(text[line_start:i])
|
|
286
|
+
offsets.append(base + line_start)
|
|
287
|
+
line_start = i + 1
|
|
288
|
+
position = i + 1
|
|
149
289
|
else:
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
290
|
+
# A quoted string is opaque: newlines inside it are content
|
|
291
|
+
end = self._skip_quoted_string(text, i)
|
|
292
|
+
position = end if end > i else i + 1
|
|
293
|
+
counted = position
|
|
153
294
|
|
|
154
295
|
# Add the last line if non-empty
|
|
155
|
-
if
|
|
156
|
-
lines.append(
|
|
296
|
+
if line_start < len(text):
|
|
297
|
+
lines.append(text[line_start:])
|
|
298
|
+
offsets.append(base + line_start)
|
|
157
299
|
|
|
158
|
-
return lines
|
|
300
|
+
return lines, offsets
|
|
159
301
|
|
|
160
302
|
def _parse_document(self) -> List[Dict]:
|
|
161
303
|
"""Parse the entire document."""
|
|
@@ -165,7 +307,7 @@ class Parser:
|
|
|
165
307
|
while self.pos < len(self.lines):
|
|
166
308
|
line = self.lines[self.pos]
|
|
167
309
|
if line.strip(): # Skip empty lines
|
|
168
|
-
element = self._parse_element(0)
|
|
310
|
+
element = self._parse_element(0, 0)
|
|
169
311
|
if element:
|
|
170
312
|
links.append(element)
|
|
171
313
|
else:
|
|
@@ -173,8 +315,13 @@ class Parser:
|
|
|
173
315
|
|
|
174
316
|
return links
|
|
175
317
|
|
|
176
|
-
def _parse_element(self, current_indent: int) -> Optional[Dict]:
|
|
177
|
-
"""
|
|
318
|
+
def _parse_element(self, current_indent: int, level: int) -> Optional[Dict]:
|
|
319
|
+
"""
|
|
320
|
+
Parse a single element (link or reference) at given indentation.
|
|
321
|
+
|
|
322
|
+
``level`` is the number of indentation levels the element sits at in
|
|
323
|
+
the context being parsed.
|
|
324
|
+
"""
|
|
178
325
|
if self.pos >= len(self.lines):
|
|
179
326
|
return None
|
|
180
327
|
|
|
@@ -196,14 +343,18 @@ class Parser:
|
|
|
196
343
|
self.pos += 1
|
|
197
344
|
return None
|
|
198
345
|
|
|
346
|
+
content_offset = self.line_offsets[self.pos] + len(line) - len(line.lstrip())
|
|
199
347
|
self.pos += 1
|
|
200
348
|
|
|
201
349
|
# Try to parse the line
|
|
202
|
-
|
|
350
|
+
line_depth = self.context_depth + level
|
|
351
|
+
self.depth = line_depth
|
|
352
|
+
element = self._parse_line_content(content, content_offset)
|
|
353
|
+
# Only a line that parsed counts, as in the other implementations
|
|
354
|
+
self._check_depth(line_depth, content_offset)
|
|
203
355
|
|
|
204
356
|
# Check for children (indented lines that follow)
|
|
205
357
|
children = []
|
|
206
|
-
child_indent = indent + 2 # Expect at least 2 spaces for child
|
|
207
358
|
|
|
208
359
|
while self.pos < len(self.lines):
|
|
209
360
|
# A line holding nothing does not close a block: the block goes on
|
|
@@ -226,7 +377,9 @@ class Parser:
|
|
|
226
377
|
|
|
227
378
|
# This is a child
|
|
228
379
|
self.pos = following
|
|
229
|
-
child
|
|
380
|
+
# A child only has to be indented deeper than its parent; asking
|
|
381
|
+
# for more left a line indented by a single space unread forever.
|
|
382
|
+
child = self._parse_element(indent + 1, level + 1)
|
|
230
383
|
if child:
|
|
231
384
|
children.append(child)
|
|
232
385
|
|
|
@@ -235,11 +388,19 @@ class Parser:
|
|
|
235
388
|
|
|
236
389
|
return element
|
|
237
390
|
|
|
238
|
-
def
|
|
239
|
-
"""
|
|
391
|
+
def _check_depth(self, depth: int, offset: int) -> None:
|
|
392
|
+
"""
|
|
393
|
+
Refuse, for good, links at ``depth`` when that is deeper than the
|
|
394
|
+
parser allows. ``offset`` is where the level that is too deep opens.
|
|
395
|
+
"""
|
|
396
|
+
if depth > self.max_depth:
|
|
397
|
+
raise ParseError.nesting_too_deep(self.source, offset, self.max_depth)
|
|
398
|
+
|
|
399
|
+
def _parse_line_content(self, content: str, offset: int) -> Dict:
|
|
400
|
+
"""Parse the content of a single line, which starts at ``offset``."""
|
|
240
401
|
# A whole parenthesized group: (id: values), (values) or a nested document
|
|
241
402
|
if content.startswith("(") and self._find_matching_paren(content, 0) == len(content) - 1:
|
|
242
|
-
return self._parse_parenthesized(content[1:-1])
|
|
403
|
+
return self._parse_parenthesized(content[1:-1], offset)
|
|
243
404
|
|
|
244
405
|
# Try indented ID syntax: id:
|
|
245
406
|
if content.endswith(":"):
|
|
@@ -251,69 +412,92 @@ class Parser:
|
|
|
251
412
|
colon_pos = self._find_colon_outside_quotes(content)
|
|
252
413
|
if colon_pos >= 0:
|
|
253
414
|
id_part = content[:colon_pos].strip()
|
|
254
|
-
|
|
415
|
+
after_colon = content[colon_pos + 1 :]
|
|
416
|
+
values_part = after_colon.strip()
|
|
417
|
+
values_offset = offset + colon_pos + 1 + len(after_colon) - len(after_colon.lstrip())
|
|
255
418
|
ref = self._extract_reference(id_part)
|
|
256
|
-
values = self._parse_values(values_part)
|
|
419
|
+
values = self._parse_values(values_part, values_offset)
|
|
257
420
|
return {"id": ref, "values": values}
|
|
258
421
|
|
|
259
422
|
# Simple value list
|
|
260
|
-
values = self._parse_values(content)
|
|
423
|
+
values = self._parse_values(content, offset)
|
|
261
424
|
return {"values": values}
|
|
262
425
|
|
|
263
|
-
def _parse_parenthesized(self, inner: str) -> Dict:
|
|
426
|
+
def _parse_parenthesized(self, inner: str, offset: int) -> Dict:
|
|
264
427
|
"""
|
|
265
|
-
Parse the content of a parenthesized group
|
|
428
|
+
Parse the content of a parenthesized group opened at ``offset``.
|
|
266
429
|
|
|
267
430
|
The group opens a nested context that starts fresh at indentation level
|
|
268
431
|
zero and follows exactly the rules used at the root of the document, so
|
|
269
|
-
line breaks separate links and indentation nests them.
|
|
432
|
+
line breaks separate links and indentation nests them. The group is one
|
|
433
|
+
level deeper than the line it is written on.
|
|
270
434
|
"""
|
|
271
|
-
|
|
435
|
+
self._check_depth(self.depth + 1, offset)
|
|
436
|
+
return {"nested": self._parse_nested_document(inner, offset + 1)}
|
|
272
437
|
|
|
273
|
-
def _parse_nested_document(self, inner: str) -> List[Dict]:
|
|
274
|
-
"""
|
|
438
|
+
def _parse_nested_document(self, inner: str, offset: int) -> List[Dict]:
|
|
439
|
+
"""
|
|
440
|
+
Parse the text of a parenthesized group, which starts at ``offset``,
|
|
441
|
+
as a document of its own.
|
|
442
|
+
"""
|
|
275
443
|
saved_lines = self.lines
|
|
444
|
+
saved_line_offsets = self.line_offsets
|
|
276
445
|
saved_pos = self.pos
|
|
277
446
|
saved_base_indentation = self.base_indentation
|
|
278
447
|
saved_indentation_stack = self.indentation_stack
|
|
448
|
+
saved_context_depth = self.context_depth
|
|
449
|
+
saved_depth = self.depth
|
|
279
450
|
try:
|
|
280
|
-
self.lines = self._split_lines_respecting_quotes(inner)
|
|
451
|
+
self.lines, self.line_offsets = self._split_lines_respecting_quotes(inner, offset)
|
|
281
452
|
self.pos = 0
|
|
282
453
|
self.base_indentation = None
|
|
283
454
|
self.indentation_stack = [0]
|
|
455
|
+
self.context_depth = self.depth + 1
|
|
284
456
|
return self._parse_document()
|
|
285
457
|
finally:
|
|
286
458
|
self.lines = saved_lines
|
|
459
|
+
self.line_offsets = saved_line_offsets
|
|
287
460
|
self.pos = saved_pos
|
|
288
461
|
self.base_indentation = saved_base_indentation
|
|
289
462
|
self.indentation_stack = saved_indentation_stack
|
|
463
|
+
self.context_depth = saved_context_depth
|
|
464
|
+
self.depth = saved_depth
|
|
290
465
|
|
|
291
466
|
def _find_matching_paren(self, text: str, start: int) -> int:
|
|
292
467
|
"""
|
|
293
468
|
Find the position of the parenthesis closing the one at start.
|
|
294
469
|
|
|
295
470
|
Quoted strings are skipped, so parentheses inside them are ignored.
|
|
296
|
-
Returns -1 when the group is not closed
|
|
297
|
-
|
|
298
|
-
depth = 0
|
|
299
|
-
i = start
|
|
471
|
+
Returns -1 when the group is not closed, or when start is not at an
|
|
472
|
+
opening parenthesis.
|
|
300
473
|
|
|
301
|
-
|
|
302
|
-
|
|
303
|
-
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
|
|
307
|
-
continue
|
|
308
|
-
elif char == "(":
|
|
309
|
-
depth += 1
|
|
310
|
-
elif char == ")":
|
|
311
|
-
depth -= 1
|
|
312
|
-
if depth == 0:
|
|
313
|
-
return i
|
|
314
|
-
i += 1
|
|
474
|
+
A run of parentheses is taken in one step, so the parentheses deeply
|
|
475
|
+
nested groups begin and end with cost a step per run rather than one
|
|
476
|
+
per character: every group is scanned once for each group around it.
|
|
477
|
+
"""
|
|
478
|
+
if not text.startswith("(", start):
|
|
479
|
+
return -1
|
|
315
480
|
|
|
316
|
-
|
|
481
|
+
depth = 0
|
|
482
|
+
position = start
|
|
483
|
+
|
|
484
|
+
while True:
|
|
485
|
+
for found in _PAREN_RUN_OR_QUOTE.finditer(text, position):
|
|
486
|
+
run_start, run_end = found.span()
|
|
487
|
+
char = text[run_start]
|
|
488
|
+
if char == "(":
|
|
489
|
+
depth += run_end - run_start
|
|
490
|
+
elif char == ")":
|
|
491
|
+
if run_end - run_start >= depth:
|
|
492
|
+
return run_start + depth - 1
|
|
493
|
+
depth -= run_end - run_start
|
|
494
|
+
else:
|
|
495
|
+
end = self._skip_quoted_string(text, run_start)
|
|
496
|
+
if end > run_start:
|
|
497
|
+
position = end
|
|
498
|
+
break
|
|
499
|
+
else:
|
|
500
|
+
return -1
|
|
317
501
|
|
|
318
502
|
def _find_colon_outside_quotes(self, text: str) -> int:
|
|
319
503
|
"""
|
|
@@ -325,28 +509,28 @@ class Parser:
|
|
|
325
509
|
because it's inside the second parenthesized expression.
|
|
326
510
|
"""
|
|
327
511
|
paren_depth = 0
|
|
328
|
-
|
|
329
|
-
|
|
330
|
-
|
|
331
|
-
|
|
332
|
-
|
|
512
|
+
counted = 0
|
|
513
|
+
position = 0
|
|
514
|
+
|
|
515
|
+
while True:
|
|
516
|
+
found = _COLON_OR_QUOTE.search(text, position)
|
|
517
|
+
if found is None:
|
|
518
|
+
return -1
|
|
519
|
+
i = found.start()
|
|
520
|
+
paren_depth += text.count("(", counted, i) - text.count(")", counted, i)
|
|
521
|
+
|
|
522
|
+
if text[i] == ":":
|
|
523
|
+
if paren_depth == 0:
|
|
524
|
+
# Only return colon if it's outside quotes AND at parenthesis depth 0
|
|
525
|
+
return i
|
|
526
|
+
position = i + 1
|
|
527
|
+
else:
|
|
333
528
|
end = self._skip_quoted_string(text, i)
|
|
334
|
-
if end > i
|
|
335
|
-
|
|
336
|
-
continue
|
|
337
|
-
elif char == "(":
|
|
338
|
-
paren_depth += 1
|
|
339
|
-
elif char == ")":
|
|
340
|
-
paren_depth -= 1
|
|
341
|
-
elif char == ":" and paren_depth == 0:
|
|
342
|
-
# Only return colon if it's outside quotes AND at parenthesis depth 0
|
|
343
|
-
return i
|
|
344
|
-
i += 1
|
|
345
|
-
|
|
346
|
-
return -1
|
|
529
|
+
position = end if end > i else i + 1
|
|
530
|
+
counted = position
|
|
347
531
|
|
|
348
|
-
def _parse_values(self, text: str) -> List[Dict]:
|
|
349
|
-
"""Parse a space-separated list of values
|
|
532
|
+
def _parse_values(self, text: str, offset: int) -> List[Dict]:
|
|
533
|
+
"""Parse a space-separated list of values, which starts at ``offset``."""
|
|
350
534
|
if not text:
|
|
351
535
|
return []
|
|
352
536
|
|
|
@@ -363,7 +547,7 @@ class Parser:
|
|
|
363
547
|
# Try to extract the next value
|
|
364
548
|
value_end, value_text = self._extract_next_value(text, i)
|
|
365
549
|
if value_text and value_text.strip():
|
|
366
|
-
values.append(self._parse_value(value_text))
|
|
550
|
+
values.append(self._parse_value(value_text, offset + i))
|
|
367
551
|
if value_end == i:
|
|
368
552
|
# No progress made - skip this character to avoid infinite loop
|
|
369
553
|
i += 1
|
|
@@ -414,11 +598,11 @@ class Parser:
|
|
|
414
598
|
|
|
415
599
|
return (i, text[start:i])
|
|
416
600
|
|
|
417
|
-
def _parse_value(self, value: str) -> Dict:
|
|
418
|
-
"""Parse a single value (could be a reference or nested link)
|
|
601
|
+
def _parse_value(self, value: str, offset: int) -> Dict:
|
|
602
|
+
"""Parse a single value (could be a reference or nested link) starting at ``offset``."""
|
|
419
603
|
# Nested link in parentheses
|
|
420
604
|
if value.startswith("(") and self._find_matching_paren(value, 0) == len(value) - 1:
|
|
421
|
-
return self._parse_parenthesized(value[1:-1])
|
|
605
|
+
return self._parse_parenthesized(value[1:-1], offset)
|
|
422
606
|
|
|
423
607
|
# Simple reference
|
|
424
608
|
ref = self._extract_reference(value)
|
|
@@ -4,7 +4,7 @@ from dataclasses import dataclass
|
|
|
4
4
|
from typing import AsyncIterable, AsyncIterator, Callable, Iterable, Iterator, List, Optional
|
|
5
5
|
|
|
6
6
|
from .link import Link
|
|
7
|
-
from .parser import ParseError, Parser
|
|
7
|
+
from .parser import DEFAULT_MAX_DEPTH, ParseError, Parser
|
|
8
8
|
from .quotes import QUOTE_CHARS, _parse_quoted_string_at
|
|
9
9
|
|
|
10
10
|
|
|
@@ -19,14 +19,22 @@ class StreamPosition:
|
|
|
19
19
|
|
|
20
20
|
|
|
21
21
|
class StreamParseError(ParseError):
|
|
22
|
-
"""A canonical parse error located within the complete stream.
|
|
22
|
+
"""A canonical parse error located within the complete stream.
|
|
23
|
+
|
|
24
|
+
``error`` is the error the parser raised for the buffered record;
|
|
25
|
+
``offset``, ``line`` and ``column`` locate it within the complete stream.
|
|
26
|
+
"""
|
|
23
27
|
|
|
24
28
|
def __init__(self, error: Exception, offset: int, line: int, column: int):
|
|
25
|
-
super().__init__(
|
|
29
|
+
super().__init__(
|
|
30
|
+
f"Stream parse error at line {line}, column {column}: {error}",
|
|
31
|
+
offset=offset,
|
|
32
|
+
line=line,
|
|
33
|
+
column=column,
|
|
34
|
+
line_text=getattr(error, "line_text", None),
|
|
35
|
+
max_depth=getattr(error, "max_depth", None),
|
|
36
|
+
)
|
|
26
37
|
self.error = error
|
|
27
|
-
self.offset = offset
|
|
28
|
-
self.line = line
|
|
29
|
-
self.column = column
|
|
30
38
|
|
|
31
39
|
|
|
32
40
|
class StreamParser:
|
|
@@ -45,8 +53,9 @@ class StreamParser:
|
|
|
45
53
|
collect: bool = True,
|
|
46
54
|
max_buffer_size: Optional[int] = None,
|
|
47
55
|
comments: bool = True,
|
|
56
|
+
max_depth: int = DEFAULT_MAX_DEPTH,
|
|
48
57
|
):
|
|
49
|
-
self.parser = parser or Parser(comments=comments)
|
|
58
|
+
self.parser = parser or Parser(comments=comments, max_depth=max_depth)
|
|
50
59
|
self.on_link = on_link
|
|
51
60
|
self.collect = collect
|
|
52
61
|
self.max_buffer_size = self.parser.max_input_size if max_buffer_size is None else max_buffer_size
|
|
@@ -95,7 +104,7 @@ class StreamParser:
|
|
|
95
104
|
try:
|
|
96
105
|
links = self.parser.parse(document)
|
|
97
106
|
except Exception as error:
|
|
98
|
-
raise
|
|
107
|
+
raise self._stream_error(error) from error
|
|
99
108
|
self._publish(links, emitted)
|
|
100
109
|
self._advance_segment(document)
|
|
101
110
|
|
|
@@ -164,6 +173,15 @@ class StreamParser:
|
|
|
164
173
|
if self.on_link is not None:
|
|
165
174
|
self.on_link(link)
|
|
166
175
|
|
|
176
|
+
def _stream_error(self, error: Exception) -> StreamParseError:
|
|
177
|
+
"""Locate an error the parser raised for the buffered record within the stream."""
|
|
178
|
+
line = getattr(error, "line", None)
|
|
179
|
+
column = getattr(error, "column", None)
|
|
180
|
+
offset = getattr(error, "offset", None)
|
|
181
|
+
if line is None or column is None or offset is None:
|
|
182
|
+
return StreamParseError(error, self._segment_offset, self._segment_line, 1)
|
|
183
|
+
return StreamParseError(error, self._segment_offset + offset, self._segment_line + line - 1, column)
|
|
184
|
+
|
|
167
185
|
def _advance_segment(self, document: str) -> None:
|
|
168
186
|
self._segment_offset += len(document)
|
|
169
187
|
self._segment_line += document.count("\n")
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: links-notation
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.22.0
|
|
4
4
|
Summary: Python implementation of the Links Notation parser
|
|
5
5
|
Author-email: "Link.Foundation" <drakonard@gmail.com>
|
|
6
6
|
License-Expression: Unlicense
|
|
@@ -140,9 +140,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
|
|
|
140
140
|
|
|
141
141
|
The main parser class for Links Notation.
|
|
142
142
|
|
|
143
|
-
- `__init__(
|
|
144
|
-
a `#` is an ordinary character instead
|
|
145
|
-
|
|
143
|
+
- `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
|
|
144
|
+
Create a parser; with `comments=False` a `#` is an ordinary character instead
|
|
145
|
+
of the start of a comment
|
|
146
|
+
- `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
|
|
147
|
+
objects; raises `ParseError` when the text does not parse or nests links
|
|
148
|
+
deeper than `max_depth`
|
|
149
|
+
- `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
|
|
150
|
+
implementation
|
|
151
|
+
|
|
152
|
+
`max_depth` is how deep links may nest (default: 64). Every parenthesized group
|
|
153
|
+
and every indentation level is one level, and the lines of a document start at
|
|
154
|
+
level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
|
|
155
|
+
group on an indented line are refused. A document nested deeper is refused with
|
|
156
|
+
a `ParseError` rather than recursed into until Python's recursion limit is hit.
|
|
157
|
+
|
|
158
|
+
### ParseError
|
|
159
|
+
|
|
160
|
+
Raised when parsing fails. When a document nests links deeper than
|
|
161
|
+
`max_depth`, it points at the group or the line that is one level too deep:
|
|
162
|
+
|
|
163
|
+
```text
|
|
164
|
+
Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
|
|
165
|
+
1 | ((((a))))
|
|
166
|
+
| ^
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
- `max_depth`: The deepest nesting the parser accepts, when the document nests
|
|
170
|
+
deeper; `None` for any other error
|
|
171
|
+
- `line`, `column`: Where the offending group or line starts, counted from 1
|
|
172
|
+
- `offset`: The same position as a character offset from the start of the
|
|
173
|
+
document
|
|
174
|
+
- `line_text`: The offending line, as written
|
|
175
|
+
|
|
176
|
+
`StreamParser` reports the same error as a `StreamParseError` whose `error` is
|
|
177
|
+
the `ParseError` and whose `line`, `column` and `offset` are counted from the
|
|
178
|
+
start of the stream.
|
|
146
179
|
|
|
147
180
|
### Link
|
|
148
181
|
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Links nested too deeply are refused with an error rather than recursed into
|
|
2
|
+
until the stack overflows
|
|
3
|
+
(https://github.com/link-foundation/links-notation/issues/315).
|
|
4
|
+
|
|
5
|
+
Every parenthesized group and every indentation level is one level, and the
|
|
6
|
+
lines of a document start at level 0. The positions asserted here are the ones
|
|
7
|
+
the Rust port reports for the same input.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
import sys
|
|
11
|
+
|
|
12
|
+
import pytest
|
|
13
|
+
|
|
14
|
+
from links_notation import DEFAULT_MAX_DEPTH, ParseError, Parser, StreamParser
|
|
15
|
+
from links_notation.stream_parser import StreamParseError
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def parens(depth):
|
|
19
|
+
return "(" * depth + "a" + ")" * depth
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def values(depth):
|
|
23
|
+
return "(a " * depth + "b" + ")" * depth
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def indentation(depth):
|
|
27
|
+
return "".join(" " * level + "a\n" for level in range(depth + 1))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def too_deep(document, max_depth):
|
|
31
|
+
with pytest.raises(ParseError) as caught:
|
|
32
|
+
Parser(max_depth=max_depth).parse(document)
|
|
33
|
+
assert caught.value.max_depth == max_depth
|
|
34
|
+
return caught.value
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def accepted(document, max_depth=DEFAULT_MAX_DEPTH):
|
|
38
|
+
return len(Parser(max_depth=max_depth).parse(document)) > 0
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_default_limit_is_shared_by_every_implementation():
|
|
42
|
+
assert DEFAULT_MAX_DEPTH == 64
|
|
43
|
+
assert Parser().max_depth == DEFAULT_MAX_DEPTH
|
|
44
|
+
assert StreamParser().parser.max_depth == DEFAULT_MAX_DEPTH
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def test_parentheses_up_to_the_limit_are_accepted():
|
|
48
|
+
assert accepted(parens(3), 3)
|
|
49
|
+
assert accepted(values(3), 3)
|
|
50
|
+
assert accepted(parens(DEFAULT_MAX_DEPTH))
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_parentheses_past_the_limit_are_refused_at_the_group_that_is_too_deep():
|
|
54
|
+
error = too_deep(parens(4), 3)
|
|
55
|
+
|
|
56
|
+
assert (error.line, error.column, error.offset) == (1, 4, 3)
|
|
57
|
+
assert str(error) == (
|
|
58
|
+
"Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3\n" "1 | ((((a))))\n" " | ^"
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def test_groups_in_value_position_count_like_any_other_group():
|
|
63
|
+
error = too_deep(values(4), 3)
|
|
64
|
+
|
|
65
|
+
assert (error.line, error.column) == (1, 10)
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def test_indentation_up_to_the_limit_is_accepted():
|
|
69
|
+
assert accepted(indentation(3), 3)
|
|
70
|
+
assert accepted(indentation(DEFAULT_MAX_DEPTH))
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_indentation_past_the_limit_is_refused_at_the_line_that_is_too_deep():
|
|
74
|
+
error = too_deep(indentation(4), 3)
|
|
75
|
+
|
|
76
|
+
assert (error.line, error.column) == (5, 5)
|
|
77
|
+
assert error.line_text == " a"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def test_groups_and_indentation_add_up():
|
|
81
|
+
# `(b)` on the line indented once is at level 2.
|
|
82
|
+
assert accepted("a\n (b)\n", 2)
|
|
83
|
+
error = too_deep("a\n (b)\n", 1)
|
|
84
|
+
|
|
85
|
+
assert (error.line, error.column) == (2, 3)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def test_limit_of_one_allows_one_group():
|
|
89
|
+
assert accepted("(a b)", 1)
|
|
90
|
+
assert accepted("a\n b\n", 1)
|
|
91
|
+
too_deep("((a))", 1)
|
|
92
|
+
too_deep("a\n b\n c\n", 1)
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def test_trailing_spaces_on_a_deep_line_are_not_a_deeper_line():
|
|
96
|
+
assert accepted("a\n b\n c \n", 2)
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def test_limit_past_the_recursion_limit_is_an_error_rather_than_a_crash():
|
|
100
|
+
# Nothing here is too deep for the limit, so the error has no max_depth.
|
|
101
|
+
with pytest.raises(ParseError) as caught:
|
|
102
|
+
Parser(max_depth=sys.maxsize).parse(parens(100_000))
|
|
103
|
+
|
|
104
|
+
assert caught.value.max_depth is None
|
|
105
|
+
assert not str(caught.value).startswith("Nesting too deep")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def test_parser_is_reusable_after_refusing_a_document():
|
|
109
|
+
parser = Parser(max_depth=2)
|
|
110
|
+
with pytest.raises(ParseError):
|
|
111
|
+
parser.parse(parens(3))
|
|
112
|
+
|
|
113
|
+
assert parser.parse(parens(2))
|
|
114
|
+
assert parser.parse(indentation(2))
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
@pytest.mark.timeout(30)
|
|
118
|
+
def test_refuses_a_document_far_past_the_limit_without_overflowing_the_stack():
|
|
119
|
+
# Before the limit existed each of these exhausted Python's recursion limit.
|
|
120
|
+
# Each group around the one too deep is scanned to its end before it is
|
|
121
|
+
# entered, so refusing values(n) reads the document once per level; it is
|
|
122
|
+
# kept short enough to be refused quickly.
|
|
123
|
+
for document in (parens(100_000), values(5_000), indentation(2_000)):
|
|
124
|
+
error = too_deep(document, DEFAULT_MAX_DEPTH)
|
|
125
|
+
assert str(error).startswith("Nesting too deep at ")
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def test_stream_parser_reports_where_the_nesting_is_too_deep():
|
|
129
|
+
stream = StreamParser(max_depth=1)
|
|
130
|
+
stream.write("a\nb ((c))\n")
|
|
131
|
+
|
|
132
|
+
with pytest.raises(StreamParseError) as caught:
|
|
133
|
+
stream.finish()
|
|
134
|
+
|
|
135
|
+
assert caught.value.error.max_depth == 1
|
|
136
|
+
assert caught.value.max_depth == 1
|
|
137
|
+
assert (caught.value.line, caught.value.column, caught.value.offset) == (2, 4, 5)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/dependency_links.txt
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|