links-notation 0.21.2__tar.gz → 0.22.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (40) hide show
  1. {links_notation-0.21.2/links_notation.egg-info → links_notation-0.22.0}/PKG-INFO +37 -4
  2. {links_notation-0.21.2 → links_notation-0.22.0}/README.md +36 -3
  3. {links_notation-0.21.2 → links_notation-0.22.0}/README.ru.md +37 -3
  4. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/__init__.py +3 -1
  5. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/parser.py +286 -102
  6. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/stream_parser.py +26 -8
  7. {links_notation-0.21.2 → links_notation-0.22.0/links_notation.egg-info}/PKG-INFO +37 -4
  8. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/SOURCES.txt +1 -0
  9. {links_notation-0.21.2 → links_notation-0.22.0}/pyproject.toml +1 -1
  10. links_notation-0.22.0/tests/test_nesting_limit.py +137 -0
  11. {links_notation-0.21.2 → links_notation-0.22.0}/LICENSE +0 -0
  12. {links_notation-0.21.2 → links_notation-0.22.0}/MANIFEST.in +0 -0
  13. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/comments.py +0 -0
  14. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/format_config.py +0 -0
  15. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/formatter.py +0 -0
  16. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/link.py +0 -0
  17. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation/quotes.py +0 -0
  18. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/dependency_links.txt +0 -0
  19. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/requires.txt +0 -0
  20. {links_notation-0.21.2 → links_notation-0.22.0}/links_notation.egg-info/top_level.txt +0 -0
  21. {links_notation-0.21.2 → links_notation-0.22.0}/setup.cfg +0 -0
  22. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_api.py +0 -0
  23. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_comments.py +0 -0
  24. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_edge_case_parser.py +0 -0
  25. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_empty_reference.py +0 -0
  26. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_format_config.py +0 -0
  27. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indentation_consistency.py +0 -0
  28. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indented_id_nested_values.py +0 -0
  29. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_indented_id_syntax.py +0 -0
  30. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_link.py +0 -0
  31. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_links_group.py +0 -0
  32. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_mixed_indentation_modes.py +0 -0
  33. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multi_quote_parser.py +0 -0
  34. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multiline_parser.py +0 -0
  35. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_multiline_quoted_string.py +0 -0
  36. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_indentation.py +0 -0
  37. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_parser.py +0 -0
  38. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_nested_self_reference.py +0 -0
  39. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_single_line_parser.py +0 -0
  40. {links_notation-0.21.2 → links_notation-0.22.0}/tests/test_stream_parser.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: links-notation
3
- Version: 0.21.2
3
+ Version: 0.22.0
4
4
  Summary: Python implementation of the Links Notation parser
5
5
  Author-email: "Link.Foundation" <drakonard@gmail.com>
6
6
  License-Expression: Unlicense
@@ -140,9 +140,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
140
140
 
141
141
  The main parser class for Links Notation.
142
142
 
143
- - `__init__(..., comments: bool = True)`: Create a parser; with `comments=False`
144
- a `#` is an ordinary character instead of the start of a comment
145
- - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link objects
143
+ - `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
144
+ Create a parser; with `comments=False` a `#` is an ordinary character instead
145
+ of the start of a comment
146
+ - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
147
+ objects; raises `ParseError` when the text does not parse or nests links
148
+ deeper than `max_depth`
149
+ - `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
150
+ implementation
151
+
152
+ `max_depth` is how deep links may nest (default: 64). Every parenthesized group
153
+ and every indentation level is one level, and the lines of a document start at
154
+ level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
155
+ group on an indented line are refused. A document nested deeper is refused with
156
+ a `ParseError` rather than recursed into until Python's recursion limit is hit.
157
+
158
+ ### ParseError
159
+
160
+ Raised when parsing fails. When a document nests links deeper than
161
+ `max_depth`, it points at the group or the line that is one level too deep:
162
+
163
+ ```text
164
+ Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
165
+ 1 | ((((a))))
166
+ | ^
167
+ ```
168
+
169
+ - `max_depth`: The deepest nesting the parser accepts, when the document nests
170
+ deeper; `None` for any other error
171
+ - `line`, `column`: Where the offending group or line starts, counted from 1
172
+ - `offset`: The same position as a character offset from the start of the
173
+ document
174
+ - `line_text`: The offending line, as written
175
+
176
+ `StreamParser` reports the same error as a `StreamParseError` whose `error` is
177
+ the `ParseError` and whose `line`, `column` and `offset` are counted from the
178
+ start of the stream.
146
179
 
147
180
  ### Link
148
181
 
@@ -102,9 +102,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
102
102
 
103
103
  The main parser class for Links Notation.
104
104
 
105
- - `__init__(..., comments: bool = True)`: Create a parser; with `comments=False`
106
- a `#` is an ordinary character instead of the start of a comment
107
- - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link objects
105
+ - `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
106
+ Create a parser; with `comments=False` a `#` is an ordinary character instead
107
+ of the start of a comment
108
+ - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
109
+ objects; raises `ParseError` when the text does not parse or nests links
110
+ deeper than `max_depth`
111
+ - `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
112
+ implementation
113
+
114
+ `max_depth` is how deep links may nest (default: 64). Every parenthesized group
115
+ and every indentation level is one level, and the lines of a document start at
116
+ level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
117
+ group on an indented line are refused. A document nested deeper is refused with
118
+ a `ParseError` rather than recursed into until Python's recursion limit is hit.
119
+
120
+ ### ParseError
121
+
122
+ Raised when parsing fails. When a document nests links deeper than
123
+ `max_depth`, it points at the group or the line that is one level too deep:
124
+
125
+ ```text
126
+ Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
127
+ 1 | ((((a))))
128
+ | ^
129
+ ```
130
+
131
+ - `max_depth`: The deepest nesting the parser accepts, when the document nests
132
+ deeper; `None` for any other error
133
+ - `line`, `column`: Where the offending group or line starts, counted from 1
134
+ - `offset`: The same position as a character offset from the start of the
135
+ document
136
+ - `line_text`: The offending line, as written
137
+
138
+ `StreamParser` reports the same error as a `StreamParseError` whose `error` is
139
+ the `ParseError` and whose `line`, `column` and `offset` are counted from the
140
+ start of the stream.
108
141
 
109
142
  ### Link
110
143
 
@@ -103,9 +103,43 @@ for link in parse_chunks(["одна свя", "зь\nвторая связь"]):
103
103
 
104
104
  Основной класс парсера для Links Notation.
105
105
 
106
- - `__init__(..., comments: bool = True)`: создать парсер; при `comments=False`
107
- `#` — обычный символ, а не начало комментария
108
- - `parse(input_text: str) -> List[Link]`: Парсинг текста Links Notation в объекты Link
106
+ - `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
107
+ создать парсер; при `comments=False` `#` — обычный символ, а не начало
108
+ комментария
109
+ - `parse(input_text: str) -> List[Link]`: Парсинг текста Links Notation в
110
+ объекты Link; выбрасывает `ParseError`, если текст не удалось разобрать или
111
+ связи в нем вложены глубже `max_depth`
112
+ - `DEFAULT_MAX_DEPTH`: значение `max_depth` по умолчанию, 64, одинаковое во всех
113
+ реализациях
114
+
115
+ `max_depth` — насколько глубоко могут вкладываться связи (по умолчанию 64).
116
+ Каждая группа в скобках и каждый уровень отступа — это один уровень, а строки
117
+ документа находятся на уровне 0, поэтому при `max_depth=1` `(a)` принимается,
118
+ а `((a))`, `(a (b))` и группа на строке с отступом отклоняются. Документ с более
119
+ глубокой вложенностью отклоняется с `ParseError`, а не разбирается рекурсивно,
120
+ пока не будет достигнут предел рекурсии Python.
121
+
122
+ ### ParseError
123
+
124
+ Выбрасывается при ошибке разбора. Когда связи в документе вложены глубже
125
+ `max_depth`, указывает на группу или строку, которая на уровень глубже
126
+ допустимого:
127
+
128
+ ```text
129
+ Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
130
+ 1 | ((((a))))
131
+ | ^
132
+ ```
133
+
134
+ - `max_depth`: наибольшая допустимая глубина вложенности, если документ вложен
135
+ глубже; `None` для любой другой ошибки
136
+ - `line`, `column`: где начинается группа или строка, считая с 1
137
+ - `offset`: та же позиция как смещение в символах от начала документа
138
+ - `line_text`: строка с ошибкой в том виде, в каком она написана
139
+
140
+ `StreamParser` сообщает о той же ошибке через `StreamParseError`, у которого
141
+ `error` — это `ParseError`, а `line`, `column` и `offset` отсчитываются от
142
+ начала потока.
109
143
 
110
144
  ### Link
111
145
 
@@ -14,7 +14,7 @@ from .comments import strip_comments
14
14
  from .format_config import FormatConfig
15
15
  from .formatter import format_links
16
16
  from .link import Link
17
- from .parser import Parser
17
+ from .parser import DEFAULT_MAX_DEPTH, ParseError, Parser
18
18
  from .stream_parser import StreamParseError, StreamParser, StreamPosition, parse_async_chunks, parse_chunks
19
19
 
20
20
 
@@ -47,7 +47,9 @@ def _read_version() -> str:
47
47
  __version__ = _read_version()
48
48
 
49
49
  __all__ = [
50
+ "DEFAULT_MAX_DEPTH",
50
51
  "Link",
52
+ "ParseError",
51
53
  "Parser",
52
54
  "StreamParseError",
53
55
  "StreamParser",
@@ -5,15 +5,135 @@ This module provides parsing functionality for Links Notation (Lino),
5
5
  converting text into structured Link objects.
6
6
  """
7
7
 
8
- from typing import Any, Dict, List, Optional
8
+ import re
9
+ from typing import Any, Dict, List, Optional, Tuple
9
10
 
10
11
  from .comments import strip_comments
11
12
  from .link import Link
12
13
  from .quotes import _parse_quoted_string_at
13
14
 
15
+ #: How deep links may nest unless a parser is told otherwise: every
16
+ #: parenthesized group and every indentation level is one level, and the lines
17
+ #: of a document are at level 0. Every implementation shares this default.
18
+ DEFAULT_MAX_DEPTH = 64
19
+
20
+ #: The number of characters a quoted line is cut down to.
21
+ _QUOTED_LINE_WIDTH = 80
22
+
23
+ #: What a message writes in place of the part of a long line it left out.
24
+ _ELLIPSIS = "..."
25
+
26
+
27
+ #: The characters that can end a line or open a quoted string.
28
+ _LINE_BREAK_OR_QUOTE = re.compile("[\n\"'`]")
29
+
30
+ #: The characters that can separate an id from its values or open a quoted string.
31
+ _COLON_OR_QUOTE = re.compile("[:\"'`]")
32
+
33
+ #: A run of opening parentheses, a run of closing ones, or a character that can
34
+ #: open a quoted string.
35
+ _PAREN_RUN_OR_QUOTE = re.compile("\\(+|\\)+|[\"'`]")
36
+
14
37
 
15
38
  class ParseError(Exception):
16
- """Exception raised when parsing fails."""
39
+ """
40
+ Exception raised when parsing fails.
41
+
42
+ An error that points at a place in the document carries that place:
43
+ ``offset`` (characters from the start of the document), ``line`` and
44
+ ``column`` (both counted from 1), and ``line_text`` (the offending line as
45
+ written, without its line ending); for any other error they are None.
46
+
47
+ A document nested deeper than the parser's ``max_depth`` is refused with
48
+ this error too; then ``max_depth`` says how deep the nesting may go. It is
49
+ None for any other error.
50
+ """
51
+
52
+ def __init__(
53
+ self,
54
+ message: str = "",
55
+ *,
56
+ offset: Optional[int] = None,
57
+ line: Optional[int] = None,
58
+ column: Optional[int] = None,
59
+ line_text: Optional[str] = None,
60
+ max_depth: Optional[int] = None,
61
+ ):
62
+ super().__init__(message)
63
+ self.offset = offset
64
+ self.line = line
65
+ self.column = column
66
+ self.line_text = line_text
67
+ self.max_depth = max_depth
68
+
69
+ @classmethod
70
+ def nesting_too_deep(cls, document: str, offset: int, max_depth: int) -> "ParseError":
71
+ """The error for a document nested deeper than ``max_depth`` at ``offset``."""
72
+ line, column, line_text = _locate(document, offset)
73
+ summary = f"line {line}, column {column}: nesting depth exceeds the maximum of {max_depth}"
74
+ return cls(
75
+ f"Nesting too deep at {summary}\n{_quote(line, line_text, column)}",
76
+ offset=offset,
77
+ line=line,
78
+ column=column,
79
+ line_text=line_text,
80
+ max_depth=max_depth,
81
+ )
82
+
83
+
84
+ def _locate(document: str, offset: int) -> Tuple[int, int, str]:
85
+ """
86
+ The line and column ``offset`` falls on, both counted from 1, and that line
87
+ without its line ending. CR, LF and CRLF all end a line.
88
+ """
89
+ offset = max(0, min(offset, len(document)))
90
+ line = 1
91
+ line_start = 0
92
+ cursor = 0
93
+ while cursor < offset:
94
+ character = document[cursor]
95
+ cursor += 1
96
+ if character == "\r":
97
+ line += 1
98
+ if cursor < offset and document[cursor] == "\n":
99
+ cursor += 1
100
+ line_start = cursor
101
+ elif character == "\n":
102
+ line += 1
103
+ line_start = cursor
104
+ line_end = len(document)
105
+ for ending in ("\r", "\n"):
106
+ found = document.find(ending, line_start)
107
+ if 0 <= found < line_end:
108
+ line_end = found
109
+ return line, offset - line_start + 1, document[line_start:line_end]
110
+
111
+
112
+ def _quote(number: int, line_text: str, column: int) -> str:
113
+ """
114
+ The offending line with a caret under the offending column, quoted the way
115
+ a compiler quotes source. A long line is shown as a window around the caret.
116
+ """
117
+ quoted, column = _window_around(line_text, column)
118
+ gutter = " " * len(str(number))
119
+ return f"{number} | {quoted}\n{gutter} | {' ' * (column - 1)}^"
120
+
121
+
122
+ def _window_around(line_text: str, column: int) -> Tuple[str, int]:
123
+ """
124
+ Cut a line down to a window around ``column``, and say which column the
125
+ offending character sits at in that window. Both columns count from 1.
126
+ """
127
+ if len(line_text) <= _QUOTED_LINE_WIDTH:
128
+ return line_text, column
129
+
130
+ target = column - 1
131
+ last_start = len(line_text) - _QUOTED_LINE_WIDTH
132
+ start = min(max(target - _QUOTED_LINE_WIDTH // 2, 0), last_start)
133
+ end = start + _QUOTED_LINE_WIDTH
134
+ quoted = (_ELLIPSIS if start > 0 else "") + line_text[start:end] + (_ELLIPSIS if end < len(line_text) else "")
135
+ shift = len(_ELLIPSIS) if start > 0 else 0
136
+ return quoted, target - start + shift + 1
17
137
 
18
138
 
19
139
  class Parser:
@@ -26,7 +146,7 @@ class Parser:
26
146
  def __init__(
27
147
  self,
28
148
  max_input_size: int = 10 * 1024 * 1024,
29
- max_depth: int = 1000,
149
+ max_depth: int = DEFAULT_MAX_DEPTH,
30
150
  comments: bool = True,
31
151
  ):
32
152
  """
@@ -34,7 +154,10 @@ class Parser:
34
154
 
35
155
  Args:
36
156
  max_input_size: Maximum input size in bytes (default: 10MB)
37
- max_depth: Maximum nesting depth (default: 1000)
157
+ max_depth: How deep links may nest (default: 64). Every
158
+ parenthesized group and every indentation level is one level,
159
+ and the lines of a document are at level 0; a document nested
160
+ deeper is refused with a ParseError whose ``max_depth`` is set
38
161
  comments: Whether ``#`` starts a comment that runs to the end of
39
162
  its line; when False it is an ordinary character (default: True)
40
163
  """
@@ -42,7 +165,15 @@ class Parser:
42
165
  self.pos = 0
43
166
  self.text = ""
44
167
  self.lines = []
168
+ self.line_offsets: List[int] = []
45
169
  self.base_indentation = None
170
+ # The document as written, for quoting the offending line in an error
171
+ self.source = ""
172
+ # Depth of the lines at level 0 of the context being parsed: the number
173
+ # of parenthesized groups around it
174
+ self.context_depth = 0
175
+ # Depth of the line being parsed
176
+ self.depth = 0
46
177
  self.max_input_size = max_input_size
47
178
  self.max_depth = max_depth
48
179
  self.comments = comments
@@ -58,7 +189,8 @@ class Parser:
58
189
  List of parsed Link objects
59
190
 
60
191
  Raises:
61
- ParseError: If parsing fails
192
+ ParseError: If parsing fails, including when the document nests
193
+ links deeper than ``max_depth``
62
194
  TypeError: If input is not a string
63
195
  ValueError: If input exceeds maximum size
64
196
  """
@@ -79,11 +211,14 @@ class Parser:
79
211
  prepared = strip_comments(input_text) if self.comments else input_text
80
212
 
81
213
  self.text = prepared
214
+ self.source = input_text
82
215
  # Use smart line splitting that respects quoted strings
83
- self.lines = self._split_lines_respecting_quotes(prepared)
216
+ self.lines, self.line_offsets = self._split_lines_respecting_quotes(prepared, 0)
84
217
  self.pos = 0
85
218
  self.indentation_stack = [0]
86
219
  self.base_indentation = None
220
+ self.context_depth = 0
221
+ self.depth = 0
87
222
 
88
223
  raw_result = self._parse_document()
89
224
  return self._transform_result(raw_result)
@@ -96,6 +231,13 @@ class Parser:
96
231
  except (KeyError, IndexError, AttributeError) as e:
97
232
  # Catch specific parsing-related exceptions
98
233
  raise ParseError(f"Parse error: {str(e)}") from e
234
+ except RecursionError:
235
+ # Only reachable with a max_depth set higher than Python's recursion
236
+ # limit allows; the default is far below it.
237
+ raise ParseError(
238
+ "Parse error: the document is nested too deeply for Python's recursion limit; "
239
+ f"lower max_depth (currently {self.max_depth}) to refuse it with a located error"
240
+ ) from None
99
241
 
100
242
  def _skip_quoted_string(self, text: str, start: int) -> int:
101
243
  """
@@ -107,7 +249,7 @@ class Parser:
107
249
  parsed = _parse_quoted_string_at(text, start)
108
250
  return -1 if parsed is None else parsed[1]
109
251
 
110
- def _split_lines_respecting_quotes(self, text: str) -> List[str]:
252
+ def _split_lines_respecting_quotes(self, text: str, base: int) -> Tuple[List[str], List[int]]:
111
253
  """
112
254
  Split text into lines, but preserve newlines inside quoted strings
113
255
  and handle multiline parenthesized expressions.
@@ -115,47 +257,47 @@ class Parser:
115
257
  Quoted strings can span multiple lines, and newlines within them
116
258
  should be preserved as part of the string value. Also, parenthesized
117
259
  expressions that span multiple lines are kept together.
260
+
261
+ Returns the lines and where each of them starts in the document, given
262
+ that text starts at ``base``.
118
263
  """
119
264
  lines = []
120
- current_line = ""
265
+ offsets = []
266
+ line_start = 0
267
+ # Parentheses are counted in bulk between the characters that matter
268
+ # here, so a long run of them costs one pass in C rather than a Python
269
+ # step per character.
121
270
  paren_depth = 0
122
- i = 0
123
-
124
- while i < len(text):
125
- char = text[i]
271
+ counted = 0
272
+ position = 0
126
273
 
127
- if char in ('"', "'", "`"):
128
- end = self._skip_quoted_string(text, i)
129
- if end > i:
130
- # A quoted string is opaque: newlines inside it are content
131
- current_line += text[i:end]
132
- i = end
133
- continue
134
- current_line += char
135
- elif char == "(":
136
- paren_depth += 1
137
- current_line += char
138
- elif char == ")":
139
- paren_depth -= 1
140
- current_line += char
141
- elif char == "\n":
142
- if paren_depth > 0:
143
- # Inside unclosed parens: preserve the newline
144
- current_line += char
145
- else:
146
- # Parentheses balanced: this is a line break
147
- lines.append(current_line)
148
- current_line = ""
274
+ while True:
275
+ found = _LINE_BREAK_OR_QUOTE.search(text, position)
276
+ if found is None:
277
+ break
278
+ i = found.start()
279
+ paren_depth += text.count("(", counted, i) - text.count(")", counted, i)
280
+
281
+ if text[i] == "\n":
282
+ # Inside unclosed parens the newline is preserved; with the
283
+ # parentheses balanced it is a line break
284
+ if paren_depth <= 0:
285
+ lines.append(text[line_start:i])
286
+ offsets.append(base + line_start)
287
+ line_start = i + 1
288
+ position = i + 1
149
289
  else:
150
- current_line += char
151
-
152
- i += 1
290
+ # A quoted string is opaque: newlines inside it are content
291
+ end = self._skip_quoted_string(text, i)
292
+ position = end if end > i else i + 1
293
+ counted = position
153
294
 
154
295
  # Add the last line if non-empty
155
- if current_line:
156
- lines.append(current_line)
296
+ if line_start < len(text):
297
+ lines.append(text[line_start:])
298
+ offsets.append(base + line_start)
157
299
 
158
- return lines
300
+ return lines, offsets
159
301
 
160
302
  def _parse_document(self) -> List[Dict]:
161
303
  """Parse the entire document."""
@@ -165,7 +307,7 @@ class Parser:
165
307
  while self.pos < len(self.lines):
166
308
  line = self.lines[self.pos]
167
309
  if line.strip(): # Skip empty lines
168
- element = self._parse_element(0)
310
+ element = self._parse_element(0, 0)
169
311
  if element:
170
312
  links.append(element)
171
313
  else:
@@ -173,8 +315,13 @@ class Parser:
173
315
 
174
316
  return links
175
317
 
176
- def _parse_element(self, current_indent: int) -> Optional[Dict]:
177
- """Parse a single element (link or reference) at given indentation."""
318
+ def _parse_element(self, current_indent: int, level: int) -> Optional[Dict]:
319
+ """
320
+ Parse a single element (link or reference) at given indentation.
321
+
322
+ ``level`` is the number of indentation levels the element sits at in
323
+ the context being parsed.
324
+ """
178
325
  if self.pos >= len(self.lines):
179
326
  return None
180
327
 
@@ -196,14 +343,18 @@ class Parser:
196
343
  self.pos += 1
197
344
  return None
198
345
 
346
+ content_offset = self.line_offsets[self.pos] + len(line) - len(line.lstrip())
199
347
  self.pos += 1
200
348
 
201
349
  # Try to parse the line
202
- element = self._parse_line_content(content)
350
+ line_depth = self.context_depth + level
351
+ self.depth = line_depth
352
+ element = self._parse_line_content(content, content_offset)
353
+ # Only a line that parsed counts, as in the other implementations
354
+ self._check_depth(line_depth, content_offset)
203
355
 
204
356
  # Check for children (indented lines that follow)
205
357
  children = []
206
- child_indent = indent + 2 # Expect at least 2 spaces for child
207
358
 
208
359
  while self.pos < len(self.lines):
209
360
  # A line holding nothing does not close a block: the block goes on
@@ -226,7 +377,9 @@ class Parser:
226
377
 
227
378
  # This is a child
228
379
  self.pos = following
229
- child = self._parse_element(child_indent if not children else indent + 2)
380
+ # A child only has to be indented deeper than its parent; asking
381
+ # for more left a line indented by a single space unread forever.
382
+ child = self._parse_element(indent + 1, level + 1)
230
383
  if child:
231
384
  children.append(child)
232
385
 
@@ -235,11 +388,19 @@ class Parser:
235
388
 
236
389
  return element
237
390
 
238
- def _parse_line_content(self, content: str) -> Dict:
239
- """Parse the content of a single line."""
391
+ def _check_depth(self, depth: int, offset: int) -> None:
392
+ """
393
+ Refuse, for good, links at ``depth`` when that is deeper than the
394
+ parser allows. ``offset`` is where the level that is too deep opens.
395
+ """
396
+ if depth > self.max_depth:
397
+ raise ParseError.nesting_too_deep(self.source, offset, self.max_depth)
398
+
399
+ def _parse_line_content(self, content: str, offset: int) -> Dict:
400
+ """Parse the content of a single line, which starts at ``offset``."""
240
401
  # A whole parenthesized group: (id: values), (values) or a nested document
241
402
  if content.startswith("(") and self._find_matching_paren(content, 0) == len(content) - 1:
242
- return self._parse_parenthesized(content[1:-1])
403
+ return self._parse_parenthesized(content[1:-1], offset)
243
404
 
244
405
  # Try indented ID syntax: id:
245
406
  if content.endswith(":"):
@@ -251,69 +412,92 @@ class Parser:
251
412
  colon_pos = self._find_colon_outside_quotes(content)
252
413
  if colon_pos >= 0:
253
414
  id_part = content[:colon_pos].strip()
254
- values_part = content[colon_pos + 1 :].strip()
415
+ after_colon = content[colon_pos + 1 :]
416
+ values_part = after_colon.strip()
417
+ values_offset = offset + colon_pos + 1 + len(after_colon) - len(after_colon.lstrip())
255
418
  ref = self._extract_reference(id_part)
256
- values = self._parse_values(values_part)
419
+ values = self._parse_values(values_part, values_offset)
257
420
  return {"id": ref, "values": values}
258
421
 
259
422
  # Simple value list
260
- values = self._parse_values(content)
423
+ values = self._parse_values(content, offset)
261
424
  return {"values": values}
262
425
 
263
- def _parse_parenthesized(self, inner: str) -> Dict:
426
+ def _parse_parenthesized(self, inner: str, offset: int) -> Dict:
264
427
  """
265
- Parse the content of a parenthesized group.
428
+ Parse the content of a parenthesized group opened at ``offset``.
266
429
 
267
430
  The group opens a nested context that starts fresh at indentation level
268
431
  zero and follows exactly the rules used at the root of the document, so
269
- line breaks separate links and indentation nests them.
432
+ line breaks separate links and indentation nests them. The group is one
433
+ level deeper than the line it is written on.
270
434
  """
271
- return {"nested": self._parse_nested_document(inner)}
435
+ self._check_depth(self.depth + 1, offset)
436
+ return {"nested": self._parse_nested_document(inner, offset + 1)}
272
437
 
273
- def _parse_nested_document(self, inner: str) -> List[Dict]:
274
- """Parse the text of a parenthesized group as a document of its own."""
438
+ def _parse_nested_document(self, inner: str, offset: int) -> List[Dict]:
439
+ """
440
+ Parse the text of a parenthesized group, which starts at ``offset``,
441
+ as a document of its own.
442
+ """
275
443
  saved_lines = self.lines
444
+ saved_line_offsets = self.line_offsets
276
445
  saved_pos = self.pos
277
446
  saved_base_indentation = self.base_indentation
278
447
  saved_indentation_stack = self.indentation_stack
448
+ saved_context_depth = self.context_depth
449
+ saved_depth = self.depth
279
450
  try:
280
- self.lines = self._split_lines_respecting_quotes(inner)
451
+ self.lines, self.line_offsets = self._split_lines_respecting_quotes(inner, offset)
281
452
  self.pos = 0
282
453
  self.base_indentation = None
283
454
  self.indentation_stack = [0]
455
+ self.context_depth = self.depth + 1
284
456
  return self._parse_document()
285
457
  finally:
286
458
  self.lines = saved_lines
459
+ self.line_offsets = saved_line_offsets
287
460
  self.pos = saved_pos
288
461
  self.base_indentation = saved_base_indentation
289
462
  self.indentation_stack = saved_indentation_stack
463
+ self.context_depth = saved_context_depth
464
+ self.depth = saved_depth
290
465
 
291
466
  def _find_matching_paren(self, text: str, start: int) -> int:
292
467
  """
293
468
  Find the position of the parenthesis closing the one at start.
294
469
 
295
470
  Quoted strings are skipped, so parentheses inside them are ignored.
296
- Returns -1 when the group is not closed.
297
- """
298
- depth = 0
299
- i = start
471
+ Returns -1 when the group is not closed, or when start is not at an
472
+ opening parenthesis.
300
473
 
301
- while i < len(text):
302
- char = text[i]
303
- if char in ('"', "'", "`"):
304
- end = self._skip_quoted_string(text, i)
305
- if end > i:
306
- i = end
307
- continue
308
- elif char == "(":
309
- depth += 1
310
- elif char == ")":
311
- depth -= 1
312
- if depth == 0:
313
- return i
314
- i += 1
474
+ A run of parentheses is taken in one step, so the parentheses deeply
475
+ nested groups begin and end with cost a step per run rather than one
476
+ per character: every group is scanned once for each group around it.
477
+ """
478
+ if not text.startswith("(", start):
479
+ return -1
315
480
 
316
- return -1
481
+ depth = 0
482
+ position = start
483
+
484
+ while True:
485
+ for found in _PAREN_RUN_OR_QUOTE.finditer(text, position):
486
+ run_start, run_end = found.span()
487
+ char = text[run_start]
488
+ if char == "(":
489
+ depth += run_end - run_start
490
+ elif char == ")":
491
+ if run_end - run_start >= depth:
492
+ return run_start + depth - 1
493
+ depth -= run_end - run_start
494
+ else:
495
+ end = self._skip_quoted_string(text, run_start)
496
+ if end > run_start:
497
+ position = end
498
+ break
499
+ else:
500
+ return -1
317
501
 
318
502
  def _find_colon_outside_quotes(self, text: str) -> int:
319
503
  """
@@ -325,28 +509,28 @@ class Parser:
325
509
  because it's inside the second parenthesized expression.
326
510
  """
327
511
  paren_depth = 0
328
- i = 0
329
-
330
- while i < len(text):
331
- char = text[i]
332
- if char in ('"', "'", "`"):
512
+ counted = 0
513
+ position = 0
514
+
515
+ while True:
516
+ found = _COLON_OR_QUOTE.search(text, position)
517
+ if found is None:
518
+ return -1
519
+ i = found.start()
520
+ paren_depth += text.count("(", counted, i) - text.count(")", counted, i)
521
+
522
+ if text[i] == ":":
523
+ if paren_depth == 0:
524
+ # Only return colon if it's outside quotes AND at parenthesis depth 0
525
+ return i
526
+ position = i + 1
527
+ else:
333
528
  end = self._skip_quoted_string(text, i)
334
- if end > i:
335
- i = end
336
- continue
337
- elif char == "(":
338
- paren_depth += 1
339
- elif char == ")":
340
- paren_depth -= 1
341
- elif char == ":" and paren_depth == 0:
342
- # Only return colon if it's outside quotes AND at parenthesis depth 0
343
- return i
344
- i += 1
345
-
346
- return -1
529
+ position = end if end > i else i + 1
530
+ counted = position
347
531
 
348
- def _parse_values(self, text: str) -> List[Dict]:
349
- """Parse a space-separated list of values."""
532
+ def _parse_values(self, text: str, offset: int) -> List[Dict]:
533
+ """Parse a space-separated list of values, which starts at ``offset``."""
350
534
  if not text:
351
535
  return []
352
536
 
@@ -363,7 +547,7 @@ class Parser:
363
547
  # Try to extract the next value
364
548
  value_end, value_text = self._extract_next_value(text, i)
365
549
  if value_text and value_text.strip():
366
- values.append(self._parse_value(value_text))
550
+ values.append(self._parse_value(value_text, offset + i))
367
551
  if value_end == i:
368
552
  # No progress made - skip this character to avoid infinite loop
369
553
  i += 1
@@ -414,11 +598,11 @@ class Parser:
414
598
 
415
599
  return (i, text[start:i])
416
600
 
417
- def _parse_value(self, value: str) -> Dict:
418
- """Parse a single value (could be a reference or nested link)."""
601
+ def _parse_value(self, value: str, offset: int) -> Dict:
602
+ """Parse a single value (could be a reference or nested link) starting at ``offset``."""
419
603
  # Nested link in parentheses
420
604
  if value.startswith("(") and self._find_matching_paren(value, 0) == len(value) - 1:
421
- return self._parse_parenthesized(value[1:-1])
605
+ return self._parse_parenthesized(value[1:-1], offset)
422
606
 
423
607
  # Simple reference
424
608
  ref = self._extract_reference(value)
@@ -4,7 +4,7 @@ from dataclasses import dataclass
4
4
  from typing import AsyncIterable, AsyncIterator, Callable, Iterable, Iterator, List, Optional
5
5
 
6
6
  from .link import Link
7
- from .parser import ParseError, Parser
7
+ from .parser import DEFAULT_MAX_DEPTH, ParseError, Parser
8
8
  from .quotes import QUOTE_CHARS, _parse_quoted_string_at
9
9
 
10
10
 
@@ -19,14 +19,22 @@ class StreamPosition:
19
19
 
20
20
 
21
21
  class StreamParseError(ParseError):
22
- """A canonical parse error located within the complete stream."""
22
+ """A canonical parse error located within the complete stream.
23
+
24
+ ``error`` is the error the parser raised for the buffered record;
25
+ ``offset``, ``line`` and ``column`` locate it within the complete stream.
26
+ """
23
27
 
24
28
  def __init__(self, error: Exception, offset: int, line: int, column: int):
25
- super().__init__(f"Stream parse error at line {line}, column {column}: {error}")
29
+ super().__init__(
30
+ f"Stream parse error at line {line}, column {column}: {error}",
31
+ offset=offset,
32
+ line=line,
33
+ column=column,
34
+ line_text=getattr(error, "line_text", None),
35
+ max_depth=getattr(error, "max_depth", None),
36
+ )
26
37
  self.error = error
27
- self.offset = offset
28
- self.line = line
29
- self.column = column
30
38
 
31
39
 
32
40
  class StreamParser:
@@ -45,8 +53,9 @@ class StreamParser:
45
53
  collect: bool = True,
46
54
  max_buffer_size: Optional[int] = None,
47
55
  comments: bool = True,
56
+ max_depth: int = DEFAULT_MAX_DEPTH,
48
57
  ):
49
- self.parser = parser or Parser(comments=comments)
58
+ self.parser = parser or Parser(comments=comments, max_depth=max_depth)
50
59
  self.on_link = on_link
51
60
  self.collect = collect
52
61
  self.max_buffer_size = self.parser.max_input_size if max_buffer_size is None else max_buffer_size
@@ -95,7 +104,7 @@ class StreamParser:
95
104
  try:
96
105
  links = self.parser.parse(document)
97
106
  except Exception as error:
98
- raise StreamParseError(error, self._segment_offset, self._segment_line, 1) from error
107
+ raise self._stream_error(error) from error
99
108
  self._publish(links, emitted)
100
109
  self._advance_segment(document)
101
110
 
@@ -164,6 +173,15 @@ class StreamParser:
164
173
  if self.on_link is not None:
165
174
  self.on_link(link)
166
175
 
176
+ def _stream_error(self, error: Exception) -> StreamParseError:
177
+ """Locate an error the parser raised for the buffered record within the stream."""
178
+ line = getattr(error, "line", None)
179
+ column = getattr(error, "column", None)
180
+ offset = getattr(error, "offset", None)
181
+ if line is None or column is None or offset is None:
182
+ return StreamParseError(error, self._segment_offset, self._segment_line, 1)
183
+ return StreamParseError(error, self._segment_offset + offset, self._segment_line + line - 1, column)
184
+
167
185
  def _advance_segment(self, document: str) -> None:
168
186
  self._segment_offset += len(document)
169
187
  self._segment_line += document.count("\n")
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: links-notation
3
- Version: 0.21.2
3
+ Version: 0.22.0
4
4
  Summary: Python implementation of the Links Notation parser
5
5
  Author-email: "Link.Foundation" <drakonard@gmail.com>
6
6
  License-Expression: Unlicense
@@ -140,9 +140,42 @@ stream also exposes `position`, `drain()`, `reset()`, and a buffer-size limit.
140
140
 
141
141
  The main parser class for Links Notation.
142
142
 
143
- - `__init__(..., comments: bool = True)`: Create a parser; with `comments=False`
144
- a `#` is an ordinary character instead of the start of a comment
145
- - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link objects
143
+ - `__init__(max_input_size: int = 10 * 1024 * 1024, max_depth: int = DEFAULT_MAX_DEPTH, comments: bool = True)`:
144
+ Create a parser; with `comments=False` a `#` is an ordinary character instead
145
+ of the start of a comment
146
+ - `parse(input_text: str) -> List[Link]`: Parse Links Notation text into Link
147
+ objects; raises `ParseError` when the text does not parse or nests links
148
+ deeper than `max_depth`
149
+ - `DEFAULT_MAX_DEPTH`: The default `max_depth`, 64, the same in every
150
+ implementation
151
+
152
+ `max_depth` is how deep links may nest (default: 64). Every parenthesized group
153
+ and every indentation level is one level, and the lines of a document start at
154
+ level 0, so with `max_depth=1` `(a)` is accepted while `((a))`, `(a (b))` and a
155
+ group on an indented line are refused. A document nested deeper is refused with
156
+ a `ParseError` rather than recursed into until Python's recursion limit is hit.
157
+
158
+ ### ParseError
159
+
160
+ Raised when parsing fails. When a document nests links deeper than
161
+ `max_depth`, it points at the group or the line that is one level too deep:
162
+
163
+ ```text
164
+ Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3
165
+ 1 | ((((a))))
166
+ | ^
167
+ ```
168
+
169
+ - `max_depth`: The deepest nesting the parser accepts, when the document nests
170
+ deeper; `None` for any other error
171
+ - `line`, `column`: Where the offending group or line starts, counted from 1
172
+ - `offset`: The same position as a character offset from the start of the
173
+ document
174
+ - `line_text`: The offending line, as written
175
+
176
+ `StreamParser` reports the same error as a `StreamParseError` whose `error` is
177
+ the `ParseError` and whose `line`, `column` and `offset` are counted from the
178
+ start of the stream.
146
179
 
147
180
  ### Link
148
181
 
@@ -34,5 +34,6 @@ tests/test_multiline_quoted_string.py
34
34
  tests/test_nested_indentation.py
35
35
  tests/test_nested_parser.py
36
36
  tests/test_nested_self_reference.py
37
+ tests/test_nesting_limit.py
37
38
  tests/test_single_line_parser.py
38
39
  tests/test_stream_parser.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "links-notation"
7
- version = "0.21.2"
7
+ version = "0.22.0"
8
8
  description = "Python implementation of the Links Notation parser"
9
9
  readme = "README.md"
10
10
  license = "Unlicense"
@@ -0,0 +1,137 @@
1
+ """Links nested too deeply are refused with an error rather than recursed into
2
+ until the stack overflows
3
+ (https://github.com/link-foundation/links-notation/issues/315).
4
+
5
+ Every parenthesized group and every indentation level is one level, and the
6
+ lines of a document start at level 0. The positions asserted here are the ones
7
+ the Rust port reports for the same input.
8
+ """
9
+
10
+ import sys
11
+
12
+ import pytest
13
+
14
+ from links_notation import DEFAULT_MAX_DEPTH, ParseError, Parser, StreamParser
15
+ from links_notation.stream_parser import StreamParseError
16
+
17
+
18
+ def parens(depth):
19
+ return "(" * depth + "a" + ")" * depth
20
+
21
+
22
+ def values(depth):
23
+ return "(a " * depth + "b" + ")" * depth
24
+
25
+
26
+ def indentation(depth):
27
+ return "".join(" " * level + "a\n" for level in range(depth + 1))
28
+
29
+
30
+ def too_deep(document, max_depth):
31
+ with pytest.raises(ParseError) as caught:
32
+ Parser(max_depth=max_depth).parse(document)
33
+ assert caught.value.max_depth == max_depth
34
+ return caught.value
35
+
36
+
37
+ def accepted(document, max_depth=DEFAULT_MAX_DEPTH):
38
+ return len(Parser(max_depth=max_depth).parse(document)) > 0
39
+
40
+
41
+ def test_default_limit_is_shared_by_every_implementation():
42
+ assert DEFAULT_MAX_DEPTH == 64
43
+ assert Parser().max_depth == DEFAULT_MAX_DEPTH
44
+ assert StreamParser().parser.max_depth == DEFAULT_MAX_DEPTH
45
+
46
+
47
+ def test_parentheses_up_to_the_limit_are_accepted():
48
+ assert accepted(parens(3), 3)
49
+ assert accepted(values(3), 3)
50
+ assert accepted(parens(DEFAULT_MAX_DEPTH))
51
+
52
+
53
+ def test_parentheses_past_the_limit_are_refused_at_the_group_that_is_too_deep():
54
+ error = too_deep(parens(4), 3)
55
+
56
+ assert (error.line, error.column, error.offset) == (1, 4, 3)
57
+ assert str(error) == (
58
+ "Nesting too deep at line 1, column 4: nesting depth exceeds the maximum of 3\n" "1 | ((((a))))\n" " | ^"
59
+ )
60
+
61
+
62
+ def test_groups_in_value_position_count_like_any_other_group():
63
+ error = too_deep(values(4), 3)
64
+
65
+ assert (error.line, error.column) == (1, 10)
66
+
67
+
68
+ def test_indentation_up_to_the_limit_is_accepted():
69
+ assert accepted(indentation(3), 3)
70
+ assert accepted(indentation(DEFAULT_MAX_DEPTH))
71
+
72
+
73
+ def test_indentation_past_the_limit_is_refused_at_the_line_that_is_too_deep():
74
+ error = too_deep(indentation(4), 3)
75
+
76
+ assert (error.line, error.column) == (5, 5)
77
+ assert error.line_text == " a"
78
+
79
+
80
+ def test_groups_and_indentation_add_up():
81
+ # `(b)` on the line indented once is at level 2.
82
+ assert accepted("a\n (b)\n", 2)
83
+ error = too_deep("a\n (b)\n", 1)
84
+
85
+ assert (error.line, error.column) == (2, 3)
86
+
87
+
88
+ def test_limit_of_one_allows_one_group():
89
+ assert accepted("(a b)", 1)
90
+ assert accepted("a\n b\n", 1)
91
+ too_deep("((a))", 1)
92
+ too_deep("a\n b\n c\n", 1)
93
+
94
+
95
+ def test_trailing_spaces_on_a_deep_line_are_not_a_deeper_line():
96
+ assert accepted("a\n b\n c \n", 2)
97
+
98
+
99
+ def test_limit_past_the_recursion_limit_is_an_error_rather_than_a_crash():
100
+ # Nothing here is too deep for the limit, so the error has no max_depth.
101
+ with pytest.raises(ParseError) as caught:
102
+ Parser(max_depth=sys.maxsize).parse(parens(100_000))
103
+
104
+ assert caught.value.max_depth is None
105
+ assert not str(caught.value).startswith("Nesting too deep")
106
+
107
+
108
+ def test_parser_is_reusable_after_refusing_a_document():
109
+ parser = Parser(max_depth=2)
110
+ with pytest.raises(ParseError):
111
+ parser.parse(parens(3))
112
+
113
+ assert parser.parse(parens(2))
114
+ assert parser.parse(indentation(2))
115
+
116
+
117
+ @pytest.mark.timeout(30)
118
+ def test_refuses_a_document_far_past_the_limit_without_overflowing_the_stack():
119
+ # Before the limit existed each of these exhausted Python's recursion limit.
120
+ # Each group around the one too deep is scanned to its end before it is
121
+ # entered, so refusing values(n) reads the document once per level; it is
122
+ # kept short enough to be refused quickly.
123
+ for document in (parens(100_000), values(5_000), indentation(2_000)):
124
+ error = too_deep(document, DEFAULT_MAX_DEPTH)
125
+ assert str(error).startswith("Nesting too deep at ")
126
+
127
+
128
+ def test_stream_parser_reports_where_the_nesting_is_too_deep():
129
+ stream = StreamParser(max_depth=1)
130
+ stream.write("a\nb ((c))\n")
131
+
132
+ with pytest.raises(StreamParseError) as caught:
133
+ stream.finish()
134
+
135
+ assert caught.value.error.max_depth == 1
136
+ assert caught.value.max_depth == 1
137
+ assert (caught.value.line, caught.value.column, caught.value.offset) == (2, 4, 5)
File without changes