linked-data-python 0.0.4__py3-none-any.whl → 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ldpy/__init__.py +35 -11
- ldpy/__main__.py +75 -250
- ldpy/build.py +91 -0
- ldpy/console.py +116 -0
- ldpy/debug.py +235 -0
- ldpy/formatter.py +329 -0
- ldpy/importer.py +110 -0
- ldpy/lsp/__init__.py +11 -0
- ldpy/lsp/__main__.py +4 -0
- ldpy/lsp/backend.py +118 -0
- ldpy/lsp/rpc.py +104 -0
- ldpy/lsp/server.py +353 -0
- ldpy/lsp/translate.py +140 -0
- ldpy/pygments_lexer.py +613 -0
- ldpy/runtime.py +931 -0
- ldpy/sparql.py +551 -0
- ldpy/transpiler/__init__.py +14 -0
- ldpy/transpiler/core.py +2558 -0
- ldpy/transpiler/errors.py +38 -0
- ldpy/transpiler/linemap.py +292 -0
- linked_data_python-0.2.0.dist-info/METADATA +158 -0
- linked_data_python-0.2.0.dist-info/RECORD +26 -0
- {linked_data_python-0.0.4.dist-info → linked_data_python-0.2.0.dist-info}/WHEEL +1 -1
- linked_data_python-0.2.0.dist-info/entry_points.txt +9 -0
- {linked_data_python-0.0.4.dist-info → linked_data_python-0.2.0.dist-info/licenses}/LICENSE.md +0 -0
- {linked_data_python-0.0.4.dist-info → linked_data_python-0.2.0.dist-info}/top_level.txt +0 -0
- ldpy/grun/lib.py +0 -63
- ldpy/grun/util.py +0 -11
- ldpy/ldpy.py +0 -183
- ldpy/rewriter/IndentedStringWriter.py +0 -54
- ldpy/rewriter/LDPythonRewriter.py +0 -677
- ldpy/rewriter/MultiChannelTokenStream.py +0 -127
- ldpy/rewriter/Result.py +0 -49
- ldpy/rewriter/__init__.py +0 -7
- ldpy/rewriter/antlr/LDPythonLexer.py +0 -870
- ldpy/rewriter/antlr/LDPythonParser.py +0 -9336
- ldpy/rewriter/antlr/LDPythonVisitor.py +0 -573
- ldpy/sparql/builtin.py +0 -326
- linked_data_python-0.0.4.dist-info/METADATA +0 -139
- linked_data_python-0.0.4.dist-info/RECORD +0 -20
- linked_data_python-0.0.4.dist-info/entry_points.txt +0 -3
ldpy/pygments_lexer.py
ADDED
|
@@ -0,0 +1,613 @@
|
|
|
1
|
+
"""Pygments lexer for Linked-Data Python.
|
|
2
|
+
|
|
3
|
+
Two ideas, and almost no third one.
|
|
4
|
+
|
|
5
|
+
**The highlighter is the transpiler.** Rather than re-specify the island
|
|
6
|
+
triggers in a third grammar (after the transpiler's scanner and the TextMate
|
|
7
|
+
grammar of the VS Code extension), this lexer transpiles the source and reads
|
|
8
|
+
the resulting :class:`~ldpy.transpiler.linemap.LanguageMap` — an ordered
|
|
9
|
+
partition of the file into ``copy`` and ``island:KIND`` segments. Islands
|
|
10
|
+
therefore highlight exactly where the transpiler sees them: a disambiguation
|
|
11
|
+
rule that changes in ``ldpy/002`` changes the colouring with no
|
|
12
|
+
edit here.
|
|
13
|
+
|
|
14
|
+
**The tokenising is Pygments'.** Nothing here re-describes Python, Turtle or
|
|
15
|
+
SPARQL:
|
|
16
|
+
|
|
17
|
+
* ``copy`` segments and every ``{expr}`` interpolation go to ``PythonLexer``;
|
|
18
|
+
* ``@prefix`` / ``@base`` are Turtle's own directives, and go to
|
|
19
|
+
``TurtleLexer``;
|
|
20
|
+
* island bodies go to ``SparqlLexer``. A ``g{ }``, ``m{ }``, ``+{ }`` or
|
|
21
|
+
``-{ }`` body is Turtle *with variables*, which is precisely SPARQL's
|
|
22
|
+
triples block — ``TurtleLexer`` rejects ``?s``, ``SparqlLexer`` does not.
|
|
23
|
+
|
|
24
|
+
What is written here is only what no existing lexer can know: where the
|
|
25
|
+
islands are (the map answers that), the ldpy-only declarations (``@graph``,
|
|
26
|
+
``@bindings``, prefix imports), and the *masking* that lets a delegated lexer
|
|
27
|
+
see a well-formed document — each ldpy-specific region is replaced by a
|
|
28
|
+
placeholder of the same length that is a valid term where it stands, then the
|
|
29
|
+
real text is lexed back in at that position.
|
|
30
|
+
|
|
31
|
+
When the source does not transpile — an editor buffer mid-keystroke, an
|
|
32
|
+
illustrative snippet — the lexer degrades to plain Python rather than guessing.
|
|
33
|
+
|
|
34
|
+
Pygments is an optional dependency: ``pip install linked-data-python[highlight]``
|
|
35
|
+
(and it comes with the ``docs`` extra). Nothing else in ldpy imports this
|
|
36
|
+
module; Pygments loads it through an entry point when it is installed.
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
try:
|
|
40
|
+
from pygments.lexer import Lexer
|
|
41
|
+
from pygments.lexers.python import PythonLexer
|
|
42
|
+
from pygments.lexers.rdf import SparqlLexer, TurtleLexer
|
|
43
|
+
from pygments.token import (Comment, Error, Generic, Keyword, Name, Number,
|
|
44
|
+
Operator, Punctuation, String, Text)
|
|
45
|
+
except ImportError as exc: # pragma: no cover
|
|
46
|
+
raise ImportError(
|
|
47
|
+
"the ldpy Pygments lexer needs Pygments: "
|
|
48
|
+
"pip install linked-data-python[highlight]") from exc
|
|
49
|
+
|
|
50
|
+
__all__ = ["LdpyLexer"]
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
# --------------------------------------------------------------- token choices
|
|
54
|
+
#
|
|
55
|
+
# Only STANDARD Pygments token types, so that every style colours ldpy with no
|
|
56
|
+
# extra stylesheet — but not always the ones pygments.lexers.rdf picks:
|
|
57
|
+
# mkdocs-material collapses Name.Label, Name.Tag, Keyword and Keyword.Pseudo
|
|
58
|
+
# onto ONE colour, which would make IRIs and local names indistinguishable from
|
|
59
|
+
# keywords. The delegated lexers' tokens are remapped (see _REMAP_*) so that
|
|
60
|
+
# these eight roles land on eight colours:
|
|
61
|
+
#
|
|
62
|
+
# keyword sigils, declarations, `a`, SPARQL keywords
|
|
63
|
+
# string IRIs, blank-node labels, RDF literals
|
|
64
|
+
# function prefixed names
|
|
65
|
+
# variable ?v $v
|
|
66
|
+
# constant SPARQL built-ins and language tags
|
|
67
|
+
# number / operator / punctuation / comment as usual
|
|
68
|
+
|
|
69
|
+
T_SIGIL = Keyword.Pseudo # g{ f< e{ ?{ m{ s{ +{ -{ and their closers
|
|
70
|
+
T_DECL = Keyword.Declaration # @prefix @base @graph @bindings
|
|
71
|
+
T_KW = Keyword # a, as, in, for, global, SELECT, WHERE...
|
|
72
|
+
T_IRI = String.Symbol # <http://...> and _:label
|
|
73
|
+
T_PREFIX = Name.Namespace # the prefix part of a prefixed name
|
|
74
|
+
T_LOCAL = Name.Class # its local part
|
|
75
|
+
T_VAR = Name.Variable # ?v $v
|
|
76
|
+
T_LANG = Name.Builtin # the language tag of "x"@en
|
|
77
|
+
T_CONST = Keyword.Constant # true false
|
|
78
|
+
|
|
79
|
+
#: SparqlLexer → the palette above. Everything absent passes through.
|
|
80
|
+
_REMAP_SPARQL = {
|
|
81
|
+
Name.Label: T_IRI, # IRIREF and BLANK_NODE_LABEL
|
|
82
|
+
Name.Tag: T_LOCAL, # local part of a prefixed name
|
|
83
|
+
Name.Function: T_LANG, # language tags *and* built-in functions
|
|
84
|
+
Keyword.Type: T_KW,
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
#: TurtleLexer → the same palette (it spells IRIREF Name.Variable).
|
|
88
|
+
_REMAP_TURTLE = {
|
|
89
|
+
Name.Variable: T_IRI,
|
|
90
|
+
Name.Label: T_IRI,
|
|
91
|
+
Name.Tag: T_LOCAL,
|
|
92
|
+
Generic.Emph: T_LANG, # language tag
|
|
93
|
+
Keyword.Type: T_KW,
|
|
94
|
+
Keyword: T_DECL, # @prefix / @base
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
# ------------------------------------------------------------------- utilities
|
|
99
|
+
|
|
100
|
+
import re # noqa: E402
|
|
101
|
+
|
|
102
|
+
_STRING_START = re.compile(r"""[rRbBuUfF]{0,3}('''|\"\"\"|'|")""")
|
|
103
|
+
_PNAME_TAIL = re.compile(r"[\wÀ-][-\w.·À-]*:"
|
|
104
|
+
r"[\w·À-%\\][-\w.·À-%\\]*$")
|
|
105
|
+
_PNAME_COLON = re.compile(r"[\wÀ-][-\w.·À-]*:$")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _skip_string(text, i):
|
|
109
|
+
"""Index just past the string literal starting at *i*, or None."""
|
|
110
|
+
m = _STRING_START.match(text, i)
|
|
111
|
+
if not m:
|
|
112
|
+
return None
|
|
113
|
+
quote, j = m.group(1), m.end()
|
|
114
|
+
while j < len(text):
|
|
115
|
+
if text[j] == "\\":
|
|
116
|
+
j += 2
|
|
117
|
+
continue
|
|
118
|
+
if text.startswith(quote, j):
|
|
119
|
+
return j + len(quote)
|
|
120
|
+
j += 1
|
|
121
|
+
return len(text)
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def _match_brace(text, i):
|
|
125
|
+
"""Index just past the ``}`` closing the ``{`` at *i*, brackets and Python
|
|
126
|
+
strings respected. Returns ``len(text)`` if unterminated."""
|
|
127
|
+
depth, j = 0, i
|
|
128
|
+
while j < len(text):
|
|
129
|
+
c = text[j]
|
|
130
|
+
if c in "\"'":
|
|
131
|
+
nxt = _skip_string(text, j)
|
|
132
|
+
j = nxt if nxt is not None else j + 1
|
|
133
|
+
continue
|
|
134
|
+
if c in "([{":
|
|
135
|
+
depth += 1
|
|
136
|
+
elif c in ")]}":
|
|
137
|
+
depth -= 1
|
|
138
|
+
if depth == 0:
|
|
139
|
+
return j + 1
|
|
140
|
+
j += 1
|
|
141
|
+
return len(text)
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _is_interpolation(body):
|
|
145
|
+
"""The transpiler's oracle for `{...}` inside s{ } (fiche 015): balanced
|
|
146
|
+
content is a Python interpolation iff it transpiles and compiles as an
|
|
147
|
+
expression. A SPARQL group never does."""
|
|
148
|
+
body = body.strip()
|
|
149
|
+
if not body:
|
|
150
|
+
return False
|
|
151
|
+
try:
|
|
152
|
+
from ldpy.transpiler import transpile
|
|
153
|
+
compile(transpile(body, "<pygments>").code, "<pygments>", "eval")
|
|
154
|
+
return True
|
|
155
|
+
except Exception:
|
|
156
|
+
return False
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _line_offsets(text):
|
|
160
|
+
offs, pos = [0], 0
|
|
161
|
+
for line in text.splitlines(True):
|
|
162
|
+
pos += len(line)
|
|
163
|
+
offs.append(pos)
|
|
164
|
+
return offs
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
# ----------------------------------------------------------------- delegation
|
|
168
|
+
|
|
169
|
+
def _delegate(sub, remap, text, offset, regions):
|
|
170
|
+
"""Lex *text* with the *sub* Pygments lexer, remapping its token types and
|
|
171
|
+
splicing the ldpy *regions* back in.
|
|
172
|
+
|
|
173
|
+
``regions`` is a list of ``(start, end, emit)``; ``emit(offset)`` yields the
|
|
174
|
+
tokens of that region. The text handed to *sub* has each region replaced by
|
|
175
|
+
a same-length placeholder, so positions are preserved exactly."""
|
|
176
|
+
masked = list(text)
|
|
177
|
+
for start, end, _ in regions:
|
|
178
|
+
masked[start:end] = _placeholder(text, start, end)
|
|
179
|
+
masked = "".join(masked)
|
|
180
|
+
assert len(masked) == len(text)
|
|
181
|
+
|
|
182
|
+
spans = [(s, e) for s, e, _ in regions]
|
|
183
|
+
pieces = []
|
|
184
|
+
for start, end, emit in regions:
|
|
185
|
+
pieces.extend(emit(offset))
|
|
186
|
+
for idx, ttype, value in sub.get_tokens_unprocessed(masked):
|
|
187
|
+
ttype = remap.get(ttype, ttype)
|
|
188
|
+
s, e = idx, idx + len(value)
|
|
189
|
+
while s < e:
|
|
190
|
+
covering = next(((a, b) for a, b in spans if a <= s < b), None)
|
|
191
|
+
if covering:
|
|
192
|
+
s = covering[1]
|
|
193
|
+
continue
|
|
194
|
+
nxt = min([a for a, b in spans if a > s] + [e])
|
|
195
|
+
pieces.append((offset + s, ttype, text[s:min(e, nxt)]))
|
|
196
|
+
s = min(e, nxt)
|
|
197
|
+
pieces.sort(key=lambda p: p[0])
|
|
198
|
+
return pieces
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _placeholder(text, start, end):
|
|
202
|
+
"""A same-length stand-in that the delegated lexer accepts as a term where
|
|
203
|
+
the real region stands."""
|
|
204
|
+
n = end - start
|
|
205
|
+
glued = text[:start] # NO rstrip: adjacency matters
|
|
206
|
+
if _PNAME_COLON.search(glued[-64:]):
|
|
207
|
+
return "x" * n # ex:{expr} -> ex:xxxxx
|
|
208
|
+
if text.startswith("_:", start):
|
|
209
|
+
return "_:" + "x" * (n - 2) # _:{expr} -> _:xxxxx
|
|
210
|
+
if glued.endswith("^^") or (text[start] in "fe" and n > 2
|
|
211
|
+
and text[start + 1] == "<"):
|
|
212
|
+
return "<" + "x" * (n - 2) + ">" # an IRI where an IRI fits
|
|
213
|
+
return '"' + "x" * (n - 2) + '"' if n >= 2 else "x" * n
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
# ------------------------------------------------------------- island scanning
|
|
217
|
+
|
|
218
|
+
_DELIMITED = { # kind -> (opener length, closer, flavour)
|
|
219
|
+
"graph": (2, "}", "sparql"),
|
|
220
|
+
"match": (2, "}", "sparql"),
|
|
221
|
+
"addto": (2, "}", "sparql"),
|
|
222
|
+
"removefrom": (2, "}", "sparql"),
|
|
223
|
+
"sparql": (2, "}", "sparql"),
|
|
224
|
+
"enode": (2, "}", "sparql"),
|
|
225
|
+
"fnode": (2, "}", "python"), # f{ or ?{
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
_DECL_KEYWORDS = frozenset(("global", "nonlocal", "as", "in", "for", "from",
|
|
229
|
+
"import"))
|
|
230
|
+
_RE_WS = re.compile(r"[ \t\r\n]+")
|
|
231
|
+
_RE_NAME = re.compile(r"[A-Za-z_]\w*")
|
|
232
|
+
_RE_IRIREF = re.compile(r"<[^<>\"{}|^`\\\x00-\x20]*>")
|
|
233
|
+
_RE_PNAME = re.compile(r"([\wÀ-][-\w.·À-]*)?(:)"
|
|
234
|
+
r"([\w·À-%\\][-\w.·À-%\\]*)?")
|
|
235
|
+
_RE_PUNCT = re.compile(r"[;,.\[\]():]")
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
#: A snippet is rarely a whole module. Documentation and papers show
|
|
239
|
+
#: `g{ ex:s ex:p 1 }` without the `@prefix` line that makes it legal, and the
|
|
240
|
+
#: transpiler — rightly — refuses it. Rather than fall back to plain Python
|
|
241
|
+
#: (which paints `ex:` and `?s` bright red), the lexer DECLARES what the
|
|
242
|
+
#: fragment is missing and tries again: the snippet then colours exactly as
|
|
243
|
+
#: the same lines would inside a complete file.
|
|
244
|
+
_UNDECLARED = re.compile(
|
|
245
|
+
r"préfixe non déclaré\s*:\s*'([^':]*):'"
|
|
246
|
+
r"|[Uu]nknown namespace prefix\s*:\s*(\S*)")
|
|
247
|
+
|
|
248
|
+
#: The declaration a fragment is missing, read off the error it raised. Only
|
|
249
|
+
#: errors that a DECLARATION can repair are listed: a snippet that is actually
|
|
250
|
+
#: ill-formed must keep failing, and fall through to plain Python.
|
|
251
|
+
_NO_CURRENT_GRAPH = re.compile(r"sans graphe courant")
|
|
252
|
+
|
|
253
|
+
#: Enough for any realistic snippet; the bound is what keeps a pathological
|
|
254
|
+
#: input from looping.
|
|
255
|
+
_MAX_SYNTHETIC = 24
|
|
256
|
+
|
|
257
|
+
#: Namespace of the synthetic prefixes. Never resolved — the map is thrown
|
|
258
|
+
#: away after the tokens are read — but it has to be a legal IRI.
|
|
259
|
+
_SYNTHETIC_NS = "urn:x-ldpy-highlight:"
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _synthetic_declaration(exc):
|
|
263
|
+
"""The declaration line that would let *exc* go away, or None."""
|
|
264
|
+
m = _UNDECLARED.search(str(exc))
|
|
265
|
+
if m is not None:
|
|
266
|
+
prefix = (m.group(1) if m.group(1) is not None else m.group(2)).strip()
|
|
267
|
+
return f"@prefix {prefix}: <{_SYNTHETIC_NS}{prefix or 'default'}#> ."
|
|
268
|
+
if _NO_CURRENT_GRAPH.search(str(exc)):
|
|
269
|
+
return "@graph as _ldpy_highlight_graph"
|
|
270
|
+
return None
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
def _transpile_for_display(text):
|
|
274
|
+
"""``(segments, preamble)`` — the language map of *text*, possibly read
|
|
275
|
+
under a synthetic preamble that declares the prefixes it never declared.
|
|
276
|
+
|
|
277
|
+
The preamble is prepended, so every coordinate it produces is shifted by
|
|
278
|
+
``len(preamble)``; the caller subtracts it. Raises whatever the transpiler
|
|
279
|
+
raises when no preamble can help."""
|
|
280
|
+
from ldpy.transpiler import transpile
|
|
281
|
+
preamble = ""
|
|
282
|
+
seen = set()
|
|
283
|
+
while True:
|
|
284
|
+
try:
|
|
285
|
+
return transpile(preamble + text, "<pygments>").map.segments, preamble
|
|
286
|
+
except Exception as exc:
|
|
287
|
+
line = _synthetic_declaration(exc)
|
|
288
|
+
if line is None or line in seen or len(seen) >= _MAX_SYNTHETIC:
|
|
289
|
+
raise
|
|
290
|
+
seen.add(line)
|
|
291
|
+
preamble += line + "\n"
|
|
292
|
+
|
|
293
|
+
|
|
294
|
+
def _without_errors(python, text):
|
|
295
|
+
"""*text* through PythonLexer, with ``Token.Error`` flattened to text.
|
|
296
|
+
|
|
297
|
+
Last resort: the fragment is not ldpy the transpiler can read, even with a
|
|
298
|
+
preamble. Whatever it is, painting it in error red is a worse answer than
|
|
299
|
+
painting it plainly — the highlighter is not the place where a syntax
|
|
300
|
+
error gets reported."""
|
|
301
|
+
for idx, ttype, value in python.get_tokens_unprocessed(text):
|
|
302
|
+
yield idx, (Text if ttype is Error else ttype), value
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
class LdpyLexer(Lexer):
|
|
306
|
+
"""Lexer for Linked-Data Python (``.ldpy``) sources."""
|
|
307
|
+
|
|
308
|
+
name = "Linked-Data Python"
|
|
309
|
+
aliases = ["ldpy", "linked-data-python"]
|
|
310
|
+
filenames = ["*.ldpy"]
|
|
311
|
+
mimetypes = ["text/x-ldpy"]
|
|
312
|
+
url = "https://linked-data-python.readthedocs.io/"
|
|
313
|
+
|
|
314
|
+
def __init__(self, **options):
|
|
315
|
+
Lexer.__init__(self, **options)
|
|
316
|
+
self.python = PythonLexer(**options)
|
|
317
|
+
self.turtle = TurtleLexer(**options)
|
|
318
|
+
self.sparql = SparqlLexer(**options)
|
|
319
|
+
|
|
320
|
+
# -- entry point ---------------------------------------------------------
|
|
321
|
+
|
|
322
|
+
def get_tokens_unprocessed(self, text):
|
|
323
|
+
try:
|
|
324
|
+
segments, preamble = _transpile_for_display(text)
|
|
325
|
+
except Exception: # not ldpy, or not yet valid: plain Python
|
|
326
|
+
yield from _without_errors(self.python, text)
|
|
327
|
+
return
|
|
328
|
+
|
|
329
|
+
shift = len(preamble)
|
|
330
|
+
offs = _line_offsets(preamble + text)
|
|
331
|
+
limit = shift + len(text)
|
|
332
|
+
|
|
333
|
+
def abs_pos(line, col):
|
|
334
|
+
at = min(offs[line] + col, limit) if line < len(offs) else limit
|
|
335
|
+
return at - shift
|
|
336
|
+
|
|
337
|
+
pos = 0
|
|
338
|
+
for seg in segments:
|
|
339
|
+
if seg.src is None: # synthetic prelude
|
|
340
|
+
continue
|
|
341
|
+
end = abs_pos(seg.src[2], seg.src[3])
|
|
342
|
+
if end <= pos: # inside the preamble
|
|
343
|
+
continue
|
|
344
|
+
start = max(abs_pos(seg.src[0], seg.src[1]), pos)
|
|
345
|
+
if start > pos: # never lose a character
|
|
346
|
+
yield pos, Text, text[pos:start]
|
|
347
|
+
chunk = text[start:end]
|
|
348
|
+
if seg.kind == "copy":
|
|
349
|
+
for idx, ttype, value in \
|
|
350
|
+
self.python.get_tokens_unprocessed(chunk):
|
|
351
|
+
yield start + idx, ttype, value
|
|
352
|
+
else:
|
|
353
|
+
yield from self._island(seg.kind[len("island:"):], chunk, start)
|
|
354
|
+
pos = end
|
|
355
|
+
if pos < len(text):
|
|
356
|
+
yield pos, Text, text[pos:]
|
|
357
|
+
|
|
358
|
+
# -- islands -------------------------------------------------------------
|
|
359
|
+
|
|
360
|
+
def _island(self, kind, text, offset):
|
|
361
|
+
if kind in _DELIMITED:
|
|
362
|
+
yield from self._delimited(kind, text, offset)
|
|
363
|
+
elif kind in ("prefix", "base"):
|
|
364
|
+
# Turtle's own directives, lexed by Turtle's own lexer
|
|
365
|
+
yield from _delegate(self.turtle, _REMAP_TURTLE, text, offset,
|
|
366
|
+
self._regions(text, "turtle"))
|
|
367
|
+
elif kind in ("firi", "eiri"):
|
|
368
|
+
yield offset, T_SIGIL, text[:2]
|
|
369
|
+
yield from self._firi_body(text, 2, len(text) - 1, offset)
|
|
370
|
+
yield offset + len(text) - 1, T_SIGIL, text[-1:]
|
|
371
|
+
elif kind in ("iri", "var", "pname", "literal"):
|
|
372
|
+
yield from _delegate(self.sparql, _REMAP_SPARQL, text, offset,
|
|
373
|
+
self._regions(text, "sparql"))
|
|
374
|
+
else: # @graph, @bindings, for @bindings, import
|
|
375
|
+
yield from self._declaration(text, offset)
|
|
376
|
+
|
|
377
|
+
def _delimited(self, kind, text, offset):
|
|
378
|
+
head, closer, flavour = _DELIMITED[kind]
|
|
379
|
+
yield offset, T_SIGIL, text[:head]
|
|
380
|
+
# the island ends at the brace matching its opener; a call suffix
|
|
381
|
+
# (fiche 019) may follow, and that is ordinary Python
|
|
382
|
+
close = _match_brace(text, head - 1) - 1
|
|
383
|
+
body_end = close if 0 <= close < len(text) and text[close] == closer \
|
|
384
|
+
else (len(text) - len(closer) if text.endswith(closer)
|
|
385
|
+
else len(text))
|
|
386
|
+
if flavour == "python":
|
|
387
|
+
yield from self._python_chunk(text, head, body_end, offset)
|
|
388
|
+
else:
|
|
389
|
+
body = text[head:body_end]
|
|
390
|
+
yield from _delegate(self.sparql, _REMAP_SPARQL, body,
|
|
391
|
+
offset + head,
|
|
392
|
+
self._regions(body, "sparql", is_query=(
|
|
393
|
+
kind == "sparql")))
|
|
394
|
+
if body_end < len(text):
|
|
395
|
+
yield offset + body_end, T_SIGIL, text[body_end:body_end + 1]
|
|
396
|
+
if body_end + 1 < len(text): # the call suffix
|
|
397
|
+
for idx, ttype, value in self.python.get_tokens_unprocessed(
|
|
398
|
+
text[body_end + 1:]):
|
|
399
|
+
yield offset + body_end + 1 + idx, ttype, value
|
|
400
|
+
|
|
401
|
+
# -- the ldpy-specific regions of an island body -------------------------
|
|
402
|
+
|
|
403
|
+
def _regions(self, text, flavour, is_query=False):
|
|
404
|
+
"""Locate what no Turtle or SPARQL lexer can read: interpolations and
|
|
405
|
+
nested islands. Each becomes a (start, end, emit) region."""
|
|
406
|
+
regions = []
|
|
407
|
+
i, n = 0, len(text)
|
|
408
|
+
while i < n:
|
|
409
|
+
c = text[i]
|
|
410
|
+
if c == "#": # comment: sub-lexer
|
|
411
|
+
i = text.find("\n", i)
|
|
412
|
+
if i < 0:
|
|
413
|
+
break
|
|
414
|
+
continue
|
|
415
|
+
if c in "\"'" or (c in "fFrRbB" and i + 1 < n
|
|
416
|
+
and text[i + 1] in "\"'"):
|
|
417
|
+
end = _skip_string(text, i)
|
|
418
|
+
if end is None:
|
|
419
|
+
i += 1
|
|
420
|
+
continue
|
|
421
|
+
if text[i] in "fF": # f-string: has holes
|
|
422
|
+
regions.append((i, end, self._emit_fstring(text, i, end)))
|
|
423
|
+
i = end
|
|
424
|
+
continue
|
|
425
|
+
if text.startswith("_:{", i):
|
|
426
|
+
end = _match_brace(text, i + 2)
|
|
427
|
+
regions.append((i, end, self._emit_bnode(text, i, end)))
|
|
428
|
+
i = end
|
|
429
|
+
continue
|
|
430
|
+
if (text.startswith("e{", i) or text.startswith("f{", i)
|
|
431
|
+
or text.startswith("?{", i)) and i + 2 <= n:
|
|
432
|
+
end = _match_brace(text, i + 1)
|
|
433
|
+
regions.append((i, end, self._emit_nested(text, i, end)))
|
|
434
|
+
i = end
|
|
435
|
+
continue
|
|
436
|
+
if text.startswith("e<", i) or text.startswith("f<", i):
|
|
437
|
+
end = text.find(">", i)
|
|
438
|
+
end = n if end < 0 else end + 1
|
|
439
|
+
regions.append((i, end, self._emit_firi(text, i, end)))
|
|
440
|
+
i = end
|
|
441
|
+
continue
|
|
442
|
+
if c == "{":
|
|
443
|
+
end = _match_brace(text, i)
|
|
444
|
+
if is_query and not _is_interpolation(text[i + 1:end - 1]):
|
|
445
|
+
i += 1 # a SPARQL group
|
|
446
|
+
continue
|
|
447
|
+
regions.append((i, end, self._emit_interp(text, i, end)))
|
|
448
|
+
i = end
|
|
449
|
+
continue
|
|
450
|
+
i += 1
|
|
451
|
+
return regions
|
|
452
|
+
|
|
453
|
+
# -- region emitters -----------------------------------------------------
|
|
454
|
+
|
|
455
|
+
def _python_chunk(self, text, start, end, offset):
|
|
456
|
+
"""The Python expression text[start:end], through PythonLexer.
|
|
457
|
+
|
|
458
|
+
An interpolation is re-scanned by the transpiler's own scanner, so it
|
|
459
|
+
may itself hold an island — ``ex:{?id}``, ``{f<http://e/{x}>}``.
|
|
460
|
+
PythonLexer would mark those an error; when it does, the chunk goes to
|
|
461
|
+
the SPARQL lexer instead."""
|
|
462
|
+
chunk = text[start:end]
|
|
463
|
+
out = list(self.python.get_tokens_unprocessed(chunk))
|
|
464
|
+
if any(ttype is Error for _, ttype, _ in out):
|
|
465
|
+
return _delegate(self.sparql, _REMAP_SPARQL, chunk, offset + start,
|
|
466
|
+
self._regions(chunk, "sparql"))
|
|
467
|
+
return [(offset + start + idx, ttype, value)
|
|
468
|
+
for idx, ttype, value in out]
|
|
469
|
+
|
|
470
|
+
def _emit_interp(self, text, start, end):
|
|
471
|
+
def emit(offset):
|
|
472
|
+
out = [(offset + start, Punctuation, "{")]
|
|
473
|
+
out.extend(self._python_chunk(text, start + 1, end - 1, offset))
|
|
474
|
+
out.append((offset + end - 1, Punctuation, "}"))
|
|
475
|
+
return out
|
|
476
|
+
return emit
|
|
477
|
+
|
|
478
|
+
def _emit_nested(self, text, start, end):
|
|
479
|
+
"""e{ … } / f{ … } / ?{ … } in term position."""
|
|
480
|
+
def emit(offset):
|
|
481
|
+
out = [(offset + start, T_SIGIL, text[start:start + 2])]
|
|
482
|
+
body = text[start + 2:end - 1]
|
|
483
|
+
if text[start] == "e":
|
|
484
|
+
out.extend(_delegate(self.sparql, _REMAP_SPARQL, body,
|
|
485
|
+
offset + start + 2,
|
|
486
|
+
self._regions(body, "sparql")))
|
|
487
|
+
else:
|
|
488
|
+
out.extend(self._python_chunk(text, start + 2, end - 1, offset))
|
|
489
|
+
out.append((offset + end - 1, T_SIGIL, text[end - 1:end]))
|
|
490
|
+
return out
|
|
491
|
+
return emit
|
|
492
|
+
|
|
493
|
+
def _emit_firi(self, text, start, end):
|
|
494
|
+
def emit(offset):
|
|
495
|
+
out = [(offset + start, T_SIGIL, text[start:start + 2])]
|
|
496
|
+
out.extend(self._firi_body(text, start + 2, end - 1, offset))
|
|
497
|
+
out.append((offset + end - 1, T_SIGIL, text[end - 1:end]))
|
|
498
|
+
return out
|
|
499
|
+
return emit
|
|
500
|
+
|
|
501
|
+
def _emit_bnode(self, text, start, end):
|
|
502
|
+
def emit(offset):
|
|
503
|
+
out = [(offset + start, T_IRI, "_:"),
|
|
504
|
+
(offset + start + 2, Punctuation, "{")]
|
|
505
|
+
out.extend(self._python_chunk(text, start + 3, end - 1, offset))
|
|
506
|
+
out.append((offset + end - 1, Punctuation, "}"))
|
|
507
|
+
return out
|
|
508
|
+
return emit
|
|
509
|
+
|
|
510
|
+
def _emit_fstring(self, text, start, end):
|
|
511
|
+
def emit(offset):
|
|
512
|
+
return list(self._fstring(text, start, end, offset))
|
|
513
|
+
return emit
|
|
514
|
+
|
|
515
|
+
def _fstring(self, text, start, end, offset):
|
|
516
|
+
"""An f-string in term position: its holes are Python."""
|
|
517
|
+
i = start
|
|
518
|
+
while i < end:
|
|
519
|
+
j = text.find("{", i, end)
|
|
520
|
+
if j < 0:
|
|
521
|
+
yield offset + i, String, text[i:end]
|
|
522
|
+
return
|
|
523
|
+
if text.startswith("{{", j):
|
|
524
|
+
yield offset + i, String, text[i:j + 2]
|
|
525
|
+
i = j + 2
|
|
526
|
+
continue
|
|
527
|
+
if j > i:
|
|
528
|
+
yield offset + i, String, text[i:j]
|
|
529
|
+
k = min(_match_brace(text, j), end)
|
|
530
|
+
yield offset + j, String.Interpol, "{"
|
|
531
|
+
for tok in self._python_chunk(text, j + 1, k - 1, offset):
|
|
532
|
+
yield tok
|
|
533
|
+
yield offset + k - 1, String.Interpol, "}"
|
|
534
|
+
i = k
|
|
535
|
+
|
|
536
|
+
def _firi_body(self, text, start, end, offset):
|
|
537
|
+
"""The body of f<...> / e<...>: literal IRI text with {holes}."""
|
|
538
|
+
i = start
|
|
539
|
+
while i < end:
|
|
540
|
+
j = text.find("{", i, end)
|
|
541
|
+
if j < 0:
|
|
542
|
+
yield offset + i, T_IRI, text[i:end]
|
|
543
|
+
return
|
|
544
|
+
if j > i:
|
|
545
|
+
yield offset + i, T_IRI, text[i:j]
|
|
546
|
+
k = min(_match_brace(text, j), end)
|
|
547
|
+
yield offset + j, Punctuation, "{"
|
|
548
|
+
body = text[j + 1:k - 1]
|
|
549
|
+
if body.lstrip()[:1] in ("?", "$"): # e< > holds SPARQL
|
|
550
|
+
for tok in _delegate(self.sparql, _REMAP_SPARQL, body,
|
|
551
|
+
offset + j + 1,
|
|
552
|
+
self._regions(body, "sparql")):
|
|
553
|
+
yield tok
|
|
554
|
+
else:
|
|
555
|
+
for tok in self._python_chunk(text, j + 1, k - 1, offset):
|
|
556
|
+
yield tok
|
|
557
|
+
yield offset + k - 1, Punctuation, "}"
|
|
558
|
+
i = k
|
|
559
|
+
|
|
560
|
+
# -- ldpy-only declarations ----------------------------------------------
|
|
561
|
+
|
|
562
|
+
def _declaration(self, text, offset):
|
|
563
|
+
"""@graph, @bindings, `for @bindings in`, prefix imports: forms that
|
|
564
|
+
exist in no other language, so no lexer to delegate to."""
|
|
565
|
+
i, n = 0, len(text)
|
|
566
|
+
while i < n:
|
|
567
|
+
c = text[i]
|
|
568
|
+
m = _RE_WS.match(text, i)
|
|
569
|
+
if m:
|
|
570
|
+
yield offset + i, Text, m.group(0)
|
|
571
|
+
i = m.end()
|
|
572
|
+
continue
|
|
573
|
+
if c == "@":
|
|
574
|
+
m = _RE_NAME.match(text, i + 1)
|
|
575
|
+
if m:
|
|
576
|
+
yield offset + i, T_DECL, text[i:m.end()]
|
|
577
|
+
i = m.end()
|
|
578
|
+
continue
|
|
579
|
+
if text.startswith("f<", i) or text.startswith("e<", i):
|
|
580
|
+
end = text.find(">", i)
|
|
581
|
+
end = n if end < 0 else end + 1
|
|
582
|
+
yield offset + i, T_SIGIL, text[i:i + 2]
|
|
583
|
+
yield from self._firi_body(text, i + 2, end - 1, offset)
|
|
584
|
+
yield offset + end - 1, T_SIGIL, text[end - 1:end]
|
|
585
|
+
i = end
|
|
586
|
+
continue
|
|
587
|
+
m = _RE_IRIREF.match(text, i)
|
|
588
|
+
if m:
|
|
589
|
+
yield offset + i, T_IRI, m.group(0)
|
|
590
|
+
i = m.end()
|
|
591
|
+
continue
|
|
592
|
+
m = _RE_PNAME.match(text, i)
|
|
593
|
+
if m and m.group(2) and (m.group(1) or m.group(3)):
|
|
594
|
+
if m.group(1):
|
|
595
|
+
yield offset + i, T_PREFIX, m.group(1)
|
|
596
|
+
yield offset + m.start(2), Punctuation, ":"
|
|
597
|
+
if m.group(3):
|
|
598
|
+
yield offset + m.start(3), T_LOCAL, m.group(3)
|
|
599
|
+
i = m.end()
|
|
600
|
+
continue
|
|
601
|
+
m = _RE_NAME.match(text, i)
|
|
602
|
+
if m:
|
|
603
|
+
word = m.group(0)
|
|
604
|
+
yield offset + i, T_KW if word in _DECL_KEYWORDS else Name, word
|
|
605
|
+
i = m.end()
|
|
606
|
+
continue
|
|
607
|
+
m = _RE_PUNCT.match(text, i)
|
|
608
|
+
if m:
|
|
609
|
+
yield offset + i, Punctuation, m.group(0)
|
|
610
|
+
i = m.end()
|
|
611
|
+
continue
|
|
612
|
+
yield offset + i, Text, c
|
|
613
|
+
i += 1
|