bpp-format 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bpp/__init__.py +36 -0
- bpp/__main__.py +7 -0
- bpp/cli.py +239 -0
- bpp/decoder.py +402 -0
- bpp/encoder.py +388 -0
- bpp/estimate.py +35 -0
- bpp/formats.py +205 -0
- bpp/lexer.py +241 -0
- bpp/markdown.py +160 -0
- bpp/tokens.py +185 -0
- bpp_format-0.3.0.dist-info/METADATA +327 -0
- bpp_format-0.3.0.dist-info/RECORD +16 -0
- bpp_format-0.3.0.dist-info/WHEEL +5 -0
- bpp_format-0.3.0.dist-info/entry_points.txt +2 -0
- bpp_format-0.3.0.dist-info/licenses/LICENSE +21 -0
- bpp_format-0.3.0.dist-info/top_level.txt +1 -0
bpp/__init__.py
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
"""bpp - a token-efficient text format for feeding structured data to LLMs.
|
|
2
|
+
|
|
3
|
+
import bpp
|
|
4
|
+
text = bpp.dumps({"users": [{"id": 1, "name": "Ayşe"}]}) # -> .bpp text
|
|
5
|
+
data = bpp.loads(text) # -> Python objects
|
|
6
|
+
data = bpp.load("config.yaml") # any supported file (.json .yaml .csv .md .bpp)
|
|
7
|
+
bpp.dump(data, "config.bpp") # format from the extension
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from .decoder import decode
|
|
13
|
+
from .encoder import encode
|
|
14
|
+
from .lexer import BppError
|
|
15
|
+
|
|
16
|
+
__version__ = "0.3.0"
|
|
17
|
+
__all__ = ["encode", "decode", "dumps", "loads", "load", "dump", "BppError", "__version__"]
|
|
18
|
+
|
|
19
|
+
dumps = encode
|
|
20
|
+
loads = decode
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def load(path):
|
|
24
|
+
"""Read a .json/.yaml/.csv/.md/.bpp file into Python objects."""
|
|
25
|
+
from .formats import detect, loads as _loads
|
|
26
|
+
|
|
27
|
+
return _loads(Path(path).read_text(encoding="utf-8"), detect(path))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def dump(data, path, **options):
|
|
31
|
+
"""Write data to a file; the extension picks the format (.bpp by default)."""
|
|
32
|
+
from .formats import detect, dumps as _dumps
|
|
33
|
+
|
|
34
|
+
fmt = detect(path) if Path(path).suffix else "bpp"
|
|
35
|
+
with open(path, "w", encoding="utf-8", newline="\n") as f: # newline=: Python 3.9
|
|
36
|
+
f.write(_dumps(data, fmt, **options))
|
bpp/__main__.py
ADDED
bpp/cli.py
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
"""Command line interface: bpp encode | decode | stats."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import os
|
|
7
|
+
import shutil
|
|
8
|
+
import subprocess
|
|
9
|
+
import sys
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from . import __version__
|
|
13
|
+
from .formats import FORMATS, detect, dumps, loads
|
|
14
|
+
|
|
15
|
+
|
|
16
|
+
def _read(path: str) -> str:
|
|
17
|
+
if path == "-":
|
|
18
|
+
return sys.stdin.read()
|
|
19
|
+
return Path(path).read_text(encoding="utf-8")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _write(path: str | None, text: str):
|
|
23
|
+
if not path or path == "-":
|
|
24
|
+
sys.stdout.write(text)
|
|
25
|
+
else:
|
|
26
|
+
with open(path, "w", encoding="utf-8", newline="\n") as f:
|
|
27
|
+
f.write(text)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _load(path: str, fmt: str | None):
|
|
31
|
+
fmt = fmt or (detect(path) if path != "-" else None)
|
|
32
|
+
if not fmt:
|
|
33
|
+
raise ValueError("reading stdin needs --from")
|
|
34
|
+
return loads(_read(path), fmt), fmt
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def cmd_encode(a) -> int:
|
|
38
|
+
data, fmt = _load(a.input, a.from_)
|
|
39
|
+
primer = {"none": False, "short": True, "long": "long"}[a.primer]
|
|
40
|
+
# CSV column order is part of the data, so never move a column there.
|
|
41
|
+
_write(a.output, dumps(data, "bpp", primer=primer, refs=not a.no_refs,
|
|
42
|
+
keep_order=a.keep_order or fmt == "csv"))
|
|
43
|
+
return 0
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
def cmd_decode(a) -> int:
|
|
47
|
+
data, _ = _load(a.input, a.from_ or "bpp")
|
|
48
|
+
to = a.to or (detect(a.output) if a.output and a.output != "-" else "json")
|
|
49
|
+
kw = {"indent": None if a.indent < 0 else a.indent} if to == "json" else {}
|
|
50
|
+
_write(a.output, dumps(data, to, **kw))
|
|
51
|
+
return 0
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
# ---------------------------------------------------------------------- stats
|
|
55
|
+
|
|
56
|
+
def _toon_bridge() -> list[str] | None:
|
|
57
|
+
"""Command that runs the reference TOON encoder, if available."""
|
|
58
|
+
node = shutil.which("node")
|
|
59
|
+
if not node:
|
|
60
|
+
return None
|
|
61
|
+
candidates = [os.environ.get("BPP_TOON_BRIDGE"),
|
|
62
|
+
Path.cwd() / "bench/toon/toon.mjs",
|
|
63
|
+
Path(__file__).resolve().parents[2] / "bench/toon/toon.mjs"]
|
|
64
|
+
for c in candidates:
|
|
65
|
+
if c and Path(c).exists() and (Path(c).parent / "node_modules").exists():
|
|
66
|
+
return [node, str(c), "encode"]
|
|
67
|
+
return None
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def renderings(data, keep_order: bool = False, markdown: bool = False) -> dict[str, str]:
|
|
71
|
+
"""The same data in every format we compare."""
|
|
72
|
+
import json
|
|
73
|
+
|
|
74
|
+
out = {
|
|
75
|
+
"JSON (indent 2)": dumps(data, "json"),
|
|
76
|
+
"JSON (minified)": dumps(data, "json", indent=None),
|
|
77
|
+
"YAML": dumps(data, "yaml"),
|
|
78
|
+
}
|
|
79
|
+
try:
|
|
80
|
+
out["CSV"] = dumps(data, "csv")
|
|
81
|
+
except ValueError:
|
|
82
|
+
pass
|
|
83
|
+
if markdown:
|
|
84
|
+
out["Markdown"] = dumps(data, "md")
|
|
85
|
+
bridge = _toon_bridge()
|
|
86
|
+
if bridge:
|
|
87
|
+
p = subprocess.run(bridge, input=json.dumps(data, ensure_ascii=False),
|
|
88
|
+
capture_output=True, text=True, encoding="utf-8")
|
|
89
|
+
if p.returncode == 0:
|
|
90
|
+
out["TOON"] = p.stdout
|
|
91
|
+
out["bpp"] = dumps(data, "bpp", keep_order=keep_order)
|
|
92
|
+
out["bpp + primer"] = dumps(data, "bpp", primer=True, keep_order=keep_order)
|
|
93
|
+
return out
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def cmd_stats(a) -> int:
|
|
97
|
+
from .tokens import available_counters
|
|
98
|
+
|
|
99
|
+
data, fmt = _load(a.input, a.from_)
|
|
100
|
+
counters = available_counters()
|
|
101
|
+
rows = [(name, text, {c: fn(text) for c, fn in counters.items()})
|
|
102
|
+
for name, text in renderings(data, keep_order=fmt == "csv",
|
|
103
|
+
markdown=fmt == "md").items()]
|
|
104
|
+
base = rows[0][2]
|
|
105
|
+
names = list(counters)
|
|
106
|
+
head = ["format", "chars"] + names + [f"vs JSON ({n})" for n in names]
|
|
107
|
+
table = [head]
|
|
108
|
+
for name, text, cnt in rows:
|
|
109
|
+
table.append([name, str(len(text))] + [str(cnt[n]) for n in names]
|
|
110
|
+
+ [f"{100 * (cnt[n] - base[n]) / base[n]:+.1f}%" for n in names])
|
|
111
|
+
if a.markdown:
|
|
112
|
+
print("| " + " | ".join(head) + " |")
|
|
113
|
+
print("|" + "---|" * len(head))
|
|
114
|
+
for r in table[1:]:
|
|
115
|
+
print("| " + " | ".join(r) + " |")
|
|
116
|
+
else:
|
|
117
|
+
w = [max(len(r[i]) for r in table) for i in range(len(head))]
|
|
118
|
+
for r in table:
|
|
119
|
+
print(" ".join(c.ljust(w[i]) if i == 0 else c.rjust(w[i]) for i, c in enumerate(r)))
|
|
120
|
+
if "estimate" in counters:
|
|
121
|
+
print("\nnote: 'estimate' is approximate; pip install tiktoken for real token counts.",
|
|
122
|
+
file=sys.stderr)
|
|
123
|
+
elif "anthropic" not in counters:
|
|
124
|
+
print("\nnote: o200k/claude2 are proxies; set ANTHROPIC_API_KEY for real Claude counts.",
|
|
125
|
+
file=sys.stderr)
|
|
126
|
+
return 0
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# ------------------------------------------------------------ one-step mode
|
|
130
|
+
|
|
131
|
+
def _savings(src: str, out: str) -> str:
|
|
132
|
+
try:
|
|
133
|
+
from .tokens import available_counters
|
|
134
|
+
|
|
135
|
+
name, fn = next(iter(available_counters().items()))
|
|
136
|
+
except Exception:
|
|
137
|
+
return ""
|
|
138
|
+
a, b = fn(src), fn(out)
|
|
139
|
+
if not a:
|
|
140
|
+
return ""
|
|
141
|
+
if name == "estimate":
|
|
142
|
+
return (f" (~{a} -> ~{b} tokens, {100 * (b - a) / a:+.0f}%, rough estimate; "
|
|
143
|
+
"pip install tiktoken for exact counts)")
|
|
144
|
+
return f" ({name}: {a} -> {b} tokens, {100 * (b - a) / a:+.0f}%)"
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def cmd_auto(a) -> int:
|
|
148
|
+
"""`bpp FILE`: .bpp files are decoded, everything else is encoded."""
|
|
149
|
+
src = Path(a.input)
|
|
150
|
+
if src.suffix.lower() == ".bpp":
|
|
151
|
+
to = a.to or (detect(a.output) if a.output and a.output != "-" else "json")
|
|
152
|
+
out = a.output or str(src.with_suffix("." + {"yaml": "yaml", "md": "md",
|
|
153
|
+
"csv": "csv"}.get(to, "json")))
|
|
154
|
+
if out != "-" and Path(out).exists() and not a.force:
|
|
155
|
+
raise ValueError(f"{out} exists; use -o to pick another name or --force to overwrite")
|
|
156
|
+
text = dumps(loads(_read(a.input), "bpp"), to)
|
|
157
|
+
_write(out, text)
|
|
158
|
+
else:
|
|
159
|
+
fmt = detect(src)
|
|
160
|
+
raw = _read(a.input)
|
|
161
|
+
out = a.output or str(src.with_suffix(".bpp"))
|
|
162
|
+
text = dumps(loads(raw, fmt), "bpp", primer=a.primer, keep_order=fmt == "csv")
|
|
163
|
+
_write(out, text)
|
|
164
|
+
if out != "-":
|
|
165
|
+
print(f"{src} -> {out}{_savings(raw, text)}", file=sys.stderr)
|
|
166
|
+
return 0
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _auto_main(argv: list[str]) -> int:
|
|
170
|
+
p = argparse.ArgumentParser(
|
|
171
|
+
prog="bpp", description="bpp FILE: encode json/yaml/csv/md to FILE.bpp, or decode "
|
|
172
|
+
"FILE.bpp to json. Subcommands: encode, decode, stats (bpp <cmd> -h).")
|
|
173
|
+
p.add_argument("input")
|
|
174
|
+
p.add_argument("-o", "--output", help="output file, '-' for stdout (default: next to input)")
|
|
175
|
+
p.add_argument("--to", choices=["json", "yaml", "csv", "md"], help="decode target format")
|
|
176
|
+
p.add_argument("--primer", action="store_true", help="add a one-line format explanation")
|
|
177
|
+
p.add_argument("-f", "--force", action="store_true", help="overwrite when decoding")
|
|
178
|
+
a = p.parse_args(argv)
|
|
179
|
+
try:
|
|
180
|
+
return cmd_auto(a)
|
|
181
|
+
except (ValueError, OSError) as ex:
|
|
182
|
+
print(f"bpp: error: {ex}", file=sys.stderr)
|
|
183
|
+
return 1
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def _utf8_console():
|
|
187
|
+
# Windows consoles default to a legacy code page; Turkish text would crash print().
|
|
188
|
+
for stream in (sys.stdout, sys.stderr):
|
|
189
|
+
try:
|
|
190
|
+
stream.reconfigure(encoding="utf-8")
|
|
191
|
+
except (AttributeError, ValueError):
|
|
192
|
+
pass
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def main(argv: list[str] | None = None) -> int:
|
|
196
|
+
_utf8_console()
|
|
197
|
+
argv = sys.argv[1:] if argv is None else argv
|
|
198
|
+
if argv and argv[0] not in ("encode", "decode", "stats", "-h", "--help", "--version"):
|
|
199
|
+
return _auto_main(argv)
|
|
200
|
+
p = argparse.ArgumentParser(prog="bpp", description="Token-efficient data format for LLMs. "
|
|
201
|
+
"Shortcut: bpp FILE (encode, or decode if FILE is .bpp).")
|
|
202
|
+
p.add_argument("--version", action="version", version=f"bpp {__version__}")
|
|
203
|
+
sub = p.add_subparsers(dest="cmd", required=True)
|
|
204
|
+
|
|
205
|
+
e = sub.add_parser("encode", help="convert json/yaml/csv/md to .bpp")
|
|
206
|
+
e.add_argument("input", help="input file ('-' for stdin with --from)")
|
|
207
|
+
e.add_argument("-o", "--output", help="output .bpp file (default stdout)")
|
|
208
|
+
e.add_argument("--from", dest="from_", choices=[f for f in FORMATS if f != "bpp"])
|
|
209
|
+
e.add_argument("--primer", choices=["none", "short", "long"], default="none",
|
|
210
|
+
help="prepend a format explanation for LLMs that have not seen bpp")
|
|
211
|
+
e.add_argument("--no-refs", action="store_true", help="disable the &n/*n dictionary")
|
|
212
|
+
e.add_argument("--keep-order", action="store_true",
|
|
213
|
+
help="never reorder keys (slightly larger output)")
|
|
214
|
+
e.set_defaults(fn=cmd_encode)
|
|
215
|
+
|
|
216
|
+
d = sub.add_parser("decode", help="convert .bpp to json/yaml/csv/md")
|
|
217
|
+
d.add_argument("input", help="input .bpp file ('-' for stdin)")
|
|
218
|
+
d.add_argument("-o", "--output", help="output file (default stdout)")
|
|
219
|
+
d.add_argument("--to", choices=["json", "yaml", "csv", "md"],
|
|
220
|
+
help="output format (default: from -o extension, else json)")
|
|
221
|
+
d.add_argument("--indent", type=int, default=2, help="JSON indent; -1 = minified")
|
|
222
|
+
d.set_defaults(fn=cmd_decode, from_=None)
|
|
223
|
+
|
|
224
|
+
s = sub.add_parser("stats", help="compare token counts across formats")
|
|
225
|
+
s.add_argument("input")
|
|
226
|
+
s.add_argument("--from", dest="from_", choices=list(FORMATS))
|
|
227
|
+
s.add_argument("--markdown", action="store_true", help="print a Markdown table")
|
|
228
|
+
s.set_defaults(fn=cmd_stats)
|
|
229
|
+
|
|
230
|
+
a = p.parse_args(argv)
|
|
231
|
+
try:
|
|
232
|
+
return a.fn(a)
|
|
233
|
+
except (ValueError, OSError) as ex:
|
|
234
|
+
print(f"bpp: error: {ex}", file=sys.stderr)
|
|
235
|
+
return 1
|
|
236
|
+
|
|
237
|
+
|
|
238
|
+
if __name__ == "__main__":
|
|
239
|
+
sys.exit(main())
|
bpp/decoder.py
ADDED
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
""".bpp text -> JSON data model (SPEC §1-§6)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from .lexer import BppError, Cursor
|
|
8
|
+
|
|
9
|
+
_COUNT_RE = re.compile(r"\[(\d+)\]")
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
def _header(text: str):
|
|
13
|
+
"""`[N]` or `[N]{...}` at the start of text -> (N, rest) or None."""
|
|
14
|
+
m = _COUNT_RE.match(text)
|
|
15
|
+
if not m or (m.end() < len(text) and text[m.end()] != "{"):
|
|
16
|
+
return None
|
|
17
|
+
return int(m.group(1)), text[m.end():]
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class _Spec:
|
|
21
|
+
"""Parsed column spec of a table or row table (see encoder._Spec)."""
|
|
22
|
+
|
|
23
|
+
def __init__(self, cols, delim, child, sub):
|
|
24
|
+
self.cols, self.delim, self.child, self.sub = cols, delim, child, sub
|
|
25
|
+
self.order = [c[0] for c in cols]
|
|
26
|
+
self.smode = {c[0]: c[3] for c in cols}
|
|
27
|
+
self.optional = {c[0] for c in cols if c[1]}
|
|
28
|
+
self.keyed = {c[0] for c in cols if c[2]}
|
|
29
|
+
self.positional = [p for p in self.order[:-1] if p not in self.keyed]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _set_path(obj: dict, path: tuple, v, line):
|
|
33
|
+
for k in path[:-1]:
|
|
34
|
+
nxt = obj.setdefault(k, {})
|
|
35
|
+
if not isinstance(nxt, dict):
|
|
36
|
+
raise BppError(f"column {'.'.join(path)!r} conflicts with {k!r}", line)
|
|
37
|
+
obj = nxt
|
|
38
|
+
obj[path[-1]] = v
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
class _Line:
|
|
42
|
+
__slots__ = ("depth", "text", "no")
|
|
43
|
+
|
|
44
|
+
def __init__(self, depth, text, no):
|
|
45
|
+
self.depth, self.text, self.no = depth, text, no
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def decode(text: str):
|
|
49
|
+
"""Decode .bpp text into Python/JSON values."""
|
|
50
|
+
raw = text.split("\n")
|
|
51
|
+
lines: list[_Line] = []
|
|
52
|
+
version = 0
|
|
53
|
+
for no, ln in enumerate(raw, 1):
|
|
54
|
+
if ln.endswith("\r"):
|
|
55
|
+
ln = ln[:-1]
|
|
56
|
+
stripped = ln.lstrip(" ")
|
|
57
|
+
if not stripped or stripped.startswith("#"):
|
|
58
|
+
continue
|
|
59
|
+
if not version:
|
|
60
|
+
if stripped not in ("bpp1", "bpp2", "bpp3"):
|
|
61
|
+
raise BppError("missing 'bpp3' header", no)
|
|
62
|
+
version = int(stripped[3])
|
|
63
|
+
continue
|
|
64
|
+
lines.append(_Line(len(ln) - len(stripped), stripped, no))
|
|
65
|
+
if not version:
|
|
66
|
+
raise BppError("missing 'bpp3' header")
|
|
67
|
+
return _Dec(lines, version).document()
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
class _Dec:
|
|
71
|
+
def __init__(self, lines: list[_Line], version: int = 2):
|
|
72
|
+
self.L = lines
|
|
73
|
+
self.version = version
|
|
74
|
+
self.i = 0
|
|
75
|
+
self.refs: list = []
|
|
76
|
+
|
|
77
|
+
# -- helpers ---------------------------------------------------------------
|
|
78
|
+
def cur(self, ln: _Line, start: int = 0) -> Cursor:
|
|
79
|
+
c = Cursor(ln.text, self.refs, ln.no)
|
|
80
|
+
c.i = start
|
|
81
|
+
return c
|
|
82
|
+
|
|
83
|
+
def depth_at(self, i: int) -> int:
|
|
84
|
+
return self.L[i].depth if i < len(self.L) else -1
|
|
85
|
+
|
|
86
|
+
# -- document ----------------------------------------------------------------
|
|
87
|
+
def document(self):
|
|
88
|
+
while self.i < len(self.L) and self.L[self.i].text.startswith("&"):
|
|
89
|
+
ln = self.L[self.i]
|
|
90
|
+
m = re.match(r"&(\d+) ", ln.text)
|
|
91
|
+
if not m or ln.depth or int(m.group(1)) != len(self.refs):
|
|
92
|
+
raise BppError("bad dictionary definition", ln.no)
|
|
93
|
+
c = self.cur(ln, m.end())
|
|
94
|
+
v = c.value()
|
|
95
|
+
if not isinstance(v, str) or not c.eof():
|
|
96
|
+
raise BppError("dictionary value must be a string", ln.no)
|
|
97
|
+
self.refs.append(v)
|
|
98
|
+
self.i += 1
|
|
99
|
+
if self.i >= len(self.L):
|
|
100
|
+
raise BppError("empty document")
|
|
101
|
+
first = self.L[self.i]
|
|
102
|
+
if first.depth:
|
|
103
|
+
raise BppError("unexpected indentation", first.no)
|
|
104
|
+
if self.i == len(self.L) - 1:
|
|
105
|
+
v = self._root_inline(first)
|
|
106
|
+
if v is not _NO:
|
|
107
|
+
self.i += 1
|
|
108
|
+
return v
|
|
109
|
+
if first.text.startswith("["):
|
|
110
|
+
v = self.keyless(first, 0)
|
|
111
|
+
else:
|
|
112
|
+
v = self.object(0)
|
|
113
|
+
if self.i < len(self.L):
|
|
114
|
+
raise BppError("unexpected content", self.L[self.i].no)
|
|
115
|
+
return v
|
|
116
|
+
|
|
117
|
+
def _root_inline(self, ln: _Line):
|
|
118
|
+
c = self.cur(ln)
|
|
119
|
+
ch = c.peek()
|
|
120
|
+
try:
|
|
121
|
+
if ch == '"':
|
|
122
|
+
v = c.jstring()
|
|
123
|
+
elif ch in "[{" or ch == "*":
|
|
124
|
+
v = c.value(stops=" ")
|
|
125
|
+
else:
|
|
126
|
+
v = c.token(" ")
|
|
127
|
+
if isinstance(v, str):
|
|
128
|
+
return _NO
|
|
129
|
+
except BppError:
|
|
130
|
+
return _NO
|
|
131
|
+
return v if c.eof() else _NO
|
|
132
|
+
|
|
133
|
+
# -- objects -----------------------------------------------------------------
|
|
134
|
+
def object(self, d: int) -> dict:
|
|
135
|
+
obj: dict = {}
|
|
136
|
+
while self.i < len(self.L):
|
|
137
|
+
ln = self.L[self.i]
|
|
138
|
+
if ln.depth < d:
|
|
139
|
+
break
|
|
140
|
+
if ln.depth > d:
|
|
141
|
+
raise BppError("unexpected indentation", ln.no)
|
|
142
|
+
if ln.text.startswith("- ") or ln.text == "-":
|
|
143
|
+
raise BppError("list item outside a list", ln.no)
|
|
144
|
+
self.i += 1
|
|
145
|
+
k, v = self.entry(ln, 0, d)
|
|
146
|
+
obj[k] = v
|
|
147
|
+
return obj
|
|
148
|
+
|
|
149
|
+
def entry(self, ln: _Line, start: int, d: int):
|
|
150
|
+
"""Parse `key ...` beginning at column `start`; `d` is its logical depth."""
|
|
151
|
+
c = self.cur(ln, start)
|
|
152
|
+
key = c.key()
|
|
153
|
+
rest = ln.text[c.i:]
|
|
154
|
+
if rest == "":
|
|
155
|
+
if self.depth_at(self.i) != d + 1:
|
|
156
|
+
raise BppError(f"key {key!r} has no value", ln.no)
|
|
157
|
+
return key, self.object(d + 1)
|
|
158
|
+
if rest[0] == " ":
|
|
159
|
+
c.i += 1
|
|
160
|
+
v = c.value()
|
|
161
|
+
if not c.eof():
|
|
162
|
+
raise BppError("trailing characters", ln.no)
|
|
163
|
+
return key, v
|
|
164
|
+
if rest[0] == "[":
|
|
165
|
+
return key, self.block(rest, ln, d)
|
|
166
|
+
raise BppError(f"expected space after key {key!r}", ln.no)
|
|
167
|
+
|
|
168
|
+
# -- arrays ------------------------------------------------------------------
|
|
169
|
+
def keyless(self, ln: _Line, d: int):
|
|
170
|
+
"""A line at depth d beginning with '[': header `[N]...` or inline list."""
|
|
171
|
+
h = _header(ln.text)
|
|
172
|
+
if h and (h[1] or self._items_follow(d)):
|
|
173
|
+
self.i += 1
|
|
174
|
+
return self.block(ln.text, ln, d)
|
|
175
|
+
c = self.cur(ln)
|
|
176
|
+
v = c.inline_list()
|
|
177
|
+
if not c.eof():
|
|
178
|
+
raise BppError("trailing characters", ln.no)
|
|
179
|
+
self.i += 1
|
|
180
|
+
return v
|
|
181
|
+
|
|
182
|
+
def _items_follow(self, d: int) -> bool:
|
|
183
|
+
j = self.i + 1
|
|
184
|
+
return j < len(self.L) and self.L[j].depth == d and (
|
|
185
|
+
self.L[j].text.startswith("- ") or self.L[j].text == "-")
|
|
186
|
+
|
|
187
|
+
def block(self, head: str, ln: _Line, d: int) -> list:
|
|
188
|
+
h = _header(head)
|
|
189
|
+
if not h:
|
|
190
|
+
raise BppError("bad array header", ln.no)
|
|
191
|
+
n, rest = h
|
|
192
|
+
if not rest:
|
|
193
|
+
return self.items(n, d, ln)
|
|
194
|
+
c = self.cur(_Line(0, rest, ln.no))
|
|
195
|
+
spec = self._spec(c)
|
|
196
|
+
if not c.eof():
|
|
197
|
+
raise BppError("bad array header", ln.no)
|
|
198
|
+
if spec.child is None and not spec.optional and (len(spec.cols) == 1 or spec.delim == ","):
|
|
199
|
+
return self.table(n, spec, d, ln)
|
|
200
|
+
return self.outline(n, spec, d, ln)
|
|
201
|
+
|
|
202
|
+
def _spec(self, c: Cursor) -> _Spec:
|
|
203
|
+
dotted = self.version >= 3
|
|
204
|
+
c.expect("{")
|
|
205
|
+
cols = []
|
|
206
|
+
delim = None
|
|
207
|
+
while True:
|
|
208
|
+
path = c.path(dotted)
|
|
209
|
+
opt = keyed = False
|
|
210
|
+
if c.peek() == "?":
|
|
211
|
+
opt = True
|
|
212
|
+
c.i += 1
|
|
213
|
+
# bpp2+: `x?` = positional, '-' when absent; `x?=` = written as x=v.
|
|
214
|
+
# bpp1 only had the x=v form, spelled `x?`.
|
|
215
|
+
if self.version == 1:
|
|
216
|
+
keyed = True
|
|
217
|
+
elif c.peek() == "=":
|
|
218
|
+
keyed = True
|
|
219
|
+
c.i += 1
|
|
220
|
+
smode = False
|
|
221
|
+
if c.s.startswith(":str", c.i):
|
|
222
|
+
smode = True
|
|
223
|
+
c.i += 4
|
|
224
|
+
cols.append((path, opt, keyed, smode))
|
|
225
|
+
sep = c.peek()
|
|
226
|
+
if sep == "}":
|
|
227
|
+
c.i += 1
|
|
228
|
+
break
|
|
229
|
+
if delim is None:
|
|
230
|
+
delim = sep
|
|
231
|
+
if sep != delim or sep not in ", ":
|
|
232
|
+
c.err("bad column list")
|
|
233
|
+
c.i += 1
|
|
234
|
+
child = sub = None
|
|
235
|
+
if c.peek() == ">":
|
|
236
|
+
c.i += 1
|
|
237
|
+
child = c.segment() if dotted else c.key()
|
|
238
|
+
if dotted and c.peek() == "{":
|
|
239
|
+
sub = self._spec(c)
|
|
240
|
+
return _Spec(cols, delim or ",", child, sub)
|
|
241
|
+
|
|
242
|
+
def table(self, n: int, spec: _Spec, d: int, ln: _Line) -> list:
|
|
243
|
+
rows = []
|
|
244
|
+
for _ in range(n):
|
|
245
|
+
if self.i >= len(self.L) or self.L[self.i].depth != d:
|
|
246
|
+
raise BppError(f"expected {n} table rows", ln.no)
|
|
247
|
+
r = self.L[self.i]
|
|
248
|
+
self.i += 1
|
|
249
|
+
c = self.cur(r)
|
|
250
|
+
obj: dict = {}
|
|
251
|
+
for j, p in enumerate(spec.order):
|
|
252
|
+
_set_path(obj, p, c.token(",", spec.smode[p]), r.no)
|
|
253
|
+
if j < len(spec.order) - 1:
|
|
254
|
+
c.expect(",")
|
|
255
|
+
if not c.eof():
|
|
256
|
+
raise BppError("too many cells", r.no)
|
|
257
|
+
rows.append(obj)
|
|
258
|
+
return rows
|
|
259
|
+
|
|
260
|
+
def outline(self, n: int, spec: _Spec, d: int, ln: _Line) -> list:
|
|
261
|
+
dotted = self.version >= 3
|
|
262
|
+
|
|
263
|
+
def check(sp):
|
|
264
|
+
if sp.order[-1] in sp.optional:
|
|
265
|
+
raise BppError("last column must be required", ln.no)
|
|
266
|
+
if sp.sub is not None:
|
|
267
|
+
check(sp.sub)
|
|
268
|
+
check(spec)
|
|
269
|
+
|
|
270
|
+
def row(r: _Line, sp: _Spec):
|
|
271
|
+
c = self.cur(r)
|
|
272
|
+
got = {}
|
|
273
|
+
child_empty = False
|
|
274
|
+
for p in sp.positional:
|
|
275
|
+
if p in sp.optional and c.s.startswith("- ", c.i):
|
|
276
|
+
c.i += 2 # '-' = this optional column is absent
|
|
277
|
+
continue
|
|
278
|
+
got[p] = self._cell(c, sp.smode[p])
|
|
279
|
+
c.expect(" ")
|
|
280
|
+
while True:
|
|
281
|
+
save = c.i
|
|
282
|
+
if c.peek() == '"' or c.peek() not in "[{*":
|
|
283
|
+
try:
|
|
284
|
+
p = c.path(dotted)
|
|
285
|
+
except BppError:
|
|
286
|
+
c.i = save
|
|
287
|
+
break
|
|
288
|
+
if c.peek() == "=":
|
|
289
|
+
if sp.child is not None and p == (sp.child,) and not child_empty:
|
|
290
|
+
c.i += 1
|
|
291
|
+
if c.value(" ") != []:
|
|
292
|
+
raise BppError("child key may only be [] inline", r.no)
|
|
293
|
+
child_empty = True
|
|
294
|
+
c.expect(" ")
|
|
295
|
+
continue
|
|
296
|
+
if p in sp.keyed and p not in got:
|
|
297
|
+
c.i += 1
|
|
298
|
+
got[p] = self._cell(c, sp.smode[p])
|
|
299
|
+
c.expect(" ")
|
|
300
|
+
continue
|
|
301
|
+
c.i = save
|
|
302
|
+
break
|
|
303
|
+
last = sp.order[-1]
|
|
304
|
+
got[last] = c.value(strmode=sp.smode[last])
|
|
305
|
+
if not c.eof():
|
|
306
|
+
raise BppError("trailing characters", r.no)
|
|
307
|
+
obj: dict = {}
|
|
308
|
+
for p in sp.order:
|
|
309
|
+
if p in got:
|
|
310
|
+
_set_path(obj, p, got[p], r.no)
|
|
311
|
+
if child_empty:
|
|
312
|
+
obj[sp.child] = []
|
|
313
|
+
return obj
|
|
314
|
+
|
|
315
|
+
def rows_at(depth: int, count, sp: _Spec) -> list:
|
|
316
|
+
out = []
|
|
317
|
+
while self.i < len(self.L) and (count is None or len(out) < count):
|
|
318
|
+
r = self.L[self.i]
|
|
319
|
+
if r.depth < depth:
|
|
320
|
+
break
|
|
321
|
+
if r.depth > depth:
|
|
322
|
+
raise BppError("unexpected indentation", r.no)
|
|
323
|
+
self.i += 1
|
|
324
|
+
obj = row(r, sp)
|
|
325
|
+
if sp.child is not None and self.depth_at(self.i) == depth + 1:
|
|
326
|
+
if sp.child in obj:
|
|
327
|
+
raise BppError("child rows after child=[]", r.no)
|
|
328
|
+
obj[sp.child] = rows_at(depth + 1, None, sp.sub or sp)
|
|
329
|
+
out.append(obj)
|
|
330
|
+
if count is not None and len(out) != count:
|
|
331
|
+
raise BppError(f"expected {count} rows", ln.no)
|
|
332
|
+
return out
|
|
333
|
+
|
|
334
|
+
return rows_at(d, n, spec)
|
|
335
|
+
|
|
336
|
+
@staticmethod
|
|
337
|
+
def _cell(c: Cursor, smode: bool):
|
|
338
|
+
if c.peek() == "[":
|
|
339
|
+
return c.inline_list(smode)
|
|
340
|
+
return c.value(" ", smode)
|
|
341
|
+
|
|
342
|
+
def items(self, n: int, d: int, ln: _Line) -> list:
|
|
343
|
+
out = []
|
|
344
|
+
for _ in range(n):
|
|
345
|
+
if self.i >= len(self.L) or self.L[self.i].depth != d:
|
|
346
|
+
raise BppError(f"expected {n} list items", ln.no)
|
|
347
|
+
it = self.L[self.i]
|
|
348
|
+
if not it.text.startswith("- "):
|
|
349
|
+
raise BppError("expected '- ' list item", it.no)
|
|
350
|
+
self.i += 1
|
|
351
|
+
out.append(self.item(it, d))
|
|
352
|
+
return out
|
|
353
|
+
|
|
354
|
+
def item(self, it: _Line, d: int):
|
|
355
|
+
body = it.text[2:]
|
|
356
|
+
sub = _Line(d + 1, body, it.no)
|
|
357
|
+
if body.startswith("["):
|
|
358
|
+
h = _header(body)
|
|
359
|
+
if h and (h[1] or (
|
|
360
|
+
self.i < len(self.L) and self.L[self.i].depth == d + 1
|
|
361
|
+
and self.L[self.i].text.startswith("- "))):
|
|
362
|
+
return self.block(body, sub, d + 1)
|
|
363
|
+
c = self.cur(sub)
|
|
364
|
+
v = c.inline_list()
|
|
365
|
+
if not c.eof():
|
|
366
|
+
raise BppError("trailing characters", it.no)
|
|
367
|
+
return v
|
|
368
|
+
if body == "{}":
|
|
369
|
+
return {}
|
|
370
|
+
c = self.cur(sub)
|
|
371
|
+
try:
|
|
372
|
+
key = c.key()
|
|
373
|
+
rest = body[c.i:]
|
|
374
|
+
except BppError:
|
|
375
|
+
key, rest = None, None
|
|
376
|
+
if key is not None:
|
|
377
|
+
is_entry = (rest.startswith(" ") or rest.startswith("[")
|
|
378
|
+
or (rest == "" and self.depth_at(self.i) == d + 2))
|
|
379
|
+
if is_entry:
|
|
380
|
+
k, v = self.entry(sub, 0, d + 1)
|
|
381
|
+
obj = {k: v}
|
|
382
|
+
more = self.object(d + 1)
|
|
383
|
+
for kk, vv in more.items():
|
|
384
|
+
if kk in obj:
|
|
385
|
+
raise BppError(f"duplicate key {kk!r}", it.no)
|
|
386
|
+
obj[kk] = vv
|
|
387
|
+
return obj
|
|
388
|
+
c = self.cur(sub)
|
|
389
|
+
if c.peek() == '"':
|
|
390
|
+
v = c.jstring()
|
|
391
|
+
else:
|
|
392
|
+
v = c.value()
|
|
393
|
+
if not c.eof():
|
|
394
|
+
raise BppError("trailing characters", it.no)
|
|
395
|
+
return v
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
class _Sentinel:
|
|
399
|
+
pass
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
_NO = _Sentinel()
|