bpp-format 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bpp/__init__.py +36 -0
- bpp/__main__.py +7 -0
- bpp/cli.py +239 -0
- bpp/decoder.py +402 -0
- bpp/encoder.py +388 -0
- bpp/estimate.py +35 -0
- bpp/formats.py +205 -0
- bpp/lexer.py +241 -0
- bpp/markdown.py +160 -0
- bpp/tokens.py +185 -0
- bpp_format-0.3.0.dist-info/METADATA +327 -0
- bpp_format-0.3.0.dist-info/RECORD +16 -0
- bpp_format-0.3.0.dist-info/WHEEL +5 -0
- bpp_format-0.3.0.dist-info/entry_points.txt +2 -0
- bpp_format-0.3.0.dist-info/licenses/LICENSE +21 -0
- bpp_format-0.3.0.dist-info/top_level.txt +1 -0
bpp/encoder.py
ADDED
|
@@ -0,0 +1,388 @@
|
|
|
1
|
+
"""JSON data model -> .bpp text (SPEC §1-§6)."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from collections import Counter
|
|
6
|
+
|
|
7
|
+
from .estimate import est_tokens
|
|
8
|
+
from .lexer import NUM_RE, Ref, fmt_inline, fmt_key, fmt_scalar, fmt_seg, is_scalar, jstr
|
|
9
|
+
|
|
10
|
+
HEADER = "bpp3"
|
|
11
|
+
PRIMER = ("# bpp3: JSON as 'key value' lines, 1-space indent nests. k[N]{a b.c}: N rows, values "
|
|
12
|
+
"in column order (b.c = key c of b), last one = rest of line; x? optional (- = absent), "
|
|
13
|
+
"x?= as x=v; >k: indented rows are k. \"...\" = JSON string, *n = &n.")
|
|
14
|
+
PRIMER_LONG = (
|
|
15
|
+
"# bpp3 = JSON data. Lines are 'key value'; a bare 'key' opens a nested object (1-space indent). [a,b] = list.\n"
|
|
16
|
+
"# k[N]{a b.c d}: N rows, values space-separated in column order, b.c = key c inside object b, "
|
|
17
|
+
"last column = rest of line;\n"
|
|
18
|
+
"# x? = optional, '-' if absent; x?= written as x=v; >kids: indented rows are kids, >kids{...} gives "
|
|
19
|
+
"them their own columns. {a,b}: comma rows. k[N]: N '- ' items. \"...\" = JSON string. *n = &n value."
|
|
20
|
+
)
|
|
21
|
+
CHILD_KEYS = ("steps", "children", "subtasks", "tasks", "items", "nodes")
|
|
22
|
+
MIN_REF_LEN = 8
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def encode(data, *, primer: bool | str = False, refs: bool = True,
|
|
26
|
+
keep_order: bool = False) -> str:
|
|
27
|
+
"""Encode a JSON-compatible value as .bpp text.
|
|
28
|
+
|
|
29
|
+
primer: add a one-line (True) or three-line ("long") explanation comment.
|
|
30
|
+
refs: allow the &n/*n dictionary for repeated long strings.
|
|
31
|
+
keep_order: never reorder object keys (disables the row-table variant that
|
|
32
|
+
moves its free-text column last).
|
|
33
|
+
"""
|
|
34
|
+
out = [HEADER]
|
|
35
|
+
if primer:
|
|
36
|
+
out.append(PRIMER_LONG if primer == "long" else PRIMER)
|
|
37
|
+
defs: list[str] = []
|
|
38
|
+
if refs:
|
|
39
|
+
data, defs = _extract_refs(data)
|
|
40
|
+
for i, s in enumerate(defs):
|
|
41
|
+
out.append(f"&{i} {fmt_scalar(s)}")
|
|
42
|
+
out.extend(_Enc(keep_order).root(data))
|
|
43
|
+
return "\n".join(out) + "\n"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# ------------------------------------------------------------------ dictionary
|
|
47
|
+
|
|
48
|
+
def _walk_strings(v, cnt: Counter):
|
|
49
|
+
if isinstance(v, str):
|
|
50
|
+
cnt[v] += 1
|
|
51
|
+
elif isinstance(v, dict):
|
|
52
|
+
for x in v.values():
|
|
53
|
+
_walk_strings(x, cnt)
|
|
54
|
+
elif isinstance(v, list):
|
|
55
|
+
for x in v:
|
|
56
|
+
_walk_strings(x, cnt)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _extract_refs(data):
|
|
60
|
+
if isinstance(data, str):
|
|
61
|
+
return data, []
|
|
62
|
+
cnt: Counter = Counter()
|
|
63
|
+
_walk_strings(data, cnt)
|
|
64
|
+
cands = []
|
|
65
|
+
for s, n in cnt.items():
|
|
66
|
+
if n < 2 or len(s) < MIN_REF_LEN:
|
|
67
|
+
continue
|
|
68
|
+
cost = est_tokens(" " + fmt_scalar(s))
|
|
69
|
+
gain = n * cost - est_tokens(f"&10 {fmt_scalar(s)}\n") - n * est_tokens(" *10")
|
|
70
|
+
if gain > 0:
|
|
71
|
+
cands.append((gain, s))
|
|
72
|
+
if not cands:
|
|
73
|
+
return data, []
|
|
74
|
+
cands.sort(key=lambda t: (-t[0], t[1]))
|
|
75
|
+
table = {s: Ref(i) for i, (_, s) in enumerate(cands)}
|
|
76
|
+
|
|
77
|
+
def sub(v):
|
|
78
|
+
if isinstance(v, str):
|
|
79
|
+
return table.get(v, v)
|
|
80
|
+
if isinstance(v, dict):
|
|
81
|
+
return {k: sub(x) for k, x in v.items()}
|
|
82
|
+
if isinstance(v, list):
|
|
83
|
+
return [sub(x) for x in v]
|
|
84
|
+
return v
|
|
85
|
+
|
|
86
|
+
return sub(data), [s for _, s in cands]
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
# ---------------------------------------------------------------------- body
|
|
90
|
+
|
|
91
|
+
def _str_col(values) -> bool:
|
|
92
|
+
"""Use `:str` when every non-null value is a string and it saves quotes."""
|
|
93
|
+
flat = []
|
|
94
|
+
for v in values:
|
|
95
|
+
if isinstance(v, list):
|
|
96
|
+
flat.extend(v)
|
|
97
|
+
else:
|
|
98
|
+
flat.append(v)
|
|
99
|
+
strs = [v for v in flat if v is not None and not isinstance(v, Ref)]
|
|
100
|
+
return bool(strs) and all(isinstance(v, str) for v in strs) and any(
|
|
101
|
+
NUM_RE.fullmatch(v) for v in strs)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
class _Enc:
|
|
105
|
+
def __init__(self, keep_order: bool = False):
|
|
106
|
+
self.keep_order = keep_order
|
|
107
|
+
|
|
108
|
+
def root(self, v) -> list[str]:
|
|
109
|
+
if isinstance(v, str):
|
|
110
|
+
return [jstr(v)]
|
|
111
|
+
if isinstance(v, dict) and v:
|
|
112
|
+
lines: list[str] = []
|
|
113
|
+
for k, x in v.items():
|
|
114
|
+
self.entry(fmt_key(k), x, 0, lines)
|
|
115
|
+
return lines
|
|
116
|
+
inl = fmt_inline(v)
|
|
117
|
+
if inl is not None:
|
|
118
|
+
return [inl]
|
|
119
|
+
return self.array("", v, 0)
|
|
120
|
+
|
|
121
|
+
def entry(self, key: str, v, d: int, out: list[str]):
|
|
122
|
+
pad = " " * d
|
|
123
|
+
inl = fmt_inline(v)
|
|
124
|
+
if inl is not None:
|
|
125
|
+
out.append(f"{pad}{key} {inl}")
|
|
126
|
+
elif isinstance(v, dict):
|
|
127
|
+
out.append(pad + key)
|
|
128
|
+
for k, x in v.items():
|
|
129
|
+
self.entry(fmt_key(k), x, d + 1, out)
|
|
130
|
+
else:
|
|
131
|
+
out.extend(self.array(key, v, d))
|
|
132
|
+
|
|
133
|
+
# --- arrays -----------------------------------------------------------
|
|
134
|
+
def array(self, key: str, arr: list, d: int) -> list[str]:
|
|
135
|
+
# Candidates in order of preference; ties go to the earlier one, so
|
|
136
|
+
# order-preserving layouts win over the one that moves a column.
|
|
137
|
+
cands = []
|
|
138
|
+
for c in (self.table(key, arr, d), self.outline(key, arr, d, keep_order=True),
|
|
139
|
+
None if self.keep_order else self.outline(key, arr, d)):
|
|
140
|
+
if c and c not in cands:
|
|
141
|
+
cands.append(c)
|
|
142
|
+
if not cands or len(arr) <= 2:
|
|
143
|
+
cands.append(self.items(key, arr, d))
|
|
144
|
+
if len(cands) == 1:
|
|
145
|
+
return cands[0]
|
|
146
|
+
return min(cands, key=lambda ls: est_tokens("\n".join(ls)))
|
|
147
|
+
|
|
148
|
+
def table(self, key, arr, d):
|
|
149
|
+
if not all(isinstance(x, dict) and x for x in arr):
|
|
150
|
+
return None
|
|
151
|
+
cols = list(arr[0])
|
|
152
|
+
for x in arr:
|
|
153
|
+
if list(x) != cols or not all(is_scalar(v) for v in x.values()):
|
|
154
|
+
return None
|
|
155
|
+
smode = [_str_col(x[c] for x in arr) for c in cols]
|
|
156
|
+
head = ",".join(fmt_seg(c) + (":str" if s else "") for c, s in zip(cols, smode))
|
|
157
|
+
pad = " " * d
|
|
158
|
+
lines = [f"{pad}{key}[{len(arr)}]{{{head}}}"]
|
|
159
|
+
for x in arr:
|
|
160
|
+
lines.append(pad + ",".join(fmt_scalar(x[c], "cell", s) for c, s in zip(cols, smode)))
|
|
161
|
+
return lines
|
|
162
|
+
|
|
163
|
+
def outline(self, key, arr, d, keep_order=False):
|
|
164
|
+
spec = _plan(arr, keep_order)
|
|
165
|
+
if spec is None:
|
|
166
|
+
return None
|
|
167
|
+
lines = [f"{' ' * d}{key}[{len(arr)}]{spec.header()}"]
|
|
168
|
+
_render(arr, spec, d, lines)
|
|
169
|
+
return lines
|
|
170
|
+
|
|
171
|
+
def items(self, key, arr, d):
|
|
172
|
+
pad = " " * d
|
|
173
|
+
lines = [f"{pad}{key}[{len(arr)}]"]
|
|
174
|
+
for x in arr:
|
|
175
|
+
inl = fmt_inline(x, "item")
|
|
176
|
+
if inl is not None:
|
|
177
|
+
lines.append(f"{pad}- {inl}")
|
|
178
|
+
continue
|
|
179
|
+
if isinstance(x, dict):
|
|
180
|
+
sub: list[str] = []
|
|
181
|
+
first = True
|
|
182
|
+
for k, v in x.items():
|
|
183
|
+
self.entry(fmt_key(k), v, d + 1, sub)
|
|
184
|
+
if first:
|
|
185
|
+
sub[0] = f"{pad}- " + sub[0][d + 1:]
|
|
186
|
+
first = False
|
|
187
|
+
lines.extend(sub)
|
|
188
|
+
else: # nested non-inline list
|
|
189
|
+
sub = self.array("", x, d + 1)
|
|
190
|
+
sub[0] = f"{pad}- " + sub[0][d + 1:]
|
|
191
|
+
lines.extend(sub)
|
|
192
|
+
return lines
|
|
193
|
+
|
|
194
|
+
|
|
195
|
+
def _tree_nodes(arr, child):
|
|
196
|
+
"""All nodes of a tree whose recursion key is `child`, or None if not a tree."""
|
|
197
|
+
nodes = []
|
|
198
|
+
stack = list(reversed(arr))
|
|
199
|
+
while stack:
|
|
200
|
+
x = stack.pop()
|
|
201
|
+
if not isinstance(x, dict) or not x:
|
|
202
|
+
return None
|
|
203
|
+
nodes.append(x)
|
|
204
|
+
if child in x:
|
|
205
|
+
kids = x[child]
|
|
206
|
+
if not isinstance(kids, list):
|
|
207
|
+
return None
|
|
208
|
+
stack.extend(reversed(kids))
|
|
209
|
+
return nodes
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
# ------------------------------------------------------------- row tables
|
|
213
|
+
# A row table (SPEC §4.3) is described by a _Spec: its columns are paths into
|
|
214
|
+
# the row object (('customer', 'name') is written customer.name), and it may
|
|
215
|
+
# have one child key whose rows are indented under their parent, either with
|
|
216
|
+
# the same spec (a tree, `>steps`) or with their own (`>items{...}`).
|
|
217
|
+
|
|
218
|
+
_ABSENT = object()
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
class _Spec:
|
|
222
|
+
def __init__(self, order, required, keyed, smode, child, sub):
|
|
223
|
+
self.order, self.required, self.keyed = order, required, keyed
|
|
224
|
+
self.smode, self.child, self.sub = smode, child, sub
|
|
225
|
+
|
|
226
|
+
def header(self) -> str:
|
|
227
|
+
cols = " ".join(
|
|
228
|
+
_fmt_path(p) + ("?=" if p in self.keyed else "" if p in self.required else "?")
|
|
229
|
+
+ (":str" if self.smode[p] else "") for p in self.order)
|
|
230
|
+
out = "{" + cols + "}"
|
|
231
|
+
if self.child is not None:
|
|
232
|
+
out += ">" + fmt_seg(self.child) + (self.sub.header() if self.sub else "")
|
|
233
|
+
return out
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def _fmt_path(p) -> str:
|
|
237
|
+
return ".".join(fmt_seg(k) for k in p)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _cellable(v) -> bool:
|
|
241
|
+
return is_scalar(v) or v == {} or (isinstance(v, list) and all(is_scalar(y) for y in v))
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _flat_node(x, skip):
|
|
245
|
+
"""[(path, value)] of a row object with nested objects flattened, or None."""
|
|
246
|
+
out = []
|
|
247
|
+
|
|
248
|
+
def go(path, v):
|
|
249
|
+
if isinstance(v, dict) and v:
|
|
250
|
+
return all(go(path + (k,), y) for k, y in v.items())
|
|
251
|
+
if not _cellable(v):
|
|
252
|
+
return False
|
|
253
|
+
out.append((path, v))
|
|
254
|
+
return True
|
|
255
|
+
|
|
256
|
+
for k, v in x.items():
|
|
257
|
+
if k != skip and not go((k,), v):
|
|
258
|
+
return None
|
|
259
|
+
return out
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def _is_rows(v) -> bool:
|
|
263
|
+
return isinstance(v, list) and all(isinstance(y, dict) and y for y in v)
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _plan(arr, keep_order):
|
|
267
|
+
if not arr or not all(isinstance(x, dict) and x for x in arr):
|
|
268
|
+
return None
|
|
269
|
+
seen = set()
|
|
270
|
+
for ck in list(CHILD_KEYS) + [k for x in arr for k in x]:
|
|
271
|
+
if ck in seen:
|
|
272
|
+
continue
|
|
273
|
+
seen.add(ck)
|
|
274
|
+
if not any(x.get(ck) for x in arr) or not all(ck not in x or _is_rows(x[ck]) for x in arr):
|
|
275
|
+
continue
|
|
276
|
+
nodes = _tree_nodes(arr, ck)
|
|
277
|
+
if nodes is not None:
|
|
278
|
+
spec = _plan_cols(arr, nodes, ck, None, keep_order)
|
|
279
|
+
if spec is not None:
|
|
280
|
+
return spec
|
|
281
|
+
sub = _plan([y for x in arr for y in x.get(ck) or []], keep_order)
|
|
282
|
+
if sub is not None:
|
|
283
|
+
spec = _plan_cols(arr, arr, ck, sub, keep_order)
|
|
284
|
+
if spec is not None:
|
|
285
|
+
return spec
|
|
286
|
+
return _plan_cols(arr, arr, None, None, keep_order)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _plan_cols(arr, nodes, child, sub, keep_order):
|
|
290
|
+
flats = []
|
|
291
|
+
for x in nodes:
|
|
292
|
+
f = _flat_node(x, child)
|
|
293
|
+
if f is None:
|
|
294
|
+
return None
|
|
295
|
+
flats.append(dict(f))
|
|
296
|
+
cols = []
|
|
297
|
+
seen = set()
|
|
298
|
+
for f in flats:
|
|
299
|
+
for p in f:
|
|
300
|
+
if p not in seen:
|
|
301
|
+
seen.add(p)
|
|
302
|
+
cols.append(p)
|
|
303
|
+
if any(p[:i] in seen for p in cols for i in range(1, len(p))):
|
|
304
|
+
return None # a key is an object in one row and a value in another
|
|
305
|
+
required = [p for p in cols if all(p in f for f in flats)]
|
|
306
|
+
if not required or (len(cols) == 1 and child is None):
|
|
307
|
+
return None # one plain column would read back as a comma table
|
|
308
|
+
|
|
309
|
+
def text_score(p):
|
|
310
|
+
# A dictionary reference stands for a (long) string, so it counts as text.
|
|
311
|
+
vals = [f[p] for f in flats]
|
|
312
|
+
if not all(isinstance(v, (str, Ref)) for v in vals):
|
|
313
|
+
return (0, 0)
|
|
314
|
+
return (1, sum(v.count(" ") + 1 if isinstance(v, str) else 1 for v in vals))
|
|
315
|
+
|
|
316
|
+
if keep_order:
|
|
317
|
+
if cols[-1] not in required:
|
|
318
|
+
return None
|
|
319
|
+
rest = cols[-1]
|
|
320
|
+
else:
|
|
321
|
+
rest = max(required, key=text_score)
|
|
322
|
+
order = [p for p in cols if p != rest] + [rest]
|
|
323
|
+
smode = {p: _str_col(f[p] for f in flats if p in f) for p in order}
|
|
324
|
+
optional = [p for p in order if p not in required]
|
|
325
|
+
# An optional column is positional with '-' for "absent" when that is
|
|
326
|
+
# cheaper than writing `name=` on every row that has it.
|
|
327
|
+
keyed = set()
|
|
328
|
+
for p in optional:
|
|
329
|
+
present = sum(p in f for f in flats)
|
|
330
|
+
if (len(flats) - present) * est_tokens(" -") >= present * est_tokens(f" {_fmt_path(p)}="):
|
|
331
|
+
keyed.add(p)
|
|
332
|
+
spec = _Spec(order, set(required), keyed, smode, child, sub)
|
|
333
|
+
if keep_order and not all(_ordered_eq(_rebuild(x, spec), x) for x in arr):
|
|
334
|
+
return None
|
|
335
|
+
return spec
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
def _rebuild(x, spec):
|
|
339
|
+
"""The object the decoder builds from x's row (to check key order)."""
|
|
340
|
+
flat = dict(_flat_node(x, spec.child))
|
|
341
|
+
out: dict = {}
|
|
342
|
+
for p in spec.order:
|
|
343
|
+
if p in flat:
|
|
344
|
+
o = out
|
|
345
|
+
for k in p[:-1]:
|
|
346
|
+
o = o.setdefault(k, {})
|
|
347
|
+
o[p[-1]] = flat[p]
|
|
348
|
+
if spec.child is not None and spec.child in x:
|
|
349
|
+
out[spec.child] = [_rebuild(y, spec.sub or spec) for y in x[spec.child]]
|
|
350
|
+
return out
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
def _ordered_eq(a, b) -> bool:
|
|
354
|
+
if isinstance(a, dict):
|
|
355
|
+
return (isinstance(b, dict) and list(a) == list(b)
|
|
356
|
+
and all(_ordered_eq(a[k], b[k]) for k in a))
|
|
357
|
+
if isinstance(a, list):
|
|
358
|
+
return isinstance(b, list) and len(a) == len(b) and all(map(_ordered_eq, a, b))
|
|
359
|
+
return a is b or a == b
|
|
360
|
+
|
|
361
|
+
|
|
362
|
+
def _cell(v, ctx, strmode):
|
|
363
|
+
if isinstance(v, list):
|
|
364
|
+
return "[" + ",".join(fmt_scalar(y, "list", strmode) for y in v) + "]"
|
|
365
|
+
if v == {}:
|
|
366
|
+
return "{}"
|
|
367
|
+
return fmt_scalar(v, ctx, strmode)
|
|
368
|
+
|
|
369
|
+
|
|
370
|
+
def _render(arr, spec, depth, lines):
|
|
371
|
+
last = spec.order[-1]
|
|
372
|
+
for x in arr:
|
|
373
|
+
flat = dict(_flat_node(x, spec.child))
|
|
374
|
+
parts = []
|
|
375
|
+
for p in spec.order[:-1]:
|
|
376
|
+
if p not in spec.keyed:
|
|
377
|
+
parts.append(_cell(flat[p], "pos", spec.smode[p]) if p in flat else "-")
|
|
378
|
+
for p in spec.order[:-1]:
|
|
379
|
+
if p in spec.keyed and p in flat:
|
|
380
|
+
parts.append(f"{_fmt_path(p)}={_cell(flat[p], 'pos', spec.smode[p])}")
|
|
381
|
+
kids = x.get(spec.child) if spec.child is not None else None
|
|
382
|
+
if spec.child is not None and spec.child in x and not kids:
|
|
383
|
+
parts.append(f"{fmt_seg(spec.child)}=[]")
|
|
384
|
+
parts.append(_cell(flat[last], "last", spec.smode[last]))
|
|
385
|
+
lines.append(" " * depth + " ".join(parts))
|
|
386
|
+
if kids:
|
|
387
|
+
_render(kids, spec.sub or spec, depth + 1, lines)
|
|
388
|
+
|
bpp/estimate.py
ADDED
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
"""Deterministic token estimator.
|
|
2
|
+
|
|
3
|
+
The encoder makes a few cost-based choices (dictionary entries, table style).
|
|
4
|
+
Those choices must not depend on which tokenizer happens to be installed, so we
|
|
5
|
+
use this small BPE-like approximation instead of tiktoken. It tracks o200k and
|
|
6
|
+
the legacy Claude tokenizer closely enough to rank alternatives.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
_PIECE = re.compile(
|
|
14
|
+
r" ?[A-Za-z]+" # ASCII word, leading space merges
|
|
15
|
+
r"| ?[^\W\d_A-Za-z]+" # non-ASCII letters (ç, ş, ı ...)
|
|
16
|
+
r"| ?\d{1,3}" # digits come in groups of up to 3
|
|
17
|
+
r"|\n"
|
|
18
|
+
r"| +"
|
|
19
|
+
r"|[^\w\s]{1,2}" # punctuation, often in pairs
|
|
20
|
+
r"|."
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def est_tokens(text: str) -> int:
|
|
25
|
+
n = 0
|
|
26
|
+
for m in _PIECE.finditer(text):
|
|
27
|
+
p = m.group()
|
|
28
|
+
L = len(p.strip())
|
|
29
|
+
if p[-1:].isalpha() and p.strip().isascii():
|
|
30
|
+
n += 1 + max(0, L - 1) // 7
|
|
31
|
+
elif p.strip() and p.strip()[0].isalpha():
|
|
32
|
+
n += 1 + max(0, L - 1) // 3
|
|
33
|
+
else:
|
|
34
|
+
n += 1
|
|
35
|
+
return n
|
bpp/formats.py
ADDED
|
@@ -0,0 +1,205 @@
|
|
|
1
|
+
"""Readers and writers for the formats bpp converts from and to."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import io
|
|
7
|
+
import json
|
|
8
|
+
import math
|
|
9
|
+
import re
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from .lexer import NUM_RE
|
|
13
|
+
|
|
14
|
+
FORMATS = ("json", "yaml", "csv", "md", "bpp")
|
|
15
|
+
_EXT = {".json": "json", ".yaml": "yaml", ".yml": "yaml", ".csv": "csv",
|
|
16
|
+
".md": "md", ".markdown": "md", ".bpp": "bpp"}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def detect(path: str | Path) -> str:
|
|
20
|
+
fmt = _EXT.get(Path(path).suffix.lower())
|
|
21
|
+
if not fmt:
|
|
22
|
+
raise ValueError(f"cannot infer format from {path!s}; pass --from/--to")
|
|
23
|
+
return fmt
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
# ----------------------------------------------------------------------- YAML
|
|
27
|
+
|
|
28
|
+
def _yaml_loader():
|
|
29
|
+
import yaml
|
|
30
|
+
|
|
31
|
+
class Loader(yaml.SafeLoader):
|
|
32
|
+
def construct_mapping(self, node, deep=False):
|
|
33
|
+
# Stringify keys before they meet in a dict: YAML `1:` and `true:`
|
|
34
|
+
# are different keys but equal (and same-hash) in Python.
|
|
35
|
+
if not isinstance(node, yaml.MappingNode):
|
|
36
|
+
raise yaml.constructor.ConstructorError(
|
|
37
|
+
None, None, "expected a mapping", node.start_mark)
|
|
38
|
+
self.flatten_mapping(node)
|
|
39
|
+
out = {}
|
|
40
|
+
for key_node, value_node in node.value:
|
|
41
|
+
key = _key_str(self.construct_object(key_node, deep=deep))
|
|
42
|
+
out[key] = self.construct_object(value_node, deep=deep)
|
|
43
|
+
return out
|
|
44
|
+
|
|
45
|
+
# Keep dates/times as the strings they were written as: JSON has no date
|
|
46
|
+
# type, and turning them into datetime objects would not round-trip.
|
|
47
|
+
Loader.yaml_implicit_resolvers = {
|
|
48
|
+
ch: [(tag, rx) for tag, rx in rs if tag != "tag:yaml.org,2002:timestamp"]
|
|
49
|
+
for ch, rs in Loader.yaml_implicit_resolvers.items()
|
|
50
|
+
}
|
|
51
|
+
return Loader
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _key_str(k) -> str:
|
|
55
|
+
"""YAML allows non-string keys; the JSON data model does not."""
|
|
56
|
+
if isinstance(k, str):
|
|
57
|
+
return k
|
|
58
|
+
if isinstance(k, (bool, int, float)) or k is None:
|
|
59
|
+
return json.dumps(k)
|
|
60
|
+
if isinstance(k, (list, tuple, dict)):
|
|
61
|
+
raise ValueError("YAML keys must be scalars")
|
|
62
|
+
return str(k)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _json_keys(v):
|
|
66
|
+
if isinstance(v, dict):
|
|
67
|
+
return {_key_str(k): _json_keys(x) for k, x in v.items()}
|
|
68
|
+
if isinstance(v, list):
|
|
69
|
+
return [_json_keys(x) for x in v]
|
|
70
|
+
if isinstance(v, (str, int, float, bool)) or v is None:
|
|
71
|
+
return v
|
|
72
|
+
raise ValueError(f"unsupported YAML value of type {type(v).__name__}")
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def load_yaml(text: str):
|
|
76
|
+
import yaml
|
|
77
|
+
|
|
78
|
+
return _json_keys(yaml.load(text, Loader=_yaml_loader()))
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
def dump_yaml(data) -> str:
|
|
82
|
+
import yaml
|
|
83
|
+
|
|
84
|
+
return yaml.safe_dump(data, allow_unicode=True, sort_keys=False, width=10**9)
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
# ------------------------------------------------------------------------ CSV
|
|
88
|
+
|
|
89
|
+
def _infer(cell: str):
|
|
90
|
+
"""CSV cell -> typed value, only when writing it back yields the same text."""
|
|
91
|
+
if cell == "":
|
|
92
|
+
return None
|
|
93
|
+
if cell in ("true", "false"):
|
|
94
|
+
return cell == "true"
|
|
95
|
+
if NUM_RE.fullmatch(cell):
|
|
96
|
+
if re.fullmatch(r"-?(?:0|[1-9]\d*)", cell):
|
|
97
|
+
return int(cell) if cell != "-0" else cell
|
|
98
|
+
f = float(cell)
|
|
99
|
+
if math.isfinite(f) and repr(f) == cell:
|
|
100
|
+
return f
|
|
101
|
+
return cell
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _cell_text(v) -> str:
|
|
105
|
+
if v is None:
|
|
106
|
+
return ""
|
|
107
|
+
if v is True:
|
|
108
|
+
return "true"
|
|
109
|
+
if v is False:
|
|
110
|
+
return "false"
|
|
111
|
+
if isinstance(v, float):
|
|
112
|
+
return repr(v)
|
|
113
|
+
if isinstance(v, (int, str)):
|
|
114
|
+
return str(v)
|
|
115
|
+
raise ValueError("CSV cells must be scalars")
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _no_nul(text: str):
|
|
119
|
+
# Python < 3.11's csv module cannot read or write NUL; reject it everywhere
|
|
120
|
+
# so behaviour does not depend on the Python version.
|
|
121
|
+
if "\x00" in text:
|
|
122
|
+
raise ValueError("CSV cells cannot contain NUL (\\x00) characters")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def load_csv(text: str) -> list[dict]:
|
|
126
|
+
_no_nul(text)
|
|
127
|
+
rows = list(csv.reader(io.StringIO(text, newline="")))
|
|
128
|
+
if not rows:
|
|
129
|
+
return []
|
|
130
|
+
header = rows[0]
|
|
131
|
+
if len(set(header)) != len(header):
|
|
132
|
+
raise ValueError("CSV header has duplicate column names")
|
|
133
|
+
out = []
|
|
134
|
+
for i, r in enumerate(rows[1:], 2):
|
|
135
|
+
if len(r) != len(header):
|
|
136
|
+
raise ValueError(f"CSV row {i} has {len(r)} cells, header has {len(header)}")
|
|
137
|
+
out.append({k: _infer(c) for k, c in zip(header, r)})
|
|
138
|
+
return out
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def dump_csv(data) -> str:
|
|
142
|
+
if not isinstance(data, list) or not all(isinstance(r, dict) for r in data):
|
|
143
|
+
raise ValueError("CSV output needs a list of flat objects")
|
|
144
|
+
cols: list[str] = []
|
|
145
|
+
for r in data:
|
|
146
|
+
for k in r:
|
|
147
|
+
if k not in cols:
|
|
148
|
+
cols.append(k)
|
|
149
|
+
_no_nul("".join(cols))
|
|
150
|
+
for r in data:
|
|
151
|
+
for v in r.values():
|
|
152
|
+
if isinstance(v, str):
|
|
153
|
+
_no_nul(v)
|
|
154
|
+
buf = io.StringIO()
|
|
155
|
+
w = csv.writer(buf, lineterminator="\n")
|
|
156
|
+
w.writerow(cols)
|
|
157
|
+
for r in data:
|
|
158
|
+
w.writerow([_cell_text(r.get(c)) for c in cols])
|
|
159
|
+
return buf.getvalue()
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
# ----------------------------------------------------------------------- JSON
|
|
163
|
+
|
|
164
|
+
def load_json(text: str):
|
|
165
|
+
return json.loads(text)
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def dump_json(data, indent: int | None = 2) -> str:
|
|
169
|
+
if indent is None:
|
|
170
|
+
return json.dumps(data, ensure_ascii=False, separators=(",", ":"))
|
|
171
|
+
return json.dumps(data, ensure_ascii=False, indent=indent) + "\n"
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
# --------------------------------------------------------------------- facade
|
|
175
|
+
|
|
176
|
+
def loads(text: str, fmt: str):
|
|
177
|
+
if fmt == "json":
|
|
178
|
+
return load_json(text)
|
|
179
|
+
if fmt == "yaml":
|
|
180
|
+
return load_yaml(text)
|
|
181
|
+
if fmt == "csv":
|
|
182
|
+
return load_csv(text)
|
|
183
|
+
if fmt == "md":
|
|
184
|
+
from .markdown import md_to_tree
|
|
185
|
+
return md_to_tree(text)
|
|
186
|
+
if fmt == "bpp":
|
|
187
|
+
from .decoder import decode
|
|
188
|
+
return decode(text)
|
|
189
|
+
raise ValueError(f"unknown format {fmt!r}")
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def dumps(data, fmt: str, **kw) -> str:
|
|
193
|
+
if fmt == "json":
|
|
194
|
+
return dump_json(data, kw.get("indent", 2))
|
|
195
|
+
if fmt == "yaml":
|
|
196
|
+
return dump_yaml(data)
|
|
197
|
+
if fmt == "csv":
|
|
198
|
+
return dump_csv(data)
|
|
199
|
+
if fmt == "md":
|
|
200
|
+
from .markdown import tree_to_md
|
|
201
|
+
return tree_to_md(data)
|
|
202
|
+
if fmt == "bpp":
|
|
203
|
+
from .encoder import encode
|
|
204
|
+
return encode(data, **{k: v for k, v in kw.items() if k in ("primer", "refs", "keep_order")})
|
|
205
|
+
raise ValueError(f"unknown format {fmt!r}")
|