fair-bioheaders 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bioheaders/__init__.py +678 -0
- bioheaders/__main__.py +6 -0
- bioheaders/cli.py +828 -0
- bioheaders/fhr_schema.json +266 -0
- fair_bioheaders-0.4.0.dist-info/METADATA +261 -0
- fair_bioheaders-0.4.0.dist-info/RECORD +10 -0
- fair_bioheaders-0.4.0.dist-info/WHEEL +4 -0
- fair_bioheaders-0.4.0.dist-info/entry_points.txt +11 -0
- fair_bioheaders-0.4.0.dist-info/licenses/LICENSE +73 -0
- fhr.py +30 -0
bioheaders/__init__.py
ADDED
|
@@ -0,0 +1,678 @@
|
|
|
1
|
+
"""FAIR-bioHeaders tools: FHR metadata parsing, serialization, and validation."""
|
|
2
|
+
|
|
3
|
+
import codecs
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Hashable
|
|
7
|
+
from copy import deepcopy
|
|
8
|
+
from datetime import date, datetime
|
|
9
|
+
from html import escape
|
|
10
|
+
from html.parser import HTMLParser
|
|
11
|
+
from importlib.resources import files
|
|
12
|
+
from itertools import chain
|
|
13
|
+
|
|
14
|
+
import yaml
|
|
15
|
+
from jsonschema import Draft202012Validator, FormatChecker
|
|
16
|
+
|
|
17
|
+
__version__ = "0.4.0"
|
|
18
|
+
SCHEMA = json.loads(
|
|
19
|
+
files(__package__).joinpath("fhr_schema.json").read_text(encoding="utf-8")
|
|
20
|
+
)
|
|
21
|
+
ITEM_TYPE = SCHEMA["$id"]
|
|
22
|
+
# YAML treats these as line breaks, but FASTA/GFA line splitting does not.
|
|
23
|
+
YAML_ONLY_LINE_BREAKS = "\x85\u2028\u2029"
|
|
24
|
+
# FASTA/GFA files are read in chunks; only FHR header lines are held whole.
|
|
25
|
+
CHUNK_SIZE = 2**20
|
|
26
|
+
MAX_HEADER_BYTES = 16 * 2**20
|
|
27
|
+
HEADER, DATA, END = "header", "data", "end"
|
|
28
|
+
_MAYBE_BLANK = "maybe blank"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _read(stream):
|
|
32
|
+
return stream.read() if hasattr(stream, "read") else stream
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _text(stream):
|
|
36
|
+
value = _read(stream)
|
|
37
|
+
if isinstance(value, bytes):
|
|
38
|
+
value = value.decode("utf-8")
|
|
39
|
+
if not isinstance(value, str):
|
|
40
|
+
raise TypeError("Expected text or a readable stream")
|
|
41
|
+
# A UTF-8 byte order mark is an encoding signature, not metadata.
|
|
42
|
+
return value[1:] if value.startswith("\ufeff") else value
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def split_lines(content):
|
|
46
|
+
"""Split FASTA/GFA bytes into lines, keeping each original terminator."""
|
|
47
|
+
if content.startswith(codecs.BOM_UTF8):
|
|
48
|
+
raise ValueError("FASTA/GFA files must not begin with a UTF-8 byte order mark")
|
|
49
|
+
return content.splitlines(keepends=True)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _bytes(value):
|
|
53
|
+
if isinstance(value, str):
|
|
54
|
+
value = value.encode("utf-8")
|
|
55
|
+
if not isinstance(value, bytes):
|
|
56
|
+
raise TypeError("Expected bytes, text, or a readable stream")
|
|
57
|
+
return value
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def read_chunks(stream, size=CHUNK_SIZE):
|
|
61
|
+
"""Yield bytes chunks from bytes, text, or a readable stream."""
|
|
62
|
+
if hasattr(stream, "read"):
|
|
63
|
+
while True:
|
|
64
|
+
chunk = stream.read(size)
|
|
65
|
+
if not chunk:
|
|
66
|
+
return
|
|
67
|
+
yield _bytes(chunk)
|
|
68
|
+
content = _bytes(stream)
|
|
69
|
+
for start in range(0, len(content), size):
|
|
70
|
+
yield content[start : start + size]
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def sequence_parts(chunks, prefix):
|
|
74
|
+
"""Yield ``(kind, bytes)`` parts of FASTA/GFA bytes given as chunks, in order.
|
|
75
|
+
|
|
76
|
+
Lines are split as ``bytes.splitlines(keepends=True)`` splits the whole input.
|
|
77
|
+
Each FHR line of the leading header block is one ``HEADER`` part; all other
|
|
78
|
+
bytes are ``DATA`` parts of any size, and ``END`` (empty) marks the first
|
|
79
|
+
record. The block ends at a FASTA ``>`` line, or a nonblank GFA line that is
|
|
80
|
+
not a ``#`` comment; ordinary comments and blank lines may be mixed in. Later
|
|
81
|
+
FHR lines are rejected. Only header lines are held whole, up to
|
|
82
|
+
``MAX_HEADER_BYTES`` in total.
|
|
83
|
+
"""
|
|
84
|
+
fasta = prefix == b";~"
|
|
85
|
+
pending = b"" # Unprocessed header block bytes: at most two.
|
|
86
|
+
started = False
|
|
87
|
+
in_header = True
|
|
88
|
+
line = None # Kind of the current header block line; None at a line start.
|
|
89
|
+
parts = []
|
|
90
|
+
header_size = 0
|
|
91
|
+
lines = 0 # Line terminators before the unprocessed bytes.
|
|
92
|
+
tail = b"" # Last bytes after the header block, to find split prefixes.
|
|
93
|
+
after_cr = False
|
|
94
|
+
for chunk in chain(chunks, [None]):
|
|
95
|
+
final = chunk is None
|
|
96
|
+
if in_header:
|
|
97
|
+
if final:
|
|
98
|
+
data = pending
|
|
99
|
+
elif pending:
|
|
100
|
+
data = pending + chunk
|
|
101
|
+
else:
|
|
102
|
+
data = chunk
|
|
103
|
+
if not started:
|
|
104
|
+
if len(data) < len(codecs.BOM_UTF8) and not final:
|
|
105
|
+
pending = data
|
|
106
|
+
continue
|
|
107
|
+
if data.startswith(codecs.BOM_UTF8):
|
|
108
|
+
raise ValueError(
|
|
109
|
+
"FASTA/GFA files must not begin with a UTF-8 byte order mark"
|
|
110
|
+
)
|
|
111
|
+
started = True
|
|
112
|
+
position, size = 0, len(data)
|
|
113
|
+
# Next \n and \r at or after position, or size if there is none.
|
|
114
|
+
newline = carriage_return = -1
|
|
115
|
+
while position < size:
|
|
116
|
+
if line is None:
|
|
117
|
+
if size - position < len(prefix) and not final:
|
|
118
|
+
break
|
|
119
|
+
if data.startswith(prefix, position):
|
|
120
|
+
line = HEADER
|
|
121
|
+
elif fasta and data.startswith(b">", position):
|
|
122
|
+
in_header = False
|
|
123
|
+
break
|
|
124
|
+
elif fasta or data.startswith(b"#", position):
|
|
125
|
+
line = DATA
|
|
126
|
+
else:
|
|
127
|
+
line = _MAYBE_BLANK
|
|
128
|
+
if newline < position:
|
|
129
|
+
newline = data.find(b"\n", position)
|
|
130
|
+
newline = size if newline < 0 else newline
|
|
131
|
+
if carriage_return < position:
|
|
132
|
+
carriage_return = data.find(b"\r", position)
|
|
133
|
+
carriage_return = size if carriage_return < 0 else carriage_return
|
|
134
|
+
end = min(newline, carriage_return) + 1
|
|
135
|
+
complete = end <= size
|
|
136
|
+
if not complete:
|
|
137
|
+
end = size
|
|
138
|
+
elif end - 1 == carriage_return: # Perhaps \r\n.
|
|
139
|
+
if end < size:
|
|
140
|
+
end += data[end] == 10
|
|
141
|
+
elif not final:
|
|
142
|
+
end -= 1
|
|
143
|
+
complete = False
|
|
144
|
+
piece = data[position:end]
|
|
145
|
+
if line is _MAYBE_BLANK and piece.strip():
|
|
146
|
+
in_header = False # A GFA record line.
|
|
147
|
+
break
|
|
148
|
+
if line is HEADER:
|
|
149
|
+
header_size += len(piece)
|
|
150
|
+
if header_size > MAX_HEADER_BYTES:
|
|
151
|
+
raise ValueError(
|
|
152
|
+
"FHR header lines exceed the "
|
|
153
|
+
f"{MAX_HEADER_BYTES // 2**20} MiB size limit"
|
|
154
|
+
)
|
|
155
|
+
parts.append(piece)
|
|
156
|
+
elif piece:
|
|
157
|
+
yield DATA, piece
|
|
158
|
+
position = end
|
|
159
|
+
if not complete:
|
|
160
|
+
break
|
|
161
|
+
lines += 1
|
|
162
|
+
if line is HEADER:
|
|
163
|
+
yield HEADER, b"".join(parts)
|
|
164
|
+
parts = []
|
|
165
|
+
line = None
|
|
166
|
+
if in_header:
|
|
167
|
+
pending = data[position:]
|
|
168
|
+
if final and line is HEADER:
|
|
169
|
+
yield HEADER, b"".join(parts)
|
|
170
|
+
continue
|
|
171
|
+
yield END, b""
|
|
172
|
+
chunk = data[position:]
|
|
173
|
+
pending = b""
|
|
174
|
+
elif final:
|
|
175
|
+
break
|
|
176
|
+
if not chunk:
|
|
177
|
+
continue
|
|
178
|
+
# After the header block, a line starts after each \n or \r.
|
|
179
|
+
late = [
|
|
180
|
+
match - len(tail) + 1
|
|
181
|
+
for match in (
|
|
182
|
+
(tail + chunk[: len(prefix)]).find(b"\n" + prefix),
|
|
183
|
+
(tail + chunk[: len(prefix)]).find(b"\r" + prefix),
|
|
184
|
+
)
|
|
185
|
+
if 0 <= match < len(tail)
|
|
186
|
+
]
|
|
187
|
+
if prefix[-1:] in chunk: # A fast check that is usually false.
|
|
188
|
+
late += [
|
|
189
|
+
match + 1
|
|
190
|
+
for match in (chunk.find(b"\n" + prefix), chunk.find(b"\r" + prefix))
|
|
191
|
+
if match >= 0
|
|
192
|
+
]
|
|
193
|
+
if late:
|
|
194
|
+
start = min(late)
|
|
195
|
+
if start > 0:
|
|
196
|
+
lines += _count_lines(chunk[:start], after_cr)
|
|
197
|
+
raise ValueError(f"FHR header line after sequence data at line {lines + 1}")
|
|
198
|
+
lines += _count_lines(chunk, after_cr)
|
|
199
|
+
after_cr = chunk.endswith(b"\r")
|
|
200
|
+
tail = (tail + chunk[-len(prefix) :])[-len(prefix) :]
|
|
201
|
+
yield DATA, chunk
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _count_lines(data, after_cr):
|
|
205
|
+
"""Count line terminators, given whether the previous byte was \\r."""
|
|
206
|
+
count = data.count(b"\n")
|
|
207
|
+
if b"\r" in data:
|
|
208
|
+
count += data.count(b"\r") - data.count(b"\r\n")
|
|
209
|
+
if after_cr and data.startswith(b"\n"):
|
|
210
|
+
count -= 1 # The second byte of a \r\n split between chunks.
|
|
211
|
+
return count
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def header_lines(lines, prefix):
|
|
215
|
+
"""Return the FHR lines of the leading header block, rejecting any later ones.
|
|
216
|
+
|
|
217
|
+
The block ends at the first record: a FASTA ``>`` line, or a nonblank GFA line
|
|
218
|
+
that is not a ``#`` comment. Ordinary comments and blank lines may be mixed in.
|
|
219
|
+
"""
|
|
220
|
+
return [part for kind, part in sequence_parts(lines, prefix) if kind is HEADER]
|
|
221
|
+
|
|
222
|
+
|
|
223
|
+
def header_text(line, prefix):
|
|
224
|
+
"""Return the YAML text of one FHR header line given as bytes."""
|
|
225
|
+
text = line[len(prefix) :].rstrip(b"\r\n").decode("utf-8")
|
|
226
|
+
if any(char in text for char in YAML_ONLY_LINE_BREAKS):
|
|
227
|
+
raise ValueError("FHR header lines must not contain U+0085, U+2028, or U+2029")
|
|
228
|
+
return text
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
class _Loader(yaml.SafeLoader):
|
|
232
|
+
"""Load JSON-compatible YAML without aliases, merge keys, or duplicate keys."""
|
|
233
|
+
|
|
234
|
+
def compose_node(self, parent, index):
|
|
235
|
+
event = self.peek_event()
|
|
236
|
+
if isinstance(event, yaml.AliasEvent) or getattr(event, "anchor", None):
|
|
237
|
+
raise yaml.composer.ComposerError(
|
|
238
|
+
None,
|
|
239
|
+
None,
|
|
240
|
+
"YAML anchors and aliases are not allowed in FHR metadata",
|
|
241
|
+
event.start_mark,
|
|
242
|
+
)
|
|
243
|
+
return super().compose_node(parent, index)
|
|
244
|
+
|
|
245
|
+
def construct_mapping(self, node, deep=False):
|
|
246
|
+
if isinstance(node, yaml.MappingNode):
|
|
247
|
+
keys = set()
|
|
248
|
+
for key_node, _ in node.value:
|
|
249
|
+
if key_node.tag == "tag:yaml.org,2002:merge":
|
|
250
|
+
raise yaml.constructor.ConstructorError(
|
|
251
|
+
None,
|
|
252
|
+
None,
|
|
253
|
+
"YAML merge keys are not allowed in FHR metadata",
|
|
254
|
+
key_node.start_mark,
|
|
255
|
+
)
|
|
256
|
+
key = self.construct_object(key_node, deep=True)
|
|
257
|
+
if not isinstance(key, Hashable):
|
|
258
|
+
continue # SafeConstructor reports unhashable keys.
|
|
259
|
+
if key in keys:
|
|
260
|
+
raise yaml.constructor.ConstructorError(
|
|
261
|
+
"while constructing a mapping",
|
|
262
|
+
node.start_mark,
|
|
263
|
+
f"found duplicate key {key!r}",
|
|
264
|
+
key_node.start_mark,
|
|
265
|
+
)
|
|
266
|
+
keys.add(key)
|
|
267
|
+
return super().construct_mapping(node, deep=deep)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def load_yaml(text):
|
|
271
|
+
return yaml.load(text, Loader=_Loader)
|
|
272
|
+
|
|
273
|
+
|
|
274
|
+
class _Dumper(yaml.SafeDumper):
|
|
275
|
+
"""Escape characters that YAML, but not FASTA/GFA, reads as line breaks."""
|
|
276
|
+
|
|
277
|
+
def represent_str(self, data):
|
|
278
|
+
if any(char in data for char in YAML_ONLY_LINE_BREAKS):
|
|
279
|
+
return self.represent_scalar("tag:yaml.org,2002:str", data, style='"')
|
|
280
|
+
return super().represent_str(data)
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
_Dumper.add_representer(str, _Dumper.represent_str)
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
def _unique_object(pairs):
|
|
287
|
+
result = {}
|
|
288
|
+
for key, value in pairs:
|
|
289
|
+
if key in result:
|
|
290
|
+
raise ValueError(f"Duplicate JSON object key: {key!r}")
|
|
291
|
+
result[key] = value
|
|
292
|
+
return result
|
|
293
|
+
|
|
294
|
+
|
|
295
|
+
def _json_value(value):
|
|
296
|
+
if isinstance(value, (date, datetime)):
|
|
297
|
+
return value.isoformat()
|
|
298
|
+
if isinstance(value, dict):
|
|
299
|
+
if not all(isinstance(key, str) for key in value):
|
|
300
|
+
raise ValueError("Metadata object keys must be strings")
|
|
301
|
+
return {key: _json_value(item) for key, item in value.items()}
|
|
302
|
+
if isinstance(value, list):
|
|
303
|
+
return [_json_value(item) for item in value]
|
|
304
|
+
return value
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
_VOID_ELEMENTS = {
|
|
308
|
+
"area",
|
|
309
|
+
"base",
|
|
310
|
+
"br",
|
|
311
|
+
"col",
|
|
312
|
+
"embed",
|
|
313
|
+
"hr",
|
|
314
|
+
"img",
|
|
315
|
+
"input",
|
|
316
|
+
"link",
|
|
317
|
+
"meta",
|
|
318
|
+
"param",
|
|
319
|
+
"source",
|
|
320
|
+
"track",
|
|
321
|
+
"wbr",
|
|
322
|
+
}
|
|
323
|
+
# Elements that stop the HTML search for an element to close implicitly.
|
|
324
|
+
_HTML_SCOPE = {
|
|
325
|
+
"applet",
|
|
326
|
+
"button",
|
|
327
|
+
"caption",
|
|
328
|
+
"html",
|
|
329
|
+
"marquee",
|
|
330
|
+
"object",
|
|
331
|
+
"table",
|
|
332
|
+
"td",
|
|
333
|
+
"template",
|
|
334
|
+
"th",
|
|
335
|
+
}
|
|
336
|
+
_CLOSES_P = {
|
|
337
|
+
"address",
|
|
338
|
+
"article",
|
|
339
|
+
"aside",
|
|
340
|
+
"blockquote",
|
|
341
|
+
"center",
|
|
342
|
+
"dd",
|
|
343
|
+
"details",
|
|
344
|
+
"dialog",
|
|
345
|
+
"dir",
|
|
346
|
+
"div",
|
|
347
|
+
"dl",
|
|
348
|
+
"dt",
|
|
349
|
+
"fieldset",
|
|
350
|
+
"figcaption",
|
|
351
|
+
"figure",
|
|
352
|
+
"footer",
|
|
353
|
+
"form",
|
|
354
|
+
"h1",
|
|
355
|
+
"h2",
|
|
356
|
+
"h3",
|
|
357
|
+
"h4",
|
|
358
|
+
"h5",
|
|
359
|
+
"h6",
|
|
360
|
+
"header",
|
|
361
|
+
"hgroup",
|
|
362
|
+
"hr",
|
|
363
|
+
"li",
|
|
364
|
+
"listing",
|
|
365
|
+
"main",
|
|
366
|
+
"menu",
|
|
367
|
+
"nav",
|
|
368
|
+
"ol",
|
|
369
|
+
"p",
|
|
370
|
+
"pre",
|
|
371
|
+
"search",
|
|
372
|
+
"section",
|
|
373
|
+
"summary",
|
|
374
|
+
"table",
|
|
375
|
+
"ul",
|
|
376
|
+
"xmp",
|
|
377
|
+
}
|
|
378
|
+
_TABLE_SECTIONS = {"tbody", "tfoot", "thead"}
|
|
379
|
+
# Start tag -> (open elements it implicitly closes, elements bounding the search).
|
|
380
|
+
_IMPLIED_END_TAGS = {
|
|
381
|
+
"li": ({"li"}, {"ol", "ul"}),
|
|
382
|
+
"dd": ({"dd", "dt"}, {"dl"}),
|
|
383
|
+
"dt": ({"dd", "dt"}, {"dl"}),
|
|
384
|
+
"option": ({"option"}, {"datalist", "optgroup", "select"}),
|
|
385
|
+
"optgroup": ({"optgroup", "option"}, {"datalist", "select"}),
|
|
386
|
+
"tr": ({"td", "th", "tr"}, _TABLE_SECTIONS),
|
|
387
|
+
"td": ({"td", "th"}, {"tr"}),
|
|
388
|
+
"th": ({"td", "th"}, {"tr"}),
|
|
389
|
+
**{tag: (_TABLE_SECTIONS | {"td", "th", "tr"}, set()) for tag in _TABLE_SECTIONS},
|
|
390
|
+
}
|
|
391
|
+
_MICRODATA_ATTRIBUTES = {
|
|
392
|
+
"meta": "content",
|
|
393
|
+
"audio": "src",
|
|
394
|
+
"embed": "src",
|
|
395
|
+
"iframe": "src",
|
|
396
|
+
"img": "src",
|
|
397
|
+
"source": "src",
|
|
398
|
+
"track": "src",
|
|
399
|
+
"video": "src",
|
|
400
|
+
"a": "href",
|
|
401
|
+
"area": "href",
|
|
402
|
+
"link": "href",
|
|
403
|
+
"object": "data",
|
|
404
|
+
"data": "value",
|
|
405
|
+
"meter": "value",
|
|
406
|
+
"time": "datetime",
|
|
407
|
+
}
|
|
408
|
+
_JSON_TYPES = {
|
|
409
|
+
"integer": lambda value: isinstance(value, int) and not isinstance(value, bool),
|
|
410
|
+
"number": lambda value: isinstance(value, (int, float))
|
|
411
|
+
and not isinstance(value, bool),
|
|
412
|
+
"boolean": lambda value: isinstance(value, bool),
|
|
413
|
+
"null": lambda value: value is None,
|
|
414
|
+
}
|
|
415
|
+
|
|
416
|
+
|
|
417
|
+
def _tokens(value):
|
|
418
|
+
"""Split an HTML attribute on ASCII whitespace."""
|
|
419
|
+
return re.findall(r"[^\t\n\f\r ]+", value or "")
|
|
420
|
+
|
|
421
|
+
|
|
422
|
+
def _attributes(pairs):
|
|
423
|
+
"""Keep the first of duplicate attributes, as HTML parsers do."""
|
|
424
|
+
result = {}
|
|
425
|
+
for name, value in pairs:
|
|
426
|
+
result.setdefault(name, value)
|
|
427
|
+
return result
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
class _MetadataHTML(HTMLParser):
|
|
431
|
+
"""Parse the FHR item scope, using explicit JSON types for lossless values."""
|
|
432
|
+
|
|
433
|
+
def __init__(self):
|
|
434
|
+
super().__init__(convert_charrefs=True)
|
|
435
|
+
# Every open element; only "active" ones are inside the FHR item scope.
|
|
436
|
+
self.stack = []
|
|
437
|
+
self.result = None
|
|
438
|
+
|
|
439
|
+
def handle_starttag(self, tag, attrs):
|
|
440
|
+
self._imply_end_tags(tag)
|
|
441
|
+
attrs = _attributes(attrs)
|
|
442
|
+
if tag in _VOID_ELEMENTS:
|
|
443
|
+
self._leaf(tag, attrs)
|
|
444
|
+
return
|
|
445
|
+
root = "itemscope" in attrs and ITEM_TYPE in _tokens(attrs.get("itemtype"))
|
|
446
|
+
active = bool(self.stack) and self.stack[-1]["active"]
|
|
447
|
+
if root and (self.result is not None or active):
|
|
448
|
+
raise ValueError("Multiple FHR item scopes found")
|
|
449
|
+
self.stack.append(
|
|
450
|
+
{
|
|
451
|
+
"tag": tag,
|
|
452
|
+
"attrs": attrs,
|
|
453
|
+
"active": root or active,
|
|
454
|
+
"text": [],
|
|
455
|
+
"values": {},
|
|
456
|
+
}
|
|
457
|
+
)
|
|
458
|
+
|
|
459
|
+
def handle_startendtag(self, tag, attrs):
|
|
460
|
+
self._imply_end_tags(tag)
|
|
461
|
+
self._leaf(tag, _attributes(attrs))
|
|
462
|
+
|
|
463
|
+
def handle_endtag(self, tag):
|
|
464
|
+
for index in range(len(self.stack) - 1, -1, -1):
|
|
465
|
+
if self.stack[index]["tag"] == tag:
|
|
466
|
+
self._close(index)
|
|
467
|
+
return
|
|
468
|
+
|
|
469
|
+
def handle_data(self, text):
|
|
470
|
+
if self.stack and self.stack[-1]["active"]:
|
|
471
|
+
self.stack[-1]["text"].append(text)
|
|
472
|
+
|
|
473
|
+
def _imply_end_tags(self, tag):
|
|
474
|
+
rules = []
|
|
475
|
+
if tag in _IMPLIED_END_TAGS:
|
|
476
|
+
rules.append(_IMPLIED_END_TAGS[tag])
|
|
477
|
+
if tag in _CLOSES_P:
|
|
478
|
+
rules.append(({"p"}, set()))
|
|
479
|
+
for closes, bounds in rules:
|
|
480
|
+
for index in range(len(self.stack) - 1, -1, -1):
|
|
481
|
+
name = self.stack[index]["tag"]
|
|
482
|
+
if name in closes:
|
|
483
|
+
self._close(index)
|
|
484
|
+
break
|
|
485
|
+
if name in bounds or name in _HTML_SCOPE:
|
|
486
|
+
break
|
|
487
|
+
|
|
488
|
+
def _close(self, index):
|
|
489
|
+
while len(self.stack) > index:
|
|
490
|
+
self._finish(self.stack.pop())
|
|
491
|
+
|
|
492
|
+
def _leaf(self, tag, attrs):
|
|
493
|
+
if self.stack and self.stack[-1]["active"]:
|
|
494
|
+
value = attrs.get(_MICRODATA_ATTRIBUTES.get(tag, ""))
|
|
495
|
+
for name in _tokens(attrs.get("itemprop")):
|
|
496
|
+
self._attach(self.stack[-1], name, value or "")
|
|
497
|
+
|
|
498
|
+
@staticmethod
|
|
499
|
+
def _attach(node, key, value):
|
|
500
|
+
node["values"].setdefault(key, []).append(value)
|
|
501
|
+
|
|
502
|
+
@staticmethod
|
|
503
|
+
def _object(values):
|
|
504
|
+
return {
|
|
505
|
+
key: items[0] if len(items) == 1 else items for key, items in values.items()
|
|
506
|
+
}
|
|
507
|
+
|
|
508
|
+
def _finish(self, node):
|
|
509
|
+
if not node["active"]:
|
|
510
|
+
return
|
|
511
|
+
tag, attrs = node["tag"], node["attrs"]
|
|
512
|
+
kind = attrs.get("data-fhr-type")
|
|
513
|
+
text = attrs.get(_MICRODATA_ATTRIBUTES.get(tag, ""))
|
|
514
|
+
if text is None:
|
|
515
|
+
text = "".join(node["text"])
|
|
516
|
+
if kind != "string":
|
|
517
|
+
text = text.strip()
|
|
518
|
+
if kind == "array":
|
|
519
|
+
value = node["values"].get("item", [])
|
|
520
|
+
elif kind == "object" or "itemscope" in attrs or node["values"]:
|
|
521
|
+
value = self._object(node["values"])
|
|
522
|
+
elif kind in _JSON_TYPES:
|
|
523
|
+
value = json.loads(text)
|
|
524
|
+
if not _JSON_TYPES[kind](value):
|
|
525
|
+
raise ValueError(f"Microdata value {text!r} is not a JSON {kind}")
|
|
526
|
+
else:
|
|
527
|
+
value = text
|
|
528
|
+
parent = self.stack[-1] if self.stack and self.stack[-1]["active"] else None
|
|
529
|
+
if parent is None:
|
|
530
|
+
self.result = value
|
|
531
|
+
elif _tokens(attrs.get("itemprop")):
|
|
532
|
+
for name in _tokens(attrs.get("itemprop")):
|
|
533
|
+
self._attach(parent, name, value)
|
|
534
|
+
elif "itemscope" not in attrs:
|
|
535
|
+
parent["text"].extend(node["text"])
|
|
536
|
+
for key, items in node["values"].items():
|
|
537
|
+
parent["values"].setdefault(key, []).extend(items)
|
|
538
|
+
|
|
539
|
+
def close(self):
|
|
540
|
+
super().close()
|
|
541
|
+
if any(node["active"] for node in self.stack):
|
|
542
|
+
raise ValueError("No complete FHR microdata item scope found")
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _html_value(key, value):
|
|
546
|
+
prop = escape(key, quote=True)
|
|
547
|
+
if isinstance(value, dict):
|
|
548
|
+
content = "".join(_html_value(k, v) for k, v in value.items())
|
|
549
|
+
return (
|
|
550
|
+
f'<span itemprop="{prop}" itemscope data-fhr-type="object">{content}</span>'
|
|
551
|
+
)
|
|
552
|
+
if isinstance(value, list):
|
|
553
|
+
content = "".join(_html_value("item", v) for v in value)
|
|
554
|
+
return f'<span itemprop="{prop}" data-fhr-type="array">{content}</span>'
|
|
555
|
+
kind = "string"
|
|
556
|
+
if value is None:
|
|
557
|
+
kind = "null"
|
|
558
|
+
elif isinstance(value, bool):
|
|
559
|
+
kind = "boolean"
|
|
560
|
+
elif isinstance(value, int):
|
|
561
|
+
kind = "integer"
|
|
562
|
+
elif isinstance(value, float):
|
|
563
|
+
kind = "number"
|
|
564
|
+
content = value if isinstance(value, str) else json.dumps(value)
|
|
565
|
+
return f'<span itemprop="{prop}" data-fhr-type="{kind}">{escape(content)}</span>'
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
class fhr:
|
|
569
|
+
"""A metadata mapping. Optional fields are absent until explicitly supplied."""
|
|
570
|
+
|
|
571
|
+
def __init__(self, **metadata):
|
|
572
|
+
self.__dict__.update(deepcopy(metadata))
|
|
573
|
+
|
|
574
|
+
def _input(self, data):
|
|
575
|
+
if not isinstance(data, dict) or not all(isinstance(key, str) for key in data):
|
|
576
|
+
raise ValueError("FHR metadata must be an object with string keys")
|
|
577
|
+
normalized = _json_value(data)
|
|
578
|
+
json.dumps(normalized, allow_nan=False)
|
|
579
|
+
self.__dict__.clear()
|
|
580
|
+
self.__dict__.update(deepcopy(normalized))
|
|
581
|
+
|
|
582
|
+
def input_yaml(self, stream):
|
|
583
|
+
self._input(load_yaml(_text(stream)))
|
|
584
|
+
|
|
585
|
+
def output_yaml(self):
|
|
586
|
+
return yaml.dump(
|
|
587
|
+
self.__dict__, Dumper=_Dumper, sort_keys=False, allow_unicode=True
|
|
588
|
+
)
|
|
589
|
+
|
|
590
|
+
def input_json(self, stream):
|
|
591
|
+
self._input(json.loads(_text(stream), object_pairs_hook=_unique_object))
|
|
592
|
+
|
|
593
|
+
def output_json(self):
|
|
594
|
+
return json.dumps(self.__dict__, ensure_ascii=False, indent=2) + "\n"
|
|
595
|
+
|
|
596
|
+
def _input_header(self, stream, prefix):
|
|
597
|
+
# Match checksum line handling: split bytes, decode only header lines.
|
|
598
|
+
marker = prefix.encode("ascii")
|
|
599
|
+
self._input_header_lines(
|
|
600
|
+
[
|
|
601
|
+
line
|
|
602
|
+
for kind, line in sequence_parts(read_chunks(stream), marker)
|
|
603
|
+
if kind is HEADER
|
|
604
|
+
],
|
|
605
|
+
marker,
|
|
606
|
+
)
|
|
607
|
+
|
|
608
|
+
def _input_header_lines(self, lines, prefix):
|
|
609
|
+
"""Load metadata from FHR header lines given as bytes with ``prefix``."""
|
|
610
|
+
if not lines:
|
|
611
|
+
raise ValueError(f"No {prefix.decode('ascii')} FHR metadata header found")
|
|
612
|
+
self._input(load_yaml("\n".join(header_text(line, prefix) for line in lines)))
|
|
613
|
+
|
|
614
|
+
def _output_header(self, prefix):
|
|
615
|
+
lines = self.output_yaml().split("\n")
|
|
616
|
+
if lines[-1] == "":
|
|
617
|
+
lines.pop()
|
|
618
|
+
return "".join(prefix + line + "\n" for line in lines)
|
|
619
|
+
|
|
620
|
+
def input_fasta(self, stream):
|
|
621
|
+
self._input_header(stream, ";~")
|
|
622
|
+
|
|
623
|
+
def output_fasta(self):
|
|
624
|
+
return self._output_header(";~")
|
|
625
|
+
|
|
626
|
+
def input_gfa(self, stream):
|
|
627
|
+
self._input_header(stream, "#~")
|
|
628
|
+
|
|
629
|
+
def output_gfa(self):
|
|
630
|
+
return self._output_header("#~")
|
|
631
|
+
|
|
632
|
+
def input_microdata(self, stream):
|
|
633
|
+
parser = _MetadataHTML()
|
|
634
|
+
parser.feed(_text(stream))
|
|
635
|
+
parser.close()
|
|
636
|
+
if parser.result is None:
|
|
637
|
+
raise ValueError("No complete FHR microdata item scope found")
|
|
638
|
+
data = parser.result
|
|
639
|
+
# Standard microdata scalar values are strings; restore schema-defined types.
|
|
640
|
+
for key, definition in SCHEMA["properties"].items():
|
|
641
|
+
if key not in data:
|
|
642
|
+
continue
|
|
643
|
+
if definition.get("type") == "array" and not isinstance(data[key], list):
|
|
644
|
+
data[key] = [data[key]]
|
|
645
|
+
if definition.get("type") in {"number", "integer"} and isinstance(
|
|
646
|
+
data[key], str
|
|
647
|
+
):
|
|
648
|
+
data[key] = json.loads(data[key])
|
|
649
|
+
if isinstance(data.get("assemblySoftware"), dict):
|
|
650
|
+
data["assemblySoftware"] = [data["assemblySoftware"]]
|
|
651
|
+
if isinstance(data.get("assemblySoftware"), list):
|
|
652
|
+
for software in data["assemblySoftware"]:
|
|
653
|
+
if isinstance(software, dict) and isinstance(
|
|
654
|
+
software.get("commandLineOption"), str
|
|
655
|
+
):
|
|
656
|
+
software["commandLineOption"] = [software["commandLineOption"]]
|
|
657
|
+
if isinstance(data.get("vitalStats"), dict):
|
|
658
|
+
for key, value in data["vitalStats"].items():
|
|
659
|
+
kind = (
|
|
660
|
+
SCHEMA["properties"]["vitalStats"]["properties"]
|
|
661
|
+
.get(key, {})
|
|
662
|
+
.get("type")
|
|
663
|
+
)
|
|
664
|
+
if kind in {"number", "integer"} and isinstance(value, str):
|
|
665
|
+
data["vitalStats"][key] = json.loads(value)
|
|
666
|
+
self._input(data)
|
|
667
|
+
|
|
668
|
+
def output_microdata(self):
|
|
669
|
+
content = "".join(
|
|
670
|
+
_html_value(key, value) for key, value in self.__dict__.items()
|
|
671
|
+
)
|
|
672
|
+
return f'<div itemscope itemtype="{ITEM_TYPE}">{content}</div>\n'
|
|
673
|
+
|
|
674
|
+
def fhr_validate(self):
|
|
675
|
+
json.dumps(self.__dict__, allow_nan=False)
|
|
676
|
+
Draft202012Validator(SCHEMA, format_checker=FormatChecker()).validate(
|
|
677
|
+
self.__dict__
|
|
678
|
+
)
|