telemetry-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- telemetry_cli/__init__.py +8 -0
- telemetry_cli/_cli.py +18 -0
- telemetry_cli/_getopt.py +114 -0
- telemetry_cli/_msg.py +39 -0
- telemetry_cli/_pack.py +183 -0
- telemetry_cli/_perl.py +183 -0
- telemetry_cli/pick/__init__.py +1 -0
- telemetry_cli/pick/__main__.py +7 -0
- telemetry_cli/pick/cli.py +157 -0
- telemetry_cli/pick/commandline.py +219 -0
- telemetry_cli/pick/fields.py +166 -0
- telemetry_cli/pick/formats.py +191 -0
- telemetry_cli/pick/help.py +536 -0
- telemetry_cli/pick/symbols/airmoc +24 -0
- telemetry_cli/pick/symbols/aux +28 -0
- telemetry_cli/pick/symbols/corr +31 -0
- telemetry_cli/pick/symbols/corrOld +29 -0
- telemetry_cli/pick/symbols/hw +16 -0
- telemetry_cli/pick/symbols/moc +42 -0
- telemetry_cli/pick/symbols/subcom +83 -0
- telemetry_cli/pick/symbols/subcom_2002 +111 -0
- telemetry_cli/recl/__init__.py +1 -0
- telemetry_cli/recl/__main__.py +7 -0
- telemetry_cli/recl/cli.py +214 -0
- telemetry_cli/recl/corr.py +83 -0
- telemetry_cli/recs/__init__.py +1 -0
- telemetry_cli/recs/__main__.py +7 -0
- telemetry_cli/recs/cli.py +182 -0
- telemetry_cli/recs/extract.py +399 -0
- telemetry_cli/tgen/__init__.py +1 -0
- telemetry_cli/tgen/__main__.py +7 -0
- telemetry_cli/tgen/cli.py +128 -0
- telemetry_cli/tgen/cmdfile.py +68 -0
- telemetry_cli/tgen/convert.py +278 -0
- telemetry_cli/tgen/examples/arb_data.tgen +15 -0
- telemetry_cli/tgen/examples/calsweep.tgen +32 -0
- telemetry_cli/tgen/examples/cond_fill.tgen +17 -0
- telemetry_cli/tgen/examples/counter.tgen +11 -0
- telemetry_cli/tgen/examples/null_data.tgen +11 -0
- telemetry_cli/tgen/examples/random_size.tgen +20 -0
- telemetry_cli/tgen/examples/subcom.tgen +14 -0
- telemetry_cli/tgen/examples/valueV_value.tgen +20 -0
- telemetry_cli/tgen/runtime.py +390 -0
- telemetry_cli-1.0.0.dist-info/METADATA +106 -0
- telemetry_cli-1.0.0.dist-info/RECORD +48 -0
- telemetry_cli-1.0.0.dist-info/WHEEL +4 -0
- telemetry_cli-1.0.0.dist-info/entry_points.txt +5 -0
- telemetry_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,182 @@
|
|
|
1
|
+
"""The recs command: extract records from a stream of data.
|
|
2
|
+
|
|
3
|
+
recs filename [ options ] record_start_match [ reclen_bytes ]
|
|
4
|
+
cat filename | recs - [ options ] record_start_match [ reclen_bytes ]
|
|
5
|
+
|
|
6
|
+
Run ``recs -help`` for help.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import os
|
|
12
|
+
import sys
|
|
13
|
+
from typing import BinaryIO, TextIO
|
|
14
|
+
|
|
15
|
+
from .._cli import broken_pipe
|
|
16
|
+
from .._getopt import OptionError, getoptions
|
|
17
|
+
from .extract import Extractor, RecsError
|
|
18
|
+
|
|
19
|
+
SPECS = ["v", "x", "z", "mml=i", "e=s", "a", "f=s", "r", "test", "help", "min_reclen=i",
|
|
20
|
+
"max_reclen=i", "append=s", "prepend=s", "nl"] # fmt: skip
|
|
21
|
+
|
|
22
|
+
HELP = """\
|
|
23
|
+
recs - record extraction utility
|
|
24
|
+
|
|
25
|
+
SYNOPSIS
|
|
26
|
+
recs filename [ options ] record_start_match [ reclen_bytes ]
|
|
27
|
+
cat filename | recs - [ options ] record_start_match [ reclen_bytes ]
|
|
28
|
+
|
|
29
|
+
DESCRIPTION
|
|
30
|
+
recs extracts records from a data stream. The arguments are:
|
|
31
|
+
|
|
32
|
+
filename
|
|
33
|
+
File from which to extract records. If filename is '-', then
|
|
34
|
+
stdin is used for input.
|
|
35
|
+
|
|
36
|
+
record_start_match
|
|
37
|
+
String to match for the beginning of a record. By default, this is
|
|
38
|
+
a simple string, but see the -r option.
|
|
39
|
+
|
|
40
|
+
reclen_bytes
|
|
41
|
+
Number of bytes for each record. If not given, records are
|
|
42
|
+
extracted from each record_start_match to the next one (or to the
|
|
43
|
+
end marker given with -e).
|
|
44
|
+
|
|
45
|
+
OPTIONS
|
|
46
|
+
Options can be abbreviated, and can go anywhere on the command line.
|
|
47
|
+
|
|
48
|
+
-a Start a new record at every match of the start marker, even if the
|
|
49
|
+
number of bytes for a record has not yet been read. Missing bytes
|
|
50
|
+
are filled with the fill string.
|
|
51
|
+
|
|
52
|
+
-f=string
|
|
53
|
+
Fill string for padded records. By default, a null byte.
|
|
54
|
+
|
|
55
|
+
-e=string
|
|
56
|
+
String to match for the end of a record.
|
|
57
|
+
|
|
58
|
+
-r Interpret match strings as regular expressions (Python syntax),
|
|
59
|
+
rather than simple strings.
|
|
60
|
+
|
|
61
|
+
-x Exclude the bytes matching the start marker from the records.
|
|
62
|
+
|
|
63
|
+
-z Exclude the bytes matching the end marker from the records.
|
|
64
|
+
|
|
65
|
+
-mml=bytes
|
|
66
|
+
The maximum expected length of a match. This keeps matches from
|
|
67
|
+
being missed when they straddle a buffer boundary. A warning is
|
|
68
|
+
printed if a longer match is found. By default, the length of the
|
|
69
|
+
match string, which is not a good guess for a regex such as '\\d+'.
|
|
70
|
+
|
|
71
|
+
-min_reclen=bytes
|
|
72
|
+
Minimum record length: shorter records are padded with the fill
|
|
73
|
+
string. Works for fixed-length and marker-to-marker extraction.
|
|
74
|
+
|
|
75
|
+
-max_reclen=bytes
|
|
76
|
+
Maximum record length: longer records are truncated.
|
|
77
|
+
|
|
78
|
+
-prepend=string
|
|
79
|
+
A string to write before each record.
|
|
80
|
+
|
|
81
|
+
-append=string
|
|
82
|
+
A string to write after each record.
|
|
83
|
+
|
|
84
|
+
-nl Write a newline after each record (after the -append string).
|
|
85
|
+
|
|
86
|
+
-v Verbose (accepted, but has no effect).
|
|
87
|
+
|
|
88
|
+
-help
|
|
89
|
+
Show this help.
|
|
90
|
+
|
|
91
|
+
The full manual is at
|
|
92
|
+
https://github.com/donalgrant/telemetry-cli/blob/main/docs/recs.md
|
|
93
|
+
"""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _bytes(s: str) -> bytes:
|
|
97
|
+
"""A command-line string as the bytes the user typed."""
|
|
98
|
+
return os.fsencode(s)
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def main(
|
|
102
|
+
argv: list[str] | None = None,
|
|
103
|
+
*,
|
|
104
|
+
stdin: BinaryIO | None = None,
|
|
105
|
+
stdout: BinaryIO | None = None,
|
|
106
|
+
stderr: TextIO | None = None,
|
|
107
|
+
) -> int:
|
|
108
|
+
argv = sys.argv[1:] if argv is None else list(argv)
|
|
109
|
+
stdin = stdin if stdin is not None else sys.stdin.buffer
|
|
110
|
+
stdout = stdout if stdout is not None else sys.stdout.buffer
|
|
111
|
+
stderr = stderr if stderr is not None else sys.stderr
|
|
112
|
+
|
|
113
|
+
def error(message: str, status: int = 1) -> int:
|
|
114
|
+
print(f"recs: {message}", file=stderr)
|
|
115
|
+
return status
|
|
116
|
+
|
|
117
|
+
try:
|
|
118
|
+
o, args = getoptions(argv, SPECS)
|
|
119
|
+
except OptionError as e:
|
|
120
|
+
return error(f"{e} (see 'recs -help')", 2)
|
|
121
|
+
if o.get("help"):
|
|
122
|
+
stdout.write(HELP.encode())
|
|
123
|
+
return 0
|
|
124
|
+
if o.get("test"):
|
|
125
|
+
print("recs: the -test self-tests are now part of telemetry-cli's test suite", file=stderr)
|
|
126
|
+
return 0
|
|
127
|
+
if not args:
|
|
128
|
+
return error("no input file given (use - for stdin); see 'recs -help'", 2)
|
|
129
|
+
if len(args) < 2:
|
|
130
|
+
return error("no record start marker given; see 'recs -help'", 2)
|
|
131
|
+
file, match = args[0], _bytes(args[1])
|
|
132
|
+
if not match:
|
|
133
|
+
return error("the record start marker is empty", 2)
|
|
134
|
+
try:
|
|
135
|
+
out_length = int(args[2]) if len(args) > 2 else -1
|
|
136
|
+
except ValueError:
|
|
137
|
+
return error(f"the record length must be a number of bytes, not {args[2]!r}", 2)
|
|
138
|
+
|
|
139
|
+
if out_length < 0 and "e" not in o:
|
|
140
|
+
# with no record length and no end marker, take records from start to start
|
|
141
|
+
o["e"] = args[1]
|
|
142
|
+
o.setdefault("a", 1)
|
|
143
|
+
append = _bytes(o.get("append", "")) + (b"\n" if o.get("nl") else b"")
|
|
144
|
+
options = {
|
|
145
|
+
"xbeg": o.get("x"),
|
|
146
|
+
"xend": o.get("z"),
|
|
147
|
+
"mml": o.get("mml", len(match)),
|
|
148
|
+
"sp": o.get("a"),
|
|
149
|
+
"fill": _bytes(o["f"]) if "f" in o else None,
|
|
150
|
+
"regex": o.get("r"),
|
|
151
|
+
"nmin": o.get("min_reclen"),
|
|
152
|
+
"nmax": o.get("max_reclen"),
|
|
153
|
+
"pre": _bytes(o["prepend"]) if "prepend" in o else None,
|
|
154
|
+
"post": append,
|
|
155
|
+
}
|
|
156
|
+
|
|
157
|
+
def warn(message: str) -> None:
|
|
158
|
+
print(f"recs: warning: {message}", file=stderr)
|
|
159
|
+
|
|
160
|
+
try:
|
|
161
|
+
if file == "-":
|
|
162
|
+
inp = stdin
|
|
163
|
+
else:
|
|
164
|
+
inp = open(file, "rb") # noqa: SIM115
|
|
165
|
+
except OSError as e:
|
|
166
|
+
return error(f"Can't open {file} for input: {e.strerror}")
|
|
167
|
+
try:
|
|
168
|
+
E = Extractor(inp, stdout, options, warn=warn)
|
|
169
|
+
if out_length < 0:
|
|
170
|
+
stop = _bytes(o["e"])
|
|
171
|
+
(E.reg_reg if o.get("r") else E.str_str)(match, stop)
|
|
172
|
+
else:
|
|
173
|
+
(E.reg_n if o.get("r") else E.str_n)(match, out_length)
|
|
174
|
+
except RecsError as e:
|
|
175
|
+
return error(str(e))
|
|
176
|
+
except BrokenPipeError:
|
|
177
|
+
return broken_pipe(stdout)
|
|
178
|
+
finally:
|
|
179
|
+
if inp is not stdin:
|
|
180
|
+
inp.close()
|
|
181
|
+
stdout.flush()
|
|
182
|
+
return 0
|
|
@@ -0,0 +1,399 @@
|
|
|
1
|
+
"""Extracting records from a byte stream by their start and end markers.
|
|
2
|
+
|
|
3
|
+
This is a close port of the Perl original's three classes: a fixed-size
|
|
4
|
+
buffer over the input (BStream), string and regex search in that buffer
|
|
5
|
+
(Match_Stream), and the extractor (Extract_Records). The buffer mechanics are
|
|
6
|
+
kept as they were, because whether a marker that straddles two buffers is
|
|
7
|
+
found depends on them; see the ``mml`` option.
|
|
8
|
+
|
|
9
|
+
Regexes are Python regexes (on bytes, with ``.`` matching newlines, as the
|
|
10
|
+
Perl's ``/s`` did). For the simple patterns recs is used with, they behave
|
|
11
|
+
like Perl's.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import re
|
|
17
|
+
from collections.abc import Callable
|
|
18
|
+
from typing import BinaryIO
|
|
19
|
+
|
|
20
|
+
from .._perl import truthy
|
|
21
|
+
|
|
22
|
+
Finder = Callable[["MatchStream", bytes], tuple[int, int]]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class RecsError(ValueError):
|
|
26
|
+
pass
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class BStream:
|
|
30
|
+
"""A buffer of up to ``size`` bytes over a binary stream, refilled as it is consumed."""
|
|
31
|
+
|
|
32
|
+
def __init__(self, fp: BinaryIO, size: int = 1024):
|
|
33
|
+
self.fp = fp
|
|
34
|
+
self.size = size or 1024 # size=0 is not allowed
|
|
35
|
+
self.data = b""
|
|
36
|
+
self.consumed = 0 # net bytes taken from the front (for spotting endless loops)
|
|
37
|
+
self._add_bytes()
|
|
38
|
+
|
|
39
|
+
def _add_bytes(self, offset: int = 0, nbytes: int | None = None) -> int:
|
|
40
|
+
nbytes = self.size if nbytes is None else nbytes
|
|
41
|
+
chunk = self.fp.read(nbytes) if nbytes > 0 else b""
|
|
42
|
+
# Perl read into substr($data, $offset, $size - $offset)
|
|
43
|
+
self.data = self.data[:offset] + chunk + self.data[self.size :]
|
|
44
|
+
return len(chunk)
|
|
45
|
+
|
|
46
|
+
def __len__(self) -> int:
|
|
47
|
+
return len(self.data)
|
|
48
|
+
|
|
49
|
+
@property
|
|
50
|
+
def full(self) -> bool:
|
|
51
|
+
return self.size <= len(self.data)
|
|
52
|
+
|
|
53
|
+
def get(self, n: int | None = None, offset: int = 0) -> bytes | None:
|
|
54
|
+
"""n bytes from offset (all by default); None if n or offset is past the end."""
|
|
55
|
+
length = len(self.data)
|
|
56
|
+
n = length if n is None else n
|
|
57
|
+
if n > length or offset > length:
|
|
58
|
+
return None
|
|
59
|
+
n = min(n, length - offset)
|
|
60
|
+
return self.data[offset : offset + n]
|
|
61
|
+
|
|
62
|
+
def shift_bytes(self, nbytes: int | None = None) -> bytes:
|
|
63
|
+
"""Remove and return up to nbytes from the front, then refill to size."""
|
|
64
|
+
nbytes = self.size if nbytes is None else nbytes
|
|
65
|
+
length = len(self.data)
|
|
66
|
+
nbytes = min(nbytes, length)
|
|
67
|
+
out = self.data[:nbytes]
|
|
68
|
+
self.data = self.data[nbytes:]
|
|
69
|
+
self.consumed += len(out)
|
|
70
|
+
to_add = self.size - (length - nbytes)
|
|
71
|
+
if to_add > 0:
|
|
72
|
+
self._add_bytes(length - nbytes, to_add)
|
|
73
|
+
return out
|
|
74
|
+
|
|
75
|
+
def unshift_bytes(self, src: bytes) -> int:
|
|
76
|
+
self.data = src + self.data
|
|
77
|
+
self.consumed -= len(src)
|
|
78
|
+
return len(src)
|
|
79
|
+
|
|
80
|
+
|
|
81
|
+
class MatchStream(BStream):
|
|
82
|
+
def str_find(self, s: bytes) -> tuple[int, int]:
|
|
83
|
+
return self.data.find(s), len(s)
|
|
84
|
+
|
|
85
|
+
def reg_find(self, pattern: bytes) -> tuple[int, int]:
|
|
86
|
+
m = _regex(pattern).search(self.data)
|
|
87
|
+
return (m.start(), m.end() - m.start()) if m else (-1, 0)
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
_REGEX_CACHE: dict[bytes, re.Pattern] = {}
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def _regex(pattern: bytes) -> re.Pattern:
|
|
94
|
+
if pattern not in _REGEX_CACHE:
|
|
95
|
+
try:
|
|
96
|
+
_REGEX_CACHE[pattern] = re.compile(pattern, re.DOTALL)
|
|
97
|
+
except re.error as e:
|
|
98
|
+
text = pattern.decode("latin-1")
|
|
99
|
+
raise RecsError(f"can't use the regex {text!r}: {e}") from None
|
|
100
|
+
return _REGEX_CACHE[pattern]
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def _perl_substr_from_end(data: bytes, n: int) -> tuple[bytes, bytes]:
|
|
104
|
+
"""Perl's (substr($r, -n), substr($r, 0, -n)). With n == 0 that is ($r, "")."""
|
|
105
|
+
if n == 0:
|
|
106
|
+
return data, b""
|
|
107
|
+
if n >= len(data):
|
|
108
|
+
return data, b""
|
|
109
|
+
return data[-n:], data[:-n]
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# Option names and defaults, as in the Perl's %OPTION table, except that the
|
|
113
|
+
# default fill is a null byte, as documented. (The Perl's default was the
|
|
114
|
+
# number 0, which padded records with the character "0".)
|
|
115
|
+
DEFAULTS = {
|
|
116
|
+
"fill": b"\0",
|
|
117
|
+
"pad": 0,
|
|
118
|
+
"regex": 1,
|
|
119
|
+
"mml": 0,
|
|
120
|
+
"sp": 0,
|
|
121
|
+
"xbeg": 0,
|
|
122
|
+
"xend": 0,
|
|
123
|
+
"pre": b"",
|
|
124
|
+
"post": b"",
|
|
125
|
+
"reclen": 0,
|
|
126
|
+
"nmin": 0,
|
|
127
|
+
"nmax": 0,
|
|
128
|
+
"bufsiz": 1024,
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
class Extractor:
|
|
133
|
+
"""Extract records from ``inp`` and write them to ``out``.
|
|
134
|
+
|
|
135
|
+
Options (all optional): fill, pad, regex, mml, sp, xbeg, xend, pre, post,
|
|
136
|
+
reclen, nmin, nmax, bufsiz, with the Perl original's meanings. ``warn`` is
|
|
137
|
+
called with the text of each warning.
|
|
138
|
+
"""
|
|
139
|
+
|
|
140
|
+
# guards against inputs that made the Perl loop forever
|
|
141
|
+
MAX_STEPS_WITHOUT_PROGRESS = 10_000
|
|
142
|
+
|
|
143
|
+
def __init__(
|
|
144
|
+
self,
|
|
145
|
+
inp: BinaryIO,
|
|
146
|
+
out: BinaryIO,
|
|
147
|
+
options: dict | None = None,
|
|
148
|
+
*,
|
|
149
|
+
defaults: dict | None = None,
|
|
150
|
+
warn: Callable[[str], None] = lambda m: None,
|
|
151
|
+
):
|
|
152
|
+
self.inp, self.out, self.warn = inp, out, warn
|
|
153
|
+
self.o: dict = {}
|
|
154
|
+
self.record = b""
|
|
155
|
+
for k, v in (defaults or DEFAULTS).items():
|
|
156
|
+
self._set(k, v)
|
|
157
|
+
for k, v in (options or {}).items():
|
|
158
|
+
if v is not None:
|
|
159
|
+
self._set(k, v)
|
|
160
|
+
|
|
161
|
+
# --- options, with the Perl setters' side effects ---------------------------
|
|
162
|
+
|
|
163
|
+
def _set(self, key: str, value) -> None:
|
|
164
|
+
setter = getattr(self, f"set_{key}", None)
|
|
165
|
+
if setter:
|
|
166
|
+
setter(value)
|
|
167
|
+
else:
|
|
168
|
+
self.o[key] = value
|
|
169
|
+
|
|
170
|
+
def set_pad(self, v) -> None:
|
|
171
|
+
self.o["pad"] = v
|
|
172
|
+
if truthy(self.o.get("reclen")) and truthy(v):
|
|
173
|
+
self.o["nmin"] = self.o["reclen"]
|
|
174
|
+
|
|
175
|
+
def set_nmin(self, v) -> None:
|
|
176
|
+
self.o["nmin"] = v
|
|
177
|
+
if truthy(v) and not truthy(self.o.get("pad")):
|
|
178
|
+
self.o["pad"] = 1
|
|
179
|
+
|
|
180
|
+
def set_reclen(self, v) -> None:
|
|
181
|
+
self.o["reclen"] = v
|
|
182
|
+
if self.o.get("mml") is None:
|
|
183
|
+
self.set_mml(DEFAULTS["mml"])
|
|
184
|
+
self._grow_buffer()
|
|
185
|
+
if truthy(v):
|
|
186
|
+
self.o["nmax"] = v
|
|
187
|
+
if truthy(self.o.get("pad")) and truthy(v):
|
|
188
|
+
self.set_nmin(v)
|
|
189
|
+
|
|
190
|
+
def set_mml(self, v) -> None:
|
|
191
|
+
self.o["mml"] = v
|
|
192
|
+
if self.o.get("reclen") is None:
|
|
193
|
+
self.set_reclen(DEFAULTS["reclen"])
|
|
194
|
+
self._grow_buffer()
|
|
195
|
+
|
|
196
|
+
def _grow_buffer(self) -> None:
|
|
197
|
+
need = 2 * ((self.o.get("reclen") or 0) + (self.o.get("mml") or 0))
|
|
198
|
+
if need > self.o.get("bufsiz", 0):
|
|
199
|
+
self.o["bufsiz"] = need
|
|
200
|
+
|
|
201
|
+
# --- the record being built -------------------------------------------------
|
|
202
|
+
|
|
203
|
+
def clear_rec(self) -> None:
|
|
204
|
+
self.record = b""
|
|
205
|
+
|
|
206
|
+
def add_rec(self, b: bytes) -> None:
|
|
207
|
+
self.record += b
|
|
208
|
+
|
|
209
|
+
def trim_rec(self) -> None:
|
|
210
|
+
if len(self.record) > self.o["nmax"]:
|
|
211
|
+
self.record = self.record[: self.o["nmax"]]
|
|
212
|
+
|
|
213
|
+
def fill_rec(self) -> None:
|
|
214
|
+
fill = _fill_bytes(self.o["fill"])
|
|
215
|
+
if not fill and len(self.record) < self.o["nmin"]:
|
|
216
|
+
raise RecsError("the fill string is empty, so records can't be padded")
|
|
217
|
+
while len(self.record) < self.o["nmin"]:
|
|
218
|
+
self.add_rec(fill)
|
|
219
|
+
|
|
220
|
+
def pop_rec(self, n: int) -> bytes:
|
|
221
|
+
popped, self.record = _perl_substr_from_end(self.record, n)
|
|
222
|
+
return popped
|
|
223
|
+
|
|
224
|
+
def dump_rec(self) -> None:
|
|
225
|
+
if truthy(self.o["nmin"]):
|
|
226
|
+
self.fill_rec()
|
|
227
|
+
if truthy(self.o["nmax"]):
|
|
228
|
+
self.trim_rec()
|
|
229
|
+
self.out.write(self.o["pre"] + self.record + self.o["post"])
|
|
230
|
+
|
|
231
|
+
def find_in_rec(self, match: bytes, offset: int | None = 0) -> tuple[int, int]:
|
|
232
|
+
offset = offset or 0
|
|
233
|
+
rest = self.record[offset:]
|
|
234
|
+
if truthy(self.o["regex"]):
|
|
235
|
+
m = _regex(match).search(rest)
|
|
236
|
+
if not m:
|
|
237
|
+
return -1, 0
|
|
238
|
+
mb = m.end() - m.start()
|
|
239
|
+
self._check_mml(mb)
|
|
240
|
+
return offset + m.start(), mb
|
|
241
|
+
return offset + rest.find(match), len(match)
|
|
242
|
+
|
|
243
|
+
def _check_mml(self, nbytes: int) -> None:
|
|
244
|
+
if nbytes > self.o["mml"]:
|
|
245
|
+
self.warn(f"Matched bytes ({nbytes}) is larger than mml parameter ({self.o['mml']})")
|
|
246
|
+
|
|
247
|
+
# --- moving through the stream ---------------------------------------------------
|
|
248
|
+
|
|
249
|
+
def _finder(self) -> Finder:
|
|
250
|
+
return MatchStream.reg_find if truthy(self.o["regex"]) else MatchStream.str_find
|
|
251
|
+
|
|
252
|
+
def _discard_to_match(self, match: bytes) -> int | None:
|
|
253
|
+
find = self._finder()
|
|
254
|
+
shift = self.M.size - self.o["mml"]
|
|
255
|
+
while True:
|
|
256
|
+
r = find(self.M, match)
|
|
257
|
+
if not (r[0] < 0 and len(self.M)):
|
|
258
|
+
break
|
|
259
|
+
self._check_mml(r[1])
|
|
260
|
+
self.M.shift_bytes(shift)
|
|
261
|
+
if r[0] < 0:
|
|
262
|
+
return None
|
|
263
|
+
self.M.shift_bytes(r[0])
|
|
264
|
+
return r[1]
|
|
265
|
+
|
|
266
|
+
def _extract_to_match(self, match: bytes) -> int | None:
|
|
267
|
+
find = self._finder()
|
|
268
|
+
shift = self.M.size - self.o["mml"]
|
|
269
|
+
while True:
|
|
270
|
+
r = find(self.M, match)
|
|
271
|
+
if not (r[0] < 0 and len(self.M)):
|
|
272
|
+
break
|
|
273
|
+
self.add_rec(self.M.shift_bytes(shift))
|
|
274
|
+
if r[0] < 0:
|
|
275
|
+
return None
|
|
276
|
+
self._check_mml(r[1])
|
|
277
|
+
self.add_rec(self.M.shift_bytes(r[0] + r[1]))
|
|
278
|
+
return r[1]
|
|
279
|
+
|
|
280
|
+
def _extract_nbytes(self, n: int) -> int:
|
|
281
|
+
"""Add n bytes to the record; return how many were missing at the end of input."""
|
|
282
|
+
while n > (length := len(self.M)):
|
|
283
|
+
if not length:
|
|
284
|
+
if not truthy(self.o["pad"]):
|
|
285
|
+
return n
|
|
286
|
+
self.fill_rec()
|
|
287
|
+
self.trim_rec()
|
|
288
|
+
return 0
|
|
289
|
+
self.add_rec(self.M.shift_bytes())
|
|
290
|
+
n -= length # (the Perl subtracted the length, even if it shifted less)
|
|
291
|
+
if n > 0:
|
|
292
|
+
self.add_rec(self.M.shift_bytes(n))
|
|
293
|
+
return 0
|
|
294
|
+
|
|
295
|
+
# --- the two kinds of extraction -------------------------------------------------
|
|
296
|
+
|
|
297
|
+
def match_n(self, match: bytes, reclen: int, mml: int | None = None, max_recs=-1) -> int:
|
|
298
|
+
"""Records of reclen bytes, each starting at a match."""
|
|
299
|
+
mml = mml or len(match)
|
|
300
|
+
if mml > self.o["mml"]:
|
|
301
|
+
self.set_mml(mml)
|
|
302
|
+
self.set_reclen(reclen)
|
|
303
|
+
self.M = MatchStream(self.inp, self.o["bufsiz"])
|
|
304
|
+
i, stuck, last = 0, 0, None
|
|
305
|
+
while (i <= max_recs or max_recs < 0) and len(self.M) > 0:
|
|
306
|
+
last, stuck = self._progress(last, stuck)
|
|
307
|
+
mb_start = self._discard_to_match(match)
|
|
308
|
+
if mb_start is None:
|
|
309
|
+
break
|
|
310
|
+
if truthy(self.o["xbeg"]):
|
|
311
|
+
self.M.shift_bytes(mb_start)
|
|
312
|
+
self.clear_rec()
|
|
313
|
+
missing = self._extract_nbytes(self.o["reclen"])
|
|
314
|
+
if truthy(self.o["sp"]):
|
|
315
|
+
rec_offset = 0 if truthy(self.o["xbeg"]) else mb_start
|
|
316
|
+
r = self.find_in_rec(match, rec_offset)
|
|
317
|
+
if r[0] >= rec_offset:
|
|
318
|
+
# another start marker: put the rest back on the stream
|
|
319
|
+
self.M.unshift_bytes(self.pop_rec(len(self.record) - r[0]))
|
|
320
|
+
if not truthy(self.o["pad"]):
|
|
321
|
+
continue
|
|
322
|
+
if missing and not truthy(self.o["pad"]):
|
|
323
|
+
break # an incomplete record, and not padding
|
|
324
|
+
self.dump_rec()
|
|
325
|
+
i += 1
|
|
326
|
+
return i
|
|
327
|
+
|
|
328
|
+
def match_match(
|
|
329
|
+
self, start: bytes, stop: bytes | None = None, mml: int | None = None, max_recs=-1
|
|
330
|
+
) -> int:
|
|
331
|
+
"""Records from each start marker to the following stop marker."""
|
|
332
|
+
stop = start if stop is None else stop
|
|
333
|
+
mml = mml or max(len(start), len(stop))
|
|
334
|
+
if mml > self.o["mml"]:
|
|
335
|
+
self.set_mml(mml)
|
|
336
|
+
self.M = MatchStream(self.inp, self.o["bufsiz"])
|
|
337
|
+
i, stuck, last = 0, 0, None
|
|
338
|
+
while (i <= max_recs or max_recs < 0) and len(self.M) > 0:
|
|
339
|
+
last, stuck = self._progress(last, stuck)
|
|
340
|
+
mb_start = self._discard_to_match(start)
|
|
341
|
+
if mb_start is None:
|
|
342
|
+
break
|
|
343
|
+
self.clear_rec()
|
|
344
|
+
mb0 = 0
|
|
345
|
+
if truthy(self.o["xbeg"]):
|
|
346
|
+
self.M.shift_bytes(mb_start)
|
|
347
|
+
else:
|
|
348
|
+
mb0 = self._extract_to_match(start)
|
|
349
|
+
mb = self._extract_to_match(stop)
|
|
350
|
+
if truthy(self.o["sp"]):
|
|
351
|
+
r = self.find_in_rec(start, mb0)
|
|
352
|
+
if r[0] >= (mb0 or 0):
|
|
353
|
+
self.M.unshift_bytes(self.pop_rec(len(self.record) - r[0]))
|
|
354
|
+
self.dump_rec() # no stop marker: dump what we have
|
|
355
|
+
i += 1
|
|
356
|
+
continue
|
|
357
|
+
if not mb:
|
|
358
|
+
self.clear_rec() # only complete records
|
|
359
|
+
return i
|
|
360
|
+
if truthy(self.o["xend"]):
|
|
361
|
+
self.pop_rec(mb)
|
|
362
|
+
self.dump_rec()
|
|
363
|
+
i += 1
|
|
364
|
+
return i
|
|
365
|
+
|
|
366
|
+
def _progress(self, most, stuck):
|
|
367
|
+
"""Stop if many passes in a row consume no new input."""
|
|
368
|
+
if most is not None and self.M.consumed <= most:
|
|
369
|
+
stuck += 1
|
|
370
|
+
if stuck > self.MAX_STEPS_WITHOUT_PROGRESS:
|
|
371
|
+
raise RecsError("the markers match without consuming any input; stopping")
|
|
372
|
+
return most, stuck
|
|
373
|
+
return self.M.consumed, 0
|
|
374
|
+
|
|
375
|
+
# Perl-style entry points: str_* use plain strings, reg_* regexes
|
|
376
|
+
def str_n(self, match, reclen, *a):
|
|
377
|
+
self.o["regex"] = 0
|
|
378
|
+
return self.match_n(match, reclen, *a)
|
|
379
|
+
|
|
380
|
+
def reg_n(self, match, reclen, *a):
|
|
381
|
+
self.o["regex"] = 1
|
|
382
|
+
return self.match_n(match, reclen, *a)
|
|
383
|
+
|
|
384
|
+
def str_str(self, start, stop=None, *a):
|
|
385
|
+
self.o["regex"] = 0
|
|
386
|
+
return self.match_match(start, stop, *a)
|
|
387
|
+
|
|
388
|
+
def reg_reg(self, start, stop=None, *a):
|
|
389
|
+
self.o["regex"] = 1
|
|
390
|
+
return self.match_match(start, stop, *a)
|
|
391
|
+
|
|
392
|
+
|
|
393
|
+
def _fill_bytes(fill) -> bytes:
|
|
394
|
+
"""The fill as bytes. (A number is its digits, as in Perl: 0 is "0".)"""
|
|
395
|
+
if isinstance(fill, bytes):
|
|
396
|
+
return fill
|
|
397
|
+
if isinstance(fill, str):
|
|
398
|
+
return fill.encode("latin-1")
|
|
399
|
+
return str(fill).encode()
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""tgen: generate telemetry records from a command file."""
|