telemetry-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (48) hide show
  1. telemetry_cli/__init__.py +8 -0
  2. telemetry_cli/_cli.py +18 -0
  3. telemetry_cli/_getopt.py +114 -0
  4. telemetry_cli/_msg.py +39 -0
  5. telemetry_cli/_pack.py +183 -0
  6. telemetry_cli/_perl.py +183 -0
  7. telemetry_cli/pick/__init__.py +1 -0
  8. telemetry_cli/pick/__main__.py +7 -0
  9. telemetry_cli/pick/cli.py +157 -0
  10. telemetry_cli/pick/commandline.py +219 -0
  11. telemetry_cli/pick/fields.py +166 -0
  12. telemetry_cli/pick/formats.py +191 -0
  13. telemetry_cli/pick/help.py +536 -0
  14. telemetry_cli/pick/symbols/airmoc +24 -0
  15. telemetry_cli/pick/symbols/aux +28 -0
  16. telemetry_cli/pick/symbols/corr +31 -0
  17. telemetry_cli/pick/symbols/corrOld +29 -0
  18. telemetry_cli/pick/symbols/hw +16 -0
  19. telemetry_cli/pick/symbols/moc +42 -0
  20. telemetry_cli/pick/symbols/subcom +83 -0
  21. telemetry_cli/pick/symbols/subcom_2002 +111 -0
  22. telemetry_cli/recl/__init__.py +1 -0
  23. telemetry_cli/recl/__main__.py +7 -0
  24. telemetry_cli/recl/cli.py +214 -0
  25. telemetry_cli/recl/corr.py +83 -0
  26. telemetry_cli/recs/__init__.py +1 -0
  27. telemetry_cli/recs/__main__.py +7 -0
  28. telemetry_cli/recs/cli.py +182 -0
  29. telemetry_cli/recs/extract.py +399 -0
  30. telemetry_cli/tgen/__init__.py +1 -0
  31. telemetry_cli/tgen/__main__.py +7 -0
  32. telemetry_cli/tgen/cli.py +128 -0
  33. telemetry_cli/tgen/cmdfile.py +68 -0
  34. telemetry_cli/tgen/convert.py +278 -0
  35. telemetry_cli/tgen/examples/arb_data.tgen +15 -0
  36. telemetry_cli/tgen/examples/calsweep.tgen +32 -0
  37. telemetry_cli/tgen/examples/cond_fill.tgen +17 -0
  38. telemetry_cli/tgen/examples/counter.tgen +11 -0
  39. telemetry_cli/tgen/examples/null_data.tgen +11 -0
  40. telemetry_cli/tgen/examples/random_size.tgen +20 -0
  41. telemetry_cli/tgen/examples/subcom.tgen +14 -0
  42. telemetry_cli/tgen/examples/valueV_value.tgen +20 -0
  43. telemetry_cli/tgen/runtime.py +390 -0
  44. telemetry_cli-1.0.0.dist-info/METADATA +106 -0
  45. telemetry_cli-1.0.0.dist-info/RECORD +48 -0
  46. telemetry_cli-1.0.0.dist-info/WHEEL +4 -0
  47. telemetry_cli-1.0.0.dist-info/entry_points.txt +5 -0
  48. telemetry_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
@@ -0,0 +1,42 @@
1
+ ; This is the symbol table for .moc files, which
2
+ ; are the radar motion compensation files for section 334.
3
+
4
+ length=21d ; length is 21 double-precision (8 byte) numbers
5
+
6
+ ; these are the components of the file
7
+
8
+ xRec = d0G
9
+ xrec = d0G
10
+ time = d1G
11
+ utc = d1G
12
+ prf = d2 ; actually 1/prf...
13
+ tprf = d2
14
+ posS = d3
15
+ posC = d4
16
+ posH = d5
17
+ pos = posS posC posH
18
+ velS = d6
19
+ velC = d7
20
+ velH = d8
21
+ vel = velS velC velH
22
+ af1S = d9
23
+ af1C = d10
24
+ af1H = d11
25
+ af1 = 3d9
26
+ af2 = 3d12
27
+ af3 = 3d15
28
+ az = d18
29
+ yaw = d18
30
+ pitch = d19
31
+ roll = d20
32
+ orient = 3d18
33
+ azD = d18D
34
+ yawD = d18D
35
+ pitchd = d19D
36
+ rollD = d20D
37
+ orientD = 3d18D
38
+ all = 21d0
39
+ af = af1 af2 af3
40
+ angles = posS orientD
41
+ velocities = posS vel
42
+ tube = pos
@@ -0,0 +1,83 @@
1
+ ;
2
+ ; pick symbol table for AIRSAR subcommutated data files
3
+ ;
4
+ length=4096
5
+ ;
6
+ sub_len = w0+0u
7
+ used_len = w0+2u
8
+ ;
9
+ ; data collection notes
10
+ ;
11
+ target = 25A0+439
12
+ invstg = 25A0+464
13
+ hp_opr = 25A0+489
14
+ crew = 25A0+514
15
+ mode = 25A0+539
16
+ status = 25A0+564
17
+ antenna = 25A0+589
18
+ telev = 8A0+638 ; target elevation in feet
19
+ dser1 = 10A0+683 ; digital tape serial number HDDR #1
20
+ dser2 = 10A0+683 ; digital tape serial number HDDR #2
21
+ dser3 = 10A0+683 ; digital tape serial number HDDR #3
22
+ notes = 256A0+1131 ; operator's notes
23
+ ;
24
+ ; data in six-gun position
25
+ ;
26
+ six_month = b0+1412u ; month (1..12)
27
+ six_day = b0+1413u ; day (1..31)
28
+ six_year = w0+1414u ; year (1900..2999)
29
+ six_hour = b0+1416u ; hour (0..23)
30
+ six_min = b0+1417u ; min (0..59)
31
+ six_sec = b0+1418u ; sec (0..60)
32
+ six_nsec = u0+1419u ; fractional seconds (0..999,999,999) [nano-seconds]
33
+ ;
34
+ six_f_cnt = u0+1476u ; 32-bit frame count at GPS time tag
35
+ ;
36
+ ; True-Time data
37
+ ;
38
+ TTmonth = b0+2188u ; month (1..12)
39
+ TTday = b0+2189u ; day (1..31)
40
+ TTyear = w0+2190u ; year (1900..2999)
41
+ TThour = b0+2192u ; hour (0..23)
42
+ TTmin = b0+2193u ; min (0..59)
43
+ TTsec = b0+2194u ; sec (0..60)
44
+ TTnsec = u0+2195u ; fractional seconds (0..999,999,999) [nano-seconds]
45
+ ;
46
+ TTf_cnt = u0+2212u ; 32-bit frame count at True-Time time tag
47
+ ;
48
+ ; EO-17 message
49
+ ;
50
+ egi17TimeTag = w0+2496u ; 1553 time tag
51
+ ;
52
+ ; EO-18 message
53
+ ;
54
+ egi18GpsTimeHi = w0+2560u ; gps time tag word
55
+ egi18GpsTime2 = w0+2562u ; gps time tag word
56
+ egi18GpsTime3 = w0+2564u ; gps time tag word
57
+ egi18GpsTimeLo = w0+2566u ; gps time tag word
58
+
59
+ egi18GpsTimeHigh = u0+2560u ; gps time tag word
60
+ egi18GpsTimeLow = u0+2564u ; gps time tag word
61
+
62
+ egi18UtcTimeHi = w0+2568u ; utc time tag word
63
+ egi18UtcTime2 = w0+2570u ; utc time tag word
64
+ egi18UtcTime3 = w0+2572u ; utc time tag word
65
+ egi18UtcTimeLo = w0+2574u ; utc time tag word
66
+
67
+ egi18UtcTimeHigh = u0+2568u ; utc time tag word
68
+ egi18UtcTimeLow = u0+2572u ; utc time tag word
69
+ ;
70
+ ; EO-19 message
71
+ ;
72
+ egi19TimeTag = w0+2622u ; Time tag
73
+ egi19PltfrmAzTimeTag = w0+2654u ; Platform azimuth time tag
74
+ egi19RollTimeTag = w0+2656u ; Roll time tag
75
+ egi19PitchTimeTag = w0+2658u ; Pitch time tag
76
+ ;
77
+ ; end of record stuff
78
+ ;
79
+ motionSensorFlag = b0+3004d ; Bit 0 set = IGI, bit 1 set = EGI
80
+ hdrVersion = 6A0+3005 ; Date of this revision, eg. 980817
81
+ eosd = 4A0+3011 ; Indicates end of used subheader
82
+
83
+
@@ -0,0 +1,111 @@
1
+ ;
2
+ ; pick symbol table for AIRSAR subcommutated data files
3
+ ;
4
+ length=4096
5
+ ;
6
+ sub_len = w0+0u
7
+ used_len = w0+2u
8
+
9
+ _ = -1 bs
10
+
11
+ ;
12
+ ; data collection notes
13
+ ;
14
+ target = 25A0+439
15
+ invstg = 25A0+464
16
+ hp_opr = 25A0+489
17
+ crew = 25A0+514
18
+ mode = 25A0+539
19
+ status = 25A0+564
20
+ antenna = 25A0+589
21
+ telev = 8A0+638 ; target elevation in feet
22
+ dser1 = 10A0+683 ; digital tape serial number HDDR #1
23
+ dser2 = 10A0+683 ; digital tape serial number HDDR #2
24
+ dser3 = 10A0+683 ; digital tape serial number HDDR #3
25
+ notes = 256A0+1131 ; operator's notes
26
+ ;
27
+ ; data in six-gun position
28
+ ;
29
+ six_month = b0+1412u ; month (1..12)
30
+ six_day = b0+1413u ; day (1..31)
31
+ six_year = w0+1414u ; year (1900..2999)
32
+ six_hour = b0+1416u ; hour (0..23)
33
+ six_min = b0+1417u ; min (0..59)
34
+ six_sec = b0+1418u ; sec (0..60)
35
+ six_nsec = u0+1419u ; fractional seconds (0..999,999,999) [nano-seconds]
36
+ ;
37
+ six_f_cnt = u0+1476u ; 32-bit frame count at GPS time tag
38
+ ;
39
+ ; True-Time data
40
+ ;
41
+ TTmonth = b0+2188u ; month (1..12)
42
+ TTday = b0+2189u ; day (1..31)
43
+ TTyear = w0+2190u ; year (1900..2999)
44
+ TThour = b0+2192u ; hour (0..23)
45
+ TTmin = b0+2193u ; min (0..59)
46
+ TTsec = b0+2194u ; sec (0..60)
47
+ TTnsec = u0+2195u ; fractional seconds (0..999,999,999) [nano-seconds]
48
+ ;
49
+ TTf_cnt = u0+2212u ; 32-bit frame count at True-Time time tag
50
+ ;
51
+ ; EO-17 message
52
+ ;
53
+ egi17TimeTag = w0+2496u ; 1553 time tag
54
+ ;
55
+ ; EO-18 message
56
+ ;
57
+ egi18GpsTimeHi = w0+2560u ; gps time tag word
58
+ egi18GpsTime2 = w0+2562u ; gps time tag word
59
+ egi18GpsTime3 = w0+2564u ; gps time tag word
60
+ egi18GpsTimeLo = w0+2566u ; gps time tag word
61
+
62
+ egi18GpsTimeHigh = u0+2560u ; gps time tag word
63
+ egi18GpsTimeLow = u0+2564u ; gps time tag word
64
+
65
+ egi18UtcTimeHi = w0+2568u ; utc time tag word
66
+ egi18UtcTime2 = w0+2570u ; utc time tag word
67
+ egi18UtcTime3 = w0+2572u ; utc time tag word
68
+ egi18UtcTimeLo = w0+2574u ; utc time tag word
69
+
70
+ egi18UtcTimeHigh = u0+2568u ; utc time tag word
71
+ egi18UtcTimeLow = u0+2572u ; utc time tag word
72
+ ;
73
+ ; EO-19 message
74
+ ;
75
+ egi19TimeTag = w0+2622u ; Time tag
76
+ egi19PltfrmAzTimeTag = w0+2654u ; Platform azimuth time tag
77
+ egi19RollTimeTag = w0+2656u ; Roll time tag
78
+ egi19PitchTimeTag = w0+2658u ; Pitch time tag
79
+ ;
80
+ ; end of record stuff
81
+ ;
82
+ motionSensorFlag = b0+3004d ; Bit 0 set = IGI, bit 1 set = EGI
83
+ hdrVersion = 6A0+3005 ; Date of this revision, eg. 980817
84
+
85
+ dcg_bw_l = 4A0+3011
86
+ dcg_bw_c = 4A0+3015
87
+ dcg_bw_p = 4A0+3019
88
+
89
+ dcg_pd_l = 3A0+3023
90
+ dcg_pd_c = 3A0+3026
91
+ dcg_pd_p = 3A0+3029
92
+
93
+ sampfreq_l = 4A0+3032
94
+ sampfreq_c = 4A0+3036
95
+ sampfreq_p = 4A0+3040
96
+
97
+ dgps_time_J2000 = d0+3428G
98
+
99
+ dgps_x = d0+3444G
100
+ dgps_y = d0+3452G
101
+ dgps_z = d0+3460G
102
+
103
+ dgps_sigma_x = d0+3468
104
+ dgps_sigma_y = d0+3476
105
+ dgps_sigma_z = d0+3484
106
+
107
+ cal_mode = b0+3492d
108
+
109
+ eosd = 4A0+3493 ; Indicates end of used subheader
110
+
111
+
@@ -0,0 +1 @@
1
+ """recl: estimate the record length of a binary file."""
@@ -0,0 +1,7 @@
1
+ """Allow ``python -m telemetry_cli.recl``."""
2
+
3
+ import sys
4
+
5
+ from .cli import main
6
+
7
+ sys.exit(main())
@@ -0,0 +1,214 @@
1
+ """The recl command: estimate the record length of a binary file.
2
+
3
+ recl filename [ options ]
4
+ cat filename | recl - [ options ]
5
+
6
+ Run ``recl -help`` for help.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import sys
12
+ from typing import BinaryIO, TextIO
13
+
14
+ import numpy as np
15
+
16
+ from .._cli import broken_pipe
17
+ from .._getopt import OptionError, getoptions
18
+ from .._msg import Messenger
19
+ from .._perl import perl_str
20
+ from .corr import agreement, choose, compared
21
+
22
+ SPECS = ["v|verbose", "h|help", "f|full", "q|quiet", "part|partial", "skip|header=i", "min=i",
23
+ "max=i", "minrecs=i", "maxbufs=i", "fact|factor|mult=i", "only=s", "limit=i",
24
+ "reduce=f"] # fmt: skip
25
+
26
+ HELP = """\
27
+ recl - estimate the record length of a binary file
28
+
29
+ SYNOPSIS
30
+ recl filename [ options ]
31
+ cat filename | recl - [ options ]
32
+
33
+ DESCRIPTION
34
+ recl compares the file with itself shifted by each possible record
35
+ length R, and reports the R at which the most bytes are equal: records
36
+ of the same format tend to have the same bytes in the same places. It
37
+ prints
38
+
39
+ RESULT R the estimated record length
40
+ CORR x the percentage of bytes equal to the byte R bytes on
41
+ (about 0.4 for unrelated random data)
42
+
43
+ and notes about the lengths it checked. Every multiple of the record
44
+ length agrees about as well as the length itself; recl reports the
45
+ smallest.
46
+
47
+ Unless -partial is given, recl assumes the file (after any header) holds
48
+ whole records, so it only checks lengths that divide its size.
49
+
50
+ OPTIONS
51
+ Options can be abbreviated.
52
+
53
+ -skip=bytes (or -header)
54
+ Skip a header of this many bytes.
55
+ -min=bytes, -max=bytes
56
+ The shortest and longest record lengths to check. By default, 1 and
57
+ half the data (see -minrecs).
58
+ -minrecs=N
59
+ The file holds at least N records (default 2): the same as
60
+ -max=(size-skip)/N. -max takes precedence.
61
+ -fact=bytes (also -factor, -mult)
62
+ The record length is a multiple of this, e.g. 8 for doubles.
63
+ -only=list
64
+ Check only these lengths (comma-separated).
65
+ -partial
66
+ Don't assume whole records: check every length in range.
67
+ -maxbufs=N
68
+ Use only the first N buffers (of twice the longest length checked).
69
+ -limit=bytes
70
+ Use only the first this many bytes of data.
71
+ -reduce=factor
72
+ Use only the first 1/factor of the data (factor >= 1).
73
+ -full
74
+ Compare the data (or the part -limit or -reduce keep) as one piece,
75
+ rather than buffer by buffer.
76
+ -verbose
77
+ Show the score of every length in every buffer, and a sorted table.
78
+ -quiet
79
+ Show only the result.
80
+ -help
81
+ Show this help.
82
+
83
+ The full manual is at
84
+ https://github.com/donalgrant/telemetry-cli/blob/main/docs/recl.md
85
+ """
86
+
87
+
88
+ class ReclError(ValueError):
89
+ pass
90
+
91
+
92
+ def candidate_lengths(databytes: int, lo: int, hi: int, fact: int, partial: bool) -> list[int]:
93
+ """Lengths in [lo, hi] that are multiples of fact (and divide databytes, unless partial)."""
94
+ return [
95
+ r for r in range(max(lo, 1), hi + 1) if r % fact == 0 and (partial or databytes % r == 0)
96
+ ]
97
+
98
+
99
+ def main(
100
+ argv: list[str] | None = None,
101
+ *,
102
+ stdin: BinaryIO | None = None,
103
+ stdout: BinaryIO | None = None,
104
+ stderr: TextIO | None = None,
105
+ ) -> int:
106
+ argv = sys.argv[1:] if argv is None else list(argv)
107
+ stdin = stdin if stdin is not None else sys.stdin.buffer
108
+ stdout = stdout if stdout is not None else sys.stdout.buffer
109
+ stderr = stderr if stderr is not None else sys.stderr
110
+
111
+ def error(message: str, status: int = 1) -> int:
112
+ print(f"recl: {message}", file=stderr)
113
+ return status
114
+
115
+ try:
116
+ o, args = getoptions(argv, SPECS)
117
+ except OptionError as e:
118
+ return error(f"{e} (see 'recl -help')", 2)
119
+ if o.get("h"):
120
+ stdout.write(HELP.encode())
121
+ return 0
122
+ if not args:
123
+ return error("no input file given (use - for stdin); see 'recl -help'", 2)
124
+ try:
125
+ return _run(args[0], o, stdin, stdout)
126
+ except ReclError as e:
127
+ return error(str(e))
128
+ except OSError as e:
129
+ return error(f"Can't open {args[0]} for input: {e.strerror}")
130
+ except BrokenPipeError:
131
+ return broken_pipe(stdout)
132
+
133
+
134
+ def _run(file: str, o: dict, stdin: BinaryIO, stdout: BinaryIO) -> int:
135
+ lines: list[str] = []
136
+
137
+ class _Out:
138
+ def write(self, s):
139
+ lines.append(s)
140
+
141
+ msg = Messenger(_Out())
142
+ if file == "-":
143
+ data = stdin.read()
144
+ else:
145
+ with open(file, "rb") as f:
146
+ data = f.read()
147
+
148
+ skip = o.get("skip", 0)
149
+ reduce = o.get("reduce", 1.0)
150
+ if reduce < 1:
151
+ raise ReclError("-reduce must be at least 1")
152
+ fact = o.get("fact", 1)
153
+ if fact < 1:
154
+ raise ReclError("-fact must be at least 1")
155
+ databytes = len(data) - skip
156
+ if databytes < 2:
157
+ raise ReclError(f"too little data: {max(databytes, 0)} bytes after the header")
158
+ hi = o["max"] if "max" in o else databytes // o.get("minrecs", 2)
159
+
160
+ if "only" in o:
161
+ try:
162
+ lengths = [int(x) for x in o["only"].split(",") if x.strip()]
163
+ except ValueError:
164
+ raise ReclError(
165
+ f"-only needs a comma-separated list of lengths: {o['only']!r}"
166
+ ) from None
167
+ else:
168
+ lengths = candidate_lengths(databytes, o.get("min", 1), hi, fact, bool(o.get("part")))
169
+ msg("Checking Record Lengths " + ", ".join(map(str, lengths)) + " bytes")
170
+ _flush(lines, stdout)
171
+ if not lengths:
172
+ raise ReclError("no record lengths to check; see the -min, -max and -fact options")
173
+ if min(lengths) < 1:
174
+ raise ReclError("record lengths must be at least 1 byte")
175
+
176
+ use = np.frombuffer(data, dtype=np.uint8)[skip:]
177
+ if "limit" in o:
178
+ use = use[: max(o["limit"], 0)]
179
+ use = use[: int(len(use) / reduce)]
180
+
181
+ bufsiz = 2 * max(lengths)
182
+ if o.get("f") or len(use) < bufsiz:
183
+ buffers = use[np.newaxis, :] # one piece
184
+ else:
185
+ nbuf = len(use) // bufsiz
186
+ if "maxbufs" in o:
187
+ nbuf = min(nbuf, max(o["maxbufs"], 1))
188
+ buffers = use[: nbuf * bufsiz].reshape(nbuf, bufsiz)
189
+
190
+ scores = agreement(buffers, lengths) # (buffers, lengths)
191
+ if o.get("v"):
192
+ running = np.cumsum(scores, axis=0)
193
+ for i in range(scores.shape[0]):
194
+ for j, r in enumerate(lengths):
195
+ msg(f"buf {i + 1} reclen {r} corr {perl_str(float(scores[i, j]))} "
196
+ f"accum-corr {perl_str(float(running[i, j]))}") # fmt: skip
197
+ mean = scores.mean(axis=0).tolist()
198
+ chosen = choose(lengths, mean, compared(buffers, lengths).tolist())
199
+ if o.get("v"):
200
+ msg("Table of Sorted Correlations")
201
+ for j in sorted(range(len(lengths)), key=lambda j: (-mean[j], lengths[j])):
202
+ msg(f"{100 * mean[j]:5.2f}% for reclen {lengths[j]}")
203
+ msg(lengths[chosen], "RESULT")
204
+ if not o.get("q"):
205
+ msg(perl_str(100.0 * mean[chosen]), "CORR")
206
+ msg(f"Based on {buffers.shape[0]} buffers, each of length {buffers.shape[1]}")
207
+ _flush(lines, stdout)
208
+ stdout.flush()
209
+ return 0
210
+
211
+
212
+ def _flush(lines: list[str], stdout: BinaryIO) -> None:
213
+ stdout.write("".join(lines).encode())
214
+ lines.clear()
@@ -0,0 +1,83 @@
1
+ """Scoring candidate record lengths.
2
+
3
+ For a record length R, the score of a buffer is the fraction of its bytes
4
+ that equal the byte R bytes further on. For unrelated random data that is
5
+ about 1/256; it is higher when records are R bytes long, since records of
6
+ the same format tend to have the same bytes in the same places (sync words,
7
+ flags, the high bytes of counters and slowly varying values). Scores are
8
+ averaged over the buffers.
9
+
10
+ (Comparing whole bytes works better than counting agreeing bits, which is
11
+ what the Perl original set out to do: a counter's low bits agree more often
12
+ at multiples of the record length than at the length itself, which pulls a
13
+ bit count toward multiples, and bits agree often at small shifts in text.)
14
+
15
+ A length R is scored only in buffers of at least 2R bytes (as in the Perl);
16
+ elsewhere its score is 0.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import numpy as np
22
+
23
+
24
+ def agreement(buffers: np.ndarray, lengths: list[int]) -> np.ndarray:
25
+ """Scores, shape (number of buffers, number of lengths)."""
26
+ nbuf, size = buffers.shape
27
+ out = np.zeros((nbuf, len(lengths)))
28
+ for j, r in enumerate(lengths):
29
+ if r < 1 or size < 2 * r:
30
+ continue
31
+ out[:, j] = (buffers[:, : size - r] == buffers[:, r:]).mean(axis=1)
32
+ return out
33
+
34
+
35
+ def compared(buffers: np.ndarray, lengths: list[int]) -> np.ndarray:
36
+ """How many byte pairs were compared for each length, over all buffers."""
37
+ nbuf, size = buffers.shape
38
+ r = np.array(lengths)
39
+ return np.where(size >= 2 * r, 1.0 * (size - r) * nbuf, 0.0)
40
+
41
+
42
+ def choose(
43
+ lengths: list[int],
44
+ scores: list[float],
45
+ counts: list[float] | None = None,
46
+ tolerance: float = 0.25,
47
+ ) -> int:
48
+ """The record length the scores point to.
49
+
50
+ Multiples of the record length score about as well as the length itself,
51
+ since the data repeat at those too; the longer ones only less precisely,
52
+ as fewer bytes overlap. So the answer is the smallest length that divides
53
+ the best-scoring one and scores nearly as well: within three standard
54
+ errors of it, or within ``tolerance`` of its height above a typical
55
+ score, whichever is more. It must also score at least half that height
56
+ above a typical score, so that a length with no structure of its own (1,
57
+ say) isn't chosen when the signal is weak.
58
+
59
+ A typical score is the lower quartile: with short records, most of the
60
+ candidates may be multiples of the record length, so the median may not
61
+ be typical. With fewer than five candidates, the lowest score stands in,
62
+ and the structure check is skipped.
63
+
64
+ ``counts`` are the numbers of byte pairs compared, for the standard errors.
65
+ """
66
+ best = max(range(len(lengths)), key=lambda j: (scores[j], -lengths[j]))
67
+ few = len(lengths) < 5
68
+ baseline = float(min(scores) if few else np.percentile(scores, 25))
69
+ height = scores[best] - baseline
70
+ p = min(max(baseline, 1 / 256), 1 - 1 / 256)
71
+
72
+ def stderr(j): # of a fraction near the baseline
73
+ return float(np.sqrt(p * (1 - p) / counts[j])) if counts and counts[j] > 0 else 0.0
74
+
75
+ for j in sorted(range(len(lengths)), key=lambda j: lengths[j]):
76
+ r = lengths[j]
77
+ if not r or lengths[best] % r:
78
+ continue
79
+ margin = max(tolerance * height, 3 * np.hypot(stderr(j), stderr(best)))
80
+ # nearly as good as the best, and clearly better than a typical length
81
+ if scores[j] >= scores[best] - margin and (few or scores[j] - baseline >= height / 2):
82
+ return j
83
+ return best
@@ -0,0 +1 @@
1
+ """recs: extract records from a stream of data by their start and end markers."""
@@ -0,0 +1,7 @@
1
+ """Allow ``python -m telemetry_cli.recs``."""
2
+
3
+ import sys
4
+
5
+ from .cli import main
6
+
7
+ sys.exit(main())