telemetry-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- telemetry_cli/__init__.py +8 -0
- telemetry_cli/_cli.py +18 -0
- telemetry_cli/_getopt.py +114 -0
- telemetry_cli/_msg.py +39 -0
- telemetry_cli/_pack.py +183 -0
- telemetry_cli/_perl.py +183 -0
- telemetry_cli/pick/__init__.py +1 -0
- telemetry_cli/pick/__main__.py +7 -0
- telemetry_cli/pick/cli.py +157 -0
- telemetry_cli/pick/commandline.py +219 -0
- telemetry_cli/pick/fields.py +166 -0
- telemetry_cli/pick/formats.py +191 -0
- telemetry_cli/pick/help.py +536 -0
- telemetry_cli/pick/symbols/airmoc +24 -0
- telemetry_cli/pick/symbols/aux +28 -0
- telemetry_cli/pick/symbols/corr +31 -0
- telemetry_cli/pick/symbols/corrOld +29 -0
- telemetry_cli/pick/symbols/hw +16 -0
- telemetry_cli/pick/symbols/moc +42 -0
- telemetry_cli/pick/symbols/subcom +83 -0
- telemetry_cli/pick/symbols/subcom_2002 +111 -0
- telemetry_cli/recl/__init__.py +1 -0
- telemetry_cli/recl/__main__.py +7 -0
- telemetry_cli/recl/cli.py +214 -0
- telemetry_cli/recl/corr.py +83 -0
- telemetry_cli/recs/__init__.py +1 -0
- telemetry_cli/recs/__main__.py +7 -0
- telemetry_cli/recs/cli.py +182 -0
- telemetry_cli/recs/extract.py +399 -0
- telemetry_cli/tgen/__init__.py +1 -0
- telemetry_cli/tgen/__main__.py +7 -0
- telemetry_cli/tgen/cli.py +128 -0
- telemetry_cli/tgen/cmdfile.py +68 -0
- telemetry_cli/tgen/convert.py +278 -0
- telemetry_cli/tgen/examples/arb_data.tgen +15 -0
- telemetry_cli/tgen/examples/calsweep.tgen +32 -0
- telemetry_cli/tgen/examples/cond_fill.tgen +17 -0
- telemetry_cli/tgen/examples/counter.tgen +11 -0
- telemetry_cli/tgen/examples/null_data.tgen +11 -0
- telemetry_cli/tgen/examples/random_size.tgen +20 -0
- telemetry_cli/tgen/examples/subcom.tgen +14 -0
- telemetry_cli/tgen/examples/valueV_value.tgen +20 -0
- telemetry_cli/tgen/runtime.py +390 -0
- telemetry_cli-1.0.0.dist-info/METADATA +106 -0
- telemetry_cli-1.0.0.dist-info/RECORD +48 -0
- telemetry_cli-1.0.0.dist-info/WHEEL +4 -0
- telemetry_cli-1.0.0.dist-info/entry_points.txt +5 -0
- telemetry_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
; This is the symbol table for .moc files, which
|
|
2
|
+
; are the radar motion compensation files for section 334.
|
|
3
|
+
|
|
4
|
+
length=21d ; length is 21 double-precision (8 byte) numbers
|
|
5
|
+
|
|
6
|
+
; these are the components of the file
|
|
7
|
+
|
|
8
|
+
xRec = d0G
|
|
9
|
+
xrec = d0G
|
|
10
|
+
time = d1G
|
|
11
|
+
utc = d1G
|
|
12
|
+
prf = d2 ; actually 1/prf...
|
|
13
|
+
tprf = d2
|
|
14
|
+
posS = d3
|
|
15
|
+
posC = d4
|
|
16
|
+
posH = d5
|
|
17
|
+
pos = posS posC posH
|
|
18
|
+
velS = d6
|
|
19
|
+
velC = d7
|
|
20
|
+
velH = d8
|
|
21
|
+
vel = velS velC velH
|
|
22
|
+
af1S = d9
|
|
23
|
+
af1C = d10
|
|
24
|
+
af1H = d11
|
|
25
|
+
af1 = 3d9
|
|
26
|
+
af2 = 3d12
|
|
27
|
+
af3 = 3d15
|
|
28
|
+
az = d18
|
|
29
|
+
yaw = d18
|
|
30
|
+
pitch = d19
|
|
31
|
+
roll = d20
|
|
32
|
+
orient = 3d18
|
|
33
|
+
azD = d18D
|
|
34
|
+
yawD = d18D
|
|
35
|
+
pitchd = d19D
|
|
36
|
+
rollD = d20D
|
|
37
|
+
orientD = 3d18D
|
|
38
|
+
all = 21d0
|
|
39
|
+
af = af1 af2 af3
|
|
40
|
+
angles = posS orientD
|
|
41
|
+
velocities = posS vel
|
|
42
|
+
tube = pos
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
;
|
|
2
|
+
; pick symbol table for AIRSAR subcommutated data files
|
|
3
|
+
;
|
|
4
|
+
length=4096
|
|
5
|
+
;
|
|
6
|
+
sub_len = w0+0u
|
|
7
|
+
used_len = w0+2u
|
|
8
|
+
;
|
|
9
|
+
; data collection notes
|
|
10
|
+
;
|
|
11
|
+
target = 25A0+439
|
|
12
|
+
invstg = 25A0+464
|
|
13
|
+
hp_opr = 25A0+489
|
|
14
|
+
crew = 25A0+514
|
|
15
|
+
mode = 25A0+539
|
|
16
|
+
status = 25A0+564
|
|
17
|
+
antenna = 25A0+589
|
|
18
|
+
telev = 8A0+638 ; target elevation in feet
|
|
19
|
+
dser1 = 10A0+683 ; digital tape serial number HDDR #1
|
|
20
|
+
dser2 = 10A0+683 ; digital tape serial number HDDR #2
|
|
21
|
+
dser3 = 10A0+683 ; digital tape serial number HDDR #3
|
|
22
|
+
notes = 256A0+1131 ; operator's notes
|
|
23
|
+
;
|
|
24
|
+
; data in six-gun position
|
|
25
|
+
;
|
|
26
|
+
six_month = b0+1412u ; month (1..12)
|
|
27
|
+
six_day = b0+1413u ; day (1..31)
|
|
28
|
+
six_year = w0+1414u ; year (1900..2999)
|
|
29
|
+
six_hour = b0+1416u ; hour (0..23)
|
|
30
|
+
six_min = b0+1417u ; min (0..59)
|
|
31
|
+
six_sec = b0+1418u ; sec (0..60)
|
|
32
|
+
six_nsec = u0+1419u ; fractional seconds (0..999,999,999) [nano-seconds]
|
|
33
|
+
;
|
|
34
|
+
six_f_cnt = u0+1476u ; 32-bit frame count at GPS time tag
|
|
35
|
+
;
|
|
36
|
+
; True-Time data
|
|
37
|
+
;
|
|
38
|
+
TTmonth = b0+2188u ; month (1..12)
|
|
39
|
+
TTday = b0+2189u ; day (1..31)
|
|
40
|
+
TTyear = w0+2190u ; year (1900..2999)
|
|
41
|
+
TThour = b0+2192u ; hour (0..23)
|
|
42
|
+
TTmin = b0+2193u ; min (0..59)
|
|
43
|
+
TTsec = b0+2194u ; sec (0..60)
|
|
44
|
+
TTnsec = u0+2195u ; fractional seconds (0..999,999,999) [nano-seconds]
|
|
45
|
+
;
|
|
46
|
+
TTf_cnt = u0+2212u ; 32-bit frame count at True-Time time tag
|
|
47
|
+
;
|
|
48
|
+
; EO-17 message
|
|
49
|
+
;
|
|
50
|
+
egi17TimeTag = w0+2496u ; 1553 time tag
|
|
51
|
+
;
|
|
52
|
+
; EO-18 message
|
|
53
|
+
;
|
|
54
|
+
egi18GpsTimeHi = w0+2560u ; gps time tag word
|
|
55
|
+
egi18GpsTime2 = w0+2562u ; gps time tag word
|
|
56
|
+
egi18GpsTime3 = w0+2564u ; gps time tag word
|
|
57
|
+
egi18GpsTimeLo = w0+2566u ; gps time tag word
|
|
58
|
+
|
|
59
|
+
egi18GpsTimeHigh = u0+2560u ; gps time tag word
|
|
60
|
+
egi18GpsTimeLow = u0+2564u ; gps time tag word
|
|
61
|
+
|
|
62
|
+
egi18UtcTimeHi = w0+2568u ; utc time tag word
|
|
63
|
+
egi18UtcTime2 = w0+2570u ; utc time tag word
|
|
64
|
+
egi18UtcTime3 = w0+2572u ; utc time tag word
|
|
65
|
+
egi18UtcTimeLo = w0+2574u ; utc time tag word
|
|
66
|
+
|
|
67
|
+
egi18UtcTimeHigh = u0+2568u ; utc time tag word
|
|
68
|
+
egi18UtcTimeLow = u0+2572u ; utc time tag word
|
|
69
|
+
;
|
|
70
|
+
; EO-19 message
|
|
71
|
+
;
|
|
72
|
+
egi19TimeTag = w0+2622u ; Time tag
|
|
73
|
+
egi19PltfrmAzTimeTag = w0+2654u ; Platform azimuth time tag
|
|
74
|
+
egi19RollTimeTag = w0+2656u ; Roll time tag
|
|
75
|
+
egi19PitchTimeTag = w0+2658u ; Pitch time tag
|
|
76
|
+
;
|
|
77
|
+
; end of record stuff
|
|
78
|
+
;
|
|
79
|
+
motionSensorFlag = b0+3004d ; Bit 0 set = IGI, bit 1 set = EGI
|
|
80
|
+
hdrVersion = 6A0+3005 ; Date of this revision, eg. 980817
|
|
81
|
+
eosd = 4A0+3011 ; Indicates end of used subheader
|
|
82
|
+
|
|
83
|
+
|
|
@@ -0,0 +1,111 @@
|
|
|
1
|
+
;
|
|
2
|
+
; pick symbol table for AIRSAR subcommutated data files
|
|
3
|
+
;
|
|
4
|
+
length=4096
|
|
5
|
+
;
|
|
6
|
+
sub_len = w0+0u
|
|
7
|
+
used_len = w0+2u
|
|
8
|
+
|
|
9
|
+
_ = -1 bs
|
|
10
|
+
|
|
11
|
+
;
|
|
12
|
+
; data collection notes
|
|
13
|
+
;
|
|
14
|
+
target = 25A0+439
|
|
15
|
+
invstg = 25A0+464
|
|
16
|
+
hp_opr = 25A0+489
|
|
17
|
+
crew = 25A0+514
|
|
18
|
+
mode = 25A0+539
|
|
19
|
+
status = 25A0+564
|
|
20
|
+
antenna = 25A0+589
|
|
21
|
+
telev = 8A0+638 ; target elevation in feet
|
|
22
|
+
dser1 = 10A0+683 ; digital tape serial number HDDR #1
|
|
23
|
+
dser2 = 10A0+683 ; digital tape serial number HDDR #2
|
|
24
|
+
dser3 = 10A0+683 ; digital tape serial number HDDR #3
|
|
25
|
+
notes = 256A0+1131 ; operator's notes
|
|
26
|
+
;
|
|
27
|
+
; data in six-gun position
|
|
28
|
+
;
|
|
29
|
+
six_month = b0+1412u ; month (1..12)
|
|
30
|
+
six_day = b0+1413u ; day (1..31)
|
|
31
|
+
six_year = w0+1414u ; year (1900..2999)
|
|
32
|
+
six_hour = b0+1416u ; hour (0..23)
|
|
33
|
+
six_min = b0+1417u ; min (0..59)
|
|
34
|
+
six_sec = b0+1418u ; sec (0..60)
|
|
35
|
+
six_nsec = u0+1419u ; fractional seconds (0..999,999,999) [nano-seconds]
|
|
36
|
+
;
|
|
37
|
+
six_f_cnt = u0+1476u ; 32-bit frame count at GPS time tag
|
|
38
|
+
;
|
|
39
|
+
; True-Time data
|
|
40
|
+
;
|
|
41
|
+
TTmonth = b0+2188u ; month (1..12)
|
|
42
|
+
TTday = b0+2189u ; day (1..31)
|
|
43
|
+
TTyear = w0+2190u ; year (1900..2999)
|
|
44
|
+
TThour = b0+2192u ; hour (0..23)
|
|
45
|
+
TTmin = b0+2193u ; min (0..59)
|
|
46
|
+
TTsec = b0+2194u ; sec (0..60)
|
|
47
|
+
TTnsec = u0+2195u ; fractional seconds (0..999,999,999) [nano-seconds]
|
|
48
|
+
;
|
|
49
|
+
TTf_cnt = u0+2212u ; 32-bit frame count at True-Time time tag
|
|
50
|
+
;
|
|
51
|
+
; EO-17 message
|
|
52
|
+
;
|
|
53
|
+
egi17TimeTag = w0+2496u ; 1553 time tag
|
|
54
|
+
;
|
|
55
|
+
; EO-18 message
|
|
56
|
+
;
|
|
57
|
+
egi18GpsTimeHi = w0+2560u ; gps time tag word
|
|
58
|
+
egi18GpsTime2 = w0+2562u ; gps time tag word
|
|
59
|
+
egi18GpsTime3 = w0+2564u ; gps time tag word
|
|
60
|
+
egi18GpsTimeLo = w0+2566u ; gps time tag word
|
|
61
|
+
|
|
62
|
+
egi18GpsTimeHigh = u0+2560u ; gps time tag word
|
|
63
|
+
egi18GpsTimeLow = u0+2564u ; gps time tag word
|
|
64
|
+
|
|
65
|
+
egi18UtcTimeHi = w0+2568u ; utc time tag word
|
|
66
|
+
egi18UtcTime2 = w0+2570u ; utc time tag word
|
|
67
|
+
egi18UtcTime3 = w0+2572u ; utc time tag word
|
|
68
|
+
egi18UtcTimeLo = w0+2574u ; utc time tag word
|
|
69
|
+
|
|
70
|
+
egi18UtcTimeHigh = u0+2568u ; utc time tag word
|
|
71
|
+
egi18UtcTimeLow = u0+2572u ; utc time tag word
|
|
72
|
+
;
|
|
73
|
+
; EO-19 message
|
|
74
|
+
;
|
|
75
|
+
egi19TimeTag = w0+2622u ; Time tag
|
|
76
|
+
egi19PltfrmAzTimeTag = w0+2654u ; Platform azimuth time tag
|
|
77
|
+
egi19RollTimeTag = w0+2656u ; Roll time tag
|
|
78
|
+
egi19PitchTimeTag = w0+2658u ; Pitch time tag
|
|
79
|
+
;
|
|
80
|
+
; end of record stuff
|
|
81
|
+
;
|
|
82
|
+
motionSensorFlag = b0+3004d ; Bit 0 set = IGI, bit 1 set = EGI
|
|
83
|
+
hdrVersion = 6A0+3005 ; Date of this revision, eg. 980817
|
|
84
|
+
|
|
85
|
+
dcg_bw_l = 4A0+3011
|
|
86
|
+
dcg_bw_c = 4A0+3015
|
|
87
|
+
dcg_bw_p = 4A0+3019
|
|
88
|
+
|
|
89
|
+
dcg_pd_l = 3A0+3023
|
|
90
|
+
dcg_pd_c = 3A0+3026
|
|
91
|
+
dcg_pd_p = 3A0+3029
|
|
92
|
+
|
|
93
|
+
sampfreq_l = 4A0+3032
|
|
94
|
+
sampfreq_c = 4A0+3036
|
|
95
|
+
sampfreq_p = 4A0+3040
|
|
96
|
+
|
|
97
|
+
dgps_time_J2000 = d0+3428G
|
|
98
|
+
|
|
99
|
+
dgps_x = d0+3444G
|
|
100
|
+
dgps_y = d0+3452G
|
|
101
|
+
dgps_z = d0+3460G
|
|
102
|
+
|
|
103
|
+
dgps_sigma_x = d0+3468
|
|
104
|
+
dgps_sigma_y = d0+3476
|
|
105
|
+
dgps_sigma_z = d0+3484
|
|
106
|
+
|
|
107
|
+
cal_mode = b0+3492d
|
|
108
|
+
|
|
109
|
+
eosd = 4A0+3493 ; Indicates end of used subheader
|
|
110
|
+
|
|
111
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""recl: estimate the record length of a binary file."""
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""The recl command: estimate the record length of a binary file.
|
|
2
|
+
|
|
3
|
+
recl filename [ options ]
|
|
4
|
+
cat filename | recl - [ options ]
|
|
5
|
+
|
|
6
|
+
Run ``recl -help`` for help.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import sys
|
|
12
|
+
from typing import BinaryIO, TextIO
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
from .._cli import broken_pipe
|
|
17
|
+
from .._getopt import OptionError, getoptions
|
|
18
|
+
from .._msg import Messenger
|
|
19
|
+
from .._perl import perl_str
|
|
20
|
+
from .corr import agreement, choose, compared
|
|
21
|
+
|
|
22
|
+
SPECS = ["v|verbose", "h|help", "f|full", "q|quiet", "part|partial", "skip|header=i", "min=i",
|
|
23
|
+
"max=i", "minrecs=i", "maxbufs=i", "fact|factor|mult=i", "only=s", "limit=i",
|
|
24
|
+
"reduce=f"] # fmt: skip
|
|
25
|
+
|
|
26
|
+
HELP = """\
|
|
27
|
+
recl - estimate the record length of a binary file
|
|
28
|
+
|
|
29
|
+
SYNOPSIS
|
|
30
|
+
recl filename [ options ]
|
|
31
|
+
cat filename | recl - [ options ]
|
|
32
|
+
|
|
33
|
+
DESCRIPTION
|
|
34
|
+
recl compares the file with itself shifted by each possible record
|
|
35
|
+
length R, and reports the R at which the most bytes are equal: records
|
|
36
|
+
of the same format tend to have the same bytes in the same places. It
|
|
37
|
+
prints
|
|
38
|
+
|
|
39
|
+
RESULT R the estimated record length
|
|
40
|
+
CORR x the percentage of bytes equal to the byte R bytes on
|
|
41
|
+
(about 0.4 for unrelated random data)
|
|
42
|
+
|
|
43
|
+
and notes about the lengths it checked. Every multiple of the record
|
|
44
|
+
length agrees about as well as the length itself; recl reports the
|
|
45
|
+
smallest.
|
|
46
|
+
|
|
47
|
+
Unless -partial is given, recl assumes the file (after any header) holds
|
|
48
|
+
whole records, so it only checks lengths that divide its size.
|
|
49
|
+
|
|
50
|
+
OPTIONS
|
|
51
|
+
Options can be abbreviated.
|
|
52
|
+
|
|
53
|
+
-skip=bytes (or -header)
|
|
54
|
+
Skip a header of this many bytes.
|
|
55
|
+
-min=bytes, -max=bytes
|
|
56
|
+
The shortest and longest record lengths to check. By default, 1 and
|
|
57
|
+
half the data (see -minrecs).
|
|
58
|
+
-minrecs=N
|
|
59
|
+
The file holds at least N records (default 2): the same as
|
|
60
|
+
-max=(size-skip)/N. -max takes precedence.
|
|
61
|
+
-fact=bytes (also -factor, -mult)
|
|
62
|
+
The record length is a multiple of this, e.g. 8 for doubles.
|
|
63
|
+
-only=list
|
|
64
|
+
Check only these lengths (comma-separated).
|
|
65
|
+
-partial
|
|
66
|
+
Don't assume whole records: check every length in range.
|
|
67
|
+
-maxbufs=N
|
|
68
|
+
Use only the first N buffers (of twice the longest length checked).
|
|
69
|
+
-limit=bytes
|
|
70
|
+
Use only the first this many bytes of data.
|
|
71
|
+
-reduce=factor
|
|
72
|
+
Use only the first 1/factor of the data (factor >= 1).
|
|
73
|
+
-full
|
|
74
|
+
Compare the data (or the part -limit or -reduce keep) as one piece,
|
|
75
|
+
rather than buffer by buffer.
|
|
76
|
+
-verbose
|
|
77
|
+
Show the score of every length in every buffer, and a sorted table.
|
|
78
|
+
-quiet
|
|
79
|
+
Show only the result.
|
|
80
|
+
-help
|
|
81
|
+
Show this help.
|
|
82
|
+
|
|
83
|
+
The full manual is at
|
|
84
|
+
https://github.com/donalgrant/telemetry-cli/blob/main/docs/recl.md
|
|
85
|
+
"""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
class ReclError(ValueError):
|
|
89
|
+
pass
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def candidate_lengths(databytes: int, lo: int, hi: int, fact: int, partial: bool) -> list[int]:
|
|
93
|
+
"""Lengths in [lo, hi] that are multiples of fact (and divide databytes, unless partial)."""
|
|
94
|
+
return [
|
|
95
|
+
r for r in range(max(lo, 1), hi + 1) if r % fact == 0 and (partial or databytes % r == 0)
|
|
96
|
+
]
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def main(
|
|
100
|
+
argv: list[str] | None = None,
|
|
101
|
+
*,
|
|
102
|
+
stdin: BinaryIO | None = None,
|
|
103
|
+
stdout: BinaryIO | None = None,
|
|
104
|
+
stderr: TextIO | None = None,
|
|
105
|
+
) -> int:
|
|
106
|
+
argv = sys.argv[1:] if argv is None else list(argv)
|
|
107
|
+
stdin = stdin if stdin is not None else sys.stdin.buffer
|
|
108
|
+
stdout = stdout if stdout is not None else sys.stdout.buffer
|
|
109
|
+
stderr = stderr if stderr is not None else sys.stderr
|
|
110
|
+
|
|
111
|
+
def error(message: str, status: int = 1) -> int:
|
|
112
|
+
print(f"recl: {message}", file=stderr)
|
|
113
|
+
return status
|
|
114
|
+
|
|
115
|
+
try:
|
|
116
|
+
o, args = getoptions(argv, SPECS)
|
|
117
|
+
except OptionError as e:
|
|
118
|
+
return error(f"{e} (see 'recl -help')", 2)
|
|
119
|
+
if o.get("h"):
|
|
120
|
+
stdout.write(HELP.encode())
|
|
121
|
+
return 0
|
|
122
|
+
if not args:
|
|
123
|
+
return error("no input file given (use - for stdin); see 'recl -help'", 2)
|
|
124
|
+
try:
|
|
125
|
+
return _run(args[0], o, stdin, stdout)
|
|
126
|
+
except ReclError as e:
|
|
127
|
+
return error(str(e))
|
|
128
|
+
except OSError as e:
|
|
129
|
+
return error(f"Can't open {args[0]} for input: {e.strerror}")
|
|
130
|
+
except BrokenPipeError:
|
|
131
|
+
return broken_pipe(stdout)
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _run(file: str, o: dict, stdin: BinaryIO, stdout: BinaryIO) -> int:
|
|
135
|
+
lines: list[str] = []
|
|
136
|
+
|
|
137
|
+
class _Out:
|
|
138
|
+
def write(self, s):
|
|
139
|
+
lines.append(s)
|
|
140
|
+
|
|
141
|
+
msg = Messenger(_Out())
|
|
142
|
+
if file == "-":
|
|
143
|
+
data = stdin.read()
|
|
144
|
+
else:
|
|
145
|
+
with open(file, "rb") as f:
|
|
146
|
+
data = f.read()
|
|
147
|
+
|
|
148
|
+
skip = o.get("skip", 0)
|
|
149
|
+
reduce = o.get("reduce", 1.0)
|
|
150
|
+
if reduce < 1:
|
|
151
|
+
raise ReclError("-reduce must be at least 1")
|
|
152
|
+
fact = o.get("fact", 1)
|
|
153
|
+
if fact < 1:
|
|
154
|
+
raise ReclError("-fact must be at least 1")
|
|
155
|
+
databytes = len(data) - skip
|
|
156
|
+
if databytes < 2:
|
|
157
|
+
raise ReclError(f"too little data: {max(databytes, 0)} bytes after the header")
|
|
158
|
+
hi = o["max"] if "max" in o else databytes // o.get("minrecs", 2)
|
|
159
|
+
|
|
160
|
+
if "only" in o:
|
|
161
|
+
try:
|
|
162
|
+
lengths = [int(x) for x in o["only"].split(",") if x.strip()]
|
|
163
|
+
except ValueError:
|
|
164
|
+
raise ReclError(
|
|
165
|
+
f"-only needs a comma-separated list of lengths: {o['only']!r}"
|
|
166
|
+
) from None
|
|
167
|
+
else:
|
|
168
|
+
lengths = candidate_lengths(databytes, o.get("min", 1), hi, fact, bool(o.get("part")))
|
|
169
|
+
msg("Checking Record Lengths " + ", ".join(map(str, lengths)) + " bytes")
|
|
170
|
+
_flush(lines, stdout)
|
|
171
|
+
if not lengths:
|
|
172
|
+
raise ReclError("no record lengths to check; see the -min, -max and -fact options")
|
|
173
|
+
if min(lengths) < 1:
|
|
174
|
+
raise ReclError("record lengths must be at least 1 byte")
|
|
175
|
+
|
|
176
|
+
use = np.frombuffer(data, dtype=np.uint8)[skip:]
|
|
177
|
+
if "limit" in o:
|
|
178
|
+
use = use[: max(o["limit"], 0)]
|
|
179
|
+
use = use[: int(len(use) / reduce)]
|
|
180
|
+
|
|
181
|
+
bufsiz = 2 * max(lengths)
|
|
182
|
+
if o.get("f") or len(use) < bufsiz:
|
|
183
|
+
buffers = use[np.newaxis, :] # one piece
|
|
184
|
+
else:
|
|
185
|
+
nbuf = len(use) // bufsiz
|
|
186
|
+
if "maxbufs" in o:
|
|
187
|
+
nbuf = min(nbuf, max(o["maxbufs"], 1))
|
|
188
|
+
buffers = use[: nbuf * bufsiz].reshape(nbuf, bufsiz)
|
|
189
|
+
|
|
190
|
+
scores = agreement(buffers, lengths) # (buffers, lengths)
|
|
191
|
+
if o.get("v"):
|
|
192
|
+
running = np.cumsum(scores, axis=0)
|
|
193
|
+
for i in range(scores.shape[0]):
|
|
194
|
+
for j, r in enumerate(lengths):
|
|
195
|
+
msg(f"buf {i + 1} reclen {r} corr {perl_str(float(scores[i, j]))} "
|
|
196
|
+
f"accum-corr {perl_str(float(running[i, j]))}") # fmt: skip
|
|
197
|
+
mean = scores.mean(axis=0).tolist()
|
|
198
|
+
chosen = choose(lengths, mean, compared(buffers, lengths).tolist())
|
|
199
|
+
if o.get("v"):
|
|
200
|
+
msg("Table of Sorted Correlations")
|
|
201
|
+
for j in sorted(range(len(lengths)), key=lambda j: (-mean[j], lengths[j])):
|
|
202
|
+
msg(f"{100 * mean[j]:5.2f}% for reclen {lengths[j]}")
|
|
203
|
+
msg(lengths[chosen], "RESULT")
|
|
204
|
+
if not o.get("q"):
|
|
205
|
+
msg(perl_str(100.0 * mean[chosen]), "CORR")
|
|
206
|
+
msg(f"Based on {buffers.shape[0]} buffers, each of length {buffers.shape[1]}")
|
|
207
|
+
_flush(lines, stdout)
|
|
208
|
+
stdout.flush()
|
|
209
|
+
return 0
|
|
210
|
+
|
|
211
|
+
|
|
212
|
+
def _flush(lines: list[str], stdout: BinaryIO) -> None:
|
|
213
|
+
stdout.write("".join(lines).encode())
|
|
214
|
+
lines.clear()
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
"""Scoring candidate record lengths.
|
|
2
|
+
|
|
3
|
+
For a record length R, the score of a buffer is the fraction of its bytes
|
|
4
|
+
that equal the byte R bytes further on. For unrelated random data that is
|
|
5
|
+
about 1/256; it is higher when records are R bytes long, since records of
|
|
6
|
+
the same format tend to have the same bytes in the same places (sync words,
|
|
7
|
+
flags, the high bytes of counters and slowly varying values). Scores are
|
|
8
|
+
averaged over the buffers.
|
|
9
|
+
|
|
10
|
+
(Comparing whole bytes works better than counting agreeing bits, which is
|
|
11
|
+
what the Perl original set out to do: a counter's low bits agree more often
|
|
12
|
+
at multiples of the record length than at the length itself, which pulls a
|
|
13
|
+
bit count toward multiples, and bits agree often at small shifts in text.)
|
|
14
|
+
|
|
15
|
+
A length R is scored only in buffers of at least 2R bytes (as in the Perl);
|
|
16
|
+
elsewhere its score is 0.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import numpy as np
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def agreement(buffers: np.ndarray, lengths: list[int]) -> np.ndarray:
|
|
25
|
+
"""Scores, shape (number of buffers, number of lengths)."""
|
|
26
|
+
nbuf, size = buffers.shape
|
|
27
|
+
out = np.zeros((nbuf, len(lengths)))
|
|
28
|
+
for j, r in enumerate(lengths):
|
|
29
|
+
if r < 1 or size < 2 * r:
|
|
30
|
+
continue
|
|
31
|
+
out[:, j] = (buffers[:, : size - r] == buffers[:, r:]).mean(axis=1)
|
|
32
|
+
return out
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def compared(buffers: np.ndarray, lengths: list[int]) -> np.ndarray:
|
|
36
|
+
"""How many byte pairs were compared for each length, over all buffers."""
|
|
37
|
+
nbuf, size = buffers.shape
|
|
38
|
+
r = np.array(lengths)
|
|
39
|
+
return np.where(size >= 2 * r, 1.0 * (size - r) * nbuf, 0.0)
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def choose(
|
|
43
|
+
lengths: list[int],
|
|
44
|
+
scores: list[float],
|
|
45
|
+
counts: list[float] | None = None,
|
|
46
|
+
tolerance: float = 0.25,
|
|
47
|
+
) -> int:
|
|
48
|
+
"""The record length the scores point to.
|
|
49
|
+
|
|
50
|
+
Multiples of the record length score about as well as the length itself,
|
|
51
|
+
since the data repeat at those too; the longer ones only less precisely,
|
|
52
|
+
as fewer bytes overlap. So the answer is the smallest length that divides
|
|
53
|
+
the best-scoring one and scores nearly as well: within three standard
|
|
54
|
+
errors of it, or within ``tolerance`` of its height above a typical
|
|
55
|
+
score, whichever is more. It must also score at least half that height
|
|
56
|
+
above a typical score, so that a length with no structure of its own (1,
|
|
57
|
+
say) isn't chosen when the signal is weak.
|
|
58
|
+
|
|
59
|
+
A typical score is the lower quartile: with short records, most of the
|
|
60
|
+
candidates may be multiples of the record length, so the median may not
|
|
61
|
+
be typical. With fewer than five candidates, the lowest score stands in,
|
|
62
|
+
and the structure check is skipped.
|
|
63
|
+
|
|
64
|
+
``counts`` are the numbers of byte pairs compared, for the standard errors.
|
|
65
|
+
"""
|
|
66
|
+
best = max(range(len(lengths)), key=lambda j: (scores[j], -lengths[j]))
|
|
67
|
+
few = len(lengths) < 5
|
|
68
|
+
baseline = float(min(scores) if few else np.percentile(scores, 25))
|
|
69
|
+
height = scores[best] - baseline
|
|
70
|
+
p = min(max(baseline, 1 / 256), 1 - 1 / 256)
|
|
71
|
+
|
|
72
|
+
def stderr(j): # of a fraction near the baseline
|
|
73
|
+
return float(np.sqrt(p * (1 - p) / counts[j])) if counts and counts[j] > 0 else 0.0
|
|
74
|
+
|
|
75
|
+
for j in sorted(range(len(lengths)), key=lambda j: lengths[j]):
|
|
76
|
+
r = lengths[j]
|
|
77
|
+
if not r or lengths[best] % r:
|
|
78
|
+
continue
|
|
79
|
+
margin = max(tolerance * height, 3 * np.hypot(stderr(j), stderr(best)))
|
|
80
|
+
# nearly as good as the best, and clearly better than a typical length
|
|
81
|
+
if scores[j] >= scores[best] - margin and (few or scores[j] - baseline >= height / 2):
|
|
82
|
+
return j
|
|
83
|
+
return best
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""recs: extract records from a stream of data by their start and end markers."""
|