bmc-sensor-audit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- bmc_sensor_audit/__init__.py +6 -0
- bmc_sensor_audit/cli.py +640 -0
- bmc_sensor_audit/detect/__init__.py +10 -0
- bmc_sensor_audit/detect/attestation.py +231 -0
- bmc_sensor_audit/detect/feeder.py +331 -0
- bmc_sensor_audit/detect/generator.py +496 -0
- bmc_sensor_audit/detect/supplemental.py +301 -0
- bmc_sensor_audit/inventory/__init__.py +1 -0
- bmc_sensor_audit/inventory/diff.py +380 -0
- bmc_sensor_audit/inventory/entity_manager.py +561 -0
- bmc_sensor_audit/inventory/redfish.py +544 -0
- bmc_sensor_audit/inventory/redfish_properties.json +248 -0
- bmc_sensor_audit/inventory/redfish_schema.py +110 -0
- bmc_sensor_audit/inventory/regression.py +266 -0
- bmc_sensor_audit/inventory/sensor_types.py +122 -0
- bmc_sensor_audit/report.py +543 -0
- bmc_sensor_audit/testing/__init__.py +5 -0
- bmc_sensor_audit/testing/mock_redfish.py +204 -0
- bmc_sensor_audit-0.1.0.dist-info/METADATA +545 -0
- bmc_sensor_audit-0.1.0.dist-info/RECORD +24 -0
- bmc_sensor_audit-0.1.0.dist-info/WHEEL +4 -0
- bmc_sensor_audit-0.1.0.dist-info/entry_points.txt +2 -0
- bmc_sensor_audit-0.1.0.dist-info/licenses/LICENSE +202 -0
- bmc_sensor_audit-0.1.0.dist-info/licenses/NOTICE +46 -0
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
"""Find the sensors that should be reporting and are not."""
|
|
2
|
+
|
|
3
|
+
# The single source. `pyproject.toml` declares the version dynamic and reads it
|
|
4
|
+
# from here, so a bump is one edit and the wheel's metadata cannot disagree with
|
|
5
|
+
# what the installed package reports about itself.
|
|
6
|
+
__version__ = "0.1.0"
|
bmc_sensor_audit/cli.py
ADDED
|
@@ -0,0 +1,640 @@
|
|
|
1
|
+
"""Command line entry point for Stage 1.
|
|
2
|
+
|
|
3
|
+
bmc-sensor-audit coverage --config <path> --target https://<bmc> [--insecure]
|
|
4
|
+
bmc-sensor-audit coverage --config <path> --walk recorded-walk.json
|
|
5
|
+
bmc-sensor-audit declare --config <path>
|
|
6
|
+
bmc-sensor-audit regression --before before.json --after after.json
|
|
7
|
+
|
|
8
|
+
`--config` accepts a file or a directory, and a directory is walked recursively,
|
|
9
|
+
because a platform's declaration is normally several files (baseboard, chassis,
|
|
10
|
+
front panel) and asking an operator to enumerate them invites them to miss one.
|
|
11
|
+
|
|
12
|
+
**Exit codes are the CI interface**: 0 clean, 1 regressions found, 2 the run
|
|
13
|
+
could not be completed. 2 is distinct from 1 on purpose -- a pipeline that
|
|
14
|
+
treats "could not reach the BMC" as "sensors are missing" will fail a good
|
|
15
|
+
firmware image, and it only has to do that once before nobody trusts the gate.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import argparse
|
|
21
|
+
import json
|
|
22
|
+
import sys
|
|
23
|
+
import tempfile
|
|
24
|
+
from datetime import datetime
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from .inventory.diff import compare
|
|
28
|
+
from .inventory.entity_manager import load_declaration
|
|
29
|
+
from .inventory.redfish import (RedfishClient, Walk, order_walks, walk_chassis,
|
|
30
|
+
walk_from_dict)
|
|
31
|
+
from .inventory.regression import compare_walks
|
|
32
|
+
from .report import (as_json, as_text, regression_as_json, regression_as_text,
|
|
33
|
+
strict_fields_as_text)
|
|
34
|
+
|
|
35
|
+
EXIT_CLEAN, EXIT_REGRESSION, EXIT_INCOMPLETE = 0, 1, 2
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _load_recorded_walk(path: str) -> Walk:
|
|
39
|
+
"""Rehydrate a walk from a recorded fixture.
|
|
40
|
+
|
|
41
|
+
Recording once and diffing repeatedly is how the firmware-upgrade gate works:
|
|
42
|
+
capture before, capture after, compare both against the config. It is also
|
|
43
|
+
how the test suite runs with no hardware in the room.
|
|
44
|
+
"""
|
|
45
|
+
return walk_from_dict(json.loads(Path(path).read_text()))
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _walk_span(walks: list[Walk]) -> str | None:
|
|
49
|
+
"""How much wall-clock time these walks cover, or nothing if it is unknowable.
|
|
50
|
+
|
|
51
|
+
Nothing rather than zero when any walk is unstamped: a run whose captures carry
|
|
52
|
+
no times covers an unknown span, and printing `0:00:00` would state the one
|
|
53
|
+
answer that is certainly wrong.
|
|
54
|
+
"""
|
|
55
|
+
if len(walks) < 2:
|
|
56
|
+
return None
|
|
57
|
+
stamps = [w.captured_at for w in walks]
|
|
58
|
+
if not all(stamps):
|
|
59
|
+
return None
|
|
60
|
+
try:
|
|
61
|
+
first = datetime.fromisoformat(stamps[0])
|
|
62
|
+
last = datetime.fromisoformat(stamps[-1])
|
|
63
|
+
except ValueError:
|
|
64
|
+
return None
|
|
65
|
+
return f"{last - first} ({stamps[0]} to {stamps[-1]})"
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _client(args: argparse.Namespace) -> RedfishClient:
|
|
69
|
+
return RedfishClient(args.target, username=args.username, password=args.password,
|
|
70
|
+
verify_tls=not args.insecure, timeout=args.timeout)
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def _cmd_capture(args: argparse.Namespace) -> int:
|
|
74
|
+
"""Record a walk to disk, for diffing later or for a before/after gate."""
|
|
75
|
+
walk = walk_chassis(_client(args))
|
|
76
|
+
Path(args.out).write_text(json.dumps(walk.to_dict(), indent=2))
|
|
77
|
+
print(f"wrote {len(walk)} sensor(s) to {args.out}")
|
|
78
|
+
print(f" chassis {len(walk.chassis)}")
|
|
79
|
+
print(f" tree shapes {sorted(walk.shapes_seen) or '(none found)'}")
|
|
80
|
+
if walk.latencies:
|
|
81
|
+
times = sorted(t for _, t in walk.latencies)
|
|
82
|
+
slowest_path, slowest = max(walk.latencies, key=lambda pair: pair[1])
|
|
83
|
+
# The TAIL, not the mean. A Redfish stack that has started to struggle
|
|
84
|
+
# answers most requests normally and a few very slowly, and a mean over a
|
|
85
|
+
# hundred fetches hides exactly that.
|
|
86
|
+
print(f" fetches {len(times)} median {times[len(times)//2]:.3f}s "
|
|
87
|
+
f"slowest {slowest:.3f}s")
|
|
88
|
+
print(f" slowest was {slowest_path}")
|
|
89
|
+
if walk.divergence:
|
|
90
|
+
print(f" {len(walk.divergence)} sensor(s) present on only one interface")
|
|
91
|
+
drifting = [s for s in walk if s.undeclared]
|
|
92
|
+
if drifting:
|
|
93
|
+
# Surfaced at capture time without a flag, because this is where the
|
|
94
|
+
# evidence is. The capture keeps the property names, so the detail is
|
|
95
|
+
# recoverable later -- but a signal nobody knows to ask for is one nobody
|
|
96
|
+
# asks for.
|
|
97
|
+
print(f" {len(drifting)} sensor(s) carry properties the published schema "
|
|
98
|
+
f"does not declare")
|
|
99
|
+
print(" coverage --strict-fields names them")
|
|
100
|
+
if not walk.complete:
|
|
101
|
+
# Written anyway: a partial capture is still evidence, and deleting it
|
|
102
|
+
# loses the record of WHICH subtree failed. But it must not be mistaken
|
|
103
|
+
# for a baseline, and a diff against it withholds absence findings.
|
|
104
|
+
print(f" ** INCOMPLETE -- {len(walk.errors)} fetch(es) failed **")
|
|
105
|
+
for path, reason in walk.errors[:5]:
|
|
106
|
+
print(f" {path}: {reason}")
|
|
107
|
+
return EXIT_INCOMPLETE
|
|
108
|
+
return EXIT_CLEAN
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _cmd_declare(args: argparse.Namespace) -> int:
|
|
112
|
+
declaration = load_declaration(args.config)
|
|
113
|
+
print(f"read {declaration.files_read} file(s) from {len(args.config)} path(s)")
|
|
114
|
+
print(f" sensors declared {len(declaration):>5}")
|
|
115
|
+
print(f" templated names {len(declaration.templated):>5}")
|
|
116
|
+
print(f" disabled in config {len(declaration.disabled):>5}")
|
|
117
|
+
print(f" anomalies {len(declaration.anomalies):>5}")
|
|
118
|
+
print(f" unreadable files {len(declaration.unreadable):>5}")
|
|
119
|
+
for source, reason in declaration.unreadable:
|
|
120
|
+
print(f" {source}: {reason}")
|
|
121
|
+
for anomaly in declaration.anomalies:
|
|
122
|
+
print(f" {anomaly}")
|
|
123
|
+
|
|
124
|
+
# A file that parses and declares nothing is a THIRD state, and the summary
|
|
125
|
+
# above cannot express it. Point this at a directory of JSON schemas and it
|
|
126
|
+
# prints `read 22 file(s)` with `0 unreadable` -- every number honest, the
|
|
127
|
+
# answer meaningless, and indistinguishable from a board that genuinely
|
|
128
|
+
# declares nothing. That is the exact shape this tool exists to catch on
|
|
129
|
+
# someone else's machine, and `coverage` already refuses it; `declare` was
|
|
130
|
+
# reporting it as clean and exiting 0.
|
|
131
|
+
#
|
|
132
|
+
# The two causes are split because they have different fixes: a path that
|
|
133
|
+
# matched no files is usually wrong, while a path that matched files
|
|
134
|
+
# declaring nothing is usually pointed at the wrong KIND of directory.
|
|
135
|
+
if not declaration.sensors:
|
|
136
|
+
if declaration.files_read == 0:
|
|
137
|
+
print("no files were read under the given paths -- check the path",
|
|
138
|
+
file=sys.stderr)
|
|
139
|
+
else:
|
|
140
|
+
print(f"{declaration.files_read} file(s) read, none of which declares "
|
|
141
|
+
"a sensor. Nothing here can be audited -- check the path names a "
|
|
142
|
+
"configuration directory and not, say, a schema directory.",
|
|
143
|
+
file=sys.stderr)
|
|
144
|
+
return EXIT_INCOMPLETE
|
|
145
|
+
|
|
146
|
+
# An unreadable config is not a clean board; it is an unknown one.
|
|
147
|
+
return EXIT_INCOMPLETE if declaration.unreadable else EXIT_CLEAN
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _report_unreadable(declaration) -> int:
|
|
151
|
+
"""An unreadable config is not a clean board; it is an unknown one.
|
|
152
|
+
|
|
153
|
+
Returns the exit code this fact floors the answer at, so a caller composes it with
|
|
154
|
+
whatever else it found instead of choosing between them.
|
|
155
|
+
|
|
156
|
+
`declare` has applied this rule at its own exit since the beginning. `coverage` and
|
|
157
|
+
`detect` did not: both printed `cannot read: ... every sensor this file declares is
|
|
158
|
+
unverifiable, not absent` and then exited 0, which is the single outcome that
|
|
159
|
+
sentence rules out. Reported from outside against `detect`; `coverage` carried the
|
|
160
|
+
same guard and the same hole. The case that matters is in neither report -- a real
|
|
161
|
+
configuration directory with one corrupt file in it, where everything else audits
|
|
162
|
+
normally and the gate goes green.
|
|
163
|
+
|
|
164
|
+
Printed from here rather than from each exit so it is reached whether or not the
|
|
165
|
+
optional engine extra is installed. At the exit it would be emitted only on the
|
|
166
|
+
path that already had a reason to fail.
|
|
167
|
+
"""
|
|
168
|
+
if not declaration.unreadable:
|
|
169
|
+
return EXIT_CLEAN
|
|
170
|
+
print(f"\n{len(declaration.unreadable)} configuration file(s) could not be read. "
|
|
171
|
+
"The sensors they declare are unverifiable, not absent, so this run cannot "
|
|
172
|
+
"report a clean board:", file=sys.stderr)
|
|
173
|
+
for source, reason in declaration.unreadable:
|
|
174
|
+
print(f" {source}: {reason}", file=sys.stderr)
|
|
175
|
+
return EXIT_INCOMPLETE
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _report_unobserved_fields(walk: Walk, requested: bool) -> int:
|
|
179
|
+
"""A strictness check that was asked for and could not run floors the exit at 2.
|
|
180
|
+
|
|
181
|
+
Returns the floor, the same shape as `_report_unreadable`, so a caller composes
|
|
182
|
+
it with whatever else it found instead of choosing between them.
|
|
183
|
+
|
|
184
|
+
**Reported from outside, and it sat on the thesis.** The report printed
|
|
185
|
+
`NOT CHECKED` and the process exited 0, so a pipeline gating on
|
|
186
|
+
`--strict-fields` over a capture written before object properties were
|
|
187
|
+
recorded went green with the strictness half never having run. Honest prose
|
|
188
|
+
beside a clean exit code is the exact failure this tool is pointed at: the
|
|
189
|
+
exit code is the claim a gate reads, and the prose is not.
|
|
190
|
+
|
|
191
|
+
The precedent is this repository's own, in three places already -- `detect`
|
|
192
|
+
without the engine prints its coverage findings and exits 2, an unreadable
|
|
193
|
+
configuration file floors at 2, and an incomplete walk exits 2. All three are
|
|
194
|
+
the same sentence: a run that could not complete the audit it was asked for
|
|
195
|
+
must not read as clean.
|
|
196
|
+
|
|
197
|
+
**Only when the check was requested.** An old capture used without the flag is
|
|
198
|
+
a perfectly complete coverage run, and flooring it would fail every gate that
|
|
199
|
+
never asked the question.
|
|
200
|
+
|
|
201
|
+
**And only for the requested check.** `regression` computes field drift
|
|
202
|
+
opportunistically when both walks happen to carry observations, says so when
|
|
203
|
+
they do not, and does NOT floor: the removal, rename and threshold comparisons
|
|
204
|
+
it was actually asked for all completed. Flooring there would turn a fully
|
|
205
|
+
answered question red because a bonus one could not be asked, which is how a
|
|
206
|
+
gate teaches people to stop reading it.
|
|
207
|
+
"""
|
|
208
|
+
if not requested or walk.fields_observed:
|
|
209
|
+
return EXIT_CLEAN
|
|
210
|
+
from .report import unobserved_reason
|
|
211
|
+
|
|
212
|
+
print(f"\nfield strictness was requested and could not be checked: "
|
|
213
|
+
f"{unobserved_reason(walk)}.\nThis run has not answered the question it "
|
|
214
|
+
f"was asked, so it does not exit clean.", file=sys.stderr)
|
|
215
|
+
return EXIT_INCOMPLETE
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def _report_uncomparable_fields(before: Walk, after: Walk, requested: bool) -> int:
|
|
219
|
+
"""The same rule, applied to the comparison rather than to one walk.
|
|
220
|
+
|
|
221
|
+
Field drift is computed opportunistically when both walks happen to carry
|
|
222
|
+
observations, and `regression` reports honestly when they do not -- but until
|
|
223
|
+
there was a flag, that was ALL it could do. A pipeline that gates firmware on
|
|
224
|
+
`regression` and needs drift covered had no handle: the run said *not
|
|
225
|
+
computed* in prose and exited on the strength of the comparisons that did run.
|
|
226
|
+
The same could-not-complete-reads-as-clean shape as the strictness finding,
|
|
227
|
+
one door over, and reported from outside in the same way.
|
|
228
|
+
|
|
229
|
+
**A flag rather than a default, and the weight is the reason.** Flooring
|
|
230
|
+
flagless would turn every regression run against an older baseline into exit
|
|
231
|
+
2, breaking the removal and rename gating that works perfectly well on those
|
|
232
|
+
captures -- a real cost paid by exactly the operators the subcommand serves
|
|
233
|
+
best. So drift stays best-effort until somebody asks for it, and asking is
|
|
234
|
+
what makes the existing rule apply.
|
|
235
|
+
|
|
236
|
+
**Which side is named**, because the fix differs: one old capture means
|
|
237
|
+
re-capture that one, and two mean the baseline predates the field entirely.
|
|
238
|
+
"""
|
|
239
|
+
if not requested or (before.fields_observed and after.fields_observed):
|
|
240
|
+
return EXIT_CLEAN
|
|
241
|
+
from .report import unobserved_reason
|
|
242
|
+
|
|
243
|
+
missing = [(label, walk) for label, walk in (("--before", before), ("--after", after))
|
|
244
|
+
if not walk.fields_observed]
|
|
245
|
+
which = ("neither capture carries a record" if len(missing) == 2
|
|
246
|
+
else "one of the two captures carries no record")
|
|
247
|
+
print(f"\nfield drift was requested and could not be compared: {which} of what "
|
|
248
|
+
f"properties each object reported.", file=sys.stderr)
|
|
249
|
+
for label, walk in missing:
|
|
250
|
+
print(f" {label}: {unobserved_reason(walk)}", file=sys.stderr)
|
|
251
|
+
print("This run has not answered the question it was asked, so it does not "
|
|
252
|
+
"exit clean.", file=sys.stderr)
|
|
253
|
+
return EXIT_INCOMPLETE
|
|
254
|
+
|
|
255
|
+
|
|
256
|
+
def _cmd_coverage(args: argparse.Namespace) -> int:
|
|
257
|
+
declaration = load_declaration(args.config)
|
|
258
|
+
if not declaration.sensors and not declaration.unreadable:
|
|
259
|
+
print("no sensors declared by any file under the given paths", file=sys.stderr)
|
|
260
|
+
return EXIT_INCOMPLETE
|
|
261
|
+
unreadable_floor = _report_unreadable(declaration)
|
|
262
|
+
|
|
263
|
+
if args.walk:
|
|
264
|
+
walk = _load_recorded_walk(args.walk)
|
|
265
|
+
target = args.walk
|
|
266
|
+
else:
|
|
267
|
+
walk = walk_chassis(_client(args))
|
|
268
|
+
target = args.target
|
|
269
|
+
|
|
270
|
+
report = compare(declaration, walk,
|
|
271
|
+
include_disabled_in_config=args.include_disabled)
|
|
272
|
+
rendered = (as_json(report, target=target,
|
|
273
|
+
walk=walk if args.strict_fields else None) if args.json
|
|
274
|
+
else as_text(report, target=target))
|
|
275
|
+
print(rendered)
|
|
276
|
+
|
|
277
|
+
if args.strict_fields and not args.json:
|
|
278
|
+
# What it FINDS is reported and never scored. A vendor extension is not a
|
|
279
|
+
# regression -- the firmware is doing something the standard permits, and
|
|
280
|
+
# a gate that failed on the first one gets switched off within a week,
|
|
281
|
+
# taking the signal with it. What DOES fail a gate is an extension that
|
|
282
|
+
# ARRIVED, which is a comparison between two firmware versions and belongs
|
|
283
|
+
# to `regression`.
|
|
284
|
+
#
|
|
285
|
+
# Whether it RAN is a different question, and it is scored. See below.
|
|
286
|
+
print(strict_fields_as_text(walk, target=target))
|
|
287
|
+
strict_floor = _report_unobserved_fields(walk, args.strict_fields)
|
|
288
|
+
|
|
289
|
+
if not report.walk_complete:
|
|
290
|
+
return EXIT_INCOMPLETE
|
|
291
|
+
# Composed the way `detect` composes its two stages: the worse wins, and 2 outranks
|
|
292
|
+
# 1 because could-not-read is a different claim from something-got-worse.
|
|
293
|
+
stage1 = EXIT_REGRESSION if report.regressions else EXIT_CLEAN
|
|
294
|
+
return max(stage1, unreadable_floor, strict_floor)
|
|
295
|
+
|
|
296
|
+
|
|
297
|
+
def _cmd_detect(args: argparse.Namespace) -> int:
|
|
298
|
+
"""Both stages in one run, and one exit code.
|
|
299
|
+
|
|
300
|
+
Stage 1 answers presence; Stage 2 answers liveness for what is present. They are
|
|
301
|
+
composed rather than merged, because they fail differently: a walk that could not
|
|
302
|
+
complete is not a board with missing sensors, and neither is an engine that is not
|
|
303
|
+
installed.
|
|
304
|
+
"""
|
|
305
|
+
declaration = load_declaration(args.config)
|
|
306
|
+
if not declaration.sensors and not declaration.unreadable:
|
|
307
|
+
print("no sensors declared by any file under the given paths", file=sys.stderr)
|
|
308
|
+
return EXIT_INCOMPLETE
|
|
309
|
+
unreadable_floor = _report_unreadable(declaration)
|
|
310
|
+
|
|
311
|
+
# `--walk` is repeatable and CHRONOLOGICAL, oldest first: stuck-at needs history,
|
|
312
|
+
# and one walk is one sample. A live target gives exactly one.
|
|
313
|
+
if args.walk:
|
|
314
|
+
walks = [_load_recorded_walk(path) for path in args.walk]
|
|
315
|
+
# The last walk supplies every current reading, so the order is not a
|
|
316
|
+
# presentation detail. A shell glob hands over lexical order, in which
|
|
317
|
+
# `walk10` precedes `walk9`.
|
|
318
|
+
walks, ordering = order_walks(walks)
|
|
319
|
+
if ordering:
|
|
320
|
+
print(f"\n{ordering}", file=sys.stderr)
|
|
321
|
+
span = _walk_span(walks)
|
|
322
|
+
if span:
|
|
323
|
+
# The verdict is over the values; this is over the clock. The engine is
|
|
324
|
+
# told every sample is a minute old regardless of when the walk was
|
|
325
|
+
# taken, so `frozen` alone does not say whether the reading held still
|
|
326
|
+
# for a minute or for a shift. That distinction is only in the stamps.
|
|
327
|
+
print(f"\n{len(walks)} walks covering {span}")
|
|
328
|
+
target = args.walk[-1]
|
|
329
|
+
else:
|
|
330
|
+
walks = [walk_chassis(_client(args))]
|
|
331
|
+
target = args.target
|
|
332
|
+
|
|
333
|
+
reports = [compare(declaration, walk,
|
|
334
|
+
include_disabled_in_config=args.include_disabled)
|
|
335
|
+
for walk in walks]
|
|
336
|
+
current = reports[-1]
|
|
337
|
+
print(as_text(current, target=target))
|
|
338
|
+
|
|
339
|
+
if not current.walk_complete:
|
|
340
|
+
# An incomplete walk is not an empty machine, and it is not a model worth
|
|
341
|
+
# feeding either. Stop before the engine sees a partial picture.
|
|
342
|
+
print("\nwalk incomplete; liveness not evaluated", file=sys.stderr)
|
|
343
|
+
return EXIT_INCOMPLETE
|
|
344
|
+
|
|
345
|
+
try:
|
|
346
|
+
import yaml
|
|
347
|
+
from arbiter_engine.api import EngineSession, check, model_describe
|
|
348
|
+
except ImportError as error:
|
|
349
|
+
print(f"\nliveness needs the optional extra, which is not installed: {error}\n"
|
|
350
|
+
" pip install 'bmc-sensor-audit[detect]'\n"
|
|
351
|
+
"Stage 1 coverage above is complete and unaffected.", file=sys.stderr)
|
|
352
|
+
return EXIT_INCOMPLETE
|
|
353
|
+
|
|
354
|
+
from .detect.feeder import evaluate, feed
|
|
355
|
+
from .detect.generator import generate
|
|
356
|
+
from .detect.supplemental import (SupplementalError, load_supplemental,
|
|
357
|
+
unmatched_names)
|
|
358
|
+
from .report import detect_as_text, supplemental_as_text
|
|
359
|
+
|
|
360
|
+
supplemental = None
|
|
361
|
+
if args.supplemental:
|
|
362
|
+
# A refusal here stops the run rather than degrading it. A supplemental file
|
|
363
|
+
# that failed to load and carried on would produce a report with no
|
|
364
|
+
# disagreements in it because nothing was ever compared -- and from the
|
|
365
|
+
# outside that is identical to a board where every declared pair agrees.
|
|
366
|
+
try:
|
|
367
|
+
supplemental = load_supplemental(args.supplemental)
|
|
368
|
+
except SupplementalError as error:
|
|
369
|
+
print(f"\n{error}", file=sys.stderr)
|
|
370
|
+
return EXIT_INCOMPLETE
|
|
371
|
+
missing = unmatched_names(supplemental,
|
|
372
|
+
{s.display_name for s in declaration.sensors})
|
|
373
|
+
if missing:
|
|
374
|
+
print(f"\n{args.supplemental} names {len(missing)} sensor(s) this "
|
|
375
|
+
f"configuration does not declare. A name that matches nothing "
|
|
376
|
+
f"creates no check, silently:", file=sys.stderr)
|
|
377
|
+
for name in missing:
|
|
378
|
+
print(f" {name}", file=sys.stderr)
|
|
379
|
+
return EXIT_INCOMPLETE
|
|
380
|
+
# Printed whether or not anything is missing a number, so the reader sees
|
|
381
|
+
# what was declared before they read a verdict that rests on it.
|
|
382
|
+
print(supplemental_as_text(supplemental))
|
|
383
|
+
|
|
384
|
+
model, manifest = generate(declaration, expect_variation=not args.no_stuck_at,
|
|
385
|
+
supplemental=supplemental)
|
|
386
|
+
if args.model_out:
|
|
387
|
+
Path(args.model_out).write_text(yaml.safe_dump(model))
|
|
388
|
+
if args.manifest_out:
|
|
389
|
+
Path(args.manifest_out).write_text(json.dumps(manifest.to_dict(), indent=2))
|
|
390
|
+
|
|
391
|
+
with tempfile.NamedTemporaryFile("w", suffix=".yaml", delete=False) as handle:
|
|
392
|
+
handle.write(yaml.safe_dump(model))
|
|
393
|
+
model_path = handle.name
|
|
394
|
+
session = EngineSession()
|
|
395
|
+
session.load_model(model_path)
|
|
396
|
+
|
|
397
|
+
feed_result = feed(session, manifest, reports)
|
|
398
|
+
envelope = check(session).to_dict()
|
|
399
|
+
described = model_describe(session).to_dict()
|
|
400
|
+
outcome = evaluate(envelope, described, manifest,
|
|
401
|
+
strict_declines=args.strict_declines,
|
|
402
|
+
feed_result=feed_result)
|
|
403
|
+
|
|
404
|
+
if args.attest_out:
|
|
405
|
+
# After `check`, never before: `attest` refuses on an unchecked session with
|
|
406
|
+
# `source: unavailable`, and that refusal reads a lot like a clean run.
|
|
407
|
+
from arbiter_engine.api import attest
|
|
408
|
+
|
|
409
|
+
from .detect.attestation import build_attestation
|
|
410
|
+
# The artifact leaves through a different door from every committed file,
|
|
411
|
+
# and the hygiene perimeter guards commits. `target` is a Redfish URL by
|
|
412
|
+
# default, so an artifact uploaded from CI can publish an internal hostname
|
|
413
|
+
# in a channel no hook ever scans. The label is the operator's override; no
|
|
414
|
+
# guessing at which hostnames look internal happens here, because that is
|
|
415
|
+
# pattern-matching a judgement only they can make.
|
|
416
|
+
artifact = build_attestation(session, envelope, described, manifest,
|
|
417
|
+
target=args.attest_target_label or target,
|
|
418
|
+
attest_fn=attest)
|
|
419
|
+
Path(args.attest_out).write_text(json.dumps(artifact, indent=2))
|
|
420
|
+
|
|
421
|
+
# Said on the terminal, not only inside the file. The artifact already
|
|
422
|
+
# accounts for this honestly -- `unattested` is a required field and the
|
|
423
|
+
# shipped validator reads it -- but an operator who asked for evidence and
|
|
424
|
+
# received an artifact carrying none finds that out only by opening it.
|
|
425
|
+
# A quiet gap is not a false claim, and it is still a gap nobody sees.
|
|
426
|
+
#
|
|
427
|
+
# No exit floor: `check` completed and its findings stand. What did not
|
|
428
|
+
# complete is the evidence the engine attaches to them, which is a weaker
|
|
429
|
+
# thing than the audit itself.
|
|
430
|
+
from .report import unattested_notice
|
|
431
|
+
|
|
432
|
+
notice = unattested_notice(artifact, args.attest_out)
|
|
433
|
+
if notice:
|
|
434
|
+
print(f"\n{notice}", file=sys.stderr)
|
|
435
|
+
print(detect_as_text(outcome, feed_result))
|
|
436
|
+
|
|
437
|
+
# Composed, not merged. The worse of the four wins, and `2` outranks `1` because
|
|
438
|
+
# could-not-complete is a different claim from something-got-worse. The config
|
|
439
|
+
# floor is one of them: a run that could not read part of its own input has
|
|
440
|
+
# not verified the board, however clean the part it could read came out.
|
|
441
|
+
#
|
|
442
|
+
# An envelope whose schema version this build does not parse floors at 2 for the
|
|
443
|
+
# same reason and not at 1: nothing was found to be worse, we were unable to
|
|
444
|
+
# read the answer. Applied here rather than inside `DetectOutcome.exit_code`,
|
|
445
|
+
# which returns 0 or 1 by contract -- `2` is the caller's to give.
|
|
446
|
+
stage1 = EXIT_REGRESSION if current.regressions else EXIT_CLEAN
|
|
447
|
+
schema_floor = EXIT_INCOMPLETE if outcome.schema_mismatch else EXIT_CLEAN
|
|
448
|
+
return max(stage1, outcome.exit_code, unreadable_floor, schema_floor)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def _cmd_regression(args: argparse.Namespace) -> int:
|
|
452
|
+
"""Compare two captures of the same machine across a firmware change.
|
|
453
|
+
|
|
454
|
+
Needs no configuration and no BMC: two files and a diff. That matters for where
|
|
455
|
+
it runs -- the flashing station has the captures and often has neither the
|
|
456
|
+
entity-manager tree nor a route back to the machine by the time anyone looks.
|
|
457
|
+
"""
|
|
458
|
+
before = _load_recorded_walk(args.before)
|
|
459
|
+
after = _load_recorded_walk(args.after)
|
|
460
|
+
|
|
461
|
+
# The two captures are named, not sorted. `order_walks` exists because a glob
|
|
462
|
+
# hands over lexical order; here the operator has typed which is which, and
|
|
463
|
+
# silently swapping them because their timestamps disagree would report every
|
|
464
|
+
# removal as an addition. So it is checked and SAID, and the run stops: a
|
|
465
|
+
# backwards regression report is worse than no report, because it reads clean.
|
|
466
|
+
if before.captured_at and after.captured_at and before.captured_at > after.captured_at:
|
|
467
|
+
print(f"--before was captured at {before.captured_at} and --after at "
|
|
468
|
+
f"{after.captured_at}, which is the wrong way round. Nothing here "
|
|
469
|
+
f"reorders them: a reversed comparison reports every removal as an "
|
|
470
|
+
f"addition and reads like a clean upgrade.", file=sys.stderr)
|
|
471
|
+
return EXIT_INCOMPLETE
|
|
472
|
+
|
|
473
|
+
report = compare_walks(before, after)
|
|
474
|
+
print(regression_as_json(report, before=args.before, after=args.after) if args.json
|
|
475
|
+
else regression_as_text(report, before=args.before, after=args.after))
|
|
476
|
+
|
|
477
|
+
if args.strict_fields and not args.json:
|
|
478
|
+
# The AFTER walk's own strictness, so the flag means the same sentence in
|
|
479
|
+
# both commands -- apply field strictness, and require it to be
|
|
480
|
+
# applicable -- rather than sharing a name with `coverage` while doing
|
|
481
|
+
# something else. Two flags spelled alike that mean different things is
|
|
482
|
+
# its own defect, and a worse one than two names.
|
|
483
|
+
#
|
|
484
|
+
# The absolute view and the delta answer different questions. `field_drift`
|
|
485
|
+
# above names what ARRIVED; this names what the firmware carries now,
|
|
486
|
+
# which is what a downstream parser actually meets.
|
|
487
|
+
print(strict_fields_as_text(after, target=args.after))
|
|
488
|
+
strict_floor = _report_uncomparable_fields(before, after, args.strict_fields)
|
|
489
|
+
|
|
490
|
+
if not report.complete:
|
|
491
|
+
return EXIT_INCOMPLETE
|
|
492
|
+
stage1 = EXIT_REGRESSION if report.regressions else EXIT_CLEAN
|
|
493
|
+
return max(stage1, strict_floor)
|
|
494
|
+
|
|
495
|
+
|
|
496
|
+
def _cmd_validate_attestation(args: argparse.Namespace) -> int:
|
|
497
|
+
"""Check an attestation artifact against the format it declares.
|
|
498
|
+
|
|
499
|
+
Needs no engine and no hardware: an artifact is JSON, so the person who
|
|
500
|
+
RECEIVES one can run this over a file somebody sent them. That is the point of
|
|
501
|
+
the command existing rather than the rule living inside a CI workflow where only
|
|
502
|
+
the producer can reach it.
|
|
503
|
+
"""
|
|
504
|
+
from .detect.attestation import validate_attestation
|
|
505
|
+
|
|
506
|
+
try:
|
|
507
|
+
artifact = json.loads(Path(args.path).read_text())
|
|
508
|
+
except OSError as error:
|
|
509
|
+
print(f"cannot read {args.path}: {error}", file=sys.stderr)
|
|
510
|
+
return EXIT_INCOMPLETE
|
|
511
|
+
except json.JSONDecodeError as error:
|
|
512
|
+
print(f"{args.path} is not parseable as JSON: {error}", file=sys.stderr)
|
|
513
|
+
return EXIT_INCOMPLETE
|
|
514
|
+
|
|
515
|
+
problems = validate_attestation(artifact)
|
|
516
|
+
if problems:
|
|
517
|
+
print(f"{args.path}: {len(problems)} problem(s)", file=sys.stderr)
|
|
518
|
+
for problem in problems:
|
|
519
|
+
print(f" {problem}", file=sys.stderr)
|
|
520
|
+
return EXIT_REGRESSION
|
|
521
|
+
|
|
522
|
+
findings = len(artifact.get("findings") or [])
|
|
523
|
+
declined = len(artifact.get("not_checked") or [])
|
|
524
|
+
print(f"{args.path}: valid {artifact['format']}")
|
|
525
|
+
print(f" {findings} finding(s), {declined} declined, "
|
|
526
|
+
f"{len(artifact.get('evidence') or [])} with measurements")
|
|
527
|
+
# Printed because a reader's next question is what the judgment rests on, and
|
|
528
|
+
# because an artifact that validates still carries the engine's own limit.
|
|
529
|
+
print(f" judged under envelope schema_version "
|
|
530
|
+
f"{artifact['engine'].get('schema_version')}")
|
|
531
|
+
print(f" boundary: {artifact['engine']['boundary']}")
|
|
532
|
+
return EXIT_CLEAN
|
|
533
|
+
|
|
534
|
+
|
|
535
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
536
|
+
parser = argparse.ArgumentParser(
|
|
537
|
+
prog="bmc-sensor-audit",
|
|
538
|
+
description="Find the sensors that should be reporting and are not.")
|
|
539
|
+
subparsers = parser.add_subparsers(dest="command", required=True)
|
|
540
|
+
|
|
541
|
+
declare = subparsers.add_parser(
|
|
542
|
+
"declare", help="read the configuration and report what it declares")
|
|
543
|
+
declare.add_argument("--config", required=True, action="append",
|
|
544
|
+
help="entity-manager JSON file or directory (repeatable)")
|
|
545
|
+
declare.set_defaults(func=_cmd_declare)
|
|
546
|
+
|
|
547
|
+
coverage = subparsers.add_parser(
|
|
548
|
+
"coverage", help="diff a declaration against what a machine reports")
|
|
549
|
+
coverage.add_argument("--config", required=True, action="append",
|
|
550
|
+
help="entity-manager JSON file or directory (repeatable)")
|
|
551
|
+
source = coverage.add_mutually_exclusive_group(required=True)
|
|
552
|
+
source.add_argument("--target", help="Redfish base URL, e.g. https://bmc.example")
|
|
553
|
+
source.add_argument("--walk", help="a recorded walk, instead of live hardware")
|
|
554
|
+
coverage.add_argument("--username")
|
|
555
|
+
coverage.add_argument("--password")
|
|
556
|
+
coverage.add_argument("--insecure", action="store_true",
|
|
557
|
+
help="do not verify TLS; BMCs ship self-signed certificates")
|
|
558
|
+
coverage.add_argument("--timeout", type=float, default=15.0)
|
|
559
|
+
coverage.add_argument("--json", action="store_true", help="machine-readable output")
|
|
560
|
+
coverage.add_argument("--include-disabled", action="store_true",
|
|
561
|
+
help="also expect sensors the config marks Status: disabled")
|
|
562
|
+
coverage.add_argument("--strict-fields", action="store_true",
|
|
563
|
+
help="also name the properties each sensor object carries "
|
|
564
|
+
"that the published Redfish schema does not declare")
|
|
565
|
+
coverage.set_defaults(func=_cmd_coverage)
|
|
566
|
+
|
|
567
|
+
regression = subparsers.add_parser(
|
|
568
|
+
"regression",
|
|
569
|
+
help="compare two captures of one machine across a firmware change")
|
|
570
|
+
regression.add_argument("--before", required=True,
|
|
571
|
+
help="a capture taken before the flash")
|
|
572
|
+
regression.add_argument("--after", required=True,
|
|
573
|
+
help="a capture taken after it")
|
|
574
|
+
regression.add_argument("--json", action="store_true", help="machine-readable output")
|
|
575
|
+
regression.add_argument("--strict-fields", action="store_true",
|
|
576
|
+
help="also apply field strictness, and exit 2 if either "
|
|
577
|
+
"capture carries no record of object properties, "
|
|
578
|
+
"so drift cannot be compared")
|
|
579
|
+
regression.set_defaults(func=_cmd_regression)
|
|
580
|
+
|
|
581
|
+
detect = subparsers.add_parser(
|
|
582
|
+
"detect", help="coverage plus liveness, in one run and one exit code")
|
|
583
|
+
detect.add_argument("--config", required=True, action="append",
|
|
584
|
+
help="entity-manager JSON file or directory (repeatable)")
|
|
585
|
+
detect_source = detect.add_mutually_exclusive_group(required=True)
|
|
586
|
+
detect_source.add_argument("--target", help="Redfish base URL")
|
|
587
|
+
detect_source.add_argument("--walk", action="append",
|
|
588
|
+
help="a recorded walk; repeatable, OLDEST FIRST -- "
|
|
589
|
+
"stuck-at needs history and one walk is one sample")
|
|
590
|
+
detect.add_argument("--username")
|
|
591
|
+
detect.add_argument("--password")
|
|
592
|
+
detect.add_argument("--insecure", action="store_true",
|
|
593
|
+
help="do not verify TLS; BMCs ship self-signed certificates")
|
|
594
|
+
detect.add_argument("--timeout", type=float, default=15.0)
|
|
595
|
+
detect.add_argument("--include-disabled", action="store_true")
|
|
596
|
+
detect.add_argument("--strict-declines", action="store_true",
|
|
597
|
+
help="fail on data-sufficiency and unrecognised declines too")
|
|
598
|
+
detect.add_argument("--no-stuck-at", action="store_true",
|
|
599
|
+
help="do not expect readings to vary; turns off liveness")
|
|
600
|
+
detect.add_argument("--supplemental",
|
|
601
|
+
help="operator declarations the configuration cannot make: "
|
|
602
|
+
"which sensors are redundant, which are counters")
|
|
603
|
+
detect.add_argument("--model-out", help="write the generated domain model here")
|
|
604
|
+
detect.add_argument("--manifest-out", help="write the generation manifest here")
|
|
605
|
+
detect.add_argument("--attest-out",
|
|
606
|
+
help="write a per-run record of what was checked, what was "
|
|
607
|
+
"declined, and the measurements behind each finding")
|
|
608
|
+
detect.add_argument("--attest-target-label",
|
|
609
|
+
help="what the artifact should call the target instead of "
|
|
610
|
+
"its URL; a BMC hostname names an internal machine and "
|
|
611
|
+
"an artifact uploaded from CI publishes it")
|
|
612
|
+
detect.set_defaults(func=_cmd_detect)
|
|
613
|
+
|
|
614
|
+
validate = subparsers.add_parser(
|
|
615
|
+
"validate-attestation",
|
|
616
|
+
help="check an attestation artifact against the format it declares")
|
|
617
|
+
validate.add_argument("path", help="the attestation JSON to check")
|
|
618
|
+
validate.set_defaults(func=_cmd_validate_attestation)
|
|
619
|
+
|
|
620
|
+
capture = subparsers.add_parser(
|
|
621
|
+
"capture", help="record a walk to disk, for a before/after gate")
|
|
622
|
+
capture.add_argument("--target", required=True)
|
|
623
|
+
capture.add_argument("--out", required=True, help="file to write")
|
|
624
|
+
capture.add_argument("--username")
|
|
625
|
+
capture.add_argument("--password")
|
|
626
|
+
capture.add_argument("--insecure", action="store_true",
|
|
627
|
+
help="do not verify TLS; BMCs ship self-signed certificates")
|
|
628
|
+
capture.add_argument("--timeout", type=float, default=15.0)
|
|
629
|
+
capture.set_defaults(func=_cmd_capture)
|
|
630
|
+
|
|
631
|
+
return parser
|
|
632
|
+
|
|
633
|
+
|
|
634
|
+
def main(argv: list[str] | None = None) -> int:
|
|
635
|
+
args = build_parser().parse_args(argv)
|
|
636
|
+
return args.func(args)
|
|
637
|
+
|
|
638
|
+
|
|
639
|
+
if __name__ == "__main__":
|
|
640
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,10 @@
|
|
|
1
|
+
"""Stage 2: liveness detection through `arbiter-engine`.
|
|
2
|
+
|
|
3
|
+
Everything in here needs the optional `[detect]` extra. Stage 1 does not import it,
|
|
4
|
+
deliberately — the coverage diff has to run on a bring-up bench with nothing
|
|
5
|
+
provisioned, and that property is easy to lose by accident and invisible until
|
|
6
|
+
somebody is standing in front of a machine that will not boot.
|
|
7
|
+
|
|
8
|
+
The generator itself has no engine dependency: it emits a model as plain data, so it
|
|
9
|
+
can be tested and golden-pinned without installing anything.
|
|
10
|
+
"""
|