stacktrace-cli 0.5.3__py3-none-any.whl → 0.6.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stacktrace_cli/__init__.py +1 -1
- stacktrace_cli/analysis.py +33 -31
- stacktrace_cli/cli.py +5 -0
- stacktrace_cli/cloud/__init__.py +10 -0
- stacktrace_cli/cloud/setup.sh +17 -4
- stacktrace_cli/correlate/acquire.py +60 -67
- stacktrace_cli/correlate/composition.py +4 -4
- stacktrace_cli/correlate/observed.py +62 -6
- stacktrace_cli/correlate/orchestrate.py +30 -38
- stacktrace_cli/daemon/cli.py +93 -2
- stacktrace_cli/daemon/config.py +85 -5
- stacktrace_cli/daemon/identity.py +17 -1
- stacktrace_cli/daemon/jobs.py +48 -1
- stacktrace_cli/daemon/observer.py +31 -2
- stacktrace_cli/daemon/presentation.py +34 -1
- stacktrace_cli/daemon/runtime.py +25 -11
- stacktrace_cli/daemon/store.py +60 -15
- stacktrace_cli/daemon/webhook.py +34 -4
- stacktrace_cli/detector/blocked.py +12 -0
- stacktrace_cli/detector/deterministic.py +1935 -27
- stacktrace_cli/detector/render.py +6 -9
- stacktrace_cli/detector/rules.py +74 -12
- stacktrace_cli/detector/run.py +11 -3
- stacktrace_cli/monitor/inventory.py +495 -0
- stacktrace_cli/monitor/reasoning.py +8 -6
- stacktrace_cli/monitor/render.py +49 -1
- stacktrace_cli/monitor/server.py +44 -20
- stacktrace_cli/monitor/site/app.js +386 -369
- stacktrace_cli/monitor/site/index.html +26 -14
- stacktrace_cli/monitor/site/inventory.js +554 -0
- stacktrace_cli/monitor/site/styles.css +139 -51
- stacktrace_cli/monitor/state.py +45 -2
- stacktrace_cli/monitor/watch.py +294 -129
- stacktrace_cli/remote/identity.py +41 -12
- stacktrace_cli/remote/payload.py +10 -5
- stacktrace_cli/remote/redact.py +100 -2
- stacktrace_cli/remote/sync.py +11 -0
- stacktrace_cli/telemetry/events.py +12 -30
- stacktrace_cli/webhook/cli.py +35 -4
- {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/METADATA +3 -3
- {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/RECORD +43 -41
- {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/WHEEL +0 -0
- {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/entry_points.txt +0 -0
stacktrace_cli/__init__.py
CHANGED
stacktrace_cli/analysis.py
CHANGED
|
@@ -348,32 +348,6 @@ def _acquire(
|
|
|
348
348
|
)
|
|
349
349
|
|
|
350
350
|
|
|
351
|
-
def _correlated(
|
|
352
|
-
*,
|
|
353
|
-
agent_kinds: tuple[str, ...],
|
|
354
|
-
window_start: datetime,
|
|
355
|
-
bom_paths: tuple[Path, ...],
|
|
356
|
-
project_map: tuple[str, ...],
|
|
357
|
-
root: Path | None,
|
|
358
|
-
session_ids: tuple[str, ...],
|
|
359
|
-
attach_advisories: Any,
|
|
360
|
-
build_all: Any = None,
|
|
361
|
-
) -> Acquired:
|
|
362
|
-
"""Everything up to the judging, shared by both entry points."""
|
|
363
|
-
return _acquire(
|
|
364
|
-
_collected(
|
|
365
|
-
agent_kinds=agent_kinds,
|
|
366
|
-
window_start=window_start,
|
|
367
|
-
root=root,
|
|
368
|
-
session_ids=session_ids,
|
|
369
|
-
),
|
|
370
|
-
bom_paths=bom_paths,
|
|
371
|
-
project_map=project_map,
|
|
372
|
-
attach_advisories=attach_advisories,
|
|
373
|
-
build_all=build_all,
|
|
374
|
-
)
|
|
375
|
-
|
|
376
|
-
|
|
377
351
|
def _reaches(detection: Any, window_start: datetime, window_end: datetime) -> bool:
|
|
378
352
|
"""Whether this finding's own span overlaps the window.
|
|
379
353
|
|
|
@@ -482,6 +456,7 @@ def analyse_progressively(
|
|
|
482
456
|
history: DriftHistory | None = None,
|
|
483
457
|
reasoning: bool = False,
|
|
484
458
|
budget: int = DEFAULT_BUDGET,
|
|
459
|
+
exclude_from_reasoning: frozenset[tuple[str, str]] = frozenset(),
|
|
485
460
|
) -> Iterator[Analysis]:
|
|
486
461
|
"""The same pipeline, delivered in instalments.
|
|
487
462
|
|
|
@@ -634,12 +609,20 @@ def analyse_progressively(
|
|
|
634
609
|
if not reasoning or not ordered:
|
|
635
610
|
return
|
|
636
611
|
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
|
|
612
|
+
excluded_count = 0
|
|
613
|
+
if exclude_from_reasoning:
|
|
614
|
+
kept = tuple(
|
|
615
|
+
s
|
|
616
|
+
for s in view.sessions
|
|
617
|
+
if (s.session.session_id, s.session.agent_kind) not in exclude_from_reasoning
|
|
618
|
+
)
|
|
619
|
+
excluded_count = len(view.sessions) - len(kept)
|
|
620
|
+
reasoning_view = replace(view, sessions=kept) if excluded_count else view
|
|
621
|
+
else:
|
|
622
|
+
reasoning_view = view
|
|
623
|
+
|
|
641
624
|
judged = run_detector(
|
|
642
|
-
|
|
625
|
+
reasoning_view,
|
|
643
626
|
reasoning=True,
|
|
644
627
|
budget=budget,
|
|
645
628
|
cache=cache,
|
|
@@ -647,4 +630,23 @@ def analyse_progressively(
|
|
|
647
630
|
jev_key=jev_key,
|
|
648
631
|
history=history,
|
|
649
632
|
)
|
|
633
|
+
|
|
634
|
+
if excluded_count:
|
|
635
|
+
extra_detections = tuple(
|
|
636
|
+
d
|
|
637
|
+
for d in accumulated.detections
|
|
638
|
+
if (d.session.session_id, d.session.agent_kind) in exclude_from_reasoning
|
|
639
|
+
)
|
|
640
|
+
extra_unknowns = tuple(
|
|
641
|
+
u
|
|
642
|
+
for u in accumulated.unknowns
|
|
643
|
+
if (u.session.session_id, u.session.agent_kind) in exclude_from_reasoning
|
|
644
|
+
)
|
|
645
|
+
judged = replace(
|
|
646
|
+
judged,
|
|
647
|
+
detections=judged.detections + extra_detections,
|
|
648
|
+
unknowns=judged.unknowns + extra_unknowns,
|
|
649
|
+
sessions=judged.sessions + excluded_count,
|
|
650
|
+
)
|
|
651
|
+
|
|
650
652
|
yield _stage(judged, tuple(ordered))
|
stacktrace_cli/cli.py
CHANGED
|
@@ -539,6 +539,11 @@ def monitor(
|
|
|
539
539
|
Loopback only, and free to leave open: a pass is skipped when nothing has
|
|
540
540
|
changed, advisory lookups are asked once per component, and the two stages
|
|
541
541
|
that need no model are the only ones that run.
|
|
542
|
+
|
|
543
|
+
The Inventory and Vulnerabilities tabs list every component your agents can
|
|
544
|
+
load. To check them for known vulnerabilities, the package names of
|
|
545
|
+
installed components -- not only the ones a session used -- are sent to
|
|
546
|
+
osv.dev. Skills and plugins with no package are never sent.
|
|
542
547
|
"""
|
|
543
548
|
analyzer = resolve_analyzer(analyzer, reasoning)
|
|
544
549
|
try:
|
stacktrace_cli/cloud/__init__.py
CHANGED
|
@@ -12,6 +12,16 @@ CLI and runs this:
|
|
|
12
12
|
uv tool install stacktrace-cli
|
|
13
13
|
stacktrace-cloud-setup
|
|
14
14
|
|
|
15
|
+
Or, equivalently, one hosted line that wraps exactly those two steps (see
|
|
16
|
+
`stacktrace-site`'s `public/remote-install.sh`). Download before running rather
|
|
17
|
+
than piping directly into `sh`: without `pipefail` (the default in the `sh`
|
|
18
|
+
that a setup-script box runs), a failed `curl` still lets `sh` read an empty
|
|
19
|
+
script and exit 0, so a broken download would otherwise snapshot an
|
|
20
|
+
environment with neither the CLI nor the boot hook and report success:
|
|
21
|
+
|
|
22
|
+
curl -fsSL https://stacktrace.ai/remote-install.sh -o /tmp/remote-install.sh \
|
|
23
|
+
&& sh /tmp/remote-install.sh
|
|
24
|
+
|
|
15
25
|
Both variables matter: setup runs as root, and `uv tool install`'s root
|
|
16
26
|
defaults put the tool's environment under `/root/.local/share/uv/tools` and the
|
|
17
27
|
linked executable under `/root/.local/bin` — both inside `/root`, which a later
|
stacktrace_cli/cloud/setup.sh
CHANGED
|
@@ -10,7 +10,10 @@
|
|
|
10
10
|
# every home so it fires whichever user the session runs as.
|
|
11
11
|
# 2. The asset identity, in every home's ~/.config/stacktrace/asset-id, so
|
|
12
12
|
# `stacktrace remote sync` resolves it at the persisted rung (ADR-0065)
|
|
13
|
-
# instead of minting a fresh asset per session.
|
|
13
|
+
# instead of minting a fresh asset per session. Beside it,
|
|
14
|
+
# asset-display-prefix makes the fleet label `cloud (<id>)` rather than
|
|
15
|
+
# the image's hostname (`vm`), which is the same on every box. Label
|
|
16
|
+
# only: the key above is what identifies the asset.
|
|
14
17
|
#
|
|
15
18
|
# The telemetry install id is deliberately NOT seeded here. Seeding it would
|
|
16
19
|
# make `read_install_id()` non-None before any event runs, so the one-time
|
|
@@ -59,7 +62,7 @@ while [ $# -gt 0 ]; do
|
|
|
59
62
|
# the loop spins forever — a malformed invocation would hang setup.
|
|
60
63
|
--asset-id) [ $# -ge 2 ] || { echo "✗ --asset-id needs a value" >&2; exit 2; }; ASSET_ID="$2"; shift 2 ;;
|
|
61
64
|
--root) [ $# -ge 2 ] || { echo "✗ --root needs a value" >&2; exit 2; }; ROOT="$2"; shift 2 ;;
|
|
62
|
-
-h|--help) sed -n '2,
|
|
65
|
+
-h|--help) sed -n '2,52p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
|
|
63
66
|
*) echo "✗ unknown argument: $1" >&2; exit 2 ;;
|
|
64
67
|
esac
|
|
65
68
|
done
|
|
@@ -78,6 +81,7 @@ _uuid() {
|
|
|
78
81
|
fi
|
|
79
82
|
}
|
|
80
83
|
|
|
84
|
+
DISPLAY_PREFIX="cloud"
|
|
81
85
|
[ -n "$ASSET_ID" ] || ASSET_ID="cloud-$(_uuid)"
|
|
82
86
|
# remote/identity.py:_is_identifier rejects whitespace, non-printable
|
|
83
87
|
# characters, and anything over 255 characters. A value this accepts but the
|
|
@@ -146,8 +150,12 @@ fi
|
|
|
146
150
|
|
|
147
151
|
# Streams redirected because own_std_streams only re-points fd 1/2 on log
|
|
148
152
|
# rollover; without this the daemon inherits the hook's pipe to Claude Code.
|
|
153
|
+
# --no-reasoning is explicit, not just `run`'s own default (ADR-0081): this
|
|
154
|
+
# hook never provisions TYPESAFE_API_KEY itself, but a cloud VM's ambient
|
|
155
|
+
# environment is not something this script controls or can audit ahead of
|
|
156
|
+
# time, and this surface has nobody present to notice if that changed.
|
|
149
157
|
if ! stacktrace daemon status 2>/dev/null | grep -q '^running: yes'; then
|
|
150
|
-
setsid nohup stacktrace daemon run </dev/null >>"$STATE/daemon.log" 2>&1 &
|
|
158
|
+
setsid nohup stacktrace daemon run --no-reasoning </dev/null >>"$STATE/daemon.log" 2>&1 &
|
|
151
159
|
fi
|
|
152
160
|
exit 0
|
|
153
161
|
BOOTSCRIPT
|
|
@@ -176,6 +184,11 @@ for home in "${ROOT}"/root "${ROOT}"/home/*; do
|
|
|
176
184
|
fi
|
|
177
185
|
chmod 600 "$cfg/asset-id" 2>/dev/null || true
|
|
178
186
|
|
|
187
|
+
if ! printf '%s\n' "$DISPLAY_PREFIX" > "$cfg/asset-display-prefix" 2>/dev/null; then
|
|
188
|
+
echo "✗ cannot write $cfg/asset-display-prefix — not seeding $home" >&2; failed=1; continue
|
|
189
|
+
fi
|
|
190
|
+
chmod 600 "$cfg/asset-display-prefix" 2>/dev/null || true
|
|
191
|
+
|
|
179
192
|
# Register the SessionStart hook, merging so any existing settings survive.
|
|
180
193
|
# The hook is the point of seeding a home — a home whose registration fails
|
|
181
194
|
# (an unwritable path, or a settings.json that is valid JSON of an unexpected
|
|
@@ -248,7 +261,7 @@ PY
|
|
|
248
261
|
for d in "$home/.config" "$cfg" "$home/.claude"; do
|
|
249
262
|
[ -e "$d" ] && chown --reference="$home" "$d" 2>/dev/null || true
|
|
250
263
|
done
|
|
251
|
-
chown --reference="$home" "$cfg/asset-id" \
|
|
264
|
+
chown --reference="$home" "$cfg/asset-id" "$cfg/asset-display-prefix" \
|
|
252
265
|
"$home/.claude/settings.json" 2>/dev/null || true
|
|
253
266
|
|
|
254
267
|
echo "✓ seeded $home"
|
|
@@ -316,10 +316,32 @@ class Built:
|
|
|
316
316
|
document: dict[str, Any]
|
|
317
317
|
|
|
318
318
|
|
|
319
|
+
#: What an advisory answer is about: one component at one version. The
|
|
320
|
+
#: `openaca:identity` this codebase assigns does not encode a version, so it is
|
|
321
|
+
#: paired with the version the component declares. Keyed by identity alone, one
|
|
322
|
+
#: scan kept one version per identity and handed its advisories to every
|
|
323
|
+
#: version of that component in the run.
|
|
324
|
+
AdvisoryKey = tuple[str, "str | None"]
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def advisory_key(component: dict[str, Any]) -> AdvisoryKey | None:
|
|
328
|
+
"""A raw BOM component's `(identity, version)`, or None with no identity.
|
|
329
|
+
|
|
330
|
+
Read off the document, where `Component` reads the same two fields through
|
|
331
|
+
OpenACA's projection; `test_advisory_keys_read_the_same_off_both` holds the
|
|
332
|
+
two equal, since a key that differed would match nothing and read clean.
|
|
333
|
+
"""
|
|
334
|
+
identity = _identity_of(component)
|
|
335
|
+
if identity is None:
|
|
336
|
+
return None
|
|
337
|
+
version = component.get("version")
|
|
338
|
+
return identity, version if isinstance(version, str) else None
|
|
339
|
+
|
|
340
|
+
|
|
319
341
|
def advisories_for(
|
|
320
|
-
documents: Sequence[dict[str, Any]],
|
|
321
|
-
) -> dict[
|
|
322
|
-
"""Advisories for these
|
|
342
|
+
documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
|
|
343
|
+
) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
|
|
344
|
+
"""Advisories for these components alone, keyed by `(identity, version)`.
|
|
323
345
|
|
|
324
346
|
**Asked after correlation, not during acquisition.** Advisory matching costs
|
|
325
347
|
a network round trip to osv.dev per package coordinate, and acquisition does
|
|
@@ -332,61 +354,62 @@ def advisories_for(
|
|
|
332
354
|
there: one scan, over the union of what ran. Same measurement, 0.66s to
|
|
333
355
|
0.14s, and osv.dev learns only about components this machine actually used.
|
|
334
356
|
|
|
335
|
-
Keyed by identity rather than `bom-ref` because a ref is local
|
|
336
|
-
document while
|
|
337
|
-
|
|
357
|
+
Keyed by `(identity, version)` rather than `bom-ref` because a ref is local
|
|
358
|
+
to one document while the pair is what OSV matched — which is what lets one
|
|
359
|
+
scan answer for every composition in the run, and each version for itself.
|
|
338
360
|
|
|
339
361
|
Returns None when the scan could not run. None is not an empty result: a
|
|
340
362
|
composition that was never checked must not read as one with nothing found.
|
|
341
363
|
"""
|
|
342
|
-
if not
|
|
364
|
+
if not keys:
|
|
343
365
|
# Nothing ran that could carry an advisory. That is a checked result with
|
|
344
366
|
# nothing in it, not a failure to look, so it is `{}` rather than None.
|
|
345
367
|
return {}
|
|
346
|
-
merged = _merge_for_scan(documents,
|
|
368
|
+
merged, key_by_ref = _merge_for_scan(documents, keys)
|
|
347
369
|
if not merged["components"]:
|
|
348
370
|
return {}
|
|
349
371
|
with tempfile.TemporaryDirectory() as scratch:
|
|
350
372
|
path = Path(scratch) / "invoked.cdx.json"
|
|
351
373
|
path.write_text(json.dumps(merged), encoding="utf-8")
|
|
352
|
-
return _scan(path,
|
|
374
|
+
return _scan(path, key_by_ref)
|
|
353
375
|
|
|
354
376
|
|
|
355
377
|
def _merge_for_scan(
|
|
356
|
-
documents: Sequence[dict[str, Any]],
|
|
357
|
-
) -> dict[str, Any]:
|
|
358
|
-
"""One document holding each
|
|
359
|
-
|
|
360
|
-
Deduplicated on identity
|
|
361
|
-
built, and scanning it once per appearance is what this
|
|
362
|
-
avoid
|
|
378
|
+
documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
|
|
379
|
+
) -> tuple[dict[str, Any], dict[str, AdvisoryKey]]:
|
|
380
|
+
"""One document holding each wanted component once, and which key each ref is.
|
|
381
|
+
|
|
382
|
+
Deduplicated on `(identity, version)`: the same component appears in every
|
|
383
|
+
composition built, and scanning it once per appearance is what this
|
|
384
|
+
function exists to avoid -- while two versions of one identity are two
|
|
385
|
+
components, each scanned for itself. `dependencies` is dropped rather than filtered — it describes
|
|
363
386
|
installation structure that advisory matching does not read, and a pruned
|
|
364
387
|
graph would be a second, worse description of one already recorded in the
|
|
365
388
|
compositions themselves.
|
|
366
389
|
|
|
367
|
-
**Every kept component's `bom-ref` is rewritten
|
|
368
|
-
CycloneDX spec scopes `bom-ref` uniqueness to a single
|
|
369
|
-
across documents, so two source BOMs are free to reuse the
|
|
370
|
-
for two different components. Carrying an original ref
|
|
371
|
-
merged document could then collide
|
|
372
|
-
the
|
|
373
|
-
|
|
374
|
-
own key, so it is already unique across every document being merged.
|
|
390
|
+
**Every kept component's `bom-ref` is rewritten, and the new ref maps back
|
|
391
|
+
to its key.** The CycloneDX spec scopes `bom-ref` uniqueness to a single
|
|
392
|
+
document, never across documents, so two source BOMs are free to reuse the
|
|
393
|
+
same local ref for two different components. Carrying an original ref
|
|
394
|
+
forward into the merged document could then collide and misattribute that
|
|
395
|
+
ref's advisories to the wrong component. A ref minted here per kept key is
|
|
396
|
+
unique across every document being merged by construction.
|
|
375
397
|
"""
|
|
376
398
|
base = documents[0] if documents else {}
|
|
377
|
-
kept: dict[
|
|
399
|
+
kept: dict[AdvisoryKey, dict[str, Any]] = {}
|
|
378
400
|
for document in documents:
|
|
379
401
|
for component in document.get("components") or []:
|
|
380
|
-
|
|
381
|
-
if
|
|
382
|
-
kept[
|
|
383
|
-
|
|
402
|
+
key = advisory_key(component)
|
|
403
|
+
if key is not None and key in keys and key not in kept:
|
|
404
|
+
kept[key] = {**component, "bom-ref": f"scan-{len(kept)}"}
|
|
405
|
+
merged = {
|
|
384
406
|
"bomFormat": base.get("bomFormat", "CycloneDX"),
|
|
385
407
|
"specVersion": base.get("specVersion", "1.6"),
|
|
386
408
|
"version": base.get("version", 1),
|
|
387
409
|
"metadata": base.get("metadata", {}),
|
|
388
410
|
"components": list(kept.values()),
|
|
389
411
|
}
|
|
412
|
+
return merged, {component["bom-ref"]: key for key, component in kept.items()}
|
|
390
413
|
|
|
391
414
|
|
|
392
415
|
def _identity_of(component: dict[str, Any]) -> str | None:
|
|
@@ -397,40 +420,10 @@ def _identity_of(component: dict[str, Any]) -> str | None:
|
|
|
397
420
|
return None
|
|
398
421
|
|
|
399
422
|
|
|
400
|
-
def identity_versions(
|
|
401
|
-
documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
|
|
402
|
-
) -> dict[str, str | None]:
|
|
403
|
-
"""Each wanted identity's own declared version, the first time it is found.
|
|
404
|
-
|
|
405
|
-
The identity this codebase assigns does not encode a version — that is
|
|
406
|
-
what lets one scan answer for a component across every composition it
|
|
407
|
-
appears in — so a cache keyed on identity alone cannot tell an upgrade
|
|
408
|
-
from no change at all. Pairing identity with this lets a long-lived
|
|
409
|
-
lookup cache do that.
|
|
410
|
-
"""
|
|
411
|
-
versions: dict[str, str | None] = {}
|
|
412
|
-
for document in documents:
|
|
413
|
-
for component in document.get("components") or []:
|
|
414
|
-
identity = _identity_of(component)
|
|
415
|
-
if identity is None or identity not in identities or identity in versions:
|
|
416
|
-
continue
|
|
417
|
-
version = component.get("version")
|
|
418
|
-
versions[identity] = version if isinstance(version, str) else None
|
|
419
|
-
return versions
|
|
420
|
-
|
|
421
|
-
|
|
422
|
-
def _identity_by_ref(document: dict[str, Any]) -> dict[str, str]:
|
|
423
|
-
return {
|
|
424
|
-
str(component["bom-ref"]): identity
|
|
425
|
-
for component in document.get("components") or []
|
|
426
|
-
if component.get("bom-ref") and (identity := _identity_of(component))
|
|
427
|
-
}
|
|
428
|
-
|
|
429
|
-
|
|
430
423
|
def _scan(
|
|
431
|
-
bom_path: Path,
|
|
432
|
-
) -> dict[
|
|
433
|
-
"""Run `openaca scan bom` and re-key its findings from refs onto
|
|
424
|
+
bom_path: Path, key_by_ref: dict[str, AdvisoryKey]
|
|
425
|
+
) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
|
|
426
|
+
"""Run `openaca scan bom` and re-key its findings from refs onto keys.
|
|
434
427
|
|
|
435
428
|
`scan bom` is used rather than `scan endpoint` because it is a pure function
|
|
436
429
|
of a document we already hold, so nothing can drift between the composition
|
|
@@ -483,7 +476,7 @@ def _scan(
|
|
|
483
476
|
if not isinstance(findings, list):
|
|
484
477
|
return None
|
|
485
478
|
|
|
486
|
-
matched: dict[
|
|
479
|
+
matched: dict[AdvisoryKey, list[Advisory]] = {}
|
|
487
480
|
for finding in findings:
|
|
488
481
|
if not isinstance(finding, dict):
|
|
489
482
|
return None
|
|
@@ -505,13 +498,13 @@ def _scan(
|
|
|
505
498
|
return None
|
|
506
499
|
if not all(_text_or_absent(value) for value in (fixed_in, source, severity, title)):
|
|
507
500
|
return None
|
|
508
|
-
|
|
509
|
-
if
|
|
501
|
+
key = key_by_ref.get(ref)
|
|
502
|
+
if key is None:
|
|
510
503
|
# Not malformed: we sent the components, and a ref we do not hold is
|
|
511
504
|
# a finding about something outside this scan rather than a response
|
|
512
505
|
# we cannot read.
|
|
513
506
|
continue
|
|
514
|
-
matched.setdefault(
|
|
507
|
+
matched.setdefault(key, []).append(
|
|
515
508
|
Advisory(
|
|
516
509
|
id=identifier,
|
|
517
510
|
severity=severity or "UNKNOWN",
|
|
@@ -520,7 +513,7 @@ def _scan(
|
|
|
520
513
|
source=source,
|
|
521
514
|
)
|
|
522
515
|
)
|
|
523
|
-
return {
|
|
516
|
+
return {key: tuple(items) for key, items in matched.items()}
|
|
524
517
|
|
|
525
518
|
|
|
526
519
|
def _text_or_absent(value: object) -> bool:
|
|
@@ -152,10 +152,10 @@ def composition_from_bom(
|
|
|
152
152
|
generated_at: datetime,
|
|
153
153
|
generated_at_is_observed: bool,
|
|
154
154
|
#: Keyed by **bom-ref**, matching `Composition.advisories`. Note that
|
|
155
|
-
#: `acquire.advisories_for` returns an
|
|
156
|
-
#: answer for every composition that way — so a caller re-keys
|
|
157
|
-
#: arriving here, as `_attach_advisories` does. Passing
|
|
158
|
-
#:
|
|
155
|
+
#: `acquire.advisories_for` returns an `(identity, version)`-keyed map — one
|
|
156
|
+
#: scan can answer for every composition that way — so a caller re-keys
|
|
157
|
+
#: before arriving here, as `_attach_advisories` does. Passing that map
|
|
158
|
+
#: straight through would silently match nothing.
|
|
159
159
|
advisories: dict[str, tuple[Advisory, ...]] | None = None,
|
|
160
160
|
) -> Composition:
|
|
161
161
|
"""Project a CycloneDX Agent BOM into a composition, or reject it."""
|
|
@@ -1381,9 +1381,13 @@ _SEPARATORS = ("&&", "||", ";", "|", "\n")
|
|
|
1381
1381
|
_WORD_START_BEFORE_COMMENT = frozenset(" \t\n;&|<>()")
|
|
1382
1382
|
|
|
1383
1383
|
|
|
1384
|
-
def _split_on_separators(command: str) -> list[str]:
|
|
1384
|
+
def _split_on_separators(command: str) -> list[tuple[str, str | None]]:
|
|
1385
1385
|
"""The commands in a shell line, split only where the shell would split it.
|
|
1386
1386
|
|
|
1387
|
+
Returns ``(segment_text, preceding_operator)`` pairs. The first segment's
|
|
1388
|
+
operator is ``None``; every later segment carries the ``&&``, ``||``, ``;``,
|
|
1389
|
+
``|`` or ``\\n`` that separated it from the one before it.
|
|
1390
|
+
|
|
1387
1391
|
A control operator is an operator **only unquoted and unescaped** — that is
|
|
1388
1392
|
the shell's own grammar, and it is verified against `bash` itself in
|
|
1389
1393
|
`tests/test_shell_separators_match_the_shell.py` rather than asserted here.
|
|
@@ -1420,10 +1424,11 @@ def _split_on_separators(command: str) -> list[str]:
|
|
|
1420
1424
|
left in, `shlex` hands the caller a bare newline where the next word should
|
|
1421
1425
|
be, and `git \\<newline> push` loses its subcommand to it.
|
|
1422
1426
|
"""
|
|
1423
|
-
parts: list[str] = []
|
|
1427
|
+
parts: list[tuple[str, str | None]] = []
|
|
1424
1428
|
current: list[str] = []
|
|
1425
1429
|
index = 0
|
|
1426
1430
|
quote: str | None = None
|
|
1431
|
+
preceding_op: str | None = None
|
|
1427
1432
|
#: Whether a `#` here would open a comment. True at the start of the string
|
|
1428
1433
|
#: and after anything that ends a word; false after any character a word is
|
|
1429
1434
|
#: made of, quotes included -- `echo "a"#b` prints `a#b`.
|
|
@@ -1475,11 +1480,20 @@ def _split_on_separators(command: str) -> list[str]:
|
|
|
1475
1480
|
index += 1
|
|
1476
1481
|
at_word_start = char in _WORD_START_BEFORE_COMMENT
|
|
1477
1482
|
continue
|
|
1478
|
-
|
|
1483
|
+
segment_text = "".join(current)
|
|
1484
|
+
if segment_text.strip():
|
|
1485
|
+
parts.append((segment_text, preceding_op))
|
|
1486
|
+
preceding_op = separator
|
|
1487
|
+
else:
|
|
1488
|
+
# Two separators with nothing between them (e.g. "&&\n"):
|
|
1489
|
+
# keep the conditional operator over a plain terminator so
|
|
1490
|
+
# the next real segment inherits the right dependency.
|
|
1491
|
+
if separator in ("&&", "||") or preceding_op is None:
|
|
1492
|
+
preceding_op = separator
|
|
1479
1493
|
current = []
|
|
1480
1494
|
index += len(separator)
|
|
1481
1495
|
at_word_start = True
|
|
1482
|
-
parts.append("".join(current))
|
|
1496
|
+
parts.append(("".join(current), preceding_op))
|
|
1483
1497
|
return parts
|
|
1484
1498
|
|
|
1485
1499
|
|
|
@@ -1521,7 +1535,7 @@ def _segments_of(text: str, depth: int) -> tuple[tuple[str, tuple[str, ...]], ..
|
|
|
1521
1535
|
`command_segments` would run the masks a second time on their own output.
|
|
1522
1536
|
"""
|
|
1523
1537
|
found: list[tuple[str, tuple[str, ...]]] = []
|
|
1524
|
-
for part in _split_on_separators(text):
|
|
1538
|
+
for part, _op in _split_on_separators(text):
|
|
1525
1539
|
# Comments are gone: `_split_on_separators` discards one at the word
|
|
1526
1540
|
# start that opens it, which is the only place the shell begins one.
|
|
1527
1541
|
# This used to skip a segment *beginning* with `#`, which caught the
|
|
@@ -1567,6 +1581,48 @@ def _segments_of(text: str, depth: int) -> tuple[tuple[str, tuple[str, ...]], ..
|
|
|
1567
1581
|
return tuple(found)
|
|
1568
1582
|
|
|
1569
1583
|
|
|
1584
|
+
def command_segments_with_ops(
|
|
1585
|
+
command: object, depth: int = 0
|
|
1586
|
+
) -> tuple[tuple[str, tuple[str, ...], str | None], ...]:
|
|
1587
|
+
"""Like `command_segments`, but each entry also carries the shell operator
|
|
1588
|
+
that preceded it (``None`` for the first segment, then ``&&``, ``||``,
|
|
1589
|
+
``;``, ``|`` or ``\\n``).
|
|
1590
|
+
|
|
1591
|
+
Only `_resolve_rm_targets` in the deterministic detector needs the operator
|
|
1592
|
+
context — to decide whether an inline ``cd`` reliably changed the directory
|
|
1593
|
+
and whether an ``rm`` segment is unconditionally reachable. Every other
|
|
1594
|
+
caller uses the plain `command_segments`, which discards the operator.
|
|
1595
|
+
"""
|
|
1596
|
+
if not isinstance(command, str) or not command:
|
|
1597
|
+
return ()
|
|
1598
|
+
text = _readable_commands(command, depth)
|
|
1599
|
+
found: list[tuple[str, tuple[str, ...], str | None]] = []
|
|
1600
|
+
for part, preceding_op in _split_on_separators(text):
|
|
1601
|
+
part = part.strip()
|
|
1602
|
+
if not part:
|
|
1603
|
+
continue
|
|
1604
|
+
try:
|
|
1605
|
+
tokens = shlex.split(part)
|
|
1606
|
+
except ValueError:
|
|
1607
|
+
continue
|
|
1608
|
+
while tokens and (_ASSIGNMENT.match(tokens[0]) or tokens[0] in _NOT_A_BINARY):
|
|
1609
|
+
if tokens.pop(0) == "time":
|
|
1610
|
+
while tokens and tokens[0] in _TIME_OPTIONS:
|
|
1611
|
+
tokens.pop(0)
|
|
1612
|
+
if not tokens:
|
|
1613
|
+
continue
|
|
1614
|
+
head = tokens[0].rsplit("/", 1)[-1]
|
|
1615
|
+
if not head or all(char in _OPERATOR_CHARS for char in head):
|
|
1616
|
+
continue
|
|
1617
|
+
arguments = tuple(tokens[1:])
|
|
1618
|
+
found.append((head, arguments, preceding_op))
|
|
1619
|
+
if head in _SHELLS and depth < _MAX_SHELL_DEPTH:
|
|
1620
|
+
body = _shell_command_body(arguments)
|
|
1621
|
+
if body is not None:
|
|
1622
|
+
found.extend(command_segments_with_ops(body, depth + 1))
|
|
1623
|
+
return tuple(found)
|
|
1624
|
+
|
|
1625
|
+
|
|
1570
1626
|
def redirect_targets(command: object) -> tuple[str, ...]:
|
|
1571
1627
|
"""Files a shell command redirects its output into.
|
|
1572
1628
|
|
|
@@ -1604,7 +1660,7 @@ def redirect_targets(command: object) -> tuple[str, ...]:
|
|
|
1604
1660
|
# body, an arithmetic comparison, a `case` frame or the body of a function
|
|
1605
1661
|
# nothing called is not a redirection, and each was named as a file before
|
|
1606
1662
|
# its mask existed.
|
|
1607
|
-
for part in _split_on_separators(_readable_commands(command, 0)):
|
|
1663
|
+
for part, _op in _split_on_separators(_readable_commands(command, 0)):
|
|
1608
1664
|
part = part.strip()
|
|
1609
1665
|
if not part:
|
|
1610
1666
|
continue
|