stacktrace-cli 0.5.3__py3-none-any.whl → 0.6.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. stacktrace_cli/__init__.py +1 -1
  2. stacktrace_cli/analysis.py +33 -31
  3. stacktrace_cli/cli.py +5 -0
  4. stacktrace_cli/cloud/__init__.py +10 -0
  5. stacktrace_cli/cloud/setup.sh +17 -4
  6. stacktrace_cli/correlate/acquire.py +60 -67
  7. stacktrace_cli/correlate/composition.py +4 -4
  8. stacktrace_cli/correlate/observed.py +62 -6
  9. stacktrace_cli/correlate/orchestrate.py +30 -38
  10. stacktrace_cli/daemon/cli.py +93 -2
  11. stacktrace_cli/daemon/config.py +85 -5
  12. stacktrace_cli/daemon/identity.py +17 -1
  13. stacktrace_cli/daemon/jobs.py +48 -1
  14. stacktrace_cli/daemon/observer.py +31 -2
  15. stacktrace_cli/daemon/presentation.py +34 -1
  16. stacktrace_cli/daemon/runtime.py +25 -11
  17. stacktrace_cli/daemon/store.py +60 -15
  18. stacktrace_cli/daemon/webhook.py +34 -4
  19. stacktrace_cli/detector/blocked.py +12 -0
  20. stacktrace_cli/detector/deterministic.py +1935 -27
  21. stacktrace_cli/detector/render.py +6 -9
  22. stacktrace_cli/detector/rules.py +74 -12
  23. stacktrace_cli/detector/run.py +11 -3
  24. stacktrace_cli/monitor/inventory.py +495 -0
  25. stacktrace_cli/monitor/reasoning.py +8 -6
  26. stacktrace_cli/monitor/render.py +49 -1
  27. stacktrace_cli/monitor/server.py +44 -20
  28. stacktrace_cli/monitor/site/app.js +386 -369
  29. stacktrace_cli/monitor/site/index.html +26 -14
  30. stacktrace_cli/monitor/site/inventory.js +554 -0
  31. stacktrace_cli/monitor/site/styles.css +139 -51
  32. stacktrace_cli/monitor/state.py +45 -2
  33. stacktrace_cli/monitor/watch.py +294 -129
  34. stacktrace_cli/remote/identity.py +41 -12
  35. stacktrace_cli/remote/payload.py +10 -5
  36. stacktrace_cli/remote/redact.py +100 -2
  37. stacktrace_cli/remote/sync.py +11 -0
  38. stacktrace_cli/telemetry/events.py +12 -30
  39. stacktrace_cli/webhook/cli.py +35 -4
  40. {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/METADATA +3 -3
  41. {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/RECORD +43 -41
  42. {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/WHEEL +0 -0
  43. {stacktrace_cli-0.5.3.dist-info → stacktrace_cli-0.6.1.dist-info}/entry_points.txt +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  import logging as _logging
4
4
 
5
- __version__ = "0.5.3"
5
+ __version__ = "0.6.1"
6
6
 
7
7
  # Only the daemon attaches a handler (ADR-0048). Without this one, a WARNING
8
8
  # logged in any other process would reach Python's last-resort handler and
@@ -348,32 +348,6 @@ def _acquire(
348
348
  )
349
349
 
350
350
 
351
- def _correlated(
352
- *,
353
- agent_kinds: tuple[str, ...],
354
- window_start: datetime,
355
- bom_paths: tuple[Path, ...],
356
- project_map: tuple[str, ...],
357
- root: Path | None,
358
- session_ids: tuple[str, ...],
359
- attach_advisories: Any,
360
- build_all: Any = None,
361
- ) -> Acquired:
362
- """Everything up to the judging, shared by both entry points."""
363
- return _acquire(
364
- _collected(
365
- agent_kinds=agent_kinds,
366
- window_start=window_start,
367
- root=root,
368
- session_ids=session_ids,
369
- ),
370
- bom_paths=bom_paths,
371
- project_map=project_map,
372
- attach_advisories=attach_advisories,
373
- build_all=build_all,
374
- )
375
-
376
-
377
351
  def _reaches(detection: Any, window_start: datetime, window_end: datetime) -> bool:
378
352
  """Whether this finding's own span overlaps the window.
379
353
 
@@ -482,6 +456,7 @@ def analyse_progressively(
482
456
  history: DriftHistory | None = None,
483
457
  reasoning: bool = False,
484
458
  budget: int = DEFAULT_BUDGET,
459
+ exclude_from_reasoning: frozenset[tuple[str, str]] = frozenset(),
485
460
  ) -> Iterator[Analysis]:
486
461
  """The same pipeline, delivered in instalments.
487
462
 
@@ -634,12 +609,20 @@ def analyse_progressively(
634
609
  if not reasoning or not ordered:
635
610
  return
636
611
 
637
- # One more instalment, one run, one budget. Everything above has already
638
- # been published, so the page is complete and readable while this is in
639
- # flight — on a large window it is several hundred sequential requests and
640
- # holding the first paint behind it is what made the page look hung.
612
+ excluded_count = 0
613
+ if exclude_from_reasoning:
614
+ kept = tuple(
615
+ s
616
+ for s in view.sessions
617
+ if (s.session.session_id, s.session.agent_kind) not in exclude_from_reasoning
618
+ )
619
+ excluded_count = len(view.sessions) - len(kept)
620
+ reasoning_view = replace(view, sessions=kept) if excluded_count else view
621
+ else:
622
+ reasoning_view = view
623
+
641
624
  judged = run_detector(
642
- view,
625
+ reasoning_view,
643
626
  reasoning=True,
644
627
  budget=budget,
645
628
  cache=cache,
@@ -647,4 +630,23 @@ def analyse_progressively(
647
630
  jev_key=jev_key,
648
631
  history=history,
649
632
  )
633
+
634
+ if excluded_count:
635
+ extra_detections = tuple(
636
+ d
637
+ for d in accumulated.detections
638
+ if (d.session.session_id, d.session.agent_kind) in exclude_from_reasoning
639
+ )
640
+ extra_unknowns = tuple(
641
+ u
642
+ for u in accumulated.unknowns
643
+ if (u.session.session_id, u.session.agent_kind) in exclude_from_reasoning
644
+ )
645
+ judged = replace(
646
+ judged,
647
+ detections=judged.detections + extra_detections,
648
+ unknowns=judged.unknowns + extra_unknowns,
649
+ sessions=judged.sessions + excluded_count,
650
+ )
651
+
650
652
  yield _stage(judged, tuple(ordered))
stacktrace_cli/cli.py CHANGED
@@ -539,6 +539,11 @@ def monitor(
539
539
  Loopback only, and free to leave open: a pass is skipped when nothing has
540
540
  changed, advisory lookups are asked once per component, and the two stages
541
541
  that need no model are the only ones that run.
542
+
543
+ The Inventory and Vulnerabilities tabs list every component your agents can
544
+ load. To check them for known vulnerabilities, the package names of
545
+ installed components -- not only the ones a session used -- are sent to
546
+ osv.dev. Skills and plugins with no package are never sent.
542
547
  """
543
548
  analyzer = resolve_analyzer(analyzer, reasoning)
544
549
  try:
@@ -12,6 +12,16 @@ CLI and runs this:
12
12
  uv tool install stacktrace-cli
13
13
  stacktrace-cloud-setup
14
14
 
15
+ Or, equivalently, one hosted line that wraps exactly those two steps (see
16
+ `stacktrace-site`'s `public/remote-install.sh`). Download before running rather
17
+ than piping directly into `sh`: without `pipefail` (the default in the `sh`
18
+ that a setup-script box runs), a failed `curl` still lets `sh` read an empty
19
+ script and exit 0, so a broken download would otherwise snapshot an
20
+ environment with neither the CLI nor the boot hook and report success:
21
+
22
+ curl -fsSL https://stacktrace.ai/remote-install.sh -o /tmp/remote-install.sh \
23
+ && sh /tmp/remote-install.sh
24
+
15
25
  Both variables matter: setup runs as root, and `uv tool install`'s root
16
26
  defaults put the tool's environment under `/root/.local/share/uv/tools` and the
17
27
  linked executable under `/root/.local/bin` — both inside `/root`, which a later
@@ -10,7 +10,10 @@
10
10
  # every home so it fires whichever user the session runs as.
11
11
  # 2. The asset identity, in every home's ~/.config/stacktrace/asset-id, so
12
12
  # `stacktrace remote sync` resolves it at the persisted rung (ADR-0065)
13
- # instead of minting a fresh asset per session.
13
+ # instead of minting a fresh asset per session. Beside it,
14
+ # asset-display-prefix makes the fleet label `cloud (<id>)` rather than
15
+ # the image's hostname (`vm`), which is the same on every box. Label
16
+ # only: the key above is what identifies the asset.
14
17
  #
15
18
  # The telemetry install id is deliberately NOT seeded here. Seeding it would
16
19
  # make `read_install_id()` non-None before any event runs, so the one-time
@@ -59,7 +62,7 @@ while [ $# -gt 0 ]; do
59
62
  # the loop spins forever — a malformed invocation would hang setup.
60
63
  --asset-id) [ $# -ge 2 ] || { echo "✗ --asset-id needs a value" >&2; exit 2; }; ASSET_ID="$2"; shift 2 ;;
61
64
  --root) [ $# -ge 2 ] || { echo "✗ --root needs a value" >&2; exit 2; }; ROOT="$2"; shift 2 ;;
62
- -h|--help) sed -n '2,49p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
65
+ -h|--help) sed -n '2,52p' "$0" | sed 's/^# \{0,1\}//'; exit 0 ;;
63
66
  *) echo "✗ unknown argument: $1" >&2; exit 2 ;;
64
67
  esac
65
68
  done
@@ -78,6 +81,7 @@ _uuid() {
78
81
  fi
79
82
  }
80
83
 
84
+ DISPLAY_PREFIX="cloud"
81
85
  [ -n "$ASSET_ID" ] || ASSET_ID="cloud-$(_uuid)"
82
86
  # remote/identity.py:_is_identifier rejects whitespace, non-printable
83
87
  # characters, and anything over 255 characters. A value this accepts but the
@@ -146,8 +150,12 @@ fi
146
150
 
147
151
  # Streams redirected because own_std_streams only re-points fd 1/2 on log
148
152
  # rollover; without this the daemon inherits the hook's pipe to Claude Code.
153
+ # --no-reasoning is explicit, not just `run`'s own default (ADR-0081): this
154
+ # hook never provisions TYPESAFE_API_KEY itself, but a cloud VM's ambient
155
+ # environment is not something this script controls or can audit ahead of
156
+ # time, and this surface has nobody present to notice if that changed.
149
157
  if ! stacktrace daemon status 2>/dev/null | grep -q '^running: yes'; then
150
- setsid nohup stacktrace daemon run </dev/null >>"$STATE/daemon.log" 2>&1 &
158
+ setsid nohup stacktrace daemon run --no-reasoning </dev/null >>"$STATE/daemon.log" 2>&1 &
151
159
  fi
152
160
  exit 0
153
161
  BOOTSCRIPT
@@ -176,6 +184,11 @@ for home in "${ROOT}"/root "${ROOT}"/home/*; do
176
184
  fi
177
185
  chmod 600 "$cfg/asset-id" 2>/dev/null || true
178
186
 
187
+ if ! printf '%s\n' "$DISPLAY_PREFIX" > "$cfg/asset-display-prefix" 2>/dev/null; then
188
+ echo "✗ cannot write $cfg/asset-display-prefix — not seeding $home" >&2; failed=1; continue
189
+ fi
190
+ chmod 600 "$cfg/asset-display-prefix" 2>/dev/null || true
191
+
179
192
  # Register the SessionStart hook, merging so any existing settings survive.
180
193
  # The hook is the point of seeding a home — a home whose registration fails
181
194
  # (an unwritable path, or a settings.json that is valid JSON of an unexpected
@@ -248,7 +261,7 @@ PY
248
261
  for d in "$home/.config" "$cfg" "$home/.claude"; do
249
262
  [ -e "$d" ] && chown --reference="$home" "$d" 2>/dev/null || true
250
263
  done
251
- chown --reference="$home" "$cfg/asset-id" \
264
+ chown --reference="$home" "$cfg/asset-id" "$cfg/asset-display-prefix" \
252
265
  "$home/.claude/settings.json" 2>/dev/null || true
253
266
 
254
267
  echo "✓ seeded $home"
@@ -316,10 +316,32 @@ class Built:
316
316
  document: dict[str, Any]
317
317
 
318
318
 
319
+ #: What an advisory answer is about: one component at one version. The
320
+ #: `openaca:identity` this codebase assigns does not encode a version, so it is
321
+ #: paired with the version the component declares. Keyed by identity alone, one
322
+ #: scan kept one version per identity and handed its advisories to every
323
+ #: version of that component in the run.
324
+ AdvisoryKey = tuple[str, "str | None"]
325
+
326
+
327
+ def advisory_key(component: dict[str, Any]) -> AdvisoryKey | None:
328
+ """A raw BOM component's `(identity, version)`, or None with no identity.
329
+
330
+ Read off the document, where `Component` reads the same two fields through
331
+ OpenACA's projection; `test_advisory_keys_read_the_same_off_both` holds the
332
+ two equal, since a key that differed would match nothing and read clean.
333
+ """
334
+ identity = _identity_of(component)
335
+ if identity is None:
336
+ return None
337
+ version = component.get("version")
338
+ return identity, version if isinstance(version, str) else None
339
+
340
+
319
341
  def advisories_for(
320
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
321
- ) -> dict[str, tuple[Advisory, ...]] | None:
322
- """Advisories for these identities alone, keyed by `openaca:identity`.
342
+ documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
343
+ ) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
344
+ """Advisories for these components alone, keyed by `(identity, version)`.
323
345
 
324
346
  **Asked after correlation, not during acquisition.** Advisory matching costs
325
347
  a network round trip to osv.dev per package coordinate, and acquisition does
@@ -332,61 +354,62 @@ def advisories_for(
332
354
  there: one scan, over the union of what ran. Same measurement, 0.66s to
333
355
  0.14s, and osv.dev learns only about components this machine actually used.
334
356
 
335
- Keyed by identity rather than `bom-ref` because a ref is local to one
336
- document while an identity is the coordinate OSV matched — which is what lets
337
- one scan answer for every composition in the run.
357
+ Keyed by `(identity, version)` rather than `bom-ref` because a ref is local
358
+ to one document while the pair is what OSV matched — which is what lets one
359
+ scan answer for every composition in the run, and each version for itself.
338
360
 
339
361
  Returns None when the scan could not run. None is not an empty result: a
340
362
  composition that was never checked must not read as one with nothing found.
341
363
  """
342
- if not identities:
364
+ if not keys:
343
365
  # Nothing ran that could carry an advisory. That is a checked result with
344
366
  # nothing in it, not a failure to look, so it is `{}` rather than None.
345
367
  return {}
346
- merged = _merge_for_scan(documents, identities)
368
+ merged, key_by_ref = _merge_for_scan(documents, keys)
347
369
  if not merged["components"]:
348
370
  return {}
349
371
  with tempfile.TemporaryDirectory() as scratch:
350
372
  path = Path(scratch) / "invoked.cdx.json"
351
373
  path.write_text(json.dumps(merged), encoding="utf-8")
352
- return _scan(path, _identity_by_ref(merged))
374
+ return _scan(path, key_by_ref)
353
375
 
354
376
 
355
377
  def _merge_for_scan(
356
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
357
- ) -> dict[str, Any]:
358
- """One document holding each invoked component once, and nothing else.
359
-
360
- Deduplicated on identity: the same component appears in every composition
361
- built, and scanning it once per appearance is what this function exists to
362
- avoid. `dependencies` is dropped rather than filtered — it describes
378
+ documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
379
+ ) -> tuple[dict[str, Any], dict[str, AdvisoryKey]]:
380
+ """One document holding each wanted component once, and which key each ref is.
381
+
382
+ Deduplicated on `(identity, version)`: the same component appears in every
383
+ composition built, and scanning it once per appearance is what this
384
+ function exists to avoid -- while two versions of one identity are two
385
+ components, each scanned for itself. `dependencies` is dropped rather than filtered — it describes
363
386
  installation structure that advisory matching does not read, and a pruned
364
387
  graph would be a second, worse description of one already recorded in the
365
388
  compositions themselves.
366
389
 
367
- **Every kept component's `bom-ref` is rewritten to its identity.** The
368
- CycloneDX spec scopes `bom-ref` uniqueness to a single document, never
369
- across documents, so two source BOMs are free to reuse the same local ref
370
- for two different components. Carrying an original ref forward into the
371
- merged document could then collide, and `_identity_by_ref` would collapse
372
- the collision onto whichever identity it saw last — misattributing that
373
- ref's advisories to the wrong component. Identity is already this dict's
374
- own key, so it is already unique across every document being merged.
390
+ **Every kept component's `bom-ref` is rewritten, and the new ref maps back
391
+ to its key.** The CycloneDX spec scopes `bom-ref` uniqueness to a single
392
+ document, never across documents, so two source BOMs are free to reuse the
393
+ same local ref for two different components. Carrying an original ref
394
+ forward into the merged document could then collide and misattribute that
395
+ ref's advisories to the wrong component. A ref minted here per kept key is
396
+ unique across every document being merged by construction.
375
397
  """
376
398
  base = documents[0] if documents else {}
377
- kept: dict[str, dict[str, Any]] = {}
399
+ kept: dict[AdvisoryKey, dict[str, Any]] = {}
378
400
  for document in documents:
379
401
  for component in document.get("components") or []:
380
- identity = _identity_of(component)
381
- if identity is not None and identity in identities and identity not in kept:
382
- kept[identity] = {**component, "bom-ref": identity}
383
- return {
402
+ key = advisory_key(component)
403
+ if key is not None and key in keys and key not in kept:
404
+ kept[key] = {**component, "bom-ref": f"scan-{len(kept)}"}
405
+ merged = {
384
406
  "bomFormat": base.get("bomFormat", "CycloneDX"),
385
407
  "specVersion": base.get("specVersion", "1.6"),
386
408
  "version": base.get("version", 1),
387
409
  "metadata": base.get("metadata", {}),
388
410
  "components": list(kept.values()),
389
411
  }
412
+ return merged, {component["bom-ref"]: key for key, component in kept.items()}
390
413
 
391
414
 
392
415
  def _identity_of(component: dict[str, Any]) -> str | None:
@@ -397,40 +420,10 @@ def _identity_of(component: dict[str, Any]) -> str | None:
397
420
  return None
398
421
 
399
422
 
400
- def identity_versions(
401
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
402
- ) -> dict[str, str | None]:
403
- """Each wanted identity's own declared version, the first time it is found.
404
-
405
- The identity this codebase assigns does not encode a version — that is
406
- what lets one scan answer for a component across every composition it
407
- appears in — so a cache keyed on identity alone cannot tell an upgrade
408
- from no change at all. Pairing identity with this lets a long-lived
409
- lookup cache do that.
410
- """
411
- versions: dict[str, str | None] = {}
412
- for document in documents:
413
- for component in document.get("components") or []:
414
- identity = _identity_of(component)
415
- if identity is None or identity not in identities or identity in versions:
416
- continue
417
- version = component.get("version")
418
- versions[identity] = version if isinstance(version, str) else None
419
- return versions
420
-
421
-
422
- def _identity_by_ref(document: dict[str, Any]) -> dict[str, str]:
423
- return {
424
- str(component["bom-ref"]): identity
425
- for component in document.get("components") or []
426
- if component.get("bom-ref") and (identity := _identity_of(component))
427
- }
428
-
429
-
430
423
  def _scan(
431
- bom_path: Path, identity_by_ref: dict[str, str]
432
- ) -> dict[str, tuple[Advisory, ...]] | None:
433
- """Run `openaca scan bom` and re-key its findings from refs onto identities.
424
+ bom_path: Path, key_by_ref: dict[str, AdvisoryKey]
425
+ ) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
426
+ """Run `openaca scan bom` and re-key its findings from refs onto keys.
434
427
 
435
428
  `scan bom` is used rather than `scan endpoint` because it is a pure function
436
429
  of a document we already hold, so nothing can drift between the composition
@@ -483,7 +476,7 @@ def _scan(
483
476
  if not isinstance(findings, list):
484
477
  return None
485
478
 
486
- matched: dict[str, list[Advisory]] = {}
479
+ matched: dict[AdvisoryKey, list[Advisory]] = {}
487
480
  for finding in findings:
488
481
  if not isinstance(finding, dict):
489
482
  return None
@@ -505,13 +498,13 @@ def _scan(
505
498
  return None
506
499
  if not all(_text_or_absent(value) for value in (fixed_in, source, severity, title)):
507
500
  return None
508
- identity = identity_by_ref.get(ref)
509
- if identity is None:
501
+ key = key_by_ref.get(ref)
502
+ if key is None:
510
503
  # Not malformed: we sent the components, and a ref we do not hold is
511
504
  # a finding about something outside this scan rather than a response
512
505
  # we cannot read.
513
506
  continue
514
- matched.setdefault(identity, []).append(
507
+ matched.setdefault(key, []).append(
515
508
  Advisory(
516
509
  id=identifier,
517
510
  severity=severity or "UNKNOWN",
@@ -520,7 +513,7 @@ def _scan(
520
513
  source=source,
521
514
  )
522
515
  )
523
- return {identity: tuple(items) for identity, items in matched.items()}
516
+ return {key: tuple(items) for key, items in matched.items()}
524
517
 
525
518
 
526
519
  def _text_or_absent(value: object) -> bool:
@@ -152,10 +152,10 @@ def composition_from_bom(
152
152
  generated_at: datetime,
153
153
  generated_at_is_observed: bool,
154
154
  #: Keyed by **bom-ref**, matching `Composition.advisories`. Note that
155
- #: `acquire.advisories_for` returns an *identity*-keyed map — one scan can
156
- #: answer for every composition that way — so a caller re-keys before
157
- #: arriving here, as `_attach_advisories` does. Passing the identity-keyed
158
- #: map straight through would silently match nothing.
155
+ #: `acquire.advisories_for` returns an `(identity, version)`-keyed map — one
156
+ #: scan can answer for every composition that way — so a caller re-keys
157
+ #: before arriving here, as `_attach_advisories` does. Passing that map
158
+ #: straight through would silently match nothing.
159
159
  advisories: dict[str, tuple[Advisory, ...]] | None = None,
160
160
  ) -> Composition:
161
161
  """Project a CycloneDX Agent BOM into a composition, or reject it."""
@@ -1381,9 +1381,13 @@ _SEPARATORS = ("&&", "||", ";", "|", "\n")
1381
1381
  _WORD_START_BEFORE_COMMENT = frozenset(" \t\n;&|<>()")
1382
1382
 
1383
1383
 
1384
- def _split_on_separators(command: str) -> list[str]:
1384
+ def _split_on_separators(command: str) -> list[tuple[str, str | None]]:
1385
1385
  """The commands in a shell line, split only where the shell would split it.
1386
1386
 
1387
+ Returns ``(segment_text, preceding_operator)`` pairs. The first segment's
1388
+ operator is ``None``; every later segment carries the ``&&``, ``||``, ``;``,
1389
+ ``|`` or ``\\n`` that separated it from the one before it.
1390
+
1387
1391
  A control operator is an operator **only unquoted and unescaped** — that is
1388
1392
  the shell's own grammar, and it is verified against `bash` itself in
1389
1393
  `tests/test_shell_separators_match_the_shell.py` rather than asserted here.
@@ -1420,10 +1424,11 @@ def _split_on_separators(command: str) -> list[str]:
1420
1424
  left in, `shlex` hands the caller a bare newline where the next word should
1421
1425
  be, and `git \\<newline> push` loses its subcommand to it.
1422
1426
  """
1423
- parts: list[str] = []
1427
+ parts: list[tuple[str, str | None]] = []
1424
1428
  current: list[str] = []
1425
1429
  index = 0
1426
1430
  quote: str | None = None
1431
+ preceding_op: str | None = None
1427
1432
  #: Whether a `#` here would open a comment. True at the start of the string
1428
1433
  #: and after anything that ends a word; false after any character a word is
1429
1434
  #: made of, quotes included -- `echo "a"#b` prints `a#b`.
@@ -1475,11 +1480,20 @@ def _split_on_separators(command: str) -> list[str]:
1475
1480
  index += 1
1476
1481
  at_word_start = char in _WORD_START_BEFORE_COMMENT
1477
1482
  continue
1478
- parts.append("".join(current))
1483
+ segment_text = "".join(current)
1484
+ if segment_text.strip():
1485
+ parts.append((segment_text, preceding_op))
1486
+ preceding_op = separator
1487
+ else:
1488
+ # Two separators with nothing between them (e.g. "&&\n"):
1489
+ # keep the conditional operator over a plain terminator so
1490
+ # the next real segment inherits the right dependency.
1491
+ if separator in ("&&", "||") or preceding_op is None:
1492
+ preceding_op = separator
1479
1493
  current = []
1480
1494
  index += len(separator)
1481
1495
  at_word_start = True
1482
- parts.append("".join(current))
1496
+ parts.append(("".join(current), preceding_op))
1483
1497
  return parts
1484
1498
 
1485
1499
 
@@ -1521,7 +1535,7 @@ def _segments_of(text: str, depth: int) -> tuple[tuple[str, tuple[str, ...]], ..
1521
1535
  `command_segments` would run the masks a second time on their own output.
1522
1536
  """
1523
1537
  found: list[tuple[str, tuple[str, ...]]] = []
1524
- for part in _split_on_separators(text):
1538
+ for part, _op in _split_on_separators(text):
1525
1539
  # Comments are gone: `_split_on_separators` discards one at the word
1526
1540
  # start that opens it, which is the only place the shell begins one.
1527
1541
  # This used to skip a segment *beginning* with `#`, which caught the
@@ -1567,6 +1581,48 @@ def _segments_of(text: str, depth: int) -> tuple[tuple[str, tuple[str, ...]], ..
1567
1581
  return tuple(found)
1568
1582
 
1569
1583
 
1584
+ def command_segments_with_ops(
1585
+ command: object, depth: int = 0
1586
+ ) -> tuple[tuple[str, tuple[str, ...], str | None], ...]:
1587
+ """Like `command_segments`, but each entry also carries the shell operator
1588
+ that preceded it (``None`` for the first segment, then ``&&``, ``||``,
1589
+ ``;``, ``|`` or ``\\n``).
1590
+
1591
+ Only `_resolve_rm_targets` in the deterministic detector needs the operator
1592
+ context — to decide whether an inline ``cd`` reliably changed the directory
1593
+ and whether an ``rm`` segment is unconditionally reachable. Every other
1594
+ caller uses the plain `command_segments`, which discards the operator.
1595
+ """
1596
+ if not isinstance(command, str) or not command:
1597
+ return ()
1598
+ text = _readable_commands(command, depth)
1599
+ found: list[tuple[str, tuple[str, ...], str | None]] = []
1600
+ for part, preceding_op in _split_on_separators(text):
1601
+ part = part.strip()
1602
+ if not part:
1603
+ continue
1604
+ try:
1605
+ tokens = shlex.split(part)
1606
+ except ValueError:
1607
+ continue
1608
+ while tokens and (_ASSIGNMENT.match(tokens[0]) or tokens[0] in _NOT_A_BINARY):
1609
+ if tokens.pop(0) == "time":
1610
+ while tokens and tokens[0] in _TIME_OPTIONS:
1611
+ tokens.pop(0)
1612
+ if not tokens:
1613
+ continue
1614
+ head = tokens[0].rsplit("/", 1)[-1]
1615
+ if not head or all(char in _OPERATOR_CHARS for char in head):
1616
+ continue
1617
+ arguments = tuple(tokens[1:])
1618
+ found.append((head, arguments, preceding_op))
1619
+ if head in _SHELLS and depth < _MAX_SHELL_DEPTH:
1620
+ body = _shell_command_body(arguments)
1621
+ if body is not None:
1622
+ found.extend(command_segments_with_ops(body, depth + 1))
1623
+ return tuple(found)
1624
+
1625
+
1570
1626
  def redirect_targets(command: object) -> tuple[str, ...]:
1571
1627
  """Files a shell command redirects its output into.
1572
1628
 
@@ -1604,7 +1660,7 @@ def redirect_targets(command: object) -> tuple[str, ...]:
1604
1660
  # body, an arithmetic comparison, a `case` frame or the body of a function
1605
1661
  # nothing called is not a redirection, and each was named as a file before
1606
1662
  # its mask existed.
1607
- for part in _split_on_separators(_readable_commands(command, 0)):
1663
+ for part, _op in _split_on_separators(_readable_commands(command, 0)):
1608
1664
  part = part.strip()
1609
1665
  if not part:
1610
1666
  continue