stacktrace-cli 0.6.0__py3-none-any.whl → 0.6.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. stacktrace_cli/__init__.py +1 -1
  2. stacktrace_cli/cli.py +5 -0
  3. stacktrace_cli/correlate/acquire.py +60 -67
  4. stacktrace_cli/correlate/composition.py +4 -4
  5. stacktrace_cli/correlate/orchestrate.py +30 -38
  6. stacktrace_cli/daemon/identity.py +17 -1
  7. stacktrace_cli/daemon/presentation.py +6 -1
  8. stacktrace_cli/daemon/store.py +60 -15
  9. stacktrace_cli/daemon/webhook.py +34 -4
  10. stacktrace_cli/detector/blocked.py +12 -0
  11. stacktrace_cli/detector/deterministic.py +241 -172
  12. stacktrace_cli/detector/render.py +5 -5
  13. stacktrace_cli/detector/rules.py +47 -8
  14. stacktrace_cli/detector/run.py +2 -2
  15. stacktrace_cli/monitor/inventory.py +495 -0
  16. stacktrace_cli/monitor/render.py +49 -1
  17. stacktrace_cli/monitor/server.py +40 -4
  18. stacktrace_cli/monitor/site/app.js +368 -78
  19. stacktrace_cli/monitor/site/index.html +27 -3
  20. stacktrace_cli/monitor/site/inventory.js +554 -0
  21. stacktrace_cli/monitor/site/styles.css +139 -16
  22. stacktrace_cli/monitor/state.py +45 -2
  23. stacktrace_cli/monitor/watch.py +156 -65
  24. stacktrace_cli/remote/payload.py +10 -5
  25. stacktrace_cli/remote/redact.py +100 -2
  26. stacktrace_cli/remote/sync.py +11 -0
  27. stacktrace_cli/telemetry/events.py +12 -30
  28. stacktrace_cli/webhook/cli.py +35 -4
  29. {stacktrace_cli-0.6.0.dist-info → stacktrace_cli-0.6.1.dist-info}/METADATA +2 -2
  30. {stacktrace_cli-0.6.0.dist-info → stacktrace_cli-0.6.1.dist-info}/RECORD +32 -30
  31. {stacktrace_cli-0.6.0.dist-info → stacktrace_cli-0.6.1.dist-info}/WHEEL +0 -0
  32. {stacktrace_cli-0.6.0.dist-info → stacktrace_cli-0.6.1.dist-info}/entry_points.txt +0 -0
@@ -2,7 +2,7 @@
2
2
 
3
3
  import logging as _logging
4
4
 
5
- __version__ = "0.6.0"
5
+ __version__ = "0.6.1"
6
6
 
7
7
  # Only the daemon attaches a handler (ADR-0048). Without this one, a WARNING
8
8
  # logged in any other process would reach Python's last-resort handler and
stacktrace_cli/cli.py CHANGED
@@ -539,6 +539,11 @@ def monitor(
539
539
  Loopback only, and free to leave open: a pass is skipped when nothing has
540
540
  changed, advisory lookups are asked once per component, and the two stages
541
541
  that need no model are the only ones that run.
542
+
543
+ The Inventory and Vulnerabilities tabs list every component your agents can
544
+ load. To check them for known vulnerabilities, the package names of
545
+ installed components -- not only the ones a session used -- are sent to
546
+ osv.dev. Skills and plugins with no package are never sent.
542
547
  """
543
548
  analyzer = resolve_analyzer(analyzer, reasoning)
544
549
  try:
@@ -316,10 +316,32 @@ class Built:
316
316
  document: dict[str, Any]
317
317
 
318
318
 
319
+ #: What an advisory answer is about: one component at one version. The
320
+ #: `openaca:identity` this codebase assigns does not encode a version, so it is
321
+ #: paired with the version the component declares. Keyed by identity alone, one
322
+ #: scan kept one version per identity and handed its advisories to every
323
+ #: version of that component in the run.
324
+ AdvisoryKey = tuple[str, "str | None"]
325
+
326
+
327
+ def advisory_key(component: dict[str, Any]) -> AdvisoryKey | None:
328
+ """A raw BOM component's `(identity, version)`, or None with no identity.
329
+
330
+ Read off the document, where `Component` reads the same two fields through
331
+ OpenACA's projection; `test_advisory_keys_read_the_same_off_both` holds the
332
+ two equal, since a key that differed would match nothing and read clean.
333
+ """
334
+ identity = _identity_of(component)
335
+ if identity is None:
336
+ return None
337
+ version = component.get("version")
338
+ return identity, version if isinstance(version, str) else None
339
+
340
+
319
341
  def advisories_for(
320
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
321
- ) -> dict[str, tuple[Advisory, ...]] | None:
322
- """Advisories for these identities alone, keyed by `openaca:identity`.
342
+ documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
343
+ ) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
344
+ """Advisories for these components alone, keyed by `(identity, version)`.
323
345
 
324
346
  **Asked after correlation, not during acquisition.** Advisory matching costs
325
347
  a network round trip to osv.dev per package coordinate, and acquisition does
@@ -332,61 +354,62 @@ def advisories_for(
332
354
  there: one scan, over the union of what ran. Same measurement, 0.66s to
333
355
  0.14s, and osv.dev learns only about components this machine actually used.
334
356
 
335
- Keyed by identity rather than `bom-ref` because a ref is local to one
336
- document while an identity is the coordinate OSV matched — which is what lets
337
- one scan answer for every composition in the run.
357
+ Keyed by `(identity, version)` rather than `bom-ref` because a ref is local
358
+ to one document while the pair is what OSV matched — which is what lets one
359
+ scan answer for every composition in the run, and each version for itself.
338
360
 
339
361
  Returns None when the scan could not run. None is not an empty result: a
340
362
  composition that was never checked must not read as one with nothing found.
341
363
  """
342
- if not identities:
364
+ if not keys:
343
365
  # Nothing ran that could carry an advisory. That is a checked result with
344
366
  # nothing in it, not a failure to look, so it is `{}` rather than None.
345
367
  return {}
346
- merged = _merge_for_scan(documents, identities)
368
+ merged, key_by_ref = _merge_for_scan(documents, keys)
347
369
  if not merged["components"]:
348
370
  return {}
349
371
  with tempfile.TemporaryDirectory() as scratch:
350
372
  path = Path(scratch) / "invoked.cdx.json"
351
373
  path.write_text(json.dumps(merged), encoding="utf-8")
352
- return _scan(path, _identity_by_ref(merged))
374
+ return _scan(path, key_by_ref)
353
375
 
354
376
 
355
377
  def _merge_for_scan(
356
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
357
- ) -> dict[str, Any]:
358
- """One document holding each invoked component once, and nothing else.
359
-
360
- Deduplicated on identity: the same component appears in every composition
361
- built, and scanning it once per appearance is what this function exists to
362
- avoid. `dependencies` is dropped rather than filtered — it describes
378
+ documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
379
+ ) -> tuple[dict[str, Any], dict[str, AdvisoryKey]]:
380
+ """One document holding each wanted component once, and which key each ref is.
381
+
382
+ Deduplicated on `(identity, version)`: the same component appears in every
383
+ composition built, and scanning it once per appearance is what this
384
+ function exists to avoid -- while two versions of one identity are two
385
+ components, each scanned for itself. `dependencies` is dropped rather than filtered — it describes
363
386
  installation structure that advisory matching does not read, and a pruned
364
387
  graph would be a second, worse description of one already recorded in the
365
388
  compositions themselves.
366
389
 
367
- **Every kept component's `bom-ref` is rewritten to its identity.** The
368
- CycloneDX spec scopes `bom-ref` uniqueness to a single document, never
369
- across documents, so two source BOMs are free to reuse the same local ref
370
- for two different components. Carrying an original ref forward into the
371
- merged document could then collide, and `_identity_by_ref` would collapse
372
- the collision onto whichever identity it saw last — misattributing that
373
- ref's advisories to the wrong component. Identity is already this dict's
374
- own key, so it is already unique across every document being merged.
390
+ **Every kept component's `bom-ref` is rewritten, and the new ref maps back
391
+ to its key.** The CycloneDX spec scopes `bom-ref` uniqueness to a single
392
+ document, never across documents, so two source BOMs are free to reuse the
393
+ same local ref for two different components. Carrying an original ref
394
+ forward into the merged document could then collide and misattribute that
395
+ ref's advisories to the wrong component. A ref minted here per kept key is
396
+ unique across every document being merged by construction.
375
397
  """
376
398
  base = documents[0] if documents else {}
377
- kept: dict[str, dict[str, Any]] = {}
399
+ kept: dict[AdvisoryKey, dict[str, Any]] = {}
378
400
  for document in documents:
379
401
  for component in document.get("components") or []:
380
- identity = _identity_of(component)
381
- if identity is not None and identity in identities and identity not in kept:
382
- kept[identity] = {**component, "bom-ref": identity}
383
- return {
402
+ key = advisory_key(component)
403
+ if key is not None and key in keys and key not in kept:
404
+ kept[key] = {**component, "bom-ref": f"scan-{len(kept)}"}
405
+ merged = {
384
406
  "bomFormat": base.get("bomFormat", "CycloneDX"),
385
407
  "specVersion": base.get("specVersion", "1.6"),
386
408
  "version": base.get("version", 1),
387
409
  "metadata": base.get("metadata", {}),
388
410
  "components": list(kept.values()),
389
411
  }
412
+ return merged, {component["bom-ref"]: key for key, component in kept.items()}
390
413
 
391
414
 
392
415
  def _identity_of(component: dict[str, Any]) -> str | None:
@@ -397,40 +420,10 @@ def _identity_of(component: dict[str, Any]) -> str | None:
397
420
  return None
398
421
 
399
422
 
400
- def identity_versions(
401
- documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
402
- ) -> dict[str, str | None]:
403
- """Each wanted identity's own declared version, the first time it is found.
404
-
405
- The identity this codebase assigns does not encode a version — that is
406
- what lets one scan answer for a component across every composition it
407
- appears in — so a cache keyed on identity alone cannot tell an upgrade
408
- from no change at all. Pairing identity with this lets a long-lived
409
- lookup cache do that.
410
- """
411
- versions: dict[str, str | None] = {}
412
- for document in documents:
413
- for component in document.get("components") or []:
414
- identity = _identity_of(component)
415
- if identity is None or identity not in identities or identity in versions:
416
- continue
417
- version = component.get("version")
418
- versions[identity] = version if isinstance(version, str) else None
419
- return versions
420
-
421
-
422
- def _identity_by_ref(document: dict[str, Any]) -> dict[str, str]:
423
- return {
424
- str(component["bom-ref"]): identity
425
- for component in document.get("components") or []
426
- if component.get("bom-ref") and (identity := _identity_of(component))
427
- }
428
-
429
-
430
423
  def _scan(
431
- bom_path: Path, identity_by_ref: dict[str, str]
432
- ) -> dict[str, tuple[Advisory, ...]] | None:
433
- """Run `openaca scan bom` and re-key its findings from refs onto identities.
424
+ bom_path: Path, key_by_ref: dict[str, AdvisoryKey]
425
+ ) -> dict[AdvisoryKey, tuple[Advisory, ...]] | None:
426
+ """Run `openaca scan bom` and re-key its findings from refs onto keys.
434
427
 
435
428
  `scan bom` is used rather than `scan endpoint` because it is a pure function
436
429
  of a document we already hold, so nothing can drift between the composition
@@ -483,7 +476,7 @@ def _scan(
483
476
  if not isinstance(findings, list):
484
477
  return None
485
478
 
486
- matched: dict[str, list[Advisory]] = {}
479
+ matched: dict[AdvisoryKey, list[Advisory]] = {}
487
480
  for finding in findings:
488
481
  if not isinstance(finding, dict):
489
482
  return None
@@ -505,13 +498,13 @@ def _scan(
505
498
  return None
506
499
  if not all(_text_or_absent(value) for value in (fixed_in, source, severity, title)):
507
500
  return None
508
- identity = identity_by_ref.get(ref)
509
- if identity is None:
501
+ key = key_by_ref.get(ref)
502
+ if key is None:
510
503
  # Not malformed: we sent the components, and a ref we do not hold is
511
504
  # a finding about something outside this scan rather than a response
512
505
  # we cannot read.
513
506
  continue
514
- matched.setdefault(identity, []).append(
507
+ matched.setdefault(key, []).append(
515
508
  Advisory(
516
509
  id=identifier,
517
510
  severity=severity or "UNKNOWN",
@@ -520,7 +513,7 @@ def _scan(
520
513
  source=source,
521
514
  )
522
515
  )
523
- return {identity: tuple(items) for identity, items in matched.items()}
516
+ return {key: tuple(items) for key, items in matched.items()}
524
517
 
525
518
 
526
519
  def _text_or_absent(value: object) -> bool:
@@ -152,10 +152,10 @@ def composition_from_bom(
152
152
  generated_at: datetime,
153
153
  generated_at_is_observed: bool,
154
154
  #: Keyed by **bom-ref**, matching `Composition.advisories`. Note that
155
- #: `acquire.advisories_for` returns an *identity*-keyed map — one scan can
156
- #: answer for every composition that way — so a caller re-keys before
157
- #: arriving here, as `_attach_advisories` does. Passing the identity-keyed
158
- #: map straight through would silently match nothing.
155
+ #: `acquire.advisories_for` returns an `(identity, version)`-keyed map — one
156
+ #: scan can answer for every composition that way — so a caller re-keys
157
+ #: before arriving here, as `_attach_advisories` does. Passing that map
158
+ #: straight through would silently match nothing.
159
159
  advisories: dict[str, tuple[Advisory, ...]] | None = None,
160
160
  ) -> Composition:
161
161
  """Project a CycloneDX Agent BOM into a composition, or reject it."""
@@ -41,10 +41,10 @@ from pathlib import Path
41
41
  from typing import Any
42
42
 
43
43
  from stacktrace_cli.correlate.acquire import (
44
+ AdvisoryKey,
44
45
  Built,
45
46
  advisories_for,
46
47
  build_bom,
47
- identity_versions,
48
48
  load_bom,
49
49
  )
50
50
  from stacktrace_cli.correlate.composition import Advisory, Composition
@@ -141,8 +141,8 @@ def _key(project: Path | None) -> str | None:
141
141
  BuildAll = Callable[[Sequence[tuple[str, Path | None]]], dict[tuple[str, str | None], Built]]
142
142
 
143
143
 
144
- def _invoked_identities(correlated: object) -> set[str]:
145
- """Every component identity a session actually reached.
144
+ def _invoked_keys(correlated: object) -> set[AdvisoryKey]:
145
+ """Every component a session actually reached, as `(identity, version)`.
146
146
 
147
147
  Every candidate of every resolved call, ambiguous ones included: the call
148
148
  happened, and which of two same-named components answered it is undecided
@@ -154,7 +154,7 @@ def _invoked_identities(correlated: object) -> set[str]:
154
154
  would still be sent to osv.dev: an unnecessary disclosure and scan cost
155
155
  for a component nothing actually ran.
156
156
  """
157
- identities: set[str] = set()
157
+ keys: set[AdvisoryKey] = set()
158
158
  for session in correlated.sessions: # type: ignore[attr-defined]
159
159
  calls = {call.span: call for turn in session.session.turns for call in turn.tool_calls}
160
160
  for span, resolution in session.resolutions.items():
@@ -163,22 +163,22 @@ def _invoked_identities(correlated: object) -> set[str]:
163
163
  continue
164
164
  for candidate in resolution.candidates:
165
165
  if candidate.identity:
166
- identities.add(candidate.identity)
167
- return identities
166
+ keys.add((candidate.identity, candidate.version))
167
+ return keys
168
168
 
169
169
 
170
- #: `advisories_for`'s shape: the documents to search and the identities that
170
+ #: `advisories_for`'s shape: the documents to search and the components that
171
171
  #: were actually invoked, answering `None` for "not checked at all", which must
172
172
  #: stay distinguishable from "checked and found nothing".
173
173
  AdvisoryLookup = Callable[
174
- ["Sequence[dict[str, Any]]", "AbstractSet[str]"],
175
- "dict[str, tuple[Advisory, ...]] | None",
174
+ ["Sequence[dict[str, Any]]", "AbstractSet[AdvisoryKey]"],
175
+ "dict[AdvisoryKey, tuple[Advisory, ...]] | None",
176
176
  ]
177
177
  AttachAdvisories = Callable[[CorrelatedView, "Sequence[Built]"], CorrelatedView]
178
178
 
179
179
 
180
180
  class CachingAdvisoryLookup:
181
- """`advisories_for`, asked once per identity for the life of the process.
181
+ """`advisories_for`, asked once per `(identity, version)` for the life of the process.
182
182
 
183
183
  `None` from the underlying lookup means *not checked at all*, which must
184
184
  stay distinguishable from *checked and found nothing* — an unchecked
@@ -194,19 +194,16 @@ class CachingAdvisoryLookup:
194
194
 
195
195
  def __init__(self, lookup: Callable[..., Any] = advisories_for) -> None:
196
196
  self._lookup = lookup
197
- self._known: dict[tuple[str, str | None], tuple[Any, ...]] = {}
198
- self._asked: set[tuple[str, str | None]] = set()
197
+ self._known: dict[AdvisoryKey, tuple[Any, ...]] = {}
198
+ self._asked: set[AdvisoryKey] = set()
199
199
 
200
200
  def __call__(
201
- self, documents: Sequence[dict[str, Any]], identities: AbstractSet[str]
202
- ) -> dict[str, tuple[Any, ...]] | None:
203
- wanted = set(identities or ())
204
- # Keyed by (identity, version) rather than identity alone: the identity
205
- # this codebase assigns does not encode a version, so an upgrade or
206
- # downgrade of an already-asked component must still be asked about.
207
- versions = identity_versions(documents, wanted)
208
- keys = {identity: (identity, versions.get(identity)) for identity in wanted}
209
- unasked = {identity for identity in wanted if keys[identity] not in self._asked}
201
+ self, documents: Sequence[dict[str, Any]], keys: AbstractSet[AdvisoryKey]
202
+ ) -> dict[AdvisoryKey, tuple[Any, ...]] | None:
203
+ # The key carries the version, so an upgrade or downgrade of an
204
+ # already-asked component is a new key and is asked about.
205
+ wanted = set(keys or ())
206
+ unasked = wanted - self._asked
210
207
  if unasked:
211
208
  matched = self._lookup(documents, unasked)
212
209
  if matched is None:
@@ -214,15 +211,9 @@ class CachingAdvisoryLookup:
214
211
  # answer for the ones we do know: a partial answer here would
215
212
  # be indistinguishable from a complete one downstream.
216
213
  return None
217
- self._asked |= {keys[identity] for identity in unasked}
218
- self._known.update(
219
- {keys[identity]: advisories for identity, advisories in matched.items()}
220
- )
221
- return {
222
- identity: self._known[keys[identity]]
223
- for identity in wanted
224
- if keys[identity] in self._known
225
- }
214
+ self._asked |= unasked
215
+ self._known.update(matched)
216
+ return {key: self._known[key] for key in wanted if key in self._known}
226
217
 
227
218
 
228
219
  def advisory_attacher(lookup: AdvisoryLookup = advisories_for) -> AttachAdvisories:
@@ -255,9 +246,10 @@ def _attach_advisories(
255
246
  component on the machine to osv.dev, once per composition built; asked here
256
247
  it sends the union of what was invoked, once.
257
248
 
258
- Re-keying is by identity, which is why one scan can answer for every
259
- composition: a `bom-ref` is local to one document, while the identity is the
260
- coordinate OSV matched on.
249
+ Re-keying is by `(identity, version)`, which is why one scan can answer for
250
+ every composition: a `bom-ref` is local to one document, while the pair is
251
+ what OSV matched on -- and a composition on another version of the same
252
+ component reads that version's answer, never this one's.
261
253
 
262
254
  **Enrichment is memoised by the original composition's identity.** Two
263
255
  sessions of the same kind (and project) share one `Composition` object —
@@ -268,10 +260,10 @@ def _attach_advisories(
268
260
  session independently would give two sessions that shared one composition
269
261
  two different enriched objects and break that dedup silently.
270
262
  """
271
- identities = _invoked_identities(correlated)
272
- matched = lookup([b.document for b in builts], identities)
263
+ keys = _invoked_keys(correlated)
264
+ matched = lookup([b.document for b in builts], keys)
273
265
  checked = matched is not None
274
- by_identity = matched or {}
266
+ by_key = matched or {}
275
267
  memo: dict[int, Composition] = {}
276
268
 
277
269
  def enriched(composition: Composition | None) -> Composition | None:
@@ -283,9 +275,9 @@ def _attach_advisories(
283
275
  built = replace(
284
276
  composition,
285
277
  advisories={
286
- component.bom_ref: by_identity[component.identity]
278
+ component.bom_ref: by_key[(component.identity, component.version)]
287
279
  for component in composition.components
288
- if component.identity and component.identity in by_identity
280
+ if component.identity and (component.identity, component.version) in by_key
289
281
  },
290
282
  advisories_checked=checked,
291
283
  )
@@ -15,7 +15,7 @@ import hashlib
15
15
  import json
16
16
  from typing import Any
17
17
 
18
- from stacktrace_cli.detector.rules import scope_of
18
+ from stacktrace_cli.detector.rules import SUCCEEDS, scope_of
19
19
 
20
20
  #: How much of the content digest is kept. Wide enough that the revisions of one
21
21
  #: finding cannot collide within the `UNIQUE (event_id, content_hash)` they share
@@ -59,6 +59,22 @@ def detection_event_id(agent_kind: str, session_id: str, detection: dict[str, An
59
59
  return _event_id(agent_kind, session_id, rule_id, anchor)
60
60
 
61
61
 
62
+ def retired_detection_event_id(
63
+ agent_kind: str, session_id: str, detection: dict[str, Any]
64
+ ) -> str | None:
65
+ """The id this finding had under the retired rule it took over from, or
66
+ `None` where its rule replaced none (ADR-0083).
67
+
68
+ Rebuilt rather than looked up: the retired rule's own scope decides the
69
+ anchor, and `agent-blocked` anchored on the first blocked call, which is the
70
+ same `evidence[0]` its successor still cites first.
71
+ """
72
+ predecessor = SUCCEEDS.get(str(detection.get("rule_id")))
73
+ if predecessor is None:
74
+ return None
75
+ return detection_event_id(agent_kind, session_id, {**detection, "rule_id": predecessor})
76
+
77
+
62
78
  def unknown_event_id(agent_kind: str, session_id: str, unknown: dict[str, Any]) -> str:
63
79
  """`hash(agent_kind, session_id, stage, reason, spans)`.
64
80
 
@@ -4,11 +4,14 @@ from __future__ import annotations
4
4
 
5
5
  import json
6
6
 
7
- from stacktrace_cli.detector.blocked import REASON_CODES, reason_for
7
+ from stacktrace_cli.detector.blocked import REASON_CODES, REASONS, reason_for
8
8
 
9
9
  # Notification prose is selected by rule ID only (ADR-0031). In particular,
10
10
  # finding titles, component names and evidence never supply notification text.
11
11
  _GUIDANCE = {
12
+ # One per block reason, in `blocked.py`'s own words: each reason is its own
13
+ # rule (ADR-0083), so the rule id alone now selects the specific sentence.
14
+ **{reason.rule_id: (reason.title, reason.remedy) for reason in REASONS},
12
15
  "stacktrace-credential-egress": (
13
16
  "Credential-shaped material appeared in an outbound tool call",
14
17
  (
@@ -20,6 +23,7 @@ _GUIDANCE = {
20
23
  "A success claim conflicts with verification evidence",
21
24
  "Inspect the cited verification results before relying on the completion claim.",
22
25
  ),
26
+ # Retired by ADR-0083. Findings stored before the split still name it.
23
27
  "stacktrace-agent-blocked": (
24
28
  "The agent encountered a block while working",
25
29
  "Inspect the recorded block reason before retrying the task.",
@@ -70,6 +74,7 @@ def render_finding(finding: dict[str, object]) -> str:
70
74
  items = (
71
75
  [item for item in evidence if isinstance(item, dict)] if isinstance(evidence, list) else []
72
76
  )
77
+ # A finding stored before ADR-0083 carries its reason in the evidence.
73
78
  if rule_id == "stacktrace-agent-blocked" and items:
74
79
  code = items[0].get("kind")
75
80
  if isinstance(code, str) and code in REASON_CODES:
@@ -50,6 +50,7 @@ from .identity import (
50
50
  canonical_json,
51
51
  content_hash,
52
52
  detection_event_id,
53
+ retired_detection_event_id,
53
54
  unknown_event_id,
54
55
  )
55
56
  from .presentation import rule_guidance
@@ -64,6 +65,11 @@ from .presentation import rule_guidance
64
65
  FINDING = "finding"
65
66
  UNKNOWN = "unknown"
66
67
 
68
+ #: One definition of what the webhook may POST. The delivery query evaluates
69
+ #: it per row so the cursor can still advance past an expired revision; the
70
+ #: configure-time preview counts the same predicate.
71
+ _WEBHOOK_DELIVERABLE_SQL = "kind = ? AND timeline_json IS NOT NULL AND created_at >= ?"
72
+
67
73
  #: How long a finding is kept locally. `daemon-uploads.md` puts a `--retention`
68
74
  #: flag in front of this; until then it is the default and the only value.
69
75
  DEFAULT_RETENTION = timedelta(days=90)
@@ -96,6 +102,9 @@ class QueuedRecord:
96
102
  #: Whether this exact revision crossed the local notification gate. Later
97
103
  #: revisions keep the same event id but do not ring again.
98
104
  notifiable: bool = False
105
+ #: Whether the webhook may POST this revision now: it crossed the gate and
106
+ #: remains inside the webhook's retention window.
107
+ deliverable: bool = False
99
108
 
100
109
 
101
110
  @dataclass(frozen=True)
@@ -190,7 +199,14 @@ class FindingStore:
190
199
  payload, home=self.home, working_directories=working_directories
191
200
  )
192
201
  event_id = detection_event_id(session.agent_kind, session.session_id, payload)
193
- already_rung = self._already_notified(event_id)
202
+ # Or rang before an upgrade, under the rule this one took
203
+ # over from (ADR-0083).
204
+ retired_id = retired_detection_event_id(
205
+ session.agent_kind, session.session_id, payload
206
+ )
207
+ already_rung = self._already_notified(event_id) or (
208
+ retired_id is not None and self._already_notified(retired_id)
209
+ )
194
210
  timeline = (
195
211
  None
196
212
  if already_rung or not self._eligible(payload)
@@ -367,12 +383,13 @@ class FindingStore:
367
383
  later revision of the same event must not hide it before an offline
368
384
  subscriber has advanced past it.
369
385
  """
386
+ cutoff = (datetime.now(UTC) - self.retention).isoformat()
370
387
  with self._lock:
371
388
  rows = self._connection.execute(
372
389
  "SELECT seq, kind, event_id, findings_json, created_at,"
373
- " timeline_json IS NOT NULL FROM findings"
390
+ " timeline_json IS NOT NULL, " + _WEBHOOK_DELIVERABLE_SQL + " FROM findings"
374
391
  " WHERE agent_kind = ? AND seq > ? ORDER BY seq LIMIT ?",
375
- (partition, seq, limit),
392
+ (FINDING, cutoff, partition, seq, limit),
376
393
  ).fetchall()
377
394
  return tuple(
378
395
  QueuedRecord(
@@ -382,10 +399,45 @@ class FindingStore:
382
399
  payload=json.loads(findings_json),
383
400
  created_at=str(created_at),
384
401
  notifiable=bool(notifiable),
402
+ deliverable=bool(deliverable),
385
403
  )
386
- for row_seq, kind, event_id, findings_json, created_at, notifiable in rows
404
+ for (
405
+ row_seq,
406
+ kind,
407
+ event_id,
408
+ findings_json,
409
+ created_at,
410
+ notifiable,
411
+ deliverable,
412
+ ) in rows
387
413
  )
388
414
 
415
+ def pending_delivery_count(self, *, subscriber: str) -> int:
416
+ """Notifiable finding revisions this subscriber has not passed yet.
417
+
418
+ A delivery drain scans every row, but it POSTs only finding revisions
419
+ with a notification timeline. This count follows that distinction so
420
+ a configure-time preview does not include unknowns, ineligible rows or
421
+ later revisions of a finding that already rang.
422
+
423
+ Uses the same deliverability predicate as `delivery_after()`: Fleet
424
+ may retain an older physical row for its own cursor, but the webhook
425
+ still delivers only what is inside its retention window (ADR-0069).
426
+ """
427
+ cutoff = (datetime.now(UTC) - self.retention).isoformat()
428
+ total = 0
429
+ for partition in self.partitions():
430
+ cursor = self.cursor(partition, subscriber=subscriber)
431
+ with self._lock:
432
+ row = self._connection.execute(
433
+ "SELECT COUNT(*) FROM findings"
434
+ " WHERE agent_kind = ? AND seq > ?"
435
+ " AND " + _WEBHOOK_DELIVERABLE_SQL,
436
+ (partition, cursor, FINDING, cutoff),
437
+ ).fetchone()
438
+ total += int(row[0])
439
+ return total
440
+
389
441
  def has_queued_rows(self, *, subscriber: str = FLEET) -> bool:
390
442
  """Whether any partition has a row this subscriber has not passed yet.
391
443
 
@@ -890,11 +942,10 @@ class FindingStore:
890
942
 
891
943
  @staticmethod
892
944
  def _delivery_properties(session: SessionKey, finding: dict[str, Any]) -> dict[str, str]:
893
- """The closed-set facts about one delivery: rule, severity, the sink, and
894
- for `stacktrace-agent-blocked` the reason code, which lives in its anchor
895
- evidence's `kind` (`detector/deterministic.py`). Nothing else on the
896
- finding is read."""
897
- properties = {
945
+ """The closed-set facts about one delivery: rule, severity and the sink.
946
+ Nothing else on the finding is read -- a block's reason is its rule
947
+ (ADR-0083), so there is no reason code to read out of its evidence."""
948
+ return {
898
949
  "rule": str(finding["rule_id"]),
899
950
  "severity": str(finding["severity"]),
900
951
  # The only sink with a delivery mark on `main`; ADR-0036's others
@@ -903,9 +954,3 @@ class FindingStore:
903
954
  "session_id": session.session_id,
904
955
  "agent_kind": session.agent_kind,
905
956
  }
906
- if properties["rule"] == "stacktrace-agent-blocked":
907
- evidence = finding.get("evidence") or [{}]
908
- reason = evidence[0].get("kind")
909
- if isinstance(reason, str):
910
- properties["reason"] = reason
911
- return properties