veltro-cli 0.15.7__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,611 @@
1
+ """Demo dataset catalogue: discovery, validation, selection and replay shaping.
2
+
3
+ The corpora live in ``content/datasets/<id>/`` (Apache-2.0) and are described by
4
+ a ``veltro-demo-dataset/v1`` manifest. This module is the only reader: the CLI
5
+ (`veltro demo seed --dataset …`), the replay engine
6
+ (`veltro_cli.lib.demo_replay`) and the CI validator
7
+ (`scripts/test_demo_datasets.py`) all go through it, so a corpus cannot ship a
8
+ manifest that one consumer accepts and another rejects.
9
+
10
+ Everything here is pure: no network, no clock except what the caller passes in.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ import json
16
+ import re
17
+ from dataclasses import dataclass
18
+ from datetime import datetime, timedelta, timezone
19
+ from pathlib import Path
20
+ from typing import Any, Iterable, Mapping, Sequence
21
+
22
+ from veltro_cli.lib.errors import VeltroCliError
23
+
24
+ MANIFEST_SCHEMA = "veltro-demo-dataset/v1"
25
+ MANIFEST_NAME = "dataset.json"
26
+ CONTENT_SUBDIR = ("content", "datasets")
27
+ GOLDEN_PATH_ENGINE = "golden-path-seed"
28
+ CORPUS_ENGINE = "corpus"
29
+ ENGINES = (GOLDEN_PATH_ENGINE, CORPUS_ENGINE)
30
+ REQUIRED_LICENSE = "Apache-2.0"
31
+ APACHE_SPDX = "SPDX-License-Identifier: Apache-2.0"
32
+ ID_PATTERN = re.compile(r"^[a-z][a-z0-9-]{1,40}$")
33
+ # The label every replayed event carries, so `veltro demo reset` can find demo
34
+ # data again and an operator can tell it apart from their own telemetry.
35
+ DEMO_LABEL_FIELD = "veltro_demo"
36
+ DEMO_TAG = "veltro-demo"
37
+ ALL_KEYWORD = "all"
38
+
39
+
40
+ class DatasetError(VeltroCliError):
41
+ """A dataset directory does not satisfy the manifest contract."""
42
+
43
+
44
+ @dataclass(frozen=True)
45
+ class Threshold:
46
+ count: int
47
+ window_minutes: int
48
+ group_by: str
49
+
50
+
51
+ @dataclass(frozen=True)
52
+ class Detection:
53
+ rule_title: str
54
+ description: str
55
+ severity: str
56
+ yaml_path: Path
57
+ field_mappings: tuple[tuple[str, str], ...]
58
+ threshold: Threshold | None
59
+ sample_event_index: int
60
+
61
+ def yaml_content(self) -> str:
62
+ return self.yaml_path.read_text(encoding="utf-8")
63
+
64
+
65
+ @dataclass(frozen=True)
66
+ class Dataset:
67
+ id: str
68
+ title: str
69
+ summary: str
70
+ engine: str
71
+ directory: Path
72
+ attack_techniques: tuple[str, ...]
73
+ events_path: Path | None
74
+ detection: Detection | None
75
+ # The CHAD rule title this dataset's detection carries. Taken from
76
+ # `detection.rule_title` for a corpus dataset and from the manifest's
77
+ # `golden_path.rule_title` for the golden path, whose rule the shell
78
+ # seeder owns: `reset` resolves the rule by title when a run died before
79
+ # it could record the id (veltro#984), so every dataset has one.
80
+ rule_title: str
81
+ min_alerts: int
82
+ expect_warden_case: bool
83
+ matching_event_ids: tuple[str, ...]
84
+
85
+ @property
86
+ def is_corpus(self) -> bool:
87
+ return self.engine == CORPUS_ENGINE
88
+
89
+ def events(self) -> list[dict[str, Any]]:
90
+ """The corpus, in authored order. Empty for the golden-path engine."""
91
+ if self.events_path is None:
92
+ return []
93
+ return [
94
+ json.loads(line)
95
+ for line in self.events_path.read_text(encoding="utf-8").splitlines()
96
+ if line.strip()
97
+ ]
98
+
99
+
100
+ def _require(condition: bool, message: str, remedy: str) -> None:
101
+ if not condition:
102
+ raise DatasetError(message, remedy=remedy)
103
+
104
+
105
+ def _string(mapping: Mapping[str, Any], key: str, where: str) -> str:
106
+ value = mapping.get(key)
107
+ _require(
108
+ isinstance(value, str) and value.strip() != "",
109
+ f"{where}: {key!r} must be a non-empty string",
110
+ remedy=f"set {key!r} in {where}",
111
+ )
112
+ return str(value).strip()
113
+
114
+
115
+ def _event_id(event: Mapping[str, Any]) -> str:
116
+ event_block = event.get("event")
117
+ if isinstance(event_block, Mapping):
118
+ candidate = event_block.get("id")
119
+ if isinstance(candidate, str) and candidate.strip():
120
+ return candidate.strip()
121
+ return ""
122
+
123
+
124
+ def field_value(event: Mapping[str, Any], dotted: str) -> Any:
125
+ """Read a dotted ECS path out of a nested event document."""
126
+ cursor: Any = event
127
+ for part in dotted.split("."):
128
+ if not isinstance(cursor, Mapping) or part not in cursor:
129
+ return None
130
+ cursor = cursor[part]
131
+ return cursor
132
+
133
+
134
+ def parse_simple_detection(rule_yaml: str) -> dict[str, str]:
135
+ """Return the ``selection`` field/value pairs of a shipped demo rule.
136
+
137
+ Demo rules are deliberately restricted to one flat ``selection`` mapping
138
+ with ``condition: selection`` — the shape a stdlib reader can parse exactly,
139
+ with no YAML dependency and no partial interpretation. CHAD's pySigma engine
140
+ remains the real evaluator; this parser exists so CI can prove a shipped
141
+ rule matches the events its manifest claims, and it refuses anything richer
142
+ rather than guessing.
143
+ """
144
+ lines = rule_yaml.splitlines()
145
+ try:
146
+ start = next(i for i, line in enumerate(lines) if line.rstrip() == "detection:")
147
+ except StopIteration as exc:
148
+ raise DatasetError(
149
+ "rule has no `detection:` block",
150
+ remedy="a demo rule needs a detection block with one flat selection",
151
+ ) from exc
152
+ selection: dict[str, str] = {}
153
+ condition = ""
154
+ in_selection = False
155
+ for line in lines[start + 1 :]:
156
+ if line.strip() == "" or line.lstrip().startswith("#"):
157
+ continue
158
+ indent = len(line) - len(line.lstrip())
159
+ if indent == 0:
160
+ break
161
+ stripped = line.strip()
162
+ if indent == 2:
163
+ in_selection = stripped == "selection:"
164
+ if stripped.startswith("condition:"):
165
+ condition = stripped.split(":", 1)[1].strip()
166
+ elif not in_selection:
167
+ raise DatasetError(
168
+ f"unsupported detection key {stripped!r}",
169
+ remedy="demo rules support exactly `selection:` and `condition: selection`",
170
+ )
171
+ continue
172
+ if indent == 4 and in_selection:
173
+ _require(
174
+ ":" in stripped,
175
+ f"unsupported selection entry {stripped!r}",
176
+ remedy="each selection entry must be `field: value`",
177
+ )
178
+ key, _, value = stripped.partition(":")
179
+ _require(
180
+ "|" not in key,
181
+ f"selection modifier in {key!r} is not supported by the demo validator",
182
+ remedy="keep demo rule selections to plain equality so CI can verify them",
183
+ )
184
+ selection[key.strip()] = value.strip().strip("'\"")
185
+ continue
186
+ raise DatasetError(
187
+ f"unsupported detection line {line!r}",
188
+ remedy="demo rules support exactly one flat selection mapping",
189
+ )
190
+ _require(
191
+ condition == "selection",
192
+ "demo rules must use `condition: selection`",
193
+ remedy="simplify the rule condition, or move the rule out of content/datasets",
194
+ )
195
+ _require(
196
+ bool(selection),
197
+ "demo rule selection is empty",
198
+ remedy="add at least one `field: value` entry under selection",
199
+ )
200
+ return selection
201
+
202
+
203
+ def selection_matches(selection: Mapping[str, str], event: Mapping[str, Any]) -> bool:
204
+ """True when every selection field equals its value on this event."""
205
+ return all(str(field_value(event, field)) == value for field, value in selection.items())
206
+
207
+
208
+ def _load_detection(directory: Path, block: Mapping[str, Any], event_count: int) -> Detection:
209
+ where = f"{directory.name}/dataset.json detection"
210
+ yaml_name = _string(block, "yaml", where)
211
+ yaml_path = directory / yaml_name
212
+ _require(
213
+ yaml_path.is_file(),
214
+ f"{where}: rule file {yaml_name!r} is missing",
215
+ remedy=f"add {directory.name}/{yaml_name}",
216
+ )
217
+ head = "".join(yaml_path.read_text(encoding="utf-8").splitlines(keepends=True)[:5])
218
+ _require(
219
+ APACHE_SPDX in head,
220
+ f"{where}: {yaml_name} has no Apache-2.0 SPDX header",
221
+ remedy=f"add `# {APACHE_SPDX}` as the first line of {yaml_name}",
222
+ )
223
+ raw_mappings = block.get("field_mappings")
224
+ _require(
225
+ isinstance(raw_mappings, list) and len(raw_mappings) > 0,
226
+ f"{where}: field_mappings must be a non-empty list",
227
+ remedy="map every Sigma field the rule keys on to its indexed field",
228
+ )
229
+ mappings: list[tuple[str, str]] = []
230
+ for entry in raw_mappings or []:
231
+ _require(
232
+ isinstance(entry, Mapping),
233
+ f"{where}: each field mapping must be an object",
234
+ remedy="use {\"sigma_field\": …, \"target_field\": …}",
235
+ )
236
+ mappings.append(
237
+ (_string(entry, "sigma_field", where), _string(entry, "target_field", where))
238
+ )
239
+ threshold_block = block.get("threshold")
240
+ threshold: Threshold | None = None
241
+ if threshold_block is not None:
242
+ _require(
243
+ isinstance(threshold_block, Mapping),
244
+ f"{where}: threshold must be an object or null",
245
+ remedy="omit the threshold with null, or give count/window_minutes/group_by",
246
+ )
247
+ assert isinstance(threshold_block, Mapping)
248
+ count = threshold_block.get("count")
249
+ window = threshold_block.get("window_minutes")
250
+ _require(
251
+ isinstance(count, int) and count >= 1,
252
+ f"{where}: threshold.count must be a positive integer",
253
+ remedy="set threshold.count",
254
+ )
255
+ _require(
256
+ isinstance(window, int) and window >= 1,
257
+ f"{where}: threshold.window_minutes must be a positive integer",
258
+ remedy="set threshold.window_minutes",
259
+ )
260
+ threshold = Threshold(
261
+ count=int(count), # type: ignore[arg-type]
262
+ window_minutes=int(window), # type: ignore[arg-type]
263
+ group_by=_string(threshold_block, "group_by", where),
264
+ )
265
+ sample = block.get("sample")
266
+ _require(
267
+ isinstance(sample, Mapping) and isinstance(sample.get("event_index"), int),
268
+ f"{where}: sample.event_index must be an integer",
269
+ remedy="name the corpus event the rule's deploy-floor sample test uses",
270
+ )
271
+ assert isinstance(sample, Mapping)
272
+ index = int(sample["event_index"])
273
+ _require(
274
+ 0 <= index < event_count,
275
+ f"{where}: sample.event_index {index} is outside the corpus (0..{event_count - 1})",
276
+ remedy="point sample.event_index at an event the rule matches",
277
+ )
278
+ return Detection(
279
+ rule_title=_string(block, "rule_title", where),
280
+ description=_string(block, "description", where),
281
+ severity=_string(block, "severity", where),
282
+ yaml_path=yaml_path,
283
+ field_mappings=tuple(mappings),
284
+ threshold=threshold,
285
+ sample_event_index=index,
286
+ )
287
+
288
+
289
+ def load_dataset(directory: Path) -> Dataset:
290
+ """Read and fully validate one dataset directory."""
291
+ directory = Path(directory)
292
+ manifest_path = directory / MANIFEST_NAME
293
+ _require(
294
+ manifest_path.is_file(),
295
+ f"{directory} has no {MANIFEST_NAME}",
296
+ remedy=f"add {directory.name}/{MANIFEST_NAME} (schema {MANIFEST_SCHEMA})",
297
+ )
298
+ try:
299
+ manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
300
+ except json.JSONDecodeError as exc:
301
+ raise DatasetError(
302
+ f"{manifest_path} is not valid JSON: {exc}",
303
+ remedy="fix the manifest syntax",
304
+ ) from exc
305
+ _require(
306
+ isinstance(manifest, Mapping),
307
+ f"{manifest_path} must contain a JSON object",
308
+ remedy="wrap the manifest in an object",
309
+ )
310
+ where = f"{directory.name}/{MANIFEST_NAME}"
311
+ _require(
312
+ manifest.get("schema") == MANIFEST_SCHEMA,
313
+ f"{where}: schema must be {MANIFEST_SCHEMA!r}",
314
+ remedy=f"set \"schema\": \"{MANIFEST_SCHEMA}\"",
315
+ )
316
+ dataset_id = _string(manifest, "id", where)
317
+ _require(
318
+ bool(ID_PATTERN.match(dataset_id)),
319
+ f"{where}: id {dataset_id!r} must be lowercase letters, digits and dashes",
320
+ remedy="rename the dataset id",
321
+ )
322
+ _require(
323
+ dataset_id == directory.name,
324
+ f"{where}: id {dataset_id!r} must equal the directory name {directory.name!r}",
325
+ remedy="rename the directory or the id so they agree",
326
+ )
327
+ _require(
328
+ manifest.get("license") == REQUIRED_LICENSE,
329
+ f"{where}: license must be {REQUIRED_LICENSE!r}",
330
+ remedy="demo corpora ship Apache-2.0; see content/datasets/LICENSE",
331
+ )
332
+ engine = _string(manifest, "engine", where)
333
+ _require(
334
+ engine in ENGINES,
335
+ f"{where}: engine must be one of {', '.join(ENGINES)}",
336
+ remedy="use \"corpus\" for a dataset that ships its own events",
337
+ )
338
+ techniques = manifest.get("attack_techniques")
339
+ _require(
340
+ isinstance(techniques, list) and all(isinstance(t, str) and t for t in techniques),
341
+ f"{where}: attack_techniques must be a list of technique ids",
342
+ remedy="list the ATT&CK technique ids the corpus depicts",
343
+ )
344
+ provenance = manifest.get("provenance")
345
+ _require(
346
+ isinstance(provenance, Mapping) and provenance.get("synthetic") is True,
347
+ f"{where}: provenance.synthetic must be true",
348
+ remedy="only authored, synthetic corpora may ship here — see content/datasets/README.md",
349
+ )
350
+ events_path: Path | None = None
351
+ detection: Detection | None = None
352
+ rule_title = ""
353
+ event_ids: list[str] = []
354
+ if engine == CORPUS_ENGINE:
355
+ corpus = manifest.get("corpus")
356
+ _require(
357
+ isinstance(corpus, Mapping),
358
+ f"{where}: engine \"corpus\" requires a corpus block",
359
+ remedy="add corpus.events and corpus.count",
360
+ )
361
+ assert isinstance(corpus, Mapping)
362
+ events_name = _string(corpus, "events", where + " corpus")
363
+ events_path = directory / events_name
364
+ _require(
365
+ events_path.is_file(),
366
+ f"{where}: corpus file {events_name!r} is missing",
367
+ remedy=f"add {directory.name}/{events_name}",
368
+ )
369
+ lines = [
370
+ line
371
+ for line in events_path.read_text(encoding="utf-8").splitlines()
372
+ if line.strip()
373
+ ]
374
+ parsed: list[dict[str, Any]] = []
375
+ for number, line in enumerate(lines, start=1):
376
+ try:
377
+ event = json.loads(line)
378
+ except json.JSONDecodeError as exc:
379
+ raise DatasetError(
380
+ f"{events_name} line {number} is not valid JSON: {exc}",
381
+ remedy="one JSON object per line, no trailing commas",
382
+ ) from exc
383
+ _require(
384
+ isinstance(event, dict),
385
+ f"{events_name} line {number} must be a JSON object",
386
+ remedy="one JSON event object per line",
387
+ )
388
+ identifier = _event_id(event)
389
+ _require(
390
+ identifier != "",
391
+ f"{events_name} line {number} has no event.id",
392
+ remedy="every corpus event needs a stable authored event.id",
393
+ )
394
+ _require(
395
+ isinstance(event.get("@timestamp"), str) and event["@timestamp"].endswith("Z"),
396
+ f"{events_name} line {number} needs a UTC @timestamp ending in Z",
397
+ remedy="use an ISO-8601 UTC timestamp such as 2026-03-11T09:00:12Z",
398
+ )
399
+ event_ids.append(identifier)
400
+ parsed.append(event)
401
+ _require(
402
+ len(set(event_ids)) == len(event_ids),
403
+ f"{events_name}: duplicate event.id values",
404
+ remedy="every authored event.id must be unique inside the corpus",
405
+ )
406
+ declared = corpus.get("count")
407
+ _require(
408
+ declared == len(parsed),
409
+ f"{where}: corpus.count is {declared!r} but the file holds {len(parsed)} events",
410
+ remedy="update corpus.count",
411
+ )
412
+ detection_block = manifest.get("detection")
413
+ _require(
414
+ isinstance(detection_block, Mapping),
415
+ f"{where}: engine \"corpus\" requires a detection block",
416
+ remedy="describe the rule this dataset deploys",
417
+ )
418
+ assert isinstance(detection_block, Mapping)
419
+ detection = _load_detection(directory, detection_block, len(parsed))
420
+ rule_title = detection.rule_title
421
+ else:
422
+ _require(
423
+ manifest.get("corpus") is None and manifest.get("detection") is None,
424
+ f"{where}: engine {engine!r} must declare corpus and detection as null",
425
+ remedy="the golden-path seeder owns this dataset's events and rule",
426
+ )
427
+ # The seeder owns the rule, but `reset` still has to find it when a run
428
+ # dies before recording its id, so the title is manifest data rather
429
+ # than a literal only the shell knows (veltro#984).
430
+ golden = manifest.get("golden_path")
431
+ _require(
432
+ isinstance(golden, Mapping),
433
+ f"{where}: engine {engine!r} requires a golden_path block",
434
+ remedy="add golden_path.rule_title — the title the seeder gives its CHAD rule",
435
+ )
436
+ assert isinstance(golden, Mapping)
437
+ rule_title = _string(golden, "rule_title", where + " golden_path")
438
+
439
+ expect = manifest.get("expect")
440
+ _require(
441
+ isinstance(expect, Mapping),
442
+ f"{where}: expect block is required",
443
+ remedy="declare min_alerts and warden_case",
444
+ )
445
+ assert isinstance(expect, Mapping)
446
+ min_alerts = expect.get("min_alerts")
447
+ _require(
448
+ isinstance(min_alerts, int) and min_alerts >= 1,
449
+ f"{where}: expect.min_alerts must be a positive integer",
450
+ remedy="a dataset that cannot raise an alert does not belong in the sandbox",
451
+ )
452
+ _require(
453
+ isinstance(expect.get("warden_case"), bool),
454
+ f"{where}: expect.warden_case must be a boolean",
455
+ remedy="state whether the alert is expected to open a Warden case",
456
+ )
457
+ matching = expect.get("matching_event_ids", [])
458
+ _require(
459
+ isinstance(matching, list) and all(isinstance(m, str) for m in matching),
460
+ f"{where}: expect.matching_event_ids must be a list of authored event ids",
461
+ remedy="name the corpus events the rule matches",
462
+ )
463
+ unknown = [m for m in matching if m not in event_ids]
464
+ _require(
465
+ not unknown,
466
+ f"{where}: expect.matching_event_ids not present in the corpus: {', '.join(unknown)}",
467
+ remedy="use authored event.id values from events.ndjson",
468
+ )
469
+
470
+
471
+ return Dataset(
472
+ id=dataset_id,
473
+ title=_string(manifest, "title", where),
474
+ summary=_string(manifest, "summary", where),
475
+ engine=engine,
476
+ directory=directory,
477
+ attack_techniques=tuple(str(t) for t in techniques or ()),
478
+ events_path=events_path,
479
+ detection=detection,
480
+ rule_title=rule_title,
481
+ min_alerts=int(min_alerts), # type: ignore[arg-type]
482
+ expect_warden_case=bool(expect["warden_case"]),
483
+ matching_event_ids=tuple(str(m) for m in matching or ()),
484
+ )
485
+
486
+
487
+ def datasets_root(root: Path) -> Path:
488
+ return Path(root).joinpath(*CONTENT_SUBDIR)
489
+
490
+
491
+ def load_catalog(root: Path) -> dict[str, Dataset]:
492
+ """Every dataset under ``<root>/content/datasets``, keyed by id."""
493
+ base = datasets_root(root)
494
+ if not base.is_dir():
495
+ raise DatasetError(
496
+ f"{base} not found",
497
+ remedy="run from an installed suite directory or a repository checkout",
498
+ )
499
+ catalog: dict[str, Dataset] = {}
500
+ for directory in sorted(p for p in base.iterdir() if p.is_dir()):
501
+ if not (directory / MANIFEST_NAME).is_file():
502
+ continue
503
+ dataset = load_dataset(directory)
504
+ catalog[dataset.id] = dataset
505
+ _require(
506
+ bool(catalog),
507
+ f"{base} holds no datasets",
508
+ remedy="restore content/datasets from the release",
509
+ )
510
+ return catalog
511
+
512
+
513
+ def split_dataset_names(values: Iterable[str]) -> list[str]:
514
+ """Accept ``--dataset a --dataset b`` and ``--dataset a,b`` alike."""
515
+ names: list[str] = []
516
+ for value in values:
517
+ for part in str(value).split(","):
518
+ part = part.strip()
519
+ if part:
520
+ names.append(part)
521
+ return names
522
+
523
+
524
+ def resolve_datasets(root: Path, names: Sequence[str]) -> list[Dataset]:
525
+ """Selected datasets, de-duplicated, golden-path first.
526
+
527
+ The golden-path dataset is ordered first deliberately: it provisions the
528
+ VectorFlow pipeline, the CHAD push data source and the Warden bindings the
529
+ corpus datasets replay through.
530
+ """
531
+ catalog = load_catalog(root)
532
+ requested = split_dataset_names(names)
533
+ if not requested or requested == [ALL_KEYWORD]:
534
+ selected = list(catalog.values()) if requested == [ALL_KEYWORD] else []
535
+ else:
536
+ selected = []
537
+ for name in requested:
538
+ if name == ALL_KEYWORD:
539
+ selected.extend(catalog.values())
540
+ continue
541
+ dataset = catalog.get(name)
542
+ if dataset is None:
543
+ raise DatasetError(
544
+ f"unknown dataset {name!r}",
545
+ remedy=f"available datasets: {', '.join(sorted(catalog))}",
546
+ )
547
+ selected.append(dataset)
548
+ seen: set[str] = set()
549
+ unique: list[Dataset] = []
550
+ for dataset in selected:
551
+ if dataset.id in seen:
552
+ continue
553
+ seen.add(dataset.id)
554
+ unique.append(dataset)
555
+ unique.sort(key=lambda d: (d.engine != GOLDEN_PATH_ENGINE, d.id))
556
+ return unique
557
+
558
+
559
+ def _parse_timestamp(value: str) -> datetime:
560
+ try:
561
+ return datetime.strptime(value, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc)
562
+ except ValueError as exc:
563
+ raise DatasetError(
564
+ f"corpus timestamp {value!r} is not `YYYY-MM-DDTHH:MM:SSZ`",
565
+ remedy="use whole-second UTC timestamps in the corpus",
566
+ ) from exc
567
+
568
+
569
+ def rebase_events(
570
+ dataset: Dataset,
571
+ *,
572
+ now: datetime,
573
+ run_token: str,
574
+ lead_seconds: int = 30,
575
+ ) -> list[dict[str, Any]]:
576
+ """Shift a corpus onto the wall clock and stamp it as demo data.
577
+
578
+ The newest authored event lands ``lead_seconds`` before ``now`` and every
579
+ earlier event keeps its exact offset, so a threshold rule sees the interval
580
+ the corpus author intended. Each ``event.id`` becomes
581
+ ``demo-<dataset>-<run token>-<authored id>``: a re-seed produces distinct
582
+ alerts, and any assertion can follow exactly its own run.
583
+ """
584
+ events = dataset.events()
585
+ if not events:
586
+ return []
587
+ stamps = [_parse_timestamp(str(event["@timestamp"])) for event in events]
588
+ newest = max(stamps)
589
+ target = now.astimezone(timezone.utc) - timedelta(seconds=lead_seconds)
590
+ shift = target - newest
591
+ prepared: list[dict[str, Any]] = []
592
+ for event, stamp in zip(events, stamps):
593
+ document = json.loads(json.dumps(event)) # deep copy, corpus stays untouched
594
+ document["@timestamp"] = (stamp + shift).strftime("%Y-%m-%dT%H:%M:%SZ")
595
+ block = document.setdefault("event", {})
596
+ authored = str(block.get("id", ""))
597
+ block["id"] = f"demo-{dataset.id}-{run_token}-{authored}"
598
+ block["original_id"] = authored
599
+ labels = document.setdefault("labels", {})
600
+ if isinstance(labels, dict):
601
+ labels[DEMO_LABEL_FIELD] = dataset.id
602
+ labels["veltro_demo_run"] = run_token
603
+ tags = document.get("tags")
604
+ document["tags"] = sorted({*(tags if isinstance(tags, list) else []), DEMO_TAG, f"{DEMO_TAG}:{dataset.id}"})
605
+ prepared.append(document)
606
+ return prepared
607
+
608
+
609
+ def replay_event_prefix(dataset: Dataset, run_token: str) -> str:
610
+ """The ``event.id`` prefix every event of one replay carries."""
611
+ return f"demo-{dataset.id}-{run_token}-"