veltro-cli 0.15.7__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- veltro_cli/__init__.py +6 -0
- veltro_cli/__main__.py +7 -0
- veltro_cli/cli.py +478 -0
- veltro_cli/lib/__init__.py +3 -0
- veltro_cli/lib/agent.py +37 -0
- veltro_cli/lib/backup.py +139 -0
- veltro_cli/lib/compose.py +276 -0
- veltro_cli/lib/datasets.py +611 -0
- veltro_cli/lib/demo.py +421 -0
- veltro_cli/lib/demo_replay.py +794 -0
- veltro_cli/lib/doctor.py +99 -0
- veltro_cli/lib/env_init.py +93 -0
- veltro_cli/lib/errors.py +40 -0
- veltro_cli/lib/host.py +346 -0
- veltro_cli/lib/install.py +400 -0
- veltro_cli/lib/k8s.py +57 -0
- veltro_cli/lib/paths.py +42 -0
- veltro_cli/lib/products.py +54 -0
- veltro_cli/lib/release.py +88 -0
- veltro_cli/lib/status.py +170 -0
- veltro_cli/lib/suite_http.py +227 -0
- veltro_cli/lib/support.py +35 -0
- veltro_cli/lib/tls.py +113 -0
- veltro_cli/lib/toml_config.py +130 -0
- veltro_cli/lib/upgrade.py +379 -0
- veltro_cli/py.typed +0 -0
- veltro_cli-0.15.7.dist-info/METADATA +40 -0
- veltro_cli-0.15.7.dist-info/RECORD +31 -0
- veltro_cli-0.15.7.dist-info/WHEEL +5 -0
- veltro_cli-0.15.7.dist-info/entry_points.txt +2 -0
- veltro_cli-0.15.7.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,611 @@
|
|
|
1
|
+
"""Demo dataset catalogue: discovery, validation, selection and replay shaping.
|
|
2
|
+
|
|
3
|
+
The corpora live in ``content/datasets/<id>/`` (Apache-2.0) and are described by
|
|
4
|
+
a ``veltro-demo-dataset/v1`` manifest. This module is the only reader: the CLI
|
|
5
|
+
(`veltro demo seed --dataset …`), the replay engine
|
|
6
|
+
(`veltro_cli.lib.demo_replay`) and the CI validator
|
|
7
|
+
(`scripts/test_demo_datasets.py`) all go through it, so a corpus cannot ship a
|
|
8
|
+
manifest that one consumer accepts and another rejects.
|
|
9
|
+
|
|
10
|
+
Everything here is pure: no network, no clock except what the caller passes in.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import re
|
|
17
|
+
from dataclasses import dataclass
|
|
18
|
+
from datetime import datetime, timedelta, timezone
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import Any, Iterable, Mapping, Sequence
|
|
21
|
+
|
|
22
|
+
from veltro_cli.lib.errors import VeltroCliError
|
|
23
|
+
|
|
24
|
+
MANIFEST_SCHEMA = "veltro-demo-dataset/v1"
|
|
25
|
+
MANIFEST_NAME = "dataset.json"
|
|
26
|
+
CONTENT_SUBDIR = ("content", "datasets")
|
|
27
|
+
GOLDEN_PATH_ENGINE = "golden-path-seed"
|
|
28
|
+
CORPUS_ENGINE = "corpus"
|
|
29
|
+
ENGINES = (GOLDEN_PATH_ENGINE, CORPUS_ENGINE)
|
|
30
|
+
REQUIRED_LICENSE = "Apache-2.0"
|
|
31
|
+
APACHE_SPDX = "SPDX-License-Identifier: Apache-2.0"
|
|
32
|
+
ID_PATTERN = re.compile(r"^[a-z][a-z0-9-]{1,40}$")
|
|
33
|
+
# The label every replayed event carries, so `veltro demo reset` can find demo
|
|
34
|
+
# data again and an operator can tell it apart from their own telemetry.
|
|
35
|
+
DEMO_LABEL_FIELD = "veltro_demo"
|
|
36
|
+
DEMO_TAG = "veltro-demo"
|
|
37
|
+
ALL_KEYWORD = "all"
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
class DatasetError(VeltroCliError):
|
|
41
|
+
"""A dataset directory does not satisfy the manifest contract."""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
@dataclass(frozen=True)
|
|
45
|
+
class Threshold:
|
|
46
|
+
count: int
|
|
47
|
+
window_minutes: int
|
|
48
|
+
group_by: str
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
@dataclass(frozen=True)
|
|
52
|
+
class Detection:
|
|
53
|
+
rule_title: str
|
|
54
|
+
description: str
|
|
55
|
+
severity: str
|
|
56
|
+
yaml_path: Path
|
|
57
|
+
field_mappings: tuple[tuple[str, str], ...]
|
|
58
|
+
threshold: Threshold | None
|
|
59
|
+
sample_event_index: int
|
|
60
|
+
|
|
61
|
+
def yaml_content(self) -> str:
|
|
62
|
+
return self.yaml_path.read_text(encoding="utf-8")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
@dataclass(frozen=True)
|
|
66
|
+
class Dataset:
|
|
67
|
+
id: str
|
|
68
|
+
title: str
|
|
69
|
+
summary: str
|
|
70
|
+
engine: str
|
|
71
|
+
directory: Path
|
|
72
|
+
attack_techniques: tuple[str, ...]
|
|
73
|
+
events_path: Path | None
|
|
74
|
+
detection: Detection | None
|
|
75
|
+
# The CHAD rule title this dataset's detection carries. Taken from
|
|
76
|
+
# `detection.rule_title` for a corpus dataset and from the manifest's
|
|
77
|
+
# `golden_path.rule_title` for the golden path, whose rule the shell
|
|
78
|
+
# seeder owns: `reset` resolves the rule by title when a run died before
|
|
79
|
+
# it could record the id (veltro#984), so every dataset has one.
|
|
80
|
+
rule_title: str
|
|
81
|
+
min_alerts: int
|
|
82
|
+
expect_warden_case: bool
|
|
83
|
+
matching_event_ids: tuple[str, ...]
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def is_corpus(self) -> bool:
|
|
87
|
+
return self.engine == CORPUS_ENGINE
|
|
88
|
+
|
|
89
|
+
def events(self) -> list[dict[str, Any]]:
|
|
90
|
+
"""The corpus, in authored order. Empty for the golden-path engine."""
|
|
91
|
+
if self.events_path is None:
|
|
92
|
+
return []
|
|
93
|
+
return [
|
|
94
|
+
json.loads(line)
|
|
95
|
+
for line in self.events_path.read_text(encoding="utf-8").splitlines()
|
|
96
|
+
if line.strip()
|
|
97
|
+
]
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _require(condition: bool, message: str, remedy: str) -> None:
|
|
101
|
+
if not condition:
|
|
102
|
+
raise DatasetError(message, remedy=remedy)
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def _string(mapping: Mapping[str, Any], key: str, where: str) -> str:
|
|
106
|
+
value = mapping.get(key)
|
|
107
|
+
_require(
|
|
108
|
+
isinstance(value, str) and value.strip() != "",
|
|
109
|
+
f"{where}: {key!r} must be a non-empty string",
|
|
110
|
+
remedy=f"set {key!r} in {where}",
|
|
111
|
+
)
|
|
112
|
+
return str(value).strip()
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _event_id(event: Mapping[str, Any]) -> str:
|
|
116
|
+
event_block = event.get("event")
|
|
117
|
+
if isinstance(event_block, Mapping):
|
|
118
|
+
candidate = event_block.get("id")
|
|
119
|
+
if isinstance(candidate, str) and candidate.strip():
|
|
120
|
+
return candidate.strip()
|
|
121
|
+
return ""
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
def field_value(event: Mapping[str, Any], dotted: str) -> Any:
|
|
125
|
+
"""Read a dotted ECS path out of a nested event document."""
|
|
126
|
+
cursor: Any = event
|
|
127
|
+
for part in dotted.split("."):
|
|
128
|
+
if not isinstance(cursor, Mapping) or part not in cursor:
|
|
129
|
+
return None
|
|
130
|
+
cursor = cursor[part]
|
|
131
|
+
return cursor
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def parse_simple_detection(rule_yaml: str) -> dict[str, str]:
|
|
135
|
+
"""Return the ``selection`` field/value pairs of a shipped demo rule.
|
|
136
|
+
|
|
137
|
+
Demo rules are deliberately restricted to one flat ``selection`` mapping
|
|
138
|
+
with ``condition: selection`` — the shape a stdlib reader can parse exactly,
|
|
139
|
+
with no YAML dependency and no partial interpretation. CHAD's pySigma engine
|
|
140
|
+
remains the real evaluator; this parser exists so CI can prove a shipped
|
|
141
|
+
rule matches the events its manifest claims, and it refuses anything richer
|
|
142
|
+
rather than guessing.
|
|
143
|
+
"""
|
|
144
|
+
lines = rule_yaml.splitlines()
|
|
145
|
+
try:
|
|
146
|
+
start = next(i for i, line in enumerate(lines) if line.rstrip() == "detection:")
|
|
147
|
+
except StopIteration as exc:
|
|
148
|
+
raise DatasetError(
|
|
149
|
+
"rule has no `detection:` block",
|
|
150
|
+
remedy="a demo rule needs a detection block with one flat selection",
|
|
151
|
+
) from exc
|
|
152
|
+
selection: dict[str, str] = {}
|
|
153
|
+
condition = ""
|
|
154
|
+
in_selection = False
|
|
155
|
+
for line in lines[start + 1 :]:
|
|
156
|
+
if line.strip() == "" or line.lstrip().startswith("#"):
|
|
157
|
+
continue
|
|
158
|
+
indent = len(line) - len(line.lstrip())
|
|
159
|
+
if indent == 0:
|
|
160
|
+
break
|
|
161
|
+
stripped = line.strip()
|
|
162
|
+
if indent == 2:
|
|
163
|
+
in_selection = stripped == "selection:"
|
|
164
|
+
if stripped.startswith("condition:"):
|
|
165
|
+
condition = stripped.split(":", 1)[1].strip()
|
|
166
|
+
elif not in_selection:
|
|
167
|
+
raise DatasetError(
|
|
168
|
+
f"unsupported detection key {stripped!r}",
|
|
169
|
+
remedy="demo rules support exactly `selection:` and `condition: selection`",
|
|
170
|
+
)
|
|
171
|
+
continue
|
|
172
|
+
if indent == 4 and in_selection:
|
|
173
|
+
_require(
|
|
174
|
+
":" in stripped,
|
|
175
|
+
f"unsupported selection entry {stripped!r}",
|
|
176
|
+
remedy="each selection entry must be `field: value`",
|
|
177
|
+
)
|
|
178
|
+
key, _, value = stripped.partition(":")
|
|
179
|
+
_require(
|
|
180
|
+
"|" not in key,
|
|
181
|
+
f"selection modifier in {key!r} is not supported by the demo validator",
|
|
182
|
+
remedy="keep demo rule selections to plain equality so CI can verify them",
|
|
183
|
+
)
|
|
184
|
+
selection[key.strip()] = value.strip().strip("'\"")
|
|
185
|
+
continue
|
|
186
|
+
raise DatasetError(
|
|
187
|
+
f"unsupported detection line {line!r}",
|
|
188
|
+
remedy="demo rules support exactly one flat selection mapping",
|
|
189
|
+
)
|
|
190
|
+
_require(
|
|
191
|
+
condition == "selection",
|
|
192
|
+
"demo rules must use `condition: selection`",
|
|
193
|
+
remedy="simplify the rule condition, or move the rule out of content/datasets",
|
|
194
|
+
)
|
|
195
|
+
_require(
|
|
196
|
+
bool(selection),
|
|
197
|
+
"demo rule selection is empty",
|
|
198
|
+
remedy="add at least one `field: value` entry under selection",
|
|
199
|
+
)
|
|
200
|
+
return selection
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
def selection_matches(selection: Mapping[str, str], event: Mapping[str, Any]) -> bool:
|
|
204
|
+
"""True when every selection field equals its value on this event."""
|
|
205
|
+
return all(str(field_value(event, field)) == value for field, value in selection.items())
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def _load_detection(directory: Path, block: Mapping[str, Any], event_count: int) -> Detection:
|
|
209
|
+
where = f"{directory.name}/dataset.json detection"
|
|
210
|
+
yaml_name = _string(block, "yaml", where)
|
|
211
|
+
yaml_path = directory / yaml_name
|
|
212
|
+
_require(
|
|
213
|
+
yaml_path.is_file(),
|
|
214
|
+
f"{where}: rule file {yaml_name!r} is missing",
|
|
215
|
+
remedy=f"add {directory.name}/{yaml_name}",
|
|
216
|
+
)
|
|
217
|
+
head = "".join(yaml_path.read_text(encoding="utf-8").splitlines(keepends=True)[:5])
|
|
218
|
+
_require(
|
|
219
|
+
APACHE_SPDX in head,
|
|
220
|
+
f"{where}: {yaml_name} has no Apache-2.0 SPDX header",
|
|
221
|
+
remedy=f"add `# {APACHE_SPDX}` as the first line of {yaml_name}",
|
|
222
|
+
)
|
|
223
|
+
raw_mappings = block.get("field_mappings")
|
|
224
|
+
_require(
|
|
225
|
+
isinstance(raw_mappings, list) and len(raw_mappings) > 0,
|
|
226
|
+
f"{where}: field_mappings must be a non-empty list",
|
|
227
|
+
remedy="map every Sigma field the rule keys on to its indexed field",
|
|
228
|
+
)
|
|
229
|
+
mappings: list[tuple[str, str]] = []
|
|
230
|
+
for entry in raw_mappings or []:
|
|
231
|
+
_require(
|
|
232
|
+
isinstance(entry, Mapping),
|
|
233
|
+
f"{where}: each field mapping must be an object",
|
|
234
|
+
remedy="use {\"sigma_field\": …, \"target_field\": …}",
|
|
235
|
+
)
|
|
236
|
+
mappings.append(
|
|
237
|
+
(_string(entry, "sigma_field", where), _string(entry, "target_field", where))
|
|
238
|
+
)
|
|
239
|
+
threshold_block = block.get("threshold")
|
|
240
|
+
threshold: Threshold | None = None
|
|
241
|
+
if threshold_block is not None:
|
|
242
|
+
_require(
|
|
243
|
+
isinstance(threshold_block, Mapping),
|
|
244
|
+
f"{where}: threshold must be an object or null",
|
|
245
|
+
remedy="omit the threshold with null, or give count/window_minutes/group_by",
|
|
246
|
+
)
|
|
247
|
+
assert isinstance(threshold_block, Mapping)
|
|
248
|
+
count = threshold_block.get("count")
|
|
249
|
+
window = threshold_block.get("window_minutes")
|
|
250
|
+
_require(
|
|
251
|
+
isinstance(count, int) and count >= 1,
|
|
252
|
+
f"{where}: threshold.count must be a positive integer",
|
|
253
|
+
remedy="set threshold.count",
|
|
254
|
+
)
|
|
255
|
+
_require(
|
|
256
|
+
isinstance(window, int) and window >= 1,
|
|
257
|
+
f"{where}: threshold.window_minutes must be a positive integer",
|
|
258
|
+
remedy="set threshold.window_minutes",
|
|
259
|
+
)
|
|
260
|
+
threshold = Threshold(
|
|
261
|
+
count=int(count), # type: ignore[arg-type]
|
|
262
|
+
window_minutes=int(window), # type: ignore[arg-type]
|
|
263
|
+
group_by=_string(threshold_block, "group_by", where),
|
|
264
|
+
)
|
|
265
|
+
sample = block.get("sample")
|
|
266
|
+
_require(
|
|
267
|
+
isinstance(sample, Mapping) and isinstance(sample.get("event_index"), int),
|
|
268
|
+
f"{where}: sample.event_index must be an integer",
|
|
269
|
+
remedy="name the corpus event the rule's deploy-floor sample test uses",
|
|
270
|
+
)
|
|
271
|
+
assert isinstance(sample, Mapping)
|
|
272
|
+
index = int(sample["event_index"])
|
|
273
|
+
_require(
|
|
274
|
+
0 <= index < event_count,
|
|
275
|
+
f"{where}: sample.event_index {index} is outside the corpus (0..{event_count - 1})",
|
|
276
|
+
remedy="point sample.event_index at an event the rule matches",
|
|
277
|
+
)
|
|
278
|
+
return Detection(
|
|
279
|
+
rule_title=_string(block, "rule_title", where),
|
|
280
|
+
description=_string(block, "description", where),
|
|
281
|
+
severity=_string(block, "severity", where),
|
|
282
|
+
yaml_path=yaml_path,
|
|
283
|
+
field_mappings=tuple(mappings),
|
|
284
|
+
threshold=threshold,
|
|
285
|
+
sample_event_index=index,
|
|
286
|
+
)
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def load_dataset(directory: Path) -> Dataset:
|
|
290
|
+
"""Read and fully validate one dataset directory."""
|
|
291
|
+
directory = Path(directory)
|
|
292
|
+
manifest_path = directory / MANIFEST_NAME
|
|
293
|
+
_require(
|
|
294
|
+
manifest_path.is_file(),
|
|
295
|
+
f"{directory} has no {MANIFEST_NAME}",
|
|
296
|
+
remedy=f"add {directory.name}/{MANIFEST_NAME} (schema {MANIFEST_SCHEMA})",
|
|
297
|
+
)
|
|
298
|
+
try:
|
|
299
|
+
manifest = json.loads(manifest_path.read_text(encoding="utf-8"))
|
|
300
|
+
except json.JSONDecodeError as exc:
|
|
301
|
+
raise DatasetError(
|
|
302
|
+
f"{manifest_path} is not valid JSON: {exc}",
|
|
303
|
+
remedy="fix the manifest syntax",
|
|
304
|
+
) from exc
|
|
305
|
+
_require(
|
|
306
|
+
isinstance(manifest, Mapping),
|
|
307
|
+
f"{manifest_path} must contain a JSON object",
|
|
308
|
+
remedy="wrap the manifest in an object",
|
|
309
|
+
)
|
|
310
|
+
where = f"{directory.name}/{MANIFEST_NAME}"
|
|
311
|
+
_require(
|
|
312
|
+
manifest.get("schema") == MANIFEST_SCHEMA,
|
|
313
|
+
f"{where}: schema must be {MANIFEST_SCHEMA!r}",
|
|
314
|
+
remedy=f"set \"schema\": \"{MANIFEST_SCHEMA}\"",
|
|
315
|
+
)
|
|
316
|
+
dataset_id = _string(manifest, "id", where)
|
|
317
|
+
_require(
|
|
318
|
+
bool(ID_PATTERN.match(dataset_id)),
|
|
319
|
+
f"{where}: id {dataset_id!r} must be lowercase letters, digits and dashes",
|
|
320
|
+
remedy="rename the dataset id",
|
|
321
|
+
)
|
|
322
|
+
_require(
|
|
323
|
+
dataset_id == directory.name,
|
|
324
|
+
f"{where}: id {dataset_id!r} must equal the directory name {directory.name!r}",
|
|
325
|
+
remedy="rename the directory or the id so they agree",
|
|
326
|
+
)
|
|
327
|
+
_require(
|
|
328
|
+
manifest.get("license") == REQUIRED_LICENSE,
|
|
329
|
+
f"{where}: license must be {REQUIRED_LICENSE!r}",
|
|
330
|
+
remedy="demo corpora ship Apache-2.0; see content/datasets/LICENSE",
|
|
331
|
+
)
|
|
332
|
+
engine = _string(manifest, "engine", where)
|
|
333
|
+
_require(
|
|
334
|
+
engine in ENGINES,
|
|
335
|
+
f"{where}: engine must be one of {', '.join(ENGINES)}",
|
|
336
|
+
remedy="use \"corpus\" for a dataset that ships its own events",
|
|
337
|
+
)
|
|
338
|
+
techniques = manifest.get("attack_techniques")
|
|
339
|
+
_require(
|
|
340
|
+
isinstance(techniques, list) and all(isinstance(t, str) and t for t in techniques),
|
|
341
|
+
f"{where}: attack_techniques must be a list of technique ids",
|
|
342
|
+
remedy="list the ATT&CK technique ids the corpus depicts",
|
|
343
|
+
)
|
|
344
|
+
provenance = manifest.get("provenance")
|
|
345
|
+
_require(
|
|
346
|
+
isinstance(provenance, Mapping) and provenance.get("synthetic") is True,
|
|
347
|
+
f"{where}: provenance.synthetic must be true",
|
|
348
|
+
remedy="only authored, synthetic corpora may ship here — see content/datasets/README.md",
|
|
349
|
+
)
|
|
350
|
+
events_path: Path | None = None
|
|
351
|
+
detection: Detection | None = None
|
|
352
|
+
rule_title = ""
|
|
353
|
+
event_ids: list[str] = []
|
|
354
|
+
if engine == CORPUS_ENGINE:
|
|
355
|
+
corpus = manifest.get("corpus")
|
|
356
|
+
_require(
|
|
357
|
+
isinstance(corpus, Mapping),
|
|
358
|
+
f"{where}: engine \"corpus\" requires a corpus block",
|
|
359
|
+
remedy="add corpus.events and corpus.count",
|
|
360
|
+
)
|
|
361
|
+
assert isinstance(corpus, Mapping)
|
|
362
|
+
events_name = _string(corpus, "events", where + " corpus")
|
|
363
|
+
events_path = directory / events_name
|
|
364
|
+
_require(
|
|
365
|
+
events_path.is_file(),
|
|
366
|
+
f"{where}: corpus file {events_name!r} is missing",
|
|
367
|
+
remedy=f"add {directory.name}/{events_name}",
|
|
368
|
+
)
|
|
369
|
+
lines = [
|
|
370
|
+
line
|
|
371
|
+
for line in events_path.read_text(encoding="utf-8").splitlines()
|
|
372
|
+
if line.strip()
|
|
373
|
+
]
|
|
374
|
+
parsed: list[dict[str, Any]] = []
|
|
375
|
+
for number, line in enumerate(lines, start=1):
|
|
376
|
+
try:
|
|
377
|
+
event = json.loads(line)
|
|
378
|
+
except json.JSONDecodeError as exc:
|
|
379
|
+
raise DatasetError(
|
|
380
|
+
f"{events_name} line {number} is not valid JSON: {exc}",
|
|
381
|
+
remedy="one JSON object per line, no trailing commas",
|
|
382
|
+
) from exc
|
|
383
|
+
_require(
|
|
384
|
+
isinstance(event, dict),
|
|
385
|
+
f"{events_name} line {number} must be a JSON object",
|
|
386
|
+
remedy="one JSON event object per line",
|
|
387
|
+
)
|
|
388
|
+
identifier = _event_id(event)
|
|
389
|
+
_require(
|
|
390
|
+
identifier != "",
|
|
391
|
+
f"{events_name} line {number} has no event.id",
|
|
392
|
+
remedy="every corpus event needs a stable authored event.id",
|
|
393
|
+
)
|
|
394
|
+
_require(
|
|
395
|
+
isinstance(event.get("@timestamp"), str) and event["@timestamp"].endswith("Z"),
|
|
396
|
+
f"{events_name} line {number} needs a UTC @timestamp ending in Z",
|
|
397
|
+
remedy="use an ISO-8601 UTC timestamp such as 2026-03-11T09:00:12Z",
|
|
398
|
+
)
|
|
399
|
+
event_ids.append(identifier)
|
|
400
|
+
parsed.append(event)
|
|
401
|
+
_require(
|
|
402
|
+
len(set(event_ids)) == len(event_ids),
|
|
403
|
+
f"{events_name}: duplicate event.id values",
|
|
404
|
+
remedy="every authored event.id must be unique inside the corpus",
|
|
405
|
+
)
|
|
406
|
+
declared = corpus.get("count")
|
|
407
|
+
_require(
|
|
408
|
+
declared == len(parsed),
|
|
409
|
+
f"{where}: corpus.count is {declared!r} but the file holds {len(parsed)} events",
|
|
410
|
+
remedy="update corpus.count",
|
|
411
|
+
)
|
|
412
|
+
detection_block = manifest.get("detection")
|
|
413
|
+
_require(
|
|
414
|
+
isinstance(detection_block, Mapping),
|
|
415
|
+
f"{where}: engine \"corpus\" requires a detection block",
|
|
416
|
+
remedy="describe the rule this dataset deploys",
|
|
417
|
+
)
|
|
418
|
+
assert isinstance(detection_block, Mapping)
|
|
419
|
+
detection = _load_detection(directory, detection_block, len(parsed))
|
|
420
|
+
rule_title = detection.rule_title
|
|
421
|
+
else:
|
|
422
|
+
_require(
|
|
423
|
+
manifest.get("corpus") is None and manifest.get("detection") is None,
|
|
424
|
+
f"{where}: engine {engine!r} must declare corpus and detection as null",
|
|
425
|
+
remedy="the golden-path seeder owns this dataset's events and rule",
|
|
426
|
+
)
|
|
427
|
+
# The seeder owns the rule, but `reset` still has to find it when a run
|
|
428
|
+
# dies before recording its id, so the title is manifest data rather
|
|
429
|
+
# than a literal only the shell knows (veltro#984).
|
|
430
|
+
golden = manifest.get("golden_path")
|
|
431
|
+
_require(
|
|
432
|
+
isinstance(golden, Mapping),
|
|
433
|
+
f"{where}: engine {engine!r} requires a golden_path block",
|
|
434
|
+
remedy="add golden_path.rule_title — the title the seeder gives its CHAD rule",
|
|
435
|
+
)
|
|
436
|
+
assert isinstance(golden, Mapping)
|
|
437
|
+
rule_title = _string(golden, "rule_title", where + " golden_path")
|
|
438
|
+
|
|
439
|
+
expect = manifest.get("expect")
|
|
440
|
+
_require(
|
|
441
|
+
isinstance(expect, Mapping),
|
|
442
|
+
f"{where}: expect block is required",
|
|
443
|
+
remedy="declare min_alerts and warden_case",
|
|
444
|
+
)
|
|
445
|
+
assert isinstance(expect, Mapping)
|
|
446
|
+
min_alerts = expect.get("min_alerts")
|
|
447
|
+
_require(
|
|
448
|
+
isinstance(min_alerts, int) and min_alerts >= 1,
|
|
449
|
+
f"{where}: expect.min_alerts must be a positive integer",
|
|
450
|
+
remedy="a dataset that cannot raise an alert does not belong in the sandbox",
|
|
451
|
+
)
|
|
452
|
+
_require(
|
|
453
|
+
isinstance(expect.get("warden_case"), bool),
|
|
454
|
+
f"{where}: expect.warden_case must be a boolean",
|
|
455
|
+
remedy="state whether the alert is expected to open a Warden case",
|
|
456
|
+
)
|
|
457
|
+
matching = expect.get("matching_event_ids", [])
|
|
458
|
+
_require(
|
|
459
|
+
isinstance(matching, list) and all(isinstance(m, str) for m in matching),
|
|
460
|
+
f"{where}: expect.matching_event_ids must be a list of authored event ids",
|
|
461
|
+
remedy="name the corpus events the rule matches",
|
|
462
|
+
)
|
|
463
|
+
unknown = [m for m in matching if m not in event_ids]
|
|
464
|
+
_require(
|
|
465
|
+
not unknown,
|
|
466
|
+
f"{where}: expect.matching_event_ids not present in the corpus: {', '.join(unknown)}",
|
|
467
|
+
remedy="use authored event.id values from events.ndjson",
|
|
468
|
+
)
|
|
469
|
+
|
|
470
|
+
|
|
471
|
+
return Dataset(
|
|
472
|
+
id=dataset_id,
|
|
473
|
+
title=_string(manifest, "title", where),
|
|
474
|
+
summary=_string(manifest, "summary", where),
|
|
475
|
+
engine=engine,
|
|
476
|
+
directory=directory,
|
|
477
|
+
attack_techniques=tuple(str(t) for t in techniques or ()),
|
|
478
|
+
events_path=events_path,
|
|
479
|
+
detection=detection,
|
|
480
|
+
rule_title=rule_title,
|
|
481
|
+
min_alerts=int(min_alerts), # type: ignore[arg-type]
|
|
482
|
+
expect_warden_case=bool(expect["warden_case"]),
|
|
483
|
+
matching_event_ids=tuple(str(m) for m in matching or ()),
|
|
484
|
+
)
|
|
485
|
+
|
|
486
|
+
|
|
487
|
+
def datasets_root(root: Path) -> Path:
|
|
488
|
+
return Path(root).joinpath(*CONTENT_SUBDIR)
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def load_catalog(root: Path) -> dict[str, Dataset]:
|
|
492
|
+
"""Every dataset under ``<root>/content/datasets``, keyed by id."""
|
|
493
|
+
base = datasets_root(root)
|
|
494
|
+
if not base.is_dir():
|
|
495
|
+
raise DatasetError(
|
|
496
|
+
f"{base} not found",
|
|
497
|
+
remedy="run from an installed suite directory or a repository checkout",
|
|
498
|
+
)
|
|
499
|
+
catalog: dict[str, Dataset] = {}
|
|
500
|
+
for directory in sorted(p for p in base.iterdir() if p.is_dir()):
|
|
501
|
+
if not (directory / MANIFEST_NAME).is_file():
|
|
502
|
+
continue
|
|
503
|
+
dataset = load_dataset(directory)
|
|
504
|
+
catalog[dataset.id] = dataset
|
|
505
|
+
_require(
|
|
506
|
+
bool(catalog),
|
|
507
|
+
f"{base} holds no datasets",
|
|
508
|
+
remedy="restore content/datasets from the release",
|
|
509
|
+
)
|
|
510
|
+
return catalog
|
|
511
|
+
|
|
512
|
+
|
|
513
|
+
def split_dataset_names(values: Iterable[str]) -> list[str]:
|
|
514
|
+
"""Accept ``--dataset a --dataset b`` and ``--dataset a,b`` alike."""
|
|
515
|
+
names: list[str] = []
|
|
516
|
+
for value in values:
|
|
517
|
+
for part in str(value).split(","):
|
|
518
|
+
part = part.strip()
|
|
519
|
+
if part:
|
|
520
|
+
names.append(part)
|
|
521
|
+
return names
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def resolve_datasets(root: Path, names: Sequence[str]) -> list[Dataset]:
|
|
525
|
+
"""Selected datasets, de-duplicated, golden-path first.
|
|
526
|
+
|
|
527
|
+
The golden-path dataset is ordered first deliberately: it provisions the
|
|
528
|
+
VectorFlow pipeline, the CHAD push data source and the Warden bindings the
|
|
529
|
+
corpus datasets replay through.
|
|
530
|
+
"""
|
|
531
|
+
catalog = load_catalog(root)
|
|
532
|
+
requested = split_dataset_names(names)
|
|
533
|
+
if not requested or requested == [ALL_KEYWORD]:
|
|
534
|
+
selected = list(catalog.values()) if requested == [ALL_KEYWORD] else []
|
|
535
|
+
else:
|
|
536
|
+
selected = []
|
|
537
|
+
for name in requested:
|
|
538
|
+
if name == ALL_KEYWORD:
|
|
539
|
+
selected.extend(catalog.values())
|
|
540
|
+
continue
|
|
541
|
+
dataset = catalog.get(name)
|
|
542
|
+
if dataset is None:
|
|
543
|
+
raise DatasetError(
|
|
544
|
+
f"unknown dataset {name!r}",
|
|
545
|
+
remedy=f"available datasets: {', '.join(sorted(catalog))}",
|
|
546
|
+
)
|
|
547
|
+
selected.append(dataset)
|
|
548
|
+
seen: set[str] = set()
|
|
549
|
+
unique: list[Dataset] = []
|
|
550
|
+
for dataset in selected:
|
|
551
|
+
if dataset.id in seen:
|
|
552
|
+
continue
|
|
553
|
+
seen.add(dataset.id)
|
|
554
|
+
unique.append(dataset)
|
|
555
|
+
unique.sort(key=lambda d: (d.engine != GOLDEN_PATH_ENGINE, d.id))
|
|
556
|
+
return unique
|
|
557
|
+
|
|
558
|
+
|
|
559
|
+
def _parse_timestamp(value: str) -> datetime:
|
|
560
|
+
try:
|
|
561
|
+
return datetime.strptime(value, "%Y-%m-%dT%H:%M:%SZ").replace(tzinfo=timezone.utc)
|
|
562
|
+
except ValueError as exc:
|
|
563
|
+
raise DatasetError(
|
|
564
|
+
f"corpus timestamp {value!r} is not `YYYY-MM-DDTHH:MM:SSZ`",
|
|
565
|
+
remedy="use whole-second UTC timestamps in the corpus",
|
|
566
|
+
) from exc
|
|
567
|
+
|
|
568
|
+
|
|
569
|
+
def rebase_events(
|
|
570
|
+
dataset: Dataset,
|
|
571
|
+
*,
|
|
572
|
+
now: datetime,
|
|
573
|
+
run_token: str,
|
|
574
|
+
lead_seconds: int = 30,
|
|
575
|
+
) -> list[dict[str, Any]]:
|
|
576
|
+
"""Shift a corpus onto the wall clock and stamp it as demo data.
|
|
577
|
+
|
|
578
|
+
The newest authored event lands ``lead_seconds`` before ``now`` and every
|
|
579
|
+
earlier event keeps its exact offset, so a threshold rule sees the interval
|
|
580
|
+
the corpus author intended. Each ``event.id`` becomes
|
|
581
|
+
``demo-<dataset>-<run token>-<authored id>``: a re-seed produces distinct
|
|
582
|
+
alerts, and any assertion can follow exactly its own run.
|
|
583
|
+
"""
|
|
584
|
+
events = dataset.events()
|
|
585
|
+
if not events:
|
|
586
|
+
return []
|
|
587
|
+
stamps = [_parse_timestamp(str(event["@timestamp"])) for event in events]
|
|
588
|
+
newest = max(stamps)
|
|
589
|
+
target = now.astimezone(timezone.utc) - timedelta(seconds=lead_seconds)
|
|
590
|
+
shift = target - newest
|
|
591
|
+
prepared: list[dict[str, Any]] = []
|
|
592
|
+
for event, stamp in zip(events, stamps):
|
|
593
|
+
document = json.loads(json.dumps(event)) # deep copy, corpus stays untouched
|
|
594
|
+
document["@timestamp"] = (stamp + shift).strftime("%Y-%m-%dT%H:%M:%SZ")
|
|
595
|
+
block = document.setdefault("event", {})
|
|
596
|
+
authored = str(block.get("id", ""))
|
|
597
|
+
block["id"] = f"demo-{dataset.id}-{run_token}-{authored}"
|
|
598
|
+
block["original_id"] = authored
|
|
599
|
+
labels = document.setdefault("labels", {})
|
|
600
|
+
if isinstance(labels, dict):
|
|
601
|
+
labels[DEMO_LABEL_FIELD] = dataset.id
|
|
602
|
+
labels["veltro_demo_run"] = run_token
|
|
603
|
+
tags = document.get("tags")
|
|
604
|
+
document["tags"] = sorted({*(tags if isinstance(tags, list) else []), DEMO_TAG, f"{DEMO_TAG}:{dataset.id}"})
|
|
605
|
+
prepared.append(document)
|
|
606
|
+
return prepared
|
|
607
|
+
|
|
608
|
+
|
|
609
|
+
def replay_event_prefix(dataset: Dataset, run_token: str) -> str:
|
|
610
|
+
"""The ``event.id`` prefix every event of one replay carries."""
|
|
611
|
+
return f"demo-{dataset.id}-{run_token}-"
|