agentmetry 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (131) hide show
  1. agentmetry/__init__.py +12 -0
  2. agentmetry/api/__init__.py +0 -0
  3. agentmetry/api/main.py +232 -0
  4. agentmetry/api/routes/__init__.py +0 -0
  5. agentmetry/api/routes/audit.py +396 -0
  6. agentmetry/api/websocket.py +53 -0
  7. agentmetry/api/ws_bridge.py +34 -0
  8. agentmetry/cli/__init__.py +941 -0
  9. agentmetry/cli/__main__.py +5 -0
  10. agentmetry/core/__init__.py +0 -0
  11. agentmetry/core/audit/__init__.py +1 -0
  12. agentmetry/core/audit/adapters/__init__.py +0 -0
  13. agentmetry/core/audit/adapters/agt.py +304 -0
  14. agentmetry/core/audit/adapters/cloudevents.py +159 -0
  15. agentmetry/core/audit/adapters/ecs.py +103 -0
  16. agentmetry/core/audit/adapters/splunk.py +40 -0
  17. agentmetry/core/audit/alerts.py +56 -0
  18. agentmetry/core/audit/canonical.py +150 -0
  19. agentmetry/core/audit/compliance_digest.py +299 -0
  20. agentmetry/core/audit/detection/__init__.py +9 -0
  21. agentmetry/core/audit/detection/benchmark.py +194 -0
  22. agentmetry/core/audit/detection/corpus/attack_approval_denied_then_executed.jsonl +3 -0
  23. agentmetry/core/audit/detection/corpus/attack_arbitrary_host_stage_execute.jsonl +3 -0
  24. agentmetry/core/audit/detection/corpus/attack_autonomous_unapproved_write.jsonl +3 -0
  25. agentmetry/core/audit/detection/corpus/attack_credential_exfil.jsonl +2 -0
  26. agentmetry/core/audit/detection/corpus/attack_credential_then_cloud_api.jsonl +2 -0
  27. agentmetry/core/audit/detection/corpus/attack_destructive_delete_burst.jsonl +6 -0
  28. agentmetry/core/audit/detection/corpus/attack_discovery_then_collect.jsonl +5 -0
  29. agentmetry/core/audit/detection/corpus/attack_dotfile_then_git_push.jsonl +2 -0
  30. agentmetry/core/audit/detection/corpus/attack_encoded_command_download.jsonl +1 -0
  31. agentmetry/core/audit/detection/corpus/attack_env_credential_exfil.jsonl +3 -0
  32. agentmetry/core/audit/detection/corpus/attack_hashed_only_no_command.jsonl +2 -0
  33. agentmetry/core/audit/detection/corpus/attack_interpreter_egress.jsonl +2 -0
  34. agentmetry/core/audit/detection/corpus/attack_pr_merged_without_review.jsonl +2 -0
  35. agentmetry/core/audit/detection/corpus/attack_proc_substitution_cradle.jsonl +2 -0
  36. agentmetry/core/audit/detection/corpus/attack_remote_pipe_to_shell.jsonl +1 -0
  37. agentmetry/core/audit/detection/corpus/attack_remote_staging_then_execute.jsonl +2 -0
  38. agentmetry/core/audit/detection/corpus/attack_session_tool_burst.jsonl +42 -0
  39. agentmetry/core/audit/detection/corpus/attack_single_command_exfil.jsonl +2 -0
  40. agentmetry/core/audit/detection/corpus/attack_ssh_directory_exfil.jsonl +3 -0
  41. agentmetry/core/audit/detection/corpus/attack_subagent_swarm.jsonl +6 -0
  42. agentmetry/core/audit/detection/corpus/attack_timestamp_collision.jsonl +2 -0
  43. agentmetry/core/audit/detection/corpus/attack_untrusted_input_then_action.jsonl +3 -0
  44. agentmetry/core/audit/detection/corpus/benign_authoring_merge_fixtures.jsonl +3 -0
  45. agentmetry/core/audit/detection/corpus/benign_autonomous_after_approval.jsonl +4 -0
  46. agentmetry/core/audit/detection/corpus/benign_build_artifact_cleanup.jsonl +5 -0
  47. agentmetry/core/audit/detection/corpus/benign_ci_artifact_download.jsonl +3 -0
  48. agentmetry/core/audit/detection/corpus/benign_database_migration.jsonl +5 -0
  49. agentmetry/core/audit/detection/corpus/benign_dependency_install_and_build.jsonl +5 -0
  50. agentmetry/core/audit/detection/corpus/benign_download_release_archive.jsonl +4 -0
  51. agentmetry/core/audit/detection/corpus/benign_fetch_data_then_run_repo_script.jsonl +3 -0
  52. agentmetry/core/audit/detection/corpus/benign_fetch_dataset_then_analyse.jsonl +3 -0
  53. agentmetry/core/audit/detection/corpus/benign_fetch_lockfile_then_install.jsonl +3 -0
  54. agentmetry/core/audit/detection/corpus/benign_git_review_and_push.jsonl +6 -0
  55. agentmetry/core/audit/detection/corpus/benign_human_driven_deletes.jsonl +6 -0
  56. agentmetry/core/audit/detection/corpus/benign_local_api_probing.jsonl +5 -0
  57. agentmetry/core/audit/detection/corpus/benign_long_but_calm_session.jsonl +30 -0
  58. agentmetry/core/audit/detection/corpus/benign_loopback_is_not_egress.jsonl +2 -0
  59. agentmetry/core/audit/detection/corpus/benign_loopback_pipe_to_interpreter.jsonl +3 -0
  60. agentmetry/core/audit/detection/corpus/benign_ordinary_development.jsonl +5 -0
  61. agentmetry/core/audit/detection/corpus/benign_package_manager_after_fetch.jsonl +2 -0
  62. agentmetry/core/audit/detection/corpus/benign_reading_config_that_is_not_secret.jsonl +5 -0
  63. agentmetry/core/audit/detection/corpus/benign_remote_api_call_no_credentials.jsonl +4 -0
  64. agentmetry/core/audit/detection/corpus/benign_research_then_docs.jsonl +5 -0
  65. agentmetry/core/audit/detection/corpus/benign_reversed_order_is_not_exfil.jsonl +2 -0
  66. agentmetry/core/audit/detection/corpus/benign_test_and_fix_loop.jsonl +6 -0
  67. agentmetry/core/audit/detection/corpus/benign_writing_about_credentials.jsonl +6 -0
  68. agentmetry/core/audit/detection/corpus/corpus.yaml +443 -0
  69. agentmetry/core/audit/detection/disposition.py +651 -0
  70. agentmetry/core/audit/detection/engine.py +78 -0
  71. agentmetry/core/audit/detection/live.py +127 -0
  72. agentmetry/core/audit/detection/live_store.py +355 -0
  73. agentmetry/core/audit/detection/models.py +53 -0
  74. agentmetry/core/audit/detection/rules.py +1314 -0
  75. agentmetry/core/audit/detection/traits.py +648 -0
  76. agentmetry/core/audit/detection/yaml_config.py +91 -0
  77. agentmetry/core/audit/detection/yaml_rules.py +83 -0
  78. agentmetry/core/audit/dlp/__init__.py +4 -0
  79. agentmetry/core/audit/dlp/loader.py +29 -0
  80. agentmetry/core/audit/dlp/models.py +29 -0
  81. agentmetry/core/audit/dlp/scanner.py +96 -0
  82. agentmetry/core/audit/dogfood.py +398 -0
  83. agentmetry/core/audit/evidence_pack.py +500 -0
  84. agentmetry/core/audit/external.py +213 -0
  85. agentmetry/core/audit/hashing.py +21 -0
  86. agentmetry/core/audit/hook_bootstrap.py +451 -0
  87. agentmetry/core/audit/identity.py +39 -0
  88. agentmetry/core/audit/ingest.py +242 -0
  89. agentmetry/core/audit/migrate.py +73 -0
  90. agentmetry/core/audit/mitre.py +244 -0
  91. agentmetry/core/audit/policy.py +99 -0
  92. agentmetry/core/audit/redaction.py +50 -0
  93. agentmetry/core/audit/replay.py +54 -0
  94. agentmetry/core/audit/run_context.py +129 -0
  95. agentmetry/core/audit/sinks.py +235 -0
  96. agentmetry/core/audit/spool.py +394 -0
  97. agentmetry/core/audit/tool_policy/__init__.py +4 -0
  98. agentmetry/core/audit/tool_policy/evaluator.py +198 -0
  99. agentmetry/core/audit/tool_policy/loader.py +44 -0
  100. agentmetry/core/audit/tool_policy/models.py +25 -0
  101. agentmetry/core/audit/trail_chain.py +300 -0
  102. agentmetry/core/audit/trail_db.py +491 -0
  103. agentmetry/core/audit/trail_merkle.py +332 -0
  104. agentmetry/core/auth.py +54 -0
  105. agentmetry/core/bus/__init__.py +5 -0
  106. agentmetry/core/bus/audit_exporter.py +107 -0
  107. agentmetry/core/bus/bridges.py +26 -0
  108. agentmetry/core/bus/bus.py +102 -0
  109. agentmetry/core/bus/events.py +50 -0
  110. agentmetry/core/bus/outbox.py +124 -0
  111. agentmetry/core/config.py +177 -0
  112. agentmetry/core/diagnostics/__init__.py +0 -0
  113. agentmetry/core/diagnostics/autostart.py +563 -0
  114. agentmetry/core/diagnostics/doctor.py +535 -0
  115. agentmetry/core/diagnostics/driver_paths.py +156 -0
  116. agentmetry/core/diagnostics/env_file.py +45 -0
  117. agentmetry/core/drivers/__init__.py +4 -0
  118. agentmetry/core/drivers/host.py +263 -0
  119. agentmetry/core/drivers/permissions.py +37 -0
  120. agentmetry/core/drivers/spec.py +118 -0
  121. agentmetry/core/extensions.py +107 -0
  122. agentmetry/core/health.py +26 -0
  123. agentmetry/core/version.py +13 -0
  124. agentmetry/policies/detection/manifest.yaml +43 -0
  125. agentmetry/policies/dlp/manifest.yaml +161 -0
  126. agentmetry/policies/opa/agent_rules.rego +33 -0
  127. agentmetry/policies/tool/manifest.yaml +117 -0
  128. agentmetry-0.4.0.dist-info/METADATA +86 -0
  129. agentmetry-0.4.0.dist-info/RECORD +131 -0
  130. agentmetry-0.4.0.dist-info/WHEEL +4 -0
  131. agentmetry-0.4.0.dist-info/entry_points.txt +2 -0
@@ -0,0 +1,535 @@
1
+ """Agentmetry doctor - SIEM preflight (manifests, trail chain, hooks, health).
2
+
3
+ Vault/drivers checks are optional-runtime extras and can only warn.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import json
9
+ import os
10
+ import shutil
11
+ import sys
12
+ from dataclasses import dataclass, field
13
+ from pathlib import Path
14
+ from typing import Literal
15
+
16
+ from agentmetry.core.config import settings
17
+ from agentmetry.core.diagnostics.driver_paths import (
18
+ default_python,
19
+ entry_has_absolute_paths,
20
+ normalize_drivers_file,
21
+ orchestrator_root,
22
+ resolve_driver_entry,
23
+ )
24
+ from agentmetry.core.drivers.spec import DriverSpec
25
+
26
+ Severity = Literal["ok", "warn", "fail"]
27
+
28
+
29
+ @dataclass
30
+ class Finding:
31
+ severity: Severity
32
+ code: str
33
+ message: str
34
+
35
+
36
+ @dataclass
37
+ class DoctorReport:
38
+ findings: list[Finding] = field(default_factory=list)
39
+
40
+ def ok(self, code: str, message: str) -> None:
41
+ self.findings.append(Finding("ok", code, message))
42
+
43
+ def warn(self, code: str, message: str) -> None:
44
+ self.findings.append(Finding("warn", code, message))
45
+
46
+ def fail(self, code: str, message: str) -> None:
47
+ self.findings.append(Finding("fail", code, message))
48
+
49
+ @property
50
+ def exit_code(self) -> int:
51
+ if any(f.severity == "fail" for f in self.findings):
52
+ return 1
53
+ return 0
54
+
55
+
56
+ def _check_health_endpoint(report: DoctorReport) -> None:
57
+ """Is the orchestrator up? A recorder that is not running records nothing."""
58
+ import urllib.request
59
+
60
+ url = settings.audit_ingest_url.rstrip("/") + "/api/v1/health"
61
+ try:
62
+ with urllib.request.urlopen(url, timeout=2) as resp:
63
+ if resp.status == 200:
64
+ report.ok("orchestrator_up", f"Orchestrator responding at {url}")
65
+ return
66
+ report.warn("orchestrator_up", f"Orchestrator returned HTTP {resp.status} at {url}")
67
+ except Exception:
68
+ report.warn(
69
+ "orchestrator_up",
70
+ f"Orchestrator not reachable at {url} - start it with `agentmetry start`",
71
+ )
72
+
73
+
74
+ #: Bind addresses that keep the API on this machine.
75
+ _LOOPBACK = frozenset({"127.0.0.1", "localhost", "::1", ""})
76
+
77
+
78
+ def _check_exposure(report: DoctorReport) -> None:
79
+ """Is the API reachable by anyone who cannot already read the trail?
80
+
81
+ `require_api_key` is deliberately a no-op when no key is set, which is the
82
+ right default for a recorder bound to loopback. Combined with a non-loopback
83
+ bind it is not a weak default, it is an open door: the ingest route accepts
84
+ forged events into the tamper-evident trail, the export route hands over the
85
+ whole evidence pack, and the disposition route lets a stranger close a
86
+ finding as accepted risk - a decision that is then written into the trail as
87
+ a legitimate human action.
88
+
89
+ The enterprise MSI reached exactly that combination by setting
90
+ AGENTMETRY_HOST=0.0.0.0 without setting a key, so this check fails rather
91
+ than warns. Any future packaging that repeats the mistake trips it here.
92
+ """
93
+ if not settings.fleet_id.strip():
94
+ report.warn(
95
+ "fleet_id",
96
+ "AGENTMETRY_FLEET_ID not set - fleet SIEM queries cannot scope to "
97
+ "this org or business unit",
98
+ )
99
+ else:
100
+ report.ok("fleet_id", f"Fleet id: {settings.fleet_id.strip()}")
101
+
102
+ host = os.environ.get("AGENTMETRY_HOST", "127.0.0.1").strip()
103
+ has_key = bool(settings.api_key.strip())
104
+
105
+ if host in _LOOPBACK:
106
+ detail = "loopback only" if has_key else "loopback only (no API key needed)"
107
+ report.ok("exposure", f"API bound to {host or '127.0.0.1'} - {detail}")
108
+ return
109
+
110
+ if has_key:
111
+ report.ok("exposure", f"API bound to {host} with an API key set")
112
+ return
113
+
114
+ report.fail(
115
+ "exposure",
116
+ f"API bound to {host} with NO API key. Anyone who can reach this host "
117
+ "can read the trail, export evidence, inject events, and close "
118
+ "detections. Set AGENTMETRY_API_KEY, or bind 127.0.0.1.",
119
+ )
120
+
121
+
122
+ def _check_hooks_installed(report: DoctorReport) -> None:
123
+ """Detect global hook installs. Absence is a warn: capture is opt-in per IDE."""
124
+ targets = {
125
+ "cursor": Path.home() / ".cursor" / "hooks.json",
126
+ "claude": Path.home() / ".claude" / "settings.json",
127
+ }
128
+ installed: list[str] = []
129
+ missing: list[str] = []
130
+ for name, path in targets.items():
131
+ try:
132
+ if path.is_file() and "agentmetry_ingest" in path.read_text(encoding="utf-8"):
133
+ installed.append(name)
134
+ else:
135
+ missing.append(name)
136
+ except OSError:
137
+ missing.append(name)
138
+ if installed:
139
+ report.ok("hooks", f"Hooks installed: {', '.join(installed)}")
140
+ if missing:
141
+ report.warn(
142
+ "hooks",
143
+ f"No hooks detected for: {', '.join(missing)} "
144
+ "(installed at orchestrator boot, or run scripts/install_*_hooks.ps1)",
145
+ )
146
+
147
+
148
+ def _check_trail(report: DoctorReport) -> None:
149
+ trail = Path(settings.audit_export_path)
150
+ if not trail.is_file():
151
+ report.warn(
152
+ "trail",
153
+ f"No trail yet at {trail.name} - run `python scripts/demo.py` or capture a session",
154
+ )
155
+ return
156
+ from agentmetry.core.audit.trail_chain import verify_trail_file
157
+
158
+ result = verify_trail_file(trail)
159
+ if result.ok:
160
+ report.ok("trail", f"Trail chain verified: {result.message}")
161
+ else:
162
+ report.fail("trail", f"Trail chain BROKEN: {result.message}")
163
+
164
+
165
+ def _check_triage(report: DoctorReport) -> None:
166
+ """Surface the triage backlog without making the operator open the UI.
167
+
168
+ A growing pile of undispositioned findings is the failure mode this product
169
+ is most exposed to: detection keeps working, nobody answers it, and the
170
+ evidence pack quietly says so. Warn, never fail - an untriaged detection is
171
+ a task, not a broken install.
172
+ """
173
+ from agentmetry.core.audit.detection.disposition import CLOSED_STATUSES, get_disposition_store
174
+
175
+ try:
176
+ counts = get_disposition_store().counts()
177
+ except Exception as exc: # a missing store must not sink the whole report
178
+ report.warn("triage", f"Could not read triage state: {exc}")
179
+ return
180
+
181
+ decided = sum(counts.values())
182
+ if not decided:
183
+ report.warn(
184
+ "triage",
185
+ "No detections have been dispositioned. Detections evidence that "
186
+ "the system noticed, not that anyone acted.",
187
+ )
188
+ return
189
+
190
+ open_findings = sum(n for s, n in counts.items() if s not in CLOSED_STATUSES)
191
+ report.ok(
192
+ "triage",
193
+ f"{decided} detection(s) dispositioned; {open_findings} still open",
194
+ )
195
+
196
+ # A decision about a rule that no longer exists is still evidence somebody
197
+ # reviewed something, so it is never dropped. It does need saying out loud,
198
+ # or an auditor reads a retired rule as an unreviewed finding.
199
+ try:
200
+ orphans = get_disposition_store().orphaned()
201
+ except Exception:
202
+ return
203
+ if orphans:
204
+ rules = sorted({str(o["rule_id"]) for o in orphans})
205
+ report.warn(
206
+ "triage_orphans",
207
+ f"{len(orphans)} disposition(s) reference rules that no longer exist "
208
+ f"({', '.join(rules[:3])}). Kept as evidence; add a RULE_ALIASES "
209
+ "entry if the rule was renamed rather than retired.",
210
+ )
211
+
212
+
213
+ def _check_autostart(report: DoctorReport) -> None:
214
+ """Say whether anything will restart the recorder without a human.
215
+
216
+ `agentmetry install` has existed for a while and nothing ever mentioned it.
217
+ On the machine where this check was written it had never been run, and the
218
+ result was five days of agent activity sitting in the hook spool while the
219
+ trail looked healthy. A capability nobody is told about is worth about as
220
+ much as one that does not exist.
221
+
222
+ A warning rather than a failure: running the recorder by hand is a
223
+ legitimate choice, and doctor should not fail an operator for making it.
224
+
225
+ A registration that exists but does not work is a different matter, and it
226
+ does fail. Nobody chose that, it looks identical to working from the
227
+ outside, and it is the state this check was in for a day: the task launched
228
+ a module path a package rename had removed, exited 1 every minute, and
229
+ doctor called it OK because something was registered.
230
+ """
231
+ from agentmetry.core.diagnostics import autostart
232
+
233
+ state = autostart.status()
234
+ if state.configured and state.healthy is False:
235
+ report.fail("autostart", f"Autostart is broken ({state.backend}): {state.detail}")
236
+ return
237
+ if state.configured:
238
+ report.ok("autostart", f"Starts automatically ({state.backend}): {state.detail}")
239
+ return
240
+ report.warn(
241
+ "autostart",
242
+ f"Nothing restarts the recorder ({state.backend}: {state.detail}). "
243
+ "Hooks keep capturing to the spool while it is down, and spooled events "
244
+ "expire after 7 days. Run `agentmetry install` to fix.",
245
+ )
246
+
247
+
248
+ # A backlog this deep, or this old, is no longer "the orchestrator restarted a
249
+ # moment ago". It means capture is not reaching the trail, and for a flight
250
+ # recorder that is a failure, not a note.
251
+ _SPOOL_FAIL_DEPTH = 100
252
+ _SPOOL_FAIL_AGE_SECONDS = 24 * 3600
253
+
254
+
255
+ def _check_spool(report: DoctorReport) -> None:
256
+ """Surface events the hooks captured but the trail has not accepted.
257
+
258
+ A small spool is normal for a moment after a restart and drains on a timer.
259
+ A large or old one means the orchestrator is not reachable from the hooks,
260
+ and the operator would otherwise see a healthy-looking trail that is quietly
261
+ missing sessions. Events past MAX_AGE_SECONDS stop being replayable, so the
262
+ age is a countdown, not a statistic.
263
+ """
264
+ from agentmetry.core.audit.spool import (
265
+ MAX_AGE_SECONDS,
266
+ expired_path,
267
+ spool_depth,
268
+ spool_oldest_age_seconds,
269
+ )
270
+
271
+ depth = spool_depth()
272
+ quarantined = expired_path()
273
+
274
+ if depth == 0:
275
+ if quarantined.is_file():
276
+ report.warn(
277
+ "spool",
278
+ f"Spool empty, but past events were quarantined unreplayed: {quarantined}",
279
+ )
280
+ else:
281
+ report.ok("spool", "No pending hook spool (capture is reaching the trail)")
282
+ return
283
+
284
+ age = spool_oldest_age_seconds() or 0.0
285
+ hours = age / 3600
286
+ message = f"{depth} event(s) pending replay; oldest {hours:.1f}h old"
287
+
288
+ if depth >= _SPOOL_FAIL_DEPTH or age >= _SPOOL_FAIL_AGE_SECONDS:
289
+ remaining = (MAX_AGE_SECONDS - age) / 3600
290
+ if remaining <= 0:
291
+ message += ". The oldest are past the replay window already"
292
+ else:
293
+ message += f". The oldest become unreplayable in {remaining:.0f}h"
294
+ message += ". Is the orchestrator running and reachable at the hook's base URL?"
295
+ report.fail("spool", message)
296
+ return
297
+
298
+ report.warn("spool", message)
299
+
300
+
301
+ def _check_manifests(report: DoctorReport) -> None:
302
+ dlp_path = Path(settings.dlp_rules_path)
303
+ if not dlp_path.is_file():
304
+ report.fail("dlp", f"DLP manifest missing at {dlp_path}")
305
+ else:
306
+ try:
307
+ from agentmetry.core.audit.dlp.loader import load_dlp_rules
308
+
309
+ rules = load_dlp_rules(dlp_path)
310
+ report.ok("dlp", f"{len(rules)} DLP rules load from {dlp_path.name}")
311
+ except Exception as exc:
312
+ report.fail("dlp", f"DLP manifest failed to load: {exc}")
313
+
314
+ tp_path = Path(settings.tool_policy_path)
315
+ if not tp_path.is_file():
316
+ report.fail("tool_policy", f"Tool policy manifest missing at {tp_path}")
317
+ else:
318
+ try:
319
+ from agentmetry.core.audit.tool_policy.loader import load_tool_policy
320
+
321
+ rules, default = load_tool_policy(tp_path)
322
+ report.ok(
323
+ "tool_policy",
324
+ f"{len(rules)} tool policy rules load (default: {default})",
325
+ )
326
+ except Exception as exc:
327
+ report.fail("tool_policy", f"Tool policy manifest failed to load: {exc}")
328
+
329
+ det_path = Path(settings.detection_rules_path)
330
+ if not det_path.is_file():
331
+ report.fail("detection", f"Detection manifest missing at {det_path}")
332
+ else:
333
+ try:
334
+ from agentmetry.core.audit.detection.yaml_config import load_manifest
335
+
336
+ manifest = load_manifest(reload=True)
337
+ thresholds = manifest.get("thresholds") or {}
338
+ count_rules = manifest.get("count_rules") or []
339
+ report.ok(
340
+ "detection",
341
+ f"{len(count_rules)} YAML count rules + {len(thresholds)} thresholds from {det_path.name}",
342
+ )
343
+ except Exception as exc:
344
+ report.fail("detection", f"Detection manifest failed to load: {exc}")
345
+
346
+ report.ok(
347
+ "hook_enforcement",
348
+ f"Tool policy={settings.tool_policy_mode}, DLP={settings.dlp_mode} "
349
+ "(set block in .env or install.ps1 -ToolPolicyBlock)",
350
+ )
351
+
352
+
353
+ def _check_optional_vault(
354
+ report: DoctorReport, vault: Path, *, fix_drivers: bool
355
+ ) -> None:
356
+ """Demo MCP vault checks - optional runtime, never a doctor failure.
357
+
358
+ The SIEM records IDE hook traffic with no vault at all. These checks only
359
+ run when a vault directory exists, and the worst they produce is a warn.
360
+ """
361
+ if not vault.is_dir():
362
+ report.ok("vault", "Demo MCP vault not present (optional) - skipped")
363
+ return
364
+
365
+ report.ok("vault", f"Demo vault found at {vault} (optional runtime)")
366
+ drivers_path = vault / ".system" / "drivers.json"
367
+ example_path = vault / ".system" / "drivers.json.example"
368
+
369
+ if not drivers_path.is_file():
370
+ if fix_drivers and example_path.is_file():
371
+ shutil.copy(example_path, drivers_path)
372
+ report.ok("drivers", f"Created {drivers_path.name} from drivers.json.example")
373
+ else:
374
+ report.warn(
375
+ "drivers",
376
+ f"No {drivers_path.name} - demo MCP drivers disabled "
377
+ "(copy drivers.json.example or run `agentmetry doctor --fix`)",
378
+ )
379
+ return
380
+
381
+ try:
382
+ raw = json.loads(drivers_path.read_text(encoding="utf-8"))
383
+ except json.JSONDecodeError as exc:
384
+ report.warn("drivers", f"drivers.json invalid JSON: {exc}")
385
+ return
386
+
387
+ drivers = raw.get("drivers") or []
388
+ report.ok("drivers", f"{len(drivers)} driver entries in drivers.json")
389
+
390
+ absolute_entries = [d.get("name", "?") for d in drivers if entry_has_absolute_paths(d)]
391
+ if absolute_entries:
392
+ if fix_drivers:
393
+ if normalize_drivers_file(drivers_path, vault_path=vault):
394
+ report.ok(
395
+ "drivers_portable",
396
+ "Rewrote drivers.json with {PYTHON}/{ORCHESTRATOR_ROOT}/{VAULT_PATH} tokens",
397
+ )
398
+ raw = json.loads(drivers_path.read_text(encoding="utf-8"))
399
+ drivers = raw.get("drivers") or []
400
+ absolute_entries = [
401
+ d.get("name", "?") for d in drivers if entry_has_absolute_paths(d)
402
+ ]
403
+ else:
404
+ report.warn("drivers_portable", "Nothing to rewrite in drivers.json")
405
+ if absolute_entries:
406
+ report.warn(
407
+ "drivers_absolute",
408
+ f"Machine-specific paths in: {', '.join(absolute_entries)} "
409
+ "(run `agentmetry doctor --fix`)",
410
+ )
411
+ else:
412
+ report.ok("drivers_portable", "drivers.json uses portable path tokens")
413
+
414
+ invalid: list[str] = []
415
+ for entry in drivers:
416
+ try:
417
+ DriverSpec.model_validate(resolve_driver_entry(entry, vault_path=vault))
418
+ except Exception:
419
+ invalid.append(str(entry.get("name", "?")))
420
+ if invalid:
421
+ report.warn("drivers_schema", f"Invalid driver entries: {', '.join(invalid)}")
422
+ else:
423
+ report.ok("drivers_schema", "All driver entries validate")
424
+
425
+
426
+ def _check_extensions(report: DoctorReport) -> None:
427
+ """Report enterprise extension packages (entry points) if installed."""
428
+ from agentmetry.core.extensions import _iter_extension_entry_points, get_extension_registry
429
+
430
+ registry = get_extension_registry()
431
+ if registry.loaded:
432
+ names = ", ".join(item.name for item in registry.loaded)
433
+ report.ok("extensions", f"Enterprise extensions loaded: {names}")
434
+ return
435
+
436
+ eps = list(_iter_extension_entry_points())
437
+ if not eps:
438
+ report.ok("extensions", "Open-source core (no enterprise extensions installed)")
439
+ return
440
+
441
+ names = ", ".join(sorted(ep.name for ep in eps))
442
+ report.ok(
443
+ "extensions",
444
+ f"Enterprise extension packages installed ({names}) - loaded on orchestrator start",
445
+ )
446
+
447
+
448
+ def run_doctor(
449
+ *,
450
+ vault_path: Path | None = None,
451
+ fix_drivers: bool = False,
452
+ ) -> DoctorReport:
453
+ """SIEM preflight. The recorder is the product; the demo vault is optional.
454
+
455
+ Order and severity reflect that: a missing DLP manifest or a broken trail
456
+ chain is a failure, a missing vault is not - the previous doctor hard-failed
457
+ on vault/drivers.json and returned early, so a recorder-only install (the
458
+ documented quick start) showed FAIL while capturing perfectly. Vault checks
459
+ now run last and can only warn.
460
+ """
461
+ report = DoctorReport()
462
+ orch = orchestrator_root()
463
+
464
+ # --- SIEM flight recorder ------------------------------------------------
465
+ # A source checkout has a pyproject next to the package; an installed one
466
+ # does not, and never will. Failing on its absence told every pip user their
467
+ # working install was broken as the first line of the first command they
468
+ # run, which is a poor way to meet someone.
469
+ if (orch / "pyproject.toml").is_file():
470
+ report.ok("orchestrator", f"Orchestrator root {orch}")
471
+ elif (Path(__file__).resolve().parents[2] / "__init__.py").is_file():
472
+ report.ok("orchestrator", f"Installed package at {Path(__file__).resolve().parents[2]}")
473
+ else:
474
+ report.fail("orchestrator", f"Expected orchestrator at {orch}")
475
+
476
+ py = Path(default_python())
477
+ if py.is_file():
478
+ report.ok("python", f"Python interpreter {py}")
479
+ else:
480
+ report.warn("python", f"Python not found at {py} - run pip install -e '.[dev]'")
481
+
482
+ env_file = orch / ".env"
483
+ if env_file.is_file():
484
+ report.ok("env", f"Found {env_file.name} (secrets stay gitignored)")
485
+ else:
486
+ report.warn("env", f"No {env_file} - copy from .env.example if needed")
487
+
488
+ data_dir = orch / "data"
489
+ try:
490
+ data_dir.mkdir(parents=True, exist_ok=True)
491
+ probe = data_dir / ".doctor-probe"
492
+ probe.write_text("ok", encoding="utf-8")
493
+ probe.unlink()
494
+ report.ok("data", f"Data directory writable: {data_dir}")
495
+ except OSError as exc:
496
+ report.fail("data", f"Data directory not writable ({data_dir}): {exc}")
497
+
498
+ _check_manifests(report)
499
+ _check_exposure(report)
500
+ _check_trail(report)
501
+ _check_triage(report)
502
+ _check_spool(report)
503
+ _check_autostart(report)
504
+ _check_health_endpoint(report)
505
+ _check_hooks_installed(report)
506
+ _check_extensions(report)
507
+
508
+ # --- Optional governed runtime (demo vault) ------------------------------
509
+ vault = Path(vault_path or settings.vault_path).resolve()
510
+ _check_optional_vault(report, vault, fix_drivers=fix_drivers)
511
+
512
+ return report
513
+
514
+
515
+ def format_report(report: DoctorReport) -> str:
516
+ lines: list[str] = []
517
+ for finding in report.findings:
518
+ prefix = {"ok": "OK", "warn": "WARN", "fail": "FAIL"}[finding.severity]
519
+ lines.append(f" [{prefix}] {finding.message}")
520
+ return "\n".join(lines)
521
+
522
+
523
+ def main(argv: list[str] | None = None) -> int:
524
+ import argparse
525
+
526
+ parser = argparse.ArgumentParser(prog="agentmetry-doctor")
527
+ parser.add_argument("--fix", action="store_true", help="rewrite drivers.json to portable tokens")
528
+ args = parser.parse_args(argv)
529
+ report = run_doctor(fix_drivers=args.fix)
530
+ print("Agentmetry doctor\n" + format_report(report))
531
+ return report.exit_code
532
+
533
+
534
+ if __name__ == "__main__":
535
+ sys.exit(main())
@@ -0,0 +1,156 @@
1
+ """Portable driver path tokens and drivers.json normalization."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import logging
7
+ import re
8
+ import sys
9
+ from pathlib import Path
10
+ from typing import Any
11
+
12
+ from pydantic import ValidationError
13
+
14
+ from agentmetry.core.config import settings
15
+ from agentmetry.core.drivers.spec import DriverSpec
16
+
17
+ logger = logging.getLogger(__name__)
18
+
19
+ _ORCH_ROOT = Path(__file__).resolve().parents[3]
20
+ _REPO_ROOT = _ORCH_ROOT.parents[1]
21
+
22
+ TOKEN_PYTHON = "{PYTHON}"
23
+ TOKEN_ORCH = "{ORCHESTRATOR_ROOT}"
24
+ TOKEN_REPO = "{REPO_ROOT}"
25
+ TOKEN_VAULT = "{VAULT_PATH}"
26
+
27
+ _PORTABLE_TOKENS = (TOKEN_PYTHON, TOKEN_ORCH, TOKEN_REPO, TOKEN_VAULT)
28
+
29
+ _ABSOLUTE_PATH_RE = re.compile(
30
+ r"(?:[A-Za-z]:[\\/]|/Users/|/home/|C:/Users/|C:\\Users\\)"
31
+ )
32
+
33
+
34
+ def orchestrator_root() -> Path:
35
+ return _ORCH_ROOT
36
+
37
+
38
+ def repo_root() -> Path:
39
+ return _REPO_ROOT
40
+
41
+
42
+ def default_python() -> str:
43
+ venv_py = _ORCH_ROOT / ".venv" / "Scripts" / "python.exe"
44
+ if venv_py.is_file():
45
+ return str(venv_py)
46
+ venv_bin = _ORCH_ROOT / ".venv" / "bin" / "python"
47
+ if venv_bin.is_file():
48
+ return str(venv_bin)
49
+ return sys.executable
50
+
51
+
52
+ def placeholder_map(vault_path: Path | None = None) -> dict[str, str]:
53
+ vault = Path(vault_path or settings.vault_path).resolve()
54
+ return {
55
+ TOKEN_PYTHON: default_python().replace("\\", "/"),
56
+ TOKEN_ORCH: str(_ORCH_ROOT.resolve()).replace("\\", "/"),
57
+ TOKEN_REPO: str(_REPO_ROOT.resolve()).replace("\\", "/"),
58
+ TOKEN_VAULT: str(vault).replace("\\", "/"),
59
+ }
60
+
61
+
62
+ def expand_placeholders(value: str, *, vault_path: Path | None = None) -> str:
63
+ text = value.replace("\\", "/")
64
+ for token, resolved in placeholder_map(vault_path).items():
65
+ text = text.replace(token, resolved)
66
+ return text
67
+
68
+
69
+ def collapse_to_portable(value: str, *, vault_path: Path | None = None) -> str:
70
+ text = value.replace("\\", "/")
71
+ mapping = placeholder_map(vault_path)
72
+ py = mapping[TOKEN_PYTHON]
73
+ if py and py in text:
74
+ return text.replace(py, TOKEN_PYTHON)
75
+ for token, resolved in sorted(mapping.items(), key=lambda kv: len(kv[1]), reverse=True):
76
+ if token == TOKEN_PYTHON:
77
+ continue
78
+ if resolved and resolved in text:
79
+ text = text.replace(resolved, token)
80
+ return text
81
+
82
+
83
+ def resolve_driver_entry(entry: dict[str, Any], *, vault_path: Path | None = None) -> dict[str, Any]:
84
+ resolved = dict(entry)
85
+ command = resolved.get("command")
86
+ if isinstance(command, str):
87
+ resolved["command"] = expand_placeholders(command, vault_path=vault_path)
88
+ args = resolved.get("args") or []
89
+ resolved["args"] = [
90
+ expand_placeholders(arg, vault_path=vault_path) if isinstance(arg, str) else arg
91
+ for arg in args
92
+ ]
93
+ return resolved
94
+
95
+
96
+ def normalize_driver_entry(entry: dict[str, Any], *, vault_path: Path | None = None) -> dict[str, Any]:
97
+ normalized = dict(entry)
98
+ command = normalized.get("command")
99
+ if isinstance(command, str):
100
+ normalized["command"] = collapse_to_portable(command, vault_path=vault_path)
101
+ args = normalized.get("args") or []
102
+ normalized["args"] = [
103
+ collapse_to_portable(arg, vault_path=vault_path) if isinstance(arg, str) else arg
104
+ for arg in args
105
+ ]
106
+ return normalized
107
+
108
+
109
+ def entry_has_absolute_paths(entry: dict[str, Any]) -> bool:
110
+ parts: list[str] = []
111
+ if isinstance(entry.get("command"), str):
112
+ parts.append(entry["command"])
113
+ for arg in entry.get("args") or []:
114
+ if isinstance(arg, str):
115
+ parts.append(arg)
116
+ combined = " ".join(parts)
117
+ if any(token in combined for token in _PORTABLE_TOKENS):
118
+ return False
119
+ return bool(_ABSOLUTE_PATH_RE.search(combined))
120
+
121
+
122
+ def load_resolved_driver_specs(config_path: Path, *, vault_path: Path | None = None) -> list[DriverSpec]:
123
+ if not config_path.exists():
124
+ return []
125
+ try:
126
+ data = json.loads(config_path.read_text(encoding="utf-8"))
127
+ except (OSError, json.JSONDecodeError) as exc:
128
+ logger.warning("drivers.json unreadable (%s) — no drivers mounted", exc)
129
+ return []
130
+
131
+ specs: list[DriverSpec] = []
132
+ for entry in data.get("drivers", []):
133
+ try:
134
+ resolved = resolve_driver_entry(entry, vault_path=vault_path)
135
+ spec = DriverSpec.model_validate(resolved)
136
+ except ValidationError as exc:
137
+ logger.warning("Skipping invalid driver entry %r: %s", entry.get("name"), exc)
138
+ continue
139
+ if not spec.enabled:
140
+ logger.info("Driver %s is disabled — skipped", spec.name)
141
+ continue
142
+ specs.append(spec)
143
+ return specs
144
+
145
+
146
+ def normalize_drivers_file(config_path: Path, *, vault_path: Path | None = None) -> bool:
147
+ if not config_path.exists():
148
+ return False
149
+ data = json.loads(config_path.read_text(encoding="utf-8"))
150
+ drivers = data.get("drivers") or []
151
+ normalized = [normalize_driver_entry(entry, vault_path=vault_path) for entry in drivers]
152
+ if normalized == drivers:
153
+ return False
154
+ data["drivers"] = normalized
155
+ config_path.write_text(json.dumps(data, indent=2) + "\n", encoding="utf-8")
156
+ return True