nat-engine 1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (299) hide show
  1. mannf/__init__.py +33 -0
  2. mannf/__main__.py +10 -0
  3. mannf/_version.py +8 -0
  4. mannf/agents/__init__.py +7 -0
  5. mannf/agents/analyzer_agent.py +9 -0
  6. mannf/agents/base.py +9 -0
  7. mannf/agents/bdi_agent.py +9 -0
  8. mannf/agents/belief_state.py +9 -0
  9. mannf/agents/coordinator_agent.py +9 -0
  10. mannf/agents/executor_agent.py +9 -0
  11. mannf/agents/monitor_agent.py +9 -0
  12. mannf/agents/oracle_agent.py +9 -0
  13. mannf/agents/planner_agent.py +9 -0
  14. mannf/agents/test_agent.py +9 -0
  15. mannf/anomaly/__init__.py +7 -0
  16. mannf/anomaly/enhanced_detector.py +9 -0
  17. mannf/cli.py +9 -0
  18. mannf/core/__init__.py +26 -0
  19. mannf/core/agents/__init__.py +52 -0
  20. mannf/core/agents/accessibility_scanner_agent.py +245 -0
  21. mannf/core/agents/analyzer_agent.py +224 -0
  22. mannf/core/agents/autonomous_loop_agent.py +1086 -0
  23. mannf/core/agents/autonomous_loop_models.py +62 -0
  24. mannf/core/agents/autonomous_run_differ.py +427 -0
  25. mannf/core/agents/base.py +128 -0
  26. mannf/core/agents/bdi_agent.py +330 -0
  27. mannf/core/agents/belief_state.py +202 -0
  28. mannf/core/agents/browser_coordinator_agent.py +224 -0
  29. mannf/core/agents/browser_executor_agent.py +410 -0
  30. mannf/core/agents/coordinator_agent.py +262 -0
  31. mannf/core/agents/executor_agent.py +222 -0
  32. mannf/core/agents/monitor_agent.py +188 -0
  33. mannf/core/agents/oracle_agent.py +150 -0
  34. mannf/core/agents/performance_testing_agent.py +279 -0
  35. mannf/core/agents/planner_agent.py +128 -0
  36. mannf/core/agents/test_agent.py +249 -0
  37. mannf/core/agents/visual_regression_agent.py +311 -0
  38. mannf/core/agents/web_crawler_agent.py +510 -0
  39. mannf/core/agents/worker_pool.py +366 -0
  40. mannf/core/anomaly/__init__.py +14 -0
  41. mannf/core/anomaly/enhanced_detector.py +541 -0
  42. mannf/core/browser/__init__.py +63 -0
  43. mannf/core/browser/accessibility_scanner.py +424 -0
  44. mannf/core/browser/discovery_model.py +178 -0
  45. mannf/core/browser/dom_snapshot.py +349 -0
  46. mannf/core/browser/ingestor_bridge.py +371 -0
  47. mannf/core/browser/performance_metrics.py +217 -0
  48. mannf/core/browser/reflection_analyzer.py +442 -0
  49. mannf/core/browser/scenario_generator.py +1100 -0
  50. mannf/core/browser/security_scenario_generator.py +695 -0
  51. mannf/core/browser/visual_comparer.py +159 -0
  52. mannf/core/diagnostics/__init__.py +28 -0
  53. mannf/core/diagnostics/failure_clusterer.py +211 -0
  54. mannf/core/diagnostics/flake_detector.py +233 -0
  55. mannf/core/diagnostics/root_cause_analyzer.py +273 -0
  56. mannf/core/distributed/__init__.py +16 -0
  57. mannf/core/distributed/endpoint.py +139 -0
  58. mannf/core/distributed/system_under_test.py +207 -0
  59. mannf/core/functional_orchestrator.py +428 -0
  60. mannf/core/messaging/__init__.py +11 -0
  61. mannf/core/messaging/bus.py +113 -0
  62. mannf/core/messaging/messages.py +89 -0
  63. mannf/core/nat_orchestrator.py +342 -0
  64. mannf/core/neural/__init__.py +183 -0
  65. mannf/core/orchestrator.py +272 -0
  66. mannf/core/prioritization/__init__.py +17 -0
  67. mannf/core/prioritization/adaptive_controller.py +509 -0
  68. mannf/core/prioritization/belief_prioritizer.py +231 -0
  69. mannf/core/prioritization/risk_scorer.py +430 -0
  70. mannf/core/reporting/__init__.py +12 -0
  71. mannf/core/reporting/unified_report.py +664 -0
  72. mannf/core/testing/__init__.py +17 -0
  73. mannf/core/testing/adaptive_controller.py +149 -0
  74. mannf/core/testing/models.py +179 -0
  75. mannf/core/validation/__init__.py +10 -0
  76. mannf/core/validation/self_validation_runner.py +180 -0
  77. mannf/dashboard/__init__.py +7 -0
  78. mannf/dashboard/app.py +9 -0
  79. mannf/dashboard/models.py +9 -0
  80. mannf/dashboard/static/index.html +2538 -0
  81. mannf/dashboard/telemetry.py +9 -0
  82. mannf/distributed/__init__.py +7 -0
  83. mannf/distributed/endpoint.py +9 -0
  84. mannf/distributed/system_under_test.py +9 -0
  85. mannf/healing/__init__.py +7 -0
  86. mannf/healing/graphql_schema_diff.py +9 -0
  87. mannf/healing/healer.py +9 -0
  88. mannf/healing/models.py +9 -0
  89. mannf/healing/schema_diff.py +9 -0
  90. mannf/integrations/__init__.py +7 -0
  91. mannf/integrations/auth.py +9 -0
  92. mannf/integrations/graphql_parser.py +9 -0
  93. mannf/integrations/graphql_sut.py +9 -0
  94. mannf/integrations/http_sut.py +9 -0
  95. mannf/integrations/openapi_parser.py +9 -0
  96. mannf/integrations/postman_parser.py +9 -0
  97. mannf/llm/__init__.py +7 -0
  98. mannf/llm/anthropic_provider.py +9 -0
  99. mannf/llm/base.py +9 -0
  100. mannf/llm/config.py +9 -0
  101. mannf/llm/factory.py +9 -0
  102. mannf/llm/openai_provider.py +9 -0
  103. mannf/llm/prompts.py +9 -0
  104. mannf/messaging/__init__.py +7 -0
  105. mannf/messaging/bus.py +9 -0
  106. mannf/messaging/messages.py +9 -0
  107. mannf/nat_orchestrator.py +9 -0
  108. mannf/neural/__init__.py +7 -0
  109. mannf/orchestrator.py +9 -0
  110. mannf/prioritization/__init__.py +7 -0
  111. mannf/prioritization/adaptive_controller.py +9 -0
  112. mannf/prioritization/belief_prioritizer.py +9 -0
  113. mannf/prioritization/risk_scorer.py +9 -0
  114. mannf/product/__init__.py +29 -0
  115. mannf/product/admin/__init__.py +3 -0
  116. mannf/product/admin/routes.py +514 -0
  117. mannf/product/auth/__init__.py +5 -0
  118. mannf/product/auth/saml.py +212 -0
  119. mannf/product/billing/__init__.py +5 -0
  120. mannf/product/billing/audit.py +160 -0
  121. mannf/product/billing/feature_gates.py +180 -0
  122. mannf/product/billing/metering.py +179 -0
  123. mannf/product/billing/notifications.py +181 -0
  124. mannf/product/billing/plans.py +133 -0
  125. mannf/product/billing/rate_limits.py +35 -0
  126. mannf/product/billing/stripe_billing.py +906 -0
  127. mannf/product/billing/tenant_auth.py +233 -0
  128. mannf/product/billing/tenant_manager.py +873 -0
  129. mannf/product/cli.py +3900 -0
  130. mannf/product/cli_admin.py +408 -0
  131. mannf/product/dashboard/__init__.py +61 -0
  132. mannf/product/dashboard/app.py +3567 -0
  133. mannf/product/dashboard/models.py +460 -0
  134. mannf/product/dashboard/static/index.html +6347 -0
  135. mannf/product/dashboard/static/manifest.json +25 -0
  136. mannf/product/dashboard/static/pwa-icon-192.png +0 -0
  137. mannf/product/dashboard/static/pwa-icon-512.png +0 -0
  138. mannf/product/dashboard/static/sw.js +64 -0
  139. mannf/product/dashboard/telemetry.py +547 -0
  140. mannf/product/database.py +145 -0
  141. mannf/product/demo.py +844 -0
  142. mannf/product/doctor.py +509 -0
  143. mannf/product/exporters/__init__.py +65 -0
  144. mannf/product/exporters/azuredevops_exporter.py +257 -0
  145. mannf/product/exporters/base.py +307 -0
  146. mannf/product/exporters/bugzilla_exporter.py +200 -0
  147. mannf/product/exporters/dedup.py +275 -0
  148. mannf/product/exporters/finding_adapter.py +216 -0
  149. mannf/product/exporters/github_exporter.py +197 -0
  150. mannf/product/exporters/gitlab_exporter.py +215 -0
  151. mannf/product/exporters/jira_exporter.py +180 -0
  152. mannf/product/exporters/linear_exporter.py +195 -0
  153. mannf/product/exporters/loader.py +233 -0
  154. mannf/product/exporters/pagerduty_exporter.py +363 -0
  155. mannf/product/exporters/sentry_exporter.py +322 -0
  156. mannf/product/exporters/servicenow_exporter.py +240 -0
  157. mannf/product/exporters/shortcut_exporter.py +231 -0
  158. mannf/product/exporters/webhook_exporter.py +383 -0
  159. mannf/product/formatters/__init__.py +18 -0
  160. mannf/product/formatters/allure_formatter.py +161 -0
  161. mannf/product/formatters/ctrf_formatter.py +149 -0
  162. mannf/product/healing/__init__.py +30 -0
  163. mannf/product/healing/graphql_schema_diff.py +152 -0
  164. mannf/product/healing/healer.py +141 -0
  165. mannf/product/healing/models.py +175 -0
  166. mannf/product/healing/schema_diff.py +251 -0
  167. mannf/product/ingestors/__init__.py +77 -0
  168. mannf/product/ingestors/base.py +256 -0
  169. mannf/product/ingestors/bgstm_ingestor.py +764 -0
  170. mannf/product/ingestors/curl_ingestor.py +1019 -0
  171. mannf/product/ingestors/cypress_ingestor.py +487 -0
  172. mannf/product/ingestors/gherkin_ingestor.py +967 -0
  173. mannf/product/ingestors/graphql_ingestor.py +845 -0
  174. mannf/product/ingestors/grpc_ingestor.py +591 -0
  175. mannf/product/ingestors/har_ingestor.py +976 -0
  176. mannf/product/ingestors/loader.py +284 -0
  177. mannf/product/ingestors/models.py +146 -0
  178. mannf/product/ingestors/openapi_ingestor.py +606 -0
  179. mannf/product/ingestors/playwright_ingestor.py +449 -0
  180. mannf/product/ingestors/postman_ingestor.py +631 -0
  181. mannf/product/ingestors/traffic_ingestor.py +679 -0
  182. mannf/product/ingestors/websocket_ingestor.py +526 -0
  183. mannf/product/integrations/__init__.py +21 -0
  184. mannf/product/integrations/auth.py +190 -0
  185. mannf/product/integrations/graphql_parser.py +436 -0
  186. mannf/product/integrations/graphql_sut.py +247 -0
  187. mannf/product/integrations/grpc_sut.py +469 -0
  188. mannf/product/integrations/http_sut.py +237 -0
  189. mannf/product/integrations/kafka_adapter.py +342 -0
  190. mannf/product/integrations/openapi_parser.py +513 -0
  191. mannf/product/integrations/postman_parser.py +467 -0
  192. mannf/product/integrations/webhook_receiver.py +344 -0
  193. mannf/product/integrations/websocket_sut.py +434 -0
  194. mannf/product/llm/__init__.py +25 -0
  195. mannf/product/llm/anthropic_provider.py +94 -0
  196. mannf/product/llm/base.py +267 -0
  197. mannf/product/llm/config.py +48 -0
  198. mannf/product/llm/factory.py +42 -0
  199. mannf/product/llm/openai_provider.py +93 -0
  200. mannf/product/llm/prompts.py +403 -0
  201. mannf/product/llm/root_cause_service.py +311 -0
  202. mannf/product/llm/test_plan_models.py +78 -0
  203. mannf/product/metrics.py +149 -0
  204. mannf/product/middleware/__init__.py +3 -0
  205. mannf/product/middleware/audit_middleware.py +112 -0
  206. mannf/product/middleware/tenant_isolation.py +114 -0
  207. mannf/product/models.py +347 -0
  208. mannf/product/notifications/__init__.py +24 -0
  209. mannf/product/notifications/dispatcher.py +411 -0
  210. mannf/product/onboarding.py +190 -0
  211. mannf/product/orchestration/__init__.py +39 -0
  212. mannf/product/orchestration/ingest_scan_orchestrator.py +339 -0
  213. mannf/product/orchestration/pipeline.py +401 -0
  214. mannf/product/orchestrator.py +987 -0
  215. mannf/product/orchestrator_models.py +269 -0
  216. mannf/product/regression/__init__.py +36 -0
  217. mannf/product/regression/differ.py +172 -0
  218. mannf/product/regression/masking.py +100 -0
  219. mannf/product/regression/models.py +232 -0
  220. mannf/product/regression/recorder.py +124 -0
  221. mannf/product/regression/replayer.py +168 -0
  222. mannf/product/reports/__init__.py +10 -0
  223. mannf/product/reports/pdf.py +132 -0
  224. mannf/product/scheduling/__init__.py +57 -0
  225. mannf/product/scheduling/cron_utils.py +251 -0
  226. mannf/product/scheduling/engine.py +473 -0
  227. mannf/product/scheduling/models.py +86 -0
  228. mannf/product/scheduling/queue.py +894 -0
  229. mannf/product/scheduling/store.py +235 -0
  230. mannf/product/security/__init__.py +21 -0
  231. mannf/product/security/belief_guided.py +143 -0
  232. mannf/product/security/checks/__init__.py +55 -0
  233. mannf/product/security/checks/base.py +69 -0
  234. mannf/product/security/checks/bfla.py +77 -0
  235. mannf/product/security/checks/bola.py +77 -0
  236. mannf/product/security/checks/bopla.py +80 -0
  237. mannf/product/security/checks/broken_auth.py +86 -0
  238. mannf/product/security/checks/graphql_security.py +299 -0
  239. mannf/product/security/checks/inventory.py +70 -0
  240. mannf/product/security/checks/misconfig.py +158 -0
  241. mannf/product/security/checks/resource_consumption.py +70 -0
  242. mannf/product/security/checks/sensitive_flows.py +80 -0
  243. mannf/product/security/checks/ssrf.py +101 -0
  244. mannf/product/security/checks/unsafe_consumption.py +120 -0
  245. mannf/product/security/models.py +92 -0
  246. mannf/product/security/plugin_loader.py +182 -0
  247. mannf/product/security/reporter.py +92 -0
  248. mannf/product/security/scanner.py +183 -0
  249. mannf/product/server.py +6220 -0
  250. mannf/product/setup_wizard.py +873 -0
  251. mannf/product/status.py +404 -0
  252. mannf/product/storage/__init__.py +10 -0
  253. mannf/product/storage/artifact_store.py +343 -0
  254. mannf/product/telemetry.py +300 -0
  255. mannf/product/uninstall.py +169 -0
  256. mannf/product/upgrade.py +139 -0
  257. mannf/product/weights/__init__.py +13 -0
  258. mannf/product/weights/blob_store.py +299 -0
  259. mannf/product/weights/factory.py +42 -0
  260. mannf/product/weights/registry.py +159 -0
  261. mannf/product/weights/store.py +210 -0
  262. mannf/regression/__init__.py +7 -0
  263. mannf/regression/differ.py +9 -0
  264. mannf/regression/masking.py +9 -0
  265. mannf/regression/models.py +9 -0
  266. mannf/regression/recorder.py +9 -0
  267. mannf/regression/replayer.py +9 -0
  268. mannf/security/__init__.py +7 -0
  269. mannf/security/belief_guided.py +9 -0
  270. mannf/security/checks/__init__.py +7 -0
  271. mannf/security/checks/base.py +9 -0
  272. mannf/security/checks/bfla.py +9 -0
  273. mannf/security/checks/bola.py +9 -0
  274. mannf/security/checks/bopla.py +9 -0
  275. mannf/security/checks/broken_auth.py +9 -0
  276. mannf/security/checks/graphql_security.py +9 -0
  277. mannf/security/checks/inventory.py +9 -0
  278. mannf/security/checks/misconfig.py +9 -0
  279. mannf/security/checks/resource_consumption.py +9 -0
  280. mannf/security/checks/sensitive_flows.py +9 -0
  281. mannf/security/checks/ssrf.py +9 -0
  282. mannf/security/checks/unsafe_consumption.py +9 -0
  283. mannf/security/models.py +9 -0
  284. mannf/security/reporter.py +9 -0
  285. mannf/security/scanner.py +9 -0
  286. mannf/server.py +9 -0
  287. mannf/testing/__init__.py +7 -0
  288. mannf/testing/adaptive_controller.py +9 -0
  289. mannf/testing/models.py +9 -0
  290. mannf/weights/__init__.py +7 -0
  291. mannf/weights/registry.py +9 -0
  292. mannf/weights/store.py +9 -0
  293. nat_engine-1.dist-info/METADATA +555 -0
  294. nat_engine-1.dist-info/RECORD +299 -0
  295. nat_engine-1.dist-info/WHEEL +5 -0
  296. nat_engine-1.dist-info/entry_points.txt +4 -0
  297. nat_engine-1.dist-info/licenses/LICENSE +651 -0
  298. nat_engine-1.dist-info/licenses/NOTICE +178 -0
  299. nat_engine-1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,159 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """VisualComparer – pixel-level screenshot comparison utility.
7
+
8
+ Provides pure functions and a dataclass for comparing PNG screenshots
9
+ against stored baselines to detect unintended visual regressions.
10
+
11
+ Pillow (``PIL``) is imported **lazily** inside :func:`compare_screenshots`
12
+ so the module can be imported without Pillow installed.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import io
18
+ import os
19
+ from dataclasses import dataclass
20
+
21
+
22
+ @dataclass
23
+ class VisualComparisonResult:
24
+ """Result of comparing two screenshots.
25
+
26
+ Attributes
27
+ ----------
28
+ matched:
29
+ ``True`` if ``diff_percentage <= threshold``.
30
+ diff_percentage:
31
+ Percentage of pixels that differ (0.0–100.0).
32
+ diff_image:
33
+ PNG bytes of the diff visualisation (differing pixels highlighted in
34
+ magenta), or ``None`` if sizes differed.
35
+ baseline_size:
36
+ ``(width, height)`` of the baseline image.
37
+ actual_size:
38
+ ``(width, height)`` of the actual screenshot.
39
+ threshold:
40
+ The diff-percentage threshold used for the comparison.
41
+ """
42
+
43
+ matched: bool
44
+ diff_percentage: float
45
+ diff_image: bytes | None
46
+ baseline_size: tuple[int, int]
47
+ actual_size: tuple[int, int]
48
+ threshold: float
49
+
50
+
51
+ async def compare_screenshots(
52
+ baseline: bytes,
53
+ actual: bytes,
54
+ threshold: float = 0.1,
55
+ ) -> VisualComparisonResult:
56
+ """Compare two PNG screenshots pixel-by-pixel using Pillow.
57
+
58
+ Parameters
59
+ ----------
60
+ baseline:
61
+ PNG bytes of the stored baseline image.
62
+ actual:
63
+ PNG bytes of the newly captured screenshot.
64
+ threshold:
65
+ Maximum allowed ``diff_percentage`` for the comparison to be
66
+ considered *matched* (default 0.1 %).
67
+
68
+ Returns
69
+ -------
70
+ VisualComparisonResult
71
+ Detailed comparison outcome including diff image bytes.
72
+ """
73
+ from PIL import Image # lazy import — Pillow optional at module level
74
+
75
+ baseline_img = Image.open(io.BytesIO(baseline)).convert("RGB")
76
+ actual_img = Image.open(io.BytesIO(actual)).convert("RGB")
77
+
78
+ baseline_size: tuple[int, int] = baseline_img.size # (width, height)
79
+ actual_size: tuple[int, int] = actual_img.size
80
+
81
+ if baseline_size != actual_size:
82
+ return VisualComparisonResult(
83
+ matched=False,
84
+ diff_percentage=100.0,
85
+ diff_image=None,
86
+ baseline_size=baseline_size,
87
+ actual_size=actual_size,
88
+ threshold=threshold,
89
+ )
90
+
91
+ width, height = baseline_size
92
+ total_pixels = width * height
93
+
94
+ baseline_pixels = list(baseline_img.get_flattened_data())
95
+ actual_pixels = list(actual_img.get_flattened_data())
96
+
97
+ diff_img = actual_img.copy()
98
+ diff_pixels = list(diff_img.get_flattened_data())
99
+
100
+ differing = 0
101
+ for idx, (bp, ap) in enumerate(zip(baseline_pixels, actual_pixels)):
102
+ r1, g1, b1 = bp[0], bp[1], bp[2]
103
+ r2, g2, b2 = ap[0], ap[1], ap[2]
104
+ if abs(r1 - r2) + abs(g1 - g2) + abs(b1 - b2) > 30:
105
+ differing += 1
106
+ diff_pixels[idx] = (255, 0, 255) # magenta highlight
107
+
108
+ diff_img.putdata(diff_pixels) # type: ignore[arg-type]
109
+
110
+ diff_percentage = (differing / total_pixels) * 100.0
111
+ matched = diff_percentage <= threshold
112
+
113
+ buf = io.BytesIO()
114
+ diff_img.save(buf, format="PNG")
115
+ diff_image_bytes = buf.getvalue()
116
+
117
+ return VisualComparisonResult(
118
+ matched=matched,
119
+ diff_percentage=diff_percentage,
120
+ diff_image=diff_image_bytes,
121
+ baseline_size=baseline_size,
122
+ actual_size=actual_size,
123
+ threshold=threshold,
124
+ )
125
+
126
+
127
+ def save_baseline(path: str, screenshot: bytes) -> None:
128
+ """Write PNG bytes to *path*, creating parent directories as needed.
129
+
130
+ Parameters
131
+ ----------
132
+ path:
133
+ Absolute or relative filesystem path for the baseline PNG.
134
+ screenshot:
135
+ Raw PNG bytes to persist.
136
+ """
137
+ os.makedirs(os.path.dirname(os.path.abspath(path)), exist_ok=True)
138
+ with open(path, "wb") as fh:
139
+ fh.write(screenshot)
140
+
141
+
142
+ def load_baseline(path: str) -> bytes | None:
143
+ """Read PNG bytes from *path*, returning ``None`` if the file is absent.
144
+
145
+ Parameters
146
+ ----------
147
+ path:
148
+ Filesystem path of the baseline PNG.
149
+
150
+ Returns
151
+ -------
152
+ bytes or None
153
+ Raw PNG bytes, or ``None`` if the file does not exist.
154
+ """
155
+ try:
156
+ with open(path, "rb") as fh:
157
+ return fh.read()
158
+ except FileNotFoundError:
159
+ return None
@@ -0,0 +1,28 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """NAT Diagnostics package (Phase 6.5).
7
+
8
+ Provides intelligent failure analysis:
9
+
10
+ - :mod:`~mannf.core.diagnostics.failure_clusterer` — groups failures by error
11
+ signature similarity using TF-IDF + cosine similarity.
12
+ - :mod:`~mannf.core.diagnostics.flake_detector` — tracks pass/fail history per
13
+ scenario across runs and auto-labels flaky scenarios.
14
+ - :mod:`~mannf.core.diagnostics.root_cause_analyzer` — calls the configured LLM
15
+ to suggest root causes for new failures.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from mannf.core.diagnostics.failure_clusterer import FailureClusterer
21
+ from mannf.core.diagnostics.flake_detector import FlakeDetector
22
+ from mannf.core.diagnostics.root_cause_analyzer import RootCauseAnalyzer
23
+
24
+ __all__ = [
25
+ "FailureClusterer",
26
+ "FlakeDetector",
27
+ "RootCauseAnalyzer",
28
+ ]
@@ -0,0 +1,211 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """Failure clustering by error-message similarity (Phase 6.5).
7
+
8
+ Groups a list of failure dicts — each with at minimum an ``"error"`` key —
9
+ into clusters whose members share a similar error signature.
10
+
11
+ Algorithm
12
+ ---------
13
+ 1. Tokenise each error message into a bag-of-words (lowercased, stripped of
14
+ punctuation, deduplicated per message — effectively a binary term vector).
15
+ 2. Compute pairwise cosine similarity between the term vectors.
16
+ 3. Single-pass greedy merge: start a new cluster for each failure whose
17
+ similarity to every existing cluster centroid is below *threshold*. The
18
+ cluster centroid is the member that is most similar to all other members
19
+ (the "medoid").
20
+
21
+ This is intentionally lightweight — no scipy / sklearn dependency. For
22
+ large corpora a proper TF-IDF matrix would be more accurate, but for the
23
+ hundreds-of-failures scale typical of an autonomous run, this is sufficient.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import math
29
+ import re
30
+ from collections import defaultdict
31
+ from typing import Any, Dict, List, Optional, Tuple
32
+
33
+
34
+ # ---------------------------------------------------------------------------
35
+ # Internal helpers
36
+ # ---------------------------------------------------------------------------
37
+
38
+
39
+ def _tokenise(text: str) -> frozenset[str]:
40
+ """Lower-case, strip punctuation, split into unique words."""
41
+ return frozenset(re.split(r"\W+", text.lower())) - {""}
42
+
43
+
44
+ def _cosine(a: frozenset[str], b: frozenset[str]) -> float:
45
+ """Binary (0/1) cosine similarity between two token sets."""
46
+ if not a or not b:
47
+ return 0.0
48
+ intersection = len(a & b)
49
+ return intersection / math.sqrt(len(a) * len(b))
50
+
51
+
52
+ # ---------------------------------------------------------------------------
53
+ # Public dataclass-style result
54
+ # ---------------------------------------------------------------------------
55
+
56
+
57
+ class FailureCluster:
58
+ """A group of failures sharing a similar error signature.
59
+
60
+ Attributes
61
+ ----------
62
+ cluster_id : str
63
+ Stable identifier ``"cluster-0"``, ``"cluster-1"``, …
64
+ label : str
65
+ Auto-generated human-readable description derived from the most
66
+ common tokens in the cluster error messages.
67
+ failures : list[dict]
68
+ The original failure dicts that belong to this cluster.
69
+ representative_error : str
70
+ The error message of the *medoid* — the member most similar to all
71
+ other members.
72
+ pages : list[str]
73
+ Unique ``page`` keys across cluster members.
74
+ size : int
75
+ Number of failures in this cluster.
76
+ """
77
+
78
+ def __init__(
79
+ self,
80
+ cluster_id: str,
81
+ failures: List[Dict[str, Any]],
82
+ representative_error: str,
83
+ label: str,
84
+ ) -> None:
85
+ self.cluster_id = cluster_id
86
+ self.failures = failures
87
+ self.representative_error = representative_error
88
+ self.label = label
89
+ self.pages: List[str] = sorted(
90
+ {f.get("page", f.get("url", "")) for f in failures if f.get("page") or f.get("url")}
91
+ )
92
+ self.size = len(failures)
93
+
94
+ def to_dict(self) -> Dict[str, Any]:
95
+ return {
96
+ "cluster_id": self.cluster_id,
97
+ "label": self.label,
98
+ "representative_error": self.representative_error,
99
+ "pages": self.pages,
100
+ "size": self.size,
101
+ "failures": self.failures,
102
+ }
103
+
104
+
105
+ # ---------------------------------------------------------------------------
106
+ # Clusterer
107
+ # ---------------------------------------------------------------------------
108
+
109
+
110
+ class FailureClusterer:
111
+ """Cluster failure dicts by error-message similarity.
112
+
113
+ Parameters
114
+ ----------
115
+ threshold : float
116
+ Minimum cosine similarity (0–1) required to merge a failure into an
117
+ existing cluster. Default ``0.4`` works well for typical error
118
+ messages.
119
+ """
120
+
121
+ def __init__(self, threshold: float = 0.4) -> None:
122
+ if not 0.0 < threshold <= 1.0:
123
+ raise ValueError("threshold must be in (0, 1]")
124
+ self.threshold = threshold
125
+
126
+ # ------------------------------------------------------------------
127
+ # Public API
128
+ # ------------------------------------------------------------------
129
+
130
+ def cluster(self, failures: List[Dict[str, Any]]) -> List[FailureCluster]:
131
+ """Group *failures* into clusters and return them.
132
+
133
+ Parameters
134
+ ----------
135
+ failures:
136
+ List of failure dicts, each expected to have at least an
137
+ ``"error"`` key (empty string is tolerated).
138
+
139
+ Returns
140
+ -------
141
+ list[FailureCluster]
142
+ Clusters sorted by descending size.
143
+ """
144
+ if not failures:
145
+ return []
146
+
147
+ # Each bucket: list of (failure_dict, token_set) pairs
148
+ buckets: List[List[Tuple[Dict[str, Any], frozenset[str]]]] = []
149
+
150
+ for f in failures:
151
+ tokens = _tokenise(f.get("error", ""))
152
+ placed = False
153
+ for bucket in buckets:
154
+ medoid_tokens = bucket[0][1] # first member as centroid approx
155
+ if _cosine(tokens, medoid_tokens) >= self.threshold:
156
+ bucket.append((f, tokens))
157
+ placed = True
158
+ break
159
+ if not placed:
160
+ buckets.append([(f, tokens)])
161
+
162
+ clusters: List[FailureCluster] = []
163
+ for idx, bucket in enumerate(buckets):
164
+ representative, rep_label = self._pick_representative(bucket)
165
+ cluster = FailureCluster(
166
+ cluster_id=f"cluster-{idx}",
167
+ failures=[pair[0] for pair in bucket],
168
+ representative_error=representative,
169
+ label=rep_label,
170
+ )
171
+ clusters.append(cluster)
172
+
173
+ # Sort by descending cluster size
174
+ clusters.sort(key=lambda c: c.size, reverse=True)
175
+ return clusters
176
+
177
+ # ------------------------------------------------------------------
178
+ # Internals
179
+ # ------------------------------------------------------------------
180
+
181
+ def _pick_representative(
182
+ self, bucket: List[Tuple[Dict[str, Any], frozenset[str]]]
183
+ ) -> Tuple[str, str]:
184
+ """Return (representative_error, human_label) for the cluster."""
185
+ if len(bucket) == 1:
186
+ error = bucket[0][0].get("error", "")
187
+ return error, self._label_from_error(error)
188
+
189
+ # Pick medoid: member with highest average similarity to all others
190
+ best_idx = 0
191
+ best_score = -1.0
192
+ for i, (_, tokens_i) in enumerate(bucket):
193
+ total = sum(
194
+ _cosine(tokens_i, tokens_j) for j, (_, tokens_j) in enumerate(bucket) if i != j
195
+ )
196
+ avg = total / (len(bucket) - 1)
197
+ if avg > best_score:
198
+ best_score = avg
199
+ best_idx = i
200
+
201
+ error = bucket[best_idx][0].get("error", "")
202
+ return error, self._label_from_error(error)
203
+
204
+ @staticmethod
205
+ def _label_from_error(error: str) -> str:
206
+ """Create a short human-readable label from an error message."""
207
+ if not error:
208
+ return "Unknown error"
209
+ # Take the first ~80 chars, stripped of excess whitespace
210
+ short = " ".join(error.split())[:80]
211
+ return short if short else "Unknown error"
@@ -0,0 +1,233 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """Flakiness detection across autonomous runs (Phase 6.5).
7
+
8
+ Tracks pass/fail history per ``(page_key, scenario_id)`` pair across multiple
9
+ autonomous runs. A scenario is labelled **flaky** if it alternates pass/fail
10
+ in 3 or more consecutive runs.
11
+
12
+ Integration points
13
+ ------------------
14
+ * **Delta alerting (Phase 6.3)**: call :meth:`FlakeDetector.is_flaky` before
15
+ dispatching regression alerts to suppress noise from known-flaky scenarios.
16
+ * **Bug filing (Phase 6.4)**: call :meth:`FlakeDetector.is_flaky` before
17
+ creating tickets to exclude flaky failures.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from collections import defaultdict, deque
23
+ from typing import Any, Dict, List, Optional, Set, Tuple
24
+
25
+
26
+ # ---------------------------------------------------------------------------
27
+ # Helpers
28
+ # ---------------------------------------------------------------------------
29
+
30
+ _FLAKY_WINDOW = 5 # look-back window of runs to assess flakiness
31
+ _FLAKY_MIN_ALTERNATIONS = 2 # minimum pass/fail flip-flops within the window
32
+
33
+
34
+ def _count_alternations(history: deque) -> int:
35
+ """Count how many consecutive pairs differ in outcome (pass vs fail)."""
36
+ if len(history) < 2:
37
+ return 0
38
+ alternations = 0
39
+ lst = list(history)
40
+ for i in range(1, len(lst)):
41
+ if lst[i] != lst[i - 1]:
42
+ alternations += 1
43
+ return alternations
44
+
45
+
46
+ # ---------------------------------------------------------------------------
47
+ # Public dataclass-style result
48
+ # ---------------------------------------------------------------------------
49
+
50
+
51
+ class FlakeSummary:
52
+ """Flakiness statistics for a single ``(page_key, scenario_id)`` pair.
53
+
54
+ Attributes
55
+ ----------
56
+ page_key : str
57
+ scenario_id : str
58
+ is_flaky : bool
59
+ flake_rate : float
60
+ ``flake_count / total_runs`` ratio.
61
+ total_runs : int
62
+ pass_count : int
63
+ fail_count : int
64
+ alternations : int
65
+ Number of pass/fail flip-flops in the observation window.
66
+ """
67
+
68
+ def __init__(
69
+ self,
70
+ page_key: str,
71
+ scenario_id: str,
72
+ history: deque,
73
+ ) -> None:
74
+ self.page_key = page_key
75
+ self.scenario_id = scenario_id
76
+ total = len(history)
77
+ passes = sum(1 for v in history if v)
78
+ fails = total - passes
79
+ alts = _count_alternations(history)
80
+
81
+ self.total_runs = total
82
+ self.pass_count = passes
83
+ self.fail_count = fails
84
+ self.alternations = alts
85
+ self.is_flaky = total >= 3 and alts >= _FLAKY_MIN_ALTERNATIONS
86
+ self.flake_rate = round(fails / total, 4) if total > 0 else 0.0
87
+
88
+ def to_dict(self) -> Dict[str, Any]:
89
+ return {
90
+ "page_key": self.page_key,
91
+ "scenario_id": self.scenario_id,
92
+ "is_flaky": self.is_flaky,
93
+ "flake_rate": self.flake_rate,
94
+ "total_runs": self.total_runs,
95
+ "pass_count": self.pass_count,
96
+ "fail_count": self.fail_count,
97
+ "alternations": self.alternations,
98
+ }
99
+
100
+
101
+ # ---------------------------------------------------------------------------
102
+ # Detector
103
+ # ---------------------------------------------------------------------------
104
+
105
+
106
+ class FlakeDetector:
107
+ """Stateful, in-process flakiness tracker.
108
+
109
+ Maintains a rolling history of pass/fail outcomes per
110
+ ``(page_key, scenario_id)`` across autonomous runs.
111
+
112
+ Parameters
113
+ ----------
114
+ window : int
115
+ Maximum number of recent runs to retain per scenario. Default ``5``.
116
+ min_alternations : int
117
+ Minimum pass/fail flip-flops within the window to label a scenario
118
+ flaky. Default ``2`` (i.e. at least 3 runs and 2 flips).
119
+ """
120
+
121
+ def __init__(self, window: int = _FLAKY_WINDOW, min_alternations: int = _FLAKY_MIN_ALTERNATIONS) -> None:
122
+ self._window = window
123
+ self._min_alternations = min_alternations
124
+ # key → deque of booleans (True = passed, False = failed)
125
+ self._history: Dict[Tuple[str, str], deque] = defaultdict(
126
+ lambda: deque(maxlen=self._window)
127
+ )
128
+
129
+ # ------------------------------------------------------------------
130
+ # Recording outcomes
131
+ # ------------------------------------------------------------------
132
+
133
+ def record(
134
+ self,
135
+ page_key: str,
136
+ scenario_id: str,
137
+ passed: bool,
138
+ ) -> None:
139
+ """Record a single scenario outcome.
140
+
141
+ Parameters
142
+ ----------
143
+ page_key:
144
+ Page path (e.g. ``"/checkout"``).
145
+ scenario_id:
146
+ Scenario identifier (task_id or similar).
147
+ passed:
148
+ ``True`` if the scenario passed, ``False`` if it failed.
149
+ """
150
+ key = (page_key, scenario_id)
151
+ self._history[key].append(passed)
152
+
153
+ def record_run_results(self, task_details: List[Dict[str, Any]]) -> None:
154
+ """Bulk-record results from a single loop iteration.
155
+
156
+ Each dict in *task_details* should have:
157
+ - ``"url"`` or ``"page"`` — page key
158
+ - ``"task_id"`` — scenario identifier
159
+ - ``"passed"`` — bool
160
+ """
161
+ for t in task_details:
162
+ page_key = t.get("page") or t.get("url", "")
163
+ scenario_id = t.get("task_id", "")
164
+ passed = bool(t.get("passed", False))
165
+ if page_key and scenario_id:
166
+ self.record(page_key, scenario_id, passed)
167
+
168
+ # ------------------------------------------------------------------
169
+ # Querying
170
+ # ------------------------------------------------------------------
171
+
172
+ def is_flaky(self, page_key: str, scenario_id: str) -> bool:
173
+ """Return ``True`` if the scenario is currently labelled flaky."""
174
+ return self.get_summary(page_key, scenario_id).is_flaky
175
+
176
+ def get_summary(self, page_key: str, scenario_id: str) -> FlakeSummary:
177
+ """Return :class:`FlakeSummary` for a specific scenario."""
178
+ history = self._history[(page_key, scenario_id)]
179
+ return FlakeSummary(page_key, scenario_id, history)
180
+
181
+ def all_summaries(self) -> List[FlakeSummary]:
182
+ """Return :class:`FlakeSummary` for every tracked scenario."""
183
+ return [
184
+ FlakeSummary(pk, sid, hist)
185
+ for (pk, sid), hist in self._history.items()
186
+ ]
187
+
188
+ def flaky_summaries(self) -> List[FlakeSummary]:
189
+ """Return only the summaries labelled flaky."""
190
+ return [s for s in self.all_summaries() if s.is_flaky]
191
+
192
+ def flaky_scenario_keys(self) -> Set[Tuple[str, str]]:
193
+ """Return the set of ``(page_key, scenario_id)`` keys labelled flaky."""
194
+ return {(s.page_key, s.scenario_id) for s in self.flaky_summaries()}
195
+
196
+ def should_suppress(
197
+ self,
198
+ page_key: str,
199
+ scenario_id: str,
200
+ suppress_flaky: bool = True,
201
+ ) -> bool:
202
+ """Return ``True`` if a failure should be suppressed from alerts/filing.
203
+
204
+ Parameters
205
+ ----------
206
+ suppress_flaky:
207
+ When ``True`` (default), flaky scenarios are suppressed.
208
+ """
209
+ if not suppress_flaky:
210
+ return False
211
+ return self.is_flaky(page_key, scenario_id)
212
+
213
+ # ------------------------------------------------------------------
214
+ # Serialisation
215
+ # ------------------------------------------------------------------
216
+
217
+ def export_state(self) -> Dict[str, Any]:
218
+ """Export tracker state for persistence / debugging."""
219
+ return {
220
+ f"{pk}::{sid}": list(hist)
221
+ for (pk, sid), hist in self._history.items()
222
+ }
223
+
224
+ def import_state(self, state: Dict[str, Any]) -> None:
225
+ """Restore tracker state from a previously exported snapshot."""
226
+ for combined_key, history_list in state.items():
227
+ if "::" in combined_key:
228
+ pk, sid = combined_key.split("::", 1)
229
+ else:
230
+ pk, sid = combined_key, ""
231
+ dq: deque = deque(maxlen=self._window)
232
+ dq.extend(history_list)
233
+ self._history[(pk, sid)] = dq