nat-engine 1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (299) hide show
  1. mannf/__init__.py +33 -0
  2. mannf/__main__.py +10 -0
  3. mannf/_version.py +8 -0
  4. mannf/agents/__init__.py +7 -0
  5. mannf/agents/analyzer_agent.py +9 -0
  6. mannf/agents/base.py +9 -0
  7. mannf/agents/bdi_agent.py +9 -0
  8. mannf/agents/belief_state.py +9 -0
  9. mannf/agents/coordinator_agent.py +9 -0
  10. mannf/agents/executor_agent.py +9 -0
  11. mannf/agents/monitor_agent.py +9 -0
  12. mannf/agents/oracle_agent.py +9 -0
  13. mannf/agents/planner_agent.py +9 -0
  14. mannf/agents/test_agent.py +9 -0
  15. mannf/anomaly/__init__.py +7 -0
  16. mannf/anomaly/enhanced_detector.py +9 -0
  17. mannf/cli.py +9 -0
  18. mannf/core/__init__.py +26 -0
  19. mannf/core/agents/__init__.py +52 -0
  20. mannf/core/agents/accessibility_scanner_agent.py +245 -0
  21. mannf/core/agents/analyzer_agent.py +224 -0
  22. mannf/core/agents/autonomous_loop_agent.py +1086 -0
  23. mannf/core/agents/autonomous_loop_models.py +62 -0
  24. mannf/core/agents/autonomous_run_differ.py +427 -0
  25. mannf/core/agents/base.py +128 -0
  26. mannf/core/agents/bdi_agent.py +330 -0
  27. mannf/core/agents/belief_state.py +202 -0
  28. mannf/core/agents/browser_coordinator_agent.py +224 -0
  29. mannf/core/agents/browser_executor_agent.py +410 -0
  30. mannf/core/agents/coordinator_agent.py +262 -0
  31. mannf/core/agents/executor_agent.py +222 -0
  32. mannf/core/agents/monitor_agent.py +188 -0
  33. mannf/core/agents/oracle_agent.py +150 -0
  34. mannf/core/agents/performance_testing_agent.py +279 -0
  35. mannf/core/agents/planner_agent.py +128 -0
  36. mannf/core/agents/test_agent.py +249 -0
  37. mannf/core/agents/visual_regression_agent.py +311 -0
  38. mannf/core/agents/web_crawler_agent.py +510 -0
  39. mannf/core/agents/worker_pool.py +366 -0
  40. mannf/core/anomaly/__init__.py +14 -0
  41. mannf/core/anomaly/enhanced_detector.py +541 -0
  42. mannf/core/browser/__init__.py +63 -0
  43. mannf/core/browser/accessibility_scanner.py +424 -0
  44. mannf/core/browser/discovery_model.py +178 -0
  45. mannf/core/browser/dom_snapshot.py +349 -0
  46. mannf/core/browser/ingestor_bridge.py +371 -0
  47. mannf/core/browser/performance_metrics.py +217 -0
  48. mannf/core/browser/reflection_analyzer.py +442 -0
  49. mannf/core/browser/scenario_generator.py +1100 -0
  50. mannf/core/browser/security_scenario_generator.py +695 -0
  51. mannf/core/browser/visual_comparer.py +159 -0
  52. mannf/core/diagnostics/__init__.py +28 -0
  53. mannf/core/diagnostics/failure_clusterer.py +211 -0
  54. mannf/core/diagnostics/flake_detector.py +233 -0
  55. mannf/core/diagnostics/root_cause_analyzer.py +273 -0
  56. mannf/core/distributed/__init__.py +16 -0
  57. mannf/core/distributed/endpoint.py +139 -0
  58. mannf/core/distributed/system_under_test.py +207 -0
  59. mannf/core/functional_orchestrator.py +428 -0
  60. mannf/core/messaging/__init__.py +11 -0
  61. mannf/core/messaging/bus.py +113 -0
  62. mannf/core/messaging/messages.py +89 -0
  63. mannf/core/nat_orchestrator.py +342 -0
  64. mannf/core/neural/__init__.py +183 -0
  65. mannf/core/orchestrator.py +272 -0
  66. mannf/core/prioritization/__init__.py +17 -0
  67. mannf/core/prioritization/adaptive_controller.py +509 -0
  68. mannf/core/prioritization/belief_prioritizer.py +231 -0
  69. mannf/core/prioritization/risk_scorer.py +430 -0
  70. mannf/core/reporting/__init__.py +12 -0
  71. mannf/core/reporting/unified_report.py +664 -0
  72. mannf/core/testing/__init__.py +17 -0
  73. mannf/core/testing/adaptive_controller.py +149 -0
  74. mannf/core/testing/models.py +179 -0
  75. mannf/core/validation/__init__.py +10 -0
  76. mannf/core/validation/self_validation_runner.py +180 -0
  77. mannf/dashboard/__init__.py +7 -0
  78. mannf/dashboard/app.py +9 -0
  79. mannf/dashboard/models.py +9 -0
  80. mannf/dashboard/static/index.html +2538 -0
  81. mannf/dashboard/telemetry.py +9 -0
  82. mannf/distributed/__init__.py +7 -0
  83. mannf/distributed/endpoint.py +9 -0
  84. mannf/distributed/system_under_test.py +9 -0
  85. mannf/healing/__init__.py +7 -0
  86. mannf/healing/graphql_schema_diff.py +9 -0
  87. mannf/healing/healer.py +9 -0
  88. mannf/healing/models.py +9 -0
  89. mannf/healing/schema_diff.py +9 -0
  90. mannf/integrations/__init__.py +7 -0
  91. mannf/integrations/auth.py +9 -0
  92. mannf/integrations/graphql_parser.py +9 -0
  93. mannf/integrations/graphql_sut.py +9 -0
  94. mannf/integrations/http_sut.py +9 -0
  95. mannf/integrations/openapi_parser.py +9 -0
  96. mannf/integrations/postman_parser.py +9 -0
  97. mannf/llm/__init__.py +7 -0
  98. mannf/llm/anthropic_provider.py +9 -0
  99. mannf/llm/base.py +9 -0
  100. mannf/llm/config.py +9 -0
  101. mannf/llm/factory.py +9 -0
  102. mannf/llm/openai_provider.py +9 -0
  103. mannf/llm/prompts.py +9 -0
  104. mannf/messaging/__init__.py +7 -0
  105. mannf/messaging/bus.py +9 -0
  106. mannf/messaging/messages.py +9 -0
  107. mannf/nat_orchestrator.py +9 -0
  108. mannf/neural/__init__.py +7 -0
  109. mannf/orchestrator.py +9 -0
  110. mannf/prioritization/__init__.py +7 -0
  111. mannf/prioritization/adaptive_controller.py +9 -0
  112. mannf/prioritization/belief_prioritizer.py +9 -0
  113. mannf/prioritization/risk_scorer.py +9 -0
  114. mannf/product/__init__.py +29 -0
  115. mannf/product/admin/__init__.py +3 -0
  116. mannf/product/admin/routes.py +514 -0
  117. mannf/product/auth/__init__.py +5 -0
  118. mannf/product/auth/saml.py +212 -0
  119. mannf/product/billing/__init__.py +5 -0
  120. mannf/product/billing/audit.py +160 -0
  121. mannf/product/billing/feature_gates.py +180 -0
  122. mannf/product/billing/metering.py +179 -0
  123. mannf/product/billing/notifications.py +181 -0
  124. mannf/product/billing/plans.py +133 -0
  125. mannf/product/billing/rate_limits.py +35 -0
  126. mannf/product/billing/stripe_billing.py +906 -0
  127. mannf/product/billing/tenant_auth.py +233 -0
  128. mannf/product/billing/tenant_manager.py +873 -0
  129. mannf/product/cli.py +3900 -0
  130. mannf/product/cli_admin.py +408 -0
  131. mannf/product/dashboard/__init__.py +61 -0
  132. mannf/product/dashboard/app.py +3567 -0
  133. mannf/product/dashboard/models.py +460 -0
  134. mannf/product/dashboard/static/index.html +6347 -0
  135. mannf/product/dashboard/static/manifest.json +25 -0
  136. mannf/product/dashboard/static/pwa-icon-192.png +0 -0
  137. mannf/product/dashboard/static/pwa-icon-512.png +0 -0
  138. mannf/product/dashboard/static/sw.js +64 -0
  139. mannf/product/dashboard/telemetry.py +547 -0
  140. mannf/product/database.py +145 -0
  141. mannf/product/demo.py +844 -0
  142. mannf/product/doctor.py +509 -0
  143. mannf/product/exporters/__init__.py +65 -0
  144. mannf/product/exporters/azuredevops_exporter.py +257 -0
  145. mannf/product/exporters/base.py +307 -0
  146. mannf/product/exporters/bugzilla_exporter.py +200 -0
  147. mannf/product/exporters/dedup.py +275 -0
  148. mannf/product/exporters/finding_adapter.py +216 -0
  149. mannf/product/exporters/github_exporter.py +197 -0
  150. mannf/product/exporters/gitlab_exporter.py +215 -0
  151. mannf/product/exporters/jira_exporter.py +180 -0
  152. mannf/product/exporters/linear_exporter.py +195 -0
  153. mannf/product/exporters/loader.py +233 -0
  154. mannf/product/exporters/pagerduty_exporter.py +363 -0
  155. mannf/product/exporters/sentry_exporter.py +322 -0
  156. mannf/product/exporters/servicenow_exporter.py +240 -0
  157. mannf/product/exporters/shortcut_exporter.py +231 -0
  158. mannf/product/exporters/webhook_exporter.py +383 -0
  159. mannf/product/formatters/__init__.py +18 -0
  160. mannf/product/formatters/allure_formatter.py +161 -0
  161. mannf/product/formatters/ctrf_formatter.py +149 -0
  162. mannf/product/healing/__init__.py +30 -0
  163. mannf/product/healing/graphql_schema_diff.py +152 -0
  164. mannf/product/healing/healer.py +141 -0
  165. mannf/product/healing/models.py +175 -0
  166. mannf/product/healing/schema_diff.py +251 -0
  167. mannf/product/ingestors/__init__.py +77 -0
  168. mannf/product/ingestors/base.py +256 -0
  169. mannf/product/ingestors/bgstm_ingestor.py +764 -0
  170. mannf/product/ingestors/curl_ingestor.py +1019 -0
  171. mannf/product/ingestors/cypress_ingestor.py +487 -0
  172. mannf/product/ingestors/gherkin_ingestor.py +967 -0
  173. mannf/product/ingestors/graphql_ingestor.py +845 -0
  174. mannf/product/ingestors/grpc_ingestor.py +591 -0
  175. mannf/product/ingestors/har_ingestor.py +976 -0
  176. mannf/product/ingestors/loader.py +284 -0
  177. mannf/product/ingestors/models.py +146 -0
  178. mannf/product/ingestors/openapi_ingestor.py +606 -0
  179. mannf/product/ingestors/playwright_ingestor.py +449 -0
  180. mannf/product/ingestors/postman_ingestor.py +631 -0
  181. mannf/product/ingestors/traffic_ingestor.py +679 -0
  182. mannf/product/ingestors/websocket_ingestor.py +526 -0
  183. mannf/product/integrations/__init__.py +21 -0
  184. mannf/product/integrations/auth.py +190 -0
  185. mannf/product/integrations/graphql_parser.py +436 -0
  186. mannf/product/integrations/graphql_sut.py +247 -0
  187. mannf/product/integrations/grpc_sut.py +469 -0
  188. mannf/product/integrations/http_sut.py +237 -0
  189. mannf/product/integrations/kafka_adapter.py +342 -0
  190. mannf/product/integrations/openapi_parser.py +513 -0
  191. mannf/product/integrations/postman_parser.py +467 -0
  192. mannf/product/integrations/webhook_receiver.py +344 -0
  193. mannf/product/integrations/websocket_sut.py +434 -0
  194. mannf/product/llm/__init__.py +25 -0
  195. mannf/product/llm/anthropic_provider.py +94 -0
  196. mannf/product/llm/base.py +267 -0
  197. mannf/product/llm/config.py +48 -0
  198. mannf/product/llm/factory.py +42 -0
  199. mannf/product/llm/openai_provider.py +93 -0
  200. mannf/product/llm/prompts.py +403 -0
  201. mannf/product/llm/root_cause_service.py +311 -0
  202. mannf/product/llm/test_plan_models.py +78 -0
  203. mannf/product/metrics.py +149 -0
  204. mannf/product/middleware/__init__.py +3 -0
  205. mannf/product/middleware/audit_middleware.py +112 -0
  206. mannf/product/middleware/tenant_isolation.py +114 -0
  207. mannf/product/models.py +347 -0
  208. mannf/product/notifications/__init__.py +24 -0
  209. mannf/product/notifications/dispatcher.py +411 -0
  210. mannf/product/onboarding.py +190 -0
  211. mannf/product/orchestration/__init__.py +39 -0
  212. mannf/product/orchestration/ingest_scan_orchestrator.py +339 -0
  213. mannf/product/orchestration/pipeline.py +401 -0
  214. mannf/product/orchestrator.py +987 -0
  215. mannf/product/orchestrator_models.py +269 -0
  216. mannf/product/regression/__init__.py +36 -0
  217. mannf/product/regression/differ.py +172 -0
  218. mannf/product/regression/masking.py +100 -0
  219. mannf/product/regression/models.py +232 -0
  220. mannf/product/regression/recorder.py +124 -0
  221. mannf/product/regression/replayer.py +168 -0
  222. mannf/product/reports/__init__.py +10 -0
  223. mannf/product/reports/pdf.py +132 -0
  224. mannf/product/scheduling/__init__.py +57 -0
  225. mannf/product/scheduling/cron_utils.py +251 -0
  226. mannf/product/scheduling/engine.py +473 -0
  227. mannf/product/scheduling/models.py +86 -0
  228. mannf/product/scheduling/queue.py +894 -0
  229. mannf/product/scheduling/store.py +235 -0
  230. mannf/product/security/__init__.py +21 -0
  231. mannf/product/security/belief_guided.py +143 -0
  232. mannf/product/security/checks/__init__.py +55 -0
  233. mannf/product/security/checks/base.py +69 -0
  234. mannf/product/security/checks/bfla.py +77 -0
  235. mannf/product/security/checks/bola.py +77 -0
  236. mannf/product/security/checks/bopla.py +80 -0
  237. mannf/product/security/checks/broken_auth.py +86 -0
  238. mannf/product/security/checks/graphql_security.py +299 -0
  239. mannf/product/security/checks/inventory.py +70 -0
  240. mannf/product/security/checks/misconfig.py +158 -0
  241. mannf/product/security/checks/resource_consumption.py +70 -0
  242. mannf/product/security/checks/sensitive_flows.py +80 -0
  243. mannf/product/security/checks/ssrf.py +101 -0
  244. mannf/product/security/checks/unsafe_consumption.py +120 -0
  245. mannf/product/security/models.py +92 -0
  246. mannf/product/security/plugin_loader.py +182 -0
  247. mannf/product/security/reporter.py +92 -0
  248. mannf/product/security/scanner.py +183 -0
  249. mannf/product/server.py +6220 -0
  250. mannf/product/setup_wizard.py +873 -0
  251. mannf/product/status.py +404 -0
  252. mannf/product/storage/__init__.py +10 -0
  253. mannf/product/storage/artifact_store.py +343 -0
  254. mannf/product/telemetry.py +300 -0
  255. mannf/product/uninstall.py +169 -0
  256. mannf/product/upgrade.py +139 -0
  257. mannf/product/weights/__init__.py +13 -0
  258. mannf/product/weights/blob_store.py +299 -0
  259. mannf/product/weights/factory.py +42 -0
  260. mannf/product/weights/registry.py +159 -0
  261. mannf/product/weights/store.py +210 -0
  262. mannf/regression/__init__.py +7 -0
  263. mannf/regression/differ.py +9 -0
  264. mannf/regression/masking.py +9 -0
  265. mannf/regression/models.py +9 -0
  266. mannf/regression/recorder.py +9 -0
  267. mannf/regression/replayer.py +9 -0
  268. mannf/security/__init__.py +7 -0
  269. mannf/security/belief_guided.py +9 -0
  270. mannf/security/checks/__init__.py +7 -0
  271. mannf/security/checks/base.py +9 -0
  272. mannf/security/checks/bfla.py +9 -0
  273. mannf/security/checks/bola.py +9 -0
  274. mannf/security/checks/bopla.py +9 -0
  275. mannf/security/checks/broken_auth.py +9 -0
  276. mannf/security/checks/graphql_security.py +9 -0
  277. mannf/security/checks/inventory.py +9 -0
  278. mannf/security/checks/misconfig.py +9 -0
  279. mannf/security/checks/resource_consumption.py +9 -0
  280. mannf/security/checks/sensitive_flows.py +9 -0
  281. mannf/security/checks/ssrf.py +9 -0
  282. mannf/security/checks/unsafe_consumption.py +9 -0
  283. mannf/security/models.py +9 -0
  284. mannf/security/reporter.py +9 -0
  285. mannf/security/scanner.py +9 -0
  286. mannf/server.py +9 -0
  287. mannf/testing/__init__.py +7 -0
  288. mannf/testing/adaptive_controller.py +9 -0
  289. mannf/testing/models.py +9 -0
  290. mannf/weights/__init__.py +7 -0
  291. mannf/weights/registry.py +9 -0
  292. mannf/weights/store.py +9 -0
  293. nat_engine-1.dist-info/METADATA +555 -0
  294. nat_engine-1.dist-info/RECORD +299 -0
  295. nat_engine-1.dist-info/WHEEL +5 -0
  296. nat_engine-1.dist-info/entry_points.txt +4 -0
  297. nat_engine-1.dist-info/licenses/LICENSE +651 -0
  298. nat_engine-1.dist-info/licenses/NOTICE +178 -0
  299. nat_engine-1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,62 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """Pydantic models for the autonomous test loop (Phase 5).
7
+
8
+ Defines the per-iteration and full-run result structures produced by
9
+ :class:`~mannf.core.agents.autonomous_loop_agent.AutonomousTestLoopAgent`.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ from typing import Any, Dict, List, Optional
15
+
16
+ from pydantic import BaseModel, Field
17
+
18
+
19
+ class LoopIterationResult(BaseModel):
20
+ """Aggregated outcomes for a single autonomous loop iteration."""
21
+
22
+ iteration: int
23
+ scenarios_generated: int
24
+ scenarios_executed: int
25
+ passed: int
26
+ failed: int
27
+ flaky: int
28
+ new_failures: int # failures not seen in previous iterations
29
+ coverage_pages: int # unique pages tested this iteration
30
+ duration_s: float
31
+
32
+
33
+ class AutonomousRunReport(BaseModel):
34
+ """Complete report for a finished (or in-progress) autonomous run."""
35
+
36
+ run_id: str
37
+ target_url: str
38
+ strategy: str
39
+ total_iterations: int
40
+ total_scenarios_executed: int
41
+ total_passed: int
42
+ total_failed: int
43
+ total_flaky: int
44
+ unique_failures: List[Dict[str, Any]] = Field(default_factory=list)
45
+ coverage_summary: Dict[str, Any] = Field(default_factory=dict)
46
+ iterations: List[LoopIterationResult] = Field(default_factory=list)
47
+ started_at: str
48
+ completed_at: str = ""
49
+ duration_s: float = 0.0
50
+ stop_reason: str = "" # "max_iterations", "max_duration", "stable", "coverage_met"
51
+ status: str = "running" # "running", "complete", "stopped", "failed"
52
+ error: str = ""
53
+ # Deploy / commit attribution (Phase 6.3)
54
+ build_id: Optional[str] = None
55
+ commit_sha: Optional[str] = None
56
+ deploy_tag: Optional[str] = None
57
+ # Belief evolution snapshots — list of per-iteration snapshots (Phase 6.5)
58
+ belief_snapshots: List[Dict[str, Any]] = Field(default_factory=list)
59
+ # LLM root cause suggestions per failure (Phase 6.5)
60
+ root_cause_suggestions: List[Dict[str, Any]] = Field(default_factory=list)
61
+ # Coverage gaps — pages blocked by auth/CAPTCHA/timeout/unreachable (Phase 7.4)
62
+ coverage_gaps: List[Dict[str, Any]] = Field(default_factory=list)
@@ -0,0 +1,427 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """AutonomousRunDiffer — run-over-run diff engine for autonomous test loop reports.
7
+
8
+ Compares two :class:`~mannf.core.agents.autonomous_loop_models.AutonomousRunReport`
9
+ objects and produces a structured :class:`AutonomousRunDiff` identifying:
10
+
11
+ - Per-page pass/fail deltas
12
+ - New failures (regressions) since the baseline run
13
+ - Resolved failures (recoveries) that no longer occur
14
+ - Flakiness changes
15
+ - Coverage delta (pages added / removed)
16
+ - Belief score changes
17
+
18
+ The comparison key for failures is the ``page_key`` (path component of the URL)
19
+ plus an error signature string, mirroring the dedup logic in
20
+ :func:`~mannf.core.agents.autonomous_loop_agent._page_key`.
21
+
22
+ Follows the same structural pattern as
23
+ :class:`~mannf.product.regression.differ.RegressionDiffer` — a class with a
24
+ :meth:`diff` method that accepts two report objects and returns a structured
25
+ result.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ from dataclasses import dataclass, field
31
+ from typing import Any, Dict, List, Optional
32
+ from urllib.parse import urlparse
33
+
34
+ from mannf.core.agents.autonomous_loop_models import AutonomousRunReport
35
+
36
+
37
+ # ---------------------------------------------------------------------------
38
+ # Structured result types
39
+ # ---------------------------------------------------------------------------
40
+
41
+
42
+ @dataclass
43
+ class PageDelta:
44
+ """Pass/fail delta for a single page between two runs."""
45
+
46
+ page_key: str
47
+ baseline_passed: int
48
+ baseline_failed: int
49
+ current_passed: int
50
+ current_failed: int
51
+
52
+ @property
53
+ def pass_delta(self) -> int:
54
+ return self.current_passed - self.baseline_passed
55
+
56
+ @property
57
+ def fail_delta(self) -> int:
58
+ return self.current_failed - self.baseline_failed
59
+
60
+
61
+ @dataclass
62
+ class FailureEntry:
63
+ """A single unique failure identified by page + error signature."""
64
+
65
+ page_key: str
66
+ error_signature: str
67
+ url: str = ""
68
+ scenario_type: str = ""
69
+ details: Dict[str, Any] = field(default_factory=dict)
70
+
71
+
72
+ @dataclass
73
+ class FlakinessDelta:
74
+ """Change in flaky scenario count between two runs."""
75
+
76
+ page_key: str
77
+ baseline_flaky: int
78
+ current_flaky: int
79
+
80
+ @property
81
+ def delta(self) -> int:
82
+ return self.current_flaky - self.baseline_flaky
83
+
84
+
85
+ @dataclass
86
+ class CoverageDelta:
87
+ """Coverage delta: pages added or removed between two runs."""
88
+
89
+ pages_added: List[str] = field(default_factory=list)
90
+ pages_removed: List[str] = field(default_factory=list)
91
+ baseline_page_count: int = 0
92
+ current_page_count: int = 0
93
+
94
+
95
+ @dataclass
96
+ class AutonomousRunDiff:
97
+ """Structured diff between a baseline and a current autonomous run."""
98
+
99
+ baseline_run_id: str
100
+ current_run_id: str
101
+ baseline_commit: Optional[str]
102
+ current_commit: Optional[str]
103
+ baseline_build: Optional[str]
104
+ current_build: Optional[str]
105
+
106
+ summary: Dict[str, Any] = field(default_factory=dict)
107
+ new_failures: List[FailureEntry] = field(default_factory=list)
108
+ resolved_failures: List[FailureEntry] = field(default_factory=list)
109
+ page_deltas: List[PageDelta] = field(default_factory=list)
110
+ flakiness_changes: List[FlakinessDelta] = field(default_factory=list)
111
+ coverage_delta: CoverageDelta = field(default_factory=CoverageDelta)
112
+
113
+ def human_readable(self) -> str:
114
+ """Return a human-readable diff summary string."""
115
+ lines: List[str] = []
116
+ s = self.summary
117
+ lines.append("=" * 60)
118
+ lines.append(" Autonomous Run Diff Report")
119
+ lines.append("=" * 60)
120
+ lines.append(f" Baseline: {self.baseline_run_id[:8]}")
121
+ lines.append(f" Current: {self.current_run_id[:8]}")
122
+ if self.baseline_commit:
123
+ lines.append(f" Baseline commit: {self.baseline_commit}")
124
+ if self.current_commit:
125
+ lines.append(f" Current commit: {self.current_commit}")
126
+ lines.append("-" * 60)
127
+ lines.append(f" New failures: {s.get('new_failure_count', 0)}")
128
+ lines.append(f" Resolved failures: {s.get('resolved_failure_count', 0)}")
129
+ lines.append(f" Pass delta: {s.get('total_pass_delta', 0):+d}")
130
+ lines.append(f" Fail delta: {s.get('total_fail_delta', 0):+d}")
131
+ lines.append(f" Coverage added: {s.get('pages_added', 0)}")
132
+ lines.append(f" Coverage removed: {s.get('pages_removed', 0)}")
133
+ lines.append("=" * 60)
134
+ if self.new_failures:
135
+ lines.append("\n New Failures:")
136
+ for f in self.new_failures:
137
+ lines.append(f" [{f.page_key}] {f.error_signature}")
138
+ if self.resolved_failures:
139
+ lines.append("\n Resolved Failures:")
140
+ for f in self.resolved_failures:
141
+ lines.append(f" [{f.page_key}] {f.error_signature}")
142
+ return "\n".join(lines)
143
+
144
+
145
+ # ---------------------------------------------------------------------------
146
+ # Differ
147
+ # ---------------------------------------------------------------------------
148
+
149
+
150
+ def _normalize_page_key(url: str, base_url: str) -> str:
151
+ """Return a normalised page-relative path (mirrors autonomous_loop_agent._page_key)."""
152
+ try:
153
+ base = urlparse(base_url)
154
+ parsed = urlparse(url)
155
+ if parsed.netloc == base.netloc or not parsed.netloc:
156
+ return parsed.path or "/"
157
+ return url
158
+ except Exception: # noqa: BLE001
159
+ return url
160
+
161
+
162
+ def _failure_signature(failure: Dict[str, Any]) -> str:
163
+ """Derive a stable dedup key from a unique-failure dict.
164
+
165
+ Uses the ``error`` field if present, otherwise falls back to
166
+ ``scenario_type`` + ``page_key``.
167
+ """
168
+ parts: List[str] = []
169
+ if failure.get("error"):
170
+ parts.append(str(failure["error"])[:120])
171
+ elif failure.get("scenario_type"):
172
+ parts.append(str(failure["scenario_type"]))
173
+ parts.append(str(failure.get("page_key", failure.get("url", ""))))
174
+ return "|".join(parts)
175
+
176
+
177
+ def _build_failure_index(
178
+ unique_failures: List[Dict[str, Any]],
179
+ base_url: str,
180
+ ) -> Dict[str, FailureEntry]:
181
+ """Build a ``{signature: FailureEntry}`` index from the unique_failures list."""
182
+ index: Dict[str, FailureEntry] = {}
183
+ for f in unique_failures:
184
+ url = f.get("url", "")
185
+ page_key = f.get("page_key") or _normalize_page_key(url, base_url)
186
+ sig = _failure_signature(f)
187
+ composite_key = f"{page_key}::{sig}"
188
+ index[composite_key] = FailureEntry(
189
+ page_key=page_key,
190
+ error_signature=sig,
191
+ url=url,
192
+ scenario_type=f.get("scenario_type", ""),
193
+ details={k: v for k, v in f.items() if k not in {"page_key", "url", "error"}},
194
+ )
195
+ return index
196
+
197
+
198
+ def _page_stats_from_failures(
199
+ unique_failures: List[Dict[str, Any]],
200
+ base_url: str,
201
+ total_passed: int,
202
+ total_failed: int,
203
+ ) -> Dict[str, Dict[str, int]]:
204
+ """Build per-page pass/fail stats from a run report.
205
+
206
+ When per-page breakdowns are not directly stored in the report, we
207
+ distribute the aggregate totals evenly across the pages that have failures
208
+ and treat the remainder as passed on an implicit "root" page.
209
+ """
210
+ page_failed: Dict[str, int] = {}
211
+ for f in unique_failures:
212
+ url = f.get("url", "")
213
+ pk = f.get("page_key") or _normalize_page_key(url, base_url)
214
+ page_failed[pk] = page_failed.get(pk, 0) + 1
215
+
216
+ stats: Dict[str, Dict[str, int]] = {}
217
+ for pk, fail_count in page_failed.items():
218
+ stats[pk] = {"passed": 0, "failed": fail_count}
219
+
220
+ # Assign remaining passed count to a synthetic root page when we have no
221
+ # page-level breakdown
222
+ if total_passed > 0 and not stats:
223
+ stats["/"] = {"passed": total_passed, "failed": 0}
224
+
225
+ return stats
226
+
227
+
228
+ class AutonomousRunDiffer:
229
+ """Compare two :class:`~mannf.core.agents.autonomous_loop_models.AutonomousRunReport`
230
+ objects and produce an :class:`AutonomousRunDiff`.
231
+
232
+ Parameters
233
+ ----------
234
+ flakiness_spike_threshold:
235
+ Minimum absolute increase in flaky scenarios on a page to be reported
236
+ as a flakiness spike. Defaults to ``2``.
237
+ """
238
+
239
+ def __init__(self, flakiness_spike_threshold: int = 2) -> None:
240
+ self.flakiness_spike_threshold = flakiness_spike_threshold
241
+
242
+ def diff(
243
+ self,
244
+ baseline: AutonomousRunReport,
245
+ current: AutonomousRunReport,
246
+ ) -> AutonomousRunDiff:
247
+ """Compute the run-over-run diff between *baseline* and *current*.
248
+
249
+ Parameters
250
+ ----------
251
+ baseline:
252
+ The earlier run to compare against.
253
+ current:
254
+ The newer run to compare.
255
+
256
+ Returns
257
+ -------
258
+ AutonomousRunDiff
259
+ Structured diff report.
260
+ """
261
+ base_url = baseline.target_url or current.target_url
262
+
263
+ # Build failure indexes
264
+ baseline_failures = _build_failure_index(
265
+ baseline.unique_failures, base_url
266
+ )
267
+ current_failures = _build_failure_index(
268
+ current.unique_failures, base_url
269
+ )
270
+
271
+ baseline_keys = set(baseline_failures.keys())
272
+ current_keys = set(current_failures.keys())
273
+
274
+ new_failure_keys = current_keys - baseline_keys
275
+ resolved_failure_keys = baseline_keys - current_keys
276
+
277
+ new_failures = [current_failures[k] for k in sorted(new_failure_keys)]
278
+ resolved_failures = [baseline_failures[k] for k in sorted(resolved_failure_keys)]
279
+
280
+ # Per-page pass/fail deltas
281
+ page_deltas = self._compute_page_deltas(baseline, current, base_url)
282
+
283
+ # Flakiness changes
284
+ flakiness_changes = self._compute_flakiness_changes(baseline, current, base_url)
285
+
286
+ # Coverage delta
287
+ coverage_delta = self._compute_coverage_delta(baseline, current)
288
+
289
+ # Summary
290
+ total_pass_delta = current.total_passed - baseline.total_passed
291
+ total_fail_delta = current.total_failed - baseline.total_failed
292
+
293
+ summary: Dict[str, Any] = {
294
+ "new_failure_count": len(new_failures),
295
+ "resolved_failure_count": len(resolved_failures),
296
+ "total_pass_delta": total_pass_delta,
297
+ "total_fail_delta": total_fail_delta,
298
+ "pages_added": len(coverage_delta.pages_added),
299
+ "pages_removed": len(coverage_delta.pages_removed),
300
+ "baseline_total_passed": baseline.total_passed,
301
+ "baseline_total_failed": baseline.total_failed,
302
+ "current_total_passed": current.total_passed,
303
+ "current_total_failed": current.total_failed,
304
+ "flakiness_spike_pages": len(
305
+ [
306
+ c
307
+ for c in flakiness_changes
308
+ if c.delta >= self.flakiness_spike_threshold
309
+ ]
310
+ ),
311
+ }
312
+
313
+ return AutonomousRunDiff(
314
+ baseline_run_id=baseline.run_id,
315
+ current_run_id=current.run_id,
316
+ baseline_commit=baseline.commit_sha,
317
+ current_commit=current.commit_sha,
318
+ baseline_build=baseline.build_id,
319
+ current_build=current.build_id,
320
+ summary=summary,
321
+ new_failures=new_failures,
322
+ resolved_failures=resolved_failures,
323
+ page_deltas=page_deltas,
324
+ flakiness_changes=flakiness_changes,
325
+ coverage_delta=coverage_delta,
326
+ )
327
+
328
+ # ------------------------------------------------------------------
329
+ # Internal helpers
330
+ # ------------------------------------------------------------------
331
+
332
+ def _compute_page_deltas(
333
+ self,
334
+ baseline: AutonomousRunReport,
335
+ current: AutonomousRunReport,
336
+ base_url: str,
337
+ ) -> List[PageDelta]:
338
+ baseline_stats = _page_stats_from_failures(
339
+ baseline.unique_failures,
340
+ base_url,
341
+ baseline.total_passed,
342
+ baseline.total_failed,
343
+ )
344
+ current_stats = _page_stats_from_failures(
345
+ current.unique_failures,
346
+ base_url,
347
+ current.total_passed,
348
+ current.total_failed,
349
+ )
350
+
351
+ all_pages = set(baseline_stats.keys()) | set(current_stats.keys())
352
+ deltas: List[PageDelta] = []
353
+ for page in sorted(all_pages):
354
+ b = baseline_stats.get(page, {"passed": 0, "failed": 0})
355
+ c = current_stats.get(page, {"passed": 0, "failed": 0})
356
+ if b != c:
357
+ deltas.append(
358
+ PageDelta(
359
+ page_key=page,
360
+ baseline_passed=b["passed"],
361
+ baseline_failed=b["failed"],
362
+ current_passed=c["passed"],
363
+ current_failed=c["failed"],
364
+ )
365
+ )
366
+ return deltas
367
+
368
+ def _compute_flakiness_changes(
369
+ self,
370
+ baseline: AutonomousRunReport,
371
+ current: AutonomousRunReport,
372
+ base_url: str,
373
+ ) -> List[FlakinessDelta]:
374
+ """Compute per-page flakiness change using iteration-level data."""
375
+
376
+ def _flaky_by_page(report: AutonomousRunReport) -> Dict[str, int]:
377
+ """Sum flaky counts by page using coverage_summary if available."""
378
+ cs = report.coverage_summary or {}
379
+ if cs.get("pages_by_risk"):
380
+ # Not per-page flakiness — fall through to aggregate
381
+ pass
382
+ # Use total_flaky as a single aggregate bucket
383
+ if report.total_flaky:
384
+ return {"/": report.total_flaky}
385
+ return {}
386
+
387
+ baseline_flaky = _flaky_by_page(baseline)
388
+ current_flaky = _flaky_by_page(current)
389
+ all_pages = set(baseline_flaky.keys()) | set(current_flaky.keys())
390
+
391
+ changes: List[FlakinessDelta] = []
392
+ for page in sorted(all_pages):
393
+ b_val = baseline_flaky.get(page, 0)
394
+ c_val = current_flaky.get(page, 0)
395
+ if b_val != c_val:
396
+ changes.append(
397
+ FlakinessDelta(
398
+ page_key=page,
399
+ baseline_flaky=b_val,
400
+ current_flaky=c_val,
401
+ )
402
+ )
403
+ return changes
404
+
405
+ def _compute_coverage_delta(
406
+ self,
407
+ baseline: AutonomousRunReport,
408
+ current: AutonomousRunReport,
409
+ ) -> CoverageDelta:
410
+ """Compare covered_pages sets from coverage_summary."""
411
+
412
+ def _pages(report: AutonomousRunReport) -> set:
413
+ cs = report.coverage_summary or {}
414
+ raw = cs.get("covered_pages", [])
415
+ if isinstance(raw, list):
416
+ return set(raw)
417
+ return set()
418
+
419
+ baseline_pages = _pages(baseline)
420
+ current_pages = _pages(current)
421
+
422
+ return CoverageDelta(
423
+ pages_added=sorted(current_pages - baseline_pages),
424
+ pages_removed=sorted(baseline_pages - current_pages),
425
+ baseline_page_count=len(baseline_pages),
426
+ current_page_count=len(current_pages),
427
+ )
@@ -0,0 +1,128 @@
1
+ # Copyright (C) 2026 Brad Guider
2
+ # This file is part of NAT (Neural Agent Testing Framework).
3
+ # Licensed under the AGPL-3.0. See LICENSE for details.
4
+ # Commercial licensing available — see COMMERCIAL_LICENSE.md.
5
+
6
+ """Abstract base class for all agents in the framework.
7
+
8
+ Every agent:
9
+ • has a unique ID and a human-readable role label
10
+ • owns a :class:`NeuralNetwork` for decision-making
11
+ • subscribes to the shared :class:`MessageBus`
12
+ • runs as an ``asyncio`` task via :meth:`run`
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import asyncio
18
+ import logging
19
+ from abc import ABC, abstractmethod
20
+ from typing import Optional, Set
21
+
22
+ from mannf.core.messaging.bus import MessageBus
23
+ from mannf.core.messaging.messages import Message, MessageType
24
+ from mannf.core.neural import NeuralNetwork
25
+
26
+ logger = logging.getLogger(__name__)
27
+
28
+
29
+ class BaseAgent(ABC):
30
+ """Abstract base for all MANNF agents.
31
+
32
+ Parameters
33
+ ----------
34
+ agent_id:
35
+ Unique identifier for this agent instance.
36
+ bus:
37
+ Shared message bus used for inter-agent communication.
38
+ network:
39
+ Optional neural network the agent uses for decision-making.
40
+ Subclasses may construct their own if *None* is passed.
41
+ """
42
+
43
+ def __init__(
44
+ self,
45
+ agent_id: str,
46
+ bus: MessageBus,
47
+ network: Optional[NeuralNetwork] = None,
48
+ ) -> None:
49
+ self.agent_id = agent_id
50
+ self._bus = bus
51
+ self.network = network
52
+
53
+ self._running = False
54
+ self._task: Optional[asyncio.Task] = None # type: ignore[type-arg]
55
+ self._inbox: asyncio.Queue[Message] = asyncio.Queue()
56
+
57
+ # ------------------------------------------------------------------
58
+ # Subscriptions – override to declare which message types to receive
59
+ # ------------------------------------------------------------------
60
+
61
+ @property
62
+ @abstractmethod
63
+ def subscribed_types(self) -> Set[MessageType]:
64
+ """The set of :class:`MessageType` values this agent subscribes to."""
65
+
66
+ # ------------------------------------------------------------------
67
+ # Lifecycle
68
+ # ------------------------------------------------------------------
69
+
70
+ async def start(self) -> None:
71
+ """Register subscriptions, announce AGENT_STARTED, then start the run loop."""
72
+ self._inbox = self._bus.subscribe(self.agent_id, self.subscribed_types)
73
+ self._running = True
74
+ await self._bus.publish(
75
+ Message(MessageType.AGENT_STARTED, sender_id=self.agent_id)
76
+ )
77
+ logger.info("Agent %s started", self.agent_id)
78
+
79
+ async def stop(self) -> None:
80
+ """Announce AGENT_STOPPED and tear down the subscription."""
81
+ self._running = False
82
+ await self._bus.publish(
83
+ Message(MessageType.AGENT_STOPPED, sender_id=self.agent_id)
84
+ )
85
+ self._bus.unsubscribe(self.agent_id)
86
+ logger.info("Agent %s stopped", self.agent_id)
87
+
88
+ # ------------------------------------------------------------------
89
+ # Main loop
90
+ # ------------------------------------------------------------------
91
+
92
+ async def run(self) -> None:
93
+ """Entry point for the agent's asyncio task.
94
+
95
+ Calls :meth:`start`, then processes incoming messages until
96
+ :meth:`stop` is called. An explicit ``asyncio.sleep(0)`` after each
97
+ message ensures other concurrent tasks (e.g. the coordinator's
98
+ award timers) always get a chance to execute.
99
+ """
100
+ await self.start()
101
+ try:
102
+ while self._running:
103
+ try:
104
+ msg = await asyncio.wait_for(self._inbox.get(), timeout=1.0)
105
+ await self._handle_message(msg)
106
+ # Explicit yield so other tasks (timers, award callbacks) can run
107
+ await asyncio.sleep(0)
108
+ except asyncio.TimeoutError:
109
+ await self._on_idle()
110
+ finally:
111
+ await self.stop()
112
+
113
+ @abstractmethod
114
+ async def _handle_message(self, message: Message) -> None:
115
+ """Process a single inbound message."""
116
+
117
+ async def _on_idle(self) -> None:
118
+ """Called each time the inbox timeout expires. Override for background work."""
119
+
120
+ # ------------------------------------------------------------------
121
+ # Helpers
122
+ # ------------------------------------------------------------------
123
+
124
+ async def _publish(self, message: Message) -> None:
125
+ await self._bus.publish(message)
126
+
127
+ def __repr__(self) -> str:
128
+ return f"{self.__class__.__name__}(id={self.agent_id!r})"