nat-engine 1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mannf/__init__.py +33 -0
- mannf/__main__.py +10 -0
- mannf/_version.py +8 -0
- mannf/agents/__init__.py +7 -0
- mannf/agents/analyzer_agent.py +9 -0
- mannf/agents/base.py +9 -0
- mannf/agents/bdi_agent.py +9 -0
- mannf/agents/belief_state.py +9 -0
- mannf/agents/coordinator_agent.py +9 -0
- mannf/agents/executor_agent.py +9 -0
- mannf/agents/monitor_agent.py +9 -0
- mannf/agents/oracle_agent.py +9 -0
- mannf/agents/planner_agent.py +9 -0
- mannf/agents/test_agent.py +9 -0
- mannf/anomaly/__init__.py +7 -0
- mannf/anomaly/enhanced_detector.py +9 -0
- mannf/cli.py +9 -0
- mannf/core/__init__.py +26 -0
- mannf/core/agents/__init__.py +52 -0
- mannf/core/agents/accessibility_scanner_agent.py +245 -0
- mannf/core/agents/analyzer_agent.py +224 -0
- mannf/core/agents/autonomous_loop_agent.py +1086 -0
- mannf/core/agents/autonomous_loop_models.py +62 -0
- mannf/core/agents/autonomous_run_differ.py +427 -0
- mannf/core/agents/base.py +128 -0
- mannf/core/agents/bdi_agent.py +330 -0
- mannf/core/agents/belief_state.py +202 -0
- mannf/core/agents/browser_coordinator_agent.py +224 -0
- mannf/core/agents/browser_executor_agent.py +410 -0
- mannf/core/agents/coordinator_agent.py +262 -0
- mannf/core/agents/executor_agent.py +222 -0
- mannf/core/agents/monitor_agent.py +188 -0
- mannf/core/agents/oracle_agent.py +150 -0
- mannf/core/agents/performance_testing_agent.py +279 -0
- mannf/core/agents/planner_agent.py +128 -0
- mannf/core/agents/test_agent.py +249 -0
- mannf/core/agents/visual_regression_agent.py +311 -0
- mannf/core/agents/web_crawler_agent.py +510 -0
- mannf/core/agents/worker_pool.py +366 -0
- mannf/core/anomaly/__init__.py +14 -0
- mannf/core/anomaly/enhanced_detector.py +541 -0
- mannf/core/browser/__init__.py +63 -0
- mannf/core/browser/accessibility_scanner.py +424 -0
- mannf/core/browser/discovery_model.py +178 -0
- mannf/core/browser/dom_snapshot.py +349 -0
- mannf/core/browser/ingestor_bridge.py +371 -0
- mannf/core/browser/performance_metrics.py +217 -0
- mannf/core/browser/reflection_analyzer.py +442 -0
- mannf/core/browser/scenario_generator.py +1100 -0
- mannf/core/browser/security_scenario_generator.py +695 -0
- mannf/core/browser/visual_comparer.py +159 -0
- mannf/core/diagnostics/__init__.py +28 -0
- mannf/core/diagnostics/failure_clusterer.py +211 -0
- mannf/core/diagnostics/flake_detector.py +233 -0
- mannf/core/diagnostics/root_cause_analyzer.py +273 -0
- mannf/core/distributed/__init__.py +16 -0
- mannf/core/distributed/endpoint.py +139 -0
- mannf/core/distributed/system_under_test.py +207 -0
- mannf/core/functional_orchestrator.py +428 -0
- mannf/core/messaging/__init__.py +11 -0
- mannf/core/messaging/bus.py +113 -0
- mannf/core/messaging/messages.py +89 -0
- mannf/core/nat_orchestrator.py +342 -0
- mannf/core/neural/__init__.py +183 -0
- mannf/core/orchestrator.py +272 -0
- mannf/core/prioritization/__init__.py +17 -0
- mannf/core/prioritization/adaptive_controller.py +509 -0
- mannf/core/prioritization/belief_prioritizer.py +231 -0
- mannf/core/prioritization/risk_scorer.py +430 -0
- mannf/core/reporting/__init__.py +12 -0
- mannf/core/reporting/unified_report.py +664 -0
- mannf/core/testing/__init__.py +17 -0
- mannf/core/testing/adaptive_controller.py +149 -0
- mannf/core/testing/models.py +179 -0
- mannf/core/validation/__init__.py +10 -0
- mannf/core/validation/self_validation_runner.py +180 -0
- mannf/dashboard/__init__.py +7 -0
- mannf/dashboard/app.py +9 -0
- mannf/dashboard/models.py +9 -0
- mannf/dashboard/static/index.html +2538 -0
- mannf/dashboard/telemetry.py +9 -0
- mannf/distributed/__init__.py +7 -0
- mannf/distributed/endpoint.py +9 -0
- mannf/distributed/system_under_test.py +9 -0
- mannf/healing/__init__.py +7 -0
- mannf/healing/graphql_schema_diff.py +9 -0
- mannf/healing/healer.py +9 -0
- mannf/healing/models.py +9 -0
- mannf/healing/schema_diff.py +9 -0
- mannf/integrations/__init__.py +7 -0
- mannf/integrations/auth.py +9 -0
- mannf/integrations/graphql_parser.py +9 -0
- mannf/integrations/graphql_sut.py +9 -0
- mannf/integrations/http_sut.py +9 -0
- mannf/integrations/openapi_parser.py +9 -0
- mannf/integrations/postman_parser.py +9 -0
- mannf/llm/__init__.py +7 -0
- mannf/llm/anthropic_provider.py +9 -0
- mannf/llm/base.py +9 -0
- mannf/llm/config.py +9 -0
- mannf/llm/factory.py +9 -0
- mannf/llm/openai_provider.py +9 -0
- mannf/llm/prompts.py +9 -0
- mannf/messaging/__init__.py +7 -0
- mannf/messaging/bus.py +9 -0
- mannf/messaging/messages.py +9 -0
- mannf/nat_orchestrator.py +9 -0
- mannf/neural/__init__.py +7 -0
- mannf/orchestrator.py +9 -0
- mannf/prioritization/__init__.py +7 -0
- mannf/prioritization/adaptive_controller.py +9 -0
- mannf/prioritization/belief_prioritizer.py +9 -0
- mannf/prioritization/risk_scorer.py +9 -0
- mannf/product/__init__.py +29 -0
- mannf/product/admin/__init__.py +3 -0
- mannf/product/admin/routes.py +514 -0
- mannf/product/auth/__init__.py +5 -0
- mannf/product/auth/saml.py +212 -0
- mannf/product/billing/__init__.py +5 -0
- mannf/product/billing/audit.py +160 -0
- mannf/product/billing/feature_gates.py +180 -0
- mannf/product/billing/metering.py +179 -0
- mannf/product/billing/notifications.py +181 -0
- mannf/product/billing/plans.py +133 -0
- mannf/product/billing/rate_limits.py +35 -0
- mannf/product/billing/stripe_billing.py +906 -0
- mannf/product/billing/tenant_auth.py +233 -0
- mannf/product/billing/tenant_manager.py +873 -0
- mannf/product/cli.py +3900 -0
- mannf/product/cli_admin.py +408 -0
- mannf/product/dashboard/__init__.py +61 -0
- mannf/product/dashboard/app.py +3567 -0
- mannf/product/dashboard/models.py +460 -0
- mannf/product/dashboard/static/index.html +6347 -0
- mannf/product/dashboard/static/manifest.json +25 -0
- mannf/product/dashboard/static/pwa-icon-192.png +0 -0
- mannf/product/dashboard/static/pwa-icon-512.png +0 -0
- mannf/product/dashboard/static/sw.js +64 -0
- mannf/product/dashboard/telemetry.py +547 -0
- mannf/product/database.py +145 -0
- mannf/product/demo.py +844 -0
- mannf/product/doctor.py +509 -0
- mannf/product/exporters/__init__.py +65 -0
- mannf/product/exporters/azuredevops_exporter.py +257 -0
- mannf/product/exporters/base.py +307 -0
- mannf/product/exporters/bugzilla_exporter.py +200 -0
- mannf/product/exporters/dedup.py +275 -0
- mannf/product/exporters/finding_adapter.py +216 -0
- mannf/product/exporters/github_exporter.py +197 -0
- mannf/product/exporters/gitlab_exporter.py +215 -0
- mannf/product/exporters/jira_exporter.py +180 -0
- mannf/product/exporters/linear_exporter.py +195 -0
- mannf/product/exporters/loader.py +233 -0
- mannf/product/exporters/pagerduty_exporter.py +363 -0
- mannf/product/exporters/sentry_exporter.py +322 -0
- mannf/product/exporters/servicenow_exporter.py +240 -0
- mannf/product/exporters/shortcut_exporter.py +231 -0
- mannf/product/exporters/webhook_exporter.py +383 -0
- mannf/product/formatters/__init__.py +18 -0
- mannf/product/formatters/allure_formatter.py +161 -0
- mannf/product/formatters/ctrf_formatter.py +149 -0
- mannf/product/healing/__init__.py +30 -0
- mannf/product/healing/graphql_schema_diff.py +152 -0
- mannf/product/healing/healer.py +141 -0
- mannf/product/healing/models.py +175 -0
- mannf/product/healing/schema_diff.py +251 -0
- mannf/product/ingestors/__init__.py +77 -0
- mannf/product/ingestors/base.py +256 -0
- mannf/product/ingestors/bgstm_ingestor.py +764 -0
- mannf/product/ingestors/curl_ingestor.py +1019 -0
- mannf/product/ingestors/cypress_ingestor.py +487 -0
- mannf/product/ingestors/gherkin_ingestor.py +967 -0
- mannf/product/ingestors/graphql_ingestor.py +845 -0
- mannf/product/ingestors/grpc_ingestor.py +591 -0
- mannf/product/ingestors/har_ingestor.py +976 -0
- mannf/product/ingestors/loader.py +284 -0
- mannf/product/ingestors/models.py +146 -0
- mannf/product/ingestors/openapi_ingestor.py +606 -0
- mannf/product/ingestors/playwright_ingestor.py +449 -0
- mannf/product/ingestors/postman_ingestor.py +631 -0
- mannf/product/ingestors/traffic_ingestor.py +679 -0
- mannf/product/ingestors/websocket_ingestor.py +526 -0
- mannf/product/integrations/__init__.py +21 -0
- mannf/product/integrations/auth.py +190 -0
- mannf/product/integrations/graphql_parser.py +436 -0
- mannf/product/integrations/graphql_sut.py +247 -0
- mannf/product/integrations/grpc_sut.py +469 -0
- mannf/product/integrations/http_sut.py +237 -0
- mannf/product/integrations/kafka_adapter.py +342 -0
- mannf/product/integrations/openapi_parser.py +513 -0
- mannf/product/integrations/postman_parser.py +467 -0
- mannf/product/integrations/webhook_receiver.py +344 -0
- mannf/product/integrations/websocket_sut.py +434 -0
- mannf/product/llm/__init__.py +25 -0
- mannf/product/llm/anthropic_provider.py +94 -0
- mannf/product/llm/base.py +267 -0
- mannf/product/llm/config.py +48 -0
- mannf/product/llm/factory.py +42 -0
- mannf/product/llm/openai_provider.py +93 -0
- mannf/product/llm/prompts.py +403 -0
- mannf/product/llm/root_cause_service.py +311 -0
- mannf/product/llm/test_plan_models.py +78 -0
- mannf/product/metrics.py +149 -0
- mannf/product/middleware/__init__.py +3 -0
- mannf/product/middleware/audit_middleware.py +112 -0
- mannf/product/middleware/tenant_isolation.py +114 -0
- mannf/product/models.py +347 -0
- mannf/product/notifications/__init__.py +24 -0
- mannf/product/notifications/dispatcher.py +411 -0
- mannf/product/onboarding.py +190 -0
- mannf/product/orchestration/__init__.py +39 -0
- mannf/product/orchestration/ingest_scan_orchestrator.py +339 -0
- mannf/product/orchestration/pipeline.py +401 -0
- mannf/product/orchestrator.py +987 -0
- mannf/product/orchestrator_models.py +269 -0
- mannf/product/regression/__init__.py +36 -0
- mannf/product/regression/differ.py +172 -0
- mannf/product/regression/masking.py +100 -0
- mannf/product/regression/models.py +232 -0
- mannf/product/regression/recorder.py +124 -0
- mannf/product/regression/replayer.py +168 -0
- mannf/product/reports/__init__.py +10 -0
- mannf/product/reports/pdf.py +132 -0
- mannf/product/scheduling/__init__.py +57 -0
- mannf/product/scheduling/cron_utils.py +251 -0
- mannf/product/scheduling/engine.py +473 -0
- mannf/product/scheduling/models.py +86 -0
- mannf/product/scheduling/queue.py +894 -0
- mannf/product/scheduling/store.py +235 -0
- mannf/product/security/__init__.py +21 -0
- mannf/product/security/belief_guided.py +143 -0
- mannf/product/security/checks/__init__.py +55 -0
- mannf/product/security/checks/base.py +69 -0
- mannf/product/security/checks/bfla.py +77 -0
- mannf/product/security/checks/bola.py +77 -0
- mannf/product/security/checks/bopla.py +80 -0
- mannf/product/security/checks/broken_auth.py +86 -0
- mannf/product/security/checks/graphql_security.py +299 -0
- mannf/product/security/checks/inventory.py +70 -0
- mannf/product/security/checks/misconfig.py +158 -0
- mannf/product/security/checks/resource_consumption.py +70 -0
- mannf/product/security/checks/sensitive_flows.py +80 -0
- mannf/product/security/checks/ssrf.py +101 -0
- mannf/product/security/checks/unsafe_consumption.py +120 -0
- mannf/product/security/models.py +92 -0
- mannf/product/security/plugin_loader.py +182 -0
- mannf/product/security/reporter.py +92 -0
- mannf/product/security/scanner.py +183 -0
- mannf/product/server.py +6220 -0
- mannf/product/setup_wizard.py +873 -0
- mannf/product/status.py +404 -0
- mannf/product/storage/__init__.py +10 -0
- mannf/product/storage/artifact_store.py +343 -0
- mannf/product/telemetry.py +300 -0
- mannf/product/uninstall.py +169 -0
- mannf/product/upgrade.py +139 -0
- mannf/product/weights/__init__.py +13 -0
- mannf/product/weights/blob_store.py +299 -0
- mannf/product/weights/factory.py +42 -0
- mannf/product/weights/registry.py +159 -0
- mannf/product/weights/store.py +210 -0
- mannf/regression/__init__.py +7 -0
- mannf/regression/differ.py +9 -0
- mannf/regression/masking.py +9 -0
- mannf/regression/models.py +9 -0
- mannf/regression/recorder.py +9 -0
- mannf/regression/replayer.py +9 -0
- mannf/security/__init__.py +7 -0
- mannf/security/belief_guided.py +9 -0
- mannf/security/checks/__init__.py +7 -0
- mannf/security/checks/base.py +9 -0
- mannf/security/checks/bfla.py +9 -0
- mannf/security/checks/bola.py +9 -0
- mannf/security/checks/bopla.py +9 -0
- mannf/security/checks/broken_auth.py +9 -0
- mannf/security/checks/graphql_security.py +9 -0
- mannf/security/checks/inventory.py +9 -0
- mannf/security/checks/misconfig.py +9 -0
- mannf/security/checks/resource_consumption.py +9 -0
- mannf/security/checks/sensitive_flows.py +9 -0
- mannf/security/checks/ssrf.py +9 -0
- mannf/security/checks/unsafe_consumption.py +9 -0
- mannf/security/models.py +9 -0
- mannf/security/reporter.py +9 -0
- mannf/security/scanner.py +9 -0
- mannf/server.py +9 -0
- mannf/testing/__init__.py +7 -0
- mannf/testing/adaptive_controller.py +9 -0
- mannf/testing/models.py +9 -0
- mannf/weights/__init__.py +7 -0
- mannf/weights/registry.py +9 -0
- mannf/weights/store.py +9 -0
- nat_engine-1.dist-info/METADATA +555 -0
- nat_engine-1.dist-info/RECORD +299 -0
- nat_engine-1.dist-info/WHEEL +5 -0
- nat_engine-1.dist-info/entry_points.txt +4 -0
- nat_engine-1.dist-info/licenses/LICENSE +651 -0
- nat_engine-1.dist-info/licenses/NOTICE +178 -0
- nat_engine-1.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,510 @@
|
|
|
1
|
+
# Copyright (C) 2026 Brad Guider
|
|
2
|
+
# This file is part of NAT (Neural Agent Testing Framework).
|
|
3
|
+
# Licensed under the AGPL-3.0. See LICENSE for details.
|
|
4
|
+
# Commercial licensing available — see COMMERCIAL_LICENSE.md.
|
|
5
|
+
|
|
6
|
+
"""WebCrawlerAgent – autonomously discovers and maps a target web application.
|
|
7
|
+
|
|
8
|
+
Extends :class:`PlannerAgent` (which itself extends :class:`BDIAgent`) and
|
|
9
|
+
**composes** a Playwright browser lifecycle following the same lazy-init pattern
|
|
10
|
+
used by :class:`BrowserExecutorAgent` (rather than inheriting from it, since the
|
|
11
|
+
crawler is semantically a planner).
|
|
12
|
+
|
|
13
|
+
The agent performs a bounded BFS crawl over the target site and produces a
|
|
14
|
+
:class:`~mannf.core.browser.discovery_model.DiscoveryModel` — a structured JSON
|
|
15
|
+
artefact consumed by downstream Phase 2 components (Scenario Config Generator).
|
|
16
|
+
|
|
17
|
+
Key responsibilities
|
|
18
|
+
--------------------
|
|
19
|
+
* **BFS crawl loop** — visits pages up to ``max_pages`` / ``max_depth`` /
|
|
20
|
+
``crawl_timeout_s`` budget limits.
|
|
21
|
+
* **Form detection** — extracts fields with ``required``, ``label``, ``pattern``,
|
|
22
|
+
and ``submit_selector`` via enhanced JS.
|
|
23
|
+
* **Interactive element discovery** — buttons, links, dropdowns, tabs.
|
|
24
|
+
* **SPA support** — detects client-side route changes after clicks.
|
|
25
|
+
* **Belief integration** — updates :class:`~mannf.core.agents.belief_state.BeliefState`
|
|
26
|
+
based on page load stability.
|
|
27
|
+
* **Progress / result publishing** — emits ``CRAWL_PROGRESS`` messages during
|
|
28
|
+
crawling and a single ``CRAWL_RESULT`` when done.
|
|
29
|
+
|
|
30
|
+
Import note: Playwright is imported **lazily** inside ``start_browser()`` so
|
|
31
|
+
this module can be imported and tested without Playwright installed.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
from __future__ import annotations
|
|
35
|
+
|
|
36
|
+
import asyncio
|
|
37
|
+
import logging
|
|
38
|
+
import time
|
|
39
|
+
from collections import deque
|
|
40
|
+
from datetime import datetime, timezone
|
|
41
|
+
from typing import Any
|
|
42
|
+
from urllib.parse import urljoin, urlparse, urlunparse
|
|
43
|
+
|
|
44
|
+
from mannf.core.agents.planner_agent import PlannerAgent
|
|
45
|
+
from mannf.core.browser.discovery_model import (
|
|
46
|
+
DiscoveredField,
|
|
47
|
+
DiscoveredForm,
|
|
48
|
+
DiscoveredPage,
|
|
49
|
+
DiscoveryModel,
|
|
50
|
+
InteractiveElement,
|
|
51
|
+
PageEdge,
|
|
52
|
+
)
|
|
53
|
+
from mannf.core.browser.dom_snapshot import _ENHANCED_FORM_EXTRACTOR_JS, build_selector
|
|
54
|
+
from mannf.core.messaging.bus import MessageBus
|
|
55
|
+
from mannf.core.messaging.messages import Message, MessageType
|
|
56
|
+
|
|
57
|
+
logger = logging.getLogger(__name__)
|
|
58
|
+
|
|
59
|
+
# JavaScript to extract all outgoing same-page links.
|
|
60
|
+
_LINK_EXTRACTOR_JS = """
|
|
61
|
+
() => {
|
|
62
|
+
const hrefs = [];
|
|
63
|
+
document.querySelectorAll('a[href]').forEach(a => {
|
|
64
|
+
const h = a.getAttribute('href');
|
|
65
|
+
if (h && !h.startsWith('javascript:') && !h.startsWith('mailto:')
|
|
66
|
+
&& !h.startsWith('tel:')) {
|
|
67
|
+
hrefs.push(h);
|
|
68
|
+
}
|
|
69
|
+
});
|
|
70
|
+
return hrefs;
|
|
71
|
+
}
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
# JavaScript to extract non-form interactive elements.
|
|
75
|
+
_INTERACTIVE_EXTRACTOR_JS = """
|
|
76
|
+
() => {
|
|
77
|
+
function buildSelector(el) {
|
|
78
|
+
if (el.id) return '#' + el.id;
|
|
79
|
+
const testid = el.getAttribute('data-testid');
|
|
80
|
+
if (testid) return '[data-testid="' + testid + '"]';
|
|
81
|
+
const name = el.getAttribute('name');
|
|
82
|
+
if (name) return el.tagName.toLowerCase() + '[name="' + name + '"]';
|
|
83
|
+
const classes = Array.from(el.classList);
|
|
84
|
+
if (classes.length) return el.tagName.toLowerCase() + '.' + classes.join('.');
|
|
85
|
+
return el.tagName.toLowerCase();
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
const elements = [];
|
|
89
|
+
const selectors = [
|
|
90
|
+
{ sel: 'button:not([type="submit"]):not([type="reset"])', type: 'button' },
|
|
91
|
+
{ sel: 'a[href]', type: 'link' },
|
|
92
|
+
{ sel: 'select', type: 'dropdown' },
|
|
93
|
+
{ sel: '[role="tab"]', type: 'tab' },
|
|
94
|
+
];
|
|
95
|
+
selectors.forEach(({ sel, type }) => {
|
|
96
|
+
document.querySelectorAll(sel).forEach(el => {
|
|
97
|
+
const text = (el.innerText || el.textContent || '').trim().slice(0, 200);
|
|
98
|
+
elements.push({
|
|
99
|
+
selector: buildSelector(el),
|
|
100
|
+
text,
|
|
101
|
+
type,
|
|
102
|
+
tag: el.tagName.toLowerCase(),
|
|
103
|
+
});
|
|
104
|
+
});
|
|
105
|
+
});
|
|
106
|
+
return elements;
|
|
107
|
+
}
|
|
108
|
+
"""
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _normalize_url(url: str) -> str:
|
|
112
|
+
"""Strip fragment and normalize trailing slash for deduplication."""
|
|
113
|
+
parsed = urlparse(url)
|
|
114
|
+
# Remove fragment
|
|
115
|
+
normalized = parsed._replace(fragment="")
|
|
116
|
+
# Normalize path: ensure non-empty paths end consistently (no trailing slash
|
|
117
|
+
# for paths that are not just "/")
|
|
118
|
+
path = normalized.path
|
|
119
|
+
if path != "/" and path.endswith("/"):
|
|
120
|
+
path = path.rstrip("/")
|
|
121
|
+
normalized = normalized._replace(path=path)
|
|
122
|
+
return urlunparse(normalized)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def _same_origin(url: str, base_url: str) -> bool:
|
|
126
|
+
"""Return True when *url* shares the scheme + netloc of *base_url*."""
|
|
127
|
+
p = urlparse(url)
|
|
128
|
+
b = urlparse(base_url)
|
|
129
|
+
return p.scheme == b.scheme and p.netloc == b.netloc
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def _url_to_path(url: str, base_url: str) -> str:
|
|
133
|
+
"""Return the path (relative to origin) for display in the discovery model."""
|
|
134
|
+
parsed = urlparse(url)
|
|
135
|
+
path = parsed.path or "/"
|
|
136
|
+
if parsed.query:
|
|
137
|
+
path = f"{path}?{parsed.query}"
|
|
138
|
+
return path
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
class WebCrawlerAgent(PlannerAgent):
|
|
142
|
+
"""Autonomously crawls a web application and produces a :class:`DiscoveryModel`.
|
|
143
|
+
|
|
144
|
+
Parameters
|
|
145
|
+
----------
|
|
146
|
+
agent_id:
|
|
147
|
+
Unique identifier for this agent.
|
|
148
|
+
bus:
|
|
149
|
+
Shared message bus.
|
|
150
|
+
service_names:
|
|
151
|
+
Services tracked by the BDI belief state.
|
|
152
|
+
base_url:
|
|
153
|
+
Root URL of the application to crawl.
|
|
154
|
+
browser_type:
|
|
155
|
+
Playwright browser family: ``"chromium"``, ``"firefox"``, or ``"webkit"``.
|
|
156
|
+
headless:
|
|
157
|
+
Run the browser in headless mode.
|
|
158
|
+
viewport_width:
|
|
159
|
+
Browser viewport width in pixels.
|
|
160
|
+
viewport_height:
|
|
161
|
+
Browser viewport height in pixels.
|
|
162
|
+
max_pages:
|
|
163
|
+
Hard cap on the number of pages to visit.
|
|
164
|
+
max_depth:
|
|
165
|
+
Maximum BFS depth to follow links.
|
|
166
|
+
crawl_timeout_s:
|
|
167
|
+
Total crawl time budget in seconds (default 3 minutes).
|
|
168
|
+
page_timeout_ms:
|
|
169
|
+
Per-page navigation timeout in milliseconds.
|
|
170
|
+
"""
|
|
171
|
+
|
|
172
|
+
def __init__(
|
|
173
|
+
self,
|
|
174
|
+
agent_id: str,
|
|
175
|
+
bus: MessageBus,
|
|
176
|
+
service_names: list[str],
|
|
177
|
+
*,
|
|
178
|
+
base_url: str,
|
|
179
|
+
browser_type: str = "chromium",
|
|
180
|
+
headless: bool = True,
|
|
181
|
+
viewport_width: int = 1280,
|
|
182
|
+
viewport_height: int = 720,
|
|
183
|
+
max_pages: int = 50,
|
|
184
|
+
max_depth: int = 5,
|
|
185
|
+
crawl_timeout_s: float = 180.0,
|
|
186
|
+
page_timeout_ms: int = 10_000,
|
|
187
|
+
) -> None:
|
|
188
|
+
super().__init__(agent_id, bus, service_names)
|
|
189
|
+
self._base_url = base_url.rstrip("/")
|
|
190
|
+
self._browser_type = browser_type
|
|
191
|
+
self._headless = headless
|
|
192
|
+
self._viewport_width = viewport_width
|
|
193
|
+
self._viewport_height = viewport_height
|
|
194
|
+
self._max_pages = max_pages
|
|
195
|
+
self._max_depth = max_depth
|
|
196
|
+
self._crawl_timeout_s = crawl_timeout_s
|
|
197
|
+
self._page_timeout_ms = page_timeout_ms
|
|
198
|
+
|
|
199
|
+
# Composed Playwright state (lazy-initialised by start_browser)
|
|
200
|
+
self._playwright: Any = None
|
|
201
|
+
self._browser: Any = None
|
|
202
|
+
self._context: Any = None
|
|
203
|
+
self._page: Any = None
|
|
204
|
+
|
|
205
|
+
# Crawl state
|
|
206
|
+
self._crawl_started = False
|
|
207
|
+
self._discovery_model: DiscoveryModel | None = None
|
|
208
|
+
|
|
209
|
+
# ------------------------------------------------------------------
|
|
210
|
+
# Subscription
|
|
211
|
+
# ------------------------------------------------------------------
|
|
212
|
+
|
|
213
|
+
@property
|
|
214
|
+
def subscribed_types(self) -> set[MessageType]:
|
|
215
|
+
return super().subscribed_types | {
|
|
216
|
+
MessageType.CRAWL_PROGRESS,
|
|
217
|
+
MessageType.CRAWL_RESULT,
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
# ------------------------------------------------------------------
|
|
221
|
+
# Idle hook — drives the crawl
|
|
222
|
+
# ------------------------------------------------------------------
|
|
223
|
+
|
|
224
|
+
async def _on_idle(self) -> None:
|
|
225
|
+
if self._crawl_started:
|
|
226
|
+
return
|
|
227
|
+
self._crawl_started = True
|
|
228
|
+
try:
|
|
229
|
+
await self._run_crawl()
|
|
230
|
+
except Exception: # noqa: BLE001
|
|
231
|
+
logger.exception(
|
|
232
|
+
"WebCrawlerAgent %s: unhandled error during crawl", self.agent_id
|
|
233
|
+
)
|
|
234
|
+
|
|
235
|
+
# ------------------------------------------------------------------
|
|
236
|
+
# Browser lifecycle (composed, not inherited)
|
|
237
|
+
# ------------------------------------------------------------------
|
|
238
|
+
|
|
239
|
+
async def start_browser(self) -> None:
|
|
240
|
+
"""Launch Playwright and open a browser context + page (idempotent)."""
|
|
241
|
+
if self._playwright is not None:
|
|
242
|
+
return
|
|
243
|
+
|
|
244
|
+
from playwright.async_api import async_playwright # lazy import
|
|
245
|
+
|
|
246
|
+
self._playwright = await async_playwright().start()
|
|
247
|
+
launcher = getattr(self._playwright, self._browser_type)
|
|
248
|
+
self._browser = await launcher.launch(headless=self._headless)
|
|
249
|
+
self._context = await self._browser.new_context(
|
|
250
|
+
viewport={"width": self._viewport_width, "height": self._viewport_height}
|
|
251
|
+
)
|
|
252
|
+
self._page = await self._context.new_page()
|
|
253
|
+
logger.info(
|
|
254
|
+
"WebCrawlerAgent %s: browser started (%s, headless=%s)",
|
|
255
|
+
self.agent_id,
|
|
256
|
+
self._browser_type,
|
|
257
|
+
self._headless,
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
async def stop_browser(self) -> None:
|
|
261
|
+
"""Close the browser and Playwright instance (idempotent)."""
|
|
262
|
+
if self._playwright is None:
|
|
263
|
+
return
|
|
264
|
+
|
|
265
|
+
try:
|
|
266
|
+
if self._context is not None:
|
|
267
|
+
await self._context.close()
|
|
268
|
+
if self._browser is not None:
|
|
269
|
+
await self._browser.close()
|
|
270
|
+
await self._playwright.stop()
|
|
271
|
+
except Exception: # noqa: BLE001
|
|
272
|
+
logger.exception(
|
|
273
|
+
"WebCrawlerAgent %s: error during stop_browser", self.agent_id
|
|
274
|
+
)
|
|
275
|
+
finally:
|
|
276
|
+
self._page = None
|
|
277
|
+
self._context = None
|
|
278
|
+
self._browser = None
|
|
279
|
+
self._playwright = None
|
|
280
|
+
logger.info("WebCrawlerAgent %s: browser stopped", self.agent_id)
|
|
281
|
+
|
|
282
|
+
# ------------------------------------------------------------------
|
|
283
|
+
# Core crawl loop
|
|
284
|
+
# ------------------------------------------------------------------
|
|
285
|
+
|
|
286
|
+
async def _run_crawl(self) -> None:
|
|
287
|
+
"""Run the full BFS crawl and publish the discovery model."""
|
|
288
|
+
start_time = time.monotonic()
|
|
289
|
+
start_norm = _normalize_url(self._base_url)
|
|
290
|
+
|
|
291
|
+
model = DiscoveryModel(base_url=self._base_url)
|
|
292
|
+
visited: set[str] = set()
|
|
293
|
+
# Queue entries: (normalized_url, depth)
|
|
294
|
+
queue: deque[tuple[str, int]] = deque()
|
|
295
|
+
queue.append((start_norm, 0))
|
|
296
|
+
|
|
297
|
+
await self.start_browser()
|
|
298
|
+
try:
|
|
299
|
+
while queue:
|
|
300
|
+
if len(visited) >= self._max_pages:
|
|
301
|
+
logger.info(
|
|
302
|
+
"WebCrawlerAgent %s: max_pages=%d reached, stopping",
|
|
303
|
+
self.agent_id, self._max_pages,
|
|
304
|
+
)
|
|
305
|
+
break
|
|
306
|
+
elapsed = time.monotonic() - start_time
|
|
307
|
+
if elapsed >= self._crawl_timeout_s:
|
|
308
|
+
logger.info(
|
|
309
|
+
"WebCrawlerAgent %s: crawl_timeout=%.1fs reached, stopping",
|
|
310
|
+
self.agent_id, self._crawl_timeout_s,
|
|
311
|
+
)
|
|
312
|
+
break
|
|
313
|
+
|
|
314
|
+
url, depth = queue.popleft()
|
|
315
|
+
if url in visited:
|
|
316
|
+
continue
|
|
317
|
+
if depth > self._max_depth:
|
|
318
|
+
continue
|
|
319
|
+
|
|
320
|
+
visited.add(url)
|
|
321
|
+
page_data = await self._visit_page(url, depth, model, visited, queue)
|
|
322
|
+
if page_data is not None:
|
|
323
|
+
model.pages.append(page_data)
|
|
324
|
+
# Register page as a "service" in the belief state (dynamic)
|
|
325
|
+
page_path = _url_to_path(url, self._base_url)
|
|
326
|
+
if page_path not in self.beliefs.fault_likelihood:
|
|
327
|
+
self.beliefs.update(page_path, 0.5)
|
|
328
|
+
|
|
329
|
+
await self._publish_progress(model)
|
|
330
|
+
finally:
|
|
331
|
+
await self.stop_browser()
|
|
332
|
+
|
|
333
|
+
elapsed = time.monotonic() - start_time
|
|
334
|
+
model.crawl_duration_s = round(elapsed, 3)
|
|
335
|
+
model.crawl_timestamp = datetime.now(timezone.utc).isoformat()
|
|
336
|
+
self._discovery_model = model
|
|
337
|
+
|
|
338
|
+
await self._publish_result(model)
|
|
339
|
+
logger.info(
|
|
340
|
+
"WebCrawlerAgent %s: crawl complete — %d pages, %d edges in %.1fs",
|
|
341
|
+
self.agent_id, len(model.pages), len(model.edges), elapsed,
|
|
342
|
+
)
|
|
343
|
+
|
|
344
|
+
async def _visit_page(
|
|
345
|
+
self,
|
|
346
|
+
url: str,
|
|
347
|
+
depth: int,
|
|
348
|
+
model: DiscoveryModel,
|
|
349
|
+
visited: set[str],
|
|
350
|
+
queue: deque[tuple[str, int]],
|
|
351
|
+
) -> DiscoveredPage | None:
|
|
352
|
+
"""Navigate to *url*, extract page data, and enqueue new links.
|
|
353
|
+
|
|
354
|
+
Returns the :class:`DiscoveredPage` on success, or ``None`` on error.
|
|
355
|
+
"""
|
|
356
|
+
assert self._page is not None, "Browser page is not initialised"
|
|
357
|
+
|
|
358
|
+
try:
|
|
359
|
+
await self._page.goto(
|
|
360
|
+
url,
|
|
361
|
+
wait_until="networkidle",
|
|
362
|
+
timeout=self._page_timeout_ms,
|
|
363
|
+
)
|
|
364
|
+
except Exception: # noqa: BLE001
|
|
365
|
+
try:
|
|
366
|
+
await self._page.goto(
|
|
367
|
+
url,
|
|
368
|
+
wait_until="load",
|
|
369
|
+
timeout=self._page_timeout_ms,
|
|
370
|
+
)
|
|
371
|
+
except Exception: # noqa: BLE001
|
|
372
|
+
logger.warning(
|
|
373
|
+
"WebCrawlerAgent %s: failed to load %r, skipping", self.agent_id, url
|
|
374
|
+
)
|
|
375
|
+
# Update belief — page failed to load → higher fault likelihood
|
|
376
|
+
page_path = _url_to_path(url, self._base_url)
|
|
377
|
+
if page_path in self.beliefs.fault_likelihood:
|
|
378
|
+
self.beliefs.update(page_path, 1.0)
|
|
379
|
+
return None
|
|
380
|
+
|
|
381
|
+
# Wait for any SPA hydration
|
|
382
|
+
try:
|
|
383
|
+
await self._page.wait_for_load_state(
|
|
384
|
+
"networkidle", timeout=self._page_timeout_ms
|
|
385
|
+
)
|
|
386
|
+
except Exception: # noqa: BLE001
|
|
387
|
+
pass # best-effort; continue with whatever the DOM has
|
|
388
|
+
|
|
389
|
+
current_url = _normalize_url(self._page.url)
|
|
390
|
+
title = await self._page.title()
|
|
391
|
+
|
|
392
|
+
# Extract links
|
|
393
|
+
raw_links: list[str] = await self._page.evaluate(_LINK_EXTRACTOR_JS)
|
|
394
|
+
links: list[str] = []
|
|
395
|
+
for href in raw_links:
|
|
396
|
+
absolute = urljoin(current_url, href)
|
|
397
|
+
norm = _normalize_url(absolute)
|
|
398
|
+
if _same_origin(norm, self._base_url):
|
|
399
|
+
links.append(_url_to_path(norm, self._base_url))
|
|
400
|
+
if norm not in visited and depth + 1 <= self._max_depth:
|
|
401
|
+
queue.append((norm, depth + 1))
|
|
402
|
+
# Record graph edge
|
|
403
|
+
parsed = urlparse(href)
|
|
404
|
+
link_sel = (
|
|
405
|
+
f'a[href="{href}"]'
|
|
406
|
+
if not parsed.fragment
|
|
407
|
+
else f'a[href="{href}"]'
|
|
408
|
+
)
|
|
409
|
+
model.edges.append(
|
|
410
|
+
PageEdge(
|
|
411
|
+
from_url=_url_to_path(current_url, self._base_url),
|
|
412
|
+
to_url=_url_to_path(norm, self._base_url),
|
|
413
|
+
action="click",
|
|
414
|
+
selector=link_sel,
|
|
415
|
+
)
|
|
416
|
+
)
|
|
417
|
+
|
|
418
|
+
# Extract enhanced forms
|
|
419
|
+
raw_forms: list[dict[str, Any]] = await self._page.evaluate(
|
|
420
|
+
_ENHANCED_FORM_EXTRACTOR_JS
|
|
421
|
+
)
|
|
422
|
+
forms = _parse_forms(raw_forms)
|
|
423
|
+
|
|
424
|
+
# Extract interactive elements
|
|
425
|
+
raw_interactive: list[dict[str, Any]] = await self._page.evaluate(
|
|
426
|
+
_INTERACTIVE_EXTRACTOR_JS
|
|
427
|
+
)
|
|
428
|
+
interactive_elements = [
|
|
429
|
+
InteractiveElement(
|
|
430
|
+
selector=r.get("selector", ""),
|
|
431
|
+
text=r.get("text", ""),
|
|
432
|
+
type=r.get("type", "button"),
|
|
433
|
+
tag=r.get("tag", ""),
|
|
434
|
+
)
|
|
435
|
+
for r in raw_interactive
|
|
436
|
+
]
|
|
437
|
+
|
|
438
|
+
# Belief update — page loaded successfully → lower fault likelihood
|
|
439
|
+
page_path = _url_to_path(current_url, self._base_url)
|
|
440
|
+
if page_path in self.beliefs.fault_likelihood:
|
|
441
|
+
self.beliefs.update(page_path, 0.0)
|
|
442
|
+
|
|
443
|
+
logger.debug(
|
|
444
|
+
"WebCrawlerAgent %s: visited %r depth=%d links=%d forms=%d",
|
|
445
|
+
self.agent_id, page_path, depth, len(links), len(forms),
|
|
446
|
+
)
|
|
447
|
+
|
|
448
|
+
return DiscoveredPage(
|
|
449
|
+
url=page_path,
|
|
450
|
+
title=title,
|
|
451
|
+
forms=forms,
|
|
452
|
+
links=links,
|
|
453
|
+
interactive_elements=interactive_elements,
|
|
454
|
+
)
|
|
455
|
+
|
|
456
|
+
# ------------------------------------------------------------------
|
|
457
|
+
# Message publishing
|
|
458
|
+
# ------------------------------------------------------------------
|
|
459
|
+
|
|
460
|
+
async def _publish_progress(self, model: DiscoveryModel) -> None:
|
|
461
|
+
"""Publish a CRAWL_PROGRESS message with current crawl statistics."""
|
|
462
|
+
payload = {
|
|
463
|
+
"pages_discovered": len(model.pages),
|
|
464
|
+
"forms_found": sum(len(p.forms) for p in model.pages),
|
|
465
|
+
"edges_found": len(model.edges),
|
|
466
|
+
}
|
|
467
|
+
await self._publish(
|
|
468
|
+
Message(MessageType.CRAWL_PROGRESS, self.agent_id, payload)
|
|
469
|
+
)
|
|
470
|
+
|
|
471
|
+
async def _publish_result(self, model: DiscoveryModel) -> None:
|
|
472
|
+
"""Publish a CRAWL_RESULT message carrying the full discovery model."""
|
|
473
|
+
await self._publish(
|
|
474
|
+
Message(MessageType.CRAWL_RESULT, self.agent_id, model.to_dict())
|
|
475
|
+
)
|
|
476
|
+
|
|
477
|
+
|
|
478
|
+
# ---------------------------------------------------------------------------
|
|
479
|
+
# Parsing helpers
|
|
480
|
+
# ---------------------------------------------------------------------------
|
|
481
|
+
|
|
482
|
+
def _parse_forms(raw_forms: list[dict[str, Any]]) -> list[DiscoveredForm]:
|
|
483
|
+
"""Convert raw JS-extracted form dicts into :class:`DiscoveredForm` objects."""
|
|
484
|
+
forms: list[DiscoveredForm] = []
|
|
485
|
+
for rf in raw_forms:
|
|
486
|
+
raw_fields = rf.get("fields", [])
|
|
487
|
+
fields: list[DiscoveredField] = []
|
|
488
|
+
for f in raw_fields:
|
|
489
|
+
# Determine the input type robustly
|
|
490
|
+
field_type = f.get("type", "") or f.get("tag", "text") or "text"
|
|
491
|
+
fields.append(
|
|
492
|
+
DiscoveredField(
|
|
493
|
+
selector=f.get("selector", ""),
|
|
494
|
+
type=field_type,
|
|
495
|
+
required=bool(f.get("required", False)),
|
|
496
|
+
label=f.get("label", ""),
|
|
497
|
+
name=f.get("name", ""),
|
|
498
|
+
placeholder=f.get("placeholder", ""),
|
|
499
|
+
pattern=f.get("pattern", ""),
|
|
500
|
+
)
|
|
501
|
+
)
|
|
502
|
+
forms.append(
|
|
503
|
+
DiscoveredForm(
|
|
504
|
+
action=rf.get("action", ""),
|
|
505
|
+
method=(rf.get("method", "GET") or "GET").upper(),
|
|
506
|
+
fields=fields,
|
|
507
|
+
submit_selector=rf.get("submit_selector", ""),
|
|
508
|
+
)
|
|
509
|
+
)
|
|
510
|
+
return forms
|