tanli 0.0.1__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- redteam/__init__.py +3 -0
- redteam/agent.py +649 -0
- redteam/auth.py +196 -0
- redteam/cli.py +1043 -0
- redteam/config.py +103 -0
- redteam/cve_lookup.py +461 -0
- redteam/cvss.py +233 -0
- redteam/engagement.py +170 -0
- redteam/findings.py +432 -0
- redteam/guardtext.py +73 -0
- redteam/interactor.py +114 -0
- redteam/judge.py +164 -0
- redteam/knowledge.py +239 -0
- redteam/planner.py +87 -0
- redteam/playbook.py +411 -0
- redteam/playbooks/llm/playbook_1.yaml +75 -0
- redteam/playbooks/llm/playbook_10.yaml +113 -0
- redteam/playbooks/llm/playbook_11.yaml +103 -0
- redteam/playbooks/llm/playbook_2.yaml +58 -0
- redteam/playbooks/llm/playbook_3.yaml +59 -0
- redteam/playbooks/llm/playbook_4.yaml +43 -0
- redteam/playbooks/llm/playbook_5.yaml +37 -0
- redteam/playbooks/llm/playbook_6.yaml +101 -0
- redteam/playbooks/llm/playbook_7.yaml +100 -0
- redteam/playbooks/llm/playbook_8.yaml +115 -0
- redteam/playbooks/llm/playbook_9.yaml +102 -0
- redteam/playbooks/web/playbook_10.yaml +21 -0
- redteam/playbooks/web/playbook_11.yaml +21 -0
- redteam/playbooks/web/playbook_12.yaml +22 -0
- redteam/playbooks/web/playbook_13.yaml +21 -0
- redteam/playbooks/web/playbook_14.yaml +20 -0
- redteam/playbooks/web/playbook_15.yaml +23 -0
- redteam/playbooks/web/playbook_16.yaml +21 -0
- redteam/playbooks/web/playbook_17.yaml +23 -0
- redteam/playbooks/web/playbook_18.yaml +25 -0
- redteam/playbooks/web/playbook_19.yaml +29 -0
- redteam/playbooks/web/playbook_20.yaml +25 -0
- redteam/playbooks/web/playbook_21.yaml +30 -0
- redteam/playbooks/web/playbook_22.yaml +24 -0
- redteam/playbooks/web/playbook_23.yaml +28 -0
- redteam/playbooks/web/playbook_24.yaml +29 -0
- redteam/playbooks/web/playbook_25.yaml +27 -0
- redteam/playbooks/web/playbook_26.yaml +29 -0
- redteam/playbooks/web/playbook_27.yaml +32 -0
- redteam/playbooks/web/playbook_28.yaml +29 -0
- redteam/playbooks/web/playbook_29.yaml +31 -0
- redteam/playbooks/web/playbook_30.yaml +29 -0
- redteam/playbooks/web/playbook_6.yaml +29 -0
- redteam/playbooks/web/playbook_7.yaml +23 -0
- redteam/playbooks/web/playbook_8.yaml +24 -0
- redteam/playbooks/web/playbook_9.yaml +22 -0
- redteam/report.py +292 -0
- redteam/sandbox.py +101 -0
- redteam/scanners.py +435 -0
- redteam/target_lab.py +287 -0
- redteam/triage.py +86 -0
- redteam/version_watch.py +218 -0
- redteam/web_config.py +289 -0
- redteam/workspace.py +167 -0
- tanli-0.0.1.dist-info/METADATA +235 -0
- tanli-0.0.1.dist-info/RECORD +65 -0
- tanli-0.0.1.dist-info/WHEEL +5 -0
- tanli-0.0.1.dist-info/entry_points.txt +3 -0
- tanli-0.0.1.dist-info/licenses/LICENSE +201 -0
- tanli-0.0.1.dist-info/top_level.txt +1 -0
redteam/__init__.py
ADDED
redteam/agent.py
ADDED
|
@@ -0,0 +1,649 @@
|
|
|
1
|
+
"""Autonomous agent loop — OpenCode/Hermes 級 tool-loop 自主性,安全圍籬內執行。
|
|
2
|
+
|
|
3
|
+
設計(spec §6 精神延伸,用戶 2026-09-19 需求):
|
|
4
|
+
- 靜態 6 節點 plan 之外,新增真正的 LLM tool-loop:模型自己決定下一步
|
|
5
|
+
(探什麼、查哪個 CVE、動態怎麼試),直到達成目標或預算耗盡。
|
|
6
|
+
- 每個工具都是「圍籬內的原語」:ScopeGuard/read-only/token budget 在工具
|
|
7
|
+
層強制,模型無法繞過(與 hermes 的 approval 分層同理:能力給足,權限收緊)。
|
|
8
|
+
- 不信任指紋版本原則寫進 system prompt:發現套件/框架 → cve_lookup 查全部
|
|
9
|
+
CVE → 以被動指紋+動態安全探測交叉核實 → 才有資格宣稱漏洞。
|
|
10
|
+
|
|
11
|
+
Brain 可注入(FakeBrain 測試);生產用 OpenAI-compatible endpoint
|
|
12
|
+
(與 LLMJudge 同源配置 REDTEAM_JUDGE_*)。
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import json
|
|
18
|
+
import re
|
|
19
|
+
from dataclasses import dataclass, field
|
|
20
|
+
from typing import Any, Protocol
|
|
21
|
+
|
|
22
|
+
from .cve_lookup import enrich as cve_enrich, lookup as cve_lookup_backend
|
|
23
|
+
from .guardtext import SYSTEM_RULE as GUARD_RULE, guard_observation
|
|
24
|
+
from .knowledge import (catalog as kb_catalog, get_by_id as kb_get,
|
|
25
|
+
load_library as kb_load, load_user_skills,
|
|
26
|
+
search as kb_search)
|
|
27
|
+
from .interactor import RedTeamHTTP
|
|
28
|
+
from .triage import triage as triage_score, sort_findings
|
|
29
|
+
from .version_watch import detect_releases, watch as version_watch
|
|
30
|
+
from .workspace import ProbeLog, Workspace
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
# Tool 層(每個工具 = 圍籬內原語)
|
|
35
|
+
# ---------------------------------------------------------------------------
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class ToolResult:
|
|
39
|
+
ok: bool
|
|
40
|
+
data: Any
|
|
41
|
+
note: str = ""
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class AgentTools:
|
|
45
|
+
"""綁定單一 target + ScopeGuard/read-only HTTP 客戶端的工具集。
|
|
46
|
+
|
|
47
|
+
add_finding / finish 由 loop 內部處理;這裡只實作探測/查詢類工具。
|
|
48
|
+
"""
|
|
49
|
+
|
|
50
|
+
def __init__(self, http: RedTeamHTTP, target: str, *, docker_bridge=None,
|
|
51
|
+
workspace: Workspace | None = None):
|
|
52
|
+
self.http = http
|
|
53
|
+
self.target = target
|
|
54
|
+
self.docker_bridge = docker_bridge
|
|
55
|
+
self.probe_count = 0
|
|
56
|
+
self.max_probes = 60 # 硬性總探測量,防失控轟炸
|
|
57
|
+
self.ws = workspace # None = 不啟用工作區(向後兼容)
|
|
58
|
+
self._kb = None # get_playbook 知識庫 lazy load
|
|
59
|
+
self._get_cache: dict[str, ToolResult] = {} # 安全方法結果快取(省預算)
|
|
60
|
+
self.probe_log: list[ProbeLog] = [] # 跨會話記憶的探測軌跡(含失敗)
|
|
61
|
+
|
|
62
|
+
# ---- schema(提供給 brain) ----
|
|
63
|
+
SCHEMAS: list[dict] = [
|
|
64
|
+
{
|
|
65
|
+
"name": "http_request",
|
|
66
|
+
"description": ("Send an HTTP request to a URL inside the authorized scope. "
|
|
67
|
+
"Returns status, selected headers, and truncated body. "
|
|
68
|
+
"In read-only mode non-safe methods are physically blocked."),
|
|
69
|
+
"parameters": {
|
|
70
|
+
"type": "object",
|
|
71
|
+
"properties": {
|
|
72
|
+
"method": {"type": "string", "enum": ["GET", "HEAD", "OPTIONS", "POST", "PUT", "DELETE"]},
|
|
73
|
+
"url": {"type": "string", "description": "Must stay within authorized scope"},
|
|
74
|
+
"headers": {"type": "object", "additionalProperties": {"type": "string"}},
|
|
75
|
+
"body": {"type": "string"},
|
|
76
|
+
},
|
|
77
|
+
"required": ["method", "url"],
|
|
78
|
+
},
|
|
79
|
+
},
|
|
80
|
+
{
|
|
81
|
+
"name": "fingerprint",
|
|
82
|
+
"description": ("Passively fingerprint the target: fetch it, extract server "
|
|
83
|
+
"headers, generator meta, and any <pkg>@<ver> release "
|
|
84
|
+
"declarations (self-reported versions — treat as hypothesis, "
|
|
85
|
+
"never as truth)."),
|
|
86
|
+
"parameters": {"type": "object", "properties": {}, "required": []},
|
|
87
|
+
},
|
|
88
|
+
{
|
|
89
|
+
"name": "cve_lookup",
|
|
90
|
+
"description": ("Query public CVE sources (GHSA + OSV + NVD) with EPSS "
|
|
91
|
+
"likelihood + CISA KEV enrichment. Give either "
|
|
92
|
+
"cve_id (CVE-YYYY-NNNNN) for exact lookup, or product "
|
|
93
|
+
"(+ecosystem/version optional) for product-level listing of "
|
|
94
|
+
"ALL published CVEs — do not trust the target's claimed "
|
|
95
|
+
"version, verify which versions are affected and probe. "
|
|
96
|
+
"kev=true means actively exploited in the wild: verify first."),
|
|
97
|
+
"parameters": {
|
|
98
|
+
"type": "object",
|
|
99
|
+
"properties": {
|
|
100
|
+
"cve_id": {"type": "string"},
|
|
101
|
+
"product": {"type": "string"},
|
|
102
|
+
"ecosystem": {"type": "string",
|
|
103
|
+
"description": "npm|pip|composer|maven|go|cargo|rubygems..."},
|
|
104
|
+
"version": {"type": "string"},
|
|
105
|
+
"likelihood": {"type": "boolean",
|
|
106
|
+
"description": "set false to skip EPSS/KEV enrichment (faster)"},
|
|
107
|
+
},
|
|
108
|
+
"required": [],
|
|
109
|
+
},
|
|
110
|
+
},
|
|
111
|
+
{
|
|
112
|
+
"name": "version_watch",
|
|
113
|
+
"description": ("Compare a specific pkg@version against GHSA for applicable "
|
|
114
|
+
"unpatched CVEs (fast, version-range matched). Only meaningful "
|
|
115
|
+
"for framework/ecosystem packages."),
|
|
116
|
+
"parameters": {
|
|
117
|
+
"type": "object",
|
|
118
|
+
"properties": {"product": {"type": "string"}, "ecosystem": {"type": "string"},
|
|
119
|
+
"version": {"type": "string"}},
|
|
120
|
+
"required": ["product", "version"],
|
|
121
|
+
},
|
|
122
|
+
},
|
|
123
|
+
{
|
|
124
|
+
"name": "web_config_probe",
|
|
125
|
+
"description": ("Run the deterministic web-configuration probe suite against "
|
|
126
|
+
"the target (security headers, cookie flags, exposed paths, "
|
|
127
|
+
"method policies). Pure read-only, no Docker. Returns "
|
|
128
|
+
"confirmed config weaknesses with OWASP mapping."),
|
|
129
|
+
"parameters": {"type": "object", "properties": {}, "required": []},
|
|
130
|
+
},
|
|
131
|
+
{
|
|
132
|
+
"name": "nuclei_cve_probe",
|
|
133
|
+
"description": ("Attempt a dynamic check of a specific CVE by running nuclei "
|
|
134
|
+
"with template tag = CVE id, inside the Docker sandbox. "
|
|
135
|
+
"Requires Docker; respects read-only & scope. This is the "
|
|
136
|
+
"'動態嘗試' path — match against the REAL detected product/version."),
|
|
137
|
+
"parameters": {
|
|
138
|
+
"type": "object",
|
|
139
|
+
"properties": {"cve_id": {"type": "string"}},
|
|
140
|
+
"required": ["cve_id"],
|
|
141
|
+
},
|
|
142
|
+
},
|
|
143
|
+
{
|
|
144
|
+
"name": "get_playbook",
|
|
145
|
+
"description": ("Query the playbook library (attack theory / defense "
|
|
146
|
+
"checklist templates: OWASP GenAI LLM playbooks + web "
|
|
147
|
+
"methodology + user-injected markdown skills from "
|
|
148
|
+
"~/.tanli/skills). Call with no args to get a compact catalog; "
|
|
149
|
+
"then fetch full steps by id, or search by keyword/owasp. "
|
|
150
|
+
"Use playbooks as REFERENCE workflows for planning — they "
|
|
151
|
+
"encode standard attack/defense procedures."),
|
|
152
|
+
"parameters": {
|
|
153
|
+
"type": "object",
|
|
154
|
+
"properties": {
|
|
155
|
+
"id": {"type": "string", "description": "Exact playbook id, e.g. llm-001"},
|
|
156
|
+
"query": {"type": "string", "description": "Keyword search across all playbooks"},
|
|
157
|
+
"owasp": {"type": "string", "description": "Filter by OWASP code, e.g. LLM01"},
|
|
158
|
+
"target_type": {"type": "string", "description": "Filter: llm_app | web_service"},
|
|
159
|
+
},
|
|
160
|
+
"required": [],
|
|
161
|
+
},
|
|
162
|
+
},
|
|
163
|
+
{
|
|
164
|
+
"name": "read_tool_output",
|
|
165
|
+
"description": ("Read back an offloaded large tool output by its file path "
|
|
166
|
+
"(only paths under this engagement's tool-outputs/ are "
|
|
167
|
+
"allowed). Use when a previous result shows [OFFLOADED ...]."),
|
|
168
|
+
"parameters": {
|
|
169
|
+
"type": "object",
|
|
170
|
+
"properties": {
|
|
171
|
+
"path": {"type": "string"},
|
|
172
|
+
"offset": {"type": "integer", "description": "char offset (default 0)"},
|
|
173
|
+
"limit": {"type": "integer", "description": "max chars (default 6000)"},
|
|
174
|
+
},
|
|
175
|
+
"required": ["path"],
|
|
176
|
+
},
|
|
177
|
+
},
|
|
178
|
+
{
|
|
179
|
+
"name": "write_note",
|
|
180
|
+
"description": ("Persist a working note in the engagement workspace "
|
|
181
|
+
"(notes/<name>.md). Survives across sessions — future "
|
|
182
|
+
"runs on this target load findings/lessons automatically. "
|
|
183
|
+
"Good for: hypotheses to test later, dead-end reasons, "
|
|
184
|
+
"attack-surface maps."),
|
|
185
|
+
"parameters": {
|
|
186
|
+
"type": "object",
|
|
187
|
+
"properties": {
|
|
188
|
+
"name": {"type": "string", "description": "filename-safe name"},
|
|
189
|
+
"content": {"type": "string"},
|
|
190
|
+
},
|
|
191
|
+
"required": ["name", "content"],
|
|
192
|
+
},
|
|
193
|
+
},
|
|
194
|
+
{
|
|
195
|
+
"name": "add_finding",
|
|
196
|
+
"description": ("Record a confirmed-or-candidate finding. Cite concrete "
|
|
197
|
+
"evidence (request results / CVE IDs / probe output). "
|
|
198
|
+
"Never claim a finding you did not observe."),
|
|
199
|
+
"parameters": {
|
|
200
|
+
"type": "object",
|
|
201
|
+
"properties": {
|
|
202
|
+
"title": {"type": "string"},
|
|
203
|
+
"severity": {"type": "string", "enum": ["critical", "high", "medium", "low", "info"]},
|
|
204
|
+
"category": {"type": "string"},
|
|
205
|
+
"description": {"type": "string"},
|
|
206
|
+
"evidence": {"type": "string"},
|
|
207
|
+
"cve": {"type": "string"},
|
|
208
|
+
"confidence": {"type": "number"},
|
|
209
|
+
"reachability": {"type": "string", "enum": [
|
|
210
|
+
"direct", "low_priv", "auth_required",
|
|
211
|
+
"user_interaction", "internal_only"],
|
|
212
|
+
"description": "how reachable the flaw is for an attacker"},
|
|
213
|
+
},
|
|
214
|
+
"required": ["title", "severity", "description", "evidence"],
|
|
215
|
+
},
|
|
216
|
+
},
|
|
217
|
+
{
|
|
218
|
+
"name": "finish",
|
|
219
|
+
"description": "End the assessment with a short summary of what was done and found.",
|
|
220
|
+
"parameters": {
|
|
221
|
+
"type": "object",
|
|
222
|
+
"properties": {"summary": {"type": "string"}},
|
|
223
|
+
"required": ["summary"],
|
|
224
|
+
},
|
|
225
|
+
},
|
|
226
|
+
]
|
|
227
|
+
|
|
228
|
+
# ---- dispatch ----
|
|
229
|
+
def call(self, name: str, args: dict) -> ToolResult:
|
|
230
|
+
if self.probe_count >= self.max_probes and name == "http_request":
|
|
231
|
+
return ToolResult(False, None, f"probe budget exhausted ({self.max_probes})")
|
|
232
|
+
fn = getattr(self, f"_t_{name}", None)
|
|
233
|
+
if fn is None:
|
|
234
|
+
return ToolResult(False, None, f"unknown tool: {name}")
|
|
235
|
+
try:
|
|
236
|
+
return fn(**args)
|
|
237
|
+
except Exception as e: # noqa: BLE001
|
|
238
|
+
return ToolResult(False, None, f"tool error ({e.__class__.__name__}): {e}")
|
|
239
|
+
|
|
240
|
+
def _t_http_request(self, method: str, url: str, headers: dict | None = None,
|
|
241
|
+
body: str | None = None) -> ToolResult:
|
|
242
|
+
# 安全方法(GET/HEAD/OPTIONS)同請求快取:省探測預算與 token,結果一致
|
|
243
|
+
m = method.upper()
|
|
244
|
+
if m in ("GET", "HEAD", "OPTIONS") and body is None and not headers:
|
|
245
|
+
hit = self._get_cache.get(f"{m} {url}")
|
|
246
|
+
if hit is not None:
|
|
247
|
+
return ToolResult(True, dict(hit.data) if isinstance(hit.data, dict) else hit.data,
|
|
248
|
+
"cached(先前已探過同一請求,結果相同,未耗預算)")
|
|
249
|
+
self.probe_count += 1
|
|
250
|
+
rec = self.http.request(method.upper(), url, headers=headers,
|
|
251
|
+
data=body.encode() if body else None)
|
|
252
|
+
if rec.blocked_by_scope:
|
|
253
|
+
self.probe_log.append(ProbeLog(m, url, "blocked-scope", ok=False))
|
|
254
|
+
return ToolResult(False, {"blocked": "scope"}, "BLOCKED by ScopeGuard — outside authorized scope")
|
|
255
|
+
if rec.blocked_by_readonly:
|
|
256
|
+
self.probe_log.append(ProbeLog(m, url, "blocked-readonly", ok=False))
|
|
257
|
+
return ToolResult(False, {"blocked": "read_only"},
|
|
258
|
+
"BLOCKED by read-only fuse — non-safe method physically blocked")
|
|
259
|
+
self.probe_log.append(ProbeLog(m, url, rec.status, ok=(rec.status or 0) < 400))
|
|
260
|
+
text = (rec.body or b"").decode("utf-8", "replace")
|
|
261
|
+
body_out = text[:3500]
|
|
262
|
+
if self.ws is not None and len(text) > 3500:
|
|
263
|
+
# 大輸出卸載(借鑑 RedAmon auto-offload):完整內容落盤,LLM 收 stub
|
|
264
|
+
body_out = self.ws.offload(f"http{rec.status}", text)
|
|
265
|
+
# 注入防護:目標內容包定界符 + 啟發式標記(借鑑 Decepticon)
|
|
266
|
+
body_out, _hits = guard_observation(body_out, source=url)
|
|
267
|
+
out = ToolResult(True, {
|
|
268
|
+
"status": rec.status,
|
|
269
|
+
"headers": {k: v for k, v in (rec.response_headers or {}).items()
|
|
270
|
+
if k.lower() in ("server", "x-powered-by", "content-type", "generator", "via")},
|
|
271
|
+
"body": body_out,
|
|
272
|
+
"body_len": len(text),
|
|
273
|
+
})
|
|
274
|
+
if m in ("GET", "HEAD", "OPTIONS") and body is None and not headers \
|
|
275
|
+
and not rec.blocked_by_scope and not rec.blocked_by_readonly:
|
|
276
|
+
self._get_cache[f"{m} {url}"] = out
|
|
277
|
+
return out
|
|
278
|
+
|
|
279
|
+
def _t_fingerprint(self) -> ToolResult:
|
|
280
|
+
self.probe_count += 1
|
|
281
|
+
rec = self.http.request("GET", self.target)
|
|
282
|
+
if rec.blocked_by_scope or rec.blocked_by_readonly:
|
|
283
|
+
self.probe_log.append(ProbeLog("GET", self.target, "blocked", ok=False))
|
|
284
|
+
return ToolResult(False, None, "target blocked by scope/read-only")
|
|
285
|
+
self.probe_log.append(ProbeLog("GET", self.target, rec.status,
|
|
286
|
+
ok=(rec.status or 0) < 400))
|
|
287
|
+
html = (rec.body or b"").decode("utf-8", "replace")
|
|
288
|
+
releases = detect_releases(html)
|
|
289
|
+
hdrs = {k: v for k, v in (rec.response_headers or {}).items()
|
|
290
|
+
if k.lower() in ("server", "x-powered-by", "generator", "via", "x-aspnet-version")}
|
|
291
|
+
meta = re.findall(r'<meta[^>]+name=["\']generator["\'][^>]+content=["\']([^"\']+)',
|
|
292
|
+
html, re.I)
|
|
293
|
+
head = html[:2500]
|
|
294
|
+
if self.ws is not None and len(html) > 2500:
|
|
295
|
+
head = self.ws.offload("fingerprint", html)
|
|
296
|
+
head, _ = guard_observation(head, source=self.target)
|
|
297
|
+
return ToolResult(True, {"status": rec.status, "headers": hdrs,
|
|
298
|
+
"generator_meta": meta, "release_decls": releases,
|
|
299
|
+
"body": head})
|
|
300
|
+
|
|
301
|
+
def _t_cve_lookup(self, cve_id: str | None = None, product: str | None = None,
|
|
302
|
+
ecosystem: str | None = None, version: str | None = None,
|
|
303
|
+
likelihood: bool = True) -> ToolResult:
|
|
304
|
+
r = cve_lookup_backend(cve_id=cve_id, product=product, ecosystem=ecosystem, version=version)
|
|
305
|
+
if not r.ok:
|
|
306
|
+
return ToolResult(False, {"error": r.error}, r.error)
|
|
307
|
+
intel = {}
|
|
308
|
+
if likelihood and r.advisories:
|
|
309
|
+
# EPSS + CISA KEV 情報加權(失敗如實標 status,不編造分數)
|
|
310
|
+
try:
|
|
311
|
+
intel = cve_enrich(r.advisories)
|
|
312
|
+
except Exception as e: # noqa: BLE001
|
|
313
|
+
intel = {"kev_status": f"error({e.__class__.__name__})"}
|
|
314
|
+
advs = [{"cve": a.cve, "src": a.source, "sev": a.severity, "product": a.product,
|
|
315
|
+
"range": a.vulnerable_range, "patched": a.first_patched,
|
|
316
|
+
"pub": a.published, "epss": a.epss, "kev": a.kev,
|
|
317
|
+
"summary": a.summary[:160]} for a in r.advisories]
|
|
318
|
+
# 高 likelihood 排前面(EPSS desc → KEV → severity),幫 agent 聚焦
|
|
319
|
+
advs.sort(key=lambda a: (-(a.get("epss") or 0), not a.get("kev"),
|
|
320
|
+
a["sev"] != "critical"))
|
|
321
|
+
return ToolResult(True, {"sources": r.sources, "count": len(advs),
|
|
322
|
+
"likelihood_intel": intel or "未啟用或不可用",
|
|
323
|
+
"latest_advisory_date": r.latest_advisory_date,
|
|
324
|
+
"advisories": advs[:40]},
|
|
325
|
+
"查無 CVE ≠ 無漏洞(注意資料源滯後;見 latest_advisory_date);"
|
|
326
|
+
"kev=true 的 CVE 已被真實利用,優先動態驗證")
|
|
327
|
+
|
|
328
|
+
def _t_version_watch(self, product: str, version: str, ecosystem: str = "npm") -> ToolResult:
|
|
329
|
+
wr = version_watch(product, ecosystem, version)
|
|
330
|
+
if wr.error:
|
|
331
|
+
return ToolResult(False, {"error": wr.error}, wr.error)
|
|
332
|
+
return ToolResult(True, {"product": wr.product, "version": wr.current_version,
|
|
333
|
+
"applicable_cves": [{"cve": m.cve, "sev": m.severity,
|
|
334
|
+
"patched": m.first_patched,
|
|
335
|
+
"range": m.vulnerable_range}
|
|
336
|
+
for m in wr.matches],
|
|
337
|
+
"latest_advisory_date": wr.latest_advisory_date})
|
|
338
|
+
|
|
339
|
+
def _t_web_config_probe(self) -> ToolResult:
|
|
340
|
+
from .web_config import WebConfigProbe
|
|
341
|
+
self.probe_count += 1
|
|
342
|
+
probes = WebConfigProbe(http=self.http).run(self.target)
|
|
343
|
+
hits = [{"name": p.name, "owasp": getattr(p, "owasp", ""),
|
|
344
|
+
"evidence": str(p.evidence)[:300]} for p in probes if p.passed]
|
|
345
|
+
return ToolResult(True, {"checked": len(probes), "hit_count": len(hits),
|
|
346
|
+
"hits": hits[:15]},
|
|
347
|
+
"deterministic 證據,可直接 add_finding")
|
|
348
|
+
|
|
349
|
+
def _t_get_playbook(self, id: str | None = None, query: str | None = None,
|
|
350
|
+
owasp: str | None = None,
|
|
351
|
+
target_type: str | None = None) -> ToolResult:
|
|
352
|
+
if self._kb is None:
|
|
353
|
+
# 內建 playbook + 使用者注入的 markdown 技能(~/.tanli/skills)
|
|
354
|
+
self._kb = kb_load() + load_user_skills()
|
|
355
|
+
if id:
|
|
356
|
+
doc = kb_get(self._kb, id)
|
|
357
|
+
if doc is None:
|
|
358
|
+
return ToolResult(False, {"catalog": kb_catalog(self._kb)},
|
|
359
|
+
f"playbook id 不存在:{id}(見 catalog 選一個)")
|
|
360
|
+
return ToolResult(True, {"playbook": doc.render(), "source": doc.source})
|
|
361
|
+
hits = kb_search(self._kb, query=query, owasp=owasp, target_type=target_type)
|
|
362
|
+
if not (query or owasp or target_type):
|
|
363
|
+
return ToolResult(True, {"catalog": kb_catalog(self._kb),
|
|
364
|
+
"count": len(self._kb)},
|
|
365
|
+
"目錄;用 id 取完整步驟,或 query/owasp 過濾")
|
|
366
|
+
if not hits:
|
|
367
|
+
return ToolResult(True, {"catalog": kb_catalog(self._kb)},
|
|
368
|
+
"無符合項;看目錄改用 id")
|
|
369
|
+
return ToolResult(True, {"count": len(hits),
|
|
370
|
+
"playbooks": [d.render() for d in hits[:4]]})
|
|
371
|
+
|
|
372
|
+
def _t_read_tool_output(self, path: str, offset: int = 0,
|
|
373
|
+
limit: int = 6000) -> ToolResult:
|
|
374
|
+
if self.ws is None:
|
|
375
|
+
return ToolResult(False, None, "工作區未啟用(--workspace 關閉)")
|
|
376
|
+
try:
|
|
377
|
+
return ToolResult(True, {"path": path,
|
|
378
|
+
"content": self.ws.read_tool_output(path, offset, limit)})
|
|
379
|
+
except ValueError as e:
|
|
380
|
+
return ToolResult(False, None, str(e))
|
|
381
|
+
|
|
382
|
+
def _t_write_note(self, name: str, content: str) -> ToolResult:
|
|
383
|
+
if self.ws is None:
|
|
384
|
+
return ToolResult(False, None, "工作區未啟用(--workspace 關閉)")
|
|
385
|
+
p = self.ws.write_note(name, str(content)[:20000])
|
|
386
|
+
return ToolResult(True, {"saved": str(p)})
|
|
387
|
+
|
|
388
|
+
def _t_nuclei_cve_probe(self, cve_id: str) -> ToolResult:
|
|
389
|
+
if not re.match(r"^(cve-\d{4}-\d{4,7}|ghsa-[a-z0-9-]+)$", cve_id, re.I):
|
|
390
|
+
return ToolResult(False, None, f"非法 tag: {cve_id!r}(僅 CVE/GHSA ID)")
|
|
391
|
+
if self.docker_bridge is None:
|
|
392
|
+
return ToolResult(False, None, "Docker 不可用 → nuclei 動態嘗試不可行(如實回報,勿臆測結果)")
|
|
393
|
+
self.probe_count += 1
|
|
394
|
+
import asyncio
|
|
395
|
+
job = self.docker_bridge.nuclei_runner(self.target, tags=cve_id.upper())
|
|
396
|
+
asyncio.run(self.docker_bridge.start(job))
|
|
397
|
+
if job.status == "failed":
|
|
398
|
+
return ToolResult(False, {"stderr": (job.stderr or "")[:400]}, "nuclei job failed")
|
|
399
|
+
# ScannerBridge._parse_nuclei_jsonl 回傳 list[dict](每筆為一條 match)
|
|
400
|
+
res = job.result
|
|
401
|
+
matches = res if isinstance(res, list) else (res or {}).get("matches", [])
|
|
402
|
+
brief = [{"name": m.get("info", {}).get("name", "?"),
|
|
403
|
+
"severity": m.get("info", {}).get("severity", "?"),
|
|
404
|
+
"matched": m.get("matched-at", "")} for m in matches[:10]]
|
|
405
|
+
return ToolResult(True, {"tag": cve_id.upper(), "match_count": len(matches),
|
|
406
|
+
"matches": brief,
|
|
407
|
+
"note": "match_count=0 只代表該模板未命中,不等於無漏洞"
|
|
408
|
+
"(模板覆蓋率有限)"})
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
# ---------------------------------------------------------------------------
|
|
412
|
+
# Brain(OpenAI-compatible + function calling)
|
|
413
|
+
# ---------------------------------------------------------------------------
|
|
414
|
+
|
|
415
|
+
class Brain(Protocol):
|
|
416
|
+
def step(self, messages: list[dict]) -> dict:
|
|
417
|
+
"""回傳 {'thought': str, 'tool': str, 'args': dict} 或 {'thought', 'final': str}."""
|
|
418
|
+
...
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
SYSTEM_PROMPT = """You are Tanli, an autonomous red-team agent assessing ONE authorized target.
|
|
422
|
+
Method rules (mandatory):
|
|
423
|
+
1. NEVER trust self-reported versions (headers/generator/release meta are hypotheses).
|
|
424
|
+
Identify the real product/framework, then call cve_lookup at PRODUCT level to get ALL
|
|
425
|
+
published CVEs, and cross-check which versions are actually affected.
|
|
426
|
+
2. Verify dynamically: after matching a plausible CVE, use nuclei_cve_probe (or safe
|
|
427
|
+
http_request probes) to attempt confirmation. A version-range match alone is a
|
|
428
|
+
CANDIDATE, not a finding.
|
|
429
|
+
3. For web targets, run web_config_probe early — it is deterministic evidence.
|
|
430
|
+
4. Before planning attack checks, consult get_playbook (catalog first, then the
|
|
431
|
+
relevant ids). The library encodes standard attack/defense procedures
|
|
432
|
+
(OWASP GenAI LLM playbooks, web methodology) — adapt them to what you
|
|
433
|
+
actually observed; never run a playbook blindly against the wrong product.
|
|
434
|
+
5. Record each real observation via add_finding with concrete evidence. Severity must
|
|
435
|
+
match CVSS/impact. Never fabricate probe results; if a tool is unavailable or a
|
|
436
|
+
probe is blocked, say so honestly.
|
|
437
|
+
6. Stay inside the authorized scope; prefer read-only observation. You have a hard
|
|
438
|
+
probe budget and a token budget.
|
|
439
|
+
7. When you have exhausted reasonable checks (or budget), call finish with a summary.
|
|
440
|
+
""" + GUARD_RULE + """
|
|
441
|
+
Keep working notes via write_note for anything worth carrying to a next session.
|
|
442
|
+
Think step by step; one tool call per message."""
|
|
443
|
+
|
|
444
|
+
|
|
445
|
+
class OpenAIBrain:
|
|
446
|
+
"""OpenAI-compatible tool-calling brain(與 LLMJudge 同配置源)。"""
|
|
447
|
+
|
|
448
|
+
def __init__(self, cfg: dict[str, Any]):
|
|
449
|
+
from .config import Config
|
|
450
|
+
env = Config.resolve_env()
|
|
451
|
+
self.model = env.get("model") or cfg.get("model") or "gpt-4o"
|
|
452
|
+
base_url = env.get("base_url") or cfg.get("base_url")
|
|
453
|
+
api_key = env.get("api_key") or cfg.get("api_key")
|
|
454
|
+
if not api_key and not base_url:
|
|
455
|
+
raise RuntimeError(
|
|
456
|
+
"agent 需要 LLM 端點:設 REDTEAM_JUDGE_BASE_URL / REDTEAM_JUDGE_API_KEY "
|
|
457
|
+
"(或 OPENAI_API_KEY)")
|
|
458
|
+
from openai import OpenAI
|
|
459
|
+
kwargs: dict[str, Any] = {"api_key": api_key or "not-needed"}
|
|
460
|
+
if base_url:
|
|
461
|
+
kwargs["base_url"] = base_url
|
|
462
|
+
self.client = OpenAI(**kwargs)
|
|
463
|
+
self.temperature = cfg.get("agent_temperature", 0.2)
|
|
464
|
+
|
|
465
|
+
def step(self, messages: list[dict]) -> dict:
|
|
466
|
+
resp = self.client.chat.completions.create(
|
|
467
|
+
model=self.model,
|
|
468
|
+
messages=messages, # type: ignore[arg-type]
|
|
469
|
+
tools=[{"type": "function", "function": s} for s in AgentTools.SCHEMAS], # type: ignore[arg-type]
|
|
470
|
+
tool_choice="auto",
|
|
471
|
+
temperature=self.temperature,
|
|
472
|
+
max_tokens=1200,
|
|
473
|
+
)
|
|
474
|
+
msg = resp.choices[0].message
|
|
475
|
+
usage = getattr(resp, "usage", None)
|
|
476
|
+
tokens = (usage.total_tokens if usage else 0)
|
|
477
|
+
if getattr(msg, "tool_calls", None):
|
|
478
|
+
tc = msg.tool_calls[0] # type: ignore[index]
|
|
479
|
+
try:
|
|
480
|
+
args = json.loads(tc.function.arguments or "{}") # type: ignore[union-attr]
|
|
481
|
+
except json.JSONDecodeError:
|
|
482
|
+
args = {}
|
|
483
|
+
return {"thought": msg.content or "", "tool": tc.function.name, # type: ignore[union-attr]
|
|
484
|
+
"args": args, "tokens": tokens, "_msg": msg}
|
|
485
|
+
return {"thought": msg.content or "", "final": msg.content or "(no summary)",
|
|
486
|
+
"tokens": tokens, "_msg": msg}
|
|
487
|
+
|
|
488
|
+
|
|
489
|
+
# ---------------------------------------------------------------------------
|
|
490
|
+
# Loop
|
|
491
|
+
# ---------------------------------------------------------------------------
|
|
492
|
+
|
|
493
|
+
@dataclass
|
|
494
|
+
class AgentFinding:
|
|
495
|
+
title: str
|
|
496
|
+
severity: str
|
|
497
|
+
description: str
|
|
498
|
+
evidence: str
|
|
499
|
+
category: str = "agent"
|
|
500
|
+
cve: str = ""
|
|
501
|
+
confidence: float = 0.6
|
|
502
|
+
triage_score: float = -1.0
|
|
503
|
+
triage_factors: dict = field(default_factory=dict)
|
|
504
|
+
kev: bool = False
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
@dataclass
|
|
508
|
+
class AgentRunResult:
|
|
509
|
+
findings: list[AgentFinding] = field(default_factory=list)
|
|
510
|
+
steps: int = 0
|
|
511
|
+
tokens: int = 0
|
|
512
|
+
probes: int = 0
|
|
513
|
+
finished: bool = False
|
|
514
|
+
summary: str = ""
|
|
515
|
+
transcript: list[dict] = field(default_factory=list)
|
|
516
|
+
workspace: str = ""
|
|
517
|
+
engagement_package: str = ""
|
|
518
|
+
|
|
519
|
+
|
|
520
|
+
def run_agent(tools: AgentTools, brain, *, goal: str, max_steps: int = 30,
|
|
521
|
+
token_budget: int = 200_000, console=None,
|
|
522
|
+
engagement_brief: str = "") -> AgentRunResult:
|
|
523
|
+
"""自主 tool-loop:brain 決定每一步,工具層圍籬兜底。
|
|
524
|
+
|
|
525
|
+
engagement_brief: RoE + OPPLAN + 跨會話記憶 的合成文本(借鑑
|
|
526
|
+
Decepticon「行動前先注入紀律」;空字串 = 純預設行為)。
|
|
527
|
+
"""
|
|
528
|
+
result = AgentRunResult()
|
|
529
|
+
if tools.ws is not None:
|
|
530
|
+
result.workspace = str(tools.ws.root)
|
|
531
|
+
opening = (f"Authorized target: {tools.target}\nGoal: {goal}\n"
|
|
532
|
+
f"Hard limits: max {max_steps} steps, {tools.max_probes} HTTP probes. ")
|
|
533
|
+
if engagement_brief:
|
|
534
|
+
opening += ("\nEngagement discipline (read before acting):\n" + engagement_brief
|
|
535
|
+
+ "\nBegin with fingerprinting, then execute the OPPLAN phases.")
|
|
536
|
+
else:
|
|
537
|
+
opening += "Begin with fingerprinting, then plan your checks."
|
|
538
|
+
messages: list[dict] = [
|
|
539
|
+
{"role": "system", "content": SYSTEM_PROMPT},
|
|
540
|
+
{"role": "user", "content": opening},
|
|
541
|
+
]
|
|
542
|
+
for step_i in range(max_steps):
|
|
543
|
+
try:
|
|
544
|
+
out = brain.step(messages)
|
|
545
|
+
except Exception as e: # noqa: BLE001
|
|
546
|
+
result.summary = f"brain 中途中斷({e.__class__.__name__}: {e})"
|
|
547
|
+
break
|
|
548
|
+
result.steps += 1
|
|
549
|
+
result.tokens += out.get("tokens", 0)
|
|
550
|
+
thought = out.get("thought", "")
|
|
551
|
+
if console:
|
|
552
|
+
console.print(f" [magenta]◈ step {step_i+1}[/] {thought[:180]}")
|
|
553
|
+
|
|
554
|
+
if "final" in out:
|
|
555
|
+
result.finished = True
|
|
556
|
+
result.summary = out["final"]
|
|
557
|
+
break
|
|
558
|
+
|
|
559
|
+
tool, args = out.get("tool", ""), out.get("args", {})
|
|
560
|
+
if tool == "add_finding":
|
|
561
|
+
cve_arg = str(args.get("cve", ""))
|
|
562
|
+
# triage(借鑑 CypherFix):固定公式風險分,確定性可稽核
|
|
563
|
+
kev_hit = False
|
|
564
|
+
epss_val: float | None = None
|
|
565
|
+
try:
|
|
566
|
+
if re.match(r"^CVE-\d{4}-\d{4,7}$", cve_arg, re.I):
|
|
567
|
+
from .cve_lookup import kev_set as _kev_set
|
|
568
|
+
kev_map, _st = _kev_set()
|
|
569
|
+
kev_hit = cve_arg.upper() in kev_map
|
|
570
|
+
except Exception: # noqa: BLE001 — KEV 不可用不阻斷記錄
|
|
571
|
+
pass
|
|
572
|
+
tr = triage_score(severity=str(args.get("severity", "info")),
|
|
573
|
+
confidence=float(args.get("confidence", 0.6)),
|
|
574
|
+
cve_epss=epss_val, kev=kev_hit,
|
|
575
|
+
reachability=str(args.get("reachability", "direct")))
|
|
576
|
+
f = AgentFinding(
|
|
577
|
+
title=str(args.get("title", ""))[:200],
|
|
578
|
+
severity=str(args.get("severity", "info")).lower(),
|
|
579
|
+
description=str(args.get("description", ""))[:2000],
|
|
580
|
+
evidence=str(args.get("evidence", ""))[:2000],
|
|
581
|
+
category=str(args.get("category", "agent")),
|
|
582
|
+
cve=cve_arg,
|
|
583
|
+
confidence=float(args.get("confidence", 0.6)),
|
|
584
|
+
triage_score=tr.score, triage_factors=tr.factors, kev=kev_hit,
|
|
585
|
+
)
|
|
586
|
+
result.findings.append(f)
|
|
587
|
+
obs = ToolResult(True, {"recorded": f.title,
|
|
588
|
+
"triage": tr.as_dict()})
|
|
589
|
+
if console:
|
|
590
|
+
console.print(f" [green]✎ finding: {f.title} ({f.severity}"
|
|
591
|
+
f" | triage {tr.score})[/]")
|
|
592
|
+
elif tool == "finish":
|
|
593
|
+
result.finished = True
|
|
594
|
+
s = str(args.get("summary", "")).strip() or thought.strip()
|
|
595
|
+
if not s:
|
|
596
|
+
# 模型沒給總結:以實際狀態合成事實性摘要(不編造內容)
|
|
597
|
+
s = (f"評估結束:{len(result.findings)} 項發現已記錄,"
|
|
598
|
+
f"{result.probes} 次探測 / {result.steps} 步。")
|
|
599
|
+
result.summary = s
|
|
600
|
+
break
|
|
601
|
+
else:
|
|
602
|
+
obs = tools.call(tool, args)
|
|
603
|
+
|
|
604
|
+
result.transcript.append({"step": step_i + 1, "thought": thought[:500],
|
|
605
|
+
"tool": tool, "args": args,
|
|
606
|
+
"ok": obs.ok, "note": obs.note,
|
|
607
|
+
"data_preview": str(obs.data)[:800]})
|
|
608
|
+
messages.append({"role": "assistant", "content": thought or f"[call {tool}]"})
|
|
609
|
+
messages.append({"role": "user", "content": (
|
|
610
|
+
f"[tool {tool} -> {'OK' if obs.ok else 'FAIL'}] "
|
|
611
|
+
f"{obs.note + ' | ' if obs.note else ''}"
|
|
612
|
+
f"{json.dumps(obs.data, ensure_ascii=False, default=str)[:4000] if obs.data is not None else ''}"
|
|
613
|
+
f"\n(remaining: {max_steps - step_i - 1} steps, "
|
|
614
|
+
f"{tools.max_probes - tools.probe_count} probes, "
|
|
615
|
+
f"~{max(token_budget - result.tokens, 0)} tokens)")})
|
|
616
|
+
|
|
617
|
+
if result.tokens > token_budget:
|
|
618
|
+
result.summary = result.summary or "token budget 耗盡,自主循環停止"
|
|
619
|
+
break
|
|
620
|
+
else:
|
|
621
|
+
result.summary = result.summary or f"達成 max_steps={max_steps} 上限,未明確 finish"
|
|
622
|
+
|
|
623
|
+
result.probes = tools.probe_count
|
|
624
|
+
|
|
625
|
+
# ---- 收尾:EPSS 回填 triage(批次一次)+ 跨會話記憶累加 ----
|
|
626
|
+
cve_findings = [f for f in result.findings
|
|
627
|
+
if re.match(r"^CVE-\d{4}-\d{4,7}$", f.cve or "", re.I)]
|
|
628
|
+
if cve_findings:
|
|
629
|
+
try:
|
|
630
|
+
from .cve_lookup import epss_scores
|
|
631
|
+
scores = epss_scores([f.cve for f in cve_findings])
|
|
632
|
+
for f in cve_findings:
|
|
633
|
+
e = scores.get(f.cve.upper())
|
|
634
|
+
if e:
|
|
635
|
+
tr = triage_score(severity=f.severity, confidence=f.confidence,
|
|
636
|
+
cve_epss=e["score"], kev=f.kev)
|
|
637
|
+
f.triage_score, f.triage_factors = tr.score, tr.factors
|
|
638
|
+
except Exception: # noqa: BLE001 — EPSS 缺援保留 severity 預設分
|
|
639
|
+
pass
|
|
640
|
+
if tools.ws is not None:
|
|
641
|
+
fdicts = [{"title": f.title, "severity": f.severity, "cve": f.cve,
|
|
642
|
+
"triage_score": f.triage_score} for f in result.findings]
|
|
643
|
+
pdicts = [p.to_dict() for p in tools.probe_log]
|
|
644
|
+
lessons = [t.get("note", "") for t in result.transcript
|
|
645
|
+
if not t.get("ok") and t.get("note")]
|
|
646
|
+
tools.ws.append_session(fdicts, pdicts,
|
|
647
|
+
[l for l in lessons if l][:10],
|
|
648
|
+
result.summary or "")
|
|
649
|
+
return result
|