tanli 0.0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (65) hide show
  1. redteam/__init__.py +3 -0
  2. redteam/agent.py +649 -0
  3. redteam/auth.py +196 -0
  4. redteam/cli.py +1043 -0
  5. redteam/config.py +103 -0
  6. redteam/cve_lookup.py +461 -0
  7. redteam/cvss.py +233 -0
  8. redteam/engagement.py +170 -0
  9. redteam/findings.py +432 -0
  10. redteam/guardtext.py +73 -0
  11. redteam/interactor.py +114 -0
  12. redteam/judge.py +164 -0
  13. redteam/knowledge.py +239 -0
  14. redteam/planner.py +87 -0
  15. redteam/playbook.py +411 -0
  16. redteam/playbooks/llm/playbook_1.yaml +75 -0
  17. redteam/playbooks/llm/playbook_10.yaml +113 -0
  18. redteam/playbooks/llm/playbook_11.yaml +103 -0
  19. redteam/playbooks/llm/playbook_2.yaml +58 -0
  20. redteam/playbooks/llm/playbook_3.yaml +59 -0
  21. redteam/playbooks/llm/playbook_4.yaml +43 -0
  22. redteam/playbooks/llm/playbook_5.yaml +37 -0
  23. redteam/playbooks/llm/playbook_6.yaml +101 -0
  24. redteam/playbooks/llm/playbook_7.yaml +100 -0
  25. redteam/playbooks/llm/playbook_8.yaml +115 -0
  26. redteam/playbooks/llm/playbook_9.yaml +102 -0
  27. redteam/playbooks/web/playbook_10.yaml +21 -0
  28. redteam/playbooks/web/playbook_11.yaml +21 -0
  29. redteam/playbooks/web/playbook_12.yaml +22 -0
  30. redteam/playbooks/web/playbook_13.yaml +21 -0
  31. redteam/playbooks/web/playbook_14.yaml +20 -0
  32. redteam/playbooks/web/playbook_15.yaml +23 -0
  33. redteam/playbooks/web/playbook_16.yaml +21 -0
  34. redteam/playbooks/web/playbook_17.yaml +23 -0
  35. redteam/playbooks/web/playbook_18.yaml +25 -0
  36. redteam/playbooks/web/playbook_19.yaml +29 -0
  37. redteam/playbooks/web/playbook_20.yaml +25 -0
  38. redteam/playbooks/web/playbook_21.yaml +30 -0
  39. redteam/playbooks/web/playbook_22.yaml +24 -0
  40. redteam/playbooks/web/playbook_23.yaml +28 -0
  41. redteam/playbooks/web/playbook_24.yaml +29 -0
  42. redteam/playbooks/web/playbook_25.yaml +27 -0
  43. redteam/playbooks/web/playbook_26.yaml +29 -0
  44. redteam/playbooks/web/playbook_27.yaml +32 -0
  45. redteam/playbooks/web/playbook_28.yaml +29 -0
  46. redteam/playbooks/web/playbook_29.yaml +31 -0
  47. redteam/playbooks/web/playbook_30.yaml +29 -0
  48. redteam/playbooks/web/playbook_6.yaml +29 -0
  49. redteam/playbooks/web/playbook_7.yaml +23 -0
  50. redteam/playbooks/web/playbook_8.yaml +24 -0
  51. redteam/playbooks/web/playbook_9.yaml +22 -0
  52. redteam/report.py +292 -0
  53. redteam/sandbox.py +101 -0
  54. redteam/scanners.py +435 -0
  55. redteam/target_lab.py +287 -0
  56. redteam/triage.py +86 -0
  57. redteam/version_watch.py +218 -0
  58. redteam/web_config.py +289 -0
  59. redteam/workspace.py +167 -0
  60. tanli-0.0.1.dist-info/METADATA +235 -0
  61. tanli-0.0.1.dist-info/RECORD +65 -0
  62. tanli-0.0.1.dist-info/WHEEL +5 -0
  63. tanli-0.0.1.dist-info/entry_points.txt +3 -0
  64. tanli-0.0.1.dist-info/licenses/LICENSE +201 -0
  65. tanli-0.0.1.dist-info/top_level.txt +1 -0
redteam/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """RedTeam Agent - Autonomous Red Team testing agent."""
2
+
3
+ __version__ = "0.0.1"
redteam/agent.py ADDED
@@ -0,0 +1,649 @@
1
+ """Autonomous agent loop — OpenCode/Hermes 級 tool-loop 自主性,安全圍籬內執行。
2
+
3
+ 設計(spec §6 精神延伸,用戶 2026-09-19 需求):
4
+ - 靜態 6 節點 plan 之外,新增真正的 LLM tool-loop:模型自己決定下一步
5
+ (探什麼、查哪個 CVE、動態怎麼試),直到達成目標或預算耗盡。
6
+ - 每個工具都是「圍籬內的原語」:ScopeGuard/read-only/token budget 在工具
7
+ 層強制,模型無法繞過(與 hermes 的 approval 分層同理:能力給足,權限收緊)。
8
+ - 不信任指紋版本原則寫進 system prompt:發現套件/框架 → cve_lookup 查全部
9
+ CVE → 以被動指紋+動態安全探測交叉核實 → 才有資格宣稱漏洞。
10
+
11
+ Brain 可注入(FakeBrain 測試);生產用 OpenAI-compatible endpoint
12
+ (與 LLMJudge 同源配置 REDTEAM_JUDGE_*)。
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import re
19
+ from dataclasses import dataclass, field
20
+ from typing import Any, Protocol
21
+
22
+ from .cve_lookup import enrich as cve_enrich, lookup as cve_lookup_backend
23
+ from .guardtext import SYSTEM_RULE as GUARD_RULE, guard_observation
24
+ from .knowledge import (catalog as kb_catalog, get_by_id as kb_get,
25
+ load_library as kb_load, load_user_skills,
26
+ search as kb_search)
27
+ from .interactor import RedTeamHTTP
28
+ from .triage import triage as triage_score, sort_findings
29
+ from .version_watch import detect_releases, watch as version_watch
30
+ from .workspace import ProbeLog, Workspace
31
+
32
+
33
+ # ---------------------------------------------------------------------------
34
+ # Tool 層(每個工具 = 圍籬內原語)
35
+ # ---------------------------------------------------------------------------
36
+
37
+ @dataclass
38
+ class ToolResult:
39
+ ok: bool
40
+ data: Any
41
+ note: str = ""
42
+
43
+
44
+ class AgentTools:
45
+ """綁定單一 target + ScopeGuard/read-only HTTP 客戶端的工具集。
46
+
47
+ add_finding / finish 由 loop 內部處理;這裡只實作探測/查詢類工具。
48
+ """
49
+
50
+ def __init__(self, http: RedTeamHTTP, target: str, *, docker_bridge=None,
51
+ workspace: Workspace | None = None):
52
+ self.http = http
53
+ self.target = target
54
+ self.docker_bridge = docker_bridge
55
+ self.probe_count = 0
56
+ self.max_probes = 60 # 硬性總探測量,防失控轟炸
57
+ self.ws = workspace # None = 不啟用工作區(向後兼容)
58
+ self._kb = None # get_playbook 知識庫 lazy load
59
+ self._get_cache: dict[str, ToolResult] = {} # 安全方法結果快取(省預算)
60
+ self.probe_log: list[ProbeLog] = [] # 跨會話記憶的探測軌跡(含失敗)
61
+
62
+ # ---- schema(提供給 brain) ----
63
+ SCHEMAS: list[dict] = [
64
+ {
65
+ "name": "http_request",
66
+ "description": ("Send an HTTP request to a URL inside the authorized scope. "
67
+ "Returns status, selected headers, and truncated body. "
68
+ "In read-only mode non-safe methods are physically blocked."),
69
+ "parameters": {
70
+ "type": "object",
71
+ "properties": {
72
+ "method": {"type": "string", "enum": ["GET", "HEAD", "OPTIONS", "POST", "PUT", "DELETE"]},
73
+ "url": {"type": "string", "description": "Must stay within authorized scope"},
74
+ "headers": {"type": "object", "additionalProperties": {"type": "string"}},
75
+ "body": {"type": "string"},
76
+ },
77
+ "required": ["method", "url"],
78
+ },
79
+ },
80
+ {
81
+ "name": "fingerprint",
82
+ "description": ("Passively fingerprint the target: fetch it, extract server "
83
+ "headers, generator meta, and any <pkg>@<ver> release "
84
+ "declarations (self-reported versions — treat as hypothesis, "
85
+ "never as truth)."),
86
+ "parameters": {"type": "object", "properties": {}, "required": []},
87
+ },
88
+ {
89
+ "name": "cve_lookup",
90
+ "description": ("Query public CVE sources (GHSA + OSV + NVD) with EPSS "
91
+ "likelihood + CISA KEV enrichment. Give either "
92
+ "cve_id (CVE-YYYY-NNNNN) for exact lookup, or product "
93
+ "(+ecosystem/version optional) for product-level listing of "
94
+ "ALL published CVEs — do not trust the target's claimed "
95
+ "version, verify which versions are affected and probe. "
96
+ "kev=true means actively exploited in the wild: verify first."),
97
+ "parameters": {
98
+ "type": "object",
99
+ "properties": {
100
+ "cve_id": {"type": "string"},
101
+ "product": {"type": "string"},
102
+ "ecosystem": {"type": "string",
103
+ "description": "npm|pip|composer|maven|go|cargo|rubygems..."},
104
+ "version": {"type": "string"},
105
+ "likelihood": {"type": "boolean",
106
+ "description": "set false to skip EPSS/KEV enrichment (faster)"},
107
+ },
108
+ "required": [],
109
+ },
110
+ },
111
+ {
112
+ "name": "version_watch",
113
+ "description": ("Compare a specific pkg@version against GHSA for applicable "
114
+ "unpatched CVEs (fast, version-range matched). Only meaningful "
115
+ "for framework/ecosystem packages."),
116
+ "parameters": {
117
+ "type": "object",
118
+ "properties": {"product": {"type": "string"}, "ecosystem": {"type": "string"},
119
+ "version": {"type": "string"}},
120
+ "required": ["product", "version"],
121
+ },
122
+ },
123
+ {
124
+ "name": "web_config_probe",
125
+ "description": ("Run the deterministic web-configuration probe suite against "
126
+ "the target (security headers, cookie flags, exposed paths, "
127
+ "method policies). Pure read-only, no Docker. Returns "
128
+ "confirmed config weaknesses with OWASP mapping."),
129
+ "parameters": {"type": "object", "properties": {}, "required": []},
130
+ },
131
+ {
132
+ "name": "nuclei_cve_probe",
133
+ "description": ("Attempt a dynamic check of a specific CVE by running nuclei "
134
+ "with template tag = CVE id, inside the Docker sandbox. "
135
+ "Requires Docker; respects read-only & scope. This is the "
136
+ "'動態嘗試' path — match against the REAL detected product/version."),
137
+ "parameters": {
138
+ "type": "object",
139
+ "properties": {"cve_id": {"type": "string"}},
140
+ "required": ["cve_id"],
141
+ },
142
+ },
143
+ {
144
+ "name": "get_playbook",
145
+ "description": ("Query the playbook library (attack theory / defense "
146
+ "checklist templates: OWASP GenAI LLM playbooks + web "
147
+ "methodology + user-injected markdown skills from "
148
+ "~/.tanli/skills). Call with no args to get a compact catalog; "
149
+ "then fetch full steps by id, or search by keyword/owasp. "
150
+ "Use playbooks as REFERENCE workflows for planning — they "
151
+ "encode standard attack/defense procedures."),
152
+ "parameters": {
153
+ "type": "object",
154
+ "properties": {
155
+ "id": {"type": "string", "description": "Exact playbook id, e.g. llm-001"},
156
+ "query": {"type": "string", "description": "Keyword search across all playbooks"},
157
+ "owasp": {"type": "string", "description": "Filter by OWASP code, e.g. LLM01"},
158
+ "target_type": {"type": "string", "description": "Filter: llm_app | web_service"},
159
+ },
160
+ "required": [],
161
+ },
162
+ },
163
+ {
164
+ "name": "read_tool_output",
165
+ "description": ("Read back an offloaded large tool output by its file path "
166
+ "(only paths under this engagement's tool-outputs/ are "
167
+ "allowed). Use when a previous result shows [OFFLOADED ...]."),
168
+ "parameters": {
169
+ "type": "object",
170
+ "properties": {
171
+ "path": {"type": "string"},
172
+ "offset": {"type": "integer", "description": "char offset (default 0)"},
173
+ "limit": {"type": "integer", "description": "max chars (default 6000)"},
174
+ },
175
+ "required": ["path"],
176
+ },
177
+ },
178
+ {
179
+ "name": "write_note",
180
+ "description": ("Persist a working note in the engagement workspace "
181
+ "(notes/<name>.md). Survives across sessions — future "
182
+ "runs on this target load findings/lessons automatically. "
183
+ "Good for: hypotheses to test later, dead-end reasons, "
184
+ "attack-surface maps."),
185
+ "parameters": {
186
+ "type": "object",
187
+ "properties": {
188
+ "name": {"type": "string", "description": "filename-safe name"},
189
+ "content": {"type": "string"},
190
+ },
191
+ "required": ["name", "content"],
192
+ },
193
+ },
194
+ {
195
+ "name": "add_finding",
196
+ "description": ("Record a confirmed-or-candidate finding. Cite concrete "
197
+ "evidence (request results / CVE IDs / probe output). "
198
+ "Never claim a finding you did not observe."),
199
+ "parameters": {
200
+ "type": "object",
201
+ "properties": {
202
+ "title": {"type": "string"},
203
+ "severity": {"type": "string", "enum": ["critical", "high", "medium", "low", "info"]},
204
+ "category": {"type": "string"},
205
+ "description": {"type": "string"},
206
+ "evidence": {"type": "string"},
207
+ "cve": {"type": "string"},
208
+ "confidence": {"type": "number"},
209
+ "reachability": {"type": "string", "enum": [
210
+ "direct", "low_priv", "auth_required",
211
+ "user_interaction", "internal_only"],
212
+ "description": "how reachable the flaw is for an attacker"},
213
+ },
214
+ "required": ["title", "severity", "description", "evidence"],
215
+ },
216
+ },
217
+ {
218
+ "name": "finish",
219
+ "description": "End the assessment with a short summary of what was done and found.",
220
+ "parameters": {
221
+ "type": "object",
222
+ "properties": {"summary": {"type": "string"}},
223
+ "required": ["summary"],
224
+ },
225
+ },
226
+ ]
227
+
228
+ # ---- dispatch ----
229
+ def call(self, name: str, args: dict) -> ToolResult:
230
+ if self.probe_count >= self.max_probes and name == "http_request":
231
+ return ToolResult(False, None, f"probe budget exhausted ({self.max_probes})")
232
+ fn = getattr(self, f"_t_{name}", None)
233
+ if fn is None:
234
+ return ToolResult(False, None, f"unknown tool: {name}")
235
+ try:
236
+ return fn(**args)
237
+ except Exception as e: # noqa: BLE001
238
+ return ToolResult(False, None, f"tool error ({e.__class__.__name__}): {e}")
239
+
240
+ def _t_http_request(self, method: str, url: str, headers: dict | None = None,
241
+ body: str | None = None) -> ToolResult:
242
+ # 安全方法(GET/HEAD/OPTIONS)同請求快取:省探測預算與 token,結果一致
243
+ m = method.upper()
244
+ if m in ("GET", "HEAD", "OPTIONS") and body is None and not headers:
245
+ hit = self._get_cache.get(f"{m} {url}")
246
+ if hit is not None:
247
+ return ToolResult(True, dict(hit.data) if isinstance(hit.data, dict) else hit.data,
248
+ "cached(先前已探過同一請求,結果相同,未耗預算)")
249
+ self.probe_count += 1
250
+ rec = self.http.request(method.upper(), url, headers=headers,
251
+ data=body.encode() if body else None)
252
+ if rec.blocked_by_scope:
253
+ self.probe_log.append(ProbeLog(m, url, "blocked-scope", ok=False))
254
+ return ToolResult(False, {"blocked": "scope"}, "BLOCKED by ScopeGuard — outside authorized scope")
255
+ if rec.blocked_by_readonly:
256
+ self.probe_log.append(ProbeLog(m, url, "blocked-readonly", ok=False))
257
+ return ToolResult(False, {"blocked": "read_only"},
258
+ "BLOCKED by read-only fuse — non-safe method physically blocked")
259
+ self.probe_log.append(ProbeLog(m, url, rec.status, ok=(rec.status or 0) < 400))
260
+ text = (rec.body or b"").decode("utf-8", "replace")
261
+ body_out = text[:3500]
262
+ if self.ws is not None and len(text) > 3500:
263
+ # 大輸出卸載(借鑑 RedAmon auto-offload):完整內容落盤,LLM 收 stub
264
+ body_out = self.ws.offload(f"http{rec.status}", text)
265
+ # 注入防護:目標內容包定界符 + 啟發式標記(借鑑 Decepticon)
266
+ body_out, _hits = guard_observation(body_out, source=url)
267
+ out = ToolResult(True, {
268
+ "status": rec.status,
269
+ "headers": {k: v for k, v in (rec.response_headers or {}).items()
270
+ if k.lower() in ("server", "x-powered-by", "content-type", "generator", "via")},
271
+ "body": body_out,
272
+ "body_len": len(text),
273
+ })
274
+ if m in ("GET", "HEAD", "OPTIONS") and body is None and not headers \
275
+ and not rec.blocked_by_scope and not rec.blocked_by_readonly:
276
+ self._get_cache[f"{m} {url}"] = out
277
+ return out
278
+
279
+ def _t_fingerprint(self) -> ToolResult:
280
+ self.probe_count += 1
281
+ rec = self.http.request("GET", self.target)
282
+ if rec.blocked_by_scope or rec.blocked_by_readonly:
283
+ self.probe_log.append(ProbeLog("GET", self.target, "blocked", ok=False))
284
+ return ToolResult(False, None, "target blocked by scope/read-only")
285
+ self.probe_log.append(ProbeLog("GET", self.target, rec.status,
286
+ ok=(rec.status or 0) < 400))
287
+ html = (rec.body or b"").decode("utf-8", "replace")
288
+ releases = detect_releases(html)
289
+ hdrs = {k: v for k, v in (rec.response_headers or {}).items()
290
+ if k.lower() in ("server", "x-powered-by", "generator", "via", "x-aspnet-version")}
291
+ meta = re.findall(r'<meta[^>]+name=["\']generator["\'][^>]+content=["\']([^"\']+)',
292
+ html, re.I)
293
+ head = html[:2500]
294
+ if self.ws is not None and len(html) > 2500:
295
+ head = self.ws.offload("fingerprint", html)
296
+ head, _ = guard_observation(head, source=self.target)
297
+ return ToolResult(True, {"status": rec.status, "headers": hdrs,
298
+ "generator_meta": meta, "release_decls": releases,
299
+ "body": head})
300
+
301
+ def _t_cve_lookup(self, cve_id: str | None = None, product: str | None = None,
302
+ ecosystem: str | None = None, version: str | None = None,
303
+ likelihood: bool = True) -> ToolResult:
304
+ r = cve_lookup_backend(cve_id=cve_id, product=product, ecosystem=ecosystem, version=version)
305
+ if not r.ok:
306
+ return ToolResult(False, {"error": r.error}, r.error)
307
+ intel = {}
308
+ if likelihood and r.advisories:
309
+ # EPSS + CISA KEV 情報加權(失敗如實標 status,不編造分數)
310
+ try:
311
+ intel = cve_enrich(r.advisories)
312
+ except Exception as e: # noqa: BLE001
313
+ intel = {"kev_status": f"error({e.__class__.__name__})"}
314
+ advs = [{"cve": a.cve, "src": a.source, "sev": a.severity, "product": a.product,
315
+ "range": a.vulnerable_range, "patched": a.first_patched,
316
+ "pub": a.published, "epss": a.epss, "kev": a.kev,
317
+ "summary": a.summary[:160]} for a in r.advisories]
318
+ # 高 likelihood 排前面(EPSS desc → KEV → severity),幫 agent 聚焦
319
+ advs.sort(key=lambda a: (-(a.get("epss") or 0), not a.get("kev"),
320
+ a["sev"] != "critical"))
321
+ return ToolResult(True, {"sources": r.sources, "count": len(advs),
322
+ "likelihood_intel": intel or "未啟用或不可用",
323
+ "latest_advisory_date": r.latest_advisory_date,
324
+ "advisories": advs[:40]},
325
+ "查無 CVE ≠ 無漏洞(注意資料源滯後;見 latest_advisory_date);"
326
+ "kev=true 的 CVE 已被真實利用,優先動態驗證")
327
+
328
+ def _t_version_watch(self, product: str, version: str, ecosystem: str = "npm") -> ToolResult:
329
+ wr = version_watch(product, ecosystem, version)
330
+ if wr.error:
331
+ return ToolResult(False, {"error": wr.error}, wr.error)
332
+ return ToolResult(True, {"product": wr.product, "version": wr.current_version,
333
+ "applicable_cves": [{"cve": m.cve, "sev": m.severity,
334
+ "patched": m.first_patched,
335
+ "range": m.vulnerable_range}
336
+ for m in wr.matches],
337
+ "latest_advisory_date": wr.latest_advisory_date})
338
+
339
+ def _t_web_config_probe(self) -> ToolResult:
340
+ from .web_config import WebConfigProbe
341
+ self.probe_count += 1
342
+ probes = WebConfigProbe(http=self.http).run(self.target)
343
+ hits = [{"name": p.name, "owasp": getattr(p, "owasp", ""),
344
+ "evidence": str(p.evidence)[:300]} for p in probes if p.passed]
345
+ return ToolResult(True, {"checked": len(probes), "hit_count": len(hits),
346
+ "hits": hits[:15]},
347
+ "deterministic 證據,可直接 add_finding")
348
+
349
+ def _t_get_playbook(self, id: str | None = None, query: str | None = None,
350
+ owasp: str | None = None,
351
+ target_type: str | None = None) -> ToolResult:
352
+ if self._kb is None:
353
+ # 內建 playbook + 使用者注入的 markdown 技能(~/.tanli/skills)
354
+ self._kb = kb_load() + load_user_skills()
355
+ if id:
356
+ doc = kb_get(self._kb, id)
357
+ if doc is None:
358
+ return ToolResult(False, {"catalog": kb_catalog(self._kb)},
359
+ f"playbook id 不存在:{id}(見 catalog 選一個)")
360
+ return ToolResult(True, {"playbook": doc.render(), "source": doc.source})
361
+ hits = kb_search(self._kb, query=query, owasp=owasp, target_type=target_type)
362
+ if not (query or owasp or target_type):
363
+ return ToolResult(True, {"catalog": kb_catalog(self._kb),
364
+ "count": len(self._kb)},
365
+ "目錄;用 id 取完整步驟,或 query/owasp 過濾")
366
+ if not hits:
367
+ return ToolResult(True, {"catalog": kb_catalog(self._kb)},
368
+ "無符合項;看目錄改用 id")
369
+ return ToolResult(True, {"count": len(hits),
370
+ "playbooks": [d.render() for d in hits[:4]]})
371
+
372
+ def _t_read_tool_output(self, path: str, offset: int = 0,
373
+ limit: int = 6000) -> ToolResult:
374
+ if self.ws is None:
375
+ return ToolResult(False, None, "工作區未啟用(--workspace 關閉)")
376
+ try:
377
+ return ToolResult(True, {"path": path,
378
+ "content": self.ws.read_tool_output(path, offset, limit)})
379
+ except ValueError as e:
380
+ return ToolResult(False, None, str(e))
381
+
382
+ def _t_write_note(self, name: str, content: str) -> ToolResult:
383
+ if self.ws is None:
384
+ return ToolResult(False, None, "工作區未啟用(--workspace 關閉)")
385
+ p = self.ws.write_note(name, str(content)[:20000])
386
+ return ToolResult(True, {"saved": str(p)})
387
+
388
+ def _t_nuclei_cve_probe(self, cve_id: str) -> ToolResult:
389
+ if not re.match(r"^(cve-\d{4}-\d{4,7}|ghsa-[a-z0-9-]+)$", cve_id, re.I):
390
+ return ToolResult(False, None, f"非法 tag: {cve_id!r}(僅 CVE/GHSA ID)")
391
+ if self.docker_bridge is None:
392
+ return ToolResult(False, None, "Docker 不可用 → nuclei 動態嘗試不可行(如實回報,勿臆測結果)")
393
+ self.probe_count += 1
394
+ import asyncio
395
+ job = self.docker_bridge.nuclei_runner(self.target, tags=cve_id.upper())
396
+ asyncio.run(self.docker_bridge.start(job))
397
+ if job.status == "failed":
398
+ return ToolResult(False, {"stderr": (job.stderr or "")[:400]}, "nuclei job failed")
399
+ # ScannerBridge._parse_nuclei_jsonl 回傳 list[dict](每筆為一條 match)
400
+ res = job.result
401
+ matches = res if isinstance(res, list) else (res or {}).get("matches", [])
402
+ brief = [{"name": m.get("info", {}).get("name", "?"),
403
+ "severity": m.get("info", {}).get("severity", "?"),
404
+ "matched": m.get("matched-at", "")} for m in matches[:10]]
405
+ return ToolResult(True, {"tag": cve_id.upper(), "match_count": len(matches),
406
+ "matches": brief,
407
+ "note": "match_count=0 只代表該模板未命中,不等於無漏洞"
408
+ "(模板覆蓋率有限)"})
409
+
410
+
411
+ # ---------------------------------------------------------------------------
412
+ # Brain(OpenAI-compatible + function calling)
413
+ # ---------------------------------------------------------------------------
414
+
415
+ class Brain(Protocol):
416
+ def step(self, messages: list[dict]) -> dict:
417
+ """回傳 {'thought': str, 'tool': str, 'args': dict} 或 {'thought', 'final': str}."""
418
+ ...
419
+
420
+
421
+ SYSTEM_PROMPT = """You are Tanli, an autonomous red-team agent assessing ONE authorized target.
422
+ Method rules (mandatory):
423
+ 1. NEVER trust self-reported versions (headers/generator/release meta are hypotheses).
424
+ Identify the real product/framework, then call cve_lookup at PRODUCT level to get ALL
425
+ published CVEs, and cross-check which versions are actually affected.
426
+ 2. Verify dynamically: after matching a plausible CVE, use nuclei_cve_probe (or safe
427
+ http_request probes) to attempt confirmation. A version-range match alone is a
428
+ CANDIDATE, not a finding.
429
+ 3. For web targets, run web_config_probe early — it is deterministic evidence.
430
+ 4. Before planning attack checks, consult get_playbook (catalog first, then the
431
+ relevant ids). The library encodes standard attack/defense procedures
432
+ (OWASP GenAI LLM playbooks, web methodology) — adapt them to what you
433
+ actually observed; never run a playbook blindly against the wrong product.
434
+ 5. Record each real observation via add_finding with concrete evidence. Severity must
435
+ match CVSS/impact. Never fabricate probe results; if a tool is unavailable or a
436
+ probe is blocked, say so honestly.
437
+ 6. Stay inside the authorized scope; prefer read-only observation. You have a hard
438
+ probe budget and a token budget.
439
+ 7. When you have exhausted reasonable checks (or budget), call finish with a summary.
440
+ """ + GUARD_RULE + """
441
+ Keep working notes via write_note for anything worth carrying to a next session.
442
+ Think step by step; one tool call per message."""
443
+
444
+
445
+ class OpenAIBrain:
446
+ """OpenAI-compatible tool-calling brain(與 LLMJudge 同配置源)。"""
447
+
448
+ def __init__(self, cfg: dict[str, Any]):
449
+ from .config import Config
450
+ env = Config.resolve_env()
451
+ self.model = env.get("model") or cfg.get("model") or "gpt-4o"
452
+ base_url = env.get("base_url") or cfg.get("base_url")
453
+ api_key = env.get("api_key") or cfg.get("api_key")
454
+ if not api_key and not base_url:
455
+ raise RuntimeError(
456
+ "agent 需要 LLM 端點:設 REDTEAM_JUDGE_BASE_URL / REDTEAM_JUDGE_API_KEY "
457
+ "(或 OPENAI_API_KEY)")
458
+ from openai import OpenAI
459
+ kwargs: dict[str, Any] = {"api_key": api_key or "not-needed"}
460
+ if base_url:
461
+ kwargs["base_url"] = base_url
462
+ self.client = OpenAI(**kwargs)
463
+ self.temperature = cfg.get("agent_temperature", 0.2)
464
+
465
+ def step(self, messages: list[dict]) -> dict:
466
+ resp = self.client.chat.completions.create(
467
+ model=self.model,
468
+ messages=messages, # type: ignore[arg-type]
469
+ tools=[{"type": "function", "function": s} for s in AgentTools.SCHEMAS], # type: ignore[arg-type]
470
+ tool_choice="auto",
471
+ temperature=self.temperature,
472
+ max_tokens=1200,
473
+ )
474
+ msg = resp.choices[0].message
475
+ usage = getattr(resp, "usage", None)
476
+ tokens = (usage.total_tokens if usage else 0)
477
+ if getattr(msg, "tool_calls", None):
478
+ tc = msg.tool_calls[0] # type: ignore[index]
479
+ try:
480
+ args = json.loads(tc.function.arguments or "{}") # type: ignore[union-attr]
481
+ except json.JSONDecodeError:
482
+ args = {}
483
+ return {"thought": msg.content or "", "tool": tc.function.name, # type: ignore[union-attr]
484
+ "args": args, "tokens": tokens, "_msg": msg}
485
+ return {"thought": msg.content or "", "final": msg.content or "(no summary)",
486
+ "tokens": tokens, "_msg": msg}
487
+
488
+
489
+ # ---------------------------------------------------------------------------
490
+ # Loop
491
+ # ---------------------------------------------------------------------------
492
+
493
+ @dataclass
494
+ class AgentFinding:
495
+ title: str
496
+ severity: str
497
+ description: str
498
+ evidence: str
499
+ category: str = "agent"
500
+ cve: str = ""
501
+ confidence: float = 0.6
502
+ triage_score: float = -1.0
503
+ triage_factors: dict = field(default_factory=dict)
504
+ kev: bool = False
505
+
506
+
507
+ @dataclass
508
+ class AgentRunResult:
509
+ findings: list[AgentFinding] = field(default_factory=list)
510
+ steps: int = 0
511
+ tokens: int = 0
512
+ probes: int = 0
513
+ finished: bool = False
514
+ summary: str = ""
515
+ transcript: list[dict] = field(default_factory=list)
516
+ workspace: str = ""
517
+ engagement_package: str = ""
518
+
519
+
520
+ def run_agent(tools: AgentTools, brain, *, goal: str, max_steps: int = 30,
521
+ token_budget: int = 200_000, console=None,
522
+ engagement_brief: str = "") -> AgentRunResult:
523
+ """自主 tool-loop:brain 決定每一步,工具層圍籬兜底。
524
+
525
+ engagement_brief: RoE + OPPLAN + 跨會話記憶 的合成文本(借鑑
526
+ Decepticon「行動前先注入紀律」;空字串 = 純預設行為)。
527
+ """
528
+ result = AgentRunResult()
529
+ if tools.ws is not None:
530
+ result.workspace = str(tools.ws.root)
531
+ opening = (f"Authorized target: {tools.target}\nGoal: {goal}\n"
532
+ f"Hard limits: max {max_steps} steps, {tools.max_probes} HTTP probes. ")
533
+ if engagement_brief:
534
+ opening += ("\nEngagement discipline (read before acting):\n" + engagement_brief
535
+ + "\nBegin with fingerprinting, then execute the OPPLAN phases.")
536
+ else:
537
+ opening += "Begin with fingerprinting, then plan your checks."
538
+ messages: list[dict] = [
539
+ {"role": "system", "content": SYSTEM_PROMPT},
540
+ {"role": "user", "content": opening},
541
+ ]
542
+ for step_i in range(max_steps):
543
+ try:
544
+ out = brain.step(messages)
545
+ except Exception as e: # noqa: BLE001
546
+ result.summary = f"brain 中途中斷({e.__class__.__name__}: {e})"
547
+ break
548
+ result.steps += 1
549
+ result.tokens += out.get("tokens", 0)
550
+ thought = out.get("thought", "")
551
+ if console:
552
+ console.print(f" [magenta]◈ step {step_i+1}[/] {thought[:180]}")
553
+
554
+ if "final" in out:
555
+ result.finished = True
556
+ result.summary = out["final"]
557
+ break
558
+
559
+ tool, args = out.get("tool", ""), out.get("args", {})
560
+ if tool == "add_finding":
561
+ cve_arg = str(args.get("cve", ""))
562
+ # triage(借鑑 CypherFix):固定公式風險分,確定性可稽核
563
+ kev_hit = False
564
+ epss_val: float | None = None
565
+ try:
566
+ if re.match(r"^CVE-\d{4}-\d{4,7}$", cve_arg, re.I):
567
+ from .cve_lookup import kev_set as _kev_set
568
+ kev_map, _st = _kev_set()
569
+ kev_hit = cve_arg.upper() in kev_map
570
+ except Exception: # noqa: BLE001 — KEV 不可用不阻斷記錄
571
+ pass
572
+ tr = triage_score(severity=str(args.get("severity", "info")),
573
+ confidence=float(args.get("confidence", 0.6)),
574
+ cve_epss=epss_val, kev=kev_hit,
575
+ reachability=str(args.get("reachability", "direct")))
576
+ f = AgentFinding(
577
+ title=str(args.get("title", ""))[:200],
578
+ severity=str(args.get("severity", "info")).lower(),
579
+ description=str(args.get("description", ""))[:2000],
580
+ evidence=str(args.get("evidence", ""))[:2000],
581
+ category=str(args.get("category", "agent")),
582
+ cve=cve_arg,
583
+ confidence=float(args.get("confidence", 0.6)),
584
+ triage_score=tr.score, triage_factors=tr.factors, kev=kev_hit,
585
+ )
586
+ result.findings.append(f)
587
+ obs = ToolResult(True, {"recorded": f.title,
588
+ "triage": tr.as_dict()})
589
+ if console:
590
+ console.print(f" [green]✎ finding: {f.title} ({f.severity}"
591
+ f" | triage {tr.score})[/]")
592
+ elif tool == "finish":
593
+ result.finished = True
594
+ s = str(args.get("summary", "")).strip() or thought.strip()
595
+ if not s:
596
+ # 模型沒給總結:以實際狀態合成事實性摘要(不編造內容)
597
+ s = (f"評估結束:{len(result.findings)} 項發現已記錄,"
598
+ f"{result.probes} 次探測 / {result.steps} 步。")
599
+ result.summary = s
600
+ break
601
+ else:
602
+ obs = tools.call(tool, args)
603
+
604
+ result.transcript.append({"step": step_i + 1, "thought": thought[:500],
605
+ "tool": tool, "args": args,
606
+ "ok": obs.ok, "note": obs.note,
607
+ "data_preview": str(obs.data)[:800]})
608
+ messages.append({"role": "assistant", "content": thought or f"[call {tool}]"})
609
+ messages.append({"role": "user", "content": (
610
+ f"[tool {tool} -> {'OK' if obs.ok else 'FAIL'}] "
611
+ f"{obs.note + ' | ' if obs.note else ''}"
612
+ f"{json.dumps(obs.data, ensure_ascii=False, default=str)[:4000] if obs.data is not None else ''}"
613
+ f"\n(remaining: {max_steps - step_i - 1} steps, "
614
+ f"{tools.max_probes - tools.probe_count} probes, "
615
+ f"~{max(token_budget - result.tokens, 0)} tokens)")})
616
+
617
+ if result.tokens > token_budget:
618
+ result.summary = result.summary or "token budget 耗盡,自主循環停止"
619
+ break
620
+ else:
621
+ result.summary = result.summary or f"達成 max_steps={max_steps} 上限,未明確 finish"
622
+
623
+ result.probes = tools.probe_count
624
+
625
+ # ---- 收尾:EPSS 回填 triage(批次一次)+ 跨會話記憶累加 ----
626
+ cve_findings = [f for f in result.findings
627
+ if re.match(r"^CVE-\d{4}-\d{4,7}$", f.cve or "", re.I)]
628
+ if cve_findings:
629
+ try:
630
+ from .cve_lookup import epss_scores
631
+ scores = epss_scores([f.cve for f in cve_findings])
632
+ for f in cve_findings:
633
+ e = scores.get(f.cve.upper())
634
+ if e:
635
+ tr = triage_score(severity=f.severity, confidence=f.confidence,
636
+ cve_epss=e["score"], kev=f.kev)
637
+ f.triage_score, f.triage_factors = tr.score, tr.factors
638
+ except Exception: # noqa: BLE001 — EPSS 缺援保留 severity 預設分
639
+ pass
640
+ if tools.ws is not None:
641
+ fdicts = [{"title": f.title, "severity": f.severity, "cve": f.cve,
642
+ "triage_score": f.triage_score} for f in result.findings]
643
+ pdicts = [p.to_dict() for p in tools.probe_log]
644
+ lessons = [t.get("note", "") for t in result.transcript
645
+ if not t.get("ok") and t.get("note")]
646
+ tools.ws.append_session(fdicts, pdicts,
647
+ [l for l in lessons if l][:10],
648
+ result.summary or "")
649
+ return result