@yottameta/yotta-intel 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +16 -0
- package/LICENSE +21 -0
- package/NOTICE +11 -0
- package/README.md +170 -0
- package/README.zh-CN.md +181 -0
- package/SKILL.md +116 -0
- package/assets/banner.png +0 -0
- package/bin/install.js +163 -0
- package/install.sh +132 -0
- package/package.json +33 -0
- package/references/defang-rules.md +66 -0
- package/references/ioc-spec.md +66 -0
- package/references/stix-lite-spec.md +76 -0
- package/scripts/test_yotta_intel.py +520 -0
- package/scripts/yotta_intel.py +584 -0
|
@@ -0,0 +1,584 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""
|
|
4
|
+
yotta-intel(元情)—— 零依赖自研威胁情报 IOC 提取与规范化引擎
|
|
5
|
+
================================================================
|
|
6
|
+
|
|
7
|
+
跨智能体的威胁情报 IOC 提取能力:从威胁情报文本 / 安全报告 / 钓鱼邮件 / 日志中
|
|
8
|
+
提取 IP(IPv4/IPv6)、域名、URL、邮箱、哈希(MD5/SHA1/SHA256/SHA512)与 CVE 编号,
|
|
9
|
+
自动识别 defang(去活性)写法并还原,去重、归一化后输出 CSV / JSON / STIX-lite。
|
|
10
|
+
|
|
11
|
+
特性
|
|
12
|
+
----
|
|
13
|
+
- 七类 IOC 提取:ipv4 / ipv6 / domain / url / email / hash / cve
|
|
14
|
+
- defang / refang:识别常见去活性写法(hxxp、[.]、(.)、[dot]、[:]、[@] 等)并还原;
|
|
15
|
+
输出时给出安全的 defang 形态,便于在邮件 / 文档 / 工单中共享
|
|
16
|
+
- 归一化:域名小写 + IDN punycode、URL 去默认端口、哈希小写、IPv6 压缩写法
|
|
17
|
+
- 去重计数:同一 IOC 只保留一条,记录出现次数与首次出现的行号 / 上下文
|
|
18
|
+
- 三种结构化输出:CSV / JSON / STIX-lite(STIX 2.1 Bundle + indicator pattern)
|
|
19
|
+
- 纯本地离线处理:不联网查证、不下载样本、不主动扫描任何系统(红线)
|
|
20
|
+
|
|
21
|
+
用法
|
|
22
|
+
----
|
|
23
|
+
python3 scripts/yotta_intel.py extract --path report.txt
|
|
24
|
+
python3 scripts/yotta_intel.py extract --stdin --format json
|
|
25
|
+
python3 scripts/yotta_intel.py extract --path intel.md --types ipv4,domain,hash --min-count 2
|
|
26
|
+
python3 scripts/yotta_intel.py extract --path intel.md --format stix --output iocs.json
|
|
27
|
+
python3 scripts/yotta_intel.py defang --path report.txt --output safe.txt
|
|
28
|
+
python3 scripts/yotta_intel.py refang --path safe.txt --output raw.txt
|
|
29
|
+
python3 scripts/yotta_intel.py --version
|
|
30
|
+
|
|
31
|
+
退出码:extract 0 = 无 IOC;1 = 发现 IOC;4 = 用法或读取错误。
|
|
32
|
+
defang / refang:0 = 成功;4 = 用法或读取错误。
|
|
33
|
+
Windows 下用 python 代替 python3。
|
|
34
|
+
"""
|
|
35
|
+
|
|
36
|
+
import argparse
|
|
37
|
+
import csv
|
|
38
|
+
import io
|
|
39
|
+
import ipaddress
|
|
40
|
+
import json
|
|
41
|
+
import os
|
|
42
|
+
import re
|
|
43
|
+
import sys
|
|
44
|
+
import uuid
|
|
45
|
+
from datetime import datetime, timezone
|
|
46
|
+
from urllib.parse import urlsplit
|
|
47
|
+
|
|
48
|
+
try:
|
|
49
|
+
sys.stdin.reconfigure(encoding="utf-8", errors="replace")
|
|
50
|
+
sys.stdout.reconfigure(encoding="utf-8", errors="replace")
|
|
51
|
+
sys.stderr.reconfigure(encoding="utf-8", errors="replace")
|
|
52
|
+
except Exception:
|
|
53
|
+
pass
|
|
54
|
+
|
|
55
|
+
VERSION = "0.1.0"
|
|
56
|
+
TOOL = "yotta-intel"
|
|
57
|
+
TOOL_CN = "元情"
|
|
58
|
+
|
|
59
|
+
# 七类 IOC:顺序即文本 / CSV / JSON 输出的分组顺序
|
|
60
|
+
IOC_TYPES = ("ipv4", "ipv6", "domain", "url", "email", "hash", "cve")
|
|
61
|
+
IOC_LABELS = {
|
|
62
|
+
"ipv4": "IPv4 地址",
|
|
63
|
+
"ipv6": "IPv6 地址",
|
|
64
|
+
"domain": "域名",
|
|
65
|
+
"url": "URL",
|
|
66
|
+
"email": "邮箱",
|
|
67
|
+
"hash": "哈希",
|
|
68
|
+
"cve": "CVE 编号",
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
# 哈希长度 -> 算法名
|
|
72
|
+
HASH_ALGO = {32: "MD5", 40: "SHA1", 64: "SHA256", 128: "SHA512"}
|
|
73
|
+
HASH_STIX = {"MD5": "MD5", "SHA1": "SHA-1", "SHA256": "SHA-256", "SHA512": "SHA-512"}
|
|
74
|
+
|
|
75
|
+
# 常见 gTLD 与主流新 gTLD(域名判定的白名单;不在列表内的多段名不当作域名)
|
|
76
|
+
GTLD = frozenset("""
|
|
77
|
+
com net org edu gov mil int info biz name pro museum coop aero asia cat jobs mobi
|
|
78
|
+
post tel travel xxx app dev io ai xyz top vip shop cloud click link site online
|
|
79
|
+
tech store live digital network systems security software tools wiki media news
|
|
80
|
+
blog press agency consulting solutions services company center care club community
|
|
81
|
+
cool design directory download email expert express fail finance fit fun gift guru
|
|
82
|
+
host institute international kim life limited lol management marketing moe monster
|
|
83
|
+
partners party photo pics pink plus rent reviews rocks sale science social studio
|
|
84
|
+
support team technology today trading training university vision voting win works
|
|
85
|
+
world zone art auto baby bank bar beauty bio buzz cab camera capital casino chat
|
|
86
|
+
city coffee college crypto date eco energy engineering events exchange faith
|
|
87
|
+
family fashion film fitness food football forex forum fun games glass gold golf
|
|
88
|
+
green group health help home hospital hotel house immo inc insure jet kaufen kids
|
|
89
|
+
kitchen lawyer lease legal life lighting llc loan love ltd luxury makeup market
|
|
90
|
+
media memorial men moda money movie music news ninja page park party pet plumbing
|
|
91
|
+
plus poker porn press pro productions promo properties racing realty repair report
|
|
92
|
+
rest restaurant review rich rip rocks run sale salon save school search services
|
|
93
|
+
sex shoes shop show singles site ski soccer social software solar solutions soy
|
|
94
|
+
space sport storage stream studio style supplies supply support surgery systems
|
|
95
|
+
tax taxi team tech tel tenis theater tickets tienda tips tires today tools town
|
|
96
|
+
toys trade training travel tube university uno vacation vegas ventures video villas
|
|
97
|
+
vision vodka vote voting voyage watch webcam website wedding wiki wine work wtf yoga
|
|
98
|
+
""".split())
|
|
99
|
+
|
|
100
|
+
# 全部 ISO 3166-1 ccTLD
|
|
101
|
+
CCTLD = frozenset("""
|
|
102
|
+
ac ad ae af ag ai al am ao aq ar as at au aw ax az ba bb bd be bf bg bh bi bj bl bm
|
|
103
|
+
bn bo bq br bs bt bv bw by bz ca cc cd cf cg ch ci ck cl cm cn co cr cu cv cw cx cy
|
|
104
|
+
cz de dj dk dm do dz ec ee eg eh er es et eu fi fj fk fm fo fr ga gb gd ge gf gg gh
|
|
105
|
+
gi gl gm gn gp gq gr gs gt gu gw gy hk hm hn hr ht hu id ie il im in io iq ir is it
|
|
106
|
+
je jm jo jp ke kg kh ki km kn kp kr kw ky kz la lb lc li lk lr ls lt lu lv ly ma mc
|
|
107
|
+
md me mf mg mh mk ml mm mn mo mp mq mr ms mt mu mv mw mx my mz na nc ne nf ng ni nl
|
|
108
|
+
no np nr nu nz om pa pe pf pg ph pk pl pm pn pr ps pt pw py qa re ro rs ru rw sa sb
|
|
109
|
+
sc sd se sg sh si sj sk sl sm sn so sr ss st su sv sx sy sz tc td tf tg th tj tk tl
|
|
110
|
+
tm tn to tr tt tv tw tz ua ug uk us uy uz va vc ve vg vi vn vu wf ws ye yt za zm zw
|
|
111
|
+
""".split())
|
|
112
|
+
|
|
113
|
+
TLD_SET = GTLD | CCTLD
|
|
114
|
+
|
|
115
|
+
# 与常见文件扩展名重叠的 TLD:二段域名命中这些时判为文件名而非域名(如 README.md、test.py)
|
|
116
|
+
FILE_EXT_TLDS = frozenset("""
|
|
117
|
+
md py sh js ts json txt log xml html css png jpg jpeg gif webp svg pdf doc docx xls
|
|
118
|
+
xlsx pptx ppt csv zip rar 7z tar gz tgz bz2 xz exe dll so dylib apk deb rpm iso bin
|
|
119
|
+
bat cmd ps1 vbs tmp bak old swp lock env gitignore pyc class jar war o a lib ini cfg
|
|
120
|
+
conf yml yaml key pem crt pfx cer db sqlite db3 mdb accdb mp3 mp4 avi mov mkv wav
|
|
121
|
+
flac map img dmg
|
|
122
|
+
""".split())
|
|
123
|
+
|
|
124
|
+
# ---------------------------------------------------------------------------
|
|
125
|
+
# defang / refang
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
# defang(去活性)是威胁情报共享的常见做法:把 IOC 中可被自动识别的分隔符替换成
|
|
128
|
+
# 「安全」写法,避免收件人 / 平台把纯文本误识别为可点击链接或可解析地址。
|
|
129
|
+
# 本引擎识别常见 defang 写法,先还原成规范形态(refang),再输出统一的 defang 形态。
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def refang_text(text):
|
|
133
|
+
"""把文本中的常见 defang 写法还原为原始形态(行数保持不变)。"""
|
|
134
|
+
t = text
|
|
135
|
+
t = re.sub(r"(?i)\[\.\]|\(\.\)|\{\.\}|\[dot\]|\(dot\)|\{dot\}", ".", t)
|
|
136
|
+
t = re.sub(r"(?i)\[:\]|\(:\)|\{:\}|\[colon\]|\(colon\)|\{colon\}", ":", t)
|
|
137
|
+
t = re.sub(r"(?i)\[@\]|\(@\)|\{@\}|\[at\]|\(at\)|\{at\}", "@", t)
|
|
138
|
+
t = re.sub(r"\[/\]|\(/\)|\[\\/\]|\[/\\\]", "/", t)
|
|
139
|
+
t = re.sub(r"\[\\\]", lambda m: "\\", t)
|
|
140
|
+
t = re.sub(r"(?i)\bhxxps", "https", t)
|
|
141
|
+
t = re.sub(r"(?i)\bhxxp", "http", t)
|
|
142
|
+
return t
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def defang_value(value, ioc_type):
|
|
146
|
+
"""把规范化后的 IOC 转为统一的 defang 形态(用于安全共享展示)。"""
|
|
147
|
+
if ioc_type == "ipv4":
|
|
148
|
+
return value.replace(".", "[.]")
|
|
149
|
+
if ioc_type == "ipv6":
|
|
150
|
+
return value.replace(":", "[:]")
|
|
151
|
+
if ioc_type == "domain":
|
|
152
|
+
return value.replace(".", "[.]")
|
|
153
|
+
if ioc_type == "url":
|
|
154
|
+
if value.lower().startswith("https://"):
|
|
155
|
+
scheme = "hxxps"
|
|
156
|
+
rest = value[8:]
|
|
157
|
+
elif value.lower().startswith("http://"):
|
|
158
|
+
scheme = "hxxp"
|
|
159
|
+
rest = value[7:]
|
|
160
|
+
elif value.lower().startswith("ftp://"):
|
|
161
|
+
scheme = "fxp"
|
|
162
|
+
rest = value[6:]
|
|
163
|
+
else:
|
|
164
|
+
scheme, _, rest = value.partition("://")
|
|
165
|
+
host, sep, tail = rest.partition("/")
|
|
166
|
+
host = host.replace(".", "[.]").replace("@", "[@]")
|
|
167
|
+
return scheme + "://" + host + sep + tail
|
|
168
|
+
if ioc_type == "email":
|
|
169
|
+
return value.replace("@", "[@]").replace(".", "[.]")
|
|
170
|
+
# hash / cve 本身不会被自动链接或解析,保持原样即可
|
|
171
|
+
return value
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def prune_overlaps(matches):
|
|
175
|
+
"""在 (start, end, type, value) 列表中保留互不重叠的最长覆盖(供 defang 流式替换用)。"""
|
|
176
|
+
ms = sorted(matches, key=lambda m: (m[0], -m[1]))
|
|
177
|
+
keep = []
|
|
178
|
+
for m in ms:
|
|
179
|
+
if keep and m[0] < keep[-1][1]:
|
|
180
|
+
continue
|
|
181
|
+
keep.append(m)
|
|
182
|
+
return keep
|
|
183
|
+
|
|
184
|
+
# ---------------------------------------------------------------------------
|
|
185
|
+
# IOC 提取(在 refang 后的文本上工作;refang 只替换分隔符,不改变行结构)
|
|
186
|
+
# ---------------------------------------------------------------------------
|
|
187
|
+
URL_RE = re.compile(r"(?i)(?:(?:https?|ftp)://)[^\s<>\"',。!?;:、()【】《》“”‘’]+")
|
|
188
|
+
EMAIL_RE = re.compile(r"(?i)[a-z0-9._%+\-]+@[a-z0-9\-]+(?:\.[a-z0-9\-]+)+")
|
|
189
|
+
DOMAIN_RE = re.compile(r"(?<![\w@.])(?:[\w\-]{1,63}\.)+[\w\-]{2,63}(?![\-\w])")
|
|
190
|
+
IPV4_RE = re.compile(r"(?<!\d)(?:\d{1,3}\.){3}\d{1,3}(?!\d)")
|
|
191
|
+
IPV6_TOKEN_RE = re.compile(r"(?i)(?<![\w:])(?:[0-9a-f:.]{2,45})(?![\w:])")
|
|
192
|
+
HASH_RE = re.compile(r"(?i)(?<![0-9a-f])[0-9a-f]{32,128}(?![0-9a-f])")
|
|
193
|
+
CVE_RE = re.compile(r"(?i)\bcve-\d{4}-\d{4,7}\b")
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def valid_domain(dom):
|
|
197
|
+
"""域名判定:TLD 白名单 + 标签结构 + 文件名误报过滤。"""
|
|
198
|
+
d = dom.lower().rstrip(".")
|
|
199
|
+
if not d or len(d) > 253 or d.count(".") < 1:
|
|
200
|
+
return False
|
|
201
|
+
labels = d.split(".")
|
|
202
|
+
if len(labels) < 2:
|
|
203
|
+
return False
|
|
204
|
+
tld = labels[-1]
|
|
205
|
+
if not tld.startswith("xn--") and tld not in TLD_SET:
|
|
206
|
+
return False
|
|
207
|
+
# 二段名 + 常见文件扩展名 -> 判为文件名(如 README.md、test.py),不算域名
|
|
208
|
+
if len(labels) == 2 and tld in FILE_EXT_TLDS:
|
|
209
|
+
return False
|
|
210
|
+
for lab in labels:
|
|
211
|
+
if len(lab) > 63 or not re.match(r"^[a-z0-9](?:[a-z0-9\-]{0,61}[a-z0-9])?$", lab):
|
|
212
|
+
return False
|
|
213
|
+
return True
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
def norm_domain(value):
|
|
217
|
+
"""域名归一:小写、去尾点、IDN -> punycode。"""
|
|
218
|
+
d = value.lower().strip().rstrip(".")
|
|
219
|
+
if not d or d.count(".") < 1:
|
|
220
|
+
return None
|
|
221
|
+
try:
|
|
222
|
+
d = d.encode("idna").decode("ascii").lower()
|
|
223
|
+
except (UnicodeError, ValueError):
|
|
224
|
+
return None
|
|
225
|
+
return d
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def norm_url(value):
|
|
229
|
+
"""URL 归一:scheme/host 小写、IDN host、去默认端口、去 fragment。"""
|
|
230
|
+
try:
|
|
231
|
+
parts = urlsplit(value)
|
|
232
|
+
port = parts.port
|
|
233
|
+
except ValueError:
|
|
234
|
+
return None
|
|
235
|
+
scheme = parts.scheme.lower()
|
|
236
|
+
if scheme not in ("http", "https", "ftp"):
|
|
237
|
+
return None
|
|
238
|
+
host = parts.hostname
|
|
239
|
+
if not host:
|
|
240
|
+
return None
|
|
241
|
+
try:
|
|
242
|
+
host = host.encode("idna").decode("ascii").lower()
|
|
243
|
+
except (UnicodeError, ValueError):
|
|
244
|
+
return None
|
|
245
|
+
is_ipv6 = False
|
|
246
|
+
try:
|
|
247
|
+
ipaddress.ip_address(host)
|
|
248
|
+
is_ipv6 = ":" in host
|
|
249
|
+
except ValueError:
|
|
250
|
+
if not valid_domain(host):
|
|
251
|
+
return None
|
|
252
|
+
if port is not None:
|
|
253
|
+
if (scheme == "http" and port == 80) or (scheme == "https" and port == 443) or (scheme == "ftp" and port == 21):
|
|
254
|
+
port = None
|
|
255
|
+
if port is not None:
|
|
256
|
+
host = "%s:%d" % (host, port)
|
|
257
|
+
if is_ipv6 and not host.startswith("["):
|
|
258
|
+
host = "[" + host + "]"
|
|
259
|
+
userinfo = ""
|
|
260
|
+
if parts.username:
|
|
261
|
+
ui = parts.username
|
|
262
|
+
if parts.password is not None:
|
|
263
|
+
ui += ":" + parts.password
|
|
264
|
+
userinfo = ui + "@"
|
|
265
|
+
path = parts.path or ""
|
|
266
|
+
query = ("?" + parts.query) if parts.query else ""
|
|
267
|
+
return "%s://%s%s%s%s" % (scheme, userinfo, host, path, query)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
def norm_email(value):
|
|
271
|
+
"""邮箱归一:小写 + 域名段 IDN。"""
|
|
272
|
+
v = value.strip().lower()
|
|
273
|
+
if "@" not in v:
|
|
274
|
+
return None
|
|
275
|
+
user, _, dom = v.partition("@")
|
|
276
|
+
if not user or not dom:
|
|
277
|
+
return None
|
|
278
|
+
nd = norm_domain(dom)
|
|
279
|
+
if not nd or not valid_domain(nd):
|
|
280
|
+
return None
|
|
281
|
+
return user + "@" + nd
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def find_iocs_in_line(line):
|
|
285
|
+
"""在一行(已 refang)中找出所有 IOC,返回 (start, end, type, canonical)。"""
|
|
286
|
+
found = []
|
|
287
|
+
for m in IPV4_RE.finditer(line):
|
|
288
|
+
raw = m.group(0)
|
|
289
|
+
try:
|
|
290
|
+
ip = ipaddress.IPv4Address(raw)
|
|
291
|
+
except ValueError:
|
|
292
|
+
continue
|
|
293
|
+
found.append((m.start(), m.end(), "ipv4", str(ip)))
|
|
294
|
+
for m in IPV6_TOKEN_RE.finditer(line):
|
|
295
|
+
tok = m.group(0)
|
|
296
|
+
if ":" not in tok:
|
|
297
|
+
continue
|
|
298
|
+
for cand in (tok.rstrip("."), tok):
|
|
299
|
+
try:
|
|
300
|
+
ip6 = ipaddress.IPv6Address(cand)
|
|
301
|
+
except ValueError:
|
|
302
|
+
continue
|
|
303
|
+
found.append((m.start(), m.start() + len(cand), "ipv6", str(ip6)))
|
|
304
|
+
break
|
|
305
|
+
for m in DOMAIN_RE.finditer(line):
|
|
306
|
+
d = norm_domain(m.group(0))
|
|
307
|
+
if d and valid_domain(d):
|
|
308
|
+
found.append((m.start(), m.end(), "domain", d))
|
|
309
|
+
for m in EMAIL_RE.finditer(line):
|
|
310
|
+
e = norm_email(m.group(0))
|
|
311
|
+
if e:
|
|
312
|
+
found.append((m.start(), m.end(), "email", e))
|
|
313
|
+
for m in URL_RE.finditer(line):
|
|
314
|
+
raw = m.group(0).rstrip(".,;:!?\"')]")
|
|
315
|
+
u = norm_url(raw)
|
|
316
|
+
if u:
|
|
317
|
+
found.append((m.start(), m.end(), "url", u))
|
|
318
|
+
for m in HASH_RE.finditer(line):
|
|
319
|
+
h = m.group(0).lower()
|
|
320
|
+
if len(h) in HASH_ALGO:
|
|
321
|
+
found.append((m.start(), m.end(), "hash", h))
|
|
322
|
+
for m in CVE_RE.finditer(line):
|
|
323
|
+
found.append((m.start(), m.end(), "cve", m.group(0).upper()))
|
|
324
|
+
return found
|
|
325
|
+
|
|
326
|
+
|
|
327
|
+
def extract_iocs(text, types=None, min_count=1):
|
|
328
|
+
"""在文本上提取 IOC:refang -> 逐行提取 -> 归一 -> 去重计数。
|
|
329
|
+
|
|
330
|
+
返回记录列表,每条约含:type / value / defanged / count / first_line / snippet。
|
|
331
|
+
"""
|
|
332
|
+
rtext = refang_text(text)
|
|
333
|
+
rlines = rtext.split("\n")
|
|
334
|
+
orig_lines = text.split("\n")
|
|
335
|
+
want = set(types) if types else set(IOC_TYPES)
|
|
336
|
+
buckets = {}
|
|
337
|
+
for idx, line in enumerate(rlines, start=1):
|
|
338
|
+
for start, end, ioc_type, canonical in find_iocs_in_line(line):
|
|
339
|
+
if ioc_type not in want:
|
|
340
|
+
continue
|
|
341
|
+
rec = buckets.setdefault(ioc_type, {})
|
|
342
|
+
entry = rec.get(canonical)
|
|
343
|
+
if entry is None:
|
|
344
|
+
snippet = ""
|
|
345
|
+
if 0 < idx <= len(orig_lines):
|
|
346
|
+
snippet = orig_lines[idx - 1].strip()
|
|
347
|
+
entry = {
|
|
348
|
+
"type": ioc_type,
|
|
349
|
+
"value": canonical,
|
|
350
|
+
"defanged": defang_value(canonical, ioc_type),
|
|
351
|
+
"count": 0,
|
|
352
|
+
"first_line": idx,
|
|
353
|
+
"snippet": snippet,
|
|
354
|
+
}
|
|
355
|
+
rec[canonical] = entry
|
|
356
|
+
entry["count"] += 1
|
|
357
|
+
records = []
|
|
358
|
+
for ioc_type in IOC_TYPES:
|
|
359
|
+
if ioc_type not in want:
|
|
360
|
+
continue
|
|
361
|
+
items = buckets.get(ioc_type, {})
|
|
362
|
+
for entry in sorted(items.values(), key=lambda r: (-r["count"], r["value"])):
|
|
363
|
+
if entry["count"] >= min_count:
|
|
364
|
+
records.append(entry)
|
|
365
|
+
return records
|
|
366
|
+
|
|
367
|
+
|
|
368
|
+
def defang_line(line):
|
|
369
|
+
"""把一行文本中识别到的 IOC 替换为 defang 形态(其余原样保留)。"""
|
|
370
|
+
rline = refang_text(line)
|
|
371
|
+
matches = prune_overlaps(find_iocs_in_line(rline))
|
|
372
|
+
if not matches:
|
|
373
|
+
return line
|
|
374
|
+
buf = list(rline)
|
|
375
|
+
for start, end, ioc_type, canonical in sorted(matches, key=lambda m: -m[0]):
|
|
376
|
+
d = defang_value(canonical, ioc_type)
|
|
377
|
+
buf[start:end] = list(d)
|
|
378
|
+
return "".join(buf)
|
|
379
|
+
|
|
380
|
+
|
|
381
|
+
def defang_text(text):
|
|
382
|
+
"""流式 defang:逐行处理,行结构不变。"""
|
|
383
|
+
return "".join(defang_line(line) for line in text.splitlines(keepends=True))
|
|
384
|
+
|
|
385
|
+
# ---------------------------------------------------------------------------
|
|
386
|
+
# 输出:文本 / JSON / CSV / STIX-lite
|
|
387
|
+
# ---------------------------------------------------------------------------
|
|
388
|
+
def stix_pattern(ioc_type, value):
|
|
389
|
+
"""STIX 2.1 indicator pattern(确定性生成)。"""
|
|
390
|
+
if ioc_type == "ipv4":
|
|
391
|
+
return "[ipv4-addr:value = '%s']" % value
|
|
392
|
+
if ioc_type == "ipv6":
|
|
393
|
+
return "[ipv6-addr:value = '%s']" % value
|
|
394
|
+
if ioc_type == "domain":
|
|
395
|
+
return "[domain-name:value = '%s']" % value
|
|
396
|
+
if ioc_type == "url":
|
|
397
|
+
return "[url:value = '%s']" % value
|
|
398
|
+
if ioc_type == "email":
|
|
399
|
+
return "[email-addr:value = '%s']" % value
|
|
400
|
+
if ioc_type == "hash":
|
|
401
|
+
algo = HASH_ALGO[len(value)]
|
|
402
|
+
return "[file:hashes.'%s' = '%s']" % (HASH_STIX[algo], value)
|
|
403
|
+
if ioc_type == "cve":
|
|
404
|
+
return "[vulnerability:name = '%s']" % value
|
|
405
|
+
return "[x-yottameta:value = '%s']" % value
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
def build_stix(records, generated):
|
|
409
|
+
"""把记录打包成 STIX 2.1 Bundle(lite:只含 indicator 对象 + 自定义扩展属性)。"""
|
|
410
|
+
objects = []
|
|
411
|
+
for r in records:
|
|
412
|
+
oid = uuid.uuid5(uuid.NAMESPACE_URL, "yotta-intel:" + r["type"] + ":" + r["value"])
|
|
413
|
+
objects.append({
|
|
414
|
+
"type": "indicator",
|
|
415
|
+
"spec_version": "2.1",
|
|
416
|
+
"id": "indicator--" + str(oid),
|
|
417
|
+
"created": generated,
|
|
418
|
+
"modified": generated,
|
|
419
|
+
"name": IOC_LABELS[r["type"]] + ": " + r["value"],
|
|
420
|
+
"pattern": stix_pattern(r["type"], r["value"]),
|
|
421
|
+
"pattern_type": "stix",
|
|
422
|
+
"valid_from": generated,
|
|
423
|
+
"labels": ["malicious-activity"],
|
|
424
|
+
"x_yottameta_type": r["type"],
|
|
425
|
+
"x_yottameta_value": r["value"],
|
|
426
|
+
"x_yottameta_defanged": r["defanged"],
|
|
427
|
+
"x_yottameta_count": r["count"],
|
|
428
|
+
})
|
|
429
|
+
return {
|
|
430
|
+
"type": "bundle",
|
|
431
|
+
"id": "bundle--" + str(uuid.uuid5(uuid.NAMESPACE_URL, "yotta-intel:" + generated)),
|
|
432
|
+
"spec_version": "2.1",
|
|
433
|
+
"objects": objects,
|
|
434
|
+
}
|
|
435
|
+
|
|
436
|
+
|
|
437
|
+
def build_json(records, generated, source):
|
|
438
|
+
by_type = {}
|
|
439
|
+
for r in records:
|
|
440
|
+
by_type[r["type"]] = by_type.get(r["type"], 0) + 1
|
|
441
|
+
return json.dumps({
|
|
442
|
+
"tool": TOOL,
|
|
443
|
+
"tool_cn": TOOL_CN,
|
|
444
|
+
"version": VERSION,
|
|
445
|
+
"generated": generated,
|
|
446
|
+
"source": source,
|
|
447
|
+
"summary": {"total": len(records), "by_type": by_type},
|
|
448
|
+
"indicators": records,
|
|
449
|
+
}, ensure_ascii=False, indent=2)
|
|
450
|
+
|
|
451
|
+
|
|
452
|
+
def build_csv(records):
|
|
453
|
+
out = io.StringIO()
|
|
454
|
+
w = csv.writer(out)
|
|
455
|
+
w.writerow(["type", "value", "defanged", "count", "first_line", "snippet"])
|
|
456
|
+
for r in records:
|
|
457
|
+
w.writerow([r["type"], r["value"], r["defanged"], r["count"], r["first_line"], r["snippet"]])
|
|
458
|
+
return out.getvalue()
|
|
459
|
+
|
|
460
|
+
|
|
461
|
+
def build_text(records, context=120):
|
|
462
|
+
lines = ["元情 %s v%s —— IOC 提取结果" % (TOOL, VERSION)]
|
|
463
|
+
if not records:
|
|
464
|
+
lines.append("未发现 IOC。")
|
|
465
|
+
return "\n".join(lines)
|
|
466
|
+
lines.append("共发现 %d 个 IOC:\n" % len(records))
|
|
467
|
+
cur = None
|
|
468
|
+
for r in records:
|
|
469
|
+
if r["type"] != cur:
|
|
470
|
+
cur = r["type"]
|
|
471
|
+
lines.append("■ %s(%s)" % (IOC_LABELS[cur], cur))
|
|
472
|
+
lines.append(" %s ×%d 行 %d" % (r["value"], r["count"], r["first_line"]))
|
|
473
|
+
lines.append(" defang: %s" % r["defanged"])
|
|
474
|
+
sn = r["snippet"]
|
|
475
|
+
if len(sn) > context:
|
|
476
|
+
sn = sn[:context] + "…"
|
|
477
|
+
lines.append(" 上下文: %s" % sn)
|
|
478
|
+
lines.append("")
|
|
479
|
+
return "\n".join(lines)
|
|
480
|
+
|
|
481
|
+
|
|
482
|
+
def emit(text, output):
|
|
483
|
+
if output:
|
|
484
|
+
with open(output, "w", encoding="utf-8") as f:
|
|
485
|
+
f.write(text)
|
|
486
|
+
else:
|
|
487
|
+
sys.stdout.write(text)
|
|
488
|
+
if not text.endswith("\n"):
|
|
489
|
+
sys.stdout.write("\n")
|
|
490
|
+
|
|
491
|
+
|
|
492
|
+
# ---------------------------------------------------------------------------
|
|
493
|
+
# CLI
|
|
494
|
+
# ---------------------------------------------------------------------------
|
|
495
|
+
def read_input(args):
|
|
496
|
+
if getattr(args, "stdin", False):
|
|
497
|
+
return sys.stdin.read(), "<stdin>"
|
|
498
|
+
path = getattr(args, "path", None)
|
|
499
|
+
if path:
|
|
500
|
+
if not os.path.isfile(path):
|
|
501
|
+
raise IOError("文件不存在或不是普通文件: %s" % path)
|
|
502
|
+
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
|
503
|
+
return f.read(), os.path.basename(path)
|
|
504
|
+
raise IOError("必须提供 --path 或 --stdin")
|
|
505
|
+
|
|
506
|
+
|
|
507
|
+
def main():
|
|
508
|
+
ap = argparse.ArgumentParser(
|
|
509
|
+
prog=TOOL,
|
|
510
|
+
description="元情 yotta-intel —— 零依赖威胁情报 IOC 提取与规范化引擎(%s v%s)" % (TOOL_CN, VERSION),
|
|
511
|
+
)
|
|
512
|
+
ap.add_argument("--version", action="store_true", help="显示版本并退出")
|
|
513
|
+
sub = ap.add_subparsers(dest="command")
|
|
514
|
+
|
|
515
|
+
p_extract = sub.add_parser("extract", help="提取 IOC 并输出结构化结果(text/json/csv/stix)")
|
|
516
|
+
p_extract.add_argument("--path", metavar="FILE", help="输入文件")
|
|
517
|
+
p_extract.add_argument("--stdin", action="store_true", help="从标准输入读取")
|
|
518
|
+
p_extract.add_argument("--types", default=",".join(IOC_TYPES),
|
|
519
|
+
help="要提取的 IOC 类型(逗号分隔),默认全部")
|
|
520
|
+
p_extract.add_argument("--format", choices=["text", "json", "csv", "stix"], default="text")
|
|
521
|
+
p_extract.add_argument("--output", metavar="FILE", help="写入文件(默认打印到 stdout)")
|
|
522
|
+
p_extract.add_argument("--min-count", type=int, default=1, help="只保留出现次数 >= N 的 IOC")
|
|
523
|
+
p_extract.add_argument("--context", type=int, default=120, help="文本输出上下文截断宽度")
|
|
524
|
+
|
|
525
|
+
p_defang = sub.add_parser("defang", help="把文本中识别到的 IOC 替换为安全 defang 形态")
|
|
526
|
+
p_defang.add_argument("--path", metavar="FILE")
|
|
527
|
+
p_defang.add_argument("--stdin", action="store_true")
|
|
528
|
+
p_defang.add_argument("--output", metavar="FILE")
|
|
529
|
+
|
|
530
|
+
p_refang = sub.add_parser("refang", help="把 defang 文本还原为原始形态")
|
|
531
|
+
p_refang.add_argument("--path", metavar="FILE")
|
|
532
|
+
p_refang.add_argument("--stdin", action="store_true")
|
|
533
|
+
p_refang.add_argument("--output", metavar="FILE")
|
|
534
|
+
|
|
535
|
+
args = ap.parse_args()
|
|
536
|
+
|
|
537
|
+
if args.version:
|
|
538
|
+
print("%s %s(%s)" % (TOOL, VERSION, TOOL_CN))
|
|
539
|
+
return 0
|
|
540
|
+
if not args.command:
|
|
541
|
+
ap.print_help()
|
|
542
|
+
return 4
|
|
543
|
+
|
|
544
|
+
try:
|
|
545
|
+
if args.command == "extract":
|
|
546
|
+
data, source = read_input(args)
|
|
547
|
+
types = [t.strip().lower() for t in args.types.split(",") if t.strip()]
|
|
548
|
+
unknown = [t for t in types if t not in IOC_TYPES]
|
|
549
|
+
if unknown:
|
|
550
|
+
sys.stderr.write("未知 IOC 类型: %s(可选: %s)\n" % (",".join(unknown), ",".join(IOC_TYPES)))
|
|
551
|
+
return 4
|
|
552
|
+
if args.min_count < 1:
|
|
553
|
+
sys.stderr.write("--min-count 必须 >= 1\n")
|
|
554
|
+
return 4
|
|
555
|
+
records = extract_iocs(data, types=types, min_count=args.min_count)
|
|
556
|
+
generated = datetime.now(timezone.utc).isoformat(timespec="seconds")
|
|
557
|
+
if args.format == "json":
|
|
558
|
+
out = build_json(records, generated, source)
|
|
559
|
+
elif args.format == "csv":
|
|
560
|
+
out = build_csv(records)
|
|
561
|
+
elif args.format == "stix":
|
|
562
|
+
out = json.dumps(build_stix(records, generated), ensure_ascii=False, indent=2)
|
|
563
|
+
else:
|
|
564
|
+
out = build_text(records, context=args.context)
|
|
565
|
+
emit(out, args.output)
|
|
566
|
+
return 1 if records else 0
|
|
567
|
+
if args.command == "defang":
|
|
568
|
+
data, _source = read_input(args)
|
|
569
|
+
emit(defang_text(data), args.output)
|
|
570
|
+
return 0
|
|
571
|
+
if args.command == "refang":
|
|
572
|
+
data, _source = read_input(args)
|
|
573
|
+
emit(refang_text(data), args.output)
|
|
574
|
+
return 0
|
|
575
|
+
except IOError as e:
|
|
576
|
+
sys.stderr.write("错误: %s\n" % e)
|
|
577
|
+
return 4
|
|
578
|
+
except KeyboardInterrupt:
|
|
579
|
+
return 130
|
|
580
|
+
return 4
|
|
581
|
+
|
|
582
|
+
|
|
583
|
+
if __name__ == "__main__":
|
|
584
|
+
sys.exit(main())
|