logspecter 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
logspecter/__init__.py ADDED
@@ -0,0 +1,34 @@
1
+ """LogSpecter —— 面向云端结构化日志的密钥泄露扫描器。
2
+
3
+ 三层技术壁垒:
4
+ 1. 多维混合检测引擎:正则初筛 + 香农熵二次校验 + 启发式降噪(``entropy`` 模块)。
5
+ 2. 云原生 Schema 结构感知:orjson 零拷贝解析,输出云端身份/动作/JSON 路径(``structured`` / ``cloud``)。
6
+ 3. 内存克制的流式引擎:字节区间行对齐 + mmap 窗口读取,内存占用与文件大小解耦(``ingest``)。
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ __all__ = [
12
+ "Finding",
13
+ "FindingGroup",
14
+ "ScanConfig",
15
+ "ScanResult",
16
+ "ScanStats",
17
+ "Severity",
18
+ "__version__",
19
+ "scan",
20
+ ]
21
+
22
+ __version__ = "0.1.0"
23
+
24
+
25
+ def __getattr__(name: str): # pragma: no cover - 惰性导出,避免导入 CLI 时拉起全部依赖
26
+ if name in {"Finding", "FindingGroup", "Severity"}:
27
+ from logspecter import findings
28
+
29
+ return getattr(findings, name)
30
+ if name in {"ScanConfig", "ScanResult", "ScanStats", "scan"}:
31
+ from logspecter import engine
32
+
33
+ return getattr(engine, name)
34
+ raise AttributeError(f"module {__name__!r} has no attribute {name!r}")
logspecter/__main__.py ADDED
@@ -0,0 +1,12 @@
1
+ """支持 ``python -m logspecter``。
2
+
3
+ 必须保留 ``__main__`` 守卫:Windows / macOS 默认使用 spawn 启动子进程,
4
+ 子进程会重新导入入口模块,没有守卫会导致递归启动。
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from logspecter.cli import main
10
+
11
+ if __name__ == "__main__":
12
+ main()
logspecter/baseline.py ADDED
@@ -0,0 +1,121 @@
1
+ """基线(baseline)与抑制列表。
2
+
3
+ 合规自测的现实:仓库里总有一批「已知、已评估、暂不处置」的命中。基线文件把
4
+ 这些命中的**指纹**记下来,后续扫描自动抑制,从而让 CI 的门禁只对**新增**泄露
5
+ 报警。基线只存指纹(SHA-256 前 16 位)与元数据,不存明文密钥。
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ import json
11
+ import time
12
+ from collections.abc import Iterable, Sequence
13
+ from dataclasses import dataclass, field
14
+ from pathlib import Path
15
+ from typing import Any
16
+
17
+ from logspecter.findings import FindingGroup
18
+
19
+ __all__ = ["Baseline", "BaselineError"]
20
+
21
+ _SCHEMA_VERSION = 1
22
+
23
+
24
+ class BaselineError(ValueError):
25
+ """基线文件不可用或格式不正确。"""
26
+
27
+
28
+ @dataclass(slots=True)
29
+ class Baseline:
30
+ """一份指纹抑制清单。"""
31
+
32
+ fingerprints: set[str] = field(default_factory=set)
33
+ generated_at: str = ""
34
+ note: str = ""
35
+ #: 指纹 → 记录时的摘要信息,仅用于人工审阅。
36
+ entries: dict[str, dict[str, Any]] = field(default_factory=dict)
37
+
38
+ def __contains__(self, fingerprint: object) -> bool:
39
+ return fingerprint in self.fingerprints
40
+
41
+ def __len__(self) -> int:
42
+ return len(self.fingerprints)
43
+
44
+ # ------------------------------------------------------------------ 读写
45
+
46
+ @classmethod
47
+ def load(cls, path: str | Path) -> Baseline:
48
+ """从 JSON 文件读取基线。
49
+
50
+ Raises:
51
+ BaselineError: 文件不存在或结构不合法。
52
+ """
53
+ file = Path(path).expanduser()
54
+ if not file.exists():
55
+ raise BaselineError(f"基线文件不存在: {file}")
56
+ try:
57
+ raw = json.loads(file.read_text(encoding="utf-8"))
58
+ except (OSError, json.JSONDecodeError) as exc:
59
+ raise BaselineError(f"基线文件无法解析: {file}: {exc}") from exc
60
+ if not isinstance(raw, dict):
61
+ raise BaselineError(f"基线文件顶层必须是对象: {file}")
62
+
63
+ version = raw.get("version", _SCHEMA_VERSION)
64
+ if int(version) != _SCHEMA_VERSION:
65
+ raise BaselineError(f"不支持的基线版本 {version}(当前支持 {_SCHEMA_VERSION})")
66
+
67
+ entries = raw.get("entries") or {}
68
+ if not isinstance(entries, dict):
69
+ raise BaselineError(f"基线 entries 必须是对象: {file}")
70
+
71
+ fingerprints = set(raw.get("fingerprints") or entries.keys())
72
+ return cls(
73
+ fingerprints={str(f) for f in fingerprints},
74
+ generated_at=str(raw.get("generated_at", "")),
75
+ note=str(raw.get("note", "")),
76
+ entries={str(k): dict(v) for k, v in entries.items() if isinstance(v, dict)},
77
+ )
78
+
79
+ @classmethod
80
+ def from_groups(cls, groups: Iterable[FindingGroup], *, note: str = "") -> Baseline:
81
+ """由一次扫描结果生成基线。"""
82
+ baseline = cls(
83
+ generated_at=time.strftime("%Y-%m-%dT%H:%M:%S%z"),
84
+ note=note,
85
+ )
86
+ for group in groups:
87
+ representative = group.representative
88
+ baseline.fingerprints.add(group.fingerprint)
89
+ baseline.entries[group.fingerprint] = {
90
+ "rule_id": representative.rule_id,
91
+ "severity": representative.severity.value,
92
+ "source": representative.source,
93
+ "line": representative.line,
94
+ "json_path": representative.json_path,
95
+ "secret_masked": representative.secret_masked,
96
+ "occurrences": group.occurrences,
97
+ }
98
+ return baseline
99
+
100
+ def save(self, path: str | Path) -> Path:
101
+ """写出基线文件,返回实际写入路径。"""
102
+ file = Path(path).expanduser()
103
+ file.parent.mkdir(parents=True, exist_ok=True)
104
+ payload = {
105
+ "version": _SCHEMA_VERSION,
106
+ "generated_at": self.generated_at or time.strftime("%Y-%m-%dT%H:%M:%S%z"),
107
+ "note": self.note,
108
+ "fingerprints": sorted(self.fingerprints),
109
+ "entries": self.entries,
110
+ }
111
+ file.write_text(json.dumps(payload, indent=2, ensure_ascii=False), encoding="utf-8")
112
+ return file
113
+
114
+ # ------------------------------------------------------------------ 过滤
115
+
116
+ def apply(self, groups: Sequence[FindingGroup]) -> tuple[list[FindingGroup], int]:
117
+ """过滤掉已在基线中的分组,返回 ``(保留的分组, 被抑制数量)``。"""
118
+ if not self.fingerprints:
119
+ return list(groups), 0
120
+ kept = [g for g in groups if g.fingerprint not in self.fingerprints]
121
+ return kept, len(groups) - len(kept)