@yottameta/yotta-compliance 0.0.0 → 0.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,1426 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """
4
+ yotta-compliance(元规)—— 零依赖确定性合规条款审查内核
5
+ ============================================================
6
+
7
+ 设计依据:项目设计稿(2026-09-24 定稿)。
8
+ S2 范围:规范化与偏移映射 / 结构解析 / 规则包加载与校验 / 匹配原语 /
9
+ 断言求值 / 证据链 / Markdown + JSON 报告 / CLI。
10
+
11
+ 可信契约(红线)
12
+ ----------------
13
+ - 结论优先于语言生成:没有原文证据,不输出风险结论;
14
+ - 核心路径不调用模型、不联网;规则包是只读 JSON 数据,绝不当作代码执行;
15
+ - 每条 finding 的 quote 必须由原文切片产生,start/end 基于原始文本;
16
+ - absence 断言必须给出搜索范围,不伪造原文引用;
17
+ - mapping_only 框架只做主题映射,不产生该框架的独立合规结论。
18
+
19
+ 用法
20
+ ----
21
+ python3 scripts/yotta_compliance.py review --input contract.md --frameworks pipl,data-export
22
+ python3 scripts/yotta_compliance.py review --input contract.md --format json --out report.json
23
+ python3 scripts/yotta_compliance.py review --stdin --frameworks pipl --gate high
24
+ python3 scripts/yotta_compliance.py rules list --framework pipl
25
+ python3 scripts/yotta_compliance.py rules show PIPL-DATA-001
26
+ python3 scripts/yotta_compliance.py rules validate --pack rules/pipl.json
27
+ python3 scripts/yotta_compliance.py --version
28
+
29
+ 退出码:0 = 审查完成未超过 gate;1 = 存在达到 gate 的 finding;
30
+ 2 = 输入缺失 / 编码错误 / 路径不可用;3 = 规则包缺失或 schema 无效;4 = CLI 用法错误。
31
+ Windows 下用 python 代替 python3。
32
+ """
33
+
34
+ import argparse
35
+ import hashlib
36
+ import json
37
+ import os
38
+ import re
39
+ import sys
40
+ from datetime import datetime, timezone
41
+
42
+ try:
43
+ sys.stdin.reconfigure(encoding="utf-8", errors="replace")
44
+ sys.stdout.reconfigure(encoding="utf-8", errors="replace")
45
+ sys.stderr.reconfigure(encoding="utf-8", errors="replace")
46
+ except Exception:
47
+ pass
48
+
49
+ VERSION = "0.1.1"
50
+ TOOL = "yotta-compliance"
51
+ TOOL_CN = "元规"
52
+ SCHEMA_VERSION = "1.0"
53
+
54
+ # 输入上限与证据上限(设计 §12 / §9)
55
+ MAX_INPUT_BYTES = 2 * 1024 * 1024
56
+ MAX_EVIDENCE_PER_FINDING = 5
57
+
58
+ DEFAULT_FRAMEWORKS = ("pipl", "data-export")
59
+ SUPPORTED_SUFFIXES = (".txt", ".md", ".markdown")
60
+ UNSUPPORTED_SUFFIXES = (".pdf", ".docx", ".doc")
61
+
62
+ SEVERITY_ORDER = {"info": 0, "low": 1, "medium": 2, "high": 3, "critical": 4}
63
+ SEVERITIES = tuple(SEVERITY_ORDER)
64
+ CONFIDENCE_LEVELS = ("low", "medium", "high")
65
+ COVERAGE_LEVELS = ("baseline_review", "mapping_only")
66
+ CONDITION_KINDS = ("keyword", "regex", "clause_type", "section_anchor")
67
+ ASSERT_OPERATORS = ("absent", "present", "numeric_compare", "date_compare", "duration_compare")
68
+ COMPARE_OPERATORS = ("gte", "lte", "gt", "lt", "eq", "ne")
69
+ EVIDENCE_KINDS = ("matched_span", "document_scope", "section_scope")
70
+ SCOPE_KINDS = ("document", "section")
71
+ DURATION_UNITS = {"day": ("天", "日"), "month": ("个月", "月"), "year": ("年",)}
72
+
73
+ DISCLAIMER = "本工具提供基于确定性规则的条款审查建议与证据链,不构成法律意见。"
74
+ MAPPING_ONLY_NOTE = "仅主题映射,不等同于对该框架的合规审查。"
75
+
76
+ HERE = os.path.dirname(os.path.abspath(__file__))
77
+ DEFAULT_RULES_DIR = os.path.normpath(os.path.join(HERE, os.pardir, "rules"))
78
+
79
+
80
+ # ---------------------------------------------------------------------------
81
+ # 异常与退出码
82
+ # ---------------------------------------------------------------------------
83
+ class ComplianceError(Exception):
84
+ """内核异常基类:exit_code 决定 CLI 退出码。"""
85
+
86
+ exit_code = 4
87
+
88
+
89
+ class InputError(ComplianceError):
90
+ exit_code = 2
91
+
92
+
93
+ class RulePackError(ComplianceError):
94
+ exit_code = 3
95
+
96
+
97
+ class UsageError(ComplianceError):
98
+ exit_code = 4
99
+
100
+
101
+ # ---------------------------------------------------------------------------
102
+ # 规范化与偏移映射(设计 §7)
103
+ # ---------------------------------------------------------------------------
104
+ class NormalizedText(object):
105
+ """规范化文本 + 每个字符回到原始文本的偏移区间。
106
+
107
+ start/end 一律基于原始文本;quote 由原始切片产生。
108
+ """
109
+
110
+ __slots__ = ("original", "text", "_spans")
111
+
112
+ def __init__(self, original, text, spans):
113
+ self.original = original
114
+ self.text = text
115
+ self._spans = spans
116
+
117
+ def span(self, start, end):
118
+ if start >= end:
119
+ raise ValueError("空区间没有原文映射")
120
+ return self._spans[start][0], self._spans[end - 1][1]
121
+
122
+ def quote(self, start, end):
123
+ original_start, original_end = self.span(start, end)
124
+ return self.original[original_start:original_end]
125
+
126
+
127
+ def normalize_text(original):
128
+ """把原文规范化为可匹配文本,同时保留回原文的偏移映射。
129
+
130
+ 规范化仅做确定性、可逆定位的转换:
131
+ 换行统一为 \\n、NBSP / 全角空格转普通空格、全角 ASCII 转半角。
132
+ 不改动中文标点语义,不删除字符,不做同义词替换。
133
+ """
134
+ if not isinstance(original, str):
135
+ raise InputError("输入必须是文本")
136
+ chars = []
137
+ spans = []
138
+ index = 0
139
+ length = len(original)
140
+ while index < length:
141
+ char = original[index]
142
+ if char == "\r":
143
+ if index + 1 < length and original[index + 1] == "\n":
144
+ chars.append("\n")
145
+ spans.append((index, index + 2))
146
+ index += 2
147
+ continue
148
+ chars.append("\n")
149
+ spans.append((index, index + 1))
150
+ elif char in ("\u00a0", "\u3000"):
151
+ chars.append(" ")
152
+ spans.append((index, index + 1))
153
+ elif "\uff01" <= char <= "\uff5e":
154
+ chars.append(chr(ord(char) - 0xfee0))
155
+ spans.append((index, index + 1))
156
+ else:
157
+ chars.append(char)
158
+ spans.append((index, index + 1))
159
+ index += 1
160
+ return NormalizedText(original, "".join(chars), spans)
161
+
162
+
163
+ def locate_span(normalized, start, end):
164
+ """把规范化区间转成原文位置(行 / 列 从 1 开始,按字符计数)。"""
165
+ original = normalized.original
166
+ original_start, original_end = normalized.span(start, end)
167
+ line = original.count("\n", 0, original_start) + 1
168
+ line_start = original.rfind("\n", 0, original_start) + 1
169
+ return {
170
+ "start": original_start,
171
+ "end": original_end,
172
+ "line": line,
173
+ "column": original_start - line_start + 1,
174
+ "quote": original[original_start:original_end],
175
+ }
176
+
177
+
178
+ def _ascii_lower(text):
179
+ """长度不变的小写化:只处理 ASCII 字母,保证偏移可用。"""
180
+ return "".join(chr(ord(ch) + 32) if "A" <= ch <= "Z" else ch for ch in text)
181
+
182
+
183
+ def _line_table(text):
184
+ """返回每行的 (内容起始, 内容结束, 下一行起始)。"""
185
+ rows = []
186
+ index = 0
187
+ length = len(text)
188
+ while index <= length:
189
+ if index == length and (not rows or rows[-1][2] == index):
190
+ break
191
+ newline = text.find("\n", index)
192
+ if newline < 0:
193
+ rows.append((index, length, length))
194
+ break
195
+ rows.append((index, newline, newline + 1))
196
+ index = newline + 1
197
+ return rows
198
+
199
+
200
+ # ---------------------------------------------------------------------------
201
+ # 结构解析(设计 §7)
202
+ # ---------------------------------------------------------------------------
203
+ HEADING_RE = re.compile(r"^(#{1,6})\s+(.*)$")
204
+ CLAUSE_RE = re.compile(r"^(第[0-9一二三四五六七八九十百零〇两]+[章节条])\s*(.*)$")
205
+ ITEM_RE = re.compile(
206
+ r"^(\s*)(\(?\d+[).、]|[((][0-9一二三四五六七八九十]+[))]|[一二三四五六七八九十]+、)\s*(.*)$"
207
+ )
208
+
209
+
210
+ def _trim_span(text, start, end):
211
+ while start < end and text[start].isspace():
212
+ start += 1
213
+ while end > start and text[end - 1].isspace():
214
+ end -= 1
215
+ return start, end
216
+
217
+
218
+ def _make_block(kind, index, level, start, end, label, normalized, id_kind=None, label_span=None):
219
+ original_start, original_end = normalized.span(start, end)
220
+ original = normalized.original
221
+ line = original.count("\n", 0, original_start) + 1
222
+ line_start = original.rfind("\n", 0, original_start) + 1
223
+ if label_span is not None:
224
+ label = normalized.quote(label_span[0], label_span[1])
225
+ return {
226
+ "block_id": "%s-%d" % (id_kind or kind, index),
227
+ "kind": kind,
228
+ "level": level,
229
+ "label": label or "",
230
+ "start": original_start,
231
+ "end": original_end,
232
+ "line": line,
233
+ "column": original_start - line_start + 1,
234
+ "text": original[original_start:original_end],
235
+ "n_start": start,
236
+ "n_end": end,
237
+ }
238
+
239
+
240
+ def parse_structure(normalized):
241
+ """解析标题 / 第X条 / 编号项 / 段落,返回带双坐标系的块列表。
242
+
243
+ kind ∈ heading / clause / item / paragraph;block_id 前缀固定为
244
+ heading- / clause- / item- / para-,标签文本取原文切片。
245
+ """
246
+ text = normalized.text
247
+ blocks = []
248
+ counters = {"heading": 0, "clause": 0, "item": 0, "para": 0}
249
+ para_start = None
250
+ para_end = None
251
+
252
+ def flush_paragraph():
253
+ state = [para_start, para_end]
254
+ start, end = state[0], state[1]
255
+ if start is None:
256
+ return None
257
+ while start < end and text[start].isspace():
258
+ start += 1
259
+ while end > start and text[end - 1].isspace():
260
+ end -= 1
261
+ para = None
262
+ if end > start:
263
+ counters["para"] += 1
264
+ para = _make_block(
265
+ "paragraph", counters["para"], 1, start, end, None, normalized, id_kind="para"
266
+ )
267
+ return para
268
+
269
+ for line_start, line_end, _next in _line_table(text):
270
+ raw_line = text[line_start:line_end]
271
+ stripped = raw_line.strip()
272
+ if not stripped:
273
+ paragraph = flush_paragraph()
274
+ if paragraph:
275
+ blocks.append(paragraph)
276
+ para_start = None
277
+ para_end = None
278
+ continue
279
+ heading = HEADING_RE.match(raw_line)
280
+ if heading:
281
+ paragraph = flush_paragraph()
282
+ if paragraph:
283
+ blocks.append(paragraph)
284
+ para_start = None
285
+ para_end = None
286
+ counters["heading"] += 1
287
+ label_span = _trim_span(
288
+ text, line_start + heading.start(2), line_start + heading.end(2)
289
+ )
290
+ blocks.append(
291
+ _make_block(
292
+ "heading", counters["heading"], len(heading.group(1)),
293
+ line_start, line_end, heading.group(2).strip(), normalized,
294
+ label_span=label_span,
295
+ )
296
+ )
297
+ continue
298
+ clause = CLAUSE_RE.match(raw_line)
299
+ if clause:
300
+ paragraph = flush_paragraph()
301
+ if paragraph:
302
+ blocks.append(paragraph)
303
+ para_start = None
304
+ para_end = None
305
+ counters["clause"] += 1
306
+ label_span = (line_start + clause.start(1), line_start + clause.end(1))
307
+ blocks.append(
308
+ _make_block(
309
+ "clause", counters["clause"], 1,
310
+ line_start, line_end, clause.group(1), normalized,
311
+ label_span=label_span,
312
+ )
313
+ )
314
+ continue
315
+ item = ITEM_RE.match(raw_line)
316
+ if item:
317
+ paragraph = flush_paragraph()
318
+ if paragraph:
319
+ blocks.append(paragraph)
320
+ para_start = None
321
+ para_end = None
322
+ counters["item"] += 1
323
+ level = 1 + (len(item.group(1)) // 4)
324
+ label_span = (line_start + item.start(2), line_start + item.end(2))
325
+ blocks.append(
326
+ _make_block(
327
+ "item", counters["item"], min(level, 4),
328
+ line_start, line_end, item.group(2), normalized,
329
+ label_span=label_span,
330
+ )
331
+ )
332
+ continue
333
+ if para_start is None:
334
+ para_start = line_start
335
+ para_end = line_end
336
+ paragraph = flush_paragraph()
337
+ if paragraph:
338
+ blocks.append(paragraph)
339
+ blocks.sort(key=lambda block: block["n_start"])
340
+ return blocks
341
+
342
+
343
+ # ---------------------------------------------------------------------------
344
+ # 规则包加载与校验(设计 §8)
345
+ # ---------------------------------------------------------------------------
346
+ def _require(condition, message):
347
+ if not condition:
348
+ raise RulePackError(message)
349
+
350
+
351
+ def _validate_condition(condition, where, lexicon):
352
+ _require(isinstance(condition, dict), "%s:条件必须是 JSON 对象" % where)
353
+ kind = condition.get("kind")
354
+ _require(kind in CONDITION_KINDS, "%s:未知条件类型 %r(允许:%s)" % (where, kind, ", ".join(CONDITION_KINDS)))
355
+ if kind == "regex":
356
+ pattern = condition.get("pattern")
357
+ _require(isinstance(pattern, str) and pattern, "%s:regex 条件缺少 pattern" % where)
358
+ try:
359
+ re.compile(pattern)
360
+ except re.error as exc:
361
+ raise RulePackError("%s:regex 无法编译(%s)" % (where, exc))
362
+ elif kind in ("keyword", "clause_type", "section_anchor"):
363
+ value = condition.get("value")
364
+ _require(isinstance(value, str) and value, "%s:%s 条件缺少 value" % (where, kind))
365
+ if kind == "clause_type":
366
+ _require(
367
+ value in lexicon,
368
+ "%s:clause_type %r 未在 clause_type_lexicon 声明" % (where, value),
369
+ )
370
+ return condition
371
+
372
+
373
+ def _validate_assert(assert_spec, where, lexicon):
374
+ _require(isinstance(assert_spec, dict), "%s:assert 必须是 JSON 对象" % where)
375
+ operator = assert_spec.get("operator")
376
+ _require(
377
+ operator in ASSERT_OPERATORS,
378
+ "%s:未知断言 operator %r(允许:%s)" % (where, operator, ", ".join(ASSERT_OPERATORS)),
379
+ )
380
+ target = assert_spec.get("target")
381
+ _require(isinstance(target, str) and target, "%s:assert 缺少 target" % where)
382
+ search = assert_spec.get("search")
383
+ _require(isinstance(search, dict), "%s:assert 缺少 search(absence 必须给出搜索范围)" % where)
384
+ _validate_condition(search, "%s.assert.search" % where, lexicon)
385
+ if operator == "absent":
386
+ scope = assert_spec.get("scope")
387
+ _require(scope in SCOPE_KINDS, "%s:absence 断言必须声明 scope(%s)" % (where, ", ".join(SCOPE_KINDS)))
388
+ elif operator in ("numeric_compare", "date_compare", "duration_compare"):
389
+ group = search.get("group", 1)
390
+ _require(isinstance(group, int) and group >= 1, "%s:compare 断言需要 search.group >= 1" % where)
391
+ pattern = search.get("pattern")
392
+ _require(isinstance(pattern, str) and re.compile(pattern).groups >= group,
393
+ "%s:search 正则缺少第 %s 个捕获组" % (where, group))
394
+ compare = assert_spec.get("compare")
395
+ _require(isinstance(compare, dict), "%s:compare 断言缺少 compare" % where)
396
+ _require(compare.get("op") in COMPARE_OPERATORS,
397
+ "%s:compare.op 非法(允许:%s)" % (where, ", ".join(COMPARE_OPERATORS)))
398
+ _require("value" in compare, "%s:compare 缺少 value" % where)
399
+ if operator == "numeric_compare":
400
+ _require(isinstance(compare["value"], (int, float)) and not isinstance(compare["value"], bool),
401
+ "%s:numeric_compare 的 value 必须是数字" % where)
402
+ elif operator == "date_compare":
403
+ _require(isinstance(compare["value"], str) and _parse_date(compare["value"]) is not None,
404
+ "%s:date_compare 的 value 必须是可解析日期(YYYY-MM-DD)" % where)
405
+ else:
406
+ _require(assert_spec.get("unit") in DURATION_UNITS,
407
+ "%s:duration_compare 必须声明 unit(%s)" % (where, ", ".join(sorted(DURATION_UNITS))))
408
+ return assert_spec
409
+
410
+
411
+ def validate_pack(pack, source=None):
412
+ """校验规则包;任何不合规直接抛 RulePackError,绝不降级继续审查。"""
413
+ where = source or "规则包"
414
+ _require(isinstance(pack, dict), "%s:根节点必须是 JSON 对象" % where)
415
+ for field in ("schema_version", "pack_id", "framework", "pack_version", "coverage_level", "rules"):
416
+ _require(field in pack and pack[field] not in (None, ""), "%s:缺少字段 %s" % (where, field))
417
+ _require(pack["schema_version"] == SCHEMA_VERSION,
418
+ "%s:schema_version 必须是 %s(当前 %r)" % (where, SCHEMA_VERSION, pack["schema_version"]))
419
+ _require(pack["coverage_level"] in COVERAGE_LEVELS,
420
+ "%s:coverage_level 非法(允许:%s)" % (where, ", ".join(COVERAGE_LEVELS)))
421
+ lexicon = pack.get("clause_type_lexicon") or {}
422
+ _require(isinstance(lexicon, dict), "%s:clause_type_lexicon 必须是对象" % where)
423
+ rules = pack["rules"]
424
+ _require(isinstance(rules, list) and rules, "%s:rules 必须是非空数组" % where)
425
+
426
+ seen_rule_ids = set()
427
+ for index, rule in enumerate(rules):
428
+ rule_where = "%s#rules[%d]" % (where, index)
429
+ _require(isinstance(rule, dict), "%s:规则必须是 JSON 对象" % rule_where)
430
+ rule_id = rule.get("rule_id")
431
+ _require(isinstance(rule_id, str) and rule_id, "%s:缺少 rule_id" % rule_where)
432
+ rule_where = "%s(%s)" % (rule_where, rule_id)
433
+ _require(rule_id not in seen_rule_ids, "%s:rule_id 重复" % rule_where)
434
+ seen_rule_ids.add(rule_id)
435
+ for field in ("title", "severity", "evidence", "rationale", "source", "remediation", "test_ids"):
436
+ _require(field in rule and rule[field] not in (None, ""), "%s:缺少字段 %s" % (rule_where, field))
437
+ _require(rule["severity"] in SEVERITY_ORDER,
438
+ "%s:severity 非法(允许:%s)" % (rule_where, ", ".join(SEVERITIES)))
439
+ source_spec = rule["source"]
440
+ _require(isinstance(source_spec, dict), "%s:source 必须是对象" % rule_where)
441
+ _require(isinstance(source_spec.get("citation"), str) and source_spec["citation"],
442
+ "%s:source 缺少 citation" % rule_where)
443
+ evidence = rule["evidence"]
444
+ _require(isinstance(evidence, list) and evidence, "%s:evidence 不能为空" % rule_where)
445
+ for kind in evidence:
446
+ _require(kind in EVIDENCE_KINDS,
447
+ "%s:未知证据类型 %r(允许:%s)" % (rule_where, kind, ", ".join(EVIDENCE_KINDS)))
448
+ test_ids = rule["test_ids"]
449
+ _require(isinstance(test_ids, list) and test_ids, "%s:test_ids 不能为空(每条规则需正反例)" % rule_where)
450
+ match = rule.get("match")
451
+ _require(isinstance(match, dict), "%s:缺少 match" % rule_where)
452
+ _require(match.get("operator") in ("any", "all"), "%s:match.operator 必须是 any / all" % rule_where)
453
+ conditions = match.get("conditions")
454
+ _require(isinstance(conditions, list) and conditions, "%s:match.conditions 不能为空" % rule_where)
455
+ regex_seen = set()
456
+ for position, condition in enumerate(conditions):
457
+ _validate_condition(condition, "%s.match.conditions[%d]" % (rule_where, position), lexicon)
458
+ if condition.get("kind") == "regex":
459
+ pattern = condition["pattern"]
460
+ _require(pattern not in regex_seen,
461
+ "%s:同一规则内重复 regex(%r)会导致同一证据重复计数" % (rule_where, pattern))
462
+ regex_seen.add(pattern)
463
+ _validate_assert(rule.get("assert"), "%s.assert" % rule_where, lexicon)
464
+ confidence_policy = rule.get("confidence_policy") or {}
465
+ _require(isinstance(confidence_policy, dict), "%s:confidence_policy 必须是对象" % rule_where)
466
+ for key in ("exact_match", "absence"):
467
+ if key in confidence_policy:
468
+ _require(confidence_policy[key] in CONFIDENCE_LEVELS,
469
+ "%s:confidence_policy.%s 非法" % (rule_where, key))
470
+
471
+ mapping_only = pack.get("mapping_only") or []
472
+ _require(isinstance(mapping_only, list), "%s:mapping_only 必须是数组" % where)
473
+ for index, entry in enumerate(mapping_only):
474
+ entry_where = "%s#mapping_only[%d]" % (where, index)
475
+ _require(isinstance(entry, dict), "%s:必须是 JSON 对象" % entry_where)
476
+ _require(isinstance(entry.get("framework"), str) and entry["framework"],
477
+ "%s:缺少 framework" % entry_where)
478
+ topics = entry.get("topics")
479
+ _require(isinstance(topics, dict) and topics, "%s:mapping_only 必须声明 topics" % entry_where)
480
+ for topic, rule_ids in topics.items():
481
+ _require(isinstance(rule_ids, list) and rule_ids,
482
+ "%s:topic %s 必须给出引用的 rule_id" % (entry_where, topic))
483
+ for rule_id in rule_ids:
484
+ _require(rule_id in seen_rule_ids,
485
+ "%s:topic %s 引用了不存在的 rule_id %s" % (entry_where, topic, rule_id))
486
+ return pack
487
+
488
+
489
+ def load_pack(path):
490
+ """读取并校验单个规则包;返回带 _path / _sha256 元数据的包。"""
491
+ if not os.path.isfile(path):
492
+ raise RulePackError("规则包不存在:%s" % path)
493
+ try:
494
+ with open(path, "rb") as handle:
495
+ raw = handle.read()
496
+ except OSError as exc:
497
+ raise RulePackError("规则包不可读(%s):%s" % (exc, path))
498
+ try:
499
+ pack = json.loads(raw.decode("utf-8-sig"))
500
+ except UnicodeDecodeError:
501
+ raise RulePackError("规则包必须是 UTF-8 编码:%s" % path)
502
+ except ValueError as exc:
503
+ raise RulePackError("规则包不是合法 JSON(%s):%s" % (exc, path))
504
+ validate_pack(pack, source=os.path.basename(path))
505
+ pack["_path"] = os.path.abspath(path)
506
+ pack["_sha256"] = hashlib.sha256(raw).hexdigest()
507
+ return pack
508
+
509
+
510
+ def available_frameworks(rules_dir):
511
+ """列出规则目录内可用的框架 slug(文件名的 stem,例如 pipl.json -> pipl)。"""
512
+ if not os.path.isdir(rules_dir):
513
+ raise RulePackError("规则包目录不存在:%s" % rules_dir)
514
+ names = []
515
+ for name in sorted(os.listdir(rules_dir)):
516
+ if name.endswith(".json"):
517
+ names.append(name[:-5])
518
+ return names
519
+
520
+
521
+ def load_rule_packs(rules_dir, frameworks=None):
522
+ """加载规则包;frameworks 为空时加载目录内全部包。"""
523
+ names = available_frameworks(rules_dir)
524
+ if not names:
525
+ raise RulePackError("规则包目录内没有 JSON 规则包:%s" % rules_dir)
526
+ if frameworks:
527
+ unknown = [item for item in frameworks if item not in names]
528
+ if unknown:
529
+ raise UsageError(
530
+ "未知框架:%s;可用框架:%s" % (", ".join(unknown), ", ".join(names))
531
+ )
532
+ selected = list(frameworks)
533
+ else:
534
+ selected = names
535
+ packs = []
536
+ seen_rule_ids = {}
537
+ for name in selected:
538
+ pack = load_pack(os.path.join(rules_dir, name + ".json"))
539
+ for rule in pack["rules"]:
540
+ rule_id = rule["rule_id"]
541
+ if rule_id in seen_rule_ids:
542
+ raise RulePackError(
543
+ "rule_id 跨包重复:%s(%s 与 %s)" % (rule_id, seen_rule_ids[rule_id], name)
544
+ )
545
+ seen_rule_ids[rule_id] = name
546
+ packs.append(pack)
547
+ return packs
548
+
549
+
550
+ # ---------------------------------------------------------------------------
551
+ # 匹配原语(设计 §8.3)
552
+ # ---------------------------------------------------------------------------
553
+ class _MatchContext(object):
554
+ __slots__ = ("normalized", "lowered", "blocks", "pack")
555
+
556
+ def __init__(self, normalized, blocks, pack):
557
+ self.normalized = normalized
558
+ self.lowered = _ascii_lower(normalized.text)
559
+ self.blocks = blocks
560
+ self.pack = pack
561
+
562
+
563
+ def _section_for_span(blocks, n_start, n_end):
564
+ for block in blocks:
565
+ if block["n_start"] <= n_start and n_end <= block["n_end"]:
566
+ return block["block_id"]
567
+ return None
568
+
569
+
570
+ def _hit(ctx, n_start, n_end, condition_kind):
571
+ locator = locate_span(ctx.normalized, n_start, n_end)
572
+ hit = {
573
+ "condition_kind": condition_kind,
574
+ "n_start": n_start,
575
+ "n_end": n_end,
576
+ "section_id": _section_for_span(ctx.blocks, n_start, n_end),
577
+ }
578
+ hit.update(locator)
579
+ return hit
580
+
581
+
582
+ def _literal_hits(ctx, keyword, condition_kind):
583
+ needle = _ascii_lower(keyword)
584
+ hits = []
585
+ start = ctx.lowered.find(needle)
586
+ while start >= 0:
587
+ hits.append(_hit(ctx, start, start + len(needle), condition_kind))
588
+ start = ctx.lowered.find(needle, start + len(needle))
589
+ return hits
590
+
591
+
592
+ def _regex_hits(ctx, pattern, condition_kind):
593
+ hits = []
594
+ for match in re.finditer(pattern, ctx.normalized.text, re.IGNORECASE):
595
+ if match.start() == match.end():
596
+ continue
597
+ hits.append(_hit(ctx, match.start(), match.end(), condition_kind))
598
+ return hits
599
+
600
+
601
+ def _condition_hits(condition, ctx):
602
+ kind = condition["kind"]
603
+ if kind == "keyword":
604
+ return _literal_hits(ctx, condition["value"], kind)
605
+ if kind == "regex":
606
+ return _regex_hits(ctx, condition["pattern"], kind)
607
+ if kind == "clause_type":
608
+ keywords = (ctx.pack.get("clause_type_lexicon") or {}).get(condition["value"], [])
609
+ hits = []
610
+ for block in ctx.blocks:
611
+ block_text = ctx.normalized.text[block["n_start"]:block["n_end"]]
612
+ lowered = _ascii_lower(block_text)
613
+ for keyword in keywords:
614
+ offset = lowered.find(_ascii_lower(keyword))
615
+ if offset >= 0:
616
+ start = block["n_start"] + offset
617
+ hits.append(_hit(ctx, start, start + len(keyword), kind))
618
+ break
619
+ return hits
620
+ # section_anchor:锚点文本命中所在块
621
+ hits = []
622
+ anchor = _ascii_lower(condition["value"])
623
+ for block in ctx.blocks:
624
+ block_text = _ascii_lower(ctx.normalized.text[block["n_start"]:block["n_end"]])
625
+ offset = block_text.find(anchor)
626
+ if offset >= 0:
627
+ start = block["n_start"] + offset
628
+ hits.append(_hit(ctx, start, start + len(anchor), kind))
629
+ return hits
630
+
631
+
632
+ def _dedupe_hits(hits):
633
+ seen = set()
634
+ unique = []
635
+ for hit in sorted(hits, key=lambda item: (item["n_start"], item["n_end"])):
636
+ key = (hit["n_start"], hit["n_end"])
637
+ if key in seen:
638
+ continue
639
+ seen.add(key)
640
+ unique.append(hit)
641
+ return unique
642
+
643
+
644
+ def match_rule(rule, normalized, blocks, pack):
645
+ """按 match 条件求候选;返回命中区间(原文坐标)与是否命中。"""
646
+ ctx = _MatchContext(normalized, blocks, pack)
647
+ conditions = rule["match"]["conditions"]
648
+ per_condition = [_condition_hits(condition, ctx) for condition in conditions]
649
+ if rule["match"]["operator"] == "all":
650
+ matched = all(per_condition)
651
+ else:
652
+ matched = any(per_condition)
653
+ hits = _dedupe_hits([hit for hits in per_condition for hit in hits])
654
+ return {"matched": bool(matched and hits), "hits": hits}
655
+
656
+
657
+ # ---------------------------------------------------------------------------
658
+ # 断言求值(设计 §8.2 / §9 / §10)
659
+ # ---------------------------------------------------------------------------
660
+ _CHINESE_DIGITS = {
661
+ "零": 0, "〇": 0, "一": 1, "二": 2, "两": 2, "三": 3, "四": 4,
662
+ "五": 5, "六": 6, "七": 7, "八": 8, "九": 9,
663
+ }
664
+ _DATE_FORMATS = ("%Y-%m-%d", "%Y/%m/%d", "%Y.%m.%d", "%Y年%m月%d日")
665
+
666
+
667
+ def _parse_number(text):
668
+ token = text.strip()
669
+ if not token:
670
+ return None
671
+ try:
672
+ return float(token)
673
+ except ValueError:
674
+ pass
675
+ if all(char in _CHINESE_DIGITS for char in token):
676
+ return float("".join(str(_CHINESE_DIGITS[char]) for char in token))
677
+ if "十" in token:
678
+ parts = token.split("十")
679
+ if len(parts) == 2 and all(part == "" or all(c in _CHINESE_DIGITS for c in part) for part in parts):
680
+ tens = 1 if parts[0] == "" else _CHINESE_DIGITS.get(parts[0], 0)
681
+ ones = 0 if parts[1] == "" else _CHINESE_DIGITS.get(parts[1], 0)
682
+ return float(tens * 10 + ones)
683
+ return None
684
+
685
+
686
+ def _parse_date(text):
687
+ token = text.strip()
688
+ for fmt in _DATE_FORMATS:
689
+ try:
690
+ return datetime.strptime(token, fmt).date()
691
+ except ValueError:
692
+ continue
693
+ return None
694
+
695
+
696
+ def _compare(left, right, operator):
697
+ if operator == "gte":
698
+ return left >= right
699
+ if operator == "lte":
700
+ return left <= right
701
+ if operator == "gt":
702
+ return left > right
703
+ if operator == "lt":
704
+ return left < right
705
+ if operator == "eq":
706
+ return left == right
707
+ return left != right
708
+
709
+
710
+ def _extract_group_span(pattern, text, n_start, n_end, group):
711
+ for match in re.finditer(pattern, text, re.IGNORECASE):
712
+ if match.span() == (n_start, n_end):
713
+ span = match.span(group)
714
+ if span[0] < 0:
715
+ return None
716
+ return span
717
+ return None
718
+
719
+
720
+ def _scope_evidence(ctx, scope, hits):
721
+ def with_range(locator, n_start, n_end):
722
+ locator.pop("quote", None)
723
+ end = locate_span(ctx.normalized, n_end - 1, n_end) if n_end > n_start else locator
724
+ locator["end_line"] = end["line"]
725
+ locator["end_column"] = end["column"]
726
+ return locator
727
+
728
+ if scope == "section" and hits:
729
+ section_id = hits[0]["section_id"]
730
+ for block in ctx.blocks:
731
+ if block["block_id"] == section_id:
732
+ locator = locate_span(ctx.normalized, block["n_start"], block["n_end"])
733
+ locator = with_range(locator, block["n_start"], block["n_end"])
734
+ locator["kind"] = "section_scope"
735
+ locator["section_id"] = block["block_id"]
736
+ locator["scope_label"] = "所在段:%s" % block["block_id"]
737
+ return locator
738
+ length = len(ctx.normalized.text)
739
+ locator = locate_span(ctx.normalized, 0, length)
740
+ locator = with_range(locator, 0, length)
741
+ locator["kind"] = "document_scope"
742
+ locator["section_id"] = None
743
+ locator["scope_label"] = "全文范围"
744
+ return locator
745
+
746
+
747
+ def _matched_evidence(hits, limit=MAX_EVIDENCE_PER_FINDING):
748
+ evidence = []
749
+ for hit in hits[:limit]:
750
+ entry = {
751
+ "kind": "matched_span",
752
+ "quote": hit["quote"],
753
+ "start": hit["start"],
754
+ "end": hit["end"],
755
+ "line": hit["line"],
756
+ "column": hit["column"],
757
+ "section_id": hit["section_id"],
758
+ }
759
+ evidence.append(entry)
760
+ return evidence
761
+
762
+
763
+ def _confidence(rule, key, default):
764
+ policy = rule.get("confidence_policy") or {}
765
+ return policy.get(key, default)
766
+
767
+
768
+ def _framework_mapping(rule, pack):
769
+ mapping = [pack["framework"]]
770
+ for entry in pack.get("mapping_only") or []:
771
+ for rule_ids in (entry.get("topics") or {}).values():
772
+ if rule["rule_id"] in rule_ids:
773
+ label = "%s:mapping_only" % entry["framework"]
774
+ if label not in mapping:
775
+ mapping.append(label)
776
+ break
777
+ return mapping
778
+
779
+
780
+ def _evaluate_assert(rule, ctx, hits):
781
+ """返回 {"outcome": finding|none, "confidence":..., "evidence":[...], "review":{...}}。"""
782
+ assert_spec = rule["assert"]
783
+ operator = assert_spec["operator"]
784
+ search = assert_spec["search"]
785
+ search_hits = _dedupe_hits(_condition_hits(search, ctx))
786
+ if search.get("kind") == "regex" and search.get("group"):
787
+ search_hits = [
788
+ hit for hit in search_hits
789
+ if _extract_group_span(
790
+ search["pattern"], ctx.normalized.text, hit["n_start"], hit["n_end"], search["group"]
791
+ ) is not None
792
+ ]
793
+
794
+ if operator == "absent":
795
+ if search_hits:
796
+ return {"outcome": "none"}
797
+ evidence = _matched_evidence(hits)
798
+ evidence.append(_scope_evidence(ctx, assert_spec.get("scope", "document"), hits))
799
+ return {
800
+ "outcome": "finding",
801
+ "confidence": _confidence(rule, "absence", "medium"),
802
+ "evidence": evidence,
803
+ "evidence_total": len(hits) + 1,
804
+ }
805
+
806
+ if operator == "present":
807
+ if not search_hits:
808
+ return {"outcome": "none"}
809
+ return {
810
+ "outcome": "finding",
811
+ "confidence": _confidence(rule, "exact_match", "high"),
812
+ "evidence": _matched_evidence(search_hits),
813
+ "evidence_total": len(search_hits),
814
+ }
815
+
816
+ if not search_hits:
817
+ return {
818
+ "outcome": "review",
819
+ "review": {
820
+ "kind": "unverifiable",
821
+ "rule_id": rule["rule_id"],
822
+ "reason": "断言目标无法在原文中定位,未输出数值结论,请人工核验。",
823
+ },
824
+ }
825
+
826
+ hit = search_hits[0]
827
+ group_span = _extract_group_span(
828
+ search["pattern"], ctx.normalized.text, hit["n_start"], hit["n_end"], search.get("group", 1)
829
+ )
830
+ if group_span is None:
831
+ return {
832
+ "outcome": "review",
833
+ "review": {
834
+ "kind": "unverifiable",
835
+ "rule_id": rule["rule_id"],
836
+ "reason": "断言目标无法从原文切片中解析,未输出结论,请人工核验。",
837
+ },
838
+ }
839
+ value_text = ctx.normalized.text[group_span[0]:group_span[1]]
840
+ compare = assert_spec["compare"]
841
+
842
+ if operator == "numeric_compare":
843
+ value = _parse_number(value_text)
844
+ if value is None:
845
+ return _unverifiable(rule, "原文数值无法解析(%s),未自动换算或补值。" % value_text.strip())
846
+ left, right = value, float(compare["value"])
847
+ elif operator == "date_compare":
848
+ value = _parse_date(value_text)
849
+ if value is None:
850
+ return _unverifiable(rule, "原文日期无法解析(%s),未输出结论。" % value_text.strip())
851
+ left, right = value, _parse_date(compare["value"])
852
+ else:
853
+ unit = assert_spec["unit"]
854
+ found_units = [
855
+ name for name, tokens in DURATION_UNITS.items()
856
+ if any(token in hit["quote"] for token in tokens)
857
+ ]
858
+ if found_units != [unit]:
859
+ return _unverifiable(
860
+ rule,
861
+ "时长单位与规则声明不一致(原文:%s;规则:%s),不做跨单位换算。"
862
+ % (", ".join(found_units) or "未识别", unit),
863
+ )
864
+ value = _parse_number(value_text)
865
+ if value is None:
866
+ return _unverifiable(rule, "原文时长数值无法解析(%s)。" % value_text.strip())
867
+ left, right = value, float(compare["value"])
868
+
869
+ if not _compare(left, right, compare["op"]):
870
+ return {"outcome": "none"}
871
+ return {
872
+ "outcome": "finding",
873
+ "confidence": _confidence(rule, "exact_match", "high"),
874
+ "evidence": _matched_evidence([hit]),
875
+ "evidence_total": 1,
876
+ }
877
+
878
+
879
+ def _unverifiable(rule, reason):
880
+ return {
881
+ "outcome": "review",
882
+ "review": {"kind": "unverifiable", "rule_id": rule["rule_id"], "reason": reason},
883
+ }
884
+
885
+
886
+ # ---------------------------------------------------------------------------
887
+ # 审查主流程(设计 §4 / §7 / §11.2)
888
+ # ---------------------------------------------------------------------------
889
+ def _severity_rank(severity):
890
+ return SEVERITY_ORDER.get(severity, 0)
891
+
892
+
893
+ def _input_summary(text, input_meta):
894
+ meta = dict(input_meta or {})
895
+ payload = text.encode("utf-8")
896
+ summary = {
897
+ "kind": meta.get("kind", "memory"),
898
+ "path": meta.get("path"),
899
+ "sha256": hashlib.sha256(payload).hexdigest(),
900
+ "bytes": len(payload),
901
+ "lines": (text.count("\n") + 1) if text else 0,
902
+ }
903
+ return summary
904
+
905
+
906
+ def _framework_coverage(packs):
907
+ coverage = []
908
+ seen = set()
909
+ for pack in packs:
910
+ coverage.append({
911
+ "framework": pack["framework"],
912
+ "coverage_level": pack["coverage_level"],
913
+ "status": "reviewed" if pack["coverage_level"] == "baseline_review" else "mapping_only",
914
+ "pack_id": pack["pack_id"],
915
+ "pack_version": pack["pack_version"],
916
+ "rule_count": len(pack["rules"]),
917
+ "note": "" if pack["coverage_level"] == "baseline_review" else MAPPING_ONLY_NOTE,
918
+ })
919
+ seen.add(pack["framework"])
920
+ for entry in pack.get("mapping_only") or []:
921
+ framework = entry["framework"]
922
+ topics = sorted((entry.get("topics") or {}).keys())
923
+ coverage.append({
924
+ "framework": framework,
925
+ "coverage_level": "mapping_only",
926
+ "status": "mapping_only",
927
+ "pack_id": pack["pack_id"],
928
+ "pack_version": pack["pack_version"],
929
+ "rule_count": sum(len(ids) for ids in (entry.get("topics") or {}).values()),
930
+ "topics": topics,
931
+ "note": MAPPING_ONLY_NOTE,
932
+ })
933
+ seen.add(framework)
934
+ return coverage
935
+
936
+
937
+ def _unique_in_order(items):
938
+ """按首次出现顺序去重,用于覆盖摘要的框架清单。"""
939
+ seen = set()
940
+ unique = []
941
+ for item in items:
942
+ if item in seen:
943
+ continue
944
+ seen.add(item)
945
+ unique.append(item)
946
+ return unique
947
+
948
+
949
+ def review_text(text, packs, min_severity="low", include_safe=False, input_meta=None, generated_at=None):
950
+ """对文本执行确定性审查,返回稳定 JSON 契约结果。"""
951
+ if min_severity not in SEVERITY_ORDER:
952
+ raise UsageError("未知严重度阈值:%s" % min_severity)
953
+ normalized = normalize_text(text)
954
+ blocks = parse_structure(normalized)
955
+ checked = []
956
+ raw_findings = []
957
+ review_items = []
958
+
959
+ for pack in packs:
960
+ ctx = _MatchContext(normalized, blocks, pack)
961
+ for rule in pack["rules"]:
962
+ matched = match_rule(rule, normalized, blocks, pack)
963
+ if not matched["matched"]:
964
+ checked.append({
965
+ "rule_id": rule["rule_id"],
966
+ "pack_id": pack["pack_id"],
967
+ "status": "no_match",
968
+ })
969
+ continue
970
+ outcome = _evaluate_assert(rule, ctx, matched["hits"])
971
+ if outcome["outcome"] == "finding":
972
+ raw_findings.append({
973
+ "rule_id": rule["rule_id"],
974
+ "title": rule["title"],
975
+ "severity": rule["severity"],
976
+ "confidence": outcome["confidence"],
977
+ "framework_mapping": _framework_mapping(rule, pack),
978
+ "clause_types": list(rule.get("clause_types") or []),
979
+ "evidence": outcome["evidence"],
980
+ "evidence_total": outcome.get("evidence_total", len(outcome["evidence"])),
981
+ "rationale": rule["rationale"],
982
+ "remediation": rule["remediation"],
983
+ "source": rule["source"],
984
+ })
985
+ checked.append({
986
+ "rule_id": rule["rule_id"],
987
+ "pack_id": pack["pack_id"],
988
+ "status": "matched",
989
+ })
990
+ elif outcome["outcome"] == "review":
991
+ item = dict(outcome["review"])
992
+ item["pack_id"] = pack["pack_id"]
993
+ review_items.append(item)
994
+ checked.append({
995
+ "rule_id": rule["rule_id"],
996
+ "pack_id": pack["pack_id"],
997
+ "status": "matched_no_finding",
998
+ })
999
+ else:
1000
+ checked.append({
1001
+ "rule_id": rule["rule_id"],
1002
+ "pack_id": pack["pack_id"],
1003
+ "status": "matched_no_finding",
1004
+ })
1005
+
1006
+ raw_findings.sort(key=lambda item: (
1007
+ -_severity_rank(item["severity"]),
1008
+ item["rule_id"],
1009
+ item["evidence"][0]["start"] if item["evidence"] else 0,
1010
+ ))
1011
+ findings = []
1012
+ for index, item in enumerate(raw_findings, start=1):
1013
+ finding = {"finding_id": "F-%04d" % index}
1014
+ finding.update(item)
1015
+ total = finding["evidence_total"]
1016
+ finding["evidence_total"] = total
1017
+ if total > MAX_EVIDENCE_PER_FINDING:
1018
+ finding["evidence_note"] = "仅显示前 %d 条证据,共 %d 条。" % (MAX_EVIDENCE_PER_FINDING, total)
1019
+ findings.append(finding)
1020
+
1021
+ for finding in findings:
1022
+ if finding["confidence"] == "low":
1023
+ review_items.insert(0, {
1024
+ "kind": "low_confidence",
1025
+ "rule_id": finding["rule_id"],
1026
+ "pack_id": None,
1027
+ "finding_id": finding["finding_id"],
1028
+ "reason": "低置信结论,必须人工核验后才能采信。",
1029
+ })
1030
+
1031
+ by_severity = {severity: 0 for severity in SEVERITIES}
1032
+ by_confidence = {level: 0 for level in CONFIDENCE_LEVELS}
1033
+ for finding in findings:
1034
+ by_severity[finding["severity"]] += 1
1035
+ by_confidence[finding["confidence"]] += 1
1036
+
1037
+ emitted = [finding for finding in findings if _severity_rank(finding["severity"]) >= _severity_rank(min_severity)]
1038
+ summary = {
1039
+ "findings_total": len(emitted),
1040
+ "findings_total_all": len(findings),
1041
+ "by_severity": by_severity,
1042
+ "by_confidence": by_confidence,
1043
+ "rules_checked": len(checked),
1044
+ "rules_matched": len([item for item in checked if item["status"] != "no_match"]),
1045
+ "review_items": len(review_items),
1046
+ }
1047
+ if include_safe:
1048
+ summary["checked_rules"] = checked
1049
+
1050
+ packs_meta = []
1051
+ for pack in packs:
1052
+ packs_meta.append({
1053
+ "pack_id": pack["pack_id"],
1054
+ "framework": pack["framework"],
1055
+ "pack_version": pack["pack_version"],
1056
+ "coverage_level": pack["coverage_level"],
1057
+ "sha256": pack.get("_sha256", ""),
1058
+ "rule_count": len(pack["rules"]),
1059
+ })
1060
+
1061
+ stamp = generated_at or datetime.now(timezone.utc).strftime("%Y-%m-%dT%H:%M:%SZ")
1062
+ return {
1063
+ "schema_version": SCHEMA_VERSION,
1064
+ "tool": TOOL,
1065
+ "tool_version": VERSION,
1066
+ "generated_at": stamp,
1067
+ "input": _input_summary(text, input_meta),
1068
+ "rule_packs": packs_meta,
1069
+ "framework_coverage": _framework_coverage(packs),
1070
+ "summary": summary,
1071
+ "findings": emitted,
1072
+ "review_items": review_items,
1073
+ "disclaimer": DISCLAIMER,
1074
+ }
1075
+
1076
+
1077
+ # ---------------------------------------------------------------------------
1078
+ # 报告渲染(设计 §11)
1079
+ # ---------------------------------------------------------------------------
1080
+ def _finding_lines(finding):
1081
+ lines = [
1082
+ "### %s %s %s" % (finding["finding_id"], finding["rule_id"], finding["title"]),
1083
+ "",
1084
+ "- 严重度:%s | 置信度:%s" % (finding["severity"], finding["confidence"]),
1085
+ "- 框架映射:%s" % "、".join(finding["framework_mapping"]),
1086
+ ]
1087
+ if finding.get("clause_types"):
1088
+ lines.append("- 条款类型:%s" % "、".join(finding["clause_types"]))
1089
+ lines.append("- 证据:")
1090
+ for evidence in finding["evidence"]:
1091
+ if evidence["kind"] == "matched_span":
1092
+ lines.append(
1093
+ " - [matched_span] 第 %s 行 第 %s 列 `%s`"
1094
+ % (evidence["line"], evidence["column"], evidence["quote"].replace("\n", "\\n"))
1095
+ )
1096
+ else:
1097
+ lines.append(
1098
+ " - [%s] %s(第 %s 行 第 %s 列 至 第 %s 行 第 %s 列)"
1099
+ % (
1100
+ evidence["kind"], evidence.get("scope_label", ""),
1101
+ evidence["line"], evidence["column"],
1102
+ evidence.get("end_line", evidence["line"]),
1103
+ evidence.get("end_column", evidence["column"]),
1104
+ )
1105
+ )
1106
+ if finding.get("evidence_note"):
1107
+ lines.append("- 证据说明:%s" % finding["evidence_note"])
1108
+ lines.extend([
1109
+ "- 规则说明:%s" % finding["rationale"],
1110
+ "- 建议动作:%s" % finding["remediation"],
1111
+ "- 规则来源:%s" % finding["source"].get("citation", ""),
1112
+ "",
1113
+ ])
1114
+ return lines
1115
+
1116
+
1117
+ def render_markdown(result):
1118
+ """渲染人类可读报告,节顺序固定(设计 §11.1)。"""
1119
+ lines = ["# 合规条款审查报告", "", "生成时间:%s" % result["generated_at"], ""]
1120
+
1121
+ lines.append("## 1. 输入摘要")
1122
+ lines.append("")
1123
+ input_info = result["input"]
1124
+ lines.append("- 来源:%s" % (input_info.get("path") or input_info.get("kind")))
1125
+ lines.append("- 大小:%s 字节 | 行数:%s" % (input_info["bytes"], input_info["lines"]))
1126
+ lines.append("- SHA-256:%s" % input_info["sha256"])
1127
+ for pack in result["rule_packs"]:
1128
+ lines.append(
1129
+ "- 规则包:%s %s(%s,%s 条规则)"
1130
+ % (pack["pack_id"], pack["pack_version"], pack["coverage_level"], pack["rule_count"])
1131
+ )
1132
+ lines.append("")
1133
+
1134
+ lines.append("## 2. 框架覆盖摘要")
1135
+ lines.append("")
1136
+ lines.append("| 规则包 | 框架 | 覆盖级别 | 状态 | 规则数 |")
1137
+ lines.append("| --- | --- | --- | --- | --- |")
1138
+ for entry in result["framework_coverage"]:
1139
+ lines.append(
1140
+ "| %s | %s | %s | %s | %s |"
1141
+ % (
1142
+ entry.get("pack_id", "-"),
1143
+ entry["framework"],
1144
+ entry["coverage_level"],
1145
+ entry["status"],
1146
+ entry["rule_count"],
1147
+ )
1148
+ )
1149
+ lines.append("")
1150
+ mapping_only = _unique_in_order(
1151
+ entry["framework"] for entry in result["framework_coverage"]
1152
+ if entry["coverage_level"] == "mapping_only"
1153
+ )
1154
+ if mapping_only:
1155
+ lines.append(
1156
+ "说明:%s 为 mapping_only(%s)。"
1157
+ % ("、".join(mapping_only), MAPPING_ONLY_NOTE.strip("。"))
1158
+ )
1159
+ lines.append("")
1160
+
1161
+ lines.append("## 3. 风险汇总")
1162
+ lines.append("")
1163
+ summary = result["summary"]
1164
+ if result["findings"]:
1165
+ lines.append("- 本次输出 %s 条发现(全部 %s 条)" % (summary["findings_total"], summary["findings_total_all"]))
1166
+ for severity in reversed(SEVERITIES):
1167
+ if summary["by_severity"].get(severity):
1168
+ lines.append("- %s:%s 条" % (severity, summary["by_severity"][severity]))
1169
+ else:
1170
+ lines.append("- 本报告未发现达到输出阈值的风险项。")
1171
+ lines.append("- 已检查规则:%s 条 | 命中规则:%s 条" % (summary["rules_checked"], summary["rules_matched"]))
1172
+ lines.append("")
1173
+
1174
+ lines.append("## 4. 逐条发现")
1175
+ lines.append("")
1176
+ if result["findings"]:
1177
+ for finding in result["findings"]:
1178
+ lines.extend(_finding_lines(finding))
1179
+ else:
1180
+ lines.append("(无达到输出阈值的发现。)")
1181
+ lines.append("")
1182
+
1183
+ lines.append("## 5. 未覆盖范围")
1184
+ lines.append("")
1185
+ if mapping_only:
1186
+ lines.append("- mapping_only 框架(仅主题映射):%s" % "、".join(mapping_only))
1187
+ lines.append("- v0.1 不直接解析 PDF / docx,请先用文档适配器转成文本或 Markdown。")
1188
+ checked_rules = summary.get("checked_rules") or []
1189
+ unmatched = [item["rule_id"] for item in checked_rules if item["status"] == "no_match"]
1190
+ if checked_rules:
1191
+ lines.append("- 本次已检查未命中的规则:%s" % ("、".join(unmatched) if unmatched else "无"))
1192
+ lines.append("")
1193
+
1194
+ lines.append("## 6. 人工复核清单")
1195
+ lines.append("")
1196
+ if result["review_items"]:
1197
+ for item in result["review_items"]:
1198
+ lines.append(
1199
+ "- [%s] %s:%s"
1200
+ % (item["kind"], item.get("rule_id") or item.get("finding_id") or "-", item["reason"])
1201
+ )
1202
+ else:
1203
+ lines.append("(无需人工复核项。)")
1204
+ lines.append("")
1205
+
1206
+ lines.append("## 7. 免责声明")
1207
+ lines.append("")
1208
+ lines.append(result["disclaimer"])
1209
+ lines.append("")
1210
+ return "\n".join(lines)
1211
+
1212
+
1213
+ def render_json(result):
1214
+ return json.dumps(result, ensure_ascii=False, indent=2)
1215
+
1216
+
1217
+ # ---------------------------------------------------------------------------
1218
+ # CLI(设计 §6)
1219
+ # ---------------------------------------------------------------------------
1220
+ class _ArgumentParser(argparse.ArgumentParser):
1221
+ def error(self, message):
1222
+ raise UsageError(message)
1223
+
1224
+
1225
+ def _read_input(args):
1226
+ if args.stdin:
1227
+ text = sys.stdin.read()
1228
+ meta = {"kind": "stdin", "path": None}
1229
+ else:
1230
+ path = args.input
1231
+ if not os.path.exists(path):
1232
+ raise InputError("输入文件不存在:%s" % path)
1233
+ if os.path.isdir(path):
1234
+ raise InputError("输入路径不是文件:%s" % path)
1235
+ suffix = os.path.splitext(path)[1].lower()
1236
+ if suffix in UNSUPPORTED_SUFFIXES:
1237
+ raise InputError(
1238
+ "v0.1 不支持直接读取 %s;请先用文档适配器转成文本或 Markdown 再审查。" % suffix
1239
+ )
1240
+ size = os.path.getsize(path)
1241
+ if size > MAX_INPUT_BYTES:
1242
+ raise InputError(
1243
+ "输入超过 2 MiB 上限(%s 字节);请分批处理或先精简文本。" % size
1244
+ )
1245
+ try:
1246
+ with open(path, "rb") as handle:
1247
+ raw = handle.read()
1248
+ except OSError as exc:
1249
+ raise InputError("输入文件不可读:%s" % exc)
1250
+ try:
1251
+ text = raw.decode("utf-8-sig")
1252
+ except UnicodeDecodeError:
1253
+ raise InputError("输入必须是 UTF-8 文本;请先转码(例如 GB18030 → UTF-8)。")
1254
+ meta = {"kind": "file", "path": os.path.abspath(path)}
1255
+ if not text.strip():
1256
+ raise InputError("输入内容为空,未执行审查。")
1257
+ return text, meta
1258
+
1259
+
1260
+ def _resolve_output_path(out_path, rules_dir, input_path):
1261
+ out_abs = os.path.realpath(os.path.abspath(out_path))
1262
+ rules_abs = os.path.realpath(os.path.abspath(rules_dir))
1263
+ if out_abs == rules_abs or out_abs.startswith(rules_abs + os.sep):
1264
+ raise InputError("输出路径不得位于规则包目录内:%s" % out_path)
1265
+ if input_path:
1266
+ input_abs = os.path.realpath(os.path.abspath(input_path))
1267
+ if out_abs == input_abs:
1268
+ raise InputError("输出路径不得与输入文件相同,避免覆盖原文。")
1269
+ parent = os.path.dirname(out_abs)
1270
+ if parent and not os.path.isdir(parent):
1271
+ raise InputError("输出目录不存在:%s" % parent)
1272
+ return out_abs
1273
+
1274
+
1275
+ def _write_output(path, content):
1276
+ temporary = path + ".tmp-yotta-compliance"
1277
+ try:
1278
+ with open(temporary, "w", encoding="utf-8", newline="\n") as handle:
1279
+ handle.write(content)
1280
+ os.replace(temporary, path)
1281
+ except OSError as exc:
1282
+ try:
1283
+ if os.path.exists(temporary):
1284
+ os.remove(temporary)
1285
+ except OSError:
1286
+ pass
1287
+ raise InputError("报告写入失败:%s" % exc)
1288
+
1289
+
1290
+ def _gate_hit(summary, gate):
1291
+ if gate == "off":
1292
+ return False
1293
+ threshold = SEVERITY_ORDER[gate]
1294
+ for severity, count in summary["by_severity"].items():
1295
+ if count and SEVERITY_ORDER[severity] >= threshold:
1296
+ return True
1297
+ return False
1298
+
1299
+
1300
+ def _cmd_review(args):
1301
+ if args.input and args.stdin:
1302
+ raise UsageError("--input 与 --stdin 互斥,只能选择一种输入方式。")
1303
+ if not args.input and not args.stdin:
1304
+ raise UsageError("必须提供 --input <文件> 或 --stdin。")
1305
+ if args.locale != "zh-CN":
1306
+ raise UsageError("v0.1 仅支持 --locale zh-CN。")
1307
+ if args.gate not in (("off",) + SEVERITIES):
1308
+ raise UsageError("--gate 非法(允许:off, %s)" % ", ".join(SEVERITIES))
1309
+ if args.min_severity not in SEVERITY_ORDER:
1310
+ raise UsageError("--min-severity 非法(允许:%s)" % ", ".join(SEVERITIES))
1311
+ frameworks = None
1312
+ if args.frameworks:
1313
+ frameworks = [item.strip() for item in args.frameworks.split(",") if item.strip()]
1314
+ packs = load_rule_packs(args.rules_dir, frameworks)
1315
+ text, meta = _read_input(args)
1316
+ result = review_text(
1317
+ text, packs,
1318
+ min_severity=args.min_severity,
1319
+ include_safe=args.include_safe,
1320
+ input_meta=meta,
1321
+ )
1322
+ if args.format == "json":
1323
+ rendered = render_json(result)
1324
+ else:
1325
+ rendered = render_markdown(result)
1326
+ if args.out:
1327
+ target = _resolve_output_path(args.out, args.rules_dir, meta.get("path"))
1328
+ _write_output(target, rendered if rendered.endswith("\n") else rendered + "\n")
1329
+ sys.stderr.write("已写入:%s\n" % target)
1330
+ else:
1331
+ sys.stdout.write(rendered if rendered.endswith("\n") else rendered + "\n")
1332
+ return 1 if _gate_hit(result["summary"], args.gate) else 0
1333
+
1334
+
1335
+ def _cmd_rules_list(args):
1336
+ frameworks = None
1337
+ if args.framework:
1338
+ frameworks = [item.strip() for item in args.framework.split(",") if item.strip()]
1339
+ packs = load_rule_packs(args.rules_dir, frameworks)
1340
+ for pack in packs:
1341
+ sys.stdout.write("# %s(%s,%s)\n" % (pack["pack_id"], pack["framework"], pack["coverage_level"]))
1342
+ for rule in pack["rules"]:
1343
+ sys.stdout.write("- %s %s [%s]\n" % (rule["rule_id"], rule["title"], rule["severity"]))
1344
+ return 0
1345
+
1346
+
1347
+ def _cmd_rules_show(args):
1348
+ packs = load_rule_packs(args.rules_dir)
1349
+ for pack in packs:
1350
+ for rule in pack["rules"]:
1351
+ if rule["rule_id"] != args.rule_id:
1352
+ continue
1353
+ sys.stdout.write("%s %s\n" % (rule["rule_id"], rule["title"]))
1354
+ sys.stdout.write("严重度:%s\n" % rule["severity"])
1355
+ sys.stdout.write("规则说明:%s\n" % rule["rationale"])
1356
+ sys.stdout.write("建议动作:%s\n" % rule["remediation"])
1357
+ sys.stdout.write("规则来源:%s\n" % rule["source"].get("citation", ""))
1358
+ if rule["source"].get("url"):
1359
+ sys.stdout.write("来源链接:%s\n" % rule["source"]["url"])
1360
+ sys.stdout.write("测试用例:%s\n" % "、".join(rule["test_ids"]))
1361
+ return 0
1362
+ raise UsageError("未找到规则:%s" % args.rule_id)
1363
+
1364
+
1365
+ def _cmd_rules_validate(args):
1366
+ pack = load_pack(args.pack)
1367
+ sys.stdout.write(
1368
+ "规则包校验通过:%s(%s,%s 条规则,sha256=%s)\n"
1369
+ % (pack["pack_id"], pack["framework"], len(pack["rules"]), pack["_sha256"][:12])
1370
+ )
1371
+ return 0
1372
+
1373
+
1374
+ def build_parser():
1375
+ parser = _ArgumentParser(prog=TOOL, description="%s(%s):确定性合规条款审查内核" % (TOOL_CN, TOOL))
1376
+ parser.add_argument("--version", action="version", version=VERSION)
1377
+ subparsers = parser.add_subparsers(dest="command")
1378
+
1379
+ review = subparsers.add_parser("review", help="审查文本 / Markdown")
1380
+ review.add_argument("--input", help="输入文件(UTF-8 .txt / .md)")
1381
+ review.add_argument("--stdin", action="store_true", help="从标准输入读取")
1382
+ review.add_argument("--frameworks", help="框架列表,逗号分隔(默认全部已装载规则包)")
1383
+ review.add_argument("--rules-dir", default=DEFAULT_RULES_DIR, help="规则包目录")
1384
+ review.add_argument("--format", choices=("md", "json"), default="md")
1385
+ review.add_argument("--out", help="报告输出路径(默认写 stdout)")
1386
+ review.add_argument("--gate", default="off", help="CI 闸门阈值:off/low/medium/high/critical")
1387
+ review.add_argument("--min-severity", default="low", help="输出最低严重度(默认 low)")
1388
+ review.add_argument("--include-safe", action="store_true", help="输出已检查未命中的规则清单")
1389
+ review.add_argument("--locale", default="zh-CN", help="报告语言(v0.1 仅 zh-CN)")
1390
+ review.set_defaults(handler=_cmd_review)
1391
+
1392
+ rules = subparsers.add_parser("rules", help="规则包操作")
1393
+ rules_sub = rules.add_subparsers(dest="rules_command")
1394
+
1395
+ rules_list = rules_sub.add_parser("list", help="列出规则")
1396
+ rules_list.add_argument("--rules-dir", default=DEFAULT_RULES_DIR)
1397
+ rules_list.add_argument("--framework", help="按框架过滤(逗号分隔)")
1398
+ rules_list.set_defaults(handler=_cmd_rules_list)
1399
+
1400
+ rules_show = rules_sub.add_parser("show", help="查看单条规则")
1401
+ rules_show.add_argument("rule_id")
1402
+ rules_show.add_argument("--rules-dir", default=DEFAULT_RULES_DIR)
1403
+ rules_show.set_defaults(handler=_cmd_rules_show)
1404
+
1405
+ rules_validate = rules_sub.add_parser("validate", help="校验规则包")
1406
+ rules_validate.add_argument("--pack", required=True, help="规则包 JSON 路径")
1407
+ rules_validate.set_defaults(handler=_cmd_rules_validate)
1408
+
1409
+ return parser
1410
+
1411
+
1412
+ def main(argv=None):
1413
+ parser = build_parser()
1414
+ try:
1415
+ args = parser.parse_args(argv)
1416
+ handler = getattr(args, "handler", None)
1417
+ if handler is None:
1418
+ raise UsageError("请指定子命令:review / rules")
1419
+ return handler(args)
1420
+ except ComplianceError as exc:
1421
+ sys.stderr.write("错误:%s\n" % exc)
1422
+ return exc.exit_code
1423
+
1424
+
1425
+ if __name__ == "__main__":
1426
+ sys.exit(main())