@furongjun1999/dsh-memory 0.4.7 → 0.4.8

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (172) hide show
  1. package/README.md +92 -37
  2. package/codebuddy/CODEBUDDY.md +195 -185
  3. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -0
  4. package/docs/README.md +111 -0
  5. package/docs/discipline/harnesses.yaml +215 -152
  6. package/docs/discipline/templates/full.md.tmpl +60 -58
  7. package/docs/eval//347/254/254/344/270/211/346/226/271/351/252/214/350/257/201/346/212/245/345/221/212_LoCoMo_/347/201/265/346/236/242_.md +127 -0
  8. package/docs/eval//347/254/254/344/270/211/346/226/271/351/252/214/350/257/201/346/212/245/345/221/212_LoCoMo_/347/201/265/346/236/242_.png +0 -0
  9. package/docs/eval//347/254/254/344/270/211/346/226/271/351/252/214/350/257/201/346/212/245/345/221/212_/347/201/265/346/236/242_vs_dejavu_/347/273/237/344/270/200/350/257/204/345/210/206_v7.md +202 -0
  10. package/docs/experiments/m4-role-probe/census.py +86 -0
  11. package/docs/experiments/m4-role-probe/probe.py +89 -0
  12. package/docs/experiments/mapped_confidence/bench_mapped_conf.py +216 -0
  13. package/docs/experiments/mapped_confidence/post_check.py +75 -0
  14. package/docs/experiments/mapped_confidence/repro_thirdparty_tol.py +83 -0
  15. package/docs/experiments/mapped_confidence/result_mapped_conf.json +215 -0
  16. package/docs/{ → hive/}/350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +14 -2
  17. package/docs/{README → mdcg/README}/350/257/246/347/273/206/347/211/210_v0.4.5.md +9 -9
  18. package/docs/{ → mdcg/}/344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md +1 -1
  19. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +261 -0
  20. package/docs/{ → mdcg/}/345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +1 -1
  21. package/docs/mdcg//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.4.md +148 -0
  22. package/docs/{ → mdcg/}/346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +2 -2
  23. package/docs/{ → mdcg/}/347/201/265/346/236/242/344/270/211/345/261/202/346/213/206/345/210/206/350/247/204/345/210/222_v0.1.md +181 -181
  24. package/docs/{ → mdcg/}/347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +2 -2
  25. package/docs/{ → mdcg/}/350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +1 -1
  26. package/docs/{ → mdcg/}/350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +4 -4
  27. package/docs/{ → mdcg/}/350/256/260/345/277/206/346/223/215/344/275/234/347/263/273/347/273/237_MdCGOS/344/270/216MCP/346/216/245/345/205/245_v0.1.md +2 -2
  28. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -0
  29. package/docs/{ → swarm/}/350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +3 -3
  30. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -0
  31. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -0
  32. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -0
  33. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -0
  34. package/docs/{ → theory/}/347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +3 -3
  35. package/docs/{ → theory/}/347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +1 -1
  36. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -0
  37. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -0
  38. package/docs//345/255/230/347/256/227/344/270/200/344/275/223/346/236/266/346/236/204/345/255/246/344/271/240_/345/244/226/351/203/250/347/220/206/350/256/272/345/257/271/347/205/247/344/270/216/350/267/257/347/272/277/344/272/244/346/216/245_20260916.md +98 -0
  39. package/docs//345/255/230/347/256/227/344/270/200/344/275/223/347/245/236/347/273/217/347/275/221/347/273/234/346/236/266/346/236/204_/346/200/273/347/272/262/344/270/216/344/272/244/346/216/245_20260916.md +109 -0
  40. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +34 -6
  41. package/docs//347/231/275/347/256/261/345/214/226/347/272/262/351/242/206_/344/270/211/351/241/271/347/233/256/345/255/246/344/271/240/346/200/273/346/236/266/346/236/204/344/270/216Ghidra/344/272/244/346/216/245_20260916.md +104 -0
  42. package/docs//350/256/260/345/277/206/350/257/204/345/256/241/347/263/273/347/273/237_/347/253/213/351/241/271/350/256/276/350/256/241/344/270/216/346/226/275/345/267/245/344/272/244/346/216/245_20260915.md +183 -0
  43. package/dsh/README.md +1 -1
  44. package/lib/hooks.d.ts +5 -0
  45. package/lib/hooks.js +10 -1
  46. package/lib/lib/prompt_safety.d.ts +48 -0
  47. package/lib/lib/prompt_safety.js +62 -0
  48. package/md_cg/audit.py +76 -2
  49. package/md_cg/bench_membench.py +10 -5
  50. package/md_cg/bench_progressive.py +48 -9
  51. package/md_cg/branches.py +275 -0
  52. package/md_cg/ccgc.py +884 -0
  53. package/md_cg/conformance.py +662 -0
  54. package/md_cg/consistency.py +1 -1
  55. package/md_cg/corpus.py +1 -1
  56. package/md_cg/eval_common.py +6 -5
  57. package/md_cg/forgetting.py +11 -3
  58. package/md_cg/fsutil.py +63 -1
  59. package/md_cg/identity.py +2 -2
  60. package/md_cg/insight.py +2 -2
  61. package/md_cg/lifecycle.py +262 -0
  62. package/md_cg/mcp_server.py +547 -140
  63. package/md_cg/mdcg.py +206 -18
  64. package/md_cg/mdcos.py +512 -50
  65. package/md_cg/metacognition.py +2 -2
  66. package/md_cg/mreview/__init__.py +25 -0
  67. package/md_cg/mreview/__main__.py +107 -0
  68. package/md_cg/mreview/bundle.py +170 -0
  69. package/md_cg/mreview/candidates.py +253 -0
  70. package/md_cg/mreview/govern.py +674 -0
  71. package/md_cg/mreview/locate.py +905 -0
  72. package/md_cg/mreview/pipeline.py +701 -0
  73. package/md_cg/mreview/rules/duplication.json +21 -0
  74. package/md_cg/mreview/rules/field_coverage.json +54 -0
  75. package/md_cg/mreview/rules/source_license.json +21 -0
  76. package/md_cg/mreview/rules/template_flow.json +21 -0
  77. package/md_cg/mreview/ruleset.py +238 -0
  78. package/md_cg/nodefile.py +38 -0
  79. package/md_cg/predict.py +24 -6
  80. package/md_cg/protect.py +4 -4
  81. package/md_cg/refindex.py +41 -2
  82. package/md_cg/scrub.py +5 -1
  83. package/md_cg/self_state.py +1 -1
  84. package/md_cg/semantic/en_normalizer.py +85 -25
  85. package/md_cg/sustain.py +18 -0
  86. package/md_cg/tasks.py +447 -0
  87. package/md_cg/test_action_derive.py +203 -0
  88. package/md_cg/test_audit_rotate.py +270 -0
  89. package/md_cg/test_branches.py +249 -0
  90. package/md_cg/test_ccg_perturb.py +1 -1
  91. package/md_cg/test_ccgc.py +423 -0
  92. package/md_cg/test_conformance.py +343 -0
  93. package/md_cg/test_en_pipeline.py +24 -0
  94. package/md_cg/test_health_scale.py +173 -0
  95. package/md_cg/test_identity_attribution.py +6 -0
  96. package/md_cg/test_index_durability.py +224 -0
  97. package/md_cg/test_lifecycle.py +309 -0
  98. package/md_cg/test_mr_m1.py +679 -0
  99. package/md_cg/test_mr_m2.py +587 -0
  100. package/md_cg/test_mr_m3.py +703 -0
  101. package/md_cg/test_mr_m4.py +485 -0
  102. package/md_cg/test_p1.py +1 -1
  103. package/md_cg/test_p11_consistency.py +1 -1
  104. package/md_cg/test_p12_metacognition.py +2 -2
  105. package/md_cg/test_p16_self_state.py +1 -1
  106. package/md_cg/test_p26_refindex.py +1 -1
  107. package/md_cg/test_p27_docindex.py +22 -2
  108. package/md_cg/test_p28_refcheck.py +1 -1
  109. package/md_cg/test_p29_session_ingest_export.py +1 -1
  110. package/md_cg/test_p2_mcp.py +7 -1
  111. package/md_cg/test_p38_contextualize.py +1 -1
  112. package/md_cg/test_p39_vision_evidence.py +1 -1
  113. package/md_cg/test_p40_refine_worklist.py +1 -1
  114. package/md_cg/test_p41_evolve_patrol.py +1 -1
  115. package/md_cg/test_p42_provenance.py +1 -1
  116. package/md_cg/test_p43_pooling.py +41 -13
  117. package/md_cg/test_read_clip.py +137 -0
  118. package/md_cg/test_review_conformance.py +310 -0
  119. package/md_cg/test_tasks.py +409 -0
  120. package/md_cg/test_tool_face.py +189 -0
  121. package/md_cg/test_twophase.py +286 -0
  122. package/md_cg/test_writepipe.py +210 -0
  123. package/md_cg/theory.py +1 -1
  124. package/md_cg/tokens.py +95 -5
  125. package/md_cg/tool_face.py +250 -0
  126. package/md_cg/twophase.py +221 -0
  127. package/md_cg/units.py +546 -0
  128. package/md_cg/vision_evidence.py +1 -1
  129. package/md_cg/weights.py +1 -1
  130. package/md_cg/whitebox.py +1 -1
  131. package/md_cg/whitebox_kb/data/verify_cache.json +28 -0
  132. package/md_cg/whitebox_kb/data/verify_savings.jsonl +46 -0
  133. package/md_cg/whitebox_kb/wisdom/audit_log/chain_heat.json +10 -10
  134. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-shm +0 -0
  135. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-wal +0 -0
  136. package/md_cg/writelimit.py +19 -3
  137. package/md_cg/writepipe.py +372 -0
  138. package/package.json +1 -1
  139. package/src/hooks.ts +10 -1
  140. package/src/lib/prompt_safety.ts +62 -0
  141. package/zcode/AGENTS.md +195 -185
  142. package/docs//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +0 -221
  143. /package/docs/{AGI → eval/AGI}/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md" +0 -0
  144. /package/docs/{AGI → eval/AGI}/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md" +0 -0
  145. /package/docs/{ → eval/}/345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md" +0 -0
  146. /package/docs/{ → eval/}/346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md" +0 -0
  147. /package/docs/{guardrail-charter.md → mdcg/guardrail-charter.md} +0 -0
  148. /package/docs/{lingshu_tutorial.html → mdcg/lingshu_tutorial.html} +0 -0
  149. /package/docs/{memory-assessment.html → mdcg/memory-assessment.html} +0 -0
  150. /package/docs/{memory_score.html → mdcg/memory_score.html} +0 -0
  151. /package/docs/{memory_score.png → mdcg/memory_score.png} +0 -0
  152. /package/docs/{release_v0.3.0.md → mdcg/release_v0.3.0.md} +0 -0
  153. /package/docs/{release_v0.4.5.md → mdcg/release_v0.4.5.md} +0 -0
  154. /package/docs/{tool_table_v0.3.0.md → mdcg/tool_table_v0.3.0.md} +0 -0
  155. /package/docs/{ → mdcg/}/344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md" +0 -0
  156. /package/docs/{ → mdcg/}/347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md" +0 -0
  157. /package/docs/{ → mdcg/}/347/201/265/346/236/242MCP/345/267/245/345/205/267/346/200/273/350/241/250_v3.4.md" +0 -0
  158. /package/docs/{ → mdcg/}/347/201/265/346/236/242_/350/207/252/346/210/221/345/261/202/345/256/232/344/271/211.md" +0 -0
  159. /package/docs/{ → mdcg/}/350/256/244/347/237/245/345/233/276_MD/347/233/256/345/275/225/346/226/271/346/241/210_v0.1.md" +0 -0
  160. /package/docs/{ → mdcg/}/350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md" +0 -0
  161. /package/docs/{ → mdcg/}/350/256/244/347/237/245/345/233/276/344/275/234/344/270/272/350/256/260/345/277/206/346/223/215/344/275/234/347/263/273/347/273/237_/350/257/204/344/274/260/344/270/216/350/267/257/347/272/277_v0.1.md" +0 -0
  162. /package/docs/{ → plans/}/345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md" +0 -0
  163. /package/docs/{ → plans/}/347/237/245/350/257/206/345/233/276/350/260/261/351/241/271/347/233/256_/344/272/244/346/216/245/346/226/207/346/241/243_20260914.md" +0 -0
  164. /package/docs/{ → swarm/}/350/234/202/347/276/244/345/244/232/346/231/272/350/203/275/344/275/223_/344/275/277/347/224/250/346/212/245/345/221/212_20260913.md" +0 -0
  165. /package/docs/{ → swarm/}/350/234/202/347/276/244/345/244/232/346/231/272/350/203/275/344/275/223_/345/212/237/350/203/275/350/257/264/346/230/216_v0.6.md" +0 -0
  166. /package/docs/{ → swarm/}/350/234/202/347/276/244/347/233/262/345/214/272/346/240/207/350/256/260_v1.0.md" +0 -0
  167. /package/docs/{ → theory/}/344/277/241/346/201/257/345/267/256/344/270/272/344/273/200/344/271/210/345/277/205/347/204/266/345/255/230/345/234/250/344/270/224/350/207/252/347/204/266/346/211/251/345/244/247.md" +0 -0
  168. /package/docs/{ → theory/}/346/231/272/350/203/275/347/232/204/345/205/254/347/220/206/345/214/226/345/237/272/347/237/263.md" +0 -0
  169. /package/docs/{ → theory/}/346/231/272/350/203/275/347/232/204/350/256/244/347/237/245/350/277/207/347/250/213.md" +0 -0
  170. /package/docs/{ → theory/}/346/231/272/350/203/275/350/256/2723.4.md" +0 -0
  171. /package/docs/{ → theory/}/347/231/275/347/256/261/346/231/272/350/203/275/346/230/257/344/273/200/344/271/210/357/274/237.md" +0 -0
  172. /package/docs/{ → theory/}/347/231/275/347/256/261/346/231/272/350/203/275/347/263/273/345/210/227/302/267/347/254/254/344/272/224/347/257/207/357/274/232/350/256/251AI/347/234/237/346/255/243/350/243/205/344/270/212/350/256/260/345/277/206.md" +0 -0
@@ -0,0 +1,703 @@
1
+ # -*- coding: utf-8 -*-
2
+ """M3 定位自测(D1 字段级定位):六类定位器 + 契约四键 + 批量/包 + 零写入。
3
+
4
+ 真源:docs/记忆评审系统_立项设计与施工交接_20260915.md §5 D1
5
+ `locate(node_id, issue_hint) -> [{field, span, issue_kind, evidence}]`
6
+
7
+ 纪律(同 test_mr_m1/m2):
8
+ · **自备数据源**——合成节点 + tempfile 合成库,零依赖真源库(外部 clone 全绿)。
9
+ · **零写入实锤**——全部相位跑完后认知图指纹逐字节不变(M3 是只读模块)。
10
+ · **确定性**——同一输入两次调用逐字节一致;`now` 显式传入,不靠墙钟。
11
+ · **不猜**——语义级矛盾归 blindspot(`contradiction_semantic`),不编造字符区间。
12
+
13
+ 覆盖:A 基础工具(纯函数) B 问题面定位器(含字段层门限、指纹不一致成因)+ 观测面
14
+ (observation_aged:观测时刻不是失效声明,**不进告警面**)
15
+ C 提示过滤与 blindspot(含 stale 的**依赖存在性**判据:载体消失/漂移)
16
+ D 契约与确定性 E 批量与包 F 零写入
17
+ 运行:python -m md_cg.test_mr_m3
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import hashlib
22
+ import json
23
+ import os
24
+ import shutil
25
+ import sys
26
+ import tempfile
27
+
28
+ from . import codeindex as CI
29
+ from . import conformance as CF
30
+ from . import nodefile as NF
31
+ from . import writelimit as WL
32
+ from .mreview import locate as LC
33
+
34
+ PASS = FAIL = SKIP = 0
35
+ FAILS = []
36
+
37
+
38
+ def ok(cond, label):
39
+ global PASS, FAIL
40
+ if cond:
41
+ PASS += 1
42
+ else:
43
+ FAIL += 1
44
+ FAILS.append(label)
45
+ print(" FAIL %s" % label)
46
+
47
+
48
+ def skip(label):
49
+ global SKIP
50
+ SKIP += 1
51
+ print(" SKIP %s" % label)
52
+
53
+
54
+ # ---------------------------- 合成库 ----------------------------
55
+
56
+ def ccg(fn="示例节点", *, 生效="载体/位置:本地仓;时间:全时窗(任意时刻成立);方法:test;约束:无",
57
+ sub="a/b", exe="python -m md_cg.demo", ver="test", neg="无", extra=""):
58
+ """六要素齐全的正文(防 `_loc_missing_field` 的 CCG 缺行噪声干扰其它判据)。"""
59
+ return ("# 功能名:%s\n# 生效条件:%s\n# 子功能:%s\n# 执行:%s\n"
60
+ "# 验证方式:%s\n# 不适用条件:%s\n%s" % (fn, 生效, sub, exe, ver, neg, extra))
61
+
62
+
63
+ def full_cs(*, pos="本地仓", tw=(0.0, 9999999999.0), tool="test", con="无"):
64
+ return {"observation_position": pos, "time_window": list(tw),
65
+ "observation_tool": tool, "existence_constraint": con}
66
+
67
+
68
+ def mkroot(root, nodes):
69
+ """合成认知图根:`_index.json` + 节点盘文件(文件文本由 NF.dumps 生成)。"""
70
+ os.makedirs(root, exist_ok=True)
71
+ idx = {}
72
+ for nid, spec in nodes.items():
73
+ fm = dict(spec.get("fm") or {})
74
+ fm.setdefault("id", nid)
75
+ content = spec.get("content") or ""
76
+ rel = spec.get("path") or ("knowledge/%s.md" % nid)
77
+ p = os.path.join(root, rel)
78
+ os.makedirs(os.path.dirname(p), exist_ok=True)
79
+ with open(p, "w", encoding="utf-8") as f:
80
+ f.write(NF.dumps(fm, content))
81
+ meta = dict(spec.get("meta") or {})
82
+ meta.setdefault("path", rel)
83
+ meta.setdefault("layer", spec.get("layer") or "knowledge")
84
+ meta.setdefault("content_hash", NF.content_hash(content))
85
+ meta.update({"id": nid, "path": rel})
86
+ meta["path"] = rel
87
+ idx[nid] = meta
88
+ with open(os.path.join(root, CF.INDEX_FILE), "w", encoding="utf-8") as f:
89
+ json.dump({"nodes": idx}, f, ensure_ascii=False)
90
+ return root
91
+
92
+
93
+ def snapshot(root):
94
+ out = {}
95
+ for dp, _dns, fns in os.walk(root):
96
+ for fn in fns:
97
+ p = os.path.join(dp, fn)
98
+ with open(p, "rb") as f:
99
+ out[os.path.relpath(p, root)] = hashlib.md5(f.read()).hexdigest()
100
+ return out
101
+
102
+
103
+ def kinds_of(hits):
104
+ return sorted({h["issue_kind"] for h in hits})
105
+
106
+
107
+ def fields_of(hits, kind=None):
108
+ return sorted({h["field"] for h in hits if kind is None or h["issue_kind"] == kind})
109
+
110
+
111
+ NOW = 1789000000.0 # 固定「当前时间」(2026-09 量级),不靠墙钟
112
+ EXPIRED = (1000.0, 2000.0)
113
+
114
+
115
+ # ---------------------------- A 组:基础工具 ----------------------------
116
+
117
+ def phase_a(tmp):
118
+ print("[A] 基础工具(纯函数)")
119
+
120
+ c = "第一句。第二句!第三句?第四句;"
121
+ sp = LC.sentence_spans(c)
122
+ ok(len(sp) == 4 and [i for i, _s, _e, _t in sp] == [0, 1, 2, 3],
123
+ "A1 sentence_spans 句索引连续(%s)" % [i for i, _s, _e, _t in sp])
124
+ ok(all(c[s:e] == t for _i, s, e, t in sp), "A2 span 与原文切片逐字对应")
125
+ ok(LC.sentence_spans("") == [] and LC.sentence_spans(" ") == [],
126
+ "A3 空/纯空白正文零句(不编号空句)")
127
+ ok([t for _i, _s, _e, t in LC.sentence_spans("甲。\n乙。")] == ["甲。", "乙。"],
128
+ "A4 换行是句尾且空句不编号(分隔符不残留在句首)")
129
+
130
+ body = ccg(extra="尾句。")
131
+ ms = LC.mark_spans(body)
132
+ ok(set(ms) == set(NF.CCG_MARKS), "A5 mark_spans 六要素全提(实得 %s)" % sorted(ms))
133
+ ok(all(body[s:e].startswith("#") for s, e in ms.values()),
134
+ "A6 要素行 span 覆盖整行")
135
+ ok(LC.mark_spans(ccg() + "# 功能名:第二个\n").get("功能名")
136
+ == LC.mark_spans(ccg()).get("功能名"), "A7 同要素取首次出现(确定性)")
137
+
138
+ text = NF.dumps({"id": "n1", "path": "knowledge/n1.md", "tags": []}, ccg())
139
+ ks = LC.key_line_spans(text)
140
+ ok(ks.get("id", (None, None))[0] == 2, "A8 key_line_spans 行号 1-based(实得 %s)"
141
+ % (ks.get("id") or (None,))[0])
142
+ ok("功能名" not in ks and "---" not in ks,
143
+ "A9 正文 `#` 行与 `---` 分隔线都排除在 frontmatter 之外")
144
+ ok(ks.get("id") and text[ks["id"][1][0]:ks["id"][1][1]].startswith('id:'),
145
+ "A10 键行 span 切片以键名开头")
146
+
147
+ root = mkroot(os.path.join(tmp, "a"), {
148
+ "n1": {"content": ccg(), "meta": {"layer": "knowledge", "tags": ["a"]}},
149
+ "n2": {"content": "短", "path": "", "meta": {"layer": "knowledge"}},
150
+ })
151
+ nd = LC.load_node("n1", root)
152
+ ok(nd and nd["meta"].get("layer") == "knowledge" and "# 功能名:" in (nd["content"] or ""),
153
+ "A11 load_node 取索引 meta + 文件正文")
154
+ ok(nd and nd["fm"].get("id") == "n1", "A12 loads 解析出的 fm 是文件真源")
155
+ ok(LC.load_node("ghost", root) is None, "A13 索引无此节点 → None(不猜路径)")
156
+ ok(LC.load_node("n1", root, index={"nodes": {"n1": {"path": "knowledge/n1.md"}}})
157
+ is not None, "A14 index 可显式注入(不读盘 index)")
158
+ ok(LC._index(root).get("n1") is not None, "A15 _index 兼容 {nodes:…} 形态")
159
+ ok(LC._index(root, index={"n1": {"path": "x"}}) == {"n1": {"path": "x"}},
160
+ "A16 裸 dict 索引原样透传")
161
+
162
+ ok(LC._snippet("甲" * 200, [0, 200]).endswith("…"), "A17 超长片段截断加省略号")
163
+ ok(LC._line_of(text, nd["content"], [0, 5]) == 6,
164
+ "A18 _line_of 定位到正文首行(实得 %s)" % LC._line_of(text, nd["content"], [0, 5]))
165
+ ok(LC._line_of(None, "x", [0, 1]) is None and LC._line_of(text, "不存在", [0, 1]) is None,
166
+ "A19 无 text / 正文不在文件内 → line=None(不编造)")
167
+ ok(LC.canonical_kind("dup_content") == "dup"
168
+ and LC.canonical_kind("template_flow_digits_only") == "template_flow",
169
+ "A20 M1/D1 用词归并到 D1 规范名")
170
+ ok(LC.canonical_kind("天外飞仙") == "天外飞仙", "A21 未知类别原样返回(不假装认路)")
171
+
172
+
173
+ # ---------------------------- B 组:六类定位器 ----------------------------
174
+
175
+ def phase_b(tmp):
176
+ print("[B] 六类定位器")
177
+
178
+ # B1-B6 missing_field
179
+ m = {"id": "b1", "layer": "knowledge", "content_hash": "h1",
180
+ "tags": ["a"], "role": "", "importance": 0.5, "evidence_count": 3,
181
+ "lifecycle_state": "active", "condition_space": full_cs()}
182
+ hits = LC.locate_ex("b1", "missing_field", meta=m, content=ccg(),
183
+ text="", fm={}, peers=[], now=NOW)["hits"]
184
+ ok(fields_of(hits) == ["role"] and all(h["span"] is None for h in hits),
185
+ "B1 字段两处皆空 → missing_field 且 span=None(实得 %s)" % fields_of(hits))
186
+
187
+ m2 = dict(m, evidence_count=0, role="knowledge-card")
188
+ hits = LC.locate_ex("b1", "missing_field", meta=m2, content=ccg(),
189
+ text="", fm={}, peers=[], now=NOW)["hits"]
190
+ ok("evidence_count" in fields_of(hits) and "evidence_zero" in {h["rule"] for h in hits},
191
+ "B2 evidence_count=0 → 专项命中(rule=evidence_zero)")
192
+
193
+ m3 = dict(m, importance=1.7, role="k")
194
+ hits = LC.locate_ex("b1", "missing_field", meta=m3, content=ccg(),
195
+ text="", fm={}, peers=[], now=NOW)["hits"]
196
+ ok("importance" in fields_of(hits) and "field_invalid" in {h["rule"] for h in hits},
197
+ "B3 importance 越界 → 字段存在但不可用")
198
+
199
+ m4 = dict(m, condition_space={"observation_position": "本地"},
200
+ role="k", tags=["a"])
201
+ body = ccg()
202
+ hits = LC.locate_ex("b1", "missing_field", meta=m4, content=body,
203
+ text="", fm={}, peers=[], now=NOW)["hits"]
204
+ h = [x for x in hits if x["field"] == "condition_space"]
205
+ ok(bool(h) and h[0]["span"] is not None
206
+ and body[h[0]["span"][0]:h[0]["span"][1]].startswith("# 生效条件"),
207
+ "B4 四槽不全 → 指向正文「# 生效条件」行(%d/4)"
208
+ % (len(NF.CONDITION_SLOTS) - 3))
209
+
210
+ cut = "# 功能名:只有一行\n正文没有其它要素。\n"
211
+ hits = LC.locate_ex("b1", "missing_field", meta=dict(m, role="k", tags=["a"]),
212
+ content=cut, text="", fm={}, peers=[], now=NOW)["hits"]
213
+ hm = [x for x in hits if x["rule"] == "ccg_incomplete"]
214
+ ok(bool(hm) and hm[0]["field"] == "content" and hm[0]["span"] is None
215
+ and "生效条件" in hm[0]["evidence"],
216
+ "B5 正文缺 CCG 要素行 → field=content 且 span=None(行不存在,不编造区间)")
217
+ ok(not [x for x in LC.locate_ex("b1", "missing_field", meta=m3, content=ccg(),
218
+ text="", fm={}, peers=[], now=NOW)["hits"]
219
+ if x["rule"] == "ccg_incomplete"],
220
+ "B6 要素齐全 → 无 ccg_incomplete 噪声")
221
+
222
+ # B7-B10 weak_source
223
+ hits = LC.locate_ex("b1", "weak_source",
224
+ meta={"verification_basis": "", "tags": ["计算机"]},
225
+ content=ccg(), text="", fm={}, peers=[], now=NOW)["hits"]
226
+ ok(len(hits) == 1 and hits[0]["rule"] == "basis_absent", "B7 基底为空 → basis_absent")
227
+
228
+ hits = LC.locate_ex("b1", "weak_source",
229
+ meta={"verification_basis": "self", "tags": ["计算机"]},
230
+ content=ccg(), text="", fm={}, peers=[], now=NOW)["hits"]
231
+ ok(len(hits) == 1 and hits[0]["rule"] == "basis_enum", "B8 基底越枚举 → basis_enum")
232
+
233
+ hits = LC.locate_ex("b1", "weak_source",
234
+ meta={"verification_basis": "textbook", "tags": ["计算机"]},
235
+ content=ccg(), text="", fm={}, peers=[], now=NOW)["hits"]
236
+ ok(len(hits) == 1 and hits[0]["rule"] == "basis_licensed"
237
+ and "science" in hits[0]["evidence"], "B9 理科×textbook → 赛道不相容(实得 %s)"
238
+ % (hits[0]["evidence"] if hits else "无"))
239
+
240
+ hits = LC.locate_ex("b1", "weak_source",
241
+ meta={"verification_basis": "textbook", "tags": ["语文"]},
242
+ content=ccg(), text="", fm={}, peers=[], now=NOW)["hits"]
243
+ ok(hits == [], "B10 文科×textbook 合规 → 零命中(不误报)")
244
+
245
+ # B11-B12c observation_aged(原「stale」的时间窗口径:**观测时刻不是失效声明**)
246
+ m5 = dict(m, role="k", tags=["a"], condition_space=full_cs(tw=EXPIRED))
247
+ body = ccg()
248
+ hits = LC.locate_ex("b1", "observation_aged", meta=m5, content=body, text="",
249
+ fm={}, peers=[], now=NOW)["hits"]
250
+ ok(len(hits) == 1 and hits[0]["field"] == "condition_space"
251
+ and hits[0]["severity"] == "info" and "观测时刻" in hits[0]["evidence"],
252
+ "B11 时间窗过期 → observation_aged(观测面·info,不冒充失效)")
253
+
254
+ m6 = dict(m5, condition_space=full_cs(tw=(0.0, NF.FULL_TIME_WINDOW_MAX)))
255
+ ok(LC.locate_ex("b1", "observation_aged", meta=m6, content=body, text="", fm={},
256
+ peers=[], now=NOW)["hits"] == [],
257
+ "B12 全时窗是合法声明 → 不判")
258
+
259
+ # B12b-B12c 时间窗来源链(真库口径:索引快照只带 time_window,**无 condition_space 键**)
260
+ m_nocs = {k: v for k, v in m.items() if k != "condition_space"}
261
+ hits = LC.locate_ex("b1", "observation_aged", meta=dict(m_nocs, role="k"),
262
+ content=body, text="",
263
+ fm={"condition_space": full_cs(tw=EXPIRED)}, peers=[],
264
+ now=NOW)["hits"]
265
+ ok(len(hits) == 1 and hits[0]["rule"] == "observation_window_passed",
266
+ "B12b 快照无条件空间 → 回退 fm 真源仍判(实得 %d 条)" % len(hits))
267
+
268
+ hits = LC.locate_ex("b1", "observation_aged",
269
+ meta=dict(m_nocs, role="k", time_window=list(EXPIRED)),
270
+ content=body, text="", fm={}, peers=[], now=NOW)["hits"]
271
+ ok(len(hits) == 1 and hits[0]["rule"] == "observation_window_passed",
272
+ "B12c 快照只带 time_window → 真库口径下仍判")
273
+
274
+ # B12d-B12e **观测面不进评审告警面**(B+C 修正的关键隔离断言)
275
+ hits = LC.locate_ex("b1", None, meta=m5, content=body, text="", fm={},
276
+ peers=[], now=NOW)["hits"]
277
+ ok(all(h["issue_kind"] != "observation_aged" for h in hits),
278
+ "B12d 默认全量定位不含 observation_aged(不进评审告警面)")
279
+ ok("observation_aged" in LC.ADVISORY_KINDS
280
+ and "observation_aged" not in LC.ISSUE_KINDS,
281
+ "B12e observation_aged 归观测面(ADVISORY_KINDS),不占问题面 D1 六类")
282
+
283
+ # C stale —— **依赖存在性**(B+C 修正:时效判定看载体是否还在,不看观测时刻)
284
+ src = tempfile.mkdtemp(prefix="m3src_")
285
+ slines = ["def f():", " return 1", "", "def g():", " return 2"]
286
+ with open(os.path.join(src, "mod.py"), "w", encoding="utf-8") as f:
287
+ f.write("\n".join(slines))
288
+ ref_ok = {"path": "mod.py", "name": "f", "kind": "def", "lineno": 1, "end": 2,
289
+ "lang": "py", "precise": True,
290
+ "hash": CI.region_hash(slines, 1, 2), "root": src}
291
+ hits = LC.locate_ex("c1", "stale", meta={}, content=body, text="",
292
+ fm={"code_ref": dict(ref_ok)}, peers=[], now=NOW)["hits"]
293
+ ok(hits == [], "C1 依赖源文件在且区间哈希吻合 → 零命中(不误报)")
294
+
295
+ hits = LC.locate_ex("c1", "stale", meta={}, content=body, text="",
296
+ fm={"code_ref": dict(ref_ok, hash="000000000000")},
297
+ peers=[], now=NOW)["hits"]
298
+ ok(len(hits) == 1 and hits[0]["rule"] == "ref_stale"
299
+ and hits[0]["field"] == "code_ref",
300
+ "C2 源文件在但区间哈希不符(已漂移)→ stale")
301
+
302
+ hits = LC.locate_ex("c1", "stale", meta={}, content=body, text="",
303
+ fm={"code_ref": dict(ref_ok, path="gone.py")},
304
+ peers=[], now=NOW)["hits"]
305
+ ok(len(hits) == 1 and hits[0]["rule"] == "ref_dangling",
306
+ "C3 依赖源文件不存在(悬空)→ stale(载体消失)")
307
+
308
+ hits = LC.locate_ex("c1", "stale", meta={}, content=body, text="",
309
+ fm={"doc_ref": {"path": "x.md", "lineno": 1, "end": 2,
310
+ "hash": "000000000000"}},
311
+ peers=[], now=NOW)["hits"]
312
+ ok(hits == [], "C4 ref 无 root(判不了)→ 零命中(观测手段不足不冒充失效)")
313
+
314
+ hits = LC.locate_ex("c1", None, meta=m5, content=body, text="",
315
+ fm={"code_ref": dict(ref_ok, path="gone.py")},
316
+ peers=[], now=NOW)["hits"]
317
+ ok(any(h["issue_kind"] == "stale" for h in hits),
318
+ "C5 默认全量定位会跑 stale(依赖存在性属问题面)")
319
+ shutil.rmtree(src, ignore_errors=True)
320
+
321
+ # B13-B14 dup
322
+ same = ccg(fn="重复节点")
323
+ peers = LC.build_peers([{"node_id": "b1", "content": same},
324
+ {"node_id": "b2", "content": same}])
325
+ hits = LC.locate_ex("b1", "dup",
326
+ meta=dict(m, role="k", content_hash=NF.content_hash(same)),
327
+ content=same, text="", fm={}, peers=peers.get("b1"),
328
+ now=NOW)["hits"]
329
+ ok(len(hits) == 1 and hits[0]["field"] == "content_hash"
330
+ and hits[0]["span"] == [0, len(same)] and hits[0]["peer"] == "b2",
331
+ "B13 同内容指纹 → dup 指整篇正文(peer=%s)"
332
+ % (hits[0]["peer"] if hits else "无"))
333
+
334
+ hits = LC.locate_ex("b1", "dup", meta=dict(m, role="k", tags=["a"]),
335
+ content=ccg(fn="甲"), text="", fm={},
336
+ peers=LC.build_peers([{"node_id": "b1", "content": ccg(fn="甲")},
337
+ {"node_id": "b2", "content": ccg(fn="乙")}]
338
+ ).get("b1"), now=NOW)["hits"]
339
+ ok(hits == [], "B14 内容不同 → 不判 dup(不误报)")
340
+
341
+ # B15 template_flow:逐句骨架相同、仅数值不同
342
+ t1 = ccg(fn="批次 1 收官", extra="本批处理 100 条记录,耗用 12 秒。第二句写 200 条。")
343
+ t2 = ccg(fn="批次 2 收官", extra="本批处理 300 条记录,耗用 45 秒。第二句写 400 条。")
344
+ peers2 = LC.build_peers([{"node_id": "b1", "content": t1},
345
+ {"node_id": "b2", "content": t2}])
346
+ hits = LC.locate_ex("b1", "template_flow", meta=dict(m, role="k", tags=["a"]),
347
+ content=t1, text="", fm={}, peers=peers2.get("b1"), now=NOW)["hits"]
348
+ ok(len(hits) >= 2 and all(h["field"] == "content" and h["span"] is not None
349
+ for h in hits),
350
+ "B15 同模板流水 → 逐句给出 span(%d 句命中)" % len(hits))
351
+ ok(all(t1[h["span"][0]:h["span"][1]].strip()[:LC.SNIPPET_MAX] == h["snippet"]
352
+ for h in hits),
353
+ "B16 片段=span 切片去空白截断(与实现同口径,可肉眼复核)")
354
+ ok(hits and hits[0]["sentence"] is not None and hits[0]["peer"] == "b2",
355
+ "B17 携带句索引与对照节点(人工可跳行)")
356
+
357
+ hits = LC.locate_ex("b1", "template_flow", meta=dict(m, role="k", tags=["a"]),
358
+ content=t1, text="", fm={},
359
+ peers=LC.build_peers([{"node_id": "b1", "content": t1}]).get("b1"),
360
+ now=NOW)["hits"]
361
+ ok(hits == [], "B18 无对照节点 → 不判流水")
362
+
363
+ # B19-B21 contradiction(确定性)
364
+ body = ccg()
365
+ hits = LC.locate_ex("b1", "contradiction",
366
+ meta={"content_hash": "declared!", "role": "k", "tags": ["a"]},
367
+ content=body, text="", fm={}, peers=[], now=NOW)["hits"]
368
+ ok(len(hits) == 1 and hits[0]["rule"] == "hash_declared_vs_actual"
369
+ and NF.content_hash(body) in hits[0]["evidence"],
370
+ "B19 索引声明指纹 ≠ 正文实算 → contradiction")
371
+
372
+ hits = LC.locate_ex("b1", "contradiction",
373
+ meta={"content_hash": NF.content_hash(body), "role": "k"},
374
+ content=body, text="", fm={"id": "别的id"}, peers=[],
375
+ now=NOW)["hits"]
376
+ ok(len(hits) == 1 and hits[0]["rule"] == "id_declared_vs_index"
377
+ and hits[0]["field"] == "id", "B20 文件 id ≠ 索引键 → contradiction")
378
+
379
+ hits = LC.locate_ex("b1", "contradiction",
380
+ meta={"content_hash": NF.content_hash(body), "role": "k"},
381
+ content=body, text="", fm={"id": "b1"}, peers=[], now=NOW)["hits"]
382
+ ok(hits == [], "B21 声明与事实一致 → 零命中")
383
+
384
+ # B22-B25 字段层门限(判据同源:派生自 M1 规则库,不另立一份)
385
+ scope = LC.field_layer_scope()
386
+ ok(scope.get("role") == ["knowledge"]
387
+ and scope.get("evidence_count") == ["knowledge"],
388
+ "B22 层门限派生自 rules/*.json 的 matcher.layer(实得 %s)" % scope)
389
+ ok("verification_basis" not in scope,
390
+ "B23 未限层的字段不入表 = 全层适用(M1 R-BASIS-MISSING 无 layer 门限)")
391
+
392
+ hits = LC.locate_ex("b1", "missing_field",
393
+ meta=dict(m, layer="contextual", role="", evidence_count=0,
394
+ tags=[]),
395
+ content=ccg(), text="", fm={}, peers=[], now=NOW)["hits"]
396
+ fld = fields_of(hits)
397
+ ok("role" not in fld and "evidence_count" not in fld,
398
+ "B24 非 knowledge 层 → 限层字段越层不报(与 M1 `_scope` 同口径,实得 %s)" % fld)
399
+ ok("tags" in fld,
400
+ "B25 未限层字段不受门限影响(tags 仍全层检查,实得 %s)" % fld)
401
+
402
+ # B26-B28 指纹不一致的**成因**(同一条命中,两种成因,处置完全不同)
403
+ cause, note = LC.hash_mismatch_cause({"content_hash": "x"}, root=None, path=None)
404
+ ok(cause == "unknown" and "成因未判定" in note,
405
+ "B26 无盘上证据 → cause=unknown(不假装知道成因)")
406
+
407
+ root = mkroot(os.path.join(tmp, "b_lag"), {
408
+ "n1": {"content": body, "meta": {"content_hash": "declared!"}},
409
+ })
410
+ node_f = os.path.join(root, "knowledge", "n1.md")
411
+ idx_f = os.path.join(root, CF.INDEX_FILE)
412
+ mt = os.path.getmtime(node_f)
413
+ os.utime(idx_f, (mt - 60.0, mt - 60.0)) # 索引快照比节点文件旧 60s
414
+ hc = [h for h in LC.locate_ex("n1", "contradiction", root=root, now=NOW)["hits"]
415
+ if h["rule"] == "hash_declared_vs_actual"]
416
+ ok(hc and hc[0]["cause"] == "index_lag" and "快照滞后" in hc[0]["evidence"],
417
+ "B27 文件比索引快照新 → cause=index_lag(正常写路径现象,实得 %s)"
418
+ % (hc[0]["cause"] if hc else "无命中"))
419
+
420
+ os.utime(idx_f, (mt + 60.0, mt + 60.0)) # 索引快照不旧于节点文件
421
+ hc = [h for h in LC.locate_ex("n1", "contradiction", root=root, now=NOW)["hits"]
422
+ if h["rule"] == "hash_declared_vs_actual"]
423
+ ok(hc and hc[0]["cause"] == "true_mismatch" and "真源相抵触" in hc[0]["evidence"],
424
+ "B28 索引快照不旧于文件 → cause=true_mismatch(需查,实得 %s)"
425
+ % (hc[0]["cause"] if hc else "无命中"))
426
+
427
+
428
+ # ---------------------------- C 组:提示过滤与 blindspot ----------------------------
429
+
430
+ def _mixed():
431
+ """同时命中多类的节点:role 空 + 四槽不全(missing_field)+ 指纹不符(contradiction)。
432
+
433
+ `condition_space` 刻意只声明 1 槽——让 D 组同时存在「可指区间」(`# 生效条件` 行)
434
+ 与「无区间」(frontmatter 声明类)两种命中,契约两侧都被覆盖。
435
+ 基底取 `textbook` + `语文`(文科档)——让 `weak_source` 真的干净,
436
+ C 组才能验证「该类无问题就返回空、不借机报别的类」。
437
+ """
438
+ body = ccg()
439
+ return dict({"content_hash": "declared!", "role": "", "tags": ["语文"],
440
+ "layer": "knowledge", "verification_basis": "textbook",
441
+ "importance": 0.5, "evidence_count": 3, "lifecycle_state": "active",
442
+ "condition_space": {"observation_position": "本地仓"}}), body
443
+
444
+
445
+ def phase_c(tmp):
446
+ print("[C] 提示过滤与 blindspot")
447
+ m, body = _mixed()
448
+ kw = dict(meta=m, content=body, text="", fm={}, peers=[], now=NOW)
449
+
450
+ all_hits = LC.locate_ex("c1", None, **kw)["hits"]
451
+ ok({"missing_field", "contradiction"} <= set(kinds_of(all_hits)),
452
+ "C1 hint=None → 全量定位(实得 %s)" % kinds_of(all_hits))
453
+
454
+ only = LC.locate_ex("c1", "weak_source", **kw)["hits"]
455
+ ok(all(h["issue_kind"] == "weak_source" for h in only),
456
+ "C2 hint=类别 → 只跑该类(实得 %s)" % kinds_of(only))
457
+ ok(only == [], "C3 该类无问题 → 空(不借机报别的类)")
458
+
459
+ by_field = LC.locate_ex("c1", "role", **kw)["hits"]
460
+ ok(by_field and all(h["field"] == "role" for h in by_field),
461
+ "C4 hint=字段名 → 按字段过滤全量结果(实得 %s)" % fields_of(by_field))
462
+
463
+ both = LC.locate_ex("c1", {"issue_kind": "missing_field", "field": "role"}, **kw)["hits"]
464
+ ok(both and len(both) == len(by_field), "C5 dict 形态 hint 同时收类别与字段")
465
+
466
+ lst = LC.locate_ex("c1", ["contradiction", "missing_field"], **kw)["hits"]
467
+ ok(len(lst) == len(all_hits)
468
+ and set(kinds_of(lst)) == {"contradiction", "missing_field"}
469
+ and LC.locate_ex("c1", ["contradiction", {"field": "role"}], **kw)["hits"] == [],
470
+ "C6 list 形态 hint 收集多个类别;类别与字段是收窄关系(交集空即空,不退回全量)"
471
+ "(实得 %d 条 %s)" % (len(lst), kinds_of(lst)))
472
+
473
+ ex = LC.locate_ex("c1", "contradiction_semantic", **kw)
474
+ ok(ex["hits"] == [] and ex["blindspot"] and "语义" in ex["blindspot"][0],
475
+ "C7 语义级矛盾 → blindspot 且零 hits(不猜、不编造区间)")
476
+ ok(ex["blindspot"] == LC.locate_ex("c1", "contradiction_semantic", **kw)["blindspot"],
477
+ "C8 blindspot 文本确定(可断言)")
478
+
479
+ ex2 = LC.locate_ex("c1", ["contradiction_semantic", "missing_field"], **kw)
480
+ ok(ex2["blindspot"] and ex2["hits"]
481
+ and all(h["issue_kind"] == "missing_field" for h in ex2["hits"]),
482
+ "C9 blindspot 与可定位类别同批共存(互不吞没)")
483
+
484
+ unk = LC.locate_ex("c1", "天外飞仙", **kw)
485
+ ok(unk["hits"] == [] and unk["blindspot"] == [],
486
+ "C10 未知提示 → 当字段过滤后为空,不炸也不假装认路")
487
+
488
+ alias = LC.locate_ex("c1", "dup_content", **kw)["hits"]
489
+ ok(all(h["issue_kind"] == "dup" for h in alias) and bool(alias) is False,
490
+ "C11 M1 用词 dup_content 归并为 dup(无重复故空)")
491
+
492
+ ok(LC.blindspot_reason("contradiction") == "" and
493
+ LC.blindspot_reason("contradiction_semantic") != "",
494
+ "C12 blindspot_reason 单一归口(可定位类别返回空串)")
495
+
496
+
497
+ # ---------------------------- D 组:契约与确定性 ----------------------------
498
+
499
+ def phase_d(tmp):
500
+ print("[D] 契约与确定性")
501
+ m, body = _mixed()
502
+ kw = dict(meta=m, content=body, text="", fm={}, peers=[], now=NOW)
503
+ hits = LC.locate("d1", None, **kw)
504
+
505
+ ok(isinstance(hits, list), "D1 locate() 契约入口返回 list")
506
+ ok(all({"field", "span", "issue_kind", "evidence"} <= set(h) for h in hits),
507
+ "D2 四键齐备(实得 %s)" % (sorted(hits[0]) if hits else "无命中"))
508
+ ok(all(h["issue_kind"] in LC.ISSUE_KINDS for h in hits),
509
+ "D3 issue_kind 全落 D1 枚举(实得 %s)" % kinds_of(hits))
510
+ ok(all(str(h["evidence"]).strip() for h in hits),
511
+ "D4 evidence 非空(白箱判据:为何算问题)")
512
+
513
+ sp = [h for h in hits if h["span"] is not None]
514
+ ok(all(isinstance(h["span"], list) and len(h["span"]) == 2
515
+ and h["span"][0] < h["span"][1] and h["span"][1] <= len(body) for h in sp),
516
+ "D5 span 是正文内的半开区间 [start,end)")
517
+ ok(all(isinstance(h["span"][0], int) and isinstance(h["span"][1], int) for h in sp),
518
+ "D6 span 端点为整数(可复算切片)")
519
+ ok(all(body[h["span"][0]:h["span"][1]].strip() != "" for h in sp),
520
+ "D7 span 指向非空片段")
521
+ ok(all(h.get("snippet") != "" for h in sp),
522
+ "D8 有 span 即有片段(人工核对可肉眼确认)")
523
+ ok(all(not h.get("snippet") for h in hits if h["span"] is None),
524
+ "D9 声明类命中无区间 → 片段为空(不编造区间)")
525
+
526
+ h2 = LC.locate("d1", None, **kw)
527
+ ok(json.dumps(hits, ensure_ascii=False) == json.dumps(h2, ensure_ascii=False),
528
+ "D10 同一输入两次调用逐字节一致(确定性)")
529
+ ok(LC.locate_ex("d1", None, **kw)["load"]["content_len"] == len(body),
530
+ "D11 load 回报正文长度(审计留痕)")
531
+
532
+ key = [(h["issue_kind"], str(h["field"])) for h in hits]
533
+ ok(key == sorted(key), "D12 命中按 (issue_kind, field, span, peer) 稳定排序(实得 %s)" % key)
534
+
535
+ try:
536
+ LC.locate_ex("d1", None, meta=m, content=None, text="", fm={}, peers=[], now=NOW)
537
+ raised = False
538
+ except ValueError:
539
+ raised = True
540
+ ok(raised, "D13 无正文且无 root → fail-closed 报错(不假装能定位)")
541
+
542
+
543
+ # ---------------------------- E 组:批量与包 ----------------------------
544
+
545
+ def _pkg(entries, bid="bE"):
546
+ return {"bundle_id": bid, "group_kind": "batch", "group_key": "g",
547
+ "size": len(entries), "entries": entries}
548
+
549
+
550
+ def _ent(nid, *, excerpt="", h="", **kw):
551
+ e = {"ref": nid, "node_id": nid, "excerpt": excerpt, "content_hash": h,
552
+ "layer": "knowledge", "tags": ["a"]}
553
+ e.update(kw)
554
+ return e
555
+
556
+
557
+ def _clean_meta(**kw):
558
+ """合规 meta(理科档 × test 基底)——「干净」必须是真干净,否则 clean 断言无意义。
559
+
560
+ `layer="knowledge"`:层门限(真源 = M1 规则库 `matcher.layer`)下,
561
+ `role`/`evidence_count` 只在本层检查;缺层节点根本不进判据,clean 断言会空转。
562
+ """
563
+ m = {"role": "knowledge-card", "tags": ["计算机"], "layer": "knowledge",
564
+ "verification_basis": "test",
565
+ "importance": 0.5, "evidence_count": 2, "lifecycle_state": "active",
566
+ "condition_space": full_cs()}
567
+ m.update(kw)
568
+ return m
569
+
570
+
571
+ def phase_e(tmp):
572
+ print("[E] 批量与包")
573
+ same = ccg(fn="重复的")
574
+ diff = ccg(fn="独一无二的甲")
575
+ peers = LC.build_peers([{"node_id": "e1", "content": same},
576
+ {"node_id": "e2", "content": same},
577
+ {"node_id": "e3", "content": diff}])
578
+ ok([p["node_id"] for p in peers["e1"]] == ["e2"]
579
+ and [p["node_id"] for p in peers["e2"]] == ["e1"],
580
+ "E1 build_peers 同内容指纹互为对照(双向)")
581
+ ok(peers["e3"] == [], "E2 内容不同 → 不同组(不是「同批即同组」)")
582
+
583
+ t1 = ccg(fn="批次 1 收官", extra="处理 100 条。")
584
+ t2 = ccg(fn="批次 2 收官", extra="处理 200 条。")
585
+ p2 = LC.build_peers([{"node_id": "e1", "content": t1},
586
+ {"node_id": "e2", "content": t2}])
587
+ ok([p["node_id"] for p in p2["e1"]] == ["e2"],
588
+ "E3 同标题模板骨架 → 互为对照(指纹不同也入组)")
589
+ ok(all(p["sk"] for p in p2["e1"]), "E4 对照项携带模板骨架(M1 同源口径)")
590
+
591
+ ok(LC.build_peers([]) == {}, "E5 空批 → 空映射(不炸)")
592
+ ok("e1" in LC.build_peers([{"node_id": "e1", "content": same}]),
593
+ "E6 单条批仍回填键(调用方不必判空)")
594
+
595
+ items = [{"node_id": "e1", "content": ccg(fn="干净的"), "meta": _clean_meta()},
596
+ {"node_id": "e2", "content": ccg(fn="有问题的"),
597
+ "meta": _clean_meta(role="")}]
598
+ rep = LC.locate_many(items=items, now=NOW)
599
+ ok(rep["nodes"] == 2 and rep["clean"] == ["e1"] and rep["missing"] == [],
600
+ "E7 locate_many 报「干净」条(没问题≠没看,clean=%s)" % rep["clean"])
601
+ ok(rep["by_kind"].get("missing_field") == 1 and rep["by_field"].get("role") == 1,
602
+ "E8 by_kind/by_field 汇总正确(%s / %s)" % (rep["by_kind"], rep["by_field"]))
603
+ ok(all(h["node_id"] == "e2" for h in rep["hits"]), "E9 命中归属到正确节点")
604
+
605
+ root = mkroot(os.path.join(tmp, "e"), {
606
+ "n1": {"content": ccg(fn="盘上节点"), "meta": _clean_meta(role="")},
607
+ "n2": {"content": ccg(fn="盘上无问题"), "meta": _clean_meta()},
608
+ })
609
+ rep2 = LC.locate_many(["n1", "n2", "ghost"], root=root, now=NOW)
610
+ ok(rep2["nodes"] == 2 and rep2["missing"] == ["ghost"],
611
+ "E10 读不到的节点单列 missing(「没看」≠「没问题」,missing=%s)" % rep2["missing"])
612
+ ok(rep2["clean"] == ["n2"], "E11 读盘形态同样分流 clean")
613
+ ok(rep2["hits"] and rep2["hits"][0]["field"] == "role",
614
+ "E12 读盘形态命中与显式 items 同判据")
615
+
616
+ exc9 = ccg(fn="只在摘录里的节点")
617
+ pkg = _pkg([_ent("n1", excerpt="摘录里没有特征码", h="过期指纹"),
618
+ _ent("p9", excerpt=exc9, h=NF.content_hash(exc9),
619
+ **_clean_meta(condition_space={"observation_position": "本地仓",
620
+ "time_window": [0.0, 9999999999.0],
621
+ "observation_tool": "test"}))])
622
+ rep3 = LC.locate_package(pkg, root=root, now=NOW)
623
+ ok(rep3["bundle_id"] == "bE" and rep3["entries"] == 2 and rep3["nodes"] == 2,
624
+ "E13 locate_package 带包标识与条目计数")
625
+ ok(any(h["node_id"] == "n1" for h in rep3["hits"]),
626
+ "E14 包内可读节点走读盘正文(准确)")
627
+ p9 = [h for h in rep3["hits"] if h["node_id"] == "p9"]
628
+ ok(p9 and all(h["field"] == "condition_space" for h in p9)
629
+ and exc9[p9[0]["span"][0]:p9[0]["span"][1]].startswith("# 生效条件"),
630
+ "E15 文件不可读 → 用包内 excerpt 仍给出正文区间(降级但仍可指,实得 %s)"
631
+ % kinds_of(p9))
632
+ ok(not [h for h in rep3["hits"] if h["node_id"] == "n1"
633
+ and h["issue_kind"] == "dup"],
634
+ "E16 excerpt 不冒充正文做重复判定(诚实降级)")
635
+
636
+ s = LC.summary(rep3["hits"])
637
+ ok(s["total"] == len(rep3["hits"]) and s["nodes"] == len(s["node_ids"])
638
+ and s["by_kind"], "E17 summary 汇总口径自洽")
639
+ ok(LC.summary([])["total"] == 0 and LC.summary([])["node_ids"] == [],
640
+ "E18 空命中 summary 不炸")
641
+
642
+ md = LC.markdown_table(rep3["hits"])
643
+ ok(md.count("\n") >= len(rep3["hits"]) + 1 and "人工判定" in md,
644
+ "E19 markdown_table 逐条一行且留人工判定列")
645
+ ok(LC.markdown_table([{"node_id": "x", "issue_kind": "dup", "field": "c",
646
+ "span": None, "evidence": "含|竖线"}]).count("\\|") == 1,
647
+ "E20 表格竖线转义(不破坏表格结构)")
648
+
649
+ code = LC.main(["--root", root, "--node", "n1", "--json"])
650
+ ok(code == 0, "E21 CLI --json 退出码 0")
651
+ code2 = LC.main(["--root", root, "--node", "n1", "--kind", "missing_field",
652
+ "--markdown"])
653
+ ok(code2 == 0, "E22 CLI --kind + --markdown 退出码 0")
654
+ ok(LC.main(["--node", "n1"]) == 2, "E23 缺 root → 退出码 2(fail-closed)")
655
+ ok(LC.main(["--root", root]) == 2, "E24 缺 node → 退出码 2(不静默空跑)")
656
+
657
+
658
+ # ---------------------------- F 组:零写入 ----------------------------
659
+
660
+ def phase_f(tmp):
661
+ print("[F] 零写入")
662
+ root = mkroot(os.path.join(tmp, "f"), {
663
+ "n1": {"content": ccg(fn="批次 1 收官", extra="处理 100 条。"),
664
+ "meta": {"role": "", "tags": ["a"], "importance": 0.5,
665
+ "evidence_count": 0, "lifecycle_state": "active",
666
+ "condition_space": full_cs(tw=EXPIRED)}},
667
+ "n2": {"content": ccg(fn="批次 2 收官", extra="处理 200 条。"),
668
+ "meta": {"role": "knowledge-card", "tags": ["语文"],
669
+ "importance": 0.5, "evidence_count": 1,
670
+ "lifecycle_state": "active",
671
+ "condition_space": full_cs()}},
672
+ })
673
+ before = snapshot(root)
674
+ LC.locate_many(["n1", "n2"], root=root, now=NOW)
675
+ LC.locate_package(_pkg([_ent("n1"), _ent("n2")]), root=root, now=NOW)
676
+ LC.locate("n1", None, root=root, now=NOW)
677
+ LC.main(["--root", root, "--node", "n1", "--node", "n2"])
678
+ ok(snapshot(root) == before,
679
+ "F1 全相位跑完认知图指纹逐字节不变(M3 只读)")
680
+ ok(not os.path.exists(os.path.join(root, "_mreview")),
681
+ "F2 定位不另立状态目录(无残留)")
682
+
683
+
684
+ # ---------------------------- main ----------------------------
685
+
686
+ def main(argv=None):
687
+ with tempfile.TemporaryDirectory(prefix="mrev_m3_") as tmp:
688
+ phase_a(tmp)
689
+ phase_b(tmp)
690
+ phase_c(tmp)
691
+ phase_d(tmp)
692
+ phase_e(tmp)
693
+ phase_f(tmp)
694
+ print("\nM3 自测:%d 通过 / %d 失败 / %d 跳过" % (PASS, FAIL, SKIP))
695
+ if FAILS:
696
+ print("失败项:")
697
+ for f in FAILS:
698
+ print(" - %s" % f)
699
+ return 1 if FAIL else 0
700
+
701
+
702
+ if __name__ == "__main__":
703
+ sys.exit(main())