@furongjun1999/dsh-memory 0.5.0 → 0.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (156) hide show
  1. package/README.md +572 -465
  2. package/docs/README.md +1 -0
  3. package/docs/eval/DSH/346/227/245/345/277/227/347/264/242/345/274/225v2_/345/217/202/350/200/203dsh-TUI_v1.0.md +248 -0
  4. package/docs/eval/DSH/346/227/245/345/277/227/347/264/242/345/274/225/346/225/210/346/236/234/351/252/214/350/257/201_v1.0.md +209 -0
  5. package/docs/eval/DSH/347/253/257/347/274/272/351/231/267/344/270/223/351/241/271_v1.0.md +254 -0
  6. package/docs/eval/P1b2_/350/257/273/351/235/242/344/273/243/351/231/205/344/277/256/345/244/215_v1.0.md +124 -0
  7. package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
  8. package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
  9. package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
  10. package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
  11. package/docs/eval//345/217/221/345/270/20306_/344/270/200/351/224/256/351/205/215/347/275/256/344/270/216DSH0172_v1.0.md +171 -0
  12. package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
  13. package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
  14. package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
  15. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
  16. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
  17. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
  18. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
  19. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
  20. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
  21. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
  22. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
  23. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
  24. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v18.md +183 -0
  25. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v19.md +207 -0
  26. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
  27. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
  28. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
  29. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
  30. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
  31. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
  32. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
  33. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
  34. package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
  35. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +19 -3
  36. package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
  37. package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
  38. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  39. package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
  40. package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
  41. package/dsh/cordis-patch-profile-web.example.yml +35 -0
  42. package/lib/bridge.d.ts +9 -0
  43. package/lib/bridge.js +35 -0
  44. package/lib/cli.d.ts +3 -0
  45. package/lib/cli.js +66 -0
  46. package/lib/index.js +7 -1
  47. package/lib/init.d.ts +88 -0
  48. package/lib/init.js +287 -0
  49. package/lib/lib/roleplay_web.js +530 -443
  50. package/lib/lib/token_store.d.ts +7 -1
  51. package/lib/lib/token_store.js +12 -3
  52. package/md_cg/audit.py +12 -1
  53. package/md_cg/backfill.py +16 -15
  54. package/md_cg/backfill_bucket_zh.py +35 -0
  55. package/md_cg/bench_e2e_judge.py +532 -0
  56. package/md_cg/bench_e2e_locomo_qa.py +368 -0
  57. package/md_cg/bench_e2e_qa.py +256 -0
  58. package/md_cg/branches.py +18 -2
  59. package/md_cg/ccgc.py +3 -2
  60. package/md_cg/chain.py +19 -4
  61. package/md_cg/consolidate.py +7 -6
  62. package/md_cg/crosscheck.py +4 -3
  63. package/md_cg/crypto.py +439 -437
  64. package/md_cg/datapath.py +395 -335
  65. package/md_cg/evidence.py +8 -3
  66. package/md_cg/export.py +3 -1
  67. package/md_cg/forgetting.py +2 -2
  68. package/md_cg/fsutil.py +48 -0
  69. package/md_cg/hotcache.py +255 -238
  70. package/md_cg/interop.py +161 -22
  71. package/md_cg/judgment_manifest.py +177 -0
  72. package/md_cg/links.py +655 -622
  73. package/md_cg/mcp_server.py +282 -31
  74. package/md_cg/mdcg.py +702 -126
  75. package/md_cg/mdcos.py +454 -68
  76. package/md_cg/mreview/govern.py +5 -4
  77. package/md_cg/postings.py +4 -2
  78. package/md_cg/readcache.py +76 -18
  79. package/md_cg/reconcile.py +228 -0
  80. package/md_cg/review_cli.py +215 -0
  81. package/md_cg/routing.py +28 -0
  82. package/md_cg/run_tests.py +211 -0
  83. package/md_cg/scrub.py +862 -852
  84. package/md_cg/security.py +385 -275
  85. package/md_cg/selfreport.py +3 -2
  86. package/md_cg/signer.py +565 -562
  87. package/md_cg/sources.py +3 -2
  88. package/md_cg/stg.py +6 -0
  89. package/md_cg/sustain.py +1168 -1138
  90. package/md_cg/test_access_hints.py +147 -0
  91. package/md_cg/test_branch_discard_tombstone.py +136 -0
  92. package/md_cg/test_branches.py +259 -249
  93. package/md_cg/test_chain_read_isolate.py +168 -0
  94. package/md_cg/test_datapath_device_name.py +203 -0
  95. package/md_cg/test_emit_negtail_cache.py +156 -0
  96. package/md_cg/test_en_pipeline.py +186 -166
  97. package/md_cg/test_govern_directread.py +421 -0
  98. package/md_cg/test_i32_hotcache_env_key.py +218 -0
  99. package/md_cg/test_identity_attribution.py +228 -147
  100. package/md_cg/test_index_crossprocess_reload.py +301 -0
  101. package/md_cg/test_index_durability.py +238 -224
  102. package/md_cg/test_interop.py +4 -2
  103. package/md_cg/test_interop_judgment.py +228 -0
  104. package/md_cg/test_issue39_utf8_stdio.py +273 -0
  105. package/md_cg/test_links_concurrent_write.py +188 -0
  106. package/md_cg/test_merge_upsert.py +168 -0
  107. package/md_cg/test_n123_derive_expiry_chain.py +205 -0
  108. package/md_cg/test_n130_verify_falsified_protect.py +185 -0
  109. package/md_cg/test_n131_merge_gate.py +205 -0
  110. package/md_cg/test_none_id_write_guard.py +165 -0
  111. package/md_cg/test_p1x_ref_root.py +160 -0
  112. package/md_cg/test_p27_docindex.py +774 -765
  113. package/md_cg/test_p2_mcp.py +3 -0
  114. package/md_cg/test_p32_backfill.py +304 -298
  115. package/md_cg/test_p39_verify_flow.py +90 -50
  116. package/md_cg/test_p47_session_view.py +60 -25
  117. package/md_cg/test_propose_tail_index.py +157 -0
  118. package/md_cg/test_read_scope_b27.py +277 -0
  119. package/md_cg/test_readcache_default_on.py +168 -0
  120. package/md_cg/test_readcache_precise_inval.py +270 -0
  121. package/md_cg/test_readcache_prodpath.py +55 -7
  122. package/md_cg/test_reconcile_v0.py +294 -0
  123. package/md_cg/test_retr_s1.py +6 -2
  124. package/md_cg/test_retr_s1b.py +67 -0
  125. package/md_cg/test_retr_s7.py +8 -0
  126. package/md_cg/test_retr_s9_entity_ctx.py +181 -175
  127. package/md_cg/test_retr_score_once.py +208 -0
  128. package/md_cg/test_review_cli_attribution.py +177 -0
  129. package/md_cg/test_review_cli_visibility.py +235 -0
  130. package/md_cg/test_review_onepass.py +170 -0
  131. package/md_cg/test_rrf_graph_seed_cache.py +154 -0
  132. package/md_cg/test_security_audit.py +155 -0
  133. package/md_cg/test_security_audit_b26.py +161 -0
  134. package/md_cg/test_security_audit_v21.py +250 -0
  135. package/md_cg/test_semantic_canonical.py +255 -241
  136. package/md_cg/test_session_isolation.py +168 -0
  137. package/md_cg/test_snapshot_autoclose.py +187 -0
  138. package/md_cg/test_tail_watermark_race.py +208 -0
  139. package/md_cg/test_tenant_env_override_warn.py +139 -0
  140. package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
  141. package/md_cg/test_v14_fixes.py +415 -397
  142. package/md_cg/test_verify_dirty_reconcile.py +157 -0
  143. package/md_cg/theory.py +276 -273
  144. package/md_cg/tokens.py +734 -677
  145. package/md_cg/units.py +3 -2
  146. package/md_cg/vision_evidence.py +4 -3
  147. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  148. package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
  149. package/md_cg/writepipe.py +554 -550
  150. package/package.json +11 -2
  151. package/src/bridge.ts +434 -401
  152. package/src/cli.ts +65 -0
  153. package/src/index.ts +526 -518
  154. package/src/init.ts +347 -0
  155. package/src/lib/roleplay_web.ts +1019 -932
  156. package/src/lib/token_store.ts +202 -192
package/md_cg/mdcg.py CHANGED
@@ -248,6 +248,12 @@ def en_zh_terms(text: str) -> list:
248
248
  独立归一化原子+Jaccard 路(REPRODUCE.md 双语双路裁定,md_cg
249
249
  char-bigram 管线跑英文实测比独立方案差 25% vs 52%),本集成为
250
250
  opt-in 实验能力与 soul hub 序列化出口,不在默认链路生效;
251
+ ⚠ 注意与**统一归一层**的区别(2026-09-24 澄清):`semantic/unify.py`
252
+ 的 `unify_query` 是另一个开关且**默认开启**(MDCG_UNIFY_QUERY,
253
+ 2026-09-23 口径转正),它在检索入口把英文 query 归一成中文原子——
254
+ 于是默认态下英文 query 经**词法路**即可命中中文节点,而本函数
255
+ (召回词扩展侧)仍是默认关。两者不能互相推断,改动其一须核对
256
+ test_en_pipeline / test_semantic_canonical 的双态断言;
251
257
  - 仅当 query 含英文字母时触发,纯中文 query 零开销零变化;
252
258
  - 代词语素剔除(我/你/他…超泛词防污染),动词/名词单字语素保留
253
259
  (「吃/雨」在中文正文检索价值高,_score 终排兜底精度);
@@ -718,25 +724,46 @@ def apply_retrieval_gates(entries, terms, big_domain, context, min_results):
718
724
  gates["s1b"] = {"keys": [], "reason": "disabled_by_topk"}
719
725
  else:
720
726
  _sizes = {}
727
+ _zh_alias = {} # issue #33:桶 → 中文别名集(英文键的跨语言面)
721
728
  for _e in entries:
722
729
  _b = _e.get("bucket")
723
730
  if _b and _b != routing.ORPHAN:
724
731
  _sizes[_b] = _sizes.get(_b, 0) + 1
732
+ _al = _zh_alias.setdefault(_b, set())
733
+ for _w in (_e.get("bucket_zh") or []):
734
+ _al.add(str(_w))
725
735
  _best = {}
726
736
  for _b in _sizes:
727
737
  _kk = routing.bucket_key_readable(_b)
728
738
  if not _kk:
729
739
  continue
740
+ # 比较面 = 可读键 ∪ 中文别名(别名只来自节点自身内容,零翻译依赖;
741
+ # 无别名的存量库行为与旧版逐位一致——别名集为空即旧口径)
742
+ _faces = [_kk] + sorted(_zh_alias.get(_b, ()))
730
743
  _sim = 0.0
731
744
  for _t in terms:
732
- _sv = routing.domain_similarity(_t, _kk)
733
- if _sv > _sim:
734
- _sim = _sv
745
+ for _f in _faces:
746
+ _sv = routing.domain_similarity(_t, _f)
747
+ if _sv > _sim:
748
+ _sim = _sv
735
749
  if _sim >= _s1b_minsim:
736
750
  _best[_b] = _sim
737
751
  if not _best:
738
- gates["s1b"] = {"keys": [], "reason": "no_key_match",
739
- "in": len(entries), "buckets": len(_sizes)}
752
+ # 审计止血(issue #33 建议 3):区分「真的无匹配」与「跨语言盲区」——
753
+ # query 带中文信号、而全部桶的比较面(键+别名)都无中文 → 前者是
754
+ # 语义不相关(正常回退),后者是修复盲区的运维可见形态。
755
+ _q_zh = any(routing._ZH_RE.search(str(_t)) for _t in terms)
756
+ _faces_all_zh = any(
757
+ routing._ZH_RE.search(_f)
758
+ for _b in _sizes
759
+ for _f in ([routing.bucket_key_readable(_b)]
760
+ + sorted(_zh_alias.get(_b, ()))))
761
+ gates["s1b"] = {
762
+ "keys": [],
763
+ "reason": ("cross_lang_no_match"
764
+ if (_q_zh and _sizes and not _faces_all_zh)
765
+ else "no_key_match"),
766
+ "in": len(entries), "buckets": len(_sizes)}
740
767
  else:
741
768
  _picked = sorted(_best.items(),
742
769
  key=lambda kv: (-kv[1], -_sizes[kv[0]],
@@ -770,9 +797,128 @@ def apply_retrieval_gates(entries, terms, big_domain, context, min_results):
770
797
  return entries, gates
771
798
 
772
799
 
800
+ # 生效条件:无独立生效条件(模块级哨兵字典类);任何变更操作(setitem/delitem/clear/pop/popitem/setdefault/update)都会使 write_gen 自增 1,读取 write_gen 不变更;setitem 的值为含非空 path 字符串的 dict 时把该 path 记入 path_gen(值为当时的 write_gen)且不动 broad_gen,值为其它形态(含 None tombstone)或经 delitem/pop/popitem/setdefault/update 变更时把 broad_gen 置为当时的 write_gen,clear() 缺省(flush 收尾)只使 write_gen 自增、不改写 path_gen/broad_gen,clear(broad=True)(rebuild_index 收尾)同时把 broad_gen 置为当时的 write_gen。
801
+ class _DirtyDict(dict):
802
+ """写代际哨兵字典(批次 23,issue #31 D-4 / v20 报告):任何变更使
803
+ `write_gen` **单调自增、永不回退**。
804
+
805
+ 为什么不用 `len(self._dirty)` 作写代际:flush 后 `_dirty` 清零,新写入
806
+ 会使长度**回到旧值**——代际巧合回退会让读缓存误命中陈旧内容(v20
807
+ `v20_d4_repro.py` stale=True 实测)。单调计数器下「任何写都使代际前进」,
808
+ 读缓存的失效判定不再依赖长度巧合。
809
+
810
+ 消费方纪律:flush/rebuild 清空必须走 `.clear()`(保住子类钩子),
811
+ **不得** `self._dirty = {}` 直接换新 dict——那会退化为普通 dict,
812
+ write_gen 恒 0、读缓存失效面随之失效。
813
+
814
+ 脏集精确失效簿记(缺陷迭代第 14 轮,high:写读交替整池重装):
815
+ 全局单一 write_gen 作读缓存哨兵会让**任意一次单节点写使全部缓存条目
816
+ 同时 miss**(10k 池写读交替实测 3rep 中位 1677.6ms vs 稳态 43.7ms,
817
+ 38.4× 退化、线性于 N)。所有节点文件写路径在落盘后必以
818
+ `_dirty[nid] = entry`(entry 带 path)标脏——本类据此把失效粒度从
819
+ 整代际细化到 path:`path_gen[path]` 记该 path 最近标脏时的 write_gen,
820
+ 读缓存按「path 最近标脏代际 ≤ 缓存代际」判新鲜(见 readcache.install)。
821
+ tombstone(值 None,_unstage 删除记录)与其余无 path 可辨的变更形态
822
+ 记 `broad_gen`(保守整池失效,与旧整代际口径等价的安全兜底)。
823
+ `clear()`(flush 收尾)**只推进 write_gen**:flush 只落索引
824
+ 派生物(_index_log 分片/快照),不改节点文件——文件内容变更已在各
825
+ 写路径 `_dirty[nid]=entry` 时按 path 精确失效;且生产写路径每次写后
826
+ `writepipe._commit_visibility→flush`(每次写必 clear),clear 若再整池
827
+ 失效会让精确失效在生产写读交替负载下恒不生效。`clear(broad=True)`
828
+ (rebuild_index 收尾)额外推进 broad_gen 整池失效:rebuild 以盘面
829
+ 扫描为准,ccgc/crosscheck/backfill 等「直写文件 + rebuild 收尾」的
830
+ 写方不经 `_dirty` 标脏,其文件改写对读缓存的可见性靠这一兜底。
831
+ """
832
+
833
+ def __init__(self, *a, **k):
834
+ super().__init__(*a, **k)
835
+ self.write_gen = 0
836
+ self.path_gen = {} # path → 该 path 最近一次标脏时的 write_gen
837
+ self.broad_gen = 0 # 最近一次无 path 可辨变更的 write_gen(整池兜底)
838
+
839
+ def _bump(self):
840
+ self.write_gen += 1
841
+
842
+ def _note(self, v):
843
+ """标脏登记:值带 path → 精确记路径代际;否则保守记整池代际。"""
844
+ p = v.get("path") if isinstance(v, dict) else None
845
+ if isinstance(p, str) and p:
846
+ self.path_gen[p] = self.write_gen
847
+ else:
848
+ self.broad_gen = self.write_gen
849
+
850
+ def __setitem__(self, k, v):
851
+ self._bump()
852
+ self._note(v)
853
+ super().__setitem__(k, v)
854
+
855
+ def __delitem__(self, k):
856
+ self._bump()
857
+ self.broad_gen = self.write_gen
858
+ super().__delitem__(k)
859
+
860
+ def clear(self, broad: bool = False):
861
+ """清空脏集。
862
+
863
+ broad=False(flush 收尾):只推进 write_gen,不整池失效——flush 只
864
+ 落索引派生物(_index_log 分片),节点文件的内容变更已在各写路径
865
+ `_dirty[nid]=entry`(带 path)时按 path 精确失效;且生产写路径每次
866
+ 写后 `writepipe._commit_visibility→flush`,整池失效会让精确失效在
867
+ 生产写读交替负载下恒不生效。
868
+ broad=True(rebuild_index 收尾):同时推进 broad_gen 整池失效——
869
+ rebuild 以**盘面扫描**为准重建索引,ccgc/crosscheck/backfill 等
870
+ 「直写文件 + rebuild 收尾」的写方不经 `_dirty` 标脏,其文件改写对
871
+ 读缓存的可见性全靠这一兜底(与 MdStore「写方落盘后显式 reload」
872
+ 同构)。
873
+ """
874
+ self._bump()
875
+ if broad:
876
+ self.broad_gen = self.write_gen
877
+ super().clear()
878
+
879
+ def pop(self, k, *d):
880
+ self._bump()
881
+ self.broad_gen = self.write_gen
882
+ return super().pop(k, *d)
883
+
884
+ def popitem(self):
885
+ self._bump()
886
+ self.broad_gen = self.write_gen
887
+ return super().popitem()
888
+
889
+ def setdefault(self, k, d=None):
890
+ self._bump()
891
+ self.broad_gen = self.write_gen
892
+ return super().setdefault(k, d)
893
+
894
+ def update(self, *a, **k):
895
+ self._bump()
896
+ self.broad_gen = self.write_gen
897
+ super().update(*a, **k)
898
+
899
+
900
+ # P0-1(批次 24)/ V21-3(批次 34,外部报告):node_id 白名单——id 拼进落盘
901
+ # 路径,穿越(..)与绝对路径/路径分隔符一律拒绝。**不含冒号**:NTFS 上文件名
902
+ # 中的 `:` 是备用数据流(ADS)分隔符(`a:b.md` 实际写 a 的 b.md 流,主文件名
903
+ # 错位)——Windows 目标平台硬约束。V21-3 放行**中文**(\u4e00-\u9fff)与**逗号**:
904
+ # tasks.slugify 设计「不翻译、中文原样保留」(task_任务-名-一 形态)、
905
+ # test_reach 用 comma,node——库自身命名链路必须自洽(V21 报告定案 5 官方红
906
+ # 同源)。中文/逗号无路径语义,P0-1 防线(..禁令+realpath 纵深闸)不变。
907
+ _NODE_ID_RE = re.compile(r"^[A-Za-z0-9_.@\u4e00-\u9fff,-]{1,128}$")
908
+
909
+ # P3(2026-09-26,DSH 在役库实测 None.md 缺陷):空语义值的**字符串化污染形态**
910
+ # 整串禁写——批次 24 前 add 无校验,`f"{node_id}.md"` 直拼,`add(None)` 直落
911
+ # `None.md`(旧版直接源头);批次 24 白名单堵住 None 本体后,字符串 "None"
912
+ # (Python `str(None)`)/"null"(TS `String(null)`)形状合法仍放行(临时库实弹
913
+ # 复现:add("None") → knowledge/orphan/None.md,经 _scan_nodes 收录进检索与
914
+ # route 候选参与打分)。只拒**整串精确等值**(大小写敏感),不扩子串——
915
+ # "NoneBot_1"/"nullify_test" 等合法 id 不受影响。
916
+ _NODE_ID_FORBIDDEN = frozenset({"None", "null"})
917
+
918
+
773
919
  # 生效条件:构造须传入 root,经 os.path.abspath 后以 exist_ok=True 创建该目录及 LAYERS 各层子目录;autoflush 无论取值(默认 64)都原样赋给实例。
774
920
  class MdCG:
775
- # 生效条件:root 传参即被 os.path.abspath 绝对化并 makedirs(exist_ok=True) 建立 root 与模块级 LAYERS 各层目录,autoflush(默认 64,含 0 等假值)原样存入 self.autoflush,随后 _load_index() 载入索引、sweep_stale_temps(self.root) 清扫,并把 self 登记进模块级 _LIVE_CGS;
921
+ # 生效条件:root 传参即被 os.path.abspath 绝对化并 makedirs(exist_ok=True) 建立 root 与模块级 LAYERS 各层目录,autoflush(默认 64,含 0 等假值)原样存入 self.autoflush,随后 _load_index() 载入索引并以 _index_signature() 记录 _index.json 签名到 self._index_sig(跨进程读面代际感知的基线,P1b-2)、sweep_stale_temps(self.root) 清扫,并把 self 登记进模块级 _LIVE_CGS;
776
922
  def __init__(self, root: str, autoflush: int = 64):
777
923
  self.root = os.path.abspath(root)
778
924
  os.makedirs(self.root, exist_ok=True)
@@ -788,9 +934,16 @@ class MdCG:
788
934
  # 近期事件滚动窗口(白箱第 5 篇第 3 章「近期事件」)
789
935
  self.recent_log = os.path.join(self.root, "_recent.jsonl")
790
936
  self.autoflush = autoflush
791
- self._dirty = {}
937
+ self._dirty = _DirtyDict()
792
938
  self._log = None
793
939
  self.index = self._load_index()
940
+ # P1b-2(2026-09-26,DSH 在役复验):索引只在本行装载一次,此后
941
+ # read/get/search 全走内存态——其它进程(review_cli autoflush=1)
942
+ # 写盘对常驻进程不可见(探针 dsh_restart_recheck_20260926:写后
943
+ # _index.json 已含探针、同进程 read 仍 null)。记录装载时的文件
944
+ # 签名作为基线,读路径入口经 _maybe_reload_index 按签名变化增量
945
+ # 重载;无变化路径只有一次 stat(微秒级)。
946
+ self._index_sig = self._index_signature()
794
947
  sweep_stale_temps(self.root)
795
948
  # 进程退出兜底登记(见模块级 _LIVE_CGS):一次性脚本漏收尾时索引仍能落盘。
796
949
  _LIVE_CGS.add(self)
@@ -798,7 +951,7 @@ class MdCG:
798
951
 
799
952
  # ---------- 索引(派生物,可重建) ----------
800
953
 
801
- # 生效条件:os.path.exists(self.index_path) 为真、json.load 成功且其 "schema" 等于模块级 SCHEMA 时以该快照为基底,否则以 self._scan_nodes() 结果与空 buckets 新建;随后重放 ShardedLog.read_all(self.index_log_dir):记录无 id 跳过、e 为 None 则 pop 该 nid(tombstone)、e 非 None 则覆盖,最终 buckets 由 _count_buckets 重算;
954
+ # 生效条件:os.path.exists(self.index_path) 为真、json.load 成功且其 "schema" 等于模块级 SCHEMA、且其 "_fingerprint"(目录树 mtime 指纹)等于当前 self._dir_fingerprint() 时以该快照为基底;快照缺失/损坏/无指纹/指纹不符(盘面在快照写入后有增删——其它存活实例未 flush 的写入或外部改盘)均回退 self._scan_nodes() 全库扫描为基底;随后重放 ShardedLog.read_all(self.index_log_dir):记录无 id 跳过、e 为 None 则 pop 该 nid(tombstone)、e 非 None 则覆盖,最终 buckets 由 _count_buckets 重算;
802
955
  def _load_index(self):
803
956
  idx = None
804
957
  if os.path.exists(self.index_path):
@@ -809,6 +962,17 @@ class MdCG:
809
962
  idx = d
810
963
  except (ValueError, OSError):
811
964
  idx = None
965
+ if idx is not None:
966
+ # 指纹校验(issue #33):快照只在「盘面与写入时一致」时可信。
967
+ # close 自动落盘常态化后,快照路径不再扫盘面——若不校验,
968
+ # 其它存活实例未 flush 的写入(文件已落盘、索引增量还在其
969
+ # _dirty)会从索引不可见。指纹不符 → 回退全库扫描(即原
970
+ # 「无快照」路径的兜底行为,成本不劣于改动前)。旧快照无
971
+ # _fingerprint 键 → 同样回退一次(迁移窗口),下次 close
972
+ # 落新快照后恢复快照路径。
973
+ fp = idx.get("_fingerprint")
974
+ if fp is None or fp != self._dir_fingerprint():
975
+ idx = None
812
976
  if idx is None:
813
977
  idx = {"schema": SCHEMA, "nodes": self._scan_nodes(), "buckets": {}}
814
978
  for rec in ShardedLog.read_all(self.index_log_dir):
@@ -824,6 +988,94 @@ class MdCG:
824
988
  idx["buckets"] = self._count_buckets(idx["nodes"])
825
989
  return idx
826
990
 
991
+ # 生效条件:以 os.stat(self.index_path) 返回 (st_mtime_ns, st_size) 二元组;文件不存在或 stat 抛 OSError 时返回 None(stat 异常静默——签名探测绝不阻塞读);
992
+ def _index_signature(self):
993
+ """_index.json 文件签名(P1b-2 跨进程读面代际的廉价哨兵)。
994
+
995
+ 只 stat 一次(微秒级、无 open 无解析);(mtime_ns, size) 二元组在
996
+ NTFS 100ns / ext4 ns 粒度下足以识别「他进程写快照」。None = 快照
997
+ 不存在(空库或写方尚未 compact),与他进程首次落快照的 None→非 None
998
+ 变化同样可判。
999
+ """
1000
+ try:
1001
+ st = os.stat(self.index_path)
1002
+ return (st.st_mtime_ns, st.st_size)
1003
+ except OSError:
1004
+ return None
1005
+
1006
+ # 生效条件:stat 对比 _index_signature() 与 self._index_sig,相等(含双侧 None)即返回 False 不做任何事;不等则调 _load_index() 重载,OSError/ValueError 时静默放弃并返回 False(重载失败不阻塞读,沿用旧内存态);成功后把 self._dirty 重放回新索引(None=tombstone pop、否则覆盖,与 _load_index 的日志重放同语义——本实例未 flush 的写入不得因重载从检索面消失)并重算 buckets,替换 self.index、刷新 self._index_sig、返回 True;
1007
+ def _maybe_reload_index(self):
1008
+ """读路径入口的索引代际感知(P1b-2):签名变化才重载。
1009
+
1010
+ 语义边界(读码定案):
1011
+ · flush() 只追加 _index_log 分片、不改 _index.json(见 flush 注释)
1012
+ → 签名不变 → 自身写入/落账**永不**触发重载(无自重载循环);
1013
+ · compact_index / rebuild_index 写 _index.json 后已主动刷新签名,
1014
+ 同理不触发;
1015
+ · 他进程只有写出新快照(compact/rebuild,典型在 close)才改变签名
1016
+ → 此时重载并重放分片日志(_load_index 既有行为),写入可见;
1017
+ 仅追加日志、尚无新快照的存活写方不在本探测面内(显式触发可用
1018
+ maintain action=reload,但日志记录只有在快照重写时才会并入);
1019
+ · 本实例 _dirty 未 flush 的条目在重载后**重放回内存索引**——
1020
+ _stage 双写 index+_dirty 的「检索看得到未落盘写入」语义保持。
1021
+ """
1022
+ sig = self._index_signature()
1023
+ if sig == self._index_sig:
1024
+ return False
1025
+ try:
1026
+ idx = self._load_index()
1027
+ except (OSError, ValueError):
1028
+ return False # 重载失败不阻塞读:沿用旧内存态
1029
+ for nid, e in self._dirty.items():
1030
+ if e is None:
1031
+ idx["nodes"].pop(nid, None)
1032
+ else:
1033
+ idx["nodes"][nid] = e
1034
+ idx["buckets"] = self._count_buckets(idx["nodes"])
1035
+ self.index = idx
1036
+ self._index_sig = sig
1037
+ # 旧索引代里的派生缓存一并失效:query 结果缓存(hotcache,默认关)
1038
+ # 缓存的是旧候选池上的结果;解析读缓存(readcache,默认开)按 path
1039
+ # 缓存解析产物——他进程重写同 path 节点时旧产物陈旧。重载是稀疏
1040
+ # 事件(签名变化才有),此处清缓存不构成热路径开销。
1041
+ from . import hotcache as _hc
1042
+ _hc.invalidate(self)
1043
+ from . import readcache as _rc
1044
+ _rc.clear(self)
1045
+ return True
1046
+
1047
+ # 生效条件:遍历 LAYERS 各层目录树(os.walk,不读文件内容),对每个可达目录记录其相对 root 的正斜杠路径到 os.stat().st_mtime_ns 的映射;stat 抛 OSError 的目录跳过;返回该映射。
1048
+ def _dir_fingerprint(self):
1049
+ """盘面目录树 mtime 指纹——检测「快照写入后盘面有增删」的廉价哨兵。
1050
+
1051
+ 只 walk 目录 + 每目录一次 stat(不 open/不解析任何 .md),成本
1052
+ 与目录数成正比、与节点数无关(平铺库 ~LAYERS 次 stat)。目录
1053
+ mtime 在直接子项增删时变化(NTFS 100ns / ext4 ns 粒度)。
1054
+ 边界:同目录**内容级**改写(不动文件名)不触发——合作写路径
1055
+ (add/update)都同步走索引增量日志,不依赖本指纹;外部直改
1056
+ 文件内容属非合作写者协议(见 readcache 同款声明)。
1057
+ """
1058
+ fp = {}
1059
+ for layer in LAYERS:
1060
+ base = os.path.join(self.root, layer)
1061
+ for dirpath, _dirs, _files in os.walk(base):
1062
+ try:
1063
+ fp[os.path.relpath(dirpath, self.root).replace("\\", "/")] = \
1064
+ os.stat(dirpath).st_mtime_ns
1065
+ except OSError:
1066
+ continue
1067
+ return fp
1068
+
1069
+ # 生效条件:遍历 LAYERS 各层目录树(os.walk,不读文件内容),统计文件名以 ".md" 结尾的文件总数并返回。
1070
+ def _count_md_files(self):
1071
+ """盘面 .md 文件计数(不 open/不解析)——compact 写快照前的账本对账。"""
1072
+ n = 0
1073
+ for layer in LAYERS:
1074
+ base = os.path.join(self.root, layer)
1075
+ for _dp, _dirs, files in os.walk(base):
1076
+ n += sum(1 for fn in files if fn.endswith(".md"))
1077
+ return n
1078
+
827
1079
  @staticmethod
828
1080
  # 生效条件:nodes 须为带 values() 的映射且各元素支持 .get("bucket");仅当 bucket 取值为真值时才计入返回计数,缺键或假值均跳过。
829
1081
  def _count_buckets(nodes):
@@ -834,30 +1086,59 @@ class MdCG:
834
1086
  buckets[b] = buckets.get(b, 0) + 1
835
1087
  return buckets
836
1088
 
837
- # 生效条件:os.path.exists(self.index_path) 为真、JSON 可解析且 "schema" 等于 SCHEMA 时以该快照为基底,否则用空骨架 {"schema":SCHEMA,"nodes":{},"buckets":{}} 作基底(不重扫目录),再对 index_log_dir 逐条重放(无 id 跳过、e 为 None 则 pop、否则覆盖),落盘并清空日志后替换 self.index 并返回 idx;
1089
+ # 生效条件:os.path.exists(self.index_path) 为真、JSON 可解析且 "schema" 等于 SCHEMA 时以该快照为基底;快照不存在用空骨架 {"schema":SCHEMA,"nodes":{},"buckets":{}}(不重扫目录);快照损坏(ValueError/OSError)或 schema 不符时降级以 self._scan_nodes() 全库扫描结果为基底(宁重扫勿清池);基底确定后对 index_log_dir 逐条重放(无 id 跳过、e 为 None 则 pop、否则覆盖),落盘并清空日志后替换 self.index、刷新 _index.json 签名(P1b-2 自写不自载)并返回 idx;
838
1090
  def compact_index(self):
839
1091
  with FileLock(self.index_path):
840
1092
  idx = {"schema": SCHEMA, "nodes": {}, "buckets": {}}
1093
+ scan_fallback = False
841
1094
  if os.path.exists(self.index_path):
842
1095
  try:
843
1096
  with open(self.index_path, encoding="utf-8") as f:
844
1097
  d = json.load(f)
845
1098
  if d.get("schema") == SCHEMA:
846
1099
  idx = d
1100
+ else:
1101
+ scan_fallback = True # schema 不符:视同损坏
847
1102
  except (ValueError, OSError):
848
- pass
849
- for rec in ShardedLog.read_all(self.index_log_dir):
850
- nid, e = rec.get("id"), rec.get("e")
851
- if not nid:
852
- continue
853
- if e is None: # 删除记录(tombstone),见 _load_index
854
- idx["nodes"].pop(nid, None)
855
- else:
856
- idx["nodes"][nid] = e
1103
+ scan_fallback = True # 损坏:不得拿空骨架覆盖
1104
+ if scan_fallback:
1105
+ # 快照损坏/schema 不符 → 降级全库扫描重建基底(宁重扫勿清池)。
1106
+ # close 自动 compact(issue #33)上生产路径后此防御必须:
1107
+ # 空骨架 + 增量日志会写出丢失全部历史节点的快照,而
1108
+ # _load_index 见快照即不扫目录——池子从此不可见。
1109
+ idx = {"schema": SCHEMA, "nodes": self._scan_nodes(),
1110
+ "buckets": {}}
1111
+
1112
+ def _apply_log(nodes):
1113
+ for rec in ShardedLog.read_all(self.index_log_dir):
1114
+ nid, e = rec.get("id"), rec.get("e")
1115
+ if not nid:
1116
+ continue
1117
+ if e is None: # 删除记录(tombstone),见 _load_index
1118
+ nodes.pop(nid, None)
1119
+ else:
1120
+ nodes[nid] = e
1121
+ return nodes
1122
+
1123
+ _apply_log(idx["nodes"])
1124
+ if not scan_fallback and self._count_md_files() != len(idx["nodes"]):
1125
+ # 计数对账(issue #33):快照+日志账本与盘面不符——其它存活
1126
+ # 实例未 flush 的写入(文件已落盘、索引增量还在其 _dirty)
1127
+ # 或外部增删文件。宁重扫勿写缺账快照:缺账快照会被重开
1128
+ # 路径的指纹校验放行,节点从此「在盘上但不可见」。
1129
+ # 与 _load_index 的指纹兜底双保险:此处保快照**完整**,
1130
+ # 指纹保快照**新鲜**。
1131
+ idx = {"schema": SCHEMA,
1132
+ "nodes": _apply_log(self._scan_nodes()),
1133
+ "buckets": {}}
857
1134
  idx["buckets"] = self._count_buckets(idx["nodes"])
1135
+ idx["_fingerprint"] = self._dir_fingerprint()
858
1136
  atomic_write(self.index_path, json.dumps(idx, ensure_ascii=False))
859
1137
  ShardedLog.clear(self.index_log_dir)
860
1138
  self.index = idx
1139
+ # P1b-2:自己写了快照 → 主动刷新签名,防止后续读路径把自己的
1140
+ # compact 误判为「他进程写入」而做一次无谓重载(自写自载循环)。
1141
+ self._index_sig = self._index_signature()
861
1142
  return idx
862
1143
 
863
1144
  # 生效条件:self._dirty 非空时才在 FileLock(index_path) 下(必要时新建 ShardedLog)逐条 append、清空 _dirty 并关闭分片句柄;self._dirty 为空时立即返回、不写任何记录;
@@ -875,23 +1156,117 @@ class MdCG:
875
1156
  self._log = ShardedLog(self.index_log_dir)
876
1157
  for nid, e in self._dirty.items():
877
1158
  self._log.append({"id": nid, "e": e})
878
- self._dirty = {}
1159
+ self._dirty.clear() # 保住 _DirtyDict 钩子(批次 23 D-4:不得换新 dict)
879
1160
  # 写完立即关分片句柄:Windows 上「被本进程打开的文件」无法删除,
880
1161
  # 若持有句柄,rebuild_index 的 ShardedLog.clear 会静默失败,已并进
881
1162
  # 快照的旧记录被永久重放(旧条目反而覆盖新快照)。append 内部已
882
1163
  # 每次 flush,句柄无需常驻;下一次 append 会按需重开。
883
1164
  self._log.close()
884
1165
 
885
- # 生效条件:随认知图对象生命周期结束调用、可重复;先 flush() 落未达 autoflush 阈值的脏索引(否则尾部写入虽在盘上但不可见),再关闭并置空日志句柄(句柄为假值时跳过关闭);
1166
+ # 生效条件:随认知图对象生命周期结束调用、可重复;先 flush() 落未达 autoflush 阈值的脏索引(否则尾部写入虽在盘上但不可见),随后自动落快照(issue #33:快照存在 → compact_index 增量并入并清日志;快照不存在且 index["nodes"] 非空 → rebuild_index 全量建快照;空库不写快照;OSError/ValueError 静默吞掉——失败时增量日志仍在、重开可重放,close 是兜底路径不该再抛),最后关闭并置空日志句柄(句柄为假值时跳过关闭);
886
1167
  def close(self):
887
1168
  # 先落脏索引再关句柄:否则未达 autoflush 阈值的尾部写入会永久丢失,
888
1169
  # 已有 _index.json 的根重开时不会重扫目录,节点将「在盘上但不可见」。
889
1170
  self.flush()
1171
+ # 自动落快照(issue #33):原 close 只 flush——「add→close」的库
1172
+ # 永远没有快照(compact_index 曾是唯一增量写快照点且生产零调用),
1173
+ # 每次重开都 _scan_nodes 全库逐文件解析(1500 池实测 ~104ms/次)
1174
+ # + 无条件重放全部增量日志(在扫描基底上纯冗余)。close 是一次性
1175
+ # 脚本(review_cli 等)与 MCP server 退出的统一优雅收尾点:
1176
+ # · 有快照 → compact(快照+日志=全量账本,增量并入+清日志);
1177
+ # · 无快照但有节点 → rebuild 全量建快照(历史节点可能无日志记录,
1178
+ # 重放拼不出全量,必须扫一次盘面——本次一次,此后重开零扫描);
1179
+ # · 空库不写(不产生空快照文件,保持原行为)。
1180
+ # 失败静默:与 atexit 兜底吞异常同风格;compact 失败时增量日志
1181
+ # 仍在,重开照常重放,语义退回改动前而不会丢数据。
1182
+ try:
1183
+ if os.path.exists(self.index_path):
1184
+ self.compact_index()
1185
+ elif self.index["nodes"]:
1186
+ self.rebuild_index()
1187
+ except (OSError, ValueError):
1188
+ pass
890
1189
  if self._log:
891
1190
  self._log.close()
892
1191
  self._log = None
893
1192
 
894
- # 生效条件:遍历模块级 LAYERS 各层目录下所有 .md 文件,读取抛 OSError 的跳过;nid 取 fm.get("id"),为假值时回落去掉 .md 的文件名;bucket 取父目录名、父目录等于层名时为 None;返回 nodes;
1193
+ # 生效条件:以节点文件绝对路径 p、其所在层目录名 layer、解析出的 fm 与 content 为入参,构造含 path(相对 root 正斜杠)/layer(fm 回落 layer)/role/content_kind/session/tags/bucket(父目录名,等于层名时 None)/bucket_zh/importance/created_at/verification_basis/has_neg_conditions/content_hash/temporal/spatial/time_window/lifecycle 与 trust 状态字段/branch_id/branched_from/evidence_count/big_domain/observation_position/subgraph/edges/protected/protection_reason/immutable/self_state/derived_from/derived_relation 各键的条目,经 _strip_empty_gate_fields 清洗后返回;
1194
+ def _node_entry(self, p, layer, fm, content):
1195
+ """单节点索引条目——_scan_nodes 与定向 upsert 共用的**唯一真源**。
1196
+
1197
+ issue #34:merge 裁决从 rebuild_index(O(N) 全库扫描)收尾改为
1198
+ 定向 upsert(O(1))后,单文件条目构造必须与全量扫描**同源**,
1199
+ 否则 upsert 与 rebuild 两条路径的字段集漂移(重建前后索引形态
1200
+ 不一致)。提取本方法即为此——_scan_nodes 循环体与 merge 的
1201
+ upsert 都调它,字段口径零漂移由构造保证。
1202
+ """
1203
+ rel = os.path.relpath(p, self.root).replace("\\", "/")
1204
+ parent = os.path.basename(os.path.dirname(p))
1205
+ return _strip_empty_gate_fields({
1206
+ "path": rel, "layer": fm.get("layer", layer),
1207
+ # role 必须回填:它写在节点 frontmatter 里(写入时 role or "user"),
1208
+ # 但索引重建时若不复制,os.roles 会全部退化为 (none),
1209
+ # 来源归因打分随之失效(实测 48 条全丢)。
1210
+ "role": fm.get("role"),
1211
+ # content_kind 入快照:角色化读取视图(第四阶段 6.1)的
1212
+ # 候选资格维度须免读文件可判(与 role 同款理由)。
1213
+ # 纯增量键:批次 C 之前零消费方,view=None 零行为变更。
1214
+ "content_kind": fm.get("content_kind"),
1215
+ # 会话归属入快照(P45 归因维度):写入路径 _stage 早已带出
1216
+ # 该键,重建路径若漏掉,rebuild_index() 之后「按会话过滤」
1217
+ # 即静默全空——快照与 _stage 必须同口径(与 role 同款理由)。
1218
+ "session": fm.get("session"),
1219
+ "tags": fm.get("tags", []),
1220
+ "bucket": parent if parent != layer else None,
1221
+ # issue #33:中文别名入快照(fm 派生,与写入路径同口径)——
1222
+ # S1b 跨语言收敛免读文件可判。
1223
+ "bucket_zh": fm.get("bucket_zh") or None,
1224
+ "importance": fm.get("importance", 0.5),
1225
+ "created_at": fm.get("created_at", 0),
1226
+ "verification_basis": fm.get("verification_basis"),
1227
+ "has_neg_conditions": nodefile.has_non_applicable(content),
1228
+ # 内容指纹走 nodefile 的唯一实现(两段式对账依赖同一算法)
1229
+ "content_hash": nodefile.content_hash(content),
1230
+ # 时空字段入索引快照:STG 查询免读文件(大域/目录索引的延伸)
1231
+ "temporal": fm.get("temporal"),
1232
+ "spatial": fm.get("spatial"),
1233
+ "time_window": (fm.get("condition_space") or {}).get("time_window"),
1234
+ # 生命周期状态(② 显式状态机):索引入快照 → 免读文件可查,
1235
+ # 写入路径也因此无需读盘就能校验迁移合法性。重建口径与
1236
+ # _stage 一致(旧库无该字段 → None → state_of 视为 active)。
1237
+ lifecycle.STATE_FIELD: fm.get(lifecycle.STATE_FIELD),
1238
+ # 可验证记忆单元(trust):重建口径与 _stage 三件同源
1239
+ # (验证态 / 依赖 / 双时间轴)——索引缺键即免读文件不可判。
1240
+ trust.STATE_FIELD: fm.get(trust.STATE_FIELD),
1241
+ trust.DEPS_FIELD: fm.get(trust.DEPS_FIELD),
1242
+ trust.FROM_FIELD: fm.get(trust.FROM_FIELD),
1243
+ trust.UNTIL_FIELD: fm.get(trust.UNTIL_FIELD),
1244
+ # 规范时间轴键(2026-09-19 阶段一):与 _stage 同口径,
1245
+ # 重建索引后新旧口径一致(检索面 validity 须免读盘可判)。
1246
+ trust.EFFECTIVE_FROM_FIELD: fm.get(trust.EFFECTIVE_FROM_FIELD),
1247
+ trust.EFFECTIVE_UNTIL_FIELD: fm.get(trust.EFFECTIVE_UNTIL_FIELD),
1248
+ # 记忆演化分支(④):重建口径与 _stage 一致
1249
+ "branch_id": fm.get("branch_id"),
1250
+ "branched_from": fm.get("branched_from"),
1251
+ "evidence_count": fm.get("evidence_count", 0),
1252
+ # S1 大域先验:域标签入索引快照 → 候选收敛零读文件(与 add() 同口径)
1253
+ "big_domain": fm.get("big_domain"),
1254
+ # S2 条件门控所需的可判定硬槽(免读文件即可门控)
1255
+ "observation_position": (fm.get("condition_space") or {}).get("observation_position"),
1256
+ # 嵌套子图 / 关系边入索引快照:递归展开与链式遍历免读文件
1257
+ "subgraph": fm.get("subgraph"),
1258
+ "edges": fm.get("edges") or [],
1259
+ "protected": fm.get("protected"),
1260
+ "protection_reason": fm.get("protection_reason"),
1261
+ "immutable": fm.get("immutable"),
1262
+ # 自我状态卡标记:protect 据此豁免「不可覆盖」(仍不可遗忘)
1263
+ "self_state": fm.get("self_state"),
1264
+ # G8 派生溯源:frontmatter 声明入索引 → 悬空巡检零读文件
1265
+ "derived_from": fm.get("derived_from") or [],
1266
+ "derived_relation": fm.get("derived_relation"),
1267
+ })
1268
+
1269
+ # 生效条件:遍历模块级 LAYERS 各层目录下所有 .md 文件,读取抛 OSError 的跳过;nid 取 fm.get("id"),为假值时回落去掉 .md 的文件名;条目经 self._node_entry(p, layer, fm, content) 构造;返回按 nid 排序的 nodes;
895
1270
  def _scan_nodes(self):
896
1271
  nodes = {}
897
1272
  for layer in LAYERS:
@@ -904,92 +1279,56 @@ class MdCG:
904
1279
  try:
905
1280
  with open(p, encoding="utf-8") as f:
906
1281
  fm, content = nodefile.loads(f.read())
907
- except OSError:
1282
+ except (OSError, UnicodeDecodeError):
1283
+ # UnicodeDecodeError(批次 53 补):非 UTF-8 字节的损坏
1284
+ # 真源此前会炸穿整库装载/重建(init → _load_index →
1285
+ # 指纹失配回落 _scan_nodes → 一个坏文件全库打不开)。
1286
+ # 与 reconcile v0 的 T9 边界同口径:真源损坏只跳过,
1287
+ # 不猜测不放大——启动对账(reconcile_state)会对其
1288
+ # 告警留痕。
908
1289
  continue
909
1290
  nid = fm.get("id") or fn[:-3]
910
- rel = os.path.relpath(p, self.root).replace("\\", "/")
911
- parent = os.path.basename(dirpath)
912
- nodes[nid] = _strip_empty_gate_fields({
913
- "path": rel, "layer": fm.get("layer", layer),
914
- # role 必须回填:它写在节点 frontmatter 里(写入时 role or "user"),
915
- # 但索引重建时若不复制,os.roles 会全部退化为 (none),
916
- # 来源归因打分随之失效(实测 48 条全丢)。
917
- "role": fm.get("role"),
918
- # content_kind 入快照:角色化读取视图(第四阶段 6.1)的
919
- # 候选资格维度须免读文件可判(与 role 同款理由)。
920
- # 纯增量键:批次 C 之前零消费方,view=None 零行为变更。
921
- "content_kind": fm.get("content_kind"),
922
- # 会话归属入快照(P45 归因维度):写入路径 _stage 早已带出
923
- # 该键,重建路径若漏掉,rebuild_index() 之后「按会话过滤」
924
- # 即静默全空——快照与 _stage 必须同口径(与 role 同款理由)。
925
- "session": fm.get("session"),
926
- "tags": fm.get("tags", []),
927
- "bucket": parent if parent != layer else None,
928
- "importance": fm.get("importance", 0.5),
929
- "created_at": fm.get("created_at", 0),
930
- "verification_basis": fm.get("verification_basis"),
931
- "has_neg_conditions": nodefile.has_non_applicable(content),
932
- # 内容指纹走 nodefile 的唯一实现(两段式对账依赖同一算法)
933
- "content_hash": nodefile.content_hash(content),
934
- # 时空字段入索引快照:STG 查询免读文件(大域/目录索引的延伸)
935
- "temporal": fm.get("temporal"),
936
- "spatial": fm.get("spatial"),
937
- "time_window": (fm.get("condition_space") or {}).get("time_window"),
938
- # 生命周期状态(② 显式状态机):索引入快照 → 免读文件可查,
939
- # 写入路径也因此无需读盘就能校验迁移合法性。重建口径与
940
- # _stage 一致(旧库无该字段 → None → state_of 视为 active)。
941
- lifecycle.STATE_FIELD: fm.get(lifecycle.STATE_FIELD),
942
- # 可验证记忆单元(trust):重建口径与 _stage 三件同源
943
- # (验证态 / 依赖 / 双时间轴)——索引缺键即免读文件不可判。
944
- trust.STATE_FIELD: fm.get(trust.STATE_FIELD),
945
- trust.DEPS_FIELD: fm.get(trust.DEPS_FIELD),
946
- trust.FROM_FIELD: fm.get(trust.FROM_FIELD),
947
- trust.UNTIL_FIELD: fm.get(trust.UNTIL_FIELD),
948
- # 规范时间轴键(2026-09-19 阶段一):与 _stage 同口径,
949
- # 重建索引后新旧口径一致(检索面 validity 须免读盘可判)。
950
- trust.EFFECTIVE_FROM_FIELD: fm.get(trust.EFFECTIVE_FROM_FIELD),
951
- trust.EFFECTIVE_UNTIL_FIELD: fm.get(trust.EFFECTIVE_UNTIL_FIELD),
952
- # 记忆演化分支(④):重建口径与 _stage 一致
953
- "branch_id": fm.get("branch_id"),
954
- "branched_from": fm.get("branched_from"),
955
- "evidence_count": fm.get("evidence_count", 0),
956
- # S1 大域先验:域标签入索引快照 → 候选收敛零读文件(与 add() 同口径)
957
- "big_domain": fm.get("big_domain"),
958
- # S2 条件门控所需的可判定硬槽(免读文件即可门控)
959
- "observation_position": (fm.get("condition_space") or {}).get("observation_position"),
960
- # 嵌套子图 / 关系边入索引快照:递归展开与链式遍历免读文件
961
- "subgraph": fm.get("subgraph"),
962
- "edges": fm.get("edges") or [],
963
- "protected": fm.get("protected"),
964
- "protection_reason": fm.get("protection_reason"),
965
- "immutable": fm.get("immutable"),
966
- # 自我状态卡标记:protect 据此豁免「不可覆盖」(仍不可遗忘)
967
- "self_state": fm.get("self_state"),
968
- # G8 派生溯源:frontmatter 声明入索引 → 悬空巡检零读文件
969
- "derived_from": fm.get("derived_from") or [],
970
- "derived_relation": fm.get("derived_relation"),
971
- })
1291
+ nodes[nid] = self._node_entry(p, layer, fm, content)
972
1292
  # 索引序确定性:按 nid 排序返回。os.walk 的遍历序是**文件系统事实**
973
1293
  # (NTFS 上常为字母序,但换 FS / 目录碎片化后不保证),若直接作为
974
1294
  # index["nodes"] 的物理序,就会让「完全并列」的结果顺序依赖重建路径。
975
1295
  # 排序后重建序恒定(增量路径另由 cut_by_relevance 的 nid 终键兜住)。
976
1296
  return {k: nodes[k] for k in sorted(nodes)}
977
1297
 
978
- # 生效条件:每次调用都以 self._scan_nodes() 的结果重建 nodes 与 buckets,在 FileLock 下 atomic_write 覆盖 index_path 并 ShardedLog.clear(index_log_dir),随后替换 self.index、清空 _dirty 并返回 idx(无 .md 时也照样覆盖为空索引);
1298
+ # 生效条件:每次调用都以 self._scan_nodes() 的结果重建 nodes 与 buckets 并附 _dir_fingerprint(),在 FileLock 下 atomic_write 覆盖 index_path 并 ShardedLog.clear(index_log_dir),随后替换 self.index、刷新 _index.json 签名(P1b-2 自写不自载)、清空 _dirty 并返回 idx(无 .md 时也照样覆盖为空索引);
979
1299
  def rebuild_index(self):
980
1300
  nodes = self._scan_nodes()
981
1301
  idx = {"schema": SCHEMA, "nodes": nodes,
982
1302
  "buckets": self._count_buckets(nodes)}
983
1303
  with FileLock(self.index_path):
1304
+ idx["_fingerprint"] = self._dir_fingerprint()
984
1305
  atomic_write(self.index_path, json.dumps(idx, ensure_ascii=False))
985
1306
  ShardedLog.clear(self.index_log_dir)
986
1307
  self.index = idx
987
- self._dirty = {}
1308
+ # P1b-2:rebuild 同 compact——写完快照即刷新签名(自写不自载)。
1309
+ self._index_sig = self._index_signature()
1310
+ # broad=True:rebuild 以盘面扫描为准——ccgc/crosscheck/backfill 等
1311
+ # 「直写文件 + rebuild 收尾」的写方不经 _dirty 标脏,其文件改写对
1312
+ # 读缓存的可见性靠这一整池失效兜底(flush 的 clear 则不失效,见
1313
+ # _DirtyDict.clear 文档串)。
1314
+ self._dirty.clear(broad=True) # 保住 _DirtyDict 钩子(批次 23 D-4)
988
1315
  return idx
989
1316
 
1317
+ # P2-20(批次 30,外部审查报告):索引中的 path 参与所有读/写落盘定位
1318
+ # ——写穿越(P0-1 历史节点/索引污染)可经「读穿越」放大。单点校验:
1319
+ # realpath 必须落在 root 内,越界抛 ValueError(宁可少读,不可越权)。
1320
+ def _node_disk_path(self, e):
1321
+ p = os.path.realpath(os.path.join(self.root, e.get("path") or ""))
1322
+ rr = os.path.realpath(self.root)
1323
+ if p != rr and not p.startswith(rr + os.sep):
1324
+ raise ValueError(
1325
+ f"节点路径越界(P2-20):{e.get('path')!r} -> {p}"
1326
+ "——拒绝读写")
1327
+ return p
1328
+
990
1329
  # ---------- 写 ----------
991
1330
 
992
- # 生效条件:node_id/content 必填,layer 不在 LAYERS 内、或 verification_basis 非 None 且不在 VERIFICATION_BASIS 内时抛 ValueError;consistency 为真且 _cons.check 判 REJECT 时,on_conflict="reject" 抛 ConsistencyError、on_conflict="defer" 返回 None,verdict 为 BLINDSPOT 且 on_conflict="defer" 同样返回 None,其余情形完成写盘/入索引后返回 node_id。
1331
+ # 生效条件:node_id/content 必填;node_id 不匹配 _NODE_ID_RE(^[A-Za-z0-9_.@-]{1,128}$)或含 ".."、或整串精确等值于 _NODE_ID_FORBIDDEN("None"/"null",空语义值字符串化污染形态,P3 None.md 缺陷禁写)、或落盘 realpath 越出 self.root 时抛 ValueError(P0-1 白名单+纵深闸);layer 不在 LAYERS 内、或 verification_basis 非 None 且不在 VERIFICATION_BASIS 内时抛 ValueError;consistency 为真且 _cons.check 判 REJECT 时,on_conflict="reject" 抛 ConsistencyError、on_conflict="defer" 返回 None,verdict 为 BLINDSPOT 且 on_conflict="defer" 同样返回 None,其余情形完成写盘/入索引后返回 node_id。
993
1332
  def add(self, node_id: str, content: str, layer: str = "knowledge",
994
1333
  tags=None, condition_space=None, importance: float = 0.5,
995
1334
  confidence: float = 0.6, edges=None, verification_basis: str = None,
@@ -1038,6 +1377,21 @@ class MdCG:
1038
1377
  """
1039
1378
  if layer not in LAYERS:
1040
1379
  raise ValueError(f"未知层:{layer}(允许:{LAYERS})")
1380
+ # P0-1(批次 24,外部审查报告):node_id 是模型可控输入,直接拼
1381
+ # 文件路径——`..` 穿越出 root、Windows 绝对路径(C:/x)在
1382
+ # os.path.join 下直接丢弃前缀 = 任意 .md 覆盖。白名单先行
1383
+ # (报告建议 1),realpath 断言在落盘前兜底(报告建议 2)。
1384
+ nid_s = str(node_id or "")
1385
+ if not nid_s or len(nid_s) > 128 or ".." in nid_s \
1386
+ or not _NODE_ID_RE.match(nid_s):
1387
+ raise ValueError(
1388
+ f"非法 node_id:{node_id!r}(须匹配 {_NODE_ID_RE.pattern} "
1389
+ f"且不含 '..'——node_id 会拼进落盘路径,穿越/绝对路径一律拒绝)")
1390
+ if nid_s in _NODE_ID_FORBIDDEN:
1391
+ raise ValueError(
1392
+ f"非法 node_id:{node_id!r}(空语义值的字符串化污染形态 "
1393
+ f"{sorted(_NODE_ID_FORBIDDEN)} 禁写——会落盘成 None.md 参与检索"
1394
+ f"(DSH 在役库实测缺陷 P3);fail-closed,如为合法业务 id 请改名)")
1041
1395
  if verification_basis is not None and verification_basis not in VERIFICATION_BASIS:
1042
1396
  raise ValueError(f"未知验证基底:{verification_basis}(允许:{VERIFICATION_BASIS})")
1043
1397
  # 写保护:self/anchor 层、protected 标记、importance≥0.7 的**既有**节点
@@ -1073,13 +1427,48 @@ class MdCG:
1073
1427
  extra["semantic_oov"] = _oov
1074
1428
  except Exception:
1075
1429
  pass
1430
+ # 桶路由键必须在**选目录之前**补齐:`routing.route_key()` 只认 tags 里的
1431
+ # `domain:` 前缀(其次才是 condition_space.observation_position),**不读
1432
+ # big_domain**;而 bucket 与落盘目录在这一段就定死了(索引的 bucket 同样由
1433
+ # 父目录名派生)。实测:把补标签放在 fm 构造之后(原 W1 的位置)会导致
1434
+ # big_domain 与 domain: 标签都在、但文件仍写进 knowledge/orphan/。
1435
+ # 只对分桶层补标签(BUCKETED_LAYERS=("knowledge",)),避免污染其它层的 tags。
1436
+ if (layer in BUCKETED_LAYERS
1437
+ and os.environ.get("MDCG_RETRIEVAL_PIPELINE") == "1"
1438
+ and not any(str(t).startswith("domain:")
1439
+ for t in (tags or []))):
1440
+ _bd_pre = extra.get("big_domain") or None
1441
+ if not _bd_pre:
1442
+ try:
1443
+ _bd_pre = routing.classify_text(content)
1444
+ except Exception:
1445
+ _bd_pre = None
1446
+ if _bd_pre:
1447
+ tags = list(tags or [])
1448
+ tags.append("domain:" + str(_bd_pre))
1076
1449
  bucket = None
1077
1450
  d = os.path.join(self.root, layer)
1451
+ bucket_zh = []
1078
1452
  if layer in BUCKETED_LAYERS:
1079
1453
  bucket = routing.bucket_dir(routing.route_key(condition_space, tags))
1080
1454
  d = os.path.join(d, bucket)
1455
+ # issue #33:英文桶键配中文别名(节点自身 tags/正文中文词,零翻译依赖),
1456
+ # 供 S1b 跨语言收敛;键含中文或无别名时空。写进 fm 保证 _scan_nodes
1457
+ # 重建后口径一致(索引派生原则,见 _stage 注释)。
1458
+ bucket_zh = routing.bucket_zh_aliases(
1459
+ routing.bucket_key_readable(bucket), tags, content)
1081
1460
  os.makedirs(d, exist_ok=True)
1082
1461
  path = os.path.join(d, f"{node_id}.md")
1462
+ # P0-1 纵深(批次 24):realpath 断言兜底——白名单已挡穿越/绝对
1463
+ # 路径,此闸防未来 node_id 规则放松或 d 被污染(任何写入路径的
1464
+ # 最后一道闸:落盘位置必须在 root 内)。
1465
+ _real_node = os.path.realpath(path)
1466
+ _real_root = os.path.realpath(self.root)
1467
+ if _real_node != _real_root \
1468
+ and not _real_node.startswith(_real_root + os.sep):
1469
+ raise ValueError(
1470
+ f"node_id 落盘路径越界(realpath={_real_node},root={_real_root})"
1471
+ "——拒绝写入(P0-1 纵深闸)")
1083
1472
  created_at = extra.pop("created_at", time.time())
1084
1473
  # 条件论「观测时间」栏:写入时必须记录观测时间窗。
1085
1474
  # 调用方未提供 time_window 时,以写入时刻为锚、默认窗口 OBSERVATION_WINDOW_SEC。
@@ -1100,6 +1489,8 @@ class MdCG:
1100
1489
  "evidence_count": 0, "positive_evidence": 0, "negative_evidence": 0,
1101
1490
  }
1102
1491
  fm.update(extra)
1492
+ if bucket_zh:
1493
+ fm["bucket_zh"] = bucket_zh
1103
1494
  # S1 大域先验:写入时固化「内容 → 大域」(契约 §3 S1)。
1104
1495
  # 为何在写入侧:检索侧要按域收敛,节点就必须带域;query 侧分类器已存在,
1105
1496
  # 缺的只是这一列节点元数据(审计偏差 4 的根因)。调用方可显式传入覆盖。
@@ -1113,6 +1504,21 @@ class MdCG:
1113
1504
  _bd = None
1114
1505
  if _bd:
1115
1506
  fm["big_domain"] = _bd
1507
+ # CCG「不适用条件」写入口径:正文声明了 `# 不适用条件:` 而调用方没给列表时,
1508
+ # 按 ccgc 的同一拆分口径([;;])结构化进 frontmatter。否则 has_non_applicable()
1509
+ # (按正文文本判,health 的 neg_conditions_set 用它)与 non_applicable_conditions
1510
+ # (judge_qualification 的 REJECT 路径用它)会长期脱钩——实测历史数据修好后
1511
+ # 新写入节点立刻复现(1/73)。正文里字面 `…` 是写入期截断尾巴,按占位丢弃。
1512
+ if not fm.get("non_applicable_conditions"):
1513
+ _na_line = nodefile.ccg_field_value(content, "不适用条件")
1514
+ if _na_line:
1515
+ _na_vals = [s.strip() for s in re.split(r"[;;]", _na_line) if s.strip()]
1516
+ _na_vals = [re.sub(r"[..…\s,,;;、]+$", "", v).strip()
1517
+ for v in _na_vals]
1518
+ _na_vals = [v for v in _na_vals
1519
+ if v and not nodefile.is_placeholder_text(v)]
1520
+ if _na_vals:
1521
+ fm["non_applicable_conditions"] = _na_vals
1116
1522
  # 生命周期状态(② 显式状态机,真源 `lifecycle.py`):add 是**全量重建
1117
1523
  # fm** 而非增量更新,故必须显式处理状态——否则已定型/已降权节点会被静默
1118
1524
  # 打回 active(与 ④ rewrite 必须重传 branch_id 同构的坑)。口径:
@@ -1270,9 +1676,34 @@ class MdCG:
1270
1676
  # 私有内容封装(默认恒等;MdCGSecure 覆盖为 AEAD 加密)。
1271
1677
  # 索引派生同样基于落盘内容,保证与 _scan_nodes 重建结果一致。
1272
1678
  sealed = self._write_node(node_id, path, fm, content)
1679
+ # 覆写且路由键变化时清理旧桶同 id 文件:不清理则磁盘留双文件,
1680
+ # rebuild_index() 后索引取哪个取决于 os.walk 枚举序——新内容可能被
1681
+ # 旧文件静默顶掉(违背「原文即真源」)。旧索引条目即旧路径唯一线索;
1682
+ # 双闸防误删:必须在 root 内(P2-20 同口径)且不等于本次新路径
1683
+ # (同桶覆写是同一文件,删了等于自毁)。先写新后删旧,顺序不可倒。
1684
+ if prev_entry and prev_entry.get("path"):
1685
+ try:
1686
+ _old_real = os.path.realpath(
1687
+ os.path.join(self.root, prev_entry["path"]))
1688
+ if _old_real != _real_node \
1689
+ and _old_real.startswith(_real_root + os.sep) \
1690
+ and os.path.isfile(_old_real):
1691
+ os.remove(_old_real)
1692
+ except OSError:
1693
+ pass # 删除失败不阻断主写路径(残留双文件退回旧行为,可重建兜底)
1694
+ # 旧桶计数递减(与 _unstage 同口径),否则 buckets 计数漂移;
1695
+ # 仅在桶变化时做——同桶覆写的 _stage 自增是既有口径,不在此对账。
1696
+ _ob = prev_entry.get("bucket")
1697
+ if _ob and _ob != bucket:
1698
+ _left = self.index["buckets"].get(_ob, 1) - 1
1699
+ if _left > 0:
1700
+ self.index["buckets"][_ob] = _left
1701
+ else:
1702
+ self.index["buckets"].pop(_ob, None)
1273
1703
  self._stage(node_id, _strip_empty_gate_fields({
1274
1704
  "path": os.path.relpath(path, self.root).replace("\\", "/"),
1275
1705
  "layer": layer, "tags": tags, "bucket": bucket,
1706
+ "bucket_zh": bucket_zh or None,
1276
1707
  "importance": importance, "created_at": fm["created_at"],
1277
1708
  "verification_basis": verification_basis,
1278
1709
  "has_neg_conditions": nodefile.has_non_applicable(sealed),
@@ -1376,6 +1807,18 @@ class MdCG:
1376
1807
  from . import twophase
1377
1808
  return twophase.pending(self, limit=limit)
1378
1809
 
1810
+ # 生效条件:每次调用都延迟导入 reconcile 并以原样 apply(含 False)转调 reconcile.reconcile_state(self, apply=apply) 并返回,自身不做参数回落;
1811
+ def reconcile_state(self, apply: bool = True) -> dict:
1812
+ """启动对账 reconcile v0(v0.3 T9 落地,落地清单 P0-3):索引(派生
1813
+ 态)vs 盘面真源的三类 diff 对账(多/缺索引条目、内容哈希漂移)。
1814
+
1815
+ 与 `reconcile_writes`(写账本清账)分工见 `md_cg/reconcile.py` 模块
1816
+ 文档串。`MDCG_RECONCILE=0` 关闭(缺省开,开关真源在 reconcile.enabled);
1817
+ `apply=False` 为 dry-run。真源自身损坏只告警留痕,不自动改写(T9 边界)。
1818
+ """
1819
+ from . import reconcile # 延迟导入:与写路径解耦,避免模块环
1820
+ return reconcile.reconcile_state(self, apply=apply)
1821
+
1379
1822
  # ------------------------------------------------------------------
1380
1823
  # 边域窄原语(写路径收口:白箱工具写 md 真源的唯一正路)。
1381
1824
  #
@@ -1608,6 +2051,7 @@ class MdCG:
1608
2051
  # 生效条件:遍历 self.index["nodes"] 中 layer=="goals" 的节点,_goal_entry 取不到者跳过;status 为假值时不过滤、否则仅保留 status 相等者;按 (-priority, -created_at) 降序,limit 为真值时截断 out[:limit],否则返回全量;
1609
2052
  def list_goals(self, status=None, limit=None):
1610
2053
  """列出目标,按 (priority, created_at) 降序。status 过滤 active/done/dropped。"""
2054
+ self._maybe_reload_index() # P1b-2:读面代际感知(goal=list/active 链)
1611
2055
  out = []
1612
2056
  for nid, e in self.index["nodes"].items():
1613
2057
  if e.get("layer") != "goals":
@@ -1626,7 +2070,7 @@ class MdCG:
1626
2070
  """当前活跃目标(默认最多 5 条)——检索定向的默认来源。"""
1627
2071
  return self.list_goals(status="active", limit=limit)
1628
2072
 
1629
- # 生效条件:status 不属于模块级 GOAL_STATUSES 时抛 ValueError;self.get(node_id) 取不到节点或其 frontmatter 的 layer 不等于 "goals" 时返回 None;否则写回 goal_status 与 status_changed_at 并返回 {"id": node_id, "status": status};
2073
+ # 生效条件:status 不属于模块级 GOAL_STATUSES 时抛 ValueError;self.get(node_id) 取不到节点或其 frontmatter 的 layer 不等于 "goals" 时返回 None;否则写回 goal_status 与 status_changed_at,置 self._dirty[node_id](读缓存代际失效,entry 取 index["nodes"] 现值或 {"path","layer":"goals"} 兜底)并返回 {"id": node_id, "status": status};
1630
2074
  def set_goal_status(self, node_id: str, status: str):
1631
2075
  """目标状态机:active → done/dropped(可回退)。"""
1632
2076
  if status not in GOAL_STATUSES:
@@ -1639,6 +2083,12 @@ class MdCG:
1639
2083
  fm["status_changed_at"] = time.time()
1640
2084
  self._write_node(node_id, os.path.join(self.root, node["path"]),
1641
2085
  fm, node["content"])
2086
+ # 标脏(读缓存代际哨兵前提:写路径必须推进 write_gen——直写不标脏
2087
+ # 会让同实例 _goal_entry/检索读到旧 goal_status,p7「done 退出
2088
+ # active」陈旧形态)。不走 _stage:节点已在索引,_stage 会虚增
2089
+ # bucket 计数;这里只推代际,entry 同值幂等。
2090
+ self._dirty[node_id] = self.index["nodes"].get(node_id) \
2091
+ or {"path": node["path"], "layer": "goals"}
1642
2092
  return {"id": node_id, "status": status}
1643
2093
 
1644
2094
  # 生效条件:goal 为真值(非 None/空串/0)时返回 str(goal);goal 为假值时返回 self.active_goals(limit=limit) 中非空 goal 文本以空格拼接的字符串;
@@ -1801,7 +2251,7 @@ class MdCG:
1801
2251
  st["written"] += 1
1802
2252
  continue
1803
2253
  fm["big_domain"] = dom
1804
- self._write_node(nid, os.path.join(self.root, e["path"]), fm, content)
2254
+ self._write_node(nid, self._node_disk_path(e), fm, content)
1805
2255
  e["big_domain"] = dom
1806
2256
  # 必须走 _stage:索引持久化靠 _dirty → flush → _index_log 重放,
1807
2257
  # 只改内存 entry 会在重启后丢掉标签(S1 失效,且二次回填因 fm 已有标签而跳过)
@@ -1813,14 +2263,64 @@ class MdCG:
1813
2263
  self.flush()
1814
2264
  return st
1815
2265
 
2266
+ # 生效条件:遍历 index["nodes"](limit 截断);bucket 缺失/orphan/键含中文或可读键为空时计 not_needed 跳过;索引已有 bucket_zh 计 already;get() 不可读计 unreadable;别名=routing.bucket_zh_aliases(可读键, tags, content)(fm 已有则沿用)为空计 not_needed;dry_run 只计 written 不落盘,否则写 fm.bucket_zh + _write_node + _stage(e)(不动 content/其它字段),最终有写入即 flush,返回统计 dict。
2267
+ def backfill_bucket_zh(self, dry_run: bool = False, limit: int = None) -> dict:
2268
+ """为英文桶键的存量节点补中文别名(issue #33;幂等、零翻译依赖)。
2269
+
2270
+ 别名与写入侧同源(`routing.bucket_zh_aliases`:节点自身 tags ∪ 正文
2271
+ 中文域词);只补 `fm.bucket_zh` 与索引快照,不动 content/条件空间;
2272
+ 加密库走 get() → _write_node() 重封装安全路径(同 backfill_big_domain)。
2273
+ 无中文面可用的节点跳过——其在 S1b 的 cross_lang_no_match 审计里可见。
2274
+ """
2275
+ st = {"seen": 0, "already": 0, "written": 0, "not_needed": 0,
2276
+ "unreadable": 0, "dry_run": bool(dry_run)}
2277
+ for nid, e in list((self.index.get("nodes") or {}).items()):
2278
+ if limit and st["written"] >= limit:
2279
+ break
2280
+ st["seen"] += 1
2281
+ b = e.get("bucket")
2282
+ if not b or b == routing.ORPHAN:
2283
+ st["not_needed"] += 1
2284
+ continue
2285
+ kk = routing.bucket_key_readable(b)
2286
+ if not kk or routing._ZH_RE.search(kk):
2287
+ st["not_needed"] += 1 # 中文键无跨语言问题
2288
+ continue
2289
+ if e.get("bucket_zh"):
2290
+ st["already"] += 1
2291
+ continue
2292
+ got = self.get(nid)
2293
+ if not got or got.get("content") is None:
2294
+ st["unreadable"] += 1
2295
+ continue
2296
+ fm, content = got["frontmatter"], got["content"]
2297
+ aliases = fm.get("bucket_zh") or routing.bucket_zh_aliases(
2298
+ kk, fm.get("tags") or e.get("tags"), content)
2299
+ if not aliases:
2300
+ st["not_needed"] += 1 # 无中文面(cross_lang 审计兜底)
2301
+ continue
2302
+ if dry_run:
2303
+ st["written"] += 1
2304
+ continue
2305
+ fm["bucket_zh"] = aliases
2306
+ self._write_node(nid, self._node_disk_path(e),
2307
+ fm, content)
2308
+ e["bucket_zh"] = aliases
2309
+ self._stage(nid, e)
2310
+ st["written"] += 1
2311
+ if not dry_run and st["written"]:
2312
+ self.flush()
2313
+ return st
2314
+
1816
2315
  # ---------- 读 ----------
1817
2316
 
1818
2317
  # 生效条件:index["nodes"].get(node_id) 为假值时回落 self._dirty.get(node_id),仍为假值返回 None;打开 root 下 e["path"] 抛 OSError 返回 None;_open_content 返回 None(无密钥/身份不符)返回 None;否则返回 {id, frontmatter, content, path};
1819
2318
  def get(self, node_id: str):
2319
+ self._maybe_reload_index() # P1b-2:读面代际感知(他进程写快照后可见)
1820
2320
  e = self.index["nodes"].get(node_id) or self._dirty.get(node_id)
1821
2321
  if not e:
1822
2322
  return None
1823
- p = os.path.join(self.root, e["path"])
2323
+ p = self._node_disk_path(e)
1824
2324
  try:
1825
2325
  with open(p, encoding="utf-8") as f:
1826
2326
  fm, content = nodefile.loads(f.read())
@@ -1831,9 +2331,12 @@ class MdCG:
1831
2331
  return None # 有节点但无密钥 → 不可读即不存在
1832
2332
  return {"id": node_id, "frontmatter": fm, "content": content, "path": e["path"]}
1833
2333
 
1834
- # 生效条件:打开 os.path.join(self.root, entry["path"]) 成功时返回 nodefile.loads 的 (fm, content);抛 OSError 时返回 (None, None)(entry 的 "path" 按源码直接取键,无 .get 回落);
2334
+ # 生效条件:经 _node_disk_path(entry) 定位(P2-20 越界抛 ValueError → 返回 (None, None),不可读即不存在——索引被污染时不得绕过统一校验读 root 外文件,2026-09-25 缺陷 #4),打开成功时返回 nodefile.loads 的 (fm, content);抛 OSError 时同样返回 (None, None);
1835
2335
  def _read(self, entry):
1836
- p = os.path.join(self.root, entry["path"])
2336
+ try:
2337
+ p = self._node_disk_path(entry)
2338
+ except ValueError:
2339
+ return None, None
1837
2340
  try:
1838
2341
  with open(p, encoding="utf-8") as f:
1839
2342
  return nodefile.loads(f.read())
@@ -2017,6 +2520,7 @@ class MdCG:
2017
2520
  q = unify_query(q)
2018
2521
  if not q:
2019
2522
  return [], {"tier": None, "reason": "empty_query", "scanned": 0}
2523
+ self._maybe_reload_index() # P1b-2:读面代际感知(空查询早退后、读索引前)
2020
2524
  pool_cfg = pooling.resolve(pooling.from_env(pools))
2021
2525
  now = time.time() if validity else None # 统一取一次 now,保判定口径一致
2022
2526
 
@@ -2027,8 +2531,14 @@ class MdCG:
2027
2531
  # 把 rejected/unresolved 视作可参与召回的特殊「候选池」
2028
2532
  # ——命中它们的结果会改变 meta 的 covered_neg(被负记忆覆盖的查询)
2029
2533
  # 默认排除掉负记忆层的节点进入正排打分,仅作为「覆盖标记」用
2030
- entries = [e for e in self.index["nodes"].values()
2031
- if (not layer or e["layer"] == layer)
2534
+ # P2-2(批次 31):单次遍历双收集——负记忆层引用与正排候选
2535
+ # 同车收集,消除此前的第二遍全索引遍历。
2536
+ neg_layer_entries = []
2537
+ entries = []
2538
+ for e in self.index["nodes"].values():
2539
+ if include_neg and e["layer"] in ("rejected", "unresolved"):
2540
+ neg_layer_entries.append(e)
2541
+ if ((not layer or e["layer"] == layer)
2032
2542
  # '"*"' = 显式跨会话(读遍所有会话);缺省 None 同义
2033
2543
  and (not session or session == "*"
2034
2544
  or e.get("session") == session)
@@ -2039,7 +2549,8 @@ class MdCG:
2039
2549
  # (非法视图 ValueError——fail-closed 只针对调用方误用)
2040
2550
  and (view is None or roleviews.matches(e, view))
2041
2551
  # 时效:只在显式启用时排除已过期(not_yet 保留)
2042
- and not (validity and trust.is_expired(e, now=now))]
2552
+ and not (validity and trust.is_expired(e, now=now))):
2553
+ entries.append(e)
2043
2554
 
2044
2555
  # 默认关:索引里可能残留门控字段(曾开启过 / 回填过)→ 返回前剥离,
2045
2556
  # 保证候选 entry 形状与「从未启用过本功能」逐字节一致(独立复核 2026-09-19)。
@@ -2047,8 +2558,12 @@ class MdCG:
2047
2558
  if (os.environ.get("MDCG_RETRIEVAL_PIPELINE") != "1" and entries
2048
2559
  and any(("big_domain" in e) or ("observation_position" in e)
2049
2560
  for e in entries)):
2050
- entries = [_strip_empty_gate_fields(dict(e), enabled=False)
2051
- for e in entries]
2561
+ # P2-3(批次 31)/ V21-5(批次 34,外部报告定案):**就地剥离**——
2562
+ # entries 里的 e 与 index["nodes"] 是同一对象,_strip_empty_gate_fields
2563
+ # 内部 del 直接落在索引条目上,天然自愈。批次 31 原写法 dict(e) 拷贝
2564
+ # 后按不存在的 "id" 键回写 = 永久 no-op(残留每查都在)。
2565
+ for e in entries:
2566
+ _strip_empty_gate_fields(e, enabled=False)
2052
2567
 
2053
2568
  # ---- 时间算子(阶段二 4.1):候选**资格**过滤(在 S1/S2 收敛之前)----
2054
2569
  # 与 validity 各司其职、互不替代:
@@ -2115,21 +2630,27 @@ class MdCG:
2115
2630
  # 负记忆覆盖:查询词是否已被否决议过
2116
2631
  neg_coverage = []
2117
2632
  if include_neg:
2118
- for e in self.index["nodes"].values():
2119
- if e["layer"] in ("rejected", "unresolved"):
2120
- got = self.get(e["path"].split("/")[-1][:-3])
2121
- # 用 path 末段作为 id 反查(_index 用 path,get 用 id)
2122
- # 上面写法是错的;改用直接路径读
2123
- full_path = os.path.join(self.root, e["path"])
2124
- if not os.path.exists(full_path):
2125
- continue
2126
- try:
2127
- with open(full_path, encoding="utf-8") as f:
2128
- fm, content = nodefile.loads(f.read())
2129
- except OSError:
2130
- continue
2131
- if any(t in content for t in terms):
2132
- neg_coverage.append(e)
2633
+ # P2-2:遍历第一轮同车收集的引用,不再全索引二遍
2634
+ for e in neg_layer_entries:
2635
+ if e["layer"] not in ("rejected", "unresolved"):
2636
+ continue
2637
+ # _index 用 path 存,直接按 path 读盘(批次 24 P2-1:删除
2638
+ # 原先的死调用 `got = self.get(...)`——结果从未使用却完成
2639
+ # 一次完整读盘+解析+解密,每次检索对每个负记忆节点双倍 IO)。
2640
+ # 批次 33 修复:P2-2 双收集重构时本段被整体缩进进 continue
2641
+ # 之后(不可达死代码)——neg_coverage 恒空 → s5 负记忆抑制
2642
+ # 全失效(test_retr_s5 8/17,Windows/Linux 同红;Docker Linux
2643
+ # 验证暴露后主机复跑定案为既有回归,非平台差异)。
2644
+ full_path = self._node_disk_path(e)
2645
+ if not os.path.exists(full_path):
2646
+ continue
2647
+ try:
2648
+ with open(full_path, encoding="utf-8") as f:
2649
+ fm, content = nodefile.loads(f.read())
2650
+ except OSError:
2651
+ continue
2652
+ if any(t in content for t in terms):
2653
+ neg_coverage.append(e)
2133
2654
 
2134
2655
  stat = {"scanned": 0, "query": q, "gates": gates}
2135
2656
  if _tf_audit:
@@ -2139,12 +2660,27 @@ class MdCG:
2139
2660
  ctx = context if isinstance(context, dict) else {}
2140
2661
  route_bucket = routing.bucket_dir(
2141
2662
  routing.route_key(ctx, ctx.get("tags")))
2663
+ # 「无域信号」等价于「没声明条件」:此时**不施加桶约束**。否则 route_key
2664
+ # 回落的 orphan 会被当成一个真桶,下面 T0/T1 阶段只读该桶,并因
2665
+ # valid >= min_results 在 TIER_BUCKET_SCAN 直接 return —— **全量阶梯
2666
+ # 永不执行**,凡被标了 domain:(落在 cond_* 目录)的节点对这条查询
2667
+ # 整片不可见。
2668
+ # 实测(带 context 但不含 domain / observation_position):置 None 前
2669
+ # tier=T1_bucket_scan/scanned=10/目标召回丢失;置 None 后
2670
+ # tier=T2_global_like/scanned=75/召回命中。
2671
+ # 与 mdcos._path_bucket 的「无域信号弃权」(X1)同构,也贴合 test_p0
2672
+ # 的「桶路由平均 recall@10 ≥ 0.9(加 context 不丢召回)」不变量。
2673
+ if route_bucket == routing.ORPHAN:
2674
+ route_bucket = None
2142
2675
 
2143
2676
  # 阶段 1 大域打分已在候选构建前算好(S1 门控要用);此处不再重复计算。
2144
2677
 
2145
2678
  # 生效条件:docs 经 self._score(docs, q, qb, pool_cfg) 后分数 >0 的条数达到闭包阈值 min_results 时返回 self._emit(scored, k, tier, stat, route_bucket, record, len(docs), judge, context, neg_coverage, big_domain, big_scores, pool_cfg)(tier 原样透传);未达阈值返回 None;
2146
- def try_stage(docs, tier):
2147
- scored = self._score(docs, q, qb, pool_cfg)
2679
+ def try_stage(docs, tier, scored=None):
2680
+ # P2-4(批次 31):支持传入已打分结果(reach 分支复用,
2681
+ # 免对同一批文档二次 _score)。
2682
+ if scored is None:
2683
+ scored = self._score(docs, q, qb, pool_cfg)
2148
2684
  valid = sum(1 for _, s in scored if s > 0)
2149
2685
  if valid >= min_results:
2150
2686
  return self._emit(scored, k, tier, stat, route_bucket, record,
@@ -2191,11 +2727,27 @@ class MdCG:
2191
2727
  stat["cap"] = GLOBAL_CAP
2192
2728
  # 与 T2 **无条件**同序调用(不可按 cap 短路:cut_by_relevance 还会写
2193
2729
  # 池账 pool_taken/cands/lost,短路会让 meta.pools.taken 缺失 → test_p43(13) 红)
2194
- hits_r, _rep = cut_by_relevance(hits_r, self._score(hits_r, q, qb, pool_cfg),
2730
+ # P2-4(批次 31):打分结果经 pre_scored 传给 try_stage,
2731
+ # 不再对同一 hits_r 二次 _score(报告 P2-4)。
2732
+ scored_r = self._score(hits_r, q, qb, pool_cfg)
2733
+ hits_r, _rep = cut_by_relevance(hits_r, scored_r,
2195
2734
  GLOBAL_CAP, pools=pool_cfg,
2196
2735
  key_of=pooling.doc_key, stat=stat)
2197
2736
  pooling.record_audit(stat, _rep)
2198
- out = try_stage(hits_r, TIER_GLOBAL_LIKE)
2737
+ # V21-6(批次 34,外部报告定案):doc_key 返回 (node_id, entry),
2738
+ # 第二元是 dict 不可哈希——整键作字典键 = TypeError 死代码。
2739
+ # 取 [0](node_id)作键。缺陷迭代第 14 轮修(两处形态错误,
2740
+ # MDCG_REACH=1 实测 KeyError: 1 崩,stash 取证 HEAD 原生在位):
2741
+ # ① scored 元素是 (card, score),card 是 dict——对它调 doc_key
2742
+ # (形参须 (entry, fm, content) 三元组)= KeyError: 1;按
2743
+ # card["id"](与 doc_key(doc)[0] 同源同值)建键表;
2744
+ # ② 键表须存 (card, score) 原对再按 hits_r 序取回——自组
2745
+ # (doc, score) 对会让 _emit 排序键 x[0]["frontmatter"] 崩
2746
+ # (x[0] 须是 card dict)。
2747
+ _smap = {c.get("id"): (c, s) for c, s in scored_r}
2748
+ out = try_stage(hits_r, TIER_GLOBAL_LIKE,
2749
+ scored=[_smap[pooling.doc_key(d)[0]]
2750
+ for d in hits_r])
2199
2751
  if out:
2200
2752
  return out
2201
2753
  # 收敛阶段未产出结果 → 继续走下方全量 T2/T3;此时必须**改写 reach 审计**,
@@ -2572,11 +3124,16 @@ class MdCG:
2572
3124
  out.append((r[0], r[1], qual))
2573
3125
  # 把被负记忆覆盖的查询也作为结果条目返回(首条),方便调用方感知
2574
3126
  for nc in neg_coverage[:3]:
2575
- full_path = os.path.join(self.root, nc["path"])
2576
- try:
2577
- with open(full_path, encoding="utf-8") as f:
2578
- fm, content = nodefile.loads(f.read())
2579
- except OSError:
3127
+ # v9 留档残余项修复(缺陷迭代第 14 轮):改走 self._read——裸
3128
+ # open+loads 不在读缓存包装面(install 只包 cg._read),每条命中
3129
+ # 负层的查询恒 ≤3 次盘读(60+8 池实测 Q2 增量恰 3);并入读缓存
3130
+ # 后增量归零,负层主面 IO 早已同款覆盖(_neg_coverage → _read)。
3131
+ # _read 同款 _node_disk_path 边界闸(P2-20:索引被污染时不得绕过
3132
+ # 统一校验读 root 外文件)、OSError→(None,None) 与旧 continue 同义;
3133
+ # 负层写路径(add/add_rejected→_stage 标脏带 path)在脏集精确
3134
+ # 失效面内——覆写后尾部条目即时见新内容,不陈旧。
3135
+ fm, content = self._read(nc)
3136
+ if content is None:
2580
3137
  continue
2581
3138
  entry = {"id": nc["path"], "frontmatter": fm,
2582
3139
  "content": content, "path": nc["path"]}
@@ -2759,14 +3316,17 @@ class MdCG:
2759
3316
  return {"id": node_id, "from": from_layer, "to": to_layer,
2760
3317
  "path": rel, "reason": reason}
2761
3318
 
2762
- # 生效条件:verdict 不在 {confirmed,weakened,falsified} 时抛 ValueError;node_id 取不到节点返回 None;verdict=="falsified" 时以 content[:200] 为假设写入 rejected 层、删除原文件并 _unstage,返回 {"action":"falsified","new_id":None,"evidence_count":None,"demoted":None};confirmed/weakened 时 confidence 分别 +0.05 / -0.15(夹到 [0,0.99] 并 round 2)、evidence_count 与 positive_evidence/negative_evidence 各 +1、追加 evidence_log,且仅当 weakened 后 confidence < DEMOTE_CONFIDENCE 且 layer=="knowledge" 时调 _move_layer 到 contextual 并 set_state("demoted"),否则 _write_node 回写并同步索引 evidence_count;
2763
- def verify(self, node_id: str, evidence: str, verdict: str):
3319
+ # 生效条件:verdict 不在 {confirmed,weakened,falsified} 时抛 ValueError;node_id 取不到节点返回 None;verdict=="falsified" 时先经 protect.guard_forget(受保护节点需显式 override=True——旧版本自动快照+留痕;拒绝抛 ProtectionError 且无任何副作用),通过后以 content[:200] 为假设写入 rejected 层、删除原文件并 _unstage,返回 {"action":"falsified","new_id":None,"evidence_count":None,"demoted":None};confirmed/weakened 时 confidence 分别 +0.05 / -0.15(夹到 [0,0.99] 并 round 2)、evidence_count 与 positive_evidence/negative_evidence 各 +1、追加 evidence_log,且仅当 weakened 后 confidence < DEMOTE_CONFIDENCE 且 layer=="knowledge" 时调 _move_layer 到 contextual 并 set_state("demoted"),否则 _write_node 回写并同步索引 evidence_count;
3320
+ def verify(self, node_id: str, evidence: str, verdict: str,
3321
+ override: bool = False):
2764
3322
  """验证单元(白箱第 3 篇第 13 章):对节点做一次外部验证裁决。
2765
3323
 
2766
3324
  verdict ∈ {"confirmed", "weakened", "falsified"}
2767
3325
  - confirmed:positive_evidence+1,confidence 上调 +0.05
2768
3326
  - weakened:negative_evidence+1,confidence 下调 -0.15(反例权重更大)
2769
- - falsified:节点移入 rejected/(幂等:相同 hypothesis 不重复)
3327
+ - falsified:节点移入 rejected/(幂等:相同 hypothesis 不重复);
3328
+ 受保护节点(self/anchor 层、protected 标记、importance≥0.7)
3329
+ 不可一键证伪删除——需显式 override=True(先快照+留痕)
2770
3330
 
2771
3331
  每次裁决累加 evidence_count(白箱第 5 篇第 4 章)。
2772
3332
  weakened 后 confidence 跌破 DEMOTE_CONFIDENCE 且原属 knowledge 层 →
@@ -2778,6 +3338,15 @@ class MdCG:
2778
3338
  if not node:
2779
3339
  return None
2780
3340
  if verdict == "falsified":
3341
+ # 写保护(N130,2026-09-25):证伪 = 对原节点的一次**硬删除**
3342
+ # (os.remove,无快照、不进 trash),与 forget/降级同受「不可遗忘」
3343
+ # 保护——self/anchor 层、protected 标记、importance≥0.7 不得被一次
3344
+ # verify 调用一键删除。对照 forget 路径(mdcos.py forget)同一闸口:
3345
+ # 受保护节点需显式 override=True(旧版本先快照进 _protected_history
3346
+ # 并留痕 _protected_audit.jsonl),拒绝抛 ProtectionError、原节点
3347
+ # 原样保留。守卫必须在 add_rejected 之前——拒绝路径零副作用。
3348
+ protect.guard_forget(self, node_id, override=override,
3349
+ actor=getattr(self, "actor", None))
2781
3350
  # 移入 rejected:负记忆化
2782
3351
  self.add_rejected(hypothesis=node["content"][:200],
2783
3352
  reason=evidence,
@@ -2825,6 +3394,13 @@ class MdCG:
2825
3394
  e = self.index["nodes"].get(node_id)
2826
3395
  if e is not None:
2827
3396
  e["evidence_count"] = fm["evidence_count"]
3397
+ # 落盘必须对索引增量日志(_dirty→flush→_index_log 重放)与
3398
+ # 读缓存代际可见:本分支写盘后若节点已在目标验证态,尾部
3399
+ # set_verification 走 noop 提前返回(trust.py stamp),不会
3400
+ # 经 _sync_index 标脏——不显式置 _dirty 会造成重开后索引
3401
+ # evidence_count 与盘面漂移、同实例 _read 命中旧 frontmatter
3402
+ #(口径同边域同步 _sync_edge_entry / goal 状态写回)。
3403
+ self._dirty[node_id] = e
2828
3404
  # 验证态流转(可验证记忆单元):外部证据裁决**就是**验证态迁移。
2829
3405
  # confirmed → verified(首次转正;已在 verified 则幂等 no-op)
2830
3406
  # weakened → doubted(证据被削弱 ⇒ 存疑,待复核)
@@ -2927,7 +3503,7 @@ class MdCG:
2927
3503
  continue
2928
3504
  fm["access_count"] = int(fm.get("access_count") or 0) + c
2929
3505
  fm["last_access"] = max(float(fm.get("last_access") or 0), last.get(nid, 0))
2930
- self._write_node(nid, os.path.join(self.root, e["path"]), fm, content)
3506
+ self._write_node(nid, self._node_disk_path(e), fm, content)
2931
3507
  n += 1
2932
3508
  with FileLock(self.access_log):
2933
3509
  atomic_write(self.access_log, "")
@@ -2944,7 +3520,7 @@ class MdCG:
2944
3520
  layer_stats = {}
2945
3521
  neg_stats = {}
2946
3522
  for e in self.index["nodes"].values():
2947
- full_path = os.path.join(self.root, e["path"])
3523
+ full_path = self._node_disk_path(e)
2948
3524
  try:
2949
3525
  with open(full_path, encoding="utf-8") as f:
2950
3526
  fm, content = nodefile.loads(f.read())