@furongjun1999/dsh-memory 0.4.8 → 0.4.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (231) hide show
  1. package/README.md +54 -26
  2. package/codebuddy/CODEBUDDY.md +196 -195
  3. package/codebuddy/README.md +13 -1
  4. package/codebuddy/mcp.json +9 -0
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -0
  6. package/docs/README.md +1 -1
  7. package/docs/discipline/harnesses.yaml +18 -7
  8. package/docs/discipline/templates/full.md.tmpl +4 -3
  9. package/docs/experiments/linkref_backfill/candidates_20260917.json +726 -0
  10. package/docs/experiments/linkref_backfill/candidates_internal_20260917.json +602 -0
  11. package/docs/experiments/linkref_backfill/candidates_internal_v2.json +603 -0
  12. package/docs/experiments/linkref_backfill/candidates_secret_20260917.json +884 -0
  13. package/docs/experiments/linkref_backfill/candidates_secret_v2.json +789 -0
  14. package/docs/hive//345/244/232/347/253/257harness/351/200/232/344/277/241/345/245/221/347/272/246_v0.1.md +42 -0
  15. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -0
  16. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -0
  17. package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S1b/344/270/216/345/217/254/345/233/236/346/235/203/350/241/241_v0.1.md +57 -0
  18. package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S7/345/200/222/346/216/222/345/200/231/351/200/211/345/261/202_v0.1.md +126 -0
  19. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -0
  20. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -0
  21. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.5.md +631 -625
  22. package/docs/mdcg//344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/345/245/221/347/272/246_v0.1.md +82 -0
  23. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -0
  24. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +47 -45
  25. package/docs/mdcg//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/{206_v0.4.md → 206_v0.5.md} +92 -4
  26. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -0
  27. package/docs/mdcg//347/216/257/344/272/214_/347/231/275/347/256/261/345/241/253/345/205/205/346/265/201/346/260/264/347/272/277_v0.1.md +31 -0
  28. package/docs/mdcg//350/267/250/347/253/257/351/252/214/350/257/201/344/270/216/345/220/214/346/255/245/345/215/217/350/256/256_v0.1.md +93 -0
  29. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +2 -2
  30. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +3 -3
  31. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +2 -2
  32. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +434 -433
  33. package/docs//347/201/265/346/236/242/350/207/252/346/210/221/346/224/271/350/277/233/345/267/245/344/275/234/350/256/241/345/210/222_/345/244/226/351/203/250/347/240/224/347/251/266/347/263/273/345/210/227/345/220/270/346/224/266_v1_20260919.md +286 -0
  34. package/dsh/README.md +33 -0
  35. package/dsh/cordis.yml.example +13 -7
  36. package/dsh/hive-mcp-probe.mjs +94 -0
  37. package/dsh/hive-mcp.example.yml +62 -0
  38. package/dsh/update-lingshu.bat +11 -0
  39. package/dsh/update-lingshu.ps1 +337 -0
  40. package/lib/hooks.d.ts +3 -0
  41. package/lib/hooks.js +17 -23
  42. package/lib/index.d.ts +4 -2
  43. package/lib/index.js +29 -6
  44. package/lib/lib/datapath.d.ts +76 -1
  45. package/lib/lib/datapath.js +199 -13
  46. package/lib/lib/mdcg_client.d.ts +6 -3
  47. package/lib/lib/mdcg_client.js +6 -5
  48. package/lib/lib/mutual.js +4 -4
  49. package/lib/lib/token_store.js +4 -5
  50. package/md_cg/audit.py +17 -2
  51. package/md_cg/autonomy.py +86 -15
  52. package/md_cg/backfill.py +36 -1
  53. package/md_cg/backfill_bigdomain.py +34 -0
  54. package/md_cg/bench6_arms.py +28 -1
  55. package/md_cg/bench6_common.py +10 -1
  56. package/md_cg/bench6_competitors.py +6 -1
  57. package/md_cg/bench_axis_domain.py +9 -1
  58. package/md_cg/bench_blind_comp.py +7 -1
  59. package/md_cg/bench_en_atoms_public.py +9 -0
  60. package/md_cg/bench_governance.py +348 -0
  61. package/md_cg/bench_lme_zh.py +16 -1
  62. package/md_cg/bench_locomo.py +2 -1
  63. package/md_cg/bench_locomo_zh.py +16 -1
  64. package/md_cg/bench_locomo_zh_public.py +4 -1
  65. package/md_cg/bench_longmem.py +2 -1
  66. package/md_cg/bench_membench.py +27 -1
  67. package/md_cg/bench_p0.py +4 -1
  68. package/md_cg/bench_progressive.py +13 -1
  69. package/md_cg/bench_role_views.py +238 -0
  70. package/md_cg/bench_task_ab.py +8 -1
  71. package/md_cg/bench_task_ab_llm.py +13 -1
  72. package/md_cg/bench_unified_en.py +6 -1
  73. package/md_cg/bench_zh_mad.py +20 -1
  74. package/md_cg/blindspot_tickets.py +123 -0
  75. package/md_cg/branches.py +12 -1
  76. package/md_cg/build_postings.py +73 -0
  77. package/md_cg/ccgc.py +67 -2
  78. package/md_cg/census.py +5 -1
  79. package/md_cg/chain.py +24 -3
  80. package/md_cg/codeindex.py +134 -17
  81. package/md_cg/coldverify.py +265 -0
  82. package/md_cg/comment_gate.py +338 -0
  83. package/md_cg/cond_compose.py +190 -0
  84. package/md_cg/cond_facts.py +155 -0
  85. package/md_cg/cond_template.json +107 -0
  86. package/md_cg/condition_anchor.py +143 -0
  87. package/md_cg/conformance.py +69 -4
  88. package/md_cg/consistency.py +24 -1
  89. package/md_cg/consolidate.py +53 -2
  90. package/md_cg/corpus.py +4 -0
  91. package/md_cg/crosscheck.py +42 -2
  92. package/md_cg/crypto.py +35 -1
  93. package/md_cg/d_meta.py +310 -0
  94. package/md_cg/datapath.py +201 -26
  95. package/md_cg/docindex.py +122 -1
  96. package/md_cg/eval_common.py +29 -1
  97. package/md_cg/evidence.py +27 -1
  98. package/md_cg/evolution.py +21 -1
  99. package/md_cg/export.py +11 -1
  100. package/md_cg/forgetting.py +23 -1
  101. package/md_cg/fsutil.py +18 -1
  102. package/md_cg/hotcache.py +214 -0
  103. package/md_cg/hyperedge.py +251 -0
  104. package/md_cg/identity.py +18 -1
  105. package/md_cg/insight.py +17 -1
  106. package/md_cg/lexicon/build_cedict_en_zh.py +9 -0
  107. package/md_cg/lexicon/build_standard_en.py +171 -168
  108. package/md_cg/lexicon/expand_en_zh.py +6 -0
  109. package/md_cg/lifecycle.py +12 -1
  110. package/md_cg/linkref.py +281 -0
  111. package/md_cg/links.py +29 -1
  112. package/md_cg/mcp_server.py +362 -43
  113. package/md_cg/md_whitebox.py +53 -1
  114. package/md_cg/mdcg.py +1003 -27
  115. package/md_cg/mdcos.py +558 -36
  116. package/md_cg/metacognition.py +37 -2
  117. package/md_cg/migrate.py +4 -0
  118. package/md_cg/migrate_aeis.py +221 -213
  119. package/md_cg/migrate_roleplay.py +8 -0
  120. package/md_cg/migrate_wisdom_graph.py +14 -1
  121. package/md_cg/mreview/__main__.py +3 -0
  122. package/md_cg/mreview/bundle.py +8 -0
  123. package/md_cg/mreview/candidates.py +9 -0
  124. package/md_cg/mreview/govern.py +21 -1
  125. package/md_cg/mreview/locate.py +34 -0
  126. package/md_cg/mreview/pipeline.py +29 -1
  127. package/md_cg/mreview/ruleset.py +16 -1
  128. package/md_cg/nodefile.py +233 -3
  129. package/md_cg/pooling.py +23 -1
  130. package/md_cg/postings.py +298 -0
  131. package/md_cg/predict.py +89 -9
  132. package/md_cg/progressive.py +3 -0
  133. package/md_cg/protect.py +14 -1
  134. package/md_cg/protocol.py +372 -0
  135. package/md_cg/provenance.py +262 -0
  136. package/md_cg/reach.py +453 -0
  137. package/md_cg/refindex.py +47 -2
  138. package/md_cg/refine.py +20 -1
  139. package/md_cg/roleviews.py +89 -0
  140. package/md_cg/routing.py +76 -0
  141. package/md_cg/scrub.py +63 -2
  142. package/md_cg/security.py +26 -1
  143. package/md_cg/self_state.py +64 -1
  144. package/md_cg/selfreport.py +151 -0
  145. package/md_cg/semantic/canonical.py +5 -0
  146. package/md_cg/semantic/en_normalizer.py +364 -355
  147. package/md_cg/semantic/zh_en_atoms.py +139 -136
  148. package/md_cg/signer.py +41 -1
  149. package/md_cg/sources.py +583 -547
  150. package/md_cg/statushdr.py +179 -0
  151. package/md_cg/stg.py +59 -18
  152. package/md_cg/subgraph.py +23 -0
  153. package/md_cg/sustain.py +56 -1
  154. package/md_cg/tasks.py +26 -2
  155. package/md_cg/test_autonomy.py +26 -0
  156. package/md_cg/test_bench_governance.py +102 -0
  157. package/md_cg/test_blindspot_tickets.py +166 -0
  158. package/md_cg/test_ccgc.py +10 -0
  159. package/md_cg/test_codeindex.py +338 -0
  160. package/md_cg/test_comment_gate.py +187 -0
  161. package/md_cg/test_cond_compose_anchors.py +76 -0
  162. package/md_cg/test_condition_anchor.py +82 -0
  163. package/md_cg/test_d_meta.py +412 -0
  164. package/md_cg/test_datapath_root.py +188 -0
  165. package/md_cg/test_gain_gate.py +47 -1
  166. package/md_cg/test_hot_cold.py +187 -0
  167. package/md_cg/test_hyperedge.py +245 -0
  168. package/md_cg/test_linkref.py +306 -0
  169. package/md_cg/test_md_access_parity.py +15 -3
  170. package/md_cg/test_mr_m1.py +108 -18
  171. package/md_cg/test_mr_m3.py +8 -1
  172. package/md_cg/test_p26_refindex.py +49 -20
  173. package/md_cg/test_p27_docindex.py +236 -2
  174. package/md_cg/test_p2_mcp.py +1 -1
  175. package/md_cg/test_p31_insight.py +24 -0
  176. package/md_cg/test_p44_md_whitebox.py +14 -1
  177. package/md_cg/test_protocol.py +243 -0
  178. package/md_cg/test_reach.py +378 -0
  179. package/md_cg/test_reach_keys.py +201 -0
  180. package/md_cg/test_reach_meta_exits.py +145 -0
  181. package/md_cg/test_read_clip.py +8 -4
  182. package/md_cg/test_retr_s1.py +340 -0
  183. package/md_cg/test_retr_s1b.py +209 -0
  184. package/md_cg/test_retr_s3.py +194 -0
  185. package/md_cg/test_retr_s4.py +163 -0
  186. package/md_cg/test_retr_s5.py +200 -0
  187. package/md_cg/test_retr_s6.py +157 -0
  188. package/md_cg/test_retr_s7.py +385 -0
  189. package/md_cg/test_retr_s8_time.py +316 -0
  190. package/md_cg/test_retr_s9_edges.py +286 -0
  191. package/md_cg/test_retr_s9_entity_ctx.py +175 -0
  192. package/md_cg/test_review_conformance.py +59 -2
  193. package/md_cg/test_role_views.py +354 -0
  194. package/md_cg/test_trust.py +361 -0
  195. package/md_cg/test_units_poll.py +71 -0
  196. package/md_cg/test_v14_fixes.py +397 -0
  197. package/md_cg/test_validity_filter.py +280 -0
  198. package/md_cg/test_wisdom_md_store.py +7 -3
  199. package/md_cg/test_writepipe.py +5 -1
  200. package/md_cg/theory.py +16 -1
  201. package/md_cg/tokens.py +40 -8
  202. package/md_cg/tool_face.py +13 -2
  203. package/md_cg/trust.py +943 -0
  204. package/md_cg/twophase.py +12 -1
  205. package/md_cg/units.py +129 -10
  206. package/md_cg/vision_evidence.py +24 -1
  207. package/md_cg/weights.py +24 -1
  208. package/md_cg/whitebox.py +32 -1
  209. package/md_cg/whitebox_kb/data/verify_cache.json +21210 -365
  210. package/md_cg/whitebox_kb/data/verify_savings.jsonl +5078 -0
  211. package/md_cg/whitebox_kb/wisdom/audit_log/chain_heat.json +10 -10
  212. package/md_cg/whitebox_kb/wisdom/code_compose.py +113 -6
  213. package/md_cg/whitebox_kb/wisdom/code_solidified.json +1 -1
  214. package/md_cg/whitebox_kb/wisdom/verifier.py +340 -55
  215. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db +0 -0
  216. package/md_cg/writelimit.py +18 -4
  217. package/md_cg/writepipe.py +178 -7
  218. package/package.json +2 -2
  219. package/skills/skills/designer-perspective/scripts/__pycache__/designer.cpython-310.pyc +0 -0
  220. package/skills/skills/designer-perspective/scripts/designer.py +17 -1
  221. package/skills/skills/designer-perspective/tests/selftest.py +3 -1
  222. package/src/hooks.ts +17 -21
  223. package/src/index.ts +33 -6
  224. package/src/lib/datapath.ts +211 -13
  225. package/src/lib/mdcg_client.ts +12 -8
  226. package/src/lib/mutual.ts +411 -411
  227. package/src/lib/token_store.ts +4 -5
  228. package/zcode/AGENTS.md +196 -195
  229. package/zcode/README.md +4 -0
  230. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-shm +0 -0
  231. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-wal +0 -0
@@ -134,6 +134,7 @@ for _c in CASES:
134
134
 
135
135
 
136
136
  # ---- LLM 客户端(OpenAI 兼容,零第三方依赖)-------------------------------
137
+ # 生效条件:给定 model/base/key/messages 即构造 JSON POST 到 base.rstrip("/")+"/chat/completions",返回 (choices[0].message 的 content 或缺失/假值回落 ""、data.get("usage") 或缺失/假值回落 {}、耗时 dt),timeout/max_tokens/temperature 仅作为请求参数传入。
137
138
  def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
138
139
  temperature=0.0):
139
140
  payload = json.dumps({
@@ -154,6 +155,7 @@ def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
154
155
  return content, (data.get("usage") or {}), dt
155
156
 
156
157
 
158
+ # 生效条件:raw(None/空串视为 "")中首个 "{" 至末个 "}" 的子串能解析出 pick,或全文匹配到单个数字 1-9,且该编号落在 1..n 内时返回该编号(pick 为数字字符串先转 int),否则返回 None。
157
159
  def _parse_pick(raw, n):
158
160
  """从模型输出里抠出候选编号(1..n);解析失败返回 None。"""
159
161
  s = raw or ""
@@ -174,6 +176,7 @@ def _parse_pick(raw, n):
174
176
  return None
175
177
 
176
178
 
179
+ # 生效条件:以 case["id"] 播种打乱 [(fix,case["fix"]),(trap,case["trap"]),(decoy,case["decoy"])] 后取三种循环排列,n<=3 返回其前 n 个,n>3 返回 6 种全排列的前 n 个。
177
180
  def _orders(case, n):
178
181
  """候选顺序。默认取 3 种循环排列(每个候选在 3 个位置各出现一次,天然去位置偏差);
179
182
  n>3 时退回全 6 种排列。基准顺序由 case id 播种打乱,避免跨用例雷同。"""
@@ -186,6 +189,7 @@ def _orders(case, n):
186
189
  return list(itertools.permutations(base))[:n]
187
190
 
188
191
 
192
+ # 生效条件:case 含 task 且 opts 为 (kind, txt) 序列时,返回含场景、编号候选与选择指令的提示文本,memory 非空时在开头插入记忆段。
189
193
  def _user_prompt(case, opts, memory=None):
190
194
  lines = []
191
195
  if memory:
@@ -203,12 +207,14 @@ def _user_prompt(case, opts, memory=None):
203
207
  return "\n".join(lines)
204
208
 
205
209
 
210
+ # 生效条件:cg 存在时把模块常量 CASES 逐条以 c["task"] 为 error、c["lesson"] 为 fix 组成列表传给 cg.mine_fix_pairs 并原样返回其结果。
206
211
  def _build_memory(cg):
207
212
  """Phase A:把 5 条「项目规范」写进记忆(真实 mine_fix_pairs)。"""
208
213
  return cg.mine_fix_pairs(
209
214
  [{"error": c["task"], "fix": c["lesson"]} for c in CASES])
210
215
 
211
216
 
217
+ # 生效条件:以 "本项目规范:"+case["query"] 且 budget_tokens=budget、k=10 调用 cg.recall,取 pack 各项 content(缺失/假值为 "")拼接,返回其前 1500 字符与 case["marker"] 是否出现在未截断拼接文本中。
212
218
  def _recall_memory(cg, case, budget):
213
219
  res = cg.recall("本项目规范:" + case["query"],
214
220
  budget_tokens=budget, k=10)
@@ -216,6 +222,7 @@ def _recall_memory(cg, case, budget):
216
222
  return text[:1500], (case["marker"] in text)
217
223
 
218
224
 
225
+ # 生效条件:在 cfg["max_turns"] 轮内以 cfg["model"]/cfg["base"]/cfg["key"] 调 llm_chat(timeout=cfg["timeout"]、max_tokens=cfg["max_tokens"])并累加 usage["total_tokens"],pick 等于 correct_pos 或为 None 时中止(否则在仍有剩余轮次时追加 assistant/user 消息重选),最终 pick 为 None 则 kind="invalid"、否则 kind=opts[pick-1][0]。
219
226
  def _run_one(cfg, case, opts, correct_pos, memory):
220
227
  messages = [{"role": "system", "content": SYS_PROMPT},
221
228
  {"role": "user", "content": _user_prompt(case, opts, memory)}]
@@ -239,6 +246,7 @@ def _run_one(cfg, case, opts, correct_pos, memory):
239
246
  "turns": turns, "tokens": tokens, "latency": dt}
240
247
 
241
248
 
249
+ # 生效条件:cfg["use_mem"] 为真时对每个 case 以 cfg["budget"] 先串行召回写入 mems、否则存 (None, False),再对每 case × _orders(case, n_perms) 的排列 × ("none","mem") 两臂建任务交 ThreadPoolExecutor(max_workers=workers) 执行,返回按 arm 分组的 {"none": [...], "mem": [...]} 行表。
242
250
  def _run_model(label, cfg, cases, cg, n_perms, workers):
243
251
  # 记忆召回先串行算好(MdCGOS 非线程安全),LLM 调用再并发。
244
252
  mems = {}
@@ -254,6 +262,7 @@ def _run_model(label, cfg, cases, cg, n_perms, workers):
254
262
  for arm in ("none", "mem"):
255
263
  tasks.append((case, oi, opts, correct_pos, arm))
256
264
 
265
+ # 生效条件:t 解包为 (case, oi, opts, correct_pos, arm),arm=="mem" 时 memory/hit 取闭包 mems[case["id"]]、否则为 (None, False),_run_one 抛异常时以 kind="error"、pick=None 的占位行替代,随后补上 case/arm/hit 并打印该行后返回 r。
257
266
  def work(t):
258
267
  case, oi, opts, correct_pos, arm = t
259
268
  memory, hit = mems[case["id"]] if arm == "mem" else (None, False)
@@ -276,10 +285,12 @@ def _run_model(label, cfg, cases, cg, n_perms, workers):
276
285
  return rows
277
286
 
278
287
 
288
+ # 生效条件:n 为真值时返回 f"{100.0*x/n:.1f}%",n 为假值(0)时返回 "-"。
279
289
  def _pct(x, n):
280
290
  return f"{100.0 * x / n:.1f}%" if n else "-"
281
291
 
282
292
 
293
+ # 生效条件:以 rows["none"] 的行数 n 汇总 none/mem 两臂的 correct、kind=="trap"、kind∈("invalid","error")、tokens、latency、hit 与 pick∈(1,2,3) 的分布并打印(n_perms 仅用于表头文字),返回该 agg 字典。
283
294
  def _report(label, rows, n_perms):
284
295
  n = len(rows["none"])
285
296
  print(f"\n{'-' * 78}")
@@ -333,6 +344,7 @@ def _report(label, rows, n_perms):
333
344
  return agg
334
345
 
335
346
 
347
+ # 生效条件:argv 为 None 时改读 sys.argv;若 --key 与 DEEPSEEK_API_KEY 均为空则返回 2,否则按 --only/--cases 从 CASES 取用例跑完双臂后返回 0。
336
348
  def main(argv=None):
337
349
  ap = argparse.ArgumentParser(
338
350
  description="先验陷阱 A/B:无记忆 vs 有记忆(真实 LLM)")
@@ -394,4 +406,4 @@ def main(argv=None):
394
406
 
395
407
 
396
408
  if __name__ == "__main__":
397
- sys.exit(main())
409
+ sys.exit(main())
@@ -48,6 +48,7 @@ ARM_ROOT = os.path.join(HERE, "_md_cg_eval_unified_en") # gitignore 已覆盖
48
48
  K = 10
49
49
 
50
50
 
51
+ # 生效条件:传入 root 使 os.path.isdir(root) 为真时先执行 shutil.rmtree(root, ignore_errors=True)、否则跳过,随后遍历 corpus 每项 c,semantic_of 不为 None 且 semantic_of(c) 返回真值时以 semantic=sem 传入,正文按 zh_body 真值取 c.get("zh")、假值取 b6.ingest_text_en(c),并以 c["id"] 为键 add 后 flush 并 install_read_cache,最终返回 cg;
51
52
  def build(root, corpus, semantic_of=None, zh_body=False):
52
53
  """单库构建:正文形态与 fm.semantic 摘要层是仅有的两个自由度。"""
53
54
  if os.path.isdir(root):
@@ -67,6 +68,7 @@ def build(root, corpus, semantic_of=None, zh_body=False):
67
68
  return cg
68
69
 
69
70
 
71
+ # 生效条件:cg 可对 questions 做 evaluate_group 时,按 semantic 设置或清除 MDCG_SEMANTIC,返回词法路 rows 及其按 qtype 汇总的 by_type。
70
72
  def run_arm(cg, questions, tag, semantic=False):
71
73
  if semantic:
72
74
  os.environ["MDCG_SEMANTIC"] = "1"
@@ -82,6 +84,7 @@ def run_arm(cg, questions, tag, semantic=False):
82
84
  return rows, by_type
83
85
 
84
86
 
87
+ # 生效条件:questions 为真时逐 q 取 q["question"](缺该键抛 KeyError)经 query_atoms,仅当 atom 含任一满足 "A" <= ch <= "z" 的字符才计入 n_keep,n_tok 为 0 时 keep_ratio 为 0.0、否则 round(n_keep/n_tok,4),questions 为空时 q_with_keep_ratio 为 0.0、否则 round(n_q_with_keep/len(questions),4),返回含 n_questions/n_atoms/n_en_keep 的 dict;
85
88
  def oov_stats(questions):
86
89
  """题级 en→zh 归一覆盖统计:英文保留词占比 = 词表覆盖瓶颈的直接量化。"""
87
90
  n_tok = n_keep = n_q_with_keep = 0
@@ -100,6 +103,7 @@ def oov_stats(questions):
100
103
  if questions else 0.0}
101
104
 
102
105
 
106
+ # 生效条件:不适用(无必需形参与模块级常量)
103
107
  def main():
104
108
  t0 = time.time()
105
109
  rows500 = list(ec.iter_jsonl(b6.QUESTIONS500))
@@ -115,6 +119,7 @@ def main():
115
119
  ec.unlock_global_cap()
116
120
  ec.use_jaccard()
117
121
 
122
+ # 生效条件:对 c 先取 (c.get("zh_fields") or {}).get("summary") 再 or "" 并 str().strip(),该值为空时回落为 (c.get("zh") or "") 的 str().strip(),所得 s 非空则返回 " ".join(semantic_atoms(s))、否则返回空串;
118
123
  def sem_of(c):
119
124
  s = str((c.get("zh_fields") or {}).get("summary") or "").strip()
120
125
  if not s:
@@ -197,4 +202,4 @@ def main():
197
202
 
198
203
 
199
204
  if __name__ == "__main__":
200
- main()
205
+ main()
@@ -62,6 +62,7 @@ QUESTIONS20 = os.path.join(ZP, "questions20.jsonl")
62
62
  AXES = ("person", "event", "time", "identity", "place", "condition")
63
63
 
64
64
 
65
+ # 生效条件:path 以 UTF-8 打开后逐行读取,仅 strip 后非空的行经 json.loads 产出,空白行被跳过。
65
66
  def iter_jsonl(path):
66
67
  with open(path, encoding="utf-8") as f:
67
68
  for line in f:
@@ -70,11 +71,13 @@ def iter_jsonl(path):
70
71
  yield json.loads(line)
71
72
 
72
73
 
74
+ # 生效条件:path 以 UTF-8 打开成功时返回 json.load(f) 的解析结果,片段内无其它分支。
73
75
  def load_json(path):
74
76
  with open(path, encoding="utf-8") as f:
75
77
  return json.load(f)
76
78
 
77
79
 
80
+ # 生效条件:raw 经 str() 后按“;”切分,仅含“=”的段被解析,其中键为身份/时间/摘要时写对应槽位、键为条件时按“|”与“:”写入 condition、键为词时按“,”拆出非空项写入 terms,其它键或未出现的槽位保持 out 的默认空值。
78
81
  def parse_zh(raw):
79
82
  """中文层串 → 槽位 dict。
80
83
 
@@ -107,6 +110,7 @@ def parse_zh(raw):
107
110
  return out
108
111
 
109
112
 
113
+ # 生效条件:当 MANUAL_ZH/MANUAL_Q/LM_Q/LM_H 可读取、hay 覆盖 want_ids 且 meta 含 man_q 全部 qid 时写出 CORPUS20/QUESTIONS20 并返回 (corpus, questions),缺 turn 或 qid 分别 raise SystemExit,verbose(默认 True)为真值时额外打印统计、假值时静默。
110
114
  def prepare(verbose=True):
111
115
  """生成 corpus20.jsonl 与 questions20.jsonl(幂等覆盖)。"""
112
116
  man_zh = load_json(MANUAL_ZH)
@@ -125,6 +129,7 @@ def prepare(verbose=True):
125
129
  raise SystemExit(f"[失败] haystack 缺以下 turn:{sorted(missing)}")
126
130
 
127
131
  # corpus:按 turn id 的数值序(= 时间序),保证「承接前一条」有确定语义
132
+ # 生效条件:tid 经 rsplit("t",1) 得到至少两段且最后一段可被 int() 解析时返回该整数,否则源码未做校验会抛错。
128
133
  def turn_no(tid):
129
134
  return int(tid.rsplit("t", 1)[1])
130
135
 
@@ -171,6 +176,7 @@ def prepare(verbose=True):
171
176
  return corpus, questions
172
177
 
173
178
 
179
+ # 生效条件:prepare(verbose=False) 返回 corpus/questions 后,对每题以 evidence_turns[0](空则 "")取 ev0,命中词限于 ev0 非空且词出现在 zh_of[ev0] 中,df1 收集池内 df==1 的词、all_df 收集 df==len(corpus) 的词,verbose(默认 True)为真值时打印后返回 per_q、为假值时直接返回 per_q。
174
180
  def analyze(verbose=True):
175
181
  """诊断:查询词能否落到 gold 中文层 / 池内其他条目(决定各轴是否有可桥接的实体)。
176
182
 
@@ -254,6 +260,7 @@ EN_NAME_RE = re.compile(r"\b[A-Z][a-z]{2,}(?:[ ][A-Z][a-z]{2,})*\b")
254
260
  DATE_RE = re.compile(r"(\d{4})[/-](\d{1,2})[/-](\d{1,2})")
255
261
 
256
262
 
263
+ # 生效条件:DATE_RE.search(str(raw or "")) 有匹配时返回零填充的 YYYY-MM-DD,无匹配(含 raw 为假值回落成 "")时返回 ""。
257
264
  def norm_time(raw):
258
265
  """'2023/05/24 04:49' → '2023-05-24'(跨条目统一格式,使日期查询可命中)。"""
259
266
  m = DATE_RE.search(str(raw or ""))
@@ -263,6 +270,7 @@ def norm_time(raw):
263
270
  return f"{int(y):04d}-{int(mo):02d}-{int(d):02d}"
264
271
 
265
272
 
273
+ # 生效条件:对 str(text or "") 用 EN_NAME_RE 扫描,匹配串按空白切分后任一部分命中 EN_STOP 即跳过,其余项按首次出现顺序去重加入 out 并返回;text 为假值时扫描空串返回 []。
266
274
  def en_names(text):
267
275
  """英文原文中的专名候选(跨语言桥:中文查询里的英文专名可经 entity 路命中)。"""
268
276
  out, seen = [], set()
@@ -276,6 +284,7 @@ def en_names(text):
276
284
  return out
277
285
 
278
286
 
287
+ # 生效条件:遍历 corpus,对每条记录的 zh_fields["terms"] 去重后逐词判断,词非空且不在 STOP_ZH 时使 df[词] 计数加一,返回 df。
279
288
  def build_tables(corpus):
280
289
  """全库统计:# 词项 df(**只读语料**,与查询集无关)。"""
281
290
  df = {}
@@ -286,6 +295,7 @@ def build_tables(corpus):
286
295
  return df
287
296
 
288
297
 
298
+ # 生效条件:c 含 zh_fields(terms/condition/time/identity)且 df 为词到频次的映射时,返回 person/event/time/identity/place/condition 六轴取值字典。
289
299
  def axis_values(c, df):
290
300
  """一条 turn 的六个关系轴取值(用户裁决:条件链 + 人物/事件/时间/身份/地点)。"""
291
301
  f = c["zh_fields"]
@@ -321,6 +331,7 @@ INTENT_CLASSES = (
321
331
  )
322
332
 
323
333
 
334
+ # 生效条件:按 INTENT_CLASSES 顺序检查 c["zh_fields"]["summary"],首个命中类目的首个包含于摘要的动词返回 (cls, v),全不命中返回 ('', '')。
324
335
  def intent_of(c):
325
336
  """意图抽象:摘要 → (规范类目, 命中的动词原形)。
326
337
 
@@ -338,6 +349,7 @@ def intent_of(c):
338
349
  return "", ""
339
350
 
340
351
 
352
+ # 生效条件:把 c["zh_fields"]["terms"] 与 en_names(c["text"]) 逐项 strip 后跳过空串、STOP_ZH(原形或小写)及已见项去重入 out,再用 norm_time(条件.get("时间窗口") or c 的 time) 得到非空且未出现的时窗串追加;df 形参在该片段内未参与条件判断。
341
353
  def normalize_terms(c, df):
342
354
  """实体规范化:去停用 + 去重 + 附时间规范式 + 附英文专名(跨语言对齐)。"""
343
355
  f = c["zh_fields"]
@@ -354,6 +366,7 @@ def normalize_terms(c, df):
354
366
  return out
355
367
 
356
368
 
369
+ # 生效条件:遍历 corpus 并以 root=os.path.join(HERE, root_base or f"_md_cg_eval_zhprobe_{arm['name']}")(root_base 为 None/空串时回落)建库,graph 分支仅当 arm.get("graph") 为真且某轴取值的同值 id 数落在 [2, max_df](默认 5)时为这些 id 两两建有向边,coref 仅当 arm.get("coref") 为真、本条 own 为空(own 只在 arm.get("norm") 为真时由 normalize_terms 生成,否则恒为 [])且 prev_own 非空时承接前一条自身 canonical,intent 分支仅当 arm.get("intent") 为真且 intent_of 返回 cls 非空时写意图行并把意图词与长度>=2 的动词加入 tags,verbose(默认 True)为真值时打印臂统计、root 已是目录时先整树删除再建 MdCGOS。
357
370
  def build_arm(corpus, arm, max_df=5, root_base=None, verbose=True):
358
371
  """按消融臂建库。每臂一个独立 root,互不污染。
359
372
 
@@ -453,6 +466,7 @@ ARMS = [
453
466
  ]
454
467
 
455
468
 
469
+ # 生效条件:先执行 prepare(verbose=False) 取得 corpus,再对 ARMS 每臂以 max_df=max_df(默认 5,原样下传不做假值回落)调用 build_arm 并汇总为 roots 返回。
456
470
  def build_all(max_df=5):
457
471
  corpus, _ = prepare(verbose=False)
458
472
  print(f"== 建库(消融 {len(ARMS)} 臂,max_df={max_df})==")
@@ -469,6 +483,7 @@ ROW_RE = re.compile(
469
483
  GROUPS = ["precise", "temporal", "interference", "reference"]
470
484
 
471
485
 
486
+ # 生效条件:os.path.exists(RUST_BIN) 为真时以 argv=[RUST_BIN,"--dataset","mad","--tag",name,"--lib",lib](extra 为真值时追加 list(extra))执行 subprocess,返回码非 0 或 ROW_RE 在 stdout 未匹配到任何组时 raise SystemExit,否则返回 {组:(n,hit@1,hit@5,MRR)}。
472
487
  def run_one(name, lib, extra=None):
473
488
  """调用 **Rust 检索器** 跑一臂,返回 {组: (n, hit@1, hit@5, MRR)}。
474
489
 
@@ -496,6 +511,7 @@ def run_one(name, lib, extra=None):
496
511
  return got
497
512
 
498
513
 
514
+ # 生效条件:先打印 title 与表头,再遍历 rows 的 (label, got, note),对 GROUPS 每组用 got[g] 取 (n,h1,h5,mrr) 并累计总体 hit@1/MRR,仅当 baseline 不为 None 且 label==baseline 时记 ref,仅当 ref 已记录且 label!=baseline 时输出 Δ。
499
515
  def print_table(title, rows, baseline=None):
500
516
  """rows: [(label, got, note)];baseline = 参照行 label(算 Δ)。"""
501
517
  print(f"\n== {title} ==")
@@ -524,6 +540,7 @@ def print_table(title, rows, baseline=None):
524
540
  + f"{ov_h1 * 100:>10.1f}%{ov_mrr:>10.3f}{delta}{suffix}")
525
541
 
526
542
 
543
+ # 生效条件:对 ARMS 每臂以 lib=f"_md_cg_eval_zhprobe_{name}"、无 extra 调用 run_one 组成 rows,并以 ARMS[0]["name"] 为 baseline 调 print_table 后返回 rows。
527
544
  def run_ablation():
528
545
  """消融主表:逐臂调用 Rust 检索器并汇总(种子口径 = 缺省,即生产现状)。"""
529
546
  rows = []
@@ -537,6 +554,7 @@ def run_ablation():
537
554
  return rows
538
555
 
539
556
 
557
+ # 生效条件:以最后一个臂 ARMS[-1]['name'] 对应的库路径,先无 extra 调 run_one("a4_index", lib)、再以 extra=("--graph-seeds","sorted") 调 run_one("a4_sorted", lib) 组成 rows,并以 baseline="a4/index" 调 print_table 后返回。
540
558
  def run_seed_control():
541
559
  """对照:**同一个 a4 库**(写入侧与边结构完全相同),只切换 graph 路种子口径。
542
560
 
@@ -555,6 +573,7 @@ def run_seed_control():
555
573
  return rows
556
574
 
557
575
 
576
+ # 生效条件:cmd 取 argv[1](len(argv)<=1 时为 "prepare"),cmd=="prepare"/"analyze"/"build"/"run"/"seed-control" 分别调用 prepare/analyze/build_all/run_ablation/run_seed_control,cmd=="all" 依次调用 prepare、analyze、build_all、run_ablation、run_seed_control,其余值 raise SystemExit。
558
577
  def main(argv):
559
578
  cmd = argv[1] if len(argv) > 1 else "prepare"
560
579
  if cmd == "prepare":
@@ -580,4 +599,4 @@ def main(argv):
580
599
 
581
600
 
582
601
  if __name__ == "__main__":
583
- main(sys.argv)
602
+ main(sys.argv)
@@ -0,0 +1,123 @@
1
+ # -*- coding: utf-8 -*-
2
+ """盲区消解票据(阶段三 §5.4):BLINDSPOT/DEFER 盲区 → 四类消解票据(任务卡形态)挂认知图。
3
+
4
+ 四类分类判据(确定性、声明式、可审计,不做语义猜测):
5
+ task — 盲区 query 精确命中某 unresolved 层节点正文首行 → 直接立工程任务卡
6
+ grilling — defer 主导(defer > blindspot,信息不完整须澄清);执行通道=宿主侧
7
+ 访谈(grill),md_cg 库内无访谈工具,本模块只建卡不越权代答(边界如实)
8
+ research — blindspot >= 3(反复未知)→ 研究议题(检索/调研补证)
9
+ prototype — 其余 → 最小探针实验验证
10
+
11
+ 幂等:tasks.upsert 同 slug 即同任务,重复生成不新建卡(updated 计数)。
12
+ 依赖挂接:票据即任务节点(认知图节点天然在图上、可检索)。「升级为正式工程
13
+ 任务时的机械 depends_on 边」由使用者在该卡更新时显式填写——本模块不伪造
14
+ 工程边,消解路径以正文声明。
15
+ 预演:apply=False 只出清单不落库(与 learn/reconstruct 的 opt-in 同构,
16
+ 默认链路零变更)。
17
+ """
18
+ from . import metacognition, tasks
19
+
20
+ TICKET_TYPES = ("research", "prototype", "grilling", "task")
21
+
22
+ _RESOLVE = {
23
+ "research": "检索/调研补证(外部研究系列吸收流程),产出知识节点后复核 gap_hint",
24
+ "prototype": "构造最小探针实验验证,结论回写认知图",
25
+ "grilling": "澄清缺位:执行通道=宿主侧访谈(grill);库内无访谈工具,不越权代答",
26
+ "task": "直接立工程任务推进(关联 unresolved 悬挂线索)",
27
+ }
28
+
29
+ _JUDGE = {
30
+ "research": "blindspot>=3(反复未知 → 研究议题)",
31
+ "prototype": "其余(偶发未知 → 最小探针)",
32
+ "grilling": "defer 主导(信息不完整 → 须澄清)",
33
+ "task": "query 精确命中 unresolved 节点正文首行",
34
+ }
35
+
36
+
37
+ def classify(item, unresolved_first_lines):
38
+ """按确定性判据给盲区分派票据类型。unresolved_first_lines={首行: node_id}。"""
39
+ q = str(item.get("query") or "").strip()
40
+ if q and q in unresolved_first_lines:
41
+ return "task"
42
+ b = int(item.get("blindspot") or 0)
43
+ d = int(item.get("defer") or 0)
44
+ if d > b:
45
+ return "grilling"
46
+ if b >= 3:
47
+ return "research"
48
+ return "prototype"
49
+
50
+
51
+ def _plan_text(item, ttype, unresolved_first_lines):
52
+ q = str(item.get("query") or "").strip()
53
+ lines = [
54
+ "来源:盲区簇 key=%s(blindspot=%s defer=%s samples=%s)" % (
55
+ metacognition._key(q), item.get("blindspot"), item.get("defer"),
56
+ item.get("samples")),
57
+ "分类判据:%s" % _JUDGE[ttype],
58
+ ]
59
+ if ttype == "task" and q in unresolved_first_lines:
60
+ lines.append("关联悬挂:unresolved 节点 %s" % unresolved_first_lines[q])
61
+ lines.append("消解路径:%s" % _RESOLVE[ttype])
62
+ lines.append("完成判据:盲区簇后续反思记录中 BLINDSPOT/DEFER 归零(gap_hint 复核)")
63
+ return "\n".join(lines)
64
+
65
+
66
+ def make_tickets(cg, *, types=None, limit=10, min_blindspot=0,
67
+ apply=False, actor=None) -> dict:
68
+ """盲区 → 消解票据。apply=False 预演(零落库);apply=True 经 tasks.upsert 落卡。"""
69
+ want = tuple(t for t in (types or TICKET_TYPES) if str(t).strip())
70
+ bad = [t for t in want if str(t).strip().lower() not in TICKET_TYPES]
71
+ if bad:
72
+ return {"ok": False,
73
+ "error": "未知票据类型:%s(可选 %s)" % (bad, list(TICKET_TYPES))}
74
+
75
+ mp = metacognition.blindspots(cg, limit=max(int(limit), 20))
76
+ items = []
77
+ for it in mp.get("items") or []:
78
+ b = int(it.get("blindspot") or 0)
79
+ d = int(it.get("defer") or 0)
80
+ if b >= int(min_blindspot) or d >= 1:
81
+ items.append(it)
82
+
83
+ unresolved_first_lines = {}
84
+ for u in mp.get("unresolved") or []:
85
+ first = str(u.get("content") or "").strip().splitlines()
86
+ if first and first[0].strip():
87
+ unresolved_first_lines.setdefault(first[0].strip(), u.get("node_id"))
88
+
89
+ out, created, updated, errors = [], 0, 0, 0
90
+ for it in items[:int(limit)]:
91
+ q = str(it.get("query") or "").strip()
92
+ if not q:
93
+ continue
94
+ ttype = classify(it, unresolved_first_lines)
95
+ if ttype not in [str(t).strip().lower() for t in want]:
96
+ out.append({"type": ttype, "query": q, "skipped": "types 过滤"})
97
+ continue
98
+ name = "票据:%s:%s" % (ttype, q[:24])
99
+ if not apply:
100
+ out.append({"type": ttype, "query": q, "name": name, "dry_run": True})
101
+ continue
102
+ r = tasks.upsert(
103
+ cg, name, plan=_plan_text(it, ttype, unresolved_first_lines),
104
+ status="active",
105
+ note="盲区簇 key=%s|blindspot=%s|defer=%s" % (
106
+ metacognition._key(q), it.get("blindspot"), it.get("defer")),
107
+ tags=["blindspot-ticket", "ticket:%s" % ttype],
108
+ actor=actor or "blindspot_tickets")
109
+ if r.get("ok"):
110
+ if r.get("created"):
111
+ created += 1
112
+ else:
113
+ updated += 1
114
+ out.append({"node_id": r.get("node_id"), "type": ttype,
115
+ "query": q, "created": bool(r.get("created"))})
116
+ else:
117
+ errors += 1
118
+ out.append({"type": ttype, "query": q, "error": r.get("error")})
119
+
120
+ return {"ok": errors == 0, "action": "tickets", "op": "insight",
121
+ "scanned": len(items), "created": created, "updated": updated,
122
+ "errors": errors, "applied": bool(apply), "tickets": out,
123
+ "unresolved_count": len(unresolved_first_lines)}
package/md_cg/branches.py CHANGED
@@ -33,16 +33,19 @@ COLD_DIR = "_branches"
33
33
  _BRANCH_RE = re.compile(r"^[a-z0-9][a-z0-9_-]{0,63}$")
34
34
 
35
35
 
36
+ # 生效条件:调用 branch_tag(branch_id) 时,返回 "branch:" 拼接 str(branch_id or "").strip().lower()(branch_id 为假值时以空串参与拼接)。
36
37
  def branch_tag(branch_id: str) -> str:
37
38
  """分支节点的标注 tag(fork 时写入;branch_summary 也带它以便反查)。"""
38
39
  return "branch:" + str(branch_id or "").strip().lower()
39
40
 
40
41
 
42
+ # 生效条件:调用 branch_node_id(orig, branch_id) 时,返回 orig 与 str(branch_id or "").strip().lower() 以 @ 拼接的字符串(branch_id 为假值时用空串)。
41
43
  def branch_node_id(orig: str, branch_id: str) -> str:
42
44
  """分支副本 id:`<主支nid>@<branch_id>`(@ 文件名合法、撞名不可能)。"""
43
45
  return "%s@%s" % (orig, str(branch_id or "").strip().lower())
44
46
 
45
47
 
48
+ # 生效条件:调用 _audit(cg, op, **meta) 时,仅当 getattr(cg, "_audit", None) 不为 None 才用 op 等调用之,且该调用抛出的任何 Exception 被静默吞掉;否则直接返回。
46
49
  def _audit(cg, op: str, **meta) -> None:
47
50
  """生命周期留痕;审计失败绝不阻断主流程(G8 同哲学)。"""
48
51
  a = getattr(cg, "_audit", None)
@@ -54,12 +57,14 @@ def _audit(cg, op: str, **meta) -> None:
54
57
  pass
55
58
 
56
59
 
60
+ # 生效条件:调用 _iter_branch_nodes(cg, branch_id) 时,返回 cg.index 的 "nodes" 字典(假值则视为空字典)中所有 v.get("branch_id") 等于 str(branch_id or "").strip().lower() 的 (k, v) 列表。
57
61
  def _iter_branch_nodes(cg, branch_id: str):
58
62
  bid = str(branch_id or "").strip().lower()
59
63
  return [(k, v) for k, v in (cg.index.get("nodes") or {}).items()
60
64
  if v.get("branch_id") == bid]
61
65
 
62
66
 
67
+ # 生效条件:调用 _fm_of(rec) 时,若 rec 为假值返回 {};否则返回 rec.get("frontmatter")(若该值真)或 rec 本身。
63
68
  def _fm_of(rec):
64
69
  """get 返回形态兼容:{'frontmatter': fm, 'content': c} 或平铺 fm。"""
65
70
  if not rec:
@@ -67,6 +72,7 @@ def _fm_of(rec):
67
72
  return rec.get("frontmatter") or rec
68
73
 
69
74
 
75
+ # 生效条件:调用 fork(cg, node_ids, branch_id=None, note=None) 时,先将 node_ids 去空转字符串,为空则返回 {"ok": False, "error": ...};否则 branch_id 为 None 时生成 "br_"+uuid4().hex[:8],再 str(branch_id).strip().lower(),若不匹配模块常量 _BRANCH_RE 则返回错误;然后对每个 nid 跳过含 "@"、cg.get(nid) 为假值或 cg.get(branch_node_id(nid, branch_id)) 已存在者,其余经 cg.add 复制并记入 forked,最终返回 ok=bool(forked) 及 branch_id/tag/forked/skipped。
70
76
  def fork(cg, node_ids, branch_id=None, note=None) -> dict:
71
77
  """把主支节点复制成分支实验副本。
72
78
 
@@ -115,11 +121,13 @@ def fork(cg, node_ids, branch_id=None, note=None) -> dict:
115
121
  "tag": branch_tag(branch_id), "forked": forked, "skipped": skipped}
116
122
 
117
123
 
124
+ # 生效条件:调用 search(cg, query, branch_id, **kw) 时,以 branch=str(branch_id or "").strip().lower() 为参数转调 cg.search,并原样返回其结果。
118
125
  def search(cg, query: str, branch_id: str, **kw):
119
126
  """分支内检索:主支 + 本分支可见、其他分支隐身(_candidates branch 过滤)。"""
120
127
  return cg.search(query, branch=str(branch_id or "").strip().lower(), **kw)
121
128
 
122
129
 
130
+ # 生效条件:当 node_id 经 str 转换并 strip 后含 '@'、cg.index['nodes'] 中存在该 id 且其 branch_id 非空、content 不是 dict 且不是 None 时返回 {'ok': True, 'node': nid, 'branch_id': ...},否则返回 {'ok': False, 'error': ...}。
123
131
  def rewrite(cg, node_id, content, tags=None, importance=None,
124
132
  actor="branch_rewrite", **extra) -> dict:
125
133
  """分支实验改写的**唯一正路**(防裸 add 丢分支归属)。
@@ -167,6 +175,7 @@ def rewrite(cg, node_id, content, tags=None, importance=None,
167
175
  return {"ok": True, "node": nid, "branch_id": e.get("branch_id")}
168
176
 
169
177
 
178
+ # 生效条件:调用 merge(cg, branch_id, reason=None) 时,先取 bid=str(branch_id or "").strip().lower() 与 _iter_branch_nodes(cg, bid),若结果为空返回 {"ok": False, "error": ...};否则遍历每个分支节点,若其 branched_from 取不到主支节点则记 skipped,否则把分支内容用 cg.add 回写主支(override=True、derived_from 追加分支 nid、relation="merged_from" 等),最终返回 ok=bool(merged) 等。
170
179
  def merge(cg, branch_id, reason=None) -> dict:
171
180
  """把分支上的改写按 branched_from 溯源回写主支。
172
181
 
@@ -210,6 +219,7 @@ def merge(cg, branch_id, reason=None) -> dict:
210
219
  "hint": "分支节点保留:可 branch_search 继续实验,或 discard 冷归档"}
211
220
 
212
221
 
222
+ # 生效条件:调用 discard(cg, branch_id, summary) 时,若 cg.principal 存在则先 require_admin("branch_discard");取 bid=str(branch_id or "").strip().lower(),若 summary 缺少模块常量 BRANCH_MARKS 中任一标记则返回错误,若 _iter_branch_nodes(cg, bid) 为空也返回错误;否则写 "branch_summary_"+bid 知识节点,把分支节点文件移入 COLD_DIR/bid 并从 cg.index["nodes"] 弹出,最后返回 ok=True 等。
213
223
  def discard(cg, branch_id, summary: str) -> dict:
214
224
  """放弃分支:校验教训必填 → 写 branch_summary 教训节点 → 分支冷归档。
215
225
 
@@ -258,6 +268,7 @@ def discard(cg, branch_id, summary: str) -> dict:
258
268
  "summary_node": sid, "cold_dir": "%s/%s" % (COLD_DIR, bid)}
259
269
 
260
270
 
271
+ # 生效条件:调用 list_branches(cg) 时,遍历 cg.index 的 "nodes"(假值则空字典),仅对 e.get("branch_id") 为真值的节点按 bid 分组,收集节点 id 和 branched_from,返回 ok=True 与按 branch_id 排序的分组列表。
261
272
  def list_branches(cg) -> dict:
262
273
  """按 branch_id 聚合现存分支(节点清单 + 溯源主支)。"""
263
274
  groups = {}
@@ -272,4 +283,4 @@ def list_branches(cg) -> dict:
272
283
  if bf and bf not in g["branched_from"]:
273
284
  g["branched_from"].append(bf)
274
285
  return {"ok": True,
275
- "branches": sorted(groups.values(), key=lambda g: g["branch_id"])}
286
+ "branches": sorted(groups.values(), key=lambda g: g["branch_id"])}
@@ -0,0 +1,73 @@
1
+ # -*- coding: utf-8 -*-
2
+ """倒排发布表构建/查看(S7 候选层)。
3
+
4
+ 用法:
5
+ python -X utf8 -m md_cg.build_postings --root <库根> # 全量重建(幂等)
6
+ python -X utf8 -m md_cg.build_postings --root <库根> --stats # 只看现状
7
+ python -X utf8 -m md_cg.build_postings --root <库根> --limit 500 # 抽样构建(冒烟用,永不算新鲜)
8
+ python -X utf8 -m md_cg.build_postings --root <库根> --if-stale # 过期才重建(守候环用)
9
+
10
+ 说明:发布表是**派生索引**(<root>/_postings.json + _postings_meta.json),
11
+ 不修改任何节点内容;节点新增/修改后需重建(v1 不做增量维护)。
12
+ """
13
+ import argparse
14
+ import json
15
+ import sys
16
+
17
+ from . import postings
18
+ from .mdcg import MdCG
19
+
20
+
21
+ # 生效条件:st 为 dict 时返回其摘要——保留 schema/built_at/nodes/terms/postings/partial/rebuilt/rebuilt_from,
22
+ # 并把 snapshot 折成 dirs/root_files 计数(快照本体含上千个目录键,直接打印会淹没日志);st 非 dict 时原样返回。
23
+ def _brief(st):
24
+ if not isinstance(st, dict):
25
+ return st
26
+ keys = ("schema", "built_at", "nodes", "terms", "postings", "partial",
27
+ "rebuilt", "rebuilt_from")
28
+ out = {k: st.get(k) for k in keys if k in st}
29
+ snap = st.get("snapshot")
30
+ if isinstance(snap, dict):
31
+ out["dirs"] = len(snap.get("dir_mtimes") or {})
32
+ out["root_files"] = len(snap.get("file_mtimes") or {})
33
+ return out
34
+
35
+
36
+ # 生效条件:无条件解析参数(root 必填;stats/limit/if-stale 可选;if-stale 与 limit 互斥则 argparse 报错并退出 2);
37
+ # stats 为真时打开库打印 stats 摘要(含 stale_reason)并返回 0;if-stale 为真且快照新鲜时打印 {"rebuilt": false} 返回 0、
38
+ # 过期时重建并打印摘要(附 rebuilt_from=过期原因);否则一律重建并打印摘要,返回 0。
39
+ # 三条路径打印的都是 _brief 摘要——不打印快照指纹本体。
40
+ def main(argv=None):
41
+ ap = argparse.ArgumentParser()
42
+ ap.add_argument("--root", required=True)
43
+ ap.add_argument("--stats", action="store_true")
44
+ ap.add_argument("--limit", type=int, default=None)
45
+ ap.add_argument("--if-stale", action="store_true", dest="if_stale")
46
+ a = ap.parse_args(argv)
47
+ if a.if_stale and a.limit:
48
+ ap.error("--if-stale 与 --limit 互斥(抽样表不可当作新鲜快照)")
49
+ cg = MdCG(a.root)
50
+ nodes = cg.index.get("nodes") or {}
51
+ if a.stats:
52
+ st = postings.stats(a.root, nodes)
53
+ print(json.dumps({"terms": st.get("terms"), "postings": st.get("postings"),
54
+ "stale_reason": st.get("stale_reason"),
55
+ "meta": _brief(st.get("meta"))}, ensure_ascii=False))
56
+ return 0
57
+ if a.if_stale:
58
+ sr = postings.stale_reason(a.root, nodes)
59
+ if not sr:
60
+ print(json.dumps({"rebuilt": False, "reason": "fresh"}, ensure_ascii=False))
61
+ return 0
62
+ st = postings.build(cg)
63
+ print(json.dumps(_brief(dict(st, rebuilt_from=sr)), ensure_ascii=False))
64
+ return 0
65
+ st = postings.build(cg, limit=a.limit)
66
+ print(json.dumps(_brief(st), ensure_ascii=False))
67
+ return 0
68
+
69
+
70
+ if __name__ == "__main__":
71
+ if hasattr(sys.stdout, "reconfigure"):
72
+ sys.stdout.reconfigure(encoding="utf-8")
73
+ sys.exit(main())