@furongjun1999/dsh-memory 0.4.8 → 0.4.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (232) hide show
  1. package/README.md +54 -26
  2. package/codebuddy/CODEBUDDY.md +196 -195
  3. package/codebuddy/README.md +13 -1
  4. package/codebuddy/mcp.json +9 -0
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -0
  6. package/docs/README.md +1 -1
  7. package/docs/discipline/harnesses.yaml +18 -7
  8. package/docs/discipline/templates/full.md.tmpl +4 -3
  9. package/docs/experiments/linkref_backfill/candidates_20260917.json +726 -0
  10. package/docs/experiments/linkref_backfill/candidates_internal_20260917.json +602 -0
  11. package/docs/experiments/linkref_backfill/candidates_internal_v2.json +603 -0
  12. package/docs/experiments/linkref_backfill/candidates_secret_20260917.json +884 -0
  13. package/docs/experiments/linkref_backfill/candidates_secret_v2.json +789 -0
  14. package/docs/hive//345/244/232/347/253/257harness/351/200/232/344/277/241/345/245/221/347/272/246_v0.1.md +42 -0
  15. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -0
  16. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -0
  17. package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S1b/344/270/216/345/217/254/345/233/236/346/235/203/350/241/241_v0.1.md +57 -0
  18. package/docs/hive//347/234/237/345/256/236/345/272/223/347/253/257/345/210/260/347/253/257/345/256/236/346/265/213_S7/345/200/222/346/216/222/345/200/231/351/200/211/345/261/202_v0.1.md +126 -0
  19. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -0
  20. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -0
  21. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.5.md +631 -625
  22. package/docs/mdcg//344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/345/245/221/347/272/246_v0.1.md +82 -0
  23. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -0
  24. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +47 -45
  25. package/docs/mdcg//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/{206_v0.4.md → 206_v0.5.md} +92 -4
  26. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -0
  27. package/docs/mdcg//347/216/257/344/272/214_/347/231/275/347/256/261/345/241/253/345/205/205/346/265/201/346/260/264/347/272/277_v0.1.md +31 -0
  28. package/docs/mdcg//350/267/250/347/253/257/351/252/214/350/257/201/344/270/216/345/220/214/346/255/245/345/215/217/350/256/256_v0.1.md +93 -0
  29. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +2 -2
  30. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +3 -3
  31. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +2 -2
  32. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +434 -433
  33. package/docs//347/201/265/346/236/242/350/207/252/346/210/221/346/224/271/350/277/233/345/267/245/344/275/234/350/256/241/345/210/222_/345/244/226/351/203/250/347/240/224/347/251/266/347/263/273/345/210/227/345/220/270/346/224/266_v1_20260919.md +286 -0
  34. package/dsh/README.md +33 -0
  35. package/dsh/cordis.yml.example +13 -7
  36. package/dsh/hive-mcp-probe.mjs +94 -0
  37. package/dsh/hive-mcp.example.yml +62 -0
  38. package/dsh/update-lingshu.bat +11 -0
  39. package/dsh/update-lingshu.ps1 +337 -0
  40. package/lib/hooks.d.ts +3 -0
  41. package/lib/hooks.js +17 -23
  42. package/lib/index.d.ts +4 -2
  43. package/lib/index.js +29 -6
  44. package/lib/lib/datapath.d.ts +76 -1
  45. package/lib/lib/datapath.js +199 -13
  46. package/lib/lib/mdcg_client.d.ts +40 -3
  47. package/lib/lib/mdcg_client.js +46 -22
  48. package/lib/lib/mutual.js +4 -4
  49. package/lib/lib/token_store.js +4 -5
  50. package/md_cg/audit.py +17 -2
  51. package/md_cg/autonomy.py +86 -15
  52. package/md_cg/backfill.py +36 -1
  53. package/md_cg/backfill_bigdomain.py +34 -0
  54. package/md_cg/bench6_arms.py +28 -1
  55. package/md_cg/bench6_common.py +10 -1
  56. package/md_cg/bench6_competitors.py +6 -1
  57. package/md_cg/bench_axis_domain.py +9 -1
  58. package/md_cg/bench_blind_comp.py +7 -1
  59. package/md_cg/bench_en_atoms_public.py +9 -0
  60. package/md_cg/bench_governance.py +348 -0
  61. package/md_cg/bench_lme_zh.py +16 -1
  62. package/md_cg/bench_locomo.py +2 -1
  63. package/md_cg/bench_locomo_zh.py +16 -1
  64. package/md_cg/bench_locomo_zh_public.py +4 -1
  65. package/md_cg/bench_longmem.py +2 -1
  66. package/md_cg/bench_membench.py +27 -1
  67. package/md_cg/bench_p0.py +4 -1
  68. package/md_cg/bench_progressive.py +13 -1
  69. package/md_cg/bench_role_views.py +238 -0
  70. package/md_cg/bench_task_ab.py +8 -1
  71. package/md_cg/bench_task_ab_llm.py +13 -1
  72. package/md_cg/bench_unified_en.py +6 -1
  73. package/md_cg/bench_zh_mad.py +20 -1
  74. package/md_cg/blindspot_tickets.py +123 -0
  75. package/md_cg/branches.py +12 -1
  76. package/md_cg/build_postings.py +73 -0
  77. package/md_cg/ccgc.py +67 -2
  78. package/md_cg/census.py +5 -1
  79. package/md_cg/chain.py +24 -3
  80. package/md_cg/codeindex.py +134 -17
  81. package/md_cg/coldverify.py +265 -0
  82. package/md_cg/comment_gate.py +338 -0
  83. package/md_cg/cond_compose.py +190 -0
  84. package/md_cg/cond_facts.py +155 -0
  85. package/md_cg/cond_template.json +107 -0
  86. package/md_cg/condition_anchor.py +143 -0
  87. package/md_cg/conformance.py +69 -4
  88. package/md_cg/consistency.py +24 -1
  89. package/md_cg/consolidate.py +53 -2
  90. package/md_cg/corpus.py +4 -0
  91. package/md_cg/crosscheck.py +42 -2
  92. package/md_cg/crypto.py +35 -1
  93. package/md_cg/d_meta.py +310 -0
  94. package/md_cg/datapath.py +201 -26
  95. package/md_cg/docindex.py +122 -1
  96. package/md_cg/eval_common.py +29 -1
  97. package/md_cg/evidence.py +27 -1
  98. package/md_cg/evolution.py +21 -1
  99. package/md_cg/export.py +11 -1
  100. package/md_cg/forgetting.py +23 -1
  101. package/md_cg/fsutil.py +18 -1
  102. package/md_cg/hotcache.py +214 -0
  103. package/md_cg/hyperedge.py +251 -0
  104. package/md_cg/identity.py +18 -1
  105. package/md_cg/insight.py +17 -1
  106. package/md_cg/lexicon/build_cedict_en_zh.py +9 -0
  107. package/md_cg/lexicon/build_standard_en.py +171 -168
  108. package/md_cg/lexicon/expand_en_zh.py +6 -0
  109. package/md_cg/lifecycle.py +12 -1
  110. package/md_cg/linkref.py +281 -0
  111. package/md_cg/links.py +29 -1
  112. package/md_cg/mcp_server.py +362 -43
  113. package/md_cg/md_whitebox.py +53 -1
  114. package/md_cg/mdcg.py +1003 -27
  115. package/md_cg/mdcos.py +558 -36
  116. package/md_cg/metacognition.py +37 -2
  117. package/md_cg/migrate.py +4 -0
  118. package/md_cg/migrate_aeis.py +221 -213
  119. package/md_cg/migrate_roleplay.py +8 -0
  120. package/md_cg/migrate_wisdom_graph.py +14 -1
  121. package/md_cg/mreview/__main__.py +3 -0
  122. package/md_cg/mreview/bundle.py +8 -0
  123. package/md_cg/mreview/candidates.py +9 -0
  124. package/md_cg/mreview/govern.py +21 -1
  125. package/md_cg/mreview/locate.py +34 -0
  126. package/md_cg/mreview/pipeline.py +29 -1
  127. package/md_cg/mreview/ruleset.py +16 -1
  128. package/md_cg/nodefile.py +233 -3
  129. package/md_cg/pooling.py +23 -1
  130. package/md_cg/postings.py +298 -0
  131. package/md_cg/predict.py +89 -9
  132. package/md_cg/progressive.py +3 -0
  133. package/md_cg/protect.py +14 -1
  134. package/md_cg/protocol.py +372 -0
  135. package/md_cg/provenance.py +262 -0
  136. package/md_cg/reach.py +453 -0
  137. package/md_cg/refindex.py +47 -2
  138. package/md_cg/refine.py +20 -1
  139. package/md_cg/roleviews.py +89 -0
  140. package/md_cg/routing.py +76 -0
  141. package/md_cg/scrub.py +63 -2
  142. package/md_cg/security.py +26 -1
  143. package/md_cg/self_state.py +64 -1
  144. package/md_cg/selfreport.py +151 -0
  145. package/md_cg/semantic/canonical.py +5 -0
  146. package/md_cg/semantic/en_normalizer.py +364 -355
  147. package/md_cg/semantic/zh_en_atoms.py +139 -136
  148. package/md_cg/signer.py +41 -1
  149. package/md_cg/sources.py +583 -547
  150. package/md_cg/statushdr.py +179 -0
  151. package/md_cg/stg.py +59 -18
  152. package/md_cg/subgraph.py +23 -0
  153. package/md_cg/sustain.py +56 -1
  154. package/md_cg/tasks.py +26 -2
  155. package/md_cg/test_autonomy.py +26 -0
  156. package/md_cg/test_bench_governance.py +102 -0
  157. package/md_cg/test_blindspot_tickets.py +166 -0
  158. package/md_cg/test_ccgc.py +10 -0
  159. package/md_cg/test_codeindex.py +338 -0
  160. package/md_cg/test_comment_gate.py +187 -0
  161. package/md_cg/test_cond_compose_anchors.py +76 -0
  162. package/md_cg/test_condition_anchor.py +82 -0
  163. package/md_cg/test_d_meta.py +412 -0
  164. package/md_cg/test_datapath_root.py +188 -0
  165. package/md_cg/test_gain_gate.py +47 -1
  166. package/md_cg/test_hot_cold.py +187 -0
  167. package/md_cg/test_hyperedge.py +245 -0
  168. package/md_cg/test_linkref.py +306 -0
  169. package/md_cg/test_md_access_parity.py +15 -3
  170. package/md_cg/test_mr_m1.py +108 -18
  171. package/md_cg/test_mr_m3.py +8 -1
  172. package/md_cg/test_p26_refindex.py +49 -20
  173. package/md_cg/test_p27_docindex.py +236 -2
  174. package/md_cg/test_p2_mcp.py +1 -1
  175. package/md_cg/test_p31_insight.py +24 -0
  176. package/md_cg/test_p44_md_whitebox.py +14 -1
  177. package/md_cg/test_protocol.py +243 -0
  178. package/md_cg/test_reach.py +378 -0
  179. package/md_cg/test_reach_keys.py +201 -0
  180. package/md_cg/test_reach_meta_exits.py +145 -0
  181. package/md_cg/test_read_clip.py +8 -4
  182. package/md_cg/test_retr_s1.py +340 -0
  183. package/md_cg/test_retr_s1b.py +209 -0
  184. package/md_cg/test_retr_s3.py +194 -0
  185. package/md_cg/test_retr_s4.py +163 -0
  186. package/md_cg/test_retr_s5.py +200 -0
  187. package/md_cg/test_retr_s6.py +157 -0
  188. package/md_cg/test_retr_s7.py +385 -0
  189. package/md_cg/test_retr_s8_time.py +316 -0
  190. package/md_cg/test_retr_s9_edges.py +286 -0
  191. package/md_cg/test_retr_s9_entity_ctx.py +175 -0
  192. package/md_cg/test_review_conformance.py +59 -2
  193. package/md_cg/test_role_views.py +354 -0
  194. package/md_cg/test_subproc_encoding.py +188 -0
  195. package/md_cg/test_trust.py +361 -0
  196. package/md_cg/test_units_poll.py +71 -0
  197. package/md_cg/test_v14_fixes.py +397 -0
  198. package/md_cg/test_validity_filter.py +280 -0
  199. package/md_cg/test_wisdom_md_store.py +7 -3
  200. package/md_cg/test_writepipe.py +5 -1
  201. package/md_cg/theory.py +16 -1
  202. package/md_cg/tokens.py +40 -8
  203. package/md_cg/tool_face.py +13 -2
  204. package/md_cg/trust.py +943 -0
  205. package/md_cg/twophase.py +12 -1
  206. package/md_cg/units.py +132 -10
  207. package/md_cg/vision_evidence.py +24 -1
  208. package/md_cg/weights.py +24 -1
  209. package/md_cg/whitebox.py +32 -1
  210. package/md_cg/whitebox_kb/data/verify_cache.json +21210 -365
  211. package/md_cg/whitebox_kb/data/verify_savings.jsonl +5078 -0
  212. package/md_cg/whitebox_kb/wisdom/audit_log/chain_heat.json +10 -10
  213. package/md_cg/whitebox_kb/wisdom/code_compose.py +113 -6
  214. package/md_cg/whitebox_kb/wisdom/code_solidified.json +1 -1
  215. package/md_cg/whitebox_kb/wisdom/verifier.py +340 -55
  216. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db +0 -0
  217. package/md_cg/writelimit.py +18 -4
  218. package/md_cg/writepipe.py +178 -7
  219. package/package.json +2 -2
  220. package/skills/skills/designer-perspective/scripts/__pycache__/designer.cpython-310.pyc +0 -0
  221. package/skills/skills/designer-perspective/scripts/designer.py +17 -1
  222. package/skills/skills/designer-perspective/tests/selftest.py +3 -1
  223. package/src/hooks.ts +17 -21
  224. package/src/index.ts +33 -6
  225. package/src/lib/datapath.ts +211 -13
  226. package/src/lib/mdcg_client.ts +64 -25
  227. package/src/lib/mutual.ts +411 -411
  228. package/src/lib/token_store.ts +4 -5
  229. package/zcode/AGENTS.md +196 -195
  230. package/zcode/README.md +4 -0
  231. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-shm +0 -0
  232. package/md_cg/whitebox_kb/wisdom/wisdom-book-cloud.db-wal +0 -0
package/md_cg/sources.py CHANGED
@@ -1,547 +1,583 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 设备驱动(记忆 OS #3):把外部会话流接进认知图
3
-
4
- 设计:
5
- Source(事件源)—— 把某种外部存储解析成统一事件流:
6
- {"t": 毫秒时间戳, "seq": 序号, "role": ..., "text": ..., "session": ..., "cwd": ...}
7
- Ingestor(摄取器)—— 增量 watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
8
-
9
- 内置源:
10
- · JsonlSource —— 通用 JSONL(每行一个事件对象)
11
- · DSHSessionSource —— DeepSeek Harness 会话(~/.dsh/sessions/**/session.jsonl[.zstd])
12
- zstd 为**可选**依赖:缺失时优雅降级(跳过 .zstd 文件并告警)
13
-
14
- 默认敏感度:会话内容是私有记忆 → sensitivity="private"(避免落入公开根)。
15
-
16
- 零第三方依赖(zstd 为可选增强)。
17
- """
18
- from __future__ import annotations
19
-
20
- import glob
21
- import json
22
- import os
23
- import time
24
-
25
- from .security import DEFAULT_SENSITIVITY
26
-
27
- # 会话事件的默认落层与敏感度
28
- SESSION_LAYER = "contextual"
29
- SESSION_SENSITIVITY = "private"
30
-
31
-
32
- # --------------------------------------------------------------------------
33
- # 事件源
34
- # --------------------------------------------------------------------------
35
-
36
- class Source:
37
- """事件源基类。"""
38
-
39
- name = "source"
40
-
41
- def events(self):
42
- raise NotImplementedError
43
-
44
- def key(self):
45
- """源的稳定标识(用于 watermark)。"""
46
- return self.name
47
-
48
-
49
- class JsonlSource(Source):
50
- """通用 JSONL 会话源。
51
-
52
- 每行是事件对象;字段映射可配置:
53
- t_key —— 时间戳字段(默认 "time",也接受 ISO 字符串)
54
- role_key —— 角色字段(默认 "role")
55
- text_key —— 文本字段(默认 "text")
56
- """
57
-
58
- def __init__(self, path: str, name: str = None, t_key="time", role_key="role",
59
- text_key="text", default_role="user"):
60
- self.path = path
61
- self.name = name or ("jsonl:" + os.path.basename(path))
62
- self.t_key, self.role_key, self.text_key = t_key, role_key, text_key
63
- self.default_role = default_role
64
-
65
- def key(self):
66
- return self.name
67
-
68
- def _ts(self, o):
69
- v = o.get(self.t_key)
70
- if isinstance(v, (int, float)):
71
- return float(v) * (1000.0 if v < 1e12 else 1.0)
72
- if isinstance(v, str):
73
- try:
74
- return time.mktime(time.strptime(v[:19], "%Y-%m-%dT%H:%M:%S")) * 1000
75
- except ValueError:
76
- return 0.0
77
- return 0.0
78
-
79
- def events(self):
80
- if not os.path.exists(self.path):
81
- return
82
- with open(self.path, encoding="utf-8", errors="replace") as f:
83
- for i, line in enumerate(f):
84
- line = line.strip()
85
- if not line:
86
- continue
87
- try:
88
- o = json.loads(line)
89
- except ValueError:
90
- continue
91
- text = o.get(self.text_key)
92
- if not text:
93
- continue
94
- yield {"t": self._ts(o), "seq": o.get("seq", i),
95
- "role": o.get(self.role_key) or self.default_role,
96
- "text": str(text), "session": o.get("session"),
97
- "cwd": o.get("cwd")}
98
-
99
-
100
- def _zstd_reader(path):
101
- """返回可读的文本迭代器;zstd 不可用返回 None。"""
102
- import io
103
- try:
104
- import zstandard as zstd
105
- except ImportError:
106
- return None
107
- with open(path, "rb") as f:
108
- raw = zstd.ZstdDecompressor().stream_reader(f).read()
109
- return io.StringIO(raw.decode("utf-8", errors="replace"))
110
-
111
-
112
- class DSHSessionSource(Source):
113
- """DeepSeek Harness 会话源。
114
-
115
- 事件映射:
116
- user/message → role=user
117
- assistant/message → role=assistant(只取 content[].text,reasoning 默认丢弃)
118
- tool/call → role=command("name(args)" 形式)
119
- tool/result → role=tool-output
120
- """
121
-
122
- def __init__(self, path: str, include_reasoning: bool = False):
123
- self.path = path
124
- self.include_reasoning = include_reasoning
125
- self.name = "dsh:" + os.path.basename(os.path.dirname(path))
126
-
127
- def key(self):
128
- return self.name
129
-
130
- @staticmethod
131
- def discover(root: str = None, limit: int = None):
132
- root = root or os.path.join(os.path.expanduser("~"), ".dsh", "sessions")
133
- files = glob.glob(os.path.join(root, "**", "session.jsonl"), recursive=True)
134
- files += glob.glob(os.path.join(root, "**", "session.jsonl.zstd"), recursive=True)
135
- files.sort(key=lambda p: -os.path.getsize(p))
136
- return files[:limit] if limit else files
137
-
138
- def _lines(self):
139
- if self.path.endswith(".zstd"):
140
- fh = _zstd_reader(self.path)
141
- if fh is None:
142
- raise RuntimeError("zstd 不可用(pip install zstandard),跳过该会话")
143
- try:
144
- yield from fh
145
- finally:
146
- fh.close()
147
- else:
148
- with open(self.path, encoding="utf-8", errors="replace") as f:
149
- yield from f
150
-
151
- @staticmethod
152
- def _text_of(content):
153
- """content 可能是 [{type,text}] 或字符串。"""
154
- if isinstance(content, str):
155
- return content
156
- if isinstance(content, list):
157
- parts = []
158
- for c in content:
159
- if isinstance(c, dict):
160
- if c.get("type") in ("text", "input-text") and c.get("text"):
161
- parts.append(str(c["text"]))
162
- elif isinstance(c, str):
163
- parts.append(c)
164
- return "\n".join(parts)
165
- return ""
166
-
167
- def events(self):
168
- sess = None
169
- cwd = None
170
- for line in self._lines():
171
- line = line.strip()
172
- if not line:
173
- continue
174
- try:
175
- o = json.loads(line)
176
- except ValueError:
177
- continue
178
- t = o.get("type")
179
- data = o.get("data") or {}
180
- if t == "session":
181
- sess, cwd = o.get("id"), o.get("cwd")
182
- continue
183
- ev = {"t": float(o.get("time") or 0), "seq": o.get("seq"),
184
- "session": sess, "cwd": cwd}
185
- if t == "user/message":
186
- ev["role"], ev["text"] = "user", self._text_of(data.get("content"))
187
- elif t == "assistant/message":
188
- msg = data.get("message") or {}
189
- ev["role"] = "assistant"
190
- ev["text"] = self._text_of(msg.get("content"))
191
- if self.include_reasoning:
192
- for c in (msg.get("content") or []):
193
- if isinstance(c, dict) and c.get("type") == "reasoning" \
194
- and c.get("text"):
195
- ev["text"] = (ev["text"] or "") + "\n[reasoning] " + str(c["text"])
196
- elif t == "tool/call":
197
- ev["role"] = "command"
198
- ev["text"] = f"{data.get('name')}({data.get('arguments') or ''})"
199
- elif t == "tool/result":
200
- msg = data.get("message") or {}
201
- inner = msg.get("content") or []
202
- txt = ""
203
- if inner and isinstance(inner[0], dict):
204
- txt = self._text_of(inner[0].get("content"))
205
- ev["role"], ev["text"] = "tool-output", txt
206
- else:
207
- continue
208
- if ev.get("text"):
209
- yield ev
210
-
211
-
212
- # --------------------------------------------------------------------------
213
- # 摄取器
214
- # --------------------------------------------------------------------------
215
-
216
- class Ingestor:
217
- """增量摄取:watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
218
-
219
- watermark 文件:<root>/_sources.json
220
- {source_key: {"t": 最后时间戳(ms), "seq": 最后序号, "count": 已摄取条数}}
221
- """
222
-
223
- def __init__(self, cg, layer: str = SESSION_LAYER,
224
- sensitivity: str = SESSION_SENSITIVITY):
225
- self.cg = cg
226
- self.layer = layer
227
- self.sensitivity = sensitivity
228
- self.path = os.path.join(cg.root, "_sources.json")
229
-
230
- # ---- watermark ----
231
-
232
- def _load(self):
233
- if os.path.exists(self.path):
234
- try:
235
- with open(self.path, encoding="utf-8") as f:
236
- d = json.load(f)
237
- if isinstance(d, dict):
238
- return d
239
- except (ValueError, OSError):
240
- pass
241
- return {"schema": 1, "sources": {}}
242
-
243
- def _save(self, d):
244
- tmp = self.path + ".tmp"
245
- with open(tmp, "w", encoding="utf-8") as f:
246
- json.dump(d, f, ensure_ascii=False, indent=1)
247
- os.replace(tmp, self.path)
248
-
249
- def watermark(self, key: str):
250
- return self._load()["sources"].get(key, {})
251
-
252
- def watermarks(self):
253
- return dict(self._load()["sources"])
254
-
255
- # ---- 摄取 ----
256
-
257
- def ingest(self, source, mine_fix_pairs: bool = True, max_events: int = None,
258
- dry_run: bool = False):
259
- """摄取一个源的新事件。返回统计。
260
-
261
- 去重键:(session, seq) —— 同一事件不重复入库。
262
- 增量:只处理 (t, seq) 大于 watermark 的事件。
263
- """
264
- key = source.key()
265
- wm = self.watermark(key)
266
- last_t, last_seq = float(wm.get("t") or 0), wm.get("seq")
267
- new_events, seen = [], set()
268
- for ev in source.events():
269
- if ev.get("t", 0) < last_t:
270
- continue
271
- if ev.get("t", 0) == last_t and last_seq is not None \
272
- and isinstance(ev.get("seq"), int) and ev["seq"] <= last_seq:
273
- continue
274
- dedup = (ev.get("session"), ev.get("seq"))
275
- if dedup in seen:
276
- continue
277
- seen.add(dedup)
278
- new_events.append(ev)
279
- if max_events and len(new_events) >= max_events:
280
- break
281
-
282
- written, ids, denied = 0, [], 0
283
- if not dry_run:
284
- for ev in new_events:
285
- nid = "src_%s_%s" % (_sig(key)[:6], _sig(
286
- f"{ev.get('session')}:{ev.get('seq')}")[:10])
287
- if nid in self.cg.index["nodes"]:
288
- continue # 幂等
289
- body = ("# 功能名:会话事件\n"
290
- f"# 生效条件:检索「{str(ev.get('text'))[:20]}」\n"
291
- "# 子功能:记录会话事件\n"
292
- f"# 执行:{str(ev.get('text'))[:80]}\n"
293
- "# 验证方式:data(会话原始记录)\n"
294
- "# 不适用条件:其它会话\n\n"
295
- f"{ev.get('text')}\n")
296
- try:
297
- self.cg.add(nid, body, layer=self.layer, role=ev.get("role"),
298
- tags=["session", key], sensitivity=self.sensitivity,
299
- verification_basis="data",
300
- condition_space={"observation_position": key,
301
- "observation_tool": "会话流",
302
- "time_window": [ev.get("t") or 0,
303
- ev.get("t") or 0]})
304
- except Exception as exc: # noqa: BLE001 —— 权限/层错误不中断整批
305
- denied += 1
306
- self.last_error = f"{type(exc).__name__}: {exc}"
307
- continue
308
- ids.append(nid)
309
- written += 1
310
-
311
- result = {"source": key, "new_events": len(new_events), "written": written,
312
- "denied": denied, "ids": ids, "dry_run": dry_run,
313
- "sensitivity": self.sensitivity}
314
- if denied and getattr(self, "last_error", None):
315
- result["last_error"] = self.last_error
316
- result["hint"] = ("会话内容默认 sensitivity=private;"
317
- "调用方需 MDCG_CLEARANCE=private 才能写入")
318
- # 自动 fix-pair 挖掘(对标 deja-vu:错误→修复)
319
- if mine_fix_pairs and new_events and not dry_run:
320
- result["fix_pairs"] = self.cg.mine_fix_pairs(
321
- [{"role": e.get("role"), "text": e.get("text")} for e in new_events])
322
-
323
- if new_events and not dry_run:
324
- d = self._load()
325
- last = new_events[-1]
326
- d["sources"][key] = {
327
- "t": last.get("t") or last_t,
328
- "seq": last.get("seq"),
329
- "count": (wm.get("count") or 0) + written,
330
- "updated_at": time.time(),
331
- "path": getattr(source, "path", None),
332
- }
333
- self._save(d)
334
- return result
335
-
336
-
337
- def _sig(text: str, n: int = 12) -> str:
338
- import hashlib
339
- return hashlib.sha1((text or "").encode("utf-8")).hexdigest()[:n]
340
-
341
-
342
- # --------------------------------------------------------------------------
343
- # 摄取分派(P0 · ingest op):按扩展名选摄取方式,单一入口吃多种文件
344
- # --------------------------------------------------------------------------
345
- #
346
- # 三条摄取链:
347
- # session —— 会话流(.jsonl):走 Ingestor(watermark + 去重 + fix-pair)
348
- # doc —— 文档(.md/.txt/...):走 docindex.extract + refindex.add_items
349
- # code —— 代码(.py/.ts/...):走 codeindex.extract + refindex.add_items
350
- #
351
- # 设计要点:
352
- # - 注册表 INGEST_REGISTRY 是唯一真源:新增后缀只改这里。
353
- # - dir 动作分链处理:目录里 doc / code / jsonl 混放时各链互不干扰。
354
- # - 幂等:沿用 refindex.Ledger(size+mtime 水位)与 Ingestor watermark。
355
- # - 预演:dry_run=True 只统计、不写入(对应计划「可预演」要求)。
356
-
357
- INGEST_ACTIONS = ("file", "dir", "jsonl", "stat")
358
-
359
- INGEST_REGISTRY = {
360
- # 会话流
361
- ".jsonl": "session", ".ndjson": "session",
362
- # 文档
363
- ".md": "doc", ".markdown": "doc", ".txt": "doc", ".rst": "doc",
364
- ".html": "doc", ".htm": "doc",
365
- # 代码
366
- ".py": "code", ".js": "code", ".mjs": "code", ".ts": "code",
367
- ".tsx": "code", ".jsx": "code", ".go": "code", ".rs": "code",
368
- ".java": "code", ".kt": "code", ".swift": "code", ".rb": "code",
369
- ".php": "code", ".cs": "code", ".cpp": "code", ".cc": "code",
370
- ".c": "code", ".h": "code", ".hpp": "code",
371
- }
372
-
373
-
374
- def dispatch_of(path: str):
375
- """按扩展名返回摄取方式(session / doc / code);未登记返回 None。"""
376
- return INGEST_REGISTRY.get(os.path.splitext(path or "")[1].lower())
377
-
378
-
379
- class FileDispatcher:
380
- """单一入口吃多种文件:按扩展名分派到会话流 / 文档 / 代码三条摄取链。"""
381
-
382
- def __init__(self, cg, sensitivity=None):
383
- self.cg = cg
384
- # 会话链默认 sensitivity=private;调用方可显式覆盖(测试/受限环境)
385
- self.ingestor = Ingestor(cg, sensitivity=sensitivity or SESSION_SENSITIVITY)
386
-
387
- # ---- stat:看水位与支持面 ----
388
-
389
- def _ledger_stat(self):
390
- try:
391
- from . import refindex
392
- return refindex.Ledger(self.cg.root).stat()
393
- except Exception: # noqa: BLE001
394
- return {}
395
-
396
- def stat(self):
397
- kinds = {}
398
- for ext, kind in INGEST_REGISTRY.items():
399
- kinds.setdefault(kind, []).append(ext)
400
- return {"ok": True, "kind": "stat", "actions": list(INGEST_ACTIONS),
401
- "extensions": {k: sorted(v) for k, v in sorted(kinds.items())},
402
- "watermarks": self.ingestor.watermarks(),
403
- "ledger": self._ledger_stat(),
404
- "note": ("注册表是唯一真源:新增后缀只改 INGEST_REGISTRY。"
405
- "watermarks=会话流水位;ledger=文档/代码文件水位。")}
406
-
407
- # ---- 单文件 ----
408
-
409
- def ingest_file(self, path, layer=None, sensitivity=None, dry_run=False):
410
- kind = dispatch_of(path)
411
- if kind is None:
412
- ext = os.path.splitext(path or "")[1] or "(无后缀)"
413
- return {"ok": False, "path": path, "error": f"不支持的后缀:{ext}",
414
- "supported": sorted(INGEST_REGISTRY)}
415
- if not os.path.isfile(path):
416
- return {"ok": False, "path": path, "error": "文件不存在"}
417
- if kind == "session":
418
- return self.ingest_jsonl(path, dry_run=dry_run)
419
- return self._ingest_doc_or_code(path, kind, layer=layer,
420
- sensitivity=sensitivity, dry_run=dry_run)
421
-
422
- def _ingest_doc_or_code(self, path, kind, layer=None, sensitivity=None,
423
- dry_run=False):
424
- from . import codeindex, docindex, refindex
425
- mod = codeindex if kind == "code" else docindex
426
- rel = os.path.basename(path)
427
- with open(path, encoding="utf-8", errors="replace") as f:
428
- src = f.read()
429
- try:
430
- items = mod.extract(src, path=rel,
431
- suffix=os.path.splitext(path)[1].lower())
432
- except ValueError as exc:
433
- return {"ok": False, "path": path, "error": f"抽取失败:{exc}"}
434
- items = items or []
435
- if dry_run:
436
- return {"ok": True, "dry_run": True, "kind": kind, "path": path,
437
- "items": len(items),
438
- "ids": [mod.node_id(i) for i in items[:20]],
439
- "note": "预演:只抽取计数,未写盘"}
440
- ref_kind = "code_ref" if kind == "code" else "doc_ref"
441
- ids, sens = refindex.add_items(
442
- self.cg, items, kind=ref_kind,
443
- root=os.path.dirname(path) or ".", layer=layer,
444
- sensitivity=sensitivity)
445
- return {"ok": True, "kind": kind, "path": path, "items": len(items),
446
- "indexed": len(ids), "ids": ids[:20], "sensitivity": sens}
447
-
448
- # ---- 目录 ----
449
-
450
- def _dry_dir(self, root):
451
- counts = {}
452
- for _dp, _dn, fns in os.walk(root):
453
- for name in fns:
454
- k = dispatch_of(name) or "unsupported"
455
- counts[k] = counts.get(k, 0) + 1
456
- return {"ok": True, "dry_run": True, "kind": "dir", "root": root,
457
- "counts": counts,
458
- "note": "预演:仅统计各链文件数,未做任何写入"}
459
-
460
- def ingest_dir(self, root, layer=None, sensitivity=None, patterns=None,
461
- max_files=500, max_items=2000, incremental=False,
462
- dry_run=False):
463
- from . import refindex
464
- if not os.path.isdir(root):
465
- return {"ok": False, "error": f"目录不存在:{root}"}
466
- if dry_run:
467
- return self._dry_dir(root)
468
- ledger = refindex.Ledger(self.cg.root)
469
- out = {"ok": True, "kind": "dir", "root": root, "chains": {}}
470
- for ref_kind, key in (("doc_ref", "doc"), ("code_ref", "code")):
471
- items, errors, stats = refindex.index_dir(
472
- root, kind=ref_kind, patterns=patterns, max_files=max_files,
473
- max_items=max_items, incremental=incremental, ledger=ledger)
474
- ids, sens = refindex.add_items(self.cg, items, kind=ref_kind,
475
- root=root, layer=layer,
476
- sensitivity=sensitivity)
477
- out["chains"][key] = {
478
- "indexed": len(ids), "errors": len(errors),
479
- "files": stats.get("files"), "truncated": stats.get("truncated"),
480
- "skipped_unchanged": stats.get("skipped_unchanged", 0),
481
- "skipped_suffixes": stats.get("skipped_suffixes", []),
482
- "sensitivity": sens}
483
- # 会话流(.jsonl)逐个增量摄取
484
- jsons = sorted(glob.glob(os.path.join(root, "**", "*.jsonl"),
485
- recursive=True))[:max_files]
486
- ses = []
487
- for p in jsons:
488
- r = self.ingest_jsonl(p)
489
- ses.append({"path": p, "written": r.get("written", 0),
490
- "new_events": r.get("new_events", 0)})
491
- out["chains"]["session"] = {"files": len(jsons), "results": ses}
492
- return out
493
-
494
- # ---- 会话流 ----
495
-
496
- @staticmethod
497
- def _auto_source(path):
498
- """通用 JSONL vs DSH 会话:按内容探测,避免调用方选错源类型。"""
499
- try:
500
- with open(path, encoding="utf-8", errors="replace") as f:
501
- head = f.read(50000)
502
- except OSError:
503
- return JsonlSource(path)
504
- for marker in ('"user/message"', '"assistant/message"', '"tool/call"'):
505
- if marker in head:
506
- return DSHSessionSource(path)
507
- return JsonlSource(path)
508
-
509
- def ingest_jsonl(self, path, dry_run=False, max_events=None):
510
- if not os.path.isfile(path):
511
- return {"ok": False, "error": f"文件不存在:{path}"}
512
- src = self._auto_source(path)
513
- res = self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events)
514
- res.update({"ok": True, "kind": "session", "path": path,
515
- "source_class": type(src).__name__})
516
- return res
517
-
518
-
519
- def run(cg, action: str = "stat", **kw):
520
- """ingest op 唯一入口。"""
521
- act = (action or "stat").strip().lower()
522
- d = FileDispatcher(cg, sensitivity=kw.get("sensitivity"))
523
- if act == "stat":
524
- return d.stat()
525
- if act == "file":
526
- p = kw.get("path")
527
- if not p:
528
- return {"ok": False, "error": "file 动作需要 path"}
529
- return d.ingest_file(p, layer=kw.get("layer"),
530
- sensitivity=kw.get("sensitivity"),
531
- dry_run=bool(kw.get("dry_run")))
532
- if act == "dir":
533
- p = kw.get("path") or cg.root
534
- return d.ingest_dir(p, layer=kw.get("layer"),
535
- sensitivity=kw.get("sensitivity"),
536
- patterns=kw.get("patterns"),
537
- max_files=int(kw.get("max_files") or 500),
538
- max_items=int(kw.get("max_items") or 2000),
539
- incremental=bool(kw.get("incremental")),
540
- dry_run=bool(kw.get("dry_run")))
541
- if act == "jsonl":
542
- p = kw.get("path")
543
- if not p:
544
- return {"ok": False, "error": "jsonl 动作需要 path"}
545
- return d.ingest_jsonl(p, dry_run=bool(kw.get("dry_run")),
546
- max_events=kw.get("max_events"))
547
- raise ValueError(f"未知 ingest action:{action!r}(允许 {INGEST_ACTIONS})")
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 设备驱动(记忆 OS #3):把外部会话流接进认知图
3
+
4
+ 设计:
5
+ Source(事件源)—— 把某种外部存储解析成统一事件流:
6
+ {"t": 毫秒时间戳, "seq": 序号, "role": ..., "text": ..., "session": ..., "cwd": ...}
7
+ Ingestor(摄取器)—— 增量 watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
8
+
9
+ 内置源:
10
+ · JsonlSource —— 通用 JSONL(每行一个事件对象)
11
+ · DSHSessionSource —— DeepSeek Harness 会话(~/.dsh/sessions/**/session.jsonl[.zstd])
12
+ zstd 为**可选**依赖:缺失时优雅降级(跳过 .zstd 文件并告警)
13
+
14
+ 默认敏感度:会话内容是私有记忆 → sensitivity="private"(避免落入公开根)。
15
+
16
+ 零第三方依赖(zstd 为可选增强)。
17
+ """
18
+ from __future__ import annotations
19
+
20
+ import glob
21
+ import json
22
+ import os
23
+ import time
24
+
25
+ from .security import DEFAULT_SENSITIVITY
26
+
27
+ # 会话事件的默认落层与敏感度
28
+ SESSION_LAYER = "contextual"
29
+ SESSION_SENSITIVITY = "private"
30
+
31
+
32
+ # --------------------------------------------------------------------------
33
+ # 事件源
34
+ # --------------------------------------------------------------------------
35
+
36
+ # 生效条件:无必填构造形参,类常量 name="source" 即实例默认;仅当子类覆写 events() 时才产出事件,基类 events() 恒抛 NotImplementedError。
37
+ class Source:
38
+ """事件源基类。"""
39
+
40
+ name = "source"
41
+
42
+ # 生效条件:任何调用都直接 raise NotImplementedError(基类占位,无其它分支)。
43
+ def events(self):
44
+ raise NotImplementedError
45
+
46
+ # 生效条件:无前置;返回类常量 self.name(基类为 "source"),作为 watermark 的稳定标识,不含路径与运行期状态;
47
+ def key(self):
48
+ """源的稳定标识(用于 watermark)。"""
49
+ return self.name
50
+
51
+
52
+ # 生效条件:required 形参 path 总被存入 self.path;可选形参 name 为假值(None/空串)时 self.name 回落到 "jsonl:"+os.path.basename(path),为真值时用 name 本身,t_key/role_key/text_key/default_role 原样存入属性。
53
+ class JsonlSource(Source):
54
+ """通用 JSONL 会话源。
55
+
56
+ 每行是事件对象;字段映射可配置:
57
+ t_key —— 时间戳字段(默认 "time",也接受 ISO 字符串)
58
+ role_key —— 角色字段(默认 "role")
59
+ text_key —— 文本字段(默认 "text")
60
+ """
61
+
62
+ # 生效条件:传入 path;name 为假值(None/空串)时回落为 "jsonl:"+os.path.basename(path),t_key/role_key/text_key/default_role 原样存为属性(默认值 "time"/"role"/"text"/"user")。
63
+ def __init__(self, path: str, name: str = None, t_key="time", role_key="role",
64
+ text_key="text", default_role="user"):
65
+ self.path = path
66
+ self.name = name or ("jsonl:" + os.path.basename(path))
67
+ self.t_key, self.role_key, self.text_key = t_key, role_key, text_key
68
+ self.default_role = default_role
69
+
70
+ # 生效条件:无前置;返回 self.name(构造时已回落为 "jsonl:"+basename(path)),只随构造参数变化、不随文件内容变化;
71
+ def key(self):
72
+ return self.name
73
+
74
+ # 生效条件:o.get(self.t_key) 为 int/float 时返回 float(v) 乘 1000(v<1e12)或乘 1(否则);为字符串且 v[:19] 按 "%Y-%m-%dT%H:%M:%S" 解析成功时返回 mktime*1000,抛 ValueError 时返回 0.0;键缺失或其它类型返回 0.0。
75
+ def _ts(self, o):
76
+ v = o.get(self.t_key)
77
+ if isinstance(v, (int, float)):
78
+ return float(v) * (1000.0 if v < 1e12 else 1.0)
79
+ if isinstance(v, str):
80
+ try:
81
+ return time.mktime(time.strptime(v[:19], "%Y-%m-%dT%H:%M:%S")) * 1000
82
+ except ValueError:
83
+ return 0.0
84
+ return 0.0
85
+
86
+ # 生效条件:os.path.exists(self.path) 为真时逐行产出,空行、json.loads 抛 ValueError、o.get(self.text_key) 为假的行被跳过,产出项为 t=self._ts(o)、seq=o.get("seq", i)、role=o.get(self.role_key) or self.default_role、text=str(text)、session/cwd 取 o 同名键;path 不存在时直接 return 不产出。
87
+ def events(self):
88
+ if not os.path.exists(self.path):
89
+ return
90
+ with open(self.path, encoding="utf-8", errors="replace") as f:
91
+ for i, line in enumerate(f):
92
+ line = line.strip()
93
+ if not line:
94
+ continue
95
+ try:
96
+ o = json.loads(line)
97
+ except ValueError:
98
+ continue
99
+ text = o.get(self.text_key)
100
+ if not text:
101
+ continue
102
+ yield {"t": self._ts(o), "seq": o.get("seq", i),
103
+ "role": o.get(self.role_key) or self.default_role,
104
+ "text": str(text), "session": o.get("session"),
105
+ "cwd": o.get("cwd")}
106
+
107
+
108
+ # 生效条件:path 指向的内容可读且 zstandard 可导入时返回 StringIO(raw.decode("utf-8", errors="replace"));ImportError 时返回 None。
109
+ def _zstd_reader(path):
110
+ """返回可读的文本迭代器;zstd 不可用返回 None。"""
111
+ import io
112
+ try:
113
+ import zstandard as zstd
114
+ except ImportError:
115
+ return None
116
+ with open(path, "rb") as f:
117
+ raw = zstd.ZstdDecompressor().stream_reader(f).read()
118
+ return io.StringIO(raw.decode("utf-8", errors="replace"))
119
+
120
+
121
+ # 生效条件:required 形参 path 总被存入 self.path,可选形参 include_reasoning 原样存入,self.name 恒为 "dsh:"+os.path.basename(os.path.dirname(path))(与 include_reasoning 取值无关)。
122
+ class DSHSessionSource(Source):
123
+ """DeepSeek Harness 会话源。
124
+
125
+ 事件映射:
126
+ user/message → role=user
127
+ assistant/message → role=assistant(只取 content[].text,reasoning 默认丢弃)
128
+ tool/call → role=command("name(args)" 形式)
129
+ tool/result → role=tool-output
130
+ """
131
+
132
+ # 生效条件:传入 path 即成立,include_reasoning 原样存为属性(默认 False),name 固定为 "dsh:"+os.path.basename(os.path.dirname(path))。
133
+ def __init__(self, path: str, include_reasoning: bool = False):
134
+ self.path = path
135
+ self.include_reasoning = include_reasoning
136
+ self.name = "dsh:" + os.path.basename(os.path.dirname(path))
137
+
138
+ # 生效条件:无前置;返回 self.name(构造时固定为 "dsh:"+basename(dirname(path))),与 include_reasoning 取值无关;
139
+ def key(self):
140
+ return self.name
141
+
142
+ @staticmethod
143
+ # 生效条件:可选形参 root 为假值(None/空串)时改用默认目录 os.path.join(expanduser("~"),".dsh","sessions"),否则用传入 root;对 root 下递归 glob 到的 session.jsonl 与 session.jsonl.zstd 按 -os.path.getsize 降序排序(两处 glob 均无命中时为空列表),可选形参 limit 为假值(None/0)时返回全部 files,否则返回 files[:limit]。
144
+ def discover(root: str = None, limit: int = None):
145
+ root = root or os.path.join(os.path.expanduser("~"), ".dsh", "sessions")
146
+ files = glob.glob(os.path.join(root, "**", "session.jsonl"), recursive=True)
147
+ files += glob.glob(os.path.join(root, "**", "session.jsonl.zstd"), recursive=True)
148
+ files.sort(key=lambda p: (-os.path.getsize(p), p))
149
+ return files[:limit] if limit else files
150
+
151
+ # 生效条件:self.path 以 ".zstd" 结尾时经 _zstd_reader 逐行产出(其返回 None 时 raise RuntimeError),否则以 utf-8/errors=replace 打开 self.path 逐行产出。
152
+ def _lines(self):
153
+ if self.path.endswith(".zstd"):
154
+ fh = _zstd_reader(self.path)
155
+ if fh is None:
156
+ raise RuntimeError("zstd 不可用(pip install zstandard),跳过该会话")
157
+ try:
158
+ yield from fh
159
+ finally:
160
+ fh.close()
161
+ else:
162
+ with open(self.path, encoding="utf-8", errors="replace") as f:
163
+ yield from f
164
+
165
+ @staticmethod
166
+ # 生效条件:required 形参 content 为 str 时原样返回 content;为 list 时收集其中 str 元素及 type 属于 ("text","input-text") 且 text 为真值的 dict 元素(取 str(c["text"])),以 "\n" 连接返回(无可收集元素时为空串 "");既非 str 也非 list 时返回 ""。
167
+ def _text_of(content):
168
+ """content 可能是 [{type,text}] 或字符串。"""
169
+ if isinstance(content, str):
170
+ return content
171
+ if isinstance(content, list):
172
+ parts = []
173
+ for c in content:
174
+ if isinstance(c, dict):
175
+ if c.get("type") in ("text", "input-text") and c.get("text"):
176
+ parts.append(str(c["text"]))
177
+ elif isinstance(c, str):
178
+ parts.append(c)
179
+ return "\n".join(parts)
180
+ return ""
181
+
182
+ # 生效条件:逐行解析后按 o.get("type") 分派——"session" 只更新 sess/cwd 不产出;"user/message" 产出 role=user 与 _text_of(data.get("content"));"assistant/message" 产出 role=assistant 与 _text_of(msg.get("content")),include_reasoning 为真时再把 content 中 type=="reasoning" 且有 text 的项追加 "\n[reasoning] "+str(c["text"]);"tool/call" 产出 role=command 与 f"{name}({arguments or ''})";"tool/result" 产出 role=tool-output,仅当 inner[0] 为 dict 时 text=_text_of(inner[0].get("content"));其它 type 或 ev 中 text 为空的事件不产出。
183
+ def events(self):
184
+ sess = None
185
+ cwd = None
186
+ for line in self._lines():
187
+ line = line.strip()
188
+ if not line:
189
+ continue
190
+ try:
191
+ o = json.loads(line)
192
+ except ValueError:
193
+ continue
194
+ t = o.get("type")
195
+ data = o.get("data") or {}
196
+ if t == "session":
197
+ sess, cwd = o.get("id"), o.get("cwd")
198
+ continue
199
+ ev = {"t": float(o.get("time") or 0), "seq": o.get("seq"),
200
+ "session": sess, "cwd": cwd}
201
+ if t == "user/message":
202
+ ev["role"], ev["text"] = "user", self._text_of(data.get("content"))
203
+ elif t == "assistant/message":
204
+ msg = data.get("message") or {}
205
+ ev["role"] = "assistant"
206
+ ev["text"] = self._text_of(msg.get("content"))
207
+ if self.include_reasoning:
208
+ for c in (msg.get("content") or []):
209
+ if isinstance(c, dict) and c.get("type") == "reasoning" \
210
+ and c.get("text"):
211
+ ev["text"] = (ev["text"] or "") + "\n[reasoning] " + str(c["text"])
212
+ elif t == "tool/call":
213
+ ev["role"] = "command"
214
+ ev["text"] = f"{data.get('name')}({data.get('arguments') or ''})"
215
+ elif t == "tool/result":
216
+ msg = data.get("message") or {}
217
+ inner = msg.get("content") or []
218
+ txt = ""
219
+ if inner and isinstance(inner[0], dict):
220
+ txt = self._text_of(inner[0].get("content"))
221
+ ev["role"], ev["text"] = "tool-output", txt
222
+ else:
223
+ continue
224
+ if ev.get("text"):
225
+ yield ev
226
+
227
+
228
+ # --------------------------------------------------------------------------
229
+ # 摄取器
230
+ # --------------------------------------------------------------------------
231
+
232
+ # 生效条件:required 形参 cg 总被存入并以其 cg.root 拼出 self.path=os.path.join(cg.root,"_sources.json");layer/sensitivity 原样存入,未传时取模块级常量 SESSION_LAYER、SESSION_SENSITIVITY 作为默认值。
233
+ class Ingestor:
234
+ """增量摄取:watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
235
+
236
+ watermark 文件:<root>/_sources.json
237
+ {source_key: {"t": 最后时间戳(ms), "seq": 最后序号, "count": 已摄取条数}}
238
+ """
239
+
240
+ # 生效条件:传入带 root 的 cg 即成立,layer/sensitivity 默认 SESSION_LAYER/SESSION_SENSITIVITY 并原样存为属性,路径为 os.path.join(cg.root, "_sources.json")。
241
+ def __init__(self, cg, layer: str = SESSION_LAYER,
242
+ sensitivity: str = SESSION_SENSITIVITY):
243
+ self.cg = cg
244
+ self.layer = layer
245
+ self.sensitivity = sensitivity
246
+ self.path = os.path.join(cg.root, "_sources.json")
247
+
248
+ # ---- watermark ----
249
+
250
+ # 生效条件:self.path 存在、json.load 成功且结果为 dict 时返回该 dict;path 不存在、抛 ValueError/OSError 或结果非 dict 时返回 {"schema": 1, "sources": {}}。
251
+ def _load(self):
252
+ if os.path.exists(self.path):
253
+ try:
254
+ with open(self.path, encoding="utf-8") as f:
255
+ d = json.load(f)
256
+ if isinstance(d, dict):
257
+ return d
258
+ except (ValueError, OSError):
259
+ pass
260
+ return {"schema": 1, "sources": {}}
261
+
262
+ # 生效条件:传入 d 时以 ensure_ascii=False/indent=1 写入 self.path+".tmp",再 os.replace 覆盖 self.path。
263
+ def _save(self, d):
264
+ tmp = self.path + ".tmp"
265
+ with open(tmp, "w", encoding="utf-8") as f:
266
+ json.dump(d, f, ensure_ascii=False, indent=1)
267
+ os.replace(tmp, self.path)
268
+
269
+ # 生效条件:key 命中 self._load()["sources"] 时返回其值,缺 key 时返回 {}(_load 结果缺 "sources" 键则抛 KeyError)。
270
+ def watermark(self, key: str):
271
+ return self._load()["sources"].get(key, {})
272
+
273
+ # 生效条件:调用即返回 dict(self._load()["sources"]) 的浅拷贝(_load 结果缺 "sources" 键则抛 KeyError)。
274
+ def watermarks(self):
275
+ return dict(self._load()["sources"])
276
+
277
+ # ---- 摄取 ----
278
+
279
+ # 生效条件:对必需形参 source,先按 watermark(source.key()) 跳过 t<last_t(wm 的 "t" 为假值时视作 0)及 t==last_t 且 last_seq 非 None 且 ev["seq"] 为 int 且 <= last_seq 的事件、并按 (session, seq) 去重,max_events 为真值(非 0/None)时取满即停;dry_run 为假时逐条 cg.add(nid 已在 cg.index["nodes"] 中则跳过,cg.add 抛异常则 denied+=1、记 last_error 并继续),denied 与 last_error 同时成立时补 last_error/hint,mine_fix_pairs 为真且 new_events 非空且非 dry_run 时结果附 fix_pairs,new_events 非空且非 dry_run 时以末事件写回水位(count 累加 written),最后返回 result。
280
+ def ingest(self, source, mine_fix_pairs: bool = True, max_events: int = None,
281
+ dry_run: bool = False):
282
+ """摄取一个源的新事件。返回统计。
283
+
284
+ 去重键:(session, seq) —— 同一事件不重复入库。
285
+ 增量:只处理 (t, seq) 大于 watermark 的事件。
286
+ """
287
+ key = source.key()
288
+ wm = self.watermark(key)
289
+ last_t, last_seq = float(wm.get("t") or 0), wm.get("seq")
290
+ new_events, seen = [], set()
291
+ for ev in source.events():
292
+ if ev.get("t", 0) < last_t:
293
+ continue
294
+ if ev.get("t", 0) == last_t and last_seq is not None \
295
+ and isinstance(ev.get("seq"), int) and ev["seq"] <= last_seq:
296
+ continue
297
+ dedup = (ev.get("session"), ev.get("seq"))
298
+ if dedup in seen:
299
+ continue
300
+ seen.add(dedup)
301
+ new_events.append(ev)
302
+ if max_events and len(new_events) >= max_events:
303
+ break
304
+
305
+ written, ids, denied = 0, [], 0
306
+ if not dry_run:
307
+ for ev in new_events:
308
+ nid = "src_%s_%s" % (_sig(key)[:6], _sig(
309
+ f"{ev.get('session')}:{ev.get('seq')}")[:10])
310
+ if nid in self.cg.index["nodes"]:
311
+ continue # 幂等
312
+ body = ("# 功能名:会话事件\n"
313
+ f"# 生效条件:检索「{str(ev.get('text'))[:20]}」\n"
314
+ "# 子功能:记录会话事件\n"
315
+ f"# 执行:{str(ev.get('text'))[:80]}\n"
316
+ "# 验证方式:data(会话原始记录)\n"
317
+ "# 不适用条件:其它会话\n\n"
318
+ f"{ev.get('text')}\n")
319
+ try:
320
+ self.cg.add(nid, body, layer=self.layer, role=ev.get("role"),
321
+ tags=["session", key], sensitivity=self.sensitivity,
322
+ verification_basis="data",
323
+ condition_space={"observation_position": key,
324
+ "observation_tool": "会话流",
325
+ "time_window": [ev.get("t") or 0,
326
+ ev.get("t") or 0]})
327
+ except Exception as exc: # noqa: BLE001 —— 权限/层错误不中断整批
328
+ denied += 1
329
+ self.last_error = f"{type(exc).__name__}: {exc}"
330
+ continue
331
+ ids.append(nid)
332
+ written += 1
333
+
334
+ result = {"source": key, "new_events": len(new_events), "written": written,
335
+ "denied": denied, "ids": ids, "dry_run": dry_run,
336
+ "sensitivity": self.sensitivity}
337
+ if denied and getattr(self, "last_error", None):
338
+ result["last_error"] = self.last_error
339
+ result["hint"] = ("会话内容默认 sensitivity=private;"
340
+ "调用方需 MDCG_CLEARANCE=private 才能写入")
341
+ # 自动 fix-pair 挖掘(对标 deja-vu:错误→修复)
342
+ if mine_fix_pairs and new_events and not dry_run:
343
+ result["fix_pairs"] = self.cg.mine_fix_pairs(
344
+ [{"role": e.get("role"), "text": e.get("text")} for e in new_events])
345
+
346
+ if new_events and not dry_run:
347
+ d = self._load()
348
+ last = new_events[-1]
349
+ d["sources"][key] = {
350
+ "t": last.get("t") or last_t,
351
+ "seq": last.get("seq"),
352
+ "count": (wm.get("count") or 0) + written,
353
+ "updated_at": time.time(),
354
+ "path": getattr(source, "path", None),
355
+ }
356
+ self._save(d)
357
+ return result
358
+
359
+
360
+ # 生效条件:text 为 None 或假值时按 "" 参与 sha1;n 默认 12,返回 hexdigest 前 n 位(n=0 得空串)。
361
+ def _sig(text: str, n: int = 12) -> str:
362
+ import hashlib
363
+ return hashlib.sha1((text or "").encode("utf-8")).hexdigest()[:n]
364
+
365
+
366
+ # --------------------------------------------------------------------------
367
+ # 摄取分派(P0 · ingest op):按扩展名选摄取方式,单一入口吃多种文件
368
+ # --------------------------------------------------------------------------
369
+ #
370
+ # 三条摄取链:
371
+ # session —— 会话流(.jsonl):走 Ingestor(watermark + 去重 + fix-pair)
372
+ # doc —— 文档(.md/.txt/...):走 docindex.extract + refindex.add_items
373
+ # code —— 代码(.py/.ts/...):走 codeindex.extract + refindex.add_items
374
+ #
375
+ # 设计要点:
376
+ # - 注册表 INGEST_REGISTRY 是唯一真源:新增后缀只改这里。
377
+ # - dir 动作分链处理:目录里 doc / code / jsonl 混放时各链互不干扰。
378
+ # - 幂等:沿用 refindex.Ledger(size+mtime 水位)与 Ingestor watermark。
379
+ # - 预演:dry_run=True 只统计、不写入(对应计划「可预演」要求)。
380
+
381
+ INGEST_ACTIONS = ("file", "dir", "jsonl", "stat")
382
+
383
+ INGEST_REGISTRY = {
384
+ # 会话流
385
+ ".jsonl": "session", ".ndjson": "session",
386
+ # 文档
387
+ ".md": "doc", ".markdown": "doc", ".txt": "doc", ".rst": "doc",
388
+ ".html": "doc", ".htm": "doc",
389
+ # 代码
390
+ ".py": "code", ".js": "code", ".mjs": "code", ".ts": "code",
391
+ ".tsx": "code", ".jsx": "code", ".go": "code", ".rs": "code",
392
+ ".java": "code", ".kt": "code", ".swift": "code", ".rb": "code",
393
+ ".php": "code", ".cs": "code", ".cpp": "code", ".cc": "code",
394
+ ".c": "code", ".h": "code", ".hpp": "code",
395
+ }
396
+
397
+
398
+ # 生效条件:path 为 None 或空串时 splitext 得 "" 且未登记 → 返回 None;扩展名(小写)存在于 INGEST_REGISTRY 时返回其 kind。
399
+ def dispatch_of(path: str):
400
+ """按扩展名返回摄取方式(session / doc / code);未登记返回 None。"""
401
+ return INGEST_REGISTRY.get(os.path.splitext(path or "")[1].lower())
402
+
403
+
404
+ # 生效条件:required 形参 cg 总被存入;可选形参 sensitivity 为假值(None/空串)时内部 Ingestor 的 sensitivity 回落模块级常量 SESSION_SENSITIVITY,为真值时用传入的 sensitivity。
405
+ class FileDispatcher:
406
+ """单一入口吃多种文件:按扩展名分派到会话流 / 文档 / 代码三条摄取链。"""
407
+
408
+ # 生效条件:传入 cg 即成立,sensitivity 为假值(None/空串)时所用 Ingestor 回落 SESSION_SENSITIVITY,否则用传入值。
409
+ def __init__(self, cg, sensitivity=None):
410
+ self.cg = cg
411
+ # 会话链默认 sensitivity=private;调用方可显式覆盖(测试/受限环境)
412
+ self.ingestor = Ingestor(cg, sensitivity=sensitivity or SESSION_SENSITIVITY)
413
+
414
+ # ---- stat:看水位与支持面 ----
415
+
416
+ # 生效条件:from . import refindex 与 refindex.Ledger(self.cg.root).stat() 均不抛异常时返回该 stat 结果,抛任何异常时返回 {}。
417
+ def _ledger_stat(self):
418
+ try:
419
+ from . import refindex
420
+ return refindex.Ledger(self.cg.root).stat()
421
+ except Exception: # noqa: BLE001
422
+ return {}
423
+
424
+ # 生效条件:调用即返回由 INGEST_REGISTRY 按 kind 分组并排序后的 extensions、list(INGEST_ACTIONS)、self.ingestor.watermarks()、self._ledger_stat() 与固定 note。
425
+ def stat(self):
426
+ kinds = {}
427
+ for ext, kind in INGEST_REGISTRY.items():
428
+ kinds.setdefault(kind, []).append(ext)
429
+ return {"ok": True, "kind": "stat", "actions": list(INGEST_ACTIONS),
430
+ "extensions": {k: sorted(v) for k, v in sorted(kinds.items())},
431
+ "watermarks": self.ingestor.watermarks(),
432
+ "ledger": self._ledger_stat(),
433
+ "note": ("注册表是唯一真源:新增后缀只改 INGEST_REGISTRY。"
434
+ "watermarks=会话流水位;ledger=文档/代码文件水位。")}
435
+
436
+ # ---- 单文件 ----
437
+
438
+ # 生效条件:dispatch_of(path) 为 None 时返回 ok=False 的「不支持的后缀」结果;path 不是文件时返回 ok=False 的「文件不存在」结果;kind=="session" 时转 ingest_jsonl(path, dry_run=dry_run)(layer/sensitivity 不参与);其它 kind 转 _ingest_doc_or_code(path, kind, layer=layer, sensitivity=sensitivity, dry_run=dry_run)。
439
+ def ingest_file(self, path, layer=None, sensitivity=None, dry_run=False):
440
+ kind = dispatch_of(path)
441
+ if kind is None:
442
+ ext = os.path.splitext(path or "")[1] or "(无后缀)"
443
+ return {"ok": False, "path": path, "error": f"不支持的后缀:{ext}",
444
+ "supported": sorted(INGEST_REGISTRY)}
445
+ if not os.path.isfile(path):
446
+ return {"ok": False, "path": path, "error": "文件不存在"}
447
+ if kind == "session":
448
+ return self.ingest_jsonl(path, dry_run=dry_run)
449
+ return self._ingest_doc_or_code(path, kind, layer=layer,
450
+ sensitivity=sensitivity, dry_run=dry_run)
451
+
452
+ # 生效条件:kind=="code" 用 codeindex 否则用 docindex;mod.extract 抛 ValueError 时返回 ok=False 的「抽取失败」;dry_run 为真时只返回 items 计数与前 20 个 node_id 不写盘;否则经 refindex.add_items(kind 为 code_ref/doc_ref)写入并返回 indexed、ids[:20] 与 sensitivity。
453
+ def _ingest_doc_or_code(self, path, kind, layer=None, sensitivity=None,
454
+ dry_run=False):
455
+ from . import codeindex, docindex, refindex
456
+ mod = codeindex if kind == "code" else docindex
457
+ rel = os.path.basename(path)
458
+ with open(path, encoding="utf-8", errors="replace") as f:
459
+ src = f.read()
460
+ try:
461
+ items = mod.extract(src, path=rel,
462
+ suffix=os.path.splitext(path)[1].lower())
463
+ except ValueError as exc:
464
+ return {"ok": False, "path": path, "error": f"抽取失败:{exc}"}
465
+ items = items or []
466
+ if dry_run:
467
+ return {"ok": True, "dry_run": True, "kind": kind, "path": path,
468
+ "items": len(items),
469
+ "ids": [mod.node_id(i) for i in items[:20]],
470
+ "note": "预演:只抽取计数,未写盘"}
471
+ ref_kind = "code_ref" if kind == "code" else "doc_ref"
472
+ ids, sens = refindex.add_items(
473
+ self.cg, items, kind=ref_kind,
474
+ root=os.path.dirname(path) or ".", layer=layer,
475
+ sensitivity=sensitivity)
476
+ return {"ok": True, "kind": kind, "path": path, "items": len(items),
477
+ "indexed": len(ids), "ids": ids[:20], "sensitivity": sens}
478
+
479
+ # ---- 目录 ----
480
+
481
+ # 生效条件:调用即 os.walk(root) 统计每个文件名经 dispatch_of 得到的 kind(无匹配记 "unsupported")并返回 dry_run 预演计数结果。
482
+ def _dry_dir(self, root):
483
+ counts = {}
484
+ for _dp, _dn, fns in os.walk(root):
485
+ for name in fns:
486
+ k = dispatch_of(name) or "unsupported"
487
+ counts[k] = counts.get(k, 0) + 1
488
+ return {"ok": True, "dry_run": True, "kind": "dir", "root": root,
489
+ "counts": counts,
490
+ "note": "预演:仅统计各链文件数,未做任何写入"}
491
+
492
+ # 生效条件:root 非目录时返回 ok=False 的「目录不存在」;dry_run 为真时返回 _dry_dir(root);否则对 doc_ref/code_ref 两链各以 patterns/max_files/max_items/incremental/ledger 调 refindex.index_dir 与 add_items,并把 root 下 **/*.jsonl 前 max_files 个逐个 ingest_jsonl 后返回 out。
493
+ def ingest_dir(self, root, layer=None, sensitivity=None, patterns=None,
494
+ max_files=500, max_items=2000, incremental=False,
495
+ dry_run=False):
496
+ from . import refindex
497
+ if not os.path.isdir(root):
498
+ return {"ok": False, "error": f"目录不存在:{root}"}
499
+ if dry_run:
500
+ return self._dry_dir(root)
501
+ ledger = refindex.Ledger(self.cg.root)
502
+ out = {"ok": True, "kind": "dir", "root": root, "chains": {}}
503
+ for ref_kind, key in (("doc_ref", "doc"), ("code_ref", "code")):
504
+ items, errors, stats = refindex.index_dir(
505
+ root, kind=ref_kind, patterns=patterns, max_files=max_files,
506
+ max_items=max_items, incremental=incremental, ledger=ledger)
507
+ ids, sens = refindex.add_items(self.cg, items, kind=ref_kind,
508
+ root=root, layer=layer,
509
+ sensitivity=sensitivity)
510
+ out["chains"][key] = {
511
+ "indexed": len(ids), "errors": len(errors),
512
+ "files": stats.get("files"), "truncated": stats.get("truncated"),
513
+ "skipped_unchanged": stats.get("skipped_unchanged", 0),
514
+ "skipped_suffixes": stats.get("skipped_suffixes", []),
515
+ "sensitivity": sens}
516
+ # 会话流(.jsonl)逐个增量摄取
517
+ jsons = sorted(glob.glob(os.path.join(root, "**", "*.jsonl"),
518
+ recursive=True))[:max_files]
519
+ ses = []
520
+ for p in jsons:
521
+ r = self.ingest_jsonl(p)
522
+ ses.append({"path": p, "written": r.get("written", 0),
523
+ "new_events": r.get("new_events", 0)})
524
+ out["chains"]["session"] = {"files": len(jsons), "results": ses}
525
+ return out
526
+
527
+ # ---- 会话流 ----
528
+
529
+ @staticmethod
530
+ # 生效条件:required 形参 path 能被 open(...,encoding="utf-8",errors="replace") 打开时读前 50000 字符,head 含 '"user/message"'、'"assistant/message"'、'"tool/call"' 任一标记则返回 DSHSessionSource(path),否则返回 JsonlSource(path);打开抛 OSError 时直接返回 JsonlSource(path)。
531
+ def _auto_source(path):
532
+ """通用 JSONL vs DSH 会话:按内容探测,避免调用方选错源类型。"""
533
+ try:
534
+ with open(path, encoding="utf-8", errors="replace") as f:
535
+ head = f.read(50000)
536
+ except OSError:
537
+ return JsonlSource(path)
538
+ for marker in ('"user/message"', '"assistant/message"', '"tool/call"'):
539
+ if marker in head:
540
+ return DSHSessionSource(path)
541
+ return JsonlSource(path)
542
+
543
+ # 生效条件:path 不是文件时返回 ok=False 的「文件不存在」;否则经 _auto_source(path) 选源后调 self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events),并补上 ok=True/kind/path/source_class 后返回。
544
+ def ingest_jsonl(self, path, dry_run=False, max_events=None):
545
+ if not os.path.isfile(path):
546
+ return {"ok": False, "error": f"文件不存在:{path}"}
547
+ src = self._auto_source(path)
548
+ res = self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events)
549
+ res.update({"ok": True, "kind": "session", "path": path,
550
+ "source_class": type(src).__name__})
551
+ return res
552
+
553
+
554
+ # 生效条件:cg 必填;action 为 None/空串时 (action or "stat") 归为 stat;file/jsonl 缺 path 返回 ok=False;dir 的 max_files/max_items 走 int(x or 500)/int(x or 2000),传 0 也变 500/2000;未知 action 抛 ValueError。
555
+ def run(cg, action: str = "stat", **kw):
556
+ """ingest op 唯一入口。"""
557
+ act = (action or "stat").strip().lower()
558
+ d = FileDispatcher(cg, sensitivity=kw.get("sensitivity"))
559
+ if act == "stat":
560
+ return d.stat()
561
+ if act == "file":
562
+ p = kw.get("path")
563
+ if not p:
564
+ return {"ok": False, "error": "file 动作需要 path"}
565
+ return d.ingest_file(p, layer=kw.get("layer"),
566
+ sensitivity=kw.get("sensitivity"),
567
+ dry_run=bool(kw.get("dry_run")))
568
+ if act == "dir":
569
+ p = kw.get("path") or cg.root
570
+ return d.ingest_dir(p, layer=kw.get("layer"),
571
+ sensitivity=kw.get("sensitivity"),
572
+ patterns=kw.get("patterns"),
573
+ max_files=int(kw.get("max_files") or 500),
574
+ max_items=int(kw.get("max_items") or 2000),
575
+ incremental=bool(kw.get("incremental")),
576
+ dry_run=bool(kw.get("dry_run")))
577
+ if act == "jsonl":
578
+ p = kw.get("path")
579
+ if not p:
580
+ return {"ok": False, "error": "jsonl 动作需要 path"}
581
+ return d.ingest_jsonl(p, dry_run=bool(kw.get("dry_run")),
582
+ max_events=kw.get("max_events"))
583
+ raise ValueError(f"未知 ingest action:{action!r}(允许 {INGEST_ACTIONS})")