compound-memory 0.2.0__tar.gz → 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (33) hide show
  1. {compound_memory-0.2.0/src/compound_memory.egg-info → compound_memory-0.3.0}/PKG-INFO +1 -1
  2. {compound_memory-0.2.0 → compound_memory-0.3.0}/pyproject.toml +1 -1
  3. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/cli.py +49 -2
  4. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/embedding.py +23 -10
  5. compound_memory-0.3.0/src/compound_memory/extraction.py +521 -0
  6. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/model.py +4 -0
  7. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/scoring.py +13 -0
  8. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/server.py +15 -4
  9. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/storage.py +86 -27
  10. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/vector_index.py +5 -0
  11. {compound_memory-0.2.0 → compound_memory-0.3.0/src/compound_memory.egg-info}/PKG-INFO +1 -1
  12. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory.egg-info/SOURCES.txt +2 -0
  13. compound_memory-0.3.0/tests/test_embedding.py +154 -0
  14. compound_memory-0.3.0/tests/test_extraction.py +548 -0
  15. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_lifecycle.py +80 -0
  16. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_ns_isolation.py +72 -8
  17. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_scoring.py +25 -0
  18. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_vector_index.py +1 -0
  19. compound_memory-0.2.0/tests/test_embedding.py +0 -58
  20. {compound_memory-0.2.0 → compound_memory-0.3.0}/LICENSE +0 -0
  21. {compound_memory-0.2.0 → compound_memory-0.3.0}/README.md +0 -0
  22. {compound_memory-0.2.0 → compound_memory-0.3.0}/setup.cfg +0 -0
  23. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/__init__.py +0 -0
  24. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/index.py +0 -0
  25. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory/review_queue.py +0 -0
  26. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory.egg-info/dependency_links.txt +0 -0
  27. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory.egg-info/entry_points.txt +0 -0
  28. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory.egg-info/requires.txt +0 -0
  29. {compound_memory-0.2.0 → compound_memory-0.3.0}/src/compound_memory.egg-info/top_level.txt +0 -0
  30. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_distill.py +0 -0
  31. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_index.py +0 -0
  32. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_mcp_tools.py +0 -0
  33. {compound_memory-0.2.0 → compound_memory-0.3.0}/tests/test_model.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: compound-memory
3
- Version: 0.2.0
3
+ Version: 0.3.0
4
4
  Summary: Local multi-agent shared memory with compounding (MCP server + CLI)
5
5
  Author: chinwe
6
6
  License-Expression: MIT
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "compound-memory"
7
- version = "0.2.0"
7
+ version = "0.3.0"
8
8
  description = "Local multi-agent shared memory with compounding (MCP server + CLI)"
9
9
  readme = "README.md"
10
10
  # PEP 639 SPDX 表达式;与 License:: classifier 互斥,故后者已移除
@@ -12,6 +12,7 @@ from typing import Any
12
12
 
13
13
  from . import __version__
14
14
  from .embedding import auto_encoder
15
+ from .extraction import extract, extract_dir
15
16
  from .storage import DISTILL_DUP_SIM_THRESHOLD, MEMORY_TYPES, MemoryStore, PROMOTION_USES_THRESHOLD, default_root
16
17
 
17
18
 
@@ -35,7 +36,17 @@ def cmd_init(args: argparse.Namespace) -> None:
35
36
 
36
37
 
37
38
  def cmd_write(args: argparse.Namespace) -> None:
38
- _emit(_open_store(args).write(content=args.content, type=args.type, source=args.source, ns=args.ns, key=args.key))
39
+ _emit(
40
+ _open_store(args).write(
41
+ content=args.content,
42
+ type=args.type,
43
+ source=args.source,
44
+ ns=args.ns,
45
+ key=args.key,
46
+ valid_from=args.valid_from,
47
+ valid_until=args.valid_until,
48
+ )
49
+ )
39
50
 
40
51
 
41
52
  def cmd_search(args: argparse.Namespace) -> None:
@@ -117,6 +128,14 @@ def cmd_distill_apply(args: argparse.Namespace) -> None:
117
128
  )
118
129
 
119
130
 
131
+ def cmd_extract(args: argparse.Namespace) -> None:
132
+ target = Path(args.transcript)
133
+ if target.is_dir():
134
+ _emit(extract_dir(target, _open_store(args)))
135
+ else:
136
+ _emit(extract(target, _open_store(args)))
137
+
138
+
120
139
  def cmd_git_log(args: argparse.Namespace) -> None:
121
140
  _emit(_open_store(args).git_log(limit=args.limit))
122
141
 
@@ -132,10 +151,16 @@ def build_parser() -> argparse.ArgumentParser:
132
151
  p = sub.add_parser("write")
133
152
  p.add_argument("content"); p.add_argument("type", choices=MEMORY_TYPES)
134
153
  p.add_argument("source"); p.add_argument("--ns", default="_shared"); p.add_argument("--key", default=None)
154
+ p.add_argument("--valid-from", default=None, help="ISO date: fact valid from (annotation)")
155
+ p.add_argument("--valid-until", default=None,
156
+ help="ISO date: fact expires after this day (excluded from search, still readable via get)")
135
157
  p.set_defaults(func=cmd_write)
136
158
 
137
159
  p = sub.add_parser("search")
138
- p.add_argument("query"); p.add_argument("--ns", default="_shared"); p.add_argument("--top-k", type=int, default=5)
160
+ p.add_argument("query")
161
+ p.add_argument("--ns", default=None,
162
+ help="scope: '_shared', 'agent-<name>', or omit for _shared + your own private ns")
163
+ p.add_argument("--top-k", type=int, default=5)
139
164
  p.add_argument("--reader", default=None, help="caller identity, required for private agent-* namespaces")
140
165
  p.add_argument("--no-neighbors", dest="include_neighbors", action="store_false",
141
166
  help="omit embedded one-hop neighbors from hits")
@@ -189,6 +214,28 @@ def build_parser() -> argparse.ArgumentParser:
189
214
  p.add_argument("--all", action="store_true", help="clear the whole queue")
190
215
  p.set_defaults(func=cmd_review_resolve)
191
216
  p = sub.add_parser("git-log"); p.add_argument("--limit", type=int, default=5); p.set_defaults(func=cmd_git_log)
217
+ p = sub.add_parser(
218
+ "extract",
219
+ help="scan a session transcript for memory candidates (deterministic pass, no LLM)",
220
+ description="Deterministic candidate extraction from a session transcript. Transcript shapes are "
221
+ "auto-detected by content (not filename): WorkBuddy session log jsonl, the ZCode session database "
222
+ "(~/.zcode/cli/db/db.sqlite, full history), Claude Code session log jsonl, and DeepSeek Harness "
223
+ "zstd-compressed session files. Pass a directory to batch-scan every supported session file under it "
224
+ "(<project>/<session>.jsonl layouts and <project>/<session>/session.jsonl.zstd; subagents/ skipped — "
225
+ "their role:user is the team-lead agent's task brief, not the human's own statement). "
226
+ "ZCode rollout/model-io snapshots and WorkBuddy traces/ are deliberately unsupported — they keep only "
227
+ "the most recent / first turns, so accepting them would look like a scan while silently dropping most "
228
+ "of the history. "
229
+ "Pattern matching only (statement -> fact, pitfall -> insight); the manifest lands in "
230
+ "<root>/extract/last-candidates.json. Writing stays with the agent: confirm each candidate "
231
+ "via memory_write (same-key conflicts still enter the review queue).",
232
+ )
233
+ p.add_argument(
234
+ "transcript",
235
+ help="path to a session log jsonl (WorkBuddy / Claude Code), the ZCode session database (.sqlite), "
236
+ "a DeepSeek Harness session file (.jsonl.zstd), or a directory of session logs",
237
+ )
238
+ p.set_defaults(func=cmd_extract)
192
239
  return parser
193
240
 
194
241
 
@@ -22,6 +22,11 @@ from typing import Callable
22
22
  MODEL_REPO_ID = os.environ.get("COMPOUND_MEMORY_EMBEDDING_MODEL", "Xenova/bge-small-zh-v1.5")
23
23
  EMBED_DIM = int(os.environ.get("COMPOUND_MEMORY_EMBEDDING_DIM", "512"))
24
24
 
25
+ # 单次 onnx run 的批量上限:rebuild 把全库一次喂进来会变成单次巨批 run
26
+ # (LongMemEval 960 条长会话实测 >40 分钟无进度),切块让成本线性可控。
27
+ # 对调用方 encode 仍是一次全量调用——这是批量面优化,与增量索引无关。
28
+ ENCODE_CHUNK = 32
29
+
25
30
 
26
31
  def _cache_glob(repo_id: str) -> str:
27
32
  """HF 缓存目录 glob:repo id 的 "/" 替换为 "--"(如 a/b → models--a--b)。"""
@@ -76,7 +81,11 @@ class BgeEncoder:
76
81
  self._tokenizer: "Tokenizer | None" = None
77
82
 
78
83
  def encode(self, texts: list[str]) -> list[list[float]]:
79
- """批量编码;L2 归一化后的 [CLS] 表示(余弦可直接用作相似度)。"""
84
+ """批量编码;L2 归一化后的 [CLS] 表示(余弦可直接用作相似度)。
85
+
86
+ 内部按 ENCODE_CHUNK 切块逐次 run:外部仍是一次全量调用、返回顺序
87
+ 与输入一一对应,只是把单次巨批拆成有界小批(issue #16)。
88
+ """
80
89
  assert VEC_AVAILABLE # 构造已保证;reassure 类型检查
81
90
  if self._session is None or self._tokenizer is None:
82
91
  self._tokenizer = Tokenizer.from_file(str(self._tokenizer_path))
@@ -84,13 +93,17 @@ class BgeEncoder:
84
93
  self._tokenizer.enable_padding()
85
94
  self._session = ort.InferenceSession(str(self._onnx_path), providers=["CPUExecutionProvider"])
86
95
  encs = self._tokenizer.encode_batch(texts)
87
- feed = {
88
- "input_ids": np.array([e.ids for e in encs], dtype=np.int64),
89
- "attention_mask": np.array([e.attention_mask for e in encs], dtype=np.int64),
90
- "token_type_ids": np.array([e.type_ids for e in encs], dtype=np.int64),
91
- }
92
96
  names = {i.name for i in self._session.get_inputs()}
93
- out = self._session.run(None, {k: v for k, v in feed.items() if k in names})[0]
94
- cls = out[:, 0, :]
95
- normed = cls / np.linalg.norm(cls, axis=1, keepdims=True)
96
- return normed.tolist()
97
+ out_rows: list[list[float]] = []
98
+ for start in range(0, len(encs), ENCODE_CHUNK):
99
+ batch = encs[start : start + ENCODE_CHUNK]
100
+ feed = {
101
+ "input_ids": np.array([e.ids for e in batch], dtype=np.int64),
102
+ "attention_mask": np.array([e.attention_mask for e in batch], dtype=np.int64),
103
+ "token_type_ids": np.array([e.type_ids for e in batch], dtype=np.int64),
104
+ }
105
+ out = self._session.run(None, {k: v for k, v in feed.items() if k in names})[0]
106
+ cls = out[:, 0, :]
107
+ normed = cls / np.linalg.norm(cls, axis=1, keepdims=True)
108
+ out_rows.extend(normed.tolist())
109
+ return out_rows
@@ -0,0 +1,521 @@
1
+ """抽取清单扫描器:会话 transcript → 记忆候选(抽取管线的确定性段,零 LLM)。
2
+
3
+ 蒸馏三段式的第二应用(spec:确定性准备自动跑、判断由 Agent 完成):
4
+ 扫描只「发现候选」,写库仍由 Agent 逐条确认走 memory_write——清单是建议、
5
+ 写入是动作,同 key 冲突照常进 review 队列(不静默原则在管线里不变)。
6
+
7
+ transcript 解析支持两种宿主格式,按内容形状分发(不靠文件名约定):
8
+
9
+ 1. **session log**(WorkBuddy 主源):`<project>/<sessionId>.jsonl`,逐轮完整
10
+ 消息(type=message / role / content[].text)。真实用户话被 `<user_query>` 或
11
+ `<session>` 包裹,同块内混着注入块(user-context / team-context / 队友消息 /
12
+ 上下文压缩摘要)——壳剥掉、注入块整块跳过。
13
+ 2. **ZCode 会话库**(`~/.zcode/cli/db/db.sqlite`):全量对话在 SQLite 里
14
+ (message+part 表,正文在 part 的 text 块)。rollout/model-io jsonl 快照
15
+ 已退役不接——它只剩最近几个会话,只接快照会产出"看似扫过、实则只盖住
16
+ 冰山一角"的假阴性,与 trace 同等对待。
17
+ 3. **Claude Code session log**(`~/.claude/projects/<项目>/<sessionId>.jsonl`):
18
+ 真实输入 = `type=='user'` 的 message.content(字符串或 text 块);isMeta
19
+ (UI 回显/命令展开)、isSidechain(子 agent 转述)、tool_result 块与
20
+ `<command-*>`/`<local-command-stdout>` 包装都不是用户话。
21
+ 4. **DeepSeek Harness session**(`~/.dsh/sessions/<项目>/<会话>/session.jsonl.zstd`):
22
+ zstd 压缩的 JSONL(经系统 zstd CLI 解压,不为此引 C 扩展依赖)。真实输入 =
23
+ `user/message` 且 `data.source.kind=='user'`(runtime-context 快照 / 技能
24
+ 注入 / 审批通知走别的 source.kind);`session.origin=='subagent'` 的子会话
25
+ 是主 agent 派活文本,整场返回空。
26
+
27
+ **刻意不支持 trace**(`~/.workbuddy/traces/<pid>/trace_*.json`):generation
28
+ span 的 toolInput 是请求快照,但被**头部**硬截到 100000 字符,整段解析必抛;
29
+ 即便逐条 raw_decode 抢救(实测 845/845 span 成功),单快照也只剩**首轮** user
30
+ 消息,多轮会话损失严重。接它会产出"看起来扫描过、实际漏掉大部分会话"的假阴性,
31
+ 比明确不支持更有害。traces/ 只作排障线索,不是抽取源。
32
+
33
+ 模式匹配面向中文宿主场景:statement 模式 → fact 候选、pitfall 模式 →
34
+ insight 候选;提供 store 时对 _shared 做词面去重标注(likely_dup_of 指向
35
+ 既有条目,Agent 复用同 key 而非新开条目),命中相似仍进清单——丢弃与否
36
+ 是判断段的事,扫描器不静默吞。
37
+ """
38
+
39
+ from __future__ import annotations
40
+
41
+ import datetime as dt
42
+ import json
43
+ import re
44
+ import shutil
45
+ import sqlite3
46
+ import subprocess
47
+ from pathlib import Path
48
+ from typing import Any
49
+
50
+ from .model import Memory
51
+ from .scoring import doc_text, tokenize
52
+ from .storage import MemoryStore
53
+
54
+ # 声明类模式(用户陈述事实/偏好/环境)→ fact 候选;踩坑类 → insight 候选。
55
+ # 面向中文宿主场景;英文会话 P0 不覆盖(模式表后续按需扩充)。
56
+ STATEMENT_PATTERNS = (
57
+ "我用", "我用的是", "默认用", "以后都", "记住", "偏好",
58
+ "换成", "部署在", "装了", "升级了", "安装了", "部署了", "迁移到",
59
+ "地址是", "密码是", "账号是", "版本是", "端口是",
60
+ )
61
+ PITFALL_PATTERNS = (
62
+ # 否定指令优先于泛坑描述:「不要用 X」比「有坑」信号更明确
63
+ "不要用", "别用", "报错", "踩坑", "坑是", "有坑", "失败", "不行",
64
+ "问题出在", "注意", "超时", "限制是",
65
+ )
66
+
67
+ # 注入块标记:命中即整块跳过(hook 注入、系统提醒、通知、命令展开都不是用户话)
68
+ INJECTION_MARKERS = (
69
+ "<system-reminder",
70
+ "[SYSTEM NOTIFICATION",
71
+ "<task-notification",
72
+ "<command-name>",
73
+ "<local-command",
74
+ "Caveat:",
75
+ "[Request interrupted",
76
+ "<command-message",
77
+ # WorkBuddy 形态:Agent Team 注入、队友派活、上下文压缩摘要、续写指令
78
+ "<teammate-message",
79
+ "<user-prompt-submit-hook",
80
+ "<conversation_history_summary>",
81
+ "Please continue with the conversation based on the summarized context",
82
+ "You are a prompt enhancement assistant",
83
+ )
84
+
85
+ # 真实用户话外壳:WorkBuddy 把用户输入包在 <user_query>(主路)或 <session>
86
+ # (远程/小程序回传路径)里;先剥壳再判注入,否则整块会被壳掩盖成"非注入"。
87
+ USER_WRAPPERS = (
88
+ re.compile(r"<user_query>\s*(.*?)\s*</user_query>", re.S),
89
+ re.compile(r"<session>\s*(.*?)\s*</session>", re.S),
90
+ )
91
+
92
+ MAX_CANDIDATES = 20 # 每次扫描的清单上限:防喋喋不休的会话产出垃圾清单
93
+ EXTRACT_MAX_SESSIONS = 500 # 批量模式单次最多吃多少个会话文件(防目录爆量)
94
+ QUOTE_CHARS = 200 # 候选摘录截断
95
+ # 去重标注阈值:查询 token 被库内条目覆盖率(containment)。不用 normalized BM25——
96
+ # 长句查询的分母惩罚使复述句也只有 ~0.12,结构性偏低;覆盖率对「复述检测」语义正确
97
+ EXTRACT_DUP_COVERAGE = 0.5
98
+ SENTENCE_SPLIT = "。!?!?;;\n"
99
+
100
+
101
+ def _texts_from_messages(messages: Any) -> list[str]:
102
+ """messages 数组 → 剥壳去注入后的真实用户话块。"""
103
+ out: list[str] = []
104
+ if not isinstance(messages, list):
105
+ return out
106
+ for message in messages:
107
+ if not isinstance(message, dict) or message.get("role") != "user":
108
+ continue
109
+ content = message.get("content")
110
+ blocks = content if isinstance(content, list) else [{"type": "text", "text": str(content)}]
111
+ for block in blocks:
112
+ if not isinstance(block, dict) or block.get("type") not in ("text", "input_text"):
113
+ continue
114
+ cleaned = _unwrap_user_text((block.get("text") or "").strip())
115
+ if cleaned:
116
+ out.append(cleaned)
117
+ return out
118
+
119
+
120
+ def _unwrap_user_text(text: str) -> str | None:
121
+ """剥 <user_query>/<session> 壳并滤注入块;不是用户话返回 None。
122
+
123
+ 壳优先于注入判定:用户话整体被壳包裹,若先按整块判注入会漏掉真实输入。
124
+ 壳内再判注入(壳里塞 system-reminder 的形态确实存在)。
125
+ """
126
+ if not text:
127
+ return None
128
+ for wrapper in USER_WRAPPERS:
129
+ matched = wrapper.search(text)
130
+ if matched:
131
+ inner = matched.group(1).strip()
132
+ if not inner or inner.startswith(INJECTION_MARKERS):
133
+ return None
134
+ return inner
135
+ if text.startswith(INJECTION_MARKERS):
136
+ return None
137
+ return text
138
+
139
+
140
+ def _dedupe(texts: list[str]) -> list[str]:
141
+ return list(dict.fromkeys(texts))
142
+
143
+
144
+ def user_texts_from_zcode_db(path: Path) -> list[str]:
145
+ """ZCode 会话库(SQLite)→ 真实用户话(保序去重)。
146
+
147
+ db 是活动 ZCode 进程的 WAL 库,只读打开(mode=ro)绝不写。用户消息由
148
+ message.data.role=='user' 定位,正文是 part 表 text 块;synthetic 与
149
+ model-only 的块是运行时注入(todo 提醒、hook),连同 system-reminder 等
150
+ 注入标记一并滤掉——注入过滤复用 _unwrap_user_text(与 WorkBuddy 同一堵墙)。
151
+ """
152
+ con = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True)
153
+ try:
154
+ rows = con.execute(
155
+ """
156
+ SELECT p.data
157
+ FROM message m JOIN part p ON p.message_id = m.id
158
+ WHERE json_valid(m.data) AND json_valid(p.data)
159
+ AND json_extract(m.data, '$.role') = 'user'
160
+ AND json_extract(p.data, '$.type') = 'text'
161
+ ORDER BY m.session_id, m.time_created, p.sequence
162
+ """
163
+ ).fetchall()
164
+ finally:
165
+ con.close()
166
+ texts: list[str] = []
167
+ for (raw,) in rows:
168
+ # json_valid + '$.type'='text' 已在 SQL 侧保证只剩合法 JSON 的 object
169
+ # 形状行(非 object 的 json_extract 返回 NULL,被 WHERE 过滤)——这里
170
+ # 只再做注入块的语义过滤;metadata 形状仍防御一次(宁跳过不崩扫描)
171
+ part: Any = json.loads(raw)
172
+ metadata = part.get("metadata")
173
+ if part.get("synthetic") or (isinstance(metadata, dict) and metadata.get("visibility") == "model-only"):
174
+ continue
175
+ cleaned = _unwrap_user_text((part.get("text") or "").strip())
176
+ if cleaned:
177
+ texts.append(cleaned)
178
+ return _dedupe(texts)
179
+
180
+
181
+ def user_texts_from_claude_log(path: Path) -> list[str]:
182
+ """Claude Code session log jsonl → 真实用户话(保序去重)。
183
+
184
+ 只取 type=='user' 的真实输入:跳过 isSidechain(子 agent 转述)与 isMeta
185
+ (UI 回显、命令展开);content 字符串或 text 块都过注入过滤——<command-*>
186
+ 包装、local-command-stdout、tool_result 块、"[Request interrupted]" 提示
187
+ 不是用户话。
188
+ """
189
+ texts: list[str] = []
190
+ for event in _iter_json_lines(path):
191
+ if not isinstance(event, dict) or event.get("type") != "user":
192
+ continue
193
+ if event.get("isSidechain") or event.get("isMeta"):
194
+ continue
195
+ message = event.get("message")
196
+ content = message.get("content") if isinstance(message, dict) else None
197
+ blocks = [content] if isinstance(content, str) else content if isinstance(content, list) else []
198
+ for block in blocks:
199
+ if isinstance(block, str):
200
+ text = block
201
+ elif isinstance(block, dict) and block.get("type") == "text":
202
+ text = block.get("text") or ""
203
+ else:
204
+ continue
205
+ cleaned = _unwrap_user_text(text.strip())
206
+ if cleaned:
207
+ texts.append(cleaned)
208
+ return _dedupe(texts)
209
+
210
+
211
+ def _zstd_decompress(path: Path) -> str:
212
+ """zstd CLI 解压出文本(dsh 会话是 zstd 压缩 JSONL;用系统 CLI 免引 C 扩展依赖)。"""
213
+ zstd = shutil.which("zstd")
214
+ if zstd is None:
215
+ raise ValueError(
216
+ f"zstd CLI not found on PATH; it is required to read DeepSeek Harness session files: {path}"
217
+ )
218
+ proc = subprocess.run([zstd, "-dc", str(path)], capture_output=True, check=False)
219
+ if proc.returncode != 0:
220
+ stderr = proc.stderr.decode("utf-8", errors="replace").strip()
221
+ raise ValueError(f"zstd failed to decompress {path}: {stderr}")
222
+ return proc.stdout.decode("utf-8", errors="replace")
223
+
224
+
225
+ def user_texts_from_dsh_session(path: Path) -> list[str]:
226
+ """DeepSeek Harness session.jsonl.zstd → 真实用户话(保序去重)。
227
+
228
+ 真实输入 = user/message 且 data.source.kind=='user'——runtime-context
229
+ 快照、技能注入、审批通知都走别的 source.kind,结构上就能分开,不必靠
230
+ 文本模式硬猜。session.origin=='subagent' 的子会话是主 agent 派活文本
231
+ (第三人称转述,与 WorkBuddy subagents/ 同型噪声),整场返回空。
232
+ """
233
+ texts: list[str] = []
234
+ subagent = False
235
+ for line in _zstd_decompress(path).splitlines():
236
+ line = line.strip()
237
+ if not line:
238
+ continue
239
+ try:
240
+ event = json.loads(line)
241
+ except json.JSONDecodeError:
242
+ continue
243
+ if not isinstance(event, dict):
244
+ continue
245
+ if event.get("type") == "session":
246
+ # session 事件在文件首行,先于所有用户消息
247
+ subagent = event.get("origin") == "subagent"
248
+ continue
249
+ if event.get("type") != "user/message":
250
+ continue
251
+ data = event.get("data")
252
+ source = data.get("source") if isinstance(data, dict) else None
253
+ if not isinstance(source, dict) or source.get("kind") != "user":
254
+ continue
255
+ content = data.get("content") if isinstance(data, dict) else None
256
+ for block in content if isinstance(content, list) else []:
257
+ if isinstance(block, dict) and block.get("type") == "text":
258
+ cleaned = _unwrap_user_text((block.get("text") or "").strip())
259
+ if cleaned:
260
+ texts.append(cleaned)
261
+ return [] if subagent else _dedupe(texts)
262
+
263
+
264
+ def user_texts_from_session_log(path: Path) -> list[str]:
265
+ """WorkBuddy session log jsonl → 真实用户话(保序去重)。
266
+
267
+ 每行一个事件,只取 type=="message" && role=="user" 的 input_text 块——
268
+ function_call / reasoning / file-history-snapshot 等事件不是用户话。
269
+ """
270
+ texts: list[str] = []
271
+ for event in _iter_json_lines(path):
272
+ if not isinstance(event, dict) or event.get("type") != "message" or event.get("role") != "user":
273
+ continue
274
+ texts.extend(_texts_from_messages([event]))
275
+ return _dedupe(texts)
276
+
277
+
278
+ def _iter_json_lines(path: Path) -> list[Any]:
279
+ """逐行读 jsonl,坏行跳过(宁少一条输入,不让整场扫描崩掉)。"""
280
+ events: list[Any] = []
281
+ with open(path, encoding="utf-8", errors="replace") as fh:
282
+ for line in fh:
283
+ line = line.strip()
284
+ if not line:
285
+ continue
286
+ try:
287
+ events.append(json.loads(line))
288
+ except json.JSONDecodeError:
289
+ continue
290
+ return events
291
+
292
+
293
+ def _sentences(text: str) -> list[str]:
294
+ buf: list[str] = []
295
+ for piece in text.split(SENTENCE_SPLIT):
296
+ piece = piece.strip()
297
+ if piece:
298
+ buf.append(piece)
299
+ return buf
300
+
301
+
302
+ def _match_pattern(sentence: str) -> tuple[str, str] | None:
303
+ """返回 (suggested_type, signal) 或 None;statement 优先于 pitfall。"""
304
+ for pattern in STATEMENT_PATTERNS:
305
+ if pattern in sentence:
306
+ return "fact", pattern
307
+ for pattern in PITFALL_PATTERNS:
308
+ if pattern in sentence:
309
+ return "insight", pattern
310
+ return None
311
+
312
+
313
+ def _dup_of(store: MemoryStore, quote: str) -> str | None:
314
+ """库内(仅 _shared)复述标注:查询 token 被单条条目覆盖率最高者达阈值即标注。
315
+
316
+ 私有 ns 无身份不读(访问控制不变量);覆盖率低于阈值返回 None——漏标由
317
+ Agent 自行 search 兜底,扫描器不静默吞候选。
318
+ """
319
+ q_tokens = set(tokenize(quote))
320
+ if not q_tokens:
321
+ return None
322
+ candidates = store._candidates(sorted(q_tokens), {"_shared"})
323
+ best_id, best_cov = None, 0.0
324
+ for mem in candidates:
325
+ coverage = len(q_tokens & set(tokenize(doc_text(mem)))) / len(q_tokens)
326
+ if coverage > best_cov:
327
+ best_id, best_cov = mem.id, coverage
328
+ return best_id if best_cov >= EXTRACT_DUP_COVERAGE else None
329
+
330
+
331
+ def scan_texts(
332
+ texts: list[str],
333
+ store: MemoryStore | None = None,
334
+ max_items: int = MAX_CANDIDATES,
335
+ ) -> list[dict[str, Any]]:
336
+ """用户话 → 候选清单:分句、模式匹配、去重标注、上限截断。
337
+
338
+ 候选字段:quote(原句摘录)、suggested_type、signal(命中模式)、
339
+ likely_dup_of(既有条目 id 或 None)。key 由 Agent 判断时定——扫描器
340
+ 不猜 key(spec 写入约定:复用既有 key 依赖对库内现状的判断)。
341
+ """
342
+ candidates: list[dict[str, Any]] = []
343
+ for text in texts:
344
+ for sentence in _sentences(text):
345
+ matched = _match_pattern(sentence)
346
+ if matched is None:
347
+ continue
348
+ suggested_type, signal = matched
349
+ quote = sentence[:QUOTE_CHARS] + ("…" if len(sentence) > QUOTE_CHARS else "")
350
+ candidates.append(
351
+ {
352
+ "quote": quote,
353
+ "suggested_type": suggested_type,
354
+ "signal": signal,
355
+ "likely_dup_of": _dup_of(store, quote) if store is not None else None,
356
+ }
357
+ )
358
+ if len(candidates) >= max_items:
359
+ return candidates
360
+ return candidates
361
+
362
+
363
+ DETECT_HEAD_CHARS = 65536 # 形态探测只读文件头:足够看清结构,避开大文件全读
364
+ DETECT_MAX_LINES = 20 # 最多探这么行(较新会话以多条 session-meta 开头)
365
+ SQLITE_MAGIC = b"SQLite format 3\x00"
366
+ ZSTD_MAGIC = b"\x28\xb5\x2f\xfd" # zstd 帧魔数(dsh 会话文件)
367
+
368
+
369
+ def _is_zcode_db(path: Path) -> bool:
370
+ """SQLite 魔数之外再验形状:必须有 message+part 两表(ZCode 会话库形状)。"""
371
+ con: sqlite3.Connection | None = None
372
+ try:
373
+ con = sqlite3.connect(f"{path.resolve().as_uri()}?mode=ro", uri=True)
374
+ names = {row[0] for row in con.execute("SELECT name FROM sqlite_master WHERE type='table'")}
375
+ except sqlite3.Error:
376
+ return False
377
+ finally:
378
+ if con is not None:
379
+ con.close()
380
+ return {"message", "part"} <= names
381
+
382
+
383
+ def detect_transcript_kind(path: Path) -> str:
384
+ """按内容形状判定 transcript 形态:session-log / zcode-db / claude-log /
385
+ dsh-session / unsupported。
386
+
387
+ 不靠文件名约定——各家都把日志叫 .jsonl / .sqlite / .zstd。也不只看第一
388
+ 行:较新会话以 session-meta / mode 等元事件开头,只看首行会把它们全判成
389
+ unsupported 静默跳过。
390
+
391
+ SQLite 按魔数 + message/part 表形状识别;zstd 按帧魔数识别(dsh 会话);
392
+ model-io 快照与 trace 是 retired 源(只剩最近几个会话 / 首轮 user 消息),
393
+ 判 unsupported 而非 unknown——给出行内理由并指向受支持源,避免"看似扫过、
394
+ 实则大面积漏"。
395
+ """
396
+ with open(path, "rb") as fh:
397
+ raw = fh.read(DETECT_HEAD_CHARS)
398
+ if raw.startswith(SQLITE_MAGIC):
399
+ return "zcode-db" if _is_zcode_db(path) else "unsupported"
400
+ if raw.startswith(ZSTD_MAGIC):
401
+ return "dsh-session"
402
+ head = raw.decode("utf-8", errors="replace")
403
+ for line in head.splitlines()[:DETECT_MAX_LINES]:
404
+ line = line.strip()
405
+ if not line.startswith("{"):
406
+ continue
407
+ try:
408
+ event = json.loads(line)
409
+ except json.JSONDecodeError:
410
+ continue
411
+ if not isinstance(event, dict):
412
+ continue
413
+ if event.get("type") == "message":
414
+ return "session-log"
415
+ if event.get("type") == "user" and isinstance(event.get("message"), dict):
416
+ return "claude-log"
417
+ return "unsupported"
418
+
419
+
420
+ PARSERS = {
421
+ "session-log": user_texts_from_session_log,
422
+ "zcode-db": user_texts_from_zcode_db,
423
+ "claude-log": user_texts_from_claude_log,
424
+ "dsh-session": user_texts_from_dsh_session,
425
+ }
426
+
427
+ UNSUPPORTED_HINT = (
428
+ "不支持的 transcript 形态:{path}。受支持的源有——"
429
+ "WorkBuddy session log(~/.workbuddy/projects/<项目>/<sessionId>.jsonl)、"
430
+ "ZCode 会话库(~/.zcode/cli/db/db.sqlite)、"
431
+ "Claude Code session log(~/.claude/projects/<项目>/<sessionId>.jsonl)、"
432
+ "DeepSeek Harness session(~/.dsh/sessions/<项目>/<会话>/session.jsonl.zstd);"
433
+ "jsonl 传目录则批量扫。"
434
+ "WorkBuddy traces/ 与 ZCode rollout/model-io 快照不接入:都只剩部分轮次,"
435
+ "接进来是'看似扫过、实则大面积漏'的假阴性。"
436
+ )
437
+
438
+
439
+ def extract(transcript: Path, store: MemoryStore) -> dict[str, Any]:
440
+ """入口:解析 transcript、扫描、清单落 <root>/extract/last-candidates.json。
441
+
442
+ extract/ 是运行时工件目录(_ensure_layout 统一 gitignore)——清单含会话
443
+ 摘录,不进记忆库的审计史;返回摘要供 CLI 打印。
444
+ """
445
+ kind = detect_transcript_kind(transcript)
446
+ parser = PARSERS.get(kind)
447
+ if parser is None:
448
+ raise ValueError(UNSUPPORTED_HINT.format(path=transcript))
449
+ texts = parser(transcript)
450
+ candidates = scan_texts(texts, store=store)
451
+ return _write_manifest(store, transcript, kind, texts, candidates)
452
+
453
+
454
+ def _write_manifest(
455
+ store: MemoryStore,
456
+ source: str | Path,
457
+ kind: str,
458
+ texts: list[str],
459
+ candidates: list[dict[str, Any]],
460
+ ) -> dict[str, Any]:
461
+ out_dir = store.root / "extract"
462
+ out_dir.mkdir(parents=True, exist_ok=True)
463
+ out_path = out_dir / "last-candidates.json"
464
+ manifest = {
465
+ "generated": dt.date.today().isoformat(),
466
+ "source": str(source),
467
+ "parser": kind,
468
+ "user_turns": len(texts),
469
+ "candidates": candidates,
470
+ }
471
+ out_path.write_text(json.dumps(manifest, ensure_ascii=False, indent=2), encoding="utf-8")
472
+ return {
473
+ "candidates": len(candidates),
474
+ "user_turns": len(texts),
475
+ "parser": kind,
476
+ "out": str(out_path),
477
+ "source": str(source),
478
+ }
479
+
480
+
481
+ def extract_dir(
482
+ root: Path,
483
+ store: MemoryStore,
484
+ max_sessions: int = EXTRACT_MAX_SESSIONS,
485
+ ) -> dict[str, Any]:
486
+ """批量扫一个宿主日志目录。
487
+
488
+ 支持的目录布局:WorkBuddy `~/.workbuddy/projects` 与 Claude Code
489
+ `~/.claude/projects`(都是 `<项目>/<会话>.jsonl`),DeepSeek Harness
490
+ `~/.dsh/sessions`(`<项目>/<会话>/session.jsonl.zstd`)。每个文件按内容
491
+ 形态分派解析器,认不出的静默跳过(目录里可能混着非会话文件)。
492
+
493
+ 只吃一级会话文件,**跳过 subagents/**:那里的 role:user 是 team-lead
494
+ agent 的派活文本("用户想要…" 是第三人称转述,不是本人陈述),
495
+ 实测 3/3 候选全是噪声——混进来只会污染清单。dsh 的 subagent 子会话在
496
+ 解析器内按 session.origin 识别并返回空(计为已扫会话)。
497
+
498
+ 跨会话合并去重后再扫(同一句话在多会话复述),单会话上限由
499
+ scan_texts 的 MAX_CANDIDATES 兜底。
500
+ """
501
+ texts: list[str] = []
502
+ sessions = 0
503
+ paths: set[Path] = set()
504
+ # dsh 会话文件有两代文件名(session.jsonl.zstd / session.v3.jsonl.zstd),
505
+ # 事件形态相同——glob 只认旧名会静默漏掉新会话(实测 25 个里 14 个是 v3)
506
+ for pattern in ("*/*.jsonl", "*/*/session*.jsonl.zstd"):
507
+ paths.update(root.glob(pattern))
508
+ for log in sorted(paths):
509
+ kind = detect_transcript_kind(log)
510
+ parser = PARSERS.get(kind)
511
+ if parser is None or "subagents" in log.parts:
512
+ continue
513
+ texts.extend(parser(log))
514
+ sessions += 1
515
+ if sessions >= max_sessions:
516
+ break
517
+ merged = _dedupe(texts)
518
+ candidates = scan_texts(merged, store=store)
519
+ summary = _write_manifest(store, f"{root} (batch)", "batch", merged, candidates)
520
+ summary["sessions"] = sessions
521
+ return summary