@yottameta/yotta-logs 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,39 +1,49 @@
1
1
  #!/usr/bin/env python3
2
2
  # -*- coding: utf-8 -*-
3
- """yotta_logs.py — YottaMeta 元史(yotta-logs):跨智能体历史会话日志检索引擎。
3
+ """yotta_logs.py — YottaMeta 元史(yotta-logs):跨智能体历史会话 / 记忆日志检索引擎。
4
4
 
5
- 零依赖(Python 3.8+ 标准库),只读检索 / 分析会话 JSONL 记录,为跨会话追溯
6
- 提供原始日志依据。与元忆(yotta-memory,语义记忆)互补:本技能只管原始会话
7
- 日志的定位、检索、提取与统计;不修改、不删除、不联网上传任何会话记录。
5
+ v0.2.0 通用化:不再只认 JSONL,按「格式族 × 字段别名归一 + 配置兜底」适配
6
+ JSONL / 单文件 JSON / SQLite / Markdown / 二进制 五大格式族;discover 全源登记;
7
+ 新增 --source / --kind / --format 过滤;默认检索范围 = 会话源 + 结构化记忆源开、
8
+ 自由笔记 / 二进制日志默认关(可显式开)。格式普查见 references/agent-formats.md。
8
9
 
9
- 子命令:
10
- locate 自动发现本机常见的会话日志目录
11
- scan [--dir D] 列出目录下所有会话(ID / 日期 / 消息数 / 大小)
12
- search <query> [--dir D] 按关键词 / 正则跨会话检索,输出时间线命中
10
+ 零依赖(Python 3.8+ 标准库),只读检索 / 分析,不修改、不删除、不联网上传。
11
+ 与元忆(yotta-memory,语义记忆)互补:本技能只管原始日志 / 记忆文件的定位、
12
+ 检索、提取与统计。
13
+
14
+ 子命令(7 个语义不变):
15
+ locate 全源登记(来源 / 格式 / 类型 / 路径 / 默认范围)
16
+ scan [--dir D] 列出所有会话(来源 / ID / 日期 / 消息数 / 大小)
17
+ search <query> [--dir D] 跨源关键词 / 正则检索,输出时间线命中
13
18
  session <sid> [--dir D] 提取单个会话原文(时间线 + 角色 + 文本)
14
- stats [--dir D] 会话统计(消息 / token / 成本 / 每日汇总)
19
+ stats [--dir D] 统计(消息 / token / 成本 / 每日汇总 / 分源)
15
20
  tools [--dir D] 工具调用次数排行
16
21
  version 打印版本
17
22
 
18
23
  通用选项:
19
- --dir PATH 日志目录(缺省读环境变量 YOTTA_LOGS_DIR,再自动定位首个候选)
24
+ --dir PATH 日志 / 记忆目录或文件(目录自动嗅探格式族;缺省读
25
+ YOTTA_LOGS_DIR,再 discover 全源登记)
26
+ --source NAME 只检索指定来源(可多次;名称见 locate 登记)
27
+ --kind KIND 只检索指定类型:session / memory / note / log
28
+ --format FMT 只检索指定格式:jsonl / json / sqlite / markdown / binary
20
29
  --json 输出纯 JSON(stdout 无其它噪音)
21
- --no-redact 关闭默认脱敏(默认会把疑似密钥 / token / 口令打码)
30
+ --no-redact 关闭默认脱敏
22
31
  --limit N 最多返回 N 条(默认 50)
23
32
 
24
33
  退出码(与元安 / 元审 / 元盾 / 元真家族一致):
25
34
  0 = 成功(检索到结果 / 操作完成)
26
35
  1 = 无匹配 / 空结果集(search 未命中、scan / stats 无会话)
27
- 4 = 用法错误 / 目录不存在 / 致命异常
36
+ 4 = 用法错误 / 路径不存在 / 致命异常
28
37
 
29
38
  用法示例:
30
39
  python3 yotta_logs.py locate
31
40
  python3 yotta_logs.py scan --dir ~/.clawdbot/agents/dashu/sessions
32
- python3 yotta_logs.py search "部署方案" --dir /path/to/sessions
33
- python3 yotta_logs.py search "CI 失败" --regex --date 2026-08-26
41
+ python3 yotta_logs.py search "部署方案"
42
+ python3 yotta_logs.py search "CI 失败" --regex --date 2026-08-26 --source opencode-db
43
+ python3 yotta_logs.py search "记住" --kind memory
34
44
  python3 yotta_logs.py session abc123 --role assistant
35
45
  python3 yotta_logs.py stats --dir /path/to/sessions --daily
36
- python3 yotta_logs.py tools --dir /path/to/sessions
46
+ python3 yotta_logs.py tools --dir /path/to/logs --format sqlite
37
47
  """
38
48
  import argparse
39
49
  import datetime as _dt
@@ -41,6 +51,7 @@ import glob
41
51
  import json
42
52
  import os
43
53
  import re
54
+ import sqlite3
44
55
  import sys
45
56
  from pathlib import Path
46
57
 
@@ -53,14 +64,30 @@ try:
53
64
  except Exception:
54
65
  pass
55
66
 
56
- VERSION = "0.1.0"
67
+ VERSION = "0.2.0"
57
68
  TOOL_NAME = "yotta-logs"
58
69
  TOOL_CN = "元史"
59
70
  DEFAULT_LIMIT = 50
60
71
  DEFAULT_CONTEXT = 40 # 命中上下文半径(字符)
61
72
  JSONL_SUFFIXES = (".jsonl", ".jsonlines", ".ndjson")
62
- ROLE_TOOL = ("tool", "toolResult", "tool_result")
63
-
73
+ JSON_SUFFIXES = (".json",)
74
+ SQLITE_SUFFIXES = (".db", ".sqlite", ".sqlite3", ".vscdb")
75
+ MD_SUFFIXES = (".md", ".markdown", ".mdown")
76
+ BINARY_SUFFIXES = (".pbtxt", ".nitrite", ".cascade", ".bin", ".enc")
77
+ ROLE_TOOL = ("tool", "toolResult", "tool_result", "toolCall", "tool_call",
78
+ "function", "functionCall", "function_call")
79
+ KIND_CHOICES = ("session", "memory", "note", "log")
80
+ FORMAT_CHOICES = ("jsonl", "json", "sqlite", "markdown", "binary")
81
+
82
+ # 字段别名(按序取首个命中)——适配一切关键字段
83
+ TIME_ALIASES = ("timestamp", "time_created", "created", "time", "ts", "date",
84
+ "created_at", "mtime", "updated")
85
+ ROLE_ALIASES = ("role", "type", "kind")
86
+ TEXT_ALIASES = ("text", "content", "body", "message", "statement",
87
+ "text_content")
88
+ SESSION_ALIASES = ("session_id", "thread_id", "sessionId", "session",
89
+ "conversation_id", "threadId")
90
+ TITLE_ALIASES = ("title", "subject", "name", "heading")
64
91
 
65
92
  # ── 脱敏(默认开启)──────────────────────────────────────────────────────
66
93
 
@@ -107,109 +134,11 @@ def redact(text):
107
134
  out.append(chunk)
108
135
  return "".join(out)
109
136
 
137
+ # ── JSONL 会话日志目录(兼容保留)───────────────────────────────────────
110
138
 
111
- # ── 记录解析(容错:字段缺失不报错,坏行由 parse_jsonl 计数跳过)─────────
112
-
113
- def _rec_ts(rec):
114
- ts = rec.get("timestamp")
115
- if not ts and isinstance(rec.get("message"), dict):
116
- ts = rec["message"].get("timestamp")
117
- return str(ts) if ts else ""
118
-
119
-
120
- def _rec_role(rec):
121
- msg = rec.get("message")
122
- if isinstance(msg, dict) and msg.get("role"):
123
- role = str(msg["role"])
124
- else:
125
- role = rec.get("role")
126
- role = str(role) if role else ""
127
- if role in ROLE_TOOL:
128
- return "tool"
129
- return role
130
-
131
-
132
- def _rec_content(rec):
133
- msg = rec.get("message")
134
- if isinstance(msg, dict):
135
- return msg.get("content")
136
- return rec.get("content")
137
-
138
-
139
- def _rec_text(rec):
140
- """提取记录里的人类可读文本(content 列表只取 type=text,字符串直接取)。"""
141
- content = _rec_content(rec)
142
- if isinstance(content, str):
143
- return content
144
- if isinstance(content, list):
145
- parts = []
146
- for item in content:
147
- if not isinstance(item, dict):
148
- continue
149
- if item.get("type") == "text" and item.get("text"):
150
- parts.append(str(item["text"]))
151
- return "\n".join(parts)
152
- return ""
153
-
154
-
155
- def _rec_tool_names(rec):
156
- """提取记录里的工具调用名(toolCall / toolResult)。"""
157
- content = _rec_content(rec)
158
- names = []
159
- if isinstance(content, list):
160
- for item in content:
161
- if not isinstance(item, dict):
162
- continue
163
- if item.get("type") in ("tool_call", "toolCall", "toolResult"):
164
- nm = item.get("name") or item.get("toolName") or ""
165
- if nm:
166
- names.append(str(nm))
167
- return names
168
-
169
-
170
- def _rec_cost(rec):
171
- msg = rec.get("message")
172
- usage = None
173
- if isinstance(msg, dict):
174
- usage = msg.get("usage")
175
- if not isinstance(usage, dict):
176
- usage = rec.get("usage")
177
- if not isinstance(usage, dict):
178
- return 0.0
179
- cost = usage.get("cost")
180
- if isinstance(cost, dict):
181
- return float(cost.get("total") or 0)
182
- try:
183
- return float(cost or 0)
184
- except (TypeError, ValueError):
185
- return 0.0
186
-
187
-
188
- def _rec_tokens(rec):
189
- msg = rec.get("message")
190
- usage = None
191
- if isinstance(msg, dict):
192
- usage = msg.get("usage")
193
- if not isinstance(usage, dict):
194
- usage = rec.get("usage")
195
- if not isinstance(usage, dict):
196
- return (0, 0)
197
- return (int(usage.get("input_tokens") or 0),
198
- int(usage.get("output_tokens") or 0))
199
-
200
-
201
- def _is_message(rec):
202
- """是否为可计入统计的消息记录(排除 session 元数据 / 空角色)。"""
203
- role = _rec_role(rec)
204
- if role in ("", "session"):
205
- return False
206
- return True
207
-
208
-
209
- # ── 会话日志目录 ─────────────────────────────────────────────────────────
210
139
 
211
140
  def discover_dirs():
212
- """自动发现本机常见会话日志目录(只返回存在且含 *.jsonl 的目录)。"""
141
+ """自动发现本机常见 JSONL 会话日志目录(只返回存在且含 *.jsonl 的目录)。"""
213
142
  home = Path.home()
214
143
  patterns = [
215
144
  home / ".clawdbot" / "agents" / "*" / "sessions",
@@ -219,6 +148,8 @@ def discover_dirs():
219
148
  home / ".gemini" / "sessions",
220
149
  home / ".agents" / "sessions",
221
150
  ]
151
+ if os.environ.get("CODEX_HOME"):
152
+ patterns.append(Path(os.environ["CODEX_HOME"]) / "sessions")
222
153
  found = []
223
154
  for pat in patterns:
224
155
  for d in glob.glob(str(pat)):
@@ -339,81 +270,1167 @@ def _human_size(n):
339
270
  return "%d B" % n
340
271
 
341
272
 
342
- # ── 检索 / 提取 / 统计 ──────────────────────────────────────────────────
273
+ # ── 记录提取(JSONL 消息形态;兼容保留)─────────────────────────────────
343
274
 
344
- def scan_sessions(dir_path):
345
- idx = load_index(dir_path)
346
- rows = []
347
- total_messages = 0
348
- total_invalid = 0
349
- for info in list_sessions(dir_path):
350
- records, invalid = parse_jsonl(info["path"])
351
- total_invalid += invalid
352
- first_ts = ""
353
- messages = 0
354
- for rec in records:
355
- if not _is_message(rec):
275
+ def _rec_ts(rec):
276
+ ts = rec.get("timestamp")
277
+ if not ts and isinstance(rec.get("message"), dict):
278
+ ts = rec["message"].get("timestamp")
279
+ if not ts and isinstance(rec.get("payload"), dict):
280
+ ts = rec["payload"].get("timestamp") or rec["payload"].get("started_at")
281
+ return str(ts) if ts else ""
282
+
283
+
284
+ def _rec_role(rec):
285
+ payload = rec.get("payload")
286
+ if isinstance(payload, dict):
287
+ ptype = payload.get("type")
288
+ if ptype == "message":
289
+ role = str(payload.get("role") or "")
290
+ elif ptype in ("function_call", "function_call_output",
291
+ "local_shell_call", "shell_call", "web_search_call"):
292
+ role = "tool"
293
+ else:
294
+ role = ""
295
+ return _norm_role(role) if role else ""
296
+ msg = rec.get("message")
297
+ if isinstance(msg, dict) and msg.get("role"):
298
+ role = str(msg["role"])
299
+ else:
300
+ role = rec.get("role")
301
+ role = str(role) if role else ""
302
+ if role in ROLE_TOOL:
303
+ return "tool"
304
+ return role
305
+
306
+
307
+ def _rec_content(rec):
308
+ payload = rec.get("payload")
309
+ if isinstance(payload, dict) and payload.get("content") is not None:
310
+ return payload["content"]
311
+ msg = rec.get("message")
312
+ if isinstance(msg, dict):
313
+ return msg.get("content")
314
+ return rec.get("content")
315
+
316
+
317
+ def _rec_text(rec):
318
+ """提取记录里的人类可读文本(content 列表只取 type=text,字符串直接取)。"""
319
+ content = _rec_content(rec)
320
+ if isinstance(content, str):
321
+ return content
322
+ if isinstance(content, list):
323
+ parts = []
324
+ for item in content:
325
+ if not isinstance(item, dict):
356
326
  continue
357
- messages += 1
358
- ts = _rec_ts(rec)
359
- if not first_ts and ts:
360
- first_ts = ts
361
- total_messages += messages
362
- rows.append({
363
- "session": info["session"],
364
- "alias": _alias_for(idx, info["session"]),
365
- "path": info["path"],
366
- "size": info["size"],
367
- "date": first_ts[:10],
368
- "messages": messages,
369
- "invalid": invalid,
370
- "mtime": info["mtime"],
371
- })
372
- rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
327
+ if item.get("type") in ("text", "input_text", "output_text") \
328
+ and item.get("text"):
329
+ parts.append(str(item["text"]))
330
+ return "\n".join(parts)
331
+ return ""
332
+
333
+
334
+ def _rec_tool_names(rec):
335
+ """提取记录里的工具调用名(toolCall / toolResult / payload function_call)。"""
336
+ payload = rec.get("payload")
337
+ if isinstance(payload, dict):
338
+ ptype = payload.get("type")
339
+ if ptype in ("function_call", "function_call_output",
340
+ "local_shell_call", "shell_call"):
341
+ nm = payload.get("name") or payload.get("tool_name") or ""
342
+ return [str(nm)] if nm else []
343
+ return []
344
+ content = _rec_content(rec)
345
+ names = []
346
+ if isinstance(content, list):
347
+ for item in content:
348
+ if not isinstance(item, dict):
349
+ continue
350
+ if item.get("type") in ("tool_call", "toolCall", "toolResult"):
351
+ nm = item.get("name") or item.get("toolName") or ""
352
+ if nm:
353
+ names.append(str(nm))
354
+ return names
355
+
356
+
357
+ def _rec_cost(rec):
358
+ payload = rec.get("payload")
359
+ usage = None
360
+ if isinstance(payload, dict):
361
+ usage = payload.get("usage")
362
+ if not isinstance(usage, dict):
363
+ msg = rec.get("message")
364
+ if isinstance(msg, dict):
365
+ usage = msg.get("usage")
366
+ if not isinstance(usage, dict):
367
+ usage = rec.get("usage")
368
+ if not isinstance(usage, dict):
369
+ return 0.0
370
+ cost = usage.get("cost")
371
+ if isinstance(cost, dict):
372
+ return float(cost.get("total") or 0)
373
+ try:
374
+ return float(cost or 0)
375
+ except (TypeError, ValueError):
376
+ return 0.0
377
+
378
+
379
+ def _rec_tokens(rec):
380
+ payload = rec.get("payload")
381
+ usage = None
382
+ if isinstance(payload, dict):
383
+ usage = payload.get("usage")
384
+ if not isinstance(usage, dict):
385
+ msg = rec.get("message")
386
+ if isinstance(msg, dict):
387
+ usage = msg.get("usage")
388
+ if not isinstance(usage, dict):
389
+ usage = rec.get("usage")
390
+ if not isinstance(usage, dict):
391
+ return (0, 0)
392
+ return (int(usage.get("input_tokens") or 0),
393
+ int(usage.get("output_tokens") or 0))
394
+
395
+
396
+ def _is_message(rec):
397
+ """是否为可计入统计的消息记录(排除 session 元数据 / 空角色)。"""
398
+ role = _rec_role(rec)
399
+ if role in ("", "session"):
400
+ return False
401
+ return True
402
+
403
+
404
+ # ── 统一记录模型 + 字段别名归一 ──────────────────────────────────────────
405
+
406
+ def _first_alias(rec, aliases):
407
+ if not isinstance(rec, dict):
408
+ return None
409
+ for k in aliases:
410
+ if k in rec and rec[k] is not None and rec[k] != "":
411
+ return rec[k]
412
+ return None
413
+
414
+
415
+ def _unpack(value):
416
+ """JSON 字符串解包(一层):content / data 可能是 JSON 编码的字符串。"""
417
+ if isinstance(value, str):
418
+ s = value.strip()
419
+ if s[:1] in ("{", "[") and (s.endswith("}") or s.endswith("]")):
420
+ try:
421
+ return json.loads(s)
422
+ except Exception:
423
+ return value
424
+ return value
425
+
426
+
427
+ def _norm_time(value):
428
+ """时间戳归一为 ISO 字符串;秒 / 毫秒自动推断。"""
429
+ if value is None:
430
+ return ""
431
+ s = str(value).strip()
432
+ if not s:
433
+ return ""
434
+ if re.fullmatch(r"\d{9,13}(\.\d+)?", s):
435
+ try:
436
+ secs = float(s)
437
+ if secs > 1e12: # 毫秒
438
+ secs /= 1000.0
439
+ return _dt.datetime.fromtimestamp(
440
+ secs, tz=_dt.timezone.utc).isoformat(timespec="seconds")
441
+ except (ValueError, OSError, OverflowError):
442
+ return s
443
+ return s[:-1] + "+00:00" if s.endswith("Z") else s
444
+
445
+
446
+ def _norm_role(role):
447
+ if role is None:
448
+ return ""
449
+ r = str(role).strip()
450
+ low = r.lower().replace("_", "").replace("-", "")
451
+ if low in ("toolresult", "toolcall", "tool", "function",
452
+ "functioncall", "toolcallresult"):
453
+ return "tool"
454
+ return r
455
+
456
+
457
+ def _extract_text(rec):
458
+ """从各种形态的记录里提取人类可读文本(含 content 列表 / 字符串 / 嵌套 message)。"""
459
+ if not isinstance(rec, dict):
460
+ return ""
461
+ msg = rec.get("message")
462
+ content = None
463
+ if isinstance(msg, dict):
464
+ content = msg.get("content")
465
+ if content is None:
466
+ content = rec.get("content")
467
+ if content is None:
468
+ for k in TEXT_ALIASES:
469
+ if k in rec and isinstance(rec[k], (str, list, dict)):
470
+ content = rec[k]
471
+ break
472
+ content = _unpack(content)
473
+ if isinstance(content, str):
474
+ return content
475
+ if isinstance(content, list):
476
+ parts = []
477
+ for item in content:
478
+ if not isinstance(item, dict):
479
+ continue
480
+ if item.get("type") in ("text", "input_text", "output_text") \
481
+ and item.get("text"):
482
+ parts.append(str(item["text"]))
483
+ elif item.get("type") == "text" and item.get("content"):
484
+ parts.append(str(item["content"]))
485
+ return "\n".join(parts)
486
+ if isinstance(content, dict):
487
+ return str(content.get("text") or content.get("content") or "")
488
+ return ""
489
+
490
+
491
+ def _norm_record(rec, source, fmt, kind, session_default, path, line=0,
492
+ extra_meta=None):
493
+ """把一个原始记录(dict)归一为统一 Record。"""
494
+ msg = rec.get("message") if isinstance(rec, dict) else None
495
+ payload = rec.get("payload") if isinstance(rec, dict) else None
496
+ if isinstance(msg, dict) or isinstance(payload, dict):
497
+ role = _rec_role(rec)
498
+ ts = _rec_ts(rec) or _first_alias(rec, TIME_ALIASES)
499
+ text = _rec_text(rec)
500
+ else:
501
+ role = _first_alias(rec, ROLE_ALIASES)
502
+ role = _norm_role(role) if role is not None else ""
503
+ ts = _first_alias(rec, TIME_ALIASES)
504
+ text = _extract_text(rec)
505
+ session_id = _first_alias(rec, SESSION_ALIASES) or session_default
506
+ title = _first_alias(rec, TITLE_ALIASES)
507
+ meta = dict(extra_meta or {})
508
+ meta["line"] = line
509
+ if title:
510
+ meta["title"] = str(title)
511
+ tools = _rec_tool_names(rec)
512
+ if tools:
513
+ meta["tools"] = tools
514
+ cost = _rec_cost(rec)
515
+ ti, to = _rec_tokens(rec)
516
+ if cost:
517
+ meta["cost"] = cost
518
+ if ti or to:
519
+ meta["tokens_in"] = ti
520
+ meta["tokens_out"] = to
373
521
  return {
374
- "dir": str(dir_path),
375
- "rows": rows,
376
- "total_sessions": len(rows),
377
- "total_messages": total_messages,
378
- "total_invalid": total_invalid,
522
+ "source": source, "format": fmt, "kind": kind,
523
+ "session": str(session_id), "time": _norm_time(ts),
524
+ "role": role, "text": text, "path": str(path), "meta": meta,
379
525
  }
380
526
 
381
527
 
382
- def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
383
- role=None, limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT,
384
- no_redact=False):
385
- """跨会话关键词 / 正则检索,返回结构化命中列表。"""
528
+ def _mk_source(name, kind, fmt, path, default_on=True, extra=None):
529
+ return {"name": name, "kind": kind, "format": fmt, "path": str(path),
530
+ "default_on": default_on, "extra": extra or {}}
531
+
532
+
533
+ # ── Reader 层:格式族可插拔 ──────────────────────────────────────────────
534
+
535
+ class JSONLReader:
536
+ FORMAT = "jsonl"
537
+ KIND = "session"
538
+
539
+ @classmethod
540
+ def discover(cls, base=None):
541
+ base = base or Path.home()
542
+ patterns = [
543
+ base / ".clawdbot" / "agents" / "*" / "sessions",
544
+ base / ".codex" / "sessions",
545
+ base / ".claude" / "projects" / "*",
546
+ base / ".config" / "opencode" / "sessions",
547
+ base / ".gemini" / "sessions",
548
+ base / ".agents" / "sessions",
549
+ ]
550
+ if os.environ.get("CODEX_HOME"):
551
+ patterns.append(Path(os.environ["CODEX_HOME"]) / "sessions")
552
+ out = []
553
+ for pat in patterns:
554
+ for d in glob.glob(str(pat)):
555
+ dp = Path(d)
556
+ if not dp.is_dir():
557
+ continue
558
+ if cls._has_jsonl(dp):
559
+ out.append(_mk_source(cls._name_for(dp), "session",
560
+ "jsonl", dp))
561
+ return out
562
+
563
+ @staticmethod
564
+ def _has_jsonl(dp, max_depth=3):
565
+ """目录(含最多 3 层子目录)内是否存在 *.jsonl 会话文件。"""
566
+ root = str(dp)
567
+ for dirpath, dirnames, filenames in os.walk(root):
568
+ depth = dirpath[len(root):].count(os.sep)
569
+ if depth >= max_depth:
570
+ dirnames[:] = []
571
+ for f in filenames:
572
+ if f.lower().endswith(JSONL_SUFFIXES):
573
+ return True
574
+ return False
575
+
576
+ @staticmethod
577
+ def _name_for(dp):
578
+ s = str(dp).replace("\\", "/")
579
+ if ".clawdbot" in s:
580
+ m = re.search(r"agents/([^/]+)/sessions", s)
581
+ return "clawdbot-" + (m.group(1) if m else "sessions")
582
+ if ".codex" in s:
583
+ return "codex-sessions"
584
+ if ".claude" in s:
585
+ return "claude-projects"
586
+ if "opencode" in s:
587
+ return "opencode-sessions"
588
+ if ".gemini" in s:
589
+ return "gemini-sessions"
590
+ if ".agents" in s:
591
+ return "agents-sessions"
592
+ return "jsonl-sessions"
593
+
594
+ @staticmethod
595
+ def _walk_jsonl(d, max_depth=5):
596
+ """递归收集目录下(含子目录)的 *.jsonl 会话文件。"""
597
+ out = []
598
+ root = str(d)
599
+ for dirpath, dirnames, filenames in os.walk(root):
600
+ depth = dirpath[len(root):].count(os.sep)
601
+ if depth >= max_depth:
602
+ dirnames[:] = []
603
+ for f in sorted(filenames):
604
+ if f.lower().endswith(JSONL_SUFFIXES):
605
+ out.append(Path(dirpath) / f)
606
+ return out
607
+
608
+ def iter_sessions(self, source, limit=0):
609
+ d = Path(source["path"])
610
+ idx = load_index(d)
611
+ rows = []
612
+ total_messages = 0
613
+ total_invalid = 0
614
+ for path in self._walk_jsonl(d):
615
+ st = path.stat()
616
+ records, invalid = parse_jsonl(path)
617
+ total_invalid += invalid
618
+ first_ts = ""
619
+ messages = 0
620
+ for rec in records:
621
+ if not _is_message(rec):
622
+ continue
623
+ messages += 1
624
+ ts = _rec_ts(rec)
625
+ if not first_ts and ts:
626
+ first_ts = ts
627
+ total_messages += messages
628
+ rows.append({
629
+ "source": source["name"], "format": "jsonl", "kind": "session",
630
+ "session": path.stem, "alias": _alias_for(idx, path.stem),
631
+ "path": str(path), "size": st.st_size,
632
+ "date": first_ts[:10], "messages": messages,
633
+ "invalid": invalid,
634
+ "mtime": _dt.datetime.fromtimestamp(st.st_mtime)
635
+ .isoformat(timespec="seconds"),
636
+ })
637
+ rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
638
+ if limit > 0:
639
+ rows = rows[:limit]
640
+ return rows, total_messages, total_invalid
641
+
642
+ def iter_records(self, source):
643
+ d = Path(source["path"])
644
+ for path in self._walk_jsonl(d):
645
+ records, _ = parse_jsonl(path)
646
+ for lineno, rec in enumerate(records, 1):
647
+ if not _is_message(rec):
648
+ continue
649
+ yield _norm_record(rec, source["name"], "jsonl", "session",
650
+ path.stem, path, line=lineno)
651
+
652
+
653
+ class JSONReader:
654
+ FORMAT = "json"
655
+ KIND = "session"
656
+
657
+ @classmethod
658
+ def discover(cls, base=None):
659
+ base = base or Path.home()
660
+ out = []
661
+ for cand in (base / ".continue" / "sessions",
662
+ base / ".config" / "continue" / "sessions"):
663
+ if cand.is_dir() and any(
664
+ p.suffix.lower() in JSON_SUFFIXES for p in cand.iterdir()):
665
+ out.append(_mk_source("continue-sessions", "session", "json", cand))
666
+ return out
667
+
668
+ def _files(self, source):
669
+ p = Path(source["path"])
670
+ if p.is_file():
671
+ return [p]
672
+ return sorted(x for x in p.iterdir()
673
+ if x.is_file() and x.suffix.lower() in JSON_SUFFIXES)
674
+
675
+ def _iter_file_records(self, path, source):
676
+ try:
677
+ data = json.loads(path.read_text(encoding="utf-8", errors="replace"))
678
+ except Exception:
679
+ return
680
+ if isinstance(data, list):
681
+ for i, item in enumerate(data, 1):
682
+ if isinstance(item, dict):
683
+ yield _norm_record(item, source["name"], "json", "session",
684
+ path.stem, path, line=i)
685
+ elif isinstance(data, dict):
686
+ for sid, val in data.items():
687
+ if isinstance(val, list):
688
+ for i, item in enumerate(val, 1):
689
+ if isinstance(item, dict):
690
+ yield _norm_record(item, source["name"], "json",
691
+ "session", sid, path, line=i)
692
+ elif isinstance(val, dict):
693
+ yield _norm_record(val, source["name"], "json", "session",
694
+ sid, path, line=1)
695
+
696
+ def iter_records(self, source):
697
+ for p in self._files(source):
698
+ for rec in self._iter_file_records(p, source):
699
+ yield rec
700
+
701
+ def iter_sessions(self, source, limit=0):
702
+ rows = []
703
+ total_messages = 0
704
+ for p in self._files(source):
705
+ recs = list(self._iter_file_records(p, source))
706
+ grouped = {}
707
+ for r in recs:
708
+ grouped.setdefault(r["session"], []).append(r)
709
+ for sid, rl in grouped.items():
710
+ first_ts = next((r["time"] for r in rl if r["time"]), "")
711
+ total_messages += len(rl)
712
+ rows.append({
713
+ "source": source["name"], "format": "json", "kind": "session",
714
+ "session": sid, "alias": "", "path": str(p),
715
+ "size": p.stat().st_size, "date": first_ts[:10],
716
+ "messages": len(rl), "invalid": 0,
717
+ "mtime": _dt.datetime.fromtimestamp(p.stat().st_mtime)
718
+ .isoformat(timespec="seconds"),
719
+ })
720
+ rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
721
+ if limit > 0:
722
+ rows = rows[:limit]
723
+ return rows, total_messages, 0
724
+
725
+
726
+ class SQLiteReader:
727
+ FORMAT = "sqlite"
728
+ KIND = "session"
729
+
730
+ @classmethod
731
+ def discover(cls, base=None):
732
+ base = base or Path.home()
733
+ out = []
734
+ cands = []
735
+ xdg = os.environ.get("XDG_DATA_HOME")
736
+ if xdg:
737
+ cands.append(("opencode-db",
738
+ Path(xdg) / "opencode" / "opencode.db"))
739
+ env = os.environ.get("OPENCODE_DATA")
740
+ if env:
741
+ cands.append(("opencode-db", Path(env) / "data" / "opencode" / "opencode.db"))
742
+ cands.append(("opencode-db", Path(env) / "opencode.db"))
743
+ cands += [
744
+ ("opencode-db", base / ".local" / "share" / "opencode" / "opencode.db"),
745
+ ("opencode-db", base / ".config" / "opencode" / "opencode.db"),
746
+ ("opencode-db", base / ".OpenCodeData" / "data" / "opencode" / "opencode.db"),
747
+ ]
748
+ seen = set()
749
+ for name, p in cands:
750
+ key = str(p)
751
+ if key in seen:
752
+ continue
753
+ seen.add(key)
754
+ if p.exists():
755
+ out.append(_mk_source(name, "session", "sqlite", p))
756
+ # VS Code / Cursor state.vscdb(Windows / Linux / macOS)
757
+ for app in ("Code", "Cursor"):
758
+ for root in (base / "AppData" / "Roaming" / app / "User" / "globalStorage",
759
+ base / ".config" / app / "User" / "globalStorage",
760
+ base / "Library" / "Application Support" / app / "User" / "globalStorage"):
761
+ for d in glob.glob(str(root / "*")):
762
+ p = Path(d) / "state.vscdb"
763
+ if p.exists():
764
+ out.append(_mk_source(app.lower() + "-state", "session",
765
+ "sqlite", p))
766
+ return out
767
+
768
+ @staticmethod
769
+ def _connect(path):
770
+ return sqlite3.connect("file:%s?mode=ro" % str(path).replace("\\", "/"),
771
+ uri=True)
772
+
773
+ def _is_opencode(self, con):
774
+ tabs = {r[0] for r in con.execute(
775
+ "SELECT name FROM sqlite_master WHERE type='table'")}
776
+ return {"session", "message", "part"}.issubset(tabs)
777
+
778
+ def _pick_generic(self, con, extra):
779
+ if extra.get("table"):
780
+ return extra["table"]
781
+ tabs = [r[0] for r in con.execute(
782
+ "SELECT name FROM sqlite_master WHERE type='table' ORDER BY name")]
783
+ best = None
784
+ for t in tabs:
785
+ if t.startswith("sqlite_"):
786
+ continue
787
+ cols = [c[1].lower() for c in con.execute("PRAGMA table_info(%s)" % t)]
788
+ if any(c in ("text", "content", "body", "message", "statement")
789
+ for c in cols):
790
+ if "id" in cols or "session" in cols or "role" in cols:
791
+ return t
792
+ if best is None:
793
+ best = t
794
+ return best
795
+
796
+ def _generic_cols(self, con, table, extra):
797
+ real = [c[1] for c in con.execute("PRAGMA table_info(%s)" % table)]
798
+ low = [c.lower() for c in real]
799
+
800
+ def pick(aliases, explicit):
801
+ if explicit and explicit in real:
802
+ return explicit
803
+ if explicit and explicit.lower() in low:
804
+ return real[low.index(explicit.lower())]
805
+ for a in aliases:
806
+ if a in real:
807
+ return a
808
+ if a.lower() in low:
809
+ return real[low.index(a.lower())]
810
+ return None
811
+
812
+ return (pick(TIME_ALIASES, extra.get("col_time")),
813
+ pick(ROLE_ALIASES, extra.get("col_role")),
814
+ pick(TEXT_ALIASES, extra.get("col_text")),
815
+ pick(SESSION_ALIASES, extra.get("col_session")),
816
+ pick(TITLE_ALIASES, extra.get("col_title")))
817
+
818
+ def _opencode_records(self, con, source):
819
+ sess_title = dict(con.execute("SELECT id, title FROM session").fetchall())
820
+ parts = {}
821
+ for mid, data in con.execute("SELECT message_id, data FROM part"):
822
+ parts.setdefault(mid, []).append(data)
823
+ for mid, sid, t_created, mdata in con.execute(
824
+ "SELECT id, session_id, time_created, data FROM message"):
825
+ try:
826
+ m = json.loads(mdata) if isinstance(mdata, str) else (mdata or {})
827
+ role = (m or {}).get("role", "")
828
+ except Exception:
829
+ role = ""
830
+ texts = []
831
+ tools = []
832
+ for pdata in parts.get(mid, []):
833
+ try:
834
+ p = json.loads(pdata) if isinstance(pdata, str) else (pdata or {})
835
+ except Exception:
836
+ continue
837
+ ptype = p.get("type", "")
838
+ if ptype == "text" and p.get("text"):
839
+ texts.append(p["text"])
840
+ elif ptype == "tool" and p.get("tool"):
841
+ tools.append(p["tool"])
842
+ if not texts and not tools:
843
+ continue
844
+ meta = {"tools": tools}
845
+ title = sess_title.get(sid)
846
+ if title:
847
+ meta["title"] = title
848
+ yield {
849
+ "source": source["name"], "format": "sqlite", "kind": "session",
850
+ "session": sid, "time": _norm_time(t_created),
851
+ "role": _norm_role(role), "text": "\n".join(texts),
852
+ "path": str(source["path"]), "meta": meta,
853
+ }
854
+
855
+ def iter_records(self, source):
856
+ con = self._connect(source["path"])
857
+ try:
858
+ if self._is_opencode(con):
859
+ for r in self._opencode_records(con, source):
860
+ yield r
861
+ return
862
+ extra = source.get("extra") or {}
863
+ table = self._pick_generic(con, extra)
864
+ if not table:
865
+ return
866
+ col_time, col_role, col_text, col_session, col_title = \
867
+ self._generic_cols(con, table, extra)
868
+ if not col_text:
869
+ return
870
+ cols = [d[0] for d in con.execute("SELECT * FROM %s LIMIT 0" % table)
871
+ .description]
872
+ for i, row in enumerate(con.execute("SELECT * FROM %s" % table), 1):
873
+ rec = dict(zip(cols, row))
874
+ text = rec.get(col_text)
875
+ if isinstance(text, (dict, list)):
876
+ text = json.dumps(text, ensure_ascii=False)
877
+ elif isinstance(text, str):
878
+ text = _unpack(text)
879
+ text = str(text) if text is not None else ""
880
+ sid = rec.get(col_session) if col_session else \
881
+ Path(source["path"]).stem
882
+ meta = {}
883
+ if col_title and rec.get(col_title) is not None:
884
+ meta["title"] = str(rec[col_title])
885
+ yield {
886
+ "source": source["name"], "format": "sqlite",
887
+ "kind": source["kind"],
888
+ "session": str(sid),
889
+ "time": _norm_time(rec.get(col_time) if col_time else None),
890
+ "role": _norm_role(rec.get(col_role) if col_role else ""),
891
+ "text": text, "path": str(source["path"]), "meta": meta,
892
+ }
893
+ finally:
894
+ con.close()
895
+
896
+ def iter_sessions(self, source, limit=0):
897
+ con = self._connect(source["path"])
898
+ try:
899
+ rows = []
900
+ if self._is_opencode(con):
901
+ cols = {c[1] for c in con.execute("PRAGMA table_info(session)")}
902
+ use_metrics = {"cost", "tokens_input", "tokens_output"}.issubset(cols)
903
+ if use_metrics:
904
+ q = ("SELECT id, title, time_created, cost, tokens_input, "
905
+ "tokens_output, (SELECT COUNT(*) FROM message m WHERE "
906
+ "m.session_id = s.id) FROM session s "
907
+ "ORDER BY time_created DESC")
908
+ else:
909
+ q = ("SELECT id, title, time_created, 0, 0, 0, "
910
+ "(SELECT COUNT(*) FROM message m WHERE "
911
+ "m.session_id = s.id) FROM session s "
912
+ "ORDER BY time_created DESC")
913
+ for sid, title, t_created, cost, ti, to, cnt in con.execute(q):
914
+ rows.append({
915
+ "source": source["name"], "format": "sqlite",
916
+ "kind": "session", "session": sid, "alias": "",
917
+ "path": str(source["path"]),
918
+ "size": os.path.getsize(source["path"]),
919
+ "date": _norm_time(t_created)[:10],
920
+ "messages": cnt, "invalid": 0,
921
+ "mtime": _norm_time(t_created)[:19].replace("T", " "),
922
+ "title": title or "",
923
+ })
924
+ return rows, sum(r["messages"] for r in rows), 0
925
+ extra = source.get("extra") or {}
926
+ table = self._pick_generic(con, extra)
927
+ if not table:
928
+ return [], 0, 0
929
+ col_time, col_role, col_text, col_session, col_title = \
930
+ self._generic_cols(con, table, extra)
931
+ if not col_session:
932
+ cnt = con.execute("SELECT COUNT(*) FROM %s" % table).fetchone()[0]
933
+ rows.append({
934
+ "source": source["name"], "format": "sqlite",
935
+ "kind": source["kind"], "session": Path(source["path"]).stem,
936
+ "alias": "", "path": str(source["path"]),
937
+ "size": os.path.getsize(source["path"]),
938
+ "date": "", "messages": cnt, "invalid": 0,
939
+ "mtime": _dt.datetime.fromtimestamp(
940
+ os.path.getmtime(source["path"])).isoformat(timespec="seconds"),
941
+ "title": "",
942
+ })
943
+ return rows, cnt, 0
944
+ q = ("SELECT %s, COUNT(*) FROM %s GROUP BY %s"
945
+ % (col_session, table, col_session))
946
+ for sid, cnt in con.execute(q):
947
+ rows.append({
948
+ "source": source["name"], "format": "sqlite",
949
+ "kind": source["kind"], "session": str(sid), "alias": "",
950
+ "path": str(source["path"]),
951
+ "size": os.path.getsize(source["path"]),
952
+ "date": "", "messages": cnt, "invalid": 0,
953
+ "mtime": _dt.datetime.fromtimestamp(
954
+ os.path.getmtime(source["path"])).isoformat(timespec="seconds"),
955
+ "title": "",
956
+ })
957
+ rows.sort(key=lambda r: r["session"])
958
+ if limit > 0:
959
+ rows = rows[:limit]
960
+ return rows, sum(r["messages"] for r in rows), 0
961
+ finally:
962
+ con.close()
963
+
964
+
965
+ class MarkdownReader:
966
+ FORMAT = "markdown"
967
+ KIND = "memory"
968
+
969
+ @classmethod
970
+ def discover(cls, base=None):
971
+ base = base or Path.home()
972
+ cwd = Path.cwd()
973
+ out = []
974
+ mem = cls._memory_home(base)
975
+ for name, sub in (("yottamemory-facts", "facts"),
976
+ ("yottamemory-private", "private"),
977
+ ("yottamemory-archive", "archive")):
978
+ p = mem / sub
979
+ if p.is_dir() and any(x.suffix.lower() in MD_SUFFIXES
980
+ for x in p.iterdir()):
981
+ out.append(_mk_source(name, "memory", "markdown", p))
982
+ codex_home = os.environ.get("CODEX_HOME")
983
+ codex_notes = Path(codex_home) / "memories" if codex_home \
984
+ else base / ".CodexData" / "memories"
985
+ if codex_notes.is_dir() and cls._has_md(codex_notes):
986
+ out.append(_mk_source("codex-notes", "note", "markdown", codex_notes,
987
+ default_on=False))
988
+ # Aider 会话历史(当前目录浅扫)
989
+ for p in sorted(cwd.glob("*.aider.*.md")) + \
990
+ sorted(cwd.glob("*.aider.*.markdown")):
991
+ out.append(_mk_source("aider-history", "session", "markdown", p))
992
+ return out
993
+
994
+ @staticmethod
995
+ def _memory_home(base):
996
+ """yotta-memory 记忆库位置:优先读引擎 config.json 的 memory_home。"""
997
+ try:
998
+ cfg_p = base / ".yottamemory" / "config.json"
999
+ cfg = json.loads(cfg_p.read_text(encoding="utf-8", errors="replace"))
1000
+ mh = cfg.get("memory_home")
1001
+ if mh:
1002
+ return Path(mh)
1003
+ except Exception:
1004
+ pass
1005
+ return base / ".yottamemory"
1006
+
1007
+ @staticmethod
1008
+ def _has_md(p, depth=3):
1009
+ for x in p.rglob("*.md"):
1010
+ rel = x.relative_to(p)
1011
+ if len(rel.parts) <= depth:
1012
+ return True
1013
+ return False
1014
+
1015
+ def _files(self, source):
1016
+ p = Path(source["path"])
1017
+ if p.is_file():
1018
+ return [p]
1019
+ if not p.is_dir():
1020
+ return []
1021
+ out = []
1022
+ for x in sorted(p.rglob("*.md")):
1023
+ rel = x.relative_to(p)
1024
+ if len(rel.parts) <= 4:
1025
+ out.append(x)
1026
+ return out
1027
+
1028
+ @staticmethod
1029
+ def _split_frontmatter(text):
1030
+ """YAML frontmatter 子集解析:--- 块 → (dict, 正文)。零依赖。"""
1031
+ if not text.startswith("---"):
1032
+ return {}, text
1033
+ lines = text.split("\n")
1034
+ end = None
1035
+ for i in range(1, min(len(lines), 300)):
1036
+ if lines[i].strip() == "---":
1037
+ end = i
1038
+ break
1039
+ if end is None:
1040
+ return {}, text
1041
+ fm = {}
1042
+ for line in lines[1:end]:
1043
+ s = line.strip()
1044
+ if not s or s.startswith("#") or ":" not in s:
1045
+ continue
1046
+ k, _, v = s.partition(":")
1047
+ k = k.strip().lower()
1048
+ v = v.strip()
1049
+ if not v:
1050
+ fm[k] = None
1051
+ continue
1052
+ if v.startswith("[") and v.endswith("]"):
1053
+ inner = v[1:-1].strip()
1054
+ fm[k] = [x.strip().strip("\"'")
1055
+ for x in inner.split(",") if x.strip()]
1056
+ elif v[:1] in ("\"", "'") and v[-1:] == v[:1]:
1057
+ fm[k] = v[1:-1]
1058
+ else:
1059
+ fm[k] = v
1060
+ body = "\n".join(lines[end + 1:])
1061
+ return fm, body
1062
+
1063
+ def _file_record(self, path, source):
1064
+ try:
1065
+ text = path.read_text(encoding="utf-8", errors="replace")
1066
+ except Exception:
1067
+ return None
1068
+ fm, body = self._split_frontmatter(text)
1069
+ st = path.stat()
1070
+ mtime = _dt.datetime.fromtimestamp(st.st_mtime) \
1071
+ .isoformat(timespec="seconds")
1072
+ kind = source["kind"]
1073
+ role = ""
1074
+ if fm:
1075
+ role = _norm_role(fm.get("type") or "")
1076
+ if not role and kind == "memory":
1077
+ role = "memory"
1078
+ title = fm.get("subject") or fm.get("title") or fm.get("name")
1079
+ content = fm.get("statement") or fm.get("content") or fm.get("text")
1080
+ if content is None:
1081
+ content = body.strip()
1082
+ else:
1083
+ content = str(content)
1084
+ if not title:
1085
+ m = re.search(r"^#\s+(.+)$", body, re.M)
1086
+ title = m.group(1).strip() if m else ""
1087
+ ts = fm.get("created") or fm.get("date") or fm.get("updated") or mtime
1088
+ meta = {}
1089
+ if title:
1090
+ meta["title"] = title
1091
+ for k in ("tags", "confidence", "scope", "owner", "immutable"):
1092
+ if fm.get(k) is not None:
1093
+ meta[k] = fm[k]
1094
+ return {
1095
+ "source": source["name"], "format": "markdown", "kind": kind,
1096
+ "session": path.stem, "time": _norm_time(ts),
1097
+ "role": role, "text": content, "path": str(path), "meta": meta,
1098
+ }
1099
+
1100
+ def iter_records(self, source):
1101
+ for p in self._files(source):
1102
+ r = self._file_record(p, source)
1103
+ if r and (r["text"] or r["meta"].get("title")):
1104
+ yield r
1105
+
1106
+ def iter_sessions(self, source, limit=0):
1107
+ rows = []
1108
+ total = 0
1109
+ for p in self._files(source):
1110
+ r = self._file_record(p, source)
1111
+ if not r:
1112
+ continue
1113
+ total += 1
1114
+ rows.append({
1115
+ "source": source["name"], "format": "markdown",
1116
+ "kind": r["kind"], "session": p.stem, "alias": "",
1117
+ "path": str(p), "size": p.stat().st_size,
1118
+ "date": r["time"][:10], "messages": 1, "invalid": 0,
1119
+ "mtime": _dt.datetime.fromtimestamp(p.stat().st_mtime)
1120
+ .isoformat(timespec="seconds"),
1121
+ "title": r["meta"].get("title", ""),
1122
+ })
1123
+ rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
1124
+ if limit > 0:
1125
+ rows = rows[:limit]
1126
+ return rows, total, 0
1127
+
1128
+
1129
+ class BinaryReader:
1130
+ FORMAT = "binary"
1131
+ KIND = "log"
1132
+
1133
+ @classmethod
1134
+ def discover(cls, base=None):
1135
+ base = base or Path.home()
1136
+ out = []
1137
+ for root in (base / ".codeium" / "windsurf", base / ".windsurf"):
1138
+ if root.is_dir():
1139
+ for p in list(root.glob("**/*.pbtxt"))[:200]:
1140
+ out.append(_mk_source("windsurf-conv", "log", "binary",
1141
+ p, default_on=False))
1142
+ return out
1143
+
1144
+ def iter_records(self, source):
1145
+ p = Path(source["path"])
1146
+ title = p.stem
1147
+ try:
1148
+ raw = p.read_bytes()[:512]
1149
+ s = raw.decode("utf-8", errors="replace")
1150
+ m = re.search(r"[A-Za-z0-9\u4e00-\u9fff][^\x00-\x1f]{2,80}", s)
1151
+ if m:
1152
+ title = m.group(0).strip()
1153
+ except Exception:
1154
+ pass
1155
+ st = p.stat()
1156
+ yield {
1157
+ "source": source["name"], "format": "binary", "kind": "log",
1158
+ "session": p.stem,
1159
+ "time": _dt.datetime.fromtimestamp(st.st_mtime)
1160
+ .isoformat(timespec="seconds"),
1161
+ "role": "", "text": title, "path": str(p),
1162
+ "meta": {"title": title},
1163
+ }
1164
+
1165
+ def iter_sessions(self, source, limit=0):
1166
+ rows = []
1167
+ for rec in self.iter_records(source):
1168
+ p = Path(source["path"])
1169
+ st = p.stat()
1170
+ rows.append({
1171
+ "source": source["name"], "format": "binary", "kind": "log",
1172
+ "session": rec["session"], "alias": "", "path": str(p),
1173
+ "size": st.st_size, "date": rec["time"][:10],
1174
+ "messages": 1, "invalid": 0,
1175
+ "mtime": _dt.datetime.fromtimestamp(st.st_mtime)
1176
+ .isoformat(timespec="seconds"),
1177
+ "title": rec["meta"].get("title", ""),
1178
+ })
1179
+ return rows, len(rows), 0
1180
+
1181
+
1182
+ READERS = (JSONLReader, JSONReader, SQLiteReader, MarkdownReader, BinaryReader)
1183
+
1184
+
1185
+ def reader_for(fmt):
1186
+ for cls in READERS:
1187
+ if cls.FORMAT == fmt:
1188
+ return cls()
1189
+ raise SystemExit("不支持的格式:%s" % fmt)
1190
+
1191
+
1192
+ # ── 嗅探(--dir 指向目录 / 文件时自动判定格式族)────────────────────────
1193
+
1194
+ def _sniff_file_format(p):
1195
+ name = p.name.lower()
1196
+ for suf in JSONL_SUFFIXES:
1197
+ if name.endswith(suf):
1198
+ return "jsonl"
1199
+ if name.endswith(JSON_SUFFIXES):
1200
+ return "json"
1201
+ if name.endswith(SQLITE_SUFFIXES):
1202
+ return "sqlite"
1203
+ if name.endswith(MD_SUFFIXES):
1204
+ return "markdown"
1205
+ if name.endswith(BINARY_SUFFIXES):
1206
+ return "binary"
1207
+ try:
1208
+ head = p.read_bytes()[:512]
1209
+ except Exception:
1210
+ return "binary"
1211
+ if head.startswith(b"SQLite format 3"):
1212
+ return "sqlite"
1213
+ try:
1214
+ s = head.decode("utf-8", errors="replace").lstrip()
1215
+ except Exception:
1216
+ return "binary"
1217
+ if s.startswith("---"):
1218
+ return "markdown"
1219
+ if s[:1] in ("{", "["):
1220
+ first_line = s.split("\n", 1)[0].strip()
1221
+ if first_line.startswith("["):
1222
+ return "json"
1223
+ try:
1224
+ obj = json.loads(first_line)
1225
+ return "jsonl" if isinstance(obj, dict) else "json"
1226
+ except Exception:
1227
+ return "json"
1228
+ if re.search(r"^#\s+\S", s, re.M):
1229
+ return "markdown"
1230
+ return "binary"
1231
+
1232
+
1233
+ def _sniff_dir_format(p):
1234
+ names = [f.name.lower() for f in p.iterdir() if f.is_file()]
1235
+ if any(n.endswith(JSONL_SUFFIXES) for n in names):
1236
+ return "jsonl"
1237
+ if any(n.endswith(SQLITE_SUFFIXES) for n in names):
1238
+ return "sqlite"
1239
+ if any(n.endswith(JSON_SUFFIXES) for n in names):
1240
+ return "json"
1241
+ if any(n.endswith(MD_SUFFIXES) for n in names):
1242
+ return "markdown"
1243
+ return None
1244
+
1245
+
1246
+ def _sniff_file_kind(p):
1247
+ fmt = _sniff_file_format(p)
1248
+ if fmt == "markdown":
1249
+ try:
1250
+ t = p.read_text(encoding="utf-8", errors="replace")
1251
+ return "memory" if t.startswith("---") else "note"
1252
+ except Exception:
1253
+ return "note"
1254
+ if fmt == "binary":
1255
+ return "log"
1256
+ return "session"
1257
+
1258
+
1259
+ def _sniff_dir_kind(p):
1260
+ low = p.name.lower()
1261
+ if any(k in low for k in ("fact", "private", "memory", "记忆")):
1262
+ return "memory"
1263
+ if any(k in low for k in ("note", "notes", "笔记", "memories")):
1264
+ return "note"
1265
+ return "session"
1266
+
1267
+
1268
+ def sniff_source(path):
1269
+ """把 --dir / YOTTA_LOGS_DIR 指向的路径嗅探为一个单一来源。"""
1270
+ p = Path(path)
1271
+ if not p.exists():
1272
+ raise SystemExit("路径不存在:%s" % p)
1273
+ if p.is_file():
1274
+ return _mk_source(p.stem or "source", _sniff_file_kind(p),
1275
+ _sniff_file_format(p), p)
1276
+ fmt = _sniff_dir_format(p)
1277
+ if fmt == "markdown":
1278
+ return _mk_source(p.name or "source", _sniff_dir_kind(p), fmt, p)
1279
+ if fmt:
1280
+ return _mk_source(p.name or "source", "session", fmt, p)
1281
+ raise SystemExit("无法识别 %s 的日志 / 记忆格式(支持 jsonl / json / "
1282
+ "sqlite / markdown)" % p)
1283
+
1284
+
1285
+ # ── 配置兜底 + discover 全源登记 ─────────────────────────────────────────
1286
+
1287
+ def load_config():
1288
+ p = os.environ.get("YOTTA_LOGS_CONFIG")
1289
+ if not p:
1290
+ p = str(Path.home() / ".config" / "yotta-logs" / "config.json")
1291
+ cfg_path = Path(p)
1292
+ if not cfg_path.exists():
1293
+ return {}
1294
+ try:
1295
+ data = json.loads(cfg_path.read_text(encoding="utf-8", errors="replace"))
1296
+ return data if isinstance(data, dict) else {}
1297
+ except Exception:
1298
+ return {}
1299
+
1300
+
1301
+ def default_scope():
1302
+ cfg = load_config()
1303
+ return list(cfg.get("default_scope") or ["session", "memory"])
1304
+
1305
+
1306
+ def _config_sources(cfg):
1307
+ out = []
1308
+ for s in (cfg.get("sources") or []):
1309
+ if not isinstance(s, dict) or not s.get("path"):
1310
+ continue
1311
+ extra = {k: v for k, v in s.items()
1312
+ if k in ("table", "col_time", "col_role", "col_text",
1313
+ "col_session", "col_title")}
1314
+ out.append(_mk_source(
1315
+ s.get("name") or Path(s["path"]).stem,
1316
+ s.get("kind") or "session",
1317
+ s.get("format") or "jsonl",
1318
+ s["path"],
1319
+ default_on=True,
1320
+ extra=extra,
1321
+ ))
1322
+ return out
1323
+
1324
+
1325
+ def discover_sources(config=None):
1326
+ cfg = config if config is not None else load_config()
1327
+ sources = []
1328
+ for cls in READERS:
1329
+ for s in cls.discover():
1330
+ sources.append(s)
1331
+ sources += _config_sources(cfg)
1332
+ seen = set()
1333
+ dedup = []
1334
+ for s in sources:
1335
+ key = (s["format"], str(s["path"]).lower())
1336
+ if key in seen:
1337
+ continue
1338
+ seen.add(key)
1339
+ dedup.append(s)
1340
+ scope = set(cfg.get("default_scope") or ["session", "memory"])
1341
+ for s in dedup:
1342
+ s["default_on"] = s["default_on"] or s["kind"] in scope
1343
+ return dedup
1344
+
1345
+
1346
+ def _candidate_ids(source, session_ids):
1347
+ """把会话 ID / 别名解析为候选会话 ID 集合(JSONL 源带 sessions.json 别名)。"""
1348
+ ids = set()
1349
+ for g in (session_ids if isinstance(session_ids, (list, tuple))
1350
+ else [session_ids]):
1351
+ ids.add(g)
1352
+ if source["format"] == "jsonl":
1353
+ idx = load_index(Path(source["path"]))
1354
+ ids |= _resolve_session_ids(idx, g)
1355
+ return ids
1356
+
1357
+
1358
+ def filter_sources(sources, args):
1359
+ explicit = False
1360
+ if getattr(args, "source", None):
1361
+ names = set(args.source)
1362
+ sources = [s for s in sources if s["name"] in names]
1363
+ explicit = True
1364
+ if getattr(args, "format", None):
1365
+ sources = [s for s in sources if s["format"] == args.format]
1366
+ explicit = True
1367
+ if getattr(args, "kind", None):
1368
+ sources = [s for s in sources if s["kind"] == args.kind]
1369
+ explicit = True
1370
+ if not explicit:
1371
+ sources = [s for s in sources if s["default_on"]]
1372
+ return sources
1373
+
1374
+
1375
+ def resolve_sources(args):
1376
+ if getattr(args, "dir", None):
1377
+ srcs = [sniff_source(args.dir)]
1378
+ else:
1379
+ env = os.environ.get("YOTTA_LOGS_DIR")
1380
+ if env:
1381
+ srcs = [sniff_source(env)]
1382
+ else:
1383
+ srcs = discover_sources()
1384
+ if not srcs:
1385
+ raise SystemExit(
1386
+ "未找到已知日志 / 记忆源:用 --dir 指定,或设 YOTTA_LOGS_DIR / "
1387
+ "YOTTA_LOGS_CONFIG;可用 locate 查看候选。")
1388
+ return filter_sources(srcs, args)
1389
+
1390
+
1391
+ # ── 统一检索 / 提取 / 统计 / 工具排行 ───────────────────────────────────
1392
+
1393
+ def scan_all(sources, limit=0):
1394
+ rows = []
1395
+ total_messages = 0
1396
+ total_invalid = 0
1397
+ for src in sources:
1398
+ reader = reader_for(src["format"])
1399
+ r, tm, inv = reader.iter_sessions(src)
1400
+ rows.extend(r)
1401
+ total_messages += tm
1402
+ total_invalid += inv
1403
+ if limit > 0:
1404
+ rows = rows[:limit]
1405
+ return {"rows": rows, "total_sessions": len(rows),
1406
+ "total_messages": total_messages, "total_invalid": total_invalid}
1407
+
1408
+
1409
+ def search_all(sources, query, regex=False, date=None, sessions=None, role=None,
1410
+ limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT, no_redact=False):
386
1411
  if regex:
387
1412
  try:
388
1413
  pat = re.compile(query, re.I)
389
1414
  except re.error as e:
390
1415
  raise SystemExit("正则无效:%s" % e)
391
- idx = load_index(dir_path)
392
- sid_filter = None
393
- if sessions:
394
- sid_filter = set()
395
- for g in sessions:
396
- sid_filter |= _resolve_session_ids(idx, g)
397
1416
  matches = []
398
- hit_sids = set()
1417
+ hit = set()
399
1418
  truncated = False
400
- for info in list_sessions(dir_path):
401
- sid = info["session"]
402
- if sid_filter is not None and sid not in sid_filter:
403
- continue
404
- records, _ = parse_jsonl(info["path"])
405
- for lineno, rec in enumerate(records, 1):
406
- if not _is_message(rec):
1419
+ for source in sources:
1420
+ reader = reader_for(source["format"])
1421
+ cands = _candidate_ids(source, sessions) if sessions else None
1422
+ for rec in reader.iter_records(source):
1423
+ text = rec["text"]
1424
+ if not text:
407
1425
  continue
408
- ts = _rec_ts(rec)
1426
+ if cands is not None and rec["session"] not in cands:
1427
+ continue
1428
+ ts = rec["time"]
409
1429
  if date and not _ts_on_date(ts, date):
410
1430
  continue
411
- rrole = _rec_role(rec)
1431
+ rrole = rec["role"]
412
1432
  if role and rrole != role:
413
1433
  continue
414
- text = _rec_text(rec)
415
- if not text:
416
- continue
417
1434
  if regex:
418
1435
  m = pat.search(text)
419
1436
  if not m:
@@ -431,164 +1448,213 @@ def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
431
1448
  snippet = redact(snippet)
432
1449
  matched = redact(matched)
433
1450
  matches.append({
434
- "session": sid,
435
- "timestamp": ts,
436
- "role": rrole,
437
- "line": lineno,
438
- "match": matched,
439
- "text": snippet,
1451
+ "source": source["name"], "format": source["format"],
1452
+ "kind": source["kind"], "session": rec["session"],
1453
+ "timestamp": ts, "role": rrole,
1454
+ "line": rec["meta"].get("line", 0),
1455
+ "match": matched, "text": snippet,
440
1456
  })
441
- hit_sids.add(sid)
1457
+ hit.add((source["name"], rec["session"]))
442
1458
  if len(matches) >= limit:
443
1459
  truncated = True
444
- return {
445
- "matches": matches,
446
- "sessions_hit": len(hit_sids),
447
- "truncated": truncated,
448
- }
449
- return {"matches": matches, "sessions_hit": len(hit_sids),
450
- "truncated": truncated}
451
-
452
-
453
- def extract_session(dir_path, session_id, role=None, with_tools=False,
454
- no_redact=False):
455
- """提取单个会话原文(时间线 + 角色 + 文本)。"""
456
- idx = load_index(dir_path)
457
- info = None
458
- for sid in _resolve_session_ids(idx, session_id):
459
- p = Path(dir_path) / (sid + ".jsonl")
460
- if p.exists() and p.is_file():
461
- info = {"session": sid, "path": str(p)}
1460
+ return {"matches": matches, "sessions_hit": len(hit),
1461
+ "truncated": truncated}
1462
+ return {"matches": matches, "sessions_hit": len(hit), "truncated": truncated}
1463
+
1464
+
1465
+ def extract_all(sources, session_id, role=None, with_tools=False,
1466
+ no_redact=False):
1467
+ """跨源提取单个会话;多个源同名会话取第一个(--source 可消歧)。"""
1468
+ source_match = None
1469
+ for source in sources:
1470
+ reader = reader_for(source["format"])
1471
+ cands = _candidate_ids(source, session_id)
1472
+ for rec in reader.iter_records(source):
1473
+ if rec["session"] in cands:
1474
+ source_match = source
1475
+ break
1476
+ if source_match:
462
1477
  break
463
- if info is None:
464
- p = Path(dir_path) / session_id
465
- if p.exists() and p.is_file():
466
- info = {"session": p.stem, "path": str(p)}
467
- if info is None:
1478
+ if source_match is None:
468
1479
  raise SystemExit("未找到会话:%s(可用 scan 列出会话 ID)" % session_id)
469
- records, invalid = parse_jsonl(info["path"])
1480
+ reader = reader_for(source_match["format"])
1481
+ cands = _candidate_ids(source_match, session_id)
1482
+ actual = None
470
1483
  messages = []
471
- for lineno, rec in enumerate(records, 1):
472
- if not _is_message(rec):
1484
+ total_records = 0
1485
+ for rec in reader.iter_records(source_match):
1486
+ if rec["session"] not in cands:
473
1487
  continue
474
- rrole = _rec_role(rec)
1488
+ if actual is None:
1489
+ actual = rec["session"]
1490
+ total_records += 1
1491
+ rrole = rec["role"]
475
1492
  if role and rrole != role:
476
1493
  continue
477
- text = _rec_text(rec)
478
- tools = _rec_tool_names(rec)
1494
+ text = rec["text"]
1495
+ tools = rec["meta"].get("tools") or []
479
1496
  if not text and not tools:
480
1497
  continue
481
1498
  if not no_redact:
482
1499
  text = redact(text)
483
1500
  messages.append({
484
- "line": lineno,
485
- "timestamp": _rec_ts(rec),
1501
+ "line": rec["meta"].get("line", 0),
1502
+ "timestamp": rec["time"],
486
1503
  "role": rrole,
487
1504
  "text": text,
488
1505
  "tools": tools,
489
1506
  })
490
1507
  return {
491
- "session": info["session"],
492
- "dir": str(dir_path),
1508
+ "session": actual,
1509
+ "dir": str(source_match["path"]),
1510
+ "source": source_match["name"],
1511
+ "format": source_match["format"],
1512
+ "kind": source_match["kind"],
493
1513
  "messages": messages,
494
- "invalid": invalid,
495
- "total_records": len(records),
1514
+ "invalid": 0,
1515
+ "total_records": total_records,
496
1516
  }
497
1517
 
498
1518
 
499
- def session_stats(dir_path, session_id=None, daily=False):
500
- """汇总统计:消息 / 角色 / token / 成本 / 时间范围 / 每日汇总。"""
501
- idx = load_index(dir_path)
502
- sessions = list_sessions(dir_path)
503
- if session_id:
504
- ids = set()
505
- for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
506
- ids |= _resolve_session_ids(idx, g)
507
- sessions = [s for s in sessions if s["session"] in ids]
1519
+ def stats_all(sources, session_id=None, daily=False):
508
1520
  agg = {
509
- "sessions": len(sessions),
510
- "messages": 0,
511
- "invalid": 0,
512
- "roles": {},
513
- "cost": 0.0,
514
- "tokens_in": 0,
515
- "tokens_out": 0,
516
- "first": "",
517
- "last": "",
518
- "days": {},
1521
+ "sessions": 0, "messages": 0, "invalid": 0,
1522
+ "roles": {}, "cost": 0.0, "tokens_in": 0, "tokens_out": 0,
1523
+ "first": "", "last": "", "days": {}, "by_source": {},
519
1524
  }
520
- for info in sessions:
521
- records, invalid = parse_jsonl(info["path"])
522
- agg["invalid"] += invalid
523
- for rec in records:
524
- if not _is_message(rec):
1525
+ for source in sources:
1526
+ reader = reader_for(source["format"])
1527
+ rows, tm, inv = reader.iter_sessions(source)
1528
+ if session_id:
1529
+ cands = _candidate_ids(source, session_id)
1530
+ rows = [r for r in rows if r["session"] in cands]
1531
+ agg["sessions"] += len(rows)
1532
+ agg["invalid"] += inv
1533
+ bs = agg["by_source"].setdefault(source["name"], {
1534
+ "format": source["format"], "kind": source["kind"],
1535
+ "sessions": 0, "messages": 0, "cost": 0.0,
1536
+ "tokens_in": 0, "tokens_out": 0, "first": "", "last": "",
1537
+ })
1538
+ bs["sessions"] += len(rows)
1539
+ cands = _candidate_ids(source, session_id) if session_id else None
1540
+ for rec in reader.iter_records(source):
1541
+ if cands is not None and rec["session"] not in cands:
525
1542
  continue
526
- agg["messages"] += 1
527
- role = _rec_role(rec)
1543
+ role = rec["role"]
528
1544
  agg["roles"][role] = agg["roles"].get(role, 0) + 1
529
- ts = _rec_ts(rec)
1545
+ bs["messages"] += 1
1546
+ ts = rec["time"]
1547
+ cost = rec["meta"].get("cost", 0.0)
1548
+ ti = rec["meta"].get("tokens_in", 0)
1549
+ to = rec["meta"].get("tokens_out", 0)
530
1550
  if ts:
531
1551
  day = ts[:10]
532
1552
  if day:
533
1553
  d = agg["days"].setdefault(day, {"messages": 0, "cost": 0.0})
534
1554
  d["messages"] += 1
535
- d["cost"] += _rec_cost(rec)
1555
+ d["cost"] += cost
536
1556
  if not agg["first"] or ts < agg["first"]:
537
1557
  agg["first"] = ts
538
1558
  if not agg["last"] or ts > agg["last"]:
539
1559
  agg["last"] = ts
540
- agg["cost"] += _rec_cost(rec)
541
- ti, to = _rec_tokens(rec)
1560
+ if not bs["first"] or ts < bs["first"]:
1561
+ bs["first"] = ts
1562
+ if not bs["last"] or ts > bs["last"]:
1563
+ bs["last"] = ts
1564
+ agg["cost"] += cost
542
1565
  agg["tokens_in"] += ti
543
1566
  agg["tokens_out"] += to
1567
+ bs["cost"] += cost
1568
+ bs["tokens_in"] += ti
1569
+ bs["tokens_out"] += to
1570
+ agg["messages"] += bs["messages"]
544
1571
  return agg
545
1572
 
546
1573
 
547
- def tool_breakdown(dir_path, session_id=None):
548
- """工具调用次数排行 → [(工具名, 次数)],按次数降序。"""
549
- idx = load_index(dir_path)
550
- sessions = list_sessions(dir_path)
551
- if session_id:
552
- ids = set()
553
- for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
554
- ids |= _resolve_session_ids(idx, g)
555
- sessions = [s for s in sessions if s["session"] in ids]
1574
+ def tools_all(sources, session_id=None):
556
1575
  counts = {}
557
- for info in sessions:
558
- records, _ = parse_jsonl(info["path"])
559
- for rec in records:
560
- for nm in _rec_tool_names(rec):
1576
+ for source in sources:
1577
+ reader = reader_for(source["format"])
1578
+ cands = _candidate_ids(source, session_id) if session_id else None
1579
+ for rec in reader.iter_records(source):
1580
+ if cands is not None and rec["session"] not in cands:
1581
+ continue
1582
+ for nm in rec["meta"].get("tools") or []:
561
1583
  counts[nm] = counts.get(nm, 0) + 1
562
1584
  return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
563
1585
 
564
1586
 
565
- def _snippet(text, span, radius):
566
- start, end = span
567
- lo = max(0, start - radius)
568
- hi = min(len(text), end + radius)
569
- pre = "…" if lo > 0 else ""
570
- post = "…" if hi < len(text) else ""
571
- return pre + text[lo:hi].replace("\n", " ") + post
1587
+ # ── 兼容保留:单目录(JSONL)旧函数签名 ─────────────────────────────────
1588
+
1589
+ def scan_sessions(dir_path):
1590
+ source = sniff_source(dir_path)
1591
+ res = scan_all([source])
1592
+ res["dir"] = str(dir_path)
1593
+ return res
1594
+
1595
+
1596
+ def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
1597
+ role=None, limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT,
1598
+ no_redact=False):
1599
+ source = sniff_source(dir_path)
1600
+ return search_all([source], query, regex=regex, date=date,
1601
+ sessions=sessions, role=role, limit=limit,
1602
+ context=context, no_redact=no_redact)
1603
+
1604
+
1605
+ def extract_session(dir_path, session_id, role=None, with_tools=False,
1606
+ no_redact=False):
1607
+ source = sniff_source(dir_path)
1608
+ return extract_all([source], session_id, role=role, with_tools=with_tools,
1609
+ no_redact=no_redact)
1610
+
1611
+
1612
+ def session_stats(dir_path, session_id=None, daily=False):
1613
+ source = sniff_source(dir_path)
1614
+ res = stats_all([source], session_id=session_id, daily=daily)
1615
+ res["dir"] = str(dir_path)
1616
+ return res
1617
+
1618
+
1619
+ def tool_breakdown(dir_path, session_id=None):
1620
+ source = sniff_source(dir_path)
1621
+ return tools_all([source], session_id=session_id)
572
1622
 
573
1623
 
574
1624
  # ── 文本格式化 ───────────────────────────────────────────────────────────
575
1625
 
1626
+ def fmt_locate(sources):
1627
+ lines = []
1628
+ scope = " + ".join(default_scope()) if default_scope() else "(无)"
1629
+ lines.append("发现的日志 / 记忆源 %d 个 | 默认范围:%s"
1630
+ % (len(sources), scope))
1631
+ lines.append("")
1632
+ lines.append("%-22s %-9s %-8s %-3s %s"
1633
+ % ("来源", "格式", "类型", "默认", "路径"))
1634
+ for s in sorted(sources, key=lambda x: (x["name"], x["path"])):
1635
+ lines.append("%-22s %-9s %-8s %-3s %s"
1636
+ % (s["name"][:22], s["format"], s["kind"],
1637
+ "开" if s["default_on"] else "关", s["path"]))
1638
+ return "\n".join(lines)
1639
+
1640
+
576
1641
  def fmt_scan(res):
577
1642
  lines = []
578
- lines.append("会话日志目录:%s" % res["dir"])
579
- lines.append("会话 %d 个 | 消息合计 %d | 无效行 %d"
580
- % (res["total_sessions"], res["total_messages"],
581
- res["total_invalid"]))
1643
+ lines.append("来源 %d 个 | 会话 %d 个 | 消息合计 %d | 无效行 %d"
1644
+ % (len(res.get("sources") or []), res["total_sessions"],
1645
+ res["total_messages"], res["total_invalid"]))
582
1646
  lines.append("")
583
- lines.append("%-24s %-12s %8s %10s %s" % ("会话 ID", "日期", "消息", "大小", "别名"))
1647
+ lines.append("%-14s %-8s %-8s %-24s %-10s %8s %10s %s"
1648
+ % ("来源", "格式", "类型", "会话 ID", "日期", "消息", "大小", "别名"))
584
1649
  for r in res["rows"]:
585
- lines.append("%-24s %-12s %8d %10s %s"
586
- % (r["session"][:24], r["date"] or "-",
587
- r["messages"], _human_size(r["size"]), r["alias"]))
1650
+ lines.append("%-14s %-8s %-8s %-24s %-10s %8d %10s %s"
1651
+ % (r["source"][:14], r["format"], r["kind"],
1652
+ r["session"][:24], r["date"] or "-", r["messages"],
1653
+ _human_size(r["size"]), r.get("alias", "")))
588
1654
  return "\n".join(lines)
589
1655
 
590
1656
 
591
- def fmt_search(res, query, dir_path, regex):
1657
+ def fmt_search(res, query, sources, regex):
592
1658
  lines = []
593
1659
  if regex:
594
1660
  desc = "正则:%s" % query
@@ -601,9 +1667,10 @@ def fmt_search(res, query, dir_path, regex):
601
1667
  lines.append("")
602
1668
  cur = None
603
1669
  for m in res["matches"]:
604
- if m["session"] != cur:
605
- cur = m["session"]
606
- lines.append("── 会话 %s ─────────────────────────" % cur)
1670
+ key = (m["source"], m["session"])
1671
+ if key != cur:
1672
+ cur = key
1673
+ lines.append("── %s / 会话 %s ─────────────────" % (m["source"], m["session"]))
607
1674
  lines.append("%s [%s] %s" % (_clock(m["timestamp"]), m["role"], m["text"]))
608
1675
  if not res["matches"]:
609
1676
  lines.append("未命中。")
@@ -616,8 +1683,9 @@ def fmt_session(res):
616
1683
  for m in res["messages"]:
617
1684
  counts[m["role"]] = counts.get(m["role"], 0) + 1
618
1685
  parts = " | ".join("%s %d" % (k, v) for k, v in sorted(counts.items()))
619
- lines.append("会话:%s | 消息 %d(%s)| 无效行 %d"
620
- % (res["session"], len(res["messages"]), parts, res["invalid"]))
1686
+ lines.append("会话:%s | 来源 %s(%s)| 消息 %d(%s)"
1687
+ % (res["session"], res.get("source", "-"),
1688
+ res.get("format", "-"), len(res["messages"]), parts))
621
1689
  lines.append("")
622
1690
  for m in res["messages"]:
623
1691
  lines.append("── %s ─────────────────────────" % _clock(m["timestamp"]))
@@ -632,7 +1700,10 @@ def fmt_session(res):
632
1700
  def fmt_stats(res, daily):
633
1701
  lines = []
634
1702
  lines.append("会话统计")
635
- lines.append("目录:%s" % res.get("dir", ""))
1703
+ srcs = " | ".join("%s %d" % (k, v["sessions"])
1704
+ for k, v in sorted(res.get("by_source", {}).items()))
1705
+ if srcs:
1706
+ lines.append("分源会话:%s" % srcs)
636
1707
  roles = " | ".join("%s %d" % (k, v)
637
1708
  for k, v in sorted(res["roles"].items()))
638
1709
  lines.append("会话 %d | 消息 %d(%s)| 无效行 %d"
@@ -665,6 +1736,15 @@ def fmt_tools(items):
665
1736
  return "\n".join(lines)
666
1737
 
667
1738
 
1739
+ def _snippet(text, span, radius):
1740
+ start, end = span
1741
+ lo = max(0, start - radius)
1742
+ hi = min(len(text), end + radius)
1743
+ pre = "…" if lo > 0 else ""
1744
+ post = "…" if hi < len(text) else ""
1745
+ return pre + text[lo:hi].replace("\n", " ") + post
1746
+
1747
+
668
1748
  # ── CLI ──────────────────────────────────────────────────────────────────
669
1749
 
670
1750
  class _Parser(argparse.ArgumentParser):
@@ -676,56 +1756,55 @@ class _Parser(argparse.ArgumentParser):
676
1756
 
677
1757
  def _add_dir(ap):
678
1758
  ap.add_argument("--dir", metavar="PATH",
679
- help="会话日志目录(缺省读 YOTTA_LOGS_DIR,再自动定位)")
1759
+ help="日志 / 记忆目录或文件(缺省读 YOTTA_LOGS_DIR,"
1760
+ "再 discover 全源登记)")
680
1761
 
681
1762
 
682
- def resolve_dir(args):
683
- if getattr(args, "dir", None):
684
- d = Path(args.dir)
685
- else:
686
- env = os.environ.get("YOTTA_LOGS_DIR")
687
- if env:
688
- d = Path(env)
689
- else:
690
- found = discover_dirs()
691
- if not found:
692
- raise SystemExit(
693
- "未找到会话日志目录:用 --dir 指定,或设 YOTTA_LOGS_DIR;"
694
- "可用 locate 查看候选。")
695
- d = Path(found[0])
696
- if not d.is_dir():
697
- raise SystemExit("目录不存在:%s" % d)
698
- return d
1763
+ def _add_filters(ap):
1764
+ ap.add_argument("--source", action="append", metavar="NAME",
1765
+ help="只检索指定来源(可多次;名称见 locate)")
1766
+ ap.add_argument("--kind", choices=KIND_CHOICES,
1767
+ help="只检索指定类型:session / memory / note / log")
1768
+ ap.add_argument("--format", choices=FORMAT_CHOICES,
1769
+ help="只检索指定格式:jsonl / json / sqlite / markdown / binary")
1770
+
1771
+
1772
+ def _source_brief(sources):
1773
+ return [{"name": s["name"], "kind": s["kind"], "format": s["format"],
1774
+ "path": s["path"], "default_on": s["default_on"]} for s in sources]
699
1775
 
700
1776
 
701
1777
  def main(argv=None):
702
1778
  ap = _Parser(
703
1779
  prog=TOOL_NAME,
704
- description="%s(%s):零依赖跨智能体会话日志检索引擎。" % (TOOL_CN, TOOL_NAME))
1780
+ description="%s(%s):零依赖跨智能体会话 / 记忆日志检索引擎。"
1781
+ % (TOOL_CN, TOOL_NAME))
705
1782
  ap.add_argument("--version", action="version",
706
1783
  version="%s %s" % (TOOL_NAME, VERSION))
707
1784
  sub = ap.add_subparsers(dest="command", required=True)
708
1785
 
709
1786
  sub.add_parser("version", help="打印版本")
710
1787
 
711
- p_locate = sub.add_parser("locate", help="自动发现本机会话日志目录")
1788
+ p_locate = sub.add_parser("locate", help="全源登记:发现本机日志 / 记忆源")
712
1789
  p_locate.add_argument("--json", action="store_true", help="输出 JSON")
713
1790
 
714
- p_scan = sub.add_parser("scan", help="列出目录下所有会话")
1791
+ p_scan = sub.add_parser("scan", help="列出所有会话(跨源)")
715
1792
  _add_dir(p_scan)
1793
+ _add_filters(p_scan)
716
1794
  p_scan.add_argument("--json", action="store_true", help="输出 JSON")
717
1795
  p_scan.add_argument("--limit", type=int, default=0,
718
1796
  help="最多列出 N 个会话(默认全部)")
719
1797
 
720
- p_search = sub.add_parser("search", help="跨会话检索关键词 / 正则")
1798
+ p_search = sub.add_parser("search", help="跨源检索关键词 / 正则")
721
1799
  p_search.add_argument("query", help="检索词(默认不区分大小写)")
722
1800
  _add_dir(p_search)
1801
+ _add_filters(p_search)
723
1802
  p_search.add_argument("--regex", action="store_true", help="把 query 当正则")
724
1803
  p_search.add_argument("--date", metavar="YYYY-MM-DD",
725
1804
  help="只检索指定日期(或 YYYY-MM)")
726
1805
  p_search.add_argument("-s", "--session", action="append", metavar="SID",
727
1806
  help="只检索指定会话 ID / 别名(可多次)")
728
- p_search.add_argument("--role", choices=("user", "assistant", "tool", "system"),
1807
+ p_search.add_argument("--role", choices=("user", "assistant", "tool", "system", "developer"),
729
1808
  help="只检索指定角色")
730
1809
  p_search.add_argument("--limit", type=int, default=DEFAULT_LIMIT,
731
1810
  help="最多返回 N 条命中(默认 %d)" % DEFAULT_LIMIT)
@@ -736,27 +1815,30 @@ def main(argv=None):
736
1815
  help="关闭默认脱敏")
737
1816
 
738
1817
  p_session = sub.add_parser("session", help="提取单个会话原文")
739
- p_session.add_argument("sid", help="会话 ID 或 sessions.json 里的别名")
1818
+ p_session.add_argument("sid", help="会话 ID / 别名")
740
1819
  _add_dir(p_session)
741
- p_session.add_argument("--role", choices=("user", "assistant", "tool", "system"),
1820
+ _add_filters(p_session)
1821
+ p_session.add_argument("--role", choices=("user", "assistant", "tool", "system", "developer"),
742
1822
  help="只提取指定角色")
743
1823
  p_session.add_argument("--tools", action="store_true",
744
- help="时间线里标注工具调用")
1824
+ help="标注工具调用")
745
1825
  p_session.add_argument("--limit", type=int, default=0,
746
- help="最多提取 N 条消息(默认全部)")
1826
+ help="最多输出 N 条消息(默认全部)")
747
1827
  p_session.add_argument("--json", action="store_true", help="输出 JSON")
748
1828
  p_session.add_argument("--no-redact", action="store_true",
749
1829
  help="关闭默认脱敏")
750
1830
 
751
- p_stats = sub.add_parser("stats", help="会话统计汇总")
1831
+ p_stats = sub.add_parser("stats", help="会话统计汇总(跨源)")
752
1832
  _add_dir(p_stats)
1833
+ _add_filters(p_stats)
753
1834
  p_stats.add_argument("-s", "--session", metavar="SID",
754
1835
  help="只统计指定会话 ID / 别名")
755
1836
  p_stats.add_argument("--daily", action="store_true", help="输出每日汇总")
756
1837
  p_stats.add_argument("--json", action="store_true", help="输出 JSON")
757
1838
 
758
- p_tools = sub.add_parser("tools", help="工具调用次数排行")
1839
+ p_tools = sub.add_parser("tools", help="工具调用次数排行(跨源)")
759
1840
  _add_dir(p_tools)
1841
+ _add_filters(p_tools)
760
1842
  p_tools.add_argument("-s", "--session", metavar="SID",
761
1843
  help="只统计指定会话 ID / 别名")
762
1844
  p_tools.add_argument("--json", action="store_true", help="输出 JSON")
@@ -768,23 +1850,24 @@ def main(argv=None):
768
1850
  return 0
769
1851
 
770
1852
  if args.command == "locate":
771
- found = discover_dirs()
1853
+ sources = discover_sources()
772
1854
  if args.json:
773
- print(json.dumps({"dirs": found}, ensure_ascii=False, indent=2))
774
- elif found:
775
- for d in found:
776
- print(d)
1855
+ print(json.dumps({
1856
+ "tool": TOOL_NAME, "version": VERSION,
1857
+ "default_scope": default_scope(),
1858
+ "sources": _source_brief(sources),
1859
+ }, ensure_ascii=False, indent=2))
1860
+ elif sources:
1861
+ print(fmt_locate(sources))
777
1862
  else:
778
- print("未发现已知会话日志目录。")
1863
+ print("未发现已知日志 / 记忆源。")
779
1864
  return 1
780
1865
  return 0
781
1866
 
782
1867
  if args.command == "scan":
783
- d = resolve_dir(args)
784
- res = scan_sessions(d)
785
- if args.limit > 0:
786
- res["rows"] = res["rows"][:args.limit]
787
- res["total_sessions"] = len(res["rows"])
1868
+ sources = resolve_sources(args)
1869
+ res = scan_all(sources, limit=args.limit)
1870
+ res["sources"] = _source_brief(sources)
788
1871
  if args.json:
789
1872
  print(json.dumps(res, ensure_ascii=False, indent=2))
790
1873
  else:
@@ -792,9 +1875,9 @@ def main(argv=None):
792
1875
  return 0 if res["rows"] else 1
793
1876
 
794
1877
  if args.command == "search":
795
- d = resolve_dir(args)
796
- res = search_sessions(
797
- d, args.query, regex=args.regex, date=args.date,
1878
+ sources = resolve_sources(args)
1879
+ res = search_all(
1880
+ sources, args.query, regex=args.regex, date=args.date,
798
1881
  sessions=args.session, role=args.role, limit=args.limit,
799
1882
  context=args.context, no_redact=args.no_redact)
800
1883
  if args.json:
@@ -804,21 +1887,21 @@ def main(argv=None):
804
1887
  "version": VERSION,
805
1888
  "query": args.query,
806
1889
  "regex": args.regex,
807
- "dir": str(d),
1890
+ "sources": _source_brief(sources),
808
1891
  "total_matches": len(res["matches"]),
809
1892
  "sessions_hit": res["sessions_hit"],
810
1893
  "truncated": res["truncated"],
811
1894
  "matches": res["matches"],
812
1895
  }, ensure_ascii=False, indent=2))
813
1896
  else:
814
- print(fmt_search(res, args.query, d, args.regex))
1897
+ print(fmt_search(res, args.query, sources, args.regex))
815
1898
  return 0 if res["matches"] else 1
816
1899
 
817
1900
  if args.command == "session":
818
- d = resolve_dir(args)
819
- res = extract_session(d, args.sid, role=args.role,
820
- with_tools=args.tools,
821
- no_redact=args.no_redact)
1901
+ sources = resolve_sources(args)
1902
+ res = extract_all(sources, args.sid, role=args.role,
1903
+ with_tools=args.tools,
1904
+ no_redact=args.no_redact)
822
1905
  if args.limit > 0:
823
1906
  res["messages"] = res["messages"][:args.limit]
824
1907
  if args.json:
@@ -828,9 +1911,8 @@ def main(argv=None):
828
1911
  return 0
829
1912
 
830
1913
  if args.command == "stats":
831
- d = resolve_dir(args)
832
- res = session_stats(d, session_id=args.session, daily=args.daily)
833
- res["dir"] = str(d)
1914
+ sources = resolve_sources(args)
1915
+ res = stats_all(sources, session_id=args.session, daily=args.daily)
834
1916
  if args.json:
835
1917
  print(json.dumps(res, ensure_ascii=False, indent=2))
836
1918
  else:
@@ -838,14 +1920,14 @@ def main(argv=None):
838
1920
  return 0 if res["sessions"] else 1
839
1921
 
840
1922
  if args.command == "tools":
841
- d = resolve_dir(args)
842
- items = tool_breakdown(d, session_id=args.session)
1923
+ sources = resolve_sources(args)
1924
+ items = tools_all(sources, session_id=args.session)
843
1925
  if args.json:
844
1926
  print(json.dumps({
845
1927
  "command": "tools",
846
1928
  "tool": TOOL_NAME,
847
1929
  "version": VERSION,
848
- "dir": str(d),
1930
+ "sources": _source_brief(sources),
849
1931
  "tools": [{"name": n, "count": c} for n, c in items],
850
1932
  }, ensure_ascii=False, indent=2))
851
1933
  else: