@yottameta/yotta-logs 0.1.0 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +12 -0
- package/README.md +40 -23
- package/SKILL.md +35 -25
- package/package.json +2 -2
- package/references/agent-formats.md +127 -0
- package/references/cli.md +23 -11
- package/references/format.md +60 -25
- package/references/security.md +8 -7
- package/scripts/test_yotta_logs.py +353 -0
- package/scripts/yotta_logs.py +1419 -337
package/scripts/yotta_logs.py
CHANGED
|
@@ -1,39 +1,49 @@
|
|
|
1
1
|
#!/usr/bin/env python3
|
|
2
2
|
# -*- coding: utf-8 -*-
|
|
3
|
-
"""yotta_logs.py — YottaMeta 元史(yotta-logs
|
|
3
|
+
"""yotta_logs.py — YottaMeta 元史(yotta-logs):跨智能体历史会话 / 记忆日志检索引擎。
|
|
4
4
|
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
5
|
+
v0.2.0 通用化:不再只认 JSONL,按「格式族 × 字段别名归一 + 配置兜底」适配
|
|
6
|
+
JSONL / 单文件 JSON / SQLite / Markdown / 二进制 五大格式族;discover 全源登记;
|
|
7
|
+
新增 --source / --kind / --format 过滤;默认检索范围 = 会话源 + 结构化记忆源开、
|
|
8
|
+
自由笔记 / 二进制日志默认关(可显式开)。格式普查见 references/agent-formats.md。
|
|
8
9
|
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
10
|
+
零依赖(Python 3.8+ 标准库),只读检索 / 分析,不修改、不删除、不联网上传。
|
|
11
|
+
与元忆(yotta-memory,语义记忆)互补:本技能只管原始日志 / 记忆文件的定位、
|
|
12
|
+
检索、提取与统计。
|
|
13
|
+
|
|
14
|
+
子命令(7 个语义不变):
|
|
15
|
+
locate 全源登记(来源 / 格式 / 类型 / 路径 / 默认范围)
|
|
16
|
+
scan [--dir D] 列出所有会话(来源 / ID / 日期 / 消息数 / 大小)
|
|
17
|
+
search <query> [--dir D] 跨源关键词 / 正则检索,输出时间线命中
|
|
13
18
|
session <sid> [--dir D] 提取单个会话原文(时间线 + 角色 + 文本)
|
|
14
|
-
stats [--dir D]
|
|
19
|
+
stats [--dir D] 统计(消息 / token / 成本 / 每日汇总 / 分源)
|
|
15
20
|
tools [--dir D] 工具调用次数排行
|
|
16
21
|
version 打印版本
|
|
17
22
|
|
|
18
23
|
通用选项:
|
|
19
|
-
--dir PATH
|
|
24
|
+
--dir PATH 日志 / 记忆目录或文件(目录自动嗅探格式族;缺省读
|
|
25
|
+
YOTTA_LOGS_DIR,再 discover 全源登记)
|
|
26
|
+
--source NAME 只检索指定来源(可多次;名称见 locate 登记)
|
|
27
|
+
--kind KIND 只检索指定类型:session / memory / note / log
|
|
28
|
+
--format FMT 只检索指定格式:jsonl / json / sqlite / markdown / binary
|
|
20
29
|
--json 输出纯 JSON(stdout 无其它噪音)
|
|
21
|
-
--no-redact
|
|
30
|
+
--no-redact 关闭默认脱敏
|
|
22
31
|
--limit N 最多返回 N 条(默认 50)
|
|
23
32
|
|
|
24
33
|
退出码(与元安 / 元审 / 元盾 / 元真家族一致):
|
|
25
34
|
0 = 成功(检索到结果 / 操作完成)
|
|
26
35
|
1 = 无匹配 / 空结果集(search 未命中、scan / stats 无会话)
|
|
27
|
-
4 = 用法错误 /
|
|
36
|
+
4 = 用法错误 / 路径不存在 / 致命异常
|
|
28
37
|
|
|
29
38
|
用法示例:
|
|
30
39
|
python3 yotta_logs.py locate
|
|
31
40
|
python3 yotta_logs.py scan --dir ~/.clawdbot/agents/dashu/sessions
|
|
32
|
-
python3 yotta_logs.py search "部署方案"
|
|
33
|
-
python3 yotta_logs.py search "CI 失败" --regex --date 2026-08-26
|
|
41
|
+
python3 yotta_logs.py search "部署方案"
|
|
42
|
+
python3 yotta_logs.py search "CI 失败" --regex --date 2026-08-26 --source opencode-db
|
|
43
|
+
python3 yotta_logs.py search "记住" --kind memory
|
|
34
44
|
python3 yotta_logs.py session abc123 --role assistant
|
|
35
45
|
python3 yotta_logs.py stats --dir /path/to/sessions --daily
|
|
36
|
-
python3 yotta_logs.py tools --dir /path/to/
|
|
46
|
+
python3 yotta_logs.py tools --dir /path/to/logs --format sqlite
|
|
37
47
|
"""
|
|
38
48
|
import argparse
|
|
39
49
|
import datetime as _dt
|
|
@@ -41,6 +51,7 @@ import glob
|
|
|
41
51
|
import json
|
|
42
52
|
import os
|
|
43
53
|
import re
|
|
54
|
+
import sqlite3
|
|
44
55
|
import sys
|
|
45
56
|
from pathlib import Path
|
|
46
57
|
|
|
@@ -53,14 +64,30 @@ try:
|
|
|
53
64
|
except Exception:
|
|
54
65
|
pass
|
|
55
66
|
|
|
56
|
-
VERSION = "0.
|
|
67
|
+
VERSION = "0.2.0"
|
|
57
68
|
TOOL_NAME = "yotta-logs"
|
|
58
69
|
TOOL_CN = "元史"
|
|
59
70
|
DEFAULT_LIMIT = 50
|
|
60
71
|
DEFAULT_CONTEXT = 40 # 命中上下文半径(字符)
|
|
61
72
|
JSONL_SUFFIXES = (".jsonl", ".jsonlines", ".ndjson")
|
|
62
|
-
|
|
63
|
-
|
|
73
|
+
JSON_SUFFIXES = (".json",)
|
|
74
|
+
SQLITE_SUFFIXES = (".db", ".sqlite", ".sqlite3", ".vscdb")
|
|
75
|
+
MD_SUFFIXES = (".md", ".markdown", ".mdown")
|
|
76
|
+
BINARY_SUFFIXES = (".pbtxt", ".nitrite", ".cascade", ".bin", ".enc")
|
|
77
|
+
ROLE_TOOL = ("tool", "toolResult", "tool_result", "toolCall", "tool_call",
|
|
78
|
+
"function", "functionCall", "function_call")
|
|
79
|
+
KIND_CHOICES = ("session", "memory", "note", "log")
|
|
80
|
+
FORMAT_CHOICES = ("jsonl", "json", "sqlite", "markdown", "binary")
|
|
81
|
+
|
|
82
|
+
# 字段别名(按序取首个命中)——适配一切关键字段
|
|
83
|
+
TIME_ALIASES = ("timestamp", "time_created", "created", "time", "ts", "date",
|
|
84
|
+
"created_at", "mtime", "updated")
|
|
85
|
+
ROLE_ALIASES = ("role", "type", "kind")
|
|
86
|
+
TEXT_ALIASES = ("text", "content", "body", "message", "statement",
|
|
87
|
+
"text_content")
|
|
88
|
+
SESSION_ALIASES = ("session_id", "thread_id", "sessionId", "session",
|
|
89
|
+
"conversation_id", "threadId")
|
|
90
|
+
TITLE_ALIASES = ("title", "subject", "name", "heading")
|
|
64
91
|
|
|
65
92
|
# ── 脱敏(默认开启)──────────────────────────────────────────────────────
|
|
66
93
|
|
|
@@ -107,109 +134,11 @@ def redact(text):
|
|
|
107
134
|
out.append(chunk)
|
|
108
135
|
return "".join(out)
|
|
109
136
|
|
|
137
|
+
# ── JSONL 会话日志目录(兼容保留)───────────────────────────────────────
|
|
110
138
|
|
|
111
|
-
# ── 记录解析(容错:字段缺失不报错,坏行由 parse_jsonl 计数跳过)─────────
|
|
112
|
-
|
|
113
|
-
def _rec_ts(rec):
|
|
114
|
-
ts = rec.get("timestamp")
|
|
115
|
-
if not ts and isinstance(rec.get("message"), dict):
|
|
116
|
-
ts = rec["message"].get("timestamp")
|
|
117
|
-
return str(ts) if ts else ""
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
def _rec_role(rec):
|
|
121
|
-
msg = rec.get("message")
|
|
122
|
-
if isinstance(msg, dict) and msg.get("role"):
|
|
123
|
-
role = str(msg["role"])
|
|
124
|
-
else:
|
|
125
|
-
role = rec.get("role")
|
|
126
|
-
role = str(role) if role else ""
|
|
127
|
-
if role in ROLE_TOOL:
|
|
128
|
-
return "tool"
|
|
129
|
-
return role
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
def _rec_content(rec):
|
|
133
|
-
msg = rec.get("message")
|
|
134
|
-
if isinstance(msg, dict):
|
|
135
|
-
return msg.get("content")
|
|
136
|
-
return rec.get("content")
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
def _rec_text(rec):
|
|
140
|
-
"""提取记录里的人类可读文本(content 列表只取 type=text,字符串直接取)。"""
|
|
141
|
-
content = _rec_content(rec)
|
|
142
|
-
if isinstance(content, str):
|
|
143
|
-
return content
|
|
144
|
-
if isinstance(content, list):
|
|
145
|
-
parts = []
|
|
146
|
-
for item in content:
|
|
147
|
-
if not isinstance(item, dict):
|
|
148
|
-
continue
|
|
149
|
-
if item.get("type") == "text" and item.get("text"):
|
|
150
|
-
parts.append(str(item["text"]))
|
|
151
|
-
return "\n".join(parts)
|
|
152
|
-
return ""
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
def _rec_tool_names(rec):
|
|
156
|
-
"""提取记录里的工具调用名(toolCall / toolResult)。"""
|
|
157
|
-
content = _rec_content(rec)
|
|
158
|
-
names = []
|
|
159
|
-
if isinstance(content, list):
|
|
160
|
-
for item in content:
|
|
161
|
-
if not isinstance(item, dict):
|
|
162
|
-
continue
|
|
163
|
-
if item.get("type") in ("tool_call", "toolCall", "toolResult"):
|
|
164
|
-
nm = item.get("name") or item.get("toolName") or ""
|
|
165
|
-
if nm:
|
|
166
|
-
names.append(str(nm))
|
|
167
|
-
return names
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
def _rec_cost(rec):
|
|
171
|
-
msg = rec.get("message")
|
|
172
|
-
usage = None
|
|
173
|
-
if isinstance(msg, dict):
|
|
174
|
-
usage = msg.get("usage")
|
|
175
|
-
if not isinstance(usage, dict):
|
|
176
|
-
usage = rec.get("usage")
|
|
177
|
-
if not isinstance(usage, dict):
|
|
178
|
-
return 0.0
|
|
179
|
-
cost = usage.get("cost")
|
|
180
|
-
if isinstance(cost, dict):
|
|
181
|
-
return float(cost.get("total") or 0)
|
|
182
|
-
try:
|
|
183
|
-
return float(cost or 0)
|
|
184
|
-
except (TypeError, ValueError):
|
|
185
|
-
return 0.0
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
def _rec_tokens(rec):
|
|
189
|
-
msg = rec.get("message")
|
|
190
|
-
usage = None
|
|
191
|
-
if isinstance(msg, dict):
|
|
192
|
-
usage = msg.get("usage")
|
|
193
|
-
if not isinstance(usage, dict):
|
|
194
|
-
usage = rec.get("usage")
|
|
195
|
-
if not isinstance(usage, dict):
|
|
196
|
-
return (0, 0)
|
|
197
|
-
return (int(usage.get("input_tokens") or 0),
|
|
198
|
-
int(usage.get("output_tokens") or 0))
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
def _is_message(rec):
|
|
202
|
-
"""是否为可计入统计的消息记录(排除 session 元数据 / 空角色)。"""
|
|
203
|
-
role = _rec_role(rec)
|
|
204
|
-
if role in ("", "session"):
|
|
205
|
-
return False
|
|
206
|
-
return True
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
# ── 会话日志目录 ─────────────────────────────────────────────────────────
|
|
210
139
|
|
|
211
140
|
def discover_dirs():
|
|
212
|
-
"""
|
|
141
|
+
"""自动发现本机常见 JSONL 会话日志目录(只返回存在且含 *.jsonl 的目录)。"""
|
|
213
142
|
home = Path.home()
|
|
214
143
|
patterns = [
|
|
215
144
|
home / ".clawdbot" / "agents" / "*" / "sessions",
|
|
@@ -219,6 +148,8 @@ def discover_dirs():
|
|
|
219
148
|
home / ".gemini" / "sessions",
|
|
220
149
|
home / ".agents" / "sessions",
|
|
221
150
|
]
|
|
151
|
+
if os.environ.get("CODEX_HOME"):
|
|
152
|
+
patterns.append(Path(os.environ["CODEX_HOME"]) / "sessions")
|
|
222
153
|
found = []
|
|
223
154
|
for pat in patterns:
|
|
224
155
|
for d in glob.glob(str(pat)):
|
|
@@ -339,81 +270,1167 @@ def _human_size(n):
|
|
|
339
270
|
return "%d B" % n
|
|
340
271
|
|
|
341
272
|
|
|
342
|
-
# ──
|
|
273
|
+
# ── 记录提取(JSONL 消息形态;兼容保留)─────────────────────────────────
|
|
343
274
|
|
|
344
|
-
def
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
275
|
+
def _rec_ts(rec):
|
|
276
|
+
ts = rec.get("timestamp")
|
|
277
|
+
if not ts and isinstance(rec.get("message"), dict):
|
|
278
|
+
ts = rec["message"].get("timestamp")
|
|
279
|
+
if not ts and isinstance(rec.get("payload"), dict):
|
|
280
|
+
ts = rec["payload"].get("timestamp") or rec["payload"].get("started_at")
|
|
281
|
+
return str(ts) if ts else ""
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def _rec_role(rec):
|
|
285
|
+
payload = rec.get("payload")
|
|
286
|
+
if isinstance(payload, dict):
|
|
287
|
+
ptype = payload.get("type")
|
|
288
|
+
if ptype == "message":
|
|
289
|
+
role = str(payload.get("role") or "")
|
|
290
|
+
elif ptype in ("function_call", "function_call_output",
|
|
291
|
+
"local_shell_call", "shell_call", "web_search_call"):
|
|
292
|
+
role = "tool"
|
|
293
|
+
else:
|
|
294
|
+
role = ""
|
|
295
|
+
return _norm_role(role) if role else ""
|
|
296
|
+
msg = rec.get("message")
|
|
297
|
+
if isinstance(msg, dict) and msg.get("role"):
|
|
298
|
+
role = str(msg["role"])
|
|
299
|
+
else:
|
|
300
|
+
role = rec.get("role")
|
|
301
|
+
role = str(role) if role else ""
|
|
302
|
+
if role in ROLE_TOOL:
|
|
303
|
+
return "tool"
|
|
304
|
+
return role
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def _rec_content(rec):
|
|
308
|
+
payload = rec.get("payload")
|
|
309
|
+
if isinstance(payload, dict) and payload.get("content") is not None:
|
|
310
|
+
return payload["content"]
|
|
311
|
+
msg = rec.get("message")
|
|
312
|
+
if isinstance(msg, dict):
|
|
313
|
+
return msg.get("content")
|
|
314
|
+
return rec.get("content")
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
def _rec_text(rec):
|
|
318
|
+
"""提取记录里的人类可读文本(content 列表只取 type=text,字符串直接取)。"""
|
|
319
|
+
content = _rec_content(rec)
|
|
320
|
+
if isinstance(content, str):
|
|
321
|
+
return content
|
|
322
|
+
if isinstance(content, list):
|
|
323
|
+
parts = []
|
|
324
|
+
for item in content:
|
|
325
|
+
if not isinstance(item, dict):
|
|
356
326
|
continue
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
372
|
-
|
|
327
|
+
if item.get("type") in ("text", "input_text", "output_text") \
|
|
328
|
+
and item.get("text"):
|
|
329
|
+
parts.append(str(item["text"]))
|
|
330
|
+
return "\n".join(parts)
|
|
331
|
+
return ""
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
def _rec_tool_names(rec):
|
|
335
|
+
"""提取记录里的工具调用名(toolCall / toolResult / payload function_call)。"""
|
|
336
|
+
payload = rec.get("payload")
|
|
337
|
+
if isinstance(payload, dict):
|
|
338
|
+
ptype = payload.get("type")
|
|
339
|
+
if ptype in ("function_call", "function_call_output",
|
|
340
|
+
"local_shell_call", "shell_call"):
|
|
341
|
+
nm = payload.get("name") or payload.get("tool_name") or ""
|
|
342
|
+
return [str(nm)] if nm else []
|
|
343
|
+
return []
|
|
344
|
+
content = _rec_content(rec)
|
|
345
|
+
names = []
|
|
346
|
+
if isinstance(content, list):
|
|
347
|
+
for item in content:
|
|
348
|
+
if not isinstance(item, dict):
|
|
349
|
+
continue
|
|
350
|
+
if item.get("type") in ("tool_call", "toolCall", "toolResult"):
|
|
351
|
+
nm = item.get("name") or item.get("toolName") or ""
|
|
352
|
+
if nm:
|
|
353
|
+
names.append(str(nm))
|
|
354
|
+
return names
|
|
355
|
+
|
|
356
|
+
|
|
357
|
+
def _rec_cost(rec):
|
|
358
|
+
payload = rec.get("payload")
|
|
359
|
+
usage = None
|
|
360
|
+
if isinstance(payload, dict):
|
|
361
|
+
usage = payload.get("usage")
|
|
362
|
+
if not isinstance(usage, dict):
|
|
363
|
+
msg = rec.get("message")
|
|
364
|
+
if isinstance(msg, dict):
|
|
365
|
+
usage = msg.get("usage")
|
|
366
|
+
if not isinstance(usage, dict):
|
|
367
|
+
usage = rec.get("usage")
|
|
368
|
+
if not isinstance(usage, dict):
|
|
369
|
+
return 0.0
|
|
370
|
+
cost = usage.get("cost")
|
|
371
|
+
if isinstance(cost, dict):
|
|
372
|
+
return float(cost.get("total") or 0)
|
|
373
|
+
try:
|
|
374
|
+
return float(cost or 0)
|
|
375
|
+
except (TypeError, ValueError):
|
|
376
|
+
return 0.0
|
|
377
|
+
|
|
378
|
+
|
|
379
|
+
def _rec_tokens(rec):
|
|
380
|
+
payload = rec.get("payload")
|
|
381
|
+
usage = None
|
|
382
|
+
if isinstance(payload, dict):
|
|
383
|
+
usage = payload.get("usage")
|
|
384
|
+
if not isinstance(usage, dict):
|
|
385
|
+
msg = rec.get("message")
|
|
386
|
+
if isinstance(msg, dict):
|
|
387
|
+
usage = msg.get("usage")
|
|
388
|
+
if not isinstance(usage, dict):
|
|
389
|
+
usage = rec.get("usage")
|
|
390
|
+
if not isinstance(usage, dict):
|
|
391
|
+
return (0, 0)
|
|
392
|
+
return (int(usage.get("input_tokens") or 0),
|
|
393
|
+
int(usage.get("output_tokens") or 0))
|
|
394
|
+
|
|
395
|
+
|
|
396
|
+
def _is_message(rec):
|
|
397
|
+
"""是否为可计入统计的消息记录(排除 session 元数据 / 空角色)。"""
|
|
398
|
+
role = _rec_role(rec)
|
|
399
|
+
if role in ("", "session"):
|
|
400
|
+
return False
|
|
401
|
+
return True
|
|
402
|
+
|
|
403
|
+
|
|
404
|
+
# ── 统一记录模型 + 字段别名归一 ──────────────────────────────────────────
|
|
405
|
+
|
|
406
|
+
def _first_alias(rec, aliases):
|
|
407
|
+
if not isinstance(rec, dict):
|
|
408
|
+
return None
|
|
409
|
+
for k in aliases:
|
|
410
|
+
if k in rec and rec[k] is not None and rec[k] != "":
|
|
411
|
+
return rec[k]
|
|
412
|
+
return None
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
def _unpack(value):
|
|
416
|
+
"""JSON 字符串解包(一层):content / data 可能是 JSON 编码的字符串。"""
|
|
417
|
+
if isinstance(value, str):
|
|
418
|
+
s = value.strip()
|
|
419
|
+
if s[:1] in ("{", "[") and (s.endswith("}") or s.endswith("]")):
|
|
420
|
+
try:
|
|
421
|
+
return json.loads(s)
|
|
422
|
+
except Exception:
|
|
423
|
+
return value
|
|
424
|
+
return value
|
|
425
|
+
|
|
426
|
+
|
|
427
|
+
def _norm_time(value):
|
|
428
|
+
"""时间戳归一为 ISO 字符串;秒 / 毫秒自动推断。"""
|
|
429
|
+
if value is None:
|
|
430
|
+
return ""
|
|
431
|
+
s = str(value).strip()
|
|
432
|
+
if not s:
|
|
433
|
+
return ""
|
|
434
|
+
if re.fullmatch(r"\d{9,13}(\.\d+)?", s):
|
|
435
|
+
try:
|
|
436
|
+
secs = float(s)
|
|
437
|
+
if secs > 1e12: # 毫秒
|
|
438
|
+
secs /= 1000.0
|
|
439
|
+
return _dt.datetime.fromtimestamp(
|
|
440
|
+
secs, tz=_dt.timezone.utc).isoformat(timespec="seconds")
|
|
441
|
+
except (ValueError, OSError, OverflowError):
|
|
442
|
+
return s
|
|
443
|
+
return s[:-1] + "+00:00" if s.endswith("Z") else s
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
def _norm_role(role):
|
|
447
|
+
if role is None:
|
|
448
|
+
return ""
|
|
449
|
+
r = str(role).strip()
|
|
450
|
+
low = r.lower().replace("_", "").replace("-", "")
|
|
451
|
+
if low in ("toolresult", "toolcall", "tool", "function",
|
|
452
|
+
"functioncall", "toolcallresult"):
|
|
453
|
+
return "tool"
|
|
454
|
+
return r
|
|
455
|
+
|
|
456
|
+
|
|
457
|
+
def _extract_text(rec):
|
|
458
|
+
"""从各种形态的记录里提取人类可读文本(含 content 列表 / 字符串 / 嵌套 message)。"""
|
|
459
|
+
if not isinstance(rec, dict):
|
|
460
|
+
return ""
|
|
461
|
+
msg = rec.get("message")
|
|
462
|
+
content = None
|
|
463
|
+
if isinstance(msg, dict):
|
|
464
|
+
content = msg.get("content")
|
|
465
|
+
if content is None:
|
|
466
|
+
content = rec.get("content")
|
|
467
|
+
if content is None:
|
|
468
|
+
for k in TEXT_ALIASES:
|
|
469
|
+
if k in rec and isinstance(rec[k], (str, list, dict)):
|
|
470
|
+
content = rec[k]
|
|
471
|
+
break
|
|
472
|
+
content = _unpack(content)
|
|
473
|
+
if isinstance(content, str):
|
|
474
|
+
return content
|
|
475
|
+
if isinstance(content, list):
|
|
476
|
+
parts = []
|
|
477
|
+
for item in content:
|
|
478
|
+
if not isinstance(item, dict):
|
|
479
|
+
continue
|
|
480
|
+
if item.get("type") in ("text", "input_text", "output_text") \
|
|
481
|
+
and item.get("text"):
|
|
482
|
+
parts.append(str(item["text"]))
|
|
483
|
+
elif item.get("type") == "text" and item.get("content"):
|
|
484
|
+
parts.append(str(item["content"]))
|
|
485
|
+
return "\n".join(parts)
|
|
486
|
+
if isinstance(content, dict):
|
|
487
|
+
return str(content.get("text") or content.get("content") or "")
|
|
488
|
+
return ""
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def _norm_record(rec, source, fmt, kind, session_default, path, line=0,
|
|
492
|
+
extra_meta=None):
|
|
493
|
+
"""把一个原始记录(dict)归一为统一 Record。"""
|
|
494
|
+
msg = rec.get("message") if isinstance(rec, dict) else None
|
|
495
|
+
payload = rec.get("payload") if isinstance(rec, dict) else None
|
|
496
|
+
if isinstance(msg, dict) or isinstance(payload, dict):
|
|
497
|
+
role = _rec_role(rec)
|
|
498
|
+
ts = _rec_ts(rec) or _first_alias(rec, TIME_ALIASES)
|
|
499
|
+
text = _rec_text(rec)
|
|
500
|
+
else:
|
|
501
|
+
role = _first_alias(rec, ROLE_ALIASES)
|
|
502
|
+
role = _norm_role(role) if role is not None else ""
|
|
503
|
+
ts = _first_alias(rec, TIME_ALIASES)
|
|
504
|
+
text = _extract_text(rec)
|
|
505
|
+
session_id = _first_alias(rec, SESSION_ALIASES) or session_default
|
|
506
|
+
title = _first_alias(rec, TITLE_ALIASES)
|
|
507
|
+
meta = dict(extra_meta or {})
|
|
508
|
+
meta["line"] = line
|
|
509
|
+
if title:
|
|
510
|
+
meta["title"] = str(title)
|
|
511
|
+
tools = _rec_tool_names(rec)
|
|
512
|
+
if tools:
|
|
513
|
+
meta["tools"] = tools
|
|
514
|
+
cost = _rec_cost(rec)
|
|
515
|
+
ti, to = _rec_tokens(rec)
|
|
516
|
+
if cost:
|
|
517
|
+
meta["cost"] = cost
|
|
518
|
+
if ti or to:
|
|
519
|
+
meta["tokens_in"] = ti
|
|
520
|
+
meta["tokens_out"] = to
|
|
373
521
|
return {
|
|
374
|
-
"
|
|
375
|
-
"
|
|
376
|
-
"
|
|
377
|
-
"total_messages": total_messages,
|
|
378
|
-
"total_invalid": total_invalid,
|
|
522
|
+
"source": source, "format": fmt, "kind": kind,
|
|
523
|
+
"session": str(session_id), "time": _norm_time(ts),
|
|
524
|
+
"role": role, "text": text, "path": str(path), "meta": meta,
|
|
379
525
|
}
|
|
380
526
|
|
|
381
527
|
|
|
382
|
-
def
|
|
383
|
-
|
|
384
|
-
|
|
385
|
-
|
|
528
|
+
def _mk_source(name, kind, fmt, path, default_on=True, extra=None):
|
|
529
|
+
return {"name": name, "kind": kind, "format": fmt, "path": str(path),
|
|
530
|
+
"default_on": default_on, "extra": extra or {}}
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
# ── Reader 层:格式族可插拔 ──────────────────────────────────────────────
|
|
534
|
+
|
|
535
|
+
class JSONLReader:
|
|
536
|
+
FORMAT = "jsonl"
|
|
537
|
+
KIND = "session"
|
|
538
|
+
|
|
539
|
+
@classmethod
|
|
540
|
+
def discover(cls, base=None):
|
|
541
|
+
base = base or Path.home()
|
|
542
|
+
patterns = [
|
|
543
|
+
base / ".clawdbot" / "agents" / "*" / "sessions",
|
|
544
|
+
base / ".codex" / "sessions",
|
|
545
|
+
base / ".claude" / "projects" / "*",
|
|
546
|
+
base / ".config" / "opencode" / "sessions",
|
|
547
|
+
base / ".gemini" / "sessions",
|
|
548
|
+
base / ".agents" / "sessions",
|
|
549
|
+
]
|
|
550
|
+
if os.environ.get("CODEX_HOME"):
|
|
551
|
+
patterns.append(Path(os.environ["CODEX_HOME"]) / "sessions")
|
|
552
|
+
out = []
|
|
553
|
+
for pat in patterns:
|
|
554
|
+
for d in glob.glob(str(pat)):
|
|
555
|
+
dp = Path(d)
|
|
556
|
+
if not dp.is_dir():
|
|
557
|
+
continue
|
|
558
|
+
if cls._has_jsonl(dp):
|
|
559
|
+
out.append(_mk_source(cls._name_for(dp), "session",
|
|
560
|
+
"jsonl", dp))
|
|
561
|
+
return out
|
|
562
|
+
|
|
563
|
+
@staticmethod
|
|
564
|
+
def _has_jsonl(dp, max_depth=3):
|
|
565
|
+
"""目录(含最多 3 层子目录)内是否存在 *.jsonl 会话文件。"""
|
|
566
|
+
root = str(dp)
|
|
567
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
568
|
+
depth = dirpath[len(root):].count(os.sep)
|
|
569
|
+
if depth >= max_depth:
|
|
570
|
+
dirnames[:] = []
|
|
571
|
+
for f in filenames:
|
|
572
|
+
if f.lower().endswith(JSONL_SUFFIXES):
|
|
573
|
+
return True
|
|
574
|
+
return False
|
|
575
|
+
|
|
576
|
+
@staticmethod
|
|
577
|
+
def _name_for(dp):
|
|
578
|
+
s = str(dp).replace("\\", "/")
|
|
579
|
+
if ".clawdbot" in s:
|
|
580
|
+
m = re.search(r"agents/([^/]+)/sessions", s)
|
|
581
|
+
return "clawdbot-" + (m.group(1) if m else "sessions")
|
|
582
|
+
if ".codex" in s:
|
|
583
|
+
return "codex-sessions"
|
|
584
|
+
if ".claude" in s:
|
|
585
|
+
return "claude-projects"
|
|
586
|
+
if "opencode" in s:
|
|
587
|
+
return "opencode-sessions"
|
|
588
|
+
if ".gemini" in s:
|
|
589
|
+
return "gemini-sessions"
|
|
590
|
+
if ".agents" in s:
|
|
591
|
+
return "agents-sessions"
|
|
592
|
+
return "jsonl-sessions"
|
|
593
|
+
|
|
594
|
+
@staticmethod
|
|
595
|
+
def _walk_jsonl(d, max_depth=5):
|
|
596
|
+
"""递归收集目录下(含子目录)的 *.jsonl 会话文件。"""
|
|
597
|
+
out = []
|
|
598
|
+
root = str(d)
|
|
599
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
600
|
+
depth = dirpath[len(root):].count(os.sep)
|
|
601
|
+
if depth >= max_depth:
|
|
602
|
+
dirnames[:] = []
|
|
603
|
+
for f in sorted(filenames):
|
|
604
|
+
if f.lower().endswith(JSONL_SUFFIXES):
|
|
605
|
+
out.append(Path(dirpath) / f)
|
|
606
|
+
return out
|
|
607
|
+
|
|
608
|
+
def iter_sessions(self, source, limit=0):
|
|
609
|
+
d = Path(source["path"])
|
|
610
|
+
idx = load_index(d)
|
|
611
|
+
rows = []
|
|
612
|
+
total_messages = 0
|
|
613
|
+
total_invalid = 0
|
|
614
|
+
for path in self._walk_jsonl(d):
|
|
615
|
+
st = path.stat()
|
|
616
|
+
records, invalid = parse_jsonl(path)
|
|
617
|
+
total_invalid += invalid
|
|
618
|
+
first_ts = ""
|
|
619
|
+
messages = 0
|
|
620
|
+
for rec in records:
|
|
621
|
+
if not _is_message(rec):
|
|
622
|
+
continue
|
|
623
|
+
messages += 1
|
|
624
|
+
ts = _rec_ts(rec)
|
|
625
|
+
if not first_ts and ts:
|
|
626
|
+
first_ts = ts
|
|
627
|
+
total_messages += messages
|
|
628
|
+
rows.append({
|
|
629
|
+
"source": source["name"], "format": "jsonl", "kind": "session",
|
|
630
|
+
"session": path.stem, "alias": _alias_for(idx, path.stem),
|
|
631
|
+
"path": str(path), "size": st.st_size,
|
|
632
|
+
"date": first_ts[:10], "messages": messages,
|
|
633
|
+
"invalid": invalid,
|
|
634
|
+
"mtime": _dt.datetime.fromtimestamp(st.st_mtime)
|
|
635
|
+
.isoformat(timespec="seconds"),
|
|
636
|
+
})
|
|
637
|
+
rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
|
|
638
|
+
if limit > 0:
|
|
639
|
+
rows = rows[:limit]
|
|
640
|
+
return rows, total_messages, total_invalid
|
|
641
|
+
|
|
642
|
+
def iter_records(self, source):
|
|
643
|
+
d = Path(source["path"])
|
|
644
|
+
for path in self._walk_jsonl(d):
|
|
645
|
+
records, _ = parse_jsonl(path)
|
|
646
|
+
for lineno, rec in enumerate(records, 1):
|
|
647
|
+
if not _is_message(rec):
|
|
648
|
+
continue
|
|
649
|
+
yield _norm_record(rec, source["name"], "jsonl", "session",
|
|
650
|
+
path.stem, path, line=lineno)
|
|
651
|
+
|
|
652
|
+
|
|
653
|
+
class JSONReader:
|
|
654
|
+
FORMAT = "json"
|
|
655
|
+
KIND = "session"
|
|
656
|
+
|
|
657
|
+
@classmethod
|
|
658
|
+
def discover(cls, base=None):
|
|
659
|
+
base = base or Path.home()
|
|
660
|
+
out = []
|
|
661
|
+
for cand in (base / ".continue" / "sessions",
|
|
662
|
+
base / ".config" / "continue" / "sessions"):
|
|
663
|
+
if cand.is_dir() and any(
|
|
664
|
+
p.suffix.lower() in JSON_SUFFIXES for p in cand.iterdir()):
|
|
665
|
+
out.append(_mk_source("continue-sessions", "session", "json", cand))
|
|
666
|
+
return out
|
|
667
|
+
|
|
668
|
+
def _files(self, source):
|
|
669
|
+
p = Path(source["path"])
|
|
670
|
+
if p.is_file():
|
|
671
|
+
return [p]
|
|
672
|
+
return sorted(x for x in p.iterdir()
|
|
673
|
+
if x.is_file() and x.suffix.lower() in JSON_SUFFIXES)
|
|
674
|
+
|
|
675
|
+
def _iter_file_records(self, path, source):
|
|
676
|
+
try:
|
|
677
|
+
data = json.loads(path.read_text(encoding="utf-8", errors="replace"))
|
|
678
|
+
except Exception:
|
|
679
|
+
return
|
|
680
|
+
if isinstance(data, list):
|
|
681
|
+
for i, item in enumerate(data, 1):
|
|
682
|
+
if isinstance(item, dict):
|
|
683
|
+
yield _norm_record(item, source["name"], "json", "session",
|
|
684
|
+
path.stem, path, line=i)
|
|
685
|
+
elif isinstance(data, dict):
|
|
686
|
+
for sid, val in data.items():
|
|
687
|
+
if isinstance(val, list):
|
|
688
|
+
for i, item in enumerate(val, 1):
|
|
689
|
+
if isinstance(item, dict):
|
|
690
|
+
yield _norm_record(item, source["name"], "json",
|
|
691
|
+
"session", sid, path, line=i)
|
|
692
|
+
elif isinstance(val, dict):
|
|
693
|
+
yield _norm_record(val, source["name"], "json", "session",
|
|
694
|
+
sid, path, line=1)
|
|
695
|
+
|
|
696
|
+
def iter_records(self, source):
|
|
697
|
+
for p in self._files(source):
|
|
698
|
+
for rec in self._iter_file_records(p, source):
|
|
699
|
+
yield rec
|
|
700
|
+
|
|
701
|
+
def iter_sessions(self, source, limit=0):
|
|
702
|
+
rows = []
|
|
703
|
+
total_messages = 0
|
|
704
|
+
for p in self._files(source):
|
|
705
|
+
recs = list(self._iter_file_records(p, source))
|
|
706
|
+
grouped = {}
|
|
707
|
+
for r in recs:
|
|
708
|
+
grouped.setdefault(r["session"], []).append(r)
|
|
709
|
+
for sid, rl in grouped.items():
|
|
710
|
+
first_ts = next((r["time"] for r in rl if r["time"]), "")
|
|
711
|
+
total_messages += len(rl)
|
|
712
|
+
rows.append({
|
|
713
|
+
"source": source["name"], "format": "json", "kind": "session",
|
|
714
|
+
"session": sid, "alias": "", "path": str(p),
|
|
715
|
+
"size": p.stat().st_size, "date": first_ts[:10],
|
|
716
|
+
"messages": len(rl), "invalid": 0,
|
|
717
|
+
"mtime": _dt.datetime.fromtimestamp(p.stat().st_mtime)
|
|
718
|
+
.isoformat(timespec="seconds"),
|
|
719
|
+
})
|
|
720
|
+
rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
|
|
721
|
+
if limit > 0:
|
|
722
|
+
rows = rows[:limit]
|
|
723
|
+
return rows, total_messages, 0
|
|
724
|
+
|
|
725
|
+
|
|
726
|
+
class SQLiteReader:
|
|
727
|
+
FORMAT = "sqlite"
|
|
728
|
+
KIND = "session"
|
|
729
|
+
|
|
730
|
+
@classmethod
|
|
731
|
+
def discover(cls, base=None):
|
|
732
|
+
base = base or Path.home()
|
|
733
|
+
out = []
|
|
734
|
+
cands = []
|
|
735
|
+
xdg = os.environ.get("XDG_DATA_HOME")
|
|
736
|
+
if xdg:
|
|
737
|
+
cands.append(("opencode-db",
|
|
738
|
+
Path(xdg) / "opencode" / "opencode.db"))
|
|
739
|
+
env = os.environ.get("OPENCODE_DATA")
|
|
740
|
+
if env:
|
|
741
|
+
cands.append(("opencode-db", Path(env) / "data" / "opencode" / "opencode.db"))
|
|
742
|
+
cands.append(("opencode-db", Path(env) / "opencode.db"))
|
|
743
|
+
cands += [
|
|
744
|
+
("opencode-db", base / ".local" / "share" / "opencode" / "opencode.db"),
|
|
745
|
+
("opencode-db", base / ".config" / "opencode" / "opencode.db"),
|
|
746
|
+
("opencode-db", base / ".OpenCodeData" / "data" / "opencode" / "opencode.db"),
|
|
747
|
+
]
|
|
748
|
+
seen = set()
|
|
749
|
+
for name, p in cands:
|
|
750
|
+
key = str(p)
|
|
751
|
+
if key in seen:
|
|
752
|
+
continue
|
|
753
|
+
seen.add(key)
|
|
754
|
+
if p.exists():
|
|
755
|
+
out.append(_mk_source(name, "session", "sqlite", p))
|
|
756
|
+
# VS Code / Cursor state.vscdb(Windows / Linux / macOS)
|
|
757
|
+
for app in ("Code", "Cursor"):
|
|
758
|
+
for root in (base / "AppData" / "Roaming" / app / "User" / "globalStorage",
|
|
759
|
+
base / ".config" / app / "User" / "globalStorage",
|
|
760
|
+
base / "Library" / "Application Support" / app / "User" / "globalStorage"):
|
|
761
|
+
for d in glob.glob(str(root / "*")):
|
|
762
|
+
p = Path(d) / "state.vscdb"
|
|
763
|
+
if p.exists():
|
|
764
|
+
out.append(_mk_source(app.lower() + "-state", "session",
|
|
765
|
+
"sqlite", p))
|
|
766
|
+
return out
|
|
767
|
+
|
|
768
|
+
@staticmethod
|
|
769
|
+
def _connect(path):
|
|
770
|
+
return sqlite3.connect("file:%s?mode=ro" % str(path).replace("\\", "/"),
|
|
771
|
+
uri=True)
|
|
772
|
+
|
|
773
|
+
def _is_opencode(self, con):
|
|
774
|
+
tabs = {r[0] for r in con.execute(
|
|
775
|
+
"SELECT name FROM sqlite_master WHERE type='table'")}
|
|
776
|
+
return {"session", "message", "part"}.issubset(tabs)
|
|
777
|
+
|
|
778
|
+
def _pick_generic(self, con, extra):
|
|
779
|
+
if extra.get("table"):
|
|
780
|
+
return extra["table"]
|
|
781
|
+
tabs = [r[0] for r in con.execute(
|
|
782
|
+
"SELECT name FROM sqlite_master WHERE type='table' ORDER BY name")]
|
|
783
|
+
best = None
|
|
784
|
+
for t in tabs:
|
|
785
|
+
if t.startswith("sqlite_"):
|
|
786
|
+
continue
|
|
787
|
+
cols = [c[1].lower() for c in con.execute("PRAGMA table_info(%s)" % t)]
|
|
788
|
+
if any(c in ("text", "content", "body", "message", "statement")
|
|
789
|
+
for c in cols):
|
|
790
|
+
if "id" in cols or "session" in cols or "role" in cols:
|
|
791
|
+
return t
|
|
792
|
+
if best is None:
|
|
793
|
+
best = t
|
|
794
|
+
return best
|
|
795
|
+
|
|
796
|
+
def _generic_cols(self, con, table, extra):
|
|
797
|
+
real = [c[1] for c in con.execute("PRAGMA table_info(%s)" % table)]
|
|
798
|
+
low = [c.lower() for c in real]
|
|
799
|
+
|
|
800
|
+
def pick(aliases, explicit):
|
|
801
|
+
if explicit and explicit in real:
|
|
802
|
+
return explicit
|
|
803
|
+
if explicit and explicit.lower() in low:
|
|
804
|
+
return real[low.index(explicit.lower())]
|
|
805
|
+
for a in aliases:
|
|
806
|
+
if a in real:
|
|
807
|
+
return a
|
|
808
|
+
if a.lower() in low:
|
|
809
|
+
return real[low.index(a.lower())]
|
|
810
|
+
return None
|
|
811
|
+
|
|
812
|
+
return (pick(TIME_ALIASES, extra.get("col_time")),
|
|
813
|
+
pick(ROLE_ALIASES, extra.get("col_role")),
|
|
814
|
+
pick(TEXT_ALIASES, extra.get("col_text")),
|
|
815
|
+
pick(SESSION_ALIASES, extra.get("col_session")),
|
|
816
|
+
pick(TITLE_ALIASES, extra.get("col_title")))
|
|
817
|
+
|
|
818
|
+
def _opencode_records(self, con, source):
|
|
819
|
+
sess_title = dict(con.execute("SELECT id, title FROM session").fetchall())
|
|
820
|
+
parts = {}
|
|
821
|
+
for mid, data in con.execute("SELECT message_id, data FROM part"):
|
|
822
|
+
parts.setdefault(mid, []).append(data)
|
|
823
|
+
for mid, sid, t_created, mdata in con.execute(
|
|
824
|
+
"SELECT id, session_id, time_created, data FROM message"):
|
|
825
|
+
try:
|
|
826
|
+
m = json.loads(mdata) if isinstance(mdata, str) else (mdata or {})
|
|
827
|
+
role = (m or {}).get("role", "")
|
|
828
|
+
except Exception:
|
|
829
|
+
role = ""
|
|
830
|
+
texts = []
|
|
831
|
+
tools = []
|
|
832
|
+
for pdata in parts.get(mid, []):
|
|
833
|
+
try:
|
|
834
|
+
p = json.loads(pdata) if isinstance(pdata, str) else (pdata or {})
|
|
835
|
+
except Exception:
|
|
836
|
+
continue
|
|
837
|
+
ptype = p.get("type", "")
|
|
838
|
+
if ptype == "text" and p.get("text"):
|
|
839
|
+
texts.append(p["text"])
|
|
840
|
+
elif ptype == "tool" and p.get("tool"):
|
|
841
|
+
tools.append(p["tool"])
|
|
842
|
+
if not texts and not tools:
|
|
843
|
+
continue
|
|
844
|
+
meta = {"tools": tools}
|
|
845
|
+
title = sess_title.get(sid)
|
|
846
|
+
if title:
|
|
847
|
+
meta["title"] = title
|
|
848
|
+
yield {
|
|
849
|
+
"source": source["name"], "format": "sqlite", "kind": "session",
|
|
850
|
+
"session": sid, "time": _norm_time(t_created),
|
|
851
|
+
"role": _norm_role(role), "text": "\n".join(texts),
|
|
852
|
+
"path": str(source["path"]), "meta": meta,
|
|
853
|
+
}
|
|
854
|
+
|
|
855
|
+
def iter_records(self, source):
|
|
856
|
+
con = self._connect(source["path"])
|
|
857
|
+
try:
|
|
858
|
+
if self._is_opencode(con):
|
|
859
|
+
for r in self._opencode_records(con, source):
|
|
860
|
+
yield r
|
|
861
|
+
return
|
|
862
|
+
extra = source.get("extra") or {}
|
|
863
|
+
table = self._pick_generic(con, extra)
|
|
864
|
+
if not table:
|
|
865
|
+
return
|
|
866
|
+
col_time, col_role, col_text, col_session, col_title = \
|
|
867
|
+
self._generic_cols(con, table, extra)
|
|
868
|
+
if not col_text:
|
|
869
|
+
return
|
|
870
|
+
cols = [d[0] for d in con.execute("SELECT * FROM %s LIMIT 0" % table)
|
|
871
|
+
.description]
|
|
872
|
+
for i, row in enumerate(con.execute("SELECT * FROM %s" % table), 1):
|
|
873
|
+
rec = dict(zip(cols, row))
|
|
874
|
+
text = rec.get(col_text)
|
|
875
|
+
if isinstance(text, (dict, list)):
|
|
876
|
+
text = json.dumps(text, ensure_ascii=False)
|
|
877
|
+
elif isinstance(text, str):
|
|
878
|
+
text = _unpack(text)
|
|
879
|
+
text = str(text) if text is not None else ""
|
|
880
|
+
sid = rec.get(col_session) if col_session else \
|
|
881
|
+
Path(source["path"]).stem
|
|
882
|
+
meta = {}
|
|
883
|
+
if col_title and rec.get(col_title) is not None:
|
|
884
|
+
meta["title"] = str(rec[col_title])
|
|
885
|
+
yield {
|
|
886
|
+
"source": source["name"], "format": "sqlite",
|
|
887
|
+
"kind": source["kind"],
|
|
888
|
+
"session": str(sid),
|
|
889
|
+
"time": _norm_time(rec.get(col_time) if col_time else None),
|
|
890
|
+
"role": _norm_role(rec.get(col_role) if col_role else ""),
|
|
891
|
+
"text": text, "path": str(source["path"]), "meta": meta,
|
|
892
|
+
}
|
|
893
|
+
finally:
|
|
894
|
+
con.close()
|
|
895
|
+
|
|
896
|
+
def iter_sessions(self, source, limit=0):
|
|
897
|
+
con = self._connect(source["path"])
|
|
898
|
+
try:
|
|
899
|
+
rows = []
|
|
900
|
+
if self._is_opencode(con):
|
|
901
|
+
cols = {c[1] for c in con.execute("PRAGMA table_info(session)")}
|
|
902
|
+
use_metrics = {"cost", "tokens_input", "tokens_output"}.issubset(cols)
|
|
903
|
+
if use_metrics:
|
|
904
|
+
q = ("SELECT id, title, time_created, cost, tokens_input, "
|
|
905
|
+
"tokens_output, (SELECT COUNT(*) FROM message m WHERE "
|
|
906
|
+
"m.session_id = s.id) FROM session s "
|
|
907
|
+
"ORDER BY time_created DESC")
|
|
908
|
+
else:
|
|
909
|
+
q = ("SELECT id, title, time_created, 0, 0, 0, "
|
|
910
|
+
"(SELECT COUNT(*) FROM message m WHERE "
|
|
911
|
+
"m.session_id = s.id) FROM session s "
|
|
912
|
+
"ORDER BY time_created DESC")
|
|
913
|
+
for sid, title, t_created, cost, ti, to, cnt in con.execute(q):
|
|
914
|
+
rows.append({
|
|
915
|
+
"source": source["name"], "format": "sqlite",
|
|
916
|
+
"kind": "session", "session": sid, "alias": "",
|
|
917
|
+
"path": str(source["path"]),
|
|
918
|
+
"size": os.path.getsize(source["path"]),
|
|
919
|
+
"date": _norm_time(t_created)[:10],
|
|
920
|
+
"messages": cnt, "invalid": 0,
|
|
921
|
+
"mtime": _norm_time(t_created)[:19].replace("T", " "),
|
|
922
|
+
"title": title or "",
|
|
923
|
+
})
|
|
924
|
+
return rows, sum(r["messages"] for r in rows), 0
|
|
925
|
+
extra = source.get("extra") or {}
|
|
926
|
+
table = self._pick_generic(con, extra)
|
|
927
|
+
if not table:
|
|
928
|
+
return [], 0, 0
|
|
929
|
+
col_time, col_role, col_text, col_session, col_title = \
|
|
930
|
+
self._generic_cols(con, table, extra)
|
|
931
|
+
if not col_session:
|
|
932
|
+
cnt = con.execute("SELECT COUNT(*) FROM %s" % table).fetchone()[0]
|
|
933
|
+
rows.append({
|
|
934
|
+
"source": source["name"], "format": "sqlite",
|
|
935
|
+
"kind": source["kind"], "session": Path(source["path"]).stem,
|
|
936
|
+
"alias": "", "path": str(source["path"]),
|
|
937
|
+
"size": os.path.getsize(source["path"]),
|
|
938
|
+
"date": "", "messages": cnt, "invalid": 0,
|
|
939
|
+
"mtime": _dt.datetime.fromtimestamp(
|
|
940
|
+
os.path.getmtime(source["path"])).isoformat(timespec="seconds"),
|
|
941
|
+
"title": "",
|
|
942
|
+
})
|
|
943
|
+
return rows, cnt, 0
|
|
944
|
+
q = ("SELECT %s, COUNT(*) FROM %s GROUP BY %s"
|
|
945
|
+
% (col_session, table, col_session))
|
|
946
|
+
for sid, cnt in con.execute(q):
|
|
947
|
+
rows.append({
|
|
948
|
+
"source": source["name"], "format": "sqlite",
|
|
949
|
+
"kind": source["kind"], "session": str(sid), "alias": "",
|
|
950
|
+
"path": str(source["path"]),
|
|
951
|
+
"size": os.path.getsize(source["path"]),
|
|
952
|
+
"date": "", "messages": cnt, "invalid": 0,
|
|
953
|
+
"mtime": _dt.datetime.fromtimestamp(
|
|
954
|
+
os.path.getmtime(source["path"])).isoformat(timespec="seconds"),
|
|
955
|
+
"title": "",
|
|
956
|
+
})
|
|
957
|
+
rows.sort(key=lambda r: r["session"])
|
|
958
|
+
if limit > 0:
|
|
959
|
+
rows = rows[:limit]
|
|
960
|
+
return rows, sum(r["messages"] for r in rows), 0
|
|
961
|
+
finally:
|
|
962
|
+
con.close()
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
class MarkdownReader:
|
|
966
|
+
FORMAT = "markdown"
|
|
967
|
+
KIND = "memory"
|
|
968
|
+
|
|
969
|
+
@classmethod
|
|
970
|
+
def discover(cls, base=None):
|
|
971
|
+
base = base or Path.home()
|
|
972
|
+
cwd = Path.cwd()
|
|
973
|
+
out = []
|
|
974
|
+
mem = cls._memory_home(base)
|
|
975
|
+
for name, sub in (("yottamemory-facts", "facts"),
|
|
976
|
+
("yottamemory-private", "private"),
|
|
977
|
+
("yottamemory-archive", "archive")):
|
|
978
|
+
p = mem / sub
|
|
979
|
+
if p.is_dir() and any(x.suffix.lower() in MD_SUFFIXES
|
|
980
|
+
for x in p.iterdir()):
|
|
981
|
+
out.append(_mk_source(name, "memory", "markdown", p))
|
|
982
|
+
codex_home = os.environ.get("CODEX_HOME")
|
|
983
|
+
codex_notes = Path(codex_home) / "memories" if codex_home \
|
|
984
|
+
else base / ".CodexData" / "memories"
|
|
985
|
+
if codex_notes.is_dir() and cls._has_md(codex_notes):
|
|
986
|
+
out.append(_mk_source("codex-notes", "note", "markdown", codex_notes,
|
|
987
|
+
default_on=False))
|
|
988
|
+
# Aider 会话历史(当前目录浅扫)
|
|
989
|
+
for p in sorted(cwd.glob("*.aider.*.md")) + \
|
|
990
|
+
sorted(cwd.glob("*.aider.*.markdown")):
|
|
991
|
+
out.append(_mk_source("aider-history", "session", "markdown", p))
|
|
992
|
+
return out
|
|
993
|
+
|
|
994
|
+
@staticmethod
|
|
995
|
+
def _memory_home(base):
|
|
996
|
+
"""yotta-memory 记忆库位置:优先读引擎 config.json 的 memory_home。"""
|
|
997
|
+
try:
|
|
998
|
+
cfg_p = base / ".yottamemory" / "config.json"
|
|
999
|
+
cfg = json.loads(cfg_p.read_text(encoding="utf-8", errors="replace"))
|
|
1000
|
+
mh = cfg.get("memory_home")
|
|
1001
|
+
if mh:
|
|
1002
|
+
return Path(mh)
|
|
1003
|
+
except Exception:
|
|
1004
|
+
pass
|
|
1005
|
+
return base / ".yottamemory"
|
|
1006
|
+
|
|
1007
|
+
@staticmethod
|
|
1008
|
+
def _has_md(p, depth=3):
|
|
1009
|
+
for x in p.rglob("*.md"):
|
|
1010
|
+
rel = x.relative_to(p)
|
|
1011
|
+
if len(rel.parts) <= depth:
|
|
1012
|
+
return True
|
|
1013
|
+
return False
|
|
1014
|
+
|
|
1015
|
+
def _files(self, source):
|
|
1016
|
+
p = Path(source["path"])
|
|
1017
|
+
if p.is_file():
|
|
1018
|
+
return [p]
|
|
1019
|
+
if not p.is_dir():
|
|
1020
|
+
return []
|
|
1021
|
+
out = []
|
|
1022
|
+
for x in sorted(p.rglob("*.md")):
|
|
1023
|
+
rel = x.relative_to(p)
|
|
1024
|
+
if len(rel.parts) <= 4:
|
|
1025
|
+
out.append(x)
|
|
1026
|
+
return out
|
|
1027
|
+
|
|
1028
|
+
@staticmethod
|
|
1029
|
+
def _split_frontmatter(text):
|
|
1030
|
+
"""YAML frontmatter 子集解析:--- 块 → (dict, 正文)。零依赖。"""
|
|
1031
|
+
if not text.startswith("---"):
|
|
1032
|
+
return {}, text
|
|
1033
|
+
lines = text.split("\n")
|
|
1034
|
+
end = None
|
|
1035
|
+
for i in range(1, min(len(lines), 300)):
|
|
1036
|
+
if lines[i].strip() == "---":
|
|
1037
|
+
end = i
|
|
1038
|
+
break
|
|
1039
|
+
if end is None:
|
|
1040
|
+
return {}, text
|
|
1041
|
+
fm = {}
|
|
1042
|
+
for line in lines[1:end]:
|
|
1043
|
+
s = line.strip()
|
|
1044
|
+
if not s or s.startswith("#") or ":" not in s:
|
|
1045
|
+
continue
|
|
1046
|
+
k, _, v = s.partition(":")
|
|
1047
|
+
k = k.strip().lower()
|
|
1048
|
+
v = v.strip()
|
|
1049
|
+
if not v:
|
|
1050
|
+
fm[k] = None
|
|
1051
|
+
continue
|
|
1052
|
+
if v.startswith("[") and v.endswith("]"):
|
|
1053
|
+
inner = v[1:-1].strip()
|
|
1054
|
+
fm[k] = [x.strip().strip("\"'")
|
|
1055
|
+
for x in inner.split(",") if x.strip()]
|
|
1056
|
+
elif v[:1] in ("\"", "'") and v[-1:] == v[:1]:
|
|
1057
|
+
fm[k] = v[1:-1]
|
|
1058
|
+
else:
|
|
1059
|
+
fm[k] = v
|
|
1060
|
+
body = "\n".join(lines[end + 1:])
|
|
1061
|
+
return fm, body
|
|
1062
|
+
|
|
1063
|
+
def _file_record(self, path, source):
|
|
1064
|
+
try:
|
|
1065
|
+
text = path.read_text(encoding="utf-8", errors="replace")
|
|
1066
|
+
except Exception:
|
|
1067
|
+
return None
|
|
1068
|
+
fm, body = self._split_frontmatter(text)
|
|
1069
|
+
st = path.stat()
|
|
1070
|
+
mtime = _dt.datetime.fromtimestamp(st.st_mtime) \
|
|
1071
|
+
.isoformat(timespec="seconds")
|
|
1072
|
+
kind = source["kind"]
|
|
1073
|
+
role = ""
|
|
1074
|
+
if fm:
|
|
1075
|
+
role = _norm_role(fm.get("type") or "")
|
|
1076
|
+
if not role and kind == "memory":
|
|
1077
|
+
role = "memory"
|
|
1078
|
+
title = fm.get("subject") or fm.get("title") or fm.get("name")
|
|
1079
|
+
content = fm.get("statement") or fm.get("content") or fm.get("text")
|
|
1080
|
+
if content is None:
|
|
1081
|
+
content = body.strip()
|
|
1082
|
+
else:
|
|
1083
|
+
content = str(content)
|
|
1084
|
+
if not title:
|
|
1085
|
+
m = re.search(r"^#\s+(.+)$", body, re.M)
|
|
1086
|
+
title = m.group(1).strip() if m else ""
|
|
1087
|
+
ts = fm.get("created") or fm.get("date") or fm.get("updated") or mtime
|
|
1088
|
+
meta = {}
|
|
1089
|
+
if title:
|
|
1090
|
+
meta["title"] = title
|
|
1091
|
+
for k in ("tags", "confidence", "scope", "owner", "immutable"):
|
|
1092
|
+
if fm.get(k) is not None:
|
|
1093
|
+
meta[k] = fm[k]
|
|
1094
|
+
return {
|
|
1095
|
+
"source": source["name"], "format": "markdown", "kind": kind,
|
|
1096
|
+
"session": path.stem, "time": _norm_time(ts),
|
|
1097
|
+
"role": role, "text": content, "path": str(path), "meta": meta,
|
|
1098
|
+
}
|
|
1099
|
+
|
|
1100
|
+
def iter_records(self, source):
|
|
1101
|
+
for p in self._files(source):
|
|
1102
|
+
r = self._file_record(p, source)
|
|
1103
|
+
if r and (r["text"] or r["meta"].get("title")):
|
|
1104
|
+
yield r
|
|
1105
|
+
|
|
1106
|
+
def iter_sessions(self, source, limit=0):
|
|
1107
|
+
rows = []
|
|
1108
|
+
total = 0
|
|
1109
|
+
for p in self._files(source):
|
|
1110
|
+
r = self._file_record(p, source)
|
|
1111
|
+
if not r:
|
|
1112
|
+
continue
|
|
1113
|
+
total += 1
|
|
1114
|
+
rows.append({
|
|
1115
|
+
"source": source["name"], "format": "markdown",
|
|
1116
|
+
"kind": r["kind"], "session": p.stem, "alias": "",
|
|
1117
|
+
"path": str(p), "size": p.stat().st_size,
|
|
1118
|
+
"date": r["time"][:10], "messages": 1, "invalid": 0,
|
|
1119
|
+
"mtime": _dt.datetime.fromtimestamp(p.stat().st_mtime)
|
|
1120
|
+
.isoformat(timespec="seconds"),
|
|
1121
|
+
"title": r["meta"].get("title", ""),
|
|
1122
|
+
})
|
|
1123
|
+
rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
|
|
1124
|
+
if limit > 0:
|
|
1125
|
+
rows = rows[:limit]
|
|
1126
|
+
return rows, total, 0
|
|
1127
|
+
|
|
1128
|
+
|
|
1129
|
+
class BinaryReader:
|
|
1130
|
+
FORMAT = "binary"
|
|
1131
|
+
KIND = "log"
|
|
1132
|
+
|
|
1133
|
+
@classmethod
|
|
1134
|
+
def discover(cls, base=None):
|
|
1135
|
+
base = base or Path.home()
|
|
1136
|
+
out = []
|
|
1137
|
+
for root in (base / ".codeium" / "windsurf", base / ".windsurf"):
|
|
1138
|
+
if root.is_dir():
|
|
1139
|
+
for p in list(root.glob("**/*.pbtxt"))[:200]:
|
|
1140
|
+
out.append(_mk_source("windsurf-conv", "log", "binary",
|
|
1141
|
+
p, default_on=False))
|
|
1142
|
+
return out
|
|
1143
|
+
|
|
1144
|
+
def iter_records(self, source):
|
|
1145
|
+
p = Path(source["path"])
|
|
1146
|
+
title = p.stem
|
|
1147
|
+
try:
|
|
1148
|
+
raw = p.read_bytes()[:512]
|
|
1149
|
+
s = raw.decode("utf-8", errors="replace")
|
|
1150
|
+
m = re.search(r"[A-Za-z0-9\u4e00-\u9fff][^\x00-\x1f]{2,80}", s)
|
|
1151
|
+
if m:
|
|
1152
|
+
title = m.group(0).strip()
|
|
1153
|
+
except Exception:
|
|
1154
|
+
pass
|
|
1155
|
+
st = p.stat()
|
|
1156
|
+
yield {
|
|
1157
|
+
"source": source["name"], "format": "binary", "kind": "log",
|
|
1158
|
+
"session": p.stem,
|
|
1159
|
+
"time": _dt.datetime.fromtimestamp(st.st_mtime)
|
|
1160
|
+
.isoformat(timespec="seconds"),
|
|
1161
|
+
"role": "", "text": title, "path": str(p),
|
|
1162
|
+
"meta": {"title": title},
|
|
1163
|
+
}
|
|
1164
|
+
|
|
1165
|
+
def iter_sessions(self, source, limit=0):
|
|
1166
|
+
rows = []
|
|
1167
|
+
for rec in self.iter_records(source):
|
|
1168
|
+
p = Path(source["path"])
|
|
1169
|
+
st = p.stat()
|
|
1170
|
+
rows.append({
|
|
1171
|
+
"source": source["name"], "format": "binary", "kind": "log",
|
|
1172
|
+
"session": rec["session"], "alias": "", "path": str(p),
|
|
1173
|
+
"size": st.st_size, "date": rec["time"][:10],
|
|
1174
|
+
"messages": 1, "invalid": 0,
|
|
1175
|
+
"mtime": _dt.datetime.fromtimestamp(st.st_mtime)
|
|
1176
|
+
.isoformat(timespec="seconds"),
|
|
1177
|
+
"title": rec["meta"].get("title", ""),
|
|
1178
|
+
})
|
|
1179
|
+
return rows, len(rows), 0
|
|
1180
|
+
|
|
1181
|
+
|
|
1182
|
+
READERS = (JSONLReader, JSONReader, SQLiteReader, MarkdownReader, BinaryReader)
|
|
1183
|
+
|
|
1184
|
+
|
|
1185
|
+
def reader_for(fmt):
|
|
1186
|
+
for cls in READERS:
|
|
1187
|
+
if cls.FORMAT == fmt:
|
|
1188
|
+
return cls()
|
|
1189
|
+
raise SystemExit("不支持的格式:%s" % fmt)
|
|
1190
|
+
|
|
1191
|
+
|
|
1192
|
+
# ── 嗅探(--dir 指向目录 / 文件时自动判定格式族)────────────────────────
|
|
1193
|
+
|
|
1194
|
+
def _sniff_file_format(p):
|
|
1195
|
+
name = p.name.lower()
|
|
1196
|
+
for suf in JSONL_SUFFIXES:
|
|
1197
|
+
if name.endswith(suf):
|
|
1198
|
+
return "jsonl"
|
|
1199
|
+
if name.endswith(JSON_SUFFIXES):
|
|
1200
|
+
return "json"
|
|
1201
|
+
if name.endswith(SQLITE_SUFFIXES):
|
|
1202
|
+
return "sqlite"
|
|
1203
|
+
if name.endswith(MD_SUFFIXES):
|
|
1204
|
+
return "markdown"
|
|
1205
|
+
if name.endswith(BINARY_SUFFIXES):
|
|
1206
|
+
return "binary"
|
|
1207
|
+
try:
|
|
1208
|
+
head = p.read_bytes()[:512]
|
|
1209
|
+
except Exception:
|
|
1210
|
+
return "binary"
|
|
1211
|
+
if head.startswith(b"SQLite format 3"):
|
|
1212
|
+
return "sqlite"
|
|
1213
|
+
try:
|
|
1214
|
+
s = head.decode("utf-8", errors="replace").lstrip()
|
|
1215
|
+
except Exception:
|
|
1216
|
+
return "binary"
|
|
1217
|
+
if s.startswith("---"):
|
|
1218
|
+
return "markdown"
|
|
1219
|
+
if s[:1] in ("{", "["):
|
|
1220
|
+
first_line = s.split("\n", 1)[0].strip()
|
|
1221
|
+
if first_line.startswith("["):
|
|
1222
|
+
return "json"
|
|
1223
|
+
try:
|
|
1224
|
+
obj = json.loads(first_line)
|
|
1225
|
+
return "jsonl" if isinstance(obj, dict) else "json"
|
|
1226
|
+
except Exception:
|
|
1227
|
+
return "json"
|
|
1228
|
+
if re.search(r"^#\s+\S", s, re.M):
|
|
1229
|
+
return "markdown"
|
|
1230
|
+
return "binary"
|
|
1231
|
+
|
|
1232
|
+
|
|
1233
|
+
def _sniff_dir_format(p):
|
|
1234
|
+
names = [f.name.lower() for f in p.iterdir() if f.is_file()]
|
|
1235
|
+
if any(n.endswith(JSONL_SUFFIXES) for n in names):
|
|
1236
|
+
return "jsonl"
|
|
1237
|
+
if any(n.endswith(SQLITE_SUFFIXES) for n in names):
|
|
1238
|
+
return "sqlite"
|
|
1239
|
+
if any(n.endswith(JSON_SUFFIXES) for n in names):
|
|
1240
|
+
return "json"
|
|
1241
|
+
if any(n.endswith(MD_SUFFIXES) for n in names):
|
|
1242
|
+
return "markdown"
|
|
1243
|
+
return None
|
|
1244
|
+
|
|
1245
|
+
|
|
1246
|
+
def _sniff_file_kind(p):
|
|
1247
|
+
fmt = _sniff_file_format(p)
|
|
1248
|
+
if fmt == "markdown":
|
|
1249
|
+
try:
|
|
1250
|
+
t = p.read_text(encoding="utf-8", errors="replace")
|
|
1251
|
+
return "memory" if t.startswith("---") else "note"
|
|
1252
|
+
except Exception:
|
|
1253
|
+
return "note"
|
|
1254
|
+
if fmt == "binary":
|
|
1255
|
+
return "log"
|
|
1256
|
+
return "session"
|
|
1257
|
+
|
|
1258
|
+
|
|
1259
|
+
def _sniff_dir_kind(p):
|
|
1260
|
+
low = p.name.lower()
|
|
1261
|
+
if any(k in low for k in ("fact", "private", "memory", "记忆")):
|
|
1262
|
+
return "memory"
|
|
1263
|
+
if any(k in low for k in ("note", "notes", "笔记", "memories")):
|
|
1264
|
+
return "note"
|
|
1265
|
+
return "session"
|
|
1266
|
+
|
|
1267
|
+
|
|
1268
|
+
def sniff_source(path):
|
|
1269
|
+
"""把 --dir / YOTTA_LOGS_DIR 指向的路径嗅探为一个单一来源。"""
|
|
1270
|
+
p = Path(path)
|
|
1271
|
+
if not p.exists():
|
|
1272
|
+
raise SystemExit("路径不存在:%s" % p)
|
|
1273
|
+
if p.is_file():
|
|
1274
|
+
return _mk_source(p.stem or "source", _sniff_file_kind(p),
|
|
1275
|
+
_sniff_file_format(p), p)
|
|
1276
|
+
fmt = _sniff_dir_format(p)
|
|
1277
|
+
if fmt == "markdown":
|
|
1278
|
+
return _mk_source(p.name or "source", _sniff_dir_kind(p), fmt, p)
|
|
1279
|
+
if fmt:
|
|
1280
|
+
return _mk_source(p.name or "source", "session", fmt, p)
|
|
1281
|
+
raise SystemExit("无法识别 %s 的日志 / 记忆格式(支持 jsonl / json / "
|
|
1282
|
+
"sqlite / markdown)" % p)
|
|
1283
|
+
|
|
1284
|
+
|
|
1285
|
+
# ── 配置兜底 + discover 全源登记 ─────────────────────────────────────────
|
|
1286
|
+
|
|
1287
|
+
def load_config():
|
|
1288
|
+
p = os.environ.get("YOTTA_LOGS_CONFIG")
|
|
1289
|
+
if not p:
|
|
1290
|
+
p = str(Path.home() / ".config" / "yotta-logs" / "config.json")
|
|
1291
|
+
cfg_path = Path(p)
|
|
1292
|
+
if not cfg_path.exists():
|
|
1293
|
+
return {}
|
|
1294
|
+
try:
|
|
1295
|
+
data = json.loads(cfg_path.read_text(encoding="utf-8", errors="replace"))
|
|
1296
|
+
return data if isinstance(data, dict) else {}
|
|
1297
|
+
except Exception:
|
|
1298
|
+
return {}
|
|
1299
|
+
|
|
1300
|
+
|
|
1301
|
+
def default_scope():
|
|
1302
|
+
cfg = load_config()
|
|
1303
|
+
return list(cfg.get("default_scope") or ["session", "memory"])
|
|
1304
|
+
|
|
1305
|
+
|
|
1306
|
+
def _config_sources(cfg):
|
|
1307
|
+
out = []
|
|
1308
|
+
for s in (cfg.get("sources") or []):
|
|
1309
|
+
if not isinstance(s, dict) or not s.get("path"):
|
|
1310
|
+
continue
|
|
1311
|
+
extra = {k: v for k, v in s.items()
|
|
1312
|
+
if k in ("table", "col_time", "col_role", "col_text",
|
|
1313
|
+
"col_session", "col_title")}
|
|
1314
|
+
out.append(_mk_source(
|
|
1315
|
+
s.get("name") or Path(s["path"]).stem,
|
|
1316
|
+
s.get("kind") or "session",
|
|
1317
|
+
s.get("format") or "jsonl",
|
|
1318
|
+
s["path"],
|
|
1319
|
+
default_on=True,
|
|
1320
|
+
extra=extra,
|
|
1321
|
+
))
|
|
1322
|
+
return out
|
|
1323
|
+
|
|
1324
|
+
|
|
1325
|
+
def discover_sources(config=None):
|
|
1326
|
+
cfg = config if config is not None else load_config()
|
|
1327
|
+
sources = []
|
|
1328
|
+
for cls in READERS:
|
|
1329
|
+
for s in cls.discover():
|
|
1330
|
+
sources.append(s)
|
|
1331
|
+
sources += _config_sources(cfg)
|
|
1332
|
+
seen = set()
|
|
1333
|
+
dedup = []
|
|
1334
|
+
for s in sources:
|
|
1335
|
+
key = (s["format"], str(s["path"]).lower())
|
|
1336
|
+
if key in seen:
|
|
1337
|
+
continue
|
|
1338
|
+
seen.add(key)
|
|
1339
|
+
dedup.append(s)
|
|
1340
|
+
scope = set(cfg.get("default_scope") or ["session", "memory"])
|
|
1341
|
+
for s in dedup:
|
|
1342
|
+
s["default_on"] = s["default_on"] or s["kind"] in scope
|
|
1343
|
+
return dedup
|
|
1344
|
+
|
|
1345
|
+
|
|
1346
|
+
def _candidate_ids(source, session_ids):
|
|
1347
|
+
"""把会话 ID / 别名解析为候选会话 ID 集合(JSONL 源带 sessions.json 别名)。"""
|
|
1348
|
+
ids = set()
|
|
1349
|
+
for g in (session_ids if isinstance(session_ids, (list, tuple))
|
|
1350
|
+
else [session_ids]):
|
|
1351
|
+
ids.add(g)
|
|
1352
|
+
if source["format"] == "jsonl":
|
|
1353
|
+
idx = load_index(Path(source["path"]))
|
|
1354
|
+
ids |= _resolve_session_ids(idx, g)
|
|
1355
|
+
return ids
|
|
1356
|
+
|
|
1357
|
+
|
|
1358
|
+
def filter_sources(sources, args):
|
|
1359
|
+
explicit = False
|
|
1360
|
+
if getattr(args, "source", None):
|
|
1361
|
+
names = set(args.source)
|
|
1362
|
+
sources = [s for s in sources if s["name"] in names]
|
|
1363
|
+
explicit = True
|
|
1364
|
+
if getattr(args, "format", None):
|
|
1365
|
+
sources = [s for s in sources if s["format"] == args.format]
|
|
1366
|
+
explicit = True
|
|
1367
|
+
if getattr(args, "kind", None):
|
|
1368
|
+
sources = [s for s in sources if s["kind"] == args.kind]
|
|
1369
|
+
explicit = True
|
|
1370
|
+
if not explicit:
|
|
1371
|
+
sources = [s for s in sources if s["default_on"]]
|
|
1372
|
+
return sources
|
|
1373
|
+
|
|
1374
|
+
|
|
1375
|
+
def resolve_sources(args):
|
|
1376
|
+
if getattr(args, "dir", None):
|
|
1377
|
+
srcs = [sniff_source(args.dir)]
|
|
1378
|
+
else:
|
|
1379
|
+
env = os.environ.get("YOTTA_LOGS_DIR")
|
|
1380
|
+
if env:
|
|
1381
|
+
srcs = [sniff_source(env)]
|
|
1382
|
+
else:
|
|
1383
|
+
srcs = discover_sources()
|
|
1384
|
+
if not srcs:
|
|
1385
|
+
raise SystemExit(
|
|
1386
|
+
"未找到已知日志 / 记忆源:用 --dir 指定,或设 YOTTA_LOGS_DIR / "
|
|
1387
|
+
"YOTTA_LOGS_CONFIG;可用 locate 查看候选。")
|
|
1388
|
+
return filter_sources(srcs, args)
|
|
1389
|
+
|
|
1390
|
+
|
|
1391
|
+
# ── 统一检索 / 提取 / 统计 / 工具排行 ───────────────────────────────────
|
|
1392
|
+
|
|
1393
|
+
def scan_all(sources, limit=0):
|
|
1394
|
+
rows = []
|
|
1395
|
+
total_messages = 0
|
|
1396
|
+
total_invalid = 0
|
|
1397
|
+
for src in sources:
|
|
1398
|
+
reader = reader_for(src["format"])
|
|
1399
|
+
r, tm, inv = reader.iter_sessions(src)
|
|
1400
|
+
rows.extend(r)
|
|
1401
|
+
total_messages += tm
|
|
1402
|
+
total_invalid += inv
|
|
1403
|
+
if limit > 0:
|
|
1404
|
+
rows = rows[:limit]
|
|
1405
|
+
return {"rows": rows, "total_sessions": len(rows),
|
|
1406
|
+
"total_messages": total_messages, "total_invalid": total_invalid}
|
|
1407
|
+
|
|
1408
|
+
|
|
1409
|
+
def search_all(sources, query, regex=False, date=None, sessions=None, role=None,
|
|
1410
|
+
limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT, no_redact=False):
|
|
386
1411
|
if regex:
|
|
387
1412
|
try:
|
|
388
1413
|
pat = re.compile(query, re.I)
|
|
389
1414
|
except re.error as e:
|
|
390
1415
|
raise SystemExit("正则无效:%s" % e)
|
|
391
|
-
idx = load_index(dir_path)
|
|
392
|
-
sid_filter = None
|
|
393
|
-
if sessions:
|
|
394
|
-
sid_filter = set()
|
|
395
|
-
for g in sessions:
|
|
396
|
-
sid_filter |= _resolve_session_ids(idx, g)
|
|
397
1416
|
matches = []
|
|
398
|
-
|
|
1417
|
+
hit = set()
|
|
399
1418
|
truncated = False
|
|
400
|
-
for
|
|
401
|
-
|
|
402
|
-
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
|
|
406
|
-
if not _is_message(rec):
|
|
1419
|
+
for source in sources:
|
|
1420
|
+
reader = reader_for(source["format"])
|
|
1421
|
+
cands = _candidate_ids(source, sessions) if sessions else None
|
|
1422
|
+
for rec in reader.iter_records(source):
|
|
1423
|
+
text = rec["text"]
|
|
1424
|
+
if not text:
|
|
407
1425
|
continue
|
|
408
|
-
|
|
1426
|
+
if cands is not None and rec["session"] not in cands:
|
|
1427
|
+
continue
|
|
1428
|
+
ts = rec["time"]
|
|
409
1429
|
if date and not _ts_on_date(ts, date):
|
|
410
1430
|
continue
|
|
411
|
-
rrole =
|
|
1431
|
+
rrole = rec["role"]
|
|
412
1432
|
if role and rrole != role:
|
|
413
1433
|
continue
|
|
414
|
-
text = _rec_text(rec)
|
|
415
|
-
if not text:
|
|
416
|
-
continue
|
|
417
1434
|
if regex:
|
|
418
1435
|
m = pat.search(text)
|
|
419
1436
|
if not m:
|
|
@@ -431,164 +1448,213 @@ def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
|
|
|
431
1448
|
snippet = redact(snippet)
|
|
432
1449
|
matched = redact(matched)
|
|
433
1450
|
matches.append({
|
|
434
|
-
"
|
|
435
|
-
"
|
|
436
|
-
"role": rrole,
|
|
437
|
-
"line":
|
|
438
|
-
"match": matched,
|
|
439
|
-
"text": snippet,
|
|
1451
|
+
"source": source["name"], "format": source["format"],
|
|
1452
|
+
"kind": source["kind"], "session": rec["session"],
|
|
1453
|
+
"timestamp": ts, "role": rrole,
|
|
1454
|
+
"line": rec["meta"].get("line", 0),
|
|
1455
|
+
"match": matched, "text": snippet,
|
|
440
1456
|
})
|
|
441
|
-
|
|
1457
|
+
hit.add((source["name"], rec["session"]))
|
|
442
1458
|
if len(matches) >= limit:
|
|
443
1459
|
truncated = True
|
|
444
|
-
return {
|
|
445
|
-
|
|
446
|
-
|
|
447
|
-
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
|
|
453
|
-
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
|
|
457
|
-
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
if
|
|
461
|
-
info = {"session": sid, "path": str(p)}
|
|
1460
|
+
return {"matches": matches, "sessions_hit": len(hit),
|
|
1461
|
+
"truncated": truncated}
|
|
1462
|
+
return {"matches": matches, "sessions_hit": len(hit), "truncated": truncated}
|
|
1463
|
+
|
|
1464
|
+
|
|
1465
|
+
def extract_all(sources, session_id, role=None, with_tools=False,
|
|
1466
|
+
no_redact=False):
|
|
1467
|
+
"""跨源提取单个会话;多个源同名会话取第一个(--source 可消歧)。"""
|
|
1468
|
+
source_match = None
|
|
1469
|
+
for source in sources:
|
|
1470
|
+
reader = reader_for(source["format"])
|
|
1471
|
+
cands = _candidate_ids(source, session_id)
|
|
1472
|
+
for rec in reader.iter_records(source):
|
|
1473
|
+
if rec["session"] in cands:
|
|
1474
|
+
source_match = source
|
|
1475
|
+
break
|
|
1476
|
+
if source_match:
|
|
462
1477
|
break
|
|
463
|
-
if
|
|
464
|
-
p = Path(dir_path) / session_id
|
|
465
|
-
if p.exists() and p.is_file():
|
|
466
|
-
info = {"session": p.stem, "path": str(p)}
|
|
467
|
-
if info is None:
|
|
1478
|
+
if source_match is None:
|
|
468
1479
|
raise SystemExit("未找到会话:%s(可用 scan 列出会话 ID)" % session_id)
|
|
469
|
-
|
|
1480
|
+
reader = reader_for(source_match["format"])
|
|
1481
|
+
cands = _candidate_ids(source_match, session_id)
|
|
1482
|
+
actual = None
|
|
470
1483
|
messages = []
|
|
471
|
-
|
|
472
|
-
|
|
1484
|
+
total_records = 0
|
|
1485
|
+
for rec in reader.iter_records(source_match):
|
|
1486
|
+
if rec["session"] not in cands:
|
|
473
1487
|
continue
|
|
474
|
-
|
|
1488
|
+
if actual is None:
|
|
1489
|
+
actual = rec["session"]
|
|
1490
|
+
total_records += 1
|
|
1491
|
+
rrole = rec["role"]
|
|
475
1492
|
if role and rrole != role:
|
|
476
1493
|
continue
|
|
477
|
-
text =
|
|
478
|
-
tools =
|
|
1494
|
+
text = rec["text"]
|
|
1495
|
+
tools = rec["meta"].get("tools") or []
|
|
479
1496
|
if not text and not tools:
|
|
480
1497
|
continue
|
|
481
1498
|
if not no_redact:
|
|
482
1499
|
text = redact(text)
|
|
483
1500
|
messages.append({
|
|
484
|
-
"line":
|
|
485
|
-
"timestamp":
|
|
1501
|
+
"line": rec["meta"].get("line", 0),
|
|
1502
|
+
"timestamp": rec["time"],
|
|
486
1503
|
"role": rrole,
|
|
487
1504
|
"text": text,
|
|
488
1505
|
"tools": tools,
|
|
489
1506
|
})
|
|
490
1507
|
return {
|
|
491
|
-
"session":
|
|
492
|
-
"dir": str(
|
|
1508
|
+
"session": actual,
|
|
1509
|
+
"dir": str(source_match["path"]),
|
|
1510
|
+
"source": source_match["name"],
|
|
1511
|
+
"format": source_match["format"],
|
|
1512
|
+
"kind": source_match["kind"],
|
|
493
1513
|
"messages": messages,
|
|
494
|
-
"invalid":
|
|
495
|
-
"total_records":
|
|
1514
|
+
"invalid": 0,
|
|
1515
|
+
"total_records": total_records,
|
|
496
1516
|
}
|
|
497
1517
|
|
|
498
1518
|
|
|
499
|
-
def
|
|
500
|
-
"""汇总统计:消息 / 角色 / token / 成本 / 时间范围 / 每日汇总。"""
|
|
501
|
-
idx = load_index(dir_path)
|
|
502
|
-
sessions = list_sessions(dir_path)
|
|
503
|
-
if session_id:
|
|
504
|
-
ids = set()
|
|
505
|
-
for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
|
|
506
|
-
ids |= _resolve_session_ids(idx, g)
|
|
507
|
-
sessions = [s for s in sessions if s["session"] in ids]
|
|
1519
|
+
def stats_all(sources, session_id=None, daily=False):
|
|
508
1520
|
agg = {
|
|
509
|
-
"sessions":
|
|
510
|
-
"
|
|
511
|
-
"
|
|
512
|
-
"roles": {},
|
|
513
|
-
"cost": 0.0,
|
|
514
|
-
"tokens_in": 0,
|
|
515
|
-
"tokens_out": 0,
|
|
516
|
-
"first": "",
|
|
517
|
-
"last": "",
|
|
518
|
-
"days": {},
|
|
1521
|
+
"sessions": 0, "messages": 0, "invalid": 0,
|
|
1522
|
+
"roles": {}, "cost": 0.0, "tokens_in": 0, "tokens_out": 0,
|
|
1523
|
+
"first": "", "last": "", "days": {}, "by_source": {},
|
|
519
1524
|
}
|
|
520
|
-
for
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
1525
|
+
for source in sources:
|
|
1526
|
+
reader = reader_for(source["format"])
|
|
1527
|
+
rows, tm, inv = reader.iter_sessions(source)
|
|
1528
|
+
if session_id:
|
|
1529
|
+
cands = _candidate_ids(source, session_id)
|
|
1530
|
+
rows = [r for r in rows if r["session"] in cands]
|
|
1531
|
+
agg["sessions"] += len(rows)
|
|
1532
|
+
agg["invalid"] += inv
|
|
1533
|
+
bs = agg["by_source"].setdefault(source["name"], {
|
|
1534
|
+
"format": source["format"], "kind": source["kind"],
|
|
1535
|
+
"sessions": 0, "messages": 0, "cost": 0.0,
|
|
1536
|
+
"tokens_in": 0, "tokens_out": 0, "first": "", "last": "",
|
|
1537
|
+
})
|
|
1538
|
+
bs["sessions"] += len(rows)
|
|
1539
|
+
cands = _candidate_ids(source, session_id) if session_id else None
|
|
1540
|
+
for rec in reader.iter_records(source):
|
|
1541
|
+
if cands is not None and rec["session"] not in cands:
|
|
525
1542
|
continue
|
|
526
|
-
|
|
527
|
-
role = _rec_role(rec)
|
|
1543
|
+
role = rec["role"]
|
|
528
1544
|
agg["roles"][role] = agg["roles"].get(role, 0) + 1
|
|
529
|
-
|
|
1545
|
+
bs["messages"] += 1
|
|
1546
|
+
ts = rec["time"]
|
|
1547
|
+
cost = rec["meta"].get("cost", 0.0)
|
|
1548
|
+
ti = rec["meta"].get("tokens_in", 0)
|
|
1549
|
+
to = rec["meta"].get("tokens_out", 0)
|
|
530
1550
|
if ts:
|
|
531
1551
|
day = ts[:10]
|
|
532
1552
|
if day:
|
|
533
1553
|
d = agg["days"].setdefault(day, {"messages": 0, "cost": 0.0})
|
|
534
1554
|
d["messages"] += 1
|
|
535
|
-
d["cost"] +=
|
|
1555
|
+
d["cost"] += cost
|
|
536
1556
|
if not agg["first"] or ts < agg["first"]:
|
|
537
1557
|
agg["first"] = ts
|
|
538
1558
|
if not agg["last"] or ts > agg["last"]:
|
|
539
1559
|
agg["last"] = ts
|
|
540
|
-
|
|
541
|
-
|
|
1560
|
+
if not bs["first"] or ts < bs["first"]:
|
|
1561
|
+
bs["first"] = ts
|
|
1562
|
+
if not bs["last"] or ts > bs["last"]:
|
|
1563
|
+
bs["last"] = ts
|
|
1564
|
+
agg["cost"] += cost
|
|
542
1565
|
agg["tokens_in"] += ti
|
|
543
1566
|
agg["tokens_out"] += to
|
|
1567
|
+
bs["cost"] += cost
|
|
1568
|
+
bs["tokens_in"] += ti
|
|
1569
|
+
bs["tokens_out"] += to
|
|
1570
|
+
agg["messages"] += bs["messages"]
|
|
544
1571
|
return agg
|
|
545
1572
|
|
|
546
1573
|
|
|
547
|
-
def
|
|
548
|
-
"""工具调用次数排行 → [(工具名, 次数)],按次数降序。"""
|
|
549
|
-
idx = load_index(dir_path)
|
|
550
|
-
sessions = list_sessions(dir_path)
|
|
551
|
-
if session_id:
|
|
552
|
-
ids = set()
|
|
553
|
-
for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
|
|
554
|
-
ids |= _resolve_session_ids(idx, g)
|
|
555
|
-
sessions = [s for s in sessions if s["session"] in ids]
|
|
1574
|
+
def tools_all(sources, session_id=None):
|
|
556
1575
|
counts = {}
|
|
557
|
-
for
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
1576
|
+
for source in sources:
|
|
1577
|
+
reader = reader_for(source["format"])
|
|
1578
|
+
cands = _candidate_ids(source, session_id) if session_id else None
|
|
1579
|
+
for rec in reader.iter_records(source):
|
|
1580
|
+
if cands is not None and rec["session"] not in cands:
|
|
1581
|
+
continue
|
|
1582
|
+
for nm in rec["meta"].get("tools") or []:
|
|
561
1583
|
counts[nm] = counts.get(nm, 0) + 1
|
|
562
1584
|
return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
|
|
563
1585
|
|
|
564
1586
|
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
return
|
|
1587
|
+
# ── 兼容保留:单目录(JSONL)旧函数签名 ─────────────────────────────────
|
|
1588
|
+
|
|
1589
|
+
def scan_sessions(dir_path):
|
|
1590
|
+
source = sniff_source(dir_path)
|
|
1591
|
+
res = scan_all([source])
|
|
1592
|
+
res["dir"] = str(dir_path)
|
|
1593
|
+
return res
|
|
1594
|
+
|
|
1595
|
+
|
|
1596
|
+
def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
|
|
1597
|
+
role=None, limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT,
|
|
1598
|
+
no_redact=False):
|
|
1599
|
+
source = sniff_source(dir_path)
|
|
1600
|
+
return search_all([source], query, regex=regex, date=date,
|
|
1601
|
+
sessions=sessions, role=role, limit=limit,
|
|
1602
|
+
context=context, no_redact=no_redact)
|
|
1603
|
+
|
|
1604
|
+
|
|
1605
|
+
def extract_session(dir_path, session_id, role=None, with_tools=False,
|
|
1606
|
+
no_redact=False):
|
|
1607
|
+
source = sniff_source(dir_path)
|
|
1608
|
+
return extract_all([source], session_id, role=role, with_tools=with_tools,
|
|
1609
|
+
no_redact=no_redact)
|
|
1610
|
+
|
|
1611
|
+
|
|
1612
|
+
def session_stats(dir_path, session_id=None, daily=False):
|
|
1613
|
+
source = sniff_source(dir_path)
|
|
1614
|
+
res = stats_all([source], session_id=session_id, daily=daily)
|
|
1615
|
+
res["dir"] = str(dir_path)
|
|
1616
|
+
return res
|
|
1617
|
+
|
|
1618
|
+
|
|
1619
|
+
def tool_breakdown(dir_path, session_id=None):
|
|
1620
|
+
source = sniff_source(dir_path)
|
|
1621
|
+
return tools_all([source], session_id=session_id)
|
|
572
1622
|
|
|
573
1623
|
|
|
574
1624
|
# ── 文本格式化 ───────────────────────────────────────────────────────────
|
|
575
1625
|
|
|
1626
|
+
def fmt_locate(sources):
|
|
1627
|
+
lines = []
|
|
1628
|
+
scope = " + ".join(default_scope()) if default_scope() else "(无)"
|
|
1629
|
+
lines.append("发现的日志 / 记忆源 %d 个 | 默认范围:%s"
|
|
1630
|
+
% (len(sources), scope))
|
|
1631
|
+
lines.append("")
|
|
1632
|
+
lines.append("%-22s %-9s %-8s %-3s %s"
|
|
1633
|
+
% ("来源", "格式", "类型", "默认", "路径"))
|
|
1634
|
+
for s in sorted(sources, key=lambda x: (x["name"], x["path"])):
|
|
1635
|
+
lines.append("%-22s %-9s %-8s %-3s %s"
|
|
1636
|
+
% (s["name"][:22], s["format"], s["kind"],
|
|
1637
|
+
"开" if s["default_on"] else "关", s["path"]))
|
|
1638
|
+
return "\n".join(lines)
|
|
1639
|
+
|
|
1640
|
+
|
|
576
1641
|
def fmt_scan(res):
|
|
577
1642
|
lines = []
|
|
578
|
-
lines.append("
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
res["total_invalid"]))
|
|
1643
|
+
lines.append("来源 %d 个 | 会话 %d 个 | 消息合计 %d | 无效行 %d"
|
|
1644
|
+
% (len(res.get("sources") or []), res["total_sessions"],
|
|
1645
|
+
res["total_messages"], res["total_invalid"]))
|
|
582
1646
|
lines.append("")
|
|
583
|
-
lines.append("%-24s %-
|
|
1647
|
+
lines.append("%-14s %-8s %-8s %-24s %-10s %8s %10s %s"
|
|
1648
|
+
% ("来源", "格式", "类型", "会话 ID", "日期", "消息", "大小", "别名"))
|
|
584
1649
|
for r in res["rows"]:
|
|
585
|
-
lines.append("%-24s %-
|
|
586
|
-
% (r["
|
|
587
|
-
r["
|
|
1650
|
+
lines.append("%-14s %-8s %-8s %-24s %-10s %8d %10s %s"
|
|
1651
|
+
% (r["source"][:14], r["format"], r["kind"],
|
|
1652
|
+
r["session"][:24], r["date"] or "-", r["messages"],
|
|
1653
|
+
_human_size(r["size"]), r.get("alias", "")))
|
|
588
1654
|
return "\n".join(lines)
|
|
589
1655
|
|
|
590
1656
|
|
|
591
|
-
def fmt_search(res, query,
|
|
1657
|
+
def fmt_search(res, query, sources, regex):
|
|
592
1658
|
lines = []
|
|
593
1659
|
if regex:
|
|
594
1660
|
desc = "正则:%s" % query
|
|
@@ -601,9 +1667,10 @@ def fmt_search(res, query, dir_path, regex):
|
|
|
601
1667
|
lines.append("")
|
|
602
1668
|
cur = None
|
|
603
1669
|
for m in res["matches"]:
|
|
604
|
-
|
|
605
|
-
|
|
606
|
-
|
|
1670
|
+
key = (m["source"], m["session"])
|
|
1671
|
+
if key != cur:
|
|
1672
|
+
cur = key
|
|
1673
|
+
lines.append("── %s / 会话 %s ─────────────────" % (m["source"], m["session"]))
|
|
607
1674
|
lines.append("%s [%s] %s" % (_clock(m["timestamp"]), m["role"], m["text"]))
|
|
608
1675
|
if not res["matches"]:
|
|
609
1676
|
lines.append("未命中。")
|
|
@@ -616,8 +1683,9 @@ def fmt_session(res):
|
|
|
616
1683
|
for m in res["messages"]:
|
|
617
1684
|
counts[m["role"]] = counts.get(m["role"], 0) + 1
|
|
618
1685
|
parts = " | ".join("%s %d" % (k, v) for k, v in sorted(counts.items()))
|
|
619
|
-
lines.append("会话:%s |
|
|
620
|
-
% (res["session"],
|
|
1686
|
+
lines.append("会话:%s | 来源 %s(%s)| 消息 %d(%s)"
|
|
1687
|
+
% (res["session"], res.get("source", "-"),
|
|
1688
|
+
res.get("format", "-"), len(res["messages"]), parts))
|
|
621
1689
|
lines.append("")
|
|
622
1690
|
for m in res["messages"]:
|
|
623
1691
|
lines.append("── %s ─────────────────────────" % _clock(m["timestamp"]))
|
|
@@ -632,7 +1700,10 @@ def fmt_session(res):
|
|
|
632
1700
|
def fmt_stats(res, daily):
|
|
633
1701
|
lines = []
|
|
634
1702
|
lines.append("会话统计")
|
|
635
|
-
|
|
1703
|
+
srcs = " | ".join("%s %d" % (k, v["sessions"])
|
|
1704
|
+
for k, v in sorted(res.get("by_source", {}).items()))
|
|
1705
|
+
if srcs:
|
|
1706
|
+
lines.append("分源会话:%s" % srcs)
|
|
636
1707
|
roles = " | ".join("%s %d" % (k, v)
|
|
637
1708
|
for k, v in sorted(res["roles"].items()))
|
|
638
1709
|
lines.append("会话 %d | 消息 %d(%s)| 无效行 %d"
|
|
@@ -665,6 +1736,15 @@ def fmt_tools(items):
|
|
|
665
1736
|
return "\n".join(lines)
|
|
666
1737
|
|
|
667
1738
|
|
|
1739
|
+
def _snippet(text, span, radius):
|
|
1740
|
+
start, end = span
|
|
1741
|
+
lo = max(0, start - radius)
|
|
1742
|
+
hi = min(len(text), end + radius)
|
|
1743
|
+
pre = "…" if lo > 0 else ""
|
|
1744
|
+
post = "…" if hi < len(text) else ""
|
|
1745
|
+
return pre + text[lo:hi].replace("\n", " ") + post
|
|
1746
|
+
|
|
1747
|
+
|
|
668
1748
|
# ── CLI ──────────────────────────────────────────────────────────────────
|
|
669
1749
|
|
|
670
1750
|
class _Parser(argparse.ArgumentParser):
|
|
@@ -676,56 +1756,55 @@ class _Parser(argparse.ArgumentParser):
|
|
|
676
1756
|
|
|
677
1757
|
def _add_dir(ap):
|
|
678
1758
|
ap.add_argument("--dir", metavar="PATH",
|
|
679
|
-
help="
|
|
1759
|
+
help="日志 / 记忆目录或文件(缺省读 YOTTA_LOGS_DIR,"
|
|
1760
|
+
"再 discover 全源登记)")
|
|
680
1761
|
|
|
681
1762
|
|
|
682
|
-
def
|
|
683
|
-
|
|
684
|
-
|
|
685
|
-
|
|
686
|
-
|
|
687
|
-
|
|
688
|
-
|
|
689
|
-
|
|
690
|
-
|
|
691
|
-
|
|
692
|
-
|
|
693
|
-
|
|
694
|
-
"可用 locate 查看候选。")
|
|
695
|
-
d = Path(found[0])
|
|
696
|
-
if not d.is_dir():
|
|
697
|
-
raise SystemExit("目录不存在:%s" % d)
|
|
698
|
-
return d
|
|
1763
|
+
def _add_filters(ap):
|
|
1764
|
+
ap.add_argument("--source", action="append", metavar="NAME",
|
|
1765
|
+
help="只检索指定来源(可多次;名称见 locate)")
|
|
1766
|
+
ap.add_argument("--kind", choices=KIND_CHOICES,
|
|
1767
|
+
help="只检索指定类型:session / memory / note / log")
|
|
1768
|
+
ap.add_argument("--format", choices=FORMAT_CHOICES,
|
|
1769
|
+
help="只检索指定格式:jsonl / json / sqlite / markdown / binary")
|
|
1770
|
+
|
|
1771
|
+
|
|
1772
|
+
def _source_brief(sources):
|
|
1773
|
+
return [{"name": s["name"], "kind": s["kind"], "format": s["format"],
|
|
1774
|
+
"path": s["path"], "default_on": s["default_on"]} for s in sources]
|
|
699
1775
|
|
|
700
1776
|
|
|
701
1777
|
def main(argv=None):
|
|
702
1778
|
ap = _Parser(
|
|
703
1779
|
prog=TOOL_NAME,
|
|
704
|
-
description="%s(%s
|
|
1780
|
+
description="%s(%s):零依赖跨智能体会话 / 记忆日志检索引擎。"
|
|
1781
|
+
% (TOOL_CN, TOOL_NAME))
|
|
705
1782
|
ap.add_argument("--version", action="version",
|
|
706
1783
|
version="%s %s" % (TOOL_NAME, VERSION))
|
|
707
1784
|
sub = ap.add_subparsers(dest="command", required=True)
|
|
708
1785
|
|
|
709
1786
|
sub.add_parser("version", help="打印版本")
|
|
710
1787
|
|
|
711
|
-
p_locate = sub.add_parser("locate", help="
|
|
1788
|
+
p_locate = sub.add_parser("locate", help="全源登记:发现本机日志 / 记忆源")
|
|
712
1789
|
p_locate.add_argument("--json", action="store_true", help="输出 JSON")
|
|
713
1790
|
|
|
714
|
-
p_scan = sub.add_parser("scan", help="
|
|
1791
|
+
p_scan = sub.add_parser("scan", help="列出所有会话(跨源)")
|
|
715
1792
|
_add_dir(p_scan)
|
|
1793
|
+
_add_filters(p_scan)
|
|
716
1794
|
p_scan.add_argument("--json", action="store_true", help="输出 JSON")
|
|
717
1795
|
p_scan.add_argument("--limit", type=int, default=0,
|
|
718
1796
|
help="最多列出 N 个会话(默认全部)")
|
|
719
1797
|
|
|
720
|
-
p_search = sub.add_parser("search", help="
|
|
1798
|
+
p_search = sub.add_parser("search", help="跨源检索关键词 / 正则")
|
|
721
1799
|
p_search.add_argument("query", help="检索词(默认不区分大小写)")
|
|
722
1800
|
_add_dir(p_search)
|
|
1801
|
+
_add_filters(p_search)
|
|
723
1802
|
p_search.add_argument("--regex", action="store_true", help="把 query 当正则")
|
|
724
1803
|
p_search.add_argument("--date", metavar="YYYY-MM-DD",
|
|
725
1804
|
help="只检索指定日期(或 YYYY-MM)")
|
|
726
1805
|
p_search.add_argument("-s", "--session", action="append", metavar="SID",
|
|
727
1806
|
help="只检索指定会话 ID / 别名(可多次)")
|
|
728
|
-
p_search.add_argument("--role", choices=("user", "assistant", "tool", "system"),
|
|
1807
|
+
p_search.add_argument("--role", choices=("user", "assistant", "tool", "system", "developer"),
|
|
729
1808
|
help="只检索指定角色")
|
|
730
1809
|
p_search.add_argument("--limit", type=int, default=DEFAULT_LIMIT,
|
|
731
1810
|
help="最多返回 N 条命中(默认 %d)" % DEFAULT_LIMIT)
|
|
@@ -736,27 +1815,30 @@ def main(argv=None):
|
|
|
736
1815
|
help="关闭默认脱敏")
|
|
737
1816
|
|
|
738
1817
|
p_session = sub.add_parser("session", help="提取单个会话原文")
|
|
739
|
-
p_session.add_argument("sid", help="会话 ID
|
|
1818
|
+
p_session.add_argument("sid", help="会话 ID / 别名")
|
|
740
1819
|
_add_dir(p_session)
|
|
741
|
-
p_session
|
|
1820
|
+
_add_filters(p_session)
|
|
1821
|
+
p_session.add_argument("--role", choices=("user", "assistant", "tool", "system", "developer"),
|
|
742
1822
|
help="只提取指定角色")
|
|
743
1823
|
p_session.add_argument("--tools", action="store_true",
|
|
744
|
-
help="
|
|
1824
|
+
help="标注工具调用")
|
|
745
1825
|
p_session.add_argument("--limit", type=int, default=0,
|
|
746
|
-
help="
|
|
1826
|
+
help="最多输出 N 条消息(默认全部)")
|
|
747
1827
|
p_session.add_argument("--json", action="store_true", help="输出 JSON")
|
|
748
1828
|
p_session.add_argument("--no-redact", action="store_true",
|
|
749
1829
|
help="关闭默认脱敏")
|
|
750
1830
|
|
|
751
|
-
p_stats = sub.add_parser("stats", help="
|
|
1831
|
+
p_stats = sub.add_parser("stats", help="会话统计汇总(跨源)")
|
|
752
1832
|
_add_dir(p_stats)
|
|
1833
|
+
_add_filters(p_stats)
|
|
753
1834
|
p_stats.add_argument("-s", "--session", metavar="SID",
|
|
754
1835
|
help="只统计指定会话 ID / 别名")
|
|
755
1836
|
p_stats.add_argument("--daily", action="store_true", help="输出每日汇总")
|
|
756
1837
|
p_stats.add_argument("--json", action="store_true", help="输出 JSON")
|
|
757
1838
|
|
|
758
|
-
p_tools = sub.add_parser("tools", help="
|
|
1839
|
+
p_tools = sub.add_parser("tools", help="工具调用次数排行(跨源)")
|
|
759
1840
|
_add_dir(p_tools)
|
|
1841
|
+
_add_filters(p_tools)
|
|
760
1842
|
p_tools.add_argument("-s", "--session", metavar="SID",
|
|
761
1843
|
help="只统计指定会话 ID / 别名")
|
|
762
1844
|
p_tools.add_argument("--json", action="store_true", help="输出 JSON")
|
|
@@ -768,23 +1850,24 @@ def main(argv=None):
|
|
|
768
1850
|
return 0
|
|
769
1851
|
|
|
770
1852
|
if args.command == "locate":
|
|
771
|
-
|
|
1853
|
+
sources = discover_sources()
|
|
772
1854
|
if args.json:
|
|
773
|
-
print(json.dumps({
|
|
774
|
-
|
|
775
|
-
|
|
776
|
-
|
|
1855
|
+
print(json.dumps({
|
|
1856
|
+
"tool": TOOL_NAME, "version": VERSION,
|
|
1857
|
+
"default_scope": default_scope(),
|
|
1858
|
+
"sources": _source_brief(sources),
|
|
1859
|
+
}, ensure_ascii=False, indent=2))
|
|
1860
|
+
elif sources:
|
|
1861
|
+
print(fmt_locate(sources))
|
|
777
1862
|
else:
|
|
778
|
-
print("
|
|
1863
|
+
print("未发现已知日志 / 记忆源。")
|
|
779
1864
|
return 1
|
|
780
1865
|
return 0
|
|
781
1866
|
|
|
782
1867
|
if args.command == "scan":
|
|
783
|
-
|
|
784
|
-
res =
|
|
785
|
-
|
|
786
|
-
res["rows"] = res["rows"][:args.limit]
|
|
787
|
-
res["total_sessions"] = len(res["rows"])
|
|
1868
|
+
sources = resolve_sources(args)
|
|
1869
|
+
res = scan_all(sources, limit=args.limit)
|
|
1870
|
+
res["sources"] = _source_brief(sources)
|
|
788
1871
|
if args.json:
|
|
789
1872
|
print(json.dumps(res, ensure_ascii=False, indent=2))
|
|
790
1873
|
else:
|
|
@@ -792,9 +1875,9 @@ def main(argv=None):
|
|
|
792
1875
|
return 0 if res["rows"] else 1
|
|
793
1876
|
|
|
794
1877
|
if args.command == "search":
|
|
795
|
-
|
|
796
|
-
res =
|
|
797
|
-
|
|
1878
|
+
sources = resolve_sources(args)
|
|
1879
|
+
res = search_all(
|
|
1880
|
+
sources, args.query, regex=args.regex, date=args.date,
|
|
798
1881
|
sessions=args.session, role=args.role, limit=args.limit,
|
|
799
1882
|
context=args.context, no_redact=args.no_redact)
|
|
800
1883
|
if args.json:
|
|
@@ -804,21 +1887,21 @@ def main(argv=None):
|
|
|
804
1887
|
"version": VERSION,
|
|
805
1888
|
"query": args.query,
|
|
806
1889
|
"regex": args.regex,
|
|
807
|
-
"
|
|
1890
|
+
"sources": _source_brief(sources),
|
|
808
1891
|
"total_matches": len(res["matches"]),
|
|
809
1892
|
"sessions_hit": res["sessions_hit"],
|
|
810
1893
|
"truncated": res["truncated"],
|
|
811
1894
|
"matches": res["matches"],
|
|
812
1895
|
}, ensure_ascii=False, indent=2))
|
|
813
1896
|
else:
|
|
814
|
-
print(fmt_search(res, args.query,
|
|
1897
|
+
print(fmt_search(res, args.query, sources, args.regex))
|
|
815
1898
|
return 0 if res["matches"] else 1
|
|
816
1899
|
|
|
817
1900
|
if args.command == "session":
|
|
818
|
-
|
|
819
|
-
res =
|
|
820
|
-
|
|
821
|
-
|
|
1901
|
+
sources = resolve_sources(args)
|
|
1902
|
+
res = extract_all(sources, args.sid, role=args.role,
|
|
1903
|
+
with_tools=args.tools,
|
|
1904
|
+
no_redact=args.no_redact)
|
|
822
1905
|
if args.limit > 0:
|
|
823
1906
|
res["messages"] = res["messages"][:args.limit]
|
|
824
1907
|
if args.json:
|
|
@@ -828,9 +1911,8 @@ def main(argv=None):
|
|
|
828
1911
|
return 0
|
|
829
1912
|
|
|
830
1913
|
if args.command == "stats":
|
|
831
|
-
|
|
832
|
-
res =
|
|
833
|
-
res["dir"] = str(d)
|
|
1914
|
+
sources = resolve_sources(args)
|
|
1915
|
+
res = stats_all(sources, session_id=args.session, daily=args.daily)
|
|
834
1916
|
if args.json:
|
|
835
1917
|
print(json.dumps(res, ensure_ascii=False, indent=2))
|
|
836
1918
|
else:
|
|
@@ -838,14 +1920,14 @@ def main(argv=None):
|
|
|
838
1920
|
return 0 if res["sessions"] else 1
|
|
839
1921
|
|
|
840
1922
|
if args.command == "tools":
|
|
841
|
-
|
|
842
|
-
items =
|
|
1923
|
+
sources = resolve_sources(args)
|
|
1924
|
+
items = tools_all(sources, session_id=args.session)
|
|
843
1925
|
if args.json:
|
|
844
1926
|
print(json.dumps({
|
|
845
1927
|
"command": "tools",
|
|
846
1928
|
"tool": TOOL_NAME,
|
|
847
1929
|
"version": VERSION,
|
|
848
|
-
"
|
|
1930
|
+
"sources": _source_brief(sources),
|
|
849
1931
|
"tools": [{"name": n, "count": c} for n, c in items],
|
|
850
1932
|
}, ensure_ascii=False, indent=2))
|
|
851
1933
|
else:
|