@yottameta/yotta-logs 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,877 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """yotta_logs.py — YottaMeta 元史(yotta-logs):跨智能体历史会话日志检索引擎。
4
+
5
+ 零依赖(Python 3.8+ 标准库),只读检索 / 分析会话 JSONL 记录,为跨会话追溯
6
+ 提供原始日志依据。与元忆(yotta-memory,语义记忆)互补:本技能只管原始会话
7
+ 日志的定位、检索、提取与统计;不修改、不删除、不联网上传任何会话记录。
8
+
9
+ 子命令:
10
+ locate 自动发现本机常见的会话日志目录
11
+ scan [--dir D] 列出目录下所有会话(ID / 日期 / 消息数 / 大小)
12
+ search <query> [--dir D] 按关键词 / 正则跨会话检索,输出时间线命中
13
+ session <sid> [--dir D] 提取单个会话原文(时间线 + 角色 + 文本)
14
+ stats [--dir D] 会话统计(消息 / token / 成本 / 每日汇总)
15
+ tools [--dir D] 工具调用次数排行
16
+ version 打印版本
17
+
18
+ 通用选项:
19
+ --dir PATH 日志目录(缺省读环境变量 YOTTA_LOGS_DIR,再自动定位首个候选)
20
+ --json 输出纯 JSON(stdout 无其它噪音)
21
+ --no-redact 关闭默认脱敏(默认会把疑似密钥 / token / 口令打码)
22
+ --limit N 最多返回 N 条(默认 50)
23
+
24
+ 退出码(与元安 / 元审 / 元盾 / 元真家族一致):
25
+ 0 = 成功(检索到结果 / 操作完成)
26
+ 1 = 无匹配 / 空结果集(search 未命中、scan / stats 无会话)
27
+ 4 = 用法错误 / 目录不存在 / 致命异常
28
+
29
+ 用法示例:
30
+ python3 yotta_logs.py locate
31
+ python3 yotta_logs.py scan --dir ~/.clawdbot/agents/dashu/sessions
32
+ python3 yotta_logs.py search "部署方案" --dir /path/to/sessions
33
+ python3 yotta_logs.py search "CI 失败" --regex --date 2026-08-26
34
+ python3 yotta_logs.py session abc123 --role assistant
35
+ python3 yotta_logs.py stats --dir /path/to/sessions --daily
36
+ python3 yotta_logs.py tools --dir /path/to/sessions
37
+ """
38
+ import argparse
39
+ import datetime as _dt
40
+ import glob
41
+ import json
42
+ import os
43
+ import re
44
+ import sys
45
+ from pathlib import Path
46
+
47
+ try:
48
+ sys.stdout.reconfigure(encoding="utf-8")
49
+ except Exception:
50
+ pass
51
+ try:
52
+ sys.stderr.reconfigure(encoding="utf-8")
53
+ except Exception:
54
+ pass
55
+
56
+ VERSION = "0.1.0"
57
+ TOOL_NAME = "yotta-logs"
58
+ TOOL_CN = "元史"
59
+ DEFAULT_LIMIT = 50
60
+ DEFAULT_CONTEXT = 40 # 命中上下文半径(字符)
61
+ JSONL_SUFFIXES = (".jsonl", ".jsonlines", ".ndjson")
62
+ ROLE_TOOL = ("tool", "toolResult", "tool_result")
63
+
64
+
65
+ # ── 脱敏(默认开启)──────────────────────────────────────────────────────
66
+
67
+ _URL_RE = re.compile(r"(https?://[^\s\"'<>]+)", re.I)
68
+ _URL_USERPASS_RE = re.compile(r"(https?://)([^/\s:@]+):([^/\s@]+)@", re.I)
69
+ _KNOWN_KEY_RE = re.compile(
70
+ r"(?i)\b("
71
+ r"sk-[a-z0-9_-]{8,}" # OpenAI 类 API key
72
+ r"|rk-[a-z0-9_-]{8,}"
73
+ r"|pk-[a-z0-9_-]{8,}"
74
+ r"|gh[pousr]_[a-z0-9]{20,}" # GitHub token
75
+ r"|xox[baprs]-[a-z0-9-]{10,}" # Slack token
76
+ r"|AKIA[0-9A-Z]{16}" # AWS access key
77
+ r"|ASIA[0-9A-Z]{16}"
78
+ r")\b")
79
+ _JWT_RE = re.compile(r"eyJ[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}\.[A-Za-z0-9_-]{10,}")
80
+ _BEARER_RE = re.compile(r"(?i)\bbearer\s+[a-z0-9._~+/-]+")
81
+ _PEM_RE = re.compile(
82
+ r"-----BEGIN [A-Z0-9 ]+ PRIVATE KEY-----.*?-----END [A-Z0-9 ]+ PRIVATE KEY-----",
83
+ re.S)
84
+ _ASSIGN_RE = re.compile(
85
+ r"(?i)\b(token|password|passwd|secret|api[_-]?key|access[_-]?key|"
86
+ r"client[_-]?secret)\b\s*[=:]\s*[\"']?[a-z0-9._~+/\-]{6,}")
87
+ _LONG_TOKEN_RE = re.compile(r"[a-z0-9+/_-]{40,}", re.I)
88
+
89
+
90
+ def redact(text):
91
+ """把疑似密钥 / token / 口令打码(默认开启;--no-redact 关闭)。"""
92
+ if not text:
93
+ return text
94
+ text = _PEM_RE.sub("[PRIVATE KEY REDACTED]", text)
95
+ text = _URL_USERPASS_RE.sub(r"\1\2:***@", text)
96
+ chunks = _URL_RE.split(text) # 奇数下标为 URL,原文保留(路径不算密钥)
97
+ out = []
98
+ for i, chunk in enumerate(chunks):
99
+ if i % 2 == 1:
100
+ out.append(chunk)
101
+ continue
102
+ chunk = _KNOWN_KEY_RE.sub("***", chunk)
103
+ chunk = _JWT_RE.sub("***", chunk)
104
+ chunk = _BEARER_RE.sub("Bearer ***", chunk)
105
+ chunk = _ASSIGN_RE.sub(lambda m: m.group(1) + "=***", chunk)
106
+ chunk = _LONG_TOKEN_RE.sub("***", chunk)
107
+ out.append(chunk)
108
+ return "".join(out)
109
+
110
+
111
+ # ── 记录解析(容错:字段缺失不报错,坏行由 parse_jsonl 计数跳过)─────────
112
+
113
+ def _rec_ts(rec):
114
+ ts = rec.get("timestamp")
115
+ if not ts and isinstance(rec.get("message"), dict):
116
+ ts = rec["message"].get("timestamp")
117
+ return str(ts) if ts else ""
118
+
119
+
120
+ def _rec_role(rec):
121
+ msg = rec.get("message")
122
+ if isinstance(msg, dict) and msg.get("role"):
123
+ role = str(msg["role"])
124
+ else:
125
+ role = rec.get("role")
126
+ role = str(role) if role else ""
127
+ if role in ROLE_TOOL:
128
+ return "tool"
129
+ return role
130
+
131
+
132
+ def _rec_content(rec):
133
+ msg = rec.get("message")
134
+ if isinstance(msg, dict):
135
+ return msg.get("content")
136
+ return rec.get("content")
137
+
138
+
139
+ def _rec_text(rec):
140
+ """提取记录里的人类可读文本(content 列表只取 type=text,字符串直接取)。"""
141
+ content = _rec_content(rec)
142
+ if isinstance(content, str):
143
+ return content
144
+ if isinstance(content, list):
145
+ parts = []
146
+ for item in content:
147
+ if not isinstance(item, dict):
148
+ continue
149
+ if item.get("type") == "text" and item.get("text"):
150
+ parts.append(str(item["text"]))
151
+ return "\n".join(parts)
152
+ return ""
153
+
154
+
155
+ def _rec_tool_names(rec):
156
+ """提取记录里的工具调用名(toolCall / toolResult)。"""
157
+ content = _rec_content(rec)
158
+ names = []
159
+ if isinstance(content, list):
160
+ for item in content:
161
+ if not isinstance(item, dict):
162
+ continue
163
+ if item.get("type") in ("tool_call", "toolCall", "toolResult"):
164
+ nm = item.get("name") or item.get("toolName") or ""
165
+ if nm:
166
+ names.append(str(nm))
167
+ return names
168
+
169
+
170
+ def _rec_cost(rec):
171
+ msg = rec.get("message")
172
+ usage = None
173
+ if isinstance(msg, dict):
174
+ usage = msg.get("usage")
175
+ if not isinstance(usage, dict):
176
+ usage = rec.get("usage")
177
+ if not isinstance(usage, dict):
178
+ return 0.0
179
+ cost = usage.get("cost")
180
+ if isinstance(cost, dict):
181
+ return float(cost.get("total") or 0)
182
+ try:
183
+ return float(cost or 0)
184
+ except (TypeError, ValueError):
185
+ return 0.0
186
+
187
+
188
+ def _rec_tokens(rec):
189
+ msg = rec.get("message")
190
+ usage = None
191
+ if isinstance(msg, dict):
192
+ usage = msg.get("usage")
193
+ if not isinstance(usage, dict):
194
+ usage = rec.get("usage")
195
+ if not isinstance(usage, dict):
196
+ return (0, 0)
197
+ return (int(usage.get("input_tokens") or 0),
198
+ int(usage.get("output_tokens") or 0))
199
+
200
+
201
+ def _is_message(rec):
202
+ """是否为可计入统计的消息记录(排除 session 元数据 / 空角色)。"""
203
+ role = _rec_role(rec)
204
+ if role in ("", "session"):
205
+ return False
206
+ return True
207
+
208
+
209
+ # ── 会话日志目录 ─────────────────────────────────────────────────────────
210
+
211
+ def discover_dirs():
212
+ """自动发现本机常见会话日志目录(只返回存在且含 *.jsonl 的目录)。"""
213
+ home = Path.home()
214
+ patterns = [
215
+ home / ".clawdbot" / "agents" / "*" / "sessions",
216
+ home / ".codex" / "sessions",
217
+ home / ".claude" / "projects" / "*",
218
+ home / ".config" / "opencode" / "sessions",
219
+ home / ".gemini" / "sessions",
220
+ home / ".agents" / "sessions",
221
+ ]
222
+ found = []
223
+ for pat in patterns:
224
+ for d in glob.glob(str(pat)):
225
+ dp = Path(d)
226
+ if not dp.is_dir():
227
+ continue
228
+ if any(p.is_file() and p.name.lower().endswith(JSONL_SUFFIXES)
229
+ for p in dp.iterdir()):
230
+ found.append(str(dp))
231
+ return sorted(set(found))
232
+
233
+
234
+ def list_sessions(dir_path):
235
+ """返回目录下所有会话文件信息(只按文件名/大小,不解析内容)。"""
236
+ d = Path(dir_path)
237
+ out = []
238
+ for p in sorted(d.iterdir()):
239
+ if not p.is_file() or not p.name.lower().endswith(JSONL_SUFFIXES):
240
+ continue
241
+ st = p.stat()
242
+ out.append({
243
+ "session": p.stem,
244
+ "path": str(p),
245
+ "size": st.st_size,
246
+ "mtime": _dt.datetime.fromtimestamp(st.st_mtime)
247
+ .isoformat(timespec="seconds"),
248
+ })
249
+ return out
250
+
251
+
252
+ def parse_jsonl(path):
253
+ """解析一个 JSONL 文件 → (records, invalid)。容错:坏行跳过并计数。"""
254
+ records = []
255
+ invalid = 0
256
+ with open(path, "r", encoding="utf-8", errors="replace") as fh:
257
+ for line in fh:
258
+ line = line.strip()
259
+ if not line:
260
+ continue
261
+ try:
262
+ obj = json.loads(line)
263
+ except Exception: # noqa: BLE001
264
+ invalid += 1
265
+ continue
266
+ if isinstance(obj, dict):
267
+ records.append(obj)
268
+ else:
269
+ invalid += 1
270
+ return records, invalid
271
+
272
+
273
+ def load_index(dir_path):
274
+ """读取 sessions.json(若有):返回 {别名: 会话ID}。"""
275
+ idx = {}
276
+ p = Path(dir_path) / "sessions.json"
277
+ if not p.exists():
278
+ return idx
279
+ try:
280
+ data = json.loads(p.read_text(encoding="utf-8", errors="replace"))
281
+ except Exception: # noqa: BLE001
282
+ return idx
283
+ if isinstance(data, dict):
284
+ for k, v in data.items():
285
+ idx[str(k)] = str(v)
286
+ elif isinstance(data, list):
287
+ for item in data:
288
+ if isinstance(item, dict):
289
+ sid = item.get("sessionId") or item.get("session_id") or item.get("id")
290
+ key = item.get("key") or item.get("name")
291
+ if sid and key:
292
+ idx[str(key)] = str(sid)
293
+ return idx
294
+
295
+
296
+ def _resolve_session_ids(idx, key):
297
+ """把别名 / 会话 ID 统一解析为候选会话 ID 集合。"""
298
+ ids = {key}
299
+ if key in idx:
300
+ ids.add(idx[key])
301
+ for alias, sid in idx.items():
302
+ if sid == key:
303
+ ids.add(alias)
304
+ ids.add(sid)
305
+ return ids
306
+
307
+
308
+ def _alias_for(idx, sid):
309
+ for alias, value in idx.items():
310
+ if value == sid:
311
+ return alias
312
+ return ""
313
+
314
+
315
+ def _ts_on_date(ts, date):
316
+ if not ts:
317
+ return False
318
+ if len(date) == 10:
319
+ return ts[:10] == date
320
+ if len(date) == 7:
321
+ return ts[:7] == date
322
+ return date in ts
323
+
324
+
325
+ def _clock(ts):
326
+ """从 ISO 时间戳取 HH:MM:SS 片段。"""
327
+ if "T" in ts:
328
+ return ts.split("T", 1)[1][:8]
329
+ if " " in ts:
330
+ return ts.split(" ", 1)[1][:8]
331
+ return ts[:8]
332
+
333
+
334
+ def _human_size(n):
335
+ for unit in ("B", "KB", "MB", "GB"):
336
+ if n < 1024 or unit == "GB":
337
+ return "%d B" % n if unit == "B" else "%.1f %s" % (n, unit)
338
+ n /= 1024.0
339
+ return "%d B" % n
340
+
341
+
342
+ # ── 检索 / 提取 / 统计 ──────────────────────────────────────────────────
343
+
344
+ def scan_sessions(dir_path):
345
+ idx = load_index(dir_path)
346
+ rows = []
347
+ total_messages = 0
348
+ total_invalid = 0
349
+ for info in list_sessions(dir_path):
350
+ records, invalid = parse_jsonl(info["path"])
351
+ total_invalid += invalid
352
+ first_ts = ""
353
+ messages = 0
354
+ for rec in records:
355
+ if not _is_message(rec):
356
+ continue
357
+ messages += 1
358
+ ts = _rec_ts(rec)
359
+ if not first_ts and ts:
360
+ first_ts = ts
361
+ total_messages += messages
362
+ rows.append({
363
+ "session": info["session"],
364
+ "alias": _alias_for(idx, info["session"]),
365
+ "path": info["path"],
366
+ "size": info["size"],
367
+ "date": first_ts[:10],
368
+ "messages": messages,
369
+ "invalid": invalid,
370
+ "mtime": info["mtime"],
371
+ })
372
+ rows.sort(key=lambda r: (r["date"] or r["mtime"]), reverse=True)
373
+ return {
374
+ "dir": str(dir_path),
375
+ "rows": rows,
376
+ "total_sessions": len(rows),
377
+ "total_messages": total_messages,
378
+ "total_invalid": total_invalid,
379
+ }
380
+
381
+
382
+ def search_sessions(dir_path, query, regex=False, date=None, sessions=None,
383
+ role=None, limit=DEFAULT_LIMIT, context=DEFAULT_CONTEXT,
384
+ no_redact=False):
385
+ """跨会话关键词 / 正则检索,返回结构化命中列表。"""
386
+ if regex:
387
+ try:
388
+ pat = re.compile(query, re.I)
389
+ except re.error as e:
390
+ raise SystemExit("正则无效:%s" % e)
391
+ idx = load_index(dir_path)
392
+ sid_filter = None
393
+ if sessions:
394
+ sid_filter = set()
395
+ for g in sessions:
396
+ sid_filter |= _resolve_session_ids(idx, g)
397
+ matches = []
398
+ hit_sids = set()
399
+ truncated = False
400
+ for info in list_sessions(dir_path):
401
+ sid = info["session"]
402
+ if sid_filter is not None and sid not in sid_filter:
403
+ continue
404
+ records, _ = parse_jsonl(info["path"])
405
+ for lineno, rec in enumerate(records, 1):
406
+ if not _is_message(rec):
407
+ continue
408
+ ts = _rec_ts(rec)
409
+ if date and not _ts_on_date(ts, date):
410
+ continue
411
+ rrole = _rec_role(rec)
412
+ if role and rrole != role:
413
+ continue
414
+ text = _rec_text(rec)
415
+ if not text:
416
+ continue
417
+ if regex:
418
+ m = pat.search(text)
419
+ if not m:
420
+ continue
421
+ span = m.span()
422
+ matched = m.group(0)
423
+ else:
424
+ idx_f = text.lower().find(query.lower())
425
+ if idx_f < 0:
426
+ continue
427
+ span = (idx_f, idx_f + len(query))
428
+ matched = query
429
+ snippet = _snippet(text, span, context)
430
+ if not no_redact:
431
+ snippet = redact(snippet)
432
+ matched = redact(matched)
433
+ matches.append({
434
+ "session": sid,
435
+ "timestamp": ts,
436
+ "role": rrole,
437
+ "line": lineno,
438
+ "match": matched,
439
+ "text": snippet,
440
+ })
441
+ hit_sids.add(sid)
442
+ if len(matches) >= limit:
443
+ truncated = True
444
+ return {
445
+ "matches": matches,
446
+ "sessions_hit": len(hit_sids),
447
+ "truncated": truncated,
448
+ }
449
+ return {"matches": matches, "sessions_hit": len(hit_sids),
450
+ "truncated": truncated}
451
+
452
+
453
+ def extract_session(dir_path, session_id, role=None, with_tools=False,
454
+ no_redact=False):
455
+ """提取单个会话原文(时间线 + 角色 + 文本)。"""
456
+ idx = load_index(dir_path)
457
+ info = None
458
+ for sid in _resolve_session_ids(idx, session_id):
459
+ p = Path(dir_path) / (sid + ".jsonl")
460
+ if p.exists() and p.is_file():
461
+ info = {"session": sid, "path": str(p)}
462
+ break
463
+ if info is None:
464
+ p = Path(dir_path) / session_id
465
+ if p.exists() and p.is_file():
466
+ info = {"session": p.stem, "path": str(p)}
467
+ if info is None:
468
+ raise SystemExit("未找到会话:%s(可用 scan 列出会话 ID)" % session_id)
469
+ records, invalid = parse_jsonl(info["path"])
470
+ messages = []
471
+ for lineno, rec in enumerate(records, 1):
472
+ if not _is_message(rec):
473
+ continue
474
+ rrole = _rec_role(rec)
475
+ if role and rrole != role:
476
+ continue
477
+ text = _rec_text(rec)
478
+ tools = _rec_tool_names(rec)
479
+ if not text and not tools:
480
+ continue
481
+ if not no_redact:
482
+ text = redact(text)
483
+ messages.append({
484
+ "line": lineno,
485
+ "timestamp": _rec_ts(rec),
486
+ "role": rrole,
487
+ "text": text,
488
+ "tools": tools,
489
+ })
490
+ return {
491
+ "session": info["session"],
492
+ "dir": str(dir_path),
493
+ "messages": messages,
494
+ "invalid": invalid,
495
+ "total_records": len(records),
496
+ }
497
+
498
+
499
+ def session_stats(dir_path, session_id=None, daily=False):
500
+ """汇总统计:消息 / 角色 / token / 成本 / 时间范围 / 每日汇总。"""
501
+ idx = load_index(dir_path)
502
+ sessions = list_sessions(dir_path)
503
+ if session_id:
504
+ ids = set()
505
+ for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
506
+ ids |= _resolve_session_ids(idx, g)
507
+ sessions = [s for s in sessions if s["session"] in ids]
508
+ agg = {
509
+ "sessions": len(sessions),
510
+ "messages": 0,
511
+ "invalid": 0,
512
+ "roles": {},
513
+ "cost": 0.0,
514
+ "tokens_in": 0,
515
+ "tokens_out": 0,
516
+ "first": "",
517
+ "last": "",
518
+ "days": {},
519
+ }
520
+ for info in sessions:
521
+ records, invalid = parse_jsonl(info["path"])
522
+ agg["invalid"] += invalid
523
+ for rec in records:
524
+ if not _is_message(rec):
525
+ continue
526
+ agg["messages"] += 1
527
+ role = _rec_role(rec)
528
+ agg["roles"][role] = agg["roles"].get(role, 0) + 1
529
+ ts = _rec_ts(rec)
530
+ if ts:
531
+ day = ts[:10]
532
+ if day:
533
+ d = agg["days"].setdefault(day, {"messages": 0, "cost": 0.0})
534
+ d["messages"] += 1
535
+ d["cost"] += _rec_cost(rec)
536
+ if not agg["first"] or ts < agg["first"]:
537
+ agg["first"] = ts
538
+ if not agg["last"] or ts > agg["last"]:
539
+ agg["last"] = ts
540
+ agg["cost"] += _rec_cost(rec)
541
+ ti, to = _rec_tokens(rec)
542
+ agg["tokens_in"] += ti
543
+ agg["tokens_out"] += to
544
+ return agg
545
+
546
+
547
+ def tool_breakdown(dir_path, session_id=None):
548
+ """工具调用次数排行 → [(工具名, 次数)],按次数降序。"""
549
+ idx = load_index(dir_path)
550
+ sessions = list_sessions(dir_path)
551
+ if session_id:
552
+ ids = set()
553
+ for g in (session_id if isinstance(session_id, (list, tuple)) else [session_id]):
554
+ ids |= _resolve_session_ids(idx, g)
555
+ sessions = [s for s in sessions if s["session"] in ids]
556
+ counts = {}
557
+ for info in sessions:
558
+ records, _ = parse_jsonl(info["path"])
559
+ for rec in records:
560
+ for nm in _rec_tool_names(rec):
561
+ counts[nm] = counts.get(nm, 0) + 1
562
+ return sorted(counts.items(), key=lambda kv: (-kv[1], kv[0]))
563
+
564
+
565
+ def _snippet(text, span, radius):
566
+ start, end = span
567
+ lo = max(0, start - radius)
568
+ hi = min(len(text), end + radius)
569
+ pre = "…" if lo > 0 else ""
570
+ post = "…" if hi < len(text) else ""
571
+ return pre + text[lo:hi].replace("\n", " ") + post
572
+
573
+
574
+ # ── 文本格式化 ───────────────────────────────────────────────────────────
575
+
576
+ def fmt_scan(res):
577
+ lines = []
578
+ lines.append("会话日志目录:%s" % res["dir"])
579
+ lines.append("会话 %d 个 | 消息合计 %d | 无效行 %d"
580
+ % (res["total_sessions"], res["total_messages"],
581
+ res["total_invalid"]))
582
+ lines.append("")
583
+ lines.append("%-24s %-12s %8s %10s %s" % ("会话 ID", "日期", "消息", "大小", "别名"))
584
+ for r in res["rows"]:
585
+ lines.append("%-24s %-12s %8d %10s %s"
586
+ % (r["session"][:24], r["date"] or "-",
587
+ r["messages"], _human_size(r["size"]), r["alias"]))
588
+ return "\n".join(lines)
589
+
590
+
591
+ def fmt_search(res, query, dir_path, regex):
592
+ lines = []
593
+ if regex:
594
+ desc = "正则:%s" % query
595
+ else:
596
+ desc = "检索词:%s" % query
597
+ lines.append("匹配 %d 处 / %d 个会话(%s)"
598
+ % (len(res["matches"]), res["sessions_hit"], desc))
599
+ if res["truncated"]:
600
+ lines.append("(已达 --limit 上限,结果被截断)")
601
+ lines.append("")
602
+ cur = None
603
+ for m in res["matches"]:
604
+ if m["session"] != cur:
605
+ cur = m["session"]
606
+ lines.append("── 会话 %s ─────────────────────────" % cur)
607
+ lines.append("%s [%s] %s" % (_clock(m["timestamp"]), m["role"], m["text"]))
608
+ if not res["matches"]:
609
+ lines.append("未命中。")
610
+ return "\n".join(lines)
611
+
612
+
613
+ def fmt_session(res):
614
+ lines = []
615
+ counts = {}
616
+ for m in res["messages"]:
617
+ counts[m["role"]] = counts.get(m["role"], 0) + 1
618
+ parts = " | ".join("%s %d" % (k, v) for k, v in sorted(counts.items()))
619
+ lines.append("会话:%s | 消息 %d(%s)| 无效行 %d"
620
+ % (res["session"], len(res["messages"]), parts, res["invalid"]))
621
+ lines.append("")
622
+ for m in res["messages"]:
623
+ lines.append("── %s ─────────────────────────" % _clock(m["timestamp"]))
624
+ tag = "[%s]" % m["role"]
625
+ if m["tools"]:
626
+ tag += " 工具:%s" % ",".join(m["tools"])
627
+ lines.append("%s %s" % (tag, m["text"]))
628
+ lines.append("")
629
+ return "\n".join(lines)
630
+
631
+
632
+ def fmt_stats(res, daily):
633
+ lines = []
634
+ lines.append("会话统计")
635
+ lines.append("目录:%s" % res.get("dir", ""))
636
+ roles = " | ".join("%s %d" % (k, v)
637
+ for k, v in sorted(res["roles"].items()))
638
+ lines.append("会话 %d | 消息 %d(%s)| 无效行 %d"
639
+ % (res["sessions"], res["messages"], roles, res["invalid"]))
640
+ if res["first"]:
641
+ lines.append("时间范围 %s → %s" % (res["first"][:19], res["last"][:19]))
642
+ lines.append("token 输入 %s | 输出 %s | 成本 $%.2f"
643
+ % (_fmt_int(res["tokens_in"]), _fmt_int(res["tokens_out"]),
644
+ res["cost"]))
645
+ if daily and res["days"]:
646
+ lines.append("")
647
+ lines.append("── 每日汇总 ──")
648
+ for day in sorted(res["days"], reverse=True):
649
+ d = res["days"][day]
650
+ lines.append("%s 消息 %d | 成本 $%.2f" % (day, d["messages"], d["cost"]))
651
+ return "\n".join(lines)
652
+
653
+
654
+ def _fmt_int(n):
655
+ return format(n, ",")
656
+
657
+
658
+ def fmt_tools(items):
659
+ lines = []
660
+ lines.append("工具调用排行(按次数降序)")
661
+ for name, count in items:
662
+ lines.append("%6d %s" % (count, name))
663
+ if not items:
664
+ lines.append("(无工具调用记录)")
665
+ return "\n".join(lines)
666
+
667
+
668
+ # ── CLI ──────────────────────────────────────────────────────────────────
669
+
670
+ class _Parser(argparse.ArgumentParser):
671
+ def error(self, message):
672
+ self.print_usage(sys.stderr)
673
+ self._print_message("%s: error: %s\n" % (self.prog, message), sys.stderr)
674
+ raise SystemExit(4)
675
+
676
+
677
+ def _add_dir(ap):
678
+ ap.add_argument("--dir", metavar="PATH",
679
+ help="会话日志目录(缺省读 YOTTA_LOGS_DIR,再自动定位)")
680
+
681
+
682
+ def resolve_dir(args):
683
+ if getattr(args, "dir", None):
684
+ d = Path(args.dir)
685
+ else:
686
+ env = os.environ.get("YOTTA_LOGS_DIR")
687
+ if env:
688
+ d = Path(env)
689
+ else:
690
+ found = discover_dirs()
691
+ if not found:
692
+ raise SystemExit(
693
+ "未找到会话日志目录:用 --dir 指定,或设 YOTTA_LOGS_DIR;"
694
+ "可用 locate 查看候选。")
695
+ d = Path(found[0])
696
+ if not d.is_dir():
697
+ raise SystemExit("目录不存在:%s" % d)
698
+ return d
699
+
700
+
701
+ def main(argv=None):
702
+ ap = _Parser(
703
+ prog=TOOL_NAME,
704
+ description="%s(%s):零依赖跨智能体会话日志检索引擎。" % (TOOL_CN, TOOL_NAME))
705
+ ap.add_argument("--version", action="version",
706
+ version="%s %s" % (TOOL_NAME, VERSION))
707
+ sub = ap.add_subparsers(dest="command", required=True)
708
+
709
+ sub.add_parser("version", help="打印版本")
710
+
711
+ p_locate = sub.add_parser("locate", help="自动发现本机会话日志目录")
712
+ p_locate.add_argument("--json", action="store_true", help="输出 JSON")
713
+
714
+ p_scan = sub.add_parser("scan", help="列出目录下所有会话")
715
+ _add_dir(p_scan)
716
+ p_scan.add_argument("--json", action="store_true", help="输出 JSON")
717
+ p_scan.add_argument("--limit", type=int, default=0,
718
+ help="最多列出 N 个会话(默认全部)")
719
+
720
+ p_search = sub.add_parser("search", help="跨会话检索关键词 / 正则")
721
+ p_search.add_argument("query", help="检索词(默认不区分大小写)")
722
+ _add_dir(p_search)
723
+ p_search.add_argument("--regex", action="store_true", help="把 query 当正则")
724
+ p_search.add_argument("--date", metavar="YYYY-MM-DD",
725
+ help="只检索指定日期(或 YYYY-MM)")
726
+ p_search.add_argument("-s", "--session", action="append", metavar="SID",
727
+ help="只检索指定会话 ID / 别名(可多次)")
728
+ p_search.add_argument("--role", choices=("user", "assistant", "tool", "system"),
729
+ help="只检索指定角色")
730
+ p_search.add_argument("--limit", type=int, default=DEFAULT_LIMIT,
731
+ help="最多返回 N 条命中(默认 %d)" % DEFAULT_LIMIT)
732
+ p_search.add_argument("--context", type=int, default=DEFAULT_CONTEXT,
733
+ help="命中上下文半径字符数(默认 %d)" % DEFAULT_CONTEXT)
734
+ p_search.add_argument("--json", action="store_true", help="输出 JSON")
735
+ p_search.add_argument("--no-redact", action="store_true",
736
+ help="关闭默认脱敏")
737
+
738
+ p_session = sub.add_parser("session", help="提取单个会话原文")
739
+ p_session.add_argument("sid", help="会话 ID 或 sessions.json 里的别名")
740
+ _add_dir(p_session)
741
+ p_session.add_argument("--role", choices=("user", "assistant", "tool", "system"),
742
+ help="只提取指定角色")
743
+ p_session.add_argument("--tools", action="store_true",
744
+ help="时间线里标注工具调用")
745
+ p_session.add_argument("--limit", type=int, default=0,
746
+ help="最多提取 N 条消息(默认全部)")
747
+ p_session.add_argument("--json", action="store_true", help="输出 JSON")
748
+ p_session.add_argument("--no-redact", action="store_true",
749
+ help="关闭默认脱敏")
750
+
751
+ p_stats = sub.add_parser("stats", help="会话统计汇总")
752
+ _add_dir(p_stats)
753
+ p_stats.add_argument("-s", "--session", metavar="SID",
754
+ help="只统计指定会话 ID / 别名")
755
+ p_stats.add_argument("--daily", action="store_true", help="输出每日汇总")
756
+ p_stats.add_argument("--json", action="store_true", help="输出 JSON")
757
+
758
+ p_tools = sub.add_parser("tools", help="工具调用次数排行")
759
+ _add_dir(p_tools)
760
+ p_tools.add_argument("-s", "--session", metavar="SID",
761
+ help="只统计指定会话 ID / 别名")
762
+ p_tools.add_argument("--json", action="store_true", help="输出 JSON")
763
+
764
+ args = ap.parse_args(argv)
765
+ try:
766
+ if args.command == "version":
767
+ print("%s %s" % (TOOL_NAME, VERSION))
768
+ return 0
769
+
770
+ if args.command == "locate":
771
+ found = discover_dirs()
772
+ if args.json:
773
+ print(json.dumps({"dirs": found}, ensure_ascii=False, indent=2))
774
+ elif found:
775
+ for d in found:
776
+ print(d)
777
+ else:
778
+ print("未发现已知会话日志目录。")
779
+ return 1
780
+ return 0
781
+
782
+ if args.command == "scan":
783
+ d = resolve_dir(args)
784
+ res = scan_sessions(d)
785
+ if args.limit > 0:
786
+ res["rows"] = res["rows"][:args.limit]
787
+ res["total_sessions"] = len(res["rows"])
788
+ if args.json:
789
+ print(json.dumps(res, ensure_ascii=False, indent=2))
790
+ else:
791
+ print(fmt_scan(res))
792
+ return 0 if res["rows"] else 1
793
+
794
+ if args.command == "search":
795
+ d = resolve_dir(args)
796
+ res = search_sessions(
797
+ d, args.query, regex=args.regex, date=args.date,
798
+ sessions=args.session, role=args.role, limit=args.limit,
799
+ context=args.context, no_redact=args.no_redact)
800
+ if args.json:
801
+ print(json.dumps({
802
+ "command": "search",
803
+ "tool": TOOL_NAME,
804
+ "version": VERSION,
805
+ "query": args.query,
806
+ "regex": args.regex,
807
+ "dir": str(d),
808
+ "total_matches": len(res["matches"]),
809
+ "sessions_hit": res["sessions_hit"],
810
+ "truncated": res["truncated"],
811
+ "matches": res["matches"],
812
+ }, ensure_ascii=False, indent=2))
813
+ else:
814
+ print(fmt_search(res, args.query, d, args.regex))
815
+ return 0 if res["matches"] else 1
816
+
817
+ if args.command == "session":
818
+ d = resolve_dir(args)
819
+ res = extract_session(d, args.sid, role=args.role,
820
+ with_tools=args.tools,
821
+ no_redact=args.no_redact)
822
+ if args.limit > 0:
823
+ res["messages"] = res["messages"][:args.limit]
824
+ if args.json:
825
+ print(json.dumps(res, ensure_ascii=False, indent=2))
826
+ else:
827
+ print(fmt_session(res))
828
+ return 0
829
+
830
+ if args.command == "stats":
831
+ d = resolve_dir(args)
832
+ res = session_stats(d, session_id=args.session, daily=args.daily)
833
+ res["dir"] = str(d)
834
+ if args.json:
835
+ print(json.dumps(res, ensure_ascii=False, indent=2))
836
+ else:
837
+ print(fmt_stats(res, args.daily))
838
+ return 0 if res["sessions"] else 1
839
+
840
+ if args.command == "tools":
841
+ d = resolve_dir(args)
842
+ items = tool_breakdown(d, session_id=args.session)
843
+ if args.json:
844
+ print(json.dumps({
845
+ "command": "tools",
846
+ "tool": TOOL_NAME,
847
+ "version": VERSION,
848
+ "dir": str(d),
849
+ "tools": [{"name": n, "count": c} for n, c in items],
850
+ }, ensure_ascii=False, indent=2))
851
+ else:
852
+ print(fmt_tools(items))
853
+ return 0
854
+ except BrokenPipeError:
855
+ # 管道被提前关闭(如 scan | head):静默收尾,不算错误
856
+ try:
857
+ sys.stdout.close()
858
+ except Exception:
859
+ pass
860
+ return 0
861
+ except SystemExit as e:
862
+ code = e.code if isinstance(e.code, int) else 4
863
+ msg = e.code if isinstance(e.code, str) else None
864
+ if msg:
865
+ print(msg, file=sys.stderr)
866
+ return code if code in (0, 4) else 4
867
+ except Exception as e: # noqa: BLE001
868
+ print("错误:%s" % e, file=sys.stderr)
869
+ return 4
870
+ return 4
871
+
872
+
873
+ if __name__ == "__main__":
874
+ try:
875
+ sys.exit(main())
876
+ except SystemExit:
877
+ raise