@yottameta/yotta-logs 0.1.0 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2,7 +2,7 @@
2
2
 
3
3
  ## 只读保证
4
4
 
5
- - 引擎只调用读取类操作(`open(path, "r")`),对会话目录不做任何写入 / 修改 / 删除;
5
+ - 引擎只调用读取类操作(`open(path, "r")`、SQLite `file:<path>?mode=ro` 只读 URI),对任何日志 / 记忆文件与数据库不做写入 / 修改 / 删除;
6
6
  - 检索、提取、统计全程无网络请求;不把日志内容上传到任何服务;
7
7
  - 测试含只读回归:跑完 scan / search / session / stats / tools 后目录文件清单与大小不变。
8
8
 
@@ -26,17 +26,18 @@ URL 路径(非凭据)原文保留,方便回溯链接。
26
26
 
27
27
  ## 边界
28
28
 
29
- - 只检索 `*.jsonl` 会话文件与 `sessions.json` 索引;不读取、不分析其它文件;
30
- - 检索范围限定用户显式传入的 `--dir`(或环境变量 / 自动定位结果),不主动扫描磁盘;
31
- - 输出可能包含会话原文中的隐私,默认脱敏且仅用于本机回溯,请勿外传。
29
+ - 只检索显式传入的 `--dir`(或环境变量 / discover 登记的来源),不主动扫描整盘;
30
+ - 默认检索范围只含会话源 + 结构化记忆源;自由笔记(kind=note)与二进制日志(kind=log)默认不读,需显式 `--kind note / --kind log` 或配置 default_scope 开启;
31
+ - discover 只登记已知根(见 agent-formats.md 第四节),命中即登记来源,不展开内容;
32
+ - 输出可能包含会话 / 记忆原文中的隐私,默认脱敏且仅用于本机回溯,请勿外传。
32
33
 
33
34
  ## 与元忆(yotta-memory)的分工
34
35
 
35
36
  | 维度 | yotta-logs(元史) | yotta-memory(元忆) |
36
37
  |---|---|---|
37
- | 定位 | 原始会话日志(JSONL 事实) | 语义记忆(结构化条目) |
38
- | 输出 | 原文片段 + 行号 + 时间戳 | 记忆条目 + 权限边界 + 画像 |
38
+ | 定位 | 原始会话日志 / 记忆文件(JSONL / JSON / SQLite / Markdown 事实) | 语义记忆(结构化条目 + 权限边界) |
39
+ | 输出 | 原文片段 + 行号 + 时间戳 + 来源 | 记忆条目 + 权限边界 + 画像 |
39
40
  | 写操作 | 无(只读) | 支持(remember / forget / archive) |
40
- | 权限 | 目录级只读 | 类型 / 属主级权限边界 |
41
+ | 权限 | 目录 / 来源级只读 | 类型 / 属主级权限边界 |
41
42
 
42
43
  回溯「原文」用元史;沉淀「长期知识 / 偏好 / 承诺」用元忆,二者互补。
@@ -374,6 +374,348 @@ def test_readonly(fx):
374
374
  "before=%s after=%s" % (before, after))
375
375
 
376
376
 
377
+ # ── v0.2.0 通用化测试 ────────────────────────────────────────────────────
378
+
379
+ def build_json_fixture(base):
380
+ """单文件 JSON 会话:一个数组文件 + 一个 dict-of-lists 文件。"""
381
+ d = base / "jsons"
382
+ d.mkdir(parents=True, exist_ok=True)
383
+ arr = [
384
+ {"role": "user", "content": "你好,单文件 JSON 测试。", "ts": "2026-08-26T09:00:00+08:00"},
385
+ {"role": "assistant", "content": "收到。", "ts": "2026-08-26T09:01:00+08:00"},
386
+ ]
387
+ (d / "convo.json").write_text(json.dumps(arr, ensure_ascii=False),
388
+ encoding="utf-8")
389
+ multi = {
390
+ "s1": [{"role": "user", "content": "s1 的第一个问题"},
391
+ {"role": "assistant", "content": "s1 的回复"}],
392
+ "s2": [{"role": "user", "content": "s2 的问题"}],
393
+ }
394
+ (d / "multi.json").write_text(json.dumps(multi, ensure_ascii=False),
395
+ encoding="utf-8")
396
+ return d
397
+
398
+
399
+ def build_sqlite_opencode(p):
400
+ import sqlite3 as _sq
401
+ con = _sq.connect(str(p))
402
+ con.execute("CREATE TABLE session (id TEXT, title TEXT, time_created INTEGER,"
403
+ " cost REAL, tokens_input INTEGER, tokens_output INTEGER)")
404
+ con.execute("CREATE TABLE message (id TEXT, session_id TEXT, time_created INTEGER, data TEXT)")
405
+ con.execute("CREATE TABLE part (id TEXT, message_id TEXT, session_id TEXT,"
406
+ " time_created INTEGER, data TEXT)")
407
+ con.execute("INSERT INTO session VALUES ('ses_a','部署讨论',1785834902916,0.05,100,50)")
408
+ con.execute("INSERT INTO session VALUES ('ses_b','CI 排查',1785835000000,0.0,0,0)")
409
+ con.execute("INSERT INTO message VALUES ('msg_1','ses_a',1785834903000,'{\"role\":\"user\"}')")
410
+ con.execute("INSERT INTO part VALUES ('prt_1','msg_1','ses_a',1785834903001,"
411
+ "'{\"type\":\"text\",\"text\":\"你好,部署方案定了吗?\"}')")
412
+ con.execute("INSERT INTO message VALUES ('msg_2','ses_a',1785834904000,'{\"role\":\"assistant\"}')")
413
+ con.execute("INSERT INTO part VALUES ('prt_2','msg_2','ses_a',1785834904001,"
414
+ "'{\"type\":\"text\",\"text\":\"定了,按灰度发布执行。\"}')")
415
+ con.execute("INSERT INTO part VALUES ('prt_3','msg_2','ses_a',1785834904002,"
416
+ "'{\"type\":\"tool\",\"tool\":\"read_file\"}')")
417
+ con.execute("INSERT INTO part VALUES ('prt_4','msg_2','ses_a',1785834904003,"
418
+ "'{\"type\":\"reasoning\",\"text\":\"隐藏推理不输出\"}')")
419
+ con.execute("INSERT INTO message VALUES ('msg_3','ses_b',1785835001000,'{\"role\":\"user\"}')")
420
+ con.execute("INSERT INTO part VALUES ('prt_5','msg_3','ses_b',1785835001001,"
421
+ "'{\"type\":\"text\",\"text\":\"CI 又失败了,看下日志。\"}')")
422
+ con.commit()
423
+ con.close()
424
+
425
+
426
+ def build_sqlite_generic(p):
427
+ import sqlite3 as _sq
428
+ con = _sq.connect(str(p))
429
+ con.execute("CREATE TABLE messages (id INTEGER, session_id TEXT, role TEXT,"
430
+ " content TEXT, created_at TEXT)")
431
+ con.execute("INSERT INTO messages VALUES (1,'g1','user','泛型表问题甲','2026-08-26T03:00:00+08:00')")
432
+ con.execute("INSERT INTO messages VALUES (2,'g1','assistant','泛型表回复甲','2026-08-26T03:01:00+08:00')")
433
+ con.execute("INSERT INTO messages VALUES (3,'g2','user','乙的问题','2026-08-27T10:00:00+08:00')")
434
+ con.commit()
435
+ con.close()
436
+
437
+
438
+ def build_md_fixture(base):
439
+ facts = base / ".yottamemory" / "facts"
440
+ facts.mkdir(parents=True, exist_ok=True)
441
+ (facts / "2026-08-25-0002.md").write_text(
442
+ "---\ntype: FACT\nsubject: 共享记忆引擎接入指南\n"
443
+ "statement: 本机运行 yotta-memory 记忆引擎,接入方式见正文。\n"
444
+ "confidence: 1\ncreated: 2026-08-25\ntags: [memory, guide]\n---\n"
445
+ "正文补充。\n", encoding="utf-8")
446
+ notes = base / ".CodexData" / "memories"
447
+ notes.mkdir(parents=True, exist_ok=True)
448
+ (notes / "note.md").write_text(
449
+ "# 推送闸门红线\n\n规则:测试通过才能推。\n", encoding="utf-8")
450
+ return facts, notes
451
+
452
+
453
+ def test_norm_time():
454
+ check("毫秒转 ISO", YL._norm_time(1785834903000)
455
+ .startswith("20") and "T" in YL._norm_time(1785834903000),
456
+ YL._norm_time(1785834903000))
457
+ check("秒时间戳", YL._norm_time(1785834903).startswith("20"),
458
+ YL._norm_time(1785834903))
459
+ check("ISO 原样", YL._norm_time("2026-08-26T03:00:00+08:00")
460
+ == "2026-08-26T03:00:00+08:00")
461
+ check("日期原样", YL._norm_time("2026-08-25") == "2026-08-25")
462
+ check("Z 归一", YL._norm_time("2026-08-26T03:00:00Z")
463
+ == "2026-08-26T03:00:00+00:00")
464
+
465
+
466
+ def test_json_reader():
467
+ with tempfile.TemporaryDirectory() as td:
468
+ d = build_json_fixture(Path(td))
469
+ src = YL.sniff_source(str(d))
470
+ check("JSON 目录嗅探", src["format"] == "json" and src["kind"] == "session",
471
+ str(src))
472
+ reader = YL.JSONReader()
473
+ rows, tm, inv = reader.iter_sessions(src)
474
+ check("JSON scan 3 会话", len(rows) == 3 and tm == 5, str(rows))
475
+ r = YL.search_all([src], "单文件")
476
+ check("JSON search 命中", len(r["matches"]) == 1
477
+ and r["matches"][0]["session"] == "convo", str(r))
478
+ r2 = YL.search_all([src], "s2")
479
+ check("JSON dict 会话检索", len(r2["matches"]) == 1
480
+ and r2["matches"][0]["session"] == "s2", str(r2))
481
+ ex = YL.extract_all([src], "convo")
482
+ check("JSON extract", len(ex["messages"]) == 2
483
+ and ex["messages"][0]["role"] == "user", str(ex))
484
+ st = YL.stats_all([src])
485
+ check("JSON stats 消息 5", st["messages"] == 5, str(st["messages"]))
486
+ check("JSON stats 角色", st["roles"].get("user") == 3, str(st["roles"]))
487
+
488
+
489
+ def test_sqlite_opencode_reader():
490
+ with tempfile.TemporaryDirectory() as td:
491
+ p = Path(td) / "opencode.db"
492
+ build_sqlite_opencode(p)
493
+ src = YL.sniff_source(str(p))
494
+ check("SQLite 文件嗅探", src["format"] == "sqlite", str(src))
495
+ reader = YL.SQLiteReader()
496
+ rows, tm, inv = reader.iter_sessions(src)
497
+ check("opencode scan 2 会话", len(rows) == 2, str(rows))
498
+ by = {r["session"]: r for r in rows}
499
+ check("opencode ses_a 消息 2", by["ses_a"]["messages"] == 2,
500
+ str(by["ses_a"]))
501
+ check("opencode 毫秒日期", by["ses_a"]["date"].startswith("20"),
502
+ by["ses_a"]["date"])
503
+ r = YL.search_all([src], "灰度")
504
+ check("opencode search 命中", len(r["matches"]) == 1
505
+ and r["matches"][0]["session"] == "ses_a", str(r))
506
+ r2 = YL.search_all([src], "隐藏推理")
507
+ check("opencode reasoning 不进文本", r2["matches"] == [], str(r2))
508
+ ex = YL.extract_all([src], "ses_a")
509
+ check("opencode extract 消息 2", len(ex["messages"]) == 2, str(ex))
510
+ msg2 = ex["messages"][1]
511
+ check("opencode 工具标注", msg2["tools"] == ["read_file"], str(msg2))
512
+ check("opencode role assistant", msg2["role"] == "assistant", str(msg2))
513
+ st = YL.stats_all([src])
514
+ check("opencode stats 消息 3", st["messages"] == 3, str(st["messages"]))
515
+ check("opencode stats 角色", st["roles"] == {"user": 2, "assistant": 1},
516
+ str(st["roles"]))
517
+ items = YL.tools_all([src])
518
+ check("opencode tools read_file 1", dict(items).get("read_file") == 1,
519
+ str(items))
520
+
521
+
522
+ def test_sqlite_generic_reader():
523
+ with tempfile.TemporaryDirectory() as td:
524
+ p = Path(td) / "app.db"
525
+ build_sqlite_generic(p)
526
+ src = YL._mk_source("generic-test", "session", "sqlite", p,
527
+ extra={"table": "messages"})
528
+ reader = YL.SQLiteReader()
529
+ rows, tm, inv = reader.iter_sessions(src)
530
+ by = {r["session"]: r for r in rows}
531
+ check("generic scan 2 会话", len(rows) == 2 and tm == 3, str(rows))
532
+ check("generic g1 消息 2", by["g1"]["messages"] == 2, str(by["g1"]))
533
+ r = YL.search_all([src], "甲")
534
+ check("generic search 命中 2", len(r["matches"]) == 2, str(r))
535
+ ex = YL.extract_all([src], "g1")
536
+ check("generic extract 2 条", len(ex["messages"]) == 2, str(ex))
537
+ st = YL.stats_all([src])
538
+ check("generic stats 角色", st["roles"] == {"user": 2, "assistant": 1},
539
+ str(st["roles"]))
540
+
541
+
542
+ def test_markdown_reader():
543
+ with tempfile.TemporaryDirectory() as td:
544
+ facts, notes = build_md_fixture(Path(td))
545
+ fsrc = YL._mk_source("yottamemory-facts", "memory", "markdown", facts)
546
+ nsrc = YL._mk_source("codex-notes", "note", "markdown", notes)
547
+ fr = list(YL.MarkdownReader().iter_records(fsrc))
548
+ check("md memory 1 条", len(fr) == 1, str(fr))
549
+ rec = fr[0]
550
+ check("md role FACT", rec["role"] == "FACT", str(rec["role"]))
551
+ check("md title subject", rec["meta"].get("title") == "共享记忆引擎接入指南",
552
+ str(rec["meta"]))
553
+ check("md text statement", "yotta-memory" in rec["text"], rec["text"][:50])
554
+ check("md created 时间", rec["time"] == "2026-08-25", rec["time"])
555
+ nr = list(YL.MarkdownReader().iter_records(nsrc))
556
+ check("md note 1 条", len(nr) == 1 and nr[0]["kind"] == "note", str(nr))
557
+ check("md note 标题", nr[0]["meta"].get("title") == "推送闸门红线",
558
+ str(nr[0]["meta"]))
559
+ check("md note 正文", "测试通过" in nr[0]["text"], nr[0]["text"][:40])
560
+ rows, tm, inv = YL.MarkdownReader().iter_sessions(fsrc)
561
+ check("md memory scan", len(rows) == 1 and tm == 1, str(rows))
562
+ # 结构化 md 文件单独 --dir → kind memory
563
+ fp = facts / "2026-08-25-0002.md"
564
+ s = YL.sniff_source(str(fp))
565
+ check("md 文件嗅探 memory", s["format"] == "markdown" and s["kind"] == "memory",
566
+ str(s))
567
+ np = notes / "note.md"
568
+ s2 = YL.sniff_source(str(np))
569
+ check("md 文件嗅探 note", s2["kind"] == "note", str(s2))
570
+
571
+
572
+ def test_binary_reader():
573
+ with tempfile.TemporaryDirectory() as td:
574
+ p = Path(td) / "conv.pbtxt"
575
+ p.write_bytes(b"\x00\x01Conversation WindSurf Title\x00\x02more")
576
+ src = YL._mk_source("windsurf-conv", "log", "binary", p, default_on=False)
577
+ recs = list(YL.BinaryReader().iter_records(src))
578
+ check("binary 1 条且不崩", len(recs) == 1, str(recs))
579
+ check("binary title 提取", "WindSurf" in recs[0]["text"], recs[0]["text"])
580
+ check("binary kind log", recs[0]["kind"] == "log", str(recs[0]["kind"]))
581
+ rows, tm, inv = YL.BinaryReader().iter_sessions(src)
582
+ check("binary scan", len(rows) == 1 and tm == 1, str(rows))
583
+
584
+
585
+ def test_sniff_and_discover():
586
+ with tempfile.TemporaryDirectory() as td:
587
+ base = Path(td)
588
+ # discover:JSONL + SQLite(opencode) + 记忆 md + 自由笔记
589
+ jd = base / ".codex" / "sessions"
590
+ jd.mkdir(parents=True, exist_ok=True)
591
+ (jd / "a.jsonl").write_text('{"type":"message","message":{"role":"user","content":"hi"}}\n',
592
+ encoding="utf-8")
593
+ db = base / ".local" / "share" / "opencode" / "opencode.db"
594
+ db.parent.mkdir(parents=True, exist_ok=True)
595
+ build_sqlite_opencode(db)
596
+ facts, notes = build_md_fixture(base)
597
+ jsrcs = YL.JSONLReader.discover(base)
598
+ check("discover JSONL 命中 codex", any(s["name"] == "codex-sessions"
599
+ for s in jsrcs), str(jsrcs))
600
+ ssrcs = YL.SQLiteReader.discover(base)
601
+ check("discover SQLite 命中 opencode", any(s["name"] == "opencode-db"
602
+ for s in ssrcs), str(ssrcs))
603
+ msrcs = YL.MarkdownReader.discover(base)
604
+ names = {s["name"] for s in msrcs}
605
+ check("discover md 命中 yottamemory-facts", "yottamemory-facts" in names,
606
+ str(msrcs))
607
+ check("discover md 命中 codex-notes", "codex-notes" in names, str(msrcs))
608
+ for s in msrcs:
609
+ if s["name"] == "codex-notes":
610
+ check("自由笔记默认关", s["default_on"] is False, str(s))
611
+ if s["name"] == "yottamemory-facts":
612
+ check("结构化记忆默认开", s["default_on"] is True, str(s))
613
+ # 配置兜底
614
+ cfg = {"sources": [{"name": "myapp", "path": str(base / "app.db"),
615
+ "format": "sqlite", "kind": "session",
616
+ "table": "messages", "col_text": "content"}]}
617
+ srcs = YL.discover_sources(cfg)
618
+ check("配置源登记", any(s["name"] == "myapp" for s in srcs), str(srcs))
619
+
620
+
621
+ def test_filters_and_scope():
622
+ srcs = [
623
+ YL._mk_source("sess", "session", "jsonl", "/x", True),
624
+ YL._mk_source("mem", "memory", "markdown", "/y", True),
625
+ YL._mk_source("note", "note", "markdown", "/z", False),
626
+ ]
627
+
628
+ class A:
629
+ source = None
630
+ kind = None
631
+ format = None
632
+
633
+ a = A()
634
+ out = YL.filter_sources(srcs, a)
635
+ check("默认范围排除 note", [s["name"] for s in out] == ["sess", "mem"],
636
+ str(out))
637
+ a.kind = "note"
638
+ out = YL.filter_sources(srcs, a)
639
+ check("--kind note 显式开", [s["name"] for s in out] == ["note"], str(out))
640
+ a.kind = None
641
+ a.format = "markdown"
642
+ out = YL.filter_sources(srcs, a)
643
+ check("--format markdown 显式含 note", {s["name"] for s in out} == {"mem", "note"},
644
+ str(out))
645
+ a.format = None
646
+ a.source = ["sess"]
647
+ out = YL.filter_sources(srcs, a)
648
+ check("--source 过滤", [s["name"] for s in out] == ["sess"], str(out))
649
+
650
+
651
+ def test_cross_source_scope():
652
+ with tempfile.TemporaryDirectory() as td:
653
+ base = Path(td)
654
+ # 用自己的 fixture:jsonl 会话 + 记忆 md + 自由笔记 都含关键词
655
+ jd = base / "sessions"
656
+ jd.mkdir(parents=True, exist_ok=True)
657
+ (jd / "s1.jsonl").write_text(
658
+ '{"type":"message","timestamp":"2026-08-26T03:00:00+08:00","message":{"role":"user","content":"部署方案定了吗"}}\n',
659
+ encoding="utf-8")
660
+ facts, notes = build_md_fixture(base)
661
+ (facts / "m1.md").write_text(
662
+ "---\ntype: FACT\nsubject: 部署\nstatement: 部署方案已拍板。\ncreated: 2026-08-25\n---\n",
663
+ encoding="utf-8")
664
+ (notes / "n1.md").write_text("# 部署笔记\n\n部署草稿。\n", encoding="utf-8")
665
+ jsrc = YL.sniff_source(str(jd))
666
+ fsrc = YL._mk_source("facts", "memory", "markdown", facts)
667
+ nsrc = YL._mk_source("notes", "note", "markdown", notes, default_on=False)
668
+ srcs = [jsrc, fsrc, nsrc]
669
+ # 默认:会话 + 记忆,排除自由笔记
670
+ default = YL.filter_sources(srcs, type("A", (), {"source": None,
671
+ "kind": None,
672
+ "format": None})())
673
+ res = YL.search_all(default, "部署")
674
+ names = {(m["source"], m["session"]) for m in res["matches"]}
675
+ check("默认范围不含自由笔记", ("notes", "n1") not in names, str(names))
676
+ check("默认范围含会话与记忆",
677
+ ("sessions", "s1") in names and ("facts", "m1") in names, str(names))
678
+ # 显式开 note
679
+ res2 = YL.search_all([nsrc], "部署")
680
+ check("显式开 note 可检索", ("notes", "n1") in
681
+ {(m["source"], m["session"]) for m in res2["matches"]}, str(res2))
682
+
683
+
684
+ def test_cli_v020(fx):
685
+ # --kind / --format / --source 参数存在且 --dir 行为不变
686
+ r = _run(["search", "部署", "--dir", str(fx), "--format", "jsonl"])
687
+ check("CLI --format jsonl", r.returncode == 0 and "部署方案" in r.stdout,
688
+ "rc=%d" % r.returncode)
689
+ r = _run(["search", "部署", "--dir", str(fx), "--kind", "session"])
690
+ check("CLI --kind session", r.returncode == 0, "rc=%d" % r.returncode)
691
+ r = _run(["scan", "--dir", str(fx), "--source", "nope"])
692
+ check("CLI --source 无命中退出码 1", r.returncode == 1,
693
+ "rc=%d" % r.returncode)
694
+ # locate --json 结构
695
+ r = _run(["locate", "--json"])
696
+ try:
697
+ obj = json.loads(r.stdout)
698
+ check("CLI locate --json", r.returncode == 0
699
+ and "sources" in obj and "default_scope" in obj, r.stdout[:120])
700
+ except Exception as e: # noqa: BLE001
701
+ check("CLI locate --json", False, str(e))
702
+ # 单文件 md --dir(自由笔记显式可查)
703
+ with tempfile.TemporaryDirectory() as td:
704
+ np = Path(td) / "note.md"
705
+ np.write_text("# 标题甲\n\n正文含关键词乙。\n", encoding="utf-8")
706
+ r = _run(["search", "关键词乙", "--dir", str(np)])
707
+ check("CLI 单 md 文件检索", r.returncode == 0 and "关键词乙" in r.stdout,
708
+ "rc=%d out=%s" % (r.returncode, r.stdout[:80]))
709
+ r = _run(["scan", "--dir", str(np), "--json"])
710
+ try:
711
+ obj = json.loads(r.stdout)
712
+ check("CLI 单 md scan", r.returncode == 0
713
+ and obj["total_sessions"] == 1, r.stdout[:120])
714
+ except Exception as e: # noqa: BLE001
715
+ check("CLI 单 md scan", False, str(e))
716
+
717
+
718
+
377
719
  def main():
378
720
  print("元史(yotta-logs)测试开始…")
379
721
  with tempfile.TemporaryDirectory() as td:
@@ -391,6 +733,16 @@ def main():
391
733
  test_cli(fx)
392
734
  test_gbk_console(fx)
393
735
  test_readonly(fx)
736
+ test_norm_time()
737
+ test_json_reader()
738
+ test_sqlite_opencode_reader()
739
+ test_sqlite_generic_reader()
740
+ test_markdown_reader()
741
+ test_binary_reader()
742
+ test_sniff_and_discover()
743
+ test_filters_and_scope()
744
+ test_cross_source_scope()
745
+ test_cli_v020(fx)
394
746
  print("")
395
747
  print("通过 %d 项,失败 %d 项" % (PASS, FAIL))
396
748
  if FAILED:
@@ -400,5 +752,6 @@ def main():
400
752
  sys.exit(1 if FAIL else 0)
401
753
 
402
754
 
755
+
403
756
  if __name__ == "__main__":
404
757
  main()