memor-cli 0.4.1__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (99) hide show
  1. {memor_cli-0.4.1/memor_cli.egg-info → memor_cli-0.5.0}/PKG-INFO +3 -2
  2. {memor_cli-0.4.1 → memor_cli-0.5.0}/README.md +2 -1
  3. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/cli.py +36 -3
  4. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/server.py +13 -0
  5. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/static/index.html +89 -1
  6. memor_cli-0.5.0/memor/eval/counterfactual.py +176 -0
  7. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/store/sqlite_store.py +50 -0
  8. {memor_cli-0.4.1 → memor_cli-0.5.0/memor_cli.egg-info}/PKG-INFO +3 -2
  9. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/SOURCES.txt +3 -0
  10. {memor_cli-0.4.1 → memor_cli-0.5.0}/pyproject.toml +1 -1
  11. memor_cli-0.5.0/tests/test_counterfactual.py +104 -0
  12. memor_cli-0.5.0/tests/test_roi_trend.py +87 -0
  13. {memor_cli-0.4.1 → memor_cli-0.5.0}/LICENSE +0 -0
  14. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/__init__.py +0 -0
  15. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/daemon.py +0 -0
  16. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/__init__.py +0 -0
  17. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/__init__.py +0 -0
  18. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/distiller.py +0 -0
  19. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/extractive.py +0 -0
  20. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/__init__.py +0 -0
  21. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/api.py +0 -0
  22. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/fake.py +0 -0
  23. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/local.py +0 -0
  24. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/__init__.py +0 -0
  25. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/__init__.py +0 -0
  26. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/base.py +0 -0
  27. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/claude_mem.py +0 -0
  28. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/graphiti.py +0 -0
  29. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/dataset.py +0 -0
  30. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/embed_benchmark.py +0 -0
  31. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/judge.py +0 -0
  32. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/metrics.py +0 -0
  33. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/runner.py +0 -0
  34. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/feedback.py +0 -0
  35. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/hook_cli.py +0 -0
  36. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/hook_server.py +0 -0
  37. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/__init__.py +0 -0
  38. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/claude_code.py +0 -0
  39. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/documents.py +0 -0
  40. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/interfaces.py +0 -0
  41. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/__init__.py +0 -0
  42. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/anthropic.py +0 -0
  43. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/base.py +0 -0
  44. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/openai_compat.py +0 -0
  45. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/project.py +0 -0
  46. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/query_complexity.py +0 -0
  47. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/recall.py +0 -0
  48. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/redact.py +0 -0
  49. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/retrieve/__init__.py +0 -0
  50. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/retrieve/retriever.py +0 -0
  51. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/service.py +0 -0
  52. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/session_context.py +0 -0
  53. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/store/__init__.py +0 -0
  54. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/tokencount.py +0 -0
  55. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/turn_metrics.py +0 -0
  56. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/types.py +0 -0
  57. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/dependency_links.txt +0 -0
  58. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/entry_points.txt +0 -0
  59. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/requires.txt +0 -0
  60. {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/top_level.txt +0 -0
  61. {memor_cli-0.4.1 → memor_cli-0.5.0}/setup.cfg +0 -0
  62. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_cli_smoke.py +0 -0
  63. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_daemon.py +0 -0
  64. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dashboard.py +0 -0
  65. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dataset_builder.py +0 -0
  66. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dimension_safety.py +0 -0
  67. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_distiller.py +0 -0
  68. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_embed.py +0 -0
  69. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_embed_benchmark.py +0 -0
  70. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_eval_ablation.py +0 -0
  71. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_eval_runner.py +0 -0
  72. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_external_baselines.py +0 -0
  73. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_extractive.py +0 -0
  74. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_feedback.py +0 -0
  75. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hook.py +0 -0
  76. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hook_server.py +0 -0
  77. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hybrid_retrieval.py +0 -0
  78. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_ingest_claude_code.py +0 -0
  79. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_ingest_documents.py +0 -0
  80. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_install_hook.py +0 -0
  81. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_interfaces.py +0 -0
  82. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_judge.py +0 -0
  83. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_metrics.py +0 -0
  84. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_noise_filter.py +0 -0
  85. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_project_resolver.py +0 -0
  86. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_quality_gate.py +0 -0
  87. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_query_complexity.py +0 -0
  88. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_recall_core.py +0 -0
  89. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_redact.py +0 -0
  90. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_retriever.py +0 -0
  91. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_semantic_feedback.py +0 -0
  92. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_service.py +0 -0
  93. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_session_context.py +0 -0
  94. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_skill_recall.py +0 -0
  95. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_store.py +0 -0
  96. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_supersession.py +0 -0
  97. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_tokencount.py +0 -0
  98. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_turn_metrics.py +0 -0
  99. {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: memor-cli
3
- Version: 0.4.1
3
+ Version: 0.5.0
4
4
  Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
5
5
  Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
6
6
  License-Expression: MIT
@@ -46,7 +46,7 @@ Dynamic: license-file
46
46
  ```
47
47
 
48
48
  [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
49
- [![Tests](https://img.shields.io/badge/tests-215%20passing-brightgreen.svg)]()
49
+ [![Tests](https://img.shields.io/badge/tests-227%20passing-brightgreen.svg)]()
50
50
  [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)]()
51
51
  [![PyPI](https://img.shields.io/pypi/v/memor-cli.svg)](https://pypi.org/project/memor-cli/)
52
52
 
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
200
200
  memor ingest-doc <file> Ingest a markdown document
201
201
  memor distill --project <name> Run distillation manually
202
202
  memor eval <cases.json> Run eval suite
203
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
203
204
  memor bench-embed --project <name> Compare embedding models
204
205
  ```
205
206
 
@@ -9,7 +9,7 @@
9
9
  ```
10
10
 
11
11
  [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
12
- [![Tests](https://img.shields.io/badge/tests-215%20passing-brightgreen.svg)]()
12
+ [![Tests](https://img.shields.io/badge/tests-227%20passing-brightgreen.svg)]()
13
13
  [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)]()
14
14
  [![PyPI](https://img.shields.io/pypi/v/memor-cli.svg)](https://pypi.org/project/memor-cli/)
15
15
 
@@ -163,6 +163,7 @@ memor ingest-project <dir> Bulk ingest a project directory
163
163
  memor ingest-doc <file> Ingest a markdown document
164
164
  memor distill --project <name> Run distillation manually
165
165
  memor eval <cases.json> Run eval suite
166
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
166
167
  memor bench-embed --project <name> Compare embedding models
167
168
  ```
168
169
 
@@ -78,9 +78,10 @@ MAINTENANCE
78
78
  memor version Print the installed version
79
79
 
80
80
  EVALUATION
81
- memor eval <cases.json> Run eval suite
82
- memor eval-judge --project <name> LLM-as-judge evaluation
83
- memor bench-embed --project <name> Compare embedding models
81
+ memor eval <cases.json> Run eval suite
82
+ memor eval-judge --project <name> LLM-as-judge evaluation
83
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
84
+ memor bench-embed --project <name> Compare embedding models
84
85
 
85
86
  CONFIGURATION
86
87
  Everything works locally with zero API keys.
@@ -191,6 +192,38 @@ def eval_judge_cmd(project: str = typer.Option(...), db: str = "memor.db",
191
192
  typer.echo(f"Judge eval complete: mean relevance = {summary['mean_relevance']:.3f}")
192
193
 
193
194
 
195
+ @app.command("eval-counterfactual")
196
+ def eval_counterfactual_cmd(project: str = typer.Option(...), db: str = "memor.db",
197
+ k: int = 8, fake: bool = False,
198
+ llm_provider: str = "anthropic",
199
+ llm_model: str = "claude-sonnet-4-6",
200
+ holdout: int = 2):
201
+ """Counterfactual eval: win/tie/loss vs no-memory baseline. Requires an LLM API key."""
202
+ from memor.eval.counterfactual import build_cases_from_store, run_suite
203
+ e = _embedder(fake); s = SqliteStore(_db_path(db), dim=e.dim)
204
+ cases = build_cases_from_store(s, project=project, holdout_turns=holdout)
205
+ if not cases:
206
+ typer.echo("No cases — need at least 2 sessions with >= 4 turns each.")
207
+ raise typer.Exit(1)
208
+ typer.echo(f"Built {len(cases)} counterfactual cases. Running evaluation...")
209
+ if llm_provider == "anthropic":
210
+ from memor.llm.anthropic import AnthropicLLM
211
+ llm = AnthropicLLM(model=llm_model)
212
+ else:
213
+ from memor.llm.openai_compat import OpenAICompatLLM
214
+ import os
215
+ llm = OpenAICompatLLM(base_url=os.environ.get("OPENAI_BASE_URL", "http://localhost:11434/v1"),
216
+ api_key=os.environ.get("OPENAI_API_KEY", ""), model=llm_model)
217
+ summary = run_suite(cases, store=s, embedder=e, llm=llm, k=k)
218
+ typer.echo(json.dumps({k: v for k, v in summary.items() if k != "cases"}, indent=2))
219
+ typer.echo("")
220
+ typer.echo(f" Win: {summary['win_count']}/{summary['n_cases']} ({summary['win_pct']}%)")
221
+ typer.echo(f" Tie: {summary['tie_count']}/{summary['n_cases']} ({summary['tie_pct']}%)")
222
+ typer.echo(f" Loss: {summary['loss_count']}/{summary['n_cases']} ({summary['loss_pct']}%)")
223
+ typer.echo(f" Do-no-harm: {summary['do_no_harm_pct']}%")
224
+ s.save_eval_run({"type": "counterfactual", "k": k, "project": project, "holdout": holdout}, summary)
225
+
226
+
194
227
  @app.command("bench-embed")
195
228
  def bench_embed(project: str = typer.Option(...), db: str = "memor.db",
196
229
  k: int = 8, fake: bool = False):
@@ -139,6 +139,19 @@ def create_app(db_path: str | None = None) -> FastAPI:
139
139
  store = _store()
140
140
  return store.get_token_roi(project=project)
141
141
 
142
+ @app.get("/api/roi-trend")
143
+ def roi_trend(project: str | None = Query(None)):
144
+ store = _store()
145
+ return store.get_roi_trend(project=project)
146
+
147
+ @app.get("/api/eval/latest")
148
+ def eval_latest(eval_type: str = Query("counterfactual")):
149
+ store = _store()
150
+ result = store.get_latest_eval(eval_type)
151
+ if not result:
152
+ return {"status": "no_runs"}
153
+ return result
154
+
142
155
  @app.get("/api/health")
143
156
  def health():
144
157
  store = _store()
@@ -404,6 +404,10 @@
404
404
  </div>
405
405
  </div>
406
406
  </div>
407
+ <div style="margin-top:16px;border-top:1px solid var(--border-light);padding-top:12px;position:relative;">
408
+ <div id="roi-sparkline" style="width:100%;height:80px;"></div>
409
+ <div id="roi-spark-tooltip" style="display:none;position:absolute;top:0;background:var(--surface3);border:1px solid var(--border);border-radius:6px;padding:6px 10px;font-size:11px;color:var(--text);white-space:nowrap;pointer-events:none;box-shadow:0 4px 12px rgba(0,0,0,0.4);z-index:10;"></div>
410
+ </div>
407
411
  </div>
408
412
  </section>
409
413
 
@@ -835,6 +839,89 @@
835
839
  }
836
840
  }
837
841
 
842
+ function renderROITrend(data) {
843
+ var container = document.getElementById('roi-sparkline');
844
+ var tooltip = document.getElementById('roi-spark-tooltip');
845
+ if (!container || !data || !data.length) return;
846
+
847
+ var W = container.clientWidth || 400;
848
+ var H = 80;
849
+ var padL = 32, padR = 8, padT = 8, padB = 20;
850
+ var plotW = W - padL - padR;
851
+ var plotH = H - padT - padB;
852
+
853
+ var vals = data.map(function(d) { return d.reduction_pct; });
854
+ var minV = Math.min.apply(null, vals);
855
+ var maxV = Math.max.apply(null, vals);
856
+ var range = (maxV - minV) || 1;
857
+ var yFloor = Math.min(0, minV);
858
+ var yCeil = Math.max(maxV, 10);
859
+ range = (yCeil - yFloor) || 1;
860
+
861
+ function x(i) { return padL + (data.length > 1 ? (i / (data.length - 1)) * plotW : plotW / 2); }
862
+ function y(v) { return padT + plotH - ((v - yFloor) / range) * plotH; }
863
+
864
+ var lineColor = vals[vals.length - 1] >= 0 ? '#3dd68c' : '#e89320';
865
+ var gradId = 'roiGrad';
866
+
867
+ var pts = data.map(function(d, i) { return x(i) + ',' + y(d.reduction_pct); });
868
+ var linePath = 'M' + pts.join(' L');
869
+ var areaPath = linePath + ' L' + x(data.length - 1) + ',' + y(yFloor) + ' L' + x(0) + ',' + y(yFloor) + ' Z';
870
+
871
+ var gridLines = '';
872
+ var steps = [yFloor, Math.round((yFloor + yCeil) / 2), yCeil];
873
+ steps.forEach(function(v) {
874
+ var yy = y(v);
875
+ gridLines += '<line x1="' + padL + '" y1="' + yy + '" x2="' + (W - padR) + '" y2="' + yy + '" stroke="var(--border-light)" stroke-width="0.5" stroke-dasharray="3,3"/>';
876
+ gridLines += '<text x="' + (padL - 4) + '" y="' + (yy + 3) + '" fill="var(--text-muted)" font-size="8" text-anchor="end">' + v + '%</text>';
877
+ });
878
+
879
+ var xLabels = '';
880
+ data.forEach(function(d, i) {
881
+ xLabels += '<text x="' + x(i) + '" y="' + (H - 2) + '" fill="var(--text-muted)" font-size="8" text-anchor="middle">' + d.day.slice(5) + '</text>';
882
+ });
883
+
884
+ var dots = data.map(function(d, i) {
885
+ return '<circle cx="' + x(i) + '" cy="' + y(d.reduction_pct) + '" r="4" fill="' + lineColor + '" stroke="var(--surface)" stroke-width="2" style="opacity:0;" class="roi-dot" data-idx="' + i + '"/>';
886
+ }).join('');
887
+
888
+ var hitAreas = data.map(function(d, i) {
889
+ return '<rect x="' + (x(i) - plotW / data.length / 2) + '" y="' + padT + '" width="' + (plotW / data.length) + '" height="' + plotH + '" fill="transparent" class="roi-hit" data-idx="' + i + '"/>';
890
+ }).join('');
891
+
892
+ container.innerHTML = '<svg width="' + W + '" height="' + H + '" style="display:block;">' +
893
+ '<defs><linearGradient id="' + gradId + '" x1="0" y1="0" x2="0" y2="1">' +
894
+ '<stop offset="0%" stop-color="' + lineColor + '" stop-opacity="0.3"/>' +
895
+ '<stop offset="100%" stop-color="' + lineColor + '" stop-opacity="0.0"/>' +
896
+ '</linearGradient></defs>' +
897
+ gridLines + xLabels +
898
+ '<path d="' + areaPath + '" fill="url(#' + gradId + ')"/>' +
899
+ '<path d="' + linePath + '" fill="none" stroke="' + lineColor + '" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"/>' +
900
+ dots + hitAreas +
901
+ '</svg>';
902
+
903
+ container.querySelectorAll('.roi-hit').forEach(function(rect) {
904
+ rect.addEventListener('mouseenter', function() {
905
+ var i = parseInt(rect.getAttribute('data-idx'));
906
+ var d = data[i];
907
+ var dot = container.querySelectorAll('.roi-dot')[i];
908
+ if (dot) dot.style.opacity = '1';
909
+ tooltip.innerHTML = '<strong>' + d.day + '</strong><br>' +
910
+ d.reduction_pct + '% fewer tools<br>' +
911
+ d.turns_with + ' recall / ' + d.turns_without + ' no-recall turns';
912
+ tooltip.style.display = 'block';
913
+ var tx = Math.min(x(i), W - 160);
914
+ tooltip.style.left = tx + 'px';
915
+ });
916
+ rect.addEventListener('mouseleave', function() {
917
+ var i = parseInt(rect.getAttribute('data-idx'));
918
+ var dot = container.querySelectorAll('.roi-dot')[i];
919
+ if (dot) dot.style.opacity = '0';
920
+ tooltip.style.display = 'none';
921
+ });
922
+ });
923
+ }
924
+
838
925
  /* ── Data loaders ──────────────────────────────────────── */
839
926
  async function loadSummary() { try { renderSummary(await api('/api/summary')); } catch(e) { console.warn('summary',e); } }
840
927
  async function loadProjects() { try { renderProjects(await api('/api/projects')); } catch(e) { console.warn('projects',e); } }
@@ -843,6 +930,7 @@
843
930
  async function loadHealth() { try { renderHealth(await api('/api/health')); } catch(e) { console.warn('health',e); } }
844
931
  async function loadTrend() { try { renderTrend(await api('/api/recall-trend?days=30')); } catch(e) { console.warn('trend',e); } }
845
932
  async function loadROI() { try { renderROI(await api('/api/roi')); } catch(e) { console.warn('roi',e); } }
933
+ async function loadROITrend() { try { renderROITrend(await api('/api/roi-trend')); } catch(e) { console.warn('roi-trend',e); } }
846
934
  async function loadRecalls() {
847
935
  try {
848
936
  var url = '/api/recalls?limit=50' + (projectFilter ? '&project=' + encodeURIComponent(projectFilter) : '');
@@ -851,7 +939,7 @@
851
939
  }
852
940
 
853
941
  async function refresh() {
854
- await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI()]);
942
+ await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI(), loadROITrend()]);
855
943
  document.getElementById('last-updated').textContent = new Date().toLocaleTimeString();
856
944
 
857
945
  renderMiniBars('mb-chunks', null);
@@ -0,0 +1,176 @@
1
+ """Counterfactual eval harness — win/tie/loss vs no-memory baseline.
2
+
3
+ For each held-out session, uses the opening turn as query, retrieves context
4
+ from prior sessions, and asks an LLM judge whether the recalled context would
5
+ have helped (win), been irrelevant (tie), or misled the agent (loss).
6
+
7
+ The key metric is do-no-harm rate = 1 - loss_rate."""
8
+ from __future__ import annotations
9
+
10
+ import enum
11
+ import json
12
+ import re
13
+ import time
14
+ from dataclasses import dataclass
15
+
16
+ from memor.types import Artifact, Scope
17
+ from memor.retrieve.retriever import Retriever
18
+
19
+
20
+ class Outcome(enum.Enum):
21
+ WIN = "win"
22
+ TIE = "tie"
23
+ LOSS = "loss"
24
+
25
+
26
+ @dataclass
27
+ class CounterfactualCase:
28
+ query: str
29
+ holdout_texts: list[str]
30
+ scope_project: str
31
+ session_id: str
32
+
33
+
34
+ @dataclass
35
+ class CounterfactualVerdict:
36
+ outcome: Outcome
37
+ reasoning: str
38
+ confidence: float
39
+
40
+
41
+ COUNTERFACTUAL_PROMPT = """You are evaluating whether recalled memory context helped or hurt a coding agent.
42
+
43
+ The agent received this task:
44
+ ---
45
+ {query}
46
+ ---
47
+
48
+ Here is what actually happened next in the session (ground truth — the agent didn't see this):
49
+ ---
50
+ {holdout}
51
+ ---
52
+
53
+ Here is the context that was recalled from prior sessions and injected into the agent's prompt:
54
+ ---
55
+ {recalled_context}
56
+ ---
57
+
58
+ Compare the recalled context against what actually happened. Score as:
59
+
60
+ - **win**: The recalled context directly anticipates what happened, would save the agent exploration/time, or would prevent repeating a past mistake. The agent would have been measurably better with this context.
61
+ - **tie**: The recalled context is irrelevant to what happened. Neither helpful nor harmful — the agent would have arrived at the same outcome either way.
62
+ - **loss**: The recalled context is stale, contradictory, or misleading. It would have sent the agent down the wrong path or caused confusion.
63
+
64
+ When uncertain, default to **tie** — only score win when the context clearly helps, and loss when it clearly hurts.
65
+
66
+ Return STRICT JSON: {{"outcome": "win"|"tie"|"loss", "reasoning": "<one sentence>", "confidence": <float 0-1>}}"""
67
+
68
+
69
+ def build_cases_from_store(store, *, project: str,
70
+ holdout_turns: int = 2,
71
+ min_session_turns: int = 4,
72
+ min_prior_sessions: int = 1) -> list[CounterfactualCase]:
73
+ rows = store.db.execute(
74
+ "SELECT * FROM artifacts WHERE project=? AND kind='session_chunk' AND active=1",
75
+ (project,)).fetchall()
76
+ artifacts = [store._row_to_artifact(r) for r in rows]
77
+
78
+ by_session: dict[str, list[Artifact]] = {}
79
+ for a in artifacts:
80
+ by_session.setdefault(a.meta.get("session_id", "?"), []).append(a)
81
+ for v in by_session.values():
82
+ v.sort(key=lambda a: a.meta.get("ord", 0))
83
+ sessions = sorted(by_session.items(), key=lambda kv: kv[1][0].created_at)
84
+
85
+ cases: list[CounterfactualCase] = []
86
+ for idx, (sid, chunks) in enumerate(sessions):
87
+ if idx < min_prior_sessions:
88
+ continue
89
+ if len(chunks) < min_session_turns:
90
+ continue
91
+ query = chunks[0].text
92
+ holdout = [c.text for c in chunks[-holdout_turns:]]
93
+ cases.append(CounterfactualCase(
94
+ query=query, holdout_texts=holdout,
95
+ scope_project=project, session_id=sid))
96
+ return cases
97
+
98
+
99
+ def _extract_json(raw: str) -> dict:
100
+ m = re.search(r"\{.*\}", raw, re.DOTALL)
101
+ if m:
102
+ return json.loads(m.group())
103
+ return json.loads(raw)
104
+
105
+
106
+ def parse_verdict_json(raw: str) -> CounterfactualVerdict:
107
+ data = _extract_json(raw)
108
+ outcome_str = data.get("outcome", "tie").lower().strip()
109
+ outcome = Outcome(outcome_str) if outcome_str in ("win", "tie", "loss") else Outcome.TIE
110
+ return CounterfactualVerdict(
111
+ outcome=outcome,
112
+ reasoning=data.get("reasoning", ""),
113
+ confidence=float(data.get("confidence", 0.5)),
114
+ )
115
+
116
+
117
+ def run_case(case: CounterfactualCase, *, store, embedder, llm,
118
+ k: int = 8) -> CounterfactualVerdict:
119
+ scope = Scope(project=case.scope_project)
120
+ r = Retriever(store, embedder, k=k)
121
+ trace = r.query(case.query, scope)
122
+
123
+ if not trace.hits:
124
+ return CounterfactualVerdict(
125
+ outcome=Outcome.TIE,
126
+ reasoning="No context recalled — nothing to evaluate",
127
+ confidence=1.0)
128
+
129
+ recalled = "\n\n".join(
130
+ f"[{h.artifact.kind}] {h.artifact.text}" for h in trace.hits)
131
+ holdout = "\n".join(case.holdout_texts)
132
+
133
+ prompt = COUNTERFACTUAL_PROMPT.format(
134
+ query=case.query, holdout=holdout, recalled_context=recalled)
135
+ raw = llm.complete(prompt)
136
+ return parse_verdict_json(raw)
137
+
138
+
139
+ def run_suite(cases: list[CounterfactualCase], *, store, embedder, llm,
140
+ k: int = 8) -> dict:
141
+ verdicts = []
142
+ for c in cases:
143
+ v = run_case(c, store=store, embedder=embedder, llm=llm, k=k)
144
+ verdicts.append((c, v))
145
+ verdict_list = [v for _, v in verdicts]
146
+ summary = summarize_verdicts(verdict_list)
147
+ summary["cases"] = [
148
+ {"session_id": c.session_id, "query_preview": c.query[:100],
149
+ "outcome": v.outcome.value, "reasoning": v.reasoning,
150
+ "confidence": v.confidence}
151
+ for c, v in verdicts
152
+ ]
153
+ return summary
154
+
155
+
156
+ def summarize_verdicts(verdicts: list[CounterfactualVerdict]) -> dict:
157
+ n = len(verdicts)
158
+ if n == 0:
159
+ return {"n_cases": 0, "win_count": 0, "tie_count": 0, "loss_count": 0,
160
+ "win_pct": 0.0, "tie_pct": 0.0, "loss_pct": 0.0,
161
+ "do_no_harm_pct": 100.0}
162
+
163
+ wins = sum(1 for v in verdicts if v.outcome == Outcome.WIN)
164
+ ties = sum(1 for v in verdicts if v.outcome == Outcome.TIE)
165
+ losses = sum(1 for v in verdicts if v.outcome == Outcome.LOSS)
166
+
167
+ return {
168
+ "n_cases": n,
169
+ "win_count": wins,
170
+ "tie_count": ties,
171
+ "loss_count": losses,
172
+ "win_pct": round(wins / n * 100, 1),
173
+ "tie_pct": round(ties / n * 100, 1),
174
+ "loss_pct": round(losses / n * 100, 1),
175
+ "do_no_harm_pct": round((wins + ties) / n * 100, 1),
176
+ }
@@ -234,6 +234,19 @@ class SqliteStore:
234
234
  self.db.commit()
235
235
  return cur.lastrowid
236
236
 
237
+ def get_latest_eval(self, eval_type: str = "counterfactual") -> dict | None:
238
+ row = self.db.execute(
239
+ "SELECT * FROM eval_runs WHERE json_extract(config, '$.type') = ? "
240
+ "ORDER BY created_at DESC LIMIT 1", (eval_type,)).fetchone()
241
+ if not row:
242
+ return None
243
+ return {
244
+ "id": row["id"],
245
+ "created_at": row["created_at"],
246
+ "config": json.loads(row["config"]),
247
+ "metrics": json.loads(row["metrics"]),
248
+ }
249
+
237
250
  def log_recall(self, project: str, query_preview: str, hits_count: int,
238
251
  top_score: float, tokens_injected: int, latency_ms: float,
239
252
  status: str, session_id: str = "") -> None:
@@ -571,6 +584,43 @@ class SqliteStore:
571
584
  "tool_call_reduction_pct": reduction,
572
585
  }
573
586
 
587
+ def get_roi_trend(self, project: str | None = None,
588
+ days: int = 30) -> list[dict]:
589
+ where = "WHERE user_timestamp > 0"
590
+ params: list = []
591
+ if project:
592
+ where += " AND project = ?"
593
+ params.append(project)
594
+
595
+ rows = self.db.execute(f"""
596
+ SELECT date(user_timestamp, 'unixepoch', 'localtime') AS day,
597
+ AVG(CASE WHEN had_recall = 1 THEN tool_call_count END) AS avg_with,
598
+ AVG(CASE WHEN had_recall = 0 THEN tool_call_count END) AS avg_without,
599
+ COUNT(CASE WHEN had_recall = 1 THEN 1 END) AS turns_with,
600
+ COUNT(CASE WHEN had_recall = 0 THEN 1 END) AS turns_without,
601
+ COUNT(*) AS turns_total
602
+ FROM turn_metrics {where}
603
+ GROUP BY day
604
+ HAVING turns_with > 0 AND turns_without > 0
605
+ ORDER BY day
606
+ """, params).fetchall()
607
+
608
+ result = []
609
+ for r in rows:
610
+ avg_w = round(r["avg_with"], 2)
611
+ avg_wo = round(r["avg_without"], 2)
612
+ reduction = round((1 - avg_w / avg_wo) * 100, 1) if avg_wo > 0 else 0
613
+ result.append({
614
+ "day": r["day"],
615
+ "avg_with": avg_w,
616
+ "avg_without": avg_wo,
617
+ "reduction_pct": reduction,
618
+ "turns_with": r["turns_with"],
619
+ "turns_without": r["turns_without"],
620
+ "turns_total": r["turns_total"],
621
+ })
622
+ return result
623
+
574
624
  def get_onboarding_status(self) -> str:
575
625
  chunks = self.db.execute(
576
626
  "SELECT COUNT(*) as c FROM artifacts WHERE kind='session_chunk' AND active=1"
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: memor-cli
3
- Version: 0.4.1
3
+ Version: 0.5.0
4
4
  Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
5
5
  Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
6
6
  License-Expression: MIT
@@ -46,7 +46,7 @@ Dynamic: license-file
46
46
  ```
47
47
 
48
48
  [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
49
- [![Tests](https://img.shields.io/badge/tests-215%20passing-brightgreen.svg)]()
49
+ [![Tests](https://img.shields.io/badge/tests-227%20passing-brightgreen.svg)]()
50
50
  [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)]()
51
51
  [![PyPI](https://img.shields.io/pypi/v/memor-cli.svg)](https://pypi.org/project/memor-cli/)
52
52
 
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
200
200
  memor ingest-doc <file> Ingest a markdown document
201
201
  memor distill --project <name> Run distillation manually
202
202
  memor eval <cases.json> Run eval suite
203
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
203
204
  memor bench-embed --project <name> Compare embedding models
204
205
  ```
205
206
 
@@ -28,6 +28,7 @@ memor/embed/api.py
28
28
  memor/embed/fake.py
29
29
  memor/embed/local.py
30
30
  memor/eval/__init__.py
31
+ memor/eval/counterfactual.py
31
32
  memor/eval/dataset.py
32
33
  memor/eval/embed_benchmark.py
33
34
  memor/eval/judge.py
@@ -55,6 +56,7 @@ memor_cli.egg-info/entry_points.txt
55
56
  memor_cli.egg-info/requires.txt
56
57
  memor_cli.egg-info/top_level.txt
57
58
  tests/test_cli_smoke.py
59
+ tests/test_counterfactual.py
58
60
  tests/test_daemon.py
59
61
  tests/test_dashboard.py
60
62
  tests/test_dataset_builder.py
@@ -83,6 +85,7 @@ tests/test_query_complexity.py
83
85
  tests/test_recall_core.py
84
86
  tests/test_redact.py
85
87
  tests/test_retriever.py
88
+ tests/test_roi_trend.py
86
89
  tests/test_semantic_feedback.py
87
90
  tests/test_service.py
88
91
  tests/test_session_context.py
@@ -4,7 +4,7 @@ build-backend = "setuptools.build_meta"
4
4
 
5
5
  [project]
6
6
  name = "memor-cli"
7
- version = "0.4.1"
7
+ version = "0.5.0"
8
8
  description = "Measured memory for coding agents. Fire and forget — no API keys needed."
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -0,0 +1,104 @@
1
+ """Tests for the counterfactual eval harness — win/tie/loss vs no-memory baseline."""
2
+ import json
3
+ import time
4
+
5
+ from memor.types import Artifact
6
+ from memor.store.sqlite_store import SqliteStore
7
+ from memor.embed.fake import FakeEmbedder
8
+ from memor.eval.counterfactual import (
9
+ build_cases_from_store, CounterfactualCase, CounterfactualVerdict,
10
+ Outcome, summarize_verdicts, parse_verdict_json,
11
+ )
12
+
13
+
14
+ def _seed_store(tmp_path, *, n_sessions=3, turns_per_session=5):
15
+ """Create a store with sessions that have recall logs and artifacts."""
16
+ db_path = str(tmp_path / "m.db")
17
+ e = FakeEmbedder(dim=16)
18
+ store = SqliteStore(db_path, dim=16)
19
+
20
+ now = time.time()
21
+ for s_idx in range(n_sessions):
22
+ sid = f"session-{s_idx}"
23
+ session_start = now - (n_sessions - s_idx) * 86400
24
+ artifacts = []
25
+ vectors = []
26
+ for t_idx in range(turns_per_session):
27
+ aid = f"a-{s_idx}-{t_idx}"
28
+ text = f"Turn {t_idx} of session {s_idx}: discussion about auth middleware and token refresh"
29
+ vec = e.embed([text])[0]
30
+ artifacts.append(Artifact(
31
+ id=aid, kind="session_chunk", project="testproj",
32
+ source="test", text=text, token_count=20,
33
+ created_at=session_start + t_idx,
34
+ meta={"session_id": sid, "ord": t_idx}))
35
+ vectors.append(vec)
36
+ store.add_artifacts(artifacts, vectors)
37
+
38
+ store.db.commit()
39
+ return store
40
+
41
+
42
+ def test_build_cases_from_store(tmp_path):
43
+ store = _seed_store(tmp_path, n_sessions=3, turns_per_session=5)
44
+ cases = build_cases_from_store(store, project="testproj",
45
+ holdout_turns=2, min_session_turns=4)
46
+ assert len(cases) >= 1
47
+ for c in cases:
48
+ assert c.query
49
+ assert c.holdout_texts
50
+ assert c.scope_project == "testproj"
51
+
52
+
53
+ def test_build_cases_needs_prior_sessions(tmp_path):
54
+ store = _seed_store(tmp_path, n_sessions=1, turns_per_session=5)
55
+ cases = build_cases_from_store(store, project="testproj")
56
+ assert len(cases) == 0
57
+
58
+
59
+ def test_parse_verdict_json_win():
60
+ raw = '{"outcome": "win", "reasoning": "Context saved rework", "confidence": 0.9}'
61
+ v = parse_verdict_json(raw)
62
+ assert v.outcome == Outcome.WIN
63
+ assert v.confidence == 0.9
64
+
65
+
66
+ def test_parse_verdict_json_loss():
67
+ raw = '{"outcome": "loss", "reasoning": "Stale info misled agent", "confidence": 0.8}'
68
+ v = parse_verdict_json(raw)
69
+ assert v.outcome == Outcome.LOSS
70
+
71
+
72
+ def test_parse_verdict_json_tie():
73
+ raw = '{"outcome": "tie", "reasoning": "Irrelevant context", "confidence": 0.7}'
74
+ v = parse_verdict_json(raw)
75
+ assert v.outcome == Outcome.TIE
76
+
77
+
78
+ def test_parse_verdict_embedded_json():
79
+ raw = 'Here is my assessment:\n```json\n{"outcome": "win", "reasoning": "Helped", "confidence": 0.85}\n```'
80
+ v = parse_verdict_json(raw)
81
+ assert v.outcome == Outcome.WIN
82
+
83
+
84
+ def test_summarize_verdicts():
85
+ verdicts = [
86
+ CounterfactualVerdict(outcome=Outcome.WIN, reasoning="a", confidence=0.9),
87
+ CounterfactualVerdict(outcome=Outcome.WIN, reasoning="b", confidence=0.8),
88
+ CounterfactualVerdict(outcome=Outcome.TIE, reasoning="c", confidence=0.7),
89
+ CounterfactualVerdict(outcome=Outcome.LOSS, reasoning="d", confidence=0.6),
90
+ ]
91
+ s = summarize_verdicts(verdicts)
92
+ assert s["win_count"] == 2
93
+ assert s["tie_count"] == 1
94
+ assert s["loss_count"] == 1
95
+ assert s["n_cases"] == 4
96
+ assert s["win_pct"] == 50.0
97
+ assert s["loss_pct"] == 25.0
98
+ assert s["do_no_harm_pct"] == 75.0
99
+
100
+
101
+ def test_summarize_empty():
102
+ s = summarize_verdicts([])
103
+ assert s["n_cases"] == 0
104
+ assert s["do_no_harm_pct"] == 100.0
@@ -0,0 +1,87 @@
1
+ """Tests for daily ROI trend — tool call reduction over time."""
2
+ from memor.store.sqlite_store import SqliteStore
3
+ from memor.turn_metrics import TurnMetric
4
+
5
+
6
+ def test_roi_trend_groups_by_day(tmp_path):
7
+ db_path = str(tmp_path / "m.db")
8
+ store = SqliteStore(db_path, dim=16)
9
+
10
+ day1_base = 1749427200.0 # 2025-06-09 00:00:00 UTC
11
+ day2_base = day1_base + 86400
12
+
13
+ store.save_turn_metrics("s1", "p", [
14
+ TurnMetric(turn_idx=0, user_timestamp=day1_base + 100, tool_call_count=1, had_recall=True),
15
+ TurnMetric(turn_idx=1, user_timestamp=day1_base + 200, tool_call_count=4, had_recall=False),
16
+ ])
17
+ store.save_turn_metrics("s2", "p", [
18
+ TurnMetric(turn_idx=0, user_timestamp=day2_base + 100, tool_call_count=2, had_recall=True),
19
+ TurnMetric(turn_idx=1, user_timestamp=day2_base + 200, tool_call_count=6, had_recall=False),
20
+ ])
21
+
22
+ trend = store.get_roi_trend()
23
+ assert len(trend) == 2
24
+ for day in trend:
25
+ assert "day" in day
26
+ assert "avg_with" in day
27
+ assert "avg_without" in day
28
+ assert "reduction_pct" in day
29
+ assert "turns_total" in day
30
+
31
+
32
+ def test_roi_trend_calculates_reduction(tmp_path):
33
+ db_path = str(tmp_path / "m.db")
34
+ store = SqliteStore(db_path, dim=16)
35
+
36
+ base = 1749427200.0
37
+ store.save_turn_metrics("s1", "p", [
38
+ TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=2, had_recall=True),
39
+ TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=2, had_recall=True),
40
+ TurnMetric(turn_idx=2, user_timestamp=base + 300, tool_call_count=4, had_recall=False),
41
+ TurnMetric(turn_idx=3, user_timestamp=base + 400, tool_call_count=4, had_recall=False),
42
+ ])
43
+
44
+ trend = store.get_roi_trend()
45
+ assert len(trend) == 1
46
+ day = trend[0]
47
+ assert day["avg_with"] == 2.0
48
+ assert day["avg_without"] == 4.0
49
+ assert day["reduction_pct"] == 50.0
50
+
51
+
52
+ def test_roi_trend_filters_by_project(tmp_path):
53
+ db_path = str(tmp_path / "m.db")
54
+ store = SqliteStore(db_path, dim=16)
55
+
56
+ base = 1749427200.0
57
+ store.save_turn_metrics("s1", "proj-a", [
58
+ TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=1, had_recall=True),
59
+ TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=5, had_recall=False),
60
+ ])
61
+ store.save_turn_metrics("s2", "proj-b", [
62
+ TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=3, had_recall=True),
63
+ TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=3, had_recall=False),
64
+ ])
65
+
66
+ trend_a = store.get_roi_trend(project="proj-a")
67
+ assert len(trend_a) == 1
68
+ assert trend_a[0]["avg_with"] == 1.0
69
+
70
+ trend_b = store.get_roi_trend(project="proj-b")
71
+ assert len(trend_b) == 1
72
+ assert trend_b[0]["avg_with"] == 3.0
73
+
74
+
75
+ def test_roi_trend_skips_days_without_both_types(tmp_path):
76
+ """Days that have only recall or only non-recall turns can't compute reduction."""
77
+ db_path = str(tmp_path / "m.db")
78
+ store = SqliteStore(db_path, dim=16)
79
+
80
+ base = 1749427200.0
81
+ store.save_turn_metrics("s1", "p", [
82
+ TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=2, had_recall=True),
83
+ TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=3, had_recall=True),
84
+ ])
85
+
86
+ trend = store.get_roi_trend()
87
+ assert len(trend) == 0
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes