memor-cli 0.4.0__tar.gz → 0.5.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. {memor_cli-0.4.0/memor_cli.egg-info → memor_cli-0.5.0}/PKG-INFO +3 -2
  2. {memor_cli-0.4.0 → memor_cli-0.5.0}/README.md +2 -1
  3. memor_cli-0.5.0/memor/__init__.py +1 -0
  4. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/cli.py +36 -3
  5. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/server.py +13 -0
  6. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/static/index.html +144 -60
  7. memor_cli-0.5.0/memor/eval/counterfactual.py +176 -0
  8. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/store/sqlite_store.py +50 -0
  9. {memor_cli-0.4.0 → memor_cli-0.5.0/memor_cli.egg-info}/PKG-INFO +3 -2
  10. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/SOURCES.txt +3 -0
  11. {memor_cli-0.4.0 → memor_cli-0.5.0}/pyproject.toml +1 -1
  12. memor_cli-0.5.0/tests/test_counterfactual.py +104 -0
  13. memor_cli-0.5.0/tests/test_roi_trend.py +87 -0
  14. memor_cli-0.4.0/memor/__init__.py +0 -1
  15. {memor_cli-0.4.0 → memor_cli-0.5.0}/LICENSE +0 -0
  16. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/daemon.py +0 -0
  17. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/__init__.py +0 -0
  18. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/__init__.py +0 -0
  19. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/distiller.py +0 -0
  20. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/extractive.py +0 -0
  21. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/__init__.py +0 -0
  22. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/api.py +0 -0
  23. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/fake.py +0 -0
  24. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/local.py +0 -0
  25. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/__init__.py +0 -0
  26. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/__init__.py +0 -0
  27. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/base.py +0 -0
  28. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/claude_mem.py +0 -0
  29. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/graphiti.py +0 -0
  30. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/dataset.py +0 -0
  31. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/embed_benchmark.py +0 -0
  32. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/judge.py +0 -0
  33. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/metrics.py +0 -0
  34. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/runner.py +0 -0
  35. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/feedback.py +0 -0
  36. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/hook_cli.py +0 -0
  37. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/hook_server.py +0 -0
  38. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/__init__.py +0 -0
  39. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/claude_code.py +0 -0
  40. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/documents.py +0 -0
  41. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/interfaces.py +0 -0
  42. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/__init__.py +0 -0
  43. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/anthropic.py +0 -0
  44. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/base.py +0 -0
  45. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/openai_compat.py +0 -0
  46. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/project.py +0 -0
  47. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/query_complexity.py +0 -0
  48. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/recall.py +0 -0
  49. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/redact.py +0 -0
  50. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/retrieve/__init__.py +0 -0
  51. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/retrieve/retriever.py +0 -0
  52. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/service.py +0 -0
  53. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/session_context.py +0 -0
  54. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/store/__init__.py +0 -0
  55. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/tokencount.py +0 -0
  56. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/turn_metrics.py +0 -0
  57. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/types.py +0 -0
  58. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/dependency_links.txt +0 -0
  59. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/entry_points.txt +0 -0
  60. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/requires.txt +0 -0
  61. {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/top_level.txt +0 -0
  62. {memor_cli-0.4.0 → memor_cli-0.5.0}/setup.cfg +0 -0
  63. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_cli_smoke.py +0 -0
  64. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_daemon.py +0 -0
  65. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dashboard.py +0 -0
  66. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dataset_builder.py +0 -0
  67. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dimension_safety.py +0 -0
  68. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_distiller.py +0 -0
  69. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_embed.py +0 -0
  70. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_embed_benchmark.py +0 -0
  71. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_eval_ablation.py +0 -0
  72. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_eval_runner.py +0 -0
  73. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_external_baselines.py +0 -0
  74. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_extractive.py +0 -0
  75. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_feedback.py +0 -0
  76. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hook.py +0 -0
  77. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hook_server.py +0 -0
  78. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hybrid_retrieval.py +0 -0
  79. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_ingest_claude_code.py +0 -0
  80. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_ingest_documents.py +0 -0
  81. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_install_hook.py +0 -0
  82. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_interfaces.py +0 -0
  83. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_judge.py +0 -0
  84. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_metrics.py +0 -0
  85. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_noise_filter.py +0 -0
  86. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_project_resolver.py +0 -0
  87. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_quality_gate.py +0 -0
  88. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_query_complexity.py +0 -0
  89. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_recall_core.py +0 -0
  90. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_redact.py +0 -0
  91. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_retriever.py +0 -0
  92. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_semantic_feedback.py +0 -0
  93. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_service.py +0 -0
  94. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_session_context.py +0 -0
  95. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_skill_recall.py +0 -0
  96. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_store.py +0 -0
  97. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_supersession.py +0 -0
  98. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_tokencount.py +0 -0
  99. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_turn_metrics.py +0 -0
  100. {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_types.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: memor-cli
3
- Version: 0.4.0
3
+ Version: 0.5.0
4
4
  Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
5
5
  Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
6
6
  License-Expression: MIT
@@ -46,7 +46,7 @@ Dynamic: license-file
46
46
  ```
47
47
 
48
48
  [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
49
- [![Tests](https://img.shields.io/badge/tests-215%20passing-brightgreen.svg)]()
49
+ [![Tests](https://img.shields.io/badge/tests-227%20passing-brightgreen.svg)]()
50
50
  [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)]()
51
51
  [![PyPI](https://img.shields.io/pypi/v/memor-cli.svg)](https://pypi.org/project/memor-cli/)
52
52
 
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
200
200
  memor ingest-doc <file> Ingest a markdown document
201
201
  memor distill --project <name> Run distillation manually
202
202
  memor eval <cases.json> Run eval suite
203
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
203
204
  memor bench-embed --project <name> Compare embedding models
204
205
  ```
205
206
 
@@ -9,7 +9,7 @@
9
9
  ```
10
10
 
11
11
  [![License: MIT](https://img.shields.io/badge/License-MIT-blue.svg)](LICENSE)
12
- [![Tests](https://img.shields.io/badge/tests-215%20passing-brightgreen.svg)]()
12
+ [![Tests](https://img.shields.io/badge/tests-227%20passing-brightgreen.svg)]()
13
13
  [![Python](https://img.shields.io/badge/python-3.11%2B-blue.svg)]()
14
14
  [![PyPI](https://img.shields.io/pypi/v/memor-cli.svg)](https://pypi.org/project/memor-cli/)
15
15
 
@@ -163,6 +163,7 @@ memor ingest-project <dir> Bulk ingest a project directory
163
163
  memor ingest-doc <file> Ingest a markdown document
164
164
  memor distill --project <name> Run distillation manually
165
165
  memor eval <cases.json> Run eval suite
166
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
166
167
  memor bench-embed --project <name> Compare embedding models
167
168
  ```
168
169
 
@@ -0,0 +1 @@
1
+ __version__ = "0.4.1"
@@ -78,9 +78,10 @@ MAINTENANCE
78
78
  memor version Print the installed version
79
79
 
80
80
  EVALUATION
81
- memor eval <cases.json> Run eval suite
82
- memor eval-judge --project <name> LLM-as-judge evaluation
83
- memor bench-embed --project <name> Compare embedding models
81
+ memor eval <cases.json> Run eval suite
82
+ memor eval-judge --project <name> LLM-as-judge evaluation
83
+ memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
84
+ memor bench-embed --project <name> Compare embedding models
84
85
 
85
86
  CONFIGURATION
86
87
  Everything works locally with zero API keys.
@@ -191,6 +192,38 @@ def eval_judge_cmd(project: str = typer.Option(...), db: str = "memor.db",
191
192
  typer.echo(f"Judge eval complete: mean relevance = {summary['mean_relevance']:.3f}")
192
193
 
193
194
 
195
+ @app.command("eval-counterfactual")
196
+ def eval_counterfactual_cmd(project: str = typer.Option(...), db: str = "memor.db",
197
+ k: int = 8, fake: bool = False,
198
+ llm_provider: str = "anthropic",
199
+ llm_model: str = "claude-sonnet-4-6",
200
+ holdout: int = 2):
201
+ """Counterfactual eval: win/tie/loss vs no-memory baseline. Requires an LLM API key."""
202
+ from memor.eval.counterfactual import build_cases_from_store, run_suite
203
+ e = _embedder(fake); s = SqliteStore(_db_path(db), dim=e.dim)
204
+ cases = build_cases_from_store(s, project=project, holdout_turns=holdout)
205
+ if not cases:
206
+ typer.echo("No cases — need at least 2 sessions with >= 4 turns each.")
207
+ raise typer.Exit(1)
208
+ typer.echo(f"Built {len(cases)} counterfactual cases. Running evaluation...")
209
+ if llm_provider == "anthropic":
210
+ from memor.llm.anthropic import AnthropicLLM
211
+ llm = AnthropicLLM(model=llm_model)
212
+ else:
213
+ from memor.llm.openai_compat import OpenAICompatLLM
214
+ import os
215
+ llm = OpenAICompatLLM(base_url=os.environ.get("OPENAI_BASE_URL", "http://localhost:11434/v1"),
216
+ api_key=os.environ.get("OPENAI_API_KEY", ""), model=llm_model)
217
+ summary = run_suite(cases, store=s, embedder=e, llm=llm, k=k)
218
+ typer.echo(json.dumps({k: v for k, v in summary.items() if k != "cases"}, indent=2))
219
+ typer.echo("")
220
+ typer.echo(f" Win: {summary['win_count']}/{summary['n_cases']} ({summary['win_pct']}%)")
221
+ typer.echo(f" Tie: {summary['tie_count']}/{summary['n_cases']} ({summary['tie_pct']}%)")
222
+ typer.echo(f" Loss: {summary['loss_count']}/{summary['n_cases']} ({summary['loss_pct']}%)")
223
+ typer.echo(f" Do-no-harm: {summary['do_no_harm_pct']}%")
224
+ s.save_eval_run({"type": "counterfactual", "k": k, "project": project, "holdout": holdout}, summary)
225
+
226
+
194
227
  @app.command("bench-embed")
195
228
  def bench_embed(project: str = typer.Option(...), db: str = "memor.db",
196
229
  k: int = 8, fake: bool = False):
@@ -139,6 +139,19 @@ def create_app(db_path: str | None = None) -> FastAPI:
139
139
  store = _store()
140
140
  return store.get_token_roi(project=project)
141
141
 
142
+ @app.get("/api/roi-trend")
143
+ def roi_trend(project: str | None = Query(None)):
144
+ store = _store()
145
+ return store.get_roi_trend(project=project)
146
+
147
+ @app.get("/api/eval/latest")
148
+ def eval_latest(eval_type: str = Query("counterfactual")):
149
+ store = _store()
150
+ result = store.get_latest_eval(eval_type)
151
+ if not result:
152
+ return {"status": "no_runs"}
153
+ return result
154
+
142
155
  @app.get("/api/health")
143
156
  def health():
144
157
  store = _store()
@@ -172,6 +172,7 @@
172
172
  background: var(--surface); border: 1px solid var(--border);
173
173
  border-radius: var(--radius); padding: 22px 24px;
174
174
  box-shadow: var(--card-shadow);
175
+ display: flex; flex-direction: column;
175
176
  }
176
177
  .chart-header {
177
178
  display: flex; align-items: center; justify-content: space-between;
@@ -199,7 +200,7 @@
199
200
 
200
201
  .bar-chart {
201
202
  display: flex; align-items: flex-end; gap: 4px;
202
- height: 180px; padding-top: 8px;
203
+ flex: 1; min-height: 180px; padding-top: 8px;
203
204
  border-bottom: 1px solid var(--border-light);
204
205
  }
205
206
  .bar-group {
@@ -207,15 +208,16 @@
207
208
  align-items: center; gap: 0; height: 100%;
208
209
  justify-content: flex-end;
209
210
  }
210
- .bar-stack { display: flex; flex-direction: column-reverse; gap: 1px; width: 100%; max-width: 22px; }
211
- .bar-hits { background: var(--accent); border-radius: 2px 2px 0 0; min-height: 0; transition: height 0.3s; }
212
- .bar-miss { background: var(--surface3); border-radius: 2px 2px 0 0; min-height: 0; transition: height 0.3s; }
211
+ .bar-stack { display: flex; flex-direction: column-reverse; gap: 2px; width: 100%; max-width: 28px; flex: 1; align-items: center; justify-content: flex-end; }
212
+ .bar-cell { width: 16px; height: 10px; border-radius: 2px; transition: opacity 0.3s; }
213
+ .bar-cell-hit { background: var(--accent); }
214
+ .bar-cell-miss { background: var(--surface3); }
213
215
  .bar-label {
214
216
  font-size: 9px; color: var(--text-muted); margin-top: 6px;
215
217
  text-align: center; white-space: nowrap;
216
218
  }
217
219
 
218
- .bar-group:hover .bar-hits { background: #f0a030; }
220
+ .bar-group:hover .bar-cell-hit { background: #f0a030; }
219
221
  .bar-group { cursor: default; position: relative; }
220
222
  .bar-tooltip {
221
223
  display: none; position: absolute; bottom: calc(100% + 8px);
@@ -378,6 +380,37 @@
378
380
  </div>
379
381
  </section>
380
382
 
383
+ <!-- ── Token ROI hero ─────────────────────────────────── -->
384
+ <section>
385
+ <div class="chart-card" id="roi-card" style="display:none;">
386
+ <div style="display:flex;align-items:center;gap:20px;flex-wrap:wrap;">
387
+ <div>
388
+ <div class="chart-title" style="margin-bottom:4px;">Token ROI</div>
389
+ <div style="font-size:32px;font-weight:700;color:var(--ok);letter-spacing:-1px;" id="roi-value">&ndash;</div>
390
+ <div style="font-size:12px;color:var(--text-muted);margin-top:2px;" id="roi-desc">fewer tool calls when Memor injects context</div>
391
+ </div>
392
+ <div style="display:flex;gap:24px;flex:1;justify-content:flex-end;">
393
+ <div style="text-align:center;">
394
+ <div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-tools-with">&ndash;</div>
395
+ <div style="font-size:10px;color:var(--text-muted);">avg tools/turn<br>with recall</div>
396
+ </div>
397
+ <div style="text-align:center;">
398
+ <div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-tools-without">&ndash;</div>
399
+ <div style="font-size:10px;color:var(--text-muted);">avg tools/turn<br>without recall</div>
400
+ </div>
401
+ <div style="text-align:center;">
402
+ <div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-turns">&ndash;</div>
403
+ <div style="font-size:10px;color:var(--text-muted);" id="roi-turns-sub">turns measured</div>
404
+ </div>
405
+ </div>
406
+ </div>
407
+ <div style="margin-top:16px;border-top:1px solid var(--border-light);padding-top:12px;position:relative;">
408
+ <div id="roi-sparkline" style="width:100%;height:80px;"></div>
409
+ <div id="roi-spark-tooltip" style="display:none;position:absolute;top:0;background:var(--surface3);border:1px solid var(--border);border-radius:6px;padding:6px 10px;font-size:11px;color:var(--text);white-space:nowrap;pointer-events:none;box-shadow:0 4px 12px rgba(0,0,0,0.4);z-index:10;"></div>
410
+ </div>
411
+ </div>
412
+ </section>
413
+
381
414
  <!-- ── Recall trend chart ─────────────────────────────── -->
382
415
  <section>
383
416
  <div class="chart-grid">
@@ -426,32 +459,6 @@
426
459
  </div>
427
460
  </div>
428
461
  </div>
429
- <div class="chart-card">
430
- <div class="chart-header">
431
- <div class="chart-title">Token ROI</div>
432
- </div>
433
- <div id="roi-banner" style="display:none;background:var(--ok-dim);border:1px solid rgba(61,214,140,0.2);border-radius:var(--radius-sm);padding:12px 14px;margin-bottom:14px;">
434
- <div style="font-size:22px;font-weight:700;color:var(--ok);letter-spacing:-0.5px;" id="roi-value">&ndash;</div>
435
- <div style="font-size:11px;color:var(--text-muted);margin-top:2px;" id="roi-desc">fewer tool calls when Memor injects context</div>
436
- </div>
437
- <div class="side-stats" id="roi-side">
438
- <div class="side-stat">
439
- <div class="side-stat-label">Avg Tools / Turn (with recall)</div>
440
- <div class="side-stat-value" id="roi-tools-with">&ndash;</div>
441
- <div class="side-stat-sub">when Memor injected context</div>
442
- </div>
443
- <div class="side-stat">
444
- <div class="side-stat-label">Avg Tools / Turn (without)</div>
445
- <div class="side-stat-value" id="roi-tools-without">&ndash;</div>
446
- <div class="side-stat-sub">when no context was injected</div>
447
- </div>
448
- <div class="side-stat">
449
- <div class="side-stat-label">Turns Measured</div>
450
- <div class="side-stat-value" id="roi-turns">&ndash;</div>
451
- <div class="side-stat-sub" id="roi-turns-sub">with vs. without recall</div>
452
- </div>
453
- </div>
454
- </div>
455
462
  </div>
456
463
  </section>
457
464
 
@@ -539,7 +546,7 @@
539
546
  function badge(status) {
540
547
  var map = {
541
548
  ok: ['badge-ok', 'Success'],
542
- extractive_only: ['badge-extractive','Extractive'],
549
+ extractive_only: ['badge-extractive','No Distill'],
543
550
  no_hits: ['badge-no_hits', 'No Hits'],
544
551
  };
545
552
  var m = map[status] || ['badge-no_hits', status];
@@ -669,21 +676,20 @@
669
676
  var recallsByDay = data.map(function(d) { return d.recalls; });
670
677
  renderMiniBars('mb-recalls', recallsByDay);
671
678
 
679
+ var maxCells = 30;
672
680
  data.forEach(function(d) {
673
- var pct = (d.recalls / maxRecalls) * 100;
674
- var hitPct = d.hits ? (d.hits / d.recalls) * 100 : 0;
675
- var missPct = 100 - hitPct;
676
- var hitH = Math.max(0, pct * hitPct / 100);
677
- var missH = Math.max(0, pct * missPct / 100);
681
+ var totalCells = Math.max(1, Math.round((d.recalls / maxRecalls) * maxCells));
682
+ var hitCells = Math.round((d.hits / d.recalls) * totalCells);
683
+ var missCells = totalCells - hitCells;
678
684
 
679
685
  var dayStr = d.day.slice(5);
680
686
  var group = el('div', {class: 'bar-group'});
687
+ var cells = '';
688
+ for (var i = 0; i < hitCells; i++) cells += '<div class="bar-cell bar-cell-hit"></div>';
689
+ for (var i = 0; i < missCells; i++) cells += '<div class="bar-cell bar-cell-miss"></div>';
681
690
  group.innerHTML =
682
691
  '<div class="bar-tooltip">' + esc(d.day) + '<br>' + d.recalls + ' recalls · ' + (d.hits||0) + ' hits</div>' +
683
- '<div class="bar-stack">' +
684
- '<div class="bar-hits" style="height:' + hitH + '%"></div>' +
685
- '<div class="bar-miss" style="height:' + missH + '%"></div>' +
686
- '</div>' +
692
+ '<div class="bar-stack">' + cells + '</div>' +
687
693
  '<div class="bar-label">' + dayStr + '</div>';
688
694
  container.appendChild(group);
689
695
  });
@@ -725,7 +731,7 @@
725
731
  document.getElementById('th-col3').textContent = 'Avg Score';
726
732
  document.getElementById('th-col4').textContent = 'OK';
727
733
  document.getElementById('th-col5').textContent = 'No Hits';
728
- document.getElementById('th-col6').textContent = 'Extractive';
734
+ document.getElementById('th-col6').textContent = 'No Distill';
729
735
  }
730
736
 
731
737
  sorted.forEach(function(row) {
@@ -809,36 +815,113 @@
809
815
 
810
816
  /* ── ROI renderer ─────────────────────────────────────── */
811
817
  function renderROI(data) {
818
+ var card = document.getElementById('roi-card');
819
+ var roiValue = document.getElementById('roi-value');
820
+ var roiDesc = document.getElementById('roi-desc');
821
+
812
822
  document.getElementById('roi-tools-with').textContent = data.avg_tools_with_recall;
813
823
  document.getElementById('roi-tools-without').textContent = data.avg_tools_without_recall;
814
- document.getElementById('roi-turns').textContent =
815
- (data.turns_with_recall + data.turns_without_recall).toLocaleString();
824
+ var totalTurns = data.turns_with_recall + data.turns_without_recall;
825
+ document.getElementById('roi-turns').textContent = totalTurns.toLocaleString();
816
826
  document.getElementById('roi-turns-sub').textContent =
817
- data.turns_with_recall + ' with recall · ' + data.turns_without_recall + ' without';
818
-
819
- var banner = document.getElementById('roi-banner');
820
- var roiValue = document.getElementById('roi-value');
821
- var roiDesc = document.getElementById('roi-desc');
822
- banner.style.display = 'none';
823
- banner.style.background = 'var(--ok-dim)';
824
- banner.style.borderColor = 'rgba(61,214,140,0.2)';
825
- roiValue.style.color = 'var(--ok)';
826
- roiValue.textContent = '–';
827
- roiDesc.textContent = 'fewer tool calls when Memor injects context';
827
+ data.turns_with_recall + ' with · ' + data.turns_without_recall + ' without';
828
828
 
829
829
  if (data.tool_call_reduction_pct > 0 && data.turns_with_recall >= 5 && data.turns_without_recall >= 5) {
830
830
  roiValue.textContent = data.tool_call_reduction_pct + '% fewer';
831
- banner.style.display = 'block';
831
+ roiValue.style.color = 'var(--ok)';
832
+ roiDesc.textContent = 'fewer tool calls when Memor injects context';
833
+ card.style.display = 'block';
832
834
  } else if (data.tool_call_reduction_pct < 0 && data.turns_with_recall >= 5) {
833
835
  roiValue.textContent = Math.abs(data.tool_call_reduction_pct) + '% more';
834
836
  roiValue.style.color = 'var(--warn)';
835
837
  roiDesc.textContent = 'tool calls with recall — investigating...';
836
- banner.style.display = 'block';
837
- banner.style.background = 'var(--warn-dim)';
838
- banner.style.borderColor = 'rgba(232,147,32,0.2)';
838
+ card.style.display = 'block';
839
839
  }
840
840
  }
841
841
 
842
+ function renderROITrend(data) {
843
+ var container = document.getElementById('roi-sparkline');
844
+ var tooltip = document.getElementById('roi-spark-tooltip');
845
+ if (!container || !data || !data.length) return;
846
+
847
+ var W = container.clientWidth || 400;
848
+ var H = 80;
849
+ var padL = 32, padR = 8, padT = 8, padB = 20;
850
+ var plotW = W - padL - padR;
851
+ var plotH = H - padT - padB;
852
+
853
+ var vals = data.map(function(d) { return d.reduction_pct; });
854
+ var minV = Math.min.apply(null, vals);
855
+ var maxV = Math.max.apply(null, vals);
856
+ var range = (maxV - minV) || 1;
857
+ var yFloor = Math.min(0, minV);
858
+ var yCeil = Math.max(maxV, 10);
859
+ range = (yCeil - yFloor) || 1;
860
+
861
+ function x(i) { return padL + (data.length > 1 ? (i / (data.length - 1)) * plotW : plotW / 2); }
862
+ function y(v) { return padT + plotH - ((v - yFloor) / range) * plotH; }
863
+
864
+ var lineColor = vals[vals.length - 1] >= 0 ? '#3dd68c' : '#e89320';
865
+ var gradId = 'roiGrad';
866
+
867
+ var pts = data.map(function(d, i) { return x(i) + ',' + y(d.reduction_pct); });
868
+ var linePath = 'M' + pts.join(' L');
869
+ var areaPath = linePath + ' L' + x(data.length - 1) + ',' + y(yFloor) + ' L' + x(0) + ',' + y(yFloor) + ' Z';
870
+
871
+ var gridLines = '';
872
+ var steps = [yFloor, Math.round((yFloor + yCeil) / 2), yCeil];
873
+ steps.forEach(function(v) {
874
+ var yy = y(v);
875
+ gridLines += '<line x1="' + padL + '" y1="' + yy + '" x2="' + (W - padR) + '" y2="' + yy + '" stroke="var(--border-light)" stroke-width="0.5" stroke-dasharray="3,3"/>';
876
+ gridLines += '<text x="' + (padL - 4) + '" y="' + (yy + 3) + '" fill="var(--text-muted)" font-size="8" text-anchor="end">' + v + '%</text>';
877
+ });
878
+
879
+ var xLabels = '';
880
+ data.forEach(function(d, i) {
881
+ xLabels += '<text x="' + x(i) + '" y="' + (H - 2) + '" fill="var(--text-muted)" font-size="8" text-anchor="middle">' + d.day.slice(5) + '</text>';
882
+ });
883
+
884
+ var dots = data.map(function(d, i) {
885
+ return '<circle cx="' + x(i) + '" cy="' + y(d.reduction_pct) + '" r="4" fill="' + lineColor + '" stroke="var(--surface)" stroke-width="2" style="opacity:0;" class="roi-dot" data-idx="' + i + '"/>';
886
+ }).join('');
887
+
888
+ var hitAreas = data.map(function(d, i) {
889
+ return '<rect x="' + (x(i) - plotW / data.length / 2) + '" y="' + padT + '" width="' + (plotW / data.length) + '" height="' + plotH + '" fill="transparent" class="roi-hit" data-idx="' + i + '"/>';
890
+ }).join('');
891
+
892
+ container.innerHTML = '<svg width="' + W + '" height="' + H + '" style="display:block;">' +
893
+ '<defs><linearGradient id="' + gradId + '" x1="0" y1="0" x2="0" y2="1">' +
894
+ '<stop offset="0%" stop-color="' + lineColor + '" stop-opacity="0.3"/>' +
895
+ '<stop offset="100%" stop-color="' + lineColor + '" stop-opacity="0.0"/>' +
896
+ '</linearGradient></defs>' +
897
+ gridLines + xLabels +
898
+ '<path d="' + areaPath + '" fill="url(#' + gradId + ')"/>' +
899
+ '<path d="' + linePath + '" fill="none" stroke="' + lineColor + '" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"/>' +
900
+ dots + hitAreas +
901
+ '</svg>';
902
+
903
+ container.querySelectorAll('.roi-hit').forEach(function(rect) {
904
+ rect.addEventListener('mouseenter', function() {
905
+ var i = parseInt(rect.getAttribute('data-idx'));
906
+ var d = data[i];
907
+ var dot = container.querySelectorAll('.roi-dot')[i];
908
+ if (dot) dot.style.opacity = '1';
909
+ tooltip.innerHTML = '<strong>' + d.day + '</strong><br>' +
910
+ d.reduction_pct + '% fewer tools<br>' +
911
+ d.turns_with + ' recall / ' + d.turns_without + ' no-recall turns';
912
+ tooltip.style.display = 'block';
913
+ var tx = Math.min(x(i), W - 160);
914
+ tooltip.style.left = tx + 'px';
915
+ });
916
+ rect.addEventListener('mouseleave', function() {
917
+ var i = parseInt(rect.getAttribute('data-idx'));
918
+ var dot = container.querySelectorAll('.roi-dot')[i];
919
+ if (dot) dot.style.opacity = '0';
920
+ tooltip.style.display = 'none';
921
+ });
922
+ });
923
+ }
924
+
842
925
  /* ── Data loaders ──────────────────────────────────────── */
843
926
  async function loadSummary() { try { renderSummary(await api('/api/summary')); } catch(e) { console.warn('summary',e); } }
844
927
  async function loadProjects() { try { renderProjects(await api('/api/projects')); } catch(e) { console.warn('projects',e); } }
@@ -847,6 +930,7 @@
847
930
  async function loadHealth() { try { renderHealth(await api('/api/health')); } catch(e) { console.warn('health',e); } }
848
931
  async function loadTrend() { try { renderTrend(await api('/api/recall-trend?days=30')); } catch(e) { console.warn('trend',e); } }
849
932
  async function loadROI() { try { renderROI(await api('/api/roi')); } catch(e) { console.warn('roi',e); } }
933
+ async function loadROITrend() { try { renderROITrend(await api('/api/roi-trend')); } catch(e) { console.warn('roi-trend',e); } }
850
934
  async function loadRecalls() {
851
935
  try {
852
936
  var url = '/api/recalls?limit=50' + (projectFilter ? '&project=' + encodeURIComponent(projectFilter) : '');
@@ -855,7 +939,7 @@
855
939
  }
856
940
 
857
941
  async function refresh() {
858
- await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI()]);
942
+ await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI(), loadROITrend()]);
859
943
  document.getElementById('last-updated').textContent = new Date().toLocaleTimeString();
860
944
 
861
945
  renderMiniBars('mb-chunks', null);
@@ -0,0 +1,176 @@
1
+ """Counterfactual eval harness — win/tie/loss vs no-memory baseline.
2
+
3
+ For each held-out session, uses the opening turn as query, retrieves context
4
+ from prior sessions, and asks an LLM judge whether the recalled context would
5
+ have helped (win), been irrelevant (tie), or misled the agent (loss).
6
+
7
+ The key metric is do-no-harm rate = 1 - loss_rate."""
8
+ from __future__ import annotations
9
+
10
+ import enum
11
+ import json
12
+ import re
13
+ import time
14
+ from dataclasses import dataclass
15
+
16
+ from memor.types import Artifact, Scope
17
+ from memor.retrieve.retriever import Retriever
18
+
19
+
20
+ class Outcome(enum.Enum):
21
+ WIN = "win"
22
+ TIE = "tie"
23
+ LOSS = "loss"
24
+
25
+
26
+ @dataclass
27
+ class CounterfactualCase:
28
+ query: str
29
+ holdout_texts: list[str]
30
+ scope_project: str
31
+ session_id: str
32
+
33
+
34
+ @dataclass
35
+ class CounterfactualVerdict:
36
+ outcome: Outcome
37
+ reasoning: str
38
+ confidence: float
39
+
40
+
41
+ COUNTERFACTUAL_PROMPT = """You are evaluating whether recalled memory context helped or hurt a coding agent.
42
+
43
+ The agent received this task:
44
+ ---
45
+ {query}
46
+ ---
47
+
48
+ Here is what actually happened next in the session (ground truth — the agent didn't see this):
49
+ ---
50
+ {holdout}
51
+ ---
52
+
53
+ Here is the context that was recalled from prior sessions and injected into the agent's prompt:
54
+ ---
55
+ {recalled_context}
56
+ ---
57
+
58
+ Compare the recalled context against what actually happened. Score as:
59
+
60
+ - **win**: The recalled context directly anticipates what happened, would save the agent exploration/time, or would prevent repeating a past mistake. The agent would have been measurably better with this context.
61
+ - **tie**: The recalled context is irrelevant to what happened. Neither helpful nor harmful — the agent would have arrived at the same outcome either way.
62
+ - **loss**: The recalled context is stale, contradictory, or misleading. It would have sent the agent down the wrong path or caused confusion.
63
+
64
+ When uncertain, default to **tie** — only score win when the context clearly helps, and loss when it clearly hurts.
65
+
66
+ Return STRICT JSON: {{"outcome": "win"|"tie"|"loss", "reasoning": "<one sentence>", "confidence": <float 0-1>}}"""
67
+
68
+
69
+ def build_cases_from_store(store, *, project: str,
70
+ holdout_turns: int = 2,
71
+ min_session_turns: int = 4,
72
+ min_prior_sessions: int = 1) -> list[CounterfactualCase]:
73
+ rows = store.db.execute(
74
+ "SELECT * FROM artifacts WHERE project=? AND kind='session_chunk' AND active=1",
75
+ (project,)).fetchall()
76
+ artifacts = [store._row_to_artifact(r) for r in rows]
77
+
78
+ by_session: dict[str, list[Artifact]] = {}
79
+ for a in artifacts:
80
+ by_session.setdefault(a.meta.get("session_id", "?"), []).append(a)
81
+ for v in by_session.values():
82
+ v.sort(key=lambda a: a.meta.get("ord", 0))
83
+ sessions = sorted(by_session.items(), key=lambda kv: kv[1][0].created_at)
84
+
85
+ cases: list[CounterfactualCase] = []
86
+ for idx, (sid, chunks) in enumerate(sessions):
87
+ if idx < min_prior_sessions:
88
+ continue
89
+ if len(chunks) < min_session_turns:
90
+ continue
91
+ query = chunks[0].text
92
+ holdout = [c.text for c in chunks[-holdout_turns:]]
93
+ cases.append(CounterfactualCase(
94
+ query=query, holdout_texts=holdout,
95
+ scope_project=project, session_id=sid))
96
+ return cases
97
+
98
+
99
+ def _extract_json(raw: str) -> dict:
100
+ m = re.search(r"\{.*\}", raw, re.DOTALL)
101
+ if m:
102
+ return json.loads(m.group())
103
+ return json.loads(raw)
104
+
105
+
106
+ def parse_verdict_json(raw: str) -> CounterfactualVerdict:
107
+ data = _extract_json(raw)
108
+ outcome_str = data.get("outcome", "tie").lower().strip()
109
+ outcome = Outcome(outcome_str) if outcome_str in ("win", "tie", "loss") else Outcome.TIE
110
+ return CounterfactualVerdict(
111
+ outcome=outcome,
112
+ reasoning=data.get("reasoning", ""),
113
+ confidence=float(data.get("confidence", 0.5)),
114
+ )
115
+
116
+
117
+ def run_case(case: CounterfactualCase, *, store, embedder, llm,
118
+ k: int = 8) -> CounterfactualVerdict:
119
+ scope = Scope(project=case.scope_project)
120
+ r = Retriever(store, embedder, k=k)
121
+ trace = r.query(case.query, scope)
122
+
123
+ if not trace.hits:
124
+ return CounterfactualVerdict(
125
+ outcome=Outcome.TIE,
126
+ reasoning="No context recalled — nothing to evaluate",
127
+ confidence=1.0)
128
+
129
+ recalled = "\n\n".join(
130
+ f"[{h.artifact.kind}] {h.artifact.text}" for h in trace.hits)
131
+ holdout = "\n".join(case.holdout_texts)
132
+
133
+ prompt = COUNTERFACTUAL_PROMPT.format(
134
+ query=case.query, holdout=holdout, recalled_context=recalled)
135
+ raw = llm.complete(prompt)
136
+ return parse_verdict_json(raw)
137
+
138
+
139
+ def run_suite(cases: list[CounterfactualCase], *, store, embedder, llm,
140
+ k: int = 8) -> dict:
141
+ verdicts = []
142
+ for c in cases:
143
+ v = run_case(c, store=store, embedder=embedder, llm=llm, k=k)
144
+ verdicts.append((c, v))
145
+ verdict_list = [v for _, v in verdicts]
146
+ summary = summarize_verdicts(verdict_list)
147
+ summary["cases"] = [
148
+ {"session_id": c.session_id, "query_preview": c.query[:100],
149
+ "outcome": v.outcome.value, "reasoning": v.reasoning,
150
+ "confidence": v.confidence}
151
+ for c, v in verdicts
152
+ ]
153
+ return summary
154
+
155
+
156
+ def summarize_verdicts(verdicts: list[CounterfactualVerdict]) -> dict:
157
+ n = len(verdicts)
158
+ if n == 0:
159
+ return {"n_cases": 0, "win_count": 0, "tie_count": 0, "loss_count": 0,
160
+ "win_pct": 0.0, "tie_pct": 0.0, "loss_pct": 0.0,
161
+ "do_no_harm_pct": 100.0}
162
+
163
+ wins = sum(1 for v in verdicts if v.outcome == Outcome.WIN)
164
+ ties = sum(1 for v in verdicts if v.outcome == Outcome.TIE)
165
+ losses = sum(1 for v in verdicts if v.outcome == Outcome.LOSS)
166
+
167
+ return {
168
+ "n_cases": n,
169
+ "win_count": wins,
170
+ "tie_count": ties,
171
+ "loss_count": losses,
172
+ "win_pct": round(wins / n * 100, 1),
173
+ "tie_pct": round(ties / n * 100, 1),
174
+ "loss_pct": round(losses / n * 100, 1),
175
+ "do_no_harm_pct": round((wins + ties) / n * 100, 1),
176
+ }
@@ -234,6 +234,19 @@ class SqliteStore:
234
234
  self.db.commit()
235
235
  return cur.lastrowid
236
236
 
237
+ def get_latest_eval(self, eval_type: str = "counterfactual") -> dict | None:
238
+ row = self.db.execute(
239
+ "SELECT * FROM eval_runs WHERE json_extract(config, '$.type') = ? "
240
+ "ORDER BY created_at DESC LIMIT 1", (eval_type,)).fetchone()
241
+ if not row:
242
+ return None
243
+ return {
244
+ "id": row["id"],
245
+ "created_at": row["created_at"],
246
+ "config": json.loads(row["config"]),
247
+ "metrics": json.loads(row["metrics"]),
248
+ }
249
+
237
250
  def log_recall(self, project: str, query_preview: str, hits_count: int,
238
251
  top_score: float, tokens_injected: int, latency_ms: float,
239
252
  status: str, session_id: str = "") -> None:
@@ -571,6 +584,43 @@ class SqliteStore:
571
584
  "tool_call_reduction_pct": reduction,
572
585
  }
573
586
 
587
+ def get_roi_trend(self, project: str | None = None,
588
+ days: int = 30) -> list[dict]:
589
+ where = "WHERE user_timestamp > 0"
590
+ params: list = []
591
+ if project:
592
+ where += " AND project = ?"
593
+ params.append(project)
594
+
595
+ rows = self.db.execute(f"""
596
+ SELECT date(user_timestamp, 'unixepoch', 'localtime') AS day,
597
+ AVG(CASE WHEN had_recall = 1 THEN tool_call_count END) AS avg_with,
598
+ AVG(CASE WHEN had_recall = 0 THEN tool_call_count END) AS avg_without,
599
+ COUNT(CASE WHEN had_recall = 1 THEN 1 END) AS turns_with,
600
+ COUNT(CASE WHEN had_recall = 0 THEN 1 END) AS turns_without,
601
+ COUNT(*) AS turns_total
602
+ FROM turn_metrics {where}
603
+ GROUP BY day
604
+ HAVING turns_with > 0 AND turns_without > 0
605
+ ORDER BY day
606
+ """, params).fetchall()
607
+
608
+ result = []
609
+ for r in rows:
610
+ avg_w = round(r["avg_with"], 2)
611
+ avg_wo = round(r["avg_without"], 2)
612
+ reduction = round((1 - avg_w / avg_wo) * 100, 1) if avg_wo > 0 else 0
613
+ result.append({
614
+ "day": r["day"],
615
+ "avg_with": avg_w,
616
+ "avg_without": avg_wo,
617
+ "reduction_pct": reduction,
618
+ "turns_with": r["turns_with"],
619
+ "turns_without": r["turns_without"],
620
+ "turns_total": r["turns_total"],
621
+ })
622
+ return result
623
+
574
624
  def get_onboarding_status(self) -> str:
575
625
  chunks = self.db.execute(
576
626
  "SELECT COUNT(*) as c FROM artifacts WHERE kind='session_chunk' AND active=1"