memor-cli 0.4.1__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memor_cli-0.4.1/memor_cli.egg-info → memor_cli-0.5.0}/PKG-INFO +3 -2
- {memor_cli-0.4.1 → memor_cli-0.5.0}/README.md +2 -1
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/cli.py +36 -3
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/server.py +13 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/static/index.html +89 -1
- memor_cli-0.5.0/memor/eval/counterfactual.py +176 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/store/sqlite_store.py +50 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0/memor_cli.egg-info}/PKG-INFO +3 -2
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/SOURCES.txt +3 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/pyproject.toml +1 -1
- memor_cli-0.5.0/tests/test_counterfactual.py +104 -0
- memor_cli-0.5.0/tests/test_roi_trend.py +87 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/LICENSE +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/daemon.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/dashboard/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/distiller.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/distill/extractive.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/api.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/fake.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/embed/local.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/base.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/claude_mem.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/baselines/graphiti.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/dataset.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/embed_benchmark.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/judge.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/metrics.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/eval/runner.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/feedback.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/hook_cli.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/hook_server.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/claude_code.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/ingest/documents.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/interfaces.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/anthropic.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/base.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/llm/openai_compat.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/project.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/query_complexity.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/recall.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/redact.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/retrieve/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/retrieve/retriever.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/service.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/session_context.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/store/__init__.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/tokencount.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/turn_metrics.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor/types.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/dependency_links.txt +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/entry_points.txt +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/requires.txt +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/memor_cli.egg-info/top_level.txt +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/setup.cfg +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_cli_smoke.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_daemon.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dashboard.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dataset_builder.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_dimension_safety.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_distiller.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_embed.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_embed_benchmark.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_eval_ablation.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_eval_runner.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_external_baselines.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_extractive.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_feedback.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hook.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hook_server.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_hybrid_retrieval.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_ingest_claude_code.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_ingest_documents.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_install_hook.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_interfaces.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_judge.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_metrics.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_noise_filter.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_project_resolver.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_quality_gate.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_query_complexity.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_recall_core.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_redact.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_retriever.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_semantic_feedback.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_service.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_session_context.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_skill_recall.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_store.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_supersession.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_tokencount.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_turn_metrics.py +0 -0
- {memor_cli-0.4.1 → memor_cli-0.5.0}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memor-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
|
|
5
5
|
Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -46,7 +46,7 @@ Dynamic: license-file
|
|
|
46
46
|
```
|
|
47
47
|
|
|
48
48
|
[](LICENSE)
|
|
49
|
-
[]()
|
|
50
50
|
[]()
|
|
51
51
|
[](https://pypi.org/project/memor-cli/)
|
|
52
52
|
|
|
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
|
|
|
200
200
|
memor ingest-doc <file> Ingest a markdown document
|
|
201
201
|
memor distill --project <name> Run distillation manually
|
|
202
202
|
memor eval <cases.json> Run eval suite
|
|
203
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
203
204
|
memor bench-embed --project <name> Compare embedding models
|
|
204
205
|
```
|
|
205
206
|
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
```
|
|
10
10
|
|
|
11
11
|
[](LICENSE)
|
|
12
|
-
[]()
|
|
13
13
|
[]()
|
|
14
14
|
[](https://pypi.org/project/memor-cli/)
|
|
15
15
|
|
|
@@ -163,6 +163,7 @@ memor ingest-project <dir> Bulk ingest a project directory
|
|
|
163
163
|
memor ingest-doc <file> Ingest a markdown document
|
|
164
164
|
memor distill --project <name> Run distillation manually
|
|
165
165
|
memor eval <cases.json> Run eval suite
|
|
166
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
166
167
|
memor bench-embed --project <name> Compare embedding models
|
|
167
168
|
```
|
|
168
169
|
|
|
@@ -78,9 +78,10 @@ MAINTENANCE
|
|
|
78
78
|
memor version Print the installed version
|
|
79
79
|
|
|
80
80
|
EVALUATION
|
|
81
|
-
memor eval <cases.json>
|
|
82
|
-
memor eval-judge --project <name>
|
|
83
|
-
memor
|
|
81
|
+
memor eval <cases.json> Run eval suite
|
|
82
|
+
memor eval-judge --project <name> LLM-as-judge evaluation
|
|
83
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
84
|
+
memor bench-embed --project <name> Compare embedding models
|
|
84
85
|
|
|
85
86
|
CONFIGURATION
|
|
86
87
|
Everything works locally with zero API keys.
|
|
@@ -191,6 +192,38 @@ def eval_judge_cmd(project: str = typer.Option(...), db: str = "memor.db",
|
|
|
191
192
|
typer.echo(f"Judge eval complete: mean relevance = {summary['mean_relevance']:.3f}")
|
|
192
193
|
|
|
193
194
|
|
|
195
|
+
@app.command("eval-counterfactual")
|
|
196
|
+
def eval_counterfactual_cmd(project: str = typer.Option(...), db: str = "memor.db",
|
|
197
|
+
k: int = 8, fake: bool = False,
|
|
198
|
+
llm_provider: str = "anthropic",
|
|
199
|
+
llm_model: str = "claude-sonnet-4-6",
|
|
200
|
+
holdout: int = 2):
|
|
201
|
+
"""Counterfactual eval: win/tie/loss vs no-memory baseline. Requires an LLM API key."""
|
|
202
|
+
from memor.eval.counterfactual import build_cases_from_store, run_suite
|
|
203
|
+
e = _embedder(fake); s = SqliteStore(_db_path(db), dim=e.dim)
|
|
204
|
+
cases = build_cases_from_store(s, project=project, holdout_turns=holdout)
|
|
205
|
+
if not cases:
|
|
206
|
+
typer.echo("No cases — need at least 2 sessions with >= 4 turns each.")
|
|
207
|
+
raise typer.Exit(1)
|
|
208
|
+
typer.echo(f"Built {len(cases)} counterfactual cases. Running evaluation...")
|
|
209
|
+
if llm_provider == "anthropic":
|
|
210
|
+
from memor.llm.anthropic import AnthropicLLM
|
|
211
|
+
llm = AnthropicLLM(model=llm_model)
|
|
212
|
+
else:
|
|
213
|
+
from memor.llm.openai_compat import OpenAICompatLLM
|
|
214
|
+
import os
|
|
215
|
+
llm = OpenAICompatLLM(base_url=os.environ.get("OPENAI_BASE_URL", "http://localhost:11434/v1"),
|
|
216
|
+
api_key=os.environ.get("OPENAI_API_KEY", ""), model=llm_model)
|
|
217
|
+
summary = run_suite(cases, store=s, embedder=e, llm=llm, k=k)
|
|
218
|
+
typer.echo(json.dumps({k: v for k, v in summary.items() if k != "cases"}, indent=2))
|
|
219
|
+
typer.echo("")
|
|
220
|
+
typer.echo(f" Win: {summary['win_count']}/{summary['n_cases']} ({summary['win_pct']}%)")
|
|
221
|
+
typer.echo(f" Tie: {summary['tie_count']}/{summary['n_cases']} ({summary['tie_pct']}%)")
|
|
222
|
+
typer.echo(f" Loss: {summary['loss_count']}/{summary['n_cases']} ({summary['loss_pct']}%)")
|
|
223
|
+
typer.echo(f" Do-no-harm: {summary['do_no_harm_pct']}%")
|
|
224
|
+
s.save_eval_run({"type": "counterfactual", "k": k, "project": project, "holdout": holdout}, summary)
|
|
225
|
+
|
|
226
|
+
|
|
194
227
|
@app.command("bench-embed")
|
|
195
228
|
def bench_embed(project: str = typer.Option(...), db: str = "memor.db",
|
|
196
229
|
k: int = 8, fake: bool = False):
|
|
@@ -139,6 +139,19 @@ def create_app(db_path: str | None = None) -> FastAPI:
|
|
|
139
139
|
store = _store()
|
|
140
140
|
return store.get_token_roi(project=project)
|
|
141
141
|
|
|
142
|
+
@app.get("/api/roi-trend")
|
|
143
|
+
def roi_trend(project: str | None = Query(None)):
|
|
144
|
+
store = _store()
|
|
145
|
+
return store.get_roi_trend(project=project)
|
|
146
|
+
|
|
147
|
+
@app.get("/api/eval/latest")
|
|
148
|
+
def eval_latest(eval_type: str = Query("counterfactual")):
|
|
149
|
+
store = _store()
|
|
150
|
+
result = store.get_latest_eval(eval_type)
|
|
151
|
+
if not result:
|
|
152
|
+
return {"status": "no_runs"}
|
|
153
|
+
return result
|
|
154
|
+
|
|
142
155
|
@app.get("/api/health")
|
|
143
156
|
def health():
|
|
144
157
|
store = _store()
|
|
@@ -404,6 +404,10 @@
|
|
|
404
404
|
</div>
|
|
405
405
|
</div>
|
|
406
406
|
</div>
|
|
407
|
+
<div style="margin-top:16px;border-top:1px solid var(--border-light);padding-top:12px;position:relative;">
|
|
408
|
+
<div id="roi-sparkline" style="width:100%;height:80px;"></div>
|
|
409
|
+
<div id="roi-spark-tooltip" style="display:none;position:absolute;top:0;background:var(--surface3);border:1px solid var(--border);border-radius:6px;padding:6px 10px;font-size:11px;color:var(--text);white-space:nowrap;pointer-events:none;box-shadow:0 4px 12px rgba(0,0,0,0.4);z-index:10;"></div>
|
|
410
|
+
</div>
|
|
407
411
|
</div>
|
|
408
412
|
</section>
|
|
409
413
|
|
|
@@ -835,6 +839,89 @@
|
|
|
835
839
|
}
|
|
836
840
|
}
|
|
837
841
|
|
|
842
|
+
function renderROITrend(data) {
|
|
843
|
+
var container = document.getElementById('roi-sparkline');
|
|
844
|
+
var tooltip = document.getElementById('roi-spark-tooltip');
|
|
845
|
+
if (!container || !data || !data.length) return;
|
|
846
|
+
|
|
847
|
+
var W = container.clientWidth || 400;
|
|
848
|
+
var H = 80;
|
|
849
|
+
var padL = 32, padR = 8, padT = 8, padB = 20;
|
|
850
|
+
var plotW = W - padL - padR;
|
|
851
|
+
var plotH = H - padT - padB;
|
|
852
|
+
|
|
853
|
+
var vals = data.map(function(d) { return d.reduction_pct; });
|
|
854
|
+
var minV = Math.min.apply(null, vals);
|
|
855
|
+
var maxV = Math.max.apply(null, vals);
|
|
856
|
+
var range = (maxV - minV) || 1;
|
|
857
|
+
var yFloor = Math.min(0, minV);
|
|
858
|
+
var yCeil = Math.max(maxV, 10);
|
|
859
|
+
range = (yCeil - yFloor) || 1;
|
|
860
|
+
|
|
861
|
+
function x(i) { return padL + (data.length > 1 ? (i / (data.length - 1)) * plotW : plotW / 2); }
|
|
862
|
+
function y(v) { return padT + plotH - ((v - yFloor) / range) * plotH; }
|
|
863
|
+
|
|
864
|
+
var lineColor = vals[vals.length - 1] >= 0 ? '#3dd68c' : '#e89320';
|
|
865
|
+
var gradId = 'roiGrad';
|
|
866
|
+
|
|
867
|
+
var pts = data.map(function(d, i) { return x(i) + ',' + y(d.reduction_pct); });
|
|
868
|
+
var linePath = 'M' + pts.join(' L');
|
|
869
|
+
var areaPath = linePath + ' L' + x(data.length - 1) + ',' + y(yFloor) + ' L' + x(0) + ',' + y(yFloor) + ' Z';
|
|
870
|
+
|
|
871
|
+
var gridLines = '';
|
|
872
|
+
var steps = [yFloor, Math.round((yFloor + yCeil) / 2), yCeil];
|
|
873
|
+
steps.forEach(function(v) {
|
|
874
|
+
var yy = y(v);
|
|
875
|
+
gridLines += '<line x1="' + padL + '" y1="' + yy + '" x2="' + (W - padR) + '" y2="' + yy + '" stroke="var(--border-light)" stroke-width="0.5" stroke-dasharray="3,3"/>';
|
|
876
|
+
gridLines += '<text x="' + (padL - 4) + '" y="' + (yy + 3) + '" fill="var(--text-muted)" font-size="8" text-anchor="end">' + v + '%</text>';
|
|
877
|
+
});
|
|
878
|
+
|
|
879
|
+
var xLabels = '';
|
|
880
|
+
data.forEach(function(d, i) {
|
|
881
|
+
xLabels += '<text x="' + x(i) + '" y="' + (H - 2) + '" fill="var(--text-muted)" font-size="8" text-anchor="middle">' + d.day.slice(5) + '</text>';
|
|
882
|
+
});
|
|
883
|
+
|
|
884
|
+
var dots = data.map(function(d, i) {
|
|
885
|
+
return '<circle cx="' + x(i) + '" cy="' + y(d.reduction_pct) + '" r="4" fill="' + lineColor + '" stroke="var(--surface)" stroke-width="2" style="opacity:0;" class="roi-dot" data-idx="' + i + '"/>';
|
|
886
|
+
}).join('');
|
|
887
|
+
|
|
888
|
+
var hitAreas = data.map(function(d, i) {
|
|
889
|
+
return '<rect x="' + (x(i) - plotW / data.length / 2) + '" y="' + padT + '" width="' + (plotW / data.length) + '" height="' + plotH + '" fill="transparent" class="roi-hit" data-idx="' + i + '"/>';
|
|
890
|
+
}).join('');
|
|
891
|
+
|
|
892
|
+
container.innerHTML = '<svg width="' + W + '" height="' + H + '" style="display:block;">' +
|
|
893
|
+
'<defs><linearGradient id="' + gradId + '" x1="0" y1="0" x2="0" y2="1">' +
|
|
894
|
+
'<stop offset="0%" stop-color="' + lineColor + '" stop-opacity="0.3"/>' +
|
|
895
|
+
'<stop offset="100%" stop-color="' + lineColor + '" stop-opacity="0.0"/>' +
|
|
896
|
+
'</linearGradient></defs>' +
|
|
897
|
+
gridLines + xLabels +
|
|
898
|
+
'<path d="' + areaPath + '" fill="url(#' + gradId + ')"/>' +
|
|
899
|
+
'<path d="' + linePath + '" fill="none" stroke="' + lineColor + '" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"/>' +
|
|
900
|
+
dots + hitAreas +
|
|
901
|
+
'</svg>';
|
|
902
|
+
|
|
903
|
+
container.querySelectorAll('.roi-hit').forEach(function(rect) {
|
|
904
|
+
rect.addEventListener('mouseenter', function() {
|
|
905
|
+
var i = parseInt(rect.getAttribute('data-idx'));
|
|
906
|
+
var d = data[i];
|
|
907
|
+
var dot = container.querySelectorAll('.roi-dot')[i];
|
|
908
|
+
if (dot) dot.style.opacity = '1';
|
|
909
|
+
tooltip.innerHTML = '<strong>' + d.day + '</strong><br>' +
|
|
910
|
+
d.reduction_pct + '% fewer tools<br>' +
|
|
911
|
+
d.turns_with + ' recall / ' + d.turns_without + ' no-recall turns';
|
|
912
|
+
tooltip.style.display = 'block';
|
|
913
|
+
var tx = Math.min(x(i), W - 160);
|
|
914
|
+
tooltip.style.left = tx + 'px';
|
|
915
|
+
});
|
|
916
|
+
rect.addEventListener('mouseleave', function() {
|
|
917
|
+
var i = parseInt(rect.getAttribute('data-idx'));
|
|
918
|
+
var dot = container.querySelectorAll('.roi-dot')[i];
|
|
919
|
+
if (dot) dot.style.opacity = '0';
|
|
920
|
+
tooltip.style.display = 'none';
|
|
921
|
+
});
|
|
922
|
+
});
|
|
923
|
+
}
|
|
924
|
+
|
|
838
925
|
/* ── Data loaders ──────────────────────────────────────── */
|
|
839
926
|
async function loadSummary() { try { renderSummary(await api('/api/summary')); } catch(e) { console.warn('summary',e); } }
|
|
840
927
|
async function loadProjects() { try { renderProjects(await api('/api/projects')); } catch(e) { console.warn('projects',e); } }
|
|
@@ -843,6 +930,7 @@
|
|
|
843
930
|
async function loadHealth() { try { renderHealth(await api('/api/health')); } catch(e) { console.warn('health',e); } }
|
|
844
931
|
async function loadTrend() { try { renderTrend(await api('/api/recall-trend?days=30')); } catch(e) { console.warn('trend',e); } }
|
|
845
932
|
async function loadROI() { try { renderROI(await api('/api/roi')); } catch(e) { console.warn('roi',e); } }
|
|
933
|
+
async function loadROITrend() { try { renderROITrend(await api('/api/roi-trend')); } catch(e) { console.warn('roi-trend',e); } }
|
|
846
934
|
async function loadRecalls() {
|
|
847
935
|
try {
|
|
848
936
|
var url = '/api/recalls?limit=50' + (projectFilter ? '&project=' + encodeURIComponent(projectFilter) : '');
|
|
@@ -851,7 +939,7 @@
|
|
|
851
939
|
}
|
|
852
940
|
|
|
853
941
|
async function refresh() {
|
|
854
|
-
await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI()]);
|
|
942
|
+
await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI(), loadROITrend()]);
|
|
855
943
|
document.getElementById('last-updated').textContent = new Date().toLocaleTimeString();
|
|
856
944
|
|
|
857
945
|
renderMiniBars('mb-chunks', null);
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Counterfactual eval harness — win/tie/loss vs no-memory baseline.
|
|
2
|
+
|
|
3
|
+
For each held-out session, uses the opening turn as query, retrieves context
|
|
4
|
+
from prior sessions, and asks an LLM judge whether the recalled context would
|
|
5
|
+
have helped (win), been irrelevant (tie), or misled the agent (loss).
|
|
6
|
+
|
|
7
|
+
The key metric is do-no-harm rate = 1 - loss_rate."""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import enum
|
|
11
|
+
import json
|
|
12
|
+
import re
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
|
|
16
|
+
from memor.types import Artifact, Scope
|
|
17
|
+
from memor.retrieve.retriever import Retriever
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Outcome(enum.Enum):
|
|
21
|
+
WIN = "win"
|
|
22
|
+
TIE = "tie"
|
|
23
|
+
LOSS = "loss"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class CounterfactualCase:
|
|
28
|
+
query: str
|
|
29
|
+
holdout_texts: list[str]
|
|
30
|
+
scope_project: str
|
|
31
|
+
session_id: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class CounterfactualVerdict:
|
|
36
|
+
outcome: Outcome
|
|
37
|
+
reasoning: str
|
|
38
|
+
confidence: float
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
COUNTERFACTUAL_PROMPT = """You are evaluating whether recalled memory context helped or hurt a coding agent.
|
|
42
|
+
|
|
43
|
+
The agent received this task:
|
|
44
|
+
---
|
|
45
|
+
{query}
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
Here is what actually happened next in the session (ground truth — the agent didn't see this):
|
|
49
|
+
---
|
|
50
|
+
{holdout}
|
|
51
|
+
---
|
|
52
|
+
|
|
53
|
+
Here is the context that was recalled from prior sessions and injected into the agent's prompt:
|
|
54
|
+
---
|
|
55
|
+
{recalled_context}
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
Compare the recalled context against what actually happened. Score as:
|
|
59
|
+
|
|
60
|
+
- **win**: The recalled context directly anticipates what happened, would save the agent exploration/time, or would prevent repeating a past mistake. The agent would have been measurably better with this context.
|
|
61
|
+
- **tie**: The recalled context is irrelevant to what happened. Neither helpful nor harmful — the agent would have arrived at the same outcome either way.
|
|
62
|
+
- **loss**: The recalled context is stale, contradictory, or misleading. It would have sent the agent down the wrong path or caused confusion.
|
|
63
|
+
|
|
64
|
+
When uncertain, default to **tie** — only score win when the context clearly helps, and loss when it clearly hurts.
|
|
65
|
+
|
|
66
|
+
Return STRICT JSON: {{"outcome": "win"|"tie"|"loss", "reasoning": "<one sentence>", "confidence": <float 0-1>}}"""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def build_cases_from_store(store, *, project: str,
|
|
70
|
+
holdout_turns: int = 2,
|
|
71
|
+
min_session_turns: int = 4,
|
|
72
|
+
min_prior_sessions: int = 1) -> list[CounterfactualCase]:
|
|
73
|
+
rows = store.db.execute(
|
|
74
|
+
"SELECT * FROM artifacts WHERE project=? AND kind='session_chunk' AND active=1",
|
|
75
|
+
(project,)).fetchall()
|
|
76
|
+
artifacts = [store._row_to_artifact(r) for r in rows]
|
|
77
|
+
|
|
78
|
+
by_session: dict[str, list[Artifact]] = {}
|
|
79
|
+
for a in artifacts:
|
|
80
|
+
by_session.setdefault(a.meta.get("session_id", "?"), []).append(a)
|
|
81
|
+
for v in by_session.values():
|
|
82
|
+
v.sort(key=lambda a: a.meta.get("ord", 0))
|
|
83
|
+
sessions = sorted(by_session.items(), key=lambda kv: kv[1][0].created_at)
|
|
84
|
+
|
|
85
|
+
cases: list[CounterfactualCase] = []
|
|
86
|
+
for idx, (sid, chunks) in enumerate(sessions):
|
|
87
|
+
if idx < min_prior_sessions:
|
|
88
|
+
continue
|
|
89
|
+
if len(chunks) < min_session_turns:
|
|
90
|
+
continue
|
|
91
|
+
query = chunks[0].text
|
|
92
|
+
holdout = [c.text for c in chunks[-holdout_turns:]]
|
|
93
|
+
cases.append(CounterfactualCase(
|
|
94
|
+
query=query, holdout_texts=holdout,
|
|
95
|
+
scope_project=project, session_id=sid))
|
|
96
|
+
return cases
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _extract_json(raw: str) -> dict:
|
|
100
|
+
m = re.search(r"\{.*\}", raw, re.DOTALL)
|
|
101
|
+
if m:
|
|
102
|
+
return json.loads(m.group())
|
|
103
|
+
return json.loads(raw)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def parse_verdict_json(raw: str) -> CounterfactualVerdict:
|
|
107
|
+
data = _extract_json(raw)
|
|
108
|
+
outcome_str = data.get("outcome", "tie").lower().strip()
|
|
109
|
+
outcome = Outcome(outcome_str) if outcome_str in ("win", "tie", "loss") else Outcome.TIE
|
|
110
|
+
return CounterfactualVerdict(
|
|
111
|
+
outcome=outcome,
|
|
112
|
+
reasoning=data.get("reasoning", ""),
|
|
113
|
+
confidence=float(data.get("confidence", 0.5)),
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def run_case(case: CounterfactualCase, *, store, embedder, llm,
|
|
118
|
+
k: int = 8) -> CounterfactualVerdict:
|
|
119
|
+
scope = Scope(project=case.scope_project)
|
|
120
|
+
r = Retriever(store, embedder, k=k)
|
|
121
|
+
trace = r.query(case.query, scope)
|
|
122
|
+
|
|
123
|
+
if not trace.hits:
|
|
124
|
+
return CounterfactualVerdict(
|
|
125
|
+
outcome=Outcome.TIE,
|
|
126
|
+
reasoning="No context recalled — nothing to evaluate",
|
|
127
|
+
confidence=1.0)
|
|
128
|
+
|
|
129
|
+
recalled = "\n\n".join(
|
|
130
|
+
f"[{h.artifact.kind}] {h.artifact.text}" for h in trace.hits)
|
|
131
|
+
holdout = "\n".join(case.holdout_texts)
|
|
132
|
+
|
|
133
|
+
prompt = COUNTERFACTUAL_PROMPT.format(
|
|
134
|
+
query=case.query, holdout=holdout, recalled_context=recalled)
|
|
135
|
+
raw = llm.complete(prompt)
|
|
136
|
+
return parse_verdict_json(raw)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def run_suite(cases: list[CounterfactualCase], *, store, embedder, llm,
|
|
140
|
+
k: int = 8) -> dict:
|
|
141
|
+
verdicts = []
|
|
142
|
+
for c in cases:
|
|
143
|
+
v = run_case(c, store=store, embedder=embedder, llm=llm, k=k)
|
|
144
|
+
verdicts.append((c, v))
|
|
145
|
+
verdict_list = [v for _, v in verdicts]
|
|
146
|
+
summary = summarize_verdicts(verdict_list)
|
|
147
|
+
summary["cases"] = [
|
|
148
|
+
{"session_id": c.session_id, "query_preview": c.query[:100],
|
|
149
|
+
"outcome": v.outcome.value, "reasoning": v.reasoning,
|
|
150
|
+
"confidence": v.confidence}
|
|
151
|
+
for c, v in verdicts
|
|
152
|
+
]
|
|
153
|
+
return summary
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def summarize_verdicts(verdicts: list[CounterfactualVerdict]) -> dict:
|
|
157
|
+
n = len(verdicts)
|
|
158
|
+
if n == 0:
|
|
159
|
+
return {"n_cases": 0, "win_count": 0, "tie_count": 0, "loss_count": 0,
|
|
160
|
+
"win_pct": 0.0, "tie_pct": 0.0, "loss_pct": 0.0,
|
|
161
|
+
"do_no_harm_pct": 100.0}
|
|
162
|
+
|
|
163
|
+
wins = sum(1 for v in verdicts if v.outcome == Outcome.WIN)
|
|
164
|
+
ties = sum(1 for v in verdicts if v.outcome == Outcome.TIE)
|
|
165
|
+
losses = sum(1 for v in verdicts if v.outcome == Outcome.LOSS)
|
|
166
|
+
|
|
167
|
+
return {
|
|
168
|
+
"n_cases": n,
|
|
169
|
+
"win_count": wins,
|
|
170
|
+
"tie_count": ties,
|
|
171
|
+
"loss_count": losses,
|
|
172
|
+
"win_pct": round(wins / n * 100, 1),
|
|
173
|
+
"tie_pct": round(ties / n * 100, 1),
|
|
174
|
+
"loss_pct": round(losses / n * 100, 1),
|
|
175
|
+
"do_no_harm_pct": round((wins + ties) / n * 100, 1),
|
|
176
|
+
}
|
|
@@ -234,6 +234,19 @@ class SqliteStore:
|
|
|
234
234
|
self.db.commit()
|
|
235
235
|
return cur.lastrowid
|
|
236
236
|
|
|
237
|
+
def get_latest_eval(self, eval_type: str = "counterfactual") -> dict | None:
|
|
238
|
+
row = self.db.execute(
|
|
239
|
+
"SELECT * FROM eval_runs WHERE json_extract(config, '$.type') = ? "
|
|
240
|
+
"ORDER BY created_at DESC LIMIT 1", (eval_type,)).fetchone()
|
|
241
|
+
if not row:
|
|
242
|
+
return None
|
|
243
|
+
return {
|
|
244
|
+
"id": row["id"],
|
|
245
|
+
"created_at": row["created_at"],
|
|
246
|
+
"config": json.loads(row["config"]),
|
|
247
|
+
"metrics": json.loads(row["metrics"]),
|
|
248
|
+
}
|
|
249
|
+
|
|
237
250
|
def log_recall(self, project: str, query_preview: str, hits_count: int,
|
|
238
251
|
top_score: float, tokens_injected: int, latency_ms: float,
|
|
239
252
|
status: str, session_id: str = "") -> None:
|
|
@@ -571,6 +584,43 @@ class SqliteStore:
|
|
|
571
584
|
"tool_call_reduction_pct": reduction,
|
|
572
585
|
}
|
|
573
586
|
|
|
587
|
+
def get_roi_trend(self, project: str | None = None,
|
|
588
|
+
days: int = 30) -> list[dict]:
|
|
589
|
+
where = "WHERE user_timestamp > 0"
|
|
590
|
+
params: list = []
|
|
591
|
+
if project:
|
|
592
|
+
where += " AND project = ?"
|
|
593
|
+
params.append(project)
|
|
594
|
+
|
|
595
|
+
rows = self.db.execute(f"""
|
|
596
|
+
SELECT date(user_timestamp, 'unixepoch', 'localtime') AS day,
|
|
597
|
+
AVG(CASE WHEN had_recall = 1 THEN tool_call_count END) AS avg_with,
|
|
598
|
+
AVG(CASE WHEN had_recall = 0 THEN tool_call_count END) AS avg_without,
|
|
599
|
+
COUNT(CASE WHEN had_recall = 1 THEN 1 END) AS turns_with,
|
|
600
|
+
COUNT(CASE WHEN had_recall = 0 THEN 1 END) AS turns_without,
|
|
601
|
+
COUNT(*) AS turns_total
|
|
602
|
+
FROM turn_metrics {where}
|
|
603
|
+
GROUP BY day
|
|
604
|
+
HAVING turns_with > 0 AND turns_without > 0
|
|
605
|
+
ORDER BY day
|
|
606
|
+
""", params).fetchall()
|
|
607
|
+
|
|
608
|
+
result = []
|
|
609
|
+
for r in rows:
|
|
610
|
+
avg_w = round(r["avg_with"], 2)
|
|
611
|
+
avg_wo = round(r["avg_without"], 2)
|
|
612
|
+
reduction = round((1 - avg_w / avg_wo) * 100, 1) if avg_wo > 0 else 0
|
|
613
|
+
result.append({
|
|
614
|
+
"day": r["day"],
|
|
615
|
+
"avg_with": avg_w,
|
|
616
|
+
"avg_without": avg_wo,
|
|
617
|
+
"reduction_pct": reduction,
|
|
618
|
+
"turns_with": r["turns_with"],
|
|
619
|
+
"turns_without": r["turns_without"],
|
|
620
|
+
"turns_total": r["turns_total"],
|
|
621
|
+
})
|
|
622
|
+
return result
|
|
623
|
+
|
|
574
624
|
def get_onboarding_status(self) -> str:
|
|
575
625
|
chunks = self.db.execute(
|
|
576
626
|
"SELECT COUNT(*) as c FROM artifacts WHERE kind='session_chunk' AND active=1"
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memor-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
|
|
5
5
|
Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -46,7 +46,7 @@ Dynamic: license-file
|
|
|
46
46
|
```
|
|
47
47
|
|
|
48
48
|
[](LICENSE)
|
|
49
|
-
[]()
|
|
50
50
|
[]()
|
|
51
51
|
[](https://pypi.org/project/memor-cli/)
|
|
52
52
|
|
|
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
|
|
|
200
200
|
memor ingest-doc <file> Ingest a markdown document
|
|
201
201
|
memor distill --project <name> Run distillation manually
|
|
202
202
|
memor eval <cases.json> Run eval suite
|
|
203
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
203
204
|
memor bench-embed --project <name> Compare embedding models
|
|
204
205
|
```
|
|
205
206
|
|
|
@@ -28,6 +28,7 @@ memor/embed/api.py
|
|
|
28
28
|
memor/embed/fake.py
|
|
29
29
|
memor/embed/local.py
|
|
30
30
|
memor/eval/__init__.py
|
|
31
|
+
memor/eval/counterfactual.py
|
|
31
32
|
memor/eval/dataset.py
|
|
32
33
|
memor/eval/embed_benchmark.py
|
|
33
34
|
memor/eval/judge.py
|
|
@@ -55,6 +56,7 @@ memor_cli.egg-info/entry_points.txt
|
|
|
55
56
|
memor_cli.egg-info/requires.txt
|
|
56
57
|
memor_cli.egg-info/top_level.txt
|
|
57
58
|
tests/test_cli_smoke.py
|
|
59
|
+
tests/test_counterfactual.py
|
|
58
60
|
tests/test_daemon.py
|
|
59
61
|
tests/test_dashboard.py
|
|
60
62
|
tests/test_dataset_builder.py
|
|
@@ -83,6 +85,7 @@ tests/test_query_complexity.py
|
|
|
83
85
|
tests/test_recall_core.py
|
|
84
86
|
tests/test_redact.py
|
|
85
87
|
tests/test_retriever.py
|
|
88
|
+
tests/test_roi_trend.py
|
|
86
89
|
tests/test_semantic_feedback.py
|
|
87
90
|
tests/test_service.py
|
|
88
91
|
tests/test_session_context.py
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Tests for the counterfactual eval harness — win/tie/loss vs no-memory baseline."""
|
|
2
|
+
import json
|
|
3
|
+
import time
|
|
4
|
+
|
|
5
|
+
from memor.types import Artifact
|
|
6
|
+
from memor.store.sqlite_store import SqliteStore
|
|
7
|
+
from memor.embed.fake import FakeEmbedder
|
|
8
|
+
from memor.eval.counterfactual import (
|
|
9
|
+
build_cases_from_store, CounterfactualCase, CounterfactualVerdict,
|
|
10
|
+
Outcome, summarize_verdicts, parse_verdict_json,
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
def _seed_store(tmp_path, *, n_sessions=3, turns_per_session=5):
|
|
15
|
+
"""Create a store with sessions that have recall logs and artifacts."""
|
|
16
|
+
db_path = str(tmp_path / "m.db")
|
|
17
|
+
e = FakeEmbedder(dim=16)
|
|
18
|
+
store = SqliteStore(db_path, dim=16)
|
|
19
|
+
|
|
20
|
+
now = time.time()
|
|
21
|
+
for s_idx in range(n_sessions):
|
|
22
|
+
sid = f"session-{s_idx}"
|
|
23
|
+
session_start = now - (n_sessions - s_idx) * 86400
|
|
24
|
+
artifacts = []
|
|
25
|
+
vectors = []
|
|
26
|
+
for t_idx in range(turns_per_session):
|
|
27
|
+
aid = f"a-{s_idx}-{t_idx}"
|
|
28
|
+
text = f"Turn {t_idx} of session {s_idx}: discussion about auth middleware and token refresh"
|
|
29
|
+
vec = e.embed([text])[0]
|
|
30
|
+
artifacts.append(Artifact(
|
|
31
|
+
id=aid, kind="session_chunk", project="testproj",
|
|
32
|
+
source="test", text=text, token_count=20,
|
|
33
|
+
created_at=session_start + t_idx,
|
|
34
|
+
meta={"session_id": sid, "ord": t_idx}))
|
|
35
|
+
vectors.append(vec)
|
|
36
|
+
store.add_artifacts(artifacts, vectors)
|
|
37
|
+
|
|
38
|
+
store.db.commit()
|
|
39
|
+
return store
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_build_cases_from_store(tmp_path):
|
|
43
|
+
store = _seed_store(tmp_path, n_sessions=3, turns_per_session=5)
|
|
44
|
+
cases = build_cases_from_store(store, project="testproj",
|
|
45
|
+
holdout_turns=2, min_session_turns=4)
|
|
46
|
+
assert len(cases) >= 1
|
|
47
|
+
for c in cases:
|
|
48
|
+
assert c.query
|
|
49
|
+
assert c.holdout_texts
|
|
50
|
+
assert c.scope_project == "testproj"
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def test_build_cases_needs_prior_sessions(tmp_path):
|
|
54
|
+
store = _seed_store(tmp_path, n_sessions=1, turns_per_session=5)
|
|
55
|
+
cases = build_cases_from_store(store, project="testproj")
|
|
56
|
+
assert len(cases) == 0
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_parse_verdict_json_win():
|
|
60
|
+
raw = '{"outcome": "win", "reasoning": "Context saved rework", "confidence": 0.9}'
|
|
61
|
+
v = parse_verdict_json(raw)
|
|
62
|
+
assert v.outcome == Outcome.WIN
|
|
63
|
+
assert v.confidence == 0.9
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def test_parse_verdict_json_loss():
|
|
67
|
+
raw = '{"outcome": "loss", "reasoning": "Stale info misled agent", "confidence": 0.8}'
|
|
68
|
+
v = parse_verdict_json(raw)
|
|
69
|
+
assert v.outcome == Outcome.LOSS
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def test_parse_verdict_json_tie():
|
|
73
|
+
raw = '{"outcome": "tie", "reasoning": "Irrelevant context", "confidence": 0.7}'
|
|
74
|
+
v = parse_verdict_json(raw)
|
|
75
|
+
assert v.outcome == Outcome.TIE
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def test_parse_verdict_embedded_json():
|
|
79
|
+
raw = 'Here is my assessment:\n```json\n{"outcome": "win", "reasoning": "Helped", "confidence": 0.85}\n```'
|
|
80
|
+
v = parse_verdict_json(raw)
|
|
81
|
+
assert v.outcome == Outcome.WIN
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def test_summarize_verdicts():
|
|
85
|
+
verdicts = [
|
|
86
|
+
CounterfactualVerdict(outcome=Outcome.WIN, reasoning="a", confidence=0.9),
|
|
87
|
+
CounterfactualVerdict(outcome=Outcome.WIN, reasoning="b", confidence=0.8),
|
|
88
|
+
CounterfactualVerdict(outcome=Outcome.TIE, reasoning="c", confidence=0.7),
|
|
89
|
+
CounterfactualVerdict(outcome=Outcome.LOSS, reasoning="d", confidence=0.6),
|
|
90
|
+
]
|
|
91
|
+
s = summarize_verdicts(verdicts)
|
|
92
|
+
assert s["win_count"] == 2
|
|
93
|
+
assert s["tie_count"] == 1
|
|
94
|
+
assert s["loss_count"] == 1
|
|
95
|
+
assert s["n_cases"] == 4
|
|
96
|
+
assert s["win_pct"] == 50.0
|
|
97
|
+
assert s["loss_pct"] == 25.0
|
|
98
|
+
assert s["do_no_harm_pct"] == 75.0
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def test_summarize_empty():
|
|
102
|
+
s = summarize_verdicts([])
|
|
103
|
+
assert s["n_cases"] == 0
|
|
104
|
+
assert s["do_no_harm_pct"] == 100.0
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""Tests for daily ROI trend — tool call reduction over time."""
|
|
2
|
+
from memor.store.sqlite_store import SqliteStore
|
|
3
|
+
from memor.turn_metrics import TurnMetric
|
|
4
|
+
|
|
5
|
+
|
|
6
|
+
def test_roi_trend_groups_by_day(tmp_path):
|
|
7
|
+
db_path = str(tmp_path / "m.db")
|
|
8
|
+
store = SqliteStore(db_path, dim=16)
|
|
9
|
+
|
|
10
|
+
day1_base = 1749427200.0 # 2025-06-09 00:00:00 UTC
|
|
11
|
+
day2_base = day1_base + 86400
|
|
12
|
+
|
|
13
|
+
store.save_turn_metrics("s1", "p", [
|
|
14
|
+
TurnMetric(turn_idx=0, user_timestamp=day1_base + 100, tool_call_count=1, had_recall=True),
|
|
15
|
+
TurnMetric(turn_idx=1, user_timestamp=day1_base + 200, tool_call_count=4, had_recall=False),
|
|
16
|
+
])
|
|
17
|
+
store.save_turn_metrics("s2", "p", [
|
|
18
|
+
TurnMetric(turn_idx=0, user_timestamp=day2_base + 100, tool_call_count=2, had_recall=True),
|
|
19
|
+
TurnMetric(turn_idx=1, user_timestamp=day2_base + 200, tool_call_count=6, had_recall=False),
|
|
20
|
+
])
|
|
21
|
+
|
|
22
|
+
trend = store.get_roi_trend()
|
|
23
|
+
assert len(trend) == 2
|
|
24
|
+
for day in trend:
|
|
25
|
+
assert "day" in day
|
|
26
|
+
assert "avg_with" in day
|
|
27
|
+
assert "avg_without" in day
|
|
28
|
+
assert "reduction_pct" in day
|
|
29
|
+
assert "turns_total" in day
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def test_roi_trend_calculates_reduction(tmp_path):
|
|
33
|
+
db_path = str(tmp_path / "m.db")
|
|
34
|
+
store = SqliteStore(db_path, dim=16)
|
|
35
|
+
|
|
36
|
+
base = 1749427200.0
|
|
37
|
+
store.save_turn_metrics("s1", "p", [
|
|
38
|
+
TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=2, had_recall=True),
|
|
39
|
+
TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=2, had_recall=True),
|
|
40
|
+
TurnMetric(turn_idx=2, user_timestamp=base + 300, tool_call_count=4, had_recall=False),
|
|
41
|
+
TurnMetric(turn_idx=3, user_timestamp=base + 400, tool_call_count=4, had_recall=False),
|
|
42
|
+
])
|
|
43
|
+
|
|
44
|
+
trend = store.get_roi_trend()
|
|
45
|
+
assert len(trend) == 1
|
|
46
|
+
day = trend[0]
|
|
47
|
+
assert day["avg_with"] == 2.0
|
|
48
|
+
assert day["avg_without"] == 4.0
|
|
49
|
+
assert day["reduction_pct"] == 50.0
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def test_roi_trend_filters_by_project(tmp_path):
|
|
53
|
+
db_path = str(tmp_path / "m.db")
|
|
54
|
+
store = SqliteStore(db_path, dim=16)
|
|
55
|
+
|
|
56
|
+
base = 1749427200.0
|
|
57
|
+
store.save_turn_metrics("s1", "proj-a", [
|
|
58
|
+
TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=1, had_recall=True),
|
|
59
|
+
TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=5, had_recall=False),
|
|
60
|
+
])
|
|
61
|
+
store.save_turn_metrics("s2", "proj-b", [
|
|
62
|
+
TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=3, had_recall=True),
|
|
63
|
+
TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=3, had_recall=False),
|
|
64
|
+
])
|
|
65
|
+
|
|
66
|
+
trend_a = store.get_roi_trend(project="proj-a")
|
|
67
|
+
assert len(trend_a) == 1
|
|
68
|
+
assert trend_a[0]["avg_with"] == 1.0
|
|
69
|
+
|
|
70
|
+
trend_b = store.get_roi_trend(project="proj-b")
|
|
71
|
+
assert len(trend_b) == 1
|
|
72
|
+
assert trend_b[0]["avg_with"] == 3.0
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def test_roi_trend_skips_days_without_both_types(tmp_path):
|
|
76
|
+
"""Days that have only recall or only non-recall turns can't compute reduction."""
|
|
77
|
+
db_path = str(tmp_path / "m.db")
|
|
78
|
+
store = SqliteStore(db_path, dim=16)
|
|
79
|
+
|
|
80
|
+
base = 1749427200.0
|
|
81
|
+
store.save_turn_metrics("s1", "p", [
|
|
82
|
+
TurnMetric(turn_idx=0, user_timestamp=base + 100, tool_call_count=2, had_recall=True),
|
|
83
|
+
TurnMetric(turn_idx=1, user_timestamp=base + 200, tool_call_count=3, had_recall=True),
|
|
84
|
+
])
|
|
85
|
+
|
|
86
|
+
trend = store.get_roi_trend()
|
|
87
|
+
assert len(trend) == 0
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|