memor-cli 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {memor_cli-0.4.0/memor_cli.egg-info → memor_cli-0.5.0}/PKG-INFO +3 -2
- {memor_cli-0.4.0 → memor_cli-0.5.0}/README.md +2 -1
- memor_cli-0.5.0/memor/__init__.py +1 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/cli.py +36 -3
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/server.py +13 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/static/index.html +144 -60
- memor_cli-0.5.0/memor/eval/counterfactual.py +176 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/store/sqlite_store.py +50 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0/memor_cli.egg-info}/PKG-INFO +3 -2
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/SOURCES.txt +3 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/pyproject.toml +1 -1
- memor_cli-0.5.0/tests/test_counterfactual.py +104 -0
- memor_cli-0.5.0/tests/test_roi_trend.py +87 -0
- memor_cli-0.4.0/memor/__init__.py +0 -1
- {memor_cli-0.4.0 → memor_cli-0.5.0}/LICENSE +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/daemon.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/dashboard/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/distiller.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/distill/extractive.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/api.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/fake.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/embed/local.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/base.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/claude_mem.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/baselines/graphiti.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/dataset.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/embed_benchmark.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/judge.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/metrics.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/eval/runner.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/feedback.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/hook_cli.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/hook_server.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/claude_code.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/ingest/documents.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/interfaces.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/anthropic.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/base.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/llm/openai_compat.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/project.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/query_complexity.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/recall.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/redact.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/retrieve/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/retrieve/retriever.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/service.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/session_context.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/store/__init__.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/tokencount.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/turn_metrics.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor/types.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/dependency_links.txt +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/entry_points.txt +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/requires.txt +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/memor_cli.egg-info/top_level.txt +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/setup.cfg +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_cli_smoke.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_daemon.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dashboard.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dataset_builder.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_dimension_safety.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_distiller.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_embed.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_embed_benchmark.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_eval_ablation.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_eval_runner.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_external_baselines.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_extractive.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_feedback.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hook.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hook_server.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_hybrid_retrieval.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_ingest_claude_code.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_ingest_documents.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_install_hook.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_interfaces.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_judge.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_metrics.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_noise_filter.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_project_resolver.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_quality_gate.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_query_complexity.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_recall_core.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_redact.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_retriever.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_semantic_feedback.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_service.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_session_context.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_skill_recall.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_store.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_supersession.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_tokencount.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_turn_metrics.py +0 -0
- {memor_cli-0.4.0 → memor_cli-0.5.0}/tests/test_types.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: memor-cli
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: Measured memory for coding agents. Fire and forget — no API keys needed.
|
|
5
5
|
Author-email: Nimit Bhandari <nimitbhandari17@gmail.com>
|
|
6
6
|
License-Expression: MIT
|
|
@@ -46,7 +46,7 @@ Dynamic: license-file
|
|
|
46
46
|
```
|
|
47
47
|
|
|
48
48
|
[](LICENSE)
|
|
49
|
-
[]()
|
|
50
50
|
[]()
|
|
51
51
|
[](https://pypi.org/project/memor-cli/)
|
|
52
52
|
|
|
@@ -200,6 +200,7 @@ memor ingest-project <dir> Bulk ingest a project directory
|
|
|
200
200
|
memor ingest-doc <file> Ingest a markdown document
|
|
201
201
|
memor distill --project <name> Run distillation manually
|
|
202
202
|
memor eval <cases.json> Run eval suite
|
|
203
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
203
204
|
memor bench-embed --project <name> Compare embedding models
|
|
204
205
|
```
|
|
205
206
|
|
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
```
|
|
10
10
|
|
|
11
11
|
[](LICENSE)
|
|
12
|
-
[]()
|
|
13
13
|
[]()
|
|
14
14
|
[](https://pypi.org/project/memor-cli/)
|
|
15
15
|
|
|
@@ -163,6 +163,7 @@ memor ingest-project <dir> Bulk ingest a project directory
|
|
|
163
163
|
memor ingest-doc <file> Ingest a markdown document
|
|
164
164
|
memor distill --project <name> Run distillation manually
|
|
165
165
|
memor eval <cases.json> Run eval suite
|
|
166
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
166
167
|
memor bench-embed --project <name> Compare embedding models
|
|
167
168
|
```
|
|
168
169
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.4.1"
|
|
@@ -78,9 +78,10 @@ MAINTENANCE
|
|
|
78
78
|
memor version Print the installed version
|
|
79
79
|
|
|
80
80
|
EVALUATION
|
|
81
|
-
memor eval <cases.json>
|
|
82
|
-
memor eval-judge --project <name>
|
|
83
|
-
memor
|
|
81
|
+
memor eval <cases.json> Run eval suite
|
|
82
|
+
memor eval-judge --project <name> LLM-as-judge evaluation
|
|
83
|
+
memor eval-counterfactual --project Win/tie/loss vs no-memory baseline
|
|
84
|
+
memor bench-embed --project <name> Compare embedding models
|
|
84
85
|
|
|
85
86
|
CONFIGURATION
|
|
86
87
|
Everything works locally with zero API keys.
|
|
@@ -191,6 +192,38 @@ def eval_judge_cmd(project: str = typer.Option(...), db: str = "memor.db",
|
|
|
191
192
|
typer.echo(f"Judge eval complete: mean relevance = {summary['mean_relevance']:.3f}")
|
|
192
193
|
|
|
193
194
|
|
|
195
|
+
@app.command("eval-counterfactual")
|
|
196
|
+
def eval_counterfactual_cmd(project: str = typer.Option(...), db: str = "memor.db",
|
|
197
|
+
k: int = 8, fake: bool = False,
|
|
198
|
+
llm_provider: str = "anthropic",
|
|
199
|
+
llm_model: str = "claude-sonnet-4-6",
|
|
200
|
+
holdout: int = 2):
|
|
201
|
+
"""Counterfactual eval: win/tie/loss vs no-memory baseline. Requires an LLM API key."""
|
|
202
|
+
from memor.eval.counterfactual import build_cases_from_store, run_suite
|
|
203
|
+
e = _embedder(fake); s = SqliteStore(_db_path(db), dim=e.dim)
|
|
204
|
+
cases = build_cases_from_store(s, project=project, holdout_turns=holdout)
|
|
205
|
+
if not cases:
|
|
206
|
+
typer.echo("No cases — need at least 2 sessions with >= 4 turns each.")
|
|
207
|
+
raise typer.Exit(1)
|
|
208
|
+
typer.echo(f"Built {len(cases)} counterfactual cases. Running evaluation...")
|
|
209
|
+
if llm_provider == "anthropic":
|
|
210
|
+
from memor.llm.anthropic import AnthropicLLM
|
|
211
|
+
llm = AnthropicLLM(model=llm_model)
|
|
212
|
+
else:
|
|
213
|
+
from memor.llm.openai_compat import OpenAICompatLLM
|
|
214
|
+
import os
|
|
215
|
+
llm = OpenAICompatLLM(base_url=os.environ.get("OPENAI_BASE_URL", "http://localhost:11434/v1"),
|
|
216
|
+
api_key=os.environ.get("OPENAI_API_KEY", ""), model=llm_model)
|
|
217
|
+
summary = run_suite(cases, store=s, embedder=e, llm=llm, k=k)
|
|
218
|
+
typer.echo(json.dumps({k: v for k, v in summary.items() if k != "cases"}, indent=2))
|
|
219
|
+
typer.echo("")
|
|
220
|
+
typer.echo(f" Win: {summary['win_count']}/{summary['n_cases']} ({summary['win_pct']}%)")
|
|
221
|
+
typer.echo(f" Tie: {summary['tie_count']}/{summary['n_cases']} ({summary['tie_pct']}%)")
|
|
222
|
+
typer.echo(f" Loss: {summary['loss_count']}/{summary['n_cases']} ({summary['loss_pct']}%)")
|
|
223
|
+
typer.echo(f" Do-no-harm: {summary['do_no_harm_pct']}%")
|
|
224
|
+
s.save_eval_run({"type": "counterfactual", "k": k, "project": project, "holdout": holdout}, summary)
|
|
225
|
+
|
|
226
|
+
|
|
194
227
|
@app.command("bench-embed")
|
|
195
228
|
def bench_embed(project: str = typer.Option(...), db: str = "memor.db",
|
|
196
229
|
k: int = 8, fake: bool = False):
|
|
@@ -139,6 +139,19 @@ def create_app(db_path: str | None = None) -> FastAPI:
|
|
|
139
139
|
store = _store()
|
|
140
140
|
return store.get_token_roi(project=project)
|
|
141
141
|
|
|
142
|
+
@app.get("/api/roi-trend")
|
|
143
|
+
def roi_trend(project: str | None = Query(None)):
|
|
144
|
+
store = _store()
|
|
145
|
+
return store.get_roi_trend(project=project)
|
|
146
|
+
|
|
147
|
+
@app.get("/api/eval/latest")
|
|
148
|
+
def eval_latest(eval_type: str = Query("counterfactual")):
|
|
149
|
+
store = _store()
|
|
150
|
+
result = store.get_latest_eval(eval_type)
|
|
151
|
+
if not result:
|
|
152
|
+
return {"status": "no_runs"}
|
|
153
|
+
return result
|
|
154
|
+
|
|
142
155
|
@app.get("/api/health")
|
|
143
156
|
def health():
|
|
144
157
|
store = _store()
|
|
@@ -172,6 +172,7 @@
|
|
|
172
172
|
background: var(--surface); border: 1px solid var(--border);
|
|
173
173
|
border-radius: var(--radius); padding: 22px 24px;
|
|
174
174
|
box-shadow: var(--card-shadow);
|
|
175
|
+
display: flex; flex-direction: column;
|
|
175
176
|
}
|
|
176
177
|
.chart-header {
|
|
177
178
|
display: flex; align-items: center; justify-content: space-between;
|
|
@@ -199,7 +200,7 @@
|
|
|
199
200
|
|
|
200
201
|
.bar-chart {
|
|
201
202
|
display: flex; align-items: flex-end; gap: 4px;
|
|
202
|
-
height: 180px; padding-top: 8px;
|
|
203
|
+
flex: 1; min-height: 180px; padding-top: 8px;
|
|
203
204
|
border-bottom: 1px solid var(--border-light);
|
|
204
205
|
}
|
|
205
206
|
.bar-group {
|
|
@@ -207,15 +208,16 @@
|
|
|
207
208
|
align-items: center; gap: 0; height: 100%;
|
|
208
209
|
justify-content: flex-end;
|
|
209
210
|
}
|
|
210
|
-
.bar-stack { display: flex; flex-direction: column-reverse; gap:
|
|
211
|
-
.bar-
|
|
212
|
-
.bar-
|
|
211
|
+
.bar-stack { display: flex; flex-direction: column-reverse; gap: 2px; width: 100%; max-width: 28px; flex: 1; align-items: center; justify-content: flex-end; }
|
|
212
|
+
.bar-cell { width: 16px; height: 10px; border-radius: 2px; transition: opacity 0.3s; }
|
|
213
|
+
.bar-cell-hit { background: var(--accent); }
|
|
214
|
+
.bar-cell-miss { background: var(--surface3); }
|
|
213
215
|
.bar-label {
|
|
214
216
|
font-size: 9px; color: var(--text-muted); margin-top: 6px;
|
|
215
217
|
text-align: center; white-space: nowrap;
|
|
216
218
|
}
|
|
217
219
|
|
|
218
|
-
.bar-group:hover .bar-
|
|
220
|
+
.bar-group:hover .bar-cell-hit { background: #f0a030; }
|
|
219
221
|
.bar-group { cursor: default; position: relative; }
|
|
220
222
|
.bar-tooltip {
|
|
221
223
|
display: none; position: absolute; bottom: calc(100% + 8px);
|
|
@@ -378,6 +380,37 @@
|
|
|
378
380
|
</div>
|
|
379
381
|
</section>
|
|
380
382
|
|
|
383
|
+
<!-- ── Token ROI hero ─────────────────────────────────── -->
|
|
384
|
+
<section>
|
|
385
|
+
<div class="chart-card" id="roi-card" style="display:none;">
|
|
386
|
+
<div style="display:flex;align-items:center;gap:20px;flex-wrap:wrap;">
|
|
387
|
+
<div>
|
|
388
|
+
<div class="chart-title" style="margin-bottom:4px;">Token ROI</div>
|
|
389
|
+
<div style="font-size:32px;font-weight:700;color:var(--ok);letter-spacing:-1px;" id="roi-value">–</div>
|
|
390
|
+
<div style="font-size:12px;color:var(--text-muted);margin-top:2px;" id="roi-desc">fewer tool calls when Memor injects context</div>
|
|
391
|
+
</div>
|
|
392
|
+
<div style="display:flex;gap:24px;flex:1;justify-content:flex-end;">
|
|
393
|
+
<div style="text-align:center;">
|
|
394
|
+
<div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-tools-with">–</div>
|
|
395
|
+
<div style="font-size:10px;color:var(--text-muted);">avg tools/turn<br>with recall</div>
|
|
396
|
+
</div>
|
|
397
|
+
<div style="text-align:center;">
|
|
398
|
+
<div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-tools-without">–</div>
|
|
399
|
+
<div style="font-size:10px;color:var(--text-muted);">avg tools/turn<br>without recall</div>
|
|
400
|
+
</div>
|
|
401
|
+
<div style="text-align:center;">
|
|
402
|
+
<div style="font-size:18px;font-weight:600;color:var(--text);" id="roi-turns">–</div>
|
|
403
|
+
<div style="font-size:10px;color:var(--text-muted);" id="roi-turns-sub">turns measured</div>
|
|
404
|
+
</div>
|
|
405
|
+
</div>
|
|
406
|
+
</div>
|
|
407
|
+
<div style="margin-top:16px;border-top:1px solid var(--border-light);padding-top:12px;position:relative;">
|
|
408
|
+
<div id="roi-sparkline" style="width:100%;height:80px;"></div>
|
|
409
|
+
<div id="roi-spark-tooltip" style="display:none;position:absolute;top:0;background:var(--surface3);border:1px solid var(--border);border-radius:6px;padding:6px 10px;font-size:11px;color:var(--text);white-space:nowrap;pointer-events:none;box-shadow:0 4px 12px rgba(0,0,0,0.4);z-index:10;"></div>
|
|
410
|
+
</div>
|
|
411
|
+
</div>
|
|
412
|
+
</section>
|
|
413
|
+
|
|
381
414
|
<!-- ── Recall trend chart ─────────────────────────────── -->
|
|
382
415
|
<section>
|
|
383
416
|
<div class="chart-grid">
|
|
@@ -426,32 +459,6 @@
|
|
|
426
459
|
</div>
|
|
427
460
|
</div>
|
|
428
461
|
</div>
|
|
429
|
-
<div class="chart-card">
|
|
430
|
-
<div class="chart-header">
|
|
431
|
-
<div class="chart-title">Token ROI</div>
|
|
432
|
-
</div>
|
|
433
|
-
<div id="roi-banner" style="display:none;background:var(--ok-dim);border:1px solid rgba(61,214,140,0.2);border-radius:var(--radius-sm);padding:12px 14px;margin-bottom:14px;">
|
|
434
|
-
<div style="font-size:22px;font-weight:700;color:var(--ok);letter-spacing:-0.5px;" id="roi-value">–</div>
|
|
435
|
-
<div style="font-size:11px;color:var(--text-muted);margin-top:2px;" id="roi-desc">fewer tool calls when Memor injects context</div>
|
|
436
|
-
</div>
|
|
437
|
-
<div class="side-stats" id="roi-side">
|
|
438
|
-
<div class="side-stat">
|
|
439
|
-
<div class="side-stat-label">Avg Tools / Turn (with recall)</div>
|
|
440
|
-
<div class="side-stat-value" id="roi-tools-with">–</div>
|
|
441
|
-
<div class="side-stat-sub">when Memor injected context</div>
|
|
442
|
-
</div>
|
|
443
|
-
<div class="side-stat">
|
|
444
|
-
<div class="side-stat-label">Avg Tools / Turn (without)</div>
|
|
445
|
-
<div class="side-stat-value" id="roi-tools-without">–</div>
|
|
446
|
-
<div class="side-stat-sub">when no context was injected</div>
|
|
447
|
-
</div>
|
|
448
|
-
<div class="side-stat">
|
|
449
|
-
<div class="side-stat-label">Turns Measured</div>
|
|
450
|
-
<div class="side-stat-value" id="roi-turns">–</div>
|
|
451
|
-
<div class="side-stat-sub" id="roi-turns-sub">with vs. without recall</div>
|
|
452
|
-
</div>
|
|
453
|
-
</div>
|
|
454
|
-
</div>
|
|
455
462
|
</div>
|
|
456
463
|
</section>
|
|
457
464
|
|
|
@@ -539,7 +546,7 @@
|
|
|
539
546
|
function badge(status) {
|
|
540
547
|
var map = {
|
|
541
548
|
ok: ['badge-ok', 'Success'],
|
|
542
|
-
extractive_only: ['badge-extractive','
|
|
549
|
+
extractive_only: ['badge-extractive','No Distill'],
|
|
543
550
|
no_hits: ['badge-no_hits', 'No Hits'],
|
|
544
551
|
};
|
|
545
552
|
var m = map[status] || ['badge-no_hits', status];
|
|
@@ -669,21 +676,20 @@
|
|
|
669
676
|
var recallsByDay = data.map(function(d) { return d.recalls; });
|
|
670
677
|
renderMiniBars('mb-recalls', recallsByDay);
|
|
671
678
|
|
|
679
|
+
var maxCells = 30;
|
|
672
680
|
data.forEach(function(d) {
|
|
673
|
-
var
|
|
674
|
-
var
|
|
675
|
-
var
|
|
676
|
-
var hitH = Math.max(0, pct * hitPct / 100);
|
|
677
|
-
var missH = Math.max(0, pct * missPct / 100);
|
|
681
|
+
var totalCells = Math.max(1, Math.round((d.recalls / maxRecalls) * maxCells));
|
|
682
|
+
var hitCells = Math.round((d.hits / d.recalls) * totalCells);
|
|
683
|
+
var missCells = totalCells - hitCells;
|
|
678
684
|
|
|
679
685
|
var dayStr = d.day.slice(5);
|
|
680
686
|
var group = el('div', {class: 'bar-group'});
|
|
687
|
+
var cells = '';
|
|
688
|
+
for (var i = 0; i < hitCells; i++) cells += '<div class="bar-cell bar-cell-hit"></div>';
|
|
689
|
+
for (var i = 0; i < missCells; i++) cells += '<div class="bar-cell bar-cell-miss"></div>';
|
|
681
690
|
group.innerHTML =
|
|
682
691
|
'<div class="bar-tooltip">' + esc(d.day) + '<br>' + d.recalls + ' recalls · ' + (d.hits||0) + ' hits</div>' +
|
|
683
|
-
'<div class="bar-stack">' +
|
|
684
|
-
'<div class="bar-hits" style="height:' + hitH + '%"></div>' +
|
|
685
|
-
'<div class="bar-miss" style="height:' + missH + '%"></div>' +
|
|
686
|
-
'</div>' +
|
|
692
|
+
'<div class="bar-stack">' + cells + '</div>' +
|
|
687
693
|
'<div class="bar-label">' + dayStr + '</div>';
|
|
688
694
|
container.appendChild(group);
|
|
689
695
|
});
|
|
@@ -725,7 +731,7 @@
|
|
|
725
731
|
document.getElementById('th-col3').textContent = 'Avg Score';
|
|
726
732
|
document.getElementById('th-col4').textContent = 'OK';
|
|
727
733
|
document.getElementById('th-col5').textContent = 'No Hits';
|
|
728
|
-
document.getElementById('th-col6').textContent = '
|
|
734
|
+
document.getElementById('th-col6').textContent = 'No Distill';
|
|
729
735
|
}
|
|
730
736
|
|
|
731
737
|
sorted.forEach(function(row) {
|
|
@@ -809,36 +815,113 @@
|
|
|
809
815
|
|
|
810
816
|
/* ── ROI renderer ─────────────────────────────────────── */
|
|
811
817
|
function renderROI(data) {
|
|
818
|
+
var card = document.getElementById('roi-card');
|
|
819
|
+
var roiValue = document.getElementById('roi-value');
|
|
820
|
+
var roiDesc = document.getElementById('roi-desc');
|
|
821
|
+
|
|
812
822
|
document.getElementById('roi-tools-with').textContent = data.avg_tools_with_recall;
|
|
813
823
|
document.getElementById('roi-tools-without').textContent = data.avg_tools_without_recall;
|
|
814
|
-
|
|
815
|
-
|
|
824
|
+
var totalTurns = data.turns_with_recall + data.turns_without_recall;
|
|
825
|
+
document.getElementById('roi-turns').textContent = totalTurns.toLocaleString();
|
|
816
826
|
document.getElementById('roi-turns-sub').textContent =
|
|
817
|
-
data.turns_with_recall + ' with
|
|
818
|
-
|
|
819
|
-
var banner = document.getElementById('roi-banner');
|
|
820
|
-
var roiValue = document.getElementById('roi-value');
|
|
821
|
-
var roiDesc = document.getElementById('roi-desc');
|
|
822
|
-
banner.style.display = 'none';
|
|
823
|
-
banner.style.background = 'var(--ok-dim)';
|
|
824
|
-
banner.style.borderColor = 'rgba(61,214,140,0.2)';
|
|
825
|
-
roiValue.style.color = 'var(--ok)';
|
|
826
|
-
roiValue.textContent = '–';
|
|
827
|
-
roiDesc.textContent = 'fewer tool calls when Memor injects context';
|
|
827
|
+
data.turns_with_recall + ' with · ' + data.turns_without_recall + ' without';
|
|
828
828
|
|
|
829
829
|
if (data.tool_call_reduction_pct > 0 && data.turns_with_recall >= 5 && data.turns_without_recall >= 5) {
|
|
830
830
|
roiValue.textContent = data.tool_call_reduction_pct + '% fewer';
|
|
831
|
-
|
|
831
|
+
roiValue.style.color = 'var(--ok)';
|
|
832
|
+
roiDesc.textContent = 'fewer tool calls when Memor injects context';
|
|
833
|
+
card.style.display = 'block';
|
|
832
834
|
} else if (data.tool_call_reduction_pct < 0 && data.turns_with_recall >= 5) {
|
|
833
835
|
roiValue.textContent = Math.abs(data.tool_call_reduction_pct) + '% more';
|
|
834
836
|
roiValue.style.color = 'var(--warn)';
|
|
835
837
|
roiDesc.textContent = 'tool calls with recall — investigating...';
|
|
836
|
-
|
|
837
|
-
banner.style.background = 'var(--warn-dim)';
|
|
838
|
-
banner.style.borderColor = 'rgba(232,147,32,0.2)';
|
|
838
|
+
card.style.display = 'block';
|
|
839
839
|
}
|
|
840
840
|
}
|
|
841
841
|
|
|
842
|
+
function renderROITrend(data) {
|
|
843
|
+
var container = document.getElementById('roi-sparkline');
|
|
844
|
+
var tooltip = document.getElementById('roi-spark-tooltip');
|
|
845
|
+
if (!container || !data || !data.length) return;
|
|
846
|
+
|
|
847
|
+
var W = container.clientWidth || 400;
|
|
848
|
+
var H = 80;
|
|
849
|
+
var padL = 32, padR = 8, padT = 8, padB = 20;
|
|
850
|
+
var plotW = W - padL - padR;
|
|
851
|
+
var plotH = H - padT - padB;
|
|
852
|
+
|
|
853
|
+
var vals = data.map(function(d) { return d.reduction_pct; });
|
|
854
|
+
var minV = Math.min.apply(null, vals);
|
|
855
|
+
var maxV = Math.max.apply(null, vals);
|
|
856
|
+
var range = (maxV - minV) || 1;
|
|
857
|
+
var yFloor = Math.min(0, minV);
|
|
858
|
+
var yCeil = Math.max(maxV, 10);
|
|
859
|
+
range = (yCeil - yFloor) || 1;
|
|
860
|
+
|
|
861
|
+
function x(i) { return padL + (data.length > 1 ? (i / (data.length - 1)) * plotW : plotW / 2); }
|
|
862
|
+
function y(v) { return padT + plotH - ((v - yFloor) / range) * plotH; }
|
|
863
|
+
|
|
864
|
+
var lineColor = vals[vals.length - 1] >= 0 ? '#3dd68c' : '#e89320';
|
|
865
|
+
var gradId = 'roiGrad';
|
|
866
|
+
|
|
867
|
+
var pts = data.map(function(d, i) { return x(i) + ',' + y(d.reduction_pct); });
|
|
868
|
+
var linePath = 'M' + pts.join(' L');
|
|
869
|
+
var areaPath = linePath + ' L' + x(data.length - 1) + ',' + y(yFloor) + ' L' + x(0) + ',' + y(yFloor) + ' Z';
|
|
870
|
+
|
|
871
|
+
var gridLines = '';
|
|
872
|
+
var steps = [yFloor, Math.round((yFloor + yCeil) / 2), yCeil];
|
|
873
|
+
steps.forEach(function(v) {
|
|
874
|
+
var yy = y(v);
|
|
875
|
+
gridLines += '<line x1="' + padL + '" y1="' + yy + '" x2="' + (W - padR) + '" y2="' + yy + '" stroke="var(--border-light)" stroke-width="0.5" stroke-dasharray="3,3"/>';
|
|
876
|
+
gridLines += '<text x="' + (padL - 4) + '" y="' + (yy + 3) + '" fill="var(--text-muted)" font-size="8" text-anchor="end">' + v + '%</text>';
|
|
877
|
+
});
|
|
878
|
+
|
|
879
|
+
var xLabels = '';
|
|
880
|
+
data.forEach(function(d, i) {
|
|
881
|
+
xLabels += '<text x="' + x(i) + '" y="' + (H - 2) + '" fill="var(--text-muted)" font-size="8" text-anchor="middle">' + d.day.slice(5) + '</text>';
|
|
882
|
+
});
|
|
883
|
+
|
|
884
|
+
var dots = data.map(function(d, i) {
|
|
885
|
+
return '<circle cx="' + x(i) + '" cy="' + y(d.reduction_pct) + '" r="4" fill="' + lineColor + '" stroke="var(--surface)" stroke-width="2" style="opacity:0;" class="roi-dot" data-idx="' + i + '"/>';
|
|
886
|
+
}).join('');
|
|
887
|
+
|
|
888
|
+
var hitAreas = data.map(function(d, i) {
|
|
889
|
+
return '<rect x="' + (x(i) - plotW / data.length / 2) + '" y="' + padT + '" width="' + (plotW / data.length) + '" height="' + plotH + '" fill="transparent" class="roi-hit" data-idx="' + i + '"/>';
|
|
890
|
+
}).join('');
|
|
891
|
+
|
|
892
|
+
container.innerHTML = '<svg width="' + W + '" height="' + H + '" style="display:block;">' +
|
|
893
|
+
'<defs><linearGradient id="' + gradId + '" x1="0" y1="0" x2="0" y2="1">' +
|
|
894
|
+
'<stop offset="0%" stop-color="' + lineColor + '" stop-opacity="0.3"/>' +
|
|
895
|
+
'<stop offset="100%" stop-color="' + lineColor + '" stop-opacity="0.0"/>' +
|
|
896
|
+
'</linearGradient></defs>' +
|
|
897
|
+
gridLines + xLabels +
|
|
898
|
+
'<path d="' + areaPath + '" fill="url(#' + gradId + ')"/>' +
|
|
899
|
+
'<path d="' + linePath + '" fill="none" stroke="' + lineColor + '" stroke-width="2" stroke-linecap="round" stroke-linejoin="round"/>' +
|
|
900
|
+
dots + hitAreas +
|
|
901
|
+
'</svg>';
|
|
902
|
+
|
|
903
|
+
container.querySelectorAll('.roi-hit').forEach(function(rect) {
|
|
904
|
+
rect.addEventListener('mouseenter', function() {
|
|
905
|
+
var i = parseInt(rect.getAttribute('data-idx'));
|
|
906
|
+
var d = data[i];
|
|
907
|
+
var dot = container.querySelectorAll('.roi-dot')[i];
|
|
908
|
+
if (dot) dot.style.opacity = '1';
|
|
909
|
+
tooltip.innerHTML = '<strong>' + d.day + '</strong><br>' +
|
|
910
|
+
d.reduction_pct + '% fewer tools<br>' +
|
|
911
|
+
d.turns_with + ' recall / ' + d.turns_without + ' no-recall turns';
|
|
912
|
+
tooltip.style.display = 'block';
|
|
913
|
+
var tx = Math.min(x(i), W - 160);
|
|
914
|
+
tooltip.style.left = tx + 'px';
|
|
915
|
+
});
|
|
916
|
+
rect.addEventListener('mouseleave', function() {
|
|
917
|
+
var i = parseInt(rect.getAttribute('data-idx'));
|
|
918
|
+
var dot = container.querySelectorAll('.roi-dot')[i];
|
|
919
|
+
if (dot) dot.style.opacity = '0';
|
|
920
|
+
tooltip.style.display = 'none';
|
|
921
|
+
});
|
|
922
|
+
});
|
|
923
|
+
}
|
|
924
|
+
|
|
842
925
|
/* ── Data loaders ──────────────────────────────────────── */
|
|
843
926
|
async function loadSummary() { try { renderSummary(await api('/api/summary')); } catch(e) { console.warn('summary',e); } }
|
|
844
927
|
async function loadProjects() { try { renderProjects(await api('/api/projects')); } catch(e) { console.warn('projects',e); } }
|
|
@@ -847,6 +930,7 @@
|
|
|
847
930
|
async function loadHealth() { try { renderHealth(await api('/api/health')); } catch(e) { console.warn('health',e); } }
|
|
848
931
|
async function loadTrend() { try { renderTrend(await api('/api/recall-trend?days=30')); } catch(e) { console.warn('trend',e); } }
|
|
849
932
|
async function loadROI() { try { renderROI(await api('/api/roi')); } catch(e) { console.warn('roi',e); } }
|
|
933
|
+
async function loadROITrend() { try { renderROITrend(await api('/api/roi-trend')); } catch(e) { console.warn('roi-trend',e); } }
|
|
850
934
|
async function loadRecalls() {
|
|
851
935
|
try {
|
|
852
936
|
var url = '/api/recalls?limit=50' + (projectFilter ? '&project=' + encodeURIComponent(projectFilter) : '');
|
|
@@ -855,7 +939,7 @@
|
|
|
855
939
|
}
|
|
856
940
|
|
|
857
941
|
async function refresh() {
|
|
858
|
-
await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI()]);
|
|
942
|
+
await Promise.allSettled([loadSummary(), loadProjects(), loadEfficiency(), loadSessionEfficiency(), loadHealth(), loadTrend(), loadRecalls(), loadROI(), loadROITrend()]);
|
|
859
943
|
document.getElementById('last-updated').textContent = new Date().toLocaleTimeString();
|
|
860
944
|
|
|
861
945
|
renderMiniBars('mb-chunks', null);
|
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Counterfactual eval harness — win/tie/loss vs no-memory baseline.
|
|
2
|
+
|
|
3
|
+
For each held-out session, uses the opening turn as query, retrieves context
|
|
4
|
+
from prior sessions, and asks an LLM judge whether the recalled context would
|
|
5
|
+
have helped (win), been irrelevant (tie), or misled the agent (loss).
|
|
6
|
+
|
|
7
|
+
The key metric is do-no-harm rate = 1 - loss_rate."""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import enum
|
|
11
|
+
import json
|
|
12
|
+
import re
|
|
13
|
+
import time
|
|
14
|
+
from dataclasses import dataclass
|
|
15
|
+
|
|
16
|
+
from memor.types import Artifact, Scope
|
|
17
|
+
from memor.retrieve.retriever import Retriever
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class Outcome(enum.Enum):
|
|
21
|
+
WIN = "win"
|
|
22
|
+
TIE = "tie"
|
|
23
|
+
LOSS = "loss"
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass
|
|
27
|
+
class CounterfactualCase:
|
|
28
|
+
query: str
|
|
29
|
+
holdout_texts: list[str]
|
|
30
|
+
scope_project: str
|
|
31
|
+
session_id: str
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class CounterfactualVerdict:
|
|
36
|
+
outcome: Outcome
|
|
37
|
+
reasoning: str
|
|
38
|
+
confidence: float
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
COUNTERFACTUAL_PROMPT = """You are evaluating whether recalled memory context helped or hurt a coding agent.
|
|
42
|
+
|
|
43
|
+
The agent received this task:
|
|
44
|
+
---
|
|
45
|
+
{query}
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
Here is what actually happened next in the session (ground truth — the agent didn't see this):
|
|
49
|
+
---
|
|
50
|
+
{holdout}
|
|
51
|
+
---
|
|
52
|
+
|
|
53
|
+
Here is the context that was recalled from prior sessions and injected into the agent's prompt:
|
|
54
|
+
---
|
|
55
|
+
{recalled_context}
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
Compare the recalled context against what actually happened. Score as:
|
|
59
|
+
|
|
60
|
+
- **win**: The recalled context directly anticipates what happened, would save the agent exploration/time, or would prevent repeating a past mistake. The agent would have been measurably better with this context.
|
|
61
|
+
- **tie**: The recalled context is irrelevant to what happened. Neither helpful nor harmful — the agent would have arrived at the same outcome either way.
|
|
62
|
+
- **loss**: The recalled context is stale, contradictory, or misleading. It would have sent the agent down the wrong path or caused confusion.
|
|
63
|
+
|
|
64
|
+
When uncertain, default to **tie** — only score win when the context clearly helps, and loss when it clearly hurts.
|
|
65
|
+
|
|
66
|
+
Return STRICT JSON: {{"outcome": "win"|"tie"|"loss", "reasoning": "<one sentence>", "confidence": <float 0-1>}}"""
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def build_cases_from_store(store, *, project: str,
|
|
70
|
+
holdout_turns: int = 2,
|
|
71
|
+
min_session_turns: int = 4,
|
|
72
|
+
min_prior_sessions: int = 1) -> list[CounterfactualCase]:
|
|
73
|
+
rows = store.db.execute(
|
|
74
|
+
"SELECT * FROM artifacts WHERE project=? AND kind='session_chunk' AND active=1",
|
|
75
|
+
(project,)).fetchall()
|
|
76
|
+
artifacts = [store._row_to_artifact(r) for r in rows]
|
|
77
|
+
|
|
78
|
+
by_session: dict[str, list[Artifact]] = {}
|
|
79
|
+
for a in artifacts:
|
|
80
|
+
by_session.setdefault(a.meta.get("session_id", "?"), []).append(a)
|
|
81
|
+
for v in by_session.values():
|
|
82
|
+
v.sort(key=lambda a: a.meta.get("ord", 0))
|
|
83
|
+
sessions = sorted(by_session.items(), key=lambda kv: kv[1][0].created_at)
|
|
84
|
+
|
|
85
|
+
cases: list[CounterfactualCase] = []
|
|
86
|
+
for idx, (sid, chunks) in enumerate(sessions):
|
|
87
|
+
if idx < min_prior_sessions:
|
|
88
|
+
continue
|
|
89
|
+
if len(chunks) < min_session_turns:
|
|
90
|
+
continue
|
|
91
|
+
query = chunks[0].text
|
|
92
|
+
holdout = [c.text for c in chunks[-holdout_turns:]]
|
|
93
|
+
cases.append(CounterfactualCase(
|
|
94
|
+
query=query, holdout_texts=holdout,
|
|
95
|
+
scope_project=project, session_id=sid))
|
|
96
|
+
return cases
|
|
97
|
+
|
|
98
|
+
|
|
99
|
+
def _extract_json(raw: str) -> dict:
|
|
100
|
+
m = re.search(r"\{.*\}", raw, re.DOTALL)
|
|
101
|
+
if m:
|
|
102
|
+
return json.loads(m.group())
|
|
103
|
+
return json.loads(raw)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def parse_verdict_json(raw: str) -> CounterfactualVerdict:
|
|
107
|
+
data = _extract_json(raw)
|
|
108
|
+
outcome_str = data.get("outcome", "tie").lower().strip()
|
|
109
|
+
outcome = Outcome(outcome_str) if outcome_str in ("win", "tie", "loss") else Outcome.TIE
|
|
110
|
+
return CounterfactualVerdict(
|
|
111
|
+
outcome=outcome,
|
|
112
|
+
reasoning=data.get("reasoning", ""),
|
|
113
|
+
confidence=float(data.get("confidence", 0.5)),
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def run_case(case: CounterfactualCase, *, store, embedder, llm,
|
|
118
|
+
k: int = 8) -> CounterfactualVerdict:
|
|
119
|
+
scope = Scope(project=case.scope_project)
|
|
120
|
+
r = Retriever(store, embedder, k=k)
|
|
121
|
+
trace = r.query(case.query, scope)
|
|
122
|
+
|
|
123
|
+
if not trace.hits:
|
|
124
|
+
return CounterfactualVerdict(
|
|
125
|
+
outcome=Outcome.TIE,
|
|
126
|
+
reasoning="No context recalled — nothing to evaluate",
|
|
127
|
+
confidence=1.0)
|
|
128
|
+
|
|
129
|
+
recalled = "\n\n".join(
|
|
130
|
+
f"[{h.artifact.kind}] {h.artifact.text}" for h in trace.hits)
|
|
131
|
+
holdout = "\n".join(case.holdout_texts)
|
|
132
|
+
|
|
133
|
+
prompt = COUNTERFACTUAL_PROMPT.format(
|
|
134
|
+
query=case.query, holdout=holdout, recalled_context=recalled)
|
|
135
|
+
raw = llm.complete(prompt)
|
|
136
|
+
return parse_verdict_json(raw)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def run_suite(cases: list[CounterfactualCase], *, store, embedder, llm,
|
|
140
|
+
k: int = 8) -> dict:
|
|
141
|
+
verdicts = []
|
|
142
|
+
for c in cases:
|
|
143
|
+
v = run_case(c, store=store, embedder=embedder, llm=llm, k=k)
|
|
144
|
+
verdicts.append((c, v))
|
|
145
|
+
verdict_list = [v for _, v in verdicts]
|
|
146
|
+
summary = summarize_verdicts(verdict_list)
|
|
147
|
+
summary["cases"] = [
|
|
148
|
+
{"session_id": c.session_id, "query_preview": c.query[:100],
|
|
149
|
+
"outcome": v.outcome.value, "reasoning": v.reasoning,
|
|
150
|
+
"confidence": v.confidence}
|
|
151
|
+
for c, v in verdicts
|
|
152
|
+
]
|
|
153
|
+
return summary
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def summarize_verdicts(verdicts: list[CounterfactualVerdict]) -> dict:
|
|
157
|
+
n = len(verdicts)
|
|
158
|
+
if n == 0:
|
|
159
|
+
return {"n_cases": 0, "win_count": 0, "tie_count": 0, "loss_count": 0,
|
|
160
|
+
"win_pct": 0.0, "tie_pct": 0.0, "loss_pct": 0.0,
|
|
161
|
+
"do_no_harm_pct": 100.0}
|
|
162
|
+
|
|
163
|
+
wins = sum(1 for v in verdicts if v.outcome == Outcome.WIN)
|
|
164
|
+
ties = sum(1 for v in verdicts if v.outcome == Outcome.TIE)
|
|
165
|
+
losses = sum(1 for v in verdicts if v.outcome == Outcome.LOSS)
|
|
166
|
+
|
|
167
|
+
return {
|
|
168
|
+
"n_cases": n,
|
|
169
|
+
"win_count": wins,
|
|
170
|
+
"tie_count": ties,
|
|
171
|
+
"loss_count": losses,
|
|
172
|
+
"win_pct": round(wins / n * 100, 1),
|
|
173
|
+
"tie_pct": round(ties / n * 100, 1),
|
|
174
|
+
"loss_pct": round(losses / n * 100, 1),
|
|
175
|
+
"do_no_harm_pct": round((wins + ties) / n * 100, 1),
|
|
176
|
+
}
|
|
@@ -234,6 +234,19 @@ class SqliteStore:
|
|
|
234
234
|
self.db.commit()
|
|
235
235
|
return cur.lastrowid
|
|
236
236
|
|
|
237
|
+
def get_latest_eval(self, eval_type: str = "counterfactual") -> dict | None:
|
|
238
|
+
row = self.db.execute(
|
|
239
|
+
"SELECT * FROM eval_runs WHERE json_extract(config, '$.type') = ? "
|
|
240
|
+
"ORDER BY created_at DESC LIMIT 1", (eval_type,)).fetchone()
|
|
241
|
+
if not row:
|
|
242
|
+
return None
|
|
243
|
+
return {
|
|
244
|
+
"id": row["id"],
|
|
245
|
+
"created_at": row["created_at"],
|
|
246
|
+
"config": json.loads(row["config"]),
|
|
247
|
+
"metrics": json.loads(row["metrics"]),
|
|
248
|
+
}
|
|
249
|
+
|
|
237
250
|
def log_recall(self, project: str, query_preview: str, hits_count: int,
|
|
238
251
|
top_score: float, tokens_injected: int, latency_ms: float,
|
|
239
252
|
status: str, session_id: str = "") -> None:
|
|
@@ -571,6 +584,43 @@ class SqliteStore:
|
|
|
571
584
|
"tool_call_reduction_pct": reduction,
|
|
572
585
|
}
|
|
573
586
|
|
|
587
|
+
def get_roi_trend(self, project: str | None = None,
|
|
588
|
+
days: int = 30) -> list[dict]:
|
|
589
|
+
where = "WHERE user_timestamp > 0"
|
|
590
|
+
params: list = []
|
|
591
|
+
if project:
|
|
592
|
+
where += " AND project = ?"
|
|
593
|
+
params.append(project)
|
|
594
|
+
|
|
595
|
+
rows = self.db.execute(f"""
|
|
596
|
+
SELECT date(user_timestamp, 'unixepoch', 'localtime') AS day,
|
|
597
|
+
AVG(CASE WHEN had_recall = 1 THEN tool_call_count END) AS avg_with,
|
|
598
|
+
AVG(CASE WHEN had_recall = 0 THEN tool_call_count END) AS avg_without,
|
|
599
|
+
COUNT(CASE WHEN had_recall = 1 THEN 1 END) AS turns_with,
|
|
600
|
+
COUNT(CASE WHEN had_recall = 0 THEN 1 END) AS turns_without,
|
|
601
|
+
COUNT(*) AS turns_total
|
|
602
|
+
FROM turn_metrics {where}
|
|
603
|
+
GROUP BY day
|
|
604
|
+
HAVING turns_with > 0 AND turns_without > 0
|
|
605
|
+
ORDER BY day
|
|
606
|
+
""", params).fetchall()
|
|
607
|
+
|
|
608
|
+
result = []
|
|
609
|
+
for r in rows:
|
|
610
|
+
avg_w = round(r["avg_with"], 2)
|
|
611
|
+
avg_wo = round(r["avg_without"], 2)
|
|
612
|
+
reduction = round((1 - avg_w / avg_wo) * 100, 1) if avg_wo > 0 else 0
|
|
613
|
+
result.append({
|
|
614
|
+
"day": r["day"],
|
|
615
|
+
"avg_with": avg_w,
|
|
616
|
+
"avg_without": avg_wo,
|
|
617
|
+
"reduction_pct": reduction,
|
|
618
|
+
"turns_with": r["turns_with"],
|
|
619
|
+
"turns_without": r["turns_without"],
|
|
620
|
+
"turns_total": r["turns_total"],
|
|
621
|
+
})
|
|
622
|
+
return result
|
|
623
|
+
|
|
574
624
|
def get_onboarding_status(self) -> str:
|
|
575
625
|
chunks = self.db.execute(
|
|
576
626
|
"SELECT COUNT(*) as c FROM artifacts WHERE kind='session_chunk' AND active=1"
|