superlocalmemory 3.4.60 → 3.4.62

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -5,6 +5,50 @@ All notable changes to SuperLocalMemory V3 will be documented in this file.
5
5
  The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
6
  and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
7
 
8
+ ## [3.4.62] - 2026-05-31 — Recall engine pre-warm on startup
9
+
10
+ Adds a `recall-warmup` background thread that fires one full 6-channel recall
11
+ immediately after daemon startup. This loads the graph_edges table (~100 MB,
12
+ 347K rows) into SQLite's page cache before the first user query arrives.
13
+
14
+ Without this, cold first query = 15-24s (reading graph_edges from disk).
15
+ After this warmup, all queries hit warm cache at <2s — for both MCP and dashboard.
16
+ Warmup is non-blocking (daemon stays available), fires after embedding warm.
17
+
18
+ ### Changed
19
+ - `server/unified_daemon.py`: `_warmup_recall()` thread fires after `_warmup_embedder()`
20
+
21
+ ---
22
+
23
+ ## [3.4.61] - 2026-05-31 — Dashboard search fix (in-process engine)
24
+
25
+ **Fixes dashboard search always timing out** with "signal is aborted without reason".
26
+
27
+ ### Root Cause
28
+ `POST /api/search` (used by the SLM dashboard memories pane) called
29
+ `WorkerPool.shared()` — the legacy subprocess-based worker pool from v3.4.32
30
+ (pre-unified-daemon). This spawned a fresh Python subprocess and loaded the full
31
+ SLM engine cold on **every single search request**, taking 15–20s. The browser's
32
+ AbortController always fired before the response arrived.
33
+
34
+ The `/recall` HTTP endpoint (used by MCP `session_init`) uses the daemon engine
35
+ directly and is warm at <1s. The dashboard used a completely different code path.
36
+
37
+ ### Fix
38
+ `search_memories` now calls `_get_engine(request).recall()` — the daemon's own
39
+ in-process engine that is already loaded and shares the warm SQLite page cache.
40
+ Falls back to direct LIKE text search if engine is unavailable during startup.
41
+
42
+ ### Result
43
+ Dashboard search: **<1s warm** (was >15s → browser abort).
44
+ GitHub sync failure shown in sidebar is a separate backup connectivity issue,
45
+ unrelated to search.
46
+
47
+ ### Changed
48
+ - `server/routes/memories.py`: `search_memories` uses daemon engine, not WorkerPool
49
+
50
+ ---
51
+
8
52
  ## [3.4.60] - 2026-05-31 — Daemon OpenMP Crash Hotfix
9
53
 
10
54
  **Hotfix for v3.4.59.** Forces `OMP_NUM_THREADS=1` and `KMP_DUPLICATE_LIB_OK=TRUE`
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "superlocalmemory",
3
- "version": "3.4.60",
3
+ "version": "3.4.62",
4
4
  "description": "Information-geometric agent memory with mathematical guarantees. 4-channel retrieval, Fisher-Rao similarity, zero-LLM mode, EU AI Act compliant. Works with Claude, Cursor, Windsurf, and 17+ AI tools.",
5
5
  "keywords": [
6
6
  "ai-memory",
package/pyproject.toml CHANGED
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "superlocalmemory"
3
- version = "3.4.60"
3
+ version = "3.4.62"
4
4
  description = "Information-geometric agent memory with mathematical guarantees"
5
5
  readme = "README.md"
6
6
  license = {text = "AGPL-3.0-or-later"}
@@ -28,7 +28,7 @@ if "OMP_NUM_THREADS" not in os.environ:
28
28
  os.environ["OMP_NUM_THREADS"] = "2"
29
29
  # ---------------------------------------------------------------------------
30
30
 
31
- __version__ = "3.4.60"
31
+ __version__ = "3.4.62"
32
32
 
33
33
  _REQUIRED_VERSIONS = {
34
34
  "sentence_transformers": "5.3.0",
@@ -398,27 +398,47 @@ async def get_graph(
398
398
 
399
399
  @router.post("/api/search")
400
400
  async def search_memories(request: Request, body: SearchRequest):
401
- """Semantic search via subprocess worker pool (memory-isolated).
401
+ """Semantic search using the daemon's in-process engine.
402
402
 
403
- v3.4.32: marks recall in-flight so the pending materializer yields.
403
+ v3.4.61: Replaced WorkerPool.shared() (subprocess-based, cold-starts on
404
+ every request, always >15s) with the daemon's own engine that is already
405
+ loaded and warm. WorkerPool.shared() was legacy from v3.4.32 before the
406
+ unified daemon architecture. Using the daemon engine matches what the /recall
407
+ HTTP endpoint does and shares its warm SQLite page cache, bringing dashboard
408
+ search from >15s timeout to <1s warm.
409
+
410
+ Falls back to direct DB LIKE search if engine is unavailable.
404
411
  """
405
412
  from superlocalmemory.core.recall_gate import begin_recall, end_recall
406
413
  begin_recall()
407
414
  try:
408
- from superlocalmemory.core.worker_pool import WorkerPool
409
- pool = WorkerPool.shared()
410
- result = pool.recall(body.query, limit=body.limit)
411
-
412
- if result.get("ok"):
415
+ # Use the daemon engine directly — already loaded, shares warm cache
416
+ engine = _get_engine(request)
417
+ if engine is not None:
418
+ import time as _time
419
+ t0 = _time.monotonic()
420
+ response = engine.recall(body.query, limit=body.limit)
421
+ elapsed_ms = round((_time.monotonic() - t0) * 1000, 1)
422
+ results = []
423
+ for r in response.results[: body.limit]:
424
+ results.append({
425
+ "fact_id": r.fact.fact_id,
426
+ "memory_id": getattr(r.fact, "memory_id", ""),
427
+ "content": r.fact.content[:300],
428
+ "score": round(r.score, 4),
429
+ "confidence": round(getattr(r, "confidence", 0.0), 4),
430
+ "channel_scores": getattr(r, "channel_scores", {}),
431
+ "created_at": getattr(r.fact, "created_at", ""),
432
+ })
413
433
  return {
414
434
  "query": body.query,
415
- "results": result.get("results", []),
416
- "total": result.get("result_count", 0),
417
- "query_type": result.get("query_type", "unknown"),
418
- "retrieval_time_ms": result.get("retrieval_time_ms", 0),
435
+ "results": results,
436
+ "total": len(results),
437
+ "query_type": getattr(response, "query_type", "semantic"),
438
+ "retrieval_time_ms": elapsed_ms,
419
439
  }
420
440
 
421
- # Fallback: direct DB text search (no engine needed)
441
+ # Fallback: direct DB text search (engine not yet initialised)
422
442
  conn = get_db_connection()
423
443
  conn.row_factory = dict_factory
424
444
  cursor = conn.cursor()
@@ -574,7 +574,37 @@ async def lifespan(application: FastAPI):
574
574
  logger.info("Embedding worker pre-warmed (model resident, keep_alive=-1)")
575
575
  except Exception as exc:
576
576
  logger.warning("Embedding warmup failed: %s", exc)
577
+
578
+ def _warmup_recall():
579
+ """v3.4.62: Fire a full 6-channel recall after embedding warms up.
580
+
581
+ Loads the graph_edges table (347K rows, ~100 MB) into the SQLite
582
+ page cache. Without this, the first user query takes 15-24s because
583
+ it reads graph_edges from disk. After this warmup completes, all
584
+ subsequent queries hit the warm page cache at <2s.
585
+
586
+ Runs after embedding warm (embed first so recall can use it).
587
+ Named 'recall-warmup' so it appears clearly in thread dumps.
588
+ """
589
+ import time as _t
590
+ # Wait for embedder to finish first (embed is needed by semantic channel)
591
+ for _ in range(60):
592
+ if _embedding_warm:
593
+ break
594
+ _t.sleep(0.5)
595
+ try:
596
+ t0 = _t.monotonic()
597
+ response = engine.recall("memory recall performance", limit=1)
598
+ elapsed = round((_t.monotonic() - t0) * 1000)
599
+ logger.info(
600
+ "Recall engine pre-warmed in %dms — graph page cache now hot "
601
+ "(results=%d)", elapsed, len(response.results),
602
+ )
603
+ except Exception as exc:
604
+ logger.warning("Recall warmup failed (non-fatal): %s", exc)
605
+
577
606
  threading.Thread(target=_warmup_embedder, daemon=True, name="embed-warmup").start()
607
+ threading.Thread(target=_warmup_recall, daemon=True, name="recall-warmup").start()
578
608
 
579
609
  # v3.4.37: QueueConsumer uses daemon's engine directly via adapter.
580
610
  # Previously routed through WorkerPool → recall_worker subprocess,
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: superlocalmemory
3
- Version: 3.4.60
3
+ Version: 3.4.62
4
4
  Summary: Information-geometric agent memory with mathematical guarantees
5
5
  Author-email: Varun Pratap Bhardwaj <admin@superlocalmemory.com>
6
6
  License: AGPL-3.0-or-later