ltcai 10.9.0 → 11.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (126) hide show
  1. package/README.md +46 -64
  2. package/docs/CHANGELOG.md +72 -237
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/ONBOARDING.md +1 -1
  6. package/docs/OPERATIONS.md +1 -1
  7. package/docs/PERFORMANCE.md +78 -7
  8. package/docs/TRUST_MODEL.md +1 -1
  9. package/docs/WHY_LATTICE.md +1 -1
  10. package/docs/kg-schema.md +1 -1
  11. package/docs/v11.1.0_PRODUCT_INTELLIGENCE_PLAN.md +313 -0
  12. package/lattice_brain/__init__.py +1 -1
  13. package/lattice_brain/graph/_kg_contract.py +11 -0
  14. package/lattice_brain/graph/curator.py +1 -1
  15. package/lattice_brain/graph/fusion.py +184 -3
  16. package/lattice_brain/graph/proactive.py +138 -1
  17. package/lattice_brain/graph/projection.py +10 -2
  18. package/lattice_brain/graph/retrieval.py +137 -7
  19. package/lattice_brain/graph/retrieval_docgen.py +20 -20
  20. package/lattice_brain/graph/retrieval_policy.py +6 -0
  21. package/lattice_brain/graph/retrieval_reads.py +188 -1
  22. package/lattice_brain/graph/retrieval_vector.py +474 -110
  23. package/lattice_brain/graph/schema.py +125 -2
  24. package/lattice_brain/graph/vector_index/__init__.py +85 -0
  25. package/lattice_brain/graph/vector_index/base.py +170 -0
  26. package/lattice_brain/graph/vector_index/brute_force.py +114 -0
  27. package/lattice_brain/graph/vector_index/hnsw.py +293 -0
  28. package/lattice_brain/graph/vector_index/jobs.py +287 -0
  29. package/lattice_brain/graph/vector_index/quantized.py +151 -0
  30. package/lattice_brain/graph/vector_index/selector.py +131 -0
  31. package/lattice_brain/ingestion.py +50 -3
  32. package/lattice_brain/portability.py +654 -2
  33. package/lattice_brain/runtime/agent_runtime.py +1 -1
  34. package/lattice_brain/runtime/contracts.py +1 -1
  35. package/lattice_brain/runtime/multi_agent.py +1 -1
  36. package/lattice_brain/self_model.py +620 -0
  37. package/lattice_brain/synthesis.py +801 -0
  38. package/latticeai/__init__.py +1 -1
  39. package/latticeai/api/brain_intelligence.py +83 -1
  40. package/latticeai/api/chat_stream.py +4 -1
  41. package/latticeai/api/local_files.py +62 -0
  42. package/latticeai/api/models.py +1 -1
  43. package/latticeai/api/portability.py +130 -1
  44. package/latticeai/api/security_dashboard.py +48 -13
  45. package/latticeai/api/voice_capture.py +4 -1
  46. package/latticeai/api/workspace.py +11 -5
  47. package/latticeai/core/embedding_providers.py +20 -1
  48. package/latticeai/core/legacy_compatibility.py +1 -1
  49. package/latticeai/core/marketplace.py +1 -1
  50. package/latticeai/core/messages.py +14 -0
  51. package/latticeai/core/model_compat.py +2 -2
  52. package/latticeai/core/tool_registry.py +0 -7
  53. package/latticeai/core/workspace_os_constants.py +1 -1
  54. package/latticeai/core/workspace_os_utils.py +4 -50
  55. package/latticeai/core/workspace_review_items.py +12 -1
  56. package/latticeai/integrations/telegram_bot.py +28 -10
  57. package/latticeai/models/router.py +1 -1
  58. package/latticeai/runtime/access_runtime.py +1 -1
  59. package/latticeai/runtime/network_boundary_wiring.py +9 -5
  60. package/latticeai/runtime/permission_mode_wiring.py +9 -6
  61. package/latticeai/runtime/router_registration.py +3 -0
  62. package/latticeai/services/architecture_readiness.py +1 -1
  63. package/latticeai/services/brain_intelligence.py +253 -0
  64. package/latticeai/services/memory_service.py +1 -1
  65. package/latticeai/services/model_catalog.py +4 -3
  66. package/latticeai/services/model_engines.py +28 -14
  67. package/latticeai/services/obsidian_bridge.py +618 -0
  68. package/latticeai/services/product_readiness.py +5 -3
  69. package/latticeai/tools/filesystem.py +4 -1
  70. package/package.json +1 -1
  71. package/scripts/bench_vector_index.py +295 -0
  72. package/scripts/check_current_release_docs.mjs +4 -2
  73. package/scripts/release_screen_claims.json +31 -0
  74. package/src-tauri/Cargo.lock +1 -1
  75. package/src-tauri/Cargo.toml +1 -1
  76. package/src-tauri/tauri.conf.json +1 -1
  77. package/static/app/asset-manifest.json +37 -37
  78. package/static/app/assets/{Act-CS9IeqUX.js → Act-D4zSxFR-.js} +1 -1
  79. package/static/app/assets/{AdminConsole-3UkIEWGA.js → AdminConsole-w5jBfPt2.js} +1 -1
  80. package/static/app/assets/{Brain-B22EmNqS.js → Brain-C2EqQg74.js} +2 -2
  81. package/static/app/assets/BrainHome-CvXS6XiQ.js +2 -0
  82. package/static/app/assets/BrainSignals-DOE_KhOU.js +1 -0
  83. package/static/app/assets/Capture-DPqpGK8d.js +1 -0
  84. package/static/app/assets/{CommandPalette-86m4FCcN.js → CommandPalette-CNf7h5fp.js} +1 -1
  85. package/static/app/assets/Library-BN0HYOfc.js +1 -0
  86. package/static/app/assets/LivingBrain-Dfq_wEDI.js +1 -0
  87. package/static/app/assets/ProductFlow-B-3O0rNV.js +1 -0
  88. package/static/app/assets/{ReviewCard-BepjSDpN.js → ReviewCard-gZ-tdqFM.js} +1 -1
  89. package/static/app/assets/System-BElUcSSw.js +1 -0
  90. package/static/app/assets/arrow-left-CFNIMjhv.js +1 -0
  91. package/static/app/assets/{bot-DQj0-LkM.js → bot--qYHMtkP.js} +1 -1
  92. package/static/app/assets/brain-DDCLjRqO.js +1 -0
  93. package/static/app/assets/{button-CmTknyAP.js → button-51Z3rsuv.js} +1 -1
  94. package/static/app/assets/{circle-pause-yTCWRziJ.js → circle-pause-CMIiMaQl.js} +1 -1
  95. package/static/app/assets/{circle-play-Ccrva84R.js → circle-play-DZoO_cfG.js} +1 -1
  96. package/static/app/assets/{cpu-CbJqWTlS.js → cpu-Bs6uc9W9.js} +1 -1
  97. package/static/app/assets/{download-CkSzbzU-.js → download-G-2olkWz.js} +1 -1
  98. package/static/app/assets/{folder-open-CKyjQ4PU.js → folder-open-CTOspnmb.js} +1 -1
  99. package/static/app/assets/{hard-drive-DAzk9um0.js → hard-drive-CewHWJhn.js} +1 -1
  100. package/static/app/assets/index-CkzokZAj.css +2 -0
  101. package/static/app/assets/{index-CxOcwsHV.js → index-D7Rr-J2Y.js} +3 -3
  102. package/static/app/assets/{input-DcMETmZ7.js → input-D4w_BZWl.js} +1 -1
  103. package/static/app/assets/{permissionCopy-BVf13_25.js → permissionCopy-CosBEXAZ.js} +1 -1
  104. package/static/app/assets/primitives-d0g9pvzS.js +1 -0
  105. package/static/app/assets/search-BLCYt75v.js +1 -0
  106. package/static/app/assets/{share-2-COWCHNZm.js → share-2-NmD7e_oV.js} +1 -1
  107. package/static/app/assets/{shield-alert-BDrvilyK.js → shield-alert-CcQeMuju.js} +1 -1
  108. package/static/app/assets/{textarea-rUmsc8cP.js → textarea-BPAJDc-0.js} +1 -1
  109. package/static/app/assets/{useFocusTrap-Bi5UY_8v.js → useFocusTrap-C7YLdTBC.js} +1 -1
  110. package/static/app/assets/{useQuery-C-AicB-3.js → useQuery-DRyD9opW.js} +1 -1
  111. package/static/app/assets/{utils-BMwWg78e.js → utils-DG1_ExrP.js} +3 -3
  112. package/static/app/assets/{workspace-Y93tls8P.js → workspace-CWVf3gsI.js} +1 -1
  113. package/static/app/index.html +4 -4
  114. package/static/sw.js +1 -1
  115. package/static/app/assets/BrainHome-CMDqgJF4.js +0 -2
  116. package/static/app/assets/BrainSignals-BeE8RJo3.js +0 -1
  117. package/static/app/assets/Capture-ZX9bQh68.js +0 -1
  118. package/static/app/assets/Library-Bhz5LUca.js +0 -1
  119. package/static/app/assets/LivingBrain-BSa0wpFG.js +0 -1
  120. package/static/app/assets/ProductFlow-IP4Q-5aQ.js +0 -1
  121. package/static/app/assets/System-Bwx1h_jT.js +0 -1
  122. package/static/app/assets/arrow-left-ig7AZU8B.js +0 -1
  123. package/static/app/assets/brain-D8OEwmVj.js +0 -1
  124. package/static/app/assets/index-BfD-jhA9.css +0 -2
  125. package/static/app/assets/primitives-CQV9Q2YM.js +0 -1
  126. package/static/app/assets/search-CXGASMMH.js +0 -1
@@ -1,9 +1,22 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import dataclasses
3
4
  from typing import TYPE_CHECKING
4
5
 
5
6
  # ruff: noqa: F403,F405
6
7
  from ._kg_common import * # noqa: F403,F401
8
+ from .vector_index import (
9
+ DEFAULT_VECTOR_INDEX,
10
+ HNSW_BACKEND,
11
+ VECTOR_INDEX_ENV,
12
+ BackendSelection,
13
+ HnswIndex,
14
+ IndexItem,
15
+ VectorEmbedQueue,
16
+ VectorIndex,
17
+ build_index,
18
+ resolve_vector_index,
19
+ )
7
20
 
8
21
  # The cross-mixin surface (`_connect`, `_upsert_node`, …) is declared in
9
22
  # `_kg_contract.KnowledgeGraphCore`. It is a typing-only base: at runtime this
@@ -33,6 +46,14 @@ DEFAULT_VECTOR_MAX_CANDIDATES = 10_000
33
46
  #: Upper bound for a configured cap; ``0``/``None`` still means uncapped.
34
47
  VECTOR_MAX_CANDIDATES_CEILING = 500_000
35
48
 
49
+ # ── scan batching (v11.1.0) ──────────────────────────────────────────────────
50
+ # The exact scan hands its candidates to a VectorIndex, which by definition
51
+ # holds what it is given. Handing it the whole result set would make peak
52
+ # memory O(rows × dim) floats, so the scan feeds it in fixed batches instead:
53
+ # exhaustive backends score every batch independently, so the union is
54
+ # identical to one big pass, at O(batch × dim) resident cost.
55
+ VECTOR_SCAN_BATCH = 512
56
+
36
57
 
37
58
  def _configured_vector_max_candidates() -> Optional[int]:
38
59
  """Resolve the candidate cap from the environment (None = uncapped).
@@ -597,6 +618,11 @@ class KnowledgeGraphVectorMixin(_Core):
597
618
  if isinstance(storage_capabilities, dict)
598
619
  else False
599
620
  ),
621
+ # v11.1.0: which in-process index scores a search, and — when
622
+ # the configured one could not be used — the reason it was
623
+ # substituted, so an unavailable optional extra is visible
624
+ # here instead of only showing up as "search feels slow".
625
+ "vector_index": self._vector_index_selection().as_dict(),
600
626
  },
601
627
  "source_items": len(source_items),
602
628
  "indexed_items": sum(vector_counts.values()),
@@ -668,6 +694,16 @@ class KnowledgeGraphVectorMixin(_Core):
668
694
  "total_items": 0,
669
695
  "detail": f"vector index status unavailable: {exc}",
670
696
  }
697
+ return self._vector_freshness_summary(status)
698
+
699
+ def _vector_freshness_summary(self, status: Dict[str, Any]) -> Dict[str, Any]:
700
+ """The freshness reduction of an already-read :meth:`index_status`.
701
+
702
+ Split out so :meth:`vector_freshness_breakdown` can report both shapes
703
+ from one index scan; ``index_status`` walks every source item, and
704
+ calling it twice to answer one question about freshness would double
705
+ the most expensive read in this module.
706
+ """
671
707
  pending = int(status.get("pending_items") or 0)
672
708
  total = int(status.get("source_items") or 0)
673
709
  embedder = status.get("embedder") or {}
@@ -720,13 +756,93 @@ class KnowledgeGraphVectorMixin(_Core):
720
756
  "detail": detail,
721
757
  }
722
758
 
759
+ @property
760
+ def vector_queue(self) -> VectorEmbedQueue:
761
+ """This store's durable pending-embed backlog (created on demand).
762
+
763
+ Built lazily rather than in ``__init__`` so opening a graph never
764
+ creates a table nobody asked for, and hung off the store so the
765
+ ingestion pipeline and the freshness report share one backlog instead
766
+ of each keeping a private view of it.
767
+ """
768
+ queue = getattr(self, "_vector_queue", None)
769
+ if queue is None:
770
+ queue = VectorEmbedQueue(
771
+ db_path=self.db_path, indexer=self.index_node_incremental
772
+ )
773
+ self._vector_queue = queue
774
+ return queue
775
+
776
+ def vector_freshness_breakdown(self) -> Dict[str, Any]:
777
+ """The four numbers behind :meth:`vector_freshness` (v11.1.0).
778
+
779
+ ``vector_freshness()`` answers one question — *is the index behind?* —
780
+ and its four keys are a frozen wire contract that surfaces already
781
+ read, so this is a sibling rather than an extension of it. The split
782
+ matters because "12 pending" hides two different situations: twelve
783
+ items never embedded (a new import) and twelve items whose text
784
+ changed under an existing embedding (edits). Only the second means
785
+ current answers are quietly wrong.
786
+
787
+ ``queued`` counts the durable background backlog
788
+ (:class:`~lattice_brain.graph.vector_index.VectorEmbedQueue`), and is
789
+ ``None`` when that queue has no database to persist to — never ``0``,
790
+ which would claim an empty backlog nobody measured.
791
+
792
+ Never raises: an unreadable index reports ``status="unavailable"``
793
+ with the cause in ``detail`` and zeroed counts.
794
+ """
795
+ status: Dict[str, Any] = {}
796
+ summary: Dict[str, Any]
797
+ try:
798
+ status = self.index_status()
799
+ except Exception as exc: # noqa: BLE001 — freshness must degrade, not fail
800
+ summary = {
801
+ "status": "unavailable",
802
+ "pending_items": 0,
803
+ "total_items": 0,
804
+ "detail": f"vector index status unavailable: {exc}",
805
+ }
806
+ else:
807
+ summary = self._vector_freshness_summary(status)
808
+ breakdown: Dict[str, Any] = {
809
+ "status": summary["status"],
810
+ "detail": summary["detail"],
811
+ "embedded": int(status.get("ready_items") or 0),
812
+ "pending": int(summary["pending_items"]),
813
+ "missing": int(status.get("missing_items") or 0),
814
+ "stale": int(status.get("stale_items") or 0),
815
+ "total": int(summary["total_items"]),
816
+ "queued": None,
817
+ }
818
+ queue = self.vector_queue
819
+ if queue.available:
820
+ breakdown["queued"] = int(queue.pending_count())
821
+ return breakdown
822
+
823
+ def _vector_index_selection(self) -> BackendSelection:
824
+ """The configured in-process index backend (``LATTICEAI_VECTOR_INDEX``).
825
+
826
+ Resolved per call rather than cached: the env var is the whole control
827
+ surface, and a cached selection would make a config change look like
828
+ it had no effect. The only expensive part — importing ``hnswlib`` —
829
+ is already cached by ``sys.modules``.
830
+ """
831
+ return resolve_vector_index()
832
+
723
833
  def _vector_search_backend(self) -> str:
724
- """Which backend actually scores the vectors, per storage capabilities.
834
+ """Which backend actually scores the vectors.
725
835
 
726
- sqlite-vec exposes an ANN index; without it this store scores rows in
727
- Python (``bruteforce-cosine``). Never raises — a capability probe
728
- failure means "we cannot claim ANN", which is the brute-force answer.
836
+ An explicitly selected in-process index (quantized / hnsw) wins,
837
+ because it is the thing that will do the scoring. Otherwise this is
838
+ the storage layer's answer: sqlite-vec exposes an ANN index; without
839
+ it this store scores rows in Python (``bruteforce-cosine``). Never
840
+ raises — a capability probe failure means "we cannot claim ANN",
841
+ which is the brute-force answer.
729
842
  """
843
+ selection = self._vector_index_selection()
844
+ if selection.name != DEFAULT_VECTOR_INDEX:
845
+ return selection.backend
730
846
  try:
731
847
  capabilities = self.storage_engine.capabilities().as_dict()
732
848
  except Exception: # noqa: BLE001 — a probe failure is not an ANN index
@@ -741,8 +857,15 @@ class KnowledgeGraphVectorMixin(_Core):
741
857
  cap: Optional[int],
742
858
  candidates_total: int,
743
859
  candidates_scanned: int,
860
+ approx_detail: Optional[str] = None,
744
861
  ) -> Dict[str, Any]:
745
- """The honest answer to "did this search see the whole index?"."""
862
+ """The honest answer to "did this search see the whole index?".
863
+
864
+ ``approx_detail`` covers the second way recall can be incomplete: an
865
+ ANN backend *visits* the whole index but is not guaranteed to return
866
+ its true top-k. "Scanned N of N" with no detail would read as an exact
867
+ answer, so the approximate backends supply their caveat here.
868
+ """
746
869
  truncated = candidates_scanned < candidates_total
747
870
  detail: Optional[str] = None
748
871
  if truncated:
@@ -750,9 +873,13 @@ class KnowledgeGraphVectorMixin(_Core):
750
873
  f"partial recall: scored the {candidates_scanned} most recently "
751
874
  f"indexed vectors of {candidates_total}. The cut is by index "
752
875
  f"recency, not similarity, so older matches were never compared. "
753
- f"Raise {VECTOR_MAX_CANDIDATES_ENV} (0 = scan everything) or "
754
- f"install sqlite-vec for an ANN index."
876
+ f"Raise {VECTOR_MAX_CANDIDATES_ENV} (0 = scan everything), or "
877
+ f"switch to an index that covers the whole set: "
878
+ f"{VECTOR_INDEX_ENV}=hnsw (needs the optional hnsw extra) or "
879
+ f"install sqlite-vec."
755
880
  )
881
+ elif approx_detail:
882
+ detail = approx_detail
756
883
  return {
757
884
  "backend": backend,
758
885
  "max_candidates": cap,
@@ -783,6 +910,306 @@ class KnowledgeGraphVectorMixin(_Core):
783
910
  # Never scan fewer rows than the caller intends to receive.
784
911
  return max(limit, cap)
785
912
 
913
+ # One row shape feeds every vector match, so both the exact scan and the
914
+ # ANN lookup project exactly the same columns; only the WHERE/ORDER tail
915
+ # differs. Bound values are always parameters — the interpolation below is
916
+ # a placeholder list, never data.
917
+ _VECTOR_ROW_SELECT = """
918
+ SELECT
919
+ ve.item_id, ve.item_type, ve.source_node, ve.embedding,
920
+ ve.embedding_dim, ve.embedding_model, ve.metadata_json AS vector_metadata,
921
+ n.type AS node_type, n.title AS node_title, n.summary AS node_summary,
922
+ n.metadata_json AS node_metadata, n.updated_at AS node_updated_at,
923
+ c.text AS chunk_text, c.source_node AS parent_node_id,
924
+ c.metadata_json AS chunk_metadata,
925
+ pn.type AS parent_type, pn.title AS parent_title,
926
+ pn.summary AS parent_summary, pn.metadata_json AS parent_metadata,
927
+ pn.updated_at AS parent_updated_at
928
+ FROM vector_embeddings ve
929
+ LEFT JOIN nodes n ON n.id=ve.source_node
930
+ LEFT JOIN chunks c ON c.id=ve.item_id
931
+ LEFT JOIN nodes pn ON pn.id=c.source_node
932
+ WHERE ve.embedding_model=? AND ve.embedding_dim=?
933
+ """
934
+
935
+ @staticmethod
936
+ def _vector_match(row: sqlite3.Row, score: float) -> Dict[str, Any]:
937
+ """One scored embedding row → one search match (pure projection)."""
938
+ is_chunk = row["item_type"] == "chunk"
939
+ summary = (
940
+ row["chunk_text"] if is_chunk and row["chunk_text"] else row["node_summary"]
941
+ )
942
+ parent_metadata = _safe_loads(row["parent_metadata"])
943
+ node_metadata = _safe_loads(row["node_metadata"])
944
+ # Citation precision (review 2026-07-27 P1 #4): a chunk hit used to
945
+ # cite only its parent document, so a 200-page PDF answered with
946
+ # "from report.pdf". The chunk's own provenance (section heading,
947
+ # page, offset) now rides along, and `locator` is the one-line
948
+ # human form — absent when the chunk carries no such metadata.
949
+ chunk_metadata = _safe_loads(row["chunk_metadata"]) if is_chunk else {}
950
+ locator = citation_locator(chunk_metadata)
951
+ return {
952
+ "id": row["item_id"],
953
+ "node_id": row["parent_node_id"]
954
+ if is_chunk and row["parent_node_id"]
955
+ else row["source_node"],
956
+ "item_type": row["item_type"],
957
+ "type": "Chunk" if is_chunk else row["node_type"],
958
+ "title": row["parent_title"]
959
+ if is_chunk and row["parent_title"]
960
+ else row["node_title"],
961
+ "summary": _clean_text(summary or "")[:1000],
962
+ "score": round(float(score), 6),
963
+ "metadata": {
964
+ **(parent_metadata if is_chunk else node_metadata),
965
+ "vector": _safe_loads(row["vector_metadata"]),
966
+ "parent_node_id": row["parent_node_id"],
967
+ "parent_type": row["parent_type"],
968
+ **({"chunk": chunk_metadata} if chunk_metadata else {}),
969
+ **({"locator": locator} if locator else {}),
970
+ },
971
+ "updated_at": row["parent_updated_at"]
972
+ if is_chunk and row["parent_updated_at"]
973
+ else row["node_updated_at"],
974
+ }
975
+
976
+ @staticmethod
977
+ def _flush_scan_batch(
978
+ index: VectorIndex,
979
+ batch: List[IndexItem],
980
+ query_vector: List[float],
981
+ min_score: float,
982
+ scores: Dict[str, float],
983
+ ) -> None:
984
+ """Score one batch into ``scores`` and empty it."""
985
+ if not batch:
986
+ return
987
+ index.rebuild(batch)
988
+ scores.update(
989
+ index.search(query_vector, len(batch), filter={"min_score": min_score})
990
+ )
991
+ batch.clear()
992
+
993
+ def _score_vector_rows(
994
+ self,
995
+ rows: List[sqlite3.Row],
996
+ query_vector: List[float],
997
+ selection: BackendSelection,
998
+ *,
999
+ min_score: float,
1000
+ ) -> Dict[str, float]:
1001
+ """``item_id -> score`` for every row that clears ``min_score``."""
1002
+ index = build_index(
1003
+ selection,
1004
+ dim=int(self._embedding_model.dim),
1005
+ similarity=self._embedding_model.similarity,
1006
+ )
1007
+ scores: Dict[str, float] = {}
1008
+ batch: List[IndexItem] = []
1009
+ for row in rows:
1010
+ batch.append(
1011
+ (
1012
+ str(row["item_id"]),
1013
+ self._embedding_model.decode(
1014
+ row["embedding"], row["embedding_dim"]
1015
+ ),
1016
+ {"item_type": row["item_type"]},
1017
+ )
1018
+ )
1019
+ if len(batch) >= VECTOR_SCAN_BATCH:
1020
+ self._flush_scan_batch(index, batch, query_vector, min_score, scores)
1021
+ self._flush_scan_batch(index, batch, query_vector, min_score, scores)
1022
+ return scores
1023
+
1024
+ def _vector_search_scan(
1025
+ self,
1026
+ query: str,
1027
+ query_vector: List[float],
1028
+ selection: BackendSelection,
1029
+ *,
1030
+ limit: int,
1031
+ min_score: float,
1032
+ backend: str,
1033
+ cap: Optional[int],
1034
+ ) -> Dict[str, Any]:
1035
+ """Exhaustive scan of (at most ``cap``) rows — the historical path."""
1036
+ sql = self._VECTOR_ROW_SELECT + " ORDER BY ve.indexed_at DESC"
1037
+ params: List[Any] = [
1038
+ self._embedding_model.model_id,
1039
+ self._embedding_model.dim,
1040
+ ]
1041
+ if cap is not None:
1042
+ sql += " LIMIT ?"
1043
+ params.append(cap)
1044
+ with self._connect() as conn:
1045
+ # Counted in the same transaction as the scan so "scanned N of M"
1046
+ # cannot describe two different index states.
1047
+ candidates_total = int(
1048
+ conn.execute(
1049
+ "SELECT COUNT(*) AS c FROM vector_embeddings "
1050
+ "WHERE embedding_model=? AND embedding_dim=?",
1051
+ (self._embedding_model.model_id, self._embedding_model.dim),
1052
+ ).fetchone()["c"]
1053
+ )
1054
+ rows = conn.execute(sql, tuple(params)).fetchall()
1055
+ recall = self._recall_report(
1056
+ backend=backend,
1057
+ cap=cap,
1058
+ candidates_total=candidates_total,
1059
+ candidates_scanned=len(rows),
1060
+ approx_detail=(
1061
+ "approximate backend: every candidate was compared, but the "
1062
+ "scores are estimates, so near-ties can reorder"
1063
+ if selection.approx
1064
+ else None
1065
+ ),
1066
+ )
1067
+ scores = self._score_vector_rows(
1068
+ rows, query_vector, selection, min_score=min_score
1069
+ )
1070
+ # Rows are walked in index order (not score order) so the sort below
1071
+ # sees exactly the input ordering the pre-11.1.0 inline loop produced:
1072
+ # a stable sort makes that the tie-break of last resort.
1073
+ scored = [
1074
+ self._vector_match(row, scores[str(row["item_id"])])
1075
+ for row in rows
1076
+ if str(row["item_id"]) in scores
1077
+ ]
1078
+ scored.sort(
1079
+ key=lambda item: (item["score"], item.get("updated_at") or ""), reverse=True
1080
+ )
1081
+ return {
1082
+ "query": query,
1083
+ "embedding_model": self._embedding_model.model_id,
1084
+ "embedding_dim": self._embedding_model.dim,
1085
+ "matches": scored[:limit],
1086
+ "recall": recall,
1087
+ "index": selection.as_dict(),
1088
+ }
1089
+
1090
+ def _iter_vector_index_items(
1091
+ self, conn: sqlite3.Connection, model_id: str, dim: int
1092
+ ) -> Iterator[IndexItem]:
1093
+ """Every embedding for ``model_id``/``dim`` as index items."""
1094
+ for row in conn.execute(
1095
+ "SELECT item_id, embedding, embedding_dim FROM vector_embeddings "
1096
+ "WHERE embedding_model=? AND embedding_dim=? ORDER BY item_id ASC",
1097
+ (model_id, dim),
1098
+ ):
1099
+ yield (
1100
+ str(row["item_id"]),
1101
+ self._embedding_model.decode(row["embedding"], row["embedding_dim"]),
1102
+ {},
1103
+ )
1104
+
1105
+ def _vector_rows_by_id(
1106
+ self, conn: sqlite3.Connection, item_ids: List[str]
1107
+ ) -> List[sqlite3.Row]:
1108
+ """Full match rows for the ids an ANN lookup returned."""
1109
+ if not item_ids:
1110
+ return []
1111
+ placeholders = ",".join("?" * len(item_ids))
1112
+ return conn.execute(
1113
+ self._VECTOR_ROW_SELECT + f" AND ve.item_id IN ({placeholders})",
1114
+ (self._embedding_model.model_id, self._embedding_model.dim, *item_ids),
1115
+ ).fetchall()
1116
+
1117
+ def _hnsw_index(
1118
+ self,
1119
+ conn: sqlite3.Connection,
1120
+ fingerprint: str,
1121
+ model_id: str,
1122
+ dim: int,
1123
+ ) -> HnswIndex:
1124
+ """The live ANN graph for ``fingerprint`` — cache, sidecar, or rebuild.
1125
+
1126
+ Held on the store for the process's lifetime, because reading a
1127
+ 50 000-vector graph off disk costs roughly as much as the search it
1128
+ enables: paying it per query gave back most of the speedup (105 ms
1129
+ instead of 15 ms at 50k). The fingerprint — model, dimension, row
1130
+ count, newest ``indexed_at`` — is what makes the cache safe: any write
1131
+ to ``vector_embeddings`` changes it, and a changed fingerprint is
1132
+ never served from the cache or from the sidecar.
1133
+ """
1134
+ cached = getattr(self, "_hnsw_cached", None)
1135
+ if cached is not None and cached[0] == fingerprint:
1136
+ return cached[1]
1137
+ index = HnswIndex(dim=dim)
1138
+ if not index.load(self.db_path, fingerprint=fingerprint):
1139
+ index.rebuild(self._iter_vector_index_items(conn, model_id, dim))
1140
+ index.save(self.db_path, fingerprint=fingerprint)
1141
+ self._hnsw_cached = (fingerprint, index)
1142
+ return index
1143
+
1144
+ def _vector_search_ann(
1145
+ self,
1146
+ query: str,
1147
+ query_vector: List[float],
1148
+ selection: BackendSelection,
1149
+ *,
1150
+ limit: int,
1151
+ min_score: float,
1152
+ backend: str,
1153
+ ) -> Optional[Dict[str, Any]]:
1154
+ """Approximate top-k via the persisted HNSW sidecar.
1155
+
1156
+ Two phases instead of one: ask the graph for ids, then read only those
1157
+ rows. That is where the speed comes from — the exact scan pays to
1158
+ decode every embedding on every query, and this pays it once per
1159
+ index generation.
1160
+
1161
+ The sidecar is keyed by ``model:dim:rows:newest`` so any write to
1162
+ ``vector_embeddings`` invalidates it and the next search rebuilds.
1163
+ Returns ``None`` when the index is empty, which the caller answers
1164
+ with the ordinary (equally empty, but honestly reported) scan.
1165
+ """
1166
+ model_id = self._embedding_model.model_id
1167
+ dim = int(self._embedding_model.dim)
1168
+ with self._connect() as conn:
1169
+ head = conn.execute(
1170
+ "SELECT COUNT(*) AS c, MAX(indexed_at) AS newest FROM vector_embeddings "
1171
+ "WHERE embedding_model=? AND embedding_dim=?",
1172
+ (model_id, dim),
1173
+ ).fetchone()
1174
+ candidates_total = int(head["c"])
1175
+ if candidates_total == 0:
1176
+ return None
1177
+ fingerprint = f"{model_id}:{dim}:{candidates_total}:{head['newest']}"
1178
+ index = self._hnsw_index(conn, fingerprint, model_id, dim)
1179
+ pairs = index.search(query_vector, limit, filter={"min_score": min_score})
1180
+ rows = {
1181
+ str(row["item_id"]): row
1182
+ for row in self._vector_rows_by_id(
1183
+ conn, [item_id for item_id, _ in pairs]
1184
+ )
1185
+ }
1186
+ scored = [
1187
+ self._vector_match(rows[item_id], score)
1188
+ for item_id, score in pairs
1189
+ if item_id in rows
1190
+ ]
1191
+ scored.sort(
1192
+ key=lambda item: (item["score"], item.get("updated_at") or ""), reverse=True
1193
+ )
1194
+ return {
1195
+ "query": query,
1196
+ "embedding_model": model_id,
1197
+ "embedding_dim": dim,
1198
+ "matches": scored[:limit],
1199
+ "recall": self._recall_report(
1200
+ backend=backend,
1201
+ cap=None,
1202
+ candidates_total=candidates_total,
1203
+ candidates_scanned=candidates_total,
1204
+ approx_detail=(
1205
+ "approximate nearest-neighbour search: the whole index is "
1206
+ "reachable but the true top-k is not guaranteed — compare "
1207
+ "with scripts/bench_vector_index.py"
1208
+ ),
1209
+ ),
1210
+ "index": {**selection.as_dict(), "sidecar": index.loaded_from_sidecar},
1211
+ }
1212
+
786
1213
  def vector_search(
787
1214
  self,
788
1215
  query: str,
@@ -791,7 +1218,7 @@ class KnowledgeGraphVectorMixin(_Core):
791
1218
  min_score: float = 0.0,
792
1219
  max_candidates: Optional[int] = None,
793
1220
  ) -> Dict[str, Any]:
794
- """Brute-force cosine search over the vector index.
1221
+ """Cosine search over the vector index (exact by default).
795
1222
 
796
1223
  ``max_candidates`` bounds how many indexed rows are scored; ``None``
797
1224
  (the default) resolves it from ``LATTICEAI_VECTOR_MAX_CANDIDATES``
@@ -802,6 +1229,17 @@ class KnowledgeGraphVectorMixin(_Core):
802
1229
  (``{backend, max_candidates, candidates_total, candidates_scanned,
803
1230
  truncated, detail}``) instead of being hidden, and callers/UIs are
804
1231
  expected to surface ``recall.truncated``.
1232
+
1233
+ v11.1.0: the scoring itself now lives in
1234
+ :mod:`lattice_brain.graph.vector_index`. ``LATTICEAI_VECTOR_INDEX``
1235
+ picks the backend — ``brute`` (default, exact, byte-compatible with
1236
+ every previous release), ``quantized`` (int8, exhaustive, approximate
1237
+ scores) or ``hnsw`` (approximate nearest neighbour, needs the optional
1238
+ ``hnsw`` extra). The resolved backend and any fallback reason ride
1239
+ along in the additive ``index`` block, whose ``approx`` flag is the
1240
+ one bit a caller needs to know whether "not found" is a fact or an
1241
+ estimate. The empty-query early return is deliberately unchanged: no
1242
+ query means no index was consulted, so there is nothing to report.
805
1243
  """
806
1244
  query = str(query or "").strip()
807
1245
  limit = max(1, min(int(limit or 30), 100))
@@ -821,109 +1259,35 @@ class KnowledgeGraphVectorMixin(_Core):
821
1259
  "detail": None,
822
1260
  },
823
1261
  }
1262
+ selection = self._vector_index_selection()
824
1263
  query_vector = self._embedding_model.embed(query)
825
- with self._connect() as conn:
826
- # Counted in the same transaction as the scan so "scanned N of M"
827
- # cannot describe two different index states.
828
- candidates_total = int(
829
- conn.execute(
830
- "SELECT COUNT(*) AS c FROM vector_embeddings "
831
- "WHERE embedding_model=? AND embedding_dim=?",
832
- (self._embedding_model.model_id, self._embedding_model.dim),
833
- ).fetchone()["c"]
834
- )
835
- rows = conn.execute(
836
- f"""
837
- SELECT
838
- ve.item_id, ve.item_type, ve.source_node, ve.embedding,
839
- ve.embedding_dim, ve.embedding_model, ve.metadata_json AS vector_metadata,
840
- n.type AS node_type, n.title AS node_title, n.summary AS node_summary,
841
- n.metadata_json AS node_metadata, n.updated_at AS node_updated_at,
842
- c.text AS chunk_text, c.source_node AS parent_node_id,
843
- c.metadata_json AS chunk_metadata,
844
- pn.type AS parent_type, pn.title AS parent_title,
845
- pn.summary AS parent_summary, pn.metadata_json AS parent_metadata,
846
- pn.updated_at AS parent_updated_at
847
- FROM vector_embeddings ve
848
- LEFT JOIN nodes n ON n.id=ve.source_node
849
- LEFT JOIN chunks c ON c.id=ve.item_id
850
- LEFT JOIN nodes pn ON pn.id=c.source_node
851
- WHERE ve.embedding_model=? AND ve.embedding_dim=?
852
- ORDER BY ve.indexed_at DESC
853
- {"LIMIT ?" if cap is not None else ""}
854
- """,
855
- (
856
- (
857
- self._embedding_model.model_id,
858
- self._embedding_model.dim,
859
- cap,
860
- )
861
- if cap is not None
862
- else (self._embedding_model.model_id, self._embedding_model.dim)
863
- ),
864
- ).fetchall()
865
- recall = self._recall_report(
1264
+ if selection.name == HNSW_BACKEND:
1265
+ try:
1266
+ approximate = self._vector_search_ann(
1267
+ query,
1268
+ query_vector,
1269
+ selection,
1270
+ limit=limit,
1271
+ min_score=min_score,
1272
+ backend=backend,
1273
+ )
1274
+ except Exception as exc: # noqa: BLE001 — a broken ANN must not lose the answer
1275
+ logging.warning("hnsw vector search failed: %s", exc)
1276
+ selection = dataclasses.replace(
1277
+ resolve_vector_index(DEFAULT_VECTOR_INDEX),
1278
+ requested=HNSW_BACKEND,
1279
+ detail=f"hnsw search failed ({exc}); used the exact scan instead",
1280
+ )
1281
+ backend = selection.backend
1282
+ else:
1283
+ if approximate is not None:
1284
+ return approximate
1285
+ return self._vector_search_scan(
1286
+ query,
1287
+ query_vector,
1288
+ selection,
1289
+ limit=limit,
1290
+ min_score=min_score,
866
1291
  backend=backend,
867
1292
  cap=cap,
868
- candidates_total=candidates_total,
869
- candidates_scanned=len(rows),
870
- )
871
- scored = []
872
- for row in rows:
873
- vector = self._embedding_model.decode(
874
- row["embedding"], row["embedding_dim"]
875
- )
876
- score = self._embedding_model.similarity(query_vector, vector)
877
- if score < min_score:
878
- continue
879
- is_chunk = row["item_type"] == "chunk"
880
- summary = (
881
- row["chunk_text"]
882
- if is_chunk and row["chunk_text"]
883
- else row["node_summary"]
884
- )
885
- parent_metadata = _safe_loads(row["parent_metadata"])
886
- node_metadata = _safe_loads(row["node_metadata"])
887
- # Citation precision (review 2026-07-27 P1 #4): a chunk hit used to
888
- # cite only its parent document, so a 200-page PDF answered with
889
- # "from report.pdf". The chunk's own provenance (section heading,
890
- # page, offset) now rides along, and `locator` is the one-line
891
- # human form — absent when the chunk carries no such metadata.
892
- chunk_metadata = _safe_loads(row["chunk_metadata"]) if is_chunk else {}
893
- locator = citation_locator(chunk_metadata)
894
- scored.append(
895
- {
896
- "id": row["item_id"],
897
- "node_id": row["parent_node_id"]
898
- if is_chunk and row["parent_node_id"]
899
- else row["source_node"],
900
- "item_type": row["item_type"],
901
- "type": "Chunk" if is_chunk else row["node_type"],
902
- "title": row["parent_title"]
903
- if is_chunk and row["parent_title"]
904
- else row["node_title"],
905
- "summary": _clean_text(summary or "")[:1000],
906
- "score": round(float(score), 6),
907
- "metadata": {
908
- **(parent_metadata if is_chunk else node_metadata),
909
- "vector": _safe_loads(row["vector_metadata"]),
910
- "parent_node_id": row["parent_node_id"],
911
- "parent_type": row["parent_type"],
912
- **({"chunk": chunk_metadata} if chunk_metadata else {}),
913
- **({"locator": locator} if locator else {}),
914
- },
915
- "updated_at": row["parent_updated_at"]
916
- if is_chunk and row["parent_updated_at"]
917
- else row["node_updated_at"],
918
- }
919
- )
920
- scored.sort(
921
- key=lambda item: (item["score"], item.get("updated_at") or ""), reverse=True
922
1293
  )
923
- return {
924
- "query": query,
925
- "embedding_model": self._embedding_model.model_id,
926
- "embedding_dim": self._embedding_model.dim,
927
- "matches": scored[:limit],
928
- "recall": recall,
929
- }