ltcai 10.6.0 → 10.6.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (93) hide show
  1. package/README.md +61 -52
  2. package/docs/CHANGELOG.md +126 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/ONBOARDING.md +1 -1
  6. package/docs/OPERATIONS.md +1 -1
  7. package/docs/TRUST_MODEL.md +1 -1
  8. package/docs/WHY_LATTICE.md +1 -1
  9. package/docs/kg-schema.md +1 -1
  10. package/lattice_brain/__init__.py +1 -1
  11. package/lattice_brain/graph/_kg_contract.py +15 -0
  12. package/lattice_brain/graph/provenance.py +61 -5
  13. package/lattice_brain/graph/retrieval.py +16 -4
  14. package/lattice_brain/graph/retrieval_reads.py +136 -23
  15. package/lattice_brain/graph/retrieval_vector.py +156 -8
  16. package/lattice_brain/ingestion.py +14 -1
  17. package/lattice_brain/ingestion_jobs.py +255 -5
  18. package/lattice_brain/portability.py +5 -2
  19. package/lattice_brain/runtime/multi_agent.py +1 -1
  20. package/latticeai/__init__.py +1 -1
  21. package/latticeai/api/browser.py +8 -14
  22. package/latticeai/api/knowledge_graph.py +46 -16
  23. package/latticeai/api/workspace.py +4 -11
  24. package/latticeai/api/workspace_scope.py +125 -0
  25. package/latticeai/core/agent.py +68 -3
  26. package/latticeai/core/config.py +6 -0
  27. package/latticeai/core/csrf.py +293 -0
  28. package/latticeai/core/legacy_compatibility.py +1 -1
  29. package/latticeai/core/marketplace.py +1 -1
  30. package/latticeai/core/workspace_os_constants.py +1 -1
  31. package/latticeai/models/router.py +61 -16
  32. package/latticeai/runtime/build_phases.py +2 -0
  33. package/latticeai/runtime/config_runtime.py +2 -0
  34. package/latticeai/runtime/runtime_context.py +1 -0
  35. package/latticeai/runtime/web_runtime.py +31 -4
  36. package/latticeai/services/architecture_readiness.py +1 -1
  37. package/latticeai/services/product_readiness.py +1 -1
  38. package/latticeai/services/search_service.py +9 -1
  39. package/latticeai/services/tool_dispatch.py +13 -7
  40. package/latticeai/tools/__init__.py +1 -0
  41. package/latticeai/tools/documents.py +44 -13
  42. package/package.json +1 -1
  43. package/scripts/build_frontend_assets.mjs +21 -4
  44. package/scripts/check_current_release_docs.mjs +1 -1
  45. package/scripts/check_frontend_build_freshness.mjs +170 -0
  46. package/src-tauri/Cargo.lock +1 -1
  47. package/src-tauri/Cargo.toml +1 -1
  48. package/src-tauri/tauri.conf.json +1 -1
  49. package/static/app/asset-manifest.json +39 -39
  50. package/static/app/assets/{Act-B3MSgNsJ.js → Act-aNud-lKL.js} +1 -1
  51. package/static/app/assets/{AdminConsole-Ds8u36lf.js → AdminConsole-QiCTH68K.js} +1 -1
  52. package/static/app/assets/{Brain-XdHCIB6a.js → Brain-CMFh5q6k.js} +1 -1
  53. package/static/app/assets/{BrainHome-Dp8gzQoF.js → BrainHome-B4Ar3atu.js} +2 -2
  54. package/static/app/assets/{BrainSignals-D4yZflVt.js → BrainSignals-RVMlmTGC.js} +1 -1
  55. package/static/app/assets/{Capture-BwSZmiZ8.js → Capture-DEBo-0vZ.js} +1 -1
  56. package/static/app/assets/{CommandPalette-Ds0DnRSC.js → CommandPalette-bW5WxWWZ.js} +1 -1
  57. package/static/app/assets/{Library-SqzHjyfx.js → Library-5gFexm83.js} +1 -1
  58. package/static/app/assets/{LivingBrain-DpKt-NKE.js → LivingBrain-FlHFfu9i.js} +1 -1
  59. package/static/app/assets/ProductFlow-CFUNDOHu.js +1 -0
  60. package/static/app/assets/{ReviewCard-DVPi1LPZ.js → ReviewCard-BHj86h2Z.js} +2 -2
  61. package/static/app/assets/{System-w-9miIG8.js → System-CeNHoZRu.js} +1 -1
  62. package/static/app/assets/{activity-C0QavXjd.js → activity-7_ZmZqN0.js} +1 -1
  63. package/static/app/assets/arrow-left-Dv2Tiwhe.js +1 -0
  64. package/static/app/assets/{bot-C1-MbzCF.js → bot-D1gX4xks.js} +1 -1
  65. package/static/app/assets/{brain-CeRqzJVc.js → brain-Dn_bDfl4.js} +1 -1
  66. package/static/app/assets/{button-Z-2N8PUb.js → button-DPFZ9lGw.js} +1 -1
  67. package/static/app/assets/{circle-pause-CgjCLdmO.js → circle-pause-B2P72YOO.js} +1 -1
  68. package/static/app/assets/{circle-play-Bj45uClm.js → circle-play-BiC8z2vg.js} +1 -1
  69. package/static/app/assets/{cpu-B-2Chwfy.js → cpu-Cx-wRR_V.js} +1 -1
  70. package/static/app/assets/{download-CKmiOG3C.js → download-CKxlzxsK.js} +1 -1
  71. package/static/app/assets/{folder-open-fE7lLvhZ.js → folder-open-Bbjju0tt.js} +1 -1
  72. package/static/app/assets/{hard-drive-rDl1xsJ1.js → hard-drive-Bon1VBvq.js} +1 -1
  73. package/static/app/assets/index-DYUs0cWy.css +2 -0
  74. package/static/app/assets/{index-AEIqmjwZ.js → index-mLP0-YNO.js} +3 -3
  75. package/static/app/assets/{input-BflJYT5-.js → input-DrMc0Xns.js} +1 -1
  76. package/static/app/assets/{permissionCopy-BNNvkXSX.js → permissionCopy-BGUsI7vw.js} +1 -1
  77. package/static/app/assets/{primitives-CP68OWk2.js → primitives-CEMTjBz1.js} +1 -1
  78. package/static/app/assets/search-HsIji1wY.js +1 -0
  79. package/static/app/assets/{share-2-CyjG2_yY.js → share-2-D7THHq5K.js} +1 -1
  80. package/static/app/assets/{shield-alert-C_JBPWYd.js → shield-alert-BoU4_8r9.js} +1 -1
  81. package/static/app/assets/{textarea-Ds1Exelb.js → textarea-rU1Lb7ia.js} +1 -1
  82. package/static/app/assets/{useFocusTrap-BgIZQ4if.js → useFocusTrap-BgvK4Nkx.js} +1 -1
  83. package/static/app/assets/{useQuery-6Bu27NQe.js → useQuery-CT2ChyuU.js} +1 -1
  84. package/static/app/assets/{users-31lxlciQ.js → users-DlbfHBQV.js} +1 -1
  85. package/static/app/assets/{utils-C0-C5mZc.js → utils-fEGWreKB.js} +1 -1
  86. package/static/app/assets/workspace-ClDBz_0f.js +1 -0
  87. package/static/app/index.html +4 -4
  88. package/static/sw.js +1 -1
  89. package/static/app/assets/ProductFlow-BimeVtWH.js +0 -1
  90. package/static/app/assets/arrow-left-TsAKz-s_.js +0 -1
  91. package/static/app/assets/index-DcMODGjM.css +0 -2
  92. package/static/app/assets/search-B4O4iIgg.js +0 -1
  93. package/static/app/assets/workspace-DAB-urHL.js +0 -1
@@ -140,6 +140,33 @@ class KnowledgeGraphReadsMixin(_Core):
140
140
  visible.append(item)
141
141
  return visible
142
142
 
143
+ @staticmethod
144
+ def _workspace_scope_sql(
145
+ allowed_workspaces,
146
+ include_legacy_global: bool,
147
+ ) -> Tuple[Optional[str], List[Any]]:
148
+ """``nodes_v2`` predicate for a caller's scope, or ``(None, [])``.
149
+
150
+ ``None`` means "no scoping" and is the unscoped single-user path.
151
+ An *empty* allowed set is not the same thing: it is a caller who may
152
+ read nothing, so it yields a predicate that matches nothing rather
153
+ than silently degrading into the unscoped query.
154
+ """
155
+ if allowed_workspaces is None:
156
+ return None, []
157
+ allowed = sorted({str(item) for item in allowed_workspaces if item})
158
+ clauses: List[str] = []
159
+ params: List[Any] = []
160
+ if allowed:
161
+ placeholders = ",".join("?" for _ in allowed)
162
+ clauses.append(f"workspace_id IN ({placeholders})")
163
+ params.extend(allowed)
164
+ if include_legacy_global:
165
+ clauses.append("workspace_id IS NULL")
166
+ if not clauses:
167
+ return "0", []
168
+ return " OR ".join(clauses), params
169
+
143
170
  def neighbors(
144
171
  self,
145
172
  node_id: str,
@@ -427,34 +454,82 @@ class KnowledgeGraphReadsMixin(_Core):
427
454
  edges = [edge for edge in edges if edge.get("from") in kept and edge.get("to") in kept]
428
455
  return {"root": node_id, "depth": depth, "nodes": nodes, "edges": edges}
429
456
 
430
- def stats(self) -> Dict[str, Any]:
457
+ def stats(
458
+ self,
459
+ *,
460
+ allowed_workspaces=None,
461
+ include_legacy_global: bool = False,
462
+ ) -> Dict[str, Any]:
463
+ """Store statistics, optionally restricted to a caller's workspaces.
464
+
465
+ ``allowed_workspaces=None`` keeps the historical whole-store counts
466
+ (single-user / no-auth mode). When a scope is given, the node and edge
467
+ histograms are counted through the authoritative ``nodes_v2``
468
+ projection, so a member of one organization workspace cannot read
469
+ another's volume off this endpoint. An edge counts only when *both*
470
+ endpoints are visible — the same rule ``graph()`` and ``neighbors()``
471
+ already apply to the rows they return.
472
+ """
431
473
  nt, et = self._read_tables()
474
+ scope_sql, scope_params = self._workspace_scope_sql(
475
+ allowed_workspaces, include_legacy_global
476
+ )
432
477
  with self._connect() as conn:
433
- node_counts = {
434
- row["type"]: row["count"]
435
- for row in conn.execute(
436
- f"SELECT type, COUNT(*) AS count FROM {nt} GROUP BY type"
437
- )
438
- }
439
- edge_counts = {
440
- row["type"]: row["count"]
441
- for row in conn.execute(
442
- f"SELECT type, COUNT(*) AS count FROM {et} GROUP BY type"
443
- )
444
- }
445
- local_sources = conn.execute(
446
- "SELECT COUNT(*) AS c FROM knowledge_sources"
447
- ).fetchone()["c"]
448
- local_file_status = {
449
- row["status"]: row["count"]
450
- for row in conn.execute(
451
- "SELECT status, COUNT(*) AS count FROM local_file_index GROUP BY status"
452
- )
453
- }
478
+ if scope_sql is None:
479
+ node_counts = {
480
+ row["type"]: row["count"]
481
+ for row in conn.execute(
482
+ f"SELECT type, COUNT(*) AS count FROM {nt} GROUP BY type"
483
+ )
484
+ }
485
+ edge_counts = {
486
+ row["type"]: row["count"]
487
+ for row in conn.execute(
488
+ f"SELECT type, COUNT(*) AS count FROM {et} GROUP BY type"
489
+ )
490
+ }
491
+ local_sources = conn.execute(
492
+ "SELECT COUNT(*) AS c FROM knowledge_sources"
493
+ ).fetchone()["c"]
494
+ local_file_status = {
495
+ row["status"]: row["count"]
496
+ for row in conn.execute(
497
+ "SELECT status, COUNT(*) AS count FROM local_file_index GROUP BY status"
498
+ )
499
+ }
500
+ else:
501
+ visible = f"SELECT id FROM nodes_v2 WHERE {scope_sql}"
502
+ node_counts = {
503
+ row["type"]: row["count"]
504
+ for row in conn.execute(
505
+ f"SELECT type, COUNT(*) AS count FROM {nt} "
506
+ f"WHERE id IN ({visible}) GROUP BY type",
507
+ scope_params,
508
+ )
509
+ }
510
+ edge_counts = {
511
+ row["type"]: row["count"]
512
+ for row in conn.execute(
513
+ f"SELECT type, COUNT(*) AS count FROM {et} "
514
+ f"WHERE from_node IN ({visible}) AND to_node IN ({visible}) "
515
+ f"GROUP BY type",
516
+ scope_params + scope_params,
517
+ )
518
+ }
519
+ # Local sources and the file index are machine-local ingestion
520
+ # bookkeeping with no workspace column. They are not another
521
+ # tenant's content, but they are also not this caller's scope,
522
+ # so a scoped read reports none rather than guessing.
523
+ local_sources = 0
524
+ local_file_status = {}
454
525
  v2 = None
455
526
  if KGStoreV2 is not None:
456
527
  try:
457
- v2 = KGStoreV2(self.db_path).stats()
528
+ v2 = (
529
+ KGStoreV2(self.db_path).stats()
530
+ if scope_sql is None
531
+ else self._scoped_v2_stats(scope_sql, scope_params)
532
+ )
458
533
  except Exception as e:
459
534
  v2 = {"available": False, "error": str(e)}
460
535
  return {
@@ -467,3 +542,41 @@ class KnowledgeGraphReadsMixin(_Core):
467
542
  "local_file_status": local_file_status,
468
543
  "v2": v2,
469
544
  }
545
+
546
+ def _scoped_v2_stats(self, scope_sql: str, scope_params: List[Any]) -> Dict[str, Any]:
547
+ """``KGStoreV2.stats()`` restricted to a caller's workspaces.
548
+
549
+ Starts from the real payload and overwrites only the counts, so the
550
+ key set stays whatever ``KGStoreV2`` defines — ``/knowledge-graph/schema``
551
+ returns this sub-object verbatim, making its shape part of the API
552
+ contract rather than something to re-derive here.
553
+ """
554
+ payload: Dict[str, Any] = dict(KGStoreV2(self.db_path).stats())
555
+ visible = f"SELECT id FROM nodes_v2 WHERE {scope_sql}"
556
+ with self._connect() as conn:
557
+ by_node_type = {
558
+ row["type"]: row["c"]
559
+ for row in conn.execute(
560
+ f"SELECT type, COUNT(*) AS c FROM nodes_v2 "
561
+ f"WHERE {scope_sql} GROUP BY type",
562
+ scope_params,
563
+ ).fetchall()
564
+ }
565
+ by_edge_type = {
566
+ row["type"]: row["c"]
567
+ for row in conn.execute(
568
+ f"SELECT type, COUNT(*) AS c FROM edges_v2 "
569
+ f"WHERE source IN ({visible}) AND target IN ({visible}) "
570
+ f"GROUP BY type",
571
+ scope_params + scope_params,
572
+ ).fetchall()
573
+ }
574
+ payload.update(
575
+ {
576
+ "nodes": sum(by_node_type.values()),
577
+ "edges": sum(by_edge_type.values()),
578
+ "by_node_type": by_node_type,
579
+ "by_edge_type": by_edge_type,
580
+ }
581
+ )
582
+ return payload
@@ -14,6 +14,47 @@ else:
14
14
  _Core = object
15
15
 
16
16
 
17
+ # ── brute-force recall cap (review 2026-08 P1 #2) ────────────────────────────
18
+ # There is no ANN index in the default build: sqlite-vec is an optional
19
+ # dependency and, when it is absent, `index_status()["storage"]` honestly
20
+ # reports ``vector_search_backend: "bruteforce-cosine"``. Brute force scores
21
+ # every candidate in Python, so *some* cap is unavoidable on a large graph.
22
+ #
23
+ # What is not acceptable is a SILENT cap. The pre-10.7 code took the 10 000
24
+ # most recently indexed rows — ordered by ``indexed_at``, i.e. by recency, not
25
+ # by similarity — and returned them as if they were the whole index, so recall
26
+ # on a 200 000-row brain quietly became "the newest 5%". The cap is now
27
+ # explicit, configurable, and reported back to the caller in ``recall``.
28
+ #
29
+ # ``LATTICEAI_VECTOR_MAX_CANDIDATES`` overrides the default; ``0`` means "no
30
+ # cap — scan the whole index" (exact recall, paid for in latency).
31
+ VECTOR_MAX_CANDIDATES_ENV = "LATTICEAI_VECTOR_MAX_CANDIDATES"
32
+ DEFAULT_VECTOR_MAX_CANDIDATES = 10_000
33
+ #: Upper bound for a configured cap; ``0``/``None`` still means uncapped.
34
+ VECTOR_MAX_CANDIDATES_CEILING = 500_000
35
+
36
+
37
+ def _configured_vector_max_candidates() -> Optional[int]:
38
+ """Resolve the candidate cap from the environment (None = uncapped).
39
+
40
+ Never raises: an unparseable value falls back to the documented default
41
+ rather than breaking every search.
42
+ """
43
+ raw = os.getenv(VECTOR_MAX_CANDIDATES_ENV)
44
+ if raw is None or not raw.strip():
45
+ return DEFAULT_VECTOR_MAX_CANDIDATES
46
+ try:
47
+ value = int(raw.strip())
48
+ except ValueError:
49
+ logging.warning(
50
+ "%s=%r is not an integer — using the default cap of %d",
51
+ VECTOR_MAX_CANDIDATES_ENV, raw, DEFAULT_VECTOR_MAX_CANDIDATES,
52
+ )
53
+ return DEFAULT_VECTOR_MAX_CANDIDATES
54
+ if value <= 0:
55
+ return None # explicit opt-in to an exhaustive scan
56
+ return min(value, VECTOR_MAX_CANDIDATES_CEILING)
57
+
17
58
 
18
59
  class KnowledgeGraphVectorMixin(_Core):
19
60
  """Vector-embedding index build/status/search, split out of retrieval.
@@ -643,24 +684,120 @@ class KnowledgeGraphVectorMixin(_Core):
643
684
  "detail": detail,
644
685
  }
645
686
 
687
+ def _vector_search_backend(self) -> str:
688
+ """Which backend actually scores the vectors, per storage capabilities.
689
+
690
+ sqlite-vec exposes an ANN index; without it this store scores rows in
691
+ Python (``bruteforce-cosine``). Never raises — a capability probe
692
+ failure means "we cannot claim ANN", which is the brute-force answer.
693
+ """
694
+ try:
695
+ capabilities = self.storage_engine.capabilities().as_dict()
696
+ except Exception: # noqa: BLE001 — a probe failure is not an ANN index
697
+ return "bruteforce-cosine"
698
+ backend = (capabilities or {}).get("vector_backend")
699
+ return str(backend) if backend else "bruteforce-cosine"
700
+
701
+ @staticmethod
702
+ def _recall_report(
703
+ *,
704
+ backend: str,
705
+ cap: Optional[int],
706
+ candidates_total: int,
707
+ candidates_scanned: int,
708
+ ) -> Dict[str, Any]:
709
+ """The honest answer to "did this search see the whole index?"."""
710
+ truncated = candidates_scanned < candidates_total
711
+ detail: Optional[str] = None
712
+ if truncated:
713
+ detail = (
714
+ f"partial recall: scored the {candidates_scanned} most recently "
715
+ f"indexed vectors of {candidates_total}. The cut is by index "
716
+ f"recency, not similarity, so older matches were never compared. "
717
+ f"Raise {VECTOR_MAX_CANDIDATES_ENV} (0 = scan everything) or "
718
+ f"install sqlite-vec for an ANN index."
719
+ )
720
+ return {
721
+ "backend": backend,
722
+ "max_candidates": cap,
723
+ "candidates_total": candidates_total,
724
+ "candidates_scanned": candidates_scanned,
725
+ "truncated": truncated,
726
+ "detail": detail,
727
+ }
728
+
729
+ def _vector_candidate_cap(
730
+ self, requested: Optional[int], *, limit: int
731
+ ) -> Optional[int]:
732
+ """Resolve the effective candidate cap (None = scan everything).
733
+
734
+ ``requested is None`` uses the configured/default cap; an explicit
735
+ ``<= 0`` is the caller asking for an exhaustive scan. Note the
736
+ ``is None`` test: ``0`` is a meaningful value here, so truthiness
737
+ would silently turn "no cap" into "the default cap".
738
+ """
739
+ if requested is None:
740
+ cap = _configured_vector_max_candidates()
741
+ elif int(requested) <= 0:
742
+ cap = None
743
+ else:
744
+ cap = min(int(requested), VECTOR_MAX_CANDIDATES_CEILING)
745
+ if cap is None:
746
+ return None
747
+ # Never scan fewer rows than the caller intends to receive.
748
+ return max(limit, cap)
749
+
646
750
  def vector_search(
647
751
  self,
648
752
  query: str,
649
753
  *,
650
754
  limit: int = 30,
651
755
  min_score: float = 0.0,
652
- max_candidates: int = 10_000,
756
+ max_candidates: Optional[int] = None,
653
757
  ) -> Dict[str, Any]:
758
+ """Brute-force cosine search over the vector index.
759
+
760
+ ``max_candidates`` bounds how many indexed rows are scored; ``None``
761
+ (the default) resolves it from ``LATTICEAI_VECTOR_MAX_CANDIDATES``
762
+ (default 10 000), and ``0`` or a negative value scans the whole index.
763
+ When the cap bites, the rows kept are the most recently indexed ones —
764
+ recency, not similarity — so the result is *partial recall*. That is
765
+ reported in the additive ``recall`` block
766
+ (``{backend, max_candidates, candidates_total, candidates_scanned,
767
+ truncated, detail}``) instead of being hidden, and callers/UIs are
768
+ expected to surface ``recall.truncated``.
769
+ """
654
770
  query = str(query or "").strip()
655
771
  limit = max(1, min(int(limit or 30), 100))
656
772
  min_score = float(min_score or 0.0)
773
+ cap = self._vector_candidate_cap(max_candidates, limit=limit)
774
+ backend = self._vector_search_backend()
657
775
  if not query:
658
- return {"query": query, "matches": []}
776
+ return {
777
+ "query": query,
778
+ "matches": [],
779
+ "recall": {
780
+ "backend": backend,
781
+ "max_candidates": cap,
782
+ "candidates_total": 0,
783
+ "candidates_scanned": 0,
784
+ "truncated": False,
785
+ "detail": None,
786
+ },
787
+ }
659
788
  query_vector = self._embedding_model.embed(query)
660
- max_candidates = max(limit, min(int(max_candidates or 10_000), 50_000))
661
789
  with self._connect() as conn:
790
+ # Counted in the same transaction as the scan so "scanned N of M"
791
+ # cannot describe two different index states.
792
+ candidates_total = int(
793
+ conn.execute(
794
+ "SELECT COUNT(*) AS c FROM vector_embeddings "
795
+ "WHERE embedding_model=? AND embedding_dim=?",
796
+ (self._embedding_model.model_id, self._embedding_model.dim),
797
+ ).fetchone()["c"]
798
+ )
662
799
  rows = conn.execute(
663
- """
800
+ f"""
664
801
  SELECT
665
802
  ve.item_id, ve.item_type, ve.source_node, ve.embedding,
666
803
  ve.embedding_dim, ve.embedding_model, ve.metadata_json AS vector_metadata,
@@ -677,14 +814,24 @@ class KnowledgeGraphVectorMixin(_Core):
677
814
  LEFT JOIN nodes pn ON pn.id=c.source_node
678
815
  WHERE ve.embedding_model=? AND ve.embedding_dim=?
679
816
  ORDER BY ve.indexed_at DESC
680
- LIMIT ?
817
+ {"LIMIT ?" if cap is not None else ""}
681
818
  """,
682
819
  (
683
- self._embedding_model.model_id,
684
- self._embedding_model.dim,
685
- max_candidates,
820
+ (
821
+ self._embedding_model.model_id,
822
+ self._embedding_model.dim,
823
+ cap,
824
+ )
825
+ if cap is not None
826
+ else (self._embedding_model.model_id, self._embedding_model.dim)
686
827
  ),
687
828
  ).fetchall()
829
+ recall = self._recall_report(
830
+ backend=backend,
831
+ cap=cap,
832
+ candidates_total=candidates_total,
833
+ candidates_scanned=len(rows),
834
+ )
688
835
  scored = []
689
836
  for row in rows:
690
837
  vector = self._embedding_model.decode(
@@ -742,4 +889,5 @@ class KnowledgeGraphVectorMixin(_Core):
742
889
  "embedding_model": self._embedding_model.model_id,
743
890
  "embedding_dim": self._embedding_model.dim,
744
891
  "matches": scored[:limit],
892
+ "recall": recall,
745
893
  }
@@ -454,7 +454,14 @@ class IngestionPipeline:
454
454
  self._audit = audit
455
455
  self._max_text_bytes = int(max_text_bytes)
456
456
  self._pipeline_name = pipeline_name
457
- self._bg_queue = bg_queue or BackgroundIngestionQueue()
457
+ # Background job state lives in the graph database by default, so a
458
+ # restart resumes from the last completed item instead of replaying the
459
+ # whole corpus. A store without a usable ``db_path`` (mocks, disabled
460
+ # graph) degrades to the historical in-memory queue, which reports
461
+ # itself as non-durable through ``BackgroundIngestionQueue.describe()``.
462
+ self._bg_queue = bg_queue or BackgroundIngestionQueue(
463
+ db_path=getattr(knowledge_graph, "db_path", None)
464
+ )
458
465
  # Incremental vector sync after each successful non-duplicate ingest.
459
466
  # Constructor opt-out AND env opt-out (LATTICEAI_AUTO_VECTOR_INDEX=0)
460
467
  # both disable it; a vector failure never fails the ingest.
@@ -786,6 +793,7 @@ class IngestionPipeline:
786
793
  job.failed = 0
787
794
  job.errors = []
788
795
  job.touch()
796
+ self._bg_queue.save(job)
789
797
  runner_email = user_email or job.user_email
790
798
  for index in job.remaining_indices():
791
799
  item = job.items[index]
@@ -800,6 +808,10 @@ class IngestionPipeline:
800
808
  job.record_error(index, item, detail or status)
801
809
  job.processed = len(job.done_indices)
802
810
  job.touch()
811
+ # Checkpoint per item: a crash here must cost at most the item in
812
+ # flight, never the whole job's progress. One small UPDATE against
813
+ # an ingest (parse + chunk + embed) is noise.
814
+ self._bg_queue.save(job)
803
815
  job.processed = len(job.done_indices)
804
816
  if job.total == 0 or job.processed >= job.total:
805
817
  job.status = "completed"
@@ -808,6 +820,7 @@ class IngestionPipeline:
808
820
  else:
809
821
  job.status = "failed"
810
822
  job.touch()
823
+ self._bg_queue.save(job)
811
824
  return job.as_dict()
812
825
 
813
826
  def ingest_web_page(