ltcai 11.9.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/README.md +80 -59
  2. package/docs/CHANGELOG.md +92 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +267 -119
  5. package/docs/ENTERPRISE.md +1 -1
  6. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  7. package/docs/ONBOARDING.md +13 -3
  8. package/docs/OPERATIONS.md +13 -4
  9. package/docs/REALTIME_COLLABORATION.md +1 -1
  10. package/docs/ROADMAP.md +113 -0
  11. package/docs/TRUST_MODEL.md +16 -2
  12. package/docs/WHY_LATTICE.md +8 -2
  13. package/docs/WORKFLOW_DESIGNER.md +2 -2
  14. package/docs/kg-schema.md +51 -5
  15. package/docs/mcp-tools.md +17 -0
  16. package/lattice_brain/__init__.py +1 -1
  17. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  18. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  19. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  20. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  21. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  22. package/lattice_brain/graph/_kg_constants.py +7 -0
  23. package/latticeai/__init__.py +1 -1
  24. package/latticeai/api/agent_worker_seam.py +32 -0
  25. package/latticeai/api/worker_compute.py +78 -3
  26. package/latticeai/core/embedding_providers/__init__.py +16 -0
  27. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  28. package/latticeai/core/embedding_providers/base.py +25 -0
  29. package/latticeai/core/embedding_providers/profiles.py +44 -0
  30. package/latticeai/core/embedding_providers/text.py +74 -8
  31. package/latticeai/core/vector_index/__init__.py +61 -0
  32. package/latticeai/core/vector_index/hnsw.py +383 -0
  33. package/latticeai/core/vector_index/sidecar.py +329 -0
  34. package/latticeai/models/router/generation.py +101 -22
  35. package/latticeai/models/router/loading.py +109 -4
  36. package/latticeai/runtime/brain_runtime.py +43 -9
  37. package/latticeai/runtime/build_phases/worker_profile.py +13 -4
  38. package/latticeai/services/architecture_readiness.py +2 -2
  39. package/latticeai/services/product_readiness.py +10 -5
  40. package/latticeai/services/search_service.py +7 -0
  41. package/latticeai/tools/__init__.py +6 -1
  42. package/latticeai/tools/documents.py +12 -0
  43. package/latticeai/tools/markup.py +152 -0
  44. package/package.json +2 -1
  45. package/scripts/check_current_release_docs.mjs +1 -1
  46. package/scripts/compose_openapi.py +2 -0
  47. package/scripts/openapi_route_families.json +7 -3
  48. package/scripts/publish_release.mjs +157 -0
  49. package/scripts/release_screen_claims.json +12 -0
  50. package/src-tauri/Cargo.lock +44 -10
  51. package/src-tauri/Cargo.toml +1 -1
  52. package/src-tauri/tauri.conf.json +1 -1
  53. package/static/app/asset-manifest.json +47 -41
  54. package/static/app/assets/Act-Cf1L2709.js +2 -0
  55. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  56. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  57. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  58. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  59. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  60. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  61. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  62. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  63. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  64. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  65. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  66. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  67. package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
  68. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  69. package/static/app/assets/System-CAxwBUXw.js +1 -0
  70. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  71. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  72. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  73. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  74. package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
  75. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  76. package/static/app/assets/button-CmaEqG1T.js +1 -0
  77. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  78. package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
  79. package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
  80. package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
  81. package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
  82. package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
  83. package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
  84. package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
  85. package/static/app/assets/index-D2H-wSl6.js +13 -0
  86. package/static/app/assets/input-Df1CAY_I.js +1 -0
  87. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  88. package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
  89. package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
  90. package/static/app/assets/primitives-BioD2slS.js +1 -0
  91. package/static/app/assets/search-BzBw8YcW.js +1 -0
  92. package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
  93. package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
  94. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  95. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  96. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  97. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  98. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  99. package/static/app/index.html +4 -4
  100. package/static/sw.js +1 -1
  101. package/static/app/assets/Act-B4WT81kh.js +0 -1
  102. package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
  103. package/static/app/assets/Brain-CtYaa26c.js +0 -321
  104. package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
  105. package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
  106. package/static/app/assets/Capture-DdNi5Peb.js +0 -1
  107. package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
  108. package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
  109. package/static/app/assets/Library-Bl5XClFV.js +0 -1
  110. package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
  111. package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
  112. package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
  113. package/static/app/assets/System-BJ6jQ_SL.js +0 -1
  114. package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
  115. package/static/app/assets/brain-B9BDrMTe.js +0 -1
  116. package/static/app/assets/button-D6JcpYcf.js +0 -1
  117. package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
  118. package/static/app/assets/index-CWKRRsLW.js +0 -10
  119. package/static/app/assets/input-CEqsxtil.js +0 -1
  120. package/static/app/assets/primitives-DORg7Z_7.js +0 -1
  121. package/static/app/assets/search-0NQ21wXe.js +0 -1
  122. package/static/app/assets/textarea-jtQcRSXo.js +0 -1
  123. package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
  124. package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
  125. package/static/app/assets/useQuery-BizqBNGw.js +0 -1
  126. package/static/app/assets/utils-WgW4V69R.js +0 -4
  127. package/static/app/assets/workspace-DSek3jCY.js +0 -1
@@ -0,0 +1,107 @@
1
+ """Heading paths for a document, so a fact can say *where* it came from.
2
+
3
+ The typed chunker already computes a `" > "`-joined heading path per chunk and
4
+ files it on the Chunk node (`heading_path`). Extraction ran on the whole
5
+ document text and had no idea which section a sentence sat in, so an edge could
6
+ say "이 문장이 근거다" but never "그 문장은 「아키텍처 > 저장소」 절에 있다".
7
+
8
+ This module closes that gap with the *same* rule the chunker uses — a line
9
+ matching `^#{1,6} ` opens a section — so the heading a triple names and the
10
+ heading its chunk carries are the same string.
11
+
12
+ Character offsets throughout, because Python slices `str` by code point and the
13
+ rest of the pipeline (chunk `start_char`, the Rust port) does too.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+ from typing import List, Optional, Sequence, Tuple
20
+
21
+ #: `^(#{1,6}) (.*)$` under `re.MULTILINE` — the chunker's heading rule.
22
+ _HEADING = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
23
+
24
+ #: `(start, end, heading_path)`; `end` is exclusive.
25
+ Span = Tuple[int, int, str]
26
+
27
+
28
+ def heading_spans(text: str) -> List[Span]:
29
+ """Every heading's span and its `" > "`-joined path, in document order.
30
+
31
+ Text before the first heading belongs to no section and is deliberately
32
+ absent from the result — an honest "no heading" beats inventing one from
33
+ the filename.
34
+
35
+ >>> heading_spans("# A\\nintro\\n## B\\nbody")
36
+ [(0, 12, 'A'), (12, 20, 'A > B')]
37
+ """
38
+ matches = list(_HEADING.finditer(text or ""))
39
+ spans: List[Span] = []
40
+ stack: List[Tuple[int, str]] = []
41
+ for index, match in enumerate(matches):
42
+ level = len(match.group(1))
43
+ title = match.group(2).strip()
44
+ while stack and stack[-1][0] >= level:
45
+ stack.pop()
46
+ stack.append((level, title))
47
+ end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
48
+ spans.append((match.start(), end, " > ".join(part for _, part in stack)))
49
+ return spans
50
+
51
+
52
+ def heading_at(spans: Sequence[Span], offset: int) -> str:
53
+ """The heading path covering ``offset``, or `""` when there is none."""
54
+ for start, end, path in spans:
55
+ if start <= offset < end:
56
+ return path
57
+ return ""
58
+
59
+
60
+ def with_section(context: str, section: str) -> str:
61
+ """``context`` prefixed with the section it came from, when there is one.
62
+
63
+ ``"[아키텍처 > 저장소] 쓰기는 GraphWriter가 담당한다."`` — one string,
64
+ because `TripleSpec.context` is the only free-text channel an extracted
65
+ edge has. Blank sections leave the context untouched rather than adding an
66
+ empty bracket.
67
+ """
68
+ section = (section or "").strip()
69
+ if not section:
70
+ return context
71
+ return f"[{section}] {context}"
72
+
73
+
74
+ def sentence_offsets(text: str, pattern: "re.Pattern[str]") -> List[Tuple[int, str]]:
75
+ """``(offset, sentence)`` for a split that keeps each piece's position.
76
+
77
+ ``re.split`` throws the offsets away, and the offset is exactly what maps a
78
+ sentence back to its heading. Walking the separators keeps both.
79
+ """
80
+ out: List[Tuple[int, str]] = []
81
+ cursor = 0
82
+ for match in pattern.finditer(text or ""):
83
+ out.append((cursor, text[cursor : match.start()]))
84
+ cursor = match.end()
85
+ out.append((cursor, (text or "")[cursor:]))
86
+ return out
87
+
88
+
89
+ def leading_offset(raw: str, stripped: str) -> Optional[int]:
90
+ """How many characters ``str.strip()`` removed from the front of ``raw``.
91
+
92
+ ``None`` when ``stripped`` is empty — there is no position to report for a
93
+ piece that stripped away entirely.
94
+ """
95
+ if not stripped:
96
+ return None
97
+ return raw.index(stripped)
98
+
99
+
100
+ __all__ = [
101
+ "Span",
102
+ "heading_at",
103
+ "heading_spans",
104
+ "leading_offset",
105
+ "sentence_offsets",
106
+ "with_section",
107
+ ]
@@ -28,6 +28,13 @@ LOCAL_CODE_EXTENSIONS = {
28
28
  ".tsx",
29
29
  ".jsx",
30
30
  ".html",
31
+ # v12.0.0: `.htm` sat beside `.html` in every other table (the chunker's
32
+ # prose list, the parser matrix) and was missing only here, so a folder of
33
+ # `.htm` pages was scanned past in silence. `.rs` was missing outright —
34
+ # this repository's own Rust half was invisible to its own folder ingest.
35
+ ".htm",
36
+ ".rs",
37
+ ".go",
31
38
  ".css",
32
39
  ".json",
33
40
  ".yaml",
@@ -1,3 +1,3 @@
1
1
  """Lattice AI - modular server package."""
2
2
 
3
- __version__ = "11.9.0"
3
+ __version__ = "12.0.0"
@@ -81,6 +81,13 @@ MAX_TEMPERATURE = 2.0
81
81
  #: How many stop strings one completion may name. The kernel sends two.
82
82
  MAX_STOP_STRINGS = 8
83
83
 
84
+ #: Longest forced completion prefix one call may name (v12.0.0). The prefix is
85
+ #: prepended to the prompt, so an unbounded one is an unbounded prompt. The
86
+ #: kernel sends two, and both are short: the executor's opening brace
87
+ #: (``{"thoughts": "``) and the guided menu's answer label — a few dozen
88
+ #: characters between them, and nothing legitimate needs a paragraph.
89
+ MAX_PREFIX_CHARS = 256
90
+
84
91
  #: Rate-limit bucket. Deliberately *not* the ``"agent"`` bucket ``/agent`` uses:
85
92
  #: that one is sized per *run* (10 burst, one refill per 10s) because one HTTP
86
93
  #: call there is a whole agent run. Here one call is a single loop step, and a
@@ -111,6 +118,19 @@ class AgentLLMRequest(BaseModel):
111
118
  #: truncate any reply carrying file content. Bounded so a caller cannot make
112
119
  #: the sampler check a thousand needles per token.
113
120
  stop: Optional[List[str]] = Field(default=None, max_length=MAX_STOP_STRINGS)
121
+ #: Characters the reply is **forced to begin with** (v12.0.0).
122
+ #:
123
+ #: The cheapest structural guarantee the seam can offer: instead of asking
124
+ #: a weak model to start at ``{`` and repairing the markdown fence it emits
125
+ #: instead, the worker prefills those characters and the model continues
126
+ #: from them. The response's ``text`` always starts with the prefix, so a
127
+ #: caller reads one shape whether the local prefill or a cloud provider's
128
+ #: assistant-prefill message produced it.
129
+ #:
130
+ #: Bounded because it is prepended to the prompt: an unbounded prefix would
131
+ #: be an unbounded prompt on the single MLX executor. The kernel sends
132
+ #: fourteen characters.
133
+ prefix: Optional[str] = Field(default=None, max_length=MAX_PREFIX_CHARS)
114
134
 
115
135
 
116
136
  class AgentToolRequest(BaseModel):
@@ -258,6 +278,17 @@ def create_agent_worker_seam_router(
258
278
  max_tokens=req.max_tokens,
259
279
  temperature=req.temperature,
260
280
  stop=req.stop or None,
281
+ prefix=req.prefix or None,
282
+ # **The seam's ``context`` is a prompt, not a corpus** (v12.0.0).
283
+ # ``_compose_system`` appends the citation mandate to any non-empty
284
+ # context, which is right for chat (where it really is retrieved
285
+ # passages) and wrong here: the Rust kernel sends its whole executor
286
+ # prompt as ``context``, so every agent turn was being told its own
287
+ # instructions were sources to "cite inline as [1], [2]". Small
288
+ # models obey that — a 0.5B wrote ``[1] …`` into a file and a 2B
289
+ # answered a tool call with the citation instruction. A caller that
290
+ # wants citations puts them in its own prompt.
291
+ cite_sources=False,
261
292
  )
262
293
  return {"text": str(text)}
263
294
 
@@ -316,6 +347,7 @@ def create_agent_worker_seam_router(
316
347
 
317
348
  __all__ = [
318
349
  "MAX_MAX_TOKENS",
350
+ "MAX_PREFIX_CHARS",
319
351
  "MAX_TEMPERATURE",
320
352
  "MIN_MAX_TOKENS",
321
353
  "MIN_TEMPERATURE",
@@ -19,7 +19,7 @@ here opens a database, writes a file, or reaches a store. Each handler is a
19
19
  function of its request body, so a caller can retry it, cache it, or run two of
20
20
  them at once without asking who else is writing.
21
21
 
22
- The eight seams, and what each was extracted from:
22
+ The nine seams, and what each was extracted from:
23
23
 
24
24
  ``POST /worker/embed``
25
25
  ``EmbeddingProvider.embed_batch`` on the *resolved* provider — the same
@@ -53,6 +53,12 @@ The eight seams, and what each was extracted from:
53
53
  consume, already classified. Rust writes the Concept/Task/Decision subgraph;
54
54
  this seam never opens a store.
55
55
 
56
+ ``POST /worker/vector/query``
57
+ Approximate nearest-neighbour over the ``.hnsw`` sidecar next to the
58
+ brain database. The one compute seam that *reads* the store (row count
59
+ + embeddings, never a write) so it can refresh the sidecar and say
60
+ ``index: "none"`` when there is nothing to serve.
61
+
56
62
  Gating is the seam gate the rest of the worker uses — ``LATTICEAI_AGENT_TOOL_SEAM``
57
63
  read per request through :func:`latticeai.api.agent_worker_seam._seam_open`,
58
64
  then ``require_user`` and the ``agent_seam`` rate bucket. Off ⇒ 404. Mounted
@@ -108,6 +114,12 @@ EXTRACT_LIMITS: Dict[str, int] = {"message": 12, "document": 15}
108
114
  #: write-side module would tie this compute seam to a module W1 is replacing.
109
115
  PASSAGE_MAX_CHARS = 50_000
110
116
 
117
+ #: Upper bound on ``texts`` for one ``POST /worker/embed``. Ingest already
118
+ #: batches; a caller that dumps a whole vault into one request is a bug, not
119
+ #: a use case. 256 sits in the 128–256 band G-RAG-A recommended. Additive:
120
+ #: the request/response shape is unchanged.
121
+ EMBED_MAX_BATCH = 256
122
+
111
123
  #: The CJK-capable fonts ``create_pdf`` looks for, as a module attribute so a
112
124
  #: test can point the probe at a font that exists (and at one that does not)
113
125
  #: instead of asserting whatever the host machine happens to have installed.
@@ -154,6 +166,14 @@ WORKER_COMPUTE_MESSAGES: Dict[str, Dict[str, str]] = {
154
166
  "ko": "'{kind}' 은(는) 추출 종류가 아닙니다. {allowed} 중 하나여야 합니다.",
155
167
  "en": "'{kind}' is not an extraction kind. Use one of {allowed}.",
156
168
  },
169
+ "worker_compute.embed_batch_too_large": {
170
+ "ko": "임베딩 배치가 {count}개입니다. 한 번에 {limit}개까지입니다.",
171
+ "en": "The embed batch has {count} texts; the limit is {limit}.",
172
+ },
173
+ "worker_compute.vector_query_invalid": {
174
+ "ko": "벡터 질의는 embedding_model, embedding_dim, vector 가 필요합니다.",
175
+ "en": "A vector query needs embedding_model, embedding_dim, and vector.",
176
+ },
157
177
  }
158
178
 
159
179
 
@@ -436,6 +456,16 @@ class ExtractRequest(BaseModel):
436
456
  kind: str = "message"
437
457
 
438
458
 
459
+ class VectorQueryRequest(BaseModel):
460
+ """Approximate neighbours from the HNSW sidecar. ``k`` is capped at 200."""
461
+
462
+ workspace: Optional[str] = None
463
+ embedding_model: str
464
+ embedding_dim: int
465
+ vector: List[float] = Field(default_factory=list)
466
+ k: int = 10
467
+
468
+
439
469
  def build_extract_reply(text: str, kind: str) -> Dict[str, Any]:
440
470
  """The structures ``ingest_message`` / ``ingest_document`` / ``ingest_source`` consume.
441
471
 
@@ -490,8 +520,9 @@ def create_worker_compute_router(
490
520
  transcriber: Optional[Callable[[str], str]] = None,
491
521
  require_user: Callable[[Request], Any],
492
522
  enforce_rate_limit: Callable[[str, str], None],
523
+ db_path: Any = None,
493
524
  ) -> APIRouter:
494
- """The eight compute seams, wired to what this worker actually resolved.
525
+ """The nine compute seams, wired to what this worker actually resolved.
495
526
 
496
527
  ``embedder`` is the :class:`~latticeai.core.embedding_providers.text.ResolvedEmbedder`
497
528
  ``phase_brain`` built (``None`` ⇒ 503, because a worker with no embedder is
@@ -580,15 +611,37 @@ def create_worker_compute_router(
580
611
  )
581
612
  provider = embedder.provider
582
613
  texts = list(req.texts)
614
+ if len(texts) > EMBED_MAX_BATCH:
615
+ raise http_error(
616
+ 422,
617
+ "worker_compute.embed_batch_too_large",
618
+ language,
619
+ count=str(len(texts)),
620
+ limit=str(EMBED_MAX_BATCH),
621
+ )
583
622
  if kind == "passage":
584
623
  texts = [text[:PASSAGE_MAX_CHARS] for text in texts]
585
- vectors = await asyncio.to_thread(provider.embed_batch, texts)
624
+ # `embed_batch_for` rather than `embed_batch`: an asymmetric model (the
625
+ # E5 family) needs to know whether this text is the question or the
626
+ # answer, and `kind` is exactly that. Symmetric providers ignore it.
627
+ # Resolved by name because the embedder arrives injected: a stand-in
628
+ # that predates the role-aware method still embeds, it just cannot be
629
+ # told which role it is embedding for.
630
+ role_aware = getattr(provider, "embed_batch_for", None)
631
+ vectors = (
632
+ await asyncio.to_thread(role_aware, texts, kind)
633
+ if callable(role_aware)
634
+ else await asyncio.to_thread(provider.embed_batch, texts)
635
+ )
586
636
  return {
587
637
  "vectors": vectors,
588
638
  "dim": provider.dim,
589
639
  "provider": embedder.active,
590
640
  "model_id": provider.model_id,
591
641
  "kind": kind,
642
+ # Additive (v12.0.0): `fallback` means these vectors are feature
643
+ # hashes, not meaning. A caller that stores them can say so.
644
+ "grade": getattr(provider, "grade", "fallback"),
592
645
  }
593
646
 
594
647
  @router.post("/worker/parse")
@@ -758,12 +811,33 @@ def create_worker_compute_router(
758
811
  )
759
812
  return await asyncio.to_thread(build_extract_reply, req.text, kind)
760
813
 
814
+ @router.post("/worker/vector/query")
815
+ async def worker_vector_query(req: VectorQueryRequest, request: Request):
816
+ """Top-k ids from the HNSW sidecar, or ``index: "none"`` honestly."""
817
+ _require_seam(request)
818
+ _admit(request)
819
+ language = resolve_language(request)
820
+ if not req.embedding_model or req.embedding_dim <= 0 or not req.vector:
821
+ raise http_error(422, "worker_compute.vector_query_invalid", language)
822
+ from latticeai.core.vector_index import query_sidecar
823
+
824
+ return await asyncio.to_thread(
825
+ query_sidecar,
826
+ workspace=req.workspace,
827
+ embedding_model=req.embedding_model,
828
+ embedding_dim=req.embedding_dim,
829
+ vector=req.vector,
830
+ k=req.k,
831
+ db_path=db_path,
832
+ )
833
+
761
834
  return router
762
835
 
763
836
 
764
837
  __all__ = [
765
838
  "CJK_FONT_CANDIDATES",
766
839
  "EMBED_KINDS",
840
+ "EMBED_MAX_BATCH",
767
841
  "EXTRACT_KINDS",
768
842
  "EXTRACT_LIMITS",
769
843
  "PASSAGE_MAX_CHARS",
@@ -772,6 +846,7 @@ __all__ = [
772
846
  "EmbedRequest",
773
847
  "ExtractRequest",
774
848
  "ParseRequest",
849
+ "VectorQueryRequest",
775
850
  "RenderDocxRequest",
776
851
  "RenderPdfRequest",
777
852
  "RenderPptxRequest",
@@ -68,6 +68,14 @@ from __future__ import annotations
68
68
  from lattice_brain.embeddings import DEFAULT_EMBEDDING_DIM as DEFAULT_EMBEDDING_DIM
69
69
  from lattice_brain.embeddings import LocalEmbeddingModel as LocalEmbeddingModel
70
70
 
71
+ from .autodetect import AUTO_PROVIDER as AUTO_PROVIDER
72
+ from .autodetect import AUTODETECT_ENV as AUTODETECT_ENV
73
+ from .autodetect import LOCAL_MLX_MODELS as LOCAL_MLX_MODELS
74
+ from .autodetect import Detection as Detection
75
+ from .autodetect import detect_embedder as detect_embedder
76
+ from .autodetect import detect_local_mlx as detect_local_mlx
77
+ from .autodetect import detect_ollama as detect_ollama
78
+ from .autodetect import resolve_auto_provider as resolve_auto_provider
71
79
  from .base import _KNOWN_DIMS as _KNOWN_DIMS
72
80
  from .base import EmbeddingProvider as EmbeddingProvider
73
81
  from .base import EmbeddingUnavailable as EmbeddingUnavailable
@@ -113,6 +121,14 @@ from .vision import build_vision_provider as build_vision_provider
113
121
  from .vision import resolve_vision_embedder as resolve_vision_embedder
114
122
 
115
123
  __all__ = [
124
+ "AUTODETECT_ENV",
125
+ "AUTO_PROVIDER",
126
+ "LOCAL_MLX_MODELS",
127
+ "Detection",
128
+ "detect_embedder",
129
+ "detect_local_mlx",
130
+ "detect_ollama",
131
+ "resolve_auto_provider",
116
132
  "DEFAULT_CAPTION_PROMPT",
117
133
  "DEFAULT_VISION_DIM",
118
134
  "VISION_CAPTION_TARGET_ENV",
@@ -0,0 +1,302 @@
1
+ """Find the real embedder that is already on this machine.
2
+
3
+ The default text embedder is :class:`~.text.HashEmbeddingProvider` — feature
4
+ hashing into 384 dimensions. It is deterministic, offline and always available,
5
+ and it is **not semantic**: two sentences that mean the same thing in different
6
+ words score near zero against each other. Every recall complaint about "왜 못
7
+ 찾지" that is not a keyword problem is this.
8
+
9
+ A real provider has always been configurable (``LATTICEAI_EMBEDDING_PROVIDER``),
10
+ but nothing ever *looked* for one, so a machine with a perfectly good embedding
11
+ model already in its Hugging Face cache still hashed. This module is that look:
12
+
13
+ 1. an explicit configuration always wins and is never second-guessed;
14
+ 2. otherwise, if a known embedding model is already **downloaded**, name it —
15
+ filesystem only, no network, no download;
16
+ 3. otherwise, if an Ollama server is reachable and has an embedding model
17
+ pulled, name that;
18
+ 4. otherwise, nothing was found, and the hash fallback stays.
19
+
20
+ Detection **reports**; adoption is a separate decision the caller makes (see
21
+ :func:`resolve_auto_provider`). ``LATTICEAI_EMBEDDING_PROVIDER=auto`` adopts
22
+ what was found; anything else leaves the resolution exactly as it was and the
23
+ finding travels to ``GET /api/embeddings/status`` as ``detected``, so the setup
24
+ surface can offer the switch instead of taking it.
25
+
26
+ ## Why adoption is not automatic
27
+
28
+ Rust files every vector under ``(embedding_model, embedding_dim)`` and searches
29
+ only rows whose identity matches the embedder it holds — today the hash model.
30
+ Switching the *worker's* provider therefore does not corrupt anything (the two
31
+ identities never mix), but until the read path can embed a query through the
32
+ same provider, provider vectors are written and never read. Adopting silently
33
+ would buy inference cost and no recall. The switch is real, tested and one env
34
+ var away; it is not a default.
35
+ """
36
+
37
+ from __future__ import annotations
38
+
39
+ import os
40
+ from dataclasses import dataclass, field
41
+ from pathlib import Path
42
+ from typing import Any, Dict, List, Optional, Tuple
43
+
44
+ #: ``LATTICEAI_EMBEDDING_PROVIDER=auto`` — adopt whatever detection found.
45
+ AUTO_PROVIDER = "auto"
46
+ #: Set to ``0``/``false`` to skip detection entirely (and its Ollama probe).
47
+ AUTODETECT_ENV = "LATTICEAI_EMBEDDING_AUTODETECT"
48
+ #: Where the Ollama probe looks when nothing else says.
49
+ OLLAMA_BASE_ENV = "LATTICEAI_EMBEDDING_BASE_URL"
50
+ DEFAULT_OLLAMA_BASE = "http://127.0.0.1:11434"
51
+
52
+ #: Local MLX embedding models this build knows how to drive, best first.
53
+ #: ``dim`` is the model's true output width; ``prefix`` marks the E5 family,
54
+ #: whose quality depends on the ``query:`` / ``passage:`` instruction.
55
+ LOCAL_MLX_MODELS: Tuple[Tuple[str, int, bool], ...] = (
56
+ ("mlx-community/multilingual-e5-small-mlx", 384, True),
57
+ ("mlx-community/multilingual-e5-base-mlx", 768, True),
58
+ ("mlx-community/multilingual-e5-large-mlx", 1024, True),
59
+ ("mlx-community/snowflake-arctic-embed-l-v2.0-8bit", 1024, False),
60
+ ("mlx-community/embeddinggemma-300m-4bit", 768, False),
61
+ ("mlx-community/bge-m3", 1024, False),
62
+ )
63
+
64
+ #: Ollama model names that are embedders rather than chat models.
65
+ OLLAMA_EMBEDDING_MODELS: Tuple[Tuple[str, int], ...] = (
66
+ ("bge-m3", 1024),
67
+ ("mxbai-embed-large", 1024),
68
+ ("nomic-embed-text", 768),
69
+ ("all-minilm", 384),
70
+ )
71
+
72
+
73
+ @dataclass
74
+ class Detection:
75
+ """What was found, and how. Every field is safe to show a user."""
76
+
77
+ #: ``"mlx"`` | ``"ollama"`` | ``""`` when nothing was found.
78
+ provider: str = ""
79
+ model: str = ""
80
+ dim: int = 0
81
+ #: ``configured`` | ``local_model`` | ``ollama`` | ``none``.
82
+ source: str = "none"
83
+ detail: str = ""
84
+ #: Every candidate that was looked for and not found, for the UI to offer.
85
+ candidates: List[Dict[str, Any]] = field(default_factory=list)
86
+
87
+ @property
88
+ def found(self) -> bool:
89
+ return bool(self.provider)
90
+
91
+ def as_dict(self) -> Dict[str, Any]:
92
+ return {
93
+ "found": self.found,
94
+ "provider": self.provider,
95
+ "model": self.model,
96
+ "dim": self.dim,
97
+ "source": self.source,
98
+ "detail": self.detail,
99
+ "candidates": list(self.candidates),
100
+ }
101
+
102
+
103
+ def autodetect_enabled(env: Optional[Dict[str, str]] = None) -> bool:
104
+ """Whether to look at all. On unless explicitly switched off."""
105
+ raw = (env or dict(os.environ)).get(AUTODETECT_ENV, "1").strip().lower()
106
+ return raw not in {"0", "false", "no", "off"}
107
+
108
+
109
+ def hf_cache_roots(env: Optional[Dict[str, str]] = None) -> List[Path]:
110
+ """Every directory a Hugging Face snapshot could be under, in order."""
111
+ values = env or dict(os.environ)
112
+ roots: List[Path] = []
113
+ for key in ("HF_HUB_CACHE", "HUGGINGFACE_HUB_CACHE"):
114
+ raw = values.get(key, "").strip()
115
+ if raw:
116
+ roots.append(Path(raw))
117
+ home = values.get("HF_HOME", "").strip()
118
+ if home:
119
+ roots.append(Path(home) / "hub")
120
+ roots.append(Path(values.get("HOME", "~")).expanduser() / ".cache/huggingface/hub")
121
+ seen: List[Path] = []
122
+ for root in roots:
123
+ if root not in seen:
124
+ seen.append(root)
125
+ return seen
126
+
127
+
128
+ def _snapshot_dir(repo_id: str, roots: List[Path]) -> Optional[Path]:
129
+ """The snapshot directory holding this repo's weights, if it is on disk.
130
+
131
+ A cache entry exists as soon as *anything* has been fetched, so presence of
132
+ the directory is not enough: a snapshot with a ``config.json`` and at least
133
+ one weight file is what "already downloaded" has to mean, or the detector
134
+ names a model that cannot load.
135
+ """
136
+ flattened = "models--" + repo_id.replace("/", "--")
137
+ for root in roots:
138
+ snapshots = root / flattened / "snapshots"
139
+ if not snapshots.is_dir():
140
+ continue
141
+ for snapshot in sorted(snapshots.iterdir()):
142
+ if not snapshot.is_dir():
143
+ continue
144
+ if not (snapshot / "config.json").exists():
145
+ continue
146
+ weights = any(
147
+ (snapshot / name).exists()
148
+ for name in ("model.safetensors", "weights.safetensors")
149
+ ) or any(snapshot.glob("*.safetensors"))
150
+ if weights:
151
+ return snapshot
152
+ return None
153
+
154
+
155
+ def detect_local_mlx(env: Optional[Dict[str, str]] = None) -> Detection:
156
+ """The best already-downloaded MLX embedding model, or an empty finding."""
157
+ roots = hf_cache_roots(env)
158
+ candidates: List[Dict[str, Any]] = []
159
+ for repo_id, dim, prefixed in LOCAL_MLX_MODELS:
160
+ snapshot = _snapshot_dir(repo_id, roots)
161
+ candidates.append(
162
+ {
163
+ "provider": "mlx",
164
+ "model": repo_id,
165
+ "dim": dim,
166
+ "downloaded": snapshot is not None,
167
+ "e5_prefixes": prefixed,
168
+ }
169
+ )
170
+ for candidate in candidates:
171
+ if candidate["downloaded"]:
172
+ return Detection(
173
+ provider="mlx",
174
+ model=str(candidate["model"]),
175
+ dim=int(candidate["dim"]),
176
+ source="local_model",
177
+ detail=f"{candidate['model']} is already in the local model cache",
178
+ candidates=candidates,
179
+ )
180
+ return Detection(
181
+ source="none",
182
+ detail="no known embedding model is downloaded yet",
183
+ candidates=candidates,
184
+ )
185
+
186
+
187
+ def detect_ollama(
188
+ base_url: str = "",
189
+ timeout: float = 1.5,
190
+ env: Optional[Dict[str, str]] = None,
191
+ ) -> Detection:
192
+ """An Ollama server with an embedding model pulled, or an empty finding.
193
+
194
+ One localhost GET with a short timeout. Any failure — no server, no httpx,
195
+ a slow answer — is "not found", never an error: detection must not be able
196
+ to delay or fail a boot.
197
+ """
198
+ values = env or dict(os.environ)
199
+ base = (base_url or values.get(OLLAMA_BASE_ENV, "") or DEFAULT_OLLAMA_BASE).rstrip("/")
200
+ try:
201
+ import httpx
202
+
203
+ with httpx.Client(timeout=timeout) as client:
204
+ response = client.get(f"{base}/api/tags")
205
+ response.raise_for_status()
206
+ payload = response.json()
207
+ except Exception as exc:
208
+ return Detection(source="none", detail=f"no Ollama server at {base}: {exc}")
209
+ names = [
210
+ str(model.get("name") or "")
211
+ for model in (payload.get("models") or [])
212
+ if isinstance(model, dict)
213
+ ]
214
+ for known, dim in OLLAMA_EMBEDDING_MODELS:
215
+ for name in names:
216
+ if name.split(":")[0] == known:
217
+ return Detection(
218
+ provider="ollama",
219
+ model=name,
220
+ dim=dim,
221
+ source="ollama",
222
+ detail=f"{name} is pulled on the Ollama server at {base}",
223
+ )
224
+ return Detection(
225
+ source="none",
226
+ detail=f"Ollama at {base} has no embedding model pulled",
227
+ )
228
+
229
+
230
+ def detect_embedder(
231
+ configured_provider: str = "",
232
+ configured_model: str = "",
233
+ env: Optional[Dict[str, str]] = None,
234
+ probe_ollama: bool = True,
235
+ ) -> Detection:
236
+ """What this machine could embed with, without downloading anything.
237
+
238
+ A configured provider is reported back as ``source="configured"`` and no
239
+ probing happens: the operator already answered the question.
240
+ """
241
+ values = env or dict(os.environ)
242
+ configured = str(configured_provider or "").strip().lower()
243
+ if configured and configured not in {"hash", "local", "fallback", AUTO_PROVIDER}:
244
+ return Detection(
245
+ provider=configured,
246
+ model=configured_model,
247
+ source="configured",
248
+ detail=f"{configured} is configured explicitly",
249
+ )
250
+ if not autodetect_enabled(values):
251
+ return Detection(source="none", detail=f"{AUTODETECT_ENV} is off")
252
+
253
+ local = detect_local_mlx(values)
254
+ if local.found:
255
+ return local
256
+ if probe_ollama:
257
+ ollama = detect_ollama(env=values)
258
+ if ollama.found:
259
+ ollama.candidates = local.candidates
260
+ return ollama
261
+ return local
262
+
263
+
264
+ def resolve_auto_provider(
265
+ configured_provider: str,
266
+ configured_model: str,
267
+ configured_dim: int,
268
+ detection: Detection,
269
+ ) -> Tuple[str, str, int]:
270
+ """The ``(provider, model, dim)`` the embedder should actually be built with.
271
+
272
+ Only ``provider == "auto"`` changes anything, and only when detection found
273
+ something; ``auto`` with nothing found resolves to ``hash``, which is the
274
+ honest answer rather than a construction that will fail its probe.
275
+ Explicit configuration is returned untouched.
276
+ """
277
+ requested = str(configured_provider or "").strip().lower()
278
+ if requested != AUTO_PROVIDER:
279
+ return configured_provider, configured_model, configured_dim
280
+ if not detection.found:
281
+ return "hash", "", configured_dim
282
+ return (
283
+ detection.provider,
284
+ configured_model or detection.model,
285
+ configured_dim or detection.dim,
286
+ )
287
+
288
+
289
+ __all__ = [
290
+ "AUTODETECT_ENV",
291
+ "AUTO_PROVIDER",
292
+ "DEFAULT_OLLAMA_BASE",
293
+ "LOCAL_MLX_MODELS",
294
+ "OLLAMA_EMBEDDING_MODELS",
295
+ "Detection",
296
+ "autodetect_enabled",
297
+ "detect_embedder",
298
+ "detect_local_mlx",
299
+ "detect_ollama",
300
+ "hf_cache_roots",
301
+ "resolve_auto_provider",
302
+ ]