ltcai 11.9.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (127) hide show
  1. package/README.md +80 -59
  2. package/docs/CHANGELOG.md +92 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +267 -119
  5. package/docs/ENTERPRISE.md +1 -1
  6. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  7. package/docs/ONBOARDING.md +13 -3
  8. package/docs/OPERATIONS.md +13 -4
  9. package/docs/REALTIME_COLLABORATION.md +1 -1
  10. package/docs/ROADMAP.md +113 -0
  11. package/docs/TRUST_MODEL.md +16 -2
  12. package/docs/WHY_LATTICE.md +8 -2
  13. package/docs/WORKFLOW_DESIGNER.md +2 -2
  14. package/docs/kg-schema.md +51 -5
  15. package/docs/mcp-tools.md +17 -0
  16. package/lattice_brain/__init__.py +1 -1
  17. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  18. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  19. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  20. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  21. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  22. package/lattice_brain/graph/_kg_constants.py +7 -0
  23. package/latticeai/__init__.py +1 -1
  24. package/latticeai/api/agent_worker_seam.py +32 -0
  25. package/latticeai/api/worker_compute.py +78 -3
  26. package/latticeai/core/embedding_providers/__init__.py +16 -0
  27. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  28. package/latticeai/core/embedding_providers/base.py +25 -0
  29. package/latticeai/core/embedding_providers/profiles.py +44 -0
  30. package/latticeai/core/embedding_providers/text.py +74 -8
  31. package/latticeai/core/vector_index/__init__.py +61 -0
  32. package/latticeai/core/vector_index/hnsw.py +383 -0
  33. package/latticeai/core/vector_index/sidecar.py +329 -0
  34. package/latticeai/models/router/generation.py +101 -22
  35. package/latticeai/models/router/loading.py +109 -4
  36. package/latticeai/runtime/brain_runtime.py +43 -9
  37. package/latticeai/runtime/build_phases/worker_profile.py +13 -4
  38. package/latticeai/services/architecture_readiness.py +2 -2
  39. package/latticeai/services/product_readiness.py +10 -5
  40. package/latticeai/services/search_service.py +7 -0
  41. package/latticeai/tools/__init__.py +6 -1
  42. package/latticeai/tools/documents.py +12 -0
  43. package/latticeai/tools/markup.py +152 -0
  44. package/package.json +2 -1
  45. package/scripts/check_current_release_docs.mjs +1 -1
  46. package/scripts/compose_openapi.py +2 -0
  47. package/scripts/openapi_route_families.json +7 -3
  48. package/scripts/publish_release.mjs +157 -0
  49. package/scripts/release_screen_claims.json +12 -0
  50. package/src-tauri/Cargo.lock +44 -10
  51. package/src-tauri/Cargo.toml +1 -1
  52. package/src-tauri/tauri.conf.json +1 -1
  53. package/static/app/asset-manifest.json +47 -41
  54. package/static/app/assets/Act-Cf1L2709.js +2 -0
  55. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  56. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  57. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  58. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  59. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  60. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  61. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  62. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  63. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  64. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  65. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  66. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  67. package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
  68. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  69. package/static/app/assets/System-CAxwBUXw.js +1 -0
  70. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  71. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  72. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  73. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  74. package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
  75. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  76. package/static/app/assets/button-CmaEqG1T.js +1 -0
  77. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  78. package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
  79. package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
  80. package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
  81. package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
  82. package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
  83. package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
  84. package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
  85. package/static/app/assets/index-D2H-wSl6.js +13 -0
  86. package/static/app/assets/input-Df1CAY_I.js +1 -0
  87. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  88. package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
  89. package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
  90. package/static/app/assets/primitives-BioD2slS.js +1 -0
  91. package/static/app/assets/search-BzBw8YcW.js +1 -0
  92. package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
  93. package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
  94. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  95. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  96. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  97. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  98. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  99. package/static/app/index.html +4 -4
  100. package/static/sw.js +1 -1
  101. package/static/app/assets/Act-B4WT81kh.js +0 -1
  102. package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
  103. package/static/app/assets/Brain-CtYaa26c.js +0 -321
  104. package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
  105. package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
  106. package/static/app/assets/Capture-DdNi5Peb.js +0 -1
  107. package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
  108. package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
  109. package/static/app/assets/Library-Bl5XClFV.js +0 -1
  110. package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
  111. package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
  112. package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
  113. package/static/app/assets/System-BJ6jQ_SL.js +0 -1
  114. package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
  115. package/static/app/assets/brain-B9BDrMTe.js +0 -1
  116. package/static/app/assets/button-D6JcpYcf.js +0 -1
  117. package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
  118. package/static/app/assets/index-CWKRRsLW.js +0 -10
  119. package/static/app/assets/input-CEqsxtil.js +0 -1
  120. package/static/app/assets/primitives-DORg7Z_7.js +0 -1
  121. package/static/app/assets/search-0NQ21wXe.js +0 -1
  122. package/static/app/assets/textarea-jtQcRSXo.js +0 -1
  123. package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
  124. package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
  125. package/static/app/assets/useQuery-BizqBNGw.js +0 -1
  126. package/static/app/assets/utils-WgW4V69R.js +0 -4
  127. package/static/app/assets/workspace-DSek3jCY.js +0 -1
@@ -15,6 +15,7 @@ from __future__ import annotations
15
15
  # under the module-wide directive `_kg_common.py` carried before the split.
16
16
  # ruff: noqa: F841
17
17
  import asyncio
18
+ import functools
18
19
  import json
19
20
  import logging
20
21
  import os
@@ -22,12 +23,15 @@ import re
22
23
  from typing import Any, Dict, List, Optional
23
24
 
24
25
  from ..runtime import get_llm_router
26
+ from .normalize import merge_entity_aliases
27
+ from .patterns import concept_positions, typed_relation
25
28
  from .relations import (
26
29
  COOCCURRENCE_CONCEPT_LIMIT,
27
30
  COOCCURRENCE_EDGE_WEIGHT,
28
31
  VERB_EDGE_WEIGHT,
29
32
  infer_edge_relation,
30
33
  )
34
+ from .sections import heading_at, heading_spans, sentence_offsets, with_section
31
35
  from .text import _clean_text
32
36
 
33
37
  _LLM_EXTRACT_CONCEPT_PROMPT = """Extract the key concepts from the following text.
@@ -288,13 +292,238 @@ _CONCEPT_STOP: set = {
288
292
 
289
293
 
290
294
  def _extract_concepts(text: str, limit: int = 12) -> List[str]:
291
- """LLM-first concept extraction with rule-based fallback."""
295
+ """LLM-first concept extraction with rule-based fallback.
296
+
297
+ Both paths are pushed through :func:`merge_entity_aliases` before they
298
+ leave, so ``"Lattice AI"`` / ``"lattice ai"`` / ``"지식그래프에서"`` become
299
+ one concept and therefore one node. The merge is deterministic and needs no
300
+ model — the LLM is free to be inconsistent about case and 조사, and the
301
+ graph still gets one entity.
302
+ """
292
303
  llm_result = _llm_extract_concepts(text, limit)
293
304
  if llm_result:
294
- return llm_result
305
+ return merge_entity_aliases(llm_result, text)[:limit]
295
306
  return _extract_concepts_rules(text, limit)
296
307
 
297
308
 
309
+ #: Longest a concept may be, in words. A name is a name; five words is already
310
+ #: generous for "OpenAI GPT-4o Mini Preview". Past that it is a clause.
311
+ _MAX_CONCEPT_WORDS = 5
312
+
313
+ # Precompiled once: each of these used to be a ``re.findall`` literal inside
314
+ # ``_extract_concepts_rules``, which recompiled them on every document.
315
+ _RE_BACKTICK = re.compile(r"`([^`]{2,40})`")
316
+ _RE_CODE_PUNCT = re.compile(r"[\(\)\[\]{}]")
317
+ _RE_DQUOTE = re.compile(r'"([^"]{2,40})"')
318
+ _RE_PROPER_MIXED = re.compile(
319
+ r"([A-Z][a-z]{1,20}(?:[ \t]+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})"
320
+ )
321
+ _RE_PROPER_CAPS = re.compile(
322
+ r"([A-Z]{2,6}(?:[ \t]+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})"
323
+ )
324
+ _RE_PROPER_SINGLE = re.compile(
325
+ r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])"
326
+ )
327
+ _RE_SENTENCE_START = re.compile(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)")
328
+ _RE_KO_COMPOUND = re.compile(
329
+ r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)"
330
+ )
331
+ _RE_KO_PARTICLE = re.compile(
332
+ r"([가-힣]{2,12})"
333
+ r"(?:에서|에게|으로|부터|까지|보다|처럼|한테|은|는|이|가|을|를|의|와|과|에|도|로|만)"
334
+ )
335
+ _RE_KO_DEFINITION = re.compile(r"([가-힣]{2,12}?)(?:이란|란)(?![가-힣])")
336
+ _RE_HYPHEN_ID = re.compile(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b")
337
+ _RE_DECISION = re.compile(r"(결정|확정|하기로|decided|decision)")
338
+ _RE_TASK = re.compile(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])")
339
+ _RE_TOPIC_TOKEN = re.compile(r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}")
340
+
341
+ #: A capitalized sentence opener is not part of the name that follows it.
342
+ #: "The Vector Index" and "Unlike Keyword Search" are the sentence's grammar,
343
+ #: not the entity's; the lowercase forms already sit in :data:`_CONCEPT_STOP`,
344
+ #: and these are the ones that also appear capitalized at a sentence start.
345
+ _LEADING_STOPWORDS: frozenset = frozenset(
346
+ {
347
+ "a",
348
+ "an",
349
+ "the",
350
+ "this",
351
+ "that",
352
+ "these",
353
+ "those",
354
+ "and",
355
+ "but",
356
+ "for",
357
+ "with",
358
+ "without",
359
+ "in",
360
+ "on",
361
+ "at",
362
+ "by",
363
+ "from",
364
+ "into",
365
+ "unlike",
366
+ "instead",
367
+ "our",
368
+ "their",
369
+ "its",
370
+ "some",
371
+ "each",
372
+ "every",
373
+ "both",
374
+ "when",
375
+ "while",
376
+ "where",
377
+ "if",
378
+ "as",
379
+ "than",
380
+ "then",
381
+ "so",
382
+ "also",
383
+ "now",
384
+ "here",
385
+ "there",
386
+ "no",
387
+ "not",
388
+ "only",
389
+ "just",
390
+ }
391
+ )
392
+
393
+ #: The character class a term's own edge must not touch. A Latin word ends at
394
+ #: the next Latin letter, **not** at the Korean particle glued to it: with a
395
+ #: blanket `\w` guard, `"Lattice AI"` never matches inside `"Lattice AI는"` and
396
+ #: the containment de-duplication silently keeps `"Lattice"` as a second node.
397
+ _ASCII_EDGE = "A-Za-z0-9_"
398
+ _HANGUL_EDGE = "가-힣"
399
+
400
+
401
+ def _edge_class(char: str) -> str:
402
+ """Which character class would continue the word ``char`` ends (or starts)."""
403
+ return _HANGUL_EDGE if "가" <= char <= "힣" else _ASCII_EDGE
404
+
405
+
406
+ def _in_edge_class(char: str, edge: str) -> bool:
407
+ """True when ``char`` would continue a word whose edge class is ``edge``."""
408
+ if edge is _HANGUL_EDGE:
409
+ return "가" <= char <= "힣"
410
+ return (
411
+ ("A" <= char <= "Z")
412
+ or ("a" <= char <= "z")
413
+ or ("0" <= char <= "9")
414
+ or char == "_"
415
+ )
416
+
417
+
418
+ def _fold_len_stable(term: str) -> bool:
419
+ """True when case-folding ``term`` cannot move character offsets.
420
+
421
+ ``re.IGNORECASE`` and ``str.lower()`` agree on span starts for these
422
+ terms; anything else (``İ``, ``ß``) falls back to the compiled regex so
423
+ the match list stays byte-identical.
424
+ """
425
+ return len(term) == len(term.lower()) == len(term.upper())
426
+
427
+
428
+ @functools.lru_cache(maxsize=4096)
429
+ def _whole_word_re(term: str) -> re.Pattern[str]:
430
+ """The compiled whole-word pattern for ``term`` (cached)."""
431
+ return re.compile(
432
+ f"(?<![{_edge_class(term[0])}])"
433
+ + re.escape(term)
434
+ + f"(?![{_edge_class(term[-1])}])",
435
+ re.IGNORECASE,
436
+ )
437
+
438
+
439
+ def _drop_leading_stopwords(phrase: str) -> str:
440
+ """``phrase`` without the sentence-grammar words in front of the name.
441
+
442
+ Stops before eating the whole phrase: a two-word match that is *entirely*
443
+ stopwords is returned unchanged and refused later by ``_add``.
444
+ """
445
+ words = phrase.split()
446
+ while len(words) > 1 and words[0].lower() in _LEADING_STOPWORDS:
447
+ words = words[1:]
448
+ return " ".join(words)
449
+
450
+
451
+ def _whole_word_spans(
452
+ term: str,
453
+ haystack: str,
454
+ haystack_lower: Optional[str] = None,
455
+ ) -> List[Any]:
456
+ """``(start, end)`` for every whole-word occurrence of ``term``.
457
+
458
+ On case-fold-stable terms this is a ``str.find`` walk with the same
459
+ lookaround rules as the compiled regex; anything whose lower/upper
460
+ length differs (or an empty term) uses the regex so the span list
461
+ cannot drift.
462
+ """
463
+ if not term:
464
+ return []
465
+ if _fold_len_stable(term):
466
+ return _spans_by_find(term, haystack, haystack_lower)
467
+ return [m.span() for m in _whole_word_re(term).finditer(haystack)]
468
+
469
+
470
+ def _spans_by_find(
471
+ term: str,
472
+ haystack: str,
473
+ haystack_lower: Optional[str] = None,
474
+ ) -> List[Any]:
475
+ """Whole-word spans via case-folded ``find`` — identical to the regex."""
476
+ needle = term.lower()
477
+ hay = haystack_lower if haystack_lower is not None else haystack.lower()
478
+ left_edge = _edge_class(term[0])
479
+ right_edge = _edge_class(term[-1])
480
+ width = len(needle)
481
+ spans: List[Any] = []
482
+ start = 0
483
+ while True:
484
+ idx = hay.find(needle, start)
485
+ if idx < 0:
486
+ return spans
487
+ end = idx + width
488
+ if (idx == 0 or not _in_edge_class(haystack[idx - 1], left_edge)) and (
489
+ end >= len(haystack) or not _in_edge_class(haystack[end], right_edge)
490
+ ):
491
+ spans.append((idx, end))
492
+ start = idx + 1
493
+
494
+
495
+ def _term_is_whole_word_in(short: str, longer: str) -> bool:
496
+ """True when ``short`` occurs as a whole word inside ``longer``."""
497
+ if not short:
498
+ return False
499
+ if _fold_len_stable(short):
500
+ return bool(_spans_by_find(short, longer))
501
+ return _whole_word_re(short).search(longer) is not None
502
+
503
+
504
+ def _always_inside(short: str, longer: List[str], spans: Dict[str, List[Any]]) -> bool:
505
+ """True when every occurrence of ``short`` sits inside a longer concept.
506
+
507
+ Span containment rather than a per-term count, because a short term can be
508
+ covered by *different* longer terms in different places — "Vector" is
509
+ inside "Vector Index" here and "Vector Search" there, and counting against
510
+ one long term at a time would keep it as a third, meaningless node.
511
+
512
+ ``spans`` is precomputed once per document by the caller. Scanning the text
513
+ again per (short, long) pair is quadratic in the *candidate count* times
514
+ linear in the document, which on a 100 KB file took minutes; the same
515
+ answer from cached spans is interval arithmetic over a few dozen tuples.
516
+ """
517
+ hits = spans.get(short) or []
518
+ if not hits:
519
+ return False
520
+ covers = [span for term in longer for span in spans.get(term) or []]
521
+ return all(
522
+ any(start >= c_start and end <= c_end for c_start, c_end in covers)
523
+ for start, end in hits
524
+ )
525
+
526
+
298
527
  def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
299
528
  """Extract meaningful named concepts from text (rule-based).
300
529
 
@@ -309,43 +538,44 @@ def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
309
538
  seen: dict = {} # concept_lower → original form
310
539
 
311
540
  def _add(term: str) -> None:
541
+ # A candidate that crosses a line, or runs past a handful of words, is
542
+ # a quoted *passage* the backtick and quote rules swept up — the graph
543
+ # was storing whole sentences as entity names.
544
+ if "\n" in term or len(term.split()) > _MAX_CONCEPT_WORDS:
545
+ return
312
546
  key = term.strip().lower()
313
547
  if key and key not in _CONCEPT_STOP and not key.isdigit() and len(key) >= 2:
314
548
  seen.setdefault(key, term.strip())
315
549
 
316
550
  # 1. Backtick-quoted code/term (highest confidence)
317
- for m in re.findall(r"`([^`]{2,40})`", text):
318
- if not re.search(r"[\(\)\[\]{}]", m): # skip code expressions
551
+ for m in _RE_BACKTICK.findall(text):
552
+ if not _RE_CODE_PUNCT.search(m): # skip code expressions
319
553
  _add(m)
320
554
 
321
555
  # 2. Double/single quoted terms
322
- for m in re.findall(r'"([^"]{2,40})"', text):
556
+ for m in _RE_DQUOTE.findall(text):
323
557
  _add(m)
324
558
 
325
559
  # 3. Multi-word English proper nouns (Title Case or ALL-CAPS first word, 2–4 words).
560
+ # The inner separator is `[ \t]+`, not `\s+`: `\s` crosses newlines, so a
561
+ # heading glued to the next line's first word produced phantom concepts
562
+ # like "Retrieval Lattice AI" — and, worse, consumed the "Lattice AI"
563
+ # the next pass would have found (regex scanning does not overlap).
326
564
  # Pattern A: Mixed-case first word — "Lattice AI", "Tool Use", "Graph RAG"
327
- for m in re.findall(
328
- r"([A-Z][a-z]{1,20}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})",
329
- text,
330
- ):
331
- _add(m)
565
+ for m in _RE_PROPER_MIXED.findall(text):
566
+ _add(_drop_leading_stopwords(m))
332
567
  # Pattern B: ALL-CAPS first word — "VS Code", "MCP Server", "GPT-4o Mini"
333
- for m in re.findall(
334
- r"([A-Z]{2,6}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})",
335
- text,
336
- ):
337
- _add(m)
568
+ for m in _RE_PROPER_CAPS.findall(text):
569
+ _add(_drop_leading_stopwords(m))
338
570
 
339
571
  # 4. Single capitalized proper noun.
340
572
  # Use ASCII-boundary lookaround instead of \b so Korean particles
341
573
  # (와, 의, 는 …) after an English word don't block the match.
342
- all_caps_words = re.findall(
343
- r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])", text
344
- )
574
+ all_caps_words = _RE_PROPER_SINGLE.findall(text)
345
575
  freq: Dict[str, int] = {}
346
576
  for w in all_caps_words:
347
577
  freq[w] = freq.get(w, 0) + 1
348
- sentence_starts = set(re.findall(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)", text))
578
+ sentence_starts = set(_RE_SENTENCE_START.findall(text))
349
579
  for m, cnt in freq.items():
350
580
  if m.lower() in _CONCEPT_STOP:
351
581
  continue
@@ -353,64 +583,102 @@ def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
353
583
  _add(m)
354
584
 
355
585
  # 5. Korean technical compound nouns (3–12 chars, no common particles)
356
- for m in re.findall(
357
- r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)",
358
- text,
359
- ):
586
+ for m in _RE_KO_COMPOUND.findall(text):
360
587
  _add(m)
361
- # Korean standalone terms that appear after topic markers (은/는/이/가 앞)
362
- for m in re.findall(
363
- r"([가-힣]{2,12})(?:은|는|이|가|을|를|의|에서|으로|와|과)", text
364
- ):
588
+ # Korean standalone terms that appear before a particle. Multi-syllable
589
+ # particles come first in the alternation: the group is greedy, so
590
+ # "지식그래프에서" has to be offered 에서 before 에 or the leftover 서
591
+ # blocks the match entirely.
592
+ ko_stems = _RE_KO_PARTICLE.findall(text)
593
+ ko_counts: Dict[str, int] = {}
594
+ for m in ko_stems:
365
595
  if m.lower() not in _CONCEPT_STOP and len(m) >= 2:
366
- # Only add if it's non-trivial (has 3+ chars or appears multiple times)
367
- cnt = text.count(m)
368
- if len(m) >= 3 or cnt >= 2:
596
+ if m not in ko_counts:
597
+ ko_counts[m] = text.count(m)
598
+ if len(m) >= 3 or ko_counts[m] >= 2:
369
599
  _add(m)
370
600
 
601
+ # 5b. The term a Korean definition sentence is *about*: `지식그래프란 …이다`.
602
+ # The stem is lazy and the marker must end the word, so `설명이란` keeps
603
+ # `설명` (the `이` belongs to `이란`) and `의견란에` is left alone
604
+ # entirely (`란` there is part of the noun, not a marker).
605
+ for m in _RE_KO_DEFINITION.findall(text):
606
+ _add(m)
607
+
371
608
  # 6. Hyphenated / versioned identifiers (gpt-4o, gemma-4, mlx-vlm)
372
- for m in re.findall(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b", text):
609
+ for m in _RE_HYPHEN_ID.findall(text):
373
610
  if len(m) >= 4:
374
611
  _add(m)
375
612
 
376
- # De-duplicate: remove shorter if ALL its occurrences in the source text
377
- # are followed immediately by the suffix that forms the longer concept.
378
- # "Lattice" → dropped when every occurrence is "Lattice AI"
379
- # "Claude" → kept because it appears as just "Claude" too.
613
+ # De-duplicate by containment: drop the shorter term when every occurrence
614
+ # of it in the source text is *inside* a longer concept.
615
+ #
616
+ # "Lattice" → dropped, every occurrence is "Lattice AI"
617
+ # "RAG" → dropped, every occurrence is "Graph RAG" (new: suffixes)
618
+ # "Claude" → kept, it also appears on its own
619
+ # "Lat" → kept, "Lattice AI" does not contain it as a whole word
620
+ #
621
+ # Until v12.0.0 this only looked at *prefixes*, so "Graph RAG" and "RAG"
622
+ # both became nodes and the graph answered one question with two entities.
380
623
  values = list(seen.values())
381
- values_lower = [v.lower() for v in values]
624
+ # Which candidates *could* swallow which is decided on the candidate
625
+ # strings alone — matching "Vector" inside "Vector Index" is a few
626
+ # characters of work. Only the terms that survive that filter are then
627
+ # searched for in the document, so a 100 KB file is scanned a handful of
628
+ # times instead of once per candidate pair.
629
+ # The pair filter is a plain substring test before it is a regex one. A
630
+ # long document yields well over a thousand raw candidates, and a regex
631
+ # compiled per *pair* is two million compilations — seconds of work to
632
+ # answer a question `in` answers in nanoseconds for all but a few pairs.
633
+ lowered = [value.lower() for value in values]
634
+ lengths = [len(value) for value in values]
635
+ swallowers: Dict[int, List[int]] = {}
636
+ for i, short in enumerate(values):
637
+ short_l = lowered[i]
638
+ short_n = lengths[i]
639
+ longer = [
640
+ j
641
+ for j, other in enumerate(values)
642
+ if lengths[j] > short_n
643
+ and short_l in lowered[j]
644
+ and _term_is_whole_word_in(short, other)
645
+ ]
646
+ if longer:
647
+ swallowers[i] = longer
648
+ wanted = {
649
+ values[index]
650
+ for i, group in swallowers.items()
651
+ for index in [i, *group]
652
+ }
653
+ text_lower = text.lower() if wanted else ""
654
+ spans = {
655
+ term: _whole_word_spans(term, text, text_lower) for term in wanted
656
+ }
657
+
382
658
  keep = set(range(len(values)))
383
- for i, v in enumerate(values):
384
- vl = v.lower()
385
- for j, wl in enumerate(values_lower):
386
- if i == j or j not in keep:
387
- continue
388
- # Check if vl is a word-prefix of wl
389
- suffix = wl[len(vl) :]
390
- if not (wl.startswith(vl) and re.match(r"^[\s\-]", suffix)):
391
- continue
392
- # Count occurrences of v NOT followed by the suffix
393
- suffix_stripped = suffix.lstrip(" -")
394
- # Escape for regex
395
- pattern_with_suffix = re.escape(v) + r"[\s\-]+" + re.escape(suffix_stripped)
396
- pattern_alone = (
397
- re.escape(v) + r"(?![\s\-]*" + re.escape(suffix_stripped) + r")"
398
- )
399
- alone_count = len(re.findall(pattern_alone, text, re.IGNORECASE))
400
- if alone_count == 0:
401
- # Shorter term never appears alone → safe to remove
402
- keep.discard(i)
403
- break
659
+ for i, longer in swallowers.items():
660
+ alive = [values[j] for j in longer if j in keep]
661
+ if alive and _always_inside(values[i], alive, spans):
662
+ keep.discard(i)
404
663
 
405
664
  final = [values[i] for i in range(len(values)) if i in keep]
406
- return final[:limit]
665
+ # The surface merge is the last word: whatever the six passes above
666
+ # produced, two spellings of one entity leave as one concept and therefore
667
+ # as one node. Applied after the prefix de-duplication so the two rules
668
+ # compose rather than fight.
669
+ return merge_entity_aliases(final, text)[:limit]
670
+
671
+
672
+ #: `(?<=[.!?\n])\s+|\n{2,}` — the sentence split, compiled so
673
+ #: :func:`sentence_offsets` can keep each piece's position.
674
+ _SENTENCE_SPLIT = re.compile(r"(?<=[.!?\n])\s+|\n{2,}")
407
675
 
408
676
 
409
677
  def _extract_triples(
410
678
  text: str,
411
679
  concepts: List[str],
412
680
  limit: int = 20,
413
- ) -> List[Dict[str, str]]:
681
+ ) -> List[Dict[str, Any]]:
414
682
  """LLM-first triple extraction with rule-based fallback."""
415
683
  llm_result = _llm_extract_triples(text, concepts, limit)
416
684
  if llm_result:
@@ -418,83 +686,171 @@ def _extract_triples(
418
686
  return _extract_triples_rules(text, concepts, limit)
419
687
 
420
688
 
689
+ def _drop_nested(present: List[Any]) -> List[Any]:
690
+ """Drop a concept whose span sits inside another concept's span.
691
+
692
+ ``"RAG"`` found *inside* ``"Graph RAG"`` is not a second participant; it is
693
+ the same eight characters counted twice, and pairing them produces an edge
694
+ from a thing to part of its own name.
695
+ """
696
+ kept: List[Any] = []
697
+ for start, concept in present:
698
+ end = start + len(concept)
699
+ if any(
700
+ other_start <= start and end <= other_start + len(other)
701
+ for other_start, other in present
702
+ if other != concept
703
+ ):
704
+ continue
705
+ kept.append((start, concept))
706
+ return kept
707
+
708
+
709
+ def _candidate_pairs(present: List[Any]) -> List[Any]:
710
+ """``(left, right, adjacent)`` for every pair worth testing, adjacent first.
711
+
712
+ Korean puts the verb last and separates subject from object with whatever
713
+ clause it likes — ``하이브리드검색은 키워드검색이 아니라 벡터검색을 쓴다``
714
+ states a relation between the *first* and the *last* concept, and adjacency
715
+ never sees it.
716
+ """
717
+ adjacent = [(present[i], present[i + 1], True) for i in range(len(present) - 1)]
718
+ distant = [
719
+ (present[i], present[j], False)
720
+ for i in range(len(present))
721
+ for j in range(i + 2, len(present))
722
+ ]
723
+ return adjacent + distant
724
+
725
+
421
726
  def _extract_triples_rules(
422
727
  text: str,
423
728
  concepts: List[str],
424
729
  limit: int = 20,
425
- ) -> List[Dict[str, str]]:
426
- """Extract (subject, verb-edge, object, context) triples from text (rule-based).
427
-
428
- For each sentence containing ≥2 concepts, infer the verb-form edge label
429
- from surrounding context and create a directed triple.
730
+ ) -> List[Dict[str, Any]]:
731
+ """Extract (subject, typed-edge, object, context) triples (rule-based).
732
+
733
+ Per sentence: try the four syntactic patterns in :mod:`.patterns` first —
734
+ they read *where* each concept sits and what stands between the two, so
735
+ ``A는 B를 사용한다`` and ``B is used by A`` agree on the subject — and fall
736
+ back to the sentence-level co-occurrence classification for pairs no
737
+ pattern claims. The context each triple carries names the markdown section
738
+ the sentence came from, so a reader can see where the fact lives.
430
739
  """
431
740
  if len(concepts) < 2:
432
741
  return []
433
742
 
434
- concept_lower = {c.lower(): c for c in concepts}
435
- triples: List[Dict[str, str]] = []
743
+ spans = heading_spans(text)
744
+ triples: List[Dict[str, Any]] = []
436
745
  seen_pairs: set = set()
437
746
 
438
- # Split on sentence boundaries
439
- sentences = re.split(r"(?<=[.!?\n])\s+|\n{2,}", text)
440
- for sent in sentences:
441
- sent = sent.strip()
747
+ for offset, raw_sentence in sentence_offsets(text, _SENTENCE_SPLIT):
748
+ sent = raw_sentence.strip()
442
749
  if len(sent) < 8:
443
750
  continue
444
- sent_lower = sent.lower()
751
+ section = heading_at(spans, offset + raw_sentence.index(sent))
445
752
 
446
- present = [concept_lower[k] for k in concept_lower if k in sent_lower]
753
+ present = _drop_nested(concept_positions(sent, concepts))
447
754
  if len(present) < 2:
448
755
  continue
449
756
 
450
757
  relation = infer_edge_relation(sent)
451
- edge = relation["relation"]
452
758
  # Enumeration guard (review 2026-07-27 P1 #6): a verb-less sentence
453
759
  # listing many concepts is a list, not a set of relations. Verb-backed
454
- # sentences keep every pair — the verb is the evidence.
455
- if (
760
+ # sentences keep every pair — the verb is the evidence. A *pattern*
761
+ # match is evidence too, so the guard only gates the fallback.
762
+ enumeration = (
456
763
  relation["evidence"] == "cooccurrence"
457
764
  and len(present) > COOCCURRENCE_CONCEPT_LIMIT
458
- ):
459
- continue
460
-
461
- for i in range(len(present) - 1):
462
- subj, obj = present[i], present[i + 1]
463
- # Deduplicate by (subj, obj) regardless of direction for same edge
464
- pair_key = tuple(sorted([subj.lower(), obj.lower()])) + (edge,)
465
- if pair_key in seen_pairs:
765
+ )
766
+
767
+ # Two passes, patterns first. A pattern names *which* concept is the
768
+ # subject; the fallback only knows text order, so once a pattern has
769
+ # spoken for a concept the coarse pairing around it is noise —
770
+ # ``A는 B가 아니라 C를 쓴다`` must not also emit "A 사용함 B".
771
+ pending: List[Any] = []
772
+ claimed: set = set()
773
+ for left, right, adjacent in _candidate_pairs(present):
774
+ triple = typed_relation(sent, left, right, adjacent)
775
+ if triple is None:
776
+ if adjacent:
777
+ pending.append((left, right))
466
778
  continue
467
- seen_pairs.add(pair_key)
468
- triples.append(
469
- {
470
- "subject": subj,
471
- "relation": edge, # verb form (동사)
472
- "object": obj,
779
+ claimed.add(left[1])
780
+ claimed.add(right[1])
781
+ if not _emit(triples, seen_pairs, triple, section, limit):
782
+ return triples[:limit]
783
+
784
+ if not enumeration:
785
+ for left, right in pending:
786
+ if left[1] in claimed or right[1] in claimed:
787
+ continue
788
+ coarse = {
789
+ "subject": left[1],
790
+ "relation": relation["relation"], # verb form (동사)
791
+ "object": right[1],
473
792
  "context": sent[:240],
474
793
  "evidence": relation["evidence"],
475
794
  "weight": relation["weight"],
476
795
  }
477
- )
478
- if len(triples) >= limit:
479
- return triples
796
+ if not _emit(triples, seen_pairs, coarse, section, limit):
797
+ return triples[:limit]
480
798
 
481
799
  return triples
482
800
 
483
801
 
484
- def _semantic_items(text: str) -> List[Dict[str, str]]:
485
- """Extract explicit decision / task items from text."""
486
- items: List[Dict[str, str]] = []
487
- for raw_line in str(text or "").splitlines():
802
+ def _emit(
803
+ triples: List[Dict[str, Any]],
804
+ seen_pairs: set,
805
+ triple: Dict[str, Any],
806
+ section: str,
807
+ limit: int,
808
+ ) -> bool:
809
+ """Append ``triple`` unless its pair is already there. False ⇒ limit hit."""
810
+ # Deduplicate by (subj, obj) regardless of direction for same edge
811
+ pair_key = tuple(
812
+ sorted([str(triple["subject"]).lower(), str(triple["object"]).lower()])
813
+ ) + (triple["relation"],)
814
+ if pair_key in seen_pairs:
815
+ return True
816
+ seen_pairs.add(pair_key)
817
+ triple["context"] = with_section(str(triple["context"]), section)[:240]
818
+ triples.append(triple)
819
+ return len(triples) < limit
820
+
821
+
822
+ def _semantic_items(text: str) -> List[Dict[str, Any]]:
823
+ """Extract explicit decision / task items from text.
824
+
825
+ Each item carries the markdown section its line sat in (when there is one)
826
+ so the Task or Decision node knows where in the document it was decided —
827
+ the ingest doors store the whole item dict in ``nodes.raw_json``.
828
+ """
829
+ body = str(text or "")
830
+ spans = heading_spans(body)
831
+ items: List[Dict[str, Any]] = []
832
+ cursor = 0
833
+ for raw_line in body.splitlines():
834
+ section = heading_at(spans, cursor)
835
+ cursor += len(raw_line) + 1
488
836
  line = _clean_text(raw_line)
489
837
  if len(line) < 6:
490
838
  continue
491
839
  lowered = line.lower()
492
- if re.search(r"(결정|확정|하기로|decided|decision)", lowered):
493
- items.append(
494
- {"type": "Decision", "title": line[:120], "summary": line[:500]}
495
- )
496
- if re.search(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])", lowered):
497
- items.append({"type": "Task", "title": line[:120], "summary": line[:500]})
840
+ for kind, pattern in (
841
+ ("Decision", _RE_DECISION),
842
+ ("Task", _RE_TASK),
843
+ ):
844
+ if not pattern.search(lowered):
845
+ continue
846
+ item: Dict[str, Any] = {
847
+ "type": kind,
848
+ "title": line[:120],
849
+ "summary": line[:500],
850
+ }
851
+ if section:
852
+ item["section"] = section
853
+ items.append(item)
498
854
  return items[:8]
499
855
 
500
856
 
@@ -504,9 +860,7 @@ def _topic_candidates(text: str, limit: int = 8) -> List[str]:
504
860
  if candidates:
505
861
  return candidates[:limit]
506
862
  seen: Dict[str, str] = {}
507
- for token in re.findall(
508
- r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}", str(text or "")
509
- ):
863
+ for token in _RE_TOPIC_TOKEN.findall(str(text or "")):
510
864
  key = token.lower()
511
865
  if key in _CONCEPT_STOP or key.isdigit():
512
866
  continue