ltcai 11.9.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +80 -59
- package/docs/CHANGELOG.md +92 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +267 -119
- package/docs/ENTERPRISE.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +13 -3
- package/docs/OPERATIONS.md +13 -4
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +16 -2
- package/docs/WHY_LATTICE.md +8 -2
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +51 -5
- package/docs/mcp-tools.md +17 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +32 -0
- package/latticeai/api/worker_compute.py +78 -3
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/generation.py +101 -22
- package/latticeai/models/router/loading.py +109 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/worker_profile.py +13 -4
- package/latticeai/services/architecture_readiness.py +2 -2
- package/latticeai/services/product_readiness.py +10 -5
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/tools/__init__.py +6 -1
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/markup.py +152 -0
- package/package.json +2 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/compose_openapi.py +2 -0
- package/scripts/openapi_route_families.json +7 -3
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +12 -0
- package/src-tauri/Cargo.lock +44 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/ReviewCard-CEHG6evf.js +3 -0
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Bc3Q27YR.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-BGMiV8UU.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-DoanLHnd.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DwzNf82m.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-Ddw49yCV.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-Brd6Kvto.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-Bu-DTJdB.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/{index-CGdg_aq9.css → index-CB93CZWW.css} +1 -1
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-DZ4OA5tJ.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-D9TR0F8b.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-BC5FirFv.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DUbR2W2s.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/static/app/assets/Act-B4WT81kh.js +0 -1
- package/static/app/assets/AdminConsole--Wf71m-o.js +0 -1
- package/static/app/assets/Brain-CtYaa26c.js +0 -321
- package/static/app/assets/BrainHome-Be2VPJEc.js +0 -2
- package/static/app/assets/BrainSignals-C0__xgpG.js +0 -1
- package/static/app/assets/Capture-DdNi5Peb.js +0 -1
- package/static/app/assets/Chronicle-B3hNveeI.js +0 -1
- package/static/app/assets/CommandPalette-CIsnSsFL.js +0 -1
- package/static/app/assets/Library-Bl5XClFV.js +0 -1
- package/static/app/assets/LivingBrain-DjkTB_gh.js +0 -1
- package/static/app/assets/ProductFlow-DCUNRNHs.js +0 -1
- package/static/app/assets/ReviewCard-CD3yWvUB.js +0 -3
- package/static/app/assets/System-BJ6jQ_SL.js +0 -1
- package/static/app/assets/arrow-left-6_28Z0qH.js +0 -1
- package/static/app/assets/brain-B9BDrMTe.js +0 -1
- package/static/app/assets/button-D6JcpYcf.js +0 -1
- package/static/app/assets/circle-check-BFu9lD-3.js +0 -1
- package/static/app/assets/index-CWKRRsLW.js +0 -10
- package/static/app/assets/input-CEqsxtil.js +0 -1
- package/static/app/assets/primitives-DORg7Z_7.js +0 -1
- package/static/app/assets/search-0NQ21wXe.js +0 -1
- package/static/app/assets/textarea-jtQcRSXo.js +0 -1
- package/static/app/assets/useFocusTrap-HRemcWId.js +0 -1
- package/static/app/assets/useMutation-DqlFE-Bw.js +0 -1
- package/static/app/assets/useQuery-BizqBNGw.js +0 -1
- package/static/app/assets/utils-WgW4V69R.js +0 -4
- package/static/app/assets/workspace-DSek3jCY.js +0 -1
|
@@ -15,6 +15,7 @@ from __future__ import annotations
|
|
|
15
15
|
# under the module-wide directive `_kg_common.py` carried before the split.
|
|
16
16
|
# ruff: noqa: F841
|
|
17
17
|
import asyncio
|
|
18
|
+
import functools
|
|
18
19
|
import json
|
|
19
20
|
import logging
|
|
20
21
|
import os
|
|
@@ -22,12 +23,15 @@ import re
|
|
|
22
23
|
from typing import Any, Dict, List, Optional
|
|
23
24
|
|
|
24
25
|
from ..runtime import get_llm_router
|
|
26
|
+
from .normalize import merge_entity_aliases
|
|
27
|
+
from .patterns import concept_positions, typed_relation
|
|
25
28
|
from .relations import (
|
|
26
29
|
COOCCURRENCE_CONCEPT_LIMIT,
|
|
27
30
|
COOCCURRENCE_EDGE_WEIGHT,
|
|
28
31
|
VERB_EDGE_WEIGHT,
|
|
29
32
|
infer_edge_relation,
|
|
30
33
|
)
|
|
34
|
+
from .sections import heading_at, heading_spans, sentence_offsets, with_section
|
|
31
35
|
from .text import _clean_text
|
|
32
36
|
|
|
33
37
|
_LLM_EXTRACT_CONCEPT_PROMPT = """Extract the key concepts from the following text.
|
|
@@ -288,13 +292,238 @@ _CONCEPT_STOP: set = {
|
|
|
288
292
|
|
|
289
293
|
|
|
290
294
|
def _extract_concepts(text: str, limit: int = 12) -> List[str]:
|
|
291
|
-
"""LLM-first concept extraction with rule-based fallback.
|
|
295
|
+
"""LLM-first concept extraction with rule-based fallback.
|
|
296
|
+
|
|
297
|
+
Both paths are pushed through :func:`merge_entity_aliases` before they
|
|
298
|
+
leave, so ``"Lattice AI"`` / ``"lattice ai"`` / ``"지식그래프에서"`` become
|
|
299
|
+
one concept and therefore one node. The merge is deterministic and needs no
|
|
300
|
+
model — the LLM is free to be inconsistent about case and 조사, and the
|
|
301
|
+
graph still gets one entity.
|
|
302
|
+
"""
|
|
292
303
|
llm_result = _llm_extract_concepts(text, limit)
|
|
293
304
|
if llm_result:
|
|
294
|
-
return llm_result
|
|
305
|
+
return merge_entity_aliases(llm_result, text)[:limit]
|
|
295
306
|
return _extract_concepts_rules(text, limit)
|
|
296
307
|
|
|
297
308
|
|
|
309
|
+
#: Longest a concept may be, in words. A name is a name; five words is already
|
|
310
|
+
#: generous for "OpenAI GPT-4o Mini Preview". Past that it is a clause.
|
|
311
|
+
_MAX_CONCEPT_WORDS = 5
|
|
312
|
+
|
|
313
|
+
# Precompiled once: each of these used to be a ``re.findall`` literal inside
|
|
314
|
+
# ``_extract_concepts_rules``, which recompiled them on every document.
|
|
315
|
+
_RE_BACKTICK = re.compile(r"`([^`]{2,40})`")
|
|
316
|
+
_RE_CODE_PUNCT = re.compile(r"[\(\)\[\]{}]")
|
|
317
|
+
_RE_DQUOTE = re.compile(r'"([^"]{2,40})"')
|
|
318
|
+
_RE_PROPER_MIXED = re.compile(
|
|
319
|
+
r"([A-Z][a-z]{1,20}(?:[ \t]+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})"
|
|
320
|
+
)
|
|
321
|
+
_RE_PROPER_CAPS = re.compile(
|
|
322
|
+
r"([A-Z]{2,6}(?:[ \t]+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})"
|
|
323
|
+
)
|
|
324
|
+
_RE_PROPER_SINGLE = re.compile(
|
|
325
|
+
r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])"
|
|
326
|
+
)
|
|
327
|
+
_RE_SENTENCE_START = re.compile(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)")
|
|
328
|
+
_RE_KO_COMPOUND = re.compile(
|
|
329
|
+
r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)"
|
|
330
|
+
)
|
|
331
|
+
_RE_KO_PARTICLE = re.compile(
|
|
332
|
+
r"([가-힣]{2,12})"
|
|
333
|
+
r"(?:에서|에게|으로|부터|까지|보다|처럼|한테|은|는|이|가|을|를|의|와|과|에|도|로|만)"
|
|
334
|
+
)
|
|
335
|
+
_RE_KO_DEFINITION = re.compile(r"([가-힣]{2,12}?)(?:이란|란)(?![가-힣])")
|
|
336
|
+
_RE_HYPHEN_ID = re.compile(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b")
|
|
337
|
+
_RE_DECISION = re.compile(r"(결정|확정|하기로|decided|decision)")
|
|
338
|
+
_RE_TASK = re.compile(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])")
|
|
339
|
+
_RE_TOPIC_TOKEN = re.compile(r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}")
|
|
340
|
+
|
|
341
|
+
#: A capitalized sentence opener is not part of the name that follows it.
|
|
342
|
+
#: "The Vector Index" and "Unlike Keyword Search" are the sentence's grammar,
|
|
343
|
+
#: not the entity's; the lowercase forms already sit in :data:`_CONCEPT_STOP`,
|
|
344
|
+
#: and these are the ones that also appear capitalized at a sentence start.
|
|
345
|
+
_LEADING_STOPWORDS: frozenset = frozenset(
|
|
346
|
+
{
|
|
347
|
+
"a",
|
|
348
|
+
"an",
|
|
349
|
+
"the",
|
|
350
|
+
"this",
|
|
351
|
+
"that",
|
|
352
|
+
"these",
|
|
353
|
+
"those",
|
|
354
|
+
"and",
|
|
355
|
+
"but",
|
|
356
|
+
"for",
|
|
357
|
+
"with",
|
|
358
|
+
"without",
|
|
359
|
+
"in",
|
|
360
|
+
"on",
|
|
361
|
+
"at",
|
|
362
|
+
"by",
|
|
363
|
+
"from",
|
|
364
|
+
"into",
|
|
365
|
+
"unlike",
|
|
366
|
+
"instead",
|
|
367
|
+
"our",
|
|
368
|
+
"their",
|
|
369
|
+
"its",
|
|
370
|
+
"some",
|
|
371
|
+
"each",
|
|
372
|
+
"every",
|
|
373
|
+
"both",
|
|
374
|
+
"when",
|
|
375
|
+
"while",
|
|
376
|
+
"where",
|
|
377
|
+
"if",
|
|
378
|
+
"as",
|
|
379
|
+
"than",
|
|
380
|
+
"then",
|
|
381
|
+
"so",
|
|
382
|
+
"also",
|
|
383
|
+
"now",
|
|
384
|
+
"here",
|
|
385
|
+
"there",
|
|
386
|
+
"no",
|
|
387
|
+
"not",
|
|
388
|
+
"only",
|
|
389
|
+
"just",
|
|
390
|
+
}
|
|
391
|
+
)
|
|
392
|
+
|
|
393
|
+
#: The character class a term's own edge must not touch. A Latin word ends at
|
|
394
|
+
#: the next Latin letter, **not** at the Korean particle glued to it: with a
|
|
395
|
+
#: blanket `\w` guard, `"Lattice AI"` never matches inside `"Lattice AI는"` and
|
|
396
|
+
#: the containment de-duplication silently keeps `"Lattice"` as a second node.
|
|
397
|
+
_ASCII_EDGE = "A-Za-z0-9_"
|
|
398
|
+
_HANGUL_EDGE = "가-힣"
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
def _edge_class(char: str) -> str:
|
|
402
|
+
"""Which character class would continue the word ``char`` ends (or starts)."""
|
|
403
|
+
return _HANGUL_EDGE if "가" <= char <= "힣" else _ASCII_EDGE
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def _in_edge_class(char: str, edge: str) -> bool:
|
|
407
|
+
"""True when ``char`` would continue a word whose edge class is ``edge``."""
|
|
408
|
+
if edge is _HANGUL_EDGE:
|
|
409
|
+
return "가" <= char <= "힣"
|
|
410
|
+
return (
|
|
411
|
+
("A" <= char <= "Z")
|
|
412
|
+
or ("a" <= char <= "z")
|
|
413
|
+
or ("0" <= char <= "9")
|
|
414
|
+
or char == "_"
|
|
415
|
+
)
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _fold_len_stable(term: str) -> bool:
|
|
419
|
+
"""True when case-folding ``term`` cannot move character offsets.
|
|
420
|
+
|
|
421
|
+
``re.IGNORECASE`` and ``str.lower()`` agree on span starts for these
|
|
422
|
+
terms; anything else (``İ``, ``ß``) falls back to the compiled regex so
|
|
423
|
+
the match list stays byte-identical.
|
|
424
|
+
"""
|
|
425
|
+
return len(term) == len(term.lower()) == len(term.upper())
|
|
426
|
+
|
|
427
|
+
|
|
428
|
+
@functools.lru_cache(maxsize=4096)
|
|
429
|
+
def _whole_word_re(term: str) -> re.Pattern[str]:
|
|
430
|
+
"""The compiled whole-word pattern for ``term`` (cached)."""
|
|
431
|
+
return re.compile(
|
|
432
|
+
f"(?<![{_edge_class(term[0])}])"
|
|
433
|
+
+ re.escape(term)
|
|
434
|
+
+ f"(?![{_edge_class(term[-1])}])",
|
|
435
|
+
re.IGNORECASE,
|
|
436
|
+
)
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
def _drop_leading_stopwords(phrase: str) -> str:
|
|
440
|
+
"""``phrase`` without the sentence-grammar words in front of the name.
|
|
441
|
+
|
|
442
|
+
Stops before eating the whole phrase: a two-word match that is *entirely*
|
|
443
|
+
stopwords is returned unchanged and refused later by ``_add``.
|
|
444
|
+
"""
|
|
445
|
+
words = phrase.split()
|
|
446
|
+
while len(words) > 1 and words[0].lower() in _LEADING_STOPWORDS:
|
|
447
|
+
words = words[1:]
|
|
448
|
+
return " ".join(words)
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def _whole_word_spans(
|
|
452
|
+
term: str,
|
|
453
|
+
haystack: str,
|
|
454
|
+
haystack_lower: Optional[str] = None,
|
|
455
|
+
) -> List[Any]:
|
|
456
|
+
"""``(start, end)`` for every whole-word occurrence of ``term``.
|
|
457
|
+
|
|
458
|
+
On case-fold-stable terms this is a ``str.find`` walk with the same
|
|
459
|
+
lookaround rules as the compiled regex; anything whose lower/upper
|
|
460
|
+
length differs (or an empty term) uses the regex so the span list
|
|
461
|
+
cannot drift.
|
|
462
|
+
"""
|
|
463
|
+
if not term:
|
|
464
|
+
return []
|
|
465
|
+
if _fold_len_stable(term):
|
|
466
|
+
return _spans_by_find(term, haystack, haystack_lower)
|
|
467
|
+
return [m.span() for m in _whole_word_re(term).finditer(haystack)]
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
def _spans_by_find(
|
|
471
|
+
term: str,
|
|
472
|
+
haystack: str,
|
|
473
|
+
haystack_lower: Optional[str] = None,
|
|
474
|
+
) -> List[Any]:
|
|
475
|
+
"""Whole-word spans via case-folded ``find`` — identical to the regex."""
|
|
476
|
+
needle = term.lower()
|
|
477
|
+
hay = haystack_lower if haystack_lower is not None else haystack.lower()
|
|
478
|
+
left_edge = _edge_class(term[0])
|
|
479
|
+
right_edge = _edge_class(term[-1])
|
|
480
|
+
width = len(needle)
|
|
481
|
+
spans: List[Any] = []
|
|
482
|
+
start = 0
|
|
483
|
+
while True:
|
|
484
|
+
idx = hay.find(needle, start)
|
|
485
|
+
if idx < 0:
|
|
486
|
+
return spans
|
|
487
|
+
end = idx + width
|
|
488
|
+
if (idx == 0 or not _in_edge_class(haystack[idx - 1], left_edge)) and (
|
|
489
|
+
end >= len(haystack) or not _in_edge_class(haystack[end], right_edge)
|
|
490
|
+
):
|
|
491
|
+
spans.append((idx, end))
|
|
492
|
+
start = idx + 1
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def _term_is_whole_word_in(short: str, longer: str) -> bool:
|
|
496
|
+
"""True when ``short`` occurs as a whole word inside ``longer``."""
|
|
497
|
+
if not short:
|
|
498
|
+
return False
|
|
499
|
+
if _fold_len_stable(short):
|
|
500
|
+
return bool(_spans_by_find(short, longer))
|
|
501
|
+
return _whole_word_re(short).search(longer) is not None
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def _always_inside(short: str, longer: List[str], spans: Dict[str, List[Any]]) -> bool:
|
|
505
|
+
"""True when every occurrence of ``short`` sits inside a longer concept.
|
|
506
|
+
|
|
507
|
+
Span containment rather than a per-term count, because a short term can be
|
|
508
|
+
covered by *different* longer terms in different places — "Vector" is
|
|
509
|
+
inside "Vector Index" here and "Vector Search" there, and counting against
|
|
510
|
+
one long term at a time would keep it as a third, meaningless node.
|
|
511
|
+
|
|
512
|
+
``spans`` is precomputed once per document by the caller. Scanning the text
|
|
513
|
+
again per (short, long) pair is quadratic in the *candidate count* times
|
|
514
|
+
linear in the document, which on a 100 KB file took minutes; the same
|
|
515
|
+
answer from cached spans is interval arithmetic over a few dozen tuples.
|
|
516
|
+
"""
|
|
517
|
+
hits = spans.get(short) or []
|
|
518
|
+
if not hits:
|
|
519
|
+
return False
|
|
520
|
+
covers = [span for term in longer for span in spans.get(term) or []]
|
|
521
|
+
return all(
|
|
522
|
+
any(start >= c_start and end <= c_end for c_start, c_end in covers)
|
|
523
|
+
for start, end in hits
|
|
524
|
+
)
|
|
525
|
+
|
|
526
|
+
|
|
298
527
|
def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
|
|
299
528
|
"""Extract meaningful named concepts from text (rule-based).
|
|
300
529
|
|
|
@@ -309,43 +538,44 @@ def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
|
|
|
309
538
|
seen: dict = {} # concept_lower → original form
|
|
310
539
|
|
|
311
540
|
def _add(term: str) -> None:
|
|
541
|
+
# A candidate that crosses a line, or runs past a handful of words, is
|
|
542
|
+
# a quoted *passage* the backtick and quote rules swept up — the graph
|
|
543
|
+
# was storing whole sentences as entity names.
|
|
544
|
+
if "\n" in term or len(term.split()) > _MAX_CONCEPT_WORDS:
|
|
545
|
+
return
|
|
312
546
|
key = term.strip().lower()
|
|
313
547
|
if key and key not in _CONCEPT_STOP and not key.isdigit() and len(key) >= 2:
|
|
314
548
|
seen.setdefault(key, term.strip())
|
|
315
549
|
|
|
316
550
|
# 1. Backtick-quoted code/term (highest confidence)
|
|
317
|
-
for m in
|
|
318
|
-
if not
|
|
551
|
+
for m in _RE_BACKTICK.findall(text):
|
|
552
|
+
if not _RE_CODE_PUNCT.search(m): # skip code expressions
|
|
319
553
|
_add(m)
|
|
320
554
|
|
|
321
555
|
# 2. Double/single quoted terms
|
|
322
|
-
for m in
|
|
556
|
+
for m in _RE_DQUOTE.findall(text):
|
|
323
557
|
_add(m)
|
|
324
558
|
|
|
325
559
|
# 3. Multi-word English proper nouns (Title Case or ALL-CAPS first word, 2–4 words).
|
|
560
|
+
# The inner separator is `[ \t]+`, not `\s+`: `\s` crosses newlines, so a
|
|
561
|
+
# heading glued to the next line's first word produced phantom concepts
|
|
562
|
+
# like "Retrieval Lattice AI" — and, worse, consumed the "Lattice AI"
|
|
563
|
+
# the next pass would have found (regex scanning does not overlap).
|
|
326
564
|
# Pattern A: Mixed-case first word — "Lattice AI", "Tool Use", "Graph RAG"
|
|
327
|
-
for m in
|
|
328
|
-
|
|
329
|
-
text,
|
|
330
|
-
):
|
|
331
|
-
_add(m)
|
|
565
|
+
for m in _RE_PROPER_MIXED.findall(text):
|
|
566
|
+
_add(_drop_leading_stopwords(m))
|
|
332
567
|
# Pattern B: ALL-CAPS first word — "VS Code", "MCP Server", "GPT-4o Mini"
|
|
333
|
-
for m in
|
|
334
|
-
|
|
335
|
-
text,
|
|
336
|
-
):
|
|
337
|
-
_add(m)
|
|
568
|
+
for m in _RE_PROPER_CAPS.findall(text):
|
|
569
|
+
_add(_drop_leading_stopwords(m))
|
|
338
570
|
|
|
339
571
|
# 4. Single capitalized proper noun.
|
|
340
572
|
# Use ASCII-boundary lookaround instead of \b so Korean particles
|
|
341
573
|
# (와, 의, 는 …) after an English word don't block the match.
|
|
342
|
-
all_caps_words =
|
|
343
|
-
r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])", text
|
|
344
|
-
)
|
|
574
|
+
all_caps_words = _RE_PROPER_SINGLE.findall(text)
|
|
345
575
|
freq: Dict[str, int] = {}
|
|
346
576
|
for w in all_caps_words:
|
|
347
577
|
freq[w] = freq.get(w, 0) + 1
|
|
348
|
-
sentence_starts = set(
|
|
578
|
+
sentence_starts = set(_RE_SENTENCE_START.findall(text))
|
|
349
579
|
for m, cnt in freq.items():
|
|
350
580
|
if m.lower() in _CONCEPT_STOP:
|
|
351
581
|
continue
|
|
@@ -353,64 +583,102 @@ def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
|
|
|
353
583
|
_add(m)
|
|
354
584
|
|
|
355
585
|
# 5. Korean technical compound nouns (3–12 chars, no common particles)
|
|
356
|
-
for m in
|
|
357
|
-
r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)",
|
|
358
|
-
text,
|
|
359
|
-
):
|
|
586
|
+
for m in _RE_KO_COMPOUND.findall(text):
|
|
360
587
|
_add(m)
|
|
361
|
-
# Korean standalone terms that appear
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
588
|
+
# Korean standalone terms that appear before a particle. Multi-syllable
|
|
589
|
+
# particles come first in the alternation: the group is greedy, so
|
|
590
|
+
# "지식그래프에서" has to be offered 에서 before 에 or the leftover 서
|
|
591
|
+
# blocks the match entirely.
|
|
592
|
+
ko_stems = _RE_KO_PARTICLE.findall(text)
|
|
593
|
+
ko_counts: Dict[str, int] = {}
|
|
594
|
+
for m in ko_stems:
|
|
365
595
|
if m.lower() not in _CONCEPT_STOP and len(m) >= 2:
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
if len(m) >= 3 or
|
|
596
|
+
if m not in ko_counts:
|
|
597
|
+
ko_counts[m] = text.count(m)
|
|
598
|
+
if len(m) >= 3 or ko_counts[m] >= 2:
|
|
369
599
|
_add(m)
|
|
370
600
|
|
|
601
|
+
# 5b. The term a Korean definition sentence is *about*: `지식그래프란 …이다`.
|
|
602
|
+
# The stem is lazy and the marker must end the word, so `설명이란` keeps
|
|
603
|
+
# `설명` (the `이` belongs to `이란`) and `의견란에` is left alone
|
|
604
|
+
# entirely (`란` there is part of the noun, not a marker).
|
|
605
|
+
for m in _RE_KO_DEFINITION.findall(text):
|
|
606
|
+
_add(m)
|
|
607
|
+
|
|
371
608
|
# 6. Hyphenated / versioned identifiers (gpt-4o, gemma-4, mlx-vlm)
|
|
372
|
-
for m in
|
|
609
|
+
for m in _RE_HYPHEN_ID.findall(text):
|
|
373
610
|
if len(m) >= 4:
|
|
374
611
|
_add(m)
|
|
375
612
|
|
|
376
|
-
# De-duplicate
|
|
377
|
-
#
|
|
378
|
-
#
|
|
379
|
-
#
|
|
613
|
+
# De-duplicate by containment: drop the shorter term when every occurrence
|
|
614
|
+
# of it in the source text is *inside* a longer concept.
|
|
615
|
+
#
|
|
616
|
+
# "Lattice" → dropped, every occurrence is "Lattice AI"
|
|
617
|
+
# "RAG" → dropped, every occurrence is "Graph RAG" (new: suffixes)
|
|
618
|
+
# "Claude" → kept, it also appears on its own
|
|
619
|
+
# "Lat" → kept, "Lattice AI" does not contain it as a whole word
|
|
620
|
+
#
|
|
621
|
+
# Until v12.0.0 this only looked at *prefixes*, so "Graph RAG" and "RAG"
|
|
622
|
+
# both became nodes and the graph answered one question with two entities.
|
|
380
623
|
values = list(seen.values())
|
|
381
|
-
|
|
624
|
+
# Which candidates *could* swallow which is decided on the candidate
|
|
625
|
+
# strings alone — matching "Vector" inside "Vector Index" is a few
|
|
626
|
+
# characters of work. Only the terms that survive that filter are then
|
|
627
|
+
# searched for in the document, so a 100 KB file is scanned a handful of
|
|
628
|
+
# times instead of once per candidate pair.
|
|
629
|
+
# The pair filter is a plain substring test before it is a regex one. A
|
|
630
|
+
# long document yields well over a thousand raw candidates, and a regex
|
|
631
|
+
# compiled per *pair* is two million compilations — seconds of work to
|
|
632
|
+
# answer a question `in` answers in nanoseconds for all but a few pairs.
|
|
633
|
+
lowered = [value.lower() for value in values]
|
|
634
|
+
lengths = [len(value) for value in values]
|
|
635
|
+
swallowers: Dict[int, List[int]] = {}
|
|
636
|
+
for i, short in enumerate(values):
|
|
637
|
+
short_l = lowered[i]
|
|
638
|
+
short_n = lengths[i]
|
|
639
|
+
longer = [
|
|
640
|
+
j
|
|
641
|
+
for j, other in enumerate(values)
|
|
642
|
+
if lengths[j] > short_n
|
|
643
|
+
and short_l in lowered[j]
|
|
644
|
+
and _term_is_whole_word_in(short, other)
|
|
645
|
+
]
|
|
646
|
+
if longer:
|
|
647
|
+
swallowers[i] = longer
|
|
648
|
+
wanted = {
|
|
649
|
+
values[index]
|
|
650
|
+
for i, group in swallowers.items()
|
|
651
|
+
for index in [i, *group]
|
|
652
|
+
}
|
|
653
|
+
text_lower = text.lower() if wanted else ""
|
|
654
|
+
spans = {
|
|
655
|
+
term: _whole_word_spans(term, text, text_lower) for term in wanted
|
|
656
|
+
}
|
|
657
|
+
|
|
382
658
|
keep = set(range(len(values)))
|
|
383
|
-
for i,
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
|
|
387
|
-
continue
|
|
388
|
-
# Check if vl is a word-prefix of wl
|
|
389
|
-
suffix = wl[len(vl) :]
|
|
390
|
-
if not (wl.startswith(vl) and re.match(r"^[\s\-]", suffix)):
|
|
391
|
-
continue
|
|
392
|
-
# Count occurrences of v NOT followed by the suffix
|
|
393
|
-
suffix_stripped = suffix.lstrip(" -")
|
|
394
|
-
# Escape for regex
|
|
395
|
-
pattern_with_suffix = re.escape(v) + r"[\s\-]+" + re.escape(suffix_stripped)
|
|
396
|
-
pattern_alone = (
|
|
397
|
-
re.escape(v) + r"(?![\s\-]*" + re.escape(suffix_stripped) + r")"
|
|
398
|
-
)
|
|
399
|
-
alone_count = len(re.findall(pattern_alone, text, re.IGNORECASE))
|
|
400
|
-
if alone_count == 0:
|
|
401
|
-
# Shorter term never appears alone → safe to remove
|
|
402
|
-
keep.discard(i)
|
|
403
|
-
break
|
|
659
|
+
for i, longer in swallowers.items():
|
|
660
|
+
alive = [values[j] for j in longer if j in keep]
|
|
661
|
+
if alive and _always_inside(values[i], alive, spans):
|
|
662
|
+
keep.discard(i)
|
|
404
663
|
|
|
405
664
|
final = [values[i] for i in range(len(values)) if i in keep]
|
|
406
|
-
|
|
665
|
+
# The surface merge is the last word: whatever the six passes above
|
|
666
|
+
# produced, two spellings of one entity leave as one concept and therefore
|
|
667
|
+
# as one node. Applied after the prefix de-duplication so the two rules
|
|
668
|
+
# compose rather than fight.
|
|
669
|
+
return merge_entity_aliases(final, text)[:limit]
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
#: `(?<=[.!?\n])\s+|\n{2,}` — the sentence split, compiled so
|
|
673
|
+
#: :func:`sentence_offsets` can keep each piece's position.
|
|
674
|
+
_SENTENCE_SPLIT = re.compile(r"(?<=[.!?\n])\s+|\n{2,}")
|
|
407
675
|
|
|
408
676
|
|
|
409
677
|
def _extract_triples(
|
|
410
678
|
text: str,
|
|
411
679
|
concepts: List[str],
|
|
412
680
|
limit: int = 20,
|
|
413
|
-
) -> List[Dict[str,
|
|
681
|
+
) -> List[Dict[str, Any]]:
|
|
414
682
|
"""LLM-first triple extraction with rule-based fallback."""
|
|
415
683
|
llm_result = _llm_extract_triples(text, concepts, limit)
|
|
416
684
|
if llm_result:
|
|
@@ -418,83 +686,171 @@ def _extract_triples(
|
|
|
418
686
|
return _extract_triples_rules(text, concepts, limit)
|
|
419
687
|
|
|
420
688
|
|
|
689
|
+
def _drop_nested(present: List[Any]) -> List[Any]:
|
|
690
|
+
"""Drop a concept whose span sits inside another concept's span.
|
|
691
|
+
|
|
692
|
+
``"RAG"`` found *inside* ``"Graph RAG"`` is not a second participant; it is
|
|
693
|
+
the same eight characters counted twice, and pairing them produces an edge
|
|
694
|
+
from a thing to part of its own name.
|
|
695
|
+
"""
|
|
696
|
+
kept: List[Any] = []
|
|
697
|
+
for start, concept in present:
|
|
698
|
+
end = start + len(concept)
|
|
699
|
+
if any(
|
|
700
|
+
other_start <= start and end <= other_start + len(other)
|
|
701
|
+
for other_start, other in present
|
|
702
|
+
if other != concept
|
|
703
|
+
):
|
|
704
|
+
continue
|
|
705
|
+
kept.append((start, concept))
|
|
706
|
+
return kept
|
|
707
|
+
|
|
708
|
+
|
|
709
|
+
def _candidate_pairs(present: List[Any]) -> List[Any]:
|
|
710
|
+
"""``(left, right, adjacent)`` for every pair worth testing, adjacent first.
|
|
711
|
+
|
|
712
|
+
Korean puts the verb last and separates subject from object with whatever
|
|
713
|
+
clause it likes — ``하이브리드검색은 키워드검색이 아니라 벡터검색을 쓴다``
|
|
714
|
+
states a relation between the *first* and the *last* concept, and adjacency
|
|
715
|
+
never sees it.
|
|
716
|
+
"""
|
|
717
|
+
adjacent = [(present[i], present[i + 1], True) for i in range(len(present) - 1)]
|
|
718
|
+
distant = [
|
|
719
|
+
(present[i], present[j], False)
|
|
720
|
+
for i in range(len(present))
|
|
721
|
+
for j in range(i + 2, len(present))
|
|
722
|
+
]
|
|
723
|
+
return adjacent + distant
|
|
724
|
+
|
|
725
|
+
|
|
421
726
|
def _extract_triples_rules(
|
|
422
727
|
text: str,
|
|
423
728
|
concepts: List[str],
|
|
424
729
|
limit: int = 20,
|
|
425
|
-
) -> List[Dict[str,
|
|
426
|
-
"""Extract (subject,
|
|
427
|
-
|
|
428
|
-
|
|
429
|
-
|
|
730
|
+
) -> List[Dict[str, Any]]:
|
|
731
|
+
"""Extract (subject, typed-edge, object, context) triples (rule-based).
|
|
732
|
+
|
|
733
|
+
Per sentence: try the four syntactic patterns in :mod:`.patterns` first —
|
|
734
|
+
they read *where* each concept sits and what stands between the two, so
|
|
735
|
+
``A는 B를 사용한다`` and ``B is used by A`` agree on the subject — and fall
|
|
736
|
+
back to the sentence-level co-occurrence classification for pairs no
|
|
737
|
+
pattern claims. The context each triple carries names the markdown section
|
|
738
|
+
the sentence came from, so a reader can see where the fact lives.
|
|
430
739
|
"""
|
|
431
740
|
if len(concepts) < 2:
|
|
432
741
|
return []
|
|
433
742
|
|
|
434
|
-
|
|
435
|
-
triples: List[Dict[str,
|
|
743
|
+
spans = heading_spans(text)
|
|
744
|
+
triples: List[Dict[str, Any]] = []
|
|
436
745
|
seen_pairs: set = set()
|
|
437
746
|
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
for sent in sentences:
|
|
441
|
-
sent = sent.strip()
|
|
747
|
+
for offset, raw_sentence in sentence_offsets(text, _SENTENCE_SPLIT):
|
|
748
|
+
sent = raw_sentence.strip()
|
|
442
749
|
if len(sent) < 8:
|
|
443
750
|
continue
|
|
444
|
-
|
|
751
|
+
section = heading_at(spans, offset + raw_sentence.index(sent))
|
|
445
752
|
|
|
446
|
-
present =
|
|
753
|
+
present = _drop_nested(concept_positions(sent, concepts))
|
|
447
754
|
if len(present) < 2:
|
|
448
755
|
continue
|
|
449
756
|
|
|
450
757
|
relation = infer_edge_relation(sent)
|
|
451
|
-
edge = relation["relation"]
|
|
452
758
|
# Enumeration guard (review 2026-07-27 P1 #6): a verb-less sentence
|
|
453
759
|
# listing many concepts is a list, not a set of relations. Verb-backed
|
|
454
|
-
# sentences keep every pair — the verb is the evidence.
|
|
455
|
-
|
|
760
|
+
# sentences keep every pair — the verb is the evidence. A *pattern*
|
|
761
|
+
# match is evidence too, so the guard only gates the fallback.
|
|
762
|
+
enumeration = (
|
|
456
763
|
relation["evidence"] == "cooccurrence"
|
|
457
764
|
and len(present) > COOCCURRENCE_CONCEPT_LIMIT
|
|
458
|
-
)
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
765
|
+
)
|
|
766
|
+
|
|
767
|
+
# Two passes, patterns first. A pattern names *which* concept is the
|
|
768
|
+
# subject; the fallback only knows text order, so once a pattern has
|
|
769
|
+
# spoken for a concept the coarse pairing around it is noise —
|
|
770
|
+
# ``A는 B가 아니라 C를 쓴다`` must not also emit "A 사용함 B".
|
|
771
|
+
pending: List[Any] = []
|
|
772
|
+
claimed: set = set()
|
|
773
|
+
for left, right, adjacent in _candidate_pairs(present):
|
|
774
|
+
triple = typed_relation(sent, left, right, adjacent)
|
|
775
|
+
if triple is None:
|
|
776
|
+
if adjacent:
|
|
777
|
+
pending.append((left, right))
|
|
466
778
|
continue
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
|
|
470
|
-
|
|
471
|
-
|
|
472
|
-
|
|
779
|
+
claimed.add(left[1])
|
|
780
|
+
claimed.add(right[1])
|
|
781
|
+
if not _emit(triples, seen_pairs, triple, section, limit):
|
|
782
|
+
return triples[:limit]
|
|
783
|
+
|
|
784
|
+
if not enumeration:
|
|
785
|
+
for left, right in pending:
|
|
786
|
+
if left[1] in claimed or right[1] in claimed:
|
|
787
|
+
continue
|
|
788
|
+
coarse = {
|
|
789
|
+
"subject": left[1],
|
|
790
|
+
"relation": relation["relation"], # verb form (동사)
|
|
791
|
+
"object": right[1],
|
|
473
792
|
"context": sent[:240],
|
|
474
793
|
"evidence": relation["evidence"],
|
|
475
794
|
"weight": relation["weight"],
|
|
476
795
|
}
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
return triples
|
|
796
|
+
if not _emit(triples, seen_pairs, coarse, section, limit):
|
|
797
|
+
return triples[:limit]
|
|
480
798
|
|
|
481
799
|
return triples
|
|
482
800
|
|
|
483
801
|
|
|
484
|
-
def
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
|
|
802
|
+
def _emit(
|
|
803
|
+
triples: List[Dict[str, Any]],
|
|
804
|
+
seen_pairs: set,
|
|
805
|
+
triple: Dict[str, Any],
|
|
806
|
+
section: str,
|
|
807
|
+
limit: int,
|
|
808
|
+
) -> bool:
|
|
809
|
+
"""Append ``triple`` unless its pair is already there. False ⇒ limit hit."""
|
|
810
|
+
# Deduplicate by (subj, obj) regardless of direction for same edge
|
|
811
|
+
pair_key = tuple(
|
|
812
|
+
sorted([str(triple["subject"]).lower(), str(triple["object"]).lower()])
|
|
813
|
+
) + (triple["relation"],)
|
|
814
|
+
if pair_key in seen_pairs:
|
|
815
|
+
return True
|
|
816
|
+
seen_pairs.add(pair_key)
|
|
817
|
+
triple["context"] = with_section(str(triple["context"]), section)[:240]
|
|
818
|
+
triples.append(triple)
|
|
819
|
+
return len(triples) < limit
|
|
820
|
+
|
|
821
|
+
|
|
822
|
+
def _semantic_items(text: str) -> List[Dict[str, Any]]:
|
|
823
|
+
"""Extract explicit decision / task items from text.
|
|
824
|
+
|
|
825
|
+
Each item carries the markdown section its line sat in (when there is one)
|
|
826
|
+
so the Task or Decision node knows where in the document it was decided —
|
|
827
|
+
the ingest doors store the whole item dict in ``nodes.raw_json``.
|
|
828
|
+
"""
|
|
829
|
+
body = str(text or "")
|
|
830
|
+
spans = heading_spans(body)
|
|
831
|
+
items: List[Dict[str, Any]] = []
|
|
832
|
+
cursor = 0
|
|
833
|
+
for raw_line in body.splitlines():
|
|
834
|
+
section = heading_at(spans, cursor)
|
|
835
|
+
cursor += len(raw_line) + 1
|
|
488
836
|
line = _clean_text(raw_line)
|
|
489
837
|
if len(line) < 6:
|
|
490
838
|
continue
|
|
491
839
|
lowered = line.lower()
|
|
492
|
-
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
|
|
496
|
-
|
|
497
|
-
|
|
840
|
+
for kind, pattern in (
|
|
841
|
+
("Decision", _RE_DECISION),
|
|
842
|
+
("Task", _RE_TASK),
|
|
843
|
+
):
|
|
844
|
+
if not pattern.search(lowered):
|
|
845
|
+
continue
|
|
846
|
+
item: Dict[str, Any] = {
|
|
847
|
+
"type": kind,
|
|
848
|
+
"title": line[:120],
|
|
849
|
+
"summary": line[:500],
|
|
850
|
+
}
|
|
851
|
+
if section:
|
|
852
|
+
item["section"] = section
|
|
853
|
+
items.append(item)
|
|
498
854
|
return items[:8]
|
|
499
855
|
|
|
500
856
|
|
|
@@ -504,9 +860,7 @@ def _topic_candidates(text: str, limit: int = 8) -> List[str]:
|
|
|
504
860
|
if candidates:
|
|
505
861
|
return candidates[:limit]
|
|
506
862
|
seen: Dict[str, str] = {}
|
|
507
|
-
for token in
|
|
508
|
-
r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}", str(text or "")
|
|
509
|
-
):
|
|
863
|
+
for token in _RE_TOPIC_TOKEN.findall(str(text or "")):
|
|
510
864
|
key = token.lower()
|
|
511
865
|
if key in _CONCEPT_STOP or key.isdigit():
|
|
512
866
|
continue
|