ltcai 11.7.0 → 12.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +100 -76
- package/docs/BENCHMARKS.md +9 -2
- package/docs/CHANGELOG.md +249 -0
- package/docs/CI_AND_RELEASE_GATES.md +126 -41
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +271 -103
- package/docs/ENTERPRISE.md +1 -1
- package/docs/LEGACY_COMPATIBILITY.md +10 -6
- package/docs/MULTI_AGENT_RUNTIME.md +4 -4
- package/docs/ONBOARDING.md +16 -4
- package/docs/OPERATIONS.md +14 -1
- package/docs/PERMISSION_MODE.md +14 -9
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/ROADMAP.md +113 -0
- package/docs/TRUST_MODEL.md +28 -7
- package/docs/USABILITY_AUDIT.md +5 -0
- package/docs/WHY_LATTICE.md +13 -5
- package/docs/WORKFLOW_DESIGNER.md +2 -2
- package/docs/kg-schema.md +57 -7
- package/docs/mcp-tools.md +93 -82
- package/docs/security-model.md +6 -3
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +1 -54
- package/lattice_brain/graph/_kg_common/extraction.py +459 -105
- package/lattice_brain/graph/_kg_common/normalize.py +305 -0
- package/lattice_brain/graph/_kg_common/patterns.py +275 -0
- package/lattice_brain/graph/_kg_common/relations.py +12 -3
- package/lattice_brain/graph/_kg_common/sections.py +107 -0
- package/lattice_brain/graph/_kg_common/text.py +14 -450
- package/lattice_brain/graph/_kg_constants.py +7 -0
- package/lattice_brain/ingestion/__init__.py +6 -3
- package/lattice_brain/multimodal/__init__.py +9 -3
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +44 -1
- package/latticeai/api/models.py +18 -110
- package/latticeai/api/search.py +7 -30
- package/latticeai/api/worker_compute.py +127 -106
- package/latticeai/api/worker_seams.py +17 -2
- package/latticeai/core/embedding_providers/__init__.py +16 -0
- package/latticeai/core/embedding_providers/autodetect.py +302 -0
- package/latticeai/core/embedding_providers/base.py +25 -0
- package/latticeai/core/embedding_providers/profiles.py +44 -0
- package/latticeai/core/embedding_providers/text.py +74 -8
- package/latticeai/core/http_origin.py +3 -3
- package/latticeai/core/messages.py +0 -5
- package/latticeai/core/policy.py +1 -6
- package/latticeai/core/quiet.py +1 -20
- package/latticeai/core/security.py +29 -83
- package/latticeai/core/sessions.py +95 -4
- package/latticeai/core/users.py +0 -38
- package/latticeai/core/vector_index/__init__.py +61 -0
- package/latticeai/core/vector_index/hnsw.py +383 -0
- package/latticeai/core/vector_index/sidecar.py +329 -0
- package/latticeai/models/router/catalog.py +2 -2
- package/latticeai/models/router/generation.py +176 -30
- package/latticeai/models/router/loading.py +150 -9
- package/latticeai/runtime/access_runtime.py +7 -4
- package/latticeai/runtime/brain_runtime.py +43 -9
- package/latticeai/runtime/build_phases/features.py +8 -31
- package/latticeai/runtime/build_phases/foundation.py +7 -16
- package/latticeai/runtime/build_phases/web.py +3 -3
- package/latticeai/runtime/build_phases/worker_profile.py +29 -27
- package/latticeai/runtime/runtime_context.py +0 -2
- package/latticeai/services/architecture_readiness.py +18 -19
- package/latticeai/services/process_audit.py +1 -22
- package/latticeai/services/product_readiness.py +39 -12
- package/latticeai/services/search_service.py +7 -0
- package/latticeai/services/voice_capture.py +8 -28
- package/latticeai/tools/__init__.py +12 -47
- package/latticeai/tools/commands.py +9 -15
- package/latticeai/tools/documents.py +12 -0
- package/latticeai/tools/knowledge.py +0 -6
- package/latticeai/tools/markup.py +152 -0
- package/package.json +4 -5
- package/requirements.txt +0 -1
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_openapi_drift.mjs +3 -2
- package/scripts/check_server_i18n.mjs +5 -4
- package/scripts/compose_openapi.py +4 -1
- package/scripts/export_openapi.py +5 -4
- package/scripts/gen_worker_allowlist_fixture.py +2 -2
- package/scripts/openapi_route_families.json +19 -74
- package/scripts/publish_release.mjs +157 -0
- package/scripts/release_screen_claims.json +144 -28
- package/src-tauri/Cargo.lock +45 -10
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +47 -41
- package/static/app/assets/Act-Cf1L2709.js +2 -0
- package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
- package/static/app/assets/Brain-DqamGrj-.js +2 -0
- package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
- package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
- package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
- package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
- package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
- package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
- package/static/app/assets/Library-C6xd1dlf.js +1 -0
- package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
- package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
- package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
- package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
- package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
- package/static/app/assets/System-CAxwBUXw.js +1 -0
- package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
- package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
- package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
- package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
- package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
- package/static/app/assets/brain-CLkhHsHF.js +1 -0
- package/static/app/assets/button-CmaEqG1T.js +1 -0
- package/static/app/assets/circle-check-CFgejkOS.js +1 -0
- package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
- package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
- package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
- package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
- package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
- package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
- package/static/app/assets/index-CB93CZWW.css +2 -0
- package/static/app/assets/index-D2H-wSl6.js +13 -0
- package/static/app/assets/input-Df1CAY_I.js +1 -0
- package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
- package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
- package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
- package/static/app/assets/primitives-BioD2slS.js +1 -0
- package/static/app/assets/search-BzBw8YcW.js +1 -0
- package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
- package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
- package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
- package/static/app/assets/textarea-P8o6pvOP.js +1 -0
- package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
- package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
- package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/ingestion/pipeline.py +0 -108
- package/latticeai/api/local_files.py +0 -44
- package/latticeai/api/tools.py +0 -126
- package/latticeai/api/voice_capture.py +0 -32
- package/latticeai/core/agent_permission.py +0 -85
- package/scripts/agent_eval.py +0 -34
- package/scripts/brain_quality_eval.py +0 -37
- package/scripts/check_legacy_debt.mjs +0 -91
- package/scripts/check_python.py +0 -100
- package/scripts/chunking_parity_corpus.py +0 -449
- package/scripts/generate_agent_parity_fixtures.py +0 -771
- package/scripts/generate_chunking_parity_fixtures.py +0 -259
- package/static/app/assets/Act-BPcVAbOL.js +0 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
- package/static/app/assets/Brain-CT92Kos0.js +0 -321
- package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
- package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
- package/static/app/assets/Capture-BsTokYkk.js +0 -1
- package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
- package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
- package/static/app/assets/Library-BGJbG9Hd.js +0 -1
- package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
- package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
- package/static/app/assets/System-CMHSO9qM.js +0 -1
- package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
- package/static/app/assets/brain-CQJberbE.js +0 -1
- package/static/app/assets/button-Ct9f2_oT.js +0 -1
- package/static/app/assets/circle-check-DruOxB-4.js +0 -1
- package/static/app/assets/index-D9x-kSNy.css +0 -2
- package/static/app/assets/index-Do83hDzJ.js +0 -10
- package/static/app/assets/input-BLXVNmj1.js +0 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
- package/static/app/assets/search-CT9aho2j.js +0 -1
- package/static/app/assets/textarea-DqwLnli4.js +0 -1
- package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
- package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
- package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
- package/static/app/assets/utils-CiFtIdZq.js +0 -4
- package/static/app/assets/workspace-DQz9vIId.js +0 -1
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""Entity-surface normalization and alias merge — deterministic, no model.
|
|
2
|
+
|
|
3
|
+
Two concepts that are the same thing must become one node, or the graph
|
|
4
|
+
answers "무엇이 무엇과 이어져 있나" with a fan of near-duplicates. Before
|
|
5
|
+
v12.0.0 the only dedup was ``concept.text.lower()`` inside the Rust writer, so
|
|
6
|
+
``"Lattice AI"`` / ``"lattice ai"`` / ``"Lattice AI의"`` were three nodes.
|
|
7
|
+
|
|
8
|
+
This module is the *surface* half of that fix, and it is deliberately boring:
|
|
9
|
+
Unicode normalization, whitespace collapse, bracket/quote/punctuation trimming,
|
|
10
|
+
English possessives, and Korean 조사 (postposition) stripping. No model, no
|
|
11
|
+
network, no randomness — the same input always produces the same node id.
|
|
12
|
+
|
|
13
|
+
## 조사 stripping, and why it is two tiers
|
|
14
|
+
|
|
15
|
+
Korean marks grammatical role with a suffix glued to the noun, so the *same*
|
|
16
|
+
entity appears as ``플랫폼은`` / ``플랫폼을`` / ``플랫폼에서``. Stripping the
|
|
17
|
+
suffix is what merges them. But Korean nouns also legitimately *end* in those
|
|
18
|
+
syllables — ``고양이`` ends in ``이``, ``전문가`` in ``가``, ``정확도`` in
|
|
19
|
+
``도`` — and a blind strip invents ``고양`` / ``전문`` / ``정확``. That is worse
|
|
20
|
+
than the duplicate it was trying to fix, because a wrong node id cannot be
|
|
21
|
+
undone by a later read.
|
|
22
|
+
|
|
23
|
+
So:
|
|
24
|
+
|
|
25
|
+
* **Tier 1 — unconditional.** Multi-syllable particles (``에서``, ``으로``,
|
|
26
|
+
``에게``, ``부터``, ``까지``, ``보다``, ``처럼`` …) plus ``을``/``를``.
|
|
27
|
+
Korean nouns essentially never end in these *as their own last syllables*
|
|
28
|
+
once a two-character stem is required, so no corroboration is needed.
|
|
29
|
+
* **Tier 2 — evidence-gated.** The single-syllable particles that collide with
|
|
30
|
+
real noun endings (``은 는 이 가 의 와 과 도 로 만 나``) are stripped only
|
|
31
|
+
when the source text itself shows the bare stem somewhere else — that is,
|
|
32
|
+
the text contains the stem *not* followed by this particle. With no text to
|
|
33
|
+
corroborate (an LLM-supplied concept, say) nothing is stripped: an
|
|
34
|
+
unmerged duplicate is recoverable, an invented stem is not.
|
|
35
|
+
|
|
36
|
+
Every stem must keep at least :data:`MIN_STEM_CHARS` characters, which by
|
|
37
|
+
itself rejects the whole ``결과 → 결`` / ``회의 → 회`` class.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
from __future__ import annotations
|
|
41
|
+
|
|
42
|
+
import functools
|
|
43
|
+
import re
|
|
44
|
+
import unicodedata
|
|
45
|
+
from typing import Dict, Iterable, List, Sequence, Tuple
|
|
46
|
+
|
|
47
|
+
#: A stripped stem shorter than this is never accepted — ``결과`` must not
|
|
48
|
+
#: become ``결``. Two characters is the shortest real Korean noun stem.
|
|
49
|
+
MIN_STEM_CHARS = 2
|
|
50
|
+
|
|
51
|
+
#: Particles stripped without asking the text. Longest first: ``에서는`` has to
|
|
52
|
+
#: be tried before ``에서`` or the leftover ``는`` stays glued on.
|
|
53
|
+
UNCONDITIONAL_PARTICLES: Tuple[str, ...] = (
|
|
54
|
+
"에서는",
|
|
55
|
+
"으로는",
|
|
56
|
+
"에게는",
|
|
57
|
+
"로부터",
|
|
58
|
+
"이라는",
|
|
59
|
+
"이라고",
|
|
60
|
+
"에서도",
|
|
61
|
+
"으로도",
|
|
62
|
+
"에서의",
|
|
63
|
+
"으로서",
|
|
64
|
+
"으로써",
|
|
65
|
+
"에게서",
|
|
66
|
+
"에게도",
|
|
67
|
+
"라고는",
|
|
68
|
+
"만큼은",
|
|
69
|
+
"에서",
|
|
70
|
+
"에게",
|
|
71
|
+
"한테",
|
|
72
|
+
"께서",
|
|
73
|
+
"으로",
|
|
74
|
+
"부터",
|
|
75
|
+
"까지",
|
|
76
|
+
"보다",
|
|
77
|
+
"처럼",
|
|
78
|
+
"만큼",
|
|
79
|
+
"마다",
|
|
80
|
+
"조차",
|
|
81
|
+
"밖에",
|
|
82
|
+
"라는",
|
|
83
|
+
"라고",
|
|
84
|
+
"라도",
|
|
85
|
+
"이나",
|
|
86
|
+
"을",
|
|
87
|
+
"를",
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
#: Particles that collide with real noun endings — stripped only when the text
|
|
91
|
+
#: shows the bare stem elsewhere (see :func:`strip_particle`).
|
|
92
|
+
EVIDENCE_PARTICLES: Tuple[str, ...] = (
|
|
93
|
+
"은",
|
|
94
|
+
"는",
|
|
95
|
+
"이",
|
|
96
|
+
"가",
|
|
97
|
+
"의",
|
|
98
|
+
"와",
|
|
99
|
+
"과",
|
|
100
|
+
"도",
|
|
101
|
+
"로",
|
|
102
|
+
"만",
|
|
103
|
+
"나",
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
#: Opening → closing delimiters unwrapped when a term is wrapped in a *matched*
|
|
107
|
+
#: pair. Matched only: ``<data_dir>/cloud_provider.json`` opens a bracket it
|
|
108
|
+
#: never closes, and stripping the lone ``<`` invents a name nobody wrote.
|
|
109
|
+
_DELIMITER_PAIRS = {
|
|
110
|
+
"(": ")",
|
|
111
|
+
"[": "]",
|
|
112
|
+
"{": "}",
|
|
113
|
+
"<": ">",
|
|
114
|
+
"«": "»",
|
|
115
|
+
"“": "”",
|
|
116
|
+
"‘": "’",
|
|
117
|
+
"「": "」",
|
|
118
|
+
"『": "』",
|
|
119
|
+
"《": "》",
|
|
120
|
+
"〈": "〉",
|
|
121
|
+
'"': '"',
|
|
122
|
+
"'": "'",
|
|
123
|
+
"`": "`",
|
|
124
|
+
}
|
|
125
|
+
#: Trailing punctuation trimmed after the brackets come off.
|
|
126
|
+
_TRAILING_PUNCT = ".,;:!?…·~-–—/\\|"
|
|
127
|
+
|
|
128
|
+
_WHITESPACE = re.compile(r"\s+")
|
|
129
|
+
_HANGUL = re.compile(r"[가-힣]")
|
|
130
|
+
_POSSESSIVE = re.compile(r"(?:['’]s|['’])$", re.IGNORECASE)
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _is_hangul_tail(text: str) -> bool:
|
|
134
|
+
"""True when the last character is a Hangul syllable."""
|
|
135
|
+
return bool(text) and bool(_HANGUL.fullmatch(text[-1]))
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def strip_particle(term: str, text: str = "") -> str:
|
|
139
|
+
"""``term`` with one trailing Korean particle removed, when that is safe.
|
|
140
|
+
|
|
141
|
+
``text`` is the passage the term came from; it is the corroboration for
|
|
142
|
+
the Tier-2 particles. Pass ``""`` and only Tier 1 fires.
|
|
143
|
+
|
|
144
|
+
>>> strip_particle("플랫폼에서")
|
|
145
|
+
'플랫폼'
|
|
146
|
+
>>> strip_particle("고양이") # no evidence → left alone
|
|
147
|
+
'고양이'
|
|
148
|
+
>>> strip_particle("플랫폼이", "플랫폼이 있고 플랫폼도 있다")
|
|
149
|
+
'플랫폼'
|
|
150
|
+
"""
|
|
151
|
+
term = term.strip()
|
|
152
|
+
if not _is_hangul_tail(term):
|
|
153
|
+
return term
|
|
154
|
+
for particle in UNCONDITIONAL_PARTICLES:
|
|
155
|
+
if term.endswith(particle):
|
|
156
|
+
stem = term[: -len(particle)]
|
|
157
|
+
if len(stem) >= MIN_STEM_CHARS and _is_hangul_tail(stem):
|
|
158
|
+
return stem
|
|
159
|
+
if not text:
|
|
160
|
+
return term
|
|
161
|
+
for particle in EVIDENCE_PARTICLES:
|
|
162
|
+
if not term.endswith(particle):
|
|
163
|
+
continue
|
|
164
|
+
stem = term[: -len(particle)]
|
|
165
|
+
if len(stem) < MIN_STEM_CHARS or not _is_hangul_tail(stem):
|
|
166
|
+
continue
|
|
167
|
+
if _stem_stands_alone(stem, particle, text):
|
|
168
|
+
return stem
|
|
169
|
+
return term
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
@functools.lru_cache(maxsize=8192)
|
|
173
|
+
def _stem_alone_re(stem: str, particle: str) -> re.Pattern[str]:
|
|
174
|
+
"""Compiled ``stem`` not-followed-by ``particle`` look-ahead."""
|
|
175
|
+
return re.compile(re.escape(stem) + "(?!" + re.escape(particle) + ")")
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _stem_stands_alone(stem: str, particle: str, text: str) -> bool:
|
|
179
|
+
"""True when ``text`` contains ``stem`` *not* followed by ``particle``.
|
|
180
|
+
|
|
181
|
+
That is the whole evidence test: if the passage only ever writes
|
|
182
|
+
``정확도``, the trailing ``도`` is part of the word. If it also writes
|
|
183
|
+
``정확`` on its own (or ``정확을``, ``정확에서`` …), the ``도`` was a
|
|
184
|
+
particle after all.
|
|
185
|
+
"""
|
|
186
|
+
return _stem_alone_re(stem, particle).search(text) is not None
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def normalize_entity(term: str, text: str = "") -> str:
|
|
190
|
+
"""The canonical *surface* form of one extracted entity.
|
|
191
|
+
|
|
192
|
+
NFKC (so ``AI`` and ``AI`` are one word), whitespace collapsed to single
|
|
193
|
+
spaces, brackets/quotes/trailing punctuation trimmed, English possessive
|
|
194
|
+
dropped, and one Korean particle stripped when :func:`strip_particle`
|
|
195
|
+
considers it safe. Returns ``""`` for anything that normalizes away.
|
|
196
|
+
|
|
197
|
+
>>> normalize_entity(" “Lattice AI” ")
|
|
198
|
+
'Lattice AI'
|
|
199
|
+
>>> normalize_entity("Anthropic's")
|
|
200
|
+
'Anthropic'
|
|
201
|
+
>>> normalize_entity("지식그래프에서")
|
|
202
|
+
'지식그래프'
|
|
203
|
+
"""
|
|
204
|
+
cleaned = unicodedata.normalize("NFKC", str(term or ""))
|
|
205
|
+
cleaned = _WHITESPACE.sub(" ", cleaned).strip()
|
|
206
|
+
# Matched wrappers come off, then trailing sentence punctuation; looped so
|
|
207
|
+
# `("Lattice AI").` unwraps fully rather than one layer at a time.
|
|
208
|
+
for _ in range(3):
|
|
209
|
+
before = cleaned
|
|
210
|
+
if len(cleaned) > 2 and cleaned.endswith(_DELIMITER_PAIRS.get(cleaned[0], "\0")):
|
|
211
|
+
cleaned = cleaned[1:-1].strip()
|
|
212
|
+
cleaned = cleaned.rstrip(_TRAILING_PUNCT).strip()
|
|
213
|
+
if cleaned == before:
|
|
214
|
+
break
|
|
215
|
+
if not cleaned:
|
|
216
|
+
return ""
|
|
217
|
+
cleaned = _POSSESSIVE.sub("", cleaned).strip()
|
|
218
|
+
return strip_particle(cleaned, text)
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def entity_key(term: str) -> str:
|
|
222
|
+
"""The dedup key two surface forms of one entity share.
|
|
223
|
+
|
|
224
|
+
Case-folded and whitespace-collapsed, and with the separators that
|
|
225
|
+
identifier styles disagree about (space, ``-``, ``_``) removed — so
|
|
226
|
+
``Graph RAG`` / ``graph-rag`` / ``graph_rag`` are one key while
|
|
227
|
+
``GraphRAG`` (already joined) matches them too.
|
|
228
|
+
"""
|
|
229
|
+
folded = unicodedata.normalize("NFKC", str(term or "")).casefold()
|
|
230
|
+
folded = _WHITESPACE.sub(" ", folded).strip()
|
|
231
|
+
return re.sub(r"[\s_\-]+", "", folded)
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _representative(surfaces: Sequence[str], text: str) -> str:
|
|
235
|
+
"""Pick the surface form an entity's node should carry.
|
|
236
|
+
|
|
237
|
+
Most frequent in the source text wins — the way the author actually writes
|
|
238
|
+
it. Ties go to the form with more capitalized words (``Lattice AI`` over
|
|
239
|
+
``lattice ai``), then to the longer form, then to the first one seen, so
|
|
240
|
+
the choice never depends on dict ordering.
|
|
241
|
+
"""
|
|
242
|
+
best = surfaces[0]
|
|
243
|
+
best_rank = (-1, -1, -1, 0)
|
|
244
|
+
for index, surface in enumerate(surfaces):
|
|
245
|
+
rank = (
|
|
246
|
+
text.count(surface) if text else 0,
|
|
247
|
+
sum(1 for word in surface.split() if word[:1].isupper()),
|
|
248
|
+
len(surface),
|
|
249
|
+
-index,
|
|
250
|
+
)
|
|
251
|
+
if rank > best_rank:
|
|
252
|
+
best_rank = rank
|
|
253
|
+
best = surface
|
|
254
|
+
return best
|
|
255
|
+
|
|
256
|
+
|
|
257
|
+
def merge_entity_aliases(terms: Iterable[str], text: str = "") -> List[str]:
|
|
258
|
+
"""Normalize every term, drop the empties, and merge exact-key aliases.
|
|
259
|
+
|
|
260
|
+
Order is the order the *first* member of each group appeared in, so the
|
|
261
|
+
caller's own priority (backticked terms first, then proper nouns, …) is
|
|
262
|
+
preserved. The value is the group's representative surface form.
|
|
263
|
+
|
|
264
|
+
>>> merge_entity_aliases(["Lattice AI", "lattice ai", "Graph RAG"])
|
|
265
|
+
['Lattice AI', 'Graph RAG']
|
|
266
|
+
"""
|
|
267
|
+
groups: Dict[str, List[str]] = {}
|
|
268
|
+
order: List[str] = []
|
|
269
|
+
for term in terms:
|
|
270
|
+
surface = normalize_entity(term, text)
|
|
271
|
+
if not surface:
|
|
272
|
+
continue
|
|
273
|
+
key = entity_key(surface)
|
|
274
|
+
if not key:
|
|
275
|
+
continue
|
|
276
|
+
if key not in groups:
|
|
277
|
+
groups[key] = []
|
|
278
|
+
order.append(key)
|
|
279
|
+
if surface not in groups[key]:
|
|
280
|
+
groups[key].append(surface)
|
|
281
|
+
return [_representative(groups[key], text) for key in order]
|
|
282
|
+
|
|
283
|
+
|
|
284
|
+
def occurrence_count(term: str, text: str) -> int:
|
|
285
|
+
"""How many times ``term`` appears in ``text``, case-insensitively.
|
|
286
|
+
|
|
287
|
+
The number an edge's ``occurrences`` metadata carries. Counted on the
|
|
288
|
+
normalized surface, so ``플랫폼은`` and ``플랫폼을`` both count toward
|
|
289
|
+
``플랫폼``. Zero-length terms count zero rather than raising.
|
|
290
|
+
"""
|
|
291
|
+
if not term or not text:
|
|
292
|
+
return 0
|
|
293
|
+
return len(re.findall(re.escape(term), text, re.IGNORECASE))
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
__all__ = [
|
|
297
|
+
"EVIDENCE_PARTICLES",
|
|
298
|
+
"MIN_STEM_CHARS",
|
|
299
|
+
"UNCONDITIONAL_PARTICLES",
|
|
300
|
+
"entity_key",
|
|
301
|
+
"merge_entity_aliases",
|
|
302
|
+
"normalize_entity",
|
|
303
|
+
"occurrence_count",
|
|
304
|
+
"strip_particle",
|
|
305
|
+
]
|
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""Directed, typed relation patterns — the part of extraction that reads syntax.
|
|
2
|
+
|
|
3
|
+
``infer_edge_relation`` classifies a *sentence*: it finds a verb anywhere in it
|
|
4
|
+
and stamps every concept pair in that sentence with the same label, in text
|
|
5
|
+
order. That is cheap and it is often wrong about **direction**, and it cannot
|
|
6
|
+
tell a definition from a passing mention.
|
|
7
|
+
|
|
8
|
+
The four rules here look at where each concept sits inside the sentence and at
|
|
9
|
+
what stands *between* the two, so ``A는 B를 사용한다`` and ``B is used by A``
|
|
10
|
+
come out with the same subject. Each rule returns the relation label the graph
|
|
11
|
+
should carry, the evidence class, and a weight:
|
|
12
|
+
|
|
13
|
+
| rule | label | `edges_v2` type | evidence | weight |
|
|
14
|
+
|---|---|---|---|---|
|
|
15
|
+
| definition | `설명함` | `MENTIONS` | `definition` | 1.0 |
|
|
16
|
+
| SVO / SOV | the matched `EDGE_VERB` label | that label's type | `verb` | 1.0 |
|
|
17
|
+
| part-of | `구성요소` | `PART_OF` | `structure` | 0.9 |
|
|
18
|
+
| contrast | `상충함` | `CONTRADICTS` | `contrast` | 0.9 |
|
|
19
|
+
|
|
20
|
+
``PART_OF`` and ``CONTRADICTS`` were reachable in the taxonomy but nothing
|
|
21
|
+
extracted them; the graph only ever produced the ten labels `EDGE_VERB` names.
|
|
22
|
+
|
|
23
|
+
Everything is regex over a single sentence — deterministic, no model, and the
|
|
24
|
+
same input always yields the same edge.
|
|
25
|
+
"""
|
|
26
|
+
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import re
|
|
30
|
+
from typing import Any, Dict, List, Optional, Sequence, Tuple
|
|
31
|
+
|
|
32
|
+
from .relations import _EDGE_VERB_COMPILED
|
|
33
|
+
|
|
34
|
+
#: Weight for a relation a syntactic pattern named outright.
|
|
35
|
+
PATTERN_EDGE_WEIGHT = 0.9
|
|
36
|
+
#: Weight for definition and verb-anchored relations — the strongest evidence.
|
|
37
|
+
STRONG_EDGE_WEIGHT = 1.0
|
|
38
|
+
|
|
39
|
+
#: `A is a B` / `A refers to B` — the copula and its cousins, English.
|
|
40
|
+
_EN_DEFINITION = re.compile(
|
|
41
|
+
r"^\s*(?:is|are|was|were)\s+(?:a|an|the)?\s*$"
|
|
42
|
+
r"|^\s*(?:refers?\s+to|means?|stands\s+for|is\s+defined\s+as|is\s+known\s+as"
|
|
43
|
+
r"|is\s+short\s+for)\s*$",
|
|
44
|
+
re.IGNORECASE,
|
|
45
|
+
)
|
|
46
|
+
#: `A란 B이다` / `A는 B를 의미한다` — the Korean definition tail.
|
|
47
|
+
_KO_DEFINITION_TAIL = re.compile(
|
|
48
|
+
r"(?:이다|입니다|이란다|이에요|예요|을 뜻한다|를 뜻한다|을 의미한다|를 의미한다"
|
|
49
|
+
r"|을 말한다|를 말한다|이라고 한다|라고 한다|이라 한다)\s*[.!?]?\s*$"
|
|
50
|
+
)
|
|
51
|
+
#: `A란`, `A이란`, `A라는 것은` — the Korean definition head marker. The
|
|
52
|
+
#: lookbehind and the trailing space keep the bare ``란`` from matching the
|
|
53
|
+
#: middle of an ordinary word (``결과란에``, ``발란스``).
|
|
54
|
+
_KO_DEFINITION_HEAD = re.compile(
|
|
55
|
+
r"(?<=[가-힣])(?:이란|란|라는 것은|라 함은)\s|(?:의 정의는|정의는)\s"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
#: `A is part of B` — part first, whole second.
|
|
59
|
+
_EN_PART_FORWARD = re.compile(
|
|
60
|
+
r"^\s*(?:is|are|was|were)?\s*(?:a|an|the)?\s*"
|
|
61
|
+
r"(?:part\s+of|component\s+of|subset\s+of|member\s+of|belongs?\s+to"
|
|
62
|
+
r"|lives?\s+(?:in|under)|sits?\s+(?:in|under))\s*$",
|
|
63
|
+
re.IGNORECASE,
|
|
64
|
+
)
|
|
65
|
+
#: `B consists of A` — whole first, part second, so the edge is reversed.
|
|
66
|
+
#: ``contains``/``includes`` are deliberately **not** here: they already route
|
|
67
|
+
#: to `포함함` → `CONTAINS`, which is the right edge pointing the right way.
|
|
68
|
+
_EN_PART_REVERSE = re.compile(
|
|
69
|
+
r"^\s*(?:consists?\s+of|comprises?|is\s+made\s+(?:up\s+)?of)\s*$",
|
|
70
|
+
re.IGNORECASE,
|
|
71
|
+
)
|
|
72
|
+
#: `A는 B의 일부` / `A는 B에 속한다` — part first, whole second.
|
|
73
|
+
_KO_PART_FORWARD = re.compile(r"(?:의 일부|의 구성요소|의 하위|에 속한|에 포함된|의 부분)")
|
|
74
|
+
#: `A는 B로 구성된다` — whole first, part second.
|
|
75
|
+
_KO_PART_REVERSE = re.compile(r"(?:로 구성|으로 구성|의 하위 항목)")
|
|
76
|
+
|
|
77
|
+
#: `A unlike B` / `A instead of B` — a stated opposition, not a comparison.
|
|
78
|
+
_EN_CONTRAST = re.compile(
|
|
79
|
+
r"(?:\bunlike\b|\binstead\s+of\b|\brather\s+than\b|\bcontrary\s+to\b"
|
|
80
|
+
r"|\bas\s+opposed\s+to\b|\bnot\b[^.]{0,20}\bbut\b)",
|
|
81
|
+
re.IGNORECASE,
|
|
82
|
+
)
|
|
83
|
+
#: `A가 아니라 B` / `A 대신 B` / `A와 달리 B`.
|
|
84
|
+
_KO_CONTRAST = re.compile(r"(?:아니라|아닌|대신|와 달리|과 달리|반면|이 아니고|가 아니고)")
|
|
85
|
+
|
|
86
|
+
#: Subject markers: the syllable that says "this noun is the actor".
|
|
87
|
+
_KO_SUBJECT_MARKS: Tuple[str, ...] = ("은", "는", "이", "가", "께서")
|
|
88
|
+
#: Object markers: the syllable that says "this noun is acted on".
|
|
89
|
+
_KO_OBJECT_MARKS: Tuple[str, ...] = ("을", "를")
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
def _marker_after(sentence: str, end: int, markers: Sequence[str]) -> bool:
|
|
93
|
+
"""True when one of ``markers`` sits immediately after ``end``.
|
|
94
|
+
|
|
95
|
+
Korean glues the particle to the noun with no space, so a single lookahead
|
|
96
|
+
character is the whole test.
|
|
97
|
+
"""
|
|
98
|
+
tail = sentence[end : end + 2]
|
|
99
|
+
return any(tail.startswith(mark) for mark in markers)
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _verb_label(span: str) -> Optional[str]:
|
|
103
|
+
"""The `EDGE_VERB` label whose pattern matches ``span``, if any."""
|
|
104
|
+
lowered = span.lower()
|
|
105
|
+
for label, pattern in _EDGE_VERB_COMPILED:
|
|
106
|
+
if pattern.search(lowered):
|
|
107
|
+
return label
|
|
108
|
+
return None
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _triple(
|
|
112
|
+
subject: str,
|
|
113
|
+
obj: str,
|
|
114
|
+
relation: str,
|
|
115
|
+
evidence: str,
|
|
116
|
+
weight: float,
|
|
117
|
+
context: str,
|
|
118
|
+
) -> Dict[str, Any]:
|
|
119
|
+
return {
|
|
120
|
+
"subject": subject,
|
|
121
|
+
"relation": relation,
|
|
122
|
+
"object": obj,
|
|
123
|
+
"context": context[:240],
|
|
124
|
+
"evidence": evidence,
|
|
125
|
+
"weight": weight,
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def concept_positions(
|
|
130
|
+
sentence: str, concepts: Sequence[str]
|
|
131
|
+
) -> List[Tuple[int, str]]:
|
|
132
|
+
"""``(offset, concept)`` for every concept present, in text order.
|
|
133
|
+
|
|
134
|
+
Case-insensitive, first occurrence only — a concept repeated in one
|
|
135
|
+
sentence is one participant, not two.
|
|
136
|
+
"""
|
|
137
|
+
lowered = sentence.lower()
|
|
138
|
+
found: List[Tuple[int, str]] = []
|
|
139
|
+
for concept in concepts:
|
|
140
|
+
index = lowered.find(concept.lower())
|
|
141
|
+
if index >= 0:
|
|
142
|
+
found.append((index, concept))
|
|
143
|
+
found.sort(key=lambda pair: (pair[0], pair[1]))
|
|
144
|
+
return found
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def typed_relation(
|
|
148
|
+
sentence: str,
|
|
149
|
+
left: Tuple[int, str],
|
|
150
|
+
right: Tuple[int, str],
|
|
151
|
+
adjacent: bool = True,
|
|
152
|
+
) -> Optional[Dict[str, Any]]:
|
|
153
|
+
"""The typed, directed relation between two concepts in one sentence.
|
|
154
|
+
|
|
155
|
+
``left``/``right`` are ``(offset, concept)`` with ``left`` first in the
|
|
156
|
+
text. Returns ``None`` when no rule fires, which is the caller's signal to
|
|
157
|
+
fall back to the sentence-level co-occurrence classification.
|
|
158
|
+
|
|
159
|
+
``adjacent=False`` means another concept sits between the two, and then
|
|
160
|
+
only the particle-marked Korean subject→object rule may fire. Everything
|
|
161
|
+
else reads the span *between* the pair, and with a third concept in there
|
|
162
|
+
that span describes somebody else's relation: ``A는 B가 아니라 C를 쓴다``
|
|
163
|
+
puts ``아니라`` between A and C without A and C being in contrast at all.
|
|
164
|
+
"""
|
|
165
|
+
left_at, subject = left
|
|
166
|
+
right_at, obj = right
|
|
167
|
+
between = sentence[left_at + len(subject) : right_at]
|
|
168
|
+
after = sentence[right_at + len(obj) :]
|
|
169
|
+
|
|
170
|
+
if not adjacent:
|
|
171
|
+
return _korean_subject_object(sentence, subject, obj, after, left_at, right_at)
|
|
172
|
+
|
|
173
|
+
definition = _definition(sentence, between, subject, obj)
|
|
174
|
+
if definition is not None:
|
|
175
|
+
return definition
|
|
176
|
+
part_of = _part_of(between, after, subject, obj, sentence)
|
|
177
|
+
if part_of is not None:
|
|
178
|
+
return part_of
|
|
179
|
+
contrast = _contrast(between, subject, obj, sentence)
|
|
180
|
+
if contrast is not None:
|
|
181
|
+
return contrast
|
|
182
|
+
return _verb_anchored(sentence, subject, obj, between, after, left_at, right_at)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
def _definition(
|
|
186
|
+
sentence: str, between: str, subject: str, obj: str
|
|
187
|
+
) -> Optional[Dict[str, Any]]:
|
|
188
|
+
english = _EN_DEFINITION.match(between)
|
|
189
|
+
korean = bool(_KO_DEFINITION_TAIL.search(sentence)) and bool(
|
|
190
|
+
_KO_DEFINITION_HEAD.search(sentence)
|
|
191
|
+
)
|
|
192
|
+
if not english and not korean:
|
|
193
|
+
return None
|
|
194
|
+
return _triple(
|
|
195
|
+
subject, obj, "설명함", "definition", STRONG_EDGE_WEIGHT, sentence
|
|
196
|
+
)
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def _part_of(
|
|
200
|
+
between: str, after: str, subject: str, obj: str, sentence: str
|
|
201
|
+
) -> Optional[Dict[str, Any]]:
|
|
202
|
+
# English states the relation *between* the two ("A is part of B"); Korean
|
|
203
|
+
# glues it to the second one ("A는 B의 일부이다"), so the tail is read too —
|
|
204
|
+
# anchored at position zero, because the marker belongs to the noun it is
|
|
205
|
+
# stuck to. A match further along the tail is some *other* noun's relation.
|
|
206
|
+
if _EN_PART_FORWARD.match(between) or _KO_PART_FORWARD.match(after):
|
|
207
|
+
return _triple(
|
|
208
|
+
subject, obj, "구성요소", "structure", PATTERN_EDGE_WEIGHT, sentence
|
|
209
|
+
)
|
|
210
|
+
if _EN_PART_REVERSE.match(between) or _KO_PART_REVERSE.match(after):
|
|
211
|
+
# `B consists of A` — the *whole* was named first, so the part-of edge
|
|
212
|
+
# points the other way. Direction is the whole point of this module.
|
|
213
|
+
return _triple(
|
|
214
|
+
obj, subject, "구성요소", "structure", PATTERN_EDGE_WEIGHT, sentence
|
|
215
|
+
)
|
|
216
|
+
return None
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def _contrast(
|
|
220
|
+
between: str, subject: str, obj: str, sentence: str
|
|
221
|
+
) -> Optional[Dict[str, Any]]:
|
|
222
|
+
if _EN_CONTRAST.search(between) or _KO_CONTRAST.search(between):
|
|
223
|
+
return _triple(
|
|
224
|
+
subject, obj, "상충함", "contrast", PATTERN_EDGE_WEIGHT, sentence
|
|
225
|
+
)
|
|
226
|
+
return None
|
|
227
|
+
|
|
228
|
+
|
|
229
|
+
def _verb_anchored(
|
|
230
|
+
sentence: str,
|
|
231
|
+
subject: str,
|
|
232
|
+
obj: str,
|
|
233
|
+
between: str,
|
|
234
|
+
after: str,
|
|
235
|
+
left_at: int,
|
|
236
|
+
right_at: int,
|
|
237
|
+
) -> Optional[Dict[str, Any]]:
|
|
238
|
+
"""A verb that sits *between* the pair (SVO) or *after* it (Korean SOV).
|
|
239
|
+
|
|
240
|
+
English puts the verb between subject and object, so ``between`` naming a
|
|
241
|
+
verb is enough. Korean puts it last: the direction comes from the particles
|
|
242
|
+
instead — ``A는 … B를 사용한다`` marks A as subject and B as object, and the
|
|
243
|
+
verb in the tail names the relation.
|
|
244
|
+
"""
|
|
245
|
+
label = _verb_label(between)
|
|
246
|
+
if label is not None:
|
|
247
|
+
return _triple(subject, obj, label, "verb", STRONG_EDGE_WEIGHT, sentence)
|
|
248
|
+
return _korean_subject_object(sentence, subject, obj, after, left_at, right_at)
|
|
249
|
+
|
|
250
|
+
|
|
251
|
+
def _korean_subject_object(
|
|
252
|
+
sentence: str,
|
|
253
|
+
subject: str,
|
|
254
|
+
obj: str,
|
|
255
|
+
after: str,
|
|
256
|
+
left_at: int,
|
|
257
|
+
right_at: int,
|
|
258
|
+
) -> Optional[Dict[str, Any]]:
|
|
259
|
+
"""``A는 … B를 <verb>`` — the particles decide, the tail verb names it."""
|
|
260
|
+
subject_marked = _marker_after(sentence, left_at + len(subject), _KO_SUBJECT_MARKS)
|
|
261
|
+
object_marked = _marker_after(sentence, right_at + len(obj), _KO_OBJECT_MARKS)
|
|
262
|
+
if not (subject_marked and object_marked):
|
|
263
|
+
return None
|
|
264
|
+
tail_label = _verb_label(after)
|
|
265
|
+
if tail_label is None:
|
|
266
|
+
return None
|
|
267
|
+
return _triple(subject, obj, tail_label, "verb", STRONG_EDGE_WEIGHT, sentence)
|
|
268
|
+
|
|
269
|
+
|
|
270
|
+
__all__ = [
|
|
271
|
+
"PATTERN_EDGE_WEIGHT",
|
|
272
|
+
"STRONG_EDGE_WEIGHT",
|
|
273
|
+
"concept_positions",
|
|
274
|
+
"typed_relation",
|
|
275
|
+
]
|
|
@@ -32,7 +32,9 @@ EDGE_VERB = {
|
|
|
32
32
|
"의존함": r"의존|depend|require|필요|based on",
|
|
33
33
|
"설명함": r"설명|explain|describe|정의|란|이란|means",
|
|
34
34
|
"비교함": r"비교|versus|vs\.?|차이|다르|compare",
|
|
35
|
-
"
|
|
35
|
+
# `쓴다`/`쓰는`/`씁니다` are the everyday Korean for "uses"; without them a
|
|
36
|
+
# sentence that never says 사용 fell through to bare co-occurrence.
|
|
37
|
+
"사용함": r"사용|use|활용|이용|apply|쓴다|쓰는|씁니다|썼다",
|
|
36
38
|
"연결함": r"연결|connect|통합|integrate|연동|link",
|
|
37
39
|
"확장함": r"확장|extend|플러그인|plugin|addon",
|
|
38
40
|
"생성함": r"생성|만들|create|generate|build|produced",
|
|
@@ -42,6 +44,13 @@ EDGE_VERB = {
|
|
|
42
44
|
"관련됨": r"관련|related|associated|연관",
|
|
43
45
|
}
|
|
44
46
|
|
|
47
|
+
#: Same table, compiled once. ``infer_edge_relation`` and the typed-relation
|
|
48
|
+
#: patterns both walk this on every pair; compiling per call was the cheap
|
|
49
|
+
#: half of the extraction regression.
|
|
50
|
+
_EDGE_VERB_COMPILED = tuple(
|
|
51
|
+
(label, re.compile(pattern)) for label, pattern in EDGE_VERB.items()
|
|
52
|
+
)
|
|
53
|
+
|
|
45
54
|
|
|
46
55
|
# Concepts in a list-like sentence ("A, B, C, D를 사용한다") sit together by
|
|
47
56
|
# enumeration, not by relation. Beyond this many concepts in one sentence, a
|
|
@@ -70,8 +79,8 @@ def infer_edge_relation(sentence: str) -> Dict[str, Any]:
|
|
|
70
79
|
label-only output erased.
|
|
71
80
|
"""
|
|
72
81
|
s = str(sentence or "").lower()
|
|
73
|
-
for label, pattern in
|
|
74
|
-
if
|
|
82
|
+
for label, pattern in _EDGE_VERB_COMPILED:
|
|
83
|
+
if pattern.search(s):
|
|
75
84
|
# "관련됨" is itself a weak, generic label: matching it by keyword
|
|
76
85
|
# ("관련", "related") is still verb evidence, but nothing stronger.
|
|
77
86
|
return {
|