ltcai 11.7.0 → 12.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (174) hide show
  1. package/README.md +100 -76
  2. package/docs/BENCHMARKS.md +9 -2
  3. package/docs/CHANGELOG.md +249 -0
  4. package/docs/CI_AND_RELEASE_GATES.md +126 -41
  5. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  6. package/docs/DEVELOPMENT.md +271 -103
  7. package/docs/ENTERPRISE.md +1 -1
  8. package/docs/LEGACY_COMPATIBILITY.md +10 -6
  9. package/docs/MULTI_AGENT_RUNTIME.md +4 -4
  10. package/docs/ONBOARDING.md +16 -4
  11. package/docs/OPERATIONS.md +14 -1
  12. package/docs/PERMISSION_MODE.md +14 -9
  13. package/docs/REALTIME_COLLABORATION.md +1 -1
  14. package/docs/ROADMAP.md +113 -0
  15. package/docs/TRUST_MODEL.md +28 -7
  16. package/docs/USABILITY_AUDIT.md +5 -0
  17. package/docs/WHY_LATTICE.md +13 -5
  18. package/docs/WORKFLOW_DESIGNER.md +2 -2
  19. package/docs/kg-schema.md +57 -7
  20. package/docs/mcp-tools.md +93 -82
  21. package/docs/security-model.md +6 -3
  22. package/lattice_brain/__init__.py +1 -1
  23. package/lattice_brain/graph/_kg_common/__init__.py +1 -54
  24. package/lattice_brain/graph/_kg_common/extraction.py +459 -105
  25. package/lattice_brain/graph/_kg_common/normalize.py +305 -0
  26. package/lattice_brain/graph/_kg_common/patterns.py +275 -0
  27. package/lattice_brain/graph/_kg_common/relations.py +12 -3
  28. package/lattice_brain/graph/_kg_common/sections.py +107 -0
  29. package/lattice_brain/graph/_kg_common/text.py +14 -450
  30. package/lattice_brain/graph/_kg_constants.py +7 -0
  31. package/lattice_brain/ingestion/__init__.py +6 -3
  32. package/lattice_brain/multimodal/__init__.py +9 -3
  33. package/latticeai/__init__.py +1 -1
  34. package/latticeai/api/agent_worker_seam.py +44 -1
  35. package/latticeai/api/models.py +18 -110
  36. package/latticeai/api/search.py +7 -30
  37. package/latticeai/api/worker_compute.py +127 -106
  38. package/latticeai/api/worker_seams.py +17 -2
  39. package/latticeai/core/embedding_providers/__init__.py +16 -0
  40. package/latticeai/core/embedding_providers/autodetect.py +302 -0
  41. package/latticeai/core/embedding_providers/base.py +25 -0
  42. package/latticeai/core/embedding_providers/profiles.py +44 -0
  43. package/latticeai/core/embedding_providers/text.py +74 -8
  44. package/latticeai/core/http_origin.py +3 -3
  45. package/latticeai/core/messages.py +0 -5
  46. package/latticeai/core/policy.py +1 -6
  47. package/latticeai/core/quiet.py +1 -20
  48. package/latticeai/core/security.py +29 -83
  49. package/latticeai/core/sessions.py +95 -4
  50. package/latticeai/core/users.py +0 -38
  51. package/latticeai/core/vector_index/__init__.py +61 -0
  52. package/latticeai/core/vector_index/hnsw.py +383 -0
  53. package/latticeai/core/vector_index/sidecar.py +329 -0
  54. package/latticeai/models/router/catalog.py +2 -2
  55. package/latticeai/models/router/generation.py +176 -30
  56. package/latticeai/models/router/loading.py +150 -9
  57. package/latticeai/runtime/access_runtime.py +7 -4
  58. package/latticeai/runtime/brain_runtime.py +43 -9
  59. package/latticeai/runtime/build_phases/features.py +8 -31
  60. package/latticeai/runtime/build_phases/foundation.py +7 -16
  61. package/latticeai/runtime/build_phases/web.py +3 -3
  62. package/latticeai/runtime/build_phases/worker_profile.py +29 -27
  63. package/latticeai/runtime/runtime_context.py +0 -2
  64. package/latticeai/services/architecture_readiness.py +18 -19
  65. package/latticeai/services/process_audit.py +1 -22
  66. package/latticeai/services/product_readiness.py +39 -12
  67. package/latticeai/services/search_service.py +7 -0
  68. package/latticeai/services/voice_capture.py +8 -28
  69. package/latticeai/tools/__init__.py +12 -47
  70. package/latticeai/tools/commands.py +9 -15
  71. package/latticeai/tools/documents.py +12 -0
  72. package/latticeai/tools/knowledge.py +0 -6
  73. package/latticeai/tools/markup.py +152 -0
  74. package/package.json +4 -5
  75. package/requirements.txt +0 -1
  76. package/scripts/check_current_release_docs.mjs +1 -1
  77. package/scripts/check_openapi_drift.mjs +3 -2
  78. package/scripts/check_server_i18n.mjs +5 -4
  79. package/scripts/compose_openapi.py +4 -1
  80. package/scripts/export_openapi.py +5 -4
  81. package/scripts/gen_worker_allowlist_fixture.py +2 -2
  82. package/scripts/openapi_route_families.json +19 -74
  83. package/scripts/publish_release.mjs +157 -0
  84. package/scripts/release_screen_claims.json +144 -28
  85. package/src-tauri/Cargo.lock +45 -10
  86. package/src-tauri/Cargo.toml +1 -1
  87. package/src-tauri/tauri.conf.json +1 -1
  88. package/static/app/asset-manifest.json +47 -41
  89. package/static/app/assets/Act-Cf1L2709.js +2 -0
  90. package/static/app/assets/AdminConsole-DPAbLTYV.js +1 -0
  91. package/static/app/assets/Brain-DqamGrj-.js +2 -0
  92. package/static/app/assets/BrainHome-MHe2_RYs.js +2 -0
  93. package/static/app/assets/BrainSignals-CQPPfyyH.js +1 -0
  94. package/static/app/assets/Capture-DGdIH_Zc.js +1 -0
  95. package/static/app/assets/Chronicle-C-UlCJoJ.js +1 -0
  96. package/static/app/assets/CommandPalette-WNT4EqUX.js +1 -0
  97. package/static/app/assets/DigitalBrainExplorer-CEBH5Cwc.js +321 -0
  98. package/static/app/assets/Library-C6xd1dlf.js +1 -0
  99. package/static/app/assets/LivingBrain-BEk-0ohw.js +1 -0
  100. package/static/app/assets/ProductFlow-CZLm5iXh.js +1 -0
  101. package/static/app/assets/QueryClientProvider-B3OjqSyJ.js +1 -0
  102. package/static/app/assets/{ReviewCard-HXRle3qq.js → ReviewCard-CEHG6evf.js} +2 -2
  103. package/static/app/assets/RunsListPanel-CLtEJSRW.js +1 -0
  104. package/static/app/assets/System-CAxwBUXw.js +1 -0
  105. package/static/app/assets/WorkflowGraph-Dj10RuGE.js +1 -0
  106. package/static/app/assets/WorkflowsPanel-Kyeh_LIT.js +2 -0
  107. package/static/app/assets/actHelpers-CtSmK9Dw.js +1 -0
  108. package/static/app/assets/arrow-left-CRl5EO4D.js +1 -0
  109. package/static/app/assets/{bot-Cn8bWRuq.js → bot-DhUGRel2.js} +1 -1
  110. package/static/app/assets/brain-CLkhHsHF.js +1 -0
  111. package/static/app/assets/button-CmaEqG1T.js +1 -0
  112. package/static/app/assets/circle-check-CFgejkOS.js +1 -0
  113. package/static/app/assets/{circle-pause-CmzC_apg.js → circle-pause-l96izbxj.js} +1 -1
  114. package/static/app/assets/{circle-play-D8mW2aQ7.js → circle-play-CrZa25_q.js} +1 -1
  115. package/static/app/assets/{cpu-DZcdd0PZ.js → cpu-BaXudqwl.js} +1 -1
  116. package/static/app/assets/{download-bv1KEPGQ.js → download-hCVFPiyc.js} +1 -1
  117. package/static/app/assets/{folder-open-d-Pip5gr.js → folder-open-CHL82Yp7.js} +1 -1
  118. package/static/app/assets/{hard-drive-D20iavUb.js → hard-drive-DDzET7lk.js} +1 -1
  119. package/static/app/assets/index-CB93CZWW.css +2 -0
  120. package/static/app/assets/index-D2H-wSl6.js +13 -0
  121. package/static/app/assets/input-Df1CAY_I.js +1 -0
  122. package/static/app/assets/jsx-runtime-bzQ4Vb5N.js +1 -0
  123. package/static/app/assets/{link-2-BPJOFlAy.js → link-2-xNnTIX1_.js} +1 -1
  124. package/static/app/assets/{permissionCopy-ChdJd493.js → permissionCopy-D3aWHco-.js} +1 -1
  125. package/static/app/assets/primitives-BioD2slS.js +1 -0
  126. package/static/app/assets/search-BzBw8YcW.js +1 -0
  127. package/static/app/assets/{share-2-YNX_NtMU.js → share-2-FkzGf8Df.js} +1 -1
  128. package/static/app/assets/{shield-alert-DuQ3zrVL.js → shield-alert-B3dwzik4.js} +1 -1
  129. package/static/app/assets/sourceMeta-DQSY_tah.js +1 -0
  130. package/static/app/assets/textarea-P8o6pvOP.js +1 -0
  131. package/static/app/assets/useFocusTrap-hswOIkXE.js +1 -0
  132. package/static/app/assets/useMutation-OJLrYSRA.js +1 -0
  133. package/static/app/assets/workspace-BCuk3Ku9.js +1 -0
  134. package/static/app/index.html +4 -4
  135. package/static/sw.js +1 -1
  136. package/lattice_brain/ingestion/pipeline.py +0 -108
  137. package/latticeai/api/local_files.py +0 -44
  138. package/latticeai/api/tools.py +0 -126
  139. package/latticeai/api/voice_capture.py +0 -32
  140. package/latticeai/core/agent_permission.py +0 -85
  141. package/scripts/agent_eval.py +0 -34
  142. package/scripts/brain_quality_eval.py +0 -37
  143. package/scripts/check_legacy_debt.mjs +0 -91
  144. package/scripts/check_python.py +0 -100
  145. package/scripts/chunking_parity_corpus.py +0 -449
  146. package/scripts/generate_agent_parity_fixtures.py +0 -771
  147. package/scripts/generate_chunking_parity_fixtures.py +0 -259
  148. package/static/app/assets/Act-BPcVAbOL.js +0 -1
  149. package/static/app/assets/AdminConsole-Bw1ATQL0.js +0 -1
  150. package/static/app/assets/Brain-CT92Kos0.js +0 -321
  151. package/static/app/assets/BrainHome-CFBkt1K_.js +0 -2
  152. package/static/app/assets/BrainSignals-ReLWF2H8.js +0 -1
  153. package/static/app/assets/Capture-BsTokYkk.js +0 -1
  154. package/static/app/assets/Chronicle-B6f0T9id.js +0 -1
  155. package/static/app/assets/CommandPalette-CuvjTv1u.js +0 -1
  156. package/static/app/assets/Library-BGJbG9Hd.js +0 -1
  157. package/static/app/assets/LivingBrain-DGYK_Jsa.js +0 -1
  158. package/static/app/assets/ProductFlow-DXBC6brE.js +0 -1
  159. package/static/app/assets/System-CMHSO9qM.js +0 -1
  160. package/static/app/assets/arrow-left-BfmkskWx.js +0 -1
  161. package/static/app/assets/brain-CQJberbE.js +0 -1
  162. package/static/app/assets/button-Ct9f2_oT.js +0 -1
  163. package/static/app/assets/circle-check-DruOxB-4.js +0 -1
  164. package/static/app/assets/index-D9x-kSNy.css +0 -2
  165. package/static/app/assets/index-Do83hDzJ.js +0 -10
  166. package/static/app/assets/input-BLXVNmj1.js +0 -1
  167. package/static/app/assets/primitives-Cv5tbZBY.js +0 -1
  168. package/static/app/assets/search-CT9aho2j.js +0 -1
  169. package/static/app/assets/textarea-DqwLnli4.js +0 -1
  170. package/static/app/assets/useFocusTrap-ZVI98jaW.js +0 -1
  171. package/static/app/assets/useMutation-CVC4qv_D.js +0 -1
  172. package/static/app/assets/useQuery-C7BeG4HU.js +0 -1
  173. package/static/app/assets/utils-CiFtIdZq.js +0 -4
  174. package/static/app/assets/workspace-DQz9vIId.js +0 -1
@@ -0,0 +1,305 @@
1
+ """Entity-surface normalization and alias merge — deterministic, no model.
2
+
3
+ Two concepts that are the same thing must become one node, or the graph
4
+ answers "무엇이 무엇과 이어져 있나" with a fan of near-duplicates. Before
5
+ v12.0.0 the only dedup was ``concept.text.lower()`` inside the Rust writer, so
6
+ ``"Lattice AI"`` / ``"lattice ai"`` / ``"Lattice AI의"`` were three nodes.
7
+
8
+ This module is the *surface* half of that fix, and it is deliberately boring:
9
+ Unicode normalization, whitespace collapse, bracket/quote/punctuation trimming,
10
+ English possessives, and Korean 조사 (postposition) stripping. No model, no
11
+ network, no randomness — the same input always produces the same node id.
12
+
13
+ ## 조사 stripping, and why it is two tiers
14
+
15
+ Korean marks grammatical role with a suffix glued to the noun, so the *same*
16
+ entity appears as ``플랫폼은`` / ``플랫폼을`` / ``플랫폼에서``. Stripping the
17
+ suffix is what merges them. But Korean nouns also legitimately *end* in those
18
+ syllables — ``고양이`` ends in ``이``, ``전문가`` in ``가``, ``정확도`` in
19
+ ``도`` — and a blind strip invents ``고양`` / ``전문`` / ``정확``. That is worse
20
+ than the duplicate it was trying to fix, because a wrong node id cannot be
21
+ undone by a later read.
22
+
23
+ So:
24
+
25
+ * **Tier 1 — unconditional.** Multi-syllable particles (``에서``, ``으로``,
26
+ ``에게``, ``부터``, ``까지``, ``보다``, ``처럼`` …) plus ``을``/``를``.
27
+ Korean nouns essentially never end in these *as their own last syllables*
28
+ once a two-character stem is required, so no corroboration is needed.
29
+ * **Tier 2 — evidence-gated.** The single-syllable particles that collide with
30
+ real noun endings (``은 는 이 가 의 와 과 도 로 만 나``) are stripped only
31
+ when the source text itself shows the bare stem somewhere else — that is,
32
+ the text contains the stem *not* followed by this particle. With no text to
33
+ corroborate (an LLM-supplied concept, say) nothing is stripped: an
34
+ unmerged duplicate is recoverable, an invented stem is not.
35
+
36
+ Every stem must keep at least :data:`MIN_STEM_CHARS` characters, which by
37
+ itself rejects the whole ``결과 → 결`` / ``회의 → 회`` class.
38
+ """
39
+
40
+ from __future__ import annotations
41
+
42
+ import functools
43
+ import re
44
+ import unicodedata
45
+ from typing import Dict, Iterable, List, Sequence, Tuple
46
+
47
+ #: A stripped stem shorter than this is never accepted — ``결과`` must not
48
+ #: become ``결``. Two characters is the shortest real Korean noun stem.
49
+ MIN_STEM_CHARS = 2
50
+
51
+ #: Particles stripped without asking the text. Longest first: ``에서는`` has to
52
+ #: be tried before ``에서`` or the leftover ``는`` stays glued on.
53
+ UNCONDITIONAL_PARTICLES: Tuple[str, ...] = (
54
+ "에서는",
55
+ "으로는",
56
+ "에게는",
57
+ "로부터",
58
+ "이라는",
59
+ "이라고",
60
+ "에서도",
61
+ "으로도",
62
+ "에서의",
63
+ "으로서",
64
+ "으로써",
65
+ "에게서",
66
+ "에게도",
67
+ "라고는",
68
+ "만큼은",
69
+ "에서",
70
+ "에게",
71
+ "한테",
72
+ "께서",
73
+ "으로",
74
+ "부터",
75
+ "까지",
76
+ "보다",
77
+ "처럼",
78
+ "만큼",
79
+ "마다",
80
+ "조차",
81
+ "밖에",
82
+ "라는",
83
+ "라고",
84
+ "라도",
85
+ "이나",
86
+ "을",
87
+ "를",
88
+ )
89
+
90
+ #: Particles that collide with real noun endings — stripped only when the text
91
+ #: shows the bare stem elsewhere (see :func:`strip_particle`).
92
+ EVIDENCE_PARTICLES: Tuple[str, ...] = (
93
+ "은",
94
+ "는",
95
+ "이",
96
+ "가",
97
+ "의",
98
+ "와",
99
+ "과",
100
+ "도",
101
+ "로",
102
+ "만",
103
+ "나",
104
+ )
105
+
106
+ #: Opening → closing delimiters unwrapped when a term is wrapped in a *matched*
107
+ #: pair. Matched only: ``<data_dir>/cloud_provider.json`` opens a bracket it
108
+ #: never closes, and stripping the lone ``<`` invents a name nobody wrote.
109
+ _DELIMITER_PAIRS = {
110
+ "(": ")",
111
+ "[": "]",
112
+ "{": "}",
113
+ "<": ">",
114
+ "«": "»",
115
+ "“": "”",
116
+ "‘": "’",
117
+ "「": "」",
118
+ "『": "』",
119
+ "《": "》",
120
+ "〈": "〉",
121
+ '"': '"',
122
+ "'": "'",
123
+ "`": "`",
124
+ }
125
+ #: Trailing punctuation trimmed after the brackets come off.
126
+ _TRAILING_PUNCT = ".,;:!?…·~-–—/\\|"
127
+
128
+ _WHITESPACE = re.compile(r"\s+")
129
+ _HANGUL = re.compile(r"[가-힣]")
130
+ _POSSESSIVE = re.compile(r"(?:['’]s|['’])$", re.IGNORECASE)
131
+
132
+
133
+ def _is_hangul_tail(text: str) -> bool:
134
+ """True when the last character is a Hangul syllable."""
135
+ return bool(text) and bool(_HANGUL.fullmatch(text[-1]))
136
+
137
+
138
+ def strip_particle(term: str, text: str = "") -> str:
139
+ """``term`` with one trailing Korean particle removed, when that is safe.
140
+
141
+ ``text`` is the passage the term came from; it is the corroboration for
142
+ the Tier-2 particles. Pass ``""`` and only Tier 1 fires.
143
+
144
+ >>> strip_particle("플랫폼에서")
145
+ '플랫폼'
146
+ >>> strip_particle("고양이") # no evidence → left alone
147
+ '고양이'
148
+ >>> strip_particle("플랫폼이", "플랫폼이 있고 플랫폼도 있다")
149
+ '플랫폼'
150
+ """
151
+ term = term.strip()
152
+ if not _is_hangul_tail(term):
153
+ return term
154
+ for particle in UNCONDITIONAL_PARTICLES:
155
+ if term.endswith(particle):
156
+ stem = term[: -len(particle)]
157
+ if len(stem) >= MIN_STEM_CHARS and _is_hangul_tail(stem):
158
+ return stem
159
+ if not text:
160
+ return term
161
+ for particle in EVIDENCE_PARTICLES:
162
+ if not term.endswith(particle):
163
+ continue
164
+ stem = term[: -len(particle)]
165
+ if len(stem) < MIN_STEM_CHARS or not _is_hangul_tail(stem):
166
+ continue
167
+ if _stem_stands_alone(stem, particle, text):
168
+ return stem
169
+ return term
170
+
171
+
172
+ @functools.lru_cache(maxsize=8192)
173
+ def _stem_alone_re(stem: str, particle: str) -> re.Pattern[str]:
174
+ """Compiled ``stem`` not-followed-by ``particle`` look-ahead."""
175
+ return re.compile(re.escape(stem) + "(?!" + re.escape(particle) + ")")
176
+
177
+
178
+ def _stem_stands_alone(stem: str, particle: str, text: str) -> bool:
179
+ """True when ``text`` contains ``stem`` *not* followed by ``particle``.
180
+
181
+ That is the whole evidence test: if the passage only ever writes
182
+ ``정확도``, the trailing ``도`` is part of the word. If it also writes
183
+ ``정확`` on its own (or ``정확을``, ``정확에서`` …), the ``도`` was a
184
+ particle after all.
185
+ """
186
+ return _stem_alone_re(stem, particle).search(text) is not None
187
+
188
+
189
+ def normalize_entity(term: str, text: str = "") -> str:
190
+ """The canonical *surface* form of one extracted entity.
191
+
192
+ NFKC (so ``AI`` and ``AI`` are one word), whitespace collapsed to single
193
+ spaces, brackets/quotes/trailing punctuation trimmed, English possessive
194
+ dropped, and one Korean particle stripped when :func:`strip_particle`
195
+ considers it safe. Returns ``""`` for anything that normalizes away.
196
+
197
+ >>> normalize_entity(" “Lattice AI” ")
198
+ 'Lattice AI'
199
+ >>> normalize_entity("Anthropic's")
200
+ 'Anthropic'
201
+ >>> normalize_entity("지식그래프에서")
202
+ '지식그래프'
203
+ """
204
+ cleaned = unicodedata.normalize("NFKC", str(term or ""))
205
+ cleaned = _WHITESPACE.sub(" ", cleaned).strip()
206
+ # Matched wrappers come off, then trailing sentence punctuation; looped so
207
+ # `("Lattice AI").` unwraps fully rather than one layer at a time.
208
+ for _ in range(3):
209
+ before = cleaned
210
+ if len(cleaned) > 2 and cleaned.endswith(_DELIMITER_PAIRS.get(cleaned[0], "\0")):
211
+ cleaned = cleaned[1:-1].strip()
212
+ cleaned = cleaned.rstrip(_TRAILING_PUNCT).strip()
213
+ if cleaned == before:
214
+ break
215
+ if not cleaned:
216
+ return ""
217
+ cleaned = _POSSESSIVE.sub("", cleaned).strip()
218
+ return strip_particle(cleaned, text)
219
+
220
+
221
+ def entity_key(term: str) -> str:
222
+ """The dedup key two surface forms of one entity share.
223
+
224
+ Case-folded and whitespace-collapsed, and with the separators that
225
+ identifier styles disagree about (space, ``-``, ``_``) removed — so
226
+ ``Graph RAG`` / ``graph-rag`` / ``graph_rag`` are one key while
227
+ ``GraphRAG`` (already joined) matches them too.
228
+ """
229
+ folded = unicodedata.normalize("NFKC", str(term or "")).casefold()
230
+ folded = _WHITESPACE.sub(" ", folded).strip()
231
+ return re.sub(r"[\s_\-]+", "", folded)
232
+
233
+
234
+ def _representative(surfaces: Sequence[str], text: str) -> str:
235
+ """Pick the surface form an entity's node should carry.
236
+
237
+ Most frequent in the source text wins — the way the author actually writes
238
+ it. Ties go to the form with more capitalized words (``Lattice AI`` over
239
+ ``lattice ai``), then to the longer form, then to the first one seen, so
240
+ the choice never depends on dict ordering.
241
+ """
242
+ best = surfaces[0]
243
+ best_rank = (-1, -1, -1, 0)
244
+ for index, surface in enumerate(surfaces):
245
+ rank = (
246
+ text.count(surface) if text else 0,
247
+ sum(1 for word in surface.split() if word[:1].isupper()),
248
+ len(surface),
249
+ -index,
250
+ )
251
+ if rank > best_rank:
252
+ best_rank = rank
253
+ best = surface
254
+ return best
255
+
256
+
257
+ def merge_entity_aliases(terms: Iterable[str], text: str = "") -> List[str]:
258
+ """Normalize every term, drop the empties, and merge exact-key aliases.
259
+
260
+ Order is the order the *first* member of each group appeared in, so the
261
+ caller's own priority (backticked terms first, then proper nouns, …) is
262
+ preserved. The value is the group's representative surface form.
263
+
264
+ >>> merge_entity_aliases(["Lattice AI", "lattice ai", "Graph RAG"])
265
+ ['Lattice AI', 'Graph RAG']
266
+ """
267
+ groups: Dict[str, List[str]] = {}
268
+ order: List[str] = []
269
+ for term in terms:
270
+ surface = normalize_entity(term, text)
271
+ if not surface:
272
+ continue
273
+ key = entity_key(surface)
274
+ if not key:
275
+ continue
276
+ if key not in groups:
277
+ groups[key] = []
278
+ order.append(key)
279
+ if surface not in groups[key]:
280
+ groups[key].append(surface)
281
+ return [_representative(groups[key], text) for key in order]
282
+
283
+
284
+ def occurrence_count(term: str, text: str) -> int:
285
+ """How many times ``term`` appears in ``text``, case-insensitively.
286
+
287
+ The number an edge's ``occurrences`` metadata carries. Counted on the
288
+ normalized surface, so ``플랫폼은`` and ``플랫폼을`` both count toward
289
+ ``플랫폼``. Zero-length terms count zero rather than raising.
290
+ """
291
+ if not term or not text:
292
+ return 0
293
+ return len(re.findall(re.escape(term), text, re.IGNORECASE))
294
+
295
+
296
+ __all__ = [
297
+ "EVIDENCE_PARTICLES",
298
+ "MIN_STEM_CHARS",
299
+ "UNCONDITIONAL_PARTICLES",
300
+ "entity_key",
301
+ "merge_entity_aliases",
302
+ "normalize_entity",
303
+ "occurrence_count",
304
+ "strip_particle",
305
+ ]
@@ -0,0 +1,275 @@
1
+ """Directed, typed relation patterns — the part of extraction that reads syntax.
2
+
3
+ ``infer_edge_relation`` classifies a *sentence*: it finds a verb anywhere in it
4
+ and stamps every concept pair in that sentence with the same label, in text
5
+ order. That is cheap and it is often wrong about **direction**, and it cannot
6
+ tell a definition from a passing mention.
7
+
8
+ The four rules here look at where each concept sits inside the sentence and at
9
+ what stands *between* the two, so ``A는 B를 사용한다`` and ``B is used by A``
10
+ come out with the same subject. Each rule returns the relation label the graph
11
+ should carry, the evidence class, and a weight:
12
+
13
+ | rule | label | `edges_v2` type | evidence | weight |
14
+ |---|---|---|---|---|
15
+ | definition | `설명함` | `MENTIONS` | `definition` | 1.0 |
16
+ | SVO / SOV | the matched `EDGE_VERB` label | that label's type | `verb` | 1.0 |
17
+ | part-of | `구성요소` | `PART_OF` | `structure` | 0.9 |
18
+ | contrast | `상충함` | `CONTRADICTS` | `contrast` | 0.9 |
19
+
20
+ ``PART_OF`` and ``CONTRADICTS`` were reachable in the taxonomy but nothing
21
+ extracted them; the graph only ever produced the ten labels `EDGE_VERB` names.
22
+
23
+ Everything is regex over a single sentence — deterministic, no model, and the
24
+ same input always yields the same edge.
25
+ """
26
+
27
+ from __future__ import annotations
28
+
29
+ import re
30
+ from typing import Any, Dict, List, Optional, Sequence, Tuple
31
+
32
+ from .relations import _EDGE_VERB_COMPILED
33
+
34
+ #: Weight for a relation a syntactic pattern named outright.
35
+ PATTERN_EDGE_WEIGHT = 0.9
36
+ #: Weight for definition and verb-anchored relations — the strongest evidence.
37
+ STRONG_EDGE_WEIGHT = 1.0
38
+
39
+ #: `A is a B` / `A refers to B` — the copula and its cousins, English.
40
+ _EN_DEFINITION = re.compile(
41
+ r"^\s*(?:is|are|was|were)\s+(?:a|an|the)?\s*$"
42
+ r"|^\s*(?:refers?\s+to|means?|stands\s+for|is\s+defined\s+as|is\s+known\s+as"
43
+ r"|is\s+short\s+for)\s*$",
44
+ re.IGNORECASE,
45
+ )
46
+ #: `A란 B이다` / `A는 B를 의미한다` — the Korean definition tail.
47
+ _KO_DEFINITION_TAIL = re.compile(
48
+ r"(?:이다|입니다|이란다|이에요|예요|을 뜻한다|를 뜻한다|을 의미한다|를 의미한다"
49
+ r"|을 말한다|를 말한다|이라고 한다|라고 한다|이라 한다)\s*[.!?]?\s*$"
50
+ )
51
+ #: `A란`, `A이란`, `A라는 것은` — the Korean definition head marker. The
52
+ #: lookbehind and the trailing space keep the bare ``란`` from matching the
53
+ #: middle of an ordinary word (``결과란에``, ``발란스``).
54
+ _KO_DEFINITION_HEAD = re.compile(
55
+ r"(?<=[가-힣])(?:이란|란|라는 것은|라 함은)\s|(?:의 정의는|정의는)\s"
56
+ )
57
+
58
+ #: `A is part of B` — part first, whole second.
59
+ _EN_PART_FORWARD = re.compile(
60
+ r"^\s*(?:is|are|was|were)?\s*(?:a|an|the)?\s*"
61
+ r"(?:part\s+of|component\s+of|subset\s+of|member\s+of|belongs?\s+to"
62
+ r"|lives?\s+(?:in|under)|sits?\s+(?:in|under))\s*$",
63
+ re.IGNORECASE,
64
+ )
65
+ #: `B consists of A` — whole first, part second, so the edge is reversed.
66
+ #: ``contains``/``includes`` are deliberately **not** here: they already route
67
+ #: to `포함함` → `CONTAINS`, which is the right edge pointing the right way.
68
+ _EN_PART_REVERSE = re.compile(
69
+ r"^\s*(?:consists?\s+of|comprises?|is\s+made\s+(?:up\s+)?of)\s*$",
70
+ re.IGNORECASE,
71
+ )
72
+ #: `A는 B의 일부` / `A는 B에 속한다` — part first, whole second.
73
+ _KO_PART_FORWARD = re.compile(r"(?:의 일부|의 구성요소|의 하위|에 속한|에 포함된|의 부분)")
74
+ #: `A는 B로 구성된다` — whole first, part second.
75
+ _KO_PART_REVERSE = re.compile(r"(?:로 구성|으로 구성|의 하위 항목)")
76
+
77
+ #: `A unlike B` / `A instead of B` — a stated opposition, not a comparison.
78
+ _EN_CONTRAST = re.compile(
79
+ r"(?:\bunlike\b|\binstead\s+of\b|\brather\s+than\b|\bcontrary\s+to\b"
80
+ r"|\bas\s+opposed\s+to\b|\bnot\b[^.]{0,20}\bbut\b)",
81
+ re.IGNORECASE,
82
+ )
83
+ #: `A가 아니라 B` / `A 대신 B` / `A와 달리 B`.
84
+ _KO_CONTRAST = re.compile(r"(?:아니라|아닌|대신|와 달리|과 달리|반면|이 아니고|가 아니고)")
85
+
86
+ #: Subject markers: the syllable that says "this noun is the actor".
87
+ _KO_SUBJECT_MARKS: Tuple[str, ...] = ("은", "는", "이", "가", "께서")
88
+ #: Object markers: the syllable that says "this noun is acted on".
89
+ _KO_OBJECT_MARKS: Tuple[str, ...] = ("을", "를")
90
+
91
+
92
+ def _marker_after(sentence: str, end: int, markers: Sequence[str]) -> bool:
93
+ """True when one of ``markers`` sits immediately after ``end``.
94
+
95
+ Korean glues the particle to the noun with no space, so a single lookahead
96
+ character is the whole test.
97
+ """
98
+ tail = sentence[end : end + 2]
99
+ return any(tail.startswith(mark) for mark in markers)
100
+
101
+
102
+ def _verb_label(span: str) -> Optional[str]:
103
+ """The `EDGE_VERB` label whose pattern matches ``span``, if any."""
104
+ lowered = span.lower()
105
+ for label, pattern in _EDGE_VERB_COMPILED:
106
+ if pattern.search(lowered):
107
+ return label
108
+ return None
109
+
110
+
111
+ def _triple(
112
+ subject: str,
113
+ obj: str,
114
+ relation: str,
115
+ evidence: str,
116
+ weight: float,
117
+ context: str,
118
+ ) -> Dict[str, Any]:
119
+ return {
120
+ "subject": subject,
121
+ "relation": relation,
122
+ "object": obj,
123
+ "context": context[:240],
124
+ "evidence": evidence,
125
+ "weight": weight,
126
+ }
127
+
128
+
129
+ def concept_positions(
130
+ sentence: str, concepts: Sequence[str]
131
+ ) -> List[Tuple[int, str]]:
132
+ """``(offset, concept)`` for every concept present, in text order.
133
+
134
+ Case-insensitive, first occurrence only — a concept repeated in one
135
+ sentence is one participant, not two.
136
+ """
137
+ lowered = sentence.lower()
138
+ found: List[Tuple[int, str]] = []
139
+ for concept in concepts:
140
+ index = lowered.find(concept.lower())
141
+ if index >= 0:
142
+ found.append((index, concept))
143
+ found.sort(key=lambda pair: (pair[0], pair[1]))
144
+ return found
145
+
146
+
147
+ def typed_relation(
148
+ sentence: str,
149
+ left: Tuple[int, str],
150
+ right: Tuple[int, str],
151
+ adjacent: bool = True,
152
+ ) -> Optional[Dict[str, Any]]:
153
+ """The typed, directed relation between two concepts in one sentence.
154
+
155
+ ``left``/``right`` are ``(offset, concept)`` with ``left`` first in the
156
+ text. Returns ``None`` when no rule fires, which is the caller's signal to
157
+ fall back to the sentence-level co-occurrence classification.
158
+
159
+ ``adjacent=False`` means another concept sits between the two, and then
160
+ only the particle-marked Korean subject→object rule may fire. Everything
161
+ else reads the span *between* the pair, and with a third concept in there
162
+ that span describes somebody else's relation: ``A는 B가 아니라 C를 쓴다``
163
+ puts ``아니라`` between A and C without A and C being in contrast at all.
164
+ """
165
+ left_at, subject = left
166
+ right_at, obj = right
167
+ between = sentence[left_at + len(subject) : right_at]
168
+ after = sentence[right_at + len(obj) :]
169
+
170
+ if not adjacent:
171
+ return _korean_subject_object(sentence, subject, obj, after, left_at, right_at)
172
+
173
+ definition = _definition(sentence, between, subject, obj)
174
+ if definition is not None:
175
+ return definition
176
+ part_of = _part_of(between, after, subject, obj, sentence)
177
+ if part_of is not None:
178
+ return part_of
179
+ contrast = _contrast(between, subject, obj, sentence)
180
+ if contrast is not None:
181
+ return contrast
182
+ return _verb_anchored(sentence, subject, obj, between, after, left_at, right_at)
183
+
184
+
185
+ def _definition(
186
+ sentence: str, between: str, subject: str, obj: str
187
+ ) -> Optional[Dict[str, Any]]:
188
+ english = _EN_DEFINITION.match(between)
189
+ korean = bool(_KO_DEFINITION_TAIL.search(sentence)) and bool(
190
+ _KO_DEFINITION_HEAD.search(sentence)
191
+ )
192
+ if not english and not korean:
193
+ return None
194
+ return _triple(
195
+ subject, obj, "설명함", "definition", STRONG_EDGE_WEIGHT, sentence
196
+ )
197
+
198
+
199
+ def _part_of(
200
+ between: str, after: str, subject: str, obj: str, sentence: str
201
+ ) -> Optional[Dict[str, Any]]:
202
+ # English states the relation *between* the two ("A is part of B"); Korean
203
+ # glues it to the second one ("A는 B의 일부이다"), so the tail is read too —
204
+ # anchored at position zero, because the marker belongs to the noun it is
205
+ # stuck to. A match further along the tail is some *other* noun's relation.
206
+ if _EN_PART_FORWARD.match(between) or _KO_PART_FORWARD.match(after):
207
+ return _triple(
208
+ subject, obj, "구성요소", "structure", PATTERN_EDGE_WEIGHT, sentence
209
+ )
210
+ if _EN_PART_REVERSE.match(between) or _KO_PART_REVERSE.match(after):
211
+ # `B consists of A` — the *whole* was named first, so the part-of edge
212
+ # points the other way. Direction is the whole point of this module.
213
+ return _triple(
214
+ obj, subject, "구성요소", "structure", PATTERN_EDGE_WEIGHT, sentence
215
+ )
216
+ return None
217
+
218
+
219
+ def _contrast(
220
+ between: str, subject: str, obj: str, sentence: str
221
+ ) -> Optional[Dict[str, Any]]:
222
+ if _EN_CONTRAST.search(between) or _KO_CONTRAST.search(between):
223
+ return _triple(
224
+ subject, obj, "상충함", "contrast", PATTERN_EDGE_WEIGHT, sentence
225
+ )
226
+ return None
227
+
228
+
229
+ def _verb_anchored(
230
+ sentence: str,
231
+ subject: str,
232
+ obj: str,
233
+ between: str,
234
+ after: str,
235
+ left_at: int,
236
+ right_at: int,
237
+ ) -> Optional[Dict[str, Any]]:
238
+ """A verb that sits *between* the pair (SVO) or *after* it (Korean SOV).
239
+
240
+ English puts the verb between subject and object, so ``between`` naming a
241
+ verb is enough. Korean puts it last: the direction comes from the particles
242
+ instead — ``A는 … B를 사용한다`` marks A as subject and B as object, and the
243
+ verb in the tail names the relation.
244
+ """
245
+ label = _verb_label(between)
246
+ if label is not None:
247
+ return _triple(subject, obj, label, "verb", STRONG_EDGE_WEIGHT, sentence)
248
+ return _korean_subject_object(sentence, subject, obj, after, left_at, right_at)
249
+
250
+
251
+ def _korean_subject_object(
252
+ sentence: str,
253
+ subject: str,
254
+ obj: str,
255
+ after: str,
256
+ left_at: int,
257
+ right_at: int,
258
+ ) -> Optional[Dict[str, Any]]:
259
+ """``A는 … B를 <verb>`` — the particles decide, the tail verb names it."""
260
+ subject_marked = _marker_after(sentence, left_at + len(subject), _KO_SUBJECT_MARKS)
261
+ object_marked = _marker_after(sentence, right_at + len(obj), _KO_OBJECT_MARKS)
262
+ if not (subject_marked and object_marked):
263
+ return None
264
+ tail_label = _verb_label(after)
265
+ if tail_label is None:
266
+ return None
267
+ return _triple(subject, obj, tail_label, "verb", STRONG_EDGE_WEIGHT, sentence)
268
+
269
+
270
+ __all__ = [
271
+ "PATTERN_EDGE_WEIGHT",
272
+ "STRONG_EDGE_WEIGHT",
273
+ "concept_positions",
274
+ "typed_relation",
275
+ ]
@@ -32,7 +32,9 @@ EDGE_VERB = {
32
32
  "의존함": r"의존|depend|require|필요|based on",
33
33
  "설명함": r"설명|explain|describe|정의|란|이란|means",
34
34
  "비교함": r"비교|versus|vs\.?|차이|다르|compare",
35
- "사용함": r"사용|use|활용|이용|apply",
35
+ # `쓴다`/`쓰는`/`씁니다` are the everyday Korean for "uses"; without them a
36
+ # sentence that never says 사용 fell through to bare co-occurrence.
37
+ "사용함": r"사용|use|활용|이용|apply|쓴다|쓰는|씁니다|썼다",
36
38
  "연결함": r"연결|connect|통합|integrate|연동|link",
37
39
  "확장함": r"확장|extend|플러그인|plugin|addon",
38
40
  "생성함": r"생성|만들|create|generate|build|produced",
@@ -42,6 +44,13 @@ EDGE_VERB = {
42
44
  "관련됨": r"관련|related|associated|연관",
43
45
  }
44
46
 
47
+ #: Same table, compiled once. ``infer_edge_relation`` and the typed-relation
48
+ #: patterns both walk this on every pair; compiling per call was the cheap
49
+ #: half of the extraction regression.
50
+ _EDGE_VERB_COMPILED = tuple(
51
+ (label, re.compile(pattern)) for label, pattern in EDGE_VERB.items()
52
+ )
53
+
45
54
 
46
55
  # Concepts in a list-like sentence ("A, B, C, D를 사용한다") sit together by
47
56
  # enumeration, not by relation. Beyond this many concepts in one sentence, a
@@ -70,8 +79,8 @@ def infer_edge_relation(sentence: str) -> Dict[str, Any]:
70
79
  label-only output erased.
71
80
  """
72
81
  s = str(sentence or "").lower()
73
- for label, pattern in EDGE_VERB.items():
74
- if re.search(pattern, s):
82
+ for label, pattern in _EDGE_VERB_COMPILED:
83
+ if pattern.search(s):
75
84
  # "관련됨" is itself a weak, generic label: matching it by keyword
76
85
  # ("관련", "related") is still verb evidence, but nothing stronger.
77
86
  return {