ltcai 11.2.0 → 11.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +50 -53
- package/docs/CHANGELOG.md +87 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +181 -0
- package/docs/v11.5.0_RUST_COMPLETE_PLAN.md +145 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/api/index_jobs.py +145 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +14 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +421 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +2 -0
- package/scripts/chunking_parity_corpus.py +449 -0
- package/scripts/generate_agent_parity_fixtures.py +752 -0
- package/scripts/generate_chunking_parity_fixtures.py +259 -0
- package/scripts/generate_rust_parity_fixtures.py +997 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +42 -2
- package/src-tauri/Cargo.lock +404 -3
- package/src-tauri/Cargo.toml +13 -1
- package/src-tauri/src/backend.rs +460 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +109 -399
- package/src-tauri/src/topology.rs +356 -0
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-CWnxSCgN.js +1 -0
- package/static/app/assets/AdminConsole-BEQYU6kF.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-DWu1BhFg.js} +2 -2
- package/static/app/assets/BrainHome-95Hilr9R.js +2 -0
- package/static/app/assets/BrainSignals-QdeqCpAF.js +1 -0
- package/static/app/assets/Capture-BHpCxnzb.js +1 -0
- package/static/app/assets/Chronicle-B4xYKoed.js +1 -0
- package/static/app/assets/CommandPalette-BVXnttSz.js +1 -0
- package/static/app/assets/Library-DgYcHome.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-CrJLDbf7.js} +1 -1
- package/static/app/assets/ProductFlow-DFlScKoJ.js +1 -0
- package/static/app/assets/ReviewCard-Cy5f48Pj.js +3 -0
- package/static/app/assets/System-NF8IfhTa.js +1 -0
- package/static/app/assets/arrow-left-DwkSYrjR.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-CucuhLhm.js} +1 -1
- package/static/app/assets/brain-BBnSryW_.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-C2GUj2Ai.js} +1 -1
- package/static/app/assets/circle-check-CxOVPwYq.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-CbkWzBmG.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-7lEaqHdJ.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DAlCXlIy.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-RNhuuJwh.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CLW4odzM.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-NKEiDIAJ.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-DMurvUuR.js +10 -0
- package/static/app/assets/input-D2UhPC1X.js +1 -0
- package/static/app/assets/link-2-6amKbP_P.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-Cu9TZtdR.js} +1 -1
- package/static/app/assets/primitives-gPsccucr.js +1 -0
- package/static/app/assets/search-Cj_TKk_2.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-Bau7KkPq.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-BufNYypi.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-BQnVWhYs.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-B3_w60si.js} +1 -1
- package/static/app/assets/useMutation-BHhCflT6.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-rBWfI-5t.js} +1 -1
- package/static/app/assets/utils-V_5-wxr5.js +4 -0
- package/static/app/assets/workspace-K1zjYUHj.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,516 @@
|
|
|
1
|
+
"""Concept and triple extraction — LLM-first, rules as the fallback.
|
|
2
|
+
|
|
3
|
+
Moved verbatim out of ``_kg_common`` (v11.3.0 decomposition). The two seams
|
|
4
|
+
tests reach for live **here**, next to the code that reads them:
|
|
5
|
+
``ENABLE_LLM_EXTRACTION`` and ``get_llm_router``. Patching them on the
|
|
6
|
+
``_kg_common`` package would rebind the package attribute and leave this
|
|
7
|
+
module's own globals untouched, so the patch target is
|
|
8
|
+
``lattice_brain.graph._kg_common.extraction``.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
# F841: `_extract_concepts_rules` builds `pattern_with_suffix` alongside the
|
|
14
|
+
# pattern it actually matches on; it documents the pair and is left as it was
|
|
15
|
+
# under the module-wide directive `_kg_common.py` carried before the split.
|
|
16
|
+
# ruff: noqa: F841
|
|
17
|
+
import asyncio
|
|
18
|
+
import json
|
|
19
|
+
import logging
|
|
20
|
+
import os
|
|
21
|
+
import re
|
|
22
|
+
from typing import Any, Dict, List, Optional
|
|
23
|
+
|
|
24
|
+
from ..runtime import get_llm_router
|
|
25
|
+
from .relations import (
|
|
26
|
+
COOCCURRENCE_CONCEPT_LIMIT,
|
|
27
|
+
COOCCURRENCE_EDGE_WEIGHT,
|
|
28
|
+
VERB_EDGE_WEIGHT,
|
|
29
|
+
infer_edge_relation,
|
|
30
|
+
)
|
|
31
|
+
from .text import _clean_text
|
|
32
|
+
|
|
33
|
+
_LLM_EXTRACT_CONCEPT_PROMPT = """Extract the key concepts from the following text.
|
|
34
|
+
Return ONLY a JSON array of objects, each with "concept" (string) and "importance" (float 0-1).
|
|
35
|
+
Extract up to {limit} concepts. Focus on named entities, technical terms, and domain-specific nouns.
|
|
36
|
+
Do NOT include common words, stop words, or generic terms.
|
|
37
|
+
|
|
38
|
+
Text:
|
|
39
|
+
{text}
|
|
40
|
+
|
|
41
|
+
JSON:"""
|
|
42
|
+
|
|
43
|
+
_LLM_EXTRACT_TRIPLE_PROMPT = """Extract relationship triples from the following text.
|
|
44
|
+
Return ONLY a JSON array of objects, each with:
|
|
45
|
+
- "subject": source concept (string)
|
|
46
|
+
- "relation": relationship verb (string, Korean or English)
|
|
47
|
+
- "object": target concept (string)
|
|
48
|
+
- "evidence": the sentence supporting this triple (string, max 240 chars)
|
|
49
|
+
- "confidence": how confident you are (float 0-1)
|
|
50
|
+
|
|
51
|
+
Extract up to {limit} triples. Focus on meaningful semantic relationships.
|
|
52
|
+
|
|
53
|
+
Text:
|
|
54
|
+
{text}
|
|
55
|
+
|
|
56
|
+
Concepts already identified: {concepts}
|
|
57
|
+
|
|
58
|
+
JSON:"""
|
|
59
|
+
|
|
60
|
+
ENABLE_LLM_EXTRACTION = os.getenv("LATTICEAI_LLM_EXTRACTION", "true").lower() in (
|
|
61
|
+
"1",
|
|
62
|
+
"true",
|
|
63
|
+
"yes",
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _llm_extract_concepts(text: str, limit: int = 12) -> Optional[List[str]]:
|
|
68
|
+
router = get_llm_router()
|
|
69
|
+
if not ENABLE_LLM_EXTRACTION or not router:
|
|
70
|
+
return None
|
|
71
|
+
if not router.current_model_id:
|
|
72
|
+
return None
|
|
73
|
+
prompt = _LLM_EXTRACT_CONCEPT_PROMPT.format(text=text[:3000], limit=limit)
|
|
74
|
+
try:
|
|
75
|
+
loop = asyncio.get_event_loop()
|
|
76
|
+
if loop.is_running():
|
|
77
|
+
import concurrent.futures
|
|
78
|
+
|
|
79
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
80
|
+
future = pool.submit(
|
|
81
|
+
asyncio.run,
|
|
82
|
+
router.generate(prompt, max_tokens=1024, temperature=0.1),
|
|
83
|
+
)
|
|
84
|
+
raw = future.result(timeout=30)
|
|
85
|
+
else:
|
|
86
|
+
raw = asyncio.run(
|
|
87
|
+
router.generate(prompt, max_tokens=1024, temperature=0.1)
|
|
88
|
+
)
|
|
89
|
+
raw = raw.strip()
|
|
90
|
+
if raw.startswith("```"):
|
|
91
|
+
raw = re.sub(r"^```(?:json)?\s*", "", raw)
|
|
92
|
+
raw = re.sub(r"\s*```$", "", raw)
|
|
93
|
+
parsed = json.loads(raw)
|
|
94
|
+
if isinstance(parsed, list):
|
|
95
|
+
concepts = []
|
|
96
|
+
for item in parsed[:limit]:
|
|
97
|
+
if isinstance(item, dict) and "concept" in item:
|
|
98
|
+
concepts.append(item["concept"])
|
|
99
|
+
elif isinstance(item, str):
|
|
100
|
+
concepts.append(item)
|
|
101
|
+
return concepts if concepts else None
|
|
102
|
+
except Exception as e:
|
|
103
|
+
logging.debug("LLM concept extraction failed (falling back to rules): %s", e)
|
|
104
|
+
return None
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# Triples carry a numeric ``weight``/``confidence`` alongside string fields,
|
|
108
|
+
# so the value type is Any rather than str.
|
|
109
|
+
def _llm_extract_triples(
|
|
110
|
+
text: str, concepts: List[str], limit: int = 20
|
|
111
|
+
) -> Optional[List[Dict[str, Any]]]:
|
|
112
|
+
router = get_llm_router()
|
|
113
|
+
if not ENABLE_LLM_EXTRACTION or not router:
|
|
114
|
+
return None
|
|
115
|
+
if not router.current_model_id:
|
|
116
|
+
return None
|
|
117
|
+
prompt = _LLM_EXTRACT_TRIPLE_PROMPT.format(
|
|
118
|
+
text=text[:3000],
|
|
119
|
+
limit=limit,
|
|
120
|
+
concepts=", ".join(concepts[:15]),
|
|
121
|
+
)
|
|
122
|
+
try:
|
|
123
|
+
loop = asyncio.get_event_loop()
|
|
124
|
+
if loop.is_running():
|
|
125
|
+
import concurrent.futures
|
|
126
|
+
|
|
127
|
+
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
128
|
+
future = pool.submit(
|
|
129
|
+
asyncio.run,
|
|
130
|
+
router.generate(prompt, max_tokens=2048, temperature=0.1),
|
|
131
|
+
)
|
|
132
|
+
raw = future.result(timeout=30)
|
|
133
|
+
else:
|
|
134
|
+
raw = asyncio.run(
|
|
135
|
+
router.generate(prompt, max_tokens=2048, temperature=0.1)
|
|
136
|
+
)
|
|
137
|
+
raw = raw.strip()
|
|
138
|
+
if raw.startswith("```"):
|
|
139
|
+
raw = re.sub(r"^```(?:json)?\s*", "", raw)
|
|
140
|
+
raw = re.sub(r"\s*```$", "", raw)
|
|
141
|
+
parsed = json.loads(raw)
|
|
142
|
+
if isinstance(parsed, list):
|
|
143
|
+
triples: List[Dict[str, Any]] = []
|
|
144
|
+
for item in parsed[:limit]:
|
|
145
|
+
if isinstance(item, dict) and "subject" in item and "object" in item:
|
|
146
|
+
relation = str(item.get("relation", "관련됨"))
|
|
147
|
+
evidence_text = str(item.get("evidence", ""))[:240]
|
|
148
|
+
confidence = float(item.get("confidence", 0.8))
|
|
149
|
+
# An LLM triple that names a real verb and cites the
|
|
150
|
+
# sentence it came from is semantic evidence; a bare
|
|
151
|
+
# "관련됨" with no quoted evidence is the model restating
|
|
152
|
+
# co-occurrence, and is weighted (and labelled) as such.
|
|
153
|
+
is_semantic = bool(evidence_text) and relation != "관련됨"
|
|
154
|
+
triples.append(
|
|
155
|
+
{
|
|
156
|
+
"subject": str(item["subject"]),
|
|
157
|
+
"relation": relation,
|
|
158
|
+
"object": str(item["object"]),
|
|
159
|
+
"context": evidence_text,
|
|
160
|
+
"confidence": confidence,
|
|
161
|
+
"evidence": "verb" if is_semantic else "cooccurrence",
|
|
162
|
+
"weight": round(
|
|
163
|
+
(VERB_EDGE_WEIGHT if is_semantic else COOCCURRENCE_EDGE_WEIGHT)
|
|
164
|
+
* max(0.1, min(confidence, 1.0)),
|
|
165
|
+
4,
|
|
166
|
+
),
|
|
167
|
+
}
|
|
168
|
+
)
|
|
169
|
+
return triples if triples else None
|
|
170
|
+
except Exception as e:
|
|
171
|
+
logging.debug("LLM triple extraction failed (falling back to rules): %s", e)
|
|
172
|
+
return None
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
_CONCEPT_STOP: set = {
|
|
176
|
+
# English stop words
|
|
177
|
+
"the",
|
|
178
|
+
"and",
|
|
179
|
+
"for",
|
|
180
|
+
"with",
|
|
181
|
+
"this",
|
|
182
|
+
"that",
|
|
183
|
+
"from",
|
|
184
|
+
"into",
|
|
185
|
+
"which",
|
|
186
|
+
"are",
|
|
187
|
+
"was",
|
|
188
|
+
"were",
|
|
189
|
+
"has",
|
|
190
|
+
"have",
|
|
191
|
+
"had",
|
|
192
|
+
"can",
|
|
193
|
+
"will",
|
|
194
|
+
"would",
|
|
195
|
+
"could",
|
|
196
|
+
"should",
|
|
197
|
+
"may",
|
|
198
|
+
"might",
|
|
199
|
+
"must",
|
|
200
|
+
"shall",
|
|
201
|
+
"being",
|
|
202
|
+
"been",
|
|
203
|
+
"also",
|
|
204
|
+
"just",
|
|
205
|
+
"then",
|
|
206
|
+
"than",
|
|
207
|
+
"when",
|
|
208
|
+
"where",
|
|
209
|
+
"what",
|
|
210
|
+
"how",
|
|
211
|
+
"why",
|
|
212
|
+
"its",
|
|
213
|
+
"their",
|
|
214
|
+
"your",
|
|
215
|
+
"our",
|
|
216
|
+
"you",
|
|
217
|
+
"they",
|
|
218
|
+
"them",
|
|
219
|
+
"these",
|
|
220
|
+
"those",
|
|
221
|
+
"use",
|
|
222
|
+
"used",
|
|
223
|
+
"using",
|
|
224
|
+
"based",
|
|
225
|
+
"like",
|
|
226
|
+
"such",
|
|
227
|
+
"via",
|
|
228
|
+
"per",
|
|
229
|
+
"let",
|
|
230
|
+
"yes",
|
|
231
|
+
"not",
|
|
232
|
+
"but",
|
|
233
|
+
"all",
|
|
234
|
+
"any",
|
|
235
|
+
"out",
|
|
236
|
+
"new",
|
|
237
|
+
"get",
|
|
238
|
+
"set",
|
|
239
|
+
# Korean stop words
|
|
240
|
+
"사용자",
|
|
241
|
+
"내용",
|
|
242
|
+
"파일",
|
|
243
|
+
"채팅",
|
|
244
|
+
"답변",
|
|
245
|
+
"입니다",
|
|
246
|
+
"그리고",
|
|
247
|
+
"처럼",
|
|
248
|
+
"있어",
|
|
249
|
+
"없어",
|
|
250
|
+
"이야",
|
|
251
|
+
"이다",
|
|
252
|
+
"한다",
|
|
253
|
+
"하다",
|
|
254
|
+
"되다",
|
|
255
|
+
"됩니다",
|
|
256
|
+
"경우",
|
|
257
|
+
"방법",
|
|
258
|
+
"부분",
|
|
259
|
+
"상태",
|
|
260
|
+
"정도",
|
|
261
|
+
"결과",
|
|
262
|
+
"이후",
|
|
263
|
+
"이전",
|
|
264
|
+
"그것",
|
|
265
|
+
"이것",
|
|
266
|
+
"저것",
|
|
267
|
+
"여기",
|
|
268
|
+
"거기",
|
|
269
|
+
"저기",
|
|
270
|
+
"우리",
|
|
271
|
+
"저희",
|
|
272
|
+
"기능",
|
|
273
|
+
"서버",
|
|
274
|
+
"모델",
|
|
275
|
+
"설정",
|
|
276
|
+
"설명",
|
|
277
|
+
"버전",
|
|
278
|
+
"지원",
|
|
279
|
+
"사용",
|
|
280
|
+
"실행",
|
|
281
|
+
"todo",
|
|
282
|
+
"fixme",
|
|
283
|
+
"note",
|
|
284
|
+
"참고",
|
|
285
|
+
"주의",
|
|
286
|
+
"warning",
|
|
287
|
+
}
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def _extract_concepts(text: str, limit: int = 12) -> List[str]:
|
|
291
|
+
"""LLM-first concept extraction with rule-based fallback."""
|
|
292
|
+
llm_result = _llm_extract_concepts(text, limit)
|
|
293
|
+
if llm_result:
|
|
294
|
+
return llm_result
|
|
295
|
+
return _extract_concepts_rules(text, limit)
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
|
|
299
|
+
"""Extract meaningful named concepts from text (rule-based).
|
|
300
|
+
|
|
301
|
+
Priority order:
|
|
302
|
+
1. Backtick / quoted terms (explicitly technical)
|
|
303
|
+
2. Multi-word proper nouns (Lattice AI, GPT-4o, Claude Sonnet)
|
|
304
|
+
3. Single capitalized proper nouns not at sentence start (Claude, Python, FastAPI)
|
|
305
|
+
4. Korean compound technical terms (멀티모달, 에이전트, 그래프RAG)
|
|
306
|
+
5. Hyphenated / versioned identifiers (gpt-4o, mlx-vlm, gemma-4)
|
|
307
|
+
"""
|
|
308
|
+
text = str(text or "")
|
|
309
|
+
seen: dict = {} # concept_lower → original form
|
|
310
|
+
|
|
311
|
+
def _add(term: str) -> None:
|
|
312
|
+
key = term.strip().lower()
|
|
313
|
+
if key and key not in _CONCEPT_STOP and not key.isdigit() and len(key) >= 2:
|
|
314
|
+
seen.setdefault(key, term.strip())
|
|
315
|
+
|
|
316
|
+
# 1. Backtick-quoted code/term (highest confidence)
|
|
317
|
+
for m in re.findall(r"`([^`]{2,40})`", text):
|
|
318
|
+
if not re.search(r"[\(\)\[\]{}]", m): # skip code expressions
|
|
319
|
+
_add(m)
|
|
320
|
+
|
|
321
|
+
# 2. Double/single quoted terms
|
|
322
|
+
for m in re.findall(r'"([^"]{2,40})"', text):
|
|
323
|
+
_add(m)
|
|
324
|
+
|
|
325
|
+
# 3. Multi-word English proper nouns (Title Case or ALL-CAPS first word, 2–4 words).
|
|
326
|
+
# Pattern A: Mixed-case first word — "Lattice AI", "Tool Use", "Graph RAG"
|
|
327
|
+
for m in re.findall(
|
|
328
|
+
r"([A-Z][a-z]{1,20}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})",
|
|
329
|
+
text,
|
|
330
|
+
):
|
|
331
|
+
_add(m)
|
|
332
|
+
# Pattern B: ALL-CAPS first word — "VS Code", "MCP Server", "GPT-4o Mini"
|
|
333
|
+
for m in re.findall(
|
|
334
|
+
r"([A-Z]{2,6}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})",
|
|
335
|
+
text,
|
|
336
|
+
):
|
|
337
|
+
_add(m)
|
|
338
|
+
|
|
339
|
+
# 4. Single capitalized proper noun.
|
|
340
|
+
# Use ASCII-boundary lookaround instead of \b so Korean particles
|
|
341
|
+
# (와, 의, 는 …) after an English word don't block the match.
|
|
342
|
+
all_caps_words = re.findall(
|
|
343
|
+
r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])", text
|
|
344
|
+
)
|
|
345
|
+
freq: Dict[str, int] = {}
|
|
346
|
+
for w in all_caps_words:
|
|
347
|
+
freq[w] = freq.get(w, 0) + 1
|
|
348
|
+
sentence_starts = set(re.findall(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)", text))
|
|
349
|
+
for m, cnt in freq.items():
|
|
350
|
+
if m.lower() in _CONCEPT_STOP:
|
|
351
|
+
continue
|
|
352
|
+
if cnt >= 2 or m not in sentence_starts:
|
|
353
|
+
_add(m)
|
|
354
|
+
|
|
355
|
+
# 5. Korean technical compound nouns (3–12 chars, no common particles)
|
|
356
|
+
for m in re.findall(
|
|
357
|
+
r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)",
|
|
358
|
+
text,
|
|
359
|
+
):
|
|
360
|
+
_add(m)
|
|
361
|
+
# Korean standalone terms that appear after topic markers (은/는/이/가 앞)
|
|
362
|
+
for m in re.findall(
|
|
363
|
+
r"([가-힣]{2,12})(?:은|는|이|가|을|를|의|에서|으로|와|과)", text
|
|
364
|
+
):
|
|
365
|
+
if m.lower() not in _CONCEPT_STOP and len(m) >= 2:
|
|
366
|
+
# Only add if it's non-trivial (has 3+ chars or appears multiple times)
|
|
367
|
+
cnt = text.count(m)
|
|
368
|
+
if len(m) >= 3 or cnt >= 2:
|
|
369
|
+
_add(m)
|
|
370
|
+
|
|
371
|
+
# 6. Hyphenated / versioned identifiers (gpt-4o, gemma-4, mlx-vlm)
|
|
372
|
+
for m in re.findall(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b", text):
|
|
373
|
+
if len(m) >= 4:
|
|
374
|
+
_add(m)
|
|
375
|
+
|
|
376
|
+
# De-duplicate: remove shorter if ALL its occurrences in the source text
|
|
377
|
+
# are followed immediately by the suffix that forms the longer concept.
|
|
378
|
+
# "Lattice" → dropped when every occurrence is "Lattice AI"
|
|
379
|
+
# "Claude" → kept because it appears as just "Claude" too.
|
|
380
|
+
values = list(seen.values())
|
|
381
|
+
values_lower = [v.lower() for v in values]
|
|
382
|
+
keep = set(range(len(values)))
|
|
383
|
+
for i, v in enumerate(values):
|
|
384
|
+
vl = v.lower()
|
|
385
|
+
for j, wl in enumerate(values_lower):
|
|
386
|
+
if i == j or j not in keep:
|
|
387
|
+
continue
|
|
388
|
+
# Check if vl is a word-prefix of wl
|
|
389
|
+
suffix = wl[len(vl) :]
|
|
390
|
+
if not (wl.startswith(vl) and re.match(r"^[\s\-]", suffix)):
|
|
391
|
+
continue
|
|
392
|
+
# Count occurrences of v NOT followed by the suffix
|
|
393
|
+
suffix_stripped = suffix.lstrip(" -")
|
|
394
|
+
# Escape for regex
|
|
395
|
+
pattern_with_suffix = re.escape(v) + r"[\s\-]+" + re.escape(suffix_stripped)
|
|
396
|
+
pattern_alone = (
|
|
397
|
+
re.escape(v) + r"(?![\s\-]*" + re.escape(suffix_stripped) + r")"
|
|
398
|
+
)
|
|
399
|
+
alone_count = len(re.findall(pattern_alone, text, re.IGNORECASE))
|
|
400
|
+
if alone_count == 0:
|
|
401
|
+
# Shorter term never appears alone → safe to remove
|
|
402
|
+
keep.discard(i)
|
|
403
|
+
break
|
|
404
|
+
|
|
405
|
+
final = [values[i] for i in range(len(values)) if i in keep]
|
|
406
|
+
return final[:limit]
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def _extract_triples(
|
|
410
|
+
text: str,
|
|
411
|
+
concepts: List[str],
|
|
412
|
+
limit: int = 20,
|
|
413
|
+
) -> List[Dict[str, str]]:
|
|
414
|
+
"""LLM-first triple extraction with rule-based fallback."""
|
|
415
|
+
llm_result = _llm_extract_triples(text, concepts, limit)
|
|
416
|
+
if llm_result:
|
|
417
|
+
return llm_result
|
|
418
|
+
return _extract_triples_rules(text, concepts, limit)
|
|
419
|
+
|
|
420
|
+
|
|
421
|
+
def _extract_triples_rules(
|
|
422
|
+
text: str,
|
|
423
|
+
concepts: List[str],
|
|
424
|
+
limit: int = 20,
|
|
425
|
+
) -> List[Dict[str, str]]:
|
|
426
|
+
"""Extract (subject, verb-edge, object, context) triples from text (rule-based).
|
|
427
|
+
|
|
428
|
+
For each sentence containing ≥2 concepts, infer the verb-form edge label
|
|
429
|
+
from surrounding context and create a directed triple.
|
|
430
|
+
"""
|
|
431
|
+
if len(concepts) < 2:
|
|
432
|
+
return []
|
|
433
|
+
|
|
434
|
+
concept_lower = {c.lower(): c for c in concepts}
|
|
435
|
+
triples: List[Dict[str, str]] = []
|
|
436
|
+
seen_pairs: set = set()
|
|
437
|
+
|
|
438
|
+
# Split on sentence boundaries
|
|
439
|
+
sentences = re.split(r"(?<=[.!?\n])\s+|\n{2,}", text)
|
|
440
|
+
for sent in sentences:
|
|
441
|
+
sent = sent.strip()
|
|
442
|
+
if len(sent) < 8:
|
|
443
|
+
continue
|
|
444
|
+
sent_lower = sent.lower()
|
|
445
|
+
|
|
446
|
+
present = [concept_lower[k] for k in concept_lower if k in sent_lower]
|
|
447
|
+
if len(present) < 2:
|
|
448
|
+
continue
|
|
449
|
+
|
|
450
|
+
relation = infer_edge_relation(sent)
|
|
451
|
+
edge = relation["relation"]
|
|
452
|
+
# Enumeration guard (review 2026-07-27 P1 #6): a verb-less sentence
|
|
453
|
+
# listing many concepts is a list, not a set of relations. Verb-backed
|
|
454
|
+
# sentences keep every pair — the verb is the evidence.
|
|
455
|
+
if (
|
|
456
|
+
relation["evidence"] == "cooccurrence"
|
|
457
|
+
and len(present) > COOCCURRENCE_CONCEPT_LIMIT
|
|
458
|
+
):
|
|
459
|
+
continue
|
|
460
|
+
|
|
461
|
+
for i in range(len(present) - 1):
|
|
462
|
+
subj, obj = present[i], present[i + 1]
|
|
463
|
+
# Deduplicate by (subj, obj) regardless of direction for same edge
|
|
464
|
+
pair_key = tuple(sorted([subj.lower(), obj.lower()])) + (edge,)
|
|
465
|
+
if pair_key in seen_pairs:
|
|
466
|
+
continue
|
|
467
|
+
seen_pairs.add(pair_key)
|
|
468
|
+
triples.append(
|
|
469
|
+
{
|
|
470
|
+
"subject": subj,
|
|
471
|
+
"relation": edge, # verb form (동사)
|
|
472
|
+
"object": obj,
|
|
473
|
+
"context": sent[:240],
|
|
474
|
+
"evidence": relation["evidence"],
|
|
475
|
+
"weight": relation["weight"],
|
|
476
|
+
}
|
|
477
|
+
)
|
|
478
|
+
if len(triples) >= limit:
|
|
479
|
+
return triples
|
|
480
|
+
|
|
481
|
+
return triples
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _semantic_items(text: str) -> List[Dict[str, str]]:
|
|
485
|
+
"""Extract explicit decision / task items from text."""
|
|
486
|
+
items: List[Dict[str, str]] = []
|
|
487
|
+
for raw_line in str(text or "").splitlines():
|
|
488
|
+
line = _clean_text(raw_line)
|
|
489
|
+
if len(line) < 6:
|
|
490
|
+
continue
|
|
491
|
+
lowered = line.lower()
|
|
492
|
+
if re.search(r"(결정|확정|하기로|decided|decision)", lowered):
|
|
493
|
+
items.append(
|
|
494
|
+
{"type": "Decision", "title": line[:120], "summary": line[:500]}
|
|
495
|
+
)
|
|
496
|
+
if re.search(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])", lowered):
|
|
497
|
+
items.append({"type": "Task", "title": line[:120], "summary": line[:500]})
|
|
498
|
+
return items[:8]
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def _topic_candidates(text: str, limit: int = 8) -> List[str]:
|
|
502
|
+
"""Return compact keyword candidates for fallback graph search."""
|
|
503
|
+
candidates = _extract_concepts(text, limit=limit)
|
|
504
|
+
if candidates:
|
|
505
|
+
return candidates[:limit]
|
|
506
|
+
seen: Dict[str, str] = {}
|
|
507
|
+
for token in re.findall(
|
|
508
|
+
r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}", str(text or "")
|
|
509
|
+
):
|
|
510
|
+
key = token.lower()
|
|
511
|
+
if key in _CONCEPT_STOP or key.isdigit():
|
|
512
|
+
continue
|
|
513
|
+
seen.setdefault(key, token)
|
|
514
|
+
if len(seen) >= limit:
|
|
515
|
+
break
|
|
516
|
+
return list(seen.values())[:limit]
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Edge-relation vocabulary and node-type classification.
|
|
2
|
+
|
|
3
|
+
The verb table (``EDGE_VERB``), the evidence weights that tell a semantic
|
|
4
|
+
relation from bare co-occurrence, and the concept → node-type classifier.
|
|
5
|
+
Moved verbatim out of ``_kg_common`` (v11.3.0 decomposition); depends on
|
|
6
|
+
nothing else in the package.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any, Dict
|
|
13
|
+
|
|
14
|
+
# ──────────────────────────────────────────────────────────────────────────────
|
|
15
|
+
# Node type taxonomy (점 = 명사)
|
|
16
|
+
# ──────────────────────────────────────────────────────────────────────────────
|
|
17
|
+
# Chat — 대화 세션
|
|
18
|
+
# Document — 파일 (PDF·PPT·Word·Excel·이미지 등)
|
|
19
|
+
# Concept — 개념·아이디어·기술 용어
|
|
20
|
+
# Person — 사람 (사용자, 언급된 인물)
|
|
21
|
+
# Error — 오류·버그·예외
|
|
22
|
+
# Code — 코드 스니펫·함수·클래스
|
|
23
|
+
# Feature — 소프트웨어 기능
|
|
24
|
+
# Task — 할 일·액션 아이템
|
|
25
|
+
# Decision — 결정 사항
|
|
26
|
+
|
|
27
|
+
# Edge type vocabulary (선 = 동사 — 과거형 서술어)
|
|
28
|
+
EDGE_VERB = {
|
|
29
|
+
"언급함": r"언급|mention|refer|cited",
|
|
30
|
+
"포함함": r"포함|include|consist|구성|탑재|contains",
|
|
31
|
+
"해결함": r"해결|resolv|fix|수정|고쳤|closed",
|
|
32
|
+
"의존함": r"의존|depend|require|필요|based on",
|
|
33
|
+
"설명함": r"설명|explain|describe|정의|란|이란|means",
|
|
34
|
+
"비교함": r"비교|versus|vs\.?|차이|다르|compare",
|
|
35
|
+
"사용함": r"사용|use|활용|이용|apply",
|
|
36
|
+
"연결함": r"연결|connect|통합|integrate|연동|link",
|
|
37
|
+
"확장함": r"확장|extend|플러그인|plugin|addon",
|
|
38
|
+
"생성함": r"생성|만들|create|generate|build|produced",
|
|
39
|
+
"대체함": r"대체|replace|instead|alternative",
|
|
40
|
+
"지원함": r"지원|support|제공|provide|offer",
|
|
41
|
+
"발생함": r"발생|occur|throw|raise|triggered",
|
|
42
|
+
"관련됨": r"관련|related|associated|연관",
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
# Concepts in a list-like sentence ("A, B, C, D를 사용한다") sit together by
|
|
47
|
+
# enumeration, not by relation. Beyond this many concepts in one sentence, a
|
|
48
|
+
# verb-less pairing is enumeration noise and is dropped outright.
|
|
49
|
+
COOCCURRENCE_CONCEPT_LIMIT = 4
|
|
50
|
+
# Verb-backed relations carry the sentence's own evidence; co-occurrence
|
|
51
|
+
# relations carry only adjacency, so they enter the graph at a lower weight
|
|
52
|
+
# and are labelled as such.
|
|
53
|
+
VERB_EDGE_WEIGHT = 1.0
|
|
54
|
+
COOCCURRENCE_EDGE_WEIGHT = 0.35
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def infer_edge_relation(sentence: str) -> Dict[str, Any]:
|
|
58
|
+
"""Classify the relation between two concepts in one sentence.
|
|
59
|
+
|
|
60
|
+
Review 2026-07-27 P1 #6: the graph drifted toward co-occurrence because a
|
|
61
|
+
verb-less sentence still produced a "관련됨" edge indistinguishable from a
|
|
62
|
+
real semantic relation. The label alone cannot carry that difference, so
|
|
63
|
+
the evidence class rides with it::
|
|
64
|
+
|
|
65
|
+
{"relation": "사용함", "evidence": "verb", "weight": 1.0}
|
|
66
|
+
{"relation": "관련됨", "evidence": "cooccurrence", "weight": 0.35}
|
|
67
|
+
|
|
68
|
+
``evidence`` is what the graph, the curator, and the UI use to tell a
|
|
69
|
+
meaning edge from an adjacency edge — the honest distinction the previous
|
|
70
|
+
label-only output erased.
|
|
71
|
+
"""
|
|
72
|
+
s = str(sentence or "").lower()
|
|
73
|
+
for label, pattern in EDGE_VERB.items():
|
|
74
|
+
if re.search(pattern, s):
|
|
75
|
+
# "관련됨" is itself a weak, generic label: matching it by keyword
|
|
76
|
+
# ("관련", "related") is still verb evidence, but nothing stronger.
|
|
77
|
+
return {
|
|
78
|
+
"relation": label,
|
|
79
|
+
"evidence": "verb",
|
|
80
|
+
"weight": VERB_EDGE_WEIGHT,
|
|
81
|
+
}
|
|
82
|
+
return {
|
|
83
|
+
"relation": "관련됨",
|
|
84
|
+
"evidence": "cooccurrence",
|
|
85
|
+
"weight": COOCCURRENCE_EDGE_WEIGHT,
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _infer_edge(sentence: str) -> str:
|
|
90
|
+
"""Back-compat wrapper: the verb label only (see :func:`infer_edge_relation`)."""
|
|
91
|
+
return infer_edge_relation(sentence)["relation"]
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# Technical words that cannot be person names
|
|
95
|
+
_NOT_PERSON_WORDS: set = {
|
|
96
|
+
"use",
|
|
97
|
+
"api",
|
|
98
|
+
"rag",
|
|
99
|
+
"sdk",
|
|
100
|
+
"ide",
|
|
101
|
+
"cli",
|
|
102
|
+
"llm",
|
|
103
|
+
"mcp",
|
|
104
|
+
"ui",
|
|
105
|
+
"ux",
|
|
106
|
+
"new",
|
|
107
|
+
"old",
|
|
108
|
+
"get",
|
|
109
|
+
"set",
|
|
110
|
+
"run",
|
|
111
|
+
"add",
|
|
112
|
+
"fix",
|
|
113
|
+
"tool",
|
|
114
|
+
"code",
|
|
115
|
+
"base",
|
|
116
|
+
"core",
|
|
117
|
+
"data",
|
|
118
|
+
"file",
|
|
119
|
+
"test",
|
|
120
|
+
"type",
|
|
121
|
+
"mode",
|
|
122
|
+
"view",
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _classify_node_type(concept: str, text: str) -> str:
|
|
127
|
+
"""Classify a concept into the node taxonomy.
|
|
128
|
+
|
|
129
|
+
Term-level signals take priority; then a tight ±60-char window is used
|
|
130
|
+
so distant keywords don't cause mis-classification.
|
|
131
|
+
"""
|
|
132
|
+
term = concept.lower()
|
|
133
|
+
|
|
134
|
+
# ── Term-level signals (highest confidence) ───────────────────────────
|
|
135
|
+
if re.search(r"(?:error|exception|traceback|오류|에러|버그)$", term, re.I):
|
|
136
|
+
return "Error"
|
|
137
|
+
if re.search(r"error|exception|err\b", term, re.I) and len(concept) < 30:
|
|
138
|
+
return "Error"
|
|
139
|
+
if re.search(r"\(\)|\.py$|\.js$|\.ts$|\.go$|::\w", term):
|
|
140
|
+
return "Code"
|
|
141
|
+
|
|
142
|
+
# Person: "First Last" pattern, neither word is a known technical term
|
|
143
|
+
if re.match(r"^[A-Z][a-z]{1,15} [A-Z][a-z]{1,15}$", concept):
|
|
144
|
+
words = term.split()
|
|
145
|
+
if not any(w in _NOT_PERSON_WORDS for w in words):
|
|
146
|
+
return "Person"
|
|
147
|
+
|
|
148
|
+
# ── Windowed context (±60 chars) — NOT used for Error to avoid false positives
|
|
149
|
+
idx = text.lower().find(term)
|
|
150
|
+
if idx >= 0:
|
|
151
|
+
win = text[max(0, idx - 60) : idx + len(concept) + 60].lower()
|
|
152
|
+
if re.search(r"def |class |function|함수|클래스|메서드|import", win):
|
|
153
|
+
return "Code"
|
|
154
|
+
# Feature: concept appears DIRECTLY adjacent to 기능/feature keyword
|
|
155
|
+
if len(concept) <= 12 and re.search(
|
|
156
|
+
rf"{re.escape(term)}.{{0,8}}(?:기능|feature)|(?:기능|feature).{{0,8}}{re.escape(term)}",
|
|
157
|
+
win,
|
|
158
|
+
):
|
|
159
|
+
return "Feature"
|
|
160
|
+
|
|
161
|
+
return "Concept"
|