ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -1,1331 +0,0 @@
|
|
|
1
|
-
"""
|
|
2
|
-
SQLite knowledge graph for Lattice AI workspace memory.
|
|
3
|
-
|
|
4
|
-
The graph keeps raw event JSON, normalized node metadata, and edges in one
|
|
5
|
-
portable database so it can later migrate to Neo4j/Postgres without changing
|
|
6
|
-
the ingestion contract.
|
|
7
|
-
"""
|
|
8
|
-
|
|
9
|
-
# ruff: noqa: F401,F841
|
|
10
|
-
|
|
11
|
-
import asyncio
|
|
12
|
-
import hashlib
|
|
13
|
-
import json
|
|
14
|
-
import logging
|
|
15
|
-
import math
|
|
16
|
-
import os
|
|
17
|
-
import platform
|
|
18
|
-
import re
|
|
19
|
-
import shutil
|
|
20
|
-
import sqlite3
|
|
21
|
-
import time
|
|
22
|
-
import zipfile
|
|
23
|
-
from collections import Counter
|
|
24
|
-
from contextlib import contextmanager
|
|
25
|
-
from datetime import datetime
|
|
26
|
-
from pathlib import Path
|
|
27
|
-
from typing import Any, Dict, Iterable, Iterator, List, Optional, Tuple
|
|
28
|
-
|
|
29
|
-
try:
|
|
30
|
-
from .schema import EdgeType, KGStoreV2, NodeType, _exec_script
|
|
31
|
-
except Exception: # pragma: no cover - v2 schema is optional at import time
|
|
32
|
-
KGStoreV2 = None # type: ignore[assignment,misc]
|
|
33
|
-
NodeType = None # type: ignore[assignment,misc]
|
|
34
|
-
EdgeType = None # type: ignore[assignment,misc]
|
|
35
|
-
_exec_script = None # type: ignore[assignment]
|
|
36
|
-
|
|
37
|
-
from ..embeddings import LocalEmbeddingModel
|
|
38
|
-
from .json_utils import _json, _safe_loads
|
|
39
|
-
from .runtime import get_llm_router, set_llm_router
|
|
40
|
-
|
|
41
|
-
# Default read source for the graph queries: v2 reconstruction views.
|
|
42
|
-
# Override with LATTICEAI_KG_READ_V2=0 to fall back to the legacy tables.
|
|
43
|
-
_READ_FROM_V2_DEFAULT = os.getenv("LATTICEAI_KG_READ_V2", "1") != "0"
|
|
44
|
-
|
|
45
|
-
# Static constants (projection/format versions, local-ingestion classification
|
|
46
|
-
# tables, OS exclusion lists) live in ._kg_constants; re-exported here so every
|
|
47
|
-
# existing ``from ._kg_common import <CONST>`` site is unaffected.
|
|
48
|
-
from ..quiet import quiet
|
|
49
|
-
from ._kg_constants import ( # noqa: E402
|
|
50
|
-
_KG_DB_FORMAT_KEY,
|
|
51
|
-
_KG_DB_FORMAT_VERSION,
|
|
52
|
-
_PROJECTION_VERSION,
|
|
53
|
-
_V2_WRITE_MASTER_KEY,
|
|
54
|
-
COMMON_EXCLUDED_DIRS,
|
|
55
|
-
COMMON_EXCLUDED_FILE_NAMES,
|
|
56
|
-
COMMON_EXCLUDED_FILE_SUFFIXES,
|
|
57
|
-
GRAPH_SCHEMA_VERSION,
|
|
58
|
-
LINUX_EXCLUDED_PREFIXES,
|
|
59
|
-
LOCAL_CODE_EXTENSIONS,
|
|
60
|
-
LOCAL_DOCUMENT_EXTENSIONS,
|
|
61
|
-
LOCAL_IMAGE_EXTENSIONS,
|
|
62
|
-
LOCAL_SIZE_LIMITS,
|
|
63
|
-
LOCAL_SLIDE_EXTENSIONS,
|
|
64
|
-
LOCAL_SPREADSHEET_EXTENSIONS,
|
|
65
|
-
LOCAL_SUPPORTED_EXTENSIONS,
|
|
66
|
-
LOCAL_TEXT_EXTENSIONS,
|
|
67
|
-
MACOS_EXCLUDED_PREFIXES,
|
|
68
|
-
SENSITIVE_PATH_KEYWORDS,
|
|
69
|
-
WINDOWS_EXCLUDED_NAMES,
|
|
70
|
-
)
|
|
71
|
-
|
|
72
|
-
# Pure fs/path/hash/classification helpers → ._kg_fsutil, re-exported so the
|
|
73
|
-
# static __all__ below forwards them to the graph mixins. Listed explicitly
|
|
74
|
-
# rather than star-imported: a star import here made every name in this
|
|
75
|
-
# module unverifiable to both ruff and mypy.
|
|
76
|
-
from ._kg_fsutil import ( # noqa: E402,F401
|
|
77
|
-
_current_os_type,
|
|
78
|
-
_drive_id_for_path,
|
|
79
|
-
_excluded_directory_reason,
|
|
80
|
-
_file_category,
|
|
81
|
-
_is_hidden_path,
|
|
82
|
-
_is_relative_to,
|
|
83
|
-
_node_type_for_category,
|
|
84
|
-
_now,
|
|
85
|
-
_parse_iso,
|
|
86
|
-
_parser_type_for_category,
|
|
87
|
-
_path_fingerprint,
|
|
88
|
-
_path_parts_lower,
|
|
89
|
-
_recency_score,
|
|
90
|
-
_root_warning,
|
|
91
|
-
_safe_iso_from_stat_mtime,
|
|
92
|
-
_sample_file,
|
|
93
|
-
_sensitive_file_reason,
|
|
94
|
-
_sha256_bytes,
|
|
95
|
-
_sha256_text,
|
|
96
|
-
_size_limit_for_category,
|
|
97
|
-
_slug,
|
|
98
|
-
)
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
def _clean_text(text: str) -> str:
|
|
102
|
-
return re.sub(r"\s+", " ", str(text or "")).strip()
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
def _chunks(text: str, size: int = 1200, overlap: int = 160) -> List[str]:
|
|
106
|
-
cleaned = str(text or "").strip()
|
|
107
|
-
if not cleaned:
|
|
108
|
-
return []
|
|
109
|
-
chunks: List[str] = []
|
|
110
|
-
start = 0
|
|
111
|
-
while start < len(cleaned):
|
|
112
|
-
end = min(len(cleaned), start + size)
|
|
113
|
-
chunks.append(cleaned[start:end])
|
|
114
|
-
if end >= len(cleaned):
|
|
115
|
-
break
|
|
116
|
-
start = max(0, end - overlap)
|
|
117
|
-
return chunks
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
# ── Typed chunking (review 2026-07-25 §5.2 S2 — Wave 2.1 + 2.4) ──────────────
|
|
121
|
-
# ``_chunks`` above is a compatibility contract (chunk ids hash over the chunk
|
|
122
|
-
# text) and stays byte-for-byte untouched. ``typed_chunks`` layers strategy-
|
|
123
|
-
# aware boundaries plus per-chunk provenance (start_char / heading_path) on
|
|
124
|
-
# top; ``strategy="plain"`` reproduces the exact ``_chunks`` boundaries so
|
|
125
|
-
# unchanged plain content keeps identical chunk ids.
|
|
126
|
-
|
|
127
|
-
_MARKDOWN_CHUNK_EXTENSIONS = {".md", ".markdown"}
|
|
128
|
-
_CODE_CHUNK_EXTENSIONS = {
|
|
129
|
-
".py", ".js", ".jsx", ".ts", ".tsx", ".go", ".rs", ".java", ".rb",
|
|
130
|
-
".c", ".h", ".cpp", ".css", ".sh", ".sql", ".vue", ".svelte",
|
|
131
|
-
".json", ".yaml", ".yml", ".toml",
|
|
132
|
-
}
|
|
133
|
-
_PROSE_CHUNK_EXTENSIONS = {
|
|
134
|
-
".txt", ".pdf", ".docx", ".doc", ".rtf", ".odt", ".epub", ".html", ".htm",
|
|
135
|
-
}
|
|
136
|
-
_CHUNK_STRATEGIES = {"plain", "markdown", "code", "prose"}
|
|
137
|
-
# Markdown sections smaller than this merge forward into the next section so
|
|
138
|
-
# heading-dense documents don't shatter into confetti chunks.
|
|
139
|
-
_MARKDOWN_MIN_SECTION_CHARS = 200
|
|
140
|
-
_MARKDOWN_HEADING_RE = re.compile(r"^(#{1,6}) (.*)$", re.MULTILINE)
|
|
141
|
-
_CODE_BOUNDARY_LINE_RE = re.compile(
|
|
142
|
-
r"^(?:def |class |function |export |const |public |private )", re.MULTILINE
|
|
143
|
-
)
|
|
144
|
-
_CODE_BLANK_RUN_RE = re.compile(r"\n\s*\n")
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
def chunk_strategy_for(filename: Any, *, content_type: str = "") -> str:
|
|
148
|
-
"""Route a filename / path / URI (plus optional MIME hint) to a strategy.
|
|
149
|
-
|
|
150
|
-
Returns ``"markdown"`` for .md/.markdown, ``"code"`` for known source-code
|
|
151
|
-
extensions, ``"prose"`` for document formats whose text is running prose
|
|
152
|
-
(.txt/.pdf/.docx/.html/…), ``"plain"`` otherwise. Case-insensitive,
|
|
153
|
-
tolerant of URLs (query/fragment stripped) and ``Path`` objects; never
|
|
154
|
-
raises — any malformed input falls back to ``"plain"``.
|
|
155
|
-
|
|
156
|
-
Unknown/extension-less input stays ``"plain"`` on purpose: the plain
|
|
157
|
-
strategy is the byte-compatible legacy walk, and guessing prose for
|
|
158
|
-
something that might be a data dump would move chunk boundaries for no
|
|
159
|
-
retrieval gain.
|
|
160
|
-
"""
|
|
161
|
-
try:
|
|
162
|
-
name = str(filename or "").strip().lower()
|
|
163
|
-
for sep in ("?", "#"):
|
|
164
|
-
name = name.split(sep, 1)[0]
|
|
165
|
-
name = name.replace("\\", "/").rstrip("/").rsplit("/", 1)[-1]
|
|
166
|
-
dot = name.rfind(".")
|
|
167
|
-
ext = name[dot:] if dot > 0 else ""
|
|
168
|
-
if ext in _MARKDOWN_CHUNK_EXTENSIONS:
|
|
169
|
-
return "markdown"
|
|
170
|
-
if ext in _CODE_CHUNK_EXTENSIONS:
|
|
171
|
-
return "code"
|
|
172
|
-
if ext in _PROSE_CHUNK_EXTENSIONS:
|
|
173
|
-
return "prose"
|
|
174
|
-
mime = str(content_type or "").strip().lower()
|
|
175
|
-
if "markdown" in mime:
|
|
176
|
-
return "markdown"
|
|
177
|
-
if mime.startswith("text/html") or mime.startswith("text/plain"):
|
|
178
|
-
return "prose"
|
|
179
|
-
except Exception:
|
|
180
|
-
quiet()
|
|
181
|
-
return "plain"
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
def _plain_windows(
|
|
185
|
-
cleaned: str,
|
|
186
|
-
size: int,
|
|
187
|
-
overlap: int,
|
|
188
|
-
*,
|
|
189
|
-
base_offset: int = 0,
|
|
190
|
-
strategy: str = "plain",
|
|
191
|
-
heading_path: Optional[str] = None,
|
|
192
|
-
) -> List[Dict[str, Any]]:
|
|
193
|
-
"""The exact ``_chunks`` walk with ``start_char`` tracked.
|
|
194
|
-
|
|
195
|
-
Boundaries and chunk texts are byte-identical to ``_chunks`` over the same
|
|
196
|
-
string — this is the plain-strategy compatibility guarantee.
|
|
197
|
-
"""
|
|
198
|
-
out: List[Dict[str, Any]] = []
|
|
199
|
-
start = 0
|
|
200
|
-
total = len(cleaned)
|
|
201
|
-
while start < total:
|
|
202
|
-
end = min(total, start + size)
|
|
203
|
-
out.append(
|
|
204
|
-
{
|
|
205
|
-
"text": cleaned[start:end],
|
|
206
|
-
"meta": {
|
|
207
|
-
"strategy": strategy,
|
|
208
|
-
"start_char": base_offset + start,
|
|
209
|
-
"heading_path": heading_path,
|
|
210
|
-
},
|
|
211
|
-
}
|
|
212
|
-
)
|
|
213
|
-
if end >= total:
|
|
214
|
-
break
|
|
215
|
-
start = max(0, end - overlap)
|
|
216
|
-
return out
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
def _markdown_section_spans(cleaned: str) -> List[Tuple[int, int, Optional[str]]]:
|
|
220
|
-
"""``(start, end, heading_path)`` spans split at ``^#{1,6} `` heading lines.
|
|
221
|
-
|
|
222
|
-
``heading_path`` is the " > "-joined path of the enclosing headings
|
|
223
|
-
including the section's own heading (e.g. ``"Guide > Setup"``); the
|
|
224
|
-
preamble before the first heading carries ``None``. Spans are contiguous
|
|
225
|
-
raw slices of ``cleaned`` so every chunk text round-trips via start_char.
|
|
226
|
-
"""
|
|
227
|
-
spans: List[Tuple[int, int, Optional[str]]] = []
|
|
228
|
-
stack: List[Tuple[int, str]] = []
|
|
229
|
-
prev_start = 0
|
|
230
|
-
prev_path: Optional[str] = None
|
|
231
|
-
for match in _MARKDOWN_HEADING_RE.finditer(cleaned):
|
|
232
|
-
offset = match.start()
|
|
233
|
-
if offset > prev_start:
|
|
234
|
-
spans.append((prev_start, offset, prev_path))
|
|
235
|
-
level = len(match.group(1))
|
|
236
|
-
while stack and stack[-1][0] >= level:
|
|
237
|
-
stack.pop()
|
|
238
|
-
stack.append((level, match.group(2).strip()))
|
|
239
|
-
prev_start = offset
|
|
240
|
-
prev_path = " > ".join(title for _, title in stack) or None
|
|
241
|
-
if len(cleaned) > prev_start:
|
|
242
|
-
spans.append((prev_start, len(cleaned), prev_path))
|
|
243
|
-
return spans
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
def _merge_small_sections(
|
|
247
|
-
spans: List[Tuple[int, int, Optional[str]]], min_chars: int
|
|
248
|
-
) -> List[Tuple[int, int, Optional[str]]]:
|
|
249
|
-
"""Merge sections under ``min_chars`` forward into the next section.
|
|
250
|
-
|
|
251
|
-
A merged section keeps the heading_path of its first constituent (the
|
|
252
|
-
path in effect at the chunk start). A trailing undersized section merges
|
|
253
|
-
backward into the previous emitted section when one exists.
|
|
254
|
-
"""
|
|
255
|
-
merged: List[Tuple[int, int, Optional[str]]] = []
|
|
256
|
-
pending: Optional[Tuple[int, int, Optional[str]]] = None
|
|
257
|
-
for start, end, path in spans:
|
|
258
|
-
if pending is None:
|
|
259
|
-
pending = (start, end, path)
|
|
260
|
-
else:
|
|
261
|
-
pending = (pending[0], end, pending[2])
|
|
262
|
-
if pending[1] - pending[0] >= min_chars:
|
|
263
|
-
merged.append(pending)
|
|
264
|
-
pending = None
|
|
265
|
-
if pending is not None:
|
|
266
|
-
if merged and pending[1] - pending[0] < min_chars:
|
|
267
|
-
last = merged.pop()
|
|
268
|
-
merged.append((last[0], pending[1], last[2]))
|
|
269
|
-
else:
|
|
270
|
-
merged.append(pending)
|
|
271
|
-
return merged
|
|
272
|
-
|
|
273
|
-
|
|
274
|
-
def _markdown_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
275
|
-
sections = _merge_small_sections(
|
|
276
|
-
_markdown_section_spans(cleaned), _MARKDOWN_MIN_SECTION_CHARS
|
|
277
|
-
)
|
|
278
|
-
out: List[Dict[str, Any]] = []
|
|
279
|
-
for start, end, path in sections:
|
|
280
|
-
body = cleaned[start:end]
|
|
281
|
-
if len(body) <= size:
|
|
282
|
-
out.append(
|
|
283
|
-
{
|
|
284
|
-
"text": body,
|
|
285
|
-
"meta": {
|
|
286
|
-
"strategy": "markdown",
|
|
287
|
-
"start_char": start,
|
|
288
|
-
"heading_path": path,
|
|
289
|
-
},
|
|
290
|
-
}
|
|
291
|
-
)
|
|
292
|
-
else:
|
|
293
|
-
out.extend(
|
|
294
|
-
_plain_windows(
|
|
295
|
-
body,
|
|
296
|
-
size,
|
|
297
|
-
overlap,
|
|
298
|
-
base_offset=start,
|
|
299
|
-
strategy="markdown",
|
|
300
|
-
heading_path=path,
|
|
301
|
-
)
|
|
302
|
-
)
|
|
303
|
-
return out
|
|
304
|
-
|
|
305
|
-
|
|
306
|
-
def _code_segment_spans(cleaned: str) -> List[Tuple[int, int]]:
|
|
307
|
-
"""Contiguous top-level segments split at blank-line runs and decl lines."""
|
|
308
|
-
boundaries = {0, len(cleaned)}
|
|
309
|
-
for match in _CODE_BLANK_RUN_RE.finditer(cleaned):
|
|
310
|
-
boundaries.add(match.end())
|
|
311
|
-
for match in _CODE_BOUNDARY_LINE_RE.finditer(cleaned):
|
|
312
|
-
boundaries.add(match.start())
|
|
313
|
-
ordered = sorted(boundaries)
|
|
314
|
-
return [
|
|
315
|
-
(ordered[i], ordered[i + 1])
|
|
316
|
-
for i in range(len(ordered) - 1)
|
|
317
|
-
if ordered[i + 1] > ordered[i]
|
|
318
|
-
]
|
|
319
|
-
|
|
320
|
-
|
|
321
|
-
def _code_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
322
|
-
hard_limit = int(size * 1.5)
|
|
323
|
-
out: List[Dict[str, Any]] = []
|
|
324
|
-
pack: Optional[Tuple[int, int]] = None
|
|
325
|
-
|
|
326
|
-
def _emit(span: Tuple[int, int]) -> None:
|
|
327
|
-
out.append(
|
|
328
|
-
{
|
|
329
|
-
"text": cleaned[span[0] : span[1]],
|
|
330
|
-
"meta": {
|
|
331
|
-
"strategy": "code",
|
|
332
|
-
"start_char": span[0],
|
|
333
|
-
"heading_path": None,
|
|
334
|
-
},
|
|
335
|
-
}
|
|
336
|
-
)
|
|
337
|
-
|
|
338
|
-
for start, end in _code_segment_spans(cleaned):
|
|
339
|
-
if end - start > hard_limit:
|
|
340
|
-
# Monster segment: flush the pack, then window it like plain text.
|
|
341
|
-
if pack is not None:
|
|
342
|
-
_emit(pack)
|
|
343
|
-
pack = None
|
|
344
|
-
out.extend(
|
|
345
|
-
_plain_windows(
|
|
346
|
-
cleaned[start:end],
|
|
347
|
-
size,
|
|
348
|
-
overlap,
|
|
349
|
-
base_offset=start,
|
|
350
|
-
strategy="code",
|
|
351
|
-
)
|
|
352
|
-
)
|
|
353
|
-
continue
|
|
354
|
-
if pack is None:
|
|
355
|
-
pack = (start, end)
|
|
356
|
-
elif end - pack[0] <= size:
|
|
357
|
-
pack = (pack[0], end)
|
|
358
|
-
else:
|
|
359
|
-
_emit(pack)
|
|
360
|
-
pack = (start, end)
|
|
361
|
-
if pack is not None:
|
|
362
|
-
_emit(pack)
|
|
363
|
-
return out
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
# ── Prose chunking (review 2026-07-27 P1 #4) ────────────────────────────────
|
|
367
|
-
# The plain walk cuts every ``size`` characters, which lands mid-sentence and
|
|
368
|
-
# — for Korean, where the verb carrying the meaning sits at the end — routinely
|
|
369
|
-
# splits a claim from its predicate. Retrieval then matches half a statement
|
|
370
|
-
# and the citation shows a fragment. The prose strategy keeps the same window
|
|
371
|
-
# budget but ends each chunk at the last sentence/paragraph boundary inside it.
|
|
372
|
-
|
|
373
|
-
# Strong: sentence-final punctuation (ASCII + CJK) with optional closing
|
|
374
|
-
# quotes/brackets, followed by whitespace; or a blank-line paragraph break.
|
|
375
|
-
_PROSE_STRONG_BOUNDARY_RE = re.compile(
|
|
376
|
-
r"(?:[.!?。!?…]+[\"'”’」』\)\]]*\s+|\n[ \t]*\n)"
|
|
377
|
-
)
|
|
378
|
-
# Weak: a single line break. Korean notes and bullet lists often carry no
|
|
379
|
-
# sentence punctuation at all; a line end is still a real boundary there.
|
|
380
|
-
_PROSE_WEAK_BOUNDARY_RE = re.compile(r"\n")
|
|
381
|
-
# Never emit a chunk shorter than this fraction of ``size`` just to hit a
|
|
382
|
-
# boundary — tiny chunks hurt recall more than a mid-sentence cut.
|
|
383
|
-
_PROSE_MIN_SPAN_RATIO = 0.5
|
|
384
|
-
|
|
385
|
-
|
|
386
|
-
def _last_boundary(cleaned: str, lo: int, hi: int) -> Optional[int]:
|
|
387
|
-
"""End offset of the last sentence/paragraph boundary in ``cleaned[lo:hi]``.
|
|
388
|
-
|
|
389
|
-
Strong boundaries win; a single line break is the fallback. Returns None
|
|
390
|
-
when the span holds neither, so the caller keeps the hard window cut.
|
|
391
|
-
"""
|
|
392
|
-
window = cleaned[lo:hi]
|
|
393
|
-
for pattern in (_PROSE_STRONG_BOUNDARY_RE, _PROSE_WEAK_BOUNDARY_RE):
|
|
394
|
-
last = None
|
|
395
|
-
for match in pattern.finditer(window):
|
|
396
|
-
last = match.end()
|
|
397
|
-
if last:
|
|
398
|
-
return lo + last
|
|
399
|
-
return None
|
|
400
|
-
|
|
401
|
-
|
|
402
|
-
def _prose_chunks(cleaned: str, size: int, overlap: int) -> List[Dict[str, Any]]:
|
|
403
|
-
out: List[Dict[str, Any]] = []
|
|
404
|
-
total = len(cleaned)
|
|
405
|
-
min_span = max(1, int(size * _PROSE_MIN_SPAN_RATIO))
|
|
406
|
-
start = 0
|
|
407
|
-
while start < total:
|
|
408
|
-
hard_end = min(total, start + size)
|
|
409
|
-
end = hard_end
|
|
410
|
-
if hard_end < total:
|
|
411
|
-
boundary = _last_boundary(cleaned, start + min_span, hard_end)
|
|
412
|
-
if boundary is not None and boundary > start:
|
|
413
|
-
end = boundary
|
|
414
|
-
out.append(
|
|
415
|
-
{
|
|
416
|
-
"text": cleaned[start:end],
|
|
417
|
-
"meta": {
|
|
418
|
-
"strategy": "prose",
|
|
419
|
-
"start_char": start,
|
|
420
|
-
"heading_path": None,
|
|
421
|
-
},
|
|
422
|
-
}
|
|
423
|
-
)
|
|
424
|
-
if end >= total:
|
|
425
|
-
break
|
|
426
|
-
# Overlap carries the tail of the previous chunk into the next one so
|
|
427
|
-
# a claim split across a boundary is still retrievable from both.
|
|
428
|
-
start = max(start + 1, end - overlap)
|
|
429
|
-
return out
|
|
430
|
-
|
|
431
|
-
|
|
432
|
-
def typed_chunks(
|
|
433
|
-
text: str,
|
|
434
|
-
*,
|
|
435
|
-
strategy: str = "plain",
|
|
436
|
-
size: int = 1200,
|
|
437
|
-
overlap: int = 160,
|
|
438
|
-
) -> List[Dict[str, Any]]:
|
|
439
|
-
"""Strategy-aware chunking with per-chunk provenance metadata.
|
|
440
|
-
|
|
441
|
-
Returns ``[{"text": str, "meta": {"strategy", "start_char", "heading_path"}}]``
|
|
442
|
-
where ``start_char`` is the offset in ``str(text or "").strip()`` (every
|
|
443
|
-
chunk text is an exact substring at that offset).
|
|
444
|
-
|
|
445
|
-
Contract: ``[c["text"] for c in typed_chunks(t)] == _chunks(t)`` for the
|
|
446
|
-
default plain strategy — unknown strategies also fall back to plain.
|
|
447
|
-
"""
|
|
448
|
-
cleaned = str(text or "").strip()
|
|
449
|
-
if not cleaned:
|
|
450
|
-
return []
|
|
451
|
-
try:
|
|
452
|
-
size = max(1, int(size))
|
|
453
|
-
except Exception:
|
|
454
|
-
size = 1200
|
|
455
|
-
try:
|
|
456
|
-
overlap = min(max(0, int(overlap)), size - 1)
|
|
457
|
-
except Exception:
|
|
458
|
-
overlap = min(160, size - 1)
|
|
459
|
-
label = strategy if strategy in _CHUNK_STRATEGIES else "plain"
|
|
460
|
-
if label == "markdown":
|
|
461
|
-
return _markdown_chunks(cleaned, size, overlap)
|
|
462
|
-
if label == "code":
|
|
463
|
-
return _code_chunks(cleaned, size, overlap)
|
|
464
|
-
if label == "prose":
|
|
465
|
-
return _prose_chunks(cleaned, size, overlap)
|
|
466
|
-
return _plain_windows(cleaned, size, overlap)
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
def typed_chunk_meta_fields(piece: Dict[str, Any]) -> Dict[str, Any]:
|
|
470
|
-
"""Additive chunk-metadata fields for one ``typed_chunks`` piece.
|
|
471
|
-
|
|
472
|
-
Ingest call sites merge this into the existing ``{"index", "source_node"}``
|
|
473
|
-
chunk metadata; ``heading_path`` is only present when known — honest
|
|
474
|
-
absence over empty labels.
|
|
475
|
-
"""
|
|
476
|
-
meta = piece.get("meta") or {}
|
|
477
|
-
fields: Dict[str, Any] = {
|
|
478
|
-
"strategy": str(meta.get("strategy") or "plain"),
|
|
479
|
-
"start_char": int(meta.get("start_char") or 0),
|
|
480
|
-
}
|
|
481
|
-
heading_path = meta.get("heading_path")
|
|
482
|
-
if heading_path:
|
|
483
|
-
fields["heading_path"] = str(heading_path)
|
|
484
|
-
return fields
|
|
485
|
-
|
|
486
|
-
|
|
487
|
-
def citation_locator(chunk_metadata: Any) -> str:
|
|
488
|
-
"""Human "where in the document" label for one chunk, or "".
|
|
489
|
-
|
|
490
|
-
Built only from provenance the chunk actually carries — a section heading
|
|
491
|
-
path and/or a page number. When neither is known the answer is the empty
|
|
492
|
-
string, so a citation never claims a location it cannot prove.
|
|
493
|
-
"""
|
|
494
|
-
if not isinstance(chunk_metadata, dict):
|
|
495
|
-
return ""
|
|
496
|
-
parts: List[str] = []
|
|
497
|
-
heading = str(chunk_metadata.get("heading_path") or "").strip()
|
|
498
|
-
if heading:
|
|
499
|
-
parts.append(heading)
|
|
500
|
-
def _page(key: str) -> int:
|
|
501
|
-
value = chunk_metadata.get(key)
|
|
502
|
-
try:
|
|
503
|
-
return int(value) if value is not None else 0
|
|
504
|
-
except (TypeError, ValueError):
|
|
505
|
-
return 0
|
|
506
|
-
|
|
507
|
-
page_number = _page("page")
|
|
508
|
-
if page_number > 0:
|
|
509
|
-
page_end = _page("page_end")
|
|
510
|
-
parts.append(
|
|
511
|
-
f"p.{page_number}–{page_end}" if page_end > page_number else f"p.{page_number}"
|
|
512
|
-
)
|
|
513
|
-
return " · ".join(parts)
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
def pdf_page_offsets(structure: Any) -> List[int]:
|
|
517
|
-
"""Start offset of each PDF page in the "\\n\\n"-joined page text.
|
|
518
|
-
|
|
519
|
-
``structure`` is the ``metadata["structure"]`` dict produced by
|
|
520
|
-
``_pdf_structure`` (``pages`` = ``[{"chars": int, ...}, ...]``); pages were
|
|
521
|
-
joined with ``"\\n\\n"`` (see ``read_document``), so page k starts at
|
|
522
|
-
``sum(chars[j] + 2 for j < k)``. Empty or malformed input returns ``[]``.
|
|
523
|
-
"""
|
|
524
|
-
if not isinstance(structure, dict):
|
|
525
|
-
return []
|
|
526
|
-
pages = structure.get("pages")
|
|
527
|
-
if not isinstance(pages, list) or not pages:
|
|
528
|
-
return []
|
|
529
|
-
offsets: List[int] = []
|
|
530
|
-
cursor = 0
|
|
531
|
-
for page in pages:
|
|
532
|
-
if not isinstance(page, dict):
|
|
533
|
-
return []
|
|
534
|
-
chars = page.get("chars")
|
|
535
|
-
if isinstance(chars, bool) or not isinstance(chars, (int, float)) or chars < 0:
|
|
536
|
-
return []
|
|
537
|
-
offsets.append(cursor)
|
|
538
|
-
cursor += int(chars) + 2 # +2 for the "\n\n" page joiner
|
|
539
|
-
return offsets
|
|
540
|
-
|
|
541
|
-
|
|
542
|
-
def page_for_offset(page_offsets: List[int], offset: int) -> Optional[int]:
|
|
543
|
-
"""1-based page number containing ``offset`` given page start offsets.
|
|
544
|
-
|
|
545
|
-
Returns ``None`` when ``page_offsets`` is empty or the offset precedes the
|
|
546
|
-
first page start (honest absence over a wrong label).
|
|
547
|
-
"""
|
|
548
|
-
if not page_offsets:
|
|
549
|
-
return None
|
|
550
|
-
try:
|
|
551
|
-
target = int(offset)
|
|
552
|
-
except Exception:
|
|
553
|
-
return None
|
|
554
|
-
page = 0
|
|
555
|
-
for index, start in enumerate(page_offsets):
|
|
556
|
-
try:
|
|
557
|
-
if target >= int(start):
|
|
558
|
-
page = index + 1
|
|
559
|
-
else:
|
|
560
|
-
break
|
|
561
|
-
except Exception:
|
|
562
|
-
return None
|
|
563
|
-
return page if page >= 1 else None
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
_LLM_EXTRACT_CONCEPT_PROMPT = """Extract the key concepts from the following text.
|
|
567
|
-
Return ONLY a JSON array of objects, each with "concept" (string) and "importance" (float 0-1).
|
|
568
|
-
Extract up to {limit} concepts. Focus on named entities, technical terms, and domain-specific nouns.
|
|
569
|
-
Do NOT include common words, stop words, or generic terms.
|
|
570
|
-
|
|
571
|
-
Text:
|
|
572
|
-
{text}
|
|
573
|
-
|
|
574
|
-
JSON:"""
|
|
575
|
-
|
|
576
|
-
_LLM_EXTRACT_TRIPLE_PROMPT = """Extract relationship triples from the following text.
|
|
577
|
-
Return ONLY a JSON array of objects, each with:
|
|
578
|
-
- "subject": source concept (string)
|
|
579
|
-
- "relation": relationship verb (string, Korean or English)
|
|
580
|
-
- "object": target concept (string)
|
|
581
|
-
- "evidence": the sentence supporting this triple (string, max 240 chars)
|
|
582
|
-
- "confidence": how confident you are (float 0-1)
|
|
583
|
-
|
|
584
|
-
Extract up to {limit} triples. Focus on meaningful semantic relationships.
|
|
585
|
-
|
|
586
|
-
Text:
|
|
587
|
-
{text}
|
|
588
|
-
|
|
589
|
-
Concepts already identified: {concepts}
|
|
590
|
-
|
|
591
|
-
JSON:"""
|
|
592
|
-
|
|
593
|
-
ENABLE_LLM_EXTRACTION = os.getenv("LATTICEAI_LLM_EXTRACTION", "true").lower() in (
|
|
594
|
-
"1",
|
|
595
|
-
"true",
|
|
596
|
-
"yes",
|
|
597
|
-
)
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
def _llm_extract_concepts(text: str, limit: int = 12) -> Optional[List[str]]:
|
|
601
|
-
router = get_llm_router()
|
|
602
|
-
if not ENABLE_LLM_EXTRACTION or not router:
|
|
603
|
-
return None
|
|
604
|
-
if not router.current_model_id:
|
|
605
|
-
return None
|
|
606
|
-
prompt = _LLM_EXTRACT_CONCEPT_PROMPT.format(text=text[:3000], limit=limit)
|
|
607
|
-
try:
|
|
608
|
-
loop = asyncio.get_event_loop()
|
|
609
|
-
if loop.is_running():
|
|
610
|
-
import concurrent.futures
|
|
611
|
-
|
|
612
|
-
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
613
|
-
future = pool.submit(
|
|
614
|
-
asyncio.run,
|
|
615
|
-
router.generate(prompt, max_tokens=1024, temperature=0.1),
|
|
616
|
-
)
|
|
617
|
-
raw = future.result(timeout=30)
|
|
618
|
-
else:
|
|
619
|
-
raw = asyncio.run(
|
|
620
|
-
router.generate(prompt, max_tokens=1024, temperature=0.1)
|
|
621
|
-
)
|
|
622
|
-
raw = raw.strip()
|
|
623
|
-
if raw.startswith("```"):
|
|
624
|
-
raw = re.sub(r"^```(?:json)?\s*", "", raw)
|
|
625
|
-
raw = re.sub(r"\s*```$", "", raw)
|
|
626
|
-
parsed = json.loads(raw)
|
|
627
|
-
if isinstance(parsed, list):
|
|
628
|
-
concepts = []
|
|
629
|
-
for item in parsed[:limit]:
|
|
630
|
-
if isinstance(item, dict) and "concept" in item:
|
|
631
|
-
concepts.append(item["concept"])
|
|
632
|
-
elif isinstance(item, str):
|
|
633
|
-
concepts.append(item)
|
|
634
|
-
return concepts if concepts else None
|
|
635
|
-
except Exception as e:
|
|
636
|
-
logging.debug("LLM concept extraction failed (falling back to rules): %s", e)
|
|
637
|
-
return None
|
|
638
|
-
|
|
639
|
-
|
|
640
|
-
# Triples carry a numeric ``weight``/``confidence`` alongside string fields,
|
|
641
|
-
# so the value type is Any rather than str.
|
|
642
|
-
def _llm_extract_triples(
|
|
643
|
-
text: str, concepts: List[str], limit: int = 20
|
|
644
|
-
) -> Optional[List[Dict[str, Any]]]:
|
|
645
|
-
router = get_llm_router()
|
|
646
|
-
if not ENABLE_LLM_EXTRACTION or not router:
|
|
647
|
-
return None
|
|
648
|
-
if not router.current_model_id:
|
|
649
|
-
return None
|
|
650
|
-
prompt = _LLM_EXTRACT_TRIPLE_PROMPT.format(
|
|
651
|
-
text=text[:3000],
|
|
652
|
-
limit=limit,
|
|
653
|
-
concepts=", ".join(concepts[:15]),
|
|
654
|
-
)
|
|
655
|
-
try:
|
|
656
|
-
loop = asyncio.get_event_loop()
|
|
657
|
-
if loop.is_running():
|
|
658
|
-
import concurrent.futures
|
|
659
|
-
|
|
660
|
-
with concurrent.futures.ThreadPoolExecutor(max_workers=1) as pool:
|
|
661
|
-
future = pool.submit(
|
|
662
|
-
asyncio.run,
|
|
663
|
-
router.generate(prompt, max_tokens=2048, temperature=0.1),
|
|
664
|
-
)
|
|
665
|
-
raw = future.result(timeout=30)
|
|
666
|
-
else:
|
|
667
|
-
raw = asyncio.run(
|
|
668
|
-
router.generate(prompt, max_tokens=2048, temperature=0.1)
|
|
669
|
-
)
|
|
670
|
-
raw = raw.strip()
|
|
671
|
-
if raw.startswith("```"):
|
|
672
|
-
raw = re.sub(r"^```(?:json)?\s*", "", raw)
|
|
673
|
-
raw = re.sub(r"\s*```$", "", raw)
|
|
674
|
-
parsed = json.loads(raw)
|
|
675
|
-
if isinstance(parsed, list):
|
|
676
|
-
triples: List[Dict[str, Any]] = []
|
|
677
|
-
for item in parsed[:limit]:
|
|
678
|
-
if isinstance(item, dict) and "subject" in item and "object" in item:
|
|
679
|
-
relation = str(item.get("relation", "관련됨"))
|
|
680
|
-
evidence_text = str(item.get("evidence", ""))[:240]
|
|
681
|
-
confidence = float(item.get("confidence", 0.8))
|
|
682
|
-
# An LLM triple that names a real verb and cites the
|
|
683
|
-
# sentence it came from is semantic evidence; a bare
|
|
684
|
-
# "관련됨" with no quoted evidence is the model restating
|
|
685
|
-
# co-occurrence, and is weighted (and labelled) as such.
|
|
686
|
-
is_semantic = bool(evidence_text) and relation != "관련됨"
|
|
687
|
-
triples.append(
|
|
688
|
-
{
|
|
689
|
-
"subject": str(item["subject"]),
|
|
690
|
-
"relation": relation,
|
|
691
|
-
"object": str(item["object"]),
|
|
692
|
-
"context": evidence_text,
|
|
693
|
-
"confidence": confidence,
|
|
694
|
-
"evidence": "verb" if is_semantic else "cooccurrence",
|
|
695
|
-
"weight": round(
|
|
696
|
-
(VERB_EDGE_WEIGHT if is_semantic else COOCCURRENCE_EDGE_WEIGHT)
|
|
697
|
-
* max(0.1, min(confidence, 1.0)),
|
|
698
|
-
4,
|
|
699
|
-
),
|
|
700
|
-
}
|
|
701
|
-
)
|
|
702
|
-
return triples if triples else None
|
|
703
|
-
except Exception as e:
|
|
704
|
-
logging.debug("LLM triple extraction failed (falling back to rules): %s", e)
|
|
705
|
-
return None
|
|
706
|
-
|
|
707
|
-
|
|
708
|
-
_CONCEPT_STOP: set = {
|
|
709
|
-
# English stop words
|
|
710
|
-
"the",
|
|
711
|
-
"and",
|
|
712
|
-
"for",
|
|
713
|
-
"with",
|
|
714
|
-
"this",
|
|
715
|
-
"that",
|
|
716
|
-
"from",
|
|
717
|
-
"into",
|
|
718
|
-
"which",
|
|
719
|
-
"are",
|
|
720
|
-
"was",
|
|
721
|
-
"were",
|
|
722
|
-
"has",
|
|
723
|
-
"have",
|
|
724
|
-
"had",
|
|
725
|
-
"can",
|
|
726
|
-
"will",
|
|
727
|
-
"would",
|
|
728
|
-
"could",
|
|
729
|
-
"should",
|
|
730
|
-
"may",
|
|
731
|
-
"might",
|
|
732
|
-
"must",
|
|
733
|
-
"shall",
|
|
734
|
-
"being",
|
|
735
|
-
"been",
|
|
736
|
-
"also",
|
|
737
|
-
"just",
|
|
738
|
-
"then",
|
|
739
|
-
"than",
|
|
740
|
-
"when",
|
|
741
|
-
"where",
|
|
742
|
-
"what",
|
|
743
|
-
"how",
|
|
744
|
-
"why",
|
|
745
|
-
"its",
|
|
746
|
-
"their",
|
|
747
|
-
"your",
|
|
748
|
-
"our",
|
|
749
|
-
"you",
|
|
750
|
-
"they",
|
|
751
|
-
"them",
|
|
752
|
-
"these",
|
|
753
|
-
"those",
|
|
754
|
-
"use",
|
|
755
|
-
"used",
|
|
756
|
-
"using",
|
|
757
|
-
"based",
|
|
758
|
-
"like",
|
|
759
|
-
"such",
|
|
760
|
-
"via",
|
|
761
|
-
"per",
|
|
762
|
-
"let",
|
|
763
|
-
"yes",
|
|
764
|
-
"not",
|
|
765
|
-
"but",
|
|
766
|
-
"all",
|
|
767
|
-
"any",
|
|
768
|
-
"out",
|
|
769
|
-
"new",
|
|
770
|
-
"get",
|
|
771
|
-
"set",
|
|
772
|
-
# Korean stop words
|
|
773
|
-
"사용자",
|
|
774
|
-
"내용",
|
|
775
|
-
"파일",
|
|
776
|
-
"채팅",
|
|
777
|
-
"답변",
|
|
778
|
-
"입니다",
|
|
779
|
-
"그리고",
|
|
780
|
-
"처럼",
|
|
781
|
-
"있어",
|
|
782
|
-
"없어",
|
|
783
|
-
"이야",
|
|
784
|
-
"이다",
|
|
785
|
-
"한다",
|
|
786
|
-
"하다",
|
|
787
|
-
"되다",
|
|
788
|
-
"됩니다",
|
|
789
|
-
"경우",
|
|
790
|
-
"방법",
|
|
791
|
-
"부분",
|
|
792
|
-
"상태",
|
|
793
|
-
"정도",
|
|
794
|
-
"결과",
|
|
795
|
-
"이후",
|
|
796
|
-
"이전",
|
|
797
|
-
"그것",
|
|
798
|
-
"이것",
|
|
799
|
-
"저것",
|
|
800
|
-
"여기",
|
|
801
|
-
"거기",
|
|
802
|
-
"저기",
|
|
803
|
-
"우리",
|
|
804
|
-
"저희",
|
|
805
|
-
"기능",
|
|
806
|
-
"서버",
|
|
807
|
-
"모델",
|
|
808
|
-
"설정",
|
|
809
|
-
"설명",
|
|
810
|
-
"버전",
|
|
811
|
-
"지원",
|
|
812
|
-
"사용",
|
|
813
|
-
"실행",
|
|
814
|
-
"todo",
|
|
815
|
-
"fixme",
|
|
816
|
-
"note",
|
|
817
|
-
"참고",
|
|
818
|
-
"주의",
|
|
819
|
-
"warning",
|
|
820
|
-
}
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
def _extract_concepts(text: str, limit: int = 12) -> List[str]:
|
|
824
|
-
"""LLM-first concept extraction with rule-based fallback."""
|
|
825
|
-
llm_result = _llm_extract_concepts(text, limit)
|
|
826
|
-
if llm_result:
|
|
827
|
-
return llm_result
|
|
828
|
-
return _extract_concepts_rules(text, limit)
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
def _extract_concepts_rules(text: str, limit: int = 12) -> List[str]:
|
|
832
|
-
"""Extract meaningful named concepts from text (rule-based).
|
|
833
|
-
|
|
834
|
-
Priority order:
|
|
835
|
-
1. Backtick / quoted terms (explicitly technical)
|
|
836
|
-
2. Multi-word proper nouns (Lattice AI, GPT-4o, Claude Sonnet)
|
|
837
|
-
3. Single capitalized proper nouns not at sentence start (Claude, Python, FastAPI)
|
|
838
|
-
4. Korean compound technical terms (멀티모달, 에이전트, 그래프RAG)
|
|
839
|
-
5. Hyphenated / versioned identifiers (gpt-4o, mlx-vlm, gemma-4)
|
|
840
|
-
"""
|
|
841
|
-
text = str(text or "")
|
|
842
|
-
seen: dict = {} # concept_lower → original form
|
|
843
|
-
|
|
844
|
-
def _add(term: str) -> None:
|
|
845
|
-
key = term.strip().lower()
|
|
846
|
-
if key and key not in _CONCEPT_STOP and not key.isdigit() and len(key) >= 2:
|
|
847
|
-
seen.setdefault(key, term.strip())
|
|
848
|
-
|
|
849
|
-
# 1. Backtick-quoted code/term (highest confidence)
|
|
850
|
-
for m in re.findall(r"`([^`]{2,40})`", text):
|
|
851
|
-
if not re.search(r"[\(\)\[\]{}]", m): # skip code expressions
|
|
852
|
-
_add(m)
|
|
853
|
-
|
|
854
|
-
# 2. Double/single quoted terms
|
|
855
|
-
for m in re.findall(r'"([^"]{2,40})"', text):
|
|
856
|
-
_add(m)
|
|
857
|
-
|
|
858
|
-
# 3. Multi-word English proper nouns (Title Case or ALL-CAPS first word, 2–4 words).
|
|
859
|
-
# Pattern A: Mixed-case first word — "Lattice AI", "Tool Use", "Graph RAG"
|
|
860
|
-
for m in re.findall(
|
|
861
|
-
r"([A-Z][a-z]{1,20}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20}|\d[\w.]{0,6})){1,3})",
|
|
862
|
-
text,
|
|
863
|
-
):
|
|
864
|
-
_add(m)
|
|
865
|
-
# Pattern B: ALL-CAPS first word — "VS Code", "MCP Server", "GPT-4o Mini"
|
|
866
|
-
for m in re.findall(
|
|
867
|
-
r"([A-Z]{2,6}(?:\s+(?:[A-Z]{2,10}|[A-Z][a-z0-9]{1,20})){1,2})",
|
|
868
|
-
text,
|
|
869
|
-
):
|
|
870
|
-
_add(m)
|
|
871
|
-
|
|
872
|
-
# 4. Single capitalized proper noun.
|
|
873
|
-
# Use ASCII-boundary lookaround instead of \b so Korean particles
|
|
874
|
-
# (와, 의, 는 …) after an English word don't block the match.
|
|
875
|
-
all_caps_words = re.findall(
|
|
876
|
-
r"(?<![A-Za-z0-9])([A-Z][A-Za-z0-9]{2,24})(?![A-Za-z0-9])", text
|
|
877
|
-
)
|
|
878
|
-
freq: Dict[str, int] = {}
|
|
879
|
-
for w in all_caps_words:
|
|
880
|
-
freq[w] = freq.get(w, 0) + 1
|
|
881
|
-
sentence_starts = set(re.findall(r"(?:^|(?<=[.!?])\s+)([A-Z][a-z]+)", text))
|
|
882
|
-
for m, cnt in freq.items():
|
|
883
|
-
if m.lower() in _CONCEPT_STOP:
|
|
884
|
-
continue
|
|
885
|
-
if cnt >= 2 or m not in sentence_starts:
|
|
886
|
-
_add(m)
|
|
887
|
-
|
|
888
|
-
# 5. Korean technical compound nouns (3–12 chars, no common particles)
|
|
889
|
-
for m in re.findall(
|
|
890
|
-
r"[가-힣]{2,12}(?:AI|LLM|API|UI|RAG|bot|Bot|기능|모델|서버|에이전트|파이프라인|워크플로)",
|
|
891
|
-
text,
|
|
892
|
-
):
|
|
893
|
-
_add(m)
|
|
894
|
-
# Korean standalone terms that appear after topic markers (은/는/이/가 앞)
|
|
895
|
-
for m in re.findall(
|
|
896
|
-
r"([가-힣]{2,12})(?:은|는|이|가|을|를|의|에서|으로|와|과)", text
|
|
897
|
-
):
|
|
898
|
-
if m.lower() not in _CONCEPT_STOP and len(m) >= 2:
|
|
899
|
-
# Only add if it's non-trivial (has 3+ chars or appears multiple times)
|
|
900
|
-
cnt = text.count(m)
|
|
901
|
-
if len(m) >= 3 or cnt >= 2:
|
|
902
|
-
_add(m)
|
|
903
|
-
|
|
904
|
-
# 6. Hyphenated / versioned identifiers (gpt-4o, gemma-4, mlx-vlm)
|
|
905
|
-
for m in re.findall(r"\b([a-zA-Z][a-zA-Z0-9]*(?:-[a-zA-Z0-9.]+)+)\b", text):
|
|
906
|
-
if len(m) >= 4:
|
|
907
|
-
_add(m)
|
|
908
|
-
|
|
909
|
-
# De-duplicate: remove shorter if ALL its occurrences in the source text
|
|
910
|
-
# are followed immediately by the suffix that forms the longer concept.
|
|
911
|
-
# "Lattice" → dropped when every occurrence is "Lattice AI"
|
|
912
|
-
# "Claude" → kept because it appears as just "Claude" too.
|
|
913
|
-
values = list(seen.values())
|
|
914
|
-
values_lower = [v.lower() for v in values]
|
|
915
|
-
keep = set(range(len(values)))
|
|
916
|
-
for i, v in enumerate(values):
|
|
917
|
-
vl = v.lower()
|
|
918
|
-
for j, wl in enumerate(values_lower):
|
|
919
|
-
if i == j or j not in keep:
|
|
920
|
-
continue
|
|
921
|
-
# Check if vl is a word-prefix of wl
|
|
922
|
-
suffix = wl[len(vl) :]
|
|
923
|
-
if not (wl.startswith(vl) and re.match(r"^[\s\-]", suffix)):
|
|
924
|
-
continue
|
|
925
|
-
# Count occurrences of v NOT followed by the suffix
|
|
926
|
-
suffix_stripped = suffix.lstrip(" -")
|
|
927
|
-
# Escape for regex
|
|
928
|
-
pattern_with_suffix = re.escape(v) + r"[\s\-]+" + re.escape(suffix_stripped)
|
|
929
|
-
pattern_alone = (
|
|
930
|
-
re.escape(v) + r"(?![\s\-]*" + re.escape(suffix_stripped) + r")"
|
|
931
|
-
)
|
|
932
|
-
alone_count = len(re.findall(pattern_alone, text, re.IGNORECASE))
|
|
933
|
-
if alone_count == 0:
|
|
934
|
-
# Shorter term never appears alone → safe to remove
|
|
935
|
-
keep.discard(i)
|
|
936
|
-
break
|
|
937
|
-
|
|
938
|
-
final = [values[i] for i in range(len(values)) if i in keep]
|
|
939
|
-
return final[:limit]
|
|
940
|
-
|
|
941
|
-
|
|
942
|
-
# ──────────────────────────────────────────────────────────────────────────────
|
|
943
|
-
# Node type taxonomy (점 = 명사)
|
|
944
|
-
# ──────────────────────────────────────────────────────────────────────────────
|
|
945
|
-
# Chat — 대화 세션
|
|
946
|
-
# Document — 파일 (PDF·PPT·Word·Excel·이미지 등)
|
|
947
|
-
# Concept — 개념·아이디어·기술 용어
|
|
948
|
-
# Person — 사람 (사용자, 언급된 인물)
|
|
949
|
-
# Error — 오류·버그·예외
|
|
950
|
-
# Code — 코드 스니펫·함수·클래스
|
|
951
|
-
# Feature — 소프트웨어 기능
|
|
952
|
-
# Task — 할 일·액션 아이템
|
|
953
|
-
# Decision — 결정 사항
|
|
954
|
-
|
|
955
|
-
# Edge type vocabulary (선 = 동사 — 과거형 서술어)
|
|
956
|
-
EDGE_VERB = {
|
|
957
|
-
"언급함": r"언급|mention|refer|cited",
|
|
958
|
-
"포함함": r"포함|include|consist|구성|탑재|contains",
|
|
959
|
-
"해결함": r"해결|resolv|fix|수정|고쳤|closed",
|
|
960
|
-
"의존함": r"의존|depend|require|필요|based on",
|
|
961
|
-
"설명함": r"설명|explain|describe|정의|란|이란|means",
|
|
962
|
-
"비교함": r"비교|versus|vs\.?|차이|다르|compare",
|
|
963
|
-
"사용함": r"사용|use|활용|이용|apply",
|
|
964
|
-
"연결함": r"연결|connect|통합|integrate|연동|link",
|
|
965
|
-
"확장함": r"확장|extend|플러그인|plugin|addon",
|
|
966
|
-
"생성함": r"생성|만들|create|generate|build|produced",
|
|
967
|
-
"대체함": r"대체|replace|instead|alternative",
|
|
968
|
-
"지원함": r"지원|support|제공|provide|offer",
|
|
969
|
-
"발생함": r"발생|occur|throw|raise|triggered",
|
|
970
|
-
"관련됨": r"관련|related|associated|연관",
|
|
971
|
-
}
|
|
972
|
-
|
|
973
|
-
|
|
974
|
-
# Concepts in a list-like sentence ("A, B, C, D를 사용한다") sit together by
|
|
975
|
-
# enumeration, not by relation. Beyond this many concepts in one sentence, a
|
|
976
|
-
# verb-less pairing is enumeration noise and is dropped outright.
|
|
977
|
-
COOCCURRENCE_CONCEPT_LIMIT = 4
|
|
978
|
-
# Verb-backed relations carry the sentence's own evidence; co-occurrence
|
|
979
|
-
# relations carry only adjacency, so they enter the graph at a lower weight
|
|
980
|
-
# and are labelled as such.
|
|
981
|
-
VERB_EDGE_WEIGHT = 1.0
|
|
982
|
-
COOCCURRENCE_EDGE_WEIGHT = 0.35
|
|
983
|
-
|
|
984
|
-
|
|
985
|
-
def infer_edge_relation(sentence: str) -> Dict[str, Any]:
|
|
986
|
-
"""Classify the relation between two concepts in one sentence.
|
|
987
|
-
|
|
988
|
-
Review 2026-07-27 P1 #6: the graph drifted toward co-occurrence because a
|
|
989
|
-
verb-less sentence still produced a "관련됨" edge indistinguishable from a
|
|
990
|
-
real semantic relation. The label alone cannot carry that difference, so
|
|
991
|
-
the evidence class rides with it::
|
|
992
|
-
|
|
993
|
-
{"relation": "사용함", "evidence": "verb", "weight": 1.0}
|
|
994
|
-
{"relation": "관련됨", "evidence": "cooccurrence", "weight": 0.35}
|
|
995
|
-
|
|
996
|
-
``evidence`` is what the graph, the curator, and the UI use to tell a
|
|
997
|
-
meaning edge from an adjacency edge — the honest distinction the previous
|
|
998
|
-
label-only output erased.
|
|
999
|
-
"""
|
|
1000
|
-
s = str(sentence or "").lower()
|
|
1001
|
-
for label, pattern in EDGE_VERB.items():
|
|
1002
|
-
if re.search(pattern, s):
|
|
1003
|
-
# "관련됨" is itself a weak, generic label: matching it by keyword
|
|
1004
|
-
# ("관련", "related") is still verb evidence, but nothing stronger.
|
|
1005
|
-
return {
|
|
1006
|
-
"relation": label,
|
|
1007
|
-
"evidence": "verb",
|
|
1008
|
-
"weight": VERB_EDGE_WEIGHT,
|
|
1009
|
-
}
|
|
1010
|
-
return {
|
|
1011
|
-
"relation": "관련됨",
|
|
1012
|
-
"evidence": "cooccurrence",
|
|
1013
|
-
"weight": COOCCURRENCE_EDGE_WEIGHT,
|
|
1014
|
-
}
|
|
1015
|
-
|
|
1016
|
-
|
|
1017
|
-
def _infer_edge(sentence: str) -> str:
|
|
1018
|
-
"""Back-compat wrapper: the verb label only (see :func:`infer_edge_relation`)."""
|
|
1019
|
-
return infer_edge_relation(sentence)["relation"]
|
|
1020
|
-
|
|
1021
|
-
|
|
1022
|
-
# Technical words that cannot be person names
|
|
1023
|
-
_NOT_PERSON_WORDS: set = {
|
|
1024
|
-
"use",
|
|
1025
|
-
"api",
|
|
1026
|
-
"rag",
|
|
1027
|
-
"sdk",
|
|
1028
|
-
"ide",
|
|
1029
|
-
"cli",
|
|
1030
|
-
"llm",
|
|
1031
|
-
"mcp",
|
|
1032
|
-
"ui",
|
|
1033
|
-
"ux",
|
|
1034
|
-
"new",
|
|
1035
|
-
"old",
|
|
1036
|
-
"get",
|
|
1037
|
-
"set",
|
|
1038
|
-
"run",
|
|
1039
|
-
"add",
|
|
1040
|
-
"fix",
|
|
1041
|
-
"tool",
|
|
1042
|
-
"code",
|
|
1043
|
-
"base",
|
|
1044
|
-
"core",
|
|
1045
|
-
"data",
|
|
1046
|
-
"file",
|
|
1047
|
-
"test",
|
|
1048
|
-
"type",
|
|
1049
|
-
"mode",
|
|
1050
|
-
"view",
|
|
1051
|
-
}
|
|
1052
|
-
|
|
1053
|
-
|
|
1054
|
-
def _classify_node_type(concept: str, text: str) -> str:
|
|
1055
|
-
"""Classify a concept into the node taxonomy.
|
|
1056
|
-
|
|
1057
|
-
Term-level signals take priority; then a tight ±60-char window is used
|
|
1058
|
-
so distant keywords don't cause mis-classification.
|
|
1059
|
-
"""
|
|
1060
|
-
term = concept.lower()
|
|
1061
|
-
|
|
1062
|
-
# ── Term-level signals (highest confidence) ───────────────────────────
|
|
1063
|
-
if re.search(r"(?:error|exception|traceback|오류|에러|버그)$", term, re.I):
|
|
1064
|
-
return "Error"
|
|
1065
|
-
if re.search(r"error|exception|err\b", term, re.I) and len(concept) < 30:
|
|
1066
|
-
return "Error"
|
|
1067
|
-
if re.search(r"\(\)|\.py$|\.js$|\.ts$|\.go$|::\w", term):
|
|
1068
|
-
return "Code"
|
|
1069
|
-
|
|
1070
|
-
# Person: "First Last" pattern, neither word is a known technical term
|
|
1071
|
-
if re.match(r"^[A-Z][a-z]{1,15} [A-Z][a-z]{1,15}$", concept):
|
|
1072
|
-
words = term.split()
|
|
1073
|
-
if not any(w in _NOT_PERSON_WORDS for w in words):
|
|
1074
|
-
return "Person"
|
|
1075
|
-
|
|
1076
|
-
# ── Windowed context (±60 chars) — NOT used for Error to avoid false positives
|
|
1077
|
-
idx = text.lower().find(term)
|
|
1078
|
-
if idx >= 0:
|
|
1079
|
-
win = text[max(0, idx - 60) : idx + len(concept) + 60].lower()
|
|
1080
|
-
if re.search(r"def |class |function|함수|클래스|메서드|import", win):
|
|
1081
|
-
return "Code"
|
|
1082
|
-
# Feature: concept appears DIRECTLY adjacent to 기능/feature keyword
|
|
1083
|
-
if len(concept) <= 12 and re.search(
|
|
1084
|
-
rf"{re.escape(term)}.{{0,8}}(?:기능|feature)|(?:기능|feature).{{0,8}}{re.escape(term)}",
|
|
1085
|
-
win,
|
|
1086
|
-
):
|
|
1087
|
-
return "Feature"
|
|
1088
|
-
|
|
1089
|
-
return "Concept"
|
|
1090
|
-
|
|
1091
|
-
|
|
1092
|
-
def _extract_triples(
|
|
1093
|
-
text: str,
|
|
1094
|
-
concepts: List[str],
|
|
1095
|
-
limit: int = 20,
|
|
1096
|
-
) -> List[Dict[str, str]]:
|
|
1097
|
-
"""LLM-first triple extraction with rule-based fallback."""
|
|
1098
|
-
llm_result = _llm_extract_triples(text, concepts, limit)
|
|
1099
|
-
if llm_result:
|
|
1100
|
-
return llm_result
|
|
1101
|
-
return _extract_triples_rules(text, concepts, limit)
|
|
1102
|
-
|
|
1103
|
-
|
|
1104
|
-
def _extract_triples_rules(
|
|
1105
|
-
text: str,
|
|
1106
|
-
concepts: List[str],
|
|
1107
|
-
limit: int = 20,
|
|
1108
|
-
) -> List[Dict[str, str]]:
|
|
1109
|
-
"""Extract (subject, verb-edge, object, context) triples from text (rule-based).
|
|
1110
|
-
|
|
1111
|
-
For each sentence containing ≥2 concepts, infer the verb-form edge label
|
|
1112
|
-
from surrounding context and create a directed triple.
|
|
1113
|
-
"""
|
|
1114
|
-
if len(concepts) < 2:
|
|
1115
|
-
return []
|
|
1116
|
-
|
|
1117
|
-
concept_lower = {c.lower(): c for c in concepts}
|
|
1118
|
-
triples: List[Dict[str, str]] = []
|
|
1119
|
-
seen_pairs: set = set()
|
|
1120
|
-
|
|
1121
|
-
# Split on sentence boundaries
|
|
1122
|
-
sentences = re.split(r"(?<=[.!?\n])\s+|\n{2,}", text)
|
|
1123
|
-
for sent in sentences:
|
|
1124
|
-
sent = sent.strip()
|
|
1125
|
-
if len(sent) < 8:
|
|
1126
|
-
continue
|
|
1127
|
-
sent_lower = sent.lower()
|
|
1128
|
-
|
|
1129
|
-
present = [concept_lower[k] for k in concept_lower if k in sent_lower]
|
|
1130
|
-
if len(present) < 2:
|
|
1131
|
-
continue
|
|
1132
|
-
|
|
1133
|
-
relation = infer_edge_relation(sent)
|
|
1134
|
-
edge = relation["relation"]
|
|
1135
|
-
# Enumeration guard (review 2026-07-27 P1 #6): a verb-less sentence
|
|
1136
|
-
# listing many concepts is a list, not a set of relations. Verb-backed
|
|
1137
|
-
# sentences keep every pair — the verb is the evidence.
|
|
1138
|
-
if (
|
|
1139
|
-
relation["evidence"] == "cooccurrence"
|
|
1140
|
-
and len(present) > COOCCURRENCE_CONCEPT_LIMIT
|
|
1141
|
-
):
|
|
1142
|
-
continue
|
|
1143
|
-
|
|
1144
|
-
for i in range(len(present) - 1):
|
|
1145
|
-
subj, obj = present[i], present[i + 1]
|
|
1146
|
-
# Deduplicate by (subj, obj) regardless of direction for same edge
|
|
1147
|
-
pair_key = tuple(sorted([subj.lower(), obj.lower()])) + (edge,)
|
|
1148
|
-
if pair_key in seen_pairs:
|
|
1149
|
-
continue
|
|
1150
|
-
seen_pairs.add(pair_key)
|
|
1151
|
-
triples.append(
|
|
1152
|
-
{
|
|
1153
|
-
"subject": subj,
|
|
1154
|
-
"relation": edge, # verb form (동사)
|
|
1155
|
-
"object": obj,
|
|
1156
|
-
"context": sent[:240],
|
|
1157
|
-
"evidence": relation["evidence"],
|
|
1158
|
-
"weight": relation["weight"],
|
|
1159
|
-
}
|
|
1160
|
-
)
|
|
1161
|
-
if len(triples) >= limit:
|
|
1162
|
-
return triples
|
|
1163
|
-
|
|
1164
|
-
return triples
|
|
1165
|
-
|
|
1166
|
-
|
|
1167
|
-
def _semantic_items(text: str) -> List[Dict[str, str]]:
|
|
1168
|
-
"""Extract explicit decision / task items from text."""
|
|
1169
|
-
items: List[Dict[str, str]] = []
|
|
1170
|
-
for raw_line in str(text or "").splitlines():
|
|
1171
|
-
line = _clean_text(raw_line)
|
|
1172
|
-
if len(line) < 6:
|
|
1173
|
-
continue
|
|
1174
|
-
lowered = line.lower()
|
|
1175
|
-
if re.search(r"(결정|확정|하기로|decided|decision)", lowered):
|
|
1176
|
-
items.append(
|
|
1177
|
-
{"type": "Decision", "title": line[:120], "summary": line[:500]}
|
|
1178
|
-
)
|
|
1179
|
-
if re.search(r"(todo|해야|하자|진행|구현|수정|확인|next|task|\[ \])", lowered):
|
|
1180
|
-
items.append({"type": "Task", "title": line[:120], "summary": line[:500]})
|
|
1181
|
-
return items[:8]
|
|
1182
|
-
|
|
1183
|
-
|
|
1184
|
-
def _topic_candidates(text: str, limit: int = 8) -> List[str]:
|
|
1185
|
-
"""Return compact keyword candidates for fallback graph search."""
|
|
1186
|
-
candidates = _extract_concepts(text, limit=limit)
|
|
1187
|
-
if candidates:
|
|
1188
|
-
return candidates[:limit]
|
|
1189
|
-
seen: Dict[str, str] = {}
|
|
1190
|
-
for token in re.findall(
|
|
1191
|
-
r"[A-Za-z][A-Za-z0-9_.:-]{2,}|[가-힣]{2,12}", str(text or "")
|
|
1192
|
-
):
|
|
1193
|
-
key = token.lower()
|
|
1194
|
-
if key in _CONCEPT_STOP or key.isdigit():
|
|
1195
|
-
continue
|
|
1196
|
-
seen.setdefault(key, token)
|
|
1197
|
-
if len(seen) >= limit:
|
|
1198
|
-
break
|
|
1199
|
-
return list(seen.values())[:limit]
|
|
1200
|
-
|
|
1201
|
-
|
|
1202
|
-
# Static export list. This used to be `[name for name in globals() if not
|
|
1203
|
-
# name.startswith("__")]`, which is invisible to a type checker: every
|
|
1204
|
-
# `from ._kg_common import *` consumer then had *no* resolvable names, and
|
|
1205
|
-
# mypy reported ~750 spurious `name-defined` errors across the graph
|
|
1206
|
-
# package. `tests/unit/test_kg_common_exports.py` asserts this list still
|
|
1207
|
-
# equals what the computed expression would produce, so it cannot drift.
|
|
1208
|
-
__all__ = [
|
|
1209
|
-
"Any",
|
|
1210
|
-
"COMMON_EXCLUDED_DIRS",
|
|
1211
|
-
"COMMON_EXCLUDED_FILE_NAMES",
|
|
1212
|
-
"COMMON_EXCLUDED_FILE_SUFFIXES",
|
|
1213
|
-
"COOCCURRENCE_CONCEPT_LIMIT",
|
|
1214
|
-
"COOCCURRENCE_EDGE_WEIGHT",
|
|
1215
|
-
"Counter",
|
|
1216
|
-
"Dict",
|
|
1217
|
-
"EDGE_VERB",
|
|
1218
|
-
"ENABLE_LLM_EXTRACTION",
|
|
1219
|
-
"EdgeType",
|
|
1220
|
-
"GRAPH_SCHEMA_VERSION",
|
|
1221
|
-
"Iterable",
|
|
1222
|
-
"Iterator",
|
|
1223
|
-
"KGStoreV2",
|
|
1224
|
-
"LINUX_EXCLUDED_PREFIXES",
|
|
1225
|
-
"LOCAL_CODE_EXTENSIONS",
|
|
1226
|
-
"LOCAL_DOCUMENT_EXTENSIONS",
|
|
1227
|
-
"LOCAL_IMAGE_EXTENSIONS",
|
|
1228
|
-
"LOCAL_SIZE_LIMITS",
|
|
1229
|
-
"LOCAL_SLIDE_EXTENSIONS",
|
|
1230
|
-
"LOCAL_SPREADSHEET_EXTENSIONS",
|
|
1231
|
-
"LOCAL_SUPPORTED_EXTENSIONS",
|
|
1232
|
-
"LOCAL_TEXT_EXTENSIONS",
|
|
1233
|
-
"List",
|
|
1234
|
-
"LocalEmbeddingModel",
|
|
1235
|
-
"MACOS_EXCLUDED_PREFIXES",
|
|
1236
|
-
"NodeType",
|
|
1237
|
-
"Optional",
|
|
1238
|
-
"Path",
|
|
1239
|
-
"SENSITIVE_PATH_KEYWORDS",
|
|
1240
|
-
"Tuple",
|
|
1241
|
-
"VERB_EDGE_WEIGHT",
|
|
1242
|
-
"WINDOWS_EXCLUDED_NAMES",
|
|
1243
|
-
"_CHUNK_STRATEGIES",
|
|
1244
|
-
"_CODE_BLANK_RUN_RE",
|
|
1245
|
-
"_CODE_BOUNDARY_LINE_RE",
|
|
1246
|
-
"_CODE_CHUNK_EXTENSIONS",
|
|
1247
|
-
"_CONCEPT_STOP",
|
|
1248
|
-
"_KG_DB_FORMAT_KEY",
|
|
1249
|
-
"_KG_DB_FORMAT_VERSION",
|
|
1250
|
-
"_LLM_EXTRACT_CONCEPT_PROMPT",
|
|
1251
|
-
"_LLM_EXTRACT_TRIPLE_PROMPT",
|
|
1252
|
-
"_MARKDOWN_CHUNK_EXTENSIONS",
|
|
1253
|
-
"_MARKDOWN_HEADING_RE",
|
|
1254
|
-
"_MARKDOWN_MIN_SECTION_CHARS",
|
|
1255
|
-
"_NOT_PERSON_WORDS",
|
|
1256
|
-
"_PROJECTION_VERSION",
|
|
1257
|
-
"_PROSE_CHUNK_EXTENSIONS",
|
|
1258
|
-
"_PROSE_MIN_SPAN_RATIO",
|
|
1259
|
-
"_PROSE_STRONG_BOUNDARY_RE",
|
|
1260
|
-
"_PROSE_WEAK_BOUNDARY_RE",
|
|
1261
|
-
"_READ_FROM_V2_DEFAULT",
|
|
1262
|
-
"_V2_WRITE_MASTER_KEY",
|
|
1263
|
-
"_chunks",
|
|
1264
|
-
"_classify_node_type",
|
|
1265
|
-
"_clean_text",
|
|
1266
|
-
"_code_chunks",
|
|
1267
|
-
"_code_segment_spans",
|
|
1268
|
-
"_current_os_type",
|
|
1269
|
-
"_drive_id_for_path",
|
|
1270
|
-
"_excluded_directory_reason",
|
|
1271
|
-
"_exec_script",
|
|
1272
|
-
"_extract_concepts",
|
|
1273
|
-
"_extract_concepts_rules",
|
|
1274
|
-
"_extract_triples",
|
|
1275
|
-
"_extract_triples_rules",
|
|
1276
|
-
"_file_category",
|
|
1277
|
-
"_infer_edge",
|
|
1278
|
-
"_is_hidden_path",
|
|
1279
|
-
"_is_relative_to",
|
|
1280
|
-
"_json",
|
|
1281
|
-
"_last_boundary",
|
|
1282
|
-
"_llm_extract_concepts",
|
|
1283
|
-
"_llm_extract_triples",
|
|
1284
|
-
"_markdown_chunks",
|
|
1285
|
-
"_markdown_section_spans",
|
|
1286
|
-
"_merge_small_sections",
|
|
1287
|
-
"_node_type_for_category",
|
|
1288
|
-
"_now",
|
|
1289
|
-
"_parse_iso",
|
|
1290
|
-
"_parser_type_for_category",
|
|
1291
|
-
"_path_fingerprint",
|
|
1292
|
-
"_path_parts_lower",
|
|
1293
|
-
"_plain_windows",
|
|
1294
|
-
"_prose_chunks",
|
|
1295
|
-
"_recency_score",
|
|
1296
|
-
"_root_warning",
|
|
1297
|
-
"_safe_iso_from_stat_mtime",
|
|
1298
|
-
"_safe_loads",
|
|
1299
|
-
"_sample_file",
|
|
1300
|
-
"_semantic_items",
|
|
1301
|
-
"_sensitive_file_reason",
|
|
1302
|
-
"_sha256_bytes",
|
|
1303
|
-
"_sha256_text",
|
|
1304
|
-
"_size_limit_for_category",
|
|
1305
|
-
"_slug",
|
|
1306
|
-
"_topic_candidates",
|
|
1307
|
-
"asyncio",
|
|
1308
|
-
"chunk_strategy_for",
|
|
1309
|
-
"citation_locator",
|
|
1310
|
-
"contextmanager",
|
|
1311
|
-
"datetime",
|
|
1312
|
-
"get_llm_router",
|
|
1313
|
-
"hashlib",
|
|
1314
|
-
"infer_edge_relation",
|
|
1315
|
-
"json",
|
|
1316
|
-
"logging",
|
|
1317
|
-
"math",
|
|
1318
|
-
"os",
|
|
1319
|
-
"page_for_offset",
|
|
1320
|
-
"pdf_page_offsets",
|
|
1321
|
-
"platform",
|
|
1322
|
-
"quiet",
|
|
1323
|
-
"re",
|
|
1324
|
-
"set_llm_router",
|
|
1325
|
-
"shutil",
|
|
1326
|
-
"sqlite3",
|
|
1327
|
-
"time",
|
|
1328
|
-
"typed_chunk_meta_fields",
|
|
1329
|
-
"typed_chunks",
|
|
1330
|
-
"zipfile",
|
|
1331
|
-
]
|