ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,209 @@
|
|
|
1
|
+
"""Advisory extraction-quality scoring, and the capture CTA built on it.
|
|
2
|
+
|
|
3
|
+
Pure heuristics over already-extracted text — no model call, no network, and
|
|
4
|
+
deterministic. The score never blocks an ingest; it annotates the result so a
|
|
5
|
+
capture surface can say "this capture is thin" and offer a way to fix it
|
|
6
|
+
instead of silently storing junk.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Any, Dict, List, Optional
|
|
12
|
+
|
|
13
|
+
# ── Extraction quality heuristics (v9.8.0 A1) ────────────────────────────────
|
|
14
|
+
# Pure heuristics over the extracted text — no model calls, no network. The
|
|
15
|
+
# score is *advisory*: it never blocks an ingest, it only annotates the result
|
|
16
|
+
# so capture surfaces (browser, folder scan) can surface low-quality warnings.
|
|
17
|
+
QUALITY_HIGH_THRESHOLD = 0.7
|
|
18
|
+
QUALITY_LOW_THRESHOLD = 0.4
|
|
19
|
+
QUALITY_LOW_WARNING = "추출 품질이 낮습니다 — 원문 확인을 권장합니다."
|
|
20
|
+
_WEB_SOURCE_TYPES = frozenset({"web_url", "browser_tab"})
|
|
21
|
+
# Standalone short lines that smell like leftover site chrome (nav/menu/footer).
|
|
22
|
+
_BOILERPLATE_LINE_MARKERS = frozenset(
|
|
23
|
+
{
|
|
24
|
+
"home", "menu", "nav", "navigation", "login", "log in", "sign in",
|
|
25
|
+
"sign up", "register", "subscribe", "search", "about", "about us",
|
|
26
|
+
"contact", "contact us", "privacy policy", "terms of service",
|
|
27
|
+
"cookie policy", "accept cookies", "accept all cookies", "share",
|
|
28
|
+
"skip to content", "copyright", "all rights reserved", "sitemap",
|
|
29
|
+
"back to top", "footer", "read more", "next", "previous",
|
|
30
|
+
}
|
|
31
|
+
)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _quality_level(score: float) -> str:
|
|
35
|
+
if score >= QUALITY_HIGH_THRESHOLD:
|
|
36
|
+
return "high"
|
|
37
|
+
if score >= QUALITY_LOW_THRESHOLD:
|
|
38
|
+
return "medium"
|
|
39
|
+
return "low"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def assess_extraction_quality(
|
|
43
|
+
text: Optional[str],
|
|
44
|
+
*,
|
|
45
|
+
source_type: Optional[str] = None,
|
|
46
|
+
upstream_confidence: Optional[Any] = None,
|
|
47
|
+
) -> Dict[str, Any]:
|
|
48
|
+
"""Score extracted text 0..1 with reasons (pure heuristic, deterministic).
|
|
49
|
+
|
|
50
|
+
Signals: text length, whitespace ratio, character/word diversity
|
|
51
|
+
(repetition), sentence structure, and — for web sources — leftover
|
|
52
|
+
nav/menu boilerplate. When the upstream extractor supplies its own
|
|
53
|
+
confidence (``upstream_confidence``), that value wins verbatim: the
|
|
54
|
+
extractor saw the raw document, this function only sees its output.
|
|
55
|
+
"""
|
|
56
|
+
if upstream_confidence is not None:
|
|
57
|
+
try:
|
|
58
|
+
score = max(0.0, min(1.0, float(upstream_confidence)))
|
|
59
|
+
except (TypeError, ValueError):
|
|
60
|
+
score = None
|
|
61
|
+
if score is not None:
|
|
62
|
+
return {
|
|
63
|
+
"score": round(score, 4),
|
|
64
|
+
"level": _quality_level(score),
|
|
65
|
+
"reasons": ["upstream_confidence"],
|
|
66
|
+
}
|
|
67
|
+
|
|
68
|
+
raw = str(text or "")
|
|
69
|
+
stripped = raw.strip()
|
|
70
|
+
if not stripped:
|
|
71
|
+
return {"score": 0.0, "level": "low", "reasons": ["empty_text"]}
|
|
72
|
+
|
|
73
|
+
reasons: List[str] = []
|
|
74
|
+
length = len(stripped)
|
|
75
|
+
sample = stripped[:4000]
|
|
76
|
+
lines = [ln.strip() for ln in stripped.splitlines() if ln.strip()]
|
|
77
|
+
words = stripped.split()
|
|
78
|
+
|
|
79
|
+
# 1) Length — very short extractions rarely carry recall value.
|
|
80
|
+
if length < 40:
|
|
81
|
+
length_factor = 0.35
|
|
82
|
+
reasons.append("very_short_text")
|
|
83
|
+
elif length < 120:
|
|
84
|
+
length_factor = 0.6
|
|
85
|
+
reasons.append("short_text")
|
|
86
|
+
elif length < 300:
|
|
87
|
+
length_factor = 0.85
|
|
88
|
+
else:
|
|
89
|
+
length_factor = 1.0
|
|
90
|
+
|
|
91
|
+
# 2) Sentence structure — prose has sentence-ending punctuation.
|
|
92
|
+
sentence_marks = sum(sample.count(mark) for mark in (".", "!", "?", "…", "。", "!", "?"))
|
|
93
|
+
if sentence_marks > 0:
|
|
94
|
+
structure_factor = 1.0
|
|
95
|
+
elif length < 200:
|
|
96
|
+
structure_factor = 0.75 # titles/snippets legitimately lack periods
|
|
97
|
+
else:
|
|
98
|
+
structure_factor = 0.45
|
|
99
|
+
reasons.append("no_sentence_structure")
|
|
100
|
+
|
|
101
|
+
# 3) Diversity — repeated characters/lines/words indicate extraction junk.
|
|
102
|
+
diversity_factor = 1.0
|
|
103
|
+
distinct_chars = len(set(sample.lower()))
|
|
104
|
+
if distinct_chars < 10:
|
|
105
|
+
diversity_factor *= 0.2
|
|
106
|
+
reasons.append("low_character_diversity")
|
|
107
|
+
elif distinct_chars < 20:
|
|
108
|
+
diversity_factor *= 0.7
|
|
109
|
+
if len(lines) >= 6:
|
|
110
|
+
top_count = max(lines.count(ln) for ln in set(lines))
|
|
111
|
+
if top_count >= max(3, len(lines) // 4):
|
|
112
|
+
diversity_factor *= 0.5
|
|
113
|
+
reasons.append("repetitive_lines")
|
|
114
|
+
if len(words) >= 30 and (len(set(w.lower() for w in words)) / len(words)) < 0.25:
|
|
115
|
+
diversity_factor *= 0.5
|
|
116
|
+
reasons.append("repetitive_words")
|
|
117
|
+
|
|
118
|
+
# 4) Cleanliness — whitespace floods, fragmented lines, site chrome.
|
|
119
|
+
cleanliness_factor = 1.0
|
|
120
|
+
whitespace_ratio = sum(1 for ch in raw if ch.isspace()) / max(1, len(raw))
|
|
121
|
+
if whitespace_ratio > 0.45:
|
|
122
|
+
cleanliness_factor *= 0.6
|
|
123
|
+
reasons.append("high_whitespace_ratio")
|
|
124
|
+
if len(lines) >= 8:
|
|
125
|
+
short_lines = sum(1 for ln in lines if len(ln.split()) <= 3)
|
|
126
|
+
if short_lines / len(lines) > 0.6:
|
|
127
|
+
cleanliness_factor *= 0.6
|
|
128
|
+
reasons.append("fragmented_lines")
|
|
129
|
+
boilerplate_hits = sum(
|
|
130
|
+
1 for ln in lines if ln.lower().strip(" .:>|•·-–—*") in _BOILERPLATE_LINE_MARKERS
|
|
131
|
+
)
|
|
132
|
+
if lines and boilerplate_hits >= 3 and (boilerplate_hits / len(lines)) > 0.2:
|
|
133
|
+
cleanliness_factor *= 0.35
|
|
134
|
+
if str(source_type or "").lower() in _WEB_SOURCE_TYPES:
|
|
135
|
+
reasons.append("nav_menu_remnants")
|
|
136
|
+
else:
|
|
137
|
+
reasons.append("boilerplate_markers")
|
|
138
|
+
|
|
139
|
+
score = length_factor * structure_factor * diversity_factor * cleanliness_factor
|
|
140
|
+
score = max(0.0, min(1.0, score))
|
|
141
|
+
if not reasons:
|
|
142
|
+
reasons.append("clean_extraction")
|
|
143
|
+
return {"score": round(score, 4), "level": _quality_level(score), "reasons": reasons}
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
# ── capture quality CTA (backlog #9, review §7.2 C) ──────────────────────────
|
|
147
|
+
# Structured verdict over the same extraction-quality schema the rest of the
|
|
148
|
+
# pipeline uses, so capture surfaces (browser extension, read-url) can render
|
|
149
|
+
# an honest "this capture is thin" CTA instead of silently storing junk.
|
|
150
|
+
CAPTURE_SUGGESTIONS_THIN = ["recapture", "paste_manually", "highlight_source"]
|
|
151
|
+
_CAPTURE_REASON_LABELS = {
|
|
152
|
+
"empty_text": "추출된 본문이 비어 있습니다",
|
|
153
|
+
"very_short_text": "추출된 본문이 매우 짧습니다",
|
|
154
|
+
"short_text": "추출된 본문이 짧습니다",
|
|
155
|
+
"no_sentence_structure": "문장 구조가 거의 없습니다",
|
|
156
|
+
"low_character_diversity": "반복 문자가 대부분입니다",
|
|
157
|
+
"repetitive_lines": "같은 줄이 반복됩니다",
|
|
158
|
+
"repetitive_words": "같은 단어가 반복됩니다",
|
|
159
|
+
"high_whitespace_ratio": "공백이 지나치게 많습니다",
|
|
160
|
+
"fragmented_lines": "줄이 잘게 조각나 있습니다",
|
|
161
|
+
"nav_menu_remnants": "메뉴/내비게이션 잔여물이 많습니다",
|
|
162
|
+
"boilerplate_markers": "상용구 텍스트가 많습니다",
|
|
163
|
+
"no_extracted_text": "추출된 텍스트가 없습니다",
|
|
164
|
+
}
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def capture_quality_verdict(
|
|
168
|
+
extraction_quality: Optional[Dict[str, Any]],
|
|
169
|
+
*,
|
|
170
|
+
source_type: Optional[str] = None,
|
|
171
|
+
) -> Dict[str, Any]:
|
|
172
|
+
"""Structured CTA verdict from a pipeline ``extraction_quality`` dict.
|
|
173
|
+
|
|
174
|
+
``{"status": "thin"|"ok", "reason": str|None, "suggestions": [...],
|
|
175
|
+
"score": float|None, "level": str|None}``. ``thin`` (level == "low", the
|
|
176
|
+
same threshold as the ingest warning) carries actionable suggestions —
|
|
177
|
+
``recapture`` / ``paste_manually`` / ``highlight_source`` — so the UI can
|
|
178
|
+
offer the user a way to fix the capture instead of hiding the problem.
|
|
179
|
+
Deterministic and never raises; ``None`` input yields an honest ``thin``.
|
|
180
|
+
"""
|
|
181
|
+
if not isinstance(extraction_quality, dict):
|
|
182
|
+
return {
|
|
183
|
+
"status": "thin",
|
|
184
|
+
"reason": _CAPTURE_REASON_LABELS["no_extracted_text"],
|
|
185
|
+
"reason_codes": ["no_extracted_text"],
|
|
186
|
+
"suggestions": list(CAPTURE_SUGGESTIONS_THIN),
|
|
187
|
+
"score": None,
|
|
188
|
+
"level": None,
|
|
189
|
+
}
|
|
190
|
+
level = str(extraction_quality.get("level") or "")
|
|
191
|
+
score = extraction_quality.get("score")
|
|
192
|
+
reasons = [str(item) for item in (extraction_quality.get("reasons") or [])]
|
|
193
|
+
thin = level == "low"
|
|
194
|
+
reason = None
|
|
195
|
+
if thin:
|
|
196
|
+
labeled = [
|
|
197
|
+
_CAPTURE_REASON_LABELS[code]
|
|
198
|
+
for code in reasons
|
|
199
|
+
if code in _CAPTURE_REASON_LABELS
|
|
200
|
+
]
|
|
201
|
+
reason = "; ".join(labeled) if labeled else QUALITY_LOW_WARNING
|
|
202
|
+
return {
|
|
203
|
+
"status": "thin" if thin else "ok",
|
|
204
|
+
"reason": reason,
|
|
205
|
+
"reason_codes": reasons if thin else [],
|
|
206
|
+
"suggestions": list(CAPTURE_SUGGESTIONS_THIN) if thin else [],
|
|
207
|
+
"score": score,
|
|
208
|
+
"level": level or None,
|
|
209
|
+
}
|
|
@@ -0,0 +1,295 @@
|
|
|
1
|
+
"""One door per kind of thing: text, chat, memory record, picture, film, file.
|
|
2
|
+
|
|
3
|
+
Every method here returns the raw store payload that ``IngestionPipeline.\
|
|
4
|
+
ingest`` normalizes; none of them decide *whether* to run. Modality routing
|
|
5
|
+
(:meth:`IngestionRoutingMixin._modality_for`) answers ``"text"`` for everything
|
|
6
|
+
while multi-modal is off, which is what makes "off" mean *unchanged* rather than
|
|
7
|
+
*slightly different*.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Any, Dict
|
|
14
|
+
|
|
15
|
+
from ..multimodal import (
|
|
16
|
+
MODALITY_AUDIO,
|
|
17
|
+
MODALITY_IMAGE,
|
|
18
|
+
MODALITY_VIDEO,
|
|
19
|
+
ImageFacts,
|
|
20
|
+
audio_quality_score,
|
|
21
|
+
detect_modality,
|
|
22
|
+
extract_image_facts,
|
|
23
|
+
image_quality_score,
|
|
24
|
+
read_video_facts,
|
|
25
|
+
transcribe_audio,
|
|
26
|
+
video_frame_dir,
|
|
27
|
+
video_quality_score,
|
|
28
|
+
write_image_memory,
|
|
29
|
+
write_video_memory,
|
|
30
|
+
)
|
|
31
|
+
from ..utils import utc_now_iso
|
|
32
|
+
from ._contract import IngestionCore as _Core
|
|
33
|
+
from .constants import (
|
|
34
|
+
_MEMORY_NODE_TYPES,
|
|
35
|
+
AUDIO_NODE_TYPE,
|
|
36
|
+
AUDIO_SOURCE_TYPES,
|
|
37
|
+
IMAGE_SOURCE_TYPES,
|
|
38
|
+
VIDEO_SOURCE_TYPES,
|
|
39
|
+
)
|
|
40
|
+
from .hashing import _file_digest
|
|
41
|
+
from .models import IngestionItem
|
|
42
|
+
from .quality import _quality_level
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
class IngestionRoutingMixin(_Core):
|
|
46
|
+
"""The per-source-type ingest doors. Mixed into ``IngestionPipeline``."""
|
|
47
|
+
|
|
48
|
+
# ── routing helpers ──────────────────────────────────────────────────────
|
|
49
|
+
def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
50
|
+
text = item.text or ""
|
|
51
|
+
if not text.strip():
|
|
52
|
+
raise ValueError(
|
|
53
|
+
f"Empty content: {source_type} ingestion requires non-empty text."
|
|
54
|
+
)
|
|
55
|
+
if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
|
|
56
|
+
raise ValueError(
|
|
57
|
+
f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
|
|
58
|
+
)
|
|
59
|
+
title = item.title or item.source_uri or source_type
|
|
60
|
+
return self._kg.ingest_source(
|
|
61
|
+
source_type=source_type,
|
|
62
|
+
title=title,
|
|
63
|
+
text=text,
|
|
64
|
+
source_uri=item.source_uri,
|
|
65
|
+
owner=owner,
|
|
66
|
+
workspace_id=item.workspace_id,
|
|
67
|
+
permissions=item.permissions,
|
|
68
|
+
captured_at=captured_at,
|
|
69
|
+
modified_at=item.modified_at,
|
|
70
|
+
conversation_id=item.conversation_id,
|
|
71
|
+
metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
75
|
+
text = item.text or ""
|
|
76
|
+
meta = item.metadata or {}
|
|
77
|
+
role = str(meta.get("role") or "user")
|
|
78
|
+
result = self._kg.ingest_message(
|
|
79
|
+
role,
|
|
80
|
+
text,
|
|
81
|
+
user_email=owner,
|
|
82
|
+
user_nickname=meta.get("user_nickname"),
|
|
83
|
+
source=meta.get("source") or source_type,
|
|
84
|
+
conversation_id=item.conversation_id,
|
|
85
|
+
workspace_id=item.workspace_id,
|
|
86
|
+
raw=meta.get("raw"),
|
|
87
|
+
)
|
|
88
|
+
# ingest_message reports message/response node ids; normalize the keys
|
|
89
|
+
# the provenance step expects.
|
|
90
|
+
result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
|
|
91
|
+
result.setdefault("title", item.title or text[:80])
|
|
92
|
+
return result
|
|
93
|
+
|
|
94
|
+
def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
95
|
+
node_type = _MEMORY_NODE_TYPES[source_type]
|
|
96
|
+
meta = item.metadata or {}
|
|
97
|
+
result = self._kg.ingest_event(
|
|
98
|
+
node_type,
|
|
99
|
+
item.title or (item.text or node_type)[:120],
|
|
100
|
+
user_email=owner,
|
|
101
|
+
source=meta.get("source") or source_type,
|
|
102
|
+
conversation_id=item.conversation_id,
|
|
103
|
+
workspace_id=item.workspace_id,
|
|
104
|
+
metadata={**meta, "detail": (item.text or "")[:2000]},
|
|
105
|
+
)
|
|
106
|
+
result.setdefault("node_id", result.get("node_id") or result.get("id"))
|
|
107
|
+
result.setdefault("title", item.title)
|
|
108
|
+
return result
|
|
109
|
+
|
|
110
|
+
# ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
|
|
111
|
+
def _modality_for(self, item: IngestionItem, source_type: str) -> str:
|
|
112
|
+
"""``image`` / ``audio`` / ``video`` / ``text`` for this item.
|
|
113
|
+
|
|
114
|
+
Always ``"text"`` while the flag is off, which is what makes "off" mean
|
|
115
|
+
*unchanged* rather than *slightly different*.
|
|
116
|
+
"""
|
|
117
|
+
if not self._allow_multimodal:
|
|
118
|
+
return "text"
|
|
119
|
+
if source_type in IMAGE_SOURCE_TYPES:
|
|
120
|
+
return MODALITY_IMAGE
|
|
121
|
+
if source_type in AUDIO_SOURCE_TYPES:
|
|
122
|
+
return MODALITY_AUDIO
|
|
123
|
+
if source_type in VIDEO_SOURCE_TYPES:
|
|
124
|
+
return MODALITY_VIDEO
|
|
125
|
+
if not item.path:
|
|
126
|
+
return "text"
|
|
127
|
+
return detect_modality(item.path, item.mime_type)
|
|
128
|
+
|
|
129
|
+
def _resolve_file_path(self, item: IngestionItem) -> Path:
|
|
130
|
+
if not item.path:
|
|
131
|
+
raise ValueError("File ingestion requires a path.")
|
|
132
|
+
path = Path(item.path)
|
|
133
|
+
if not path.exists():
|
|
134
|
+
raise FileNotFoundError(f"File not found: {path}")
|
|
135
|
+
if path.is_dir():
|
|
136
|
+
raise ValueError(f"File ingestion requires a file, got a directory: {path}")
|
|
137
|
+
return path
|
|
138
|
+
|
|
139
|
+
def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
140
|
+
"""Store one picture as an ``Image`` node — OCR, caption, vector.
|
|
141
|
+
|
|
142
|
+
The image vector (when a vision model produced one) goes to its own
|
|
143
|
+
index; the OCR/caption text rides the ordinary text index. That split
|
|
144
|
+
is what lets a typed question find a screenshot without ever comparing
|
|
145
|
+
a text vector to an image vector.
|
|
146
|
+
"""
|
|
147
|
+
path = self._resolve_file_path(item)
|
|
148
|
+
facts = extract_image_facts(str(path), ports=self._multimodal)
|
|
149
|
+
result = write_image_memory(
|
|
150
|
+
self._kg,
|
|
151
|
+
path=path,
|
|
152
|
+
facts=facts,
|
|
153
|
+
title=item.title or path.name,
|
|
154
|
+
source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
|
|
155
|
+
source_uri=item.source_uri,
|
|
156
|
+
owner=owner,
|
|
157
|
+
workspace_id=item.workspace_id,
|
|
158
|
+
conversation_id=item.conversation_id,
|
|
159
|
+
captured_at=captured_at,
|
|
160
|
+
modified_at=item.modified_at,
|
|
161
|
+
permissions=item.permissions,
|
|
162
|
+
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
163
|
+
)
|
|
164
|
+
self._record_image_vector(result["node_id"], facts)
|
|
165
|
+
quality = image_quality_score(facts)
|
|
166
|
+
result["extraction_quality"] = {
|
|
167
|
+
"score": quality["score"],
|
|
168
|
+
"level": _quality_level(quality["score"]),
|
|
169
|
+
"reasons": quality["reasons"],
|
|
170
|
+
}
|
|
171
|
+
return result
|
|
172
|
+
|
|
173
|
+
def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
|
|
174
|
+
"""File the image-space vector, if a vision model actually made one."""
|
|
175
|
+
if facts.embedding is None:
|
|
176
|
+
return
|
|
177
|
+
from ..graph.image_vectors import record_image_vector
|
|
178
|
+
|
|
179
|
+
record_image_vector(
|
|
180
|
+
self._kg,
|
|
181
|
+
node_id=node_id,
|
|
182
|
+
vector=facts.embedding,
|
|
183
|
+
model_id=self._multimodal.vision_model_id or "vision:unnamed",
|
|
184
|
+
space=self._multimodal.vision_space,
|
|
185
|
+
updated_at=utc_now_iso(),
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
189
|
+
"""Store one recording as an ``Audio`` node, transcribed when possible.
|
|
190
|
+
|
|
191
|
+
The transcript is text and rides the ordinary text index — chunks,
|
|
192
|
+
concepts, provenance, dedupe all unchanged — but the node itself is a
|
|
193
|
+
recording, because that is what it is whether or not anyone could hear
|
|
194
|
+
it. The recording's own facts stay in the metadata (``modality``,
|
|
195
|
+
``audio_path``, ``transcription``, ``searchable``). Without a
|
|
196
|
+
transcriber the memory is still kept, and its body says plainly that
|
|
197
|
+
the words were never recognized instead of leaving a blank note.
|
|
198
|
+
"""
|
|
199
|
+
path = self._resolve_file_path(item)
|
|
200
|
+
facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
|
|
201
|
+
title = item.title or path.stem
|
|
202
|
+
body = facts.transcript or (
|
|
203
|
+
f"[{MODALITY_AUDIO}] {title}\n"
|
|
204
|
+
"이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
|
|
205
|
+
)
|
|
206
|
+
result = self._kg.ingest_source(
|
|
207
|
+
source_type=source_type,
|
|
208
|
+
title=title,
|
|
209
|
+
text=body,
|
|
210
|
+
source_uri=item.source_uri or str(path),
|
|
211
|
+
owner=owner,
|
|
212
|
+
workspace_id=item.workspace_id,
|
|
213
|
+
permissions=item.permissions,
|
|
214
|
+
captured_at=captured_at,
|
|
215
|
+
modified_at=item.modified_at,
|
|
216
|
+
conversation_id=item.conversation_id,
|
|
217
|
+
node_type=AUDIO_NODE_TYPE,
|
|
218
|
+
metadata={
|
|
219
|
+
"mime_type": item.mime_type,
|
|
220
|
+
"modality": MODALITY_AUDIO,
|
|
221
|
+
"audio_path": str(path),
|
|
222
|
+
"audio_bytes": path.stat().st_size,
|
|
223
|
+
"transcription": facts.transcription_status,
|
|
224
|
+
"searchable": facts.searchable,
|
|
225
|
+
**({"transcription_detail": facts.detail} if facts.detail else {}),
|
|
226
|
+
**(item.metadata or {}),
|
|
227
|
+
},
|
|
228
|
+
)
|
|
229
|
+
result.setdefault("title", title)
|
|
230
|
+
quality = audio_quality_score(facts)
|
|
231
|
+
result["extraction_quality"] = {
|
|
232
|
+
"score": quality["score"],
|
|
233
|
+
"level": _quality_level(quality["score"]),
|
|
234
|
+
"reasons": quality["reasons"],
|
|
235
|
+
}
|
|
236
|
+
return result
|
|
237
|
+
|
|
238
|
+
def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
239
|
+
"""Store one video as keyframes through the image door plus subtitles.
|
|
240
|
+
|
|
241
|
+
Nothing here is a new retrieval path: the stills become ordinary
|
|
242
|
+
``Image`` nodes (OCR, caption, vector, thumbnail) joined by
|
|
243
|
+
``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
|
|
244
|
+
the ``Video`` node adds is the thing they belong to — and an honest
|
|
245
|
+
body when there were no subtitles to read.
|
|
246
|
+
"""
|
|
247
|
+
path = self._resolve_file_path(item)
|
|
248
|
+
facts = read_video_facts(
|
|
249
|
+
str(path),
|
|
250
|
+
video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
|
|
251
|
+
count=self._keyframes,
|
|
252
|
+
ports=self._multimodal,
|
|
253
|
+
subtitle_text=item.text,
|
|
254
|
+
)
|
|
255
|
+
result = write_video_memory(
|
|
256
|
+
self._kg,
|
|
257
|
+
path=path,
|
|
258
|
+
facts=facts,
|
|
259
|
+
title=item.title or path.stem,
|
|
260
|
+
source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
|
|
261
|
+
source_uri=item.source_uri,
|
|
262
|
+
owner=owner,
|
|
263
|
+
workspace_id=item.workspace_id,
|
|
264
|
+
conversation_id=item.conversation_id,
|
|
265
|
+
captured_at=captured_at,
|
|
266
|
+
modified_at=item.modified_at,
|
|
267
|
+
permissions=item.permissions,
|
|
268
|
+
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
269
|
+
ports=self._multimodal,
|
|
270
|
+
)
|
|
271
|
+
quality = video_quality_score(facts)
|
|
272
|
+
result["extraction_quality"] = {
|
|
273
|
+
"score": quality["score"],
|
|
274
|
+
"level": _quality_level(quality["score"]),
|
|
275
|
+
"reasons": quality["reasons"],
|
|
276
|
+
}
|
|
277
|
+
return result
|
|
278
|
+
|
|
279
|
+
def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
280
|
+
path = self._resolve_file_path(item)
|
|
281
|
+
return self._kg.ingest_document(
|
|
282
|
+
path,
|
|
283
|
+
original_filename=item.title or path.name,
|
|
284
|
+
mime_type=item.mime_type,
|
|
285
|
+
uploader=owner,
|
|
286
|
+
conversation_id=item.conversation_id,
|
|
287
|
+
extracted=item.metadata.get("extracted") if item.metadata else None,
|
|
288
|
+
source_type=source_type,
|
|
289
|
+
source_uri=item.source_uri or str(path),
|
|
290
|
+
captured_at=captured_at,
|
|
291
|
+
modified_at=item.modified_at,
|
|
292
|
+
owner=owner,
|
|
293
|
+
workspace_id=item.workspace_id,
|
|
294
|
+
permissions=item.permissions,
|
|
295
|
+
)
|
|
@@ -0,0 +1,164 @@
|
|
|
1
|
+
"""Images and audio as first-class memories (v11.1.0, Track 3).
|
|
2
|
+
|
|
3
|
+
Before this module the Brain could only remember things that arrived as text.
|
|
4
|
+
A screenshot of a whiteboard, a photo of a receipt, a voice memo — all of them
|
|
5
|
+
either bounced off the ingestion pipeline or landed as an opaque ``Document``
|
|
6
|
+
node whose only searchable content was its filename.
|
|
7
|
+
|
|
8
|
+
What routes here
|
|
9
|
+
----------------
|
|
10
|
+
:func:`detect_modality` reads the MIME type first and the extension second, and
|
|
11
|
+
answers with one of ``text`` / ``image`` / ``audio`` / ``video``. ``video`` is
|
|
12
|
+
deliberately a *recognized but unsupported* answer in this release: keyframe
|
|
13
|
+
extraction needs a decoder this project does not ship, and returning "video,
|
|
14
|
+
out of scope" is worth more than pretending a ``.mov`` is a picture.
|
|
15
|
+
|
|
16
|
+
What an image memory contains
|
|
17
|
+
-----------------------------
|
|
18
|
+
:func:`extract_image_facts` gathers only what it can actually observe:
|
|
19
|
+
|
|
20
|
+
* **dimensions/format** from Pillow (a core dependency);
|
|
21
|
+
* **ocr_text** from ``pytesseract`` when it is installed — otherwise
|
|
22
|
+
``ocr_status="unavailable"`` and no text, never an empty string dressed up as
|
|
23
|
+
a successful read;
|
|
24
|
+
* **caption** from an injected vision-language port, and *only* from there. No
|
|
25
|
+
VLM means ``caption is None``. Composing "Image IMG_2381.png (JPEG 3024x4032)"
|
|
26
|
+
and storing it in the caption field would make metadata indistinguishable
|
|
27
|
+
from a model's description forever after;
|
|
28
|
+
* **embedding** from an injected vision port, which lives in its own vector
|
|
29
|
+
space (see :mod:`latticeai.core.embedding_providers`) and therefore its own
|
|
30
|
+
index — text queries reach images through OCR/caption text, not by scoring a
|
|
31
|
+
BGE vector against CLIP vectors.
|
|
32
|
+
|
|
33
|
+
Brain Core owns none of those models. Every heavy dependency arrives as an
|
|
34
|
+
injected callable (:class:`MultimodalPorts`), which is also why this module
|
|
35
|
+
imports nothing from ``latticeai``.
|
|
36
|
+
|
|
37
|
+
Split into cohesive submodules in v11.3.0 (no behaviour change): ``common``
|
|
38
|
+
(taxonomy + shared helpers), ``ports`` (injected capabilities + the ffmpeg
|
|
39
|
+
fallback), ``images``, ``audio``, ``video``. This module re-exports every name
|
|
40
|
+
the single file exposed, so ``lattice_brain.multimodal.X`` keeps working.
|
|
41
|
+
"""
|
|
42
|
+
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
# Internals that predate the split. They are not public API, but they were
|
|
46
|
+
# reachable as ``lattice_brain.multimodal.<name>`` before it and callers (and
|
|
47
|
+
# tests) still reach them that way, so they are re-exported explicitly. The
|
|
48
|
+
# redundant-alias form says "this is a re-export", not a leftover import.
|
|
49
|
+
#
|
|
50
|
+
# Stubbing note: replacing one of these *here* rebinds only this module's name.
|
|
51
|
+
# The submodule that calls it holds its own reference, so a test that wants to
|
|
52
|
+
# stand in for ``_which_ffmpeg`` patches ``lattice_brain.multimodal.ports``.
|
|
53
|
+
from ..quiet import quiet as quiet
|
|
54
|
+
from ..utils import utc_now_iso as utc_now_iso
|
|
55
|
+
from .audio import AudioFacts, audio_quality_score, transcribe_audio
|
|
56
|
+
from .common import (
|
|
57
|
+
AUDIO_EXTENSIONS,
|
|
58
|
+
IMAGE_CHUNK_CHARS,
|
|
59
|
+
IMAGE_EXTENSIONS,
|
|
60
|
+
MAX_INDEX_TEXT_CHARS,
|
|
61
|
+
MAX_THUMBNAIL_CHARS,
|
|
62
|
+
MODALITY_AUDIO,
|
|
63
|
+
MODALITY_IMAGE,
|
|
64
|
+
MODALITY_TEXT,
|
|
65
|
+
MODALITY_VIDEO,
|
|
66
|
+
SUBTITLE_EXTENSIONS,
|
|
67
|
+
SUMMARY_CHARS,
|
|
68
|
+
THUMBNAIL_EDGE,
|
|
69
|
+
VIDEO_EXTENSIONS,
|
|
70
|
+
VIDEO_OUT_OF_SCOPE,
|
|
71
|
+
VIDEO_UNAVAILABLE_DETAIL,
|
|
72
|
+
detect_modality,
|
|
73
|
+
)
|
|
74
|
+
from .common import _sha256_file as _sha256_file
|
|
75
|
+
from .common import _sha256_text as _sha256_text
|
|
76
|
+
from .common import _split_index_text as _split_index_text
|
|
77
|
+
from .images import (
|
|
78
|
+
ImageFacts,
|
|
79
|
+
extract_image_facts,
|
|
80
|
+
image_node_id,
|
|
81
|
+
image_quality_score,
|
|
82
|
+
write_image_memory,
|
|
83
|
+
)
|
|
84
|
+
from .images import _apply_vision_embedding as _apply_vision_embedding
|
|
85
|
+
from .images import _attach_concepts as _attach_concepts
|
|
86
|
+
from .images import _attach_source as _attach_source
|
|
87
|
+
from .images import _open_image as _open_image
|
|
88
|
+
from .images import _run_ocr as _run_ocr
|
|
89
|
+
from .images import _safe_caption as _safe_caption
|
|
90
|
+
from .images import _thumbnail_data_uri as _thumbnail_data_uri
|
|
91
|
+
from .ports import (
|
|
92
|
+
DEFAULT_KEYFRAMES,
|
|
93
|
+
FFMPEG_BINARY,
|
|
94
|
+
MultimodalPorts,
|
|
95
|
+
extract_keyframes,
|
|
96
|
+
ffmpeg_available,
|
|
97
|
+
)
|
|
98
|
+
from .ports import KEYFRAME_TIMEOUT_SECONDS as KEYFRAME_TIMEOUT_SECONDS
|
|
99
|
+
from .ports import KEYFRAME_WINDOW as KEYFRAME_WINDOW
|
|
100
|
+
from .ports import _injected_keyframes as _injected_keyframes
|
|
101
|
+
from .ports import _run_ffmpeg as _run_ffmpeg
|
|
102
|
+
from .ports import _which_ffmpeg as _which_ffmpeg
|
|
103
|
+
from .video import _CUE_TAG_RE as _CUE_TAG_RE
|
|
104
|
+
from .video import _SRT_INDEX_RE as _SRT_INDEX_RE
|
|
105
|
+
from .video import _TIMECODE_RE as _TIMECODE_RE
|
|
106
|
+
from .video import (
|
|
107
|
+
MAX_SUBTITLE_CHARS,
|
|
108
|
+
VIDEO_FRAME_RELATION,
|
|
109
|
+
VIDEO_FRAME_SOURCE_TYPE,
|
|
110
|
+
VIDEO_NODE_TYPE,
|
|
111
|
+
VideoFacts,
|
|
112
|
+
find_subtitle,
|
|
113
|
+
parse_subtitles,
|
|
114
|
+
read_video_facts,
|
|
115
|
+
video_frame_dir,
|
|
116
|
+
video_node_id,
|
|
117
|
+
video_quality_score,
|
|
118
|
+
write_video_memory,
|
|
119
|
+
)
|
|
120
|
+
from .video import _write_keyframes as _write_keyframes
|
|
121
|
+
|
|
122
|
+
__all__ = [
|
|
123
|
+
"AUDIO_EXTENSIONS",
|
|
124
|
+
"DEFAULT_KEYFRAMES",
|
|
125
|
+
"FFMPEG_BINARY",
|
|
126
|
+
"IMAGE_CHUNK_CHARS",
|
|
127
|
+
"IMAGE_EXTENSIONS",
|
|
128
|
+
"MAX_INDEX_TEXT_CHARS",
|
|
129
|
+
"MAX_SUBTITLE_CHARS",
|
|
130
|
+
"MAX_THUMBNAIL_CHARS",
|
|
131
|
+
"MODALITY_AUDIO",
|
|
132
|
+
"MODALITY_IMAGE",
|
|
133
|
+
"MODALITY_TEXT",
|
|
134
|
+
"MODALITY_VIDEO",
|
|
135
|
+
"SUBTITLE_EXTENSIONS",
|
|
136
|
+
"SUMMARY_CHARS",
|
|
137
|
+
"THUMBNAIL_EDGE",
|
|
138
|
+
"VIDEO_EXTENSIONS",
|
|
139
|
+
"VIDEO_FRAME_RELATION",
|
|
140
|
+
"VIDEO_FRAME_SOURCE_TYPE",
|
|
141
|
+
"VIDEO_NODE_TYPE",
|
|
142
|
+
"VIDEO_OUT_OF_SCOPE",
|
|
143
|
+
"VIDEO_UNAVAILABLE_DETAIL",
|
|
144
|
+
"AudioFacts",
|
|
145
|
+
"ImageFacts",
|
|
146
|
+
"MultimodalPorts",
|
|
147
|
+
"VideoFacts",
|
|
148
|
+
"audio_quality_score",
|
|
149
|
+
"detect_modality",
|
|
150
|
+
"extract_image_facts",
|
|
151
|
+
"extract_keyframes",
|
|
152
|
+
"ffmpeg_available",
|
|
153
|
+
"find_subtitle",
|
|
154
|
+
"image_node_id",
|
|
155
|
+
"image_quality_score",
|
|
156
|
+
"parse_subtitles",
|
|
157
|
+
"read_video_facts",
|
|
158
|
+
"transcribe_audio",
|
|
159
|
+
"video_frame_dir",
|
|
160
|
+
"video_node_id",
|
|
161
|
+
"video_quality_score",
|
|
162
|
+
"write_image_memory",
|
|
163
|
+
"write_video_memory",
|
|
164
|
+
]
|