ltcai 11.2.0 → 11.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +46 -53
- package/docs/CHANGELOG.md +61 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +1 -1
- package/docs/MULTI_AGENT_RUNTIME.md +1 -1
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -2
- package/docs/PERMISSION_MODE.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/kg-schema.md +2 -2
- package/docs/v11.3.0_PLAN.md +202 -0
- package/docs/v11.4.0_RUST_FOUNDATION_PLAN.md +176 -0
- package/lattice_brain/__init__.py +1 -1
- package/lattice_brain/graph/_kg_common/__init__.py +287 -0
- package/lattice_brain/graph/_kg_common/extraction.py +516 -0
- package/lattice_brain/graph/_kg_common/relations.py +161 -0
- package/lattice_brain/graph/_kg_common/text.py +479 -0
- package/lattice_brain/graph/discovery_index/__init__.py +35 -0
- package/lattice_brain/graph/discovery_index/cleanup.py +182 -0
- package/lattice_brain/graph/discovery_index/extract.py +137 -0
- package/lattice_brain/graph/discovery_index/scan.py +411 -0
- package/lattice_brain/graph/discovery_index/upsert.py +495 -0
- package/lattice_brain/graph/projection/__init__.py +42 -0
- package/lattice_brain/graph/projection/curation.py +500 -0
- package/lattice_brain/graph/{projection.py → projection/v2_schema.py} +15 -477
- package/lattice_brain/graph/retrieval/__init__.py +54 -0
- package/lattice_brain/graph/retrieval/context.py +197 -0
- package/lattice_brain/graph/retrieval/graph_view.py +319 -0
- package/lattice_brain/graph/retrieval/hybrid.py +488 -0
- package/lattice_brain/graph/retrieval/maintenance.py +121 -0
- package/lattice_brain/graph/retrieval/signals.py +95 -0
- package/lattice_brain/graph/retrieval_vector/__init__.py +42 -0
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +97 -0
- package/lattice_brain/graph/retrieval_vector/indexing.py +347 -0
- package/lattice_brain/graph/retrieval_vector/search.py +560 -0
- package/lattice_brain/graph/retrieval_vector/status.py +374 -0
- package/lattice_brain/ingestion/__init__.py +130 -0
- package/lattice_brain/ingestion/_contract.py +90 -0
- package/lattice_brain/ingestion/constants.py +127 -0
- package/lattice_brain/ingestion/folder_scan.py +57 -0
- package/lattice_brain/ingestion/folders.py +258 -0
- package/lattice_brain/ingestion/hashing.py +26 -0
- package/lattice_brain/ingestion/jobs_api.py +107 -0
- package/lattice_brain/ingestion/models.py +80 -0
- package/lattice_brain/ingestion/pipeline.py +486 -0
- package/lattice_brain/ingestion/quality.py +209 -0
- package/lattice_brain/ingestion/routing.py +295 -0
- package/lattice_brain/multimodal/__init__.py +164 -0
- package/lattice_brain/multimodal/audio.py +77 -0
- package/lattice_brain/multimodal/common.py +118 -0
- package/lattice_brain/multimodal/images.py +498 -0
- package/lattice_brain/multimodal/ports.py +169 -0
- package/lattice_brain/multimodal/video.py +410 -0
- package/lattice_brain/portability/__init__.py +90 -0
- package/lattice_brain/portability/_contract.py +42 -0
- package/lattice_brain/portability/backups.py +338 -0
- package/lattice_brain/portability/bundles.py +136 -0
- package/lattice_brain/portability/constants.py +93 -0
- package/lattice_brain/portability/fsops.py +138 -0
- package/lattice_brain/portability/service.py +41 -0
- package/lattice_brain/{portability.py → portability/sharing.py} +44 -677
- package/lattice_brain/runtime/__init__.py +1 -1
- package/lattice_brain/runtime/multi_agent.py +1 -1
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/chronicle.py +63 -0
- package/latticeai/core/agent/__init__.py +93 -0
- package/latticeai/core/agent/_contract.py +79 -0
- package/latticeai/core/agent/context.py +57 -0
- package/latticeai/core/agent/deps.py +125 -0
- package/latticeai/core/agent/execution.py +622 -0
- package/latticeai/core/agent/planning.py +145 -0
- package/latticeai/core/agent/recovery.py +157 -0
- package/latticeai/core/agent/runtime.py +210 -0
- package/latticeai/core/agent/verification.py +231 -0
- package/latticeai/core/embedding_providers/__init__.py +151 -0
- package/latticeai/core/embedding_providers/base.py +199 -0
- package/latticeai/core/embedding_providers/captions.py +162 -0
- package/latticeai/core/embedding_providers/profiles.py +126 -0
- package/latticeai/core/embedding_providers/text.py +350 -0
- package/latticeai/core/embedding_providers/vision.py +352 -0
- package/latticeai/core/file_generation/__init__.py +115 -0
- package/latticeai/core/file_generation/bundles.py +76 -0
- package/latticeai/core/file_generation/extraction.py +154 -0
- package/latticeai/core/file_generation/inference.py +235 -0
- package/latticeai/core/file_generation/orchestration.py +152 -0
- package/latticeai/core/file_generation/prompting.py +117 -0
- package/latticeai/core/file_generation/repair.py +114 -0
- package/latticeai/core/file_generation/sanitize.py +61 -0
- package/latticeai/core/file_generation/validation.py +201 -0
- package/latticeai/core/legacy_compatibility.py +1 -1
- package/latticeai/core/marketplace.py +1 -1
- package/latticeai/core/messages.py +9 -0
- package/latticeai/core/workspace_os_constants.py +1 -1
- package/latticeai/integrations/telegram_bot/__init__.py +123 -0
- package/latticeai/integrations/telegram_bot/__main__.py +17 -0
- package/latticeai/integrations/telegram_bot/config.py +86 -0
- package/latticeai/integrations/telegram_bot/dispatch.py +311 -0
- package/latticeai/integrations/telegram_bot/flows.py +478 -0
- package/latticeai/integrations/telegram_bot/helpers.py +322 -0
- package/latticeai/integrations/telegram_bot/screens.py +394 -0
- package/latticeai/models/router/__init__.py +88 -0
- package/latticeai/models/router/_contract.py +66 -0
- package/latticeai/models/router/branding.py +56 -0
- package/latticeai/models/router/catalog.py +69 -0
- package/latticeai/models/router/documents.py +199 -0
- package/latticeai/models/router/errors.py +37 -0
- package/latticeai/models/router/generation.py +258 -0
- package/latticeai/models/router/loading.py +291 -0
- package/latticeai/models/router/local_models.py +85 -0
- package/latticeai/models/router/registry.py +147 -0
- package/latticeai/runtime/build_phases/__init__.py +82 -0
- package/latticeai/runtime/build_phases/features.py +407 -0
- package/latticeai/runtime/build_phases/foundation.py +555 -0
- package/latticeai/runtime/build_phases/web.py +492 -0
- package/latticeai/runtime/runtime_context.py +1 -0
- package/latticeai/services/architecture_readiness.py +48 -19
- package/latticeai/services/brain_intelligence/__init__.py +58 -0
- package/latticeai/services/brain_intelligence/_contract.py +71 -0
- package/latticeai/services/brain_intelligence/consistency.py +193 -0
- package/latticeai/services/brain_intelligence/constants.py +47 -0
- package/latticeai/services/brain_intelligence/digest.py +258 -0
- package/latticeai/services/brain_intelligence/health.py +331 -0
- package/latticeai/services/brain_intelligence/proposals.py +264 -0
- package/latticeai/services/brain_intelligence/sampling.py +84 -0
- package/latticeai/services/brain_intelligence/service.py +48 -0
- package/latticeai/services/chronicle.py +557 -0
- package/latticeai/services/memory_service/__init__.py +52 -0
- package/latticeai/services/memory_service/_contract.py +100 -0
- package/latticeai/services/memory_service/brief.py +431 -0
- package/latticeai/services/memory_service/constants.py +57 -0
- package/latticeai/services/memory_service/maintenance.py +138 -0
- package/latticeai/services/memory_service/manager.py +186 -0
- package/latticeai/services/memory_service/proof.py +136 -0
- package/latticeai/services/memory_service/recall.py +225 -0
- package/latticeai/services/memory_service/service.py +48 -0
- package/latticeai/services/memory_service/stores.py +110 -0
- package/latticeai/services/model_runtime/__init__.py +322 -0
- package/latticeai/services/model_runtime/cloud.py +87 -0
- package/latticeai/services/model_runtime/download.py +282 -0
- package/latticeai/services/model_runtime/engines.py +341 -0
- package/latticeai/services/model_runtime/loading.py +178 -0
- package/latticeai/services/model_runtime/service.py +129 -0
- package/latticeai/services/model_runtime/state.py +131 -0
- package/latticeai/services/model_runtime/status.py +255 -0
- package/latticeai/services/product_readiness.py +15 -7
- package/latticeai/setup/wizard/__init__.py +126 -0
- package/latticeai/setup/wizard/catalog.py +172 -0
- package/latticeai/setup/wizard/detect.py +323 -0
- package/latticeai/setup/wizard/install.py +348 -0
- package/latticeai/setup/wizard/paths.py +168 -0
- package/latticeai/setup/wizard/plans.py +74 -0
- package/latticeai/setup/wizard/recommend.py +320 -0
- package/package.json +6 -2
- package/scripts/bump_version.py +14 -0
- package/scripts/capture_release_evidence.mjs +33 -21
- package/scripts/check_current_release_docs.mjs +1 -1
- package/scripts/check_i18n_namespace_coverage.mjs +41 -4
- package/scripts/check_max_file_lines.mjs +102 -0
- package/scripts/check_release_evidence_bound.mjs +30 -15
- package/scripts/check_screenshot_pixel_delta.py +34 -4
- package/scripts/check_server_i18n.mjs +1 -0
- package/scripts/generate_rust_parity_fixtures.py +562 -0
- package/scripts/lib/mock_server_fingerprint.mjs +94 -0
- package/scripts/release_screen_claims.json +31 -2
- package/src-tauri/Cargo.lock +361 -3
- package/src-tauri/Cargo.toml +6 -1
- package/src-tauri/src/backend.rs +349 -0
- package/src-tauri/src/folder.rs +33 -0
- package/src-tauri/src/main.rs +97 -399
- package/src-tauri/tauri.conf.json +1 -1
- package/static/app/asset-manifest.json +41 -37
- package/static/app/assets/Act-yYpYnn0v.js +1 -0
- package/static/app/assets/AdminConsole-DL3Cr5pL.js +1 -0
- package/static/app/assets/{Brain-tuhI4sOC.js → Brain-C1HBN0Wf.js} +2 -2
- package/static/app/assets/BrainHome-DoXRhUUC.js +2 -0
- package/static/app/assets/BrainSignals-6yR6ir5t.js +1 -0
- package/static/app/assets/Capture-CFIRsFNE.js +1 -0
- package/static/app/assets/Chronicle-BZbEgiwN.js +1 -0
- package/static/app/assets/CommandPalette-D2pMxC2I.js +1 -0
- package/static/app/assets/Library-DwO3yZST.js +1 -0
- package/static/app/assets/{LivingBrain-DBwhto14.js → LivingBrain-Jn1GK0-S.js} +1 -1
- package/static/app/assets/ProductFlow-B-w1R4Oo.js +1 -0
- package/static/app/assets/ReviewCard-6B27X8Vg.js +3 -0
- package/static/app/assets/System-DW8F-2xL.js +1 -0
- package/static/app/assets/arrow-left-DXvKg9U6.js +1 -0
- package/static/app/assets/{bot-Cia42c2h.js → bot-IM_E_Y12.js} +1 -1
- package/static/app/assets/brain-Ci1CkWjM.js +1 -0
- package/static/app/assets/{button-2j2Ijzgq.js → button-COwyqfHM.js} +1 -1
- package/static/app/assets/circle-check-DfInj-qD.js +1 -0
- package/static/app/assets/{circle-pause-BEFeWpVW.js → circle-pause-DEM4A1Y5.js} +1 -1
- package/static/app/assets/{circle-play-ujXMcHxl.js → circle-play-C9djDuLd.js} +1 -1
- package/static/app/assets/{cpu-k4awryFq.js → cpu-DFdo1gw-.js} +1 -1
- package/static/app/assets/{download-DFbLJ_ig.js → download-SnJL6oqk.js} +1 -1
- package/static/app/assets/{folder-open-7y_b6xkM.js → folder-open-CqZeDkjE.js} +1 -1
- package/static/app/assets/{hard-drive-Bidh02Kr.js → hard-drive-j1jJXYYf.js} +1 -1
- package/static/app/assets/{index-DwDl9-8Y.css → index-BLPb5lmE.css} +1 -1
- package/static/app/assets/index-_u5iUHDr.js +10 -0
- package/static/app/assets/input-B0lPdRQZ.js +1 -0
- package/static/app/assets/link-2-CoFbooHS.js +1 -0
- package/static/app/assets/{permissionCopy-Bpb83Hx9.js → permissionCopy-BsyLxtao.js} +1 -1
- package/static/app/assets/primitives-DEbN-d6p.js +1 -0
- package/static/app/assets/search-BybIWPNd.js +1 -0
- package/static/app/assets/{share-2-BH1M-WNi.js → share-2-CVtZ_ewX.js} +1 -1
- package/static/app/assets/{shield-alert-BlKdBXcG.js → shield-alert-CBi2GNWM.js} +1 -1
- package/static/app/assets/{textarea-CCWbUfFB.js → textarea-DNMpB5ih.js} +1 -1
- package/static/app/assets/{useFocusTrap-YdHQ7pJ1.js → useFocusTrap-C83t3GXF.js} +1 -1
- package/static/app/assets/useMutation-DtbJDoyz.js +1 -0
- package/static/app/assets/{useQuery-CXQiwbVT.js → useQuery-Dcp1OChy.js} +1 -1
- package/static/app/assets/utils-BlZr7Pd4.js +4 -0
- package/static/app/assets/workspace-jJY4RuAV.js +1 -0
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/graph/_kg_common.py +0 -1331
- package/lattice_brain/graph/discovery_index.py +0 -1141
- package/lattice_brain/graph/retrieval.py +0 -1120
- package/lattice_brain/graph/retrieval_vector.py +0 -1293
- package/lattice_brain/ingestion.py +0 -1525
- package/lattice_brain/multimodal.py +0 -1258
- package/latticeai/core/agent.py +0 -1465
- package/latticeai/core/embedding_providers.py +0 -1196
- package/latticeai/core/file_generation.py +0 -1047
- package/latticeai/integrations/telegram_bot.py +0 -1390
- package/latticeai/models/router.py +0 -1007
- package/latticeai/runtime/build_phases.py +0 -1450
- package/latticeai/services/brain_intelligence.py +0 -1083
- package/latticeai/services/memory_service.py +0 -1177
- package/latticeai/services/model_runtime.py +0 -1281
- package/latticeai/setup/wizard.py +0 -1310
- package/static/app/assets/Act-AWf0SAKp.js +0 -1
- package/static/app/assets/AdminConsole-D0u8Tiyj.js +0 -1
- package/static/app/assets/BrainHome-Ts7G_Ila.js +0 -2
- package/static/app/assets/BrainSignals-jMYgQ2Ar.js +0 -1
- package/static/app/assets/Capture-CqOSzyPr.js +0 -1
- package/static/app/assets/CommandPalette-DC0Bzh-I.js +0 -1
- package/static/app/assets/Library-CX-bbhmK.js +0 -1
- package/static/app/assets/ProductFlow-BHA2cfKI.js +0 -1
- package/static/app/assets/ReviewCard-BUhCKRNM.js +0 -3
- package/static/app/assets/System-Bu2t5hn1.js +0 -1
- package/static/app/assets/arrow-left-Dzwa5zRb.js +0 -1
- package/static/app/assets/brain-DJMoqrwx.js +0 -1
- package/static/app/assets/index-BpYkzcVm.js +0 -10
- package/static/app/assets/input-DSlJJxRs.js +0 -1
- package/static/app/assets/primitives-BCx6TvfG.js +0 -1
- package/static/app/assets/search-Cgy8cCFJ.js +0 -1
- package/static/app/assets/utils-zqPZJxdx.js +0 -4
- package/static/app/assets/workspace-DXTihhfU.js +0 -1
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""What routes where, what a folder scan admits, and which gates decide.
|
|
2
|
+
|
|
3
|
+
Source-type sets, folder-scan filters, size budgets and the four feature gates,
|
|
4
|
+
with no logic beyond them. Every other submodule imports this one; this one
|
|
5
|
+
imports nothing from the package, which is what keeps the layering acyclic.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from ..gates import FeatureGate
|
|
11
|
+
from ..multimodal import AUDIO_EXTENSIONS, IMAGE_EXTENSIONS, VIDEO_EXTENSIONS
|
|
12
|
+
|
|
13
|
+
# Source types that arrive as a file on disk (read via ingest_document).
|
|
14
|
+
FILE_SOURCE_TYPES = frozenset({"file", "local_file", "upload", "pdf"})
|
|
15
|
+
# Source types that arrive as extracted text (read via ingest_source).
|
|
16
|
+
TEXT_SOURCE_TYPES = frozenset(
|
|
17
|
+
{"web_url", "browser_tab", "text", "markdown", "note", "code", "clipboard"}
|
|
18
|
+
)
|
|
19
|
+
# Conversational exchanges (read via ingest_message — role/content semantics,
|
|
20
|
+
# conversation chaining). v4: chat and MCP messages stop bypassing the
|
|
21
|
+
# pipeline, so they carry provenance and fire the hook lifecycle like every
|
|
22
|
+
# other source.
|
|
23
|
+
CHAT_SOURCE_TYPES = frozenset({"chat_message", "mcp_message"})
|
|
24
|
+
# Typed memory records (read via ingest_event → Decision/Experience/Event
|
|
25
|
+
# nodes). The Memory System writes through the same door as everything else.
|
|
26
|
+
MEMORY_SOURCE_TYPES = frozenset({"decision", "experience", "workspace_event"})
|
|
27
|
+
_MEMORY_NODE_TYPES = {"decision": "Decision", "experience": "Experience", "workspace_event": "Event"}
|
|
28
|
+
|
|
29
|
+
DEFAULT_MAX_TEXT_BYTES = 5 * 1024 * 1024 # 5 MB of extracted text per item
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# ── Folder ingestion (ingest_folder) filters ─────────────────────────────────
|
|
33
|
+
# Directories that are always pruned regardless of .latticeignore.
|
|
34
|
+
FOLDER_DEFAULT_SKIP_DIRS = frozenset(
|
|
35
|
+
{
|
|
36
|
+
".git",
|
|
37
|
+
"node_modules",
|
|
38
|
+
"__pycache__",
|
|
39
|
+
".venv",
|
|
40
|
+
"venv",
|
|
41
|
+
"env",
|
|
42
|
+
".pytest_cache",
|
|
43
|
+
".mypy_cache",
|
|
44
|
+
".ruff_cache",
|
|
45
|
+
"dist",
|
|
46
|
+
"build",
|
|
47
|
+
".next",
|
|
48
|
+
"target",
|
|
49
|
+
".cache",
|
|
50
|
+
".idea",
|
|
51
|
+
".vscode",
|
|
52
|
+
}
|
|
53
|
+
)
|
|
54
|
+
# Extension filter matching FILE_SOURCE_TYPES conventions: text/markdown/code
|
|
55
|
+
# are read inline (extracted content → chunks); .pdf routes as source_type
|
|
56
|
+
# "pdf" through ingest_document (content extraction is upstream's concern).
|
|
57
|
+
FOLDER_TEXT_EXTENSIONS = frozenset(
|
|
58
|
+
{".txt", ".md", ".markdown", ".rst", ".csv", ".json", ".yaml", ".yml", ".toml", ".ini"}
|
|
59
|
+
)
|
|
60
|
+
FOLDER_CODE_EXTENSIONS = frozenset(
|
|
61
|
+
{
|
|
62
|
+
".py", ".js", ".ts", ".tsx", ".jsx", ".html", ".css", ".go", ".rs",
|
|
63
|
+
".java", ".c", ".h", ".cpp", ".hpp", ".rb", ".php", ".swift", ".kt",
|
|
64
|
+
".sh", ".sql",
|
|
65
|
+
}
|
|
66
|
+
)
|
|
67
|
+
FOLDER_DOCUMENT_EXTENSIONS = frozenset({".pdf"})
|
|
68
|
+
DEFAULT_FOLDER_EXTENSIONS = (
|
|
69
|
+
FOLDER_TEXT_EXTENSIONS | FOLDER_CODE_EXTENSIONS | FOLDER_DOCUMENT_EXTENSIONS
|
|
70
|
+
)
|
|
71
|
+
DEFAULT_MAX_FILE_BYTES = 4_000_000 # matches the local-index text/code budget
|
|
72
|
+
LATTICEIGNORE_FILENAME = ".latticeignore"
|
|
73
|
+
# Opt-out escape hatch for the post-ingest incremental vector sync.
|
|
74
|
+
AUTO_VECTOR_INDEX_ENV = "LATTICEAI_AUTO_VECTOR_INDEX"
|
|
75
|
+
#: Default *on*, unlike every other gate here: new material has always been made
|
|
76
|
+
#: searchable straight away, and this exists so a settings surface can turn that
|
|
77
|
+
#: off (batch reindex later) without a restart. ``FeatureGate`` parses the env
|
|
78
|
+
#: var with the same words the hand-written opt-out check used, so an untouched
|
|
79
|
+
#: install — including one with a nonsense value — answers exactly as before.
|
|
80
|
+
AUTO_VECTOR_INDEX_GATE = FeatureGate(
|
|
81
|
+
AUTO_VECTOR_INDEX_ENV,
|
|
82
|
+
default=True,
|
|
83
|
+
name="auto_vector_index",
|
|
84
|
+
detail="New material is prepared for semantic search as soon as it lands.",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
# ── Multi-modal ingestion (v11.1.0 Track 3) ──────────────────────────────────
|
|
88
|
+
# Opt-in, default off, on purpose. Turning it on changes what a folder scan
|
|
89
|
+
# *stores* (pictures and recordings, with OCR and — if a model is loaded —
|
|
90
|
+
# captions and vectors), and that is the user's call, not a default. With the
|
|
91
|
+
# flag off every routing decision below is skipped and behaviour is byte-for-
|
|
92
|
+
# byte what it was before this release.
|
|
93
|
+
ALLOW_MULTIMODAL_ENV = "LATTICEAI_ALLOW_MULTIMODAL"
|
|
94
|
+
#: The multi-modal switch, resolved when it is asked rather than frozen into
|
|
95
|
+
#: ``self`` at construction (v11.2.0). The environment variable is still the
|
|
96
|
+
#: answer for an untouched install — same var, same words, same default off —
|
|
97
|
+
#: but a settings surface can bind a resolver and move it without a restart.
|
|
98
|
+
MULTIMODAL_GATE = FeatureGate(
|
|
99
|
+
ALLOW_MULTIMODAL_ENV,
|
|
100
|
+
default=False,
|
|
101
|
+
name="allow_multimodal",
|
|
102
|
+
detail="Pictures and recordings are only ingested when this is turned on.",
|
|
103
|
+
)
|
|
104
|
+
#: Video is a *sub-switch* of the one above: with multi-modal off nothing about
|
|
105
|
+
#: video happens at all, and with it on video is included unless this is
|
|
106
|
+
#: explicitly turned off. The effective default is therefore still "no video",
|
|
107
|
+
#: and the seam exists so a settings screen can offer pictures without films.
|
|
108
|
+
ALLOW_VIDEO_ENV = "LATTICEAI_ALLOW_VIDEO"
|
|
109
|
+
VIDEO_GATE = FeatureGate(
|
|
110
|
+
ALLOW_VIDEO_ENV,
|
|
111
|
+
default=True,
|
|
112
|
+
name="allow_video",
|
|
113
|
+
detail="Videos are ingested as keyframes plus subtitles when multi-modal is on.",
|
|
114
|
+
)
|
|
115
|
+
#: Source types that name a modality outright (a caller who already knows).
|
|
116
|
+
IMAGE_SOURCE_TYPES = frozenset({"image", "screenshot", "photo"})
|
|
117
|
+
AUDIO_SOURCE_TYPES = frozenset({"audio", "voice_memo", "recording"})
|
|
118
|
+
VIDEO_SOURCE_TYPES = frozenset({"video", "screen_recording", "movie"})
|
|
119
|
+
#: Added to the folder-scan allow-list only while multimodal is enabled.
|
|
120
|
+
FOLDER_MULTIMODAL_EXTENSIONS = IMAGE_EXTENSIONS | AUDIO_EXTENSIONS
|
|
121
|
+
#: Videos join the folder allow-list only when this machine can decode one —
|
|
122
|
+
#: scanning a folder into a pile of refusals is not a feature.
|
|
123
|
+
FOLDER_VIDEO_EXTENSIONS = VIDEO_EXTENSIONS
|
|
124
|
+
#: Graph node type for a recording. ``NodeType.AUDIO`` normalizes this on the
|
|
125
|
+
#: KG v2 write side; the legacy tables keep the label verbatim, which is what
|
|
126
|
+
#: every type-aware read (graph view, context sections, doc-gen) matches on.
|
|
127
|
+
AUDIO_NODE_TYPE = "Audio"
|
|
@@ -0,0 +1,57 @@
|
|
|
1
|
+
"""``.latticeignore`` parsing and matching for the folder walk.
|
|
2
|
+
|
|
3
|
+
A gitignore-like subset: blank lines and ``#`` comments are dropped, patterns
|
|
4
|
+
are ``fnmatch`` globs, and a trailing ``/`` restricts a pattern to directories.
|
|
5
|
+
Patterns match against both the root-relative posix path and the basename, so
|
|
6
|
+
``*.log`` and ``docs/draft.md`` both behave the way a reader expects.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import fnmatch
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from typing import Iterable, List
|
|
14
|
+
|
|
15
|
+
from .constants import LATTICEIGNORE_FILENAME
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _load_latticeignore(root: Path) -> List[str]:
|
|
19
|
+
"""Parse ``root/.latticeignore`` → glob patterns (gitignore-like subset)."""
|
|
20
|
+
ignore_file = root / LATTICEIGNORE_FILENAME
|
|
21
|
+
patterns: List[str] = []
|
|
22
|
+
if not ignore_file.is_file():
|
|
23
|
+
return patterns
|
|
24
|
+
try:
|
|
25
|
+
lines = ignore_file.read_text(encoding="utf-8", errors="ignore").splitlines()
|
|
26
|
+
except OSError:
|
|
27
|
+
return patterns
|
|
28
|
+
for raw in lines:
|
|
29
|
+
line = raw.strip()
|
|
30
|
+
if not line or line.startswith("#"):
|
|
31
|
+
continue
|
|
32
|
+
patterns.append(line)
|
|
33
|
+
return patterns
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _matches_ignore(
|
|
37
|
+
rel_posix: str, name: str, *, is_dir: bool, patterns: Iterable[str]
|
|
38
|
+
) -> bool:
|
|
39
|
+
"""fnmatch-based .latticeignore matching.
|
|
40
|
+
|
|
41
|
+
- ``pattern/`` matches directories only (files under it never appear
|
|
42
|
+
because ignored directories are pruned during the walk).
|
|
43
|
+
- Patterns match against both the root-relative posix path and the
|
|
44
|
+
basename, so ``*.log`` and ``docs/draft.md`` both behave as expected.
|
|
45
|
+
"""
|
|
46
|
+
for raw in patterns:
|
|
47
|
+
pattern = raw
|
|
48
|
+
if pattern.endswith("/"):
|
|
49
|
+
if not is_dir:
|
|
50
|
+
continue
|
|
51
|
+
pattern = pattern.rstrip("/")
|
|
52
|
+
pattern = pattern.lstrip("/")
|
|
53
|
+
if not pattern:
|
|
54
|
+
continue
|
|
55
|
+
if fnmatch.fnmatch(rel_posix, pattern) or fnmatch.fnmatch(name, pattern):
|
|
56
|
+
return True
|
|
57
|
+
return False
|
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
"""Walking a folder, and the one-page web hand-off, into the standard door.
|
|
2
|
+
|
|
3
|
+
Neither method is a second ingest path: both build ordinary ``IngestionItem``
|
|
4
|
+
values and hand them to ``IngestionPipeline.ingest``. The folder walk owns the
|
|
5
|
+
filtering order (hard skip-list → hidden → ``.latticeignore`` → extension →
|
|
6
|
+
size) and the choice between ingesting inline and scheduling in the background;
|
|
7
|
+
the web hand-off owns the refusal to fetch or parse anything itself.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import os
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, Dict, Iterable, List, Optional
|
|
15
|
+
|
|
16
|
+
from ._contract import IngestionCore as _Core
|
|
17
|
+
from .constants import (
|
|
18
|
+
DEFAULT_FOLDER_EXTENSIONS,
|
|
19
|
+
DEFAULT_MAX_FILE_BYTES,
|
|
20
|
+
FOLDER_DEFAULT_SKIP_DIRS,
|
|
21
|
+
FOLDER_DOCUMENT_EXTENSIONS,
|
|
22
|
+
FOLDER_MULTIMODAL_EXTENSIONS,
|
|
23
|
+
FOLDER_VIDEO_EXTENSIONS,
|
|
24
|
+
LATTICEIGNORE_FILENAME,
|
|
25
|
+
)
|
|
26
|
+
from .folder_scan import _load_latticeignore, _matches_ignore
|
|
27
|
+
from .models import IngestionItem, IngestionResult
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class IngestionFolderMixin(_Core):
|
|
31
|
+
"""Folder walk + web hand-off. Mixed into ``IngestionPipeline``."""
|
|
32
|
+
|
|
33
|
+
def ingest_web_page(
|
|
34
|
+
self,
|
|
35
|
+
url: str,
|
|
36
|
+
extracted_text: str,
|
|
37
|
+
*,
|
|
38
|
+
title: Optional[str] = None,
|
|
39
|
+
metadata: Optional[Dict[str, Any]] = None,
|
|
40
|
+
owner: Optional[str] = None,
|
|
41
|
+
workspace_id: Optional[str] = None,
|
|
42
|
+
captured_at: Optional[str] = None,
|
|
43
|
+
user_email: Optional[str] = None,
|
|
44
|
+
) -> IngestionResult:
|
|
45
|
+
"""Ingest an *already-extracted* web page (see module docstring seam).
|
|
46
|
+
|
|
47
|
+
Fetching/parsing is upstream's responsibility (browser extension /
|
|
48
|
+
tools layer); this wrapper only normalizes ``(url, extracted_text)``
|
|
49
|
+
into an ``IngestionItem(source_type="web_url")`` and routes it through
|
|
50
|
+
the standard :meth:`ingest` door.
|
|
51
|
+
"""
|
|
52
|
+
url = str(url or "").strip()
|
|
53
|
+
if not url:
|
|
54
|
+
return IngestionResult(
|
|
55
|
+
status="failed", source_type="web_url",
|
|
56
|
+
indexing_status="skipped", detail="url required",
|
|
57
|
+
)
|
|
58
|
+
text = str(extracted_text or "")
|
|
59
|
+
if not text.strip():
|
|
60
|
+
return IngestionResult(
|
|
61
|
+
status="failed", source_type="web_url",
|
|
62
|
+
indexing_status="skipped",
|
|
63
|
+
detail=(
|
|
64
|
+
"extracted_text required — the graph layer does not fetch or "
|
|
65
|
+
"parse the web; extraction happens upstream."
|
|
66
|
+
),
|
|
67
|
+
)
|
|
68
|
+
item = IngestionItem(
|
|
69
|
+
source_type="web_url",
|
|
70
|
+
title=title or url,
|
|
71
|
+
text=text,
|
|
72
|
+
source_uri=url,
|
|
73
|
+
owner=owner,
|
|
74
|
+
workspace_id=workspace_id,
|
|
75
|
+
captured_at=captured_at,
|
|
76
|
+
metadata=dict(metadata or {}),
|
|
77
|
+
)
|
|
78
|
+
return self.ingest(item, user_email=user_email or owner)
|
|
79
|
+
|
|
80
|
+
def ingest_folder(
|
|
81
|
+
self,
|
|
82
|
+
root_path: Any,
|
|
83
|
+
*,
|
|
84
|
+
recursive: bool = True,
|
|
85
|
+
background: bool = False,
|
|
86
|
+
extensions: Optional[Iterable[str]] = None,
|
|
87
|
+
max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
|
|
88
|
+
include_hidden: bool = False,
|
|
89
|
+
max_files: int = 1000,
|
|
90
|
+
max_errors: int = 25,
|
|
91
|
+
owner: Optional[str] = None,
|
|
92
|
+
workspace_id: Optional[str] = None,
|
|
93
|
+
user_email: Optional[str] = None,
|
|
94
|
+
) -> Dict[str, Any]:
|
|
95
|
+
"""Walk ``root_path`` and ingest every eligible file through the pipeline.
|
|
96
|
+
|
|
97
|
+
Filtering, in order: hard skip-list directories (``.git`` …), hidden
|
|
98
|
+
entries (unless ``include_hidden``), root ``.latticeignore`` patterns
|
|
99
|
+
(fnmatch globs; ``dir/`` suffix prunes directories), extension
|
|
100
|
+
allow-list, then ``max_file_bytes``. Text/code files are read inline so
|
|
101
|
+
their content is chunked; ``.pdf`` routes through the file door without
|
|
102
|
+
inline extraction.
|
|
103
|
+
|
|
104
|
+
``background=True`` schedules the built items on the existing
|
|
105
|
+
:class:`BackgroundIngestionQueue` instead of ingesting inline.
|
|
106
|
+
Returns a summary dict with counts and per-file errors (capped at
|
|
107
|
+
``max_errors``).
|
|
108
|
+
"""
|
|
109
|
+
summary: Dict[str, Any] = {
|
|
110
|
+
"root": str(root_path),
|
|
111
|
+
"recursive": bool(recursive),
|
|
112
|
+
"background": bool(background),
|
|
113
|
+
"scanned": 0,
|
|
114
|
+
"matched": 0,
|
|
115
|
+
"ingested": 0,
|
|
116
|
+
"duplicate": 0,
|
|
117
|
+
"failed": 0,
|
|
118
|
+
"skipped": {"ignored": 0, "extension": 0, "too_large": 0, "hidden": 0},
|
|
119
|
+
"truncated": False,
|
|
120
|
+
"errors": [],
|
|
121
|
+
}
|
|
122
|
+
try:
|
|
123
|
+
root = Path(root_path).expanduser()
|
|
124
|
+
except TypeError:
|
|
125
|
+
summary.update(status="failed", detail=f"invalid root path: {root_path!r}")
|
|
126
|
+
return summary
|
|
127
|
+
if not root.is_dir():
|
|
128
|
+
summary.update(status="failed", detail=f"not a directory: {root}")
|
|
129
|
+
return summary
|
|
130
|
+
if not self.available():
|
|
131
|
+
summary.update(
|
|
132
|
+
status="unavailable",
|
|
133
|
+
detail="Knowledge Graph is disabled (LATTICEAI_ENABLE_GRAPH).",
|
|
134
|
+
)
|
|
135
|
+
return summary
|
|
136
|
+
summary["root"] = str(root)
|
|
137
|
+
max_files = max(1, int(max_files))
|
|
138
|
+
max_errors = max(0, int(max_errors))
|
|
139
|
+
max_file_bytes = max(1, int(max_file_bytes))
|
|
140
|
+
allowed_exts = (
|
|
141
|
+
frozenset(str(e).lower() if str(e).startswith(".") else f".{str(e).lower()}" for e in extensions)
|
|
142
|
+
if extensions
|
|
143
|
+
else self._folder_extensions()
|
|
144
|
+
)
|
|
145
|
+
patterns = _load_latticeignore(root)
|
|
146
|
+
errors: List[Dict[str, Any]] = summary["errors"]
|
|
147
|
+
skipped = summary["skipped"]
|
|
148
|
+
items: List[IngestionItem] = []
|
|
149
|
+
|
|
150
|
+
def _record_error(path: Path, detail: str, status: str = "failed") -> None:
|
|
151
|
+
summary["failed"] += 1
|
|
152
|
+
if len(errors) < max_errors:
|
|
153
|
+
errors.append({"path": str(path), "status": status, "detail": detail})
|
|
154
|
+
|
|
155
|
+
for dirpath, dirnames, filenames in os.walk(root):
|
|
156
|
+
current = Path(dirpath)
|
|
157
|
+
rel_dir = current.relative_to(root)
|
|
158
|
+
kept_dirs: List[str] = []
|
|
159
|
+
for name in sorted(dirnames):
|
|
160
|
+
if name in FOLDER_DEFAULT_SKIP_DIRS:
|
|
161
|
+
continue
|
|
162
|
+
if name.startswith(".") and not include_hidden:
|
|
163
|
+
continue
|
|
164
|
+
rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
|
|
165
|
+
if _matches_ignore(rel, name, is_dir=True, patterns=patterns):
|
|
166
|
+
skipped["ignored"] += 1
|
|
167
|
+
continue
|
|
168
|
+
kept_dirs.append(name)
|
|
169
|
+
dirnames[:] = kept_dirs if recursive else []
|
|
170
|
+
|
|
171
|
+
for name in sorted(filenames):
|
|
172
|
+
if name == LATTICEIGNORE_FILENAME:
|
|
173
|
+
continue
|
|
174
|
+
summary["scanned"] += 1
|
|
175
|
+
path = current / name
|
|
176
|
+
rel = name if str(rel_dir) == "." else (rel_dir / name).as_posix()
|
|
177
|
+
if name.startswith(".") and not include_hidden:
|
|
178
|
+
skipped["hidden"] += 1
|
|
179
|
+
continue
|
|
180
|
+
if _matches_ignore(rel, name, is_dir=False, patterns=patterns):
|
|
181
|
+
skipped["ignored"] += 1
|
|
182
|
+
continue
|
|
183
|
+
ext = path.suffix.lower()
|
|
184
|
+
if ext not in allowed_exts:
|
|
185
|
+
skipped["extension"] += 1
|
|
186
|
+
continue
|
|
187
|
+
try:
|
|
188
|
+
size = path.stat().st_size
|
|
189
|
+
except OSError as exc:
|
|
190
|
+
_record_error(path, f"stat failed: {exc}")
|
|
191
|
+
continue
|
|
192
|
+
if size > max_file_bytes:
|
|
193
|
+
skipped["too_large"] += 1
|
|
194
|
+
continue
|
|
195
|
+
if len(items) >= max_files:
|
|
196
|
+
summary["truncated"] = True
|
|
197
|
+
break
|
|
198
|
+
item_metadata: Dict[str, Any] = {"relative_path": rel}
|
|
199
|
+
if ext in (FOLDER_MULTIMODAL_EXTENSIONS | FOLDER_VIDEO_EXTENSIONS) and self._allow_multimodal:
|
|
200
|
+
# Routed by modality inside ``ingest``; reading the bytes as
|
|
201
|
+
# UTF-8 here would only produce mojibake.
|
|
202
|
+
source_type = "file"
|
|
203
|
+
elif ext in FOLDER_DOCUMENT_EXTENSIONS:
|
|
204
|
+
source_type = "pdf"
|
|
205
|
+
else:
|
|
206
|
+
source_type = "file"
|
|
207
|
+
try:
|
|
208
|
+
content = path.read_text(encoding="utf-8", errors="ignore")
|
|
209
|
+
except OSError as exc:
|
|
210
|
+
_record_error(path, f"read failed: {exc}")
|
|
211
|
+
continue
|
|
212
|
+
item_metadata["extracted"] = {"content": content, "chars": len(content)}
|
|
213
|
+
items.append(
|
|
214
|
+
IngestionItem(
|
|
215
|
+
source_type=source_type,
|
|
216
|
+
title=name,
|
|
217
|
+
path=str(path),
|
|
218
|
+
source_uri=str(path),
|
|
219
|
+
owner=owner,
|
|
220
|
+
workspace_id=workspace_id,
|
|
221
|
+
metadata=item_metadata,
|
|
222
|
+
)
|
|
223
|
+
)
|
|
224
|
+
if summary["truncated"]:
|
|
225
|
+
break
|
|
226
|
+
|
|
227
|
+
summary["matched"] = len(items)
|
|
228
|
+
if background:
|
|
229
|
+
job = self.schedule_background(
|
|
230
|
+
items, incremental=True, user_email=user_email or owner,
|
|
231
|
+
)
|
|
232
|
+
summary.update(status="scheduled", job_id=job.job_id, scheduled=len(items))
|
|
233
|
+
return summary
|
|
234
|
+
|
|
235
|
+
for item in items:
|
|
236
|
+
result = self.ingest(item, user_email=user_email or owner)
|
|
237
|
+
if result.status == "ok":
|
|
238
|
+
if result.duplicate:
|
|
239
|
+
summary["duplicate"] += 1
|
|
240
|
+
else:
|
|
241
|
+
summary["ingested"] += 1
|
|
242
|
+
else:
|
|
243
|
+
_record_error(Path(item.path or ""), result.detail or result.status, result.status)
|
|
244
|
+
summary["status"] = "ok" if summary["failed"] == 0 else "partial"
|
|
245
|
+
return summary
|
|
246
|
+
|
|
247
|
+
def _folder_extensions(self) -> frozenset:
|
|
248
|
+
"""Folder-scan allow-list — pictures, recordings and films when enabled.
|
|
249
|
+
|
|
250
|
+
Video joins only when this machine can actually decode one, so a scan
|
|
251
|
+
never fills the error list with files it was always going to refuse.
|
|
252
|
+
"""
|
|
253
|
+
if not self._allow_multimodal:
|
|
254
|
+
return DEFAULT_FOLDER_EXTENSIONS
|
|
255
|
+
allowed = DEFAULT_FOLDER_EXTENSIONS | FOLDER_MULTIMODAL_EXTENSIONS
|
|
256
|
+
if self._allow_video:
|
|
257
|
+
return allowed | FOLDER_VIDEO_EXTENSIONS
|
|
258
|
+
return allowed
|
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""The two content hashes the pipeline names things by.
|
|
2
|
+
|
|
3
|
+
:func:`content_hash_text` matches the store's own hashing scheme, so a text
|
|
4
|
+
payload hashed here and a text payload hashed there dedupe against each other.
|
|
5
|
+
:func:`_file_digest` streams a file instead of reading it whole — it is the key
|
|
6
|
+
a video's keyframe folder is named by, and videos are large.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import hashlib
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def content_hash_text(text: str) -> str:
|
|
16
|
+
"""Canonical content hash for a text payload (matches store hashing scheme)."""
|
|
17
|
+
return hashlib.sha256((text or "").encode("utf-8", "ignore")).hexdigest()
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _file_digest(path: Path) -> str:
|
|
21
|
+
"""Streaming sha256 of a file — the key a video's frame folder is named by."""
|
|
22
|
+
digest = hashlib.sha256()
|
|
23
|
+
with path.open("rb") as handle:
|
|
24
|
+
for block in iter(lambda: handle.read(1024 * 1024), b""):
|
|
25
|
+
digest.update(block)
|
|
26
|
+
return digest.hexdigest()
|
|
@@ -0,0 +1,107 @@
|
|
|
1
|
+
"""Scheduling many items, and running (or resuming) what was scheduled.
|
|
2
|
+
|
|
3
|
+
The pipeline owns "ingest one item"; ``BackgroundIngestionQueue`` owns
|
|
4
|
+
"schedule many and report progress". This mixin is the seam between them:
|
|
5
|
+
per-item errors are recorded and never abort a job, progress is checkpointed
|
|
6
|
+
after every item, and the same method powers both the first run and a resume
|
|
7
|
+
because already-completed items are simply skipped.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
from typing import Any, Dict, List, Optional
|
|
13
|
+
|
|
14
|
+
from ..ingestion_jobs import BackgroundIngestionJob
|
|
15
|
+
from ._contract import IngestionCore as _Core
|
|
16
|
+
from .models import IngestionItem
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class IngestionJobsMixin(_Core):
|
|
20
|
+
"""Background job scheduling. Mixed into ``IngestionPipeline``."""
|
|
21
|
+
|
|
22
|
+
# --- Large candidate #1: background / incremental scheduling (slice) ---
|
|
23
|
+
def schedule_background(
|
|
24
|
+
self,
|
|
25
|
+
items: List[IngestionItem],
|
|
26
|
+
*,
|
|
27
|
+
incremental: bool = True,
|
|
28
|
+
user_email: Optional[str] = None,
|
|
29
|
+
) -> BackgroundIngestionJob:
|
|
30
|
+
"""Schedule items for background incremental indexing.
|
|
31
|
+
|
|
32
|
+
Returns a job handle. Actual execution can be driven by caller
|
|
33
|
+
(or future worker) calling pipeline.ingest on each — or through
|
|
34
|
+
:meth:`run_background_job`. This seam enables large-corpus scale
|
|
35
|
+
without blocking user requests.
|
|
36
|
+
"""
|
|
37
|
+
job = self._bg_queue.schedule(items, incremental=incremental, user_email=user_email)
|
|
38
|
+
# mark initial status on results concept (jobs track)
|
|
39
|
+
return job
|
|
40
|
+
|
|
41
|
+
def get_background_job(self, job_id: str) -> Optional[BackgroundIngestionJob]:
|
|
42
|
+
return self._bg_queue.get(job_id)
|
|
43
|
+
|
|
44
|
+
def list_background_jobs(self, limit: int = 20) -> List[Dict[str, Any]]:
|
|
45
|
+
"""Recent jobs (newest first) in the frozen ``/api/ingestion`` schema."""
|
|
46
|
+
return [job.as_dict() for job in self._bg_queue.list_recent(limit=limit)]
|
|
47
|
+
|
|
48
|
+
def run_background_job(
|
|
49
|
+
self, job_id: str, *, user_email: Optional[str] = None
|
|
50
|
+
) -> Dict[str, Any]:
|
|
51
|
+
"""Execute a queued/interrupted job's remaining items.
|
|
52
|
+
|
|
53
|
+
Per-item errors are recorded (capped) and never abort the job. The
|
|
54
|
+
final status is ``completed`` (all done), ``partial`` (some done),
|
|
55
|
+
or ``failed`` (nothing done). Already-completed items are skipped, so
|
|
56
|
+
the same method safely powers both first-run and resume.
|
|
57
|
+
"""
|
|
58
|
+
job = self._bg_queue.get(job_id)
|
|
59
|
+
if job is None:
|
|
60
|
+
return {"status": "not_found", "job_id": job_id}
|
|
61
|
+
if job.status == "running":
|
|
62
|
+
return job.as_dict()
|
|
63
|
+
return self._execute_background_job(job, user_email=user_email)
|
|
64
|
+
|
|
65
|
+
def resume_background_job(
|
|
66
|
+
self, job_id: str, *, user_email: Optional[str] = None
|
|
67
|
+
) -> Dict[str, Any]:
|
|
68
|
+
"""Resume an interrupted/partial/failed job from its remaining items."""
|
|
69
|
+
return self.run_background_job(job_id, user_email=user_email)
|
|
70
|
+
|
|
71
|
+
def _execute_background_job(
|
|
72
|
+
self, job: BackgroundIngestionJob, *, user_email: Optional[str] = None
|
|
73
|
+
) -> Dict[str, Any]:
|
|
74
|
+
job.status = "running"
|
|
75
|
+
# Retried items get a fresh verdict: reset failure state for this run.
|
|
76
|
+
job.failed = 0
|
|
77
|
+
job.errors = []
|
|
78
|
+
job.touch()
|
|
79
|
+
self._bg_queue.save(job)
|
|
80
|
+
runner_email = user_email or job.user_email
|
|
81
|
+
for index in job.remaining_indices():
|
|
82
|
+
item = job.items[index]
|
|
83
|
+
try:
|
|
84
|
+
result = self.ingest(item, user_email=runner_email or item.owner)
|
|
85
|
+
status, detail = result.status, result.detail
|
|
86
|
+
except Exception as exc: # noqa: BLE001 — per-item isolation: keep going
|
|
87
|
+
status, detail = "failed", str(exc)
|
|
88
|
+
if status == "ok":
|
|
89
|
+
job.done_indices.add(index)
|
|
90
|
+
else:
|
|
91
|
+
job.record_error(index, item, detail or status)
|
|
92
|
+
job.processed = len(job.done_indices)
|
|
93
|
+
job.touch()
|
|
94
|
+
# Checkpoint per item: a crash here must cost at most the item in
|
|
95
|
+
# flight, never the whole job's progress. One small UPDATE against
|
|
96
|
+
# an ingest (parse + chunk + embed) is noise.
|
|
97
|
+
self._bg_queue.save(job)
|
|
98
|
+
job.processed = len(job.done_indices)
|
|
99
|
+
if job.total == 0 or job.processed >= job.total:
|
|
100
|
+
job.status = "completed"
|
|
101
|
+
elif job.processed > 0:
|
|
102
|
+
job.status = "partial"
|
|
103
|
+
else:
|
|
104
|
+
job.status = "failed"
|
|
105
|
+
job.touch()
|
|
106
|
+
self._bg_queue.save(job)
|
|
107
|
+
return job.as_dict()
|
|
@@ -0,0 +1,80 @@
|
|
|
1
|
+
"""The two records the pipeline speaks in: one item in, one result out.
|
|
2
|
+
|
|
3
|
+
:class:`IngestionItem` is what every source normalizes to before the pipeline
|
|
4
|
+
sees it; :class:`IngestionResult` is what every source normalizes to after. The
|
|
5
|
+
result's ``as_dict`` is the frozen ``/api/ingestion`` payload shape — additive
|
|
6
|
+
keys appear only when populated, so pre-v9.8 consumers see what they always did.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass, field
|
|
12
|
+
from typing import Any, Dict, List, Optional
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass
|
|
16
|
+
class IngestionItem:
|
|
17
|
+
"""A single thing to ingest, normalized across every source type."""
|
|
18
|
+
|
|
19
|
+
source_type: str
|
|
20
|
+
title: Optional[str] = None
|
|
21
|
+
text: Optional[str] = None # text/web sources
|
|
22
|
+
path: Optional[str] = None # file sources
|
|
23
|
+
source_uri: Optional[str] = None
|
|
24
|
+
mime_type: Optional[str] = None
|
|
25
|
+
owner: Optional[str] = None
|
|
26
|
+
workspace_id: Optional[str] = None
|
|
27
|
+
permissions: Optional[Dict[str, Any]] = None
|
|
28
|
+
captured_at: Optional[str] = None
|
|
29
|
+
modified_at: Optional[str] = None
|
|
30
|
+
conversation_id: Optional[str] = None
|
|
31
|
+
agent_used: Optional[str] = None
|
|
32
|
+
metadata: Dict[str, Any] = field(default_factory=dict)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
@dataclass
|
|
36
|
+
class IngestionResult:
|
|
37
|
+
"""The outcome of one ingestion, including provenance and idempotency."""
|
|
38
|
+
|
|
39
|
+
status: str # ok | unavailable | blocked | failed
|
|
40
|
+
source_type: str
|
|
41
|
+
node_id: Optional[str] = None
|
|
42
|
+
source_node_id: Optional[str] = None
|
|
43
|
+
content_hash: Optional[str] = None
|
|
44
|
+
title: Optional[str] = None
|
|
45
|
+
chunk_ids: List[str] = field(default_factory=list)
|
|
46
|
+
chunk_count: int = 0
|
|
47
|
+
duplicate: bool = False
|
|
48
|
+
embedded: bool = False
|
|
49
|
+
indexing_status: str = "pending" # indexed | skipped | failed | pending
|
|
50
|
+
provenance_id: Optional[str] = None
|
|
51
|
+
detail: Optional[str] = None
|
|
52
|
+
# v9.8.0 additive quality fields — advisory only, never gate behavior.
|
|
53
|
+
extraction_quality: Optional[Dict[str, Any]] = None
|
|
54
|
+
warnings: List[str] = field(default_factory=list)
|
|
55
|
+
quality_gate: Optional[Dict[str, Any]] = None
|
|
56
|
+
|
|
57
|
+
def as_dict(self) -> Dict[str, Any]:
|
|
58
|
+
payload: Dict[str, Any] = {
|
|
59
|
+
"status": self.status,
|
|
60
|
+
"source_type": self.source_type,
|
|
61
|
+
"node_id": self.node_id,
|
|
62
|
+
"source_node_id": self.source_node_id,
|
|
63
|
+
"content_hash": self.content_hash,
|
|
64
|
+
"title": self.title,
|
|
65
|
+
"chunk_ids": self.chunk_ids,
|
|
66
|
+
"chunk_count": self.chunk_count,
|
|
67
|
+
"duplicate": self.duplicate,
|
|
68
|
+
"embedded": self.embedded,
|
|
69
|
+
"indexing_status": self.indexing_status,
|
|
70
|
+
"provenance_id": self.provenance_id,
|
|
71
|
+
"detail": self.detail,
|
|
72
|
+
}
|
|
73
|
+
# Additive keys only when populated so pre-v9.8 payloads are unchanged.
|
|
74
|
+
if self.extraction_quality is not None:
|
|
75
|
+
payload["extraction_quality"] = self.extraction_quality
|
|
76
|
+
if self.warnings:
|
|
77
|
+
payload["warnings"] = list(self.warnings)
|
|
78
|
+
if self.quality_gate is not None:
|
|
79
|
+
payload["quality_gate"] = self.quality_gate
|
|
80
|
+
return payload
|