ltcai 11.0.1 → 11.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (140) hide show
  1. package/README.md +55 -43
  2. package/docs/CHANGELOG.md +61 -0
  3. package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
  4. package/docs/DEVELOPMENT.md +1 -1
  5. package/docs/FEATURE_AUDIT_v11.2.0.md +393 -0
  6. package/docs/LAYOUT_REBUILD_SPEC.md +9 -1
  7. package/docs/ONBOARDING.md +1 -1
  8. package/docs/OPERATIONS.md +1 -1
  9. package/docs/PERFORMANCE.md +71 -18
  10. package/docs/TRUST_MODEL.md +1 -1
  11. package/docs/WHY_LATTICE.md +1 -1
  12. package/docs/architecture.md +6 -2
  13. package/docs/kg-schema.md +1 -1
  14. package/lattice_brain/__init__.py +1 -1
  15. package/lattice_brain/embeddings.py +12 -37
  16. package/lattice_brain/gates.py +125 -0
  17. package/lattice_brain/graph/discovery_index.py +30 -32
  18. package/lattice_brain/graph/fusion.py +35 -4
  19. package/lattice_brain/graph/image_vectors.py +230 -0
  20. package/lattice_brain/graph/ingest.py +11 -5
  21. package/lattice_brain/graph/projection.py +66 -8
  22. package/lattice_brain/graph/provenance.py +27 -2
  23. package/lattice_brain/graph/retrieval.py +113 -2
  24. package/lattice_brain/graph/retrieval_docgen.py +6 -6
  25. package/lattice_brain/graph/schema.py +18 -0
  26. package/lattice_brain/graph/store.py +9 -0
  27. package/lattice_brain/graph/vector_index/selector.py +32 -2
  28. package/lattice_brain/ingestion.py +363 -10
  29. package/lattice_brain/multimodal.py +1258 -0
  30. package/lattice_brain/portability.py +169 -32
  31. package/lattice_brain/runtime/multi_agent.py +1 -1
  32. package/lattice_brain/sealed_box.py +244 -0
  33. package/lattice_brain/self_model.py +77 -22
  34. package/lattice_brain/synthesis.py +24 -1
  35. package/latticeai/__init__.py +1 -1
  36. package/latticeai/api/brain_intelligence.py +4 -0
  37. package/latticeai/api/chat.py +11 -0
  38. package/latticeai/api/chat_helpers.py +16 -3
  39. package/latticeai/api/chat_hybrid.py +32 -1
  40. package/latticeai/api/features.py +70 -0
  41. package/latticeai/api/local_files.py +102 -0
  42. package/latticeai/api/memory.py +128 -1
  43. package/latticeai/api/portability.py +39 -4
  44. package/latticeai/api/review_queue.py +126 -0
  45. package/latticeai/api/search.py +16 -2
  46. package/latticeai/core/agent.py +59 -2
  47. package/latticeai/core/agent_prompts.py +66 -0
  48. package/latticeai/core/config.py +4 -1
  49. package/latticeai/core/context_builder.py +98 -11
  50. package/latticeai/core/embedding_providers.py +528 -0
  51. package/latticeai/core/legacy_compatibility.py +1 -1
  52. package/latticeai/core/marketplace.py +1 -1
  53. package/latticeai/core/messages.py +180 -0
  54. package/latticeai/core/model_compat.py +73 -2
  55. package/latticeai/core/workspace_os.py +43 -0
  56. package/latticeai/core/workspace_os_constants.py +1 -1
  57. package/latticeai/core/workspace_reorganization.py +335 -0
  58. package/latticeai/models/model_providers.py +12 -4
  59. package/latticeai/runtime/build_phases.py +33 -2
  60. package/latticeai/runtime/chat_wiring.py +4 -0
  61. package/latticeai/runtime/feature_toggle_wiring.py +163 -0
  62. package/latticeai/runtime/persistence_runtime.py +41 -4
  63. package/latticeai/runtime/router_registration.py +11 -0
  64. package/latticeai/runtime/runtime_context.py +1 -0
  65. package/latticeai/services/app_context.py +8 -0
  66. package/latticeai/services/architecture_readiness.py +1 -1
  67. package/latticeai/services/automation_intelligence.py +22 -2
  68. package/latticeai/services/brain_intelligence.py +123 -7
  69. package/latticeai/services/change_proposals.py +50 -10
  70. package/latticeai/services/command_center.py +10 -4
  71. package/latticeai/services/feature_toggles.py +502 -0
  72. package/latticeai/services/folder_watch.py +122 -1
  73. package/latticeai/services/hybrid_chat.py +56 -5
  74. package/latticeai/services/interop_bridges.py +978 -0
  75. package/latticeai/services/memory_service.py +34 -0
  76. package/latticeai/services/model_capability_registry.py +434 -261
  77. package/latticeai/services/model_catalog.py +95 -61
  78. package/latticeai/services/model_recommendation.py +18 -11
  79. package/latticeai/services/model_runtime.py +1 -1
  80. package/latticeai/services/multimodal_ports.py +112 -0
  81. package/latticeai/services/obsidian_bridge.py +16 -25
  82. package/latticeai/services/product_readiness.py +1 -1
  83. package/latticeai/services/search_service.py +149 -2
  84. package/latticeai/services/self_model_service.py +171 -0
  85. package/latticeai/services/tool_dispatch.py +4 -0
  86. package/latticeai/services/voice_capture.py +27 -1
  87. package/latticeai/setup/auto_setup.py +27 -30
  88. package/latticeai/setup/wizard.py +77 -44
  89. package/package.json +1 -1
  90. package/scripts/check_current_release_docs.mjs +1 -1
  91. package/scripts/check_server_i18n.mjs +1 -0
  92. package/scripts/release_screen_claims.json +22 -0
  93. package/scripts/verify_hf_model_registry.py +253 -218
  94. package/src-tauri/Cargo.lock +1 -1
  95. package/src-tauri/Cargo.toml +1 -1
  96. package/src-tauri/tauri.conf.json +1 -1
  97. package/static/app/asset-manifest.json +37 -37
  98. package/static/app/assets/{Act-D4zSxFR-.js → Act-AWf0SAKp.js} +1 -1
  99. package/static/app/assets/{AdminConsole-w5jBfPt2.js → AdminConsole-D0u8Tiyj.js} +1 -1
  100. package/static/app/assets/{Brain-C2EqQg74.js → Brain-tuhI4sOC.js} +1 -1
  101. package/static/app/assets/BrainHome-Ts7G_Ila.js +2 -0
  102. package/static/app/assets/BrainSignals-jMYgQ2Ar.js +1 -0
  103. package/static/app/assets/{Capture-DPqpGK8d.js → Capture-CqOSzyPr.js} +1 -1
  104. package/static/app/assets/{CommandPalette-CNf7h5fp.js → CommandPalette-DC0Bzh-I.js} +1 -1
  105. package/static/app/assets/{Library-BN0HYOfc.js → Library-CX-bbhmK.js} +1 -1
  106. package/static/app/assets/{LivingBrain-Dfq_wEDI.js → LivingBrain-DBwhto14.js} +1 -1
  107. package/static/app/assets/{ProductFlow-B-3O0rNV.js → ProductFlow-BHA2cfKI.js} +1 -1
  108. package/static/app/assets/{ReviewCard-gZ-tdqFM.js → ReviewCard-BUhCKRNM.js} +1 -1
  109. package/static/app/assets/{System-BElUcSSw.js → System-Bu2t5hn1.js} +1 -1
  110. package/static/app/assets/arrow-left-Dzwa5zRb.js +1 -0
  111. package/static/app/assets/{bot--qYHMtkP.js → bot-Cia42c2h.js} +1 -1
  112. package/static/app/assets/brain-DJMoqrwx.js +1 -0
  113. package/static/app/assets/{button-51Z3rsuv.js → button-2j2Ijzgq.js} +1 -1
  114. package/static/app/assets/{circle-pause-CMIiMaQl.js → circle-pause-BEFeWpVW.js} +1 -1
  115. package/static/app/assets/{circle-play-DZoO_cfG.js → circle-play-ujXMcHxl.js} +1 -1
  116. package/static/app/assets/{cpu-Bs6uc9W9.js → cpu-k4awryFq.js} +1 -1
  117. package/static/app/assets/{download-G-2olkWz.js → download-DFbLJ_ig.js} +1 -1
  118. package/static/app/assets/{folder-open-CTOspnmb.js → folder-open-7y_b6xkM.js} +1 -1
  119. package/static/app/assets/{hard-drive-CewHWJhn.js → hard-drive-Bidh02Kr.js} +1 -1
  120. package/static/app/assets/{index-D7Rr-J2Y.js → index-BpYkzcVm.js} +3 -3
  121. package/static/app/assets/index-DwDl9-8Y.css +2 -0
  122. package/static/app/assets/{input-D4w_BZWl.js → input-DSlJJxRs.js} +1 -1
  123. package/static/app/assets/{permissionCopy-CosBEXAZ.js → permissionCopy-Bpb83Hx9.js} +1 -1
  124. package/static/app/assets/{primitives-d0g9pvzS.js → primitives-BCx6TvfG.js} +1 -1
  125. package/static/app/assets/search-Cgy8cCFJ.js +1 -0
  126. package/static/app/assets/{share-2-NmD7e_oV.js → share-2-BH1M-WNi.js} +1 -1
  127. package/static/app/assets/{shield-alert-CcQeMuju.js → shield-alert-BlKdBXcG.js} +1 -1
  128. package/static/app/assets/{textarea-BPAJDc-0.js → textarea-CCWbUfFB.js} +1 -1
  129. package/static/app/assets/{useFocusTrap-C7YLdTBC.js → useFocusTrap-YdHQ7pJ1.js} +1 -1
  130. package/static/app/assets/{useQuery-DRyD9opW.js → useQuery-CXQiwbVT.js} +1 -1
  131. package/static/app/assets/{utils-DG1_ExrP.js → utils-zqPZJxdx.js} +2 -2
  132. package/static/app/assets/{workspace-CWVf3gsI.js → workspace-DXTihhfU.js} +1 -1
  133. package/static/app/index.html +4 -4
  134. package/static/sw.js +1 -1
  135. package/static/app/assets/BrainHome-CvXS6XiQ.js +0 -2
  136. package/static/app/assets/BrainSignals-DOE_KhOU.js +0 -1
  137. package/static/app/assets/arrow-left-CFNIMjhv.js +0 -1
  138. package/static/app/assets/brain-DDCLjRqO.js +0 -1
  139. package/static/app/assets/index-CkzokZAj.css +0 -2
  140. package/static/app/assets/search-BLCYt75v.js +0 -1
@@ -0,0 +1,978 @@
1
+ """Interop bridges — other people's formats, through the one ingestion gate.
2
+
3
+ Through 11.1.0 Obsidian was the only external source with a bridge, and the
4
+ release said so plainly: Notion, email, calendar, and Git were *scoped out*
5
+ rather than stubbed. This module closes that gap, and it does it the same way
6
+ :mod:`latticeai.services.obsidian_bridge` does — by refusing to open a second
7
+ door into the graph.
8
+
9
+ Every bridge here:
10
+
11
+ * reads **local files the user already owns** (a Notion export, an ``.eml`` on
12
+ disk, a repository path). Nothing calls a vendor API, nothing needs a token,
13
+ and nothing leaves the machine;
14
+ * pushes every item through :meth:`IngestionPipeline.ingest`, so content
15
+ hashing, hooks, provenance, extraction quality and workspace scoping are the
16
+ ones the rest of the product already has;
17
+ * supports ``dry_run``, which reports exactly what a real run would touch and
18
+ writes nothing;
19
+ * reports what it could **not** do — an unresolvable link, a missing decoder, a
20
+ file it could not read — instead of quietly dropping it.
21
+
22
+ What is still out of scope, stated rather than implied: **system integration**.
23
+ There is no macOS Calendar / Mail permission dance, no IMAP, no Google
24
+ Calendar, no Notion API. Those need credentials and background sync, and the
25
+ honest version of this release is "point me at files you exported".
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import email
31
+ import email.policy
32
+ import json
33
+ import os
34
+ import re
35
+ import shutil
36
+ import subprocess # noqa: S404 — one fixed binary, argv list, never a shell
37
+ import tempfile
38
+ import zipfile
39
+ from dataclasses import dataclass, field
40
+ from pathlib import Path, PurePath
41
+ from typing import Any, Dict, List, Optional, Tuple
42
+ from urllib.parse import unquote
43
+
44
+ from lattice_brain.graph.ingest import _scoped_slug_id
45
+ from lattice_brain.ingestion import IngestionItem
46
+
47
+ # ── shared vocabulary ────────────────────────────────────────────────────────
48
+ SOURCE_NOTION = "notion"
49
+ SOURCE_GIT = "git_commit"
50
+ SOURCE_EMAIL = "email"
51
+ SOURCE_CALENDAR = "calendar_event"
52
+ #: Every bridge in this module, for a status surface that wants to list them.
53
+ BRIDGE_SOURCE_TYPES = (SOURCE_NOTION, SOURCE_GIT, SOURCE_EMAIL, SOURCE_CALENDAR)
54
+
55
+ LINK_RELATION = "REFERENCES"
56
+ TAG_RELATION = "TAGGED_AS"
57
+ TOPIC_NODE_TYPE = "Topic"
58
+
59
+ ERROR_REPORT_CAP = 25
60
+ UNRESOLVED_REPORT_CAP = 50
61
+ DEFAULT_MAX_ITEMS = 2000
62
+ DEFAULT_MAX_FILE_BYTES = 2_000_000
63
+ DEFAULT_GIT_COMMITS = 500
64
+
65
+ GRAPH_DISABLED_DETAIL = "Knowledge Graph ingestion is disabled (LATTICEAI_ENABLE_GRAPH)."
66
+
67
+
68
+ def record_error(errors: List[Dict[str, Any]], entry: Dict[str, Any]) -> None:
69
+ """Append one failure to a capped report list.
70
+
71
+ The *count* of failures is always exact; this list is a sample, so one
72
+ unreadable directory cannot turn a summary into a megabyte of paths. One
73
+ helper rather than three copies of the same ``if len(...) <`` check.
74
+ """
75
+ if len(errors) < ERROR_REPORT_CAP:
76
+ errors.append(entry)
77
+
78
+
79
+ def edge_row(
80
+ from_id: str, to_id: str, relation: str, metadata: Dict[str, Any]
81
+ ) -> Dict[str, Any]:
82
+ """One ``import_graph_data`` edge row — the shape every bridge writes.
83
+
84
+ Shared with :mod:`~latticeai.services.obsidian_bridge` on purpose: two
85
+ copies of "what an edge row looks like" is two places to forget the
86
+ ``metadata_json`` encoding.
87
+ """
88
+ return {
89
+ "from_node": from_id,
90
+ "to_node": to_id,
91
+ "type": relation,
92
+ "weight": 1.0,
93
+ "metadata_json": json.dumps(metadata, ensure_ascii=False),
94
+ }
95
+
96
+
97
+ def topic_row(
98
+ topic_id: str, label: str, *, summary: str, metadata: Dict[str, Any]
99
+ ) -> Dict[str, Any]:
100
+ """One ``Topic`` node row, identified by the store's own scoped slug."""
101
+ return {
102
+ "id": topic_id,
103
+ "type": TOPIC_NODE_TYPE,
104
+ "title": label,
105
+ "summary": summary,
106
+ "metadata_json": json.dumps(metadata, ensure_ascii=False),
107
+ "raw_json": "{}",
108
+ }
109
+
110
+
111
+ @dataclass
112
+ class BridgeItem:
113
+ """One thing a bridge found, before it reaches the pipeline."""
114
+
115
+ key: str
116
+ item: IngestionItem
117
+ #: Keys of other items this one points at (resolved after the full scan).
118
+ links: List[str] = field(default_factory=list)
119
+ #: Free-text labels that become ``Topic`` nodes (file paths, calendars…).
120
+ topics: List[str] = field(default_factory=list)
121
+
122
+
123
+ class InteropBridge:
124
+ """Shared skeleton: scan → (dry_run?) → ingest → wire structure → report.
125
+
126
+ Subclasses own exactly one thing — how to turn a local path into
127
+ :class:`BridgeItem` values. Everything after that (the pipeline gate, the
128
+ counters, the edge writes, the honest failure list) is identical across
129
+ sources, which is the whole reason a second bridge did not become a second
130
+ ingestion path.
131
+ """
132
+
133
+ source_type = "interop"
134
+ label = "interop source"
135
+
136
+ def __init__(
137
+ self,
138
+ *,
139
+ pipeline: Any,
140
+ knowledge_graph: Any = None,
141
+ max_items: int = DEFAULT_MAX_ITEMS,
142
+ max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
143
+ ) -> None:
144
+ self._pipeline = pipeline
145
+ self._kg = knowledge_graph
146
+ self._max_items = max(1, int(max_items))
147
+ self._max_file_bytes = max(1, int(max_file_bytes))
148
+
149
+ def available(self) -> bool:
150
+ return self._pipeline is not None and bool(self._pipeline.available())
151
+
152
+ # ── subclass seam ────────────────────────────────────────────────────────
153
+ def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
154
+ """``{"status", "target", "items": [BridgeItem], "errors": [...] , …}``."""
155
+ raise NotImplementedError # pragma: no cover - abstract seam
156
+
157
+ # ── the one public entry point ───────────────────────────────────────────
158
+ def sync(
159
+ self,
160
+ target: Any,
161
+ *,
162
+ owner: Optional[str] = None,
163
+ workspace_id: Optional[str] = None,
164
+ user_email: Optional[str] = None,
165
+ dry_run: bool = False,
166
+ **options: Any,
167
+ ) -> Dict[str, Any]:
168
+ """Ingest everything at ``target`` and wire whatever structure it has."""
169
+ if not self.available():
170
+ return {
171
+ "status": "unavailable",
172
+ "source": self.source_type,
173
+ "target": str(target),
174
+ "detail": GRAPH_DISABLED_DETAIL,
175
+ }
176
+ scan = self.scan(target, **options)
177
+ summary: Dict[str, Any] = {
178
+ "status": scan.get("status", "ok"),
179
+ "source": self.source_type,
180
+ "target": scan.get("target", str(target)),
181
+ "dry_run": bool(dry_run),
182
+ "scanned": int(scan.get("scanned") or 0),
183
+ "items": len(scan.get("items") or []),
184
+ "ingested": 0,
185
+ "duplicate": 0,
186
+ "failed": 0,
187
+ "truncated": bool(scan.get("truncated")),
188
+ "skipped": dict(scan.get("skipped") or {}),
189
+ "links": {
190
+ "resolved": 0,
191
+ "written": 0,
192
+ "unresolved_count": len(scan.get("unresolved") or []),
193
+ "unresolved": list(scan.get("unresolved") or [])[:UNRESOLVED_REPORT_CAP],
194
+ },
195
+ "edges": {"status": "none", "references": 0, "topics": 0, "detail": None},
196
+ "errors": list(scan.get("errors") or []),
197
+ }
198
+ if scan.get("status") != "ok":
199
+ summary["status"] = "failed"
200
+ summary["detail"] = scan.get("detail")
201
+ return summary
202
+ items: List[BridgeItem] = list(scan.get("items") or [])
203
+ known = {entry.key for entry in items}
204
+ resolved = {
205
+ entry.key: [link for link in entry.links if link in known and link != entry.key]
206
+ for entry in items
207
+ }
208
+ summary["links"]["resolved"] = sum(len(v) for v in resolved.values())
209
+ summary["topics"] = len({topic for entry in items for topic in entry.topics})
210
+ if dry_run:
211
+ summary["status"] = "dry_run"
212
+ return summary
213
+
214
+ node_ids = self._ingest_items(items, summary=summary, user_email=user_email or owner)
215
+ self._write_structure(
216
+ items,
217
+ node_ids=node_ids,
218
+ resolved=resolved,
219
+ owner=owner,
220
+ workspace_id=workspace_id,
221
+ summary=summary,
222
+ )
223
+ if summary["failed"] or summary["edges"]["status"] == "failed":
224
+ summary["status"] = "partial"
225
+ return summary
226
+
227
+ # ── internals ────────────────────────────────────────────────────────────
228
+ def _ingest_items(
229
+ self,
230
+ items: List[BridgeItem],
231
+ *,
232
+ summary: Dict[str, Any],
233
+ user_email: Optional[str],
234
+ ) -> Dict[str, str]:
235
+ node_ids: Dict[str, str] = {}
236
+ errors: List[Dict[str, Any]] = summary["errors"]
237
+ for entry in items:
238
+ result = self._pipeline.ingest(entry.item, user_email=user_email)
239
+ if result.status != "ok":
240
+ summary["failed"] += 1
241
+ record_error(errors, {
242
+ "key": entry.key,
243
+ "status": result.status,
244
+ "detail": result.detail,
245
+ })
246
+ continue
247
+ if result.duplicate:
248
+ summary["duplicate"] += 1
249
+ else:
250
+ summary["ingested"] += 1
251
+ if result.node_id:
252
+ node_ids[entry.key] = result.node_id
253
+ return node_ids
254
+
255
+ def _write_structure(
256
+ self,
257
+ items: List[BridgeItem],
258
+ *,
259
+ node_ids: Dict[str, str],
260
+ resolved: Dict[str, List[str]],
261
+ owner: Optional[str],
262
+ workspace_id: Optional[str],
263
+ summary: Dict[str, Any],
264
+ ) -> None:
265
+ edges: List[Dict[str, Any]] = []
266
+ topics: Dict[str, Dict[str, Any]] = {}
267
+ for entry in items:
268
+ from_id = node_ids.get(entry.key)
269
+ if from_id is None:
270
+ continue
271
+ for target_key in resolved.get(entry.key, []):
272
+ to_id = node_ids.get(target_key)
273
+ if to_id is None:
274
+ continue
275
+ edges.append(edge_row(from_id, to_id, LINK_RELATION, {
276
+ "source": self.source_type,
277
+ "from": entry.key,
278
+ "to": target_key,
279
+ }))
280
+ for label in entry.topics:
281
+ topic_id = _scoped_slug_id("topic", label, workspace_id)
282
+ topics.setdefault(topic_id, topic_row(
283
+ topic_id,
284
+ label,
285
+ summary=f"{self.label}: {label}",
286
+ metadata={
287
+ "topic": label,
288
+ "source": self.source_type,
289
+ "owner": owner,
290
+ "workspace_id": workspace_id,
291
+ },
292
+ ))
293
+ edges.append(edge_row(from_id, topic_id, TAG_RELATION, {
294
+ "source": self.source_type,
295
+ "topic": label,
296
+ }))
297
+ if not edges:
298
+ return
299
+ references = sum(1 for edge in edges if edge["type"] == LINK_RELATION)
300
+ if self._kg is None:
301
+ summary["edges"] = {
302
+ "status": "skipped",
303
+ "references": 0,
304
+ "topics": 0,
305
+ "detail": "no Knowledge Graph store is bound; items were ingested without relations",
306
+ }
307
+ return
308
+ try:
309
+ outcome = self._kg.import_graph_data(
310
+ {
311
+ "nodes": list(topics.values()),
312
+ "edges": edges,
313
+ "chunks": [],
314
+ "knowledge_sources": [],
315
+ "provenance": [],
316
+ },
317
+ mode="merge",
318
+ )
319
+ except Exception as exc: # noqa: BLE001 — items already landed; report, never crash
320
+ summary["edges"] = {
321
+ "status": "failed",
322
+ "references": 0,
323
+ "topics": 0,
324
+ "detail": f"relations could not be written: {exc}",
325
+ }
326
+ return
327
+ summary["links"]["written"] = references
328
+ summary["edges"] = {
329
+ "status": "written",
330
+ "references": references,
331
+ "topics": len(topics),
332
+ "detail": None,
333
+ "index": outcome.get("index"),
334
+ }
335
+
336
+ def _failed_scan(self, target: Any, detail: str) -> Dict[str, Any]:
337
+ return {"status": "failed", "target": str(target), "detail": detail, "items": []}
338
+
339
+
340
+ # ── Notion export ────────────────────────────────────────────────────────────
341
+ NOTION_EXTENSIONS = frozenset({".md", ".markdown", ".csv"})
342
+ #: Notion appends a 32-hex page id to every exported filename. Two exports of
343
+ #: the same page produce two different ids, so the id is stripped from the
344
+ #: *title* and kept in metadata rather than shown to a reader.
345
+ _NOTION_ID_RE = re.compile(r"^(?P<title>.*?)[ _-]?(?P<page_id>[0-9a-f]{32})$", re.IGNORECASE)
346
+ _MD_LINK_RE = re.compile(r"\[[^\]\n]{0,200}\]\(([^()\s]{1,400})\)")
347
+ _EXTERNAL_PREFIXES = ("http://", "https://", "mailto:", "notion://", "//")
348
+
349
+
350
+ def notion_title(stem: str) -> Tuple[str, Optional[str]]:
351
+ """``("Roadmap", "1a2b…")`` from ``"Roadmap 1a2b…"`` — id split off, not lost."""
352
+ match = _NOTION_ID_RE.match(str(stem or "").strip())
353
+ if match is None:
354
+ return str(stem or "").strip(), None
355
+ title = match.group("title").strip()
356
+ page_id = match.group("page_id").lower()
357
+ return (title or page_id), page_id
358
+
359
+
360
+ def notion_key(relative_path: str) -> str:
361
+ """Stable identity for one exported page: its path with the id stripped."""
362
+ path = PurePath(str(relative_path or "").replace("\\", "/"))
363
+ title, _ = notion_title(path.stem)
364
+ parent = "/".join(part for part in path.parent.parts if part not in (".", ""))
365
+ normalized = f"{parent}/{title}" if parent else title
366
+ return normalized.strip("/").lower()
367
+
368
+
369
+ def notion_links(body: str) -> List[str]:
370
+ """Relative page links inside an exported page, in document order."""
371
+ found: List[str] = []
372
+ seen: set = set()
373
+ for match in _MD_LINK_RE.finditer(str(body or "")):
374
+ raw = unquote(match.group(1)).strip()
375
+ if not raw or raw.startswith(_EXTERNAL_PREFIXES):
376
+ continue
377
+ suffix = PurePath(raw).suffix.lower()
378
+ if suffix and suffix not in NOTION_EXTENSIONS:
379
+ continue
380
+ key = notion_key(raw)
381
+ if not key or key in seen:
382
+ continue
383
+ seen.add(key)
384
+ found.append(key)
385
+ return found
386
+
387
+
388
+ class NotionExportBridge(InteropBridge):
389
+ """A Notion **export** (directory or ``.zip``) through the one gate.
390
+
391
+ Deliberately not the Notion API: an API bridge needs an integration token,
392
+ a network round trip per page, and a background sync to stay current — all
393
+ of which are the opposite of "local-first, opt-in, off by default". An
394
+ export is a folder the user already downloaded, and it contains the same
395
+ words.
396
+ """
397
+
398
+ source_type = SOURCE_NOTION
399
+ label = "Notion page"
400
+
401
+ def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
402
+ report: Dict[str, Any] = {
403
+ "status": "ok",
404
+ "target": str(target),
405
+ "scanned": 0,
406
+ "items": [],
407
+ "skipped": {"empty": 0, "too_large": 0, "unreadable": 0},
408
+ "truncated": False,
409
+ "errors": [],
410
+ }
411
+ try:
412
+ root = Path(target).expanduser()
413
+ except TypeError:
414
+ return self._failed_scan(target, f"invalid export path: {target!r}")
415
+ if root.is_file() and root.suffix.lower() == ".zip":
416
+ return self._scan_zip(root, report)
417
+ if not root.is_dir():
418
+ return self._failed_scan(root, f"not a Notion export directory or zip: {root}")
419
+ report["target"] = str(root)
420
+ self._walk(root, report)
421
+ return report
422
+
423
+ def _scan_zip(self, archive: Path, report: Dict[str, Any]) -> Dict[str, Any]:
424
+ """Read a ``.zip`` export by extracting it to a temp dir first.
425
+
426
+ A zip member is not a path the pipeline can hash and re-read later, so
427
+ the export is materialized once and the ordinary directory walk runs
428
+ over it. Unsafe member names are refused outright.
429
+ """
430
+ try:
431
+ with zipfile.ZipFile(archive) as zf:
432
+ names = zf.namelist()
433
+ for name in names:
434
+ member = PurePath(name.replace("\\", "/"))
435
+ if member.is_absolute() or ".." in member.parts:
436
+ return self._failed_scan(
437
+ archive, f"export archive contains an unsafe path: {name}"
438
+ )
439
+ staging = Path(tempfile.mkdtemp(prefix="notion-export-"))
440
+ zf.extractall(staging)
441
+ except (zipfile.BadZipFile, OSError) as exc:
442
+ return self._failed_scan(archive, f"export archive could not be read: {exc}")
443
+ report["target"] = str(archive)
444
+ report["extracted_to"] = str(staging)
445
+ self._walk(staging, report)
446
+ return report
447
+
448
+ def _walk(self, root: Path, report: Dict[str, Any]) -> None:
449
+ items: List[BridgeItem] = report["items"]
450
+ skipped = report["skipped"]
451
+ errors: List[Dict[str, Any]] = report["errors"]
452
+ for dirpath, dirnames, filenames in os.walk(root):
453
+ dirnames[:] = sorted(name for name in dirnames if not name.startswith("."))
454
+ current = Path(dirpath)
455
+ for name in sorted(filenames):
456
+ if name.startswith(".") or Path(name).suffix.lower() not in NOTION_EXTENSIONS:
457
+ continue
458
+ report["scanned"] += 1
459
+ path = current / name
460
+ if len(items) >= self._max_items:
461
+ report["truncated"] = True
462
+ continue
463
+ try:
464
+ if path.stat().st_size > self._max_file_bytes:
465
+ skipped["too_large"] += 1
466
+ continue
467
+ text = path.read_text(encoding="utf-8", errors="ignore")
468
+ except OSError as exc:
469
+ skipped["unreadable"] += 1
470
+ record_error(errors, {"key": name, "status": "unreadable", "detail": str(exc)})
471
+ continue
472
+ if not text.strip():
473
+ skipped["empty"] += 1
474
+ continue
475
+ relative = path.relative_to(root).as_posix()
476
+ title, page_id = notion_title(path.stem)
477
+ items.append(BridgeItem(
478
+ key=notion_key(relative),
479
+ links=notion_links(text),
480
+ item=IngestionItem(
481
+ source_type=SOURCE_NOTION,
482
+ title=title,
483
+ text=text,
484
+ source_uri=str(path),
485
+ mime_type="text/markdown" if path.suffix.lower() != ".csv" else "text/csv",
486
+ metadata={
487
+ "relative_path": relative,
488
+ "notion_page_id": page_id,
489
+ "export_kind": "database" if path.suffix.lower() == ".csv" else "page",
490
+ },
491
+ ),
492
+ ))
493
+
494
+
495
+ # ── Git repository ───────────────────────────────────────────────────────────
496
+ GIT_BINARY = "git"
497
+ _GIT_RECORD_SEPARATOR = "\x00"
498
+ _GIT_FIELD_SEPARATOR = "\x1f"
499
+ _GIT_PRETTY = (
500
+ f"format:{_GIT_RECORD_SEPARATOR}%H{_GIT_FIELD_SEPARATOR}%an{_GIT_FIELD_SEPARATOR}"
501
+ f"%ae{_GIT_FIELD_SEPARATOR}%aI{_GIT_FIELD_SEPARATOR}%s{_GIT_FIELD_SEPARATOR}%b"
502
+ f"{_GIT_RECORD_SEPARATOR}"
503
+ )
504
+ GIT_UNAVAILABLE_DETAIL = (
505
+ "reading a repository's history needs git on this machine and none was "
506
+ "found; nothing was ingested"
507
+ )
508
+ GIT_TIMEOUT_SECONDS = 60
509
+
510
+
511
+ def _which_git() -> Optional[str]:
512
+ """Absolute path to git, or ``None``. The one probe, seamed for tests."""
513
+ return shutil.which(GIT_BINARY)
514
+
515
+
516
+ def _run_git(binary: str, args: List[str], cwd: Path) -> Tuple[int, str]:
517
+ """Run git with an argv list in a fixed directory — no shell, ever."""
518
+ completed = subprocess.run( # noqa: S603 — argv list, fixed binary, no shell
519
+ [binary, *args],
520
+ cwd=str(cwd),
521
+ capture_output=True,
522
+ text=True,
523
+ timeout=GIT_TIMEOUT_SECONDS,
524
+ check=False,
525
+ )
526
+ return int(completed.returncode), str(completed.stdout or "")
527
+
528
+
529
+ def parse_git_log(output: str) -> List[Dict[str, Any]]:
530
+ """Parse the ``--pretty``/``--name-only`` stream into commit records.
531
+
532
+ The separators are NUL and unit-separator rather than newlines because a
533
+ commit body contains newlines and a filename may contain almost anything
534
+ else. Anything that does not parse into six fields is skipped rather than
535
+ guessed at.
536
+ """
537
+ chunks = str(output or "").split(_GIT_RECORD_SEPARATOR)
538
+ commits: List[Dict[str, Any]] = []
539
+ index = 1
540
+ while index < len(chunks):
541
+ fields = chunks[index].split(_GIT_FIELD_SEPARATOR)
542
+ trailer = chunks[index + 1] if index + 1 < len(chunks) else ""
543
+ index += 2
544
+ if len(fields) != 6:
545
+ continue
546
+ sha, author, mail, when, subject, body = fields
547
+ files = [line.strip() for line in trailer.splitlines() if line.strip()]
548
+ commits.append({
549
+ "sha": sha.strip(),
550
+ "author": author.strip(),
551
+ "author_email": mail.strip(),
552
+ "date": when.strip(),
553
+ "subject": subject.strip(),
554
+ "body": body.strip(),
555
+ "files": files,
556
+ })
557
+ return commits
558
+
559
+
560
+ class GitHistoryBridge(InteropBridge):
561
+ """A local repository's commit history as project memory.
562
+
563
+ One node per commit — message, author, date, and the files it touched —
564
+ with each file path joined as a ``Topic``, so "what changed in the auth
565
+ module" is a graph traversal rather than a shell command. The commit hash
566
+ rides in the source URI and the body, so re-running reports duplicates
567
+ instead of writing a second copy of the same commit.
568
+
569
+ Nothing is cloned or fetched: this reads a path the user approved, with
570
+ ``git log``, and says so when git is not installed.
571
+ """
572
+
573
+ source_type = SOURCE_GIT
574
+ label = "repository file"
575
+
576
+ def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
577
+ limit = max(1, int(options.get("max_commits") or DEFAULT_GIT_COMMITS))
578
+ report: Dict[str, Any] = {
579
+ "status": "ok",
580
+ "target": str(target),
581
+ "scanned": 0,
582
+ "items": [],
583
+ "truncated": False,
584
+ "errors": [],
585
+ }
586
+ try:
587
+ root = Path(target).expanduser()
588
+ except TypeError:
589
+ return self._failed_scan(target, f"invalid repository path: {target!r}")
590
+ if not root.is_dir():
591
+ return self._failed_scan(root, f"not a directory: {root}")
592
+ if not (root / ".git").exists():
593
+ return self._failed_scan(root, f"not a git repository (no .git): {root}")
594
+ binary = _which_git()
595
+ if binary is None:
596
+ return self._failed_scan(root, GIT_UNAVAILABLE_DETAIL)
597
+ args = [
598
+ "log", f"--max-count={limit}", "--name-only",
599
+ f"--pretty={_GIT_PRETTY}",
600
+ ]
601
+ try:
602
+ code, output = _run_git(binary, args, root)
603
+ except Exception as exc: # noqa: BLE001 — a broken git is a state, not a crash
604
+ return self._failed_scan(root, f"git log failed: {exc}")
605
+ if code != 0:
606
+ return self._failed_scan(root, f"git log exited with status {code}")
607
+ report["target"] = str(root)
608
+ commits = parse_git_log(output)
609
+ report["scanned"] = len(commits)
610
+ items: List[BridgeItem] = report["items"]
611
+ for commit in commits:
612
+ if len(items) >= self._max_items:
613
+ report["truncated"] = True
614
+ break
615
+ items.append(self._commit_item(root, commit))
616
+ return report
617
+
618
+ def _commit_item(self, root: Path, commit: Dict[str, Any]) -> BridgeItem:
619
+ sha = commit["sha"]
620
+ files = commit["files"]
621
+ lines = [
622
+ f"commit {sha}",
623
+ f"작성자: {commit['author']} <{commit['author_email']}>",
624
+ f"시각: {commit['date']}",
625
+ "",
626
+ commit["subject"],
627
+ ]
628
+ if commit["body"]:
629
+ lines.extend(["", commit["body"]])
630
+ if files:
631
+ lines.extend(["", "변경된 파일:", *[f"- {name}" for name in files]])
632
+ return BridgeItem(
633
+ key=sha,
634
+ topics=list(files),
635
+ item=IngestionItem(
636
+ source_type=SOURCE_GIT,
637
+ title=f"{commit['subject'] or sha[:12]} ({sha[:8]})",
638
+ text="\n".join(lines),
639
+ source_uri=f"git:{root}#{sha}",
640
+ mime_type="text/plain",
641
+ modified_at=commit["date"] or None,
642
+ metadata={
643
+ "repository": str(root),
644
+ "commit": sha,
645
+ "author": commit["author"],
646
+ "author_email": commit["author_email"],
647
+ "committed_at": commit["date"],
648
+ "files": files,
649
+ "file_count": len(files),
650
+ },
651
+ ),
652
+ )
653
+
654
+
655
+ # ── email (.eml) and calendar (.ics) ─────────────────────────────────────────
656
+ EMAIL_EXTENSIONS = frozenset({".eml"})
657
+ CALENDAR_EXTENSIONS = frozenset({".ics"})
658
+ MAILBOX_EXTENSIONS = EMAIL_EXTENSIONS | CALENDAR_EXTENSIONS
659
+ _ICS_ESCAPES = (("\\n", "\n"), ("\\N", "\n"), ("\\,", ","), ("\\;", ";"), ("\\\\", "\\"))
660
+
661
+
662
+ def _unfold_ics(text: str) -> List[str]:
663
+ """RFC 5545 line unfolding: a leading space continues the previous line."""
664
+ lines: List[str] = []
665
+ for raw in str(text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n"):
666
+ if raw[:1] in (" ", "\t") and lines:
667
+ lines[-1] += raw[1:]
668
+ continue
669
+ lines.append(raw)
670
+ return lines
671
+
672
+
673
+ def _ics_value(raw: str) -> str:
674
+ value = raw
675
+ for token, replacement in _ICS_ESCAPES:
676
+ value = value.replace(token, replacement)
677
+ return value.strip()
678
+
679
+
680
+ def parse_ics(text: str) -> List[Dict[str, str]]:
681
+ """Every ``VEVENT`` in an ``.ics`` file as a flat dict of its properties.
682
+
683
+ A deliberately small parser rather than a dependency: this needs five
684
+ properties out of a format whose grammar is line folding plus
685
+ ``NAME;PARAM=…:VALUE``, and a calendar library in the ingest path would be
686
+ a permanent cost for that. Properties it does not understand are ignored,
687
+ never guessed at, and an unterminated block is dropped rather than
688
+ half-reported.
689
+ """
690
+ events: List[Dict[str, str]] = []
691
+ current: Optional[Dict[str, str]] = None
692
+ for line in _unfold_ics(text):
693
+ stripped = line.strip()
694
+ if stripped.upper() == "BEGIN:VEVENT":
695
+ current = {}
696
+ continue
697
+ if stripped.upper() == "END:VEVENT":
698
+ # A stray END with no BEGIN closes nothing rather than inventing an
699
+ # empty event.
700
+ if current is not None:
701
+ events.append(current)
702
+ current = None
703
+ continue
704
+ if current is None or ":" not in stripped:
705
+ continue
706
+ name, _, value = stripped.partition(":")
707
+ key = name.split(";", 1)[0].strip().upper()
708
+ current[key] = _ics_value(value)
709
+ return events
710
+
711
+
712
+ def _preferred_part(message: Any, kind: str) -> Any:
713
+ """One ``get_body`` lookup; a malformed message answers ``None``, not a raise."""
714
+ try:
715
+ return message.get_body(preferencelist=(kind,))
716
+ except Exception: # noqa: BLE001 — a malformed message is a state
717
+ return None
718
+
719
+
720
+ def email_body(message: Any) -> Tuple[str, str]:
721
+ """``(body, status)`` for a parsed message — plain text only, honestly.
722
+
723
+ HTML-only mail yields an empty body and ``html_only``: turning markup into
724
+ prose is a job for the extraction layer the capture surfaces already own,
725
+ and a naive tag-strip here would store navigation chrome as if it were the
726
+ message.
727
+ """
728
+ part = _preferred_part(message, "plain")
729
+ if part is None:
730
+ return "", "html_only" if _preferred_part(message, "html") else "empty"
731
+ try:
732
+ content = str(part.get_content() or "").strip()
733
+ except Exception as exc: # noqa: BLE001 — undecodable charset is a state
734
+ return "", f"undecodable: {exc}"
735
+ return content, "ok" if content else "empty"
736
+
737
+
738
+ class MailCalendarBridge(InteropBridge):
739
+ """Local ``.eml`` messages and ``.ics`` calendars, through the one gate.
740
+
741
+ Both are file formats with standard-library (or five-line) parsers, which
742
+ is exactly why they are in this release and a live mailbox connection is
743
+ not: reading ``~/Mail`` needs an OS permission grant and a sync loop, and
744
+ neither belongs behind a feature that claims to only read what it was
745
+ pointed at. Point this at a folder of exported messages.
746
+ """
747
+
748
+ source_type = SOURCE_EMAIL
749
+ label = "calendar"
750
+
751
+ def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
752
+ report: Dict[str, Any] = {
753
+ "status": "ok",
754
+ "target": str(target),
755
+ "scanned": 0,
756
+ "items": [],
757
+ "skipped": {"empty": 0, "too_large": 0, "unreadable": 0},
758
+ "truncated": False,
759
+ "errors": [],
760
+ }
761
+ try:
762
+ root = Path(target).expanduser()
763
+ except TypeError:
764
+ return self._failed_scan(target, f"invalid path: {target!r}")
765
+ if root.is_file():
766
+ paths = [root] if root.suffix.lower() in MAILBOX_EXTENSIONS else []
767
+ if not paths:
768
+ return self._failed_scan(root, f"not an .eml or .ics file: {root}")
769
+ elif root.is_dir():
770
+ paths = sorted(
771
+ path for path in root.rglob("*")
772
+ if path.is_file()
773
+ and not path.name.startswith(".")
774
+ and path.suffix.lower() in MAILBOX_EXTENSIONS
775
+ )
776
+ else:
777
+ return self._failed_scan(root, f"no such file or folder: {root}")
778
+ report["target"] = str(root)
779
+ self._read_all(paths, report)
780
+ return report
781
+
782
+ def _read_all(self, paths: List[Path], report: Dict[str, Any]) -> None:
783
+ items: List[BridgeItem] = report["items"]
784
+ skipped = report["skipped"]
785
+ errors: List[Dict[str, Any]] = report["errors"]
786
+ for path in paths:
787
+ report["scanned"] += 1
788
+ if len(items) >= self._max_items:
789
+ report["truncated"] = True
790
+ continue
791
+ try:
792
+ if path.stat().st_size > self._max_file_bytes:
793
+ skipped["too_large"] += 1
794
+ continue
795
+ raw = path.read_bytes()
796
+ except OSError as exc:
797
+ skipped["unreadable"] += 1
798
+ record_error(errors, {"key": path.name, "status": "unreadable", "detail": str(exc)})
799
+ continue
800
+ produced = (
801
+ self._calendar_items(path, raw)
802
+ if path.suffix.lower() in CALENDAR_EXTENSIONS
803
+ else self._message_items(path, raw)
804
+ )
805
+ if not produced:
806
+ skipped["empty"] += 1
807
+ continue
808
+ items.extend(produced)
809
+
810
+ def _message_items(self, path: Path, raw: bytes) -> List[BridgeItem]:
811
+ try:
812
+ message = email.message_from_bytes(raw, policy=email.policy.default)
813
+ except Exception as exc: # noqa: BLE001 — an unparseable message is a state
814
+ return [self._unreadable_item(path, f"message could not be parsed: {exc}")]
815
+ body, status = email_body(message)
816
+ subject = str(message.get("Subject") or path.stem).strip() or path.stem
817
+ sender = str(message.get("From") or "").strip()
818
+ recipients = str(message.get("To") or "").strip()
819
+ sent_at = str(message.get("Date") or "").strip()
820
+ message_id = str(message.get("Message-ID") or "").strip()
821
+ header_block = "\n".join(
822
+ line for line in (
823
+ f"보낸 사람: {sender}" if sender else "",
824
+ f"받는 사람: {recipients}" if recipients else "",
825
+ f"보낸 시각: {sent_at}" if sent_at else "",
826
+ ) if line
827
+ )
828
+ text = f"{header_block}\n\n{body}".strip() if body else (
829
+ f"{header_block}\n\n[본문 없음] 이 메일은 일반 텍스트 본문이 없어 "
830
+ "내용 검색은 되지 않습니다."
831
+ ).strip()
832
+ return [BridgeItem(
833
+ key=message_id or str(path),
834
+ item=IngestionItem(
835
+ source_type=SOURCE_EMAIL,
836
+ title=subject,
837
+ text=text,
838
+ source_uri=str(path),
839
+ mime_type="message/rfc822",
840
+ modified_at=sent_at or None,
841
+ metadata={
842
+ "from": sender,
843
+ "to": recipients,
844
+ "sent_at": sent_at,
845
+ "message_id": message_id,
846
+ "body_status": status,
847
+ "searchable": bool(body),
848
+ },
849
+ ),
850
+ )]
851
+
852
+ def _calendar_items(self, path: Path, raw: bytes) -> List[BridgeItem]:
853
+ events = parse_ics(raw.decode("utf-8", errors="ignore"))
854
+ produced: List[BridgeItem] = []
855
+ for index, event in enumerate(events):
856
+ title = event.get("SUMMARY") or f"{path.stem} 일정 {index + 1}"
857
+ uid = event.get("UID") or f"{path}#{index}"
858
+ lines = [title]
859
+ for label, key in (("시작", "DTSTART"), ("종료", "DTEND"), ("장소", "LOCATION")):
860
+ if event.get(key):
861
+ lines.append(f"{label}: {event[key]}")
862
+ if event.get("DESCRIPTION"):
863
+ lines.extend(["", event["DESCRIPTION"]])
864
+ produced.append(BridgeItem(
865
+ key=uid,
866
+ topics=[event["LOCATION"]] if event.get("LOCATION") else [],
867
+ item=IngestionItem(
868
+ source_type=SOURCE_CALENDAR,
869
+ title=title,
870
+ text="\n".join(lines),
871
+ source_uri=f"{path}#{uid}",
872
+ mime_type="text/calendar",
873
+ modified_at=event.get("DTSTART") or None,
874
+ metadata={
875
+ "calendar_file": str(path),
876
+ "uid": uid,
877
+ "starts_at": event.get("DTSTART", ""),
878
+ "ends_at": event.get("DTEND", ""),
879
+ "location": event.get("LOCATION", ""),
880
+ },
881
+ ),
882
+ ))
883
+ return produced
884
+
885
+ @staticmethod
886
+ def _unreadable_item(path: Path, detail: str) -> BridgeItem:
887
+ """Keep the fact that a file existed, and say what went wrong with it."""
888
+ return BridgeItem(
889
+ key=str(path),
890
+ item=IngestionItem(
891
+ source_type=SOURCE_EMAIL,
892
+ title=path.stem,
893
+ text=f"[읽을 수 없는 메일] {path.name}\n{detail}",
894
+ source_uri=str(path),
895
+ mime_type="message/rfc822",
896
+ metadata={"body_status": "unreadable", "detail": detail, "searchable": False},
897
+ ),
898
+ )
899
+
900
+
901
+ BRIDGES: Dict[str, Any] = {
902
+ SOURCE_NOTION: NotionExportBridge,
903
+ "git": GitHistoryBridge,
904
+ "mail": MailCalendarBridge,
905
+ }
906
+
907
+
908
+ def build_bridge(
909
+ kind: str, *, pipeline: Any, knowledge_graph: Any = None, **options: Any
910
+ ) -> InteropBridge:
911
+ """One named bridge, or a ``ValueError`` naming the ones that exist."""
912
+ factory = BRIDGES.get(str(kind or "").strip().lower())
913
+ if factory is None:
914
+ raise ValueError(
915
+ f"unknown interop source '{kind}'; available: {', '.join(sorted(BRIDGES))}"
916
+ )
917
+ return factory(pipeline=pipeline, knowledge_graph=knowledge_graph, **options)
918
+
919
+
920
+ def bridge_status() -> Dict[str, Any]:
921
+ """What each bridge can do on *this* machine, right now.
922
+
923
+ Git is the only one with a runtime prerequisite, and it is reported rather
924
+ than discovered when someone tries to use it.
925
+ """
926
+ return {
927
+ "sources": {
928
+ SOURCE_NOTION: {
929
+ "available": True,
930
+ "accepts": ["directory", ".zip"],
931
+ "detail": "a Notion export you downloaded — never the Notion API",
932
+ },
933
+ "git": {
934
+ "available": _which_git() is not None,
935
+ "accepts": ["repository directory"],
936
+ "detail": None if _which_git() else GIT_UNAVAILABLE_DETAIL,
937
+ },
938
+ "mail": {
939
+ "available": True,
940
+ "accepts": [".eml", ".ics", "a folder of either"],
941
+ "detail": (
942
+ "local files only; connecting a live mailbox or system "
943
+ "calendar is deliberately out of scope"
944
+ ),
945
+ },
946
+ },
947
+ }
948
+
949
+
950
+ __all__ = [
951
+ "BRIDGES",
952
+ "BRIDGE_SOURCE_TYPES",
953
+ "CALENDAR_EXTENSIONS",
954
+ "DEFAULT_GIT_COMMITS",
955
+ "EMAIL_EXTENSIONS",
956
+ "GIT_UNAVAILABLE_DETAIL",
957
+ "NOTION_EXTENSIONS",
958
+ "SOURCE_CALENDAR",
959
+ "SOURCE_EMAIL",
960
+ "SOURCE_GIT",
961
+ "SOURCE_NOTION",
962
+ "BridgeItem",
963
+ "GitHistoryBridge",
964
+ "InteropBridge",
965
+ "MailCalendarBridge",
966
+ "NotionExportBridge",
967
+ "bridge_status",
968
+ "build_bridge",
969
+ "edge_row",
970
+ "email_body",
971
+ "notion_key",
972
+ "notion_links",
973
+ "notion_title",
974
+ "parse_git_log",
975
+ "record_error",
976
+ "parse_ics",
977
+ "topic_row",
978
+ ]