ltcai 11.5.2 → 11.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -148
- package/bin/ltcai.js +234 -24
- package/docs/CHANGELOG.md +119 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +8 -6
- package/docs/ENTERPRISE.md +2 -1
- package/docs/MULTI_AGENT_RUNTIME.md +12 -5
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -3
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/WORKFLOW_DESIGNER.md +3 -2
- package/docs/kg-schema.md +1 -1
- package/docs/v11.6.0_ONE_DOOR_PLAN.md +170 -0
- package/lattice_brain/__init__.py +42 -79
- package/lattice_brain/graph/__init__.py +9 -24
- package/lattice_brain/graph/_kg_common/__init__.py +13 -24
- package/lattice_brain/ingestion/__init__.py +19 -58
- package/lattice_brain/ingestion/pipeline.py +20 -398
- package/lattice_brain/multimodal/__init__.py +7 -18
- package/lattice_brain/multimodal/images.py +6 -264
- package/lattice_brain/multimodal/video.py +8 -247
- package/lattice_brain/runtime/__init__.py +8 -79
- package/lattice_brain/runtime/hooks.py +22 -584
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +7 -80
- package/latticeai/api/health.py +6 -20
- package/latticeai/api/local_files.py +16 -613
- package/latticeai/api/models.py +12 -87
- package/latticeai/api/search.py +17 -255
- package/latticeai/api/tools.py +69 -750
- package/latticeai/api/voice_capture.py +8 -70
- package/latticeai/api/worker_compute.py +842 -0
- package/latticeai/api/worker_seams.py +216 -0
- package/latticeai/app_factory.py +23 -232
- package/latticeai/cli/entrypoint.py +36 -249
- package/latticeai/core/agent_permission.py +16 -91
- package/latticeai/core/messages.py +89 -531
- package/latticeai/runtime/access_runtime.py +0 -14
- package/latticeai/runtime/bootstrap.py +10 -21
- package/latticeai/runtime/brain_runtime.py +35 -43
- package/latticeai/runtime/build_phases/__init__.py +21 -37
- package/latticeai/runtime/build_phases/features.py +99 -389
- package/latticeai/runtime/build_phases/foundation.py +82 -392
- package/latticeai/runtime/build_phases/web.py +91 -408
- package/latticeai/runtime/build_phases/worker_profile.py +244 -0
- package/latticeai/runtime/lifespan_runtime.py +7 -16
- package/latticeai/runtime/platform_services_runtime.py +1 -15
- package/latticeai/runtime/runtime_context.py +13 -130
- package/latticeai/runtime/security_runtime.py +10 -103
- package/latticeai/services/architecture_readiness.py +82 -43
- package/latticeai/services/model_runtime/__init__.py +2 -11
- package/latticeai/services/model_runtime/service.py +4 -28
- package/latticeai/services/p_reinforce.py +10 -261
- package/latticeai/services/product_readiness.py +31 -38
- package/latticeai/services/search_service.py +37 -795
- package/latticeai/services/tool_dispatch.py +30 -386
- package/latticeai/services/voice_capture.py +13 -107
- package/latticeai/tools/__init__.py +22 -49
- package/latticeai/tools/commands.py +0 -163
- package/latticeai/tools/computer.py +0 -39
- package/latticeai/tools/documents.py +1 -134
- package/latticeai/tools/filesystem.py +1 -247
- package/latticeai/tools/knowledge.py +1 -52
- package/latticeai/tools/local_files.py +0 -20
- package/latticeai/worker_app.py +75 -0
- package/package.json +2 -3
- package/requirements.txt +0 -5
- package/scripts/agent_eval.py +16 -28
- package/scripts/brain_quality_eval.py +20 -183
- package/scripts/bump_version.py +5 -5
- package/scripts/check_current_release_docs.mjs +7 -4
- package/scripts/check_openapi_drift.mjs +10 -0
- package/scripts/check_server_i18n.mjs +2 -18
- package/scripts/compose_openapi.py +377 -0
- package/scripts/export_openapi.py +32 -4
- package/scripts/gen_messages_catalog_fixture.py +302 -0
- package/scripts/gen_openapi_fragments.py +365 -0
- package/scripts/gen_redact_fixture.py +196 -0
- package/scripts/gen_worker_allowlist_fixture.py +115 -0
- package/scripts/generate_agent_parity_fixtures.py +33 -14
- package/scripts/openapi_route_families.json +2161 -0
- package/scripts/release_screen_claims.json +54 -0
- package/scripts/run_integration_tests.mjs +134 -29
- package/scripts/run_sidecar_e2e.mjs +104 -11
- package/scripts/wheel_smoke.py +32 -22
- package/src-tauri/Cargo.lock +375 -9
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/src/backend.rs +151 -45
- package/src-tauri/src/main.rs +18 -18
- package/src-tauri/src/topology.rs +58 -88
- package/src-tauri/tauri.conf.json +1 -2
- package/static/app/asset-manifest.json +41 -41
- package/static/app/assets/{Act-DcQizkl1.js → Act-BPcVAbOL.js} +1 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +1 -0
- package/static/app/assets/{Brain-3VCSHFcn.js → Brain-CT92Kos0.js} +2 -2
- package/static/app/assets/{BrainHome-Qm8eaztx.js → BrainHome-CFBkt1K_.js} +1 -1
- package/static/app/assets/{BrainSignals-DS9BtKOW.js → BrainSignals-ReLWF2H8.js} +1 -1
- package/static/app/assets/Capture-BsTokYkk.js +1 -0
- package/static/app/assets/{Chronicle-BGvuAchH.js → Chronicle-B6f0T9id.js} +1 -1
- package/static/app/assets/{CommandPalette-Bqhm0Urn.js → CommandPalette-CuvjTv1u.js} +1 -1
- package/static/app/assets/{Library-BV6NnF0a.js → Library-BGJbG9Hd.js} +1 -1
- package/static/app/assets/{LivingBrain-GzenJchP.js → LivingBrain-DGYK_Jsa.js} +1 -1
- package/static/app/assets/{ProductFlow-DEP6-vML.js → ProductFlow-DXBC6brE.js} +1 -1
- package/static/app/assets/{ReviewCard-CNZ7XjWG.js → ReviewCard-HXRle3qq.js} +2 -2
- package/static/app/assets/System-CMHSO9qM.js +1 -0
- package/static/app/assets/arrow-left-BfmkskWx.js +1 -0
- package/static/app/assets/{bot-B_K1Tdmw.js → bot-Cn8bWRuq.js} +1 -1
- package/static/app/assets/{brain-DWyaV1L1.js → brain-CQJberbE.js} +1 -1
- package/static/app/assets/{button-aTn4s84A.js → button-Ct9f2_oT.js} +1 -1
- package/static/app/assets/circle-check-DruOxB-4.js +1 -0
- package/static/app/assets/{circle-pause-xKgeGXkT.js → circle-pause-CmzC_apg.js} +1 -1
- package/static/app/assets/{circle-play-DkT6tYPX.js → circle-play-D8mW2aQ7.js} +1 -1
- package/static/app/assets/{cpu-85xYObUC.js → cpu-DZcdd0PZ.js} +1 -1
- package/static/app/assets/{download-B5Fm7YXo.js → download-bv1KEPGQ.js} +1 -1
- package/static/app/assets/{folder-open-kk2Xa52u.js → folder-open-d-Pip5gr.js} +1 -1
- package/static/app/assets/{hard-drive-DkA3zBW_.js → hard-drive-D20iavUb.js} +1 -1
- package/static/app/assets/index-D9x-kSNy.css +2 -0
- package/static/app/assets/{index-BMPdTmlY.js → index-Do83hDzJ.js} +3 -3
- package/static/app/assets/{input-B0nRf2jO.js → input-BLXVNmj1.js} +1 -1
- package/static/app/assets/{link-2-Dwb4gnTc.js → link-2-BPJOFlAy.js} +1 -1
- package/static/app/assets/{permissionCopy-CQDUBrOZ.js → permissionCopy-ChdJd493.js} +1 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +1 -0
- package/static/app/assets/search-CT9aho2j.js +1 -0
- package/static/app/assets/{share-2-BsrxFglO.js → share-2-YNX_NtMU.js} +1 -1
- package/static/app/assets/{shield-alert-5BStfp2_.js → shield-alert-DuQ3zrVL.js} +1 -1
- package/static/app/assets/{textarea-Cg8IUA-k.js → textarea-DqwLnli4.js} +1 -1
- package/static/app/assets/{useFocusTrap-CYKvE46M.js → useFocusTrap-ZVI98jaW.js} +1 -1
- package/static/app/assets/{useMutation-CSn9t1op.js → useMutation-CVC4qv_D.js} +1 -1
- package/static/app/assets/{useQuery-CY2OI2uy.js → useQuery-C7BeG4HU.js} +1 -1
- package/static/app/assets/{utils-Ddol2RWD.js → utils-CiFtIdZq.js} +1 -1
- package/static/app/assets/{workspace-BqDwOz_p.js → workspace-DQz9vIId.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/archive.py +0 -522
- package/lattice_brain/context.py +0 -325
- package/lattice_brain/conversations.py +0 -382
- package/lattice_brain/core.py +0 -82
- package/lattice_brain/graph/_kg_contract.py +0 -249
- package/lattice_brain/graph/curator.py +0 -676
- package/lattice_brain/graph/discovery.py +0 -595
- package/lattice_brain/graph/discovery_index/__init__.py +0 -35
- package/lattice_brain/graph/discovery_index/cleanup.py +0 -182
- package/lattice_brain/graph/discovery_index/extract.py +0 -137
- package/lattice_brain/graph/discovery_index/scan.py +0 -411
- package/lattice_brain/graph/discovery_index/upsert.py +0 -495
- package/lattice_brain/graph/documents.py +0 -380
- package/lattice_brain/graph/fusion.py +0 -395
- package/lattice_brain/graph/identity.py +0 -175
- package/lattice_brain/graph/image_vectors.py +0 -230
- package/lattice_brain/graph/ingest.py +0 -829
- package/lattice_brain/graph/network.py +0 -205
- package/lattice_brain/graph/proactive.py +0 -724
- package/lattice_brain/graph/projection/__init__.py +0 -42
- package/lattice_brain/graph/projection/curation.py +0 -500
- package/lattice_brain/graph/projection/v2_schema.py +0 -518
- package/lattice_brain/graph/provenance.py +0 -524
- package/lattice_brain/graph/rerank.py +0 -163
- package/lattice_brain/graph/retrieval/__init__.py +0 -54
- package/lattice_brain/graph/retrieval/context.py +0 -197
- package/lattice_brain/graph/retrieval/graph_view.py +0 -319
- package/lattice_brain/graph/retrieval/hybrid.py +0 -488
- package/lattice_brain/graph/retrieval/maintenance.py +0 -121
- package/lattice_brain/graph/retrieval/signals.py +0 -95
- package/lattice_brain/graph/retrieval_docgen.py +0 -253
- package/lattice_brain/graph/retrieval_policy.py +0 -180
- package/lattice_brain/graph/retrieval_reads.py +0 -769
- package/lattice_brain/graph/retrieval_vector/__init__.py +0 -42
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +0 -97
- package/lattice_brain/graph/retrieval_vector/indexing.py +0 -347
- package/lattice_brain/graph/retrieval_vector/search.py +0 -560
- package/lattice_brain/graph/retrieval_vector/status.py +0 -374
- package/lattice_brain/graph/schema.py +0 -792
- package/lattice_brain/graph/store.py +0 -268
- package/lattice_brain/graph/vector_index/__init__.py +0 -85
- package/lattice_brain/graph/vector_index/base.py +0 -167
- package/lattice_brain/graph/vector_index/brute_force.py +0 -110
- package/lattice_brain/graph/vector_index/hnsw.py +0 -290
- package/lattice_brain/graph/vector_index/jobs.py +0 -287
- package/lattice_brain/graph/vector_index/quantized.py +0 -148
- package/lattice_brain/graph/vector_index/selector.py +0 -161
- package/lattice_brain/graph/write_master.py +0 -308
- package/lattice_brain/ingestion/_contract.py +0 -90
- package/lattice_brain/ingestion/folder_scan.py +0 -57
- package/lattice_brain/ingestion/folders.py +0 -258
- package/lattice_brain/ingestion/jobs_api.py +0 -107
- package/lattice_brain/ingestion/routing.py +0 -295
- package/lattice_brain/ingestion_jobs.py +0 -380
- package/lattice_brain/memory.py +0 -75
- package/lattice_brain/portability/__init__.py +0 -90
- package/lattice_brain/portability/_contract.py +0 -42
- package/lattice_brain/portability/backups.py +0 -338
- package/lattice_brain/portability/bundles.py +0 -136
- package/lattice_brain/portability/constants.py +0 -93
- package/lattice_brain/portability/fsops.py +0 -138
- package/lattice_brain/portability/service.py +0 -41
- package/lattice_brain/portability/sharing.py +0 -710
- package/lattice_brain/quality.py +0 -543
- package/lattice_brain/retrieval_benchmark_fixtures.py +0 -92
- package/lattice_brain/runtime/agent_runtime.py +0 -859
- package/lattice_brain/runtime/contracts.py +0 -460
- package/lattice_brain/runtime/multi_agent.py +0 -942
- package/lattice_brain/runtime/statuses.py +0 -10
- package/lattice_brain/sealed_box.py +0 -240
- package/lattice_brain/self_model.py +0 -675
- package/lattice_brain/sensitivity.py +0 -94
- package/lattice_brain/storage/__init__.py +0 -22
- package/lattice_brain/storage/base.py +0 -100
- package/lattice_brain/storage/docker.py +0 -105
- package/lattice_brain/storage/factory.py +0 -31
- package/lattice_brain/storage/migration.py +0 -191
- package/lattice_brain/storage/postgres.py +0 -123
- package/lattice_brain/storage/sqlite.py +0 -143
- package/lattice_brain/synthesis.py +0 -824
- package/lattice_brain/workflow.py +0 -497
- package/latticeai/api/admin.py +0 -471
- package/latticeai/api/agent_registry.py +0 -105
- package/latticeai/api/agents.py +0 -228
- package/latticeai/api/auth.py +0 -383
- package/latticeai/api/automation_intelligence.py +0 -401
- package/latticeai/api/brain_intelligence.py +0 -199
- package/latticeai/api/browser.py +0 -493
- package/latticeai/api/change_proposals.py +0 -89
- package/latticeai/api/chat.py +0 -572
- package/latticeai/api/chat_agent_http.py +0 -892
- package/latticeai/api/chat_contracts.py +0 -73
- package/latticeai/api/chat_documents.py +0 -276
- package/latticeai/api/chat_helpers.py +0 -460
- package/latticeai/api/chat_history.py +0 -101
- package/latticeai/api/chat_hybrid.py +0 -113
- package/latticeai/api/chat_intents.py +0 -646
- package/latticeai/api/chat_stream.py +0 -216
- package/latticeai/api/chronicle.py +0 -63
- package/latticeai/api/command_center.py +0 -51
- package/latticeai/api/computer_use.py +0 -474
- package/latticeai/api/evidence_actions.py +0 -48
- package/latticeai/api/features.py +0 -70
- package/latticeai/api/funnel_metrics.py +0 -31
- package/latticeai/api/garden.py +0 -34
- package/latticeai/api/hooks.py +0 -165
- package/latticeai/api/index_jobs.py +0 -145
- package/latticeai/api/invitations.py +0 -100
- package/latticeai/api/knowledge_graph.py +0 -536
- package/latticeai/api/marketplace.py +0 -105
- package/latticeai/api/mcp.py +0 -482
- package/latticeai/api/memory.py +0 -270
- package/latticeai/api/network.py +0 -81
- package/latticeai/api/network_boundary.py +0 -225
- package/latticeai/api/permission_mode.py +0 -61
- package/latticeai/api/permissions.py +0 -436
- package/latticeai/api/plugins.py +0 -126
- package/latticeai/api/portability.py +0 -391
- package/latticeai/api/project_sessions.py +0 -114
- package/latticeai/api/realtime.py +0 -118
- package/latticeai/api/review_queue.py +0 -364
- package/latticeai/api/security_dashboard.py +0 -604
- package/latticeai/api/setup.py +0 -319
- package/latticeai/api/static_routes.py +0 -354
- package/latticeai/api/ui_redirects.py +0 -26
- package/latticeai/api/workflow_designer.py +0 -394
- package/latticeai/api/workspace.py +0 -856
- package/latticeai/api/workspace_scope.py +0 -125
- package/latticeai/core/agent/__init__.py +0 -93
- package/latticeai/core/agent/_contract.py +0 -79
- package/latticeai/core/agent/context.py +0 -57
- package/latticeai/core/agent/deps.py +0 -125
- package/latticeai/core/agent/execution.py +0 -622
- package/latticeai/core/agent/planning.py +0 -145
- package/latticeai/core/agent/recovery.py +0 -157
- package/latticeai/core/agent/runtime.py +0 -210
- package/latticeai/core/agent/verification.py +0 -231
- package/latticeai/core/agent_eval.py +0 -739
- package/latticeai/core/agent_helpers.py +0 -493
- package/latticeai/core/agent_profiles.py +0 -110
- package/latticeai/core/agent_prompts.py +0 -171
- package/latticeai/core/agent_registry.py +0 -232
- package/latticeai/core/agent_state.py +0 -41
- package/latticeai/core/agent_trace.py +0 -104
- package/latticeai/core/artifact_ledger.py +0 -109
- package/latticeai/core/audit.py +0 -260
- package/latticeai/core/builtin_hooks.py +0 -105
- package/latticeai/core/context_builder.py +0 -394
- package/latticeai/core/document_generator.py +0 -103
- package/latticeai/core/enterprise.py +0 -154
- package/latticeai/core/enterprise_admin.py +0 -158
- package/latticeai/core/file_generation/__init__.py +0 -115
- package/latticeai/core/file_generation/bundles.py +0 -76
- package/latticeai/core/file_generation/extraction.py +0 -154
- package/latticeai/core/file_generation/inference.py +0 -235
- package/latticeai/core/file_generation/orchestration.py +0 -152
- package/latticeai/core/file_generation/prompting.py +0 -117
- package/latticeai/core/file_generation/repair.py +0 -114
- package/latticeai/core/file_generation/sanitize.py +0 -61
- package/latticeai/core/file_generation/validation.py +0 -201
- package/latticeai/core/invitations.py +0 -132
- package/latticeai/core/legacy_compatibility.py +0 -243
- package/latticeai/core/logging_safety.py +0 -46
- package/latticeai/core/marketplace.py +0 -293
- package/latticeai/core/mcp_catalog.py +0 -452
- package/latticeai/core/mcp_registry.py +0 -506
- package/latticeai/core/network_boundary.py +0 -168
- package/latticeai/core/oidc.py +0 -208
- package/latticeai/core/plugins.py +0 -432
- package/latticeai/core/product_hardening.py +0 -218
- package/latticeai/core/project_sessions.py +0 -337
- package/latticeai/core/realtime.py +0 -238
- package/latticeai/core/run_explain.py +0 -426
- package/latticeai/core/run_store.py +0 -252
- package/latticeai/core/timezones.py +0 -80
- package/latticeai/core/workspace_computer_memory.py +0 -84
- package/latticeai/core/workspace_graph_trace.py +0 -155
- package/latticeai/core/workspace_indexing.py +0 -102
- package/latticeai/core/workspace_memory.py +0 -77
- package/latticeai/core/workspace_onboarding.py +0 -104
- package/latticeai/core/workspace_os.py +0 -978
- package/latticeai/core/workspace_os_constants.py +0 -126
- package/latticeai/core/workspace_os_state.py +0 -180
- package/latticeai/core/workspace_os_utils.py +0 -103
- package/latticeai/core/workspace_permissions.py +0 -101
- package/latticeai/core/workspace_plugins.py +0 -97
- package/latticeai/core/workspace_relationships.py +0 -99
- package/latticeai/core/workspace_reorganization.py +0 -335
- package/latticeai/core/workspace_review_items.py +0 -112
- package/latticeai/core/workspace_runs.py +0 -726
- package/latticeai/core/workspace_skills.py +0 -109
- package/latticeai/core/workspace_snapshots.py +0 -198
- package/latticeai/core/workspace_timeline.py +0 -110
- package/latticeai/integrations/__init__.py +0 -0
- package/latticeai/integrations/telegram_bot/__init__.py +0 -123
- package/latticeai/integrations/telegram_bot/__main__.py +0 -17
- package/latticeai/integrations/telegram_bot/config.py +0 -86
- package/latticeai/integrations/telegram_bot/dispatch.py +0 -311
- package/latticeai/integrations/telegram_bot/flows.py +0 -478
- package/latticeai/integrations/telegram_bot/helpers.py +0 -322
- package/latticeai/integrations/telegram_bot/screens.py +0 -394
- package/latticeai/runtime/audit_runtime.py +0 -76
- package/latticeai/runtime/automation_runtime.py +0 -81
- package/latticeai/runtime/chat_wiring.py +0 -141
- package/latticeai/runtime/context_runtime.py +0 -66
- package/latticeai/runtime/feature_toggle_wiring.py +0 -157
- package/latticeai/runtime/history_runtime.py +0 -163
- package/latticeai/runtime/history_writer.py +0 -138
- package/latticeai/runtime/hooks_runtime.py +0 -77
- package/latticeai/runtime/model_wiring.py +0 -68
- package/latticeai/runtime/namespace_runtime.py +0 -163
- package/latticeai/runtime/network_boundary_wiring.py +0 -117
- package/latticeai/runtime/network_config_runtime.py +0 -56
- package/latticeai/runtime/permission_mode_wiring.py +0 -112
- package/latticeai/runtime/persistence_runtime.py +0 -159
- package/latticeai/runtime/platform_runtime_wiring.py +0 -89
- package/latticeai/runtime/review_wiring.py +0 -42
- package/latticeai/runtime/router_registration.py +0 -693
- package/latticeai/runtime/service_singletons.py +0 -55
- package/latticeai/runtime/sso_config_runtime.py +0 -128
- package/latticeai/runtime/user_key_runtime.py +0 -106
- package/latticeai/runtime/web_runtime.py +0 -92
- package/latticeai/server_app.py +0 -51
- package/latticeai/services/app_context.py +0 -130
- package/latticeai/services/automation_execution.py +0 -266
- package/latticeai/services/automation_intelligence.py +0 -614
- package/latticeai/services/brain_automation.py +0 -191
- package/latticeai/services/brain_intelligence/__init__.py +0 -58
- package/latticeai/services/brain_intelligence/_contract.py +0 -71
- package/latticeai/services/brain_intelligence/consistency.py +0 -193
- package/latticeai/services/brain_intelligence/constants.py +0 -47
- package/latticeai/services/brain_intelligence/digest.py +0 -258
- package/latticeai/services/brain_intelligence/health.py +0 -331
- package/latticeai/services/brain_intelligence/proposals.py +0 -259
- package/latticeai/services/brain_intelligence/sampling.py +0 -84
- package/latticeai/services/brain_intelligence/service.py +0 -48
- package/latticeai/services/change_proposals.py +0 -471
- package/latticeai/services/chat_service.py +0 -243
- package/latticeai/services/chronicle.py +0 -555
- package/latticeai/services/cloud_egress_audit.py +0 -85
- package/latticeai/services/cloud_extraction.py +0 -129
- package/latticeai/services/cloud_streaming.py +0 -268
- package/latticeai/services/cloud_token_guard.py +0 -84
- package/latticeai/services/command_center.py +0 -548
- package/latticeai/services/evidence_actions.py +0 -258
- package/latticeai/services/feature_toggles.py +0 -502
- package/latticeai/services/folder_watch.py +0 -520
- package/latticeai/services/funnel_metrics.py +0 -307
- package/latticeai/services/hybrid_chat.py +0 -316
- package/latticeai/services/hybrid_context.py +0 -228
- package/latticeai/services/hybrid_policy.py +0 -129
- package/latticeai/services/interop_bridges.py +0 -978
- package/latticeai/services/local_knowledge.py +0 -465
- package/latticeai/services/memory_service/__init__.py +0 -52
- package/latticeai/services/memory_service/_contract.py +0 -100
- package/latticeai/services/memory_service/brief.py +0 -431
- package/latticeai/services/memory_service/constants.py +0 -57
- package/latticeai/services/memory_service/maintenance.py +0 -138
- package/latticeai/services/memory_service/manager.py +0 -186
- package/latticeai/services/memory_service/proof.py +0 -136
- package/latticeai/services/memory_service/recall.py +0 -225
- package/latticeai/services/memory_service/service.py +0 -48
- package/latticeai/services/memory_service/stores.py +0 -110
- package/latticeai/services/mode_store.py +0 -132
- package/latticeai/services/model_recommendation.py +0 -224
- package/latticeai/services/model_runtime/cloud.py +0 -87
- package/latticeai/services/network_boundary_service.py +0 -117
- package/latticeai/services/obsidian_bridge.py +0 -609
- package/latticeai/services/openai_compatible_adapter.py +0 -101
- package/latticeai/services/permission_mode_service.py +0 -122
- package/latticeai/services/platform_runtime.py +0 -366
- package/latticeai/services/review_queue.py +0 -380
- package/latticeai/services/router_context.py +0 -59
- package/latticeai/services/run_executor.py +0 -387
- package/latticeai/services/self_model_service.py +0 -171
- package/latticeai/services/setup_detection.py +0 -147
- package/latticeai/services/triggers.py +0 -378
- package/latticeai/services/upload_service.py +0 -172
- package/latticeai/services/workspace_service.py +0 -165
- package/latticeai/setup/__init__.py +0 -25
- package/latticeai/setup/auto_setup.py +0 -846
- package/latticeai/setup/demo_corpus.py +0 -98
- package/latticeai/setup/wizard/__init__.py +0 -126
- package/latticeai/setup/wizard/catalog.py +0 -175
- package/latticeai/setup/wizard/detect.py +0 -302
- package/latticeai/setup/wizard/install.py +0 -348
- package/latticeai/setup/wizard/paths.py +0 -165
- package/latticeai/setup/wizard/plans.py +0 -74
- package/latticeai/setup/wizard/recommend.py +0 -320
- package/scripts/bench_agent_smoke.py +0 -409
- package/scripts/bench_models.py +0 -540
- package/scripts/bench_vector_index.py +0 -295
- package/scripts/funnel_soft_gate.py +0 -192
- package/scripts/generate_agent_loop_fixtures.py +0 -994
- package/scripts/generate_rust_parity_fixtures.py +0 -908
- package/scripts/migrate_brain_storage.py +0 -57
- package/scripts/parity_fixture_corpus_context.py +0 -162
- package/scripts/parity_fixture_corpus_docgen.py +0 -341
- package/scripts/profile_kg.py +0 -355
- package/server.py +0 -30
- package/static/app/assets/AdminConsole-cf4npybT.js +0 -1
- package/static/app/assets/Capture-DiQ219jW.js +0 -1
- package/static/app/assets/System-CieofHQa.js +0 -1
- package/static/app/assets/arrow-left-kfsrk0mv.js +0 -1
- package/static/app/assets/circle-check-qqLug9nU.js +0 -1
- package/static/app/assets/index-DxmOfNRi.css +0 -2
- package/static/app/assets/primitives-SNp0LRJz.js +0 -1
- package/static/app/assets/search-BcHqkjoy.js +0 -1
|
@@ -1,978 +0,0 @@
|
|
|
1
|
-
"""Interop bridges — other people's formats, through the one ingestion gate.
|
|
2
|
-
|
|
3
|
-
Through 11.1.0 Obsidian was the only external source with a bridge, and the
|
|
4
|
-
release said so plainly: Notion, email, calendar, and Git were *scoped out*
|
|
5
|
-
rather than stubbed. This module closes that gap, and it does it the same way
|
|
6
|
-
:mod:`latticeai.services.obsidian_bridge` does — by refusing to open a second
|
|
7
|
-
door into the graph.
|
|
8
|
-
|
|
9
|
-
Every bridge here:
|
|
10
|
-
|
|
11
|
-
* reads **local files the user already owns** (a Notion export, an ``.eml`` on
|
|
12
|
-
disk, a repository path). Nothing calls a vendor API, nothing needs a token,
|
|
13
|
-
and nothing leaves the machine;
|
|
14
|
-
* pushes every item through :meth:`IngestionPipeline.ingest`, so content
|
|
15
|
-
hashing, hooks, provenance, extraction quality and workspace scoping are the
|
|
16
|
-
ones the rest of the product already has;
|
|
17
|
-
* supports ``dry_run``, which reports exactly what a real run would touch and
|
|
18
|
-
writes nothing;
|
|
19
|
-
* reports what it could **not** do — an unresolvable link, a missing decoder, a
|
|
20
|
-
file it could not read — instead of quietly dropping it.
|
|
21
|
-
|
|
22
|
-
What is still out of scope, stated rather than implied: **system integration**.
|
|
23
|
-
There is no macOS Calendar / Mail permission dance, no IMAP, no Google
|
|
24
|
-
Calendar, no Notion API. Those need credentials and background sync, and the
|
|
25
|
-
honest version of this release is "point me at files you exported".
|
|
26
|
-
"""
|
|
27
|
-
|
|
28
|
-
from __future__ import annotations
|
|
29
|
-
|
|
30
|
-
import email
|
|
31
|
-
import email.policy
|
|
32
|
-
import json
|
|
33
|
-
import os
|
|
34
|
-
import re
|
|
35
|
-
import shutil
|
|
36
|
-
import subprocess # noqa: S404 — one fixed binary, argv list, never a shell
|
|
37
|
-
import tempfile
|
|
38
|
-
import zipfile
|
|
39
|
-
from dataclasses import dataclass, field
|
|
40
|
-
from pathlib import Path, PurePath
|
|
41
|
-
from typing import Any, Dict, List, Optional, Tuple
|
|
42
|
-
from urllib.parse import unquote
|
|
43
|
-
|
|
44
|
-
from lattice_brain.graph.ingest import _scoped_slug_id
|
|
45
|
-
from lattice_brain.ingestion import IngestionItem
|
|
46
|
-
|
|
47
|
-
# ── shared vocabulary ────────────────────────────────────────────────────────
|
|
48
|
-
SOURCE_NOTION = "notion"
|
|
49
|
-
SOURCE_GIT = "git_commit"
|
|
50
|
-
SOURCE_EMAIL = "email"
|
|
51
|
-
SOURCE_CALENDAR = "calendar_event"
|
|
52
|
-
#: Every bridge in this module, for a status surface that wants to list them.
|
|
53
|
-
BRIDGE_SOURCE_TYPES = (SOURCE_NOTION, SOURCE_GIT, SOURCE_EMAIL, SOURCE_CALENDAR)
|
|
54
|
-
|
|
55
|
-
LINK_RELATION = "REFERENCES"
|
|
56
|
-
TAG_RELATION = "TAGGED_AS"
|
|
57
|
-
TOPIC_NODE_TYPE = "Topic"
|
|
58
|
-
|
|
59
|
-
ERROR_REPORT_CAP = 25
|
|
60
|
-
UNRESOLVED_REPORT_CAP = 50
|
|
61
|
-
DEFAULT_MAX_ITEMS = 2000
|
|
62
|
-
DEFAULT_MAX_FILE_BYTES = 2_000_000
|
|
63
|
-
DEFAULT_GIT_COMMITS = 500
|
|
64
|
-
|
|
65
|
-
GRAPH_DISABLED_DETAIL = "Knowledge Graph ingestion is disabled (LATTICEAI_ENABLE_GRAPH)."
|
|
66
|
-
|
|
67
|
-
|
|
68
|
-
def record_error(errors: List[Dict[str, Any]], entry: Dict[str, Any]) -> None:
|
|
69
|
-
"""Append one failure to a capped report list.
|
|
70
|
-
|
|
71
|
-
The *count* of failures is always exact; this list is a sample, so one
|
|
72
|
-
unreadable directory cannot turn a summary into a megabyte of paths. One
|
|
73
|
-
helper rather than three copies of the same ``if len(...) <`` check.
|
|
74
|
-
"""
|
|
75
|
-
if len(errors) < ERROR_REPORT_CAP:
|
|
76
|
-
errors.append(entry)
|
|
77
|
-
|
|
78
|
-
|
|
79
|
-
def edge_row(
|
|
80
|
-
from_id: str, to_id: str, relation: str, metadata: Dict[str, Any]
|
|
81
|
-
) -> Dict[str, Any]:
|
|
82
|
-
"""One ``import_graph_data`` edge row — the shape every bridge writes.
|
|
83
|
-
|
|
84
|
-
Shared with :mod:`~latticeai.services.obsidian_bridge` on purpose: two
|
|
85
|
-
copies of "what an edge row looks like" is two places to forget the
|
|
86
|
-
``metadata_json`` encoding.
|
|
87
|
-
"""
|
|
88
|
-
return {
|
|
89
|
-
"from_node": from_id,
|
|
90
|
-
"to_node": to_id,
|
|
91
|
-
"type": relation,
|
|
92
|
-
"weight": 1.0,
|
|
93
|
-
"metadata_json": json.dumps(metadata, ensure_ascii=False),
|
|
94
|
-
}
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
def topic_row(
|
|
98
|
-
topic_id: str, label: str, *, summary: str, metadata: Dict[str, Any]
|
|
99
|
-
) -> Dict[str, Any]:
|
|
100
|
-
"""One ``Topic`` node row, identified by the store's own scoped slug."""
|
|
101
|
-
return {
|
|
102
|
-
"id": topic_id,
|
|
103
|
-
"type": TOPIC_NODE_TYPE,
|
|
104
|
-
"title": label,
|
|
105
|
-
"summary": summary,
|
|
106
|
-
"metadata_json": json.dumps(metadata, ensure_ascii=False),
|
|
107
|
-
"raw_json": "{}",
|
|
108
|
-
}
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
@dataclass
|
|
112
|
-
class BridgeItem:
|
|
113
|
-
"""One thing a bridge found, before it reaches the pipeline."""
|
|
114
|
-
|
|
115
|
-
key: str
|
|
116
|
-
item: IngestionItem
|
|
117
|
-
#: Keys of other items this one points at (resolved after the full scan).
|
|
118
|
-
links: List[str] = field(default_factory=list)
|
|
119
|
-
#: Free-text labels that become ``Topic`` nodes (file paths, calendars…).
|
|
120
|
-
topics: List[str] = field(default_factory=list)
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
class InteropBridge:
|
|
124
|
-
"""Shared skeleton: scan → (dry_run?) → ingest → wire structure → report.
|
|
125
|
-
|
|
126
|
-
Subclasses own exactly one thing — how to turn a local path into
|
|
127
|
-
:class:`BridgeItem` values. Everything after that (the pipeline gate, the
|
|
128
|
-
counters, the edge writes, the honest failure list) is identical across
|
|
129
|
-
sources, which is the whole reason a second bridge did not become a second
|
|
130
|
-
ingestion path.
|
|
131
|
-
"""
|
|
132
|
-
|
|
133
|
-
source_type = "interop"
|
|
134
|
-
label = "interop source"
|
|
135
|
-
|
|
136
|
-
def __init__(
|
|
137
|
-
self,
|
|
138
|
-
*,
|
|
139
|
-
pipeline: Any,
|
|
140
|
-
knowledge_graph: Any = None,
|
|
141
|
-
max_items: int = DEFAULT_MAX_ITEMS,
|
|
142
|
-
max_file_bytes: int = DEFAULT_MAX_FILE_BYTES,
|
|
143
|
-
) -> None:
|
|
144
|
-
self._pipeline = pipeline
|
|
145
|
-
self._kg = knowledge_graph
|
|
146
|
-
self._max_items = max(1, int(max_items))
|
|
147
|
-
self._max_file_bytes = max(1, int(max_file_bytes))
|
|
148
|
-
|
|
149
|
-
def available(self) -> bool:
|
|
150
|
-
return self._pipeline is not None and bool(self._pipeline.available())
|
|
151
|
-
|
|
152
|
-
# ── subclass seam ────────────────────────────────────────────────────────
|
|
153
|
-
def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
|
|
154
|
-
"""``{"status", "target", "items": [BridgeItem], "errors": [...] , …}``."""
|
|
155
|
-
raise NotImplementedError # pragma: no cover - abstract seam
|
|
156
|
-
|
|
157
|
-
# ── the one public entry point ───────────────────────────────────────────
|
|
158
|
-
def sync(
|
|
159
|
-
self,
|
|
160
|
-
target: Any,
|
|
161
|
-
*,
|
|
162
|
-
owner: Optional[str] = None,
|
|
163
|
-
workspace_id: Optional[str] = None,
|
|
164
|
-
user_email: Optional[str] = None,
|
|
165
|
-
dry_run: bool = False,
|
|
166
|
-
**options: Any,
|
|
167
|
-
) -> Dict[str, Any]:
|
|
168
|
-
"""Ingest everything at ``target`` and wire whatever structure it has."""
|
|
169
|
-
if not self.available():
|
|
170
|
-
return {
|
|
171
|
-
"status": "unavailable",
|
|
172
|
-
"source": self.source_type,
|
|
173
|
-
"target": str(target),
|
|
174
|
-
"detail": GRAPH_DISABLED_DETAIL,
|
|
175
|
-
}
|
|
176
|
-
scan = self.scan(target, **options)
|
|
177
|
-
summary: Dict[str, Any] = {
|
|
178
|
-
"status": scan.get("status", "ok"),
|
|
179
|
-
"source": self.source_type,
|
|
180
|
-
"target": scan.get("target", str(target)),
|
|
181
|
-
"dry_run": bool(dry_run),
|
|
182
|
-
"scanned": int(scan.get("scanned") or 0),
|
|
183
|
-
"items": len(scan.get("items") or []),
|
|
184
|
-
"ingested": 0,
|
|
185
|
-
"duplicate": 0,
|
|
186
|
-
"failed": 0,
|
|
187
|
-
"truncated": bool(scan.get("truncated")),
|
|
188
|
-
"skipped": dict(scan.get("skipped") or {}),
|
|
189
|
-
"links": {
|
|
190
|
-
"resolved": 0,
|
|
191
|
-
"written": 0,
|
|
192
|
-
"unresolved_count": len(scan.get("unresolved") or []),
|
|
193
|
-
"unresolved": list(scan.get("unresolved") or [])[:UNRESOLVED_REPORT_CAP],
|
|
194
|
-
},
|
|
195
|
-
"edges": {"status": "none", "references": 0, "topics": 0, "detail": None},
|
|
196
|
-
"errors": list(scan.get("errors") or []),
|
|
197
|
-
}
|
|
198
|
-
if scan.get("status") != "ok":
|
|
199
|
-
summary["status"] = "failed"
|
|
200
|
-
summary["detail"] = scan.get("detail")
|
|
201
|
-
return summary
|
|
202
|
-
items: List[BridgeItem] = list(scan.get("items") or [])
|
|
203
|
-
known = {entry.key for entry in items}
|
|
204
|
-
resolved = {
|
|
205
|
-
entry.key: [link for link in entry.links if link in known and link != entry.key]
|
|
206
|
-
for entry in items
|
|
207
|
-
}
|
|
208
|
-
summary["links"]["resolved"] = sum(len(v) for v in resolved.values())
|
|
209
|
-
summary["topics"] = len({topic for entry in items for topic in entry.topics})
|
|
210
|
-
if dry_run:
|
|
211
|
-
summary["status"] = "dry_run"
|
|
212
|
-
return summary
|
|
213
|
-
|
|
214
|
-
node_ids = self._ingest_items(items, summary=summary, user_email=user_email or owner)
|
|
215
|
-
self._write_structure(
|
|
216
|
-
items,
|
|
217
|
-
node_ids=node_ids,
|
|
218
|
-
resolved=resolved,
|
|
219
|
-
owner=owner,
|
|
220
|
-
workspace_id=workspace_id,
|
|
221
|
-
summary=summary,
|
|
222
|
-
)
|
|
223
|
-
if summary["failed"] or summary["edges"]["status"] == "failed":
|
|
224
|
-
summary["status"] = "partial"
|
|
225
|
-
return summary
|
|
226
|
-
|
|
227
|
-
# ── internals ────────────────────────────────────────────────────────────
|
|
228
|
-
def _ingest_items(
|
|
229
|
-
self,
|
|
230
|
-
items: List[BridgeItem],
|
|
231
|
-
*,
|
|
232
|
-
summary: Dict[str, Any],
|
|
233
|
-
user_email: Optional[str],
|
|
234
|
-
) -> Dict[str, str]:
|
|
235
|
-
node_ids: Dict[str, str] = {}
|
|
236
|
-
errors: List[Dict[str, Any]] = summary["errors"]
|
|
237
|
-
for entry in items:
|
|
238
|
-
result = self._pipeline.ingest(entry.item, user_email=user_email)
|
|
239
|
-
if result.status != "ok":
|
|
240
|
-
summary["failed"] += 1
|
|
241
|
-
record_error(errors, {
|
|
242
|
-
"key": entry.key,
|
|
243
|
-
"status": result.status,
|
|
244
|
-
"detail": result.detail,
|
|
245
|
-
})
|
|
246
|
-
continue
|
|
247
|
-
if result.duplicate:
|
|
248
|
-
summary["duplicate"] += 1
|
|
249
|
-
else:
|
|
250
|
-
summary["ingested"] += 1
|
|
251
|
-
if result.node_id:
|
|
252
|
-
node_ids[entry.key] = result.node_id
|
|
253
|
-
return node_ids
|
|
254
|
-
|
|
255
|
-
def _write_structure(
|
|
256
|
-
self,
|
|
257
|
-
items: List[BridgeItem],
|
|
258
|
-
*,
|
|
259
|
-
node_ids: Dict[str, str],
|
|
260
|
-
resolved: Dict[str, List[str]],
|
|
261
|
-
owner: Optional[str],
|
|
262
|
-
workspace_id: Optional[str],
|
|
263
|
-
summary: Dict[str, Any],
|
|
264
|
-
) -> None:
|
|
265
|
-
edges: List[Dict[str, Any]] = []
|
|
266
|
-
topics: Dict[str, Dict[str, Any]] = {}
|
|
267
|
-
for entry in items:
|
|
268
|
-
from_id = node_ids.get(entry.key)
|
|
269
|
-
if from_id is None:
|
|
270
|
-
continue
|
|
271
|
-
for target_key in resolved.get(entry.key, []):
|
|
272
|
-
to_id = node_ids.get(target_key)
|
|
273
|
-
if to_id is None:
|
|
274
|
-
continue
|
|
275
|
-
edges.append(edge_row(from_id, to_id, LINK_RELATION, {
|
|
276
|
-
"source": self.source_type,
|
|
277
|
-
"from": entry.key,
|
|
278
|
-
"to": target_key,
|
|
279
|
-
}))
|
|
280
|
-
for label in entry.topics:
|
|
281
|
-
topic_id = _scoped_slug_id("topic", label, workspace_id)
|
|
282
|
-
topics.setdefault(topic_id, topic_row(
|
|
283
|
-
topic_id,
|
|
284
|
-
label,
|
|
285
|
-
summary=f"{self.label}: {label}",
|
|
286
|
-
metadata={
|
|
287
|
-
"topic": label,
|
|
288
|
-
"source": self.source_type,
|
|
289
|
-
"owner": owner,
|
|
290
|
-
"workspace_id": workspace_id,
|
|
291
|
-
},
|
|
292
|
-
))
|
|
293
|
-
edges.append(edge_row(from_id, topic_id, TAG_RELATION, {
|
|
294
|
-
"source": self.source_type,
|
|
295
|
-
"topic": label,
|
|
296
|
-
}))
|
|
297
|
-
if not edges:
|
|
298
|
-
return
|
|
299
|
-
references = sum(1 for edge in edges if edge["type"] == LINK_RELATION)
|
|
300
|
-
if self._kg is None:
|
|
301
|
-
summary["edges"] = {
|
|
302
|
-
"status": "skipped",
|
|
303
|
-
"references": 0,
|
|
304
|
-
"topics": 0,
|
|
305
|
-
"detail": "no Knowledge Graph store is bound; items were ingested without relations",
|
|
306
|
-
}
|
|
307
|
-
return
|
|
308
|
-
try:
|
|
309
|
-
outcome = self._kg.import_graph_data(
|
|
310
|
-
{
|
|
311
|
-
"nodes": list(topics.values()),
|
|
312
|
-
"edges": edges,
|
|
313
|
-
"chunks": [],
|
|
314
|
-
"knowledge_sources": [],
|
|
315
|
-
"provenance": [],
|
|
316
|
-
},
|
|
317
|
-
mode="merge",
|
|
318
|
-
)
|
|
319
|
-
except Exception as exc: # noqa: BLE001 — items already landed; report, never crash
|
|
320
|
-
summary["edges"] = {
|
|
321
|
-
"status": "failed",
|
|
322
|
-
"references": 0,
|
|
323
|
-
"topics": 0,
|
|
324
|
-
"detail": f"relations could not be written: {exc}",
|
|
325
|
-
}
|
|
326
|
-
return
|
|
327
|
-
summary["links"]["written"] = references
|
|
328
|
-
summary["edges"] = {
|
|
329
|
-
"status": "written",
|
|
330
|
-
"references": references,
|
|
331
|
-
"topics": len(topics),
|
|
332
|
-
"detail": None,
|
|
333
|
-
"index": outcome.get("index"),
|
|
334
|
-
}
|
|
335
|
-
|
|
336
|
-
def _failed_scan(self, target: Any, detail: str) -> Dict[str, Any]:
|
|
337
|
-
return {"status": "failed", "target": str(target), "detail": detail, "items": []}
|
|
338
|
-
|
|
339
|
-
|
|
340
|
-
# ── Notion export ────────────────────────────────────────────────────────────
|
|
341
|
-
NOTION_EXTENSIONS = frozenset({".md", ".markdown", ".csv"})
|
|
342
|
-
#: Notion appends a 32-hex page id to every exported filename. Two exports of
|
|
343
|
-
#: the same page produce two different ids, so the id is stripped from the
|
|
344
|
-
#: *title* and kept in metadata rather than shown to a reader.
|
|
345
|
-
_NOTION_ID_RE = re.compile(r"^(?P<title>.*?)[ _-]?(?P<page_id>[0-9a-f]{32})$", re.IGNORECASE)
|
|
346
|
-
_MD_LINK_RE = re.compile(r"\[[^\]\n]{0,200}\]\(([^()\s]{1,400})\)")
|
|
347
|
-
_EXTERNAL_PREFIXES = ("http://", "https://", "mailto:", "notion://", "//")
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
def notion_title(stem: str) -> Tuple[str, Optional[str]]:
|
|
351
|
-
"""``("Roadmap", "1a2b…")`` from ``"Roadmap 1a2b…"`` — id split off, not lost."""
|
|
352
|
-
match = _NOTION_ID_RE.match(str(stem or "").strip())
|
|
353
|
-
if match is None:
|
|
354
|
-
return str(stem or "").strip(), None
|
|
355
|
-
title = match.group("title").strip()
|
|
356
|
-
page_id = match.group("page_id").lower()
|
|
357
|
-
return (title or page_id), page_id
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
def notion_key(relative_path: str) -> str:
|
|
361
|
-
"""Stable identity for one exported page: its path with the id stripped."""
|
|
362
|
-
path = PurePath(str(relative_path or "").replace("\\", "/"))
|
|
363
|
-
title, _ = notion_title(path.stem)
|
|
364
|
-
parent = "/".join(part for part in path.parent.parts if part not in (".", ""))
|
|
365
|
-
normalized = f"{parent}/{title}" if parent else title
|
|
366
|
-
return normalized.strip("/").lower()
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
def notion_links(body: str) -> List[str]:
|
|
370
|
-
"""Relative page links inside an exported page, in document order."""
|
|
371
|
-
found: List[str] = []
|
|
372
|
-
seen: set = set()
|
|
373
|
-
for match in _MD_LINK_RE.finditer(str(body or "")):
|
|
374
|
-
raw = unquote(match.group(1)).strip()
|
|
375
|
-
if not raw or raw.startswith(_EXTERNAL_PREFIXES):
|
|
376
|
-
continue
|
|
377
|
-
suffix = PurePath(raw).suffix.lower()
|
|
378
|
-
if suffix and suffix not in NOTION_EXTENSIONS:
|
|
379
|
-
continue
|
|
380
|
-
key = notion_key(raw)
|
|
381
|
-
if not key or key in seen:
|
|
382
|
-
continue
|
|
383
|
-
seen.add(key)
|
|
384
|
-
found.append(key)
|
|
385
|
-
return found
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
class NotionExportBridge(InteropBridge):
|
|
389
|
-
"""A Notion **export** (directory or ``.zip``) through the one gate.
|
|
390
|
-
|
|
391
|
-
Deliberately not the Notion API: an API bridge needs an integration token,
|
|
392
|
-
a network round trip per page, and a background sync to stay current — all
|
|
393
|
-
of which are the opposite of "local-first, opt-in, off by default". An
|
|
394
|
-
export is a folder the user already downloaded, and it contains the same
|
|
395
|
-
words.
|
|
396
|
-
"""
|
|
397
|
-
|
|
398
|
-
source_type = SOURCE_NOTION
|
|
399
|
-
label = "Notion page"
|
|
400
|
-
|
|
401
|
-
def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
|
|
402
|
-
report: Dict[str, Any] = {
|
|
403
|
-
"status": "ok",
|
|
404
|
-
"target": str(target),
|
|
405
|
-
"scanned": 0,
|
|
406
|
-
"items": [],
|
|
407
|
-
"skipped": {"empty": 0, "too_large": 0, "unreadable": 0},
|
|
408
|
-
"truncated": False,
|
|
409
|
-
"errors": [],
|
|
410
|
-
}
|
|
411
|
-
try:
|
|
412
|
-
root = Path(target).expanduser()
|
|
413
|
-
except TypeError:
|
|
414
|
-
return self._failed_scan(target, f"invalid export path: {target!r}")
|
|
415
|
-
if root.is_file() and root.suffix.lower() == ".zip":
|
|
416
|
-
return self._scan_zip(root, report)
|
|
417
|
-
if not root.is_dir():
|
|
418
|
-
return self._failed_scan(root, f"not a Notion export directory or zip: {root}")
|
|
419
|
-
report["target"] = str(root)
|
|
420
|
-
self._walk(root, report)
|
|
421
|
-
return report
|
|
422
|
-
|
|
423
|
-
def _scan_zip(self, archive: Path, report: Dict[str, Any]) -> Dict[str, Any]:
|
|
424
|
-
"""Read a ``.zip`` export by extracting it to a temp dir first.
|
|
425
|
-
|
|
426
|
-
A zip member is not a path the pipeline can hash and re-read later, so
|
|
427
|
-
the export is materialized once and the ordinary directory walk runs
|
|
428
|
-
over it. Unsafe member names are refused outright.
|
|
429
|
-
"""
|
|
430
|
-
try:
|
|
431
|
-
with zipfile.ZipFile(archive) as zf:
|
|
432
|
-
names = zf.namelist()
|
|
433
|
-
for name in names:
|
|
434
|
-
member = PurePath(name.replace("\\", "/"))
|
|
435
|
-
if member.is_absolute() or ".." in member.parts:
|
|
436
|
-
return self._failed_scan(
|
|
437
|
-
archive, f"export archive contains an unsafe path: {name}"
|
|
438
|
-
)
|
|
439
|
-
staging = Path(tempfile.mkdtemp(prefix="notion-export-"))
|
|
440
|
-
zf.extractall(staging)
|
|
441
|
-
except (zipfile.BadZipFile, OSError) as exc:
|
|
442
|
-
return self._failed_scan(archive, f"export archive could not be read: {exc}")
|
|
443
|
-
report["target"] = str(archive)
|
|
444
|
-
report["extracted_to"] = str(staging)
|
|
445
|
-
self._walk(staging, report)
|
|
446
|
-
return report
|
|
447
|
-
|
|
448
|
-
def _walk(self, root: Path, report: Dict[str, Any]) -> None:
|
|
449
|
-
items: List[BridgeItem] = report["items"]
|
|
450
|
-
skipped = report["skipped"]
|
|
451
|
-
errors: List[Dict[str, Any]] = report["errors"]
|
|
452
|
-
for dirpath, dirnames, filenames in os.walk(root):
|
|
453
|
-
dirnames[:] = sorted(name for name in dirnames if not name.startswith("."))
|
|
454
|
-
current = Path(dirpath)
|
|
455
|
-
for name in sorted(filenames):
|
|
456
|
-
if name.startswith(".") or Path(name).suffix.lower() not in NOTION_EXTENSIONS:
|
|
457
|
-
continue
|
|
458
|
-
report["scanned"] += 1
|
|
459
|
-
path = current / name
|
|
460
|
-
if len(items) >= self._max_items:
|
|
461
|
-
report["truncated"] = True
|
|
462
|
-
continue
|
|
463
|
-
try:
|
|
464
|
-
if path.stat().st_size > self._max_file_bytes:
|
|
465
|
-
skipped["too_large"] += 1
|
|
466
|
-
continue
|
|
467
|
-
text = path.read_text(encoding="utf-8", errors="ignore")
|
|
468
|
-
except OSError as exc:
|
|
469
|
-
skipped["unreadable"] += 1
|
|
470
|
-
record_error(errors, {"key": name, "status": "unreadable", "detail": str(exc)})
|
|
471
|
-
continue
|
|
472
|
-
if not text.strip():
|
|
473
|
-
skipped["empty"] += 1
|
|
474
|
-
continue
|
|
475
|
-
relative = path.relative_to(root).as_posix()
|
|
476
|
-
title, page_id = notion_title(path.stem)
|
|
477
|
-
items.append(BridgeItem(
|
|
478
|
-
key=notion_key(relative),
|
|
479
|
-
links=notion_links(text),
|
|
480
|
-
item=IngestionItem(
|
|
481
|
-
source_type=SOURCE_NOTION,
|
|
482
|
-
title=title,
|
|
483
|
-
text=text,
|
|
484
|
-
source_uri=str(path),
|
|
485
|
-
mime_type="text/markdown" if path.suffix.lower() != ".csv" else "text/csv",
|
|
486
|
-
metadata={
|
|
487
|
-
"relative_path": relative,
|
|
488
|
-
"notion_page_id": page_id,
|
|
489
|
-
"export_kind": "database" if path.suffix.lower() == ".csv" else "page",
|
|
490
|
-
},
|
|
491
|
-
),
|
|
492
|
-
))
|
|
493
|
-
|
|
494
|
-
|
|
495
|
-
# ── Git repository ───────────────────────────────────────────────────────────
|
|
496
|
-
GIT_BINARY = "git"
|
|
497
|
-
_GIT_RECORD_SEPARATOR = "\x00"
|
|
498
|
-
_GIT_FIELD_SEPARATOR = "\x1f"
|
|
499
|
-
_GIT_PRETTY = (
|
|
500
|
-
f"format:{_GIT_RECORD_SEPARATOR}%H{_GIT_FIELD_SEPARATOR}%an{_GIT_FIELD_SEPARATOR}"
|
|
501
|
-
f"%ae{_GIT_FIELD_SEPARATOR}%aI{_GIT_FIELD_SEPARATOR}%s{_GIT_FIELD_SEPARATOR}%b"
|
|
502
|
-
f"{_GIT_RECORD_SEPARATOR}"
|
|
503
|
-
)
|
|
504
|
-
GIT_UNAVAILABLE_DETAIL = (
|
|
505
|
-
"reading a repository's history needs git on this machine and none was "
|
|
506
|
-
"found; nothing was ingested"
|
|
507
|
-
)
|
|
508
|
-
GIT_TIMEOUT_SECONDS = 60
|
|
509
|
-
|
|
510
|
-
|
|
511
|
-
def _which_git() -> Optional[str]:
|
|
512
|
-
"""Absolute path to git, or ``None``. The one probe, seamed for tests."""
|
|
513
|
-
return shutil.which(GIT_BINARY)
|
|
514
|
-
|
|
515
|
-
|
|
516
|
-
def _run_git(binary: str, args: List[str], cwd: Path) -> Tuple[int, str]:
|
|
517
|
-
"""Run git with an argv list in a fixed directory — no shell, ever."""
|
|
518
|
-
completed = subprocess.run( # noqa: S603 — argv list, fixed binary, no shell
|
|
519
|
-
[binary, *args],
|
|
520
|
-
cwd=str(cwd),
|
|
521
|
-
capture_output=True,
|
|
522
|
-
text=True,
|
|
523
|
-
timeout=GIT_TIMEOUT_SECONDS,
|
|
524
|
-
check=False,
|
|
525
|
-
)
|
|
526
|
-
return int(completed.returncode), str(completed.stdout or "")
|
|
527
|
-
|
|
528
|
-
|
|
529
|
-
def parse_git_log(output: str) -> List[Dict[str, Any]]:
|
|
530
|
-
"""Parse the ``--pretty``/``--name-only`` stream into commit records.
|
|
531
|
-
|
|
532
|
-
The separators are NUL and unit-separator rather than newlines because a
|
|
533
|
-
commit body contains newlines and a filename may contain almost anything
|
|
534
|
-
else. Anything that does not parse into six fields is skipped rather than
|
|
535
|
-
guessed at.
|
|
536
|
-
"""
|
|
537
|
-
chunks = str(output or "").split(_GIT_RECORD_SEPARATOR)
|
|
538
|
-
commits: List[Dict[str, Any]] = []
|
|
539
|
-
index = 1
|
|
540
|
-
while index < len(chunks):
|
|
541
|
-
fields = chunks[index].split(_GIT_FIELD_SEPARATOR)
|
|
542
|
-
trailer = chunks[index + 1] if index + 1 < len(chunks) else ""
|
|
543
|
-
index += 2
|
|
544
|
-
if len(fields) != 6:
|
|
545
|
-
continue
|
|
546
|
-
sha, author, mail, when, subject, body = fields
|
|
547
|
-
files = [line.strip() for line in trailer.splitlines() if line.strip()]
|
|
548
|
-
commits.append({
|
|
549
|
-
"sha": sha.strip(),
|
|
550
|
-
"author": author.strip(),
|
|
551
|
-
"author_email": mail.strip(),
|
|
552
|
-
"date": when.strip(),
|
|
553
|
-
"subject": subject.strip(),
|
|
554
|
-
"body": body.strip(),
|
|
555
|
-
"files": files,
|
|
556
|
-
})
|
|
557
|
-
return commits
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
class GitHistoryBridge(InteropBridge):
|
|
561
|
-
"""A local repository's commit history as project memory.
|
|
562
|
-
|
|
563
|
-
One node per commit — message, author, date, and the files it touched —
|
|
564
|
-
with each file path joined as a ``Topic``, so "what changed in the auth
|
|
565
|
-
module" is a graph traversal rather than a shell command. The commit hash
|
|
566
|
-
rides in the source URI and the body, so re-running reports duplicates
|
|
567
|
-
instead of writing a second copy of the same commit.
|
|
568
|
-
|
|
569
|
-
Nothing is cloned or fetched: this reads a path the user approved, with
|
|
570
|
-
``git log``, and says so when git is not installed.
|
|
571
|
-
"""
|
|
572
|
-
|
|
573
|
-
source_type = SOURCE_GIT
|
|
574
|
-
label = "repository file"
|
|
575
|
-
|
|
576
|
-
def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
|
|
577
|
-
limit = max(1, int(options.get("max_commits") or DEFAULT_GIT_COMMITS))
|
|
578
|
-
report: Dict[str, Any] = {
|
|
579
|
-
"status": "ok",
|
|
580
|
-
"target": str(target),
|
|
581
|
-
"scanned": 0,
|
|
582
|
-
"items": [],
|
|
583
|
-
"truncated": False,
|
|
584
|
-
"errors": [],
|
|
585
|
-
}
|
|
586
|
-
try:
|
|
587
|
-
root = Path(target).expanduser()
|
|
588
|
-
except TypeError:
|
|
589
|
-
return self._failed_scan(target, f"invalid repository path: {target!r}")
|
|
590
|
-
if not root.is_dir():
|
|
591
|
-
return self._failed_scan(root, f"not a directory: {root}")
|
|
592
|
-
if not (root / ".git").exists():
|
|
593
|
-
return self._failed_scan(root, f"not a git repository (no .git): {root}")
|
|
594
|
-
binary = _which_git()
|
|
595
|
-
if binary is None:
|
|
596
|
-
return self._failed_scan(root, GIT_UNAVAILABLE_DETAIL)
|
|
597
|
-
args = [
|
|
598
|
-
"log", f"--max-count={limit}", "--name-only",
|
|
599
|
-
f"--pretty={_GIT_PRETTY}",
|
|
600
|
-
]
|
|
601
|
-
try:
|
|
602
|
-
code, output = _run_git(binary, args, root)
|
|
603
|
-
except Exception as exc: # noqa: BLE001 — a broken git is a state, not a crash
|
|
604
|
-
return self._failed_scan(root, f"git log failed: {exc}")
|
|
605
|
-
if code != 0:
|
|
606
|
-
return self._failed_scan(root, f"git log exited with status {code}")
|
|
607
|
-
report["target"] = str(root)
|
|
608
|
-
commits = parse_git_log(output)
|
|
609
|
-
report["scanned"] = len(commits)
|
|
610
|
-
items: List[BridgeItem] = report["items"]
|
|
611
|
-
for commit in commits:
|
|
612
|
-
if len(items) >= self._max_items:
|
|
613
|
-
report["truncated"] = True
|
|
614
|
-
break
|
|
615
|
-
items.append(self._commit_item(root, commit))
|
|
616
|
-
return report
|
|
617
|
-
|
|
618
|
-
def _commit_item(self, root: Path, commit: Dict[str, Any]) -> BridgeItem:
|
|
619
|
-
sha = commit["sha"]
|
|
620
|
-
files = commit["files"]
|
|
621
|
-
lines = [
|
|
622
|
-
f"commit {sha}",
|
|
623
|
-
f"작성자: {commit['author']} <{commit['author_email']}>",
|
|
624
|
-
f"시각: {commit['date']}",
|
|
625
|
-
"",
|
|
626
|
-
commit["subject"],
|
|
627
|
-
]
|
|
628
|
-
if commit["body"]:
|
|
629
|
-
lines.extend(["", commit["body"]])
|
|
630
|
-
if files:
|
|
631
|
-
lines.extend(["", "변경된 파일:", *[f"- {name}" for name in files]])
|
|
632
|
-
return BridgeItem(
|
|
633
|
-
key=sha,
|
|
634
|
-
topics=list(files),
|
|
635
|
-
item=IngestionItem(
|
|
636
|
-
source_type=SOURCE_GIT,
|
|
637
|
-
title=f"{commit['subject'] or sha[:12]} ({sha[:8]})",
|
|
638
|
-
text="\n".join(lines),
|
|
639
|
-
source_uri=f"git:{root}#{sha}",
|
|
640
|
-
mime_type="text/plain",
|
|
641
|
-
modified_at=commit["date"] or None,
|
|
642
|
-
metadata={
|
|
643
|
-
"repository": str(root),
|
|
644
|
-
"commit": sha,
|
|
645
|
-
"author": commit["author"],
|
|
646
|
-
"author_email": commit["author_email"],
|
|
647
|
-
"committed_at": commit["date"],
|
|
648
|
-
"files": files,
|
|
649
|
-
"file_count": len(files),
|
|
650
|
-
},
|
|
651
|
-
),
|
|
652
|
-
)
|
|
653
|
-
|
|
654
|
-
|
|
655
|
-
# ── email (.eml) and calendar (.ics) ─────────────────────────────────────────
|
|
656
|
-
EMAIL_EXTENSIONS = frozenset({".eml"})
|
|
657
|
-
CALENDAR_EXTENSIONS = frozenset({".ics"})
|
|
658
|
-
MAILBOX_EXTENSIONS = EMAIL_EXTENSIONS | CALENDAR_EXTENSIONS
|
|
659
|
-
_ICS_ESCAPES = (("\\n", "\n"), ("\\N", "\n"), ("\\,", ","), ("\\;", ";"), ("\\\\", "\\"))
|
|
660
|
-
|
|
661
|
-
|
|
662
|
-
def _unfold_ics(text: str) -> List[str]:
|
|
663
|
-
"""RFC 5545 line unfolding: a leading space continues the previous line."""
|
|
664
|
-
lines: List[str] = []
|
|
665
|
-
for raw in str(text or "").replace("\r\n", "\n").replace("\r", "\n").split("\n"):
|
|
666
|
-
if raw[:1] in (" ", "\t") and lines:
|
|
667
|
-
lines[-1] += raw[1:]
|
|
668
|
-
continue
|
|
669
|
-
lines.append(raw)
|
|
670
|
-
return lines
|
|
671
|
-
|
|
672
|
-
|
|
673
|
-
def _ics_value(raw: str) -> str:
|
|
674
|
-
value = raw
|
|
675
|
-
for token, replacement in _ICS_ESCAPES:
|
|
676
|
-
value = value.replace(token, replacement)
|
|
677
|
-
return value.strip()
|
|
678
|
-
|
|
679
|
-
|
|
680
|
-
def parse_ics(text: str) -> List[Dict[str, str]]:
|
|
681
|
-
"""Every ``VEVENT`` in an ``.ics`` file as a flat dict of its properties.
|
|
682
|
-
|
|
683
|
-
A deliberately small parser rather than a dependency: this needs five
|
|
684
|
-
properties out of a format whose grammar is line folding plus
|
|
685
|
-
``NAME;PARAM=…:VALUE``, and a calendar library in the ingest path would be
|
|
686
|
-
a permanent cost for that. Properties it does not understand are ignored,
|
|
687
|
-
never guessed at, and an unterminated block is dropped rather than
|
|
688
|
-
half-reported.
|
|
689
|
-
"""
|
|
690
|
-
events: List[Dict[str, str]] = []
|
|
691
|
-
current: Optional[Dict[str, str]] = None
|
|
692
|
-
for line in _unfold_ics(text):
|
|
693
|
-
stripped = line.strip()
|
|
694
|
-
if stripped.upper() == "BEGIN:VEVENT":
|
|
695
|
-
current = {}
|
|
696
|
-
continue
|
|
697
|
-
if stripped.upper() == "END:VEVENT":
|
|
698
|
-
# A stray END with no BEGIN closes nothing rather than inventing an
|
|
699
|
-
# empty event.
|
|
700
|
-
if current is not None:
|
|
701
|
-
events.append(current)
|
|
702
|
-
current = None
|
|
703
|
-
continue
|
|
704
|
-
if current is None or ":" not in stripped:
|
|
705
|
-
continue
|
|
706
|
-
name, _, value = stripped.partition(":")
|
|
707
|
-
key = name.split(";", 1)[0].strip().upper()
|
|
708
|
-
current[key] = _ics_value(value)
|
|
709
|
-
return events
|
|
710
|
-
|
|
711
|
-
|
|
712
|
-
def _preferred_part(message: Any, kind: str) -> Any:
|
|
713
|
-
"""One ``get_body`` lookup; a malformed message answers ``None``, not a raise."""
|
|
714
|
-
try:
|
|
715
|
-
return message.get_body(preferencelist=(kind,))
|
|
716
|
-
except Exception: # noqa: BLE001 — a malformed message is a state
|
|
717
|
-
return None
|
|
718
|
-
|
|
719
|
-
|
|
720
|
-
def email_body(message: Any) -> Tuple[str, str]:
|
|
721
|
-
"""``(body, status)`` for a parsed message — plain text only, honestly.
|
|
722
|
-
|
|
723
|
-
HTML-only mail yields an empty body and ``html_only``: turning markup into
|
|
724
|
-
prose is a job for the extraction layer the capture surfaces already own,
|
|
725
|
-
and a naive tag-strip here would store navigation chrome as if it were the
|
|
726
|
-
message.
|
|
727
|
-
"""
|
|
728
|
-
part = _preferred_part(message, "plain")
|
|
729
|
-
if part is None:
|
|
730
|
-
return "", "html_only" if _preferred_part(message, "html") else "empty"
|
|
731
|
-
try:
|
|
732
|
-
content = str(part.get_content() or "").strip()
|
|
733
|
-
except Exception as exc: # noqa: BLE001 — undecodable charset is a state
|
|
734
|
-
return "", f"undecodable: {exc}"
|
|
735
|
-
return content, "ok" if content else "empty"
|
|
736
|
-
|
|
737
|
-
|
|
738
|
-
class MailCalendarBridge(InteropBridge):
|
|
739
|
-
"""Local ``.eml`` messages and ``.ics`` calendars, through the one gate.
|
|
740
|
-
|
|
741
|
-
Both are file formats with standard-library (or five-line) parsers, which
|
|
742
|
-
is exactly why they are in this release and a live mailbox connection is
|
|
743
|
-
not: reading ``~/Mail`` needs an OS permission grant and a sync loop, and
|
|
744
|
-
neither belongs behind a feature that claims to only read what it was
|
|
745
|
-
pointed at. Point this at a folder of exported messages.
|
|
746
|
-
"""
|
|
747
|
-
|
|
748
|
-
source_type = SOURCE_EMAIL
|
|
749
|
-
label = "calendar"
|
|
750
|
-
|
|
751
|
-
def scan(self, target: Any, **options: Any) -> Dict[str, Any]:
|
|
752
|
-
report: Dict[str, Any] = {
|
|
753
|
-
"status": "ok",
|
|
754
|
-
"target": str(target),
|
|
755
|
-
"scanned": 0,
|
|
756
|
-
"items": [],
|
|
757
|
-
"skipped": {"empty": 0, "too_large": 0, "unreadable": 0},
|
|
758
|
-
"truncated": False,
|
|
759
|
-
"errors": [],
|
|
760
|
-
}
|
|
761
|
-
try:
|
|
762
|
-
root = Path(target).expanduser()
|
|
763
|
-
except TypeError:
|
|
764
|
-
return self._failed_scan(target, f"invalid path: {target!r}")
|
|
765
|
-
if root.is_file():
|
|
766
|
-
paths = [root] if root.suffix.lower() in MAILBOX_EXTENSIONS else []
|
|
767
|
-
if not paths:
|
|
768
|
-
return self._failed_scan(root, f"not an .eml or .ics file: {root}")
|
|
769
|
-
elif root.is_dir():
|
|
770
|
-
paths = sorted(
|
|
771
|
-
path for path in root.rglob("*")
|
|
772
|
-
if path.is_file()
|
|
773
|
-
and not path.name.startswith(".")
|
|
774
|
-
and path.suffix.lower() in MAILBOX_EXTENSIONS
|
|
775
|
-
)
|
|
776
|
-
else:
|
|
777
|
-
return self._failed_scan(root, f"no such file or folder: {root}")
|
|
778
|
-
report["target"] = str(root)
|
|
779
|
-
self._read_all(paths, report)
|
|
780
|
-
return report
|
|
781
|
-
|
|
782
|
-
def _read_all(self, paths: List[Path], report: Dict[str, Any]) -> None:
|
|
783
|
-
items: List[BridgeItem] = report["items"]
|
|
784
|
-
skipped = report["skipped"]
|
|
785
|
-
errors: List[Dict[str, Any]] = report["errors"]
|
|
786
|
-
for path in paths:
|
|
787
|
-
report["scanned"] += 1
|
|
788
|
-
if len(items) >= self._max_items:
|
|
789
|
-
report["truncated"] = True
|
|
790
|
-
continue
|
|
791
|
-
try:
|
|
792
|
-
if path.stat().st_size > self._max_file_bytes:
|
|
793
|
-
skipped["too_large"] += 1
|
|
794
|
-
continue
|
|
795
|
-
raw = path.read_bytes()
|
|
796
|
-
except OSError as exc:
|
|
797
|
-
skipped["unreadable"] += 1
|
|
798
|
-
record_error(errors, {"key": path.name, "status": "unreadable", "detail": str(exc)})
|
|
799
|
-
continue
|
|
800
|
-
produced = (
|
|
801
|
-
self._calendar_items(path, raw)
|
|
802
|
-
if path.suffix.lower() in CALENDAR_EXTENSIONS
|
|
803
|
-
else self._message_items(path, raw)
|
|
804
|
-
)
|
|
805
|
-
if not produced:
|
|
806
|
-
skipped["empty"] += 1
|
|
807
|
-
continue
|
|
808
|
-
items.extend(produced)
|
|
809
|
-
|
|
810
|
-
def _message_items(self, path: Path, raw: bytes) -> List[BridgeItem]:
|
|
811
|
-
try:
|
|
812
|
-
message = email.message_from_bytes(raw, policy=email.policy.default)
|
|
813
|
-
except Exception as exc: # noqa: BLE001 — an unparseable message is a state
|
|
814
|
-
return [self._unreadable_item(path, f"message could not be parsed: {exc}")]
|
|
815
|
-
body, status = email_body(message)
|
|
816
|
-
subject = str(message.get("Subject") or path.stem).strip() or path.stem
|
|
817
|
-
sender = str(message.get("From") or "").strip()
|
|
818
|
-
recipients = str(message.get("To") or "").strip()
|
|
819
|
-
sent_at = str(message.get("Date") or "").strip()
|
|
820
|
-
message_id = str(message.get("Message-ID") or "").strip()
|
|
821
|
-
header_block = "\n".join(
|
|
822
|
-
line for line in (
|
|
823
|
-
f"보낸 사람: {sender}" if sender else "",
|
|
824
|
-
f"받는 사람: {recipients}" if recipients else "",
|
|
825
|
-
f"보낸 시각: {sent_at}" if sent_at else "",
|
|
826
|
-
) if line
|
|
827
|
-
)
|
|
828
|
-
text = f"{header_block}\n\n{body}".strip() if body else (
|
|
829
|
-
f"{header_block}\n\n[본문 없음] 이 메일은 일반 텍스트 본문이 없어 "
|
|
830
|
-
"내용 검색은 되지 않습니다."
|
|
831
|
-
).strip()
|
|
832
|
-
return [BridgeItem(
|
|
833
|
-
key=message_id or str(path),
|
|
834
|
-
item=IngestionItem(
|
|
835
|
-
source_type=SOURCE_EMAIL,
|
|
836
|
-
title=subject,
|
|
837
|
-
text=text,
|
|
838
|
-
source_uri=str(path),
|
|
839
|
-
mime_type="message/rfc822",
|
|
840
|
-
modified_at=sent_at or None,
|
|
841
|
-
metadata={
|
|
842
|
-
"from": sender,
|
|
843
|
-
"to": recipients,
|
|
844
|
-
"sent_at": sent_at,
|
|
845
|
-
"message_id": message_id,
|
|
846
|
-
"body_status": status,
|
|
847
|
-
"searchable": bool(body),
|
|
848
|
-
},
|
|
849
|
-
),
|
|
850
|
-
)]
|
|
851
|
-
|
|
852
|
-
def _calendar_items(self, path: Path, raw: bytes) -> List[BridgeItem]:
|
|
853
|
-
events = parse_ics(raw.decode("utf-8", errors="ignore"))
|
|
854
|
-
produced: List[BridgeItem] = []
|
|
855
|
-
for index, event in enumerate(events):
|
|
856
|
-
title = event.get("SUMMARY") or f"{path.stem} 일정 {index + 1}"
|
|
857
|
-
uid = event.get("UID") or f"{path}#{index}"
|
|
858
|
-
lines = [title]
|
|
859
|
-
for label, key in (("시작", "DTSTART"), ("종료", "DTEND"), ("장소", "LOCATION")):
|
|
860
|
-
if event.get(key):
|
|
861
|
-
lines.append(f"{label}: {event[key]}")
|
|
862
|
-
if event.get("DESCRIPTION"):
|
|
863
|
-
lines.extend(["", event["DESCRIPTION"]])
|
|
864
|
-
produced.append(BridgeItem(
|
|
865
|
-
key=uid,
|
|
866
|
-
topics=[event["LOCATION"]] if event.get("LOCATION") else [],
|
|
867
|
-
item=IngestionItem(
|
|
868
|
-
source_type=SOURCE_CALENDAR,
|
|
869
|
-
title=title,
|
|
870
|
-
text="\n".join(lines),
|
|
871
|
-
source_uri=f"{path}#{uid}",
|
|
872
|
-
mime_type="text/calendar",
|
|
873
|
-
modified_at=event.get("DTSTART") or None,
|
|
874
|
-
metadata={
|
|
875
|
-
"calendar_file": str(path),
|
|
876
|
-
"uid": uid,
|
|
877
|
-
"starts_at": event.get("DTSTART", ""),
|
|
878
|
-
"ends_at": event.get("DTEND", ""),
|
|
879
|
-
"location": event.get("LOCATION", ""),
|
|
880
|
-
},
|
|
881
|
-
),
|
|
882
|
-
))
|
|
883
|
-
return produced
|
|
884
|
-
|
|
885
|
-
@staticmethod
|
|
886
|
-
def _unreadable_item(path: Path, detail: str) -> BridgeItem:
|
|
887
|
-
"""Keep the fact that a file existed, and say what went wrong with it."""
|
|
888
|
-
return BridgeItem(
|
|
889
|
-
key=str(path),
|
|
890
|
-
item=IngestionItem(
|
|
891
|
-
source_type=SOURCE_EMAIL,
|
|
892
|
-
title=path.stem,
|
|
893
|
-
text=f"[읽을 수 없는 메일] {path.name}\n{detail}",
|
|
894
|
-
source_uri=str(path),
|
|
895
|
-
mime_type="message/rfc822",
|
|
896
|
-
metadata={"body_status": "unreadable", "detail": detail, "searchable": False},
|
|
897
|
-
),
|
|
898
|
-
)
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
BRIDGES: Dict[str, Any] = {
|
|
902
|
-
SOURCE_NOTION: NotionExportBridge,
|
|
903
|
-
"git": GitHistoryBridge,
|
|
904
|
-
"mail": MailCalendarBridge,
|
|
905
|
-
}
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
def build_bridge(
|
|
909
|
-
kind: str, *, pipeline: Any, knowledge_graph: Any = None, **options: Any
|
|
910
|
-
) -> InteropBridge:
|
|
911
|
-
"""One named bridge, or a ``ValueError`` naming the ones that exist."""
|
|
912
|
-
factory = BRIDGES.get(str(kind or "").strip().lower())
|
|
913
|
-
if factory is None:
|
|
914
|
-
raise ValueError(
|
|
915
|
-
f"unknown interop source '{kind}'; available: {', '.join(sorted(BRIDGES))}"
|
|
916
|
-
)
|
|
917
|
-
return factory(pipeline=pipeline, knowledge_graph=knowledge_graph, **options)
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
def bridge_status() -> Dict[str, Any]:
|
|
921
|
-
"""What each bridge can do on *this* machine, right now.
|
|
922
|
-
|
|
923
|
-
Git is the only one with a runtime prerequisite, and it is reported rather
|
|
924
|
-
than discovered when someone tries to use it.
|
|
925
|
-
"""
|
|
926
|
-
return {
|
|
927
|
-
"sources": {
|
|
928
|
-
SOURCE_NOTION: {
|
|
929
|
-
"available": True,
|
|
930
|
-
"accepts": ["directory", ".zip"],
|
|
931
|
-
"detail": "a Notion export you downloaded — never the Notion API",
|
|
932
|
-
},
|
|
933
|
-
"git": {
|
|
934
|
-
"available": _which_git() is not None,
|
|
935
|
-
"accepts": ["repository directory"],
|
|
936
|
-
"detail": None if _which_git() else GIT_UNAVAILABLE_DETAIL,
|
|
937
|
-
},
|
|
938
|
-
"mail": {
|
|
939
|
-
"available": True,
|
|
940
|
-
"accepts": [".eml", ".ics", "a folder of either"],
|
|
941
|
-
"detail": (
|
|
942
|
-
"local files only; connecting a live mailbox or system "
|
|
943
|
-
"calendar is deliberately out of scope"
|
|
944
|
-
),
|
|
945
|
-
},
|
|
946
|
-
},
|
|
947
|
-
}
|
|
948
|
-
|
|
949
|
-
|
|
950
|
-
__all__ = [
|
|
951
|
-
"BRIDGES",
|
|
952
|
-
"BRIDGE_SOURCE_TYPES",
|
|
953
|
-
"CALENDAR_EXTENSIONS",
|
|
954
|
-
"DEFAULT_GIT_COMMITS",
|
|
955
|
-
"EMAIL_EXTENSIONS",
|
|
956
|
-
"GIT_UNAVAILABLE_DETAIL",
|
|
957
|
-
"NOTION_EXTENSIONS",
|
|
958
|
-
"SOURCE_CALENDAR",
|
|
959
|
-
"SOURCE_EMAIL",
|
|
960
|
-
"SOURCE_GIT",
|
|
961
|
-
"SOURCE_NOTION",
|
|
962
|
-
"BridgeItem",
|
|
963
|
-
"GitHistoryBridge",
|
|
964
|
-
"InteropBridge",
|
|
965
|
-
"MailCalendarBridge",
|
|
966
|
-
"NotionExportBridge",
|
|
967
|
-
"bridge_status",
|
|
968
|
-
"build_bridge",
|
|
969
|
-
"edge_row",
|
|
970
|
-
"email_body",
|
|
971
|
-
"notion_key",
|
|
972
|
-
"notion_links",
|
|
973
|
-
"notion_title",
|
|
974
|
-
"parse_git_log",
|
|
975
|
-
"record_error",
|
|
976
|
-
"parse_ics",
|
|
977
|
-
"topic_row",
|
|
978
|
-
]
|