ltcai 11.5.2 → 11.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +93 -148
- package/bin/ltcai.js +234 -24
- package/docs/CHANGELOG.md +119 -0
- package/docs/COMMUNITY_AND_PLUGINS.md +1 -1
- package/docs/DEVELOPMENT.md +8 -6
- package/docs/ENTERPRISE.md +2 -1
- package/docs/MULTI_AGENT_RUNTIME.md +12 -5
- package/docs/ONBOARDING.md +1 -1
- package/docs/OPERATIONS.md +6 -3
- package/docs/REALTIME_COLLABORATION.md +1 -1
- package/docs/TRUST_MODEL.md +1 -1
- package/docs/WHY_LATTICE.md +1 -1
- package/docs/WORKFLOW_DESIGNER.md +3 -2
- package/docs/kg-schema.md +1 -1
- package/docs/v11.6.0_ONE_DOOR_PLAN.md +170 -0
- package/lattice_brain/__init__.py +42 -79
- package/lattice_brain/graph/__init__.py +9 -24
- package/lattice_brain/graph/_kg_common/__init__.py +13 -24
- package/lattice_brain/ingestion/__init__.py +19 -58
- package/lattice_brain/ingestion/pipeline.py +20 -398
- package/lattice_brain/multimodal/__init__.py +7 -18
- package/lattice_brain/multimodal/images.py +6 -264
- package/lattice_brain/multimodal/video.py +8 -247
- package/lattice_brain/runtime/__init__.py +8 -79
- package/lattice_brain/runtime/hooks.py +22 -584
- package/latticeai/__init__.py +1 -1
- package/latticeai/api/agent_worker_seam.py +7 -80
- package/latticeai/api/health.py +6 -20
- package/latticeai/api/local_files.py +16 -613
- package/latticeai/api/models.py +12 -87
- package/latticeai/api/search.py +17 -255
- package/latticeai/api/tools.py +69 -750
- package/latticeai/api/voice_capture.py +8 -70
- package/latticeai/api/worker_compute.py +842 -0
- package/latticeai/api/worker_seams.py +216 -0
- package/latticeai/app_factory.py +23 -232
- package/latticeai/cli/entrypoint.py +36 -249
- package/latticeai/core/agent_permission.py +16 -91
- package/latticeai/core/messages.py +89 -531
- package/latticeai/runtime/access_runtime.py +0 -14
- package/latticeai/runtime/bootstrap.py +10 -21
- package/latticeai/runtime/brain_runtime.py +35 -43
- package/latticeai/runtime/build_phases/__init__.py +21 -37
- package/latticeai/runtime/build_phases/features.py +99 -389
- package/latticeai/runtime/build_phases/foundation.py +82 -392
- package/latticeai/runtime/build_phases/web.py +91 -408
- package/latticeai/runtime/build_phases/worker_profile.py +244 -0
- package/latticeai/runtime/lifespan_runtime.py +7 -16
- package/latticeai/runtime/platform_services_runtime.py +1 -15
- package/latticeai/runtime/runtime_context.py +13 -130
- package/latticeai/runtime/security_runtime.py +10 -103
- package/latticeai/services/architecture_readiness.py +82 -43
- package/latticeai/services/model_runtime/__init__.py +2 -11
- package/latticeai/services/model_runtime/service.py +4 -28
- package/latticeai/services/p_reinforce.py +10 -261
- package/latticeai/services/product_readiness.py +31 -38
- package/latticeai/services/search_service.py +37 -795
- package/latticeai/services/tool_dispatch.py +30 -386
- package/latticeai/services/voice_capture.py +13 -107
- package/latticeai/tools/__init__.py +22 -49
- package/latticeai/tools/commands.py +0 -163
- package/latticeai/tools/computer.py +0 -39
- package/latticeai/tools/documents.py +1 -134
- package/latticeai/tools/filesystem.py +1 -247
- package/latticeai/tools/knowledge.py +1 -52
- package/latticeai/tools/local_files.py +0 -20
- package/latticeai/worker_app.py +75 -0
- package/package.json +2 -3
- package/requirements.txt +0 -5
- package/scripts/agent_eval.py +16 -28
- package/scripts/brain_quality_eval.py +20 -183
- package/scripts/bump_version.py +5 -5
- package/scripts/check_current_release_docs.mjs +7 -4
- package/scripts/check_openapi_drift.mjs +10 -0
- package/scripts/check_server_i18n.mjs +2 -18
- package/scripts/compose_openapi.py +377 -0
- package/scripts/export_openapi.py +32 -4
- package/scripts/gen_messages_catalog_fixture.py +302 -0
- package/scripts/gen_openapi_fragments.py +365 -0
- package/scripts/gen_redact_fixture.py +196 -0
- package/scripts/gen_worker_allowlist_fixture.py +115 -0
- package/scripts/generate_agent_parity_fixtures.py +33 -14
- package/scripts/openapi_route_families.json +2161 -0
- package/scripts/release_screen_claims.json +54 -0
- package/scripts/run_integration_tests.mjs +134 -29
- package/scripts/run_sidecar_e2e.mjs +104 -11
- package/scripts/wheel_smoke.py +32 -22
- package/src-tauri/Cargo.lock +375 -9
- package/src-tauri/Cargo.toml +1 -1
- package/src-tauri/src/backend.rs +151 -45
- package/src-tauri/src/main.rs +18 -18
- package/src-tauri/src/topology.rs +58 -88
- package/src-tauri/tauri.conf.json +1 -2
- package/static/app/asset-manifest.json +41 -41
- package/static/app/assets/{Act-DcQizkl1.js → Act-BPcVAbOL.js} +1 -1
- package/static/app/assets/AdminConsole-Bw1ATQL0.js +1 -0
- package/static/app/assets/{Brain-3VCSHFcn.js → Brain-CT92Kos0.js} +2 -2
- package/static/app/assets/{BrainHome-Qm8eaztx.js → BrainHome-CFBkt1K_.js} +1 -1
- package/static/app/assets/{BrainSignals-DS9BtKOW.js → BrainSignals-ReLWF2H8.js} +1 -1
- package/static/app/assets/Capture-BsTokYkk.js +1 -0
- package/static/app/assets/{Chronicle-BGvuAchH.js → Chronicle-B6f0T9id.js} +1 -1
- package/static/app/assets/{CommandPalette-Bqhm0Urn.js → CommandPalette-CuvjTv1u.js} +1 -1
- package/static/app/assets/{Library-BV6NnF0a.js → Library-BGJbG9Hd.js} +1 -1
- package/static/app/assets/{LivingBrain-GzenJchP.js → LivingBrain-DGYK_Jsa.js} +1 -1
- package/static/app/assets/{ProductFlow-DEP6-vML.js → ProductFlow-DXBC6brE.js} +1 -1
- package/static/app/assets/{ReviewCard-CNZ7XjWG.js → ReviewCard-HXRle3qq.js} +2 -2
- package/static/app/assets/System-CMHSO9qM.js +1 -0
- package/static/app/assets/arrow-left-BfmkskWx.js +1 -0
- package/static/app/assets/{bot-B_K1Tdmw.js → bot-Cn8bWRuq.js} +1 -1
- package/static/app/assets/{brain-DWyaV1L1.js → brain-CQJberbE.js} +1 -1
- package/static/app/assets/{button-aTn4s84A.js → button-Ct9f2_oT.js} +1 -1
- package/static/app/assets/circle-check-DruOxB-4.js +1 -0
- package/static/app/assets/{circle-pause-xKgeGXkT.js → circle-pause-CmzC_apg.js} +1 -1
- package/static/app/assets/{circle-play-DkT6tYPX.js → circle-play-D8mW2aQ7.js} +1 -1
- package/static/app/assets/{cpu-85xYObUC.js → cpu-DZcdd0PZ.js} +1 -1
- package/static/app/assets/{download-B5Fm7YXo.js → download-bv1KEPGQ.js} +1 -1
- package/static/app/assets/{folder-open-kk2Xa52u.js → folder-open-d-Pip5gr.js} +1 -1
- package/static/app/assets/{hard-drive-DkA3zBW_.js → hard-drive-D20iavUb.js} +1 -1
- package/static/app/assets/index-D9x-kSNy.css +2 -0
- package/static/app/assets/{index-BMPdTmlY.js → index-Do83hDzJ.js} +3 -3
- package/static/app/assets/{input-B0nRf2jO.js → input-BLXVNmj1.js} +1 -1
- package/static/app/assets/{link-2-Dwb4gnTc.js → link-2-BPJOFlAy.js} +1 -1
- package/static/app/assets/{permissionCopy-CQDUBrOZ.js → permissionCopy-ChdJd493.js} +1 -1
- package/static/app/assets/primitives-Cv5tbZBY.js +1 -0
- package/static/app/assets/search-CT9aho2j.js +1 -0
- package/static/app/assets/{share-2-BsrxFglO.js → share-2-YNX_NtMU.js} +1 -1
- package/static/app/assets/{shield-alert-5BStfp2_.js → shield-alert-DuQ3zrVL.js} +1 -1
- package/static/app/assets/{textarea-Cg8IUA-k.js → textarea-DqwLnli4.js} +1 -1
- package/static/app/assets/{useFocusTrap-CYKvE46M.js → useFocusTrap-ZVI98jaW.js} +1 -1
- package/static/app/assets/{useMutation-CSn9t1op.js → useMutation-CVC4qv_D.js} +1 -1
- package/static/app/assets/{useQuery-CY2OI2uy.js → useQuery-C7BeG4HU.js} +1 -1
- package/static/app/assets/{utils-Ddol2RWD.js → utils-CiFtIdZq.js} +1 -1
- package/static/app/assets/{workspace-BqDwOz_p.js → workspace-DQz9vIId.js} +1 -1
- package/static/app/index.html +4 -4
- package/static/sw.js +1 -1
- package/lattice_brain/archive.py +0 -522
- package/lattice_brain/context.py +0 -325
- package/lattice_brain/conversations.py +0 -382
- package/lattice_brain/core.py +0 -82
- package/lattice_brain/graph/_kg_contract.py +0 -249
- package/lattice_brain/graph/curator.py +0 -676
- package/lattice_brain/graph/discovery.py +0 -595
- package/lattice_brain/graph/discovery_index/__init__.py +0 -35
- package/lattice_brain/graph/discovery_index/cleanup.py +0 -182
- package/lattice_brain/graph/discovery_index/extract.py +0 -137
- package/lattice_brain/graph/discovery_index/scan.py +0 -411
- package/lattice_brain/graph/discovery_index/upsert.py +0 -495
- package/lattice_brain/graph/documents.py +0 -380
- package/lattice_brain/graph/fusion.py +0 -395
- package/lattice_brain/graph/identity.py +0 -175
- package/lattice_brain/graph/image_vectors.py +0 -230
- package/lattice_brain/graph/ingest.py +0 -829
- package/lattice_brain/graph/network.py +0 -205
- package/lattice_brain/graph/proactive.py +0 -724
- package/lattice_brain/graph/projection/__init__.py +0 -42
- package/lattice_brain/graph/projection/curation.py +0 -500
- package/lattice_brain/graph/projection/v2_schema.py +0 -518
- package/lattice_brain/graph/provenance.py +0 -524
- package/lattice_brain/graph/rerank.py +0 -163
- package/lattice_brain/graph/retrieval/__init__.py +0 -54
- package/lattice_brain/graph/retrieval/context.py +0 -197
- package/lattice_brain/graph/retrieval/graph_view.py +0 -319
- package/lattice_brain/graph/retrieval/hybrid.py +0 -488
- package/lattice_brain/graph/retrieval/maintenance.py +0 -121
- package/lattice_brain/graph/retrieval/signals.py +0 -95
- package/lattice_brain/graph/retrieval_docgen.py +0 -253
- package/lattice_brain/graph/retrieval_policy.py +0 -180
- package/lattice_brain/graph/retrieval_reads.py +0 -769
- package/lattice_brain/graph/retrieval_vector/__init__.py +0 -42
- package/lattice_brain/graph/retrieval_vector/fingerprint.py +0 -97
- package/lattice_brain/graph/retrieval_vector/indexing.py +0 -347
- package/lattice_brain/graph/retrieval_vector/search.py +0 -560
- package/lattice_brain/graph/retrieval_vector/status.py +0 -374
- package/lattice_brain/graph/schema.py +0 -792
- package/lattice_brain/graph/store.py +0 -268
- package/lattice_brain/graph/vector_index/__init__.py +0 -85
- package/lattice_brain/graph/vector_index/base.py +0 -167
- package/lattice_brain/graph/vector_index/brute_force.py +0 -110
- package/lattice_brain/graph/vector_index/hnsw.py +0 -290
- package/lattice_brain/graph/vector_index/jobs.py +0 -287
- package/lattice_brain/graph/vector_index/quantized.py +0 -148
- package/lattice_brain/graph/vector_index/selector.py +0 -161
- package/lattice_brain/graph/write_master.py +0 -308
- package/lattice_brain/ingestion/_contract.py +0 -90
- package/lattice_brain/ingestion/folder_scan.py +0 -57
- package/lattice_brain/ingestion/folders.py +0 -258
- package/lattice_brain/ingestion/jobs_api.py +0 -107
- package/lattice_brain/ingestion/routing.py +0 -295
- package/lattice_brain/ingestion_jobs.py +0 -380
- package/lattice_brain/memory.py +0 -75
- package/lattice_brain/portability/__init__.py +0 -90
- package/lattice_brain/portability/_contract.py +0 -42
- package/lattice_brain/portability/backups.py +0 -338
- package/lattice_brain/portability/bundles.py +0 -136
- package/lattice_brain/portability/constants.py +0 -93
- package/lattice_brain/portability/fsops.py +0 -138
- package/lattice_brain/portability/service.py +0 -41
- package/lattice_brain/portability/sharing.py +0 -710
- package/lattice_brain/quality.py +0 -543
- package/lattice_brain/retrieval_benchmark_fixtures.py +0 -92
- package/lattice_brain/runtime/agent_runtime.py +0 -859
- package/lattice_brain/runtime/contracts.py +0 -460
- package/lattice_brain/runtime/multi_agent.py +0 -942
- package/lattice_brain/runtime/statuses.py +0 -10
- package/lattice_brain/sealed_box.py +0 -240
- package/lattice_brain/self_model.py +0 -675
- package/lattice_brain/sensitivity.py +0 -94
- package/lattice_brain/storage/__init__.py +0 -22
- package/lattice_brain/storage/base.py +0 -100
- package/lattice_brain/storage/docker.py +0 -105
- package/lattice_brain/storage/factory.py +0 -31
- package/lattice_brain/storage/migration.py +0 -191
- package/lattice_brain/storage/postgres.py +0 -123
- package/lattice_brain/storage/sqlite.py +0 -143
- package/lattice_brain/synthesis.py +0 -824
- package/lattice_brain/workflow.py +0 -497
- package/latticeai/api/admin.py +0 -471
- package/latticeai/api/agent_registry.py +0 -105
- package/latticeai/api/agents.py +0 -228
- package/latticeai/api/auth.py +0 -383
- package/latticeai/api/automation_intelligence.py +0 -401
- package/latticeai/api/brain_intelligence.py +0 -199
- package/latticeai/api/browser.py +0 -493
- package/latticeai/api/change_proposals.py +0 -89
- package/latticeai/api/chat.py +0 -572
- package/latticeai/api/chat_agent_http.py +0 -892
- package/latticeai/api/chat_contracts.py +0 -73
- package/latticeai/api/chat_documents.py +0 -276
- package/latticeai/api/chat_helpers.py +0 -460
- package/latticeai/api/chat_history.py +0 -101
- package/latticeai/api/chat_hybrid.py +0 -113
- package/latticeai/api/chat_intents.py +0 -646
- package/latticeai/api/chat_stream.py +0 -216
- package/latticeai/api/chronicle.py +0 -63
- package/latticeai/api/command_center.py +0 -51
- package/latticeai/api/computer_use.py +0 -474
- package/latticeai/api/evidence_actions.py +0 -48
- package/latticeai/api/features.py +0 -70
- package/latticeai/api/funnel_metrics.py +0 -31
- package/latticeai/api/garden.py +0 -34
- package/latticeai/api/hooks.py +0 -165
- package/latticeai/api/index_jobs.py +0 -145
- package/latticeai/api/invitations.py +0 -100
- package/latticeai/api/knowledge_graph.py +0 -536
- package/latticeai/api/marketplace.py +0 -105
- package/latticeai/api/mcp.py +0 -482
- package/latticeai/api/memory.py +0 -270
- package/latticeai/api/network.py +0 -81
- package/latticeai/api/network_boundary.py +0 -225
- package/latticeai/api/permission_mode.py +0 -61
- package/latticeai/api/permissions.py +0 -436
- package/latticeai/api/plugins.py +0 -126
- package/latticeai/api/portability.py +0 -391
- package/latticeai/api/project_sessions.py +0 -114
- package/latticeai/api/realtime.py +0 -118
- package/latticeai/api/review_queue.py +0 -364
- package/latticeai/api/security_dashboard.py +0 -604
- package/latticeai/api/setup.py +0 -319
- package/latticeai/api/static_routes.py +0 -354
- package/latticeai/api/ui_redirects.py +0 -26
- package/latticeai/api/workflow_designer.py +0 -394
- package/latticeai/api/workspace.py +0 -856
- package/latticeai/api/workspace_scope.py +0 -125
- package/latticeai/core/agent/__init__.py +0 -93
- package/latticeai/core/agent/_contract.py +0 -79
- package/latticeai/core/agent/context.py +0 -57
- package/latticeai/core/agent/deps.py +0 -125
- package/latticeai/core/agent/execution.py +0 -622
- package/latticeai/core/agent/planning.py +0 -145
- package/latticeai/core/agent/recovery.py +0 -157
- package/latticeai/core/agent/runtime.py +0 -210
- package/latticeai/core/agent/verification.py +0 -231
- package/latticeai/core/agent_eval.py +0 -739
- package/latticeai/core/agent_helpers.py +0 -493
- package/latticeai/core/agent_profiles.py +0 -110
- package/latticeai/core/agent_prompts.py +0 -171
- package/latticeai/core/agent_registry.py +0 -232
- package/latticeai/core/agent_state.py +0 -41
- package/latticeai/core/agent_trace.py +0 -104
- package/latticeai/core/artifact_ledger.py +0 -109
- package/latticeai/core/audit.py +0 -260
- package/latticeai/core/builtin_hooks.py +0 -105
- package/latticeai/core/context_builder.py +0 -394
- package/latticeai/core/document_generator.py +0 -103
- package/latticeai/core/enterprise.py +0 -154
- package/latticeai/core/enterprise_admin.py +0 -158
- package/latticeai/core/file_generation/__init__.py +0 -115
- package/latticeai/core/file_generation/bundles.py +0 -76
- package/latticeai/core/file_generation/extraction.py +0 -154
- package/latticeai/core/file_generation/inference.py +0 -235
- package/latticeai/core/file_generation/orchestration.py +0 -152
- package/latticeai/core/file_generation/prompting.py +0 -117
- package/latticeai/core/file_generation/repair.py +0 -114
- package/latticeai/core/file_generation/sanitize.py +0 -61
- package/latticeai/core/file_generation/validation.py +0 -201
- package/latticeai/core/invitations.py +0 -132
- package/latticeai/core/legacy_compatibility.py +0 -243
- package/latticeai/core/logging_safety.py +0 -46
- package/latticeai/core/marketplace.py +0 -293
- package/latticeai/core/mcp_catalog.py +0 -452
- package/latticeai/core/mcp_registry.py +0 -506
- package/latticeai/core/network_boundary.py +0 -168
- package/latticeai/core/oidc.py +0 -208
- package/latticeai/core/plugins.py +0 -432
- package/latticeai/core/product_hardening.py +0 -218
- package/latticeai/core/project_sessions.py +0 -337
- package/latticeai/core/realtime.py +0 -238
- package/latticeai/core/run_explain.py +0 -426
- package/latticeai/core/run_store.py +0 -252
- package/latticeai/core/timezones.py +0 -80
- package/latticeai/core/workspace_computer_memory.py +0 -84
- package/latticeai/core/workspace_graph_trace.py +0 -155
- package/latticeai/core/workspace_indexing.py +0 -102
- package/latticeai/core/workspace_memory.py +0 -77
- package/latticeai/core/workspace_onboarding.py +0 -104
- package/latticeai/core/workspace_os.py +0 -978
- package/latticeai/core/workspace_os_constants.py +0 -126
- package/latticeai/core/workspace_os_state.py +0 -180
- package/latticeai/core/workspace_os_utils.py +0 -103
- package/latticeai/core/workspace_permissions.py +0 -101
- package/latticeai/core/workspace_plugins.py +0 -97
- package/latticeai/core/workspace_relationships.py +0 -99
- package/latticeai/core/workspace_reorganization.py +0 -335
- package/latticeai/core/workspace_review_items.py +0 -112
- package/latticeai/core/workspace_runs.py +0 -726
- package/latticeai/core/workspace_skills.py +0 -109
- package/latticeai/core/workspace_snapshots.py +0 -198
- package/latticeai/core/workspace_timeline.py +0 -110
- package/latticeai/integrations/__init__.py +0 -0
- package/latticeai/integrations/telegram_bot/__init__.py +0 -123
- package/latticeai/integrations/telegram_bot/__main__.py +0 -17
- package/latticeai/integrations/telegram_bot/config.py +0 -86
- package/latticeai/integrations/telegram_bot/dispatch.py +0 -311
- package/latticeai/integrations/telegram_bot/flows.py +0 -478
- package/latticeai/integrations/telegram_bot/helpers.py +0 -322
- package/latticeai/integrations/telegram_bot/screens.py +0 -394
- package/latticeai/runtime/audit_runtime.py +0 -76
- package/latticeai/runtime/automation_runtime.py +0 -81
- package/latticeai/runtime/chat_wiring.py +0 -141
- package/latticeai/runtime/context_runtime.py +0 -66
- package/latticeai/runtime/feature_toggle_wiring.py +0 -157
- package/latticeai/runtime/history_runtime.py +0 -163
- package/latticeai/runtime/history_writer.py +0 -138
- package/latticeai/runtime/hooks_runtime.py +0 -77
- package/latticeai/runtime/model_wiring.py +0 -68
- package/latticeai/runtime/namespace_runtime.py +0 -163
- package/latticeai/runtime/network_boundary_wiring.py +0 -117
- package/latticeai/runtime/network_config_runtime.py +0 -56
- package/latticeai/runtime/permission_mode_wiring.py +0 -112
- package/latticeai/runtime/persistence_runtime.py +0 -159
- package/latticeai/runtime/platform_runtime_wiring.py +0 -89
- package/latticeai/runtime/review_wiring.py +0 -42
- package/latticeai/runtime/router_registration.py +0 -693
- package/latticeai/runtime/service_singletons.py +0 -55
- package/latticeai/runtime/sso_config_runtime.py +0 -128
- package/latticeai/runtime/user_key_runtime.py +0 -106
- package/latticeai/runtime/web_runtime.py +0 -92
- package/latticeai/server_app.py +0 -51
- package/latticeai/services/app_context.py +0 -130
- package/latticeai/services/automation_execution.py +0 -266
- package/latticeai/services/automation_intelligence.py +0 -614
- package/latticeai/services/brain_automation.py +0 -191
- package/latticeai/services/brain_intelligence/__init__.py +0 -58
- package/latticeai/services/brain_intelligence/_contract.py +0 -71
- package/latticeai/services/brain_intelligence/consistency.py +0 -193
- package/latticeai/services/brain_intelligence/constants.py +0 -47
- package/latticeai/services/brain_intelligence/digest.py +0 -258
- package/latticeai/services/brain_intelligence/health.py +0 -331
- package/latticeai/services/brain_intelligence/proposals.py +0 -259
- package/latticeai/services/brain_intelligence/sampling.py +0 -84
- package/latticeai/services/brain_intelligence/service.py +0 -48
- package/latticeai/services/change_proposals.py +0 -471
- package/latticeai/services/chat_service.py +0 -243
- package/latticeai/services/chronicle.py +0 -555
- package/latticeai/services/cloud_egress_audit.py +0 -85
- package/latticeai/services/cloud_extraction.py +0 -129
- package/latticeai/services/cloud_streaming.py +0 -268
- package/latticeai/services/cloud_token_guard.py +0 -84
- package/latticeai/services/command_center.py +0 -548
- package/latticeai/services/evidence_actions.py +0 -258
- package/latticeai/services/feature_toggles.py +0 -502
- package/latticeai/services/folder_watch.py +0 -520
- package/latticeai/services/funnel_metrics.py +0 -307
- package/latticeai/services/hybrid_chat.py +0 -316
- package/latticeai/services/hybrid_context.py +0 -228
- package/latticeai/services/hybrid_policy.py +0 -129
- package/latticeai/services/interop_bridges.py +0 -978
- package/latticeai/services/local_knowledge.py +0 -465
- package/latticeai/services/memory_service/__init__.py +0 -52
- package/latticeai/services/memory_service/_contract.py +0 -100
- package/latticeai/services/memory_service/brief.py +0 -431
- package/latticeai/services/memory_service/constants.py +0 -57
- package/latticeai/services/memory_service/maintenance.py +0 -138
- package/latticeai/services/memory_service/manager.py +0 -186
- package/latticeai/services/memory_service/proof.py +0 -136
- package/latticeai/services/memory_service/recall.py +0 -225
- package/latticeai/services/memory_service/service.py +0 -48
- package/latticeai/services/memory_service/stores.py +0 -110
- package/latticeai/services/mode_store.py +0 -132
- package/latticeai/services/model_recommendation.py +0 -224
- package/latticeai/services/model_runtime/cloud.py +0 -87
- package/latticeai/services/network_boundary_service.py +0 -117
- package/latticeai/services/obsidian_bridge.py +0 -609
- package/latticeai/services/openai_compatible_adapter.py +0 -101
- package/latticeai/services/permission_mode_service.py +0 -122
- package/latticeai/services/platform_runtime.py +0 -366
- package/latticeai/services/review_queue.py +0 -380
- package/latticeai/services/router_context.py +0 -59
- package/latticeai/services/run_executor.py +0 -387
- package/latticeai/services/self_model_service.py +0 -171
- package/latticeai/services/setup_detection.py +0 -147
- package/latticeai/services/triggers.py +0 -378
- package/latticeai/services/upload_service.py +0 -172
- package/latticeai/services/workspace_service.py +0 -165
- package/latticeai/setup/__init__.py +0 -25
- package/latticeai/setup/auto_setup.py +0 -846
- package/latticeai/setup/demo_corpus.py +0 -98
- package/latticeai/setup/wizard/__init__.py +0 -126
- package/latticeai/setup/wizard/catalog.py +0 -175
- package/latticeai/setup/wizard/detect.py +0 -302
- package/latticeai/setup/wizard/install.py +0 -348
- package/latticeai/setup/wizard/paths.py +0 -165
- package/latticeai/setup/wizard/plans.py +0 -74
- package/latticeai/setup/wizard/recommend.py +0 -320
- package/scripts/bench_agent_smoke.py +0 -409
- package/scripts/bench_models.py +0 -540
- package/scripts/bench_vector_index.py +0 -295
- package/scripts/funnel_soft_gate.py +0 -192
- package/scripts/generate_agent_loop_fixtures.py +0 -994
- package/scripts/generate_rust_parity_fixtures.py +0 -908
- package/scripts/migrate_brain_storage.py +0 -57
- package/scripts/parity_fixture_corpus_context.py +0 -162
- package/scripts/parity_fixture_corpus_docgen.py +0 -341
- package/scripts/profile_kg.py +0 -355
- package/server.py +0 -30
- package/static/app/assets/AdminConsole-cf4npybT.js +0 -1
- package/static/app/assets/Capture-DiQ219jW.js +0 -1
- package/static/app/assets/System-CieofHQa.js +0 -1
- package/static/app/assets/arrow-left-kfsrk0mv.js +0 -1
- package/static/app/assets/circle-check-qqLug9nU.js +0 -1
- package/static/app/assets/index-DxmOfNRi.css +0 -2
- package/static/app/assets/primitives-SNp0LRJz.js +0 -1
- package/static/app/assets/search-BcHqkjoy.js +0 -1
|
@@ -1,295 +0,0 @@
|
|
|
1
|
-
"""One door per kind of thing: text, chat, memory record, picture, film, file.
|
|
2
|
-
|
|
3
|
-
Every method here returns the raw store payload that ``IngestionPipeline.\
|
|
4
|
-
ingest`` normalizes; none of them decide *whether* to run. Modality routing
|
|
5
|
-
(:meth:`IngestionRoutingMixin._modality_for`) answers ``"text"`` for everything
|
|
6
|
-
while multi-modal is off, which is what makes "off" mean *unchanged* rather than
|
|
7
|
-
*slightly different*.
|
|
8
|
-
"""
|
|
9
|
-
|
|
10
|
-
from __future__ import annotations
|
|
11
|
-
|
|
12
|
-
from pathlib import Path
|
|
13
|
-
from typing import Any, Dict
|
|
14
|
-
|
|
15
|
-
from ..multimodal import (
|
|
16
|
-
MODALITY_AUDIO,
|
|
17
|
-
MODALITY_IMAGE,
|
|
18
|
-
MODALITY_VIDEO,
|
|
19
|
-
ImageFacts,
|
|
20
|
-
audio_quality_score,
|
|
21
|
-
detect_modality,
|
|
22
|
-
extract_image_facts,
|
|
23
|
-
image_quality_score,
|
|
24
|
-
read_video_facts,
|
|
25
|
-
transcribe_audio,
|
|
26
|
-
video_frame_dir,
|
|
27
|
-
video_quality_score,
|
|
28
|
-
write_image_memory,
|
|
29
|
-
write_video_memory,
|
|
30
|
-
)
|
|
31
|
-
from ..utils import utc_now_iso
|
|
32
|
-
from ._contract import IngestionCore as _Core
|
|
33
|
-
from .constants import (
|
|
34
|
-
_MEMORY_NODE_TYPES,
|
|
35
|
-
AUDIO_NODE_TYPE,
|
|
36
|
-
AUDIO_SOURCE_TYPES,
|
|
37
|
-
IMAGE_SOURCE_TYPES,
|
|
38
|
-
VIDEO_SOURCE_TYPES,
|
|
39
|
-
)
|
|
40
|
-
from .hashing import _file_digest
|
|
41
|
-
from .models import IngestionItem
|
|
42
|
-
from .quality import _quality_level
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
class IngestionRoutingMixin(_Core):
|
|
46
|
-
"""The per-source-type ingest doors. Mixed into ``IngestionPipeline``."""
|
|
47
|
-
|
|
48
|
-
# ── routing helpers ──────────────────────────────────────────────────────
|
|
49
|
-
def _ingest_text(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
50
|
-
text = item.text or ""
|
|
51
|
-
if not text.strip():
|
|
52
|
-
raise ValueError(
|
|
53
|
-
f"Empty content: {source_type} ingestion requires non-empty text."
|
|
54
|
-
)
|
|
55
|
-
if len(text.encode("utf-8", "ignore")) > self._max_text_bytes:
|
|
56
|
-
raise ValueError(
|
|
57
|
-
f"Text payload exceeds the {self._max_text_bytes // (1024 * 1024)}MB ingestion limit."
|
|
58
|
-
)
|
|
59
|
-
title = item.title or item.source_uri or source_type
|
|
60
|
-
return self._kg.ingest_source(
|
|
61
|
-
source_type=source_type,
|
|
62
|
-
title=title,
|
|
63
|
-
text=text,
|
|
64
|
-
source_uri=item.source_uri,
|
|
65
|
-
owner=owner,
|
|
66
|
-
workspace_id=item.workspace_id,
|
|
67
|
-
permissions=item.permissions,
|
|
68
|
-
captured_at=captured_at,
|
|
69
|
-
modified_at=item.modified_at,
|
|
70
|
-
conversation_id=item.conversation_id,
|
|
71
|
-
metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
72
|
-
)
|
|
73
|
-
|
|
74
|
-
def _ingest_chat(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
75
|
-
text = item.text or ""
|
|
76
|
-
meta = item.metadata or {}
|
|
77
|
-
role = str(meta.get("role") or "user")
|
|
78
|
-
result = self._kg.ingest_message(
|
|
79
|
-
role,
|
|
80
|
-
text,
|
|
81
|
-
user_email=owner,
|
|
82
|
-
user_nickname=meta.get("user_nickname"),
|
|
83
|
-
source=meta.get("source") or source_type,
|
|
84
|
-
conversation_id=item.conversation_id,
|
|
85
|
-
workspace_id=item.workspace_id,
|
|
86
|
-
raw=meta.get("raw"),
|
|
87
|
-
)
|
|
88
|
-
# ingest_message reports message/response node ids; normalize the keys
|
|
89
|
-
# the provenance step expects.
|
|
90
|
-
result.setdefault("node_id", result.get("node_id") or result.get("message_node_id") or result.get("id"))
|
|
91
|
-
result.setdefault("title", item.title or text[:80])
|
|
92
|
-
return result
|
|
93
|
-
|
|
94
|
-
def _ingest_memory_record(self, item, *, source_type, owner) -> Dict[str, Any]:
|
|
95
|
-
node_type = _MEMORY_NODE_TYPES[source_type]
|
|
96
|
-
meta = item.metadata or {}
|
|
97
|
-
result = self._kg.ingest_event(
|
|
98
|
-
node_type,
|
|
99
|
-
item.title or (item.text or node_type)[:120],
|
|
100
|
-
user_email=owner,
|
|
101
|
-
source=meta.get("source") or source_type,
|
|
102
|
-
conversation_id=item.conversation_id,
|
|
103
|
-
workspace_id=item.workspace_id,
|
|
104
|
-
metadata={**meta, "detail": (item.text or "")[:2000]},
|
|
105
|
-
)
|
|
106
|
-
result.setdefault("node_id", result.get("node_id") or result.get("id"))
|
|
107
|
-
result.setdefault("title", item.title)
|
|
108
|
-
return result
|
|
109
|
-
|
|
110
|
-
# ── multi-modal routing (v11.1.0 Track 3) ────────────────────────────────
|
|
111
|
-
def _modality_for(self, item: IngestionItem, source_type: str) -> str:
|
|
112
|
-
"""``image`` / ``audio`` / ``video`` / ``text`` for this item.
|
|
113
|
-
|
|
114
|
-
Always ``"text"`` while the flag is off, which is what makes "off" mean
|
|
115
|
-
*unchanged* rather than *slightly different*.
|
|
116
|
-
"""
|
|
117
|
-
if not self._allow_multimodal:
|
|
118
|
-
return "text"
|
|
119
|
-
if source_type in IMAGE_SOURCE_TYPES:
|
|
120
|
-
return MODALITY_IMAGE
|
|
121
|
-
if source_type in AUDIO_SOURCE_TYPES:
|
|
122
|
-
return MODALITY_AUDIO
|
|
123
|
-
if source_type in VIDEO_SOURCE_TYPES:
|
|
124
|
-
return MODALITY_VIDEO
|
|
125
|
-
if not item.path:
|
|
126
|
-
return "text"
|
|
127
|
-
return detect_modality(item.path, item.mime_type)
|
|
128
|
-
|
|
129
|
-
def _resolve_file_path(self, item: IngestionItem) -> Path:
|
|
130
|
-
if not item.path:
|
|
131
|
-
raise ValueError("File ingestion requires a path.")
|
|
132
|
-
path = Path(item.path)
|
|
133
|
-
if not path.exists():
|
|
134
|
-
raise FileNotFoundError(f"File not found: {path}")
|
|
135
|
-
if path.is_dir():
|
|
136
|
-
raise ValueError(f"File ingestion requires a file, got a directory: {path}")
|
|
137
|
-
return path
|
|
138
|
-
|
|
139
|
-
def _ingest_image(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
140
|
-
"""Store one picture as an ``Image`` node — OCR, caption, vector.
|
|
141
|
-
|
|
142
|
-
The image vector (when a vision model produced one) goes to its own
|
|
143
|
-
index; the OCR/caption text rides the ordinary text index. That split
|
|
144
|
-
is what lets a typed question find a screenshot without ever comparing
|
|
145
|
-
a text vector to an image vector.
|
|
146
|
-
"""
|
|
147
|
-
path = self._resolve_file_path(item)
|
|
148
|
-
facts = extract_image_facts(str(path), ports=self._multimodal)
|
|
149
|
-
result = write_image_memory(
|
|
150
|
-
self._kg,
|
|
151
|
-
path=path,
|
|
152
|
-
facts=facts,
|
|
153
|
-
title=item.title or path.name,
|
|
154
|
-
source_type=source_type if source_type in IMAGE_SOURCE_TYPES else MODALITY_IMAGE,
|
|
155
|
-
source_uri=item.source_uri,
|
|
156
|
-
owner=owner,
|
|
157
|
-
workspace_id=item.workspace_id,
|
|
158
|
-
conversation_id=item.conversation_id,
|
|
159
|
-
captured_at=captured_at,
|
|
160
|
-
modified_at=item.modified_at,
|
|
161
|
-
permissions=item.permissions,
|
|
162
|
-
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
163
|
-
)
|
|
164
|
-
self._record_image_vector(result["node_id"], facts)
|
|
165
|
-
quality = image_quality_score(facts)
|
|
166
|
-
result["extraction_quality"] = {
|
|
167
|
-
"score": quality["score"],
|
|
168
|
-
"level": _quality_level(quality["score"]),
|
|
169
|
-
"reasons": quality["reasons"],
|
|
170
|
-
}
|
|
171
|
-
return result
|
|
172
|
-
|
|
173
|
-
def _record_image_vector(self, node_id: str, facts: ImageFacts) -> None:
|
|
174
|
-
"""File the image-space vector, if a vision model actually made one."""
|
|
175
|
-
if facts.embedding is None:
|
|
176
|
-
return
|
|
177
|
-
from ..graph.image_vectors import record_image_vector
|
|
178
|
-
|
|
179
|
-
record_image_vector(
|
|
180
|
-
self._kg,
|
|
181
|
-
node_id=node_id,
|
|
182
|
-
vector=facts.embedding,
|
|
183
|
-
model_id=self._multimodal.vision_model_id or "vision:unnamed",
|
|
184
|
-
space=self._multimodal.vision_space,
|
|
185
|
-
updated_at=utc_now_iso(),
|
|
186
|
-
)
|
|
187
|
-
|
|
188
|
-
def _ingest_audio(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
189
|
-
"""Store one recording as an ``Audio`` node, transcribed when possible.
|
|
190
|
-
|
|
191
|
-
The transcript is text and rides the ordinary text index — chunks,
|
|
192
|
-
concepts, provenance, dedupe all unchanged — but the node itself is a
|
|
193
|
-
recording, because that is what it is whether or not anyone could hear
|
|
194
|
-
it. The recording's own facts stay in the metadata (``modality``,
|
|
195
|
-
``audio_path``, ``transcription``, ``searchable``). Without a
|
|
196
|
-
transcriber the memory is still kept, and its body says plainly that
|
|
197
|
-
the words were never recognized instead of leaving a blank note.
|
|
198
|
-
"""
|
|
199
|
-
path = self._resolve_file_path(item)
|
|
200
|
-
facts = transcribe_audio(str(path), ports=self._multimodal, transcript=item.text)
|
|
201
|
-
title = item.title or path.stem
|
|
202
|
-
body = facts.transcript or (
|
|
203
|
-
f"[{MODALITY_AUDIO}] {title}\n"
|
|
204
|
-
"이 녹음은 아직 글로 바뀌지 않았습니다 — 음성 인식기가 없어 내용 검색은 되지 않습니다."
|
|
205
|
-
)
|
|
206
|
-
result = self._kg.ingest_source(
|
|
207
|
-
source_type=source_type,
|
|
208
|
-
title=title,
|
|
209
|
-
text=body,
|
|
210
|
-
source_uri=item.source_uri or str(path),
|
|
211
|
-
owner=owner,
|
|
212
|
-
workspace_id=item.workspace_id,
|
|
213
|
-
permissions=item.permissions,
|
|
214
|
-
captured_at=captured_at,
|
|
215
|
-
modified_at=item.modified_at,
|
|
216
|
-
conversation_id=item.conversation_id,
|
|
217
|
-
node_type=AUDIO_NODE_TYPE,
|
|
218
|
-
metadata={
|
|
219
|
-
"mime_type": item.mime_type,
|
|
220
|
-
"modality": MODALITY_AUDIO,
|
|
221
|
-
"audio_path": str(path),
|
|
222
|
-
"audio_bytes": path.stat().st_size,
|
|
223
|
-
"transcription": facts.transcription_status,
|
|
224
|
-
"searchable": facts.searchable,
|
|
225
|
-
**({"transcription_detail": facts.detail} if facts.detail else {}),
|
|
226
|
-
**(item.metadata or {}),
|
|
227
|
-
},
|
|
228
|
-
)
|
|
229
|
-
result.setdefault("title", title)
|
|
230
|
-
quality = audio_quality_score(facts)
|
|
231
|
-
result["extraction_quality"] = {
|
|
232
|
-
"score": quality["score"],
|
|
233
|
-
"level": _quality_level(quality["score"]),
|
|
234
|
-
"reasons": quality["reasons"],
|
|
235
|
-
}
|
|
236
|
-
return result
|
|
237
|
-
|
|
238
|
-
def _ingest_video(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
239
|
-
"""Store one video as keyframes through the image door plus subtitles.
|
|
240
|
-
|
|
241
|
-
Nothing here is a new retrieval path: the stills become ordinary
|
|
242
|
-
``Image`` nodes (OCR, caption, vector, thumbnail) joined by
|
|
243
|
-
``CONTAINS_IMAGE``, and the subtitle text becomes ordinary chunks. What
|
|
244
|
-
the ``Video`` node adds is the thing they belong to — and an honest
|
|
245
|
-
body when there were no subtitles to read.
|
|
246
|
-
"""
|
|
247
|
-
path = self._resolve_file_path(item)
|
|
248
|
-
facts = read_video_facts(
|
|
249
|
-
str(path),
|
|
250
|
-
video_frame_dir(getattr(self._kg, "blob_dir", path.parent), _file_digest(path)),
|
|
251
|
-
count=self._keyframes,
|
|
252
|
-
ports=self._multimodal,
|
|
253
|
-
subtitle_text=item.text,
|
|
254
|
-
)
|
|
255
|
-
result = write_video_memory(
|
|
256
|
-
self._kg,
|
|
257
|
-
path=path,
|
|
258
|
-
facts=facts,
|
|
259
|
-
title=item.title or path.stem,
|
|
260
|
-
source_type=source_type if source_type in VIDEO_SOURCE_TYPES else MODALITY_VIDEO,
|
|
261
|
-
source_uri=item.source_uri,
|
|
262
|
-
owner=owner,
|
|
263
|
-
workspace_id=item.workspace_id,
|
|
264
|
-
conversation_id=item.conversation_id,
|
|
265
|
-
captured_at=captured_at,
|
|
266
|
-
modified_at=item.modified_at,
|
|
267
|
-
permissions=item.permissions,
|
|
268
|
-
extra_metadata={"mime_type": item.mime_type, **(item.metadata or {})},
|
|
269
|
-
ports=self._multimodal,
|
|
270
|
-
)
|
|
271
|
-
quality = video_quality_score(facts)
|
|
272
|
-
result["extraction_quality"] = {
|
|
273
|
-
"score": quality["score"],
|
|
274
|
-
"level": _quality_level(quality["score"]),
|
|
275
|
-
"reasons": quality["reasons"],
|
|
276
|
-
}
|
|
277
|
-
return result
|
|
278
|
-
|
|
279
|
-
def _ingest_file(self, item, *, source_type, owner, captured_at) -> Dict[str, Any]:
|
|
280
|
-
path = self._resolve_file_path(item)
|
|
281
|
-
return self._kg.ingest_document(
|
|
282
|
-
path,
|
|
283
|
-
original_filename=item.title or path.name,
|
|
284
|
-
mime_type=item.mime_type,
|
|
285
|
-
uploader=owner,
|
|
286
|
-
conversation_id=item.conversation_id,
|
|
287
|
-
extracted=item.metadata.get("extracted") if item.metadata else None,
|
|
288
|
-
source_type=source_type,
|
|
289
|
-
source_uri=item.source_uri or str(path),
|
|
290
|
-
captured_at=captured_at,
|
|
291
|
-
modified_at=item.modified_at,
|
|
292
|
-
owner=owner,
|
|
293
|
-
workspace_id=item.workspace_id,
|
|
294
|
-
permissions=item.permissions,
|
|
295
|
-
)
|
|
@@ -1,380 +0,0 @@
|
|
|
1
|
-
"""Background ingestion jobs — queue + per-job progress (v9.9.6 extraction).
|
|
2
|
-
|
|
3
|
-
Split out of :mod:`lattice_brain.ingestion` (behaviour-preserving move, review
|
|
4
|
-
2026-07-27 P2 #8 "ingestion job/watch 분리"). The pipeline owns *what it means
|
|
5
|
-
to ingest one item*; this module owns *scheduling many of them and reporting
|
|
6
|
-
progress* — a genuinely separate concern with its own frozen wire schema
|
|
7
|
-
(``/api/ingestion/jobs*``) and its own resume semantics.
|
|
8
|
-
|
|
9
|
-
The seam is also where a real scheduler (thread pool, rq, celery) would plug
|
|
10
|
-
in without touching the pipeline.
|
|
11
|
-
|
|
12
|
-
Job state is **durable** (review 2026-08 P1 #3): ``done_indices`` used to live
|
|
13
|
-
only in this process's heap, so a restart mid-import silently lost the resume
|
|
14
|
-
point and re-ingesting meant replaying every item. :class:`IngestionJobStore`
|
|
15
|
-
persists the queue to SQLite — by default the same database file the knowledge
|
|
16
|
-
graph and :mod:`lattice_brain.conversations` use, so the existing
|
|
17
|
-
backup/restore covers it with no manifest change. A queue built without a
|
|
18
|
-
``db_path`` stays purely in-memory and says so through :meth:`describe`.
|
|
19
|
-
|
|
20
|
-
``IngestionItem`` is imported only for type checking: the pipeline module owns
|
|
21
|
-
that dataclass, and a runtime import here would be circular.
|
|
22
|
-
"""
|
|
23
|
-
|
|
24
|
-
from __future__ import annotations
|
|
25
|
-
|
|
26
|
-
import dataclasses
|
|
27
|
-
import json
|
|
28
|
-
import logging
|
|
29
|
-
import sqlite3
|
|
30
|
-
import threading
|
|
31
|
-
from contextlib import contextmanager
|
|
32
|
-
from dataclasses import dataclass, field
|
|
33
|
-
from pathlib import Path
|
|
34
|
-
from typing import TYPE_CHECKING, Any, Dict, Iterator, List, Optional, Set
|
|
35
|
-
|
|
36
|
-
from .utils import utc_now_iso
|
|
37
|
-
|
|
38
|
-
if TYPE_CHECKING: # pragma: no cover - typing only
|
|
39
|
-
from .ingestion import IngestionItem
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
JOB_ERRORS_CAP = 50 # per-job error records kept (failed count keeps counting)
|
|
43
|
-
|
|
44
|
-
#: Statuses a job may hold on disk. Frozen wire schema of ``/api/ingestion/jobs*``.
|
|
45
|
-
JOB_STATUSES = ("queued", "running", "completed", "failed", "partial")
|
|
46
|
-
|
|
47
|
-
|
|
48
|
-
@dataclass
|
|
49
|
-
class BackgroundIngestionJob:
|
|
50
|
-
"""Job descriptor + progress state for background/incremental indexing.
|
|
51
|
-
|
|
52
|
-
``done_indices`` tracks per-item completion so an interrupted or partially
|
|
53
|
-
failed job can be *resumed* from the remaining items instead of restarting.
|
|
54
|
-
``errors`` is capped at ``max_errors`` records; ``failed`` keeps counting.
|
|
55
|
-
"""
|
|
56
|
-
job_id: str
|
|
57
|
-
items: List[IngestionItem]
|
|
58
|
-
status: str = "queued" # queued | running | completed | failed | partial
|
|
59
|
-
created_at: str = field(default_factory=utc_now_iso)
|
|
60
|
-
updated_at: str = field(default_factory=utc_now_iso)
|
|
61
|
-
processed: int = 0
|
|
62
|
-
failed: int = 0
|
|
63
|
-
total: int = 0
|
|
64
|
-
errors: List[Dict[str, Any]] = field(default_factory=list)
|
|
65
|
-
incremental: bool = True
|
|
66
|
-
user_email: Optional[str] = None
|
|
67
|
-
max_errors: int = JOB_ERRORS_CAP
|
|
68
|
-
done_indices: Set[int] = field(default_factory=set)
|
|
69
|
-
|
|
70
|
-
def touch(self) -> None:
|
|
71
|
-
self.updated_at = utc_now_iso()
|
|
72
|
-
|
|
73
|
-
def record_error(self, index: int, item: IngestionItem, detail: Any) -> None:
|
|
74
|
-
self.failed += 1
|
|
75
|
-
if len(self.errors) < self.max_errors:
|
|
76
|
-
self.errors.append({
|
|
77
|
-
"index": index,
|
|
78
|
-
"source": item.source_uri or item.path or item.title or item.source_type,
|
|
79
|
-
"detail": str(detail)[:500],
|
|
80
|
-
})
|
|
81
|
-
|
|
82
|
-
def remaining_indices(self) -> List[int]:
|
|
83
|
-
return [i for i in range(len(self.items)) if i not in self.done_indices]
|
|
84
|
-
|
|
85
|
-
def as_dict(self) -> Dict[str, Any]:
|
|
86
|
-
"""Frozen job schema consumed by ``/api/ingestion/jobs*``."""
|
|
87
|
-
return {
|
|
88
|
-
"job_id": self.job_id,
|
|
89
|
-
"status": self.status,
|
|
90
|
-
"total": self.total,
|
|
91
|
-
"processed": self.processed,
|
|
92
|
-
"failed": self.failed,
|
|
93
|
-
"errors": list(self.errors),
|
|
94
|
-
"created_at": self.created_at,
|
|
95
|
-
"updated_at": self.updated_at,
|
|
96
|
-
}
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
def _item_payload(item: IngestionItem) -> Dict[str, Any]:
|
|
100
|
-
"""One ingestion item as JSON-safe data (dataclass fields only)."""
|
|
101
|
-
return dataclasses.asdict(item)
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
def _item_from_payload(payload: Dict[str, Any]) -> IngestionItem:
|
|
105
|
-
"""Rebuild an ``IngestionItem`` from persisted data.
|
|
106
|
-
|
|
107
|
-
Unknown keys are dropped rather than raising: a job written by an older
|
|
108
|
-
build must still resume on a newer one. Imported here (not at module
|
|
109
|
-
scope) because ``lattice_brain.ingestion`` imports *this* module.
|
|
110
|
-
"""
|
|
111
|
-
from .ingestion import IngestionItem as _Item
|
|
112
|
-
|
|
113
|
-
known = {f.name for f in dataclasses.fields(_Item)}
|
|
114
|
-
kwargs = {key: value for key, value in payload.items() if key in known}
|
|
115
|
-
metadata = kwargs.get("metadata")
|
|
116
|
-
if not isinstance(metadata, dict):
|
|
117
|
-
kwargs["metadata"] = {}
|
|
118
|
-
return _Item(**kwargs)
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
class IngestionJobStore:
|
|
122
|
-
"""SQLite persistence for :class:`BackgroundIngestionJob`.
|
|
123
|
-
|
|
124
|
-
Own connection per operation, always closed — ``with sqlite3.connect(...)``
|
|
125
|
-
commits but never closes (same note as
|
|
126
|
-
:meth:`lattice_brain.conversations.ConversationStore._connect`).
|
|
127
|
-
"""
|
|
128
|
-
|
|
129
|
-
def __init__(self, db_path: Path) -> None:
|
|
130
|
-
self.db_path = Path(db_path)
|
|
131
|
-
self.db_path.parent.mkdir(parents=True, exist_ok=True)
|
|
132
|
-
self._init_db()
|
|
133
|
-
|
|
134
|
-
@contextmanager
|
|
135
|
-
def _connect(self) -> Iterator[sqlite3.Connection]:
|
|
136
|
-
conn = sqlite3.connect(str(self.db_path))
|
|
137
|
-
conn.row_factory = sqlite3.Row
|
|
138
|
-
conn.execute("PRAGMA journal_mode=WAL")
|
|
139
|
-
try:
|
|
140
|
-
with conn:
|
|
141
|
-
yield conn
|
|
142
|
-
finally:
|
|
143
|
-
conn.close()
|
|
144
|
-
|
|
145
|
-
def _init_db(self) -> None:
|
|
146
|
-
with self._connect() as conn:
|
|
147
|
-
conn.executescript(
|
|
148
|
-
"""
|
|
149
|
-
CREATE TABLE IF NOT EXISTS ingestion_jobs (
|
|
150
|
-
job_id TEXT PRIMARY KEY,
|
|
151
|
-
status TEXT NOT NULL,
|
|
152
|
-
total INTEGER NOT NULL DEFAULT 0,
|
|
153
|
-
processed INTEGER NOT NULL DEFAULT 0,
|
|
154
|
-
failed INTEGER NOT NULL DEFAULT 0,
|
|
155
|
-
incremental INTEGER NOT NULL DEFAULT 1,
|
|
156
|
-
user_email TEXT,
|
|
157
|
-
max_errors INTEGER NOT NULL DEFAULT 50,
|
|
158
|
-
items_json TEXT NOT NULL DEFAULT '[]',
|
|
159
|
-
done_indices_json TEXT NOT NULL DEFAULT '[]',
|
|
160
|
-
errors_json TEXT NOT NULL DEFAULT '[]',
|
|
161
|
-
created_at TEXT NOT NULL,
|
|
162
|
-
updated_at TEXT NOT NULL
|
|
163
|
-
);
|
|
164
|
-
CREATE INDEX IF NOT EXISTS idx_ingestion_jobs_status
|
|
165
|
-
ON ingestion_jobs(status);
|
|
166
|
-
CREATE INDEX IF NOT EXISTS idx_ingestion_jobs_created
|
|
167
|
-
ON ingestion_jobs(created_at);
|
|
168
|
-
"""
|
|
169
|
-
)
|
|
170
|
-
|
|
171
|
-
def save(self, job: BackgroundIngestionJob) -> None:
|
|
172
|
-
with self._connect() as conn:
|
|
173
|
-
conn.execute(
|
|
174
|
-
"""
|
|
175
|
-
INSERT INTO ingestion_jobs(
|
|
176
|
-
job_id, status, total, processed, failed, incremental, user_email,
|
|
177
|
-
max_errors, items_json, done_indices_json, errors_json,
|
|
178
|
-
created_at, updated_at)
|
|
179
|
-
VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
|
|
180
|
-
ON CONFLICT(job_id) DO UPDATE SET
|
|
181
|
-
status=excluded.status,
|
|
182
|
-
total=excluded.total,
|
|
183
|
-
processed=excluded.processed,
|
|
184
|
-
failed=excluded.failed,
|
|
185
|
-
incremental=excluded.incremental,
|
|
186
|
-
user_email=excluded.user_email,
|
|
187
|
-
max_errors=excluded.max_errors,
|
|
188
|
-
items_json=excluded.items_json,
|
|
189
|
-
done_indices_json=excluded.done_indices_json,
|
|
190
|
-
errors_json=excluded.errors_json,
|
|
191
|
-
updated_at=excluded.updated_at
|
|
192
|
-
""",
|
|
193
|
-
(
|
|
194
|
-
job.job_id,
|
|
195
|
-
job.status,
|
|
196
|
-
int(job.total),
|
|
197
|
-
int(job.processed),
|
|
198
|
-
int(job.failed),
|
|
199
|
-
1 if job.incremental else 0,
|
|
200
|
-
job.user_email,
|
|
201
|
-
int(job.max_errors),
|
|
202
|
-
json.dumps(
|
|
203
|
-
[_item_payload(item) for item in job.items],
|
|
204
|
-
ensure_ascii=False,
|
|
205
|
-
default=str,
|
|
206
|
-
),
|
|
207
|
-
json.dumps(sorted(job.done_indices)),
|
|
208
|
-
json.dumps(list(job.errors), ensure_ascii=False, default=str),
|
|
209
|
-
job.created_at,
|
|
210
|
-
job.updated_at,
|
|
211
|
-
),
|
|
212
|
-
)
|
|
213
|
-
|
|
214
|
-
def load_all(self) -> List[BackgroundIngestionJob]:
|
|
215
|
-
"""Every persisted job, oldest first.
|
|
216
|
-
|
|
217
|
-
A row still marked ``running`` means the process died mid-run. It is
|
|
218
|
-
reported as ``partial``/``queued`` (whichever the recorded progress
|
|
219
|
-
supports) so it is honestly *not* running and so
|
|
220
|
-
``run_background_job`` will pick it up instead of refusing.
|
|
221
|
-
"""
|
|
222
|
-
with self._connect() as conn:
|
|
223
|
-
rows = conn.execute(
|
|
224
|
-
"SELECT * FROM ingestion_jobs ORDER BY created_at ASC, job_id ASC"
|
|
225
|
-
).fetchall()
|
|
226
|
-
jobs: List[BackgroundIngestionJob] = []
|
|
227
|
-
for row in rows:
|
|
228
|
-
try:
|
|
229
|
-
jobs.append(self._row_to_job(row))
|
|
230
|
-
except Exception as exc: # noqa: BLE001 — one bad row must not hide the rest
|
|
231
|
-
logging.warning(
|
|
232
|
-
"ingestion job %s could not be restored: %s", row["job_id"], exc
|
|
233
|
-
)
|
|
234
|
-
return jobs
|
|
235
|
-
|
|
236
|
-
@staticmethod
|
|
237
|
-
def _row_to_job(row: sqlite3.Row) -> BackgroundIngestionJob:
|
|
238
|
-
items = [
|
|
239
|
-
_item_from_payload(payload)
|
|
240
|
-
for payload in json.loads(row["items_json"] or "[]")
|
|
241
|
-
]
|
|
242
|
-
done = {int(i) for i in json.loads(row["done_indices_json"] or "[]")}
|
|
243
|
-
status = str(row["status"] or "queued")
|
|
244
|
-
if status == "running":
|
|
245
|
-
status = "partial" if done else "queued"
|
|
246
|
-
return BackgroundIngestionJob(
|
|
247
|
-
job_id=str(row["job_id"]),
|
|
248
|
-
items=items,
|
|
249
|
-
status=status,
|
|
250
|
-
created_at=str(row["created_at"]),
|
|
251
|
-
updated_at=str(row["updated_at"]),
|
|
252
|
-
processed=len(done),
|
|
253
|
-
failed=int(row["failed"] or 0),
|
|
254
|
-
total=int(row["total"] or 0),
|
|
255
|
-
errors=list(json.loads(row["errors_json"] or "[]")),
|
|
256
|
-
incremental=bool(row["incremental"]),
|
|
257
|
-
user_email=row["user_email"],
|
|
258
|
-
max_errors=int(row["max_errors"] or JOB_ERRORS_CAP),
|
|
259
|
-
done_indices=done,
|
|
260
|
-
)
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
class BackgroundIngestionQueue:
|
|
264
|
-
"""Queue for background incremental ingestion, durable when given a db.
|
|
265
|
-
|
|
266
|
-
For large corpus: this is the seam where a real scheduler / worker pool
|
|
267
|
-
(celery, rq, or internal thread) can be plugged later without changing callers.
|
|
268
|
-
Supports incremental (skip duplicates) vs force reindex.
|
|
269
|
-
|
|
270
|
-
``db_path`` makes job state survive a restart. The in-process dict stays
|
|
271
|
-
the authority *within* a process (callers hold job references and mutate
|
|
272
|
-
them), and every mutation is mirrored to SQLite through :meth:`save`.
|
|
273
|
-
Without ``db_path`` — or when the database cannot be opened — the queue is
|
|
274
|
-
memory-only and :meth:`describe` reports that instead of implying
|
|
275
|
-
durability it does not have.
|
|
276
|
-
"""
|
|
277
|
-
|
|
278
|
-
def __init__(self, db_path: Optional[Any] = None) -> None:
|
|
279
|
-
self._jobs: Dict[str, BackgroundIngestionJob] = {}
|
|
280
|
-
self._counter = 0
|
|
281
|
-
self._lock = threading.RLock()
|
|
282
|
-
self._store: Optional[IngestionJobStore] = None
|
|
283
|
-
self._persistence_detail: Optional[str] = None
|
|
284
|
-
if db_path is None:
|
|
285
|
-
self._persistence_detail = "no database configured; job state is in-memory only"
|
|
286
|
-
elif not isinstance(db_path, (str, Path)):
|
|
287
|
-
self._persistence_detail = (
|
|
288
|
-
f"unusable db_path {type(db_path).__name__}; job state is in-memory only"
|
|
289
|
-
)
|
|
290
|
-
else:
|
|
291
|
-
try:
|
|
292
|
-
self._store = IngestionJobStore(Path(db_path))
|
|
293
|
-
for job in self._store.load_all():
|
|
294
|
-
self._jobs[job.job_id] = job
|
|
295
|
-
self._counter = max(self._counter, _job_sequence(job.job_id))
|
|
296
|
-
except Exception as exc: # noqa: BLE001 — degrade to memory, never block ingestion
|
|
297
|
-
self._store = None
|
|
298
|
-
self._persistence_detail = f"job persistence unavailable: {exc}"
|
|
299
|
-
logging.warning("ingestion job persistence unavailable: %s", exc)
|
|
300
|
-
|
|
301
|
-
# ── honesty surface ──────────────────────────────────────────────────────
|
|
302
|
-
def describe(self) -> Dict[str, Any]:
|
|
303
|
-
"""Whether resume state actually survives a restart, and where."""
|
|
304
|
-
return {
|
|
305
|
-
"persistent": self._store is not None,
|
|
306
|
-
"db_path": str(self._store.db_path) if self._store is not None else None,
|
|
307
|
-
"jobs": len(self._jobs),
|
|
308
|
-
"detail": self._persistence_detail,
|
|
309
|
-
}
|
|
310
|
-
|
|
311
|
-
def save(self, job: BackgroundIngestionJob) -> None:
|
|
312
|
-
"""Mirror a job's current state to disk (no-op when memory-only).
|
|
313
|
-
|
|
314
|
-
A persistence failure degrades durability, never the run in progress:
|
|
315
|
-
the item work already succeeded and must not be rolled back by a
|
|
316
|
-
bookkeeping error.
|
|
317
|
-
"""
|
|
318
|
-
if self._store is None:
|
|
319
|
-
return
|
|
320
|
-
try:
|
|
321
|
-
self._store.save(job)
|
|
322
|
-
except Exception as exc: # noqa: BLE001 — durability is best-effort mid-run
|
|
323
|
-
logging.warning("ingestion job %s could not be persisted: %s", job.job_id, exc)
|
|
324
|
-
|
|
325
|
-
def schedule(
|
|
326
|
-
self,
|
|
327
|
-
items: List[IngestionItem],
|
|
328
|
-
*,
|
|
329
|
-
incremental: bool = True,
|
|
330
|
-
user_email: Optional[str] = None,
|
|
331
|
-
) -> BackgroundIngestionJob:
|
|
332
|
-
with self._lock:
|
|
333
|
-
self._counter += 1
|
|
334
|
-
job_id = f"bg_ingest_{self._counter:04d}"
|
|
335
|
-
job = BackgroundIngestionJob(
|
|
336
|
-
job_id=job_id,
|
|
337
|
-
items=items,
|
|
338
|
-
total=len(items),
|
|
339
|
-
incremental=incremental,
|
|
340
|
-
user_email=user_email,
|
|
341
|
-
)
|
|
342
|
-
# annotate items for downstream
|
|
343
|
-
for it in job.items:
|
|
344
|
-
# attach flag without breaking dataclass defaults (use metadata)
|
|
345
|
-
it.metadata = {**it.metadata, "incremental": incremental, "bg_job": job_id}
|
|
346
|
-
with self._lock:
|
|
347
|
-
self._jobs[job_id] = job
|
|
348
|
-
self.save(job)
|
|
349
|
-
return job
|
|
350
|
-
|
|
351
|
-
def get(self, job_id: str) -> Optional[BackgroundIngestionJob]:
|
|
352
|
-
return self._jobs.get(job_id)
|
|
353
|
-
|
|
354
|
-
def list_recent(self, limit: int = 20) -> List[BackgroundIngestionJob]:
|
|
355
|
-
"""Most recent jobs first (insertion order is schedule order)."""
|
|
356
|
-
limit = max(1, int(limit))
|
|
357
|
-
with self._lock:
|
|
358
|
-
jobs = list(self._jobs.values())
|
|
359
|
-
return list(reversed(jobs))[:limit]
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
def _job_sequence(job_id: str) -> int:
|
|
363
|
-
"""The numeric suffix of ``bg_ingest_0007`` → 7 (0 when unparseable).
|
|
364
|
-
|
|
365
|
-
Restored jobs must not have their ids handed out again after a restart.
|
|
366
|
-
"""
|
|
367
|
-
_, _, suffix = str(job_id or "").rpartition("_")
|
|
368
|
-
try:
|
|
369
|
-
return int(suffix)
|
|
370
|
-
except ValueError:
|
|
371
|
-
return 0
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
__all__ = [
|
|
375
|
-
"JOB_ERRORS_CAP",
|
|
376
|
-
"JOB_STATUSES",
|
|
377
|
-
"BackgroundIngestionJob",
|
|
378
|
-
"BackgroundIngestionQueue",
|
|
379
|
-
"IngestionJobStore",
|
|
380
|
-
]
|