marvisx-cli 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- core/api/__init__.py +0 -0
- core/api/agents/__init__.py +0 -0
- core/api/agents/session_health.py +59 -0
- core/api/agents/session_manager.py +206 -0
- core/api/bin/marvisx-state-hook.py +182 -0
- core/api/config.py +533 -0
- core/api/db.py +1516 -0
- core/api/dependencies/__init__.py +0 -0
- core/api/dependencies/tenant.py +34 -0
- core/api/main.py +1641 -0
- core/api/mcp/__init__.py +8 -0
- core/api/mcp/_adapter.py +184 -0
- core/api/mcp/server.py +58 -0
- core/api/mcp/tools/__init__.py +59 -0
- core/api/mcp/tools/brain.py +599 -0
- core/api/mcp/tools/graph.py +380 -0
- core/api/mcp/tools/handoffs.py +112 -0
- core/api/mcp/tools/ingest.py +326 -0
- core/api/mcp/tools/learnings.py +144 -0
- core/api/mcp/tools/projects.py +99 -0
- core/api/mcp/tools/pull_requests.py +173 -0
- core/api/mcp/tools/safety.py +111 -0
- core/api/mcp/tools/search.py +79 -0
- core/api/mcp/tools/tasks.py +258 -0
- core/api/middleware/__init__.py +0 -0
- core/api/middleware/tool_call_audit.py +111 -0
- core/api/models/__init__.py +346 -0
- core/api/models/auth.py +48 -0
- core/api/models/brain.py +1006 -0
- core/api/models/common.py +76 -0
- core/api/models/costs.py +91 -0
- core/api/models/graph.py +66 -0
- core/api/models/graph_cosmo.py +125 -0
- core/api/models/graph_pr_impact.py +257 -0
- core/api/models/graph_ux.py +141 -0
- core/api/models/inbox.py +230 -0
- core/api/models/ingest_keys.py +108 -0
- core/api/models/kg.py +41 -0
- core/api/models/llm_config.py +56 -0
- core/api/models/monitoring.py +234 -0
- core/api/models/projects.py +161 -0
- core/api/models/search.py +42 -0
- core/api/models/sessions.py +322 -0
- core/api/models/tasks.py +184 -0
- core/api/models/teams.py +63 -0
- core/api/models/users.py +184 -0
- core/api/observability/__init__.py +0 -0
- core/api/observability/tracing.py +92 -0
- core/api/paths.py +26 -0
- core/api/rate_limit.py +24 -0
- core/api/rbac.py +112 -0
- core/api/routers/__init__.py +0 -0
- core/api/routers/_adapter.py +24 -0
- core/api/routers/admin_pr_impact.py +230 -0
- core/api/routers/admin_settings.py +147 -0
- core/api/routers/agent.py +1079 -0
- core/api/routers/agent_tokens.py +276 -0
- core/api/routers/app_settings.py +112 -0
- core/api/routers/audit.py +89 -0
- core/api/routers/auth.py +586 -0
- core/api/routers/bench.py +161 -0
- core/api/routers/brain.py +881 -0
- core/api/routers/brain_directions.py +527 -0
- core/api/routers/ci_checks.py +140 -0
- core/api/routers/comments.py +273 -0
- core/api/routers/costs.py +148 -0
- core/api/routers/docs_coverage.py +217 -0
- core/api/routers/docs_governance.py +63 -0
- core/api/routers/documents.py +318 -0
- core/api/routers/files.py +163 -0
- core/api/routers/finder.py +987 -0
- core/api/routers/graph.py +836 -0
- core/api/routers/handoffs.py +156 -0
- core/api/routers/inbox.py +496 -0
- core/api/routers/ingest_api_keys.py +205 -0
- core/api/routers/ingest_triage.py +1227 -0
- core/api/routers/judge.py +306 -0
- core/api/routers/kg.py +336 -0
- core/api/routers/learnings.py +253 -0
- core/api/routers/llm_config.py +130 -0
- core/api/routers/monitoring.py +347 -0
- core/api/routers/notifications.py +125 -0
- core/api/routers/pr_impact.py +315 -0
- core/api/routers/projects.py +1061 -0
- core/api/routers/pull_requests.py +312 -0
- core/api/routers/push.py +67 -0
- core/api/routers/raci.py +228 -0
- core/api/routers/search.py +125 -0
- core/api/routers/sessions.py +3100 -0
- core/api/routers/settings.py +90 -0
- core/api/routers/share_repo.py +68 -0
- core/api/routers/status_updates.py +96 -0
- core/api/routers/tags.py +45 -0
- core/api/routers/tasks.py +526 -0
- core/api/routers/teams.py +425 -0
- core/api/routers/terminal.py +105 -0
- core/api/routers/users.py +331 -0
- core/api/routers/webhooks.py +330 -0
- core/api/runtime_settings.py +84 -0
- core/api/security.py +652 -0
- core/api/services/__init__.py +0 -0
- core/api/services/audit.py +58 -0
- core/api/services/auto_approval.py +82 -0
- core/api/services/brain/__init__.py +52 -0
- core/api/services/brain/baseline.py +230 -0
- core/api/services/brain/capabilities.py +75 -0
- core/api/services/brain/cascade_rollup.py +388 -0
- core/api/services/brain/compound_bridge.py +215 -0
- core/api/services/brain/cycle.py +1242 -0
- core/api/services/brain/cycle_snapshot.py +371 -0
- core/api/services/brain/digest_collector.py +147 -0
- core/api/services/brain/direction.py +421 -0
- core/api/services/brain/drift.py +356 -0
- core/api/services/brain/drift_router.py +409 -0
- core/api/services/brain/edge_metrics.py +79 -0
- core/api/services/brain/events_reader.py +222 -0
- core/api/services/brain/findings.py +1379 -0
- core/api/services/brain/findings_reader.py +1006 -0
- core/api/services/brain/jobs.py +733 -0
- core/api/services/brain/journal.py +206 -0
- core/api/services/brain/knowledge_forms.py +92 -0
- core/api/services/brain/llm/__init__.py +37 -0
- core/api/services/brain/llm/_runner.py +62 -0
- core/api/services/brain/llm/base.py +70 -0
- core/api/services/brain/llm/cache.py +99 -0
- core/api/services/brain/llm/constants.py +46 -0
- core/api/services/brain/llm/direction_alignment.py +289 -0
- core/api/services/brain/llm/factory.py +132 -0
- core/api/services/brain/llm/finding_reasoning.py +98 -0
- core/api/services/brain/llm/finding_summary.py +92 -0
- core/api/services/brain/llm/grounding.py +46 -0
- core/api/services/brain/llm/journal_polish.py +96 -0
- core/api/services/brain/llm/local_gateway.py +426 -0
- core/api/services/brain/llm/parsers.py +71 -0
- core/api/services/brain/llm/router_glue.py +422 -0
- core/api/services/brain/memory_ops.py +1677 -0
- core/api/services/brain/models.py +140 -0
- core/api/services/brain/owner_hint.py +211 -0
- core/api/services/brain/recap.py +307 -0
- core/api/services/brain/rules/__init__.py +65 -0
- core/api/services/brain/rules/_signals.py +205 -0
- core/api/services/brain/rules/dr1_activity_without_status.py +108 -0
- core/api/services/brain/rules/dr2_decision_without_adr.py +110 -0
- core/api/services/brain/rules/dr3_stale_open_loop.py +127 -0
- core/api/services/brain/rules/dr4_docs_governance_drift.py +79 -0
- core/api/services/brain/rules/dr5_playbook_changed.py +99 -0
- core/api/services/brain/rules/dr6_external_update_unpropagated.py +100 -0
- core/api/services/brain/rules/dr7_claimed_decision_gap.py +123 -0
- core/api/services/brain/rules/dr8_direction_misalignment.py +230 -0
- core/api/services/brain/runs_reader.py +485 -0
- core/api/services/brain/scope.py +79 -0
- core/api/services/brain/sources/__init__.py +46 -0
- core/api/services/brain/sources/base.py +86 -0
- core/api/services/brain/sources/git_kg.py +393 -0
- core/api/services/brain/sources/handoffs.py +157 -0
- core/api/services/brain/sources/ingestor.py +130 -0
- core/api/services/brain/sources/learnings.py +121 -0
- core/api/services/brain/sources/pir_tasks.py +245 -0
- core/api/services/brain/watermarks.py +147 -0
- core/api/services/brain/ws_emitter.py +170 -0
- core/api/services/cc_tasks_reader.py +76 -0
- core/api/services/ci_service.py +263 -0
- core/api/services/claude_metrics.py +796 -0
- core/api/services/codex_metrics.py +364 -0
- core/api/services/conversation_reader.py +102 -0
- core/api/services/cost_service.py +243 -0
- core/api/services/crypto.py +147 -0
- core/api/services/docs_governance/__init__.py +1 -0
- core/api/services/docs_governance/confidence.py +230 -0
- core/api/services/docs_governance/config.py +87 -0
- core/api/services/docs_governance/enrichment.py +65 -0
- core/api/services/docs_governance/frontmatter_validator.py +83 -0
- core/api/services/docs_governance/hard_gates.py +221 -0
- core/api/services/docs_governance/triage_orchestrator.py +98 -0
- core/api/services/embedding_internal.py +395 -0
- core/api/services/embedding_service.py +832 -0
- core/api/services/event_dispatcher.py +167 -0
- core/api/services/events.py +70 -0
- core/api/services/git_ops.py +621 -0
- core/api/services/graph_cosmo_service.py +440 -0
- core/api/services/graph_ranker.py +306 -0
- core/api/services/graph_service.py +1589 -0
- core/api/services/inbox.py +800 -0
- core/api/services/inbox_digest.py +221 -0
- core/api/services/inbox_digest_deep_research.py +80 -0
- core/api/services/inbox_digest_jobs.py +595 -0
- core/api/services/inbox_gmail_sync.py +167 -0
- core/api/services/inbox_llm_classifier.py +906 -0
- core/api/services/inbox_source_identity.py +116 -0
- core/api/services/inbox_sources.py +456 -0
- core/api/services/inbox_taxonomy.py +195 -0
- core/api/services/inbox_tldr.py +1079 -0
- core/api/services/inbox_triage.py +899 -0
- core/api/services/ingest/__init__.py +13 -0
- core/api/services/ingest/api_key_auth.py +136 -0
- core/api/services/ingest/auto_approve.py +120 -0
- core/api/services/ingest/classifier.py +138 -0
- core/api/services/ingest/confidence.py +173 -0
- core/api/services/ingest/dispatch.py +88 -0
- core/api/services/ingest/embedding_router.py +272 -0
- core/api/services/ingest/events.py +33 -0
- core/api/services/ingest/ignore_patterns.py +79 -0
- core/api/services/ingest/image_probe.py +218 -0
- core/api/services/ingest/ingress.py +263 -0
- core/api/services/ingest/insert_saga.py +793 -0
- core/api/services/ingest/llm/__init__.py +13 -0
- core/api/services/ingest/llm/anthropic_haiku.py +23 -0
- core/api/services/ingest/llm/base.py +59 -0
- core/api/services/ingest/llm/byok_provider.py +130 -0
- core/api/services/ingest/llm/classification_context.py +301 -0
- core/api/services/ingest/llm/config_store.py +246 -0
- core/api/services/ingest/llm/factory.py +24 -0
- core/api/services/ingest/llm/kg_enricher.py +306 -0
- core/api/services/ingest/llm/local_gateway.py +821 -0
- core/api/services/ingest/llm/local_vllm.py +23 -0
- core/api/services/ingest/llm/openai_nano.py +349 -0
- core/api/services/ingest/lock_advisory.py +57 -0
- core/api/services/ingest/parser_router.py +1756 -0
- core/api/services/ingest/parsers/__init__.py +1 -0
- core/api/services/ingest/parsers/docling_parser.py +142 -0
- core/api/services/ingest/parsers/docparse_gateway.py +178 -0
- core/api/services/ingest/parsers/docx_parser.py +127 -0
- core/api/services/ingest/parsers/folder_unpacker.py +85 -0
- core/api/services/ingest/parsers/gateway_aux.py +147 -0
- core/api/services/ingest/parsers/image_parser.py +251 -0
- core/api/services/ingest/parsers/internal_markdown.py +89 -0
- core/api/services/ingest/parsers/ocr_gateway.py +117 -0
- core/api/services/ingest/parsers/ocr_pdf_parser.py +112 -0
- core/api/services/ingest/parsers/pdf_types.py +13 -0
- core/api/services/ingest/parsers/transcript_parser.py +445 -0
- core/api/services/ingest/parsers/vision_gateway.py +186 -0
- core/api/services/ingest/parsers/xlsx_parser.py +91 -0
- core/api/services/ingest/parsers/zip_unpacker.py +126 -0
- core/api/services/ingest/preflight.py +393 -0
- core/api/services/ingest/retry_voyage.py +88 -0
- core/api/services/ingest/routing_policy.py +307 -0
- core/api/services/ingest/serializers/__init__.py +1 -0
- core/api/services/ingest/serializers/xlsx_to_markdown.py +80 -0
- core/api/services/ingest/skip_log.py +74 -0
- core/api/services/ingest/watcher.py +637 -0
- core/api/services/kg/__init__.py +0 -0
- core/api/services/kg/audit.py +49 -0
- core/api/services/kg/hybrid_search.py +691 -0
- core/api/services/kg/lens.py +339 -0
- core/api/services/kg/pr_impact.py +770 -0
- core/api/services/kg/queries.py +152 -0
- core/api/services/kg/ranking.py +89 -0
- core/api/services/kg/rrf.py +143 -0
- core/api/services/kg_watcher_control.py +161 -0
- core/api/services/local_llm/__init__.py +19 -0
- core/api/services/local_llm/async_client.py +385 -0
- core/api/services/local_llm/client.py +173 -0
- core/api/services/local_llm/url_validator.py +44 -0
- core/api/services/metrics_collector.py +646 -0
- core/api/services/metrics_providers.py +65 -0
- core/api/services/model_registry.py +266 -0
- core/api/services/model_router.py +137 -0
- core/api/services/n8n_client.py +77 -0
- core/api/services/newsletter_llm_gateway.py +66 -0
- core/api/services/notification_service.py +134 -0
- core/api/services/openai_responses.py +55 -0
- core/api/services/opencode_metrics.py +375 -0
- core/api/services/opencode_sessions.py +173 -0
- core/api/services/pii_redactor.py +138 -0
- core/api/services/pr_impact_pipeline/__init__.py +21 -0
- core/api/services/pr_impact_pipeline/differ.py +421 -0
- core/api/services/pr_impact_pipeline/dispatcher.py +415 -0
- core/api/services/pr_impact_pipeline/gc.py +93 -0
- core/api/services/pr_impact_pipeline/languages.py +192 -0
- core/api/services/pr_impact_pipeline/parser.py +178 -0
- core/api/services/pr_impact_pipeline/writer.py +394 -0
- core/api/services/pr_service.py +1393 -0
- core/api/services/project_paths.py +70 -0
- core/api/services/project_status_updates.py +265 -0
- core/api/services/providers.py +276 -0
- core/api/services/push_service.py +170 -0
- core/api/services/reminder_service.py +89 -0
- core/api/services/runas.py +41 -0
- core/api/services/salience_service.py +69 -0
- core/api/services/security_collector.py +281 -0
- core/api/services/session_catalog.py +385 -0
- core/api/services/session_metrics_service.py +301 -0
- core/api/services/session_ops.py +272 -0
- core/api/services/session_state.py +173 -0
- core/api/services/share_links.py +222 -0
- core/api/services/task_transitions.py +146 -0
- core/api/services/terminal_metrics.py +462 -0
- core/api/services/terminal_metrics_dump.py +203 -0
- core/api/services/tmux.py +1205 -0
- core/api/services/webhook_service.py +422 -0
- core/api/services/workspace_sync.py +164 -0
- core/api/templates/__init__.py +1 -0
- core/api/templates/markdown_share.py +164 -0
- core/api/terminal.py +1031 -0
- core/api/tests/__init__.py +0 -0
- core/api/tests/test_agent_facing_auth_dependencies.py +132 -0
- core/api/tests/test_audit_permissions.py +133 -0
- core/api/tests/test_backfill_session_conversations.py +90 -0
- core/api/tests/test_backfill_working_seconds_msg.py +129 -0
- core/api/tests/test_claude_metrics.py +326 -0
- core/api/tests/test_codex_metrics.py +189 -0
- core/api/tests/test_finder_paths.py +74 -0
- core/api/tests/test_git_ops_merge.py +155 -0
- core/api/tests/test_learnings_check_search.py +81 -0
- core/api/tests/test_metrics_providers.py +133 -0
- core/api/tests/test_migration_087.py +164 -0
- core/api/tests/test_migration_088.py +94 -0
- core/api/tests/test_migration_089.py +116 -0
- core/api/tests/test_openai_responses.py +24 -0
- core/api/tests/test_opencode_metrics.py +740 -0
- core/api/tests/test_opencode_sessions.py +321 -0
- core/api/tests/test_pr_workflow_e2e.py +457 -0
- core/api/tests/test_projects_handoffs.py +31 -0
- core/api/tests/test_providers.py +138 -0
- core/api/tests/test_safety_bridge.py +347 -0
- core/api/tests/test_session_catalog.py +142 -0
- core/api/tests/test_session_conversations.py +512 -0
- core/api/tests/test_session_metrics_service.py +270 -0
- core/api/tests/test_session_resume_paths.py +548 -0
- core/api/tests/test_session_theme_mode_migration.py +56 -0
- core/api/tests/test_sessions_rbac.py +131 -0
- core/api/tests/test_share_edit.py +398 -0
- core/api/tests/test_share_repo.py +200 -0
- core/api/tests/test_terminal_session_manager.py +98 -0
- core/api/tests/test_terminal_upload.py +34 -0
- core/api/tests/test_tmux.py +272 -0
- core/api/tests/test_workspace_sync.py +186 -0
- core/api/tests/test_ws_ticket_in_memory.py +73 -0
- core/api/use_cases/__init__.py +11 -0
- core/api/use_cases/_context.py +89 -0
- core/api/use_cases/_errors.py +62 -0
- core/api/use_cases/_roles.py +16 -0
- core/api/use_cases/audit.py +171 -0
- core/api/use_cases/brain.py +1232 -0
- core/api/use_cases/costs.py +249 -0
- core/api/use_cases/graph.py +1153 -0
- core/api/use_cases/handoffs.py +506 -0
- core/api/use_cases/ingest_triage.py +1229 -0
- core/api/use_cases/learnings.py +538 -0
- core/api/use_cases/projects.py +705 -0
- core/api/use_cases/pull_requests.py +415 -0
- core/api/use_cases/search.py +926 -0
- core/api/use_cases/tasks.py +1495 -0
- core/api/visibility.py +141 -0
- core/cli/__init__.py +5 -0
- core/cli/_index_source.py +632 -0
- core/cli/_runtime_ctx.py +160 -0
- core/cli/_transmute.py +241 -0
- core/cli/marvis_doctor.py +704 -0
- core/cli/marvis_feedback.py +396 -0
- core/cli/marvis_governance.py +315 -0
- core/cli/marvis_hooks.py +515 -0
- core/cli/marvis_init.py +757 -0
- core/cli/marvis_mcp.py +401 -0
- core/cli/marvis_runtime.py +855 -0
- core/cli/marvis_telemetry.py +228 -0
- core/scripts/_drift_check.py +716 -0
- core/scripts/_frontmatter.py +66 -0
- core/scripts/_graph_writer.py +189 -0
- core/scripts/ast_parser.py +1553 -0
- core/scripts/install_hooks/__init__.py +1 -0
- core/scripts/install_hooks/_config.sh +109 -0
- core/scripts/install_hooks/block-dangerous-bash.sh +23 -0
- core/scripts/install_hooks/block-db-direct-write.sh +23 -0
- core/scripts/install_hooks/block-push-no-task.sh +23 -0
- core/scripts/install_hooks/block-staging-to-prod.sh +23 -0
- core/scripts/install_hooks/block-subtree-push.sh +23 -0
- core/scripts/install_hooks/config.json +53 -0
- core/scripts/install_hooks/enforce-no-merge-main.sh +23 -0
- core/scripts/install_hooks/enforce-worktree.sh +23 -0
- core/scripts/install_hooks/quality-gate.sh +170 -0
- core/scripts/install_hooks/safety_bridge.py +968 -0
- core/scripts/install_hooks/secret-scan.sh +23 -0
- core/scripts/migrate_spike_node_ids.py +122 -0
- core/scripts/populate_artifacts.py +2198 -0
- core/scripts/populate_cross_project.py +2457 -0
- core/scripts/populate_inbox_nodes.py +357 -0
- core/scripts/populate_pr_impact.py +267 -0
- core/scripts/populate_project_nodes.py +603 -0
- core/scripts/populate_touch_counter.py +337 -0
- core/scripts/reparse_failed.py +57 -0
- core/scripts/safety_bridge.py +968 -0
- core/telemetry/__init__.py +9 -0
- core/telemetry/client.py +405 -0
- core/telemetry/schema.py +122 -0
- core/wizard/__init__.py +65 -0
- core/wizard/byok_vault.py +147 -0
- core/wizard/defaults.py +58 -0
- core/wizard/state.py +117 -0
- core/wizard/steps.py +70 -0
- core/wizard/validation.py +136 -0
- marvisx_cli-0.1.0.dist-info/METADATA +201 -0
- marvisx_cli-0.1.0.dist-info/RECORD +587 -0
- marvisx_cli-0.1.0.dist-info/WHEEL +5 -0
- marvisx_cli-0.1.0.dist-info/entry_points.txt +3 -0
- marvisx_cli-0.1.0.dist-info/licenses/LICENSE +98 -0
- marvisx_cli-0.1.0.dist-info/top_level.txt +3 -0
- migrations/001_initial.sql +33 -0
- migrations/002_tasks.sql +30 -0
- migrations/003_session_management.sql +7 -0
- migrations/004_projects_comments.sql +65 -0
- migrations/005_session_intelligence.sql +15 -0
- migrations/006_settings.sql +12 -0
- migrations/007_task_scoring.sql +12 -0
- migrations/008_cost_tracking.sql +31 -0
- migrations/009_session_card_metrics.sql +3 -0
- migrations/010_monitoring.sql +55 -0
- migrations/012_agent_api.sql +21 -0
- migrations/013_session_complete.sql +8 -0
- migrations/015_pull_requests.sql +43 -0
- migrations/015_pull_requests_down.sql +5 -0
- migrations/016_users_raci.sql +116 -0
- migrations/017_task_cost_entries.sql +87 -0
- migrations/018_agents.sql +73 -0
- migrations/018_agents_down.sql +13 -0
- migrations/019_review_feedback.sql +18 -0
- migrations/020_pr_commit_sha.sql +4 -0
- migrations/021_webhook_events.sql +18 -0
- migrations/022_devx_agent_managed.sql +11 -0
- migrations/022_devx_agent_managed_down.sql +6 -0
- migrations/023_devx_p1_gate.sql +7 -0
- migrations/023_devx_p1_gate_down.sql +3 -0
- migrations/024_chat_messages.sql +16 -0
- migrations/024_pr_conversation_id.sql +8 -0
- migrations/024_task_indexes.sql +21 -0
- migrations/024_task_indexes_down.sql +7 -0
- migrations/025_audit_log.sql +17 -0
- migrations/026_agent_tokens.sql +20 -0
- migrations/027_teams_auth_phase_b.sql +35 -0
- migrations/028_learnings.sql +23 -0
- migrations/029_team_roles.sql +14 -0
- migrations/030_finder_pins.sql +10 -0
- migrations/031_pr_deploy_status.sql +9 -0
- migrations/032_task_reminders.sql +7 -0
- migrations/033_events_retry_count.sql +6 -0
- migrations/033_session_owner.sql +9 -0
- migrations/034_notifications.sql +38 -0
- migrations/035_shared_links.sql +15 -0
- migrations/036_session_index_upgrade.sql +29 -0
- migrations/037_pr_approval.sql +15 -0
- migrations/038_pr_submitted_by.sql +6 -0
- migrations/039_push_subscriptions.sql +17 -0
- migrations/040_semantic_search.sql +16 -0
- migrations/041_workspaces.sql +63 -0
- migrations/042_oidc_providers.sql +24 -0
- migrations/043_ci_checks.sql +31 -0
- migrations/044_agent_metrics.sql +30 -0
- migrations/045_documents_doc_type.sql +5 -0
- migrations/046_salience.sql +13 -0
- migrations/047_seed_missing_agents.sql +6 -0
- migrations/048_fix_agent_paths_roles.sql +5 -0
- migrations/049_agent_role_and_learnings_schema.sql +3 -0
- migrations/050_session_provider.sql +2 -0
- migrations/051_session_launch_profile.sql +4 -0
- migrations/052_session_theme_mode.sql +2 -0
- migrations/052_task_kind.sql +4 -0
- migrations/053_inbox_items.sql +31 -0
- migrations/054_inbox_triage_contract.sql +30 -0
- migrations/055_inbox_topic_treatment.sql +12 -0
- migrations/056_inbox_treatment_read_save.sql +57 -0
- migrations/057_session_theme_mode_backfill.sql +4 -0
- migrations/058_inbox_item_status_lifecycle.sql +13 -0
- migrations/059_inbox_tldr_and_source_scores.sql +18 -0
- migrations/060_newsletter.sql +16 -0
- migrations/061_inbox_redesign.sql +69 -0
- migrations/062_fix_inbox_sources_backfill.sql +37 -0
- migrations/063_task_completion_mode.sql +23 -0
- migrations/064_judge_mode_setting.sql +4 -0
- migrations/065_knowledge_graph_spike.sql +40 -0
- migrations/066_digest_ranking_inputs.sql +10 -0
- migrations/066_kg_artifact_nodes.sql +129 -0
- migrations/067_inbox_digest_selections.sql +28 -0
- migrations/067_kg_temporal.sql +53 -0
- migrations/068_inbox_digest_app_settings.sql +9 -0
- migrations/068_kg_touch_counter.sql +52 -0
- migrations/069_kg_doc_types.sql +117 -0
- migrations/070_digest_ranking_inputs_recovery.sql +3 -0
- migrations/071_inbox_digest_selections_recovery.sql +3 -0
- migrations/072_inbox_digest_app_settings_recovery.sql +3 -0
- migrations/073_kg_cross_project.sql +216 -0
- migrations/073_kg_cross_project_down.sql +77 -0
- migrations/074_kg_infra_types.sql +208 -0
- migrations/074_kg_infra_types_down.sql +80 -0
- migrations/075_kg_file_state_recovery.sql +35 -0
- migrations/075_kg_file_state_recovery_down.sql +5 -0
- migrations/076_kg_watcher_state.sql +33 -0
- migrations/076_kg_watcher_state_down.sql +3 -0
- migrations/077_kg_doc_types_extend.sql +226 -0
- migrations/077_kg_doc_types_extend_down.sql +80 -0
- migrations/078_kg_fts5.sql +102 -0
- migrations/078_kg_fts5_down.sql +14 -0
- migrations/079_kg_missing_indexes.sql +31 -0
- migrations/079_kg_missing_indexes_down.sql +10 -0
- migrations/080_kg_fts5_extended.sql +232 -0
- migrations/080_kg_fts5_extended_down.sql +25 -0
- migrations/081_kg_lens_indexes.sql +9 -0
- migrations/081_kg_lens_indexes_down.sql +3 -0
- migrations/082_kg_pins.sql +26 -0
- migrations/082_kg_pins_down.sql +14 -0
- migrations/083_kg_graph_nodes_degree.sql +20 -0
- migrations/083_kg_graph_nodes_degree_down.sql +15 -0
- migrations/084_drop_legacy_scheduler_tables.sql +58 -0
- migrations/084_drop_legacy_scheduler_tables_down.sql +112 -0
- migrations/085_kg_edge_resolves_to.sql +142 -0
- migrations/085_kg_edge_resolves_to_down.sql +66 -0
- migrations/086_project_status_updates_feed.sql +20 -0
- migrations/086_project_status_updates_feed_down.sql +36 -0
- migrations/087_session_metrics_dual.sql +50 -0
- migrations/087_session_metrics_dual_down.sql +21 -0
- migrations/088_rename_context_pct_legacy.sql +23 -0
- migrations/088_rename_context_pct_legacy_down.sql +8 -0
- migrations/089_session_metrics_equivalent_cost.sql +26 -0
- migrations/089_session_metrics_equivalent_cost_down.sql +11 -0
- migrations/090_kg_inbox_node_type.sql +26 -0
- migrations/090_kg_inbox_node_type_down.sql +20 -0
- migrations/091_kg_inbox_node_type_check.sql +265 -0
- migrations/091_kg_inbox_node_type_check_down.sql +129 -0
- migrations/092_sessions_activity_state_ts.sql +29 -0
- migrations/092_sessions_activity_state_ts_down.sql +14 -0
- migrations/093_sessions_activity_state_column.sql +29 -0
- migrations/093_sessions_activity_state_column_down.sql +10 -0
- migrations/094_ingest_pending.sql +55 -0
- migrations/094_ingest_pending_down.sql +15 -0
- migrations/095_kg_intent_first.sql +77 -0
- migrations/095_kg_intent_first_down.sql +25 -0
- migrations/096_kg_xlsx_artifact_prefix.sql +17 -0
- migrations/096_kg_xlsx_artifact_prefix_down.sql +11 -0
- migrations/097_ingest_change_history.sql +37 -0
- migrations/097_ingest_change_history_down.sql +13 -0
- migrations/098_kg_node_type_business.sql +254 -0
- migrations/098_kg_node_type_business_down.sql +195 -0
- migrations/099_kg_edges_restore_weight.sql +58 -0
- migrations/099_kg_edges_restore_weight_down.sql +12 -0
- migrations/100_kg_enriched_at.sql +25 -0
- migrations/100_kg_enriched_at_down.sql +12 -0
- migrations/101_local_llm_shadow_comparisons.sql +66 -0
- migrations/101_local_llm_shadow_comparisons_down.sql +15 -0
- migrations/102_promote_llm_costs.sql +69 -0
- migrations/102_promote_llm_costs_down.sql +19 -0
- migrations/103_ingest_skipped_log.sql +46 -0
- migrations/103_ingest_skipped_log_down.sql +15 -0
- migrations/120_docs_governance.sql +50 -0
- migrations/120_docs_governance_down.sql +11 -0
- migrations/121_notification_event_fk_cleanup.sql +21 -0
- migrations/121_notification_event_fk_cleanup_down.sql +10 -0
- migrations/122_docs_drift_history.sql +34 -0
- migrations/122_docs_drift_history_down.sql +15 -0
- migrations/123_ingest_parser_waiting_status.sql +69 -0
- migrations/123_ingest_parser_waiting_status_down.sql +69 -0
- migrations/124_heypocket_recordings.sql +63 -0
- migrations/124_heypocket_recordings_down.sql +13 -0
- migrations/125_kg_node_type_record.sql +219 -0
- migrations/125_kg_node_type_record_down.sql +205 -0
- migrations/126_ingest_terminal_upload_source_kind.sql +69 -0
- migrations/126_ingest_terminal_upload_source_kind_down.sql +69 -0
- migrations/127_brain_v1_substrate.sql +200 -0
- migrations/127_brain_v1_substrate_down.sql +32 -0
- migrations/128_brain_drift_signals.sql +157 -0
- migrations/128_brain_drift_signals_down.sql +23 -0
- migrations/129_brain_memory_operations.sql +232 -0
- migrations/129_brain_memory_operations_down.sql +27 -0
- migrations/130_brain_findings.sql +258 -0
- migrations/130_brain_findings_down.sql +29 -0
- migrations/132_kg_pr_modifies.sql +242 -0
- migrations/132_kg_pr_modifies_down.sql +99 -0
- migrations/133_brain_v1_2_direction_schema.sql +476 -0
- migrations/133_brain_v1_2_direction_schema_down.sql +273 -0
- migrations/134_brain_journal_narrative_polished.sql +8 -0
- migrations/134_brain_journal_narrative_polished_down.sql +6 -0
- migrations/135_kg_edges_provider.sql +21 -0
- migrations/136_documents_fts.sql +56 -0
- migrations/137_promote_llm_costs.sql +59 -0
- migrations/137_promote_llm_costs_down.sql +19 -0
- migrations/138_ingest_api_keys.sql +39 -0
- migrations/138_ingest_api_keys_down.sql +8 -0
- migrations/139_ingest_pending_ingress.sql +91 -0
- migrations/139_ingest_pending_ingress_down.sql +73 -0
- migrations/140_ingest_idempotency_quota.sql +45 -0
- migrations/140_ingest_idempotency_quota_down.sql +9 -0
- migrations/141_ingest_pending_metadata.sql +16 -0
- migrations/141_ingest_pending_metadata_down.sql +7 -0
- migrations/142_llm_function_config.sql +36 -0
- migrations/142_llm_function_config_down.sql +8 -0
- migrations/143_kg_code_embeddings.sql +25 -0
- migrations/143_kg_code_embeddings_down.sql +5 -0
- migrations/__init__.py +4 -0
- projects/_template/project.yaml +46 -0
|
@@ -0,0 +1,1756 @@
|
|
|
1
|
+
"""Parser dispatch for ingest_pending rows."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import asyncio
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import mimetypes
|
|
9
|
+
import os
|
|
10
|
+
import sqlite3
|
|
11
|
+
from collections.abc import Awaitable, Callable
|
|
12
|
+
from dataclasses import dataclass, replace
|
|
13
|
+
from pathlib import Path
|
|
14
|
+
from typing import Any, TypeVar
|
|
15
|
+
|
|
16
|
+
from core.api.config import settings
|
|
17
|
+
from core.api.db import acquire_db, acquire_write_db
|
|
18
|
+
from core.api.services import pii_redactor
|
|
19
|
+
from core.api.services.ingest.auto_approve import decide_ingress_routing, should_auto_approve
|
|
20
|
+
from core.api.services.ingest.classifier import ALLOWED_TARGETS, classify_markdown
|
|
21
|
+
from core.api.services.ingest.confidence import (
|
|
22
|
+
compute_composite_confidence,
|
|
23
|
+
estimate_parser_quality,
|
|
24
|
+
)
|
|
25
|
+
from core.api.services.ingest.events import broadcast_ingest_changed
|
|
26
|
+
from core.api.services.ingest.parsers.docling_parser import parse_pdf_file
|
|
27
|
+
from core.api.services.ingest.parsers.docparse_gateway import parse_pdf_docparse
|
|
28
|
+
from core.api.services.ingest.parsers.docx_parser import DOCX_MIME_TYPE, parse_docx
|
|
29
|
+
from core.api.services.ingest.parsers.gateway_aux import MissingGatewayConfig
|
|
30
|
+
from core.api.services.ingest.parsers.image_parser import (
|
|
31
|
+
SUPPORTED_IMAGE_SUFFIXES,
|
|
32
|
+
UNSUPPORTED_PHASE1_SUFFIXES,
|
|
33
|
+
parse_image_with_gateway,
|
|
34
|
+
)
|
|
35
|
+
from core.api.services.ingest.parsers.internal_markdown import parse_markdown_file
|
|
36
|
+
from core.api.services.ingest.parsers.ocr_pdf_parser import parse_pdf_ocr
|
|
37
|
+
from core.api.services.ingest.parsers.transcript_parser import (
|
|
38
|
+
AUDIO_MIME_BY_SUFFIX,
|
|
39
|
+
SUPPORTED_AUDIO_SUFFIXES,
|
|
40
|
+
SUPPORTED_VIDEO_SUFFIXES,
|
|
41
|
+
VIDEO_MIME_BY_SUFFIX,
|
|
42
|
+
parse_media_transcript,
|
|
43
|
+
)
|
|
44
|
+
from core.api.services.ingest.parsers.xlsx_parser import parse_xlsx
|
|
45
|
+
from core.api.services.ingest.parsers.vision_gateway import parse_vision_with_gateway
|
|
46
|
+
from core.api.services.ingest.preflight import build_classifier_content, build_preflight
|
|
47
|
+
from core.api.services.ingest.routing_policy import IngestRoute, choose_route
|
|
48
|
+
|
|
49
|
+
logger = logging.getLogger(__name__)
|
|
50
|
+
T = TypeVar("T")
|
|
51
|
+
|
|
52
|
+
try:
|
|
53
|
+
import magic
|
|
54
|
+
|
|
55
|
+
_MAGIC = magic.Magic(mime=True)
|
|
56
|
+
except Exception: # pragma: no cover - python-magic/libmagic may be absent in old envs
|
|
57
|
+
_MAGIC = None
|
|
58
|
+
_MARKDOWN_SUFFIXES = {".md", ".markdown", ".txt"}
|
|
59
|
+
_PDF_SUFFIXES = {".pdf"}
|
|
60
|
+
# Inlined to avoid circular import with api.services.ingest.watcher (which imports
|
|
61
|
+
# parse_pending from this module). Same constant lives in watcher.py:28 + insert_saga.py:22.
|
|
62
|
+
PROJECTS_ROOT = Path("/data/projects")
|
|
63
|
+
_XLSX_MIME_BY_SUFFIX = {
|
|
64
|
+
".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
|
|
65
|
+
".xlsm": "application/vnd.ms-excel.sheet.macroEnabled.12",
|
|
66
|
+
}
|
|
67
|
+
_DOCX_SUFFIXES = {".docx"}
|
|
68
|
+
_IMAGE_MIME_BY_SUFFIX = {
|
|
69
|
+
".avif": "image/avif",
|
|
70
|
+
".heic": "image/heic",
|
|
71
|
+
".heif": "image/heif",
|
|
72
|
+
".jpeg": "image/jpeg",
|
|
73
|
+
".jpg": "image/jpeg",
|
|
74
|
+
".png": "image/png",
|
|
75
|
+
".webp": "image/webp",
|
|
76
|
+
}
|
|
77
|
+
_MEDIA_SUFFIXES = SUPPORTED_AUDIO_SUFFIXES | SUPPORTED_VIDEO_SUFFIXES
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def detect_mime(path: Path) -> str:
|
|
81
|
+
if path.suffix.lower() in _MARKDOWN_SUFFIXES:
|
|
82
|
+
return "text/markdown"
|
|
83
|
+
if path.suffix.lower() in _PDF_SUFFIXES:
|
|
84
|
+
return "application/pdf"
|
|
85
|
+
if path.suffix.lower() in _XLSX_MIME_BY_SUFFIX:
|
|
86
|
+
return _XLSX_MIME_BY_SUFFIX[path.suffix.lower()]
|
|
87
|
+
if path.suffix.lower() in _DOCX_SUFFIXES:
|
|
88
|
+
return DOCX_MIME_TYPE
|
|
89
|
+
if path.suffix.lower() in _IMAGE_MIME_BY_SUFFIX:
|
|
90
|
+
return _IMAGE_MIME_BY_SUFFIX[path.suffix.lower()]
|
|
91
|
+
if path.suffix.lower() in AUDIO_MIME_BY_SUFFIX:
|
|
92
|
+
return AUDIO_MIME_BY_SUFFIX[path.suffix.lower()]
|
|
93
|
+
if path.suffix.lower() in VIDEO_MIME_BY_SUFFIX:
|
|
94
|
+
return VIDEO_MIME_BY_SUFFIX[path.suffix.lower()]
|
|
95
|
+
if _MAGIC is not None:
|
|
96
|
+
try:
|
|
97
|
+
return str(_MAGIC.from_file(str(path)))
|
|
98
|
+
except Exception:
|
|
99
|
+
logger.exception("python-magic failed for %s", path)
|
|
100
|
+
guessed, _ = mimetypes.guess_type(str(path))
|
|
101
|
+
return guessed or "application/octet-stream"
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _classification_filename(path: Path, mime_type: str) -> str:
|
|
105
|
+
if mime_type == "application/pdf":
|
|
106
|
+
return f"{path.stem}.md"
|
|
107
|
+
if path.suffix.lower() in _MEDIA_SUFFIXES or mime_type.startswith(
|
|
108
|
+
("audio/", "video/")
|
|
109
|
+
):
|
|
110
|
+
return f"{path.stem}.md"
|
|
111
|
+
return path.name
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _is_image(path: Path, mime_type: str) -> bool:
|
|
115
|
+
suffix = path.suffix.lower()
|
|
116
|
+
return (
|
|
117
|
+
suffix in SUPPORTED_IMAGE_SUFFIXES
|
|
118
|
+
or suffix in UNSUPPORTED_PHASE1_SUFFIXES
|
|
119
|
+
or mime_type.startswith("image/")
|
|
120
|
+
)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _is_xlsx(path: Path, mime_type: str) -> bool:
|
|
124
|
+
return path.suffix.lower() in _XLSX_MIME_BY_SUFFIX or mime_type in set(
|
|
125
|
+
_XLSX_MIME_BY_SUFFIX.values()
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def _is_docx(path: Path, mime_type: str) -> bool:
|
|
130
|
+
return path.suffix.lower() in _DOCX_SUFFIXES or mime_type == DOCX_MIME_TYPE
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def _is_transcript_media(path: Path, mime_type: str) -> bool:
|
|
134
|
+
return path.suffix.lower() in _MEDIA_SUFFIXES or mime_type.startswith(
|
|
135
|
+
("audio/", "video/")
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def _metadata_sidecar_path(path: Path) -> Path:
|
|
140
|
+
return path.with_suffix(".metadata.json")
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _image_classification(
|
|
144
|
+
*,
|
|
145
|
+
path: Path,
|
|
146
|
+
auto_approve: bool,
|
|
147
|
+
reason: str,
|
|
148
|
+
rules_matched: list[str],
|
|
149
|
+
) -> dict[str, Any]:
|
|
150
|
+
return {
|
|
151
|
+
"type": "file",
|
|
152
|
+
"title": path.name,
|
|
153
|
+
"tags": ["image", path.suffix.lower().lstrip(".")],
|
|
154
|
+
"target_folder": "docs/assets",
|
|
155
|
+
"target_filename": path.name,
|
|
156
|
+
"confidence": 0.82 if auto_approve else 0.52,
|
|
157
|
+
"reason": reason,
|
|
158
|
+
"auto_approve": auto_approve,
|
|
159
|
+
"rules_matched": rules_matched,
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _xlsx_classification(path: Path) -> dict[str, Any]:
|
|
164
|
+
return {
|
|
165
|
+
"type": "file",
|
|
166
|
+
"title": path.name,
|
|
167
|
+
"tags": ["xlsx", "spreadsheet"],
|
|
168
|
+
"target_folder": "docs/assets",
|
|
169
|
+
"target_filename": path.name,
|
|
170
|
+
"confidence": 0.74,
|
|
171
|
+
"reason": "spreadsheet requires manual triage",
|
|
172
|
+
"auto_approve": False,
|
|
173
|
+
}
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
LLM_AUTO_APPROVE_THRESHOLD = 0.80
|
|
177
|
+
IDENTITY_AUTO_APPROVE_TERMS = (
|
|
178
|
+
"atto di nascita",
|
|
179
|
+
"carta d'ident",
|
|
180
|
+
"carta ident",
|
|
181
|
+
"codice fiscale",
|
|
182
|
+
"cognome e nome",
|
|
183
|
+
"documento d'ident",
|
|
184
|
+
"documento ident",
|
|
185
|
+
"fiscal code",
|
|
186
|
+
"name and surname",
|
|
187
|
+
"passport",
|
|
188
|
+
"passaporto",
|
|
189
|
+
)
|
|
190
|
+
_PARSER_LANE_LIMITS = {
|
|
191
|
+
"local": max(
|
|
192
|
+
1,
|
|
193
|
+
int(
|
|
194
|
+
settings.ingest_local_parser_max_concurrency
|
|
195
|
+
or settings.ingest_parser_max_concurrency
|
|
196
|
+
),
|
|
197
|
+
),
|
|
198
|
+
"ocr": max(1, int(settings.ingest_ocr_max_concurrency)),
|
|
199
|
+
"docparse": max(1, int(settings.ingest_docparse_max_concurrency)),
|
|
200
|
+
"transcribe": max(1, int(settings.ingest_transcribe_max_concurrency)),
|
|
201
|
+
"vision": max(1, int(settings.ingest_vision_max_concurrency)),
|
|
202
|
+
}
|
|
203
|
+
_PARSER_LANE_SEMAPHORES = {
|
|
204
|
+
lane: asyncio.Semaphore(limit) for lane, limit in _PARSER_LANE_LIMITS.items()
|
|
205
|
+
}
|
|
206
|
+
_TRANSIENT_PARSE_ERROR_MARKERS = (
|
|
207
|
+
"rate limited",
|
|
208
|
+
"unavailable after",
|
|
209
|
+
"unavailable: http 408",
|
|
210
|
+
"unavailable: http 409",
|
|
211
|
+
"unavailable: http 425",
|
|
212
|
+
"unavailable: http 429",
|
|
213
|
+
"unavailable: http 500",
|
|
214
|
+
"unavailable: http 502",
|
|
215
|
+
"unavailable: http 503",
|
|
216
|
+
"unavailable: http 504",
|
|
217
|
+
)
|
|
218
|
+
_OCR_EMPTY_TEXT_MARKER = "tier-ocr returned empty text"
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
@dataclass(frozen=True)
|
|
222
|
+
class ParseDispatchResult:
|
|
223
|
+
parsed: Any
|
|
224
|
+
parser_used: str
|
|
225
|
+
route: IngestRoute
|
|
226
|
+
preflight: dict[str, Any]
|
|
227
|
+
parser_quality: dict[str, Any]
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
class _ParserWaitCancelled(Exception):
|
|
231
|
+
"""Raised when a row leaves parser_waiting before the parser slot starts."""
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
def _parser_lane_for_workflow(workflow: str) -> str:
|
|
235
|
+
if workflow in {"ocr", "docparse", "transcribe", "vision"}:
|
|
236
|
+
return workflow
|
|
237
|
+
return "local"
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _parser_lane_waiter_count(lane: str) -> int:
|
|
241
|
+
waiters = getattr(_PARSER_LANE_SEMAPHORES[lane], "_waiters", None)
|
|
242
|
+
if waiters is None:
|
|
243
|
+
return 0
|
|
244
|
+
return sum(1 for waiter in waiters if not waiter.done())
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
async def _mark_parser_waiting(ingest_id: str, project_slug: str) -> None:
|
|
248
|
+
async with acquire_write_db() as db:
|
|
249
|
+
await db.execute(
|
|
250
|
+
"""
|
|
251
|
+
UPDATE ingest_pending
|
|
252
|
+
SET status = 'parser_waiting',
|
|
253
|
+
error_message = NULL,
|
|
254
|
+
updated_at = datetime('now')
|
|
255
|
+
WHERE id = ?
|
|
256
|
+
AND status IN ('queued', 'parse_error', 'parsing')
|
|
257
|
+
""",
|
|
258
|
+
(ingest_id,),
|
|
259
|
+
)
|
|
260
|
+
await db.commit()
|
|
261
|
+
await broadcast_ingest_changed(
|
|
262
|
+
"parser_waiting",
|
|
263
|
+
ingest_id=ingest_id,
|
|
264
|
+
project_slug=project_slug,
|
|
265
|
+
status="parser_waiting",
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
async def _mark_parser_active(ingest_id: str, project_slug: str) -> None:
|
|
270
|
+
async with acquire_write_db() as db:
|
|
271
|
+
cursor = await db.execute(
|
|
272
|
+
"""
|
|
273
|
+
UPDATE ingest_pending
|
|
274
|
+
SET status = 'parsing',
|
|
275
|
+
updated_at = datetime('now')
|
|
276
|
+
WHERE id = ?
|
|
277
|
+
AND status = 'parser_waiting'
|
|
278
|
+
""",
|
|
279
|
+
(ingest_id,),
|
|
280
|
+
)
|
|
281
|
+
await db.commit()
|
|
282
|
+
if cursor.rowcount != 1:
|
|
283
|
+
raise _ParserWaitCancelled(ingest_id)
|
|
284
|
+
await broadcast_ingest_changed(
|
|
285
|
+
"parsing",
|
|
286
|
+
ingest_id=ingest_id,
|
|
287
|
+
project_slug=project_slug,
|
|
288
|
+
status="parsing",
|
|
289
|
+
)
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
async def _mark_parse_error(
|
|
293
|
+
ingest_id: str,
|
|
294
|
+
project_slug: str,
|
|
295
|
+
message: str,
|
|
296
|
+
*,
|
|
297
|
+
attempts: int = 3,
|
|
298
|
+
) -> None:
|
|
299
|
+
for attempt in range(1, attempts + 1):
|
|
300
|
+
try:
|
|
301
|
+
async with acquire_write_db(label="ingest.parse_error") as db:
|
|
302
|
+
await db.execute(
|
|
303
|
+
"""
|
|
304
|
+
UPDATE ingest_pending
|
|
305
|
+
SET status = 'parse_error',
|
|
306
|
+
error_message = ?,
|
|
307
|
+
updated_at = datetime('now')
|
|
308
|
+
WHERE id = ?
|
|
309
|
+
""",
|
|
310
|
+
(message[:1000], ingest_id),
|
|
311
|
+
)
|
|
312
|
+
await db.commit()
|
|
313
|
+
await broadcast_ingest_changed(
|
|
314
|
+
"parse_error",
|
|
315
|
+
ingest_id=ingest_id,
|
|
316
|
+
project_slug=project_slug,
|
|
317
|
+
status="parse_error",
|
|
318
|
+
)
|
|
319
|
+
return
|
|
320
|
+
except (sqlite3.OperationalError, RuntimeError) as exc:
|
|
321
|
+
if attempt >= attempts:
|
|
322
|
+
logger.exception(
|
|
323
|
+
"ingest parse_error write failed permanently: id=%s",
|
|
324
|
+
ingest_id,
|
|
325
|
+
)
|
|
326
|
+
return
|
|
327
|
+
delay = min(0.2 * (2 ** (attempt - 1)), 1.0)
|
|
328
|
+
logger.warning(
|
|
329
|
+
"ingest parse_error write failed; retrying attempt=%d/%d id=%s error=%s",
|
|
330
|
+
attempt,
|
|
331
|
+
attempts,
|
|
332
|
+
ingest_id,
|
|
333
|
+
exc,
|
|
334
|
+
)
|
|
335
|
+
await asyncio.sleep(delay)
|
|
336
|
+
|
|
337
|
+
|
|
338
|
+
async def _run_heavy_parser(
|
|
339
|
+
*,
|
|
340
|
+
ingest_id: str,
|
|
341
|
+
project_slug: str,
|
|
342
|
+
lane: str,
|
|
343
|
+
parser_name: str,
|
|
344
|
+
path: Path,
|
|
345
|
+
parse: Callable[[], Awaitable[T]],
|
|
346
|
+
) -> T:
|
|
347
|
+
semaphore = _PARSER_LANE_SEMAPHORES[lane]
|
|
348
|
+
limit = _PARSER_LANE_LIMITS[lane]
|
|
349
|
+
waiters = _parser_lane_waiter_count(lane)
|
|
350
|
+
if semaphore.locked() or waiters:
|
|
351
|
+
logger.info(
|
|
352
|
+
"ingest parser waiting: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
|
|
353
|
+
lane,
|
|
354
|
+
parser_name,
|
|
355
|
+
path,
|
|
356
|
+
limit,
|
|
357
|
+
waiters,
|
|
358
|
+
)
|
|
359
|
+
|
|
360
|
+
async with semaphore:
|
|
361
|
+
await _mark_parser_active(ingest_id, project_slug)
|
|
362
|
+
logger.info(
|
|
363
|
+
"ingest parser acquired: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
|
|
364
|
+
lane,
|
|
365
|
+
parser_name,
|
|
366
|
+
path,
|
|
367
|
+
limit,
|
|
368
|
+
_parser_lane_waiter_count(lane),
|
|
369
|
+
)
|
|
370
|
+
try:
|
|
371
|
+
return await parse()
|
|
372
|
+
finally:
|
|
373
|
+
logger.info(
|
|
374
|
+
"ingest parser released: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
|
|
375
|
+
lane,
|
|
376
|
+
parser_name,
|
|
377
|
+
path,
|
|
378
|
+
limit,
|
|
379
|
+
_parser_lane_waiter_count(lane),
|
|
380
|
+
)
|
|
381
|
+
|
|
382
|
+
|
|
383
|
+
def _llm_classifier_enabled() -> bool:
|
|
384
|
+
"""Read LLM_CLASSIFIER_ENABLED env (true|shadow|false). Default: false."""
|
|
385
|
+
value = (os.environ.get("LLM_CLASSIFIER_ENABLED", "false") or "").strip().lower()
|
|
386
|
+
return value == "true"
|
|
387
|
+
|
|
388
|
+
|
|
389
|
+
def _llm_classifier_shadow() -> bool:
|
|
390
|
+
value = (os.environ.get("LLM_CLASSIFIER_ENABLED", "false") or "").strip().lower()
|
|
391
|
+
return value == "shadow"
|
|
392
|
+
|
|
393
|
+
|
|
394
|
+
def _ingest_llm_provider() -> str:
|
|
395
|
+
return settings.ingest_llm_provider.strip().lower()
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
def _ingest_llm_model() -> str:
|
|
399
|
+
return settings.ingest_llm_classifier_model
|
|
400
|
+
|
|
401
|
+
|
|
402
|
+
async def _resolve_classify_provider():
|
|
403
|
+
"""Resolve the BYOK 'classify' provider from llm_function_config, or None.
|
|
404
|
+
|
|
405
|
+
Reads on the read pool, fail-soft (None on any error) so the deterministic
|
|
406
|
+
path is never broken by a config/DB hiccup.
|
|
407
|
+
"""
|
|
408
|
+
try:
|
|
409
|
+
from core.api.services.ingest.llm.config_store import resolve_function_provider
|
|
410
|
+
|
|
411
|
+
async with acquire_db() as cfg_db:
|
|
412
|
+
return await resolve_function_provider(cfg_db, "classify")
|
|
413
|
+
except Exception: # noqa: BLE001
|
|
414
|
+
logger.debug("byok classify provider resolution failed", exc_info=True)
|
|
415
|
+
return None
|
|
416
|
+
|
|
417
|
+
|
|
418
|
+
def _with_llm_no_result(
|
|
419
|
+
base_classification: dict[str, Any],
|
|
420
|
+
*,
|
|
421
|
+
status: str,
|
|
422
|
+
reason: str,
|
|
423
|
+
extra: dict[str, Any] | None = None,
|
|
424
|
+
) -> dict[str, Any]:
|
|
425
|
+
merged = dict(base_classification)
|
|
426
|
+
metadata = dict(merged.get("llm_metadata") or {})
|
|
427
|
+
metadata.update(
|
|
428
|
+
{
|
|
429
|
+
"status": status,
|
|
430
|
+
"model": _ingest_llm_model(),
|
|
431
|
+
"provider": _ingest_llm_provider(),
|
|
432
|
+
"auto_approved": False,
|
|
433
|
+
"reason": reason,
|
|
434
|
+
}
|
|
435
|
+
)
|
|
436
|
+
if extra:
|
|
437
|
+
metadata.update(extra)
|
|
438
|
+
merged["llm_metadata"] = metadata
|
|
439
|
+
return merged
|
|
440
|
+
|
|
441
|
+
|
|
442
|
+
_LLM_HARD_FAILURE_STATUSES = {
|
|
443
|
+
"api_error",
|
|
444
|
+
"bad_response",
|
|
445
|
+
"client_init_failed",
|
|
446
|
+
"exception",
|
|
447
|
+
"factory_failed",
|
|
448
|
+
"json_parse_failed",
|
|
449
|
+
"no_result",
|
|
450
|
+
"unavailable",
|
|
451
|
+
}
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def _llm_failure_error_message(classification_json: dict[str, Any]) -> str | None:
|
|
455
|
+
metadata = classification_json.get("llm_metadata")
|
|
456
|
+
if not isinstance(metadata, dict):
|
|
457
|
+
return None
|
|
458
|
+
status = str(metadata.get("status") or "")
|
|
459
|
+
if status not in _LLM_HARD_FAILURE_STATUSES:
|
|
460
|
+
return None
|
|
461
|
+
reason = str(metadata.get("reason") or "llm_classifier_failed")
|
|
462
|
+
message = str(
|
|
463
|
+
metadata.get("gateway_error_message")
|
|
464
|
+
or metadata.get("error_message")
|
|
465
|
+
or reason
|
|
466
|
+
)
|
|
467
|
+
return f"E5 LLM enrichment failed after retries: {status} ({message})"[:1000]
|
|
468
|
+
|
|
469
|
+
|
|
470
|
+
async def _maybe_summarize_transcript(
|
|
471
|
+
*,
|
|
472
|
+
ingest_id: str,
|
|
473
|
+
parser_used: str,
|
|
474
|
+
extracted_text: str,
|
|
475
|
+
structure: dict[str, Any],
|
|
476
|
+
) -> dict[str, Any] | None:
|
|
477
|
+
if parser_used != "tier_transcribe" or not extracted_text.strip():
|
|
478
|
+
return None
|
|
479
|
+
if not _llm_classifier_enabled():
|
|
480
|
+
return None
|
|
481
|
+
|
|
482
|
+
try:
|
|
483
|
+
from core.api.services.ingest.llm.local_gateway import (
|
|
484
|
+
summarize_transcript_with_local_gateway,
|
|
485
|
+
)
|
|
486
|
+
except Exception: # noqa: BLE001
|
|
487
|
+
logger.debug("transcript summarizer import failed", exc_info=True)
|
|
488
|
+
return {
|
|
489
|
+
"status": "import_failed",
|
|
490
|
+
"reason": "transcript_summarizer_import_failed",
|
|
491
|
+
}
|
|
492
|
+
|
|
493
|
+
try:
|
|
494
|
+
result, diagnostics = await summarize_transcript_with_local_gateway(
|
|
495
|
+
extracted_text,
|
|
496
|
+
structure=structure,
|
|
497
|
+
idempotency_scope=f"ingest:{ingest_id}",
|
|
498
|
+
)
|
|
499
|
+
except Exception: # noqa: BLE001 - summary must never break ingest
|
|
500
|
+
logger.warning("transcript summarizer raised", exc_info=True)
|
|
501
|
+
return {
|
|
502
|
+
"status": "exception",
|
|
503
|
+
"reason": "transcript_summarizer_exception",
|
|
504
|
+
}
|
|
505
|
+
|
|
506
|
+
if result is None:
|
|
507
|
+
return _transcript_summary_failure_metadata(diagnostics)
|
|
508
|
+
|
|
509
|
+
return {
|
|
510
|
+
"status": "ok",
|
|
511
|
+
"model": _ingest_llm_model(),
|
|
512
|
+
"provider": _ingest_llm_provider(),
|
|
513
|
+
"summary": result.summary,
|
|
514
|
+
"topics": list(result.topics),
|
|
515
|
+
"participants": list(result.participants),
|
|
516
|
+
"keywords": list(result.keywords),
|
|
517
|
+
"action_items": list(result.action_items),
|
|
518
|
+
"confidence": result.confidence,
|
|
519
|
+
}
|
|
520
|
+
|
|
521
|
+
|
|
522
|
+
def _transcript_summary_failure_metadata(
|
|
523
|
+
diagnostics: dict[str, Any] | None,
|
|
524
|
+
) -> dict[str, Any]:
|
|
525
|
+
metadata = {
|
|
526
|
+
"status": "no_result",
|
|
527
|
+
"model": _ingest_llm_model(),
|
|
528
|
+
"provider": _ingest_llm_provider(),
|
|
529
|
+
"reason": "transcript_summarizer_returned_none",
|
|
530
|
+
}
|
|
531
|
+
if isinstance(diagnostics, dict):
|
|
532
|
+
for key in (
|
|
533
|
+
"status",
|
|
534
|
+
"reason",
|
|
535
|
+
"gateway_status_code",
|
|
536
|
+
"gateway_error_code",
|
|
537
|
+
"gateway_error_message",
|
|
538
|
+
"schema_retry_attempted",
|
|
539
|
+
"raw_excerpt",
|
|
540
|
+
"first_raw_excerpt",
|
|
541
|
+
):
|
|
542
|
+
if key in diagnostics:
|
|
543
|
+
metadata[key] = diagnostics[key]
|
|
544
|
+
return metadata
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def _ingest_event_for_status(status: str) -> str:
|
|
548
|
+
if status == "done":
|
|
549
|
+
return "done"
|
|
550
|
+
if status == "rejected":
|
|
551
|
+
return "rejected"
|
|
552
|
+
if status == "parse_error":
|
|
553
|
+
return "parse_error"
|
|
554
|
+
return "parsed"
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
async def _maybe_llm_enrich(
|
|
558
|
+
*,
|
|
559
|
+
ingest_id: str | None = None,
|
|
560
|
+
extracted_text: str,
|
|
561
|
+
base_classification: dict[str, Any],
|
|
562
|
+
preflight: dict[str, Any],
|
|
563
|
+
parser_quality: dict[str, Any],
|
|
564
|
+
route: IngestRoute,
|
|
565
|
+
) -> dict[str, Any] | None:
|
|
566
|
+
"""Run the local/cloud LLM classifier on a bounded evidence packet.
|
|
567
|
+
|
|
568
|
+
Returns the enriched classification dict (with auto_approve, llm_metadata,
|
|
569
|
+
suggested project_slug) only when the project_slug resolves to a valid
|
|
570
|
+
project and the composite confidence gate passes.
|
|
571
|
+
|
|
572
|
+
Reads context OUTSIDE the writer lock (M-D7).
|
|
573
|
+
"""
|
|
574
|
+
enabled = _llm_classifier_enabled()
|
|
575
|
+
shadow = _llm_classifier_shadow()
|
|
576
|
+
# BYOK (U4): a DB-configured provider for the 'classify' function also enables
|
|
577
|
+
# auto-classify (no env flag needed). The deterministic classifier stays
|
|
578
|
+
# PRIMARY; this only gates the optional LLM override.
|
|
579
|
+
resolved_provider = await _resolve_classify_provider()
|
|
580
|
+
byok = resolved_provider is not None
|
|
581
|
+
if not (enabled or shadow or byok):
|
|
582
|
+
# No env flag and no configured BYOK provider: auto-classify is disabled.
|
|
583
|
+
# The deterministic classification (already computed) remains and the item
|
|
584
|
+
# routes to triage — no heuristic semantic guess (R10/D6).
|
|
585
|
+
return None
|
|
586
|
+
|
|
587
|
+
try:
|
|
588
|
+
from core.api.services.ingest.llm.classification_context import (
|
|
589
|
+
gather_classification_context,
|
|
590
|
+
)
|
|
591
|
+
from core.api.services.ingest.llm.factory import get_classifier
|
|
592
|
+
except Exception: # noqa: BLE001 - module unavailable in some test envs
|
|
593
|
+
logger.debug("llm classifier import failed", exc_info=True)
|
|
594
|
+
return None
|
|
595
|
+
|
|
596
|
+
classifier_content = build_classifier_content(
|
|
597
|
+
extracted_text=extracted_text,
|
|
598
|
+
preflight=preflight,
|
|
599
|
+
parser_quality=parser_quality,
|
|
600
|
+
)
|
|
601
|
+
|
|
602
|
+
try:
|
|
603
|
+
async with acquire_db() as read_db:
|
|
604
|
+
ctx = await gather_classification_context(classifier_content, read_db)
|
|
605
|
+
except Exception: # noqa: BLE001
|
|
606
|
+
logger.warning("llm context gather failed", exc_info=True)
|
|
607
|
+
ctx = {"projects": [], "similar_artifacts": [], "hotspots": []}
|
|
608
|
+
if preflight.get("source_context"):
|
|
609
|
+
ctx["source_context"] = preflight["source_context"]
|
|
610
|
+
if ingest_id:
|
|
611
|
+
ctx["_idempotency_scope"] = f"ingest:{ingest_id}"
|
|
612
|
+
|
|
613
|
+
classifier = None
|
|
614
|
+
if byok:
|
|
615
|
+
try:
|
|
616
|
+
from core.api.services.ingest.llm.byok_provider import build_classifier
|
|
617
|
+
|
|
618
|
+
classifier = build_classifier(resolved_provider)
|
|
619
|
+
except Exception: # noqa: BLE001
|
|
620
|
+
logger.warning("byok classifier build failed", exc_info=True)
|
|
621
|
+
classifier = None
|
|
622
|
+
if classifier is None and (enabled or shadow):
|
|
623
|
+
# Legacy env-driven provider (OSS first-boot fallback).
|
|
624
|
+
try:
|
|
625
|
+
classifier = get_classifier()
|
|
626
|
+
except Exception: # noqa: BLE001
|
|
627
|
+
logger.warning("llm classifier factory failed", exc_info=True)
|
|
628
|
+
classifier = None
|
|
629
|
+
if classifier is None:
|
|
630
|
+
# Gate produced no usable provider: disabled, surfaced, never heuristic.
|
|
631
|
+
return _with_llm_no_result(
|
|
632
|
+
base_classification,
|
|
633
|
+
status="disabled_no_provider",
|
|
634
|
+
reason="no_llm_provider_configured",
|
|
635
|
+
)
|
|
636
|
+
|
|
637
|
+
try:
|
|
638
|
+
llm_result = await classifier.classify(classifier_content, ctx)
|
|
639
|
+
except Exception: # noqa: BLE001 - defensive: classify() should never raise
|
|
640
|
+
logger.warning("llm classify raised", exc_info=True)
|
|
641
|
+
return _with_llm_no_result(
|
|
642
|
+
base_classification,
|
|
643
|
+
status="exception",
|
|
644
|
+
reason="llm_classifier_exception",
|
|
645
|
+
)
|
|
646
|
+
|
|
647
|
+
if llm_result is None:
|
|
648
|
+
diagnostics = getattr(classifier, "last_error", None)
|
|
649
|
+
if isinstance(diagnostics, dict):
|
|
650
|
+
return _with_llm_no_result(
|
|
651
|
+
base_classification,
|
|
652
|
+
status=str(diagnostics.get("status") or "no_result"),
|
|
653
|
+
reason=str(diagnostics.get("reason") or "llm_classifier_returned_none"),
|
|
654
|
+
extra=diagnostics,
|
|
655
|
+
)
|
|
656
|
+
return _with_llm_no_result(
|
|
657
|
+
base_classification,
|
|
658
|
+
status="no_result",
|
|
659
|
+
reason="llm_classifier_returned_none",
|
|
660
|
+
)
|
|
661
|
+
|
|
662
|
+
# Validate project_slug actually exists on disk (avoid hallucinated slugs).
|
|
663
|
+
valid_slug: str | None = None
|
|
664
|
+
try:
|
|
665
|
+
from core.api.services.ingest.insert_saga import _load_project_entry
|
|
666
|
+
|
|
667
|
+
ptype, _repo = _load_project_entry(llm_result.project_slug)
|
|
668
|
+
if ptype:
|
|
669
|
+
valid_slug = llm_result.project_slug
|
|
670
|
+
except Exception: # noqa: BLE001 - any failure -> reject
|
|
671
|
+
valid_slug = None
|
|
672
|
+
|
|
673
|
+
provider = _ingest_llm_provider()
|
|
674
|
+
model = _ingest_llm_model()
|
|
675
|
+
target_folder = ALLOWED_TARGETS.get(llm_result.document_type)
|
|
676
|
+
confidence_decision = compute_composite_confidence(
|
|
677
|
+
route=route,
|
|
678
|
+
parser_quality=parser_quality,
|
|
679
|
+
llm_confidence=float(llm_result.confidence),
|
|
680
|
+
valid_project=valid_slug is not None,
|
|
681
|
+
document_type=llm_result.document_type,
|
|
682
|
+
extracted_text=extracted_text,
|
|
683
|
+
)
|
|
684
|
+
metadata = {
|
|
685
|
+
"model": model,
|
|
686
|
+
"provider": provider,
|
|
687
|
+
"project_slug": llm_result.project_slug,
|
|
688
|
+
"valid_slug": valid_slug,
|
|
689
|
+
"document_type": llm_result.document_type,
|
|
690
|
+
"title": llm_result.title,
|
|
691
|
+
"tags": llm_result.tags,
|
|
692
|
+
"llm_confidence": llm_result.confidence,
|
|
693
|
+
"composite_confidence": confidence_decision.score,
|
|
694
|
+
"confidence_gate": confidence_decision.as_json(),
|
|
695
|
+
"reasoning": llm_result.reasoning,
|
|
696
|
+
"pii_detected": bool(pii_redactor.analyze(extracted_text[:6000])),
|
|
697
|
+
}
|
|
698
|
+
source_context = preflight.get("source_context") or {}
|
|
699
|
+
source_project_slug = source_context.get("project_slug")
|
|
700
|
+
if source_project_slug:
|
|
701
|
+
metadata.update(
|
|
702
|
+
{
|
|
703
|
+
"source_project_slug": source_project_slug,
|
|
704
|
+
"source_project_prior": source_context.get("prior"),
|
|
705
|
+
"source_project_reason": source_context.get("reason"),
|
|
706
|
+
"source_project_followed": llm_result.project_slug == source_project_slug,
|
|
707
|
+
"source_project_overridden": llm_result.project_slug != source_project_slug,
|
|
708
|
+
}
|
|
709
|
+
)
|
|
710
|
+
|
|
711
|
+
# Shadow mode: log decision but never override the deterministic classifier.
|
|
712
|
+
if shadow:
|
|
713
|
+
shadow_blob = {
|
|
714
|
+
"shadow_mode": True,
|
|
715
|
+
"llm_metadata": metadata,
|
|
716
|
+
}
|
|
717
|
+
merged = dict(base_classification)
|
|
718
|
+
merged.setdefault("llm_shadow", shadow_blob)
|
|
719
|
+
return merged
|
|
720
|
+
|
|
721
|
+
if valid_slug is None:
|
|
722
|
+
logger.info(
|
|
723
|
+
"llm classifier returned unknown slug=%s; falling back to deterministic",
|
|
724
|
+
llm_result.project_slug,
|
|
725
|
+
)
|
|
726
|
+
return _with_llm_no_result(
|
|
727
|
+
base_classification,
|
|
728
|
+
status="invalid_project",
|
|
729
|
+
reason="llm_classifier_invalid_project_slug",
|
|
730
|
+
extra={
|
|
731
|
+
"project_slug": llm_result.project_slug,
|
|
732
|
+
"document_type": llm_result.document_type,
|
|
733
|
+
"llm_confidence": llm_result.confidence,
|
|
734
|
+
"reasoning": llm_result.reasoning,
|
|
735
|
+
},
|
|
736
|
+
)
|
|
737
|
+
|
|
738
|
+
if target_folder is None:
|
|
739
|
+
return _with_llm_no_result(
|
|
740
|
+
base_classification,
|
|
741
|
+
status="invalid_document_type",
|
|
742
|
+
reason="llm_classifier_invalid_document_type",
|
|
743
|
+
extra={
|
|
744
|
+
"project_slug": llm_result.project_slug,
|
|
745
|
+
"document_type": llm_result.document_type,
|
|
746
|
+
"llm_confidence": llm_result.confidence,
|
|
747
|
+
"reasoning": llm_result.reasoning,
|
|
748
|
+
},
|
|
749
|
+
)
|
|
750
|
+
|
|
751
|
+
privacy_block_reason = _llm_auto_approve_privacy_block_reason(
|
|
752
|
+
preflight=preflight,
|
|
753
|
+
extracted_text=extracted_text,
|
|
754
|
+
document_type=llm_result.document_type,
|
|
755
|
+
)
|
|
756
|
+
if privacy_block_reason:
|
|
757
|
+
merged = dict(base_classification)
|
|
758
|
+
merged["llm_metadata"] = {
|
|
759
|
+
**metadata,
|
|
760
|
+
"auto_approved": False,
|
|
761
|
+
"auto_approve_blocked_reason": privacy_block_reason,
|
|
762
|
+
}
|
|
763
|
+
return merged
|
|
764
|
+
|
|
765
|
+
if (
|
|
766
|
+
llm_result.confidence < LLM_AUTO_APPROVE_THRESHOLD
|
|
767
|
+
or not confidence_decision.auto_approve
|
|
768
|
+
):
|
|
769
|
+
# Keep deterministic decision but surface the LLM hint so the human
|
|
770
|
+
# triage UI can show it.
|
|
771
|
+
merged = dict(base_classification)
|
|
772
|
+
merged["llm_metadata"] = {**metadata, "auto_approved": False}
|
|
773
|
+
return merged
|
|
774
|
+
|
|
775
|
+
enriched = dict(base_classification)
|
|
776
|
+
enriched.update(
|
|
777
|
+
{
|
|
778
|
+
"type": llm_result.document_type,
|
|
779
|
+
"title": llm_result.title,
|
|
780
|
+
"tags": list(llm_result.tags),
|
|
781
|
+
"target_folder": target_folder,
|
|
782
|
+
"confidence": confidence_decision.score,
|
|
783
|
+
"reason": "llm_routing",
|
|
784
|
+
"auto_approve": True,
|
|
785
|
+
"suggested_project_slug": valid_slug,
|
|
786
|
+
"llm_metadata": {
|
|
787
|
+
**metadata,
|
|
788
|
+
"project_slug": valid_slug,
|
|
789
|
+
"auto_approved": True,
|
|
790
|
+
},
|
|
791
|
+
}
|
|
792
|
+
)
|
|
793
|
+
return enriched
|
|
794
|
+
|
|
795
|
+
|
|
796
|
+
def _llm_auto_approve_privacy_block_reason(
|
|
797
|
+
*,
|
|
798
|
+
preflight: dict[str, Any],
|
|
799
|
+
extracted_text: str,
|
|
800
|
+
document_type: str | None = None,
|
|
801
|
+
) -> str | None:
|
|
802
|
+
packet = preflight or {}
|
|
803
|
+
pf = packet.get("preflight") or {}
|
|
804
|
+
file_info = packet.get("file") or {}
|
|
805
|
+
filename_text = " ".join(
|
|
806
|
+
str(file_info.get(key) or "") for key in ("filename", "stem")
|
|
807
|
+
)
|
|
808
|
+
if _contains_identity_signal(filename_text, minimum=1):
|
|
809
|
+
return "identity_document_requires_manual_triage"
|
|
810
|
+
if bool(pf.get("identity_hint")) and _contains_identity_signal(
|
|
811
|
+
extracted_text, minimum=1
|
|
812
|
+
):
|
|
813
|
+
return "identity_document_requires_manual_triage"
|
|
814
|
+
if _contains_identity_signal(extracted_text, minimum=2):
|
|
815
|
+
return "identity_document_requires_manual_triage"
|
|
816
|
+
|
|
817
|
+
# E5 already redacts PII before sending the prompt to the local Gateway.
|
|
818
|
+
# Generic business documents regularly contain email, phone, IBAN, VAT or
|
|
819
|
+
# fiscal-code-like strings; those should not block auto-triage by
|
|
820
|
+
# themselves. Keep the conservative fallback only when the LLM did not
|
|
821
|
+
# return a document type.
|
|
822
|
+
if document_type == "record" and pii_redactor.analyze(extracted_text[:6000]):
|
|
823
|
+
return "sensitive_record_requires_manual_triage"
|
|
824
|
+
if document_type is None and pii_redactor.analyze(extracted_text[:6000]):
|
|
825
|
+
return "pii_requires_manual_triage"
|
|
826
|
+
return None
|
|
827
|
+
|
|
828
|
+
|
|
829
|
+
def _contains_identity_signal(text: str, *, minimum: int = 1) -> bool:
|
|
830
|
+
lowered = (text or "").lower()
|
|
831
|
+
matches = sum(1 for term in IDENTITY_AUTO_APPROVE_TERMS if term in lowered)
|
|
832
|
+
return matches >= minimum
|
|
833
|
+
|
|
834
|
+
|
|
835
|
+
def _gateway_aux_configured() -> bool:
|
|
836
|
+
key = settings.ingest_llm_gateway_api_key
|
|
837
|
+
key_value = (
|
|
838
|
+
key.get_secret_value() if hasattr(key, "get_secret_value") else str(key or "")
|
|
839
|
+
)
|
|
840
|
+
return bool(
|
|
841
|
+
settings.pir_env != "test"
|
|
842
|
+
and key_value
|
|
843
|
+
and (settings.llm_gateway_aux_base_url or settings.llm_gateway_base_url)
|
|
844
|
+
)
|
|
845
|
+
|
|
846
|
+
|
|
847
|
+
def _route_for(path: Path, mime_type: str, preflight: dict[str, Any]) -> IngestRoute:
|
|
848
|
+
gateway_enabled = _gateway_aux_configured()
|
|
849
|
+
return choose_route(
|
|
850
|
+
path=path,
|
|
851
|
+
mime_type=mime_type,
|
|
852
|
+
preflight=preflight,
|
|
853
|
+
docparse_enabled=(
|
|
854
|
+
gateway_enabled
|
|
855
|
+
and settings.ingest_docparse_enabled
|
|
856
|
+
and (
|
|
857
|
+
settings.ingest_docparse_pdfs_enabled
|
|
858
|
+
if mime_type == "application/pdf"
|
|
859
|
+
else settings.ingest_docparse_images_enabled
|
|
860
|
+
)
|
|
861
|
+
),
|
|
862
|
+
ocr_enabled=gateway_enabled,
|
|
863
|
+
vision_enabled=bool(gateway_enabled and settings.ingest_vision_images_enabled),
|
|
864
|
+
mode_override=settings.ingest_docparse_mode_override or None,
|
|
865
|
+
)
|
|
866
|
+
|
|
867
|
+
|
|
868
|
+
async def _maybe_llm_route(
|
|
869
|
+
*,
|
|
870
|
+
route: IngestRoute,
|
|
871
|
+
preflight: dict[str, Any],
|
|
872
|
+
mime_type: str,
|
|
873
|
+
) -> IngestRoute:
|
|
874
|
+
if route.confidence >= 0.70 or not _llm_classifier_enabled():
|
|
875
|
+
return route
|
|
876
|
+
try:
|
|
877
|
+
from core.api.services.ingest.llm.local_gateway import (
|
|
878
|
+
classify_route_with_local_gateway,
|
|
879
|
+
)
|
|
880
|
+
except Exception: # noqa: BLE001
|
|
881
|
+
logger.debug("local route classifier import failed", exc_info=True)
|
|
882
|
+
return route
|
|
883
|
+
|
|
884
|
+
decision = await classify_route_with_local_gateway(
|
|
885
|
+
preflight=preflight,
|
|
886
|
+
deterministic_route=route.as_json(),
|
|
887
|
+
)
|
|
888
|
+
if decision is None:
|
|
889
|
+
return route
|
|
890
|
+
if decision.workflow not in _allowed_workflows_for_mime(mime_type):
|
|
891
|
+
logger.info(
|
|
892
|
+
"route classifier ignored invalid workflow=%s for mime=%s",
|
|
893
|
+
decision.workflow,
|
|
894
|
+
mime_type,
|
|
895
|
+
)
|
|
896
|
+
return route
|
|
897
|
+
return replace(
|
|
898
|
+
route,
|
|
899
|
+
workflow=decision.workflow,
|
|
900
|
+
tier=_tier_for_workflow(decision.workflow),
|
|
901
|
+
mode=decision.mode if decision.workflow == "docparse" else None,
|
|
902
|
+
reason=f"tier-fast route classifier: {decision.reason}",
|
|
903
|
+
confidence=float(decision.confidence),
|
|
904
|
+
features_used=[*route.features_used, "tier_fast_route_classifier"],
|
|
905
|
+
)
|
|
906
|
+
|
|
907
|
+
|
|
908
|
+
def _allowed_workflows_for_mime(mime_type: str) -> set[str]:
|
|
909
|
+
if mime_type.startswith(("audio/", "video/")):
|
|
910
|
+
return {"transcribe"}
|
|
911
|
+
if mime_type == "application/pdf":
|
|
912
|
+
return {"local", "ocr", "docparse"}
|
|
913
|
+
if mime_type.startswith("image/"):
|
|
914
|
+
return {"ocr", "docparse", "vision"}
|
|
915
|
+
return {"local"}
|
|
916
|
+
|
|
917
|
+
|
|
918
|
+
def _tier_for_workflow(workflow: str) -> str | None:
|
|
919
|
+
return {
|
|
920
|
+
"ocr": "tier-ocr",
|
|
921
|
+
"docparse": "tier-docparse",
|
|
922
|
+
"transcribe": "tier-transcribe",
|
|
923
|
+
"vision": "tier-vision",
|
|
924
|
+
}.get(workflow)
|
|
925
|
+
|
|
926
|
+
|
|
927
|
+
def _block_llm_auto_approve(
|
|
928
|
+
classification_json: dict[str, Any],
|
|
929
|
+
*,
|
|
930
|
+
reason: str,
|
|
931
|
+
existing_ingest_id: str | None = None,
|
|
932
|
+
) -> dict[str, Any]:
|
|
933
|
+
blocked = dict(classification_json)
|
|
934
|
+
blocked["auto_approve"] = False
|
|
935
|
+
blocked["reason"] = reason
|
|
936
|
+
metadata = dict(blocked.get("llm_metadata") or {})
|
|
937
|
+
metadata["auto_approved"] = False
|
|
938
|
+
metadata["auto_approve_blocked_reason"] = reason
|
|
939
|
+
if existing_ingest_id:
|
|
940
|
+
metadata["existing_ingest_id"] = existing_ingest_id
|
|
941
|
+
blocked["llm_metadata"] = metadata
|
|
942
|
+
return blocked
|
|
943
|
+
|
|
944
|
+
|
|
945
|
+
def _reject_llm_auto_approve_duplicate(
|
|
946
|
+
classification_json: dict[str, Any],
|
|
947
|
+
*,
|
|
948
|
+
reason: str,
|
|
949
|
+
existing_ingest_id: str,
|
|
950
|
+
) -> dict[str, Any]:
|
|
951
|
+
rejected = _block_llm_auto_approve(
|
|
952
|
+
classification_json,
|
|
953
|
+
reason=reason,
|
|
954
|
+
existing_ingest_id=existing_ingest_id,
|
|
955
|
+
)
|
|
956
|
+
rejected["auto_reject"] = True
|
|
957
|
+
metadata = dict(rejected.get("llm_metadata") or {})
|
|
958
|
+
metadata["auto_rejected"] = True
|
|
959
|
+
metadata["auto_reject_reason"] = reason
|
|
960
|
+
metadata["existing_ingest_id"] = existing_ingest_id
|
|
961
|
+
rejected["llm_metadata"] = metadata
|
|
962
|
+
return rejected
|
|
963
|
+
|
|
964
|
+
|
|
965
|
+
async def _find_ingest_duplicate_for_project(
|
|
966
|
+
db: Any,
|
|
967
|
+
*,
|
|
968
|
+
ingest_id: str,
|
|
969
|
+
sha256: str | None,
|
|
970
|
+
project_slug: str,
|
|
971
|
+
) -> str | None:
|
|
972
|
+
if not sha256:
|
|
973
|
+
return None
|
|
974
|
+
async with db.execute(
|
|
975
|
+
"""
|
|
976
|
+
SELECT id
|
|
977
|
+
FROM ingest_pending
|
|
978
|
+
WHERE sha256 = ?
|
|
979
|
+
AND project_slug = ?
|
|
980
|
+
AND id != ?
|
|
981
|
+
LIMIT 1
|
|
982
|
+
""",
|
|
983
|
+
(sha256, project_slug, ingest_id),
|
|
984
|
+
) as cursor:
|
|
985
|
+
row = await cursor.fetchone()
|
|
986
|
+
return str(row["id"]) if row is not None else None
|
|
987
|
+
|
|
988
|
+
|
|
989
|
+
async def _apply_llm_project_switch_if_safe(
|
|
990
|
+
db: Any,
|
|
991
|
+
*,
|
|
992
|
+
ingest_id: str,
|
|
993
|
+
sha256: str | None,
|
|
994
|
+
current_project_slug: str,
|
|
995
|
+
target_project_slug: str,
|
|
996
|
+
path: Path,
|
|
997
|
+
classification_json: dict[str, Any],
|
|
998
|
+
) -> tuple[str, Path, dict[str, Any], str, str | None]:
|
|
999
|
+
new_root = PROJECTS_ROOT / target_project_slug
|
|
1000
|
+
if not new_root.is_dir():
|
|
1001
|
+
logger.warning(
|
|
1002
|
+
"llm_routing project switch aborted: %s not a project root",
|
|
1003
|
+
new_root,
|
|
1004
|
+
)
|
|
1005
|
+
return (
|
|
1006
|
+
current_project_slug,
|
|
1007
|
+
path,
|
|
1008
|
+
_block_llm_auto_approve(
|
|
1009
|
+
classification_json,
|
|
1010
|
+
reason="llm_routing_project_switch_invalid_project",
|
|
1011
|
+
),
|
|
1012
|
+
"awaiting_triage",
|
|
1013
|
+
None,
|
|
1014
|
+
)
|
|
1015
|
+
|
|
1016
|
+
duplicate_id = await _find_ingest_duplicate_for_project(
|
|
1017
|
+
db,
|
|
1018
|
+
ingest_id=ingest_id,
|
|
1019
|
+
sha256=sha256,
|
|
1020
|
+
project_slug=target_project_slug,
|
|
1021
|
+
)
|
|
1022
|
+
if duplicate_id is not None:
|
|
1023
|
+
logger.warning(
|
|
1024
|
+
"llm_routing project switch aborted: sha256 already exists in %s as %s",
|
|
1025
|
+
target_project_slug,
|
|
1026
|
+
duplicate_id,
|
|
1027
|
+
)
|
|
1028
|
+
return (
|
|
1029
|
+
current_project_slug,
|
|
1030
|
+
path,
|
|
1031
|
+
_reject_llm_auto_approve_duplicate(
|
|
1032
|
+
classification_json,
|
|
1033
|
+
reason="llm_routing_project_switch_dedup_collision",
|
|
1034
|
+
existing_ingest_id=duplicate_id,
|
|
1035
|
+
),
|
|
1036
|
+
"rejected",
|
|
1037
|
+
"auto_reject:llm_routing_duplicate",
|
|
1038
|
+
)
|
|
1039
|
+
|
|
1040
|
+
new_input = new_root / "input"
|
|
1041
|
+
new_input.mkdir(parents=True, exist_ok=True)
|
|
1042
|
+
new_source = new_input / path.name
|
|
1043
|
+
source_sidecar = _metadata_sidecar_path(path)
|
|
1044
|
+
target_sidecar = _metadata_sidecar_path(new_source)
|
|
1045
|
+
if new_source.exists():
|
|
1046
|
+
logger.warning(
|
|
1047
|
+
"llm_routing project switch aborted: %s already exists",
|
|
1048
|
+
new_source,
|
|
1049
|
+
)
|
|
1050
|
+
return (
|
|
1051
|
+
current_project_slug,
|
|
1052
|
+
path,
|
|
1053
|
+
_block_llm_auto_approve(
|
|
1054
|
+
classification_json,
|
|
1055
|
+
reason="llm_routing_project_switch_path_collision",
|
|
1056
|
+
),
|
|
1057
|
+
"awaiting_triage",
|
|
1058
|
+
None,
|
|
1059
|
+
)
|
|
1060
|
+
if source_sidecar.exists() and target_sidecar.exists():
|
|
1061
|
+
logger.warning(
|
|
1062
|
+
"llm_routing project switch aborted: %s already exists",
|
|
1063
|
+
target_sidecar,
|
|
1064
|
+
)
|
|
1065
|
+
return (
|
|
1066
|
+
current_project_slug,
|
|
1067
|
+
path,
|
|
1068
|
+
_block_llm_auto_approve(
|
|
1069
|
+
classification_json,
|
|
1070
|
+
reason="llm_routing_project_switch_sidecar_collision",
|
|
1071
|
+
),
|
|
1072
|
+
"awaiting_triage",
|
|
1073
|
+
None,
|
|
1074
|
+
)
|
|
1075
|
+
|
|
1076
|
+
path.replace(new_source)
|
|
1077
|
+
if source_sidecar.exists():
|
|
1078
|
+
source_sidecar.replace(target_sidecar)
|
|
1079
|
+
return (
|
|
1080
|
+
target_project_slug,
|
|
1081
|
+
new_source,
|
|
1082
|
+
classification_json,
|
|
1083
|
+
"approved",
|
|
1084
|
+
"auto_approve:llm_routing",
|
|
1085
|
+
)
|
|
1086
|
+
|
|
1087
|
+
|
|
1088
|
+
async def _parse_pdf_local(path: Path):
|
|
1089
|
+
try:
|
|
1090
|
+
return await parse_pdf_file(path, allow_docparse=False)
|
|
1091
|
+
except TypeError as exc:
|
|
1092
|
+
if "allow_docparse" not in str(exc):
|
|
1093
|
+
raise
|
|
1094
|
+
return await parse_pdf_file(path)
|
|
1095
|
+
|
|
1096
|
+
|
|
1097
|
+
def _transient_parse_max_attempts() -> int:
|
|
1098
|
+
raw = os.environ.get("INGEST_TRANSIENT_PARSE_MAX_ATTEMPTS", "3")
|
|
1099
|
+
try:
|
|
1100
|
+
return max(1, int(raw))
|
|
1101
|
+
except ValueError:
|
|
1102
|
+
return 3
|
|
1103
|
+
|
|
1104
|
+
|
|
1105
|
+
def _transient_parse_delay_seconds(attempt: int) -> float:
|
|
1106
|
+
if settings.pir_env == "test":
|
|
1107
|
+
return 0.0
|
|
1108
|
+
raw = os.environ.get("INGEST_TRANSIENT_PARSE_RETRY_BASE_SECONDS", "5")
|
|
1109
|
+
try:
|
|
1110
|
+
base = max(0.0, float(raw))
|
|
1111
|
+
except ValueError:
|
|
1112
|
+
base = 5.0
|
|
1113
|
+
return min(base * attempt, 30.0)
|
|
1114
|
+
|
|
1115
|
+
|
|
1116
|
+
def _is_transient_parse_error(exc: BaseException) -> bool:
|
|
1117
|
+
message = str(exc).lower()
|
|
1118
|
+
return any(marker in message for marker in _TRANSIENT_PARSE_ERROR_MARKERS)
|
|
1119
|
+
|
|
1120
|
+
|
|
1121
|
+
def _is_empty_ocr_result(exc: BaseException) -> bool:
|
|
1122
|
+
return _OCR_EMPTY_TEXT_MARKER in str(exc).lower()
|
|
1123
|
+
|
|
1124
|
+
|
|
1125
|
+
def _docparse_enabled_for_pdf() -> bool:
|
|
1126
|
+
return bool(
|
|
1127
|
+
_gateway_aux_configured()
|
|
1128
|
+
and settings.ingest_docparse_enabled
|
|
1129
|
+
and settings.ingest_docparse_pdfs_enabled
|
|
1130
|
+
)
|
|
1131
|
+
|
|
1132
|
+
|
|
1133
|
+
def _docparse_fallback_mode(route: IngestRoute) -> str:
|
|
1134
|
+
return (
|
|
1135
|
+
route.mode
|
|
1136
|
+
or settings.ingest_docparse_mode_override
|
|
1137
|
+
or settings.ingest_docparse_mode
|
|
1138
|
+
)
|
|
1139
|
+
|
|
1140
|
+
|
|
1141
|
+
def _ocr_to_docparse_route(route: IngestRoute) -> IngestRoute:
|
|
1142
|
+
return replace(
|
|
1143
|
+
route,
|
|
1144
|
+
workflow="docparse",
|
|
1145
|
+
tier="tier-docparse",
|
|
1146
|
+
mode=_docparse_fallback_mode(route),
|
|
1147
|
+
reason=f"{route.reason}; tier-ocr empty result fell back to docparse",
|
|
1148
|
+
confidence=max(route.confidence, 0.86),
|
|
1149
|
+
features_used=[*route.features_used, "ocr_empty_docparse_fallback"],
|
|
1150
|
+
)
|
|
1151
|
+
|
|
1152
|
+
|
|
1153
|
+
async def _parse_file(
|
|
1154
|
+
ingest_id: str,
|
|
1155
|
+
project_slug: str,
|
|
1156
|
+
path: Path,
|
|
1157
|
+
mime_type: str,
|
|
1158
|
+
*,
|
|
1159
|
+
preflight: dict[str, Any],
|
|
1160
|
+
route: IngestRoute,
|
|
1161
|
+
) -> ParseDispatchResult:
|
|
1162
|
+
if route.workflow == "skip":
|
|
1163
|
+
raise ValueError(route.reason)
|
|
1164
|
+
|
|
1165
|
+
if mime_type == "application/pdf":
|
|
1166
|
+
effective_route = route
|
|
1167
|
+
|
|
1168
|
+
async def parse_pdf_route():
|
|
1169
|
+
nonlocal effective_route
|
|
1170
|
+
if route.workflow == "docparse":
|
|
1171
|
+
try:
|
|
1172
|
+
return await parse_pdf_docparse(path, mode=route.mode)
|
|
1173
|
+
except MissingGatewayConfig:
|
|
1174
|
+
logger.warning(
|
|
1175
|
+
"tier-docparse not configured; falling back to local PDF"
|
|
1176
|
+
)
|
|
1177
|
+
return await _parse_pdf_local(path)
|
|
1178
|
+
if route.workflow == "ocr":
|
|
1179
|
+
try:
|
|
1180
|
+
return await parse_pdf_ocr(path)
|
|
1181
|
+
except MissingGatewayConfig:
|
|
1182
|
+
logger.warning("tier-ocr not configured; falling back to local PDF")
|
|
1183
|
+
return await _parse_pdf_local(path)
|
|
1184
|
+
except RuntimeError as exc:
|
|
1185
|
+
if not _is_empty_ocr_result(exc) or not _docparse_enabled_for_pdf():
|
|
1186
|
+
raise
|
|
1187
|
+
effective_route = _ocr_to_docparse_route(route)
|
|
1188
|
+
logger.warning(
|
|
1189
|
+
"tier-ocr returned empty text; falling back to tier-docparse: path=%s mode=%s",
|
|
1190
|
+
path,
|
|
1191
|
+
effective_route.mode,
|
|
1192
|
+
)
|
|
1193
|
+
return await parse_pdf_docparse(path, mode=effective_route.mode)
|
|
1194
|
+
return await _parse_pdf_local(path)
|
|
1195
|
+
|
|
1196
|
+
parsed = await _run_heavy_parser(
|
|
1197
|
+
ingest_id=ingest_id,
|
|
1198
|
+
project_slug=project_slug,
|
|
1199
|
+
lane=_parser_lane_for_workflow(route.workflow),
|
|
1200
|
+
parser_name=f"pdf:{route.workflow}",
|
|
1201
|
+
path=path,
|
|
1202
|
+
parse=parse_pdf_route,
|
|
1203
|
+
)
|
|
1204
|
+
parser_quality = estimate_parser_quality(
|
|
1205
|
+
parser_used=parsed.parser_used,
|
|
1206
|
+
extracted_text=parsed.text,
|
|
1207
|
+
structure=parsed.structure,
|
|
1208
|
+
)
|
|
1209
|
+
return ParseDispatchResult(
|
|
1210
|
+
parsed,
|
|
1211
|
+
parsed.parser_used,
|
|
1212
|
+
effective_route,
|
|
1213
|
+
preflight,
|
|
1214
|
+
parser_quality,
|
|
1215
|
+
)
|
|
1216
|
+
|
|
1217
|
+
if _is_image(path, mime_type):
|
|
1218
|
+
if route.workflow == "vision":
|
|
1219
|
+
parsed = await _run_heavy_parser(
|
|
1220
|
+
ingest_id=ingest_id,
|
|
1221
|
+
project_slug=project_slug,
|
|
1222
|
+
lane="vision",
|
|
1223
|
+
parser_name="image:vision",
|
|
1224
|
+
path=path,
|
|
1225
|
+
parse=lambda: parse_vision_with_gateway(path, mime_type),
|
|
1226
|
+
)
|
|
1227
|
+
parser_quality = estimate_parser_quality(
|
|
1228
|
+
parser_used=str(parsed["parser_used"]),
|
|
1229
|
+
extracted_text=str(
|
|
1230
|
+
parsed.get("text") or parsed.get("extracted_text") or ""
|
|
1231
|
+
),
|
|
1232
|
+
structure=parsed.get("structure") or {},
|
|
1233
|
+
)
|
|
1234
|
+
return ParseDispatchResult(
|
|
1235
|
+
parsed,
|
|
1236
|
+
str(parsed["parser_used"]),
|
|
1237
|
+
route,
|
|
1238
|
+
preflight,
|
|
1239
|
+
parser_quality,
|
|
1240
|
+
)
|
|
1241
|
+
|
|
1242
|
+
prefer_docparse = route.workflow == "docparse"
|
|
1243
|
+
parsed = await _run_heavy_parser(
|
|
1244
|
+
ingest_id=ingest_id,
|
|
1245
|
+
project_slug=project_slug,
|
|
1246
|
+
lane=_parser_lane_for_workflow(route.workflow),
|
|
1247
|
+
parser_name=f"image:{route.workflow}",
|
|
1248
|
+
path=path,
|
|
1249
|
+
parse=lambda: parse_image_with_gateway(
|
|
1250
|
+
path,
|
|
1251
|
+
mime_type,
|
|
1252
|
+
prefer_docparse=prefer_docparse,
|
|
1253
|
+
docparse_mode=route.mode,
|
|
1254
|
+
),
|
|
1255
|
+
)
|
|
1256
|
+
parser_quality = estimate_parser_quality(
|
|
1257
|
+
parser_used=str(parsed["parser_used"]),
|
|
1258
|
+
extracted_text=str(
|
|
1259
|
+
parsed.get("text") or parsed.get("extracted_text") or ""
|
|
1260
|
+
),
|
|
1261
|
+
structure=parsed.get("structure") or {},
|
|
1262
|
+
)
|
|
1263
|
+
return ParseDispatchResult(
|
|
1264
|
+
parsed,
|
|
1265
|
+
str(parsed["parser_used"]),
|
|
1266
|
+
route,
|
|
1267
|
+
preflight,
|
|
1268
|
+
parser_quality,
|
|
1269
|
+
)
|
|
1270
|
+
if _is_xlsx(path, mime_type):
|
|
1271
|
+
await _mark_parser_active(ingest_id, project_slug)
|
|
1272
|
+
parsed = await asyncio.to_thread(parse_xlsx, path)
|
|
1273
|
+
parser_quality = estimate_parser_quality(
|
|
1274
|
+
parser_used=str(parsed["parser_used"]),
|
|
1275
|
+
extracted_text=str(parsed.get("text") or ""),
|
|
1276
|
+
structure=parsed.get("structure") or {},
|
|
1277
|
+
)
|
|
1278
|
+
return ParseDispatchResult(
|
|
1279
|
+
parsed,
|
|
1280
|
+
str(parsed["parser_used"]),
|
|
1281
|
+
route,
|
|
1282
|
+
preflight,
|
|
1283
|
+
parser_quality,
|
|
1284
|
+
)
|
|
1285
|
+
if _is_docx(path, mime_type):
|
|
1286
|
+
await _mark_parser_active(ingest_id, project_slug)
|
|
1287
|
+
parsed = await asyncio.to_thread(parse_docx, path)
|
|
1288
|
+
parser_quality = estimate_parser_quality(
|
|
1289
|
+
parser_used="internal_docx",
|
|
1290
|
+
extracted_text=parsed.text,
|
|
1291
|
+
structure=parsed.structure,
|
|
1292
|
+
)
|
|
1293
|
+
return ParseDispatchResult(
|
|
1294
|
+
parsed, "internal_docx", route, preflight, parser_quality
|
|
1295
|
+
)
|
|
1296
|
+
if _is_transcript_media(path, mime_type):
|
|
1297
|
+
parsed = await _run_heavy_parser(
|
|
1298
|
+
ingest_id=ingest_id,
|
|
1299
|
+
project_slug=project_slug,
|
|
1300
|
+
lane="transcribe",
|
|
1301
|
+
parser_name="transcript",
|
|
1302
|
+
path=path,
|
|
1303
|
+
parse=lambda: parse_media_transcript(path, mime_type),
|
|
1304
|
+
)
|
|
1305
|
+
parser_quality = estimate_parser_quality(
|
|
1306
|
+
parser_used="tier_transcribe",
|
|
1307
|
+
extracted_text=parsed.text,
|
|
1308
|
+
structure=parsed.structure,
|
|
1309
|
+
)
|
|
1310
|
+
return ParseDispatchResult(
|
|
1311
|
+
parsed, "tier_transcribe", route, preflight, parser_quality
|
|
1312
|
+
)
|
|
1313
|
+
if path.suffix.lower() in _MARKDOWN_SUFFIXES or mime_type in {
|
|
1314
|
+
"text/markdown",
|
|
1315
|
+
"text/plain",
|
|
1316
|
+
}:
|
|
1317
|
+
await _mark_parser_active(ingest_id, project_slug)
|
|
1318
|
+
parsed = parse_markdown_file(path)
|
|
1319
|
+
parser_quality = estimate_parser_quality(
|
|
1320
|
+
parser_used="internal_markdown",
|
|
1321
|
+
extracted_text=parsed.text,
|
|
1322
|
+
structure=parsed.structure,
|
|
1323
|
+
)
|
|
1324
|
+
return ParseDispatchResult(
|
|
1325
|
+
parsed, "internal_markdown", route, preflight, parser_quality
|
|
1326
|
+
)
|
|
1327
|
+
raise ValueError(f"Unsupported phase-1 file type: {mime_type}")
|
|
1328
|
+
|
|
1329
|
+
|
|
1330
|
+
async def _parse_file_with_transient_retries(
|
|
1331
|
+
ingest_id: str,
|
|
1332
|
+
project_slug: str,
|
|
1333
|
+
path: Path,
|
|
1334
|
+
mime_type: str,
|
|
1335
|
+
*,
|
|
1336
|
+
preflight: dict[str, Any],
|
|
1337
|
+
route: IngestRoute,
|
|
1338
|
+
) -> ParseDispatchResult:
|
|
1339
|
+
attempts = _transient_parse_max_attempts()
|
|
1340
|
+
for attempt in range(1, attempts + 1):
|
|
1341
|
+
try:
|
|
1342
|
+
return await _parse_file(
|
|
1343
|
+
ingest_id,
|
|
1344
|
+
project_slug,
|
|
1345
|
+
path,
|
|
1346
|
+
mime_type,
|
|
1347
|
+
preflight=preflight,
|
|
1348
|
+
route=route,
|
|
1349
|
+
)
|
|
1350
|
+
except Exception as exc:
|
|
1351
|
+
if attempt >= attempts or not _is_transient_parse_error(exc):
|
|
1352
|
+
raise
|
|
1353
|
+
delay = _transient_parse_delay_seconds(attempt)
|
|
1354
|
+
await _mark_parser_waiting(ingest_id, project_slug)
|
|
1355
|
+
logger.warning(
|
|
1356
|
+
"ingest parser transient failure; retrying attempt=%d/%d delay=%.1fs route=%s path=%s error=%s",
|
|
1357
|
+
attempt,
|
|
1358
|
+
attempts,
|
|
1359
|
+
delay,
|
|
1360
|
+
route.workflow,
|
|
1361
|
+
path,
|
|
1362
|
+
exc,
|
|
1363
|
+
)
|
|
1364
|
+
await asyncio.sleep(delay)
|
|
1365
|
+
raise RuntimeError("transient parse retry loop exited unexpectedly")
|
|
1366
|
+
|
|
1367
|
+
|
|
1368
|
+
def _with_ingest_v2_diagnostics(
|
|
1369
|
+
structure: dict[str, Any],
|
|
1370
|
+
*,
|
|
1371
|
+
preflight: dict[str, Any],
|
|
1372
|
+
route: IngestRoute,
|
|
1373
|
+
parser_quality: dict[str, Any],
|
|
1374
|
+
) -> dict[str, Any]:
|
|
1375
|
+
merged = dict(structure or {})
|
|
1376
|
+
merged["ingest_v2"] = {
|
|
1377
|
+
"route": route.as_json(),
|
|
1378
|
+
"parser_quality": parser_quality,
|
|
1379
|
+
"preflight": _bounded_preflight_for_storage(preflight),
|
|
1380
|
+
}
|
|
1381
|
+
if preflight.get("source_context"):
|
|
1382
|
+
merged["ingest_v2"]["source_context"] = dict(preflight["source_context"])
|
|
1383
|
+
if (preflight.get("preflight") or {}).get("image_kind"):
|
|
1384
|
+
merged["ingest_v2"]["image_probe"] = {
|
|
1385
|
+
key: value
|
|
1386
|
+
for key, value in dict(preflight.get("preflight") or {}).items()
|
|
1387
|
+
if key
|
|
1388
|
+
in {
|
|
1389
|
+
"image_kind",
|
|
1390
|
+
"document_likelihood",
|
|
1391
|
+
"screenshot_likelihood",
|
|
1392
|
+
"photo_likelihood",
|
|
1393
|
+
"text_likelihood",
|
|
1394
|
+
"signals",
|
|
1395
|
+
"white_background_ratio",
|
|
1396
|
+
"edge_density",
|
|
1397
|
+
"brightness",
|
|
1398
|
+
"contrast",
|
|
1399
|
+
}
|
|
1400
|
+
}
|
|
1401
|
+
return merged
|
|
1402
|
+
|
|
1403
|
+
|
|
1404
|
+
def _bounded_preflight_for_storage(preflight: dict[str, Any]) -> dict[str, Any]:
|
|
1405
|
+
stored = {
|
|
1406
|
+
"file": dict(preflight.get("file") or {}),
|
|
1407
|
+
"preflight": dict(preflight.get("preflight") or {}),
|
|
1408
|
+
"content_sample": dict(preflight.get("content_sample") or {}),
|
|
1409
|
+
}
|
|
1410
|
+
sample = stored["content_sample"]
|
|
1411
|
+
for key in ("first_excerpt", "middle_excerpt", "last_excerpt", "parser_excerpt"):
|
|
1412
|
+
if key in sample:
|
|
1413
|
+
sample[key] = str(sample[key])[:500]
|
|
1414
|
+
return stored
|
|
1415
|
+
|
|
1416
|
+
|
|
1417
|
+
def _reconcile_ingress_metadata(
|
|
1418
|
+
structure: dict[str, Any] | None, ingress_metadata_raw: str | None
|
|
1419
|
+
) -> dict[str, Any] | None:
|
|
1420
|
+
"""Merge an api_ingress row's payload metadata into structure_json (U3).
|
|
1421
|
+
|
|
1422
|
+
Stored under the ``ingress_metadata`` key so one canonical place carries it
|
|
1423
|
+
downstream (Triage view + KG). No-op when there is no metadata or it is not
|
|
1424
|
+
a JSON object.
|
|
1425
|
+
"""
|
|
1426
|
+
if not ingress_metadata_raw:
|
|
1427
|
+
return structure
|
|
1428
|
+
try:
|
|
1429
|
+
parsed = json.loads(ingress_metadata_raw)
|
|
1430
|
+
except (json.JSONDecodeError, TypeError):
|
|
1431
|
+
return structure
|
|
1432
|
+
if not isinstance(parsed, dict) or not parsed:
|
|
1433
|
+
return structure
|
|
1434
|
+
return {**(structure or {}), "ingress_metadata": parsed}
|
|
1435
|
+
|
|
1436
|
+
|
|
1437
|
+
async def parse_pending(ingest_id: str) -> None:
|
|
1438
|
+
row = None
|
|
1439
|
+
async with acquire_write_db() as db:
|
|
1440
|
+
async with db.execute(
|
|
1441
|
+
"SELECT * FROM ingest_pending WHERE id = ?", (ingest_id,)
|
|
1442
|
+
) as cursor:
|
|
1443
|
+
row = await cursor.fetchone()
|
|
1444
|
+
if row is None:
|
|
1445
|
+
return
|
|
1446
|
+
if row["status"] not in {"queued", "parse_error"}:
|
|
1447
|
+
return
|
|
1448
|
+
await db.execute(
|
|
1449
|
+
"""
|
|
1450
|
+
UPDATE ingest_pending
|
|
1451
|
+
SET status = 'parser_waiting',
|
|
1452
|
+
error_message = NULL,
|
|
1453
|
+
updated_at = datetime('now')
|
|
1454
|
+
WHERE id = ?
|
|
1455
|
+
""",
|
|
1456
|
+
(ingest_id,),
|
|
1457
|
+
)
|
|
1458
|
+
await db.commit()
|
|
1459
|
+
|
|
1460
|
+
project_slug = row["project_slug"]
|
|
1461
|
+
await broadcast_ingest_changed(
|
|
1462
|
+
"parser_waiting",
|
|
1463
|
+
ingest_id=ingest_id,
|
|
1464
|
+
project_slug=project_slug,
|
|
1465
|
+
status="parser_waiting",
|
|
1466
|
+
)
|
|
1467
|
+
try:
|
|
1468
|
+
path = Path(row["file_path"])
|
|
1469
|
+
if not path.exists():
|
|
1470
|
+
raise FileNotFoundError(str(path))
|
|
1471
|
+
pending_llm_project_slug: str | None = None
|
|
1472
|
+
error_message: str | None = None
|
|
1473
|
+
mime_type = detect_mime(path)
|
|
1474
|
+
preflight = build_preflight(path, mime_type)
|
|
1475
|
+
source_context = _source_context_for_row(
|
|
1476
|
+
project_slug=project_slug,
|
|
1477
|
+
source_kind=row["source_kind"],
|
|
1478
|
+
path=path,
|
|
1479
|
+
)
|
|
1480
|
+
if source_context:
|
|
1481
|
+
preflight["source_context"] = source_context
|
|
1482
|
+
route = await _maybe_llm_route(
|
|
1483
|
+
route=_route_for(path, mime_type, preflight),
|
|
1484
|
+
preflight=preflight,
|
|
1485
|
+
mime_type=mime_type,
|
|
1486
|
+
)
|
|
1487
|
+
dispatch = await _parse_file_with_transient_retries(
|
|
1488
|
+
ingest_id,
|
|
1489
|
+
project_slug,
|
|
1490
|
+
path,
|
|
1491
|
+
mime_type,
|
|
1492
|
+
preflight=preflight,
|
|
1493
|
+
route=route,
|
|
1494
|
+
)
|
|
1495
|
+
parsed = dispatch.parsed
|
|
1496
|
+
parser_used = dispatch.parser_used
|
|
1497
|
+
is_image = _is_image(path, mime_type)
|
|
1498
|
+
is_xlsx = _is_xlsx(path, mime_type)
|
|
1499
|
+
is_docx = _is_docx(path, mime_type)
|
|
1500
|
+
if is_image:
|
|
1501
|
+
item = {
|
|
1502
|
+
"file_path": row["file_path"],
|
|
1503
|
+
"file_size_bytes": row["file_size_bytes"],
|
|
1504
|
+
}
|
|
1505
|
+
auto_approve = should_auto_approve(item, parsed)
|
|
1506
|
+
classification_json = _image_classification(
|
|
1507
|
+
path=path,
|
|
1508
|
+
auto_approve=auto_approve,
|
|
1509
|
+
reason=(
|
|
1510
|
+
"safe image fast-lane"
|
|
1511
|
+
if auto_approve
|
|
1512
|
+
else "image requires manual triage"
|
|
1513
|
+
),
|
|
1514
|
+
rules_matched=(
|
|
1515
|
+
["safe_ext", "under_1mb", "exif_redacted", "no_pii"]
|
|
1516
|
+
if auto_approve
|
|
1517
|
+
else []
|
|
1518
|
+
),
|
|
1519
|
+
)
|
|
1520
|
+
next_status = "done" if auto_approve else "awaiting_triage"
|
|
1521
|
+
extracted_text = str(
|
|
1522
|
+
parsed.get("text") or parsed.get("extracted_text") or ""
|
|
1523
|
+
)
|
|
1524
|
+
structure = parsed.get("structure") or {}
|
|
1525
|
+
target_folder = classification_json["target_folder"]
|
|
1526
|
+
target_filename = classification_json["target_filename"]
|
|
1527
|
+
triage_decision_id = "auto_approve:image_parser" if auto_approve else None
|
|
1528
|
+
elif is_xlsx:
|
|
1529
|
+
classification_json = _xlsx_classification(path)
|
|
1530
|
+
next_status = "awaiting_triage"
|
|
1531
|
+
extracted_text = str(parsed.get("text") or "")
|
|
1532
|
+
structure = parsed.get("structure") or {}
|
|
1533
|
+
target_folder = classification_json["target_folder"]
|
|
1534
|
+
target_filename = classification_json["target_filename"]
|
|
1535
|
+
triage_decision_id = None
|
|
1536
|
+
elif is_docx:
|
|
1537
|
+
classification = classify_markdown(
|
|
1538
|
+
frontmatter=parsed.frontmatter,
|
|
1539
|
+
original_filename=f"{path.stem}.md",
|
|
1540
|
+
)
|
|
1541
|
+
classification_json = classification.as_json()
|
|
1542
|
+
next_status = "awaiting_triage"
|
|
1543
|
+
extracted_text = parsed.text
|
|
1544
|
+
structure = parsed.structure
|
|
1545
|
+
target_folder = classification.target_folder
|
|
1546
|
+
target_filename = classification.target_filename
|
|
1547
|
+
triage_decision_id = None
|
|
1548
|
+
else:
|
|
1549
|
+
classification = classify_markdown(
|
|
1550
|
+
frontmatter=parsed.frontmatter,
|
|
1551
|
+
original_filename=_classification_filename(path, mime_type),
|
|
1552
|
+
)
|
|
1553
|
+
classification_json = classification.as_json()
|
|
1554
|
+
next_status = "awaiting_triage"
|
|
1555
|
+
extracted_text = parsed.text
|
|
1556
|
+
structure = parsed.structure
|
|
1557
|
+
target_folder = classification.target_folder
|
|
1558
|
+
target_filename = classification.target_filename
|
|
1559
|
+
triage_decision_id = None
|
|
1560
|
+
|
|
1561
|
+
parser_quality = dict(dispatch.parser_quality)
|
|
1562
|
+
transcript_summary = await _maybe_summarize_transcript(
|
|
1563
|
+
ingest_id=ingest_id,
|
|
1564
|
+
parser_used=parser_used,
|
|
1565
|
+
extracted_text=extracted_text,
|
|
1566
|
+
structure=structure,
|
|
1567
|
+
)
|
|
1568
|
+
if transcript_summary is not None:
|
|
1569
|
+
structure = dict(structure or {})
|
|
1570
|
+
structure["transcript_summary"] = transcript_summary
|
|
1571
|
+
if transcript_summary.get("status") == "ok":
|
|
1572
|
+
classification_json = dict(classification_json)
|
|
1573
|
+
classification_json["transcript_summary"] = transcript_summary
|
|
1574
|
+
parser_quality["transcript_summary"] = transcript_summary
|
|
1575
|
+
|
|
1576
|
+
structure = _with_ingest_v2_diagnostics(
|
|
1577
|
+
structure,
|
|
1578
|
+
preflight=dispatch.preflight,
|
|
1579
|
+
route=dispatch.route,
|
|
1580
|
+
parser_quality=parser_quality,
|
|
1581
|
+
)
|
|
1582
|
+
|
|
1583
|
+
if next_status == "awaiting_triage" and extracted_text.strip():
|
|
1584
|
+
# Ingestor 2.0: optional local tier-fast project routing +
|
|
1585
|
+
# frontmatter inference. Auto-approval requires composite
|
|
1586
|
+
# confidence >= 0.80, not just LLM self-confidence.
|
|
1587
|
+
try:
|
|
1588
|
+
enriched = await _maybe_llm_enrich(
|
|
1589
|
+
ingest_id=ingest_id,
|
|
1590
|
+
extracted_text=extracted_text,
|
|
1591
|
+
base_classification=classification_json,
|
|
1592
|
+
preflight=dispatch.preflight,
|
|
1593
|
+
parser_quality=parser_quality,
|
|
1594
|
+
route=dispatch.route,
|
|
1595
|
+
)
|
|
1596
|
+
except Exception: # noqa: BLE001 - never break the saga
|
|
1597
|
+
logger.exception("llm enrichment failed")
|
|
1598
|
+
enriched = None
|
|
1599
|
+
if enriched is not None:
|
|
1600
|
+
classification_json = enriched
|
|
1601
|
+
target_folder = enriched.get("target_folder", target_folder)
|
|
1602
|
+
llm_error_message = _llm_failure_error_message(enriched)
|
|
1603
|
+
if llm_error_message:
|
|
1604
|
+
next_status = "parse_error"
|
|
1605
|
+
error_message = llm_error_message
|
|
1606
|
+
triage_decision_id = None
|
|
1607
|
+
elif enriched.get("auto_approve") is True:
|
|
1608
|
+
# Stay in 'approved' so execute_saga (scheduled below)
|
|
1609
|
+
# picks the row up; saga moves the file to target_folder,
|
|
1610
|
+
# populates the KG, indexes the embedding, then flips
|
|
1611
|
+
# status to 'inserted' and finally 'done'.
|
|
1612
|
+
next_status = "approved"
|
|
1613
|
+
triage_decision_id = "auto_approve:llm_routing"
|
|
1614
|
+
# Move row + file to LLM-suggested project so the saga
|
|
1615
|
+
# processes the artifact under the new project root
|
|
1616
|
+
# (saga rejects file_path that escapes project_root).
|
|
1617
|
+
suggested_slug = enriched.get("suggested_project_slug")
|
|
1618
|
+
if suggested_slug and suggested_slug != project_slug:
|
|
1619
|
+
pending_llm_project_slug = str(suggested_slug)
|
|
1620
|
+
|
|
1621
|
+
# --- U3 per-source policy gate (single authority; saga only asserts) ---
|
|
1622
|
+
# Owner surfaces are policy-exempt (decide_ingress_routing returns None).
|
|
1623
|
+
# api_ingress: 'open'/unknown -> always triage (default-deny); 'trusted'
|
|
1624
|
+
# -> keeps the intrinsic auto-insert decision (necessary-not-sufficient).
|
|
1625
|
+
# An 'open' downgrade flips next_status to awaiting_triage, which also
|
|
1626
|
+
# disables the saga/project-switch blocks below (they require 'approved').
|
|
1627
|
+
ingress_routing = decide_ingress_routing(
|
|
1628
|
+
source_kind=row["source_kind"],
|
|
1629
|
+
ingest_policy=row["ingest_policy"],
|
|
1630
|
+
intrinsic_status=next_status,
|
|
1631
|
+
intrinsic_basis=triage_decision_id,
|
|
1632
|
+
)
|
|
1633
|
+
if ingress_routing is not None:
|
|
1634
|
+
next_status = ingress_routing.status
|
|
1635
|
+
triage_decision_id = ingress_routing.triage_decision_id
|
|
1636
|
+
if pending_llm_project_slug and not ingress_routing.auto_insert:
|
|
1637
|
+
pending_llm_project_slug = None
|
|
1638
|
+
classification_json = dict(classification_json)
|
|
1639
|
+
classification_json["ingest_policy"] = {
|
|
1640
|
+
"effective_policy": row["ingest_policy"] or "open",
|
|
1641
|
+
"auto_insert": ingress_routing.auto_insert,
|
|
1642
|
+
"decision": ingress_routing.decision,
|
|
1643
|
+
}
|
|
1644
|
+
if "auto_approve" in classification_json:
|
|
1645
|
+
classification_json["auto_approve"] = ingress_routing.auto_insert
|
|
1646
|
+
|
|
1647
|
+
# --- U3 reconcile ingress payload metadata into structure_json ---
|
|
1648
|
+
ingress_metadata_raw = (
|
|
1649
|
+
row["ingress_metadata"] if "ingress_metadata" in row.keys() else None
|
|
1650
|
+
)
|
|
1651
|
+
structure = _reconcile_ingress_metadata(structure, ingress_metadata_raw)
|
|
1652
|
+
|
|
1653
|
+
async with acquire_write_db() as db:
|
|
1654
|
+
if (
|
|
1655
|
+
pending_llm_project_slug
|
|
1656
|
+
and next_status == "approved"
|
|
1657
|
+
and triage_decision_id == "auto_approve:llm_routing"
|
|
1658
|
+
):
|
|
1659
|
+
(
|
|
1660
|
+
project_slug,
|
|
1661
|
+
path,
|
|
1662
|
+
classification_json,
|
|
1663
|
+
next_status,
|
|
1664
|
+
triage_decision_id,
|
|
1665
|
+
) = await _apply_llm_project_switch_if_safe(
|
|
1666
|
+
db,
|
|
1667
|
+
ingest_id=ingest_id,
|
|
1668
|
+
sha256=row["sha256"],
|
|
1669
|
+
current_project_slug=project_slug,
|
|
1670
|
+
target_project_slug=pending_llm_project_slug,
|
|
1671
|
+
path=path,
|
|
1672
|
+
classification_json=classification_json,
|
|
1673
|
+
)
|
|
1674
|
+
await db.execute(
|
|
1675
|
+
"""
|
|
1676
|
+
UPDATE ingest_pending
|
|
1677
|
+
SET status = ?,
|
|
1678
|
+
project_slug = ?,
|
|
1679
|
+
file_path = ?,
|
|
1680
|
+
mime_type = ?,
|
|
1681
|
+
parser_used = ?,
|
|
1682
|
+
extracted_text = ?,
|
|
1683
|
+
structure_json = ?,
|
|
1684
|
+
classification_json = ?,
|
|
1685
|
+
target_folder = ?,
|
|
1686
|
+
target_filename = ?,
|
|
1687
|
+
triage_decision_id = ?,
|
|
1688
|
+
error_message = ?,
|
|
1689
|
+
updated_at = datetime('now')
|
|
1690
|
+
WHERE id = ?
|
|
1691
|
+
""",
|
|
1692
|
+
(
|
|
1693
|
+
next_status,
|
|
1694
|
+
project_slug,
|
|
1695
|
+
str(path),
|
|
1696
|
+
mime_type,
|
|
1697
|
+
parser_used,
|
|
1698
|
+
extracted_text,
|
|
1699
|
+
json.dumps(structure, ensure_ascii=False),
|
|
1700
|
+
json.dumps(classification_json, ensure_ascii=False),
|
|
1701
|
+
target_folder,
|
|
1702
|
+
target_filename,
|
|
1703
|
+
triage_decision_id,
|
|
1704
|
+
error_message,
|
|
1705
|
+
ingest_id,
|
|
1706
|
+
),
|
|
1707
|
+
)
|
|
1708
|
+
await db.commit()
|
|
1709
|
+
await broadcast_ingest_changed(
|
|
1710
|
+
_ingest_event_for_status(next_status),
|
|
1711
|
+
ingest_id=ingest_id,
|
|
1712
|
+
project_slug=project_slug,
|
|
1713
|
+
status=next_status,
|
|
1714
|
+
)
|
|
1715
|
+
if (
|
|
1716
|
+
next_status == "approved"
|
|
1717
|
+
and triage_decision_id == "auto_approve:llm_routing"
|
|
1718
|
+
):
|
|
1719
|
+
# Lazy import to avoid circular dependency (insert_saga imports
|
|
1720
|
+
# nothing here, but parse_pending is the canonical entry point
|
|
1721
|
+
# so we keep the boundary explicit).
|
|
1722
|
+
from core.api.services.ingest.insert_saga import execute_saga
|
|
1723
|
+
|
|
1724
|
+
asyncio.create_task(execute_saga(ingest_id))
|
|
1725
|
+
except _ParserWaitCancelled:
|
|
1726
|
+
logger.info("ingest parse cancelled before parser slot: id=%s", ingest_id)
|
|
1727
|
+
return
|
|
1728
|
+
except Exception as exc:
|
|
1729
|
+
logger.exception("ingest parse failed: id=%s", ingest_id)
|
|
1730
|
+
await _mark_parse_error(ingest_id, project_slug, str(exc))
|
|
1731
|
+
|
|
1732
|
+
|
|
1733
|
+
def _source_context_for_row(
|
|
1734
|
+
*,
|
|
1735
|
+
project_slug: str | None,
|
|
1736
|
+
source_kind: str | None,
|
|
1737
|
+
path: Path,
|
|
1738
|
+
) -> dict[str, Any]:
|
|
1739
|
+
if not project_slug or source_kind != "terminal_upload":
|
|
1740
|
+
return {}
|
|
1741
|
+
try:
|
|
1742
|
+
project_root = PROJECTS_ROOT / project_slug
|
|
1743
|
+
in_project_input = path.resolve().is_relative_to((project_root / "input").resolve())
|
|
1744
|
+
except Exception:
|
|
1745
|
+
in_project_input = False
|
|
1746
|
+
if in_project_input:
|
|
1747
|
+
return {
|
|
1748
|
+
"project_slug": project_slug,
|
|
1749
|
+
"prior": 0.95,
|
|
1750
|
+
"reason": f"{source_kind or 'ingest'}_project_input",
|
|
1751
|
+
}
|
|
1752
|
+
return {
|
|
1753
|
+
"project_slug": project_slug,
|
|
1754
|
+
"prior": 0.75,
|
|
1755
|
+
"reason": f"{source_kind or 'ingest'}_row_project",
|
|
1756
|
+
}
|