marvisx-cli 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (587) hide show
  1. core/api/__init__.py +0 -0
  2. core/api/agents/__init__.py +0 -0
  3. core/api/agents/session_health.py +59 -0
  4. core/api/agents/session_manager.py +206 -0
  5. core/api/bin/marvisx-state-hook.py +182 -0
  6. core/api/config.py +533 -0
  7. core/api/db.py +1516 -0
  8. core/api/dependencies/__init__.py +0 -0
  9. core/api/dependencies/tenant.py +34 -0
  10. core/api/main.py +1641 -0
  11. core/api/mcp/__init__.py +8 -0
  12. core/api/mcp/_adapter.py +184 -0
  13. core/api/mcp/server.py +58 -0
  14. core/api/mcp/tools/__init__.py +59 -0
  15. core/api/mcp/tools/brain.py +599 -0
  16. core/api/mcp/tools/graph.py +380 -0
  17. core/api/mcp/tools/handoffs.py +112 -0
  18. core/api/mcp/tools/ingest.py +326 -0
  19. core/api/mcp/tools/learnings.py +144 -0
  20. core/api/mcp/tools/projects.py +99 -0
  21. core/api/mcp/tools/pull_requests.py +173 -0
  22. core/api/mcp/tools/safety.py +111 -0
  23. core/api/mcp/tools/search.py +79 -0
  24. core/api/mcp/tools/tasks.py +258 -0
  25. core/api/middleware/__init__.py +0 -0
  26. core/api/middleware/tool_call_audit.py +111 -0
  27. core/api/models/__init__.py +346 -0
  28. core/api/models/auth.py +48 -0
  29. core/api/models/brain.py +1006 -0
  30. core/api/models/common.py +76 -0
  31. core/api/models/costs.py +91 -0
  32. core/api/models/graph.py +66 -0
  33. core/api/models/graph_cosmo.py +125 -0
  34. core/api/models/graph_pr_impact.py +257 -0
  35. core/api/models/graph_ux.py +141 -0
  36. core/api/models/inbox.py +230 -0
  37. core/api/models/ingest_keys.py +108 -0
  38. core/api/models/kg.py +41 -0
  39. core/api/models/llm_config.py +56 -0
  40. core/api/models/monitoring.py +234 -0
  41. core/api/models/projects.py +161 -0
  42. core/api/models/search.py +42 -0
  43. core/api/models/sessions.py +322 -0
  44. core/api/models/tasks.py +184 -0
  45. core/api/models/teams.py +63 -0
  46. core/api/models/users.py +184 -0
  47. core/api/observability/__init__.py +0 -0
  48. core/api/observability/tracing.py +92 -0
  49. core/api/paths.py +26 -0
  50. core/api/rate_limit.py +24 -0
  51. core/api/rbac.py +112 -0
  52. core/api/routers/__init__.py +0 -0
  53. core/api/routers/_adapter.py +24 -0
  54. core/api/routers/admin_pr_impact.py +230 -0
  55. core/api/routers/admin_settings.py +147 -0
  56. core/api/routers/agent.py +1079 -0
  57. core/api/routers/agent_tokens.py +276 -0
  58. core/api/routers/app_settings.py +112 -0
  59. core/api/routers/audit.py +89 -0
  60. core/api/routers/auth.py +586 -0
  61. core/api/routers/bench.py +161 -0
  62. core/api/routers/brain.py +881 -0
  63. core/api/routers/brain_directions.py +527 -0
  64. core/api/routers/ci_checks.py +140 -0
  65. core/api/routers/comments.py +273 -0
  66. core/api/routers/costs.py +148 -0
  67. core/api/routers/docs_coverage.py +217 -0
  68. core/api/routers/docs_governance.py +63 -0
  69. core/api/routers/documents.py +318 -0
  70. core/api/routers/files.py +163 -0
  71. core/api/routers/finder.py +987 -0
  72. core/api/routers/graph.py +836 -0
  73. core/api/routers/handoffs.py +156 -0
  74. core/api/routers/inbox.py +496 -0
  75. core/api/routers/ingest_api_keys.py +205 -0
  76. core/api/routers/ingest_triage.py +1227 -0
  77. core/api/routers/judge.py +306 -0
  78. core/api/routers/kg.py +336 -0
  79. core/api/routers/learnings.py +253 -0
  80. core/api/routers/llm_config.py +130 -0
  81. core/api/routers/monitoring.py +347 -0
  82. core/api/routers/notifications.py +125 -0
  83. core/api/routers/pr_impact.py +315 -0
  84. core/api/routers/projects.py +1061 -0
  85. core/api/routers/pull_requests.py +312 -0
  86. core/api/routers/push.py +67 -0
  87. core/api/routers/raci.py +228 -0
  88. core/api/routers/search.py +125 -0
  89. core/api/routers/sessions.py +3100 -0
  90. core/api/routers/settings.py +90 -0
  91. core/api/routers/share_repo.py +68 -0
  92. core/api/routers/status_updates.py +96 -0
  93. core/api/routers/tags.py +45 -0
  94. core/api/routers/tasks.py +526 -0
  95. core/api/routers/teams.py +425 -0
  96. core/api/routers/terminal.py +105 -0
  97. core/api/routers/users.py +331 -0
  98. core/api/routers/webhooks.py +330 -0
  99. core/api/runtime_settings.py +84 -0
  100. core/api/security.py +652 -0
  101. core/api/services/__init__.py +0 -0
  102. core/api/services/audit.py +58 -0
  103. core/api/services/auto_approval.py +82 -0
  104. core/api/services/brain/__init__.py +52 -0
  105. core/api/services/brain/baseline.py +230 -0
  106. core/api/services/brain/capabilities.py +75 -0
  107. core/api/services/brain/cascade_rollup.py +388 -0
  108. core/api/services/brain/compound_bridge.py +215 -0
  109. core/api/services/brain/cycle.py +1242 -0
  110. core/api/services/brain/cycle_snapshot.py +371 -0
  111. core/api/services/brain/digest_collector.py +147 -0
  112. core/api/services/brain/direction.py +421 -0
  113. core/api/services/brain/drift.py +356 -0
  114. core/api/services/brain/drift_router.py +409 -0
  115. core/api/services/brain/edge_metrics.py +79 -0
  116. core/api/services/brain/events_reader.py +222 -0
  117. core/api/services/brain/findings.py +1379 -0
  118. core/api/services/brain/findings_reader.py +1006 -0
  119. core/api/services/brain/jobs.py +733 -0
  120. core/api/services/brain/journal.py +206 -0
  121. core/api/services/brain/knowledge_forms.py +92 -0
  122. core/api/services/brain/llm/__init__.py +37 -0
  123. core/api/services/brain/llm/_runner.py +62 -0
  124. core/api/services/brain/llm/base.py +70 -0
  125. core/api/services/brain/llm/cache.py +99 -0
  126. core/api/services/brain/llm/constants.py +46 -0
  127. core/api/services/brain/llm/direction_alignment.py +289 -0
  128. core/api/services/brain/llm/factory.py +132 -0
  129. core/api/services/brain/llm/finding_reasoning.py +98 -0
  130. core/api/services/brain/llm/finding_summary.py +92 -0
  131. core/api/services/brain/llm/grounding.py +46 -0
  132. core/api/services/brain/llm/journal_polish.py +96 -0
  133. core/api/services/brain/llm/local_gateway.py +426 -0
  134. core/api/services/brain/llm/parsers.py +71 -0
  135. core/api/services/brain/llm/router_glue.py +422 -0
  136. core/api/services/brain/memory_ops.py +1677 -0
  137. core/api/services/brain/models.py +140 -0
  138. core/api/services/brain/owner_hint.py +211 -0
  139. core/api/services/brain/recap.py +307 -0
  140. core/api/services/brain/rules/__init__.py +65 -0
  141. core/api/services/brain/rules/_signals.py +205 -0
  142. core/api/services/brain/rules/dr1_activity_without_status.py +108 -0
  143. core/api/services/brain/rules/dr2_decision_without_adr.py +110 -0
  144. core/api/services/brain/rules/dr3_stale_open_loop.py +127 -0
  145. core/api/services/brain/rules/dr4_docs_governance_drift.py +79 -0
  146. core/api/services/brain/rules/dr5_playbook_changed.py +99 -0
  147. core/api/services/brain/rules/dr6_external_update_unpropagated.py +100 -0
  148. core/api/services/brain/rules/dr7_claimed_decision_gap.py +123 -0
  149. core/api/services/brain/rules/dr8_direction_misalignment.py +230 -0
  150. core/api/services/brain/runs_reader.py +485 -0
  151. core/api/services/brain/scope.py +79 -0
  152. core/api/services/brain/sources/__init__.py +46 -0
  153. core/api/services/brain/sources/base.py +86 -0
  154. core/api/services/brain/sources/git_kg.py +393 -0
  155. core/api/services/brain/sources/handoffs.py +157 -0
  156. core/api/services/brain/sources/ingestor.py +130 -0
  157. core/api/services/brain/sources/learnings.py +121 -0
  158. core/api/services/brain/sources/pir_tasks.py +245 -0
  159. core/api/services/brain/watermarks.py +147 -0
  160. core/api/services/brain/ws_emitter.py +170 -0
  161. core/api/services/cc_tasks_reader.py +76 -0
  162. core/api/services/ci_service.py +263 -0
  163. core/api/services/claude_metrics.py +796 -0
  164. core/api/services/codex_metrics.py +364 -0
  165. core/api/services/conversation_reader.py +102 -0
  166. core/api/services/cost_service.py +243 -0
  167. core/api/services/crypto.py +147 -0
  168. core/api/services/docs_governance/__init__.py +1 -0
  169. core/api/services/docs_governance/confidence.py +230 -0
  170. core/api/services/docs_governance/config.py +87 -0
  171. core/api/services/docs_governance/enrichment.py +65 -0
  172. core/api/services/docs_governance/frontmatter_validator.py +83 -0
  173. core/api/services/docs_governance/hard_gates.py +221 -0
  174. core/api/services/docs_governance/triage_orchestrator.py +98 -0
  175. core/api/services/embedding_internal.py +395 -0
  176. core/api/services/embedding_service.py +832 -0
  177. core/api/services/event_dispatcher.py +167 -0
  178. core/api/services/events.py +70 -0
  179. core/api/services/git_ops.py +621 -0
  180. core/api/services/graph_cosmo_service.py +440 -0
  181. core/api/services/graph_ranker.py +306 -0
  182. core/api/services/graph_service.py +1589 -0
  183. core/api/services/inbox.py +800 -0
  184. core/api/services/inbox_digest.py +221 -0
  185. core/api/services/inbox_digest_deep_research.py +80 -0
  186. core/api/services/inbox_digest_jobs.py +595 -0
  187. core/api/services/inbox_gmail_sync.py +167 -0
  188. core/api/services/inbox_llm_classifier.py +906 -0
  189. core/api/services/inbox_source_identity.py +116 -0
  190. core/api/services/inbox_sources.py +456 -0
  191. core/api/services/inbox_taxonomy.py +195 -0
  192. core/api/services/inbox_tldr.py +1079 -0
  193. core/api/services/inbox_triage.py +899 -0
  194. core/api/services/ingest/__init__.py +13 -0
  195. core/api/services/ingest/api_key_auth.py +136 -0
  196. core/api/services/ingest/auto_approve.py +120 -0
  197. core/api/services/ingest/classifier.py +138 -0
  198. core/api/services/ingest/confidence.py +173 -0
  199. core/api/services/ingest/dispatch.py +88 -0
  200. core/api/services/ingest/embedding_router.py +272 -0
  201. core/api/services/ingest/events.py +33 -0
  202. core/api/services/ingest/ignore_patterns.py +79 -0
  203. core/api/services/ingest/image_probe.py +218 -0
  204. core/api/services/ingest/ingress.py +263 -0
  205. core/api/services/ingest/insert_saga.py +793 -0
  206. core/api/services/ingest/llm/__init__.py +13 -0
  207. core/api/services/ingest/llm/anthropic_haiku.py +23 -0
  208. core/api/services/ingest/llm/base.py +59 -0
  209. core/api/services/ingest/llm/byok_provider.py +130 -0
  210. core/api/services/ingest/llm/classification_context.py +301 -0
  211. core/api/services/ingest/llm/config_store.py +246 -0
  212. core/api/services/ingest/llm/factory.py +24 -0
  213. core/api/services/ingest/llm/kg_enricher.py +306 -0
  214. core/api/services/ingest/llm/local_gateway.py +821 -0
  215. core/api/services/ingest/llm/local_vllm.py +23 -0
  216. core/api/services/ingest/llm/openai_nano.py +349 -0
  217. core/api/services/ingest/lock_advisory.py +57 -0
  218. core/api/services/ingest/parser_router.py +1756 -0
  219. core/api/services/ingest/parsers/__init__.py +1 -0
  220. core/api/services/ingest/parsers/docling_parser.py +142 -0
  221. core/api/services/ingest/parsers/docparse_gateway.py +178 -0
  222. core/api/services/ingest/parsers/docx_parser.py +127 -0
  223. core/api/services/ingest/parsers/folder_unpacker.py +85 -0
  224. core/api/services/ingest/parsers/gateway_aux.py +147 -0
  225. core/api/services/ingest/parsers/image_parser.py +251 -0
  226. core/api/services/ingest/parsers/internal_markdown.py +89 -0
  227. core/api/services/ingest/parsers/ocr_gateway.py +117 -0
  228. core/api/services/ingest/parsers/ocr_pdf_parser.py +112 -0
  229. core/api/services/ingest/parsers/pdf_types.py +13 -0
  230. core/api/services/ingest/parsers/transcript_parser.py +445 -0
  231. core/api/services/ingest/parsers/vision_gateway.py +186 -0
  232. core/api/services/ingest/parsers/xlsx_parser.py +91 -0
  233. core/api/services/ingest/parsers/zip_unpacker.py +126 -0
  234. core/api/services/ingest/preflight.py +393 -0
  235. core/api/services/ingest/retry_voyage.py +88 -0
  236. core/api/services/ingest/routing_policy.py +307 -0
  237. core/api/services/ingest/serializers/__init__.py +1 -0
  238. core/api/services/ingest/serializers/xlsx_to_markdown.py +80 -0
  239. core/api/services/ingest/skip_log.py +74 -0
  240. core/api/services/ingest/watcher.py +637 -0
  241. core/api/services/kg/__init__.py +0 -0
  242. core/api/services/kg/audit.py +49 -0
  243. core/api/services/kg/hybrid_search.py +691 -0
  244. core/api/services/kg/lens.py +339 -0
  245. core/api/services/kg/pr_impact.py +770 -0
  246. core/api/services/kg/queries.py +152 -0
  247. core/api/services/kg/ranking.py +89 -0
  248. core/api/services/kg/rrf.py +143 -0
  249. core/api/services/kg_watcher_control.py +161 -0
  250. core/api/services/local_llm/__init__.py +19 -0
  251. core/api/services/local_llm/async_client.py +385 -0
  252. core/api/services/local_llm/client.py +173 -0
  253. core/api/services/local_llm/url_validator.py +44 -0
  254. core/api/services/metrics_collector.py +646 -0
  255. core/api/services/metrics_providers.py +65 -0
  256. core/api/services/model_registry.py +266 -0
  257. core/api/services/model_router.py +137 -0
  258. core/api/services/n8n_client.py +77 -0
  259. core/api/services/newsletter_llm_gateway.py +66 -0
  260. core/api/services/notification_service.py +134 -0
  261. core/api/services/openai_responses.py +55 -0
  262. core/api/services/opencode_metrics.py +375 -0
  263. core/api/services/opencode_sessions.py +173 -0
  264. core/api/services/pii_redactor.py +138 -0
  265. core/api/services/pr_impact_pipeline/__init__.py +21 -0
  266. core/api/services/pr_impact_pipeline/differ.py +421 -0
  267. core/api/services/pr_impact_pipeline/dispatcher.py +415 -0
  268. core/api/services/pr_impact_pipeline/gc.py +93 -0
  269. core/api/services/pr_impact_pipeline/languages.py +192 -0
  270. core/api/services/pr_impact_pipeline/parser.py +178 -0
  271. core/api/services/pr_impact_pipeline/writer.py +394 -0
  272. core/api/services/pr_service.py +1393 -0
  273. core/api/services/project_paths.py +70 -0
  274. core/api/services/project_status_updates.py +265 -0
  275. core/api/services/providers.py +276 -0
  276. core/api/services/push_service.py +170 -0
  277. core/api/services/reminder_service.py +89 -0
  278. core/api/services/runas.py +41 -0
  279. core/api/services/salience_service.py +69 -0
  280. core/api/services/security_collector.py +281 -0
  281. core/api/services/session_catalog.py +385 -0
  282. core/api/services/session_metrics_service.py +301 -0
  283. core/api/services/session_ops.py +272 -0
  284. core/api/services/session_state.py +173 -0
  285. core/api/services/share_links.py +222 -0
  286. core/api/services/task_transitions.py +146 -0
  287. core/api/services/terminal_metrics.py +462 -0
  288. core/api/services/terminal_metrics_dump.py +203 -0
  289. core/api/services/tmux.py +1205 -0
  290. core/api/services/webhook_service.py +422 -0
  291. core/api/services/workspace_sync.py +164 -0
  292. core/api/templates/__init__.py +1 -0
  293. core/api/templates/markdown_share.py +164 -0
  294. core/api/terminal.py +1031 -0
  295. core/api/tests/__init__.py +0 -0
  296. core/api/tests/test_agent_facing_auth_dependencies.py +132 -0
  297. core/api/tests/test_audit_permissions.py +133 -0
  298. core/api/tests/test_backfill_session_conversations.py +90 -0
  299. core/api/tests/test_backfill_working_seconds_msg.py +129 -0
  300. core/api/tests/test_claude_metrics.py +326 -0
  301. core/api/tests/test_codex_metrics.py +189 -0
  302. core/api/tests/test_finder_paths.py +74 -0
  303. core/api/tests/test_git_ops_merge.py +155 -0
  304. core/api/tests/test_learnings_check_search.py +81 -0
  305. core/api/tests/test_metrics_providers.py +133 -0
  306. core/api/tests/test_migration_087.py +164 -0
  307. core/api/tests/test_migration_088.py +94 -0
  308. core/api/tests/test_migration_089.py +116 -0
  309. core/api/tests/test_openai_responses.py +24 -0
  310. core/api/tests/test_opencode_metrics.py +740 -0
  311. core/api/tests/test_opencode_sessions.py +321 -0
  312. core/api/tests/test_pr_workflow_e2e.py +457 -0
  313. core/api/tests/test_projects_handoffs.py +31 -0
  314. core/api/tests/test_providers.py +138 -0
  315. core/api/tests/test_safety_bridge.py +347 -0
  316. core/api/tests/test_session_catalog.py +142 -0
  317. core/api/tests/test_session_conversations.py +512 -0
  318. core/api/tests/test_session_metrics_service.py +270 -0
  319. core/api/tests/test_session_resume_paths.py +548 -0
  320. core/api/tests/test_session_theme_mode_migration.py +56 -0
  321. core/api/tests/test_sessions_rbac.py +131 -0
  322. core/api/tests/test_share_edit.py +398 -0
  323. core/api/tests/test_share_repo.py +200 -0
  324. core/api/tests/test_terminal_session_manager.py +98 -0
  325. core/api/tests/test_terminal_upload.py +34 -0
  326. core/api/tests/test_tmux.py +272 -0
  327. core/api/tests/test_workspace_sync.py +186 -0
  328. core/api/tests/test_ws_ticket_in_memory.py +73 -0
  329. core/api/use_cases/__init__.py +11 -0
  330. core/api/use_cases/_context.py +89 -0
  331. core/api/use_cases/_errors.py +62 -0
  332. core/api/use_cases/_roles.py +16 -0
  333. core/api/use_cases/audit.py +171 -0
  334. core/api/use_cases/brain.py +1232 -0
  335. core/api/use_cases/costs.py +249 -0
  336. core/api/use_cases/graph.py +1153 -0
  337. core/api/use_cases/handoffs.py +506 -0
  338. core/api/use_cases/ingest_triage.py +1229 -0
  339. core/api/use_cases/learnings.py +538 -0
  340. core/api/use_cases/projects.py +705 -0
  341. core/api/use_cases/pull_requests.py +415 -0
  342. core/api/use_cases/search.py +926 -0
  343. core/api/use_cases/tasks.py +1495 -0
  344. core/api/visibility.py +141 -0
  345. core/cli/__init__.py +5 -0
  346. core/cli/_index_source.py +632 -0
  347. core/cli/_runtime_ctx.py +160 -0
  348. core/cli/_transmute.py +241 -0
  349. core/cli/marvis_doctor.py +704 -0
  350. core/cli/marvis_feedback.py +396 -0
  351. core/cli/marvis_governance.py +315 -0
  352. core/cli/marvis_hooks.py +515 -0
  353. core/cli/marvis_init.py +757 -0
  354. core/cli/marvis_mcp.py +401 -0
  355. core/cli/marvis_runtime.py +855 -0
  356. core/cli/marvis_telemetry.py +228 -0
  357. core/scripts/_drift_check.py +716 -0
  358. core/scripts/_frontmatter.py +66 -0
  359. core/scripts/_graph_writer.py +189 -0
  360. core/scripts/ast_parser.py +1553 -0
  361. core/scripts/install_hooks/__init__.py +1 -0
  362. core/scripts/install_hooks/_config.sh +109 -0
  363. core/scripts/install_hooks/block-dangerous-bash.sh +23 -0
  364. core/scripts/install_hooks/block-db-direct-write.sh +23 -0
  365. core/scripts/install_hooks/block-push-no-task.sh +23 -0
  366. core/scripts/install_hooks/block-staging-to-prod.sh +23 -0
  367. core/scripts/install_hooks/block-subtree-push.sh +23 -0
  368. core/scripts/install_hooks/config.json +53 -0
  369. core/scripts/install_hooks/enforce-no-merge-main.sh +23 -0
  370. core/scripts/install_hooks/enforce-worktree.sh +23 -0
  371. core/scripts/install_hooks/quality-gate.sh +170 -0
  372. core/scripts/install_hooks/safety_bridge.py +968 -0
  373. core/scripts/install_hooks/secret-scan.sh +23 -0
  374. core/scripts/migrate_spike_node_ids.py +122 -0
  375. core/scripts/populate_artifacts.py +2198 -0
  376. core/scripts/populate_cross_project.py +2457 -0
  377. core/scripts/populate_inbox_nodes.py +357 -0
  378. core/scripts/populate_pr_impact.py +267 -0
  379. core/scripts/populate_project_nodes.py +603 -0
  380. core/scripts/populate_touch_counter.py +337 -0
  381. core/scripts/reparse_failed.py +57 -0
  382. core/scripts/safety_bridge.py +968 -0
  383. core/telemetry/__init__.py +9 -0
  384. core/telemetry/client.py +405 -0
  385. core/telemetry/schema.py +122 -0
  386. core/wizard/__init__.py +65 -0
  387. core/wizard/byok_vault.py +147 -0
  388. core/wizard/defaults.py +58 -0
  389. core/wizard/state.py +117 -0
  390. core/wizard/steps.py +70 -0
  391. core/wizard/validation.py +136 -0
  392. marvisx_cli-0.1.0.dist-info/METADATA +201 -0
  393. marvisx_cli-0.1.0.dist-info/RECORD +587 -0
  394. marvisx_cli-0.1.0.dist-info/WHEEL +5 -0
  395. marvisx_cli-0.1.0.dist-info/entry_points.txt +3 -0
  396. marvisx_cli-0.1.0.dist-info/licenses/LICENSE +98 -0
  397. marvisx_cli-0.1.0.dist-info/top_level.txt +3 -0
  398. migrations/001_initial.sql +33 -0
  399. migrations/002_tasks.sql +30 -0
  400. migrations/003_session_management.sql +7 -0
  401. migrations/004_projects_comments.sql +65 -0
  402. migrations/005_session_intelligence.sql +15 -0
  403. migrations/006_settings.sql +12 -0
  404. migrations/007_task_scoring.sql +12 -0
  405. migrations/008_cost_tracking.sql +31 -0
  406. migrations/009_session_card_metrics.sql +3 -0
  407. migrations/010_monitoring.sql +55 -0
  408. migrations/012_agent_api.sql +21 -0
  409. migrations/013_session_complete.sql +8 -0
  410. migrations/015_pull_requests.sql +43 -0
  411. migrations/015_pull_requests_down.sql +5 -0
  412. migrations/016_users_raci.sql +116 -0
  413. migrations/017_task_cost_entries.sql +87 -0
  414. migrations/018_agents.sql +73 -0
  415. migrations/018_agents_down.sql +13 -0
  416. migrations/019_review_feedback.sql +18 -0
  417. migrations/020_pr_commit_sha.sql +4 -0
  418. migrations/021_webhook_events.sql +18 -0
  419. migrations/022_devx_agent_managed.sql +11 -0
  420. migrations/022_devx_agent_managed_down.sql +6 -0
  421. migrations/023_devx_p1_gate.sql +7 -0
  422. migrations/023_devx_p1_gate_down.sql +3 -0
  423. migrations/024_chat_messages.sql +16 -0
  424. migrations/024_pr_conversation_id.sql +8 -0
  425. migrations/024_task_indexes.sql +21 -0
  426. migrations/024_task_indexes_down.sql +7 -0
  427. migrations/025_audit_log.sql +17 -0
  428. migrations/026_agent_tokens.sql +20 -0
  429. migrations/027_teams_auth_phase_b.sql +35 -0
  430. migrations/028_learnings.sql +23 -0
  431. migrations/029_team_roles.sql +14 -0
  432. migrations/030_finder_pins.sql +10 -0
  433. migrations/031_pr_deploy_status.sql +9 -0
  434. migrations/032_task_reminders.sql +7 -0
  435. migrations/033_events_retry_count.sql +6 -0
  436. migrations/033_session_owner.sql +9 -0
  437. migrations/034_notifications.sql +38 -0
  438. migrations/035_shared_links.sql +15 -0
  439. migrations/036_session_index_upgrade.sql +29 -0
  440. migrations/037_pr_approval.sql +15 -0
  441. migrations/038_pr_submitted_by.sql +6 -0
  442. migrations/039_push_subscriptions.sql +17 -0
  443. migrations/040_semantic_search.sql +16 -0
  444. migrations/041_workspaces.sql +63 -0
  445. migrations/042_oidc_providers.sql +24 -0
  446. migrations/043_ci_checks.sql +31 -0
  447. migrations/044_agent_metrics.sql +30 -0
  448. migrations/045_documents_doc_type.sql +5 -0
  449. migrations/046_salience.sql +13 -0
  450. migrations/047_seed_missing_agents.sql +6 -0
  451. migrations/048_fix_agent_paths_roles.sql +5 -0
  452. migrations/049_agent_role_and_learnings_schema.sql +3 -0
  453. migrations/050_session_provider.sql +2 -0
  454. migrations/051_session_launch_profile.sql +4 -0
  455. migrations/052_session_theme_mode.sql +2 -0
  456. migrations/052_task_kind.sql +4 -0
  457. migrations/053_inbox_items.sql +31 -0
  458. migrations/054_inbox_triage_contract.sql +30 -0
  459. migrations/055_inbox_topic_treatment.sql +12 -0
  460. migrations/056_inbox_treatment_read_save.sql +57 -0
  461. migrations/057_session_theme_mode_backfill.sql +4 -0
  462. migrations/058_inbox_item_status_lifecycle.sql +13 -0
  463. migrations/059_inbox_tldr_and_source_scores.sql +18 -0
  464. migrations/060_newsletter.sql +16 -0
  465. migrations/061_inbox_redesign.sql +69 -0
  466. migrations/062_fix_inbox_sources_backfill.sql +37 -0
  467. migrations/063_task_completion_mode.sql +23 -0
  468. migrations/064_judge_mode_setting.sql +4 -0
  469. migrations/065_knowledge_graph_spike.sql +40 -0
  470. migrations/066_digest_ranking_inputs.sql +10 -0
  471. migrations/066_kg_artifact_nodes.sql +129 -0
  472. migrations/067_inbox_digest_selections.sql +28 -0
  473. migrations/067_kg_temporal.sql +53 -0
  474. migrations/068_inbox_digest_app_settings.sql +9 -0
  475. migrations/068_kg_touch_counter.sql +52 -0
  476. migrations/069_kg_doc_types.sql +117 -0
  477. migrations/070_digest_ranking_inputs_recovery.sql +3 -0
  478. migrations/071_inbox_digest_selections_recovery.sql +3 -0
  479. migrations/072_inbox_digest_app_settings_recovery.sql +3 -0
  480. migrations/073_kg_cross_project.sql +216 -0
  481. migrations/073_kg_cross_project_down.sql +77 -0
  482. migrations/074_kg_infra_types.sql +208 -0
  483. migrations/074_kg_infra_types_down.sql +80 -0
  484. migrations/075_kg_file_state_recovery.sql +35 -0
  485. migrations/075_kg_file_state_recovery_down.sql +5 -0
  486. migrations/076_kg_watcher_state.sql +33 -0
  487. migrations/076_kg_watcher_state_down.sql +3 -0
  488. migrations/077_kg_doc_types_extend.sql +226 -0
  489. migrations/077_kg_doc_types_extend_down.sql +80 -0
  490. migrations/078_kg_fts5.sql +102 -0
  491. migrations/078_kg_fts5_down.sql +14 -0
  492. migrations/079_kg_missing_indexes.sql +31 -0
  493. migrations/079_kg_missing_indexes_down.sql +10 -0
  494. migrations/080_kg_fts5_extended.sql +232 -0
  495. migrations/080_kg_fts5_extended_down.sql +25 -0
  496. migrations/081_kg_lens_indexes.sql +9 -0
  497. migrations/081_kg_lens_indexes_down.sql +3 -0
  498. migrations/082_kg_pins.sql +26 -0
  499. migrations/082_kg_pins_down.sql +14 -0
  500. migrations/083_kg_graph_nodes_degree.sql +20 -0
  501. migrations/083_kg_graph_nodes_degree_down.sql +15 -0
  502. migrations/084_drop_legacy_scheduler_tables.sql +58 -0
  503. migrations/084_drop_legacy_scheduler_tables_down.sql +112 -0
  504. migrations/085_kg_edge_resolves_to.sql +142 -0
  505. migrations/085_kg_edge_resolves_to_down.sql +66 -0
  506. migrations/086_project_status_updates_feed.sql +20 -0
  507. migrations/086_project_status_updates_feed_down.sql +36 -0
  508. migrations/087_session_metrics_dual.sql +50 -0
  509. migrations/087_session_metrics_dual_down.sql +21 -0
  510. migrations/088_rename_context_pct_legacy.sql +23 -0
  511. migrations/088_rename_context_pct_legacy_down.sql +8 -0
  512. migrations/089_session_metrics_equivalent_cost.sql +26 -0
  513. migrations/089_session_metrics_equivalent_cost_down.sql +11 -0
  514. migrations/090_kg_inbox_node_type.sql +26 -0
  515. migrations/090_kg_inbox_node_type_down.sql +20 -0
  516. migrations/091_kg_inbox_node_type_check.sql +265 -0
  517. migrations/091_kg_inbox_node_type_check_down.sql +129 -0
  518. migrations/092_sessions_activity_state_ts.sql +29 -0
  519. migrations/092_sessions_activity_state_ts_down.sql +14 -0
  520. migrations/093_sessions_activity_state_column.sql +29 -0
  521. migrations/093_sessions_activity_state_column_down.sql +10 -0
  522. migrations/094_ingest_pending.sql +55 -0
  523. migrations/094_ingest_pending_down.sql +15 -0
  524. migrations/095_kg_intent_first.sql +77 -0
  525. migrations/095_kg_intent_first_down.sql +25 -0
  526. migrations/096_kg_xlsx_artifact_prefix.sql +17 -0
  527. migrations/096_kg_xlsx_artifact_prefix_down.sql +11 -0
  528. migrations/097_ingest_change_history.sql +37 -0
  529. migrations/097_ingest_change_history_down.sql +13 -0
  530. migrations/098_kg_node_type_business.sql +254 -0
  531. migrations/098_kg_node_type_business_down.sql +195 -0
  532. migrations/099_kg_edges_restore_weight.sql +58 -0
  533. migrations/099_kg_edges_restore_weight_down.sql +12 -0
  534. migrations/100_kg_enriched_at.sql +25 -0
  535. migrations/100_kg_enriched_at_down.sql +12 -0
  536. migrations/101_local_llm_shadow_comparisons.sql +66 -0
  537. migrations/101_local_llm_shadow_comparisons_down.sql +15 -0
  538. migrations/102_promote_llm_costs.sql +69 -0
  539. migrations/102_promote_llm_costs_down.sql +19 -0
  540. migrations/103_ingest_skipped_log.sql +46 -0
  541. migrations/103_ingest_skipped_log_down.sql +15 -0
  542. migrations/120_docs_governance.sql +50 -0
  543. migrations/120_docs_governance_down.sql +11 -0
  544. migrations/121_notification_event_fk_cleanup.sql +21 -0
  545. migrations/121_notification_event_fk_cleanup_down.sql +10 -0
  546. migrations/122_docs_drift_history.sql +34 -0
  547. migrations/122_docs_drift_history_down.sql +15 -0
  548. migrations/123_ingest_parser_waiting_status.sql +69 -0
  549. migrations/123_ingest_parser_waiting_status_down.sql +69 -0
  550. migrations/124_heypocket_recordings.sql +63 -0
  551. migrations/124_heypocket_recordings_down.sql +13 -0
  552. migrations/125_kg_node_type_record.sql +219 -0
  553. migrations/125_kg_node_type_record_down.sql +205 -0
  554. migrations/126_ingest_terminal_upload_source_kind.sql +69 -0
  555. migrations/126_ingest_terminal_upload_source_kind_down.sql +69 -0
  556. migrations/127_brain_v1_substrate.sql +200 -0
  557. migrations/127_brain_v1_substrate_down.sql +32 -0
  558. migrations/128_brain_drift_signals.sql +157 -0
  559. migrations/128_brain_drift_signals_down.sql +23 -0
  560. migrations/129_brain_memory_operations.sql +232 -0
  561. migrations/129_brain_memory_operations_down.sql +27 -0
  562. migrations/130_brain_findings.sql +258 -0
  563. migrations/130_brain_findings_down.sql +29 -0
  564. migrations/132_kg_pr_modifies.sql +242 -0
  565. migrations/132_kg_pr_modifies_down.sql +99 -0
  566. migrations/133_brain_v1_2_direction_schema.sql +476 -0
  567. migrations/133_brain_v1_2_direction_schema_down.sql +273 -0
  568. migrations/134_brain_journal_narrative_polished.sql +8 -0
  569. migrations/134_brain_journal_narrative_polished_down.sql +6 -0
  570. migrations/135_kg_edges_provider.sql +21 -0
  571. migrations/136_documents_fts.sql +56 -0
  572. migrations/137_promote_llm_costs.sql +59 -0
  573. migrations/137_promote_llm_costs_down.sql +19 -0
  574. migrations/138_ingest_api_keys.sql +39 -0
  575. migrations/138_ingest_api_keys_down.sql +8 -0
  576. migrations/139_ingest_pending_ingress.sql +91 -0
  577. migrations/139_ingest_pending_ingress_down.sql +73 -0
  578. migrations/140_ingest_idempotency_quota.sql +45 -0
  579. migrations/140_ingest_idempotency_quota_down.sql +9 -0
  580. migrations/141_ingest_pending_metadata.sql +16 -0
  581. migrations/141_ingest_pending_metadata_down.sql +7 -0
  582. migrations/142_llm_function_config.sql +36 -0
  583. migrations/142_llm_function_config_down.sql +8 -0
  584. migrations/143_kg_code_embeddings.sql +25 -0
  585. migrations/143_kg_code_embeddings_down.sql +5 -0
  586. migrations/__init__.py +4 -0
  587. projects/_template/project.yaml +46 -0
@@ -0,0 +1,1756 @@
1
+ """Parser dispatch for ingest_pending rows."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import asyncio
6
+ import json
7
+ import logging
8
+ import mimetypes
9
+ import os
10
+ import sqlite3
11
+ from collections.abc import Awaitable, Callable
12
+ from dataclasses import dataclass, replace
13
+ from pathlib import Path
14
+ from typing import Any, TypeVar
15
+
16
+ from core.api.config import settings
17
+ from core.api.db import acquire_db, acquire_write_db
18
+ from core.api.services import pii_redactor
19
+ from core.api.services.ingest.auto_approve import decide_ingress_routing, should_auto_approve
20
+ from core.api.services.ingest.classifier import ALLOWED_TARGETS, classify_markdown
21
+ from core.api.services.ingest.confidence import (
22
+ compute_composite_confidence,
23
+ estimate_parser_quality,
24
+ )
25
+ from core.api.services.ingest.events import broadcast_ingest_changed
26
+ from core.api.services.ingest.parsers.docling_parser import parse_pdf_file
27
+ from core.api.services.ingest.parsers.docparse_gateway import parse_pdf_docparse
28
+ from core.api.services.ingest.parsers.docx_parser import DOCX_MIME_TYPE, parse_docx
29
+ from core.api.services.ingest.parsers.gateway_aux import MissingGatewayConfig
30
+ from core.api.services.ingest.parsers.image_parser import (
31
+ SUPPORTED_IMAGE_SUFFIXES,
32
+ UNSUPPORTED_PHASE1_SUFFIXES,
33
+ parse_image_with_gateway,
34
+ )
35
+ from core.api.services.ingest.parsers.internal_markdown import parse_markdown_file
36
+ from core.api.services.ingest.parsers.ocr_pdf_parser import parse_pdf_ocr
37
+ from core.api.services.ingest.parsers.transcript_parser import (
38
+ AUDIO_MIME_BY_SUFFIX,
39
+ SUPPORTED_AUDIO_SUFFIXES,
40
+ SUPPORTED_VIDEO_SUFFIXES,
41
+ VIDEO_MIME_BY_SUFFIX,
42
+ parse_media_transcript,
43
+ )
44
+ from core.api.services.ingest.parsers.xlsx_parser import parse_xlsx
45
+ from core.api.services.ingest.parsers.vision_gateway import parse_vision_with_gateway
46
+ from core.api.services.ingest.preflight import build_classifier_content, build_preflight
47
+ from core.api.services.ingest.routing_policy import IngestRoute, choose_route
48
+
49
+ logger = logging.getLogger(__name__)
50
+ T = TypeVar("T")
51
+
52
+ try:
53
+ import magic
54
+
55
+ _MAGIC = magic.Magic(mime=True)
56
+ except Exception: # pragma: no cover - python-magic/libmagic may be absent in old envs
57
+ _MAGIC = None
58
+ _MARKDOWN_SUFFIXES = {".md", ".markdown", ".txt"}
59
+ _PDF_SUFFIXES = {".pdf"}
60
+ # Inlined to avoid circular import with api.services.ingest.watcher (which imports
61
+ # parse_pending from this module). Same constant lives in watcher.py:28 + insert_saga.py:22.
62
+ PROJECTS_ROOT = Path("/data/projects")
63
+ _XLSX_MIME_BY_SUFFIX = {
64
+ ".xlsx": "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet",
65
+ ".xlsm": "application/vnd.ms-excel.sheet.macroEnabled.12",
66
+ }
67
+ _DOCX_SUFFIXES = {".docx"}
68
+ _IMAGE_MIME_BY_SUFFIX = {
69
+ ".avif": "image/avif",
70
+ ".heic": "image/heic",
71
+ ".heif": "image/heif",
72
+ ".jpeg": "image/jpeg",
73
+ ".jpg": "image/jpeg",
74
+ ".png": "image/png",
75
+ ".webp": "image/webp",
76
+ }
77
+ _MEDIA_SUFFIXES = SUPPORTED_AUDIO_SUFFIXES | SUPPORTED_VIDEO_SUFFIXES
78
+
79
+
80
+ def detect_mime(path: Path) -> str:
81
+ if path.suffix.lower() in _MARKDOWN_SUFFIXES:
82
+ return "text/markdown"
83
+ if path.suffix.lower() in _PDF_SUFFIXES:
84
+ return "application/pdf"
85
+ if path.suffix.lower() in _XLSX_MIME_BY_SUFFIX:
86
+ return _XLSX_MIME_BY_SUFFIX[path.suffix.lower()]
87
+ if path.suffix.lower() in _DOCX_SUFFIXES:
88
+ return DOCX_MIME_TYPE
89
+ if path.suffix.lower() in _IMAGE_MIME_BY_SUFFIX:
90
+ return _IMAGE_MIME_BY_SUFFIX[path.suffix.lower()]
91
+ if path.suffix.lower() in AUDIO_MIME_BY_SUFFIX:
92
+ return AUDIO_MIME_BY_SUFFIX[path.suffix.lower()]
93
+ if path.suffix.lower() in VIDEO_MIME_BY_SUFFIX:
94
+ return VIDEO_MIME_BY_SUFFIX[path.suffix.lower()]
95
+ if _MAGIC is not None:
96
+ try:
97
+ return str(_MAGIC.from_file(str(path)))
98
+ except Exception:
99
+ logger.exception("python-magic failed for %s", path)
100
+ guessed, _ = mimetypes.guess_type(str(path))
101
+ return guessed or "application/octet-stream"
102
+
103
+
104
+ def _classification_filename(path: Path, mime_type: str) -> str:
105
+ if mime_type == "application/pdf":
106
+ return f"{path.stem}.md"
107
+ if path.suffix.lower() in _MEDIA_SUFFIXES or mime_type.startswith(
108
+ ("audio/", "video/")
109
+ ):
110
+ return f"{path.stem}.md"
111
+ return path.name
112
+
113
+
114
+ def _is_image(path: Path, mime_type: str) -> bool:
115
+ suffix = path.suffix.lower()
116
+ return (
117
+ suffix in SUPPORTED_IMAGE_SUFFIXES
118
+ or suffix in UNSUPPORTED_PHASE1_SUFFIXES
119
+ or mime_type.startswith("image/")
120
+ )
121
+
122
+
123
+ def _is_xlsx(path: Path, mime_type: str) -> bool:
124
+ return path.suffix.lower() in _XLSX_MIME_BY_SUFFIX or mime_type in set(
125
+ _XLSX_MIME_BY_SUFFIX.values()
126
+ )
127
+
128
+
129
+ def _is_docx(path: Path, mime_type: str) -> bool:
130
+ return path.suffix.lower() in _DOCX_SUFFIXES or mime_type == DOCX_MIME_TYPE
131
+
132
+
133
+ def _is_transcript_media(path: Path, mime_type: str) -> bool:
134
+ return path.suffix.lower() in _MEDIA_SUFFIXES or mime_type.startswith(
135
+ ("audio/", "video/")
136
+ )
137
+
138
+
139
+ def _metadata_sidecar_path(path: Path) -> Path:
140
+ return path.with_suffix(".metadata.json")
141
+
142
+
143
+ def _image_classification(
144
+ *,
145
+ path: Path,
146
+ auto_approve: bool,
147
+ reason: str,
148
+ rules_matched: list[str],
149
+ ) -> dict[str, Any]:
150
+ return {
151
+ "type": "file",
152
+ "title": path.name,
153
+ "tags": ["image", path.suffix.lower().lstrip(".")],
154
+ "target_folder": "docs/assets",
155
+ "target_filename": path.name,
156
+ "confidence": 0.82 if auto_approve else 0.52,
157
+ "reason": reason,
158
+ "auto_approve": auto_approve,
159
+ "rules_matched": rules_matched,
160
+ }
161
+
162
+
163
+ def _xlsx_classification(path: Path) -> dict[str, Any]:
164
+ return {
165
+ "type": "file",
166
+ "title": path.name,
167
+ "tags": ["xlsx", "spreadsheet"],
168
+ "target_folder": "docs/assets",
169
+ "target_filename": path.name,
170
+ "confidence": 0.74,
171
+ "reason": "spreadsheet requires manual triage",
172
+ "auto_approve": False,
173
+ }
174
+
175
+
176
+ LLM_AUTO_APPROVE_THRESHOLD = 0.80
177
+ IDENTITY_AUTO_APPROVE_TERMS = (
178
+ "atto di nascita",
179
+ "carta d'ident",
180
+ "carta ident",
181
+ "codice fiscale",
182
+ "cognome e nome",
183
+ "documento d'ident",
184
+ "documento ident",
185
+ "fiscal code",
186
+ "name and surname",
187
+ "passport",
188
+ "passaporto",
189
+ )
190
+ _PARSER_LANE_LIMITS = {
191
+ "local": max(
192
+ 1,
193
+ int(
194
+ settings.ingest_local_parser_max_concurrency
195
+ or settings.ingest_parser_max_concurrency
196
+ ),
197
+ ),
198
+ "ocr": max(1, int(settings.ingest_ocr_max_concurrency)),
199
+ "docparse": max(1, int(settings.ingest_docparse_max_concurrency)),
200
+ "transcribe": max(1, int(settings.ingest_transcribe_max_concurrency)),
201
+ "vision": max(1, int(settings.ingest_vision_max_concurrency)),
202
+ }
203
+ _PARSER_LANE_SEMAPHORES = {
204
+ lane: asyncio.Semaphore(limit) for lane, limit in _PARSER_LANE_LIMITS.items()
205
+ }
206
+ _TRANSIENT_PARSE_ERROR_MARKERS = (
207
+ "rate limited",
208
+ "unavailable after",
209
+ "unavailable: http 408",
210
+ "unavailable: http 409",
211
+ "unavailable: http 425",
212
+ "unavailable: http 429",
213
+ "unavailable: http 500",
214
+ "unavailable: http 502",
215
+ "unavailable: http 503",
216
+ "unavailable: http 504",
217
+ )
218
+ _OCR_EMPTY_TEXT_MARKER = "tier-ocr returned empty text"
219
+
220
+
221
+ @dataclass(frozen=True)
222
+ class ParseDispatchResult:
223
+ parsed: Any
224
+ parser_used: str
225
+ route: IngestRoute
226
+ preflight: dict[str, Any]
227
+ parser_quality: dict[str, Any]
228
+
229
+
230
+ class _ParserWaitCancelled(Exception):
231
+ """Raised when a row leaves parser_waiting before the parser slot starts."""
232
+
233
+
234
+ def _parser_lane_for_workflow(workflow: str) -> str:
235
+ if workflow in {"ocr", "docparse", "transcribe", "vision"}:
236
+ return workflow
237
+ return "local"
238
+
239
+
240
+ def _parser_lane_waiter_count(lane: str) -> int:
241
+ waiters = getattr(_PARSER_LANE_SEMAPHORES[lane], "_waiters", None)
242
+ if waiters is None:
243
+ return 0
244
+ return sum(1 for waiter in waiters if not waiter.done())
245
+
246
+
247
+ async def _mark_parser_waiting(ingest_id: str, project_slug: str) -> None:
248
+ async with acquire_write_db() as db:
249
+ await db.execute(
250
+ """
251
+ UPDATE ingest_pending
252
+ SET status = 'parser_waiting',
253
+ error_message = NULL,
254
+ updated_at = datetime('now')
255
+ WHERE id = ?
256
+ AND status IN ('queued', 'parse_error', 'parsing')
257
+ """,
258
+ (ingest_id,),
259
+ )
260
+ await db.commit()
261
+ await broadcast_ingest_changed(
262
+ "parser_waiting",
263
+ ingest_id=ingest_id,
264
+ project_slug=project_slug,
265
+ status="parser_waiting",
266
+ )
267
+
268
+
269
+ async def _mark_parser_active(ingest_id: str, project_slug: str) -> None:
270
+ async with acquire_write_db() as db:
271
+ cursor = await db.execute(
272
+ """
273
+ UPDATE ingest_pending
274
+ SET status = 'parsing',
275
+ updated_at = datetime('now')
276
+ WHERE id = ?
277
+ AND status = 'parser_waiting'
278
+ """,
279
+ (ingest_id,),
280
+ )
281
+ await db.commit()
282
+ if cursor.rowcount != 1:
283
+ raise _ParserWaitCancelled(ingest_id)
284
+ await broadcast_ingest_changed(
285
+ "parsing",
286
+ ingest_id=ingest_id,
287
+ project_slug=project_slug,
288
+ status="parsing",
289
+ )
290
+
291
+
292
+ async def _mark_parse_error(
293
+ ingest_id: str,
294
+ project_slug: str,
295
+ message: str,
296
+ *,
297
+ attempts: int = 3,
298
+ ) -> None:
299
+ for attempt in range(1, attempts + 1):
300
+ try:
301
+ async with acquire_write_db(label="ingest.parse_error") as db:
302
+ await db.execute(
303
+ """
304
+ UPDATE ingest_pending
305
+ SET status = 'parse_error',
306
+ error_message = ?,
307
+ updated_at = datetime('now')
308
+ WHERE id = ?
309
+ """,
310
+ (message[:1000], ingest_id),
311
+ )
312
+ await db.commit()
313
+ await broadcast_ingest_changed(
314
+ "parse_error",
315
+ ingest_id=ingest_id,
316
+ project_slug=project_slug,
317
+ status="parse_error",
318
+ )
319
+ return
320
+ except (sqlite3.OperationalError, RuntimeError) as exc:
321
+ if attempt >= attempts:
322
+ logger.exception(
323
+ "ingest parse_error write failed permanently: id=%s",
324
+ ingest_id,
325
+ )
326
+ return
327
+ delay = min(0.2 * (2 ** (attempt - 1)), 1.0)
328
+ logger.warning(
329
+ "ingest parse_error write failed; retrying attempt=%d/%d id=%s error=%s",
330
+ attempt,
331
+ attempts,
332
+ ingest_id,
333
+ exc,
334
+ )
335
+ await asyncio.sleep(delay)
336
+
337
+
338
+ async def _run_heavy_parser(
339
+ *,
340
+ ingest_id: str,
341
+ project_slug: str,
342
+ lane: str,
343
+ parser_name: str,
344
+ path: Path,
345
+ parse: Callable[[], Awaitable[T]],
346
+ ) -> T:
347
+ semaphore = _PARSER_LANE_SEMAPHORES[lane]
348
+ limit = _PARSER_LANE_LIMITS[lane]
349
+ waiters = _parser_lane_waiter_count(lane)
350
+ if semaphore.locked() or waiters:
351
+ logger.info(
352
+ "ingest parser waiting: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
353
+ lane,
354
+ parser_name,
355
+ path,
356
+ limit,
357
+ waiters,
358
+ )
359
+
360
+ async with semaphore:
361
+ await _mark_parser_active(ingest_id, project_slug)
362
+ logger.info(
363
+ "ingest parser acquired: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
364
+ lane,
365
+ parser_name,
366
+ path,
367
+ limit,
368
+ _parser_lane_waiter_count(lane),
369
+ )
370
+ try:
371
+ return await parse()
372
+ finally:
373
+ logger.info(
374
+ "ingest parser released: lane=%s parser=%s path=%s max_concurrency=%d waiters=%d",
375
+ lane,
376
+ parser_name,
377
+ path,
378
+ limit,
379
+ _parser_lane_waiter_count(lane),
380
+ )
381
+
382
+
383
+ def _llm_classifier_enabled() -> bool:
384
+ """Read LLM_CLASSIFIER_ENABLED env (true|shadow|false). Default: false."""
385
+ value = (os.environ.get("LLM_CLASSIFIER_ENABLED", "false") or "").strip().lower()
386
+ return value == "true"
387
+
388
+
389
+ def _llm_classifier_shadow() -> bool:
390
+ value = (os.environ.get("LLM_CLASSIFIER_ENABLED", "false") or "").strip().lower()
391
+ return value == "shadow"
392
+
393
+
394
+ def _ingest_llm_provider() -> str:
395
+ return settings.ingest_llm_provider.strip().lower()
396
+
397
+
398
+ def _ingest_llm_model() -> str:
399
+ return settings.ingest_llm_classifier_model
400
+
401
+
402
+ async def _resolve_classify_provider():
403
+ """Resolve the BYOK 'classify' provider from llm_function_config, or None.
404
+
405
+ Reads on the read pool, fail-soft (None on any error) so the deterministic
406
+ path is never broken by a config/DB hiccup.
407
+ """
408
+ try:
409
+ from core.api.services.ingest.llm.config_store import resolve_function_provider
410
+
411
+ async with acquire_db() as cfg_db:
412
+ return await resolve_function_provider(cfg_db, "classify")
413
+ except Exception: # noqa: BLE001
414
+ logger.debug("byok classify provider resolution failed", exc_info=True)
415
+ return None
416
+
417
+
418
+ def _with_llm_no_result(
419
+ base_classification: dict[str, Any],
420
+ *,
421
+ status: str,
422
+ reason: str,
423
+ extra: dict[str, Any] | None = None,
424
+ ) -> dict[str, Any]:
425
+ merged = dict(base_classification)
426
+ metadata = dict(merged.get("llm_metadata") or {})
427
+ metadata.update(
428
+ {
429
+ "status": status,
430
+ "model": _ingest_llm_model(),
431
+ "provider": _ingest_llm_provider(),
432
+ "auto_approved": False,
433
+ "reason": reason,
434
+ }
435
+ )
436
+ if extra:
437
+ metadata.update(extra)
438
+ merged["llm_metadata"] = metadata
439
+ return merged
440
+
441
+
442
+ _LLM_HARD_FAILURE_STATUSES = {
443
+ "api_error",
444
+ "bad_response",
445
+ "client_init_failed",
446
+ "exception",
447
+ "factory_failed",
448
+ "json_parse_failed",
449
+ "no_result",
450
+ "unavailable",
451
+ }
452
+
453
+
454
+ def _llm_failure_error_message(classification_json: dict[str, Any]) -> str | None:
455
+ metadata = classification_json.get("llm_metadata")
456
+ if not isinstance(metadata, dict):
457
+ return None
458
+ status = str(metadata.get("status") or "")
459
+ if status not in _LLM_HARD_FAILURE_STATUSES:
460
+ return None
461
+ reason = str(metadata.get("reason") or "llm_classifier_failed")
462
+ message = str(
463
+ metadata.get("gateway_error_message")
464
+ or metadata.get("error_message")
465
+ or reason
466
+ )
467
+ return f"E5 LLM enrichment failed after retries: {status} ({message})"[:1000]
468
+
469
+
470
+ async def _maybe_summarize_transcript(
471
+ *,
472
+ ingest_id: str,
473
+ parser_used: str,
474
+ extracted_text: str,
475
+ structure: dict[str, Any],
476
+ ) -> dict[str, Any] | None:
477
+ if parser_used != "tier_transcribe" or not extracted_text.strip():
478
+ return None
479
+ if not _llm_classifier_enabled():
480
+ return None
481
+
482
+ try:
483
+ from core.api.services.ingest.llm.local_gateway import (
484
+ summarize_transcript_with_local_gateway,
485
+ )
486
+ except Exception: # noqa: BLE001
487
+ logger.debug("transcript summarizer import failed", exc_info=True)
488
+ return {
489
+ "status": "import_failed",
490
+ "reason": "transcript_summarizer_import_failed",
491
+ }
492
+
493
+ try:
494
+ result, diagnostics = await summarize_transcript_with_local_gateway(
495
+ extracted_text,
496
+ structure=structure,
497
+ idempotency_scope=f"ingest:{ingest_id}",
498
+ )
499
+ except Exception: # noqa: BLE001 - summary must never break ingest
500
+ logger.warning("transcript summarizer raised", exc_info=True)
501
+ return {
502
+ "status": "exception",
503
+ "reason": "transcript_summarizer_exception",
504
+ }
505
+
506
+ if result is None:
507
+ return _transcript_summary_failure_metadata(diagnostics)
508
+
509
+ return {
510
+ "status": "ok",
511
+ "model": _ingest_llm_model(),
512
+ "provider": _ingest_llm_provider(),
513
+ "summary": result.summary,
514
+ "topics": list(result.topics),
515
+ "participants": list(result.participants),
516
+ "keywords": list(result.keywords),
517
+ "action_items": list(result.action_items),
518
+ "confidence": result.confidence,
519
+ }
520
+
521
+
522
+ def _transcript_summary_failure_metadata(
523
+ diagnostics: dict[str, Any] | None,
524
+ ) -> dict[str, Any]:
525
+ metadata = {
526
+ "status": "no_result",
527
+ "model": _ingest_llm_model(),
528
+ "provider": _ingest_llm_provider(),
529
+ "reason": "transcript_summarizer_returned_none",
530
+ }
531
+ if isinstance(diagnostics, dict):
532
+ for key in (
533
+ "status",
534
+ "reason",
535
+ "gateway_status_code",
536
+ "gateway_error_code",
537
+ "gateway_error_message",
538
+ "schema_retry_attempted",
539
+ "raw_excerpt",
540
+ "first_raw_excerpt",
541
+ ):
542
+ if key in diagnostics:
543
+ metadata[key] = diagnostics[key]
544
+ return metadata
545
+
546
+
547
+ def _ingest_event_for_status(status: str) -> str:
548
+ if status == "done":
549
+ return "done"
550
+ if status == "rejected":
551
+ return "rejected"
552
+ if status == "parse_error":
553
+ return "parse_error"
554
+ return "parsed"
555
+
556
+
557
+ async def _maybe_llm_enrich(
558
+ *,
559
+ ingest_id: str | None = None,
560
+ extracted_text: str,
561
+ base_classification: dict[str, Any],
562
+ preflight: dict[str, Any],
563
+ parser_quality: dict[str, Any],
564
+ route: IngestRoute,
565
+ ) -> dict[str, Any] | None:
566
+ """Run the local/cloud LLM classifier on a bounded evidence packet.
567
+
568
+ Returns the enriched classification dict (with auto_approve, llm_metadata,
569
+ suggested project_slug) only when the project_slug resolves to a valid
570
+ project and the composite confidence gate passes.
571
+
572
+ Reads context OUTSIDE the writer lock (M-D7).
573
+ """
574
+ enabled = _llm_classifier_enabled()
575
+ shadow = _llm_classifier_shadow()
576
+ # BYOK (U4): a DB-configured provider for the 'classify' function also enables
577
+ # auto-classify (no env flag needed). The deterministic classifier stays
578
+ # PRIMARY; this only gates the optional LLM override.
579
+ resolved_provider = await _resolve_classify_provider()
580
+ byok = resolved_provider is not None
581
+ if not (enabled or shadow or byok):
582
+ # No env flag and no configured BYOK provider: auto-classify is disabled.
583
+ # The deterministic classification (already computed) remains and the item
584
+ # routes to triage — no heuristic semantic guess (R10/D6).
585
+ return None
586
+
587
+ try:
588
+ from core.api.services.ingest.llm.classification_context import (
589
+ gather_classification_context,
590
+ )
591
+ from core.api.services.ingest.llm.factory import get_classifier
592
+ except Exception: # noqa: BLE001 - module unavailable in some test envs
593
+ logger.debug("llm classifier import failed", exc_info=True)
594
+ return None
595
+
596
+ classifier_content = build_classifier_content(
597
+ extracted_text=extracted_text,
598
+ preflight=preflight,
599
+ parser_quality=parser_quality,
600
+ )
601
+
602
+ try:
603
+ async with acquire_db() as read_db:
604
+ ctx = await gather_classification_context(classifier_content, read_db)
605
+ except Exception: # noqa: BLE001
606
+ logger.warning("llm context gather failed", exc_info=True)
607
+ ctx = {"projects": [], "similar_artifacts": [], "hotspots": []}
608
+ if preflight.get("source_context"):
609
+ ctx["source_context"] = preflight["source_context"]
610
+ if ingest_id:
611
+ ctx["_idempotency_scope"] = f"ingest:{ingest_id}"
612
+
613
+ classifier = None
614
+ if byok:
615
+ try:
616
+ from core.api.services.ingest.llm.byok_provider import build_classifier
617
+
618
+ classifier = build_classifier(resolved_provider)
619
+ except Exception: # noqa: BLE001
620
+ logger.warning("byok classifier build failed", exc_info=True)
621
+ classifier = None
622
+ if classifier is None and (enabled or shadow):
623
+ # Legacy env-driven provider (OSS first-boot fallback).
624
+ try:
625
+ classifier = get_classifier()
626
+ except Exception: # noqa: BLE001
627
+ logger.warning("llm classifier factory failed", exc_info=True)
628
+ classifier = None
629
+ if classifier is None:
630
+ # Gate produced no usable provider: disabled, surfaced, never heuristic.
631
+ return _with_llm_no_result(
632
+ base_classification,
633
+ status="disabled_no_provider",
634
+ reason="no_llm_provider_configured",
635
+ )
636
+
637
+ try:
638
+ llm_result = await classifier.classify(classifier_content, ctx)
639
+ except Exception: # noqa: BLE001 - defensive: classify() should never raise
640
+ logger.warning("llm classify raised", exc_info=True)
641
+ return _with_llm_no_result(
642
+ base_classification,
643
+ status="exception",
644
+ reason="llm_classifier_exception",
645
+ )
646
+
647
+ if llm_result is None:
648
+ diagnostics = getattr(classifier, "last_error", None)
649
+ if isinstance(diagnostics, dict):
650
+ return _with_llm_no_result(
651
+ base_classification,
652
+ status=str(diagnostics.get("status") or "no_result"),
653
+ reason=str(diagnostics.get("reason") or "llm_classifier_returned_none"),
654
+ extra=diagnostics,
655
+ )
656
+ return _with_llm_no_result(
657
+ base_classification,
658
+ status="no_result",
659
+ reason="llm_classifier_returned_none",
660
+ )
661
+
662
+ # Validate project_slug actually exists on disk (avoid hallucinated slugs).
663
+ valid_slug: str | None = None
664
+ try:
665
+ from core.api.services.ingest.insert_saga import _load_project_entry
666
+
667
+ ptype, _repo = _load_project_entry(llm_result.project_slug)
668
+ if ptype:
669
+ valid_slug = llm_result.project_slug
670
+ except Exception: # noqa: BLE001 - any failure -> reject
671
+ valid_slug = None
672
+
673
+ provider = _ingest_llm_provider()
674
+ model = _ingest_llm_model()
675
+ target_folder = ALLOWED_TARGETS.get(llm_result.document_type)
676
+ confidence_decision = compute_composite_confidence(
677
+ route=route,
678
+ parser_quality=parser_quality,
679
+ llm_confidence=float(llm_result.confidence),
680
+ valid_project=valid_slug is not None,
681
+ document_type=llm_result.document_type,
682
+ extracted_text=extracted_text,
683
+ )
684
+ metadata = {
685
+ "model": model,
686
+ "provider": provider,
687
+ "project_slug": llm_result.project_slug,
688
+ "valid_slug": valid_slug,
689
+ "document_type": llm_result.document_type,
690
+ "title": llm_result.title,
691
+ "tags": llm_result.tags,
692
+ "llm_confidence": llm_result.confidence,
693
+ "composite_confidence": confidence_decision.score,
694
+ "confidence_gate": confidence_decision.as_json(),
695
+ "reasoning": llm_result.reasoning,
696
+ "pii_detected": bool(pii_redactor.analyze(extracted_text[:6000])),
697
+ }
698
+ source_context = preflight.get("source_context") or {}
699
+ source_project_slug = source_context.get("project_slug")
700
+ if source_project_slug:
701
+ metadata.update(
702
+ {
703
+ "source_project_slug": source_project_slug,
704
+ "source_project_prior": source_context.get("prior"),
705
+ "source_project_reason": source_context.get("reason"),
706
+ "source_project_followed": llm_result.project_slug == source_project_slug,
707
+ "source_project_overridden": llm_result.project_slug != source_project_slug,
708
+ }
709
+ )
710
+
711
+ # Shadow mode: log decision but never override the deterministic classifier.
712
+ if shadow:
713
+ shadow_blob = {
714
+ "shadow_mode": True,
715
+ "llm_metadata": metadata,
716
+ }
717
+ merged = dict(base_classification)
718
+ merged.setdefault("llm_shadow", shadow_blob)
719
+ return merged
720
+
721
+ if valid_slug is None:
722
+ logger.info(
723
+ "llm classifier returned unknown slug=%s; falling back to deterministic",
724
+ llm_result.project_slug,
725
+ )
726
+ return _with_llm_no_result(
727
+ base_classification,
728
+ status="invalid_project",
729
+ reason="llm_classifier_invalid_project_slug",
730
+ extra={
731
+ "project_slug": llm_result.project_slug,
732
+ "document_type": llm_result.document_type,
733
+ "llm_confidence": llm_result.confidence,
734
+ "reasoning": llm_result.reasoning,
735
+ },
736
+ )
737
+
738
+ if target_folder is None:
739
+ return _with_llm_no_result(
740
+ base_classification,
741
+ status="invalid_document_type",
742
+ reason="llm_classifier_invalid_document_type",
743
+ extra={
744
+ "project_slug": llm_result.project_slug,
745
+ "document_type": llm_result.document_type,
746
+ "llm_confidence": llm_result.confidence,
747
+ "reasoning": llm_result.reasoning,
748
+ },
749
+ )
750
+
751
+ privacy_block_reason = _llm_auto_approve_privacy_block_reason(
752
+ preflight=preflight,
753
+ extracted_text=extracted_text,
754
+ document_type=llm_result.document_type,
755
+ )
756
+ if privacy_block_reason:
757
+ merged = dict(base_classification)
758
+ merged["llm_metadata"] = {
759
+ **metadata,
760
+ "auto_approved": False,
761
+ "auto_approve_blocked_reason": privacy_block_reason,
762
+ }
763
+ return merged
764
+
765
+ if (
766
+ llm_result.confidence < LLM_AUTO_APPROVE_THRESHOLD
767
+ or not confidence_decision.auto_approve
768
+ ):
769
+ # Keep deterministic decision but surface the LLM hint so the human
770
+ # triage UI can show it.
771
+ merged = dict(base_classification)
772
+ merged["llm_metadata"] = {**metadata, "auto_approved": False}
773
+ return merged
774
+
775
+ enriched = dict(base_classification)
776
+ enriched.update(
777
+ {
778
+ "type": llm_result.document_type,
779
+ "title": llm_result.title,
780
+ "tags": list(llm_result.tags),
781
+ "target_folder": target_folder,
782
+ "confidence": confidence_decision.score,
783
+ "reason": "llm_routing",
784
+ "auto_approve": True,
785
+ "suggested_project_slug": valid_slug,
786
+ "llm_metadata": {
787
+ **metadata,
788
+ "project_slug": valid_slug,
789
+ "auto_approved": True,
790
+ },
791
+ }
792
+ )
793
+ return enriched
794
+
795
+
796
+ def _llm_auto_approve_privacy_block_reason(
797
+ *,
798
+ preflight: dict[str, Any],
799
+ extracted_text: str,
800
+ document_type: str | None = None,
801
+ ) -> str | None:
802
+ packet = preflight or {}
803
+ pf = packet.get("preflight") or {}
804
+ file_info = packet.get("file") or {}
805
+ filename_text = " ".join(
806
+ str(file_info.get(key) or "") for key in ("filename", "stem")
807
+ )
808
+ if _contains_identity_signal(filename_text, minimum=1):
809
+ return "identity_document_requires_manual_triage"
810
+ if bool(pf.get("identity_hint")) and _contains_identity_signal(
811
+ extracted_text, minimum=1
812
+ ):
813
+ return "identity_document_requires_manual_triage"
814
+ if _contains_identity_signal(extracted_text, minimum=2):
815
+ return "identity_document_requires_manual_triage"
816
+
817
+ # E5 already redacts PII before sending the prompt to the local Gateway.
818
+ # Generic business documents regularly contain email, phone, IBAN, VAT or
819
+ # fiscal-code-like strings; those should not block auto-triage by
820
+ # themselves. Keep the conservative fallback only when the LLM did not
821
+ # return a document type.
822
+ if document_type == "record" and pii_redactor.analyze(extracted_text[:6000]):
823
+ return "sensitive_record_requires_manual_triage"
824
+ if document_type is None and pii_redactor.analyze(extracted_text[:6000]):
825
+ return "pii_requires_manual_triage"
826
+ return None
827
+
828
+
829
+ def _contains_identity_signal(text: str, *, minimum: int = 1) -> bool:
830
+ lowered = (text or "").lower()
831
+ matches = sum(1 for term in IDENTITY_AUTO_APPROVE_TERMS if term in lowered)
832
+ return matches >= minimum
833
+
834
+
835
+ def _gateway_aux_configured() -> bool:
836
+ key = settings.ingest_llm_gateway_api_key
837
+ key_value = (
838
+ key.get_secret_value() if hasattr(key, "get_secret_value") else str(key or "")
839
+ )
840
+ return bool(
841
+ settings.pir_env != "test"
842
+ and key_value
843
+ and (settings.llm_gateway_aux_base_url or settings.llm_gateway_base_url)
844
+ )
845
+
846
+
847
+ def _route_for(path: Path, mime_type: str, preflight: dict[str, Any]) -> IngestRoute:
848
+ gateway_enabled = _gateway_aux_configured()
849
+ return choose_route(
850
+ path=path,
851
+ mime_type=mime_type,
852
+ preflight=preflight,
853
+ docparse_enabled=(
854
+ gateway_enabled
855
+ and settings.ingest_docparse_enabled
856
+ and (
857
+ settings.ingest_docparse_pdfs_enabled
858
+ if mime_type == "application/pdf"
859
+ else settings.ingest_docparse_images_enabled
860
+ )
861
+ ),
862
+ ocr_enabled=gateway_enabled,
863
+ vision_enabled=bool(gateway_enabled and settings.ingest_vision_images_enabled),
864
+ mode_override=settings.ingest_docparse_mode_override or None,
865
+ )
866
+
867
+
868
+ async def _maybe_llm_route(
869
+ *,
870
+ route: IngestRoute,
871
+ preflight: dict[str, Any],
872
+ mime_type: str,
873
+ ) -> IngestRoute:
874
+ if route.confidence >= 0.70 or not _llm_classifier_enabled():
875
+ return route
876
+ try:
877
+ from core.api.services.ingest.llm.local_gateway import (
878
+ classify_route_with_local_gateway,
879
+ )
880
+ except Exception: # noqa: BLE001
881
+ logger.debug("local route classifier import failed", exc_info=True)
882
+ return route
883
+
884
+ decision = await classify_route_with_local_gateway(
885
+ preflight=preflight,
886
+ deterministic_route=route.as_json(),
887
+ )
888
+ if decision is None:
889
+ return route
890
+ if decision.workflow not in _allowed_workflows_for_mime(mime_type):
891
+ logger.info(
892
+ "route classifier ignored invalid workflow=%s for mime=%s",
893
+ decision.workflow,
894
+ mime_type,
895
+ )
896
+ return route
897
+ return replace(
898
+ route,
899
+ workflow=decision.workflow,
900
+ tier=_tier_for_workflow(decision.workflow),
901
+ mode=decision.mode if decision.workflow == "docparse" else None,
902
+ reason=f"tier-fast route classifier: {decision.reason}",
903
+ confidence=float(decision.confidence),
904
+ features_used=[*route.features_used, "tier_fast_route_classifier"],
905
+ )
906
+
907
+
908
+ def _allowed_workflows_for_mime(mime_type: str) -> set[str]:
909
+ if mime_type.startswith(("audio/", "video/")):
910
+ return {"transcribe"}
911
+ if mime_type == "application/pdf":
912
+ return {"local", "ocr", "docparse"}
913
+ if mime_type.startswith("image/"):
914
+ return {"ocr", "docparse", "vision"}
915
+ return {"local"}
916
+
917
+
918
+ def _tier_for_workflow(workflow: str) -> str | None:
919
+ return {
920
+ "ocr": "tier-ocr",
921
+ "docparse": "tier-docparse",
922
+ "transcribe": "tier-transcribe",
923
+ "vision": "tier-vision",
924
+ }.get(workflow)
925
+
926
+
927
+ def _block_llm_auto_approve(
928
+ classification_json: dict[str, Any],
929
+ *,
930
+ reason: str,
931
+ existing_ingest_id: str | None = None,
932
+ ) -> dict[str, Any]:
933
+ blocked = dict(classification_json)
934
+ blocked["auto_approve"] = False
935
+ blocked["reason"] = reason
936
+ metadata = dict(blocked.get("llm_metadata") or {})
937
+ metadata["auto_approved"] = False
938
+ metadata["auto_approve_blocked_reason"] = reason
939
+ if existing_ingest_id:
940
+ metadata["existing_ingest_id"] = existing_ingest_id
941
+ blocked["llm_metadata"] = metadata
942
+ return blocked
943
+
944
+
945
+ def _reject_llm_auto_approve_duplicate(
946
+ classification_json: dict[str, Any],
947
+ *,
948
+ reason: str,
949
+ existing_ingest_id: str,
950
+ ) -> dict[str, Any]:
951
+ rejected = _block_llm_auto_approve(
952
+ classification_json,
953
+ reason=reason,
954
+ existing_ingest_id=existing_ingest_id,
955
+ )
956
+ rejected["auto_reject"] = True
957
+ metadata = dict(rejected.get("llm_metadata") or {})
958
+ metadata["auto_rejected"] = True
959
+ metadata["auto_reject_reason"] = reason
960
+ metadata["existing_ingest_id"] = existing_ingest_id
961
+ rejected["llm_metadata"] = metadata
962
+ return rejected
963
+
964
+
965
+ async def _find_ingest_duplicate_for_project(
966
+ db: Any,
967
+ *,
968
+ ingest_id: str,
969
+ sha256: str | None,
970
+ project_slug: str,
971
+ ) -> str | None:
972
+ if not sha256:
973
+ return None
974
+ async with db.execute(
975
+ """
976
+ SELECT id
977
+ FROM ingest_pending
978
+ WHERE sha256 = ?
979
+ AND project_slug = ?
980
+ AND id != ?
981
+ LIMIT 1
982
+ """,
983
+ (sha256, project_slug, ingest_id),
984
+ ) as cursor:
985
+ row = await cursor.fetchone()
986
+ return str(row["id"]) if row is not None else None
987
+
988
+
989
+ async def _apply_llm_project_switch_if_safe(
990
+ db: Any,
991
+ *,
992
+ ingest_id: str,
993
+ sha256: str | None,
994
+ current_project_slug: str,
995
+ target_project_slug: str,
996
+ path: Path,
997
+ classification_json: dict[str, Any],
998
+ ) -> tuple[str, Path, dict[str, Any], str, str | None]:
999
+ new_root = PROJECTS_ROOT / target_project_slug
1000
+ if not new_root.is_dir():
1001
+ logger.warning(
1002
+ "llm_routing project switch aborted: %s not a project root",
1003
+ new_root,
1004
+ )
1005
+ return (
1006
+ current_project_slug,
1007
+ path,
1008
+ _block_llm_auto_approve(
1009
+ classification_json,
1010
+ reason="llm_routing_project_switch_invalid_project",
1011
+ ),
1012
+ "awaiting_triage",
1013
+ None,
1014
+ )
1015
+
1016
+ duplicate_id = await _find_ingest_duplicate_for_project(
1017
+ db,
1018
+ ingest_id=ingest_id,
1019
+ sha256=sha256,
1020
+ project_slug=target_project_slug,
1021
+ )
1022
+ if duplicate_id is not None:
1023
+ logger.warning(
1024
+ "llm_routing project switch aborted: sha256 already exists in %s as %s",
1025
+ target_project_slug,
1026
+ duplicate_id,
1027
+ )
1028
+ return (
1029
+ current_project_slug,
1030
+ path,
1031
+ _reject_llm_auto_approve_duplicate(
1032
+ classification_json,
1033
+ reason="llm_routing_project_switch_dedup_collision",
1034
+ existing_ingest_id=duplicate_id,
1035
+ ),
1036
+ "rejected",
1037
+ "auto_reject:llm_routing_duplicate",
1038
+ )
1039
+
1040
+ new_input = new_root / "input"
1041
+ new_input.mkdir(parents=True, exist_ok=True)
1042
+ new_source = new_input / path.name
1043
+ source_sidecar = _metadata_sidecar_path(path)
1044
+ target_sidecar = _metadata_sidecar_path(new_source)
1045
+ if new_source.exists():
1046
+ logger.warning(
1047
+ "llm_routing project switch aborted: %s already exists",
1048
+ new_source,
1049
+ )
1050
+ return (
1051
+ current_project_slug,
1052
+ path,
1053
+ _block_llm_auto_approve(
1054
+ classification_json,
1055
+ reason="llm_routing_project_switch_path_collision",
1056
+ ),
1057
+ "awaiting_triage",
1058
+ None,
1059
+ )
1060
+ if source_sidecar.exists() and target_sidecar.exists():
1061
+ logger.warning(
1062
+ "llm_routing project switch aborted: %s already exists",
1063
+ target_sidecar,
1064
+ )
1065
+ return (
1066
+ current_project_slug,
1067
+ path,
1068
+ _block_llm_auto_approve(
1069
+ classification_json,
1070
+ reason="llm_routing_project_switch_sidecar_collision",
1071
+ ),
1072
+ "awaiting_triage",
1073
+ None,
1074
+ )
1075
+
1076
+ path.replace(new_source)
1077
+ if source_sidecar.exists():
1078
+ source_sidecar.replace(target_sidecar)
1079
+ return (
1080
+ target_project_slug,
1081
+ new_source,
1082
+ classification_json,
1083
+ "approved",
1084
+ "auto_approve:llm_routing",
1085
+ )
1086
+
1087
+
1088
+ async def _parse_pdf_local(path: Path):
1089
+ try:
1090
+ return await parse_pdf_file(path, allow_docparse=False)
1091
+ except TypeError as exc:
1092
+ if "allow_docparse" not in str(exc):
1093
+ raise
1094
+ return await parse_pdf_file(path)
1095
+
1096
+
1097
+ def _transient_parse_max_attempts() -> int:
1098
+ raw = os.environ.get("INGEST_TRANSIENT_PARSE_MAX_ATTEMPTS", "3")
1099
+ try:
1100
+ return max(1, int(raw))
1101
+ except ValueError:
1102
+ return 3
1103
+
1104
+
1105
+ def _transient_parse_delay_seconds(attempt: int) -> float:
1106
+ if settings.pir_env == "test":
1107
+ return 0.0
1108
+ raw = os.environ.get("INGEST_TRANSIENT_PARSE_RETRY_BASE_SECONDS", "5")
1109
+ try:
1110
+ base = max(0.0, float(raw))
1111
+ except ValueError:
1112
+ base = 5.0
1113
+ return min(base * attempt, 30.0)
1114
+
1115
+
1116
+ def _is_transient_parse_error(exc: BaseException) -> bool:
1117
+ message = str(exc).lower()
1118
+ return any(marker in message for marker in _TRANSIENT_PARSE_ERROR_MARKERS)
1119
+
1120
+
1121
+ def _is_empty_ocr_result(exc: BaseException) -> bool:
1122
+ return _OCR_EMPTY_TEXT_MARKER in str(exc).lower()
1123
+
1124
+
1125
+ def _docparse_enabled_for_pdf() -> bool:
1126
+ return bool(
1127
+ _gateway_aux_configured()
1128
+ and settings.ingest_docparse_enabled
1129
+ and settings.ingest_docparse_pdfs_enabled
1130
+ )
1131
+
1132
+
1133
+ def _docparse_fallback_mode(route: IngestRoute) -> str:
1134
+ return (
1135
+ route.mode
1136
+ or settings.ingest_docparse_mode_override
1137
+ or settings.ingest_docparse_mode
1138
+ )
1139
+
1140
+
1141
+ def _ocr_to_docparse_route(route: IngestRoute) -> IngestRoute:
1142
+ return replace(
1143
+ route,
1144
+ workflow="docparse",
1145
+ tier="tier-docparse",
1146
+ mode=_docparse_fallback_mode(route),
1147
+ reason=f"{route.reason}; tier-ocr empty result fell back to docparse",
1148
+ confidence=max(route.confidence, 0.86),
1149
+ features_used=[*route.features_used, "ocr_empty_docparse_fallback"],
1150
+ )
1151
+
1152
+
1153
+ async def _parse_file(
1154
+ ingest_id: str,
1155
+ project_slug: str,
1156
+ path: Path,
1157
+ mime_type: str,
1158
+ *,
1159
+ preflight: dict[str, Any],
1160
+ route: IngestRoute,
1161
+ ) -> ParseDispatchResult:
1162
+ if route.workflow == "skip":
1163
+ raise ValueError(route.reason)
1164
+
1165
+ if mime_type == "application/pdf":
1166
+ effective_route = route
1167
+
1168
+ async def parse_pdf_route():
1169
+ nonlocal effective_route
1170
+ if route.workflow == "docparse":
1171
+ try:
1172
+ return await parse_pdf_docparse(path, mode=route.mode)
1173
+ except MissingGatewayConfig:
1174
+ logger.warning(
1175
+ "tier-docparse not configured; falling back to local PDF"
1176
+ )
1177
+ return await _parse_pdf_local(path)
1178
+ if route.workflow == "ocr":
1179
+ try:
1180
+ return await parse_pdf_ocr(path)
1181
+ except MissingGatewayConfig:
1182
+ logger.warning("tier-ocr not configured; falling back to local PDF")
1183
+ return await _parse_pdf_local(path)
1184
+ except RuntimeError as exc:
1185
+ if not _is_empty_ocr_result(exc) or not _docparse_enabled_for_pdf():
1186
+ raise
1187
+ effective_route = _ocr_to_docparse_route(route)
1188
+ logger.warning(
1189
+ "tier-ocr returned empty text; falling back to tier-docparse: path=%s mode=%s",
1190
+ path,
1191
+ effective_route.mode,
1192
+ )
1193
+ return await parse_pdf_docparse(path, mode=effective_route.mode)
1194
+ return await _parse_pdf_local(path)
1195
+
1196
+ parsed = await _run_heavy_parser(
1197
+ ingest_id=ingest_id,
1198
+ project_slug=project_slug,
1199
+ lane=_parser_lane_for_workflow(route.workflow),
1200
+ parser_name=f"pdf:{route.workflow}",
1201
+ path=path,
1202
+ parse=parse_pdf_route,
1203
+ )
1204
+ parser_quality = estimate_parser_quality(
1205
+ parser_used=parsed.parser_used,
1206
+ extracted_text=parsed.text,
1207
+ structure=parsed.structure,
1208
+ )
1209
+ return ParseDispatchResult(
1210
+ parsed,
1211
+ parsed.parser_used,
1212
+ effective_route,
1213
+ preflight,
1214
+ parser_quality,
1215
+ )
1216
+
1217
+ if _is_image(path, mime_type):
1218
+ if route.workflow == "vision":
1219
+ parsed = await _run_heavy_parser(
1220
+ ingest_id=ingest_id,
1221
+ project_slug=project_slug,
1222
+ lane="vision",
1223
+ parser_name="image:vision",
1224
+ path=path,
1225
+ parse=lambda: parse_vision_with_gateway(path, mime_type),
1226
+ )
1227
+ parser_quality = estimate_parser_quality(
1228
+ parser_used=str(parsed["parser_used"]),
1229
+ extracted_text=str(
1230
+ parsed.get("text") or parsed.get("extracted_text") or ""
1231
+ ),
1232
+ structure=parsed.get("structure") or {},
1233
+ )
1234
+ return ParseDispatchResult(
1235
+ parsed,
1236
+ str(parsed["parser_used"]),
1237
+ route,
1238
+ preflight,
1239
+ parser_quality,
1240
+ )
1241
+
1242
+ prefer_docparse = route.workflow == "docparse"
1243
+ parsed = await _run_heavy_parser(
1244
+ ingest_id=ingest_id,
1245
+ project_slug=project_slug,
1246
+ lane=_parser_lane_for_workflow(route.workflow),
1247
+ parser_name=f"image:{route.workflow}",
1248
+ path=path,
1249
+ parse=lambda: parse_image_with_gateway(
1250
+ path,
1251
+ mime_type,
1252
+ prefer_docparse=prefer_docparse,
1253
+ docparse_mode=route.mode,
1254
+ ),
1255
+ )
1256
+ parser_quality = estimate_parser_quality(
1257
+ parser_used=str(parsed["parser_used"]),
1258
+ extracted_text=str(
1259
+ parsed.get("text") or parsed.get("extracted_text") or ""
1260
+ ),
1261
+ structure=parsed.get("structure") or {},
1262
+ )
1263
+ return ParseDispatchResult(
1264
+ parsed,
1265
+ str(parsed["parser_used"]),
1266
+ route,
1267
+ preflight,
1268
+ parser_quality,
1269
+ )
1270
+ if _is_xlsx(path, mime_type):
1271
+ await _mark_parser_active(ingest_id, project_slug)
1272
+ parsed = await asyncio.to_thread(parse_xlsx, path)
1273
+ parser_quality = estimate_parser_quality(
1274
+ parser_used=str(parsed["parser_used"]),
1275
+ extracted_text=str(parsed.get("text") or ""),
1276
+ structure=parsed.get("structure") or {},
1277
+ )
1278
+ return ParseDispatchResult(
1279
+ parsed,
1280
+ str(parsed["parser_used"]),
1281
+ route,
1282
+ preflight,
1283
+ parser_quality,
1284
+ )
1285
+ if _is_docx(path, mime_type):
1286
+ await _mark_parser_active(ingest_id, project_slug)
1287
+ parsed = await asyncio.to_thread(parse_docx, path)
1288
+ parser_quality = estimate_parser_quality(
1289
+ parser_used="internal_docx",
1290
+ extracted_text=parsed.text,
1291
+ structure=parsed.structure,
1292
+ )
1293
+ return ParseDispatchResult(
1294
+ parsed, "internal_docx", route, preflight, parser_quality
1295
+ )
1296
+ if _is_transcript_media(path, mime_type):
1297
+ parsed = await _run_heavy_parser(
1298
+ ingest_id=ingest_id,
1299
+ project_slug=project_slug,
1300
+ lane="transcribe",
1301
+ parser_name="transcript",
1302
+ path=path,
1303
+ parse=lambda: parse_media_transcript(path, mime_type),
1304
+ )
1305
+ parser_quality = estimate_parser_quality(
1306
+ parser_used="tier_transcribe",
1307
+ extracted_text=parsed.text,
1308
+ structure=parsed.structure,
1309
+ )
1310
+ return ParseDispatchResult(
1311
+ parsed, "tier_transcribe", route, preflight, parser_quality
1312
+ )
1313
+ if path.suffix.lower() in _MARKDOWN_SUFFIXES or mime_type in {
1314
+ "text/markdown",
1315
+ "text/plain",
1316
+ }:
1317
+ await _mark_parser_active(ingest_id, project_slug)
1318
+ parsed = parse_markdown_file(path)
1319
+ parser_quality = estimate_parser_quality(
1320
+ parser_used="internal_markdown",
1321
+ extracted_text=parsed.text,
1322
+ structure=parsed.structure,
1323
+ )
1324
+ return ParseDispatchResult(
1325
+ parsed, "internal_markdown", route, preflight, parser_quality
1326
+ )
1327
+ raise ValueError(f"Unsupported phase-1 file type: {mime_type}")
1328
+
1329
+
1330
+ async def _parse_file_with_transient_retries(
1331
+ ingest_id: str,
1332
+ project_slug: str,
1333
+ path: Path,
1334
+ mime_type: str,
1335
+ *,
1336
+ preflight: dict[str, Any],
1337
+ route: IngestRoute,
1338
+ ) -> ParseDispatchResult:
1339
+ attempts = _transient_parse_max_attempts()
1340
+ for attempt in range(1, attempts + 1):
1341
+ try:
1342
+ return await _parse_file(
1343
+ ingest_id,
1344
+ project_slug,
1345
+ path,
1346
+ mime_type,
1347
+ preflight=preflight,
1348
+ route=route,
1349
+ )
1350
+ except Exception as exc:
1351
+ if attempt >= attempts or not _is_transient_parse_error(exc):
1352
+ raise
1353
+ delay = _transient_parse_delay_seconds(attempt)
1354
+ await _mark_parser_waiting(ingest_id, project_slug)
1355
+ logger.warning(
1356
+ "ingest parser transient failure; retrying attempt=%d/%d delay=%.1fs route=%s path=%s error=%s",
1357
+ attempt,
1358
+ attempts,
1359
+ delay,
1360
+ route.workflow,
1361
+ path,
1362
+ exc,
1363
+ )
1364
+ await asyncio.sleep(delay)
1365
+ raise RuntimeError("transient parse retry loop exited unexpectedly")
1366
+
1367
+
1368
+ def _with_ingest_v2_diagnostics(
1369
+ structure: dict[str, Any],
1370
+ *,
1371
+ preflight: dict[str, Any],
1372
+ route: IngestRoute,
1373
+ parser_quality: dict[str, Any],
1374
+ ) -> dict[str, Any]:
1375
+ merged = dict(structure or {})
1376
+ merged["ingest_v2"] = {
1377
+ "route": route.as_json(),
1378
+ "parser_quality": parser_quality,
1379
+ "preflight": _bounded_preflight_for_storage(preflight),
1380
+ }
1381
+ if preflight.get("source_context"):
1382
+ merged["ingest_v2"]["source_context"] = dict(preflight["source_context"])
1383
+ if (preflight.get("preflight") or {}).get("image_kind"):
1384
+ merged["ingest_v2"]["image_probe"] = {
1385
+ key: value
1386
+ for key, value in dict(preflight.get("preflight") or {}).items()
1387
+ if key
1388
+ in {
1389
+ "image_kind",
1390
+ "document_likelihood",
1391
+ "screenshot_likelihood",
1392
+ "photo_likelihood",
1393
+ "text_likelihood",
1394
+ "signals",
1395
+ "white_background_ratio",
1396
+ "edge_density",
1397
+ "brightness",
1398
+ "contrast",
1399
+ }
1400
+ }
1401
+ return merged
1402
+
1403
+
1404
+ def _bounded_preflight_for_storage(preflight: dict[str, Any]) -> dict[str, Any]:
1405
+ stored = {
1406
+ "file": dict(preflight.get("file") or {}),
1407
+ "preflight": dict(preflight.get("preflight") or {}),
1408
+ "content_sample": dict(preflight.get("content_sample") or {}),
1409
+ }
1410
+ sample = stored["content_sample"]
1411
+ for key in ("first_excerpt", "middle_excerpt", "last_excerpt", "parser_excerpt"):
1412
+ if key in sample:
1413
+ sample[key] = str(sample[key])[:500]
1414
+ return stored
1415
+
1416
+
1417
+ def _reconcile_ingress_metadata(
1418
+ structure: dict[str, Any] | None, ingress_metadata_raw: str | None
1419
+ ) -> dict[str, Any] | None:
1420
+ """Merge an api_ingress row's payload metadata into structure_json (U3).
1421
+
1422
+ Stored under the ``ingress_metadata`` key so one canonical place carries it
1423
+ downstream (Triage view + KG). No-op when there is no metadata or it is not
1424
+ a JSON object.
1425
+ """
1426
+ if not ingress_metadata_raw:
1427
+ return structure
1428
+ try:
1429
+ parsed = json.loads(ingress_metadata_raw)
1430
+ except (json.JSONDecodeError, TypeError):
1431
+ return structure
1432
+ if not isinstance(parsed, dict) or not parsed:
1433
+ return structure
1434
+ return {**(structure or {}), "ingress_metadata": parsed}
1435
+
1436
+
1437
+ async def parse_pending(ingest_id: str) -> None:
1438
+ row = None
1439
+ async with acquire_write_db() as db:
1440
+ async with db.execute(
1441
+ "SELECT * FROM ingest_pending WHERE id = ?", (ingest_id,)
1442
+ ) as cursor:
1443
+ row = await cursor.fetchone()
1444
+ if row is None:
1445
+ return
1446
+ if row["status"] not in {"queued", "parse_error"}:
1447
+ return
1448
+ await db.execute(
1449
+ """
1450
+ UPDATE ingest_pending
1451
+ SET status = 'parser_waiting',
1452
+ error_message = NULL,
1453
+ updated_at = datetime('now')
1454
+ WHERE id = ?
1455
+ """,
1456
+ (ingest_id,),
1457
+ )
1458
+ await db.commit()
1459
+
1460
+ project_slug = row["project_slug"]
1461
+ await broadcast_ingest_changed(
1462
+ "parser_waiting",
1463
+ ingest_id=ingest_id,
1464
+ project_slug=project_slug,
1465
+ status="parser_waiting",
1466
+ )
1467
+ try:
1468
+ path = Path(row["file_path"])
1469
+ if not path.exists():
1470
+ raise FileNotFoundError(str(path))
1471
+ pending_llm_project_slug: str | None = None
1472
+ error_message: str | None = None
1473
+ mime_type = detect_mime(path)
1474
+ preflight = build_preflight(path, mime_type)
1475
+ source_context = _source_context_for_row(
1476
+ project_slug=project_slug,
1477
+ source_kind=row["source_kind"],
1478
+ path=path,
1479
+ )
1480
+ if source_context:
1481
+ preflight["source_context"] = source_context
1482
+ route = await _maybe_llm_route(
1483
+ route=_route_for(path, mime_type, preflight),
1484
+ preflight=preflight,
1485
+ mime_type=mime_type,
1486
+ )
1487
+ dispatch = await _parse_file_with_transient_retries(
1488
+ ingest_id,
1489
+ project_slug,
1490
+ path,
1491
+ mime_type,
1492
+ preflight=preflight,
1493
+ route=route,
1494
+ )
1495
+ parsed = dispatch.parsed
1496
+ parser_used = dispatch.parser_used
1497
+ is_image = _is_image(path, mime_type)
1498
+ is_xlsx = _is_xlsx(path, mime_type)
1499
+ is_docx = _is_docx(path, mime_type)
1500
+ if is_image:
1501
+ item = {
1502
+ "file_path": row["file_path"],
1503
+ "file_size_bytes": row["file_size_bytes"],
1504
+ }
1505
+ auto_approve = should_auto_approve(item, parsed)
1506
+ classification_json = _image_classification(
1507
+ path=path,
1508
+ auto_approve=auto_approve,
1509
+ reason=(
1510
+ "safe image fast-lane"
1511
+ if auto_approve
1512
+ else "image requires manual triage"
1513
+ ),
1514
+ rules_matched=(
1515
+ ["safe_ext", "under_1mb", "exif_redacted", "no_pii"]
1516
+ if auto_approve
1517
+ else []
1518
+ ),
1519
+ )
1520
+ next_status = "done" if auto_approve else "awaiting_triage"
1521
+ extracted_text = str(
1522
+ parsed.get("text") or parsed.get("extracted_text") or ""
1523
+ )
1524
+ structure = parsed.get("structure") or {}
1525
+ target_folder = classification_json["target_folder"]
1526
+ target_filename = classification_json["target_filename"]
1527
+ triage_decision_id = "auto_approve:image_parser" if auto_approve else None
1528
+ elif is_xlsx:
1529
+ classification_json = _xlsx_classification(path)
1530
+ next_status = "awaiting_triage"
1531
+ extracted_text = str(parsed.get("text") or "")
1532
+ structure = parsed.get("structure") or {}
1533
+ target_folder = classification_json["target_folder"]
1534
+ target_filename = classification_json["target_filename"]
1535
+ triage_decision_id = None
1536
+ elif is_docx:
1537
+ classification = classify_markdown(
1538
+ frontmatter=parsed.frontmatter,
1539
+ original_filename=f"{path.stem}.md",
1540
+ )
1541
+ classification_json = classification.as_json()
1542
+ next_status = "awaiting_triage"
1543
+ extracted_text = parsed.text
1544
+ structure = parsed.structure
1545
+ target_folder = classification.target_folder
1546
+ target_filename = classification.target_filename
1547
+ triage_decision_id = None
1548
+ else:
1549
+ classification = classify_markdown(
1550
+ frontmatter=parsed.frontmatter,
1551
+ original_filename=_classification_filename(path, mime_type),
1552
+ )
1553
+ classification_json = classification.as_json()
1554
+ next_status = "awaiting_triage"
1555
+ extracted_text = parsed.text
1556
+ structure = parsed.structure
1557
+ target_folder = classification.target_folder
1558
+ target_filename = classification.target_filename
1559
+ triage_decision_id = None
1560
+
1561
+ parser_quality = dict(dispatch.parser_quality)
1562
+ transcript_summary = await _maybe_summarize_transcript(
1563
+ ingest_id=ingest_id,
1564
+ parser_used=parser_used,
1565
+ extracted_text=extracted_text,
1566
+ structure=structure,
1567
+ )
1568
+ if transcript_summary is not None:
1569
+ structure = dict(structure or {})
1570
+ structure["transcript_summary"] = transcript_summary
1571
+ if transcript_summary.get("status") == "ok":
1572
+ classification_json = dict(classification_json)
1573
+ classification_json["transcript_summary"] = transcript_summary
1574
+ parser_quality["transcript_summary"] = transcript_summary
1575
+
1576
+ structure = _with_ingest_v2_diagnostics(
1577
+ structure,
1578
+ preflight=dispatch.preflight,
1579
+ route=dispatch.route,
1580
+ parser_quality=parser_quality,
1581
+ )
1582
+
1583
+ if next_status == "awaiting_triage" and extracted_text.strip():
1584
+ # Ingestor 2.0: optional local tier-fast project routing +
1585
+ # frontmatter inference. Auto-approval requires composite
1586
+ # confidence >= 0.80, not just LLM self-confidence.
1587
+ try:
1588
+ enriched = await _maybe_llm_enrich(
1589
+ ingest_id=ingest_id,
1590
+ extracted_text=extracted_text,
1591
+ base_classification=classification_json,
1592
+ preflight=dispatch.preflight,
1593
+ parser_quality=parser_quality,
1594
+ route=dispatch.route,
1595
+ )
1596
+ except Exception: # noqa: BLE001 - never break the saga
1597
+ logger.exception("llm enrichment failed")
1598
+ enriched = None
1599
+ if enriched is not None:
1600
+ classification_json = enriched
1601
+ target_folder = enriched.get("target_folder", target_folder)
1602
+ llm_error_message = _llm_failure_error_message(enriched)
1603
+ if llm_error_message:
1604
+ next_status = "parse_error"
1605
+ error_message = llm_error_message
1606
+ triage_decision_id = None
1607
+ elif enriched.get("auto_approve") is True:
1608
+ # Stay in 'approved' so execute_saga (scheduled below)
1609
+ # picks the row up; saga moves the file to target_folder,
1610
+ # populates the KG, indexes the embedding, then flips
1611
+ # status to 'inserted' and finally 'done'.
1612
+ next_status = "approved"
1613
+ triage_decision_id = "auto_approve:llm_routing"
1614
+ # Move row + file to LLM-suggested project so the saga
1615
+ # processes the artifact under the new project root
1616
+ # (saga rejects file_path that escapes project_root).
1617
+ suggested_slug = enriched.get("suggested_project_slug")
1618
+ if suggested_slug and suggested_slug != project_slug:
1619
+ pending_llm_project_slug = str(suggested_slug)
1620
+
1621
+ # --- U3 per-source policy gate (single authority; saga only asserts) ---
1622
+ # Owner surfaces are policy-exempt (decide_ingress_routing returns None).
1623
+ # api_ingress: 'open'/unknown -> always triage (default-deny); 'trusted'
1624
+ # -> keeps the intrinsic auto-insert decision (necessary-not-sufficient).
1625
+ # An 'open' downgrade flips next_status to awaiting_triage, which also
1626
+ # disables the saga/project-switch blocks below (they require 'approved').
1627
+ ingress_routing = decide_ingress_routing(
1628
+ source_kind=row["source_kind"],
1629
+ ingest_policy=row["ingest_policy"],
1630
+ intrinsic_status=next_status,
1631
+ intrinsic_basis=triage_decision_id,
1632
+ )
1633
+ if ingress_routing is not None:
1634
+ next_status = ingress_routing.status
1635
+ triage_decision_id = ingress_routing.triage_decision_id
1636
+ if pending_llm_project_slug and not ingress_routing.auto_insert:
1637
+ pending_llm_project_slug = None
1638
+ classification_json = dict(classification_json)
1639
+ classification_json["ingest_policy"] = {
1640
+ "effective_policy": row["ingest_policy"] or "open",
1641
+ "auto_insert": ingress_routing.auto_insert,
1642
+ "decision": ingress_routing.decision,
1643
+ }
1644
+ if "auto_approve" in classification_json:
1645
+ classification_json["auto_approve"] = ingress_routing.auto_insert
1646
+
1647
+ # --- U3 reconcile ingress payload metadata into structure_json ---
1648
+ ingress_metadata_raw = (
1649
+ row["ingress_metadata"] if "ingress_metadata" in row.keys() else None
1650
+ )
1651
+ structure = _reconcile_ingress_metadata(structure, ingress_metadata_raw)
1652
+
1653
+ async with acquire_write_db() as db:
1654
+ if (
1655
+ pending_llm_project_slug
1656
+ and next_status == "approved"
1657
+ and triage_decision_id == "auto_approve:llm_routing"
1658
+ ):
1659
+ (
1660
+ project_slug,
1661
+ path,
1662
+ classification_json,
1663
+ next_status,
1664
+ triage_decision_id,
1665
+ ) = await _apply_llm_project_switch_if_safe(
1666
+ db,
1667
+ ingest_id=ingest_id,
1668
+ sha256=row["sha256"],
1669
+ current_project_slug=project_slug,
1670
+ target_project_slug=pending_llm_project_slug,
1671
+ path=path,
1672
+ classification_json=classification_json,
1673
+ )
1674
+ await db.execute(
1675
+ """
1676
+ UPDATE ingest_pending
1677
+ SET status = ?,
1678
+ project_slug = ?,
1679
+ file_path = ?,
1680
+ mime_type = ?,
1681
+ parser_used = ?,
1682
+ extracted_text = ?,
1683
+ structure_json = ?,
1684
+ classification_json = ?,
1685
+ target_folder = ?,
1686
+ target_filename = ?,
1687
+ triage_decision_id = ?,
1688
+ error_message = ?,
1689
+ updated_at = datetime('now')
1690
+ WHERE id = ?
1691
+ """,
1692
+ (
1693
+ next_status,
1694
+ project_slug,
1695
+ str(path),
1696
+ mime_type,
1697
+ parser_used,
1698
+ extracted_text,
1699
+ json.dumps(structure, ensure_ascii=False),
1700
+ json.dumps(classification_json, ensure_ascii=False),
1701
+ target_folder,
1702
+ target_filename,
1703
+ triage_decision_id,
1704
+ error_message,
1705
+ ingest_id,
1706
+ ),
1707
+ )
1708
+ await db.commit()
1709
+ await broadcast_ingest_changed(
1710
+ _ingest_event_for_status(next_status),
1711
+ ingest_id=ingest_id,
1712
+ project_slug=project_slug,
1713
+ status=next_status,
1714
+ )
1715
+ if (
1716
+ next_status == "approved"
1717
+ and triage_decision_id == "auto_approve:llm_routing"
1718
+ ):
1719
+ # Lazy import to avoid circular dependency (insert_saga imports
1720
+ # nothing here, but parse_pending is the canonical entry point
1721
+ # so we keep the boundary explicit).
1722
+ from core.api.services.ingest.insert_saga import execute_saga
1723
+
1724
+ asyncio.create_task(execute_saga(ingest_id))
1725
+ except _ParserWaitCancelled:
1726
+ logger.info("ingest parse cancelled before parser slot: id=%s", ingest_id)
1727
+ return
1728
+ except Exception as exc:
1729
+ logger.exception("ingest parse failed: id=%s", ingest_id)
1730
+ await _mark_parse_error(ingest_id, project_slug, str(exc))
1731
+
1732
+
1733
+ def _source_context_for_row(
1734
+ *,
1735
+ project_slug: str | None,
1736
+ source_kind: str | None,
1737
+ path: Path,
1738
+ ) -> dict[str, Any]:
1739
+ if not project_slug or source_kind != "terminal_upload":
1740
+ return {}
1741
+ try:
1742
+ project_root = PROJECTS_ROOT / project_slug
1743
+ in_project_input = path.resolve().is_relative_to((project_root / "input").resolve())
1744
+ except Exception:
1745
+ in_project_input = False
1746
+ if in_project_input:
1747
+ return {
1748
+ "project_slug": project_slug,
1749
+ "prior": 0.95,
1750
+ "reason": f"{source_kind or 'ingest'}_project_input",
1751
+ }
1752
+ return {
1753
+ "project_slug": project_slug,
1754
+ "prior": 0.75,
1755
+ "reason": f"{source_kind or 'ingest'}_row_project",
1756
+ }