contextsynapse 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (776) hide show
  1. contextsynapse/__init__.py +107 -0
  2. contextsynapse/__main__.py +9 -0
  3. contextsynapse/a2a/__init__.py +28 -0
  4. contextsynapse/a2a/client.py +120 -0
  5. contextsynapse/a2a/discovery.py +128 -0
  6. contextsynapse/a2a/handlers.py +164 -0
  7. contextsynapse/a2a/models.py +246 -0
  8. contextsynapse/a2a/streaming.py +92 -0
  9. contextsynapse/a2a/task_manager.py +275 -0
  10. contextsynapse/adapters/__init__.py +98 -0
  11. contextsynapse/adapters/_base.py +298 -0
  12. contextsynapse/adapters/autogen/__init__.py +4 -0
  13. contextsynapse/adapters/autogen/tools.py +34 -0
  14. contextsynapse/adapters/crewai/__init__.py +4 -0
  15. contextsynapse/adapters/crewai/tools.py +63 -0
  16. contextsynapse/adapters/langchain/__init__.py +19 -0
  17. contextsynapse/adapters/langchain/message_history.py +17 -0
  18. contextsynapse/adapters/langchain/retriever.py +158 -0
  19. contextsynapse/adapters/langchain/tools.py +43 -0
  20. contextsynapse/adapters/langgraph/__init__.py +21 -0
  21. contextsynapse/adapters/langgraph/checkpoint.py +223 -0
  22. contextsynapse/adapters/langgraph/context_tools.py +341 -0
  23. contextsynapse/adapters/langgraph/message_history.py +138 -0
  24. contextsynapse/adapters/langgraph/tools.py +43 -0
  25. contextsynapse/adapters/llamaindex/__init__.py +4 -0
  26. contextsynapse/adapters/llamaindex/tools.py +64 -0
  27. contextsynapse/adapters/openai/__init__.py +24 -0
  28. contextsynapse/adapters/openai/tools.py +206 -0
  29. contextsynapse/adapters/pydantic_ai/__init__.py +4 -0
  30. contextsynapse/adapters/pydantic_ai/tools.py +36 -0
  31. contextsynapse/adapters/swarm/__init__.py +4 -0
  32. contextsynapse/adapters/swarm/tools.py +44 -0
  33. contextsynapse/agents/__init__.py +6 -0
  34. contextsynapse/agents/auto_capture.py +80 -0
  35. contextsynapse/agents/dispatcher.py +299 -0
  36. contextsynapse/agents/notifications.py +112 -0
  37. contextsynapse/aiql/__init__.py +71 -0
  38. contextsynapse/aiql/cache.py +341 -0
  39. contextsynapse/aiql/compiler/__init__.py +62 -0
  40. contextsynapse/aiql/compiler/optimizer.py +126 -0
  41. contextsynapse/aiql/compiler/planner.py +366 -0
  42. contextsynapse/aiql/compiler/statistics.py +235 -0
  43. contextsynapse/aiql/compiler/validator.py +425 -0
  44. contextsynapse/aiql/engine/__init__.py +11 -0
  45. contextsynapse/aiql/engine/chunk_storage.py +2584 -0
  46. contextsynapse/aiql/engine/executor.py +7737 -0
  47. contextsynapse/aiql/engine/extraction_integration.py +442 -0
  48. contextsynapse/aiql/engine/graph_builder.py +405 -0
  49. contextsynapse/aiql/engine/node_creation_from_normalized.py +691 -0
  50. contextsynapse/aiql/engine/performance_monitor.py +289 -0
  51. contextsynapse/aiql/engine/postprocessing.py +365 -0
  52. contextsynapse/aiql/engine/ray_stage_executors.py +918 -0
  53. contextsynapse/aiql/engine/redis_cache.py +106 -0
  54. contextsynapse/aiql/engine/streaming_json.py +366 -0
  55. contextsynapse/aiql/engine/streaming_parquet.py +568 -0
  56. contextsynapse/aiql/engine/streaming_ray_chunking.py +419 -0
  57. contextsynapse/aiql/grammar/__init__.py +14 -0
  58. contextsynapse/aiql/grammar/aiql_grammar.py +1822 -0
  59. contextsynapse/aiql/grammar/base.py +71 -0
  60. contextsynapse/aiql/parser/__init__.py +46 -0
  61. contextsynapse/aiql/parser/aiql_parser.py +11330 -0
  62. contextsynapse/aiql/parser/base_parser.py +47 -0
  63. contextsynapse/api/__init__.py +5 -0
  64. contextsynapse/api/a2a_router.py +233 -0
  65. contextsynapse/api/admin_router.py +140 -0
  66. contextsynapse/api/agent_worker_router.py +978 -0
  67. contextsynapse/api/algorithms_router.py +317 -0
  68. contextsynapse/api/api.py +4144 -0
  69. contextsynapse/api/audit.py +204 -0
  70. contextsynapse/api/auth.py +441 -0
  71. contextsynapse/api/auth_router.py +537 -0
  72. contextsynapse/api/billing.py +261 -0
  73. contextsynapse/api/billing_router.py +275 -0
  74. contextsynapse/api/boundary_router.py +452 -0
  75. contextsynapse/api/cognition_router.py +116 -0
  76. contextsynapse/api/connector_router.py +237 -0
  77. contextsynapse/api/context_router.py +1619 -0
  78. contextsynapse/api/dashboard_router.py +11367 -0
  79. contextsynapse/api/errors.py +183 -0
  80. contextsynapse/api/events.py +88 -0
  81. contextsynapse/api/experiment_bootstrap.py +321 -0
  82. contextsynapse/api/experiment_orchestrator.py +1078 -0
  83. contextsynapse/api/experiment_router.py +2089 -0
  84. contextsynapse/api/feature_gate.py +89 -0
  85. contextsynapse/api/federation_router.py +108 -0
  86. contextsynapse/api/integrations.py +567 -0
  87. contextsynapse/api/integrations_router.py +461 -0
  88. contextsynapse/api/intelligence_router.py +1028 -0
  89. contextsynapse/api/metering.py +155 -0
  90. contextsynapse/api/metrics.py +112 -0
  91. contextsynapse/api/models.py +38 -0
  92. contextsynapse/api/monitoring_router.py +174 -0
  93. contextsynapse/api/pipeline_router.py +1437 -0
  94. contextsynapse/api/playground_conn.py +116 -0
  95. contextsynapse/api/projects_router.py +131 -0
  96. contextsynapse/api/queries_router.py +282 -0
  97. contextsynapse/api/quota.py +95 -0
  98. contextsynapse/api/rate_limit.py +64 -0
  99. contextsynapse/api/search_router.py +261 -0
  100. contextsynapse/api/shield_router.py +447 -0
  101. contextsynapse/api/tenant_guard.py +103 -0
  102. contextsynapse/api/tenant_quotas.py +231 -0
  103. contextsynapse/api/tenants.py +305 -0
  104. contextsynapse/api/users.py +646 -0
  105. contextsynapse/api/v1.py +167 -0
  106. contextsynapse/api/vertical_builder_router.py +387 -0
  107. contextsynapse/benchmark/__init__.py +15 -0
  108. contextsynapse/benchmark/__main__.py +64 -0
  109. contextsynapse/benchmark/agents.py +338 -0
  110. contextsynapse/benchmark/live_ab.py +338 -0
  111. contextsynapse/benchmark/reporter.py +149 -0
  112. contextsynapse/benchmark/runner.py +264 -0
  113. contextsynapse/benchmark/savings.py +266 -0
  114. contextsynapse/benchmark/token_counter.py +212 -0
  115. contextsynapse/benchmark/workloads.py +289 -0
  116. contextsynapse/buffer/__init__.py +154 -0
  117. contextsynapse/buffer/base.py +257 -0
  118. contextsynapse/buffer/duckdb_buffer.py +246 -0
  119. contextsynapse/buffer/file_buffer.py +229 -0
  120. contextsynapse/buffer/lmdb_buffer.py +286 -0
  121. contextsynapse/buffer/local_buffer.py +154 -0
  122. contextsynapse/buffer/redis_buffer.py +211 -0
  123. contextsynapse/buffer/sqlite_buffer.py +266 -0
  124. contextsynapse/cache.py +2 -0
  125. contextsynapse/cli/__init__.py +40 -0
  126. contextsynapse/cli/cli.py +475 -0
  127. contextsynapse/cli/config_cli.py +177 -0
  128. contextsynapse/cli/scaffold.py +465 -0
  129. contextsynapse/cognition/__init__.py +31 -0
  130. contextsynapse/cognition/derivation.py +55 -0
  131. contextsynapse/cognition/events.py +135 -0
  132. contextsynapse/cognition/feedback.py +52 -0
  133. contextsynapse/cognition/hallucination.py +90 -0
  134. contextsynapse/cognition/invalidation.py +70 -0
  135. contextsynapse/cognition/read_tracker.py +50 -0
  136. contextsynapse/collaboration.py +2 -0
  137. contextsynapse/comparison.py +2 -0
  138. contextsynapse/config/__init__.py +10 -0
  139. contextsynapse/config/connectors/amfi_nav.yaml +18 -0
  140. contextsynapse/config/connectors/auto_sector.yaml +20 -0
  141. contextsynapse/config/connectors/banking_sector.yaml +20 -0
  142. contextsynapse/config/connectors/energy_sector.yaml +20 -0
  143. contextsynapse/config/connectors/fmcg_sector.yaml +20 -0
  144. contextsynapse/config/connectors/nsdl_fpi.yaml +9 -0
  145. contextsynapse/config/connectors/nse_price.yaml +61 -0
  146. contextsynapse/config/connectors/pharma_sector.yaml +20 -0
  147. contextsynapse/config/connectors/rbi_dbie.yaml +12 -0
  148. contextsynapse/config/context_presets/ai_research.json +25 -0
  149. contextsynapse/config/context_presets/chatgpt_conversations.json +19 -0
  150. contextsynapse/config/context_presets/claude_conversations.json +19 -0
  151. contextsynapse/config/context_presets/legal_compliance.json +21 -0
  152. contextsynapse/config/context_presets/meeting_notes.json +21 -0
  153. contextsynapse/config/context_presets/slack_engineering.json +21 -0
  154. contextsynapse/config/context_presets/tech_climate.json +24 -0
  155. contextsynapse/config/context_presets/toi_politics.json +25 -0
  156. contextsynapse/config/database_config.py +453 -0
  157. contextsynapse/config/deployment.py +221 -0
  158. contextsynapse/config/gpt_action_schema.json +164 -0
  159. contextsynapse/config/indicators/india_macro.yaml +213 -0
  160. contextsynapse/config/logging_config.py +88 -0
  161. contextsynapse/config/mode_config.py +223 -0
  162. contextsynapse/config/schemas/api_spec.yaml +48 -0
  163. contextsynapse/config/schemas/architecture_doc.yaml +51 -0
  164. contextsynapse/config/schemas/chat_history.yaml +90 -0
  165. contextsynapse/config/schemas/chatgpt_conversation.yaml +101 -0
  166. contextsynapse/config/schemas/claude_conversation.yaml +148 -0
  167. contextsynapse/config/schemas/contract.yaml +48 -0
  168. contextsynapse/config/schemas/conversation.yaml +167 -0
  169. contextsynapse/config/schemas/database.yaml +74 -0
  170. contextsynapse/config/schemas/decision.yaml +72 -0
  171. contextsynapse/config/schemas/ecommerce.yaml +56 -0
  172. contextsynapse/config/schemas/education.yaml +42 -0
  173. contextsynapse/config/schemas/healthcare.yaml +133 -0
  174. contextsynapse/config/schemas/hr_document.yaml +48 -0
  175. contextsynapse/config/schemas/invoice.yaml +40 -0
  176. contextsynapse/config/schemas/knowledge_base.yaml +75 -0
  177. contextsynapse/config/schemas/legal.yaml +42 -0
  178. contextsynapse/config/schemas/marketing_content.yaml +48 -0
  179. contextsynapse/config/schemas/meeting_notes.yaml +51 -0
  180. contextsynapse/config/schemas/news_article.yaml +61 -0
  181. contextsynapse/config/schemas/real_estate.yaml +49 -0
  182. contextsynapse/config/schemas/requirements_doc.yaml +48 -0
  183. contextsynapse/config/schemas/research_paper.yaml +56 -0
  184. contextsynapse/config/schemas/resume.yaml +49 -0
  185. contextsynapse/config/schemas/retail.yaml +42 -0
  186. contextsynapse/config/schemas/rule_meta.yaml +173 -0
  187. contextsynapse/config/schemas/rules.yaml +72 -0
  188. contextsynapse/config/schemas/screener_meta.yaml +236 -0
  189. contextsynapse/config/schemas/sdlc.yaml +103 -0
  190. contextsynapse/config/schemas/session_graph.yaml +172 -0
  191. contextsynapse/config/schemas/support_ticket.yaml +48 -0
  192. contextsynapse/config/schemas/web.yaml +64 -0
  193. contextsynapse/config/sensors/earnings_calendar.yaml +52 -0
  194. contextsynapse/config/sensors/global_risk.yaml +53 -0
  195. contextsynapse/config/sensors/government_policy.yaml +54 -0
  196. contextsynapse/config/sensors/supply_chain.yaml +50 -0
  197. contextsynapse/config/sensors/templates/behavioral.yaml +56 -0
  198. contextsynapse/config/sensors/templates/equity_sentiment.yaml +48 -0
  199. contextsynapse/config/sensors/templates/global_macro.yaml +47 -0
  200. contextsynapse/config/sensors/templates/market_data.yaml +118 -0
  201. contextsynapse/config/sensors/templates/sector_technical.yaml +47 -0
  202. contextsynapse/config/sensors/weather.yaml +52 -0
  203. contextsynapse/config/storage_strategy.py +202 -0
  204. contextsynapse/config/templates/competitive_intel.yaml +39 -0
  205. contextsynapse/config/templates/geopolitical_risk.yaml +46 -0
  206. contextsynapse/config/templates/market_analysis.yaml +58 -0
  207. contextsynapse/connectors/__init__.py +15 -0
  208. contextsynapse/connectors/api_router.py +121 -0
  209. contextsynapse/connectors/base.py +357 -0
  210. contextsynapse/connectors/financial/audio.py +170 -0
  211. contextsynapse/connectors/financial/youtube.py +272 -0
  212. contextsynapse/connectors/github.py +164 -0
  213. contextsynapse/connectors/jira.py +120 -0
  214. contextsynapse/connectors/kafka_connector.py +124 -0
  215. contextsynapse/connectors/pipeline_connector.py +747 -0
  216. contextsynapse/connectors/registry.py +116 -0
  217. contextsynapse/connectors/scheduler.py +87 -0
  218. contextsynapse/connectors/universal.py +360 -0
  219. contextsynapse/context/__init__.py +96 -0
  220. contextsynapse/context/__main__.py +4 -0
  221. contextsynapse/context/acl.py +125 -0
  222. contextsynapse/context/agent_card.py +201 -0
  223. contextsynapse/context/agent_memory.py +814 -0
  224. contextsynapse/context/agents.py +472 -0
  225. contextsynapse/context/agents_redis.py +391 -0
  226. contextsynapse/context/assembled.py +204 -0
  227. contextsynapse/context/attribution.py +113 -0
  228. contextsynapse/context/blob.py +184 -0
  229. contextsynapse/context/boundaries.py +187 -0
  230. contextsynapse/context/boundary.py +507 -0
  231. contextsynapse/context/bundle.py +155 -0
  232. contextsynapse/context/cli.py +519 -0
  233. contextsynapse/context/collaboration.py +165 -0
  234. contextsynapse/context/compiler.py +604 -0
  235. contextsynapse/context/composite.py +536 -0
  236. contextsynapse/context/compression.py +147 -0
  237. contextsynapse/context/context_manager.py +888 -0
  238. contextsynapse/context/context_schema.py +90 -0
  239. contextsynapse/context/context_schema.yaml +105 -0
  240. contextsynapse/context/context_units.py +891 -0
  241. contextsynapse/context/conversation.py +488 -0
  242. contextsynapse/context/dedup.py +207 -0
  243. contextsynapse/context/document_processor.py +698 -0
  244. contextsynapse/context/embedding_hooks.py +338 -0
  245. contextsynapse/context/execution_schema.py +274 -0
  246. contextsynapse/context/fan_out.py +138 -0
  247. contextsynapse/context/frozen.py +80 -0
  248. contextsynapse/context/gravity.py +227 -0
  249. contextsynapse/context/hub.py +1401 -0
  250. contextsynapse/context/ingest.py +391 -0
  251. contextsynapse/context/injection.py +226 -0
  252. contextsynapse/context/interaction_graphifier.py +562 -0
  253. contextsynapse/context/layers.py +574 -0
  254. contextsynapse/context/projection.py +1195 -0
  255. contextsynapse/context/promotion.py +634 -0
  256. contextsynapse/context/proof.py +131 -0
  257. contextsynapse/context/propagation.py +577 -0
  258. contextsynapse/context/pubsub.py +135 -0
  259. contextsynapse/context/quality.py +1054 -0
  260. contextsynapse/context/quality_tracker.py +270 -0
  261. contextsynapse/context/scoping.py +370 -0
  262. contextsynapse/context/seed.py +992 -0
  263. contextsynapse/context/session.py +1142 -0
  264. contextsynapse/context/session_graph.py +593 -0
  265. contextsynapse/context/session_memory.py +309 -0
  266. contextsynapse/context/session_resolver.py +111 -0
  267. contextsynapse/context/skill_learning.py +237 -0
  268. contextsynapse/context/smart_budget.py +129 -0
  269. contextsynapse/context/source_policy.py +242 -0
  270. contextsynapse/context/store_factory.py +53 -0
  271. contextsynapse/context/sync.py +648 -0
  272. contextsynapse/context/token.py +236 -0
  273. contextsynapse/context/vector_integration.py +297 -0
  274. contextsynapse/context/webhooks.py +194 -0
  275. contextsynapse/context/working_memory.py +224 -0
  276. contextsynapse/core/__init__.py +18 -0
  277. contextsynapse/core/archiver.py +237 -0
  278. contextsynapse/core/change_stream.py +282 -0
  279. contextsynapse/core/checkpoint.py +310 -0
  280. contextsynapse/core/cloud_storage.py +387 -0
  281. contextsynapse/core/context_state.py +651 -0
  282. contextsynapse/core/cron.py +324 -0
  283. contextsynapse/core/datasource.py +643 -0
  284. contextsynapse/core/db.py +289 -0
  285. contextsynapse/core/federation.py +283 -0
  286. contextsynapse/core/graph_coordinator.py +201 -0
  287. contextsynapse/core/graph_intelligence.py +388 -0
  288. contextsynapse/core/graph_structures.py +106 -0
  289. contextsynapse/core/graph_sync.py +157 -0
  290. contextsynapse/core/hybrid_graph_storage.py +2998 -0
  291. contextsynapse/core/lifecycle.py +691 -0
  292. contextsynapse/core/multiworker.py +378 -0
  293. contextsynapse/core/pruning.py +312 -0
  294. contextsynapse/core/redis_registry.py +524 -0
  295. contextsynapse/core/registry.py +1043 -0
  296. contextsynapse/core/registry_factory.py +51 -0
  297. contextsynapse/core/registry_metadata.py +294 -0
  298. contextsynapse/core/replay.py +354 -0
  299. contextsynapse/core/replication.py +339 -0
  300. contextsynapse/core/time_travel.py +260 -0
  301. contextsynapse/core/write_behind.py +152 -0
  302. contextsynapse/core/write_context.py +39 -0
  303. contextsynapse/db/__init__.py +1 -0
  304. contextsynapse/db/postgres.py +267 -0
  305. contextsynapse/db/profile.py +236 -0
  306. contextsynapse/db/rules.py +328 -0
  307. contextsynapse/demo/__init__.py +0 -0
  308. contextsynapse/demo/report.py +147 -0
  309. contextsynapse/demo/runner.py +520 -0
  310. contextsynapse/demo/scenario.py +165 -0
  311. contextsynapse/demo/session_demo.py +280 -0
  312. contextsynapse/demo/session_report.py +143 -0
  313. contextsynapse/demo/universal_runner.py +257 -0
  314. contextsynapse/engine/__init__.py +15 -0
  315. contextsynapse/engine/core.py +555 -0
  316. contextsynapse/extraction/__init__.py +69 -0
  317. contextsynapse/extraction/cli.py +274 -0
  318. contextsynapse/extraction/config.py +107 -0
  319. contextsynapse/extraction/examples/example_usage.py +118 -0
  320. contextsynapse/extraction/examples/sample_manifest.json +53 -0
  321. contextsynapse/extraction/examples/sample_normalized.json +185 -0
  322. contextsynapse/extraction/extract_engine.py +440 -0
  323. contextsynapse/extraction/extractors/__init__.py +37 -0
  324. contextsynapse/extraction/extractors/chat_export_extractor.py +189 -0
  325. contextsynapse/extraction/extractors/csv_extractor.py +83 -0
  326. contextsynapse/extraction/extractors/docx_extractor.py +129 -0
  327. contextsynapse/extraction/extractors/excel_extractor.py +152 -0
  328. contextsynapse/extraction/extractors/html_extractor.py +146 -0
  329. contextsynapse/extraction/extractors/pdf_extractor.py +635 -0
  330. contextsynapse/extraction/extractors/pdf_extractor_layoutparser.py +408 -0
  331. contextsynapse/extraction/extractors/register_extractors.py +63 -0
  332. contextsynapse/extraction/extractors/text_extractor.py +65 -0
  333. contextsynapse/extraction/extractors/txt_extractor.py +120 -0
  334. contextsynapse/extraction/extractors/website_extractor.py +246 -0
  335. contextsynapse/extraction/fact_extractor.py +236 -0
  336. contextsynapse/extraction/file_manager.py +353 -0
  337. contextsynapse/extraction/hierarchy.py +239 -0
  338. contextsynapse/extraction/id_generator.py +450 -0
  339. contextsynapse/extraction/layout_detector.py +409 -0
  340. contextsynapse/extraction/llm_entity_extractor.py +301 -0
  341. contextsynapse/extraction/normalized_store.py +480 -0
  342. contextsynapse/extraction/normalizer.py +594 -0
  343. contextsynapse/extraction/parquet_writer.py +598 -0
  344. contextsynapse/extraction/post_processors/__init__.py +40 -0
  345. contextsynapse/extraction/post_processors/column_reorganizer.py +507 -0
  346. contextsynapse/extraction/ray_normalization.py +233 -0
  347. contextsynapse/extraction/ray_runner.py +207 -0
  348. contextsynapse/extraction/redis_schema_registry.py +222 -0
  349. contextsynapse/extraction/registry.py +204 -0
  350. contextsynapse/extraction/schema_loader.py +663 -0
  351. contextsynapse/extraction/schema_registry.py +230 -0
  352. contextsynapse/extraction/semantic_normalizer.py +245 -0
  353. contextsynapse/extraction/version_manager.py +314 -0
  354. contextsynapse/gateway/__init__.py +24 -0
  355. contextsynapse/gateway/action_emitter.py +212 -0
  356. contextsynapse/gateway/cost_tracker.py +508 -0
  357. contextsynapse/gateway/gateway.py +377 -0
  358. contextsynapse/gateway/policy.py +303 -0
  359. contextsynapse/governance/__init__.py +13 -0
  360. contextsynapse/governance/layer.py +513 -0
  361. contextsynapse/ingestion/__init__.py +15 -0
  362. contextsynapse/ingestion/amplifier.py +288 -0
  363. contextsynapse/ingestion/async_ingest.py +463 -0
  364. contextsynapse/ingestion/chunker.py +303 -0
  365. contextsynapse/ingestion/chunking.py +58 -0
  366. contextsynapse/ingestion/cleaner.py +190 -0
  367. contextsynapse/ingestion/cleanse.py +252 -0
  368. contextsynapse/ingestion/connectors/__init__.py +38 -0
  369. contextsynapse/ingestion/connectors/base.py +300 -0
  370. contextsynapse/ingestion/connectors/chatgpt_memory.py +247 -0
  371. contextsynapse/ingestion/connectors/claude_memory.py +274 -0
  372. contextsynapse/ingestion/connectors/sap_ingestor.py +208 -0
  373. contextsynapse/ingestion/content_detector.py +80 -0
  374. contextsynapse/ingestion/content_monitor.py +502 -0
  375. contextsynapse/ingestion/dedup.py +106 -0
  376. contextsynapse/ingestion/feed.py +640 -0
  377. contextsynapse/ingestion/filters.py +461 -0
  378. contextsynapse/ingestion/graph_builder.py +705 -0
  379. contextsynapse/ingestion/input_classifier.py +307 -0
  380. contextsynapse/ingestion/job_manager.py +571 -0
  381. contextsynapse/ingestion/llm_extractor.py +610 -0
  382. contextsynapse/ingestion/orchestrator.py +116 -0
  383. contextsynapse/ingestion/parsers/__init__.py +19 -0
  384. contextsynapse/ingestion/parsers/graph_parser_registry.py +103 -0
  385. contextsynapse/ingestion/parsers/graphml_parser.py +141 -0
  386. contextsynapse/ingestion/parsers/jsonld_parser.py +171 -0
  387. contextsynapse/ingestion/parsers/rdf_parser.py +150 -0
  388. contextsynapse/ingestion/pipeline_context.py +228 -0
  389. contextsynapse/ingestion/pipeline_store.py +1195 -0
  390. contextsynapse/ingestion/queue.py +133 -0
  391. contextsynapse/ingestion/reprocess.py +179 -0
  392. contextsynapse/ingestion/scenario_router.py +224 -0
  393. contextsynapse/ingestion/schema_extractor.py +759 -0
  394. contextsynapse/ingestion/schema_validator.py +274 -0
  395. contextsynapse/ingestion/sdlc_ingest.py +87 -0
  396. contextsynapse/ingestion/smart_ingest.py +2451 -0
  397. contextsynapse/ingestion/source_connector.py +171 -0
  398. contextsynapse/ingestion/stage_executor.py +2095 -0
  399. contextsynapse/ingestion/stages/__init__.py +9 -0
  400. contextsynapse/ingestion/stages/tabular_to_graph.py +289 -0
  401. contextsynapse/ingestion/strategies.py +386 -0
  402. contextsynapse/ingestion/universal/__init__.py +4 -0
  403. contextsynapse/ingestion/universal/_operator_registry.py +59 -0
  404. contextsynapse/ingestion/universal/ingest_content.py +32 -0
  405. contextsynapse/ingestion/universal/operators/__init__.py +1 -0
  406. contextsynapse/ingestion/universal/operators/base.py +19 -0
  407. contextsynapse/ingestion/universal/operators/build_edges.py +75 -0
  408. contextsynapse/ingestion/universal/operators/cluster_topics.py +397 -0
  409. contextsynapse/ingestion/universal/operators/deduplicate.py +45 -0
  410. contextsynapse/ingestion/universal/operators/detect_signals.py +90 -0
  411. contextsynapse/ingestion/universal/operators/embed.py +80 -0
  412. contextsynapse/ingestion/universal/operators/extract_entities.py +253 -0
  413. contextsynapse/ingestion/universal/operators/extract_preferences.py +117 -0
  414. contextsynapse/ingestion/universal/operators/filter_content.py +69 -0
  415. contextsynapse/ingestion/universal/operators/index_bm25.py +52 -0
  416. contextsynapse/ingestion/universal/operators/infer_domains.py +69 -0
  417. contextsynapse/ingestion/universal/operators/link_cross_reference.py +210 -0
  418. contextsynapse/ingestion/universal/operators/parse_records.py +66 -0
  419. contextsynapse/ingestion/universal/operators/resolve_entities.py +261 -0
  420. contextsynapse/ingestion/universal/operators/scanners/__init__.py +16 -0
  421. contextsynapse/ingestion/universal/operators/scanners/edges.py +173 -0
  422. contextsynapse/ingestion/universal/operators/scanners/git.py +187 -0
  423. contextsynapse/ingestion/universal/operators/scanners/github.py +135 -0
  424. contextsynapse/ingestion/universal/operators/scanners/github_api.py +509 -0
  425. contextsynapse/ingestion/universal/operators/scanners/helpers.py +67 -0
  426. contextsynapse/ingestion/universal/operators/scanners/jira.py +65 -0
  427. contextsynapse/ingestion/universal/operators/scanners/llm_enrichment.py +82 -0
  428. contextsynapse/ingestion/universal/operators/scanners/repo.py +149 -0
  429. contextsynapse/ingestion/universal/operators/sdlc_scan.py +213 -0
  430. contextsynapse/ingestion/universal/operators/spec_scanner.py +149 -0
  431. contextsynapse/ingestion/universal/operators/store_documents.py +72 -0
  432. contextsynapse/ingestion/universal/operators/synthesize_cu.py +180 -0
  433. contextsynapse/ingestion/universal/operators/validate_gate.py +65 -0
  434. contextsynapse/ingestion/universal/parsers/__init__.py +5 -0
  435. contextsynapse/ingestion/universal/parsers/record_parser.py +203 -0
  436. contextsynapse/ingestion/universal/parsers/turn_parser.py +329 -0
  437. contextsynapse/ingestion/universal/shared_signals.py +161 -0
  438. contextsynapse/ingestion/universal/stage_executor.py +219 -0
  439. contextsynapse/ingestion/web_crawler.py +265 -0
  440. contextsynapse/intelligence/__init__.py +53 -0
  441. contextsynapse/intelligence/alerts.py +198 -0
  442. contextsynapse/intelligence/collector.py +655 -0
  443. contextsynapse/intelligence/comparison.py +198 -0
  444. contextsynapse/intelligence/config.py +56 -0
  445. contextsynapse/intelligence/conflict_detector.py +182 -0
  446. contextsynapse/intelligence/context_gaps.py +192 -0
  447. contextsynapse/intelligence/context_radar.py +135 -0
  448. contextsynapse/intelligence/correlation_engine.py +416 -0
  449. contextsynapse/intelligence/correlation_template.py +120 -0
  450. contextsynapse/intelligence/dedup.py +296 -0
  451. contextsynapse/intelligence/event_bus.py +91 -0
  452. contextsynapse/intelligence/feedback_loop.py +101 -0
  453. contextsynapse/intelligence/freshness.py +307 -0
  454. contextsynapse/intelligence/fundamental_feed.py +551 -0
  455. contextsynapse/intelligence/geo_data.py +170 -0
  456. contextsynapse/intelligence/geo_reference.py +112 -0
  457. contextsynapse/intelligence/graph_fusion.py +465 -0
  458. contextsynapse/intelligence/heuristics.py +176 -0
  459. contextsynapse/intelligence/impact_tracker.py +336 -0
  460. contextsynapse/intelligence/macro_indicators.py +319 -0
  461. contextsynapse/intelligence/ocr.py +104 -0
  462. contextsynapse/intelligence/persistence.py +267 -0
  463. contextsynapse/intelligence/pipeline_health.py +130 -0
  464. contextsynapse/intelligence/price_feed.py +221 -0
  465. contextsynapse/intelligence/rate_limiter.py +130 -0
  466. contextsynapse/intelligence/reactive.py +411 -0
  467. contextsynapse/intelligence/runtime_context.py +390 -0
  468. contextsynapse/intelligence/sentiment_decay.py +426 -0
  469. contextsynapse/intelligence/session.py +277 -0
  470. contextsynapse/intelligence/signal_hierarchy.py +438 -0
  471. contextsynapse/intelligence/signals.py +185 -0
  472. contextsynapse/intelligence/source_watcher.py +132 -0
  473. contextsynapse/intelligence/table_enricher.py +256 -0
  474. contextsynapse/intelligence/tagger.py +319 -0
  475. contextsynapse/intelligence/timeseries.py +209 -0
  476. contextsynapse/intelligence/tracker.py +318 -0
  477. contextsynapse/intelligence/watchdog.py +392 -0
  478. contextsynapse/intelligence/watchdog_manager.py +263 -0
  479. contextsynapse/intelligence/webhook_receiver.py +165 -0
  480. contextsynapse/intelligence/weight_learner.py +234 -0
  481. contextsynapse/llm/__init__.py +8 -0
  482. contextsynapse/llm/client.py +348 -0
  483. contextsynapse/llm/rate_limiter.py +166 -0
  484. contextsynapse/logging_config.py +2 -0
  485. contextsynapse/marketplace/__init__.py +7 -0
  486. contextsynapse/marketplace/registry.py +173 -0
  487. contextsynapse/mcp/__init__.py +11 -0
  488. contextsynapse/mcp/__main__.py +4 -0
  489. contextsynapse/mcp/auth_middleware.py +212 -0
  490. contextsynapse/mcp/connection_pool.py +128 -0
  491. contextsynapse/mcp/server.py +986 -0
  492. contextsynapse/mcp/session_context.py +67 -0
  493. contextsynapse/memory/__init__.py +34 -0
  494. contextsynapse/memory/engine.py +475 -0
  495. contextsynapse/memory/temporal.py +327 -0
  496. contextsynapse/metadata/__init__.py +12 -0
  497. contextsynapse/metadata/metadata_db.py +272 -0
  498. contextsynapse/metadata/metadata_query_tool.py +300 -0
  499. contextsynapse/metadata/metadata_tracker.py +425 -0
  500. contextsynapse/models/__init__.py +27 -0
  501. contextsynapse/models/embedding_service.py +360 -0
  502. contextsynapse/models/model_registry.py +228 -0
  503. contextsynapse/multiworker.py +2 -0
  504. contextsynapse/pipelines/__init__.py +1 -0
  505. contextsynapse/pipelines/executor.py +851 -0
  506. contextsynapse/pipelines/filters.py +189 -0
  507. contextsynapse/pipelines/models.py +176 -0
  508. contextsynapse/pipelines/scheduler.py +268 -0
  509. contextsynapse/playground/__init__.py +1 -0
  510. contextsynapse/playground/templates.py +85 -0
  511. contextsynapse/plugins/__init__.py +43 -0
  512. contextsynapse/plugins/api.py +66 -0
  513. contextsynapse/plugins/base.py +350 -0
  514. contextsynapse/plugins/domain_interface.py +508 -0
  515. contextsynapse/plugins/domain_loader.py +241 -0
  516. contextsynapse/plugins/loader.py +148 -0
  517. contextsynapse/plugins/registry.py +265 -0
  518. contextsynapse/project/__init__.py +11 -0
  519. contextsynapse/project/code_context.py +807 -0
  520. contextsynapse/project/connectors/__init__.py +4 -0
  521. contextsynapse/project/connectors/github_connector.py +113 -0
  522. contextsynapse/project/connectors/jira_connector.py +160 -0
  523. contextsynapse/project/graph.py +1023 -0
  524. contextsynapse/project/project_context.py +545 -0
  525. contextsynapse/project/scanner_registry.py +61 -0
  526. contextsynapse/project/schema_manager.py +322 -0
  527. contextsynapse/project/schemas/knowledge.yaml +59 -0
  528. contextsynapse/project/schemas/sdlc.yaml +369 -0
  529. contextsynapse/project/sdlc_schema.py +172 -0
  530. contextsynapse/project/spec_parser.py +85 -0
  531. contextsynapse/project/task_lock.py +181 -0
  532. contextsynapse/project/task_stream.py +203 -0
  533. contextsynapse/project/team_tools.py +537 -0
  534. contextsynapse/py.typed +0 -0
  535. contextsynapse/realtime/__init__.py +6 -0
  536. contextsynapse/realtime/collaboration.py +147 -0
  537. contextsynapse/rules/__init__.py +10 -0
  538. contextsynapse/rules/engine.py +116 -0
  539. contextsynapse/rules/evaluators/__init__.py +6 -0
  540. contextsynapse/rules/evaluators/aggregate.py +205 -0
  541. contextsynapse/rules/evaluators/custom.py +170 -0
  542. contextsynapse/rules/evaluators/pre_action.py +125 -0
  543. contextsynapse/rules/evaluators/temporal.py +129 -0
  544. contextsynapse/rules/loader.py +52 -0
  545. contextsynapse/rules/models.py +124 -0
  546. contextsynapse/rules/universal.py +393 -0
  547. contextsynapse/scheduler.py +229 -0
  548. contextsynapse/schema/__init__.py +42 -0
  549. contextsynapse/schema/compiler.py +252 -0
  550. contextsynapse/schema/composer.py +158 -0
  551. contextsynapse/schema/dedup.py +73 -0
  552. contextsynapse/schema/derivation.py +172 -0
  553. contextsynapse/schema/sdl.py +326 -0
  554. contextsynapse/schema/validation_gate.py +124 -0
  555. contextsynapse/sdk/__init__.py +63 -0
  556. contextsynapse/sdk/agent.py +214 -0
  557. contextsynapse/sdk/client.py +269 -0
  558. contextsynapse/sdk/exceptions.py +23 -0
  559. contextsynapse/sdk/models.py +78 -0
  560. contextsynapse/sdk/session.py +408 -0
  561. contextsynapse/sdk/tracked.py +218 -0
  562. contextsynapse/sdk/worker.py +266 -0
  563. contextsynapse/sdk/workspace_client.py +98 -0
  564. contextsynapse/sdk/ws.py +41 -0
  565. contextsynapse/search/__init__.py +20 -0
  566. contextsynapse/search/embedding_cache.py +140 -0
  567. contextsynapse/search/enhanced_search.py +152 -0
  568. contextsynapse/search/fulltext.py +130 -0
  569. contextsynapse/search/graph_search.py +634 -0
  570. contextsynapse/search/lmdb_index.py +1218 -0
  571. contextsynapse/search/nl_to_aiql.py +199 -0
  572. contextsynapse/search/rag.py +1346 -0
  573. contextsynapse/search/rag_cache.py +167 -0
  574. contextsynapse/search/reasoning_chain.py +231 -0
  575. contextsynapse/search/redis_search.py +207 -0
  576. contextsynapse/search/retrieval_quality.py +314 -0
  577. contextsynapse/search/semantic_query.py +455 -0
  578. contextsynapse/search/text_resolver.py +159 -0
  579. contextsynapse/search/whoosh_search.py +373 -0
  580. contextsynapse/security/__init__.py +17 -0
  581. contextsynapse/security/audit.py +297 -0
  582. contextsynapse/security/audit_logger.py +97 -0
  583. contextsynapse/security/audit_trail.py +270 -0
  584. contextsynapse/security/auth_unified.py +284 -0
  585. contextsynapse/security/auto_tagger.py +230 -0
  586. contextsynapse/security/data_security.py +218 -0
  587. contextsynapse/security/encryption.py +207 -0
  588. contextsynapse/security/identity.py +447 -0
  589. contextsynapse/security/jwt_identity.py +59 -0
  590. contextsynapse/security/middleware.py +336 -0
  591. contextsynapse/security/pii.py +233 -0
  592. contextsynapse/security/rbac.py +96 -0
  593. contextsynapse/security/rls.py +205 -0
  594. contextsynapse/security/sanitize.py +233 -0
  595. contextsynapse/security/scoped_encryption.py +131 -0
  596. contextsynapse/security/tenant.py +401 -0
  597. contextsynapse/shield/__init__.py +21 -0
  598. contextsynapse/shield/anomaly.py +56 -0
  599. contextsynapse/shield/permissions.py +99 -0
  600. contextsynapse/shield/profile.py +103 -0
  601. contextsynapse/shield/shield.py +108 -0
  602. contextsynapse/shield/trust_engine.py +69 -0
  603. contextsynapse/skills/__init__.py +14 -0
  604. contextsynapse/skills/engine.py +233 -0
  605. contextsynapse/skills/loader.py +59 -0
  606. contextsynapse/skills/models.py +139 -0
  607. contextsynapse/storage/__init__.py +7 -0
  608. contextsynapse/storage/cache.py +208 -0
  609. contextsynapse/storage/columnar_store_v2.py +178 -0
  610. contextsynapse/storage/csr_graph_storage.py +587 -0
  611. contextsynapse/storage/csr_store.py +388 -0
  612. contextsynapse/storage/document_store.py +491 -0
  613. contextsynapse/storage/hnsw_index.py +294 -0
  614. contextsynapse/storage/hybrid_store_v2.py +278 -0
  615. contextsynapse/storage/lmdb_graph_storage.py +366 -0
  616. contextsynapse/storage/migration.py +279 -0
  617. contextsynapse/storage/namespace_store.py +346 -0
  618. contextsynapse/storage/pure_graph_storage.py +185 -0
  619. contextsynapse/storage/redis_graph_adapter.py +186 -0
  620. contextsynapse/storage/redis_graph_storage.py +375 -0
  621. contextsynapse/storage/router/__init__.py +29 -0
  622. contextsynapse/storage/router/duckdb_store.py +398 -0
  623. contextsynapse/storage/router/factory.py +83 -0
  624. contextsynapse/storage/router/interface.py +125 -0
  625. contextsynapse/storage/router/postgres_store.py +339 -0
  626. contextsynapse/storage/router/resolver.py +192 -0
  627. contextsynapse/storage/router/router.py +300 -0
  628. contextsynapse/storage/single_file_storage.py +391 -0
  629. contextsynapse/storage/storage_manager.py +214 -0
  630. contextsynapse/storage/storage_query_optimizer.py +314 -0
  631. contextsynapse/storage/wal.py +359 -0
  632. contextsynapse/temporal/__init__.py +0 -0
  633. contextsynapse/temporal/storage.py +106 -0
  634. contextsynapse/tests/__init__.py +11 -0
  635. contextsynapse/tests/regression/__init__.py +1 -0
  636. contextsynapse/tests/regression/add_metadata.py +395 -0
  637. contextsynapse/tests/regression/add_skip_flags.py +85 -0
  638. contextsynapse/tests/regression/cases/test_basic_create_node.yaml +20 -0
  639. contextsynapse/tests/regression/cases/test_basic_select.yaml +21 -0
  640. contextsynapse/tests/regression/cases/test_blockchain_audit_trail.yaml +37 -0
  641. contextsynapse/tests/regression/cases/test_blockchain_audit_trail_filtered.yaml +39 -0
  642. contextsynapse/tests/regression/cases/test_blockchain_blocks_range.yaml +39 -0
  643. contextsynapse/tests/regression/cases/test_blockchain_get_block.yaml +35 -0
  644. contextsynapse/tests/regression/cases/test_blockchain_latest_block.yaml +35 -0
  645. contextsynapse/tests/regression/cases/test_blockchain_length.yaml +37 -0
  646. contextsynapse/tests/regression/cases/test_blockchain_merkle_root.yaml +37 -0
  647. contextsynapse/tests/regression/cases/test_blockchain_verify.yaml +35 -0
  648. contextsynapse/tests/regression/cases/test_blockchain_verify_block.yaml +35 -0
  649. contextsynapse/tests/regression/cases/test_create_edge.yaml +28 -0
  650. contextsynapse/tests/regression/cases/test_delete_edge_basic.yaml +29 -0
  651. contextsynapse/tests/regression/cases/test_delete_node_basic.yaml +29 -0
  652. contextsynapse/tests/regression/cases/test_evaluation_mrr.yaml +40 -0
  653. contextsynapse/tests/regression/cases/test_evaluation_ndcg.yaml +40 -0
  654. contextsynapse/tests/regression/cases/test_evaluation_precision.yaml +40 -0
  655. contextsynapse/tests/regression/cases/test_evaluation_recall.yaml +40 -0
  656. contextsynapse/tests/regression/cases/test_group_by_basic.yaml +56 -0
  657. contextsynapse/tests/regression/cases/test_match_node_basic.yaml +29 -0
  658. contextsynapse/tests/regression/cases/test_match_node_with_order_limit.yaml +31 -0
  659. contextsynapse/tests/regression/cases/test_namespace_switch.yaml +18 -0
  660. contextsynapse/tests/regression/cases/test_pipeline_cascaded.yaml +61 -0
  661. contextsynapse/tests/regression/cases/test_pipeline_fixed_size_chunking.yaml +48 -0
  662. contextsynapse/tests/regression/cases/test_pipeline_gpt4_chunking.yaml +54 -0
  663. contextsynapse/tests/regression/cases/test_pipeline_hybrid_indexing.yaml +57 -0
  664. contextsynapse/tests/regression/cases/test_pipeline_large_embedding.yaml +52 -0
  665. contextsynapse/tests/regression/cases/test_pipeline_paragraph_chunking.yaml +49 -0
  666. contextsynapse/tests/regression/cases/test_pipeline_recursive_chunking.yaml +47 -0
  667. contextsynapse/tests/regression/cases/test_pipeline_run_pipeline.yaml +56 -0
  668. contextsynapse/tests/regression/cases/test_pipeline_section_based_chunking.yaml +53 -0
  669. contextsynapse/tests/regression/cases/test_pipeline_semantic_chunking.yaml +51 -0
  670. contextsynapse/tests/regression/cases/test_pipeline_with_entity_extraction.yaml +57 -0
  671. contextsynapse/tests/regression/cases/test_pipeline_with_relationships.yaml +62 -0
  672. contextsynapse/tests/regression/cases/test_rag_evaluation.yaml +38 -0
  673. contextsynapse/tests/regression/cases/test_rag_graph_bfs_with_eval.yaml +48 -0
  674. contextsynapse/tests/regression/cases/test_rag_graph_centrality.yaml +40 -0
  675. contextsynapse/tests/regression/cases/test_rag_graph_centrality_eval.yaml +35 -0
  676. contextsynapse/tests/regression/cases/test_rag_graph_community.yaml +40 -0
  677. contextsynapse/tests/regression/cases/test_rag_graph_community_eval.yaml +35 -0
  678. contextsynapse/tests/regression/cases/test_rag_graph_dfs_with_eval.yaml +35 -0
  679. contextsynapse/tests/regression/cases/test_rag_graph_multi_hop_eval.yaml +35 -0
  680. contextsynapse/tests/regression/cases/test_rag_graph_path.yaml +40 -0
  681. contextsynapse/tests/regression/cases/test_rag_graph_search_basic.yaml +38 -0
  682. contextsynapse/tests/regression/cases/test_rag_graph_search_bfs.yaml +52 -0
  683. contextsynapse/tests/regression/cases/test_rag_graph_search_dfs.yaml +41 -0
  684. contextsynapse/tests/regression/cases/test_rag_graph_search_multi_hop.yaml +43 -0
  685. contextsynapse/tests/regression/cases/test_rag_graph_search_shortest_path.yaml +41 -0
  686. contextsynapse/tests/regression/cases/test_rag_graph_search_weighted.yaml +43 -0
  687. contextsynapse/tests/regression/cases/test_rag_graph_shortest_path_eval.yaml +35 -0
  688. contextsynapse/tests/regression/cases/test_rag_graph_traversal.yaml +30 -0
  689. contextsynapse/tests/regression/cases/test_rag_graph_weighted_eval.yaml +35 -0
  690. contextsynapse/tests/regression/cases/test_rag_hybrid_search.yaml +33 -0
  691. contextsynapse/tests/regression/cases/test_rag_hybrid_with_prompt_embedding.yaml +44 -0
  692. contextsynapse/tests/regression/cases/test_rag_keyword_search.yaml +32 -0
  693. contextsynapse/tests/regression/cases/test_rag_multi_hop_graph.yaml +32 -0
  694. contextsynapse/tests/regression/cases/test_rag_path_based.yaml +34 -0
  695. contextsynapse/tests/regression/cases/test_rag_profile_based.yaml +32 -0
  696. contextsynapse/tests/regression/cases/test_rag_vector_search.yaml +32 -0
  697. contextsynapse/tests/regression/cases/test_rag_weighted_graph.yaml +34 -0
  698. contextsynapse/tests/regression/cases/test_rag_weighted_hybrid.yaml +31 -0
  699. contextsynapse/tests/regression/cases/test_semantic_boundary_chunking.yaml +57 -0
  700. contextsynapse/tests/regression/cases/test_semantic_boundary_detection.yaml +33 -0
  701. contextsynapse/tests/regression/cases/test_semantic_hash_function.yaml +30 -0
  702. contextsynapse/tests/regression/cases/test_semantic_hash_with_model.yaml +31 -0
  703. contextsynapse/tests/regression/cases/test_tcs_ingestion_fixed_size.yaml +44 -0
  704. contextsynapse/tests/regression/cases/test_tcs_ingestion_recursive.yaml +46 -0
  705. contextsynapse/tests/regression/cases/test_tcs_ingestion_section.yaml +46 -0
  706. contextsynapse/tests/regression/cases/test_tcs_ingestion_semantic.yaml +44 -0
  707. contextsynapse/tests/regression/cases/test_time_travel_as_of.yaml +35 -0
  708. contextsynapse/tests/regression/cases/test_time_travel_at_commit.yaml +33 -0
  709. contextsynapse/tests/regression/cases/test_time_travel_at_timestamp.yaml +33 -0
  710. contextsynapse/tests/regression/cases/test_time_travel_for_system_time_all.yaml +37 -0
  711. contextsynapse/tests/regression/cases/test_time_travel_versions_between.yaml +39 -0
  712. contextsynapse/tests/regression/cases/test_traverse_basic.yaml +31 -0
  713. contextsynapse/tests/regression/cases/test_traverse_max_depth.yaml +34 -0
  714. contextsynapse/tests/regression/cases/test_update_edge_basic.yaml +29 -0
  715. contextsynapse/tests/regression/evaluation_metrics.py +227 -0
  716. contextsynapse/tests/regression/regression_runner.py +1081 -0
  717. contextsynapse/tests/regression/setup_regression_data.py +212 -0
  718. contextsynapse/tests/regression/test_basic_only.py +84 -0
  719. contextsynapse/tests/regression/test_namespace_creation.py +29 -0
  720. contextsynapse/tests/regression/test_single_fix.py +61 -0
  721. contextsynapse/tests/regression/test_single_query.py +20 -0
  722. contextsynapse/tests/regression/validate_test_syntax.py +179 -0
  723. contextsynapse/tests/test_advanced_features.py +347 -0
  724. contextsynapse/tests/test_aiql_summary.py +283 -0
  725. contextsynapse/tests/test_all_aiql_queries.py +948 -0
  726. contextsynapse/tests/test_all_create_queries.py +213 -0
  727. contextsynapse/tests/test_api_vs_native.py +280 -0
  728. contextsynapse/tests/test_chunking_dedup.py +199 -0
  729. contextsynapse/tests/test_connectors.py +171 -0
  730. contextsynapse/tests/test_edges_and_match.py +202 -0
  731. contextsynapse/tests/test_extraction_pipeline_integration.py +581 -0
  732. contextsynapse/tests/test_name_property.py +224 -0
  733. contextsynapse/tests/test_node_edge_metadata.py +250 -0
  734. contextsynapse/tests/test_playground_conn.py +114 -0
  735. contextsynapse/tests/test_query_cache.py +213 -0
  736. contextsynapse/tests/test_source_policy_auth.py +348 -0
  737. contextsynapse/tests/test_temporal_and_domain.py +224 -0
  738. contextsynapse/tests/test_time_travel_e2e.py +571 -0
  739. contextsynapse/tests/test_traversal_and_aggregation.py +275 -0
  740. contextsynapse/tools/__init__.py +36 -0
  741. contextsynapse/tools/a2a_tools.py +106 -0
  742. contextsynapse/tools/cognition_tools.py +118 -0
  743. contextsynapse/tools/context.py +905 -0
  744. contextsynapse/tools/formatting.py +92 -0
  745. contextsynapse/tools/gateway_tools.py +186 -0
  746. contextsynapse/tools/graph.py +1700 -0
  747. contextsynapse/tools/intelligence.py +235 -0
  748. contextsynapse/tools/memory.py +118 -0
  749. contextsynapse/tools/registry.py +373 -0
  750. contextsynapse/tools/shield_tools.py +159 -0
  751. contextsynapse/tools/tasks.py +341 -0
  752. contextsynapse/tools/workspace_tools.py +124 -0
  753. contextsynapse/utils/__init__.py +15 -0
  754. contextsynapse/utils/config_loader.py +198 -0
  755. contextsynapse/utils/env_loader.py +156 -0
  756. contextsynapse/utils/file_utils.py +288 -0
  757. contextsynapse/utils/logging_optimizer.py +119 -0
  758. contextsynapse/vector/__init__.py +30 -0
  759. contextsynapse/vector/chroma_vector_store.py +152 -0
  760. contextsynapse/vector/custom_vector_store.py +537 -0
  761. contextsynapse/vector/faiss_vector_store.py +296 -0
  762. contextsynapse/vector/qdrant_vector_store.py +163 -0
  763. contextsynapse/vector/vector_db.py +953 -0
  764. contextsynapse/vector/vector_db_manager.py +197 -0
  765. contextsynapse/workspace/__init__.py +27 -0
  766. contextsynapse/workspace/base.py +94 -0
  767. contextsynapse/workspace/git.py +228 -0
  768. contextsynapse/workspace/github.py +78 -0
  769. contextsynapse/workspace/local.py +73 -0
  770. contextsynapse/workspace/tools.py +222 -0
  771. contextsynapse-1.0.0.dist-info/METADATA +847 -0
  772. contextsynapse-1.0.0.dist-info/RECORD +776 -0
  773. contextsynapse-1.0.0.dist-info/WHEEL +5 -0
  774. contextsynapse-1.0.0.dist-info/entry_points.txt +4 -0
  775. contextsynapse-1.0.0.dist-info/licenses/LICENSE +200 -0
  776. contextsynapse-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,2095 @@
1
+ """
2
+ Stage Executor
3
+ ==============
4
+ Unified execution engine for scenario-driven pipelines.
5
+
6
+ Replaces the three disconnected pipeline systems with a single composable
7
+ stage executor that supports both run-all and step-by-step execution.
8
+
9
+ Each stage reads from PipelineContext and writes results back to it.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import json
16
+ import logging
17
+ import os
18
+ import re
19
+ import tempfile
20
+ import uuid
21
+ from dataclasses import asdict
22
+ from datetime import datetime, timezone
23
+ from pathlib import Path
24
+ from typing import Any, Callable, Dict, List, Optional, Tuple
25
+
26
+ from .pipeline_context import ClassificationResult, PipelineContext
27
+
28
+ logger = logging.getLogger(__name__)
29
+
30
+
31
+ # ---------------------------------------------------------------------------
32
+ # Helpers (reused from pipeline_runner.py)
33
+ # ---------------------------------------------------------------------------
34
+
35
+ from .chunking import chunk_text as _chunk_text # canonical implementation
36
+
37
+
38
+ def _simhash(text: str, hash_bits: int = 64) -> int:
39
+ """Compute a SimHash fingerprint for near-duplicate detection.
40
+
41
+ Uses word-level 3-grams hashed to bit vectors. Two texts with a small
42
+ Hamming distance between their SimHashes are likely near-duplicates.
43
+ """
44
+ normalised = re.sub(r"\s+", " ", text.lower().strip())
45
+ tokens = normalised.split()
46
+ if len(tokens) < 3:
47
+ tokens = list(normalised) # character-level fallback for short text
48
+
49
+ v = [0] * hash_bits
50
+ for i in range(max(1, len(tokens) - 2)):
51
+ gram = " ".join(tokens[i:i + 3])
52
+ h = int(hashlib.md5(gram.encode()).hexdigest(), 16)
53
+ for j in range(hash_bits):
54
+ if h & (1 << j):
55
+ v[j] += 1
56
+ else:
57
+ v[j] -= 1
58
+
59
+ fingerprint = 0
60
+ for j in range(hash_bits):
61
+ if v[j] > 0:
62
+ fingerprint |= (1 << j)
63
+ return fingerprint
64
+
65
+
66
+ def _hamming_distance(a: int, b: int) -> int:
67
+ """Number of differing bits between two integers."""
68
+ return bin(a ^ b).count("1")
69
+
70
+
71
+ def _content_hash(text: str) -> str:
72
+ """SHA-256 of raw text."""
73
+ return hashlib.sha256(text.encode("utf-8")).hexdigest()
74
+
75
+
76
+ def _normalised_hash(text: str) -> str:
77
+ """SHA-256 of normalised text (lowercase, collapsed whitespace)."""
78
+ normalised = re.sub(r"\s+", " ", text.lower().strip())
79
+ return hashlib.sha256(normalised.encode("utf-8")).hexdigest()
80
+
81
+
82
+ def _entity_id(entity_type: str, entity_name: str) -> str:
83
+ """Deterministic entity ID from type + canonical name."""
84
+ canonical = f"{entity_type}:{entity_name.lower().strip()}"
85
+ h = hashlib.sha256(canonical.encode()).hexdigest()[:12]
86
+ return f"entity_{h}"
87
+
88
+
89
+ def _parse_entities_fallback(text: str) -> List[Dict[str, Any]]:
90
+ """Regex fallback to extract entities from non-JSON LLM responses."""
91
+ entities = []
92
+ patterns = [
93
+ r'(\w+):\s*([^(]+)\s*\(([^)]+)\)',
94
+ r'Name:\s*([^,]+),\s*Type:\s*([^,]+)',
95
+ ]
96
+ for pattern in patterns:
97
+ matches = re.findall(pattern, text)
98
+ for match in matches:
99
+ if len(match) >= 2:
100
+ entities.append({
101
+ "name": match[1].strip() if len(match) > 1 else match[0].strip(),
102
+ "type": match[2].strip() if len(match) > 2 else "Entity",
103
+ })
104
+ return entities
105
+
106
+
107
+ # ---------------------------------------------------------------------------
108
+ # Stage Executor
109
+ # ---------------------------------------------------------------------------
110
+
111
+ class StageExecutor:
112
+ """
113
+ Executes pipeline stages against a shared PipelineContext.
114
+
115
+ Supports:
116
+ - execute_stage(name, ctx) — run one stage
117
+ - execute_next(ctx) — run the next pending stage, then pause
118
+ - execute_all(ctx) — run all remaining stages without pausing
119
+ """
120
+
121
+ def __init__(self, graph_registry=None, on_sub_step=None, aiql_executor=None):
122
+ self.graph_registry = graph_registry
123
+ self.on_sub_step = on_sub_step # Callable(run_id, stage, message, progress)
124
+ self._llm = None
125
+ self._embeddings = None
126
+ self._aiql_executor = aiql_executor
127
+
128
+ # Dispatch table: stage_name -> method
129
+ self._stage_handlers = {
130
+ "PARSE_FILE": self._stage_parse_file,
131
+ "CLASSIFY": self._stage_classify,
132
+ "CHUNK": self._stage_chunk,
133
+ "DEDUP": self._stage_dedup,
134
+ "MAP_TABULAR": self._stage_map_tabular,
135
+ "EXTRACT": self._stage_extract,
136
+ "EXTRACT_FACTS": self._stage_extract_facts,
137
+ "EMBED": self._stage_embed,
138
+ "INDEX_BM25": self._stage_index_bm25,
139
+ "STORE_VECTORS": self._stage_store_vectors,
140
+ "CANONICALIZE": self._stage_canonicalize,
141
+ "ENHANCE_GRAPH": self._stage_enhance_graph,
142
+ "PERSIST": self._stage_persist,
143
+ "IMPORT_GRAPH": self._stage_import_graph,
144
+ "VALIDATE_SCHEMA": self._stage_validate_schema,
145
+ "SDLC_SCAN": self._stage_sdlc_scan,
146
+ # SDLC sub-stages (resumable)
147
+ "SDLC_SCAN_FILES": self._stage_sdlc_scan_files,
148
+ "SDLC_SCAN_GITHUB": self._stage_sdlc_scan_github,
149
+ "SDLC_SCAN_GIT": self._stage_sdlc_scan_git,
150
+ "SDLC_SCAN_EDGES": self._stage_sdlc_scan_edges,
151
+ "SDLC_SCAN_CU": self._stage_sdlc_scan_cu,
152
+ }
153
+
154
+ # ------------------------------------------------------------------
155
+ # LLM / Embedding accessors (lazy)
156
+ # ------------------------------------------------------------------
157
+
158
+ def _get_aiql_executor(self):
159
+ """Get or lazily create an AIQLExecutor."""
160
+ if self._aiql_executor is None and self.graph_registry:
161
+ try:
162
+ from ..aiql.engine.executor import AIQLExecutor
163
+ self._aiql_executor = AIQLExecutor(graph_registry=self.graph_registry)
164
+ except Exception as e:
165
+ logger.warning("Could not init AIQLExecutor: %s", e)
166
+ return self._aiql_executor
167
+
168
+ def _get_llm(self, model: Optional[str] = None):
169
+ if self._llm is None and model:
170
+ try:
171
+ from ..llm import get_llm_client
172
+ # Support "provider:model" format (e.g. "groq:gpt-oss-120b")
173
+ provider, _, model_name = (model or "").partition(":")
174
+ if model_name:
175
+ self._llm = get_llm_client(provider=provider, model=model_name)
176
+ else:
177
+ self._llm = get_llm_client()
178
+ except ImportError:
179
+ pass
180
+ except Exception as e:
181
+ logger.debug("LLM not available: %s", e)
182
+ return self._llm
183
+
184
+ def _get_embeddings(self, model: Optional[str] = None):
185
+ if self._embeddings is None and model:
186
+ try:
187
+ from ..models.embedding_service import EmbeddingService
188
+ self._embeddings = EmbeddingService()
189
+ except ImportError:
190
+ pass
191
+ except Exception as e:
192
+ logger.debug("Embeddings not available: %s", e)
193
+ return self._embeddings
194
+
195
+ # ------------------------------------------------------------------
196
+ # Sub-step emission
197
+ # ------------------------------------------------------------------
198
+
199
+ def _emit_sub_step(self, ctx: PipelineContext, message: str, progress: float = None):
200
+ """Record a sub-step on the context and notify listeners."""
201
+ ctx.record_sub_step(message, progress)
202
+ if self.on_sub_step:
203
+ try:
204
+ self.on_sub_step(ctx.run_id, ctx.current_stage or "unknown", message, progress)
205
+ except Exception:
206
+ pass # never let sub-step emission break execution
207
+
208
+ # ------------------------------------------------------------------
209
+ # Public API
210
+ # ------------------------------------------------------------------
211
+
212
+ def execute_stage(self, stage_name: str, ctx: PipelineContext) -> PipelineContext:
213
+ """Execute a single named stage."""
214
+ handler = self._stage_handlers.get(stage_name)
215
+ if not handler:
216
+ ctx.record_stage(stage_name, "failed", error=f"Unknown stage: {stage_name}")
217
+ return ctx
218
+
219
+ ctx.status = "running"
220
+ started = datetime.now(timezone.utc).isoformat()
221
+
222
+ try:
223
+ summary = handler(ctx)
224
+ ctx.record_stage(stage_name, "completed", summary=summary or {})
225
+ except Exception as e:
226
+ logger.error("Stage %s failed: %s", stage_name, e, exc_info=True)
227
+ ctx.record_stage(stage_name, "failed", error=str(e))
228
+ ctx.status = "failed"
229
+
230
+ return ctx
231
+
232
+ def execute_next(self, ctx: PipelineContext) -> PipelineContext:
233
+ """Execute the next pending stage, then pause."""
234
+ if ctx.is_done:
235
+ ctx.status = "completed"
236
+ return ctx
237
+
238
+ stage_name = ctx.current_stage
239
+ if not stage_name:
240
+ ctx.status = "completed"
241
+ return ctx
242
+
243
+ self.execute_stage(stage_name, ctx)
244
+
245
+ # Advance pointer
246
+ ctx.advance()
247
+
248
+ if ctx.status != "failed":
249
+ if ctx.is_done:
250
+ ctx.status = "completed"
251
+ else:
252
+ ctx.status = "paused"
253
+
254
+ return ctx
255
+
256
+ def execute_all(
257
+ self,
258
+ ctx: PipelineContext,
259
+ on_stage: Optional[Callable] = None,
260
+ ) -> PipelineContext:
261
+ """Execute all remaining stages without pausing."""
262
+ while not ctx.is_done and ctx.status != "failed":
263
+ stage_name = ctx.current_stage
264
+ if on_stage:
265
+ on_stage(stage_name, "running", ctx)
266
+
267
+ self.execute_stage(stage_name, ctx)
268
+ ctx.advance()
269
+
270
+ if on_stage and ctx.stage_results:
271
+ last = ctx.stage_results[-1]
272
+ on_stage(stage_name, last.get("status", "completed"), ctx)
273
+
274
+ if ctx.status == "failed":
275
+ break
276
+
277
+ if ctx.status != "failed":
278
+ ctx.status = "completed"
279
+
280
+ return ctx
281
+
282
+ # ------------------------------------------------------------------
283
+ # PARSE_FILE — extract content from uploaded file
284
+ # ------------------------------------------------------------------
285
+
286
+ def _stage_parse_file(self, ctx: PipelineContext) -> Dict[str, Any]:
287
+ """Use ExtractorRegistry to parse the uploaded file."""
288
+ from ..extraction.registry import ExtractorRegistry, ExtractionResult
289
+
290
+ file_path = ctx.source_bytes_path
291
+ filename = ctx.source_filename or ""
292
+
293
+ # If we have text input instead of a file, create a minimal extraction result
294
+ if not file_path and ctx.source_text:
295
+ ctx.extraction_result = {
296
+ "pages": [{"page_no": 1, "text": ctx.source_text, "content": ctx.source_text}],
297
+ "tables": [],
298
+ "metadata": {"title": filename or "Text Input", "file_type": "txt"},
299
+ "errors": [],
300
+ }
301
+ return {"source": "text", "pages": 1, "tables": 0}
302
+
303
+ if not file_path or not os.path.exists(file_path):
304
+ raise FileNotFoundError(f"Source file not found: {file_path}")
305
+
306
+ self._emit_sub_step(ctx, f"Parsing file: {filename or file_path}")
307
+
308
+ # Try extension-based extractor lookup
309
+ ext = Path(filename).suffix.lower() if filename else Path(file_path).suffix.lower()
310
+
311
+ # Direct extractor mapping
312
+ extractor = None
313
+ extractor_map = {
314
+ ".xlsx": "excel", ".xls": "excel",
315
+ ".csv": "csv", ".tsv": "csv",
316
+ ".pdf": "pdf", ".docx": "docx",
317
+ ".txt": "txt", ".md": "txt",
318
+ ".html": "html", ".htm": "html",
319
+ ".zip": "chat_export",
320
+ }
321
+
322
+ extractor_type = extractor_map.get(ext)
323
+
324
+ # Auto-detect .json files — could be chat export or regular data
325
+ if ext == ".json" and not extractor_type:
326
+ try:
327
+ from ..extraction.extractors.chat_export_extractor import ChatExportExtractor
328
+ _test_ext = ChatExportExtractor(type("C", (), {"file_path": file_path})())
329
+ if _test_ext.supports_format(file_path):
330
+ extractor_type = "chat_export"
331
+ except Exception:
332
+ pass
333
+ if not extractor_type:
334
+ extractor_type = "txt" # fallback: treat JSON as text
335
+
336
+ if extractor_type:
337
+ extractor = self._create_extractor(extractor_type, file_path)
338
+
339
+ if not extractor:
340
+ # Fallback: try auto-detect from registry
341
+ try:
342
+ from ..extraction.registry import ExtractionConfig
343
+ config = ExtractionConfig()
344
+ except ImportError:
345
+ config = type("Config", (), {})()
346
+ extractor = ExtractorRegistry.auto_detect(file_path, config)
347
+
348
+ if not extractor:
349
+ raise ValueError(f"No extractor available for file type: {ext}")
350
+
351
+ self._emit_sub_step(ctx, f"Using {extractor_type or 'auto'} extractor for {ext}")
352
+
353
+ # Pass sub-step callback to extractor if supported
354
+ if hasattr(extractor, 'progress_callback'):
355
+ extractor.progress_callback = lambda msg, prog=None: self._emit_sub_step(ctx, msg, prog)
356
+
357
+ result = extractor.extract(file_path)
358
+
359
+ self._emit_sub_step(ctx, f"Extracted {len(result.pages)} pages, {len(result.tables)} tables")
360
+
361
+ # Serialize ExtractionResult to dict for PipelineContext
362
+ ctx.extraction_result = {
363
+ "pages": result.pages,
364
+ "tables": result.tables,
365
+ "metadata": result.metadata,
366
+ "images": result.images if hasattr(result, "images") else [],
367
+ "errors": result.errors,
368
+ }
369
+ ctx.tables = result.tables
370
+
371
+ return {
372
+ "file_type": ext,
373
+ "pages": len(result.pages),
374
+ "tables": len(result.tables),
375
+ "errors": len(result.errors),
376
+ }
377
+
378
+ def _create_extractor(self, extractor_type: str, file_path: str):
379
+ """Create an extractor instance by type name."""
380
+ try:
381
+ # Create a minimal config object
382
+ try:
383
+ from ..extraction.config import ExtractionConfig
384
+ config = ExtractionConfig(file_path=file_path)
385
+ except Exception:
386
+ config = type("Config", (), {
387
+ "file_path": file_path,
388
+ "pages_mode": type("Mode", (), {"value": "full"})(),
389
+ "pages_range": None,
390
+ "pages_list": None,
391
+ "detect": ["TEXT"],
392
+ "reader": "AUTO",
393
+ "parse_metadata": True,
394
+ "store_intermediate": False,
395
+ "output_dir": "contextcore_data",
396
+ "namespace": None,
397
+ "document_id": None,
398
+ "use_layout_parser": False,
399
+ "layout_detection_method": "auto",
400
+ "post_process": True,
401
+ "post_processing_config": None,
402
+ })()
403
+
404
+ if extractor_type == "excel":
405
+ from ..extraction.extractors.excel_extractor import ExcelExtractor
406
+ return ExcelExtractor(config)
407
+ elif extractor_type == "csv":
408
+ from ..extraction.extractors.csv_extractor import CsvExtractor
409
+ return CsvExtractor(config)
410
+ elif extractor_type == "pdf":
411
+ from ..extraction.extractors.pdf_extractor import PdfExtractor
412
+ return PdfExtractor(config)
413
+ elif extractor_type == "docx":
414
+ from ..extraction.extractors.docx_extractor import DocxExtractor
415
+ return DocxExtractor(config)
416
+ elif extractor_type == "txt":
417
+ from ..extraction.extractors.txt_extractor import TxtExtractor
418
+ return TxtExtractor(config)
419
+ elif extractor_type == "html":
420
+ from ..extraction.extractors.html_extractor import HtmlExtractor
421
+ return HtmlExtractor(config)
422
+ elif extractor_type == "chat_export":
423
+ from ..extraction.extractors.chat_export_extractor import ChatExportExtractor
424
+ return ChatExportExtractor(config)
425
+ except Exception as e:
426
+ logger.warning("Failed to create %s extractor: %s", extractor_type, e)
427
+ return None
428
+
429
+ # ------------------------------------------------------------------
430
+ # CLASSIFY — detect file type and content structure
431
+ # ------------------------------------------------------------------
432
+
433
+ def _stage_classify(self, ctx: PipelineContext) -> Dict[str, Any]:
434
+ """Run InputClassifier on extraction result."""
435
+ from .input_classifier import InputClassifier
436
+
437
+ if not ctx.extraction_result:
438
+ raise ValueError("No extraction result to classify — run PARSE_FILE first")
439
+
440
+ classifier = InputClassifier()
441
+ classification = classifier.classify(
442
+ ctx.extraction_result,
443
+ ctx.source_filename or "",
444
+ )
445
+
446
+ ctx.classification = classification.to_dict()
447
+
448
+ return {
449
+ "file_type": classification.file_type,
450
+ "structure_type": classification.structure_type,
451
+ "confidence": classification.confidence,
452
+ }
453
+
454
+ # ------------------------------------------------------------------
455
+ # CHUNK — split text content into chunks
456
+ # ------------------------------------------------------------------
457
+
458
+ def _stage_chunk(self, ctx: PipelineContext) -> Dict[str, Any]:
459
+ """Chunk text content from extraction result."""
460
+ if not ctx.extraction_result:
461
+ # Fallback: if source_text is set, create extraction_result on the fly
462
+ if ctx.source_text:
463
+ ctx.extraction_result = {
464
+ "pages": [{"page_no": 1, "text": ctx.source_text}],
465
+ "tables": [],
466
+ "metadata": {"title": ctx.source_filename or "Text Input", "file_type": "txt"},
467
+ }
468
+ else:
469
+ raise ValueError("No extraction result — run PARSE_FILE first")
470
+
471
+ pages = ctx.extraction_result.get("pages", [])
472
+ all_text = "\n\n".join(p.get("text", "") for p in pages).strip()
473
+
474
+ if not all_text:
475
+ ctx.chunks = []
476
+ return {"chunks": 0, "source": "no_text"}
477
+
478
+ max_chars = ctx.pipeline_params.get("chunk_size", 2000)
479
+ raw_chunks = _chunk_text(all_text, max_chars)
480
+
481
+ ctx.chunks = [
482
+ {
483
+ "id": str(uuid.uuid4()),
484
+ "index": i,
485
+ "text": chunk,
486
+ "char_count": len(chunk),
487
+ }
488
+ for i, chunk in enumerate(raw_chunks)
489
+ ]
490
+
491
+ return {"chunks": len(ctx.chunks), "total_chars": len(all_text)}
492
+
493
+ # ------------------------------------------------------------------
494
+ # DEDUP — content and semantic hash deduplication
495
+ # ------------------------------------------------------------------
496
+
497
+ def _stage_dedup(self, ctx: PipelineContext) -> Dict[str, Any]:
498
+ """Deduplicate chunks using content hashing and SimHash fingerprinting.
499
+
500
+ Two layers:
501
+ 1. **Exact hash** — SHA-256 of raw and normalised text. Catches exact
502
+ duplicates and trivial reformattings (whitespace, case changes).
503
+ 2. **SimHash** — 64-bit fingerprint based on word 3-grams. Catches
504
+ near-duplicates (paraphrases, minor edits). Configurable Hamming
505
+ distance threshold (default: 3 bits out of 64).
506
+
507
+ Checks both within the current batch (cross-chunk) and against
508
+ existing nodes already in the target graph.
509
+
510
+ Pipeline params:
511
+ dedup_enabled: bool (default True)
512
+ dedup_hamming_threshold: int (default 6 — bits that may differ out of 64)
513
+ dedup_skip_short: int (default 100 — skip chunks < N chars)
514
+ """
515
+ if not ctx.chunks:
516
+ return {"chunks": 0, "skipped": 0, "reason": "no_chunks"}
517
+
518
+ enabled = ctx.pipeline_params.get("dedup_enabled", True)
519
+ if not enabled:
520
+ return {"chunks": len(ctx.chunks), "skipped": 0, "reason": "disabled"}
521
+
522
+ hamming_threshold = ctx.pipeline_params.get("dedup_hamming_threshold", 6)
523
+ skip_short = ctx.pipeline_params.get("dedup_skip_short", 100)
524
+
525
+ # Collect existing hashes from the target graph
526
+ existing_hashes = set() # SHA-256 content hashes
527
+ existing_norm_hashes = set() # SHA-256 normalised hashes
528
+ existing_simhashes = [] # (node_id, simhash_int) pairs
529
+
530
+ graph = None
531
+ ns = getattr(ctx, "graph_namespace", "") or getattr(ctx, "namespace", "")
532
+ if self.graph_registry and ns:
533
+ try:
534
+ graph = self.graph_registry.get_graph(ns, load_if_missing=True)
535
+ if graph is None and ":" in ns:
536
+ graph = self.graph_registry.get_graph(ns.split(":", 1)[1], load_if_missing=True)
537
+ except Exception:
538
+ pass
539
+
540
+ if graph:
541
+ for node in graph.get_all_nodes():
542
+ props = node.properties if hasattr(node, "properties") else {}
543
+ ch = props.get("content_hash")
544
+ if ch:
545
+ existing_hashes.add(ch)
546
+ nh = props.get("normalised_hash")
547
+ if nh:
548
+ existing_norm_hashes.add(nh)
549
+ sh = props.get("simhash")
550
+ if sh is not None:
551
+ try:
552
+ existing_simhashes.append((node.id, int(sh)))
553
+ except (ValueError, TypeError):
554
+ pass
555
+
556
+ # Dedup within current batch + against existing graph
557
+ batch_hashes = set()
558
+ batch_norm_hashes = set()
559
+ batch_simhashes = [] # (chunk_id, simhash_int)
560
+
561
+ kept = []
562
+ skipped_exact = 0
563
+ skipped_near = 0
564
+ skipped_short = 0
565
+
566
+ for chunk in ctx.chunks:
567
+ text = chunk.get("text", "")
568
+
569
+ # Skip very short chunks (headers, footers, etc.)
570
+ if len(text) < skip_short:
571
+ chunk["content_hash"] = _content_hash(text)
572
+ chunk["normalised_hash"] = _normalised_hash(text)
573
+ chunk["simhash"] = str(_simhash(text))
574
+ kept.append(chunk)
575
+ skipped_short += 1 # counted but still kept
576
+ continue
577
+
578
+ ch = _content_hash(text)
579
+ nh = _normalised_hash(text)
580
+ sh = _simhash(text)
581
+
582
+ # Layer 1: exact hash check
583
+ if ch in existing_hashes or ch in batch_hashes:
584
+ skipped_exact += 1
585
+ logger.info("[DEDUP] Exact duplicate skipped (chunk %s, %d chars)",
586
+ chunk.get("index", "?"), len(text))
587
+ continue
588
+
589
+ if nh in existing_norm_hashes or nh in batch_norm_hashes:
590
+ skipped_exact += 1
591
+ logger.info("[DEDUP] Normalised duplicate skipped (chunk %s, %d chars)",
592
+ chunk.get("index", "?"), len(text))
593
+ continue
594
+
595
+ # Layer 2: SimHash near-duplicate check
596
+ is_near_dup = False
597
+ for _, existing_sh in existing_simhashes:
598
+ if _hamming_distance(sh, existing_sh) <= hamming_threshold:
599
+ is_near_dup = True
600
+ break
601
+ if not is_near_dup:
602
+ for _, batch_sh in batch_simhashes:
603
+ if _hamming_distance(sh, batch_sh) <= hamming_threshold:
604
+ is_near_dup = True
605
+ break
606
+
607
+ if is_near_dup:
608
+ skipped_near += 1
609
+ logger.info("[DEDUP] Near-duplicate skipped (chunk %s, hamming ≤ %d)",
610
+ chunk.get("index", "?"), hamming_threshold)
611
+ continue
612
+
613
+ # Not a duplicate — stamp hashes and keep
614
+ chunk["content_hash"] = ch
615
+ chunk["normalised_hash"] = nh
616
+ chunk["simhash"] = str(sh)
617
+ batch_hashes.add(ch)
618
+ batch_norm_hashes.add(nh)
619
+ batch_simhashes.append((chunk["id"], sh))
620
+ kept.append(chunk)
621
+
622
+ # Re-index kept chunks
623
+ for i, chunk in enumerate(kept):
624
+ chunk["index"] = i
625
+
626
+ ctx.chunks = kept
627
+
628
+ total_skipped = skipped_exact + skipped_near
629
+ if total_skipped:
630
+ logger.info("[DEDUP] Kept %d/%d chunks (exact=%d, near=%d skipped)",
631
+ len(kept), len(kept) + total_skipped, skipped_exact, skipped_near)
632
+
633
+ return {
634
+ "chunks_before": len(kept) + total_skipped,
635
+ "chunks_after": len(kept),
636
+ "skipped_exact": skipped_exact,
637
+ "skipped_near_duplicate": skipped_near,
638
+ "short_chunks": skipped_short,
639
+ "hamming_threshold": hamming_threshold,
640
+ }
641
+
642
+ # ------------------------------------------------------------------
643
+ # MAP_TABULAR — convert structured tables to nodes/edges
644
+ # ------------------------------------------------------------------
645
+
646
+ def _stage_map_tabular(self, ctx: PipelineContext) -> Dict[str, Any]:
647
+ """Map tabular data to graph nodes and edges."""
648
+ from .stages.tabular_to_graph import TabularToGraphMapper
649
+
650
+ tables = ctx.tables or (ctx.extraction_result or {}).get("tables", [])
651
+ if not tables:
652
+ return {"nodes": 0, "edges": 0, "reason": "no_tables"}
653
+
654
+ # Need classification for column role mapping
655
+ classification = ClassificationResult.from_dict(ctx.classification) if ctx.classification else None
656
+ if not classification:
657
+ # Quick classify if not done yet
658
+ from .input_classifier import InputClassifier
659
+ classifier = InputClassifier()
660
+ classification = classifier.classify(
661
+ ctx.extraction_result or {"tables": tables},
662
+ ctx.source_filename or "",
663
+ )
664
+ ctx.classification = classification.to_dict()
665
+
666
+ mapper = TabularToGraphMapper()
667
+ nodes, edges = mapper.map(tables, classification)
668
+
669
+ # Merge with any existing nodes/edges (from other stages)
670
+ ctx.nodes.extend(nodes)
671
+ ctx.edges.extend(edges)
672
+
673
+ return {"nodes": len(nodes), "edges": len(edges)}
674
+
675
+ # ------------------------------------------------------------------
676
+ # EXTRACT — LLM entity & relationship extraction from chunks
677
+ # ------------------------------------------------------------------
678
+
679
+ def _stage_extract(self, ctx: PipelineContext) -> Dict[str, Any]:
680
+ """Extract entities and relationships from text chunks using LLM."""
681
+ llm_model = ctx.pipeline_params.get("llm_model")
682
+ llm = self._get_llm(llm_model)
683
+
684
+ if not llm or not llm_model:
685
+ return {"entities": 0, "relationships": 0, "reason": "no_llm_configured"}
686
+
687
+ chunks = ctx.chunks
688
+ if not chunks:
689
+ return {"entities": 0, "relationships": 0, "reason": "no_chunks"}
690
+
691
+ entity_map: Dict[str, Dict] = {}
692
+ all_relationships = []
693
+ all_facts = []
694
+ chunk_entity_links = []
695
+
696
+ schema = ctx.pipeline_params.get("schema")
697
+
698
+ # Parallel chunk extraction — up to 3 concurrent LLM calls per document
699
+ from concurrent.futures import ThreadPoolExecutor, as_completed
700
+ import os
701
+ _extract_workers = int(os.environ.get("CONTEXTSYNAPSE_EXTRACT_WORKERS") or os.environ.get("AICONTEXTDB_EXTRACT_WORKERS", "3"))
702
+
703
+ def _extract_one(chunk):
704
+ text = chunk.get("text", "")
705
+ if not text.strip():
706
+ return chunk, {"entities": [], "relationships": []}
707
+ return chunk, self._extract_from_chunk(llm, llm_model, text, schema)
708
+
709
+ valid_chunks = [c for c in chunks if c.get("text", "").strip()]
710
+ chunk_results = []
711
+ with ThreadPoolExecutor(max_workers=min(_extract_workers, len(valid_chunks) or 1)) as pool:
712
+ futures = {pool.submit(_extract_one, c): c for c in valid_chunks}
713
+ for future in as_completed(futures):
714
+ try:
715
+ chunk_results.append(future.result())
716
+ except Exception as e:
717
+ logger.warning("[EXTRACT] Chunk extraction failed: %s", e)
718
+
719
+ for chunk, result in chunk_results:
720
+ for ent in result.get("entities", []):
721
+ # Accept "name" or "title" as the entity name (SDLC types use title)
722
+ name = (ent.get("name") or ent.get("title", "")).strip()
723
+ etype = ent.get("type", "Entity").strip()
724
+ if not name:
725
+ continue
726
+ key = f"{etype}:{name.lower()}"
727
+ # Collect all properties from the LLM response
728
+ extra_props = ent.get("properties", {})
729
+ if isinstance(extra_props, dict):
730
+ # Also grab top-level keys that aren't metadata
731
+ for k, v in ent.items():
732
+ if k not in ("name", "title", "type", "properties", "description") and v:
733
+ extra_props[k] = v
734
+ if key not in entity_map:
735
+ entity_map[key] = {
736
+ "name": name,
737
+ "type": etype,
738
+ "description": ent.get("description", ""),
739
+ "properties": extra_props,
740
+ "mention_count": 0,
741
+ }
742
+ else:
743
+ # Merge properties from subsequent mentions
744
+ for pk, pv in extra_props.items():
745
+ if pv and not entity_map[key]["properties"].get(pk):
746
+ entity_map[key]["properties"][pk] = pv
747
+ entity_map[key]["mention_count"] += 1
748
+ chunk_entity_links.append((chunk["id"], key))
749
+
750
+ for rel in result.get("relationships", []):
751
+ src = rel.get("source", "").strip()
752
+ tgt = rel.get("target", "").strip()
753
+ # Accept both "relation" and "type" keys (schema prompt uses "type")
754
+ rtype = (rel.get("type") or rel.get("relation", "RELATED_TO")).strip()
755
+ if src and tgt:
756
+ all_relationships.append({
757
+ "source_name": src,
758
+ "target_name": tgt,
759
+ "relation": rtype,
760
+ "properties": rel.get("properties", {}),
761
+ })
762
+
763
+ # Collect facts from schema-guided extraction
764
+ for fact in result.get("facts", []):
765
+ ftype = fact.get("type", "Fact")
766
+ statement = fact.get("statement", "")
767
+ if not statement:
768
+ continue
769
+ fact_props = {k: v for k, v in fact.items()
770
+ if k not in ("type",) and v}
771
+ fact_props["extraction_method"] = "llm_extraction"
772
+ fact_props["source"] = ctx.source_url or ctx.source_filename or "llm_extraction"
773
+ all_facts.append({"type": ftype, "properties": fact_props})
774
+
775
+ # Build entity nodes with full properties
776
+ name_to_id: Dict[str, str] = {}
777
+ for key, ent in entity_map.items():
778
+ eid = _entity_id(ent["type"], ent["name"])
779
+ name_to_id[ent["name"].lower()] = eid
780
+ props = {
781
+ "name": ent["name"],
782
+ "description": ent.get("description", ""),
783
+ "mention_count": ent["mention_count"],
784
+ "extraction_method": "llm_extraction",
785
+ "source": ctx.source_url or ctx.source_filename or "llm_extraction",
786
+ }
787
+ # Merge in all extracted properties (priority, status, etc.)
788
+ props.update(ent.get("properties", {}))
789
+ ctx.nodes.append({
790
+ "id": eid,
791
+ "label": ent["type"],
792
+ "properties": props,
793
+ })
794
+
795
+ # Build chunk → entity MENTIONS edges
796
+ seen_mention = set()
797
+ for chunk_id, entity_key in chunk_entity_links:
798
+ ent = entity_map[entity_key]
799
+ eid = _entity_id(ent["type"], ent["name"])
800
+ edge_key = f"{chunk_id}:{eid}"
801
+ if edge_key in seen_mention:
802
+ continue
803
+ seen_mention.add(edge_key)
804
+ ctx.edges.append({
805
+ "id": str(uuid.uuid4()),
806
+ "source": chunk_id,
807
+ "target": eid,
808
+ "label": "MENTIONS",
809
+ "properties": {},
810
+ })
811
+
812
+ # Build relationship edges
813
+ rel_count = 0
814
+ for rel in all_relationships:
815
+ src_id = name_to_id.get(rel["source_name"].lower())
816
+ tgt_id = name_to_id.get(rel["target_name"].lower())
817
+ if src_id and tgt_id and src_id != tgt_id:
818
+ ctx.edges.append({
819
+ "id": str(uuid.uuid4()),
820
+ "source": src_id,
821
+ "target": tgt_id,
822
+ "label": rel["relation"],
823
+ "properties": rel.get("properties", {}),
824
+ })
825
+ rel_count += 1
826
+
827
+ # Build fact nodes + link to mentioning chunks
828
+ fact_count = 0
829
+ for fact in all_facts:
830
+ fid = f"fact_{hashlib.sha256(fact['properties'].get('statement', str(uuid.uuid4())).encode()).hexdigest()[:12]}"
831
+ ctx.nodes.append({
832
+ "id": fid,
833
+ "label": fact["type"],
834
+ "properties": fact["properties"],
835
+ })
836
+ fact_count += 1
837
+
838
+ return {
839
+ "entities": len(entity_map),
840
+ "relationships": rel_count,
841
+ "facts": fact_count,
842
+ "chunks_processed": len(chunks),
843
+ }
844
+
845
+ def _extract_from_chunk(
846
+ self, llm, model: str, text: str, schema=None
847
+ ) -> Dict[str, Any]:
848
+ """Extract entities and relationships from a single chunk via LLM."""
849
+
850
+ # Use schema_to_prompt if we have an ExtractionSchema object
851
+ try:
852
+ from ..extraction.schema_loader import ExtractionSchema, schema_to_prompt
853
+ if isinstance(schema, ExtractionSchema):
854
+ prompt = schema_to_prompt(schema) + f"\n{text[:3000]}"
855
+ try:
856
+ raw = llm.generate(prompt=prompt, max_tokens=4000)
857
+ if raw and raw.strip():
858
+ text_to_parse = raw.strip()
859
+ json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
860
+ if json_match:
861
+ text_to_parse = json_match.group(1).strip()
862
+ data = json.loads(text_to_parse)
863
+ return {
864
+ "entities": data.get("entities", []),
865
+ "relationships": data.get("relationships", []),
866
+ "facts": data.get("facts", []),
867
+ }
868
+ except Exception:
869
+ pass
870
+ return {"entities": [], "relationships": [], "facts": []}
871
+ except ImportError:
872
+ pass
873
+
874
+ schema_hint = ""
875
+ if schema:
876
+ node_types = list(schema.get("node_types", {}).keys()) if isinstance(schema, dict) else []
877
+ edge_types = list(schema.get("edge_types", {}).keys()) if isinstance(schema, dict) else []
878
+ if node_types:
879
+ schema_hint += f"\nAllowed entity types: {', '.join(node_types)}"
880
+ if edge_types:
881
+ schema_hint += f"\nAllowed relationship types: {', '.join(edge_types)}"
882
+ schema_hint += "\nOnly extract entities and relationships matching these types.\n"
883
+
884
+ prompt = f"""Extract all named entities and relationships from the text below.
885
+ Return ONLY valid JSON with this exact structure:
886
+ {{
887
+ "entities": [
888
+ {{"name": "entity name", "type": "Person|Organization|Location|Concept|Technology|Event|Other", "description": "brief description"}}
889
+ ],
890
+ "relationships": [
891
+ {{"source": "entity name", "target": "entity name", "relation": "WORKS_FOR|LOCATED_IN|RELATED_TO|PART_OF|CREATED_BY|USES|etc", "properties": {{}}}}
892
+ ]
893
+ }}
894
+ {schema_hint}
895
+ Rules:
896
+ - Extract ALL meaningful entities (people, organizations, places, concepts, technologies, dates, events)
897
+ - Extract relationships between the entities you found
898
+ - Use consistent entity names
899
+ - Return empty lists if no entities/relationships found
900
+ - Return ONLY the JSON, no other text
901
+
902
+ Text:
903
+ {text[:3000]}"""
904
+
905
+ for attempt in range(3):
906
+ try:
907
+ raw = llm.generate(prompt=prompt, max_tokens=4000)
908
+ if not raw or not raw.strip():
909
+ return {"entities": [], "relationships": []}
910
+
911
+ # Strip markdown fences if present
912
+ text_to_parse = raw.strip()
913
+ json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
914
+ if json_match:
915
+ text_to_parse = json_match.group(1).strip()
916
+
917
+ data = json.loads(text_to_parse)
918
+ entities = data.get("entities", [])
919
+ relationships = data.get("relationships", [])
920
+ if not isinstance(entities, list):
921
+ entities = []
922
+ if not isinstance(relationships, list):
923
+ relationships = []
924
+ return {"entities": entities, "relationships": relationships}
925
+
926
+ except json.JSONDecodeError:
927
+ return {"entities": [], "relationships": []}
928
+ except Exception as e:
929
+ # Retry on rate limit / transient errors
930
+ err_str = str(e).lower()
931
+ if attempt < 2 and ("rate" in err_str or "429" in err_str or "limit" in err_str or "timeout" in err_str):
932
+ import time
933
+ wait = (attempt + 1) * 5
934
+ logger.info("Rate limited on extract (attempt %d), retrying in %ds", attempt + 1, wait)
935
+ time.sleep(wait)
936
+ continue
937
+ logger.warning("Extraction failed for chunk: %s", e)
938
+ return {"entities": [], "relationships": []}
939
+
940
+ # ------------------------------------------------------------------
941
+ # EXTRACT_FACTS — decompose passages into atomic factual statements
942
+ # ------------------------------------------------------------------
943
+
944
+ def _stage_extract_facts(self, ctx: PipelineContext) -> Dict[str, Any]:
945
+ """Extract atomic facts from text chunks using LLM.
946
+
947
+ Each fact is a self-contained statement that can be independently
948
+ verified or searched. Facts bridge passages and entities:
949
+ Passage -[STATES]-> Fact -[MENTIONS]-> Entity.
950
+ """
951
+ llm_model = ctx.pipeline_params.get("llm_model")
952
+ llm = self._get_llm(llm_model)
953
+
954
+ if not llm or not llm_model:
955
+ return {"facts": 0, "reason": "no_llm_configured"}
956
+
957
+ chunks = ctx.chunks
958
+ if not chunks:
959
+ return {"facts": 0, "reason": "no_chunks"}
960
+
961
+ fact_map: Dict[str, Dict] = {} # canonical_key -> fact dict
962
+ chunk_fact_links: List[tuple] = [] # (chunk_id, fact_key)
963
+
964
+ total = len(chunks)
965
+ for i, chunk in enumerate(chunks):
966
+ text = chunk.get("text", "")
967
+ if not text.strip():
968
+ continue
969
+
970
+ self._emit_sub_step(ctx, f"Extracting facts from chunk {i+1}/{total}", (i + 1) / total)
971
+
972
+ result = self._extract_facts_from_chunk(llm, llm_model, text)
973
+
974
+ for fact in result.get("facts", []):
975
+ statement = fact.get("statement", "").strip()
976
+ if not statement or len(statement) < 10:
977
+ continue
978
+ ftype = fact.get("type", "fact").strip().lower()
979
+ confidence = float(fact.get("confidence", 0.8))
980
+
981
+ # Canonical key for dedup
982
+ key = f"fact:{statement.lower()[:120]}"
983
+ if key not in fact_map:
984
+ fid = f"fact_{hashlib.sha256(key.encode()).hexdigest()[:12]}"
985
+ fact_map[key] = {
986
+ "id": fid,
987
+ "statement": statement,
988
+ "type": ftype,
989
+ "confidence": confidence,
990
+ "mention_count": 0,
991
+ "entity_names": fact.get("entities", []),
992
+ }
993
+ fact_map[key]["mention_count"] += 1
994
+ chunk_fact_links.append((chunk["id"], key))
995
+
996
+ # Store facts on PipelineContext for later stages
997
+ ctx.facts = []
998
+ for key, f in fact_map.items():
999
+ ctx.facts.append(f)
1000
+
1001
+ # Also create chunk→fact links for PERSIST to use
1002
+ ctx.pipeline_params["_chunk_fact_links"] = chunk_fact_links
1003
+ ctx.pipeline_params["_fact_keys"] = {k: v["id"] for k, v in fact_map.items()}
1004
+
1005
+ self._emit_sub_step(ctx, f"Extracted {len(fact_map)} facts from {total} chunks")
1006
+
1007
+ return {
1008
+ "facts": len(fact_map),
1009
+ "chunks_processed": total,
1010
+ }
1011
+
1012
+ def _extract_facts_from_chunk(
1013
+ self, llm, model: str, text: str
1014
+ ) -> Dict[str, Any]:
1015
+ """Extract atomic facts from a single chunk via LLM."""
1016
+ prompt = f"""Decompose the following text into atomic factual statements.
1017
+ Each fact should be a single, self-contained sentence that can be independently verified.
1018
+ Also list which named entities each fact mentions.
1019
+
1020
+ Return ONLY valid JSON:
1021
+ {{
1022
+ "facts": [
1023
+ {{
1024
+ "statement": "Alice joined Acme Corp as CTO in 2024.",
1025
+ "type": "claim",
1026
+ "confidence": 0.95,
1027
+ "entities": ["Alice", "Acme Corp"]
1028
+ }}
1029
+ ]
1030
+ }}
1031
+
1032
+ Rules:
1033
+ - Each fact must be one clear, atomic statement
1034
+ - Type is one of: claim, evidence, definition, event, relationship
1035
+ - Confidence 0.0-1.0 based on how clearly the text states the fact
1036
+ - entities: list the exact entity names mentioned in that fact
1037
+ - Skip trivial/filler statements
1038
+ - Return ONLY the JSON
1039
+
1040
+ Text:
1041
+ {text[:3000]}"""
1042
+
1043
+ for attempt in range(3):
1044
+ try:
1045
+ raw = llm.generate(prompt=prompt, max_tokens=4000)
1046
+ if not raw or not raw.strip():
1047
+ logger.warning("Fact extraction: LLM returned empty response (attempt %d)", attempt + 1)
1048
+ return {"facts": []}
1049
+
1050
+ text_to_parse = raw.strip()
1051
+ json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
1052
+ if json_match:
1053
+ text_to_parse = json_match.group(1).strip()
1054
+
1055
+ # Try to find JSON object if response has extra text
1056
+ if not text_to_parse.startswith("{"):
1057
+ brace_match = re.search(r'\{[\s\S]*\}', text_to_parse)
1058
+ if brace_match:
1059
+ text_to_parse = brace_match.group(0)
1060
+
1061
+ data = json.loads(text_to_parse)
1062
+ facts = data.get("facts", [])
1063
+ if not isinstance(facts, list):
1064
+ facts = []
1065
+ logger.info("Fact extraction: got %d facts from chunk", len(facts))
1066
+ return {"facts": facts}
1067
+
1068
+ except json.JSONDecodeError as jde:
1069
+ logger.warning("Fact extraction: JSON parse error (attempt %d): %s -- raw[:200]: %s",
1070
+ attempt + 1, jde, text_to_parse[:200] if text_to_parse else "empty")
1071
+ if attempt < 2:
1072
+ continue
1073
+ return {"facts": []}
1074
+ except Exception as e:
1075
+ err_str = str(e).lower()
1076
+ if attempt < 2 and ("rate" in err_str or "429" in err_str or "limit" in err_str or "timeout" in err_str):
1077
+ import time
1078
+ wait = (attempt + 1) * 5
1079
+ logger.info("Rate limited on fact extraction (attempt %d), retrying in %ds", attempt + 1, wait)
1080
+ time.sleep(wait)
1081
+ continue
1082
+ logger.warning("Fact extraction failed for chunk: %s", e)
1083
+ return {"facts": []}
1084
+
1085
+ # ------------------------------------------------------------------
1086
+ # INDEX_BM25 — index facts into BM25 sparse search
1087
+ # ------------------------------------------------------------------
1088
+
1089
+ def _stage_index_bm25(self, ctx: PipelineContext) -> Dict[str, Any]:
1090
+ """Index extracted facts into the BM25/Whoosh search engine.
1091
+
1092
+ Each fact gets a BM25 index entry. The fact's graph node stores
1093
+ a ``bm25_ref`` pointing to the indexed document ID.
1094
+ """
1095
+ if not ctx.facts:
1096
+ return {"indexed": 0, "reason": "no_facts"}
1097
+
1098
+ try:
1099
+ from ..search.whoosh_search import WhooshSearchEngine, WhooshConfig
1100
+ except ImportError:
1101
+ return {"indexed": 0, "reason": "whoosh_not_available"}
1102
+
1103
+ graph_ns = ctx.graph_namespace or "default"
1104
+ # Sanitize namespace for use as directory name (Windows forbids ':' in paths)
1105
+ safe_ns = graph_ns.replace(":", "_")
1106
+ index_dir = os.path.join("contextcore_data", "bm25_index", safe_ns)
1107
+ engine = WhooshSearchEngine(WhooshConfig(index_dir=index_dir))
1108
+
1109
+ count = 0
1110
+ for fact in ctx.facts:
1111
+ fid = fact["id"]
1112
+ engine.index_node(
1113
+ node_id=fid,
1114
+ label="Fact",
1115
+ properties={
1116
+ "name": fact["statement"],
1117
+ "statement": fact["statement"],
1118
+ "type": fact["type"],
1119
+ "confidence": fact["confidence"],
1120
+ },
1121
+ )
1122
+ # Store the BM25 reference back on the fact
1123
+ fact["bm25_ref"] = f"bm25://{graph_ns}/{fid}"
1124
+ count += 1
1125
+
1126
+ ctx.bm25_indexed = count
1127
+ self._emit_sub_step(ctx, f"BM25-indexed {count} facts (backend={engine._backend})")
1128
+
1129
+ # Keep engine reference for potential later querying
1130
+ ctx.pipeline_params["_bm25_engine"] = engine
1131
+
1132
+ # Update BM25Index pointer node in graph
1133
+ try:
1134
+ if graph_ns and self.graph_registry:
1135
+ db = self.graph_registry.get_graph(graph_ns)
1136
+ if db:
1137
+ for n in db.get_all_nodes():
1138
+ if getattr(n, "label", "") == "BM25Index":
1139
+ n.properties["status"] = "active"
1140
+ n.properties["count"] = count
1141
+ n.properties["backend"] = engine._backend
1142
+ n.properties["index_path"] = index_dir
1143
+ db.add_node(n, write_through=True)
1144
+ break
1145
+ except Exception:
1146
+ pass
1147
+
1148
+ # Also build comprehensive indexes (keyword + full BM25 + context state)
1149
+ # This indexes ALL node types, not just facts
1150
+ try:
1151
+ from ..core.graph_intelligence import build_indexes
1152
+ graph_db = None
1153
+ if self.graph_registry:
1154
+ graph_db = self.graph_registry.get_graph_for_request(graph_ns)
1155
+ if graph_db:
1156
+ idx_result = build_indexes(graph_db, graph_ns)
1157
+ self._emit_sub_step(ctx, f"Full index build: {idx_result.get('total_ms', '?')}ms")
1158
+ except Exception as e:
1159
+ logger.debug("[INDEX_BM25] Full index build failed: %s", e)
1160
+
1161
+ return {"indexed": count, "backend": engine._backend}
1162
+
1163
+ # ------------------------------------------------------------------
1164
+ # STORE_VECTORS — store passage embeddings in vector DB
1165
+ # ------------------------------------------------------------------
1166
+
1167
+ def _stage_store_vectors(self, ctx: PipelineContext) -> Dict[str, Any]:
1168
+ """Store passage chunk embeddings in the vector DB.
1169
+
1170
+ The full passage text is stored as vector metadata. The graph
1171
+ node keeps only a ``vector_ref`` pointer and a content preview.
1172
+ """
1173
+ if not ctx.chunks:
1174
+ return {"stored": 0, "reason": "no_chunks"}
1175
+
1176
+ # Only process chunks that have embeddings
1177
+ # Collect embedded items from chunks AND nodes
1178
+ embedded_chunks = [c for c in ctx.chunks if c.get("embedding")]
1179
+ embedded_nodes = [n for n in ctx.nodes if n.get("properties", {}).get("embedding")]
1180
+ # Convert nodes to chunk-like format for storage
1181
+ for node in embedded_nodes:
1182
+ props = node.get("properties", {})
1183
+ embedded_chunks.append({
1184
+ "id": node.get("id", ""),
1185
+ "embedding": props["embedding"],
1186
+ "text": props.get("content") or props.get("description") or props.get("name", ""),
1187
+ "label": node.get("label", ""),
1188
+ })
1189
+ if not embedded_chunks:
1190
+ return {"stored": 0, "reason": "no_embeddings_on_chunks_or_nodes"}
1191
+
1192
+ try:
1193
+ from ..vector.vector_db_manager import get_vector_db_manager
1194
+ except ImportError:
1195
+ return {"stored": 0, "reason": "vector_db_not_available"}
1196
+
1197
+ graph_ns = ctx.graph_namespace or "default"
1198
+
1199
+ try:
1200
+ # Determine embedding dimension from first chunk
1201
+ sample_emb = embedded_chunks[0]["embedding"]
1202
+ if hasattr(sample_emb, "tolist"):
1203
+ sample_emb = sample_emb.tolist()
1204
+ dim = len(sample_emb)
1205
+
1206
+ manager = get_vector_db_manager()
1207
+ safe_ns = graph_ns.replace(":", "_")
1208
+ collection = f"{safe_ns}_passages"
1209
+ store = manager.create_store(
1210
+ dimension=dim, metric="cosine",
1211
+ collection_name=collection,
1212
+ )
1213
+
1214
+ # Batch all embeddings for a single add_vectors call
1215
+ ids = []
1216
+ vectors = []
1217
+ metadatas = []
1218
+ for chunk in embedded_chunks:
1219
+ cid = chunk["id"]
1220
+ embedding = chunk["embedding"]
1221
+ if hasattr(embedding, "tolist"):
1222
+ embedding = embedding.tolist()
1223
+ ids.append(cid)
1224
+ vectors.append(embedding)
1225
+ metadatas.append({
1226
+ "text": chunk["text"],
1227
+ "chunk_index": chunk.get("index", 0),
1228
+ "char_count": chunk.get("char_count", len(chunk["text"])),
1229
+ })
1230
+
1231
+ store.add_vectors(node_ids=ids, vectors=vectors, metadata_list=metadatas)
1232
+
1233
+ # Persist vector store to disk (namespace directory)
1234
+ try:
1235
+ from pathlib import Path
1236
+ vec_dir = Path(f"contextcore_data/namespaces/{graph_ns}/vectors")
1237
+ vec_dir.mkdir(parents=True, exist_ok=True)
1238
+ store.save(str(vec_dir / f"{collection}.npz"))
1239
+ logger.info("Persisted %d vectors to %s", len(ids), vec_dir / f"{collection}.npz")
1240
+ except Exception as save_err:
1241
+ logger.warning("Failed to persist vectors: %s", save_err)
1242
+
1243
+ # Replace full text on chunks with preview + ref
1244
+ for chunk in embedded_chunks:
1245
+ chunk["vector_ref"] = f"vec://{collection}/{chunk['id']}"
1246
+ chunk["content_preview"] = chunk["text"][:200]
1247
+
1248
+ ctx.vectors_stored = len(ids)
1249
+ self._emit_sub_step(ctx, f"Stored {len(ids)} passage embeddings in vector DB")
1250
+
1251
+ # Update VectorIndex pointer node in graph
1252
+ try:
1253
+ graph_ns = getattr(ctx, "graph_namespace", "") or getattr(ctx, "namespace", "")
1254
+ if graph_ns and self.graph_registry:
1255
+ db = self.graph_registry.get_graph(graph_ns)
1256
+ if db:
1257
+ for n in db.get_all_nodes():
1258
+ if getattr(n, "label", "") == "VectorIndex":
1259
+ n.properties["status"] = "active"
1260
+ n.properties["count"] = len(ids)
1261
+ n.properties["collection"] = collection
1262
+ n.properties["embedding_model"] = ctx.pipeline_params.get("_resolved_embedding_model", "")
1263
+ db.add_node(n, write_through=True)
1264
+ break
1265
+ except Exception:
1266
+ pass
1267
+
1268
+ return {"stored": len(ids), "collection": collection}
1269
+
1270
+ except Exception as e:
1271
+ logger.warning("Vector store failed: %s", e)
1272
+ # Non-fatal: embeddings stay on chunk nodes as fallback
1273
+ return {"stored": 0, "error": str(e)}
1274
+
1275
+ # ------------------------------------------------------------------
1276
+ # EMBED — generate embeddings for nodes and chunks
1277
+ # ------------------------------------------------------------------
1278
+
1279
+ def _stage_embed(self, ctx: PipelineContext) -> Dict[str, Any]:
1280
+ """Embed chunks and content-rich nodes for vector search.
1281
+
1282
+ Uses batched embedding (embed_batch) when available — sends up to
1283
+ 32 texts per API call instead of 1, reducing API calls by 30x.
1284
+ """
1285
+ embedding_model = ctx.pipeline_params.get("embedding_model")
1286
+ emb = self._get_embeddings(embedding_model)
1287
+
1288
+ if not emb or not embedding_model:
1289
+ return {"embedded": 0, "reason": "no_embedding_configured"}
1290
+
1291
+ BATCH_SIZE = 32
1292
+ has_batch = hasattr(emb, "embed_batch")
1293
+
1294
+ # Collect all texts to embed
1295
+ items = [] # list of (target_dict, key_for_embedding, text)
1296
+
1297
+ # 1. Chunks
1298
+ for chunk in ctx.chunks:
1299
+ text = (chunk.get("text", "") or chunk.get("content", ""))[:8000]
1300
+ if text.strip():
1301
+ items.append((chunk, "embedding", text, True)) # True = is_chunk
1302
+
1303
+ # 2. Content-rich nodes
1304
+ embeddable_labels = {"Passage", "TextChunk", "Fact", "Feature", "Requirement",
1305
+ "Knowledge", "Finding", "Insight", "Decision", "Document"}
1306
+ for node in ctx.nodes:
1307
+ props = node.get("properties", {})
1308
+ label = node.get("label", "")
1309
+ if props.get("embedding"):
1310
+ continue
1311
+ text = ""
1312
+ if label in embeddable_labels:
1313
+ text = (props.get("content") or props.get("description") or
1314
+ props.get("statement") or props.get("name", ""))
1315
+ elif props.get("content") or props.get("description"):
1316
+ text = props.get("content") or props.get("description", "")
1317
+ text = (text or "")[:8000]
1318
+ if text.strip() and len(text.strip()) >= 20:
1319
+ items.append((props, "embedding", text, False))
1320
+
1321
+ if not items:
1322
+ return {"embedded": 0, "reason": "nothing_to_embed"}
1323
+
1324
+ self._emit_sub_step(ctx, f"Embedding {len(items)} items (batch_size={BATCH_SIZE})")
1325
+ count = 0
1326
+ api_calls = 0
1327
+
1328
+ if has_batch:
1329
+ # Batched embedding — much faster
1330
+ for i in range(0, len(items), BATCH_SIZE):
1331
+ batch = items[i:i + BATCH_SIZE]
1332
+ texts = [item[2] for item in batch]
1333
+ try:
1334
+ resp = emb.embed_batch(texts, model_name=embedding_model)
1335
+ api_calls += 1
1336
+ embeddings = resp.get("embeddings", [])
1337
+ dim = resp.get("dimension", 0)
1338
+ for j, emb_vec in enumerate(embeddings):
1339
+ if emb_vec and j < len(batch):
1340
+ target, key, _, is_chunk = batch[j]
1341
+ target[key] = emb_vec
1342
+ target["embedding_model"] = embedding_model
1343
+ if is_chunk:
1344
+ target["embedding_dim"] = dim or len(emb_vec)
1345
+ count += 1
1346
+ except Exception as e:
1347
+ logger.warning("Batch embed failed (batch %d): %s", i // BATCH_SIZE, e)
1348
+ # Fall back to individual for this batch
1349
+ for target, key, text, is_chunk in batch:
1350
+ try:
1351
+ resp = emb.embed_text(text, model_name=embedding_model)
1352
+ api_calls += 1
1353
+ if resp.get("success") and resp.get("embedding"):
1354
+ target[key] = resp["embedding"]
1355
+ target["embedding_model"] = embedding_model
1356
+ if is_chunk:
1357
+ target["embedding_dim"] = resp.get("dimension", len(resp["embedding"]))
1358
+ count += 1
1359
+ except Exception:
1360
+ pass
1361
+ else:
1362
+ # Individual embedding (legacy fallback)
1363
+ for target, key, text, is_chunk in items:
1364
+ try:
1365
+ resp = emb.embed_text(text, model_name=embedding_model)
1366
+ api_calls += 1
1367
+ if resp.get("success") and resp.get("embedding"):
1368
+ target[key] = resp["embedding"]
1369
+ target["embedding_model"] = embedding_model
1370
+ if is_chunk:
1371
+ target["embedding_dim"] = resp.get("dimension", len(resp["embedding"]))
1372
+ count += 1
1373
+ except Exception as e:
1374
+ logger.warning("Failed to embed: %s", e)
1375
+
1376
+ ctx.embeddings_count = count
1377
+ if count > 0:
1378
+ ctx.pipeline_params["_resolved_embedding_model"] = embedding_model
1379
+ return {"embedded": count, "api_calls": api_calls, "batch_size": BATCH_SIZE, "embedding_model": embedding_model}
1380
+
1381
+ # ------------------------------------------------------------------
1382
+ # CANONICALIZE — deduplicate nodes by name similarity
1383
+ # ------------------------------------------------------------------
1384
+
1385
+ def _stage_canonicalize(self, ctx: PipelineContext) -> Dict[str, Any]:
1386
+ """Deduplicate nodes by canonical name (cross-label), merge edges."""
1387
+ if not ctx.nodes:
1388
+ return {"original": 0, "deduped": 0}
1389
+
1390
+ original_count = len(ctx.nodes)
1391
+ canonical_map: Dict[str, str] = {} # name_lower -> chosen node id
1392
+ deduped_nodes: Dict[str, Dict] = {} # node_id -> node dict
1393
+ id_remap: Dict[str, str] = {} # old_id -> canonical_id
1394
+
1395
+ entity_labels = {"Person", "Organization", "Location", "Event", "Entity"}
1396
+
1397
+ for node in ctx.nodes:
1398
+ name = node.get("properties", {}).get("name", "")
1399
+ label = node.get("label", "Entity")
1400
+
1401
+ # Non-entity nodes pass through unchanged
1402
+ if label not in entity_labels:
1403
+ deduped_nodes[node["id"]] = node
1404
+ id_remap[node["id"]] = node["id"]
1405
+ continue
1406
+
1407
+ # Canonical key: name only (allows cross-label merge like "Iran" Location + Organization)
1408
+ canonical_key = name.strip().lower()
1409
+ if not canonical_key or len(canonical_key) < 2:
1410
+ deduped_nodes[node["id"]] = node
1411
+ id_remap[node["id"]] = node["id"]
1412
+ continue
1413
+
1414
+ if canonical_key in canonical_map:
1415
+ existing_id = canonical_map[canonical_key]
1416
+ id_remap[node["id"]] = existing_id
1417
+ existing = deduped_nodes[existing_id]
1418
+ existing_mc = existing.get("properties", {}).get("mention_count", 0) or 0
1419
+ node_mc = node.get("properties", {}).get("mention_count", 0) or 0
1420
+ existing["properties"]["mention_count"] = existing_mc + node_mc
1421
+ else:
1422
+ canonical_map[canonical_key] = node["id"]
1423
+ deduped_nodes[node["id"]] = node
1424
+ id_remap[node["id"]] = node["id"]
1425
+
1426
+ # Remap edge source/target IDs and deduplicate edges
1427
+ seen_edges = set()
1428
+ deduped_edges = []
1429
+ for edge in ctx.edges:
1430
+ src = id_remap.get(edge.get("source", ""), edge.get("source", ""))
1431
+ tgt = id_remap.get(edge.get("target", ""), edge.get("target", ""))
1432
+ if src == tgt:
1433
+ continue # Skip self-loops created by merge
1434
+ edge_key = f"{src}:{tgt}:{edge.get('label', '')}"
1435
+ if edge_key in seen_edges:
1436
+ continue
1437
+ seen_edges.add(edge_key)
1438
+ deduped_edges.append({
1439
+ **edge,
1440
+ "source": src,
1441
+ "target": tgt,
1442
+ })
1443
+
1444
+ ctx.nodes = list(deduped_nodes.values())
1445
+ ctx.edges = deduped_edges
1446
+
1447
+ return {
1448
+ "original_nodes": original_count,
1449
+ "deduped_nodes": len(ctx.nodes),
1450
+ "removed": original_count - len(ctx.nodes),
1451
+ "original_edges": len(ctx.edges),
1452
+ }
1453
+
1454
+ # ------------------------------------------------------------------
1455
+ # ENHANCE_GRAPH — enrich graph with additional relationships
1456
+ # ------------------------------------------------------------------
1457
+
1458
+ def _stage_enhance_graph(self, ctx: PipelineContext) -> Dict[str, Any]:
1459
+ """Enhance graph by adding inferred edges (co-occurrence, hierarchy)."""
1460
+ if not ctx.nodes:
1461
+ return {"edges_added": 0}
1462
+
1463
+ edges_added = 0
1464
+
1465
+ # Co-occurrence: if two entities appear in the same chunk, link them
1466
+ if ctx.chunks:
1467
+ chunk_entities: Dict[str, List[str]] = {}
1468
+ for edge in ctx.edges:
1469
+ if edge.get("label") == "MENTIONS":
1470
+ chunk_id = edge["source"]
1471
+ entity_id = edge["target"]
1472
+ chunk_entities.setdefault(chunk_id, []).append(entity_id)
1473
+
1474
+ seen = set()
1475
+ for chunk_id, entities in chunk_entities.items():
1476
+ for i, e1 in enumerate(entities):
1477
+ for e2 in entities[i + 1:]:
1478
+ pair = tuple(sorted([e1, e2]))
1479
+ if pair in seen:
1480
+ continue
1481
+ seen.add(pair)
1482
+ ctx.edges.append({
1483
+ "id": str(uuid.uuid4()),
1484
+ "source": e1,
1485
+ "target": e2,
1486
+ "label": "CO_OCCURS_WITH",
1487
+ "properties": {"inferred": True},
1488
+ })
1489
+ edges_added += 1
1490
+
1491
+ return {"edges_added": edges_added}
1492
+
1493
+ # ------------------------------------------------------------------
1494
+ # PERSIST — save nodes and edges to the graph (via AIQL)
1495
+ # ------------------------------------------------------------------
1496
+
1497
+ def _stage_persist(self, ctx: PipelineContext) -> Dict[str, Any]:
1498
+ """Persist the full GraphRAG hierarchy to graph storage via AIQLExecutor.
1499
+
1500
+ Hierarchy created:
1501
+ Document -[HAS_PASSAGE]-> Passage -[STATES]-> Fact -[MENTIONS]-> Entity
1502
+ | |
1503
+ vector_ref bm25_ref
1504
+ (vector DB) (BM25 index)
1505
+
1506
+ Falls back to the simpler Document-[CONTAINS]->TextChunk layout
1507
+ when no facts are present (legacy pipelines).
1508
+
1509
+ All graph mutations go through ``AIQLExecutor.bulk_ingest()`` so they
1510
+ participate in time-travel versioning, audit, and are consistent with
1511
+ AIQL CREATE NODE / CREATE EDGE semantics.
1512
+ """
1513
+ if not self.graph_registry:
1514
+ raise ValueError("No graph_registry configured")
1515
+
1516
+ graph_ns = ctx.graph_namespace
1517
+ if not graph_ns:
1518
+ raise ValueError("No graph_namespace set on PipelineContext")
1519
+
1520
+ aiql = self._get_aiql_executor()
1521
+ if not aiql:
1522
+ raise ValueError("Could not initialize AIQLExecutor")
1523
+
1524
+ # Activate the target namespace
1525
+ aiql.active_namespace = graph_ns
1526
+
1527
+ # Collect all nodes and edges, then send through AIQL in one batch
1528
+ all_nodes: List[Dict[str, Any]] = []
1529
+ all_edges: List[Dict[str, Any]] = []
1530
+ facts_created = 0
1531
+
1532
+ has_facts = bool(ctx.facts)
1533
+ chunk_label = "Passage" if has_facts else "TextChunk"
1534
+ chunk_edge_label = "HAS_PASSAGE" if has_facts else "CONTAINS"
1535
+
1536
+ # ---- 1. Document node ----
1537
+ doc_id = None
1538
+ if ctx.chunks:
1539
+ doc_id = str(uuid.uuid4())
1540
+ title = ctx.source_filename or "Untitled"
1541
+ self._emit_sub_step(ctx, f"Building Document node: {title}")
1542
+ all_nodes.append({
1543
+ "id": doc_id,
1544
+ "label": "Document",
1545
+ "properties": {
1546
+ "name": title,
1547
+ "source": ctx.source_url or ctx.source_filename or "upload",
1548
+ "chunk_count": len(ctx.chunks),
1549
+ "fact_count": len(ctx.facts) if has_facts else 0,
1550
+ },
1551
+ })
1552
+
1553
+ # ---- 2. Passage / TextChunk nodes ----
1554
+ self._emit_sub_step(ctx, f"Building {len(ctx.chunks)} {chunk_label} nodes")
1555
+ for chunk in ctx.chunks:
1556
+ props = {
1557
+ "name": f"{title} ({chunk_label.lower()} {chunk['index'] + 1}/{len(ctx.chunks)})",
1558
+ "char_count": chunk.get("char_count", len(chunk["text"])),
1559
+ "chunk_index": chunk["index"],
1560
+ "document_id": doc_id,
1561
+ }
1562
+ if chunk.get("vector_ref"):
1563
+ # Text lives in vector store — graph stores only the pointer
1564
+ props["vector_ref"] = chunk["vector_ref"]
1565
+ else:
1566
+ # No vector store — keep full text in graph node
1567
+ props["content"] = chunk["text"]
1568
+
1569
+ if "embedding" in chunk and not chunk.get("vector_ref"):
1570
+ props["embedding"] = chunk["embedding"]
1571
+ props["embedding_model"] = chunk.get("embedding_model", "")
1572
+ props["embedding_dim"] = chunk.get("embedding_dim", 0)
1573
+
1574
+ # Dedup hashes (set by DEDUP stage)
1575
+ if chunk.get("content_hash"):
1576
+ props["content_hash"] = chunk["content_hash"]
1577
+ if chunk.get("normalised_hash"):
1578
+ props["normalised_hash"] = chunk["normalised_hash"]
1579
+ if chunk.get("simhash"):
1580
+ props["simhash"] = chunk["simhash"]
1581
+
1582
+ all_nodes.append({"id": chunk["id"], "label": chunk_label, "properties": props})
1583
+
1584
+ # Document → Passage edge
1585
+ all_edges.append({
1586
+ "id": str(uuid.uuid4()),
1587
+ "source": doc_id,
1588
+ "target": chunk["id"],
1589
+ "label": chunk_edge_label,
1590
+ "properties": {"chunk_index": chunk["index"]},
1591
+ })
1592
+
1593
+ # ---- 3. Fact nodes + Passage→Fact edges ----
1594
+ if has_facts:
1595
+ self._emit_sub_step(ctx, f"Building {len(ctx.facts)} Fact nodes")
1596
+ chunk_fact_links = ctx.pipeline_params.get("_chunk_fact_links", [])
1597
+ fact_keys = ctx.pipeline_params.get("_fact_keys", {})
1598
+
1599
+ for fact in ctx.facts:
1600
+ fid = fact["id"]
1601
+ fprops = {
1602
+ "name": fact["statement"][:80],
1603
+ "statement": fact["statement"],
1604
+ "fact_type": fact["type"],
1605
+ "confidence": fact["confidence"],
1606
+ "mention_count": fact.get("mention_count", 1),
1607
+ "extraction_method": "llm_extraction",
1608
+ "source": ctx.source_url or ctx.source_filename or "llm_extraction",
1609
+ }
1610
+ if fact.get("bm25_ref"):
1611
+ fprops["bm25_ref"] = fact["bm25_ref"]
1612
+
1613
+ all_nodes.append({"id": fid, "label": "Fact", "properties": fprops})
1614
+ facts_created += 1
1615
+
1616
+ # Passage → Fact STATES edges (deduplicated)
1617
+ seen_states = set()
1618
+ for chunk_id, fact_key in chunk_fact_links:
1619
+ fid = fact_keys.get(fact_key)
1620
+ if not fid:
1621
+ continue
1622
+ edge_key = f"{chunk_id}:{fid}"
1623
+ if edge_key in seen_states:
1624
+ continue
1625
+ seen_states.add(edge_key)
1626
+ all_edges.append({
1627
+ "id": str(uuid.uuid4()),
1628
+ "source": chunk_id,
1629
+ "target": fid,
1630
+ "label": "STATES",
1631
+ "properties": {},
1632
+ })
1633
+
1634
+ # Fact → Entity MENTIONS edges
1635
+ entity_name_to_id = {}
1636
+ for node in ctx.nodes:
1637
+ ename = node.get("properties", {}).get("name", "").lower()
1638
+ if ename:
1639
+ entity_name_to_id[ename] = node["id"]
1640
+
1641
+ seen_fact_ent = set()
1642
+ for fact in ctx.facts:
1643
+ fid = fact["id"]
1644
+ for ename in fact.get("entity_names", []):
1645
+ eid = entity_name_to_id.get(ename.lower())
1646
+ if not eid:
1647
+ continue
1648
+ edge_key = f"{fid}:{eid}"
1649
+ if edge_key in seen_fact_ent:
1650
+ continue
1651
+ seen_fact_ent.add(edge_key)
1652
+ all_edges.append({
1653
+ "id": str(uuid.uuid4()),
1654
+ "source": fid,
1655
+ "target": eid,
1656
+ "label": "MENTIONS",
1657
+ "properties": {},
1658
+ })
1659
+
1660
+ # ---- 4. Entity / tabular nodes ----
1661
+ if ctx.nodes:
1662
+ self._emit_sub_step(ctx, f"Building {len(ctx.nodes)} entity nodes")
1663
+ all_nodes.extend(ctx.nodes)
1664
+
1665
+ # ---- 5. Relationship / other edges ----
1666
+ all_edges.extend(ctx.edges)
1667
+
1668
+ # ---- 5b. Tag category + IN_CATEGORY edges ----
1669
+ category_boundary_id = ctx.pipeline_params.get("_category_boundary_id")
1670
+ if category_boundary_id:
1671
+ from ..context.boundaries import tag_nodes_with_category, build_category_edges, BOUNDARY_NODE_LABELS
1672
+ tag_nodes_with_category(all_nodes, "knowledge_base")
1673
+ cat_edges = build_category_edges(all_nodes, category_boundary_id)
1674
+ all_edges.extend(cat_edges)
1675
+ self._emit_sub_step(ctx, f"Tagged {len(all_nodes)} nodes as knowledge_base, {len(cat_edges)} IN_CATEGORY edges")
1676
+
1677
+ # ---- 6. Bulk ingest through AIQL ----
1678
+ self._emit_sub_step(ctx, f"Persisting via AIQL: {len(all_nodes)} nodes, {len(all_edges)} edges")
1679
+
1680
+ def _on_progress(msg, prog):
1681
+ self._emit_sub_step(ctx, msg, prog)
1682
+
1683
+ result = aiql.bulk_ingest(
1684
+ namespace=graph_ns,
1685
+ nodes=all_nodes,
1686
+ edges=all_edges,
1687
+ merge_existing=True,
1688
+ on_progress=_on_progress,
1689
+ )
1690
+
1691
+ nodes_created = result.get("nodes_created", 0)
1692
+ nodes_merged = result.get("nodes_merged", 0)
1693
+ edges_created = result.get("edges_created", 0)
1694
+ errors = result.get("errors", [])
1695
+
1696
+ # ---- 7. Store embedding model on graph metadata ----
1697
+ emb_model = ctx.pipeline_params.get("_resolved_embedding_model") or ctx.pipeline_params.get("embedding_model")
1698
+ if emb_model and self.graph_registry:
1699
+ meta = self.graph_registry.metadata.get(graph_ns)
1700
+ if meta:
1701
+ meta.embedding_model = emb_model
1702
+ try:
1703
+ self.graph_registry._save_metadata()
1704
+ except Exception:
1705
+ pass
1706
+
1707
+ self._emit_sub_step(ctx, f"Saved: {nodes_created} created, {nodes_merged} merged, {edges_created} edges")
1708
+
1709
+ # Build search indexes (keyword, BM25, context state) in background
1710
+ # so they're ready before any agent searches
1711
+ import threading
1712
+ def _post_persist_index_build():
1713
+ try:
1714
+ from ..core.graph_intelligence import build_indexes
1715
+ graph_db = aiql._graph if hasattr(aiql, '_graph') else None
1716
+ if not graph_db and self.graph_registry:
1717
+ graph_db = self.graph_registry.get_graph_for_request(graph_ns)
1718
+ if graph_db:
1719
+ result = build_indexes(graph_db, graph_ns)
1720
+ logger.info("[PERSIST] Post-ingest index build: %s", result)
1721
+ except Exception as e:
1722
+ logger.debug("[PERSIST] Index build failed: %s", e)
1723
+
1724
+ t = threading.Thread(target=_post_persist_index_build, daemon=True, name=f"index-{graph_ns[:12]}")
1725
+ t.start()
1726
+ self._emit_sub_step(ctx, "Index build started (background)")
1727
+
1728
+ return {
1729
+ "nodes_created": nodes_created,
1730
+ "nodes_merged": nodes_merged,
1731
+ "edges_created": edges_created,
1732
+ "facts_created": facts_created,
1733
+ "errors": errors[:10] if errors else [],
1734
+ }
1735
+
1736
+ # ------------------------------------------------------------------
1737
+ # VALIDATE_SCHEMA — validate nodes/edges against a GraphSchema
1738
+ # ------------------------------------------------------------------
1739
+
1740
+ def _stage_validate_schema(self, ctx: PipelineContext) -> Dict[str, Any]:
1741
+ """Validate extracted nodes/edges against a GraphSchema definition."""
1742
+ schema_yaml = ctx.pipeline_params.get("schema_yaml", "")
1743
+ if not schema_yaml:
1744
+ return {"skipped": True, "reason": "no_schema_provided"}
1745
+
1746
+ try:
1747
+ import yaml
1748
+ from ..schema.schema_parser import SchemaParser
1749
+ from .schema_validator import SchemaValidator
1750
+
1751
+ # Parse schema from YAML string
1752
+ import tempfile, os
1753
+ fd, tmp = tempfile.mkstemp(suffix=".yaml", prefix="schema_")
1754
+ os.close(fd)
1755
+ try:
1756
+ with open(tmp, "w", encoding="utf-8") as f:
1757
+ f.write(schema_yaml)
1758
+ parser = SchemaParser(tmp)
1759
+ schema = parser.parse()
1760
+ finally:
1761
+ os.unlink(tmp)
1762
+
1763
+ strict = ctx.pipeline_params.get("schema_strict", False)
1764
+ validator = SchemaValidator(schema)
1765
+ result = validator.validate(ctx.nodes, ctx.edges, strict=strict)
1766
+
1767
+ # Replace context nodes/edges with validated ones
1768
+ ctx.nodes = result.valid_nodes
1769
+ ctx.edges = result.valid_edges
1770
+
1771
+ return result.to_dict()
1772
+
1773
+ except Exception as e:
1774
+ logger.warning("Schema validation failed: %s", e)
1775
+ return {"error": str(e), "nodes_kept": len(ctx.nodes), "edges_kept": len(ctx.edges)}
1776
+
1777
+ # ------------------------------------------------------------------
1778
+ # IMPORT_GRAPH — import graph files (GraphML, JSON, RDF)
1779
+ # ------------------------------------------------------------------
1780
+
1781
+ def _stage_import_graph(self, ctx: PipelineContext) -> Dict[str, Any]:
1782
+ """Import a graph interchange file (GraphML, RDF/Turtle, JSON-LD, plain JSON)."""
1783
+ from .parsers import GraphParserRegistry
1784
+
1785
+ pages = (ctx.extraction_result or {}).get("pages", [])
1786
+ if not pages:
1787
+ return {"nodes": 0, "edges": 0, "reason": "no_content"}
1788
+
1789
+ text = pages[0].get("text", "")
1790
+ if not text.strip():
1791
+ return {"nodes": 0, "edges": 0, "reason": "empty_content"}
1792
+
1793
+ filename = ctx.source_filename or ""
1794
+
1795
+ try:
1796
+ registry = GraphParserRegistry()
1797
+ parsed_nodes, parsed_edges = registry.parse(text, filename=filename)
1798
+
1799
+ ctx.nodes.extend(parsed_nodes)
1800
+ ctx.edges.extend(parsed_edges)
1801
+
1802
+ return {
1803
+ "nodes": len(parsed_nodes),
1804
+ "edges": len(parsed_edges),
1805
+ "format": filename.rsplit(".", 1)[-1] if "." in filename else "unknown",
1806
+ }
1807
+ except Exception as e:
1808
+ logger.warning("Graph import failed: %s", e)
1809
+ return {"nodes": 0, "edges": 0, "error": str(e)}
1810
+
1811
+ # ------------------------------------------------------------------
1812
+ # SDLC_SCAN — scan a repo directory into SDLC-typed graph nodes
1813
+ # ------------------------------------------------------------------
1814
+
1815
+ def _stage_sdlc_scan(self, ctx: PipelineContext) -> Dict[str, Any]:
1816
+ """Scan a repo directory into SDLC-typed graph nodes."""
1817
+ from .universal.operators.sdlc_scan import SDLCScanOperator
1818
+ from .universal.operators.index_bm25 import IndexBM25Operator
1819
+ from .universal.operators.embed import EmbedOperator
1820
+ from .universal.ingest_content import Chunk
1821
+ from .universal.stage_executor import GraphContext as UniversalGraphContext
1822
+
1823
+ repo_path = ctx.source_text or ctx.source_url or ""
1824
+ if not repo_path:
1825
+ return {"nodes": 0, "edges": 0, "reason": "no_repo_path"}
1826
+
1827
+ # Get or create graph for this namespace
1828
+ graph_ns = ctx.graph_namespace or "default"
1829
+ db = None
1830
+ if self.graph_registry:
1831
+ db = self.graph_registry.get_graph(graph_ns)
1832
+ if db is None:
1833
+ db = self.graph_registry.create_graph(graph_ns)
1834
+
1835
+ if db is None:
1836
+ return {"nodes": 0, "edges": 0, "reason": "no_graph"}
1837
+
1838
+ graph_ctx = UniversalGraphContext(db=db, namespace=graph_ns)
1839
+
1840
+ # Pass config + source URL through chunk metadata
1841
+ chunk_meta = {"repo_path": repo_path}
1842
+ if ctx.source_url:
1843
+ chunk_meta["source_url"] = ctx.source_url
1844
+ if ctx.pipeline_params:
1845
+ chunk_meta["pipeline_params"] = ctx.pipeline_params
1846
+ chunks = [Chunk(content="", index=0, metadata=chunk_meta)]
1847
+
1848
+ op = SDLCScanOperator(config=ctx.pipeline_params or {})
1849
+ op.process(chunks, graph_ctx)
1850
+
1851
+ # Build full node dicts for PipelineContext (needed by downstream consumers)
1852
+ full_nodes = []
1853
+ for nid in graph_ctx.node_ids:
1854
+ node = graph_ctx.get_node(nid)
1855
+ if node is not None:
1856
+ full_nodes.append({
1857
+ "id": nid,
1858
+ "label": node.label,
1859
+ "properties": dict(node.properties),
1860
+ })
1861
+ else:
1862
+ full_nodes.append({"id": nid, "label": "", "properties": {}})
1863
+ ctx.nodes = full_nodes
1864
+
1865
+ # Edges are already committed to the graph — no need to track in ctx
1866
+ ctx.edges = []
1867
+
1868
+ # Run BM25 indexing and embedding inline (nodes already in graph)
1869
+ try:
1870
+ IndexBM25Operator().process([], graph_ctx)
1871
+ except Exception as exc:
1872
+ logger.debug("[SDLC_SCAN] BM25 indexing skipped: %s", exc)
1873
+
1874
+ try:
1875
+ EmbedOperator().process([], graph_ctx)
1876
+ except Exception as exc:
1877
+ logger.debug("[SDLC_SCAN] Embed skipped: %s", exc)
1878
+
1879
+ # Build Context Units from SDLC nodes
1880
+ try:
1881
+ from .universal.operators.synthesize_cu import SynthesizeCUOperator
1882
+ # max_cus configurable via pipeline_params, default 8
1883
+ _max_cus = (ctx.pipeline_params or {}).get("max_cus", 8)
1884
+ SynthesizeCUOperator(max_cus=_max_cus).process([], graph_ctx)
1885
+ except Exception as exc:
1886
+ logger.debug("[SDLC_SCAN] CU synthesis skipped: %s", exc)
1887
+
1888
+ return {
1889
+ "nodes": len(graph_ctx.node_ids),
1890
+ "edges": graph_ctx.edge_count,
1891
+ "errors": graph_ctx.errors,
1892
+ }
1893
+
1894
+ # ------------------------------------------------------------------
1895
+ # SDLC sub-stages (resumable pipeline)
1896
+ # ------------------------------------------------------------------
1897
+
1898
+ def _get_sdlc_graph_ctx(self, ctx: PipelineContext):
1899
+ """Get or create graph context for SDLC sub-stages."""
1900
+ from .universal.stage_executor import GraphContext as UniversalGraphContext
1901
+ graph_ns = ctx.graph_namespace or "default"
1902
+ db = None
1903
+ if self.graph_registry:
1904
+ db = self.graph_registry.get_graph(graph_ns)
1905
+ if db is None:
1906
+ db = self.graph_registry.create_graph(graph_ns)
1907
+ if db is None:
1908
+ return None, graph_ns
1909
+ return UniversalGraphContext(db=db, namespace=graph_ns), graph_ns
1910
+
1911
+ def _stage_sdlc_scan_files(self, ctx: PipelineContext) -> Dict[str, Any]:
1912
+ """Sub-stage 1: Scan repo files — README, source, tests, docs, API routes."""
1913
+ graph_ctx, graph_ns = self._get_sdlc_graph_ctx(ctx)
1914
+ if graph_ctx is None:
1915
+ return {"nodes": 0, "reason": "no_graph"}
1916
+
1917
+ repo_path = ctx.source_text or ""
1918
+ if not repo_path or not os.path.isdir(repo_path):
1919
+ return {"nodes": 0, "reason": "no_repo_path"}
1920
+
1921
+ from .universal.operators.scanners.repo import (
1922
+ scan_readme, scan_source_modules, scan_tests, scan_docs, scan_api_routes,
1923
+ )
1924
+ from .universal.operators.scanners.llm_enrichment import scan_with_llm
1925
+ from .universal.operators.scanners.jira import scan_jira_issues
1926
+ from pathlib import Path
1927
+
1928
+ rp = Path(repo_path)
1929
+ all_nodes = (
1930
+ scan_readme(rp) + scan_source_modules(rp) + scan_tests(rp)
1931
+ + scan_docs(rp) + scan_api_routes(rp)
1932
+ )
1933
+
1934
+ # Optional LLM enrichment
1935
+ skip_llm = (ctx.pipeline_params or {}).get("skip_llm", True) or os.environ.get("SDLC_SKIP_LLM")
1936
+ if not skip_llm:
1937
+ code = [n for n in all_nodes if n["label"] == "CodeModule"]
1938
+ non_code = [n for n in all_nodes if n["label"] != "CodeModule"]
1939
+ all_nodes = non_code + scan_with_llm(rp, code)
1940
+
1941
+ # Jira (API-based, no clone needed)
1942
+ all_nodes += scan_jira_issues()
1943
+
1944
+ for n in all_nodes:
1945
+ graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
1946
+
1947
+ # Store node defs in ctx for edge inference later
1948
+ ctx.nodes = [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in all_nodes]
1949
+ return {"nodes": len(all_nodes)}
1950
+
1951
+ def _stage_sdlc_scan_github(self, ctx: PipelineContext) -> Dict[str, Any]:
1952
+ """Sub-stage 2: Fetch GitHub issues, PRs, contributors via API."""
1953
+ graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
1954
+ if graph_ctx is None:
1955
+ return {"nodes": 0, "reason": "no_graph"}
1956
+
1957
+ source_url = ctx.source_url or ctx.source_text or ""
1958
+ repo_path = ctx.source_text or "."
1959
+
1960
+ from .universal.operators.scanners.github import scan_github_issues_full
1961
+ from .universal.operators.scanners.github_api import quick_scan
1962
+ from pathlib import Path
1963
+
1964
+ token = getattr(ctx, "github_token", None) or None
1965
+ # Try full GitHub scan (issues + PRs + contributors + metadata)
1966
+ nodes = (
1967
+ quick_scan(source_url, github_token=token)
1968
+ if source_url.startswith("https://")
1969
+ else scan_github_issues_full(Path(repo_path), source_url, github_token=token)
1970
+ )
1971
+
1972
+ for n in nodes:
1973
+ graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
1974
+
1975
+ ctx.nodes = (ctx.nodes or []) + [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in nodes]
1976
+ return {"nodes": len(nodes)}
1977
+
1978
+ def _stage_sdlc_scan_git(self, ctx: PipelineContext) -> Dict[str, Any]:
1979
+ """Sub-stage 3: Git history + repo metadata."""
1980
+ graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
1981
+ if graph_ctx is None:
1982
+ return {"nodes": 0, "reason": "no_graph"}
1983
+
1984
+ repo_path = ctx.source_text or ""
1985
+ if not repo_path or not os.path.isdir(repo_path):
1986
+ return {"nodes": 0, "reason": "no_repo_path"}
1987
+
1988
+ from .universal.operators.scanners.git import scan_git_history, scan_repo_metadata
1989
+ from pathlib import Path
1990
+
1991
+ rp = Path(repo_path)
1992
+ meta_nodes = scan_repo_metadata(rp)
1993
+ history_nodes, history_edges = scan_git_history(rp)
1994
+ all_nodes = meta_nodes + history_nodes
1995
+
1996
+ for n in all_nodes:
1997
+ graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
1998
+ for e in history_edges:
1999
+ graph_ctx.add_edge(source_id=e["source"], target_id=e["target"], label=e["label"])
2000
+
2001
+ ctx.nodes = (ctx.nodes or []) + [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in all_nodes]
2002
+ return {"nodes": len(all_nodes), "edges": len(history_edges)}
2003
+
2004
+ def _stage_sdlc_scan_edges(self, ctx: PipelineContext) -> Dict[str, Any]:
2005
+ """Sub-stage 4: Infer edges + BM25 index + embed."""
2006
+ graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
2007
+ if graph_ctx is None:
2008
+ return {"edges": 0, "reason": "no_graph"}
2009
+
2010
+ from .universal.operators.scanners.edges import infer_edges
2011
+ from .universal.operators.index_bm25 import IndexBM25Operator
2012
+ from .universal.operators.embed import EmbedOperator
2013
+
2014
+ # Infer edges from all nodes accumulated in ctx.nodes
2015
+ edges = infer_edges(ctx.nodes or [])
2016
+ for e in edges:
2017
+ graph_ctx.add_edge(source_id=e["source"], target_id=e["target"], label=e["label"])
2018
+
2019
+ # BM25 + embed
2020
+ # Rebuild node_ids from ctx.nodes since graph_ctx is fresh
2021
+ for n in (ctx.nodes or []):
2022
+ if n["id"] not in graph_ctx.node_ids:
2023
+ graph_ctx.node_ids.append(n["id"])
2024
+
2025
+ try:
2026
+ IndexBM25Operator().process([], graph_ctx)
2027
+ except Exception:
2028
+ pass
2029
+ try:
2030
+ EmbedOperator().process([], graph_ctx)
2031
+ except Exception:
2032
+ pass
2033
+
2034
+ return {"edges": len(edges), "indexed": len(graph_ctx.node_ids)}
2035
+
2036
+ def _stage_sdlc_scan_cu(self, ctx: PipelineContext) -> Dict[str, Any]:
2037
+ """Sub-stage 5: Build Context Units."""
2038
+ graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
2039
+ if graph_ctx is None:
2040
+ return {"cus": 0, "reason": "no_graph"}
2041
+
2042
+ from .universal.operators.synthesize_cu import SynthesizeCUOperator
2043
+
2044
+ # Rebuild node_ids
2045
+ for n in (ctx.nodes or []):
2046
+ if n["id"] not in graph_ctx.node_ids:
2047
+ graph_ctx.node_ids.append(n["id"])
2048
+
2049
+ max_cus = (ctx.pipeline_params or {}).get("max_cus", 8)
2050
+ SynthesizeCUOperator(max_cus=max_cus).process([], graph_ctx)
2051
+
2052
+ cu_count = sum(1 for nid in graph_ctx.node_ids if nid.startswith("cu_"))
2053
+ return {"cus": cu_count}
2054
+
2055
+
2056
+ # ---------------------------------------------------------------------------
2057
+ # Convenience: create a context and execute a full pipeline
2058
+ # ---------------------------------------------------------------------------
2059
+
2060
+ def run_pipeline(
2061
+ graph_registry,
2062
+ graph_namespace: str,
2063
+ stages: List[str],
2064
+ source_text: Optional[str] = None,
2065
+ source_file_path: Optional[str] = None,
2066
+ source_filename: Optional[str] = None,
2067
+ intent: str = "graph_rag",
2068
+ mode: str = "run_all",
2069
+ pipeline_params: Optional[Dict[str, Any]] = None,
2070
+ on_stage: Optional[Callable] = None,
2071
+ ) -> PipelineContext:
2072
+ """
2073
+ Convenience function to create a PipelineContext and execute stages.
2074
+
2075
+ Returns the final PipelineContext with all results.
2076
+ """
2077
+ ctx = PipelineContext(
2078
+ graph_namespace=graph_namespace,
2079
+ stages=stages,
2080
+ source_text=source_text,
2081
+ source_bytes_path=source_file_path,
2082
+ source_filename=source_filename,
2083
+ intent=intent,
2084
+ mode=mode,
2085
+ pipeline_params=pipeline_params or {},
2086
+ )
2087
+
2088
+ executor = StageExecutor(graph_registry=graph_registry)
2089
+
2090
+ if mode == "step_by_step":
2091
+ executor.execute_next(ctx)
2092
+ else:
2093
+ executor.execute_all(ctx, on_stage=on_stage)
2094
+
2095
+ return ctx