contextsynapse 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contextsynapse/__init__.py +107 -0
- contextsynapse/__main__.py +9 -0
- contextsynapse/a2a/__init__.py +28 -0
- contextsynapse/a2a/client.py +120 -0
- contextsynapse/a2a/discovery.py +128 -0
- contextsynapse/a2a/handlers.py +164 -0
- contextsynapse/a2a/models.py +246 -0
- contextsynapse/a2a/streaming.py +92 -0
- contextsynapse/a2a/task_manager.py +275 -0
- contextsynapse/adapters/__init__.py +98 -0
- contextsynapse/adapters/_base.py +298 -0
- contextsynapse/adapters/autogen/__init__.py +4 -0
- contextsynapse/adapters/autogen/tools.py +34 -0
- contextsynapse/adapters/crewai/__init__.py +4 -0
- contextsynapse/adapters/crewai/tools.py +63 -0
- contextsynapse/adapters/langchain/__init__.py +19 -0
- contextsynapse/adapters/langchain/message_history.py +17 -0
- contextsynapse/adapters/langchain/retriever.py +158 -0
- contextsynapse/adapters/langchain/tools.py +43 -0
- contextsynapse/adapters/langgraph/__init__.py +21 -0
- contextsynapse/adapters/langgraph/checkpoint.py +223 -0
- contextsynapse/adapters/langgraph/context_tools.py +341 -0
- contextsynapse/adapters/langgraph/message_history.py +138 -0
- contextsynapse/adapters/langgraph/tools.py +43 -0
- contextsynapse/adapters/llamaindex/__init__.py +4 -0
- contextsynapse/adapters/llamaindex/tools.py +64 -0
- contextsynapse/adapters/openai/__init__.py +24 -0
- contextsynapse/adapters/openai/tools.py +206 -0
- contextsynapse/adapters/pydantic_ai/__init__.py +4 -0
- contextsynapse/adapters/pydantic_ai/tools.py +36 -0
- contextsynapse/adapters/swarm/__init__.py +4 -0
- contextsynapse/adapters/swarm/tools.py +44 -0
- contextsynapse/agents/__init__.py +6 -0
- contextsynapse/agents/auto_capture.py +80 -0
- contextsynapse/agents/dispatcher.py +299 -0
- contextsynapse/agents/notifications.py +112 -0
- contextsynapse/aiql/__init__.py +71 -0
- contextsynapse/aiql/cache.py +341 -0
- contextsynapse/aiql/compiler/__init__.py +62 -0
- contextsynapse/aiql/compiler/optimizer.py +126 -0
- contextsynapse/aiql/compiler/planner.py +366 -0
- contextsynapse/aiql/compiler/statistics.py +235 -0
- contextsynapse/aiql/compiler/validator.py +425 -0
- contextsynapse/aiql/engine/__init__.py +11 -0
- contextsynapse/aiql/engine/chunk_storage.py +2584 -0
- contextsynapse/aiql/engine/executor.py +7737 -0
- contextsynapse/aiql/engine/extraction_integration.py +442 -0
- contextsynapse/aiql/engine/graph_builder.py +405 -0
- contextsynapse/aiql/engine/node_creation_from_normalized.py +691 -0
- contextsynapse/aiql/engine/performance_monitor.py +289 -0
- contextsynapse/aiql/engine/postprocessing.py +365 -0
- contextsynapse/aiql/engine/ray_stage_executors.py +918 -0
- contextsynapse/aiql/engine/redis_cache.py +106 -0
- contextsynapse/aiql/engine/streaming_json.py +366 -0
- contextsynapse/aiql/engine/streaming_parquet.py +568 -0
- contextsynapse/aiql/engine/streaming_ray_chunking.py +419 -0
- contextsynapse/aiql/grammar/__init__.py +14 -0
- contextsynapse/aiql/grammar/aiql_grammar.py +1822 -0
- contextsynapse/aiql/grammar/base.py +71 -0
- contextsynapse/aiql/parser/__init__.py +46 -0
- contextsynapse/aiql/parser/aiql_parser.py +11330 -0
- contextsynapse/aiql/parser/base_parser.py +47 -0
- contextsynapse/api/__init__.py +5 -0
- contextsynapse/api/a2a_router.py +233 -0
- contextsynapse/api/admin_router.py +140 -0
- contextsynapse/api/agent_worker_router.py +978 -0
- contextsynapse/api/algorithms_router.py +317 -0
- contextsynapse/api/api.py +4144 -0
- contextsynapse/api/audit.py +204 -0
- contextsynapse/api/auth.py +441 -0
- contextsynapse/api/auth_router.py +537 -0
- contextsynapse/api/billing.py +261 -0
- contextsynapse/api/billing_router.py +275 -0
- contextsynapse/api/boundary_router.py +452 -0
- contextsynapse/api/cognition_router.py +116 -0
- contextsynapse/api/connector_router.py +237 -0
- contextsynapse/api/context_router.py +1619 -0
- contextsynapse/api/dashboard_router.py +11367 -0
- contextsynapse/api/errors.py +183 -0
- contextsynapse/api/events.py +88 -0
- contextsynapse/api/experiment_bootstrap.py +321 -0
- contextsynapse/api/experiment_orchestrator.py +1078 -0
- contextsynapse/api/experiment_router.py +2089 -0
- contextsynapse/api/feature_gate.py +89 -0
- contextsynapse/api/federation_router.py +108 -0
- contextsynapse/api/integrations.py +567 -0
- contextsynapse/api/integrations_router.py +461 -0
- contextsynapse/api/intelligence_router.py +1028 -0
- contextsynapse/api/metering.py +155 -0
- contextsynapse/api/metrics.py +112 -0
- contextsynapse/api/models.py +38 -0
- contextsynapse/api/monitoring_router.py +174 -0
- contextsynapse/api/pipeline_router.py +1437 -0
- contextsynapse/api/playground_conn.py +116 -0
- contextsynapse/api/projects_router.py +131 -0
- contextsynapse/api/queries_router.py +282 -0
- contextsynapse/api/quota.py +95 -0
- contextsynapse/api/rate_limit.py +64 -0
- contextsynapse/api/search_router.py +261 -0
- contextsynapse/api/shield_router.py +447 -0
- contextsynapse/api/tenant_guard.py +103 -0
- contextsynapse/api/tenant_quotas.py +231 -0
- contextsynapse/api/tenants.py +305 -0
- contextsynapse/api/users.py +646 -0
- contextsynapse/api/v1.py +167 -0
- contextsynapse/api/vertical_builder_router.py +387 -0
- contextsynapse/benchmark/__init__.py +15 -0
- contextsynapse/benchmark/__main__.py +64 -0
- contextsynapse/benchmark/agents.py +338 -0
- contextsynapse/benchmark/live_ab.py +338 -0
- contextsynapse/benchmark/reporter.py +149 -0
- contextsynapse/benchmark/runner.py +264 -0
- contextsynapse/benchmark/savings.py +266 -0
- contextsynapse/benchmark/token_counter.py +212 -0
- contextsynapse/benchmark/workloads.py +289 -0
- contextsynapse/buffer/__init__.py +154 -0
- contextsynapse/buffer/base.py +257 -0
- contextsynapse/buffer/duckdb_buffer.py +246 -0
- contextsynapse/buffer/file_buffer.py +229 -0
- contextsynapse/buffer/lmdb_buffer.py +286 -0
- contextsynapse/buffer/local_buffer.py +154 -0
- contextsynapse/buffer/redis_buffer.py +211 -0
- contextsynapse/buffer/sqlite_buffer.py +266 -0
- contextsynapse/cache.py +2 -0
- contextsynapse/cli/__init__.py +40 -0
- contextsynapse/cli/cli.py +475 -0
- contextsynapse/cli/config_cli.py +177 -0
- contextsynapse/cli/scaffold.py +465 -0
- contextsynapse/cognition/__init__.py +31 -0
- contextsynapse/cognition/derivation.py +55 -0
- contextsynapse/cognition/events.py +135 -0
- contextsynapse/cognition/feedback.py +52 -0
- contextsynapse/cognition/hallucination.py +90 -0
- contextsynapse/cognition/invalidation.py +70 -0
- contextsynapse/cognition/read_tracker.py +50 -0
- contextsynapse/collaboration.py +2 -0
- contextsynapse/comparison.py +2 -0
- contextsynapse/config/__init__.py +10 -0
- contextsynapse/config/connectors/amfi_nav.yaml +18 -0
- contextsynapse/config/connectors/auto_sector.yaml +20 -0
- contextsynapse/config/connectors/banking_sector.yaml +20 -0
- contextsynapse/config/connectors/energy_sector.yaml +20 -0
- contextsynapse/config/connectors/fmcg_sector.yaml +20 -0
- contextsynapse/config/connectors/nsdl_fpi.yaml +9 -0
- contextsynapse/config/connectors/nse_price.yaml +61 -0
- contextsynapse/config/connectors/pharma_sector.yaml +20 -0
- contextsynapse/config/connectors/rbi_dbie.yaml +12 -0
- contextsynapse/config/context_presets/ai_research.json +25 -0
- contextsynapse/config/context_presets/chatgpt_conversations.json +19 -0
- contextsynapse/config/context_presets/claude_conversations.json +19 -0
- contextsynapse/config/context_presets/legal_compliance.json +21 -0
- contextsynapse/config/context_presets/meeting_notes.json +21 -0
- contextsynapse/config/context_presets/slack_engineering.json +21 -0
- contextsynapse/config/context_presets/tech_climate.json +24 -0
- contextsynapse/config/context_presets/toi_politics.json +25 -0
- contextsynapse/config/database_config.py +453 -0
- contextsynapse/config/deployment.py +221 -0
- contextsynapse/config/gpt_action_schema.json +164 -0
- contextsynapse/config/indicators/india_macro.yaml +213 -0
- contextsynapse/config/logging_config.py +88 -0
- contextsynapse/config/mode_config.py +223 -0
- contextsynapse/config/schemas/api_spec.yaml +48 -0
- contextsynapse/config/schemas/architecture_doc.yaml +51 -0
- contextsynapse/config/schemas/chat_history.yaml +90 -0
- contextsynapse/config/schemas/chatgpt_conversation.yaml +101 -0
- contextsynapse/config/schemas/claude_conversation.yaml +148 -0
- contextsynapse/config/schemas/contract.yaml +48 -0
- contextsynapse/config/schemas/conversation.yaml +167 -0
- contextsynapse/config/schemas/database.yaml +74 -0
- contextsynapse/config/schemas/decision.yaml +72 -0
- contextsynapse/config/schemas/ecommerce.yaml +56 -0
- contextsynapse/config/schemas/education.yaml +42 -0
- contextsynapse/config/schemas/healthcare.yaml +133 -0
- contextsynapse/config/schemas/hr_document.yaml +48 -0
- contextsynapse/config/schemas/invoice.yaml +40 -0
- contextsynapse/config/schemas/knowledge_base.yaml +75 -0
- contextsynapse/config/schemas/legal.yaml +42 -0
- contextsynapse/config/schemas/marketing_content.yaml +48 -0
- contextsynapse/config/schemas/meeting_notes.yaml +51 -0
- contextsynapse/config/schemas/news_article.yaml +61 -0
- contextsynapse/config/schemas/real_estate.yaml +49 -0
- contextsynapse/config/schemas/requirements_doc.yaml +48 -0
- contextsynapse/config/schemas/research_paper.yaml +56 -0
- contextsynapse/config/schemas/resume.yaml +49 -0
- contextsynapse/config/schemas/retail.yaml +42 -0
- contextsynapse/config/schemas/rule_meta.yaml +173 -0
- contextsynapse/config/schemas/rules.yaml +72 -0
- contextsynapse/config/schemas/screener_meta.yaml +236 -0
- contextsynapse/config/schemas/sdlc.yaml +103 -0
- contextsynapse/config/schemas/session_graph.yaml +172 -0
- contextsynapse/config/schemas/support_ticket.yaml +48 -0
- contextsynapse/config/schemas/web.yaml +64 -0
- contextsynapse/config/sensors/earnings_calendar.yaml +52 -0
- contextsynapse/config/sensors/global_risk.yaml +53 -0
- contextsynapse/config/sensors/government_policy.yaml +54 -0
- contextsynapse/config/sensors/supply_chain.yaml +50 -0
- contextsynapse/config/sensors/templates/behavioral.yaml +56 -0
- contextsynapse/config/sensors/templates/equity_sentiment.yaml +48 -0
- contextsynapse/config/sensors/templates/global_macro.yaml +47 -0
- contextsynapse/config/sensors/templates/market_data.yaml +118 -0
- contextsynapse/config/sensors/templates/sector_technical.yaml +47 -0
- contextsynapse/config/sensors/weather.yaml +52 -0
- contextsynapse/config/storage_strategy.py +202 -0
- contextsynapse/config/templates/competitive_intel.yaml +39 -0
- contextsynapse/config/templates/geopolitical_risk.yaml +46 -0
- contextsynapse/config/templates/market_analysis.yaml +58 -0
- contextsynapse/connectors/__init__.py +15 -0
- contextsynapse/connectors/api_router.py +121 -0
- contextsynapse/connectors/base.py +357 -0
- contextsynapse/connectors/financial/audio.py +170 -0
- contextsynapse/connectors/financial/youtube.py +272 -0
- contextsynapse/connectors/github.py +164 -0
- contextsynapse/connectors/jira.py +120 -0
- contextsynapse/connectors/kafka_connector.py +124 -0
- contextsynapse/connectors/pipeline_connector.py +747 -0
- contextsynapse/connectors/registry.py +116 -0
- contextsynapse/connectors/scheduler.py +87 -0
- contextsynapse/connectors/universal.py +360 -0
- contextsynapse/context/__init__.py +96 -0
- contextsynapse/context/__main__.py +4 -0
- contextsynapse/context/acl.py +125 -0
- contextsynapse/context/agent_card.py +201 -0
- contextsynapse/context/agent_memory.py +814 -0
- contextsynapse/context/agents.py +472 -0
- contextsynapse/context/agents_redis.py +391 -0
- contextsynapse/context/assembled.py +204 -0
- contextsynapse/context/attribution.py +113 -0
- contextsynapse/context/blob.py +184 -0
- contextsynapse/context/boundaries.py +187 -0
- contextsynapse/context/boundary.py +507 -0
- contextsynapse/context/bundle.py +155 -0
- contextsynapse/context/cli.py +519 -0
- contextsynapse/context/collaboration.py +165 -0
- contextsynapse/context/compiler.py +604 -0
- contextsynapse/context/composite.py +536 -0
- contextsynapse/context/compression.py +147 -0
- contextsynapse/context/context_manager.py +888 -0
- contextsynapse/context/context_schema.py +90 -0
- contextsynapse/context/context_schema.yaml +105 -0
- contextsynapse/context/context_units.py +891 -0
- contextsynapse/context/conversation.py +488 -0
- contextsynapse/context/dedup.py +207 -0
- contextsynapse/context/document_processor.py +698 -0
- contextsynapse/context/embedding_hooks.py +338 -0
- contextsynapse/context/execution_schema.py +274 -0
- contextsynapse/context/fan_out.py +138 -0
- contextsynapse/context/frozen.py +80 -0
- contextsynapse/context/gravity.py +227 -0
- contextsynapse/context/hub.py +1401 -0
- contextsynapse/context/ingest.py +391 -0
- contextsynapse/context/injection.py +226 -0
- contextsynapse/context/interaction_graphifier.py +562 -0
- contextsynapse/context/layers.py +574 -0
- contextsynapse/context/projection.py +1195 -0
- contextsynapse/context/promotion.py +634 -0
- contextsynapse/context/proof.py +131 -0
- contextsynapse/context/propagation.py +577 -0
- contextsynapse/context/pubsub.py +135 -0
- contextsynapse/context/quality.py +1054 -0
- contextsynapse/context/quality_tracker.py +270 -0
- contextsynapse/context/scoping.py +370 -0
- contextsynapse/context/seed.py +992 -0
- contextsynapse/context/session.py +1142 -0
- contextsynapse/context/session_graph.py +593 -0
- contextsynapse/context/session_memory.py +309 -0
- contextsynapse/context/session_resolver.py +111 -0
- contextsynapse/context/skill_learning.py +237 -0
- contextsynapse/context/smart_budget.py +129 -0
- contextsynapse/context/source_policy.py +242 -0
- contextsynapse/context/store_factory.py +53 -0
- contextsynapse/context/sync.py +648 -0
- contextsynapse/context/token.py +236 -0
- contextsynapse/context/vector_integration.py +297 -0
- contextsynapse/context/webhooks.py +194 -0
- contextsynapse/context/working_memory.py +224 -0
- contextsynapse/core/__init__.py +18 -0
- contextsynapse/core/archiver.py +237 -0
- contextsynapse/core/change_stream.py +282 -0
- contextsynapse/core/checkpoint.py +310 -0
- contextsynapse/core/cloud_storage.py +387 -0
- contextsynapse/core/context_state.py +651 -0
- contextsynapse/core/cron.py +324 -0
- contextsynapse/core/datasource.py +643 -0
- contextsynapse/core/db.py +289 -0
- contextsynapse/core/federation.py +283 -0
- contextsynapse/core/graph_coordinator.py +201 -0
- contextsynapse/core/graph_intelligence.py +388 -0
- contextsynapse/core/graph_structures.py +106 -0
- contextsynapse/core/graph_sync.py +157 -0
- contextsynapse/core/hybrid_graph_storage.py +2998 -0
- contextsynapse/core/lifecycle.py +691 -0
- contextsynapse/core/multiworker.py +378 -0
- contextsynapse/core/pruning.py +312 -0
- contextsynapse/core/redis_registry.py +524 -0
- contextsynapse/core/registry.py +1043 -0
- contextsynapse/core/registry_factory.py +51 -0
- contextsynapse/core/registry_metadata.py +294 -0
- contextsynapse/core/replay.py +354 -0
- contextsynapse/core/replication.py +339 -0
- contextsynapse/core/time_travel.py +260 -0
- contextsynapse/core/write_behind.py +152 -0
- contextsynapse/core/write_context.py +39 -0
- contextsynapse/db/__init__.py +1 -0
- contextsynapse/db/postgres.py +267 -0
- contextsynapse/db/profile.py +236 -0
- contextsynapse/db/rules.py +328 -0
- contextsynapse/demo/__init__.py +0 -0
- contextsynapse/demo/report.py +147 -0
- contextsynapse/demo/runner.py +520 -0
- contextsynapse/demo/scenario.py +165 -0
- contextsynapse/demo/session_demo.py +280 -0
- contextsynapse/demo/session_report.py +143 -0
- contextsynapse/demo/universal_runner.py +257 -0
- contextsynapse/engine/__init__.py +15 -0
- contextsynapse/engine/core.py +555 -0
- contextsynapse/extraction/__init__.py +69 -0
- contextsynapse/extraction/cli.py +274 -0
- contextsynapse/extraction/config.py +107 -0
- contextsynapse/extraction/examples/example_usage.py +118 -0
- contextsynapse/extraction/examples/sample_manifest.json +53 -0
- contextsynapse/extraction/examples/sample_normalized.json +185 -0
- contextsynapse/extraction/extract_engine.py +440 -0
- contextsynapse/extraction/extractors/__init__.py +37 -0
- contextsynapse/extraction/extractors/chat_export_extractor.py +189 -0
- contextsynapse/extraction/extractors/csv_extractor.py +83 -0
- contextsynapse/extraction/extractors/docx_extractor.py +129 -0
- contextsynapse/extraction/extractors/excel_extractor.py +152 -0
- contextsynapse/extraction/extractors/html_extractor.py +146 -0
- contextsynapse/extraction/extractors/pdf_extractor.py +635 -0
- contextsynapse/extraction/extractors/pdf_extractor_layoutparser.py +408 -0
- contextsynapse/extraction/extractors/register_extractors.py +63 -0
- contextsynapse/extraction/extractors/text_extractor.py +65 -0
- contextsynapse/extraction/extractors/txt_extractor.py +120 -0
- contextsynapse/extraction/extractors/website_extractor.py +246 -0
- contextsynapse/extraction/fact_extractor.py +236 -0
- contextsynapse/extraction/file_manager.py +353 -0
- contextsynapse/extraction/hierarchy.py +239 -0
- contextsynapse/extraction/id_generator.py +450 -0
- contextsynapse/extraction/layout_detector.py +409 -0
- contextsynapse/extraction/llm_entity_extractor.py +301 -0
- contextsynapse/extraction/normalized_store.py +480 -0
- contextsynapse/extraction/normalizer.py +594 -0
- contextsynapse/extraction/parquet_writer.py +598 -0
- contextsynapse/extraction/post_processors/__init__.py +40 -0
- contextsynapse/extraction/post_processors/column_reorganizer.py +507 -0
- contextsynapse/extraction/ray_normalization.py +233 -0
- contextsynapse/extraction/ray_runner.py +207 -0
- contextsynapse/extraction/redis_schema_registry.py +222 -0
- contextsynapse/extraction/registry.py +204 -0
- contextsynapse/extraction/schema_loader.py +663 -0
- contextsynapse/extraction/schema_registry.py +230 -0
- contextsynapse/extraction/semantic_normalizer.py +245 -0
- contextsynapse/extraction/version_manager.py +314 -0
- contextsynapse/gateway/__init__.py +24 -0
- contextsynapse/gateway/action_emitter.py +212 -0
- contextsynapse/gateway/cost_tracker.py +508 -0
- contextsynapse/gateway/gateway.py +377 -0
- contextsynapse/gateway/policy.py +303 -0
- contextsynapse/governance/__init__.py +13 -0
- contextsynapse/governance/layer.py +513 -0
- contextsynapse/ingestion/__init__.py +15 -0
- contextsynapse/ingestion/amplifier.py +288 -0
- contextsynapse/ingestion/async_ingest.py +463 -0
- contextsynapse/ingestion/chunker.py +303 -0
- contextsynapse/ingestion/chunking.py +58 -0
- contextsynapse/ingestion/cleaner.py +190 -0
- contextsynapse/ingestion/cleanse.py +252 -0
- contextsynapse/ingestion/connectors/__init__.py +38 -0
- contextsynapse/ingestion/connectors/base.py +300 -0
- contextsynapse/ingestion/connectors/chatgpt_memory.py +247 -0
- contextsynapse/ingestion/connectors/claude_memory.py +274 -0
- contextsynapse/ingestion/connectors/sap_ingestor.py +208 -0
- contextsynapse/ingestion/content_detector.py +80 -0
- contextsynapse/ingestion/content_monitor.py +502 -0
- contextsynapse/ingestion/dedup.py +106 -0
- contextsynapse/ingestion/feed.py +640 -0
- contextsynapse/ingestion/filters.py +461 -0
- contextsynapse/ingestion/graph_builder.py +705 -0
- contextsynapse/ingestion/input_classifier.py +307 -0
- contextsynapse/ingestion/job_manager.py +571 -0
- contextsynapse/ingestion/llm_extractor.py +610 -0
- contextsynapse/ingestion/orchestrator.py +116 -0
- contextsynapse/ingestion/parsers/__init__.py +19 -0
- contextsynapse/ingestion/parsers/graph_parser_registry.py +103 -0
- contextsynapse/ingestion/parsers/graphml_parser.py +141 -0
- contextsynapse/ingestion/parsers/jsonld_parser.py +171 -0
- contextsynapse/ingestion/parsers/rdf_parser.py +150 -0
- contextsynapse/ingestion/pipeline_context.py +228 -0
- contextsynapse/ingestion/pipeline_store.py +1195 -0
- contextsynapse/ingestion/queue.py +133 -0
- contextsynapse/ingestion/reprocess.py +179 -0
- contextsynapse/ingestion/scenario_router.py +224 -0
- contextsynapse/ingestion/schema_extractor.py +759 -0
- contextsynapse/ingestion/schema_validator.py +274 -0
- contextsynapse/ingestion/sdlc_ingest.py +87 -0
- contextsynapse/ingestion/smart_ingest.py +2451 -0
- contextsynapse/ingestion/source_connector.py +171 -0
- contextsynapse/ingestion/stage_executor.py +2095 -0
- contextsynapse/ingestion/stages/__init__.py +9 -0
- contextsynapse/ingestion/stages/tabular_to_graph.py +289 -0
- contextsynapse/ingestion/strategies.py +386 -0
- contextsynapse/ingestion/universal/__init__.py +4 -0
- contextsynapse/ingestion/universal/_operator_registry.py +59 -0
- contextsynapse/ingestion/universal/ingest_content.py +32 -0
- contextsynapse/ingestion/universal/operators/__init__.py +1 -0
- contextsynapse/ingestion/universal/operators/base.py +19 -0
- contextsynapse/ingestion/universal/operators/build_edges.py +75 -0
- contextsynapse/ingestion/universal/operators/cluster_topics.py +397 -0
- contextsynapse/ingestion/universal/operators/deduplicate.py +45 -0
- contextsynapse/ingestion/universal/operators/detect_signals.py +90 -0
- contextsynapse/ingestion/universal/operators/embed.py +80 -0
- contextsynapse/ingestion/universal/operators/extract_entities.py +253 -0
- contextsynapse/ingestion/universal/operators/extract_preferences.py +117 -0
- contextsynapse/ingestion/universal/operators/filter_content.py +69 -0
- contextsynapse/ingestion/universal/operators/index_bm25.py +52 -0
- contextsynapse/ingestion/universal/operators/infer_domains.py +69 -0
- contextsynapse/ingestion/universal/operators/link_cross_reference.py +210 -0
- contextsynapse/ingestion/universal/operators/parse_records.py +66 -0
- contextsynapse/ingestion/universal/operators/resolve_entities.py +261 -0
- contextsynapse/ingestion/universal/operators/scanners/__init__.py +16 -0
- contextsynapse/ingestion/universal/operators/scanners/edges.py +173 -0
- contextsynapse/ingestion/universal/operators/scanners/git.py +187 -0
- contextsynapse/ingestion/universal/operators/scanners/github.py +135 -0
- contextsynapse/ingestion/universal/operators/scanners/github_api.py +509 -0
- contextsynapse/ingestion/universal/operators/scanners/helpers.py +67 -0
- contextsynapse/ingestion/universal/operators/scanners/jira.py +65 -0
- contextsynapse/ingestion/universal/operators/scanners/llm_enrichment.py +82 -0
- contextsynapse/ingestion/universal/operators/scanners/repo.py +149 -0
- contextsynapse/ingestion/universal/operators/sdlc_scan.py +213 -0
- contextsynapse/ingestion/universal/operators/spec_scanner.py +149 -0
- contextsynapse/ingestion/universal/operators/store_documents.py +72 -0
- contextsynapse/ingestion/universal/operators/synthesize_cu.py +180 -0
- contextsynapse/ingestion/universal/operators/validate_gate.py +65 -0
- contextsynapse/ingestion/universal/parsers/__init__.py +5 -0
- contextsynapse/ingestion/universal/parsers/record_parser.py +203 -0
- contextsynapse/ingestion/universal/parsers/turn_parser.py +329 -0
- contextsynapse/ingestion/universal/shared_signals.py +161 -0
- contextsynapse/ingestion/universal/stage_executor.py +219 -0
- contextsynapse/ingestion/web_crawler.py +265 -0
- contextsynapse/intelligence/__init__.py +53 -0
- contextsynapse/intelligence/alerts.py +198 -0
- contextsynapse/intelligence/collector.py +655 -0
- contextsynapse/intelligence/comparison.py +198 -0
- contextsynapse/intelligence/config.py +56 -0
- contextsynapse/intelligence/conflict_detector.py +182 -0
- contextsynapse/intelligence/context_gaps.py +192 -0
- contextsynapse/intelligence/context_radar.py +135 -0
- contextsynapse/intelligence/correlation_engine.py +416 -0
- contextsynapse/intelligence/correlation_template.py +120 -0
- contextsynapse/intelligence/dedup.py +296 -0
- contextsynapse/intelligence/event_bus.py +91 -0
- contextsynapse/intelligence/feedback_loop.py +101 -0
- contextsynapse/intelligence/freshness.py +307 -0
- contextsynapse/intelligence/fundamental_feed.py +551 -0
- contextsynapse/intelligence/geo_data.py +170 -0
- contextsynapse/intelligence/geo_reference.py +112 -0
- contextsynapse/intelligence/graph_fusion.py +465 -0
- contextsynapse/intelligence/heuristics.py +176 -0
- contextsynapse/intelligence/impact_tracker.py +336 -0
- contextsynapse/intelligence/macro_indicators.py +319 -0
- contextsynapse/intelligence/ocr.py +104 -0
- contextsynapse/intelligence/persistence.py +267 -0
- contextsynapse/intelligence/pipeline_health.py +130 -0
- contextsynapse/intelligence/price_feed.py +221 -0
- contextsynapse/intelligence/rate_limiter.py +130 -0
- contextsynapse/intelligence/reactive.py +411 -0
- contextsynapse/intelligence/runtime_context.py +390 -0
- contextsynapse/intelligence/sentiment_decay.py +426 -0
- contextsynapse/intelligence/session.py +277 -0
- contextsynapse/intelligence/signal_hierarchy.py +438 -0
- contextsynapse/intelligence/signals.py +185 -0
- contextsynapse/intelligence/source_watcher.py +132 -0
- contextsynapse/intelligence/table_enricher.py +256 -0
- contextsynapse/intelligence/tagger.py +319 -0
- contextsynapse/intelligence/timeseries.py +209 -0
- contextsynapse/intelligence/tracker.py +318 -0
- contextsynapse/intelligence/watchdog.py +392 -0
- contextsynapse/intelligence/watchdog_manager.py +263 -0
- contextsynapse/intelligence/webhook_receiver.py +165 -0
- contextsynapse/intelligence/weight_learner.py +234 -0
- contextsynapse/llm/__init__.py +8 -0
- contextsynapse/llm/client.py +348 -0
- contextsynapse/llm/rate_limiter.py +166 -0
- contextsynapse/logging_config.py +2 -0
- contextsynapse/marketplace/__init__.py +7 -0
- contextsynapse/marketplace/registry.py +173 -0
- contextsynapse/mcp/__init__.py +11 -0
- contextsynapse/mcp/__main__.py +4 -0
- contextsynapse/mcp/auth_middleware.py +212 -0
- contextsynapse/mcp/connection_pool.py +128 -0
- contextsynapse/mcp/server.py +986 -0
- contextsynapse/mcp/session_context.py +67 -0
- contextsynapse/memory/__init__.py +34 -0
- contextsynapse/memory/engine.py +475 -0
- contextsynapse/memory/temporal.py +327 -0
- contextsynapse/metadata/__init__.py +12 -0
- contextsynapse/metadata/metadata_db.py +272 -0
- contextsynapse/metadata/metadata_query_tool.py +300 -0
- contextsynapse/metadata/metadata_tracker.py +425 -0
- contextsynapse/models/__init__.py +27 -0
- contextsynapse/models/embedding_service.py +360 -0
- contextsynapse/models/model_registry.py +228 -0
- contextsynapse/multiworker.py +2 -0
- contextsynapse/pipelines/__init__.py +1 -0
- contextsynapse/pipelines/executor.py +851 -0
- contextsynapse/pipelines/filters.py +189 -0
- contextsynapse/pipelines/models.py +176 -0
- contextsynapse/pipelines/scheduler.py +268 -0
- contextsynapse/playground/__init__.py +1 -0
- contextsynapse/playground/templates.py +85 -0
- contextsynapse/plugins/__init__.py +43 -0
- contextsynapse/plugins/api.py +66 -0
- contextsynapse/plugins/base.py +350 -0
- contextsynapse/plugins/domain_interface.py +508 -0
- contextsynapse/plugins/domain_loader.py +241 -0
- contextsynapse/plugins/loader.py +148 -0
- contextsynapse/plugins/registry.py +265 -0
- contextsynapse/project/__init__.py +11 -0
- contextsynapse/project/code_context.py +807 -0
- contextsynapse/project/connectors/__init__.py +4 -0
- contextsynapse/project/connectors/github_connector.py +113 -0
- contextsynapse/project/connectors/jira_connector.py +160 -0
- contextsynapse/project/graph.py +1023 -0
- contextsynapse/project/project_context.py +545 -0
- contextsynapse/project/scanner_registry.py +61 -0
- contextsynapse/project/schema_manager.py +322 -0
- contextsynapse/project/schemas/knowledge.yaml +59 -0
- contextsynapse/project/schemas/sdlc.yaml +369 -0
- contextsynapse/project/sdlc_schema.py +172 -0
- contextsynapse/project/spec_parser.py +85 -0
- contextsynapse/project/task_lock.py +181 -0
- contextsynapse/project/task_stream.py +203 -0
- contextsynapse/project/team_tools.py +537 -0
- contextsynapse/py.typed +0 -0
- contextsynapse/realtime/__init__.py +6 -0
- contextsynapse/realtime/collaboration.py +147 -0
- contextsynapse/rules/__init__.py +10 -0
- contextsynapse/rules/engine.py +116 -0
- contextsynapse/rules/evaluators/__init__.py +6 -0
- contextsynapse/rules/evaluators/aggregate.py +205 -0
- contextsynapse/rules/evaluators/custom.py +170 -0
- contextsynapse/rules/evaluators/pre_action.py +125 -0
- contextsynapse/rules/evaluators/temporal.py +129 -0
- contextsynapse/rules/loader.py +52 -0
- contextsynapse/rules/models.py +124 -0
- contextsynapse/rules/universal.py +393 -0
- contextsynapse/scheduler.py +229 -0
- contextsynapse/schema/__init__.py +42 -0
- contextsynapse/schema/compiler.py +252 -0
- contextsynapse/schema/composer.py +158 -0
- contextsynapse/schema/dedup.py +73 -0
- contextsynapse/schema/derivation.py +172 -0
- contextsynapse/schema/sdl.py +326 -0
- contextsynapse/schema/validation_gate.py +124 -0
- contextsynapse/sdk/__init__.py +63 -0
- contextsynapse/sdk/agent.py +214 -0
- contextsynapse/sdk/client.py +269 -0
- contextsynapse/sdk/exceptions.py +23 -0
- contextsynapse/sdk/models.py +78 -0
- contextsynapse/sdk/session.py +408 -0
- contextsynapse/sdk/tracked.py +218 -0
- contextsynapse/sdk/worker.py +266 -0
- contextsynapse/sdk/workspace_client.py +98 -0
- contextsynapse/sdk/ws.py +41 -0
- contextsynapse/search/__init__.py +20 -0
- contextsynapse/search/embedding_cache.py +140 -0
- contextsynapse/search/enhanced_search.py +152 -0
- contextsynapse/search/fulltext.py +130 -0
- contextsynapse/search/graph_search.py +634 -0
- contextsynapse/search/lmdb_index.py +1218 -0
- contextsynapse/search/nl_to_aiql.py +199 -0
- contextsynapse/search/rag.py +1346 -0
- contextsynapse/search/rag_cache.py +167 -0
- contextsynapse/search/reasoning_chain.py +231 -0
- contextsynapse/search/redis_search.py +207 -0
- contextsynapse/search/retrieval_quality.py +314 -0
- contextsynapse/search/semantic_query.py +455 -0
- contextsynapse/search/text_resolver.py +159 -0
- contextsynapse/search/whoosh_search.py +373 -0
- contextsynapse/security/__init__.py +17 -0
- contextsynapse/security/audit.py +297 -0
- contextsynapse/security/audit_logger.py +97 -0
- contextsynapse/security/audit_trail.py +270 -0
- contextsynapse/security/auth_unified.py +284 -0
- contextsynapse/security/auto_tagger.py +230 -0
- contextsynapse/security/data_security.py +218 -0
- contextsynapse/security/encryption.py +207 -0
- contextsynapse/security/identity.py +447 -0
- contextsynapse/security/jwt_identity.py +59 -0
- contextsynapse/security/middleware.py +336 -0
- contextsynapse/security/pii.py +233 -0
- contextsynapse/security/rbac.py +96 -0
- contextsynapse/security/rls.py +205 -0
- contextsynapse/security/sanitize.py +233 -0
- contextsynapse/security/scoped_encryption.py +131 -0
- contextsynapse/security/tenant.py +401 -0
- contextsynapse/shield/__init__.py +21 -0
- contextsynapse/shield/anomaly.py +56 -0
- contextsynapse/shield/permissions.py +99 -0
- contextsynapse/shield/profile.py +103 -0
- contextsynapse/shield/shield.py +108 -0
- contextsynapse/shield/trust_engine.py +69 -0
- contextsynapse/skills/__init__.py +14 -0
- contextsynapse/skills/engine.py +233 -0
- contextsynapse/skills/loader.py +59 -0
- contextsynapse/skills/models.py +139 -0
- contextsynapse/storage/__init__.py +7 -0
- contextsynapse/storage/cache.py +208 -0
- contextsynapse/storage/columnar_store_v2.py +178 -0
- contextsynapse/storage/csr_graph_storage.py +587 -0
- contextsynapse/storage/csr_store.py +388 -0
- contextsynapse/storage/document_store.py +491 -0
- contextsynapse/storage/hnsw_index.py +294 -0
- contextsynapse/storage/hybrid_store_v2.py +278 -0
- contextsynapse/storage/lmdb_graph_storage.py +366 -0
- contextsynapse/storage/migration.py +279 -0
- contextsynapse/storage/namespace_store.py +346 -0
- contextsynapse/storage/pure_graph_storage.py +185 -0
- contextsynapse/storage/redis_graph_adapter.py +186 -0
- contextsynapse/storage/redis_graph_storage.py +375 -0
- contextsynapse/storage/router/__init__.py +29 -0
- contextsynapse/storage/router/duckdb_store.py +398 -0
- contextsynapse/storage/router/factory.py +83 -0
- contextsynapse/storage/router/interface.py +125 -0
- contextsynapse/storage/router/postgres_store.py +339 -0
- contextsynapse/storage/router/resolver.py +192 -0
- contextsynapse/storage/router/router.py +300 -0
- contextsynapse/storage/single_file_storage.py +391 -0
- contextsynapse/storage/storage_manager.py +214 -0
- contextsynapse/storage/storage_query_optimizer.py +314 -0
- contextsynapse/storage/wal.py +359 -0
- contextsynapse/temporal/__init__.py +0 -0
- contextsynapse/temporal/storage.py +106 -0
- contextsynapse/tests/__init__.py +11 -0
- contextsynapse/tests/regression/__init__.py +1 -0
- contextsynapse/tests/regression/add_metadata.py +395 -0
- contextsynapse/tests/regression/add_skip_flags.py +85 -0
- contextsynapse/tests/regression/cases/test_basic_create_node.yaml +20 -0
- contextsynapse/tests/regression/cases/test_basic_select.yaml +21 -0
- contextsynapse/tests/regression/cases/test_blockchain_audit_trail.yaml +37 -0
- contextsynapse/tests/regression/cases/test_blockchain_audit_trail_filtered.yaml +39 -0
- contextsynapse/tests/regression/cases/test_blockchain_blocks_range.yaml +39 -0
- contextsynapse/tests/regression/cases/test_blockchain_get_block.yaml +35 -0
- contextsynapse/tests/regression/cases/test_blockchain_latest_block.yaml +35 -0
- contextsynapse/tests/regression/cases/test_blockchain_length.yaml +37 -0
- contextsynapse/tests/regression/cases/test_blockchain_merkle_root.yaml +37 -0
- contextsynapse/tests/regression/cases/test_blockchain_verify.yaml +35 -0
- contextsynapse/tests/regression/cases/test_blockchain_verify_block.yaml +35 -0
- contextsynapse/tests/regression/cases/test_create_edge.yaml +28 -0
- contextsynapse/tests/regression/cases/test_delete_edge_basic.yaml +29 -0
- contextsynapse/tests/regression/cases/test_delete_node_basic.yaml +29 -0
- contextsynapse/tests/regression/cases/test_evaluation_mrr.yaml +40 -0
- contextsynapse/tests/regression/cases/test_evaluation_ndcg.yaml +40 -0
- contextsynapse/tests/regression/cases/test_evaluation_precision.yaml +40 -0
- contextsynapse/tests/regression/cases/test_evaluation_recall.yaml +40 -0
- contextsynapse/tests/regression/cases/test_group_by_basic.yaml +56 -0
- contextsynapse/tests/regression/cases/test_match_node_basic.yaml +29 -0
- contextsynapse/tests/regression/cases/test_match_node_with_order_limit.yaml +31 -0
- contextsynapse/tests/regression/cases/test_namespace_switch.yaml +18 -0
- contextsynapse/tests/regression/cases/test_pipeline_cascaded.yaml +61 -0
- contextsynapse/tests/regression/cases/test_pipeline_fixed_size_chunking.yaml +48 -0
- contextsynapse/tests/regression/cases/test_pipeline_gpt4_chunking.yaml +54 -0
- contextsynapse/tests/regression/cases/test_pipeline_hybrid_indexing.yaml +57 -0
- contextsynapse/tests/regression/cases/test_pipeline_large_embedding.yaml +52 -0
- contextsynapse/tests/regression/cases/test_pipeline_paragraph_chunking.yaml +49 -0
- contextsynapse/tests/regression/cases/test_pipeline_recursive_chunking.yaml +47 -0
- contextsynapse/tests/regression/cases/test_pipeline_run_pipeline.yaml +56 -0
- contextsynapse/tests/regression/cases/test_pipeline_section_based_chunking.yaml +53 -0
- contextsynapse/tests/regression/cases/test_pipeline_semantic_chunking.yaml +51 -0
- contextsynapse/tests/regression/cases/test_pipeline_with_entity_extraction.yaml +57 -0
- contextsynapse/tests/regression/cases/test_pipeline_with_relationships.yaml +62 -0
- contextsynapse/tests/regression/cases/test_rag_evaluation.yaml +38 -0
- contextsynapse/tests/regression/cases/test_rag_graph_bfs_with_eval.yaml +48 -0
- contextsynapse/tests/regression/cases/test_rag_graph_centrality.yaml +40 -0
- contextsynapse/tests/regression/cases/test_rag_graph_centrality_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_graph_community.yaml +40 -0
- contextsynapse/tests/regression/cases/test_rag_graph_community_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_graph_dfs_with_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_graph_multi_hop_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_graph_path.yaml +40 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_basic.yaml +38 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_bfs.yaml +52 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_dfs.yaml +41 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_multi_hop.yaml +43 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_shortest_path.yaml +41 -0
- contextsynapse/tests/regression/cases/test_rag_graph_search_weighted.yaml +43 -0
- contextsynapse/tests/regression/cases/test_rag_graph_shortest_path_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_graph_traversal.yaml +30 -0
- contextsynapse/tests/regression/cases/test_rag_graph_weighted_eval.yaml +35 -0
- contextsynapse/tests/regression/cases/test_rag_hybrid_search.yaml +33 -0
- contextsynapse/tests/regression/cases/test_rag_hybrid_with_prompt_embedding.yaml +44 -0
- contextsynapse/tests/regression/cases/test_rag_keyword_search.yaml +32 -0
- contextsynapse/tests/regression/cases/test_rag_multi_hop_graph.yaml +32 -0
- contextsynapse/tests/regression/cases/test_rag_path_based.yaml +34 -0
- contextsynapse/tests/regression/cases/test_rag_profile_based.yaml +32 -0
- contextsynapse/tests/regression/cases/test_rag_vector_search.yaml +32 -0
- contextsynapse/tests/regression/cases/test_rag_weighted_graph.yaml +34 -0
- contextsynapse/tests/regression/cases/test_rag_weighted_hybrid.yaml +31 -0
- contextsynapse/tests/regression/cases/test_semantic_boundary_chunking.yaml +57 -0
- contextsynapse/tests/regression/cases/test_semantic_boundary_detection.yaml +33 -0
- contextsynapse/tests/regression/cases/test_semantic_hash_function.yaml +30 -0
- contextsynapse/tests/regression/cases/test_semantic_hash_with_model.yaml +31 -0
- contextsynapse/tests/regression/cases/test_tcs_ingestion_fixed_size.yaml +44 -0
- contextsynapse/tests/regression/cases/test_tcs_ingestion_recursive.yaml +46 -0
- contextsynapse/tests/regression/cases/test_tcs_ingestion_section.yaml +46 -0
- contextsynapse/tests/regression/cases/test_tcs_ingestion_semantic.yaml +44 -0
- contextsynapse/tests/regression/cases/test_time_travel_as_of.yaml +35 -0
- contextsynapse/tests/regression/cases/test_time_travel_at_commit.yaml +33 -0
- contextsynapse/tests/regression/cases/test_time_travel_at_timestamp.yaml +33 -0
- contextsynapse/tests/regression/cases/test_time_travel_for_system_time_all.yaml +37 -0
- contextsynapse/tests/regression/cases/test_time_travel_versions_between.yaml +39 -0
- contextsynapse/tests/regression/cases/test_traverse_basic.yaml +31 -0
- contextsynapse/tests/regression/cases/test_traverse_max_depth.yaml +34 -0
- contextsynapse/tests/regression/cases/test_update_edge_basic.yaml +29 -0
- contextsynapse/tests/regression/evaluation_metrics.py +227 -0
- contextsynapse/tests/regression/regression_runner.py +1081 -0
- contextsynapse/tests/regression/setup_regression_data.py +212 -0
- contextsynapse/tests/regression/test_basic_only.py +84 -0
- contextsynapse/tests/regression/test_namespace_creation.py +29 -0
- contextsynapse/tests/regression/test_single_fix.py +61 -0
- contextsynapse/tests/regression/test_single_query.py +20 -0
- contextsynapse/tests/regression/validate_test_syntax.py +179 -0
- contextsynapse/tests/test_advanced_features.py +347 -0
- contextsynapse/tests/test_aiql_summary.py +283 -0
- contextsynapse/tests/test_all_aiql_queries.py +948 -0
- contextsynapse/tests/test_all_create_queries.py +213 -0
- contextsynapse/tests/test_api_vs_native.py +280 -0
- contextsynapse/tests/test_chunking_dedup.py +199 -0
- contextsynapse/tests/test_connectors.py +171 -0
- contextsynapse/tests/test_edges_and_match.py +202 -0
- contextsynapse/tests/test_extraction_pipeline_integration.py +581 -0
- contextsynapse/tests/test_name_property.py +224 -0
- contextsynapse/tests/test_node_edge_metadata.py +250 -0
- contextsynapse/tests/test_playground_conn.py +114 -0
- contextsynapse/tests/test_query_cache.py +213 -0
- contextsynapse/tests/test_source_policy_auth.py +348 -0
- contextsynapse/tests/test_temporal_and_domain.py +224 -0
- contextsynapse/tests/test_time_travel_e2e.py +571 -0
- contextsynapse/tests/test_traversal_and_aggregation.py +275 -0
- contextsynapse/tools/__init__.py +36 -0
- contextsynapse/tools/a2a_tools.py +106 -0
- contextsynapse/tools/cognition_tools.py +118 -0
- contextsynapse/tools/context.py +905 -0
- contextsynapse/tools/formatting.py +92 -0
- contextsynapse/tools/gateway_tools.py +186 -0
- contextsynapse/tools/graph.py +1700 -0
- contextsynapse/tools/intelligence.py +235 -0
- contextsynapse/tools/memory.py +118 -0
- contextsynapse/tools/registry.py +373 -0
- contextsynapse/tools/shield_tools.py +159 -0
- contextsynapse/tools/tasks.py +341 -0
- contextsynapse/tools/workspace_tools.py +124 -0
- contextsynapse/utils/__init__.py +15 -0
- contextsynapse/utils/config_loader.py +198 -0
- contextsynapse/utils/env_loader.py +156 -0
- contextsynapse/utils/file_utils.py +288 -0
- contextsynapse/utils/logging_optimizer.py +119 -0
- contextsynapse/vector/__init__.py +30 -0
- contextsynapse/vector/chroma_vector_store.py +152 -0
- contextsynapse/vector/custom_vector_store.py +537 -0
- contextsynapse/vector/faiss_vector_store.py +296 -0
- contextsynapse/vector/qdrant_vector_store.py +163 -0
- contextsynapse/vector/vector_db.py +953 -0
- contextsynapse/vector/vector_db_manager.py +197 -0
- contextsynapse/workspace/__init__.py +27 -0
- contextsynapse/workspace/base.py +94 -0
- contextsynapse/workspace/git.py +228 -0
- contextsynapse/workspace/github.py +78 -0
- contextsynapse/workspace/local.py +73 -0
- contextsynapse/workspace/tools.py +222 -0
- contextsynapse-1.0.0.dist-info/METADATA +847 -0
- contextsynapse-1.0.0.dist-info/RECORD +776 -0
- contextsynapse-1.0.0.dist-info/WHEEL +5 -0
- contextsynapse-1.0.0.dist-info/entry_points.txt +4 -0
- contextsynapse-1.0.0.dist-info/licenses/LICENSE +200 -0
- contextsynapse-1.0.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,2095 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Stage Executor
|
|
3
|
+
==============
|
|
4
|
+
Unified execution engine for scenario-driven pipelines.
|
|
5
|
+
|
|
6
|
+
Replaces the three disconnected pipeline systems with a single composable
|
|
7
|
+
stage executor that supports both run-all and step-by-step execution.
|
|
8
|
+
|
|
9
|
+
Each stage reads from PipelineContext and writes results back to it.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import hashlib
|
|
15
|
+
import json
|
|
16
|
+
import logging
|
|
17
|
+
import os
|
|
18
|
+
import re
|
|
19
|
+
import tempfile
|
|
20
|
+
import uuid
|
|
21
|
+
from dataclasses import asdict
|
|
22
|
+
from datetime import datetime, timezone
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any, Callable, Dict, List, Optional, Tuple
|
|
25
|
+
|
|
26
|
+
from .pipeline_context import ClassificationResult, PipelineContext
|
|
27
|
+
|
|
28
|
+
logger = logging.getLogger(__name__)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
# ---------------------------------------------------------------------------
|
|
32
|
+
# Helpers (reused from pipeline_runner.py)
|
|
33
|
+
# ---------------------------------------------------------------------------
|
|
34
|
+
|
|
35
|
+
from .chunking import chunk_text as _chunk_text # canonical implementation
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _simhash(text: str, hash_bits: int = 64) -> int:
|
|
39
|
+
"""Compute a SimHash fingerprint for near-duplicate detection.
|
|
40
|
+
|
|
41
|
+
Uses word-level 3-grams hashed to bit vectors. Two texts with a small
|
|
42
|
+
Hamming distance between their SimHashes are likely near-duplicates.
|
|
43
|
+
"""
|
|
44
|
+
normalised = re.sub(r"\s+", " ", text.lower().strip())
|
|
45
|
+
tokens = normalised.split()
|
|
46
|
+
if len(tokens) < 3:
|
|
47
|
+
tokens = list(normalised) # character-level fallback for short text
|
|
48
|
+
|
|
49
|
+
v = [0] * hash_bits
|
|
50
|
+
for i in range(max(1, len(tokens) - 2)):
|
|
51
|
+
gram = " ".join(tokens[i:i + 3])
|
|
52
|
+
h = int(hashlib.md5(gram.encode()).hexdigest(), 16)
|
|
53
|
+
for j in range(hash_bits):
|
|
54
|
+
if h & (1 << j):
|
|
55
|
+
v[j] += 1
|
|
56
|
+
else:
|
|
57
|
+
v[j] -= 1
|
|
58
|
+
|
|
59
|
+
fingerprint = 0
|
|
60
|
+
for j in range(hash_bits):
|
|
61
|
+
if v[j] > 0:
|
|
62
|
+
fingerprint |= (1 << j)
|
|
63
|
+
return fingerprint
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _hamming_distance(a: int, b: int) -> int:
|
|
67
|
+
"""Number of differing bits between two integers."""
|
|
68
|
+
return bin(a ^ b).count("1")
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _content_hash(text: str) -> str:
|
|
72
|
+
"""SHA-256 of raw text."""
|
|
73
|
+
return hashlib.sha256(text.encode("utf-8")).hexdigest()
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _normalised_hash(text: str) -> str:
|
|
77
|
+
"""SHA-256 of normalised text (lowercase, collapsed whitespace)."""
|
|
78
|
+
normalised = re.sub(r"\s+", " ", text.lower().strip())
|
|
79
|
+
return hashlib.sha256(normalised.encode("utf-8")).hexdigest()
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def _entity_id(entity_type: str, entity_name: str) -> str:
|
|
83
|
+
"""Deterministic entity ID from type + canonical name."""
|
|
84
|
+
canonical = f"{entity_type}:{entity_name.lower().strip()}"
|
|
85
|
+
h = hashlib.sha256(canonical.encode()).hexdigest()[:12]
|
|
86
|
+
return f"entity_{h}"
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _parse_entities_fallback(text: str) -> List[Dict[str, Any]]:
|
|
90
|
+
"""Regex fallback to extract entities from non-JSON LLM responses."""
|
|
91
|
+
entities = []
|
|
92
|
+
patterns = [
|
|
93
|
+
r'(\w+):\s*([^(]+)\s*\(([^)]+)\)',
|
|
94
|
+
r'Name:\s*([^,]+),\s*Type:\s*([^,]+)',
|
|
95
|
+
]
|
|
96
|
+
for pattern in patterns:
|
|
97
|
+
matches = re.findall(pattern, text)
|
|
98
|
+
for match in matches:
|
|
99
|
+
if len(match) >= 2:
|
|
100
|
+
entities.append({
|
|
101
|
+
"name": match[1].strip() if len(match) > 1 else match[0].strip(),
|
|
102
|
+
"type": match[2].strip() if len(match) > 2 else "Entity",
|
|
103
|
+
})
|
|
104
|
+
return entities
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
# ---------------------------------------------------------------------------
|
|
108
|
+
# Stage Executor
|
|
109
|
+
# ---------------------------------------------------------------------------
|
|
110
|
+
|
|
111
|
+
class StageExecutor:
|
|
112
|
+
"""
|
|
113
|
+
Executes pipeline stages against a shared PipelineContext.
|
|
114
|
+
|
|
115
|
+
Supports:
|
|
116
|
+
- execute_stage(name, ctx) — run one stage
|
|
117
|
+
- execute_next(ctx) — run the next pending stage, then pause
|
|
118
|
+
- execute_all(ctx) — run all remaining stages without pausing
|
|
119
|
+
"""
|
|
120
|
+
|
|
121
|
+
def __init__(self, graph_registry=None, on_sub_step=None, aiql_executor=None):
|
|
122
|
+
self.graph_registry = graph_registry
|
|
123
|
+
self.on_sub_step = on_sub_step # Callable(run_id, stage, message, progress)
|
|
124
|
+
self._llm = None
|
|
125
|
+
self._embeddings = None
|
|
126
|
+
self._aiql_executor = aiql_executor
|
|
127
|
+
|
|
128
|
+
# Dispatch table: stage_name -> method
|
|
129
|
+
self._stage_handlers = {
|
|
130
|
+
"PARSE_FILE": self._stage_parse_file,
|
|
131
|
+
"CLASSIFY": self._stage_classify,
|
|
132
|
+
"CHUNK": self._stage_chunk,
|
|
133
|
+
"DEDUP": self._stage_dedup,
|
|
134
|
+
"MAP_TABULAR": self._stage_map_tabular,
|
|
135
|
+
"EXTRACT": self._stage_extract,
|
|
136
|
+
"EXTRACT_FACTS": self._stage_extract_facts,
|
|
137
|
+
"EMBED": self._stage_embed,
|
|
138
|
+
"INDEX_BM25": self._stage_index_bm25,
|
|
139
|
+
"STORE_VECTORS": self._stage_store_vectors,
|
|
140
|
+
"CANONICALIZE": self._stage_canonicalize,
|
|
141
|
+
"ENHANCE_GRAPH": self._stage_enhance_graph,
|
|
142
|
+
"PERSIST": self._stage_persist,
|
|
143
|
+
"IMPORT_GRAPH": self._stage_import_graph,
|
|
144
|
+
"VALIDATE_SCHEMA": self._stage_validate_schema,
|
|
145
|
+
"SDLC_SCAN": self._stage_sdlc_scan,
|
|
146
|
+
# SDLC sub-stages (resumable)
|
|
147
|
+
"SDLC_SCAN_FILES": self._stage_sdlc_scan_files,
|
|
148
|
+
"SDLC_SCAN_GITHUB": self._stage_sdlc_scan_github,
|
|
149
|
+
"SDLC_SCAN_GIT": self._stage_sdlc_scan_git,
|
|
150
|
+
"SDLC_SCAN_EDGES": self._stage_sdlc_scan_edges,
|
|
151
|
+
"SDLC_SCAN_CU": self._stage_sdlc_scan_cu,
|
|
152
|
+
}
|
|
153
|
+
|
|
154
|
+
# ------------------------------------------------------------------
|
|
155
|
+
# LLM / Embedding accessors (lazy)
|
|
156
|
+
# ------------------------------------------------------------------
|
|
157
|
+
|
|
158
|
+
def _get_aiql_executor(self):
|
|
159
|
+
"""Get or lazily create an AIQLExecutor."""
|
|
160
|
+
if self._aiql_executor is None and self.graph_registry:
|
|
161
|
+
try:
|
|
162
|
+
from ..aiql.engine.executor import AIQLExecutor
|
|
163
|
+
self._aiql_executor = AIQLExecutor(graph_registry=self.graph_registry)
|
|
164
|
+
except Exception as e:
|
|
165
|
+
logger.warning("Could not init AIQLExecutor: %s", e)
|
|
166
|
+
return self._aiql_executor
|
|
167
|
+
|
|
168
|
+
def _get_llm(self, model: Optional[str] = None):
|
|
169
|
+
if self._llm is None and model:
|
|
170
|
+
try:
|
|
171
|
+
from ..llm import get_llm_client
|
|
172
|
+
# Support "provider:model" format (e.g. "groq:gpt-oss-120b")
|
|
173
|
+
provider, _, model_name = (model or "").partition(":")
|
|
174
|
+
if model_name:
|
|
175
|
+
self._llm = get_llm_client(provider=provider, model=model_name)
|
|
176
|
+
else:
|
|
177
|
+
self._llm = get_llm_client()
|
|
178
|
+
except ImportError:
|
|
179
|
+
pass
|
|
180
|
+
except Exception as e:
|
|
181
|
+
logger.debug("LLM not available: %s", e)
|
|
182
|
+
return self._llm
|
|
183
|
+
|
|
184
|
+
def _get_embeddings(self, model: Optional[str] = None):
|
|
185
|
+
if self._embeddings is None and model:
|
|
186
|
+
try:
|
|
187
|
+
from ..models.embedding_service import EmbeddingService
|
|
188
|
+
self._embeddings = EmbeddingService()
|
|
189
|
+
except ImportError:
|
|
190
|
+
pass
|
|
191
|
+
except Exception as e:
|
|
192
|
+
logger.debug("Embeddings not available: %s", e)
|
|
193
|
+
return self._embeddings
|
|
194
|
+
|
|
195
|
+
# ------------------------------------------------------------------
|
|
196
|
+
# Sub-step emission
|
|
197
|
+
# ------------------------------------------------------------------
|
|
198
|
+
|
|
199
|
+
def _emit_sub_step(self, ctx: PipelineContext, message: str, progress: float = None):
|
|
200
|
+
"""Record a sub-step on the context and notify listeners."""
|
|
201
|
+
ctx.record_sub_step(message, progress)
|
|
202
|
+
if self.on_sub_step:
|
|
203
|
+
try:
|
|
204
|
+
self.on_sub_step(ctx.run_id, ctx.current_stage or "unknown", message, progress)
|
|
205
|
+
except Exception:
|
|
206
|
+
pass # never let sub-step emission break execution
|
|
207
|
+
|
|
208
|
+
# ------------------------------------------------------------------
|
|
209
|
+
# Public API
|
|
210
|
+
# ------------------------------------------------------------------
|
|
211
|
+
|
|
212
|
+
def execute_stage(self, stage_name: str, ctx: PipelineContext) -> PipelineContext:
|
|
213
|
+
"""Execute a single named stage."""
|
|
214
|
+
handler = self._stage_handlers.get(stage_name)
|
|
215
|
+
if not handler:
|
|
216
|
+
ctx.record_stage(stage_name, "failed", error=f"Unknown stage: {stage_name}")
|
|
217
|
+
return ctx
|
|
218
|
+
|
|
219
|
+
ctx.status = "running"
|
|
220
|
+
started = datetime.now(timezone.utc).isoformat()
|
|
221
|
+
|
|
222
|
+
try:
|
|
223
|
+
summary = handler(ctx)
|
|
224
|
+
ctx.record_stage(stage_name, "completed", summary=summary or {})
|
|
225
|
+
except Exception as e:
|
|
226
|
+
logger.error("Stage %s failed: %s", stage_name, e, exc_info=True)
|
|
227
|
+
ctx.record_stage(stage_name, "failed", error=str(e))
|
|
228
|
+
ctx.status = "failed"
|
|
229
|
+
|
|
230
|
+
return ctx
|
|
231
|
+
|
|
232
|
+
def execute_next(self, ctx: PipelineContext) -> PipelineContext:
|
|
233
|
+
"""Execute the next pending stage, then pause."""
|
|
234
|
+
if ctx.is_done:
|
|
235
|
+
ctx.status = "completed"
|
|
236
|
+
return ctx
|
|
237
|
+
|
|
238
|
+
stage_name = ctx.current_stage
|
|
239
|
+
if not stage_name:
|
|
240
|
+
ctx.status = "completed"
|
|
241
|
+
return ctx
|
|
242
|
+
|
|
243
|
+
self.execute_stage(stage_name, ctx)
|
|
244
|
+
|
|
245
|
+
# Advance pointer
|
|
246
|
+
ctx.advance()
|
|
247
|
+
|
|
248
|
+
if ctx.status != "failed":
|
|
249
|
+
if ctx.is_done:
|
|
250
|
+
ctx.status = "completed"
|
|
251
|
+
else:
|
|
252
|
+
ctx.status = "paused"
|
|
253
|
+
|
|
254
|
+
return ctx
|
|
255
|
+
|
|
256
|
+
def execute_all(
|
|
257
|
+
self,
|
|
258
|
+
ctx: PipelineContext,
|
|
259
|
+
on_stage: Optional[Callable] = None,
|
|
260
|
+
) -> PipelineContext:
|
|
261
|
+
"""Execute all remaining stages without pausing."""
|
|
262
|
+
while not ctx.is_done and ctx.status != "failed":
|
|
263
|
+
stage_name = ctx.current_stage
|
|
264
|
+
if on_stage:
|
|
265
|
+
on_stage(stage_name, "running", ctx)
|
|
266
|
+
|
|
267
|
+
self.execute_stage(stage_name, ctx)
|
|
268
|
+
ctx.advance()
|
|
269
|
+
|
|
270
|
+
if on_stage and ctx.stage_results:
|
|
271
|
+
last = ctx.stage_results[-1]
|
|
272
|
+
on_stage(stage_name, last.get("status", "completed"), ctx)
|
|
273
|
+
|
|
274
|
+
if ctx.status == "failed":
|
|
275
|
+
break
|
|
276
|
+
|
|
277
|
+
if ctx.status != "failed":
|
|
278
|
+
ctx.status = "completed"
|
|
279
|
+
|
|
280
|
+
return ctx
|
|
281
|
+
|
|
282
|
+
# ------------------------------------------------------------------
|
|
283
|
+
# PARSE_FILE — extract content from uploaded file
|
|
284
|
+
# ------------------------------------------------------------------
|
|
285
|
+
|
|
286
|
+
def _stage_parse_file(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
287
|
+
"""Use ExtractorRegistry to parse the uploaded file."""
|
|
288
|
+
from ..extraction.registry import ExtractorRegistry, ExtractionResult
|
|
289
|
+
|
|
290
|
+
file_path = ctx.source_bytes_path
|
|
291
|
+
filename = ctx.source_filename or ""
|
|
292
|
+
|
|
293
|
+
# If we have text input instead of a file, create a minimal extraction result
|
|
294
|
+
if not file_path and ctx.source_text:
|
|
295
|
+
ctx.extraction_result = {
|
|
296
|
+
"pages": [{"page_no": 1, "text": ctx.source_text, "content": ctx.source_text}],
|
|
297
|
+
"tables": [],
|
|
298
|
+
"metadata": {"title": filename or "Text Input", "file_type": "txt"},
|
|
299
|
+
"errors": [],
|
|
300
|
+
}
|
|
301
|
+
return {"source": "text", "pages": 1, "tables": 0}
|
|
302
|
+
|
|
303
|
+
if not file_path or not os.path.exists(file_path):
|
|
304
|
+
raise FileNotFoundError(f"Source file not found: {file_path}")
|
|
305
|
+
|
|
306
|
+
self._emit_sub_step(ctx, f"Parsing file: {filename or file_path}")
|
|
307
|
+
|
|
308
|
+
# Try extension-based extractor lookup
|
|
309
|
+
ext = Path(filename).suffix.lower() if filename else Path(file_path).suffix.lower()
|
|
310
|
+
|
|
311
|
+
# Direct extractor mapping
|
|
312
|
+
extractor = None
|
|
313
|
+
extractor_map = {
|
|
314
|
+
".xlsx": "excel", ".xls": "excel",
|
|
315
|
+
".csv": "csv", ".tsv": "csv",
|
|
316
|
+
".pdf": "pdf", ".docx": "docx",
|
|
317
|
+
".txt": "txt", ".md": "txt",
|
|
318
|
+
".html": "html", ".htm": "html",
|
|
319
|
+
".zip": "chat_export",
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
extractor_type = extractor_map.get(ext)
|
|
323
|
+
|
|
324
|
+
# Auto-detect .json files — could be chat export or regular data
|
|
325
|
+
if ext == ".json" and not extractor_type:
|
|
326
|
+
try:
|
|
327
|
+
from ..extraction.extractors.chat_export_extractor import ChatExportExtractor
|
|
328
|
+
_test_ext = ChatExportExtractor(type("C", (), {"file_path": file_path})())
|
|
329
|
+
if _test_ext.supports_format(file_path):
|
|
330
|
+
extractor_type = "chat_export"
|
|
331
|
+
except Exception:
|
|
332
|
+
pass
|
|
333
|
+
if not extractor_type:
|
|
334
|
+
extractor_type = "txt" # fallback: treat JSON as text
|
|
335
|
+
|
|
336
|
+
if extractor_type:
|
|
337
|
+
extractor = self._create_extractor(extractor_type, file_path)
|
|
338
|
+
|
|
339
|
+
if not extractor:
|
|
340
|
+
# Fallback: try auto-detect from registry
|
|
341
|
+
try:
|
|
342
|
+
from ..extraction.registry import ExtractionConfig
|
|
343
|
+
config = ExtractionConfig()
|
|
344
|
+
except ImportError:
|
|
345
|
+
config = type("Config", (), {})()
|
|
346
|
+
extractor = ExtractorRegistry.auto_detect(file_path, config)
|
|
347
|
+
|
|
348
|
+
if not extractor:
|
|
349
|
+
raise ValueError(f"No extractor available for file type: {ext}")
|
|
350
|
+
|
|
351
|
+
self._emit_sub_step(ctx, f"Using {extractor_type or 'auto'} extractor for {ext}")
|
|
352
|
+
|
|
353
|
+
# Pass sub-step callback to extractor if supported
|
|
354
|
+
if hasattr(extractor, 'progress_callback'):
|
|
355
|
+
extractor.progress_callback = lambda msg, prog=None: self._emit_sub_step(ctx, msg, prog)
|
|
356
|
+
|
|
357
|
+
result = extractor.extract(file_path)
|
|
358
|
+
|
|
359
|
+
self._emit_sub_step(ctx, f"Extracted {len(result.pages)} pages, {len(result.tables)} tables")
|
|
360
|
+
|
|
361
|
+
# Serialize ExtractionResult to dict for PipelineContext
|
|
362
|
+
ctx.extraction_result = {
|
|
363
|
+
"pages": result.pages,
|
|
364
|
+
"tables": result.tables,
|
|
365
|
+
"metadata": result.metadata,
|
|
366
|
+
"images": result.images if hasattr(result, "images") else [],
|
|
367
|
+
"errors": result.errors,
|
|
368
|
+
}
|
|
369
|
+
ctx.tables = result.tables
|
|
370
|
+
|
|
371
|
+
return {
|
|
372
|
+
"file_type": ext,
|
|
373
|
+
"pages": len(result.pages),
|
|
374
|
+
"tables": len(result.tables),
|
|
375
|
+
"errors": len(result.errors),
|
|
376
|
+
}
|
|
377
|
+
|
|
378
|
+
def _create_extractor(self, extractor_type: str, file_path: str):
|
|
379
|
+
"""Create an extractor instance by type name."""
|
|
380
|
+
try:
|
|
381
|
+
# Create a minimal config object
|
|
382
|
+
try:
|
|
383
|
+
from ..extraction.config import ExtractionConfig
|
|
384
|
+
config = ExtractionConfig(file_path=file_path)
|
|
385
|
+
except Exception:
|
|
386
|
+
config = type("Config", (), {
|
|
387
|
+
"file_path": file_path,
|
|
388
|
+
"pages_mode": type("Mode", (), {"value": "full"})(),
|
|
389
|
+
"pages_range": None,
|
|
390
|
+
"pages_list": None,
|
|
391
|
+
"detect": ["TEXT"],
|
|
392
|
+
"reader": "AUTO",
|
|
393
|
+
"parse_metadata": True,
|
|
394
|
+
"store_intermediate": False,
|
|
395
|
+
"output_dir": "contextcore_data",
|
|
396
|
+
"namespace": None,
|
|
397
|
+
"document_id": None,
|
|
398
|
+
"use_layout_parser": False,
|
|
399
|
+
"layout_detection_method": "auto",
|
|
400
|
+
"post_process": True,
|
|
401
|
+
"post_processing_config": None,
|
|
402
|
+
})()
|
|
403
|
+
|
|
404
|
+
if extractor_type == "excel":
|
|
405
|
+
from ..extraction.extractors.excel_extractor import ExcelExtractor
|
|
406
|
+
return ExcelExtractor(config)
|
|
407
|
+
elif extractor_type == "csv":
|
|
408
|
+
from ..extraction.extractors.csv_extractor import CsvExtractor
|
|
409
|
+
return CsvExtractor(config)
|
|
410
|
+
elif extractor_type == "pdf":
|
|
411
|
+
from ..extraction.extractors.pdf_extractor import PdfExtractor
|
|
412
|
+
return PdfExtractor(config)
|
|
413
|
+
elif extractor_type == "docx":
|
|
414
|
+
from ..extraction.extractors.docx_extractor import DocxExtractor
|
|
415
|
+
return DocxExtractor(config)
|
|
416
|
+
elif extractor_type == "txt":
|
|
417
|
+
from ..extraction.extractors.txt_extractor import TxtExtractor
|
|
418
|
+
return TxtExtractor(config)
|
|
419
|
+
elif extractor_type == "html":
|
|
420
|
+
from ..extraction.extractors.html_extractor import HtmlExtractor
|
|
421
|
+
return HtmlExtractor(config)
|
|
422
|
+
elif extractor_type == "chat_export":
|
|
423
|
+
from ..extraction.extractors.chat_export_extractor import ChatExportExtractor
|
|
424
|
+
return ChatExportExtractor(config)
|
|
425
|
+
except Exception as e:
|
|
426
|
+
logger.warning("Failed to create %s extractor: %s", extractor_type, e)
|
|
427
|
+
return None
|
|
428
|
+
|
|
429
|
+
# ------------------------------------------------------------------
|
|
430
|
+
# CLASSIFY — detect file type and content structure
|
|
431
|
+
# ------------------------------------------------------------------
|
|
432
|
+
|
|
433
|
+
def _stage_classify(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
434
|
+
"""Run InputClassifier on extraction result."""
|
|
435
|
+
from .input_classifier import InputClassifier
|
|
436
|
+
|
|
437
|
+
if not ctx.extraction_result:
|
|
438
|
+
raise ValueError("No extraction result to classify — run PARSE_FILE first")
|
|
439
|
+
|
|
440
|
+
classifier = InputClassifier()
|
|
441
|
+
classification = classifier.classify(
|
|
442
|
+
ctx.extraction_result,
|
|
443
|
+
ctx.source_filename or "",
|
|
444
|
+
)
|
|
445
|
+
|
|
446
|
+
ctx.classification = classification.to_dict()
|
|
447
|
+
|
|
448
|
+
return {
|
|
449
|
+
"file_type": classification.file_type,
|
|
450
|
+
"structure_type": classification.structure_type,
|
|
451
|
+
"confidence": classification.confidence,
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
# ------------------------------------------------------------------
|
|
455
|
+
# CHUNK — split text content into chunks
|
|
456
|
+
# ------------------------------------------------------------------
|
|
457
|
+
|
|
458
|
+
def _stage_chunk(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
459
|
+
"""Chunk text content from extraction result."""
|
|
460
|
+
if not ctx.extraction_result:
|
|
461
|
+
# Fallback: if source_text is set, create extraction_result on the fly
|
|
462
|
+
if ctx.source_text:
|
|
463
|
+
ctx.extraction_result = {
|
|
464
|
+
"pages": [{"page_no": 1, "text": ctx.source_text}],
|
|
465
|
+
"tables": [],
|
|
466
|
+
"metadata": {"title": ctx.source_filename or "Text Input", "file_type": "txt"},
|
|
467
|
+
}
|
|
468
|
+
else:
|
|
469
|
+
raise ValueError("No extraction result — run PARSE_FILE first")
|
|
470
|
+
|
|
471
|
+
pages = ctx.extraction_result.get("pages", [])
|
|
472
|
+
all_text = "\n\n".join(p.get("text", "") for p in pages).strip()
|
|
473
|
+
|
|
474
|
+
if not all_text:
|
|
475
|
+
ctx.chunks = []
|
|
476
|
+
return {"chunks": 0, "source": "no_text"}
|
|
477
|
+
|
|
478
|
+
max_chars = ctx.pipeline_params.get("chunk_size", 2000)
|
|
479
|
+
raw_chunks = _chunk_text(all_text, max_chars)
|
|
480
|
+
|
|
481
|
+
ctx.chunks = [
|
|
482
|
+
{
|
|
483
|
+
"id": str(uuid.uuid4()),
|
|
484
|
+
"index": i,
|
|
485
|
+
"text": chunk,
|
|
486
|
+
"char_count": len(chunk),
|
|
487
|
+
}
|
|
488
|
+
for i, chunk in enumerate(raw_chunks)
|
|
489
|
+
]
|
|
490
|
+
|
|
491
|
+
return {"chunks": len(ctx.chunks), "total_chars": len(all_text)}
|
|
492
|
+
|
|
493
|
+
# ------------------------------------------------------------------
|
|
494
|
+
# DEDUP — content and semantic hash deduplication
|
|
495
|
+
# ------------------------------------------------------------------
|
|
496
|
+
|
|
497
|
+
def _stage_dedup(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
498
|
+
"""Deduplicate chunks using content hashing and SimHash fingerprinting.
|
|
499
|
+
|
|
500
|
+
Two layers:
|
|
501
|
+
1. **Exact hash** — SHA-256 of raw and normalised text. Catches exact
|
|
502
|
+
duplicates and trivial reformattings (whitespace, case changes).
|
|
503
|
+
2. **SimHash** — 64-bit fingerprint based on word 3-grams. Catches
|
|
504
|
+
near-duplicates (paraphrases, minor edits). Configurable Hamming
|
|
505
|
+
distance threshold (default: 3 bits out of 64).
|
|
506
|
+
|
|
507
|
+
Checks both within the current batch (cross-chunk) and against
|
|
508
|
+
existing nodes already in the target graph.
|
|
509
|
+
|
|
510
|
+
Pipeline params:
|
|
511
|
+
dedup_enabled: bool (default True)
|
|
512
|
+
dedup_hamming_threshold: int (default 6 — bits that may differ out of 64)
|
|
513
|
+
dedup_skip_short: int (default 100 — skip chunks < N chars)
|
|
514
|
+
"""
|
|
515
|
+
if not ctx.chunks:
|
|
516
|
+
return {"chunks": 0, "skipped": 0, "reason": "no_chunks"}
|
|
517
|
+
|
|
518
|
+
enabled = ctx.pipeline_params.get("dedup_enabled", True)
|
|
519
|
+
if not enabled:
|
|
520
|
+
return {"chunks": len(ctx.chunks), "skipped": 0, "reason": "disabled"}
|
|
521
|
+
|
|
522
|
+
hamming_threshold = ctx.pipeline_params.get("dedup_hamming_threshold", 6)
|
|
523
|
+
skip_short = ctx.pipeline_params.get("dedup_skip_short", 100)
|
|
524
|
+
|
|
525
|
+
# Collect existing hashes from the target graph
|
|
526
|
+
existing_hashes = set() # SHA-256 content hashes
|
|
527
|
+
existing_norm_hashes = set() # SHA-256 normalised hashes
|
|
528
|
+
existing_simhashes = [] # (node_id, simhash_int) pairs
|
|
529
|
+
|
|
530
|
+
graph = None
|
|
531
|
+
ns = getattr(ctx, "graph_namespace", "") or getattr(ctx, "namespace", "")
|
|
532
|
+
if self.graph_registry and ns:
|
|
533
|
+
try:
|
|
534
|
+
graph = self.graph_registry.get_graph(ns, load_if_missing=True)
|
|
535
|
+
if graph is None and ":" in ns:
|
|
536
|
+
graph = self.graph_registry.get_graph(ns.split(":", 1)[1], load_if_missing=True)
|
|
537
|
+
except Exception:
|
|
538
|
+
pass
|
|
539
|
+
|
|
540
|
+
if graph:
|
|
541
|
+
for node in graph.get_all_nodes():
|
|
542
|
+
props = node.properties if hasattr(node, "properties") else {}
|
|
543
|
+
ch = props.get("content_hash")
|
|
544
|
+
if ch:
|
|
545
|
+
existing_hashes.add(ch)
|
|
546
|
+
nh = props.get("normalised_hash")
|
|
547
|
+
if nh:
|
|
548
|
+
existing_norm_hashes.add(nh)
|
|
549
|
+
sh = props.get("simhash")
|
|
550
|
+
if sh is not None:
|
|
551
|
+
try:
|
|
552
|
+
existing_simhashes.append((node.id, int(sh)))
|
|
553
|
+
except (ValueError, TypeError):
|
|
554
|
+
pass
|
|
555
|
+
|
|
556
|
+
# Dedup within current batch + against existing graph
|
|
557
|
+
batch_hashes = set()
|
|
558
|
+
batch_norm_hashes = set()
|
|
559
|
+
batch_simhashes = [] # (chunk_id, simhash_int)
|
|
560
|
+
|
|
561
|
+
kept = []
|
|
562
|
+
skipped_exact = 0
|
|
563
|
+
skipped_near = 0
|
|
564
|
+
skipped_short = 0
|
|
565
|
+
|
|
566
|
+
for chunk in ctx.chunks:
|
|
567
|
+
text = chunk.get("text", "")
|
|
568
|
+
|
|
569
|
+
# Skip very short chunks (headers, footers, etc.)
|
|
570
|
+
if len(text) < skip_short:
|
|
571
|
+
chunk["content_hash"] = _content_hash(text)
|
|
572
|
+
chunk["normalised_hash"] = _normalised_hash(text)
|
|
573
|
+
chunk["simhash"] = str(_simhash(text))
|
|
574
|
+
kept.append(chunk)
|
|
575
|
+
skipped_short += 1 # counted but still kept
|
|
576
|
+
continue
|
|
577
|
+
|
|
578
|
+
ch = _content_hash(text)
|
|
579
|
+
nh = _normalised_hash(text)
|
|
580
|
+
sh = _simhash(text)
|
|
581
|
+
|
|
582
|
+
# Layer 1: exact hash check
|
|
583
|
+
if ch in existing_hashes or ch in batch_hashes:
|
|
584
|
+
skipped_exact += 1
|
|
585
|
+
logger.info("[DEDUP] Exact duplicate skipped (chunk %s, %d chars)",
|
|
586
|
+
chunk.get("index", "?"), len(text))
|
|
587
|
+
continue
|
|
588
|
+
|
|
589
|
+
if nh in existing_norm_hashes or nh in batch_norm_hashes:
|
|
590
|
+
skipped_exact += 1
|
|
591
|
+
logger.info("[DEDUP] Normalised duplicate skipped (chunk %s, %d chars)",
|
|
592
|
+
chunk.get("index", "?"), len(text))
|
|
593
|
+
continue
|
|
594
|
+
|
|
595
|
+
# Layer 2: SimHash near-duplicate check
|
|
596
|
+
is_near_dup = False
|
|
597
|
+
for _, existing_sh in existing_simhashes:
|
|
598
|
+
if _hamming_distance(sh, existing_sh) <= hamming_threshold:
|
|
599
|
+
is_near_dup = True
|
|
600
|
+
break
|
|
601
|
+
if not is_near_dup:
|
|
602
|
+
for _, batch_sh in batch_simhashes:
|
|
603
|
+
if _hamming_distance(sh, batch_sh) <= hamming_threshold:
|
|
604
|
+
is_near_dup = True
|
|
605
|
+
break
|
|
606
|
+
|
|
607
|
+
if is_near_dup:
|
|
608
|
+
skipped_near += 1
|
|
609
|
+
logger.info("[DEDUP] Near-duplicate skipped (chunk %s, hamming ≤ %d)",
|
|
610
|
+
chunk.get("index", "?"), hamming_threshold)
|
|
611
|
+
continue
|
|
612
|
+
|
|
613
|
+
# Not a duplicate — stamp hashes and keep
|
|
614
|
+
chunk["content_hash"] = ch
|
|
615
|
+
chunk["normalised_hash"] = nh
|
|
616
|
+
chunk["simhash"] = str(sh)
|
|
617
|
+
batch_hashes.add(ch)
|
|
618
|
+
batch_norm_hashes.add(nh)
|
|
619
|
+
batch_simhashes.append((chunk["id"], sh))
|
|
620
|
+
kept.append(chunk)
|
|
621
|
+
|
|
622
|
+
# Re-index kept chunks
|
|
623
|
+
for i, chunk in enumerate(kept):
|
|
624
|
+
chunk["index"] = i
|
|
625
|
+
|
|
626
|
+
ctx.chunks = kept
|
|
627
|
+
|
|
628
|
+
total_skipped = skipped_exact + skipped_near
|
|
629
|
+
if total_skipped:
|
|
630
|
+
logger.info("[DEDUP] Kept %d/%d chunks (exact=%d, near=%d skipped)",
|
|
631
|
+
len(kept), len(kept) + total_skipped, skipped_exact, skipped_near)
|
|
632
|
+
|
|
633
|
+
return {
|
|
634
|
+
"chunks_before": len(kept) + total_skipped,
|
|
635
|
+
"chunks_after": len(kept),
|
|
636
|
+
"skipped_exact": skipped_exact,
|
|
637
|
+
"skipped_near_duplicate": skipped_near,
|
|
638
|
+
"short_chunks": skipped_short,
|
|
639
|
+
"hamming_threshold": hamming_threshold,
|
|
640
|
+
}
|
|
641
|
+
|
|
642
|
+
# ------------------------------------------------------------------
|
|
643
|
+
# MAP_TABULAR — convert structured tables to nodes/edges
|
|
644
|
+
# ------------------------------------------------------------------
|
|
645
|
+
|
|
646
|
+
def _stage_map_tabular(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
647
|
+
"""Map tabular data to graph nodes and edges."""
|
|
648
|
+
from .stages.tabular_to_graph import TabularToGraphMapper
|
|
649
|
+
|
|
650
|
+
tables = ctx.tables or (ctx.extraction_result or {}).get("tables", [])
|
|
651
|
+
if not tables:
|
|
652
|
+
return {"nodes": 0, "edges": 0, "reason": "no_tables"}
|
|
653
|
+
|
|
654
|
+
# Need classification for column role mapping
|
|
655
|
+
classification = ClassificationResult.from_dict(ctx.classification) if ctx.classification else None
|
|
656
|
+
if not classification:
|
|
657
|
+
# Quick classify if not done yet
|
|
658
|
+
from .input_classifier import InputClassifier
|
|
659
|
+
classifier = InputClassifier()
|
|
660
|
+
classification = classifier.classify(
|
|
661
|
+
ctx.extraction_result or {"tables": tables},
|
|
662
|
+
ctx.source_filename or "",
|
|
663
|
+
)
|
|
664
|
+
ctx.classification = classification.to_dict()
|
|
665
|
+
|
|
666
|
+
mapper = TabularToGraphMapper()
|
|
667
|
+
nodes, edges = mapper.map(tables, classification)
|
|
668
|
+
|
|
669
|
+
# Merge with any existing nodes/edges (from other stages)
|
|
670
|
+
ctx.nodes.extend(nodes)
|
|
671
|
+
ctx.edges.extend(edges)
|
|
672
|
+
|
|
673
|
+
return {"nodes": len(nodes), "edges": len(edges)}
|
|
674
|
+
|
|
675
|
+
# ------------------------------------------------------------------
|
|
676
|
+
# EXTRACT — LLM entity & relationship extraction from chunks
|
|
677
|
+
# ------------------------------------------------------------------
|
|
678
|
+
|
|
679
|
+
def _stage_extract(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
680
|
+
"""Extract entities and relationships from text chunks using LLM."""
|
|
681
|
+
llm_model = ctx.pipeline_params.get("llm_model")
|
|
682
|
+
llm = self._get_llm(llm_model)
|
|
683
|
+
|
|
684
|
+
if not llm or not llm_model:
|
|
685
|
+
return {"entities": 0, "relationships": 0, "reason": "no_llm_configured"}
|
|
686
|
+
|
|
687
|
+
chunks = ctx.chunks
|
|
688
|
+
if not chunks:
|
|
689
|
+
return {"entities": 0, "relationships": 0, "reason": "no_chunks"}
|
|
690
|
+
|
|
691
|
+
entity_map: Dict[str, Dict] = {}
|
|
692
|
+
all_relationships = []
|
|
693
|
+
all_facts = []
|
|
694
|
+
chunk_entity_links = []
|
|
695
|
+
|
|
696
|
+
schema = ctx.pipeline_params.get("schema")
|
|
697
|
+
|
|
698
|
+
# Parallel chunk extraction — up to 3 concurrent LLM calls per document
|
|
699
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
700
|
+
import os
|
|
701
|
+
_extract_workers = int(os.environ.get("CONTEXTSYNAPSE_EXTRACT_WORKERS") or os.environ.get("AICONTEXTDB_EXTRACT_WORKERS", "3"))
|
|
702
|
+
|
|
703
|
+
def _extract_one(chunk):
|
|
704
|
+
text = chunk.get("text", "")
|
|
705
|
+
if not text.strip():
|
|
706
|
+
return chunk, {"entities": [], "relationships": []}
|
|
707
|
+
return chunk, self._extract_from_chunk(llm, llm_model, text, schema)
|
|
708
|
+
|
|
709
|
+
valid_chunks = [c for c in chunks if c.get("text", "").strip()]
|
|
710
|
+
chunk_results = []
|
|
711
|
+
with ThreadPoolExecutor(max_workers=min(_extract_workers, len(valid_chunks) or 1)) as pool:
|
|
712
|
+
futures = {pool.submit(_extract_one, c): c for c in valid_chunks}
|
|
713
|
+
for future in as_completed(futures):
|
|
714
|
+
try:
|
|
715
|
+
chunk_results.append(future.result())
|
|
716
|
+
except Exception as e:
|
|
717
|
+
logger.warning("[EXTRACT] Chunk extraction failed: %s", e)
|
|
718
|
+
|
|
719
|
+
for chunk, result in chunk_results:
|
|
720
|
+
for ent in result.get("entities", []):
|
|
721
|
+
# Accept "name" or "title" as the entity name (SDLC types use title)
|
|
722
|
+
name = (ent.get("name") or ent.get("title", "")).strip()
|
|
723
|
+
etype = ent.get("type", "Entity").strip()
|
|
724
|
+
if not name:
|
|
725
|
+
continue
|
|
726
|
+
key = f"{etype}:{name.lower()}"
|
|
727
|
+
# Collect all properties from the LLM response
|
|
728
|
+
extra_props = ent.get("properties", {})
|
|
729
|
+
if isinstance(extra_props, dict):
|
|
730
|
+
# Also grab top-level keys that aren't metadata
|
|
731
|
+
for k, v in ent.items():
|
|
732
|
+
if k not in ("name", "title", "type", "properties", "description") and v:
|
|
733
|
+
extra_props[k] = v
|
|
734
|
+
if key not in entity_map:
|
|
735
|
+
entity_map[key] = {
|
|
736
|
+
"name": name,
|
|
737
|
+
"type": etype,
|
|
738
|
+
"description": ent.get("description", ""),
|
|
739
|
+
"properties": extra_props,
|
|
740
|
+
"mention_count": 0,
|
|
741
|
+
}
|
|
742
|
+
else:
|
|
743
|
+
# Merge properties from subsequent mentions
|
|
744
|
+
for pk, pv in extra_props.items():
|
|
745
|
+
if pv and not entity_map[key]["properties"].get(pk):
|
|
746
|
+
entity_map[key]["properties"][pk] = pv
|
|
747
|
+
entity_map[key]["mention_count"] += 1
|
|
748
|
+
chunk_entity_links.append((chunk["id"], key))
|
|
749
|
+
|
|
750
|
+
for rel in result.get("relationships", []):
|
|
751
|
+
src = rel.get("source", "").strip()
|
|
752
|
+
tgt = rel.get("target", "").strip()
|
|
753
|
+
# Accept both "relation" and "type" keys (schema prompt uses "type")
|
|
754
|
+
rtype = (rel.get("type") or rel.get("relation", "RELATED_TO")).strip()
|
|
755
|
+
if src and tgt:
|
|
756
|
+
all_relationships.append({
|
|
757
|
+
"source_name": src,
|
|
758
|
+
"target_name": tgt,
|
|
759
|
+
"relation": rtype,
|
|
760
|
+
"properties": rel.get("properties", {}),
|
|
761
|
+
})
|
|
762
|
+
|
|
763
|
+
# Collect facts from schema-guided extraction
|
|
764
|
+
for fact in result.get("facts", []):
|
|
765
|
+
ftype = fact.get("type", "Fact")
|
|
766
|
+
statement = fact.get("statement", "")
|
|
767
|
+
if not statement:
|
|
768
|
+
continue
|
|
769
|
+
fact_props = {k: v for k, v in fact.items()
|
|
770
|
+
if k not in ("type",) and v}
|
|
771
|
+
fact_props["extraction_method"] = "llm_extraction"
|
|
772
|
+
fact_props["source"] = ctx.source_url or ctx.source_filename or "llm_extraction"
|
|
773
|
+
all_facts.append({"type": ftype, "properties": fact_props})
|
|
774
|
+
|
|
775
|
+
# Build entity nodes with full properties
|
|
776
|
+
name_to_id: Dict[str, str] = {}
|
|
777
|
+
for key, ent in entity_map.items():
|
|
778
|
+
eid = _entity_id(ent["type"], ent["name"])
|
|
779
|
+
name_to_id[ent["name"].lower()] = eid
|
|
780
|
+
props = {
|
|
781
|
+
"name": ent["name"],
|
|
782
|
+
"description": ent.get("description", ""),
|
|
783
|
+
"mention_count": ent["mention_count"],
|
|
784
|
+
"extraction_method": "llm_extraction",
|
|
785
|
+
"source": ctx.source_url or ctx.source_filename or "llm_extraction",
|
|
786
|
+
}
|
|
787
|
+
# Merge in all extracted properties (priority, status, etc.)
|
|
788
|
+
props.update(ent.get("properties", {}))
|
|
789
|
+
ctx.nodes.append({
|
|
790
|
+
"id": eid,
|
|
791
|
+
"label": ent["type"],
|
|
792
|
+
"properties": props,
|
|
793
|
+
})
|
|
794
|
+
|
|
795
|
+
# Build chunk → entity MENTIONS edges
|
|
796
|
+
seen_mention = set()
|
|
797
|
+
for chunk_id, entity_key in chunk_entity_links:
|
|
798
|
+
ent = entity_map[entity_key]
|
|
799
|
+
eid = _entity_id(ent["type"], ent["name"])
|
|
800
|
+
edge_key = f"{chunk_id}:{eid}"
|
|
801
|
+
if edge_key in seen_mention:
|
|
802
|
+
continue
|
|
803
|
+
seen_mention.add(edge_key)
|
|
804
|
+
ctx.edges.append({
|
|
805
|
+
"id": str(uuid.uuid4()),
|
|
806
|
+
"source": chunk_id,
|
|
807
|
+
"target": eid,
|
|
808
|
+
"label": "MENTIONS",
|
|
809
|
+
"properties": {},
|
|
810
|
+
})
|
|
811
|
+
|
|
812
|
+
# Build relationship edges
|
|
813
|
+
rel_count = 0
|
|
814
|
+
for rel in all_relationships:
|
|
815
|
+
src_id = name_to_id.get(rel["source_name"].lower())
|
|
816
|
+
tgt_id = name_to_id.get(rel["target_name"].lower())
|
|
817
|
+
if src_id and tgt_id and src_id != tgt_id:
|
|
818
|
+
ctx.edges.append({
|
|
819
|
+
"id": str(uuid.uuid4()),
|
|
820
|
+
"source": src_id,
|
|
821
|
+
"target": tgt_id,
|
|
822
|
+
"label": rel["relation"],
|
|
823
|
+
"properties": rel.get("properties", {}),
|
|
824
|
+
})
|
|
825
|
+
rel_count += 1
|
|
826
|
+
|
|
827
|
+
# Build fact nodes + link to mentioning chunks
|
|
828
|
+
fact_count = 0
|
|
829
|
+
for fact in all_facts:
|
|
830
|
+
fid = f"fact_{hashlib.sha256(fact['properties'].get('statement', str(uuid.uuid4())).encode()).hexdigest()[:12]}"
|
|
831
|
+
ctx.nodes.append({
|
|
832
|
+
"id": fid,
|
|
833
|
+
"label": fact["type"],
|
|
834
|
+
"properties": fact["properties"],
|
|
835
|
+
})
|
|
836
|
+
fact_count += 1
|
|
837
|
+
|
|
838
|
+
return {
|
|
839
|
+
"entities": len(entity_map),
|
|
840
|
+
"relationships": rel_count,
|
|
841
|
+
"facts": fact_count,
|
|
842
|
+
"chunks_processed": len(chunks),
|
|
843
|
+
}
|
|
844
|
+
|
|
845
|
+
def _extract_from_chunk(
|
|
846
|
+
self, llm, model: str, text: str, schema=None
|
|
847
|
+
) -> Dict[str, Any]:
|
|
848
|
+
"""Extract entities and relationships from a single chunk via LLM."""
|
|
849
|
+
|
|
850
|
+
# Use schema_to_prompt if we have an ExtractionSchema object
|
|
851
|
+
try:
|
|
852
|
+
from ..extraction.schema_loader import ExtractionSchema, schema_to_prompt
|
|
853
|
+
if isinstance(schema, ExtractionSchema):
|
|
854
|
+
prompt = schema_to_prompt(schema) + f"\n{text[:3000]}"
|
|
855
|
+
try:
|
|
856
|
+
raw = llm.generate(prompt=prompt, max_tokens=4000)
|
|
857
|
+
if raw and raw.strip():
|
|
858
|
+
text_to_parse = raw.strip()
|
|
859
|
+
json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
|
|
860
|
+
if json_match:
|
|
861
|
+
text_to_parse = json_match.group(1).strip()
|
|
862
|
+
data = json.loads(text_to_parse)
|
|
863
|
+
return {
|
|
864
|
+
"entities": data.get("entities", []),
|
|
865
|
+
"relationships": data.get("relationships", []),
|
|
866
|
+
"facts": data.get("facts", []),
|
|
867
|
+
}
|
|
868
|
+
except Exception:
|
|
869
|
+
pass
|
|
870
|
+
return {"entities": [], "relationships": [], "facts": []}
|
|
871
|
+
except ImportError:
|
|
872
|
+
pass
|
|
873
|
+
|
|
874
|
+
schema_hint = ""
|
|
875
|
+
if schema:
|
|
876
|
+
node_types = list(schema.get("node_types", {}).keys()) if isinstance(schema, dict) else []
|
|
877
|
+
edge_types = list(schema.get("edge_types", {}).keys()) if isinstance(schema, dict) else []
|
|
878
|
+
if node_types:
|
|
879
|
+
schema_hint += f"\nAllowed entity types: {', '.join(node_types)}"
|
|
880
|
+
if edge_types:
|
|
881
|
+
schema_hint += f"\nAllowed relationship types: {', '.join(edge_types)}"
|
|
882
|
+
schema_hint += "\nOnly extract entities and relationships matching these types.\n"
|
|
883
|
+
|
|
884
|
+
prompt = f"""Extract all named entities and relationships from the text below.
|
|
885
|
+
Return ONLY valid JSON with this exact structure:
|
|
886
|
+
{{
|
|
887
|
+
"entities": [
|
|
888
|
+
{{"name": "entity name", "type": "Person|Organization|Location|Concept|Technology|Event|Other", "description": "brief description"}}
|
|
889
|
+
],
|
|
890
|
+
"relationships": [
|
|
891
|
+
{{"source": "entity name", "target": "entity name", "relation": "WORKS_FOR|LOCATED_IN|RELATED_TO|PART_OF|CREATED_BY|USES|etc", "properties": {{}}}}
|
|
892
|
+
]
|
|
893
|
+
}}
|
|
894
|
+
{schema_hint}
|
|
895
|
+
Rules:
|
|
896
|
+
- Extract ALL meaningful entities (people, organizations, places, concepts, technologies, dates, events)
|
|
897
|
+
- Extract relationships between the entities you found
|
|
898
|
+
- Use consistent entity names
|
|
899
|
+
- Return empty lists if no entities/relationships found
|
|
900
|
+
- Return ONLY the JSON, no other text
|
|
901
|
+
|
|
902
|
+
Text:
|
|
903
|
+
{text[:3000]}"""
|
|
904
|
+
|
|
905
|
+
for attempt in range(3):
|
|
906
|
+
try:
|
|
907
|
+
raw = llm.generate(prompt=prompt, max_tokens=4000)
|
|
908
|
+
if not raw or not raw.strip():
|
|
909
|
+
return {"entities": [], "relationships": []}
|
|
910
|
+
|
|
911
|
+
# Strip markdown fences if present
|
|
912
|
+
text_to_parse = raw.strip()
|
|
913
|
+
json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
|
|
914
|
+
if json_match:
|
|
915
|
+
text_to_parse = json_match.group(1).strip()
|
|
916
|
+
|
|
917
|
+
data = json.loads(text_to_parse)
|
|
918
|
+
entities = data.get("entities", [])
|
|
919
|
+
relationships = data.get("relationships", [])
|
|
920
|
+
if not isinstance(entities, list):
|
|
921
|
+
entities = []
|
|
922
|
+
if not isinstance(relationships, list):
|
|
923
|
+
relationships = []
|
|
924
|
+
return {"entities": entities, "relationships": relationships}
|
|
925
|
+
|
|
926
|
+
except json.JSONDecodeError:
|
|
927
|
+
return {"entities": [], "relationships": []}
|
|
928
|
+
except Exception as e:
|
|
929
|
+
# Retry on rate limit / transient errors
|
|
930
|
+
err_str = str(e).lower()
|
|
931
|
+
if attempt < 2 and ("rate" in err_str or "429" in err_str or "limit" in err_str or "timeout" in err_str):
|
|
932
|
+
import time
|
|
933
|
+
wait = (attempt + 1) * 5
|
|
934
|
+
logger.info("Rate limited on extract (attempt %d), retrying in %ds", attempt + 1, wait)
|
|
935
|
+
time.sleep(wait)
|
|
936
|
+
continue
|
|
937
|
+
logger.warning("Extraction failed for chunk: %s", e)
|
|
938
|
+
return {"entities": [], "relationships": []}
|
|
939
|
+
|
|
940
|
+
# ------------------------------------------------------------------
|
|
941
|
+
# EXTRACT_FACTS — decompose passages into atomic factual statements
|
|
942
|
+
# ------------------------------------------------------------------
|
|
943
|
+
|
|
944
|
+
def _stage_extract_facts(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
945
|
+
"""Extract atomic facts from text chunks using LLM.
|
|
946
|
+
|
|
947
|
+
Each fact is a self-contained statement that can be independently
|
|
948
|
+
verified or searched. Facts bridge passages and entities:
|
|
949
|
+
Passage -[STATES]-> Fact -[MENTIONS]-> Entity.
|
|
950
|
+
"""
|
|
951
|
+
llm_model = ctx.pipeline_params.get("llm_model")
|
|
952
|
+
llm = self._get_llm(llm_model)
|
|
953
|
+
|
|
954
|
+
if not llm or not llm_model:
|
|
955
|
+
return {"facts": 0, "reason": "no_llm_configured"}
|
|
956
|
+
|
|
957
|
+
chunks = ctx.chunks
|
|
958
|
+
if not chunks:
|
|
959
|
+
return {"facts": 0, "reason": "no_chunks"}
|
|
960
|
+
|
|
961
|
+
fact_map: Dict[str, Dict] = {} # canonical_key -> fact dict
|
|
962
|
+
chunk_fact_links: List[tuple] = [] # (chunk_id, fact_key)
|
|
963
|
+
|
|
964
|
+
total = len(chunks)
|
|
965
|
+
for i, chunk in enumerate(chunks):
|
|
966
|
+
text = chunk.get("text", "")
|
|
967
|
+
if not text.strip():
|
|
968
|
+
continue
|
|
969
|
+
|
|
970
|
+
self._emit_sub_step(ctx, f"Extracting facts from chunk {i+1}/{total}", (i + 1) / total)
|
|
971
|
+
|
|
972
|
+
result = self._extract_facts_from_chunk(llm, llm_model, text)
|
|
973
|
+
|
|
974
|
+
for fact in result.get("facts", []):
|
|
975
|
+
statement = fact.get("statement", "").strip()
|
|
976
|
+
if not statement or len(statement) < 10:
|
|
977
|
+
continue
|
|
978
|
+
ftype = fact.get("type", "fact").strip().lower()
|
|
979
|
+
confidence = float(fact.get("confidence", 0.8))
|
|
980
|
+
|
|
981
|
+
# Canonical key for dedup
|
|
982
|
+
key = f"fact:{statement.lower()[:120]}"
|
|
983
|
+
if key not in fact_map:
|
|
984
|
+
fid = f"fact_{hashlib.sha256(key.encode()).hexdigest()[:12]}"
|
|
985
|
+
fact_map[key] = {
|
|
986
|
+
"id": fid,
|
|
987
|
+
"statement": statement,
|
|
988
|
+
"type": ftype,
|
|
989
|
+
"confidence": confidence,
|
|
990
|
+
"mention_count": 0,
|
|
991
|
+
"entity_names": fact.get("entities", []),
|
|
992
|
+
}
|
|
993
|
+
fact_map[key]["mention_count"] += 1
|
|
994
|
+
chunk_fact_links.append((chunk["id"], key))
|
|
995
|
+
|
|
996
|
+
# Store facts on PipelineContext for later stages
|
|
997
|
+
ctx.facts = []
|
|
998
|
+
for key, f in fact_map.items():
|
|
999
|
+
ctx.facts.append(f)
|
|
1000
|
+
|
|
1001
|
+
# Also create chunk→fact links for PERSIST to use
|
|
1002
|
+
ctx.pipeline_params["_chunk_fact_links"] = chunk_fact_links
|
|
1003
|
+
ctx.pipeline_params["_fact_keys"] = {k: v["id"] for k, v in fact_map.items()}
|
|
1004
|
+
|
|
1005
|
+
self._emit_sub_step(ctx, f"Extracted {len(fact_map)} facts from {total} chunks")
|
|
1006
|
+
|
|
1007
|
+
return {
|
|
1008
|
+
"facts": len(fact_map),
|
|
1009
|
+
"chunks_processed": total,
|
|
1010
|
+
}
|
|
1011
|
+
|
|
1012
|
+
def _extract_facts_from_chunk(
|
|
1013
|
+
self, llm, model: str, text: str
|
|
1014
|
+
) -> Dict[str, Any]:
|
|
1015
|
+
"""Extract atomic facts from a single chunk via LLM."""
|
|
1016
|
+
prompt = f"""Decompose the following text into atomic factual statements.
|
|
1017
|
+
Each fact should be a single, self-contained sentence that can be independently verified.
|
|
1018
|
+
Also list which named entities each fact mentions.
|
|
1019
|
+
|
|
1020
|
+
Return ONLY valid JSON:
|
|
1021
|
+
{{
|
|
1022
|
+
"facts": [
|
|
1023
|
+
{{
|
|
1024
|
+
"statement": "Alice joined Acme Corp as CTO in 2024.",
|
|
1025
|
+
"type": "claim",
|
|
1026
|
+
"confidence": 0.95,
|
|
1027
|
+
"entities": ["Alice", "Acme Corp"]
|
|
1028
|
+
}}
|
|
1029
|
+
]
|
|
1030
|
+
}}
|
|
1031
|
+
|
|
1032
|
+
Rules:
|
|
1033
|
+
- Each fact must be one clear, atomic statement
|
|
1034
|
+
- Type is one of: claim, evidence, definition, event, relationship
|
|
1035
|
+
- Confidence 0.0-1.0 based on how clearly the text states the fact
|
|
1036
|
+
- entities: list the exact entity names mentioned in that fact
|
|
1037
|
+
- Skip trivial/filler statements
|
|
1038
|
+
- Return ONLY the JSON
|
|
1039
|
+
|
|
1040
|
+
Text:
|
|
1041
|
+
{text[:3000]}"""
|
|
1042
|
+
|
|
1043
|
+
for attempt in range(3):
|
|
1044
|
+
try:
|
|
1045
|
+
raw = llm.generate(prompt=prompt, max_tokens=4000)
|
|
1046
|
+
if not raw or not raw.strip():
|
|
1047
|
+
logger.warning("Fact extraction: LLM returned empty response (attempt %d)", attempt + 1)
|
|
1048
|
+
return {"facts": []}
|
|
1049
|
+
|
|
1050
|
+
text_to_parse = raw.strip()
|
|
1051
|
+
json_match = re.search(r'```(?:json)?\s*([\s\S]*?)```', text_to_parse)
|
|
1052
|
+
if json_match:
|
|
1053
|
+
text_to_parse = json_match.group(1).strip()
|
|
1054
|
+
|
|
1055
|
+
# Try to find JSON object if response has extra text
|
|
1056
|
+
if not text_to_parse.startswith("{"):
|
|
1057
|
+
brace_match = re.search(r'\{[\s\S]*\}', text_to_parse)
|
|
1058
|
+
if brace_match:
|
|
1059
|
+
text_to_parse = brace_match.group(0)
|
|
1060
|
+
|
|
1061
|
+
data = json.loads(text_to_parse)
|
|
1062
|
+
facts = data.get("facts", [])
|
|
1063
|
+
if not isinstance(facts, list):
|
|
1064
|
+
facts = []
|
|
1065
|
+
logger.info("Fact extraction: got %d facts from chunk", len(facts))
|
|
1066
|
+
return {"facts": facts}
|
|
1067
|
+
|
|
1068
|
+
except json.JSONDecodeError as jde:
|
|
1069
|
+
logger.warning("Fact extraction: JSON parse error (attempt %d): %s -- raw[:200]: %s",
|
|
1070
|
+
attempt + 1, jde, text_to_parse[:200] if text_to_parse else "empty")
|
|
1071
|
+
if attempt < 2:
|
|
1072
|
+
continue
|
|
1073
|
+
return {"facts": []}
|
|
1074
|
+
except Exception as e:
|
|
1075
|
+
err_str = str(e).lower()
|
|
1076
|
+
if attempt < 2 and ("rate" in err_str or "429" in err_str or "limit" in err_str or "timeout" in err_str):
|
|
1077
|
+
import time
|
|
1078
|
+
wait = (attempt + 1) * 5
|
|
1079
|
+
logger.info("Rate limited on fact extraction (attempt %d), retrying in %ds", attempt + 1, wait)
|
|
1080
|
+
time.sleep(wait)
|
|
1081
|
+
continue
|
|
1082
|
+
logger.warning("Fact extraction failed for chunk: %s", e)
|
|
1083
|
+
return {"facts": []}
|
|
1084
|
+
|
|
1085
|
+
# ------------------------------------------------------------------
|
|
1086
|
+
# INDEX_BM25 — index facts into BM25 sparse search
|
|
1087
|
+
# ------------------------------------------------------------------
|
|
1088
|
+
|
|
1089
|
+
def _stage_index_bm25(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1090
|
+
"""Index extracted facts into the BM25/Whoosh search engine.
|
|
1091
|
+
|
|
1092
|
+
Each fact gets a BM25 index entry. The fact's graph node stores
|
|
1093
|
+
a ``bm25_ref`` pointing to the indexed document ID.
|
|
1094
|
+
"""
|
|
1095
|
+
if not ctx.facts:
|
|
1096
|
+
return {"indexed": 0, "reason": "no_facts"}
|
|
1097
|
+
|
|
1098
|
+
try:
|
|
1099
|
+
from ..search.whoosh_search import WhooshSearchEngine, WhooshConfig
|
|
1100
|
+
except ImportError:
|
|
1101
|
+
return {"indexed": 0, "reason": "whoosh_not_available"}
|
|
1102
|
+
|
|
1103
|
+
graph_ns = ctx.graph_namespace or "default"
|
|
1104
|
+
# Sanitize namespace for use as directory name (Windows forbids ':' in paths)
|
|
1105
|
+
safe_ns = graph_ns.replace(":", "_")
|
|
1106
|
+
index_dir = os.path.join("contextcore_data", "bm25_index", safe_ns)
|
|
1107
|
+
engine = WhooshSearchEngine(WhooshConfig(index_dir=index_dir))
|
|
1108
|
+
|
|
1109
|
+
count = 0
|
|
1110
|
+
for fact in ctx.facts:
|
|
1111
|
+
fid = fact["id"]
|
|
1112
|
+
engine.index_node(
|
|
1113
|
+
node_id=fid,
|
|
1114
|
+
label="Fact",
|
|
1115
|
+
properties={
|
|
1116
|
+
"name": fact["statement"],
|
|
1117
|
+
"statement": fact["statement"],
|
|
1118
|
+
"type": fact["type"],
|
|
1119
|
+
"confidence": fact["confidence"],
|
|
1120
|
+
},
|
|
1121
|
+
)
|
|
1122
|
+
# Store the BM25 reference back on the fact
|
|
1123
|
+
fact["bm25_ref"] = f"bm25://{graph_ns}/{fid}"
|
|
1124
|
+
count += 1
|
|
1125
|
+
|
|
1126
|
+
ctx.bm25_indexed = count
|
|
1127
|
+
self._emit_sub_step(ctx, f"BM25-indexed {count} facts (backend={engine._backend})")
|
|
1128
|
+
|
|
1129
|
+
# Keep engine reference for potential later querying
|
|
1130
|
+
ctx.pipeline_params["_bm25_engine"] = engine
|
|
1131
|
+
|
|
1132
|
+
# Update BM25Index pointer node in graph
|
|
1133
|
+
try:
|
|
1134
|
+
if graph_ns and self.graph_registry:
|
|
1135
|
+
db = self.graph_registry.get_graph(graph_ns)
|
|
1136
|
+
if db:
|
|
1137
|
+
for n in db.get_all_nodes():
|
|
1138
|
+
if getattr(n, "label", "") == "BM25Index":
|
|
1139
|
+
n.properties["status"] = "active"
|
|
1140
|
+
n.properties["count"] = count
|
|
1141
|
+
n.properties["backend"] = engine._backend
|
|
1142
|
+
n.properties["index_path"] = index_dir
|
|
1143
|
+
db.add_node(n, write_through=True)
|
|
1144
|
+
break
|
|
1145
|
+
except Exception:
|
|
1146
|
+
pass
|
|
1147
|
+
|
|
1148
|
+
# Also build comprehensive indexes (keyword + full BM25 + context state)
|
|
1149
|
+
# This indexes ALL node types, not just facts
|
|
1150
|
+
try:
|
|
1151
|
+
from ..core.graph_intelligence import build_indexes
|
|
1152
|
+
graph_db = None
|
|
1153
|
+
if self.graph_registry:
|
|
1154
|
+
graph_db = self.graph_registry.get_graph_for_request(graph_ns)
|
|
1155
|
+
if graph_db:
|
|
1156
|
+
idx_result = build_indexes(graph_db, graph_ns)
|
|
1157
|
+
self._emit_sub_step(ctx, f"Full index build: {idx_result.get('total_ms', '?')}ms")
|
|
1158
|
+
except Exception as e:
|
|
1159
|
+
logger.debug("[INDEX_BM25] Full index build failed: %s", e)
|
|
1160
|
+
|
|
1161
|
+
return {"indexed": count, "backend": engine._backend}
|
|
1162
|
+
|
|
1163
|
+
# ------------------------------------------------------------------
|
|
1164
|
+
# STORE_VECTORS — store passage embeddings in vector DB
|
|
1165
|
+
# ------------------------------------------------------------------
|
|
1166
|
+
|
|
1167
|
+
def _stage_store_vectors(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1168
|
+
"""Store passage chunk embeddings in the vector DB.
|
|
1169
|
+
|
|
1170
|
+
The full passage text is stored as vector metadata. The graph
|
|
1171
|
+
node keeps only a ``vector_ref`` pointer and a content preview.
|
|
1172
|
+
"""
|
|
1173
|
+
if not ctx.chunks:
|
|
1174
|
+
return {"stored": 0, "reason": "no_chunks"}
|
|
1175
|
+
|
|
1176
|
+
# Only process chunks that have embeddings
|
|
1177
|
+
# Collect embedded items from chunks AND nodes
|
|
1178
|
+
embedded_chunks = [c for c in ctx.chunks if c.get("embedding")]
|
|
1179
|
+
embedded_nodes = [n for n in ctx.nodes if n.get("properties", {}).get("embedding")]
|
|
1180
|
+
# Convert nodes to chunk-like format for storage
|
|
1181
|
+
for node in embedded_nodes:
|
|
1182
|
+
props = node.get("properties", {})
|
|
1183
|
+
embedded_chunks.append({
|
|
1184
|
+
"id": node.get("id", ""),
|
|
1185
|
+
"embedding": props["embedding"],
|
|
1186
|
+
"text": props.get("content") or props.get("description") or props.get("name", ""),
|
|
1187
|
+
"label": node.get("label", ""),
|
|
1188
|
+
})
|
|
1189
|
+
if not embedded_chunks:
|
|
1190
|
+
return {"stored": 0, "reason": "no_embeddings_on_chunks_or_nodes"}
|
|
1191
|
+
|
|
1192
|
+
try:
|
|
1193
|
+
from ..vector.vector_db_manager import get_vector_db_manager
|
|
1194
|
+
except ImportError:
|
|
1195
|
+
return {"stored": 0, "reason": "vector_db_not_available"}
|
|
1196
|
+
|
|
1197
|
+
graph_ns = ctx.graph_namespace or "default"
|
|
1198
|
+
|
|
1199
|
+
try:
|
|
1200
|
+
# Determine embedding dimension from first chunk
|
|
1201
|
+
sample_emb = embedded_chunks[0]["embedding"]
|
|
1202
|
+
if hasattr(sample_emb, "tolist"):
|
|
1203
|
+
sample_emb = sample_emb.tolist()
|
|
1204
|
+
dim = len(sample_emb)
|
|
1205
|
+
|
|
1206
|
+
manager = get_vector_db_manager()
|
|
1207
|
+
safe_ns = graph_ns.replace(":", "_")
|
|
1208
|
+
collection = f"{safe_ns}_passages"
|
|
1209
|
+
store = manager.create_store(
|
|
1210
|
+
dimension=dim, metric="cosine",
|
|
1211
|
+
collection_name=collection,
|
|
1212
|
+
)
|
|
1213
|
+
|
|
1214
|
+
# Batch all embeddings for a single add_vectors call
|
|
1215
|
+
ids = []
|
|
1216
|
+
vectors = []
|
|
1217
|
+
metadatas = []
|
|
1218
|
+
for chunk in embedded_chunks:
|
|
1219
|
+
cid = chunk["id"]
|
|
1220
|
+
embedding = chunk["embedding"]
|
|
1221
|
+
if hasattr(embedding, "tolist"):
|
|
1222
|
+
embedding = embedding.tolist()
|
|
1223
|
+
ids.append(cid)
|
|
1224
|
+
vectors.append(embedding)
|
|
1225
|
+
metadatas.append({
|
|
1226
|
+
"text": chunk["text"],
|
|
1227
|
+
"chunk_index": chunk.get("index", 0),
|
|
1228
|
+
"char_count": chunk.get("char_count", len(chunk["text"])),
|
|
1229
|
+
})
|
|
1230
|
+
|
|
1231
|
+
store.add_vectors(node_ids=ids, vectors=vectors, metadata_list=metadatas)
|
|
1232
|
+
|
|
1233
|
+
# Persist vector store to disk (namespace directory)
|
|
1234
|
+
try:
|
|
1235
|
+
from pathlib import Path
|
|
1236
|
+
vec_dir = Path(f"contextcore_data/namespaces/{graph_ns}/vectors")
|
|
1237
|
+
vec_dir.mkdir(parents=True, exist_ok=True)
|
|
1238
|
+
store.save(str(vec_dir / f"{collection}.npz"))
|
|
1239
|
+
logger.info("Persisted %d vectors to %s", len(ids), vec_dir / f"{collection}.npz")
|
|
1240
|
+
except Exception as save_err:
|
|
1241
|
+
logger.warning("Failed to persist vectors: %s", save_err)
|
|
1242
|
+
|
|
1243
|
+
# Replace full text on chunks with preview + ref
|
|
1244
|
+
for chunk in embedded_chunks:
|
|
1245
|
+
chunk["vector_ref"] = f"vec://{collection}/{chunk['id']}"
|
|
1246
|
+
chunk["content_preview"] = chunk["text"][:200]
|
|
1247
|
+
|
|
1248
|
+
ctx.vectors_stored = len(ids)
|
|
1249
|
+
self._emit_sub_step(ctx, f"Stored {len(ids)} passage embeddings in vector DB")
|
|
1250
|
+
|
|
1251
|
+
# Update VectorIndex pointer node in graph
|
|
1252
|
+
try:
|
|
1253
|
+
graph_ns = getattr(ctx, "graph_namespace", "") or getattr(ctx, "namespace", "")
|
|
1254
|
+
if graph_ns and self.graph_registry:
|
|
1255
|
+
db = self.graph_registry.get_graph(graph_ns)
|
|
1256
|
+
if db:
|
|
1257
|
+
for n in db.get_all_nodes():
|
|
1258
|
+
if getattr(n, "label", "") == "VectorIndex":
|
|
1259
|
+
n.properties["status"] = "active"
|
|
1260
|
+
n.properties["count"] = len(ids)
|
|
1261
|
+
n.properties["collection"] = collection
|
|
1262
|
+
n.properties["embedding_model"] = ctx.pipeline_params.get("_resolved_embedding_model", "")
|
|
1263
|
+
db.add_node(n, write_through=True)
|
|
1264
|
+
break
|
|
1265
|
+
except Exception:
|
|
1266
|
+
pass
|
|
1267
|
+
|
|
1268
|
+
return {"stored": len(ids), "collection": collection}
|
|
1269
|
+
|
|
1270
|
+
except Exception as e:
|
|
1271
|
+
logger.warning("Vector store failed: %s", e)
|
|
1272
|
+
# Non-fatal: embeddings stay on chunk nodes as fallback
|
|
1273
|
+
return {"stored": 0, "error": str(e)}
|
|
1274
|
+
|
|
1275
|
+
# ------------------------------------------------------------------
|
|
1276
|
+
# EMBED — generate embeddings for nodes and chunks
|
|
1277
|
+
# ------------------------------------------------------------------
|
|
1278
|
+
|
|
1279
|
+
def _stage_embed(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1280
|
+
"""Embed chunks and content-rich nodes for vector search.
|
|
1281
|
+
|
|
1282
|
+
Uses batched embedding (embed_batch) when available — sends up to
|
|
1283
|
+
32 texts per API call instead of 1, reducing API calls by 30x.
|
|
1284
|
+
"""
|
|
1285
|
+
embedding_model = ctx.pipeline_params.get("embedding_model")
|
|
1286
|
+
emb = self._get_embeddings(embedding_model)
|
|
1287
|
+
|
|
1288
|
+
if not emb or not embedding_model:
|
|
1289
|
+
return {"embedded": 0, "reason": "no_embedding_configured"}
|
|
1290
|
+
|
|
1291
|
+
BATCH_SIZE = 32
|
|
1292
|
+
has_batch = hasattr(emb, "embed_batch")
|
|
1293
|
+
|
|
1294
|
+
# Collect all texts to embed
|
|
1295
|
+
items = [] # list of (target_dict, key_for_embedding, text)
|
|
1296
|
+
|
|
1297
|
+
# 1. Chunks
|
|
1298
|
+
for chunk in ctx.chunks:
|
|
1299
|
+
text = (chunk.get("text", "") or chunk.get("content", ""))[:8000]
|
|
1300
|
+
if text.strip():
|
|
1301
|
+
items.append((chunk, "embedding", text, True)) # True = is_chunk
|
|
1302
|
+
|
|
1303
|
+
# 2. Content-rich nodes
|
|
1304
|
+
embeddable_labels = {"Passage", "TextChunk", "Fact", "Feature", "Requirement",
|
|
1305
|
+
"Knowledge", "Finding", "Insight", "Decision", "Document"}
|
|
1306
|
+
for node in ctx.nodes:
|
|
1307
|
+
props = node.get("properties", {})
|
|
1308
|
+
label = node.get("label", "")
|
|
1309
|
+
if props.get("embedding"):
|
|
1310
|
+
continue
|
|
1311
|
+
text = ""
|
|
1312
|
+
if label in embeddable_labels:
|
|
1313
|
+
text = (props.get("content") or props.get("description") or
|
|
1314
|
+
props.get("statement") or props.get("name", ""))
|
|
1315
|
+
elif props.get("content") or props.get("description"):
|
|
1316
|
+
text = props.get("content") or props.get("description", "")
|
|
1317
|
+
text = (text or "")[:8000]
|
|
1318
|
+
if text.strip() and len(text.strip()) >= 20:
|
|
1319
|
+
items.append((props, "embedding", text, False))
|
|
1320
|
+
|
|
1321
|
+
if not items:
|
|
1322
|
+
return {"embedded": 0, "reason": "nothing_to_embed"}
|
|
1323
|
+
|
|
1324
|
+
self._emit_sub_step(ctx, f"Embedding {len(items)} items (batch_size={BATCH_SIZE})")
|
|
1325
|
+
count = 0
|
|
1326
|
+
api_calls = 0
|
|
1327
|
+
|
|
1328
|
+
if has_batch:
|
|
1329
|
+
# Batched embedding — much faster
|
|
1330
|
+
for i in range(0, len(items), BATCH_SIZE):
|
|
1331
|
+
batch = items[i:i + BATCH_SIZE]
|
|
1332
|
+
texts = [item[2] for item in batch]
|
|
1333
|
+
try:
|
|
1334
|
+
resp = emb.embed_batch(texts, model_name=embedding_model)
|
|
1335
|
+
api_calls += 1
|
|
1336
|
+
embeddings = resp.get("embeddings", [])
|
|
1337
|
+
dim = resp.get("dimension", 0)
|
|
1338
|
+
for j, emb_vec in enumerate(embeddings):
|
|
1339
|
+
if emb_vec and j < len(batch):
|
|
1340
|
+
target, key, _, is_chunk = batch[j]
|
|
1341
|
+
target[key] = emb_vec
|
|
1342
|
+
target["embedding_model"] = embedding_model
|
|
1343
|
+
if is_chunk:
|
|
1344
|
+
target["embedding_dim"] = dim or len(emb_vec)
|
|
1345
|
+
count += 1
|
|
1346
|
+
except Exception as e:
|
|
1347
|
+
logger.warning("Batch embed failed (batch %d): %s", i // BATCH_SIZE, e)
|
|
1348
|
+
# Fall back to individual for this batch
|
|
1349
|
+
for target, key, text, is_chunk in batch:
|
|
1350
|
+
try:
|
|
1351
|
+
resp = emb.embed_text(text, model_name=embedding_model)
|
|
1352
|
+
api_calls += 1
|
|
1353
|
+
if resp.get("success") and resp.get("embedding"):
|
|
1354
|
+
target[key] = resp["embedding"]
|
|
1355
|
+
target["embedding_model"] = embedding_model
|
|
1356
|
+
if is_chunk:
|
|
1357
|
+
target["embedding_dim"] = resp.get("dimension", len(resp["embedding"]))
|
|
1358
|
+
count += 1
|
|
1359
|
+
except Exception:
|
|
1360
|
+
pass
|
|
1361
|
+
else:
|
|
1362
|
+
# Individual embedding (legacy fallback)
|
|
1363
|
+
for target, key, text, is_chunk in items:
|
|
1364
|
+
try:
|
|
1365
|
+
resp = emb.embed_text(text, model_name=embedding_model)
|
|
1366
|
+
api_calls += 1
|
|
1367
|
+
if resp.get("success") and resp.get("embedding"):
|
|
1368
|
+
target[key] = resp["embedding"]
|
|
1369
|
+
target["embedding_model"] = embedding_model
|
|
1370
|
+
if is_chunk:
|
|
1371
|
+
target["embedding_dim"] = resp.get("dimension", len(resp["embedding"]))
|
|
1372
|
+
count += 1
|
|
1373
|
+
except Exception as e:
|
|
1374
|
+
logger.warning("Failed to embed: %s", e)
|
|
1375
|
+
|
|
1376
|
+
ctx.embeddings_count = count
|
|
1377
|
+
if count > 0:
|
|
1378
|
+
ctx.pipeline_params["_resolved_embedding_model"] = embedding_model
|
|
1379
|
+
return {"embedded": count, "api_calls": api_calls, "batch_size": BATCH_SIZE, "embedding_model": embedding_model}
|
|
1380
|
+
|
|
1381
|
+
# ------------------------------------------------------------------
|
|
1382
|
+
# CANONICALIZE — deduplicate nodes by name similarity
|
|
1383
|
+
# ------------------------------------------------------------------
|
|
1384
|
+
|
|
1385
|
+
def _stage_canonicalize(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1386
|
+
"""Deduplicate nodes by canonical name (cross-label), merge edges."""
|
|
1387
|
+
if not ctx.nodes:
|
|
1388
|
+
return {"original": 0, "deduped": 0}
|
|
1389
|
+
|
|
1390
|
+
original_count = len(ctx.nodes)
|
|
1391
|
+
canonical_map: Dict[str, str] = {} # name_lower -> chosen node id
|
|
1392
|
+
deduped_nodes: Dict[str, Dict] = {} # node_id -> node dict
|
|
1393
|
+
id_remap: Dict[str, str] = {} # old_id -> canonical_id
|
|
1394
|
+
|
|
1395
|
+
entity_labels = {"Person", "Organization", "Location", "Event", "Entity"}
|
|
1396
|
+
|
|
1397
|
+
for node in ctx.nodes:
|
|
1398
|
+
name = node.get("properties", {}).get("name", "")
|
|
1399
|
+
label = node.get("label", "Entity")
|
|
1400
|
+
|
|
1401
|
+
# Non-entity nodes pass through unchanged
|
|
1402
|
+
if label not in entity_labels:
|
|
1403
|
+
deduped_nodes[node["id"]] = node
|
|
1404
|
+
id_remap[node["id"]] = node["id"]
|
|
1405
|
+
continue
|
|
1406
|
+
|
|
1407
|
+
# Canonical key: name only (allows cross-label merge like "Iran" Location + Organization)
|
|
1408
|
+
canonical_key = name.strip().lower()
|
|
1409
|
+
if not canonical_key or len(canonical_key) < 2:
|
|
1410
|
+
deduped_nodes[node["id"]] = node
|
|
1411
|
+
id_remap[node["id"]] = node["id"]
|
|
1412
|
+
continue
|
|
1413
|
+
|
|
1414
|
+
if canonical_key in canonical_map:
|
|
1415
|
+
existing_id = canonical_map[canonical_key]
|
|
1416
|
+
id_remap[node["id"]] = existing_id
|
|
1417
|
+
existing = deduped_nodes[existing_id]
|
|
1418
|
+
existing_mc = existing.get("properties", {}).get("mention_count", 0) or 0
|
|
1419
|
+
node_mc = node.get("properties", {}).get("mention_count", 0) or 0
|
|
1420
|
+
existing["properties"]["mention_count"] = existing_mc + node_mc
|
|
1421
|
+
else:
|
|
1422
|
+
canonical_map[canonical_key] = node["id"]
|
|
1423
|
+
deduped_nodes[node["id"]] = node
|
|
1424
|
+
id_remap[node["id"]] = node["id"]
|
|
1425
|
+
|
|
1426
|
+
# Remap edge source/target IDs and deduplicate edges
|
|
1427
|
+
seen_edges = set()
|
|
1428
|
+
deduped_edges = []
|
|
1429
|
+
for edge in ctx.edges:
|
|
1430
|
+
src = id_remap.get(edge.get("source", ""), edge.get("source", ""))
|
|
1431
|
+
tgt = id_remap.get(edge.get("target", ""), edge.get("target", ""))
|
|
1432
|
+
if src == tgt:
|
|
1433
|
+
continue # Skip self-loops created by merge
|
|
1434
|
+
edge_key = f"{src}:{tgt}:{edge.get('label', '')}"
|
|
1435
|
+
if edge_key in seen_edges:
|
|
1436
|
+
continue
|
|
1437
|
+
seen_edges.add(edge_key)
|
|
1438
|
+
deduped_edges.append({
|
|
1439
|
+
**edge,
|
|
1440
|
+
"source": src,
|
|
1441
|
+
"target": tgt,
|
|
1442
|
+
})
|
|
1443
|
+
|
|
1444
|
+
ctx.nodes = list(deduped_nodes.values())
|
|
1445
|
+
ctx.edges = deduped_edges
|
|
1446
|
+
|
|
1447
|
+
return {
|
|
1448
|
+
"original_nodes": original_count,
|
|
1449
|
+
"deduped_nodes": len(ctx.nodes),
|
|
1450
|
+
"removed": original_count - len(ctx.nodes),
|
|
1451
|
+
"original_edges": len(ctx.edges),
|
|
1452
|
+
}
|
|
1453
|
+
|
|
1454
|
+
# ------------------------------------------------------------------
|
|
1455
|
+
# ENHANCE_GRAPH — enrich graph with additional relationships
|
|
1456
|
+
# ------------------------------------------------------------------
|
|
1457
|
+
|
|
1458
|
+
def _stage_enhance_graph(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1459
|
+
"""Enhance graph by adding inferred edges (co-occurrence, hierarchy)."""
|
|
1460
|
+
if not ctx.nodes:
|
|
1461
|
+
return {"edges_added": 0}
|
|
1462
|
+
|
|
1463
|
+
edges_added = 0
|
|
1464
|
+
|
|
1465
|
+
# Co-occurrence: if two entities appear in the same chunk, link them
|
|
1466
|
+
if ctx.chunks:
|
|
1467
|
+
chunk_entities: Dict[str, List[str]] = {}
|
|
1468
|
+
for edge in ctx.edges:
|
|
1469
|
+
if edge.get("label") == "MENTIONS":
|
|
1470
|
+
chunk_id = edge["source"]
|
|
1471
|
+
entity_id = edge["target"]
|
|
1472
|
+
chunk_entities.setdefault(chunk_id, []).append(entity_id)
|
|
1473
|
+
|
|
1474
|
+
seen = set()
|
|
1475
|
+
for chunk_id, entities in chunk_entities.items():
|
|
1476
|
+
for i, e1 in enumerate(entities):
|
|
1477
|
+
for e2 in entities[i + 1:]:
|
|
1478
|
+
pair = tuple(sorted([e1, e2]))
|
|
1479
|
+
if pair in seen:
|
|
1480
|
+
continue
|
|
1481
|
+
seen.add(pair)
|
|
1482
|
+
ctx.edges.append({
|
|
1483
|
+
"id": str(uuid.uuid4()),
|
|
1484
|
+
"source": e1,
|
|
1485
|
+
"target": e2,
|
|
1486
|
+
"label": "CO_OCCURS_WITH",
|
|
1487
|
+
"properties": {"inferred": True},
|
|
1488
|
+
})
|
|
1489
|
+
edges_added += 1
|
|
1490
|
+
|
|
1491
|
+
return {"edges_added": edges_added}
|
|
1492
|
+
|
|
1493
|
+
# ------------------------------------------------------------------
|
|
1494
|
+
# PERSIST — save nodes and edges to the graph (via AIQL)
|
|
1495
|
+
# ------------------------------------------------------------------
|
|
1496
|
+
|
|
1497
|
+
def _stage_persist(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1498
|
+
"""Persist the full GraphRAG hierarchy to graph storage via AIQLExecutor.
|
|
1499
|
+
|
|
1500
|
+
Hierarchy created:
|
|
1501
|
+
Document -[HAS_PASSAGE]-> Passage -[STATES]-> Fact -[MENTIONS]-> Entity
|
|
1502
|
+
| |
|
|
1503
|
+
vector_ref bm25_ref
|
|
1504
|
+
(vector DB) (BM25 index)
|
|
1505
|
+
|
|
1506
|
+
Falls back to the simpler Document-[CONTAINS]->TextChunk layout
|
|
1507
|
+
when no facts are present (legacy pipelines).
|
|
1508
|
+
|
|
1509
|
+
All graph mutations go through ``AIQLExecutor.bulk_ingest()`` so they
|
|
1510
|
+
participate in time-travel versioning, audit, and are consistent with
|
|
1511
|
+
AIQL CREATE NODE / CREATE EDGE semantics.
|
|
1512
|
+
"""
|
|
1513
|
+
if not self.graph_registry:
|
|
1514
|
+
raise ValueError("No graph_registry configured")
|
|
1515
|
+
|
|
1516
|
+
graph_ns = ctx.graph_namespace
|
|
1517
|
+
if not graph_ns:
|
|
1518
|
+
raise ValueError("No graph_namespace set on PipelineContext")
|
|
1519
|
+
|
|
1520
|
+
aiql = self._get_aiql_executor()
|
|
1521
|
+
if not aiql:
|
|
1522
|
+
raise ValueError("Could not initialize AIQLExecutor")
|
|
1523
|
+
|
|
1524
|
+
# Activate the target namespace
|
|
1525
|
+
aiql.active_namespace = graph_ns
|
|
1526
|
+
|
|
1527
|
+
# Collect all nodes and edges, then send through AIQL in one batch
|
|
1528
|
+
all_nodes: List[Dict[str, Any]] = []
|
|
1529
|
+
all_edges: List[Dict[str, Any]] = []
|
|
1530
|
+
facts_created = 0
|
|
1531
|
+
|
|
1532
|
+
has_facts = bool(ctx.facts)
|
|
1533
|
+
chunk_label = "Passage" if has_facts else "TextChunk"
|
|
1534
|
+
chunk_edge_label = "HAS_PASSAGE" if has_facts else "CONTAINS"
|
|
1535
|
+
|
|
1536
|
+
# ---- 1. Document node ----
|
|
1537
|
+
doc_id = None
|
|
1538
|
+
if ctx.chunks:
|
|
1539
|
+
doc_id = str(uuid.uuid4())
|
|
1540
|
+
title = ctx.source_filename or "Untitled"
|
|
1541
|
+
self._emit_sub_step(ctx, f"Building Document node: {title}")
|
|
1542
|
+
all_nodes.append({
|
|
1543
|
+
"id": doc_id,
|
|
1544
|
+
"label": "Document",
|
|
1545
|
+
"properties": {
|
|
1546
|
+
"name": title,
|
|
1547
|
+
"source": ctx.source_url or ctx.source_filename or "upload",
|
|
1548
|
+
"chunk_count": len(ctx.chunks),
|
|
1549
|
+
"fact_count": len(ctx.facts) if has_facts else 0,
|
|
1550
|
+
},
|
|
1551
|
+
})
|
|
1552
|
+
|
|
1553
|
+
# ---- 2. Passage / TextChunk nodes ----
|
|
1554
|
+
self._emit_sub_step(ctx, f"Building {len(ctx.chunks)} {chunk_label} nodes")
|
|
1555
|
+
for chunk in ctx.chunks:
|
|
1556
|
+
props = {
|
|
1557
|
+
"name": f"{title} ({chunk_label.lower()} {chunk['index'] + 1}/{len(ctx.chunks)})",
|
|
1558
|
+
"char_count": chunk.get("char_count", len(chunk["text"])),
|
|
1559
|
+
"chunk_index": chunk["index"],
|
|
1560
|
+
"document_id": doc_id,
|
|
1561
|
+
}
|
|
1562
|
+
if chunk.get("vector_ref"):
|
|
1563
|
+
# Text lives in vector store — graph stores only the pointer
|
|
1564
|
+
props["vector_ref"] = chunk["vector_ref"]
|
|
1565
|
+
else:
|
|
1566
|
+
# No vector store — keep full text in graph node
|
|
1567
|
+
props["content"] = chunk["text"]
|
|
1568
|
+
|
|
1569
|
+
if "embedding" in chunk and not chunk.get("vector_ref"):
|
|
1570
|
+
props["embedding"] = chunk["embedding"]
|
|
1571
|
+
props["embedding_model"] = chunk.get("embedding_model", "")
|
|
1572
|
+
props["embedding_dim"] = chunk.get("embedding_dim", 0)
|
|
1573
|
+
|
|
1574
|
+
# Dedup hashes (set by DEDUP stage)
|
|
1575
|
+
if chunk.get("content_hash"):
|
|
1576
|
+
props["content_hash"] = chunk["content_hash"]
|
|
1577
|
+
if chunk.get("normalised_hash"):
|
|
1578
|
+
props["normalised_hash"] = chunk["normalised_hash"]
|
|
1579
|
+
if chunk.get("simhash"):
|
|
1580
|
+
props["simhash"] = chunk["simhash"]
|
|
1581
|
+
|
|
1582
|
+
all_nodes.append({"id": chunk["id"], "label": chunk_label, "properties": props})
|
|
1583
|
+
|
|
1584
|
+
# Document → Passage edge
|
|
1585
|
+
all_edges.append({
|
|
1586
|
+
"id": str(uuid.uuid4()),
|
|
1587
|
+
"source": doc_id,
|
|
1588
|
+
"target": chunk["id"],
|
|
1589
|
+
"label": chunk_edge_label,
|
|
1590
|
+
"properties": {"chunk_index": chunk["index"]},
|
|
1591
|
+
})
|
|
1592
|
+
|
|
1593
|
+
# ---- 3. Fact nodes + Passage→Fact edges ----
|
|
1594
|
+
if has_facts:
|
|
1595
|
+
self._emit_sub_step(ctx, f"Building {len(ctx.facts)} Fact nodes")
|
|
1596
|
+
chunk_fact_links = ctx.pipeline_params.get("_chunk_fact_links", [])
|
|
1597
|
+
fact_keys = ctx.pipeline_params.get("_fact_keys", {})
|
|
1598
|
+
|
|
1599
|
+
for fact in ctx.facts:
|
|
1600
|
+
fid = fact["id"]
|
|
1601
|
+
fprops = {
|
|
1602
|
+
"name": fact["statement"][:80],
|
|
1603
|
+
"statement": fact["statement"],
|
|
1604
|
+
"fact_type": fact["type"],
|
|
1605
|
+
"confidence": fact["confidence"],
|
|
1606
|
+
"mention_count": fact.get("mention_count", 1),
|
|
1607
|
+
"extraction_method": "llm_extraction",
|
|
1608
|
+
"source": ctx.source_url or ctx.source_filename or "llm_extraction",
|
|
1609
|
+
}
|
|
1610
|
+
if fact.get("bm25_ref"):
|
|
1611
|
+
fprops["bm25_ref"] = fact["bm25_ref"]
|
|
1612
|
+
|
|
1613
|
+
all_nodes.append({"id": fid, "label": "Fact", "properties": fprops})
|
|
1614
|
+
facts_created += 1
|
|
1615
|
+
|
|
1616
|
+
# Passage → Fact STATES edges (deduplicated)
|
|
1617
|
+
seen_states = set()
|
|
1618
|
+
for chunk_id, fact_key in chunk_fact_links:
|
|
1619
|
+
fid = fact_keys.get(fact_key)
|
|
1620
|
+
if not fid:
|
|
1621
|
+
continue
|
|
1622
|
+
edge_key = f"{chunk_id}:{fid}"
|
|
1623
|
+
if edge_key in seen_states:
|
|
1624
|
+
continue
|
|
1625
|
+
seen_states.add(edge_key)
|
|
1626
|
+
all_edges.append({
|
|
1627
|
+
"id": str(uuid.uuid4()),
|
|
1628
|
+
"source": chunk_id,
|
|
1629
|
+
"target": fid,
|
|
1630
|
+
"label": "STATES",
|
|
1631
|
+
"properties": {},
|
|
1632
|
+
})
|
|
1633
|
+
|
|
1634
|
+
# Fact → Entity MENTIONS edges
|
|
1635
|
+
entity_name_to_id = {}
|
|
1636
|
+
for node in ctx.nodes:
|
|
1637
|
+
ename = node.get("properties", {}).get("name", "").lower()
|
|
1638
|
+
if ename:
|
|
1639
|
+
entity_name_to_id[ename] = node["id"]
|
|
1640
|
+
|
|
1641
|
+
seen_fact_ent = set()
|
|
1642
|
+
for fact in ctx.facts:
|
|
1643
|
+
fid = fact["id"]
|
|
1644
|
+
for ename in fact.get("entity_names", []):
|
|
1645
|
+
eid = entity_name_to_id.get(ename.lower())
|
|
1646
|
+
if not eid:
|
|
1647
|
+
continue
|
|
1648
|
+
edge_key = f"{fid}:{eid}"
|
|
1649
|
+
if edge_key in seen_fact_ent:
|
|
1650
|
+
continue
|
|
1651
|
+
seen_fact_ent.add(edge_key)
|
|
1652
|
+
all_edges.append({
|
|
1653
|
+
"id": str(uuid.uuid4()),
|
|
1654
|
+
"source": fid,
|
|
1655
|
+
"target": eid,
|
|
1656
|
+
"label": "MENTIONS",
|
|
1657
|
+
"properties": {},
|
|
1658
|
+
})
|
|
1659
|
+
|
|
1660
|
+
# ---- 4. Entity / tabular nodes ----
|
|
1661
|
+
if ctx.nodes:
|
|
1662
|
+
self._emit_sub_step(ctx, f"Building {len(ctx.nodes)} entity nodes")
|
|
1663
|
+
all_nodes.extend(ctx.nodes)
|
|
1664
|
+
|
|
1665
|
+
# ---- 5. Relationship / other edges ----
|
|
1666
|
+
all_edges.extend(ctx.edges)
|
|
1667
|
+
|
|
1668
|
+
# ---- 5b. Tag category + IN_CATEGORY edges ----
|
|
1669
|
+
category_boundary_id = ctx.pipeline_params.get("_category_boundary_id")
|
|
1670
|
+
if category_boundary_id:
|
|
1671
|
+
from ..context.boundaries import tag_nodes_with_category, build_category_edges, BOUNDARY_NODE_LABELS
|
|
1672
|
+
tag_nodes_with_category(all_nodes, "knowledge_base")
|
|
1673
|
+
cat_edges = build_category_edges(all_nodes, category_boundary_id)
|
|
1674
|
+
all_edges.extend(cat_edges)
|
|
1675
|
+
self._emit_sub_step(ctx, f"Tagged {len(all_nodes)} nodes as knowledge_base, {len(cat_edges)} IN_CATEGORY edges")
|
|
1676
|
+
|
|
1677
|
+
# ---- 6. Bulk ingest through AIQL ----
|
|
1678
|
+
self._emit_sub_step(ctx, f"Persisting via AIQL: {len(all_nodes)} nodes, {len(all_edges)} edges")
|
|
1679
|
+
|
|
1680
|
+
def _on_progress(msg, prog):
|
|
1681
|
+
self._emit_sub_step(ctx, msg, prog)
|
|
1682
|
+
|
|
1683
|
+
result = aiql.bulk_ingest(
|
|
1684
|
+
namespace=graph_ns,
|
|
1685
|
+
nodes=all_nodes,
|
|
1686
|
+
edges=all_edges,
|
|
1687
|
+
merge_existing=True,
|
|
1688
|
+
on_progress=_on_progress,
|
|
1689
|
+
)
|
|
1690
|
+
|
|
1691
|
+
nodes_created = result.get("nodes_created", 0)
|
|
1692
|
+
nodes_merged = result.get("nodes_merged", 0)
|
|
1693
|
+
edges_created = result.get("edges_created", 0)
|
|
1694
|
+
errors = result.get("errors", [])
|
|
1695
|
+
|
|
1696
|
+
# ---- 7. Store embedding model on graph metadata ----
|
|
1697
|
+
emb_model = ctx.pipeline_params.get("_resolved_embedding_model") or ctx.pipeline_params.get("embedding_model")
|
|
1698
|
+
if emb_model and self.graph_registry:
|
|
1699
|
+
meta = self.graph_registry.metadata.get(graph_ns)
|
|
1700
|
+
if meta:
|
|
1701
|
+
meta.embedding_model = emb_model
|
|
1702
|
+
try:
|
|
1703
|
+
self.graph_registry._save_metadata()
|
|
1704
|
+
except Exception:
|
|
1705
|
+
pass
|
|
1706
|
+
|
|
1707
|
+
self._emit_sub_step(ctx, f"Saved: {nodes_created} created, {nodes_merged} merged, {edges_created} edges")
|
|
1708
|
+
|
|
1709
|
+
# Build search indexes (keyword, BM25, context state) in background
|
|
1710
|
+
# so they're ready before any agent searches
|
|
1711
|
+
import threading
|
|
1712
|
+
def _post_persist_index_build():
|
|
1713
|
+
try:
|
|
1714
|
+
from ..core.graph_intelligence import build_indexes
|
|
1715
|
+
graph_db = aiql._graph if hasattr(aiql, '_graph') else None
|
|
1716
|
+
if not graph_db and self.graph_registry:
|
|
1717
|
+
graph_db = self.graph_registry.get_graph_for_request(graph_ns)
|
|
1718
|
+
if graph_db:
|
|
1719
|
+
result = build_indexes(graph_db, graph_ns)
|
|
1720
|
+
logger.info("[PERSIST] Post-ingest index build: %s", result)
|
|
1721
|
+
except Exception as e:
|
|
1722
|
+
logger.debug("[PERSIST] Index build failed: %s", e)
|
|
1723
|
+
|
|
1724
|
+
t = threading.Thread(target=_post_persist_index_build, daemon=True, name=f"index-{graph_ns[:12]}")
|
|
1725
|
+
t.start()
|
|
1726
|
+
self._emit_sub_step(ctx, "Index build started (background)")
|
|
1727
|
+
|
|
1728
|
+
return {
|
|
1729
|
+
"nodes_created": nodes_created,
|
|
1730
|
+
"nodes_merged": nodes_merged,
|
|
1731
|
+
"edges_created": edges_created,
|
|
1732
|
+
"facts_created": facts_created,
|
|
1733
|
+
"errors": errors[:10] if errors else [],
|
|
1734
|
+
}
|
|
1735
|
+
|
|
1736
|
+
# ------------------------------------------------------------------
|
|
1737
|
+
# VALIDATE_SCHEMA — validate nodes/edges against a GraphSchema
|
|
1738
|
+
# ------------------------------------------------------------------
|
|
1739
|
+
|
|
1740
|
+
def _stage_validate_schema(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1741
|
+
"""Validate extracted nodes/edges against a GraphSchema definition."""
|
|
1742
|
+
schema_yaml = ctx.pipeline_params.get("schema_yaml", "")
|
|
1743
|
+
if not schema_yaml:
|
|
1744
|
+
return {"skipped": True, "reason": "no_schema_provided"}
|
|
1745
|
+
|
|
1746
|
+
try:
|
|
1747
|
+
import yaml
|
|
1748
|
+
from ..schema.schema_parser import SchemaParser
|
|
1749
|
+
from .schema_validator import SchemaValidator
|
|
1750
|
+
|
|
1751
|
+
# Parse schema from YAML string
|
|
1752
|
+
import tempfile, os
|
|
1753
|
+
fd, tmp = tempfile.mkstemp(suffix=".yaml", prefix="schema_")
|
|
1754
|
+
os.close(fd)
|
|
1755
|
+
try:
|
|
1756
|
+
with open(tmp, "w", encoding="utf-8") as f:
|
|
1757
|
+
f.write(schema_yaml)
|
|
1758
|
+
parser = SchemaParser(tmp)
|
|
1759
|
+
schema = parser.parse()
|
|
1760
|
+
finally:
|
|
1761
|
+
os.unlink(tmp)
|
|
1762
|
+
|
|
1763
|
+
strict = ctx.pipeline_params.get("schema_strict", False)
|
|
1764
|
+
validator = SchemaValidator(schema)
|
|
1765
|
+
result = validator.validate(ctx.nodes, ctx.edges, strict=strict)
|
|
1766
|
+
|
|
1767
|
+
# Replace context nodes/edges with validated ones
|
|
1768
|
+
ctx.nodes = result.valid_nodes
|
|
1769
|
+
ctx.edges = result.valid_edges
|
|
1770
|
+
|
|
1771
|
+
return result.to_dict()
|
|
1772
|
+
|
|
1773
|
+
except Exception as e:
|
|
1774
|
+
logger.warning("Schema validation failed: %s", e)
|
|
1775
|
+
return {"error": str(e), "nodes_kept": len(ctx.nodes), "edges_kept": len(ctx.edges)}
|
|
1776
|
+
|
|
1777
|
+
# ------------------------------------------------------------------
|
|
1778
|
+
# IMPORT_GRAPH — import graph files (GraphML, JSON, RDF)
|
|
1779
|
+
# ------------------------------------------------------------------
|
|
1780
|
+
|
|
1781
|
+
def _stage_import_graph(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1782
|
+
"""Import a graph interchange file (GraphML, RDF/Turtle, JSON-LD, plain JSON)."""
|
|
1783
|
+
from .parsers import GraphParserRegistry
|
|
1784
|
+
|
|
1785
|
+
pages = (ctx.extraction_result or {}).get("pages", [])
|
|
1786
|
+
if not pages:
|
|
1787
|
+
return {"nodes": 0, "edges": 0, "reason": "no_content"}
|
|
1788
|
+
|
|
1789
|
+
text = pages[0].get("text", "")
|
|
1790
|
+
if not text.strip():
|
|
1791
|
+
return {"nodes": 0, "edges": 0, "reason": "empty_content"}
|
|
1792
|
+
|
|
1793
|
+
filename = ctx.source_filename or ""
|
|
1794
|
+
|
|
1795
|
+
try:
|
|
1796
|
+
registry = GraphParserRegistry()
|
|
1797
|
+
parsed_nodes, parsed_edges = registry.parse(text, filename=filename)
|
|
1798
|
+
|
|
1799
|
+
ctx.nodes.extend(parsed_nodes)
|
|
1800
|
+
ctx.edges.extend(parsed_edges)
|
|
1801
|
+
|
|
1802
|
+
return {
|
|
1803
|
+
"nodes": len(parsed_nodes),
|
|
1804
|
+
"edges": len(parsed_edges),
|
|
1805
|
+
"format": filename.rsplit(".", 1)[-1] if "." in filename else "unknown",
|
|
1806
|
+
}
|
|
1807
|
+
except Exception as e:
|
|
1808
|
+
logger.warning("Graph import failed: %s", e)
|
|
1809
|
+
return {"nodes": 0, "edges": 0, "error": str(e)}
|
|
1810
|
+
|
|
1811
|
+
# ------------------------------------------------------------------
|
|
1812
|
+
# SDLC_SCAN — scan a repo directory into SDLC-typed graph nodes
|
|
1813
|
+
# ------------------------------------------------------------------
|
|
1814
|
+
|
|
1815
|
+
def _stage_sdlc_scan(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1816
|
+
"""Scan a repo directory into SDLC-typed graph nodes."""
|
|
1817
|
+
from .universal.operators.sdlc_scan import SDLCScanOperator
|
|
1818
|
+
from .universal.operators.index_bm25 import IndexBM25Operator
|
|
1819
|
+
from .universal.operators.embed import EmbedOperator
|
|
1820
|
+
from .universal.ingest_content import Chunk
|
|
1821
|
+
from .universal.stage_executor import GraphContext as UniversalGraphContext
|
|
1822
|
+
|
|
1823
|
+
repo_path = ctx.source_text or ctx.source_url or ""
|
|
1824
|
+
if not repo_path:
|
|
1825
|
+
return {"nodes": 0, "edges": 0, "reason": "no_repo_path"}
|
|
1826
|
+
|
|
1827
|
+
# Get or create graph for this namespace
|
|
1828
|
+
graph_ns = ctx.graph_namespace or "default"
|
|
1829
|
+
db = None
|
|
1830
|
+
if self.graph_registry:
|
|
1831
|
+
db = self.graph_registry.get_graph(graph_ns)
|
|
1832
|
+
if db is None:
|
|
1833
|
+
db = self.graph_registry.create_graph(graph_ns)
|
|
1834
|
+
|
|
1835
|
+
if db is None:
|
|
1836
|
+
return {"nodes": 0, "edges": 0, "reason": "no_graph"}
|
|
1837
|
+
|
|
1838
|
+
graph_ctx = UniversalGraphContext(db=db, namespace=graph_ns)
|
|
1839
|
+
|
|
1840
|
+
# Pass config + source URL through chunk metadata
|
|
1841
|
+
chunk_meta = {"repo_path": repo_path}
|
|
1842
|
+
if ctx.source_url:
|
|
1843
|
+
chunk_meta["source_url"] = ctx.source_url
|
|
1844
|
+
if ctx.pipeline_params:
|
|
1845
|
+
chunk_meta["pipeline_params"] = ctx.pipeline_params
|
|
1846
|
+
chunks = [Chunk(content="", index=0, metadata=chunk_meta)]
|
|
1847
|
+
|
|
1848
|
+
op = SDLCScanOperator(config=ctx.pipeline_params or {})
|
|
1849
|
+
op.process(chunks, graph_ctx)
|
|
1850
|
+
|
|
1851
|
+
# Build full node dicts for PipelineContext (needed by downstream consumers)
|
|
1852
|
+
full_nodes = []
|
|
1853
|
+
for nid in graph_ctx.node_ids:
|
|
1854
|
+
node = graph_ctx.get_node(nid)
|
|
1855
|
+
if node is not None:
|
|
1856
|
+
full_nodes.append({
|
|
1857
|
+
"id": nid,
|
|
1858
|
+
"label": node.label,
|
|
1859
|
+
"properties": dict(node.properties),
|
|
1860
|
+
})
|
|
1861
|
+
else:
|
|
1862
|
+
full_nodes.append({"id": nid, "label": "", "properties": {}})
|
|
1863
|
+
ctx.nodes = full_nodes
|
|
1864
|
+
|
|
1865
|
+
# Edges are already committed to the graph — no need to track in ctx
|
|
1866
|
+
ctx.edges = []
|
|
1867
|
+
|
|
1868
|
+
# Run BM25 indexing and embedding inline (nodes already in graph)
|
|
1869
|
+
try:
|
|
1870
|
+
IndexBM25Operator().process([], graph_ctx)
|
|
1871
|
+
except Exception as exc:
|
|
1872
|
+
logger.debug("[SDLC_SCAN] BM25 indexing skipped: %s", exc)
|
|
1873
|
+
|
|
1874
|
+
try:
|
|
1875
|
+
EmbedOperator().process([], graph_ctx)
|
|
1876
|
+
except Exception as exc:
|
|
1877
|
+
logger.debug("[SDLC_SCAN] Embed skipped: %s", exc)
|
|
1878
|
+
|
|
1879
|
+
# Build Context Units from SDLC nodes
|
|
1880
|
+
try:
|
|
1881
|
+
from .universal.operators.synthesize_cu import SynthesizeCUOperator
|
|
1882
|
+
# max_cus configurable via pipeline_params, default 8
|
|
1883
|
+
_max_cus = (ctx.pipeline_params or {}).get("max_cus", 8)
|
|
1884
|
+
SynthesizeCUOperator(max_cus=_max_cus).process([], graph_ctx)
|
|
1885
|
+
except Exception as exc:
|
|
1886
|
+
logger.debug("[SDLC_SCAN] CU synthesis skipped: %s", exc)
|
|
1887
|
+
|
|
1888
|
+
return {
|
|
1889
|
+
"nodes": len(graph_ctx.node_ids),
|
|
1890
|
+
"edges": graph_ctx.edge_count,
|
|
1891
|
+
"errors": graph_ctx.errors,
|
|
1892
|
+
}
|
|
1893
|
+
|
|
1894
|
+
# ------------------------------------------------------------------
|
|
1895
|
+
# SDLC sub-stages (resumable pipeline)
|
|
1896
|
+
# ------------------------------------------------------------------
|
|
1897
|
+
|
|
1898
|
+
def _get_sdlc_graph_ctx(self, ctx: PipelineContext):
|
|
1899
|
+
"""Get or create graph context for SDLC sub-stages."""
|
|
1900
|
+
from .universal.stage_executor import GraphContext as UniversalGraphContext
|
|
1901
|
+
graph_ns = ctx.graph_namespace or "default"
|
|
1902
|
+
db = None
|
|
1903
|
+
if self.graph_registry:
|
|
1904
|
+
db = self.graph_registry.get_graph(graph_ns)
|
|
1905
|
+
if db is None:
|
|
1906
|
+
db = self.graph_registry.create_graph(graph_ns)
|
|
1907
|
+
if db is None:
|
|
1908
|
+
return None, graph_ns
|
|
1909
|
+
return UniversalGraphContext(db=db, namespace=graph_ns), graph_ns
|
|
1910
|
+
|
|
1911
|
+
def _stage_sdlc_scan_files(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1912
|
+
"""Sub-stage 1: Scan repo files — README, source, tests, docs, API routes."""
|
|
1913
|
+
graph_ctx, graph_ns = self._get_sdlc_graph_ctx(ctx)
|
|
1914
|
+
if graph_ctx is None:
|
|
1915
|
+
return {"nodes": 0, "reason": "no_graph"}
|
|
1916
|
+
|
|
1917
|
+
repo_path = ctx.source_text or ""
|
|
1918
|
+
if not repo_path or not os.path.isdir(repo_path):
|
|
1919
|
+
return {"nodes": 0, "reason": "no_repo_path"}
|
|
1920
|
+
|
|
1921
|
+
from .universal.operators.scanners.repo import (
|
|
1922
|
+
scan_readme, scan_source_modules, scan_tests, scan_docs, scan_api_routes,
|
|
1923
|
+
)
|
|
1924
|
+
from .universal.operators.scanners.llm_enrichment import scan_with_llm
|
|
1925
|
+
from .universal.operators.scanners.jira import scan_jira_issues
|
|
1926
|
+
from pathlib import Path
|
|
1927
|
+
|
|
1928
|
+
rp = Path(repo_path)
|
|
1929
|
+
all_nodes = (
|
|
1930
|
+
scan_readme(rp) + scan_source_modules(rp) + scan_tests(rp)
|
|
1931
|
+
+ scan_docs(rp) + scan_api_routes(rp)
|
|
1932
|
+
)
|
|
1933
|
+
|
|
1934
|
+
# Optional LLM enrichment
|
|
1935
|
+
skip_llm = (ctx.pipeline_params or {}).get("skip_llm", True) or os.environ.get("SDLC_SKIP_LLM")
|
|
1936
|
+
if not skip_llm:
|
|
1937
|
+
code = [n for n in all_nodes if n["label"] == "CodeModule"]
|
|
1938
|
+
non_code = [n for n in all_nodes if n["label"] != "CodeModule"]
|
|
1939
|
+
all_nodes = non_code + scan_with_llm(rp, code)
|
|
1940
|
+
|
|
1941
|
+
# Jira (API-based, no clone needed)
|
|
1942
|
+
all_nodes += scan_jira_issues()
|
|
1943
|
+
|
|
1944
|
+
for n in all_nodes:
|
|
1945
|
+
graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
|
|
1946
|
+
|
|
1947
|
+
# Store node defs in ctx for edge inference later
|
|
1948
|
+
ctx.nodes = [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in all_nodes]
|
|
1949
|
+
return {"nodes": len(all_nodes)}
|
|
1950
|
+
|
|
1951
|
+
def _stage_sdlc_scan_github(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1952
|
+
"""Sub-stage 2: Fetch GitHub issues, PRs, contributors via API."""
|
|
1953
|
+
graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
|
|
1954
|
+
if graph_ctx is None:
|
|
1955
|
+
return {"nodes": 0, "reason": "no_graph"}
|
|
1956
|
+
|
|
1957
|
+
source_url = ctx.source_url or ctx.source_text or ""
|
|
1958
|
+
repo_path = ctx.source_text or "."
|
|
1959
|
+
|
|
1960
|
+
from .universal.operators.scanners.github import scan_github_issues_full
|
|
1961
|
+
from .universal.operators.scanners.github_api import quick_scan
|
|
1962
|
+
from pathlib import Path
|
|
1963
|
+
|
|
1964
|
+
token = getattr(ctx, "github_token", None) or None
|
|
1965
|
+
# Try full GitHub scan (issues + PRs + contributors + metadata)
|
|
1966
|
+
nodes = (
|
|
1967
|
+
quick_scan(source_url, github_token=token)
|
|
1968
|
+
if source_url.startswith("https://")
|
|
1969
|
+
else scan_github_issues_full(Path(repo_path), source_url, github_token=token)
|
|
1970
|
+
)
|
|
1971
|
+
|
|
1972
|
+
for n in nodes:
|
|
1973
|
+
graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
|
|
1974
|
+
|
|
1975
|
+
ctx.nodes = (ctx.nodes or []) + [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in nodes]
|
|
1976
|
+
return {"nodes": len(nodes)}
|
|
1977
|
+
|
|
1978
|
+
def _stage_sdlc_scan_git(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
1979
|
+
"""Sub-stage 3: Git history + repo metadata."""
|
|
1980
|
+
graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
|
|
1981
|
+
if graph_ctx is None:
|
|
1982
|
+
return {"nodes": 0, "reason": "no_graph"}
|
|
1983
|
+
|
|
1984
|
+
repo_path = ctx.source_text or ""
|
|
1985
|
+
if not repo_path or not os.path.isdir(repo_path):
|
|
1986
|
+
return {"nodes": 0, "reason": "no_repo_path"}
|
|
1987
|
+
|
|
1988
|
+
from .universal.operators.scanners.git import scan_git_history, scan_repo_metadata
|
|
1989
|
+
from pathlib import Path
|
|
1990
|
+
|
|
1991
|
+
rp = Path(repo_path)
|
|
1992
|
+
meta_nodes = scan_repo_metadata(rp)
|
|
1993
|
+
history_nodes, history_edges = scan_git_history(rp)
|
|
1994
|
+
all_nodes = meta_nodes + history_nodes
|
|
1995
|
+
|
|
1996
|
+
for n in all_nodes:
|
|
1997
|
+
graph_ctx.add_node(label=n["label"], properties=dict(n.get("properties", {})), node_id=n["id"])
|
|
1998
|
+
for e in history_edges:
|
|
1999
|
+
graph_ctx.add_edge(source_id=e["source"], target_id=e["target"], label=e["label"])
|
|
2000
|
+
|
|
2001
|
+
ctx.nodes = (ctx.nodes or []) + [{"id": n["id"], "label": n["label"], "properties": n.get("properties", {})} for n in all_nodes]
|
|
2002
|
+
return {"nodes": len(all_nodes), "edges": len(history_edges)}
|
|
2003
|
+
|
|
2004
|
+
def _stage_sdlc_scan_edges(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
2005
|
+
"""Sub-stage 4: Infer edges + BM25 index + embed."""
|
|
2006
|
+
graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
|
|
2007
|
+
if graph_ctx is None:
|
|
2008
|
+
return {"edges": 0, "reason": "no_graph"}
|
|
2009
|
+
|
|
2010
|
+
from .universal.operators.scanners.edges import infer_edges
|
|
2011
|
+
from .universal.operators.index_bm25 import IndexBM25Operator
|
|
2012
|
+
from .universal.operators.embed import EmbedOperator
|
|
2013
|
+
|
|
2014
|
+
# Infer edges from all nodes accumulated in ctx.nodes
|
|
2015
|
+
edges = infer_edges(ctx.nodes or [])
|
|
2016
|
+
for e in edges:
|
|
2017
|
+
graph_ctx.add_edge(source_id=e["source"], target_id=e["target"], label=e["label"])
|
|
2018
|
+
|
|
2019
|
+
# BM25 + embed
|
|
2020
|
+
# Rebuild node_ids from ctx.nodes since graph_ctx is fresh
|
|
2021
|
+
for n in (ctx.nodes or []):
|
|
2022
|
+
if n["id"] not in graph_ctx.node_ids:
|
|
2023
|
+
graph_ctx.node_ids.append(n["id"])
|
|
2024
|
+
|
|
2025
|
+
try:
|
|
2026
|
+
IndexBM25Operator().process([], graph_ctx)
|
|
2027
|
+
except Exception:
|
|
2028
|
+
pass
|
|
2029
|
+
try:
|
|
2030
|
+
EmbedOperator().process([], graph_ctx)
|
|
2031
|
+
except Exception:
|
|
2032
|
+
pass
|
|
2033
|
+
|
|
2034
|
+
return {"edges": len(edges), "indexed": len(graph_ctx.node_ids)}
|
|
2035
|
+
|
|
2036
|
+
def _stage_sdlc_scan_cu(self, ctx: PipelineContext) -> Dict[str, Any]:
|
|
2037
|
+
"""Sub-stage 5: Build Context Units."""
|
|
2038
|
+
graph_ctx, _ = self._get_sdlc_graph_ctx(ctx)
|
|
2039
|
+
if graph_ctx is None:
|
|
2040
|
+
return {"cus": 0, "reason": "no_graph"}
|
|
2041
|
+
|
|
2042
|
+
from .universal.operators.synthesize_cu import SynthesizeCUOperator
|
|
2043
|
+
|
|
2044
|
+
# Rebuild node_ids
|
|
2045
|
+
for n in (ctx.nodes or []):
|
|
2046
|
+
if n["id"] not in graph_ctx.node_ids:
|
|
2047
|
+
graph_ctx.node_ids.append(n["id"])
|
|
2048
|
+
|
|
2049
|
+
max_cus = (ctx.pipeline_params or {}).get("max_cus", 8)
|
|
2050
|
+
SynthesizeCUOperator(max_cus=max_cus).process([], graph_ctx)
|
|
2051
|
+
|
|
2052
|
+
cu_count = sum(1 for nid in graph_ctx.node_ids if nid.startswith("cu_"))
|
|
2053
|
+
return {"cus": cu_count}
|
|
2054
|
+
|
|
2055
|
+
|
|
2056
|
+
# ---------------------------------------------------------------------------
|
|
2057
|
+
# Convenience: create a context and execute a full pipeline
|
|
2058
|
+
# ---------------------------------------------------------------------------
|
|
2059
|
+
|
|
2060
|
+
def run_pipeline(
|
|
2061
|
+
graph_registry,
|
|
2062
|
+
graph_namespace: str,
|
|
2063
|
+
stages: List[str],
|
|
2064
|
+
source_text: Optional[str] = None,
|
|
2065
|
+
source_file_path: Optional[str] = None,
|
|
2066
|
+
source_filename: Optional[str] = None,
|
|
2067
|
+
intent: str = "graph_rag",
|
|
2068
|
+
mode: str = "run_all",
|
|
2069
|
+
pipeline_params: Optional[Dict[str, Any]] = None,
|
|
2070
|
+
on_stage: Optional[Callable] = None,
|
|
2071
|
+
) -> PipelineContext:
|
|
2072
|
+
"""
|
|
2073
|
+
Convenience function to create a PipelineContext and execute stages.
|
|
2074
|
+
|
|
2075
|
+
Returns the final PipelineContext with all results.
|
|
2076
|
+
"""
|
|
2077
|
+
ctx = PipelineContext(
|
|
2078
|
+
graph_namespace=graph_namespace,
|
|
2079
|
+
stages=stages,
|
|
2080
|
+
source_text=source_text,
|
|
2081
|
+
source_bytes_path=source_file_path,
|
|
2082
|
+
source_filename=source_filename,
|
|
2083
|
+
intent=intent,
|
|
2084
|
+
mode=mode,
|
|
2085
|
+
pipeline_params=pipeline_params or {},
|
|
2086
|
+
)
|
|
2087
|
+
|
|
2088
|
+
executor = StageExecutor(graph_registry=graph_registry)
|
|
2089
|
+
|
|
2090
|
+
if mode == "step_by_step":
|
|
2091
|
+
executor.execute_next(ctx)
|
|
2092
|
+
else:
|
|
2093
|
+
executor.execute_all(ctx, on_stage=on_stage)
|
|
2094
|
+
|
|
2095
|
+
return ctx
|