apsimo 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apsimo/__init__.py +38 -0
- apsimo/__main__.py +6 -0
- apsimo/agent/__init__.py +6 -0
- apsimo/agent/client.py +276 -0
- apsimo/agent/models.py +46 -0
- apsimo/agents/__init__.py +20 -0
- apsimo/agents/models.py +264 -0
- apsimo/agents/store.py +861 -0
- apsimo/agents/websocket.py +522 -0
- apsimo/api/__init__.py +1 -0
- apsimo/api/auth_telemetry.py +287 -0
- apsimo/api/authority.py +1203 -0
- apsimo/api/contact_grants.py +347 -0
- apsimo/api/middleware.py +483 -0
- apsimo/api/routers/__init__.py +1 -0
- apsimo/api/routers/commitment_work.py +265 -0
- apsimo/api/routers/context_gate.py +123 -0
- apsimo/api/routers/executions.py +140 -0
- apsimo/api/routers/followup_plans.py +147 -0
- apsimo/api/routers/governed_actions.py +162 -0
- apsimo/api/routers/host.py +14473 -0
- apsimo/api/routers/initiative_work.py +115 -0
- apsimo/api/routers/mining.py +104 -0
- apsimo/api/routers/observations.py +110 -0
- apsimo/api/routers/social_state.py +225 -0
- apsimo/api/routers/task_queue.py +2715 -0
- apsimo/api/routers/temporal_followups.py +251 -0
- apsimo/api/routers/transport.py +110 -0
- apsimo/api/routers/transport_ingress_api.py +240 -0
- apsimo/api/schemas/__init__.py +1 -0
- apsimo/api/schemas/host.py +1949 -0
- apsimo/autonomy/cli.py +110 -0
- apsimo/autonomy/condition_worker.py +437 -0
- apsimo/autonomy/config.py +424 -0
- apsimo/autonomy/loop.py +4316 -0
- apsimo/autonomy/registry.py +339 -0
- apsimo/autonomy/scheduler.py +1822 -0
- apsimo/autonomy/synthesis.py +449 -0
- apsimo/backup.py +962 -0
- apsimo/beliefs/__init__.py +23 -0
- apsimo/beliefs/contradictions.py +109 -0
- apsimo/beliefs/decay.py +61 -0
- apsimo/beliefs/engine.py +479 -0
- apsimo/beliefs/models.py +67 -0
- apsimo/beliefs/promotion.py +41 -0
- apsimo/beliefs/resolve.py +58 -0
- apsimo/beliefs/source_claims.py +690 -0
- apsimo/beliefs/source_projection.py +883 -0
- apsimo/beliefs/source_time.py +208 -0
- apsimo/beliefs/store.py +133 -0
- apsimo/briefings/aggregators.py +824 -0
- apsimo/briefings/composer.py +420 -0
- apsimo/briefings/config.py +55 -0
- apsimo/briefings/delivery.py +439 -0
- apsimo/briefings/engagement.py +97 -0
- apsimo/briefings/engine.py +274 -0
- apsimo/briefings/enhancer.py +99 -0
- apsimo/briefings/models.py +183 -0
- apsimo/briefings/scheduler.py +382 -0
- apsimo/briefings/store.py +435 -0
- apsimo/chain/__init__.py +48 -0
- apsimo/chain/block.py +100 -0
- apsimo/chain/cli.py +704 -0
- apsimo/chain/genesis.py +443 -0
- apsimo/chain/identity.py +416 -0
- apsimo/chain/keys.py +1025 -0
- apsimo/chain/local_keys.py +187 -0
- apsimo/chain/manager.py +290 -0
- apsimo/chain/node.py +163 -0
- apsimo/chain/plugin_transactions.py +371 -0
- apsimo/chain/protocol.py +220 -0
- apsimo/chain/state_machine.py +676 -0
- apsimo/chain/storage.py +503 -0
- apsimo/chain/transactions.py +250 -0
- apsimo/chain/validation.py +397 -0
- apsimo/channels/__init__.py +1 -0
- apsimo/channels/manifest.py +31 -0
- apsimo/channels/migrations/001_channels_schema.sql +12 -0
- apsimo/channels/phone_gateways.py +42 -0
- apsimo/channels/presence.py +188 -0
- apsimo/channels/router.py +235 -0
- apsimo/channels/store.py +231 -0
- apsimo/cli.py +2688 -0
- apsimo/cognition/__init__.py +11 -0
- apsimo/cognition/charter.py +398 -0
- apsimo/cognition/drive_governance.py +3530 -0
- apsimo/cognition/evidence_pipeline.py +1627 -0
- apsimo/cognition/external_events.py +932 -0
- apsimo/cognition/goal_spine.py +3488 -0
- apsimo/cognition/introspection.py +214 -0
- apsimo/cognition/prompt.py +150 -0
- apsimo/cognition/runtime.py +108 -0
- apsimo/cognition/trigger.py +154 -0
- apsimo/commitments/__init__.py +18 -0
- apsimo/commitments/local_work.py +355 -0
- apsimo/commitments/store.py +1052 -0
- apsimo/commitments/work.py +91 -0
- apsimo/compat.py +53 -0
- apsimo/compression/__init__.py +467 -0
- apsimo/connectors/__init__.py +21 -0
- apsimo/connectors/base.py +152 -0
- apsimo/connectors/caldav_calendar.py +125 -0
- apsimo/connectors/fs_documents.py +85 -0
- apsimo/connectors/imap_email.py +138 -0
- apsimo/connectors/manager.py +218 -0
- apsimo/connectors/webhook_pull.py +88 -0
- apsimo/contacts/__init__.py +33 -0
- apsimo/contacts/comms.py +357 -0
- apsimo/contacts/config.py +79 -0
- apsimo/contacts/exporters/__init__.py +1 -0
- apsimo/contacts/exporters/vcard.py +71 -0
- apsimo/contacts/identity_links.py +251 -0
- apsimo/contacts/importer.py +280 -0
- apsimo/contacts/importers/__init__.py +1 -0
- apsimo/contacts/importers/batch.py +43 -0
- apsimo/contacts/importers/macos_contacts.py +101 -0
- apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
- apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
- apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
- apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
- apsimo/contacts/migrations/005_identity_links.sql +27 -0
- apsimo/contacts/models.py +308 -0
- apsimo/contacts/scoring.py +16 -0
- apsimo/contacts/store.py +1623 -0
- apsimo/contacts/transport_ingress.py +252 -0
- apsimo/contacts/world_bridge.py +314 -0
- apsimo/contextgate/__init__.py +69 -0
- apsimo/contextgate/chunker.py +169 -0
- apsimo/contextgate/estimate.py +54 -0
- apsimo/contextgate/gate.py +313 -0
- apsimo/contextgate/retrieve.py +115 -0
- apsimo/delivery/__init__.py +16 -0
- apsimo/delivery/bridge.py +1260 -0
- apsimo/delivery/channels.py +526 -0
- apsimo/delivery/classification.py +50 -0
- apsimo/delivery/rate_limiter.py +268 -0
- apsimo/delivery/reachout_policy.py +206 -0
- apsimo/directed/__init__.py +22 -0
- apsimo/directed/audit.py +167 -0
- apsimo/directed/intake.py +95 -0
- apsimo/directed/models.py +191 -0
- apsimo/directed/service.py +509 -0
- apsimo/directives/__init__.py +25 -0
- apsimo/directives/evidence.py +87 -0
- apsimo/directives/extractor.py +188 -0
- apsimo/directives/guard.py +364 -0
- apsimo/directives/models.py +206 -0
- apsimo/directives/service.py +372 -0
- apsimo/directives/store.py +167 -0
- apsimo/doctor.py +2173 -0
- apsimo/environment.py +43 -0
- apsimo/events/__init__.py +33 -0
- apsimo/events/broadcaster.py +98 -0
- apsimo/events/bus.py +217 -0
- apsimo/events/journal.py +863 -0
- apsimo/events/stream.py +131 -0
- apsimo/events/types.py +150 -0
- apsimo/execution_results.py +357 -0
- apsimo/feedback/__init__.py +5 -0
- apsimo/feedback/store.py +76 -0
- apsimo/feeds/__init__.py +19 -0
- apsimo/feeds/cli.py +84 -0
- apsimo/feeds/engine.py +437 -0
- apsimo/feeds/example-feed.yaml +77 -0
- apsimo/feeds/hermes_cron.py +126 -0
- apsimo/feeds/manager.py +235 -0
- apsimo/feeds/spec.py +250 -0
- apsimo/feeds/template.py +202 -0
- apsimo/gate/__init__.py +18 -0
- apsimo/gate/audit.py +61 -0
- apsimo/gate/communication_policy.py +166 -0
- apsimo/gate/config.py +72 -0
- apsimo/gate/context_provenance.py +170 -0
- apsimo/gate/env_risk.py +226 -0
- apsimo/gate/guard_audit.py +353 -0
- apsimo/gate/layers/__init__.py +1 -0
- apsimo/gate/layers/base.py +15 -0
- apsimo/gate/layers/l1_recipient.py +66 -0
- apsimo/gate/layers/l2_pii.py +134 -0
- apsimo/gate/layers/l3_cross_context.py +50 -0
- apsimo/gate/layers/l4_trust_tier.py +78 -0
- apsimo/gate/layers/l5_injection.py +199 -0
- apsimo/gate/layers/l6_review.py +86 -0
- apsimo/gate/layers/l7_delay.py +100 -0
- apsimo/gate/layers/tom2_epistemic.py +185 -0
- apsimo/gate/models.py +64 -0
- apsimo/gate/pending_dispatch.py +5 -0
- apsimo/gate/pipeline.py +206 -0
- apsimo/gate/rejection.py +259 -0
- apsimo/gate/response_guard.py +700 -0
- apsimo/gate/rulesets/injection_v1.yaml +51 -0
- apsimo/gate/surface_policy.py +189 -0
- apsimo/gate/taint.py +226 -0
- apsimo/genesis.json +9 -0
- apsimo/goals/__init__.py +100 -0
- apsimo/goals/config.py +38 -0
- apsimo/goals/decomposer.py +421 -0
- apsimo/goals/engine.py +617 -0
- apsimo/goals/inference.py +354 -0
- apsimo/goals/models.py +302 -0
- apsimo/goals/priority.py +270 -0
- apsimo/goals/queue_bridge.py +149 -0
- apsimo/goals/replan.py +450 -0
- apsimo/goals/schema.sql +89 -0
- apsimo/goals/store.py +692 -0
- apsimo/governed_actions.py +1708 -0
- apsimo/harness_integration/__init__.py +45 -0
- apsimo/harness_integration/context.py +41 -0
- apsimo/harness_integration/skills.py +231 -0
- apsimo/identity/__init__.py +26 -0
- apsimo/identity/participants.py +181 -0
- apsimo/identity/resolver.py +329 -0
- apsimo/identity_bootstrap/__init__.py +5 -0
- apsimo/identity_bootstrap/builder.py +208 -0
- apsimo/identity_bootstrap/corpus.py +443 -0
- apsimo/identity_bootstrap/models.py +54 -0
- apsimo/identity_bootstrap/runner.py +353 -0
- apsimo/identity_bootstrap/seeders/__init__.py +25 -0
- apsimo/identity_bootstrap/seeders/briefings.py +109 -0
- apsimo/identity_bootstrap/seeders/chain.py +57 -0
- apsimo/identity_bootstrap/seeders/goals.py +128 -0
- apsimo/identity_bootstrap/seeders/memory.py +191 -0
- apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
- apsimo/identity_bootstrap/seeders/relationship.py +152 -0
- apsimo/identity_bootstrap/seeders/sessions.py +67 -0
- apsimo/identity_bootstrap/seeders/skills.py +92 -0
- apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
- apsimo/identity_bootstrap/seeders/world_model.py +143 -0
- apsimo/identity_bootstrap/self_query.py +92 -0
- apsimo/identity_bootstrap/self_reflection.py +155 -0
- apsimo/identity_bootstrap/skill.py +37 -0
- apsimo/identity_bootstrap/verifier.py +436 -0
- apsimo/initiatives/__init__.py +20 -0
- apsimo/initiatives/action_registry.py +454 -0
- apsimo/initiatives/approval_authority.py +2105 -0
- apsimo/initiatives/approval_policy.py +123 -0
- apsimo/initiatives/assignment.py +263 -0
- apsimo/initiatives/backup_evidence.py +100 -0
- apsimo/initiatives/context_freshness.py +103 -0
- apsimo/initiatives/models.py +318 -0
- apsimo/initiatives/native_work.py +270 -0
- apsimo/initiatives/standing_approvals.py +232 -0
- apsimo/initiatives/store.py +1081 -0
- apsimo/initiatives/temporal_followup.py +410 -0
- apsimo/intelligence/__init__.py +1 -0
- apsimo/intelligence/cognition/__init__.py +24 -0
- apsimo/intelligence/cognition/gap_detector.py +148 -0
- apsimo/intelligence/cognition/metalearner.py +547 -0
- apsimo/intelligence/cognition/metrics_collector.py +217 -0
- apsimo/intelligence/cognition/performance_index.py +299 -0
- apsimo/intelligence/cognition/registry.py +192 -0
- apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
- apsimo/intelligence/cognition/types.py +16 -0
- apsimo/intelligence/components/__init__.py +66 -0
- apsimo/intelligence/components/anomaly_detector.py +413 -0
- apsimo/intelligence/components/initiative_engine.py +2643 -0
- apsimo/intelligence/components/preference_learner.py +521 -0
- apsimo/intelligence/components/research_orchestrator.py +358 -0
- apsimo/intelligence/components/self_directed_thinker.py +221 -0
- apsimo/intelligence/components/self_reflector.py +252 -0
- apsimo/intelligence/components/session_continuity.py +154 -0
- apsimo/intelligence/components/task_planner.py +320 -0
- apsimo/intelligence/components/tool_learner.py +217 -0
- apsimo/intelligence/graph/__init__.py +79 -0
- apsimo/intelligence/graph/client.py +2483 -0
- apsimo/intelligence/graph/consolidator.py +405 -0
- apsimo/intelligence/graph/distiller.py +312 -0
- apsimo/intelligence/graph/migrations.py +129 -0
- apsimo/intelligence/graph/queries.py +248 -0
- apsimo/intelligence/graph/recall.py +281 -0
- apsimo/intelligence/graph/reconciler.py +144 -0
- apsimo/intelligence/graph/schema.py +337 -0
- apsimo/intelligence/graph/selection.py +252 -0
- apsimo/intelligence/learning/__init__.py +17 -0
- apsimo/intelligence/learning/continuous_learner.py +245 -0
- apsimo/intelligence/learning/feedback_store.py +321 -0
- apsimo/intelligence/mind_model/__init__.py +1 -0
- apsimo/intelligence/mind_model/graph_baseline.py +136 -0
- apsimo/intelligence/mind_model/signal_collector.py +361 -0
- apsimo/intelligence/relationships/__init__.py +11 -0
- apsimo/intelligence/relationships/profiler.py +389 -0
- apsimo/intelligence/relationships/scorer.py +560 -0
- apsimo/intelligence/relationships/signal_floor.py +66 -0
- apsimo/intelligence/relationships/trust_tiers.py +300 -0
- apsimo/intelligence/synthesis/__init__.py +40 -0
- apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
- apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
- apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
- apsimo/intelligence/synthesis/insight_store.py +79 -0
- apsimo/intelligence/synthesis/insight_validator.py +183 -0
- apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
- apsimo/intelligence/turn_middleware/__init__.py +15 -0
- apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
- apsimo/mcp/__init__.py +41 -0
- apsimo/mcp/__main__.py +6 -0
- apsimo/mcp/config.py +287 -0
- apsimo/mcp/server.py +501 -0
- apsimo/migrations.py +187 -0
- apsimo/mining/__init__.py +27 -0
- apsimo/mining/corpus.py +239 -0
- apsimo/mining/escalations.py +289 -0
- apsimo/mining/models.py +169 -0
- apsimo/mining/store.py +210 -0
- apsimo/models/__init__.py +30 -0
- apsimo/models/memory.py +80 -0
- apsimo/models/mesh.py +72 -0
- apsimo/models/person.py +104 -0
- apsimo/models/signal.py +108 -0
- apsimo/observations/__init__.py +15 -0
- apsimo/observations/store.py +277 -0
- apsimo/patterns/__init__.py +6 -0
- apsimo/patterns/extract.py +187 -0
- apsimo/patterns/store.py +227 -0
- apsimo/persona/__init__.py +1 -0
- apsimo/persona/engine.py +611 -0
- apsimo/persona/manifest.py +140 -0
- apsimo/projects/__init__.py +28 -0
- apsimo/projects/engine.py +1681 -0
- apsimo/projects/event_outbox.py +188 -0
- apsimo/projects/models.py +216 -0
- apsimo/projects/planner.py +181 -0
- apsimo/projects/store.py +1446 -0
- apsimo/proposals/__init__.py +12 -0
- apsimo/proposals/engine.py +114 -0
- apsimo/proposals/models.py +207 -0
- apsimo/qualification/__init__.py +1 -0
- apsimo/qualification/cases.py +75 -0
- apsimo/qualification/cli.py +51 -0
- apsimo/qualification/memory_cases.py +209 -0
- apsimo/qualification/records.py +92 -0
- apsimo/qualification/report.py +87 -0
- apsimo/qualification/runner.py +311 -0
- apsimo/qualification/structured_cases.py +131 -0
- apsimo/reasoning/__init__.py +13 -0
- apsimo/reasoning/executor.py +506 -0
- apsimo/reasoning/loop.py +373 -0
- apsimo/reasoning/native_tools/__init__.py +16 -0
- apsimo/reasoning/native_tools/calculate.py +141 -0
- apsimo/reasoning/native_tools/file_ops.py +150 -0
- apsimo/reasoning/native_tools/web_search.py +49 -0
- apsimo/reasoning/tool_policy.py +182 -0
- apsimo/redact/__init__.py +176 -0
- apsimo/repos/__init__.py +5 -0
- apsimo/repos/mirrors.py +204 -0
- apsimo/research/__init__.py +41 -0
- apsimo/research/artifact.py +482 -0
- apsimo/research/gatherer.py +387 -0
- apsimo/research/pipeline.py +513 -0
- apsimo/research/search/__init__.py +7 -0
- apsimo/research/search/base.py +41 -0
- apsimo/research/search/brave.py +59 -0
- apsimo/research/search/cache.py +51 -0
- apsimo/research/search/duckduckgo.py +103 -0
- apsimo/research/search/orchestrator.py +119 -0
- apsimo/research/search/serpapi.py +59 -0
- apsimo/research/search/tavily.py +59 -0
- apsimo/research/synthesizer.py +309 -0
- apsimo/router/__init__.py +30 -0
- apsimo/router/complexity_scorer.py +148 -0
- apsimo/router/endpoints.py +153 -0
- apsimo/router/fallback.py +58 -0
- apsimo/router/functions.py +243 -0
- apsimo/router/native_policy.py +52 -0
- apsimo/router/router.py +762 -0
- apsimo/router/self_learning.py +174 -0
- apsimo/router/tiers.py +677 -0
- apsimo/sandbox/__init__.py +21 -0
- apsimo/sandbox/backend.py +195 -0
- apsimo/sandbox/manager.py +173 -0
- apsimo/scope_bounds.py +7 -0
- apsimo/secrets/__init__.py +6 -0
- apsimo/secrets/backends/__init__.py +8 -0
- apsimo/secrets/backends/base.py +42 -0
- apsimo/secrets/backends/env.py +110 -0
- apsimo/secrets/backends/keyring.py +72 -0
- apsimo/secrets/backends/onepassword.py +232 -0
- apsimo/secrets/cli.py +191 -0
- apsimo/secrets/manager.py +160 -0
- apsimo/secrets/migration.py +101 -0
- apsimo/secrets/types.py +98 -0
- apsimo/seed.py +41 -0
- apsimo/self_model/__init__.py +37 -0
- apsimo/self_model/appraisals.py +673 -0
- apsimo/self_model/benchmark.py +1314 -0
- apsimo/self_model/brief.py +40 -0
- apsimo/self_model/event_concerns.py +1128 -0
- apsimo/self_model/execution_forecasts.py +353 -0
- apsimo/self_model/expectations.py +1595 -0
- apsimo/self_model/experiments.py +1150 -0
- apsimo/self_model/journal.py +148 -0
- apsimo/self_model/judgments.py +705 -0
- apsimo/self_model/native_outcomes.py +55 -0
- apsimo/self_model/params.py +220 -0
- apsimo/self_model/perspective.py +246 -0
- apsimo/self_model/reconcile.py +183 -0
- apsimo/self_model/reply_forecasts.py +381 -0
- apsimo/self_model/runtime_forecasts.py +296 -0
- apsimo/self_model/runtime_models.py +67 -0
- apsimo/self_model/settlement.py +207 -0
- apsimo/self_model/situation.py +1731 -0
- apsimo/self_model/store.py +883 -0
- apsimo/self_model/supervised.py +137 -0
- apsimo/self_model/thinker.py +99 -0
- apsimo/self_model/trust.py +388 -0
- apsimo/self_model/workspace.py +2388 -0
- apsimo/server.py +4197 -0
- apsimo/services/__init__.py +1 -0
- apsimo/services/agent_bridge.py +474 -0
- apsimo/services/initiative_executor.py +914 -0
- apsimo/services/instance.py +297 -0
- apsimo/sessions/__init__.py +22 -0
- apsimo/sessions/config.py +13 -0
- apsimo/sessions/context_loader.py +88 -0
- apsimo/sessions/federation_session.py +75 -0
- apsimo/sessions/isolated_session.py +98 -0
- apsimo/sessions/reports.py +84 -0
- apsimo/sessions/store.py +148 -0
- apsimo/setup.py +2818 -0
- apsimo/setup_hermes.py +879 -0
- apsimo/setup_local_work.py +218 -0
- apsimo/setup_native_goals.py +134 -0
- apsimo/setup_native_reviews.py +115 -0
- apsimo/skills/__init__.py +10 -0
- apsimo/skills/base.py +108 -0
- apsimo/skills/budget.py +28 -0
- apsimo/skills/executor.py +493 -0
- apsimo/skills/executors/__init__.py +1 -0
- apsimo/skills/executors/behavioral_correction.py +75 -0
- apsimo/skills/executors/capability_gap.py +38 -0
- apsimo/skills/executors/data_quality.py +163 -0
- apsimo/skills/executors/knowledge_acquisition.py +41 -0
- apsimo/skills/executors/operational_hygiene.py +185 -0
- apsimo/skills/executors/subsystem_health.py +169 -0
- apsimo/skills/hermes_export.py +431 -0
- apsimo/skills/index.py +123 -0
- apsimo/skills/learning/__init__.py +21 -0
- apsimo/skills/learning/novelty_detector.py +206 -0
- apsimo/skills/learning/pattern_extractor.py +199 -0
- apsimo/skills/learning/triggers.py +159 -0
- apsimo/skills/loader.py +246 -0
- apsimo/skills/migrations/002_progressive_loading.sql +6 -0
- apsimo/skills/migrations/backfill_triggers.py +20 -0
- apsimo/skills/models.py +202 -0
- apsimo/skills/packager.py +128 -0
- apsimo/skills/protocols.py +70 -0
- apsimo/skills/registry.py +191 -0
- apsimo/skills/runtime.py +58 -0
- apsimo/skills/sandbox_runner.py +229 -0
- apsimo/skills/scheduler.py +129 -0
- apsimo/skills/schema.py +79 -0
- apsimo/skills/security/__init__.py +12 -0
- apsimo/skills/security/guards.py +53 -0
- apsimo/skills/security/scanner.py +223 -0
- apsimo/skills_memory/__init__.py +26 -0
- apsimo/skills_memory/distill.py +159 -0
- apsimo/skills_memory/models.py +85 -0
- apsimo/skills_memory/retrieve.py +62 -0
- apsimo/skills_memory/store.py +172 -0
- apsimo/surprise/__init__.py +6 -0
- apsimo/surprise/accumulation.py +57 -0
- apsimo/surprise/scorer.py +102 -0
- apsimo/surprise/store.py +203 -0
- apsimo/task_queue/__init__.py +69 -0
- apsimo/task_queue/action_receipts.py +148 -0
- apsimo/task_queue/approval_relay_canary.py +108 -0
- apsimo/task_queue/config.py +85 -0
- apsimo/task_queue/contract.py +361 -0
- apsimo/task_queue/events.py +130 -0
- apsimo/task_queue/governor.py +1031 -0
- apsimo/task_queue/handlers/__init__.py +16 -0
- apsimo/task_queue/handlers/base.py +37 -0
- apsimo/task_queue/handlers/inference.py +640 -0
- apsimo/task_queue/handlers/monitoring.py +116 -0
- apsimo/task_queue/handlers/registry.py +75 -0
- apsimo/task_queue/handlers/subtask_handler.py +173 -0
- apsimo/task_queue/handlers/system_maintenance.py +147 -0
- apsimo/task_queue/mesh_integration.py +111 -0
- apsimo/task_queue/models.py +317 -0
- apsimo/task_queue/queue_manager.py +8286 -0
- apsimo/task_queue/routing.py +287 -0
- apsimo/task_queue/scheduler.py +252 -0
- apsimo/task_queue/schema.sql +197 -0
- apsimo/task_queue/work_control.py +342 -0
- apsimo/task_queue/worker.py +993 -0
- apsimo/telemetry.py +145 -0
- apsimo/tom/__init__.py +6 -0
- apsimo/tom/affect.py +387 -0
- apsimo/tom/approvals.py +171 -0
- apsimo/tom/arcs.py +896 -0
- apsimo/tom/asymmetry.py +131 -0
- apsimo/tom/eligibility.py +248 -0
- apsimo/tom/engagement.py +214 -0
- apsimo/tom/exposure.py +214 -0
- apsimo/tom/extractor.py +306 -0
- apsimo/tom/fact_adapters.py +144 -0
- apsimo/tom/facts.py +326 -0
- apsimo/tom/integration.py +592 -0
- apsimo/tom/leveled.py +118 -0
- apsimo/tom/levels.py +247 -0
- apsimo/tom/recipient_audit.py +995 -0
- apsimo/tom/recipient_simulator.py +593 -0
- apsimo/tom/source_lineage.py +93 -0
- apsimo/tom/tom2.py +277 -0
- apsimo/tom/visibility.py +559 -0
- apsimo/tom/visibility_store.py +414 -0
- apsimo/tools/__init__.py +0 -0
- apsimo/tools/definitions.py +740 -0
- apsimo/tools/handlers.py +943 -0
- apsimo/toolsmith/__init__.py +26 -0
- apsimo/toolsmith/authority.py +166 -0
- apsimo/toolsmith/engine.py +559 -0
- apsimo/toolsmith/integrity.py +100 -0
- apsimo/toolsmith/miner.py +145 -0
- apsimo/toolsmith/policy.py +110 -0
- apsimo/toolsmith/registry.py +635 -0
- apsimo/turns/__init__.py +17 -0
- apsimo/turns/audio.py +134 -0
- apsimo/turns/documents.py +235 -0
- apsimo/turns/executions.py +486 -0
- apsimo/turns/hermes_history.py +245 -0
- apsimo/turns/hermes_kanban.py +268 -0
- apsimo/turns/hermes_work.py +96 -0
- apsimo/turns/idempotency.py +752 -0
- apsimo/turns/local_work.py +115 -0
- apsimo/turns/media.py +581 -0
- apsimo/turns/reported_workers.py +196 -0
- apsimo/turns/source_annotations.py +283 -0
- apsimo/turns/source_attribution.py +154 -0
- apsimo/turns/source_read.py +351 -0
- apsimo/turns/source_vectors.py +263 -0
- apsimo/turns/video.py +210 -0
- apsimo/util/autonomy_preset.py +220 -0
- apsimo/util/instance.py +92 -0
- apsimo/util/model_output.py +25 -0
- apsimo/util/quiet_hours.py +27 -0
- apsimo/util/session_safety.py +37 -0
- apsimo/util/temporal.py +343 -0
- apsimo/vector/__init__.py +75 -0
- apsimo/vector/backfill.py +171 -0
- apsimo/vector/caption.py +114 -0
- apsimo/vector/collections.py +51 -0
- apsimo/vector/config.py +102 -0
- apsimo/vector/embedder.py +670 -0
- apsimo/vector/image_preprocess.py +406 -0
- apsimo/vector/image_store.py +296 -0
- apsimo/vector/indexes.py +162 -0
- apsimo/vector/migrate.py +334 -0
- apsimo/vector/multimodal_provider.py +417 -0
- apsimo/vector/multimodal_types.py +87 -0
- apsimo/vector/openai_provider.py +119 -0
- apsimo/vector/query.py +49 -0
- apsimo/vector/reranker.py +565 -0
- apsimo/vector/safety_image.py +159 -0
- apsimo/vector/scanner.py +197 -0
- apsimo/vector/setup.py +289 -0
- apsimo/vector/store.py +533 -0
- apsimo/vector/tiers.py +263 -0
- apsimo/work_orders.py +925 -0
- apsimo/workers/__init__.py +21 -0
- apsimo/workers/agent_bridge.py +640 -0
- apsimo/workers/colony_worker.py +382 -0
- apsimo/workers/queue_worker.py +441 -0
- apsimo/workers/skills_sync.py +152 -0
- apsimo/world_model/__init__.py +71 -0
- apsimo/world_model/causal_maintenance.py +131 -0
- apsimo/world_model/causal_policy.py +43 -0
- apsimo/world_model/causal_query.py +125 -0
- apsimo/world_model/confidence.py +54 -0
- apsimo/world_model/config.py +64 -0
- apsimo/world_model/constants.py +97 -0
- apsimo/world_model/entities.py +145 -0
- apsimo/world_model/expectation_resolvers.py +177 -0
- apsimo/world_model/extraction/__init__.py +7 -0
- apsimo/world_model/extraction/base.py +62 -0
- apsimo/world_model/extraction/conversation_extractor.py +262 -0
- apsimo/world_model/extraction/detector.py +74 -0
- apsimo/world_model/extraction/document_extractor.py +78 -0
- apsimo/world_model/extraction/formats/__init__.py +24 -0
- apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
- apsimo/world_model/extraction/formats/html_fmt.py +72 -0
- apsimo/world_model/extraction/formats/json_fmt.py +68 -0
- apsimo/world_model/extraction/formats/pdf.py +43 -0
- apsimo/world_model/extraction/formats/text.py +27 -0
- apsimo/world_model/extraction/llm_extractor.py +164 -0
- apsimo/world_model/extraction/pipeline.py +73 -0
- apsimo/world_model/integrations/__init__.py +5 -0
- apsimo/world_model/integrations/mind_model_bridge.py +115 -0
- apsimo/world_model/integrations/social_intel_bridge.py +120 -0
- apsimo/world_model/jobs/__init__.py +4 -0
- apsimo/world_model/jobs/extraction_job.py +168 -0
- apsimo/world_model/llm_extract.py +572 -0
- apsimo/world_model/neo4j/__init__.py +5 -0
- apsimo/world_model/neo4j/backend.py +654 -0
- apsimo/world_model/observations.py +155 -0
- apsimo/world_model/populator.py +307 -0
- apsimo/world_model/postgres/__init__.py +1 -0
- apsimo/world_model/postgres/backend.py +683 -0
- apsimo/world_model/relationships.py +25 -0
- apsimo/world_model/resolution/__init__.py +13 -0
- apsimo/world_model/resolution/entity_resolver.py +232 -0
- apsimo/world_model/resolution/merge_audit.py +16 -0
- apsimo/world_model/resolution/merge_workflow.py +117 -0
- apsimo/world_model/source_reports.py +121 -0
- apsimo/world_model/sqlite/__init__.py +4 -0
- apsimo/world_model/sqlite/backend.py +855 -0
- apsimo/world_model/sqlite/schema.sql +132 -0
- apsimo/world_model/store.py +545 -0
- apsimo-1.3.0.dist-info/METADATA +78 -0
- apsimo-1.3.0.dist-info/RECORD +614 -0
- apsimo-1.3.0.dist-info/WHEEL +5 -0
- apsimo-1.3.0.dist-info/entry_points.txt +11 -0
- apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
- apsimo-1.3.0.dist-info/top_level.txt +2 -0
- colony_sidecar/__init__.py +4 -0
|
@@ -0,0 +1,2483 @@
|
|
|
1
|
+
"""Colony Graph Memory System — Neo4j async client.
|
|
2
|
+
|
|
3
|
+
Replaces Hermes MEMORY.md with a persistent graph database that models
|
|
4
|
+
relationships, events, and behavioral patterns.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import ast
|
|
10
|
+
import asyncio
|
|
11
|
+
import hashlib
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import math
|
|
15
|
+
import os
|
|
16
|
+
import time
|
|
17
|
+
from collections import deque
|
|
18
|
+
from dataclasses import dataclass, field
|
|
19
|
+
from enum import Enum
|
|
20
|
+
from typing import Any, Callable, Coroutine, Dict, List, Optional, TYPE_CHECKING
|
|
21
|
+
|
|
22
|
+
try:
|
|
23
|
+
from neo4j import AsyncGraphDatabase, AsyncDriver
|
|
24
|
+
except ImportError:
|
|
25
|
+
pass
|
|
26
|
+
from pydantic import SecretStr
|
|
27
|
+
|
|
28
|
+
if TYPE_CHECKING:
|
|
29
|
+
from apsimo.vector.store import VectorStore
|
|
30
|
+
|
|
31
|
+
logger = logging.getLogger(__name__)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _recency_factor(days_old: float) -> float:
|
|
35
|
+
"""Recency weight for retrieval ranking (v0.21.0).
|
|
36
|
+
|
|
37
|
+
Configurable exponential half-life with a floor, so recent memories surface
|
|
38
|
+
over stale ones without fully suppressing older context. Defaults:
|
|
39
|
+
half-life 90d, floor 0.5 (a year-old memory keeps ~0.53 weight; fresh = 1.0).
|
|
40
|
+
Set COLONY_RECENCY_HALF_LIFE_DAYS<=0 to disable. The previous behaviour was a
|
|
41
|
+
near-flat ~10%/year discount that barely affected ranking.
|
|
42
|
+
"""
|
|
43
|
+
import os
|
|
44
|
+
try:
|
|
45
|
+
half_life = float(os.environ.get("COLONY_RECENCY_HALF_LIFE_DAYS", "90"))
|
|
46
|
+
except (ValueError, TypeError):
|
|
47
|
+
half_life = 90.0
|
|
48
|
+
if half_life <= 0:
|
|
49
|
+
return 1.0
|
|
50
|
+
try:
|
|
51
|
+
floor = float(os.environ.get("COLONY_RECENCY_FLOOR", "0.5"))
|
|
52
|
+
except (ValueError, TypeError):
|
|
53
|
+
floor = 0.5
|
|
54
|
+
floor = min(max(floor, 0.0), 1.0)
|
|
55
|
+
return floor + (1.0 - floor) * (0.5 ** (max(days_old, 0.0) / half_life))
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def distill_turn_summary(summary: str) -> str:
|
|
59
|
+
"""The distilled form of a turn summary (what COLONY_DISTILL_TURNS=1 stores).
|
|
60
|
+
|
|
61
|
+
PRESERVES BOTH speaker labels ("User:" and "Agent:", with "assistant"
|
|
62
|
+
normalized to "Agent:"). Attribution matters in both directions:
|
|
63
|
+
"User: my X is Y" says whose preference it is, and "Agent: ..." marks
|
|
64
|
+
the agent's own prose so downstream consumers (e.g. the belief claim
|
|
65
|
+
extractor) never mistake something the agent SAID for a fact a contact
|
|
66
|
+
ASSERTED. Lines are joined with "; " — this text is injected into
|
|
67
|
+
prompts, so it must never introduce an em dash.
|
|
68
|
+
"""
|
|
69
|
+
_lines = []
|
|
70
|
+
for ln in (summary or "").splitlines():
|
|
71
|
+
if ":" in ln:
|
|
72
|
+
_speaker, _rest = ln.split(":", 1)
|
|
73
|
+
_rest = _rest.strip()
|
|
74
|
+
sp = _speaker.strip().lower()
|
|
75
|
+
if sp == "user" and _rest:
|
|
76
|
+
_lines.append(f"User: {_rest}")
|
|
77
|
+
elif sp in ("agent", "assistant") and _rest:
|
|
78
|
+
_lines.append(f"Agent: {_rest}")
|
|
79
|
+
else:
|
|
80
|
+
_lines.append(_rest)
|
|
81
|
+
else:
|
|
82
|
+
_lines.append(ln)
|
|
83
|
+
return "; ".join(x for x in _lines if x) or (summary or "")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass
|
|
87
|
+
class GraphConfig:
|
|
88
|
+
"""Connection settings for the Neo4j graph database."""
|
|
89
|
+
|
|
90
|
+
uri: str = "bolt://localhost:7687"
|
|
91
|
+
database: str = "colony"
|
|
92
|
+
auth: Optional[tuple[str, SecretStr]] = None # (user, password) — password masked in logs
|
|
93
|
+
max_pool_size: int = 50
|
|
94
|
+
connection_timeout_secs: float = 10.0
|
|
95
|
+
max_retry_secs: float = 30.0
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
class MemorySourceType(str, Enum):
|
|
99
|
+
CONVERSATION = "conversation"
|
|
100
|
+
FILE = "file"
|
|
101
|
+
TOOL_OUTPUT = "tool_output"
|
|
102
|
+
USER_ASSERTION = "user_assertion"
|
|
103
|
+
INFERENCE = "inference"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
class EpistemicState(str, Enum):
|
|
107
|
+
INFERRED = "inferred"
|
|
108
|
+
OBSERVED = "observed"
|
|
109
|
+
CORROBORATED = "corroborated"
|
|
110
|
+
VERIFIED = "verified"
|
|
111
|
+
STALE = "stale"
|
|
112
|
+
SUPERSEDED = "superseded"
|
|
113
|
+
DEPRECATED = "deprecated"
|
|
114
|
+
ARCHIVED = "archived"
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
SOURCE_RELIABILITY: Dict[str, float] = {
|
|
118
|
+
MemorySourceType.USER_ASSERTION: 1.0,
|
|
119
|
+
MemorySourceType.FILE: 0.9,
|
|
120
|
+
MemorySourceType.TOOL_OUTPUT: 0.85,
|
|
121
|
+
MemorySourceType.CONVERSATION: 0.7,
|
|
122
|
+
MemorySourceType.INFERENCE: 0.5,
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
MAX_IMPORTANCE: Dict[str, float] = {
|
|
126
|
+
MemorySourceType.USER_ASSERTION: 1.0,
|
|
127
|
+
MemorySourceType.FILE: 0.95,
|
|
128
|
+
MemorySourceType.TOOL_OUTPUT: 0.9,
|
|
129
|
+
MemorySourceType.CONVERSATION: 0.8,
|
|
130
|
+
MemorySourceType.INFERENCE: 0.7,
|
|
131
|
+
}
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
class ColonyGraph:
|
|
135
|
+
"""Neo4j graph memory system replacing Hermes MEMORY.md.
|
|
136
|
+
|
|
137
|
+
Provides:
|
|
138
|
+
- Memory storage with automatic entity linking
|
|
139
|
+
- Semantic recall via vector index + strength decay
|
|
140
|
+
- Ebbinghaus forgetting‐curve decay
|
|
141
|
+
- Pruning of weak / stale memories
|
|
142
|
+
- Multi-hop traversal across memory connections
|
|
143
|
+
"""
|
|
144
|
+
|
|
145
|
+
def __init__(self, config: GraphConfig) -> None:
|
|
146
|
+
self._config = config
|
|
147
|
+
driver_auth = (
|
|
148
|
+
(config.auth[0], config.auth[1].get_secret_value())
|
|
149
|
+
if config.auth is not None
|
|
150
|
+
else None
|
|
151
|
+
)
|
|
152
|
+
self.driver: AsyncDriver = AsyncGraphDatabase.driver(
|
|
153
|
+
config.uri,
|
|
154
|
+
auth=driver_auth,
|
|
155
|
+
max_connection_pool_size=config.max_pool_size,
|
|
156
|
+
connection_timeout=config.connection_timeout_secs,
|
|
157
|
+
max_transaction_retry_time=config.max_retry_secs,
|
|
158
|
+
keep_alive=True,
|
|
159
|
+
# Suppress benign DBMS notifications (unknown-label/property warnings
|
|
160
|
+
# for nodes not yet created) — pure log noise, not errors.
|
|
161
|
+
notifications_min_severity="OFF",
|
|
162
|
+
)
|
|
163
|
+
self.database: str = config.database
|
|
164
|
+
self._embed_fn: Optional[
|
|
165
|
+
Callable[[str], Coroutine[Any, Any, List[float]]]
|
|
166
|
+
] = None
|
|
167
|
+
self._vector_store: Optional["VectorStore"] = None
|
|
168
|
+
self._rerank_fn: Optional[Callable[..., Coroutine[Any, Any, Any]]] = None
|
|
169
|
+
# Runtime-wide hard exclusions protect untyped recall consumers (for
|
|
170
|
+
# example model tools and background synthesis) that cannot safely
|
|
171
|
+
# render a governed source directly. Empty preserves legacy behavior.
|
|
172
|
+
self._recall_source_exclusions: tuple[str, ...] = ()
|
|
173
|
+
self._recall_metadata_exclusions: tuple[str, ...] = ()
|
|
174
|
+
|
|
175
|
+
@staticmethod
|
|
176
|
+
def _bounded_source_uris(source_uris) -> tuple[str, ...]:
|
|
177
|
+
"""Normalize a small exact-match source-URI exclusion set."""
|
|
178
|
+
return tuple(sorted(dict.fromkeys(
|
|
179
|
+
str(value).strip() for value in (source_uris or ())
|
|
180
|
+
if 0 < len(str(value).strip()) <= 256
|
|
181
|
+
)))[:16]
|
|
182
|
+
|
|
183
|
+
def set_recall_source_exclusions(
|
|
184
|
+
self,
|
|
185
|
+
source_uris,
|
|
186
|
+
*,
|
|
187
|
+
legacy_metadata_markers=(),
|
|
188
|
+
) -> None:
|
|
189
|
+
"""Set graph-wide hard recall exclusions for runtime policy.
|
|
190
|
+
|
|
191
|
+
Per-call exclusions are additive and can never relax this boundary.
|
|
192
|
+
``legacy_metadata_markers`` covers rows written before a governed
|
|
193
|
+
source URI existed. Both sets are bounded exact strings; passing empty
|
|
194
|
+
iterables restores the historical default behavior.
|
|
195
|
+
"""
|
|
196
|
+
self._recall_source_exclusions = self._bounded_source_uris(source_uris)
|
|
197
|
+
self._recall_metadata_exclusions = self._bounded_source_uris(
|
|
198
|
+
legacy_metadata_markers)
|
|
199
|
+
|
|
200
|
+
# ------------------------------------------------------------------
|
|
201
|
+
# Lifecycle
|
|
202
|
+
# ------------------------------------------------------------------
|
|
203
|
+
|
|
204
|
+
async def connect(self) -> None:
|
|
205
|
+
"""Verify connectivity to Neo4j."""
|
|
206
|
+
await self.driver.verify_connectivity()
|
|
207
|
+
|
|
208
|
+
async def ensure_colony_self(self) -> None:
|
|
209
|
+
"""Ensure Colony's self-representation exists in the graph (v0.11.0).
|
|
210
|
+
|
|
211
|
+
Creates an Agent node for Colony with DEPENDS_ON edges to
|
|
212
|
+
known Subsystem nodes. Idempotent — safe to call multiple times.
|
|
213
|
+
"""
|
|
214
|
+
try:
|
|
215
|
+
async with self.driver.session(database=self.database) as session:
|
|
216
|
+
# Create Agent node
|
|
217
|
+
await session.run("""
|
|
218
|
+
MERGE (a:Agent {id: 'colony-sidecar'})
|
|
219
|
+
SET a.name = 'Colony',
|
|
220
|
+
a.version = '0.11.1',
|
|
221
|
+
a.status = 'active',
|
|
222
|
+
a.created_at = coalesce(a.created_at, datetime())
|
|
223
|
+
""")
|
|
224
|
+
|
|
225
|
+
# Create Subsystem nodes for known components
|
|
226
|
+
subsystems = [
|
|
227
|
+
("embed_pipeline", "Embedding Pipeline"),
|
|
228
|
+
("delivery_bridge", "Delivery Bridge"),
|
|
229
|
+
("event_bus", "Event Bus"),
|
|
230
|
+
("graph_client", "Graph Client"),
|
|
231
|
+
("initiative_engine", "Initiative Engine"),
|
|
232
|
+
("mind_model", "Mind Model"),
|
|
233
|
+
]
|
|
234
|
+
for sub_id, sub_name in subsystems:
|
|
235
|
+
await session.run("""
|
|
236
|
+
MERGE (s:Subsystem {id: $id})
|
|
237
|
+
SET s.name = $name,
|
|
238
|
+
s.status = coalesce(s.status, 'active'),
|
|
239
|
+
s.created_at = coalesce(s.created_at, datetime())
|
|
240
|
+
""", id=sub_id, name=sub_name)
|
|
241
|
+
|
|
242
|
+
# Create DEPENDS_ON edge from Agent to Subsystem
|
|
243
|
+
await session.run("""
|
|
244
|
+
MATCH (a:Agent {id: 'colony-sidecar'}), (s:Subsystem {id: $id})
|
|
245
|
+
MERGE (a)-[r:DEPENDS_ON]->(s)
|
|
246
|
+
SET r.created_at = coalesce(r.created_at, datetime())
|
|
247
|
+
""", id=sub_id)
|
|
248
|
+
|
|
249
|
+
logger.info("Colony self-representation verified in graph")
|
|
250
|
+
except Exception as e:
|
|
251
|
+
logger.warning("Failed to ensure Colony self-representation: %s", e)
|
|
252
|
+
|
|
253
|
+
async def close(self) -> None:
|
|
254
|
+
"""Cleanly shut down the driver."""
|
|
255
|
+
await self.driver.close()
|
|
256
|
+
|
|
257
|
+
# ------------------------------------------------------------------
|
|
258
|
+
# Embedding helper
|
|
259
|
+
# ------------------------------------------------------------------
|
|
260
|
+
|
|
261
|
+
def set_embed_fn(
|
|
262
|
+
self,
|
|
263
|
+
fn: Callable[[str], Coroutine[Any, Any, List[float]]],
|
|
264
|
+
) -> None:
|
|
265
|
+
"""Register an async embedding function used by *recall*.
|
|
266
|
+
|
|
267
|
+
Args:
|
|
268
|
+
fn: async callable that maps a string to a float vector.
|
|
269
|
+
"""
|
|
270
|
+
self._embed_fn = fn
|
|
271
|
+
owner = getattr(fn, '__self__', None)
|
|
272
|
+
self._embed_query_fn = getattr(owner, 'embed_query', None)
|
|
273
|
+
|
|
274
|
+
def set_vector_store(self, store: "VectorStore") -> None:
|
|
275
|
+
"""Register a VectorStore for ANN search (replaces Neo4j vector index)."""
|
|
276
|
+
self._vector_store = store
|
|
277
|
+
|
|
278
|
+
def set_rerank_fn(
|
|
279
|
+
self,
|
|
280
|
+
fn: Callable[..., Coroutine[Any, Any, Any]],
|
|
281
|
+
*,
|
|
282
|
+
calibration_metadata: Optional[Callable[[], Dict[str, Any]]] = None,
|
|
283
|
+
) -> None:
|
|
284
|
+
"""Register an async rerank function used by *recall* (mirrors
|
|
285
|
+
:meth:`set_embed_fn`).
|
|
286
|
+
|
|
287
|
+
Args:
|
|
288
|
+
fn: async callable ``fn(query, documents, top_k=N)`` returning a
|
|
289
|
+
list of objects with ``index`` and ``score`` attributes (the
|
|
290
|
+
RerankerProvider.rerank contract). Only consulted when
|
|
291
|
+
COLONY_RECALL_RERANK is ``shadow`` or ``on``.
|
|
292
|
+
calibration_metadata: current provider/model/format configuration.
|
|
293
|
+
Optional calibrated abstention is invalidated when this changes.
|
|
294
|
+
"""
|
|
295
|
+
self._rerank_fn = fn
|
|
296
|
+
self._rerank_calibration_metadata = calibration_metadata
|
|
297
|
+
self._recall_selector = None
|
|
298
|
+
|
|
299
|
+
def set_adaptive_params(self, params: Any) -> None:
|
|
300
|
+
"""Register an AdaptiveParamStore consulted by *recall* for the
|
|
301
|
+
meta-learned relevance floor (recall.min_relevance)."""
|
|
302
|
+
self._adaptive_params = params
|
|
303
|
+
|
|
304
|
+
async def _embed(self, text: str) -> List[float]:
|
|
305
|
+
"""Produce an embedding vector for *text* (document side).
|
|
306
|
+
|
|
307
|
+
Raises:
|
|
308
|
+
RuntimeError: If no embedding function has been registered.
|
|
309
|
+
"""
|
|
310
|
+
if self._embed_fn is None:
|
|
311
|
+
raise RuntimeError(
|
|
312
|
+
"No embedding function registered. Call set_embed_fn() first."
|
|
313
|
+
)
|
|
314
|
+
return await self._embed_fn(text)
|
|
315
|
+
|
|
316
|
+
async def _embed_query(self, query: str) -> List[float]:
|
|
317
|
+
"""Embed a *query* for retrieval (v0.21.1).
|
|
318
|
+
|
|
319
|
+
Instruct-tuned embedders (Qwen3-Embedding, E5, BGE, …) are ASYMMETRIC:
|
|
320
|
+
the query gets an instruction prefix while documents do not. Without it,
|
|
321
|
+
retrieval quality collapses (every result lands at ~0.9 cosine distance
|
|
322
|
+
and the right memory never surfaces). Configurable via
|
|
323
|
+
COLONY_EMBED_QUERY_INSTRUCTION (set to empty for symmetric models).
|
|
324
|
+
"""
|
|
325
|
+
if callable(getattr(self, '_embed_query_fn', None)):
|
|
326
|
+
return await self._embed_query_fn(query)
|
|
327
|
+
import os
|
|
328
|
+
default_instr = ("Instruct: Given a search query, retrieve relevant "
|
|
329
|
+
"memories that answer it\nQuery: ")
|
|
330
|
+
instr = os.environ.get("COLONY_EMBED_QUERY_INSTRUCTION", default_instr)
|
|
331
|
+
return await self._embed((instr + query) if instr else query)
|
|
332
|
+
|
|
333
|
+
@staticmethod
|
|
334
|
+
def compute_effective_confidence(
|
|
335
|
+
base_confidence: float,
|
|
336
|
+
source_reliability: float,
|
|
337
|
+
corroboration_count: int,
|
|
338
|
+
contradiction_count: int,
|
|
339
|
+
recalls: int,
|
|
340
|
+
last_verified_at: Optional[Any],
|
|
341
|
+
created_at: Any,
|
|
342
|
+
epistemic_state: str,
|
|
343
|
+
now: Any,
|
|
344
|
+
) -> float:
|
|
345
|
+
"""Compute effective confidence from multiple signals.
|
|
346
|
+
|
|
347
|
+
Args:
|
|
348
|
+
base_confidence: Initial confidence (0-1)
|
|
349
|
+
source_reliability: Reliability of the source (0-1)
|
|
350
|
+
corroboration_count: Number of corroborating memories
|
|
351
|
+
contradiction_count: Number of contradicting memories
|
|
352
|
+
recalls: Number of times recalled
|
|
353
|
+
last_verified_at: Last verification timestamp or None
|
|
354
|
+
created_at: Creation timestamp
|
|
355
|
+
epistemic_state: Current epistemic state string
|
|
356
|
+
now: Current timestamp
|
|
357
|
+
|
|
358
|
+
Returns:
|
|
359
|
+
Effective confidence in [0, 1].
|
|
360
|
+
"""
|
|
361
|
+
from datetime import datetime as _dt, timezone as _tz
|
|
362
|
+
|
|
363
|
+
if now is None:
|
|
364
|
+
now = _dt.now(_tz.utc)
|
|
365
|
+
if hasattr(now, "to_native"):
|
|
366
|
+
now = now.to_native()
|
|
367
|
+
if isinstance(now, str):
|
|
368
|
+
now = _dt.fromisoformat(now.replace("Z", "+00:00"))
|
|
369
|
+
if hasattr(created_at, "to_native"):
|
|
370
|
+
created_at = created_at.to_native()
|
|
371
|
+
if isinstance(created_at, str):
|
|
372
|
+
created_at = _dt.fromisoformat(created_at.replace("Z", "+00:00"))
|
|
373
|
+
|
|
374
|
+
# Source weight
|
|
375
|
+
confidence = base_confidence * source_reliability
|
|
376
|
+
|
|
377
|
+
# Corroboration / contradiction adjustment
|
|
378
|
+
net_support = corroboration_count - contradiction_count
|
|
379
|
+
confidence *= min(1.0, 1.0 + net_support * 0.1)
|
|
380
|
+
|
|
381
|
+
# Recall reinforcement (diminishing returns)
|
|
382
|
+
confidence *= min(1.3, 1.0 + recalls * 0.03)
|
|
383
|
+
|
|
384
|
+
# Recency weighting (v0.21.0, configurable half-life + floor — see
|
|
385
|
+
# _recency_factor). Recent memories surface; VERIFIED memories are
|
|
386
|
+
# additionally floored at 0.9 by the epistemic-state clamp below.
|
|
387
|
+
days_old = max(0, (now - created_at).days)
|
|
388
|
+
recency_factor = _recency_factor(days_old)
|
|
389
|
+
confidence *= recency_factor
|
|
390
|
+
|
|
391
|
+
# Verification boost
|
|
392
|
+
if last_verified_at:
|
|
393
|
+
if hasattr(last_verified_at, "to_native"):
|
|
394
|
+
last_verified_at = last_verified_at.to_native()
|
|
395
|
+
if isinstance(last_verified_at, str):
|
|
396
|
+
last_verified_at = _dt.fromisoformat(last_verified_at.replace("Z", "+00:00"))
|
|
397
|
+
if (now - last_verified_at).days < 7:
|
|
398
|
+
confidence *= 1.2
|
|
399
|
+
|
|
400
|
+
# State clamp
|
|
401
|
+
if epistemic_state == EpistemicState.VERIFIED.value:
|
|
402
|
+
confidence = max(confidence, 0.9)
|
|
403
|
+
elif epistemic_state in (EpistemicState.STALE.value, EpistemicState.SUPERSEDED.value):
|
|
404
|
+
confidence *= 0.3
|
|
405
|
+
elif epistemic_state == EpistemicState.DEPRECATED.value:
|
|
406
|
+
confidence *= 0.1
|
|
407
|
+
|
|
408
|
+
return min(1.0, max(0.0, confidence))
|
|
409
|
+
|
|
410
|
+
# ------------------------------------------------------------------
|
|
411
|
+
# Core operations
|
|
412
|
+
# ------------------------------------------------------------------
|
|
413
|
+
|
|
414
|
+
@staticmethod
|
|
415
|
+
def _source_projection_erased(source_uri: str | None) -> bool:
|
|
416
|
+
if not source_uri or not source_uri.startswith("turn:"):
|
|
417
|
+
return False
|
|
418
|
+
from apsimo import get_state_dir
|
|
419
|
+
from apsimo.turns import get_turn_idempotency_ledger
|
|
420
|
+
return get_turn_idempotency_ledger(get_state_dir()).is_projection_erased(source_uri[5:])
|
|
421
|
+
|
|
422
|
+
async def _filter_erased_source_memories(self, memories: list[dict]) -> list[dict]:
|
|
423
|
+
"""A durable erase fence remains effective during projection outages."""
|
|
424
|
+
return [memory for memory in memories
|
|
425
|
+
if not self._source_projection_erased(memory.get("source_uri"))]
|
|
426
|
+
|
|
427
|
+
async def delete_source_memories(self, turn_ids: list[str]) -> int:
|
|
428
|
+
"""Remove graph and vector copies after the source fence commits."""
|
|
429
|
+
uris = ["turn:" + item for item in dict.fromkeys(turn_ids)]
|
|
430
|
+
if not uris:
|
|
431
|
+
return 0
|
|
432
|
+
# Keep IDs until both stores have been addressed. The graph's source
|
|
433
|
+
# tombstone filter blocks recall even if vector deletion must be retried.
|
|
434
|
+
async with self.driver.session(database=self.database) as session:
|
|
435
|
+
result = await session.run("MATCH (m:Memory) WHERE m.source_uri IN $uris RETURN m.id AS id", uris=uris)
|
|
436
|
+
ids = [record["id"] async for record in result]
|
|
437
|
+
for memory_id in ids:
|
|
438
|
+
if self._vector_store is not None:
|
|
439
|
+
from apsimo.vector.collections import Collection
|
|
440
|
+
await self._vector_store.delete(collection=Collection.MEMORIES, id=memory_id)
|
|
441
|
+
async with self.driver.session(database=self.database) as session:
|
|
442
|
+
await session.run("MATCH (m:Memory {id: $id}) DETACH DELETE m", id=memory_id)
|
|
443
|
+
ring = getattr(self, "_distill_preview", None)
|
|
444
|
+
if ring is not None:
|
|
445
|
+
retained = [item for item in ring if item.get("source_turn_id") not in turn_ids]
|
|
446
|
+
ring.clear()
|
|
447
|
+
ring.extend(retained)
|
|
448
|
+
return len(ids)
|
|
449
|
+
|
|
450
|
+
async def iter_indexable_memories(self, batch_size: int = 128):
|
|
451
|
+
"""Current graph evidence for a projection rebuild, without old vectors."""
|
|
452
|
+
after = ''
|
|
453
|
+
while True:
|
|
454
|
+
async with self.driver.session(database=self.database) as session:
|
|
455
|
+
result = await session.run('''
|
|
456
|
+
MATCH (m:Memory) WHERE m.id > $after
|
|
457
|
+
WITH m ORDER BY m.id LIMIT $limit
|
|
458
|
+
OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
|
|
459
|
+
RETURN properties(m) AS memory, collect(p.id) AS people
|
|
460
|
+
ORDER BY memory.id
|
|
461
|
+
''', after=after, limit=batch_size)
|
|
462
|
+
rows = [dict(row) async for row in result]
|
|
463
|
+
if not rows:
|
|
464
|
+
return
|
|
465
|
+
for row in rows:
|
|
466
|
+
memory = row['memory']
|
|
467
|
+
after = memory['id']
|
|
468
|
+
if self._source_projection_erased(memory.get('source_uri')):
|
|
469
|
+
continue
|
|
470
|
+
if memory.get('superseded_by') or not memory.get('content'):
|
|
471
|
+
continue
|
|
472
|
+
metadata = memory.get('metadata') or {}
|
|
473
|
+
if isinstance(metadata, str):
|
|
474
|
+
try:
|
|
475
|
+
metadata = json.loads(metadata)
|
|
476
|
+
except json.JSONDecodeError:
|
|
477
|
+
# Earlier store_memory writers used str(dict), not JSON.
|
|
478
|
+
# Read those retained records without rewriting evidence.
|
|
479
|
+
metadata = ast.literal_eval(metadata)
|
|
480
|
+
if not isinstance(metadata, dict):
|
|
481
|
+
raise ValueError('Graph memory metadata must be an object')
|
|
482
|
+
metadata = {**metadata, **{key: memory.get(key) for key in
|
|
483
|
+
('source_uri', 'source_type', 'source_version', 'content_hash', 'session_id',
|
|
484
|
+
'type', 'strength', 'effective_confidence', 'epistemic_state', 'protected')}}
|
|
485
|
+
metadata['person_id'] = (row['people'] or [None])[0]
|
|
486
|
+
metadata['memory_id'] = memory['id']
|
|
487
|
+
yield {'id': memory['id'], 'text': memory['content'], 'metadata': metadata}
|
|
488
|
+
|
|
489
|
+
async def store_memory(
|
|
490
|
+
self,
|
|
491
|
+
content: str,
|
|
492
|
+
memory_type: str,
|
|
493
|
+
entities: List[str],
|
|
494
|
+
metadata: Dict[str, Any] | None = None,
|
|
495
|
+
importance: float = 1.0,
|
|
496
|
+
person_id: Optional[str] = None,
|
|
497
|
+
session_id: Optional[str] = None,
|
|
498
|
+
source_type: str = "inference",
|
|
499
|
+
source_uri: Optional[str] = None,
|
|
500
|
+
source_version: Optional[str] = None,
|
|
501
|
+
content_hash: Optional[str] = None,
|
|
502
|
+
) -> str:
|
|
503
|
+
"""Store a memory with automatic entity linking.
|
|
504
|
+
|
|
505
|
+
Creates a :Memory node and :MENTIONS edges to each :Entity.
|
|
506
|
+
If an embedding function is registered the memory's vector is stored
|
|
507
|
+
on the node so it participates in semantic search.
|
|
508
|
+
|
|
509
|
+
Args:
|
|
510
|
+
content: Memory content text
|
|
511
|
+
memory_type: Type of memory (episodic, semantic, procedural, identity)
|
|
512
|
+
entities: Named entities to link to this memory
|
|
513
|
+
metadata: Optional key-value metadata
|
|
514
|
+
importance: Initial importance / strength (0-1, default 1.0)
|
|
515
|
+
person_id: Optional person ID to link this memory to via (Memory)-[:ABOUT]->(Person)
|
|
516
|
+
source_type: Origin of this memory (conversation, file, tool_output, user_assertion, inference)
|
|
517
|
+
source_uri: Optional URI referencing the source
|
|
518
|
+
source_version: Optional version string for the source
|
|
519
|
+
content_hash: Optional SHA-256 hash of the content
|
|
520
|
+
|
|
521
|
+
Returns:
|
|
522
|
+
The UUID of the newly created Memory node.
|
|
523
|
+
"""
|
|
524
|
+
metadata = metadata or {}
|
|
525
|
+
if self._source_projection_erased(source_uri):
|
|
526
|
+
return ""
|
|
527
|
+
# Guard (v0.21.1): never store empty/whitespace memories — they were
|
|
528
|
+
# accumulating as duplicate junk nodes.
|
|
529
|
+
if not content or not content.strip():
|
|
530
|
+
return ""
|
|
531
|
+
max_importance = MAX_IMPORTANCE.get(source_type, 0.7)
|
|
532
|
+
if importance > max_importance:
|
|
533
|
+
logger.warning(
|
|
534
|
+
"Importance %.2f for source_type '%s' exceeds max %.2f; clamping.",
|
|
535
|
+
importance, source_type, max_importance,
|
|
536
|
+
)
|
|
537
|
+
importance = max_importance
|
|
538
|
+
importance = max(0.0, min(1.0, importance))
|
|
539
|
+
|
|
540
|
+
# Resolve person_id from explicit arg or metadata fallback
|
|
541
|
+
person_id = person_id or metadata.get("person_id")
|
|
542
|
+
# Never mint a :Person node for a non-contact sentinel. Ids like 'default'/'unknown'/empty are
|
|
543
|
+
# NOT real people; attributing memories to them created a catch-all junk Person that polluted
|
|
544
|
+
# per-person recall (a host whose resolver fell back to "default" dumped most of its memory
|
|
545
|
+
# onto one pseudo-contact). Such memories are stored UNATTRIBUTED (no :ABOUT edge) instead.
|
|
546
|
+
# Real contact ids still create/link a Person normally (that is the graph's discovery design).
|
|
547
|
+
if isinstance(person_id, str):
|
|
548
|
+
person_id = person_id.strip() or None
|
|
549
|
+
if person_id and person_id.lower() in {"default", "unknown", "none", "null", "anonymous"}:
|
|
550
|
+
person_id = None
|
|
551
|
+
elif person_id is not None:
|
|
552
|
+
person_id = None
|
|
553
|
+
# Resolve session_id from explicit arg or metadata fallback
|
|
554
|
+
session_id = session_id or (metadata.get("session_id") if metadata else None)
|
|
555
|
+
|
|
556
|
+
source_type = (source_type or MemorySourceType.INFERENCE.value).lower()
|
|
557
|
+
source_uri = source_uri or None
|
|
558
|
+
source_version = source_version or None
|
|
559
|
+
# Always derive a content hash so identical memories can be deduped
|
|
560
|
+
# (v0.21.1 — previously null, so every write created a duplicate node).
|
|
561
|
+
content_hash = content_hash or hashlib.sha256(content.encode("utf-8")).hexdigest()
|
|
562
|
+
source_reliability = SOURCE_RELIABILITY.get(source_type, 0.5)
|
|
563
|
+
protected = source_type == MemorySourceType.USER_ASSERTION
|
|
564
|
+
base_confidence = importance
|
|
565
|
+
epistemic_state = EpistemicState.INFERRED
|
|
566
|
+
created_at = self._utcnow()
|
|
567
|
+
|
|
568
|
+
effective_confidence = self.compute_effective_confidence(
|
|
569
|
+
base_confidence=base_confidence,
|
|
570
|
+
source_reliability=source_reliability,
|
|
571
|
+
corroboration_count=0,
|
|
572
|
+
contradiction_count=0,
|
|
573
|
+
recalls=0,
|
|
574
|
+
last_verified_at=None,
|
|
575
|
+
created_at=created_at,
|
|
576
|
+
epistemic_state=epistemic_state.value,
|
|
577
|
+
now=created_at,
|
|
578
|
+
)
|
|
579
|
+
|
|
580
|
+
# Dedup (v0.21.1): if an identical memory already exists (same content
|
|
581
|
+
# hash), reinforce it instead of creating a duplicate — and skip the
|
|
582
|
+
# embed cost entirely. This collapses the runaway duplication (e.g. a
|
|
583
|
+
# recurring cron prompt that had been stored 70+ times).
|
|
584
|
+
if content_hash:
|
|
585
|
+
async with self.driver.session(database=self.database) as session:
|
|
586
|
+
dq = await session.run(
|
|
587
|
+
"""
|
|
588
|
+
MATCH (m:Memory {content_hash: $content_hash})
|
|
589
|
+
WHERE m.superseded_by IS NULL
|
|
590
|
+
AND ($source_uri IS NULL OR m.source_uri = $source_uri)
|
|
591
|
+
SET m.accessed_at = datetime(),
|
|
592
|
+
m.corroboration_count = coalesce(m.corroboration_count, 0) + 1,
|
|
593
|
+
m.strength = CASE WHEN coalesce(m.strength, 0.0) < 1.0
|
|
594
|
+
THEN coalesce(m.strength, 0.0) + 0.05 ELSE 1.0 END
|
|
595
|
+
RETURN m.id AS id
|
|
596
|
+
ORDER BY m.created_at ASC
|
|
597
|
+
LIMIT 1
|
|
598
|
+
""",
|
|
599
|
+
content_hash=content_hash,
|
|
600
|
+
source_uri=source_uri,
|
|
601
|
+
)
|
|
602
|
+
existing = await dq.single()
|
|
603
|
+
if existing is not None:
|
|
604
|
+
logger.debug("store_memory dedup: reinforced %s", existing["id"])
|
|
605
|
+
return existing["id"]
|
|
606
|
+
|
|
607
|
+
# Compute embedding. If an embedder is configured but we can't get a
|
|
608
|
+
# usable vector (outage that survived retries), FAIL the write rather than
|
|
609
|
+
# create an unsearchable memory (silent loss). The vector lives in the
|
|
610
|
+
# LanceDB store; the Neo4j m.embedding property is secondary/best-effort.
|
|
611
|
+
embedding: Optional[List[float]] = None
|
|
612
|
+
if self._embed_fn is not None:
|
|
613
|
+
embedding = await self._embed(content)
|
|
614
|
+
if not embedding:
|
|
615
|
+
raise RuntimeError(
|
|
616
|
+
"embedding unavailable — refusing to store unsearchable memory")
|
|
617
|
+
|
|
618
|
+
# Preserve dict for vector store before stringifying for Neo4j
|
|
619
|
+
metadata_dict = metadata
|
|
620
|
+
metadata_str = json.dumps(metadata_dict, default=str)
|
|
621
|
+
|
|
622
|
+
async with self.driver.session(database=self.database) as session:
|
|
623
|
+
result = await session.run(
|
|
624
|
+
"""
|
|
625
|
+
CREATE (m:Memory {
|
|
626
|
+
id: randomUUID(),
|
|
627
|
+
content: $content,
|
|
628
|
+
type: $memory_type,
|
|
629
|
+
importance: $importance,
|
|
630
|
+
strength: $importance,
|
|
631
|
+
recalls: 0,
|
|
632
|
+
created_at: datetime(),
|
|
633
|
+
accessed_at: datetime(),
|
|
634
|
+
embedding: $embedding,
|
|
635
|
+
metadata: $metadata,
|
|
636
|
+
session_id: $session_id,
|
|
637
|
+
source_type: $source_type,
|
|
638
|
+
source_uri: $source_uri,
|
|
639
|
+
source_version: $source_version,
|
|
640
|
+
content_hash: $content_hash,
|
|
641
|
+
base_confidence: $base_confidence,
|
|
642
|
+
source_reliability: $source_reliability,
|
|
643
|
+
corroboration_count: 0,
|
|
644
|
+
contradiction_count: 0,
|
|
645
|
+
effective_confidence: $effective_confidence,
|
|
646
|
+
epistemic_state: $epistemic_state,
|
|
647
|
+
protected: $protected,
|
|
648
|
+
last_verified_at: null,
|
|
649
|
+
superseded_by: null,
|
|
650
|
+
provenance: []
|
|
651
|
+
})
|
|
652
|
+
WITH m
|
|
653
|
+
FOREACH (entity_name IN $entities |
|
|
654
|
+
MERGE (e:Entity {name: entity_name})
|
|
655
|
+
CREATE (m)-[:MENTIONS]->(e)
|
|
656
|
+
)
|
|
657
|
+
WITH m
|
|
658
|
+
FOREACH (_ IN CASE WHEN $person_id IS NOT NULL THEN [1] ELSE [] END |
|
|
659
|
+
MERGE (p:Person {id: $person_id})
|
|
660
|
+
CREATE (m)-[:ABOUT]->(p)
|
|
661
|
+
)
|
|
662
|
+
WITH m
|
|
663
|
+
FOREACH (_ IN CASE WHEN $source_uri IS NOT NULL AND $source_type = "file" THEN [1] ELSE [] END |
|
|
664
|
+
MERGE (fa:FileAnchor {uri: $source_uri})
|
|
665
|
+
ON CREATE SET fa.first_seen = datetime()
|
|
666
|
+
CREATE (m)-[:DERIVED_FROM {derivation_type: "file_read"}]->(fa)
|
|
667
|
+
)
|
|
668
|
+
RETURN m.id AS id
|
|
669
|
+
""",
|
|
670
|
+
content=content,
|
|
671
|
+
memory_type=memory_type,
|
|
672
|
+
importance=importance,
|
|
673
|
+
entities=entities,
|
|
674
|
+
embedding=embedding,
|
|
675
|
+
metadata=metadata_str,
|
|
676
|
+
person_id=person_id,
|
|
677
|
+
session_id=session_id,
|
|
678
|
+
source_type=source_type,
|
|
679
|
+
source_uri=source_uri,
|
|
680
|
+
source_version=source_version,
|
|
681
|
+
content_hash=content_hash,
|
|
682
|
+
base_confidence=base_confidence,
|
|
683
|
+
source_reliability=source_reliability,
|
|
684
|
+
effective_confidence=effective_confidence,
|
|
685
|
+
epistemic_state=epistemic_state,
|
|
686
|
+
protected=protected,
|
|
687
|
+
)
|
|
688
|
+
record = await result.single()
|
|
689
|
+
if record is None:
|
|
690
|
+
raise RuntimeError("Failed to create memory node")
|
|
691
|
+
memory_id = record["id"]
|
|
692
|
+
|
|
693
|
+
# Write to LanceDB vector store (if configured)
|
|
694
|
+
if self._vector_store is not None and embedding is not None:
|
|
695
|
+
try:
|
|
696
|
+
from apsimo.vector.collections import Collection
|
|
697
|
+
await self._vector_store.add(
|
|
698
|
+
collection=Collection.MEMORIES,
|
|
699
|
+
id=memory_id,
|
|
700
|
+
text=content,
|
|
701
|
+
vector=embedding,
|
|
702
|
+
metadata={
|
|
703
|
+
"memory_id": memory_id,
|
|
704
|
+
"type": memory_type,
|
|
705
|
+
"strength": importance,
|
|
706
|
+
"importance": importance,
|
|
707
|
+
# Keep the vector candidate's scope aligned with the
|
|
708
|
+
# authoritative graph ABOUT edge. Historically this
|
|
709
|
+
# read only metadata["person_id"], so callers using the
|
|
710
|
+
# explicit argument produced unscoped vector rows.
|
|
711
|
+
"person_id": person_id,
|
|
712
|
+
"tags": metadata_dict.get("tags", []) if metadata_dict else [],
|
|
713
|
+
"created_at": metadata_dict.get("created_at") if metadata_dict else None,
|
|
714
|
+
"session_id": session_id,
|
|
715
|
+
"source_type": source_type,
|
|
716
|
+
"source_uri": source_uri,
|
|
717
|
+
"source_version": source_version,
|
|
718
|
+
"content_hash": content_hash,
|
|
719
|
+
"effective_confidence": effective_confidence,
|
|
720
|
+
"epistemic_state": epistemic_state.value,
|
|
721
|
+
"protected": protected,
|
|
722
|
+
},
|
|
723
|
+
)
|
|
724
|
+
except Exception as exc:
|
|
725
|
+
logger.warning("Failed to write memory to vector store: %s", exc)
|
|
726
|
+
|
|
727
|
+
# A deletion may have committed while embedding or graph I/O awaited.
|
|
728
|
+
if self._source_projection_erased(source_uri):
|
|
729
|
+
await self.delete_source_memories([source_uri[5:]])
|
|
730
|
+
return ""
|
|
731
|
+
return memory_id
|
|
732
|
+
|
|
733
|
+
def _distill_preview_ring(self) -> "deque":
|
|
734
|
+
"""Bounded ring of shadow distill previews (created on first use so
|
|
735
|
+
alternate construction paths, e.g. tests, still work)."""
|
|
736
|
+
ring = getattr(self, "_distill_preview", None)
|
|
737
|
+
if ring is None:
|
|
738
|
+
ring = deque(maxlen=50)
|
|
739
|
+
self._distill_preview = ring
|
|
740
|
+
return ring
|
|
741
|
+
|
|
742
|
+
def distill_preview(self) -> List[Dict[str, Any]]:
|
|
743
|
+
"""Newest-first shadow distill previews (empty once the flag is live)."""
|
|
744
|
+
return list(reversed(self._distill_preview_ring()))
|
|
745
|
+
|
|
746
|
+
async def record_turn(
|
|
747
|
+
self,
|
|
748
|
+
session_id: str,
|
|
749
|
+
contact_id: Optional[str],
|
|
750
|
+
topics: List[str],
|
|
751
|
+
entities: List[str],
|
|
752
|
+
tools_used: List[str],
|
|
753
|
+
summary: Optional[str],
|
|
754
|
+
turn_id: Optional[str] = None,
|
|
755
|
+
) -> Optional[str]:
|
|
756
|
+
"""Store a conversation turn as an episodic memory.
|
|
757
|
+
|
|
758
|
+
Creates a :Memory node of type ``episodic`` linked to the
|
|
759
|
+
conversation session and contact. Entities and topics are
|
|
760
|
+
merged as :Entity nodes. Tools used are stored in metadata.
|
|
761
|
+
|
|
762
|
+
Args:
|
|
763
|
+
session_id: Hermes session identifier
|
|
764
|
+
contact_id: Optional contact / person identifier
|
|
765
|
+
topics: Extracted topics from the turn
|
|
766
|
+
entities: Named entities mentioned
|
|
767
|
+
tools_used: Tool names invoked during the turn
|
|
768
|
+
summary: Human-readable summary of the exchange
|
|
769
|
+
|
|
770
|
+
Returns:
|
|
771
|
+
The UUID of the created Memory node, or None if storage fails.
|
|
772
|
+
"""
|
|
773
|
+
if not summary or self._source_projection_erased("turn:" + turn_id if turn_id else None):
|
|
774
|
+
return None
|
|
775
|
+
# Salience gate: don't memorialize internal-plumbing turns (context-compaction references and
|
|
776
|
+
# host-specific system-prompt wrappers / self-checks). Generic markers are built in; a
|
|
777
|
+
# deployment adds its own via COLONY_MEMORY_SKIP_MARKERS ('|'-separated, case-insensitive).
|
|
778
|
+
# This is what keeps the memory graph facts-and-events, not a verbatim transcript log.
|
|
779
|
+
_sl = summary.lower()
|
|
780
|
+
_skip = ("[context compaction", "[post-compaction", "reference only]", "[context summary]")
|
|
781
|
+
_env = os.environ.get("COLONY_MEMORY_SKIP_MARKERS", "")
|
|
782
|
+
if any(m in _sl for m in _skip) or any(m.strip().lower() in _sl for m in _env.split("|") if m.strip()):
|
|
783
|
+
logger.debug("record_turn: skipped low-salience / internal-marker turn")
|
|
784
|
+
return None
|
|
785
|
+
|
|
786
|
+
# Real salience score (attribution redesign Phase 2), replacing the old hardcoded
|
|
787
|
+
# importance=0.85 that overrode the computed value. Signal from what the turn
|
|
788
|
+
# actually carries: named entities (facts about people/things), tool use (an
|
|
789
|
+
# action happened), and substance (length). A throwaway "ok thanks" scores low
|
|
790
|
+
# and decays fast; a fact-dense exchange scores high and persists.
|
|
791
|
+
_ent_n = len(entities or [])
|
|
792
|
+
_score = 0.35
|
|
793
|
+
_score += min(0.30, 0.10 * _ent_n) # up to +0.30 for entities
|
|
794
|
+
if tools_used:
|
|
795
|
+
_score += 0.15 # an action was taken
|
|
796
|
+
if len(summary) > 240:
|
|
797
|
+
_score += 0.10 # substantive exchange
|
|
798
|
+
if "?" in summary:
|
|
799
|
+
_score += 0.05 # a question = intent/curiosity worth recalling
|
|
800
|
+
importance = round(min(_score, 0.95), 3)
|
|
801
|
+
|
|
802
|
+
# Optional distillation (shadow by default): store the salient content rather
|
|
803
|
+
# than the verbatim "User:/Agent:" wrapper. The distilled form is ALWAYS
|
|
804
|
+
# computed; off => stored content is unchanged and the would-be result goes
|
|
805
|
+
# into a bounded in-memory preview ring (GET /v1/host/memory/distill-preview)
|
|
806
|
+
# so the flip can be validated on real traffic first. On => store it.
|
|
807
|
+
content = summary
|
|
808
|
+
distilled = distill_turn_summary(summary)
|
|
809
|
+
_distill = os.environ.get("COLONY_DISTILL_TURNS", "0") not in ("0", "false", "no")
|
|
810
|
+
if _distill:
|
|
811
|
+
content = distilled
|
|
812
|
+
else:
|
|
813
|
+
try:
|
|
814
|
+
self._distill_preview_ring().append({
|
|
815
|
+
"session_id": session_id,
|
|
816
|
+
"source_turn_id": turn_id,
|
|
817
|
+
"original": summary[:400],
|
|
818
|
+
"distilled": distilled[:400],
|
|
819
|
+
"importance": importance,
|
|
820
|
+
"ts": time.time(),
|
|
821
|
+
})
|
|
822
|
+
except Exception:
|
|
823
|
+
logger.debug("distill preview append failed", exc_info=True)
|
|
824
|
+
logger.debug("distill(shadow): would store salient content for session %s (imp=%.2f)",
|
|
825
|
+
session_id, importance)
|
|
826
|
+
|
|
827
|
+
metadata: Dict[str, Any] = {
|
|
828
|
+
"turn": True,
|
|
829
|
+
"source_turn_id": turn_id,
|
|
830
|
+
"topics": topics,
|
|
831
|
+
"tools_used": tools_used,
|
|
832
|
+
"salience": importance,
|
|
833
|
+
}
|
|
834
|
+
|
|
835
|
+
try:
|
|
836
|
+
content_hash = hashlib.sha256(content.encode("utf-8")).hexdigest()
|
|
837
|
+
return await self.store_memory(
|
|
838
|
+
content=content,
|
|
839
|
+
memory_type="episodic",
|
|
840
|
+
entities=entities or [],
|
|
841
|
+
metadata=metadata,
|
|
842
|
+
importance=importance,
|
|
843
|
+
person_id=contact_id,
|
|
844
|
+
source_type=MemorySourceType.CONVERSATION.value,
|
|
845
|
+
source_uri=f"turn:{turn_id}" if turn_id else f"session:{session_id}",
|
|
846
|
+
session_id=session_id,
|
|
847
|
+
content_hash=content_hash,
|
|
848
|
+
)
|
|
849
|
+
except Exception as exc:
|
|
850
|
+
logger.warning("record_turn failed: %s", exc)
|
|
851
|
+
return None
|
|
852
|
+
|
|
853
|
+
async def read_memories(
|
|
854
|
+
self,
|
|
855
|
+
*,
|
|
856
|
+
person_id: Optional[str] = None,
|
|
857
|
+
memory_id: Optional[str] = None,
|
|
858
|
+
limit: int = 20,
|
|
859
|
+
) -> List[Dict[str, Any]]:
|
|
860
|
+
"""Read recent memories, preserving the same hard ABOUT boundary as recall.
|
|
861
|
+
|
|
862
|
+
``person_id`` is intentionally a single exact lane. A scoped miss is
|
|
863
|
+
empty and never retries against the unscoped graph. Omitting the person
|
|
864
|
+
remains only for the deprecated legacy/internal authority path; the API
|
|
865
|
+
boundary always supplies a server-derived person for scoped callers.
|
|
866
|
+
"""
|
|
867
|
+
|
|
868
|
+
person_scope = str(person_id).strip() if person_id else None
|
|
869
|
+
memory_scope = str(memory_id).strip() if memory_id else None
|
|
870
|
+
safe_limit = max(1, min(int(limit or 20), 100))
|
|
871
|
+
excluded_sources = getattr(self, "_recall_source_exclusions", ())
|
|
872
|
+
excluded_metadata_markers = getattr(
|
|
873
|
+
self, "_recall_metadata_exclusions", ())
|
|
874
|
+
policy_active = bool(
|
|
875
|
+
excluded_sources or excluded_metadata_markers)
|
|
876
|
+
async with self.driver.session(database=self.database) as session:
|
|
877
|
+
if person_scope:
|
|
878
|
+
if policy_active:
|
|
879
|
+
result = await session.run(
|
|
880
|
+
"""
|
|
881
|
+
MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
|
|
882
|
+
WHERE ($memory_id IS NULL OR m.id = $memory_id)
|
|
883
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
884
|
+
coalesce(m.source_uri, "") IN
|
|
885
|
+
$exclude_source_uris))
|
|
886
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
887
|
+
NONE(marker IN $exclude_metadata_markers
|
|
888
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
889
|
+
CONTAINS marker))
|
|
890
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
891
|
+
WITH m, collect(DISTINCT e.name) AS entity_names
|
|
892
|
+
RETURN m {.*, entities: entity_names,
|
|
893
|
+
person_id: $person_id} AS memory
|
|
894
|
+
ORDER BY m.created_at DESC
|
|
895
|
+
LIMIT $limit
|
|
896
|
+
""",
|
|
897
|
+
person_id=person_scope,
|
|
898
|
+
memory_id=memory_scope,
|
|
899
|
+
limit=safe_limit,
|
|
900
|
+
exclude_source_uris=excluded_sources,
|
|
901
|
+
exclude_metadata_markers=excluded_metadata_markers,
|
|
902
|
+
)
|
|
903
|
+
else:
|
|
904
|
+
result = await session.run(
|
|
905
|
+
"""
|
|
906
|
+
MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
|
|
907
|
+
WHERE $memory_id IS NULL OR m.id = $memory_id
|
|
908
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
909
|
+
WITH m, collect(DISTINCT e.name) AS entity_names
|
|
910
|
+
RETURN m {.*, entities: entity_names,
|
|
911
|
+
person_id: $person_id} AS memory
|
|
912
|
+
ORDER BY m.created_at DESC
|
|
913
|
+
LIMIT $limit
|
|
914
|
+
""",
|
|
915
|
+
person_id=person_scope,
|
|
916
|
+
memory_id=memory_scope,
|
|
917
|
+
limit=safe_limit,
|
|
918
|
+
)
|
|
919
|
+
else:
|
|
920
|
+
if policy_active:
|
|
921
|
+
result = await session.run(
|
|
922
|
+
"""
|
|
923
|
+
MATCH (m:Memory)
|
|
924
|
+
WHERE ($memory_id IS NULL OR m.id = $memory_id)
|
|
925
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
926
|
+
coalesce(m.source_uri, "") IN
|
|
927
|
+
$exclude_source_uris))
|
|
928
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
929
|
+
NONE(marker IN $exclude_metadata_markers
|
|
930
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
931
|
+
CONTAINS marker))
|
|
932
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
933
|
+
OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
|
|
934
|
+
WITH m, collect(DISTINCT e.name) AS entity_names,
|
|
935
|
+
collect(DISTINCT p.id) AS person_ids
|
|
936
|
+
RETURN m {.*, entities: entity_names,
|
|
937
|
+
person_id: head(person_ids)} AS memory
|
|
938
|
+
ORDER BY m.created_at DESC
|
|
939
|
+
LIMIT $limit
|
|
940
|
+
""",
|
|
941
|
+
memory_id=memory_scope,
|
|
942
|
+
limit=safe_limit,
|
|
943
|
+
exclude_source_uris=excluded_sources,
|
|
944
|
+
exclude_metadata_markers=excluded_metadata_markers,
|
|
945
|
+
)
|
|
946
|
+
else:
|
|
947
|
+
result = await session.run(
|
|
948
|
+
"""
|
|
949
|
+
MATCH (m:Memory)
|
|
950
|
+
WHERE $memory_id IS NULL OR m.id = $memory_id
|
|
951
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
952
|
+
OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
|
|
953
|
+
WITH m, collect(DISTINCT e.name) AS entity_names,
|
|
954
|
+
collect(DISTINCT p.id) AS person_ids
|
|
955
|
+
RETURN m {.*, entities: entity_names,
|
|
956
|
+
person_id: head(person_ids)} AS memory
|
|
957
|
+
ORDER BY m.created_at DESC
|
|
958
|
+
LIMIT $limit
|
|
959
|
+
""",
|
|
960
|
+
memory_id=memory_scope,
|
|
961
|
+
limit=safe_limit,
|
|
962
|
+
)
|
|
963
|
+
memories: List[Dict[str, Any]] = []
|
|
964
|
+
async for record in result:
|
|
965
|
+
memories.append(dict(record["memory"]))
|
|
966
|
+
return await self._filter_erased_source_memories(memories)
|
|
967
|
+
|
|
968
|
+
def _memory_is_recall_excluded(self, memory: Dict[str, Any]) -> bool:
|
|
969
|
+
if self._source_projection_erased(memory.get("source_uri")):
|
|
970
|
+
return True
|
|
971
|
+
sources = set(getattr(self, "_recall_source_exclusions", ()))
|
|
972
|
+
if str(memory.get("source_uri") or "") in sources:
|
|
973
|
+
return True
|
|
974
|
+
metadata = memory.get("metadata")
|
|
975
|
+
raw_metadata = str(
|
|
976
|
+
memory if metadata is None else metadata).lower()
|
|
977
|
+
return any(
|
|
978
|
+
marker in raw_metadata
|
|
979
|
+
for marker in getattr(self, "_recall_metadata_exclusions", ())
|
|
980
|
+
)
|
|
981
|
+
|
|
982
|
+
async def filter_memory_vector_results(
|
|
983
|
+
self,
|
|
984
|
+
rows: List[Dict[str, Any]],
|
|
985
|
+
) -> List[Dict[str, Any]]:
|
|
986
|
+
"""Apply the graph's hard content policy to raw vector result text.
|
|
987
|
+
|
|
988
|
+
Vector metadata catches current mirrors cheaply. Ambiguous/legacy rows
|
|
989
|
+
are hydrated by ID so their authoritative graph metadata is checked.
|
|
990
|
+
Non-graph vectors (for example image captions) remain available.
|
|
991
|
+
"""
|
|
992
|
+
rows = [row for row in rows if not self._source_projection_erased(
|
|
993
|
+
row.get("source_uri") or (row.get("metadata") or {}).get("source_uri"))]
|
|
994
|
+
if not (
|
|
995
|
+
getattr(self, "_recall_source_exclusions", ())
|
|
996
|
+
or getattr(self, "_recall_metadata_exclusions", ())
|
|
997
|
+
):
|
|
998
|
+
return rows
|
|
999
|
+
|
|
1000
|
+
candidates = [row for row in rows[:512] if isinstance(row, dict)]
|
|
1001
|
+
ids = tuple(dict.fromkeys(
|
|
1002
|
+
str(row.get("id") or "").strip() for row in candidates
|
|
1003
|
+
if str(row.get("id") or "").strip()
|
|
1004
|
+
))
|
|
1005
|
+
graph_rows: Dict[str, Dict[str, Any]] = {}
|
|
1006
|
+
if ids:
|
|
1007
|
+
async with self.driver.session(database=self.database) as session:
|
|
1008
|
+
result = await session.run(
|
|
1009
|
+
"""
|
|
1010
|
+
MATCH (m:Memory) WHERE m.id IN $ids
|
|
1011
|
+
RETURN m {.*} AS memory
|
|
1012
|
+
""",
|
|
1013
|
+
ids=ids,
|
|
1014
|
+
)
|
|
1015
|
+
async for record in result:
|
|
1016
|
+
memory = dict(record["memory"])
|
|
1017
|
+
mid = str(memory.get("id") or "")
|
|
1018
|
+
if mid:
|
|
1019
|
+
graph_rows[mid] = memory
|
|
1020
|
+
|
|
1021
|
+
allowed: List[Dict[str, Any]] = []
|
|
1022
|
+
for row in candidates:
|
|
1023
|
+
mid = str(row.get("id") or "").strip()
|
|
1024
|
+
if not mid:
|
|
1025
|
+
continue
|
|
1026
|
+
vector_metadata = row.get("metadata")
|
|
1027
|
+
vector_view = dict(vector_metadata) if isinstance(
|
|
1028
|
+
vector_metadata, dict) else {"metadata": vector_metadata}
|
|
1029
|
+
if self._memory_is_recall_excluded(vector_view):
|
|
1030
|
+
continue
|
|
1031
|
+
graph_view = graph_rows.get(mid)
|
|
1032
|
+
if graph_view is not None and self._memory_is_recall_excluded(
|
|
1033
|
+
graph_view
|
|
1034
|
+
):
|
|
1035
|
+
continue
|
|
1036
|
+
if graph_view is None:
|
|
1037
|
+
metadata = (
|
|
1038
|
+
vector_metadata
|
|
1039
|
+
if isinstance(vector_metadata, dict) else {}
|
|
1040
|
+
)
|
|
1041
|
+
modality = str(metadata.get("modality") or "").lower()
|
|
1042
|
+
source_uri = str(metadata.get("source_uri") or "").strip()
|
|
1043
|
+
# Missing graph hydration is not authority for ambiguous text.
|
|
1044
|
+
# Preserve explicit non-graph media and text with an identified,
|
|
1045
|
+
# non-excluded source; unknown text fails closed.
|
|
1046
|
+
if modality != "image" and not source_uri:
|
|
1047
|
+
continue
|
|
1048
|
+
allowed.append(row)
|
|
1049
|
+
return allowed
|
|
1050
|
+
|
|
1051
|
+
async def recall_candidates(self, **kwargs):
|
|
1052
|
+
"""Authorized graph candidates without reranking or reinforcement."""
|
|
1053
|
+
return await self.recall(**kwargs, candidates_only=True)
|
|
1054
|
+
|
|
1055
|
+
def record_recall_use(self, memories):
|
|
1056
|
+
"""Reinforce only graph records that actually entered turn context."""
|
|
1057
|
+
for memory in memories:
|
|
1058
|
+
if memory.get("kind", "belief") == "belief" and memory.get("id"):
|
|
1059
|
+
# Annotated packets have a presentation ID, separate from the
|
|
1060
|
+
# graph record whose use should slow subsequent memory decay.
|
|
1061
|
+
memory_id = memory.get("_recall_memory_id") or memory["id"]
|
|
1062
|
+
self._retain_task(asyncio.create_task(
|
|
1063
|
+
self._touch_memory_safe(memory_id)))
|
|
1064
|
+
|
|
1065
|
+
async def recall(
|
|
1066
|
+
self,
|
|
1067
|
+
query: str,
|
|
1068
|
+
limit: int = 10,
|
|
1069
|
+
min_strength: Optional[float] = None,
|
|
1070
|
+
min_confidence: float = 0.1,
|
|
1071
|
+
person_id: Optional[str] = None,
|
|
1072
|
+
exclude_source_uris: Optional[List[str]] = None,
|
|
1073
|
+
*,
|
|
1074
|
+
candidates_only: bool = False,
|
|
1075
|
+
) -> List[Dict[str, Any]]:
|
|
1076
|
+
"""Retrieve memories by semantic similarity with strength decay.
|
|
1077
|
+
|
|
1078
|
+
Uses LanceDB for ANN search, then hydrates full Memory nodes from
|
|
1079
|
+
Neo4j with entity mentions. Falls back to graph-only keyword
|
|
1080
|
+
recall if no vector store or embedding function is configured.
|
|
1081
|
+
|
|
1082
|
+
``min_strength`` is the hard exclusion floor for decayed memories.
|
|
1083
|
+
An explicit caller value always wins; when omitted it comes from
|
|
1084
|
+
COLONY_RECALL_MIN_STRENGTH (default 0.1, the historical floor).
|
|
1085
|
+
The floor stays hard — decayed junk is excluded, not demoted — and
|
|
1086
|
+
is only lowered stepwise as live pruning cleans the graph.
|
|
1087
|
+
|
|
1088
|
+
An explicit ``person_id`` is a hard candidate boundary: only memories
|
|
1089
|
+
linked ``ABOUT`` that exact person may be hydrated or ranked. A scoped
|
|
1090
|
+
miss stays empty and never falls back to global recall. Trusted owner
|
|
1091
|
+
or internal callers that intentionally need global memory must omit
|
|
1092
|
+
``person_id`` after establishing that authority outside this method.
|
|
1093
|
+
|
|
1094
|
+
``exclude_source_uris`` is a bounded hard hydration filter applied
|
|
1095
|
+
before confidence/relevance ranking and reranking. The ANN candidate
|
|
1096
|
+
fetch is widened when this filter is present so excluded mirrors do
|
|
1097
|
+
not ordinarily starve the requested result window.
|
|
1098
|
+
|
|
1099
|
+
``candidates_only`` skips final reranking and recall-strength updates;
|
|
1100
|
+
the caller must select authorized candidates and record actual use.
|
|
1101
|
+
|
|
1102
|
+
Returns:
|
|
1103
|
+
A list of memory dicts, each annotated with ``entities`` and
|
|
1104
|
+
sorted by relevance descending.
|
|
1105
|
+
"""
|
|
1106
|
+
if min_strength is None:
|
|
1107
|
+
try:
|
|
1108
|
+
min_strength = float(os.environ.get(
|
|
1109
|
+
"COLONY_RECALL_MIN_STRENGTH", "0.1"))
|
|
1110
|
+
except (TypeError, ValueError):
|
|
1111
|
+
min_strength = 0.1
|
|
1112
|
+
# Strength-blended ranking (COLONY_RECALL_STRENGTH_RANKING, default
|
|
1113
|
+
# off = legacy vector_score * effective_confidence). When on, decayed
|
|
1114
|
+
# memories are demoted smoothly ABOVE the hard floor:
|
|
1115
|
+
# relevance = vector_score * effective_confidence * (0.5 + 0.5*strength)
|
|
1116
|
+
strength_ranking = os.environ.get(
|
|
1117
|
+
"COLONY_RECALL_STRENGTH_RANKING", "off").strip().lower() in (
|
|
1118
|
+
"on", "1", "true", "yes")
|
|
1119
|
+
person_scope = str(person_id).strip() if person_id else None
|
|
1120
|
+
excluded_sources = self._bounded_source_uris((
|
|
1121
|
+
*getattr(self, "_recall_source_exclusions", ()),
|
|
1122
|
+
*(exclude_source_uris or ()),
|
|
1123
|
+
))
|
|
1124
|
+
excluded_metadata_markers = self._bounded_source_uris(
|
|
1125
|
+
getattr(self, "_recall_metadata_exclusions", ()))
|
|
1126
|
+
hybrid = os.environ.get("COLONY_RECALL_HYBRID", "off").strip().lower() in (
|
|
1127
|
+
"on", "1", "true", "yes")
|
|
1128
|
+
# Vector search path: embed query (with instruction) → LanceDB ANN → Neo4j hydration
|
|
1129
|
+
if self._vector_store is not None and self._embed_fn is not None:
|
|
1130
|
+
try:
|
|
1131
|
+
embedding = await self._embed_query(query)
|
|
1132
|
+
from apsimo.vector.collections import Collection
|
|
1133
|
+
# Oversample the ANN fetch (COLONY_RECALL_OVERSAMPLE, default 1
|
|
1134
|
+
# = legacy exact-limit fetch) so the post-hydration filters
|
|
1135
|
+
# (strength floor, terminal epistemic states, low confidence)
|
|
1136
|
+
# can't silently shrink the result set below the requested
|
|
1137
|
+
# limit. Oversampled fetches are capped at 100 candidates and
|
|
1138
|
+
# trimmed back to `limit` after filtering.
|
|
1139
|
+
try:
|
|
1140
|
+
oversample = int(os.environ.get(
|
|
1141
|
+
"COLONY_RECALL_OVERSAMPLE", "1"))
|
|
1142
|
+
except (TypeError, ValueError):
|
|
1143
|
+
oversample = 1
|
|
1144
|
+
fetch_limit = (limit if oversample <= 1
|
|
1145
|
+
else min(limit * oversample, max(100, limit)))
|
|
1146
|
+
# Existing vector rows predate structured recipient metadata,
|
|
1147
|
+
# so the authoritative scope check happens during Neo4j
|
|
1148
|
+
# hydration. Widen scoped ANN candidate generation to reduce
|
|
1149
|
+
# false-empty results without ever relaxing that graph gate.
|
|
1150
|
+
if person_scope:
|
|
1151
|
+
try:
|
|
1152
|
+
scope_oversample = max(1, int(os.environ.get(
|
|
1153
|
+
"COLONY_RECALL_SCOPE_OVERSAMPLE", "20")))
|
|
1154
|
+
except (TypeError, ValueError):
|
|
1155
|
+
scope_oversample = 20
|
|
1156
|
+
fetch_limit = max(
|
|
1157
|
+
fetch_limit,
|
|
1158
|
+
min(limit * scope_oversample, max(200, limit)),
|
|
1159
|
+
)
|
|
1160
|
+
elif excluded_sources or excluded_metadata_markers:
|
|
1161
|
+
fetch_limit = max(
|
|
1162
|
+
fetch_limit, min(limit * 20, max(200, limit)))
|
|
1163
|
+
results = await self._vector_store.search(
|
|
1164
|
+
collection=Collection.MEMORIES,
|
|
1165
|
+
query_vector=embedding,
|
|
1166
|
+
limit=fetch_limit,
|
|
1167
|
+
# metadata is stored as a JSON string (pa.utf8()); LanceDB's
|
|
1168
|
+
# filter dialect does not support json_extract on utf8 columns.
|
|
1169
|
+
# Strength filtering is applied post-hydration from Neo4j below.
|
|
1170
|
+
filter=None,
|
|
1171
|
+
)
|
|
1172
|
+
# Adaptive relevance floor (meta-learning knob): vector hits
|
|
1173
|
+
# scoring below recall.min_relevance are dropped before
|
|
1174
|
+
# hydration. Default 0.0 = no filter; store caps at 0.5 so a
|
|
1175
|
+
# self-adjustment can never starve retrieval.
|
|
1176
|
+
min_relevance = 0.0
|
|
1177
|
+
params = getattr(self, "_adaptive_params", None)
|
|
1178
|
+
if params is not None:
|
|
1179
|
+
try:
|
|
1180
|
+
from apsimo.self_model.params import (
|
|
1181
|
+
PARAM_RECALL_MIN_RELEVANCE,
|
|
1182
|
+
)
|
|
1183
|
+
min_relevance = float(params.get(
|
|
1184
|
+
PARAM_RECALL_MIN_RELEVANCE, default=0.0))
|
|
1185
|
+
except Exception:
|
|
1186
|
+
min_relevance = 0.0
|
|
1187
|
+
if min_relevance > 0.0:
|
|
1188
|
+
results = [r for r in results if r.score >= min_relevance]
|
|
1189
|
+
memories = []
|
|
1190
|
+
if results:
|
|
1191
|
+
memory_ids = [r.id for r in results]
|
|
1192
|
+
score_map = {r.id: r.score for r in results}
|
|
1193
|
+
|
|
1194
|
+
# Hydrate from Neo4j
|
|
1195
|
+
async with self.driver.session(database=self.database) as session:
|
|
1196
|
+
if person_scope:
|
|
1197
|
+
result = await session.run(
|
|
1198
|
+
"""
|
|
1199
|
+
MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
|
|
1200
|
+
WHERE m.id IN $ids
|
|
1201
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
1202
|
+
coalesce(m.source_uri, "") IN
|
|
1203
|
+
$exclude_source_uris))
|
|
1204
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
1205
|
+
NONE(marker IN $exclude_metadata_markers
|
|
1206
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
1207
|
+
CONTAINS marker))
|
|
1208
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1209
|
+
WITH m, collect(e.name) AS entity_names
|
|
1210
|
+
RETURN m {.*, entities: entity_names,
|
|
1211
|
+
person_id: $person_id} AS memory
|
|
1212
|
+
""",
|
|
1213
|
+
ids=memory_ids,
|
|
1214
|
+
person_id=person_scope,
|
|
1215
|
+
exclude_source_uris=excluded_sources,
|
|
1216
|
+
exclude_metadata_markers=(
|
|
1217
|
+
excluded_metadata_markers),
|
|
1218
|
+
)
|
|
1219
|
+
else:
|
|
1220
|
+
result = await session.run(
|
|
1221
|
+
"""
|
|
1222
|
+
MATCH (m:Memory) WHERE m.id IN $ids
|
|
1223
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
1224
|
+
coalesce(m.source_uri, "") IN
|
|
1225
|
+
$exclude_source_uris))
|
|
1226
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
1227
|
+
NONE(marker IN $exclude_metadata_markers
|
|
1228
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
1229
|
+
CONTAINS marker))
|
|
1230
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1231
|
+
WITH m, collect(e.name) AS entity_names
|
|
1232
|
+
RETURN m {.*, entities: entity_names} AS memory
|
|
1233
|
+
""",
|
|
1234
|
+
ids=memory_ids,
|
|
1235
|
+
exclude_source_uris=excluded_sources,
|
|
1236
|
+
exclude_metadata_markers=(
|
|
1237
|
+
excluded_metadata_markers),
|
|
1238
|
+
)
|
|
1239
|
+
memories = []
|
|
1240
|
+
async for record in result:
|
|
1241
|
+
mem = record["memory"]
|
|
1242
|
+
mid = mem.get("id", "")
|
|
1243
|
+
vector_score = score_map.get(mid, 0.0)
|
|
1244
|
+
strength = float(mem.get("strength", 1.0))
|
|
1245
|
+
if strength < min_strength:
|
|
1246
|
+
continue
|
|
1247
|
+
# Filter terminal epistemic states and low confidence
|
|
1248
|
+
epistemic_state = mem.get("epistemic_state", "inferred")
|
|
1249
|
+
if mem.get("superseded_by") or epistemic_state in ("stale", "superseded", "deprecated", "archived"):
|
|
1250
|
+
continue
|
|
1251
|
+
effective_confidence = float(mem.get("effective_confidence", mem.get("strength", 1.0)))
|
|
1252
|
+
if effective_confidence < min_confidence:
|
|
1253
|
+
continue
|
|
1254
|
+
if strength_ranking:
|
|
1255
|
+
mem["relevance"] = (vector_score
|
|
1256
|
+
* effective_confidence
|
|
1257
|
+
* (0.5 + 0.5 * strength))
|
|
1258
|
+
else:
|
|
1259
|
+
mem["relevance"] = vector_score * effective_confidence
|
|
1260
|
+
memories.append(mem)
|
|
1261
|
+
|
|
1262
|
+
if hybrid:
|
|
1263
|
+
from .recall import fuse_candidates
|
|
1264
|
+
candidate_limit = min(max(limit * 5, limit), max(100, limit))
|
|
1265
|
+
lexical = await self._recall_lexical(
|
|
1266
|
+
query, candidate_limit, min_strength, min_confidence,
|
|
1267
|
+
person_scope, excluded_sources, excluded_metadata_markers)
|
|
1268
|
+
memories = fuse_candidates(
|
|
1269
|
+
memories, lexical, limit=candidate_limit,
|
|
1270
|
+
strength_ranking=strength_ranking)
|
|
1271
|
+
if results or memories:
|
|
1272
|
+
memories = await self._filter_erased_source_memories(memories)
|
|
1273
|
+
if candidates_only:
|
|
1274
|
+
return sorted(memories, key=lambda m: m.get("relevance", 0), reverse=True)[:limit]
|
|
1275
|
+
memories = await self._maybe_rerank(
|
|
1276
|
+
query, memories, limit,
|
|
1277
|
+
strength_ranking=strength_ranking)
|
|
1278
|
+
memories = await self._filter_erased_source_memories(memories)
|
|
1279
|
+
memories.sort(key=lambda m: m.get("relevance", 0), reverse=True)
|
|
1280
|
+
memories = memories[:limit]
|
|
1281
|
+
# Fire-and-forget touch_memory for each recalled result
|
|
1282
|
+
for mem in memories:
|
|
1283
|
+
mid = mem.get("id")
|
|
1284
|
+
if mid:
|
|
1285
|
+
self._retain_task(asyncio.create_task(self._touch_memory_safe(mid)))
|
|
1286
|
+
return memories
|
|
1287
|
+
except Exception as exc:
|
|
1288
|
+
logger.warning("Vector recall failed, falling back to graph-only: %s", exc)
|
|
1289
|
+
|
|
1290
|
+
# Fallback: graph-only keyword/entity recall
|
|
1291
|
+
if hybrid:
|
|
1292
|
+
memories = await self._recall_lexical(
|
|
1293
|
+
query, min(max(limit * 5, limit), max(100, limit)),
|
|
1294
|
+
min_strength, min_confidence, person_scope,
|
|
1295
|
+
excluded_sources, excluded_metadata_markers)
|
|
1296
|
+
if memories:
|
|
1297
|
+
memories = await self._filter_erased_source_memories(memories)
|
|
1298
|
+
if candidates_only:
|
|
1299
|
+
return memories[:limit]
|
|
1300
|
+
memories = await self._maybe_rerank(
|
|
1301
|
+
query, memories, limit, strength_ranking=strength_ranking)
|
|
1302
|
+
memories = await self._filter_erased_source_memories(memories)
|
|
1303
|
+
memories.sort(key=lambda m: m.get("relevance", 0), reverse=True)
|
|
1304
|
+
memories = memories[:limit]
|
|
1305
|
+
for mem in memories:
|
|
1306
|
+
if mem.get("id"):
|
|
1307
|
+
self._retain_task(asyncio.create_task(self._touch_memory_safe(mem["id"])))
|
|
1308
|
+
return memories
|
|
1309
|
+
async with self.driver.session(database=self.database) as session:
|
|
1310
|
+
if person_scope:
|
|
1311
|
+
result = await session.run(
|
|
1312
|
+
"""
|
|
1313
|
+
MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
|
|
1314
|
+
WHERE m.strength >= $min_strength
|
|
1315
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
1316
|
+
coalesce(m.source_uri, "") IN
|
|
1317
|
+
$exclude_source_uris))
|
|
1318
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
1319
|
+
NONE(marker IN $exclude_metadata_markers
|
|
1320
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
1321
|
+
CONTAINS marker))
|
|
1322
|
+
AND toLower(m.content) CONTAINS toLower($search_text)
|
|
1323
|
+
AND m.superseded_by IS NULL
|
|
1324
|
+
AND NOT m.epistemic_state IN ["stale", "superseded", "deprecated", "archived"]
|
|
1325
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1326
|
+
WITH m, collect(e.name) AS entity_names
|
|
1327
|
+
RETURN m {.*, entities: entity_names,
|
|
1328
|
+
person_id: $person_id} AS memory,
|
|
1329
|
+
m.effective_confidence AS relevance
|
|
1330
|
+
ORDER BY relevance DESC
|
|
1331
|
+
LIMIT $limit
|
|
1332
|
+
""",
|
|
1333
|
+
search_text=query,
|
|
1334
|
+
limit=limit,
|
|
1335
|
+
min_strength=min_strength,
|
|
1336
|
+
person_id=person_scope,
|
|
1337
|
+
exclude_source_uris=excluded_sources,
|
|
1338
|
+
exclude_metadata_markers=excluded_metadata_markers,
|
|
1339
|
+
)
|
|
1340
|
+
else:
|
|
1341
|
+
result = await session.run(
|
|
1342
|
+
"""
|
|
1343
|
+
MATCH (m:Memory)
|
|
1344
|
+
WHERE m.strength >= $min_strength
|
|
1345
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
1346
|
+
coalesce(m.source_uri, "") IN
|
|
1347
|
+
$exclude_source_uris))
|
|
1348
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
1349
|
+
NONE(marker IN $exclude_metadata_markers
|
|
1350
|
+
WHERE toLower(coalesce(m.metadata, ""))
|
|
1351
|
+
CONTAINS marker))
|
|
1352
|
+
AND toLower(m.content) CONTAINS toLower($search_text)
|
|
1353
|
+
AND m.superseded_by IS NULL
|
|
1354
|
+
AND NOT m.epistemic_state IN ["stale", "superseded", "deprecated", "archived"]
|
|
1355
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1356
|
+
WITH m, collect(e.name) AS entity_names
|
|
1357
|
+
RETURN m {.*, entities: entity_names} AS memory,
|
|
1358
|
+
m.effective_confidence AS relevance
|
|
1359
|
+
ORDER BY relevance DESC
|
|
1360
|
+
LIMIT $limit
|
|
1361
|
+
""",
|
|
1362
|
+
search_text=query,
|
|
1363
|
+
limit=limit,
|
|
1364
|
+
min_strength=min_strength,
|
|
1365
|
+
exclude_source_uris=excluded_sources,
|
|
1366
|
+
exclude_metadata_markers=excluded_metadata_markers,
|
|
1367
|
+
)
|
|
1368
|
+
memories = []
|
|
1369
|
+
async for record in result:
|
|
1370
|
+
mem = record["memory"]
|
|
1371
|
+
mem["relevance"] = record.get("relevance", mem.get("strength", 0.5))
|
|
1372
|
+
effective_confidence = float(mem.get("effective_confidence", mem.get("strength", 1.0)))
|
|
1373
|
+
if effective_confidence < min_confidence:
|
|
1374
|
+
continue
|
|
1375
|
+
memories.append(mem)
|
|
1376
|
+
memories = await self._filter_erased_source_memories(memories)
|
|
1377
|
+
if candidates_only:
|
|
1378
|
+
return memories[:limit]
|
|
1379
|
+
# Fire-and-forget touch_memory for each recalled result
|
|
1380
|
+
for mem in memories:
|
|
1381
|
+
mid = mem.get("id")
|
|
1382
|
+
if mid:
|
|
1383
|
+
self._retain_task(asyncio.create_task(self._touch_memory_safe(mid)))
|
|
1384
|
+
return memories
|
|
1385
|
+
|
|
1386
|
+
async def _recall_lexical(
|
|
1387
|
+
self, query, limit, min_strength, min_confidence,
|
|
1388
|
+
person_scope, excluded_sources, excluded_metadata_markers,
|
|
1389
|
+
):
|
|
1390
|
+
"""Native source-text candidates with the same authoritative boundary.
|
|
1391
|
+
|
|
1392
|
+
The full-text index is maintained with graph writes, so an embedding
|
|
1393
|
+
backlog does not hide newly stored facts. Filtering happens before
|
|
1394
|
+
content reaches Python/reranking. A missing/populating index degrades
|
|
1395
|
+
to the existing vector/keyword path.
|
|
1396
|
+
"""
|
|
1397
|
+
from .recall import FULLTEXT_INDEX, lexical_query
|
|
1398
|
+
search_text = lexical_query(query)
|
|
1399
|
+
if not search_text:
|
|
1400
|
+
return []
|
|
1401
|
+
|
|
1402
|
+
async def search():
|
|
1403
|
+
async with self.driver.session(database=self.database) as session:
|
|
1404
|
+
result = await session.run(
|
|
1405
|
+
"""
|
|
1406
|
+
CALL db.index.fulltext.queryNodes($index_name, $search_text)
|
|
1407
|
+
YIELD node AS m, score
|
|
1408
|
+
WHERE m:Memory
|
|
1409
|
+
AND ($person_id IS NULL OR EXISTS {
|
|
1410
|
+
MATCH (m)-[:ABOUT]->(:Person {id: $person_id})
|
|
1411
|
+
})
|
|
1412
|
+
AND coalesce(m.strength, 1.0) >= $min_strength
|
|
1413
|
+
AND coalesce(m.effective_confidence, m.strength, 1.0) >= $min_confidence
|
|
1414
|
+
AND m.superseded_by IS NULL
|
|
1415
|
+
AND NOT coalesce(m.epistemic_state, "inferred") IN
|
|
1416
|
+
["stale", "superseded", "deprecated", "archived"]
|
|
1417
|
+
AND (size($exclude_source_uris) = 0 OR NOT (
|
|
1418
|
+
coalesce(m.source_uri, "") IN $exclude_source_uris))
|
|
1419
|
+
AND (size($exclude_metadata_markers) = 0 OR
|
|
1420
|
+
NONE(marker IN $exclude_metadata_markers
|
|
1421
|
+
WHERE toLower(coalesce(m.metadata, "")) CONTAINS marker))
|
|
1422
|
+
WITH m, score
|
|
1423
|
+
ORDER BY score DESC
|
|
1424
|
+
LIMIT $limit
|
|
1425
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1426
|
+
WITH m, score, collect(DISTINCT e.name) AS entity_names
|
|
1427
|
+
RETURN m {.*, entities: entity_names} AS memory,
|
|
1428
|
+
score AS lexical_score
|
|
1429
|
+
ORDER BY lexical_score DESC
|
|
1430
|
+
""",
|
|
1431
|
+
index_name=FULLTEXT_INDEX, search_text=search_text,
|
|
1432
|
+
limit=limit, person_id=person_scope,
|
|
1433
|
+
min_strength=min_strength, min_confidence=min_confidence,
|
|
1434
|
+
exclude_source_uris=excluded_sources,
|
|
1435
|
+
exclude_metadata_markers=excluded_metadata_markers,
|
|
1436
|
+
)
|
|
1437
|
+
memories = []
|
|
1438
|
+
async for record in result:
|
|
1439
|
+
mem = dict(record["memory"])
|
|
1440
|
+
mem["relevance"] = float(record["lexical_score"])
|
|
1441
|
+
mem["retrieval_method"] = "lexical"
|
|
1442
|
+
memories.append(mem)
|
|
1443
|
+
return memories
|
|
1444
|
+
try:
|
|
1445
|
+
return await asyncio.wait_for(search(), timeout=1.2)
|
|
1446
|
+
except Exception as exc:
|
|
1447
|
+
logger.warning("Lexical recall unavailable, using existing recall: %s", type(exc).__name__)
|
|
1448
|
+
return []
|
|
1449
|
+
|
|
1450
|
+
def _retain_task(self, task) -> None:
|
|
1451
|
+
"""Keep a strong ref to a fire-and-forget task until it finishes; the
|
|
1452
|
+
loop only weak-refs it, so an unreferenced recall-touch task can be
|
|
1453
|
+
GC-cancelled mid neo4j session ("Task was destroyed but it is
|
|
1454
|
+
pending" + an abandoned session)."""
|
|
1455
|
+
if not hasattr(self, "_bg_tasks"):
|
|
1456
|
+
self._bg_tasks = set()
|
|
1457
|
+
self._bg_tasks.add(task)
|
|
1458
|
+
task.add_done_callback(self._bg_tasks.discard)
|
|
1459
|
+
|
|
1460
|
+
async def _touch_memory_safe(self, memory_id: str) -> None:
|
|
1461
|
+
"""Touch a memory, logging but not raising on failure."""
|
|
1462
|
+
try:
|
|
1463
|
+
await self.touch_memory(memory_id)
|
|
1464
|
+
except Exception as exc:
|
|
1465
|
+
logger.debug("touch_memory failed for %s: %s", memory_id, exc)
|
|
1466
|
+
|
|
1467
|
+
async def _maybe_rerank(
|
|
1468
|
+
self, query, memories, limit, *, strength_ranking=False,
|
|
1469
|
+
):
|
|
1470
|
+
# The same selector is used for mixed source/belief turn context.
|
|
1471
|
+
from .selection import RecallSelector
|
|
1472
|
+
selector = getattr(self, "_recall_selector", None)
|
|
1473
|
+
if selector is None:
|
|
1474
|
+
selector = self._recall_selector = RecallSelector(
|
|
1475
|
+
getattr(self, "_rerank_fn", None),
|
|
1476
|
+
calibration_metadata=getattr(self, "_rerank_calibration_metadata", None),
|
|
1477
|
+
logger=logger)
|
|
1478
|
+
return await selector.rerank(
|
|
1479
|
+
query, memories, limit, strength_ranking=strength_ranking)
|
|
1480
|
+
|
|
1481
|
+
# ------------------------------------------------------------------
|
|
1482
|
+
# Decay & pruning
|
|
1483
|
+
# ------------------------------------------------------------------
|
|
1484
|
+
|
|
1485
|
+
@staticmethod
|
|
1486
|
+
def _compute_decay_factor(
|
|
1487
|
+
importance: float,
|
|
1488
|
+
days_elapsed: float,
|
|
1489
|
+
recalls: int,
|
|
1490
|
+
half_life_days: float,
|
|
1491
|
+
memory_type: str = "episodic",
|
|
1492
|
+
semantic_half_life_days: Optional[float] = None,
|
|
1493
|
+
) -> float:
|
|
1494
|
+
"""Compute Ebbinghaus decay for a memory.
|
|
1495
|
+
|
|
1496
|
+
Formula: strength = importance * e^(-lambda * days) * (1 + recalls * 0.2)
|
|
1497
|
+
|
|
1498
|
+
Where lambda = ln(2) / half_life_days. Identity memories never decay;
|
|
1499
|
+
procedural memories decay at half the normal rate; fact/semantic
|
|
1500
|
+
memories use their own half-life when one is given (defaults to the
|
|
1501
|
+
episodic value, i.e. no behavior change). Result is capped at 1.0.
|
|
1502
|
+
|
|
1503
|
+
Args:
|
|
1504
|
+
importance: Initial importance value (0-1)
|
|
1505
|
+
days_elapsed: Days since last access
|
|
1506
|
+
recalls: Number of times the memory has been recalled
|
|
1507
|
+
half_life_days: Days for strength to halve (default 7)
|
|
1508
|
+
memory_type: One of "identity", "procedural", "episodic",
|
|
1509
|
+
"semantic", "fact"
|
|
1510
|
+
semantic_half_life_days: Half-life for fact/semantic memories
|
|
1511
|
+
(None = same as half_life_days)
|
|
1512
|
+
|
|
1513
|
+
Returns:
|
|
1514
|
+
New strength value in [0, 1].
|
|
1515
|
+
"""
|
|
1516
|
+
if memory_type == "identity":
|
|
1517
|
+
return float(importance)
|
|
1518
|
+
|
|
1519
|
+
lambda_base = math.log(2) / max(half_life_days, 0.001)
|
|
1520
|
+
if memory_type == "procedural":
|
|
1521
|
+
# Procedural memories decay at half the normal rate
|
|
1522
|
+
lambda_val = lambda_base / 2
|
|
1523
|
+
elif memory_type in ("fact", "semantic"):
|
|
1524
|
+
sem_half_life = (semantic_half_life_days
|
|
1525
|
+
if semantic_half_life_days is not None
|
|
1526
|
+
else half_life_days)
|
|
1527
|
+
lambda_val = math.log(2) / max(sem_half_life, 0.001)
|
|
1528
|
+
else:
|
|
1529
|
+
lambda_val = lambda_base
|
|
1530
|
+
|
|
1531
|
+
strength = importance * math.exp(-lambda_val * max(days_elapsed, 0)) * (1.0 + recalls * 0.2)
|
|
1532
|
+
return min(1.0, max(0.0, strength))
|
|
1533
|
+
|
|
1534
|
+
async def decay_memories(
|
|
1535
|
+
self,
|
|
1536
|
+
half_life_days: Optional[float] = None,
|
|
1537
|
+
semantic_half_life_days: Optional[float] = None,
|
|
1538
|
+
) -> None:
|
|
1539
|
+
"""Apply Ebbinghaus forgetting curve to all non-identity, non-protected memories.
|
|
1540
|
+
|
|
1541
|
+
Formula: strength = importance * e^(-lambda * days) * (1 + recalls * 0.2)
|
|
1542
|
+
|
|
1543
|
+
Where lambda = ln(2) / half_life_days.
|
|
1544
|
+
- Identity memories are skipped (never decay).
|
|
1545
|
+
- Protected memories are skipped.
|
|
1546
|
+
- Procedural memories use lambda / 2 (half rate).
|
|
1547
|
+
- Fact/semantic memories use their own half-life (defaults to the
|
|
1548
|
+
episodic value — distilled knowledge can be made to outlive the
|
|
1549
|
+
episodes it came from by raising COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS).
|
|
1550
|
+
- Result is capped at 1.0.
|
|
1551
|
+
|
|
1552
|
+
Half-lives resolve, in order: explicit argument, environment
|
|
1553
|
+
(COLONY_DECAY_HALF_LIFE_DAYS / COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS),
|
|
1554
|
+
then the historical default of 7 days. Strength is RECOMPUTED from
|
|
1555
|
+
importance on every pass (not compounded), so raising a half-life
|
|
1556
|
+
retroactively resurrects previously-decayed strength — which is why
|
|
1557
|
+
half-life tuning must land before pruning goes live, never after.
|
|
1558
|
+
|
|
1559
|
+
This pass is the single writer for memory decay; nothing else may
|
|
1560
|
+
call it as a side effect (see StrategyAdjuster._decay_signals,
|
|
1561
|
+
retired for exactly that reason).
|
|
1562
|
+
"""
|
|
1563
|
+
if half_life_days is None:
|
|
1564
|
+
try:
|
|
1565
|
+
half_life_days = float(os.environ.get(
|
|
1566
|
+
"COLONY_DECAY_HALF_LIFE_DAYS", "7"))
|
|
1567
|
+
except (TypeError, ValueError):
|
|
1568
|
+
half_life_days = 7.0
|
|
1569
|
+
if semantic_half_life_days is None:
|
|
1570
|
+
_sem_env = os.environ.get("COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS", "")
|
|
1571
|
+
try:
|
|
1572
|
+
semantic_half_life_days = (
|
|
1573
|
+
float(_sem_env) if _sem_env.strip() else half_life_days)
|
|
1574
|
+
except (TypeError, ValueError):
|
|
1575
|
+
semantic_half_life_days = half_life_days
|
|
1576
|
+
|
|
1577
|
+
lambda_normal = math.log(2) / max(half_life_days, 0.001)
|
|
1578
|
+
lambda_procedural = lambda_normal / 2
|
|
1579
|
+
lambda_semantic = math.log(2) / max(semantic_half_life_days, 0.001)
|
|
1580
|
+
|
|
1581
|
+
async with self.driver.session(database=self.database) as session:
|
|
1582
|
+
# First pass: update strength
|
|
1583
|
+
await session.run(
|
|
1584
|
+
"""
|
|
1585
|
+
MATCH (m:Memory)
|
|
1586
|
+
WHERE m.type <> 'identity' AND coalesce(m.protected, false) = false
|
|
1587
|
+
WITH m,
|
|
1588
|
+
toFloat(duration.inDays(coalesce(m.accessed_at, m.created_at, datetime()), datetime()).days) AS days_since,
|
|
1589
|
+
CASE WHEN m.type = 'procedural'
|
|
1590
|
+
THEN $lambda_proc
|
|
1591
|
+
WHEN m.type IN ['fact', 'semantic']
|
|
1592
|
+
THEN $lambda_sem
|
|
1593
|
+
ELSE $lambda_norm
|
|
1594
|
+
END AS lam
|
|
1595
|
+
WITH m,
|
|
1596
|
+
coalesce(m.importance, 1.0) *
|
|
1597
|
+
exp(-lam * days_since) *
|
|
1598
|
+
(1.0 + coalesce(m.recalls, 0) * 0.2) AS new_strength
|
|
1599
|
+
SET m.strength = CASE
|
|
1600
|
+
WHEN new_strength > 1.0 THEN 1.0
|
|
1601
|
+
WHEN new_strength < 0.0 THEN 0.0
|
|
1602
|
+
ELSE new_strength
|
|
1603
|
+
END
|
|
1604
|
+
""",
|
|
1605
|
+
lambda_norm=lambda_normal,
|
|
1606
|
+
lambda_proc=lambda_procedural,
|
|
1607
|
+
lambda_sem=lambda_semantic,
|
|
1608
|
+
)
|
|
1609
|
+
|
|
1610
|
+
# Second pass: update effective_confidence in batches
|
|
1611
|
+
await self._update_effective_confidence_batch()
|
|
1612
|
+
|
|
1613
|
+
async def prune_weak_memories(
|
|
1614
|
+
self,
|
|
1615
|
+
threshold: float = 0.05,
|
|
1616
|
+
*,
|
|
1617
|
+
dry_run: bool = False,
|
|
1618
|
+
max_delete: int = 500,
|
|
1619
|
+
) -> Dict[str, Any]:
|
|
1620
|
+
"""Delete memories whose strength has decayed below *threshold*.
|
|
1621
|
+
|
|
1622
|
+
Only targets memories in ``inferred``, ``observed``, or ``stale``
|
|
1623
|
+
epistemic states. Skips protected memories, ``corroborated``,
|
|
1624
|
+
``verified``, and fully terminal states (``superseded``,
|
|
1625
|
+
``deprecated``, ``archived``).
|
|
1626
|
+
|
|
1627
|
+
A deleted memory's vector-store entry is removed too (same coupling
|
|
1628
|
+
as :meth:`archive_memories`), so pruning does not accumulate orphan
|
|
1629
|
+
vectors that keep matching in ANN search. Deletion is capped at
|
|
1630
|
+
*max_delete* per call (weakest first); ``dry_run=True`` only counts.
|
|
1631
|
+
|
|
1632
|
+
Fails closed on Neo4j errors: exceptions propagate to the caller
|
|
1633
|
+
and nothing further is deleted; a vector is only removed after its
|
|
1634
|
+
graph node is gone.
|
|
1635
|
+
|
|
1636
|
+
Returns:
|
|
1637
|
+
Dict with ``matched`` (total below threshold), ``deleted``,
|
|
1638
|
+
``dry_run``, and the capped candidate ``ids``.
|
|
1639
|
+
"""
|
|
1640
|
+
async with self.driver.session(database=self.database) as session:
|
|
1641
|
+
result = await session.run(
|
|
1642
|
+
"""
|
|
1643
|
+
MATCH (m:Memory)
|
|
1644
|
+
WHERE m.strength < $threshold
|
|
1645
|
+
AND coalesce(m.protected, false) = false
|
|
1646
|
+
AND m.epistemic_state IN ["inferred", "observed", "stale"]
|
|
1647
|
+
WITH m ORDER BY m.strength ASC
|
|
1648
|
+
RETURN collect(m.id)[0..$max_delete] AS ids,
|
|
1649
|
+
count(m) AS matched
|
|
1650
|
+
""",
|
|
1651
|
+
threshold=threshold,
|
|
1652
|
+
max_delete=max_delete,
|
|
1653
|
+
)
|
|
1654
|
+
record = await result.single()
|
|
1655
|
+
ids = list(record["ids"]) if record else []
|
|
1656
|
+
matched = int(record["matched"]) if record else 0
|
|
1657
|
+
|
|
1658
|
+
if dry_run:
|
|
1659
|
+
return {"matched": matched, "deleted": 0, "dry_run": True,
|
|
1660
|
+
"ids": ids}
|
|
1661
|
+
|
|
1662
|
+
deleted = 0
|
|
1663
|
+
for memory_id in ids:
|
|
1664
|
+
async with self.driver.session(database=self.database) as session:
|
|
1665
|
+
await session.run(
|
|
1666
|
+
"MATCH (m:Memory {id: $memory_id}) DETACH DELETE m",
|
|
1667
|
+
memory_id=memory_id,
|
|
1668
|
+
)
|
|
1669
|
+
deleted += 1
|
|
1670
|
+
# Vector removal only after the graph node is gone; a failure
|
|
1671
|
+
# here leaves an orphan vector (swept later), never a memory
|
|
1672
|
+
# that recalls without a backing node.
|
|
1673
|
+
if self._vector_store is not None:
|
|
1674
|
+
try:
|
|
1675
|
+
from apsimo.vector.collections import Collection
|
|
1676
|
+
await self._vector_store.delete(
|
|
1677
|
+
collection=Collection.MEMORIES,
|
|
1678
|
+
id=memory_id,
|
|
1679
|
+
)
|
|
1680
|
+
except Exception as exc:
|
|
1681
|
+
logger.debug(
|
|
1682
|
+
"Failed to remove pruned memory %s from vector "
|
|
1683
|
+
"store: %s", memory_id, exc)
|
|
1684
|
+
return {"matched": matched, "deleted": deleted, "dry_run": False,
|
|
1685
|
+
"ids": ids}
|
|
1686
|
+
|
|
1687
|
+
async def vacuum_orphan_vectors(
|
|
1688
|
+
self,
|
|
1689
|
+
*,
|
|
1690
|
+
dry_run: bool = False,
|
|
1691
|
+
max_delete: Optional[int] = None,
|
|
1692
|
+
batch_size: int = 200,
|
|
1693
|
+
batch_sleep_secs: float = 0.05,
|
|
1694
|
+
) -> Dict[str, Any]:
|
|
1695
|
+
"""Delete MEMORIES-collection vectors whose graph node no longer exists.
|
|
1696
|
+
|
|
1697
|
+
Orphan vectors (a node deleted without its vector — pre-coupling
|
|
1698
|
+
prunes, or a vector removal that failed after node deletion) keep
|
|
1699
|
+
matching in ANN search and then silently vanish at hydration,
|
|
1700
|
+
stealing recall slots from real memories.
|
|
1701
|
+
|
|
1702
|
+
Fails closed: the graph-id read is NOT exception-wrapped — a Neo4j
|
|
1703
|
+
failure aborts the vacuum before any deletion, because an empty or
|
|
1704
|
+
partial graph-id set would classify every vector as an orphan and
|
|
1705
|
+
wipe the store. Deletion runs in batches of *batch_size* with
|
|
1706
|
+
*batch_sleep_secs* between batches; *max_delete* bounds one run.
|
|
1707
|
+
|
|
1708
|
+
Returns:
|
|
1709
|
+
Dict with ``vectors`` (total scanned), ``orphans`` (total found),
|
|
1710
|
+
``deleted``, ``dry_run``, and a capped ``ids`` sample.
|
|
1711
|
+
"""
|
|
1712
|
+
if self._vector_store is None:
|
|
1713
|
+
return {"available": False, "vectors": 0, "orphans": 0,
|
|
1714
|
+
"deleted": 0, "dry_run": dry_run, "ids": []}
|
|
1715
|
+
|
|
1716
|
+
from apsimo.vector.collections import Collection
|
|
1717
|
+
vector_ids = await self._vector_store.list_ids(Collection.MEMORIES)
|
|
1718
|
+
|
|
1719
|
+
graph_ids: set = set()
|
|
1720
|
+
async with self.driver.session(database=self.database) as session:
|
|
1721
|
+
result = await session.run("MATCH (m:Memory) RETURN m.id AS id")
|
|
1722
|
+
async for record in result:
|
|
1723
|
+
if record["id"]:
|
|
1724
|
+
graph_ids.add(record["id"])
|
|
1725
|
+
|
|
1726
|
+
orphans = [vid for vid in vector_ids if vid not in graph_ids]
|
|
1727
|
+
total_orphans = len(orphans)
|
|
1728
|
+
if max_delete is not None:
|
|
1729
|
+
orphans = orphans[:max_delete]
|
|
1730
|
+
|
|
1731
|
+
if dry_run:
|
|
1732
|
+
return {"available": True, "vectors": len(vector_ids),
|
|
1733
|
+
"orphans": total_orphans, "deleted": 0, "dry_run": True,
|
|
1734
|
+
"ids": orphans[:20]}
|
|
1735
|
+
|
|
1736
|
+
deleted = 0
|
|
1737
|
+
for start in range(0, len(orphans), batch_size):
|
|
1738
|
+
for vid in orphans[start:start + batch_size]:
|
|
1739
|
+
await self._vector_store.delete(
|
|
1740
|
+
collection=Collection.MEMORIES, id=vid)
|
|
1741
|
+
deleted += 1
|
|
1742
|
+
if start + batch_size < len(orphans) and batch_sleep_secs > 0:
|
|
1743
|
+
await asyncio.sleep(batch_sleep_secs)
|
|
1744
|
+
|
|
1745
|
+
return {"available": True, "vectors": len(vector_ids),
|
|
1746
|
+
"orphans": total_orphans, "deleted": deleted,
|
|
1747
|
+
"dry_run": False, "ids": orphans[:20]}
|
|
1748
|
+
|
|
1749
|
+
async def touch_memory(self, memory_id: str) -> None:
|
|
1750
|
+
"""Record a memory recall, incrementing its recall counter and updating accessed_at.
|
|
1751
|
+
|
|
1752
|
+
Calling this after retrieval ensures the recall bonus in the Ebbinghaus
|
|
1753
|
+
formula is applied on the next decay pass, slowing future decay.
|
|
1754
|
+
|
|
1755
|
+
Args:
|
|
1756
|
+
memory_id: UUID of the Memory node to touch.
|
|
1757
|
+
"""
|
|
1758
|
+
async with self.driver.session(database=self.database) as session:
|
|
1759
|
+
await session.run(
|
|
1760
|
+
"""
|
|
1761
|
+
MATCH (m:Memory {id: $memory_id})
|
|
1762
|
+
SET m.recalls = coalesce(m.recalls, 0) + 1,
|
|
1763
|
+
m.accessed_at = datetime()
|
|
1764
|
+
""",
|
|
1765
|
+
memory_id=memory_id,
|
|
1766
|
+
)
|
|
1767
|
+
|
|
1768
|
+
async def _update_effective_confidence_batch(self, batch_size: int = 1000) -> None:
|
|
1769
|
+
"""Update effective_confidence for all memories in batches.
|
|
1770
|
+
|
|
1771
|
+
Each batch is one read and one write round trip: the confidences are
|
|
1772
|
+
computed in Python, then written back with a single UNWIND. Writing
|
|
1773
|
+
one auto-commit statement per memory made this pass scale with the
|
|
1774
|
+
memory count (thousands of round trips), which overran the autonomy
|
|
1775
|
+
tick budget and cancelled the whole tick every time it ran. The read
|
|
1776
|
+
is ordered by id so SKIP/LIMIT paging is stable while we write.
|
|
1777
|
+
"""
|
|
1778
|
+
from datetime import datetime as _dt, timezone as _tz
|
|
1779
|
+
now = _dt.now(_tz.utc)
|
|
1780
|
+
offset = 0
|
|
1781
|
+
while True:
|
|
1782
|
+
async with self.driver.session(database=self.database) as session:
|
|
1783
|
+
result = await session.run(
|
|
1784
|
+
"""
|
|
1785
|
+
MATCH (m:Memory)
|
|
1786
|
+
RETURN m {
|
|
1787
|
+
.id, .base_confidence, .source_reliability,
|
|
1788
|
+
.corroboration_count, .contradiction_count,
|
|
1789
|
+
.recalls, .last_verified_at, .created_at,
|
|
1790
|
+
.epistemic_state
|
|
1791
|
+
} AS mem
|
|
1792
|
+
ORDER BY m.id
|
|
1793
|
+
SKIP $offset LIMIT $limit
|
|
1794
|
+
""",
|
|
1795
|
+
offset=offset,
|
|
1796
|
+
limit=batch_size,
|
|
1797
|
+
)
|
|
1798
|
+
rows = [dict(r["mem"]) async for r in result]
|
|
1799
|
+
if not rows:
|
|
1800
|
+
break
|
|
1801
|
+
updates = []
|
|
1802
|
+
for row in rows:
|
|
1803
|
+
if row.get("id") is None:
|
|
1804
|
+
continue
|
|
1805
|
+
new_confidence = self.compute_effective_confidence(
|
|
1806
|
+
base_confidence=row.get("base_confidence") or 1.0,
|
|
1807
|
+
source_reliability=row.get("source_reliability") or 0.5,
|
|
1808
|
+
corroboration_count=row.get("corroboration_count") or 0,
|
|
1809
|
+
contradiction_count=row.get("contradiction_count") or 0,
|
|
1810
|
+
recalls=row.get("recalls") or 0,
|
|
1811
|
+
last_verified_at=row.get("last_verified_at"),
|
|
1812
|
+
created_at=row.get("created_at") or now,
|
|
1813
|
+
epistemic_state=row.get("epistemic_state") or "inferred",
|
|
1814
|
+
now=now,
|
|
1815
|
+
)
|
|
1816
|
+
updates.append({"id": row["id"], "effective_confidence": new_confidence})
|
|
1817
|
+
if updates:
|
|
1818
|
+
await session.run(
|
|
1819
|
+
"""
|
|
1820
|
+
UNWIND $updates AS u
|
|
1821
|
+
MATCH (m:Memory {id: u.id})
|
|
1822
|
+
SET m.effective_confidence = u.effective_confidence
|
|
1823
|
+
""",
|
|
1824
|
+
updates=updates,
|
|
1825
|
+
)
|
|
1826
|
+
if len(rows) < batch_size:
|
|
1827
|
+
break
|
|
1828
|
+
offset += batch_size
|
|
1829
|
+
|
|
1830
|
+
async def verify_memory(self, memory_id: str) -> None:
|
|
1831
|
+
"""Mark a memory as manually verified.
|
|
1832
|
+
|
|
1833
|
+
Sets last_verified_at, transitions epistemic_state to ``verified``
|
|
1834
|
+
if currently in an active state, and floors effective_confidence
|
|
1835
|
+
at 0.9.
|
|
1836
|
+
"""
|
|
1837
|
+
async with self.driver.session(database=self.database) as session:
|
|
1838
|
+
await session.run(
|
|
1839
|
+
"""
|
|
1840
|
+
MATCH (m:Memory {id: $memory_id})
|
|
1841
|
+
SET m.last_verified_at = datetime(),
|
|
1842
|
+
m.epistemic_state = CASE
|
|
1843
|
+
WHEN m.epistemic_state IN ["inferred", "observed", "corroborated"]
|
|
1844
|
+
THEN "verified"
|
|
1845
|
+
ELSE m.epistemic_state
|
|
1846
|
+
END,
|
|
1847
|
+
m.effective_confidence = CASE
|
|
1848
|
+
WHEN m.effective_confidence < 0.9 THEN 0.9
|
|
1849
|
+
ELSE m.effective_confidence
|
|
1850
|
+
END
|
|
1851
|
+
""",
|
|
1852
|
+
memory_id=memory_id,
|
|
1853
|
+
)
|
|
1854
|
+
|
|
1855
|
+
async def get_memory(self, memory_id: str) -> Optional[Dict[str, Any]]:
|
|
1856
|
+
"""Fetch a single memory by ID.
|
|
1857
|
+
|
|
1858
|
+
Returns:
|
|
1859
|
+
Memory dict or None if not found.
|
|
1860
|
+
"""
|
|
1861
|
+
async with self.driver.session(database=self.database) as session:
|
|
1862
|
+
result = await session.run(
|
|
1863
|
+
"""
|
|
1864
|
+
MATCH (m:Memory {id: $memory_id})
|
|
1865
|
+
OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
|
|
1866
|
+
WITH m, collect(e.name) AS entity_names
|
|
1867
|
+
RETURN m {.*, entities: entity_names} AS memory
|
|
1868
|
+
""",
|
|
1869
|
+
memory_id=memory_id,
|
|
1870
|
+
)
|
|
1871
|
+
record = await result.single()
|
|
1872
|
+
return dict(record["memory"]) if record else None
|
|
1873
|
+
|
|
1874
|
+
async def transition_epistemic_state(
|
|
1875
|
+
self,
|
|
1876
|
+
memory_id: str,
|
|
1877
|
+
new_state: str,
|
|
1878
|
+
superseded_by: Optional[str] = None,
|
|
1879
|
+
) -> None:
|
|
1880
|
+
"""Transition a memory to a new epistemic state.
|
|
1881
|
+
|
|
1882
|
+
Args:
|
|
1883
|
+
memory_id: UUID of the memory to transition.
|
|
1884
|
+
new_state: Target EpistemicState value.
|
|
1885
|
+
superseded_by: Optional UUID of the memory that superseded this one.
|
|
1886
|
+
"""
|
|
1887
|
+
async with self.driver.session(database=self.database) as session:
|
|
1888
|
+
await session.run(
|
|
1889
|
+
"""
|
|
1890
|
+
MATCH (m:Memory {id: $memory_id})
|
|
1891
|
+
SET m.epistemic_state = $new_state
|
|
1892
|
+
WITH m
|
|
1893
|
+
FOREACH (_ IN CASE WHEN $superseded_by IS NOT NULL THEN [1] ELSE [] END |
|
|
1894
|
+
SET m.superseded_by = $superseded_by
|
|
1895
|
+
)
|
|
1896
|
+
""",
|
|
1897
|
+
memory_id=memory_id,
|
|
1898
|
+
new_state=new_state,
|
|
1899
|
+
superseded_by=superseded_by,
|
|
1900
|
+
)
|
|
1901
|
+
|
|
1902
|
+
async def archive_memories(self, max_age_days: int = 30) -> int:
|
|
1903
|
+
"""Archive memories that have been in a terminal state for too long.
|
|
1904
|
+
|
|
1905
|
+
Relabels :Memory to :ArchivedMemory, copies key relationships,
|
|
1906
|
+
removes from vector store, and deletes the original.
|
|
1907
|
+
|
|
1908
|
+
Returns:
|
|
1909
|
+
Number of memories archived.
|
|
1910
|
+
"""
|
|
1911
|
+
archived = 0
|
|
1912
|
+
async with self.driver.session(database=self.database) as session:
|
|
1913
|
+
result = await session.run(
|
|
1914
|
+
"""
|
|
1915
|
+
MATCH (m:Memory)
|
|
1916
|
+
WHERE m.epistemic_state IN ["superseded", "deprecated", "stale"]
|
|
1917
|
+
AND duration.inDays(m.accessed_at, datetime()).days >= $max_age_days
|
|
1918
|
+
RETURN m.id AS id
|
|
1919
|
+
""",
|
|
1920
|
+
max_age_days=max_age_days,
|
|
1921
|
+
)
|
|
1922
|
+
memory_ids = [r["id"] async for r in result]
|
|
1923
|
+
|
|
1924
|
+
for memory_id in memory_ids:
|
|
1925
|
+
try:
|
|
1926
|
+
async with self.driver.session(database=self.database) as session:
|
|
1927
|
+
# Copy to ArchivedMemory with key relationships
|
|
1928
|
+
await session.run(
|
|
1929
|
+
"""
|
|
1930
|
+
MATCH (m:Memory {id: $memory_id})
|
|
1931
|
+
CREATE (a:ArchivedMemory)
|
|
1932
|
+
SET a = properties(m)
|
|
1933
|
+
WITH m, a
|
|
1934
|
+
OPTIONAL MATCH (m)-[r:MENTIONS]->(e:Entity)
|
|
1935
|
+
FOREACH (_ IN CASE WHEN e IS NOT NULL THEN [1] ELSE [] END |
|
|
1936
|
+
CREATE (a)-[:MENTIONS {created_at: datetime()}]->(e)
|
|
1937
|
+
)
|
|
1938
|
+
WITH m, a
|
|
1939
|
+
OPTIONAL MATCH (m)-[r:ABOUT]->(p:Person)
|
|
1940
|
+
FOREACH (_ IN CASE WHEN p IS NOT NULL THEN [1] ELSE [] END |
|
|
1941
|
+
CREATE (a)-[:ABOUT {created_at: datetime()}]->(p)
|
|
1942
|
+
)
|
|
1943
|
+
WITH m, a
|
|
1944
|
+
OPTIONAL MATCH (m)-[r:SUPERSEDES]->(old:Memory)
|
|
1945
|
+
FOREACH (_ IN CASE WHEN old IS NOT NULL THEN [1] ELSE [] END |
|
|
1946
|
+
CREATE (a)-[:SUPERSEDES {superseded_at: r.superseded_at}]->(old)
|
|
1947
|
+
)
|
|
1948
|
+
WITH m, a
|
|
1949
|
+
OPTIONAL MATCH (m)-[r:DERIVED_FROM]->(fa:FileAnchor)
|
|
1950
|
+
FOREACH (_ IN CASE WHEN fa IS NOT NULL THEN [1] ELSE [] END |
|
|
1951
|
+
CREATE (a)-[:DERIVED_FROM {derivation_type: r.derivation_type}]->(fa)
|
|
1952
|
+
)
|
|
1953
|
+
DETACH DELETE m
|
|
1954
|
+
""",
|
|
1955
|
+
memory_id=memory_id,
|
|
1956
|
+
)
|
|
1957
|
+
# Remove from vector store
|
|
1958
|
+
if self._vector_store is not None:
|
|
1959
|
+
try:
|
|
1960
|
+
from apsimo.vector.collections import Collection
|
|
1961
|
+
await self._vector_store.delete(
|
|
1962
|
+
collection=Collection.MEMORIES,
|
|
1963
|
+
id=memory_id,
|
|
1964
|
+
)
|
|
1965
|
+
except Exception as exc:
|
|
1966
|
+
logger.debug("Failed to remove archived memory from vector store: %s", exc)
|
|
1967
|
+
archived += 1
|
|
1968
|
+
except Exception as exc:
|
|
1969
|
+
logger.warning("Failed to archive memory %s: %s", memory_id, exc)
|
|
1970
|
+
|
|
1971
|
+
return archived
|
|
1972
|
+
|
|
1973
|
+
# ------------------------------------------------------------------
|
|
1974
|
+
# Traversal
|
|
1975
|
+
# ------------------------------------------------------------------
|
|
1976
|
+
|
|
1977
|
+
# Allowlist of permitted Cypher templates. Each entry is an exact string
|
|
1978
|
+
# match against the cypher argument. This prevents run_query from being
|
|
1979
|
+
# used as a Cypher injection sink while retaining the escape hatch for
|
|
1980
|
+
# known safe ad-hoc queries added here explicitly.
|
|
1981
|
+
_ALLOWED_CYPHER: frozenset = frozenset({
|
|
1982
|
+
# (similarity_threshold Config write removed: adjustments now go
|
|
1983
|
+
# through the AdaptiveParamStore, which consumers actually read.)
|
|
1984
|
+
# StrategyAdjuster._recalibrate_baselines — recalculate baseline signals
|
|
1985
|
+
"""MATCH (p:Person)-[:EXHIBITED]->(s:Signal)
|
|
1986
|
+
WHERE s.timestamp >= datetime() - duration({days: 30})
|
|
1987
|
+
WITH p.id AS pid, s.signal_type AS stype, avg(s.normalized_value) AS baseline_val
|
|
1988
|
+
MERGE (b:Baseline {person_id: pid, signal_type: stype})
|
|
1989
|
+
SET b.value = baseline_val, b.updated_at = datetime()
|
|
1990
|
+
RETURN count(b) AS updated_baselines""",
|
|
1991
|
+
# Neo4jCognitionSeeder.seed — upsert BootstrapEvent node on first boot
|
|
1992
|
+
"""
|
|
1993
|
+
MERGE (b:BootstrapEvent {colony_id: $colony_id})
|
|
1994
|
+
SET b.colony_name = $colony_name,
|
|
1995
|
+
b.colony_version = $colony_version,
|
|
1996
|
+
b.network_id = $network_id,
|
|
1997
|
+
b.corpus_version = $corpus_version,
|
|
1998
|
+
b.bootstrapped_at = $bootstrapped_at,
|
|
1999
|
+
b.layer_count = $layer_count,
|
|
2000
|
+
b.endpoint_count = $endpoint_count
|
|
2001
|
+
RETURN b.colony_id
|
|
2002
|
+
""",
|
|
2003
|
+
|
|
2004
|
+
# ConnectionDiscoverer._find_temporal_patterns (with person_id)
|
|
2005
|
+
"MATCH (m1:Memory)-[:ABOUT]->(p:Person {id: $person_id})\n"
|
|
2006
|
+
"MATCH (m2:Memory)-[:ABOUT]->(p)\n"
|
|
2007
|
+
"WHERE m1.id < m2.id\n"
|
|
2008
|
+
" AND abs(duration.between(m1.created_at, m2.created_at).hours) <= $window_hours\n"
|
|
2009
|
+
" AND m1.created_at >= datetime() - duration({days: $lookback_days})\n"
|
|
2010
|
+
"WITH m1, m2, count(*) AS co_occurrences\n"
|
|
2011
|
+
"WHERE co_occurrences >= $min_count\n"
|
|
2012
|
+
"RETURN m1.id AS source_id, m2.id AS target_id,\n"
|
|
2013
|
+
" m1.type AS source_type, m2.type AS target_type,\n"
|
|
2014
|
+
" m1.metadata AS source_meta, m2.metadata AS target_meta,\n"
|
|
2015
|
+
" co_occurrences,\n"
|
|
2016
|
+
" toFloat(co_occurrences) / $lookback_days AS daily_rate\n"
|
|
2017
|
+
"ORDER BY daily_rate DESC\n"
|
|
2018
|
+
"LIMIT 20",
|
|
2019
|
+
# ConnectionDiscoverer._find_temporal_patterns (without person_id)
|
|
2020
|
+
"MATCH (m1:Memory), (m2:Memory)\n"
|
|
2021
|
+
"WHERE m1.id < m2.id\n"
|
|
2022
|
+
" AND abs(duration.between(m1.created_at, m2.created_at).hours) <= $window_hours\n"
|
|
2023
|
+
" AND m1.created_at >= datetime() - duration({days: $lookback_days})\n"
|
|
2024
|
+
"WITH m1, m2, count(*) AS co_occurrences\n"
|
|
2025
|
+
"WHERE co_occurrences >= $min_count\n"
|
|
2026
|
+
"RETURN m1.id AS source_id, m2.id AS target_id,\n"
|
|
2027
|
+
" m1.type AS source_type, m2.type AS target_type,\n"
|
|
2028
|
+
" m1.metadata AS source_meta, m2.metadata AS target_meta,\n"
|
|
2029
|
+
" co_occurrences,\n"
|
|
2030
|
+
" toFloat(co_occurrences) / $lookback_days AS daily_rate\n"
|
|
2031
|
+
"ORDER BY daily_rate DESC\n"
|
|
2032
|
+
"LIMIT 20",
|
|
2033
|
+
# ConnectionDiscoverer._find_entity_patterns
|
|
2034
|
+
"MATCH (e:Entity)<-[:MENTIONS]-(m:Memory)\n"
|
|
2035
|
+
"WHERE ($person_id IS NULL OR (m)-[:ABOUT]->(:Person {id: $person_id}))\n"
|
|
2036
|
+
"WITH e, collect(DISTINCT m.id) AS mems\n"
|
|
2037
|
+
"WITH e, mems, size(mems) AS mem_count\n"
|
|
2038
|
+
"WHERE mem_count >= 2\n"
|
|
2039
|
+
"RETURN e.name AS entity_name,\n"
|
|
2040
|
+
" mem_count AS occurrence_count,\n"
|
|
2041
|
+
" mems[0..5] AS evidence_sample\n"
|
|
2042
|
+
"ORDER BY mem_count DESC\n"
|
|
2043
|
+
"LIMIT 20",
|
|
2044
|
+
# ConnectionDiscoverer._find_behavioral_patterns
|
|
2045
|
+
"MATCH (p:Person {id: $person_id})-[:EXHIBITED]->(s1:Signal)\n"
|
|
2046
|
+
"MATCH (p)-[:EXHIBITED]->(s2:Signal)\n"
|
|
2047
|
+
"WHERE s1.signal_type <> s2.signal_type\n"
|
|
2048
|
+
" AND s2.timestamp > s1.timestamp\n"
|
|
2049
|
+
" AND duration.between(s1.timestamp, s2.timestamp).hours <= $window_hours\n"
|
|
2050
|
+
"WITH s1.signal_type AS type_a, s2.signal_type AS type_b,\n"
|
|
2051
|
+
" count(*) AS occurrences,\n"
|
|
2052
|
+
" avg(s2.normalized_value - s1.normalized_value) AS avg_delta,\n"
|
|
2053
|
+
" collect(s1.id)[0..5] AS evidence\n"
|
|
2054
|
+
"WHERE occurrences >= $min_occurrences\n"
|
|
2055
|
+
"RETURN type_a, type_b, occurrences, avg_delta, evidence\n"
|
|
2056
|
+
"ORDER BY occurrences DESC\n"
|
|
2057
|
+
"LIMIT 15",
|
|
2058
|
+
# Consolidator._detect_conflicts
|
|
2059
|
+
"MATCH (m1:Memory)-[:MENTIONS]->(e:Entity)<-[:MENTIONS]-(m2:Memory) "
|
|
2060
|
+
"WHERE id(m1) < id(m2) "
|
|
2061
|
+
"AND NOT (m1)-[:MERGED_INTO]-() AND NOT (m2)-[:MERGED_INTO]-() "
|
|
2062
|
+
"RETURN m1.id AS id_a, m2.id AS id_b, e.name AS entity, "
|
|
2063
|
+
" m1.content AS content_a, m2.content AS content_b",
|
|
2064
|
+
})
|
|
2065
|
+
|
|
2066
|
+
# Additional allowlisted queries registered by callers via
|
|
2067
|
+
# register_allowed_cypher(). Same discipline as _ALLOWED_CYPHER: exact,
|
|
2068
|
+
# fully-parameterized strings only, never built from user input. This
|
|
2069
|
+
# keeps each query single-sourced next to the code that owns it instead
|
|
2070
|
+
# of duplicating string literals here.
|
|
2071
|
+
_EXTRA_ALLOWED_CYPHER: set = set()
|
|
2072
|
+
|
|
2073
|
+
@classmethod
|
|
2074
|
+
def register_allowed_cypher(cls, cypher: str) -> str:
|
|
2075
|
+
"""Register one exact, parameterized Cypher string for run_query."""
|
|
2076
|
+
cls._EXTRA_ALLOWED_CYPHER.add(cypher)
|
|
2077
|
+
return cypher
|
|
2078
|
+
|
|
2079
|
+
async def run_query(self, cypher: str, params: dict) -> List[Dict[str, Any]]:
|
|
2080
|
+
"""Execute a Cypher query and return results as dicts.
|
|
2081
|
+
|
|
2082
|
+
WARNING: The ``cypher`` argument must never be constructed from
|
|
2083
|
+
user-controlled input. Only queries listed in ``_ALLOWED_CYPHER``
|
|
2084
|
+
are permitted; all others raise ``ValueError``. To add a new query,
|
|
2085
|
+
add its exact string to ``_ALLOWED_CYPHER`` after security review.
|
|
2086
|
+
|
|
2087
|
+
Args:
|
|
2088
|
+
cypher: Exact Cypher query string (must be in _ALLOWED_CYPHER).
|
|
2089
|
+
params: Dict of $param bindings — always parameterized, never interpolated.
|
|
2090
|
+
|
|
2091
|
+
Returns:
|
|
2092
|
+
List of result records as plain dicts.
|
|
2093
|
+
|
|
2094
|
+
Raises:
|
|
2095
|
+
ValueError: If ``cypher`` is not in the allowlist.
|
|
2096
|
+
"""
|
|
2097
|
+
if (cypher not in self._ALLOWED_CYPHER
|
|
2098
|
+
and cypher not in self._EXTRA_ALLOWED_CYPHER):
|
|
2099
|
+
raise ValueError(
|
|
2100
|
+
"run_query: cypher string not in allowlist. "
|
|
2101
|
+
"Add to GraphClient._ALLOWED_CYPHER or register via "
|
|
2102
|
+
"register_allowed_cypher() after security review."
|
|
2103
|
+
)
|
|
2104
|
+
async with self.driver.session(database=self.database) as session:
|
|
2105
|
+
result = await session.run(cypher, **params)
|
|
2106
|
+
return [dict(r) async for r in result]
|
|
2107
|
+
|
|
2108
|
+
# GRAPH-01: server-enforced maximum traversal depth
|
|
2109
|
+
MAX_GRAPH_DEPTH = 10
|
|
2110
|
+
|
|
2111
|
+
async def traverse_memory_connections(
|
|
2112
|
+
self,
|
|
2113
|
+
memory_id: str,
|
|
2114
|
+
max_depth: int = 3,
|
|
2115
|
+
min_strength: float = 0.3,
|
|
2116
|
+
limit: int = 20,
|
|
2117
|
+
) -> List[Dict[str, Any]]:
|
|
2118
|
+
"""Walk multi-hop causal / supporting chains from a memory.
|
|
2119
|
+
|
|
2120
|
+
Uses ``CAUSED_BY``, ``LED_TO``, and ``SUPPORTS`` edge types up to
|
|
2121
|
+
*max_depth* hops, filtering out nodes below *min_strength*.
|
|
2122
|
+
|
|
2123
|
+
Returns:
|
|
2124
|
+
A list of dicts with ``memory``, ``distance``, and
|
|
2125
|
+
``path_weight`` keys.
|
|
2126
|
+
"""
|
|
2127
|
+
# GRAPH-01: clamp depth to server-enforced maximum
|
|
2128
|
+
max_depth = min(int(max_depth), self.MAX_GRAPH_DEPTH)
|
|
2129
|
+
async with self.driver.session(database=self.database) as session:
|
|
2130
|
+
result = await session.run(
|
|
2131
|
+
"""
|
|
2132
|
+
MATCH path = (m1:Memory)-[:CAUSED_BY|LED_TO|SUPPORTS*1..$max_depth]->(m2:Memory)
|
|
2133
|
+
WHERE m1.id = $memory_id
|
|
2134
|
+
AND all(node IN nodes(path) WHERE node.strength >= $min_strength)
|
|
2135
|
+
RETURN m2 {.*} AS memory,
|
|
2136
|
+
length(path) AS distance,
|
|
2137
|
+
reduce(w = 1.0, r IN relationships(path) | w * r.weight) AS path_weight
|
|
2138
|
+
ORDER BY path_weight DESC
|
|
2139
|
+
LIMIT $limit
|
|
2140
|
+
""",
|
|
2141
|
+
memory_id=memory_id,
|
|
2142
|
+
max_depth=max_depth,
|
|
2143
|
+
min_strength=min_strength,
|
|
2144
|
+
limit=limit,
|
|
2145
|
+
)
|
|
2146
|
+
return [
|
|
2147
|
+
{
|
|
2148
|
+
"memory": record["memory"],
|
|
2149
|
+
"distance": record["distance"],
|
|
2150
|
+
"path_weight": record["path_weight"],
|
|
2151
|
+
}
|
|
2152
|
+
async for record in result
|
|
2153
|
+
]
|
|
2154
|
+
|
|
2155
|
+
|
|
2156
|
+
# ------------------------------------------------------------------
|
|
2157
|
+
# Baseline methods (GraphBaselineStore)
|
|
2158
|
+
# ------------------------------------------------------------------
|
|
2159
|
+
|
|
2160
|
+
async def _run_get_baseline(self, person_id: str) -> dict | None:
|
|
2161
|
+
"""Read baseline properties from a Person node. Returns None if not found."""
|
|
2162
|
+
try:
|
|
2163
|
+
async with self.driver.session(database=self.database) as session:
|
|
2164
|
+
result = await session.run(GET_BASELINE, person_id=person_id)
|
|
2165
|
+
record = await result.single()
|
|
2166
|
+
if record is None:
|
|
2167
|
+
return None
|
|
2168
|
+
return dict(record)
|
|
2169
|
+
except Exception as exc:
|
|
2170
|
+
logger.debug("_run_get_baseline failed for %s: %s", person_id, exc)
|
|
2171
|
+
return None
|
|
2172
|
+
|
|
2173
|
+
async def _run_update_baseline(
|
|
2174
|
+
self,
|
|
2175
|
+
person_id: str,
|
|
2176
|
+
msg_count: int,
|
|
2177
|
+
length_mean: float,
|
|
2178
|
+
length_m2: float,
|
|
2179
|
+
length_std: float,
|
|
2180
|
+
hour_histogram: str,
|
|
2181
|
+
) -> None:
|
|
2182
|
+
"""Write updated baseline properties to a Person node."""
|
|
2183
|
+
try:
|
|
2184
|
+
async with self.driver.session(database=self.database) as session:
|
|
2185
|
+
await session.run(
|
|
2186
|
+
UPDATE_BASELINE,
|
|
2187
|
+
person_id=person_id,
|
|
2188
|
+
msg_count=msg_count,
|
|
2189
|
+
length_mean=length_mean,
|
|
2190
|
+
length_m2=length_m2,
|
|
2191
|
+
length_std=length_std,
|
|
2192
|
+
hour_histogram=hour_histogram,
|
|
2193
|
+
)
|
|
2194
|
+
except Exception as exc:
|
|
2195
|
+
logger.debug("_run_update_baseline failed for %s: %s", person_id, exc)
|
|
2196
|
+
|
|
2197
|
+
async def list_person_ids(self) -> List[str]:
|
|
2198
|
+
"""Return all person IDs that have a Person node in the graph."""
|
|
2199
|
+
query = "MATCH (p:Person) RETURN p.id AS id"
|
|
2200
|
+
async with self.driver.session(database=self.database) as session:
|
|
2201
|
+
result = await session.run(query)
|
|
2202
|
+
records = await result.values()
|
|
2203
|
+
return [r[0] for r in records if r[0]]
|
|
2204
|
+
|
|
2205
|
+
async def delete_person(self, person_id: str) -> bool:
|
|
2206
|
+
"""Permanently delete a Person node and all attached relationships."""
|
|
2207
|
+
t0 = time.monotonic()
|
|
2208
|
+
try:
|
|
2209
|
+
async with self.driver.session(database=self.database) as session:
|
|
2210
|
+
result = await session.run(
|
|
2211
|
+
"MATCH (p:Person {id: $person_id}) DETACH DELETE p RETURN count(p) AS deleted",
|
|
2212
|
+
person_id=person_id,
|
|
2213
|
+
)
|
|
2214
|
+
record = await result.single()
|
|
2215
|
+
deleted = record["deleted"] if record else 0
|
|
2216
|
+
logger.debug(
|
|
2217
|
+
"delete_person %s deleted=%d %.1fms",
|
|
2218
|
+
person_id, deleted, (time.monotonic() - t0) * 1000,
|
|
2219
|
+
)
|
|
2220
|
+
return deleted > 0
|
|
2221
|
+
except Exception as exc:
|
|
2222
|
+
logger.error("delete_person failed for %s: %s", person_id, exc)
|
|
2223
|
+
return False
|
|
2224
|
+
|
|
2225
|
+
async def get_people_with_substance(
|
|
2226
|
+
self,
|
|
2227
|
+
min_signals: int = 2,
|
|
2228
|
+
min_memories: int = 2,
|
|
2229
|
+
) -> List[Dict[str, Any]]:
|
|
2230
|
+
"""Return Person nodes that have enough substance to become contacts.
|
|
2231
|
+
|
|
2232
|
+
A person has substance if they have:
|
|
2233
|
+
- a name AND (a phone or email)
|
|
2234
|
+
- OR a name AND (>= min_signals OR >= min_memories)
|
|
2235
|
+
"""
|
|
2236
|
+
t0 = time.monotonic()
|
|
2237
|
+
try:
|
|
2238
|
+
async with self.driver.session(database=self.database) as session:
|
|
2239
|
+
result = await session.run(
|
|
2240
|
+
"""
|
|
2241
|
+
MATCH (p:Person)
|
|
2242
|
+
WHERE p.name IS NOT NULL
|
|
2243
|
+
OPTIONAL MATCH (p)-[:EXHIBITED]->(s:Signal)
|
|
2244
|
+
OPTIONAL MATCH (m:Memory)-[:ABOUT]->(p)
|
|
2245
|
+
WITH p, count(s) AS sigs, count(m) AS mems
|
|
2246
|
+
WHERE p.phone IS NOT NULL OR p.email IS NOT NULL
|
|
2247
|
+
OR sigs >= $min_signals OR mems >= $min_memories
|
|
2248
|
+
RETURN p.id AS id,
|
|
2249
|
+
p.name AS name,
|
|
2250
|
+
p.phone AS phone,
|
|
2251
|
+
p.email AS email,
|
|
2252
|
+
coalesce(p.score, 0.0) AS score,
|
|
2253
|
+
coalesce(p.tier, 'regular') AS tier,
|
|
2254
|
+
sigs,
|
|
2255
|
+
mems
|
|
2256
|
+
""",
|
|
2257
|
+
min_signals=min_signals,
|
|
2258
|
+
min_memories=min_memories,
|
|
2259
|
+
)
|
|
2260
|
+
people = [dict(r) async for r in result]
|
|
2261
|
+
logger.debug(
|
|
2262
|
+
"get_people_with_substance count=%d %.1fms",
|
|
2263
|
+
len(people), (time.monotonic() - t0) * 1000,
|
|
2264
|
+
)
|
|
2265
|
+
return people
|
|
2266
|
+
except Exception as exc:
|
|
2267
|
+
logger.error("get_people_with_substance failed: %s", exc)
|
|
2268
|
+
return []
|
|
2269
|
+
|
|
2270
|
+
# ------------------------------------------------------------------
|
|
2271
|
+
# Signal & relationship methods (required by SignalCollector,
|
|
2272
|
+
# BaselineStore, and RelationshipScorer)
|
|
2273
|
+
# ------------------------------------------------------------------
|
|
2274
|
+
|
|
2275
|
+
async def store_signal(self, signal: Any) -> str:
|
|
2276
|
+
"""Persist a behavioral signal to the graph.
|
|
2277
|
+
|
|
2278
|
+
Creates a :Signal node linked to the :Person via [:EXHIBITED].
|
|
2279
|
+
"""
|
|
2280
|
+
from apsimo.intelligence.graph.queries import STORE_SIGNAL, GET_BASELINE, UPDATE_BASELINE
|
|
2281
|
+
|
|
2282
|
+
t0 = time.monotonic()
|
|
2283
|
+
try:
|
|
2284
|
+
async with self.driver.session(database=self.database) as session:
|
|
2285
|
+
result = await session.run(
|
|
2286
|
+
STORE_SIGNAL,
|
|
2287
|
+
person_id=signal.person_id,
|
|
2288
|
+
signal_type=signal.signal_type,
|
|
2289
|
+
raw_value=signal.raw_value,
|
|
2290
|
+
normalized_value=signal.normalized_value,
|
|
2291
|
+
timestamp=signal.timestamp.isoformat(),
|
|
2292
|
+
source=signal.source,
|
|
2293
|
+
)
|
|
2294
|
+
record = await result.single()
|
|
2295
|
+
sid = record["id"] if record else ""
|
|
2296
|
+
logger.debug(
|
|
2297
|
+
"store_signal person=%s type=%s %.1fms",
|
|
2298
|
+
signal.person_id, signal.signal_type, (time.monotonic() - t0) * 1000,
|
|
2299
|
+
)
|
|
2300
|
+
return sid
|
|
2301
|
+
except Exception as exc:
|
|
2302
|
+
logger.error("store_signal failed for person %s: %s", signal.person_id, exc)
|
|
2303
|
+
return ""
|
|
2304
|
+
|
|
2305
|
+
async def get_recent_signals(
|
|
2306
|
+
self, person_id: str, hours: int = 24, signal_type: Optional[str] = None
|
|
2307
|
+
) -> List[Any]:
|
|
2308
|
+
"""Fetch recent signals for a person within a time window.
|
|
2309
|
+
|
|
2310
|
+
Returns a list of Signal dataclass instances from the
|
|
2311
|
+
signal_collector module.
|
|
2312
|
+
"""
|
|
2313
|
+
from apsimo.intelligence.mind_model.signal_collector import Signal
|
|
2314
|
+
from datetime import timedelta
|
|
2315
|
+
|
|
2316
|
+
cutoff = (self._utcnow() - timedelta(hours=hours)).isoformat()
|
|
2317
|
+
|
|
2318
|
+
cypher = (
|
|
2319
|
+
"MATCH (p:Person {id: $person_id})-[:EXHIBITED]->(s:Signal)\n"
|
|
2320
|
+
"WHERE s.timestamp >= datetime($cutoff)\n"
|
|
2321
|
+
)
|
|
2322
|
+
if signal_type:
|
|
2323
|
+
cypher += "AND s.signal_type = $signal_type\n"
|
|
2324
|
+
cypher += (
|
|
2325
|
+
"RETURN s.signal_type AS signal_type,\n"
|
|
2326
|
+
" s.raw_value AS raw_value,\n"
|
|
2327
|
+
" s.normalized_value AS normalized_value,\n"
|
|
2328
|
+
" s.timestamp AS timestamp,\n"
|
|
2329
|
+
" s.source AS source\n"
|
|
2330
|
+
"ORDER BY s.timestamp DESC"
|
|
2331
|
+
)
|
|
2332
|
+
|
|
2333
|
+
t0 = time.monotonic()
|
|
2334
|
+
try:
|
|
2335
|
+
async with self.driver.session(database=self.database) as session:
|
|
2336
|
+
result = await session.run(
|
|
2337
|
+
cypher,
|
|
2338
|
+
person_id=person_id,
|
|
2339
|
+
cutoff=cutoff,
|
|
2340
|
+
signal_type=signal_type,
|
|
2341
|
+
)
|
|
2342
|
+
signals = []
|
|
2343
|
+
async for record in result:
|
|
2344
|
+
ts = record["timestamp"]
|
|
2345
|
+
if hasattr(ts, "to_native"):
|
|
2346
|
+
ts = ts.to_native()
|
|
2347
|
+
signals.append(Signal(
|
|
2348
|
+
signal_type=record["signal_type"],
|
|
2349
|
+
raw_value=float(record["raw_value"]),
|
|
2350
|
+
normalized_value=float(record["normalized_value"]),
|
|
2351
|
+
timestamp=ts,
|
|
2352
|
+
person_id=person_id,
|
|
2353
|
+
source=record["source"] or "message",
|
|
2354
|
+
))
|
|
2355
|
+
logger.debug(
|
|
2356
|
+
"get_recent_signals person=%s count=%d %.1fms",
|
|
2357
|
+
person_id, len(signals), (time.monotonic() - t0) * 1000,
|
|
2358
|
+
)
|
|
2359
|
+
return signals
|
|
2360
|
+
except Exception as exc:
|
|
2361
|
+
logger.error("get_recent_signals failed for person %s: %s", person_id, exc)
|
|
2362
|
+
return []
|
|
2363
|
+
|
|
2364
|
+
async def get_all_people(self) -> List[Dict[str, Any]]:
|
|
2365
|
+
"""Return all Person nodes with their current scores."""
|
|
2366
|
+
t0 = time.monotonic()
|
|
2367
|
+
try:
|
|
2368
|
+
async with self.driver.session(database=self.database) as session:
|
|
2369
|
+
result = await session.run(
|
|
2370
|
+
"MATCH (p:Person) "
|
|
2371
|
+
"RETURN p.id AS id, p.name AS name, "
|
|
2372
|
+
" coalesce(p.score, 50.0) AS score, "
|
|
2373
|
+
" coalesce(p.tier, 'regular') AS tier, "
|
|
2374
|
+
" p.lastInteraction AS last_interaction"
|
|
2375
|
+
)
|
|
2376
|
+
people = [dict(r) async for r in result]
|
|
2377
|
+
logger.debug("get_all_people count=%d %.1fms", len(people), (time.monotonic() - t0) * 1000)
|
|
2378
|
+
return people
|
|
2379
|
+
except Exception as exc:
|
|
2380
|
+
logger.error("get_all_people failed: %s", exc)
|
|
2381
|
+
return []
|
|
2382
|
+
|
|
2383
|
+
async def record_score_change(
|
|
2384
|
+
self,
|
|
2385
|
+
person_id: str,
|
|
2386
|
+
new_score: float,
|
|
2387
|
+
new_tier: str,
|
|
2388
|
+
old_score: float,
|
|
2389
|
+
reason: str,
|
|
2390
|
+
store = None, # Optional SQLiteContactStore for reverse sync
|
|
2391
|
+
) -> None:
|
|
2392
|
+
"""Persist a relationship score change with audit trail."""
|
|
2393
|
+
from apsimo.intelligence.graph.queries import RECORD_SCORE_CHANGE
|
|
2394
|
+
|
|
2395
|
+
t0 = time.monotonic()
|
|
2396
|
+
try:
|
|
2397
|
+
async with self.driver.session(database=self.database) as session:
|
|
2398
|
+
await session.run(
|
|
2399
|
+
RECORD_SCORE_CHANGE,
|
|
2400
|
+
person_id=person_id,
|
|
2401
|
+
new_score=new_score,
|
|
2402
|
+
new_tier=new_tier,
|
|
2403
|
+
delta=new_score - old_score,
|
|
2404
|
+
reason=reason,
|
|
2405
|
+
)
|
|
2406
|
+
logger.debug(
|
|
2407
|
+
"record_score_change person=%s %.1f→%.1f (%s) %.1fms",
|
|
2408
|
+
person_id, old_score, new_score, new_tier, (time.monotonic() - t0) * 1000,
|
|
2409
|
+
)
|
|
2410
|
+
# Sync to SQLite if linked contact exists
|
|
2411
|
+
if store is not None:
|
|
2412
|
+
try:
|
|
2413
|
+
contact = await store.find_by_person_node_id(person_id)
|
|
2414
|
+
if contact:
|
|
2415
|
+
# scorer works in 0-100; the contact field is 0-1
|
|
2416
|
+
_norm = new_score / 100.0 if new_score > 1.0 else new_score
|
|
2417
|
+
await store.update_relationship_score(contact.contact_id, _norm)
|
|
2418
|
+
except Exception as exc:
|
|
2419
|
+
logger.debug("Score sync to SQLite failed for %s: %s", person_id, exc)
|
|
2420
|
+
except Exception as exc:
|
|
2421
|
+
logger.error("record_score_change failed for person %s: %s", person_id, exc)
|
|
2422
|
+
|
|
2423
|
+
async def get_person(self, person_id: str) -> Optional[Dict[str, Any]]:
|
|
2424
|
+
"""Fetch a Person node with all properties."""
|
|
2425
|
+
t0 = time.monotonic()
|
|
2426
|
+
try:
|
|
2427
|
+
async with self.driver.session(database=self.database) as session:
|
|
2428
|
+
result = await session.run(
|
|
2429
|
+
"MATCH (p:Person {id: $person_id}) RETURN p {.*} AS person",
|
|
2430
|
+
person_id=person_id,
|
|
2431
|
+
)
|
|
2432
|
+
record = await result.single()
|
|
2433
|
+
logger.debug("get_person %s %.1fms", person_id, (time.monotonic() - t0) * 1000)
|
|
2434
|
+
return dict(record["person"]) if record else None
|
|
2435
|
+
except Exception as exc:
|
|
2436
|
+
logger.error("get_person failed for %s: %s", person_id, exc)
|
|
2437
|
+
return None
|
|
2438
|
+
|
|
2439
|
+
# Property names permitted on Person nodes. Values are always passed as
|
|
2440
|
+
# parameters; this allowlist guards the one remaining interpolation point
|
|
2441
|
+
# (the property name itself) against accidental misuse from a caller that
|
|
2442
|
+
# forwards an attacker-controlled dict.
|
|
2443
|
+
_PERSON_PROPS_ALLOWED = frozenset({
|
|
2444
|
+
"name", "tier", "score", "lastInteraction", "created_at",
|
|
2445
|
+
"baseline_msg_count",
|
|
2446
|
+
"baseline_length_mean",
|
|
2447
|
+
"baseline_length_m2",
|
|
2448
|
+
"baseline_length_std",
|
|
2449
|
+
"baseline_hour_histogram",
|
|
2450
|
+
"baseline_updated_at",
|
|
2451
|
+
})
|
|
2452
|
+
|
|
2453
|
+
async def update_person(self, person_id: str, **props: Any) -> None:
|
|
2454
|
+
"""Update arbitrary properties on a Person node.
|
|
2455
|
+
|
|
2456
|
+
Only called from trusted internal code (BaselineStore).
|
|
2457
|
+
All values are passed as Neo4j parameters; property names are
|
|
2458
|
+
validated against ``_PERSON_PROPS_ALLOWED``.
|
|
2459
|
+
"""
|
|
2460
|
+
if not props:
|
|
2461
|
+
return
|
|
2462
|
+
unknown = set(props) - self._PERSON_PROPS_ALLOWED
|
|
2463
|
+
if unknown:
|
|
2464
|
+
raise ValueError(
|
|
2465
|
+
f"update_person rejected unknown properties: {sorted(unknown)}"
|
|
2466
|
+
)
|
|
2467
|
+
set_clauses = ", ".join(f"p.{k} = ${k}" for k in props)
|
|
2468
|
+
cypher = f"MATCH (p:Person {{id: $person_id}}) SET {set_clauses}"
|
|
2469
|
+
t0 = time.monotonic()
|
|
2470
|
+
try:
|
|
2471
|
+
async with self.driver.session(database=self.database) as session:
|
|
2472
|
+
await session.run(cypher, person_id=person_id, **props)
|
|
2473
|
+
logger.debug("update_person %s props=%s %.1fms", person_id, list(props), (time.monotonic() - t0) * 1000)
|
|
2474
|
+
except Exception as exc:
|
|
2475
|
+
logger.error("update_person failed for %s: %s", person_id, exc)
|
|
2476
|
+
|
|
2477
|
+
# ------------------------------------------------------------------
|
|
2478
|
+
|
|
2479
|
+
@staticmethod
|
|
2480
|
+
def _utcnow() -> "datetime":
|
|
2481
|
+
"""Return timezone-aware UTC now (isolated for testability)."""
|
|
2482
|
+
from datetime import datetime as _dt, timezone as _tz
|
|
2483
|
+
return _dt.now(_tz.utc)
|