apsimo 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (614) hide show
  1. apsimo/__init__.py +38 -0
  2. apsimo/__main__.py +6 -0
  3. apsimo/agent/__init__.py +6 -0
  4. apsimo/agent/client.py +276 -0
  5. apsimo/agent/models.py +46 -0
  6. apsimo/agents/__init__.py +20 -0
  7. apsimo/agents/models.py +264 -0
  8. apsimo/agents/store.py +861 -0
  9. apsimo/agents/websocket.py +522 -0
  10. apsimo/api/__init__.py +1 -0
  11. apsimo/api/auth_telemetry.py +287 -0
  12. apsimo/api/authority.py +1203 -0
  13. apsimo/api/contact_grants.py +347 -0
  14. apsimo/api/middleware.py +483 -0
  15. apsimo/api/routers/__init__.py +1 -0
  16. apsimo/api/routers/commitment_work.py +265 -0
  17. apsimo/api/routers/context_gate.py +123 -0
  18. apsimo/api/routers/executions.py +140 -0
  19. apsimo/api/routers/followup_plans.py +147 -0
  20. apsimo/api/routers/governed_actions.py +162 -0
  21. apsimo/api/routers/host.py +14473 -0
  22. apsimo/api/routers/initiative_work.py +115 -0
  23. apsimo/api/routers/mining.py +104 -0
  24. apsimo/api/routers/observations.py +110 -0
  25. apsimo/api/routers/social_state.py +225 -0
  26. apsimo/api/routers/task_queue.py +2715 -0
  27. apsimo/api/routers/temporal_followups.py +251 -0
  28. apsimo/api/routers/transport.py +110 -0
  29. apsimo/api/routers/transport_ingress_api.py +240 -0
  30. apsimo/api/schemas/__init__.py +1 -0
  31. apsimo/api/schemas/host.py +1949 -0
  32. apsimo/autonomy/cli.py +110 -0
  33. apsimo/autonomy/condition_worker.py +437 -0
  34. apsimo/autonomy/config.py +424 -0
  35. apsimo/autonomy/loop.py +4316 -0
  36. apsimo/autonomy/registry.py +339 -0
  37. apsimo/autonomy/scheduler.py +1822 -0
  38. apsimo/autonomy/synthesis.py +449 -0
  39. apsimo/backup.py +962 -0
  40. apsimo/beliefs/__init__.py +23 -0
  41. apsimo/beliefs/contradictions.py +109 -0
  42. apsimo/beliefs/decay.py +61 -0
  43. apsimo/beliefs/engine.py +479 -0
  44. apsimo/beliefs/models.py +67 -0
  45. apsimo/beliefs/promotion.py +41 -0
  46. apsimo/beliefs/resolve.py +58 -0
  47. apsimo/beliefs/source_claims.py +690 -0
  48. apsimo/beliefs/source_projection.py +883 -0
  49. apsimo/beliefs/source_time.py +208 -0
  50. apsimo/beliefs/store.py +133 -0
  51. apsimo/briefings/aggregators.py +824 -0
  52. apsimo/briefings/composer.py +420 -0
  53. apsimo/briefings/config.py +55 -0
  54. apsimo/briefings/delivery.py +439 -0
  55. apsimo/briefings/engagement.py +97 -0
  56. apsimo/briefings/engine.py +274 -0
  57. apsimo/briefings/enhancer.py +99 -0
  58. apsimo/briefings/models.py +183 -0
  59. apsimo/briefings/scheduler.py +382 -0
  60. apsimo/briefings/store.py +435 -0
  61. apsimo/chain/__init__.py +48 -0
  62. apsimo/chain/block.py +100 -0
  63. apsimo/chain/cli.py +704 -0
  64. apsimo/chain/genesis.py +443 -0
  65. apsimo/chain/identity.py +416 -0
  66. apsimo/chain/keys.py +1025 -0
  67. apsimo/chain/local_keys.py +187 -0
  68. apsimo/chain/manager.py +290 -0
  69. apsimo/chain/node.py +163 -0
  70. apsimo/chain/plugin_transactions.py +371 -0
  71. apsimo/chain/protocol.py +220 -0
  72. apsimo/chain/state_machine.py +676 -0
  73. apsimo/chain/storage.py +503 -0
  74. apsimo/chain/transactions.py +250 -0
  75. apsimo/chain/validation.py +397 -0
  76. apsimo/channels/__init__.py +1 -0
  77. apsimo/channels/manifest.py +31 -0
  78. apsimo/channels/migrations/001_channels_schema.sql +12 -0
  79. apsimo/channels/phone_gateways.py +42 -0
  80. apsimo/channels/presence.py +188 -0
  81. apsimo/channels/router.py +235 -0
  82. apsimo/channels/store.py +231 -0
  83. apsimo/cli.py +2688 -0
  84. apsimo/cognition/__init__.py +11 -0
  85. apsimo/cognition/charter.py +398 -0
  86. apsimo/cognition/drive_governance.py +3530 -0
  87. apsimo/cognition/evidence_pipeline.py +1627 -0
  88. apsimo/cognition/external_events.py +932 -0
  89. apsimo/cognition/goal_spine.py +3488 -0
  90. apsimo/cognition/introspection.py +214 -0
  91. apsimo/cognition/prompt.py +150 -0
  92. apsimo/cognition/runtime.py +108 -0
  93. apsimo/cognition/trigger.py +154 -0
  94. apsimo/commitments/__init__.py +18 -0
  95. apsimo/commitments/local_work.py +355 -0
  96. apsimo/commitments/store.py +1052 -0
  97. apsimo/commitments/work.py +91 -0
  98. apsimo/compat.py +53 -0
  99. apsimo/compression/__init__.py +467 -0
  100. apsimo/connectors/__init__.py +21 -0
  101. apsimo/connectors/base.py +152 -0
  102. apsimo/connectors/caldav_calendar.py +125 -0
  103. apsimo/connectors/fs_documents.py +85 -0
  104. apsimo/connectors/imap_email.py +138 -0
  105. apsimo/connectors/manager.py +218 -0
  106. apsimo/connectors/webhook_pull.py +88 -0
  107. apsimo/contacts/__init__.py +33 -0
  108. apsimo/contacts/comms.py +357 -0
  109. apsimo/contacts/config.py +79 -0
  110. apsimo/contacts/exporters/__init__.py +1 -0
  111. apsimo/contacts/exporters/vcard.py +71 -0
  112. apsimo/contacts/identity_links.py +251 -0
  113. apsimo/contacts/importer.py +280 -0
  114. apsimo/contacts/importers/__init__.py +1 -0
  115. apsimo/contacts/importers/batch.py +43 -0
  116. apsimo/contacts/importers/macos_contacts.py +101 -0
  117. apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
  118. apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
  119. apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
  120. apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
  121. apsimo/contacts/migrations/005_identity_links.sql +27 -0
  122. apsimo/contacts/models.py +308 -0
  123. apsimo/contacts/scoring.py +16 -0
  124. apsimo/contacts/store.py +1623 -0
  125. apsimo/contacts/transport_ingress.py +252 -0
  126. apsimo/contacts/world_bridge.py +314 -0
  127. apsimo/contextgate/__init__.py +69 -0
  128. apsimo/contextgate/chunker.py +169 -0
  129. apsimo/contextgate/estimate.py +54 -0
  130. apsimo/contextgate/gate.py +313 -0
  131. apsimo/contextgate/retrieve.py +115 -0
  132. apsimo/delivery/__init__.py +16 -0
  133. apsimo/delivery/bridge.py +1260 -0
  134. apsimo/delivery/channels.py +526 -0
  135. apsimo/delivery/classification.py +50 -0
  136. apsimo/delivery/rate_limiter.py +268 -0
  137. apsimo/delivery/reachout_policy.py +206 -0
  138. apsimo/directed/__init__.py +22 -0
  139. apsimo/directed/audit.py +167 -0
  140. apsimo/directed/intake.py +95 -0
  141. apsimo/directed/models.py +191 -0
  142. apsimo/directed/service.py +509 -0
  143. apsimo/directives/__init__.py +25 -0
  144. apsimo/directives/evidence.py +87 -0
  145. apsimo/directives/extractor.py +188 -0
  146. apsimo/directives/guard.py +364 -0
  147. apsimo/directives/models.py +206 -0
  148. apsimo/directives/service.py +372 -0
  149. apsimo/directives/store.py +167 -0
  150. apsimo/doctor.py +2173 -0
  151. apsimo/environment.py +43 -0
  152. apsimo/events/__init__.py +33 -0
  153. apsimo/events/broadcaster.py +98 -0
  154. apsimo/events/bus.py +217 -0
  155. apsimo/events/journal.py +863 -0
  156. apsimo/events/stream.py +131 -0
  157. apsimo/events/types.py +150 -0
  158. apsimo/execution_results.py +357 -0
  159. apsimo/feedback/__init__.py +5 -0
  160. apsimo/feedback/store.py +76 -0
  161. apsimo/feeds/__init__.py +19 -0
  162. apsimo/feeds/cli.py +84 -0
  163. apsimo/feeds/engine.py +437 -0
  164. apsimo/feeds/example-feed.yaml +77 -0
  165. apsimo/feeds/hermes_cron.py +126 -0
  166. apsimo/feeds/manager.py +235 -0
  167. apsimo/feeds/spec.py +250 -0
  168. apsimo/feeds/template.py +202 -0
  169. apsimo/gate/__init__.py +18 -0
  170. apsimo/gate/audit.py +61 -0
  171. apsimo/gate/communication_policy.py +166 -0
  172. apsimo/gate/config.py +72 -0
  173. apsimo/gate/context_provenance.py +170 -0
  174. apsimo/gate/env_risk.py +226 -0
  175. apsimo/gate/guard_audit.py +353 -0
  176. apsimo/gate/layers/__init__.py +1 -0
  177. apsimo/gate/layers/base.py +15 -0
  178. apsimo/gate/layers/l1_recipient.py +66 -0
  179. apsimo/gate/layers/l2_pii.py +134 -0
  180. apsimo/gate/layers/l3_cross_context.py +50 -0
  181. apsimo/gate/layers/l4_trust_tier.py +78 -0
  182. apsimo/gate/layers/l5_injection.py +199 -0
  183. apsimo/gate/layers/l6_review.py +86 -0
  184. apsimo/gate/layers/l7_delay.py +100 -0
  185. apsimo/gate/layers/tom2_epistemic.py +185 -0
  186. apsimo/gate/models.py +64 -0
  187. apsimo/gate/pending_dispatch.py +5 -0
  188. apsimo/gate/pipeline.py +206 -0
  189. apsimo/gate/rejection.py +259 -0
  190. apsimo/gate/response_guard.py +700 -0
  191. apsimo/gate/rulesets/injection_v1.yaml +51 -0
  192. apsimo/gate/surface_policy.py +189 -0
  193. apsimo/gate/taint.py +226 -0
  194. apsimo/genesis.json +9 -0
  195. apsimo/goals/__init__.py +100 -0
  196. apsimo/goals/config.py +38 -0
  197. apsimo/goals/decomposer.py +421 -0
  198. apsimo/goals/engine.py +617 -0
  199. apsimo/goals/inference.py +354 -0
  200. apsimo/goals/models.py +302 -0
  201. apsimo/goals/priority.py +270 -0
  202. apsimo/goals/queue_bridge.py +149 -0
  203. apsimo/goals/replan.py +450 -0
  204. apsimo/goals/schema.sql +89 -0
  205. apsimo/goals/store.py +692 -0
  206. apsimo/governed_actions.py +1708 -0
  207. apsimo/harness_integration/__init__.py +45 -0
  208. apsimo/harness_integration/context.py +41 -0
  209. apsimo/harness_integration/skills.py +231 -0
  210. apsimo/identity/__init__.py +26 -0
  211. apsimo/identity/participants.py +181 -0
  212. apsimo/identity/resolver.py +329 -0
  213. apsimo/identity_bootstrap/__init__.py +5 -0
  214. apsimo/identity_bootstrap/builder.py +208 -0
  215. apsimo/identity_bootstrap/corpus.py +443 -0
  216. apsimo/identity_bootstrap/models.py +54 -0
  217. apsimo/identity_bootstrap/runner.py +353 -0
  218. apsimo/identity_bootstrap/seeders/__init__.py +25 -0
  219. apsimo/identity_bootstrap/seeders/briefings.py +109 -0
  220. apsimo/identity_bootstrap/seeders/chain.py +57 -0
  221. apsimo/identity_bootstrap/seeders/goals.py +128 -0
  222. apsimo/identity_bootstrap/seeders/memory.py +191 -0
  223. apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
  224. apsimo/identity_bootstrap/seeders/relationship.py +152 -0
  225. apsimo/identity_bootstrap/seeders/sessions.py +67 -0
  226. apsimo/identity_bootstrap/seeders/skills.py +92 -0
  227. apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
  228. apsimo/identity_bootstrap/seeders/world_model.py +143 -0
  229. apsimo/identity_bootstrap/self_query.py +92 -0
  230. apsimo/identity_bootstrap/self_reflection.py +155 -0
  231. apsimo/identity_bootstrap/skill.py +37 -0
  232. apsimo/identity_bootstrap/verifier.py +436 -0
  233. apsimo/initiatives/__init__.py +20 -0
  234. apsimo/initiatives/action_registry.py +454 -0
  235. apsimo/initiatives/approval_authority.py +2105 -0
  236. apsimo/initiatives/approval_policy.py +123 -0
  237. apsimo/initiatives/assignment.py +263 -0
  238. apsimo/initiatives/backup_evidence.py +100 -0
  239. apsimo/initiatives/context_freshness.py +103 -0
  240. apsimo/initiatives/models.py +318 -0
  241. apsimo/initiatives/native_work.py +270 -0
  242. apsimo/initiatives/standing_approvals.py +232 -0
  243. apsimo/initiatives/store.py +1081 -0
  244. apsimo/initiatives/temporal_followup.py +410 -0
  245. apsimo/intelligence/__init__.py +1 -0
  246. apsimo/intelligence/cognition/__init__.py +24 -0
  247. apsimo/intelligence/cognition/gap_detector.py +148 -0
  248. apsimo/intelligence/cognition/metalearner.py +547 -0
  249. apsimo/intelligence/cognition/metrics_collector.py +217 -0
  250. apsimo/intelligence/cognition/performance_index.py +299 -0
  251. apsimo/intelligence/cognition/registry.py +192 -0
  252. apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
  253. apsimo/intelligence/cognition/types.py +16 -0
  254. apsimo/intelligence/components/__init__.py +66 -0
  255. apsimo/intelligence/components/anomaly_detector.py +413 -0
  256. apsimo/intelligence/components/initiative_engine.py +2643 -0
  257. apsimo/intelligence/components/preference_learner.py +521 -0
  258. apsimo/intelligence/components/research_orchestrator.py +358 -0
  259. apsimo/intelligence/components/self_directed_thinker.py +221 -0
  260. apsimo/intelligence/components/self_reflector.py +252 -0
  261. apsimo/intelligence/components/session_continuity.py +154 -0
  262. apsimo/intelligence/components/task_planner.py +320 -0
  263. apsimo/intelligence/components/tool_learner.py +217 -0
  264. apsimo/intelligence/graph/__init__.py +79 -0
  265. apsimo/intelligence/graph/client.py +2483 -0
  266. apsimo/intelligence/graph/consolidator.py +405 -0
  267. apsimo/intelligence/graph/distiller.py +312 -0
  268. apsimo/intelligence/graph/migrations.py +129 -0
  269. apsimo/intelligence/graph/queries.py +248 -0
  270. apsimo/intelligence/graph/recall.py +281 -0
  271. apsimo/intelligence/graph/reconciler.py +144 -0
  272. apsimo/intelligence/graph/schema.py +337 -0
  273. apsimo/intelligence/graph/selection.py +252 -0
  274. apsimo/intelligence/learning/__init__.py +17 -0
  275. apsimo/intelligence/learning/continuous_learner.py +245 -0
  276. apsimo/intelligence/learning/feedback_store.py +321 -0
  277. apsimo/intelligence/mind_model/__init__.py +1 -0
  278. apsimo/intelligence/mind_model/graph_baseline.py +136 -0
  279. apsimo/intelligence/mind_model/signal_collector.py +361 -0
  280. apsimo/intelligence/relationships/__init__.py +11 -0
  281. apsimo/intelligence/relationships/profiler.py +389 -0
  282. apsimo/intelligence/relationships/scorer.py +560 -0
  283. apsimo/intelligence/relationships/signal_floor.py +66 -0
  284. apsimo/intelligence/relationships/trust_tiers.py +300 -0
  285. apsimo/intelligence/synthesis/__init__.py +40 -0
  286. apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
  287. apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
  288. apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
  289. apsimo/intelligence/synthesis/insight_store.py +79 -0
  290. apsimo/intelligence/synthesis/insight_validator.py +183 -0
  291. apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
  292. apsimo/intelligence/turn_middleware/__init__.py +15 -0
  293. apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
  294. apsimo/mcp/__init__.py +41 -0
  295. apsimo/mcp/__main__.py +6 -0
  296. apsimo/mcp/config.py +287 -0
  297. apsimo/mcp/server.py +501 -0
  298. apsimo/migrations.py +187 -0
  299. apsimo/mining/__init__.py +27 -0
  300. apsimo/mining/corpus.py +239 -0
  301. apsimo/mining/escalations.py +289 -0
  302. apsimo/mining/models.py +169 -0
  303. apsimo/mining/store.py +210 -0
  304. apsimo/models/__init__.py +30 -0
  305. apsimo/models/memory.py +80 -0
  306. apsimo/models/mesh.py +72 -0
  307. apsimo/models/person.py +104 -0
  308. apsimo/models/signal.py +108 -0
  309. apsimo/observations/__init__.py +15 -0
  310. apsimo/observations/store.py +277 -0
  311. apsimo/patterns/__init__.py +6 -0
  312. apsimo/patterns/extract.py +187 -0
  313. apsimo/patterns/store.py +227 -0
  314. apsimo/persona/__init__.py +1 -0
  315. apsimo/persona/engine.py +611 -0
  316. apsimo/persona/manifest.py +140 -0
  317. apsimo/projects/__init__.py +28 -0
  318. apsimo/projects/engine.py +1681 -0
  319. apsimo/projects/event_outbox.py +188 -0
  320. apsimo/projects/models.py +216 -0
  321. apsimo/projects/planner.py +181 -0
  322. apsimo/projects/store.py +1446 -0
  323. apsimo/proposals/__init__.py +12 -0
  324. apsimo/proposals/engine.py +114 -0
  325. apsimo/proposals/models.py +207 -0
  326. apsimo/qualification/__init__.py +1 -0
  327. apsimo/qualification/cases.py +75 -0
  328. apsimo/qualification/cli.py +51 -0
  329. apsimo/qualification/memory_cases.py +209 -0
  330. apsimo/qualification/records.py +92 -0
  331. apsimo/qualification/report.py +87 -0
  332. apsimo/qualification/runner.py +311 -0
  333. apsimo/qualification/structured_cases.py +131 -0
  334. apsimo/reasoning/__init__.py +13 -0
  335. apsimo/reasoning/executor.py +506 -0
  336. apsimo/reasoning/loop.py +373 -0
  337. apsimo/reasoning/native_tools/__init__.py +16 -0
  338. apsimo/reasoning/native_tools/calculate.py +141 -0
  339. apsimo/reasoning/native_tools/file_ops.py +150 -0
  340. apsimo/reasoning/native_tools/web_search.py +49 -0
  341. apsimo/reasoning/tool_policy.py +182 -0
  342. apsimo/redact/__init__.py +176 -0
  343. apsimo/repos/__init__.py +5 -0
  344. apsimo/repos/mirrors.py +204 -0
  345. apsimo/research/__init__.py +41 -0
  346. apsimo/research/artifact.py +482 -0
  347. apsimo/research/gatherer.py +387 -0
  348. apsimo/research/pipeline.py +513 -0
  349. apsimo/research/search/__init__.py +7 -0
  350. apsimo/research/search/base.py +41 -0
  351. apsimo/research/search/brave.py +59 -0
  352. apsimo/research/search/cache.py +51 -0
  353. apsimo/research/search/duckduckgo.py +103 -0
  354. apsimo/research/search/orchestrator.py +119 -0
  355. apsimo/research/search/serpapi.py +59 -0
  356. apsimo/research/search/tavily.py +59 -0
  357. apsimo/research/synthesizer.py +309 -0
  358. apsimo/router/__init__.py +30 -0
  359. apsimo/router/complexity_scorer.py +148 -0
  360. apsimo/router/endpoints.py +153 -0
  361. apsimo/router/fallback.py +58 -0
  362. apsimo/router/functions.py +243 -0
  363. apsimo/router/native_policy.py +52 -0
  364. apsimo/router/router.py +762 -0
  365. apsimo/router/self_learning.py +174 -0
  366. apsimo/router/tiers.py +677 -0
  367. apsimo/sandbox/__init__.py +21 -0
  368. apsimo/sandbox/backend.py +195 -0
  369. apsimo/sandbox/manager.py +173 -0
  370. apsimo/scope_bounds.py +7 -0
  371. apsimo/secrets/__init__.py +6 -0
  372. apsimo/secrets/backends/__init__.py +8 -0
  373. apsimo/secrets/backends/base.py +42 -0
  374. apsimo/secrets/backends/env.py +110 -0
  375. apsimo/secrets/backends/keyring.py +72 -0
  376. apsimo/secrets/backends/onepassword.py +232 -0
  377. apsimo/secrets/cli.py +191 -0
  378. apsimo/secrets/manager.py +160 -0
  379. apsimo/secrets/migration.py +101 -0
  380. apsimo/secrets/types.py +98 -0
  381. apsimo/seed.py +41 -0
  382. apsimo/self_model/__init__.py +37 -0
  383. apsimo/self_model/appraisals.py +673 -0
  384. apsimo/self_model/benchmark.py +1314 -0
  385. apsimo/self_model/brief.py +40 -0
  386. apsimo/self_model/event_concerns.py +1128 -0
  387. apsimo/self_model/execution_forecasts.py +353 -0
  388. apsimo/self_model/expectations.py +1595 -0
  389. apsimo/self_model/experiments.py +1150 -0
  390. apsimo/self_model/journal.py +148 -0
  391. apsimo/self_model/judgments.py +705 -0
  392. apsimo/self_model/native_outcomes.py +55 -0
  393. apsimo/self_model/params.py +220 -0
  394. apsimo/self_model/perspective.py +246 -0
  395. apsimo/self_model/reconcile.py +183 -0
  396. apsimo/self_model/reply_forecasts.py +381 -0
  397. apsimo/self_model/runtime_forecasts.py +296 -0
  398. apsimo/self_model/runtime_models.py +67 -0
  399. apsimo/self_model/settlement.py +207 -0
  400. apsimo/self_model/situation.py +1731 -0
  401. apsimo/self_model/store.py +883 -0
  402. apsimo/self_model/supervised.py +137 -0
  403. apsimo/self_model/thinker.py +99 -0
  404. apsimo/self_model/trust.py +388 -0
  405. apsimo/self_model/workspace.py +2388 -0
  406. apsimo/server.py +4197 -0
  407. apsimo/services/__init__.py +1 -0
  408. apsimo/services/agent_bridge.py +474 -0
  409. apsimo/services/initiative_executor.py +914 -0
  410. apsimo/services/instance.py +297 -0
  411. apsimo/sessions/__init__.py +22 -0
  412. apsimo/sessions/config.py +13 -0
  413. apsimo/sessions/context_loader.py +88 -0
  414. apsimo/sessions/federation_session.py +75 -0
  415. apsimo/sessions/isolated_session.py +98 -0
  416. apsimo/sessions/reports.py +84 -0
  417. apsimo/sessions/store.py +148 -0
  418. apsimo/setup.py +2818 -0
  419. apsimo/setup_hermes.py +879 -0
  420. apsimo/setup_local_work.py +218 -0
  421. apsimo/setup_native_goals.py +134 -0
  422. apsimo/setup_native_reviews.py +115 -0
  423. apsimo/skills/__init__.py +10 -0
  424. apsimo/skills/base.py +108 -0
  425. apsimo/skills/budget.py +28 -0
  426. apsimo/skills/executor.py +493 -0
  427. apsimo/skills/executors/__init__.py +1 -0
  428. apsimo/skills/executors/behavioral_correction.py +75 -0
  429. apsimo/skills/executors/capability_gap.py +38 -0
  430. apsimo/skills/executors/data_quality.py +163 -0
  431. apsimo/skills/executors/knowledge_acquisition.py +41 -0
  432. apsimo/skills/executors/operational_hygiene.py +185 -0
  433. apsimo/skills/executors/subsystem_health.py +169 -0
  434. apsimo/skills/hermes_export.py +431 -0
  435. apsimo/skills/index.py +123 -0
  436. apsimo/skills/learning/__init__.py +21 -0
  437. apsimo/skills/learning/novelty_detector.py +206 -0
  438. apsimo/skills/learning/pattern_extractor.py +199 -0
  439. apsimo/skills/learning/triggers.py +159 -0
  440. apsimo/skills/loader.py +246 -0
  441. apsimo/skills/migrations/002_progressive_loading.sql +6 -0
  442. apsimo/skills/migrations/backfill_triggers.py +20 -0
  443. apsimo/skills/models.py +202 -0
  444. apsimo/skills/packager.py +128 -0
  445. apsimo/skills/protocols.py +70 -0
  446. apsimo/skills/registry.py +191 -0
  447. apsimo/skills/runtime.py +58 -0
  448. apsimo/skills/sandbox_runner.py +229 -0
  449. apsimo/skills/scheduler.py +129 -0
  450. apsimo/skills/schema.py +79 -0
  451. apsimo/skills/security/__init__.py +12 -0
  452. apsimo/skills/security/guards.py +53 -0
  453. apsimo/skills/security/scanner.py +223 -0
  454. apsimo/skills_memory/__init__.py +26 -0
  455. apsimo/skills_memory/distill.py +159 -0
  456. apsimo/skills_memory/models.py +85 -0
  457. apsimo/skills_memory/retrieve.py +62 -0
  458. apsimo/skills_memory/store.py +172 -0
  459. apsimo/surprise/__init__.py +6 -0
  460. apsimo/surprise/accumulation.py +57 -0
  461. apsimo/surprise/scorer.py +102 -0
  462. apsimo/surprise/store.py +203 -0
  463. apsimo/task_queue/__init__.py +69 -0
  464. apsimo/task_queue/action_receipts.py +148 -0
  465. apsimo/task_queue/approval_relay_canary.py +108 -0
  466. apsimo/task_queue/config.py +85 -0
  467. apsimo/task_queue/contract.py +361 -0
  468. apsimo/task_queue/events.py +130 -0
  469. apsimo/task_queue/governor.py +1031 -0
  470. apsimo/task_queue/handlers/__init__.py +16 -0
  471. apsimo/task_queue/handlers/base.py +37 -0
  472. apsimo/task_queue/handlers/inference.py +640 -0
  473. apsimo/task_queue/handlers/monitoring.py +116 -0
  474. apsimo/task_queue/handlers/registry.py +75 -0
  475. apsimo/task_queue/handlers/subtask_handler.py +173 -0
  476. apsimo/task_queue/handlers/system_maintenance.py +147 -0
  477. apsimo/task_queue/mesh_integration.py +111 -0
  478. apsimo/task_queue/models.py +317 -0
  479. apsimo/task_queue/queue_manager.py +8286 -0
  480. apsimo/task_queue/routing.py +287 -0
  481. apsimo/task_queue/scheduler.py +252 -0
  482. apsimo/task_queue/schema.sql +197 -0
  483. apsimo/task_queue/work_control.py +342 -0
  484. apsimo/task_queue/worker.py +993 -0
  485. apsimo/telemetry.py +145 -0
  486. apsimo/tom/__init__.py +6 -0
  487. apsimo/tom/affect.py +387 -0
  488. apsimo/tom/approvals.py +171 -0
  489. apsimo/tom/arcs.py +896 -0
  490. apsimo/tom/asymmetry.py +131 -0
  491. apsimo/tom/eligibility.py +248 -0
  492. apsimo/tom/engagement.py +214 -0
  493. apsimo/tom/exposure.py +214 -0
  494. apsimo/tom/extractor.py +306 -0
  495. apsimo/tom/fact_adapters.py +144 -0
  496. apsimo/tom/facts.py +326 -0
  497. apsimo/tom/integration.py +592 -0
  498. apsimo/tom/leveled.py +118 -0
  499. apsimo/tom/levels.py +247 -0
  500. apsimo/tom/recipient_audit.py +995 -0
  501. apsimo/tom/recipient_simulator.py +593 -0
  502. apsimo/tom/source_lineage.py +93 -0
  503. apsimo/tom/tom2.py +277 -0
  504. apsimo/tom/visibility.py +559 -0
  505. apsimo/tom/visibility_store.py +414 -0
  506. apsimo/tools/__init__.py +0 -0
  507. apsimo/tools/definitions.py +740 -0
  508. apsimo/tools/handlers.py +943 -0
  509. apsimo/toolsmith/__init__.py +26 -0
  510. apsimo/toolsmith/authority.py +166 -0
  511. apsimo/toolsmith/engine.py +559 -0
  512. apsimo/toolsmith/integrity.py +100 -0
  513. apsimo/toolsmith/miner.py +145 -0
  514. apsimo/toolsmith/policy.py +110 -0
  515. apsimo/toolsmith/registry.py +635 -0
  516. apsimo/turns/__init__.py +17 -0
  517. apsimo/turns/audio.py +134 -0
  518. apsimo/turns/documents.py +235 -0
  519. apsimo/turns/executions.py +486 -0
  520. apsimo/turns/hermes_history.py +245 -0
  521. apsimo/turns/hermes_kanban.py +268 -0
  522. apsimo/turns/hermes_work.py +96 -0
  523. apsimo/turns/idempotency.py +752 -0
  524. apsimo/turns/local_work.py +115 -0
  525. apsimo/turns/media.py +581 -0
  526. apsimo/turns/reported_workers.py +196 -0
  527. apsimo/turns/source_annotations.py +283 -0
  528. apsimo/turns/source_attribution.py +154 -0
  529. apsimo/turns/source_read.py +351 -0
  530. apsimo/turns/source_vectors.py +263 -0
  531. apsimo/turns/video.py +210 -0
  532. apsimo/util/autonomy_preset.py +220 -0
  533. apsimo/util/instance.py +92 -0
  534. apsimo/util/model_output.py +25 -0
  535. apsimo/util/quiet_hours.py +27 -0
  536. apsimo/util/session_safety.py +37 -0
  537. apsimo/util/temporal.py +343 -0
  538. apsimo/vector/__init__.py +75 -0
  539. apsimo/vector/backfill.py +171 -0
  540. apsimo/vector/caption.py +114 -0
  541. apsimo/vector/collections.py +51 -0
  542. apsimo/vector/config.py +102 -0
  543. apsimo/vector/embedder.py +670 -0
  544. apsimo/vector/image_preprocess.py +406 -0
  545. apsimo/vector/image_store.py +296 -0
  546. apsimo/vector/indexes.py +162 -0
  547. apsimo/vector/migrate.py +334 -0
  548. apsimo/vector/multimodal_provider.py +417 -0
  549. apsimo/vector/multimodal_types.py +87 -0
  550. apsimo/vector/openai_provider.py +119 -0
  551. apsimo/vector/query.py +49 -0
  552. apsimo/vector/reranker.py +565 -0
  553. apsimo/vector/safety_image.py +159 -0
  554. apsimo/vector/scanner.py +197 -0
  555. apsimo/vector/setup.py +289 -0
  556. apsimo/vector/store.py +533 -0
  557. apsimo/vector/tiers.py +263 -0
  558. apsimo/work_orders.py +925 -0
  559. apsimo/workers/__init__.py +21 -0
  560. apsimo/workers/agent_bridge.py +640 -0
  561. apsimo/workers/colony_worker.py +382 -0
  562. apsimo/workers/queue_worker.py +441 -0
  563. apsimo/workers/skills_sync.py +152 -0
  564. apsimo/world_model/__init__.py +71 -0
  565. apsimo/world_model/causal_maintenance.py +131 -0
  566. apsimo/world_model/causal_policy.py +43 -0
  567. apsimo/world_model/causal_query.py +125 -0
  568. apsimo/world_model/confidence.py +54 -0
  569. apsimo/world_model/config.py +64 -0
  570. apsimo/world_model/constants.py +97 -0
  571. apsimo/world_model/entities.py +145 -0
  572. apsimo/world_model/expectation_resolvers.py +177 -0
  573. apsimo/world_model/extraction/__init__.py +7 -0
  574. apsimo/world_model/extraction/base.py +62 -0
  575. apsimo/world_model/extraction/conversation_extractor.py +262 -0
  576. apsimo/world_model/extraction/detector.py +74 -0
  577. apsimo/world_model/extraction/document_extractor.py +78 -0
  578. apsimo/world_model/extraction/formats/__init__.py +24 -0
  579. apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
  580. apsimo/world_model/extraction/formats/html_fmt.py +72 -0
  581. apsimo/world_model/extraction/formats/json_fmt.py +68 -0
  582. apsimo/world_model/extraction/formats/pdf.py +43 -0
  583. apsimo/world_model/extraction/formats/text.py +27 -0
  584. apsimo/world_model/extraction/llm_extractor.py +164 -0
  585. apsimo/world_model/extraction/pipeline.py +73 -0
  586. apsimo/world_model/integrations/__init__.py +5 -0
  587. apsimo/world_model/integrations/mind_model_bridge.py +115 -0
  588. apsimo/world_model/integrations/social_intel_bridge.py +120 -0
  589. apsimo/world_model/jobs/__init__.py +4 -0
  590. apsimo/world_model/jobs/extraction_job.py +168 -0
  591. apsimo/world_model/llm_extract.py +572 -0
  592. apsimo/world_model/neo4j/__init__.py +5 -0
  593. apsimo/world_model/neo4j/backend.py +654 -0
  594. apsimo/world_model/observations.py +155 -0
  595. apsimo/world_model/populator.py +307 -0
  596. apsimo/world_model/postgres/__init__.py +1 -0
  597. apsimo/world_model/postgres/backend.py +683 -0
  598. apsimo/world_model/relationships.py +25 -0
  599. apsimo/world_model/resolution/__init__.py +13 -0
  600. apsimo/world_model/resolution/entity_resolver.py +232 -0
  601. apsimo/world_model/resolution/merge_audit.py +16 -0
  602. apsimo/world_model/resolution/merge_workflow.py +117 -0
  603. apsimo/world_model/source_reports.py +121 -0
  604. apsimo/world_model/sqlite/__init__.py +4 -0
  605. apsimo/world_model/sqlite/backend.py +855 -0
  606. apsimo/world_model/sqlite/schema.sql +132 -0
  607. apsimo/world_model/store.py +545 -0
  608. apsimo-1.3.0.dist-info/METADATA +78 -0
  609. apsimo-1.3.0.dist-info/RECORD +614 -0
  610. apsimo-1.3.0.dist-info/WHEEL +5 -0
  611. apsimo-1.3.0.dist-info/entry_points.txt +11 -0
  612. apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
  613. apsimo-1.3.0.dist-info/top_level.txt +2 -0
  614. colony_sidecar/__init__.py +4 -0
@@ -0,0 +1,2483 @@
1
+ """Colony Graph Memory System — Neo4j async client.
2
+
3
+ Replaces Hermes MEMORY.md with a persistent graph database that models
4
+ relationships, events, and behavioral patterns.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import ast
10
+ import asyncio
11
+ import hashlib
12
+ import json
13
+ import logging
14
+ import math
15
+ import os
16
+ import time
17
+ from collections import deque
18
+ from dataclasses import dataclass, field
19
+ from enum import Enum
20
+ from typing import Any, Callable, Coroutine, Dict, List, Optional, TYPE_CHECKING
21
+
22
+ try:
23
+ from neo4j import AsyncGraphDatabase, AsyncDriver
24
+ except ImportError:
25
+ pass
26
+ from pydantic import SecretStr
27
+
28
+ if TYPE_CHECKING:
29
+ from apsimo.vector.store import VectorStore
30
+
31
+ logger = logging.getLogger(__name__)
32
+
33
+
34
+ def _recency_factor(days_old: float) -> float:
35
+ """Recency weight for retrieval ranking (v0.21.0).
36
+
37
+ Configurable exponential half-life with a floor, so recent memories surface
38
+ over stale ones without fully suppressing older context. Defaults:
39
+ half-life 90d, floor 0.5 (a year-old memory keeps ~0.53 weight; fresh = 1.0).
40
+ Set COLONY_RECENCY_HALF_LIFE_DAYS<=0 to disable. The previous behaviour was a
41
+ near-flat ~10%/year discount that barely affected ranking.
42
+ """
43
+ import os
44
+ try:
45
+ half_life = float(os.environ.get("COLONY_RECENCY_HALF_LIFE_DAYS", "90"))
46
+ except (ValueError, TypeError):
47
+ half_life = 90.0
48
+ if half_life <= 0:
49
+ return 1.0
50
+ try:
51
+ floor = float(os.environ.get("COLONY_RECENCY_FLOOR", "0.5"))
52
+ except (ValueError, TypeError):
53
+ floor = 0.5
54
+ floor = min(max(floor, 0.0), 1.0)
55
+ return floor + (1.0 - floor) * (0.5 ** (max(days_old, 0.0) / half_life))
56
+
57
+
58
+ def distill_turn_summary(summary: str) -> str:
59
+ """The distilled form of a turn summary (what COLONY_DISTILL_TURNS=1 stores).
60
+
61
+ PRESERVES BOTH speaker labels ("User:" and "Agent:", with "assistant"
62
+ normalized to "Agent:"). Attribution matters in both directions:
63
+ "User: my X is Y" says whose preference it is, and "Agent: ..." marks
64
+ the agent's own prose so downstream consumers (e.g. the belief claim
65
+ extractor) never mistake something the agent SAID for a fact a contact
66
+ ASSERTED. Lines are joined with "; " — this text is injected into
67
+ prompts, so it must never introduce an em dash.
68
+ """
69
+ _lines = []
70
+ for ln in (summary or "").splitlines():
71
+ if ":" in ln:
72
+ _speaker, _rest = ln.split(":", 1)
73
+ _rest = _rest.strip()
74
+ sp = _speaker.strip().lower()
75
+ if sp == "user" and _rest:
76
+ _lines.append(f"User: {_rest}")
77
+ elif sp in ("agent", "assistant") and _rest:
78
+ _lines.append(f"Agent: {_rest}")
79
+ else:
80
+ _lines.append(_rest)
81
+ else:
82
+ _lines.append(ln)
83
+ return "; ".join(x for x in _lines if x) or (summary or "")
84
+
85
+
86
+ @dataclass
87
+ class GraphConfig:
88
+ """Connection settings for the Neo4j graph database."""
89
+
90
+ uri: str = "bolt://localhost:7687"
91
+ database: str = "colony"
92
+ auth: Optional[tuple[str, SecretStr]] = None # (user, password) — password masked in logs
93
+ max_pool_size: int = 50
94
+ connection_timeout_secs: float = 10.0
95
+ max_retry_secs: float = 30.0
96
+
97
+
98
+ class MemorySourceType(str, Enum):
99
+ CONVERSATION = "conversation"
100
+ FILE = "file"
101
+ TOOL_OUTPUT = "tool_output"
102
+ USER_ASSERTION = "user_assertion"
103
+ INFERENCE = "inference"
104
+
105
+
106
+ class EpistemicState(str, Enum):
107
+ INFERRED = "inferred"
108
+ OBSERVED = "observed"
109
+ CORROBORATED = "corroborated"
110
+ VERIFIED = "verified"
111
+ STALE = "stale"
112
+ SUPERSEDED = "superseded"
113
+ DEPRECATED = "deprecated"
114
+ ARCHIVED = "archived"
115
+
116
+
117
+ SOURCE_RELIABILITY: Dict[str, float] = {
118
+ MemorySourceType.USER_ASSERTION: 1.0,
119
+ MemorySourceType.FILE: 0.9,
120
+ MemorySourceType.TOOL_OUTPUT: 0.85,
121
+ MemorySourceType.CONVERSATION: 0.7,
122
+ MemorySourceType.INFERENCE: 0.5,
123
+ }
124
+
125
+ MAX_IMPORTANCE: Dict[str, float] = {
126
+ MemorySourceType.USER_ASSERTION: 1.0,
127
+ MemorySourceType.FILE: 0.95,
128
+ MemorySourceType.TOOL_OUTPUT: 0.9,
129
+ MemorySourceType.CONVERSATION: 0.8,
130
+ MemorySourceType.INFERENCE: 0.7,
131
+ }
132
+
133
+
134
+ class ColonyGraph:
135
+ """Neo4j graph memory system replacing Hermes MEMORY.md.
136
+
137
+ Provides:
138
+ - Memory storage with automatic entity linking
139
+ - Semantic recall via vector index + strength decay
140
+ - Ebbinghaus forgetting‐curve decay
141
+ - Pruning of weak / stale memories
142
+ - Multi-hop traversal across memory connections
143
+ """
144
+
145
+ def __init__(self, config: GraphConfig) -> None:
146
+ self._config = config
147
+ driver_auth = (
148
+ (config.auth[0], config.auth[1].get_secret_value())
149
+ if config.auth is not None
150
+ else None
151
+ )
152
+ self.driver: AsyncDriver = AsyncGraphDatabase.driver(
153
+ config.uri,
154
+ auth=driver_auth,
155
+ max_connection_pool_size=config.max_pool_size,
156
+ connection_timeout=config.connection_timeout_secs,
157
+ max_transaction_retry_time=config.max_retry_secs,
158
+ keep_alive=True,
159
+ # Suppress benign DBMS notifications (unknown-label/property warnings
160
+ # for nodes not yet created) — pure log noise, not errors.
161
+ notifications_min_severity="OFF",
162
+ )
163
+ self.database: str = config.database
164
+ self._embed_fn: Optional[
165
+ Callable[[str], Coroutine[Any, Any, List[float]]]
166
+ ] = None
167
+ self._vector_store: Optional["VectorStore"] = None
168
+ self._rerank_fn: Optional[Callable[..., Coroutine[Any, Any, Any]]] = None
169
+ # Runtime-wide hard exclusions protect untyped recall consumers (for
170
+ # example model tools and background synthesis) that cannot safely
171
+ # render a governed source directly. Empty preserves legacy behavior.
172
+ self._recall_source_exclusions: tuple[str, ...] = ()
173
+ self._recall_metadata_exclusions: tuple[str, ...] = ()
174
+
175
+ @staticmethod
176
+ def _bounded_source_uris(source_uris) -> tuple[str, ...]:
177
+ """Normalize a small exact-match source-URI exclusion set."""
178
+ return tuple(sorted(dict.fromkeys(
179
+ str(value).strip() for value in (source_uris or ())
180
+ if 0 < len(str(value).strip()) <= 256
181
+ )))[:16]
182
+
183
+ def set_recall_source_exclusions(
184
+ self,
185
+ source_uris,
186
+ *,
187
+ legacy_metadata_markers=(),
188
+ ) -> None:
189
+ """Set graph-wide hard recall exclusions for runtime policy.
190
+
191
+ Per-call exclusions are additive and can never relax this boundary.
192
+ ``legacy_metadata_markers`` covers rows written before a governed
193
+ source URI existed. Both sets are bounded exact strings; passing empty
194
+ iterables restores the historical default behavior.
195
+ """
196
+ self._recall_source_exclusions = self._bounded_source_uris(source_uris)
197
+ self._recall_metadata_exclusions = self._bounded_source_uris(
198
+ legacy_metadata_markers)
199
+
200
+ # ------------------------------------------------------------------
201
+ # Lifecycle
202
+ # ------------------------------------------------------------------
203
+
204
+ async def connect(self) -> None:
205
+ """Verify connectivity to Neo4j."""
206
+ await self.driver.verify_connectivity()
207
+
208
+ async def ensure_colony_self(self) -> None:
209
+ """Ensure Colony's self-representation exists in the graph (v0.11.0).
210
+
211
+ Creates an Agent node for Colony with DEPENDS_ON edges to
212
+ known Subsystem nodes. Idempotent — safe to call multiple times.
213
+ """
214
+ try:
215
+ async with self.driver.session(database=self.database) as session:
216
+ # Create Agent node
217
+ await session.run("""
218
+ MERGE (a:Agent {id: 'colony-sidecar'})
219
+ SET a.name = 'Colony',
220
+ a.version = '0.11.1',
221
+ a.status = 'active',
222
+ a.created_at = coalesce(a.created_at, datetime())
223
+ """)
224
+
225
+ # Create Subsystem nodes for known components
226
+ subsystems = [
227
+ ("embed_pipeline", "Embedding Pipeline"),
228
+ ("delivery_bridge", "Delivery Bridge"),
229
+ ("event_bus", "Event Bus"),
230
+ ("graph_client", "Graph Client"),
231
+ ("initiative_engine", "Initiative Engine"),
232
+ ("mind_model", "Mind Model"),
233
+ ]
234
+ for sub_id, sub_name in subsystems:
235
+ await session.run("""
236
+ MERGE (s:Subsystem {id: $id})
237
+ SET s.name = $name,
238
+ s.status = coalesce(s.status, 'active'),
239
+ s.created_at = coalesce(s.created_at, datetime())
240
+ """, id=sub_id, name=sub_name)
241
+
242
+ # Create DEPENDS_ON edge from Agent to Subsystem
243
+ await session.run("""
244
+ MATCH (a:Agent {id: 'colony-sidecar'}), (s:Subsystem {id: $id})
245
+ MERGE (a)-[r:DEPENDS_ON]->(s)
246
+ SET r.created_at = coalesce(r.created_at, datetime())
247
+ """, id=sub_id)
248
+
249
+ logger.info("Colony self-representation verified in graph")
250
+ except Exception as e:
251
+ logger.warning("Failed to ensure Colony self-representation: %s", e)
252
+
253
+ async def close(self) -> None:
254
+ """Cleanly shut down the driver."""
255
+ await self.driver.close()
256
+
257
+ # ------------------------------------------------------------------
258
+ # Embedding helper
259
+ # ------------------------------------------------------------------
260
+
261
+ def set_embed_fn(
262
+ self,
263
+ fn: Callable[[str], Coroutine[Any, Any, List[float]]],
264
+ ) -> None:
265
+ """Register an async embedding function used by *recall*.
266
+
267
+ Args:
268
+ fn: async callable that maps a string to a float vector.
269
+ """
270
+ self._embed_fn = fn
271
+ owner = getattr(fn, '__self__', None)
272
+ self._embed_query_fn = getattr(owner, 'embed_query', None)
273
+
274
+ def set_vector_store(self, store: "VectorStore") -> None:
275
+ """Register a VectorStore for ANN search (replaces Neo4j vector index)."""
276
+ self._vector_store = store
277
+
278
+ def set_rerank_fn(
279
+ self,
280
+ fn: Callable[..., Coroutine[Any, Any, Any]],
281
+ *,
282
+ calibration_metadata: Optional[Callable[[], Dict[str, Any]]] = None,
283
+ ) -> None:
284
+ """Register an async rerank function used by *recall* (mirrors
285
+ :meth:`set_embed_fn`).
286
+
287
+ Args:
288
+ fn: async callable ``fn(query, documents, top_k=N)`` returning a
289
+ list of objects with ``index`` and ``score`` attributes (the
290
+ RerankerProvider.rerank contract). Only consulted when
291
+ COLONY_RECALL_RERANK is ``shadow`` or ``on``.
292
+ calibration_metadata: current provider/model/format configuration.
293
+ Optional calibrated abstention is invalidated when this changes.
294
+ """
295
+ self._rerank_fn = fn
296
+ self._rerank_calibration_metadata = calibration_metadata
297
+ self._recall_selector = None
298
+
299
+ def set_adaptive_params(self, params: Any) -> None:
300
+ """Register an AdaptiveParamStore consulted by *recall* for the
301
+ meta-learned relevance floor (recall.min_relevance)."""
302
+ self._adaptive_params = params
303
+
304
+ async def _embed(self, text: str) -> List[float]:
305
+ """Produce an embedding vector for *text* (document side).
306
+
307
+ Raises:
308
+ RuntimeError: If no embedding function has been registered.
309
+ """
310
+ if self._embed_fn is None:
311
+ raise RuntimeError(
312
+ "No embedding function registered. Call set_embed_fn() first."
313
+ )
314
+ return await self._embed_fn(text)
315
+
316
+ async def _embed_query(self, query: str) -> List[float]:
317
+ """Embed a *query* for retrieval (v0.21.1).
318
+
319
+ Instruct-tuned embedders (Qwen3-Embedding, E5, BGE, …) are ASYMMETRIC:
320
+ the query gets an instruction prefix while documents do not. Without it,
321
+ retrieval quality collapses (every result lands at ~0.9 cosine distance
322
+ and the right memory never surfaces). Configurable via
323
+ COLONY_EMBED_QUERY_INSTRUCTION (set to empty for symmetric models).
324
+ """
325
+ if callable(getattr(self, '_embed_query_fn', None)):
326
+ return await self._embed_query_fn(query)
327
+ import os
328
+ default_instr = ("Instruct: Given a search query, retrieve relevant "
329
+ "memories that answer it\nQuery: ")
330
+ instr = os.environ.get("COLONY_EMBED_QUERY_INSTRUCTION", default_instr)
331
+ return await self._embed((instr + query) if instr else query)
332
+
333
+ @staticmethod
334
+ def compute_effective_confidence(
335
+ base_confidence: float,
336
+ source_reliability: float,
337
+ corroboration_count: int,
338
+ contradiction_count: int,
339
+ recalls: int,
340
+ last_verified_at: Optional[Any],
341
+ created_at: Any,
342
+ epistemic_state: str,
343
+ now: Any,
344
+ ) -> float:
345
+ """Compute effective confidence from multiple signals.
346
+
347
+ Args:
348
+ base_confidence: Initial confidence (0-1)
349
+ source_reliability: Reliability of the source (0-1)
350
+ corroboration_count: Number of corroborating memories
351
+ contradiction_count: Number of contradicting memories
352
+ recalls: Number of times recalled
353
+ last_verified_at: Last verification timestamp or None
354
+ created_at: Creation timestamp
355
+ epistemic_state: Current epistemic state string
356
+ now: Current timestamp
357
+
358
+ Returns:
359
+ Effective confidence in [0, 1].
360
+ """
361
+ from datetime import datetime as _dt, timezone as _tz
362
+
363
+ if now is None:
364
+ now = _dt.now(_tz.utc)
365
+ if hasattr(now, "to_native"):
366
+ now = now.to_native()
367
+ if isinstance(now, str):
368
+ now = _dt.fromisoformat(now.replace("Z", "+00:00"))
369
+ if hasattr(created_at, "to_native"):
370
+ created_at = created_at.to_native()
371
+ if isinstance(created_at, str):
372
+ created_at = _dt.fromisoformat(created_at.replace("Z", "+00:00"))
373
+
374
+ # Source weight
375
+ confidence = base_confidence * source_reliability
376
+
377
+ # Corroboration / contradiction adjustment
378
+ net_support = corroboration_count - contradiction_count
379
+ confidence *= min(1.0, 1.0 + net_support * 0.1)
380
+
381
+ # Recall reinforcement (diminishing returns)
382
+ confidence *= min(1.3, 1.0 + recalls * 0.03)
383
+
384
+ # Recency weighting (v0.21.0, configurable half-life + floor — see
385
+ # _recency_factor). Recent memories surface; VERIFIED memories are
386
+ # additionally floored at 0.9 by the epistemic-state clamp below.
387
+ days_old = max(0, (now - created_at).days)
388
+ recency_factor = _recency_factor(days_old)
389
+ confidence *= recency_factor
390
+
391
+ # Verification boost
392
+ if last_verified_at:
393
+ if hasattr(last_verified_at, "to_native"):
394
+ last_verified_at = last_verified_at.to_native()
395
+ if isinstance(last_verified_at, str):
396
+ last_verified_at = _dt.fromisoformat(last_verified_at.replace("Z", "+00:00"))
397
+ if (now - last_verified_at).days < 7:
398
+ confidence *= 1.2
399
+
400
+ # State clamp
401
+ if epistemic_state == EpistemicState.VERIFIED.value:
402
+ confidence = max(confidence, 0.9)
403
+ elif epistemic_state in (EpistemicState.STALE.value, EpistemicState.SUPERSEDED.value):
404
+ confidence *= 0.3
405
+ elif epistemic_state == EpistemicState.DEPRECATED.value:
406
+ confidence *= 0.1
407
+
408
+ return min(1.0, max(0.0, confidence))
409
+
410
+ # ------------------------------------------------------------------
411
+ # Core operations
412
+ # ------------------------------------------------------------------
413
+
414
+ @staticmethod
415
+ def _source_projection_erased(source_uri: str | None) -> bool:
416
+ if not source_uri or not source_uri.startswith("turn:"):
417
+ return False
418
+ from apsimo import get_state_dir
419
+ from apsimo.turns import get_turn_idempotency_ledger
420
+ return get_turn_idempotency_ledger(get_state_dir()).is_projection_erased(source_uri[5:])
421
+
422
+ async def _filter_erased_source_memories(self, memories: list[dict]) -> list[dict]:
423
+ """A durable erase fence remains effective during projection outages."""
424
+ return [memory for memory in memories
425
+ if not self._source_projection_erased(memory.get("source_uri"))]
426
+
427
+ async def delete_source_memories(self, turn_ids: list[str]) -> int:
428
+ """Remove graph and vector copies after the source fence commits."""
429
+ uris = ["turn:" + item for item in dict.fromkeys(turn_ids)]
430
+ if not uris:
431
+ return 0
432
+ # Keep IDs until both stores have been addressed. The graph's source
433
+ # tombstone filter blocks recall even if vector deletion must be retried.
434
+ async with self.driver.session(database=self.database) as session:
435
+ result = await session.run("MATCH (m:Memory) WHERE m.source_uri IN $uris RETURN m.id AS id", uris=uris)
436
+ ids = [record["id"] async for record in result]
437
+ for memory_id in ids:
438
+ if self._vector_store is not None:
439
+ from apsimo.vector.collections import Collection
440
+ await self._vector_store.delete(collection=Collection.MEMORIES, id=memory_id)
441
+ async with self.driver.session(database=self.database) as session:
442
+ await session.run("MATCH (m:Memory {id: $id}) DETACH DELETE m", id=memory_id)
443
+ ring = getattr(self, "_distill_preview", None)
444
+ if ring is not None:
445
+ retained = [item for item in ring if item.get("source_turn_id") not in turn_ids]
446
+ ring.clear()
447
+ ring.extend(retained)
448
+ return len(ids)
449
+
450
+ async def iter_indexable_memories(self, batch_size: int = 128):
451
+ """Current graph evidence for a projection rebuild, without old vectors."""
452
+ after = ''
453
+ while True:
454
+ async with self.driver.session(database=self.database) as session:
455
+ result = await session.run('''
456
+ MATCH (m:Memory) WHERE m.id > $after
457
+ WITH m ORDER BY m.id LIMIT $limit
458
+ OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
459
+ RETURN properties(m) AS memory, collect(p.id) AS people
460
+ ORDER BY memory.id
461
+ ''', after=after, limit=batch_size)
462
+ rows = [dict(row) async for row in result]
463
+ if not rows:
464
+ return
465
+ for row in rows:
466
+ memory = row['memory']
467
+ after = memory['id']
468
+ if self._source_projection_erased(memory.get('source_uri')):
469
+ continue
470
+ if memory.get('superseded_by') or not memory.get('content'):
471
+ continue
472
+ metadata = memory.get('metadata') or {}
473
+ if isinstance(metadata, str):
474
+ try:
475
+ metadata = json.loads(metadata)
476
+ except json.JSONDecodeError:
477
+ # Earlier store_memory writers used str(dict), not JSON.
478
+ # Read those retained records without rewriting evidence.
479
+ metadata = ast.literal_eval(metadata)
480
+ if not isinstance(metadata, dict):
481
+ raise ValueError('Graph memory metadata must be an object')
482
+ metadata = {**metadata, **{key: memory.get(key) for key in
483
+ ('source_uri', 'source_type', 'source_version', 'content_hash', 'session_id',
484
+ 'type', 'strength', 'effective_confidence', 'epistemic_state', 'protected')}}
485
+ metadata['person_id'] = (row['people'] or [None])[0]
486
+ metadata['memory_id'] = memory['id']
487
+ yield {'id': memory['id'], 'text': memory['content'], 'metadata': metadata}
488
+
489
+ async def store_memory(
490
+ self,
491
+ content: str,
492
+ memory_type: str,
493
+ entities: List[str],
494
+ metadata: Dict[str, Any] | None = None,
495
+ importance: float = 1.0,
496
+ person_id: Optional[str] = None,
497
+ session_id: Optional[str] = None,
498
+ source_type: str = "inference",
499
+ source_uri: Optional[str] = None,
500
+ source_version: Optional[str] = None,
501
+ content_hash: Optional[str] = None,
502
+ ) -> str:
503
+ """Store a memory with automatic entity linking.
504
+
505
+ Creates a :Memory node and :MENTIONS edges to each :Entity.
506
+ If an embedding function is registered the memory's vector is stored
507
+ on the node so it participates in semantic search.
508
+
509
+ Args:
510
+ content: Memory content text
511
+ memory_type: Type of memory (episodic, semantic, procedural, identity)
512
+ entities: Named entities to link to this memory
513
+ metadata: Optional key-value metadata
514
+ importance: Initial importance / strength (0-1, default 1.0)
515
+ person_id: Optional person ID to link this memory to via (Memory)-[:ABOUT]->(Person)
516
+ source_type: Origin of this memory (conversation, file, tool_output, user_assertion, inference)
517
+ source_uri: Optional URI referencing the source
518
+ source_version: Optional version string for the source
519
+ content_hash: Optional SHA-256 hash of the content
520
+
521
+ Returns:
522
+ The UUID of the newly created Memory node.
523
+ """
524
+ metadata = metadata or {}
525
+ if self._source_projection_erased(source_uri):
526
+ return ""
527
+ # Guard (v0.21.1): never store empty/whitespace memories — they were
528
+ # accumulating as duplicate junk nodes.
529
+ if not content or not content.strip():
530
+ return ""
531
+ max_importance = MAX_IMPORTANCE.get(source_type, 0.7)
532
+ if importance > max_importance:
533
+ logger.warning(
534
+ "Importance %.2f for source_type '%s' exceeds max %.2f; clamping.",
535
+ importance, source_type, max_importance,
536
+ )
537
+ importance = max_importance
538
+ importance = max(0.0, min(1.0, importance))
539
+
540
+ # Resolve person_id from explicit arg or metadata fallback
541
+ person_id = person_id or metadata.get("person_id")
542
+ # Never mint a :Person node for a non-contact sentinel. Ids like 'default'/'unknown'/empty are
543
+ # NOT real people; attributing memories to them created a catch-all junk Person that polluted
544
+ # per-person recall (a host whose resolver fell back to "default" dumped most of its memory
545
+ # onto one pseudo-contact). Such memories are stored UNATTRIBUTED (no :ABOUT edge) instead.
546
+ # Real contact ids still create/link a Person normally (that is the graph's discovery design).
547
+ if isinstance(person_id, str):
548
+ person_id = person_id.strip() or None
549
+ if person_id and person_id.lower() in {"default", "unknown", "none", "null", "anonymous"}:
550
+ person_id = None
551
+ elif person_id is not None:
552
+ person_id = None
553
+ # Resolve session_id from explicit arg or metadata fallback
554
+ session_id = session_id or (metadata.get("session_id") if metadata else None)
555
+
556
+ source_type = (source_type or MemorySourceType.INFERENCE.value).lower()
557
+ source_uri = source_uri or None
558
+ source_version = source_version or None
559
+ # Always derive a content hash so identical memories can be deduped
560
+ # (v0.21.1 — previously null, so every write created a duplicate node).
561
+ content_hash = content_hash or hashlib.sha256(content.encode("utf-8")).hexdigest()
562
+ source_reliability = SOURCE_RELIABILITY.get(source_type, 0.5)
563
+ protected = source_type == MemorySourceType.USER_ASSERTION
564
+ base_confidence = importance
565
+ epistemic_state = EpistemicState.INFERRED
566
+ created_at = self._utcnow()
567
+
568
+ effective_confidence = self.compute_effective_confidence(
569
+ base_confidence=base_confidence,
570
+ source_reliability=source_reliability,
571
+ corroboration_count=0,
572
+ contradiction_count=0,
573
+ recalls=0,
574
+ last_verified_at=None,
575
+ created_at=created_at,
576
+ epistemic_state=epistemic_state.value,
577
+ now=created_at,
578
+ )
579
+
580
+ # Dedup (v0.21.1): if an identical memory already exists (same content
581
+ # hash), reinforce it instead of creating a duplicate — and skip the
582
+ # embed cost entirely. This collapses the runaway duplication (e.g. a
583
+ # recurring cron prompt that had been stored 70+ times).
584
+ if content_hash:
585
+ async with self.driver.session(database=self.database) as session:
586
+ dq = await session.run(
587
+ """
588
+ MATCH (m:Memory {content_hash: $content_hash})
589
+ WHERE m.superseded_by IS NULL
590
+ AND ($source_uri IS NULL OR m.source_uri = $source_uri)
591
+ SET m.accessed_at = datetime(),
592
+ m.corroboration_count = coalesce(m.corroboration_count, 0) + 1,
593
+ m.strength = CASE WHEN coalesce(m.strength, 0.0) < 1.0
594
+ THEN coalesce(m.strength, 0.0) + 0.05 ELSE 1.0 END
595
+ RETURN m.id AS id
596
+ ORDER BY m.created_at ASC
597
+ LIMIT 1
598
+ """,
599
+ content_hash=content_hash,
600
+ source_uri=source_uri,
601
+ )
602
+ existing = await dq.single()
603
+ if existing is not None:
604
+ logger.debug("store_memory dedup: reinforced %s", existing["id"])
605
+ return existing["id"]
606
+
607
+ # Compute embedding. If an embedder is configured but we can't get a
608
+ # usable vector (outage that survived retries), FAIL the write rather than
609
+ # create an unsearchable memory (silent loss). The vector lives in the
610
+ # LanceDB store; the Neo4j m.embedding property is secondary/best-effort.
611
+ embedding: Optional[List[float]] = None
612
+ if self._embed_fn is not None:
613
+ embedding = await self._embed(content)
614
+ if not embedding:
615
+ raise RuntimeError(
616
+ "embedding unavailable — refusing to store unsearchable memory")
617
+
618
+ # Preserve dict for vector store before stringifying for Neo4j
619
+ metadata_dict = metadata
620
+ metadata_str = json.dumps(metadata_dict, default=str)
621
+
622
+ async with self.driver.session(database=self.database) as session:
623
+ result = await session.run(
624
+ """
625
+ CREATE (m:Memory {
626
+ id: randomUUID(),
627
+ content: $content,
628
+ type: $memory_type,
629
+ importance: $importance,
630
+ strength: $importance,
631
+ recalls: 0,
632
+ created_at: datetime(),
633
+ accessed_at: datetime(),
634
+ embedding: $embedding,
635
+ metadata: $metadata,
636
+ session_id: $session_id,
637
+ source_type: $source_type,
638
+ source_uri: $source_uri,
639
+ source_version: $source_version,
640
+ content_hash: $content_hash,
641
+ base_confidence: $base_confidence,
642
+ source_reliability: $source_reliability,
643
+ corroboration_count: 0,
644
+ contradiction_count: 0,
645
+ effective_confidence: $effective_confidence,
646
+ epistemic_state: $epistemic_state,
647
+ protected: $protected,
648
+ last_verified_at: null,
649
+ superseded_by: null,
650
+ provenance: []
651
+ })
652
+ WITH m
653
+ FOREACH (entity_name IN $entities |
654
+ MERGE (e:Entity {name: entity_name})
655
+ CREATE (m)-[:MENTIONS]->(e)
656
+ )
657
+ WITH m
658
+ FOREACH (_ IN CASE WHEN $person_id IS NOT NULL THEN [1] ELSE [] END |
659
+ MERGE (p:Person {id: $person_id})
660
+ CREATE (m)-[:ABOUT]->(p)
661
+ )
662
+ WITH m
663
+ FOREACH (_ IN CASE WHEN $source_uri IS NOT NULL AND $source_type = "file" THEN [1] ELSE [] END |
664
+ MERGE (fa:FileAnchor {uri: $source_uri})
665
+ ON CREATE SET fa.first_seen = datetime()
666
+ CREATE (m)-[:DERIVED_FROM {derivation_type: "file_read"}]->(fa)
667
+ )
668
+ RETURN m.id AS id
669
+ """,
670
+ content=content,
671
+ memory_type=memory_type,
672
+ importance=importance,
673
+ entities=entities,
674
+ embedding=embedding,
675
+ metadata=metadata_str,
676
+ person_id=person_id,
677
+ session_id=session_id,
678
+ source_type=source_type,
679
+ source_uri=source_uri,
680
+ source_version=source_version,
681
+ content_hash=content_hash,
682
+ base_confidence=base_confidence,
683
+ source_reliability=source_reliability,
684
+ effective_confidence=effective_confidence,
685
+ epistemic_state=epistemic_state,
686
+ protected=protected,
687
+ )
688
+ record = await result.single()
689
+ if record is None:
690
+ raise RuntimeError("Failed to create memory node")
691
+ memory_id = record["id"]
692
+
693
+ # Write to LanceDB vector store (if configured)
694
+ if self._vector_store is not None and embedding is not None:
695
+ try:
696
+ from apsimo.vector.collections import Collection
697
+ await self._vector_store.add(
698
+ collection=Collection.MEMORIES,
699
+ id=memory_id,
700
+ text=content,
701
+ vector=embedding,
702
+ metadata={
703
+ "memory_id": memory_id,
704
+ "type": memory_type,
705
+ "strength": importance,
706
+ "importance": importance,
707
+ # Keep the vector candidate's scope aligned with the
708
+ # authoritative graph ABOUT edge. Historically this
709
+ # read only metadata["person_id"], so callers using the
710
+ # explicit argument produced unscoped vector rows.
711
+ "person_id": person_id,
712
+ "tags": metadata_dict.get("tags", []) if metadata_dict else [],
713
+ "created_at": metadata_dict.get("created_at") if metadata_dict else None,
714
+ "session_id": session_id,
715
+ "source_type": source_type,
716
+ "source_uri": source_uri,
717
+ "source_version": source_version,
718
+ "content_hash": content_hash,
719
+ "effective_confidence": effective_confidence,
720
+ "epistemic_state": epistemic_state.value,
721
+ "protected": protected,
722
+ },
723
+ )
724
+ except Exception as exc:
725
+ logger.warning("Failed to write memory to vector store: %s", exc)
726
+
727
+ # A deletion may have committed while embedding or graph I/O awaited.
728
+ if self._source_projection_erased(source_uri):
729
+ await self.delete_source_memories([source_uri[5:]])
730
+ return ""
731
+ return memory_id
732
+
733
+ def _distill_preview_ring(self) -> "deque":
734
+ """Bounded ring of shadow distill previews (created on first use so
735
+ alternate construction paths, e.g. tests, still work)."""
736
+ ring = getattr(self, "_distill_preview", None)
737
+ if ring is None:
738
+ ring = deque(maxlen=50)
739
+ self._distill_preview = ring
740
+ return ring
741
+
742
+ def distill_preview(self) -> List[Dict[str, Any]]:
743
+ """Newest-first shadow distill previews (empty once the flag is live)."""
744
+ return list(reversed(self._distill_preview_ring()))
745
+
746
+ async def record_turn(
747
+ self,
748
+ session_id: str,
749
+ contact_id: Optional[str],
750
+ topics: List[str],
751
+ entities: List[str],
752
+ tools_used: List[str],
753
+ summary: Optional[str],
754
+ turn_id: Optional[str] = None,
755
+ ) -> Optional[str]:
756
+ """Store a conversation turn as an episodic memory.
757
+
758
+ Creates a :Memory node of type ``episodic`` linked to the
759
+ conversation session and contact. Entities and topics are
760
+ merged as :Entity nodes. Tools used are stored in metadata.
761
+
762
+ Args:
763
+ session_id: Hermes session identifier
764
+ contact_id: Optional contact / person identifier
765
+ topics: Extracted topics from the turn
766
+ entities: Named entities mentioned
767
+ tools_used: Tool names invoked during the turn
768
+ summary: Human-readable summary of the exchange
769
+
770
+ Returns:
771
+ The UUID of the created Memory node, or None if storage fails.
772
+ """
773
+ if not summary or self._source_projection_erased("turn:" + turn_id if turn_id else None):
774
+ return None
775
+ # Salience gate: don't memorialize internal-plumbing turns (context-compaction references and
776
+ # host-specific system-prompt wrappers / self-checks). Generic markers are built in; a
777
+ # deployment adds its own via COLONY_MEMORY_SKIP_MARKERS ('|'-separated, case-insensitive).
778
+ # This is what keeps the memory graph facts-and-events, not a verbatim transcript log.
779
+ _sl = summary.lower()
780
+ _skip = ("[context compaction", "[post-compaction", "reference only]", "[context summary]")
781
+ _env = os.environ.get("COLONY_MEMORY_SKIP_MARKERS", "")
782
+ if any(m in _sl for m in _skip) or any(m.strip().lower() in _sl for m in _env.split("|") if m.strip()):
783
+ logger.debug("record_turn: skipped low-salience / internal-marker turn")
784
+ return None
785
+
786
+ # Real salience score (attribution redesign Phase 2), replacing the old hardcoded
787
+ # importance=0.85 that overrode the computed value. Signal from what the turn
788
+ # actually carries: named entities (facts about people/things), tool use (an
789
+ # action happened), and substance (length). A throwaway "ok thanks" scores low
790
+ # and decays fast; a fact-dense exchange scores high and persists.
791
+ _ent_n = len(entities or [])
792
+ _score = 0.35
793
+ _score += min(0.30, 0.10 * _ent_n) # up to +0.30 for entities
794
+ if tools_used:
795
+ _score += 0.15 # an action was taken
796
+ if len(summary) > 240:
797
+ _score += 0.10 # substantive exchange
798
+ if "?" in summary:
799
+ _score += 0.05 # a question = intent/curiosity worth recalling
800
+ importance = round(min(_score, 0.95), 3)
801
+
802
+ # Optional distillation (shadow by default): store the salient content rather
803
+ # than the verbatim "User:/Agent:" wrapper. The distilled form is ALWAYS
804
+ # computed; off => stored content is unchanged and the would-be result goes
805
+ # into a bounded in-memory preview ring (GET /v1/host/memory/distill-preview)
806
+ # so the flip can be validated on real traffic first. On => store it.
807
+ content = summary
808
+ distilled = distill_turn_summary(summary)
809
+ _distill = os.environ.get("COLONY_DISTILL_TURNS", "0") not in ("0", "false", "no")
810
+ if _distill:
811
+ content = distilled
812
+ else:
813
+ try:
814
+ self._distill_preview_ring().append({
815
+ "session_id": session_id,
816
+ "source_turn_id": turn_id,
817
+ "original": summary[:400],
818
+ "distilled": distilled[:400],
819
+ "importance": importance,
820
+ "ts": time.time(),
821
+ })
822
+ except Exception:
823
+ logger.debug("distill preview append failed", exc_info=True)
824
+ logger.debug("distill(shadow): would store salient content for session %s (imp=%.2f)",
825
+ session_id, importance)
826
+
827
+ metadata: Dict[str, Any] = {
828
+ "turn": True,
829
+ "source_turn_id": turn_id,
830
+ "topics": topics,
831
+ "tools_used": tools_used,
832
+ "salience": importance,
833
+ }
834
+
835
+ try:
836
+ content_hash = hashlib.sha256(content.encode("utf-8")).hexdigest()
837
+ return await self.store_memory(
838
+ content=content,
839
+ memory_type="episodic",
840
+ entities=entities or [],
841
+ metadata=metadata,
842
+ importance=importance,
843
+ person_id=contact_id,
844
+ source_type=MemorySourceType.CONVERSATION.value,
845
+ source_uri=f"turn:{turn_id}" if turn_id else f"session:{session_id}",
846
+ session_id=session_id,
847
+ content_hash=content_hash,
848
+ )
849
+ except Exception as exc:
850
+ logger.warning("record_turn failed: %s", exc)
851
+ return None
852
+
853
+ async def read_memories(
854
+ self,
855
+ *,
856
+ person_id: Optional[str] = None,
857
+ memory_id: Optional[str] = None,
858
+ limit: int = 20,
859
+ ) -> List[Dict[str, Any]]:
860
+ """Read recent memories, preserving the same hard ABOUT boundary as recall.
861
+
862
+ ``person_id`` is intentionally a single exact lane. A scoped miss is
863
+ empty and never retries against the unscoped graph. Omitting the person
864
+ remains only for the deprecated legacy/internal authority path; the API
865
+ boundary always supplies a server-derived person for scoped callers.
866
+ """
867
+
868
+ person_scope = str(person_id).strip() if person_id else None
869
+ memory_scope = str(memory_id).strip() if memory_id else None
870
+ safe_limit = max(1, min(int(limit or 20), 100))
871
+ excluded_sources = getattr(self, "_recall_source_exclusions", ())
872
+ excluded_metadata_markers = getattr(
873
+ self, "_recall_metadata_exclusions", ())
874
+ policy_active = bool(
875
+ excluded_sources or excluded_metadata_markers)
876
+ async with self.driver.session(database=self.database) as session:
877
+ if person_scope:
878
+ if policy_active:
879
+ result = await session.run(
880
+ """
881
+ MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
882
+ WHERE ($memory_id IS NULL OR m.id = $memory_id)
883
+ AND (size($exclude_source_uris) = 0 OR NOT (
884
+ coalesce(m.source_uri, "") IN
885
+ $exclude_source_uris))
886
+ AND (size($exclude_metadata_markers) = 0 OR
887
+ NONE(marker IN $exclude_metadata_markers
888
+ WHERE toLower(coalesce(m.metadata, ""))
889
+ CONTAINS marker))
890
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
891
+ WITH m, collect(DISTINCT e.name) AS entity_names
892
+ RETURN m {.*, entities: entity_names,
893
+ person_id: $person_id} AS memory
894
+ ORDER BY m.created_at DESC
895
+ LIMIT $limit
896
+ """,
897
+ person_id=person_scope,
898
+ memory_id=memory_scope,
899
+ limit=safe_limit,
900
+ exclude_source_uris=excluded_sources,
901
+ exclude_metadata_markers=excluded_metadata_markers,
902
+ )
903
+ else:
904
+ result = await session.run(
905
+ """
906
+ MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
907
+ WHERE $memory_id IS NULL OR m.id = $memory_id
908
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
909
+ WITH m, collect(DISTINCT e.name) AS entity_names
910
+ RETURN m {.*, entities: entity_names,
911
+ person_id: $person_id} AS memory
912
+ ORDER BY m.created_at DESC
913
+ LIMIT $limit
914
+ """,
915
+ person_id=person_scope,
916
+ memory_id=memory_scope,
917
+ limit=safe_limit,
918
+ )
919
+ else:
920
+ if policy_active:
921
+ result = await session.run(
922
+ """
923
+ MATCH (m:Memory)
924
+ WHERE ($memory_id IS NULL OR m.id = $memory_id)
925
+ AND (size($exclude_source_uris) = 0 OR NOT (
926
+ coalesce(m.source_uri, "") IN
927
+ $exclude_source_uris))
928
+ AND (size($exclude_metadata_markers) = 0 OR
929
+ NONE(marker IN $exclude_metadata_markers
930
+ WHERE toLower(coalesce(m.metadata, ""))
931
+ CONTAINS marker))
932
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
933
+ OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
934
+ WITH m, collect(DISTINCT e.name) AS entity_names,
935
+ collect(DISTINCT p.id) AS person_ids
936
+ RETURN m {.*, entities: entity_names,
937
+ person_id: head(person_ids)} AS memory
938
+ ORDER BY m.created_at DESC
939
+ LIMIT $limit
940
+ """,
941
+ memory_id=memory_scope,
942
+ limit=safe_limit,
943
+ exclude_source_uris=excluded_sources,
944
+ exclude_metadata_markers=excluded_metadata_markers,
945
+ )
946
+ else:
947
+ result = await session.run(
948
+ """
949
+ MATCH (m:Memory)
950
+ WHERE $memory_id IS NULL OR m.id = $memory_id
951
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
952
+ OPTIONAL MATCH (m)-[:ABOUT]->(p:Person)
953
+ WITH m, collect(DISTINCT e.name) AS entity_names,
954
+ collect(DISTINCT p.id) AS person_ids
955
+ RETURN m {.*, entities: entity_names,
956
+ person_id: head(person_ids)} AS memory
957
+ ORDER BY m.created_at DESC
958
+ LIMIT $limit
959
+ """,
960
+ memory_id=memory_scope,
961
+ limit=safe_limit,
962
+ )
963
+ memories: List[Dict[str, Any]] = []
964
+ async for record in result:
965
+ memories.append(dict(record["memory"]))
966
+ return await self._filter_erased_source_memories(memories)
967
+
968
+ def _memory_is_recall_excluded(self, memory: Dict[str, Any]) -> bool:
969
+ if self._source_projection_erased(memory.get("source_uri")):
970
+ return True
971
+ sources = set(getattr(self, "_recall_source_exclusions", ()))
972
+ if str(memory.get("source_uri") or "") in sources:
973
+ return True
974
+ metadata = memory.get("metadata")
975
+ raw_metadata = str(
976
+ memory if metadata is None else metadata).lower()
977
+ return any(
978
+ marker in raw_metadata
979
+ for marker in getattr(self, "_recall_metadata_exclusions", ())
980
+ )
981
+
982
+ async def filter_memory_vector_results(
983
+ self,
984
+ rows: List[Dict[str, Any]],
985
+ ) -> List[Dict[str, Any]]:
986
+ """Apply the graph's hard content policy to raw vector result text.
987
+
988
+ Vector metadata catches current mirrors cheaply. Ambiguous/legacy rows
989
+ are hydrated by ID so their authoritative graph metadata is checked.
990
+ Non-graph vectors (for example image captions) remain available.
991
+ """
992
+ rows = [row for row in rows if not self._source_projection_erased(
993
+ row.get("source_uri") or (row.get("metadata") or {}).get("source_uri"))]
994
+ if not (
995
+ getattr(self, "_recall_source_exclusions", ())
996
+ or getattr(self, "_recall_metadata_exclusions", ())
997
+ ):
998
+ return rows
999
+
1000
+ candidates = [row for row in rows[:512] if isinstance(row, dict)]
1001
+ ids = tuple(dict.fromkeys(
1002
+ str(row.get("id") or "").strip() for row in candidates
1003
+ if str(row.get("id") or "").strip()
1004
+ ))
1005
+ graph_rows: Dict[str, Dict[str, Any]] = {}
1006
+ if ids:
1007
+ async with self.driver.session(database=self.database) as session:
1008
+ result = await session.run(
1009
+ """
1010
+ MATCH (m:Memory) WHERE m.id IN $ids
1011
+ RETURN m {.*} AS memory
1012
+ """,
1013
+ ids=ids,
1014
+ )
1015
+ async for record in result:
1016
+ memory = dict(record["memory"])
1017
+ mid = str(memory.get("id") or "")
1018
+ if mid:
1019
+ graph_rows[mid] = memory
1020
+
1021
+ allowed: List[Dict[str, Any]] = []
1022
+ for row in candidates:
1023
+ mid = str(row.get("id") or "").strip()
1024
+ if not mid:
1025
+ continue
1026
+ vector_metadata = row.get("metadata")
1027
+ vector_view = dict(vector_metadata) if isinstance(
1028
+ vector_metadata, dict) else {"metadata": vector_metadata}
1029
+ if self._memory_is_recall_excluded(vector_view):
1030
+ continue
1031
+ graph_view = graph_rows.get(mid)
1032
+ if graph_view is not None and self._memory_is_recall_excluded(
1033
+ graph_view
1034
+ ):
1035
+ continue
1036
+ if graph_view is None:
1037
+ metadata = (
1038
+ vector_metadata
1039
+ if isinstance(vector_metadata, dict) else {}
1040
+ )
1041
+ modality = str(metadata.get("modality") or "").lower()
1042
+ source_uri = str(metadata.get("source_uri") or "").strip()
1043
+ # Missing graph hydration is not authority for ambiguous text.
1044
+ # Preserve explicit non-graph media and text with an identified,
1045
+ # non-excluded source; unknown text fails closed.
1046
+ if modality != "image" and not source_uri:
1047
+ continue
1048
+ allowed.append(row)
1049
+ return allowed
1050
+
1051
+ async def recall_candidates(self, **kwargs):
1052
+ """Authorized graph candidates without reranking or reinforcement."""
1053
+ return await self.recall(**kwargs, candidates_only=True)
1054
+
1055
+ def record_recall_use(self, memories):
1056
+ """Reinforce only graph records that actually entered turn context."""
1057
+ for memory in memories:
1058
+ if memory.get("kind", "belief") == "belief" and memory.get("id"):
1059
+ # Annotated packets have a presentation ID, separate from the
1060
+ # graph record whose use should slow subsequent memory decay.
1061
+ memory_id = memory.get("_recall_memory_id") or memory["id"]
1062
+ self._retain_task(asyncio.create_task(
1063
+ self._touch_memory_safe(memory_id)))
1064
+
1065
+ async def recall(
1066
+ self,
1067
+ query: str,
1068
+ limit: int = 10,
1069
+ min_strength: Optional[float] = None,
1070
+ min_confidence: float = 0.1,
1071
+ person_id: Optional[str] = None,
1072
+ exclude_source_uris: Optional[List[str]] = None,
1073
+ *,
1074
+ candidates_only: bool = False,
1075
+ ) -> List[Dict[str, Any]]:
1076
+ """Retrieve memories by semantic similarity with strength decay.
1077
+
1078
+ Uses LanceDB for ANN search, then hydrates full Memory nodes from
1079
+ Neo4j with entity mentions. Falls back to graph-only keyword
1080
+ recall if no vector store or embedding function is configured.
1081
+
1082
+ ``min_strength`` is the hard exclusion floor for decayed memories.
1083
+ An explicit caller value always wins; when omitted it comes from
1084
+ COLONY_RECALL_MIN_STRENGTH (default 0.1, the historical floor).
1085
+ The floor stays hard — decayed junk is excluded, not demoted — and
1086
+ is only lowered stepwise as live pruning cleans the graph.
1087
+
1088
+ An explicit ``person_id`` is a hard candidate boundary: only memories
1089
+ linked ``ABOUT`` that exact person may be hydrated or ranked. A scoped
1090
+ miss stays empty and never falls back to global recall. Trusted owner
1091
+ or internal callers that intentionally need global memory must omit
1092
+ ``person_id`` after establishing that authority outside this method.
1093
+
1094
+ ``exclude_source_uris`` is a bounded hard hydration filter applied
1095
+ before confidence/relevance ranking and reranking. The ANN candidate
1096
+ fetch is widened when this filter is present so excluded mirrors do
1097
+ not ordinarily starve the requested result window.
1098
+
1099
+ ``candidates_only`` skips final reranking and recall-strength updates;
1100
+ the caller must select authorized candidates and record actual use.
1101
+
1102
+ Returns:
1103
+ A list of memory dicts, each annotated with ``entities`` and
1104
+ sorted by relevance descending.
1105
+ """
1106
+ if min_strength is None:
1107
+ try:
1108
+ min_strength = float(os.environ.get(
1109
+ "COLONY_RECALL_MIN_STRENGTH", "0.1"))
1110
+ except (TypeError, ValueError):
1111
+ min_strength = 0.1
1112
+ # Strength-blended ranking (COLONY_RECALL_STRENGTH_RANKING, default
1113
+ # off = legacy vector_score * effective_confidence). When on, decayed
1114
+ # memories are demoted smoothly ABOVE the hard floor:
1115
+ # relevance = vector_score * effective_confidence * (0.5 + 0.5*strength)
1116
+ strength_ranking = os.environ.get(
1117
+ "COLONY_RECALL_STRENGTH_RANKING", "off").strip().lower() in (
1118
+ "on", "1", "true", "yes")
1119
+ person_scope = str(person_id).strip() if person_id else None
1120
+ excluded_sources = self._bounded_source_uris((
1121
+ *getattr(self, "_recall_source_exclusions", ()),
1122
+ *(exclude_source_uris or ()),
1123
+ ))
1124
+ excluded_metadata_markers = self._bounded_source_uris(
1125
+ getattr(self, "_recall_metadata_exclusions", ()))
1126
+ hybrid = os.environ.get("COLONY_RECALL_HYBRID", "off").strip().lower() in (
1127
+ "on", "1", "true", "yes")
1128
+ # Vector search path: embed query (with instruction) → LanceDB ANN → Neo4j hydration
1129
+ if self._vector_store is not None and self._embed_fn is not None:
1130
+ try:
1131
+ embedding = await self._embed_query(query)
1132
+ from apsimo.vector.collections import Collection
1133
+ # Oversample the ANN fetch (COLONY_RECALL_OVERSAMPLE, default 1
1134
+ # = legacy exact-limit fetch) so the post-hydration filters
1135
+ # (strength floor, terminal epistemic states, low confidence)
1136
+ # can't silently shrink the result set below the requested
1137
+ # limit. Oversampled fetches are capped at 100 candidates and
1138
+ # trimmed back to `limit` after filtering.
1139
+ try:
1140
+ oversample = int(os.environ.get(
1141
+ "COLONY_RECALL_OVERSAMPLE", "1"))
1142
+ except (TypeError, ValueError):
1143
+ oversample = 1
1144
+ fetch_limit = (limit if oversample <= 1
1145
+ else min(limit * oversample, max(100, limit)))
1146
+ # Existing vector rows predate structured recipient metadata,
1147
+ # so the authoritative scope check happens during Neo4j
1148
+ # hydration. Widen scoped ANN candidate generation to reduce
1149
+ # false-empty results without ever relaxing that graph gate.
1150
+ if person_scope:
1151
+ try:
1152
+ scope_oversample = max(1, int(os.environ.get(
1153
+ "COLONY_RECALL_SCOPE_OVERSAMPLE", "20")))
1154
+ except (TypeError, ValueError):
1155
+ scope_oversample = 20
1156
+ fetch_limit = max(
1157
+ fetch_limit,
1158
+ min(limit * scope_oversample, max(200, limit)),
1159
+ )
1160
+ elif excluded_sources or excluded_metadata_markers:
1161
+ fetch_limit = max(
1162
+ fetch_limit, min(limit * 20, max(200, limit)))
1163
+ results = await self._vector_store.search(
1164
+ collection=Collection.MEMORIES,
1165
+ query_vector=embedding,
1166
+ limit=fetch_limit,
1167
+ # metadata is stored as a JSON string (pa.utf8()); LanceDB's
1168
+ # filter dialect does not support json_extract on utf8 columns.
1169
+ # Strength filtering is applied post-hydration from Neo4j below.
1170
+ filter=None,
1171
+ )
1172
+ # Adaptive relevance floor (meta-learning knob): vector hits
1173
+ # scoring below recall.min_relevance are dropped before
1174
+ # hydration. Default 0.0 = no filter; store caps at 0.5 so a
1175
+ # self-adjustment can never starve retrieval.
1176
+ min_relevance = 0.0
1177
+ params = getattr(self, "_adaptive_params", None)
1178
+ if params is not None:
1179
+ try:
1180
+ from apsimo.self_model.params import (
1181
+ PARAM_RECALL_MIN_RELEVANCE,
1182
+ )
1183
+ min_relevance = float(params.get(
1184
+ PARAM_RECALL_MIN_RELEVANCE, default=0.0))
1185
+ except Exception:
1186
+ min_relevance = 0.0
1187
+ if min_relevance > 0.0:
1188
+ results = [r for r in results if r.score >= min_relevance]
1189
+ memories = []
1190
+ if results:
1191
+ memory_ids = [r.id for r in results]
1192
+ score_map = {r.id: r.score for r in results}
1193
+
1194
+ # Hydrate from Neo4j
1195
+ async with self.driver.session(database=self.database) as session:
1196
+ if person_scope:
1197
+ result = await session.run(
1198
+ """
1199
+ MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
1200
+ WHERE m.id IN $ids
1201
+ AND (size($exclude_source_uris) = 0 OR NOT (
1202
+ coalesce(m.source_uri, "") IN
1203
+ $exclude_source_uris))
1204
+ AND (size($exclude_metadata_markers) = 0 OR
1205
+ NONE(marker IN $exclude_metadata_markers
1206
+ WHERE toLower(coalesce(m.metadata, ""))
1207
+ CONTAINS marker))
1208
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1209
+ WITH m, collect(e.name) AS entity_names
1210
+ RETURN m {.*, entities: entity_names,
1211
+ person_id: $person_id} AS memory
1212
+ """,
1213
+ ids=memory_ids,
1214
+ person_id=person_scope,
1215
+ exclude_source_uris=excluded_sources,
1216
+ exclude_metadata_markers=(
1217
+ excluded_metadata_markers),
1218
+ )
1219
+ else:
1220
+ result = await session.run(
1221
+ """
1222
+ MATCH (m:Memory) WHERE m.id IN $ids
1223
+ AND (size($exclude_source_uris) = 0 OR NOT (
1224
+ coalesce(m.source_uri, "") IN
1225
+ $exclude_source_uris))
1226
+ AND (size($exclude_metadata_markers) = 0 OR
1227
+ NONE(marker IN $exclude_metadata_markers
1228
+ WHERE toLower(coalesce(m.metadata, ""))
1229
+ CONTAINS marker))
1230
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1231
+ WITH m, collect(e.name) AS entity_names
1232
+ RETURN m {.*, entities: entity_names} AS memory
1233
+ """,
1234
+ ids=memory_ids,
1235
+ exclude_source_uris=excluded_sources,
1236
+ exclude_metadata_markers=(
1237
+ excluded_metadata_markers),
1238
+ )
1239
+ memories = []
1240
+ async for record in result:
1241
+ mem = record["memory"]
1242
+ mid = mem.get("id", "")
1243
+ vector_score = score_map.get(mid, 0.0)
1244
+ strength = float(mem.get("strength", 1.0))
1245
+ if strength < min_strength:
1246
+ continue
1247
+ # Filter terminal epistemic states and low confidence
1248
+ epistemic_state = mem.get("epistemic_state", "inferred")
1249
+ if mem.get("superseded_by") or epistemic_state in ("stale", "superseded", "deprecated", "archived"):
1250
+ continue
1251
+ effective_confidence = float(mem.get("effective_confidence", mem.get("strength", 1.0)))
1252
+ if effective_confidence < min_confidence:
1253
+ continue
1254
+ if strength_ranking:
1255
+ mem["relevance"] = (vector_score
1256
+ * effective_confidence
1257
+ * (0.5 + 0.5 * strength))
1258
+ else:
1259
+ mem["relevance"] = vector_score * effective_confidence
1260
+ memories.append(mem)
1261
+
1262
+ if hybrid:
1263
+ from .recall import fuse_candidates
1264
+ candidate_limit = min(max(limit * 5, limit), max(100, limit))
1265
+ lexical = await self._recall_lexical(
1266
+ query, candidate_limit, min_strength, min_confidence,
1267
+ person_scope, excluded_sources, excluded_metadata_markers)
1268
+ memories = fuse_candidates(
1269
+ memories, lexical, limit=candidate_limit,
1270
+ strength_ranking=strength_ranking)
1271
+ if results or memories:
1272
+ memories = await self._filter_erased_source_memories(memories)
1273
+ if candidates_only:
1274
+ return sorted(memories, key=lambda m: m.get("relevance", 0), reverse=True)[:limit]
1275
+ memories = await self._maybe_rerank(
1276
+ query, memories, limit,
1277
+ strength_ranking=strength_ranking)
1278
+ memories = await self._filter_erased_source_memories(memories)
1279
+ memories.sort(key=lambda m: m.get("relevance", 0), reverse=True)
1280
+ memories = memories[:limit]
1281
+ # Fire-and-forget touch_memory for each recalled result
1282
+ for mem in memories:
1283
+ mid = mem.get("id")
1284
+ if mid:
1285
+ self._retain_task(asyncio.create_task(self._touch_memory_safe(mid)))
1286
+ return memories
1287
+ except Exception as exc:
1288
+ logger.warning("Vector recall failed, falling back to graph-only: %s", exc)
1289
+
1290
+ # Fallback: graph-only keyword/entity recall
1291
+ if hybrid:
1292
+ memories = await self._recall_lexical(
1293
+ query, min(max(limit * 5, limit), max(100, limit)),
1294
+ min_strength, min_confidence, person_scope,
1295
+ excluded_sources, excluded_metadata_markers)
1296
+ if memories:
1297
+ memories = await self._filter_erased_source_memories(memories)
1298
+ if candidates_only:
1299
+ return memories[:limit]
1300
+ memories = await self._maybe_rerank(
1301
+ query, memories, limit, strength_ranking=strength_ranking)
1302
+ memories = await self._filter_erased_source_memories(memories)
1303
+ memories.sort(key=lambda m: m.get("relevance", 0), reverse=True)
1304
+ memories = memories[:limit]
1305
+ for mem in memories:
1306
+ if mem.get("id"):
1307
+ self._retain_task(asyncio.create_task(self._touch_memory_safe(mem["id"])))
1308
+ return memories
1309
+ async with self.driver.session(database=self.database) as session:
1310
+ if person_scope:
1311
+ result = await session.run(
1312
+ """
1313
+ MATCH (m:Memory)-[:ABOUT]->(:Person {id: $person_id})
1314
+ WHERE m.strength >= $min_strength
1315
+ AND (size($exclude_source_uris) = 0 OR NOT (
1316
+ coalesce(m.source_uri, "") IN
1317
+ $exclude_source_uris))
1318
+ AND (size($exclude_metadata_markers) = 0 OR
1319
+ NONE(marker IN $exclude_metadata_markers
1320
+ WHERE toLower(coalesce(m.metadata, ""))
1321
+ CONTAINS marker))
1322
+ AND toLower(m.content) CONTAINS toLower($search_text)
1323
+ AND m.superseded_by IS NULL
1324
+ AND NOT m.epistemic_state IN ["stale", "superseded", "deprecated", "archived"]
1325
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1326
+ WITH m, collect(e.name) AS entity_names
1327
+ RETURN m {.*, entities: entity_names,
1328
+ person_id: $person_id} AS memory,
1329
+ m.effective_confidence AS relevance
1330
+ ORDER BY relevance DESC
1331
+ LIMIT $limit
1332
+ """,
1333
+ search_text=query,
1334
+ limit=limit,
1335
+ min_strength=min_strength,
1336
+ person_id=person_scope,
1337
+ exclude_source_uris=excluded_sources,
1338
+ exclude_metadata_markers=excluded_metadata_markers,
1339
+ )
1340
+ else:
1341
+ result = await session.run(
1342
+ """
1343
+ MATCH (m:Memory)
1344
+ WHERE m.strength >= $min_strength
1345
+ AND (size($exclude_source_uris) = 0 OR NOT (
1346
+ coalesce(m.source_uri, "") IN
1347
+ $exclude_source_uris))
1348
+ AND (size($exclude_metadata_markers) = 0 OR
1349
+ NONE(marker IN $exclude_metadata_markers
1350
+ WHERE toLower(coalesce(m.metadata, ""))
1351
+ CONTAINS marker))
1352
+ AND toLower(m.content) CONTAINS toLower($search_text)
1353
+ AND m.superseded_by IS NULL
1354
+ AND NOT m.epistemic_state IN ["stale", "superseded", "deprecated", "archived"]
1355
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1356
+ WITH m, collect(e.name) AS entity_names
1357
+ RETURN m {.*, entities: entity_names} AS memory,
1358
+ m.effective_confidence AS relevance
1359
+ ORDER BY relevance DESC
1360
+ LIMIT $limit
1361
+ """,
1362
+ search_text=query,
1363
+ limit=limit,
1364
+ min_strength=min_strength,
1365
+ exclude_source_uris=excluded_sources,
1366
+ exclude_metadata_markers=excluded_metadata_markers,
1367
+ )
1368
+ memories = []
1369
+ async for record in result:
1370
+ mem = record["memory"]
1371
+ mem["relevance"] = record.get("relevance", mem.get("strength", 0.5))
1372
+ effective_confidence = float(mem.get("effective_confidence", mem.get("strength", 1.0)))
1373
+ if effective_confidence < min_confidence:
1374
+ continue
1375
+ memories.append(mem)
1376
+ memories = await self._filter_erased_source_memories(memories)
1377
+ if candidates_only:
1378
+ return memories[:limit]
1379
+ # Fire-and-forget touch_memory for each recalled result
1380
+ for mem in memories:
1381
+ mid = mem.get("id")
1382
+ if mid:
1383
+ self._retain_task(asyncio.create_task(self._touch_memory_safe(mid)))
1384
+ return memories
1385
+
1386
+ async def _recall_lexical(
1387
+ self, query, limit, min_strength, min_confidence,
1388
+ person_scope, excluded_sources, excluded_metadata_markers,
1389
+ ):
1390
+ """Native source-text candidates with the same authoritative boundary.
1391
+
1392
+ The full-text index is maintained with graph writes, so an embedding
1393
+ backlog does not hide newly stored facts. Filtering happens before
1394
+ content reaches Python/reranking. A missing/populating index degrades
1395
+ to the existing vector/keyword path.
1396
+ """
1397
+ from .recall import FULLTEXT_INDEX, lexical_query
1398
+ search_text = lexical_query(query)
1399
+ if not search_text:
1400
+ return []
1401
+
1402
+ async def search():
1403
+ async with self.driver.session(database=self.database) as session:
1404
+ result = await session.run(
1405
+ """
1406
+ CALL db.index.fulltext.queryNodes($index_name, $search_text)
1407
+ YIELD node AS m, score
1408
+ WHERE m:Memory
1409
+ AND ($person_id IS NULL OR EXISTS {
1410
+ MATCH (m)-[:ABOUT]->(:Person {id: $person_id})
1411
+ })
1412
+ AND coalesce(m.strength, 1.0) >= $min_strength
1413
+ AND coalesce(m.effective_confidence, m.strength, 1.0) >= $min_confidence
1414
+ AND m.superseded_by IS NULL
1415
+ AND NOT coalesce(m.epistemic_state, "inferred") IN
1416
+ ["stale", "superseded", "deprecated", "archived"]
1417
+ AND (size($exclude_source_uris) = 0 OR NOT (
1418
+ coalesce(m.source_uri, "") IN $exclude_source_uris))
1419
+ AND (size($exclude_metadata_markers) = 0 OR
1420
+ NONE(marker IN $exclude_metadata_markers
1421
+ WHERE toLower(coalesce(m.metadata, "")) CONTAINS marker))
1422
+ WITH m, score
1423
+ ORDER BY score DESC
1424
+ LIMIT $limit
1425
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1426
+ WITH m, score, collect(DISTINCT e.name) AS entity_names
1427
+ RETURN m {.*, entities: entity_names} AS memory,
1428
+ score AS lexical_score
1429
+ ORDER BY lexical_score DESC
1430
+ """,
1431
+ index_name=FULLTEXT_INDEX, search_text=search_text,
1432
+ limit=limit, person_id=person_scope,
1433
+ min_strength=min_strength, min_confidence=min_confidence,
1434
+ exclude_source_uris=excluded_sources,
1435
+ exclude_metadata_markers=excluded_metadata_markers,
1436
+ )
1437
+ memories = []
1438
+ async for record in result:
1439
+ mem = dict(record["memory"])
1440
+ mem["relevance"] = float(record["lexical_score"])
1441
+ mem["retrieval_method"] = "lexical"
1442
+ memories.append(mem)
1443
+ return memories
1444
+ try:
1445
+ return await asyncio.wait_for(search(), timeout=1.2)
1446
+ except Exception as exc:
1447
+ logger.warning("Lexical recall unavailable, using existing recall: %s", type(exc).__name__)
1448
+ return []
1449
+
1450
+ def _retain_task(self, task) -> None:
1451
+ """Keep a strong ref to a fire-and-forget task until it finishes; the
1452
+ loop only weak-refs it, so an unreferenced recall-touch task can be
1453
+ GC-cancelled mid neo4j session ("Task was destroyed but it is
1454
+ pending" + an abandoned session)."""
1455
+ if not hasattr(self, "_bg_tasks"):
1456
+ self._bg_tasks = set()
1457
+ self._bg_tasks.add(task)
1458
+ task.add_done_callback(self._bg_tasks.discard)
1459
+
1460
+ async def _touch_memory_safe(self, memory_id: str) -> None:
1461
+ """Touch a memory, logging but not raising on failure."""
1462
+ try:
1463
+ await self.touch_memory(memory_id)
1464
+ except Exception as exc:
1465
+ logger.debug("touch_memory failed for %s: %s", memory_id, exc)
1466
+
1467
+ async def _maybe_rerank(
1468
+ self, query, memories, limit, *, strength_ranking=False,
1469
+ ):
1470
+ # The same selector is used for mixed source/belief turn context.
1471
+ from .selection import RecallSelector
1472
+ selector = getattr(self, "_recall_selector", None)
1473
+ if selector is None:
1474
+ selector = self._recall_selector = RecallSelector(
1475
+ getattr(self, "_rerank_fn", None),
1476
+ calibration_metadata=getattr(self, "_rerank_calibration_metadata", None),
1477
+ logger=logger)
1478
+ return await selector.rerank(
1479
+ query, memories, limit, strength_ranking=strength_ranking)
1480
+
1481
+ # ------------------------------------------------------------------
1482
+ # Decay & pruning
1483
+ # ------------------------------------------------------------------
1484
+
1485
+ @staticmethod
1486
+ def _compute_decay_factor(
1487
+ importance: float,
1488
+ days_elapsed: float,
1489
+ recalls: int,
1490
+ half_life_days: float,
1491
+ memory_type: str = "episodic",
1492
+ semantic_half_life_days: Optional[float] = None,
1493
+ ) -> float:
1494
+ """Compute Ebbinghaus decay for a memory.
1495
+
1496
+ Formula: strength = importance * e^(-lambda * days) * (1 + recalls * 0.2)
1497
+
1498
+ Where lambda = ln(2) / half_life_days. Identity memories never decay;
1499
+ procedural memories decay at half the normal rate; fact/semantic
1500
+ memories use their own half-life when one is given (defaults to the
1501
+ episodic value, i.e. no behavior change). Result is capped at 1.0.
1502
+
1503
+ Args:
1504
+ importance: Initial importance value (0-1)
1505
+ days_elapsed: Days since last access
1506
+ recalls: Number of times the memory has been recalled
1507
+ half_life_days: Days for strength to halve (default 7)
1508
+ memory_type: One of "identity", "procedural", "episodic",
1509
+ "semantic", "fact"
1510
+ semantic_half_life_days: Half-life for fact/semantic memories
1511
+ (None = same as half_life_days)
1512
+
1513
+ Returns:
1514
+ New strength value in [0, 1].
1515
+ """
1516
+ if memory_type == "identity":
1517
+ return float(importance)
1518
+
1519
+ lambda_base = math.log(2) / max(half_life_days, 0.001)
1520
+ if memory_type == "procedural":
1521
+ # Procedural memories decay at half the normal rate
1522
+ lambda_val = lambda_base / 2
1523
+ elif memory_type in ("fact", "semantic"):
1524
+ sem_half_life = (semantic_half_life_days
1525
+ if semantic_half_life_days is not None
1526
+ else half_life_days)
1527
+ lambda_val = math.log(2) / max(sem_half_life, 0.001)
1528
+ else:
1529
+ lambda_val = lambda_base
1530
+
1531
+ strength = importance * math.exp(-lambda_val * max(days_elapsed, 0)) * (1.0 + recalls * 0.2)
1532
+ return min(1.0, max(0.0, strength))
1533
+
1534
+ async def decay_memories(
1535
+ self,
1536
+ half_life_days: Optional[float] = None,
1537
+ semantic_half_life_days: Optional[float] = None,
1538
+ ) -> None:
1539
+ """Apply Ebbinghaus forgetting curve to all non-identity, non-protected memories.
1540
+
1541
+ Formula: strength = importance * e^(-lambda * days) * (1 + recalls * 0.2)
1542
+
1543
+ Where lambda = ln(2) / half_life_days.
1544
+ - Identity memories are skipped (never decay).
1545
+ - Protected memories are skipped.
1546
+ - Procedural memories use lambda / 2 (half rate).
1547
+ - Fact/semantic memories use their own half-life (defaults to the
1548
+ episodic value — distilled knowledge can be made to outlive the
1549
+ episodes it came from by raising COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS).
1550
+ - Result is capped at 1.0.
1551
+
1552
+ Half-lives resolve, in order: explicit argument, environment
1553
+ (COLONY_DECAY_HALF_LIFE_DAYS / COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS),
1554
+ then the historical default of 7 days. Strength is RECOMPUTED from
1555
+ importance on every pass (not compounded), so raising a half-life
1556
+ retroactively resurrects previously-decayed strength — which is why
1557
+ half-life tuning must land before pruning goes live, never after.
1558
+
1559
+ This pass is the single writer for memory decay; nothing else may
1560
+ call it as a side effect (see StrategyAdjuster._decay_signals,
1561
+ retired for exactly that reason).
1562
+ """
1563
+ if half_life_days is None:
1564
+ try:
1565
+ half_life_days = float(os.environ.get(
1566
+ "COLONY_DECAY_HALF_LIFE_DAYS", "7"))
1567
+ except (TypeError, ValueError):
1568
+ half_life_days = 7.0
1569
+ if semantic_half_life_days is None:
1570
+ _sem_env = os.environ.get("COLONY_DECAY_HALF_LIFE_SEMANTIC_DAYS", "")
1571
+ try:
1572
+ semantic_half_life_days = (
1573
+ float(_sem_env) if _sem_env.strip() else half_life_days)
1574
+ except (TypeError, ValueError):
1575
+ semantic_half_life_days = half_life_days
1576
+
1577
+ lambda_normal = math.log(2) / max(half_life_days, 0.001)
1578
+ lambda_procedural = lambda_normal / 2
1579
+ lambda_semantic = math.log(2) / max(semantic_half_life_days, 0.001)
1580
+
1581
+ async with self.driver.session(database=self.database) as session:
1582
+ # First pass: update strength
1583
+ await session.run(
1584
+ """
1585
+ MATCH (m:Memory)
1586
+ WHERE m.type <> 'identity' AND coalesce(m.protected, false) = false
1587
+ WITH m,
1588
+ toFloat(duration.inDays(coalesce(m.accessed_at, m.created_at, datetime()), datetime()).days) AS days_since,
1589
+ CASE WHEN m.type = 'procedural'
1590
+ THEN $lambda_proc
1591
+ WHEN m.type IN ['fact', 'semantic']
1592
+ THEN $lambda_sem
1593
+ ELSE $lambda_norm
1594
+ END AS lam
1595
+ WITH m,
1596
+ coalesce(m.importance, 1.0) *
1597
+ exp(-lam * days_since) *
1598
+ (1.0 + coalesce(m.recalls, 0) * 0.2) AS new_strength
1599
+ SET m.strength = CASE
1600
+ WHEN new_strength > 1.0 THEN 1.0
1601
+ WHEN new_strength < 0.0 THEN 0.0
1602
+ ELSE new_strength
1603
+ END
1604
+ """,
1605
+ lambda_norm=lambda_normal,
1606
+ lambda_proc=lambda_procedural,
1607
+ lambda_sem=lambda_semantic,
1608
+ )
1609
+
1610
+ # Second pass: update effective_confidence in batches
1611
+ await self._update_effective_confidence_batch()
1612
+
1613
+ async def prune_weak_memories(
1614
+ self,
1615
+ threshold: float = 0.05,
1616
+ *,
1617
+ dry_run: bool = False,
1618
+ max_delete: int = 500,
1619
+ ) -> Dict[str, Any]:
1620
+ """Delete memories whose strength has decayed below *threshold*.
1621
+
1622
+ Only targets memories in ``inferred``, ``observed``, or ``stale``
1623
+ epistemic states. Skips protected memories, ``corroborated``,
1624
+ ``verified``, and fully terminal states (``superseded``,
1625
+ ``deprecated``, ``archived``).
1626
+
1627
+ A deleted memory's vector-store entry is removed too (same coupling
1628
+ as :meth:`archive_memories`), so pruning does not accumulate orphan
1629
+ vectors that keep matching in ANN search. Deletion is capped at
1630
+ *max_delete* per call (weakest first); ``dry_run=True`` only counts.
1631
+
1632
+ Fails closed on Neo4j errors: exceptions propagate to the caller
1633
+ and nothing further is deleted; a vector is only removed after its
1634
+ graph node is gone.
1635
+
1636
+ Returns:
1637
+ Dict with ``matched`` (total below threshold), ``deleted``,
1638
+ ``dry_run``, and the capped candidate ``ids``.
1639
+ """
1640
+ async with self.driver.session(database=self.database) as session:
1641
+ result = await session.run(
1642
+ """
1643
+ MATCH (m:Memory)
1644
+ WHERE m.strength < $threshold
1645
+ AND coalesce(m.protected, false) = false
1646
+ AND m.epistemic_state IN ["inferred", "observed", "stale"]
1647
+ WITH m ORDER BY m.strength ASC
1648
+ RETURN collect(m.id)[0..$max_delete] AS ids,
1649
+ count(m) AS matched
1650
+ """,
1651
+ threshold=threshold,
1652
+ max_delete=max_delete,
1653
+ )
1654
+ record = await result.single()
1655
+ ids = list(record["ids"]) if record else []
1656
+ matched = int(record["matched"]) if record else 0
1657
+
1658
+ if dry_run:
1659
+ return {"matched": matched, "deleted": 0, "dry_run": True,
1660
+ "ids": ids}
1661
+
1662
+ deleted = 0
1663
+ for memory_id in ids:
1664
+ async with self.driver.session(database=self.database) as session:
1665
+ await session.run(
1666
+ "MATCH (m:Memory {id: $memory_id}) DETACH DELETE m",
1667
+ memory_id=memory_id,
1668
+ )
1669
+ deleted += 1
1670
+ # Vector removal only after the graph node is gone; a failure
1671
+ # here leaves an orphan vector (swept later), never a memory
1672
+ # that recalls without a backing node.
1673
+ if self._vector_store is not None:
1674
+ try:
1675
+ from apsimo.vector.collections import Collection
1676
+ await self._vector_store.delete(
1677
+ collection=Collection.MEMORIES,
1678
+ id=memory_id,
1679
+ )
1680
+ except Exception as exc:
1681
+ logger.debug(
1682
+ "Failed to remove pruned memory %s from vector "
1683
+ "store: %s", memory_id, exc)
1684
+ return {"matched": matched, "deleted": deleted, "dry_run": False,
1685
+ "ids": ids}
1686
+
1687
+ async def vacuum_orphan_vectors(
1688
+ self,
1689
+ *,
1690
+ dry_run: bool = False,
1691
+ max_delete: Optional[int] = None,
1692
+ batch_size: int = 200,
1693
+ batch_sleep_secs: float = 0.05,
1694
+ ) -> Dict[str, Any]:
1695
+ """Delete MEMORIES-collection vectors whose graph node no longer exists.
1696
+
1697
+ Orphan vectors (a node deleted without its vector — pre-coupling
1698
+ prunes, or a vector removal that failed after node deletion) keep
1699
+ matching in ANN search and then silently vanish at hydration,
1700
+ stealing recall slots from real memories.
1701
+
1702
+ Fails closed: the graph-id read is NOT exception-wrapped — a Neo4j
1703
+ failure aborts the vacuum before any deletion, because an empty or
1704
+ partial graph-id set would classify every vector as an orphan and
1705
+ wipe the store. Deletion runs in batches of *batch_size* with
1706
+ *batch_sleep_secs* between batches; *max_delete* bounds one run.
1707
+
1708
+ Returns:
1709
+ Dict with ``vectors`` (total scanned), ``orphans`` (total found),
1710
+ ``deleted``, ``dry_run``, and a capped ``ids`` sample.
1711
+ """
1712
+ if self._vector_store is None:
1713
+ return {"available": False, "vectors": 0, "orphans": 0,
1714
+ "deleted": 0, "dry_run": dry_run, "ids": []}
1715
+
1716
+ from apsimo.vector.collections import Collection
1717
+ vector_ids = await self._vector_store.list_ids(Collection.MEMORIES)
1718
+
1719
+ graph_ids: set = set()
1720
+ async with self.driver.session(database=self.database) as session:
1721
+ result = await session.run("MATCH (m:Memory) RETURN m.id AS id")
1722
+ async for record in result:
1723
+ if record["id"]:
1724
+ graph_ids.add(record["id"])
1725
+
1726
+ orphans = [vid for vid in vector_ids if vid not in graph_ids]
1727
+ total_orphans = len(orphans)
1728
+ if max_delete is not None:
1729
+ orphans = orphans[:max_delete]
1730
+
1731
+ if dry_run:
1732
+ return {"available": True, "vectors": len(vector_ids),
1733
+ "orphans": total_orphans, "deleted": 0, "dry_run": True,
1734
+ "ids": orphans[:20]}
1735
+
1736
+ deleted = 0
1737
+ for start in range(0, len(orphans), batch_size):
1738
+ for vid in orphans[start:start + batch_size]:
1739
+ await self._vector_store.delete(
1740
+ collection=Collection.MEMORIES, id=vid)
1741
+ deleted += 1
1742
+ if start + batch_size < len(orphans) and batch_sleep_secs > 0:
1743
+ await asyncio.sleep(batch_sleep_secs)
1744
+
1745
+ return {"available": True, "vectors": len(vector_ids),
1746
+ "orphans": total_orphans, "deleted": deleted,
1747
+ "dry_run": False, "ids": orphans[:20]}
1748
+
1749
+ async def touch_memory(self, memory_id: str) -> None:
1750
+ """Record a memory recall, incrementing its recall counter and updating accessed_at.
1751
+
1752
+ Calling this after retrieval ensures the recall bonus in the Ebbinghaus
1753
+ formula is applied on the next decay pass, slowing future decay.
1754
+
1755
+ Args:
1756
+ memory_id: UUID of the Memory node to touch.
1757
+ """
1758
+ async with self.driver.session(database=self.database) as session:
1759
+ await session.run(
1760
+ """
1761
+ MATCH (m:Memory {id: $memory_id})
1762
+ SET m.recalls = coalesce(m.recalls, 0) + 1,
1763
+ m.accessed_at = datetime()
1764
+ """,
1765
+ memory_id=memory_id,
1766
+ )
1767
+
1768
+ async def _update_effective_confidence_batch(self, batch_size: int = 1000) -> None:
1769
+ """Update effective_confidence for all memories in batches.
1770
+
1771
+ Each batch is one read and one write round trip: the confidences are
1772
+ computed in Python, then written back with a single UNWIND. Writing
1773
+ one auto-commit statement per memory made this pass scale with the
1774
+ memory count (thousands of round trips), which overran the autonomy
1775
+ tick budget and cancelled the whole tick every time it ran. The read
1776
+ is ordered by id so SKIP/LIMIT paging is stable while we write.
1777
+ """
1778
+ from datetime import datetime as _dt, timezone as _tz
1779
+ now = _dt.now(_tz.utc)
1780
+ offset = 0
1781
+ while True:
1782
+ async with self.driver.session(database=self.database) as session:
1783
+ result = await session.run(
1784
+ """
1785
+ MATCH (m:Memory)
1786
+ RETURN m {
1787
+ .id, .base_confidence, .source_reliability,
1788
+ .corroboration_count, .contradiction_count,
1789
+ .recalls, .last_verified_at, .created_at,
1790
+ .epistemic_state
1791
+ } AS mem
1792
+ ORDER BY m.id
1793
+ SKIP $offset LIMIT $limit
1794
+ """,
1795
+ offset=offset,
1796
+ limit=batch_size,
1797
+ )
1798
+ rows = [dict(r["mem"]) async for r in result]
1799
+ if not rows:
1800
+ break
1801
+ updates = []
1802
+ for row in rows:
1803
+ if row.get("id") is None:
1804
+ continue
1805
+ new_confidence = self.compute_effective_confidence(
1806
+ base_confidence=row.get("base_confidence") or 1.0,
1807
+ source_reliability=row.get("source_reliability") or 0.5,
1808
+ corroboration_count=row.get("corroboration_count") or 0,
1809
+ contradiction_count=row.get("contradiction_count") or 0,
1810
+ recalls=row.get("recalls") or 0,
1811
+ last_verified_at=row.get("last_verified_at"),
1812
+ created_at=row.get("created_at") or now,
1813
+ epistemic_state=row.get("epistemic_state") or "inferred",
1814
+ now=now,
1815
+ )
1816
+ updates.append({"id": row["id"], "effective_confidence": new_confidence})
1817
+ if updates:
1818
+ await session.run(
1819
+ """
1820
+ UNWIND $updates AS u
1821
+ MATCH (m:Memory {id: u.id})
1822
+ SET m.effective_confidence = u.effective_confidence
1823
+ """,
1824
+ updates=updates,
1825
+ )
1826
+ if len(rows) < batch_size:
1827
+ break
1828
+ offset += batch_size
1829
+
1830
+ async def verify_memory(self, memory_id: str) -> None:
1831
+ """Mark a memory as manually verified.
1832
+
1833
+ Sets last_verified_at, transitions epistemic_state to ``verified``
1834
+ if currently in an active state, and floors effective_confidence
1835
+ at 0.9.
1836
+ """
1837
+ async with self.driver.session(database=self.database) as session:
1838
+ await session.run(
1839
+ """
1840
+ MATCH (m:Memory {id: $memory_id})
1841
+ SET m.last_verified_at = datetime(),
1842
+ m.epistemic_state = CASE
1843
+ WHEN m.epistemic_state IN ["inferred", "observed", "corroborated"]
1844
+ THEN "verified"
1845
+ ELSE m.epistemic_state
1846
+ END,
1847
+ m.effective_confidence = CASE
1848
+ WHEN m.effective_confidence < 0.9 THEN 0.9
1849
+ ELSE m.effective_confidence
1850
+ END
1851
+ """,
1852
+ memory_id=memory_id,
1853
+ )
1854
+
1855
+ async def get_memory(self, memory_id: str) -> Optional[Dict[str, Any]]:
1856
+ """Fetch a single memory by ID.
1857
+
1858
+ Returns:
1859
+ Memory dict or None if not found.
1860
+ """
1861
+ async with self.driver.session(database=self.database) as session:
1862
+ result = await session.run(
1863
+ """
1864
+ MATCH (m:Memory {id: $memory_id})
1865
+ OPTIONAL MATCH (m)-[:MENTIONS]->(e:Entity)
1866
+ WITH m, collect(e.name) AS entity_names
1867
+ RETURN m {.*, entities: entity_names} AS memory
1868
+ """,
1869
+ memory_id=memory_id,
1870
+ )
1871
+ record = await result.single()
1872
+ return dict(record["memory"]) if record else None
1873
+
1874
+ async def transition_epistemic_state(
1875
+ self,
1876
+ memory_id: str,
1877
+ new_state: str,
1878
+ superseded_by: Optional[str] = None,
1879
+ ) -> None:
1880
+ """Transition a memory to a new epistemic state.
1881
+
1882
+ Args:
1883
+ memory_id: UUID of the memory to transition.
1884
+ new_state: Target EpistemicState value.
1885
+ superseded_by: Optional UUID of the memory that superseded this one.
1886
+ """
1887
+ async with self.driver.session(database=self.database) as session:
1888
+ await session.run(
1889
+ """
1890
+ MATCH (m:Memory {id: $memory_id})
1891
+ SET m.epistemic_state = $new_state
1892
+ WITH m
1893
+ FOREACH (_ IN CASE WHEN $superseded_by IS NOT NULL THEN [1] ELSE [] END |
1894
+ SET m.superseded_by = $superseded_by
1895
+ )
1896
+ """,
1897
+ memory_id=memory_id,
1898
+ new_state=new_state,
1899
+ superseded_by=superseded_by,
1900
+ )
1901
+
1902
+ async def archive_memories(self, max_age_days: int = 30) -> int:
1903
+ """Archive memories that have been in a terminal state for too long.
1904
+
1905
+ Relabels :Memory to :ArchivedMemory, copies key relationships,
1906
+ removes from vector store, and deletes the original.
1907
+
1908
+ Returns:
1909
+ Number of memories archived.
1910
+ """
1911
+ archived = 0
1912
+ async with self.driver.session(database=self.database) as session:
1913
+ result = await session.run(
1914
+ """
1915
+ MATCH (m:Memory)
1916
+ WHERE m.epistemic_state IN ["superseded", "deprecated", "stale"]
1917
+ AND duration.inDays(m.accessed_at, datetime()).days >= $max_age_days
1918
+ RETURN m.id AS id
1919
+ """,
1920
+ max_age_days=max_age_days,
1921
+ )
1922
+ memory_ids = [r["id"] async for r in result]
1923
+
1924
+ for memory_id in memory_ids:
1925
+ try:
1926
+ async with self.driver.session(database=self.database) as session:
1927
+ # Copy to ArchivedMemory with key relationships
1928
+ await session.run(
1929
+ """
1930
+ MATCH (m:Memory {id: $memory_id})
1931
+ CREATE (a:ArchivedMemory)
1932
+ SET a = properties(m)
1933
+ WITH m, a
1934
+ OPTIONAL MATCH (m)-[r:MENTIONS]->(e:Entity)
1935
+ FOREACH (_ IN CASE WHEN e IS NOT NULL THEN [1] ELSE [] END |
1936
+ CREATE (a)-[:MENTIONS {created_at: datetime()}]->(e)
1937
+ )
1938
+ WITH m, a
1939
+ OPTIONAL MATCH (m)-[r:ABOUT]->(p:Person)
1940
+ FOREACH (_ IN CASE WHEN p IS NOT NULL THEN [1] ELSE [] END |
1941
+ CREATE (a)-[:ABOUT {created_at: datetime()}]->(p)
1942
+ )
1943
+ WITH m, a
1944
+ OPTIONAL MATCH (m)-[r:SUPERSEDES]->(old:Memory)
1945
+ FOREACH (_ IN CASE WHEN old IS NOT NULL THEN [1] ELSE [] END |
1946
+ CREATE (a)-[:SUPERSEDES {superseded_at: r.superseded_at}]->(old)
1947
+ )
1948
+ WITH m, a
1949
+ OPTIONAL MATCH (m)-[r:DERIVED_FROM]->(fa:FileAnchor)
1950
+ FOREACH (_ IN CASE WHEN fa IS NOT NULL THEN [1] ELSE [] END |
1951
+ CREATE (a)-[:DERIVED_FROM {derivation_type: r.derivation_type}]->(fa)
1952
+ )
1953
+ DETACH DELETE m
1954
+ """,
1955
+ memory_id=memory_id,
1956
+ )
1957
+ # Remove from vector store
1958
+ if self._vector_store is not None:
1959
+ try:
1960
+ from apsimo.vector.collections import Collection
1961
+ await self._vector_store.delete(
1962
+ collection=Collection.MEMORIES,
1963
+ id=memory_id,
1964
+ )
1965
+ except Exception as exc:
1966
+ logger.debug("Failed to remove archived memory from vector store: %s", exc)
1967
+ archived += 1
1968
+ except Exception as exc:
1969
+ logger.warning("Failed to archive memory %s: %s", memory_id, exc)
1970
+
1971
+ return archived
1972
+
1973
+ # ------------------------------------------------------------------
1974
+ # Traversal
1975
+ # ------------------------------------------------------------------
1976
+
1977
+ # Allowlist of permitted Cypher templates. Each entry is an exact string
1978
+ # match against the cypher argument. This prevents run_query from being
1979
+ # used as a Cypher injection sink while retaining the escape hatch for
1980
+ # known safe ad-hoc queries added here explicitly.
1981
+ _ALLOWED_CYPHER: frozenset = frozenset({
1982
+ # (similarity_threshold Config write removed: adjustments now go
1983
+ # through the AdaptiveParamStore, which consumers actually read.)
1984
+ # StrategyAdjuster._recalibrate_baselines — recalculate baseline signals
1985
+ """MATCH (p:Person)-[:EXHIBITED]->(s:Signal)
1986
+ WHERE s.timestamp >= datetime() - duration({days: 30})
1987
+ WITH p.id AS pid, s.signal_type AS stype, avg(s.normalized_value) AS baseline_val
1988
+ MERGE (b:Baseline {person_id: pid, signal_type: stype})
1989
+ SET b.value = baseline_val, b.updated_at = datetime()
1990
+ RETURN count(b) AS updated_baselines""",
1991
+ # Neo4jCognitionSeeder.seed — upsert BootstrapEvent node on first boot
1992
+ """
1993
+ MERGE (b:BootstrapEvent {colony_id: $colony_id})
1994
+ SET b.colony_name = $colony_name,
1995
+ b.colony_version = $colony_version,
1996
+ b.network_id = $network_id,
1997
+ b.corpus_version = $corpus_version,
1998
+ b.bootstrapped_at = $bootstrapped_at,
1999
+ b.layer_count = $layer_count,
2000
+ b.endpoint_count = $endpoint_count
2001
+ RETURN b.colony_id
2002
+ """,
2003
+
2004
+ # ConnectionDiscoverer._find_temporal_patterns (with person_id)
2005
+ "MATCH (m1:Memory)-[:ABOUT]->(p:Person {id: $person_id})\n"
2006
+ "MATCH (m2:Memory)-[:ABOUT]->(p)\n"
2007
+ "WHERE m1.id < m2.id\n"
2008
+ " AND abs(duration.between(m1.created_at, m2.created_at).hours) <= $window_hours\n"
2009
+ " AND m1.created_at >= datetime() - duration({days: $lookback_days})\n"
2010
+ "WITH m1, m2, count(*) AS co_occurrences\n"
2011
+ "WHERE co_occurrences >= $min_count\n"
2012
+ "RETURN m1.id AS source_id, m2.id AS target_id,\n"
2013
+ " m1.type AS source_type, m2.type AS target_type,\n"
2014
+ " m1.metadata AS source_meta, m2.metadata AS target_meta,\n"
2015
+ " co_occurrences,\n"
2016
+ " toFloat(co_occurrences) / $lookback_days AS daily_rate\n"
2017
+ "ORDER BY daily_rate DESC\n"
2018
+ "LIMIT 20",
2019
+ # ConnectionDiscoverer._find_temporal_patterns (without person_id)
2020
+ "MATCH (m1:Memory), (m2:Memory)\n"
2021
+ "WHERE m1.id < m2.id\n"
2022
+ " AND abs(duration.between(m1.created_at, m2.created_at).hours) <= $window_hours\n"
2023
+ " AND m1.created_at >= datetime() - duration({days: $lookback_days})\n"
2024
+ "WITH m1, m2, count(*) AS co_occurrences\n"
2025
+ "WHERE co_occurrences >= $min_count\n"
2026
+ "RETURN m1.id AS source_id, m2.id AS target_id,\n"
2027
+ " m1.type AS source_type, m2.type AS target_type,\n"
2028
+ " m1.metadata AS source_meta, m2.metadata AS target_meta,\n"
2029
+ " co_occurrences,\n"
2030
+ " toFloat(co_occurrences) / $lookback_days AS daily_rate\n"
2031
+ "ORDER BY daily_rate DESC\n"
2032
+ "LIMIT 20",
2033
+ # ConnectionDiscoverer._find_entity_patterns
2034
+ "MATCH (e:Entity)<-[:MENTIONS]-(m:Memory)\n"
2035
+ "WHERE ($person_id IS NULL OR (m)-[:ABOUT]->(:Person {id: $person_id}))\n"
2036
+ "WITH e, collect(DISTINCT m.id) AS mems\n"
2037
+ "WITH e, mems, size(mems) AS mem_count\n"
2038
+ "WHERE mem_count >= 2\n"
2039
+ "RETURN e.name AS entity_name,\n"
2040
+ " mem_count AS occurrence_count,\n"
2041
+ " mems[0..5] AS evidence_sample\n"
2042
+ "ORDER BY mem_count DESC\n"
2043
+ "LIMIT 20",
2044
+ # ConnectionDiscoverer._find_behavioral_patterns
2045
+ "MATCH (p:Person {id: $person_id})-[:EXHIBITED]->(s1:Signal)\n"
2046
+ "MATCH (p)-[:EXHIBITED]->(s2:Signal)\n"
2047
+ "WHERE s1.signal_type <> s2.signal_type\n"
2048
+ " AND s2.timestamp > s1.timestamp\n"
2049
+ " AND duration.between(s1.timestamp, s2.timestamp).hours <= $window_hours\n"
2050
+ "WITH s1.signal_type AS type_a, s2.signal_type AS type_b,\n"
2051
+ " count(*) AS occurrences,\n"
2052
+ " avg(s2.normalized_value - s1.normalized_value) AS avg_delta,\n"
2053
+ " collect(s1.id)[0..5] AS evidence\n"
2054
+ "WHERE occurrences >= $min_occurrences\n"
2055
+ "RETURN type_a, type_b, occurrences, avg_delta, evidence\n"
2056
+ "ORDER BY occurrences DESC\n"
2057
+ "LIMIT 15",
2058
+ # Consolidator._detect_conflicts
2059
+ "MATCH (m1:Memory)-[:MENTIONS]->(e:Entity)<-[:MENTIONS]-(m2:Memory) "
2060
+ "WHERE id(m1) < id(m2) "
2061
+ "AND NOT (m1)-[:MERGED_INTO]-() AND NOT (m2)-[:MERGED_INTO]-() "
2062
+ "RETURN m1.id AS id_a, m2.id AS id_b, e.name AS entity, "
2063
+ " m1.content AS content_a, m2.content AS content_b",
2064
+ })
2065
+
2066
+ # Additional allowlisted queries registered by callers via
2067
+ # register_allowed_cypher(). Same discipline as _ALLOWED_CYPHER: exact,
2068
+ # fully-parameterized strings only, never built from user input. This
2069
+ # keeps each query single-sourced next to the code that owns it instead
2070
+ # of duplicating string literals here.
2071
+ _EXTRA_ALLOWED_CYPHER: set = set()
2072
+
2073
+ @classmethod
2074
+ def register_allowed_cypher(cls, cypher: str) -> str:
2075
+ """Register one exact, parameterized Cypher string for run_query."""
2076
+ cls._EXTRA_ALLOWED_CYPHER.add(cypher)
2077
+ return cypher
2078
+
2079
+ async def run_query(self, cypher: str, params: dict) -> List[Dict[str, Any]]:
2080
+ """Execute a Cypher query and return results as dicts.
2081
+
2082
+ WARNING: The ``cypher`` argument must never be constructed from
2083
+ user-controlled input. Only queries listed in ``_ALLOWED_CYPHER``
2084
+ are permitted; all others raise ``ValueError``. To add a new query,
2085
+ add its exact string to ``_ALLOWED_CYPHER`` after security review.
2086
+
2087
+ Args:
2088
+ cypher: Exact Cypher query string (must be in _ALLOWED_CYPHER).
2089
+ params: Dict of $param bindings — always parameterized, never interpolated.
2090
+
2091
+ Returns:
2092
+ List of result records as plain dicts.
2093
+
2094
+ Raises:
2095
+ ValueError: If ``cypher`` is not in the allowlist.
2096
+ """
2097
+ if (cypher not in self._ALLOWED_CYPHER
2098
+ and cypher not in self._EXTRA_ALLOWED_CYPHER):
2099
+ raise ValueError(
2100
+ "run_query: cypher string not in allowlist. "
2101
+ "Add to GraphClient._ALLOWED_CYPHER or register via "
2102
+ "register_allowed_cypher() after security review."
2103
+ )
2104
+ async with self.driver.session(database=self.database) as session:
2105
+ result = await session.run(cypher, **params)
2106
+ return [dict(r) async for r in result]
2107
+
2108
+ # GRAPH-01: server-enforced maximum traversal depth
2109
+ MAX_GRAPH_DEPTH = 10
2110
+
2111
+ async def traverse_memory_connections(
2112
+ self,
2113
+ memory_id: str,
2114
+ max_depth: int = 3,
2115
+ min_strength: float = 0.3,
2116
+ limit: int = 20,
2117
+ ) -> List[Dict[str, Any]]:
2118
+ """Walk multi-hop causal / supporting chains from a memory.
2119
+
2120
+ Uses ``CAUSED_BY``, ``LED_TO``, and ``SUPPORTS`` edge types up to
2121
+ *max_depth* hops, filtering out nodes below *min_strength*.
2122
+
2123
+ Returns:
2124
+ A list of dicts with ``memory``, ``distance``, and
2125
+ ``path_weight`` keys.
2126
+ """
2127
+ # GRAPH-01: clamp depth to server-enforced maximum
2128
+ max_depth = min(int(max_depth), self.MAX_GRAPH_DEPTH)
2129
+ async with self.driver.session(database=self.database) as session:
2130
+ result = await session.run(
2131
+ """
2132
+ MATCH path = (m1:Memory)-[:CAUSED_BY|LED_TO|SUPPORTS*1..$max_depth]->(m2:Memory)
2133
+ WHERE m1.id = $memory_id
2134
+ AND all(node IN nodes(path) WHERE node.strength >= $min_strength)
2135
+ RETURN m2 {.*} AS memory,
2136
+ length(path) AS distance,
2137
+ reduce(w = 1.0, r IN relationships(path) | w * r.weight) AS path_weight
2138
+ ORDER BY path_weight DESC
2139
+ LIMIT $limit
2140
+ """,
2141
+ memory_id=memory_id,
2142
+ max_depth=max_depth,
2143
+ min_strength=min_strength,
2144
+ limit=limit,
2145
+ )
2146
+ return [
2147
+ {
2148
+ "memory": record["memory"],
2149
+ "distance": record["distance"],
2150
+ "path_weight": record["path_weight"],
2151
+ }
2152
+ async for record in result
2153
+ ]
2154
+
2155
+
2156
+ # ------------------------------------------------------------------
2157
+ # Baseline methods (GraphBaselineStore)
2158
+ # ------------------------------------------------------------------
2159
+
2160
+ async def _run_get_baseline(self, person_id: str) -> dict | None:
2161
+ """Read baseline properties from a Person node. Returns None if not found."""
2162
+ try:
2163
+ async with self.driver.session(database=self.database) as session:
2164
+ result = await session.run(GET_BASELINE, person_id=person_id)
2165
+ record = await result.single()
2166
+ if record is None:
2167
+ return None
2168
+ return dict(record)
2169
+ except Exception as exc:
2170
+ logger.debug("_run_get_baseline failed for %s: %s", person_id, exc)
2171
+ return None
2172
+
2173
+ async def _run_update_baseline(
2174
+ self,
2175
+ person_id: str,
2176
+ msg_count: int,
2177
+ length_mean: float,
2178
+ length_m2: float,
2179
+ length_std: float,
2180
+ hour_histogram: str,
2181
+ ) -> None:
2182
+ """Write updated baseline properties to a Person node."""
2183
+ try:
2184
+ async with self.driver.session(database=self.database) as session:
2185
+ await session.run(
2186
+ UPDATE_BASELINE,
2187
+ person_id=person_id,
2188
+ msg_count=msg_count,
2189
+ length_mean=length_mean,
2190
+ length_m2=length_m2,
2191
+ length_std=length_std,
2192
+ hour_histogram=hour_histogram,
2193
+ )
2194
+ except Exception as exc:
2195
+ logger.debug("_run_update_baseline failed for %s: %s", person_id, exc)
2196
+
2197
+ async def list_person_ids(self) -> List[str]:
2198
+ """Return all person IDs that have a Person node in the graph."""
2199
+ query = "MATCH (p:Person) RETURN p.id AS id"
2200
+ async with self.driver.session(database=self.database) as session:
2201
+ result = await session.run(query)
2202
+ records = await result.values()
2203
+ return [r[0] for r in records if r[0]]
2204
+
2205
+ async def delete_person(self, person_id: str) -> bool:
2206
+ """Permanently delete a Person node and all attached relationships."""
2207
+ t0 = time.monotonic()
2208
+ try:
2209
+ async with self.driver.session(database=self.database) as session:
2210
+ result = await session.run(
2211
+ "MATCH (p:Person {id: $person_id}) DETACH DELETE p RETURN count(p) AS deleted",
2212
+ person_id=person_id,
2213
+ )
2214
+ record = await result.single()
2215
+ deleted = record["deleted"] if record else 0
2216
+ logger.debug(
2217
+ "delete_person %s deleted=%d %.1fms",
2218
+ person_id, deleted, (time.monotonic() - t0) * 1000,
2219
+ )
2220
+ return deleted > 0
2221
+ except Exception as exc:
2222
+ logger.error("delete_person failed for %s: %s", person_id, exc)
2223
+ return False
2224
+
2225
+ async def get_people_with_substance(
2226
+ self,
2227
+ min_signals: int = 2,
2228
+ min_memories: int = 2,
2229
+ ) -> List[Dict[str, Any]]:
2230
+ """Return Person nodes that have enough substance to become contacts.
2231
+
2232
+ A person has substance if they have:
2233
+ - a name AND (a phone or email)
2234
+ - OR a name AND (>= min_signals OR >= min_memories)
2235
+ """
2236
+ t0 = time.monotonic()
2237
+ try:
2238
+ async with self.driver.session(database=self.database) as session:
2239
+ result = await session.run(
2240
+ """
2241
+ MATCH (p:Person)
2242
+ WHERE p.name IS NOT NULL
2243
+ OPTIONAL MATCH (p)-[:EXHIBITED]->(s:Signal)
2244
+ OPTIONAL MATCH (m:Memory)-[:ABOUT]->(p)
2245
+ WITH p, count(s) AS sigs, count(m) AS mems
2246
+ WHERE p.phone IS NOT NULL OR p.email IS NOT NULL
2247
+ OR sigs >= $min_signals OR mems >= $min_memories
2248
+ RETURN p.id AS id,
2249
+ p.name AS name,
2250
+ p.phone AS phone,
2251
+ p.email AS email,
2252
+ coalesce(p.score, 0.0) AS score,
2253
+ coalesce(p.tier, 'regular') AS tier,
2254
+ sigs,
2255
+ mems
2256
+ """,
2257
+ min_signals=min_signals,
2258
+ min_memories=min_memories,
2259
+ )
2260
+ people = [dict(r) async for r in result]
2261
+ logger.debug(
2262
+ "get_people_with_substance count=%d %.1fms",
2263
+ len(people), (time.monotonic() - t0) * 1000,
2264
+ )
2265
+ return people
2266
+ except Exception as exc:
2267
+ logger.error("get_people_with_substance failed: %s", exc)
2268
+ return []
2269
+
2270
+ # ------------------------------------------------------------------
2271
+ # Signal & relationship methods (required by SignalCollector,
2272
+ # BaselineStore, and RelationshipScorer)
2273
+ # ------------------------------------------------------------------
2274
+
2275
+ async def store_signal(self, signal: Any) -> str:
2276
+ """Persist a behavioral signal to the graph.
2277
+
2278
+ Creates a :Signal node linked to the :Person via [:EXHIBITED].
2279
+ """
2280
+ from apsimo.intelligence.graph.queries import STORE_SIGNAL, GET_BASELINE, UPDATE_BASELINE
2281
+
2282
+ t0 = time.monotonic()
2283
+ try:
2284
+ async with self.driver.session(database=self.database) as session:
2285
+ result = await session.run(
2286
+ STORE_SIGNAL,
2287
+ person_id=signal.person_id,
2288
+ signal_type=signal.signal_type,
2289
+ raw_value=signal.raw_value,
2290
+ normalized_value=signal.normalized_value,
2291
+ timestamp=signal.timestamp.isoformat(),
2292
+ source=signal.source,
2293
+ )
2294
+ record = await result.single()
2295
+ sid = record["id"] if record else ""
2296
+ logger.debug(
2297
+ "store_signal person=%s type=%s %.1fms",
2298
+ signal.person_id, signal.signal_type, (time.monotonic() - t0) * 1000,
2299
+ )
2300
+ return sid
2301
+ except Exception as exc:
2302
+ logger.error("store_signal failed for person %s: %s", signal.person_id, exc)
2303
+ return ""
2304
+
2305
+ async def get_recent_signals(
2306
+ self, person_id: str, hours: int = 24, signal_type: Optional[str] = None
2307
+ ) -> List[Any]:
2308
+ """Fetch recent signals for a person within a time window.
2309
+
2310
+ Returns a list of Signal dataclass instances from the
2311
+ signal_collector module.
2312
+ """
2313
+ from apsimo.intelligence.mind_model.signal_collector import Signal
2314
+ from datetime import timedelta
2315
+
2316
+ cutoff = (self._utcnow() - timedelta(hours=hours)).isoformat()
2317
+
2318
+ cypher = (
2319
+ "MATCH (p:Person {id: $person_id})-[:EXHIBITED]->(s:Signal)\n"
2320
+ "WHERE s.timestamp >= datetime($cutoff)\n"
2321
+ )
2322
+ if signal_type:
2323
+ cypher += "AND s.signal_type = $signal_type\n"
2324
+ cypher += (
2325
+ "RETURN s.signal_type AS signal_type,\n"
2326
+ " s.raw_value AS raw_value,\n"
2327
+ " s.normalized_value AS normalized_value,\n"
2328
+ " s.timestamp AS timestamp,\n"
2329
+ " s.source AS source\n"
2330
+ "ORDER BY s.timestamp DESC"
2331
+ )
2332
+
2333
+ t0 = time.monotonic()
2334
+ try:
2335
+ async with self.driver.session(database=self.database) as session:
2336
+ result = await session.run(
2337
+ cypher,
2338
+ person_id=person_id,
2339
+ cutoff=cutoff,
2340
+ signal_type=signal_type,
2341
+ )
2342
+ signals = []
2343
+ async for record in result:
2344
+ ts = record["timestamp"]
2345
+ if hasattr(ts, "to_native"):
2346
+ ts = ts.to_native()
2347
+ signals.append(Signal(
2348
+ signal_type=record["signal_type"],
2349
+ raw_value=float(record["raw_value"]),
2350
+ normalized_value=float(record["normalized_value"]),
2351
+ timestamp=ts,
2352
+ person_id=person_id,
2353
+ source=record["source"] or "message",
2354
+ ))
2355
+ logger.debug(
2356
+ "get_recent_signals person=%s count=%d %.1fms",
2357
+ person_id, len(signals), (time.monotonic() - t0) * 1000,
2358
+ )
2359
+ return signals
2360
+ except Exception as exc:
2361
+ logger.error("get_recent_signals failed for person %s: %s", person_id, exc)
2362
+ return []
2363
+
2364
+ async def get_all_people(self) -> List[Dict[str, Any]]:
2365
+ """Return all Person nodes with their current scores."""
2366
+ t0 = time.monotonic()
2367
+ try:
2368
+ async with self.driver.session(database=self.database) as session:
2369
+ result = await session.run(
2370
+ "MATCH (p:Person) "
2371
+ "RETURN p.id AS id, p.name AS name, "
2372
+ " coalesce(p.score, 50.0) AS score, "
2373
+ " coalesce(p.tier, 'regular') AS tier, "
2374
+ " p.lastInteraction AS last_interaction"
2375
+ )
2376
+ people = [dict(r) async for r in result]
2377
+ logger.debug("get_all_people count=%d %.1fms", len(people), (time.monotonic() - t0) * 1000)
2378
+ return people
2379
+ except Exception as exc:
2380
+ logger.error("get_all_people failed: %s", exc)
2381
+ return []
2382
+
2383
+ async def record_score_change(
2384
+ self,
2385
+ person_id: str,
2386
+ new_score: float,
2387
+ new_tier: str,
2388
+ old_score: float,
2389
+ reason: str,
2390
+ store = None, # Optional SQLiteContactStore for reverse sync
2391
+ ) -> None:
2392
+ """Persist a relationship score change with audit trail."""
2393
+ from apsimo.intelligence.graph.queries import RECORD_SCORE_CHANGE
2394
+
2395
+ t0 = time.monotonic()
2396
+ try:
2397
+ async with self.driver.session(database=self.database) as session:
2398
+ await session.run(
2399
+ RECORD_SCORE_CHANGE,
2400
+ person_id=person_id,
2401
+ new_score=new_score,
2402
+ new_tier=new_tier,
2403
+ delta=new_score - old_score,
2404
+ reason=reason,
2405
+ )
2406
+ logger.debug(
2407
+ "record_score_change person=%s %.1f→%.1f (%s) %.1fms",
2408
+ person_id, old_score, new_score, new_tier, (time.monotonic() - t0) * 1000,
2409
+ )
2410
+ # Sync to SQLite if linked contact exists
2411
+ if store is not None:
2412
+ try:
2413
+ contact = await store.find_by_person_node_id(person_id)
2414
+ if contact:
2415
+ # scorer works in 0-100; the contact field is 0-1
2416
+ _norm = new_score / 100.0 if new_score > 1.0 else new_score
2417
+ await store.update_relationship_score(contact.contact_id, _norm)
2418
+ except Exception as exc:
2419
+ logger.debug("Score sync to SQLite failed for %s: %s", person_id, exc)
2420
+ except Exception as exc:
2421
+ logger.error("record_score_change failed for person %s: %s", person_id, exc)
2422
+
2423
+ async def get_person(self, person_id: str) -> Optional[Dict[str, Any]]:
2424
+ """Fetch a Person node with all properties."""
2425
+ t0 = time.monotonic()
2426
+ try:
2427
+ async with self.driver.session(database=self.database) as session:
2428
+ result = await session.run(
2429
+ "MATCH (p:Person {id: $person_id}) RETURN p {.*} AS person",
2430
+ person_id=person_id,
2431
+ )
2432
+ record = await result.single()
2433
+ logger.debug("get_person %s %.1fms", person_id, (time.monotonic() - t0) * 1000)
2434
+ return dict(record["person"]) if record else None
2435
+ except Exception as exc:
2436
+ logger.error("get_person failed for %s: %s", person_id, exc)
2437
+ return None
2438
+
2439
+ # Property names permitted on Person nodes. Values are always passed as
2440
+ # parameters; this allowlist guards the one remaining interpolation point
2441
+ # (the property name itself) against accidental misuse from a caller that
2442
+ # forwards an attacker-controlled dict.
2443
+ _PERSON_PROPS_ALLOWED = frozenset({
2444
+ "name", "tier", "score", "lastInteraction", "created_at",
2445
+ "baseline_msg_count",
2446
+ "baseline_length_mean",
2447
+ "baseline_length_m2",
2448
+ "baseline_length_std",
2449
+ "baseline_hour_histogram",
2450
+ "baseline_updated_at",
2451
+ })
2452
+
2453
+ async def update_person(self, person_id: str, **props: Any) -> None:
2454
+ """Update arbitrary properties on a Person node.
2455
+
2456
+ Only called from trusted internal code (BaselineStore).
2457
+ All values are passed as Neo4j parameters; property names are
2458
+ validated against ``_PERSON_PROPS_ALLOWED``.
2459
+ """
2460
+ if not props:
2461
+ return
2462
+ unknown = set(props) - self._PERSON_PROPS_ALLOWED
2463
+ if unknown:
2464
+ raise ValueError(
2465
+ f"update_person rejected unknown properties: {sorted(unknown)}"
2466
+ )
2467
+ set_clauses = ", ".join(f"p.{k} = ${k}" for k in props)
2468
+ cypher = f"MATCH (p:Person {{id: $person_id}}) SET {set_clauses}"
2469
+ t0 = time.monotonic()
2470
+ try:
2471
+ async with self.driver.session(database=self.database) as session:
2472
+ await session.run(cypher, person_id=person_id, **props)
2473
+ logger.debug("update_person %s props=%s %.1fms", person_id, list(props), (time.monotonic() - t0) * 1000)
2474
+ except Exception as exc:
2475
+ logger.error("update_person failed for %s: %s", person_id, exc)
2476
+
2477
+ # ------------------------------------------------------------------
2478
+
2479
+ @staticmethod
2480
+ def _utcnow() -> "datetime":
2481
+ """Return timezone-aware UTC now (isolated for testability)."""
2482
+ from datetime import datetime as _dt, timezone as _tz
2483
+ return _dt.now(_tz.utc)