apsimo 1.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (614) hide show
  1. apsimo/__init__.py +38 -0
  2. apsimo/__main__.py +6 -0
  3. apsimo/agent/__init__.py +6 -0
  4. apsimo/agent/client.py +276 -0
  5. apsimo/agent/models.py +46 -0
  6. apsimo/agents/__init__.py +20 -0
  7. apsimo/agents/models.py +264 -0
  8. apsimo/agents/store.py +861 -0
  9. apsimo/agents/websocket.py +522 -0
  10. apsimo/api/__init__.py +1 -0
  11. apsimo/api/auth_telemetry.py +287 -0
  12. apsimo/api/authority.py +1203 -0
  13. apsimo/api/contact_grants.py +347 -0
  14. apsimo/api/middleware.py +483 -0
  15. apsimo/api/routers/__init__.py +1 -0
  16. apsimo/api/routers/commitment_work.py +265 -0
  17. apsimo/api/routers/context_gate.py +123 -0
  18. apsimo/api/routers/executions.py +140 -0
  19. apsimo/api/routers/followup_plans.py +147 -0
  20. apsimo/api/routers/governed_actions.py +162 -0
  21. apsimo/api/routers/host.py +14473 -0
  22. apsimo/api/routers/initiative_work.py +115 -0
  23. apsimo/api/routers/mining.py +104 -0
  24. apsimo/api/routers/observations.py +110 -0
  25. apsimo/api/routers/social_state.py +225 -0
  26. apsimo/api/routers/task_queue.py +2715 -0
  27. apsimo/api/routers/temporal_followups.py +251 -0
  28. apsimo/api/routers/transport.py +110 -0
  29. apsimo/api/routers/transport_ingress_api.py +240 -0
  30. apsimo/api/schemas/__init__.py +1 -0
  31. apsimo/api/schemas/host.py +1949 -0
  32. apsimo/autonomy/cli.py +110 -0
  33. apsimo/autonomy/condition_worker.py +437 -0
  34. apsimo/autonomy/config.py +424 -0
  35. apsimo/autonomy/loop.py +4316 -0
  36. apsimo/autonomy/registry.py +339 -0
  37. apsimo/autonomy/scheduler.py +1822 -0
  38. apsimo/autonomy/synthesis.py +449 -0
  39. apsimo/backup.py +962 -0
  40. apsimo/beliefs/__init__.py +23 -0
  41. apsimo/beliefs/contradictions.py +109 -0
  42. apsimo/beliefs/decay.py +61 -0
  43. apsimo/beliefs/engine.py +479 -0
  44. apsimo/beliefs/models.py +67 -0
  45. apsimo/beliefs/promotion.py +41 -0
  46. apsimo/beliefs/resolve.py +58 -0
  47. apsimo/beliefs/source_claims.py +690 -0
  48. apsimo/beliefs/source_projection.py +883 -0
  49. apsimo/beliefs/source_time.py +208 -0
  50. apsimo/beliefs/store.py +133 -0
  51. apsimo/briefings/aggregators.py +824 -0
  52. apsimo/briefings/composer.py +420 -0
  53. apsimo/briefings/config.py +55 -0
  54. apsimo/briefings/delivery.py +439 -0
  55. apsimo/briefings/engagement.py +97 -0
  56. apsimo/briefings/engine.py +274 -0
  57. apsimo/briefings/enhancer.py +99 -0
  58. apsimo/briefings/models.py +183 -0
  59. apsimo/briefings/scheduler.py +382 -0
  60. apsimo/briefings/store.py +435 -0
  61. apsimo/chain/__init__.py +48 -0
  62. apsimo/chain/block.py +100 -0
  63. apsimo/chain/cli.py +704 -0
  64. apsimo/chain/genesis.py +443 -0
  65. apsimo/chain/identity.py +416 -0
  66. apsimo/chain/keys.py +1025 -0
  67. apsimo/chain/local_keys.py +187 -0
  68. apsimo/chain/manager.py +290 -0
  69. apsimo/chain/node.py +163 -0
  70. apsimo/chain/plugin_transactions.py +371 -0
  71. apsimo/chain/protocol.py +220 -0
  72. apsimo/chain/state_machine.py +676 -0
  73. apsimo/chain/storage.py +503 -0
  74. apsimo/chain/transactions.py +250 -0
  75. apsimo/chain/validation.py +397 -0
  76. apsimo/channels/__init__.py +1 -0
  77. apsimo/channels/manifest.py +31 -0
  78. apsimo/channels/migrations/001_channels_schema.sql +12 -0
  79. apsimo/channels/phone_gateways.py +42 -0
  80. apsimo/channels/presence.py +188 -0
  81. apsimo/channels/router.py +235 -0
  82. apsimo/channels/store.py +231 -0
  83. apsimo/cli.py +2688 -0
  84. apsimo/cognition/__init__.py +11 -0
  85. apsimo/cognition/charter.py +398 -0
  86. apsimo/cognition/drive_governance.py +3530 -0
  87. apsimo/cognition/evidence_pipeline.py +1627 -0
  88. apsimo/cognition/external_events.py +932 -0
  89. apsimo/cognition/goal_spine.py +3488 -0
  90. apsimo/cognition/introspection.py +214 -0
  91. apsimo/cognition/prompt.py +150 -0
  92. apsimo/cognition/runtime.py +108 -0
  93. apsimo/cognition/trigger.py +154 -0
  94. apsimo/commitments/__init__.py +18 -0
  95. apsimo/commitments/local_work.py +355 -0
  96. apsimo/commitments/store.py +1052 -0
  97. apsimo/commitments/work.py +91 -0
  98. apsimo/compat.py +53 -0
  99. apsimo/compression/__init__.py +467 -0
  100. apsimo/connectors/__init__.py +21 -0
  101. apsimo/connectors/base.py +152 -0
  102. apsimo/connectors/caldav_calendar.py +125 -0
  103. apsimo/connectors/fs_documents.py +85 -0
  104. apsimo/connectors/imap_email.py +138 -0
  105. apsimo/connectors/manager.py +218 -0
  106. apsimo/connectors/webhook_pull.py +88 -0
  107. apsimo/contacts/__init__.py +33 -0
  108. apsimo/contacts/comms.py +357 -0
  109. apsimo/contacts/config.py +79 -0
  110. apsimo/contacts/exporters/__init__.py +1 -0
  111. apsimo/contacts/exporters/vcard.py +71 -0
  112. apsimo/contacts/identity_links.py +251 -0
  113. apsimo/contacts/importer.py +280 -0
  114. apsimo/contacts/importers/__init__.py +1 -0
  115. apsimo/contacts/importers/batch.py +43 -0
  116. apsimo/contacts/importers/macos_contacts.py +101 -0
  117. apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
  118. apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
  119. apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
  120. apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
  121. apsimo/contacts/migrations/005_identity_links.sql +27 -0
  122. apsimo/contacts/models.py +308 -0
  123. apsimo/contacts/scoring.py +16 -0
  124. apsimo/contacts/store.py +1623 -0
  125. apsimo/contacts/transport_ingress.py +252 -0
  126. apsimo/contacts/world_bridge.py +314 -0
  127. apsimo/contextgate/__init__.py +69 -0
  128. apsimo/contextgate/chunker.py +169 -0
  129. apsimo/contextgate/estimate.py +54 -0
  130. apsimo/contextgate/gate.py +313 -0
  131. apsimo/contextgate/retrieve.py +115 -0
  132. apsimo/delivery/__init__.py +16 -0
  133. apsimo/delivery/bridge.py +1260 -0
  134. apsimo/delivery/channels.py +526 -0
  135. apsimo/delivery/classification.py +50 -0
  136. apsimo/delivery/rate_limiter.py +268 -0
  137. apsimo/delivery/reachout_policy.py +206 -0
  138. apsimo/directed/__init__.py +22 -0
  139. apsimo/directed/audit.py +167 -0
  140. apsimo/directed/intake.py +95 -0
  141. apsimo/directed/models.py +191 -0
  142. apsimo/directed/service.py +509 -0
  143. apsimo/directives/__init__.py +25 -0
  144. apsimo/directives/evidence.py +87 -0
  145. apsimo/directives/extractor.py +188 -0
  146. apsimo/directives/guard.py +364 -0
  147. apsimo/directives/models.py +206 -0
  148. apsimo/directives/service.py +372 -0
  149. apsimo/directives/store.py +167 -0
  150. apsimo/doctor.py +2173 -0
  151. apsimo/environment.py +43 -0
  152. apsimo/events/__init__.py +33 -0
  153. apsimo/events/broadcaster.py +98 -0
  154. apsimo/events/bus.py +217 -0
  155. apsimo/events/journal.py +863 -0
  156. apsimo/events/stream.py +131 -0
  157. apsimo/events/types.py +150 -0
  158. apsimo/execution_results.py +357 -0
  159. apsimo/feedback/__init__.py +5 -0
  160. apsimo/feedback/store.py +76 -0
  161. apsimo/feeds/__init__.py +19 -0
  162. apsimo/feeds/cli.py +84 -0
  163. apsimo/feeds/engine.py +437 -0
  164. apsimo/feeds/example-feed.yaml +77 -0
  165. apsimo/feeds/hermes_cron.py +126 -0
  166. apsimo/feeds/manager.py +235 -0
  167. apsimo/feeds/spec.py +250 -0
  168. apsimo/feeds/template.py +202 -0
  169. apsimo/gate/__init__.py +18 -0
  170. apsimo/gate/audit.py +61 -0
  171. apsimo/gate/communication_policy.py +166 -0
  172. apsimo/gate/config.py +72 -0
  173. apsimo/gate/context_provenance.py +170 -0
  174. apsimo/gate/env_risk.py +226 -0
  175. apsimo/gate/guard_audit.py +353 -0
  176. apsimo/gate/layers/__init__.py +1 -0
  177. apsimo/gate/layers/base.py +15 -0
  178. apsimo/gate/layers/l1_recipient.py +66 -0
  179. apsimo/gate/layers/l2_pii.py +134 -0
  180. apsimo/gate/layers/l3_cross_context.py +50 -0
  181. apsimo/gate/layers/l4_trust_tier.py +78 -0
  182. apsimo/gate/layers/l5_injection.py +199 -0
  183. apsimo/gate/layers/l6_review.py +86 -0
  184. apsimo/gate/layers/l7_delay.py +100 -0
  185. apsimo/gate/layers/tom2_epistemic.py +185 -0
  186. apsimo/gate/models.py +64 -0
  187. apsimo/gate/pending_dispatch.py +5 -0
  188. apsimo/gate/pipeline.py +206 -0
  189. apsimo/gate/rejection.py +259 -0
  190. apsimo/gate/response_guard.py +700 -0
  191. apsimo/gate/rulesets/injection_v1.yaml +51 -0
  192. apsimo/gate/surface_policy.py +189 -0
  193. apsimo/gate/taint.py +226 -0
  194. apsimo/genesis.json +9 -0
  195. apsimo/goals/__init__.py +100 -0
  196. apsimo/goals/config.py +38 -0
  197. apsimo/goals/decomposer.py +421 -0
  198. apsimo/goals/engine.py +617 -0
  199. apsimo/goals/inference.py +354 -0
  200. apsimo/goals/models.py +302 -0
  201. apsimo/goals/priority.py +270 -0
  202. apsimo/goals/queue_bridge.py +149 -0
  203. apsimo/goals/replan.py +450 -0
  204. apsimo/goals/schema.sql +89 -0
  205. apsimo/goals/store.py +692 -0
  206. apsimo/governed_actions.py +1708 -0
  207. apsimo/harness_integration/__init__.py +45 -0
  208. apsimo/harness_integration/context.py +41 -0
  209. apsimo/harness_integration/skills.py +231 -0
  210. apsimo/identity/__init__.py +26 -0
  211. apsimo/identity/participants.py +181 -0
  212. apsimo/identity/resolver.py +329 -0
  213. apsimo/identity_bootstrap/__init__.py +5 -0
  214. apsimo/identity_bootstrap/builder.py +208 -0
  215. apsimo/identity_bootstrap/corpus.py +443 -0
  216. apsimo/identity_bootstrap/models.py +54 -0
  217. apsimo/identity_bootstrap/runner.py +353 -0
  218. apsimo/identity_bootstrap/seeders/__init__.py +25 -0
  219. apsimo/identity_bootstrap/seeders/briefings.py +109 -0
  220. apsimo/identity_bootstrap/seeders/chain.py +57 -0
  221. apsimo/identity_bootstrap/seeders/goals.py +128 -0
  222. apsimo/identity_bootstrap/seeders/memory.py +191 -0
  223. apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
  224. apsimo/identity_bootstrap/seeders/relationship.py +152 -0
  225. apsimo/identity_bootstrap/seeders/sessions.py +67 -0
  226. apsimo/identity_bootstrap/seeders/skills.py +92 -0
  227. apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
  228. apsimo/identity_bootstrap/seeders/world_model.py +143 -0
  229. apsimo/identity_bootstrap/self_query.py +92 -0
  230. apsimo/identity_bootstrap/self_reflection.py +155 -0
  231. apsimo/identity_bootstrap/skill.py +37 -0
  232. apsimo/identity_bootstrap/verifier.py +436 -0
  233. apsimo/initiatives/__init__.py +20 -0
  234. apsimo/initiatives/action_registry.py +454 -0
  235. apsimo/initiatives/approval_authority.py +2105 -0
  236. apsimo/initiatives/approval_policy.py +123 -0
  237. apsimo/initiatives/assignment.py +263 -0
  238. apsimo/initiatives/backup_evidence.py +100 -0
  239. apsimo/initiatives/context_freshness.py +103 -0
  240. apsimo/initiatives/models.py +318 -0
  241. apsimo/initiatives/native_work.py +270 -0
  242. apsimo/initiatives/standing_approvals.py +232 -0
  243. apsimo/initiatives/store.py +1081 -0
  244. apsimo/initiatives/temporal_followup.py +410 -0
  245. apsimo/intelligence/__init__.py +1 -0
  246. apsimo/intelligence/cognition/__init__.py +24 -0
  247. apsimo/intelligence/cognition/gap_detector.py +148 -0
  248. apsimo/intelligence/cognition/metalearner.py +547 -0
  249. apsimo/intelligence/cognition/metrics_collector.py +217 -0
  250. apsimo/intelligence/cognition/performance_index.py +299 -0
  251. apsimo/intelligence/cognition/registry.py +192 -0
  252. apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
  253. apsimo/intelligence/cognition/types.py +16 -0
  254. apsimo/intelligence/components/__init__.py +66 -0
  255. apsimo/intelligence/components/anomaly_detector.py +413 -0
  256. apsimo/intelligence/components/initiative_engine.py +2643 -0
  257. apsimo/intelligence/components/preference_learner.py +521 -0
  258. apsimo/intelligence/components/research_orchestrator.py +358 -0
  259. apsimo/intelligence/components/self_directed_thinker.py +221 -0
  260. apsimo/intelligence/components/self_reflector.py +252 -0
  261. apsimo/intelligence/components/session_continuity.py +154 -0
  262. apsimo/intelligence/components/task_planner.py +320 -0
  263. apsimo/intelligence/components/tool_learner.py +217 -0
  264. apsimo/intelligence/graph/__init__.py +79 -0
  265. apsimo/intelligence/graph/client.py +2483 -0
  266. apsimo/intelligence/graph/consolidator.py +405 -0
  267. apsimo/intelligence/graph/distiller.py +312 -0
  268. apsimo/intelligence/graph/migrations.py +129 -0
  269. apsimo/intelligence/graph/queries.py +248 -0
  270. apsimo/intelligence/graph/recall.py +281 -0
  271. apsimo/intelligence/graph/reconciler.py +144 -0
  272. apsimo/intelligence/graph/schema.py +337 -0
  273. apsimo/intelligence/graph/selection.py +252 -0
  274. apsimo/intelligence/learning/__init__.py +17 -0
  275. apsimo/intelligence/learning/continuous_learner.py +245 -0
  276. apsimo/intelligence/learning/feedback_store.py +321 -0
  277. apsimo/intelligence/mind_model/__init__.py +1 -0
  278. apsimo/intelligence/mind_model/graph_baseline.py +136 -0
  279. apsimo/intelligence/mind_model/signal_collector.py +361 -0
  280. apsimo/intelligence/relationships/__init__.py +11 -0
  281. apsimo/intelligence/relationships/profiler.py +389 -0
  282. apsimo/intelligence/relationships/scorer.py +560 -0
  283. apsimo/intelligence/relationships/signal_floor.py +66 -0
  284. apsimo/intelligence/relationships/trust_tiers.py +300 -0
  285. apsimo/intelligence/synthesis/__init__.py +40 -0
  286. apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
  287. apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
  288. apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
  289. apsimo/intelligence/synthesis/insight_store.py +79 -0
  290. apsimo/intelligence/synthesis/insight_validator.py +183 -0
  291. apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
  292. apsimo/intelligence/turn_middleware/__init__.py +15 -0
  293. apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
  294. apsimo/mcp/__init__.py +41 -0
  295. apsimo/mcp/__main__.py +6 -0
  296. apsimo/mcp/config.py +287 -0
  297. apsimo/mcp/server.py +501 -0
  298. apsimo/migrations.py +187 -0
  299. apsimo/mining/__init__.py +27 -0
  300. apsimo/mining/corpus.py +239 -0
  301. apsimo/mining/escalations.py +289 -0
  302. apsimo/mining/models.py +169 -0
  303. apsimo/mining/store.py +210 -0
  304. apsimo/models/__init__.py +30 -0
  305. apsimo/models/memory.py +80 -0
  306. apsimo/models/mesh.py +72 -0
  307. apsimo/models/person.py +104 -0
  308. apsimo/models/signal.py +108 -0
  309. apsimo/observations/__init__.py +15 -0
  310. apsimo/observations/store.py +277 -0
  311. apsimo/patterns/__init__.py +6 -0
  312. apsimo/patterns/extract.py +187 -0
  313. apsimo/patterns/store.py +227 -0
  314. apsimo/persona/__init__.py +1 -0
  315. apsimo/persona/engine.py +611 -0
  316. apsimo/persona/manifest.py +140 -0
  317. apsimo/projects/__init__.py +28 -0
  318. apsimo/projects/engine.py +1681 -0
  319. apsimo/projects/event_outbox.py +188 -0
  320. apsimo/projects/models.py +216 -0
  321. apsimo/projects/planner.py +181 -0
  322. apsimo/projects/store.py +1446 -0
  323. apsimo/proposals/__init__.py +12 -0
  324. apsimo/proposals/engine.py +114 -0
  325. apsimo/proposals/models.py +207 -0
  326. apsimo/qualification/__init__.py +1 -0
  327. apsimo/qualification/cases.py +75 -0
  328. apsimo/qualification/cli.py +51 -0
  329. apsimo/qualification/memory_cases.py +209 -0
  330. apsimo/qualification/records.py +92 -0
  331. apsimo/qualification/report.py +87 -0
  332. apsimo/qualification/runner.py +311 -0
  333. apsimo/qualification/structured_cases.py +131 -0
  334. apsimo/reasoning/__init__.py +13 -0
  335. apsimo/reasoning/executor.py +506 -0
  336. apsimo/reasoning/loop.py +373 -0
  337. apsimo/reasoning/native_tools/__init__.py +16 -0
  338. apsimo/reasoning/native_tools/calculate.py +141 -0
  339. apsimo/reasoning/native_tools/file_ops.py +150 -0
  340. apsimo/reasoning/native_tools/web_search.py +49 -0
  341. apsimo/reasoning/tool_policy.py +182 -0
  342. apsimo/redact/__init__.py +176 -0
  343. apsimo/repos/__init__.py +5 -0
  344. apsimo/repos/mirrors.py +204 -0
  345. apsimo/research/__init__.py +41 -0
  346. apsimo/research/artifact.py +482 -0
  347. apsimo/research/gatherer.py +387 -0
  348. apsimo/research/pipeline.py +513 -0
  349. apsimo/research/search/__init__.py +7 -0
  350. apsimo/research/search/base.py +41 -0
  351. apsimo/research/search/brave.py +59 -0
  352. apsimo/research/search/cache.py +51 -0
  353. apsimo/research/search/duckduckgo.py +103 -0
  354. apsimo/research/search/orchestrator.py +119 -0
  355. apsimo/research/search/serpapi.py +59 -0
  356. apsimo/research/search/tavily.py +59 -0
  357. apsimo/research/synthesizer.py +309 -0
  358. apsimo/router/__init__.py +30 -0
  359. apsimo/router/complexity_scorer.py +148 -0
  360. apsimo/router/endpoints.py +153 -0
  361. apsimo/router/fallback.py +58 -0
  362. apsimo/router/functions.py +243 -0
  363. apsimo/router/native_policy.py +52 -0
  364. apsimo/router/router.py +762 -0
  365. apsimo/router/self_learning.py +174 -0
  366. apsimo/router/tiers.py +677 -0
  367. apsimo/sandbox/__init__.py +21 -0
  368. apsimo/sandbox/backend.py +195 -0
  369. apsimo/sandbox/manager.py +173 -0
  370. apsimo/scope_bounds.py +7 -0
  371. apsimo/secrets/__init__.py +6 -0
  372. apsimo/secrets/backends/__init__.py +8 -0
  373. apsimo/secrets/backends/base.py +42 -0
  374. apsimo/secrets/backends/env.py +110 -0
  375. apsimo/secrets/backends/keyring.py +72 -0
  376. apsimo/secrets/backends/onepassword.py +232 -0
  377. apsimo/secrets/cli.py +191 -0
  378. apsimo/secrets/manager.py +160 -0
  379. apsimo/secrets/migration.py +101 -0
  380. apsimo/secrets/types.py +98 -0
  381. apsimo/seed.py +41 -0
  382. apsimo/self_model/__init__.py +37 -0
  383. apsimo/self_model/appraisals.py +673 -0
  384. apsimo/self_model/benchmark.py +1314 -0
  385. apsimo/self_model/brief.py +40 -0
  386. apsimo/self_model/event_concerns.py +1128 -0
  387. apsimo/self_model/execution_forecasts.py +353 -0
  388. apsimo/self_model/expectations.py +1595 -0
  389. apsimo/self_model/experiments.py +1150 -0
  390. apsimo/self_model/journal.py +148 -0
  391. apsimo/self_model/judgments.py +705 -0
  392. apsimo/self_model/native_outcomes.py +55 -0
  393. apsimo/self_model/params.py +220 -0
  394. apsimo/self_model/perspective.py +246 -0
  395. apsimo/self_model/reconcile.py +183 -0
  396. apsimo/self_model/reply_forecasts.py +381 -0
  397. apsimo/self_model/runtime_forecasts.py +296 -0
  398. apsimo/self_model/runtime_models.py +67 -0
  399. apsimo/self_model/settlement.py +207 -0
  400. apsimo/self_model/situation.py +1731 -0
  401. apsimo/self_model/store.py +883 -0
  402. apsimo/self_model/supervised.py +137 -0
  403. apsimo/self_model/thinker.py +99 -0
  404. apsimo/self_model/trust.py +388 -0
  405. apsimo/self_model/workspace.py +2388 -0
  406. apsimo/server.py +4197 -0
  407. apsimo/services/__init__.py +1 -0
  408. apsimo/services/agent_bridge.py +474 -0
  409. apsimo/services/initiative_executor.py +914 -0
  410. apsimo/services/instance.py +297 -0
  411. apsimo/sessions/__init__.py +22 -0
  412. apsimo/sessions/config.py +13 -0
  413. apsimo/sessions/context_loader.py +88 -0
  414. apsimo/sessions/federation_session.py +75 -0
  415. apsimo/sessions/isolated_session.py +98 -0
  416. apsimo/sessions/reports.py +84 -0
  417. apsimo/sessions/store.py +148 -0
  418. apsimo/setup.py +2818 -0
  419. apsimo/setup_hermes.py +879 -0
  420. apsimo/setup_local_work.py +218 -0
  421. apsimo/setup_native_goals.py +134 -0
  422. apsimo/setup_native_reviews.py +115 -0
  423. apsimo/skills/__init__.py +10 -0
  424. apsimo/skills/base.py +108 -0
  425. apsimo/skills/budget.py +28 -0
  426. apsimo/skills/executor.py +493 -0
  427. apsimo/skills/executors/__init__.py +1 -0
  428. apsimo/skills/executors/behavioral_correction.py +75 -0
  429. apsimo/skills/executors/capability_gap.py +38 -0
  430. apsimo/skills/executors/data_quality.py +163 -0
  431. apsimo/skills/executors/knowledge_acquisition.py +41 -0
  432. apsimo/skills/executors/operational_hygiene.py +185 -0
  433. apsimo/skills/executors/subsystem_health.py +169 -0
  434. apsimo/skills/hermes_export.py +431 -0
  435. apsimo/skills/index.py +123 -0
  436. apsimo/skills/learning/__init__.py +21 -0
  437. apsimo/skills/learning/novelty_detector.py +206 -0
  438. apsimo/skills/learning/pattern_extractor.py +199 -0
  439. apsimo/skills/learning/triggers.py +159 -0
  440. apsimo/skills/loader.py +246 -0
  441. apsimo/skills/migrations/002_progressive_loading.sql +6 -0
  442. apsimo/skills/migrations/backfill_triggers.py +20 -0
  443. apsimo/skills/models.py +202 -0
  444. apsimo/skills/packager.py +128 -0
  445. apsimo/skills/protocols.py +70 -0
  446. apsimo/skills/registry.py +191 -0
  447. apsimo/skills/runtime.py +58 -0
  448. apsimo/skills/sandbox_runner.py +229 -0
  449. apsimo/skills/scheduler.py +129 -0
  450. apsimo/skills/schema.py +79 -0
  451. apsimo/skills/security/__init__.py +12 -0
  452. apsimo/skills/security/guards.py +53 -0
  453. apsimo/skills/security/scanner.py +223 -0
  454. apsimo/skills_memory/__init__.py +26 -0
  455. apsimo/skills_memory/distill.py +159 -0
  456. apsimo/skills_memory/models.py +85 -0
  457. apsimo/skills_memory/retrieve.py +62 -0
  458. apsimo/skills_memory/store.py +172 -0
  459. apsimo/surprise/__init__.py +6 -0
  460. apsimo/surprise/accumulation.py +57 -0
  461. apsimo/surprise/scorer.py +102 -0
  462. apsimo/surprise/store.py +203 -0
  463. apsimo/task_queue/__init__.py +69 -0
  464. apsimo/task_queue/action_receipts.py +148 -0
  465. apsimo/task_queue/approval_relay_canary.py +108 -0
  466. apsimo/task_queue/config.py +85 -0
  467. apsimo/task_queue/contract.py +361 -0
  468. apsimo/task_queue/events.py +130 -0
  469. apsimo/task_queue/governor.py +1031 -0
  470. apsimo/task_queue/handlers/__init__.py +16 -0
  471. apsimo/task_queue/handlers/base.py +37 -0
  472. apsimo/task_queue/handlers/inference.py +640 -0
  473. apsimo/task_queue/handlers/monitoring.py +116 -0
  474. apsimo/task_queue/handlers/registry.py +75 -0
  475. apsimo/task_queue/handlers/subtask_handler.py +173 -0
  476. apsimo/task_queue/handlers/system_maintenance.py +147 -0
  477. apsimo/task_queue/mesh_integration.py +111 -0
  478. apsimo/task_queue/models.py +317 -0
  479. apsimo/task_queue/queue_manager.py +8286 -0
  480. apsimo/task_queue/routing.py +287 -0
  481. apsimo/task_queue/scheduler.py +252 -0
  482. apsimo/task_queue/schema.sql +197 -0
  483. apsimo/task_queue/work_control.py +342 -0
  484. apsimo/task_queue/worker.py +993 -0
  485. apsimo/telemetry.py +145 -0
  486. apsimo/tom/__init__.py +6 -0
  487. apsimo/tom/affect.py +387 -0
  488. apsimo/tom/approvals.py +171 -0
  489. apsimo/tom/arcs.py +896 -0
  490. apsimo/tom/asymmetry.py +131 -0
  491. apsimo/tom/eligibility.py +248 -0
  492. apsimo/tom/engagement.py +214 -0
  493. apsimo/tom/exposure.py +214 -0
  494. apsimo/tom/extractor.py +306 -0
  495. apsimo/tom/fact_adapters.py +144 -0
  496. apsimo/tom/facts.py +326 -0
  497. apsimo/tom/integration.py +592 -0
  498. apsimo/tom/leveled.py +118 -0
  499. apsimo/tom/levels.py +247 -0
  500. apsimo/tom/recipient_audit.py +995 -0
  501. apsimo/tom/recipient_simulator.py +593 -0
  502. apsimo/tom/source_lineage.py +93 -0
  503. apsimo/tom/tom2.py +277 -0
  504. apsimo/tom/visibility.py +559 -0
  505. apsimo/tom/visibility_store.py +414 -0
  506. apsimo/tools/__init__.py +0 -0
  507. apsimo/tools/definitions.py +740 -0
  508. apsimo/tools/handlers.py +943 -0
  509. apsimo/toolsmith/__init__.py +26 -0
  510. apsimo/toolsmith/authority.py +166 -0
  511. apsimo/toolsmith/engine.py +559 -0
  512. apsimo/toolsmith/integrity.py +100 -0
  513. apsimo/toolsmith/miner.py +145 -0
  514. apsimo/toolsmith/policy.py +110 -0
  515. apsimo/toolsmith/registry.py +635 -0
  516. apsimo/turns/__init__.py +17 -0
  517. apsimo/turns/audio.py +134 -0
  518. apsimo/turns/documents.py +235 -0
  519. apsimo/turns/executions.py +486 -0
  520. apsimo/turns/hermes_history.py +245 -0
  521. apsimo/turns/hermes_kanban.py +268 -0
  522. apsimo/turns/hermes_work.py +96 -0
  523. apsimo/turns/idempotency.py +752 -0
  524. apsimo/turns/local_work.py +115 -0
  525. apsimo/turns/media.py +581 -0
  526. apsimo/turns/reported_workers.py +196 -0
  527. apsimo/turns/source_annotations.py +283 -0
  528. apsimo/turns/source_attribution.py +154 -0
  529. apsimo/turns/source_read.py +351 -0
  530. apsimo/turns/source_vectors.py +263 -0
  531. apsimo/turns/video.py +210 -0
  532. apsimo/util/autonomy_preset.py +220 -0
  533. apsimo/util/instance.py +92 -0
  534. apsimo/util/model_output.py +25 -0
  535. apsimo/util/quiet_hours.py +27 -0
  536. apsimo/util/session_safety.py +37 -0
  537. apsimo/util/temporal.py +343 -0
  538. apsimo/vector/__init__.py +75 -0
  539. apsimo/vector/backfill.py +171 -0
  540. apsimo/vector/caption.py +114 -0
  541. apsimo/vector/collections.py +51 -0
  542. apsimo/vector/config.py +102 -0
  543. apsimo/vector/embedder.py +670 -0
  544. apsimo/vector/image_preprocess.py +406 -0
  545. apsimo/vector/image_store.py +296 -0
  546. apsimo/vector/indexes.py +162 -0
  547. apsimo/vector/migrate.py +334 -0
  548. apsimo/vector/multimodal_provider.py +417 -0
  549. apsimo/vector/multimodal_types.py +87 -0
  550. apsimo/vector/openai_provider.py +119 -0
  551. apsimo/vector/query.py +49 -0
  552. apsimo/vector/reranker.py +565 -0
  553. apsimo/vector/safety_image.py +159 -0
  554. apsimo/vector/scanner.py +197 -0
  555. apsimo/vector/setup.py +289 -0
  556. apsimo/vector/store.py +533 -0
  557. apsimo/vector/tiers.py +263 -0
  558. apsimo/work_orders.py +925 -0
  559. apsimo/workers/__init__.py +21 -0
  560. apsimo/workers/agent_bridge.py +640 -0
  561. apsimo/workers/colony_worker.py +382 -0
  562. apsimo/workers/queue_worker.py +441 -0
  563. apsimo/workers/skills_sync.py +152 -0
  564. apsimo/world_model/__init__.py +71 -0
  565. apsimo/world_model/causal_maintenance.py +131 -0
  566. apsimo/world_model/causal_policy.py +43 -0
  567. apsimo/world_model/causal_query.py +125 -0
  568. apsimo/world_model/confidence.py +54 -0
  569. apsimo/world_model/config.py +64 -0
  570. apsimo/world_model/constants.py +97 -0
  571. apsimo/world_model/entities.py +145 -0
  572. apsimo/world_model/expectation_resolvers.py +177 -0
  573. apsimo/world_model/extraction/__init__.py +7 -0
  574. apsimo/world_model/extraction/base.py +62 -0
  575. apsimo/world_model/extraction/conversation_extractor.py +262 -0
  576. apsimo/world_model/extraction/detector.py +74 -0
  577. apsimo/world_model/extraction/document_extractor.py +78 -0
  578. apsimo/world_model/extraction/formats/__init__.py +24 -0
  579. apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
  580. apsimo/world_model/extraction/formats/html_fmt.py +72 -0
  581. apsimo/world_model/extraction/formats/json_fmt.py +68 -0
  582. apsimo/world_model/extraction/formats/pdf.py +43 -0
  583. apsimo/world_model/extraction/formats/text.py +27 -0
  584. apsimo/world_model/extraction/llm_extractor.py +164 -0
  585. apsimo/world_model/extraction/pipeline.py +73 -0
  586. apsimo/world_model/integrations/__init__.py +5 -0
  587. apsimo/world_model/integrations/mind_model_bridge.py +115 -0
  588. apsimo/world_model/integrations/social_intel_bridge.py +120 -0
  589. apsimo/world_model/jobs/__init__.py +4 -0
  590. apsimo/world_model/jobs/extraction_job.py +168 -0
  591. apsimo/world_model/llm_extract.py +572 -0
  592. apsimo/world_model/neo4j/__init__.py +5 -0
  593. apsimo/world_model/neo4j/backend.py +654 -0
  594. apsimo/world_model/observations.py +155 -0
  595. apsimo/world_model/populator.py +307 -0
  596. apsimo/world_model/postgres/__init__.py +1 -0
  597. apsimo/world_model/postgres/backend.py +683 -0
  598. apsimo/world_model/relationships.py +25 -0
  599. apsimo/world_model/resolution/__init__.py +13 -0
  600. apsimo/world_model/resolution/entity_resolver.py +232 -0
  601. apsimo/world_model/resolution/merge_audit.py +16 -0
  602. apsimo/world_model/resolution/merge_workflow.py +117 -0
  603. apsimo/world_model/source_reports.py +121 -0
  604. apsimo/world_model/sqlite/__init__.py +4 -0
  605. apsimo/world_model/sqlite/backend.py +855 -0
  606. apsimo/world_model/sqlite/schema.sql +132 -0
  607. apsimo/world_model/store.py +545 -0
  608. apsimo-1.3.0.dist-info/METADATA +78 -0
  609. apsimo-1.3.0.dist-info/RECORD +614 -0
  610. apsimo-1.3.0.dist-info/WHEEL +5 -0
  611. apsimo-1.3.0.dist-info/entry_points.txt +11 -0
  612. apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
  613. apsimo-1.3.0.dist-info/top_level.txt +2 -0
  614. colony_sidecar/__init__.py +4 -0
@@ -0,0 +1,1314 @@
1
+ """Selfhood benchmark: falsifiable self-improvement metrics (Mind M0a).
2
+
3
+ Derives a weekly scorecard entirely from journals and stores that already
4
+ exist. Nothing is self-reported by the LLM; every metric is computed from
5
+ recorded outcomes, and a metric whose source is unavailable is SKIPPED
6
+ rather than defaulted (the same fail-unknown discipline the doctor uses).
7
+
8
+ Metrics (stable ids):
9
+ commitments.fulfillment fulfilled / (fulfilled + open-overdue) in window
10
+ initiative.acceptance owner responded within 24h of a delivery success
11
+ delivery.success delivery-domain outcome rate (competence events)
12
+ actions.success all-domain outcome rate, per-domain detail
13
+ journal.acted_share acted / (acted+asked+held+blocked) decision mix
14
+ recall.fact_coverage probe: high-confidence shared facts re-queried
15
+ against graph recall, token-coverage graded
16
+ latency.jobs_p50_secs completed queue-job durations (p50; p95 detail)
17
+ latency.* / surface.* host-submitted samples (POST .../samples) rolled
18
+ up automatically: latency.* -> p50 (+p95),
19
+ everything else -> mean
20
+
21
+ Storage: colony-benchmark.db (samples append-only + weekly rollups).
22
+ Weeks are ISO (%G-W%V), windows are Monday 00:00 UTC half-open.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ import asyncio
28
+ from dataclasses import dataclass
29
+ import hashlib
30
+ import json
31
+ import logging
32
+ import math
33
+ import os
34
+ import random
35
+ import re
36
+ import sqlite3
37
+ import threading
38
+ import time
39
+ import uuid
40
+ from datetime import datetime, timedelta, timezone
41
+ from typing import Any, Dict, List, Optional, Tuple
42
+
43
+ logger = logging.getLogger(__name__)
44
+
45
+ _METRIC_RE = re.compile(r"^[a-z0-9_]+(\.[a-z0-9_]+)+$")
46
+ _DEFINITION_VERSION_RE = re.compile(r"^v[1-9][0-9]{0,5}$")
47
+ _P4_MODES = frozenset({"off", "shadow", "live"})
48
+
49
+
50
+ def cognition_p4_mode() -> str:
51
+ """Controlled-learning mode. New deployments are deliberately dark."""
52
+
53
+ value = os.environ.get("COLONY_COGNITION_P4_MODE", "off").strip().lower()
54
+ return value if value in _P4_MODES else "off"
55
+
56
+
57
+ def _canonical(value: Any) -> str:
58
+ return json.dumps(value, sort_keys=True, separators=(",", ":"),
59
+ ensure_ascii=True)
60
+
61
+
62
+ @dataclass(frozen=True)
63
+ class MetricDefinition:
64
+ """Immutable measurement contract for one benchmark metric version."""
65
+
66
+ metric: str
67
+ version: str
68
+ direction: str
69
+ unit: str
70
+ evidence_query: str
71
+ minimum_samples: int
72
+ description: str = ""
73
+
74
+ def normalized(self) -> Dict[str, Any]:
75
+ metric = (self.metric or "").strip().lower()
76
+ version = (self.version or "").strip().lower()
77
+ direction = (self.direction or "").strip().lower()
78
+ unit = (self.unit or "").strip().lower()
79
+ evidence_query = (self.evidence_query or "").strip()
80
+ description = (self.description or "").strip()
81
+ if not _METRIC_RE.fullmatch(metric):
82
+ raise ValueError("metric definition has an invalid metric id")
83
+ if not _DEFINITION_VERSION_RE.fullmatch(version):
84
+ raise ValueError("metric definition version must look like v1")
85
+ if direction not in {"higher", "lower"}:
86
+ raise ValueError("metric direction must be higher or lower")
87
+ if not unit or len(unit) > 64:
88
+ raise ValueError("metric unit is required")
89
+ if not evidence_query or len(evidence_query) > 2000:
90
+ raise ValueError("metric evidence_query is required")
91
+ minimum = int(self.minimum_samples)
92
+ if minimum < 1 or minimum > 1_000_000:
93
+ raise ValueError("metric minimum_samples is out of bounds")
94
+ return {
95
+ "metric": metric,
96
+ "version": version,
97
+ "direction": direction,
98
+ "unit": unit,
99
+ "evidence_query": evidence_query,
100
+ "minimum_samples": minimum,
101
+ "description": description[:1000],
102
+ }
103
+
104
+
105
+ BUILTIN_METRIC_DEFINITIONS = (
106
+ MetricDefinition(
107
+ "commitments.fulfillment", "v2", "higher", "ratio",
108
+ "commitment.due_at in cohort_week AND terminal evidence is recorded",
109
+ 5, "On-time fulfillment for the commitments due in one coherent cohort."),
110
+ MetricDefinition(
111
+ "delivery.success", "v1", "higher", "ratio",
112
+ "competence.domain=delivery AND evidence_status=verified", 5,
113
+ "Receipt-backed transport success."),
114
+ MetricDefinition(
115
+ "actions.success", "v2", "higher", "ratio",
116
+ "competence evidence is available and outcome is non-neutral", 10,
117
+ "Verified, versioned action outcomes."),
118
+ MetricDefinition(
119
+ "journal.acted_share", "v1", "higher", "ratio",
120
+ "action_journal.decision in (acted,asked,held,blocked)", 5,
121
+ "Decision mix; diagnostic rather than a success claim."),
122
+ MetricDefinition(
123
+ "initiative.acceptance", "v2", "higher", "ratio",
124
+ "owner reaction explicitly names the delivered initiative or message", 5,
125
+ "Message-bound owner acceptance; unrelated inbound turns never count."),
126
+ MetricDefinition(
127
+ "responses.correction_rate", "v1", "lower", "ratio",
128
+ "owner correction context_hash names a receipt-backed outbound response", 10,
129
+ "Owner-corrected responses divided by the same outbound cohort."),
130
+ MetricDefinition(
131
+ "recall.fact_coverage", "v2", "higher", "ratio",
132
+ "fact is viewer-allowed and recall uses the fact subject scope", 8,
133
+ "Viewer-scoped recall probe coverage."),
134
+ MetricDefinition(
135
+ "latency.jobs_p50_secs", "v1", "lower", "seconds",
136
+ "queue completion has a start and terminal timestamp", 5,
137
+ "Median queue-job completion latency."),
138
+ )
139
+
140
+
141
+ def benchmark_enabled() -> bool:
142
+ return os.environ.get(
143
+ "COLONY_BENCHMARK_ENABLED", "true").strip().lower() != "false"
144
+
145
+
146
+ def _now() -> float:
147
+ return time.time()
148
+
149
+
150
+ def week_id(dt: Optional[datetime] = None) -> str:
151
+ dt = dt or datetime.now(timezone.utc)
152
+ return dt.strftime("%G-W%V")
153
+
154
+
155
+ def week_window(week: str) -> Tuple[datetime, datetime]:
156
+ """[Monday 00:00 UTC, next Monday) for an ISO week id like 2026-W27."""
157
+ year, wk = week.split("-W")
158
+ start = datetime.fromisocalendar(int(year), int(wk), 1).replace(
159
+ tzinfo=timezone.utc)
160
+ return start, start + timedelta(days=7)
161
+
162
+
163
+ def previous_week(dt: Optional[datetime] = None) -> str:
164
+ dt = dt or datetime.now(timezone.utc)
165
+ return week_id(dt - timedelta(days=7))
166
+
167
+
168
+ def _percentile(values: List[float], pct: float) -> float:
169
+ if not values:
170
+ return 0.0
171
+ vs = sorted(values)
172
+ k = max(0, min(len(vs) - 1, int(round((pct / 100.0) * (len(vs) - 1)))))
173
+ return vs[k]
174
+
175
+
176
+ class BenchmarkStore:
177
+ """SQLite persistence: append-only samples + weekly rollups."""
178
+
179
+ def __init__(self, db_path: str) -> None:
180
+ self._conn = sqlite3.connect(db_path, check_same_thread=False)
181
+ self._conn.row_factory = sqlite3.Row
182
+ self._conn.execute("PRAGMA journal_mode=WAL")
183
+ self._lock = threading.Lock()
184
+ self._conn.executescript(
185
+ """
186
+ CREATE TABLE IF NOT EXISTS benchmark_samples (
187
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
188
+ metric TEXT NOT NULL,
189
+ value REAL NOT NULL,
190
+ source TEXT NOT NULL,
191
+ ts REAL NOT NULL,
192
+ meta TEXT
193
+ );
194
+ CREATE INDEX IF NOT EXISTS idx_bench_metric_ts
195
+ ON benchmark_samples(metric, ts);
196
+ CREATE TABLE IF NOT EXISTS benchmark_rollups (
197
+ week TEXT NOT NULL,
198
+ metric TEXT NOT NULL,
199
+ value REAL,
200
+ numerator REAL,
201
+ denominator REAL,
202
+ detail TEXT,
203
+ computed_at REAL NOT NULL,
204
+ PRIMARY KEY (week, metric)
205
+ );
206
+ CREATE TABLE IF NOT EXISTS benchmark_metric_definitions (
207
+ metric TEXT NOT NULL,
208
+ version TEXT NOT NULL,
209
+ direction TEXT NOT NULL,
210
+ unit TEXT NOT NULL,
211
+ evidence_query TEXT NOT NULL,
212
+ minimum_samples INTEGER NOT NULL,
213
+ description TEXT,
214
+ definition_hash TEXT NOT NULL,
215
+ created_at REAL NOT NULL,
216
+ PRIMARY KEY (metric, version)
217
+ );
218
+ CREATE TRIGGER IF NOT EXISTS benchmark_definition_no_update
219
+ BEFORE UPDATE ON benchmark_metric_definitions
220
+ BEGIN
221
+ SELECT RAISE(ABORT, 'benchmark definition is immutable');
222
+ END;
223
+ CREATE TRIGGER IF NOT EXISTS benchmark_definition_no_delete
224
+ BEFORE DELETE ON benchmark_metric_definitions
225
+ BEGIN
226
+ SELECT RAISE(ABORT, 'benchmark definition is immutable');
227
+ END;
228
+ """
229
+ )
230
+ self._additive_columns(
231
+ "benchmark_samples",
232
+ {
233
+ "sample_id": "TEXT",
234
+ "definition_version": "TEXT",
235
+ "sample_principal": "TEXT",
236
+ "source_ref": "TEXT",
237
+ "receipt_ref": "TEXT",
238
+ "evidence_status": "TEXT",
239
+ "exposure_id": "TEXT",
240
+ },
241
+ )
242
+ self._additive_columns(
243
+ "benchmark_rollups",
244
+ {
245
+ "definition_version": "TEXT",
246
+ "definition_hash": "TEXT",
247
+ "evidence_count": "INTEGER",
248
+ },
249
+ )
250
+ self._conn.execute(
251
+ "CREATE UNIQUE INDEX IF NOT EXISTS idx_bench_sample_id "
252
+ "ON benchmark_samples(sample_id) WHERE sample_id IS NOT NULL"
253
+ )
254
+ self._conn.commit()
255
+ for definition in BUILTIN_METRIC_DEFINITIONS:
256
+ self.register_definition(definition)
257
+
258
+ def _additive_columns(self, table: str,
259
+ columns: Dict[str, str]) -> None:
260
+ existing = {str(row[1]) for row in self._conn.execute(
261
+ f"PRAGMA table_info({table})").fetchall()}
262
+ for name, sql_type in columns.items():
263
+ if name not in existing:
264
+ self._conn.execute(
265
+ f"ALTER TABLE {table} ADD COLUMN {name} {sql_type}")
266
+
267
+ def register_definition(self, definition: MetricDefinition
268
+ ) -> Dict[str, Any]:
269
+ """Idempotently register an immutable metric definition."""
270
+
271
+ normalized = definition.normalized()
272
+ digest = hashlib.sha256(
273
+ _canonical(normalized).encode("utf-8")).hexdigest()
274
+ with self._lock:
275
+ current = self._conn.execute(
276
+ "SELECT * FROM benchmark_metric_definitions "
277
+ "WHERE metric=? AND version=?",
278
+ (normalized["metric"], normalized["version"]),
279
+ ).fetchone()
280
+ if current is not None:
281
+ result = dict(current)
282
+ if result["definition_hash"] != digest:
283
+ raise ValueError(
284
+ "metric definition is immutable; publish a new version")
285
+ return result
286
+ self._conn.execute(
287
+ "INSERT INTO benchmark_metric_definitions "
288
+ "(metric,version,direction,unit,evidence_query,minimum_samples,"
289
+ "description,definition_hash,created_at) VALUES (?,?,?,?,?,?,?,?,?)",
290
+ (
291
+ normalized["metric"], normalized["version"],
292
+ normalized["direction"], normalized["unit"],
293
+ normalized["evidence_query"], normalized["minimum_samples"],
294
+ normalized["description"], digest, _now(),
295
+ ),
296
+ )
297
+ self._conn.commit()
298
+ row = self._conn.execute(
299
+ "SELECT * FROM benchmark_metric_definitions "
300
+ "WHERE metric=? AND version=?",
301
+ (normalized["metric"], normalized["version"]),
302
+ ).fetchone()
303
+ assert row is not None
304
+ return dict(row)
305
+
306
+ def definition(self, metric: str, version: str) -> Optional[Dict[str, Any]]:
307
+ with self._lock:
308
+ row = self._conn.execute(
309
+ "SELECT * FROM benchmark_metric_definitions "
310
+ "WHERE metric=? AND version=?",
311
+ ((metric or "").strip().lower(),
312
+ (version or "").strip().lower()),
313
+ ).fetchone()
314
+ return dict(row) if row is not None else None
315
+
316
+ def definitions(self) -> List[Dict[str, Any]]:
317
+ with self._lock:
318
+ rows = self._conn.execute(
319
+ "SELECT * FROM benchmark_metric_definitions "
320
+ "ORDER BY metric, version").fetchall()
321
+ return [dict(row) for row in rows]
322
+
323
+ def add_sample(self, metric: str, value: float, *, source: str = "host",
324
+ ts: Optional[float] = None,
325
+ meta: Optional[Dict[str, Any]] = None) -> bool:
326
+ """Compatibility ingestion.
327
+
328
+ Legacy samples remain queryable but are explicitly unattested and are
329
+ never eligible for a P4 causal decision.
330
+ """
331
+ metric = (metric or "").strip().lower()
332
+ if not _METRIC_RE.match(metric):
333
+ return False
334
+ try:
335
+ value = float(value)
336
+ except (TypeError, ValueError):
337
+ return False
338
+ if not math.isfinite(value):
339
+ return False
340
+ with self._lock:
341
+ self._conn.execute(
342
+ "INSERT INTO benchmark_samples "
343
+ "(metric,value,source,ts,meta,sample_id,definition_version,"
344
+ "sample_principal,source_ref,receipt_ref,evidence_status,exposure_id)"
345
+ " VALUES (?,?,?,?,?,?,?,?,?,?,?,?)",
346
+ (
347
+ metric, value, (source or "host")[:64],
348
+ ts if ts is not None else _now(),
349
+ json.dumps(meta) if meta else None,
350
+ f"legacy-{uuid.uuid4().hex}", "legacy.unversioned",
351
+ f"legacy:{(source or 'host')[:96]}", None, None,
352
+ "legacy_unverified", None,
353
+ ))
354
+ self._conn.commit()
355
+ return True
356
+
357
+ def add_evidence_sample(
358
+ self,
359
+ metric: str,
360
+ value: float,
361
+ *,
362
+ definition_version: str,
363
+ sample_principal: str,
364
+ source_ref: str,
365
+ receipt_ref: Optional[str] = None,
366
+ sample_id: Optional[str] = None,
367
+ exposure_id: Optional[str] = None,
368
+ ts: Optional[float] = None,
369
+ meta: Optional[Dict[str, Any]] = None,
370
+ ) -> bool:
371
+ """Append one attested sample under a registered evidence contract.
372
+
373
+ A stable ``sample_id`` makes transport retries idempotent. Reusing it
374
+ with changed content is refused rather than silently replacing proof.
375
+ """
376
+
377
+ normalized_metric = (metric or "").strip().lower()
378
+ version = (definition_version or "").strip().lower()
379
+ definition = self.definition(normalized_metric, version)
380
+ if definition is None:
381
+ raise ValueError("registered metric definition is required")
382
+ principal = (sample_principal or "").strip()
383
+ source = (source_ref or "").strip()
384
+ receipt = (receipt_ref or "").strip() or None
385
+ exposure = (exposure_id or "").strip() or None
386
+ if not principal or len(principal) > 192:
387
+ raise ValueError("sample_principal is required")
388
+ if not source or len(source) > 512:
389
+ raise ValueError("source_ref is required")
390
+ try:
391
+ number = float(value)
392
+ except (TypeError, ValueError) as exc:
393
+ raise ValueError("sample value must be numeric") from exc
394
+ if not math.isfinite(number):
395
+ raise ValueError("sample value must be finite")
396
+ sid = (sample_id or f"bms-{uuid.uuid4().hex}").strip()
397
+ if not sid or len(sid) > 192:
398
+ raise ValueError("sample_id is malformed")
399
+ stamp = float(ts) if ts is not None else _now()
400
+ payload = {
401
+ "metric": normalized_metric,
402
+ "value": number,
403
+ "source": "evidence",
404
+ "ts": stamp,
405
+ "meta": meta or None,
406
+ "sample_id": sid,
407
+ "definition_version": version,
408
+ "sample_principal": principal,
409
+ "source_ref": source,
410
+ "receipt_ref": receipt,
411
+ "evidence_status": "verified" if receipt else "observed",
412
+ "exposure_id": exposure,
413
+ }
414
+ with self._lock:
415
+ existing = self._conn.execute(
416
+ "SELECT * FROM benchmark_samples WHERE sample_id=?", (sid,)
417
+ ).fetchone()
418
+ if existing is not None:
419
+ row = dict(existing)
420
+ comparable = {
421
+ key: row.get(key) for key in (
422
+ "metric", "value", "source", "sample_id",
423
+ "definition_version", "sample_principal", "source_ref",
424
+ "receipt_ref", "evidence_status", "exposure_id")
425
+ }
426
+ if comparable != {key: payload.get(key) for key in comparable}:
427
+ raise ValueError("sample_id replay changed immutable evidence")
428
+ try:
429
+ stored_meta = json.loads(row["meta"]) if row.get("meta") else None
430
+ except ValueError:
431
+ stored_meta = row.get("meta")
432
+ if _canonical(stored_meta) != _canonical(meta or None):
433
+ raise ValueError("sample_id replay changed immutable evidence")
434
+ if ts is not None and abs(float(row["ts"]) - stamp) > 1e-9:
435
+ raise ValueError("sample_id replay changed immutable evidence")
436
+ return True
437
+ self._conn.execute(
438
+ "INSERT INTO benchmark_samples "
439
+ "(metric,value,source,ts,meta,sample_id,definition_version,"
440
+ "sample_principal,source_ref,receipt_ref,evidence_status,exposure_id)"
441
+ " VALUES (?,?,?,?,?,?,?,?,?,?,?,?)",
442
+ (
443
+ payload["metric"], payload["value"], payload["source"],
444
+ payload["ts"], json.dumps(meta) if meta else None,
445
+ payload["sample_id"], payload["definition_version"],
446
+ payload["sample_principal"], payload["source_ref"],
447
+ payload["receipt_ref"], payload["evidence_status"],
448
+ payload["exposure_id"],
449
+ ),
450
+ )
451
+ self._conn.commit()
452
+ return True
453
+
454
+ def samples_in(self, since: float, until: float,
455
+ metric: Optional[str] = None) -> List[Dict[str, Any]]:
456
+ q = ("SELECT * FROM benchmark_samples"
457
+ " WHERE ts >= ? AND ts < ?")
458
+ params: List[Any] = [since, until]
459
+ if metric:
460
+ q += " AND metric = ?"
461
+ params.append(metric)
462
+ q += " ORDER BY ts ASC LIMIT 100000"
463
+ with self._lock:
464
+ rows = self._conn.execute(q, params).fetchall()
465
+ return [dict(r) for r in rows]
466
+
467
+ def evidence_samples_in(
468
+ self,
469
+ since: float,
470
+ until: float,
471
+ metric: Optional[str] = None,
472
+ *,
473
+ definition_version: Optional[str] = None,
474
+ exposure_id: Optional[str] = None,
475
+ require_receipt: bool = False,
476
+ ) -> List[Dict[str, Any]]:
477
+ q = ("SELECT * FROM benchmark_samples WHERE ts>=? AND ts<? "
478
+ "AND evidence_status IN ('observed','verified')")
479
+ params: List[Any] = [float(since), float(until)]
480
+ if metric:
481
+ q += " AND metric=?"
482
+ params.append((metric or "").strip().lower())
483
+ if definition_version:
484
+ q += " AND definition_version=?"
485
+ params.append((definition_version or "").strip().lower())
486
+ if exposure_id:
487
+ q += " AND exposure_id=?"
488
+ params.append(exposure_id)
489
+ if require_receipt:
490
+ q += " AND receipt_ref IS NOT NULL AND receipt_ref!=''"
491
+ q += " ORDER BY ts,id LIMIT 100000"
492
+ with self._lock:
493
+ rows = self._conn.execute(q, params).fetchall()
494
+ return [dict(row) for row in rows]
495
+
496
+ def write_rollup(self, week: str, metric: str, value: Optional[float], *,
497
+ numerator: Optional[float] = None,
498
+ denominator: Optional[float] = None,
499
+ detail: Optional[Dict[str, Any]] = None,
500
+ definition_version: Optional[str] = None,
501
+ evidence_count: Optional[int] = None) -> None:
502
+ definition_hash = None
503
+ if definition_version:
504
+ definition = self.definition(metric, definition_version)
505
+ if definition is None:
506
+ raise ValueError("registered metric definition is required")
507
+ definition_hash = definition["definition_hash"]
508
+ with self._lock:
509
+ self._conn.execute(
510
+ "INSERT OR REPLACE INTO benchmark_rollups"
511
+ " (week, metric, value, numerator, denominator, detail,"
512
+ " computed_at,definition_version,definition_hash,evidence_count)"
513
+ " VALUES (?,?,?,?,?,?,?,?,?,?)",
514
+ (week, metric, value, numerator, denominator,
515
+ json.dumps(detail) if detail else None, _now(),
516
+ definition_version, definition_hash, evidence_count))
517
+ self._conn.commit()
518
+
519
+ def rollups(self, weeks: int = 8) -> Dict[str, Dict[str, Any]]:
520
+ """{week: {metric: {value, numerator, denominator, detail}}},
521
+ newest weeks first, at most `weeks` distinct weeks."""
522
+ with self._lock:
523
+ rows = self._conn.execute(
524
+ "SELECT * FROM benchmark_rollups ORDER BY week DESC"
525
+ ).fetchall()
526
+ out: Dict[str, Dict[str, Any]] = {}
527
+ for r in rows:
528
+ wk = r["week"]
529
+ if wk not in out:
530
+ if len(out) >= weeks:
531
+ continue
532
+ out[wk] = {}
533
+ out[wk][r["metric"]] = {
534
+ "value": r["value"],
535
+ "numerator": r["numerator"],
536
+ "denominator": r["denominator"],
537
+ "detail": json.loads(r["detail"]) if r["detail"] else None,
538
+ "definition_version": r["definition_version"],
539
+ "definition_hash": r["definition_hash"],
540
+ "evidence_count": r["evidence_count"],
541
+ }
542
+ return out
543
+
544
+
545
+ class SelfhoodBenchmark:
546
+ """Weekly metric derivation over the shipped stores.
547
+
548
+ Dependencies may be injected (tests) or resolved lazily from the host
549
+ module globals at compute time (production), so construction order in
550
+ the server lifespan does not matter.
551
+ """
552
+
553
+ def __init__(self, store: BenchmarkStore, *,
554
+ commitments: Any = None, competence: Any = None,
555
+ journal: Any = None, comms: Any = None, graph: Any = None,
556
+ facts: Any = None, queue: Any = None,
557
+ corrections: Any = None,
558
+ owner_contact_id: Optional[str] = None,
559
+ probes: Optional[int] = None) -> None:
560
+ self.store = store
561
+ self._deps = {
562
+ "commitments": commitments, "competence": competence,
563
+ "journal": journal, "comms": comms, "graph": graph,
564
+ "facts": facts, "queue": queue, "corrections": corrections,
565
+ }
566
+ self._owner = owner_contact_id
567
+ self._probes = probes
568
+
569
+ # -- lazy dependency resolution -------------------------------------
570
+ _HOST_GLOBALS = {
571
+ "commitments": "_commitment_store", "comms": "_comms_log",
572
+ "graph": "_graph", "facts": "_facts_store", "queue": "_task_queue",
573
+ "corrections": "_learning_feedback_store",
574
+ }
575
+
576
+ def _dep(self, name: str) -> Any:
577
+ if self._deps.get(name) is not None:
578
+ return self._deps[name]
579
+ if name == "competence":
580
+ sm = self._host_attr("_self_model")
581
+ if sm is not None:
582
+ # SelfModel keeps its CompetenceStore as `.store`
583
+ return (getattr(sm, "store", None)
584
+ or getattr(sm, "competence", None))
585
+ return None
586
+ if name == "journal":
587
+ sm = self._host_attr("_self_model")
588
+ return getattr(sm, "journal", None) if sm is not None else None
589
+ g = self._HOST_GLOBALS.get(name)
590
+ return self._host_attr(g) if g else None
591
+
592
+ @staticmethod
593
+ def _host_attr(name: str) -> Any:
594
+ try:
595
+ from apsimo.api.routers import host
596
+ return getattr(host, name, None)
597
+ except Exception:
598
+ return None
599
+
600
+ @property
601
+ def owner_contact_id(self) -> str:
602
+ return (self._owner
603
+ or os.environ.get("COLONY_OWNER_CONTACT_ID", "").strip())
604
+
605
+ @property
606
+ def probe_count(self) -> int:
607
+ if self._probes is not None:
608
+ return self._probes
609
+ try:
610
+ return int(os.environ.get("COLONY_BENCHMARK_PROBES", "8"))
611
+ except ValueError:
612
+ return 8
613
+
614
+ # -- derivations ------------------------------------------------------
615
+ async def compute_week(self, week: Optional[str] = None) -> Dict[str, Any]:
616
+ """Derive every computable metric for `week` (default: the previous
617
+ completed ISO week), persist rollups, and return them. Metrics whose
618
+ source is unavailable are omitted, never zero-filled."""
619
+ wk = week or previous_week()
620
+ start, end = week_window(wk)
621
+ since, until = start.timestamp(), end.timestamp()
622
+ out: Dict[str, Any] = {}
623
+
624
+ for name, fn in (
625
+ ("commitments.fulfillment", self._m_commitments),
626
+ ("delivery.success", self._m_delivery),
627
+ ("actions.success", self._m_actions),
628
+ ("journal.acted_share", self._m_journal),
629
+ ("initiative.acceptance", self._m_acceptance),
630
+ ("responses.correction_rate", self._m_corrections),
631
+ ):
632
+ try:
633
+ res = fn(start, end, since, until)
634
+ if res is not None:
635
+ out[name] = res
636
+ except Exception as exc:
637
+ logger.warning("benchmark %s failed: %s", name, exc)
638
+ for name, coro in (
639
+ ("recall.fact_coverage", self._m_recall(since, until)),
640
+ ("latency.jobs_p50_secs", self._m_jobs(start, end)),
641
+ ):
642
+ try:
643
+ res = await coro
644
+ if res is not None:
645
+ out[name] = res
646
+ except Exception as exc:
647
+ logger.warning("benchmark %s failed: %s", name, exc)
648
+ try:
649
+ out.update(self._m_calibration(since))
650
+ except Exception as exc:
651
+ logger.warning("benchmark calibration failed: %s", exc)
652
+ out.update(self._m_submitted(since, until, skip=set(out)))
653
+
654
+ for metric, r in out.items():
655
+ definition_version = self._definition_version(metric)
656
+ self.store.write_rollup(
657
+ wk, metric, r.get("value"), numerator=r.get("numerator"),
658
+ denominator=r.get("denominator"), detail=r.get("detail"),
659
+ definition_version=definition_version,
660
+ evidence_count=(int(r.get("denominator"))
661
+ if r.get("denominator") is not None else None))
662
+ logger.info("benchmark week %s: %d metrics", wk, len(out))
663
+ return {"week": wk, "metrics": out}
664
+
665
+ @staticmethod
666
+ def _definition_version(metric: str) -> Optional[str]:
667
+ if cognition_p4_mode() != "live":
668
+ # Existing rollups are intentionally left labelled as legacy
669
+ # until the controlled path is explicitly enabled.
670
+ return None
671
+ versions = {
672
+ "commitments.fulfillment": "v2",
673
+ "delivery.success": "v1",
674
+ "actions.success": "v2",
675
+ "journal.acted_share": "v1",
676
+ "initiative.acceptance": "v2",
677
+ "responses.correction_rate": "v1",
678
+ "recall.fact_coverage": "v2",
679
+ "latency.jobs_p50_secs": "v1",
680
+ }
681
+ return versions.get(metric)
682
+
683
+ def _m_commitments(self, start, end, since, until):
684
+ cs = self._dep("commitments")
685
+ if cs is None:
686
+ return None
687
+ if cognition_p4_mode() == "live":
688
+ try:
689
+ rows = cs.list(limit=10000).get("commitments", [])
690
+ except (AttributeError, TypeError):
691
+ return None
692
+ due_cohort: List[Dict[str, Any]] = []
693
+ for raw in rows:
694
+ row = raw if isinstance(raw, dict) else vars(raw)
695
+ due = self._parse_instant(row.get("due_at"))
696
+ if due is None or not (start <= due < end):
697
+ continue
698
+ if str(row.get("status") or "").lower() == "cancelled":
699
+ continue
700
+ due_cohort.append(row)
701
+ if not due_cohort:
702
+ return None
703
+ fulfilled = 0
704
+ late = 0
705
+ for row in due_cohort:
706
+ due = self._parse_instant(row.get("due_at"))
707
+ completed = self._parse_instant(row.get("fulfilled_at"))
708
+ if completed is not None and due is not None:
709
+ if completed <= due:
710
+ fulfilled += 1
711
+ else:
712
+ late += 1
713
+ return {
714
+ "value": fulfilled / len(due_cohort),
715
+ "numerator": fulfilled,
716
+ "denominator": len(due_cohort),
717
+ "detail": {
718
+ "metric_definition": "commitments.fulfillment/v2",
719
+ "cohort": "due_at_in_iso_week",
720
+ "late": late,
721
+ "open_or_missed": len(due_cohort) - fulfilled - late,
722
+ },
723
+ }
724
+ fulfilled = 0
725
+ for c in (cs.list(status=["fulfilled"], limit=500)
726
+ .get("commitments", [])):
727
+ fat = (c.get("fulfilled_at") or "") if isinstance(c, dict) else\
728
+ (getattr(c, "fulfilled_at", "") or "")
729
+ if fat and start.isoformat() <= str(fat) < end.isoformat():
730
+ fulfilled += 1
731
+ overdue_open = len(cs.get_overdue())
732
+ den = fulfilled + overdue_open
733
+ if den == 0:
734
+ return None
735
+ return {"value": fulfilled / den, "numerator": fulfilled,
736
+ "denominator": den, "detail": {"overdue_open": overdue_open}}
737
+
738
+ @staticmethod
739
+ def _parse_instant(value: Any) -> Optional[datetime]:
740
+ if not value:
741
+ return None
742
+ try:
743
+ parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
744
+ except (TypeError, ValueError):
745
+ return None
746
+ if parsed.tzinfo is None:
747
+ parsed = parsed.replace(tzinfo=timezone.utc)
748
+ return parsed.astimezone(timezone.utc)
749
+
750
+ def _events(self, domain: str, since: float):
751
+ comp = self._dep("competence")
752
+ if comp is None:
753
+ return None
754
+ return [e for e in comp.events(domain, since=since,
755
+ include_shadow=False)]
756
+
757
+ def _competence_state(self, domains: List[str], since: float,
758
+ until: float) -> Tuple[int, List[Dict[str, Any]]]:
759
+ """Correction revision and unresolved provenance gaps for a slice."""
760
+ comp = self._dep("competence")
761
+ if comp is None:
762
+ return 0, []
763
+ revision = 0
764
+ gaps: List[Dict[str, Any]] = []
765
+ try:
766
+ revision = int(comp.reconciliation_revision(
767
+ domains=domains, since=since, until=until))
768
+ except (AttributeError, TypeError, ValueError):
769
+ pass
770
+ try:
771
+ for domain in domains:
772
+ gaps.extend(comp.active_evidence_gaps(
773
+ domain, since=since, until=until))
774
+ except (AttributeError, TypeError):
775
+ pass
776
+ return revision, gaps
777
+
778
+ @staticmethod
779
+ def _unavailable_competence(
780
+ revision: int, gaps: List[Dict[str, Any]]) -> Dict[str, Any]:
781
+ return {
782
+ "value": None, "numerator": None, "denominator": None,
783
+ "detail": {
784
+ "available": False,
785
+ "reason": "competence_evidence_gap",
786
+ "competence_reconciliation_revision": revision,
787
+ "gaps": [{
788
+ "id": g.get("reconciliation_id"),
789
+ "domain": g.get("domain"),
790
+ "since_ts": g.get("since_ts"),
791
+ "until_ts": g.get("until_ts"),
792
+ "reason": g.get("reason"),
793
+ } for g in gaps],
794
+ },
795
+ }
796
+
797
+ def _m_delivery(self, start, end, since, until):
798
+ revision, gaps = self._competence_state(
799
+ ["delivery"], since, until)
800
+ if gaps:
801
+ return self._unavailable_competence(revision, gaps)
802
+ evs = self._events("delivery", since)
803
+ if evs is None:
804
+ return None
805
+ evs = [e for e in evs if e["ts"] < until]
806
+ if cognition_p4_mode() == "live":
807
+ evs = [e for e in evs if e.get("evidence_status") == "verified"]
808
+ if not evs:
809
+ return None
810
+ ok = sum(1 for e in evs if e["outcome"] == "success")
811
+ return {"value": ok / len(evs), "numerator": ok,
812
+ "denominator": len(evs),
813
+ "detail": {
814
+ "n": len(evs),
815
+ "metric_definition": "colony.delivery-success/v1",
816
+ "competence_reconciliation_revision": revision,
817
+ }}
818
+
819
+ def _m_actions(self, start, end, since, until):
820
+ comp = self._dep("competence")
821
+ if comp is None:
822
+ return None
823
+ domains = []
824
+ for row in comp.snapshot():
825
+ dom = row.get("domain") if isinstance(row, dict) else None
826
+ if dom and dom != "delivery":
827
+ domains.append(dom)
828
+ revision, gaps = self._competence_state(domains, since, until)
829
+ if gaps:
830
+ return self._unavailable_competence(revision, gaps)
831
+ per: Dict[str, Dict[str, int]] = {}
832
+ ok = n = 0
833
+ for dom in domains:
834
+ evs = [e for e in comp.events(dom, since=since,
835
+ include_shadow=False)
836
+ if e["ts"] < until]
837
+ if cognition_p4_mode() == "live":
838
+ evs = [e for e in evs
839
+ if e.get("evidence_status") == "verified"
840
+ and str(e.get("outcome_contract") or "").lower()
841
+ not in {"", "legacy.unversioned"}]
842
+ if not evs:
843
+ continue
844
+ d_ok = sum(1 for e in evs if e["outcome"] == "success")
845
+ per[dom] = {"success": d_ok, "n": len(evs)}
846
+ ok += d_ok
847
+ n += len(evs)
848
+ if n == 0:
849
+ return None
850
+ return {"value": ok / n, "numerator": ok, "denominator": n,
851
+ "detail": {
852
+ "domains": per,
853
+ "metric_definition": "colony.actions-success/v2",
854
+ "competence_reconciliation_revision": revision,
855
+ }}
856
+
857
+ def _m_journal(self, start, end, since, until):
858
+ j = self._dep("journal")
859
+ if j is None:
860
+ return None
861
+ entries = j.recent(limit=2000, since=since)
862
+ counts: Dict[str, int] = {}
863
+ for e in entries:
864
+ if e.get("ts", 0) >= until:
865
+ continue
866
+ d = e.get("decision") or "unknown"
867
+ counts[d] = counts.get(d, 0) + 1
868
+ gated = sum(counts.get(k, 0)
869
+ for k in ("acted", "asked", "held", "blocked"))
870
+ if gated == 0:
871
+ return None
872
+ return {"value": counts.get("acted", 0) / gated,
873
+ "numerator": counts.get("acted", 0), "denominator": gated,
874
+ "detail": {"decisions": counts}}
875
+
876
+ def _m_acceptance(self, start, end, since, until):
877
+ """Owner responded (inbound comm) within 24h of a delivery success."""
878
+ owner = self.owner_contact_id
879
+ comms = self._dep("comms")
880
+ if not owner or comms is None:
881
+ return None
882
+ revision, gaps = self._competence_state(
883
+ ["delivery"], since, until)
884
+ if gaps:
885
+ return self._unavailable_competence(revision, gaps)
886
+ evs = self._events("delivery", since)
887
+ if evs is None:
888
+ return None
889
+ deliveries = [e for e in evs
890
+ if e["outcome"] == "success" and e["ts"] < until]
891
+ if not deliveries:
892
+ return None
893
+ if cognition_p4_mode() == "live":
894
+ # An unrelated inbound message is not evidence that an initiative
895
+ # was useful. Only an explicit reaction naming the delivery or its
896
+ # source message enters this cohort.
897
+ refs: List[str] = []
898
+ for event in deliveries:
899
+ if event.get("evidence_status") != "verified":
900
+ continue
901
+ evidence = event.get("evidence") or {}
902
+ if isinstance(evidence, str):
903
+ try:
904
+ evidence = json.loads(evidence)
905
+ except ValueError:
906
+ evidence = {}
907
+ ref = (evidence.get("delivery_id")
908
+ if isinstance(evidence, dict) else None)
909
+ ref = ref or event.get("source_ref")
910
+ if ref:
911
+ refs.append(str(ref))
912
+ refs = list(dict.fromkeys(refs))
913
+ if not refs or not hasattr(comms, "reactions_for_refs"):
914
+ return None
915
+ reactions = comms.reactions_for_refs(
916
+ owner, refs, since_iso=start.isoformat(),
917
+ until_iso=end.isoformat())
918
+ by_ref = {str(row.get("reply_to_ref")): row
919
+ for row in reactions if row.get("reply_to_ref")}
920
+ accepted_names = {"accepted", "acknowledged", "actioned"}
921
+ negative_names = {"negative", "dismissed", "corrected", "rejected"}
922
+ accepted = sum(
923
+ 1 for ref in refs
924
+ if str((by_ref.get(ref) or {}).get("reaction") or "").lower()
925
+ in accepted_names)
926
+ negative = sum(
927
+ 1 for ref in refs
928
+ if str((by_ref.get(ref) or {}).get("reaction") or "").lower()
929
+ in negative_names)
930
+ return {
931
+ "value": accepted / len(refs),
932
+ "numerator": accepted,
933
+ "denominator": len(refs),
934
+ "detail": {
935
+ "deliveries": len(refs),
936
+ "negative": negative,
937
+ "unanswered": len(refs) - len(by_ref),
938
+ "metric_definition": "initiative.acceptance/v2",
939
+ "binding": "reply_to_ref",
940
+ "competence_reconciliation_revision": revision,
941
+ },
942
+ }
943
+ inbound = comms.inbound_since(owner, start.isoformat())
944
+ in_ts = []
945
+ for t in inbound:
946
+ try:
947
+ in_ts.append(datetime.fromisoformat(
948
+ str(t).replace("Z", "+00:00")).timestamp())
949
+ except ValueError:
950
+ continue
951
+ accepted = sum(
952
+ 1 for e in deliveries
953
+ if any(e["ts"] < t <= e["ts"] + 86400 for t in in_ts))
954
+ return {"value": accepted / len(deliveries), "numerator": accepted,
955
+ "denominator": len(deliveries),
956
+ "detail": {
957
+ "deliveries": len(deliveries),
958
+ "metric_definition": "colony.initiative-acceptance/v1",
959
+ "competence_reconciliation_revision": revision,
960
+ }}
961
+
962
+ def _m_corrections(self, start, end, since, until):
963
+ """Corrections over the same receipt-backed outbound response cohort."""
964
+
965
+ if cognition_p4_mode() != "live":
966
+ return None
967
+ owner = self.owner_contact_id
968
+ comms = self._dep("comms")
969
+ corrections = self._dep("corrections")
970
+ if (not owner or comms is None or corrections is None
971
+ or not hasattr(comms, "outbound_between")
972
+ or not hasattr(corrections, "between")):
973
+ return None
974
+ outbound = comms.outbound_between(
975
+ owner, start.isoformat(), end.isoformat(), require_receipt=True)
976
+ cohort: Dict[str, Dict[str, Any]] = {}
977
+ for row in outbound:
978
+ ref = row.get("external_ref") or row.get("receipt_ref")
979
+ if ref:
980
+ cohort[str(ref)] = row
981
+ if not cohort:
982
+ return None
983
+ corrected: set[str] = set()
984
+ for item in corrections.between(
985
+ start.isoformat(), end.isoformat(), person_id=owner):
986
+ context_ref = (item.get("context_hash") if isinstance(item, dict)
987
+ else getattr(item, "context_hash", ""))
988
+ if context_ref in cohort:
989
+ corrected.add(str(context_ref))
990
+ return {
991
+ "value": len(corrected) / len(cohort),
992
+ "numerator": len(corrected),
993
+ "denominator": len(cohort),
994
+ "detail": {
995
+ "metric_definition": "responses.correction_rate/v1",
996
+ "cohort": "receipt_backed_owner_outbound",
997
+ "correction_binding": "context_hash_to_external_ref",
998
+ },
999
+ }
1000
+
1001
+ async def _m_recall(self, since, until):
1002
+ """Probe: re-query high-confidence shared facts against graph recall
1003
+ and grade by token coverage. Records each probe as a sample."""
1004
+ rows = self._probe_rows()
1005
+ if rows is None or not rows:
1006
+ return None
1007
+ picks = random.sample(rows, min(self.probe_count, len(rows)))
1008
+ return await self._run_probes(picks, source="benchmark")
1009
+
1010
+ async def run_recall_probe(self, probes: int = 50,
1011
+ seed: Optional[int] = None
1012
+ ) -> Optional[Dict[str, Any]]:
1013
+ """On-demand recall probe: the same derivation as the weekly
1014
+ recall.fact_coverage metric, but with a seeded, deterministic fact
1015
+ pick so before/after comparisons measure the recall path rather than
1016
+ sampling noise. Read-only against the graph. Samples are recorded
1017
+ with source="manual-probe"; rollups never read recall.probe samples,
1018
+ so manual probing cannot distort the weekly scorecard."""
1019
+ rows = self._probe_rows()
1020
+ if rows is None or not rows:
1021
+ return None
1022
+ n = max(1, min(int(probes), 100))
1023
+ picks = random.Random(seed).sample(rows, min(n, len(rows)))
1024
+ out = await self._run_probes(picks, source="manual-probe")
1025
+ if out is not None:
1026
+ out["detail"]["seed"] = seed
1027
+ out["detail"]["source"] = "manual-probe"
1028
+ return out
1029
+
1030
+ def _probe_rows(self) -> Optional[List[Dict[str, Any]]]:
1031
+ """High-confidence shared facts to probe, or None when a source
1032
+ (graph or facts store) is unavailable — honest-skip, never zero."""
1033
+ graph = self._dep("graph")
1034
+ facts = self._dep("facts")
1035
+ if graph is None or facts is None:
1036
+ return None
1037
+ kwargs: Dict[str, Any] = {"min_confidence": 0.75, "limit": 200}
1038
+ if cognition_p4_mode() == "live":
1039
+ owner = self.owner_contact_id
1040
+ if not owner:
1041
+ return None
1042
+ # The store may call this key contact_id or subject_person_id; the
1043
+ # read is narrowed in both the query and the post-filter.
1044
+ kwargs["contact_id"] = owner
1045
+ rows = facts.list_facts(**kwargs).get("facts", [])
1046
+ if cognition_p4_mode() != "live":
1047
+ return rows
1048
+ allowed: List[Dict[str, Any]] = []
1049
+ owner = self.owner_contact_id
1050
+ for raw in rows:
1051
+ row = raw if isinstance(raw, dict) else vars(raw)
1052
+ subject = (row.get("subject_person_id")
1053
+ or row.get("contact_id") or "")
1054
+ if not row.get("id"):
1055
+ continue
1056
+ shareability = str(
1057
+ row.get("shareability") or "owner_private").lower()
1058
+ if str(subject) != owner:
1059
+ continue
1060
+ if shareability not in {"owner_private", "shared", "public"}:
1061
+ continue
1062
+ allowed.append(row)
1063
+ return allowed
1064
+
1065
+ async def _run_probes(self, picks: List[Any],
1066
+ source: str) -> Optional[Dict[str, Any]]:
1067
+ """Grade each picked fact against graph recall (token coverage),
1068
+ recording one recall.probe sample per fact under `source`."""
1069
+ graph = self._dep("graph")
1070
+ hits = 0
1071
+ judged = 0
1072
+ for f in picks:
1073
+ fact = (f.get("fact") if isinstance(f, dict)
1074
+ else getattr(f, "fact", "")) or ""
1075
+ if not fact.strip():
1076
+ continue
1077
+ subject = (f.get("subject_person_id") or f.get("contact_id")
1078
+ if isinstance(f, dict) else
1079
+ getattr(f, "subject_person_id", None)
1080
+ or getattr(f, "contact_id", None))
1081
+ try:
1082
+ # bound each probe so a wedged graph connection can't hang the
1083
+ # benchmark (and, through it, the autonomy tick)
1084
+ recall_kwargs = {"limit": 5, "min_confidence": 0.1}
1085
+ if cognition_p4_mode() == "live":
1086
+ if not subject:
1087
+ continue
1088
+ recall_kwargs["person_id"] = str(subject)
1089
+ results = await asyncio.wait_for(
1090
+ graph.recall(fact, **recall_kwargs), timeout=8.0)
1091
+ except (Exception, asyncio.TimeoutError):
1092
+ continue
1093
+ hit = 1.0 if self._covered(fact, results) else 0.0
1094
+ hits += int(hit)
1095
+ judged += 1
1096
+ fact_id = (f.get("id") if isinstance(f, dict)
1097
+ else getattr(f, "id", None))
1098
+ if cognition_p4_mode() == "live":
1099
+ self.store.add_evidence_sample(
1100
+ "recall.fact_coverage", hit, definition_version="v2",
1101
+ sample_principal="benchmark:recall-probe",
1102
+ source_ref=f"fact:{fact_id}",
1103
+ sample_id=(f"recall-{source}-{fact_id}-"
1104
+ f"{int(_now() * 1000000)}"),
1105
+ meta={"fact_id": fact_id, "subject_person_id": subject},
1106
+ )
1107
+ else:
1108
+ self.store.add_sample(
1109
+ "recall.probe", hit, source=source,
1110
+ meta={"fact_id": fact_id})
1111
+ n = judged if cognition_p4_mode() == "live" else len(picks)
1112
+ if n == 0:
1113
+ return None
1114
+ return {"value": hits / n, "numerator": hits, "denominator": n,
1115
+ "detail": {
1116
+ "probes": n,
1117
+ **({"metric_definition": "recall.fact_coverage/v2",
1118
+ "viewer_scope": self.owner_contact_id}
1119
+ if cognition_p4_mode() == "live" else {}),
1120
+ }}
1121
+
1122
+ @staticmethod
1123
+ def _covered(fact: str, results: List[Dict[str, Any]],
1124
+ threshold: float = 0.5) -> bool:
1125
+ words = {w for w in re.findall(r"[a-z0-9]+", fact.lower())
1126
+ if len(w) > 3}
1127
+ if not words:
1128
+ return False
1129
+ for r in results or []:
1130
+ content = str((r or {}).get("content", "")).lower()
1131
+ if not content:
1132
+ continue
1133
+ got = sum(1 for w in words if w in content)
1134
+ if got / len(words) >= threshold:
1135
+ return True
1136
+ return False
1137
+
1138
+ async def _m_jobs(self, start, end):
1139
+ queue = self._dep("queue")
1140
+ if queue is None:
1141
+ return None
1142
+ # host wires the TaskQueueManager wrapper; the raw QueueManager
1143
+ # (which owns completed_durations) sits at .queue
1144
+ if not hasattr(queue, "completed_durations"):
1145
+ queue = getattr(queue, "queue", None)
1146
+ if queue is None or not hasattr(queue, "completed_durations"):
1147
+ return None
1148
+ durs = [d for d in await queue.completed_durations(
1149
+ start.isoformat(), end.isoformat()) if d >= 0]
1150
+ if not durs:
1151
+ return None
1152
+ return {"value": _percentile(durs, 50),
1153
+ "numerator": None, "denominator": None,
1154
+ "detail": {"p50": _percentile(durs, 50),
1155
+ "p95": _percentile(durs, 95), "n": len(durs)}}
1156
+
1157
+ def _m_calibration(self, since: float) -> Dict[str, Any]:
1158
+ """Per-domain prediction calibration from the expectation engine
1159
+ (Mind M3a), expressed as accuracy = 1 - Brier so higher is better and
1160
+ it fits the benchmark's higher-is-better convention."""
1161
+ eng = self._host_attr("_expectations")
1162
+ if eng is None:
1163
+ return {}
1164
+ out: Dict[str, Any] = {}
1165
+ try:
1166
+ cal = eng.calibration(since=since)
1167
+ except Exception:
1168
+ return {}
1169
+ for domain, r in (cal or {}).items():
1170
+ brier = r.get("brier")
1171
+ if brier is None:
1172
+ continue
1173
+ out[f"calibration.{domain}"] = {
1174
+ "value": max(0.0, 1.0 - float(brier)),
1175
+ "numerator": None, "denominator": None,
1176
+ "detail": {"brier": brier, "n": r.get("n"),
1177
+ "hit_rate": r.get("hit_rate")}}
1178
+ return out
1179
+
1180
+ def _m_submitted(self, since: float, until: float,
1181
+ skip: Optional[set] = None) -> Dict[str, Any]:
1182
+ """Roll up host-submitted samples generically."""
1183
+ skip = skip or set()
1184
+ by_metric: Dict[str, List[float]] = {}
1185
+ for s in self.store.samples_in(since, until):
1186
+ if s["metric"] == "recall.probe" or s["metric"] in skip:
1187
+ continue
1188
+ if cognition_p4_mode() == "live":
1189
+ definition = self.store.definition(
1190
+ s["metric"], s.get("definition_version") or "")
1191
+ if (definition is None
1192
+ or s.get("evidence_status") not in {"observed", "verified"}
1193
+ or not s.get("sample_principal")
1194
+ or not s.get("source_ref")):
1195
+ continue
1196
+ by_metric.setdefault(s["metric"], []).append(s["value"])
1197
+ out: Dict[str, Any] = {}
1198
+ for metric, vals in by_metric.items():
1199
+ if metric.startswith("latency."):
1200
+ out[metric] = {
1201
+ "value": _percentile(vals, 50),
1202
+ "numerator": None, "denominator": None,
1203
+ "detail": {"p50": _percentile(vals, 50),
1204
+ "p95": _percentile(vals, 95),
1205
+ "n": len(vals)}}
1206
+ else:
1207
+ out[metric] = {
1208
+ "value": sum(vals) / len(vals),
1209
+ "numerator": None, "denominator": None,
1210
+ "detail": {"n": len(vals),
1211
+ "min": min(vals), "max": max(vals)}}
1212
+ return out
1213
+
1214
+ # -- read side --------------------------------------------------------
1215
+ def snapshot(self, weeks: int = 8) -> Dict[str, Any]:
1216
+ """Rollups for the last N weeks plus latest-vs-previous deltas."""
1217
+ rolls = self.store.rollups(weeks=weeks)
1218
+ ordered = sorted(rolls.keys(), reverse=True)
1219
+ # A persisted score computed before a correction is unsafe to show as
1220
+ # current truth. Hide it until compute_week rebuilds that exact slice.
1221
+ comp = self._dep("competence")
1222
+ if comp is not None:
1223
+ all_action_domains: List[str] = []
1224
+ try:
1225
+ all_action_domains = [
1226
+ str(r.get("domain")) for r in comp.snapshot()
1227
+ if isinstance(r, dict) and r.get("domain")
1228
+ and r.get("domain") != "delivery"]
1229
+ except Exception:
1230
+ pass
1231
+ for wk, metrics in rolls.items():
1232
+ try:
1233
+ start, end = week_window(wk)
1234
+ except (TypeError, ValueError):
1235
+ continue
1236
+ since, until = start.timestamp(), end.timestamp()
1237
+ for metric, domains in (
1238
+ ("delivery.success", ["delivery"]),
1239
+ ("initiative.acceptance", ["delivery"]),
1240
+ ("actions.success", all_action_domains),
1241
+ ):
1242
+ row = metrics.get(metric)
1243
+ if row is None:
1244
+ continue
1245
+ detail = dict(row.get("detail") or {})
1246
+ metric_domains = list(domains)
1247
+ if metric == "actions.success":
1248
+ stored_domains = detail.get("domains") or {}
1249
+ if isinstance(stored_domains, dict):
1250
+ metric_domains = sorted(
1251
+ set(metric_domains) | set(stored_domains))
1252
+ revision, gaps = self._competence_state(
1253
+ metric_domains, since, until)
1254
+ recorded = int(detail.get(
1255
+ "competence_reconciliation_revision") or 0)
1256
+ if gaps:
1257
+ row.update(self._unavailable_competence(
1258
+ revision, gaps))
1259
+ elif revision > recorded:
1260
+ row["value"] = None
1261
+ row["numerator"] = None
1262
+ row["denominator"] = None
1263
+ detail.update({
1264
+ "available": False,
1265
+ "reason": (
1266
+ "stale_after_competence_reconciliation"),
1267
+ "computed_revision": recorded,
1268
+ "required_revision": revision,
1269
+ })
1270
+ row["detail"] = detail
1271
+ trends: Dict[str, Any] = {}
1272
+ if len(ordered) >= 2:
1273
+ cur, prev = rolls[ordered[0]], rolls[ordered[1]]
1274
+ for metric, r in cur.items():
1275
+ pv = (prev.get(metric) or {}).get("value")
1276
+ if r.get("value") is not None and pv is not None:
1277
+ trends[metric] = round(r["value"] - pv, 4)
1278
+ return {
1279
+ "format": "colony.selfhood-benchmark/v2",
1280
+ "mode": cognition_p4_mode(),
1281
+ "weeks": ordered,
1282
+ "rollups": rolls,
1283
+ "trends": trends,
1284
+ "latest": ordered[0] if ordered else None,
1285
+ "definitions": self.store.definitions(),
1286
+ }
1287
+
1288
+ def canonical_summary(self, weeks: int = 8) -> Dict[str, Any]:
1289
+ """Canonical replacement payload for the deprecated legacy CPI API."""
1290
+
1291
+ return {
1292
+ "deprecated_cpi": True,
1293
+ "canonical": "selfhood_benchmark",
1294
+ "canonical_endpoint": "/v1/host/self/benchmark",
1295
+ **self.snapshot(weeks=weeks),
1296
+ }
1297
+
1298
+
1299
+ def legacy_cpi_payload(benchmark: Optional[SelfhoodBenchmark],
1300
+ weeks: int = 8) -> Dict[str, Any]:
1301
+ """A truthful compatibility response; no fabricated CPI dimensions."""
1302
+
1303
+ if benchmark is None:
1304
+ return {
1305
+ "deprecated": True,
1306
+ "available": False,
1307
+ "canonical_endpoint": "/v1/host/self/benchmark",
1308
+ "reason": "canonical benchmark is not wired",
1309
+ }
1310
+ return {
1311
+ "deprecated": True,
1312
+ "available": True,
1313
+ **benchmark.canonical_summary(weeks=weeks),
1314
+ }