apsimo 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apsimo/__init__.py +38 -0
- apsimo/__main__.py +6 -0
- apsimo/agent/__init__.py +6 -0
- apsimo/agent/client.py +276 -0
- apsimo/agent/models.py +46 -0
- apsimo/agents/__init__.py +20 -0
- apsimo/agents/models.py +264 -0
- apsimo/agents/store.py +861 -0
- apsimo/agents/websocket.py +522 -0
- apsimo/api/__init__.py +1 -0
- apsimo/api/auth_telemetry.py +287 -0
- apsimo/api/authority.py +1203 -0
- apsimo/api/contact_grants.py +347 -0
- apsimo/api/middleware.py +483 -0
- apsimo/api/routers/__init__.py +1 -0
- apsimo/api/routers/commitment_work.py +265 -0
- apsimo/api/routers/context_gate.py +123 -0
- apsimo/api/routers/executions.py +140 -0
- apsimo/api/routers/followup_plans.py +147 -0
- apsimo/api/routers/governed_actions.py +162 -0
- apsimo/api/routers/host.py +14473 -0
- apsimo/api/routers/initiative_work.py +115 -0
- apsimo/api/routers/mining.py +104 -0
- apsimo/api/routers/observations.py +110 -0
- apsimo/api/routers/social_state.py +225 -0
- apsimo/api/routers/task_queue.py +2715 -0
- apsimo/api/routers/temporal_followups.py +251 -0
- apsimo/api/routers/transport.py +110 -0
- apsimo/api/routers/transport_ingress_api.py +240 -0
- apsimo/api/schemas/__init__.py +1 -0
- apsimo/api/schemas/host.py +1949 -0
- apsimo/autonomy/cli.py +110 -0
- apsimo/autonomy/condition_worker.py +437 -0
- apsimo/autonomy/config.py +424 -0
- apsimo/autonomy/loop.py +4316 -0
- apsimo/autonomy/registry.py +339 -0
- apsimo/autonomy/scheduler.py +1822 -0
- apsimo/autonomy/synthesis.py +449 -0
- apsimo/backup.py +962 -0
- apsimo/beliefs/__init__.py +23 -0
- apsimo/beliefs/contradictions.py +109 -0
- apsimo/beliefs/decay.py +61 -0
- apsimo/beliefs/engine.py +479 -0
- apsimo/beliefs/models.py +67 -0
- apsimo/beliefs/promotion.py +41 -0
- apsimo/beliefs/resolve.py +58 -0
- apsimo/beliefs/source_claims.py +690 -0
- apsimo/beliefs/source_projection.py +883 -0
- apsimo/beliefs/source_time.py +208 -0
- apsimo/beliefs/store.py +133 -0
- apsimo/briefings/aggregators.py +824 -0
- apsimo/briefings/composer.py +420 -0
- apsimo/briefings/config.py +55 -0
- apsimo/briefings/delivery.py +439 -0
- apsimo/briefings/engagement.py +97 -0
- apsimo/briefings/engine.py +274 -0
- apsimo/briefings/enhancer.py +99 -0
- apsimo/briefings/models.py +183 -0
- apsimo/briefings/scheduler.py +382 -0
- apsimo/briefings/store.py +435 -0
- apsimo/chain/__init__.py +48 -0
- apsimo/chain/block.py +100 -0
- apsimo/chain/cli.py +704 -0
- apsimo/chain/genesis.py +443 -0
- apsimo/chain/identity.py +416 -0
- apsimo/chain/keys.py +1025 -0
- apsimo/chain/local_keys.py +187 -0
- apsimo/chain/manager.py +290 -0
- apsimo/chain/node.py +163 -0
- apsimo/chain/plugin_transactions.py +371 -0
- apsimo/chain/protocol.py +220 -0
- apsimo/chain/state_machine.py +676 -0
- apsimo/chain/storage.py +503 -0
- apsimo/chain/transactions.py +250 -0
- apsimo/chain/validation.py +397 -0
- apsimo/channels/__init__.py +1 -0
- apsimo/channels/manifest.py +31 -0
- apsimo/channels/migrations/001_channels_schema.sql +12 -0
- apsimo/channels/phone_gateways.py +42 -0
- apsimo/channels/presence.py +188 -0
- apsimo/channels/router.py +235 -0
- apsimo/channels/store.py +231 -0
- apsimo/cli.py +2688 -0
- apsimo/cognition/__init__.py +11 -0
- apsimo/cognition/charter.py +398 -0
- apsimo/cognition/drive_governance.py +3530 -0
- apsimo/cognition/evidence_pipeline.py +1627 -0
- apsimo/cognition/external_events.py +932 -0
- apsimo/cognition/goal_spine.py +3488 -0
- apsimo/cognition/introspection.py +214 -0
- apsimo/cognition/prompt.py +150 -0
- apsimo/cognition/runtime.py +108 -0
- apsimo/cognition/trigger.py +154 -0
- apsimo/commitments/__init__.py +18 -0
- apsimo/commitments/local_work.py +355 -0
- apsimo/commitments/store.py +1052 -0
- apsimo/commitments/work.py +91 -0
- apsimo/compat.py +53 -0
- apsimo/compression/__init__.py +467 -0
- apsimo/connectors/__init__.py +21 -0
- apsimo/connectors/base.py +152 -0
- apsimo/connectors/caldav_calendar.py +125 -0
- apsimo/connectors/fs_documents.py +85 -0
- apsimo/connectors/imap_email.py +138 -0
- apsimo/connectors/manager.py +218 -0
- apsimo/connectors/webhook_pull.py +88 -0
- apsimo/contacts/__init__.py +33 -0
- apsimo/contacts/comms.py +357 -0
- apsimo/contacts/config.py +79 -0
- apsimo/contacts/exporters/__init__.py +1 -0
- apsimo/contacts/exporters/vcard.py +71 -0
- apsimo/contacts/identity_links.py +251 -0
- apsimo/contacts/importer.py +280 -0
- apsimo/contacts/importers/__init__.py +1 -0
- apsimo/contacts/importers/batch.py +43 -0
- apsimo/contacts/importers/macos_contacts.py +101 -0
- apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
- apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
- apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
- apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
- apsimo/contacts/migrations/005_identity_links.sql +27 -0
- apsimo/contacts/models.py +308 -0
- apsimo/contacts/scoring.py +16 -0
- apsimo/contacts/store.py +1623 -0
- apsimo/contacts/transport_ingress.py +252 -0
- apsimo/contacts/world_bridge.py +314 -0
- apsimo/contextgate/__init__.py +69 -0
- apsimo/contextgate/chunker.py +169 -0
- apsimo/contextgate/estimate.py +54 -0
- apsimo/contextgate/gate.py +313 -0
- apsimo/contextgate/retrieve.py +115 -0
- apsimo/delivery/__init__.py +16 -0
- apsimo/delivery/bridge.py +1260 -0
- apsimo/delivery/channels.py +526 -0
- apsimo/delivery/classification.py +50 -0
- apsimo/delivery/rate_limiter.py +268 -0
- apsimo/delivery/reachout_policy.py +206 -0
- apsimo/directed/__init__.py +22 -0
- apsimo/directed/audit.py +167 -0
- apsimo/directed/intake.py +95 -0
- apsimo/directed/models.py +191 -0
- apsimo/directed/service.py +509 -0
- apsimo/directives/__init__.py +25 -0
- apsimo/directives/evidence.py +87 -0
- apsimo/directives/extractor.py +188 -0
- apsimo/directives/guard.py +364 -0
- apsimo/directives/models.py +206 -0
- apsimo/directives/service.py +372 -0
- apsimo/directives/store.py +167 -0
- apsimo/doctor.py +2173 -0
- apsimo/environment.py +43 -0
- apsimo/events/__init__.py +33 -0
- apsimo/events/broadcaster.py +98 -0
- apsimo/events/bus.py +217 -0
- apsimo/events/journal.py +863 -0
- apsimo/events/stream.py +131 -0
- apsimo/events/types.py +150 -0
- apsimo/execution_results.py +357 -0
- apsimo/feedback/__init__.py +5 -0
- apsimo/feedback/store.py +76 -0
- apsimo/feeds/__init__.py +19 -0
- apsimo/feeds/cli.py +84 -0
- apsimo/feeds/engine.py +437 -0
- apsimo/feeds/example-feed.yaml +77 -0
- apsimo/feeds/hermes_cron.py +126 -0
- apsimo/feeds/manager.py +235 -0
- apsimo/feeds/spec.py +250 -0
- apsimo/feeds/template.py +202 -0
- apsimo/gate/__init__.py +18 -0
- apsimo/gate/audit.py +61 -0
- apsimo/gate/communication_policy.py +166 -0
- apsimo/gate/config.py +72 -0
- apsimo/gate/context_provenance.py +170 -0
- apsimo/gate/env_risk.py +226 -0
- apsimo/gate/guard_audit.py +353 -0
- apsimo/gate/layers/__init__.py +1 -0
- apsimo/gate/layers/base.py +15 -0
- apsimo/gate/layers/l1_recipient.py +66 -0
- apsimo/gate/layers/l2_pii.py +134 -0
- apsimo/gate/layers/l3_cross_context.py +50 -0
- apsimo/gate/layers/l4_trust_tier.py +78 -0
- apsimo/gate/layers/l5_injection.py +199 -0
- apsimo/gate/layers/l6_review.py +86 -0
- apsimo/gate/layers/l7_delay.py +100 -0
- apsimo/gate/layers/tom2_epistemic.py +185 -0
- apsimo/gate/models.py +64 -0
- apsimo/gate/pending_dispatch.py +5 -0
- apsimo/gate/pipeline.py +206 -0
- apsimo/gate/rejection.py +259 -0
- apsimo/gate/response_guard.py +700 -0
- apsimo/gate/rulesets/injection_v1.yaml +51 -0
- apsimo/gate/surface_policy.py +189 -0
- apsimo/gate/taint.py +226 -0
- apsimo/genesis.json +9 -0
- apsimo/goals/__init__.py +100 -0
- apsimo/goals/config.py +38 -0
- apsimo/goals/decomposer.py +421 -0
- apsimo/goals/engine.py +617 -0
- apsimo/goals/inference.py +354 -0
- apsimo/goals/models.py +302 -0
- apsimo/goals/priority.py +270 -0
- apsimo/goals/queue_bridge.py +149 -0
- apsimo/goals/replan.py +450 -0
- apsimo/goals/schema.sql +89 -0
- apsimo/goals/store.py +692 -0
- apsimo/governed_actions.py +1708 -0
- apsimo/harness_integration/__init__.py +45 -0
- apsimo/harness_integration/context.py +41 -0
- apsimo/harness_integration/skills.py +231 -0
- apsimo/identity/__init__.py +26 -0
- apsimo/identity/participants.py +181 -0
- apsimo/identity/resolver.py +329 -0
- apsimo/identity_bootstrap/__init__.py +5 -0
- apsimo/identity_bootstrap/builder.py +208 -0
- apsimo/identity_bootstrap/corpus.py +443 -0
- apsimo/identity_bootstrap/models.py +54 -0
- apsimo/identity_bootstrap/runner.py +353 -0
- apsimo/identity_bootstrap/seeders/__init__.py +25 -0
- apsimo/identity_bootstrap/seeders/briefings.py +109 -0
- apsimo/identity_bootstrap/seeders/chain.py +57 -0
- apsimo/identity_bootstrap/seeders/goals.py +128 -0
- apsimo/identity_bootstrap/seeders/memory.py +191 -0
- apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
- apsimo/identity_bootstrap/seeders/relationship.py +152 -0
- apsimo/identity_bootstrap/seeders/sessions.py +67 -0
- apsimo/identity_bootstrap/seeders/skills.py +92 -0
- apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
- apsimo/identity_bootstrap/seeders/world_model.py +143 -0
- apsimo/identity_bootstrap/self_query.py +92 -0
- apsimo/identity_bootstrap/self_reflection.py +155 -0
- apsimo/identity_bootstrap/skill.py +37 -0
- apsimo/identity_bootstrap/verifier.py +436 -0
- apsimo/initiatives/__init__.py +20 -0
- apsimo/initiatives/action_registry.py +454 -0
- apsimo/initiatives/approval_authority.py +2105 -0
- apsimo/initiatives/approval_policy.py +123 -0
- apsimo/initiatives/assignment.py +263 -0
- apsimo/initiatives/backup_evidence.py +100 -0
- apsimo/initiatives/context_freshness.py +103 -0
- apsimo/initiatives/models.py +318 -0
- apsimo/initiatives/native_work.py +270 -0
- apsimo/initiatives/standing_approvals.py +232 -0
- apsimo/initiatives/store.py +1081 -0
- apsimo/initiatives/temporal_followup.py +410 -0
- apsimo/intelligence/__init__.py +1 -0
- apsimo/intelligence/cognition/__init__.py +24 -0
- apsimo/intelligence/cognition/gap_detector.py +148 -0
- apsimo/intelligence/cognition/metalearner.py +547 -0
- apsimo/intelligence/cognition/metrics_collector.py +217 -0
- apsimo/intelligence/cognition/performance_index.py +299 -0
- apsimo/intelligence/cognition/registry.py +192 -0
- apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
- apsimo/intelligence/cognition/types.py +16 -0
- apsimo/intelligence/components/__init__.py +66 -0
- apsimo/intelligence/components/anomaly_detector.py +413 -0
- apsimo/intelligence/components/initiative_engine.py +2643 -0
- apsimo/intelligence/components/preference_learner.py +521 -0
- apsimo/intelligence/components/research_orchestrator.py +358 -0
- apsimo/intelligence/components/self_directed_thinker.py +221 -0
- apsimo/intelligence/components/self_reflector.py +252 -0
- apsimo/intelligence/components/session_continuity.py +154 -0
- apsimo/intelligence/components/task_planner.py +320 -0
- apsimo/intelligence/components/tool_learner.py +217 -0
- apsimo/intelligence/graph/__init__.py +79 -0
- apsimo/intelligence/graph/client.py +2483 -0
- apsimo/intelligence/graph/consolidator.py +405 -0
- apsimo/intelligence/graph/distiller.py +312 -0
- apsimo/intelligence/graph/migrations.py +129 -0
- apsimo/intelligence/graph/queries.py +248 -0
- apsimo/intelligence/graph/recall.py +281 -0
- apsimo/intelligence/graph/reconciler.py +144 -0
- apsimo/intelligence/graph/schema.py +337 -0
- apsimo/intelligence/graph/selection.py +252 -0
- apsimo/intelligence/learning/__init__.py +17 -0
- apsimo/intelligence/learning/continuous_learner.py +245 -0
- apsimo/intelligence/learning/feedback_store.py +321 -0
- apsimo/intelligence/mind_model/__init__.py +1 -0
- apsimo/intelligence/mind_model/graph_baseline.py +136 -0
- apsimo/intelligence/mind_model/signal_collector.py +361 -0
- apsimo/intelligence/relationships/__init__.py +11 -0
- apsimo/intelligence/relationships/profiler.py +389 -0
- apsimo/intelligence/relationships/scorer.py +560 -0
- apsimo/intelligence/relationships/signal_floor.py +66 -0
- apsimo/intelligence/relationships/trust_tiers.py +300 -0
- apsimo/intelligence/synthesis/__init__.py +40 -0
- apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
- apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
- apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
- apsimo/intelligence/synthesis/insight_store.py +79 -0
- apsimo/intelligence/synthesis/insight_validator.py +183 -0
- apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
- apsimo/intelligence/turn_middleware/__init__.py +15 -0
- apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
- apsimo/mcp/__init__.py +41 -0
- apsimo/mcp/__main__.py +6 -0
- apsimo/mcp/config.py +287 -0
- apsimo/mcp/server.py +501 -0
- apsimo/migrations.py +187 -0
- apsimo/mining/__init__.py +27 -0
- apsimo/mining/corpus.py +239 -0
- apsimo/mining/escalations.py +289 -0
- apsimo/mining/models.py +169 -0
- apsimo/mining/store.py +210 -0
- apsimo/models/__init__.py +30 -0
- apsimo/models/memory.py +80 -0
- apsimo/models/mesh.py +72 -0
- apsimo/models/person.py +104 -0
- apsimo/models/signal.py +108 -0
- apsimo/observations/__init__.py +15 -0
- apsimo/observations/store.py +277 -0
- apsimo/patterns/__init__.py +6 -0
- apsimo/patterns/extract.py +187 -0
- apsimo/patterns/store.py +227 -0
- apsimo/persona/__init__.py +1 -0
- apsimo/persona/engine.py +611 -0
- apsimo/persona/manifest.py +140 -0
- apsimo/projects/__init__.py +28 -0
- apsimo/projects/engine.py +1681 -0
- apsimo/projects/event_outbox.py +188 -0
- apsimo/projects/models.py +216 -0
- apsimo/projects/planner.py +181 -0
- apsimo/projects/store.py +1446 -0
- apsimo/proposals/__init__.py +12 -0
- apsimo/proposals/engine.py +114 -0
- apsimo/proposals/models.py +207 -0
- apsimo/qualification/__init__.py +1 -0
- apsimo/qualification/cases.py +75 -0
- apsimo/qualification/cli.py +51 -0
- apsimo/qualification/memory_cases.py +209 -0
- apsimo/qualification/records.py +92 -0
- apsimo/qualification/report.py +87 -0
- apsimo/qualification/runner.py +311 -0
- apsimo/qualification/structured_cases.py +131 -0
- apsimo/reasoning/__init__.py +13 -0
- apsimo/reasoning/executor.py +506 -0
- apsimo/reasoning/loop.py +373 -0
- apsimo/reasoning/native_tools/__init__.py +16 -0
- apsimo/reasoning/native_tools/calculate.py +141 -0
- apsimo/reasoning/native_tools/file_ops.py +150 -0
- apsimo/reasoning/native_tools/web_search.py +49 -0
- apsimo/reasoning/tool_policy.py +182 -0
- apsimo/redact/__init__.py +176 -0
- apsimo/repos/__init__.py +5 -0
- apsimo/repos/mirrors.py +204 -0
- apsimo/research/__init__.py +41 -0
- apsimo/research/artifact.py +482 -0
- apsimo/research/gatherer.py +387 -0
- apsimo/research/pipeline.py +513 -0
- apsimo/research/search/__init__.py +7 -0
- apsimo/research/search/base.py +41 -0
- apsimo/research/search/brave.py +59 -0
- apsimo/research/search/cache.py +51 -0
- apsimo/research/search/duckduckgo.py +103 -0
- apsimo/research/search/orchestrator.py +119 -0
- apsimo/research/search/serpapi.py +59 -0
- apsimo/research/search/tavily.py +59 -0
- apsimo/research/synthesizer.py +309 -0
- apsimo/router/__init__.py +30 -0
- apsimo/router/complexity_scorer.py +148 -0
- apsimo/router/endpoints.py +153 -0
- apsimo/router/fallback.py +58 -0
- apsimo/router/functions.py +243 -0
- apsimo/router/native_policy.py +52 -0
- apsimo/router/router.py +762 -0
- apsimo/router/self_learning.py +174 -0
- apsimo/router/tiers.py +677 -0
- apsimo/sandbox/__init__.py +21 -0
- apsimo/sandbox/backend.py +195 -0
- apsimo/sandbox/manager.py +173 -0
- apsimo/scope_bounds.py +7 -0
- apsimo/secrets/__init__.py +6 -0
- apsimo/secrets/backends/__init__.py +8 -0
- apsimo/secrets/backends/base.py +42 -0
- apsimo/secrets/backends/env.py +110 -0
- apsimo/secrets/backends/keyring.py +72 -0
- apsimo/secrets/backends/onepassword.py +232 -0
- apsimo/secrets/cli.py +191 -0
- apsimo/secrets/manager.py +160 -0
- apsimo/secrets/migration.py +101 -0
- apsimo/secrets/types.py +98 -0
- apsimo/seed.py +41 -0
- apsimo/self_model/__init__.py +37 -0
- apsimo/self_model/appraisals.py +673 -0
- apsimo/self_model/benchmark.py +1314 -0
- apsimo/self_model/brief.py +40 -0
- apsimo/self_model/event_concerns.py +1128 -0
- apsimo/self_model/execution_forecasts.py +353 -0
- apsimo/self_model/expectations.py +1595 -0
- apsimo/self_model/experiments.py +1150 -0
- apsimo/self_model/journal.py +148 -0
- apsimo/self_model/judgments.py +705 -0
- apsimo/self_model/native_outcomes.py +55 -0
- apsimo/self_model/params.py +220 -0
- apsimo/self_model/perspective.py +246 -0
- apsimo/self_model/reconcile.py +183 -0
- apsimo/self_model/reply_forecasts.py +381 -0
- apsimo/self_model/runtime_forecasts.py +296 -0
- apsimo/self_model/runtime_models.py +67 -0
- apsimo/self_model/settlement.py +207 -0
- apsimo/self_model/situation.py +1731 -0
- apsimo/self_model/store.py +883 -0
- apsimo/self_model/supervised.py +137 -0
- apsimo/self_model/thinker.py +99 -0
- apsimo/self_model/trust.py +388 -0
- apsimo/self_model/workspace.py +2388 -0
- apsimo/server.py +4197 -0
- apsimo/services/__init__.py +1 -0
- apsimo/services/agent_bridge.py +474 -0
- apsimo/services/initiative_executor.py +914 -0
- apsimo/services/instance.py +297 -0
- apsimo/sessions/__init__.py +22 -0
- apsimo/sessions/config.py +13 -0
- apsimo/sessions/context_loader.py +88 -0
- apsimo/sessions/federation_session.py +75 -0
- apsimo/sessions/isolated_session.py +98 -0
- apsimo/sessions/reports.py +84 -0
- apsimo/sessions/store.py +148 -0
- apsimo/setup.py +2818 -0
- apsimo/setup_hermes.py +879 -0
- apsimo/setup_local_work.py +218 -0
- apsimo/setup_native_goals.py +134 -0
- apsimo/setup_native_reviews.py +115 -0
- apsimo/skills/__init__.py +10 -0
- apsimo/skills/base.py +108 -0
- apsimo/skills/budget.py +28 -0
- apsimo/skills/executor.py +493 -0
- apsimo/skills/executors/__init__.py +1 -0
- apsimo/skills/executors/behavioral_correction.py +75 -0
- apsimo/skills/executors/capability_gap.py +38 -0
- apsimo/skills/executors/data_quality.py +163 -0
- apsimo/skills/executors/knowledge_acquisition.py +41 -0
- apsimo/skills/executors/operational_hygiene.py +185 -0
- apsimo/skills/executors/subsystem_health.py +169 -0
- apsimo/skills/hermes_export.py +431 -0
- apsimo/skills/index.py +123 -0
- apsimo/skills/learning/__init__.py +21 -0
- apsimo/skills/learning/novelty_detector.py +206 -0
- apsimo/skills/learning/pattern_extractor.py +199 -0
- apsimo/skills/learning/triggers.py +159 -0
- apsimo/skills/loader.py +246 -0
- apsimo/skills/migrations/002_progressive_loading.sql +6 -0
- apsimo/skills/migrations/backfill_triggers.py +20 -0
- apsimo/skills/models.py +202 -0
- apsimo/skills/packager.py +128 -0
- apsimo/skills/protocols.py +70 -0
- apsimo/skills/registry.py +191 -0
- apsimo/skills/runtime.py +58 -0
- apsimo/skills/sandbox_runner.py +229 -0
- apsimo/skills/scheduler.py +129 -0
- apsimo/skills/schema.py +79 -0
- apsimo/skills/security/__init__.py +12 -0
- apsimo/skills/security/guards.py +53 -0
- apsimo/skills/security/scanner.py +223 -0
- apsimo/skills_memory/__init__.py +26 -0
- apsimo/skills_memory/distill.py +159 -0
- apsimo/skills_memory/models.py +85 -0
- apsimo/skills_memory/retrieve.py +62 -0
- apsimo/skills_memory/store.py +172 -0
- apsimo/surprise/__init__.py +6 -0
- apsimo/surprise/accumulation.py +57 -0
- apsimo/surprise/scorer.py +102 -0
- apsimo/surprise/store.py +203 -0
- apsimo/task_queue/__init__.py +69 -0
- apsimo/task_queue/action_receipts.py +148 -0
- apsimo/task_queue/approval_relay_canary.py +108 -0
- apsimo/task_queue/config.py +85 -0
- apsimo/task_queue/contract.py +361 -0
- apsimo/task_queue/events.py +130 -0
- apsimo/task_queue/governor.py +1031 -0
- apsimo/task_queue/handlers/__init__.py +16 -0
- apsimo/task_queue/handlers/base.py +37 -0
- apsimo/task_queue/handlers/inference.py +640 -0
- apsimo/task_queue/handlers/monitoring.py +116 -0
- apsimo/task_queue/handlers/registry.py +75 -0
- apsimo/task_queue/handlers/subtask_handler.py +173 -0
- apsimo/task_queue/handlers/system_maintenance.py +147 -0
- apsimo/task_queue/mesh_integration.py +111 -0
- apsimo/task_queue/models.py +317 -0
- apsimo/task_queue/queue_manager.py +8286 -0
- apsimo/task_queue/routing.py +287 -0
- apsimo/task_queue/scheduler.py +252 -0
- apsimo/task_queue/schema.sql +197 -0
- apsimo/task_queue/work_control.py +342 -0
- apsimo/task_queue/worker.py +993 -0
- apsimo/telemetry.py +145 -0
- apsimo/tom/__init__.py +6 -0
- apsimo/tom/affect.py +387 -0
- apsimo/tom/approvals.py +171 -0
- apsimo/tom/arcs.py +896 -0
- apsimo/tom/asymmetry.py +131 -0
- apsimo/tom/eligibility.py +248 -0
- apsimo/tom/engagement.py +214 -0
- apsimo/tom/exposure.py +214 -0
- apsimo/tom/extractor.py +306 -0
- apsimo/tom/fact_adapters.py +144 -0
- apsimo/tom/facts.py +326 -0
- apsimo/tom/integration.py +592 -0
- apsimo/tom/leveled.py +118 -0
- apsimo/tom/levels.py +247 -0
- apsimo/tom/recipient_audit.py +995 -0
- apsimo/tom/recipient_simulator.py +593 -0
- apsimo/tom/source_lineage.py +93 -0
- apsimo/tom/tom2.py +277 -0
- apsimo/tom/visibility.py +559 -0
- apsimo/tom/visibility_store.py +414 -0
- apsimo/tools/__init__.py +0 -0
- apsimo/tools/definitions.py +740 -0
- apsimo/tools/handlers.py +943 -0
- apsimo/toolsmith/__init__.py +26 -0
- apsimo/toolsmith/authority.py +166 -0
- apsimo/toolsmith/engine.py +559 -0
- apsimo/toolsmith/integrity.py +100 -0
- apsimo/toolsmith/miner.py +145 -0
- apsimo/toolsmith/policy.py +110 -0
- apsimo/toolsmith/registry.py +635 -0
- apsimo/turns/__init__.py +17 -0
- apsimo/turns/audio.py +134 -0
- apsimo/turns/documents.py +235 -0
- apsimo/turns/executions.py +486 -0
- apsimo/turns/hermes_history.py +245 -0
- apsimo/turns/hermes_kanban.py +268 -0
- apsimo/turns/hermes_work.py +96 -0
- apsimo/turns/idempotency.py +752 -0
- apsimo/turns/local_work.py +115 -0
- apsimo/turns/media.py +581 -0
- apsimo/turns/reported_workers.py +196 -0
- apsimo/turns/source_annotations.py +283 -0
- apsimo/turns/source_attribution.py +154 -0
- apsimo/turns/source_read.py +351 -0
- apsimo/turns/source_vectors.py +263 -0
- apsimo/turns/video.py +210 -0
- apsimo/util/autonomy_preset.py +220 -0
- apsimo/util/instance.py +92 -0
- apsimo/util/model_output.py +25 -0
- apsimo/util/quiet_hours.py +27 -0
- apsimo/util/session_safety.py +37 -0
- apsimo/util/temporal.py +343 -0
- apsimo/vector/__init__.py +75 -0
- apsimo/vector/backfill.py +171 -0
- apsimo/vector/caption.py +114 -0
- apsimo/vector/collections.py +51 -0
- apsimo/vector/config.py +102 -0
- apsimo/vector/embedder.py +670 -0
- apsimo/vector/image_preprocess.py +406 -0
- apsimo/vector/image_store.py +296 -0
- apsimo/vector/indexes.py +162 -0
- apsimo/vector/migrate.py +334 -0
- apsimo/vector/multimodal_provider.py +417 -0
- apsimo/vector/multimodal_types.py +87 -0
- apsimo/vector/openai_provider.py +119 -0
- apsimo/vector/query.py +49 -0
- apsimo/vector/reranker.py +565 -0
- apsimo/vector/safety_image.py +159 -0
- apsimo/vector/scanner.py +197 -0
- apsimo/vector/setup.py +289 -0
- apsimo/vector/store.py +533 -0
- apsimo/vector/tiers.py +263 -0
- apsimo/work_orders.py +925 -0
- apsimo/workers/__init__.py +21 -0
- apsimo/workers/agent_bridge.py +640 -0
- apsimo/workers/colony_worker.py +382 -0
- apsimo/workers/queue_worker.py +441 -0
- apsimo/workers/skills_sync.py +152 -0
- apsimo/world_model/__init__.py +71 -0
- apsimo/world_model/causal_maintenance.py +131 -0
- apsimo/world_model/causal_policy.py +43 -0
- apsimo/world_model/causal_query.py +125 -0
- apsimo/world_model/confidence.py +54 -0
- apsimo/world_model/config.py +64 -0
- apsimo/world_model/constants.py +97 -0
- apsimo/world_model/entities.py +145 -0
- apsimo/world_model/expectation_resolvers.py +177 -0
- apsimo/world_model/extraction/__init__.py +7 -0
- apsimo/world_model/extraction/base.py +62 -0
- apsimo/world_model/extraction/conversation_extractor.py +262 -0
- apsimo/world_model/extraction/detector.py +74 -0
- apsimo/world_model/extraction/document_extractor.py +78 -0
- apsimo/world_model/extraction/formats/__init__.py +24 -0
- apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
- apsimo/world_model/extraction/formats/html_fmt.py +72 -0
- apsimo/world_model/extraction/formats/json_fmt.py +68 -0
- apsimo/world_model/extraction/formats/pdf.py +43 -0
- apsimo/world_model/extraction/formats/text.py +27 -0
- apsimo/world_model/extraction/llm_extractor.py +164 -0
- apsimo/world_model/extraction/pipeline.py +73 -0
- apsimo/world_model/integrations/__init__.py +5 -0
- apsimo/world_model/integrations/mind_model_bridge.py +115 -0
- apsimo/world_model/integrations/social_intel_bridge.py +120 -0
- apsimo/world_model/jobs/__init__.py +4 -0
- apsimo/world_model/jobs/extraction_job.py +168 -0
- apsimo/world_model/llm_extract.py +572 -0
- apsimo/world_model/neo4j/__init__.py +5 -0
- apsimo/world_model/neo4j/backend.py +654 -0
- apsimo/world_model/observations.py +155 -0
- apsimo/world_model/populator.py +307 -0
- apsimo/world_model/postgres/__init__.py +1 -0
- apsimo/world_model/postgres/backend.py +683 -0
- apsimo/world_model/relationships.py +25 -0
- apsimo/world_model/resolution/__init__.py +13 -0
- apsimo/world_model/resolution/entity_resolver.py +232 -0
- apsimo/world_model/resolution/merge_audit.py +16 -0
- apsimo/world_model/resolution/merge_workflow.py +117 -0
- apsimo/world_model/source_reports.py +121 -0
- apsimo/world_model/sqlite/__init__.py +4 -0
- apsimo/world_model/sqlite/backend.py +855 -0
- apsimo/world_model/sqlite/schema.sql +132 -0
- apsimo/world_model/store.py +545 -0
- apsimo-1.3.0.dist-info/METADATA +78 -0
- apsimo-1.3.0.dist-info/RECORD +614 -0
- apsimo-1.3.0.dist-info/WHEEL +5 -0
- apsimo-1.3.0.dist-info/entry_points.txt +11 -0
- apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
- apsimo-1.3.0.dist-info/top_level.txt +2 -0
- colony_sidecar/__init__.py +4 -0
|
@@ -0,0 +1,1314 @@
|
|
|
1
|
+
"""Selfhood benchmark: falsifiable self-improvement metrics (Mind M0a).
|
|
2
|
+
|
|
3
|
+
Derives a weekly scorecard entirely from journals and stores that already
|
|
4
|
+
exist. Nothing is self-reported by the LLM; every metric is computed from
|
|
5
|
+
recorded outcomes, and a metric whose source is unavailable is SKIPPED
|
|
6
|
+
rather than defaulted (the same fail-unknown discipline the doctor uses).
|
|
7
|
+
|
|
8
|
+
Metrics (stable ids):
|
|
9
|
+
commitments.fulfillment fulfilled / (fulfilled + open-overdue) in window
|
|
10
|
+
initiative.acceptance owner responded within 24h of a delivery success
|
|
11
|
+
delivery.success delivery-domain outcome rate (competence events)
|
|
12
|
+
actions.success all-domain outcome rate, per-domain detail
|
|
13
|
+
journal.acted_share acted / (acted+asked+held+blocked) decision mix
|
|
14
|
+
recall.fact_coverage probe: high-confidence shared facts re-queried
|
|
15
|
+
against graph recall, token-coverage graded
|
|
16
|
+
latency.jobs_p50_secs completed queue-job durations (p50; p95 detail)
|
|
17
|
+
latency.* / surface.* host-submitted samples (POST .../samples) rolled
|
|
18
|
+
up automatically: latency.* -> p50 (+p95),
|
|
19
|
+
everything else -> mean
|
|
20
|
+
|
|
21
|
+
Storage: colony-benchmark.db (samples append-only + weekly rollups).
|
|
22
|
+
Weeks are ISO (%G-W%V), windows are Monday 00:00 UTC half-open.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import asyncio
|
|
28
|
+
from dataclasses import dataclass
|
|
29
|
+
import hashlib
|
|
30
|
+
import json
|
|
31
|
+
import logging
|
|
32
|
+
import math
|
|
33
|
+
import os
|
|
34
|
+
import random
|
|
35
|
+
import re
|
|
36
|
+
import sqlite3
|
|
37
|
+
import threading
|
|
38
|
+
import time
|
|
39
|
+
import uuid
|
|
40
|
+
from datetime import datetime, timedelta, timezone
|
|
41
|
+
from typing import Any, Dict, List, Optional, Tuple
|
|
42
|
+
|
|
43
|
+
logger = logging.getLogger(__name__)
|
|
44
|
+
|
|
45
|
+
_METRIC_RE = re.compile(r"^[a-z0-9_]+(\.[a-z0-9_]+)+$")
|
|
46
|
+
_DEFINITION_VERSION_RE = re.compile(r"^v[1-9][0-9]{0,5}$")
|
|
47
|
+
_P4_MODES = frozenset({"off", "shadow", "live"})
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def cognition_p4_mode() -> str:
|
|
51
|
+
"""Controlled-learning mode. New deployments are deliberately dark."""
|
|
52
|
+
|
|
53
|
+
value = os.environ.get("COLONY_COGNITION_P4_MODE", "off").strip().lower()
|
|
54
|
+
return value if value in _P4_MODES else "off"
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _canonical(value: Any) -> str:
|
|
58
|
+
return json.dumps(value, sort_keys=True, separators=(",", ":"),
|
|
59
|
+
ensure_ascii=True)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
@dataclass(frozen=True)
|
|
63
|
+
class MetricDefinition:
|
|
64
|
+
"""Immutable measurement contract for one benchmark metric version."""
|
|
65
|
+
|
|
66
|
+
metric: str
|
|
67
|
+
version: str
|
|
68
|
+
direction: str
|
|
69
|
+
unit: str
|
|
70
|
+
evidence_query: str
|
|
71
|
+
minimum_samples: int
|
|
72
|
+
description: str = ""
|
|
73
|
+
|
|
74
|
+
def normalized(self) -> Dict[str, Any]:
|
|
75
|
+
metric = (self.metric or "").strip().lower()
|
|
76
|
+
version = (self.version or "").strip().lower()
|
|
77
|
+
direction = (self.direction or "").strip().lower()
|
|
78
|
+
unit = (self.unit or "").strip().lower()
|
|
79
|
+
evidence_query = (self.evidence_query or "").strip()
|
|
80
|
+
description = (self.description or "").strip()
|
|
81
|
+
if not _METRIC_RE.fullmatch(metric):
|
|
82
|
+
raise ValueError("metric definition has an invalid metric id")
|
|
83
|
+
if not _DEFINITION_VERSION_RE.fullmatch(version):
|
|
84
|
+
raise ValueError("metric definition version must look like v1")
|
|
85
|
+
if direction not in {"higher", "lower"}:
|
|
86
|
+
raise ValueError("metric direction must be higher or lower")
|
|
87
|
+
if not unit or len(unit) > 64:
|
|
88
|
+
raise ValueError("metric unit is required")
|
|
89
|
+
if not evidence_query or len(evidence_query) > 2000:
|
|
90
|
+
raise ValueError("metric evidence_query is required")
|
|
91
|
+
minimum = int(self.minimum_samples)
|
|
92
|
+
if minimum < 1 or minimum > 1_000_000:
|
|
93
|
+
raise ValueError("metric minimum_samples is out of bounds")
|
|
94
|
+
return {
|
|
95
|
+
"metric": metric,
|
|
96
|
+
"version": version,
|
|
97
|
+
"direction": direction,
|
|
98
|
+
"unit": unit,
|
|
99
|
+
"evidence_query": evidence_query,
|
|
100
|
+
"minimum_samples": minimum,
|
|
101
|
+
"description": description[:1000],
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
BUILTIN_METRIC_DEFINITIONS = (
|
|
106
|
+
MetricDefinition(
|
|
107
|
+
"commitments.fulfillment", "v2", "higher", "ratio",
|
|
108
|
+
"commitment.due_at in cohort_week AND terminal evidence is recorded",
|
|
109
|
+
5, "On-time fulfillment for the commitments due in one coherent cohort."),
|
|
110
|
+
MetricDefinition(
|
|
111
|
+
"delivery.success", "v1", "higher", "ratio",
|
|
112
|
+
"competence.domain=delivery AND evidence_status=verified", 5,
|
|
113
|
+
"Receipt-backed transport success."),
|
|
114
|
+
MetricDefinition(
|
|
115
|
+
"actions.success", "v2", "higher", "ratio",
|
|
116
|
+
"competence evidence is available and outcome is non-neutral", 10,
|
|
117
|
+
"Verified, versioned action outcomes."),
|
|
118
|
+
MetricDefinition(
|
|
119
|
+
"journal.acted_share", "v1", "higher", "ratio",
|
|
120
|
+
"action_journal.decision in (acted,asked,held,blocked)", 5,
|
|
121
|
+
"Decision mix; diagnostic rather than a success claim."),
|
|
122
|
+
MetricDefinition(
|
|
123
|
+
"initiative.acceptance", "v2", "higher", "ratio",
|
|
124
|
+
"owner reaction explicitly names the delivered initiative or message", 5,
|
|
125
|
+
"Message-bound owner acceptance; unrelated inbound turns never count."),
|
|
126
|
+
MetricDefinition(
|
|
127
|
+
"responses.correction_rate", "v1", "lower", "ratio",
|
|
128
|
+
"owner correction context_hash names a receipt-backed outbound response", 10,
|
|
129
|
+
"Owner-corrected responses divided by the same outbound cohort."),
|
|
130
|
+
MetricDefinition(
|
|
131
|
+
"recall.fact_coverage", "v2", "higher", "ratio",
|
|
132
|
+
"fact is viewer-allowed and recall uses the fact subject scope", 8,
|
|
133
|
+
"Viewer-scoped recall probe coverage."),
|
|
134
|
+
MetricDefinition(
|
|
135
|
+
"latency.jobs_p50_secs", "v1", "lower", "seconds",
|
|
136
|
+
"queue completion has a start and terminal timestamp", 5,
|
|
137
|
+
"Median queue-job completion latency."),
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def benchmark_enabled() -> bool:
|
|
142
|
+
return os.environ.get(
|
|
143
|
+
"COLONY_BENCHMARK_ENABLED", "true").strip().lower() != "false"
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _now() -> float:
|
|
147
|
+
return time.time()
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def week_id(dt: Optional[datetime] = None) -> str:
|
|
151
|
+
dt = dt or datetime.now(timezone.utc)
|
|
152
|
+
return dt.strftime("%G-W%V")
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def week_window(week: str) -> Tuple[datetime, datetime]:
|
|
156
|
+
"""[Monday 00:00 UTC, next Monday) for an ISO week id like 2026-W27."""
|
|
157
|
+
year, wk = week.split("-W")
|
|
158
|
+
start = datetime.fromisocalendar(int(year), int(wk), 1).replace(
|
|
159
|
+
tzinfo=timezone.utc)
|
|
160
|
+
return start, start + timedelta(days=7)
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def previous_week(dt: Optional[datetime] = None) -> str:
|
|
164
|
+
dt = dt or datetime.now(timezone.utc)
|
|
165
|
+
return week_id(dt - timedelta(days=7))
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def _percentile(values: List[float], pct: float) -> float:
|
|
169
|
+
if not values:
|
|
170
|
+
return 0.0
|
|
171
|
+
vs = sorted(values)
|
|
172
|
+
k = max(0, min(len(vs) - 1, int(round((pct / 100.0) * (len(vs) - 1)))))
|
|
173
|
+
return vs[k]
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
class BenchmarkStore:
|
|
177
|
+
"""SQLite persistence: append-only samples + weekly rollups."""
|
|
178
|
+
|
|
179
|
+
def __init__(self, db_path: str) -> None:
|
|
180
|
+
self._conn = sqlite3.connect(db_path, check_same_thread=False)
|
|
181
|
+
self._conn.row_factory = sqlite3.Row
|
|
182
|
+
self._conn.execute("PRAGMA journal_mode=WAL")
|
|
183
|
+
self._lock = threading.Lock()
|
|
184
|
+
self._conn.executescript(
|
|
185
|
+
"""
|
|
186
|
+
CREATE TABLE IF NOT EXISTS benchmark_samples (
|
|
187
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
188
|
+
metric TEXT NOT NULL,
|
|
189
|
+
value REAL NOT NULL,
|
|
190
|
+
source TEXT NOT NULL,
|
|
191
|
+
ts REAL NOT NULL,
|
|
192
|
+
meta TEXT
|
|
193
|
+
);
|
|
194
|
+
CREATE INDEX IF NOT EXISTS idx_bench_metric_ts
|
|
195
|
+
ON benchmark_samples(metric, ts);
|
|
196
|
+
CREATE TABLE IF NOT EXISTS benchmark_rollups (
|
|
197
|
+
week TEXT NOT NULL,
|
|
198
|
+
metric TEXT NOT NULL,
|
|
199
|
+
value REAL,
|
|
200
|
+
numerator REAL,
|
|
201
|
+
denominator REAL,
|
|
202
|
+
detail TEXT,
|
|
203
|
+
computed_at REAL NOT NULL,
|
|
204
|
+
PRIMARY KEY (week, metric)
|
|
205
|
+
);
|
|
206
|
+
CREATE TABLE IF NOT EXISTS benchmark_metric_definitions (
|
|
207
|
+
metric TEXT NOT NULL,
|
|
208
|
+
version TEXT NOT NULL,
|
|
209
|
+
direction TEXT NOT NULL,
|
|
210
|
+
unit TEXT NOT NULL,
|
|
211
|
+
evidence_query TEXT NOT NULL,
|
|
212
|
+
minimum_samples INTEGER NOT NULL,
|
|
213
|
+
description TEXT,
|
|
214
|
+
definition_hash TEXT NOT NULL,
|
|
215
|
+
created_at REAL NOT NULL,
|
|
216
|
+
PRIMARY KEY (metric, version)
|
|
217
|
+
);
|
|
218
|
+
CREATE TRIGGER IF NOT EXISTS benchmark_definition_no_update
|
|
219
|
+
BEFORE UPDATE ON benchmark_metric_definitions
|
|
220
|
+
BEGIN
|
|
221
|
+
SELECT RAISE(ABORT, 'benchmark definition is immutable');
|
|
222
|
+
END;
|
|
223
|
+
CREATE TRIGGER IF NOT EXISTS benchmark_definition_no_delete
|
|
224
|
+
BEFORE DELETE ON benchmark_metric_definitions
|
|
225
|
+
BEGIN
|
|
226
|
+
SELECT RAISE(ABORT, 'benchmark definition is immutable');
|
|
227
|
+
END;
|
|
228
|
+
"""
|
|
229
|
+
)
|
|
230
|
+
self._additive_columns(
|
|
231
|
+
"benchmark_samples",
|
|
232
|
+
{
|
|
233
|
+
"sample_id": "TEXT",
|
|
234
|
+
"definition_version": "TEXT",
|
|
235
|
+
"sample_principal": "TEXT",
|
|
236
|
+
"source_ref": "TEXT",
|
|
237
|
+
"receipt_ref": "TEXT",
|
|
238
|
+
"evidence_status": "TEXT",
|
|
239
|
+
"exposure_id": "TEXT",
|
|
240
|
+
},
|
|
241
|
+
)
|
|
242
|
+
self._additive_columns(
|
|
243
|
+
"benchmark_rollups",
|
|
244
|
+
{
|
|
245
|
+
"definition_version": "TEXT",
|
|
246
|
+
"definition_hash": "TEXT",
|
|
247
|
+
"evidence_count": "INTEGER",
|
|
248
|
+
},
|
|
249
|
+
)
|
|
250
|
+
self._conn.execute(
|
|
251
|
+
"CREATE UNIQUE INDEX IF NOT EXISTS idx_bench_sample_id "
|
|
252
|
+
"ON benchmark_samples(sample_id) WHERE sample_id IS NOT NULL"
|
|
253
|
+
)
|
|
254
|
+
self._conn.commit()
|
|
255
|
+
for definition in BUILTIN_METRIC_DEFINITIONS:
|
|
256
|
+
self.register_definition(definition)
|
|
257
|
+
|
|
258
|
+
def _additive_columns(self, table: str,
|
|
259
|
+
columns: Dict[str, str]) -> None:
|
|
260
|
+
existing = {str(row[1]) for row in self._conn.execute(
|
|
261
|
+
f"PRAGMA table_info({table})").fetchall()}
|
|
262
|
+
for name, sql_type in columns.items():
|
|
263
|
+
if name not in existing:
|
|
264
|
+
self._conn.execute(
|
|
265
|
+
f"ALTER TABLE {table} ADD COLUMN {name} {sql_type}")
|
|
266
|
+
|
|
267
|
+
def register_definition(self, definition: MetricDefinition
|
|
268
|
+
) -> Dict[str, Any]:
|
|
269
|
+
"""Idempotently register an immutable metric definition."""
|
|
270
|
+
|
|
271
|
+
normalized = definition.normalized()
|
|
272
|
+
digest = hashlib.sha256(
|
|
273
|
+
_canonical(normalized).encode("utf-8")).hexdigest()
|
|
274
|
+
with self._lock:
|
|
275
|
+
current = self._conn.execute(
|
|
276
|
+
"SELECT * FROM benchmark_metric_definitions "
|
|
277
|
+
"WHERE metric=? AND version=?",
|
|
278
|
+
(normalized["metric"], normalized["version"]),
|
|
279
|
+
).fetchone()
|
|
280
|
+
if current is not None:
|
|
281
|
+
result = dict(current)
|
|
282
|
+
if result["definition_hash"] != digest:
|
|
283
|
+
raise ValueError(
|
|
284
|
+
"metric definition is immutable; publish a new version")
|
|
285
|
+
return result
|
|
286
|
+
self._conn.execute(
|
|
287
|
+
"INSERT INTO benchmark_metric_definitions "
|
|
288
|
+
"(metric,version,direction,unit,evidence_query,minimum_samples,"
|
|
289
|
+
"description,definition_hash,created_at) VALUES (?,?,?,?,?,?,?,?,?)",
|
|
290
|
+
(
|
|
291
|
+
normalized["metric"], normalized["version"],
|
|
292
|
+
normalized["direction"], normalized["unit"],
|
|
293
|
+
normalized["evidence_query"], normalized["minimum_samples"],
|
|
294
|
+
normalized["description"], digest, _now(),
|
|
295
|
+
),
|
|
296
|
+
)
|
|
297
|
+
self._conn.commit()
|
|
298
|
+
row = self._conn.execute(
|
|
299
|
+
"SELECT * FROM benchmark_metric_definitions "
|
|
300
|
+
"WHERE metric=? AND version=?",
|
|
301
|
+
(normalized["metric"], normalized["version"]),
|
|
302
|
+
).fetchone()
|
|
303
|
+
assert row is not None
|
|
304
|
+
return dict(row)
|
|
305
|
+
|
|
306
|
+
def definition(self, metric: str, version: str) -> Optional[Dict[str, Any]]:
|
|
307
|
+
with self._lock:
|
|
308
|
+
row = self._conn.execute(
|
|
309
|
+
"SELECT * FROM benchmark_metric_definitions "
|
|
310
|
+
"WHERE metric=? AND version=?",
|
|
311
|
+
((metric or "").strip().lower(),
|
|
312
|
+
(version or "").strip().lower()),
|
|
313
|
+
).fetchone()
|
|
314
|
+
return dict(row) if row is not None else None
|
|
315
|
+
|
|
316
|
+
def definitions(self) -> List[Dict[str, Any]]:
|
|
317
|
+
with self._lock:
|
|
318
|
+
rows = self._conn.execute(
|
|
319
|
+
"SELECT * FROM benchmark_metric_definitions "
|
|
320
|
+
"ORDER BY metric, version").fetchall()
|
|
321
|
+
return [dict(row) for row in rows]
|
|
322
|
+
|
|
323
|
+
def add_sample(self, metric: str, value: float, *, source: str = "host",
|
|
324
|
+
ts: Optional[float] = None,
|
|
325
|
+
meta: Optional[Dict[str, Any]] = None) -> bool:
|
|
326
|
+
"""Compatibility ingestion.
|
|
327
|
+
|
|
328
|
+
Legacy samples remain queryable but are explicitly unattested and are
|
|
329
|
+
never eligible for a P4 causal decision.
|
|
330
|
+
"""
|
|
331
|
+
metric = (metric or "").strip().lower()
|
|
332
|
+
if not _METRIC_RE.match(metric):
|
|
333
|
+
return False
|
|
334
|
+
try:
|
|
335
|
+
value = float(value)
|
|
336
|
+
except (TypeError, ValueError):
|
|
337
|
+
return False
|
|
338
|
+
if not math.isfinite(value):
|
|
339
|
+
return False
|
|
340
|
+
with self._lock:
|
|
341
|
+
self._conn.execute(
|
|
342
|
+
"INSERT INTO benchmark_samples "
|
|
343
|
+
"(metric,value,source,ts,meta,sample_id,definition_version,"
|
|
344
|
+
"sample_principal,source_ref,receipt_ref,evidence_status,exposure_id)"
|
|
345
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
346
|
+
(
|
|
347
|
+
metric, value, (source or "host")[:64],
|
|
348
|
+
ts if ts is not None else _now(),
|
|
349
|
+
json.dumps(meta) if meta else None,
|
|
350
|
+
f"legacy-{uuid.uuid4().hex}", "legacy.unversioned",
|
|
351
|
+
f"legacy:{(source or 'host')[:96]}", None, None,
|
|
352
|
+
"legacy_unverified", None,
|
|
353
|
+
))
|
|
354
|
+
self._conn.commit()
|
|
355
|
+
return True
|
|
356
|
+
|
|
357
|
+
def add_evidence_sample(
|
|
358
|
+
self,
|
|
359
|
+
metric: str,
|
|
360
|
+
value: float,
|
|
361
|
+
*,
|
|
362
|
+
definition_version: str,
|
|
363
|
+
sample_principal: str,
|
|
364
|
+
source_ref: str,
|
|
365
|
+
receipt_ref: Optional[str] = None,
|
|
366
|
+
sample_id: Optional[str] = None,
|
|
367
|
+
exposure_id: Optional[str] = None,
|
|
368
|
+
ts: Optional[float] = None,
|
|
369
|
+
meta: Optional[Dict[str, Any]] = None,
|
|
370
|
+
) -> bool:
|
|
371
|
+
"""Append one attested sample under a registered evidence contract.
|
|
372
|
+
|
|
373
|
+
A stable ``sample_id`` makes transport retries idempotent. Reusing it
|
|
374
|
+
with changed content is refused rather than silently replacing proof.
|
|
375
|
+
"""
|
|
376
|
+
|
|
377
|
+
normalized_metric = (metric or "").strip().lower()
|
|
378
|
+
version = (definition_version or "").strip().lower()
|
|
379
|
+
definition = self.definition(normalized_metric, version)
|
|
380
|
+
if definition is None:
|
|
381
|
+
raise ValueError("registered metric definition is required")
|
|
382
|
+
principal = (sample_principal or "").strip()
|
|
383
|
+
source = (source_ref or "").strip()
|
|
384
|
+
receipt = (receipt_ref or "").strip() or None
|
|
385
|
+
exposure = (exposure_id or "").strip() or None
|
|
386
|
+
if not principal or len(principal) > 192:
|
|
387
|
+
raise ValueError("sample_principal is required")
|
|
388
|
+
if not source or len(source) > 512:
|
|
389
|
+
raise ValueError("source_ref is required")
|
|
390
|
+
try:
|
|
391
|
+
number = float(value)
|
|
392
|
+
except (TypeError, ValueError) as exc:
|
|
393
|
+
raise ValueError("sample value must be numeric") from exc
|
|
394
|
+
if not math.isfinite(number):
|
|
395
|
+
raise ValueError("sample value must be finite")
|
|
396
|
+
sid = (sample_id or f"bms-{uuid.uuid4().hex}").strip()
|
|
397
|
+
if not sid or len(sid) > 192:
|
|
398
|
+
raise ValueError("sample_id is malformed")
|
|
399
|
+
stamp = float(ts) if ts is not None else _now()
|
|
400
|
+
payload = {
|
|
401
|
+
"metric": normalized_metric,
|
|
402
|
+
"value": number,
|
|
403
|
+
"source": "evidence",
|
|
404
|
+
"ts": stamp,
|
|
405
|
+
"meta": meta or None,
|
|
406
|
+
"sample_id": sid,
|
|
407
|
+
"definition_version": version,
|
|
408
|
+
"sample_principal": principal,
|
|
409
|
+
"source_ref": source,
|
|
410
|
+
"receipt_ref": receipt,
|
|
411
|
+
"evidence_status": "verified" if receipt else "observed",
|
|
412
|
+
"exposure_id": exposure,
|
|
413
|
+
}
|
|
414
|
+
with self._lock:
|
|
415
|
+
existing = self._conn.execute(
|
|
416
|
+
"SELECT * FROM benchmark_samples WHERE sample_id=?", (sid,)
|
|
417
|
+
).fetchone()
|
|
418
|
+
if existing is not None:
|
|
419
|
+
row = dict(existing)
|
|
420
|
+
comparable = {
|
|
421
|
+
key: row.get(key) for key in (
|
|
422
|
+
"metric", "value", "source", "sample_id",
|
|
423
|
+
"definition_version", "sample_principal", "source_ref",
|
|
424
|
+
"receipt_ref", "evidence_status", "exposure_id")
|
|
425
|
+
}
|
|
426
|
+
if comparable != {key: payload.get(key) for key in comparable}:
|
|
427
|
+
raise ValueError("sample_id replay changed immutable evidence")
|
|
428
|
+
try:
|
|
429
|
+
stored_meta = json.loads(row["meta"]) if row.get("meta") else None
|
|
430
|
+
except ValueError:
|
|
431
|
+
stored_meta = row.get("meta")
|
|
432
|
+
if _canonical(stored_meta) != _canonical(meta or None):
|
|
433
|
+
raise ValueError("sample_id replay changed immutable evidence")
|
|
434
|
+
if ts is not None and abs(float(row["ts"]) - stamp) > 1e-9:
|
|
435
|
+
raise ValueError("sample_id replay changed immutable evidence")
|
|
436
|
+
return True
|
|
437
|
+
self._conn.execute(
|
|
438
|
+
"INSERT INTO benchmark_samples "
|
|
439
|
+
"(metric,value,source,ts,meta,sample_id,definition_version,"
|
|
440
|
+
"sample_principal,source_ref,receipt_ref,evidence_status,exposure_id)"
|
|
441
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?,?,?)",
|
|
442
|
+
(
|
|
443
|
+
payload["metric"], payload["value"], payload["source"],
|
|
444
|
+
payload["ts"], json.dumps(meta) if meta else None,
|
|
445
|
+
payload["sample_id"], payload["definition_version"],
|
|
446
|
+
payload["sample_principal"], payload["source_ref"],
|
|
447
|
+
payload["receipt_ref"], payload["evidence_status"],
|
|
448
|
+
payload["exposure_id"],
|
|
449
|
+
),
|
|
450
|
+
)
|
|
451
|
+
self._conn.commit()
|
|
452
|
+
return True
|
|
453
|
+
|
|
454
|
+
def samples_in(self, since: float, until: float,
|
|
455
|
+
metric: Optional[str] = None) -> List[Dict[str, Any]]:
|
|
456
|
+
q = ("SELECT * FROM benchmark_samples"
|
|
457
|
+
" WHERE ts >= ? AND ts < ?")
|
|
458
|
+
params: List[Any] = [since, until]
|
|
459
|
+
if metric:
|
|
460
|
+
q += " AND metric = ?"
|
|
461
|
+
params.append(metric)
|
|
462
|
+
q += " ORDER BY ts ASC LIMIT 100000"
|
|
463
|
+
with self._lock:
|
|
464
|
+
rows = self._conn.execute(q, params).fetchall()
|
|
465
|
+
return [dict(r) for r in rows]
|
|
466
|
+
|
|
467
|
+
def evidence_samples_in(
|
|
468
|
+
self,
|
|
469
|
+
since: float,
|
|
470
|
+
until: float,
|
|
471
|
+
metric: Optional[str] = None,
|
|
472
|
+
*,
|
|
473
|
+
definition_version: Optional[str] = None,
|
|
474
|
+
exposure_id: Optional[str] = None,
|
|
475
|
+
require_receipt: bool = False,
|
|
476
|
+
) -> List[Dict[str, Any]]:
|
|
477
|
+
q = ("SELECT * FROM benchmark_samples WHERE ts>=? AND ts<? "
|
|
478
|
+
"AND evidence_status IN ('observed','verified')")
|
|
479
|
+
params: List[Any] = [float(since), float(until)]
|
|
480
|
+
if metric:
|
|
481
|
+
q += " AND metric=?"
|
|
482
|
+
params.append((metric or "").strip().lower())
|
|
483
|
+
if definition_version:
|
|
484
|
+
q += " AND definition_version=?"
|
|
485
|
+
params.append((definition_version or "").strip().lower())
|
|
486
|
+
if exposure_id:
|
|
487
|
+
q += " AND exposure_id=?"
|
|
488
|
+
params.append(exposure_id)
|
|
489
|
+
if require_receipt:
|
|
490
|
+
q += " AND receipt_ref IS NOT NULL AND receipt_ref!=''"
|
|
491
|
+
q += " ORDER BY ts,id LIMIT 100000"
|
|
492
|
+
with self._lock:
|
|
493
|
+
rows = self._conn.execute(q, params).fetchall()
|
|
494
|
+
return [dict(row) for row in rows]
|
|
495
|
+
|
|
496
|
+
def write_rollup(self, week: str, metric: str, value: Optional[float], *,
|
|
497
|
+
numerator: Optional[float] = None,
|
|
498
|
+
denominator: Optional[float] = None,
|
|
499
|
+
detail: Optional[Dict[str, Any]] = None,
|
|
500
|
+
definition_version: Optional[str] = None,
|
|
501
|
+
evidence_count: Optional[int] = None) -> None:
|
|
502
|
+
definition_hash = None
|
|
503
|
+
if definition_version:
|
|
504
|
+
definition = self.definition(metric, definition_version)
|
|
505
|
+
if definition is None:
|
|
506
|
+
raise ValueError("registered metric definition is required")
|
|
507
|
+
definition_hash = definition["definition_hash"]
|
|
508
|
+
with self._lock:
|
|
509
|
+
self._conn.execute(
|
|
510
|
+
"INSERT OR REPLACE INTO benchmark_rollups"
|
|
511
|
+
" (week, metric, value, numerator, denominator, detail,"
|
|
512
|
+
" computed_at,definition_version,definition_hash,evidence_count)"
|
|
513
|
+
" VALUES (?,?,?,?,?,?,?,?,?,?)",
|
|
514
|
+
(week, metric, value, numerator, denominator,
|
|
515
|
+
json.dumps(detail) if detail else None, _now(),
|
|
516
|
+
definition_version, definition_hash, evidence_count))
|
|
517
|
+
self._conn.commit()
|
|
518
|
+
|
|
519
|
+
def rollups(self, weeks: int = 8) -> Dict[str, Dict[str, Any]]:
|
|
520
|
+
"""{week: {metric: {value, numerator, denominator, detail}}},
|
|
521
|
+
newest weeks first, at most `weeks` distinct weeks."""
|
|
522
|
+
with self._lock:
|
|
523
|
+
rows = self._conn.execute(
|
|
524
|
+
"SELECT * FROM benchmark_rollups ORDER BY week DESC"
|
|
525
|
+
).fetchall()
|
|
526
|
+
out: Dict[str, Dict[str, Any]] = {}
|
|
527
|
+
for r in rows:
|
|
528
|
+
wk = r["week"]
|
|
529
|
+
if wk not in out:
|
|
530
|
+
if len(out) >= weeks:
|
|
531
|
+
continue
|
|
532
|
+
out[wk] = {}
|
|
533
|
+
out[wk][r["metric"]] = {
|
|
534
|
+
"value": r["value"],
|
|
535
|
+
"numerator": r["numerator"],
|
|
536
|
+
"denominator": r["denominator"],
|
|
537
|
+
"detail": json.loads(r["detail"]) if r["detail"] else None,
|
|
538
|
+
"definition_version": r["definition_version"],
|
|
539
|
+
"definition_hash": r["definition_hash"],
|
|
540
|
+
"evidence_count": r["evidence_count"],
|
|
541
|
+
}
|
|
542
|
+
return out
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
class SelfhoodBenchmark:
|
|
546
|
+
"""Weekly metric derivation over the shipped stores.
|
|
547
|
+
|
|
548
|
+
Dependencies may be injected (tests) or resolved lazily from the host
|
|
549
|
+
module globals at compute time (production), so construction order in
|
|
550
|
+
the server lifespan does not matter.
|
|
551
|
+
"""
|
|
552
|
+
|
|
553
|
+
def __init__(self, store: BenchmarkStore, *,
|
|
554
|
+
commitments: Any = None, competence: Any = None,
|
|
555
|
+
journal: Any = None, comms: Any = None, graph: Any = None,
|
|
556
|
+
facts: Any = None, queue: Any = None,
|
|
557
|
+
corrections: Any = None,
|
|
558
|
+
owner_contact_id: Optional[str] = None,
|
|
559
|
+
probes: Optional[int] = None) -> None:
|
|
560
|
+
self.store = store
|
|
561
|
+
self._deps = {
|
|
562
|
+
"commitments": commitments, "competence": competence,
|
|
563
|
+
"journal": journal, "comms": comms, "graph": graph,
|
|
564
|
+
"facts": facts, "queue": queue, "corrections": corrections,
|
|
565
|
+
}
|
|
566
|
+
self._owner = owner_contact_id
|
|
567
|
+
self._probes = probes
|
|
568
|
+
|
|
569
|
+
# -- lazy dependency resolution -------------------------------------
|
|
570
|
+
_HOST_GLOBALS = {
|
|
571
|
+
"commitments": "_commitment_store", "comms": "_comms_log",
|
|
572
|
+
"graph": "_graph", "facts": "_facts_store", "queue": "_task_queue",
|
|
573
|
+
"corrections": "_learning_feedback_store",
|
|
574
|
+
}
|
|
575
|
+
|
|
576
|
+
def _dep(self, name: str) -> Any:
|
|
577
|
+
if self._deps.get(name) is not None:
|
|
578
|
+
return self._deps[name]
|
|
579
|
+
if name == "competence":
|
|
580
|
+
sm = self._host_attr("_self_model")
|
|
581
|
+
if sm is not None:
|
|
582
|
+
# SelfModel keeps its CompetenceStore as `.store`
|
|
583
|
+
return (getattr(sm, "store", None)
|
|
584
|
+
or getattr(sm, "competence", None))
|
|
585
|
+
return None
|
|
586
|
+
if name == "journal":
|
|
587
|
+
sm = self._host_attr("_self_model")
|
|
588
|
+
return getattr(sm, "journal", None) if sm is not None else None
|
|
589
|
+
g = self._HOST_GLOBALS.get(name)
|
|
590
|
+
return self._host_attr(g) if g else None
|
|
591
|
+
|
|
592
|
+
@staticmethod
|
|
593
|
+
def _host_attr(name: str) -> Any:
|
|
594
|
+
try:
|
|
595
|
+
from apsimo.api.routers import host
|
|
596
|
+
return getattr(host, name, None)
|
|
597
|
+
except Exception:
|
|
598
|
+
return None
|
|
599
|
+
|
|
600
|
+
@property
|
|
601
|
+
def owner_contact_id(self) -> str:
|
|
602
|
+
return (self._owner
|
|
603
|
+
or os.environ.get("COLONY_OWNER_CONTACT_ID", "").strip())
|
|
604
|
+
|
|
605
|
+
@property
|
|
606
|
+
def probe_count(self) -> int:
|
|
607
|
+
if self._probes is not None:
|
|
608
|
+
return self._probes
|
|
609
|
+
try:
|
|
610
|
+
return int(os.environ.get("COLONY_BENCHMARK_PROBES", "8"))
|
|
611
|
+
except ValueError:
|
|
612
|
+
return 8
|
|
613
|
+
|
|
614
|
+
# -- derivations ------------------------------------------------------
|
|
615
|
+
async def compute_week(self, week: Optional[str] = None) -> Dict[str, Any]:
|
|
616
|
+
"""Derive every computable metric for `week` (default: the previous
|
|
617
|
+
completed ISO week), persist rollups, and return them. Metrics whose
|
|
618
|
+
source is unavailable are omitted, never zero-filled."""
|
|
619
|
+
wk = week or previous_week()
|
|
620
|
+
start, end = week_window(wk)
|
|
621
|
+
since, until = start.timestamp(), end.timestamp()
|
|
622
|
+
out: Dict[str, Any] = {}
|
|
623
|
+
|
|
624
|
+
for name, fn in (
|
|
625
|
+
("commitments.fulfillment", self._m_commitments),
|
|
626
|
+
("delivery.success", self._m_delivery),
|
|
627
|
+
("actions.success", self._m_actions),
|
|
628
|
+
("journal.acted_share", self._m_journal),
|
|
629
|
+
("initiative.acceptance", self._m_acceptance),
|
|
630
|
+
("responses.correction_rate", self._m_corrections),
|
|
631
|
+
):
|
|
632
|
+
try:
|
|
633
|
+
res = fn(start, end, since, until)
|
|
634
|
+
if res is not None:
|
|
635
|
+
out[name] = res
|
|
636
|
+
except Exception as exc:
|
|
637
|
+
logger.warning("benchmark %s failed: %s", name, exc)
|
|
638
|
+
for name, coro in (
|
|
639
|
+
("recall.fact_coverage", self._m_recall(since, until)),
|
|
640
|
+
("latency.jobs_p50_secs", self._m_jobs(start, end)),
|
|
641
|
+
):
|
|
642
|
+
try:
|
|
643
|
+
res = await coro
|
|
644
|
+
if res is not None:
|
|
645
|
+
out[name] = res
|
|
646
|
+
except Exception as exc:
|
|
647
|
+
logger.warning("benchmark %s failed: %s", name, exc)
|
|
648
|
+
try:
|
|
649
|
+
out.update(self._m_calibration(since))
|
|
650
|
+
except Exception as exc:
|
|
651
|
+
logger.warning("benchmark calibration failed: %s", exc)
|
|
652
|
+
out.update(self._m_submitted(since, until, skip=set(out)))
|
|
653
|
+
|
|
654
|
+
for metric, r in out.items():
|
|
655
|
+
definition_version = self._definition_version(metric)
|
|
656
|
+
self.store.write_rollup(
|
|
657
|
+
wk, metric, r.get("value"), numerator=r.get("numerator"),
|
|
658
|
+
denominator=r.get("denominator"), detail=r.get("detail"),
|
|
659
|
+
definition_version=definition_version,
|
|
660
|
+
evidence_count=(int(r.get("denominator"))
|
|
661
|
+
if r.get("denominator") is not None else None))
|
|
662
|
+
logger.info("benchmark week %s: %d metrics", wk, len(out))
|
|
663
|
+
return {"week": wk, "metrics": out}
|
|
664
|
+
|
|
665
|
+
@staticmethod
|
|
666
|
+
def _definition_version(metric: str) -> Optional[str]:
|
|
667
|
+
if cognition_p4_mode() != "live":
|
|
668
|
+
# Existing rollups are intentionally left labelled as legacy
|
|
669
|
+
# until the controlled path is explicitly enabled.
|
|
670
|
+
return None
|
|
671
|
+
versions = {
|
|
672
|
+
"commitments.fulfillment": "v2",
|
|
673
|
+
"delivery.success": "v1",
|
|
674
|
+
"actions.success": "v2",
|
|
675
|
+
"journal.acted_share": "v1",
|
|
676
|
+
"initiative.acceptance": "v2",
|
|
677
|
+
"responses.correction_rate": "v1",
|
|
678
|
+
"recall.fact_coverage": "v2",
|
|
679
|
+
"latency.jobs_p50_secs": "v1",
|
|
680
|
+
}
|
|
681
|
+
return versions.get(metric)
|
|
682
|
+
|
|
683
|
+
def _m_commitments(self, start, end, since, until):
|
|
684
|
+
cs = self._dep("commitments")
|
|
685
|
+
if cs is None:
|
|
686
|
+
return None
|
|
687
|
+
if cognition_p4_mode() == "live":
|
|
688
|
+
try:
|
|
689
|
+
rows = cs.list(limit=10000).get("commitments", [])
|
|
690
|
+
except (AttributeError, TypeError):
|
|
691
|
+
return None
|
|
692
|
+
due_cohort: List[Dict[str, Any]] = []
|
|
693
|
+
for raw in rows:
|
|
694
|
+
row = raw if isinstance(raw, dict) else vars(raw)
|
|
695
|
+
due = self._parse_instant(row.get("due_at"))
|
|
696
|
+
if due is None or not (start <= due < end):
|
|
697
|
+
continue
|
|
698
|
+
if str(row.get("status") or "").lower() == "cancelled":
|
|
699
|
+
continue
|
|
700
|
+
due_cohort.append(row)
|
|
701
|
+
if not due_cohort:
|
|
702
|
+
return None
|
|
703
|
+
fulfilled = 0
|
|
704
|
+
late = 0
|
|
705
|
+
for row in due_cohort:
|
|
706
|
+
due = self._parse_instant(row.get("due_at"))
|
|
707
|
+
completed = self._parse_instant(row.get("fulfilled_at"))
|
|
708
|
+
if completed is not None and due is not None:
|
|
709
|
+
if completed <= due:
|
|
710
|
+
fulfilled += 1
|
|
711
|
+
else:
|
|
712
|
+
late += 1
|
|
713
|
+
return {
|
|
714
|
+
"value": fulfilled / len(due_cohort),
|
|
715
|
+
"numerator": fulfilled,
|
|
716
|
+
"denominator": len(due_cohort),
|
|
717
|
+
"detail": {
|
|
718
|
+
"metric_definition": "commitments.fulfillment/v2",
|
|
719
|
+
"cohort": "due_at_in_iso_week",
|
|
720
|
+
"late": late,
|
|
721
|
+
"open_or_missed": len(due_cohort) - fulfilled - late,
|
|
722
|
+
},
|
|
723
|
+
}
|
|
724
|
+
fulfilled = 0
|
|
725
|
+
for c in (cs.list(status=["fulfilled"], limit=500)
|
|
726
|
+
.get("commitments", [])):
|
|
727
|
+
fat = (c.get("fulfilled_at") or "") if isinstance(c, dict) else\
|
|
728
|
+
(getattr(c, "fulfilled_at", "") or "")
|
|
729
|
+
if fat and start.isoformat() <= str(fat) < end.isoformat():
|
|
730
|
+
fulfilled += 1
|
|
731
|
+
overdue_open = len(cs.get_overdue())
|
|
732
|
+
den = fulfilled + overdue_open
|
|
733
|
+
if den == 0:
|
|
734
|
+
return None
|
|
735
|
+
return {"value": fulfilled / den, "numerator": fulfilled,
|
|
736
|
+
"denominator": den, "detail": {"overdue_open": overdue_open}}
|
|
737
|
+
|
|
738
|
+
@staticmethod
|
|
739
|
+
def _parse_instant(value: Any) -> Optional[datetime]:
|
|
740
|
+
if not value:
|
|
741
|
+
return None
|
|
742
|
+
try:
|
|
743
|
+
parsed = datetime.fromisoformat(str(value).replace("Z", "+00:00"))
|
|
744
|
+
except (TypeError, ValueError):
|
|
745
|
+
return None
|
|
746
|
+
if parsed.tzinfo is None:
|
|
747
|
+
parsed = parsed.replace(tzinfo=timezone.utc)
|
|
748
|
+
return parsed.astimezone(timezone.utc)
|
|
749
|
+
|
|
750
|
+
def _events(self, domain: str, since: float):
|
|
751
|
+
comp = self._dep("competence")
|
|
752
|
+
if comp is None:
|
|
753
|
+
return None
|
|
754
|
+
return [e for e in comp.events(domain, since=since,
|
|
755
|
+
include_shadow=False)]
|
|
756
|
+
|
|
757
|
+
def _competence_state(self, domains: List[str], since: float,
|
|
758
|
+
until: float) -> Tuple[int, List[Dict[str, Any]]]:
|
|
759
|
+
"""Correction revision and unresolved provenance gaps for a slice."""
|
|
760
|
+
comp = self._dep("competence")
|
|
761
|
+
if comp is None:
|
|
762
|
+
return 0, []
|
|
763
|
+
revision = 0
|
|
764
|
+
gaps: List[Dict[str, Any]] = []
|
|
765
|
+
try:
|
|
766
|
+
revision = int(comp.reconciliation_revision(
|
|
767
|
+
domains=domains, since=since, until=until))
|
|
768
|
+
except (AttributeError, TypeError, ValueError):
|
|
769
|
+
pass
|
|
770
|
+
try:
|
|
771
|
+
for domain in domains:
|
|
772
|
+
gaps.extend(comp.active_evidence_gaps(
|
|
773
|
+
domain, since=since, until=until))
|
|
774
|
+
except (AttributeError, TypeError):
|
|
775
|
+
pass
|
|
776
|
+
return revision, gaps
|
|
777
|
+
|
|
778
|
+
@staticmethod
|
|
779
|
+
def _unavailable_competence(
|
|
780
|
+
revision: int, gaps: List[Dict[str, Any]]) -> Dict[str, Any]:
|
|
781
|
+
return {
|
|
782
|
+
"value": None, "numerator": None, "denominator": None,
|
|
783
|
+
"detail": {
|
|
784
|
+
"available": False,
|
|
785
|
+
"reason": "competence_evidence_gap",
|
|
786
|
+
"competence_reconciliation_revision": revision,
|
|
787
|
+
"gaps": [{
|
|
788
|
+
"id": g.get("reconciliation_id"),
|
|
789
|
+
"domain": g.get("domain"),
|
|
790
|
+
"since_ts": g.get("since_ts"),
|
|
791
|
+
"until_ts": g.get("until_ts"),
|
|
792
|
+
"reason": g.get("reason"),
|
|
793
|
+
} for g in gaps],
|
|
794
|
+
},
|
|
795
|
+
}
|
|
796
|
+
|
|
797
|
+
def _m_delivery(self, start, end, since, until):
|
|
798
|
+
revision, gaps = self._competence_state(
|
|
799
|
+
["delivery"], since, until)
|
|
800
|
+
if gaps:
|
|
801
|
+
return self._unavailable_competence(revision, gaps)
|
|
802
|
+
evs = self._events("delivery", since)
|
|
803
|
+
if evs is None:
|
|
804
|
+
return None
|
|
805
|
+
evs = [e for e in evs if e["ts"] < until]
|
|
806
|
+
if cognition_p4_mode() == "live":
|
|
807
|
+
evs = [e for e in evs if e.get("evidence_status") == "verified"]
|
|
808
|
+
if not evs:
|
|
809
|
+
return None
|
|
810
|
+
ok = sum(1 for e in evs if e["outcome"] == "success")
|
|
811
|
+
return {"value": ok / len(evs), "numerator": ok,
|
|
812
|
+
"denominator": len(evs),
|
|
813
|
+
"detail": {
|
|
814
|
+
"n": len(evs),
|
|
815
|
+
"metric_definition": "colony.delivery-success/v1",
|
|
816
|
+
"competence_reconciliation_revision": revision,
|
|
817
|
+
}}
|
|
818
|
+
|
|
819
|
+
def _m_actions(self, start, end, since, until):
|
|
820
|
+
comp = self._dep("competence")
|
|
821
|
+
if comp is None:
|
|
822
|
+
return None
|
|
823
|
+
domains = []
|
|
824
|
+
for row in comp.snapshot():
|
|
825
|
+
dom = row.get("domain") if isinstance(row, dict) else None
|
|
826
|
+
if dom and dom != "delivery":
|
|
827
|
+
domains.append(dom)
|
|
828
|
+
revision, gaps = self._competence_state(domains, since, until)
|
|
829
|
+
if gaps:
|
|
830
|
+
return self._unavailable_competence(revision, gaps)
|
|
831
|
+
per: Dict[str, Dict[str, int]] = {}
|
|
832
|
+
ok = n = 0
|
|
833
|
+
for dom in domains:
|
|
834
|
+
evs = [e for e in comp.events(dom, since=since,
|
|
835
|
+
include_shadow=False)
|
|
836
|
+
if e["ts"] < until]
|
|
837
|
+
if cognition_p4_mode() == "live":
|
|
838
|
+
evs = [e for e in evs
|
|
839
|
+
if e.get("evidence_status") == "verified"
|
|
840
|
+
and str(e.get("outcome_contract") or "").lower()
|
|
841
|
+
not in {"", "legacy.unversioned"}]
|
|
842
|
+
if not evs:
|
|
843
|
+
continue
|
|
844
|
+
d_ok = sum(1 for e in evs if e["outcome"] == "success")
|
|
845
|
+
per[dom] = {"success": d_ok, "n": len(evs)}
|
|
846
|
+
ok += d_ok
|
|
847
|
+
n += len(evs)
|
|
848
|
+
if n == 0:
|
|
849
|
+
return None
|
|
850
|
+
return {"value": ok / n, "numerator": ok, "denominator": n,
|
|
851
|
+
"detail": {
|
|
852
|
+
"domains": per,
|
|
853
|
+
"metric_definition": "colony.actions-success/v2",
|
|
854
|
+
"competence_reconciliation_revision": revision,
|
|
855
|
+
}}
|
|
856
|
+
|
|
857
|
+
def _m_journal(self, start, end, since, until):
|
|
858
|
+
j = self._dep("journal")
|
|
859
|
+
if j is None:
|
|
860
|
+
return None
|
|
861
|
+
entries = j.recent(limit=2000, since=since)
|
|
862
|
+
counts: Dict[str, int] = {}
|
|
863
|
+
for e in entries:
|
|
864
|
+
if e.get("ts", 0) >= until:
|
|
865
|
+
continue
|
|
866
|
+
d = e.get("decision") or "unknown"
|
|
867
|
+
counts[d] = counts.get(d, 0) + 1
|
|
868
|
+
gated = sum(counts.get(k, 0)
|
|
869
|
+
for k in ("acted", "asked", "held", "blocked"))
|
|
870
|
+
if gated == 0:
|
|
871
|
+
return None
|
|
872
|
+
return {"value": counts.get("acted", 0) / gated,
|
|
873
|
+
"numerator": counts.get("acted", 0), "denominator": gated,
|
|
874
|
+
"detail": {"decisions": counts}}
|
|
875
|
+
|
|
876
|
+
def _m_acceptance(self, start, end, since, until):
|
|
877
|
+
"""Owner responded (inbound comm) within 24h of a delivery success."""
|
|
878
|
+
owner = self.owner_contact_id
|
|
879
|
+
comms = self._dep("comms")
|
|
880
|
+
if not owner or comms is None:
|
|
881
|
+
return None
|
|
882
|
+
revision, gaps = self._competence_state(
|
|
883
|
+
["delivery"], since, until)
|
|
884
|
+
if gaps:
|
|
885
|
+
return self._unavailable_competence(revision, gaps)
|
|
886
|
+
evs = self._events("delivery", since)
|
|
887
|
+
if evs is None:
|
|
888
|
+
return None
|
|
889
|
+
deliveries = [e for e in evs
|
|
890
|
+
if e["outcome"] == "success" and e["ts"] < until]
|
|
891
|
+
if not deliveries:
|
|
892
|
+
return None
|
|
893
|
+
if cognition_p4_mode() == "live":
|
|
894
|
+
# An unrelated inbound message is not evidence that an initiative
|
|
895
|
+
# was useful. Only an explicit reaction naming the delivery or its
|
|
896
|
+
# source message enters this cohort.
|
|
897
|
+
refs: List[str] = []
|
|
898
|
+
for event in deliveries:
|
|
899
|
+
if event.get("evidence_status") != "verified":
|
|
900
|
+
continue
|
|
901
|
+
evidence = event.get("evidence") or {}
|
|
902
|
+
if isinstance(evidence, str):
|
|
903
|
+
try:
|
|
904
|
+
evidence = json.loads(evidence)
|
|
905
|
+
except ValueError:
|
|
906
|
+
evidence = {}
|
|
907
|
+
ref = (evidence.get("delivery_id")
|
|
908
|
+
if isinstance(evidence, dict) else None)
|
|
909
|
+
ref = ref or event.get("source_ref")
|
|
910
|
+
if ref:
|
|
911
|
+
refs.append(str(ref))
|
|
912
|
+
refs = list(dict.fromkeys(refs))
|
|
913
|
+
if not refs or not hasattr(comms, "reactions_for_refs"):
|
|
914
|
+
return None
|
|
915
|
+
reactions = comms.reactions_for_refs(
|
|
916
|
+
owner, refs, since_iso=start.isoformat(),
|
|
917
|
+
until_iso=end.isoformat())
|
|
918
|
+
by_ref = {str(row.get("reply_to_ref")): row
|
|
919
|
+
for row in reactions if row.get("reply_to_ref")}
|
|
920
|
+
accepted_names = {"accepted", "acknowledged", "actioned"}
|
|
921
|
+
negative_names = {"negative", "dismissed", "corrected", "rejected"}
|
|
922
|
+
accepted = sum(
|
|
923
|
+
1 for ref in refs
|
|
924
|
+
if str((by_ref.get(ref) or {}).get("reaction") or "").lower()
|
|
925
|
+
in accepted_names)
|
|
926
|
+
negative = sum(
|
|
927
|
+
1 for ref in refs
|
|
928
|
+
if str((by_ref.get(ref) or {}).get("reaction") or "").lower()
|
|
929
|
+
in negative_names)
|
|
930
|
+
return {
|
|
931
|
+
"value": accepted / len(refs),
|
|
932
|
+
"numerator": accepted,
|
|
933
|
+
"denominator": len(refs),
|
|
934
|
+
"detail": {
|
|
935
|
+
"deliveries": len(refs),
|
|
936
|
+
"negative": negative,
|
|
937
|
+
"unanswered": len(refs) - len(by_ref),
|
|
938
|
+
"metric_definition": "initiative.acceptance/v2",
|
|
939
|
+
"binding": "reply_to_ref",
|
|
940
|
+
"competence_reconciliation_revision": revision,
|
|
941
|
+
},
|
|
942
|
+
}
|
|
943
|
+
inbound = comms.inbound_since(owner, start.isoformat())
|
|
944
|
+
in_ts = []
|
|
945
|
+
for t in inbound:
|
|
946
|
+
try:
|
|
947
|
+
in_ts.append(datetime.fromisoformat(
|
|
948
|
+
str(t).replace("Z", "+00:00")).timestamp())
|
|
949
|
+
except ValueError:
|
|
950
|
+
continue
|
|
951
|
+
accepted = sum(
|
|
952
|
+
1 for e in deliveries
|
|
953
|
+
if any(e["ts"] < t <= e["ts"] + 86400 for t in in_ts))
|
|
954
|
+
return {"value": accepted / len(deliveries), "numerator": accepted,
|
|
955
|
+
"denominator": len(deliveries),
|
|
956
|
+
"detail": {
|
|
957
|
+
"deliveries": len(deliveries),
|
|
958
|
+
"metric_definition": "colony.initiative-acceptance/v1",
|
|
959
|
+
"competence_reconciliation_revision": revision,
|
|
960
|
+
}}
|
|
961
|
+
|
|
962
|
+
def _m_corrections(self, start, end, since, until):
|
|
963
|
+
"""Corrections over the same receipt-backed outbound response cohort."""
|
|
964
|
+
|
|
965
|
+
if cognition_p4_mode() != "live":
|
|
966
|
+
return None
|
|
967
|
+
owner = self.owner_contact_id
|
|
968
|
+
comms = self._dep("comms")
|
|
969
|
+
corrections = self._dep("corrections")
|
|
970
|
+
if (not owner or comms is None or corrections is None
|
|
971
|
+
or not hasattr(comms, "outbound_between")
|
|
972
|
+
or not hasattr(corrections, "between")):
|
|
973
|
+
return None
|
|
974
|
+
outbound = comms.outbound_between(
|
|
975
|
+
owner, start.isoformat(), end.isoformat(), require_receipt=True)
|
|
976
|
+
cohort: Dict[str, Dict[str, Any]] = {}
|
|
977
|
+
for row in outbound:
|
|
978
|
+
ref = row.get("external_ref") or row.get("receipt_ref")
|
|
979
|
+
if ref:
|
|
980
|
+
cohort[str(ref)] = row
|
|
981
|
+
if not cohort:
|
|
982
|
+
return None
|
|
983
|
+
corrected: set[str] = set()
|
|
984
|
+
for item in corrections.between(
|
|
985
|
+
start.isoformat(), end.isoformat(), person_id=owner):
|
|
986
|
+
context_ref = (item.get("context_hash") if isinstance(item, dict)
|
|
987
|
+
else getattr(item, "context_hash", ""))
|
|
988
|
+
if context_ref in cohort:
|
|
989
|
+
corrected.add(str(context_ref))
|
|
990
|
+
return {
|
|
991
|
+
"value": len(corrected) / len(cohort),
|
|
992
|
+
"numerator": len(corrected),
|
|
993
|
+
"denominator": len(cohort),
|
|
994
|
+
"detail": {
|
|
995
|
+
"metric_definition": "responses.correction_rate/v1",
|
|
996
|
+
"cohort": "receipt_backed_owner_outbound",
|
|
997
|
+
"correction_binding": "context_hash_to_external_ref",
|
|
998
|
+
},
|
|
999
|
+
}
|
|
1000
|
+
|
|
1001
|
+
async def _m_recall(self, since, until):
|
|
1002
|
+
"""Probe: re-query high-confidence shared facts against graph recall
|
|
1003
|
+
and grade by token coverage. Records each probe as a sample."""
|
|
1004
|
+
rows = self._probe_rows()
|
|
1005
|
+
if rows is None or not rows:
|
|
1006
|
+
return None
|
|
1007
|
+
picks = random.sample(rows, min(self.probe_count, len(rows)))
|
|
1008
|
+
return await self._run_probes(picks, source="benchmark")
|
|
1009
|
+
|
|
1010
|
+
async def run_recall_probe(self, probes: int = 50,
|
|
1011
|
+
seed: Optional[int] = None
|
|
1012
|
+
) -> Optional[Dict[str, Any]]:
|
|
1013
|
+
"""On-demand recall probe: the same derivation as the weekly
|
|
1014
|
+
recall.fact_coverage metric, but with a seeded, deterministic fact
|
|
1015
|
+
pick so before/after comparisons measure the recall path rather than
|
|
1016
|
+
sampling noise. Read-only against the graph. Samples are recorded
|
|
1017
|
+
with source="manual-probe"; rollups never read recall.probe samples,
|
|
1018
|
+
so manual probing cannot distort the weekly scorecard."""
|
|
1019
|
+
rows = self._probe_rows()
|
|
1020
|
+
if rows is None or not rows:
|
|
1021
|
+
return None
|
|
1022
|
+
n = max(1, min(int(probes), 100))
|
|
1023
|
+
picks = random.Random(seed).sample(rows, min(n, len(rows)))
|
|
1024
|
+
out = await self._run_probes(picks, source="manual-probe")
|
|
1025
|
+
if out is not None:
|
|
1026
|
+
out["detail"]["seed"] = seed
|
|
1027
|
+
out["detail"]["source"] = "manual-probe"
|
|
1028
|
+
return out
|
|
1029
|
+
|
|
1030
|
+
def _probe_rows(self) -> Optional[List[Dict[str, Any]]]:
|
|
1031
|
+
"""High-confidence shared facts to probe, or None when a source
|
|
1032
|
+
(graph or facts store) is unavailable — honest-skip, never zero."""
|
|
1033
|
+
graph = self._dep("graph")
|
|
1034
|
+
facts = self._dep("facts")
|
|
1035
|
+
if graph is None or facts is None:
|
|
1036
|
+
return None
|
|
1037
|
+
kwargs: Dict[str, Any] = {"min_confidence": 0.75, "limit": 200}
|
|
1038
|
+
if cognition_p4_mode() == "live":
|
|
1039
|
+
owner = self.owner_contact_id
|
|
1040
|
+
if not owner:
|
|
1041
|
+
return None
|
|
1042
|
+
# The store may call this key contact_id or subject_person_id; the
|
|
1043
|
+
# read is narrowed in both the query and the post-filter.
|
|
1044
|
+
kwargs["contact_id"] = owner
|
|
1045
|
+
rows = facts.list_facts(**kwargs).get("facts", [])
|
|
1046
|
+
if cognition_p4_mode() != "live":
|
|
1047
|
+
return rows
|
|
1048
|
+
allowed: List[Dict[str, Any]] = []
|
|
1049
|
+
owner = self.owner_contact_id
|
|
1050
|
+
for raw in rows:
|
|
1051
|
+
row = raw if isinstance(raw, dict) else vars(raw)
|
|
1052
|
+
subject = (row.get("subject_person_id")
|
|
1053
|
+
or row.get("contact_id") or "")
|
|
1054
|
+
if not row.get("id"):
|
|
1055
|
+
continue
|
|
1056
|
+
shareability = str(
|
|
1057
|
+
row.get("shareability") or "owner_private").lower()
|
|
1058
|
+
if str(subject) != owner:
|
|
1059
|
+
continue
|
|
1060
|
+
if shareability not in {"owner_private", "shared", "public"}:
|
|
1061
|
+
continue
|
|
1062
|
+
allowed.append(row)
|
|
1063
|
+
return allowed
|
|
1064
|
+
|
|
1065
|
+
async def _run_probes(self, picks: List[Any],
|
|
1066
|
+
source: str) -> Optional[Dict[str, Any]]:
|
|
1067
|
+
"""Grade each picked fact against graph recall (token coverage),
|
|
1068
|
+
recording one recall.probe sample per fact under `source`."""
|
|
1069
|
+
graph = self._dep("graph")
|
|
1070
|
+
hits = 0
|
|
1071
|
+
judged = 0
|
|
1072
|
+
for f in picks:
|
|
1073
|
+
fact = (f.get("fact") if isinstance(f, dict)
|
|
1074
|
+
else getattr(f, "fact", "")) or ""
|
|
1075
|
+
if not fact.strip():
|
|
1076
|
+
continue
|
|
1077
|
+
subject = (f.get("subject_person_id") or f.get("contact_id")
|
|
1078
|
+
if isinstance(f, dict) else
|
|
1079
|
+
getattr(f, "subject_person_id", None)
|
|
1080
|
+
or getattr(f, "contact_id", None))
|
|
1081
|
+
try:
|
|
1082
|
+
# bound each probe so a wedged graph connection can't hang the
|
|
1083
|
+
# benchmark (and, through it, the autonomy tick)
|
|
1084
|
+
recall_kwargs = {"limit": 5, "min_confidence": 0.1}
|
|
1085
|
+
if cognition_p4_mode() == "live":
|
|
1086
|
+
if not subject:
|
|
1087
|
+
continue
|
|
1088
|
+
recall_kwargs["person_id"] = str(subject)
|
|
1089
|
+
results = await asyncio.wait_for(
|
|
1090
|
+
graph.recall(fact, **recall_kwargs), timeout=8.0)
|
|
1091
|
+
except (Exception, asyncio.TimeoutError):
|
|
1092
|
+
continue
|
|
1093
|
+
hit = 1.0 if self._covered(fact, results) else 0.0
|
|
1094
|
+
hits += int(hit)
|
|
1095
|
+
judged += 1
|
|
1096
|
+
fact_id = (f.get("id") if isinstance(f, dict)
|
|
1097
|
+
else getattr(f, "id", None))
|
|
1098
|
+
if cognition_p4_mode() == "live":
|
|
1099
|
+
self.store.add_evidence_sample(
|
|
1100
|
+
"recall.fact_coverage", hit, definition_version="v2",
|
|
1101
|
+
sample_principal="benchmark:recall-probe",
|
|
1102
|
+
source_ref=f"fact:{fact_id}",
|
|
1103
|
+
sample_id=(f"recall-{source}-{fact_id}-"
|
|
1104
|
+
f"{int(_now() * 1000000)}"),
|
|
1105
|
+
meta={"fact_id": fact_id, "subject_person_id": subject},
|
|
1106
|
+
)
|
|
1107
|
+
else:
|
|
1108
|
+
self.store.add_sample(
|
|
1109
|
+
"recall.probe", hit, source=source,
|
|
1110
|
+
meta={"fact_id": fact_id})
|
|
1111
|
+
n = judged if cognition_p4_mode() == "live" else len(picks)
|
|
1112
|
+
if n == 0:
|
|
1113
|
+
return None
|
|
1114
|
+
return {"value": hits / n, "numerator": hits, "denominator": n,
|
|
1115
|
+
"detail": {
|
|
1116
|
+
"probes": n,
|
|
1117
|
+
**({"metric_definition": "recall.fact_coverage/v2",
|
|
1118
|
+
"viewer_scope": self.owner_contact_id}
|
|
1119
|
+
if cognition_p4_mode() == "live" else {}),
|
|
1120
|
+
}}
|
|
1121
|
+
|
|
1122
|
+
@staticmethod
|
|
1123
|
+
def _covered(fact: str, results: List[Dict[str, Any]],
|
|
1124
|
+
threshold: float = 0.5) -> bool:
|
|
1125
|
+
words = {w for w in re.findall(r"[a-z0-9]+", fact.lower())
|
|
1126
|
+
if len(w) > 3}
|
|
1127
|
+
if not words:
|
|
1128
|
+
return False
|
|
1129
|
+
for r in results or []:
|
|
1130
|
+
content = str((r or {}).get("content", "")).lower()
|
|
1131
|
+
if not content:
|
|
1132
|
+
continue
|
|
1133
|
+
got = sum(1 for w in words if w in content)
|
|
1134
|
+
if got / len(words) >= threshold:
|
|
1135
|
+
return True
|
|
1136
|
+
return False
|
|
1137
|
+
|
|
1138
|
+
async def _m_jobs(self, start, end):
|
|
1139
|
+
queue = self._dep("queue")
|
|
1140
|
+
if queue is None:
|
|
1141
|
+
return None
|
|
1142
|
+
# host wires the TaskQueueManager wrapper; the raw QueueManager
|
|
1143
|
+
# (which owns completed_durations) sits at .queue
|
|
1144
|
+
if not hasattr(queue, "completed_durations"):
|
|
1145
|
+
queue = getattr(queue, "queue", None)
|
|
1146
|
+
if queue is None or not hasattr(queue, "completed_durations"):
|
|
1147
|
+
return None
|
|
1148
|
+
durs = [d for d in await queue.completed_durations(
|
|
1149
|
+
start.isoformat(), end.isoformat()) if d >= 0]
|
|
1150
|
+
if not durs:
|
|
1151
|
+
return None
|
|
1152
|
+
return {"value": _percentile(durs, 50),
|
|
1153
|
+
"numerator": None, "denominator": None,
|
|
1154
|
+
"detail": {"p50": _percentile(durs, 50),
|
|
1155
|
+
"p95": _percentile(durs, 95), "n": len(durs)}}
|
|
1156
|
+
|
|
1157
|
+
def _m_calibration(self, since: float) -> Dict[str, Any]:
|
|
1158
|
+
"""Per-domain prediction calibration from the expectation engine
|
|
1159
|
+
(Mind M3a), expressed as accuracy = 1 - Brier so higher is better and
|
|
1160
|
+
it fits the benchmark's higher-is-better convention."""
|
|
1161
|
+
eng = self._host_attr("_expectations")
|
|
1162
|
+
if eng is None:
|
|
1163
|
+
return {}
|
|
1164
|
+
out: Dict[str, Any] = {}
|
|
1165
|
+
try:
|
|
1166
|
+
cal = eng.calibration(since=since)
|
|
1167
|
+
except Exception:
|
|
1168
|
+
return {}
|
|
1169
|
+
for domain, r in (cal or {}).items():
|
|
1170
|
+
brier = r.get("brier")
|
|
1171
|
+
if brier is None:
|
|
1172
|
+
continue
|
|
1173
|
+
out[f"calibration.{domain}"] = {
|
|
1174
|
+
"value": max(0.0, 1.0 - float(brier)),
|
|
1175
|
+
"numerator": None, "denominator": None,
|
|
1176
|
+
"detail": {"brier": brier, "n": r.get("n"),
|
|
1177
|
+
"hit_rate": r.get("hit_rate")}}
|
|
1178
|
+
return out
|
|
1179
|
+
|
|
1180
|
+
def _m_submitted(self, since: float, until: float,
|
|
1181
|
+
skip: Optional[set] = None) -> Dict[str, Any]:
|
|
1182
|
+
"""Roll up host-submitted samples generically."""
|
|
1183
|
+
skip = skip or set()
|
|
1184
|
+
by_metric: Dict[str, List[float]] = {}
|
|
1185
|
+
for s in self.store.samples_in(since, until):
|
|
1186
|
+
if s["metric"] == "recall.probe" or s["metric"] in skip:
|
|
1187
|
+
continue
|
|
1188
|
+
if cognition_p4_mode() == "live":
|
|
1189
|
+
definition = self.store.definition(
|
|
1190
|
+
s["metric"], s.get("definition_version") or "")
|
|
1191
|
+
if (definition is None
|
|
1192
|
+
or s.get("evidence_status") not in {"observed", "verified"}
|
|
1193
|
+
or not s.get("sample_principal")
|
|
1194
|
+
or not s.get("source_ref")):
|
|
1195
|
+
continue
|
|
1196
|
+
by_metric.setdefault(s["metric"], []).append(s["value"])
|
|
1197
|
+
out: Dict[str, Any] = {}
|
|
1198
|
+
for metric, vals in by_metric.items():
|
|
1199
|
+
if metric.startswith("latency."):
|
|
1200
|
+
out[metric] = {
|
|
1201
|
+
"value": _percentile(vals, 50),
|
|
1202
|
+
"numerator": None, "denominator": None,
|
|
1203
|
+
"detail": {"p50": _percentile(vals, 50),
|
|
1204
|
+
"p95": _percentile(vals, 95),
|
|
1205
|
+
"n": len(vals)}}
|
|
1206
|
+
else:
|
|
1207
|
+
out[metric] = {
|
|
1208
|
+
"value": sum(vals) / len(vals),
|
|
1209
|
+
"numerator": None, "denominator": None,
|
|
1210
|
+
"detail": {"n": len(vals),
|
|
1211
|
+
"min": min(vals), "max": max(vals)}}
|
|
1212
|
+
return out
|
|
1213
|
+
|
|
1214
|
+
# -- read side --------------------------------------------------------
|
|
1215
|
+
def snapshot(self, weeks: int = 8) -> Dict[str, Any]:
|
|
1216
|
+
"""Rollups for the last N weeks plus latest-vs-previous deltas."""
|
|
1217
|
+
rolls = self.store.rollups(weeks=weeks)
|
|
1218
|
+
ordered = sorted(rolls.keys(), reverse=True)
|
|
1219
|
+
# A persisted score computed before a correction is unsafe to show as
|
|
1220
|
+
# current truth. Hide it until compute_week rebuilds that exact slice.
|
|
1221
|
+
comp = self._dep("competence")
|
|
1222
|
+
if comp is not None:
|
|
1223
|
+
all_action_domains: List[str] = []
|
|
1224
|
+
try:
|
|
1225
|
+
all_action_domains = [
|
|
1226
|
+
str(r.get("domain")) for r in comp.snapshot()
|
|
1227
|
+
if isinstance(r, dict) and r.get("domain")
|
|
1228
|
+
and r.get("domain") != "delivery"]
|
|
1229
|
+
except Exception:
|
|
1230
|
+
pass
|
|
1231
|
+
for wk, metrics in rolls.items():
|
|
1232
|
+
try:
|
|
1233
|
+
start, end = week_window(wk)
|
|
1234
|
+
except (TypeError, ValueError):
|
|
1235
|
+
continue
|
|
1236
|
+
since, until = start.timestamp(), end.timestamp()
|
|
1237
|
+
for metric, domains in (
|
|
1238
|
+
("delivery.success", ["delivery"]),
|
|
1239
|
+
("initiative.acceptance", ["delivery"]),
|
|
1240
|
+
("actions.success", all_action_domains),
|
|
1241
|
+
):
|
|
1242
|
+
row = metrics.get(metric)
|
|
1243
|
+
if row is None:
|
|
1244
|
+
continue
|
|
1245
|
+
detail = dict(row.get("detail") or {})
|
|
1246
|
+
metric_domains = list(domains)
|
|
1247
|
+
if metric == "actions.success":
|
|
1248
|
+
stored_domains = detail.get("domains") or {}
|
|
1249
|
+
if isinstance(stored_domains, dict):
|
|
1250
|
+
metric_domains = sorted(
|
|
1251
|
+
set(metric_domains) | set(stored_domains))
|
|
1252
|
+
revision, gaps = self._competence_state(
|
|
1253
|
+
metric_domains, since, until)
|
|
1254
|
+
recorded = int(detail.get(
|
|
1255
|
+
"competence_reconciliation_revision") or 0)
|
|
1256
|
+
if gaps:
|
|
1257
|
+
row.update(self._unavailable_competence(
|
|
1258
|
+
revision, gaps))
|
|
1259
|
+
elif revision > recorded:
|
|
1260
|
+
row["value"] = None
|
|
1261
|
+
row["numerator"] = None
|
|
1262
|
+
row["denominator"] = None
|
|
1263
|
+
detail.update({
|
|
1264
|
+
"available": False,
|
|
1265
|
+
"reason": (
|
|
1266
|
+
"stale_after_competence_reconciliation"),
|
|
1267
|
+
"computed_revision": recorded,
|
|
1268
|
+
"required_revision": revision,
|
|
1269
|
+
})
|
|
1270
|
+
row["detail"] = detail
|
|
1271
|
+
trends: Dict[str, Any] = {}
|
|
1272
|
+
if len(ordered) >= 2:
|
|
1273
|
+
cur, prev = rolls[ordered[0]], rolls[ordered[1]]
|
|
1274
|
+
for metric, r in cur.items():
|
|
1275
|
+
pv = (prev.get(metric) or {}).get("value")
|
|
1276
|
+
if r.get("value") is not None and pv is not None:
|
|
1277
|
+
trends[metric] = round(r["value"] - pv, 4)
|
|
1278
|
+
return {
|
|
1279
|
+
"format": "colony.selfhood-benchmark/v2",
|
|
1280
|
+
"mode": cognition_p4_mode(),
|
|
1281
|
+
"weeks": ordered,
|
|
1282
|
+
"rollups": rolls,
|
|
1283
|
+
"trends": trends,
|
|
1284
|
+
"latest": ordered[0] if ordered else None,
|
|
1285
|
+
"definitions": self.store.definitions(),
|
|
1286
|
+
}
|
|
1287
|
+
|
|
1288
|
+
def canonical_summary(self, weeks: int = 8) -> Dict[str, Any]:
|
|
1289
|
+
"""Canonical replacement payload for the deprecated legacy CPI API."""
|
|
1290
|
+
|
|
1291
|
+
return {
|
|
1292
|
+
"deprecated_cpi": True,
|
|
1293
|
+
"canonical": "selfhood_benchmark",
|
|
1294
|
+
"canonical_endpoint": "/v1/host/self/benchmark",
|
|
1295
|
+
**self.snapshot(weeks=weeks),
|
|
1296
|
+
}
|
|
1297
|
+
|
|
1298
|
+
|
|
1299
|
+
def legacy_cpi_payload(benchmark: Optional[SelfhoodBenchmark],
|
|
1300
|
+
weeks: int = 8) -> Dict[str, Any]:
|
|
1301
|
+
"""A truthful compatibility response; no fabricated CPI dimensions."""
|
|
1302
|
+
|
|
1303
|
+
if benchmark is None:
|
|
1304
|
+
return {
|
|
1305
|
+
"deprecated": True,
|
|
1306
|
+
"available": False,
|
|
1307
|
+
"canonical_endpoint": "/v1/host/self/benchmark",
|
|
1308
|
+
"reason": "canonical benchmark is not wired",
|
|
1309
|
+
}
|
|
1310
|
+
return {
|
|
1311
|
+
"deprecated": True,
|
|
1312
|
+
"available": True,
|
|
1313
|
+
**benchmark.canonical_summary(weeks=weeks),
|
|
1314
|
+
}
|