apsimo 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apsimo/__init__.py +38 -0
- apsimo/__main__.py +6 -0
- apsimo/agent/__init__.py +6 -0
- apsimo/agent/client.py +276 -0
- apsimo/agent/models.py +46 -0
- apsimo/agents/__init__.py +20 -0
- apsimo/agents/models.py +264 -0
- apsimo/agents/store.py +861 -0
- apsimo/agents/websocket.py +522 -0
- apsimo/api/__init__.py +1 -0
- apsimo/api/auth_telemetry.py +287 -0
- apsimo/api/authority.py +1203 -0
- apsimo/api/contact_grants.py +347 -0
- apsimo/api/middleware.py +483 -0
- apsimo/api/routers/__init__.py +1 -0
- apsimo/api/routers/commitment_work.py +265 -0
- apsimo/api/routers/context_gate.py +123 -0
- apsimo/api/routers/executions.py +140 -0
- apsimo/api/routers/followup_plans.py +147 -0
- apsimo/api/routers/governed_actions.py +162 -0
- apsimo/api/routers/host.py +14473 -0
- apsimo/api/routers/initiative_work.py +115 -0
- apsimo/api/routers/mining.py +104 -0
- apsimo/api/routers/observations.py +110 -0
- apsimo/api/routers/social_state.py +225 -0
- apsimo/api/routers/task_queue.py +2715 -0
- apsimo/api/routers/temporal_followups.py +251 -0
- apsimo/api/routers/transport.py +110 -0
- apsimo/api/routers/transport_ingress_api.py +240 -0
- apsimo/api/schemas/__init__.py +1 -0
- apsimo/api/schemas/host.py +1949 -0
- apsimo/autonomy/cli.py +110 -0
- apsimo/autonomy/condition_worker.py +437 -0
- apsimo/autonomy/config.py +424 -0
- apsimo/autonomy/loop.py +4316 -0
- apsimo/autonomy/registry.py +339 -0
- apsimo/autonomy/scheduler.py +1822 -0
- apsimo/autonomy/synthesis.py +449 -0
- apsimo/backup.py +962 -0
- apsimo/beliefs/__init__.py +23 -0
- apsimo/beliefs/contradictions.py +109 -0
- apsimo/beliefs/decay.py +61 -0
- apsimo/beliefs/engine.py +479 -0
- apsimo/beliefs/models.py +67 -0
- apsimo/beliefs/promotion.py +41 -0
- apsimo/beliefs/resolve.py +58 -0
- apsimo/beliefs/source_claims.py +690 -0
- apsimo/beliefs/source_projection.py +883 -0
- apsimo/beliefs/source_time.py +208 -0
- apsimo/beliefs/store.py +133 -0
- apsimo/briefings/aggregators.py +824 -0
- apsimo/briefings/composer.py +420 -0
- apsimo/briefings/config.py +55 -0
- apsimo/briefings/delivery.py +439 -0
- apsimo/briefings/engagement.py +97 -0
- apsimo/briefings/engine.py +274 -0
- apsimo/briefings/enhancer.py +99 -0
- apsimo/briefings/models.py +183 -0
- apsimo/briefings/scheduler.py +382 -0
- apsimo/briefings/store.py +435 -0
- apsimo/chain/__init__.py +48 -0
- apsimo/chain/block.py +100 -0
- apsimo/chain/cli.py +704 -0
- apsimo/chain/genesis.py +443 -0
- apsimo/chain/identity.py +416 -0
- apsimo/chain/keys.py +1025 -0
- apsimo/chain/local_keys.py +187 -0
- apsimo/chain/manager.py +290 -0
- apsimo/chain/node.py +163 -0
- apsimo/chain/plugin_transactions.py +371 -0
- apsimo/chain/protocol.py +220 -0
- apsimo/chain/state_machine.py +676 -0
- apsimo/chain/storage.py +503 -0
- apsimo/chain/transactions.py +250 -0
- apsimo/chain/validation.py +397 -0
- apsimo/channels/__init__.py +1 -0
- apsimo/channels/manifest.py +31 -0
- apsimo/channels/migrations/001_channels_schema.sql +12 -0
- apsimo/channels/phone_gateways.py +42 -0
- apsimo/channels/presence.py +188 -0
- apsimo/channels/router.py +235 -0
- apsimo/channels/store.py +231 -0
- apsimo/cli.py +2688 -0
- apsimo/cognition/__init__.py +11 -0
- apsimo/cognition/charter.py +398 -0
- apsimo/cognition/drive_governance.py +3530 -0
- apsimo/cognition/evidence_pipeline.py +1627 -0
- apsimo/cognition/external_events.py +932 -0
- apsimo/cognition/goal_spine.py +3488 -0
- apsimo/cognition/introspection.py +214 -0
- apsimo/cognition/prompt.py +150 -0
- apsimo/cognition/runtime.py +108 -0
- apsimo/cognition/trigger.py +154 -0
- apsimo/commitments/__init__.py +18 -0
- apsimo/commitments/local_work.py +355 -0
- apsimo/commitments/store.py +1052 -0
- apsimo/commitments/work.py +91 -0
- apsimo/compat.py +53 -0
- apsimo/compression/__init__.py +467 -0
- apsimo/connectors/__init__.py +21 -0
- apsimo/connectors/base.py +152 -0
- apsimo/connectors/caldav_calendar.py +125 -0
- apsimo/connectors/fs_documents.py +85 -0
- apsimo/connectors/imap_email.py +138 -0
- apsimo/connectors/manager.py +218 -0
- apsimo/connectors/webhook_pull.py +88 -0
- apsimo/contacts/__init__.py +33 -0
- apsimo/contacts/comms.py +357 -0
- apsimo/contacts/config.py +79 -0
- apsimo/contacts/exporters/__init__.py +1 -0
- apsimo/contacts/exporters/vcard.py +71 -0
- apsimo/contacts/identity_links.py +251 -0
- apsimo/contacts/importer.py +280 -0
- apsimo/contacts/importers/__init__.py +1 -0
- apsimo/contacts/importers/batch.py +43 -0
- apsimo/contacts/importers/macos_contacts.py +101 -0
- apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
- apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
- apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
- apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
- apsimo/contacts/migrations/005_identity_links.sql +27 -0
- apsimo/contacts/models.py +308 -0
- apsimo/contacts/scoring.py +16 -0
- apsimo/contacts/store.py +1623 -0
- apsimo/contacts/transport_ingress.py +252 -0
- apsimo/contacts/world_bridge.py +314 -0
- apsimo/contextgate/__init__.py +69 -0
- apsimo/contextgate/chunker.py +169 -0
- apsimo/contextgate/estimate.py +54 -0
- apsimo/contextgate/gate.py +313 -0
- apsimo/contextgate/retrieve.py +115 -0
- apsimo/delivery/__init__.py +16 -0
- apsimo/delivery/bridge.py +1260 -0
- apsimo/delivery/channels.py +526 -0
- apsimo/delivery/classification.py +50 -0
- apsimo/delivery/rate_limiter.py +268 -0
- apsimo/delivery/reachout_policy.py +206 -0
- apsimo/directed/__init__.py +22 -0
- apsimo/directed/audit.py +167 -0
- apsimo/directed/intake.py +95 -0
- apsimo/directed/models.py +191 -0
- apsimo/directed/service.py +509 -0
- apsimo/directives/__init__.py +25 -0
- apsimo/directives/evidence.py +87 -0
- apsimo/directives/extractor.py +188 -0
- apsimo/directives/guard.py +364 -0
- apsimo/directives/models.py +206 -0
- apsimo/directives/service.py +372 -0
- apsimo/directives/store.py +167 -0
- apsimo/doctor.py +2173 -0
- apsimo/environment.py +43 -0
- apsimo/events/__init__.py +33 -0
- apsimo/events/broadcaster.py +98 -0
- apsimo/events/bus.py +217 -0
- apsimo/events/journal.py +863 -0
- apsimo/events/stream.py +131 -0
- apsimo/events/types.py +150 -0
- apsimo/execution_results.py +357 -0
- apsimo/feedback/__init__.py +5 -0
- apsimo/feedback/store.py +76 -0
- apsimo/feeds/__init__.py +19 -0
- apsimo/feeds/cli.py +84 -0
- apsimo/feeds/engine.py +437 -0
- apsimo/feeds/example-feed.yaml +77 -0
- apsimo/feeds/hermes_cron.py +126 -0
- apsimo/feeds/manager.py +235 -0
- apsimo/feeds/spec.py +250 -0
- apsimo/feeds/template.py +202 -0
- apsimo/gate/__init__.py +18 -0
- apsimo/gate/audit.py +61 -0
- apsimo/gate/communication_policy.py +166 -0
- apsimo/gate/config.py +72 -0
- apsimo/gate/context_provenance.py +170 -0
- apsimo/gate/env_risk.py +226 -0
- apsimo/gate/guard_audit.py +353 -0
- apsimo/gate/layers/__init__.py +1 -0
- apsimo/gate/layers/base.py +15 -0
- apsimo/gate/layers/l1_recipient.py +66 -0
- apsimo/gate/layers/l2_pii.py +134 -0
- apsimo/gate/layers/l3_cross_context.py +50 -0
- apsimo/gate/layers/l4_trust_tier.py +78 -0
- apsimo/gate/layers/l5_injection.py +199 -0
- apsimo/gate/layers/l6_review.py +86 -0
- apsimo/gate/layers/l7_delay.py +100 -0
- apsimo/gate/layers/tom2_epistemic.py +185 -0
- apsimo/gate/models.py +64 -0
- apsimo/gate/pending_dispatch.py +5 -0
- apsimo/gate/pipeline.py +206 -0
- apsimo/gate/rejection.py +259 -0
- apsimo/gate/response_guard.py +700 -0
- apsimo/gate/rulesets/injection_v1.yaml +51 -0
- apsimo/gate/surface_policy.py +189 -0
- apsimo/gate/taint.py +226 -0
- apsimo/genesis.json +9 -0
- apsimo/goals/__init__.py +100 -0
- apsimo/goals/config.py +38 -0
- apsimo/goals/decomposer.py +421 -0
- apsimo/goals/engine.py +617 -0
- apsimo/goals/inference.py +354 -0
- apsimo/goals/models.py +302 -0
- apsimo/goals/priority.py +270 -0
- apsimo/goals/queue_bridge.py +149 -0
- apsimo/goals/replan.py +450 -0
- apsimo/goals/schema.sql +89 -0
- apsimo/goals/store.py +692 -0
- apsimo/governed_actions.py +1708 -0
- apsimo/harness_integration/__init__.py +45 -0
- apsimo/harness_integration/context.py +41 -0
- apsimo/harness_integration/skills.py +231 -0
- apsimo/identity/__init__.py +26 -0
- apsimo/identity/participants.py +181 -0
- apsimo/identity/resolver.py +329 -0
- apsimo/identity_bootstrap/__init__.py +5 -0
- apsimo/identity_bootstrap/builder.py +208 -0
- apsimo/identity_bootstrap/corpus.py +443 -0
- apsimo/identity_bootstrap/models.py +54 -0
- apsimo/identity_bootstrap/runner.py +353 -0
- apsimo/identity_bootstrap/seeders/__init__.py +25 -0
- apsimo/identity_bootstrap/seeders/briefings.py +109 -0
- apsimo/identity_bootstrap/seeders/chain.py +57 -0
- apsimo/identity_bootstrap/seeders/goals.py +128 -0
- apsimo/identity_bootstrap/seeders/memory.py +191 -0
- apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
- apsimo/identity_bootstrap/seeders/relationship.py +152 -0
- apsimo/identity_bootstrap/seeders/sessions.py +67 -0
- apsimo/identity_bootstrap/seeders/skills.py +92 -0
- apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
- apsimo/identity_bootstrap/seeders/world_model.py +143 -0
- apsimo/identity_bootstrap/self_query.py +92 -0
- apsimo/identity_bootstrap/self_reflection.py +155 -0
- apsimo/identity_bootstrap/skill.py +37 -0
- apsimo/identity_bootstrap/verifier.py +436 -0
- apsimo/initiatives/__init__.py +20 -0
- apsimo/initiatives/action_registry.py +454 -0
- apsimo/initiatives/approval_authority.py +2105 -0
- apsimo/initiatives/approval_policy.py +123 -0
- apsimo/initiatives/assignment.py +263 -0
- apsimo/initiatives/backup_evidence.py +100 -0
- apsimo/initiatives/context_freshness.py +103 -0
- apsimo/initiatives/models.py +318 -0
- apsimo/initiatives/native_work.py +270 -0
- apsimo/initiatives/standing_approvals.py +232 -0
- apsimo/initiatives/store.py +1081 -0
- apsimo/initiatives/temporal_followup.py +410 -0
- apsimo/intelligence/__init__.py +1 -0
- apsimo/intelligence/cognition/__init__.py +24 -0
- apsimo/intelligence/cognition/gap_detector.py +148 -0
- apsimo/intelligence/cognition/metalearner.py +547 -0
- apsimo/intelligence/cognition/metrics_collector.py +217 -0
- apsimo/intelligence/cognition/performance_index.py +299 -0
- apsimo/intelligence/cognition/registry.py +192 -0
- apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
- apsimo/intelligence/cognition/types.py +16 -0
- apsimo/intelligence/components/__init__.py +66 -0
- apsimo/intelligence/components/anomaly_detector.py +413 -0
- apsimo/intelligence/components/initiative_engine.py +2643 -0
- apsimo/intelligence/components/preference_learner.py +521 -0
- apsimo/intelligence/components/research_orchestrator.py +358 -0
- apsimo/intelligence/components/self_directed_thinker.py +221 -0
- apsimo/intelligence/components/self_reflector.py +252 -0
- apsimo/intelligence/components/session_continuity.py +154 -0
- apsimo/intelligence/components/task_planner.py +320 -0
- apsimo/intelligence/components/tool_learner.py +217 -0
- apsimo/intelligence/graph/__init__.py +79 -0
- apsimo/intelligence/graph/client.py +2483 -0
- apsimo/intelligence/graph/consolidator.py +405 -0
- apsimo/intelligence/graph/distiller.py +312 -0
- apsimo/intelligence/graph/migrations.py +129 -0
- apsimo/intelligence/graph/queries.py +248 -0
- apsimo/intelligence/graph/recall.py +281 -0
- apsimo/intelligence/graph/reconciler.py +144 -0
- apsimo/intelligence/graph/schema.py +337 -0
- apsimo/intelligence/graph/selection.py +252 -0
- apsimo/intelligence/learning/__init__.py +17 -0
- apsimo/intelligence/learning/continuous_learner.py +245 -0
- apsimo/intelligence/learning/feedback_store.py +321 -0
- apsimo/intelligence/mind_model/__init__.py +1 -0
- apsimo/intelligence/mind_model/graph_baseline.py +136 -0
- apsimo/intelligence/mind_model/signal_collector.py +361 -0
- apsimo/intelligence/relationships/__init__.py +11 -0
- apsimo/intelligence/relationships/profiler.py +389 -0
- apsimo/intelligence/relationships/scorer.py +560 -0
- apsimo/intelligence/relationships/signal_floor.py +66 -0
- apsimo/intelligence/relationships/trust_tiers.py +300 -0
- apsimo/intelligence/synthesis/__init__.py +40 -0
- apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
- apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
- apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
- apsimo/intelligence/synthesis/insight_store.py +79 -0
- apsimo/intelligence/synthesis/insight_validator.py +183 -0
- apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
- apsimo/intelligence/turn_middleware/__init__.py +15 -0
- apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
- apsimo/mcp/__init__.py +41 -0
- apsimo/mcp/__main__.py +6 -0
- apsimo/mcp/config.py +287 -0
- apsimo/mcp/server.py +501 -0
- apsimo/migrations.py +187 -0
- apsimo/mining/__init__.py +27 -0
- apsimo/mining/corpus.py +239 -0
- apsimo/mining/escalations.py +289 -0
- apsimo/mining/models.py +169 -0
- apsimo/mining/store.py +210 -0
- apsimo/models/__init__.py +30 -0
- apsimo/models/memory.py +80 -0
- apsimo/models/mesh.py +72 -0
- apsimo/models/person.py +104 -0
- apsimo/models/signal.py +108 -0
- apsimo/observations/__init__.py +15 -0
- apsimo/observations/store.py +277 -0
- apsimo/patterns/__init__.py +6 -0
- apsimo/patterns/extract.py +187 -0
- apsimo/patterns/store.py +227 -0
- apsimo/persona/__init__.py +1 -0
- apsimo/persona/engine.py +611 -0
- apsimo/persona/manifest.py +140 -0
- apsimo/projects/__init__.py +28 -0
- apsimo/projects/engine.py +1681 -0
- apsimo/projects/event_outbox.py +188 -0
- apsimo/projects/models.py +216 -0
- apsimo/projects/planner.py +181 -0
- apsimo/projects/store.py +1446 -0
- apsimo/proposals/__init__.py +12 -0
- apsimo/proposals/engine.py +114 -0
- apsimo/proposals/models.py +207 -0
- apsimo/qualification/__init__.py +1 -0
- apsimo/qualification/cases.py +75 -0
- apsimo/qualification/cli.py +51 -0
- apsimo/qualification/memory_cases.py +209 -0
- apsimo/qualification/records.py +92 -0
- apsimo/qualification/report.py +87 -0
- apsimo/qualification/runner.py +311 -0
- apsimo/qualification/structured_cases.py +131 -0
- apsimo/reasoning/__init__.py +13 -0
- apsimo/reasoning/executor.py +506 -0
- apsimo/reasoning/loop.py +373 -0
- apsimo/reasoning/native_tools/__init__.py +16 -0
- apsimo/reasoning/native_tools/calculate.py +141 -0
- apsimo/reasoning/native_tools/file_ops.py +150 -0
- apsimo/reasoning/native_tools/web_search.py +49 -0
- apsimo/reasoning/tool_policy.py +182 -0
- apsimo/redact/__init__.py +176 -0
- apsimo/repos/__init__.py +5 -0
- apsimo/repos/mirrors.py +204 -0
- apsimo/research/__init__.py +41 -0
- apsimo/research/artifact.py +482 -0
- apsimo/research/gatherer.py +387 -0
- apsimo/research/pipeline.py +513 -0
- apsimo/research/search/__init__.py +7 -0
- apsimo/research/search/base.py +41 -0
- apsimo/research/search/brave.py +59 -0
- apsimo/research/search/cache.py +51 -0
- apsimo/research/search/duckduckgo.py +103 -0
- apsimo/research/search/orchestrator.py +119 -0
- apsimo/research/search/serpapi.py +59 -0
- apsimo/research/search/tavily.py +59 -0
- apsimo/research/synthesizer.py +309 -0
- apsimo/router/__init__.py +30 -0
- apsimo/router/complexity_scorer.py +148 -0
- apsimo/router/endpoints.py +153 -0
- apsimo/router/fallback.py +58 -0
- apsimo/router/functions.py +243 -0
- apsimo/router/native_policy.py +52 -0
- apsimo/router/router.py +762 -0
- apsimo/router/self_learning.py +174 -0
- apsimo/router/tiers.py +677 -0
- apsimo/sandbox/__init__.py +21 -0
- apsimo/sandbox/backend.py +195 -0
- apsimo/sandbox/manager.py +173 -0
- apsimo/scope_bounds.py +7 -0
- apsimo/secrets/__init__.py +6 -0
- apsimo/secrets/backends/__init__.py +8 -0
- apsimo/secrets/backends/base.py +42 -0
- apsimo/secrets/backends/env.py +110 -0
- apsimo/secrets/backends/keyring.py +72 -0
- apsimo/secrets/backends/onepassword.py +232 -0
- apsimo/secrets/cli.py +191 -0
- apsimo/secrets/manager.py +160 -0
- apsimo/secrets/migration.py +101 -0
- apsimo/secrets/types.py +98 -0
- apsimo/seed.py +41 -0
- apsimo/self_model/__init__.py +37 -0
- apsimo/self_model/appraisals.py +673 -0
- apsimo/self_model/benchmark.py +1314 -0
- apsimo/self_model/brief.py +40 -0
- apsimo/self_model/event_concerns.py +1128 -0
- apsimo/self_model/execution_forecasts.py +353 -0
- apsimo/self_model/expectations.py +1595 -0
- apsimo/self_model/experiments.py +1150 -0
- apsimo/self_model/journal.py +148 -0
- apsimo/self_model/judgments.py +705 -0
- apsimo/self_model/native_outcomes.py +55 -0
- apsimo/self_model/params.py +220 -0
- apsimo/self_model/perspective.py +246 -0
- apsimo/self_model/reconcile.py +183 -0
- apsimo/self_model/reply_forecasts.py +381 -0
- apsimo/self_model/runtime_forecasts.py +296 -0
- apsimo/self_model/runtime_models.py +67 -0
- apsimo/self_model/settlement.py +207 -0
- apsimo/self_model/situation.py +1731 -0
- apsimo/self_model/store.py +883 -0
- apsimo/self_model/supervised.py +137 -0
- apsimo/self_model/thinker.py +99 -0
- apsimo/self_model/trust.py +388 -0
- apsimo/self_model/workspace.py +2388 -0
- apsimo/server.py +4197 -0
- apsimo/services/__init__.py +1 -0
- apsimo/services/agent_bridge.py +474 -0
- apsimo/services/initiative_executor.py +914 -0
- apsimo/services/instance.py +297 -0
- apsimo/sessions/__init__.py +22 -0
- apsimo/sessions/config.py +13 -0
- apsimo/sessions/context_loader.py +88 -0
- apsimo/sessions/federation_session.py +75 -0
- apsimo/sessions/isolated_session.py +98 -0
- apsimo/sessions/reports.py +84 -0
- apsimo/sessions/store.py +148 -0
- apsimo/setup.py +2818 -0
- apsimo/setup_hermes.py +879 -0
- apsimo/setup_local_work.py +218 -0
- apsimo/setup_native_goals.py +134 -0
- apsimo/setup_native_reviews.py +115 -0
- apsimo/skills/__init__.py +10 -0
- apsimo/skills/base.py +108 -0
- apsimo/skills/budget.py +28 -0
- apsimo/skills/executor.py +493 -0
- apsimo/skills/executors/__init__.py +1 -0
- apsimo/skills/executors/behavioral_correction.py +75 -0
- apsimo/skills/executors/capability_gap.py +38 -0
- apsimo/skills/executors/data_quality.py +163 -0
- apsimo/skills/executors/knowledge_acquisition.py +41 -0
- apsimo/skills/executors/operational_hygiene.py +185 -0
- apsimo/skills/executors/subsystem_health.py +169 -0
- apsimo/skills/hermes_export.py +431 -0
- apsimo/skills/index.py +123 -0
- apsimo/skills/learning/__init__.py +21 -0
- apsimo/skills/learning/novelty_detector.py +206 -0
- apsimo/skills/learning/pattern_extractor.py +199 -0
- apsimo/skills/learning/triggers.py +159 -0
- apsimo/skills/loader.py +246 -0
- apsimo/skills/migrations/002_progressive_loading.sql +6 -0
- apsimo/skills/migrations/backfill_triggers.py +20 -0
- apsimo/skills/models.py +202 -0
- apsimo/skills/packager.py +128 -0
- apsimo/skills/protocols.py +70 -0
- apsimo/skills/registry.py +191 -0
- apsimo/skills/runtime.py +58 -0
- apsimo/skills/sandbox_runner.py +229 -0
- apsimo/skills/scheduler.py +129 -0
- apsimo/skills/schema.py +79 -0
- apsimo/skills/security/__init__.py +12 -0
- apsimo/skills/security/guards.py +53 -0
- apsimo/skills/security/scanner.py +223 -0
- apsimo/skills_memory/__init__.py +26 -0
- apsimo/skills_memory/distill.py +159 -0
- apsimo/skills_memory/models.py +85 -0
- apsimo/skills_memory/retrieve.py +62 -0
- apsimo/skills_memory/store.py +172 -0
- apsimo/surprise/__init__.py +6 -0
- apsimo/surprise/accumulation.py +57 -0
- apsimo/surprise/scorer.py +102 -0
- apsimo/surprise/store.py +203 -0
- apsimo/task_queue/__init__.py +69 -0
- apsimo/task_queue/action_receipts.py +148 -0
- apsimo/task_queue/approval_relay_canary.py +108 -0
- apsimo/task_queue/config.py +85 -0
- apsimo/task_queue/contract.py +361 -0
- apsimo/task_queue/events.py +130 -0
- apsimo/task_queue/governor.py +1031 -0
- apsimo/task_queue/handlers/__init__.py +16 -0
- apsimo/task_queue/handlers/base.py +37 -0
- apsimo/task_queue/handlers/inference.py +640 -0
- apsimo/task_queue/handlers/monitoring.py +116 -0
- apsimo/task_queue/handlers/registry.py +75 -0
- apsimo/task_queue/handlers/subtask_handler.py +173 -0
- apsimo/task_queue/handlers/system_maintenance.py +147 -0
- apsimo/task_queue/mesh_integration.py +111 -0
- apsimo/task_queue/models.py +317 -0
- apsimo/task_queue/queue_manager.py +8286 -0
- apsimo/task_queue/routing.py +287 -0
- apsimo/task_queue/scheduler.py +252 -0
- apsimo/task_queue/schema.sql +197 -0
- apsimo/task_queue/work_control.py +342 -0
- apsimo/task_queue/worker.py +993 -0
- apsimo/telemetry.py +145 -0
- apsimo/tom/__init__.py +6 -0
- apsimo/tom/affect.py +387 -0
- apsimo/tom/approvals.py +171 -0
- apsimo/tom/arcs.py +896 -0
- apsimo/tom/asymmetry.py +131 -0
- apsimo/tom/eligibility.py +248 -0
- apsimo/tom/engagement.py +214 -0
- apsimo/tom/exposure.py +214 -0
- apsimo/tom/extractor.py +306 -0
- apsimo/tom/fact_adapters.py +144 -0
- apsimo/tom/facts.py +326 -0
- apsimo/tom/integration.py +592 -0
- apsimo/tom/leveled.py +118 -0
- apsimo/tom/levels.py +247 -0
- apsimo/tom/recipient_audit.py +995 -0
- apsimo/tom/recipient_simulator.py +593 -0
- apsimo/tom/source_lineage.py +93 -0
- apsimo/tom/tom2.py +277 -0
- apsimo/tom/visibility.py +559 -0
- apsimo/tom/visibility_store.py +414 -0
- apsimo/tools/__init__.py +0 -0
- apsimo/tools/definitions.py +740 -0
- apsimo/tools/handlers.py +943 -0
- apsimo/toolsmith/__init__.py +26 -0
- apsimo/toolsmith/authority.py +166 -0
- apsimo/toolsmith/engine.py +559 -0
- apsimo/toolsmith/integrity.py +100 -0
- apsimo/toolsmith/miner.py +145 -0
- apsimo/toolsmith/policy.py +110 -0
- apsimo/toolsmith/registry.py +635 -0
- apsimo/turns/__init__.py +17 -0
- apsimo/turns/audio.py +134 -0
- apsimo/turns/documents.py +235 -0
- apsimo/turns/executions.py +486 -0
- apsimo/turns/hermes_history.py +245 -0
- apsimo/turns/hermes_kanban.py +268 -0
- apsimo/turns/hermes_work.py +96 -0
- apsimo/turns/idempotency.py +752 -0
- apsimo/turns/local_work.py +115 -0
- apsimo/turns/media.py +581 -0
- apsimo/turns/reported_workers.py +196 -0
- apsimo/turns/source_annotations.py +283 -0
- apsimo/turns/source_attribution.py +154 -0
- apsimo/turns/source_read.py +351 -0
- apsimo/turns/source_vectors.py +263 -0
- apsimo/turns/video.py +210 -0
- apsimo/util/autonomy_preset.py +220 -0
- apsimo/util/instance.py +92 -0
- apsimo/util/model_output.py +25 -0
- apsimo/util/quiet_hours.py +27 -0
- apsimo/util/session_safety.py +37 -0
- apsimo/util/temporal.py +343 -0
- apsimo/vector/__init__.py +75 -0
- apsimo/vector/backfill.py +171 -0
- apsimo/vector/caption.py +114 -0
- apsimo/vector/collections.py +51 -0
- apsimo/vector/config.py +102 -0
- apsimo/vector/embedder.py +670 -0
- apsimo/vector/image_preprocess.py +406 -0
- apsimo/vector/image_store.py +296 -0
- apsimo/vector/indexes.py +162 -0
- apsimo/vector/migrate.py +334 -0
- apsimo/vector/multimodal_provider.py +417 -0
- apsimo/vector/multimodal_types.py +87 -0
- apsimo/vector/openai_provider.py +119 -0
- apsimo/vector/query.py +49 -0
- apsimo/vector/reranker.py +565 -0
- apsimo/vector/safety_image.py +159 -0
- apsimo/vector/scanner.py +197 -0
- apsimo/vector/setup.py +289 -0
- apsimo/vector/store.py +533 -0
- apsimo/vector/tiers.py +263 -0
- apsimo/work_orders.py +925 -0
- apsimo/workers/__init__.py +21 -0
- apsimo/workers/agent_bridge.py +640 -0
- apsimo/workers/colony_worker.py +382 -0
- apsimo/workers/queue_worker.py +441 -0
- apsimo/workers/skills_sync.py +152 -0
- apsimo/world_model/__init__.py +71 -0
- apsimo/world_model/causal_maintenance.py +131 -0
- apsimo/world_model/causal_policy.py +43 -0
- apsimo/world_model/causal_query.py +125 -0
- apsimo/world_model/confidence.py +54 -0
- apsimo/world_model/config.py +64 -0
- apsimo/world_model/constants.py +97 -0
- apsimo/world_model/entities.py +145 -0
- apsimo/world_model/expectation_resolvers.py +177 -0
- apsimo/world_model/extraction/__init__.py +7 -0
- apsimo/world_model/extraction/base.py +62 -0
- apsimo/world_model/extraction/conversation_extractor.py +262 -0
- apsimo/world_model/extraction/detector.py +74 -0
- apsimo/world_model/extraction/document_extractor.py +78 -0
- apsimo/world_model/extraction/formats/__init__.py +24 -0
- apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
- apsimo/world_model/extraction/formats/html_fmt.py +72 -0
- apsimo/world_model/extraction/formats/json_fmt.py +68 -0
- apsimo/world_model/extraction/formats/pdf.py +43 -0
- apsimo/world_model/extraction/formats/text.py +27 -0
- apsimo/world_model/extraction/llm_extractor.py +164 -0
- apsimo/world_model/extraction/pipeline.py +73 -0
- apsimo/world_model/integrations/__init__.py +5 -0
- apsimo/world_model/integrations/mind_model_bridge.py +115 -0
- apsimo/world_model/integrations/social_intel_bridge.py +120 -0
- apsimo/world_model/jobs/__init__.py +4 -0
- apsimo/world_model/jobs/extraction_job.py +168 -0
- apsimo/world_model/llm_extract.py +572 -0
- apsimo/world_model/neo4j/__init__.py +5 -0
- apsimo/world_model/neo4j/backend.py +654 -0
- apsimo/world_model/observations.py +155 -0
- apsimo/world_model/populator.py +307 -0
- apsimo/world_model/postgres/__init__.py +1 -0
- apsimo/world_model/postgres/backend.py +683 -0
- apsimo/world_model/relationships.py +25 -0
- apsimo/world_model/resolution/__init__.py +13 -0
- apsimo/world_model/resolution/entity_resolver.py +232 -0
- apsimo/world_model/resolution/merge_audit.py +16 -0
- apsimo/world_model/resolution/merge_workflow.py +117 -0
- apsimo/world_model/source_reports.py +121 -0
- apsimo/world_model/sqlite/__init__.py +4 -0
- apsimo/world_model/sqlite/backend.py +855 -0
- apsimo/world_model/sqlite/schema.sql +132 -0
- apsimo/world_model/store.py +545 -0
- apsimo-1.3.0.dist-info/METADATA +78 -0
- apsimo-1.3.0.dist-info/RECORD +614 -0
- apsimo-1.3.0.dist-info/WHEEL +5 -0
- apsimo-1.3.0.dist-info/entry_points.txt +11 -0
- apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
- apsimo-1.3.0.dist-info/top_level.txt +2 -0
- colony_sidecar/__init__.py +4 -0
apsimo/autonomy/loop.py
ADDED
|
@@ -0,0 +1,4316 @@
|
|
|
1
|
+
"""AutonomyLoop — Colony's continuous operating cycle.
|
|
2
|
+
|
|
3
|
+
Wires existing subsystems into a coherent tick-based loop that runs
|
|
4
|
+
as a background asyncio task inside the sidecar. Each tick drains
|
|
5
|
+
events, checks goals, runs cognition, and executes initiatives.
|
|
6
|
+
|
|
7
|
+
Design principle: wire what exists. The loop is pure glue — it
|
|
8
|
+
orchestrates subsystems that are already wired in the sidecar.
|
|
9
|
+
|
|
10
|
+
State lives in Neo4j + SQLite. The loop is stateless and restartable.
|
|
11
|
+
Kill it at any point and it picks up cleanly on restart.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
import asyncio
|
|
17
|
+
import contextlib
|
|
18
|
+
import copy
|
|
19
|
+
import functools
|
|
20
|
+
import hashlib
|
|
21
|
+
import json
|
|
22
|
+
import logging
|
|
23
|
+
import os
|
|
24
|
+
import time
|
|
25
|
+
import uuid
|
|
26
|
+
from dataclasses import dataclass, field
|
|
27
|
+
from datetime import datetime, timezone
|
|
28
|
+
from typing import Any, List, Optional
|
|
29
|
+
from zoneinfo import ZoneInfo
|
|
30
|
+
|
|
31
|
+
from apsimo.autonomy.config import AutonomyConfig, AutonomyMode
|
|
32
|
+
from apsimo.autonomy.registry import SubsystemRegistry
|
|
33
|
+
from apsimo.events.bus import EventBus
|
|
34
|
+
from apsimo.events.types import Event
|
|
35
|
+
|
|
36
|
+
# Lazy import to avoid circular dependency — broadcast_event is defined
|
|
37
|
+
# in the host router which imports from this module.
|
|
38
|
+
_broadcast = None
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def _get_broadcast():
|
|
42
|
+
global _broadcast
|
|
43
|
+
if _broadcast is None:
|
|
44
|
+
try:
|
|
45
|
+
from apsimo.api.routers.host import broadcast_event
|
|
46
|
+
_broadcast = broadcast_event
|
|
47
|
+
except ImportError:
|
|
48
|
+
# A circular import during startup must not permanently replace the
|
|
49
|
+
# durable publisher for the process lifetime.
|
|
50
|
+
return lambda _event: None
|
|
51
|
+
return _broadcast
|
|
52
|
+
|
|
53
|
+
logger = logging.getLogger(__name__)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True)
|
|
57
|
+
class _ProactiveFindingSnapshot:
|
|
58
|
+
"""Immutable, scalar-only finding projection used for feedback/logging."""
|
|
59
|
+
|
|
60
|
+
check: str
|
|
61
|
+
severity: str
|
|
62
|
+
reason: str
|
|
63
|
+
excerpt: Optional[str]
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
@dataclass(frozen=True)
|
|
67
|
+
class _ProactiveVerdictSnapshot:
|
|
68
|
+
"""One-read authority snapshot of a guard result.
|
|
69
|
+
|
|
70
|
+
A guard result may be an arbitrary object with descriptors rather than a
|
|
71
|
+
plain ``GuardResult``. The send boundary therefore reads every field it
|
|
72
|
+
uses exactly once, rejects non-built-in scalar types, and never consults
|
|
73
|
+
the source object again. Frozen scalar values close the validation/use
|
|
74
|
+
gap even for stateful or adversarial properties.
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
decision: Optional[str]
|
|
78
|
+
mode: Optional[str]
|
|
79
|
+
surface: Optional[str]
|
|
80
|
+
surface_family: Optional[str]
|
|
81
|
+
applicability: Optional[str]
|
|
82
|
+
guard_status: Optional[str]
|
|
83
|
+
policy_id: Optional[str]
|
|
84
|
+
policy_digest: Optional[str]
|
|
85
|
+
candidate_digest: Optional[str]
|
|
86
|
+
blocked: Optional[bool]
|
|
87
|
+
findings: tuple[_ProactiveFindingSnapshot, ...]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _exact_builtin_text(value: Any) -> Optional[str]:
|
|
91
|
+
return value if type(value) is str else None
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def _snapshot_proactive_findings(value: Any) -> tuple[_ProactiveFindingSnapshot, ...]:
|
|
95
|
+
# GuardResult.findings is a list. Reject custom iterables at this
|
|
96
|
+
# authority boundary instead of executing caller-controlled iteration.
|
|
97
|
+
if type(value) not in {list, tuple}:
|
|
98
|
+
return ()
|
|
99
|
+
snapshots = []
|
|
100
|
+
for finding in tuple(value)[:64]:
|
|
101
|
+
raw_check = getattr(finding, "check", None)
|
|
102
|
+
raw_severity = getattr(finding, "severity", None)
|
|
103
|
+
raw_reason = getattr(finding, "reason", None)
|
|
104
|
+
raw_excerpt = getattr(finding, "excerpt", None)
|
|
105
|
+
snapshots.append(_ProactiveFindingSnapshot(
|
|
106
|
+
check=_exact_builtin_text(raw_check) or "blocked",
|
|
107
|
+
severity=_exact_builtin_text(raw_severity) or "",
|
|
108
|
+
reason=_exact_builtin_text(raw_reason) or "",
|
|
109
|
+
excerpt=_exact_builtin_text(raw_excerpt),
|
|
110
|
+
))
|
|
111
|
+
return tuple(snapshots)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _snapshot_proactive_verdict(result: Any) -> _ProactiveVerdictSnapshot:
|
|
115
|
+
"""Read every egress-relevant raw verdict field exactly once."""
|
|
116
|
+
|
|
117
|
+
raw_decision = getattr(result, "decision", None)
|
|
118
|
+
raw_mode = getattr(result, "mode", None)
|
|
119
|
+
raw_surface = getattr(result, "surface", None)
|
|
120
|
+
raw_surface_family = getattr(result, "surface_family", None)
|
|
121
|
+
raw_applicability = getattr(result, "applicability", None)
|
|
122
|
+
raw_guard_status = getattr(result, "guard_status", None)
|
|
123
|
+
raw_policy_id = getattr(result, "policy_id", None)
|
|
124
|
+
raw_policy_digest = getattr(result, "policy_digest", None)
|
|
125
|
+
raw_candidate_digest = getattr(result, "candidate_digest", None)
|
|
126
|
+
raw_blocked = getattr(result, "blocked", None)
|
|
127
|
+
raw_findings = getattr(result, "findings", None)
|
|
128
|
+
return _ProactiveVerdictSnapshot(
|
|
129
|
+
decision=_exact_builtin_text(raw_decision),
|
|
130
|
+
mode=_exact_builtin_text(raw_mode),
|
|
131
|
+
surface=_exact_builtin_text(raw_surface),
|
|
132
|
+
surface_family=_exact_builtin_text(raw_surface_family),
|
|
133
|
+
applicability=_exact_builtin_text(raw_applicability),
|
|
134
|
+
guard_status=_exact_builtin_text(raw_guard_status),
|
|
135
|
+
policy_id=_exact_builtin_text(raw_policy_id),
|
|
136
|
+
policy_digest=_exact_builtin_text(raw_policy_digest),
|
|
137
|
+
candidate_digest=_exact_builtin_text(raw_candidate_digest),
|
|
138
|
+
blocked=raw_blocked if type(raw_blocked) is bool else None,
|
|
139
|
+
findings=_snapshot_proactive_findings(raw_findings),
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def _proactive_enforce_verdict_error(
|
|
144
|
+
result: _ProactiveVerdictSnapshot,
|
|
145
|
+
candidate: str,
|
|
146
|
+
) -> Optional[str]:
|
|
147
|
+
"""Return why an enforce verdict cannot authorize this exact candidate.
|
|
148
|
+
|
|
149
|
+
The guard object is an in-process dependency, but its result still crosses
|
|
150
|
+
an authority boundary: only a canonical policy decision over the exact
|
|
151
|
+
UTF-8 candidate may reach a delivery adapter. Keep this validation local
|
|
152
|
+
to the send path so a malformed result object cannot inherit authority from
|
|
153
|
+
a truthy/falsey ``blocked`` attribute.
|
|
154
|
+
"""
|
|
155
|
+
|
|
156
|
+
from apsimo.gate.response_guard import (
|
|
157
|
+
GuardDecision,
|
|
158
|
+
response_text_digest,
|
|
159
|
+
)
|
|
160
|
+
from apsimo.gate.surface_policy import ResponseGuardSurfacePolicyV1
|
|
161
|
+
|
|
162
|
+
expected = ResponseGuardSurfacePolicyV1().resolve(
|
|
163
|
+
"proactive_text",
|
|
164
|
+
configured_mode="enforce",
|
|
165
|
+
requested_mode="enforce",
|
|
166
|
+
)
|
|
167
|
+
decision = result.decision
|
|
168
|
+
valid_decisions = {item.value for item in GuardDecision}
|
|
169
|
+
if type(decision) is not str or decision not in valid_decisions:
|
|
170
|
+
return "decision is absent or invalid"
|
|
171
|
+
|
|
172
|
+
guard_status = result.guard_status
|
|
173
|
+
if type(guard_status) is not str or guard_status not in {
|
|
174
|
+
"evaluated",
|
|
175
|
+
"degraded",
|
|
176
|
+
}:
|
|
177
|
+
return "guard_status is absent or invalid"
|
|
178
|
+
if (
|
|
179
|
+
guard_status == "degraded"
|
|
180
|
+
and decision != GuardDecision.BLOCK.value
|
|
181
|
+
):
|
|
182
|
+
return "degraded enforce verdict is not a block"
|
|
183
|
+
|
|
184
|
+
expected_fields = {
|
|
185
|
+
"mode": expected.effective_mode,
|
|
186
|
+
"surface": expected.surface,
|
|
187
|
+
"surface_family": expected.family,
|
|
188
|
+
"applicability": expected.disposition,
|
|
189
|
+
"policy_id": expected.policy_id,
|
|
190
|
+
"policy_digest": expected.policy_digest,
|
|
191
|
+
"candidate_digest": response_text_digest(candidate),
|
|
192
|
+
}
|
|
193
|
+
for field_name, expected_value in expected_fields.items():
|
|
194
|
+
observed = getattr(result, field_name)
|
|
195
|
+
if type(observed) is not str or observed != expected_value:
|
|
196
|
+
return "%s does not match the canonical proactive policy" % field_name
|
|
197
|
+
|
|
198
|
+
blocked = result.blocked
|
|
199
|
+
if type(blocked) is not bool or blocked != (
|
|
200
|
+
decision != GuardDecision.ALLOW.value
|
|
201
|
+
):
|
|
202
|
+
return "blocked projection conflicts with decision"
|
|
203
|
+
return None
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _record_p3_thinker_candidates(workspace: Any, initiatives: list[Any]) -> int:
|
|
207
|
+
"""Record SelfDirectedThinker output as shadow provenance, never live.
|
|
208
|
+
|
|
209
|
+
Workspace mode describes the scheduler consuming a concern. It cannot
|
|
210
|
+
relabel the trust/mode of the producer that originated the candidate.
|
|
211
|
+
"""
|
|
212
|
+
|
|
213
|
+
recorded = 0
|
|
214
|
+
for init in initiatives:
|
|
215
|
+
key = str(getattr(init, "dedup_key", "") or "").strip()
|
|
216
|
+
if not key:
|
|
217
|
+
import hashlib
|
|
218
|
+
key = hashlib.sha256(
|
|
219
|
+
str(getattr(init, "description", init)).encode("utf-8")
|
|
220
|
+
).hexdigest()[:24]
|
|
221
|
+
workspace.bump(
|
|
222
|
+
kind="question",
|
|
223
|
+
summary=str(getattr(init, "description", init))[:300],
|
|
224
|
+
dedup_key=f"self-directed:{key}"[:200],
|
|
225
|
+
salience=max(0.3, min(0.85, float(
|
|
226
|
+
getattr(init, "priority", 0.5) or 0.5))),
|
|
227
|
+
sources=[f"self-directed-thinker:{key}"],
|
|
228
|
+
producer_name="self_directed_thinker",
|
|
229
|
+
producer_mode="shadow",
|
|
230
|
+
producer_revision="self-directed-thinker:v1",
|
|
231
|
+
)
|
|
232
|
+
recorded += 1
|
|
233
|
+
return recorded
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
@dataclass
|
|
237
|
+
class LoopStats:
|
|
238
|
+
"""Lightweight counters updated each tick for observability."""
|
|
239
|
+
|
|
240
|
+
ticks: int = 0
|
|
241
|
+
events_processed: int = 0
|
|
242
|
+
goals_checked: int = 0
|
|
243
|
+
initiatives_generated: int = 0
|
|
244
|
+
actions_executed: int = 0
|
|
245
|
+
errors: int = 0
|
|
246
|
+
actions_this_hour: int = 0
|
|
247
|
+
hour_bucket: int = field(default_factory=lambda: datetime.now(timezone.utc).hour)
|
|
248
|
+
skills_loaded: int = 0
|
|
249
|
+
skills_evicted: int = 0
|
|
250
|
+
signals_collected: int = 0
|
|
251
|
+
scoring_runs: int = 0
|
|
252
|
+
tier_changes: int = 0
|
|
253
|
+
memories_promoted: int = 0
|
|
254
|
+
task_follow_ups: int = 0
|
|
255
|
+
scheduled_runs: int = 0
|
|
256
|
+
phases_skipped: int = 0
|
|
257
|
+
boundary_check_errors: int = 0
|
|
258
|
+
phases_cancelled: int = 0
|
|
259
|
+
last_cancelled_phase: Optional[str] = None
|
|
260
|
+
|
|
261
|
+
def as_dict(self) -> dict:
|
|
262
|
+
return {
|
|
263
|
+
"ticks": self.ticks,
|
|
264
|
+
"events_processed": self.events_processed,
|
|
265
|
+
"goals_checked": self.goals_checked,
|
|
266
|
+
"initiatives_generated": self.initiatives_generated,
|
|
267
|
+
"actions_executed": self.actions_executed,
|
|
268
|
+
"errors": self.errors,
|
|
269
|
+
"actions_this_hour": self.actions_this_hour,
|
|
270
|
+
"hour_bucket": self.hour_bucket,
|
|
271
|
+
"skills_loaded": self.skills_loaded,
|
|
272
|
+
"skills_evicted": self.skills_evicted,
|
|
273
|
+
"signals_collected": self.signals_collected,
|
|
274
|
+
"scoring_runs": self.scoring_runs,
|
|
275
|
+
"tier_changes": self.tier_changes,
|
|
276
|
+
"memories_promoted": self.memories_promoted,
|
|
277
|
+
"task_follow_ups": self.task_follow_ups,
|
|
278
|
+
"scheduled_runs": self.scheduled_runs,
|
|
279
|
+
"phases_skipped": self.phases_skipped,
|
|
280
|
+
"boundary_check_errors": self.boundary_check_errors,
|
|
281
|
+
"phases_cancelled": self.phases_cancelled,
|
|
282
|
+
"last_cancelled_phase": self.last_cancelled_phase,
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
|
|
286
|
+
class AutonomyLoop:
|
|
287
|
+
"""Colony's continuous operating cycle.
|
|
288
|
+
|
|
289
|
+
Takes a SubsystemRegistry and AutonomyConfig. The registry provides
|
|
290
|
+
lazy access to all wired subsystems — if something isn't wired,
|
|
291
|
+
the corresponding phase is a no-op.
|
|
292
|
+
|
|
293
|
+
The loop does NOT auto-start. The host calls ``start()`` or uses
|
|
294
|
+
the ``/v1/host/autonomy/start`` API endpoint.
|
|
295
|
+
|
|
296
|
+
Each tick:
|
|
297
|
+
1. Drain pending events
|
|
298
|
+
2. Check goal engine for goals needing attention
|
|
299
|
+
3. Check anomaly detections above severity threshold
|
|
300
|
+
4. Run initiative engine — Colony decides whether to act
|
|
301
|
+
5. Execute approved actions
|
|
302
|
+
6. Run cognition pipeline tick
|
|
303
|
+
7. (memory consolidation runs via the memory_consolidate scheduler
|
|
304
|
+
task, not a tick phase)
|
|
305
|
+
8. Memory decay (daily)
|
|
306
|
+
9. Memory pruning (weekly)
|
|
307
|
+
10. Task completion follow-ups
|
|
308
|
+
11. Frustration back-off update
|
|
309
|
+
12. Bootstrap self-check (daily)
|
|
310
|
+
13. Self-reflection (weekly)
|
|
311
|
+
14. Relationship scoring
|
|
312
|
+
15. Synthesis (connection discovery)
|
|
313
|
+
16. Skill trigger evaluation + eviction
|
|
314
|
+
17. Sleep until next tick or event wakes early
|
|
315
|
+
"""
|
|
316
|
+
|
|
317
|
+
def __init__(
|
|
318
|
+
self,
|
|
319
|
+
registry: SubsystemRegistry,
|
|
320
|
+
config: Optional[AutonomyConfig] = None,
|
|
321
|
+
event_bus: Optional[EventBus] = None,
|
|
322
|
+
scheduler: Any = None,
|
|
323
|
+
) -> None:
|
|
324
|
+
self._registry = registry
|
|
325
|
+
self.config = config or AutonomyConfig()
|
|
326
|
+
for phase in self.config.enabled_phases or ():
|
|
327
|
+
if not callable(getattr(self, '_phase_' + phase, None)):
|
|
328
|
+
raise ValueError(f'Unknown autonomy phase: {phase}')
|
|
329
|
+
self.events = event_bus or EventBus()
|
|
330
|
+
self.stats = LoopStats()
|
|
331
|
+
self._scheduler = scheduler
|
|
332
|
+
|
|
333
|
+
self._running = False
|
|
334
|
+
self._wake_event = asyncio.Event()
|
|
335
|
+
self._stop_event = asyncio.Event()
|
|
336
|
+
self._wake_sub: Any = None
|
|
337
|
+
self._pending_initiatives: List[Any] = []
|
|
338
|
+
# Per-domain timestamps of the last observation-sync request, so
|
|
339
|
+
# a slow agent isn't spammed with duplicate sync jobs every tick.
|
|
340
|
+
self._last_sync_request: dict = {}
|
|
341
|
+
self._periodic_last: dict = {}
|
|
342
|
+
self._last_task_completion_check: Optional[datetime] = None
|
|
343
|
+
# Phases already warned about skipping (warn once, count always).
|
|
344
|
+
self._phase_skip_warned: set = set()
|
|
345
|
+
# Which tick phase is running right now (None between ticks), and how
|
|
346
|
+
# long each phase took on its last run and at its worst. A tick that
|
|
347
|
+
# blows its budget is cancelled as a whole, so without this the log
|
|
348
|
+
# could not say which phase ate the budget.
|
|
349
|
+
self._current_phase: Optional[str] = None
|
|
350
|
+
self._phase_seconds: dict[str, float] = {}
|
|
351
|
+
self._phase_seconds_max: dict[str, float] = {}
|
|
352
|
+
# High-water mark for _phase_events so each event is counted once.
|
|
353
|
+
self._last_event_seen_id: Optional[str] = None
|
|
354
|
+
# Exact already-admitted requests awaiting a terminal gateway outcome.
|
|
355
|
+
# This is intentionally bounded and in-memory; the initiative store +
|
|
356
|
+
# startup re-push rebuild it after a proactive-mode restart.
|
|
357
|
+
self._governed_delivery_replays: dict[str, dict] = {}
|
|
358
|
+
self._governed_reconcile_task: Optional[asyncio.Task] = None
|
|
359
|
+
|
|
360
|
+
# ------------------------------------------------------------------
|
|
361
|
+
# Lifecycle
|
|
362
|
+
# ------------------------------------------------------------------
|
|
363
|
+
|
|
364
|
+
async def start(self) -> None:
|
|
365
|
+
"""Start the autonomy loop. Runs until stop() is called."""
|
|
366
|
+
self._running = True
|
|
367
|
+
self._stop_event.clear()
|
|
368
|
+
|
|
369
|
+
# Fail loudly at startup if the owner identity is missing or
|
|
370
|
+
# unresolvable (v0.16.0). Relationship generation fails closed at
|
|
371
|
+
# tick time either way; this surfaces the misconfiguration once,
|
|
372
|
+
# at CRITICAL, instead of letting it hide in per-tick noise.
|
|
373
|
+
try:
|
|
374
|
+
from apsimo.identity.resolver import (
|
|
375
|
+
OwnerIdentityError,
|
|
376
|
+
get_identity_resolver,
|
|
377
|
+
)
|
|
378
|
+
await get_identity_resolver().owner_identities()
|
|
379
|
+
except OwnerIdentityError as exc:
|
|
380
|
+
logger.critical(
|
|
381
|
+
"OWNER IDENTITY NOT RESOLVED — relationship initiative "
|
|
382
|
+
"generation will be disabled until fixed: %s", exc,
|
|
383
|
+
)
|
|
384
|
+
except Exception as exc:
|
|
385
|
+
logger.warning("Owner identity startup check failed: %s", exc)
|
|
386
|
+
|
|
387
|
+
# Boot self-check: a periodic phase that dispatches on a graph
|
|
388
|
+
# capability which does not exist would otherwise no-op silently
|
|
389
|
+
# forever (exactly how the old consolidation and pruning phases went
|
|
390
|
+
# dead for months). Surface any such mismatch once, loudly, at start.
|
|
391
|
+
self._check_phase_capabilities()
|
|
392
|
+
|
|
393
|
+
# Governed message lifecycle reconciliation is independent of the
|
|
394
|
+
# main autonomy cadence. It must also run in REACTIVE mode: an already
|
|
395
|
+
# admitted message can reach provider delivery while no new event or
|
|
396
|
+
# initiative wakes the ordinary loop.
|
|
397
|
+
self._start_governed_delivery_reconciler()
|
|
398
|
+
|
|
399
|
+
# Reactive mode: just mark as running, no timer
|
|
400
|
+
if self.config.mode == AutonomyMode.REACTIVE:
|
|
401
|
+
logger.info(
|
|
402
|
+
"Autonomy loop started in REACTIVE mode (on-demand only, tz=%s)",
|
|
403
|
+
self.config.timezone,
|
|
404
|
+
)
|
|
405
|
+
return
|
|
406
|
+
|
|
407
|
+
# Proactive mode: start timer loop
|
|
408
|
+
logger.info(
|
|
409
|
+
"Autonomy loop starting in PROACTIVE mode (tick=%.0fs, quiet=%s-%s %s)",
|
|
410
|
+
self.config.tick_interval_secs,
|
|
411
|
+
self.config.quiet_hours_start,
|
|
412
|
+
self.config.quiet_hours_end,
|
|
413
|
+
self.config.timezone,
|
|
414
|
+
)
|
|
415
|
+
|
|
416
|
+
self._wake_sub = self.events.subscribe(
|
|
417
|
+
handler=self._on_wake_signal,
|
|
418
|
+
event_types=[Event],
|
|
419
|
+
)
|
|
420
|
+
|
|
421
|
+
try:
|
|
422
|
+
while not self._stop_event.is_set():
|
|
423
|
+
# Bound the whole tick so no single slow/hung phase (a wedged
|
|
424
|
+
# LLM, graph, or Docker call) can freeze the loop forever. The
|
|
425
|
+
# phases are re-run each tick, so a cancelled tick is safe; the
|
|
426
|
+
# loop advances and retries next cycle.
|
|
427
|
+
try:
|
|
428
|
+
await asyncio.wait_for(
|
|
429
|
+
self._tick(), timeout=self._tick_budget_secs())
|
|
430
|
+
except asyncio.TimeoutError:
|
|
431
|
+
self._note_tick_cancelled()
|
|
432
|
+
await self._sleep_until_next_tick()
|
|
433
|
+
finally:
|
|
434
|
+
await self._stop_governed_delivery_reconciler()
|
|
435
|
+
if self._wake_sub is not None:
|
|
436
|
+
self.events.unsubscribe(self._wake_sub)
|
|
437
|
+
self._running = False
|
|
438
|
+
logger.info("Autonomy loop stopped. Stats: %s", self.stats.as_dict())
|
|
439
|
+
|
|
440
|
+
async def stop(self) -> None:
|
|
441
|
+
"""Signal the loop to stop after the current tick completes."""
|
|
442
|
+
logger.info("Autonomy loop stop requested")
|
|
443
|
+
self._stop_event.set()
|
|
444
|
+
self._wake_event.set()
|
|
445
|
+
await self._stop_governed_delivery_reconciler()
|
|
446
|
+
|
|
447
|
+
def _start_governed_delivery_reconciler(self) -> None:
|
|
448
|
+
delivery = getattr(self._registry, "delivery", None)
|
|
449
|
+
enabled = False
|
|
450
|
+
if delivery is not None and hasattr(
|
|
451
|
+
delivery, "governed_gateway_admission_enabled"
|
|
452
|
+
):
|
|
453
|
+
try:
|
|
454
|
+
enabled = delivery.governed_gateway_admission_enabled() is True
|
|
455
|
+
except Exception:
|
|
456
|
+
enabled = False
|
|
457
|
+
if not enabled:
|
|
458
|
+
return
|
|
459
|
+
task = self._governed_reconcile_task
|
|
460
|
+
if task is None or task.done():
|
|
461
|
+
self._governed_reconcile_task = asyncio.create_task(
|
|
462
|
+
self._governed_delivery_reconciliation_loop(),
|
|
463
|
+
name="colony-governed-delivery-reconciliation",
|
|
464
|
+
)
|
|
465
|
+
|
|
466
|
+
async def _stop_governed_delivery_reconciler(self) -> None:
|
|
467
|
+
task = self._governed_reconcile_task
|
|
468
|
+
self._governed_reconcile_task = None
|
|
469
|
+
if task is None or task.done():
|
|
470
|
+
return
|
|
471
|
+
task.cancel()
|
|
472
|
+
try:
|
|
473
|
+
await task
|
|
474
|
+
except asyncio.CancelledError:
|
|
475
|
+
pass
|
|
476
|
+
|
|
477
|
+
async def _governed_delivery_reconciliation_loop(self) -> None:
|
|
478
|
+
"""Poll admitted requests until the boundary reports a terminal state."""
|
|
479
|
+
|
|
480
|
+
while not self._stop_event.is_set():
|
|
481
|
+
delivery = getattr(self._registry, "delivery", None)
|
|
482
|
+
interval = 5.0
|
|
483
|
+
getter = getattr(delivery, "governed_gateway_poll_seconds", None)
|
|
484
|
+
if callable(getter):
|
|
485
|
+
try:
|
|
486
|
+
interval = max(0.01, min(300.0, float(getter())))
|
|
487
|
+
except (TypeError, ValueError):
|
|
488
|
+
interval = 5.0
|
|
489
|
+
try:
|
|
490
|
+
await asyncio.wait_for(self._stop_event.wait(), timeout=interval)
|
|
491
|
+
continue
|
|
492
|
+
except asyncio.TimeoutError:
|
|
493
|
+
pass
|
|
494
|
+
try:
|
|
495
|
+
await self._phase_governed_delivery_reconciliation()
|
|
496
|
+
except asyncio.CancelledError:
|
|
497
|
+
raise
|
|
498
|
+
except Exception:
|
|
499
|
+
self.stats.errors += 1
|
|
500
|
+
logger.exception("Governed delivery reconciliation failed")
|
|
501
|
+
|
|
502
|
+
def wake(self) -> None:
|
|
503
|
+
"""Wake the loop early from its sleep. Thread-safe."""
|
|
504
|
+
self._wake_event.set()
|
|
505
|
+
|
|
506
|
+
@property
|
|
507
|
+
def is_running(self) -> bool:
|
|
508
|
+
return self._running
|
|
509
|
+
|
|
510
|
+
# ------------------------------------------------------------------
|
|
511
|
+
# Main tick
|
|
512
|
+
# ------------------------------------------------------------------
|
|
513
|
+
|
|
514
|
+
async def _tick(self) -> None:
|
|
515
|
+
"""One autonomy tick. The running-phase marker is cleared on every
|
|
516
|
+
exit except cancellation, where _note_tick_cancelled reads it."""
|
|
517
|
+
try:
|
|
518
|
+
await self._tick_phases()
|
|
519
|
+
except asyncio.CancelledError:
|
|
520
|
+
raise
|
|
521
|
+
except BaseException:
|
|
522
|
+
self._current_phase = None
|
|
523
|
+
raise
|
|
524
|
+
else:
|
|
525
|
+
self._current_phase = None
|
|
526
|
+
|
|
527
|
+
async def _tick_phases(self) -> None:
|
|
528
|
+
self.stats.ticks += 1
|
|
529
|
+
self._reset_hour_bucket()
|
|
530
|
+
tick_start = datetime.now(timezone.utc)
|
|
531
|
+
|
|
532
|
+
logger.debug("Tick #%d starting", self.stats.ticks)
|
|
533
|
+
|
|
534
|
+
# Phase -1: reconcile terminal WorkOrder truth before any phase that
|
|
535
|
+
# may invoke an LLM. This is bounded separately and never dispatches
|
|
536
|
+
# work, so a slow thinking/planning phase (or whole-tick cancellation)
|
|
537
|
+
# cannot starve durable queue-result projection.
|
|
538
|
+
await self._run_phase("project_result_reconciliation", self._phase_project_result_reconciliation())
|
|
539
|
+
|
|
540
|
+
# Phase 0: evaluate skill triggers
|
|
541
|
+
event_text = self._gather_event_text()
|
|
542
|
+
await self._run_phase("skill_triggers", self._phase_skill_triggers(event_text))
|
|
543
|
+
|
|
544
|
+
# Phase 1: drain pending events
|
|
545
|
+
await self._run_phase("events", self._phase_events())
|
|
546
|
+
|
|
547
|
+
# Phase 2: check goals needing attention
|
|
548
|
+
await self._run_phase("goals", self._phase_goals())
|
|
549
|
+
|
|
550
|
+
# Phase 2b: system-condition sweep (hourly) — commitment overdue flip,
|
|
551
|
+
# affect decline, surprise accumulation
|
|
552
|
+
await self._run_phase("condition_checks", self._phase_condition_checks())
|
|
553
|
+
|
|
554
|
+
# Phase 3: check anomalies
|
|
555
|
+
await self._run_phase("anomalies", self._phase_anomalies())
|
|
556
|
+
|
|
557
|
+
# Phase 4: scheduled periodic tasks (memory consolidate, briefing, etc.)
|
|
558
|
+
await self._run_phase("scheduled", self._phase_scheduled())
|
|
559
|
+
|
|
560
|
+
# Phase 5: run initiative engine
|
|
561
|
+
await self._run_phase("initiative", self._phase_initiative())
|
|
562
|
+
|
|
563
|
+
# Phase 5b: self-directed thinking (v0.17.0) — novel work the
|
|
564
|
+
# data-reactive generators can't see. Appends to the same
|
|
565
|
+
# pending-initiative batch Phase 6 consumes.
|
|
566
|
+
await self._run_phase("thinking", self._phase_thinking())
|
|
567
|
+
|
|
568
|
+
# Phase 6: execute approved actions
|
|
569
|
+
await self._run_phase("execute", self._phase_execute())
|
|
570
|
+
|
|
571
|
+
# Phase 6a: sustained project pursuit (cognition item 1)
|
|
572
|
+
await self._run_phase("projects", self._phase_projects())
|
|
573
|
+
|
|
574
|
+
# Phase 6a2: trust-engine graduation/demotion notices (Amendment 1)
|
|
575
|
+
await self._run_phase("trust_notices", self._phase_trust_notices())
|
|
576
|
+
|
|
577
|
+
# Phase 6b: request fresh observations for stale domains (v0.16.0)
|
|
578
|
+
await self._run_phase("observation_sync", self._phase_observation_sync())
|
|
579
|
+
|
|
580
|
+
# Phase 6c: feed completed agent work back into memory (v0.17.0)
|
|
581
|
+
await self._run_phase("job_writeback", self._phase_job_writeback())
|
|
582
|
+
|
|
583
|
+
# Phase 7: cognition pipeline tick
|
|
584
|
+
await self._run_phase("cognition", self._phase_cognition())
|
|
585
|
+
|
|
586
|
+
# (memory consolidation is NOT a tick phase: the consolidator runs
|
|
587
|
+
# hourly via the memory_consolidate scheduler task. The old phase
|
|
588
|
+
# here dispatched on graph.consolidate_memories, which never
|
|
589
|
+
# existed — deleted rather than repointed to avoid a double-run.)
|
|
590
|
+
|
|
591
|
+
# Phase 9: memory decay (daily)
|
|
592
|
+
await self._run_phase("memory_decay", self._phase_memory_decay())
|
|
593
|
+
|
|
594
|
+
# Phase 10: memory reconciliation (daily)
|
|
595
|
+
await self._run_phase("memory_reconciliation", self._phase_memory_reconciliation())
|
|
596
|
+
|
|
597
|
+
# Phase 11: memory pruning (weekly)
|
|
598
|
+
await self._run_phase("memory_pruning", self._phase_memory_pruning())
|
|
599
|
+
|
|
600
|
+
# Phase 11b: memory distillation — promote recalled episodics into semantic facts (daily)
|
|
601
|
+
await self._run_phase("memory_distillation", self._phase_memory_distillation())
|
|
602
|
+
|
|
603
|
+
# Phase 12: memory archive (weekly)
|
|
604
|
+
await self._run_phase("memory_archive", self._phase_memory_archive())
|
|
605
|
+
|
|
606
|
+
# Phase 11c: belief maintenance (daily, cognition item 7)
|
|
607
|
+
await self._run_phase("belief_maintenance", self._phase_belief_maintenance())
|
|
608
|
+
|
|
609
|
+
# Phase 11d: LLM-assisted world-model extraction (daily, batch)
|
|
610
|
+
await self._run_phase("world_llm_extract", self._phase_world_llm_extract())
|
|
611
|
+
|
|
612
|
+
# Phase 11d2: tom2 knowledge asymmetry (daily; COLONY_TOM2, default off)
|
|
613
|
+
await self._run_phase("tom2_asymmetry", self._phase_tom2_asymmetry())
|
|
614
|
+
|
|
615
|
+
# Phase 11e: connector ingest (read-only pull senses, cognition item 2)
|
|
616
|
+
await self._run_phase("connectors", self._phase_connectors())
|
|
617
|
+
|
|
618
|
+
# Phase 11f: relationship profiling (standing/psyche/approach briefs)
|
|
619
|
+
await self._run_phase("relationship_profiling", self._phase_relationship_profiling())
|
|
620
|
+
|
|
621
|
+
# Phase 12: task completion follow-ups
|
|
622
|
+
# (memory distillation already ran as Phase 11b — it was called twice)
|
|
623
|
+
await self._run_phase("task_completion", self._phase_task_completion())
|
|
624
|
+
|
|
625
|
+
# Phase 13: frustration back-off
|
|
626
|
+
await self._run_phase("frustration_update", self._phase_frustration_update())
|
|
627
|
+
|
|
628
|
+
# Phase 14: relationship scoring
|
|
629
|
+
await self._run_phase("relationships", self._phase_relationships())
|
|
630
|
+
|
|
631
|
+
# Phase 15: synthesis
|
|
632
|
+
await self._run_phase("synthesis", self._phase_synthesis())
|
|
633
|
+
|
|
634
|
+
# Phase 16: bootstrap self-check (daily)
|
|
635
|
+
await self._run_phase("bootstrap_check", self._phase_bootstrap_check())
|
|
636
|
+
|
|
637
|
+
# Phase 17: self-reflection (weekly)
|
|
638
|
+
await self._run_phase("self_reflection", self._phase_self_reflection())
|
|
639
|
+
|
|
640
|
+
# Phase 17b: selfhood benchmark (weekly, Mind M0a)
|
|
641
|
+
await self._run_phase("selfhood_benchmark", self._phase_selfhood_benchmark())
|
|
642
|
+
|
|
643
|
+
# Phase 17c: experiment decisions (daily, Mind M0b)
|
|
644
|
+
await self._run_phase("experiments", self._phase_experiments())
|
|
645
|
+
|
|
646
|
+
# Phase 17d: toolsmith (daily, Mind M1)
|
|
647
|
+
await self._run_phase("toolsmith", self._phase_toolsmith())
|
|
648
|
+
|
|
649
|
+
# Phase 18: skill eviction
|
|
650
|
+
await self._run_phase("skill_evict", self._phase_skill_evict())
|
|
651
|
+
|
|
652
|
+
# === Multi-Agent Phases (v0.7.0) ===
|
|
653
|
+
|
|
654
|
+
# Phase 19: startup re-push (first tick only)
|
|
655
|
+
await self._run_phase("startup_repush", self._phase_startup_repush())
|
|
656
|
+
|
|
657
|
+
# Phase 20: agent heartbeat (every tick)
|
|
658
|
+
await self._run_phase("agent_heartbeat", self._phase_agent_heartbeat())
|
|
659
|
+
|
|
660
|
+
# Phase 21: initiative timeout (every tick)
|
|
661
|
+
await self._run_phase("initiative_timeout", self._phase_initiative_timeout())
|
|
662
|
+
|
|
663
|
+
# Phase 21b: owner-approval timeout for blocked jobs (every tick)
|
|
664
|
+
await self._run_phase("approval_timeout", self._phase_approval_timeout())
|
|
665
|
+
|
|
666
|
+
# Phase 21: stale initiative cleanup (every 5 ticks)
|
|
667
|
+
if self.stats.ticks % 5 == 0:
|
|
668
|
+
await self._run_phase("stale_initiative_cleanup", self._phase_stale_initiative_cleanup())
|
|
669
|
+
|
|
670
|
+
# Phase 22: ghost agent cleanup (every 10 ticks)
|
|
671
|
+
if self.stats.ticks % 10 == 0:
|
|
672
|
+
await self._run_phase("ghost_cleanup", self._phase_ghost_cleanup())
|
|
673
|
+
|
|
674
|
+
# Phase 22b: cognitive workspace (every 3 ticks, Mind M2)
|
|
675
|
+
if self.stats.ticks % 3 == 0:
|
|
676
|
+
await self._run_phase("workspace", self._phase_workspace())
|
|
677
|
+
|
|
678
|
+
# Phase 22c: expectations — predict + check (every 2 ticks, Mind M3a)
|
|
679
|
+
if self.stats.ticks % 2 == 0:
|
|
680
|
+
await self._run_phase("expectations", self._phase_expectations())
|
|
681
|
+
|
|
682
|
+
# Phase 23: database backup (every 100 ticks)
|
|
683
|
+
if self.stats.ticks % 100 == 0:
|
|
684
|
+
await self._run_phase("database_backup", self._phase_database_backup())
|
|
685
|
+
|
|
686
|
+
# Telemetry liveness stamp — at the END of the tick, so last_tick_at
|
|
687
|
+
# means "a tick actually completed". Stamping at the top made a tick
|
|
688
|
+
# whose phases all threw (or that was cancelled on its budget) report
|
|
689
|
+
# fresh forever.
|
|
690
|
+
await self._run_phase("telemetry", self._phase_telemetry())
|
|
691
|
+
|
|
692
|
+
elapsed = (datetime.now(timezone.utc) - tick_start).total_seconds()
|
|
693
|
+
logger.debug("Tick #%d complete in %.2fs", self.stats.ticks, elapsed)
|
|
694
|
+
|
|
695
|
+
async def _phase_telemetry(self) -> None:
|
|
696
|
+
try:
|
|
697
|
+
from apsimo.api.routers.host import _telemetry
|
|
698
|
+
if _telemetry is not None:
|
|
699
|
+
await _telemetry.touch("last_tick_at")
|
|
700
|
+
except Exception:
|
|
701
|
+
logger.warning("Telemetry touch failed (non-critical)")
|
|
702
|
+
|
|
703
|
+
# ------------------------------------------------------------------
|
|
704
|
+
# Phase implementations
|
|
705
|
+
# ------------------------------------------------------------------
|
|
706
|
+
|
|
707
|
+
async def _phase_events(self) -> None:
|
|
708
|
+
"""Reduce the durable host journal, with the private bus as fallback.
|
|
709
|
+
|
|
710
|
+
When the event-to-concern reducer is enabled its persisted cursor is
|
|
711
|
+
authoritative across restarts. The old in-memory history remains only
|
|
712
|
+
for deployments that have not enabled the migration flag.
|
|
713
|
+
"""
|
|
714
|
+
try:
|
|
715
|
+
registry = getattr(self, "_registry", None)
|
|
716
|
+
workspace = getattr(registry, "workspace", None) if registry is not None else None
|
|
717
|
+
external_reducer = (
|
|
718
|
+
getattr(workspace, "external_event_reducer", None)
|
|
719
|
+
if workspace else None
|
|
720
|
+
)
|
|
721
|
+
if (
|
|
722
|
+
external_reducer is not None
|
|
723
|
+
and getattr(external_reducer, "mode", "off") != "off"
|
|
724
|
+
):
|
|
725
|
+
try:
|
|
726
|
+
result = external_reducer.run_once(limit=100)
|
|
727
|
+
self.stats.events_processed += int(
|
|
728
|
+
result.get("processed") or 0
|
|
729
|
+
)
|
|
730
|
+
if result.get("error"):
|
|
731
|
+
self.stats.errors += 1
|
|
732
|
+
logger.warning(
|
|
733
|
+
"External event concern reducer paused: %s",
|
|
734
|
+
result["error"],
|
|
735
|
+
)
|
|
736
|
+
except Exception:
|
|
737
|
+
self.stats.errors += 1
|
|
738
|
+
logger.exception(
|
|
739
|
+
"External event concern reducer failed this tick"
|
|
740
|
+
)
|
|
741
|
+
|
|
742
|
+
turn_reducer = (
|
|
743
|
+
getattr(workspace, "turn_event_reducer", None)
|
|
744
|
+
if workspace else None
|
|
745
|
+
)
|
|
746
|
+
if (
|
|
747
|
+
turn_reducer is not None
|
|
748
|
+
and getattr(turn_reducer, "mode", "off") != "off"
|
|
749
|
+
):
|
|
750
|
+
try:
|
|
751
|
+
result = turn_reducer.run_once(limit=100)
|
|
752
|
+
self.stats.events_processed += int(
|
|
753
|
+
result.get("processed") or 0
|
|
754
|
+
)
|
|
755
|
+
if result.get("error"):
|
|
756
|
+
self.stats.errors += 1
|
|
757
|
+
logger.warning(
|
|
758
|
+
"Conversation turn concern reducer paused: %s",
|
|
759
|
+
result["error"],
|
|
760
|
+
)
|
|
761
|
+
except Exception:
|
|
762
|
+
self.stats.errors += 1
|
|
763
|
+
logger.exception(
|
|
764
|
+
"Conversation turn concern reducer failed this tick"
|
|
765
|
+
)
|
|
766
|
+
|
|
767
|
+
reducer = getattr(workspace, "event_reducer", None) if workspace else None
|
|
768
|
+
if reducer is not None and getattr(reducer, "mode", "off") != "off":
|
|
769
|
+
result = reducer.run_once(limit=100)
|
|
770
|
+
self.stats.events_processed += int(result.get("processed") or 0)
|
|
771
|
+
if result.get("error"):
|
|
772
|
+
self.stats.errors += 1
|
|
773
|
+
logger.warning("Durable event reducer paused: %s", result["error"])
|
|
774
|
+
return
|
|
775
|
+
|
|
776
|
+
recent = list(self.events.get_history(limit=50))
|
|
777
|
+
if self._last_event_seen_id is not None:
|
|
778
|
+
for i in range(len(recent) - 1, -1, -1):
|
|
779
|
+
if getattr(recent[i], "id", None) == self._last_event_seen_id:
|
|
780
|
+
recent = recent[i + 1:]
|
|
781
|
+
break
|
|
782
|
+
# marker not found (aged out of the window): count the whole
|
|
783
|
+
# window, same bounded over-count the old code always had
|
|
784
|
+
if recent:
|
|
785
|
+
self._last_event_seen_id = getattr(recent[-1], "id", None)
|
|
786
|
+
self.stats.events_processed += len(recent)
|
|
787
|
+
except Exception as exc:
|
|
788
|
+
self.stats.errors += 1
|
|
789
|
+
logger.error("Phase events error: %s", exc, exc_info=True)
|
|
790
|
+
|
|
791
|
+
async def _phase_goals(self) -> None:
|
|
792
|
+
"""Check goal engine for goals needing attention."""
|
|
793
|
+
goals = self._registry.goals
|
|
794
|
+
if goals is None:
|
|
795
|
+
return
|
|
796
|
+
try:
|
|
797
|
+
from apsimo.cognition.goal_spine import cognition_spine_exclusive
|
|
798
|
+
legacy_read_only = cognition_spine_exclusive()
|
|
799
|
+
blocked = goals.list_goals(status="blocked", limit=20) if hasattr(goals, "list_goals") else []
|
|
800
|
+
accepted = goals.list_goals(status="accepted", limit=20) if hasattr(goals, "list_goals") else []
|
|
801
|
+
active = goals.list_goals(status="active", limit=50) if hasattr(goals, "list_goals") else []
|
|
802
|
+
|
|
803
|
+
for goal in ([] if legacy_read_only else accepted):
|
|
804
|
+
try:
|
|
805
|
+
if hasattr(goals, "activate_goal"):
|
|
806
|
+
goals.activate_goal(goal.get("goal_id", goal.get("id")))
|
|
807
|
+
logger.info("Loop activated goal: %r", goal.get("title"))
|
|
808
|
+
except Exception as exc:
|
|
809
|
+
logger.warning("Failed to activate goal: %s", exc)
|
|
810
|
+
|
|
811
|
+
total = len(blocked) + len(accepted) + len(active)
|
|
812
|
+
self.stats.goals_checked += total
|
|
813
|
+
except Exception as exc:
|
|
814
|
+
self.stats.errors += 1
|
|
815
|
+
logger.error("Phase goals error: %s", exc, exc_info=True)
|
|
816
|
+
|
|
817
|
+
async def _phase_anomalies(self) -> None:
|
|
818
|
+
"""Check anomaly detector for signals above severity threshold."""
|
|
819
|
+
try:
|
|
820
|
+
detector = self._registry.anomalies
|
|
821
|
+
if detector is None:
|
|
822
|
+
return
|
|
823
|
+
if hasattr(detector, "detect"):
|
|
824
|
+
recent_anomalies = await detector.detect(
|
|
825
|
+
threshold=self.config.anomaly_severity_threshold,
|
|
826
|
+
)
|
|
827
|
+
elif hasattr(detector, "get_recent"):
|
|
828
|
+
recent_anomalies = detector.get_recent(
|
|
829
|
+
min_severity=self.config.anomaly_severity_threshold,
|
|
830
|
+
limit=20,
|
|
831
|
+
)
|
|
832
|
+
else:
|
|
833
|
+
return
|
|
834
|
+
|
|
835
|
+
if recent_anomalies:
|
|
836
|
+
logger.info("Phase anomalies: %d above threshold", len(recent_anomalies))
|
|
837
|
+
_get_broadcast()({
|
|
838
|
+
"type": "anomaly",
|
|
839
|
+
"occurred_at": datetime.now(timezone.utc).isoformat(),
|
|
840
|
+
"payload": {"count": len(recent_anomalies)},
|
|
841
|
+
})
|
|
842
|
+
except Exception as exc:
|
|
843
|
+
self.stats.errors += 1
|
|
844
|
+
logger.error("Phase anomalies error: %s", exc, exc_info=True)
|
|
845
|
+
|
|
846
|
+
async def _phase_scheduled(self) -> None:
|
|
847
|
+
"""Run scheduled periodic tasks that are due (cron-style)."""
|
|
848
|
+
scheduler = self._scheduler or self._registry.scheduler
|
|
849
|
+
if scheduler is None:
|
|
850
|
+
return
|
|
851
|
+
try:
|
|
852
|
+
results = await scheduler.tick()
|
|
853
|
+
if results:
|
|
854
|
+
ok = sum(1 for r in results if r.get("status") == "ok")
|
|
855
|
+
skipped = sum(
|
|
856
|
+
1 for r in results if r.get("status") == "skipped")
|
|
857
|
+
self.stats.scheduled_runs += ok
|
|
858
|
+
errs = len(results) - ok - skipped
|
|
859
|
+
if errs:
|
|
860
|
+
self.stats.errors += errs
|
|
861
|
+
for r in results:
|
|
862
|
+
if r.get("status") not in ("ok", "skipped"):
|
|
863
|
+
logger.warning(
|
|
864
|
+
"Scheduled task failed: %s — %s",
|
|
865
|
+
r.get("task"), r.get("error"),
|
|
866
|
+
)
|
|
867
|
+
if ok:
|
|
868
|
+
logger.info("Phase scheduled: %d task(s) ran", ok)
|
|
869
|
+
except Exception as exc:
|
|
870
|
+
self.stats.errors += 1
|
|
871
|
+
logger.error("Phase scheduled error: %s", exc, exc_info=True)
|
|
872
|
+
|
|
873
|
+
async def _phase_initiative(self) -> None:
|
|
874
|
+
"""Run initiative engine to generate autonomous action proposals."""
|
|
875
|
+
engine = self._registry.initiative_engine
|
|
876
|
+
if engine is None:
|
|
877
|
+
return
|
|
878
|
+
|
|
879
|
+
try:
|
|
880
|
+
engine.clear_context()
|
|
881
|
+
await self._feed_pending_tasks(engine)
|
|
882
|
+
await self._feed_neglected_contacts(engine)
|
|
883
|
+
await self._feed_commitment_reminders(engine)
|
|
884
|
+
await self._feed_introduction_candidates(engine)
|
|
885
|
+
|
|
886
|
+
initiatives = await engine.generate(
|
|
887
|
+
min_priority=self.config.initiative_confidence_threshold,
|
|
888
|
+
cooldown_tasks=float(os.environ.get(
|
|
889
|
+
"COLONY_INITIATIVE_COOLDOWN_TASKS", "12",
|
|
890
|
+
)),
|
|
891
|
+
cooldown_contacts=float(os.environ.get(
|
|
892
|
+
"COLONY_INITIATIVE_COOLDOWN_CONTACTS", "72",
|
|
893
|
+
)),
|
|
894
|
+
)
|
|
895
|
+
|
|
896
|
+
if self._in_quiet_hours():
|
|
897
|
+
initiatives = [i for i in initiatives if getattr(i, "priority", 0) >= 0.9]
|
|
898
|
+
|
|
899
|
+
# Deferred initiatives queued by later phases of the previous
|
|
900
|
+
# tick (e.g. skill-capture reviews from Phase 6c).
|
|
901
|
+
deferred = getattr(self, "_deferred_initiatives", None)
|
|
902
|
+
if deferred:
|
|
903
|
+
initiatives = list(initiatives) + deferred
|
|
904
|
+
self._deferred_initiatives = []
|
|
905
|
+
|
|
906
|
+
if initiatives:
|
|
907
|
+
logger.info("Phase initiative: %d new proposals", len(initiatives))
|
|
908
|
+
sm = getattr(self._registry, 'self_model', None)
|
|
909
|
+
perspective = getattr(sm, 'perspective', None)
|
|
910
|
+
if perspective is not None:
|
|
911
|
+
initiatives = perspective.rank(initiatives, load=sm.load())
|
|
912
|
+
self._pending_initiatives = initiatives
|
|
913
|
+
self.stats.initiatives_generated += len(initiatives)
|
|
914
|
+
|
|
915
|
+
# Capture context for payload building in _phase_execute
|
|
916
|
+
self._last_initiative_context = dict(getattr(engine, "_context", {}))
|
|
917
|
+
except Exception as exc:
|
|
918
|
+
self.stats.errors += 1
|
|
919
|
+
logger.error("Phase initiative error: %s", exc, exc_info=True)
|
|
920
|
+
self._pending_initiatives = []
|
|
921
|
+
|
|
922
|
+
async def _phase_thinking(self) -> None:
|
|
923
|
+
"""Phase 5b: self-directed thinking (v0.17.0).
|
|
924
|
+
|
|
925
|
+
On a slow cadence (COLONY_THINKING_INTERVAL_SECS), hand the LLM a
|
|
926
|
+
situation report and let it propose novel initiatives the
|
|
927
|
+
data-reactive generators can't see. Results join the same
|
|
928
|
+
pending batch Phase 6 stores/delivers, so they inherit identical
|
|
929
|
+
dedup, quiet-hours, rate-limit, and approval treatment.
|
|
930
|
+
Disabled unless COLONY_ENABLE_INTERNAL_THINKING=true.
|
|
931
|
+
"""
|
|
932
|
+
# Mode: off | shadow | live (COLONY_THINKING_MODE). Back-compat:
|
|
933
|
+
# COLONY_ENABLE_INTERNAL_THINKING=true means "live". The autonomy
|
|
934
|
+
# preset fills the unset case (explicit env always wins).
|
|
935
|
+
mode = os.environ.get("COLONY_THINKING_MODE", "").strip().lower()
|
|
936
|
+
if mode not in ("off", "shadow", "live"):
|
|
937
|
+
if os.environ.get(
|
|
938
|
+
"COLONY_ENABLE_INTERNAL_THINKING", "false").lower() == "true":
|
|
939
|
+
mode = "live"
|
|
940
|
+
else:
|
|
941
|
+
from apsimo.util.autonomy_preset import resolve
|
|
942
|
+
mode = resolve("COLONY_THINKING_MODE",
|
|
943
|
+
("off", "shadow", "live"), "off")
|
|
944
|
+
if mode == "off":
|
|
945
|
+
return
|
|
946
|
+
router = self._registry.llm_router
|
|
947
|
+
if router is None:
|
|
948
|
+
return
|
|
949
|
+
thinker = getattr(self, "_thinker", None)
|
|
950
|
+
if thinker is None:
|
|
951
|
+
from apsimo.intelligence.components.self_directed_thinker import (
|
|
952
|
+
SelfDirectedThinker,
|
|
953
|
+
)
|
|
954
|
+
|
|
955
|
+
def _brief():
|
|
956
|
+
sm = getattr(self._registry, "self_model", None)
|
|
957
|
+
return sm.brief() if sm is not None else ""
|
|
958
|
+
|
|
959
|
+
def _bounds():
|
|
960
|
+
dm = getattr(self._registry, "directives", None)
|
|
961
|
+
return dm.context_brief() if dm is not None else ""
|
|
962
|
+
|
|
963
|
+
thinker = SelfDirectedThinker(router, self_brief_fn=_brief,
|
|
964
|
+
boundaries_fn=_bounds)
|
|
965
|
+
self._thinker = thinker
|
|
966
|
+
if not thinker.due():
|
|
967
|
+
return
|
|
968
|
+
thinker.mark_ran()
|
|
969
|
+
try:
|
|
970
|
+
situation = self._build_thinking_situation()
|
|
971
|
+
initiatives = await thinker.think(situation)
|
|
972
|
+
if not initiatives:
|
|
973
|
+
return
|
|
974
|
+
|
|
975
|
+
# P3 turns the legacy free-running thinker into a candidate
|
|
976
|
+
# generator only. Its observations enter the scoped workspace;
|
|
977
|
+
# it may no longer write initiatives/projects in parallel with
|
|
978
|
+
# the canonical Concern -> ThoughtJob -> Project path.
|
|
979
|
+
try:
|
|
980
|
+
from apsimo.cognition.goal_spine import cognition_spine_exclusive
|
|
981
|
+
if cognition_spine_exclusive():
|
|
982
|
+
workspace = getattr(self._registry, "workspace", None)
|
|
983
|
+
if workspace is None:
|
|
984
|
+
logger.warning(
|
|
985
|
+
"P3 thinker candidates held: workspace unavailable")
|
|
986
|
+
return
|
|
987
|
+
_record_p3_thinker_candidates(workspace, initiatives)
|
|
988
|
+
logger.info(
|
|
989
|
+
"P3 thinker: converted %d proposal(s) to workspace candidates",
|
|
990
|
+
len(initiatives),
|
|
991
|
+
)
|
|
992
|
+
return
|
|
993
|
+
except Exception:
|
|
994
|
+
logger.exception("P3 thinker candidate conversion failed")
|
|
995
|
+
return
|
|
996
|
+
|
|
997
|
+
# Package each thought into a well-formed Proposal and route it
|
|
998
|
+
# through the guarded (shadow-held, boundary-checked, rate-limited)
|
|
999
|
+
# delivery path. Nothing is sent while delivery shadow is on.
|
|
1000
|
+
from apsimo.proposals import build_from_thinker, proposal_to_payload
|
|
1001
|
+
delivery = self._registry.delivery
|
|
1002
|
+
pstore = getattr(self._registry, "proposal_store", None)
|
|
1003
|
+
fb = getattr(self._registry, "feedback_store", None)
|
|
1004
|
+
n = 0
|
|
1005
|
+
for init in initiatives:
|
|
1006
|
+
try:
|
|
1007
|
+
prop = build_from_thinker(init)
|
|
1008
|
+
if prop is None:
|
|
1009
|
+
# Ungrounded thought: no honest why_it_helps, so it
|
|
1010
|
+
# does not ship (item 4).
|
|
1011
|
+
continue
|
|
1012
|
+
# Outcome-driven priority: decay proposal classes the owner
|
|
1013
|
+
# ignores/dismisses, boost the ones he acts on (item 3b).
|
|
1014
|
+
if fb is not None:
|
|
1015
|
+
try:
|
|
1016
|
+
prop.confidence = max(0.0, min(1.0,
|
|
1017
|
+
prop.confidence * fb.multiplier(prop.initiative_type)))
|
|
1018
|
+
except Exception:
|
|
1019
|
+
pass
|
|
1020
|
+
if delivery is not None:
|
|
1021
|
+
await self._route_reachout_delivery(proposal_to_payload(prop), delivery)
|
|
1022
|
+
if pstore is not None:
|
|
1023
|
+
pstore.add(prop)
|
|
1024
|
+
n += 1
|
|
1025
|
+
except Exception:
|
|
1026
|
+
logger.debug("proposal routing failed", exc_info=True)
|
|
1027
|
+
logger.info("Phase thinking[%s]: %d proposal(s) generated", mode, n)
|
|
1028
|
+
|
|
1029
|
+
# Only in LIVE mode do the thought-up items ALSO become internal
|
|
1030
|
+
# work (research/knowledge initiatives the executor will run).
|
|
1031
|
+
if mode == "live":
|
|
1032
|
+
self._pending_initiatives = list(
|
|
1033
|
+
self._pending_initiatives or []) + initiatives
|
|
1034
|
+
self.stats.initiatives_generated += len(initiatives)
|
|
1035
|
+
except Exception as exc:
|
|
1036
|
+
self.stats.errors += 1
|
|
1037
|
+
logger.error("Phase thinking error: %s", exc, exc_info=True)
|
|
1038
|
+
|
|
1039
|
+
def _build_thinking_situation(self) -> dict:
|
|
1040
|
+
"""Assemble the situation report for the thinking phase."""
|
|
1041
|
+
situation: dict = {}
|
|
1042
|
+
ctx = dict(getattr(self, "_last_initiative_context", {}) or {})
|
|
1043
|
+
for key in ("pending_tasks", "neglected_contacts",
|
|
1044
|
+
"commitment_reminders"):
|
|
1045
|
+
if ctx.get(key):
|
|
1046
|
+
situation[key] = list(ctx[key])[:10]
|
|
1047
|
+
|
|
1048
|
+
goals = self._registry.goals
|
|
1049
|
+
if goals is not None and hasattr(goals, "list_goals"):
|
|
1050
|
+
try:
|
|
1051
|
+
situation["active_goals"] = goals.list_goals(
|
|
1052
|
+
status="active", limit=20)
|
|
1053
|
+
situation["blocked_goals"] = goals.list_goals(
|
|
1054
|
+
status="blocked", limit=10)
|
|
1055
|
+
except Exception:
|
|
1056
|
+
pass
|
|
1057
|
+
|
|
1058
|
+
pending = getattr(self, "_pending_initiatives", None) or []
|
|
1059
|
+
situation["current_initiatives"] = [
|
|
1060
|
+
getattr(i, "description", "") for i in pending][:20]
|
|
1061
|
+
|
|
1062
|
+
# Agent capability awareness (v0.18.0): the plugin reports the
|
|
1063
|
+
# Hermes skill index into the "skills" observation domain, so the
|
|
1064
|
+
# thinker proposes work the agent can actually do — and spots
|
|
1065
|
+
# genuine skill gaps instead of guessing.
|
|
1066
|
+
try:
|
|
1067
|
+
from apsimo.api.routers.observations import (
|
|
1068
|
+
get_observation_store,
|
|
1069
|
+
)
|
|
1070
|
+
obs_store = get_observation_store()
|
|
1071
|
+
if obs_store is not None:
|
|
1072
|
+
skills = obs_store.list("skills", limit=40)
|
|
1073
|
+
if skills:
|
|
1074
|
+
situation["agent_skills"] = [
|
|
1075
|
+
{"name": o.entity_id,
|
|
1076
|
+
"description": str(
|
|
1077
|
+
(o.payload or {}).get("description", ""))[:120]}
|
|
1078
|
+
for o in skills]
|
|
1079
|
+
except Exception:
|
|
1080
|
+
pass
|
|
1081
|
+
return situation
|
|
1082
|
+
|
|
1083
|
+
def _build_initiative_context(self, initiative: Any, type_value: str) -> dict:
|
|
1084
|
+
"""Build a focused, per-initiative context dict.
|
|
1085
|
+
|
|
1086
|
+
Instead of dumping the entire engine state (which leaks internal
|
|
1087
|
+
context like all pending tasks, all neglected contacts, etc.), we
|
|
1088
|
+
look up only the item relevant to this specific initiative.
|
|
1089
|
+
"""
|
|
1090
|
+
raw_ctx = getattr(self, "_last_initiative_context", {})
|
|
1091
|
+
entity_id = getattr(initiative, "entity_id", None)
|
|
1092
|
+
desc = getattr(initiative, "description", "")
|
|
1093
|
+
|
|
1094
|
+
if type_value == "follow_up":
|
|
1095
|
+
for item in raw_ctx.get("pending_tasks", []):
|
|
1096
|
+
if item.get("entity_id") == entity_id:
|
|
1097
|
+
return {
|
|
1098
|
+
"blocked_goal": {
|
|
1099
|
+
"goal_id": entity_id,
|
|
1100
|
+
"title": item.get("description", desc),
|
|
1101
|
+
"days_pending": item.get("days_pending", 0),
|
|
1102
|
+
}
|
|
1103
|
+
}
|
|
1104
|
+
return {}
|
|
1105
|
+
|
|
1106
|
+
if type_value == "relationship":
|
|
1107
|
+
for contact in raw_ctx.get("neglected_contacts", []):
|
|
1108
|
+
if contact.get("entity_id") == entity_id:
|
|
1109
|
+
return {
|
|
1110
|
+
"neglected_contact": {
|
|
1111
|
+
"contact_id": entity_id,
|
|
1112
|
+
"contact_name": contact.get("name"),
|
|
1113
|
+
"days_since_contact": contact.get("days_since_contact", 0),
|
|
1114
|
+
}
|
|
1115
|
+
}
|
|
1116
|
+
return {}
|
|
1117
|
+
|
|
1118
|
+
if type_value == "commitment":
|
|
1119
|
+
for c in raw_ctx.get("upcoming_commitments", []):
|
|
1120
|
+
if c.get("commitment_id") == entity_id:
|
|
1121
|
+
return {
|
|
1122
|
+
"commitment": {
|
|
1123
|
+
"commitment_id": entity_id,
|
|
1124
|
+
"commitment_text": c.get("description"),
|
|
1125
|
+
"deadline": c.get("due_at"),
|
|
1126
|
+
"status": c.get("status"),
|
|
1127
|
+
"person_id": c.get("person_id"),
|
|
1128
|
+
"hours_until_due": c.get("hours_until_due"),
|
|
1129
|
+
}
|
|
1130
|
+
}
|
|
1131
|
+
return {}
|
|
1132
|
+
|
|
1133
|
+
if type_value == "scheduling":
|
|
1134
|
+
for slot in raw_ctx.get("scheduling_opportunities", []):
|
|
1135
|
+
if slot.get("description") == desc:
|
|
1136
|
+
return {
|
|
1137
|
+
"upcoming_commitment": {
|
|
1138
|
+
"description": slot.get("description", ""),
|
|
1139
|
+
"hours_until_due": 0, # not stored in opportunity dict
|
|
1140
|
+
}
|
|
1141
|
+
}
|
|
1142
|
+
return {}
|
|
1143
|
+
|
|
1144
|
+
if type_value == "health":
|
|
1145
|
+
for alert in raw_ctx.get("health_alerts", []):
|
|
1146
|
+
if alert.get("metric") == entity_id:
|
|
1147
|
+
return {
|
|
1148
|
+
"health_alert": {
|
|
1149
|
+
"metric": entity_id,
|
|
1150
|
+
"value": alert.get("value"),
|
|
1151
|
+
"target": alert.get("target"),
|
|
1152
|
+
}
|
|
1153
|
+
}
|
|
1154
|
+
return {}
|
|
1155
|
+
|
|
1156
|
+
if type_value == "capability_gap":
|
|
1157
|
+
for gap in raw_ctx.get("capability_gaps", []):
|
|
1158
|
+
if gap.get("id") == entity_id:
|
|
1159
|
+
return {"capability_gap": gap}
|
|
1160
|
+
return {}
|
|
1161
|
+
|
|
1162
|
+
if type_value == "knowledge_acquisition":
|
|
1163
|
+
for gap in raw_ctx.get("knowledge_gaps", []):
|
|
1164
|
+
if gap.get("id") == entity_id:
|
|
1165
|
+
return {"knowledge_gap": gap}
|
|
1166
|
+
return {}
|
|
1167
|
+
|
|
1168
|
+
if type_value == "behavioral_correction":
|
|
1169
|
+
for pattern in raw_ctx.get("behavioral_patterns", []):
|
|
1170
|
+
if pattern.get("id") == entity_id:
|
|
1171
|
+
return {"behavioral_pattern": pattern}
|
|
1172
|
+
return {}
|
|
1173
|
+
|
|
1174
|
+
return {}
|
|
1175
|
+
|
|
1176
|
+
async def _phase_execute(self) -> None:
|
|
1177
|
+
"""Execute self-initiatives in the sidecar, then push remaining to delivery."""
|
|
1178
|
+
engine = self._registry.initiative_engine
|
|
1179
|
+
delivery = self._registry.delivery
|
|
1180
|
+
|
|
1181
|
+
for initiative in list(self._pending_initiatives):
|
|
1182
|
+
if (not self.config.proposals_only
|
|
1183
|
+
and self.stats.actions_this_hour >= self.config.max_actions_per_hour):
|
|
1184
|
+
logger.warning("Hourly action limit reached")
|
|
1185
|
+
break
|
|
1186
|
+
|
|
1187
|
+
initiative_type = getattr(initiative, "type", "unknown")
|
|
1188
|
+
type_value = initiative_type.value if hasattr(initiative_type, "value") else str(initiative_type)
|
|
1189
|
+
|
|
1190
|
+
is_self_initiative = type_value in {
|
|
1191
|
+
"subsystem_health", "data_quality", "operational",
|
|
1192
|
+
"capability_gap", "knowledge_acquisition", "behavioral_correction",
|
|
1193
|
+
}
|
|
1194
|
+
|
|
1195
|
+
# Try auto-execute for self-initiatives
|
|
1196
|
+
if is_self_initiative and engine is not None and not self.config.proposals_only:
|
|
1197
|
+
try:
|
|
1198
|
+
exec_result = await engine.execute_initiative(initiative.id)
|
|
1199
|
+
result_status = exec_result.get("status")
|
|
1200
|
+
skill_result = exec_result.get("result")
|
|
1201
|
+
|
|
1202
|
+
if result_status == "executed" and skill_result == "auto_fixed":
|
|
1203
|
+
self.stats.actions_executed += 1
|
|
1204
|
+
self.stats.actions_this_hour += 1
|
|
1205
|
+
logger.info("Auto-fixed initiative: %s", initiative.id)
|
|
1206
|
+
continue # Don't push to delivery
|
|
1207
|
+
|
|
1208
|
+
if result_status == "executed" and skill_result == "proposal_created":
|
|
1209
|
+
# Still push to delivery, but mark as proposed
|
|
1210
|
+
pass
|
|
1211
|
+
|
|
1212
|
+
if result_status in ("no_skill", "not_self_initiative"):
|
|
1213
|
+
# No skill matched — push to delivery for human decision
|
|
1214
|
+
pass
|
|
1215
|
+
except Exception as exc:
|
|
1216
|
+
logger.error("Auto-execution failed for %s: %s", initiative.id, exc)
|
|
1217
|
+
|
|
1218
|
+
# Build and push payload
|
|
1219
|
+
try:
|
|
1220
|
+
# Situational context snapshot (v0.16.0): persisted with the
|
|
1221
|
+
# initiative so the agent gets it over the REST API, not just
|
|
1222
|
+
# in push payloads. Carries the rationale and a capture
|
|
1223
|
+
# timestamp (volatile types check it against their TTL).
|
|
1224
|
+
initiative_context = self._build_initiative_context(initiative, type_value)
|
|
1225
|
+
trigger_data = getattr(initiative, "trigger_data", None)
|
|
1226
|
+
if trigger_data and not initiative_context:
|
|
1227
|
+
initiative_context = dict(trigger_data)
|
|
1228
|
+
rationale = getattr(initiative, "rationale", "")
|
|
1229
|
+
if rationale:
|
|
1230
|
+
initiative_context.setdefault("rationale", rationale)
|
|
1231
|
+
initiative_context.setdefault(
|
|
1232
|
+
"context_captured_at",
|
|
1233
|
+
datetime.now(timezone.utc).isoformat(),
|
|
1234
|
+
)
|
|
1235
|
+
initiative_context.setdefault("candidate_id", str(initiative.id))
|
|
1236
|
+
|
|
1237
|
+
# Persist initiative before dispatch so it survives restarts
|
|
1238
|
+
store = getattr(self._registry, "initiative_store", None)
|
|
1239
|
+
if store:
|
|
1240
|
+
try:
|
|
1241
|
+
loop = asyncio.get_event_loop()
|
|
1242
|
+
create_call = functools.partial(
|
|
1243
|
+
store.create_with_outcome,
|
|
1244
|
+
type=type_value,
|
|
1245
|
+
description=getattr(initiative, "description", ""),
|
|
1246
|
+
priority=getattr(initiative, "priority", 0.5),
|
|
1247
|
+
rationale=rationale,
|
|
1248
|
+
action_hint=getattr(initiative, "action_hint", None),
|
|
1249
|
+
entity_id=getattr(initiative, "entity_id", None),
|
|
1250
|
+
dedup_key=getattr(initiative, "dedup_key", None),
|
|
1251
|
+
dedup_base=getattr(initiative, "dedup_base", None),
|
|
1252
|
+
context=initiative_context or None,
|
|
1253
|
+
expires_at=getattr(initiative, "expires_at", None),
|
|
1254
|
+
source_type=type_value,
|
|
1255
|
+
created_by="autonomy_loop",
|
|
1256
|
+
)
|
|
1257
|
+
stored, outcome = await loop.run_in_executor(None, create_call)
|
|
1258
|
+
# Dispatch ONLY genuinely-new work: a fresh row ("created") or a retried
|
|
1259
|
+
# failure ("reactivated"). An already-active instance, or one that already
|
|
1260
|
+
# ran this period, must not be re-dispatched. (This replaces the old
|
|
1261
|
+
# id-comparison guard, which compared the store's uuid to the engine's
|
|
1262
|
+
# logical id — never equal — and so skipped every fresh initiative.)
|
|
1263
|
+
if outcome not in ("created", "reactivated"):
|
|
1264
|
+
logger.debug("Initiative %s: %s, skipping dispatch",
|
|
1265
|
+
getattr(initiative, "id", "?"), outcome)
|
|
1266
|
+
continue
|
|
1267
|
+
# Use the persisted id for the payload
|
|
1268
|
+
initiative_id = stored.id if stored else getattr(initiative, "id", str(uuid.uuid4()))
|
|
1269
|
+
except Exception as exc:
|
|
1270
|
+
logger.error("Failed to persist initiative, skipping dispatch: %s", exc)
|
|
1271
|
+
continue
|
|
1272
|
+
else:
|
|
1273
|
+
initiative_id = getattr(initiative, "id", str(uuid.uuid4()))
|
|
1274
|
+
|
|
1275
|
+
if self.config.proposals_only:
|
|
1276
|
+
# Ranking and durable proposal creation are the complete
|
|
1277
|
+
# effect of this selected rollout. Do not wake legacy
|
|
1278
|
+
# executor skills or enqueue/deliver the proposal here.
|
|
1279
|
+
continue
|
|
1280
|
+
|
|
1281
|
+
# --- v0.13.0: Route AGENT_ACTION initiatives to task queue ---
|
|
1282
|
+
action_hint = getattr(initiative, "action_hint", None) or ""
|
|
1283
|
+
is_agent_action = (
|
|
1284
|
+
type_value == "agent_action"
|
|
1285
|
+
or action_hint.startswith("agent_")
|
|
1286
|
+
)
|
|
1287
|
+
|
|
1288
|
+
if is_agent_action:
|
|
1289
|
+
await self._post_agent_action_to_queue(
|
|
1290
|
+
initiative, initiative_id, type_value, action_hint
|
|
1291
|
+
)
|
|
1292
|
+
continue # Do NOT push to delivery bridge
|
|
1293
|
+
|
|
1294
|
+
payload = {
|
|
1295
|
+
"id": initiative_id,
|
|
1296
|
+
"type": type_value,
|
|
1297
|
+
"priority": getattr(initiative, "priority", 0.5),
|
|
1298
|
+
"title": getattr(initiative, "description", "").split(".")[0][:80] if getattr(initiative, "description", "") else "(no title)",
|
|
1299
|
+
"description": getattr(initiative, "description", ""),
|
|
1300
|
+
"rationale": getattr(initiative, "rationale", ""),
|
|
1301
|
+
"suggested_action": action_hint or "review_and_decide",
|
|
1302
|
+
"entity_id": getattr(initiative, "entity_id", None),
|
|
1303
|
+
"entity_type": type_value,
|
|
1304
|
+
"channel_hint": "home" if is_self_initiative else "dm",
|
|
1305
|
+
"context": initiative_context,
|
|
1306
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
1307
|
+
}
|
|
1308
|
+
|
|
1309
|
+
if delivery:
|
|
1310
|
+
await self._route_reachout_delivery(payload, delivery)
|
|
1311
|
+
|
|
1312
|
+
# WebSocket broadcast
|
|
1313
|
+
try:
|
|
1314
|
+
broadcast = _get_broadcast()
|
|
1315
|
+
broadcast({
|
|
1316
|
+
"type": "initiative",
|
|
1317
|
+
"occurred_at": datetime.now(timezone.utc).isoformat(),
|
|
1318
|
+
"payload": payload,
|
|
1319
|
+
})
|
|
1320
|
+
except Exception:
|
|
1321
|
+
logger.warning("WebSocket broadcast failed (non-critical)")
|
|
1322
|
+
except Exception as exc:
|
|
1323
|
+
logger.error("Failed to push initiative: %s", exc)
|
|
1324
|
+
|
|
1325
|
+
self._pending_initiatives = []
|
|
1326
|
+
|
|
1327
|
+
@staticmethod
|
|
1328
|
+
def _stored_initiative_delivery_payload(initiative: Any) -> dict:
|
|
1329
|
+
initiative_type = getattr(initiative, "type", "unknown")
|
|
1330
|
+
type_value = (
|
|
1331
|
+
initiative_type.value
|
|
1332
|
+
if hasattr(initiative_type, "value")
|
|
1333
|
+
else str(initiative_type)
|
|
1334
|
+
)
|
|
1335
|
+
created_at = getattr(initiative, "created_at", None)
|
|
1336
|
+
description = getattr(initiative, "description", "") or ""
|
|
1337
|
+
return {
|
|
1338
|
+
"id": str(getattr(initiative, "id", "")),
|
|
1339
|
+
"type": type_value,
|
|
1340
|
+
"priority": getattr(initiative, "priority", 0.5),
|
|
1341
|
+
"title": description.split(".")[0][:80] if description else "(no title)",
|
|
1342
|
+
"description": description,
|
|
1343
|
+
"rationale": getattr(initiative, "rationale", "") or "",
|
|
1344
|
+
"suggested_action": (
|
|
1345
|
+
getattr(initiative, "action_hint", None) or "review_and_decide"
|
|
1346
|
+
),
|
|
1347
|
+
"entity_id": getattr(initiative, "entity_id", None),
|
|
1348
|
+
"entity_type": type_value,
|
|
1349
|
+
"context": getattr(initiative, "context", None) or {},
|
|
1350
|
+
"generated_at": (
|
|
1351
|
+
created_at.isoformat()
|
|
1352
|
+
if created_at is not None
|
|
1353
|
+
else datetime.now(timezone.utc).isoformat()
|
|
1354
|
+
),
|
|
1355
|
+
}
|
|
1356
|
+
|
|
1357
|
+
async def _rebuild_governed_delivery_replays(self, delivery: Any) -> None:
|
|
1358
|
+
"""Rebuild exact admitted replays after restart without admitting new work."""
|
|
1359
|
+
|
|
1360
|
+
getter = getattr(delivery, "governed_pending_delivery_ids", None)
|
|
1361
|
+
initiative_store = getattr(self._registry, "initiative_store", None)
|
|
1362
|
+
if not callable(getter) or initiative_store is None:
|
|
1363
|
+
return
|
|
1364
|
+
try:
|
|
1365
|
+
pending_delivery_ids = set(getter(limit=100))
|
|
1366
|
+
except Exception:
|
|
1367
|
+
logger.exception("Could not read durable governed delivery identities")
|
|
1368
|
+
return
|
|
1369
|
+
if not pending_delivery_ids:
|
|
1370
|
+
return
|
|
1371
|
+
loop = asyncio.get_running_loop()
|
|
1372
|
+
try:
|
|
1373
|
+
initiatives = await loop.run_in_executor(
|
|
1374
|
+
None,
|
|
1375
|
+
functools.partial(
|
|
1376
|
+
initiative_store.list, status=["pending"], limit=1000
|
|
1377
|
+
),
|
|
1378
|
+
)
|
|
1379
|
+
except Exception:
|
|
1380
|
+
logger.exception("Could not read pending initiatives for delivery rebuild")
|
|
1381
|
+
return
|
|
1382
|
+
for initiative in initiatives:
|
|
1383
|
+
initiative_id = str(getattr(initiative, "id", "") or "")
|
|
1384
|
+
initiative_type = getattr(initiative, "type", "unknown")
|
|
1385
|
+
type_value = (
|
|
1386
|
+
initiative_type.value
|
|
1387
|
+
if hasattr(initiative_type, "value")
|
|
1388
|
+
else str(initiative_type)
|
|
1389
|
+
)
|
|
1390
|
+
if not initiative_id:
|
|
1391
|
+
continue
|
|
1392
|
+
source_id = "initiative:" + initiative_id
|
|
1393
|
+
delivery_id = "initiative:" + hashlib.sha256(
|
|
1394
|
+
(type_value + "\0" + source_id).encode("utf-8")
|
|
1395
|
+
).hexdigest()
|
|
1396
|
+
if delivery_id not in pending_delivery_ids:
|
|
1397
|
+
continue
|
|
1398
|
+
self._governed_delivery_replays[delivery_id] = (
|
|
1399
|
+
self._stored_initiative_delivery_payload(initiative)
|
|
1400
|
+
)
|
|
1401
|
+
if len(self._governed_delivery_replays) >= 100:
|
|
1402
|
+
break
|
|
1403
|
+
|
|
1404
|
+
async def _phase_governed_delivery_reconciliation(self) -> None:
|
|
1405
|
+
"""Re-poll a bounded set of admitted messages without a new trigger."""
|
|
1406
|
+
|
|
1407
|
+
delivery = getattr(self._registry, "delivery", None)
|
|
1408
|
+
if delivery is None:
|
|
1409
|
+
return
|
|
1410
|
+
if not self._governed_delivery_replays:
|
|
1411
|
+
await self._rebuild_governed_delivery_replays(delivery)
|
|
1412
|
+
if not self._governed_delivery_replays:
|
|
1413
|
+
return
|
|
1414
|
+
# A hard bound prevents a large approval backlog from monopolizing the
|
|
1415
|
+
# event loop. The next cadence continues from the same stable requests;
|
|
1416
|
+
# the bridge itself suppresses HTTP until each row's next_poll_at.
|
|
1417
|
+
for _delivery_id, payload in list(
|
|
1418
|
+
self._governed_delivery_replays.items()
|
|
1419
|
+
)[:25]:
|
|
1420
|
+
try:
|
|
1421
|
+
await self._route_reachout_delivery(copy.deepcopy(payload), delivery)
|
|
1422
|
+
except Exception:
|
|
1423
|
+
self.stats.errors += 1
|
|
1424
|
+
logger.exception(
|
|
1425
|
+
"Governed delivery lifecycle replay failed for %s", _delivery_id
|
|
1426
|
+
)
|
|
1427
|
+
|
|
1428
|
+
async def _recipient_is_owner(self, person_id: str, payload: dict) -> bool:
|
|
1429
|
+
"""True when a delivery is owner-directed (exempt from the outbound
|
|
1430
|
+
third-party approval gate).
|
|
1431
|
+
|
|
1432
|
+
Proposals target the owner by construction. Otherwise resolve the
|
|
1433
|
+
recipient identity: fail closed (treat as NON-owner, so the approval
|
|
1434
|
+
gate engages) if the owner identity cannot be established.
|
|
1435
|
+
"""
|
|
1436
|
+
if payload.get("entity_type") == "proposal":
|
|
1437
|
+
return True
|
|
1438
|
+
if not person_id or person_id == "owner":
|
|
1439
|
+
return True
|
|
1440
|
+
try:
|
|
1441
|
+
from apsimo.identity.resolver import get_identity_resolver
|
|
1442
|
+
resolver = get_identity_resolver()
|
|
1443
|
+
return bool(await resolver.is_owner(person_id))
|
|
1444
|
+
except Exception:
|
|
1445
|
+
# Owner unresolved -> cannot prove owner-directed -> gate engages.
|
|
1446
|
+
return False
|
|
1447
|
+
|
|
1448
|
+
async def _route_reachout_delivery(self, payload: dict, delivery: Any) -> bool:
|
|
1449
|
+
"""Sanitise, staleness-guard, rate-check and (shadow-)deliver ONE
|
|
1450
|
+
reach-out initiative through the guarded Hermes path.
|
|
1451
|
+
|
|
1452
|
+
Shared by _phase_execute and _phase_startup_repush so both apply
|
|
1453
|
+
identical gating (classification, sanitisation, staleness, quiet-hours
|
|
1454
|
+
urgency cap, per-recipient rate limit). Internal initiatives are a
|
|
1455
|
+
no-op here. Returns True only when a real push was sent.
|
|
1456
|
+
"""
|
|
1457
|
+
from apsimo.delivery.classification import is_reachout
|
|
1458
|
+
from apsimo.delivery import reachout_policy as rp
|
|
1459
|
+
|
|
1460
|
+
type_value = payload.get("type", "")
|
|
1461
|
+
iid = payload.get("id")
|
|
1462
|
+
|
|
1463
|
+
# Reach-out initiatives AND proposals (a dedicated type) leave the
|
|
1464
|
+
# machine; everything else is internal. reachout_types() is NOT
|
|
1465
|
+
# overloaded -- proposals are handled as their own delivered type.
|
|
1466
|
+
is_proposal = (type_value == "proposal")
|
|
1467
|
+
if not is_reachout(type_value) and not is_proposal:
|
|
1468
|
+
logger.debug("Initiative %s (%s) is internal — not routed to delivery",
|
|
1469
|
+
iid, type_value)
|
|
1470
|
+
return False
|
|
1471
|
+
|
|
1472
|
+
# Boundary gate: never message about a subject the owner set off-limits.
|
|
1473
|
+
directives = getattr(self._registry, "directives", None)
|
|
1474
|
+
if directives is not None:
|
|
1475
|
+
try:
|
|
1476
|
+
from apsimo.directives import Action
|
|
1477
|
+
verdict = directives.check(Action(
|
|
1478
|
+
kind="deliver",
|
|
1479
|
+
text=f"{payload.get('title','')} {payload.get('description','')}",
|
|
1480
|
+
target=payload.get("entity_id", "") or "",
|
|
1481
|
+
entity_id=payload.get("entity_id", "") or "",
|
|
1482
|
+
high_risk=True,
|
|
1483
|
+
))
|
|
1484
|
+
if not verdict.allowed:
|
|
1485
|
+
logger.warning(
|
|
1486
|
+
"Reach-out %s (%s) REFUSED by boundary: %s",
|
|
1487
|
+
iid, type_value, verdict.reason,
|
|
1488
|
+
)
|
|
1489
|
+
return False
|
|
1490
|
+
except Exception:
|
|
1491
|
+
# An owner boundary we cannot evaluate must not be assumed
|
|
1492
|
+
# permissive: fail CLOSED by default (a missed delivery is
|
|
1493
|
+
# recoverable; messaging about a forbidden subject is not).
|
|
1494
|
+
self.stats.boundary_check_errors += 1
|
|
1495
|
+
from apsimo.directives.guard import boundary_fail_closed
|
|
1496
|
+
if boundary_fail_closed():
|
|
1497
|
+
logger.warning(
|
|
1498
|
+
"Reach-out %s (%s) REFUSED: boundary_check_error "
|
|
1499
|
+
"(boundary check raised; failing closed)",
|
|
1500
|
+
iid, type_value, exc_info=True,
|
|
1501
|
+
)
|
|
1502
|
+
return False
|
|
1503
|
+
logger.debug("delivery boundary check failed (allowing)", exc_info=True)
|
|
1504
|
+
|
|
1505
|
+
# Staleness guard: a long-overdue reach-out is noise, not a timely ping.
|
|
1506
|
+
# Proposals are timely by construction and exempt.
|
|
1507
|
+
if not is_proposal and rp.is_aged_out(payload):
|
|
1508
|
+
logger.info(
|
|
1509
|
+
"Reach-out %s (%s) aged out (%.1fd > %.1fd) — not delivered",
|
|
1510
|
+
iid, type_value, rp.reachout_age_days(payload), rp.max_age_days(),
|
|
1511
|
+
)
|
|
1512
|
+
return False
|
|
1513
|
+
|
|
1514
|
+
# Clean the outward-facing text before it reaches Hermes.
|
|
1515
|
+
payload = rp.sanitize_payload(payload)
|
|
1516
|
+
|
|
1517
|
+
# Resolve the ACTUAL recipient bucket + target the same way a real send
|
|
1518
|
+
# would, so the rate gate binds per recipient and the shadow view
|
|
1519
|
+
# matches reality.
|
|
1520
|
+
transport = os.environ.get(
|
|
1521
|
+
"COLONY_DELIVERY_TRANSPORT", "hermes_webhook"
|
|
1522
|
+
).strip().lower()
|
|
1523
|
+
preview = None
|
|
1524
|
+
if (
|
|
1525
|
+
transport == "gateway"
|
|
1526
|
+
and hasattr(delivery, "preview_initiative_async")
|
|
1527
|
+
):
|
|
1528
|
+
try:
|
|
1529
|
+
preview = await delivery.preview_initiative_async(payload)
|
|
1530
|
+
except Exception as exc:
|
|
1531
|
+
logger.warning("Delivery preview failed for %s: %s", iid, exc)
|
|
1532
|
+
elif hasattr(delivery, "preview_initiative"):
|
|
1533
|
+
try:
|
|
1534
|
+
preview = delivery.preview_initiative(payload)
|
|
1535
|
+
except Exception as exc:
|
|
1536
|
+
logger.warning("Delivery preview failed for %s: %s", iid, exc)
|
|
1537
|
+
person_id = (preview or {}).get("person_id") or payload.get("entity_id") or "owner"
|
|
1538
|
+
raw_urgency = float(
|
|
1539
|
+
(preview or {}).get("urgency", payload.get("priority", 0.5)) or 0.5
|
|
1540
|
+
)
|
|
1541
|
+
# Reach-out respects quiet hours unless explicitly urgent.
|
|
1542
|
+
gate_urgency = rp.quiet_hours_urgency(payload, raw_urgency)
|
|
1543
|
+
|
|
1544
|
+
rate_limiter = getattr(delivery, "_rate_limiter", None)
|
|
1545
|
+
allowed, reason = True, "ok"
|
|
1546
|
+
if rate_limiter is not None:
|
|
1547
|
+
allowed, reason = rate_limiter.can_deliver(person_id, urgency=gate_urgency)
|
|
1548
|
+
|
|
1549
|
+
if getattr(self.config, "delivery_shadow_mode", False):
|
|
1550
|
+
# Shadow: log the intended (sanitised) delivery; send nothing,
|
|
1551
|
+
# consume no budget.
|
|
1552
|
+
self._observe_p8_outbound(payload, preview or {})
|
|
1553
|
+
target = (preview or {}).get("target", {})
|
|
1554
|
+
logger.info(
|
|
1555
|
+
"SHADOW-DELIVERY reach-out id=%s type=%s recipient=%s target=%s "
|
|
1556
|
+
"rate_allowed=%s(%s) urgency=%.2f(gate=%.2f) age=%.1fd title=%r",
|
|
1557
|
+
iid, type_value, person_id, target, allowed, reason,
|
|
1558
|
+
raw_urgency, gate_urgency, rp.reachout_age_days(payload),
|
|
1559
|
+
(payload.get("title") or payload.get("description", ""))[:200],
|
|
1560
|
+
)
|
|
1561
|
+
return False
|
|
1562
|
+
|
|
1563
|
+
governed_admission_route = False
|
|
1564
|
+
preview_target = (preview or {}).get("target", {})
|
|
1565
|
+
preview_chat = (
|
|
1566
|
+
preview_target.get("user_chat")
|
|
1567
|
+
or preview_target.get("home_chat")
|
|
1568
|
+
or ""
|
|
1569
|
+
)
|
|
1570
|
+
preview_platform, _, _preview_chat_id = preview_chat.partition(":")
|
|
1571
|
+
if transport == "gateway" and hasattr(
|
|
1572
|
+
delivery, "governed_gateway_admission_enabled"
|
|
1573
|
+
):
|
|
1574
|
+
try:
|
|
1575
|
+
governed_admission_route = (
|
|
1576
|
+
delivery.governed_gateway_admission_enabled(
|
|
1577
|
+
preview_platform,
|
|
1578
|
+
) is True
|
|
1579
|
+
)
|
|
1580
|
+
except Exception:
|
|
1581
|
+
governed_admission_route = False
|
|
1582
|
+
|
|
1583
|
+
# Legacy outbound third-party gate remains the default. A deployment
|
|
1584
|
+
# may instead opt one gateway transport into a downstream governed
|
|
1585
|
+
# admission contract: that boundary owns bounded route/one-off
|
|
1586
|
+
# authorization and must later return an exact, non-delivery admission
|
|
1587
|
+
# receipt. A mere gateway HTTP 200 is never enough in that mode.
|
|
1588
|
+
recipient_is_owner = await self._recipient_is_owner(person_id, payload)
|
|
1589
|
+
if not recipient_is_owner:
|
|
1590
|
+
from apsimo.initiatives import standing_approvals
|
|
1591
|
+
if (
|
|
1592
|
+
not governed_admission_route
|
|
1593
|
+
and not standing_approvals.is_approved(
|
|
1594
|
+
"outbound_third_party_delivery"
|
|
1595
|
+
)
|
|
1596
|
+
):
|
|
1597
|
+
logger.warning(
|
|
1598
|
+
"Reach-out %s (%s) to non-owner recipient %s BLOCKED: "
|
|
1599
|
+
"outbound third-party delivery requires owner approval "
|
|
1600
|
+
"(grant 'outbound_third_party_delivery')",
|
|
1601
|
+
iid, type_value, person_id,
|
|
1602
|
+
)
|
|
1603
|
+
return False
|
|
1604
|
+
|
|
1605
|
+
if not allowed:
|
|
1606
|
+
logger.debug("Reach-out push rate-limited for %s: %s (urgency=%.2f)",
|
|
1607
|
+
person_id, reason, gate_urgency)
|
|
1608
|
+
return False
|
|
1609
|
+
|
|
1610
|
+
# Content safety on the actual outbound text (secret leak / disclosure
|
|
1611
|
+
# tier / injection / provenance). The ResponseGuard runs here on the
|
|
1612
|
+
# real send path so it is not merely an opt-in endpoint: in shadow it
|
|
1613
|
+
# logs, in enforce (COLONY_GUARD_MODE=enforce) it blocks a leaking
|
|
1614
|
+
# proactive message before it leaves. Owner-directed delivery is
|
|
1615
|
+
# authorized; third-party delivery already spent a bounded approval
|
|
1616
|
+
# gate above.
|
|
1617
|
+
guard_enforcing = (
|
|
1618
|
+
os.environ.get("COLONY_GUARD_MODE", "").strip().lower()
|
|
1619
|
+
== "enforce"
|
|
1620
|
+
)
|
|
1621
|
+
try:
|
|
1622
|
+
from apsimo.api.routers.host import _response_guard
|
|
1623
|
+
if _response_guard is None:
|
|
1624
|
+
if guard_enforcing:
|
|
1625
|
+
logger.warning(
|
|
1626
|
+
"Reach-out %s BLOCKED: configured ResponseGuard is unavailable",
|
|
1627
|
+
iid,
|
|
1628
|
+
)
|
|
1629
|
+
return False
|
|
1630
|
+
else:
|
|
1631
|
+
configured_mode = getattr(_response_guard, "configured_mode", None)
|
|
1632
|
+
configured_value = str(
|
|
1633
|
+
getattr(configured_mode, "value", configured_mode) or ""
|
|
1634
|
+
).strip().lower()
|
|
1635
|
+
# Configuration sources are monotonic at this boundary: a
|
|
1636
|
+
# stale in-process shadow instance cannot weaken an explicit
|
|
1637
|
+
# deployment-level enforce request, while an enforce instance
|
|
1638
|
+
# still strengthens an unset/shadow environment.
|
|
1639
|
+
guard_enforcing = (
|
|
1640
|
+
guard_enforcing or configured_value == "enforce"
|
|
1641
|
+
)
|
|
1642
|
+
_msg = (payload.get("description") or payload.get("title") or "")
|
|
1643
|
+
_tgt = (preview or {}).get("target", {})
|
|
1644
|
+
_chat = _tgt.get("user_chat") or _tgt.get("home_chat") or ""
|
|
1645
|
+
_plat, _, _cid = _chat.partition(":")
|
|
1646
|
+
_authorized = recipient_is_owner
|
|
1647
|
+
_guard_context = {
|
|
1648
|
+
"surface": "proactive_text",
|
|
1649
|
+
"target_contact_id": person_id,
|
|
1650
|
+
"target_gateway": _plat,
|
|
1651
|
+
"session_id": str(iid),
|
|
1652
|
+
"turn_id": str(iid),
|
|
1653
|
+
"authorized": _authorized,
|
|
1654
|
+
}
|
|
1655
|
+
_guard_result = await _response_guard.evaluate(
|
|
1656
|
+
response_text=_msg,
|
|
1657
|
+
**_guard_context,
|
|
1658
|
+
)
|
|
1659
|
+
_guard = _snapshot_proactive_verdict(_guard_result)
|
|
1660
|
+
verdict_error = (
|
|
1661
|
+
_proactive_enforce_verdict_error(_guard, _msg)
|
|
1662
|
+
if guard_enforcing else None
|
|
1663
|
+
)
|
|
1664
|
+
if verdict_error is not None:
|
|
1665
|
+
logger.warning(
|
|
1666
|
+
"Reach-out %s BLOCKED: ResponseGuard returned an invalid "
|
|
1667
|
+
"enforce verdict (%s)",
|
|
1668
|
+
iid, verdict_error,
|
|
1669
|
+
)
|
|
1670
|
+
return False
|
|
1671
|
+
guard_blocked = (
|
|
1672
|
+
_guard.decision != "allow"
|
|
1673
|
+
if guard_enforcing else _guard.blocked is True
|
|
1674
|
+
)
|
|
1675
|
+
if guard_blocked:
|
|
1676
|
+
# Enforce branch: one guarded regeneration attempt through
|
|
1677
|
+
# the rejection feedback loop. A loop-internal error never
|
|
1678
|
+
# overrides the guard BLOCK (fail closed on the block
|
|
1679
|
+
# side); only a revision the guard itself clears may ship.
|
|
1680
|
+
revised = None
|
|
1681
|
+
try:
|
|
1682
|
+
revised = await self._run_rejection_feedback(
|
|
1683
|
+
_msg, _guard, person_id=person_id, gateway=_plat,
|
|
1684
|
+
iid=iid, authorized=_authorized)
|
|
1685
|
+
except Exception:
|
|
1686
|
+
logger.debug("rejection feedback loop failed "
|
|
1687
|
+
"(block stands)", exc_info=True)
|
|
1688
|
+
if revised and guard_enforcing:
|
|
1689
|
+
# The feedback loop is not an egress authority. Recheck
|
|
1690
|
+
# its exact revision at this boundary and require a
|
|
1691
|
+
# canonical ALLOW bound to those bytes before mutation.
|
|
1692
|
+
revised_result = await _response_guard.evaluate(
|
|
1693
|
+
response_text=revised,
|
|
1694
|
+
**_guard_context,
|
|
1695
|
+
)
|
|
1696
|
+
revised_guard = _snapshot_proactive_verdict(
|
|
1697
|
+
revised_result
|
|
1698
|
+
)
|
|
1699
|
+
revised_error = _proactive_enforce_verdict_error(
|
|
1700
|
+
revised_guard, revised
|
|
1701
|
+
)
|
|
1702
|
+
if (
|
|
1703
|
+
revised_error is not None
|
|
1704
|
+
or revised_guard.decision != "allow"
|
|
1705
|
+
):
|
|
1706
|
+
logger.warning(
|
|
1707
|
+
"Reach-out %s BLOCKED: ResponseGuard did not "
|
|
1708
|
+
"authorize the exact revision (%s)",
|
|
1709
|
+
iid, revised_error or "decision is not allow",
|
|
1710
|
+
)
|
|
1711
|
+
return False
|
|
1712
|
+
if revised:
|
|
1713
|
+
logger.info(
|
|
1714
|
+
"Reach-out %s revised by rejection feedback loop "
|
|
1715
|
+
"after ResponseGuard block", iid)
|
|
1716
|
+
if payload.get("description"):
|
|
1717
|
+
payload["description"] = revised
|
|
1718
|
+
else:
|
|
1719
|
+
payload["title"] = revised
|
|
1720
|
+
else:
|
|
1721
|
+
logger.warning(
|
|
1722
|
+
"Reach-out %s BLOCKED by ResponseGuard (%s): %s",
|
|
1723
|
+
iid, _guard.decision,
|
|
1724
|
+
"; ".join(str(getattr(f, "name", f))
|
|
1725
|
+
for f in (_guard.findings or [])[:4]))
|
|
1726
|
+
return False
|
|
1727
|
+
except Exception:
|
|
1728
|
+
if guard_enforcing:
|
|
1729
|
+
logger.warning(
|
|
1730
|
+
"Reach-out %s BLOCKED: enforce ResponseGuard check failed",
|
|
1731
|
+
iid,
|
|
1732
|
+
exc_info=True,
|
|
1733
|
+
)
|
|
1734
|
+
return False
|
|
1735
|
+
logger.debug(
|
|
1736
|
+
"delivery ResponseGuard shadow check failed (allowing)",
|
|
1737
|
+
exc_info=True,
|
|
1738
|
+
)
|
|
1739
|
+
|
|
1740
|
+
# P8 is a write-only observer at this boundary: it sees a detached,
|
|
1741
|
+
# bounded snapshot of the final sanitized/revised text. Its result is
|
|
1742
|
+
# deliberately ignored and it cannot mutate transport-owned objects.
|
|
1743
|
+
self._observe_p8_outbound(payload, preview or {})
|
|
1744
|
+
|
|
1745
|
+
if not getattr(self.config, "proactive_delivery_enabled", False):
|
|
1746
|
+
logger.debug("Proactive delivery disabled — initiative stored for agent polling")
|
|
1747
|
+
return False
|
|
1748
|
+
|
|
1749
|
+
# Transport selection (env-driven, generic):
|
|
1750
|
+
# hermes_webhook (default) -- POST the structured initiative to a
|
|
1751
|
+
# composing agent's webhook (push_initiative), which writes the
|
|
1752
|
+
# final owner-facing message itself.
|
|
1753
|
+
# gateway -- POST the sanitised text directly to the deployment's
|
|
1754
|
+
# message gateway /internal/deliver (push_to_gateway), for
|
|
1755
|
+
# deployments whose channel transport speaks the flat
|
|
1756
|
+
# {platform, chat_id, message} contract.
|
|
1757
|
+
if transport == "gateway" and hasattr(delivery, "push_to_gateway"):
|
|
1758
|
+
target = (preview or {}).get("target", {})
|
|
1759
|
+
chat = target.get("user_chat") or target.get("home_chat") or ""
|
|
1760
|
+
platform, _, chat_id = chat.partition(":")
|
|
1761
|
+
message = (payload.get("description") or payload.get("title") or "").strip()
|
|
1762
|
+
if not (platform and chat_id and message):
|
|
1763
|
+
logger.warning(
|
|
1764
|
+
"Reach-out %s: gateway transport missing target/message "
|
|
1765
|
+
"(target=%r) — not delivered", iid, target,
|
|
1766
|
+
)
|
|
1767
|
+
return False
|
|
1768
|
+
raw_source_id = str(iid or "")
|
|
1769
|
+
if raw_source_id:
|
|
1770
|
+
source_id = "initiative:" + raw_source_id
|
|
1771
|
+
else:
|
|
1772
|
+
# Older/custom producers may omit an initiative ID. Preserve
|
|
1773
|
+
# their delivery behavior while deriving a retry-stable,
|
|
1774
|
+
# deployment-neutral source identity from the final sanitized
|
|
1775
|
+
# initiative instead of generating a fresh UUID per attempt.
|
|
1776
|
+
source_id = "initiative-derived:" + hashlib.sha256(
|
|
1777
|
+
json.dumps(
|
|
1778
|
+
payload,
|
|
1779
|
+
ensure_ascii=False,
|
|
1780
|
+
allow_nan=False,
|
|
1781
|
+
separators=(",", ":"),
|
|
1782
|
+
sort_keys=True,
|
|
1783
|
+
default=str,
|
|
1784
|
+
).encode("utf-8")
|
|
1785
|
+
).hexdigest()
|
|
1786
|
+
delivery_id = "initiative:" + hashlib.sha256(
|
|
1787
|
+
(str(type_value) + "\0" + source_id).encode("utf-8")
|
|
1788
|
+
).hexdigest()
|
|
1789
|
+
if (
|
|
1790
|
+
governed_admission_route
|
|
1791
|
+
and delivery_id not in self._governed_delivery_replays
|
|
1792
|
+
and len(self._governed_delivery_replays) >= 100
|
|
1793
|
+
):
|
|
1794
|
+
logger.error(
|
|
1795
|
+
"Governed delivery reconciliation backlog is full; "
|
|
1796
|
+
"refusing a new admission for %s", iid,
|
|
1797
|
+
)
|
|
1798
|
+
return False
|
|
1799
|
+
outcome = await delivery.push_to_gateway(
|
|
1800
|
+
platform=platform, chat_id=chat_id, message=message,
|
|
1801
|
+
source=type_value,
|
|
1802
|
+
delivery_id=delivery_id,
|
|
1803
|
+
source_id=source_id,
|
|
1804
|
+
)
|
|
1805
|
+
else:
|
|
1806
|
+
outcome = await delivery.push_initiative(payload)
|
|
1807
|
+
|
|
1808
|
+
transport_accepted = bool(outcome)
|
|
1809
|
+
outcome_state = str(getattr(outcome, "admission_state", ""))
|
|
1810
|
+
outcome_provider_delivered = getattr(outcome, "provider_delivered", None)
|
|
1811
|
+
outcome_terminal = getattr(outcome, "terminal", None)
|
|
1812
|
+
observation_new = getattr(outcome, "observation_new", True) is True
|
|
1813
|
+
governed_admitted = bool(
|
|
1814
|
+
governed_admission_route
|
|
1815
|
+
and getattr(outcome, "contract", "") == "governed_admission_v1"
|
|
1816
|
+
and transport_accepted
|
|
1817
|
+
and outcome_provider_delivered is False
|
|
1818
|
+
and outcome_state in {"accepted", "awaiting_approval"}
|
|
1819
|
+
and outcome_terminal is False
|
|
1820
|
+
)
|
|
1821
|
+
governed_delivered = bool(
|
|
1822
|
+
governed_admission_route
|
|
1823
|
+
and getattr(outcome, "contract", "") == "governed_admission_v1"
|
|
1824
|
+
and transport_accepted
|
|
1825
|
+
and outcome_provider_delivered is True
|
|
1826
|
+
and outcome_state == "delivered"
|
|
1827
|
+
and outcome_terminal is True
|
|
1828
|
+
)
|
|
1829
|
+
governed_terminal_failure = bool(
|
|
1830
|
+
governed_admission_route
|
|
1831
|
+
and getattr(outcome, "contract", "") == "governed_admission_v1"
|
|
1832
|
+
and not transport_accepted
|
|
1833
|
+
and outcome_provider_delivered is False
|
|
1834
|
+
and outcome_state in {"failed", "ambiguous"}
|
|
1835
|
+
and outcome_terminal is True
|
|
1836
|
+
)
|
|
1837
|
+
governed_boundary_attested = bool(
|
|
1838
|
+
governed_admitted or governed_delivered or governed_terminal_failure
|
|
1839
|
+
)
|
|
1840
|
+
if governed_admission_route and not governed_boundary_attested:
|
|
1841
|
+
logger.warning(
|
|
1842
|
+
"Governed gateway returned no permitted exact boundary attestation; "
|
|
1843
|
+
"treating the attempt as failed"
|
|
1844
|
+
)
|
|
1845
|
+
transport_accepted = False
|
|
1846
|
+
outcome_delivery_id = str(getattr(outcome, "delivery_id", "") or "")
|
|
1847
|
+
if governed_admitted and outcome_delivery_id:
|
|
1848
|
+
self._governed_delivery_replays[outcome_delivery_id] = copy.deepcopy(payload)
|
|
1849
|
+
elif (governed_delivered or governed_terminal_failure) and outcome_delivery_id:
|
|
1850
|
+
self._governed_delivery_replays.pop(outcome_delivery_id, None)
|
|
1851
|
+
provider_delivered = bool(
|
|
1852
|
+
governed_delivered
|
|
1853
|
+
if governed_admission_route
|
|
1854
|
+
else (
|
|
1855
|
+
transport_accepted
|
|
1856
|
+
and getattr(outcome, "provider_delivered", transport_accepted) is True
|
|
1857
|
+
)
|
|
1858
|
+
)
|
|
1859
|
+
|
|
1860
|
+
# Self-model delivery success means provider delivery, never admission
|
|
1861
|
+
# to an approval/dispatch queue. A rejected transport remains a real
|
|
1862
|
+
# failure; a governed pending admission is neither.
|
|
1863
|
+
sm = getattr(self._registry, "self_model", None)
|
|
1864
|
+
if sm is not None and observation_new:
|
|
1865
|
+
try:
|
|
1866
|
+
if provider_delivered:
|
|
1867
|
+
sm.record("delivery", "success")
|
|
1868
|
+
elif governed_terminal_failure or not transport_accepted:
|
|
1869
|
+
sm.record("delivery", "failure")
|
|
1870
|
+
except Exception:
|
|
1871
|
+
pass
|
|
1872
|
+
if provider_delivered and observation_new:
|
|
1873
|
+
# Consume the per-recipient rate budget so the 3/day + cooldown
|
|
1874
|
+
# caps actually bind (the push path previously never recorded).
|
|
1875
|
+
if rate_limiter is not None:
|
|
1876
|
+
try:
|
|
1877
|
+
rate_limiter.record_delivery(person_id)
|
|
1878
|
+
except Exception:
|
|
1879
|
+
logger.debug("record_delivery failed", exc_info=True)
|
|
1880
|
+
self.stats.actions_executed += 1
|
|
1881
|
+
self.stats.actions_this_hour += 1
|
|
1882
|
+
logger.info("Pushed initiative: %s -> %s", iid, person_id)
|
|
1883
|
+
try:
|
|
1884
|
+
from apsimo.api.routers.host import _telemetry
|
|
1885
|
+
if _telemetry is not None:
|
|
1886
|
+
await _telemetry.touch("last_initiative_at")
|
|
1887
|
+
except Exception:
|
|
1888
|
+
logger.warning("Telemetry touch failed (non-critical)")
|
|
1889
|
+
elif governed_admitted and observation_new:
|
|
1890
|
+
logger.info(
|
|
1891
|
+
"Admitted initiative pending provider delivery: %s -> %s (%s)",
|
|
1892
|
+
iid, person_id, getattr(outcome, "admission_state", "pending"),
|
|
1893
|
+
)
|
|
1894
|
+
elif governed_terminal_failure and observation_new:
|
|
1895
|
+
logger.warning(
|
|
1896
|
+
"Governed initiative reached terminal non-delivery: %s -> %s (%s)",
|
|
1897
|
+
iid, person_id, outcome_state,
|
|
1898
|
+
)
|
|
1899
|
+
return provider_delivered
|
|
1900
|
+
|
|
1901
|
+
def _observe_p8_outbound(self, payload: dict, preview: dict) -> None:
|
|
1902
|
+
"""Best-effort, synchronous journal hook for non-real-time text.
|
|
1903
|
+
|
|
1904
|
+
The runtime itself enforces explicit shadow mode and excludes all
|
|
1905
|
+
real-time voice surfaces. Exceptions remain advisory and therefore
|
|
1906
|
+
never change the established delivery result.
|
|
1907
|
+
"""
|
|
1908
|
+
runtime = getattr(self._registry, "p8", None)
|
|
1909
|
+
if runtime is None:
|
|
1910
|
+
return
|
|
1911
|
+
try:
|
|
1912
|
+
context = payload.get("context")
|
|
1913
|
+
context = context if isinstance(context, dict) else {}
|
|
1914
|
+
values = context.get("fact_refs") or payload.get("fact_refs") or ()
|
|
1915
|
+
if not isinstance(values, (list, tuple)):
|
|
1916
|
+
values = ()
|
|
1917
|
+
fact_refs = [
|
|
1918
|
+
str(value).strip()[:256]
|
|
1919
|
+
for value in values[:64]
|
|
1920
|
+
if str(value).strip()
|
|
1921
|
+
]
|
|
1922
|
+
target = preview.get("target")
|
|
1923
|
+
target = target if isinstance(target, dict) else {}
|
|
1924
|
+
payload_snapshot = {
|
|
1925
|
+
"id": str(payload.get("id") or "")[:256],
|
|
1926
|
+
# Strings are immutable; preserve the exact text that may ship
|
|
1927
|
+
# so the sample digest is truthful. The simulator's own size
|
|
1928
|
+
# bound may reject it after sampling, leaving incomplete
|
|
1929
|
+
# coverage rather than a false truncated evaluation.
|
|
1930
|
+
"title": str(payload.get("title") or ""),
|
|
1931
|
+
"description": str(payload.get("description") or ""),
|
|
1932
|
+
"context": {"fact_refs": fact_refs},
|
|
1933
|
+
}
|
|
1934
|
+
preview_snapshot = {
|
|
1935
|
+
"person_id": str(
|
|
1936
|
+
preview.get("person_id") or "")[:256],
|
|
1937
|
+
"target": {
|
|
1938
|
+
"user_chat": str(
|
|
1939
|
+
target.get("user_chat") or "")[:512],
|
|
1940
|
+
"home_chat": str(
|
|
1941
|
+
target.get("home_chat") or "")[:512],
|
|
1942
|
+
},
|
|
1943
|
+
}
|
|
1944
|
+
runtime.observe_outbound_payload(
|
|
1945
|
+
payload_snapshot, preview_snapshot)
|
|
1946
|
+
except Exception:
|
|
1947
|
+
logger.debug("P8 outbound shadow observation failed", exc_info=True)
|
|
1948
|
+
|
|
1949
|
+
def _get_rejection_store(self) -> Any:
|
|
1950
|
+
"""Lazy durable RejectionStore under the state dir (best-effort)."""
|
|
1951
|
+
if not hasattr(self, "_rejection_store"):
|
|
1952
|
+
self._rejection_store = None
|
|
1953
|
+
try:
|
|
1954
|
+
from pathlib import Path
|
|
1955
|
+
from apsimo.gate.rejection import RejectionStore
|
|
1956
|
+
state_dir = Path(os.environ.get(
|
|
1957
|
+
"COLONY_STATE_DIR", os.path.expanduser("~/.colony")))
|
|
1958
|
+
self._rejection_store = RejectionStore(
|
|
1959
|
+
str(state_dir / "colony-gate-rejections.db"))
|
|
1960
|
+
except Exception:
|
|
1961
|
+
logger.debug("RejectionStore unavailable", exc_info=True)
|
|
1962
|
+
return self._rejection_store
|
|
1963
|
+
|
|
1964
|
+
async def _run_rejection_feedback(self, text: str, guard_result: Any, *,
|
|
1965
|
+
person_id: str, gateway: str,
|
|
1966
|
+
iid: Any, authorized: bool) -> Optional[str]:
|
|
1967
|
+
"""One rejection-feedback cycle for a proactive message the guard
|
|
1968
|
+
blocked in enforce mode. Returns the guard-cleared revision, or None
|
|
1969
|
+
when no clean revision exists (the block stands). Regeneration uses
|
|
1970
|
+
the registry's LLM router when wired; without one the loop records
|
|
1971
|
+
the rejection and the block stands."""
|
|
1972
|
+
from apsimo.api.routers.host import _response_guard
|
|
1973
|
+
from apsimo.gate.rejection import RejectionFeedbackLoop
|
|
1974
|
+
if _response_guard is None:
|
|
1975
|
+
return None
|
|
1976
|
+
|
|
1977
|
+
regen = None
|
|
1978
|
+
llm = getattr(self._registry, "llm_router", None)
|
|
1979
|
+
if llm is not None and hasattr(llm, "complete"):
|
|
1980
|
+
async def regen(prompt_fragment: str, blocked_text: str):
|
|
1981
|
+
resp = await llm.complete(
|
|
1982
|
+
[{"role": "system",
|
|
1983
|
+
"content": ("You revise outbound messages that were "
|
|
1984
|
+
"blocked by a safety gate. Reply with ONLY "
|
|
1985
|
+
"the revised message text.")},
|
|
1986
|
+
{"role": "user",
|
|
1987
|
+
"content": (f"{prompt_fragment}\n\n"
|
|
1988
|
+
f"Original message:\n{blocked_text}")}],
|
|
1989
|
+
context={"task": "guard_revision"})
|
|
1990
|
+
return getattr(resp, "content", None)
|
|
1991
|
+
|
|
1992
|
+
floop = RejectionFeedbackLoop(
|
|
1993
|
+
_response_guard, store=self._get_rejection_store(),
|
|
1994
|
+
regenerate=regen)
|
|
1995
|
+
res = await floop.run(
|
|
1996
|
+
text, initial_result=guard_result,
|
|
1997
|
+
surface="proactive_text",
|
|
1998
|
+
target_contact_id=person_id, target_gateway=gateway,
|
|
1999
|
+
session_id=str(iid), turn_id=str(iid), authorized=authorized)
|
|
2000
|
+
return res.payload if (res.passed and res.payload) else None
|
|
2001
|
+
|
|
2002
|
+
async def _phase_observation_sync(self) -> None:
|
|
2003
|
+
"""Request fresh observations for stale domains (v0.16.0).
|
|
2004
|
+
|
|
2005
|
+
The agent is Colony's sensor array: when a domain's newest
|
|
2006
|
+
observation outlives its sync interval, post a read-only
|
|
2007
|
+
``agent_sync_<domain>`` job to the task queue. The agent claims
|
|
2008
|
+
it, observes through its own Hermes connections, and POSTs
|
|
2009
|
+
snapshots back to /v1/host/observations. Colony never calls
|
|
2010
|
+
external APIs itself.
|
|
2011
|
+
"""
|
|
2012
|
+
task_queue = getattr(self._registry, "task_queue", None)
|
|
2013
|
+
if task_queue is None:
|
|
2014
|
+
return
|
|
2015
|
+
try:
|
|
2016
|
+
from apsimo.api.routers.observations import get_observation_store
|
|
2017
|
+
obs_store = get_observation_store()
|
|
2018
|
+
except Exception:
|
|
2019
|
+
obs_store = None
|
|
2020
|
+
if obs_store is None:
|
|
2021
|
+
return
|
|
2022
|
+
|
|
2023
|
+
from apsimo.initiatives.action_registry import OBSERVATION_SYNC_ACTIONS
|
|
2024
|
+
from apsimo.observations.store import OBSERVATION_SYNC_INTERVALS
|
|
2025
|
+
|
|
2026
|
+
enabled = os.environ.get(
|
|
2027
|
+
"COLONY_SYNC_DOMAINS",
|
|
2028
|
+
"coding,task,calendar,research,project,system",
|
|
2029
|
+
)
|
|
2030
|
+
now = datetime.now(timezone.utc)
|
|
2031
|
+
|
|
2032
|
+
for domain in (d.strip() for d in enabled.split(",")):
|
|
2033
|
+
action = OBSERVATION_SYNC_ACTIONS.get(domain)
|
|
2034
|
+
if action is None:
|
|
2035
|
+
continue
|
|
2036
|
+
interval = OBSERVATION_SYNC_INTERVALS.get(domain, 3600)
|
|
2037
|
+
try:
|
|
2038
|
+
age = obs_store.domain_age_seconds(domain)
|
|
2039
|
+
except Exception:
|
|
2040
|
+
continue
|
|
2041
|
+
if age is not None and age < interval:
|
|
2042
|
+
continue
|
|
2043
|
+
last_request = self._last_sync_request.get(domain)
|
|
2044
|
+
if last_request and (now - last_request).total_seconds() < interval:
|
|
2045
|
+
continue # already asked; the agent may just be slow
|
|
2046
|
+
try:
|
|
2047
|
+
bucket = int(now.timestamp() // max(interval, 300))
|
|
2048
|
+
await task_queue.submit(
|
|
2049
|
+
task_type="agent_action",
|
|
2050
|
+
priority="normal",
|
|
2051
|
+
params={
|
|
2052
|
+
"action_hint": action,
|
|
2053
|
+
"domain": domain,
|
|
2054
|
+
"risk": "read_only",
|
|
2055
|
+
"description": (
|
|
2056
|
+
f"Observe the {domain} domain through your own "
|
|
2057
|
+
f"connections and report snapshots to Colony"
|
|
2058
|
+
),
|
|
2059
|
+
"report_to": "/v1/host/observations",
|
|
2060
|
+
"report_example": {
|
|
2061
|
+
"domain": domain,
|
|
2062
|
+
"reported_by": "<your agent id>",
|
|
2063
|
+
"observations": [
|
|
2064
|
+
{"entity_id": "<stable id>", "payload": {}}
|
|
2065
|
+
],
|
|
2066
|
+
},
|
|
2067
|
+
},
|
|
2068
|
+
idempotency_key=f"agent_sync:{domain}:{bucket}",
|
|
2069
|
+
)
|
|
2070
|
+
self._last_sync_request[domain] = now
|
|
2071
|
+
logger.info(
|
|
2072
|
+
"Requested %s observation sync (domain age: %s)",
|
|
2073
|
+
domain,
|
|
2074
|
+
f"{age:.0f}s" if age is not None else "never observed",
|
|
2075
|
+
)
|
|
2076
|
+
except Exception as exc:
|
|
2077
|
+
logger.warning("Observation sync request failed for %s: %s", domain, exc)
|
|
2078
|
+
|
|
2079
|
+
async def _post_agent_action_to_queue(
|
|
2080
|
+
self,
|
|
2081
|
+
initiative: Any,
|
|
2082
|
+
initiative_id: str,
|
|
2083
|
+
type_value: str,
|
|
2084
|
+
action_hint: str,
|
|
2085
|
+
) -> None:
|
|
2086
|
+
"""Post an AGENT_ACTION initiative to the task queue (v0.13.0).
|
|
2087
|
+
|
|
2088
|
+
Gated actions are posted as BLOCKED awaiting owner approval; the
|
|
2089
|
+
rest are posted as QUEUED for immediate claiming. What counts as
|
|
2090
|
+
gated depends on the immutable effect floor: every non-read-only
|
|
2091
|
+
mutation, disclosure, destructive, or outbound action requires a
|
|
2092
|
+
canonical direct decision or exact bounded grant. Graduated policy is
|
|
2093
|
+
retained for presentation compatibility, not execution authority.
|
|
2094
|
+
"""
|
|
2095
|
+
task_queue = getattr(self._registry, "task_queue", None)
|
|
2096
|
+
if task_queue is None:
|
|
2097
|
+
logger.warning("No task_queue available, skipping agent_action: %s", initiative_id)
|
|
2098
|
+
return
|
|
2099
|
+
|
|
2100
|
+
# v0.16.0: action_hint must be a named capability in the action
|
|
2101
|
+
# registry. Initiatives are built from graph data that can include
|
|
2102
|
+
# untrusted content — an unregistered hint NEVER reaches the queue.
|
|
2103
|
+
# The initiative stays stored (visible to the agent as information)
|
|
2104
|
+
# but nothing executes it.
|
|
2105
|
+
from apsimo.initiatives.action_registry import (
|
|
2106
|
+
RiskTier,
|
|
2107
|
+
classify_agent_action,
|
|
2108
|
+
get_approval_policy,
|
|
2109
|
+
)
|
|
2110
|
+
|
|
2111
|
+
policy = get_approval_policy()
|
|
2112
|
+
verdict = classify_agent_action(action_hint, policy=policy)
|
|
2113
|
+
if not verdict["executable"]:
|
|
2114
|
+
logger.warning(
|
|
2115
|
+
"action_hint %r is not in the action registry — initiative "
|
|
2116
|
+
"%s stored but NOT queued for execution",
|
|
2117
|
+
action_hint, initiative_id,
|
|
2118
|
+
)
|
|
2119
|
+
return
|
|
2120
|
+
|
|
2121
|
+
legacy_auto_approve_requested = (
|
|
2122
|
+
os.environ.get("COLONY_AGENT_AUTO_APPROVE", "false").lower()
|
|
2123
|
+
== "true"
|
|
2124
|
+
)
|
|
2125
|
+
|
|
2126
|
+
# Mirror _phase_execute's fallback: when there is no focused context for this type,
|
|
2127
|
+
# carry the initiative's own trigger_data. Agent_action initiatives (e.g. a deliverable)
|
|
2128
|
+
# stash their executable params there — including the recipient the graduated policy
|
|
2129
|
+
# resolves to decide auto-approval — so the worker and the gate both see them.
|
|
2130
|
+
action_context = self._build_initiative_context(initiative, type_value)
|
|
2131
|
+
if not action_context:
|
|
2132
|
+
trigger_data = getattr(initiative, "trigger_data", None)
|
|
2133
|
+
if trigger_data:
|
|
2134
|
+
action_context = dict(trigger_data)
|
|
2135
|
+
|
|
2136
|
+
job_payload = {
|
|
2137
|
+
"initiative_id": initiative_id,
|
|
2138
|
+
"action_hint": action_hint,
|
|
2139
|
+
"description": getattr(initiative, "description", ""),
|
|
2140
|
+
"entity_id": getattr(initiative, "entity_id", None),
|
|
2141
|
+
"risk": verdict["risk"],
|
|
2142
|
+
# Compatibility field retained for old workers, but the historical
|
|
2143
|
+
# env toggle is no longer authority and is always projected false.
|
|
2144
|
+
"auto_approve": False,
|
|
2145
|
+
"context": action_context,
|
|
2146
|
+
}
|
|
2147
|
+
|
|
2148
|
+
# A contact-store match is identity/context, not execution authority.
|
|
2149
|
+
# Outbound actions remain gated until a durable human/bounded decision
|
|
2150
|
+
# (phone or Operator Deck) is consumed. A future server-issued target
|
|
2151
|
+
# and transport attestation may safely restore a graduated fast path.
|
|
2152
|
+
|
|
2153
|
+
# Gated actions require HUMAN OWNER approval — the agent and legacy
|
|
2154
|
+
# environment toggles cannot approve mutations.
|
|
2155
|
+
is_gated = bool(verdict["requires_approval"])
|
|
2156
|
+
job_payload["destructive"] = is_gated # legacy field name, kept for workers
|
|
2157
|
+
if legacy_auto_approve_requested and is_gated:
|
|
2158
|
+
logger.warning(
|
|
2159
|
+
"Ignoring retired COLONY_AGENT_AUTO_APPROVE for effectful %s",
|
|
2160
|
+
action_hint,
|
|
2161
|
+
)
|
|
2162
|
+
|
|
2163
|
+
# v0.17.0: gated jobs are created directly in BLOCKED so no worker
|
|
2164
|
+
# can claim them in the window before a post-hoc transition lands.
|
|
2165
|
+
gate_pending = is_gated
|
|
2166
|
+
|
|
2167
|
+
# v0.18.0: non-read-only jobs that the POLICY (not the legacy env
|
|
2168
|
+
# bypass) waved through get a visible audit trail.
|
|
2169
|
+
policy_auto_pass = (
|
|
2170
|
+
not is_gated and verdict["risk"] != RiskTier.READ_ONLY.value
|
|
2171
|
+
)
|
|
2172
|
+
if gate_pending:
|
|
2173
|
+
job_tags = {"blocked_reason": "awaiting_owner_approval"}
|
|
2174
|
+
elif policy_auto_pass:
|
|
2175
|
+
job_tags = {
|
|
2176
|
+
"auto_approved_by_policy": policy,
|
|
2177
|
+
"risk": str(verdict["risk"]),
|
|
2178
|
+
}
|
|
2179
|
+
else:
|
|
2180
|
+
job_tags = None
|
|
2181
|
+
|
|
2182
|
+
try:
|
|
2183
|
+
from apsimo.task_queue.models import JobStatus
|
|
2184
|
+
|
|
2185
|
+
job_result = await task_queue.submit(
|
|
2186
|
+
task_type="agent_action",
|
|
2187
|
+
priority="high" if getattr(initiative, "priority", 0.5) > 0.7 else "normal",
|
|
2188
|
+
params=job_payload,
|
|
2189
|
+
idempotency_key=f"agent_action:{action_hint}:{getattr(initiative, 'entity_id', 'global')}",
|
|
2190
|
+
initial_status=JobStatus.BLOCKED if gate_pending else None,
|
|
2191
|
+
tags=job_tags,
|
|
2192
|
+
)
|
|
2193
|
+
job_id = job_result.get("id")
|
|
2194
|
+
logger.info("Posted agent_action job %s for initiative %s", job_id, initiative_id)
|
|
2195
|
+
|
|
2196
|
+
# QueueManager is the sole approval-at-birth owner. Read its
|
|
2197
|
+
# server-stamped result instead of independently consuming grants
|
|
2198
|
+
# or creating a second request in the autonomy loop.
|
|
2199
|
+
approval_request = None
|
|
2200
|
+
bounded_grant = None
|
|
2201
|
+
if gate_pending and job_id:
|
|
2202
|
+
stored_job = await task_queue.queue.get_job(job_id)
|
|
2203
|
+
stamped = dict(stored_job.tags or {}) if stored_job else {}
|
|
2204
|
+
if (
|
|
2205
|
+
stored_job is not None
|
|
2206
|
+
and stored_job.status is JobStatus.QUEUED
|
|
2207
|
+
and stamped.get("approval_provenance")
|
|
2208
|
+
== "server_bounded_grant"
|
|
2209
|
+
):
|
|
2210
|
+
bounded_grant = {
|
|
2211
|
+
"grant_id": stamped.get("bounded_grant_id"),
|
|
2212
|
+
}
|
|
2213
|
+
gate_pending = False
|
|
2214
|
+
elif (
|
|
2215
|
+
stored_job is not None
|
|
2216
|
+
and stored_job.status is JobStatus.QUEUED
|
|
2217
|
+
and stamped.get("approval_provenance")
|
|
2218
|
+
== "server_direct_decision"
|
|
2219
|
+
):
|
|
2220
|
+
gate_pending = False
|
|
2221
|
+
elif stamped.get("approval_request_id"):
|
|
2222
|
+
approval_request = {
|
|
2223
|
+
"request_id": stamped["approval_request_id"],
|
|
2224
|
+
"action_digest": stamped.get("action_digest"),
|
|
2225
|
+
}
|
|
2226
|
+
|
|
2227
|
+
if bounded_grant is not None and job_id:
|
|
2228
|
+
logger.info(
|
|
2229
|
+
"Bounded grant %s authorized %s job %s",
|
|
2230
|
+
bounded_grant["grant_id"], verdict["risk"], job_id,
|
|
2231
|
+
)
|
|
2232
|
+
try:
|
|
2233
|
+
from apsimo.events.broadcaster import emit as broadcast
|
|
2234
|
+
broadcast("action_auto_approved", {
|
|
2235
|
+
"job_id": job_id,
|
|
2236
|
+
"initiative_id": initiative_id,
|
|
2237
|
+
"action_hint": action_hint,
|
|
2238
|
+
"risk": verdict["risk"],
|
|
2239
|
+
"policy": "bounded_grant",
|
|
2240
|
+
"grant_id": bounded_grant["grant_id"],
|
|
2241
|
+
})
|
|
2242
|
+
except Exception:
|
|
2243
|
+
pass
|
|
2244
|
+
|
|
2245
|
+
if policy_auto_pass and job_id:
|
|
2246
|
+
logger.info(
|
|
2247
|
+
"Auto-approved %s job %s (%s: %s)",
|
|
2248
|
+
verdict["risk"], job_id, policy, verdict["reason"],
|
|
2249
|
+
)
|
|
2250
|
+
try:
|
|
2251
|
+
from apsimo.events.broadcaster import emit as broadcast
|
|
2252
|
+
broadcast("action_auto_approved", {
|
|
2253
|
+
"job_id": job_id,
|
|
2254
|
+
"initiative_id": initiative_id,
|
|
2255
|
+
"action_hint": action_hint,
|
|
2256
|
+
"risk": verdict["risk"],
|
|
2257
|
+
"policy": policy,
|
|
2258
|
+
"reason": verdict["reason"],
|
|
2259
|
+
})
|
|
2260
|
+
except Exception:
|
|
2261
|
+
pass
|
|
2262
|
+
|
|
2263
|
+
# Update initiative with job_id
|
|
2264
|
+
store = getattr(self._registry, "initiative_store", None)
|
|
2265
|
+
if store and job_id:
|
|
2266
|
+
try:
|
|
2267
|
+
loop = asyncio.get_event_loop()
|
|
2268
|
+
await loop.run_in_executor(
|
|
2269
|
+
None,
|
|
2270
|
+
lambda sid=initiative_id, jid=job_id: store.update(sid, job_id=jid, status="assigned"),
|
|
2271
|
+
)
|
|
2272
|
+
except Exception as exc:
|
|
2273
|
+
logger.warning("Failed to link initiative %s to job %s: %s", initiative_id, job_id, exc)
|
|
2274
|
+
|
|
2275
|
+
# mutating/outbound and not auto-approved → blocked awaiting owner
|
|
2276
|
+
if gate_pending and job_id:
|
|
2277
|
+
logger.info(
|
|
2278
|
+
"Blocked %s job %s awaiting owner approval",
|
|
2279
|
+
verdict["risk"], job_id,
|
|
2280
|
+
)
|
|
2281
|
+
# Push approval request to delivery
|
|
2282
|
+
delivery = self._registry.delivery
|
|
2283
|
+
if delivery and hasattr(delivery, "push_initiative"):
|
|
2284
|
+
# Approval requests must ALWAYS surface — proactive_delivery_enabled
|
|
2285
|
+
# gates Colony's *own* proactive outreach, not the owner's need to
|
|
2286
|
+
# unblock a gated job. Gating this dropped the request silently and
|
|
2287
|
+
# left the job blocked forever.
|
|
2288
|
+
await delivery.push_initiative({
|
|
2289
|
+
"id": initiative_id,
|
|
2290
|
+
"type": "agent_action",
|
|
2291
|
+
"priority": getattr(initiative, "priority", 0.5),
|
|
2292
|
+
"title": f"Approval required: {getattr(initiative, 'description', '')[:60]}",
|
|
2293
|
+
"description": getattr(initiative, "description", ""),
|
|
2294
|
+
"rationale": getattr(initiative, "rationale", ""),
|
|
2295
|
+
"suggested_action": "colony_approve_initiative",
|
|
2296
|
+
"entity_id": getattr(initiative, "entity_id", None),
|
|
2297
|
+
"channel_hint": "dm",
|
|
2298
|
+
"context": {
|
|
2299
|
+
"job_id": job_id,
|
|
2300
|
+
"action_hint": action_hint,
|
|
2301
|
+
"approval_request_id": (
|
|
2302
|
+
approval_request.get("request_id")
|
|
2303
|
+
if approval_request else None
|
|
2304
|
+
),
|
|
2305
|
+
"action_digest": (
|
|
2306
|
+
approval_request.get("action_digest")
|
|
2307
|
+
if approval_request else None
|
|
2308
|
+
),
|
|
2309
|
+
},
|
|
2310
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
2311
|
+
})
|
|
2312
|
+
|
|
2313
|
+
# Only count as executed if the job was not blocked awaiting approval
|
|
2314
|
+
if not gate_pending:
|
|
2315
|
+
self.stats.actions_executed += 1
|
|
2316
|
+
self.stats.actions_this_hour += 1
|
|
2317
|
+
except Exception as exc:
|
|
2318
|
+
logger.error("Failed to post agent_action to queue: %s", exc)
|
|
2319
|
+
|
|
2320
|
+
async def _feed_pending_tasks(self, engine: Any) -> None:
|
|
2321
|
+
"""Feed active goals as pending tasks, respecting cooldown.
|
|
2322
|
+
|
|
2323
|
+
Filters out abandoned and completed goals so the autonomy loop
|
|
2324
|
+
does not generate follow-up initiatives for dead tasks forever.
|
|
2325
|
+
An inferred proposal is awaiting acceptance, not an undertaking.
|
|
2326
|
+
"""
|
|
2327
|
+
goals = self._registry.goals
|
|
2328
|
+
if goals is None:
|
|
2329
|
+
return
|
|
2330
|
+
|
|
2331
|
+
try:
|
|
2332
|
+
from apsimo.cognition.goal_spine import cognition_spine_exclusive
|
|
2333
|
+
if cognition_spine_exclusive():
|
|
2334
|
+
engine.add_context("pending_tasks", [])
|
|
2335
|
+
return
|
|
2336
|
+
except Exception:
|
|
2337
|
+
pass
|
|
2338
|
+
|
|
2339
|
+
try:
|
|
2340
|
+
# Use get_active_tasks which respects cooldown and snooze (v0.7.10)
|
|
2341
|
+
cooldown_tasks = float(os.environ.get(
|
|
2342
|
+
"COLONY_INITIATIVE_COOLDOWN_TASKS", "12",
|
|
2343
|
+
))
|
|
2344
|
+
|
|
2345
|
+
if hasattr(goals, "get_active_tasks"):
|
|
2346
|
+
active = goals.get_active_tasks(cooldown_hours=cooldown_tasks)
|
|
2347
|
+
pending_tasks = []
|
|
2348
|
+
for goal in active:
|
|
2349
|
+
# Skip abandoned / completed / cancelled goals
|
|
2350
|
+
g_status = getattr(goal, "status", None)
|
|
2351
|
+
if g_status in ("abandoned", "completed", "cancelled"):
|
|
2352
|
+
continue
|
|
2353
|
+
if g_status == "proposed" and getattr(goal, "source", None) == "inferred":
|
|
2354
|
+
continue
|
|
2355
|
+
days_pending = 0
|
|
2356
|
+
if goal.created_at:
|
|
2357
|
+
days_pending = (datetime.now(timezone.utc) - goal.created_at).total_seconds() / 86400
|
|
2358
|
+
pending_tasks.append({
|
|
2359
|
+
"description": goal.title or "pending task",
|
|
2360
|
+
"days_pending": days_pending,
|
|
2361
|
+
"entity_id": goal.goal_id,
|
|
2362
|
+
})
|
|
2363
|
+
else:
|
|
2364
|
+
# Fallback for stores without get_active_tasks
|
|
2365
|
+
blocked = goals.list_goals(status="blocked", limit=20) if hasattr(goals, "list_goals") else []
|
|
2366
|
+
pending_tasks = []
|
|
2367
|
+
for goal in blocked:
|
|
2368
|
+
# Skip abandoned / completed / cancelled goals
|
|
2369
|
+
if isinstance(goal, dict):
|
|
2370
|
+
g_status = goal.get("status", "")
|
|
2371
|
+
else:
|
|
2372
|
+
g_status = getattr(goal, "status", "")
|
|
2373
|
+
if g_status in ("abandoned", "completed", "cancelled"):
|
|
2374
|
+
continue
|
|
2375
|
+
created = goal.created_at
|
|
2376
|
+
days_pending = 0
|
|
2377
|
+
if created:
|
|
2378
|
+
days_pending = (datetime.now(timezone.utc) - created).total_seconds() / 86400
|
|
2379
|
+
# Handle both dict and object representations
|
|
2380
|
+
if isinstance(goal, dict):
|
|
2381
|
+
entity_id = goal.get("context", {}).get("contact_id") if goal.get("context") else goal.get("goal_id")
|
|
2382
|
+
else:
|
|
2383
|
+
ctx = getattr(goal, "context", None)
|
|
2384
|
+
entity_id = ctx.get("contact_id") if ctx else getattr(goal, "goal_id", None)
|
|
2385
|
+
pending_tasks.append({
|
|
2386
|
+
"description": goal.title or "blocked goal",
|
|
2387
|
+
"days_pending": days_pending,
|
|
2388
|
+
"entity_id": entity_id,
|
|
2389
|
+
})
|
|
2390
|
+
|
|
2391
|
+
# Always set pending_tasks so graph loader doesn't fall back to stale
|
|
2392
|
+
# graph data when the SQL store has no active goals (Bug 41).
|
|
2393
|
+
engine.add_context("pending_tasks", pending_tasks)
|
|
2394
|
+
except Exception as e:
|
|
2395
|
+
logger.warning("Failed to feed pending tasks: %s", e)
|
|
2396
|
+
|
|
2397
|
+
async def _feed_neglected_contacts(self, engine: Any) -> None:
|
|
2398
|
+
"""Feed contacts with declining affect AND genuine neglect.
|
|
2399
|
+
|
|
2400
|
+
Combines affect-store signals with graph-based days-since-contact.
|
|
2401
|
+
Only feeds contacts that have both declining affect AND no recent
|
|
2402
|
+
interaction (≥7 days). Skips the host's own contact.
|
|
2403
|
+
"""
|
|
2404
|
+
affect = self._registry.affect_store
|
|
2405
|
+
if affect is None:
|
|
2406
|
+
return
|
|
2407
|
+
|
|
2408
|
+
# Owner exclusion is a relationship-domain policy (the agent must
|
|
2409
|
+
# not "check in with" its own operator) — fail closed when the
|
|
2410
|
+
# owner identity can't be established. Other domains (commitment,
|
|
2411
|
+
# calendar, agent_action) legitimately target the owner and must
|
|
2412
|
+
# NOT inherit this filter.
|
|
2413
|
+
from apsimo.identity.resolver import (
|
|
2414
|
+
OwnerIdentityError,
|
|
2415
|
+
get_identity_resolver,
|
|
2416
|
+
)
|
|
2417
|
+
resolver = get_identity_resolver()
|
|
2418
|
+
try:
|
|
2419
|
+
await resolver.owner_identities()
|
|
2420
|
+
except OwnerIdentityError as exc:
|
|
2421
|
+
logger.critical(
|
|
2422
|
+
"Owner identity unresolved — neglected-contact feed "
|
|
2423
|
+
"disabled (fail closed): %s", exc,
|
|
2424
|
+
)
|
|
2425
|
+
return
|
|
2426
|
+
|
|
2427
|
+
try:
|
|
2428
|
+
states = affect.get_all_states() if hasattr(affect, "get_all_states") else []
|
|
2429
|
+
neglected = []
|
|
2430
|
+
|
|
2431
|
+
for state in states[:20]:
|
|
2432
|
+
contact_id = state.get("contact_id")
|
|
2433
|
+
if not contact_id or await resolver.is_owner(contact_id):
|
|
2434
|
+
continue
|
|
2435
|
+
|
|
2436
|
+
# Only sustained decline, not a single bad event
|
|
2437
|
+
if hasattr(affect, "detect_sustained_decline"):
|
|
2438
|
+
if not affect.detect_sustained_decline(contact_id, min_events=3):
|
|
2439
|
+
continue
|
|
2440
|
+
else:
|
|
2441
|
+
# Fallback: declining trend + negative valence
|
|
2442
|
+
if state.get("trend") != "declining":
|
|
2443
|
+
continue
|
|
2444
|
+
if state.get("current_valence", 0) >= -0.3:
|
|
2445
|
+
continue
|
|
2446
|
+
|
|
2447
|
+
neglected.append({
|
|
2448
|
+
"name": contact_id,
|
|
2449
|
+
"entity_id": contact_id,
|
|
2450
|
+
"days_since_contact": 7, # minimum threshold; engine loads exact days from graph
|
|
2451
|
+
})
|
|
2452
|
+
|
|
2453
|
+
if neglected:
|
|
2454
|
+
engine.add_context("neglected_contacts", neglected)
|
|
2455
|
+
except Exception as e:
|
|
2456
|
+
logger.warning("Failed to feed neglected contacts: %s", e)
|
|
2457
|
+
|
|
2458
|
+
async def _feed_introduction_candidates(self, engine: Any) -> None:
|
|
2459
|
+
"""Feed pairs of contacts the agent might organically introduce.
|
|
2460
|
+
|
|
2461
|
+
Pairs share related work and both sit above the trust floor; the engine
|
|
2462
|
+
turns each into an OWNER-APPROVED introduction proposal (never an
|
|
2463
|
+
auto-executed action). Owner exclusion fails closed. Disabled by setting
|
|
2464
|
+
COLONY_INTROS_ENABLED=false; trust floor via COLONY_INTRO_TRUST_FLOOR.
|
|
2465
|
+
"""
|
|
2466
|
+
if os.environ.get("COLONY_INTROS_ENABLED", "true").lower() != "true":
|
|
2467
|
+
return
|
|
2468
|
+
contacts = getattr(self._registry, "contacts", None)
|
|
2469
|
+
if contacts is None or not hasattr(contacts, "introduction_candidates"):
|
|
2470
|
+
return
|
|
2471
|
+
|
|
2472
|
+
from apsimo.identity.resolver import (
|
|
2473
|
+
OwnerIdentityError,
|
|
2474
|
+
get_identity_resolver,
|
|
2475
|
+
get_owner_contact_id,
|
|
2476
|
+
)
|
|
2477
|
+
resolver = get_identity_resolver()
|
|
2478
|
+
try:
|
|
2479
|
+
await resolver.owner_identities()
|
|
2480
|
+
except OwnerIdentityError as exc:
|
|
2481
|
+
logger.critical(
|
|
2482
|
+
"Owner identity unresolved — introduction feed disabled "
|
|
2483
|
+
"(fail closed): %s", exc,
|
|
2484
|
+
)
|
|
2485
|
+
return
|
|
2486
|
+
|
|
2487
|
+
try:
|
|
2488
|
+
floor = os.environ.get("COLONY_INTRO_TRUST_FLOOR", "regular")
|
|
2489
|
+
candidates = await contacts.introduction_candidates(
|
|
2490
|
+
trust_floor=floor,
|
|
2491
|
+
owner_contact_id=get_owner_contact_id(),
|
|
2492
|
+
limit=10,
|
|
2493
|
+
)
|
|
2494
|
+
if candidates:
|
|
2495
|
+
engine.add_context("introduction_candidates", candidates)
|
|
2496
|
+
except Exception as e:
|
|
2497
|
+
logger.warning("Failed to feed introduction candidates: %s", e)
|
|
2498
|
+
|
|
2499
|
+
async def _feed_commitment_reminders(self, engine: Any) -> None:
|
|
2500
|
+
"""Feed upcoming/overdue commitments for COMMITMENT initiatives.
|
|
2501
|
+
|
|
2502
|
+
v0.16.0: commitments are first-class COMMITMENT initiatives
|
|
2503
|
+
(durable context, dedup ``commitment:{id}``) instead of being
|
|
2504
|
+
flattened into anonymous scheduling opportunities. The owner is a
|
|
2505
|
+
legitimate subject here.
|
|
2506
|
+
"""
|
|
2507
|
+
commitments = self._registry.commitment_store
|
|
2508
|
+
if commitments is None:
|
|
2509
|
+
return
|
|
2510
|
+
|
|
2511
|
+
try:
|
|
2512
|
+
# CommitmentStore.list() returns {"commitments": [...], "total": N}.
|
|
2513
|
+
# Include 'overdue': a flipped item is MORE deserving of an
|
|
2514
|
+
# initiative, not invisible to it.
|
|
2515
|
+
result = commitments.list(status=["pending", "overdue"], limit=20) if hasattr(commitments, "list") else {"commitments": []}
|
|
2516
|
+
active = result.get("commitments", [])
|
|
2517
|
+
|
|
2518
|
+
now = datetime.now(timezone.utc)
|
|
2519
|
+
upcoming = []
|
|
2520
|
+
|
|
2521
|
+
for c in active:
|
|
2522
|
+
due = c.get("due_at")
|
|
2523
|
+
if not due:
|
|
2524
|
+
continue
|
|
2525
|
+
|
|
2526
|
+
if isinstance(due, str):
|
|
2527
|
+
due = datetime.fromisoformat(due.replace("Z", "+00:00"))
|
|
2528
|
+
|
|
2529
|
+
hours_until = (due - now).total_seconds() / 3600
|
|
2530
|
+
|
|
2531
|
+
# Surface anything due in the next 48h, plus overdue
|
|
2532
|
+
# commitments up to a week old (they need follow-up most).
|
|
2533
|
+
if -168 < hours_until < 48:
|
|
2534
|
+
upcoming.append({
|
|
2535
|
+
"commitment_id": c.get("id"),
|
|
2536
|
+
"description": c.get("description", "untitled"),
|
|
2537
|
+
"due_at": due.isoformat(),
|
|
2538
|
+
"hours_until_due": hours_until,
|
|
2539
|
+
"overdue": hours_until <= 0,
|
|
2540
|
+
"status": c.get("status", "pending"),
|
|
2541
|
+
"person_id": c.get("person_id"),
|
|
2542
|
+
# carried through so the initiative engine can route a
|
|
2543
|
+
# deliverable (metadata.kind == "deliverable") to an
|
|
2544
|
+
# agent_action that actually SENDS it, vs a plain reminder.
|
|
2545
|
+
"metadata": c.get("metadata") or {},
|
|
2546
|
+
"source_type": c.get("source_type"),
|
|
2547
|
+
})
|
|
2548
|
+
|
|
2549
|
+
if upcoming:
|
|
2550
|
+
engine.add_context("upcoming_commitments", upcoming)
|
|
2551
|
+
except Exception as e:
|
|
2552
|
+
logger.warning("Failed to feed commitment reminders: %s", e)
|
|
2553
|
+
|
|
2554
|
+
async def _phase_job_writeback(self) -> None:
|
|
2555
|
+
"""Phase 6c (v0.17.0): close the act → learn loop.
|
|
2556
|
+
|
|
2557
|
+
Completed/failed agent jobs become episodic memories, advance
|
|
2558
|
+
their goals, complete their linked initiatives, and broadcast
|
|
2559
|
+
events. Before this phase, agent work landed in the queue DB and
|
|
2560
|
+
was invisible to memory — Colony could act but never learn from
|
|
2561
|
+
acting. Idempotent via the ``memory_synced`` job tag; a poison
|
|
2562
|
+
job is retried up to 3 ticks then tagged off.
|
|
2563
|
+
"""
|
|
2564
|
+
task_queue = getattr(self._registry, "task_queue", None)
|
|
2565
|
+
if task_queue is None:
|
|
2566
|
+
return
|
|
2567
|
+
qm = getattr(task_queue, "queue", None) or task_queue
|
|
2568
|
+
if not hasattr(qm, "get_jobs_by_status"):
|
|
2569
|
+
return
|
|
2570
|
+
from apsimo.task_queue.models import JobStatus
|
|
2571
|
+
|
|
2572
|
+
try:
|
|
2573
|
+
done = list(await qm.get_jobs_by_status(JobStatus.COMPLETED))
|
|
2574
|
+
done += list(await qm.get_jobs_by_status(JobStatus.FAILED))
|
|
2575
|
+
except Exception as exc:
|
|
2576
|
+
logger.debug("Job writeback: queue scan failed: %s", exc)
|
|
2577
|
+
return
|
|
2578
|
+
|
|
2579
|
+
synced = 0
|
|
2580
|
+
for job in done:
|
|
2581
|
+
tags = job.tags or {}
|
|
2582
|
+
if tags.get("memory_synced") == "true":
|
|
2583
|
+
continue
|
|
2584
|
+
if job.job_type != "agent_action":
|
|
2585
|
+
continue
|
|
2586
|
+
action_hint = (job.payload or {}).get("action_hint", "")
|
|
2587
|
+
if str(action_hint).startswith("agent_sync_"):
|
|
2588
|
+
# Observation syncs already land in the observation store;
|
|
2589
|
+
# recording them as memories would be routine-plumbing noise.
|
|
2590
|
+
await self._tag_job_synced(qm, job)
|
|
2591
|
+
continue
|
|
2592
|
+
try:
|
|
2593
|
+
await self._writeback_one_job(job)
|
|
2594
|
+
await self._tag_job_synced(qm, job)
|
|
2595
|
+
synced += 1
|
|
2596
|
+
except Exception as exc:
|
|
2597
|
+
attempts = int(tags.get("memory_sync_attempts", "0")) + 1
|
|
2598
|
+
logger.warning("Job writeback failed for %s (attempt %d): %s",
|
|
2599
|
+
job.job_id, attempts, exc)
|
|
2600
|
+
new_tags = {"memory_sync_attempts": str(attempts)}
|
|
2601
|
+
if attempts >= 3:
|
|
2602
|
+
new_tags["memory_synced"] = "true" # give up, stop retrying
|
|
2603
|
+
try:
|
|
2604
|
+
if hasattr(qm, "merge_job_tags"):
|
|
2605
|
+
await qm.merge_job_tags(job.job_id, new_tags)
|
|
2606
|
+
else:
|
|
2607
|
+
await qm.update_job_status(job.job_id, job.status,
|
|
2608
|
+
tags=new_tags)
|
|
2609
|
+
except Exception:
|
|
2610
|
+
pass
|
|
2611
|
+
if synced:
|
|
2612
|
+
logger.info("Phase job-writeback: %d agent job(s) fed back to memory",
|
|
2613
|
+
synced)
|
|
2614
|
+
|
|
2615
|
+
@staticmethod
|
|
2616
|
+
async def _tag_job_synced(qm: Any, job: Any) -> None:
|
|
2617
|
+
# Tag-only merge: the job is terminal (completed/failed), so
|
|
2618
|
+
# update_job_status would refuse it. merge_job_tags persists the
|
|
2619
|
+
# idempotency marker so this finished job is not re-written every
|
|
2620
|
+
# cycle. Falls back to update_job_status on older queue managers.
|
|
2621
|
+
# A failed tag write means this job WILL be re-processed next cycle —
|
|
2622
|
+
# log it loudly rather than silently looping (the bug class this guards).
|
|
2623
|
+
if hasattr(qm, "merge_job_tags"):
|
|
2624
|
+
ok = await qm.merge_job_tags(job.job_id, {"memory_synced": "true"})
|
|
2625
|
+
if ok is False:
|
|
2626
|
+
logger.warning(
|
|
2627
|
+
"Writeback could not persist memory_synced on job %s; it "
|
|
2628
|
+
"will be re-processed next cycle (re-written to memory).",
|
|
2629
|
+
job.job_id)
|
|
2630
|
+
else:
|
|
2631
|
+
await qm.update_job_status(job.job_id, job.status,
|
|
2632
|
+
tags={"memory_synced": "true"})
|
|
2633
|
+
|
|
2634
|
+
async def _writeback_one_job(self, job: Any) -> None:
|
|
2635
|
+
"""Propagate one finished agent job to goals, memory, initiatives."""
|
|
2636
|
+
result = job.result
|
|
2637
|
+
payload = job.payload or {}
|
|
2638
|
+
reported_succeeded = bool(result is not None and result.succeeded)
|
|
2639
|
+
tags = job.tags or {}
|
|
2640
|
+
operational_only = (
|
|
2641
|
+
tags.get("operational_completion_only") == "true"
|
|
2642
|
+
and tags.get("success_attested") != "true"
|
|
2643
|
+
)
|
|
2644
|
+
try:
|
|
2645
|
+
from apsimo.task_queue.governor import job_declares_effect
|
|
2646
|
+
effectful = job_declares_effect(job)
|
|
2647
|
+
except Exception:
|
|
2648
|
+
# Unknown classification cannot authorize downstream effects.
|
|
2649
|
+
effectful = True
|
|
2650
|
+
verification_pending = bool(
|
|
2651
|
+
reported_succeeded and operational_only and effectful
|
|
2652
|
+
)
|
|
2653
|
+
succeeded = bool(reported_succeeded and not verification_pending)
|
|
2654
|
+
action = payload.get("action_hint") or job.job_type
|
|
2655
|
+
description = payload.get("description", "")
|
|
2656
|
+
|
|
2657
|
+
# 1. Goal progress — the engine method existed since v0.13 but
|
|
2658
|
+
# nothing ever called it.
|
|
2659
|
+
goals = self._registry.goals
|
|
2660
|
+
if (goals is not None and result is not None
|
|
2661
|
+
and hasattr(goals, "on_job_completed")):
|
|
2662
|
+
output = result.output or {}
|
|
2663
|
+
if (
|
|
2664
|
+
not verification_pending
|
|
2665
|
+
and output.get("goal_id")
|
|
2666
|
+
and output.get("subtask_id")
|
|
2667
|
+
):
|
|
2668
|
+
try:
|
|
2669
|
+
goals.on_job_completed(result)
|
|
2670
|
+
except Exception as exc:
|
|
2671
|
+
logger.warning("Goal writeback failed for %s: %s",
|
|
2672
|
+
job.job_id, exc)
|
|
2673
|
+
|
|
2674
|
+
# 2. Episodic memory of what the agent did.
|
|
2675
|
+
graph = self._registry.graph
|
|
2676
|
+
if graph is not None and hasattr(graph, "store_memory"):
|
|
2677
|
+
if verification_pending:
|
|
2678
|
+
outcome = "reported completion; verification pending"
|
|
2679
|
+
else:
|
|
2680
|
+
outcome = "completed" if succeeded else (
|
|
2681
|
+
f"FAILED ({(result.error if result else None) or 'unknown error'})")
|
|
2682
|
+
summary = ""
|
|
2683
|
+
if result is not None and isinstance(result.output, dict):
|
|
2684
|
+
raw = result.output.get("summary") or result.output.get("result")
|
|
2685
|
+
if raw:
|
|
2686
|
+
summary = f" Result: {str(raw)[:300]}"
|
|
2687
|
+
content = (f"Agent {outcome} action '{action}'"
|
|
2688
|
+
+ (f" — {description}" if description else "")
|
|
2689
|
+
+ f".{summary}")
|
|
2690
|
+
await graph.store_memory(
|
|
2691
|
+
content=content,
|
|
2692
|
+
memory_type="episodic",
|
|
2693
|
+
entities=[],
|
|
2694
|
+
metadata={"job_id": job.job_id, "action_hint": str(action),
|
|
2695
|
+
"succeeded": succeeded,
|
|
2696
|
+
"verification_pending": verification_pending},
|
|
2697
|
+
importance=0.6 if succeeded else 0.7,
|
|
2698
|
+
source_type="tool_output",
|
|
2699
|
+
source_uri=f"colony://jobs/{job.job_id}",
|
|
2700
|
+
)
|
|
2701
|
+
|
|
2702
|
+
# 3. Linked initiative closure.
|
|
2703
|
+
initiative_id = payload.get("initiative_id")
|
|
2704
|
+
store = getattr(self._registry, "initiative_store", None)
|
|
2705
|
+
if initiative_id and store is not None:
|
|
2706
|
+
try:
|
|
2707
|
+
if succeeded and hasattr(store, "complete"):
|
|
2708
|
+
store.complete(initiative_id,
|
|
2709
|
+
agent_id=job.claimed_by or "agent",
|
|
2710
|
+
result=f"job {job.job_id} completed")
|
|
2711
|
+
elif not succeeded and not verification_pending and hasattr(store, "update"):
|
|
2712
|
+
store.update(initiative_id, status="failed",
|
|
2713
|
+
failed_reason=f"job {job.job_id} failed")
|
|
2714
|
+
except Exception as exc:
|
|
2715
|
+
logger.warning("Initiative closure failed for %s: %s",
|
|
2716
|
+
initiative_id, exc)
|
|
2717
|
+
|
|
2718
|
+
# 3b. Deliverable commitment fulfillment. A completed delivery flips its linked
|
|
2719
|
+
# commitment to fulfilled so it stops being re-surfaced; a FAILED one is left pending
|
|
2720
|
+
# so the next tick regenerates the agent_action and retries.
|
|
2721
|
+
if action == "agent_deliver_message" and succeeded:
|
|
2722
|
+
commitment_id = payload.get("entity_id")
|
|
2723
|
+
commitments = getattr(self._registry, "commitment_store", None)
|
|
2724
|
+
if commitment_id and commitments is not None and hasattr(commitments, "update"):
|
|
2725
|
+
try:
|
|
2726
|
+
commitments.update(
|
|
2727
|
+
commitment_id, status="fulfilled",
|
|
2728
|
+
fulfilled_at=datetime.now(timezone.utc).isoformat())
|
|
2729
|
+
except Exception as exc:
|
|
2730
|
+
logger.warning("Deliverable commitment %s fulfill failed: %s",
|
|
2731
|
+
commitment_id, exc)
|
|
2732
|
+
|
|
2733
|
+
# 4. Skill capture (v0.17.0, COLONY_ENABLE_SKILL_SYNTHESIS) — feed
|
|
2734
|
+
# successful novel work into the existing learning pipeline
|
|
2735
|
+
# (novelty gate → pattern extraction → DRAFT skill package).
|
|
2736
|
+
# Captured skills are DRAFT and deny-by-default; the v0.13
|
|
2737
|
+
# approval workflow gates activation, so nothing synthesized can
|
|
2738
|
+
# execute without the owner.
|
|
2739
|
+
if succeeded and not operational_only:
|
|
2740
|
+
await self._maybe_capture_skill(job, action, description)
|
|
2741
|
+
|
|
2742
|
+
# 5. Broadcast for anything listening (WS clients, audit log).
|
|
2743
|
+
try:
|
|
2744
|
+
from apsimo.events.broadcaster import emit as broadcast
|
|
2745
|
+
event_type = (
|
|
2746
|
+
"job_verification_pending"
|
|
2747
|
+
if verification_pending
|
|
2748
|
+
else ("job_completed" if succeeded else "job_failed")
|
|
2749
|
+
)
|
|
2750
|
+
broadcast(event_type,
|
|
2751
|
+
{"job_id": job.job_id, "action_hint": str(action),
|
|
2752
|
+
"initiative_id": initiative_id,
|
|
2753
|
+
"verification_pending": verification_pending})
|
|
2754
|
+
except Exception:
|
|
2755
|
+
pass
|
|
2756
|
+
|
|
2757
|
+
def _get_skill_learning(self) -> Any:
|
|
2758
|
+
"""Lazily build the SkillLearningService (or None if disabled)."""
|
|
2759
|
+
if os.environ.get("COLONY_ENABLE_SKILL_SYNTHESIS",
|
|
2760
|
+
"false").lower() != "true":
|
|
2761
|
+
return None
|
|
2762
|
+
service = getattr(self, "_skill_learning", None)
|
|
2763
|
+
if service is not None:
|
|
2764
|
+
return service
|
|
2765
|
+
skills_registry = self._registry.skills
|
|
2766
|
+
if skills_registry is None:
|
|
2767
|
+
return None
|
|
2768
|
+
try:
|
|
2769
|
+
import pathlib
|
|
2770
|
+
|
|
2771
|
+
from apsimo.skills.learning import (
|
|
2772
|
+
NoveltyDetector,
|
|
2773
|
+
PatternExtractor,
|
|
2774
|
+
SkillLearningService,
|
|
2775
|
+
)
|
|
2776
|
+
from apsimo.skills.packager import SkillPackager
|
|
2777
|
+
|
|
2778
|
+
library = pathlib.Path(
|
|
2779
|
+
os.environ.get("COLONY_SKILL_LIBRARY")
|
|
2780
|
+
or os.path.join(os.environ.get("COLONY_STATE_DIR", "."),
|
|
2781
|
+
"skill_library"))
|
|
2782
|
+
packager = SkillPackager(
|
|
2783
|
+
registry=skills_registry,
|
|
2784
|
+
colony_id=os.environ.get("COLONY_NODE_ID", "colony"),
|
|
2785
|
+
library_root=library,
|
|
2786
|
+
)
|
|
2787
|
+
service = SkillLearningService(
|
|
2788
|
+
detector=NoveltyDetector(skills_registry),
|
|
2789
|
+
extractor=PatternExtractor(),
|
|
2790
|
+
packager=packager,
|
|
2791
|
+
)
|
|
2792
|
+
self._skill_learning = service
|
|
2793
|
+
logger.info("Skill synthesis enabled (library=%s)", library)
|
|
2794
|
+
return service
|
|
2795
|
+
except Exception as exc:
|
|
2796
|
+
logger.warning("Skill synthesis unavailable: %s", exc)
|
|
2797
|
+
self._skill_learning = None
|
|
2798
|
+
return None
|
|
2799
|
+
|
|
2800
|
+
async def _maybe_capture_skill(self, job: Any, action: str,
|
|
2801
|
+
description: str) -> None:
|
|
2802
|
+
service = self._get_skill_learning()
|
|
2803
|
+
if service is None:
|
|
2804
|
+
return
|
|
2805
|
+
try:
|
|
2806
|
+
from datetime import datetime, timezone
|
|
2807
|
+
|
|
2808
|
+
from apsimo.skills.learning.triggers import (
|
|
2809
|
+
LearningTriggerEvent,
|
|
2810
|
+
TriggerSource,
|
|
2811
|
+
)
|
|
2812
|
+
from apsimo.skills.models import TaskSolution
|
|
2813
|
+
|
|
2814
|
+
result = job.result
|
|
2815
|
+
output = (result.output or {}) if result is not None else {}
|
|
2816
|
+
solution = TaskSolution(
|
|
2817
|
+
task_id=job.job_id,
|
|
2818
|
+
task_description=description or str(action),
|
|
2819
|
+
inputs=dict(job.payload or {}),
|
|
2820
|
+
output=output,
|
|
2821
|
+
trace=list(output.get("trace", [])),
|
|
2822
|
+
dependencies=[],
|
|
2823
|
+
embedding=None,
|
|
2824
|
+
step_fingerprint=None,
|
|
2825
|
+
duration_secs=float(
|
|
2826
|
+
getattr(result, "duration_seconds", None) or 0.0),
|
|
2827
|
+
completed_at=getattr(result, "completed_at", None)
|
|
2828
|
+
or datetime.now(timezone.utc),
|
|
2829
|
+
)
|
|
2830
|
+
skill_id = await service.handle(LearningTriggerEvent(
|
|
2831
|
+
source=TriggerSource.POST_TASK_HOOK, solution=solution))
|
|
2832
|
+
if skill_id:
|
|
2833
|
+
self._queue_deferred_initiative(skill_id, description or action)
|
|
2834
|
+
except Exception as exc:
|
|
2835
|
+
logger.warning("Skill capture failed for %s: %s", job.job_id, exc)
|
|
2836
|
+
|
|
2837
|
+
def _queue_deferred_initiative(self, skill_id: str, task_desc: str) -> None:
|
|
2838
|
+
"""Surface a captured DRAFT skill to the owner next tick."""
|
|
2839
|
+
from apsimo.intelligence.components.initiative_engine import (
|
|
2840
|
+
Initiative,
|
|
2841
|
+
InitiativeType,
|
|
2842
|
+
)
|
|
2843
|
+
deferred = getattr(self, "_deferred_initiatives", None)
|
|
2844
|
+
if deferred is None:
|
|
2845
|
+
deferred = []
|
|
2846
|
+
self._deferred_initiatives = deferred
|
|
2847
|
+
deferred.append(Initiative(
|
|
2848
|
+
id=f"init-skill-{skill_id[:24]}",
|
|
2849
|
+
type=InitiativeType.CAPABILITY_GAP,
|
|
2850
|
+
description=f"Review new draft skill '{skill_id}' captured from: "
|
|
2851
|
+
f"{task_desc[:120]}",
|
|
2852
|
+
priority=0.7,
|
|
2853
|
+
rationale="[skill synthesis] novel successful work was captured "
|
|
2854
|
+
"as a DRAFT skill; it cannot run until you approve it.",
|
|
2855
|
+
action_hint=None,
|
|
2856
|
+
dedup_key=f"skill_review:{skill_id}",
|
|
2857
|
+
))
|
|
2858
|
+
|
|
2859
|
+
async def _phase_cognition(self) -> None:
|
|
2860
|
+
"""Run cognition pipeline tick."""
|
|
2861
|
+
cognition = self._registry.cognition
|
|
2862
|
+
if cognition is None:
|
|
2863
|
+
return
|
|
2864
|
+
try:
|
|
2865
|
+
if hasattr(cognition, "run_cycle"):
|
|
2866
|
+
result = await cognition.run_cycle()
|
|
2867
|
+
# run_cycle() catches each internal step's failure into
|
|
2868
|
+
# result.errors and returns "successfully"; without inspecting
|
|
2869
|
+
# them the self-improvement loop can be fully degraded while this
|
|
2870
|
+
# phase reports a clean tick. Surface them (errors gate nothing —
|
|
2871
|
+
# this is purely observability).
|
|
2872
|
+
cycle_errors = list(getattr(result, "errors", None) or [])
|
|
2873
|
+
if cycle_errors:
|
|
2874
|
+
self.stats.errors += len(cycle_errors)
|
|
2875
|
+
logger.warning(
|
|
2876
|
+
"Phase cognition: cycle completed with %d step error(s): %s",
|
|
2877
|
+
len(cycle_errors),
|
|
2878
|
+
"; ".join(str(e) for e in cycle_errors[:5]),
|
|
2879
|
+
)
|
|
2880
|
+
else:
|
|
2881
|
+
logger.debug("Phase cognition: cycle complete")
|
|
2882
|
+
except Exception as exc:
|
|
2883
|
+
self.stats.errors += 1
|
|
2884
|
+
logger.error("Phase cognition error: %s", exc, exc_info=True)
|
|
2885
|
+
|
|
2886
|
+
async def _phase_projects(self) -> None:
|
|
2887
|
+
"""Phase 6a: sustained multi-tick project pursuit (cognition item 1).
|
|
2888
|
+
|
|
2889
|
+
The engine plans (LLM, validated), boundary-checks and advances one
|
|
2890
|
+
ready step per due project; every step dispatch routes through that
|
|
2891
|
+
action kind's own gated sub-path. Shadow mode simulates and logs.
|
|
2892
|
+
"""
|
|
2893
|
+
try:
|
|
2894
|
+
from apsimo.projects.models import projects_mode
|
|
2895
|
+
if projects_mode() == "off":
|
|
2896
|
+
return
|
|
2897
|
+
except Exception:
|
|
2898
|
+
return
|
|
2899
|
+
engine = getattr(self._registry, "project_engine", None)
|
|
2900
|
+
if engine is None:
|
|
2901
|
+
return
|
|
2902
|
+
try:
|
|
2903
|
+
report = await engine.tick()
|
|
2904
|
+
if (report.get("adopted") or report.get("planned")
|
|
2905
|
+
or report.get("steps_dispatched")):
|
|
2906
|
+
logger.info("Phase projects[%s]: adopted=%d planned=%d steps=%d",
|
|
2907
|
+
report.get("mode"), report.get("adopted", 0),
|
|
2908
|
+
report.get("planned", 0),
|
|
2909
|
+
report.get("steps_dispatched", 0))
|
|
2910
|
+
except Exception as exc:
|
|
2911
|
+
self.stats.errors += 1
|
|
2912
|
+
logger.error("Phase projects error: %s", exc, exc_info=True)
|
|
2913
|
+
|
|
2914
|
+
async def _phase_project_result_reconciliation(self) -> None:
|
|
2915
|
+
"""Bounded early projection of already-terminal WorkOrder results."""
|
|
2916
|
+
|
|
2917
|
+
engine = getattr(self._registry, "project_engine", None)
|
|
2918
|
+
reconcile = getattr(engine, "reconcile_terminal_results", None)
|
|
2919
|
+
if not callable(reconcile):
|
|
2920
|
+
return
|
|
2921
|
+
try:
|
|
2922
|
+
raw_limit = int(os.environ.get(
|
|
2923
|
+
"COLONY_PROJECT_RECONCILIATION_LIMIT", "25",
|
|
2924
|
+
))
|
|
2925
|
+
except (TypeError, ValueError):
|
|
2926
|
+
raw_limit = 25
|
|
2927
|
+
limit = max(1, min(100, raw_limit))
|
|
2928
|
+
try:
|
|
2929
|
+
raw_budget = float(os.environ.get(
|
|
2930
|
+
"COLONY_PROJECT_RECONCILIATION_BUDGET_SECS", "5",
|
|
2931
|
+
))
|
|
2932
|
+
except (TypeError, ValueError):
|
|
2933
|
+
raw_budget = 5.0
|
|
2934
|
+
# Preserve most of the whole-tick budget for ordinary cognition while
|
|
2935
|
+
# guaranteeing reconciliation gets the first bounded slice.
|
|
2936
|
+
budget = max(0.05, min(10.0, raw_budget,
|
|
2937
|
+
self._tick_budget_secs() * 0.25))
|
|
2938
|
+
try:
|
|
2939
|
+
report = await asyncio.wait_for(
|
|
2940
|
+
reconcile(limit=limit), timeout=budget,
|
|
2941
|
+
)
|
|
2942
|
+
except asyncio.TimeoutError:
|
|
2943
|
+
self.stats.errors += 1
|
|
2944
|
+
logger.warning(
|
|
2945
|
+
"project result reconciliation exceeded %.2fs budget; "
|
|
2946
|
+
"durable rows remain retryable", budget,
|
|
2947
|
+
)
|
|
2948
|
+
return
|
|
2949
|
+
except Exception as exc:
|
|
2950
|
+
self.stats.errors += 1
|
|
2951
|
+
logger.error(
|
|
2952
|
+
"Phase project result reconciliation error: %s",
|
|
2953
|
+
exc, exc_info=True,
|
|
2954
|
+
)
|
|
2955
|
+
return
|
|
2956
|
+
errors = int(report.get("errors") or 0) if isinstance(report, dict) else 0
|
|
2957
|
+
self.stats.errors += errors
|
|
2958
|
+
if isinstance(report, dict) and (
|
|
2959
|
+
report.get("projected") or errors
|
|
2960
|
+
):
|
|
2961
|
+
logger.info(
|
|
2962
|
+
"Phase project-result reconciliation: checked=%d "
|
|
2963
|
+
"terminal=%d projected=%d errors=%d",
|
|
2964
|
+
int(report.get("checked") or 0),
|
|
2965
|
+
int(report.get("terminal") or 0),
|
|
2966
|
+
int(report.get("projected") or 0),
|
|
2967
|
+
errors,
|
|
2968
|
+
)
|
|
2969
|
+
|
|
2970
|
+
async def _phase_trust_notices(self) -> None:
|
|
2971
|
+
"""Deliver trust-engine graduation/demotion notices to the owner
|
|
2972
|
+
(Amendment 1.2: notifications, not permission requests). The queue is
|
|
2973
|
+
durable: a notice retries every tick until a real delivery succeeds,
|
|
2974
|
+
surviving restarts and rate-limit windows."""
|
|
2975
|
+
sm = getattr(self._registry, "self_model", None)
|
|
2976
|
+
trust = getattr(sm, "trust", None) if sm is not None else None
|
|
2977
|
+
if trust is None or not hasattr(trust, "undelivered_notices"):
|
|
2978
|
+
return
|
|
2979
|
+
delivery = self._registry.delivery
|
|
2980
|
+
if delivery is None:
|
|
2981
|
+
return
|
|
2982
|
+
try:
|
|
2983
|
+
from apsimo.proposals import Proposal, proposal_to_payload
|
|
2984
|
+
except Exception:
|
|
2985
|
+
return
|
|
2986
|
+
for n in trust.undelivered_notices(limit=3):
|
|
2987
|
+
try:
|
|
2988
|
+
if n.get("demotion"):
|
|
2989
|
+
title = f"Autonomy pulled back: {n['domain']}"
|
|
2990
|
+
finding = (
|
|
2991
|
+
f"I demoted myself to ask-first on {n['domain']}: "
|
|
2992
|
+
f"{n.get('reason', 'circuit breaker')}. I will ask "
|
|
2993
|
+
"before doing this class of work again.")
|
|
2994
|
+
else:
|
|
2995
|
+
stage_txt = ("asking you first before"
|
|
2996
|
+
if n.get("stage") == "ask_first"
|
|
2997
|
+
else "handling autonomously")
|
|
2998
|
+
title = f"Autonomy update: {n['domain']}"
|
|
2999
|
+
finding = (
|
|
3000
|
+
f"My track record on {n['domain']} crossed the "
|
|
3001
|
+
f"threshold ({n.get('reason', '')}), so I am now "
|
|
3002
|
+
f"{stage_txt} this class of work. Say stop if you "
|
|
3003
|
+
"do not want that.")
|
|
3004
|
+
prop = Proposal(
|
|
3005
|
+
title=title[:100], finding=finding,
|
|
3006
|
+
why_it_helps="you always know exactly what I do on my own",
|
|
3007
|
+
suggested_action="Say 'stop acting' any time to pause "
|
|
3008
|
+
"all autonomy.",
|
|
3009
|
+
source="trust-engine", initiative_type="proposal",
|
|
3010
|
+
confidence=0.85)
|
|
3011
|
+
delivered = await self._route_reachout_delivery(
|
|
3012
|
+
proposal_to_payload(prop), delivery)
|
|
3013
|
+
if delivered:
|
|
3014
|
+
trust.mark_notice_delivered(n["id"])
|
|
3015
|
+
pstore = getattr(self._registry, "proposal_store", None)
|
|
3016
|
+
if pstore is not None:
|
|
3017
|
+
prop.status = "delivered"
|
|
3018
|
+
pstore.add(prop)
|
|
3019
|
+
except Exception:
|
|
3020
|
+
logger.debug("trust notice delivery failed", exc_info=True)
|
|
3021
|
+
|
|
3022
|
+
async def _phase_selfhood_benchmark(self) -> None:
|
|
3023
|
+
"""Weekly (Mind M0a): compute the previous week's selfhood-benchmark
|
|
3024
|
+
rollups and deliver the scorecard to the owner. Own weekly dedup
|
|
3025
|
+
(not _run_periodic_phase) because the benchmark must run even when
|
|
3026
|
+
graph memory is absent; the dedup key is only advanced on success."""
|
|
3027
|
+
bench = getattr(self._registry, "benchmark", None)
|
|
3028
|
+
if bench is None:
|
|
3029
|
+
return
|
|
3030
|
+
key = datetime.now(timezone.utc).strftime("%Y-W%W")
|
|
3031
|
+
if self._periodic_last.get("selfhood_benchmark") == key:
|
|
3032
|
+
return
|
|
3033
|
+
try:
|
|
3034
|
+
# compute_week runs recall probes against the graph; bound it so a
|
|
3035
|
+
# wedged graph connection can't stall the tick.
|
|
3036
|
+
with self._periodic_attempt("selfhood_benchmark", key):
|
|
3037
|
+
result = await asyncio.wait_for(
|
|
3038
|
+
bench.compute_week(), timeout=self._phase_budget_secs())
|
|
3039
|
+
except asyncio.TimeoutError:
|
|
3040
|
+
# counted as this week's attempt: retrying every tick for the rest
|
|
3041
|
+
# of the week would spend a phase budget per tick on the same stall
|
|
3042
|
+
self._periodic_last["selfhood_benchmark"] = key
|
|
3043
|
+
logger.warning("selfhood_benchmark exceeded budget; skipping "
|
|
3044
|
+
"until next week")
|
|
3045
|
+
return
|
|
3046
|
+
except Exception as exc:
|
|
3047
|
+
self.stats.errors += 1
|
|
3048
|
+
logger.error("Phase selfhood_benchmark error: %s", exc,
|
|
3049
|
+
exc_info=True)
|
|
3050
|
+
return
|
|
3051
|
+
if os.environ.get("COLONY_BENCHMARK_REPORT",
|
|
3052
|
+
"true").strip().lower() == "false":
|
|
3053
|
+
return
|
|
3054
|
+
metrics = result.get("metrics") or {}
|
|
3055
|
+
delivery = self._registry.delivery
|
|
3056
|
+
if not metrics or delivery is None:
|
|
3057
|
+
return
|
|
3058
|
+
try:
|
|
3059
|
+
from apsimo.proposals import Proposal, proposal_to_payload
|
|
3060
|
+
except Exception:
|
|
3061
|
+
return
|
|
3062
|
+
try:
|
|
3063
|
+
trends = bench.snapshot(weeks=2).get("trends", {})
|
|
3064
|
+
lines = []
|
|
3065
|
+
for m, r in sorted(metrics.items()):
|
|
3066
|
+
v = r.get("value")
|
|
3067
|
+
if v is None:
|
|
3068
|
+
continue
|
|
3069
|
+
d = trends.get(m)
|
|
3070
|
+
wow = "" if d is None else f" ({'+' if d >= 0 else ''}{d:.2f} wow)"
|
|
3071
|
+
lines.append(f"{m} {v:.2f}{wow}")
|
|
3072
|
+
prop = Proposal(
|
|
3073
|
+
title=f"Weekly selfhood report {result.get('week')}"[:100],
|
|
3074
|
+
finding="; ".join(lines)[:900],
|
|
3075
|
+
why_it_helps="a falsifiable trend line on whether I am "
|
|
3076
|
+
"actually getting better, straight from my "
|
|
3077
|
+
"journals",
|
|
3078
|
+
suggested_action="Reply if any line looks wrong; every "
|
|
3079
|
+
"number is derived from recorded outcomes.",
|
|
3080
|
+
source="selfhood-benchmark", initiative_type="proposal",
|
|
3081
|
+
confidence=0.9)
|
|
3082
|
+
await self._route_reachout_delivery(
|
|
3083
|
+
proposal_to_payload(prop), delivery)
|
|
3084
|
+
except Exception:
|
|
3085
|
+
logger.debug("benchmark report delivery failed", exc_info=True)
|
|
3086
|
+
|
|
3087
|
+
async def _phase_workspace(self) -> None:
|
|
3088
|
+
"""Every few ticks (Mind M2): feed concerns from live signals, decay
|
|
3089
|
+
the salience field, and run bounded thinking jobs. More thinking runs
|
|
3090
|
+
during the sleep window when the cluster is idle."""
|
|
3091
|
+
ws = getattr(self._registry, "workspace", None)
|
|
3092
|
+
if ws is None:
|
|
3093
|
+
return
|
|
3094
|
+
try:
|
|
3095
|
+
from apsimo.self_model.workspace import (
|
|
3096
|
+
in_sleep_window, workspace_mode,
|
|
3097
|
+
)
|
|
3098
|
+
except Exception:
|
|
3099
|
+
return
|
|
3100
|
+
# During migration, polling remains only when the durable reducer is
|
|
3101
|
+
# absent/off. Re-polling the same live state otherwise manufactures
|
|
3102
|
+
# salience without a material event.
|
|
3103
|
+
reducer = getattr(ws, "event_reducer", None)
|
|
3104
|
+
if reducer is None or getattr(reducer, "mode", "off") == "off":
|
|
3105
|
+
try:
|
|
3106
|
+
self._workspace_ingest(ws)
|
|
3107
|
+
except Exception:
|
|
3108
|
+
logger.debug("workspace ingest failed", exc_info=True)
|
|
3109
|
+
# decay + evict
|
|
3110
|
+
try:
|
|
3111
|
+
ws.decay()
|
|
3112
|
+
except Exception:
|
|
3113
|
+
logger.debug("workspace decay failed", exc_info=True)
|
|
3114
|
+
# think: 1 per pass normally, a few during the sleep window. Each
|
|
3115
|
+
# thought calls the LLM; bound it so a slow model can't stall the tick.
|
|
3116
|
+
rounds = 4 if in_sleep_window() else 1
|
|
3117
|
+
live = workspace_mode() == "live"
|
|
3118
|
+
try:
|
|
3119
|
+
from apsimo.cognition.goal_spine import (
|
|
3120
|
+
cognition_spine_enabled, cognition_spine_exclusive,
|
|
3121
|
+
)
|
|
3122
|
+
if cognition_spine_enabled():
|
|
3123
|
+
spine = getattr(ws, "cognition_spine", None)
|
|
3124
|
+
if spine is None:
|
|
3125
|
+
logger.warning("P3 cognition spine enabled but not wired")
|
|
3126
|
+
if cognition_spine_exclusive():
|
|
3127
|
+
return
|
|
3128
|
+
else:
|
|
3129
|
+
for _ in range(rounds):
|
|
3130
|
+
try:
|
|
3131
|
+
result = await asyncio.wait_for(
|
|
3132
|
+
spine.run_once(), timeout=self._phase_budget_secs())
|
|
3133
|
+
except asyncio.TimeoutError:
|
|
3134
|
+
logger.warning("P3 thought phase exceeded budget; stopping")
|
|
3135
|
+
break
|
|
3136
|
+
if result.get("status") in {"idle", "off"}:
|
|
3137
|
+
break
|
|
3138
|
+
return
|
|
3139
|
+
except Exception:
|
|
3140
|
+
logger.exception("P3 workspace phase failed")
|
|
3141
|
+
return
|
|
3142
|
+
for _ in range(rounds):
|
|
3143
|
+
try:
|
|
3144
|
+
outcome = await asyncio.wait_for(
|
|
3145
|
+
ws.think_once(), timeout=self._phase_budget_secs())
|
|
3146
|
+
except asyncio.TimeoutError:
|
|
3147
|
+
logger.warning("workspace thought exceeded budget; stopping")
|
|
3148
|
+
break
|
|
3149
|
+
if outcome is None:
|
|
3150
|
+
break
|
|
3151
|
+
if live and outcome.get("action"):
|
|
3152
|
+
try:
|
|
3153
|
+
await self._workspace_act(outcome["action"])
|
|
3154
|
+
except Exception:
|
|
3155
|
+
logger.debug("workspace act failed", exc_info=True)
|
|
3156
|
+
|
|
3157
|
+
async def _workspace_act(self, action: dict) -> None:
|
|
3158
|
+
"""Live-mode: turn a thought's action into a real, gated effect. An
|
|
3159
|
+
initiative surfaces to the owner (through the same reachout gates);
|
|
3160
|
+
an experiment is proposed to the experiment framework."""
|
|
3161
|
+
try:
|
|
3162
|
+
from apsimo.cognition.goal_spine import cognition_spine_exclusive
|
|
3163
|
+
if cognition_spine_exclusive():
|
|
3164
|
+
return
|
|
3165
|
+
except Exception:
|
|
3166
|
+
pass
|
|
3167
|
+
kind = (action or {}).get("kind")
|
|
3168
|
+
if kind == "initiative":
|
|
3169
|
+
delivery = self._registry.delivery
|
|
3170
|
+
if delivery is None:
|
|
3171
|
+
return
|
|
3172
|
+
try:
|
|
3173
|
+
from apsimo.proposals import Proposal, proposal_to_payload
|
|
3174
|
+
prop = Proposal(
|
|
3175
|
+
title=str(action.get("title", "A thought"))[:100],
|
|
3176
|
+
finding=str(action.get("detail", ""))[:600],
|
|
3177
|
+
why_it_helps="something on my mind that seemed worth "
|
|
3178
|
+
"raising with you",
|
|
3179
|
+
suggested_action="No action needed unless you want to "
|
|
3180
|
+
"weigh in.",
|
|
3181
|
+
source="workspace", initiative_type="proposal",
|
|
3182
|
+
confidence=0.7)
|
|
3183
|
+
await self._route_reachout_delivery(
|
|
3184
|
+
proposal_to_payload(prop), delivery)
|
|
3185
|
+
except Exception:
|
|
3186
|
+
logger.debug("workspace initiative failed", exc_info=True)
|
|
3187
|
+
elif kind == "experiment":
|
|
3188
|
+
engine = getattr(self._registry, "experiments", None)
|
|
3189
|
+
if engine is None:
|
|
3190
|
+
return
|
|
3191
|
+
try:
|
|
3192
|
+
engine.propose_and_start(
|
|
3193
|
+
hypothesis=str(action.get("hypothesis", ""))[:300],
|
|
3194
|
+
ref=str(action.get("ref", "")),
|
|
3195
|
+
variant=float(action.get("variant", 0.0)),
|
|
3196
|
+
metric=str(action.get("metric", "")),
|
|
3197
|
+
source="workspace")
|
|
3198
|
+
except (ValueError, TypeError):
|
|
3199
|
+
pass # invalid experiment spec is simply not started
|
|
3200
|
+
|
|
3201
|
+
async def _phase_expectations(self) -> None:
|
|
3202
|
+
"""Hourly (Mind M3a): form predictions from live signals and resolve
|
|
3203
|
+
the ones whose horizon passed. Misses become surprises on her mind."""
|
|
3204
|
+
eng = getattr(self._registry, "expectations", None)
|
|
3205
|
+
if eng is None:
|
|
3206
|
+
return
|
|
3207
|
+
# link the workspace so a miss raises salience there
|
|
3208
|
+
if getattr(eng, "_workspace", None) is None:
|
|
3209
|
+
eng._workspace = getattr(self._registry, "workspace", None)
|
|
3210
|
+
# generate + check every run: generate_from_commitments dedups on a
|
|
3211
|
+
# stable key (create() refuses an existing pending prediction), so a
|
|
3212
|
+
# newly due-dated commitment gets a prediction promptly instead of
|
|
3213
|
+
# waiting for an hour boundary.
|
|
3214
|
+
try:
|
|
3215
|
+
eng.generate_from_commitments()
|
|
3216
|
+
eng.check()
|
|
3217
|
+
except Exception as exc:
|
|
3218
|
+
self.stats.errors += 1
|
|
3219
|
+
logger.error("Phase expectations error: %s", exc, exc_info=True)
|
|
3220
|
+
|
|
3221
|
+
def _workspace_ingest(self, ws) -> None:
|
|
3222
|
+
"""Turn live signals into concerns. Each source dedups on a stable
|
|
3223
|
+
key so repeated ticks raise salience rather than pile up."""
|
|
3224
|
+
# overdue commitments -> concerns
|
|
3225
|
+
cstore = getattr(self._registry, "commitment_store", None)
|
|
3226
|
+
if cstore is not None:
|
|
3227
|
+
try:
|
|
3228
|
+
for c in cstore.get_overdue()[:10]:
|
|
3229
|
+
desc = (c.get("description") if isinstance(c, dict)
|
|
3230
|
+
else getattr(c, "description", "")) or "commitment"
|
|
3231
|
+
cid = (c.get("id") if isinstance(c, dict)
|
|
3232
|
+
else getattr(c, "id", "")) or desc
|
|
3233
|
+
ws.bump(kind="goal",
|
|
3234
|
+
summary=f"overdue commitment: {desc}",
|
|
3235
|
+
dedup_key=f"commitment:{cid}", salience=0.7,
|
|
3236
|
+
sources=[f"commitment:{cid}"])
|
|
3237
|
+
except Exception:
|
|
3238
|
+
pass
|
|
3239
|
+
# recent anomalies -> concerns (registry property is `anomalies`;
|
|
3240
|
+
# the detector exposes get_recent() -> List[Anomaly dataclass])
|
|
3241
|
+
detector = getattr(self._registry, "anomalies", None)
|
|
3242
|
+
get_recent = getattr(detector, "get_recent", None) if detector else None
|
|
3243
|
+
if callable(get_recent):
|
|
3244
|
+
try:
|
|
3245
|
+
for a in (get_recent(limit=10) or [])[:10]:
|
|
3246
|
+
summary = getattr(a, "description", None) or str(a)
|
|
3247
|
+
key = getattr(a, "id", None) or summary
|
|
3248
|
+
ws.bump(kind="anomaly", summary=str(summary)[:200],
|
|
3249
|
+
dedup_key=f"anomaly:{key}",
|
|
3250
|
+
salience=min(0.9, 0.4 + float(getattr(a, "severity", 0.4))),
|
|
3251
|
+
sources=[f"anomaly:{key}"])
|
|
3252
|
+
except Exception:
|
|
3253
|
+
logger.debug("workspace anomaly ingest failed", exc_info=True)
|
|
3254
|
+
# benchmark regressions -> a concern to look into
|
|
3255
|
+
bench = getattr(self._registry, "benchmark", None)
|
|
3256
|
+
if bench is not None:
|
|
3257
|
+
try:
|
|
3258
|
+
trends = bench.snapshot(weeks=2).get("trends", {})
|
|
3259
|
+
for metric, d in trends.items():
|
|
3260
|
+
if isinstance(d, (int, float)) and d < -0.15\
|
|
3261
|
+
and not str(metric).startswith("latency."):
|
|
3262
|
+
ws.bump(kind="question",
|
|
3263
|
+
summary=f"my {metric} regressed "
|
|
3264
|
+
f"{d:.2f} week-over-week; why, and "
|
|
3265
|
+
"can I improve it",
|
|
3266
|
+
dedup_key=f"benchmark:{metric}",
|
|
3267
|
+
salience=0.65, sources=[f"benchmark:{metric}"])
|
|
3268
|
+
except Exception:
|
|
3269
|
+
pass
|
|
3270
|
+
|
|
3271
|
+
async def _phase_toolsmith(self) -> None:
|
|
3272
|
+
"""Daily (Mind M1): mine the journal for repeated procedures, draft +
|
|
3273
|
+
sandbox-verify a tool, exercise verified tools in shadow, and propose
|
|
3274
|
+
graduation once a tool has enough clean shadow runs. Bounded per run:
|
|
3275
|
+
at most one new draft, to keep LLM+sandbox cost predictable. Wrapped in
|
|
3276
|
+
a hard timeout so a slow LLM draft or a wedged Docker run can never
|
|
3277
|
+
freeze the autonomy loop (the draft/verify calls are off-loop, but the
|
|
3278
|
+
tick still awaits the phase)."""
|
|
3279
|
+
try:
|
|
3280
|
+
await asyncio.wait_for(self._toolsmith_body(),
|
|
3281
|
+
timeout=self._phase_budget_secs())
|
|
3282
|
+
except asyncio.TimeoutError:
|
|
3283
|
+
logger.warning("toolsmith phase exceeded budget; skipping this tick")
|
|
3284
|
+
|
|
3285
|
+
def _phase_budget_secs(self) -> float:
|
|
3286
|
+
try:
|
|
3287
|
+
return float(os.environ.get("COLONY_PHASE_BUDGET_SECS", "150"))
|
|
3288
|
+
except ValueError:
|
|
3289
|
+
return 150.0
|
|
3290
|
+
|
|
3291
|
+
def _tick_budget_secs(self) -> float:
|
|
3292
|
+
"""Whole-tick wall-clock ceiling; kept under the tick interval so the
|
|
3293
|
+
loop always reaches the next cycle."""
|
|
3294
|
+
try:
|
|
3295
|
+
v = float(os.environ.get("COLONY_TICK_BUDGET_SECS", "0") or 0)
|
|
3296
|
+
except ValueError:
|
|
3297
|
+
v = 0.0
|
|
3298
|
+
if v > 0:
|
|
3299
|
+
return v
|
|
3300
|
+
return max(60.0, self.config.tick_interval_secs * 0.8)
|
|
3301
|
+
|
|
3302
|
+
async def _toolsmith_body(self) -> None:
|
|
3303
|
+
ts = getattr(self._registry, "toolsmith", None)
|
|
3304
|
+
if ts is None:
|
|
3305
|
+
return
|
|
3306
|
+
key = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
|
3307
|
+
if self._periodic_last.get("toolsmith") == key:
|
|
3308
|
+
return
|
|
3309
|
+
self._periodic_last["toolsmith"] = key
|
|
3310
|
+
try:
|
|
3311
|
+
from apsimo.toolsmith.registry import ToolStatus
|
|
3312
|
+
# 1. mine + draft one candidate that has no tool yet
|
|
3313
|
+
candidates = ts.miner.mine(limit=1000)
|
|
3314
|
+
drafted = None
|
|
3315
|
+
for cand in candidates[:3]:
|
|
3316
|
+
tool = await ts.draft(cand)
|
|
3317
|
+
if tool is not None:
|
|
3318
|
+
drafted = tool
|
|
3319
|
+
break
|
|
3320
|
+
# 2. verify any draft
|
|
3321
|
+
for tool in ts.registry.list(status=ToolStatus.DRAFT):
|
|
3322
|
+
await ts.verify(tool)
|
|
3323
|
+
# 3. Shadow evidence is supplied only when the incumbent and the
|
|
3324
|
+
# candidate can be compared on the same bounded captured input.
|
|
3325
|
+
# Replaying a generated self-test is verification, not operational
|
|
3326
|
+
# evidence, so the daily loop intentionally does not manufacture
|
|
3327
|
+
# shadow wins here.
|
|
3328
|
+
# 4. auto-retire failing tools
|
|
3329
|
+
for tool in ts.retirement_candidates():
|
|
3330
|
+
ts.retire(tool.tool_id, reason="failing/unused")
|
|
3331
|
+
# 5. propose graduation for eligible shadow tools
|
|
3332
|
+
await self._toolsmith_propose_graduations(ts)
|
|
3333
|
+
except Exception as exc:
|
|
3334
|
+
self.stats.errors += 1
|
|
3335
|
+
logger.error("Phase toolsmith error: %s", exc, exc_info=True)
|
|
3336
|
+
|
|
3337
|
+
async def _toolsmith_propose_graduations(self, ts) -> None:
|
|
3338
|
+
delivery = self._registry.delivery
|
|
3339
|
+
cands = ts.graduation_candidates()
|
|
3340
|
+
if not cands:
|
|
3341
|
+
return
|
|
3342
|
+
for tool in cands:
|
|
3343
|
+
# Trust is useful evidence, but never publication authority. A
|
|
3344
|
+
# one-shot owner-scoped graduation envelope is required at every
|
|
3345
|
+
# trust stage, including act_first.
|
|
3346
|
+
title = f"New tool ready: {tool.name}"
|
|
3347
|
+
comparison_count = ts.registry.clean_comparison_count(tool.tool_id)
|
|
3348
|
+
finding = (f"I built a tool, {tool.name}: {tool.description}. "
|
|
3349
|
+
f"It passed sandbox verification and "
|
|
3350
|
+
f"{comparison_count} clean same-input comparisons. "
|
|
3351
|
+
"Approve its exact artifact to let me use it for real.")
|
|
3352
|
+
if delivery is None:
|
|
3353
|
+
continue
|
|
3354
|
+
try:
|
|
3355
|
+
from apsimo.proposals import Proposal, proposal_to_payload
|
|
3356
|
+
prop = Proposal(
|
|
3357
|
+
title=title[:100], finding=finding[:600],
|
|
3358
|
+
why_it_helps="I get more capable at things I do often, "
|
|
3359
|
+
"and you see every new capability first",
|
|
3360
|
+
suggested_action=(
|
|
3361
|
+
f"Approve the bounded artifact digest via the tools "
|
|
3362
|
+
f"API to graduate {tool.name}."),
|
|
3363
|
+
source="toolsmith", initiative_type="proposal",
|
|
3364
|
+
confidence=0.85)
|
|
3365
|
+
await self._route_reachout_delivery(
|
|
3366
|
+
proposal_to_payload(prop), delivery)
|
|
3367
|
+
except Exception:
|
|
3368
|
+
logger.debug("toolsmith graduation notice failed",
|
|
3369
|
+
exc_info=True)
|
|
3370
|
+
|
|
3371
|
+
async def _phase_experiments(self) -> None:
|
|
3372
|
+
"""Daily (Mind M0b): decide running self-experiments whose window
|
|
3373
|
+
ended (adopt within guard, auto-revert on regression, abort when
|
|
3374
|
+
superseded). Notifies the owner of every decision."""
|
|
3375
|
+
engine = getattr(self._registry, "experiments", None)
|
|
3376
|
+
if engine is None:
|
|
3377
|
+
return
|
|
3378
|
+
key = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
|
3379
|
+
if self._periodic_last.get("experiments") == key:
|
|
3380
|
+
return
|
|
3381
|
+
try:
|
|
3382
|
+
decided = engine.evaluate()
|
|
3383
|
+
self._periodic_last["experiments"] = key
|
|
3384
|
+
except Exception as exc:
|
|
3385
|
+
self.stats.errors += 1
|
|
3386
|
+
logger.error("Phase experiments error: %s", exc, exc_info=True)
|
|
3387
|
+
return
|
|
3388
|
+
if not decided:
|
|
3389
|
+
return
|
|
3390
|
+
delivery = self._registry.delivery
|
|
3391
|
+
if delivery is None:
|
|
3392
|
+
return
|
|
3393
|
+
try:
|
|
3394
|
+
from apsimo.proposals import Proposal, proposal_to_payload
|
|
3395
|
+
except Exception:
|
|
3396
|
+
return
|
|
3397
|
+
for exp in decided:
|
|
3398
|
+
try:
|
|
3399
|
+
prop = Proposal(
|
|
3400
|
+
title=f"Experiment {exp.get('status')}: "
|
|
3401
|
+
f"{exp.get('ref')}"[:100],
|
|
3402
|
+
finding=(f"{exp.get('hypothesis', '')[:200]}. Decision: "
|
|
3403
|
+
f"{exp.get('decision_reason', '')[:400]}"),
|
|
3404
|
+
why_it_helps="self-changes stay measured and reversible",
|
|
3405
|
+
suggested_action="No action needed; abort any running "
|
|
3406
|
+
"experiment via the experiments API.",
|
|
3407
|
+
source="experiment-framework",
|
|
3408
|
+
initiative_type="proposal", confidence=0.85)
|
|
3409
|
+
await self._route_reachout_delivery(
|
|
3410
|
+
proposal_to_payload(prop), delivery)
|
|
3411
|
+
except Exception:
|
|
3412
|
+
logger.debug("experiment notice delivery failed",
|
|
3413
|
+
exc_info=True)
|
|
3414
|
+
|
|
3415
|
+
async def _phase_belief_maintenance(self) -> None:
|
|
3416
|
+
"""Phase 11c (daily): belief maintenance (cognition item 7)."""
|
|
3417
|
+
engine = getattr(self._registry, "belief_engine", None)
|
|
3418
|
+
if engine is None:
|
|
3419
|
+
return
|
|
3420
|
+
key = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
|
3421
|
+
if self._periodic_last.get("belief_maintenance") == key:
|
|
3422
|
+
return
|
|
3423
|
+
try:
|
|
3424
|
+
with self._periodic_attempt("belief_maintenance", key):
|
|
3425
|
+
await engine.run()
|
|
3426
|
+
except Exception as exc:
|
|
3427
|
+
self.stats.errors += 1
|
|
3428
|
+
logger.error("Phase belief_maintenance error: %s", exc,
|
|
3429
|
+
exc_info=True)
|
|
3430
|
+
|
|
3431
|
+
async def _phase_relationship_profiling(self) -> None:
|
|
3432
|
+
"""Phase 11f (periodic): refresh RelationshipBriefs for contacts that
|
|
3433
|
+
accrued enough new interactions (docs/RELATIONSHIPS.md #7)."""
|
|
3434
|
+
try:
|
|
3435
|
+
from apsimo.api.routers.host import _relationship_profiler
|
|
3436
|
+
except ImportError:
|
|
3437
|
+
return
|
|
3438
|
+
if _relationship_profiler is None:
|
|
3439
|
+
return
|
|
3440
|
+
refresh_secs = 21600.0
|
|
3441
|
+
try:
|
|
3442
|
+
refresh_secs = float(os.environ.get(
|
|
3443
|
+
"COLONY_RELATIONSHIP_PROFILE_REFRESH_SECS", "21600"))
|
|
3444
|
+
except ValueError:
|
|
3445
|
+
pass
|
|
3446
|
+
now = time.time()
|
|
3447
|
+
last = self._periodic_last.get("relationship_profiling", 0.0)
|
|
3448
|
+
if isinstance(last, str):
|
|
3449
|
+
last = 0.0
|
|
3450
|
+
if now - float(last or 0.0) < refresh_secs:
|
|
3451
|
+
return
|
|
3452
|
+
try:
|
|
3453
|
+
with self._periodic_attempt("relationship_profiling", now):
|
|
3454
|
+
report = await _relationship_profiler.refresh_due()
|
|
3455
|
+
if report.get("profiled"):
|
|
3456
|
+
logger.info("relationship profiling: %s", report)
|
|
3457
|
+
except Exception as exc:
|
|
3458
|
+
self.stats.errors += 1
|
|
3459
|
+
logger.error("Phase relationship_profiling error: %s", exc,
|
|
3460
|
+
exc_info=True)
|
|
3461
|
+
|
|
3462
|
+
async def _phase_world_llm_extract(self) -> None:
|
|
3463
|
+
"""Phase 11d (daily): LLM-assisted world-model extraction (batch,
|
|
3464
|
+
journaled; piggybacks the daily memory-distillation cadence)."""
|
|
3465
|
+
extractor = getattr(self._registry, "world_llm_extractor", None)
|
|
3466
|
+
if extractor is None:
|
|
3467
|
+
return
|
|
3468
|
+
try:
|
|
3469
|
+
from apsimo.world_model.llm_extract import llm_extract_mode
|
|
3470
|
+
if llm_extract_mode() == "off":
|
|
3471
|
+
return
|
|
3472
|
+
except Exception:
|
|
3473
|
+
return
|
|
3474
|
+
key = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
|
3475
|
+
if self._periodic_last.get("world_llm_extract") == key:
|
|
3476
|
+
return
|
|
3477
|
+
try:
|
|
3478
|
+
with self._periodic_attempt("world_llm_extract", key):
|
|
3479
|
+
await extractor.run()
|
|
3480
|
+
except Exception as exc:
|
|
3481
|
+
self.stats.errors += 1
|
|
3482
|
+
logger.error("Phase world_llm_extract error: %s", exc,
|
|
3483
|
+
exc_info=True)
|
|
3484
|
+
|
|
3485
|
+
async def _phase_tom2_asymmetry(self) -> None:
|
|
3486
|
+
"""Phase 11d2 (daily): tom2 knowledge-asymmetry sweep. Inert unless
|
|
3487
|
+
COLONY_TOM2 is set (default off; shadow computes counts only, live
|
|
3488
|
+
writes refs-not-content inference rows)."""
|
|
3489
|
+
engine = getattr(self._registry, "tom2_engine", None)
|
|
3490
|
+
if engine is None:
|
|
3491
|
+
return
|
|
3492
|
+
try:
|
|
3493
|
+
from apsimo.tom.asymmetry import tom2_mode
|
|
3494
|
+
if tom2_mode() == "off":
|
|
3495
|
+
return
|
|
3496
|
+
except Exception:
|
|
3497
|
+
return
|
|
3498
|
+
key = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
|
3499
|
+
if self._periodic_last.get("tom2_asymmetry") == key:
|
|
3500
|
+
return
|
|
3501
|
+
try:
|
|
3502
|
+
engine.run()
|
|
3503
|
+
self._periodic_last["tom2_asymmetry"] = key
|
|
3504
|
+
except Exception as exc:
|
|
3505
|
+
self.stats.errors += 1
|
|
3506
|
+
logger.error("Phase tom2_asymmetry error: %s", exc, exc_info=True)
|
|
3507
|
+
|
|
3508
|
+
async def _phase_connectors(self) -> None:
|
|
3509
|
+
"""Phase 11e: poll due read-only connectors (cognition item 2).
|
|
3510
|
+
|
|
3511
|
+
The manager owns per-connector cadence and mode gating (off/shadow/
|
|
3512
|
+
live); this phase just gives it a tick. Off by default -- a no-op
|
|
3513
|
+
until connectors are configured and enabled per deployment."""
|
|
3514
|
+
manager = getattr(self._registry, "connector_manager", None)
|
|
3515
|
+
if manager is None:
|
|
3516
|
+
return
|
|
3517
|
+
try:
|
|
3518
|
+
from apsimo.connectors import connectors_mode
|
|
3519
|
+
if connectors_mode() == "off":
|
|
3520
|
+
return
|
|
3521
|
+
except Exception:
|
|
3522
|
+
return
|
|
3523
|
+
try:
|
|
3524
|
+
await manager.poll_due()
|
|
3525
|
+
except Exception as exc:
|
|
3526
|
+
self.stats.errors += 1
|
|
3527
|
+
logger.error("Phase connectors error: %s", exc, exc_info=True)
|
|
3528
|
+
|
|
3529
|
+
async def _run_phase(self, name: str, awaitable) -> None:
|
|
3530
|
+
"""Await one tick phase, recording its name while it runs and its
|
|
3531
|
+
wall-clock duration (last and worst) afterwards. The marker is left in
|
|
3532
|
+
place on cancellation so the budget-cancel path can name the phase."""
|
|
3533
|
+
selected = self.config.enabled_phases
|
|
3534
|
+
if selected is not None and name not in selected:
|
|
3535
|
+
awaitable.close()
|
|
3536
|
+
self.stats.phases_skipped += 1
|
|
3537
|
+
return
|
|
3538
|
+
self._current_phase = name
|
|
3539
|
+
started = time.monotonic()
|
|
3540
|
+
try:
|
|
3541
|
+
await awaitable
|
|
3542
|
+
finally:
|
|
3543
|
+
elapsed = time.monotonic() - started
|
|
3544
|
+
self._phase_seconds[name] = elapsed
|
|
3545
|
+
if elapsed > self._phase_seconds_max.get(name, 0.0):
|
|
3546
|
+
self._phase_seconds_max[name] = elapsed
|
|
3547
|
+
|
|
3548
|
+
def _note_tick_cancelled(self) -> None:
|
|
3549
|
+
"""The whole tick overran its budget and was cancelled mid-phase."""
|
|
3550
|
+
phase = self._current_phase or "unknown"
|
|
3551
|
+
self._current_phase = None
|
|
3552
|
+
self.stats.errors += 1
|
|
3553
|
+
self.stats.phases_cancelled += 1
|
|
3554
|
+
self.stats.last_cancelled_phase = phase
|
|
3555
|
+
logger.error(
|
|
3556
|
+
"Tick #%d exceeded budget (%.0fs) in phase %s (%.1fs so far) and "
|
|
3557
|
+
"was cancelled; loop continues", self.stats.ticks,
|
|
3558
|
+
self._tick_budget_secs(), phase,
|
|
3559
|
+
self._phase_seconds.get(phase, 0.0))
|
|
3560
|
+
|
|
3561
|
+
def phase_timings(self) -> dict:
|
|
3562
|
+
"""Per-phase wall-clock seconds: last run and worst run, plus the
|
|
3563
|
+
phase running right now (None between ticks)."""
|
|
3564
|
+
return {
|
|
3565
|
+
"current": self._current_phase,
|
|
3566
|
+
"seconds_last": {k: round(v, 3) for k, v in self._phase_seconds.items()},
|
|
3567
|
+
"seconds_max": {k: round(v, 3) for k, v in self._phase_seconds_max.items()},
|
|
3568
|
+
}
|
|
3569
|
+
|
|
3570
|
+
@contextlib.contextmanager
|
|
3571
|
+
def _periodic_attempt(self, name: str, key):
|
|
3572
|
+
"""Mark a periodic phase attempted for its period.
|
|
3573
|
+
|
|
3574
|
+
The body is the phase's work. On normal exit the period key is stored;
|
|
3575
|
+
on cancellation (the tick overran its budget) the key is stored too,
|
|
3576
|
+
because otherwise the key never advances and the slow phase runs again
|
|
3577
|
+
on the very next tick, eating every tick and starving every phase
|
|
3578
|
+
after it until the period rolls over. An ordinary exception leaves the
|
|
3579
|
+
key alone so the phase retries next tick, as before.
|
|
3580
|
+
"""
|
|
3581
|
+
try:
|
|
3582
|
+
yield
|
|
3583
|
+
except asyncio.CancelledError:
|
|
3584
|
+
self._periodic_last[name] = key
|
|
3585
|
+
logger.error(
|
|
3586
|
+
"Phase %s was cancelled before it finished; counted as "
|
|
3587
|
+
"attempted for this period (%s), next attempt when the "
|
|
3588
|
+
"period rolls", name, key)
|
|
3589
|
+
raise
|
|
3590
|
+
else:
|
|
3591
|
+
self._periodic_last[name] = key
|
|
3592
|
+
|
|
3593
|
+
async def _run_periodic_phase(self, name: str, period: str, work) -> None:
|
|
3594
|
+
"""Run a memory-lifecycle phase at most once per period ("hour"|"day"|"week").
|
|
3595
|
+
work(graph) does the phase-specific work; the dedup key is cached per name.
|
|
3596
|
+
Replaces six near-identical _phase_memory_* skeletons (behavior preserved).
|
|
3597
|
+
See _periodic_attempt for why a cancelled run still counts."""
|
|
3598
|
+
graph = self._registry.graph
|
|
3599
|
+
if graph is None:
|
|
3600
|
+
return
|
|
3601
|
+
now = datetime.now(timezone.utc)
|
|
3602
|
+
key = {"hour": now.hour, "day": now.strftime("%Y-%m-%d"),
|
|
3603
|
+
"week": now.strftime("%Y-W%W")}[period]
|
|
3604
|
+
if self._periodic_last.get(name) == key:
|
|
3605
|
+
return
|
|
3606
|
+
try:
|
|
3607
|
+
with self._periodic_attempt(name, key):
|
|
3608
|
+
await work(graph)
|
|
3609
|
+
except Exception as exc:
|
|
3610
|
+
self.stats.errors += 1
|
|
3611
|
+
logger.error("Phase %s error: %s", name, exc, exc_info=True)
|
|
3612
|
+
|
|
3613
|
+
async def _phase_condition_checks(self) -> None:
|
|
3614
|
+
"""System-condition sweep (hourly) + blocked-goal condition polling
|
|
3615
|
+
(every tick, per-condition cadence).
|
|
3616
|
+
|
|
3617
|
+
System checks: the pending→overdue commitment flip (fires
|
|
3618
|
+
commitment.overdue exactly once per item), sustained affect decline,
|
|
3619
|
+
and surprise accumulation — written for a queue-scheduled path that
|
|
3620
|
+
nothing enqueues; the loop is the reliable place to run them.
|
|
3621
|
+
|
|
3622
|
+
Blocked goals: a goal blocked with context.condition_type is polled
|
|
3623
|
+
via handle_check_condition at get_check_interval() cadence and
|
|
3624
|
+
auto-unblocks when the condition is met. This is the missing PRODUCER
|
|
3625
|
+
for the per-goal condition path."""
|
|
3626
|
+
now = datetime.now(timezone.utc)
|
|
3627
|
+
key = now.strftime("%Y-%m-%dT%H")
|
|
3628
|
+
if self._periodic_last.get("condition_checks") != key:
|
|
3629
|
+
self._periodic_last["condition_checks"] = key
|
|
3630
|
+
from apsimo.autonomy.condition_worker import (
|
|
3631
|
+
_check_affect_decline,
|
|
3632
|
+
_check_commitment_overdue,
|
|
3633
|
+
_check_surprise_accumulation,
|
|
3634
|
+
)
|
|
3635
|
+
for checker in (_check_commitment_overdue, _check_affect_decline,
|
|
3636
|
+
_check_surprise_accumulation):
|
|
3637
|
+
try:
|
|
3638
|
+
await checker({})
|
|
3639
|
+
except Exception:
|
|
3640
|
+
logger.debug("condition check %s failed",
|
|
3641
|
+
getattr(checker, "__name__", "?"), exc_info=True)
|
|
3642
|
+
await self._poll_blocked_goal_conditions()
|
|
3643
|
+
|
|
3644
|
+
async def _poll_blocked_goal_conditions(self) -> None:
|
|
3645
|
+
"""Poll every BLOCKED goal that carries an external condition.
|
|
3646
|
+
|
|
3647
|
+
Cadence is the condition type's interval, floored by the tick
|
|
3648
|
+
interval (a 30s deployment_health check can only fire once per tick).
|
|
3649
|
+
Every step is per-goal fault-isolated; one malformed goal can never
|
|
3650
|
+
take down the loop."""
|
|
3651
|
+
goals = getattr(self._registry, "goals", None)
|
|
3652
|
+
if goals is None:
|
|
3653
|
+
return
|
|
3654
|
+
from apsimo.autonomy.condition_worker import (
|
|
3655
|
+
get_check_interval, handle_check_condition)
|
|
3656
|
+
from apsimo.goals.models import GoalStatus
|
|
3657
|
+
try:
|
|
3658
|
+
blocked = goals.list_goals(status="blocked", limit=100) or []
|
|
3659
|
+
except Exception:
|
|
3660
|
+
logger.debug("blocked-goal listing failed", exc_info=True)
|
|
3661
|
+
return
|
|
3662
|
+
now = time.time()
|
|
3663
|
+
for goal in blocked:
|
|
3664
|
+
try:
|
|
3665
|
+
ctx = getattr(goal, "context", None) or {}
|
|
3666
|
+
ctype = ctx.get("condition_type")
|
|
3667
|
+
if not ctype:
|
|
3668
|
+
continue # blocked on a human/approval, not a condition
|
|
3669
|
+
interval = get_check_interval(ctype, getattr(goal, "deadline", None))
|
|
3670
|
+
try:
|
|
3671
|
+
last = float(ctx.get("condition_last_check") or 0.0)
|
|
3672
|
+
except (TypeError, ValueError):
|
|
3673
|
+
last = 0.0
|
|
3674
|
+
if now - last < interval:
|
|
3675
|
+
continue
|
|
3676
|
+
await handle_check_condition(
|
|
3677
|
+
{"goal_id": goal.goal_id, "condition_type": ctype,
|
|
3678
|
+
"condition_params": ctx.get("condition_params") or {}},
|
|
3679
|
+
goal_engine=goals,
|
|
3680
|
+
)
|
|
3681
|
+
# RE-LOAD before persisting the poll time: the check may have
|
|
3682
|
+
# awaited a slow network call, and writing the pre-await
|
|
3683
|
+
# object back would clobber any concurrent update (lost
|
|
3684
|
+
# update). GoalEngine exposes no public save; _store is the
|
|
3685
|
+
# sanctioned persistence path here. If the goal is still
|
|
3686
|
+
# BLOCKED — condition not met, OR met but unblock failed —
|
|
3687
|
+
# stamp the fresh object so the cadence holds either way (a
|
|
3688
|
+
# failed unblock must not busy-repoll every tick).
|
|
3689
|
+
fresh = goals._store.get_goal(goal.goal_id)
|
|
3690
|
+
if fresh is not None and fresh.status == GoalStatus.BLOCKED\
|
|
3691
|
+
and fresh.context.get("condition_type"):
|
|
3692
|
+
fresh.context["condition_last_check"] = now
|
|
3693
|
+
goals._store.save_goal(fresh)
|
|
3694
|
+
except Exception:
|
|
3695
|
+
logger.debug("condition poll failed for goal %s",
|
|
3696
|
+
getattr(goal, "goal_id", "?"), exc_info=True)
|
|
3697
|
+
|
|
3698
|
+
# Graph capabilities the periodic memory phases dispatch on. A missing
|
|
3699
|
+
# method means the phase silently does nothing every cycle — checked
|
|
3700
|
+
# once at loop start and counted (stats.phases_skipped) at skip time.
|
|
3701
|
+
_GRAPH_PHASE_CAPABILITIES = {
|
|
3702
|
+
"memory_decay": "decay_memories",
|
|
3703
|
+
"memory_pruning": "prune_weak_memories",
|
|
3704
|
+
"memory_archive": "archive_memories",
|
|
3705
|
+
}
|
|
3706
|
+
|
|
3707
|
+
def _check_phase_capabilities(self) -> None:
|
|
3708
|
+
"""Boot self-check: warn once per phase whose graph capability is
|
|
3709
|
+
missing, so a renamed/removed backend method can never turn a
|
|
3710
|
+
maintenance phase into a silent no-op again."""
|
|
3711
|
+
try:
|
|
3712
|
+
graph = self._registry.graph
|
|
3713
|
+
except Exception:
|
|
3714
|
+
graph = None
|
|
3715
|
+
if graph is None:
|
|
3716
|
+
return
|
|
3717
|
+
for phase, attr in self._GRAPH_PHASE_CAPABILITIES.items():
|
|
3718
|
+
if not hasattr(graph, attr):
|
|
3719
|
+
self._note_phase_skipped(
|
|
3720
|
+
phase, f"graph backend lacks {attr}()", count=False)
|
|
3721
|
+
|
|
3722
|
+
def _note_phase_skipped(self, name: str, reason: str,
|
|
3723
|
+
count: bool = True) -> None:
|
|
3724
|
+
"""Record a phase skip: count every occurrence, warn only once."""
|
|
3725
|
+
if count:
|
|
3726
|
+
self.stats.phases_skipped += 1
|
|
3727
|
+
if name not in self._phase_skip_warned:
|
|
3728
|
+
self._phase_skip_warned.add(name)
|
|
3729
|
+
logger.warning("Phase %s skipped: %s", name, reason)
|
|
3730
|
+
|
|
3731
|
+
async def _phase_memory_decay(self) -> None:
|
|
3732
|
+
async def work(graph):
|
|
3733
|
+
if hasattr(graph, "decay_memories"):
|
|
3734
|
+
await graph.decay_memories()
|
|
3735
|
+
else:
|
|
3736
|
+
self._note_phase_skipped(
|
|
3737
|
+
"memory_decay", "graph backend lacks decay_memories()")
|
|
3738
|
+
await self._run_periodic_phase("memory_decay", "day", work)
|
|
3739
|
+
|
|
3740
|
+
async def _phase_memory_pruning(self) -> None:
|
|
3741
|
+
"""Weekly: prune memories whose strength decayed below threshold.
|
|
3742
|
+
|
|
3743
|
+
COLONY_MEMORY_PRUNE_MODE gates the phase: ``off`` disables it,
|
|
3744
|
+
``shadow`` (default) counts what WOULD be pruned without deleting
|
|
3745
|
+
anything, ``live`` deletes for real (capped per pass, graph node +
|
|
3746
|
+
vector together). The default preserves shipped behavior — nothing
|
|
3747
|
+
is ever deleted until a deployment flips the flag deliberately.
|
|
3748
|
+
"""
|
|
3749
|
+
mode = os.environ.get(
|
|
3750
|
+
"COLONY_MEMORY_PRUNE_MODE", "shadow").strip().lower()
|
|
3751
|
+
if mode == "off":
|
|
3752
|
+
return
|
|
3753
|
+
if mode not in ("shadow", "live"):
|
|
3754
|
+
mode = "shadow" # unknown values fail safe to counting only
|
|
3755
|
+
|
|
3756
|
+
async def work(graph):
|
|
3757
|
+
if not hasattr(graph, "prune_weak_memories"):
|
|
3758
|
+
self._note_phase_skipped(
|
|
3759
|
+
"memory_pruning",
|
|
3760
|
+
"graph backend lacks prune_weak_memories()")
|
|
3761
|
+
return
|
|
3762
|
+
result = await graph.prune_weak_memories(dry_run=(mode != "live"))
|
|
3763
|
+
logger.info(
|
|
3764
|
+
"Phase memory_pruning (%s): matched=%s deleted=%s",
|
|
3765
|
+
mode, result.get("matched"), result.get("deleted"))
|
|
3766
|
+
# Post-prune orphan-vector sweep (live mode only, bounded):
|
|
3767
|
+
# pruning couples vector deletion to node deletion, but any
|
|
3768
|
+
# vector delete that failed leaves an orphan that keeps
|
|
3769
|
+
# matching in ANN search. Best-effort — a sweep failure never
|
|
3770
|
+
# fails the prune pass; leftovers are caught next week or via
|
|
3771
|
+
# the explicit /memory/vector-vacuum endpoint.
|
|
3772
|
+
if mode == "live" and hasattr(graph, "vacuum_orphan_vectors"):
|
|
3773
|
+
try:
|
|
3774
|
+
sweep = await graph.vacuum_orphan_vectors(
|
|
3775
|
+
dry_run=False, max_delete=2000)
|
|
3776
|
+
logger.info(
|
|
3777
|
+
"Phase memory_pruning orphan sweep: orphans=%s "
|
|
3778
|
+
"deleted=%s", sweep.get("orphans"),
|
|
3779
|
+
sweep.get("deleted"))
|
|
3780
|
+
except Exception:
|
|
3781
|
+
logger.warning("post-prune orphan-vector sweep failed",
|
|
3782
|
+
exc_info=True)
|
|
3783
|
+
await self._run_periodic_phase("memory_pruning", "week", work)
|
|
3784
|
+
|
|
3785
|
+
async def _phase_memory_distillation(self) -> None:
|
|
3786
|
+
"""Daily: promote frequently-recalled episodic memories into durable semantic facts via
|
|
3787
|
+
MemoryDistiller. The distiller was built but never scheduled — this is the missing wiring
|
|
3788
|
+
that turns accumulated conversation history into lasting knowledge instead of decaying logs."""
|
|
3789
|
+
async def work(graph):
|
|
3790
|
+
try:
|
|
3791
|
+
from apsimo.intelligence.graph.distiller import MemoryDistiller
|
|
3792
|
+
result = await MemoryDistiller(graph).run()
|
|
3793
|
+
self.stats.memories_promoted += result.memories_promoted
|
|
3794
|
+
if result.memories_promoted:
|
|
3795
|
+
logger.info("memory distillation: %d semantic fact(s) from %d cluster(s)",
|
|
3796
|
+
result.memories_promoted, result.clusters_found)
|
|
3797
|
+
except Exception:
|
|
3798
|
+
logger.debug("memory distillation phase failed", exc_info=True)
|
|
3799
|
+
await self._run_periodic_phase("memory_distillation", "day", work)
|
|
3800
|
+
|
|
3801
|
+
async def _phase_memory_reconciliation(self) -> None:
|
|
3802
|
+
async def work(graph):
|
|
3803
|
+
from apsimo.intelligence.graph.reconciler import FileReconciler
|
|
3804
|
+
result = await FileReconciler(graph).reconcile(dry_run=False)
|
|
3805
|
+
logger.info(
|
|
3806
|
+
"Phase memory_reconciliation: checked=%d verified=%d staled=%d superseded=%d errors=%d",
|
|
3807
|
+
result["files_checked"], result["memories_verified"],
|
|
3808
|
+
result["memories_staled"], result["memories_superseded"], len(result["errors"]),
|
|
3809
|
+
)
|
|
3810
|
+
await self._run_periodic_phase("memory_reconciliation", "day", work)
|
|
3811
|
+
|
|
3812
|
+
async def _phase_memory_archive(self) -> None:
|
|
3813
|
+
async def work(graph):
|
|
3814
|
+
if hasattr(graph, "archive_memories"):
|
|
3815
|
+
archived = await graph.archive_memories(max_age_days=30)
|
|
3816
|
+
logger.info("Phase memory_archive: archived=%d", archived)
|
|
3817
|
+
else:
|
|
3818
|
+
self._note_phase_skipped(
|
|
3819
|
+
"memory_archive",
|
|
3820
|
+
"graph backend lacks archive_memories()")
|
|
3821
|
+
await self._run_periodic_phase("memory_archive", "week", work)
|
|
3822
|
+
|
|
3823
|
+
async def _phase_task_completion(self) -> None:
|
|
3824
|
+
"""Emit follow-up events for goals that completed since the last check.
|
|
3825
|
+
|
|
3826
|
+
Runs at most hourly. Emits one ``task_completed_followup`` event per
|
|
3827
|
+
newly-completed goal and asks the connection discoverer for
|
|
3828
|
+
reflection insights when a backlog accumulates.
|
|
3829
|
+
"""
|
|
3830
|
+
now = datetime.now(timezone.utc)
|
|
3831
|
+
if self._last_task_completion_check is not None:
|
|
3832
|
+
elapsed = (now - self._last_task_completion_check).total_seconds()
|
|
3833
|
+
if elapsed < 3600: # Check hourly
|
|
3834
|
+
return
|
|
3835
|
+
|
|
3836
|
+
goals = self._registry.goals
|
|
3837
|
+
if goals is None:
|
|
3838
|
+
self._last_task_completion_check = now
|
|
3839
|
+
return
|
|
3840
|
+
|
|
3841
|
+
try:
|
|
3842
|
+
from apsimo.goals.models import GoalStatus
|
|
3843
|
+
completed = goals.list_goals(status=GoalStatus.COMPLETED, limit=50)
|
|
3844
|
+
|
|
3845
|
+
window_start = self._last_task_completion_check
|
|
3846
|
+
new_completions = []
|
|
3847
|
+
for g in completed:
|
|
3848
|
+
cat = getattr(g, "completed_at", None)
|
|
3849
|
+
if cat is None:
|
|
3850
|
+
continue
|
|
3851
|
+
if window_start is None or cat >= window_start:
|
|
3852
|
+
new_completions.append(g)
|
|
3853
|
+
|
|
3854
|
+
for g in new_completions:
|
|
3855
|
+
try:
|
|
3856
|
+
self.events.emit(Event(
|
|
3857
|
+
id=f"task-followup-{getattr(g, 'goal_id', uuid.uuid4())}",
|
|
3858
|
+
source="autonomy.task_completion",
|
|
3859
|
+
))
|
|
3860
|
+
broadcast = _get_broadcast()
|
|
3861
|
+
if broadcast is not None:
|
|
3862
|
+
try:
|
|
3863
|
+
broadcast({
|
|
3864
|
+
"type": "task_followup",
|
|
3865
|
+
"goal_id": getattr(g, "goal_id", ""),
|
|
3866
|
+
"title": getattr(g, "title", ""),
|
|
3867
|
+
})
|
|
3868
|
+
except Exception:
|
|
3869
|
+
logger.debug("broadcast task_followup failed", exc_info=True)
|
|
3870
|
+
except Exception:
|
|
3871
|
+
logger.debug("emit task_followup failed", exc_info=True)
|
|
3872
|
+
|
|
3873
|
+
# If several goals finished in the window, ask synthesis for
|
|
3874
|
+
# reflection connections to surface patterns.
|
|
3875
|
+
if len(new_completions) >= 3:
|
|
3876
|
+
discoverer = self._registry.connection_discoverer
|
|
3877
|
+
if discoverer is not None and hasattr(discoverer, "discover_connections"):
|
|
3878
|
+
try:
|
|
3879
|
+
await discoverer.discover_connections(min_novelty=0.3)
|
|
3880
|
+
except Exception:
|
|
3881
|
+
logger.debug("reflection discovery failed", exc_info=True)
|
|
3882
|
+
|
|
3883
|
+
self.stats.task_follow_ups += len(new_completions)
|
|
3884
|
+
self._last_task_completion_check = now
|
|
3885
|
+
if new_completions:
|
|
3886
|
+
logger.info(
|
|
3887
|
+
"Phase task_completion: %d follow-up(s)", len(new_completions)
|
|
3888
|
+
)
|
|
3889
|
+
except Exception as exc:
|
|
3890
|
+
self.stats.errors += 1
|
|
3891
|
+
logger.error("Phase task_completion error: %s", exc, exc_info=True)
|
|
3892
|
+
|
|
3893
|
+
async def _phase_frustration_update(self) -> None:
|
|
3894
|
+
"""Update delivery rate limiter based on engagement feedback."""
|
|
3895
|
+
delivery = self._registry.delivery
|
|
3896
|
+
if delivery is None:
|
|
3897
|
+
return
|
|
3898
|
+
try:
|
|
3899
|
+
learner = self._registry.learner
|
|
3900
|
+
if learner is not None and hasattr(delivery, "update_rate_limiter"):
|
|
3901
|
+
await delivery.update_rate_limiter(learner)
|
|
3902
|
+
except Exception as exc:
|
|
3903
|
+
logger.debug("Phase frustration_update error (non-fatal): %s", exc)
|
|
3904
|
+
|
|
3905
|
+
async def _phase_relationships(self) -> None:
|
|
3906
|
+
"""Legacy scheduler slot: scalar relationship scoring is retired.
|
|
3907
|
+
|
|
3908
|
+
Attributed preferences and appraisals are updated by the canonical
|
|
3909
|
+
source worker. Quiet periods and contact mood cannot lower standing.
|
|
3910
|
+
"""
|
|
3911
|
+
return
|
|
3912
|
+
|
|
3913
|
+
async def _phase_synthesis(self) -> None:
|
|
3914
|
+
"""Discover cross-domain connections."""
|
|
3915
|
+
discoverer = self._registry.connection_discoverer
|
|
3916
|
+
if discoverer is None:
|
|
3917
|
+
return
|
|
3918
|
+
try:
|
|
3919
|
+
connections = await discoverer.discover_connections()
|
|
3920
|
+
if connections:
|
|
3921
|
+
logger.info("Phase synthesis: %d new connections", len(connections))
|
|
3922
|
+
_get_broadcast()({
|
|
3923
|
+
"type": "insight",
|
|
3924
|
+
"occurred_at": datetime.now(timezone.utc).isoformat(),
|
|
3925
|
+
"payload": {"new_connections": len(connections)},
|
|
3926
|
+
})
|
|
3927
|
+
except Exception as exc:
|
|
3928
|
+
self.stats.errors += 1
|
|
3929
|
+
logger.error("Phase synthesis error: %s", exc, exc_info=True)
|
|
3930
|
+
|
|
3931
|
+
async def _phase_bootstrap_check(self) -> None:
|
|
3932
|
+
"""Run identity bootstrap self-check daily."""
|
|
3933
|
+
chain = self._registry.chain
|
|
3934
|
+
if chain is None:
|
|
3935
|
+
return
|
|
3936
|
+
now = datetime.now(timezone.utc)
|
|
3937
|
+
interval_hours = self.config.bootstrap_check_interval_hours
|
|
3938
|
+
last = self._periodic_last.get("bootstrap_check")
|
|
3939
|
+
if isinstance(last, datetime):
|
|
3940
|
+
elapsed = (now - last).total_seconds() / 3600
|
|
3941
|
+
if elapsed < interval_hours:
|
|
3942
|
+
return
|
|
3943
|
+
try:
|
|
3944
|
+
if hasattr(chain, "health_check"):
|
|
3945
|
+
with self._periodic_attempt("bootstrap_check", now):
|
|
3946
|
+
healthy = await chain.health_check()
|
|
3947
|
+
if not healthy:
|
|
3948
|
+
logger.warning("Phase bootstrap_check: chain health degraded")
|
|
3949
|
+
except Exception as exc:
|
|
3950
|
+
self.stats.errors += 1
|
|
3951
|
+
logger.error("Phase bootstrap_check error: %s", exc, exc_info=True)
|
|
3952
|
+
|
|
3953
|
+
async def _phase_self_reflection(self) -> None:
|
|
3954
|
+
"""Run self-reflection component weekly."""
|
|
3955
|
+
now = datetime.now(timezone.utc)
|
|
3956
|
+
interval_days = self.config.self_reflection_interval_days
|
|
3957
|
+
last = self._periodic_last.get("self_reflection")
|
|
3958
|
+
if isinstance(last, datetime):
|
|
3959
|
+
elapsed = (now - last).total_seconds() / 86400
|
|
3960
|
+
if elapsed < interval_days:
|
|
3961
|
+
return
|
|
3962
|
+
try:
|
|
3963
|
+
cognition = self._registry.cognition
|
|
3964
|
+
if cognition is not None and hasattr(cognition, "self_reflect"):
|
|
3965
|
+
with self._periodic_attempt("self_reflection", now):
|
|
3966
|
+
await cognition.self_reflect()
|
|
3967
|
+
logger.info("Phase self_reflection: complete")
|
|
3968
|
+
except Exception as exc:
|
|
3969
|
+
self.stats.errors += 1
|
|
3970
|
+
logger.error("Phase self_reflection error: %s", exc, exc_info=True)
|
|
3971
|
+
|
|
3972
|
+
async def _phase_skill_triggers(self, event_text: str) -> None:
|
|
3973
|
+
"""Evaluate skill triggers from recent events."""
|
|
3974
|
+
skills = self._registry.skills
|
|
3975
|
+
if skills is None:
|
|
3976
|
+
return
|
|
3977
|
+
try:
|
|
3978
|
+
if hasattr(skills, "evaluate_triggers"):
|
|
3979
|
+
loaded = await skills.evaluate_triggers(event_text)
|
|
3980
|
+
self.stats.skills_loaded = len(loaded)
|
|
3981
|
+
except Exception as exc:
|
|
3982
|
+
self.stats.errors += 1
|
|
3983
|
+
logger.error("Phase skill_triggers error: %s", exc, exc_info=True)
|
|
3984
|
+
|
|
3985
|
+
async def _phase_skill_evict(self) -> None:
|
|
3986
|
+
"""Evict cold skills after execution."""
|
|
3987
|
+
skills = self._registry.skills
|
|
3988
|
+
if skills is None:
|
|
3989
|
+
return
|
|
3990
|
+
try:
|
|
3991
|
+
if hasattr(skills, "evict_cold"):
|
|
3992
|
+
evicted = await skills.evict_cold()
|
|
3993
|
+
self.stats.skills_evicted += evicted
|
|
3994
|
+
except Exception as exc:
|
|
3995
|
+
logger.debug("Phase skill_evict error (non-fatal): %s", exc)
|
|
3996
|
+
|
|
3997
|
+
# ------------------------------------------------------------------
|
|
3998
|
+
# Multi-Agent Phases (v0.7.0)
|
|
3999
|
+
# ------------------------------------------------------------------
|
|
4000
|
+
|
|
4001
|
+
async def _phase_agent_heartbeat(self) -> None:
|
|
4002
|
+
"""Check agent status and mark offline if heartbeat timeout."""
|
|
4003
|
+
agent_store = self._registry.agent_store
|
|
4004
|
+
if agent_store is None:
|
|
4005
|
+
return
|
|
4006
|
+
try:
|
|
4007
|
+
# Get agents with old last_seen_at
|
|
4008
|
+
from datetime import timedelta
|
|
4009
|
+
threshold = datetime.now(timezone.utc) - timedelta(minutes=5)
|
|
4010
|
+
|
|
4011
|
+
agents = agent_store.list(status=["online", "busy"])
|
|
4012
|
+
for agent in agents:
|
|
4013
|
+
if agent.last_seen_at and agent.last_seen_at < threshold:
|
|
4014
|
+
logger.info("Agent %s marked offline (no heartbeat)", agent.name)
|
|
4015
|
+
await agent_store.set_offline(agent.agent_id)
|
|
4016
|
+
|
|
4017
|
+
# Reassign pending initiatives
|
|
4018
|
+
initiative_store = self._registry.initiative_store
|
|
4019
|
+
if initiative_store:
|
|
4020
|
+
reassigned = initiative_store.reassign_from_agent(
|
|
4021
|
+
agent.agent_id,
|
|
4022
|
+
only_pending=True,
|
|
4023
|
+
)
|
|
4024
|
+
if reassigned:
|
|
4025
|
+
logger.info("Reassigned %d initiatives from offline agent %s", reassigned, agent.name)
|
|
4026
|
+
except Exception as exc:
|
|
4027
|
+
logger.debug("Phase agent_heartbeat error (non-fatal): %s", exc)
|
|
4028
|
+
|
|
4029
|
+
async def _phase_startup_repush(self) -> None:
|
|
4030
|
+
"""On first tick: prune orphaned initiatives and re-push pending to delivery."""
|
|
4031
|
+
if self.stats.ticks != 1:
|
|
4032
|
+
return
|
|
4033
|
+
|
|
4034
|
+
initiative_store = self._registry.initiative_store
|
|
4035
|
+
delivery = self._registry.delivery
|
|
4036
|
+
graph = self._registry.graph
|
|
4037
|
+
|
|
4038
|
+
if initiative_store is None:
|
|
4039
|
+
return
|
|
4040
|
+
|
|
4041
|
+
try:
|
|
4042
|
+
# 1. Cancel initiatives whose entity no longer exists in graph
|
|
4043
|
+
if graph is not None and hasattr(graph, "driver"):
|
|
4044
|
+
pending = initiative_store.list(status=["pending"], limit=1000)
|
|
4045
|
+
pruned = 0
|
|
4046
|
+
for initiative in pending:
|
|
4047
|
+
entity_id = initiative.entity_id
|
|
4048
|
+
if not entity_id:
|
|
4049
|
+
continue
|
|
4050
|
+
try:
|
|
4051
|
+
async with graph.driver.session(database=graph.database) as session:
|
|
4052
|
+
result = await session.run(
|
|
4053
|
+
"MATCH (n {id: $id}) RETURN count(n) as c",
|
|
4054
|
+
{"id": entity_id},
|
|
4055
|
+
)
|
|
4056
|
+
record = await result.single()
|
|
4057
|
+
if record is None or record["c"] == 0:
|
|
4058
|
+
initiative_store.cancel(
|
|
4059
|
+
initiative.id,
|
|
4060
|
+
cancelled_by="autonomy_loop",
|
|
4061
|
+
reason="entity_no_longer_exists",
|
|
4062
|
+
)
|
|
4063
|
+
pruned += 1
|
|
4064
|
+
logger.info(
|
|
4065
|
+
"Pruned orphaned initiative %s (entity %s not in graph)",
|
|
4066
|
+
initiative.id,
|
|
4067
|
+
entity_id,
|
|
4068
|
+
)
|
|
4069
|
+
except Exception as exc:
|
|
4070
|
+
logger.debug("Graph check failed for %s: %s", entity_id, exc)
|
|
4071
|
+
|
|
4072
|
+
if pruned:
|
|
4073
|
+
logger.info("Pruned %d orphaned initiatives on startup", pruned)
|
|
4074
|
+
|
|
4075
|
+
# 2. Re-push remaining pending initiatives to delivery bridge.
|
|
4076
|
+
# Routed through the SAME gated path as the main loop, so a
|
|
4077
|
+
# go-live restart cannot flush an unfiltered / unrated backlog:
|
|
4078
|
+
# only reach-out types, sanitised, staleness-guarded, and rate-
|
|
4079
|
+
# limited per recipient are (shadow-)delivered.
|
|
4080
|
+
if delivery is not None:
|
|
4081
|
+
pending = initiative_store.list(status=["pending"], limit=100)
|
|
4082
|
+
repushed = 0
|
|
4083
|
+
for initiative in pending:
|
|
4084
|
+
payload = self._stored_initiative_delivery_payload(initiative)
|
|
4085
|
+
try:
|
|
4086
|
+
if await self._route_reachout_delivery(payload, delivery):
|
|
4087
|
+
repushed += 1
|
|
4088
|
+
except Exception as exc:
|
|
4089
|
+
logger.debug("Failed to re-push initiative %s: %s", initiative.id, exc)
|
|
4090
|
+
|
|
4091
|
+
if repushed:
|
|
4092
|
+
logger.info("Re-pushed %d pending reach-out initiatives to delivery bridge", repushed)
|
|
4093
|
+
|
|
4094
|
+
except Exception as exc:
|
|
4095
|
+
self.stats.errors += 1
|
|
4096
|
+
logger.error("Startup re-push phase error: %s", exc, exc_info=True)
|
|
4097
|
+
|
|
4098
|
+
async def _phase_initiative_timeout(self) -> None:
|
|
4099
|
+
"""Check for timed-out initiatives."""
|
|
4100
|
+
initiative_store = self._registry.initiative_store
|
|
4101
|
+
if initiative_store is None:
|
|
4102
|
+
return
|
|
4103
|
+
try:
|
|
4104
|
+
now = datetime.now(timezone.utc)
|
|
4105
|
+
timed_out = initiative_store.find_timed_out(now)
|
|
4106
|
+
|
|
4107
|
+
for initiative in timed_out:
|
|
4108
|
+
logger.warning(
|
|
4109
|
+
"Initiative %s timed out after %ds",
|
|
4110
|
+
initiative.id,
|
|
4111
|
+
initiative.timeout_seconds,
|
|
4112
|
+
)
|
|
4113
|
+
|
|
4114
|
+
initiative_store.update(
|
|
4115
|
+
initiative.id,
|
|
4116
|
+
status="failed",
|
|
4117
|
+
failed_at=now.isoformat(),
|
|
4118
|
+
failed_reason="timeout_exceeded",
|
|
4119
|
+
)
|
|
4120
|
+
|
|
4121
|
+
initiative_store.log_history(
|
|
4122
|
+
initiative.id,
|
|
4123
|
+
action="timed_out",
|
|
4124
|
+
agent_id=initiative.assigned_agent_id,
|
|
4125
|
+
details={"timeout_seconds": initiative.timeout_seconds},
|
|
4126
|
+
)
|
|
4127
|
+
except Exception as exc:
|
|
4128
|
+
logger.debug("Phase initiative_timeout error (non-fatal): %s", exc)
|
|
4129
|
+
|
|
4130
|
+
async def _phase_approval_timeout(self) -> None:
|
|
4131
|
+
"""Fail BLOCKED jobs whose owner-approval window expired (v0.17.0).
|
|
4132
|
+
|
|
4133
|
+
Jobs blocked with ``awaiting_owner_approval`` older than
|
|
4134
|
+
COLONY_APPROVAL_TIMEOUT_HOURS (default 72) are failed with reason
|
|
4135
|
+
``owner_approval_timeout`` so they never execute silently later.
|
|
4136
|
+
"""
|
|
4137
|
+
task_queue = getattr(self._registry, "task_queue", None)
|
|
4138
|
+
if task_queue is None:
|
|
4139
|
+
return
|
|
4140
|
+
try:
|
|
4141
|
+
timeout_hours = float(os.environ.get("COLONY_APPROVAL_TIMEOUT_HOURS", "72"))
|
|
4142
|
+
expired = await task_queue.queue.expire_blocked_approvals(
|
|
4143
|
+
datetime.now(timezone.utc), timeout_hours,
|
|
4144
|
+
)
|
|
4145
|
+
if expired:
|
|
4146
|
+
logger.info(
|
|
4147
|
+
"Failed %d blocked job(s) after %.0fh without owner approval",
|
|
4148
|
+
expired, timeout_hours,
|
|
4149
|
+
)
|
|
4150
|
+
except Exception as exc:
|
|
4151
|
+
logger.debug("Phase approval_timeout error (non-fatal): %s", exc)
|
|
4152
|
+
|
|
4153
|
+
async def _phase_stale_initiative_cleanup(self) -> None:
|
|
4154
|
+
"""Clean up initiatives stuck in acknowledged state."""
|
|
4155
|
+
initiative_store = self._registry.initiative_store
|
|
4156
|
+
agent_store = self._registry.agent_store
|
|
4157
|
+
if initiative_store is None or agent_store is None:
|
|
4158
|
+
return
|
|
4159
|
+
try:
|
|
4160
|
+
from datetime import timedelta
|
|
4161
|
+
threshold = datetime.now(timezone.utc) - timedelta(hours=1)
|
|
4162
|
+
|
|
4163
|
+
stale = initiative_store.find_stale_acknowledged(threshold)
|
|
4164
|
+
|
|
4165
|
+
for initiative in stale:
|
|
4166
|
+
agent = agent_store.get(initiative.assigned_agent_id)
|
|
4167
|
+
|
|
4168
|
+
if agent is None or agent.status != "online":
|
|
4169
|
+
logger.warning(
|
|
4170
|
+
"Initiative %s stuck in acknowledged, reassigning",
|
|
4171
|
+
initiative.id,
|
|
4172
|
+
)
|
|
4173
|
+
initiative_store.update(
|
|
4174
|
+
initiative.id,
|
|
4175
|
+
status="pending",
|
|
4176
|
+
assigned_agent_id=None,
|
|
4177
|
+
stale_reason="agent_offline_with_acknowledged",
|
|
4178
|
+
)
|
|
4179
|
+
except Exception as exc:
|
|
4180
|
+
logger.debug("Phase stale_initiative_cleanup error (non-fatal): %s", exc)
|
|
4181
|
+
|
|
4182
|
+
async def _phase_ghost_cleanup(self) -> None:
|
|
4183
|
+
"""Remove agents that registered but never connected."""
|
|
4184
|
+
agent_store = self._registry.agent_store
|
|
4185
|
+
initiative_store = self._registry.initiative_store
|
|
4186
|
+
if agent_store is None:
|
|
4187
|
+
return
|
|
4188
|
+
try:
|
|
4189
|
+
from datetime import timedelta
|
|
4190
|
+
threshold = datetime.now(timezone.utc) - timedelta(minutes=10)
|
|
4191
|
+
|
|
4192
|
+
ghosts = agent_store.list_ghosts(registered_before=threshold)
|
|
4193
|
+
|
|
4194
|
+
for ghost in ghosts:
|
|
4195
|
+
# Reassign initiatives first
|
|
4196
|
+
if initiative_store:
|
|
4197
|
+
initiatives = initiative_store.list(assigned_agent_id=ghost.agent_id)
|
|
4198
|
+
for init in initiatives:
|
|
4199
|
+
initiative_store.update(
|
|
4200
|
+
init.id,
|
|
4201
|
+
status="pending",
|
|
4202
|
+
assigned_agent_id=None,
|
|
4203
|
+
recovery_reason="agent_ghost",
|
|
4204
|
+
)
|
|
4205
|
+
|
|
4206
|
+
# Remove ghost
|
|
4207
|
+
agent_store.delete(ghost.agent_id)
|
|
4208
|
+
logger.info("Removed ghost agent %s", ghost.agent_id)
|
|
4209
|
+
|
|
4210
|
+
# Also expire stale in_session deliveries (v0.13.0)
|
|
4211
|
+
delivery = getattr(self._registry, "delivery", None)
|
|
4212
|
+
if delivery and hasattr(delivery, "expire_in_session_deliveries"):
|
|
4213
|
+
expired = delivery.expire_in_session_deliveries(max_age_hours=24)
|
|
4214
|
+
if expired:
|
|
4215
|
+
logger.info("Expired %d stale in_session deliveries", expired)
|
|
4216
|
+
except Exception as exc:
|
|
4217
|
+
logger.debug("Phase ghost_cleanup error (non-fatal): %s", exc)
|
|
4218
|
+
|
|
4219
|
+
async def _phase_database_backup(self) -> None:
|
|
4220
|
+
"""Periodic database backup for crash recovery."""
|
|
4221
|
+
try:
|
|
4222
|
+
agent_store = self._registry.agent_store
|
|
4223
|
+
initiative_store = self._registry.initiative_store
|
|
4224
|
+
|
|
4225
|
+
if agent_store and hasattr(agent_store, "backup"):
|
|
4226
|
+
agent_store.backup()
|
|
4227
|
+
|
|
4228
|
+
if initiative_store and hasattr(initiative_store, "backup"):
|
|
4229
|
+
initiative_store.backup()
|
|
4230
|
+
|
|
4231
|
+
logger.debug("Database backup complete")
|
|
4232
|
+
except Exception as exc:
|
|
4233
|
+
logger.warning("Phase database_backup error: %s", exc)
|
|
4234
|
+
|
|
4235
|
+
# ------------------------------------------------------------------
|
|
4236
|
+
# Sleep / wake
|
|
4237
|
+
# ------------------------------------------------------------------
|
|
4238
|
+
|
|
4239
|
+
async def _sleep_until_next_tick(self) -> None:
|
|
4240
|
+
self._wake_event.clear()
|
|
4241
|
+
try:
|
|
4242
|
+
await asyncio.wait_for(
|
|
4243
|
+
asyncio.shield(self._wake_event.wait()),
|
|
4244
|
+
timeout=self.config.tick_interval_secs,
|
|
4245
|
+
)
|
|
4246
|
+
except asyncio.TimeoutError:
|
|
4247
|
+
pass
|
|
4248
|
+
|
|
4249
|
+
def _on_wake_signal(self, event: Event) -> None:
|
|
4250
|
+
self._wake_event.set()
|
|
4251
|
+
|
|
4252
|
+
# ------------------------------------------------------------------
|
|
4253
|
+
# Helpers
|
|
4254
|
+
# ------------------------------------------------------------------
|
|
4255
|
+
|
|
4256
|
+
def _in_quiet_hours(self) -> bool:
|
|
4257
|
+
"""Check if current time is within quiet hours (in configured timezone)."""
|
|
4258
|
+
try:
|
|
4259
|
+
# Use configured timezone, fallback to UTC
|
|
4260
|
+
tz = ZoneInfo(self.config.timezone)
|
|
4261
|
+
now = datetime.now(tz)
|
|
4262
|
+
except Exception:
|
|
4263
|
+
now = datetime.now(timezone.utc)
|
|
4264
|
+
|
|
4265
|
+
try:
|
|
4266
|
+
start_h, start_m = map(int, self.config.quiet_hours_start.split(":"))
|
|
4267
|
+
end_h, end_m = map(int, self.config.quiet_hours_end.split(":"))
|
|
4268
|
+
except (ValueError, AttributeError):
|
|
4269
|
+
return False
|
|
4270
|
+
|
|
4271
|
+
from apsimo.util.quiet_hours import in_quiet_window
|
|
4272
|
+
return in_quiet_window(
|
|
4273
|
+
now.hour * 60 + now.minute,
|
|
4274
|
+
start_h * 60 + start_m,
|
|
4275
|
+
end_h * 60 + end_m,
|
|
4276
|
+
)
|
|
4277
|
+
|
|
4278
|
+
def _reset_hour_bucket(self) -> None:
|
|
4279
|
+
current_hour = datetime.now(timezone.utc).hour
|
|
4280
|
+
if current_hour != self.stats.hour_bucket:
|
|
4281
|
+
self.stats.actions_this_hour = 0
|
|
4282
|
+
self.stats.hour_bucket = current_hour
|
|
4283
|
+
|
|
4284
|
+
def _gather_event_text(self) -> str:
|
|
4285
|
+
try:
|
|
4286
|
+
recent = self.events.get_history(limit=10)
|
|
4287
|
+
parts = []
|
|
4288
|
+
for event in recent:
|
|
4289
|
+
event_type = getattr(event, "event_type", "")
|
|
4290
|
+
if event_type:
|
|
4291
|
+
parts.append(event_type)
|
|
4292
|
+
return " ".join(parts)
|
|
4293
|
+
except Exception:
|
|
4294
|
+
return ""
|
|
4295
|
+
|
|
4296
|
+
def status(self) -> dict:
|
|
4297
|
+
return {
|
|
4298
|
+
"running": self._running,
|
|
4299
|
+
"mode": self.config.mode.value,
|
|
4300
|
+
"timezone": self.config.timezone,
|
|
4301
|
+
"in_quiet_hours": self._in_quiet_hours(),
|
|
4302
|
+
"config": {
|
|
4303
|
+
"mode": self.config.mode.value,
|
|
4304
|
+
"enabled_phases": (list(self.config.enabled_phases)
|
|
4305
|
+
if self.config.enabled_phases is not None else None),
|
|
4306
|
+
"proposals_only": self.config.proposals_only,
|
|
4307
|
+
"timezone": self.config.timezone,
|
|
4308
|
+
"tick_interval_secs": self.config.tick_interval_secs,
|
|
4309
|
+
"initiative_confidence_threshold": self.config.initiative_confidence_threshold,
|
|
4310
|
+
"max_actions_per_hour": self.config.max_actions_per_hour,
|
|
4311
|
+
"quiet_hours_start": self.config.quiet_hours_start,
|
|
4312
|
+
"quiet_hours_end": self.config.quiet_hours_end,
|
|
4313
|
+
},
|
|
4314
|
+
"stats": self.stats.as_dict(),
|
|
4315
|
+
"phases": self.phase_timings(),
|
|
4316
|
+
}
|