apsimo 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apsimo/__init__.py +38 -0
- apsimo/__main__.py +6 -0
- apsimo/agent/__init__.py +6 -0
- apsimo/agent/client.py +276 -0
- apsimo/agent/models.py +46 -0
- apsimo/agents/__init__.py +20 -0
- apsimo/agents/models.py +264 -0
- apsimo/agents/store.py +861 -0
- apsimo/agents/websocket.py +522 -0
- apsimo/api/__init__.py +1 -0
- apsimo/api/auth_telemetry.py +287 -0
- apsimo/api/authority.py +1203 -0
- apsimo/api/contact_grants.py +347 -0
- apsimo/api/middleware.py +483 -0
- apsimo/api/routers/__init__.py +1 -0
- apsimo/api/routers/commitment_work.py +265 -0
- apsimo/api/routers/context_gate.py +123 -0
- apsimo/api/routers/executions.py +140 -0
- apsimo/api/routers/followup_plans.py +147 -0
- apsimo/api/routers/governed_actions.py +162 -0
- apsimo/api/routers/host.py +14473 -0
- apsimo/api/routers/initiative_work.py +115 -0
- apsimo/api/routers/mining.py +104 -0
- apsimo/api/routers/observations.py +110 -0
- apsimo/api/routers/social_state.py +225 -0
- apsimo/api/routers/task_queue.py +2715 -0
- apsimo/api/routers/temporal_followups.py +251 -0
- apsimo/api/routers/transport.py +110 -0
- apsimo/api/routers/transport_ingress_api.py +240 -0
- apsimo/api/schemas/__init__.py +1 -0
- apsimo/api/schemas/host.py +1949 -0
- apsimo/autonomy/cli.py +110 -0
- apsimo/autonomy/condition_worker.py +437 -0
- apsimo/autonomy/config.py +424 -0
- apsimo/autonomy/loop.py +4316 -0
- apsimo/autonomy/registry.py +339 -0
- apsimo/autonomy/scheduler.py +1822 -0
- apsimo/autonomy/synthesis.py +449 -0
- apsimo/backup.py +962 -0
- apsimo/beliefs/__init__.py +23 -0
- apsimo/beliefs/contradictions.py +109 -0
- apsimo/beliefs/decay.py +61 -0
- apsimo/beliefs/engine.py +479 -0
- apsimo/beliefs/models.py +67 -0
- apsimo/beliefs/promotion.py +41 -0
- apsimo/beliefs/resolve.py +58 -0
- apsimo/beliefs/source_claims.py +690 -0
- apsimo/beliefs/source_projection.py +883 -0
- apsimo/beliefs/source_time.py +208 -0
- apsimo/beliefs/store.py +133 -0
- apsimo/briefings/aggregators.py +824 -0
- apsimo/briefings/composer.py +420 -0
- apsimo/briefings/config.py +55 -0
- apsimo/briefings/delivery.py +439 -0
- apsimo/briefings/engagement.py +97 -0
- apsimo/briefings/engine.py +274 -0
- apsimo/briefings/enhancer.py +99 -0
- apsimo/briefings/models.py +183 -0
- apsimo/briefings/scheduler.py +382 -0
- apsimo/briefings/store.py +435 -0
- apsimo/chain/__init__.py +48 -0
- apsimo/chain/block.py +100 -0
- apsimo/chain/cli.py +704 -0
- apsimo/chain/genesis.py +443 -0
- apsimo/chain/identity.py +416 -0
- apsimo/chain/keys.py +1025 -0
- apsimo/chain/local_keys.py +187 -0
- apsimo/chain/manager.py +290 -0
- apsimo/chain/node.py +163 -0
- apsimo/chain/plugin_transactions.py +371 -0
- apsimo/chain/protocol.py +220 -0
- apsimo/chain/state_machine.py +676 -0
- apsimo/chain/storage.py +503 -0
- apsimo/chain/transactions.py +250 -0
- apsimo/chain/validation.py +397 -0
- apsimo/channels/__init__.py +1 -0
- apsimo/channels/manifest.py +31 -0
- apsimo/channels/migrations/001_channels_schema.sql +12 -0
- apsimo/channels/phone_gateways.py +42 -0
- apsimo/channels/presence.py +188 -0
- apsimo/channels/router.py +235 -0
- apsimo/channels/store.py +231 -0
- apsimo/cli.py +2688 -0
- apsimo/cognition/__init__.py +11 -0
- apsimo/cognition/charter.py +398 -0
- apsimo/cognition/drive_governance.py +3530 -0
- apsimo/cognition/evidence_pipeline.py +1627 -0
- apsimo/cognition/external_events.py +932 -0
- apsimo/cognition/goal_spine.py +3488 -0
- apsimo/cognition/introspection.py +214 -0
- apsimo/cognition/prompt.py +150 -0
- apsimo/cognition/runtime.py +108 -0
- apsimo/cognition/trigger.py +154 -0
- apsimo/commitments/__init__.py +18 -0
- apsimo/commitments/local_work.py +355 -0
- apsimo/commitments/store.py +1052 -0
- apsimo/commitments/work.py +91 -0
- apsimo/compat.py +53 -0
- apsimo/compression/__init__.py +467 -0
- apsimo/connectors/__init__.py +21 -0
- apsimo/connectors/base.py +152 -0
- apsimo/connectors/caldav_calendar.py +125 -0
- apsimo/connectors/fs_documents.py +85 -0
- apsimo/connectors/imap_email.py +138 -0
- apsimo/connectors/manager.py +218 -0
- apsimo/connectors/webhook_pull.py +88 -0
- apsimo/contacts/__init__.py +33 -0
- apsimo/contacts/comms.py +357 -0
- apsimo/contacts/config.py +79 -0
- apsimo/contacts/exporters/__init__.py +1 -0
- apsimo/contacts/exporters/vcard.py +71 -0
- apsimo/contacts/identity_links.py +251 -0
- apsimo/contacts/importer.py +280 -0
- apsimo/contacts/importers/__init__.py +1 -0
- apsimo/contacts/importers/batch.py +43 -0
- apsimo/contacts/importers/macos_contacts.py +101 -0
- apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
- apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
- apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
- apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
- apsimo/contacts/migrations/005_identity_links.sql +27 -0
- apsimo/contacts/models.py +308 -0
- apsimo/contacts/scoring.py +16 -0
- apsimo/contacts/store.py +1623 -0
- apsimo/contacts/transport_ingress.py +252 -0
- apsimo/contacts/world_bridge.py +314 -0
- apsimo/contextgate/__init__.py +69 -0
- apsimo/contextgate/chunker.py +169 -0
- apsimo/contextgate/estimate.py +54 -0
- apsimo/contextgate/gate.py +313 -0
- apsimo/contextgate/retrieve.py +115 -0
- apsimo/delivery/__init__.py +16 -0
- apsimo/delivery/bridge.py +1260 -0
- apsimo/delivery/channels.py +526 -0
- apsimo/delivery/classification.py +50 -0
- apsimo/delivery/rate_limiter.py +268 -0
- apsimo/delivery/reachout_policy.py +206 -0
- apsimo/directed/__init__.py +22 -0
- apsimo/directed/audit.py +167 -0
- apsimo/directed/intake.py +95 -0
- apsimo/directed/models.py +191 -0
- apsimo/directed/service.py +509 -0
- apsimo/directives/__init__.py +25 -0
- apsimo/directives/evidence.py +87 -0
- apsimo/directives/extractor.py +188 -0
- apsimo/directives/guard.py +364 -0
- apsimo/directives/models.py +206 -0
- apsimo/directives/service.py +372 -0
- apsimo/directives/store.py +167 -0
- apsimo/doctor.py +2173 -0
- apsimo/environment.py +43 -0
- apsimo/events/__init__.py +33 -0
- apsimo/events/broadcaster.py +98 -0
- apsimo/events/bus.py +217 -0
- apsimo/events/journal.py +863 -0
- apsimo/events/stream.py +131 -0
- apsimo/events/types.py +150 -0
- apsimo/execution_results.py +357 -0
- apsimo/feedback/__init__.py +5 -0
- apsimo/feedback/store.py +76 -0
- apsimo/feeds/__init__.py +19 -0
- apsimo/feeds/cli.py +84 -0
- apsimo/feeds/engine.py +437 -0
- apsimo/feeds/example-feed.yaml +77 -0
- apsimo/feeds/hermes_cron.py +126 -0
- apsimo/feeds/manager.py +235 -0
- apsimo/feeds/spec.py +250 -0
- apsimo/feeds/template.py +202 -0
- apsimo/gate/__init__.py +18 -0
- apsimo/gate/audit.py +61 -0
- apsimo/gate/communication_policy.py +166 -0
- apsimo/gate/config.py +72 -0
- apsimo/gate/context_provenance.py +170 -0
- apsimo/gate/env_risk.py +226 -0
- apsimo/gate/guard_audit.py +353 -0
- apsimo/gate/layers/__init__.py +1 -0
- apsimo/gate/layers/base.py +15 -0
- apsimo/gate/layers/l1_recipient.py +66 -0
- apsimo/gate/layers/l2_pii.py +134 -0
- apsimo/gate/layers/l3_cross_context.py +50 -0
- apsimo/gate/layers/l4_trust_tier.py +78 -0
- apsimo/gate/layers/l5_injection.py +199 -0
- apsimo/gate/layers/l6_review.py +86 -0
- apsimo/gate/layers/l7_delay.py +100 -0
- apsimo/gate/layers/tom2_epistemic.py +185 -0
- apsimo/gate/models.py +64 -0
- apsimo/gate/pending_dispatch.py +5 -0
- apsimo/gate/pipeline.py +206 -0
- apsimo/gate/rejection.py +259 -0
- apsimo/gate/response_guard.py +700 -0
- apsimo/gate/rulesets/injection_v1.yaml +51 -0
- apsimo/gate/surface_policy.py +189 -0
- apsimo/gate/taint.py +226 -0
- apsimo/genesis.json +9 -0
- apsimo/goals/__init__.py +100 -0
- apsimo/goals/config.py +38 -0
- apsimo/goals/decomposer.py +421 -0
- apsimo/goals/engine.py +617 -0
- apsimo/goals/inference.py +354 -0
- apsimo/goals/models.py +302 -0
- apsimo/goals/priority.py +270 -0
- apsimo/goals/queue_bridge.py +149 -0
- apsimo/goals/replan.py +450 -0
- apsimo/goals/schema.sql +89 -0
- apsimo/goals/store.py +692 -0
- apsimo/governed_actions.py +1708 -0
- apsimo/harness_integration/__init__.py +45 -0
- apsimo/harness_integration/context.py +41 -0
- apsimo/harness_integration/skills.py +231 -0
- apsimo/identity/__init__.py +26 -0
- apsimo/identity/participants.py +181 -0
- apsimo/identity/resolver.py +329 -0
- apsimo/identity_bootstrap/__init__.py +5 -0
- apsimo/identity_bootstrap/builder.py +208 -0
- apsimo/identity_bootstrap/corpus.py +443 -0
- apsimo/identity_bootstrap/models.py +54 -0
- apsimo/identity_bootstrap/runner.py +353 -0
- apsimo/identity_bootstrap/seeders/__init__.py +25 -0
- apsimo/identity_bootstrap/seeders/briefings.py +109 -0
- apsimo/identity_bootstrap/seeders/chain.py +57 -0
- apsimo/identity_bootstrap/seeders/goals.py +128 -0
- apsimo/identity_bootstrap/seeders/memory.py +191 -0
- apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
- apsimo/identity_bootstrap/seeders/relationship.py +152 -0
- apsimo/identity_bootstrap/seeders/sessions.py +67 -0
- apsimo/identity_bootstrap/seeders/skills.py +92 -0
- apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
- apsimo/identity_bootstrap/seeders/world_model.py +143 -0
- apsimo/identity_bootstrap/self_query.py +92 -0
- apsimo/identity_bootstrap/self_reflection.py +155 -0
- apsimo/identity_bootstrap/skill.py +37 -0
- apsimo/identity_bootstrap/verifier.py +436 -0
- apsimo/initiatives/__init__.py +20 -0
- apsimo/initiatives/action_registry.py +454 -0
- apsimo/initiatives/approval_authority.py +2105 -0
- apsimo/initiatives/approval_policy.py +123 -0
- apsimo/initiatives/assignment.py +263 -0
- apsimo/initiatives/backup_evidence.py +100 -0
- apsimo/initiatives/context_freshness.py +103 -0
- apsimo/initiatives/models.py +318 -0
- apsimo/initiatives/native_work.py +270 -0
- apsimo/initiatives/standing_approvals.py +232 -0
- apsimo/initiatives/store.py +1081 -0
- apsimo/initiatives/temporal_followup.py +410 -0
- apsimo/intelligence/__init__.py +1 -0
- apsimo/intelligence/cognition/__init__.py +24 -0
- apsimo/intelligence/cognition/gap_detector.py +148 -0
- apsimo/intelligence/cognition/metalearner.py +547 -0
- apsimo/intelligence/cognition/metrics_collector.py +217 -0
- apsimo/intelligence/cognition/performance_index.py +299 -0
- apsimo/intelligence/cognition/registry.py +192 -0
- apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
- apsimo/intelligence/cognition/types.py +16 -0
- apsimo/intelligence/components/__init__.py +66 -0
- apsimo/intelligence/components/anomaly_detector.py +413 -0
- apsimo/intelligence/components/initiative_engine.py +2643 -0
- apsimo/intelligence/components/preference_learner.py +521 -0
- apsimo/intelligence/components/research_orchestrator.py +358 -0
- apsimo/intelligence/components/self_directed_thinker.py +221 -0
- apsimo/intelligence/components/self_reflector.py +252 -0
- apsimo/intelligence/components/session_continuity.py +154 -0
- apsimo/intelligence/components/task_planner.py +320 -0
- apsimo/intelligence/components/tool_learner.py +217 -0
- apsimo/intelligence/graph/__init__.py +79 -0
- apsimo/intelligence/graph/client.py +2483 -0
- apsimo/intelligence/graph/consolidator.py +405 -0
- apsimo/intelligence/graph/distiller.py +312 -0
- apsimo/intelligence/graph/migrations.py +129 -0
- apsimo/intelligence/graph/queries.py +248 -0
- apsimo/intelligence/graph/recall.py +281 -0
- apsimo/intelligence/graph/reconciler.py +144 -0
- apsimo/intelligence/graph/schema.py +337 -0
- apsimo/intelligence/graph/selection.py +252 -0
- apsimo/intelligence/learning/__init__.py +17 -0
- apsimo/intelligence/learning/continuous_learner.py +245 -0
- apsimo/intelligence/learning/feedback_store.py +321 -0
- apsimo/intelligence/mind_model/__init__.py +1 -0
- apsimo/intelligence/mind_model/graph_baseline.py +136 -0
- apsimo/intelligence/mind_model/signal_collector.py +361 -0
- apsimo/intelligence/relationships/__init__.py +11 -0
- apsimo/intelligence/relationships/profiler.py +389 -0
- apsimo/intelligence/relationships/scorer.py +560 -0
- apsimo/intelligence/relationships/signal_floor.py +66 -0
- apsimo/intelligence/relationships/trust_tiers.py +300 -0
- apsimo/intelligence/synthesis/__init__.py +40 -0
- apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
- apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
- apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
- apsimo/intelligence/synthesis/insight_store.py +79 -0
- apsimo/intelligence/synthesis/insight_validator.py +183 -0
- apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
- apsimo/intelligence/turn_middleware/__init__.py +15 -0
- apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
- apsimo/mcp/__init__.py +41 -0
- apsimo/mcp/__main__.py +6 -0
- apsimo/mcp/config.py +287 -0
- apsimo/mcp/server.py +501 -0
- apsimo/migrations.py +187 -0
- apsimo/mining/__init__.py +27 -0
- apsimo/mining/corpus.py +239 -0
- apsimo/mining/escalations.py +289 -0
- apsimo/mining/models.py +169 -0
- apsimo/mining/store.py +210 -0
- apsimo/models/__init__.py +30 -0
- apsimo/models/memory.py +80 -0
- apsimo/models/mesh.py +72 -0
- apsimo/models/person.py +104 -0
- apsimo/models/signal.py +108 -0
- apsimo/observations/__init__.py +15 -0
- apsimo/observations/store.py +277 -0
- apsimo/patterns/__init__.py +6 -0
- apsimo/patterns/extract.py +187 -0
- apsimo/patterns/store.py +227 -0
- apsimo/persona/__init__.py +1 -0
- apsimo/persona/engine.py +611 -0
- apsimo/persona/manifest.py +140 -0
- apsimo/projects/__init__.py +28 -0
- apsimo/projects/engine.py +1681 -0
- apsimo/projects/event_outbox.py +188 -0
- apsimo/projects/models.py +216 -0
- apsimo/projects/planner.py +181 -0
- apsimo/projects/store.py +1446 -0
- apsimo/proposals/__init__.py +12 -0
- apsimo/proposals/engine.py +114 -0
- apsimo/proposals/models.py +207 -0
- apsimo/qualification/__init__.py +1 -0
- apsimo/qualification/cases.py +75 -0
- apsimo/qualification/cli.py +51 -0
- apsimo/qualification/memory_cases.py +209 -0
- apsimo/qualification/records.py +92 -0
- apsimo/qualification/report.py +87 -0
- apsimo/qualification/runner.py +311 -0
- apsimo/qualification/structured_cases.py +131 -0
- apsimo/reasoning/__init__.py +13 -0
- apsimo/reasoning/executor.py +506 -0
- apsimo/reasoning/loop.py +373 -0
- apsimo/reasoning/native_tools/__init__.py +16 -0
- apsimo/reasoning/native_tools/calculate.py +141 -0
- apsimo/reasoning/native_tools/file_ops.py +150 -0
- apsimo/reasoning/native_tools/web_search.py +49 -0
- apsimo/reasoning/tool_policy.py +182 -0
- apsimo/redact/__init__.py +176 -0
- apsimo/repos/__init__.py +5 -0
- apsimo/repos/mirrors.py +204 -0
- apsimo/research/__init__.py +41 -0
- apsimo/research/artifact.py +482 -0
- apsimo/research/gatherer.py +387 -0
- apsimo/research/pipeline.py +513 -0
- apsimo/research/search/__init__.py +7 -0
- apsimo/research/search/base.py +41 -0
- apsimo/research/search/brave.py +59 -0
- apsimo/research/search/cache.py +51 -0
- apsimo/research/search/duckduckgo.py +103 -0
- apsimo/research/search/orchestrator.py +119 -0
- apsimo/research/search/serpapi.py +59 -0
- apsimo/research/search/tavily.py +59 -0
- apsimo/research/synthesizer.py +309 -0
- apsimo/router/__init__.py +30 -0
- apsimo/router/complexity_scorer.py +148 -0
- apsimo/router/endpoints.py +153 -0
- apsimo/router/fallback.py +58 -0
- apsimo/router/functions.py +243 -0
- apsimo/router/native_policy.py +52 -0
- apsimo/router/router.py +762 -0
- apsimo/router/self_learning.py +174 -0
- apsimo/router/tiers.py +677 -0
- apsimo/sandbox/__init__.py +21 -0
- apsimo/sandbox/backend.py +195 -0
- apsimo/sandbox/manager.py +173 -0
- apsimo/scope_bounds.py +7 -0
- apsimo/secrets/__init__.py +6 -0
- apsimo/secrets/backends/__init__.py +8 -0
- apsimo/secrets/backends/base.py +42 -0
- apsimo/secrets/backends/env.py +110 -0
- apsimo/secrets/backends/keyring.py +72 -0
- apsimo/secrets/backends/onepassword.py +232 -0
- apsimo/secrets/cli.py +191 -0
- apsimo/secrets/manager.py +160 -0
- apsimo/secrets/migration.py +101 -0
- apsimo/secrets/types.py +98 -0
- apsimo/seed.py +41 -0
- apsimo/self_model/__init__.py +37 -0
- apsimo/self_model/appraisals.py +673 -0
- apsimo/self_model/benchmark.py +1314 -0
- apsimo/self_model/brief.py +40 -0
- apsimo/self_model/event_concerns.py +1128 -0
- apsimo/self_model/execution_forecasts.py +353 -0
- apsimo/self_model/expectations.py +1595 -0
- apsimo/self_model/experiments.py +1150 -0
- apsimo/self_model/journal.py +148 -0
- apsimo/self_model/judgments.py +705 -0
- apsimo/self_model/native_outcomes.py +55 -0
- apsimo/self_model/params.py +220 -0
- apsimo/self_model/perspective.py +246 -0
- apsimo/self_model/reconcile.py +183 -0
- apsimo/self_model/reply_forecasts.py +381 -0
- apsimo/self_model/runtime_forecasts.py +296 -0
- apsimo/self_model/runtime_models.py +67 -0
- apsimo/self_model/settlement.py +207 -0
- apsimo/self_model/situation.py +1731 -0
- apsimo/self_model/store.py +883 -0
- apsimo/self_model/supervised.py +137 -0
- apsimo/self_model/thinker.py +99 -0
- apsimo/self_model/trust.py +388 -0
- apsimo/self_model/workspace.py +2388 -0
- apsimo/server.py +4197 -0
- apsimo/services/__init__.py +1 -0
- apsimo/services/agent_bridge.py +474 -0
- apsimo/services/initiative_executor.py +914 -0
- apsimo/services/instance.py +297 -0
- apsimo/sessions/__init__.py +22 -0
- apsimo/sessions/config.py +13 -0
- apsimo/sessions/context_loader.py +88 -0
- apsimo/sessions/federation_session.py +75 -0
- apsimo/sessions/isolated_session.py +98 -0
- apsimo/sessions/reports.py +84 -0
- apsimo/sessions/store.py +148 -0
- apsimo/setup.py +2818 -0
- apsimo/setup_hermes.py +879 -0
- apsimo/setup_local_work.py +218 -0
- apsimo/setup_native_goals.py +134 -0
- apsimo/setup_native_reviews.py +115 -0
- apsimo/skills/__init__.py +10 -0
- apsimo/skills/base.py +108 -0
- apsimo/skills/budget.py +28 -0
- apsimo/skills/executor.py +493 -0
- apsimo/skills/executors/__init__.py +1 -0
- apsimo/skills/executors/behavioral_correction.py +75 -0
- apsimo/skills/executors/capability_gap.py +38 -0
- apsimo/skills/executors/data_quality.py +163 -0
- apsimo/skills/executors/knowledge_acquisition.py +41 -0
- apsimo/skills/executors/operational_hygiene.py +185 -0
- apsimo/skills/executors/subsystem_health.py +169 -0
- apsimo/skills/hermes_export.py +431 -0
- apsimo/skills/index.py +123 -0
- apsimo/skills/learning/__init__.py +21 -0
- apsimo/skills/learning/novelty_detector.py +206 -0
- apsimo/skills/learning/pattern_extractor.py +199 -0
- apsimo/skills/learning/triggers.py +159 -0
- apsimo/skills/loader.py +246 -0
- apsimo/skills/migrations/002_progressive_loading.sql +6 -0
- apsimo/skills/migrations/backfill_triggers.py +20 -0
- apsimo/skills/models.py +202 -0
- apsimo/skills/packager.py +128 -0
- apsimo/skills/protocols.py +70 -0
- apsimo/skills/registry.py +191 -0
- apsimo/skills/runtime.py +58 -0
- apsimo/skills/sandbox_runner.py +229 -0
- apsimo/skills/scheduler.py +129 -0
- apsimo/skills/schema.py +79 -0
- apsimo/skills/security/__init__.py +12 -0
- apsimo/skills/security/guards.py +53 -0
- apsimo/skills/security/scanner.py +223 -0
- apsimo/skills_memory/__init__.py +26 -0
- apsimo/skills_memory/distill.py +159 -0
- apsimo/skills_memory/models.py +85 -0
- apsimo/skills_memory/retrieve.py +62 -0
- apsimo/skills_memory/store.py +172 -0
- apsimo/surprise/__init__.py +6 -0
- apsimo/surprise/accumulation.py +57 -0
- apsimo/surprise/scorer.py +102 -0
- apsimo/surprise/store.py +203 -0
- apsimo/task_queue/__init__.py +69 -0
- apsimo/task_queue/action_receipts.py +148 -0
- apsimo/task_queue/approval_relay_canary.py +108 -0
- apsimo/task_queue/config.py +85 -0
- apsimo/task_queue/contract.py +361 -0
- apsimo/task_queue/events.py +130 -0
- apsimo/task_queue/governor.py +1031 -0
- apsimo/task_queue/handlers/__init__.py +16 -0
- apsimo/task_queue/handlers/base.py +37 -0
- apsimo/task_queue/handlers/inference.py +640 -0
- apsimo/task_queue/handlers/monitoring.py +116 -0
- apsimo/task_queue/handlers/registry.py +75 -0
- apsimo/task_queue/handlers/subtask_handler.py +173 -0
- apsimo/task_queue/handlers/system_maintenance.py +147 -0
- apsimo/task_queue/mesh_integration.py +111 -0
- apsimo/task_queue/models.py +317 -0
- apsimo/task_queue/queue_manager.py +8286 -0
- apsimo/task_queue/routing.py +287 -0
- apsimo/task_queue/scheduler.py +252 -0
- apsimo/task_queue/schema.sql +197 -0
- apsimo/task_queue/work_control.py +342 -0
- apsimo/task_queue/worker.py +993 -0
- apsimo/telemetry.py +145 -0
- apsimo/tom/__init__.py +6 -0
- apsimo/tom/affect.py +387 -0
- apsimo/tom/approvals.py +171 -0
- apsimo/tom/arcs.py +896 -0
- apsimo/tom/asymmetry.py +131 -0
- apsimo/tom/eligibility.py +248 -0
- apsimo/tom/engagement.py +214 -0
- apsimo/tom/exposure.py +214 -0
- apsimo/tom/extractor.py +306 -0
- apsimo/tom/fact_adapters.py +144 -0
- apsimo/tom/facts.py +326 -0
- apsimo/tom/integration.py +592 -0
- apsimo/tom/leveled.py +118 -0
- apsimo/tom/levels.py +247 -0
- apsimo/tom/recipient_audit.py +995 -0
- apsimo/tom/recipient_simulator.py +593 -0
- apsimo/tom/source_lineage.py +93 -0
- apsimo/tom/tom2.py +277 -0
- apsimo/tom/visibility.py +559 -0
- apsimo/tom/visibility_store.py +414 -0
- apsimo/tools/__init__.py +0 -0
- apsimo/tools/definitions.py +740 -0
- apsimo/tools/handlers.py +943 -0
- apsimo/toolsmith/__init__.py +26 -0
- apsimo/toolsmith/authority.py +166 -0
- apsimo/toolsmith/engine.py +559 -0
- apsimo/toolsmith/integrity.py +100 -0
- apsimo/toolsmith/miner.py +145 -0
- apsimo/toolsmith/policy.py +110 -0
- apsimo/toolsmith/registry.py +635 -0
- apsimo/turns/__init__.py +17 -0
- apsimo/turns/audio.py +134 -0
- apsimo/turns/documents.py +235 -0
- apsimo/turns/executions.py +486 -0
- apsimo/turns/hermes_history.py +245 -0
- apsimo/turns/hermes_kanban.py +268 -0
- apsimo/turns/hermes_work.py +96 -0
- apsimo/turns/idempotency.py +752 -0
- apsimo/turns/local_work.py +115 -0
- apsimo/turns/media.py +581 -0
- apsimo/turns/reported_workers.py +196 -0
- apsimo/turns/source_annotations.py +283 -0
- apsimo/turns/source_attribution.py +154 -0
- apsimo/turns/source_read.py +351 -0
- apsimo/turns/source_vectors.py +263 -0
- apsimo/turns/video.py +210 -0
- apsimo/util/autonomy_preset.py +220 -0
- apsimo/util/instance.py +92 -0
- apsimo/util/model_output.py +25 -0
- apsimo/util/quiet_hours.py +27 -0
- apsimo/util/session_safety.py +37 -0
- apsimo/util/temporal.py +343 -0
- apsimo/vector/__init__.py +75 -0
- apsimo/vector/backfill.py +171 -0
- apsimo/vector/caption.py +114 -0
- apsimo/vector/collections.py +51 -0
- apsimo/vector/config.py +102 -0
- apsimo/vector/embedder.py +670 -0
- apsimo/vector/image_preprocess.py +406 -0
- apsimo/vector/image_store.py +296 -0
- apsimo/vector/indexes.py +162 -0
- apsimo/vector/migrate.py +334 -0
- apsimo/vector/multimodal_provider.py +417 -0
- apsimo/vector/multimodal_types.py +87 -0
- apsimo/vector/openai_provider.py +119 -0
- apsimo/vector/query.py +49 -0
- apsimo/vector/reranker.py +565 -0
- apsimo/vector/safety_image.py +159 -0
- apsimo/vector/scanner.py +197 -0
- apsimo/vector/setup.py +289 -0
- apsimo/vector/store.py +533 -0
- apsimo/vector/tiers.py +263 -0
- apsimo/work_orders.py +925 -0
- apsimo/workers/__init__.py +21 -0
- apsimo/workers/agent_bridge.py +640 -0
- apsimo/workers/colony_worker.py +382 -0
- apsimo/workers/queue_worker.py +441 -0
- apsimo/workers/skills_sync.py +152 -0
- apsimo/world_model/__init__.py +71 -0
- apsimo/world_model/causal_maintenance.py +131 -0
- apsimo/world_model/causal_policy.py +43 -0
- apsimo/world_model/causal_query.py +125 -0
- apsimo/world_model/confidence.py +54 -0
- apsimo/world_model/config.py +64 -0
- apsimo/world_model/constants.py +97 -0
- apsimo/world_model/entities.py +145 -0
- apsimo/world_model/expectation_resolvers.py +177 -0
- apsimo/world_model/extraction/__init__.py +7 -0
- apsimo/world_model/extraction/base.py +62 -0
- apsimo/world_model/extraction/conversation_extractor.py +262 -0
- apsimo/world_model/extraction/detector.py +74 -0
- apsimo/world_model/extraction/document_extractor.py +78 -0
- apsimo/world_model/extraction/formats/__init__.py +24 -0
- apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
- apsimo/world_model/extraction/formats/html_fmt.py +72 -0
- apsimo/world_model/extraction/formats/json_fmt.py +68 -0
- apsimo/world_model/extraction/formats/pdf.py +43 -0
- apsimo/world_model/extraction/formats/text.py +27 -0
- apsimo/world_model/extraction/llm_extractor.py +164 -0
- apsimo/world_model/extraction/pipeline.py +73 -0
- apsimo/world_model/integrations/__init__.py +5 -0
- apsimo/world_model/integrations/mind_model_bridge.py +115 -0
- apsimo/world_model/integrations/social_intel_bridge.py +120 -0
- apsimo/world_model/jobs/__init__.py +4 -0
- apsimo/world_model/jobs/extraction_job.py +168 -0
- apsimo/world_model/llm_extract.py +572 -0
- apsimo/world_model/neo4j/__init__.py +5 -0
- apsimo/world_model/neo4j/backend.py +654 -0
- apsimo/world_model/observations.py +155 -0
- apsimo/world_model/populator.py +307 -0
- apsimo/world_model/postgres/__init__.py +1 -0
- apsimo/world_model/postgres/backend.py +683 -0
- apsimo/world_model/relationships.py +25 -0
- apsimo/world_model/resolution/__init__.py +13 -0
- apsimo/world_model/resolution/entity_resolver.py +232 -0
- apsimo/world_model/resolution/merge_audit.py +16 -0
- apsimo/world_model/resolution/merge_workflow.py +117 -0
- apsimo/world_model/source_reports.py +121 -0
- apsimo/world_model/sqlite/__init__.py +4 -0
- apsimo/world_model/sqlite/backend.py +855 -0
- apsimo/world_model/sqlite/schema.sql +132 -0
- apsimo/world_model/store.py +545 -0
- apsimo-1.3.0.dist-info/METADATA +78 -0
- apsimo-1.3.0.dist-info/RECORD +614 -0
- apsimo-1.3.0.dist-info/WHEEL +5 -0
- apsimo-1.3.0.dist-info/entry_points.txt +11 -0
- apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
- apsimo-1.3.0.dist-info/top_level.txt +2 -0
- colony_sidecar/__init__.py +4 -0
apsimo/doctor.py
ADDED
|
@@ -0,0 +1,2173 @@
|
|
|
1
|
+
"""Colony doctor — configuration and health diagnostics (v0.19.0).
|
|
2
|
+
|
|
3
|
+
A check engine for every misconfiguration class the sidecar can have,
|
|
4
|
+
including the exact footguns that bit the production deployment:
|
|
5
|
+
|
|
6
|
+
- persisted LLM config ``baseUrl`` missing the ``/v1`` suffix (LiteLLM
|
|
7
|
+
then 404s against vllm and the entire internal cognition stack dies
|
|
8
|
+
silently with "all tiers exhausted")
|
|
9
|
+
- empty ``apiKey`` in that config (``OPENAI_API_KEY`` never exported,
|
|
10
|
+
LiteLLM refuses every call)
|
|
11
|
+
- contact store running ``:memory:`` (owner record lost on restart)
|
|
12
|
+
- ``COLONY_OWNER_CONTACT_ID`` unset or unresolvable (relationship +
|
|
13
|
+
thinking degraded, CRITICAL at autonomy loop start)
|
|
14
|
+
- launchd plist env changes applied with ``kickstart`` instead of
|
|
15
|
+
``bootout``/``bootstrap`` (process restarts with the stale env)
|
|
16
|
+
|
|
17
|
+
Local checks read the filesystem/environment only; server checks talk
|
|
18
|
+
HTTP (stdlib urllib — zero extra deps) to a running sidecar and degrade
|
|
19
|
+
to ``skip`` when it is down. Every check runs defensively: an exception
|
|
20
|
+
inside a check becomes a ``fail`` result, never a crashed run.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import re
|
|
28
|
+
import sqlite3
|
|
29
|
+
import urllib.error
|
|
30
|
+
import urllib.request
|
|
31
|
+
from urllib.parse import urlencode
|
|
32
|
+
from dataclasses import asdict, dataclass
|
|
33
|
+
from datetime import datetime, timezone
|
|
34
|
+
from pathlib import Path
|
|
35
|
+
from typing import Any, Callable, List, Optional, Tuple
|
|
36
|
+
|
|
37
|
+
# Check statuses
|
|
38
|
+
PASS = "pass"
|
|
39
|
+
WARN = "warn"
|
|
40
|
+
FAIL = "fail"
|
|
41
|
+
SKIP = "skip"
|
|
42
|
+
|
|
43
|
+
#: Providers that route through LiteLLM's openai/* path — for these the
|
|
44
|
+
#: persisted baseUrl becomes OPENAI_API_BASE and MUST end with /v1
|
|
45
|
+
#: (LiteLLM appends /chat/completions to it).
|
|
46
|
+
OPENAI_COMPAT_PROVIDERS = frozenset({"zai", "local", "custom", "lmstudio", "vllm", "openai"})
|
|
47
|
+
|
|
48
|
+
#: Providers that need no API key at all.
|
|
49
|
+
KEYLESS_PROVIDERS = frozenset({"ollama"})
|
|
50
|
+
|
|
51
|
+
_TRUTHY = frozenset({"1", "true", "yes", "on"})
|
|
52
|
+
|
|
53
|
+
_HOME_CHANNEL_RE = re.compile(r"^(\w+)_HOME_CHANNEL$")
|
|
54
|
+
|
|
55
|
+
#: Names of the server-side checks, in run order — used to emit skips
|
|
56
|
+
#: when the sidecar is unreachable.
|
|
57
|
+
SERVER_CHECK_NAMES = (
|
|
58
|
+
"server-health",
|
|
59
|
+
"server-auth",
|
|
60
|
+
"server-auth-migration",
|
|
61
|
+
"server-owner-contact",
|
|
62
|
+
"server-llm-router",
|
|
63
|
+
"server-embedder",
|
|
64
|
+
"server-memory-graph",
|
|
65
|
+
"server-fd-limit",
|
|
66
|
+
"server-blocked-approvals",
|
|
67
|
+
"server-worker-liveness",
|
|
68
|
+
"server-skills-observations",
|
|
69
|
+
"server-autonomy-posture",
|
|
70
|
+
"server-grant-envelope",
|
|
71
|
+
"server-self-model",
|
|
72
|
+
"server-adaptive-params",
|
|
73
|
+
"server-executor",
|
|
74
|
+
"server-projects",
|
|
75
|
+
"server-beliefs",
|
|
76
|
+
"server-workers-governor",
|
|
77
|
+
"server-sandbox",
|
|
78
|
+
"server-connectors",
|
|
79
|
+
"server-mining",
|
|
80
|
+
"server-directives",
|
|
81
|
+
"server-benchmark",
|
|
82
|
+
"server-toolsmith",
|
|
83
|
+
"server-workspace",
|
|
84
|
+
"server-expectations",
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
#: A QUEUED agent_action job older than this means no queue worker is
|
|
88
|
+
#: claiming — auto-approved jobs would sit QUEUED forever.
|
|
89
|
+
WORKER_LIVENESS_THRESHOLD_MINUTES = 15
|
|
90
|
+
|
|
91
|
+
#: How to get the queue worker scheduled (v0.20.0).
|
|
92
|
+
WORKER_CRON_REMEDY = (
|
|
93
|
+
"install the cron: re-run 'colony init' (Step 10e installs it), or add "
|
|
94
|
+
"'*/5 * * * * colony-queue-worker' to your crontab — the console script "
|
|
95
|
+
"ships with the pip package (or use "
|
|
96
|
+
"'python -m apsimo.workers.queue_worker')"
|
|
97
|
+
)
|
|
98
|
+
|
|
99
|
+
#: The launchd footgun: `launchctl kickstart` restarts the process with
|
|
100
|
+
#: the OLD environment, so .plist env edits silently do not apply.
|
|
101
|
+
PLIST_ENV_REMEDY = (
|
|
102
|
+
"If the sidecar runs under launchd and you changed plist env vars, re-apply them with "
|
|
103
|
+
"'launchctl bootout gui/$(id -u)/ai.aevonix.colony-sidecar && "
|
|
104
|
+
"launchctl bootstrap gui/$(id -u) ~/Library/LaunchAgents/ai.aevonix.colony-sidecar.plist' — "
|
|
105
|
+
"'launchctl kickstart' restarts the process with the stale environment."
|
|
106
|
+
)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
@dataclass
|
|
110
|
+
class CheckResult:
|
|
111
|
+
"""One diagnostic verdict."""
|
|
112
|
+
|
|
113
|
+
name: str
|
|
114
|
+
status: str # "pass" | "warn" | "fail" | "skip"
|
|
115
|
+
detail: str = ""
|
|
116
|
+
remedy: str = ""
|
|
117
|
+
|
|
118
|
+
def to_dict(self) -> dict:
|
|
119
|
+
return asdict(self)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
# ---------------------------------------------------------------------------
|
|
123
|
+
# Small helpers
|
|
124
|
+
# ---------------------------------------------------------------------------
|
|
125
|
+
|
|
126
|
+
def _state_dir() -> Path:
|
|
127
|
+
"""Resolve the state dir WITHOUT creating it (get_state_dir mkdirs)."""
|
|
128
|
+
explicit = os.environ.get("COLONY_STATE_DIR")
|
|
129
|
+
if explicit:
|
|
130
|
+
return Path(explicit).expanduser()
|
|
131
|
+
return Path.home() / ".colony" / "data"
|
|
132
|
+
|
|
133
|
+
|
|
134
|
+
def _run(name: str, fn: Callable[..., Any], *args: Any) -> List[CheckResult]:
|
|
135
|
+
"""Run one check defensively — an exception becomes a fail result."""
|
|
136
|
+
try:
|
|
137
|
+
result = fn(*args)
|
|
138
|
+
except Exception as exc: # noqa: BLE001 — the whole point
|
|
139
|
+
return [CheckResult(name=name, status=FAIL,
|
|
140
|
+
detail=f"check crashed: {type(exc).__name__}: {exc}")]
|
|
141
|
+
if isinstance(result, CheckResult):
|
|
142
|
+
return [result]
|
|
143
|
+
return list(result)
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _maybe_json(raw: str) -> Any:
|
|
147
|
+
try:
|
|
148
|
+
return json.loads(raw)
|
|
149
|
+
except (ValueError, TypeError):
|
|
150
|
+
return raw
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def _http_get(url: str, api_key: str = "", timeout: float = 10.0) -> Tuple[int, Any]:
|
|
154
|
+
"""GET a sidecar endpoint with X-API-Key auth.
|
|
155
|
+
|
|
156
|
+
Returns ``(status_code, parsed_body)``. HTTP error statuses are
|
|
157
|
+
returned, not raised; connection-level failures (server down)
|
|
158
|
+
propagate as ``urllib.error.URLError``/``OSError`` for the caller's
|
|
159
|
+
reachability handling.
|
|
160
|
+
"""
|
|
161
|
+
headers = {"X-API-Key": api_key} if api_key else {}
|
|
162
|
+
req = urllib.request.Request(url, headers=headers)
|
|
163
|
+
try:
|
|
164
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
165
|
+
raw = resp.read().decode("utf-8", "replace")
|
|
166
|
+
return resp.status, _maybe_json(raw)
|
|
167
|
+
except urllib.error.HTTPError as exc:
|
|
168
|
+
try:
|
|
169
|
+
raw = exc.read().decode("utf-8", "replace")
|
|
170
|
+
except Exception: # noqa: BLE001
|
|
171
|
+
raw = ""
|
|
172
|
+
return exc.code, _maybe_json(raw)
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def _reported_error(name: str, body: Any) -> Optional[CheckResult]:
|
|
176
|
+
"""WARN when a status body carries an ``error`` field.
|
|
177
|
+
|
|
178
|
+
Several status endpoints return ``{"available": true, "error": ...}``
|
|
179
|
+
when the subsystem crashed while reporting. Ignoring that field scored a
|
|
180
|
+
hard-crashing subsystem as healthy-but-idle; it is a degradation.
|
|
181
|
+
"""
|
|
182
|
+
if isinstance(body, dict) and body.get("error"):
|
|
183
|
+
return CheckResult(
|
|
184
|
+
name, WARN,
|
|
185
|
+
detail=f"status endpoint reported an internal error: {body['error']}",
|
|
186
|
+
remedy="check the sidecar log for the traceback behind this error",
|
|
187
|
+
)
|
|
188
|
+
return None
|
|
189
|
+
|
|
190
|
+
|
|
191
|
+
# ---------------------------------------------------------------------------
|
|
192
|
+
# Local checks (filesystem/env — no server needed)
|
|
193
|
+
# ---------------------------------------------------------------------------
|
|
194
|
+
|
|
195
|
+
def check_state_dir() -> CheckResult:
|
|
196
|
+
"""1. COLONY_STATE_DIR exists and is writable."""
|
|
197
|
+
path = _state_dir()
|
|
198
|
+
if not path.exists():
|
|
199
|
+
return CheckResult(
|
|
200
|
+
"state-dir", FAIL,
|
|
201
|
+
detail=f"state dir {path} does not exist",
|
|
202
|
+
remedy=f"mkdir -p {path} (or run 'colony init'); set COLONY_STATE_DIR if it should live elsewhere",
|
|
203
|
+
)
|
|
204
|
+
if not path.is_dir():
|
|
205
|
+
return CheckResult(
|
|
206
|
+
"state-dir", FAIL,
|
|
207
|
+
detail=f"{path} exists but is not a directory",
|
|
208
|
+
remedy="point COLONY_STATE_DIR at a writable directory",
|
|
209
|
+
)
|
|
210
|
+
if not os.access(path, os.W_OK):
|
|
211
|
+
return CheckResult(
|
|
212
|
+
"state-dir", FAIL,
|
|
213
|
+
detail=f"state dir {path} is not writable by uid {os.getuid()}",
|
|
214
|
+
remedy=f"chown/chmod {path} so the sidecar user can write to it",
|
|
215
|
+
)
|
|
216
|
+
return CheckResult("state-dir", PASS, detail=f"{path} exists and is writable")
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def check_llm_config() -> List[CheckResult]:
|
|
220
|
+
"""2. Persisted LLM config: exists, baseUrl /v1 suffix, apiKey, models."""
|
|
221
|
+
path = _state_dir() / ".colony-llm-config.json"
|
|
222
|
+
sub_names = ("llm-config-baseurl", "llm-config-apikey", "llm-config-models")
|
|
223
|
+
|
|
224
|
+
if not path.exists():
|
|
225
|
+
results = [CheckResult(
|
|
226
|
+
"llm-config", WARN,
|
|
227
|
+
detail=f"{path} not found — the router falls back to default Anthropic tiers "
|
|
228
|
+
"(needs ANTHROPIC_API_KEY in the sidecar env)",
|
|
229
|
+
remedy="run 'colony init' or POST /v1/host/configure from the host to persist an LLM config",
|
|
230
|
+
)]
|
|
231
|
+
results += [CheckResult(n, SKIP, detail="no persisted LLM config") for n in sub_names]
|
|
232
|
+
return results
|
|
233
|
+
|
|
234
|
+
try:
|
|
235
|
+
cfg = json.loads(path.read_text(encoding="utf-8"))
|
|
236
|
+
if not isinstance(cfg, dict):
|
|
237
|
+
raise ValueError("top-level JSON value is not an object")
|
|
238
|
+
except (OSError, ValueError) as exc:
|
|
239
|
+
results = [CheckResult(
|
|
240
|
+
"llm-config", FAIL,
|
|
241
|
+
detail=f"{path} is unreadable/corrupt: {exc}",
|
|
242
|
+
remedy="fix or delete the file, then re-run 'colony init' / POST /v1/host/configure",
|
|
243
|
+
)]
|
|
244
|
+
results += [CheckResult(n, SKIP, detail="LLM config unparseable") for n in sub_names]
|
|
245
|
+
return results
|
|
246
|
+
|
|
247
|
+
provider = str(cfg.get("provider", "anthropic") or "").strip().lower()
|
|
248
|
+
base_url = str(cfg.get("baseUrl", "") or "").strip()
|
|
249
|
+
api_key = str(cfg.get("apiKey", "") or "")
|
|
250
|
+
models = cfg.get("models") or {}
|
|
251
|
+
|
|
252
|
+
results = [CheckResult("llm-config", PASS, detail=f"{path} parsed (provider={provider})")]
|
|
253
|
+
|
|
254
|
+
# --- baseUrl /v1 suffix (the production vllm footgun) ---
|
|
255
|
+
if provider in OPENAI_COMPAT_PROVIDERS and base_url:
|
|
256
|
+
if base_url.rstrip("/").endswith("/v1"):
|
|
257
|
+
results.append(CheckResult(
|
|
258
|
+
"llm-config-baseurl", PASS,
|
|
259
|
+
detail=f"baseUrl {base_url} carries the /v1 suffix",
|
|
260
|
+
))
|
|
261
|
+
else:
|
|
262
|
+
fixed = base_url.rstrip("/") + "/v1"
|
|
263
|
+
results.append(CheckResult(
|
|
264
|
+
"llm-config-baseurl", WARN,
|
|
265
|
+
detail=f"baseUrl {base_url!r} (provider={provider}) does not end with /v1 — "
|
|
266
|
+
"LiteLLM posts to <baseUrl>/chat/completions, so vllm/OpenAI-compatible "
|
|
267
|
+
"servers return 404 on every call and the whole cognition stack dies "
|
|
268
|
+
'silently with "all tiers exhausted"',
|
|
269
|
+
remedy=f'edit {path}: set "baseUrl" to "{fixed}", then restart the sidecar. '
|
|
270
|
+
+ PLIST_ENV_REMEDY,
|
|
271
|
+
))
|
|
272
|
+
elif provider in OPENAI_COMPAT_PROVIDERS and not base_url and provider not in ("openai", "zai"):
|
|
273
|
+
results.append(CheckResult(
|
|
274
|
+
"llm-config-baseurl", WARN,
|
|
275
|
+
detail=f"provider={provider} but baseUrl is empty — LiteLLM will target the real "
|
|
276
|
+
"OpenAI API instead of your local server",
|
|
277
|
+
remedy=f'set "baseUrl" in {path} to your server\'s OpenAI-compatible endpoint '
|
|
278
|
+
'(ending in /v1, e.g. "http://127.0.0.1:8000/v1")',
|
|
279
|
+
))
|
|
280
|
+
else:
|
|
281
|
+
results.append(CheckResult(
|
|
282
|
+
"llm-config-baseurl", PASS,
|
|
283
|
+
detail=f"provider={provider} — no /v1 suffix requirement"
|
|
284
|
+
+ (f" (baseUrl={base_url})" if base_url else ""),
|
|
285
|
+
))
|
|
286
|
+
|
|
287
|
+
# --- apiKey non-empty (the OPENAI_API_KEY footgun) ---
|
|
288
|
+
if provider in KEYLESS_PROVIDERS:
|
|
289
|
+
results.append(CheckResult(
|
|
290
|
+
"llm-config-apikey", PASS, detail=f"provider={provider} needs no apiKey",
|
|
291
|
+
))
|
|
292
|
+
elif api_key.strip():
|
|
293
|
+
results.append(CheckResult("llm-config-apikey", PASS, detail="apiKey is set"))
|
|
294
|
+
else:
|
|
295
|
+
results.append(CheckResult(
|
|
296
|
+
"llm-config-apikey", FAIL,
|
|
297
|
+
detail=f"apiKey in {path} is empty — the router only exports OPENAI_API_KEY / "
|
|
298
|
+
"ANTHROPIC_API_KEY when apiKey is non-empty, so LiteLLM refuses every call",
|
|
299
|
+
remedy=f'set "apiKey" in {path} (for a local vllm any non-empty placeholder works, '
|
|
300
|
+
'e.g. "local"), then restart the sidecar. Alternatively export the provider '
|
|
301
|
+
"key directly in the sidecar environment.",
|
|
302
|
+
))
|
|
303
|
+
|
|
304
|
+
# --- models map ---
|
|
305
|
+
if isinstance(models, dict) and models:
|
|
306
|
+
results.append(CheckResult(
|
|
307
|
+
"llm-config-models", PASS,
|
|
308
|
+
detail="models: " + ", ".join(f"{k}={v}" for k, v in sorted(models.items())),
|
|
309
|
+
))
|
|
310
|
+
else:
|
|
311
|
+
results.append(CheckResult(
|
|
312
|
+
"llm-config-models", WARN,
|
|
313
|
+
detail="models map is empty — the router falls back to provider presets or "
|
|
314
|
+
"auto-discovery; for local providers the placeholder model IDs likely "
|
|
315
|
+
"do not exist on your server",
|
|
316
|
+
remedy=f'add "models" to {path}, e.g. '
|
|
317
|
+
'{"small": "llama3.2", "medium": "mistral", "large": "deepseek-r1"}',
|
|
318
|
+
))
|
|
319
|
+
return results
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def check_contacts_db() -> CheckResult:
|
|
323
|
+
"""3. Contact store must be persistent and (when present) a real DB."""
|
|
324
|
+
from apsimo.contacts.config import ContactsConfig
|
|
325
|
+
|
|
326
|
+
sqlite_path = ContactsConfig.from_env().sqlite_path
|
|
327
|
+
if sqlite_path == ":memory:":
|
|
328
|
+
return CheckResult(
|
|
329
|
+
"contacts-db", FAIL,
|
|
330
|
+
detail="contact store is configured as :memory: — every contact (including the "
|
|
331
|
+
"owner record the IdentityResolver depends on) is lost on restart",
|
|
332
|
+
remedy="set COLONY_CONTACTS_DB to a file path, or set COLONY_STATE_DIR so the "
|
|
333
|
+
"default $COLONY_STATE_DIR/colony-contacts.db applies",
|
|
334
|
+
)
|
|
335
|
+
|
|
336
|
+
path = Path(sqlite_path).expanduser()
|
|
337
|
+
if not path.parent.exists():
|
|
338
|
+
return CheckResult(
|
|
339
|
+
"contacts-db", FAIL,
|
|
340
|
+
detail=f"parent directory {path.parent} of contacts DB does not exist",
|
|
341
|
+
remedy=f"mkdir -p {path.parent} (or fix COLONY_CONTACTS_DB / COLONY_STATE_DIR)",
|
|
342
|
+
)
|
|
343
|
+
if not path.exists():
|
|
344
|
+
# A sibling contacts DB under another name is the signature of an
|
|
345
|
+
# env mismatch: the doctor shell resolves a different path than the
|
|
346
|
+
# running service (whose COLONY_CONTACTS_DB/COLONY_STATE_DIR live in
|
|
347
|
+
# its unit/plist, not this shell).
|
|
348
|
+
siblings = [p for p in path.parent.glob("*contacts*.db")
|
|
349
|
+
if p.exists()] if path.parent.exists() else []
|
|
350
|
+
if siblings:
|
|
351
|
+
return CheckResult(
|
|
352
|
+
"contacts-db", WARN,
|
|
353
|
+
detail=f"{path} does not exist, but {siblings[0]} does — this "
|
|
354
|
+
"doctor shell likely resolves a different path than the "
|
|
355
|
+
"running service",
|
|
356
|
+
remedy="run doctor with the service's env (COLONY_CONTACTS_DB / "
|
|
357
|
+
"COLONY_STATE_DIR), or align the two",
|
|
358
|
+
)
|
|
359
|
+
return CheckResult(
|
|
360
|
+
"contacts-db", PASS,
|
|
361
|
+
detail=f"{path} not created yet — the sidecar creates it on first start",
|
|
362
|
+
)
|
|
363
|
+
|
|
364
|
+
# Non-trivially sized (the DB itself or its -wal sibling) OR openable.
|
|
365
|
+
wal = path.with_name(path.name + "-wal")
|
|
366
|
+
size = path.stat().st_size
|
|
367
|
+
wal_size = wal.stat().st_size if wal.exists() else 0
|
|
368
|
+
if size >= 1024 or wal_size >= 1024:
|
|
369
|
+
return CheckResult(
|
|
370
|
+
"contacts-db", PASS,
|
|
371
|
+
detail=f"{path} present ({size} bytes, wal {wal_size} bytes)",
|
|
372
|
+
)
|
|
373
|
+
# An (almost) empty resolved DB next to a substantive sibling contacts DB
|
|
374
|
+
# is the env-mismatch signature again: the running service points at the
|
|
375
|
+
# sibling (via its unit/plist env) while this shell resolves a stale stub.
|
|
376
|
+
big_siblings = [p for p in path.parent.glob("*contacts*.db")
|
|
377
|
+
if p != path and p.stat().st_size >= 1024]
|
|
378
|
+
if big_siblings:
|
|
379
|
+
return CheckResult(
|
|
380
|
+
"contacts-db", WARN,
|
|
381
|
+
detail=f"{path} is empty ({size} bytes) but {big_siblings[0]} holds "
|
|
382
|
+
"real data — this doctor shell likely resolves a different "
|
|
383
|
+
"path than the running service",
|
|
384
|
+
remedy="run doctor with the service's env (COLONY_CONTACTS_DB / "
|
|
385
|
+
"COLONY_STATE_DIR), or delete the stale empty stub",
|
|
386
|
+
)
|
|
387
|
+
try:
|
|
388
|
+
conn = sqlite3.connect(str(path))
|
|
389
|
+
try:
|
|
390
|
+
conn.execute("PRAGMA schema_version").fetchone()
|
|
391
|
+
finally:
|
|
392
|
+
conn.close()
|
|
393
|
+
except sqlite3.Error as exc:
|
|
394
|
+
return CheckResult(
|
|
395
|
+
"contacts-db", FAIL,
|
|
396
|
+
detail=f"{path} exists but is not a readable SQLite database: {exc}",
|
|
397
|
+
remedy=f"move the corrupt file aside (mv {path} {path}.bad) and restart the "
|
|
398
|
+
"sidecar to recreate it; re-import contacts afterwards",
|
|
399
|
+
)
|
|
400
|
+
return CheckResult(
|
|
401
|
+
"contacts-db", PASS,
|
|
402
|
+
detail=f"{path} present and openable ({size} bytes)",
|
|
403
|
+
)
|
|
404
|
+
|
|
405
|
+
|
|
406
|
+
def check_owner_contact_id() -> CheckResult:
|
|
407
|
+
"""4. COLONY_OWNER_CONTACT_ID must be set for owner-aware subsystems."""
|
|
408
|
+
canonical = os.environ.get("COLONY_OWNER_CONTACT_ID", "")
|
|
409
|
+
legacy = os.environ.get("COLONY_HOST_CONTACT_ID", "")
|
|
410
|
+
if canonical:
|
|
411
|
+
return CheckResult(
|
|
412
|
+
"owner-contact-id", PASS, detail=f"COLONY_OWNER_CONTACT_ID={canonical}",
|
|
413
|
+
)
|
|
414
|
+
if legacy:
|
|
415
|
+
return CheckResult(
|
|
416
|
+
"owner-contact-id", WARN,
|
|
417
|
+
detail=f"owner only set via deprecated COLONY_HOST_CONTACT_ID={legacy}",
|
|
418
|
+
remedy="rename the variable to COLONY_OWNER_CONTACT_ID (same value)",
|
|
419
|
+
)
|
|
420
|
+
return CheckResult(
|
|
421
|
+
"owner-contact-id", WARN,
|
|
422
|
+
detail="COLONY_OWNER_CONTACT_ID is not set — owner-exclusion filters fail closed, so "
|
|
423
|
+
"relationship inference and self-directed thinking run degraded and the "
|
|
424
|
+
"autonomy loop logs CRITICAL at start (no owner-directed initiatives generated)",
|
|
425
|
+
remedy="set COLONY_OWNER_CONTACT_ID to the owner's contact CID (create one via "
|
|
426
|
+
"POST /v1/host/contacts if needed), then restart the sidecar",
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
def _effects_on_subsystems() -> List[str]:
|
|
431
|
+
"""Subsystems currently configured to produce real (live) effects."""
|
|
432
|
+
found: List[str] = []
|
|
433
|
+
try:
|
|
434
|
+
from apsimo.task_queue.governor import workers_mode
|
|
435
|
+
if workers_mode() == "live":
|
|
436
|
+
found.append("workers=live")
|
|
437
|
+
except Exception:
|
|
438
|
+
pass
|
|
439
|
+
try:
|
|
440
|
+
from apsimo.sandbox.manager import sandbox_mode
|
|
441
|
+
if sandbox_mode() == "live":
|
|
442
|
+
found.append("sandbox=live")
|
|
443
|
+
except Exception:
|
|
444
|
+
pass
|
|
445
|
+
return found
|
|
446
|
+
|
|
447
|
+
|
|
448
|
+
def _approval_result(detail: str, authority_mode: str) -> CheckResult:
|
|
449
|
+
"""PASS, unless approval authority is shadow while effects run live —
|
|
450
|
+
then WARN: approvals are recorded but NOT enforced against real effects."""
|
|
451
|
+
if authority_mode == "shadow":
|
|
452
|
+
effects = _effects_on_subsystems()
|
|
453
|
+
if effects:
|
|
454
|
+
return CheckResult(
|
|
455
|
+
"approval-policy", WARN,
|
|
456
|
+
detail=(
|
|
457
|
+
f"{detail}; approval authority mode=shadow while "
|
|
458
|
+
f"effects-on subsystem(s) run: {', '.join(effects)} — "
|
|
459
|
+
"approval decisions are observational, NOT enforced"
|
|
460
|
+
),
|
|
461
|
+
remedy="set COLONY_APPROVAL_AUTHORITY_MODE=enforce (or take "
|
|
462
|
+
"the effects-on subsystem out of live mode)",
|
|
463
|
+
)
|
|
464
|
+
return CheckResult("approval-policy", PASS, detail=detail)
|
|
465
|
+
|
|
466
|
+
|
|
467
|
+
def check_approval_policy() -> CheckResult:
|
|
468
|
+
"""5. Validate action policy and approval-authority migration mode."""
|
|
469
|
+
authority_raw = os.environ.get("COLONY_APPROVAL_AUTHORITY_MODE")
|
|
470
|
+
if authority_raw and authority_raw.strip().lower() not in ("shadow", "enforce"):
|
|
471
|
+
return CheckResult(
|
|
472
|
+
"approval-policy", FAIL,
|
|
473
|
+
detail=(
|
|
474
|
+
f"COLONY_APPROVAL_AUTHORITY_MODE={authority_raw!r} is invalid; "
|
|
475
|
+
"approval control surfaces fail closed with invalid configuration"
|
|
476
|
+
),
|
|
477
|
+
remedy="set COLONY_APPROVAL_AUTHORITY_MODE to 'shadow' or 'enforce'",
|
|
478
|
+
)
|
|
479
|
+
authority_mode = (authority_raw or "shadow").strip().lower()
|
|
480
|
+
raw = os.environ.get("COLONY_APPROVAL_POLICY")
|
|
481
|
+
if raw is None or not raw.strip():
|
|
482
|
+
return _approval_result(
|
|
483
|
+
"COLONY_APPROVAL_POLICY unset — defaults to strict; approval "
|
|
484
|
+
f"authority mode={authority_mode}",
|
|
485
|
+
authority_mode,
|
|
486
|
+
)
|
|
487
|
+
value = raw.strip().lower()
|
|
488
|
+
if value in ("strict", "graduated"):
|
|
489
|
+
return _approval_result(
|
|
490
|
+
f"COLONY_APPROVAL_POLICY={value}; approval authority "
|
|
491
|
+
f"mode={authority_mode}",
|
|
492
|
+
authority_mode,
|
|
493
|
+
)
|
|
494
|
+
return CheckResult(
|
|
495
|
+
"approval-policy", FAIL,
|
|
496
|
+
detail=f"COLONY_APPROVAL_POLICY={raw!r} is not a valid mode — the gate fails closed "
|
|
497
|
+
"to strict, so the policy you intended is silently NOT active",
|
|
498
|
+
remedy="set COLONY_APPROVAL_POLICY to 'strict' or 'graduated' (or unset it for strict)",
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
|
|
502
|
+
def check_standing_approvals() -> CheckResult:
|
|
503
|
+
"""6. Legacy standing JSON must be parseable and bounded when present."""
|
|
504
|
+
from apsimo.initiatives import standing_approvals
|
|
505
|
+
|
|
506
|
+
path = _state_dir() / standing_approvals._FILENAME
|
|
507
|
+
if not path.exists():
|
|
508
|
+
return CheckResult(
|
|
509
|
+
"standing-approvals", PASS, detail=f"{path.name} not present — no standing grants",
|
|
510
|
+
)
|
|
511
|
+
try:
|
|
512
|
+
data = json.loads(path.read_text(encoding="utf-8"))
|
|
513
|
+
if not isinstance(data, dict):
|
|
514
|
+
raise ValueError("not a JSON object")
|
|
515
|
+
except (OSError, ValueError) as exc:
|
|
516
|
+
return CheckResult(
|
|
517
|
+
"standing-approvals", FAIL,
|
|
518
|
+
detail=f"{path} is corrupt ({exc}) — the gate treats it as empty (fail closed), "
|
|
519
|
+
"so every previously granted 'always allow' is silently inactive",
|
|
520
|
+
remedy=f"fix or delete {path}; re-grant via POST /v1/host/queue/jobs/{{id}}/approve "
|
|
521
|
+
'with {"always": true}',
|
|
522
|
+
)
|
|
523
|
+
unbounded = [
|
|
524
|
+
name for name, entry in data.items()
|
|
525
|
+
if not isinstance(entry, dict)
|
|
526
|
+
or not entry.get("expires_at")
|
|
527
|
+
or not isinstance(entry.get("max_uses"), int)
|
|
528
|
+
]
|
|
529
|
+
if unbounded:
|
|
530
|
+
return CheckResult(
|
|
531
|
+
"standing-approvals", WARN,
|
|
532
|
+
detail=(
|
|
533
|
+
f"{len(unbounded)} legacy unbounded approval(s) will be treated "
|
|
534
|
+
"as one-use/24-hour migration grants"
|
|
535
|
+
),
|
|
536
|
+
remedy=(
|
|
537
|
+
"replace queue standing approvals with durable bounded grants; "
|
|
538
|
+
"see docs/BOUNDED-APPROVAL-AUTHORITY.md"
|
|
539
|
+
),
|
|
540
|
+
)
|
|
541
|
+
return CheckResult(
|
|
542
|
+
"standing-approvals", PASS,
|
|
543
|
+
detail=f"{len(data)} bounded compatibility approval(s) parseable",
|
|
544
|
+
)
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def check_feature_gates() -> CheckResult:
|
|
548
|
+
"""7. Gate env values are true/false-ish; thinking needs the LLM router."""
|
|
549
|
+
problems: List[str] = []
|
|
550
|
+
notes: List[str] = []
|
|
551
|
+
for var in ("COLONY_ENABLE_INTERNAL_THINKING", "COLONY_ENABLE_SKILL_SYNTHESIS"):
|
|
552
|
+
raw = os.environ.get(var)
|
|
553
|
+
if raw is None or not raw.strip():
|
|
554
|
+
continue
|
|
555
|
+
value = raw.strip().lower()
|
|
556
|
+
if value == "true":
|
|
557
|
+
notes.append(f"{var}=true")
|
|
558
|
+
elif value == "false":
|
|
559
|
+
notes.append(f"{var}=false")
|
|
560
|
+
else:
|
|
561
|
+
problems.append(
|
|
562
|
+
f"{var}={raw!r} — only the literal 'true' enables it, so this value is "
|
|
563
|
+
"silently treated as false"
|
|
564
|
+
)
|
|
565
|
+
thinking_on = (
|
|
566
|
+
os.environ.get("COLONY_ENABLE_INTERNAL_THINKING", "false").strip().lower() == "true"
|
|
567
|
+
)
|
|
568
|
+
if thinking_on:
|
|
569
|
+
notes.append(
|
|
570
|
+
"internal thinking is enabled — it requires a working LLM router "
|
|
571
|
+
"(see the llm-config and server-llm-router checks)"
|
|
572
|
+
)
|
|
573
|
+
preset = os.environ.get("COLONY_AUTONOMY_PRESET", "").strip().lower()
|
|
574
|
+
if preset:
|
|
575
|
+
try:
|
|
576
|
+
from apsimo.util.autonomy_preset import PRESETS
|
|
577
|
+
if preset in PRESETS:
|
|
578
|
+
notes.append(f"COLONY_AUTONOMY_PRESET={preset}")
|
|
579
|
+
else:
|
|
580
|
+
problems.append(
|
|
581
|
+
f"COLONY_AUTONOMY_PRESET={preset!r} is not a known preset "
|
|
582
|
+
f"({'/'.join(sorted(PRESETS))}) — it is silently ignored")
|
|
583
|
+
except ImportError:
|
|
584
|
+
pass
|
|
585
|
+
if problems:
|
|
586
|
+
return CheckResult(
|
|
587
|
+
"feature-gates", WARN,
|
|
588
|
+
detail="; ".join(problems),
|
|
589
|
+
remedy="set the variable to exactly 'true' or 'false' (presets: "
|
|
590
|
+
"passive/calibration/autonomous)",
|
|
591
|
+
)
|
|
592
|
+
return CheckResult(
|
|
593
|
+
"feature-gates", PASS,
|
|
594
|
+
detail="; ".join(notes)
|
|
595
|
+
or "gates unset (features off; set COLONY_AUTONOMY_PRESET or "
|
|
596
|
+
"individual flags in this shell/service env)",
|
|
597
|
+
)
|
|
598
|
+
|
|
599
|
+
|
|
600
|
+
def check_tom2_cross_context() -> CheckResult:
|
|
601
|
+
"""Cross-contact tom2 rendering (H3.5) must never run ahead of chat
|
|
602
|
+
enforcement. Configuration intent is not applied capability: current
|
|
603
|
+
Hermes cannot mutate a post-LLM reply and the Colony plugin honestly
|
|
604
|
+
downgrades requested enforce to shadow. Until a transport-owned mediator
|
|
605
|
+
exposes real applied-enforcement truth, any enabled cross-context posture
|
|
606
|
+
is incoherent — 'X hasn't heard this' carries implication-leak risk only
|
|
607
|
+
an enforcing outbound guard can backstop."""
|
|
608
|
+
on = os.environ.get("COLONY_TOM2_CROSS_CONTEXT", "0").strip().lower()\
|
|
609
|
+
in ("1", "true", "yes", "on")
|
|
610
|
+
if not on:
|
|
611
|
+
return CheckResult(
|
|
612
|
+
"tom2-cross-context", PASS,
|
|
613
|
+
detail="COLONY_TOM2_CROSS_CONTEXT off (default; the "
|
|
614
|
+
"cross-contact render path ships dark)")
|
|
615
|
+
chat_mode = os.environ.get("COLONY_GUARD_CHAT_MODE", "").strip().lower()
|
|
616
|
+
if chat_mode != "enforce":
|
|
617
|
+
return CheckResult(
|
|
618
|
+
"tom2-cross-context", WARN,
|
|
619
|
+
detail="COLONY_TOM2_CROSS_CONTEXT is ON while the chat guard is "
|
|
620
|
+
f"not enforcing (COLONY_GUARD_CHAT_MODE={chat_mode or '(unset)'}) — "
|
|
621
|
+
"cross-contact epistemic rendering without an enforcing "
|
|
622
|
+
"outbound guard risks implication leaks",
|
|
623
|
+
remedy="set COLONY_TOM2_CROSS_CONTEXT=0 (recommended; this path "
|
|
624
|
+
"is built but deliberately unwired) or finish a "
|
|
625
|
+
"transport-owned applied-enforcement mediator; setting "
|
|
626
|
+
"COLONY_GUARD_CHAT_MODE=enforce alone does not qualify")
|
|
627
|
+
return CheckResult(
|
|
628
|
+
"tom2-cross-context", WARN,
|
|
629
|
+
detail="COLONY_TOM2_CROSS_CONTEXT is ON and chat enforce was "
|
|
630
|
+
"requested, but no applied-enforcement capability truth "
|
|
631
|
+
"exists: current Hermes cannot apply post-LLM reply "
|
|
632
|
+
"mutations, so the Colony plugin runs the chat guard in "
|
|
633
|
+
"SHADOW",
|
|
634
|
+
remedy="set COLONY_TOM2_CROSS_CONTEXT=0; do not treat "
|
|
635
|
+
"COLONY_GUARD_CHAT_MODE=enforce as applied enforcement. "
|
|
636
|
+
"Graduate only after a transport-owned mediator both applies "
|
|
637
|
+
"the verdict and exposes receipt-backed capability truth")
|
|
638
|
+
|
|
639
|
+
|
|
640
|
+
def check_tom2_risk_caps() -> CheckResult:
|
|
641
|
+
"""Leveled tom2 (L1.3): COLONY_TOM2_RISK_CAPS must parse. A malformed
|
|
642
|
+
value fails closed at runtime (all environments cap at level 0), which
|
|
643
|
+
is SAFE but silently ignores whatever caps the owner intended — a
|
|
644
|
+
posture mismatch the doctor must surface."""
|
|
645
|
+
from apsimo.tom.levels import DEFAULT_RISK_CAPS, risk_caps_valid
|
|
646
|
+
raw = os.environ.get("COLONY_TOM2_RISK_CAPS", "").strip()
|
|
647
|
+
if risk_caps_valid():
|
|
648
|
+
return CheckResult(
|
|
649
|
+
"tom2-risk-caps", PASS,
|
|
650
|
+
detail=f"COLONY_TOM2_RISK_CAPS={raw or f'(default {DEFAULT_RISK_CAPS})'}")
|
|
651
|
+
return CheckResult(
|
|
652
|
+
"tom2-risk-caps", WARN,
|
|
653
|
+
detail=f"COLONY_TOM2_RISK_CAPS={raw!r} is malformed — the level "
|
|
654
|
+
"resolver fails closed to all-0 caps (every environment "
|
|
655
|
+
"renders level 0), NOT the caps you configured",
|
|
656
|
+
remedy="set COLONY_TOM2_RISK_CAPS to four 'risk:cap' pairs covering "
|
|
657
|
+
f"risks 0-3 with caps 0-2, e.g. {DEFAULT_RISK_CAPS}, or unset "
|
|
658
|
+
"it for the default")
|
|
659
|
+
|
|
660
|
+
|
|
661
|
+
def check_tom2_level_coherence() -> CheckResult:
|
|
662
|
+
"""Leveled cross-contact tom2 (L4.3): a raised COLONY_TOM2_LEVEL must
|
|
663
|
+
be COHERENT — reachable under the other brakes and backed by live
|
|
664
|
+
enforcement evidence. The min-chain silently degrades an incoherent
|
|
665
|
+
posture every turn (which is SAFE); what the doctor surfaces is the
|
|
666
|
+
owner believing a level is live when it can never actually render."""
|
|
667
|
+
from apsimo.gate.response_guard import enforce_allowlist
|
|
668
|
+
from apsimo.tom.levels import (
|
|
669
|
+
configured_level, configured_max_level, risk_caps_valid)
|
|
670
|
+
from apsimo.tom.tom2 import tom2_cross_context_enabled
|
|
671
|
+
|
|
672
|
+
lvl = configured_level()
|
|
673
|
+
if lvl == 0:
|
|
674
|
+
return CheckResult(
|
|
675
|
+
"tom2-level-coherence", PASS,
|
|
676
|
+
detail="COLONY_TOM2_LEVEL=0 (default; leveled rendering off — "
|
|
677
|
+
"this variable is also the single-var kill switch)")
|
|
678
|
+
|
|
679
|
+
problems: List[str] = []
|
|
680
|
+
if not risk_caps_valid():
|
|
681
|
+
problems.append("COLONY_TOM2_RISK_CAPS is malformed — every "
|
|
682
|
+
"environment caps at level 0")
|
|
683
|
+
if lvl >= 2:
|
|
684
|
+
if configured_max_level() < 2:
|
|
685
|
+
problems.append("level 2 unreachable: COLONY_TOM2_MAX_LEVEL "
|
|
686
|
+
f"caps at {configured_max_level()}")
|
|
687
|
+
if not tom2_cross_context_enabled():
|
|
688
|
+
problems.append("level 2 unreachable: "
|
|
689
|
+
"COLONY_TOM2_CROSS_CONTEXT is off")
|
|
690
|
+
allowed = enforce_allowlist()
|
|
691
|
+
if allowed is not None and "tom2_epistemic" not in allowed:
|
|
692
|
+
problems.append("level 2 unreachable: tom2_epistemic is not on "
|
|
693
|
+
"COLONY_GUARD_ENFORCE_CHECKS, so enforce "
|
|
694
|
+
"evidence can never accrue")
|
|
695
|
+
else:
|
|
696
|
+
problems.extend(_tom2_enforce_evidence_problem())
|
|
697
|
+
if problems:
|
|
698
|
+
return CheckResult(
|
|
699
|
+
"tom2-level-coherence", WARN,
|
|
700
|
+
detail=f"COLONY_TOM2_LEVEL={lvl} but " + "; ".join(problems),
|
|
701
|
+
remedy="either lower COLONY_TOM2_LEVEL to the level you can "
|
|
702
|
+
"actually reach, or complete the graduation ladder "
|
|
703
|
+
"(docs/TOM2-LEVELS.md): MAX_LEVEL=2 + CROSS_CONTEXT=1 + "
|
|
704
|
+
"tom2_epistemic allowlisted + a receipt-backed egress "
|
|
705
|
+
"mediator proving the exact applied output")
|
|
706
|
+
return CheckResult(
|
|
707
|
+
"tom2-level-coherence", PASS,
|
|
708
|
+
detail=f"COLONY_TOM2_LEVEL={lvl} with a coherent brake posture")
|
|
709
|
+
|
|
710
|
+
|
|
711
|
+
def _tom2_enforce_evidence_problem() -> List[str]:
|
|
712
|
+
"""Report the intentionally missing receipt-backed egress authority.
|
|
713
|
+
|
|
714
|
+
GuardAuditStore rows attest candidate evaluation only. Row count,
|
|
715
|
+
recency, enforce mode, and decision cannot prove which bytes a transport
|
|
716
|
+
withheld or emitted, so they must never make the doctor claim level 2 is
|
|
717
|
+
reachable. The runtime resolver likewise leaves its evidence probe unset
|
|
718
|
+
until a transport-owned applied-output mediator exists.
|
|
719
|
+
"""
|
|
720
|
+
|
|
721
|
+
return [
|
|
722
|
+
"no receipt-backed applied-output egress mediator is wired; "
|
|
723
|
+
"GuardAuditStore evaluation rows never qualify as enforce evidence "
|
|
724
|
+
"and level 2 caps at 1"
|
|
725
|
+
]
|
|
726
|
+
|
|
727
|
+
|
|
728
|
+
def check_home_channel() -> CheckResult:
|
|
729
|
+
"""8. At least one *_HOME_CHANNEL so initiatives can be delivered."""
|
|
730
|
+
found = sorted(
|
|
731
|
+
key for key, value in os.environ.items()
|
|
732
|
+
if _HOME_CHANNEL_RE.match(key) and value.strip()
|
|
733
|
+
)
|
|
734
|
+
if found:
|
|
735
|
+
return CheckResult("home-channel", PASS, detail="configured: " + ", ".join(found))
|
|
736
|
+
return CheckResult(
|
|
737
|
+
"home-channel", WARN,
|
|
738
|
+
detail="no *_HOME_CHANNEL configured — initiatives will queue but never deliver",
|
|
739
|
+
remedy="set one of TELEGRAM_HOME_CHANNEL / WHATSAPP_HOME_CHANNEL / DISCORD_HOME_CHANNEL "
|
|
740
|
+
"/ SLACK_HOME_CHANNEL / SIGNAL_HOME_CHANNEL to the owner's chat id",
|
|
741
|
+
)
|
|
742
|
+
|
|
743
|
+
|
|
744
|
+
def check_hermes_skills_dir() -> CheckResult:
|
|
745
|
+
"""9. When skill export is on, the export base's parent must exist."""
|
|
746
|
+
from apsimo.skills.hermes_export import hermes_base_dir, hermes_export_enabled
|
|
747
|
+
|
|
748
|
+
if not hermes_export_enabled():
|
|
749
|
+
return CheckResult(
|
|
750
|
+
"hermes-skills-dir", SKIP, detail="COLONY_EMIT_HERMES_SKILLS not enabled",
|
|
751
|
+
)
|
|
752
|
+
base = hermes_base_dir()
|
|
753
|
+
parent = base.parent
|
|
754
|
+
if parent.exists():
|
|
755
|
+
return CheckResult(
|
|
756
|
+
"hermes-skills-dir", PASS, detail=f"export base {base} (parent {parent} exists)",
|
|
757
|
+
)
|
|
758
|
+
return CheckResult(
|
|
759
|
+
"hermes-skills-dir", WARN,
|
|
760
|
+
detail=f"COLONY_EMIT_HERMES_SKILLS is on but {parent} does not exist — skill exports "
|
|
761
|
+
"will fail",
|
|
762
|
+
remedy=f"mkdir -p {parent} (or point COLONY_HERMES_SKILLS_DIR at an existing skills tree)",
|
|
763
|
+
)
|
|
764
|
+
|
|
765
|
+
|
|
766
|
+
def check_relationship_attribution() -> CheckResult:
|
|
767
|
+
"""10. Attribution health: recent communications must land on real
|
|
768
|
+
contacts, not the default/system placeholders (docs/RELATIONSHIPS.md).
|
|
769
|
+
|
|
770
|
+
A live audit found 62% of all traffic filed under 'default', which
|
|
771
|
+
starves every relationship surface. Warn when the recent placeholder
|
|
772
|
+
fraction is high; skip when there is no comms ledger yet."""
|
|
773
|
+
comms = _state_dir() / "colony-comms.db"
|
|
774
|
+
if not comms.exists():
|
|
775
|
+
return CheckResult("relationship-attribution", SKIP,
|
|
776
|
+
detail="no comms ledger yet")
|
|
777
|
+
try:
|
|
778
|
+
conn = sqlite3.connect(f"file:{comms}?mode=ro", uri=True)
|
|
779
|
+
try:
|
|
780
|
+
total, placeholder = conn.execute(
|
|
781
|
+
"SELECT COUNT(*), "
|
|
782
|
+
"SUM(contact_id IN ('default','system') OR contact_id='') "
|
|
783
|
+
"FROM communications WHERE ts >= datetime('now','-7 day')"
|
|
784
|
+
).fetchone()
|
|
785
|
+
finally:
|
|
786
|
+
conn.close()
|
|
787
|
+
except sqlite3.Error as exc:
|
|
788
|
+
return CheckResult("relationship-attribution", WARN,
|
|
789
|
+
detail=f"comms ledger unreadable: {exc}")
|
|
790
|
+
total = int(total or 0)
|
|
791
|
+
placeholder = int(placeholder or 0)
|
|
792
|
+
if total == 0:
|
|
793
|
+
return CheckResult("relationship-attribution", PASS,
|
|
794
|
+
detail="no communications in the last 7 days")
|
|
795
|
+
frac = placeholder / total
|
|
796
|
+
# 'system' rows are expected for machine turns; only an outsized share
|
|
797
|
+
# (or any legacy 'default' writes) signals an attribution regression.
|
|
798
|
+
if frac > 0.5:
|
|
799
|
+
return CheckResult(
|
|
800
|
+
"relationship-attribution", WARN,
|
|
801
|
+
detail=f"{placeholder}/{total} recent communications attribute to "
|
|
802
|
+
"placeholder contacts — senders are not resolving",
|
|
803
|
+
remedy="ensure the host passes `sender` on turns/sync (the Hermes "
|
|
804
|
+
"provider does) and that senders' handles exist; unknown "
|
|
805
|
+
"senders should be creating shadow contacts")
|
|
806
|
+
return CheckResult(
|
|
807
|
+
"relationship-attribution", PASS,
|
|
808
|
+
detail=f"{total - placeholder}/{total} recent communications attribute "
|
|
809
|
+
"to real contacts")
|
|
810
|
+
|
|
811
|
+
|
|
812
|
+
def check_optional_video() -> CheckResult:
|
|
813
|
+
"""Report the optional local decoder, without installing or running it."""
|
|
814
|
+
try:
|
|
815
|
+
import av
|
|
816
|
+
except (ImportError, OSError):
|
|
817
|
+
return CheckResult('optional-video', SKIP, detail=
|
|
818
|
+
"Decoder unavailable in this interpreter. For selected clip memory install 'colonyai[video]' here; see docs/SOURCE-VIDEOS.md.")
|
|
819
|
+
return CheckResult('optional-video', PASS, detail=
|
|
820
|
+
f'PyAV {av.__version__} imports in this interpreter; no clip, model or camera was tested.')
|
|
821
|
+
|
|
822
|
+
|
|
823
|
+
def run_local_checks() -> List[CheckResult]:
|
|
824
|
+
results: List[CheckResult] = []
|
|
825
|
+
results += _run('optional-video', check_optional_video)
|
|
826
|
+
results += _run("state-dir", check_state_dir)
|
|
827
|
+
results += _run("llm-config", check_llm_config)
|
|
828
|
+
results += _run("contacts-db", check_contacts_db)
|
|
829
|
+
results += _run("owner-contact-id", check_owner_contact_id)
|
|
830
|
+
results += _run("approval-policy", check_approval_policy)
|
|
831
|
+
results += _run("standing-approvals", check_standing_approvals)
|
|
832
|
+
results += _run("feature-gates", check_feature_gates)
|
|
833
|
+
results += _run("tom2-cross-context", check_tom2_cross_context)
|
|
834
|
+
results += _run("tom2-risk-caps", check_tom2_risk_caps)
|
|
835
|
+
results += _run("tom2-level-coherence", check_tom2_level_coherence)
|
|
836
|
+
results += _run("home-channel", check_home_channel)
|
|
837
|
+
results += _run("hermes-skills-dir", check_hermes_skills_dir)
|
|
838
|
+
results += _run("relationship-attribution", check_relationship_attribution)
|
|
839
|
+
return results
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
# ---------------------------------------------------------------------------
|
|
843
|
+
# Server checks (HTTP against the running sidecar)
|
|
844
|
+
# ---------------------------------------------------------------------------
|
|
845
|
+
|
|
846
|
+
def check_server_auth(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
847
|
+
"""11. Auth round-trip: only a genuine 2xx success may PASS."""
|
|
848
|
+
status, _ = _http_get(f"{base_url}/v1/host/queue/stats", api_key, timeout)
|
|
849
|
+
if status in (401, 403):
|
|
850
|
+
return CheckResult(
|
|
851
|
+
"server-auth", FAIL,
|
|
852
|
+
detail=f"API key rejected ({status}) by /v1/host/queue/stats",
|
|
853
|
+
remedy="set COLONY_API_KEY (CLI side) to the key in the sidecar's environment "
|
|
854
|
+
"(~/.colony/.env), or pass --api-key. " + PLIST_ENV_REMEDY,
|
|
855
|
+
)
|
|
856
|
+
if not 200 <= status < 300:
|
|
857
|
+
return CheckResult(
|
|
858
|
+
"server-auth", WARN,
|
|
859
|
+
detail=f"auth round-trip inconclusive — /v1/host/queue/stats "
|
|
860
|
+
f"returned HTTP {status}, not a success",
|
|
861
|
+
remedy="check the sidecar log; an authenticated request to the "
|
|
862
|
+
"queue stats endpoint should return 200",
|
|
863
|
+
)
|
|
864
|
+
note = "" if api_key else " (no API key configured — sidecar is in dev mode)"
|
|
865
|
+
return CheckResult(
|
|
866
|
+
"server-auth", PASS,
|
|
867
|
+
detail=f"authenticated request accepted (HTTP {status}){note}",
|
|
868
|
+
)
|
|
869
|
+
|
|
870
|
+
|
|
871
|
+
def check_server_auth_migration(
|
|
872
|
+
base_url: str, api_key: str, timeout: float,
|
|
873
|
+
) -> CheckResult:
|
|
874
|
+
"""Scoped/legacy usage and exact-contact grant readiness."""
|
|
875
|
+
|
|
876
|
+
status, body = _http_get(
|
|
877
|
+
f"{base_url}/v1/host/admin/auth/status", api_key, timeout,
|
|
878
|
+
)
|
|
879
|
+
if status == 404:
|
|
880
|
+
return CheckResult(
|
|
881
|
+
"server-auth-migration", SKIP,
|
|
882
|
+
detail="auth migration status is not available on this sidecar",
|
|
883
|
+
)
|
|
884
|
+
if status in (401, 403):
|
|
885
|
+
return CheckResult(
|
|
886
|
+
"server-auth-migration", WARN,
|
|
887
|
+
detail=f"auth migration status requires auth:admin (HTTP {status})",
|
|
888
|
+
remedy="run doctor with the legacy migration credential or a scoped auth:admin principal",
|
|
889
|
+
)
|
|
890
|
+
if status != 200 or not isinstance(body, dict):
|
|
891
|
+
return CheckResult(
|
|
892
|
+
"server-auth-migration", WARN,
|
|
893
|
+
detail=f"unexpected auth migration status response (HTTP {status})",
|
|
894
|
+
)
|
|
895
|
+
|
|
896
|
+
auth = body.get("auth") if isinstance(body.get("auth"), dict) else {}
|
|
897
|
+
telemetry = body.get("telemetry") if isinstance(body.get("telemetry"), dict) else {}
|
|
898
|
+
keyring = body.get("keyring") if isinstance(body.get("keyring"), dict) else {}
|
|
899
|
+
grants = body.get("contact_grants") if isinstance(body.get("contact_grants"), dict) else {}
|
|
900
|
+
totals = telemetry.get("totals") if isinstance(telemetry.get("totals"), dict) else {}
|
|
901
|
+
principals = (
|
|
902
|
+
telemetry.get("principals")
|
|
903
|
+
if isinstance(telemetry.get("principals"), dict)
|
|
904
|
+
else {}
|
|
905
|
+
)
|
|
906
|
+
legacy = int(totals.get("legacy_allow") or 0)
|
|
907
|
+
scoped = int(totals.get("scoped_allow") or 0)
|
|
908
|
+
denied = int(totals.get("deny") or 0)
|
|
909
|
+
|
|
910
|
+
problems: list[str] = []
|
|
911
|
+
if not telemetry.get("persistent") or telemetry.get("error"):
|
|
912
|
+
problems.append("auth telemetry is not durably healthy")
|
|
913
|
+
if auth.get("scoped_configured") and (
|
|
914
|
+
not keyring.get("available") or keyring.get("error")
|
|
915
|
+
):
|
|
916
|
+
problems.append("scoped keyring is unavailable")
|
|
917
|
+
if grants.get("error"):
|
|
918
|
+
problems.append("exact-contact grant projection is unhealthy")
|
|
919
|
+
legacy_last_seen = str((principals.get("legacy") or {}).get("last_seen_at") or "")
|
|
920
|
+
if auth.get("dual_accept") and legacy:
|
|
921
|
+
try:
|
|
922
|
+
parsed = legacy_last_seen
|
|
923
|
+
if parsed.endswith("Z"):
|
|
924
|
+
parsed = parsed[:-1] + "+00:00"
|
|
925
|
+
last_seen = datetime.fromisoformat(parsed).astimezone(timezone.utc)
|
|
926
|
+
quiet_hours = max(
|
|
927
|
+
0.0, float(os.environ.get("COLONY_AUTH_LEGACY_QUIET_HOURS", "24")),
|
|
928
|
+
)
|
|
929
|
+
age_hours = max(
|
|
930
|
+
0.0,
|
|
931
|
+
(datetime.now(timezone.utc) - last_seen).total_seconds() / 3600,
|
|
932
|
+
)
|
|
933
|
+
if age_hours < quiet_hours:
|
|
934
|
+
problems.append(
|
|
935
|
+
f"legacy bearer was used {age_hours:.1f}h ago "
|
|
936
|
+
f"(quiet window {quiet_hours:g}h)"
|
|
937
|
+
)
|
|
938
|
+
except (TypeError, ValueError, OverflowError):
|
|
939
|
+
problems.append("legacy last-seen evidence is missing or invalid")
|
|
940
|
+
if auth.get("scoped_configured") and not scoped:
|
|
941
|
+
problems.append("no scoped principal traffic has been observed")
|
|
942
|
+
if auth.get("legacy_configured") and not auth.get("scoped_configured"):
|
|
943
|
+
problems.append("scoped principals are not configured")
|
|
944
|
+
|
|
945
|
+
detail = (
|
|
946
|
+
f"legacy_allow={legacy}, scoped_allow={scoped}, denied={denied}, "
|
|
947
|
+
f"exact_contacts={int(grants.get('total_exact_person_ids') or 0)}, "
|
|
948
|
+
f"legacy_last_seen={legacy_last_seen or 'never'}"
|
|
949
|
+
)
|
|
950
|
+
if problems:
|
|
951
|
+
return CheckResult(
|
|
952
|
+
"server-auth-migration", WARN,
|
|
953
|
+
detail=detail + "; " + "; ".join(problems),
|
|
954
|
+
remedy=(
|
|
955
|
+
"keep dual acceptance, migrate one role at a time, repair telemetry/grants, "
|
|
956
|
+
"and revoke the legacy bearer only after a complete zero-legacy window"
|
|
957
|
+
),
|
|
958
|
+
)
|
|
959
|
+
return CheckResult("server-auth-migration", PASS, detail=detail)
|
|
960
|
+
|
|
961
|
+
|
|
962
|
+
def check_server_owner_contact(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
963
|
+
"""12. The configured owner CID must resolve end-to-end."""
|
|
964
|
+
from apsimo.identity.resolver import get_owner_contact_id
|
|
965
|
+
|
|
966
|
+
owner = get_owner_contact_id()
|
|
967
|
+
if not owner:
|
|
968
|
+
return CheckResult(
|
|
969
|
+
"server-owner-contact", SKIP,
|
|
970
|
+
detail="COLONY_OWNER_CONTACT_ID not set (see owner-contact-id)",
|
|
971
|
+
)
|
|
972
|
+
if not owner.startswith("cid-"):
|
|
973
|
+
return CheckResult(
|
|
974
|
+
"server-owner-contact", SKIP,
|
|
975
|
+
detail=f"owner id {owner!r} is not a cid- — name/UUID resolution happens "
|
|
976
|
+
"in-process and cannot be verified over HTTP",
|
|
977
|
+
)
|
|
978
|
+
status, _ = _http_get(f"{base_url}/v1/host/contacts/{owner}", api_key, timeout)
|
|
979
|
+
if status == 200:
|
|
980
|
+
return CheckResult("server-owner-contact", PASS, detail=f"{owner} resolves (HTTP 200)")
|
|
981
|
+
if status == 401:
|
|
982
|
+
return CheckResult(
|
|
983
|
+
"server-owner-contact", SKIP, detail="auth failed — see server-auth",
|
|
984
|
+
)
|
|
985
|
+
if status == 404:
|
|
986
|
+
return CheckResult(
|
|
987
|
+
"server-owner-contact", FAIL,
|
|
988
|
+
detail=f"COLONY_OWNER_CONTACT_ID={owner} does not resolve to any contact (404) — "
|
|
989
|
+
"owner-aware subsystems fail closed (relationship + thinking degraded, "
|
|
990
|
+
"CRITICAL at autonomy loop start)",
|
|
991
|
+
remedy="create the owner contact via POST /v1/host/contacts and set "
|
|
992
|
+
"COLONY_OWNER_CONTACT_ID to the returned contact_id. " + PLIST_ENV_REMEDY,
|
|
993
|
+
)
|
|
994
|
+
return CheckResult(
|
|
995
|
+
"server-owner-contact", FAIL,
|
|
996
|
+
detail=f"unexpected HTTP {status} looking up {owner}",
|
|
997
|
+
)
|
|
998
|
+
|
|
999
|
+
|
|
1000
|
+
def check_server_llm(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1001
|
+
"""13. Live-fire the LLM router with one tiny completion.
|
|
1002
|
+
|
|
1003
|
+
A live completion inherently takes longer than the metadata checks: on a
|
|
1004
|
+
busy local model the answer can exceed the standard timeout while the
|
|
1005
|
+
router is perfectly healthy, so this check gets a higher floor rather
|
|
1006
|
+
than failing on load."""
|
|
1007
|
+
status, body = _http_get(f"{base_url}/v1/host/health/llm", api_key,
|
|
1008
|
+
max(timeout, 45.0))
|
|
1009
|
+
if status == 404:
|
|
1010
|
+
return CheckResult(
|
|
1011
|
+
"server-llm-router", SKIP,
|
|
1012
|
+
detail="/v1/host/health/llm not available (sidecar predates v0.19)",
|
|
1013
|
+
)
|
|
1014
|
+
if status != 200 or not isinstance(body, dict):
|
|
1015
|
+
return CheckResult(
|
|
1016
|
+
"server-llm-router", FAIL, detail=f"HTTP {status}: {body}",
|
|
1017
|
+
)
|
|
1018
|
+
if body.get("ok"):
|
|
1019
|
+
return CheckResult(
|
|
1020
|
+
"server-llm-router", PASS,
|
|
1021
|
+
detail=f"router answered (tier={body.get('tier')}, "
|
|
1022
|
+
f"latency={body.get('latency_ms')}ms)",
|
|
1023
|
+
)
|
|
1024
|
+
return CheckResult(
|
|
1025
|
+
"server-llm-router", FAIL,
|
|
1026
|
+
detail=f"LLM router live-fire failed: {body.get('error')}",
|
|
1027
|
+
remedy="the common causes are a baseUrl missing the /v1 suffix (LiteLLM 404s against "
|
|
1028
|
+
"vllm) and an empty apiKey (OPENAI_API_KEY never exported) — see the "
|
|
1029
|
+
"llm-config-baseurl / llm-config-apikey checks, fix "
|
|
1030
|
+
".colony-llm-config.json, and restart the sidecar. " + PLIST_ENV_REMEDY,
|
|
1031
|
+
)
|
|
1032
|
+
|
|
1033
|
+
|
|
1034
|
+
def check_server_embedder(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1035
|
+
"""14. Embedder health — the same health_check the startup path runs."""
|
|
1036
|
+
status, body = _http_get(f"{base_url}/v1/host/embed/health", api_key, timeout)
|
|
1037
|
+
if status in (404, 501):
|
|
1038
|
+
return CheckResult(
|
|
1039
|
+
"server-embedder", SKIP,
|
|
1040
|
+
detail=f"embedder health not exposed (HTTP {status})",
|
|
1041
|
+
)
|
|
1042
|
+
if status == 200 and isinstance(body, dict):
|
|
1043
|
+
if body.get("status") == "ok":
|
|
1044
|
+
extra = []
|
|
1045
|
+
if body.get("dims"):
|
|
1046
|
+
extra.append(f"dims={body['dims']}")
|
|
1047
|
+
if body.get("latency_ms") is not None:
|
|
1048
|
+
extra.append(f"latency={body['latency_ms']}ms")
|
|
1049
|
+
return CheckResult(
|
|
1050
|
+
"server-embedder", PASS,
|
|
1051
|
+
detail="embedder healthy" + (f" ({', '.join(extra)})" if extra else ""),
|
|
1052
|
+
)
|
|
1053
|
+
return CheckResult(
|
|
1054
|
+
"server-embedder", WARN,
|
|
1055
|
+
detail=f"embedder degraded: status={body.get('status')} "
|
|
1056
|
+
f"error={body.get('error') or 'n/a'} — memory recall falls back to "
|
|
1057
|
+
"keyword search",
|
|
1058
|
+
remedy="check the sidecar log for EmbeddingPipeline init errors; verify "
|
|
1059
|
+
"COLONY_EMBED_PROVIDER / COLONY_EMBED_MODEL and restart the sidecar",
|
|
1060
|
+
)
|
|
1061
|
+
return CheckResult("server-embedder", WARN, detail=f"unexpected HTTP {status}: {body}")
|
|
1062
|
+
|
|
1063
|
+
|
|
1064
|
+
def check_server_memory_graph(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1065
|
+
"""14b. Graph/memory backend reachability.
|
|
1066
|
+
|
|
1067
|
+
A dead Neo4j means every memory read/write fails while the API keeps
|
|
1068
|
+
answering — the one degradation the doctor previously never looked at
|
|
1069
|
+
(it reported ok:true with the graph completely down).
|
|
1070
|
+
"""
|
|
1071
|
+
status, body = _http_get(f"{base_url}/v1/host/memory/status", api_key, timeout)
|
|
1072
|
+
if status in (404, 501):
|
|
1073
|
+
return CheckResult(
|
|
1074
|
+
"server-memory-graph", SKIP,
|
|
1075
|
+
detail=f"memory status not exposed (HTTP {status})",
|
|
1076
|
+
)
|
|
1077
|
+
if status != 200 or not isinstance(body, dict):
|
|
1078
|
+
return CheckResult(
|
|
1079
|
+
"server-memory-graph", FAIL,
|
|
1080
|
+
detail=f"/v1/host/memory/status returned HTTP {status}: {body}",
|
|
1081
|
+
)
|
|
1082
|
+
if body.get("graph_wired") is False:
|
|
1083
|
+
return CheckResult(
|
|
1084
|
+
"server-memory-graph", WARN,
|
|
1085
|
+
detail="no graph backend wired — memory endpoints return stubs",
|
|
1086
|
+
remedy="configure the graph backend (Neo4j) if this deployment "
|
|
1087
|
+
"is supposed to have persistent memory",
|
|
1088
|
+
)
|
|
1089
|
+
if not body.get("neo4j_connected"):
|
|
1090
|
+
return CheckResult(
|
|
1091
|
+
"server-memory-graph", FAIL,
|
|
1092
|
+
detail="graph backend is wired but UNREACHABLE — every memory "
|
|
1093
|
+
"read/write is failing",
|
|
1094
|
+
remedy="start/repair Neo4j (or fix its credentials/URI), then "
|
|
1095
|
+
"re-run 'colony doctor'",
|
|
1096
|
+
)
|
|
1097
|
+
if not body.get("wired"):
|
|
1098
|
+
missing = [k for k in ("embeddings_ready", "vector_store_ready")
|
|
1099
|
+
if not body.get(k)]
|
|
1100
|
+
return CheckResult(
|
|
1101
|
+
"server-memory-graph", WARN,
|
|
1102
|
+
detail="graph reachable but memory pipeline incomplete: "
|
|
1103
|
+
+ (", ".join(missing) or "unknown component"),
|
|
1104
|
+
remedy="check embedder/vector-store wiring in the sidecar log",
|
|
1105
|
+
)
|
|
1106
|
+
return CheckResult(
|
|
1107
|
+
"server-memory-graph", PASS,
|
|
1108
|
+
detail="graph backend reachable; memory pipeline fully wired",
|
|
1109
|
+
)
|
|
1110
|
+
|
|
1111
|
+
|
|
1112
|
+
def check_server_blocked_approvals(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1113
|
+
"""15. Surface jobs stuck waiting for owner approval."""
|
|
1114
|
+
status, body = _http_get(f"{base_url}/v1/host/queue/jobs/blocked", api_key, timeout)
|
|
1115
|
+
if status == 501:
|
|
1116
|
+
return CheckResult(
|
|
1117
|
+
"server-blocked-approvals", SKIP, detail="task queue not wired (501)",
|
|
1118
|
+
)
|
|
1119
|
+
if status != 200 or not isinstance(body, list):
|
|
1120
|
+
return CheckResult(
|
|
1121
|
+
"server-blocked-approvals", FAIL, detail=f"HTTP {status}: {body}",
|
|
1122
|
+
)
|
|
1123
|
+
if not body:
|
|
1124
|
+
return CheckResult(
|
|
1125
|
+
"server-blocked-approvals", PASS, detail="no jobs blocked on owner approval",
|
|
1126
|
+
)
|
|
1127
|
+
hints = ", ".join(
|
|
1128
|
+
str(j.get("action_hint") or j.get("id")) for j in body[:5] if isinstance(j, dict)
|
|
1129
|
+
)
|
|
1130
|
+
return CheckResult(
|
|
1131
|
+
"server-blocked-approvals", WARN,
|
|
1132
|
+
detail=f"{len(body)} job(s) pending owner approval ({hints})",
|
|
1133
|
+
remedy="review them and POST /v1/host/queue/jobs/{id}/approve (or .../reject); "
|
|
1134
|
+
"use the request ID and action digest; an optional grant must have "
|
|
1135
|
+
"an expiry and use cap",
|
|
1136
|
+
)
|
|
1137
|
+
|
|
1138
|
+
|
|
1139
|
+
def check_server_worker_liveness(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1140
|
+
"""16. A queue worker must be claiming agent_action jobs.
|
|
1141
|
+
|
|
1142
|
+
Uses the existing authed ``/v1/host/queue/jobs/pending`` surface
|
|
1143
|
+
(QUEUED jobs come first and carry ``posted_at``) and computes the
|
|
1144
|
+
age client-side: any QUEUED agent_action job older than
|
|
1145
|
+
``WORKER_LIVENESS_THRESHOLD_MINUTES`` means nothing is claiming —
|
|
1146
|
+
the cron-driven ``colony-queue-worker`` is absent or broken.
|
|
1147
|
+
"""
|
|
1148
|
+
status, body = _http_get(
|
|
1149
|
+
f"{base_url}/v1/host/queue/jobs/pending?task_type=agent_action&limit=200",
|
|
1150
|
+
api_key, timeout,
|
|
1151
|
+
)
|
|
1152
|
+
if status in (404, 501, 503):
|
|
1153
|
+
return CheckResult(
|
|
1154
|
+
"server-worker-liveness", SKIP,
|
|
1155
|
+
detail=f"task queue not available (HTTP {status})",
|
|
1156
|
+
)
|
|
1157
|
+
if status != 200 or not isinstance(body, list):
|
|
1158
|
+
return CheckResult(
|
|
1159
|
+
"server-worker-liveness", FAIL, detail=f"HTTP {status}: {body}",
|
|
1160
|
+
)
|
|
1161
|
+
|
|
1162
|
+
now = datetime.now(timezone.utc)
|
|
1163
|
+
threshold = WORKER_LIVENESS_THRESHOLD_MINUTES
|
|
1164
|
+
stale: List[dict] = []
|
|
1165
|
+
queued = 0
|
|
1166
|
+
for job in body:
|
|
1167
|
+
if not isinstance(job, dict) or job.get("status") != "queued":
|
|
1168
|
+
continue
|
|
1169
|
+
queued += 1
|
|
1170
|
+
raw = job.get("posted_at")
|
|
1171
|
+
if not raw:
|
|
1172
|
+
continue
|
|
1173
|
+
try:
|
|
1174
|
+
posted = datetime.fromisoformat(str(raw).replace("Z", "+00:00"))
|
|
1175
|
+
except ValueError:
|
|
1176
|
+
continue
|
|
1177
|
+
if posted.tzinfo is None:
|
|
1178
|
+
posted = posted.replace(tzinfo=timezone.utc)
|
|
1179
|
+
if (now - posted).total_seconds() > threshold * 60:
|
|
1180
|
+
stale.append(job)
|
|
1181
|
+
|
|
1182
|
+
if stale:
|
|
1183
|
+
oldest_mins = max(
|
|
1184
|
+
(now - datetime.fromisoformat(str(j["posted_at"]).replace("Z", "+00:00"))
|
|
1185
|
+
).total_seconds() / 60
|
|
1186
|
+
for j in stale
|
|
1187
|
+
)
|
|
1188
|
+
hints = ", ".join(
|
|
1189
|
+
str((j.get("payload") or {}).get("action_hint") or j.get("job_id"))
|
|
1190
|
+
for j in stale[:5]
|
|
1191
|
+
)
|
|
1192
|
+
return CheckResult(
|
|
1193
|
+
"server-worker-liveness", WARN,
|
|
1194
|
+
detail=f"{len(stale)} QUEUED agent_action job(s) older than {threshold} minutes "
|
|
1195
|
+
f"(oldest {oldest_mins:.0f}m: {hints}) — queue worker appears absent, so "
|
|
1196
|
+
"approved jobs are never claimed or executed",
|
|
1197
|
+
remedy=WORKER_CRON_REMEDY,
|
|
1198
|
+
)
|
|
1199
|
+
if queued:
|
|
1200
|
+
return CheckResult(
|
|
1201
|
+
"server-worker-liveness", PASS,
|
|
1202
|
+
detail=f"{queued} QUEUED agent_action job(s), all younger than "
|
|
1203
|
+
f"{threshold} minutes",
|
|
1204
|
+
)
|
|
1205
|
+
return CheckResult(
|
|
1206
|
+
"server-worker-liveness", PASS,
|
|
1207
|
+
detail="no QUEUED agent_action jobs waiting on a worker",
|
|
1208
|
+
)
|
|
1209
|
+
|
|
1210
|
+
|
|
1211
|
+
def check_server_skills_observations(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1212
|
+
"""17. The agent's skill index must be reported and reasonably fresh."""
|
|
1213
|
+
status, body = _http_get(f"{base_url}/v1/host/observations/skills", api_key, timeout)
|
|
1214
|
+
if status == 501:
|
|
1215
|
+
return CheckResult(
|
|
1216
|
+
"server-skills-observations", SKIP, detail="observation store not wired (501)",
|
|
1217
|
+
)
|
|
1218
|
+
if status != 200 or not isinstance(body, dict):
|
|
1219
|
+
return CheckResult(
|
|
1220
|
+
"server-skills-observations", FAIL, detail=f"HTTP {status}: {body}",
|
|
1221
|
+
)
|
|
1222
|
+
observations = body.get("observations") or []
|
|
1223
|
+
if not observations:
|
|
1224
|
+
return CheckResult(
|
|
1225
|
+
"server-skills-observations", WARN,
|
|
1226
|
+
detail="no skill observations recorded — Colony does not know which skills the "
|
|
1227
|
+
"agent has installed",
|
|
1228
|
+
remedy="run the colony-skills-sync console command (installed with the pip "
|
|
1229
|
+
"package; the wizard schedules it daily) or enable the plugin skills "
|
|
1230
|
+
"sync so the agent reports its ~/.hermes/skills index",
|
|
1231
|
+
)
|
|
1232
|
+
newest: Optional[datetime] = None
|
|
1233
|
+
for obs in observations:
|
|
1234
|
+
raw = (obs or {}).get("observed_at")
|
|
1235
|
+
if not raw:
|
|
1236
|
+
continue
|
|
1237
|
+
try:
|
|
1238
|
+
ts = datetime.fromisoformat(str(raw).replace("Z", "+00:00"))
|
|
1239
|
+
except ValueError:
|
|
1240
|
+
continue
|
|
1241
|
+
if ts.tzinfo is None:
|
|
1242
|
+
ts = ts.replace(tzinfo=timezone.utc)
|
|
1243
|
+
if newest is None or ts > newest:
|
|
1244
|
+
newest = ts
|
|
1245
|
+
if newest is None:
|
|
1246
|
+
return CheckResult(
|
|
1247
|
+
"server-skills-observations", WARN,
|
|
1248
|
+
detail=f"{len(observations)} observation(s) but none carry a parseable observed_at",
|
|
1249
|
+
)
|
|
1250
|
+
age_days = (datetime.now(timezone.utc) - newest).total_seconds() / 86400
|
|
1251
|
+
if age_days > 7:
|
|
1252
|
+
return CheckResult(
|
|
1253
|
+
"server-skills-observations", WARN,
|
|
1254
|
+
detail=f"skill observations are stale — newest is {age_days:.1f} days old",
|
|
1255
|
+
remedy="run the colony-skills-sync console command (or re-enable the plugin "
|
|
1256
|
+
"skills sync / the wizard's daily cron entry)",
|
|
1257
|
+
)
|
|
1258
|
+
return CheckResult(
|
|
1259
|
+
"server-skills-observations", PASS,
|
|
1260
|
+
detail=f"{len(observations)} skill observation(s), newest {age_days:.1f} days old",
|
|
1261
|
+
)
|
|
1262
|
+
|
|
1263
|
+
|
|
1264
|
+
# ---------------------------------------------------------------------------
|
|
1265
|
+
# Cognition / autonomy checks (v0.22.0) — the seven-capability program gets
|
|
1266
|
+
# doctor visibility. All of these read the RUNNING server, not the local env,
|
|
1267
|
+
# so mode flags pinned in a service unit/plist are never invisible here.
|
|
1268
|
+
# ---------------------------------------------------------------------------
|
|
1269
|
+
|
|
1270
|
+
def check_server_grant_envelope(
|
|
1271
|
+
base_url: str,
|
|
1272
|
+
api_key: str,
|
|
1273
|
+
timeout: float,
|
|
1274
|
+
) -> CheckResult:
|
|
1275
|
+
"""Report the running process's effective approval-grant envelope."""
|
|
1276
|
+
|
|
1277
|
+
status, body = _http_get(
|
|
1278
|
+
f"{base_url}/v1/host/autonomy/posture", api_key, timeout,
|
|
1279
|
+
)
|
|
1280
|
+
if status == 404:
|
|
1281
|
+
return CheckResult(
|
|
1282
|
+
"server-grant-envelope", SKIP,
|
|
1283
|
+
detail="server predates grant-envelope posture reporting",
|
|
1284
|
+
)
|
|
1285
|
+
if status != 200 or not isinstance(body, dict) or not body.get("available"):
|
|
1286
|
+
return CheckResult(
|
|
1287
|
+
"server-grant-envelope", WARN,
|
|
1288
|
+
detail=f"grant envelope could not be verified (HTTP {status}: {body})",
|
|
1289
|
+
)
|
|
1290
|
+
posture = body.get("posture")
|
|
1291
|
+
envelope = posture.get("grant_envelope") if isinstance(posture, dict) else None
|
|
1292
|
+
if envelope is None:
|
|
1293
|
+
return CheckResult(
|
|
1294
|
+
"server-grant-envelope", SKIP,
|
|
1295
|
+
detail="running server does not report its effective grant envelope",
|
|
1296
|
+
)
|
|
1297
|
+
if not isinstance(envelope, dict):
|
|
1298
|
+
return CheckResult(
|
|
1299
|
+
"server-grant-envelope", FAIL,
|
|
1300
|
+
detail="running server reported an invalid grant envelope posture",
|
|
1301
|
+
)
|
|
1302
|
+
|
|
1303
|
+
ttl_state = envelope.get("max_ttl_state")
|
|
1304
|
+
uses_state = envelope.get("max_uses_state")
|
|
1305
|
+
ttl = envelope.get("max_ttl_seconds")
|
|
1306
|
+
uses = envelope.get("max_uses")
|
|
1307
|
+
standing = envelope.get("standing")
|
|
1308
|
+
active_standing = envelope.get("active_standing_grants")
|
|
1309
|
+
active_no_expiry = envelope.get("active_no_expiry_grants")
|
|
1310
|
+
active_no_use_cap = envelope.get("active_no_use_cap_grants")
|
|
1311
|
+
ttl_valid = (
|
|
1312
|
+
(ttl_state == "unbounded" and ttl is None)
|
|
1313
|
+
or (
|
|
1314
|
+
ttl_state == "bounded"
|
|
1315
|
+
and isinstance(ttl, int) and not isinstance(ttl, bool) and ttl >= 60
|
|
1316
|
+
)
|
|
1317
|
+
)
|
|
1318
|
+
uses_valid = (
|
|
1319
|
+
(uses_state == "unbounded" and uses is None)
|
|
1320
|
+
or (
|
|
1321
|
+
uses_state == "bounded"
|
|
1322
|
+
and isinstance(uses, int) and not isinstance(uses, bool) and uses >= 1
|
|
1323
|
+
)
|
|
1324
|
+
)
|
|
1325
|
+
expected_standing = "unbounded" in {ttl_state, uses_state}
|
|
1326
|
+
counts_valid = all(
|
|
1327
|
+
isinstance(value, int) and not isinstance(value, bool) and value >= 0
|
|
1328
|
+
for value in (active_standing, active_no_expiry, active_no_use_cap)
|
|
1329
|
+
)
|
|
1330
|
+
if (
|
|
1331
|
+
not ttl_valid
|
|
1332
|
+
or not uses_valid
|
|
1333
|
+
or not isinstance(standing, bool)
|
|
1334
|
+
or standing != expected_standing
|
|
1335
|
+
or envelope.get("sentinel") != "unlimited"
|
|
1336
|
+
or not counts_valid
|
|
1337
|
+
or active_no_expiry > active_standing
|
|
1338
|
+
or active_no_use_cap > active_standing
|
|
1339
|
+
or active_standing > active_no_expiry + active_no_use_cap
|
|
1340
|
+
):
|
|
1341
|
+
return CheckResult(
|
|
1342
|
+
"server-grant-envelope", FAIL,
|
|
1343
|
+
detail="running server reported an inconsistent grant envelope posture",
|
|
1344
|
+
remedy="check the sidecar log and restart with valid COLONY_GRANT_MAX_* settings",
|
|
1345
|
+
)
|
|
1346
|
+
|
|
1347
|
+
if expected_standing or active_standing:
|
|
1348
|
+
dimensions = []
|
|
1349
|
+
if ttl_state == "unbounded":
|
|
1350
|
+
dimensions.append("COLONY_GRANT_MAX_TTL_SECONDS=unlimited (no expiry)")
|
|
1351
|
+
if uses_state == "unbounded":
|
|
1352
|
+
dimensions.append("COLONY_GRANT_MAX_USES=unlimited (no use cap)")
|
|
1353
|
+
if active_standing:
|
|
1354
|
+
dimensions.append(
|
|
1355
|
+
f"{active_standing} active standing grant(s) remain "
|
|
1356
|
+
f"({active_no_expiry} no-expiry; "
|
|
1357
|
+
f"{active_no_use_cap} no-use-cap)"
|
|
1358
|
+
)
|
|
1359
|
+
return CheckResult(
|
|
1360
|
+
"server-grant-envelope", WARN,
|
|
1361
|
+
detail=(
|
|
1362
|
+
"STANDING AUTHORITY — " + "; ".join(dimensions)
|
|
1363
|
+
+ "; exact-scope standing dimensions persist until revoked"
|
|
1364
|
+
),
|
|
1365
|
+
remedy=(
|
|
1366
|
+
"keep revocation and receipt monitoring operational; setting finite "
|
|
1367
|
+
"limits affects new grants, so revoke any active standing grants "
|
|
1368
|
+
"that should no longer persist"
|
|
1369
|
+
),
|
|
1370
|
+
)
|
|
1371
|
+
return CheckResult(
|
|
1372
|
+
"server-grant-envelope", PASS,
|
|
1373
|
+
detail=(
|
|
1374
|
+
"bounded grant envelope: "
|
|
1375
|
+
f"COLONY_GRANT_MAX_TTL_SECONDS={ttl}, "
|
|
1376
|
+
f"COLONY_GRANT_MAX_USES={uses}"
|
|
1377
|
+
),
|
|
1378
|
+
)
|
|
1379
|
+
|
|
1380
|
+
def check_server_autonomy_posture(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1381
|
+
"""18. The effective autonomy posture as the running process resolves it."""
|
|
1382
|
+
status, body = _http_get(f"{base_url}/v1/host/autonomy/posture", api_key, timeout)
|
|
1383
|
+
if status == 404:
|
|
1384
|
+
return CheckResult(
|
|
1385
|
+
"server-autonomy-posture", SKIP,
|
|
1386
|
+
detail="server predates the posture endpoint (upgrade to >=0.22)")
|
|
1387
|
+
if status != 200 or not isinstance(body, dict) or not body.get("available"):
|
|
1388
|
+
return CheckResult(
|
|
1389
|
+
"server-autonomy-posture", WARN, detail=f"HTTP {status}: {body}")
|
|
1390
|
+
posture = body.get("posture") or {}
|
|
1391
|
+
preset = posture.get("preset", "(none)")
|
|
1392
|
+
|
|
1393
|
+
# Posture coherence: the presets exist to make subsystems run on loop
|
|
1394
|
+
# ticks; a reactive loop never ticks, so a calibration/autonomous preset
|
|
1395
|
+
# with a reactive loop is a Colony that never thinks, calibrates, or
|
|
1396
|
+
# earns trust — the worst kind of misconfiguration because everything
|
|
1397
|
+
# LOOKS enabled. Since preset-loop coupling (default on), the preset
|
|
1398
|
+
# supplies the mode itself, so a reactive loop under such a preset can
|
|
1399
|
+
# only mean an explicit COLONY_AUTONOMY_MODE=reactive pin, coupling
|
|
1400
|
+
# switched off, or an older server without coupling — those FAIL; a
|
|
1401
|
+
# reactive loop the coupling machinery failed to raise only WARNs
|
|
1402
|
+
# (fail-safe worked as designed). Older servers that do not report the
|
|
1403
|
+
# loop mode are not failed on a missing key.
|
|
1404
|
+
autonomy_mode = str(posture.get("COLONY_AUTONOMY_MODE", "") or "").lower()
|
|
1405
|
+
mode_source = str(posture.get("COLONY_AUTONOMY_MODE_SOURCE", "") or "").lower()
|
|
1406
|
+
coupling = str(posture.get("COLONY_PRESET_LOOP_COUPLING", "") or "").lower()
|
|
1407
|
+
if preset in ("calibration", "autonomous") and autonomy_mode == "reactive":
|
|
1408
|
+
base = (f"preset={preset} but the autonomy loop is REACTIVE — "
|
|
1409
|
+
"preset-enabled subsystems only run on loop ticks, so "
|
|
1410
|
+
"nothing thinks, calibrates, or earns trust on its own")
|
|
1411
|
+
if mode_source == "env":
|
|
1412
|
+
return CheckResult(
|
|
1413
|
+
"server-autonomy-posture", FAIL,
|
|
1414
|
+
detail=base + " (COLONY_AUTONOMY_MODE=reactive is explicitly "
|
|
1415
|
+
"pinned below the preset)",
|
|
1416
|
+
remedy="unset COLONY_AUTONOMY_MODE and restart (the preset "
|
|
1417
|
+
"supplies proactive via COLONY_PRESET_LOOP_COUPLING), "
|
|
1418
|
+
"or leave the pin if the rollback is deliberate — "
|
|
1419
|
+
"then this FAIL is the reminder")
|
|
1420
|
+
if coupling == "off":
|
|
1421
|
+
return CheckResult(
|
|
1422
|
+
"server-autonomy-posture", FAIL,
|
|
1423
|
+
detail=base + " (COLONY_PRESET_LOOP_COUPLING=off disabled "
|
|
1424
|
+
"the preset's mode default)",
|
|
1425
|
+
remedy="unset COLONY_PRESET_LOOP_COUPLING (it defaults on) "
|
|
1426
|
+
"or set COLONY_AUTONOMY_MODE=proactive and restart")
|
|
1427
|
+
if not mode_source:
|
|
1428
|
+
# Older server: no coupling, no source reporting — the original
|
|
1429
|
+
# misconfiguration, still a FAIL.
|
|
1430
|
+
return CheckResult(
|
|
1431
|
+
"server-autonomy-posture", FAIL, detail=base,
|
|
1432
|
+
remedy="set COLONY_AUTONOMY_MODE=proactive and restart (or "
|
|
1433
|
+
"upgrade to a server with preset-loop coupling)")
|
|
1434
|
+
# Coupling on, mode not pinned, yet still reactive: the coupling
|
|
1435
|
+
# machinery failed safe toward reactive. Honest but degraded.
|
|
1436
|
+
return CheckResult(
|
|
1437
|
+
"server-autonomy-posture", WARN,
|
|
1438
|
+
detail=base + f" (mode_source={mode_source}; coupling is on but "
|
|
1439
|
+
"did not resolve — it fails toward reactive by design)",
|
|
1440
|
+
remedy="set COLONY_AUTONOMY_MODE=proactive explicitly and "
|
|
1441
|
+
"check server logs for preset resolution errors")
|
|
1442
|
+
|
|
1443
|
+
live = sorted(k.replace("COLONY_", "").replace("_MODE", "").lower()
|
|
1444
|
+
for k, v in posture.items() if v == "live")
|
|
1445
|
+
shadow = sorted(k.replace("COLONY_", "").replace("_MODE", "").lower()
|
|
1446
|
+
for k, v in posture.items() if v in ("shadow", "dry_run"))
|
|
1447
|
+
on = sorted(k.replace("COLONY_", "").replace("_ENABLED", "").lower()
|
|
1448
|
+
for k, v in posture.items() if v in ("true", "on"))
|
|
1449
|
+
parts = [f"preset={preset}"]
|
|
1450
|
+
if on:
|
|
1451
|
+
parts.append("on: " + ",".join(on))
|
|
1452
|
+
if live:
|
|
1453
|
+
parts.append("live: " + ",".join(live))
|
|
1454
|
+
if shadow:
|
|
1455
|
+
# Under the calibration preset, shadow IS the design: subsystems are
|
|
1456
|
+
# trust-gated and graduate via the trust engine. Label it expected
|
|
1457
|
+
# so an owner reading the doctor does not "fix" a healthy posture.
|
|
1458
|
+
label = ("calibrating (expected: trust-gated shadow under the "
|
|
1459
|
+
"calibration preset): "
|
|
1460
|
+
if preset == "calibration" else "calibrating: ")
|
|
1461
|
+
parts.append(label + ",".join(shadow))
|
|
1462
|
+
if not (on or live or shadow):
|
|
1463
|
+
return CheckResult(
|
|
1464
|
+
"server-autonomy-posture", WARN,
|
|
1465
|
+
detail="everything is off — this Colony observes but never thinks or acts",
|
|
1466
|
+
remedy="set COLONY_AUTONOMY_PRESET=calibration (shadow everything, earn "
|
|
1467
|
+
"autonomy via the trust engine) or flip individual COLONY_*_MODE flags")
|
|
1468
|
+
|
|
1469
|
+
# Flags explicitly overridden BELOW the preset's default (env always
|
|
1470
|
+
# wins over the preset, so this only happens by explicit override).
|
|
1471
|
+
downgraded = _preset_downgrades(preset, posture)
|
|
1472
|
+
if downgraded:
|
|
1473
|
+
return CheckResult(
|
|
1474
|
+
"server-autonomy-posture", WARN,
|
|
1475
|
+
detail="; ".join(parts) + " — below preset default (explicit env "
|
|
1476
|
+
"override): " + ",".join(downgraded),
|
|
1477
|
+
remedy="unset the overriding COLONY_* env var(s) to inherit the "
|
|
1478
|
+
"preset default, or leave them if the downgrade is "
|
|
1479
|
+
"deliberate")
|
|
1480
|
+
return CheckResult("server-autonomy-posture", PASS, detail="; ".join(parts))
|
|
1481
|
+
|
|
1482
|
+
|
|
1483
|
+
def _preset_downgrades(preset: str, posture: dict) -> List[str]:
|
|
1484
|
+
"""Flags whose effective value sits below the active preset's default."""
|
|
1485
|
+
try:
|
|
1486
|
+
from apsimo.util.autonomy_preset import PRESETS
|
|
1487
|
+
except Exception:
|
|
1488
|
+
return []
|
|
1489
|
+
# "on" ranks with shadow/dry_run: for a binary flag (expectations)
|
|
1490
|
+
# anything but "off" is fully enabled, so only an explicit "off"
|
|
1491
|
+
# counts as a downgrade from an "on" preset default.
|
|
1492
|
+
rank = {"off": 0, "false": 0, "shadow": 1, "dry_run": 1, "on": 1,
|
|
1493
|
+
"true": 2, "live": 2}
|
|
1494
|
+
out = []
|
|
1495
|
+
for flag, want in PRESETS.get(preset, {}).items():
|
|
1496
|
+
have = str(posture.get(flag, "") or "").lower()
|
|
1497
|
+
if have and rank.get(have, 99) < rank.get(want, 0):
|
|
1498
|
+
out.append(
|
|
1499
|
+
flag.replace("COLONY_", "").replace("_MODE", "").lower()
|
|
1500
|
+
+ f"={have}")
|
|
1501
|
+
return sorted(out)
|
|
1502
|
+
|
|
1503
|
+
|
|
1504
|
+
def check_server_self_model(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1505
|
+
"""19. Self-model/trust engine: wired, and no tripped circuit breakers."""
|
|
1506
|
+
status, body = _http_get(f"{base_url}/v1/host/self", api_key, timeout)
|
|
1507
|
+
if status == 404:
|
|
1508
|
+
return CheckResult("server-self-model", SKIP, detail="endpoint absent (older server)")
|
|
1509
|
+
if status != 200 or not isinstance(body, dict):
|
|
1510
|
+
return CheckResult("server-self-model", FAIL, detail=f"HTTP {status}: {body}")
|
|
1511
|
+
if not body.get("available"):
|
|
1512
|
+
return CheckResult(
|
|
1513
|
+
"server-self-model", WARN,
|
|
1514
|
+
detail="self-model not wired — no earned-autonomy gating, no action journal",
|
|
1515
|
+
remedy="unset COLONY_SELF_MODEL_ENABLED=false (it defaults on) and restart")
|
|
1516
|
+
err = _reported_error("server-self-model", body)
|
|
1517
|
+
if err:
|
|
1518
|
+
return err
|
|
1519
|
+
domains = body.get("domains") or []
|
|
1520
|
+
trust = body.get("trust") or []
|
|
1521
|
+
demoted = [t["domain"] for t in trust if t.get("demotions", 0) > 0
|
|
1522
|
+
and t.get("stage") != "act_first"]
|
|
1523
|
+
if demoted:
|
|
1524
|
+
return CheckResult(
|
|
1525
|
+
"server-self-model", WARN,
|
|
1526
|
+
detail=f"{len(domains)} competence domain(s); circuit breaker has demoted: "
|
|
1527
|
+
+ ", ".join(demoted),
|
|
1528
|
+
remedy="review the action journal (GET /v1/host/self/journal) for the "
|
|
1529
|
+
"failures that tripped the breaker; the class re-graduates on a "
|
|
1530
|
+
"clean track record")
|
|
1531
|
+
return CheckResult(
|
|
1532
|
+
"server-self-model", PASS,
|
|
1533
|
+
detail=f"{len(domains)} competence domain(s), {len(trust)} trust stage(s)")
|
|
1534
|
+
|
|
1535
|
+
|
|
1536
|
+
def check_server_fd_limit(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1537
|
+
"""The SIDECAR's own open-file soft limit (from /health notes). LanceDB
|
|
1538
|
+
opens many files under load; a low limit (macOS default 256) makes vector
|
|
1539
|
+
recall fail with 'LanceError(IO): Too many open files' and fall back to the
|
|
1540
|
+
slow keyword path. Reads the sidecar process's real limit, not the shell's."""
|
|
1541
|
+
status, body = _http_get(f"{base_url}/v1/host/health", api_key, timeout)
|
|
1542
|
+
if status != 200 or not isinstance(body, dict):
|
|
1543
|
+
return CheckResult("server-fd-limit", SKIP, detail=f"HTTP {status}")
|
|
1544
|
+
notes = body.get("notes") or {}
|
|
1545
|
+
raw = notes.get("fd_limit")
|
|
1546
|
+
if raw is None:
|
|
1547
|
+
return CheckResult("server-fd-limit", SKIP, detail="not reported (older server)")
|
|
1548
|
+
if raw == "unlimited":
|
|
1549
|
+
return CheckResult("server-fd-limit", PASS, detail="open-file limit unlimited")
|
|
1550
|
+
try:
|
|
1551
|
+
soft = int(raw)
|
|
1552
|
+
except (TypeError, ValueError):
|
|
1553
|
+
return CheckResult("server-fd-limit", SKIP, detail=f"unparseable: {raw}")
|
|
1554
|
+
if soft < 1024:
|
|
1555
|
+
return CheckResult(
|
|
1556
|
+
"server-fd-limit", WARN,
|
|
1557
|
+
detail=f"sidecar open-file limit is {soft} (< 1024); LanceDB vector "
|
|
1558
|
+
"recall will fail under load and fall back to slow keyword search",
|
|
1559
|
+
remedy="raise the sidecar's limit. launchd (macOS): add "
|
|
1560
|
+
"SoftResourceLimits/HardResourceLimits {NumberOfFiles: 16384} "
|
|
1561
|
+
"to the plist and bootout/bootstrap. systemd: LimitNOFILE=16384. "
|
|
1562
|
+
"shell: ulimit -n 16384 before starting.")
|
|
1563
|
+
return CheckResult("server-fd-limit", PASS,
|
|
1564
|
+
detail=f"sidecar open-file limit {soft}")
|
|
1565
|
+
|
|
1566
|
+
|
|
1567
|
+
def check_server_benchmark(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1568
|
+
"""Selfhood benchmark: wired and producing weekly rollups."""
|
|
1569
|
+
status, body = _http_get(f"{base_url}/v1/host/self/benchmark", api_key, timeout)
|
|
1570
|
+
if status == 404:
|
|
1571
|
+
return CheckResult("server-benchmark", SKIP, detail="endpoint absent (older server)")
|
|
1572
|
+
if status != 200 or not isinstance(body, dict):
|
|
1573
|
+
return CheckResult("server-benchmark", FAIL, detail=f"HTTP {status}: {body}")
|
|
1574
|
+
if not body.get("available"):
|
|
1575
|
+
return CheckResult(
|
|
1576
|
+
"server-benchmark", WARN,
|
|
1577
|
+
detail="benchmark not wired — self-improvement has no falsifiable trend line",
|
|
1578
|
+
remedy="unset COLONY_BENCHMARK_ENABLED=false (it defaults on) and restart")
|
|
1579
|
+
weeks = body.get("weeks") or []
|
|
1580
|
+
if not weeks:
|
|
1581
|
+
return CheckResult(
|
|
1582
|
+
"server-benchmark", PASS,
|
|
1583
|
+
detail="wired; no rollups yet (first weekly compute pending)")
|
|
1584
|
+
trends = body.get("trends") or {}
|
|
1585
|
+
falling = [m for m, d in trends.items()
|
|
1586
|
+
if isinstance(d, (int, float)) and d < -0.15
|
|
1587
|
+
and not m.startswith("latency.")]
|
|
1588
|
+
if falling:
|
|
1589
|
+
return CheckResult(
|
|
1590
|
+
"server-benchmark", WARN,
|
|
1591
|
+
detail=f"{len(weeks)} week(s) of rollups; regressing week-over-week: "
|
|
1592
|
+
+ ", ".join(sorted(falling)),
|
|
1593
|
+
remedy="run `colony benchmark` and review the regressing metrics' detail")
|
|
1594
|
+
return CheckResult(
|
|
1595
|
+
"server-benchmark", PASS,
|
|
1596
|
+
detail=f"{len(weeks)} week(s) of rollups, latest {body.get('latest')}")
|
|
1597
|
+
|
|
1598
|
+
|
|
1599
|
+
def check_server_expectations(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1600
|
+
"""Expectation engine (Mind M3a): wired; calibration present once resolved."""
|
|
1601
|
+
status, body = _http_get(f"{base_url}/v1/host/self/expectations", api_key, timeout)
|
|
1602
|
+
if status == 404:
|
|
1603
|
+
return CheckResult("server-expectations", SKIP, detail="endpoint absent (older server)")
|
|
1604
|
+
if status != 200 or not isinstance(body, dict):
|
|
1605
|
+
return CheckResult("server-expectations", FAIL, detail=f"HTTP {status}: {body}")
|
|
1606
|
+
if not body.get("available"):
|
|
1607
|
+
return CheckResult(
|
|
1608
|
+
"server-expectations", SKIP,
|
|
1609
|
+
detail="expectations off (COLONY_EXPECTATIONS=off) — no calibration signal")
|
|
1610
|
+
pending = body.get("pending") or []
|
|
1611
|
+
cal = body.get("calibration") or {}
|
|
1612
|
+
if not cal:
|
|
1613
|
+
return CheckResult(
|
|
1614
|
+
"server-expectations", PASS,
|
|
1615
|
+
detail=f"mode={body.get('mode')}, {len(pending)} open prediction(s), "
|
|
1616
|
+
"no resolved calibration yet")
|
|
1617
|
+
worst = max(cal.values(), key=lambda r: r.get("brier", 0)) if cal else {}
|
|
1618
|
+
return CheckResult(
|
|
1619
|
+
"server-expectations", PASS,
|
|
1620
|
+
detail=f"mode={body.get('mode')}, {len(pending)} open, "
|
|
1621
|
+
f"{len(cal)} calibrated domain(s), worst brier {worst.get('brier')}")
|
|
1622
|
+
|
|
1623
|
+
|
|
1624
|
+
def check_server_workspace(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1625
|
+
"""Cognitive workspace (Mind M2): wired; concern count within capacity."""
|
|
1626
|
+
status, body = _http_get(f"{base_url}/v1/host/self/workspace", api_key, timeout)
|
|
1627
|
+
if status == 404:
|
|
1628
|
+
return CheckResult("server-workspace", SKIP, detail="endpoint absent (older server)")
|
|
1629
|
+
if status != 200 or not isinstance(body, dict):
|
|
1630
|
+
return CheckResult("server-workspace", FAIL, detail=f"HTTP {status}: {body}")
|
|
1631
|
+
if not body.get("available"):
|
|
1632
|
+
return CheckResult(
|
|
1633
|
+
"server-workspace", SKIP,
|
|
1634
|
+
detail="workspace off (COLONY_WORKSPACE=off) — no continuous thought")
|
|
1635
|
+
concerns = body.get("concerns") or []
|
|
1636
|
+
cap = body.get("capacity", 24)
|
|
1637
|
+
sleeping = " (sleep window active)" if body.get("sleeping") else ""
|
|
1638
|
+
if len(concerns) > cap:
|
|
1639
|
+
return CheckResult(
|
|
1640
|
+
"server-workspace", WARN,
|
|
1641
|
+
detail=f"{len(concerns)} concerns over capacity {cap} — decay/evict not keeping up")
|
|
1642
|
+
return CheckResult(
|
|
1643
|
+
"server-workspace", PASS,
|
|
1644
|
+
detail=f"mode={body.get('mode')}, {len(concerns)}/{cap} on her mind{sleeping}")
|
|
1645
|
+
|
|
1646
|
+
|
|
1647
|
+
def check_server_toolsmith(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1648
|
+
"""Toolsmith (Mind M1): wired, and no tool is stuck failing."""
|
|
1649
|
+
status, body = _http_get(f"{base_url}/v1/host/self/tools", api_key, timeout)
|
|
1650
|
+
if status == 404:
|
|
1651
|
+
return CheckResult("server-toolsmith", SKIP, detail="endpoint absent (older server)")
|
|
1652
|
+
if status != 200 or not isinstance(body, dict):
|
|
1653
|
+
return CheckResult("server-toolsmith", FAIL, detail=f"HTTP {status}: {body}")
|
|
1654
|
+
if not body.get("available"):
|
|
1655
|
+
return CheckResult(
|
|
1656
|
+
"server-toolsmith", SKIP,
|
|
1657
|
+
detail="toolsmith off (COLONY_TOOLSMITH=off) — no self-built tools")
|
|
1658
|
+
tools = body.get("tools") or []
|
|
1659
|
+
live = [t for t in tools if t.get("status") == "live"]
|
|
1660
|
+
shadow = [t for t in tools if t.get("status") == "shadow"]
|
|
1661
|
+
failing = [t.get("name") for t in tools
|
|
1662
|
+
if t.get("status") in ("shadow", "live")
|
|
1663
|
+
and (t.get("failures") or 0) >= 3]
|
|
1664
|
+
unreceipted_live = [
|
|
1665
|
+
t.get("name") for t in live
|
|
1666
|
+
if not (t.get("graduation_receipts") or 0)
|
|
1667
|
+
]
|
|
1668
|
+
if failing:
|
|
1669
|
+
return CheckResult(
|
|
1670
|
+
"server-toolsmith", WARN,
|
|
1671
|
+
detail=f"{len(live)} live, {len(shadow)} shadow; failing: "
|
|
1672
|
+
+ ", ".join(str(f) for f in failing),
|
|
1673
|
+
remedy="retire the failing tool(s) via POST /v1/host/self/tools/{id}/retire")
|
|
1674
|
+
if unreceipted_live:
|
|
1675
|
+
return CheckResult(
|
|
1676
|
+
"server-toolsmith", WARN,
|
|
1677
|
+
detail=(f"{len(unreceipted_live)} live tool(s) predate bounded "
|
|
1678
|
+
"graduation receipts: "
|
|
1679
|
+
+ ", ".join(str(name) for name in unreceipted_live)),
|
|
1680
|
+
remedy=("retire, compare, and re-graduate each legacy tool under "
|
|
1681
|
+
"the P5 canary runbook"),
|
|
1682
|
+
)
|
|
1683
|
+
return CheckResult(
|
|
1684
|
+
"server-toolsmith", PASS,
|
|
1685
|
+
detail=f"mode={body.get('mode')}, trust={body.get('trust_stage')}, "
|
|
1686
|
+
f"{len(live)} live / {len(shadow)} shadow tool(s)")
|
|
1687
|
+
|
|
1688
|
+
|
|
1689
|
+
def check_server_adaptive_params(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1690
|
+
"""20. Meta-learning knobs: present, within bounds, attribution visible."""
|
|
1691
|
+
status, body = _http_get(f"{base_url}/v1/host/self/params", api_key, timeout)
|
|
1692
|
+
if status == 404:
|
|
1693
|
+
return CheckResult("server-adaptive-params", SKIP, detail="endpoint absent (older server)")
|
|
1694
|
+
if status != 200 or not isinstance(body, dict) or not body.get("available"):
|
|
1695
|
+
return CheckResult(
|
|
1696
|
+
"server-adaptive-params", WARN,
|
|
1697
|
+
detail=f"adaptive param store not wired (HTTP {status})",
|
|
1698
|
+
remedy="check boot logs for 'AdaptiveParamStore init failed'")
|
|
1699
|
+
params = body.get("params") or []
|
|
1700
|
+
tuned = [p for p in params if p.get("value") is not None]
|
|
1701
|
+
detail = f"{len(params)} knob(s) registered"
|
|
1702
|
+
if tuned:
|
|
1703
|
+
detail += "; tuned: " + ", ".join(
|
|
1704
|
+
f"{p['name']}={p['effective']}" for p in tuned)
|
|
1705
|
+
return CheckResult("server-adaptive-params", PASS, detail=detail)
|
|
1706
|
+
|
|
1707
|
+
|
|
1708
|
+
def check_server_executor(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1709
|
+
"""21. Initiative executor: the acting brain is wired and cycling."""
|
|
1710
|
+
status, body = _http_get(f"{base_url}/v1/host/executor/status", api_key, timeout)
|
|
1711
|
+
if status == 404:
|
|
1712
|
+
return CheckResult("server-executor", SKIP, detail="endpoint absent (older server)")
|
|
1713
|
+
if status != 200 or not isinstance(body, dict):
|
|
1714
|
+
return CheckResult("server-executor", FAIL, detail=f"HTTP {status}: {body}")
|
|
1715
|
+
if not body.get("wired"):
|
|
1716
|
+
return CheckResult(
|
|
1717
|
+
"server-executor", WARN,
|
|
1718
|
+
detail="initiative executor not wired — initiatives are generated but "
|
|
1719
|
+
"nothing acts on them with tools",
|
|
1720
|
+
remedy="set COLONY_EXECUTOR_ENABLED=true (or COLONY_AUTONOMY_PRESET="
|
|
1721
|
+
"calibration) and restart")
|
|
1722
|
+
if not body.get("running"):
|
|
1723
|
+
return CheckResult(
|
|
1724
|
+
"server-executor", WARN,
|
|
1725
|
+
detail="executor wired but not running",
|
|
1726
|
+
remedy="check the sidecar log for executor startup errors")
|
|
1727
|
+
stats = body.get("stats") or {}
|
|
1728
|
+
return CheckResult(
|
|
1729
|
+
"server-executor", PASS,
|
|
1730
|
+
detail=f"running (cycles={stats.get('cycles', 0)}, "
|
|
1731
|
+
f"completed={stats.get('initiatives_completed', 0)}, "
|
|
1732
|
+
f"failed={stats.get('initiatives_failed', 0)})")
|
|
1733
|
+
|
|
1734
|
+
|
|
1735
|
+
def check_server_projects(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1736
|
+
"""22. Goal persistence: project engine mode + blocked projects."""
|
|
1737
|
+
status, body = _http_get(f"{base_url}/v1/host/projects", api_key, timeout)
|
|
1738
|
+
if status == 404:
|
|
1739
|
+
return CheckResult("server-projects", SKIP, detail="endpoint absent (older server)")
|
|
1740
|
+
if status != 200 or not isinstance(body, dict):
|
|
1741
|
+
return CheckResult("server-projects", FAIL, detail=f"HTTP {status}: {body}")
|
|
1742
|
+
if not body.get("available"):
|
|
1743
|
+
return CheckResult("server-projects", SKIP, detail="project engine not wired")
|
|
1744
|
+
mode = body.get("mode", "?")
|
|
1745
|
+
projects = body.get("projects") or []
|
|
1746
|
+
blocked = [p for p in projects if p.get("status") == "blocked"]
|
|
1747
|
+
if blocked:
|
|
1748
|
+
return CheckResult(
|
|
1749
|
+
"server-projects", WARN,
|
|
1750
|
+
detail=f"mode={mode}, {len(projects)} project(s); "
|
|
1751
|
+
f"{len(blocked)} BLOCKED: "
|
|
1752
|
+
+ ", ".join(p.get("title", "?")[:40] for p in blocked[:3]),
|
|
1753
|
+
remedy="a blocked project hit an owner boundary or repeated failure — "
|
|
1754
|
+
"review with the project_status tool or GET /v1/host/projects")
|
|
1755
|
+
return CheckResult(
|
|
1756
|
+
"server-projects", PASS, detail=f"mode={mode}, {len(projects)} project(s)")
|
|
1757
|
+
|
|
1758
|
+
|
|
1759
|
+
def check_server_beliefs(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1760
|
+
"""23. Belief maintenance: mode + unresolved contradictions."""
|
|
1761
|
+
status, body = _http_get(f"{base_url}/v1/host/beliefs", api_key, timeout)
|
|
1762
|
+
if status == 404:
|
|
1763
|
+
return CheckResult("server-beliefs", SKIP, detail="endpoint absent (older server)")
|
|
1764
|
+
if status != 200 or not isinstance(body, dict):
|
|
1765
|
+
return CheckResult("server-beliefs", FAIL, detail=f"HTTP {status}: {body}")
|
|
1766
|
+
if not body.get("available"):
|
|
1767
|
+
return CheckResult("server-beliefs", SKIP, detail="belief engine not wired")
|
|
1768
|
+
err = _reported_error("server-beliefs", body)
|
|
1769
|
+
if err:
|
|
1770
|
+
return err
|
|
1771
|
+
mode = body.get("mode", "?")
|
|
1772
|
+
open_conflicts = int(body.get("open_conflicts") or 0)
|
|
1773
|
+
review = int(body.get("review_conflicts") or 0)
|
|
1774
|
+
if review > 0:
|
|
1775
|
+
return CheckResult(
|
|
1776
|
+
"server-beliefs", WARN,
|
|
1777
|
+
detail=f"mode={mode}; {review} contradiction(s) await owner review",
|
|
1778
|
+
remedy="use the belief_conflicts tool (or GET /v1/host/beliefs) and "
|
|
1779
|
+
"resolve or dismiss them")
|
|
1780
|
+
return CheckResult(
|
|
1781
|
+
"server-beliefs", PASS,
|
|
1782
|
+
detail=f"mode={mode}, {open_conflicts} open conflict(s)")
|
|
1783
|
+
|
|
1784
|
+
|
|
1785
|
+
def check_server_workers_governor(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1786
|
+
"""24. Worker governor: server-side enforcement posture."""
|
|
1787
|
+
status, body = _http_get(f"{base_url}/v1/host/queue/governor", api_key, timeout)
|
|
1788
|
+
if status == 404:
|
|
1789
|
+
return CheckResult("server-workers-governor", SKIP, detail="endpoint absent (older server)")
|
|
1790
|
+
if status != 200 or not isinstance(body, dict):
|
|
1791
|
+
return CheckResult("server-workers-governor", FAIL, detail=f"HTTP {status}: {body}")
|
|
1792
|
+
if not body.get("available"):
|
|
1793
|
+
return CheckResult("server-workers-governor", SKIP, detail="governor not wired")
|
|
1794
|
+
err = _reported_error("server-workers-governor", body)
|
|
1795
|
+
if err:
|
|
1796
|
+
return err
|
|
1797
|
+
mode = body.get("mode", "?")
|
|
1798
|
+
domains = body.get("worker_domains") or []
|
|
1799
|
+
if mode == "off" and domains:
|
|
1800
|
+
return CheckResult(
|
|
1801
|
+
"server-workers-governor", WARN,
|
|
1802
|
+
detail=f"workers have a track record ({len(domains)} domain(s)) but the "
|
|
1803
|
+
"governor is OFF — worker claims/completions are not re-checked "
|
|
1804
|
+
"server-side",
|
|
1805
|
+
remedy="set COLONY_WORKERS_MODE=shadow (observe) or live (enforce)")
|
|
1806
|
+
return CheckResult(
|
|
1807
|
+
"server-workers-governor", PASS,
|
|
1808
|
+
detail=f"mode={mode}, {len(domains)} worker trust domain(s)")
|
|
1809
|
+
|
|
1810
|
+
|
|
1811
|
+
def check_server_sandbox(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1812
|
+
"""25. Sandbox: mode/backends consistent."""
|
|
1813
|
+
status, body = _http_get(f"{base_url}/v1/host/sandbox/status", api_key, timeout)
|
|
1814
|
+
if status == 404:
|
|
1815
|
+
return CheckResult("server-sandbox", SKIP, detail="endpoint absent (older server)")
|
|
1816
|
+
if status != 200 or not isinstance(body, dict):
|
|
1817
|
+
return CheckResult("server-sandbox", FAIL, detail=f"HTTP {status}: {body}")
|
|
1818
|
+
if not body.get("available"):
|
|
1819
|
+
return CheckResult("server-sandbox", SKIP, detail="sandbox not wired")
|
|
1820
|
+
err = _reported_error("server-sandbox", body)
|
|
1821
|
+
if err:
|
|
1822
|
+
return err
|
|
1823
|
+
mode = body.get("mode", "off")
|
|
1824
|
+
backend_ok = bool(body.get("backend_available"))
|
|
1825
|
+
if mode != "off" and not backend_ok:
|
|
1826
|
+
return CheckResult(
|
|
1827
|
+
"server-sandbox", WARN,
|
|
1828
|
+
detail=f"mode={mode} but the container backend is unavailable — "
|
|
1829
|
+
"sandbox_run will refuse every request",
|
|
1830
|
+
remedy="install/start Docker on the sidecar host, or set "
|
|
1831
|
+
"COLONY_SANDBOX_MODE=off")
|
|
1832
|
+
return CheckResult(
|
|
1833
|
+
"server-sandbox", PASS,
|
|
1834
|
+
detail=f"mode={mode}, backend {'available' if backend_ok else 'absent'}")
|
|
1835
|
+
|
|
1836
|
+
|
|
1837
|
+
def check_server_connectors(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1838
|
+
"""26. Senses: connector framework mode vs registered connectors."""
|
|
1839
|
+
status, body = _http_get(f"{base_url}/v1/host/connectors/status", api_key, timeout)
|
|
1840
|
+
if status == 404:
|
|
1841
|
+
return CheckResult("server-connectors", SKIP, detail="endpoint absent (older server)")
|
|
1842
|
+
if status != 200 or not isinstance(body, dict):
|
|
1843
|
+
return CheckResult("server-connectors", FAIL, detail=f"HTTP {status}: {body}")
|
|
1844
|
+
if not body.get("available"):
|
|
1845
|
+
return CheckResult("server-connectors", SKIP, detail="connector manager not wired")
|
|
1846
|
+
err = _reported_error("server-connectors", body)
|
|
1847
|
+
if err:
|
|
1848
|
+
return err
|
|
1849
|
+
mode = body.get("mode", "off")
|
|
1850
|
+
connectors = body.get("connectors") or []
|
|
1851
|
+
if mode != "off" and not connectors:
|
|
1852
|
+
return CheckResult(
|
|
1853
|
+
"server-connectors", WARN,
|
|
1854
|
+
detail=f"mode={mode} but no connector is registered — the agent has no "
|
|
1855
|
+
"senses configured",
|
|
1856
|
+
remedy="enable at least one: COLONY_CONNECTOR_FS_PATH (folder watch), "
|
|
1857
|
+
"COLONY_CONNECTOR_IMAP_* (email), COLONY_CONNECTOR_CALENDAR_ICS_URL, "
|
|
1858
|
+
"or COLONY_CONNECTOR_WEBHOOK_URL; each also needs "
|
|
1859
|
+
"COLONY_CONNECTOR_<NAME>_ENABLED=true")
|
|
1860
|
+
names = ",".join(str(c.get("name", "?")) for c in connectors) or "none"
|
|
1861
|
+
return CheckResult(
|
|
1862
|
+
"server-connectors", PASS, detail=f"mode={mode}, connectors: {names}")
|
|
1863
|
+
|
|
1864
|
+
|
|
1865
|
+
def check_server_mining(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1866
|
+
"""27. Self-improvement mining: escalation miner reachable."""
|
|
1867
|
+
status, body = _http_get(
|
|
1868
|
+
f"{base_url}/v1/host/mining/escalations?limit=1", api_key, timeout)
|
|
1869
|
+
if status in (404, 501):
|
|
1870
|
+
return CheckResult("server-mining", SKIP, detail="mining not wired (or mode off)")
|
|
1871
|
+
if status != 200 or not isinstance(body, dict):
|
|
1872
|
+
return CheckResult("server-mining", WARN, detail=f"HTTP {status}: {body}")
|
|
1873
|
+
mode = body.get("mode", "?")
|
|
1874
|
+
recent = len(body.get("escalations") or [])
|
|
1875
|
+
stats = body.get("stats") or {}
|
|
1876
|
+
total = stats.get("total", stats.get("recorded", recent))
|
|
1877
|
+
return CheckResult(
|
|
1878
|
+
"server-mining", PASS,
|
|
1879
|
+
detail=f"mode={mode}, escalation miner reachable ({total} recorded)")
|
|
1880
|
+
|
|
1881
|
+
|
|
1882
|
+
def check_server_directives(base_url: str, api_key: str, timeout: float) -> CheckResult:
|
|
1883
|
+
"""28. Directive-store hygiene: no fragment boundaries, no duplicate piles.
|
|
1884
|
+
|
|
1885
|
+
Guards against the 2026-07-05 self-poisoning class: anaphora fragments
|
|
1886
|
+
("attempt them", "that and wipe it from colony") and repeated captures
|
|
1887
|
+
that, once the worker governor enforces, block routine work wholesale.
|
|
1888
|
+
"""
|
|
1889
|
+
status, body = _http_get(f"{base_url}/v1/host/directives", api_key, timeout)
|
|
1890
|
+
if status == 404:
|
|
1891
|
+
return CheckResult("server-directives", SKIP, detail="endpoint absent (older server)")
|
|
1892
|
+
if status != 200 or not isinstance(body, dict):
|
|
1893
|
+
return CheckResult("server-directives", WARN, detail=f"HTTP {status}: {body}")
|
|
1894
|
+
if not body.get("available", True):
|
|
1895
|
+
return CheckResult("server-directives", SKIP, detail="directives not wired")
|
|
1896
|
+
directives = body.get("directives") or []
|
|
1897
|
+
import re as _re
|
|
1898
|
+
frag_lead = _re.compile(
|
|
1899
|
+
r"^(?:that|this|it|them|those|these|him|her|or|and|but|so|then|there|y|x)\b",
|
|
1900
|
+
_re.IGNORECASE)
|
|
1901
|
+
fragments = [d for d in directives
|
|
1902
|
+
if frag_lead.match((d.get("subject") or "").strip())]
|
|
1903
|
+
counts: dict = {}
|
|
1904
|
+
for d in directives:
|
|
1905
|
+
s = (d.get("subject") or "").strip().lower()
|
|
1906
|
+
counts[s] = counts.get(s, 0) + 1
|
|
1907
|
+
piles = {s: n for s, n in counts.items() if n >= 3}
|
|
1908
|
+
problems = []
|
|
1909
|
+
if fragments:
|
|
1910
|
+
problems.append(f"{len(fragments)} fragment subject(s) e.g. "
|
|
1911
|
+
+ repr((fragments[0].get("subject") or "")[:40]))
|
|
1912
|
+
if piles:
|
|
1913
|
+
worst = max(piles.items(), key=lambda kv: kv[1])
|
|
1914
|
+
problems.append(f"{len(piles)} subject(s) duplicated (worst: "
|
|
1915
|
+
f"{worst[1]}x {worst[0][:40]!r})")
|
|
1916
|
+
if problems:
|
|
1917
|
+
return CheckResult(
|
|
1918
|
+
"server-directives", WARN,
|
|
1919
|
+
detail=f"{len(directives)} directive(s); " + "; ".join(problems),
|
|
1920
|
+
remedy="review GET /v1/host/directives and revoke the poisoned "
|
|
1921
|
+
"entries (POST /directives/{id}/revoke); capture-quality "
|
|
1922
|
+
"gates prevent new ones")
|
|
1923
|
+
return CheckResult(
|
|
1924
|
+
"server-directives", PASS,
|
|
1925
|
+
detail=f"{len(directives)} directive(s), no fragments or piles")
|
|
1926
|
+
|
|
1927
|
+
|
|
1928
|
+
def run_server_checks(base_url: str, api_key: str, timeout: float = 10.0) -> List[CheckResult]:
|
|
1929
|
+
"""Run all HTTP checks, skipping the rest when the sidecar is down."""
|
|
1930
|
+
base_url = base_url.rstrip("/")
|
|
1931
|
+
results: List[CheckResult] = []
|
|
1932
|
+
|
|
1933
|
+
# 10. Connectivity — everything else skips when this fails.
|
|
1934
|
+
try:
|
|
1935
|
+
status, body = _http_get(f"{base_url}/v1/host/health", api_key, timeout)
|
|
1936
|
+
except Exception as exc: # noqa: BLE001 — URLError, OSError, timeouts
|
|
1937
|
+
results.append(CheckResult(
|
|
1938
|
+
"server-health", FAIL,
|
|
1939
|
+
detail=f"sidecar not reachable at {base_url}: {exc}",
|
|
1940
|
+
remedy="start it with 'colony start' (or 'colony service start'), then re-run "
|
|
1941
|
+
"'colony doctor'",
|
|
1942
|
+
))
|
|
1943
|
+
reason = f"sidecar unreachable at {base_url}"
|
|
1944
|
+
for name in SERVER_CHECK_NAMES[1:]:
|
|
1945
|
+
results.append(CheckResult(name, SKIP, detail=reason))
|
|
1946
|
+
return results
|
|
1947
|
+
|
|
1948
|
+
if status == 200 and isinstance(body, dict):
|
|
1949
|
+
health_status = body.get("status", "unknown")
|
|
1950
|
+
if health_status == "ok":
|
|
1951
|
+
results.append(CheckResult(
|
|
1952
|
+
"server-health", PASS,
|
|
1953
|
+
detail=f"sidecar healthy ({len(body.get('capabilities') or [])} capabilities)",
|
|
1954
|
+
))
|
|
1955
|
+
else:
|
|
1956
|
+
degraded = "; ".join(
|
|
1957
|
+
f"{k}: {v}" for k, v in (body.get("notes") or {}).items()
|
|
1958
|
+
if any(w in str(v).lower() for w in ("fail", "error", "not wired", "warning"))
|
|
1959
|
+
)
|
|
1960
|
+
results.append(CheckResult(
|
|
1961
|
+
"server-health", WARN,
|
|
1962
|
+
detail=f"sidecar reports status={health_status}"
|
|
1963
|
+
+ (f" — {degraded}" if degraded else ""),
|
|
1964
|
+
remedy="check the sidecar log; 'colony status' shows the degraded subsystems",
|
|
1965
|
+
))
|
|
1966
|
+
else:
|
|
1967
|
+
results.append(CheckResult(
|
|
1968
|
+
"server-health", FAIL,
|
|
1969
|
+
detail=f"/v1/host/health returned HTTP {status}: {body}",
|
|
1970
|
+
))
|
|
1971
|
+
|
|
1972
|
+
results += _run("server-auth", check_server_auth, base_url, api_key, timeout)
|
|
1973
|
+
results += _run(
|
|
1974
|
+
"server-auth-migration", check_server_auth_migration,
|
|
1975
|
+
base_url, api_key, timeout,
|
|
1976
|
+
)
|
|
1977
|
+
results += _run("server-owner-contact", check_server_owner_contact, base_url, api_key, timeout)
|
|
1978
|
+
results += _run("server-llm-router", check_server_llm, base_url, api_key, timeout)
|
|
1979
|
+
results += _run("server-embedder", check_server_embedder, base_url, api_key, timeout)
|
|
1980
|
+
results += _run("server-memory-graph", check_server_memory_graph,
|
|
1981
|
+
base_url, api_key, timeout)
|
|
1982
|
+
results += _run("server-fd-limit", check_server_fd_limit, base_url, api_key, timeout)
|
|
1983
|
+
results += _run("server-blocked-approvals", check_server_blocked_approvals,
|
|
1984
|
+
base_url, api_key, timeout)
|
|
1985
|
+
results += _run("server-worker-liveness", check_server_worker_liveness,
|
|
1986
|
+
base_url, api_key, timeout)
|
|
1987
|
+
results += _run("server-skills-observations", check_server_skills_observations,
|
|
1988
|
+
base_url, api_key, timeout)
|
|
1989
|
+
# Cognition / autonomy visibility (v0.22.0)
|
|
1990
|
+
results += _run("server-autonomy-posture", check_server_autonomy_posture,
|
|
1991
|
+
base_url, api_key, timeout)
|
|
1992
|
+
results += _run("server-grant-envelope", check_server_grant_envelope,
|
|
1993
|
+
base_url, api_key, timeout)
|
|
1994
|
+
results += _run("server-self-model", check_server_self_model,
|
|
1995
|
+
base_url, api_key, timeout)
|
|
1996
|
+
results += _run("server-adaptive-params", check_server_adaptive_params,
|
|
1997
|
+
base_url, api_key, timeout)
|
|
1998
|
+
results += _run("server-executor", check_server_executor,
|
|
1999
|
+
base_url, api_key, timeout)
|
|
2000
|
+
results += _run("server-projects", check_server_projects,
|
|
2001
|
+
base_url, api_key, timeout)
|
|
2002
|
+
results += _run("server-beliefs", check_server_beliefs,
|
|
2003
|
+
base_url, api_key, timeout)
|
|
2004
|
+
results += _run("server-workers-governor", check_server_workers_governor,
|
|
2005
|
+
base_url, api_key, timeout)
|
|
2006
|
+
results += _run("server-sandbox", check_server_sandbox,
|
|
2007
|
+
base_url, api_key, timeout)
|
|
2008
|
+
results += _run("server-connectors", check_server_connectors,
|
|
2009
|
+
base_url, api_key, timeout)
|
|
2010
|
+
results += _run("server-mining", check_server_mining,
|
|
2011
|
+
base_url, api_key, timeout)
|
|
2012
|
+
results += _run("server-directives", check_server_directives,
|
|
2013
|
+
base_url, api_key, timeout)
|
|
2014
|
+
results += _run("server-benchmark", check_server_benchmark,
|
|
2015
|
+
base_url, api_key, timeout)
|
|
2016
|
+
results += _run("server-toolsmith", check_server_toolsmith,
|
|
2017
|
+
base_url, api_key, timeout)
|
|
2018
|
+
results += _run("server-workspace", check_server_workspace,
|
|
2019
|
+
base_url, api_key, timeout)
|
|
2020
|
+
results += _run("server-expectations", check_server_expectations,
|
|
2021
|
+
base_url, api_key, timeout)
|
|
2022
|
+
return results
|
|
2023
|
+
|
|
2024
|
+
|
|
2025
|
+
# ---------------------------------------------------------------------------
|
|
2026
|
+
# Engine entry point + reporting
|
|
2027
|
+
# ---------------------------------------------------------------------------
|
|
2028
|
+
|
|
2029
|
+
def run_private_instance_checks(base_url: str, api_key: str, timeout: float) -> List[CheckResult]:
|
|
2030
|
+
"""Diagnose the local profile through its existing exact-person API."""
|
|
2031
|
+
results = []
|
|
2032
|
+
for name, check in (("state-dir", check_state_dir), ("llm-config", check_llm_config),
|
|
2033
|
+
("contacts-db", check_contacts_db), ("owner-contact-id", check_owner_contact_id),
|
|
2034
|
+
("optional-video", check_optional_video)):
|
|
2035
|
+
results += _run(name, check)
|
|
2036
|
+
try:
|
|
2037
|
+
status, body = _http_get(base_url + '/v1/host/health', api_key, timeout)
|
|
2038
|
+
healthy = status == 200 and isinstance(body, dict) and body.get('status') == 'ok'
|
|
2039
|
+
results.append(CheckResult('server-health', PASS if healthy else FAIL,
|
|
2040
|
+
detail='Sidecar HTTP health is ready' if healthy else f'Sidecar health is not ready (HTTP {status})'))
|
|
2041
|
+
except Exception as exc:
|
|
2042
|
+
results.append(CheckResult('server-health', FAIL, detail='Sidecar is unavailable: ' + type(exc).__name__,
|
|
2043
|
+
remedy='Start this private instance and inspect its sidecar.log.'))
|
|
2044
|
+
healthy = False
|
|
2045
|
+
owner = os.environ.get('COLONY_OWNER_CONTACT_ID', '')
|
|
2046
|
+
if not healthy or not owner or not api_key:
|
|
2047
|
+
results.append(CheckResult('server-source-memory', FAIL,
|
|
2048
|
+
detail='A ready instance, owner contact and scoped client credential are required.',
|
|
2049
|
+
remedy='Use the selected instance environment and its COLONY_CLIENT_API_KEY.'))
|
|
2050
|
+
return results
|
|
2051
|
+
try:
|
|
2052
|
+
status, body = _http_get(base_url + '/v1/host/memory/sources/claims/status?' +
|
|
2053
|
+
urlencode({'contact_id': owner}), api_key, timeout)
|
|
2054
|
+
if (status != 200 or not isinstance(body, dict) or
|
|
2055
|
+
not all(isinstance(body.get(name), list) for name in ('sources', 'media'))):
|
|
2056
|
+
results.append(CheckResult('server-source-memory', FAIL,
|
|
2057
|
+
detail=f'Scoped source status unavailable (HTTP {status}); check the selected owner and client credential.'))
|
|
2058
|
+
else:
|
|
2059
|
+
jobs = body['sources'] + body['media']
|
|
2060
|
+
errors = sum(bool(row.get('error')) for row in jobs if isinstance(row, dict))
|
|
2061
|
+
pending = sum(row.get('status') != 'complete' for row in jobs if isinstance(row, dict))
|
|
2062
|
+
results.append(CheckResult('server-source-memory', WARN if errors else PASS,
|
|
2063
|
+
detail=f'Scoped source status accepted: {len(jobs)} recent jobs, {pending} pending, {errors} with errors. '
|
|
2064
|
+
'This verifies access and reported jobs, not model recall quality.'))
|
|
2065
|
+
except Exception as exc:
|
|
2066
|
+
results.append(CheckResult('server-source-memory', FAIL,
|
|
2067
|
+
detail='Scoped source status unavailable: ' + type(exc).__name__))
|
|
2068
|
+
return results
|
|
2069
|
+
|
|
2070
|
+
|
|
2071
|
+
def run_doctor(
|
|
2072
|
+
colony_url: Optional[str] = None,
|
|
2073
|
+
api_key: Optional[str] = None,
|
|
2074
|
+
timeout: float = 10.0,
|
|
2075
|
+
) -> List[CheckResult]:
|
|
2076
|
+
"""Run every local and server check; never raises."""
|
|
2077
|
+
url = colony_url or default_colony_url()
|
|
2078
|
+
key = api_key if api_key is not None else default_api_key()
|
|
2079
|
+
if os.environ.get('COLONY_INSTALL_PROFILE') == 'local':
|
|
2080
|
+
return run_private_instance_checks(url, key, timeout)
|
|
2081
|
+
results = run_local_checks()
|
|
2082
|
+
results += run_server_checks(url, key, timeout=timeout)
|
|
2083
|
+
return results
|
|
2084
|
+
|
|
2085
|
+
|
|
2086
|
+
def default_api_key() -> str:
|
|
2087
|
+
if os.environ.get('COLONY_INSTALL_PROFILE') == 'local':
|
|
2088
|
+
return os.environ.get('COLONY_CLIENT_API_KEY', '')
|
|
2089
|
+
return os.environ.get('COLONY_API_KEY', '')
|
|
2090
|
+
|
|
2091
|
+
|
|
2092
|
+
def default_colony_url() -> str:
|
|
2093
|
+
"""Resolve the sidecar URL the same way the other CLI commands do."""
|
|
2094
|
+
explicit = os.environ.get("COLONY_URL") or os.environ.get("COLONY_SIDECAR_URL")
|
|
2095
|
+
if explicit:
|
|
2096
|
+
return explicit
|
|
2097
|
+
host = os.environ.get("COLONY_SIDECAR_HOST", "127.0.0.1")
|
|
2098
|
+
port = os.environ.get("COLONY_SIDECAR_PORT", "7777")
|
|
2099
|
+
return f"http://{host}:{port}"
|
|
2100
|
+
|
|
2101
|
+
|
|
2102
|
+
def summarize(results: List[CheckResult]) -> dict:
|
|
2103
|
+
counts = {PASS: 0, WARN: 0, FAIL: 0, SKIP: 0}
|
|
2104
|
+
for r in results:
|
|
2105
|
+
counts[r.status] = counts.get(r.status, 0) + 1
|
|
2106
|
+
return counts
|
|
2107
|
+
|
|
2108
|
+
|
|
2109
|
+
def exit_code(results: List[CheckResult]) -> int:
|
|
2110
|
+
"""0 when nothing failed (warns are OK), 1 otherwise."""
|
|
2111
|
+
return 1 if any(r.status == FAIL for r in results) else 0
|
|
2112
|
+
|
|
2113
|
+
|
|
2114
|
+
def skip_dominated(results: List[CheckResult]) -> bool:
|
|
2115
|
+
"""True when more checks skipped than passed — the run verified almost
|
|
2116
|
+
nothing, so it must not claim health."""
|
|
2117
|
+
counts = summarize(results)
|
|
2118
|
+
return counts[SKIP] > counts[PASS]
|
|
2119
|
+
|
|
2120
|
+
|
|
2121
|
+
def results_to_json(results: List[CheckResult]) -> dict:
|
|
2122
|
+
counts = summarize(results)
|
|
2123
|
+
return {
|
|
2124
|
+
"results": [r.to_dict() for r in results],
|
|
2125
|
+
"summary": counts,
|
|
2126
|
+
"ok": exit_code(results) == 0 and not skip_dominated(results),
|
|
2127
|
+
}
|
|
2128
|
+
|
|
2129
|
+
|
|
2130
|
+
_ICONS = {PASS: "✅", WARN: "⚠️ ", FAIL: "❌", SKIP: "⚪"}
|
|
2131
|
+
_COLORS = {PASS: "\033[92m", WARN: "\033[93m", FAIL: "\033[91m", SKIP: "\033[90m"}
|
|
2132
|
+
_RESET = "\033[0m"
|
|
2133
|
+
|
|
2134
|
+
|
|
2135
|
+
def format_report(results: List[CheckResult], colony_url: str = "", color: bool = True) -> str:
|
|
2136
|
+
"""Human-readable report: aligned status lines, remedies indented."""
|
|
2137
|
+
lines: List[str] = []
|
|
2138
|
+
header = "🩺 Colony Doctor"
|
|
2139
|
+
if colony_url:
|
|
2140
|
+
header += f" — {colony_url}"
|
|
2141
|
+
lines.append(header)
|
|
2142
|
+
lines.append("")
|
|
2143
|
+
|
|
2144
|
+
width = max((len(r.name) for r in results), default=0)
|
|
2145
|
+
for r in results:
|
|
2146
|
+
label = r.status.upper().ljust(4)
|
|
2147
|
+
if color:
|
|
2148
|
+
label = f"{_COLORS.get(r.status, '')}{label}{_RESET}"
|
|
2149
|
+
line = f" {_ICONS.get(r.status, ' ')} {label} {r.name.ljust(width)}"
|
|
2150
|
+
if r.detail:
|
|
2151
|
+
line += f" {r.detail}"
|
|
2152
|
+
lines.append(line)
|
|
2153
|
+
if r.remedy and r.status in (WARN, FAIL):
|
|
2154
|
+
lines.append(f" ↳ {r.remedy}")
|
|
2155
|
+
|
|
2156
|
+
counts = summarize(results)
|
|
2157
|
+
lines.append("")
|
|
2158
|
+
lines.append(
|
|
2159
|
+
f" {counts[PASS]} pass, {counts[WARN]} warn, {counts[FAIL]} fail, {counts[SKIP]} skip"
|
|
2160
|
+
)
|
|
2161
|
+
if counts[FAIL]:
|
|
2162
|
+
verdict = f" 🔴 {counts[FAIL]} check(s) failing — fix the remedies above"
|
|
2163
|
+
elif skip_dominated(results):
|
|
2164
|
+
verdict = (
|
|
2165
|
+
f" ⚪ inconclusive — {counts[SKIP]} check(s) skipped, only "
|
|
2166
|
+
f"{counts[PASS]} verified; not claiming health"
|
|
2167
|
+
)
|
|
2168
|
+
elif counts[WARN]:
|
|
2169
|
+
verdict = f" 🟡 healthy with {counts[WARN]} warning(s)"
|
|
2170
|
+
else:
|
|
2171
|
+
verdict = " 🟢 all checks healthy"
|
|
2172
|
+
lines.append(verdict)
|
|
2173
|
+
return "\n".join(lines)
|