apsimo 1.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- apsimo/__init__.py +38 -0
- apsimo/__main__.py +6 -0
- apsimo/agent/__init__.py +6 -0
- apsimo/agent/client.py +276 -0
- apsimo/agent/models.py +46 -0
- apsimo/agents/__init__.py +20 -0
- apsimo/agents/models.py +264 -0
- apsimo/agents/store.py +861 -0
- apsimo/agents/websocket.py +522 -0
- apsimo/api/__init__.py +1 -0
- apsimo/api/auth_telemetry.py +287 -0
- apsimo/api/authority.py +1203 -0
- apsimo/api/contact_grants.py +347 -0
- apsimo/api/middleware.py +483 -0
- apsimo/api/routers/__init__.py +1 -0
- apsimo/api/routers/commitment_work.py +265 -0
- apsimo/api/routers/context_gate.py +123 -0
- apsimo/api/routers/executions.py +140 -0
- apsimo/api/routers/followup_plans.py +147 -0
- apsimo/api/routers/governed_actions.py +162 -0
- apsimo/api/routers/host.py +14473 -0
- apsimo/api/routers/initiative_work.py +115 -0
- apsimo/api/routers/mining.py +104 -0
- apsimo/api/routers/observations.py +110 -0
- apsimo/api/routers/social_state.py +225 -0
- apsimo/api/routers/task_queue.py +2715 -0
- apsimo/api/routers/temporal_followups.py +251 -0
- apsimo/api/routers/transport.py +110 -0
- apsimo/api/routers/transport_ingress_api.py +240 -0
- apsimo/api/schemas/__init__.py +1 -0
- apsimo/api/schemas/host.py +1949 -0
- apsimo/autonomy/cli.py +110 -0
- apsimo/autonomy/condition_worker.py +437 -0
- apsimo/autonomy/config.py +424 -0
- apsimo/autonomy/loop.py +4316 -0
- apsimo/autonomy/registry.py +339 -0
- apsimo/autonomy/scheduler.py +1822 -0
- apsimo/autonomy/synthesis.py +449 -0
- apsimo/backup.py +962 -0
- apsimo/beliefs/__init__.py +23 -0
- apsimo/beliefs/contradictions.py +109 -0
- apsimo/beliefs/decay.py +61 -0
- apsimo/beliefs/engine.py +479 -0
- apsimo/beliefs/models.py +67 -0
- apsimo/beliefs/promotion.py +41 -0
- apsimo/beliefs/resolve.py +58 -0
- apsimo/beliefs/source_claims.py +690 -0
- apsimo/beliefs/source_projection.py +883 -0
- apsimo/beliefs/source_time.py +208 -0
- apsimo/beliefs/store.py +133 -0
- apsimo/briefings/aggregators.py +824 -0
- apsimo/briefings/composer.py +420 -0
- apsimo/briefings/config.py +55 -0
- apsimo/briefings/delivery.py +439 -0
- apsimo/briefings/engagement.py +97 -0
- apsimo/briefings/engine.py +274 -0
- apsimo/briefings/enhancer.py +99 -0
- apsimo/briefings/models.py +183 -0
- apsimo/briefings/scheduler.py +382 -0
- apsimo/briefings/store.py +435 -0
- apsimo/chain/__init__.py +48 -0
- apsimo/chain/block.py +100 -0
- apsimo/chain/cli.py +704 -0
- apsimo/chain/genesis.py +443 -0
- apsimo/chain/identity.py +416 -0
- apsimo/chain/keys.py +1025 -0
- apsimo/chain/local_keys.py +187 -0
- apsimo/chain/manager.py +290 -0
- apsimo/chain/node.py +163 -0
- apsimo/chain/plugin_transactions.py +371 -0
- apsimo/chain/protocol.py +220 -0
- apsimo/chain/state_machine.py +676 -0
- apsimo/chain/storage.py +503 -0
- apsimo/chain/transactions.py +250 -0
- apsimo/chain/validation.py +397 -0
- apsimo/channels/__init__.py +1 -0
- apsimo/channels/manifest.py +31 -0
- apsimo/channels/migrations/001_channels_schema.sql +12 -0
- apsimo/channels/phone_gateways.py +42 -0
- apsimo/channels/presence.py +188 -0
- apsimo/channels/router.py +235 -0
- apsimo/channels/store.py +231 -0
- apsimo/cli.py +2688 -0
- apsimo/cognition/__init__.py +11 -0
- apsimo/cognition/charter.py +398 -0
- apsimo/cognition/drive_governance.py +3530 -0
- apsimo/cognition/evidence_pipeline.py +1627 -0
- apsimo/cognition/external_events.py +932 -0
- apsimo/cognition/goal_spine.py +3488 -0
- apsimo/cognition/introspection.py +214 -0
- apsimo/cognition/prompt.py +150 -0
- apsimo/cognition/runtime.py +108 -0
- apsimo/cognition/trigger.py +154 -0
- apsimo/commitments/__init__.py +18 -0
- apsimo/commitments/local_work.py +355 -0
- apsimo/commitments/store.py +1052 -0
- apsimo/commitments/work.py +91 -0
- apsimo/compat.py +53 -0
- apsimo/compression/__init__.py +467 -0
- apsimo/connectors/__init__.py +21 -0
- apsimo/connectors/base.py +152 -0
- apsimo/connectors/caldav_calendar.py +125 -0
- apsimo/connectors/fs_documents.py +85 -0
- apsimo/connectors/imap_email.py +138 -0
- apsimo/connectors/manager.py +218 -0
- apsimo/connectors/webhook_pull.py +88 -0
- apsimo/contacts/__init__.py +33 -0
- apsimo/contacts/comms.py +357 -0
- apsimo/contacts/config.py +79 -0
- apsimo/contacts/exporters/__init__.py +1 -0
- apsimo/contacts/exporters/vcard.py +71 -0
- apsimo/contacts/identity_links.py +251 -0
- apsimo/contacts/importer.py +280 -0
- apsimo/contacts/importers/__init__.py +1 -0
- apsimo/contacts/importers/batch.py +43 -0
- apsimo/contacts/importers/macos_contacts.py +101 -0
- apsimo/contacts/migrations/001_contacts_schema.sql +141 -0
- apsimo/contacts/migrations/002_trust_scopes.sql +36 -0
- apsimo/contacts/migrations/003_open_gateway_enum.sql +32 -0
- apsimo/contacts/migrations/004_contact_provision_operations.sql +18 -0
- apsimo/contacts/migrations/005_identity_links.sql +27 -0
- apsimo/contacts/models.py +308 -0
- apsimo/contacts/scoring.py +16 -0
- apsimo/contacts/store.py +1623 -0
- apsimo/contacts/transport_ingress.py +252 -0
- apsimo/contacts/world_bridge.py +314 -0
- apsimo/contextgate/__init__.py +69 -0
- apsimo/contextgate/chunker.py +169 -0
- apsimo/contextgate/estimate.py +54 -0
- apsimo/contextgate/gate.py +313 -0
- apsimo/contextgate/retrieve.py +115 -0
- apsimo/delivery/__init__.py +16 -0
- apsimo/delivery/bridge.py +1260 -0
- apsimo/delivery/channels.py +526 -0
- apsimo/delivery/classification.py +50 -0
- apsimo/delivery/rate_limiter.py +268 -0
- apsimo/delivery/reachout_policy.py +206 -0
- apsimo/directed/__init__.py +22 -0
- apsimo/directed/audit.py +167 -0
- apsimo/directed/intake.py +95 -0
- apsimo/directed/models.py +191 -0
- apsimo/directed/service.py +509 -0
- apsimo/directives/__init__.py +25 -0
- apsimo/directives/evidence.py +87 -0
- apsimo/directives/extractor.py +188 -0
- apsimo/directives/guard.py +364 -0
- apsimo/directives/models.py +206 -0
- apsimo/directives/service.py +372 -0
- apsimo/directives/store.py +167 -0
- apsimo/doctor.py +2173 -0
- apsimo/environment.py +43 -0
- apsimo/events/__init__.py +33 -0
- apsimo/events/broadcaster.py +98 -0
- apsimo/events/bus.py +217 -0
- apsimo/events/journal.py +863 -0
- apsimo/events/stream.py +131 -0
- apsimo/events/types.py +150 -0
- apsimo/execution_results.py +357 -0
- apsimo/feedback/__init__.py +5 -0
- apsimo/feedback/store.py +76 -0
- apsimo/feeds/__init__.py +19 -0
- apsimo/feeds/cli.py +84 -0
- apsimo/feeds/engine.py +437 -0
- apsimo/feeds/example-feed.yaml +77 -0
- apsimo/feeds/hermes_cron.py +126 -0
- apsimo/feeds/manager.py +235 -0
- apsimo/feeds/spec.py +250 -0
- apsimo/feeds/template.py +202 -0
- apsimo/gate/__init__.py +18 -0
- apsimo/gate/audit.py +61 -0
- apsimo/gate/communication_policy.py +166 -0
- apsimo/gate/config.py +72 -0
- apsimo/gate/context_provenance.py +170 -0
- apsimo/gate/env_risk.py +226 -0
- apsimo/gate/guard_audit.py +353 -0
- apsimo/gate/layers/__init__.py +1 -0
- apsimo/gate/layers/base.py +15 -0
- apsimo/gate/layers/l1_recipient.py +66 -0
- apsimo/gate/layers/l2_pii.py +134 -0
- apsimo/gate/layers/l3_cross_context.py +50 -0
- apsimo/gate/layers/l4_trust_tier.py +78 -0
- apsimo/gate/layers/l5_injection.py +199 -0
- apsimo/gate/layers/l6_review.py +86 -0
- apsimo/gate/layers/l7_delay.py +100 -0
- apsimo/gate/layers/tom2_epistemic.py +185 -0
- apsimo/gate/models.py +64 -0
- apsimo/gate/pending_dispatch.py +5 -0
- apsimo/gate/pipeline.py +206 -0
- apsimo/gate/rejection.py +259 -0
- apsimo/gate/response_guard.py +700 -0
- apsimo/gate/rulesets/injection_v1.yaml +51 -0
- apsimo/gate/surface_policy.py +189 -0
- apsimo/gate/taint.py +226 -0
- apsimo/genesis.json +9 -0
- apsimo/goals/__init__.py +100 -0
- apsimo/goals/config.py +38 -0
- apsimo/goals/decomposer.py +421 -0
- apsimo/goals/engine.py +617 -0
- apsimo/goals/inference.py +354 -0
- apsimo/goals/models.py +302 -0
- apsimo/goals/priority.py +270 -0
- apsimo/goals/queue_bridge.py +149 -0
- apsimo/goals/replan.py +450 -0
- apsimo/goals/schema.sql +89 -0
- apsimo/goals/store.py +692 -0
- apsimo/governed_actions.py +1708 -0
- apsimo/harness_integration/__init__.py +45 -0
- apsimo/harness_integration/context.py +41 -0
- apsimo/harness_integration/skills.py +231 -0
- apsimo/identity/__init__.py +26 -0
- apsimo/identity/participants.py +181 -0
- apsimo/identity/resolver.py +329 -0
- apsimo/identity_bootstrap/__init__.py +5 -0
- apsimo/identity_bootstrap/builder.py +208 -0
- apsimo/identity_bootstrap/corpus.py +443 -0
- apsimo/identity_bootstrap/models.py +54 -0
- apsimo/identity_bootstrap/runner.py +353 -0
- apsimo/identity_bootstrap/seeders/__init__.py +25 -0
- apsimo/identity_bootstrap/seeders/briefings.py +109 -0
- apsimo/identity_bootstrap/seeders/chain.py +57 -0
- apsimo/identity_bootstrap/seeders/goals.py +128 -0
- apsimo/identity_bootstrap/seeders/memory.py +191 -0
- apsimo/identity_bootstrap/seeders/neo4j_cognition.py +79 -0
- apsimo/identity_bootstrap/seeders/relationship.py +152 -0
- apsimo/identity_bootstrap/seeders/sessions.py +67 -0
- apsimo/identity_bootstrap/seeders/skills.py +92 -0
- apsimo/identity_bootstrap/seeders/task_queue.py +72 -0
- apsimo/identity_bootstrap/seeders/world_model.py +143 -0
- apsimo/identity_bootstrap/self_query.py +92 -0
- apsimo/identity_bootstrap/self_reflection.py +155 -0
- apsimo/identity_bootstrap/skill.py +37 -0
- apsimo/identity_bootstrap/verifier.py +436 -0
- apsimo/initiatives/__init__.py +20 -0
- apsimo/initiatives/action_registry.py +454 -0
- apsimo/initiatives/approval_authority.py +2105 -0
- apsimo/initiatives/approval_policy.py +123 -0
- apsimo/initiatives/assignment.py +263 -0
- apsimo/initiatives/backup_evidence.py +100 -0
- apsimo/initiatives/context_freshness.py +103 -0
- apsimo/initiatives/models.py +318 -0
- apsimo/initiatives/native_work.py +270 -0
- apsimo/initiatives/standing_approvals.py +232 -0
- apsimo/initiatives/store.py +1081 -0
- apsimo/initiatives/temporal_followup.py +410 -0
- apsimo/intelligence/__init__.py +1 -0
- apsimo/intelligence/cognition/__init__.py +24 -0
- apsimo/intelligence/cognition/gap_detector.py +148 -0
- apsimo/intelligence/cognition/metalearner.py +547 -0
- apsimo/intelligence/cognition/metrics_collector.py +217 -0
- apsimo/intelligence/cognition/performance_index.py +299 -0
- apsimo/intelligence/cognition/registry.py +192 -0
- apsimo/intelligence/cognition/strategy_adjuster.py +222 -0
- apsimo/intelligence/cognition/types.py +16 -0
- apsimo/intelligence/components/__init__.py +66 -0
- apsimo/intelligence/components/anomaly_detector.py +413 -0
- apsimo/intelligence/components/initiative_engine.py +2643 -0
- apsimo/intelligence/components/preference_learner.py +521 -0
- apsimo/intelligence/components/research_orchestrator.py +358 -0
- apsimo/intelligence/components/self_directed_thinker.py +221 -0
- apsimo/intelligence/components/self_reflector.py +252 -0
- apsimo/intelligence/components/session_continuity.py +154 -0
- apsimo/intelligence/components/task_planner.py +320 -0
- apsimo/intelligence/components/tool_learner.py +217 -0
- apsimo/intelligence/graph/__init__.py +79 -0
- apsimo/intelligence/graph/client.py +2483 -0
- apsimo/intelligence/graph/consolidator.py +405 -0
- apsimo/intelligence/graph/distiller.py +312 -0
- apsimo/intelligence/graph/migrations.py +129 -0
- apsimo/intelligence/graph/queries.py +248 -0
- apsimo/intelligence/graph/recall.py +281 -0
- apsimo/intelligence/graph/reconciler.py +144 -0
- apsimo/intelligence/graph/schema.py +337 -0
- apsimo/intelligence/graph/selection.py +252 -0
- apsimo/intelligence/learning/__init__.py +17 -0
- apsimo/intelligence/learning/continuous_learner.py +245 -0
- apsimo/intelligence/learning/feedback_store.py +321 -0
- apsimo/intelligence/mind_model/__init__.py +1 -0
- apsimo/intelligence/mind_model/graph_baseline.py +136 -0
- apsimo/intelligence/mind_model/signal_collector.py +361 -0
- apsimo/intelligence/relationships/__init__.py +11 -0
- apsimo/intelligence/relationships/profiler.py +389 -0
- apsimo/intelligence/relationships/scorer.py +560 -0
- apsimo/intelligence/relationships/signal_floor.py +66 -0
- apsimo/intelligence/relationships/trust_tiers.py +300 -0
- apsimo/intelligence/synthesis/__init__.py +40 -0
- apsimo/intelligence/synthesis/connection_discoverer.py +379 -0
- apsimo/intelligence/synthesis/cross_domain_analyzer.py +287 -0
- apsimo/intelligence/synthesis/insight_deliverer.py +171 -0
- apsimo/intelligence/synthesis/insight_store.py +79 -0
- apsimo/intelligence/synthesis/insight_validator.py +183 -0
- apsimo/intelligence/synthesis/novelty_scorer.py +267 -0
- apsimo/intelligence/turn_middleware/__init__.py +15 -0
- apsimo/intelligence/turn_middleware/memory_sync.py +119 -0
- apsimo/mcp/__init__.py +41 -0
- apsimo/mcp/__main__.py +6 -0
- apsimo/mcp/config.py +287 -0
- apsimo/mcp/server.py +501 -0
- apsimo/migrations.py +187 -0
- apsimo/mining/__init__.py +27 -0
- apsimo/mining/corpus.py +239 -0
- apsimo/mining/escalations.py +289 -0
- apsimo/mining/models.py +169 -0
- apsimo/mining/store.py +210 -0
- apsimo/models/__init__.py +30 -0
- apsimo/models/memory.py +80 -0
- apsimo/models/mesh.py +72 -0
- apsimo/models/person.py +104 -0
- apsimo/models/signal.py +108 -0
- apsimo/observations/__init__.py +15 -0
- apsimo/observations/store.py +277 -0
- apsimo/patterns/__init__.py +6 -0
- apsimo/patterns/extract.py +187 -0
- apsimo/patterns/store.py +227 -0
- apsimo/persona/__init__.py +1 -0
- apsimo/persona/engine.py +611 -0
- apsimo/persona/manifest.py +140 -0
- apsimo/projects/__init__.py +28 -0
- apsimo/projects/engine.py +1681 -0
- apsimo/projects/event_outbox.py +188 -0
- apsimo/projects/models.py +216 -0
- apsimo/projects/planner.py +181 -0
- apsimo/projects/store.py +1446 -0
- apsimo/proposals/__init__.py +12 -0
- apsimo/proposals/engine.py +114 -0
- apsimo/proposals/models.py +207 -0
- apsimo/qualification/__init__.py +1 -0
- apsimo/qualification/cases.py +75 -0
- apsimo/qualification/cli.py +51 -0
- apsimo/qualification/memory_cases.py +209 -0
- apsimo/qualification/records.py +92 -0
- apsimo/qualification/report.py +87 -0
- apsimo/qualification/runner.py +311 -0
- apsimo/qualification/structured_cases.py +131 -0
- apsimo/reasoning/__init__.py +13 -0
- apsimo/reasoning/executor.py +506 -0
- apsimo/reasoning/loop.py +373 -0
- apsimo/reasoning/native_tools/__init__.py +16 -0
- apsimo/reasoning/native_tools/calculate.py +141 -0
- apsimo/reasoning/native_tools/file_ops.py +150 -0
- apsimo/reasoning/native_tools/web_search.py +49 -0
- apsimo/reasoning/tool_policy.py +182 -0
- apsimo/redact/__init__.py +176 -0
- apsimo/repos/__init__.py +5 -0
- apsimo/repos/mirrors.py +204 -0
- apsimo/research/__init__.py +41 -0
- apsimo/research/artifact.py +482 -0
- apsimo/research/gatherer.py +387 -0
- apsimo/research/pipeline.py +513 -0
- apsimo/research/search/__init__.py +7 -0
- apsimo/research/search/base.py +41 -0
- apsimo/research/search/brave.py +59 -0
- apsimo/research/search/cache.py +51 -0
- apsimo/research/search/duckduckgo.py +103 -0
- apsimo/research/search/orchestrator.py +119 -0
- apsimo/research/search/serpapi.py +59 -0
- apsimo/research/search/tavily.py +59 -0
- apsimo/research/synthesizer.py +309 -0
- apsimo/router/__init__.py +30 -0
- apsimo/router/complexity_scorer.py +148 -0
- apsimo/router/endpoints.py +153 -0
- apsimo/router/fallback.py +58 -0
- apsimo/router/functions.py +243 -0
- apsimo/router/native_policy.py +52 -0
- apsimo/router/router.py +762 -0
- apsimo/router/self_learning.py +174 -0
- apsimo/router/tiers.py +677 -0
- apsimo/sandbox/__init__.py +21 -0
- apsimo/sandbox/backend.py +195 -0
- apsimo/sandbox/manager.py +173 -0
- apsimo/scope_bounds.py +7 -0
- apsimo/secrets/__init__.py +6 -0
- apsimo/secrets/backends/__init__.py +8 -0
- apsimo/secrets/backends/base.py +42 -0
- apsimo/secrets/backends/env.py +110 -0
- apsimo/secrets/backends/keyring.py +72 -0
- apsimo/secrets/backends/onepassword.py +232 -0
- apsimo/secrets/cli.py +191 -0
- apsimo/secrets/manager.py +160 -0
- apsimo/secrets/migration.py +101 -0
- apsimo/secrets/types.py +98 -0
- apsimo/seed.py +41 -0
- apsimo/self_model/__init__.py +37 -0
- apsimo/self_model/appraisals.py +673 -0
- apsimo/self_model/benchmark.py +1314 -0
- apsimo/self_model/brief.py +40 -0
- apsimo/self_model/event_concerns.py +1128 -0
- apsimo/self_model/execution_forecasts.py +353 -0
- apsimo/self_model/expectations.py +1595 -0
- apsimo/self_model/experiments.py +1150 -0
- apsimo/self_model/journal.py +148 -0
- apsimo/self_model/judgments.py +705 -0
- apsimo/self_model/native_outcomes.py +55 -0
- apsimo/self_model/params.py +220 -0
- apsimo/self_model/perspective.py +246 -0
- apsimo/self_model/reconcile.py +183 -0
- apsimo/self_model/reply_forecasts.py +381 -0
- apsimo/self_model/runtime_forecasts.py +296 -0
- apsimo/self_model/runtime_models.py +67 -0
- apsimo/self_model/settlement.py +207 -0
- apsimo/self_model/situation.py +1731 -0
- apsimo/self_model/store.py +883 -0
- apsimo/self_model/supervised.py +137 -0
- apsimo/self_model/thinker.py +99 -0
- apsimo/self_model/trust.py +388 -0
- apsimo/self_model/workspace.py +2388 -0
- apsimo/server.py +4197 -0
- apsimo/services/__init__.py +1 -0
- apsimo/services/agent_bridge.py +474 -0
- apsimo/services/initiative_executor.py +914 -0
- apsimo/services/instance.py +297 -0
- apsimo/sessions/__init__.py +22 -0
- apsimo/sessions/config.py +13 -0
- apsimo/sessions/context_loader.py +88 -0
- apsimo/sessions/federation_session.py +75 -0
- apsimo/sessions/isolated_session.py +98 -0
- apsimo/sessions/reports.py +84 -0
- apsimo/sessions/store.py +148 -0
- apsimo/setup.py +2818 -0
- apsimo/setup_hermes.py +879 -0
- apsimo/setup_local_work.py +218 -0
- apsimo/setup_native_goals.py +134 -0
- apsimo/setup_native_reviews.py +115 -0
- apsimo/skills/__init__.py +10 -0
- apsimo/skills/base.py +108 -0
- apsimo/skills/budget.py +28 -0
- apsimo/skills/executor.py +493 -0
- apsimo/skills/executors/__init__.py +1 -0
- apsimo/skills/executors/behavioral_correction.py +75 -0
- apsimo/skills/executors/capability_gap.py +38 -0
- apsimo/skills/executors/data_quality.py +163 -0
- apsimo/skills/executors/knowledge_acquisition.py +41 -0
- apsimo/skills/executors/operational_hygiene.py +185 -0
- apsimo/skills/executors/subsystem_health.py +169 -0
- apsimo/skills/hermes_export.py +431 -0
- apsimo/skills/index.py +123 -0
- apsimo/skills/learning/__init__.py +21 -0
- apsimo/skills/learning/novelty_detector.py +206 -0
- apsimo/skills/learning/pattern_extractor.py +199 -0
- apsimo/skills/learning/triggers.py +159 -0
- apsimo/skills/loader.py +246 -0
- apsimo/skills/migrations/002_progressive_loading.sql +6 -0
- apsimo/skills/migrations/backfill_triggers.py +20 -0
- apsimo/skills/models.py +202 -0
- apsimo/skills/packager.py +128 -0
- apsimo/skills/protocols.py +70 -0
- apsimo/skills/registry.py +191 -0
- apsimo/skills/runtime.py +58 -0
- apsimo/skills/sandbox_runner.py +229 -0
- apsimo/skills/scheduler.py +129 -0
- apsimo/skills/schema.py +79 -0
- apsimo/skills/security/__init__.py +12 -0
- apsimo/skills/security/guards.py +53 -0
- apsimo/skills/security/scanner.py +223 -0
- apsimo/skills_memory/__init__.py +26 -0
- apsimo/skills_memory/distill.py +159 -0
- apsimo/skills_memory/models.py +85 -0
- apsimo/skills_memory/retrieve.py +62 -0
- apsimo/skills_memory/store.py +172 -0
- apsimo/surprise/__init__.py +6 -0
- apsimo/surprise/accumulation.py +57 -0
- apsimo/surprise/scorer.py +102 -0
- apsimo/surprise/store.py +203 -0
- apsimo/task_queue/__init__.py +69 -0
- apsimo/task_queue/action_receipts.py +148 -0
- apsimo/task_queue/approval_relay_canary.py +108 -0
- apsimo/task_queue/config.py +85 -0
- apsimo/task_queue/contract.py +361 -0
- apsimo/task_queue/events.py +130 -0
- apsimo/task_queue/governor.py +1031 -0
- apsimo/task_queue/handlers/__init__.py +16 -0
- apsimo/task_queue/handlers/base.py +37 -0
- apsimo/task_queue/handlers/inference.py +640 -0
- apsimo/task_queue/handlers/monitoring.py +116 -0
- apsimo/task_queue/handlers/registry.py +75 -0
- apsimo/task_queue/handlers/subtask_handler.py +173 -0
- apsimo/task_queue/handlers/system_maintenance.py +147 -0
- apsimo/task_queue/mesh_integration.py +111 -0
- apsimo/task_queue/models.py +317 -0
- apsimo/task_queue/queue_manager.py +8286 -0
- apsimo/task_queue/routing.py +287 -0
- apsimo/task_queue/scheduler.py +252 -0
- apsimo/task_queue/schema.sql +197 -0
- apsimo/task_queue/work_control.py +342 -0
- apsimo/task_queue/worker.py +993 -0
- apsimo/telemetry.py +145 -0
- apsimo/tom/__init__.py +6 -0
- apsimo/tom/affect.py +387 -0
- apsimo/tom/approvals.py +171 -0
- apsimo/tom/arcs.py +896 -0
- apsimo/tom/asymmetry.py +131 -0
- apsimo/tom/eligibility.py +248 -0
- apsimo/tom/engagement.py +214 -0
- apsimo/tom/exposure.py +214 -0
- apsimo/tom/extractor.py +306 -0
- apsimo/tom/fact_adapters.py +144 -0
- apsimo/tom/facts.py +326 -0
- apsimo/tom/integration.py +592 -0
- apsimo/tom/leveled.py +118 -0
- apsimo/tom/levels.py +247 -0
- apsimo/tom/recipient_audit.py +995 -0
- apsimo/tom/recipient_simulator.py +593 -0
- apsimo/tom/source_lineage.py +93 -0
- apsimo/tom/tom2.py +277 -0
- apsimo/tom/visibility.py +559 -0
- apsimo/tom/visibility_store.py +414 -0
- apsimo/tools/__init__.py +0 -0
- apsimo/tools/definitions.py +740 -0
- apsimo/tools/handlers.py +943 -0
- apsimo/toolsmith/__init__.py +26 -0
- apsimo/toolsmith/authority.py +166 -0
- apsimo/toolsmith/engine.py +559 -0
- apsimo/toolsmith/integrity.py +100 -0
- apsimo/toolsmith/miner.py +145 -0
- apsimo/toolsmith/policy.py +110 -0
- apsimo/toolsmith/registry.py +635 -0
- apsimo/turns/__init__.py +17 -0
- apsimo/turns/audio.py +134 -0
- apsimo/turns/documents.py +235 -0
- apsimo/turns/executions.py +486 -0
- apsimo/turns/hermes_history.py +245 -0
- apsimo/turns/hermes_kanban.py +268 -0
- apsimo/turns/hermes_work.py +96 -0
- apsimo/turns/idempotency.py +752 -0
- apsimo/turns/local_work.py +115 -0
- apsimo/turns/media.py +581 -0
- apsimo/turns/reported_workers.py +196 -0
- apsimo/turns/source_annotations.py +283 -0
- apsimo/turns/source_attribution.py +154 -0
- apsimo/turns/source_read.py +351 -0
- apsimo/turns/source_vectors.py +263 -0
- apsimo/turns/video.py +210 -0
- apsimo/util/autonomy_preset.py +220 -0
- apsimo/util/instance.py +92 -0
- apsimo/util/model_output.py +25 -0
- apsimo/util/quiet_hours.py +27 -0
- apsimo/util/session_safety.py +37 -0
- apsimo/util/temporal.py +343 -0
- apsimo/vector/__init__.py +75 -0
- apsimo/vector/backfill.py +171 -0
- apsimo/vector/caption.py +114 -0
- apsimo/vector/collections.py +51 -0
- apsimo/vector/config.py +102 -0
- apsimo/vector/embedder.py +670 -0
- apsimo/vector/image_preprocess.py +406 -0
- apsimo/vector/image_store.py +296 -0
- apsimo/vector/indexes.py +162 -0
- apsimo/vector/migrate.py +334 -0
- apsimo/vector/multimodal_provider.py +417 -0
- apsimo/vector/multimodal_types.py +87 -0
- apsimo/vector/openai_provider.py +119 -0
- apsimo/vector/query.py +49 -0
- apsimo/vector/reranker.py +565 -0
- apsimo/vector/safety_image.py +159 -0
- apsimo/vector/scanner.py +197 -0
- apsimo/vector/setup.py +289 -0
- apsimo/vector/store.py +533 -0
- apsimo/vector/tiers.py +263 -0
- apsimo/work_orders.py +925 -0
- apsimo/workers/__init__.py +21 -0
- apsimo/workers/agent_bridge.py +640 -0
- apsimo/workers/colony_worker.py +382 -0
- apsimo/workers/queue_worker.py +441 -0
- apsimo/workers/skills_sync.py +152 -0
- apsimo/world_model/__init__.py +71 -0
- apsimo/world_model/causal_maintenance.py +131 -0
- apsimo/world_model/causal_policy.py +43 -0
- apsimo/world_model/causal_query.py +125 -0
- apsimo/world_model/confidence.py +54 -0
- apsimo/world_model/config.py +64 -0
- apsimo/world_model/constants.py +97 -0
- apsimo/world_model/entities.py +145 -0
- apsimo/world_model/expectation_resolvers.py +177 -0
- apsimo/world_model/extraction/__init__.py +7 -0
- apsimo/world_model/extraction/base.py +62 -0
- apsimo/world_model/extraction/conversation_extractor.py +262 -0
- apsimo/world_model/extraction/detector.py +74 -0
- apsimo/world_model/extraction/document_extractor.py +78 -0
- apsimo/world_model/extraction/formats/__init__.py +24 -0
- apsimo/world_model/extraction/formats/csv_fmt.py +68 -0
- apsimo/world_model/extraction/formats/html_fmt.py +72 -0
- apsimo/world_model/extraction/formats/json_fmt.py +68 -0
- apsimo/world_model/extraction/formats/pdf.py +43 -0
- apsimo/world_model/extraction/formats/text.py +27 -0
- apsimo/world_model/extraction/llm_extractor.py +164 -0
- apsimo/world_model/extraction/pipeline.py +73 -0
- apsimo/world_model/integrations/__init__.py +5 -0
- apsimo/world_model/integrations/mind_model_bridge.py +115 -0
- apsimo/world_model/integrations/social_intel_bridge.py +120 -0
- apsimo/world_model/jobs/__init__.py +4 -0
- apsimo/world_model/jobs/extraction_job.py +168 -0
- apsimo/world_model/llm_extract.py +572 -0
- apsimo/world_model/neo4j/__init__.py +5 -0
- apsimo/world_model/neo4j/backend.py +654 -0
- apsimo/world_model/observations.py +155 -0
- apsimo/world_model/populator.py +307 -0
- apsimo/world_model/postgres/__init__.py +1 -0
- apsimo/world_model/postgres/backend.py +683 -0
- apsimo/world_model/relationships.py +25 -0
- apsimo/world_model/resolution/__init__.py +13 -0
- apsimo/world_model/resolution/entity_resolver.py +232 -0
- apsimo/world_model/resolution/merge_audit.py +16 -0
- apsimo/world_model/resolution/merge_workflow.py +117 -0
- apsimo/world_model/source_reports.py +121 -0
- apsimo/world_model/sqlite/__init__.py +4 -0
- apsimo/world_model/sqlite/backend.py +855 -0
- apsimo/world_model/sqlite/schema.sql +132 -0
- apsimo/world_model/store.py +545 -0
- apsimo-1.3.0.dist-info/METADATA +78 -0
- apsimo-1.3.0.dist-info/RECORD +614 -0
- apsimo-1.3.0.dist-info/WHEEL +5 -0
- apsimo-1.3.0.dist-info/entry_points.txt +11 -0
- apsimo-1.3.0.dist-info/licenses/LICENSE +21 -0
- apsimo-1.3.0.dist-info/top_level.txt +2 -0
- colony_sidecar/__init__.py +4 -0
|
@@ -0,0 +1,169 @@
|
|
|
1
|
+
"""Structure-aware text chunking.
|
|
2
|
+
|
|
3
|
+
Splits text along natural boundaries — fenced code blocks stay atomic,
|
|
4
|
+
then markdown headings, then blank-line paragraphs — and greedily packs
|
|
5
|
+
blocks into chunks near a target token size, with configurable overlap
|
|
6
|
+
between consecutive chunks so facts straddling a boundary survive
|
|
7
|
+
retrieval.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass
|
|
14
|
+
|
|
15
|
+
from apsimo.contextgate.estimate import estimate_tokens
|
|
16
|
+
|
|
17
|
+
__all__ = ["Chunk", "chunk_text"]
|
|
18
|
+
|
|
19
|
+
_FENCE_RE = re.compile(r"^(```|~~~)", re.MULTILINE)
|
|
20
|
+
_HEADING_RE = re.compile(r"^#{1,6}\s", re.MULTILINE)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass
|
|
24
|
+
class Chunk:
|
|
25
|
+
"""One packed chunk of a source text."""
|
|
26
|
+
|
|
27
|
+
index: int # 0-based position in document order
|
|
28
|
+
text: str # chunk content (includes any overlap prefix)
|
|
29
|
+
start: int # char offset of the core (non-overlap) content
|
|
30
|
+
end: int # char offset one past the core content
|
|
31
|
+
tokens: int = 0 # estimated tokens of ``text``
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _split_blocks(text: str) -> list[tuple[int, str]]:
|
|
35
|
+
"""Split *text* into (offset, block) pairs along structural boundaries.
|
|
36
|
+
|
|
37
|
+
Fenced code blocks are kept whole. Outside fences, headings start new
|
|
38
|
+
blocks and blank lines separate paragraphs.
|
|
39
|
+
"""
|
|
40
|
+
blocks: list[tuple[int, str]] = []
|
|
41
|
+
pos = 0
|
|
42
|
+
n = len(text)
|
|
43
|
+
|
|
44
|
+
while pos < n:
|
|
45
|
+
fence = _FENCE_RE.search(text, pos)
|
|
46
|
+
prose_end = fence.start() if fence else n
|
|
47
|
+
|
|
48
|
+
# Prose region: split on blank lines and headings
|
|
49
|
+
prose = text[pos:prose_end]
|
|
50
|
+
if prose.strip():
|
|
51
|
+
offset = pos
|
|
52
|
+
# Insert split points before headings so each heading opens a block
|
|
53
|
+
paragraphs = re.split(r"(\n\s*\n)", prose)
|
|
54
|
+
cursor = 0
|
|
55
|
+
for part in paragraphs:
|
|
56
|
+
if part.strip() and not re.fullmatch(r"\n\s*\n", part):
|
|
57
|
+
# Further split on headings inside the paragraph run
|
|
58
|
+
last = 0
|
|
59
|
+
for m in _HEADING_RE.finditer(part):
|
|
60
|
+
if m.start() > last and part[last:m.start()].strip():
|
|
61
|
+
blocks.append((offset + cursor + last, part[last:m.start()]))
|
|
62
|
+
last = m.start()
|
|
63
|
+
if part[last:].strip():
|
|
64
|
+
blocks.append((offset + cursor + last, part[last:]))
|
|
65
|
+
cursor += len(part)
|
|
66
|
+
|
|
67
|
+
if fence is None:
|
|
68
|
+
break
|
|
69
|
+
|
|
70
|
+
# Fenced block: find the closing fence of the same kind
|
|
71
|
+
marker = fence.group(1)
|
|
72
|
+
close = text.find("\n" + marker, fence.end())
|
|
73
|
+
if close == -1:
|
|
74
|
+
blocks.append((fence.start(), text[fence.start():]))
|
|
75
|
+
break
|
|
76
|
+
# Include through the end of the closing-fence line
|
|
77
|
+
fend = text.find("\n", close + 1 + len(marker))
|
|
78
|
+
fend = n if fend == -1 else fend + 1
|
|
79
|
+
blocks.append((fence.start(), text[fence.start():fend]))
|
|
80
|
+
pos = fend
|
|
81
|
+
|
|
82
|
+
return blocks
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
def _hard_split(offset: int, block: str, target_tokens: int) -> list[tuple[int, str]]:
|
|
86
|
+
"""Split an oversized block on sentence/newline boundaries, then windows."""
|
|
87
|
+
limit_chars = max(1, target_tokens * 4)
|
|
88
|
+
pieces: list[tuple[int, str]] = []
|
|
89
|
+
cursor = 0
|
|
90
|
+
n = len(block)
|
|
91
|
+
while cursor < n:
|
|
92
|
+
end = min(cursor + limit_chars, n)
|
|
93
|
+
if end < n:
|
|
94
|
+
# Back up to the last sentence end or newline in the window
|
|
95
|
+
window = block[cursor:end]
|
|
96
|
+
cut = max(window.rfind(". "), window.rfind("\n"))
|
|
97
|
+
if cut > limit_chars // 4:
|
|
98
|
+
end = cursor + cut + 1
|
|
99
|
+
pieces.append((offset + cursor, block[cursor:end]))
|
|
100
|
+
cursor = end
|
|
101
|
+
return pieces
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def chunk_text(
|
|
105
|
+
text: str,
|
|
106
|
+
target_tokens: int = 1024,
|
|
107
|
+
overlap_tokens: int = 128,
|
|
108
|
+
) -> list[Chunk]:
|
|
109
|
+
"""Chunk *text* into ~*target_tokens* pieces along structural boundaries.
|
|
110
|
+
|
|
111
|
+
Consecutive chunks share an overlap of roughly *overlap_tokens* (the
|
|
112
|
+
tail of the previous chunk is prepended to the next), so information
|
|
113
|
+
spanning a boundary remains retrievable. ``start``/``end`` offsets
|
|
114
|
+
always refer to the core (non-overlap) content.
|
|
115
|
+
"""
|
|
116
|
+
if not text.strip():
|
|
117
|
+
return []
|
|
118
|
+
|
|
119
|
+
blocks: list[tuple[int, str]] = []
|
|
120
|
+
for offset, block in _split_blocks(text):
|
|
121
|
+
if estimate_tokens(block) > target_tokens:
|
|
122
|
+
blocks.extend(_hard_split(offset, block, target_tokens))
|
|
123
|
+
else:
|
|
124
|
+
blocks.append((offset, block))
|
|
125
|
+
|
|
126
|
+
chunks: list[Chunk] = []
|
|
127
|
+
cur_parts: list[tuple[int, str]] = []
|
|
128
|
+
cur_tokens = 0
|
|
129
|
+
|
|
130
|
+
def _flush() -> None:
|
|
131
|
+
nonlocal cur_parts, cur_tokens
|
|
132
|
+
if not cur_parts:
|
|
133
|
+
return
|
|
134
|
+
start = cur_parts[0][0]
|
|
135
|
+
last_off, last_text = cur_parts[-1]
|
|
136
|
+
end = last_off + len(last_text)
|
|
137
|
+
core = text[start:end]
|
|
138
|
+
prefix = ""
|
|
139
|
+
if chunks and overlap_tokens > 0:
|
|
140
|
+
prev = chunks[-1]
|
|
141
|
+
tail_chars = overlap_tokens * 4
|
|
142
|
+
tail = text[max(prev.start, prev.end - tail_chars):prev.end]
|
|
143
|
+
# Cut at a word boundary so the overlap reads cleanly
|
|
144
|
+
sp = tail.find(" ")
|
|
145
|
+
if 0 <= sp < len(tail) - 1:
|
|
146
|
+
tail = tail[sp + 1:]
|
|
147
|
+
prefix = tail + "\n"
|
|
148
|
+
body = prefix + core
|
|
149
|
+
chunks.append(
|
|
150
|
+
Chunk(
|
|
151
|
+
index=len(chunks),
|
|
152
|
+
text=body,
|
|
153
|
+
start=start,
|
|
154
|
+
end=end,
|
|
155
|
+
tokens=estimate_tokens(body),
|
|
156
|
+
)
|
|
157
|
+
)
|
|
158
|
+
cur_parts = []
|
|
159
|
+
cur_tokens = 0
|
|
160
|
+
|
|
161
|
+
for offset, block in blocks:
|
|
162
|
+
btok = estimate_tokens(block)
|
|
163
|
+
if cur_parts and cur_tokens + btok > target_tokens:
|
|
164
|
+
_flush()
|
|
165
|
+
cur_parts.append((offset, block))
|
|
166
|
+
cur_tokens += btok
|
|
167
|
+
_flush()
|
|
168
|
+
|
|
169
|
+
return chunks
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Cheap token estimation — no tokenizer dependency.
|
|
2
|
+
|
|
3
|
+
The gate only needs estimates good to ~±20%; the decision headroom
|
|
4
|
+
(default 0.8) absorbs the imprecision. English prose runs ~4 chars per
|
|
5
|
+
token; code and symbol-dense text runs denser (~3 chars per token), so a
|
|
6
|
+
crude density probe adjusts the ratio.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import math
|
|
12
|
+
import os
|
|
13
|
+
|
|
14
|
+
__all__ = ["estimate_tokens"]
|
|
15
|
+
|
|
16
|
+
_DEFAULT_CHARS_PER_TOKEN = 4.0
|
|
17
|
+
_CODE_CHARS_PER_TOKEN = 3.0
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def _chars_per_token() -> float:
|
|
21
|
+
try:
|
|
22
|
+
v = float(os.environ.get("COLONY_CONTEXT_CHARS_PER_TOKEN", ""))
|
|
23
|
+
if v > 0:
|
|
24
|
+
return v
|
|
25
|
+
except ValueError:
|
|
26
|
+
pass
|
|
27
|
+
return _DEFAULT_CHARS_PER_TOKEN
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _looks_dense(text: str, sample_limit: int = 20000) -> bool:
|
|
31
|
+
"""True when the text is symbol/whitespace-dense (code, logs, JSON)."""
|
|
32
|
+
sample = text[:sample_limit]
|
|
33
|
+
if not sample:
|
|
34
|
+
return False
|
|
35
|
+
symbolish = sum(
|
|
36
|
+
1 for c in sample if not (c.isalpha() or c in " .,;:'\"!?-")
|
|
37
|
+
)
|
|
38
|
+
return symbolish / len(sample) > 0.25
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def estimate_tokens(text: str) -> int:
|
|
42
|
+
"""Estimate the token count of *text*.
|
|
43
|
+
|
|
44
|
+
Uses a chars-per-token heuristic (configurable via
|
|
45
|
+
``COLONY_CONTEXT_CHARS_PER_TOKEN``), with a denser ratio for
|
|
46
|
+
code-like input. Deliberately dependency-free; accuracy within
|
|
47
|
+
~±20% is sufficient for gating decisions.
|
|
48
|
+
"""
|
|
49
|
+
if not text:
|
|
50
|
+
return 0
|
|
51
|
+
cpt = _chars_per_token()
|
|
52
|
+
if _looks_dense(text):
|
|
53
|
+
cpt = min(cpt, _CODE_CHARS_PER_TOKEN)
|
|
54
|
+
return math.ceil(len(text) / cpt)
|
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
"""Gate decision + context preparation service.
|
|
2
|
+
|
|
3
|
+
``decide()`` picks a strategy from (estimated size, budget, task kind);
|
|
4
|
+
``prepare_context()`` executes it and returns the final context text plus
|
|
5
|
+
metadata. See the package docstring for the full model.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import logging
|
|
11
|
+
import os
|
|
12
|
+
import re
|
|
13
|
+
from dataclasses import dataclass, field
|
|
14
|
+
from enum import Enum
|
|
15
|
+
from typing import Awaitable, Callable, Optional
|
|
16
|
+
|
|
17
|
+
from apsimo.contextgate.chunker import Chunk, chunk_text
|
|
18
|
+
from apsimo.contextgate.estimate import estimate_tokens
|
|
19
|
+
from apsimo.contextgate.retrieve import EmbedFn, rank_chunks
|
|
20
|
+
|
|
21
|
+
logger = logging.getLogger(__name__)
|
|
22
|
+
|
|
23
|
+
__all__ = [
|
|
24
|
+
"GateConfig",
|
|
25
|
+
"GateDecision",
|
|
26
|
+
"PreparedContext",
|
|
27
|
+
"classify_task",
|
|
28
|
+
"decide",
|
|
29
|
+
"prepare_context",
|
|
30
|
+
]
|
|
31
|
+
|
|
32
|
+
SummarizeFn = Callable[[str, str], Awaitable[str]] # (chunk_text, query) -> summary
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class GateDecision(str, Enum):
|
|
36
|
+
PASS_THROUGH = "pass_through"
|
|
37
|
+
RETRIEVE = "retrieve"
|
|
38
|
+
MAP_REDUCE = "map_reduce"
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass
|
|
42
|
+
class GateConfig:
|
|
43
|
+
"""Tunables for the context gate. All values env-overridable."""
|
|
44
|
+
|
|
45
|
+
mode: str = "auto" # auto | on | off
|
|
46
|
+
headroom: float = 0.8 # gate when est > budget * headroom
|
|
47
|
+
default_budget_tokens: int = 0 # used when caller/tier give none (0 = don't gate)
|
|
48
|
+
chunk_tokens: int = 1024
|
|
49
|
+
overlap_tokens: int = 128
|
|
50
|
+
min_score: float = 0.05 # drop retrieval chunks scoring below this
|
|
51
|
+
|
|
52
|
+
@classmethod
|
|
53
|
+
def from_env(cls) -> "GateConfig":
|
|
54
|
+
cfg = cls()
|
|
55
|
+
mode = os.environ.get("COLONY_CONTEXT_GATE", "").strip().lower()
|
|
56
|
+
if mode in ("auto", "on", "off"):
|
|
57
|
+
cfg.mode = mode
|
|
58
|
+
for attr, env, cast in (
|
|
59
|
+
("headroom", "COLONY_CONTEXT_GATE_HEADROOM", float),
|
|
60
|
+
("default_budget_tokens", "COLONY_CONTEXT_GATE_BUDGET", int),
|
|
61
|
+
("chunk_tokens", "COLONY_CONTEXT_CHUNK_TOKENS", int),
|
|
62
|
+
("overlap_tokens", "COLONY_CONTEXT_OVERLAP_TOKENS", int),
|
|
63
|
+
):
|
|
64
|
+
raw = os.environ.get(env, "")
|
|
65
|
+
if raw:
|
|
66
|
+
try:
|
|
67
|
+
setattr(cfg, attr, cast(raw))
|
|
68
|
+
except ValueError:
|
|
69
|
+
logger.warning("Invalid %s=%r — using default", env, raw)
|
|
70
|
+
return cfg
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
@dataclass
|
|
74
|
+
class PreparedContext:
|
|
75
|
+
"""Result of ``prepare_context``."""
|
|
76
|
+
|
|
77
|
+
text: str
|
|
78
|
+
decision: GateDecision
|
|
79
|
+
est_tokens_in: int
|
|
80
|
+
est_tokens_out: int
|
|
81
|
+
budget_tokens: int
|
|
82
|
+
chunks_total: int = 0
|
|
83
|
+
chunks_used: int = 0
|
|
84
|
+
coverage: float = 1.0 # fraction of source chars represented in output
|
|
85
|
+
task_kind: str = ""
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
# ---------------------------------------------------------------------------
|
|
89
|
+
# Task classification — the smarter-than-size signal
|
|
90
|
+
# ---------------------------------------------------------------------------
|
|
91
|
+
|
|
92
|
+
_HOLISTIC_RE = re.compile(
|
|
93
|
+
r"\b(summar[iy][sz]e|overview|review|rewrite|rephrase|translate|proofread|"
|
|
94
|
+
r"critique|tl;?dr|digest|condense|outline|abstract)\b",
|
|
95
|
+
re.IGNORECASE,
|
|
96
|
+
)
|
|
97
|
+
_RETRIEVAL_RE = re.compile(
|
|
98
|
+
r"(\?|\b(what|when|where|who|whom|whose|which|why|how|find|look ?up|search|"
|
|
99
|
+
r"locate|extract|quote|list all|did|does|do|is|are|was|were)\b)",
|
|
100
|
+
re.IGNORECASE,
|
|
101
|
+
)
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def classify_task(query: str) -> str:
|
|
105
|
+
"""Classify the caller's intent as ``retrieval`` or ``holistic``.
|
|
106
|
+
|
|
107
|
+
Retrieval tasks (needle questions, lookups) benefit from ranked-chunk
|
|
108
|
+
RAG; holistic tasks (summarize/review the whole document) need
|
|
109
|
+
coverage of everything and get map-reduce instead. Callers that know
|
|
110
|
+
their intent should pass ``task_kind`` explicitly — this heuristic is
|
|
111
|
+
only the fallback.
|
|
112
|
+
"""
|
|
113
|
+
q = (query or "").strip()
|
|
114
|
+
if not q:
|
|
115
|
+
return "holistic"
|
|
116
|
+
if _HOLISTIC_RE.search(q):
|
|
117
|
+
return "holistic"
|
|
118
|
+
if _RETRIEVAL_RE.search(q):
|
|
119
|
+
return "retrieval"
|
|
120
|
+
# A short, specific query usually names the thing to find.
|
|
121
|
+
return "retrieval" if len(q) < 200 else "holistic"
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
# ---------------------------------------------------------------------------
|
|
125
|
+
# Decision
|
|
126
|
+
# ---------------------------------------------------------------------------
|
|
127
|
+
|
|
128
|
+
def decide(
|
|
129
|
+
est_tokens: int,
|
|
130
|
+
budget_tokens: int,
|
|
131
|
+
query: str = "",
|
|
132
|
+
task_kind: Optional[str] = None,
|
|
133
|
+
config: Optional[GateConfig] = None,
|
|
134
|
+
) -> GateDecision:
|
|
135
|
+
"""Pick a strategy for content of *est_tokens* against *budget_tokens*.
|
|
136
|
+
|
|
137
|
+
``mode=off`` or an unknown budget (0) always passes through; otherwise
|
|
138
|
+
content within ``budget * headroom`` passes through, and oversized
|
|
139
|
+
content is routed to RETRIEVE or MAP_REDUCE by task kind.
|
|
140
|
+
"""
|
|
141
|
+
cfg = config or GateConfig.from_env()
|
|
142
|
+
if cfg.mode == "off":
|
|
143
|
+
return GateDecision.PASS_THROUGH
|
|
144
|
+
budget = budget_tokens or cfg.default_budget_tokens
|
|
145
|
+
if budget <= 0:
|
|
146
|
+
return GateDecision.PASS_THROUGH
|
|
147
|
+
if est_tokens <= budget * cfg.headroom:
|
|
148
|
+
return GateDecision.PASS_THROUGH
|
|
149
|
+
kind = task_kind if task_kind in ("retrieval", "holistic") else classify_task(query)
|
|
150
|
+
return GateDecision.RETRIEVE if kind == "retrieval" else GateDecision.MAP_REDUCE
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# ---------------------------------------------------------------------------
|
|
154
|
+
# Preparation
|
|
155
|
+
# ---------------------------------------------------------------------------
|
|
156
|
+
|
|
157
|
+
def _assemble(
|
|
158
|
+
selected: list[Chunk],
|
|
159
|
+
total: int,
|
|
160
|
+
source_chars: int,
|
|
161
|
+
label: str,
|
|
162
|
+
) -> tuple[str, float]:
|
|
163
|
+
"""Join *selected* chunks (document order) with provenance markers."""
|
|
164
|
+
selected = sorted(selected, key=lambda c: c.index)
|
|
165
|
+
covered = sum(c.end - c.start for c in selected)
|
|
166
|
+
coverage = min(1.0, covered / source_chars) if source_chars else 1.0
|
|
167
|
+
parts = [
|
|
168
|
+
f"[context gate: {label} — {len(selected)} of {total} chunks, "
|
|
169
|
+
f"~{coverage:.0%} of the source shown]"
|
|
170
|
+
]
|
|
171
|
+
for c in selected:
|
|
172
|
+
parts.append(f"--- chunk {c.index + 1}/{total} (chars {c.start}-{c.end}) ---")
|
|
173
|
+
parts.append(c.text)
|
|
174
|
+
return "\n".join(parts), coverage
|
|
175
|
+
|
|
176
|
+
|
|
177
|
+
def _pack_to_budget(
|
|
178
|
+
ranked: list[tuple[Chunk, float]],
|
|
179
|
+
budget_tokens: int,
|
|
180
|
+
min_score: float,
|
|
181
|
+
) -> list[Chunk]:
|
|
182
|
+
selected: list[Chunk] = []
|
|
183
|
+
used = 0
|
|
184
|
+
for chunk, score in ranked:
|
|
185
|
+
if score < min_score and selected:
|
|
186
|
+
break
|
|
187
|
+
if used + chunk.tokens > budget_tokens:
|
|
188
|
+
if not selected:
|
|
189
|
+
selected.append(chunk) # always include at least the best chunk
|
|
190
|
+
break
|
|
191
|
+
selected.append(chunk)
|
|
192
|
+
used += chunk.tokens
|
|
193
|
+
return selected
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _sample_evenly(chunks: list[Chunk], budget_tokens: int) -> list[Chunk]:
|
|
197
|
+
"""Pick evenly-spaced chunks so the selection spans the whole source."""
|
|
198
|
+
if not chunks:
|
|
199
|
+
return []
|
|
200
|
+
avg = max(1, sum(c.tokens for c in chunks) // len(chunks))
|
|
201
|
+
k = max(1, min(len(chunks), budget_tokens // avg))
|
|
202
|
+
if k >= len(chunks):
|
|
203
|
+
return list(chunks)
|
|
204
|
+
step = len(chunks) / k
|
|
205
|
+
picked = []
|
|
206
|
+
used = 0
|
|
207
|
+
for i in range(k):
|
|
208
|
+
c = chunks[int(i * step)]
|
|
209
|
+
if used + c.tokens > budget_tokens and picked:
|
|
210
|
+
break
|
|
211
|
+
picked.append(c)
|
|
212
|
+
used += c.tokens
|
|
213
|
+
return picked
|
|
214
|
+
|
|
215
|
+
|
|
216
|
+
async def prepare_context(
|
|
217
|
+
content: str,
|
|
218
|
+
query: str = "",
|
|
219
|
+
budget_tokens: int = 0,
|
|
220
|
+
task_kind: Optional[str] = None,
|
|
221
|
+
config: Optional[GateConfig] = None,
|
|
222
|
+
embed_fn: Optional[EmbedFn] = None,
|
|
223
|
+
summarize_fn: Optional[SummarizeFn] = None,
|
|
224
|
+
) -> PreparedContext:
|
|
225
|
+
"""Prepare *content* for a model call within *budget_tokens*.
|
|
226
|
+
|
|
227
|
+
Returns the content unchanged when it fits (or gating is off).
|
|
228
|
+
Otherwise chunks it and either retrieves the chunks most relevant to
|
|
229
|
+
*query* (retrieval tasks) or map-reduces via *summarize_fn* /
|
|
230
|
+
coverage-samples (holistic tasks). Never raises on ranking or
|
|
231
|
+
summarization failure — degrades toward coverage sampling.
|
|
232
|
+
"""
|
|
233
|
+
cfg = config or GateConfig.from_env()
|
|
234
|
+
est = estimate_tokens(content)
|
|
235
|
+
budget = budget_tokens or cfg.default_budget_tokens
|
|
236
|
+
decision = decide(est, budget, query, task_kind, cfg)
|
|
237
|
+
kind = task_kind if task_kind in ("retrieval", "holistic") else classify_task(query)
|
|
238
|
+
|
|
239
|
+
if decision == GateDecision.PASS_THROUGH:
|
|
240
|
+
return PreparedContext(
|
|
241
|
+
text=content,
|
|
242
|
+
decision=decision,
|
|
243
|
+
est_tokens_in=est,
|
|
244
|
+
est_tokens_out=est,
|
|
245
|
+
budget_tokens=budget,
|
|
246
|
+
task_kind=kind,
|
|
247
|
+
)
|
|
248
|
+
|
|
249
|
+
chunks = chunk_text(content, cfg.chunk_tokens, cfg.overlap_tokens)
|
|
250
|
+
total = len(chunks)
|
|
251
|
+
effective_budget = max(1, int(budget * cfg.headroom))
|
|
252
|
+
|
|
253
|
+
if decision == GateDecision.RETRIEVE:
|
|
254
|
+
ranked = await rank_chunks(chunks, query, embed_fn)
|
|
255
|
+
selected = _pack_to_budget(ranked, effective_budget, cfg.min_score)
|
|
256
|
+
text, coverage = _assemble(
|
|
257
|
+
selected, total, len(content), "chunks selected for relevance to the query"
|
|
258
|
+
)
|
|
259
|
+
else: # MAP_REDUCE
|
|
260
|
+
if summarize_fn is not None:
|
|
261
|
+
summaries: list[Chunk] = []
|
|
262
|
+
for c in chunks:
|
|
263
|
+
try:
|
|
264
|
+
s = await summarize_fn(c.text, query)
|
|
265
|
+
except Exception:
|
|
266
|
+
logger.warning(
|
|
267
|
+
"summarize_fn failed on chunk %d — using head of chunk",
|
|
268
|
+
c.index, exc_info=True,
|
|
269
|
+
)
|
|
270
|
+
s = c.text[: cfg.chunk_tokens]
|
|
271
|
+
summaries.append(
|
|
272
|
+
Chunk(
|
|
273
|
+
index=c.index,
|
|
274
|
+
text=s,
|
|
275
|
+
start=c.start,
|
|
276
|
+
end=c.end,
|
|
277
|
+
tokens=estimate_tokens(s),
|
|
278
|
+
)
|
|
279
|
+
)
|
|
280
|
+
# Keep summaries within budget (they normally fit; sample if not)
|
|
281
|
+
if sum(s.tokens for s in summaries) > effective_budget:
|
|
282
|
+
summaries = _sample_evenly(summaries, effective_budget)
|
|
283
|
+
text, coverage = _assemble(
|
|
284
|
+
summaries, total, len(content), "per-chunk summaries (map-reduce)"
|
|
285
|
+
)
|
|
286
|
+
else:
|
|
287
|
+
selected = _sample_evenly(chunks, effective_budget)
|
|
288
|
+
text, coverage = _assemble(
|
|
289
|
+
selected, total, len(content), "evenly-spaced coverage sample"
|
|
290
|
+
)
|
|
291
|
+
|
|
292
|
+
prepared = PreparedContext(
|
|
293
|
+
text=text,
|
|
294
|
+
decision=decision,
|
|
295
|
+
est_tokens_in=est,
|
|
296
|
+
est_tokens_out=estimate_tokens(text),
|
|
297
|
+
budget_tokens=budget,
|
|
298
|
+
chunks_total=total,
|
|
299
|
+
chunks_used=text.count("--- chunk "),
|
|
300
|
+
coverage=coverage,
|
|
301
|
+
task_kind=kind,
|
|
302
|
+
)
|
|
303
|
+
logger.info(
|
|
304
|
+
"context gate: %s (%s) %d -> %d est tokens, %d/%d chunks, %.0f%% coverage",
|
|
305
|
+
prepared.decision.value,
|
|
306
|
+
prepared.task_kind,
|
|
307
|
+
prepared.est_tokens_in,
|
|
308
|
+
prepared.est_tokens_out,
|
|
309
|
+
prepared.chunks_used,
|
|
310
|
+
prepared.chunks_total,
|
|
311
|
+
prepared.coverage * 100,
|
|
312
|
+
)
|
|
313
|
+
return prepared
|
|
@@ -0,0 +1,115 @@
|
|
|
1
|
+
"""Chunk ranking for the context gate.
|
|
2
|
+
|
|
3
|
+
Embeddings when the caller provides an ``embed_fn`` (any async
|
|
4
|
+
``list[str] -> list[list[float]]``); otherwise a dependency-free lexical
|
|
5
|
+
TF-IDF cosine ranker, so the gate works even where no embedding stack is
|
|
6
|
+
configured.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import logging
|
|
12
|
+
import math
|
|
13
|
+
import re
|
|
14
|
+
from collections import Counter
|
|
15
|
+
from typing import Awaitable, Callable, Optional, Sequence
|
|
16
|
+
|
|
17
|
+
from apsimo.contextgate.chunker import Chunk
|
|
18
|
+
|
|
19
|
+
logger = logging.getLogger(__name__)
|
|
20
|
+
|
|
21
|
+
__all__ = ["rank_chunks", "lexical_scores"]
|
|
22
|
+
|
|
23
|
+
EmbedFn = Callable[[list[str]], Awaitable[list[list[float]]]]
|
|
24
|
+
|
|
25
|
+
_WORD_RE = re.compile(r"[a-z0-9]+")
|
|
26
|
+
|
|
27
|
+
# Minimal English stopword set — enough to stop function words from
|
|
28
|
+
# dominating TF-IDF; deliberately tiny to stay language-tolerant.
|
|
29
|
+
_STOPWORDS = frozenset(
|
|
30
|
+
"a an and are as at be but by for from has have if in into is it its of on "
|
|
31
|
+
"or that the their there these they this to was were what when where which "
|
|
32
|
+
"who will with you your".split()
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _terms(text: str) -> list[str]:
|
|
37
|
+
return [w for w in _WORD_RE.findall(text.lower()) if w not in _STOPWORDS]
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def lexical_scores(chunks: Sequence[Chunk], query: str) -> list[float]:
|
|
41
|
+
"""TF-IDF cosine similarity of each chunk against *query* (0..1)."""
|
|
42
|
+
q_terms = _terms(query)
|
|
43
|
+
if not q_terms or not chunks:
|
|
44
|
+
return [0.0] * len(chunks)
|
|
45
|
+
|
|
46
|
+
chunk_terms = [_terms(c.text) for c in chunks]
|
|
47
|
+
n = len(chunks)
|
|
48
|
+
df: Counter[str] = Counter()
|
|
49
|
+
for terms in chunk_terms:
|
|
50
|
+
df.update(set(terms))
|
|
51
|
+
|
|
52
|
+
def _idf(term: str) -> float:
|
|
53
|
+
return math.log(1.0 + n / (1.0 + df.get(term, 0)))
|
|
54
|
+
|
|
55
|
+
q_tf = Counter(q_terms)
|
|
56
|
+
q_vec = {t: (1.0 + math.log(c)) * _idf(t) for t, c in q_tf.items()}
|
|
57
|
+
q_norm = math.sqrt(sum(v * v for v in q_vec.values())) or 1.0
|
|
58
|
+
|
|
59
|
+
scores: list[float] = []
|
|
60
|
+
for terms in chunk_terms:
|
|
61
|
+
tf = Counter(terms)
|
|
62
|
+
dot = 0.0
|
|
63
|
+
norm_sq = 0.0
|
|
64
|
+
for t, c in tf.items():
|
|
65
|
+
w = (1.0 + math.log(c)) * _idf(t)
|
|
66
|
+
norm_sq += w * w
|
|
67
|
+
qv = q_vec.get(t)
|
|
68
|
+
if qv:
|
|
69
|
+
dot += w * qv
|
|
70
|
+
norm = math.sqrt(norm_sq) or 1.0
|
|
71
|
+
scores.append(dot / (norm * q_norm))
|
|
72
|
+
return scores
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _cosine(a: Sequence[float], b: Sequence[float]) -> float:
|
|
76
|
+
dot = sum(x * y for x, y in zip(a, b))
|
|
77
|
+
na = math.sqrt(sum(x * x for x in a)) or 1.0
|
|
78
|
+
nb = math.sqrt(sum(y * y for y in b)) or 1.0
|
|
79
|
+
return dot / (na * nb)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
async def rank_chunks(
|
|
83
|
+
chunks: Sequence[Chunk],
|
|
84
|
+
query: str,
|
|
85
|
+
embed_fn: Optional[EmbedFn] = None,
|
|
86
|
+
) -> list[tuple[Chunk, float]]:
|
|
87
|
+
"""Rank *chunks* by relevance to *query*, best first.
|
|
88
|
+
|
|
89
|
+
Uses *embed_fn* (batched: query + chunks in one call) when provided,
|
|
90
|
+
falling back to lexical TF-IDF on any embedding failure so ranking
|
|
91
|
+
never hard-fails.
|
|
92
|
+
"""
|
|
93
|
+
if not chunks:
|
|
94
|
+
return []
|
|
95
|
+
|
|
96
|
+
if embed_fn is not None and query:
|
|
97
|
+
try:
|
|
98
|
+
vectors = await embed_fn([query] + [c.text for c in chunks])
|
|
99
|
+
if len(vectors) == len(chunks) + 1:
|
|
100
|
+
q_vec = vectors[0]
|
|
101
|
+
scored = [
|
|
102
|
+
(c, _cosine(q_vec, v)) for c, v in zip(chunks, vectors[1:])
|
|
103
|
+
]
|
|
104
|
+
scored.sort(key=lambda cs: cs[1], reverse=True)
|
|
105
|
+
return scored
|
|
106
|
+
logger.warning(
|
|
107
|
+
"embed_fn returned %d vectors for %d texts — falling back to lexical",
|
|
108
|
+
len(vectors), len(chunks) + 1,
|
|
109
|
+
)
|
|
110
|
+
except Exception:
|
|
111
|
+
logger.warning("embed_fn failed — falling back to lexical ranking", exc_info=True)
|
|
112
|
+
|
|
113
|
+
scored = list(zip(chunks, lexical_scores(chunks, query)))
|
|
114
|
+
scored.sort(key=lambda cs: cs[1], reverse=True)
|
|
115
|
+
return scored
|
|
@@ -0,0 +1,16 @@
|
|
|
1
|
+
"""Colony proactive delivery system.
|
|
2
|
+
|
|
3
|
+
Bridges the autonomy loop's initiative/insight generation with the gateway's
|
|
4
|
+
messaging adapters so Colony can proactively reach users when it has something
|
|
5
|
+
worth saying.
|
|
6
|
+
|
|
7
|
+
Components:
|
|
8
|
+
- ProactiveDeliveryBridge: Queues and manages pending deliveries
|
|
9
|
+
- RateLimiter: Per-person rate limiting (max 3/day, quiet hours, 2h cooldown)
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from apsimo.delivery.bridge import ProactiveDeliveryBridge
|
|
13
|
+
from apsimo.delivery.rate_limiter import DeliveryRateLimiter
|
|
14
|
+
from apsimo.delivery.channels import ChannelRegistry
|
|
15
|
+
|
|
16
|
+
__all__ = ["ProactiveDeliveryBridge", "DeliveryRateLimiter", "ChannelRegistry"]
|