superlocalmemory 3.7.7 → 3.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/ATTRIBUTION.md +1 -3
- package/CHANGELOG.md +85 -0
- package/README.md +199 -29
- package/package.json +4 -2
- package/plugin/.claude-plugin/plugin.json +2 -2
- package/plugin/CLAUDE.md +8 -8
- package/plugin/agents/slm-governance-advisor.md +80 -0
- package/plugin/agents/slm-loop-runner.md +71 -0
- package/plugin/agents/slm-memory-advisor.md +10 -5
- package/plugin/agents/slm-optimize-advisor.md +9 -3
- package/plugin/commands/slm-loop.md +31 -0
- package/plugin/hooks/hooks.json +79 -0
- package/plugin/requirements.txt +1 -1
- package/plugin/scripts/slm-launch +46 -7
- package/plugin/settings.json +9 -0
- package/plugin/skills/slm-cache/SKILL.md +9 -1
- package/plugin/skills/slm-compress/SKILL.md +8 -1
- package/plugin/skills/slm-governance/SKILL.md +248 -0
- package/plugin/skills/slm-graph/SKILL.md +17 -3
- package/plugin/skills/slm-loop/SKILL.md +99 -0
- package/plugin/skills/slm-mesh/SKILL.md +282 -0
- package/plugin/skills/slm-profile/SKILL.md +148 -0
- package/plugin/skills/slm-recall/SKILL.md +46 -10
- package/plugin/skills/slm-remember/SKILL.md +48 -1
- package/plugin/skills/slm-scope/SKILL.md +176 -0
- package/plugin/skills/slm-session/SKILL.md +24 -1
- package/plugin/skills/slm-status/SKILL.md +18 -1
- package/plugin-src/agents/slm-governance-advisor.md +80 -0
- package/plugin-src/agents/slm-loop-runner.md +71 -0
- package/plugin-src/agents/slm-memory-advisor.md +10 -5
- package/plugin-src/agents/slm-optimize-advisor.md +9 -3
- package/plugin-src/commands/slm-loop.md +31 -0
- package/plugin-src/hooks/hooks.json +79 -0
- package/plugin-src/manifest.json +7 -2
- package/plugin-src/requirements.txt +1 -1
- package/plugin-src/rules/AGENTS.md +57 -18
- package/plugin-src/rules/CLAUDE.md.fragment +8 -8
- package/plugin-src/scripts/slm-launch +46 -7
- package/plugin-src/settings.json +9 -0
- package/plugin-src/skills/slm-cache/SKILL.md +9 -1
- package/plugin-src/skills/slm-compress/SKILL.md +8 -1
- package/plugin-src/skills/slm-governance/SKILL.md +248 -0
- package/plugin-src/skills/slm-graph/SKILL.md +17 -3
- package/plugin-src/skills/slm-loop/SKILL.md +99 -0
- package/plugin-src/skills/slm-mesh/SKILL.md +282 -0
- package/plugin-src/skills/slm-profile/SKILL.md +148 -0
- package/plugin-src/skills/slm-recall/SKILL.md +46 -10
- package/plugin-src/skills/slm-remember/SKILL.md +48 -1
- package/plugin-src/skills/slm-scope/SKILL.md +176 -0
- package/plugin-src/skills/slm-session/SKILL.md +24 -1
- package/plugin-src/skills/slm-status/SKILL.md +18 -1
- package/pyproject.toml +1 -2
- package/scripts/postinstall/validation.js +2 -0
- package/scripts/postinstall-interactive.js +74 -2
- package/src/superlocalmemory/__init__.py +1 -1
- package/src/superlocalmemory/access/__init__.py +3 -0
- package/src/superlocalmemory/access/rbac.py +477 -0
- package/src/superlocalmemory/cli/commands.py +96 -12
- package/src/superlocalmemory/cli/compress_cmd.py +17 -7
- package/src/superlocalmemory/cli/loop_cmd.py +192 -0
- package/src/superlocalmemory/cli/main.py +39 -4
- package/src/superlocalmemory/cli/mesh_cmd.py +38 -0
- package/src/superlocalmemory/cli/optimize_cmd.py +3 -0
- package/src/superlocalmemory/cli/pending_store.py +49 -13
- package/src/superlocalmemory/cli/proxy_cmd.py +4 -0
- package/src/superlocalmemory/cli/scale_engine_cmd.py +6 -0
- package/src/superlocalmemory/cli/setup_wizard.py +22 -13
- package/src/superlocalmemory/compliance/audit.py +6 -0
- package/src/superlocalmemory/compliance/gdpr.py +128 -138
- package/src/superlocalmemory/compliance/retention.py +176 -45
- package/src/superlocalmemory/core/backend_orchestrator.py +5 -43
- package/src/superlocalmemory/core/community_summary.py +267 -0
- package/src/superlocalmemory/core/config.py +216 -3
- package/src/superlocalmemory/core/consolidation_engine.py +95 -22
- package/src/superlocalmemory/core/context_cache.py +61 -18
- package/src/superlocalmemory/core/embedding_worker.py +17 -2
- package/src/superlocalmemory/core/embeddings.py +12 -1
- package/src/superlocalmemory/core/engine.py +17 -1
- package/src/superlocalmemory/core/engine_ingestion.py +29 -0
- package/src/superlocalmemory/core/engine_wiring.py +13 -0
- package/src/superlocalmemory/core/entity_community.py +178 -0
- package/src/superlocalmemory/core/graph_analyzer.py +39 -2
- package/src/superlocalmemory/core/graph_pruner.py +13 -8
- package/src/superlocalmemory/core/key_expander.py +138 -0
- package/src/superlocalmemory/core/maintenance.py +23 -0
- package/src/superlocalmemory/core/modes.py +1 -1
- package/src/superlocalmemory/core/mutations.py +2 -2
- package/src/superlocalmemory/core/pii.py +105 -0
- package/src/superlocalmemory/core/progressive_abstraction.py +208 -0
- package/src/superlocalmemory/core/recall_pipeline.py +2 -0
- package/src/superlocalmemory/core/recall_worker.py +20 -6
- package/src/superlocalmemory/core/scale_engine.py +60 -1
- package/src/superlocalmemory/core/security_primitives.py +40 -2
- package/src/superlocalmemory/core/store_pipeline.py +35 -11
- package/src/superlocalmemory/core/worker_pool.py +21 -6
- package/src/superlocalmemory/encoding/entity_reflexion.py +200 -0
- package/src/superlocalmemory/encoding/entity_resolver.py +34 -24
- package/src/superlocalmemory/encoding/fact_extractor.py +26 -1
- package/src/superlocalmemory/encoding/temporal_validator.py +64 -1
- package/src/superlocalmemory/evolution/evolution_store.py +122 -45
- package/src/superlocalmemory/evolution/llm_dispatch.py +12 -1
- package/src/superlocalmemory/evolution/model_selection.py +160 -0
- package/src/superlocalmemory/evolution/mutation_generator.py +16 -0
- package/src/superlocalmemory/evolution/skill_evolver.py +127 -42
- package/src/superlocalmemory/evolution/triggers.py +22 -13
- package/src/superlocalmemory/graph/cozo_backend.py +43 -20
- package/src/superlocalmemory/hooks/adapter_base.py +5 -1
- package/src/superlocalmemory/hooks/auto_recall.py +13 -1
- package/src/superlocalmemory/hooks/claude_code_hooks.py +11 -0
- package/src/superlocalmemory/hooks/codex_assets.py +64 -5
- package/src/superlocalmemory/hooks/hook_daemon.py +20 -3
- package/src/superlocalmemory/hooks/memory_protocol.py +54 -0
- package/src/superlocalmemory/hooks/portable_kit.py +114 -1
- package/src/superlocalmemory/infra/auth_middleware.py +28 -0
- package/src/superlocalmemory/infra/backup.py +12 -1
- package/src/superlocalmemory/infra/daemon_identity.py +40 -4
- package/src/superlocalmemory/infra/data_root.py +43 -4
- package/src/superlocalmemory/infra/event_bus.py +107 -24
- package/src/superlocalmemory/infra/rate_limiter.py +93 -0
- package/src/superlocalmemory/ingestion/adapter_manager.py +4 -1
- package/src/superlocalmemory/ingestion/credentials.py +1 -1
- package/src/superlocalmemory/learning/cross_project.py +28 -19
- package/src/superlocalmemory/learning/reward_proxy.py +42 -9
- package/src/superlocalmemory/loops/__init__.py +56 -0
- package/src/superlocalmemory/loops/budget.py +58 -0
- package/src/superlocalmemory/loops/engine.py +164 -0
- package/src/superlocalmemory/loops/ledger.py +243 -0
- package/src/superlocalmemory/loops/models.py +152 -0
- package/src/superlocalmemory/loops/rules.py +52 -0
- package/src/superlocalmemory/mcp/_daemon_proxy.py +3 -0
- package/src/superlocalmemory/mcp/_pool_adapter.py +2 -0
- package/src/superlocalmemory/mcp/profiles.py +103 -0
- package/src/superlocalmemory/mcp/server.py +21 -49
- package/src/superlocalmemory/mcp/tools_active.py +4 -7
- package/src/superlocalmemory/mcp/tools_code_graph.py +51 -5
- package/src/superlocalmemory/mcp/tools_core.py +50 -5
- package/src/superlocalmemory/mcp/tools_evolution.py +6 -3
- package/src/superlocalmemory/mcp/tools_loops.py +300 -0
- package/src/superlocalmemory/mcp/tools_mesh.py +140 -4
- package/src/superlocalmemory/mcp/tools_optimize.py +15 -8
- package/src/superlocalmemory/mesh/broker.py +237 -129
- package/src/superlocalmemory/mesh/remote_sync.py +50 -8
- package/src/superlocalmemory/optimize/NOTICE +1 -6
- package/src/superlocalmemory/optimize/adapters/anthropic_adapter.py +1 -4
- package/src/superlocalmemory/optimize/adapters/openai_adapter.py +1 -4
- package/src/superlocalmemory/optimize/cache/semantic.py +27 -19
- package/src/superlocalmemory/optimize/compress/align.py +32 -26
- package/src/superlocalmemory/optimize/compress/ccr.py +14 -71
- package/src/superlocalmemory/optimize/compress/router.py +105 -22
- package/src/superlocalmemory/optimize/config/defaults.py +1 -1
- package/src/superlocalmemory/optimize/config/schema.py +87 -4
- package/src/superlocalmemory/optimize/metrics/counters.py +13 -4
- package/src/superlocalmemory/optimize/metrics/estimator.py +0 -3
- package/src/superlocalmemory/optimize/proxy/_helpers.py +31 -4
- package/src/superlocalmemory/optimize/storage/db.py +38 -9
- package/src/superlocalmemory/optimize/storage/schema.py +10 -0
- package/src/superlocalmemory/parameterization/pattern_extractor.py +6 -3
- package/src/superlocalmemory/retrieval/agentic.py +1 -1
- package/src/superlocalmemory/retrieval/bm25_channel.py +68 -10
- package/src/superlocalmemory/retrieval/engine.py +168 -26
- package/src/superlocalmemory/retrieval/entity_channel.py +7 -5
- package/src/superlocalmemory/retrieval/hopfield_channel.py +9 -2
- package/src/superlocalmemory/retrieval/semantic_channel.py +114 -21
- package/src/superlocalmemory/retrieval/spreading_activation.py +11 -2
- package/src/superlocalmemory/retrieval/temporal_channel.py +48 -9
- package/src/superlocalmemory/retrieval/temporal_frame.py +102 -0
- package/src/superlocalmemory/retrieval/temporal_validity_filter.py +135 -0
- package/src/superlocalmemory/retrieval/time_window.py +181 -0
- package/src/superlocalmemory/server/api.py +21 -4
- package/src/superlocalmemory/server/profile_runtime.py +125 -8
- package/src/superlocalmemory/server/rbac_enforce.py +142 -0
- package/src/superlocalmemory/server/recall_health.py +24 -3
- package/src/superlocalmemory/server/recall_serializer.py +19 -1
- package/src/superlocalmemory/server/routes/abstraction.py +115 -0
- package/src/superlocalmemory/server/routes/agents.py +128 -38
- package/src/superlocalmemory/server/routes/backup.py +34 -10
- package/src/superlocalmemory/server/routes/behavioral.py +13 -12
- package/src/superlocalmemory/server/routes/brain.py +21 -5
- package/src/superlocalmemory/server/routes/chat.py +72 -16
- package/src/superlocalmemory/server/routes/compliance.py +171 -21
- package/src/superlocalmemory/server/routes/config_api.py +436 -0
- package/src/superlocalmemory/server/routes/data_io.py +30 -8
- package/src/superlocalmemory/server/routes/entity.py +9 -4
- package/src/superlocalmemory/server/routes/events.py +24 -8
- package/src/superlocalmemory/server/routes/evolution.py +135 -17
- package/src/superlocalmemory/server/routes/helpers.py +16 -1
- package/src/superlocalmemory/server/routes/ingest.py +7 -4
- package/src/superlocalmemory/server/routes/insights.py +3 -3
- package/src/superlocalmemory/server/routes/learning.py +14 -14
- package/src/superlocalmemory/server/routes/lifecycle.py +59 -8
- package/src/superlocalmemory/server/routes/memories.py +221 -49
- package/src/superlocalmemory/server/routes/mesh.py +95 -15
- package/src/superlocalmemory/server/routes/optimize.py +33 -1
- package/src/superlocalmemory/server/routes/prewarm.py +2 -0
- package/src/superlocalmemory/server/routes/profiles.py +63 -17
- package/src/superlocalmemory/server/routes/ratelimit.py +124 -0
- package/src/superlocalmemory/server/routes/rbac.py +367 -0
- package/src/superlocalmemory/server/routes/stats.py +13 -6
- package/src/superlocalmemory/server/routes/tiers.py +11 -9
- package/src/superlocalmemory/server/routes/v3_api.py +194 -81
- package/src/superlocalmemory/server/routes/ws.py +5 -2
- package/src/superlocalmemory/server/security_middleware.py +12 -5
- package/src/superlocalmemory/server/ui.py +30 -5
- package/src/superlocalmemory/server/unified_daemon.py +431 -75
- package/src/superlocalmemory/server/write_identity.py +38 -8
- package/src/superlocalmemory/storage/database.py +265 -53
- package/src/superlocalmemory/storage/migration_runner.py +53 -0
- package/src/superlocalmemory/storage/migrations/M021_ingestion_log_profile.py +108 -0
- package/src/superlocalmemory/storage/migrations/M022_entity_aliases_profile.py +86 -0
- package/src/superlocalmemory/storage/migrations/M023_mesh_profile_isolation.py +194 -0
- package/src/superlocalmemory/storage/migrations/M024_rbac_users_roles.py +87 -0
- package/src/superlocalmemory/storage/migrations/M025_perf_indexes.py +90 -0
- package/src/superlocalmemory/storage/migrations/M026_rbac_memberships_fk.py +136 -0
- package/src/superlocalmemory/storage/migrations/M027_transferable_patterns_profile.py +163 -0
- package/src/superlocalmemory/storage/models.py +4 -0
- package/src/superlocalmemory/storage/schema.py +87 -0
- package/src/superlocalmemory/storage/schema_v32.py +0 -9
- package/src/superlocalmemory/storage/schema_v343.py +24 -12
- package/src/superlocalmemory/trust/gate.py +49 -8
- package/src/superlocalmemory/ui/assets/slm-icon-white.svg +64 -0
- package/src/superlocalmemory/ui/assets/slm-icon.svg +36 -0
- package/src/superlocalmemory/ui/css/design-system.css +621 -0
- package/src/superlocalmemory/ui/css/neural-glass.css +6 -0
- package/src/superlocalmemory/ui/css/od-bridge.css +158 -0
- package/src/superlocalmemory/ui/favicon.svg +35 -4
- package/src/superlocalmemory/ui/index.html +306 -173
- package/src/superlocalmemory/ui/js/brain.js +5 -20
- package/src/superlocalmemory/ui/js/core.js +47 -31
- package/src/superlocalmemory/ui/js/dashboard.js +314 -63
- package/src/superlocalmemory/ui/js/event-delegation.js +102 -0
- package/src/superlocalmemory/ui/js/knowledge-graph.js +11 -11
- package/src/superlocalmemory/ui/js/math-health.js +1 -1
- package/src/superlocalmemory/ui/js/memories.js +15 -4
- package/src/superlocalmemory/ui/js/memory-chat.js +7 -7
- package/src/superlocalmemory/ui/js/ng-entities.js +6 -8
- package/src/superlocalmemory/ui/js/ng-ingestion.js +4 -4
- package/src/superlocalmemory/ui/js/ng-mesh.js +4 -9
- package/src/superlocalmemory/ui/js/ng-shell.js +8 -8
- package/src/superlocalmemory/ui/js/ng-skills.js +54 -2
- package/src/superlocalmemory/ui/js/od-agents.js +544 -0
- package/src/superlocalmemory/ui/js/od-auth-gate.js +257 -0
- package/src/superlocalmemory/ui/js/od-backup.js +780 -0
- package/src/superlocalmemory/ui/js/od-brain.js +779 -0
- package/src/superlocalmemory/ui/js/od-entities.js +579 -0
- package/src/superlocalmemory/ui/js/od-graph.js +593 -0
- package/src/superlocalmemory/ui/js/od-health.js +539 -0
- package/src/superlocalmemory/ui/js/od-mcp.js +508 -0
- package/src/superlocalmemory/ui/js/od-memories.js +887 -0
- package/src/superlocalmemory/ui/js/od-mesh.js +539 -0
- package/src/superlocalmemory/ui/js/od-operations.js +1250 -0
- package/src/superlocalmemory/ui/js/od-optimize.js +787 -0
- package/src/superlocalmemory/ui/js/od-settings.js +1053 -0
- package/src/superlocalmemory/ui/js/od-shell.js +593 -0
- package/src/superlocalmemory/ui/js/od-skills.js +573 -0
- package/src/superlocalmemory/ui/js/od-team.js +258 -0
- package/src/superlocalmemory/ui/js/profiles.js +159 -46
- package/src/superlocalmemory/ui/js/settings.js +2 -2
- package/src/superlocalmemory/ui/js/timeline.js +34 -5
- package/src/superlocalmemory/ui/js/trust-dashboard.js +2 -2
- package/src/superlocalmemory/vector/lancedb_backend.py +8 -6
- package/src/superlocalmemory/learning/behavioral_listener.py +0 -94
|
@@ -13,6 +13,7 @@ independently of the main memory lifecycle system.
|
|
|
13
13
|
|
|
14
14
|
from __future__ import annotations
|
|
15
15
|
|
|
16
|
+
import json
|
|
16
17
|
import logging
|
|
17
18
|
import sqlite3
|
|
18
19
|
from datetime import datetime, timezone
|
|
@@ -27,11 +28,24 @@ CREATE TABLE IF NOT EXISTS retention_rules (
|
|
|
27
28
|
rule_name TEXT NOT NULL,
|
|
28
29
|
days INTEGER NOT NULL,
|
|
29
30
|
description TEXT DEFAULT '',
|
|
31
|
+
framework TEXT NOT NULL DEFAULT 'custom',
|
|
32
|
+
action TEXT NOT NULL DEFAULT 'archive',
|
|
33
|
+
applies_to TEXT NOT NULL DEFAULT '{}',
|
|
30
34
|
created_at TIMESTAMP DEFAULT CURRENT_TIMESTAMP,
|
|
31
35
|
UNIQUE(profile_id, rule_name)
|
|
32
36
|
)
|
|
33
37
|
"""
|
|
34
38
|
|
|
39
|
+
# Terminal lifecycle zone each action moves an expired fact into. 'notify'
|
|
40
|
+
# changes nothing — it only surfaces the count so an operator can act.
|
|
41
|
+
# atomic_facts.lifecycle CHECK allows only active/warm/cold/archived, so both
|
|
42
|
+
# retention actions land in the 'archived' lifecycle zone. 'tombstone' is
|
|
43
|
+
# additionally flagged purgeable via archive_status so a purge job can find it —
|
|
44
|
+
# this avoids a risky rebuild of the (large) atomic_facts CHECK constraint.
|
|
45
|
+
_ACTION_LIFECYCLE = {"archive": "archived", "tombstone": "archived"}
|
|
46
|
+
_TOMBSTONE_ACTIONS = frozenset({"tombstone"})
|
|
47
|
+
_VALID_ACTIONS = ("archive", "tombstone", "notify")
|
|
48
|
+
|
|
35
49
|
_FACTS_TABLE_CHECK = """
|
|
36
50
|
SELECT name FROM sqlite_master
|
|
37
51
|
WHERE type='table' AND name='atomic_facts'
|
|
@@ -54,8 +68,22 @@ class RetentionEngine:
|
|
|
54
68
|
# ------------------------------------------------------------------
|
|
55
69
|
|
|
56
70
|
def _ensure_table(self) -> None:
|
|
57
|
-
"""Create the retention_rules table
|
|
71
|
+
"""Create the retention_rules table, self-migrating older schemas.
|
|
72
|
+
|
|
73
|
+
Pre-GDPR-model tables had only (rule_name, days, description). Add the
|
|
74
|
+
framework/action/applies_to columns in place so existing rules keep
|
|
75
|
+
working (defaults: framework='custom', action='archive').
|
|
76
|
+
"""
|
|
58
77
|
self._db.execute(_RETENTION_RULES_TABLE)
|
|
78
|
+
cols = {r[1] for r in self._db.execute(
|
|
79
|
+
"PRAGMA table_info(retention_rules)").fetchall()}
|
|
80
|
+
for col, ddl in (
|
|
81
|
+
("framework", "framework TEXT NOT NULL DEFAULT 'custom'"),
|
|
82
|
+
("action", "action TEXT NOT NULL DEFAULT 'archive'"),
|
|
83
|
+
("applies_to", "applies_to TEXT NOT NULL DEFAULT '{}'"),
|
|
84
|
+
):
|
|
85
|
+
if col not in cols:
|
|
86
|
+
self._db.execute(f"ALTER TABLE retention_rules ADD COLUMN {ddl}")
|
|
59
87
|
self._db.commit()
|
|
60
88
|
|
|
61
89
|
def _has_facts_table(self) -> bool:
|
|
@@ -128,29 +156,89 @@ class RetentionEngine:
|
|
|
128
156
|
)
|
|
129
157
|
|
|
130
158
|
def get_rules(self, profile_id: str) -> list[dict[str, Any]]:
|
|
131
|
-
"""Get all retention rules for a profile.
|
|
159
|
+
"""Get all retention rules for a profile (full GDPR shape)."""
|
|
160
|
+
rows = self._db.execute(
|
|
161
|
+
"SELECT id, rule_name, days, description, framework, action, "
|
|
162
|
+
"applies_to, created_at FROM retention_rules "
|
|
163
|
+
"WHERE profile_id = ? ORDER BY id",
|
|
164
|
+
(profile_id,),
|
|
165
|
+
).fetchall()
|
|
166
|
+
out: list[dict[str, Any]] = []
|
|
167
|
+
for r in rows:
|
|
168
|
+
try:
|
|
169
|
+
applies = json.loads(r[6]) if r[6] else {}
|
|
170
|
+
except (ValueError, TypeError):
|
|
171
|
+
applies = {}
|
|
172
|
+
out.append({
|
|
173
|
+
"id": r[0],
|
|
174
|
+
"name": r[1],
|
|
175
|
+
"rule_name": r[1], # legacy alias
|
|
176
|
+
"retention_days": r[2],
|
|
177
|
+
"days": r[2], # legacy alias
|
|
178
|
+
"description": r[3],
|
|
179
|
+
"framework": r[4],
|
|
180
|
+
"action": r[5],
|
|
181
|
+
"applies_to": applies,
|
|
182
|
+
"created_at": r[7],
|
|
183
|
+
})
|
|
184
|
+
return out
|
|
132
185
|
|
|
133
|
-
|
|
134
|
-
|
|
186
|
+
# ------------------------------------------------------------------
|
|
187
|
+
# GDPR-model API (used by the compliance route + dashboard)
|
|
188
|
+
# ------------------------------------------------------------------
|
|
135
189
|
|
|
136
|
-
|
|
137
|
-
|
|
190
|
+
def create_rule(
|
|
191
|
+
self,
|
|
192
|
+
*,
|
|
193
|
+
name: str,
|
|
194
|
+
framework: str,
|
|
195
|
+
retention_days: int,
|
|
196
|
+
action: str,
|
|
197
|
+
applies_to: dict | None,
|
|
198
|
+
profile_id: str,
|
|
199
|
+
) -> int:
|
|
200
|
+
"""Create a retention rule and return its row id.
|
|
201
|
+
|
|
202
|
+
``action`` is one of archive | tombstone | notify. Raises ValueError on
|
|
203
|
+
an invalid action so the route returns a clear 4xx rather than storing
|
|
204
|
+
an unenforceable rule.
|
|
138
205
|
"""
|
|
206
|
+
if action not in _VALID_ACTIONS:
|
|
207
|
+
raise ValueError(f"action must be one of {_VALID_ACTIONS}")
|
|
208
|
+
cur = self._db.execute(
|
|
209
|
+
"INSERT OR REPLACE INTO retention_rules "
|
|
210
|
+
"(profile_id, rule_name, days, description, framework, action, applies_to) "
|
|
211
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?)",
|
|
212
|
+
(profile_id, name, int(retention_days), "", framework, action,
|
|
213
|
+
json.dumps(applies_to or {})),
|
|
214
|
+
)
|
|
215
|
+
self._db.commit()
|
|
216
|
+
logger.info(
|
|
217
|
+
"Created retention rule '%s' (%dd, %s/%s) for profile '%s'",
|
|
218
|
+
name, retention_days, framework, action, profile_id,
|
|
219
|
+
)
|
|
220
|
+
return int(cur.lastrowid or 0)
|
|
221
|
+
|
|
222
|
+
def list_rules(self, profile_id: str | None = None) -> list[dict[str, Any]]:
|
|
223
|
+
"""List rules for one profile, or all rules when profile_id is None."""
|
|
224
|
+
if profile_id is not None:
|
|
225
|
+
return self.get_rules(profile_id)
|
|
139
226
|
rows = self._db.execute(
|
|
140
|
-
"SELECT
|
|
141
|
-
"FROM retention_rules WHERE profile_id = ? "
|
|
142
|
-
"ORDER BY id",
|
|
143
|
-
(profile_id,),
|
|
227
|
+
"SELECT DISTINCT profile_id FROM retention_rules"
|
|
144
228
|
).fetchall()
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
229
|
+
out: list[dict[str, Any]] = []
|
|
230
|
+
for r in rows:
|
|
231
|
+
out.extend(self.get_rules(r[0]))
|
|
232
|
+
return out
|
|
233
|
+
|
|
234
|
+
def delete_rule(self, profile_id: str, name: str) -> bool:
|
|
235
|
+
"""Delete a rule by (profile_id, name). Returns True if a row was removed."""
|
|
236
|
+
cur = self._db.execute(
|
|
237
|
+
"DELETE FROM retention_rules WHERE profile_id = ? AND rule_name = ?",
|
|
238
|
+
(profile_id, name),
|
|
239
|
+
)
|
|
240
|
+
self._db.commit()
|
|
241
|
+
return cur.rowcount > 0
|
|
154
242
|
|
|
155
243
|
# ------------------------------------------------------------------
|
|
156
244
|
# Expiration detection
|
|
@@ -174,14 +262,14 @@ class RetentionEngine:
|
|
|
174
262
|
return []
|
|
175
263
|
|
|
176
264
|
# Use the shortest retention period (most restrictive)
|
|
177
|
-
min_days = min(r["
|
|
265
|
+
min_days = min(r["retention_days"] for r in rules)
|
|
178
266
|
|
|
179
267
|
if not self._has_facts_table():
|
|
180
268
|
return []
|
|
181
269
|
|
|
270
|
+
# atomic_facts is keyed by fact_id (TEXT hex), NOT an integer id.
|
|
182
271
|
rows = self._db.execute(
|
|
183
|
-
"SELECT
|
|
184
|
-
"WHERE profile_id = ?",
|
|
272
|
+
"SELECT fact_id, created_at FROM atomic_facts WHERE profile_id = ?",
|
|
185
273
|
(profile_id,),
|
|
186
274
|
).fetchall()
|
|
187
275
|
|
|
@@ -199,34 +287,77 @@ class RetentionEngine:
|
|
|
199
287
|
# ------------------------------------------------------------------
|
|
200
288
|
|
|
201
289
|
def enforce(self, profile_id: str) -> dict[str, Any]:
|
|
202
|
-
"""Enforce retention
|
|
290
|
+
"""Enforce every retention rule for a profile, honoring each action.
|
|
203
291
|
|
|
204
|
-
|
|
292
|
+
For each rule, facts in the profile older than its retention_days are
|
|
293
|
+
moved to the rule's terminal lifecycle zone:
|
|
294
|
+
* archive → lifecycle 'archived' (retained, hidden from recall)
|
|
295
|
+
* tombstone → lifecycle 'tombstoned' (soft-deleted, purgeable)
|
|
296
|
+
* notify → unchanged (only counted, so an operator can act)
|
|
205
297
|
|
|
206
|
-
|
|
207
|
-
|
|
298
|
+
GDPR-safe: soft-state transitions, never a raw DELETE — erasure is an
|
|
299
|
+
explicit separate operation. All updates are scoped to ``profile_id``
|
|
300
|
+
and keyed by ``fact_id`` (TEXT), fixing the prior ``id``/int() bug that
|
|
301
|
+
made enforcement crash outright.
|
|
208
302
|
|
|
209
|
-
Returns
|
|
210
|
-
Dict with keys: deleted_count, expired_ids, profile_id.
|
|
303
|
+
Returns per-action counts + the affected fact ids.
|
|
211
304
|
"""
|
|
212
|
-
|
|
213
|
-
|
|
305
|
+
rules = self.get_rules(profile_id)
|
|
306
|
+
result: dict[str, Any] = {
|
|
307
|
+
"profile_id": profile_id,
|
|
308
|
+
"archived": 0, "tombstoned": 0, "notified": 0,
|
|
309
|
+
"deleted_count": 0, # legacy alias (== tombstoned)
|
|
310
|
+
"affected_ids": [],
|
|
311
|
+
}
|
|
312
|
+
if not rules or not self._has_facts_table():
|
|
313
|
+
return result
|
|
314
|
+
|
|
315
|
+
rows = self._db.execute(
|
|
316
|
+
"SELECT fact_id, created_at FROM atomic_facts WHERE profile_id = ?",
|
|
317
|
+
(profile_id,),
|
|
318
|
+
).fetchall()
|
|
214
319
|
|
|
215
|
-
|
|
216
|
-
|
|
320
|
+
affected: set[str] = set()
|
|
321
|
+
for rule in rules:
|
|
322
|
+
action = rule.get("action", "archive")
|
|
323
|
+
days = rule.get("retention_days", rule.get("days", 0))
|
|
324
|
+
expired = [
|
|
325
|
+
str(r[0]) for r in rows
|
|
326
|
+
if r[1] and self._age_in_days(r[1]) > days
|
|
327
|
+
]
|
|
328
|
+
if not expired:
|
|
329
|
+
continue
|
|
330
|
+
if action == "notify":
|
|
331
|
+
result["notified"] += len(expired)
|
|
332
|
+
affected.update(expired)
|
|
333
|
+
continue
|
|
334
|
+
zone = _ACTION_LIFECYCLE.get(action)
|
|
335
|
+
if zone is None:
|
|
336
|
+
continue
|
|
337
|
+
placeholders = ",".join("?" for _ in expired)
|
|
217
338
|
self._db.execute(
|
|
218
|
-
f"
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
self._db.commit()
|
|
222
|
-
deleted_count = len(expired_ids)
|
|
223
|
-
logger.info(
|
|
224
|
-
"Retention enforcement: deleted %d facts from profile '%s'",
|
|
225
|
-
deleted_count, profile_id,
|
|
339
|
+
f"UPDATE atomic_facts SET lifecycle = ? "
|
|
340
|
+
f"WHERE profile_id = ? AND fact_id IN ({placeholders})",
|
|
341
|
+
[zone, profile_id, *expired],
|
|
226
342
|
)
|
|
343
|
+
if action in _TOMBSTONE_ACTIONS:
|
|
344
|
+
# Flag purgeable without violating the lifecycle CHECK.
|
|
345
|
+
try:
|
|
346
|
+
self._db.execute(
|
|
347
|
+
f"UPDATE atomic_facts SET archive_status = 'tombstoned' "
|
|
348
|
+
f"WHERE profile_id = ? AND fact_id IN ({placeholders})",
|
|
349
|
+
[profile_id, *expired],
|
|
350
|
+
)
|
|
351
|
+
except Exception:
|
|
352
|
+
pass
|
|
353
|
+
result["archived" if action == "archive" else "tombstoned"] += len(expired)
|
|
354
|
+
affected.update(expired)
|
|
227
355
|
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
356
|
+
self._db.commit()
|
|
357
|
+
result["affected_ids"] = sorted(affected)
|
|
358
|
+
result["deleted_count"] = result["tombstoned"] # legacy alias
|
|
359
|
+
logger.info(
|
|
360
|
+
"Retention enforcement for '%s': archived=%d tombstoned=%d notified=%d",
|
|
361
|
+
profile_id, result["archived"], result["tombstoned"], result["notified"],
|
|
362
|
+
)
|
|
363
|
+
return result
|
|
@@ -17,8 +17,6 @@ Part of Qualixar | Author: Varun Pratap Bhardwaj
|
|
|
17
17
|
from __future__ import annotations
|
|
18
18
|
|
|
19
19
|
import logging
|
|
20
|
-
import sqlite3
|
|
21
|
-
import threading
|
|
22
20
|
from pathlib import Path
|
|
23
21
|
from typing import TYPE_CHECKING, Any
|
|
24
22
|
|
|
@@ -366,47 +364,11 @@ class BackendOrchestrator:
|
|
|
366
364
|
logger.warning("LanceDB init failed: %s", exc)
|
|
367
365
|
self._lancedb = None
|
|
368
366
|
|
|
369
|
-
#
|
|
370
|
-
#
|
|
371
|
-
#
|
|
372
|
-
|
|
373
|
-
|
|
374
|
-
self._update_status("cozo", "migrating")
|
|
375
|
-
|
|
376
|
-
def _run():
|
|
377
|
-
conn = sqlite3.connect(str(self._data_dir / "memory.db"))
|
|
378
|
-
conn.execute("PRAGMA journal_mode=WAL")
|
|
379
|
-
conn.execute("PRAGMA query_only=ON") # F-07: read-only in migration thread
|
|
380
|
-
try:
|
|
381
|
-
count = self._cozo.bulk_import_from_sqlite(conn)
|
|
382
|
-
self._update_status("cozo", "active", count)
|
|
383
|
-
logger.info("CozoDB migration complete: %d edges", count)
|
|
384
|
-
except Exception as exc:
|
|
385
|
-
logger.error("CozoDB migration failed: %s", exc)
|
|
386
|
-
self._update_status("cozo", "failed", error=str(exc))
|
|
387
|
-
finally:
|
|
388
|
-
conn.close()
|
|
389
|
-
|
|
390
|
-
threading.Thread(target=_run, daemon=True).start()
|
|
391
|
-
|
|
392
|
-
def _migrate_lancedb(self) -> None:
|
|
393
|
-
self._update_status("lancedb", "migrating")
|
|
394
|
-
|
|
395
|
-
def _run():
|
|
396
|
-
conn = sqlite3.connect(str(self._data_dir / "memory.db"))
|
|
397
|
-
conn.execute("PRAGMA journal_mode=WAL")
|
|
398
|
-
conn.execute("PRAGMA query_only=ON")
|
|
399
|
-
try:
|
|
400
|
-
count = self._lancedb.bulk_import_from_sqlite(conn)
|
|
401
|
-
self._update_status("lancedb", "active", count)
|
|
402
|
-
logger.info("LanceDB migration complete: %d vectors", count)
|
|
403
|
-
except Exception as exc:
|
|
404
|
-
logger.error("LanceDB migration failed: %s", exc)
|
|
405
|
-
self._update_status("lancedb", "failed", error=str(exc))
|
|
406
|
-
finally:
|
|
407
|
-
conn.close()
|
|
408
|
-
|
|
409
|
-
threading.Thread(target=_run, daemon=True).start()
|
|
367
|
+
# v3.7.9 (scale MEDIUM-2): _migrate_cozo/_migrate_lancedb were dead code —
|
|
368
|
+
# never called anywhere — and bypassed the staged prepare→verify→promote
|
|
369
|
+
# safety envelope (no fingerprint, no parity, no backup). Removed so a future
|
|
370
|
+
# caller cannot re-import outside the lifecycle. Emergency re-imports must go
|
|
371
|
+
# through the Scale Engine lifecycle.
|
|
410
372
|
|
|
411
373
|
# ------------------------------------------------------------------
|
|
412
374
|
# Internal: Status
|
|
@@ -0,0 +1,267 @@
|
|
|
1
|
+
# Copyright (c) 2026 Varun Pratap Bhardwaj / Qualixar
|
|
2
|
+
# Licensed under AGPL-3.0-or-later - see LICENSE file
|
|
3
|
+
# Part of SuperLocalMemory V3 | https://qualixar.com | https://varunpratap.com
|
|
4
|
+
|
|
5
|
+
"""Community summaries (Wave Q2) — one synthesized report per entity community.
|
|
6
|
+
|
|
7
|
+
Rides on the entity-community backbone (core.entity_community). For each
|
|
8
|
+
community it gathers the member entities' facts, EXCLUDES superseded facts
|
|
9
|
+
(bi-temporal — market CRIT-3), then produces:
|
|
10
|
+
|
|
11
|
+
- a keyword-dense signal (always, Mode A, zero-LLM), and
|
|
12
|
+
- a summary: Mode B/C LLM synthesis via the shared core.Summarizer
|
|
13
|
+
(which itself falls back to a heuristic), else the Mode A keyword-dense
|
|
14
|
+
line. Fail-open — a summarizer error never breaks generation.
|
|
15
|
+
|
|
16
|
+
Surfacing is on-device-safe: summaries are PRECOMPUTED here in the background
|
|
17
|
+
and later matched to a query as a single thematic-context block (Q2b) — never
|
|
18
|
+
a GraphRAG-style per-query LLM fan-out (market CRIT-1). member_fact_ids gives
|
|
19
|
+
drill-down back to the source atoms (Q3).
|
|
20
|
+
|
|
21
|
+
Part of Qualixar | Author: Varun Pratap Bhardwaj
|
|
22
|
+
License: AGPL-3.0-or-later
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import logging
|
|
29
|
+
from collections import Counter, defaultdict
|
|
30
|
+
from typing import Any
|
|
31
|
+
|
|
32
|
+
from superlocalmemory.core.entity_community import EntityCommunityBuilder
|
|
33
|
+
|
|
34
|
+
logger = logging.getLogger(__name__)
|
|
35
|
+
|
|
36
|
+
_STOPWORDS = frozenset({
|
|
37
|
+
"the", "a", "an", "is", "was", "were", "are", "be", "been", "being",
|
|
38
|
+
"have", "has", "had", "do", "does", "did", "will", "would", "could",
|
|
39
|
+
"should", "may", "might", "shall", "can", "to", "of", "in", "for", "on",
|
|
40
|
+
"with", "at", "by", "from", "as", "into", "through", "and", "but", "or",
|
|
41
|
+
"not", "no", "this", "that", "these", "those", "it", "its", "they",
|
|
42
|
+
"them", "their", "he", "she", "his", "her", "we", "our", "you", "your",
|
|
43
|
+
"i", "my", "me", "his", "was", "who", "what", "when", "where", "how",
|
|
44
|
+
})
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
class CommunitySummaryBuilder:
|
|
48
|
+
"""Generate + persist one summary per entity community (background)."""
|
|
49
|
+
|
|
50
|
+
def __init__(
|
|
51
|
+
self,
|
|
52
|
+
db: Any,
|
|
53
|
+
summarizer: Any = None,
|
|
54
|
+
max_communities: int = 50,
|
|
55
|
+
min_facts: int = 2,
|
|
56
|
+
max_facts_per_community: int = 30,
|
|
57
|
+
max_keywords: int = 8,
|
|
58
|
+
summary_max_chars: int = 512,
|
|
59
|
+
) -> None:
|
|
60
|
+
self._db = db
|
|
61
|
+
self._summarizer = summarizer
|
|
62
|
+
self._max_communities = max(1, int(max_communities))
|
|
63
|
+
self._min_facts = max(1, int(min_facts))
|
|
64
|
+
self._max_facts = max(1, int(max_facts_per_community))
|
|
65
|
+
self._max_keywords = max(1, int(max_keywords))
|
|
66
|
+
self._summary_max_chars = max(64, int(summary_max_chars))
|
|
67
|
+
|
|
68
|
+
# ------------------------------------------------------------------
|
|
69
|
+
# Generation
|
|
70
|
+
# ------------------------------------------------------------------
|
|
71
|
+
|
|
72
|
+
def compute_and_store(self, profile_id: str) -> dict[str, int]:
|
|
73
|
+
communities = EntityCommunityBuilder(self._db).get_communities(profile_id)
|
|
74
|
+
try:
|
|
75
|
+
self._db.execute(
|
|
76
|
+
"DELETE FROM community_summaries WHERE profile_id = ?",
|
|
77
|
+
(profile_id,),
|
|
78
|
+
)
|
|
79
|
+
except Exception as exc:
|
|
80
|
+
logger.debug("community_summaries clear failed: %s", exc)
|
|
81
|
+
if not communities:
|
|
82
|
+
return {"summaries_written": 0, "communities": 0}
|
|
83
|
+
|
|
84
|
+
entity_to_cid: dict[str, int] = {
|
|
85
|
+
e: cid for cid, ents in communities.items() for e in ents
|
|
86
|
+
}
|
|
87
|
+
cid_facts = self._gather_facts(profile_id, entity_to_cid)
|
|
88
|
+
name_map = self._entity_names(profile_id, communities)
|
|
89
|
+
|
|
90
|
+
written = 0
|
|
91
|
+
ordered = sorted(
|
|
92
|
+
cid_facts.items(), key=lambda kv: len(kv[1]), reverse=True,
|
|
93
|
+
)
|
|
94
|
+
for cid, facts in ordered:
|
|
95
|
+
if written >= self._max_communities:
|
|
96
|
+
break
|
|
97
|
+
seen: set[str] = set()
|
|
98
|
+
vf: list[tuple[str, str]] = []
|
|
99
|
+
for fid, content in facts:
|
|
100
|
+
if fid in seen:
|
|
101
|
+
continue
|
|
102
|
+
seen.add(fid)
|
|
103
|
+
vf.append((fid, content))
|
|
104
|
+
if len(vf) < self._min_facts:
|
|
105
|
+
continue
|
|
106
|
+
|
|
107
|
+
fact_ids = [fid for fid, _ in vf]
|
|
108
|
+
contents = [c for _, c in vf][: self._max_facts]
|
|
109
|
+
entity_ids = list(communities.get(cid, []))
|
|
110
|
+
entity_names = [name_map.get(e, e) for e in entity_ids]
|
|
111
|
+
keywords = self._keywords(contents)
|
|
112
|
+
summary = self._summary(contents, entity_names, keywords)
|
|
113
|
+
|
|
114
|
+
try:
|
|
115
|
+
self._db.execute(
|
|
116
|
+
"INSERT OR REPLACE INTO community_summaries "
|
|
117
|
+
"(profile_id, community_id, summary, keywords, "
|
|
118
|
+
" entity_ids_json, fact_ids_json, fact_count, computed_at) "
|
|
119
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, datetime('now'))",
|
|
120
|
+
(
|
|
121
|
+
profile_id, cid, summary, keywords,
|
|
122
|
+
json.dumps(entity_ids), json.dumps(fact_ids),
|
|
123
|
+
len(fact_ids),
|
|
124
|
+
),
|
|
125
|
+
)
|
|
126
|
+
written += 1
|
|
127
|
+
except Exception as exc:
|
|
128
|
+
logger.debug("community_summaries write failed (%s): %s", cid, exc)
|
|
129
|
+
|
|
130
|
+
return {"summaries_written": written, "communities": len(communities)}
|
|
131
|
+
|
|
132
|
+
# ------------------------------------------------------------------
|
|
133
|
+
# Read API
|
|
134
|
+
# ------------------------------------------------------------------
|
|
135
|
+
|
|
136
|
+
def get_summaries(self, profile_id: str) -> list[dict]:
|
|
137
|
+
try:
|
|
138
|
+
rows = self._db.execute(
|
|
139
|
+
"SELECT * FROM community_summaries WHERE profile_id = ? "
|
|
140
|
+
"ORDER BY fact_count DESC",
|
|
141
|
+
(profile_id,),
|
|
142
|
+
)
|
|
143
|
+
except Exception as exc:
|
|
144
|
+
logger.debug("get_summaries failed: %s", exc)
|
|
145
|
+
return []
|
|
146
|
+
return [dict(r) for r in rows]
|
|
147
|
+
|
|
148
|
+
def get_summary(self, profile_id: str, community_id: int) -> dict | None:
|
|
149
|
+
try:
|
|
150
|
+
rows = self._db.execute(
|
|
151
|
+
"SELECT * FROM community_summaries "
|
|
152
|
+
"WHERE profile_id = ? AND community_id = ?",
|
|
153
|
+
(profile_id, int(community_id)),
|
|
154
|
+
)
|
|
155
|
+
except Exception as exc:
|
|
156
|
+
logger.debug("get_summary failed: %s", exc)
|
|
157
|
+
return None
|
|
158
|
+
return dict(rows[0]) if rows else None
|
|
159
|
+
|
|
160
|
+
# ------------------------------------------------------------------
|
|
161
|
+
# Internal
|
|
162
|
+
# ------------------------------------------------------------------
|
|
163
|
+
|
|
164
|
+
def _gather_facts(
|
|
165
|
+
self, profile_id: str, entity_to_cid: dict[str, int],
|
|
166
|
+
) -> dict[int, list[tuple[str, str]]]:
|
|
167
|
+
"""One scan → {community_id -> [(fact_id, content)]}, superseded dropped."""
|
|
168
|
+
try:
|
|
169
|
+
rows = self._db.execute(
|
|
170
|
+
"SELECT fact_id, canonical_entities_json, content "
|
|
171
|
+
"FROM atomic_facts WHERE profile_id = ?",
|
|
172
|
+
(profile_id,),
|
|
173
|
+
)
|
|
174
|
+
except Exception as exc:
|
|
175
|
+
logger.debug("community fact scan failed: %s", exc)
|
|
176
|
+
return {}
|
|
177
|
+
|
|
178
|
+
cid_facts: dict[int, list[tuple[str, str]]] = defaultdict(list)
|
|
179
|
+
for row in rows:
|
|
180
|
+
d = dict(row)
|
|
181
|
+
raw = d.get("canonical_entities_json")
|
|
182
|
+
if not raw:
|
|
183
|
+
continue
|
|
184
|
+
try:
|
|
185
|
+
ents = json.loads(raw)
|
|
186
|
+
except (ValueError, TypeError):
|
|
187
|
+
continue
|
|
188
|
+
if not isinstance(ents, list):
|
|
189
|
+
continue
|
|
190
|
+
cids = {
|
|
191
|
+
entity_to_cid[str(e).strip()]
|
|
192
|
+
for e in ents
|
|
193
|
+
if str(e).strip() in entity_to_cid
|
|
194
|
+
}
|
|
195
|
+
if not cids:
|
|
196
|
+
continue
|
|
197
|
+
fid = str(d["fact_id"])
|
|
198
|
+
content = d.get("content") or ""
|
|
199
|
+
for cid in cids:
|
|
200
|
+
cid_facts[cid].append((fid, content))
|
|
201
|
+
|
|
202
|
+
all_fids = list({fid for lst in cid_facts.values() for fid, _ in lst})
|
|
203
|
+
invalid: set[str] = set()
|
|
204
|
+
if all_fids:
|
|
205
|
+
try:
|
|
206
|
+
invalid = self._db.get_invalidated_fact_ids(all_fids, profile_id)
|
|
207
|
+
except Exception as exc:
|
|
208
|
+
logger.debug("invalidated-fact lookup failed: %s", exc)
|
|
209
|
+
if invalid:
|
|
210
|
+
cid_facts = {
|
|
211
|
+
cid: [(fid, c) for fid, c in lst if fid not in invalid]
|
|
212
|
+
for cid, lst in cid_facts.items()
|
|
213
|
+
}
|
|
214
|
+
return cid_facts
|
|
215
|
+
|
|
216
|
+
def _entity_names(
|
|
217
|
+
self, profile_id: str, communities: dict[int, list[str]],
|
|
218
|
+
) -> dict[str, str]:
|
|
219
|
+
all_eids = list({e for ents in communities.values() for e in ents})
|
|
220
|
+
name_map: dict[str, str] = {}
|
|
221
|
+
chunk = 900
|
|
222
|
+
for start in range(0, len(all_eids), chunk):
|
|
223
|
+
batch = all_eids[start:start + chunk]
|
|
224
|
+
ph = ",".join("?" for _ in batch)
|
|
225
|
+
try:
|
|
226
|
+
rows = self._db.execute(
|
|
227
|
+
"SELECT entity_id, canonical_name FROM canonical_entities "
|
|
228
|
+
f"WHERE profile_id = ? AND entity_id IN ({ph})",
|
|
229
|
+
(profile_id, *batch),
|
|
230
|
+
)
|
|
231
|
+
except Exception as exc:
|
|
232
|
+
logger.debug("entity-name lookup failed: %s", exc)
|
|
233
|
+
continue
|
|
234
|
+
for r in rows:
|
|
235
|
+
d = dict(r)
|
|
236
|
+
name_map[str(d["entity_id"])] = str(d.get("canonical_name") or "")
|
|
237
|
+
return name_map
|
|
238
|
+
|
|
239
|
+
def _keywords(self, contents: list[str]) -> str:
|
|
240
|
+
tokens: list[str] = []
|
|
241
|
+
for text in contents:
|
|
242
|
+
for word in text.lower().split():
|
|
243
|
+
w = word.strip(".,;:!?\"'()[]{}")
|
|
244
|
+
if len(w) > 2 and w not in _STOPWORDS:
|
|
245
|
+
tokens.append(w)
|
|
246
|
+
top = [w for w, _ in Counter(tokens).most_common(self._max_keywords)]
|
|
247
|
+
return ", ".join(top)
|
|
248
|
+
|
|
249
|
+
def _summary(
|
|
250
|
+
self, contents: list[str], entity_names: list[str], keywords: str,
|
|
251
|
+
) -> str:
|
|
252
|
+
if self._summarizer is not None:
|
|
253
|
+
try:
|
|
254
|
+
text = self._summarizer.summarize_cluster(
|
|
255
|
+
[{"content": c} for c in contents],
|
|
256
|
+
)
|
|
257
|
+
if text and text.strip():
|
|
258
|
+
return text.strip()[: self._summary_max_chars]
|
|
259
|
+
except Exception as exc:
|
|
260
|
+
logger.debug("community summarizer failed (fail-open): %s", exc)
|
|
261
|
+
# Mode A keyword-dense fallback.
|
|
262
|
+
names = [n for n in entity_names if n]
|
|
263
|
+
topic = ", ".join(names[:6]) if names else "related memories"
|
|
264
|
+
base = f"Topics: {topic}."
|
|
265
|
+
if keywords:
|
|
266
|
+
base += f" Key terms: {keywords}."
|
|
267
|
+
return base[: self._summary_max_chars]
|