superlocalmemory 3.6.22 → 3.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +52 -0
- package/README.md +275 -72
- package/bin/slm-npm +43 -89
- package/docs/pi-dev-integration.md +43 -0
- package/ide/configs/antigravity-mcp.json +2 -2
- package/ide/configs/chatgpt-desktop-mcp.json +1 -1
- package/ide/configs/claude-desktop-mcp.json +2 -2
- package/ide/configs/windsurf-mcp.json +2 -2
- package/ide/hooks/context-hook.js +6 -2
- package/ide/hooks/post-recall-hook.js +7 -3
- package/ide/hooks/tool-event-hook.sh +2 -1
- package/package.json +19 -10
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/_GENERATED.md +1 -1
- package/plugin/agents/slm-memory-advisor.md +1 -1
- package/plugin/requirements.txt +1 -1
- package/plugin/skills/slm-session/SKILL.md +1 -1
- package/plugin-src/rules/AGENTS.md +1 -1
- package/pyproject.toml +40 -8
- package/scripts/postinstall-interactive.js +17 -94
- package/scripts/postinstall.js +185 -258
- package/scripts/preuninstall.js +9 -50
- package/src/superlocalmemory/__init__.py +2 -2
- package/src/superlocalmemory/attribution/mathematical_dna.py +1 -1
- package/src/superlocalmemory/attribution/signer.py +34 -19
- package/src/superlocalmemory/attribution/watermark.py +1 -1
- package/src/superlocalmemory/cli/_lazy_init.py +3 -5
- package/src/superlocalmemory/cli/commands.py +490 -195
- package/src/superlocalmemory/cli/context_commands.py +5 -4
- package/src/superlocalmemory/cli/daemon.py +282 -187
- package/src/superlocalmemory/cli/db_migrate.py +3 -1
- package/src/superlocalmemory/cli/diagnostics_cmd.py +28 -0
- package/src/superlocalmemory/cli/evidence_cmd.py +103 -0
- package/src/superlocalmemory/cli/ingest_cmd.py +7 -3
- package/src/superlocalmemory/cli/main.py +128 -31
- package/src/superlocalmemory/cli/pending_store.py +54 -38
- package/src/superlocalmemory/cli/scale_engine_cmd.py +37 -0
- package/src/superlocalmemory/cli/service_installer.py +57 -52
- package/src/superlocalmemory/cli/setup_wizard.py +142 -88
- package/src/superlocalmemory/cli/version_banner.py +2 -1
- package/src/superlocalmemory/code_graph/config.py +3 -1
- package/src/superlocalmemory/core/backend_orchestrator.py +81 -21
- package/src/superlocalmemory/core/config.py +65 -20
- package/src/superlocalmemory/core/consolidation_engine.py +9 -7
- package/src/superlocalmemory/core/context_cache.py +56 -8
- package/src/superlocalmemory/core/derivation_lineage.py +246 -0
- package/src/superlocalmemory/core/embedding_worker.py +32 -20
- package/src/superlocalmemory/core/embeddings.py +54 -18
- package/src/superlocalmemory/core/engine.py +150 -104
- package/src/superlocalmemory/core/engine_ingestion.py +513 -0
- package/src/superlocalmemory/core/engine_wiring.py +2 -0
- package/src/superlocalmemory/core/evidence_bundle.py +526 -0
- package/src/superlocalmemory/core/fact_consolidator.py +5 -11
- package/src/superlocalmemory/core/graph_analyzer.py +2 -2
- package/src/superlocalmemory/core/health_monitor.py +4 -2
- package/src/superlocalmemory/core/ingestion_command.py +636 -0
- package/src/superlocalmemory/core/injection.py +69 -18
- package/src/superlocalmemory/core/lifecycle_state.py +153 -0
- package/src/superlocalmemory/core/maintenance.py +23 -22
- package/src/superlocalmemory/core/maintenance_scheduler.py +51 -35
- package/src/superlocalmemory/core/mutations.py +143 -0
- package/src/superlocalmemory/core/platform_utils.py +7 -4
- package/src/superlocalmemory/core/ram_lock.py +16 -5
- package/src/superlocalmemory/core/rate_limit.py +1 -1
- package/src/superlocalmemory/core/recall_pipeline.py +60 -101
- package/src/superlocalmemory/core/recall_worker.py +76 -59
- package/src/superlocalmemory/core/registry.py +1 -1
- package/src/superlocalmemory/core/scale_engine.py +293 -0
- package/src/superlocalmemory/core/score_contract.py +62 -0
- package/src/superlocalmemory/core/security_primitives.py +3 -1
- package/src/superlocalmemory/core/slm_disabled.py +3 -5
- package/src/superlocalmemory/core/store_pipeline.py +172 -40
- package/src/superlocalmemory/core/tier_manager.py +32 -20
- package/src/superlocalmemory/core/worker_pool.py +13 -4
- package/src/superlocalmemory/dynamics/activation_guided_quantization.py +1 -1
- package/src/superlocalmemory/dynamics/eap_scheduler.py +10 -3
- package/src/superlocalmemory/dynamics/ebbinghaus_langevin_coupling.py +1 -1
- package/src/superlocalmemory/dynamics/fisher_langevin_coupling.py +1 -1
- package/src/superlocalmemory/encoding/auto_linker.py +1 -1
- package/src/superlocalmemory/encoding/cognitive_consolidator.py +7 -16
- package/src/superlocalmemory/encoding/consolidator.py +22 -5
- package/src/superlocalmemory/encoding/fact_extractor.py +1 -1
- package/src/superlocalmemory/encoding/foresight.py +2 -0
- package/src/superlocalmemory/encoding/graph_builder.py +1 -1
- package/src/superlocalmemory/encoding/temporal_parser.py +2 -0
- package/src/superlocalmemory/evaluation/__init__.py +13 -0
- package/src/superlocalmemory/evaluation/calibration.py +308 -0
- package/src/superlocalmemory/evolution/skill_evolver.py +2 -1
- package/src/superlocalmemory/graph/cozo_backend.py +256 -23
- package/src/superlocalmemory/hooks/_outcome_common.py +21 -11
- package/src/superlocalmemory/hooks/antigravity_adapter.py +10 -31
- package/src/superlocalmemory/hooks/auto_invoker.py +25 -27
- package/src/superlocalmemory/hooks/auto_recall.py +31 -6
- package/src/superlocalmemory/hooks/auto_recall_hook.py +13 -33
- package/src/superlocalmemory/hooks/before_web_hook.py +9 -7
- package/src/superlocalmemory/hooks/claude_code_hooks.py +126 -39
- package/src/superlocalmemory/hooks/codex_assets.py +59 -0
- package/src/superlocalmemory/hooks/codex_hooks.py +186 -0
- package/src/superlocalmemory/hooks/context_payload.py +1 -1
- package/src/superlocalmemory/hooks/copilot_adapter.py +9 -24
- package/src/superlocalmemory/hooks/cursor_adapter.py +10 -32
- package/src/superlocalmemory/hooks/hook_daemon.py +4 -2
- package/src/superlocalmemory/hooks/hook_handlers.py +241 -55
- package/src/superlocalmemory/hooks/memory_protocol.py +5 -3
- package/src/superlocalmemory/hooks/post_tool_async_hook.py +4 -2
- package/src/superlocalmemory/hooks/session_registry.py +15 -8
- package/src/superlocalmemory/hooks/stop_outcome_hook.py +10 -6
- package/src/superlocalmemory/hooks/topic_shift_hook.py +42 -12
- package/src/superlocalmemory/hooks/user_prompt_hook.py +9 -14
- package/src/superlocalmemory/hooks/user_prompt_rehash_hook.py +19 -11
- package/src/superlocalmemory/infra/auth_middleware.py +38 -5
- package/src/superlocalmemory/infra/backup.py +7 -5
- package/src/superlocalmemory/infra/cloud_backup.py +18 -8
- package/src/superlocalmemory/infra/daemon_identity.py +248 -0
- package/src/superlocalmemory/infra/data_root.py +199 -0
- package/src/superlocalmemory/infra/event_bus.py +3 -1
- package/src/superlocalmemory/infra/local_diagnostics.py +327 -0
- package/src/superlocalmemory/infra/process_reaper.py +23 -0
- package/src/superlocalmemory/ingestion/adapter_manager.py +27 -9
- package/src/superlocalmemory/ingestion/base_adapter.py +25 -31
- package/src/superlocalmemory/ingestion/calendar_adapter.py +13 -4
- package/src/superlocalmemory/ingestion/credentials.py +14 -7
- package/src/superlocalmemory/ingestion/gmail_adapter.py +13 -4
- package/src/superlocalmemory/ingestion/transcript_adapter.py +7 -2
- package/src/superlocalmemory/learning/consolidation_quantization_worker.py +1 -1
- package/src/superlocalmemory/learning/ensemble.py +11 -0
- package/src/superlocalmemory/learning/entity_compiler.py +1 -1
- package/src/superlocalmemory/learning/feedback.py +1 -1
- package/src/superlocalmemory/learning/forgetting_scheduler.py +12 -7
- package/src/superlocalmemory/learning/quantization_scheduler.py +1 -1
- package/src/superlocalmemory/learning/ranker.py +4 -1
- package/src/superlocalmemory/learning/source_quality.py +1 -1
- package/src/superlocalmemory/learning/trigram_index.py +3 -2
- package/src/superlocalmemory/llm/backbone.py +13 -8
- package/src/superlocalmemory/math/ebbinghaus.py +1 -1
- package/src/superlocalmemory/math/fisher.py +1 -1
- package/src/superlocalmemory/math/fisher_quantized.py +1 -1
- package/src/superlocalmemory/math/hopfield.py +1 -1
- package/src/superlocalmemory/math/langevin.py +1 -1
- package/src/superlocalmemory/math/polar_quant.py +3 -4
- package/src/superlocalmemory/math/qjl.py +1 -1
- package/src/superlocalmemory/math/sheaf.py +1 -1
- package/src/superlocalmemory/math/turbo_quant.py +3 -2
- package/src/superlocalmemory/mcp/_daemon_proxy.py +12 -11
- package/src/superlocalmemory/mcp/_pool_adapter.py +27 -0
- package/src/superlocalmemory/mcp/http_transport.py +53 -0
- package/src/superlocalmemory/mcp/server.py +39 -13
- package/src/superlocalmemory/mcp/shared.py +69 -3
- package/src/superlocalmemory/mcp/tools_active.py +141 -31
- package/src/superlocalmemory/mcp/tools_core.py +128 -29
- package/src/superlocalmemory/mcp/tools_evolution.py +5 -7
- package/src/superlocalmemory/mcp/tools_learning.py +42 -2
- package/src/superlocalmemory/mcp/tools_mesh.py +7 -23
- package/src/superlocalmemory/mcp/tools_optimize.py +8 -1
- package/src/superlocalmemory/mcp/tools_v28.py +23 -2
- package/src/superlocalmemory/mcp/tools_v3.py +26 -1
- package/src/superlocalmemory/mcp/tools_v33.py +56 -17
- package/src/superlocalmemory/mesh/broker.py +2 -0
- package/src/superlocalmemory/mesh/remote_sync.py +50 -12
- package/src/superlocalmemory/optimize/cache/manager.py +77 -1
- package/src/superlocalmemory/optimize/cache/semantic.py +23 -3
- package/src/superlocalmemory/optimize/compress/ccr.py +4 -0
- package/src/superlocalmemory/optimize/compress/router.py +6 -1
- package/src/superlocalmemory/optimize/config/__init__.py +5 -0
- package/src/superlocalmemory/optimize/config/store.py +6 -4
- package/src/superlocalmemory/optimize/proxy/_helpers.py +15 -5
- package/src/superlocalmemory/optimize/proxy/capture.py +3 -2
- package/src/superlocalmemory/optimize/proxy/server.py +2 -2
- package/src/superlocalmemory/optimize/storage/db.py +14 -13
- package/src/superlocalmemory/retrieval/agentic.py +1 -1
- package/src/superlocalmemory/retrieval/ann_index.py +1 -1
- package/src/superlocalmemory/retrieval/bm25_channel.py +35 -11
- package/src/superlocalmemory/retrieval/bridge_discovery.py +73 -8
- package/src/superlocalmemory/retrieval/engine.py +169 -79
- package/src/superlocalmemory/retrieval/entity_channel.py +289 -67
- package/src/superlocalmemory/retrieval/forgetting_filter.py +1 -1
- package/src/superlocalmemory/retrieval/fusion.py +1 -1
- package/src/superlocalmemory/retrieval/hopfield_channel.py +118 -30
- package/src/superlocalmemory/retrieval/profile_channel.py +1 -1
- package/src/superlocalmemory/retrieval/quantization_aware_search.py +16 -10
- package/src/superlocalmemory/retrieval/reranker.py +56 -20
- package/src/superlocalmemory/retrieval/scope_policy.py +85 -0
- package/src/superlocalmemory/retrieval/semantic_channel.py +122 -14
- package/src/superlocalmemory/retrieval/spreading_activation.py +141 -25
- package/src/superlocalmemory/retrieval/strategy.py +1 -1
- package/src/superlocalmemory/retrieval/temporal_channel.py +30 -15
- package/src/superlocalmemory/retrieval/vector_store.py +1 -1
- package/src/superlocalmemory/server/api.py +10 -7
- package/src/superlocalmemory/server/bandit_loops.py +4 -2
- package/src/superlocalmemory/server/recall_serializer.py +24 -0
- package/src/superlocalmemory/server/route_mutations.py +84 -0
- package/src/superlocalmemory/server/routes/agents.py +8 -6
- package/src/superlocalmemory/server/routes/brain.py +14 -12
- package/src/superlocalmemory/server/routes/chat.py +29 -12
- package/src/superlocalmemory/server/routes/data_io.py +55 -24
- package/src/superlocalmemory/server/routes/helpers.py +29 -4
- package/src/superlocalmemory/server/routes/ingest.py +53 -36
- package/src/superlocalmemory/server/routes/memories.py +104 -43
- package/src/superlocalmemory/server/routes/mesh.py +31 -0
- package/src/superlocalmemory/server/routes/profiles.py +26 -4
- package/src/superlocalmemory/server/routes/tiers.py +43 -11
- package/src/superlocalmemory/server/routes/timeline.py +5 -1
- package/src/superlocalmemory/server/routes/v3_api.py +76 -21
- package/src/superlocalmemory/server/security_middleware.py +1 -1
- package/src/superlocalmemory/server/ui.py +6 -3
- package/src/superlocalmemory/server/unified_daemon.py +680 -293
- package/src/superlocalmemory/server/write_identity.py +147 -0
- package/src/superlocalmemory/storage/access_log.py +4 -3
- package/src/superlocalmemory/storage/database.py +118 -25
- package/src/superlocalmemory/storage/migration_runner.py +84 -1
- package/src/superlocalmemory/storage/migration_v33.py +1 -1
- package/src/superlocalmemory/storage/migrations/M002_model_state_history.py +6 -60
- package/src/superlocalmemory/storage/migrations/M018_ingestion_operations.py +120 -0
- package/src/superlocalmemory/storage/migrations/M019_derivation_lineage.py +54 -0
- package/src/superlocalmemory/storage/migrations/M020_model_state_integrity.py +52 -0
- package/src/superlocalmemory/storage/migrations/__init__.py +5 -0
- package/src/superlocalmemory/storage/models.py +16 -0
- package/src/superlocalmemory/storage/quantized_store.py +20 -3
- package/src/superlocalmemory/storage/v2_migrator.py +5 -3
- package/src/superlocalmemory/ui/favicon.svg +5 -0
- package/src/superlocalmemory/ui/index.html +1 -0
- package/src/superlocalmemory/ui/js/compliance.js +1 -1
- package/src/superlocalmemory/ui/js/core.js +49 -8
- package/src/superlocalmemory/ui/js/dashboard.js +23 -2
- package/src/superlocalmemory/ui/js/feedback.js +1 -1
- package/src/superlocalmemory/ui/js/graph-filters.js +1 -1
- package/src/superlocalmemory/ui/js/graph-ui.js +1 -1
- package/src/superlocalmemory/ui/js/lifecycle.js +1 -1
- package/src/superlocalmemory/ui/js/ng-mesh.js +15 -49
- package/src/superlocalmemory/ui/js/settings.js +4 -2
- package/src/superlocalmemory/vector/lancedb_backend.py +57 -9
- package/bin/slm +0 -59
- package/bin/slm.bat +0 -77
- package/bin/slm.cmd +0 -5
- package/ide/integrations/langchain/README.md +0 -106
- package/ide/integrations/langchain/langchain_superlocalmemory/__init__.py +0 -9
- package/ide/integrations/langchain/langchain_superlocalmemory/chat_message_history.py +0 -201
- package/ide/integrations/langchain/pyproject.toml +0 -38
- package/ide/integrations/langchain/tests/__init__.py +0 -3
- package/ide/integrations/langchain/tests/test_chat_message_history.py +0 -215
- package/ide/integrations/langchain/tests/test_security.py +0 -117
- package/ide/integrations/llamaindex/README.md +0 -81
- package/ide/integrations/llamaindex/llama_index/storage/chat_store/superlocalmemory/__init__.py +0 -9
- package/ide/integrations/llamaindex/llama_index/storage/chat_store/superlocalmemory/base.py +0 -316
- package/ide/integrations/llamaindex/pyproject.toml +0 -43
- package/ide/integrations/llamaindex/tests/__init__.py +0 -3
- package/ide/integrations/llamaindex/tests/test_chat_store.py +0 -294
- package/ide/integrations/llamaindex/tests/test_security.py +0 -241
- package/plugin-src/.mcp.json +0 -12
- package/plugin-src/agents/slm-memory-advisor.md +0 -44
- package/plugin-src/agents/slm-optimize-advisor.md +0 -38
- package/plugin-src/hooks/.gitkeep +0 -0
- package/plugin-src/hooks/hooks.json +0 -23
- package/plugin-src/manifest.json +0 -25
- package/plugin-src/requirements.txt +0 -1
- package/plugin-src/rules/CLAUDE.md.fragment +0 -44
- package/plugin-src/scripts/ensure-venv.bat +0 -122
- package/plugin-src/scripts/ensure-venv.sh +0 -105
- package/plugin-src/scripts/slm-launch +0 -15
- package/plugin-src/scripts/slm-launch.bat +0 -17
- package/plugin-src/settings.json +0 -16
- package/plugin-src/skills/slm-cache/SKILL.md +0 -140
- package/plugin-src/skills/slm-compress/SKILL.md +0 -143
- package/plugin-src/skills/slm-graph/SKILL.md +0 -300
- package/plugin-src/skills/slm-recall/SKILL.md +0 -204
- package/plugin-src/skills/slm-remember/SKILL.md +0 -194
- package/plugin-src/skills/slm-session/SKILL.md +0 -207
- package/plugin-src/skills/slm-status/SKILL.md +0 -149
- package/scripts/__tests__/build-plugin.test.mjs +0 -613
- package/scripts/_savings_math.py +0 -270
- package/scripts/build-dmg.sh +0 -417
- package/scripts/build-plugin.js +0 -742
- package/scripts/build-slm-hook.ps1 +0 -40
- package/scripts/build-slm-hook.sh +0 -45
- package/scripts/build_entry.py +0 -452
- package/scripts/ci/stage5b_gate.sh +0 -50
- package/scripts/dogfood_savings.py +0 -490
- package/scripts/generate-thumbnails.py +0 -218
- package/scripts/install-skills.ps1 +0 -4
- package/scripts/install-skills.sh +0 -5
- package/scripts/install.ps1 +0 -701
- package/scripts/install.sh +0 -1015
- package/scripts/postinstall_binary.js +0 -287
- package/scripts/prepack.js +0 -33
- package/scripts/release_manifest.py +0 -273
- package/scripts/slm-hook.spec +0 -56
- package/scripts/start-dashboard.ps1 +0 -52
- package/scripts/start-dashboard.sh +0 -41
- package/scripts/sync-wiki.ps1 +0 -127
- package/scripts/sync-wiki.sh +0 -82
- package/scripts/test-dmg.sh +0 -161
- package/scripts/test-npm-package.ps1 +0 -252
- package/scripts/test-npm-package.sh +0 -207
- package/scripts/verify-install.ps1 +0 -294
- package/scripts/verify-install.sh +0 -266
- package/scripts/verify-v27.ps1 +0 -301
- package/scripts/verify-v27.sh +0 -233
- package/src/superlocalmemory.egg-info/PKG-INFO +0 -513
- package/src/superlocalmemory.egg-info/SOURCES.txt +0 -529
- package/src/superlocalmemory.egg-info/dependency_links.txt +0 -1
- package/src/superlocalmemory.egg-info/entry_points.txt +0 -2
- package/src/superlocalmemory.egg-info/requires.txt +0 -71
- package/src/superlocalmemory.egg-info/top_level.txt +0 -1
|
@@ -0,0 +1,526 @@
|
|
|
1
|
+
# Copyright (c) 2026 Varun Pratap Bhardwaj / Qualixar
|
|
2
|
+
# Licensed under AGPL-3.0-or-later - see LICENSE file
|
|
3
|
+
|
|
4
|
+
"""Versioned, reviewable evidence export and rebuild contract.
|
|
5
|
+
|
|
6
|
+
The bundle keeps raw ingestion evidence and human-reviewable relational truth;
|
|
7
|
+
embeddings, Fisher parameters, lexical tokens, and optional backend indexes are
|
|
8
|
+
deliberately excluded because they are derived and rebuildable.
|
|
9
|
+
"""
|
|
10
|
+
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import hashlib
|
|
14
|
+
import json
|
|
15
|
+
import re
|
|
16
|
+
from dataclasses import dataclass
|
|
17
|
+
from pathlib import Path
|
|
18
|
+
from typing import Any, Iterable
|
|
19
|
+
|
|
20
|
+
BUNDLE_FORMAT = "superlocalmemory-evidence-bundle"
|
|
21
|
+
BUNDLE_SCHEMA_VERSION = 1
|
|
22
|
+
|
|
23
|
+
_FACT_DERIVED_COLUMNS = frozenset({
|
|
24
|
+
"embedding", "fisher_mean", "fisher_variance", "langevin_position",
|
|
25
|
+
})
|
|
26
|
+
|
|
27
|
+
# Import order is dependency-safe. Every row is profile-scoped except the
|
|
28
|
+
# profile record; aliases are excluded until their cross-table identity
|
|
29
|
+
# contract is versioned.
|
|
30
|
+
_TABLES: tuple[tuple[str, str, str], ...] = (
|
|
31
|
+
("profile.jsonl", "profiles", "profile_id"),
|
|
32
|
+
("memories.jsonl", "memories", "memory_id"),
|
|
33
|
+
("facts.jsonl", "atomic_facts", "fact_id"),
|
|
34
|
+
("entities.jsonl", "canonical_entities", "entity_id"),
|
|
35
|
+
("entity_profiles.jsonl", "entity_profiles", "profile_entry_id"),
|
|
36
|
+
("memory_scenes.jsonl", "memory_scenes", "scene_id"),
|
|
37
|
+
("graph_edges.jsonl", "graph_edges", "edge_id"),
|
|
38
|
+
("temporal_events.jsonl", "temporal_events", "event_id"),
|
|
39
|
+
("provenance.jsonl", "provenance", "provenance_id"),
|
|
40
|
+
("ingestion_operations.jsonl", "ingestion_operations", "operation_id"),
|
|
41
|
+
("derivation_lineage.jsonl", "derivation_lineage", "lineage_id"),
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
@dataclass(frozen=True, slots=True)
|
|
46
|
+
class BundleReport:
|
|
47
|
+
valid: bool
|
|
48
|
+
bundle_id: str
|
|
49
|
+
counts: dict[str, int]
|
|
50
|
+
errors: tuple[str, ...] = ()
|
|
51
|
+
warnings: tuple[str, ...] = ()
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def _canonical(value: Any) -> str:
|
|
55
|
+
return json.dumps(value, sort_keys=True, separators=(",", ":"), ensure_ascii=False)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _sha256_bytes(value: bytes) -> str:
|
|
59
|
+
return hashlib.sha256(value).hexdigest()
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _table_exists(db: Any, table: str) -> bool:
|
|
63
|
+
rows = db.execute(
|
|
64
|
+
"SELECT 1 FROM sqlite_master WHERE type='table' AND name=?", (table,),
|
|
65
|
+
)
|
|
66
|
+
if hasattr(rows, "fetchall"):
|
|
67
|
+
rows = rows.fetchall()
|
|
68
|
+
return bool(rows)
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
def _rows_for_profile(db: Any, table: str, profile_id: str, order: str) -> list[dict]:
|
|
72
|
+
if not _table_exists(db, table):
|
|
73
|
+
return []
|
|
74
|
+
rows = db.execute(
|
|
75
|
+
f'SELECT * FROM "{table}" WHERE profile_id=? ORDER BY "{order}"',
|
|
76
|
+
(profile_id,),
|
|
77
|
+
)
|
|
78
|
+
if hasattr(rows, "fetchall"):
|
|
79
|
+
rows = rows.fetchall()
|
|
80
|
+
result = [dict(row) for row in rows]
|
|
81
|
+
if table == "atomic_facts":
|
|
82
|
+
result = [
|
|
83
|
+
{key: value for key, value in row.items() if key not in _FACT_DERIVED_COLUMNS}
|
|
84
|
+
for row in result
|
|
85
|
+
]
|
|
86
|
+
return result
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def _write_jsonl(path: Path, rows: Iterable[dict]) -> tuple[str, int]:
|
|
90
|
+
payload = "".join(f"{_canonical(row)}\n" for row in rows).encode("utf-8")
|
|
91
|
+
path.write_bytes(payload)
|
|
92
|
+
return _sha256_bytes(payload), payload.count(b"\n")
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def _load_jsonl(path: Path) -> list[dict]:
|
|
96
|
+
if not path.exists():
|
|
97
|
+
return []
|
|
98
|
+
return [json.loads(line) for line in path.read_text(encoding="utf-8").splitlines() if line]
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _source_spans(
|
|
102
|
+
operations: list[dict], facts: list[dict], lineage: list[dict],
|
|
103
|
+
) -> tuple[list[dict], dict[str, dict], list[str]]:
|
|
104
|
+
facts_by_id = {str(row["fact_id"]): row for row in facts}
|
|
105
|
+
spans: list[dict] = []
|
|
106
|
+
by_fact: dict[str, dict] = {}
|
|
107
|
+
unresolved: list[str] = []
|
|
108
|
+
durable = {
|
|
109
|
+
(str(row.get("object_id")), str(row.get("operation_id"))): row
|
|
110
|
+
for row in lineage
|
|
111
|
+
if row.get("object_type") == "fact"
|
|
112
|
+
}
|
|
113
|
+
for operation in operations:
|
|
114
|
+
raw = str(operation.get("raw_content") or "")
|
|
115
|
+
ids: list[str] = []
|
|
116
|
+
for key in ("final_fact_ids_json", "queryable_fact_ids_json"):
|
|
117
|
+
try:
|
|
118
|
+
ids.extend(str(value) for value in json.loads(operation.get(key) or "[]"))
|
|
119
|
+
except (TypeError, ValueError):
|
|
120
|
+
continue
|
|
121
|
+
for fact_id in dict.fromkeys(ids):
|
|
122
|
+
fact = facts_by_id.get(fact_id)
|
|
123
|
+
if fact is None:
|
|
124
|
+
unresolved.append(
|
|
125
|
+
f"operation {operation['operation_id']} references "
|
|
126
|
+
f"missing fact {fact_id}"
|
|
127
|
+
)
|
|
128
|
+
continue
|
|
129
|
+
content = str(fact.get("content") or "")
|
|
130
|
+
recorded = durable.get((fact_id, str(operation["operation_id"])))
|
|
131
|
+
if recorded is not None and recorded.get("source_status") == "unresolved":
|
|
132
|
+
unresolved.append(
|
|
133
|
+
f"fact {fact_id} has no exact span in operation {operation['operation_id']}"
|
|
134
|
+
)
|
|
135
|
+
continue
|
|
136
|
+
start = (
|
|
137
|
+
int(recorded["source_start"])
|
|
138
|
+
if recorded is not None
|
|
139
|
+
and recorded.get("source_status") == "exact"
|
|
140
|
+
and recorded.get("source_start") is not None
|
|
141
|
+
else raw.find(content)
|
|
142
|
+
)
|
|
143
|
+
if start < 0:
|
|
144
|
+
unresolved.append(
|
|
145
|
+
f"fact {fact_id} has no exact span in operation {operation['operation_id']}"
|
|
146
|
+
)
|
|
147
|
+
continue
|
|
148
|
+
span = {
|
|
149
|
+
"end": start + len(content),
|
|
150
|
+
"fact_id": fact_id,
|
|
151
|
+
"operation_id": str(operation["operation_id"]),
|
|
152
|
+
"start": start,
|
|
153
|
+
"text_sha256": _sha256_bytes(content.encode("utf-8")),
|
|
154
|
+
}
|
|
155
|
+
spans.append(span)
|
|
156
|
+
by_fact[fact_id] = span
|
|
157
|
+
spans.sort(key=lambda row: (row["fact_id"], row["operation_id"], row["start"]))
|
|
158
|
+
return spans, by_fact, unresolved
|
|
159
|
+
|
|
160
|
+
|
|
161
|
+
def _lineage_coverage(
|
|
162
|
+
rows_by_file: dict[str, list[dict]], span_map: dict[str, dict],
|
|
163
|
+
) -> dict[str, dict[str, int]]:
|
|
164
|
+
lineage = rows_by_file.get("derivation_lineage.jsonl", [])
|
|
165
|
+
by_type: dict[str, dict[str, list[dict]]] = {}
|
|
166
|
+
for row in lineage:
|
|
167
|
+
by_type.setdefault(str(row.get("object_type")), {}).setdefault(
|
|
168
|
+
str(row.get("object_id")), []
|
|
169
|
+
).append(row)
|
|
170
|
+
|
|
171
|
+
objects = {
|
|
172
|
+
"fact": [str(row["fact_id"]) for row in rows_by_file.get("facts.jsonl", [])],
|
|
173
|
+
"profile": [str(row["profile_id"]) for row in rows_by_file.get("profile.jsonl", [])],
|
|
174
|
+
"entity_summary": [
|
|
175
|
+
str(row["profile_entry_id"])
|
|
176
|
+
for row in rows_by_file.get("entity_profiles.jsonl", [])
|
|
177
|
+
],
|
|
178
|
+
"graph_edge": [
|
|
179
|
+
str(row["edge_id"]) for row in rows_by_file.get("graph_edges.jsonl", [])
|
|
180
|
+
],
|
|
181
|
+
"memory_scene": [
|
|
182
|
+
str(row["scene_id"])
|
|
183
|
+
for row in rows_by_file.get("memory_scenes.jsonl", [])
|
|
184
|
+
],
|
|
185
|
+
"index_bm25": [
|
|
186
|
+
str(row["fact_id"]) for row in rows_by_file.get("facts.jsonl", [])
|
|
187
|
+
],
|
|
188
|
+
}
|
|
189
|
+
coverage: dict[str, dict[str, int]] = {}
|
|
190
|
+
for object_type, object_ids in objects.items():
|
|
191
|
+
counts = {
|
|
192
|
+
"total": len(object_ids),
|
|
193
|
+
"durable": 0,
|
|
194
|
+
"derived_from_facts": 0,
|
|
195
|
+
"not_applicable": 0,
|
|
196
|
+
"legacy_exact_inference": 0,
|
|
197
|
+
"unresolved": 0,
|
|
198
|
+
}
|
|
199
|
+
for object_id in object_ids:
|
|
200
|
+
records = by_type.get(object_type, {}).get(object_id, [])
|
|
201
|
+
if records:
|
|
202
|
+
statuses = {str(row.get("source_status")) for row in records}
|
|
203
|
+
if statuses - {"unresolved"}:
|
|
204
|
+
counts["durable"] += 1
|
|
205
|
+
if "derived_from_facts" in statuses:
|
|
206
|
+
counts["derived_from_facts"] += 1
|
|
207
|
+
if "not_applicable" in statuses:
|
|
208
|
+
counts["not_applicable"] += 1
|
|
209
|
+
else:
|
|
210
|
+
counts["unresolved"] += 1
|
|
211
|
+
elif object_type == "fact" and object_id in span_map:
|
|
212
|
+
counts["legacy_exact_inference"] += 1
|
|
213
|
+
else:
|
|
214
|
+
counts["unresolved"] += 1
|
|
215
|
+
coverage[object_type] = counts
|
|
216
|
+
return coverage
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def export_evidence_bundle(
|
|
220
|
+
db: Any, profile_id: str, destination: str | Path,
|
|
221
|
+
) -> dict:
|
|
222
|
+
"""Export deterministic JSONL truth plus a checksum manifest."""
|
|
223
|
+
root = Path(destination)
|
|
224
|
+
root.mkdir(parents=True, exist_ok=True)
|
|
225
|
+
if any(root.iterdir()):
|
|
226
|
+
raise ValueError(f"destination must be empty: {root}")
|
|
227
|
+
|
|
228
|
+
rows_by_file: dict[str, list[dict]] = {}
|
|
229
|
+
for filename, table, order in _TABLES:
|
|
230
|
+
rows_by_file[filename] = _rows_for_profile(db, table, profile_id, order)
|
|
231
|
+
|
|
232
|
+
spans, span_map, unresolved = _source_spans(
|
|
233
|
+
rows_by_file["ingestion_operations.jsonl"],
|
|
234
|
+
rows_by_file["facts.jsonl"],
|
|
235
|
+
rows_by_file["derivation_lineage.jsonl"],
|
|
236
|
+
)
|
|
237
|
+
rows_by_file["source_spans.jsonl"] = spans
|
|
238
|
+
|
|
239
|
+
files: dict[str, dict[str, Any]] = {}
|
|
240
|
+
for filename in sorted(rows_by_file):
|
|
241
|
+
digest, count = _write_jsonl(root / filename, rows_by_file[filename])
|
|
242
|
+
files[filename] = {"count": count, "sha256": digest}
|
|
243
|
+
|
|
244
|
+
derivation_versions = sorted({
|
|
245
|
+
str(row.get("derivation_version") or "")
|
|
246
|
+
for row in rows_by_file["ingestion_operations.jsonl"]
|
|
247
|
+
if row.get("derivation_version")
|
|
248
|
+
})
|
|
249
|
+
identity = {
|
|
250
|
+
"format": BUNDLE_FORMAT,
|
|
251
|
+
"schema_version": BUNDLE_SCHEMA_VERSION,
|
|
252
|
+
"profile_id": profile_id,
|
|
253
|
+
"files": files,
|
|
254
|
+
"derivation_versions": derivation_versions,
|
|
255
|
+
}
|
|
256
|
+
bundle_id = _sha256_bytes(_canonical(identity).encode("utf-8"))
|
|
257
|
+
manifest = {
|
|
258
|
+
**identity,
|
|
259
|
+
"bundle_id": bundle_id,
|
|
260
|
+
"source_spans": span_map,
|
|
261
|
+
"lineage_coverage": _lineage_coverage(rows_by_file, span_map),
|
|
262
|
+
"unresolved_source_links": unresolved,
|
|
263
|
+
}
|
|
264
|
+
(root / "manifest.json").write_text(
|
|
265
|
+
json.dumps(manifest, sort_keys=True, indent=2, ensure_ascii=False) + "\n",
|
|
266
|
+
encoding="utf-8",
|
|
267
|
+
)
|
|
268
|
+
return manifest
|
|
269
|
+
|
|
270
|
+
|
|
271
|
+
def verify_evidence_bundle(bundle: str | Path) -> BundleReport:
|
|
272
|
+
"""Verify checksums, counts, identities, and exact source spans."""
|
|
273
|
+
root = Path(bundle)
|
|
274
|
+
errors: list[str] = []
|
|
275
|
+
warnings: list[str] = []
|
|
276
|
+
try:
|
|
277
|
+
manifest = json.loads((root / "manifest.json").read_text(encoding="utf-8"))
|
|
278
|
+
except Exception as exc:
|
|
279
|
+
return BundleReport(False, "", {}, (f"manifest: {exc}",), ())
|
|
280
|
+
if manifest.get("format") != BUNDLE_FORMAT:
|
|
281
|
+
errors.append("unsupported bundle format")
|
|
282
|
+
if manifest.get("schema_version") != BUNDLE_SCHEMA_VERSION:
|
|
283
|
+
errors.append("unsupported bundle schema version")
|
|
284
|
+
|
|
285
|
+
counts: dict[str, int] = {}
|
|
286
|
+
for filename, expected in sorted((manifest.get("files") or {}).items()):
|
|
287
|
+
path = root / filename
|
|
288
|
+
if not path.is_file():
|
|
289
|
+
errors.append(f"missing file: {filename}")
|
|
290
|
+
continue
|
|
291
|
+
payload = path.read_bytes()
|
|
292
|
+
actual_hash = _sha256_bytes(payload)
|
|
293
|
+
if actual_hash != expected.get("sha256"):
|
|
294
|
+
errors.append(f"sha256 mismatch: {filename}")
|
|
295
|
+
try:
|
|
296
|
+
rows = _load_jsonl(path)
|
|
297
|
+
except Exception as exc:
|
|
298
|
+
errors.append(f"invalid JSONL {filename}: {exc}")
|
|
299
|
+
continue
|
|
300
|
+
counts[filename] = len(rows)
|
|
301
|
+
if len(rows) != int(expected.get("count", -1)):
|
|
302
|
+
errors.append(f"count mismatch: {filename}")
|
|
303
|
+
|
|
304
|
+
identity = {
|
|
305
|
+
"format": manifest.get("format"),
|
|
306
|
+
"schema_version": manifest.get("schema_version"),
|
|
307
|
+
"profile_id": manifest.get("profile_id"),
|
|
308
|
+
"files": manifest.get("files"),
|
|
309
|
+
"derivation_versions": manifest.get("derivation_versions") or [],
|
|
310
|
+
}
|
|
311
|
+
bundle_id = _sha256_bytes(_canonical(identity).encode("utf-8"))
|
|
312
|
+
if bundle_id != manifest.get("bundle_id"):
|
|
313
|
+
errors.append("bundle_id mismatch")
|
|
314
|
+
|
|
315
|
+
try:
|
|
316
|
+
facts = {str(row["fact_id"]): row for row in _load_jsonl(root / "facts.jsonl")}
|
|
317
|
+
operations = {
|
|
318
|
+
str(row["operation_id"]): row
|
|
319
|
+
for row in _load_jsonl(root / "ingestion_operations.jsonl")
|
|
320
|
+
}
|
|
321
|
+
for span in _load_jsonl(root / "source_spans.jsonl"):
|
|
322
|
+
fact = facts.get(str(span.get("fact_id")))
|
|
323
|
+
operation = operations.get(str(span.get("operation_id")))
|
|
324
|
+
if fact is None or operation is None:
|
|
325
|
+
errors.append(f"orphan source span: {span.get('fact_id')}")
|
|
326
|
+
continue
|
|
327
|
+
content = str(fact.get("content") or "")
|
|
328
|
+
raw = str(operation.get("raw_content") or "")
|
|
329
|
+
start, end = int(span["start"]), int(span["end"])
|
|
330
|
+
if raw[start:end] != content:
|
|
331
|
+
errors.append(f"source span content mismatch: {span['fact_id']}")
|
|
332
|
+
if _sha256_bytes(content.encode("utf-8")) != span.get("text_sha256"):
|
|
333
|
+
errors.append(f"source span hash mismatch: {span['fact_id']}")
|
|
334
|
+
warnings.extend(str(item) for item in manifest.get("unresolved_source_links") or [])
|
|
335
|
+
except Exception as exc:
|
|
336
|
+
errors.append(f"source reconciliation failed: {exc}")
|
|
337
|
+
|
|
338
|
+
return BundleReport(
|
|
339
|
+
valid=not errors,
|
|
340
|
+
bundle_id=bundle_id,
|
|
341
|
+
counts=counts,
|
|
342
|
+
errors=tuple(errors),
|
|
343
|
+
warnings=tuple(warnings),
|
|
344
|
+
)
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _table_columns(db: Any, table: str) -> set[str]:
|
|
348
|
+
rows = db.execute(f'PRAGMA table_info("{table}")')
|
|
349
|
+
if hasattr(rows, "fetchall"):
|
|
350
|
+
rows = rows.fetchall()
|
|
351
|
+
return {str(row["name"] if isinstance(row, dict) else row[1]) for row in rows}
|
|
352
|
+
|
|
353
|
+
|
|
354
|
+
def _insert_rows(db: Any, table: str, rows: list[dict], profile_id: str) -> None:
|
|
355
|
+
columns = _table_columns(db, table)
|
|
356
|
+
for source in rows:
|
|
357
|
+
row = dict(source)
|
|
358
|
+
if "profile_id" in row:
|
|
359
|
+
row["profile_id"] = profile_id
|
|
360
|
+
row = {key: value for key, value in row.items() if key in columns}
|
|
361
|
+
names = tuple(sorted(row))
|
|
362
|
+
if not names:
|
|
363
|
+
continue
|
|
364
|
+
marks = ",".join("?" for _ in names)
|
|
365
|
+
quoted = ",".join(f'"{name}"' for name in names)
|
|
366
|
+
db.execute(
|
|
367
|
+
f'INSERT OR REPLACE INTO "{table}" ({quoted}) VALUES ({marks})',
|
|
368
|
+
tuple(row[name] for name in names),
|
|
369
|
+
)
|
|
370
|
+
|
|
371
|
+
|
|
372
|
+
def import_evidence_bundle(
|
|
373
|
+
db: Any,
|
|
374
|
+
bundle: str | Path,
|
|
375
|
+
*,
|
|
376
|
+
target_profile_id: str | None = None,
|
|
377
|
+
replace: bool = False,
|
|
378
|
+
rollback_dir: str | Path | None = None,
|
|
379
|
+
) -> BundleReport:
|
|
380
|
+
"""Import relational truth; replacement always writes a rollback bundle."""
|
|
381
|
+
report = verify_evidence_bundle(bundle)
|
|
382
|
+
if not report.valid:
|
|
383
|
+
raise ValueError("invalid evidence bundle: " + "; ".join(report.errors))
|
|
384
|
+
manifest = json.loads((Path(bundle) / "manifest.json").read_text(encoding="utf-8"))
|
|
385
|
+
profile_id = target_profile_id or str(manifest["profile_id"])
|
|
386
|
+
existing = db.execute(
|
|
387
|
+
"SELECT COUNT(*) AS count FROM atomic_facts WHERE profile_id=?", (profile_id,),
|
|
388
|
+
)
|
|
389
|
+
existing_count = int(existing[0]["count"]) if existing else 0
|
|
390
|
+
if existing_count and not replace:
|
|
391
|
+
raise ValueError("target profile is not empty; use replace with rollback_dir")
|
|
392
|
+
if replace and rollback_dir is None:
|
|
393
|
+
raise ValueError("rollback_dir is required for replace")
|
|
394
|
+
if replace:
|
|
395
|
+
export_evidence_bundle(db, profile_id, rollback_dir)
|
|
396
|
+
|
|
397
|
+
delete_order = tuple(reversed(_TABLES[1:]))
|
|
398
|
+
with db.transaction():
|
|
399
|
+
if replace:
|
|
400
|
+
for _filename, table, _order in delete_order:
|
|
401
|
+
if _table_exists(db, table):
|
|
402
|
+
db.execute(f'DELETE FROM "{table}" WHERE profile_id=?', (profile_id,))
|
|
403
|
+
for filename, table, _order in _TABLES:
|
|
404
|
+
if not _table_exists(db, table):
|
|
405
|
+
continue
|
|
406
|
+
_insert_rows(db, table, _load_jsonl(Path(bundle) / filename), profile_id)
|
|
407
|
+
return verify_evidence_bundle(bundle)
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
_TOKEN_RE = re.compile(r"[\w'-]+", re.UNICODE)
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def rebuild_derived_state(
|
|
414
|
+
db: Any, profile_id: str, *, embedder: Any | None = None,
|
|
415
|
+
) -> dict[str, int]:
|
|
416
|
+
"""Rebuild deterministic lexical state and optional content embeddings."""
|
|
417
|
+
rows = db.execute(
|
|
418
|
+
"SELECT fact_id, content FROM atomic_facts WHERE profile_id=? "
|
|
419
|
+
"ORDER BY fact_id",
|
|
420
|
+
(profile_id,),
|
|
421
|
+
)
|
|
422
|
+
facts = [dict(row) for row in rows]
|
|
423
|
+
bm25_rows = 0
|
|
424
|
+
embeddings = 0
|
|
425
|
+
with db.transaction():
|
|
426
|
+
db.execute("DELETE FROM bm25_tokens WHERE profile_id=?", (profile_id,))
|
|
427
|
+
for fact in facts:
|
|
428
|
+
tokens = [token.lower() for token in _TOKEN_RE.findall(str(fact["content"]))]
|
|
429
|
+
db.execute(
|
|
430
|
+
"INSERT INTO bm25_tokens (fact_id, profile_id, tokens) VALUES (?,?,?)",
|
|
431
|
+
(fact["fact_id"], profile_id, json.dumps(tokens, separators=(",", ":"))),
|
|
432
|
+
)
|
|
433
|
+
bm25_rows += 1
|
|
434
|
+
if embedder is not None:
|
|
435
|
+
vector = embedder.embed(str(fact["content"]))
|
|
436
|
+
db.execute(
|
|
437
|
+
"UPDATE atomic_facts SET embedding=? WHERE fact_id=? AND profile_id=?",
|
|
438
|
+
(json.dumps(vector, separators=(",", ":")), fact["fact_id"], profile_id),
|
|
439
|
+
)
|
|
440
|
+
embeddings += 1
|
|
441
|
+
return {"bm25_rows": bm25_rows, "embeddings": embeddings}
|
|
442
|
+
|
|
443
|
+
|
|
444
|
+
def _normalized_bundle_rows(
|
|
445
|
+
bundle: Path, filename: str, profile_id: str,
|
|
446
|
+
) -> list[dict]:
|
|
447
|
+
rows = _load_jsonl(bundle / filename)
|
|
448
|
+
for row in rows:
|
|
449
|
+
if "profile_id" in row:
|
|
450
|
+
row["profile_id"] = profile_id
|
|
451
|
+
return rows
|
|
452
|
+
|
|
453
|
+
|
|
454
|
+
def verify_rebuild_equivalence(
|
|
455
|
+
db: Any, bundle: str | Path, *, profile_id: str,
|
|
456
|
+
) -> dict[str, Any]:
|
|
457
|
+
"""Compare deterministic local surfaces; leave model-dependent retrieval unverified."""
|
|
458
|
+
root = Path(bundle)
|
|
459
|
+
bundle_report = verify_evidence_bundle(root)
|
|
460
|
+
if not bundle_report.valid:
|
|
461
|
+
raise ValueError("invalid evidence bundle: " + "; ".join(bundle_report.errors))
|
|
462
|
+
|
|
463
|
+
mismatches: list[str] = []
|
|
464
|
+
relational = (
|
|
465
|
+
("profile.jsonl", "profiles", "profile_id"),
|
|
466
|
+
("memories.jsonl", "memories", "memory_id"),
|
|
467
|
+
("facts.jsonl", "atomic_facts", "fact_id"),
|
|
468
|
+
("entities.jsonl", "canonical_entities", "entity_id"),
|
|
469
|
+
("entity_profiles.jsonl", "entity_profiles", "profile_entry_id"),
|
|
470
|
+
("memory_scenes.jsonl", "memory_scenes", "scene_id"),
|
|
471
|
+
("graph_edges.jsonl", "graph_edges", "edge_id"),
|
|
472
|
+
)
|
|
473
|
+
for filename, table, order in relational:
|
|
474
|
+
expected = _normalized_bundle_rows(root, filename, profile_id)
|
|
475
|
+
actual = _rows_for_profile(db, table, profile_id, order)
|
|
476
|
+
if expected != actual:
|
|
477
|
+
mismatches.append(table)
|
|
478
|
+
|
|
479
|
+
facts = _normalized_bundle_rows(root, "facts.jsonl", profile_id)
|
|
480
|
+
fact_ids = {str(row["fact_id"]) for row in facts}
|
|
481
|
+
fts_rows = db.execute(
|
|
482
|
+
"SELECT fact_id FROM atomic_facts_fts ORDER BY fact_id"
|
|
483
|
+
) if _table_exists(db, "atomic_facts_fts") else []
|
|
484
|
+
fts_ids = {str(row["fact_id"]) for row in fts_rows}
|
|
485
|
+
if not fact_ids <= fts_ids:
|
|
486
|
+
mismatches.append("atomic_facts_fts")
|
|
487
|
+
|
|
488
|
+
expected_tokens = {
|
|
489
|
+
str(row["fact_id"]): [
|
|
490
|
+
token.lower() for token in _TOKEN_RE.findall(str(row.get("content") or ""))
|
|
491
|
+
]
|
|
492
|
+
for row in facts
|
|
493
|
+
}
|
|
494
|
+
actual_tokens = {
|
|
495
|
+
str(row["fact_id"]): json.loads(row["tokens"] or "[]")
|
|
496
|
+
for row in db.execute(
|
|
497
|
+
"SELECT fact_id,tokens FROM bm25_tokens WHERE profile_id=? ORDER BY fact_id",
|
|
498
|
+
(profile_id,),
|
|
499
|
+
)
|
|
500
|
+
}
|
|
501
|
+
if expected_tokens != actual_tokens:
|
|
502
|
+
mismatches.append("bm25_tokens")
|
|
503
|
+
|
|
504
|
+
return {
|
|
505
|
+
"status": "partial" if not mismatches else "failed",
|
|
506
|
+
"equivalent_on_verified_surfaces": not mismatches,
|
|
507
|
+
"verified_surfaces": [
|
|
508
|
+
"relational_truth", "fts_membership", "bm25_tokens",
|
|
509
|
+
],
|
|
510
|
+
"unverified_surfaces": [
|
|
511
|
+
"semantic_vector_retrieval", "ann_index", "reranker_behavior",
|
|
512
|
+
],
|
|
513
|
+
"mismatches": mismatches,
|
|
514
|
+
}
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
__all__ = [
|
|
518
|
+
"BUNDLE_FORMAT",
|
|
519
|
+
"BUNDLE_SCHEMA_VERSION",
|
|
520
|
+
"BundleReport",
|
|
521
|
+
"export_evidence_bundle",
|
|
522
|
+
"import_evidence_bundle",
|
|
523
|
+
"rebuild_derived_state",
|
|
524
|
+
"verify_evidence_bundle",
|
|
525
|
+
"verify_rebuild_equivalence",
|
|
526
|
+
]
|
|
@@ -289,11 +289,11 @@ def _consolidate_cluster(
|
|
|
289
289
|
""", (consolidation_id, profile_id, new_fact_id,
|
|
290
290
|
json.dumps(fact_ids), now))
|
|
291
291
|
|
|
292
|
-
# Archive the original facts (NEVER delete)
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
292
|
+
# Archive the original facts (NEVER delete) through the canonical
|
|
293
|
+
# lifecycle writer so missing retention rows are created too.
|
|
294
|
+
from superlocalmemory.core.lifecycle_state import set_fact_lifecycle_zone
|
|
295
|
+
set_fact_lifecycle_zone(
|
|
296
|
+
conn, fact_ids, "archive", profile_id=profile_id,
|
|
297
297
|
)
|
|
298
298
|
|
|
299
299
|
# P1-4 (graph-integrity-01): archived facts must stop influencing
|
|
@@ -309,12 +309,6 @@ def _consolidate_cluster(
|
|
|
309
309
|
f" OR target_fact_id IN ({placeholders}))",
|
|
310
310
|
(profile_id, *fact_ids, *fact_ids),
|
|
311
311
|
)
|
|
312
|
-
c.execute(
|
|
313
|
-
f"UPDATE fact_retention SET lifecycle_zone = 'archive' "
|
|
314
|
-
f"WHERE profile_id = ? AND fact_id IN ({placeholders})",
|
|
315
|
-
(profile_id, *fact_ids),
|
|
316
|
-
)
|
|
317
|
-
|
|
318
312
|
c.execute(f"RELEASE SAVEPOINT {savepoint_name}")
|
|
319
313
|
|
|
320
314
|
except Exception:
|
|
@@ -111,8 +111,8 @@ class GraphAnalyzer:
|
|
|
111
111
|
|
|
112
112
|
# v3.4.1: Persist community labels to JSON sidecar
|
|
113
113
|
try:
|
|
114
|
-
from
|
|
115
|
-
labels_dir =
|
|
114
|
+
from superlocalmemory.infra.data_root import canonical_data_root
|
|
115
|
+
labels_dir = canonical_data_root()
|
|
116
116
|
labels_dir.mkdir(parents=True, exist_ok=True)
|
|
117
117
|
labels_path = labels_dir / f"{profile_id}_community_labels.json"
|
|
118
118
|
labels_path.write_text(json.dumps(labels, indent=2))
|
|
@@ -11,7 +11,7 @@ Monitors:
|
|
|
11
11
|
- Extensible health check registry (Phase C/D/E add checks)
|
|
12
12
|
|
|
13
13
|
Part of Qualixar | Author: Varun Pratap Bhardwaj
|
|
14
|
-
License:
|
|
14
|
+
License: AGPL-3.0-or-later
|
|
15
15
|
"""
|
|
16
16
|
|
|
17
17
|
from __future__ import annotations
|
|
@@ -25,6 +25,8 @@ from datetime import datetime, timezone
|
|
|
25
25
|
from pathlib import Path
|
|
26
26
|
from typing import Callable
|
|
27
27
|
|
|
28
|
+
from superlocalmemory.infra.data_root import state_path
|
|
29
|
+
|
|
28
30
|
logger = logging.getLogger("superlocalmemory.health_monitor")
|
|
29
31
|
|
|
30
32
|
# Try psutil — graceful fallback if not available
|
|
@@ -78,7 +80,7 @@ def setup_structured_logging(log_dir: Path | None = None) -> None:
|
|
|
78
80
|
"""
|
|
79
81
|
global _json_logger
|
|
80
82
|
|
|
81
|
-
log_dir = log_dir or (
|
|
83
|
+
log_dir = log_dir or state_path("logs")
|
|
82
84
|
log_dir.mkdir(parents=True, exist_ok=True)
|
|
83
85
|
json_log_path = log_dir / "daemon.json.log"
|
|
84
86
|
|