contextos-memory-runtime 1.0.0rc2__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (157) hide show
  1. contextos_memory_runtime-1.0.0rc2/.gitignore +22 -0
  2. contextos_memory_runtime-1.0.0rc2/CHANGELOG.md +44 -0
  3. contextos_memory_runtime-1.0.0rc2/Makefile +44 -0
  4. contextos_memory_runtime-1.0.0rc2/PKG-INFO +143 -0
  5. contextos_memory_runtime-1.0.0rc2/README.md +100 -0
  6. contextos_memory_runtime-1.0.0rc2/docs/architecture.md +37 -0
  7. contextos_memory_runtime-1.0.0rc2/docs/benchmarking.md +13 -0
  8. contextos_memory_runtime-1.0.0rc2/docs/connectors.md +25 -0
  9. contextos_memory_runtime-1.0.0rc2/docs/demo.md +7 -0
  10. contextos_memory_runtime-1.0.0rc2/docs/explainability.md +73 -0
  11. contextos_memory_runtime-1.0.0rc2/docs/final-benchmark.json +46 -0
  12. contextos_memory_runtime-1.0.0rc2/docs/final-benchmark.md +22 -0
  13. contextos_memory_runtime-1.0.0rc2/docs/mcp.md +51 -0
  14. contextos_memory_runtime-1.0.0rc2/docs/performance.md +11 -0
  15. contextos_memory_runtime-1.0.0rc2/docs/phase12-benchmark.md +21 -0
  16. contextos_memory_runtime-1.0.0rc2/docs/phase13-audit.md +29 -0
  17. contextos_memory_runtime-1.0.0rc2/docs/phase16-coverage.md +30 -0
  18. contextos_memory_runtime-1.0.0rc2/docs/phase3-privacy.md +36 -0
  19. contextos_memory_runtime-1.0.0rc2/docs/rag-inspector.md +17 -0
  20. contextos_memory_runtime-1.0.0rc2/docs/release-checklist.md +17 -0
  21. contextos_memory_runtime-1.0.0rc2/docs/releases/v1.0.0-rc1.md +110 -0
  22. contextos_memory_runtime-1.0.0rc2/docs/releases/v1.0.0-rc2.md +116 -0
  23. contextos_memory_runtime-1.0.0rc2/docs/security.md +7 -0
  24. contextos_memory_runtime-1.0.0rc2/pyproject.toml +120 -0
  25. contextos_memory_runtime-1.0.0rc2/src/contextos/__init__.py +3 -0
  26. contextos_memory_runtime-1.0.0rc2/src/contextos/__main__.py +6 -0
  27. contextos_memory_runtime-1.0.0rc2/src/contextos/api/__init__.py +1 -0
  28. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/__init__.py +1 -0
  29. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/desktop.py +322 -0
  30. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/ingest.py +17 -0
  31. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/memories.py +84 -0
  32. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/models.py +81 -0
  33. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/retrieval.py +89 -0
  34. contextos_memory_runtime-1.0.0rc2/src/contextos/api/routes/system.py +216 -0
  35. contextos_memory_runtime-1.0.0rc2/src/contextos/api/server.py +195 -0
  36. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/__init__.py +1 -0
  37. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/compilation.py +245 -0
  38. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/connectors.py +423 -0
  39. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/explainability.py +103 -0
  40. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/final.py +406 -0
  41. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/graph.py +310 -0
  42. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/graph_adversarial.py +525 -0
  43. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/mcp.py +324 -0
  44. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/model_routing.py +203 -0
  45. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/optimization.py +305 -0
  46. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/rescue_integration.py +127 -0
  47. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/retrieval.py +266 -0
  48. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/temporal.py +377 -0
  49. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/temporal_hotpath.py +76 -0
  50. contextos_memory_runtime-1.0.0rc2/src/contextos/benchmarks/terminal.py +62 -0
  51. contextos_memory_runtime-1.0.0rc2/src/contextos/cli/__init__.py +1 -0
  52. contextos_memory_runtime-1.0.0rc2/src/contextos/cli/app.py +932 -0
  53. contextos_memory_runtime-1.0.0rc2/src/contextos/cli/dashboard.py +174 -0
  54. contextos_memory_runtime-1.0.0rc2/src/contextos/cli/formatters.py +299 -0
  55. contextos_memory_runtime-1.0.0rc2/src/contextos/config/__init__.py +1 -0
  56. contextos_memory_runtime-1.0.0rc2/src/contextos/config/settings.py +160 -0
  57. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/__init__.py +6 -0
  58. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/fake.py +11 -0
  59. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/json_import.py +125 -0
  60. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/local_files.py +102 -0
  61. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/manager.py +293 -0
  62. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/models.py +62 -0
  63. contextos_memory_runtime-1.0.0rc2/src/contextos/connectors/protocols.py +11 -0
  64. contextos_memory_runtime-1.0.0rc2/src/contextos/core/__init__.py +103 -0
  65. contextos_memory_runtime-1.0.0rc2/src/contextos/core/enums.py +489 -0
  66. contextos_memory_runtime-1.0.0rc2/src/contextos/core/exceptions.py +293 -0
  67. contextos_memory_runtime-1.0.0rc2/src/contextos/core/models.py +1147 -0
  68. contextos_memory_runtime-1.0.0rc2/src/contextos/core/protocols.py +549 -0
  69. contextos_memory_runtime-1.0.0rc2/src/contextos/daemon/__init__.py +1 -0
  70. contextos_memory_runtime-1.0.0rc2/src/contextos/daemon/manager.py +510 -0
  71. contextos_memory_runtime-1.0.0rc2/src/contextos/daemon/state.py +127 -0
  72. contextos_memory_runtime-1.0.0rc2/src/contextos/daemon/wiring.py +296 -0
  73. contextos_memory_runtime-1.0.0rc2/src/contextos/demo.py +217 -0
  74. contextos_memory_runtime-1.0.0rc2/src/contextos/embedding/__init__.py +1 -0
  75. contextos_memory_runtime-1.0.0rc2/src/contextos/embedding/deterministic.py +76 -0
  76. contextos_memory_runtime-1.0.0rc2/src/contextos/embedding/sentence_transformers.py +80 -0
  77. contextos_memory_runtime-1.0.0rc2/src/contextos/mcp/__init__.py +5 -0
  78. contextos_memory_runtime-1.0.0rc2/src/contextos/mcp/server.py +269 -0
  79. contextos_memory_runtime-1.0.0rc2/src/contextos/providers/__init__.py +13 -0
  80. contextos_memory_runtime-1.0.0rc2/src/contextos/providers/fake.py +217 -0
  81. contextos_memory_runtime-1.0.0rc2/src/contextos/providers/ollama.py +297 -0
  82. contextos_memory_runtime-1.0.0rc2/src/contextos/providers/openai_compatible.py +337 -0
  83. contextos_memory_runtime-1.0.0rc2/src/contextos/services/__init__.py +1 -0
  84. contextos_memory_runtime-1.0.0rc2/src/contextos/services/compilation.py +535 -0
  85. contextos_memory_runtime-1.0.0rc2/src/contextos/services/explainability.py +553 -0
  86. contextos_memory_runtime-1.0.0rc2/src/contextos/services/extraction.py +311 -0
  87. contextos_memory_runtime-1.0.0rc2/src/contextos/services/graph.py +524 -0
  88. contextos_memory_runtime-1.0.0rc2/src/contextos/services/graph_retrieval.py +143 -0
  89. contextos_memory_runtime-1.0.0rc2/src/contextos/services/ingestion.py +143 -0
  90. contextos_memory_runtime-1.0.0rc2/src/contextos/services/inspection.py +174 -0
  91. contextos_memory_runtime-1.0.0rc2/src/contextos/services/memory.py +291 -0
  92. contextos_memory_runtime-1.0.0rc2/src/contextos/services/model_service.py +409 -0
  93. contextos_memory_runtime-1.0.0rc2/src/contextos/services/optimization.py +426 -0
  94. contextos_memory_runtime-1.0.0rc2/src/contextos/services/privacy.py +331 -0
  95. contextos_memory_runtime-1.0.0rc2/src/contextos/services/retrieval.py +302 -0
  96. contextos_memory_runtime-1.0.0rc2/src/contextos/services/retrieval_index.py +88 -0
  97. contextos_memory_runtime-1.0.0rc2/src/contextos/services/router.py +302 -0
  98. contextos_memory_runtime-1.0.0rc2/src/contextos/services/secret_scanner.py +207 -0
  99. contextos_memory_runtime-1.0.0rc2/src/contextos/services/telemetry_query.py +102 -0
  100. contextos_memory_runtime-1.0.0rc2/src/contextos/services/temporal.py +500 -0
  101. contextos_memory_runtime-1.0.0rc2/src/contextos/services/token_counter.py +222 -0
  102. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/__init__.py +1 -0
  103. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/connector_repo.py +67 -0
  104. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/database.py +497 -0
  105. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/event_repo.py +137 -0
  106. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/graph_repo.py +228 -0
  107. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/lexical/__init__.py +1 -0
  108. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/lexical/bm25.py +134 -0
  109. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/memory_repo.py +589 -0
  110. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/relation_repo.py +80 -0
  111. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/telemetry_repo.py +481 -0
  112. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/vector/__init__.py +1 -0
  113. contextos_memory_runtime-1.0.0rc2/src/contextos/storage/vector/in_memory.py +162 -0
  114. contextos_memory_runtime-1.0.0rc2/tests/conftest.py +102 -0
  115. contextos_memory_runtime-1.0.0rc2/tests/unit/mcp_stdio_server.py +21 -0
  116. contextos_memory_runtime-1.0.0rc2/tests/unit/test_daemon_lifecycle.py +350 -0
  117. contextos_memory_runtime-1.0.0rc2/tests/unit/test_daemon_readiness.py +115 -0
  118. contextos_memory_runtime-1.0.0rc2/tests/unit/test_daemon_recovery.py +313 -0
  119. contextos_memory_runtime-1.0.0rc2/tests/unit/test_memory_extractor.py +140 -0
  120. contextos_memory_runtime-1.0.0rc2/tests/unit/test_memory_lifecycle.py +133 -0
  121. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase10_mcp.py +150 -0
  122. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase10_mcp_adversarial.py +61 -0
  123. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase10_mcp_comprehensive.py +1380 -0
  124. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase10_mcp_sqlite.py +92 -0
  125. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase10_mcp_stdio_sqlite.py +54 -0
  126. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase11_connectors.py +535 -0
  127. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase11_connectors_adversarial.py +433 -0
  128. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase11_connectors_concurrency.py +290 -0
  129. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase11_connectors_sqlite.py +1017 -0
  130. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase12_terminal.py +166 -0
  131. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase13_explainability.py +666 -0
  132. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase13_explainability_adversarial.py +464 -0
  133. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase14_inspector.py +159 -0
  134. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase15_benchmark.py +38 -0
  135. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase16_security.py +228 -0
  136. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase17_temporal_peer.py +68 -0
  137. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase18_dashboard.py +272 -0
  138. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase18_demo.py +27 -0
  139. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase1_memory_engine.py +192 -0
  140. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase2_extraction.py +266 -0
  141. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase3_privacy.py +387 -0
  142. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase4_retrieval.py +406 -0
  143. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase5_optimizer.py +463 -0
  144. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase6_compilation.py +450 -0
  145. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase6_rescue_integration.py +281 -0
  146. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase7_adversarial.py +804 -0
  147. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase7_temporal.py +561 -0
  148. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase8_graph.py +493 -0
  149. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase9_adversarial.py +936 -0
  150. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase9_api.py +170 -0
  151. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase9_pipeline.py +409 -0
  152. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase9_router.py +417 -0
  153. contextos_memory_runtime-1.0.0rc2/tests/unit/test_phase9_telemetry.py +351 -0
  154. contextos_memory_runtime-1.0.0rc2/tests/unit/test_release_artifacts.py +17 -0
  155. contextos_memory_runtime-1.0.0rc2/tests/unit/test_release_core.py +126 -0
  156. contextos_memory_runtime-1.0.0rc2/tests/unit/test_secret_scanner.py +218 -0
  157. contextos_memory_runtime-1.0.0rc2/tools/validate_release_artifacts.py +106 -0
@@ -0,0 +1,22 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .pytest_cache/
4
+ .coverage
5
+ htmlcov/
6
+
7
+ .venv/
8
+ venv/
9
+
10
+ .env
11
+ .env.*
12
+
13
+ *.db
14
+ *.sqlite
15
+ *.sqlite3
16
+
17
+ dist/
18
+ build/
19
+ *.egg-info/
20
+
21
+ .idea/
22
+ .vscode/
@@ -0,0 +1,44 @@
1
+ # ContextOS Changelog
2
+
3
+ All notable changes to ContextOS will be documented in this file.
4
+
5
+ The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
6
+ and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
7
+
8
+ ## [1.0.0-rc2] - 2026-10-03
9
+
10
+ - Distribution renamed to `contextos-memory-runtime`; imports and commands remain `contextos`.
11
+ - Bounded startup serialization using cross-platform OS lifecycle locks (`msvcrt` on Windows, `flock` on POSIX) with configured timeouts (30s readiness, 45s lock).
12
+ - OS TCP listener ownership verification: verifies that the configured loopback listener belongs to the verified ContextOS daemon process tree, rejecting spoofed endpoint PID claims.
13
+ - Recoverable startup ownership: daemon publishes durable identity prior to binding, ensuring that readiness implies discoverable lifecycle registration even if the starter process exits prematurely.
14
+ - Compare-before-delete lifecycle cleanup preserving survivor state during concurrent start or failed attempts.
15
+
16
+ ## [1.0.0-rc1] - 2026-10-02
17
+
18
+ ### Added
19
+ - **Core Architecture & Persistence:** Local SQLite storage engine (Schema v7) with append-only event logging, memory lifecycle tracking, and deterministic local defaults. The runtime has Python package dependencies but needs no cloud account or model/tokenizer download with those defaults.
20
+ - **Privacy & Security Boundary:** `PatternSecretScanner` detects supported credential shapes across CLI, API, MCP, and connectors; policy rejects, quarantines, or redacts findings. Selected graph/provenance outputs suppress path-shaped labels and terminal sanitization removes control sequences; universal private-path redaction is not a scanner capability. Attack matrix verified against ANSI/OSC escape injection, zero-width chars, malformed JSONL, SQLite contention, and credential leakage.
21
+ - **Temporal Memory:** Slot-key state tracking supporting supersession, contradiction, and coexistence relations with optimized candidate peer lookup beyond 500 records.
22
+ - **Hybrid RAG & Retrieval Engine:** BM25 lexical indexing + dense vector search combined via Reciprocal Rank Fusion (RRF), with batched candidate hydration.
23
+ - **Token-aware Optimizer & Compiler:** Knapsack-based context selection under token budgets with fact/provenance evidence formatting and contextual framing.
24
+ - **Deterministic Memory Graph:** Graph projection engine supporting bounded path traversal, entity/relation links, dirty-state tracking, and opt-in graph-assisted retrieval.
25
+ - **Model Router & Telemetry:** Provider-agnostic router (`FakeProvider`, `Ollama`, `OpenAICompatible`) with per-invocation token telemetry and measurement basis provenance.
26
+ - **Model Context Protocol (MCP):** STDIO protocol implementation (`contextos-mcp`) with permission checks and bounded output schemas.
27
+ - **Connectors System:** Local file and JSON/JSONL connectors with incremental sync, content hashing, and skip invariants for unchanged files.
28
+ - **Terminal UX & Dashboard:** Interactive CLI (`contextos stats`, `monitor`, `desktop`), rich terminal formatting, and diagnostic health checks (`contextos doctor`).
29
+ - **Explainability & RAG Inspector:** Non-intrusive trace generation and context inspection CLI (`contextos inspect`) for single-pass and side-by-side retrieval mode comparison.
30
+ - **Canonical Benchmarking & Evaluation Suite:** Synthetic benchmarking framework (`contextos.benchmarks.final`) evaluating 100 and 1,000 memory corpora across Vector, Hybrid, Hybrid+Graph, and ContextOS strategies.
31
+ - **Packaging & Modular Distribution:** Core `contextos` package decoupled from heavyweight ML frameworks, making `sentence-transformers` an optional extra (`pip install "contextos-memory-runtime[embeddings]"`).
32
+
33
+ ### Release blocker repairs
34
+ - Core defaults use hashed deterministic embeddings and word-token accounting labeled `APPROXIMATED`. Optional SentenceTransformer embeddings and exact tiktoken counting require explicit configuration.
35
+ - First-run demo needs no model/tokenizer assets or external network. Startup failure closes the initialized database before propagating the error.
36
+ - Wheel/sdist rules exclude local state and nested checkouts; `python tools/validate_release_artifacts.py` validates rebuilt archives.
37
+ - The canonical synthetic benchmark retains explicit `cl100k_base` counting; token reductions are not provider-billed savings.
38
+
39
+ ### Known limitations
40
+ - Answer quality is **NOT VALIDATED**. Deterministic embeddings are an offline baseline, not a SentenceTransformer-quality claim.
41
+ - Graph retrieval stays opt-in because the current synthetic benchmark shows a ranking/latency tradeoff.
42
+ - Live providers have not been comprehensively validated.
43
+ - Latest independent cold CLI startup: approximately **2.27 seconds median** on Windows.
44
+ - Python 3.12 is declared but not validated; release validation uses Windows/Python 3.13.5.
@@ -0,0 +1,44 @@
1
+ .PHONY: test lint typecheck check format install dev clean benchmark
2
+
3
+ install:
4
+ pip install -e .
5
+
6
+ dev:
7
+ pip install -e ".[dev]"
8
+
9
+ test:
10
+ pytest tests/ -v --tb=short -m "not slow and not benchmark"
11
+
12
+ test-all:
13
+ pytest tests/ -v --tb=short
14
+
15
+ test-unit:
16
+ pytest tests/unit/ -v --tb=short
17
+
18
+ test-integration:
19
+ pytest tests/integration/ -v --tb=short
20
+
21
+ test-adversarial:
22
+ pytest tests/adversarial/ -v --tb=short
23
+
24
+ test-cov:
25
+ pytest tests/ -v --tb=short --cov=contextos --cov-report=term-missing --cov-report=html -m "not slow and not benchmark"
26
+
27
+ benchmark:
28
+ pytest tests/benchmarks/ -v --tb=short -m benchmark
29
+
30
+ lint:
31
+ ruff check src/ tests/
32
+
33
+ format:
34
+ ruff format src/ tests/
35
+ ruff check --fix src/ tests/
36
+
37
+ typecheck:
38
+ mypy src/contextos/
39
+
40
+ check: lint typecheck test
41
+
42
+ clean:
43
+ rm -rf build/ dist/ *.egg-info .pytest_cache .mypy_cache .ruff_cache htmlcov/
44
+ find . -type d -name __pycache__ -exec rm -rf {} + 2>/dev/null || true
@@ -0,0 +1,143 @@
1
+ Metadata-Version: 2.5
2
+ Name: contextos-memory-runtime
3
+ Version: 1.0.0rc2
4
+ Summary: Local-first personal AI memory runtime
5
+ Author: ContextOS Contributors
6
+ License-Expression: MIT
7
+ Keywords: ai,context,local-first,memory,rag
8
+ Classifier: Development Status :: 4 - Beta
9
+ Classifier: Intended Audience :: Developers
10
+ Classifier: License :: OSI Approved :: MIT License
11
+ Classifier: Programming Language :: Python :: 3.12
12
+ Classifier: Programming Language :: Python :: 3.13
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Requires-Python: >=3.12
15
+ Requires-Dist: aiosqlite>=0.20.0
16
+ Requires-Dist: fastapi>=0.115.0
17
+ Requires-Dist: httpx>=0.27.0
18
+ Requires-Dist: mcp<3,>=2
19
+ Requires-Dist: numpy>=1.26.0
20
+ Requires-Dist: psutil>=6.0.0
21
+ Requires-Dist: pydantic-settings>=2.5.0
22
+ Requires-Dist: pydantic>=2.9.0
23
+ Requires-Dist: rank-bm25>=0.2.2
24
+ Requires-Dist: rich>=13.8.0
25
+ Requires-Dist: sqlite-vec>=0.1.0
26
+ Requires-Dist: tiktoken>=0.7.0
27
+ Requires-Dist: typer>=0.12.0
28
+ Requires-Dist: uvicorn[standard]>=0.30.0
29
+ Provides-Extra: dev
30
+ Requires-Dist: mypy>=1.11.0; extra == 'dev'
31
+ Requires-Dist: pre-commit>=3.8.0; extra == 'dev'
32
+ Requires-Dist: pytest-asyncio>=0.24.0; extra == 'dev'
33
+ Requires-Dist: pytest-cov>=5.0.0; extra == 'dev'
34
+ Requires-Dist: pytest>=8.3.0; extra == 'dev'
35
+ Requires-Dist: ruff>=0.6.0; extra == 'dev'
36
+ Provides-Extra: embeddings
37
+ Requires-Dist: sentence-transformers>=3.0.0; extra == 'embeddings'
38
+ Provides-Extra: lancedb
39
+ Requires-Dist: lancedb>=0.12.0; extra == 'lancedb'
40
+ Provides-Extra: tui
41
+ Requires-Dist: textual>=0.79.0; extra == 'tui'
42
+ Description-Content-Type: text/markdown
43
+
44
+ # ContextOS
45
+
46
+ ContextOS is a local-first, model-independent AI memory runtime. It is not a chatbot. It accepts activity through a privacy boundary, extracts candidate facts, resolves temporal state, persists memories in SQLite, and supplies bounded context to a chosen model. A local daemon exposes a CLI and API; optional STDIO MCP and explicitly configured connectors use the same memory pipeline.
47
+
48
+ ## Why it exists
49
+
50
+ Long conversation histories are expensive and can carry stale or contradictory facts. ContextOS combines lexical and dense retrieval, temporal eligibility, a token-aware selector, and a context compiler. It records what it selected and what it could prove, without claiming that fewer context tokens automatically improve answers or provider billing.
51
+
52
+ ## Quick start
53
+
54
+ Python 3.12+ is declared; this release has been verified on Windows with Python 3.13. Python 3.12 is not yet validated. Core defaults and the offline demo require no commercial API key, model download, or tokenizer cache.
55
+
56
+ ### Installation
57
+
58
+ Core lightweight installation (uses deterministic/local capabilities):
59
+ ```powershell
60
+ pip install contextos-memory-runtime
61
+ ```
62
+
63
+ Core defaults use hashed deterministic embeddings and word-token accounting labeled `APPROXIMATED`. These embeddings are an offline baseline; SentenceTransformer retrieval quality is not implied. FakeProvider is a local simulation.
64
+
65
+ Rebuilt RC2 archives are approximately **220 KB for the wheel** and **300 KB for the sdist** (decimal units). These sizes exclude installed dependencies. Run `python tools/validate_release_artifacts.py` after a rebuild for exact byte counts.
66
+
67
+ With optional local sentence-transformers embeddings extra:
68
+ ```powershell
69
+ pip install "contextos-memory-runtime[embeddings]"
70
+ ```
71
+
72
+ Installing the extra does not change defaults. Explicitly set `[embedding]` with `model = "all-MiniLM-L6-v2"` in your ContextOS `config.toml` to select it; first use may download that model unless provisioned locally. Exact context counting is also opt-in: set `[token_counter]` with `encoding = "cl100k_base"` or `"o200k_base"`, and provision its tokenizer cache before offline use. Default `encoding = "deterministic"` needs no tokenizer assets.
73
+
74
+ For development:
75
+ ```powershell
76
+ pip install -e ".[dev]"
77
+ ```
78
+
79
+ ### Usage
80
+
81
+ ```powershell
82
+ contextos demo
83
+ contextos start
84
+ contextos doctor
85
+ contextos stats
86
+ "I prefer concise technical explanations." | contextos memories remember
87
+ contextos inspect "What explanation style do I prefer?"
88
+ contextos stop
89
+ ```
90
+
91
+ The daemon listens on loopback by default. Review your local configuration before changing its host, connector roots, MCP permissions, or provider settings. Prefer `memories remember` for private text: a command-line argument may be visible in process lists and shell history.
92
+
93
+ ## Architecture and memory lifecycle
94
+
95
+ Input flows through privacy scanning, extraction, temporal acceptance, and SQLite persistence. Retrieval reads the memory store through ephemeral BM25 and dense indexes. The graph is a separate deterministic projection. The optimizer selects within a token budget; the compiler turns selected memories into context with fact/provenance evidence. The router sends the prepared request to a configured local or optional remote provider. Graph-assisted retrieval remains opt-in because its ranking performance is mixed in the local benchmark.
96
+
97
+ See [architecture](docs/architecture.md), [explainability](docs/explainability.md), and [RAG inspector](docs/rag-inspector.md) for exact boundaries. Explainability and inspection describe ContextOS preparation, not an unseen provider wire payload. Provider dispatch is `NOT_ATTEMPTED` during inspection.
98
+
99
+ ## Privacy and integrations
100
+
101
+ Secret scanning does not universally redact private filesystem paths. Selected graph/provenance output surfaces suppress path-shaped labels; terminal sanitization removes control sequences. These are separate output protections.
102
+
103
+ Secrets are scanned before accepted memory is stored; the legacy `--skip-secret-scan` request field does not disable that boundary. Metadata dashboards omit memory bodies, prompts, source paths, and credentials by default. Explicit content-view commands can reveal private text on your terminal; use them deliberately. Local storage is not a substitute for OS disk encryption or trusted local-user access.
104
+
105
+ Credential-free local file and JSON/JSONL connectors must be configured explicitly; none are registered by default. See [connectors](docs/connectors.md). MCP is disabled by default, uses STDIO, and separates read, write, and telemetry permissions; see [MCP](docs/mcp.md). Ollama and compatible local endpoints are first-class options; external providers are optional. FakeProvider supports deterministic tests/demo and is identified as simulated.
106
+
107
+ ## Terminal product
108
+
109
+ | Command | Purpose |
110
+ | --- | --- |
111
+ | `contextos status`, `health`, `doctor` | Daemon state and non-destructive diagnostics |
112
+ | `contextos stats [--model ID] [--provider ID] [--compare] [--today/--week]` | Measured activity by provider/model and token basis |
113
+ | `contextos monitor [--model ID] [--provider ID]` | Poll the same bounded local dashboard |
114
+ | `contextos inspect "query" [--mode hybrid] [--graph] [--memory UUID] [--compare] [--json]` | Retrieval-to-compiler decision evidence |
115
+ | `contextos explain "query"`, `preview "query"` | Explain or preview context without provider dispatch |
116
+ | `contextos graph stats/search/show` | Projection statistics and bounded graph paths |
117
+ | `contextos memories current/conflicts/history` | Temporal metadata; content is opt-in |
118
+ | `contextos connectors list/status/sync` | Registered connector state and explicit sync |
119
+ | `contextos models list`, `contextos telemetry` | Provider inventory and measured invocation records |
120
+ | `contextos benchmark [--extended]`, `contextos demo` | Isolated synthetic evaluation and offline walkthrough |
121
+
122
+ The dashboard groups provider and model together and keeps context-token counts separate from provider-reported usage. A reduction bar appears only when a single known context tokenizer basis can be compared. Older telemetry with unknown provenance is not silently combined. `--today` is the current UTC day; `--week` is a rolling seven-day window. Session-wide history is not yet persisted as a distinct aggregate.
123
+ Connector status includes currently tracked source-item count; the existing schema does not retain last-sync accepted/unchanged/failed totals for historical display.
124
+
125
+ ## Evidence and limits
126
+
127
+ The [benchmark](docs/benchmarking.md) compares full history, vector, hybrid, hybrid+graph, and compiled ContextOS context on ten fixed questions at 100 and 1,000 synthetic memories. It reports relevance ranking and context tokens, not answer quality. The [security](docs/security.md), [performance](docs/performance.md), [demo](docs/demo.md), and [release checklist](docs/release-checklist.md) documents separate measured, simulated, and unverified claims. A 5,000-memory run is opt-in. No cloud deployment or commercial-provider proof is implied by the local test suite.
128
+
129
+ Answer quality is **NOT VALIDATED**. Live providers have not been comprehensively validated. Graph retrieval stays opt-in because the current synthetic benchmark shows a recall/latency tradeoff. Latest independent cold CLI startup was approximately **2.27 seconds median** on Windows; this is a local measurement, not a guarantee.
130
+
131
+ ## Development
132
+
133
+ ```powershell
134
+ python -m pytest tests -q -ra
135
+ python -m compileall -q src tests
136
+ python -m contextos.benchmarks.final
137
+ python -m pip check
138
+ git diff --check
139
+ ```
140
+
141
+ `make test`, `make lint`, `make typecheck`, and `make check` are available when `make` is installed. The package is currently version 1.0.0-rc2. License: MIT.
142
+
143
+ Background `contextos start` polls the initialized daemon for up to 10 seconds before reporting success. Existing-daemon readiness requires the health PID to match the verified recorded PID at the configured host and port. Startup and stop share a Windows/POSIX lifecycle lock with a 15-second acquisition timeout; concurrent starters wait, then report the verified daemon as already running. The persistent `contextos.lock` file is reusable and OS lock ownership is released on process exit. PID publication is atomic, and failed startup cleans up only its own process tree and matching PID state. `contextos doctor` can be run immediately after a successful start.
@@ -0,0 +1,100 @@
1
+ # ContextOS
2
+
3
+ ContextOS is a local-first, model-independent AI memory runtime. It is not a chatbot. It accepts activity through a privacy boundary, extracts candidate facts, resolves temporal state, persists memories in SQLite, and supplies bounded context to a chosen model. A local daemon exposes a CLI and API; optional STDIO MCP and explicitly configured connectors use the same memory pipeline.
4
+
5
+ ## Why it exists
6
+
7
+ Long conversation histories are expensive and can carry stale or contradictory facts. ContextOS combines lexical and dense retrieval, temporal eligibility, a token-aware selector, and a context compiler. It records what it selected and what it could prove, without claiming that fewer context tokens automatically improve answers or provider billing.
8
+
9
+ ## Quick start
10
+
11
+ Python 3.12+ is declared; this release has been verified on Windows with Python 3.13. Python 3.12 is not yet validated. Core defaults and the offline demo require no commercial API key, model download, or tokenizer cache.
12
+
13
+ ### Installation
14
+
15
+ Core lightweight installation (uses deterministic/local capabilities):
16
+ ```powershell
17
+ pip install contextos-memory-runtime
18
+ ```
19
+
20
+ Core defaults use hashed deterministic embeddings and word-token accounting labeled `APPROXIMATED`. These embeddings are an offline baseline; SentenceTransformer retrieval quality is not implied. FakeProvider is a local simulation.
21
+
22
+ Rebuilt RC2 archives are approximately **220 KB for the wheel** and **300 KB for the sdist** (decimal units). These sizes exclude installed dependencies. Run `python tools/validate_release_artifacts.py` after a rebuild for exact byte counts.
23
+
24
+ With optional local sentence-transformers embeddings extra:
25
+ ```powershell
26
+ pip install "contextos-memory-runtime[embeddings]"
27
+ ```
28
+
29
+ Installing the extra does not change defaults. Explicitly set `[embedding]` with `model = "all-MiniLM-L6-v2"` in your ContextOS `config.toml` to select it; first use may download that model unless provisioned locally. Exact context counting is also opt-in: set `[token_counter]` with `encoding = "cl100k_base"` or `"o200k_base"`, and provision its tokenizer cache before offline use. Default `encoding = "deterministic"` needs no tokenizer assets.
30
+
31
+ For development:
32
+ ```powershell
33
+ pip install -e ".[dev]"
34
+ ```
35
+
36
+ ### Usage
37
+
38
+ ```powershell
39
+ contextos demo
40
+ contextos start
41
+ contextos doctor
42
+ contextos stats
43
+ "I prefer concise technical explanations." | contextos memories remember
44
+ contextos inspect "What explanation style do I prefer?"
45
+ contextos stop
46
+ ```
47
+
48
+ The daemon listens on loopback by default. Review your local configuration before changing its host, connector roots, MCP permissions, or provider settings. Prefer `memories remember` for private text: a command-line argument may be visible in process lists and shell history.
49
+
50
+ ## Architecture and memory lifecycle
51
+
52
+ Input flows through privacy scanning, extraction, temporal acceptance, and SQLite persistence. Retrieval reads the memory store through ephemeral BM25 and dense indexes. The graph is a separate deterministic projection. The optimizer selects within a token budget; the compiler turns selected memories into context with fact/provenance evidence. The router sends the prepared request to a configured local or optional remote provider. Graph-assisted retrieval remains opt-in because its ranking performance is mixed in the local benchmark.
53
+
54
+ See [architecture](docs/architecture.md), [explainability](docs/explainability.md), and [RAG inspector](docs/rag-inspector.md) for exact boundaries. Explainability and inspection describe ContextOS preparation, not an unseen provider wire payload. Provider dispatch is `NOT_ATTEMPTED` during inspection.
55
+
56
+ ## Privacy and integrations
57
+
58
+ Secret scanning does not universally redact private filesystem paths. Selected graph/provenance output surfaces suppress path-shaped labels; terminal sanitization removes control sequences. These are separate output protections.
59
+
60
+ Secrets are scanned before accepted memory is stored; the legacy `--skip-secret-scan` request field does not disable that boundary. Metadata dashboards omit memory bodies, prompts, source paths, and credentials by default. Explicit content-view commands can reveal private text on your terminal; use them deliberately. Local storage is not a substitute for OS disk encryption or trusted local-user access.
61
+
62
+ Credential-free local file and JSON/JSONL connectors must be configured explicitly; none are registered by default. See [connectors](docs/connectors.md). MCP is disabled by default, uses STDIO, and separates read, write, and telemetry permissions; see [MCP](docs/mcp.md). Ollama and compatible local endpoints are first-class options; external providers are optional. FakeProvider supports deterministic tests/demo and is identified as simulated.
63
+
64
+ ## Terminal product
65
+
66
+ | Command | Purpose |
67
+ | --- | --- |
68
+ | `contextos status`, `health`, `doctor` | Daemon state and non-destructive diagnostics |
69
+ | `contextos stats [--model ID] [--provider ID] [--compare] [--today/--week]` | Measured activity by provider/model and token basis |
70
+ | `contextos monitor [--model ID] [--provider ID]` | Poll the same bounded local dashboard |
71
+ | `contextos inspect "query" [--mode hybrid] [--graph] [--memory UUID] [--compare] [--json]` | Retrieval-to-compiler decision evidence |
72
+ | `contextos explain "query"`, `preview "query"` | Explain or preview context without provider dispatch |
73
+ | `contextos graph stats/search/show` | Projection statistics and bounded graph paths |
74
+ | `contextos memories current/conflicts/history` | Temporal metadata; content is opt-in |
75
+ | `contextos connectors list/status/sync` | Registered connector state and explicit sync |
76
+ | `contextos models list`, `contextos telemetry` | Provider inventory and measured invocation records |
77
+ | `contextos benchmark [--extended]`, `contextos demo` | Isolated synthetic evaluation and offline walkthrough |
78
+
79
+ The dashboard groups provider and model together and keeps context-token counts separate from provider-reported usage. A reduction bar appears only when a single known context tokenizer basis can be compared. Older telemetry with unknown provenance is not silently combined. `--today` is the current UTC day; `--week` is a rolling seven-day window. Session-wide history is not yet persisted as a distinct aggregate.
80
+ Connector status includes currently tracked source-item count; the existing schema does not retain last-sync accepted/unchanged/failed totals for historical display.
81
+
82
+ ## Evidence and limits
83
+
84
+ The [benchmark](docs/benchmarking.md) compares full history, vector, hybrid, hybrid+graph, and compiled ContextOS context on ten fixed questions at 100 and 1,000 synthetic memories. It reports relevance ranking and context tokens, not answer quality. The [security](docs/security.md), [performance](docs/performance.md), [demo](docs/demo.md), and [release checklist](docs/release-checklist.md) documents separate measured, simulated, and unverified claims. A 5,000-memory run is opt-in. No cloud deployment or commercial-provider proof is implied by the local test suite.
85
+
86
+ Answer quality is **NOT VALIDATED**. Live providers have not been comprehensively validated. Graph retrieval stays opt-in because the current synthetic benchmark shows a recall/latency tradeoff. Latest independent cold CLI startup was approximately **2.27 seconds median** on Windows; this is a local measurement, not a guarantee.
87
+
88
+ ## Development
89
+
90
+ ```powershell
91
+ python -m pytest tests -q -ra
92
+ python -m compileall -q src tests
93
+ python -m contextos.benchmarks.final
94
+ python -m pip check
95
+ git diff --check
96
+ ```
97
+
98
+ `make test`, `make lint`, `make typecheck`, and `make check` are available when `make` is installed. The package is currently version 1.0.0-rc2. License: MIT.
99
+
100
+ Background `contextos start` polls the initialized daemon for up to 10 seconds before reporting success. Existing-daemon readiness requires the health PID to match the verified recorded PID at the configured host and port. Startup and stop share a Windows/POSIX lifecycle lock with a 15-second acquisition timeout; concurrent starters wait, then report the verified daemon as already running. The persistent `contextos.lock` file is reusable and OS lock ownership is released on process exit. PID publication is atomic, and failed startup cleans up only its own process tree and matching PID state. `contextos doctor` can be run immediately after a successful start.
@@ -0,0 +1,37 @@
1
+ # Architecture
2
+
3
+ ContextOS is a single-machine daemon, not a chatbot or hosted account service. The CLI talks to the FastAPI daemon on the configured loopback address. `wire_services()` composes SQLite repositories, privacy/extraction/temporal services, retrieval indexes, graph projection, optimizer, compiler, model routing, telemetry, connectors, explainability, and inspection.
4
+
5
+ ## Write and read paths
6
+
7
+ ```text
8
+ CLI / API / opt-in MCP / configured connector
9
+ -> IngestionPipeline (privacy scan, rule-based candidate extraction, provenance event)
10
+ -> TemporalMemoryService.accept() (per-candidate durable transaction)
11
+ -> SQLite memories and relations
12
+ -> retrieval/graph projection invalidation
13
+
14
+ query -> index synchronization -> BM25 + local dense embeddings -> eligibility
15
+ -> rank fusion -> optional graph augmentation -> token-aware selection
16
+ -> fact compiler -> optional provider routing -> invocation telemetry
17
+ ```
18
+
19
+ The rule-based extractor and deterministic graph entity/relation extractor are deliberately narrower than an LLM. Ambiguous or unsupported facts may not be extracted. Multiple candidate writes are individually atomic, not one transaction for the entire request. Connector cursor advancement follows its own failure semantics; see [connectors](connectors.md).
20
+
21
+ ## Storage, retrieval, and time
22
+
23
+ SQLite schema version 7 stores memories, events, temporal relations, graph projection/supports, connector state, and model-invocation telemetry. The database uses WAL. BM25 and in-memory dense indexes are rebuildable from SQLite. The retrieval index synchronizer checks a persisted-memory fingerprint, then rebuilds as needed. Candidate hydration uses bounded batches on the SQLite repository, with the repository-protocol fallback still supported. This avoids per-candidate SQLite round trips but index synchronization still scales with corpus size.
24
+
25
+ The current temporal scope excludes stale states by default. Explicit historical/all scopes can retrieve them. The temporal peer lookup checks the latest active same-subject/property peer, without the old 500-row truncation. The graph is a persisted deterministic projection of memories and relations with a dirty bit and bounded traversal (maximum three hops, explicit node/edge limits). Graph-assisted ranking is opt-in, not a global default.
26
+
27
+ ## Context and evidence
28
+
29
+ The optimizer chooses candidate memories under a budget; the compiler emits bounded facts/context. These are distinct operations: selected memories need not contribute a compiled fact. Phase 13 explainability uses recorded retrieval, temporal, graph, optimizer, and compiler evidence. Phase 14 inspection adds a structured, bounded read-only view over one such execution. Its optional comparison runs additional retrieval strategies explicitly. Neither mechanism sends a provider request. A logical request fingerprint is not a hash of provider wire bytes; downstream adapter transformations cannot be inferred from it. Unknown candidate or graph membership remains `NOT_AVAILABLE`.
30
+
31
+ Telemetry records model/provider identifiers, counts, timing, measurement source/tokenizer, status, and errors, not raw prompts/responses. A provider-reported input count and a local preflight/context count are different measurement bases and must not be added together. The dashboard groups by provider, model, and context measurement basis. See [RAG inspector](rag-inspector.md) and [explainability](explainability.md).
32
+
33
+ ## Integrations and trust
34
+
35
+ MCP is optional STDIO with separately configured permissions. Local-file and JSON import connectors are explicit and credential-free; no connector is registered by default. FakeProvider is a simulation for tests/demo. Ollama and OpenAI-compatible local endpoints can be configured; external providers remain optional. The daemon is not a multi-user authorization boundary. Loopback binding, trusted local-user access, filesystem permissions, and disk encryption remain operational responsibilities.
36
+
37
+ No schema migration was added for Phases 14-18. The inspector, dashboards, benchmark, and demo use existing v7 data and temporary benchmark/demo databases.
@@ -0,0 +1,13 @@
1
+ # Benchmarking and claim boundaries
2
+
3
+ Run `python -m contextos.benchmarks.final` or `contextos benchmark --json`. The default measures isolated 100- and 1,000-memory SQLite corpora, ten fixed questions, and two iterations. `--extended` adds 5,000 records and is deliberately optional. Direct benchmark seeding creates synthetic fixture rows only in a disposable database; it is not a production ingestion path. All index sizes are checked against the intended corpus to prevent a false scale claim.
4
+
5
+ The five comparisons are full-history corpus order, dense/vector retrieval, hybrid retrieval, hybrid+graph, and ContextOS hybrid retrieval followed by optimizer/compiler (graph off by default). Fixed relevant memory IDs are declared before retrieval. Recall/Precision/Hit at 1/3/5/10, MRR, and NDCG@10 use those IDs. Duplicate relevant records are distinct IDs. Full history has no measured retrieval latency because it is represented as the corpus-order context baseline rather than an executed retrieval engine. Its token total includes all records, including a stale one; other strategies use current scope.
6
+
7
+ For compiled ContextOS context, `required_source_coverage` counts relevant labeled memory IDs that actually appear as source IDs of emitted facts, divided by all relevant IDs across queries. It is an evidence-coverage proxy, not semantic answer correctness; duplicate relevant memories are counted separately. `stale_source_facts` counts emitted facts attributed to explicitly stale fixture IDs. The separate temporal suite evaluates relation classification and current-state handling.
8
+
9
+ The run records candidate/optimized/compiled context token totals with one tokenizer basis, weighted reduction, selected-memory and emitted/excluded/duplicate-fact counts, provenance coverage, budget utilization, stale inclusion query count, mean/median/p95 local latency (including separate final retrieval/optimizer/compiler), top-10 hybrid/graph overlap, graph rebuild time, inspector time, index counts, SQLite bytes, process RSS, and graph counts. The fixture directly seeds memory rows without provenance events, so its zero provenance coverage is a property of the fixture, not of normal ingestion. Temporal relation classification is evaluated in the separate existing 17-case synthetic temporal harness included in final JSON. No generated answers are graded, so answer quality is `NOT AVAILABLE`. The benchmark does not prove realistic production quality, provider billing reduction, or universal graph value. Fixed data/IDs reduce variance, but timing and tied ranking can still vary by run and platform.
10
+
11
+ See [final measured run](final-benchmark.md) and its [condensed machine-readable snapshot](final-benchmark.json) for actual Windows/Python results, and [performance](performance.md) for before/after optimization scope. The benchmark command itself emits the full JSON schema. Existing canonical harnesses remain `python -m contextos.benchmarks.connectors`, `.terminal`, and `.explainability`.
12
+
13
+ The canonical final benchmark explicitly selects exact `cl100k_base` counting to preserve the published snapshot basis. Provision its tokenizer cache before offline benchmarking; first use otherwise requires an asset download. Core startup and the offline demo use the deterministic word approximation labeled `APPROXIMATED`, with no tokenizer assets.
@@ -0,0 +1,25 @@
1
+ # ContextOS connectors
2
+
3
+ Phase 11 supports credential-free local activity sources: explicitly configured
4
+ text/Markdown files, JSON/JSONL imports, and a deterministic fake connector for
5
+ tests. A connector returns bounded `ConnectorItem` values; it cannot write a
6
+ memory row. The manager sends every item through privacy, extraction, temporal
7
+ acceptance, persistence, and normal graph/index invalidation.
8
+
9
+ Source identity is `(connector_id, external_id)` plus revision/content hash.
10
+ Unchanged items are skipped before ingestion. Cursor and source identity state
11
+ are persisted without raw source content. Failed items stop cursor advancement;
12
+ the next sync retries from the prior safe cursor.
13
+
14
+ Local files are restricted to configured resolved roots, an extension allowlist
15
+ (`.txt`, `.md`, `.json`, `.jsonl`), deterministic ordering, UTF-8 decoding, and
16
+ a file-size limit. Symlink targets outside an allowed root are excluded.
17
+
18
+ Source deletion records the source item as deleted. The default policy keeps
19
+ derived memories: deleting a source is not a user request to purge memory.
20
+ Disabling a connector stops sync and retains state and memories. Future policy
21
+ may explicitly expire source-bound memory; automatic purge is never performed.
22
+
23
+ Connector input, metadata, Markdown, HTML-like text, and URLs are untrusted
24
+ data. Privacy scanner findings, raw content, credentials, and secret values are
25
+ not stored in cursor state or connector telemetry.
@@ -0,0 +1,7 @@
1
+ # Offline golden demo
2
+
3
+ Run `contextos demo --json` or `python -m contextos.demo` without starting a daemon, cloud account, API key, or local model server. It creates a temporary SQLite database, wires the normal deterministic local services, and removes the directory when finished. No real user data is read or modified.
4
+
5
+ The demo sends a preference and a later changed preference through ingestion and temporal acceptance, syncs a fake connector twice to show accepted then unchanged behavior, builds and traverses the graph, runs explainability and inspection without provider dispatch, invokes FakeProvider once, queries its telemetry, and confirms a secret-shaped input is rejected. The output reports measured IDs/counts/statuses and inspector token diff. FakeProvider response is simulated, not a production-provider proof. The synthetic demo is a walkthrough, not a benchmark or answer-quality assessment.
6
+
7
+ The first run needs no network, model weights, or existing tokenizer cache. It explicitly selects deterministic embeddings and word-token accounting labeled `APPROXIMATED` with tokenizer `deterministic-word-approximation`. FakeProvider usage is simulated; approximate context counts are not exact BPE counts or provider-billed savings.
@@ -0,0 +1,73 @@
1
+ # ContextOS explainability
2
+
3
+ Phase 13 exposes an ephemeral, deterministic account of one retrieval, optimizer, and compiler execution. The final context is what ContextOS prepared for a downstream model. Standalone explanations and previews mark provider dispatch as `NOT_ATTEMPTED`; actual provider dispatch evidence is captured only when a real model invocation executes through `ModelService.ask(..., explain=True)`. The explanation trace has a server-generated UUID; timings and trace IDs can vary while ranking and decision fields remain deterministic for unchanged inputs and configuration.
4
+
5
+ ## What the trace can establish
6
+
7
+ ### 1. Retrieval & Channels
8
+ For memories returned by the retrieval stage, the trace reports lexical BM25 rank and raw score, dense rank and raw score, fused score, reciprocal rank contributions, metadata adjustments, rank, and source strategies. BM25 and dense scores use different scales; compare ranks/fused scores rather than interpreting one as a calibrated percentage.
9
+
10
+ Candidate absence is categorized truthfully without guessing:
11
+ - `SELECTED`: Memory was retrieved, selected by the optimizer, and included in the compiled context.
12
+ - `COMPILER_EXCLUDED`: Memory was selected by the optimizer but the compiler's persisted-in-memory `included_memory_ids` evidence does not include it.
13
+ - `RETRIEVED_BUT_OPTIMIZER_EXCLUDED`: Memory was retrieved, but optimizer excluded it due to token budget or redundancy constraints.
14
+ - `RESULT_LIMIT`: Memory was present in the fused/reranked candidate set before the final top-K truncation cut it.
15
+ - `CHANNEL_NOT_RETRIEVED`: Memory is indexed and eligible, but was not surfaced by bounded candidate channels for this query.
16
+ - `INDEX_NOT_PRESENT`: Memory is directly proven absent from both lexical BM25 and dense vector indices, and graph retrieval was not active for this explanation. Graph-channel membership is not inferred from those two stores.
17
+ - `TEMPORALLY_INELIGIBLE`: Stored memory state or temporal scope policy excluded the candidate.
18
+ - `NOT_AVAILABLE`: Evidence is insufficient to establish why the memory was absent.
19
+
20
+ Channel candidate IDs and pre-limit IDs are retained with explicit truncation flags. If a requested memory is outside a truncated snapshot, the explanation returns `NOT_AVAILABLE`; it does not guess between channel absence and result-limit exclusion.
21
+
22
+ ### 2. Graph Evidence
23
+ Graph traces capture structured path nodes and edges collected during graph traversal without second-pass querying:
24
+ - Traversal bounds: strictly bounded to `max_hops <= 3` (up to 3 edges and up to 4 nodes per path), `max_nodes <= 100`, and `max_edges <= 250`.
25
+ - `GraphPathNode`: `node_id`, `node_type`, safe sanitized label, and `project_scope` (if stored). Missing labels are reported as `null` or excluded; labels are never invented.
26
+ - `GraphPathEdge`: `edge_type`, `confidence`, `supporting_memory_ids`, and `project_scope`.
27
+ - `GraphCandidateEvidence`: `seed_memory_id`, `candidate_memory_id`, `hop_count`, `graph_score`, `path_nodes`, `path_edges`, and `scope_match` evidence.
28
+
29
+ ### 3. Temporal Evidence
30
+ The structured `TemporalEvidenceResolver` reports provable persisted state and relations:
31
+ - Lifecycle status (`active`, `superseded`, `expired`, `deleted`, `purged`).
32
+ - Stored relations (`CORRECTS`, `COEXISTS_WITH`, `CONTRADICTS`, `SUPERSEDES`).
33
+ - Related endpoint state (`present`, `deleted`, or `missing`) resolved for either direction of a persisted relation.
34
+ - Replacement memory ID (`replacement_memory_id`) when superseded.
35
+ - Observed and effective timestamps.
36
+ - Acceptance rationale is strictly `null` (no fictional historical reasoning or motives are inferred).
37
+
38
+ ### 4. Compiler transformations
39
+ Compiler actions are recorded from emitted and excluded compiler facts: `RAW_INCLUDED`, `DEDUPLICATED`, `MERGED`, `COMPRESSED`, `RESCUED_FROM_OVERSIZED_MEMORY`, and `DROPPED`. Exact raw inclusion, merge source count, rescue input kind, and exclusion reason are evidenced. When output differs from the original memory but the compiler does not record whether it was clause extraction or compression, the trace reports `UNKNOWN` rather than claiming `COMPRESSED`. The trace does not invent a transformation for a memory with no compiler fact; optimizer selection and compiler inclusion are reported separately.
40
+
41
+ ### 5. Provider Dispatch Proof & Receipt
42
+ For standalone explanations (`contextos explain`, MCP `contextos_explain_context`, `POST /api/v1/explain`), the dispatch state is strictly:
43
+ - `ProviderDispatchState.NOT_ATTEMPTED`: "prepared by ContextOS; no provider dispatch attempted"
44
+
45
+ For real model invocations (`ModelService.ask`), evidence tracks:
46
+ - `state`:
47
+ - `REQUEST_CONSTRUCTED`: Context and ModelRequest constructed and fingerprinted.
48
+ - `DISPATCH_ATTEMPTED`: Provider invocation `generate()` was attempted at the adapter boundary (does not claim verified wire delivery).
49
+ - `RESPONSE_RECEIVED`: Downstream provider returned a successful response.
50
+ - `DISPATCH_FAILED`: Provider invocation raised an error or timed out. Evidence is attached to the raised Python exception and returned in the structured `/ask` API error response when available. SQLite telemetry records only generic `status="error"`, not the receipt.
51
+ - `compiled_context_sha256`: SHA-256 fingerprint of the exact compiled context text.
52
+ - `logical_request_sha256`: SHA-256 fingerprint of the actual `ModelRequest` fields observed immediately before provider invocation, represented as system prompt, compiled context, and user prompt. Provider adapters subsequently transform this into provider-specific payloads; this is not a wire-payload hash.
53
+ - `context_match`: Boolean proof confirming the actual downstream `ModelRequest` object carried the exact compiled ContextOS context string before hashing.
54
+ - `preflight_input_tokens`: Preflight token count calculated before dispatch.
55
+ - `provider_input_tokens`: Token count returned directly by the downstream provider (kept separate from ContextOS context tokens).
56
+
57
+ ## Privacy
58
+
59
+ Responses omit raw prompt text, private credentials, source URIs, connector metadata, and filesystem paths by default. Provenance includes only whitelisted source categories and event UUIDs; unknown source values are replaced with `unknown`. Terminal output is sanitized against ANSI CSI, OSC titles, OSC hyperlinks, carriage returns, backspaces, nulls, and bidi overrides. Traces are ephemeral (never persisted), and no schema v8 migration is required.
60
+
61
+ ## Use
62
+
63
+ ```powershell
64
+ contextos explain "What are my Python preferences?" --budget 1000 --mode hybrid --limit 25
65
+ contextos explain "What are my Python preferences?" --json
66
+ contextos explain "What are my Python preferences?" --show-content
67
+ contextos explain "What are my Python preferences?" --memory-id 12345678-1234-4234-8234-123456789012
68
+ contextos explain "What did I previously prefer?" --temporal-scope historical
69
+ contextos preview "What are my Python preferences?" --explain
70
+ ```
71
+
72
+ The API accepts `POST /api/v1/explain` with `query`, `mode`, `budget`, `limit`, `graph`, `temporal_scope`, `include_content`, and optional `target_memory_id`. Query length is capped at 10,000 characters, budget at 8,000 tokens, and candidates at 100.
73
+
@@ -0,0 +1,46 @@
1
+ {
2
+ "label": "LOCAL SYNTHETIC DETERMINISTIC BENCHMARK",
3
+ "command": "python -m contextos.benchmarks.final",
4
+ "environment": {"python": "3.13.5", "platform": "Windows-11-10.0.26200-SP0"},
5
+ "queries": 10,
6
+ "iterations": 2,
7
+ "tokenizer": "cl100k_base",
8
+ "scope": "One measured local run; condensed snapshot; full JSON emitted by command",
9
+ "corpora": {
10
+ "100": {
11
+ "lexical_index_count": 100,
12
+ "dense_index_count": 100,
13
+ "graph_nodes": 48,
14
+ "graph_edges": 55,
15
+ "graph_rebuild_ms": 76.887,
16
+ "sqlite_bytes": 278528,
17
+ "process_rss_bytes": 108990464,
18
+ "strategies": {
19
+ "full_history": {"recall_at_5": 0.3, "mrr": 0.2482, "ndcg_at_10": 0.2312, "candidate_tokens": 34040, "compiled_tokens": 34040, "mean_ms": null},
20
+ "vector": {"recall_at_5": 0.8, "mrr": 0.6867, "ndcg_at_10": 0.6932, "candidate_tokens": 3102, "compiled_tokens": 3102, "mean_ms": 20.369},
21
+ "hybrid": {"recall_at_5": 0.95, "mrr": 0.7833, "ndcg_at_10": 0.8263, "candidate_tokens": 2854, "compiled_tokens": 2854, "mean_ms": 20.852},
22
+ "hybrid_graph": {"recall_at_5": 0.6, "mrr": 0.4472, "ndcg_at_10": 0.5388, "candidate_tokens": 2956, "compiled_tokens": 2956, "mean_ms": 72.693},
23
+ "contextos": {"recall_at_5": 0.9, "mrr": 0.8, "ndcg_at_10": 0.7806, "candidate_tokens": 2854, "optimized_tokens": 1082, "compiled_tokens": 868, "weighted_token_reduction": 0.6959, "mean_ms": 24.419, "facts_emitted": 82, "required_source_coverage": 0.8571, "stale_source_facts": 0, "provenance_coverage": 0}
24
+ }
25
+ },
26
+ "1000": {
27
+ "lexical_index_count": 1000,
28
+ "dense_index_count": 1000,
29
+ "graph_nodes": 228,
30
+ "graph_edges": 235,
31
+ "graph_rebuild_ms": 544.252,
32
+ "sqlite_bytes": 1069056,
33
+ "process_rss_bytes": 122015744,
34
+ "strategies": {
35
+ "full_history": {"recall_at_5": 0.3, "mrr": 0.2482, "ndcg_at_10": 0.2312, "candidate_tokens": 343640, "compiled_tokens": 343640, "mean_ms": null},
36
+ "vector": {"recall_at_5": 0.6, "mrr": 0.6461, "ndcg_at_10": 0.6378, "candidate_tokens": 3098, "compiled_tokens": 3098, "mean_ms": 193.776},
37
+ "hybrid": {"recall_at_5": 0.9, "mrr": 0.7833, "ndcg_at_10": 0.7971, "candidate_tokens": 2884, "compiled_tokens": 2884, "mean_ms": 196.635},
38
+ "hybrid_graph": {"recall_at_5": 0.5, "mrr": 0.3483, "ndcg_at_10": 0.4069, "candidate_tokens": 3022, "compiled_tokens": 3022, "mean_ms": 435.317},
39
+ "contextos": {"recall_at_5": 0.85, "mrr": 0.8, "ndcg_at_10": 0.7568, "candidate_tokens": 2884, "optimized_tokens": 762, "compiled_tokens": 650, "weighted_token_reduction": 0.7746, "mean_ms": 207.143, "facts_emitted": 58, "required_source_coverage": 0.7857, "stale_source_facts": 0, "provenance_coverage": 0}
40
+ }
41
+ }
42
+ },
43
+ "temporal_fixture": {"cases": 17, "contextos_relation_accuracy": 1.0, "naive_relation_accuracy": 0.4117647058823529},
44
+ "answer_quality": null,
45
+ "answer_quality_reason": "No generated answers or independent grading"
46
+ }
@@ -0,0 +1,22 @@
1
+ # Final local benchmark snapshot
2
+
3
+ Command: `python -m contextos.benchmarks.final` (default: 100 and 1,000 records; two iterations of ten fixed queries). Environment: Windows 11 `10.0.26200`, Python 3.13.5, deterministic local embedding, SQLite, `cl100k_base` context counter. The temporary synthetic corpora include duplicate, stale, scoped preference, project/tool, negated deployment, and long-memory records. Ground-truth relevant IDs were specified before retrieval. BM25 and dense index counts were asserted equal to each corpus size. This is one local measured run, not production or model-answer evidence.
4
+
5
+ | Memories | Strategy | Recall@5 | MRR | NDCG@10 | Candidate tokens | Compiled tokens | Mean ms |
6
+ | ---: | --- | ---: | ---: | ---: | ---: | ---: | ---: |
7
+ | 100 | Full history | 0.300 | 0.248 | 0.231 | 34,040 | 34,040 | not measured |
8
+ | 100 | Vector | 0.800 | 0.687 | 0.693 | 3,102 | 3,102 | 20.369 |
9
+ | 100 | Hybrid | 0.950 | 0.783 | 0.826 | 2,854 | 2,854 | 20.852 |
10
+ | 100 | Hybrid + graph | 0.600 | 0.447 | 0.539 | 2,956 | 2,956 | 72.693 |
11
+ | 100 | ContextOS final | 0.900 | 0.800 | 0.781 | 2,854 | 868 | 24.419 |
12
+ | 1,000 | Full history | 0.300 | 0.248 | 0.231 | 343,640 | 343,640 | not measured |
13
+ | 1,000 | Vector | 0.600 | 0.646 | 0.638 | 3,098 | 3,098 | 193.776 |
14
+ | 1,000 | Hybrid | 0.900 | 0.783 | 0.797 | 2,884 | 2,884 | 196.635 |
15
+ | 1,000 | Hybrid + graph | 0.500 | 0.348 | 0.407 | 3,022 | 3,022 | 435.317 |
16
+ | 1,000 | ContextOS final | 0.850 | 0.800 | 0.757 | 2,884 | 650 | 207.143 |
17
+
18
+ ContextOS's weighted reduction of supplied context bodies was 69.59% at 100 and 77.46% at 1,000; compiled-body totals were 868 and 650 tokens. This is not provider-billed-token reduction. Over twenty query executions, the compiler emitted 82/58 facts and selected memories; ID-based required-source coverage was 85.71%/78.57%, stale-source facts were zero, and provenance coverage was 0 because direct fixture rows have no provenance event IDs. This does not describe normal ingestion. Context budget utilization was 21.7%/16.25% under a 200-token limit. The latest 1,000-record graph had 228 nodes/235 edges; SQLite size was 1,069,056 bytes and whole-process RSS was 122,015,744 bytes. These are synthetic evidence proxies, not answer correctness.
19
+
20
+ All twenty full-history query iterations included stale records, whereas the current-scope strategies did not. The separate seventeen-case temporal fixture yielded 1.0 ContextOS relation-classification/current-state accuracy versus 0.412/0.471 for two naive baselines; this is fixture-specific, not a real-world accuracy estimate.
21
+
22
+ Supported: ContextOS reduced measured supplied context tokens on this corpus; hybrid Recall@5 exceeded vector. Mixed: final compilation lost 0.05 Recall@5 versus raw hybrid at both sizes but had slightly higher MRR (0.800 versus 0.783). Opt-in graph augmentation degraded ranking and added latency here, so no graph-default change is justified. Answer quality was not measured. See [methodology](benchmarking.md) and the [condensed JSON snapshot](final-benchmark.json) before generalizing.