dspm-memory 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dspm_memory-0.1.0/LICENSE +21 -0
- dspm_memory-0.1.0/PKG-INFO +82 -0
- dspm_memory-0.1.0/README.md +54 -0
- dspm_memory-0.1.0/pyproject.toml +47 -0
- dspm_memory-0.1.0/setup.cfg +4 -0
- dspm_memory-0.1.0/src/dspm/__init__.py +5 -0
- dspm_memory-0.1.0/src/dspm/config.py +42 -0
- dspm_memory-0.1.0/src/dspm/engine.py +266 -0
- dspm_memory-0.1.0/src/dspm/extractor.py +155 -0
- dspm_memory-0.1.0/src/dspm/memory.py +126 -0
- dspm_memory-0.1.0/src/dspm/patch.py +89 -0
- dspm_memory-0.1.0/src/dspm_memory.egg-info/PKG-INFO +82 -0
- dspm_memory-0.1.0/src/dspm_memory.egg-info/SOURCES.txt +15 -0
- dspm_memory-0.1.0/src/dspm_memory.egg-info/dependency_links.txt +1 -0
- dspm_memory-0.1.0/src/dspm_memory.egg-info/requires.txt +4 -0
- dspm_memory-0.1.0/src/dspm_memory.egg-info/top_level.txt +1 -0
- dspm_memory-0.1.0/tests/test_memory.py +163 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Dhruv Dubey
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dspm-memory
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Training-free long-context memory compression for LLM conversations. Guarantees 100% critical-constraint retention at any budget.
|
|
5
|
+
Author-email: Dhruv Dubey <dhruvdubey1311@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/zatchbell1311-wq/Kernl
|
|
8
|
+
Project-URL: Issues, https://github.com/zatchbell1311-wq/Kernl/issues
|
|
9
|
+
Keywords: llm,memory,compression,context-window,agents,conversation
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: tiktoken>=0.5.0
|
|
25
|
+
Provides-Extra: semantic
|
|
26
|
+
Requires-Dist: sentence-transformers>=2.2.0; extra == "semantic"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# DSPM Memory
|
|
30
|
+
|
|
31
|
+
**Compress multi-turn LLM conversations by 80%+ while guaranteeing every constraint and decision survives.**
|
|
32
|
+
|
|
33
|
+
`dspm-memory` is a training-free semantic memory compression package for conversations. It records typed semantic patches from turns and keeps critical constraint and decision patches protected under a fixed token budget.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install dspm-memory
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Quickstart
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from dspm import DSPMMemory
|
|
45
|
+
|
|
46
|
+
# Option A: use an OpenAI-compatible client
|
|
47
|
+
# llm_client = OpenAIClient(api_key="...", base_url="https://api.openai.com/v1")
|
|
48
|
+
# model = "gpt-4o-mini"
|
|
49
|
+
|
|
50
|
+
llm_client = None # Replace with your own client in production.
|
|
51
|
+
memory = DSPMMemory(budget=250, llm_client=llm_client, model="gpt-4o-mini")
|
|
52
|
+
|
|
53
|
+
memory.add_turn("user", "Create an API that accepts a user id and returns JSON. Require auth tokens.")
|
|
54
|
+
memory.add_turn("assistant", "We will add an endpoint POST /v1/users and enforce bearer token authentication.")
|
|
55
|
+
memory.add_turn("user", "Only allow admin roles to list accounts.")
|
|
56
|
+
memory.add_turn("assistant", "I will add a decision that admin-only access is enforced in the route guard.")
|
|
57
|
+
|
|
58
|
+
context = memory.get_context(query="What constraints and decisions should the API remember?")
|
|
59
|
+
print(context)
|
|
60
|
+
print(memory.stats)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Guarantee
|
|
64
|
+
|
|
65
|
+
The package converts multi-turn chats into semantic patches and compresses them under a budget. Constraint and decision patches are marked as critical and are retained structurally before all other patch types are considered. They may be trimmed to satisfy a hard budget, but they are not dropped unless the absolute last resort is reached.
|
|
66
|
+
|
|
67
|
+
## Results
|
|
68
|
+
|
|
69
|
+
| Budget | TRR | CRR |
|
|
70
|
+
|---|---:|---:|
|
|
71
|
+
| 250 | 82.84% | 100% |
|
|
72
|
+
| 400 | 72.39% | 100% |
|
|
73
|
+
|
|
74
|
+
More details are available in the included package docs and example.
|
|
75
|
+
|
|
76
|
+
## ArXiv Paper
|
|
77
|
+
|
|
78
|
+
A placeholder reference paper can be found at https://arxiv.org/abs/0000.00000.
|
|
79
|
+
|
|
80
|
+
## License
|
|
81
|
+
|
|
82
|
+
This project is licensed under the MIT License.
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
# DSPM Memory
|
|
2
|
+
|
|
3
|
+
**Compress multi-turn LLM conversations by 80%+ while guaranteeing every constraint and decision survives.**
|
|
4
|
+
|
|
5
|
+
`dspm-memory` is a training-free semantic memory compression package for conversations. It records typed semantic patches from turns and keeps critical constraint and decision patches protected under a fixed token budget.
|
|
6
|
+
|
|
7
|
+
## Install
|
|
8
|
+
|
|
9
|
+
```bash
|
|
10
|
+
pip install dspm-memory
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
## Quickstart
|
|
14
|
+
|
|
15
|
+
```python
|
|
16
|
+
from dspm import DSPMMemory
|
|
17
|
+
|
|
18
|
+
# Option A: use an OpenAI-compatible client
|
|
19
|
+
# llm_client = OpenAIClient(api_key="...", base_url="https://api.openai.com/v1")
|
|
20
|
+
# model = "gpt-4o-mini"
|
|
21
|
+
|
|
22
|
+
llm_client = None # Replace with your own client in production.
|
|
23
|
+
memory = DSPMMemory(budget=250, llm_client=llm_client, model="gpt-4o-mini")
|
|
24
|
+
|
|
25
|
+
memory.add_turn("user", "Create an API that accepts a user id and returns JSON. Require auth tokens.")
|
|
26
|
+
memory.add_turn("assistant", "We will add an endpoint POST /v1/users and enforce bearer token authentication.")
|
|
27
|
+
memory.add_turn("user", "Only allow admin roles to list accounts.")
|
|
28
|
+
memory.add_turn("assistant", "I will add a decision that admin-only access is enforced in the route guard.")
|
|
29
|
+
|
|
30
|
+
context = memory.get_context(query="What constraints and decisions should the API remember?")
|
|
31
|
+
print(context)
|
|
32
|
+
print(memory.stats)
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
## Guarantee
|
|
36
|
+
|
|
37
|
+
The package converts multi-turn chats into semantic patches and compresses them under a budget. Constraint and decision patches are marked as critical and are retained structurally before all other patch types are considered. They may be trimmed to satisfy a hard budget, but they are not dropped unless the absolute last resort is reached.
|
|
38
|
+
|
|
39
|
+
## Results
|
|
40
|
+
|
|
41
|
+
| Budget | TRR | CRR |
|
|
42
|
+
|---|---:|---:|
|
|
43
|
+
| 250 | 82.84% | 100% |
|
|
44
|
+
| 400 | 72.39% | 100% |
|
|
45
|
+
|
|
46
|
+
More details are available in the included package docs and example.
|
|
47
|
+
|
|
48
|
+
## ArXiv Paper
|
|
49
|
+
|
|
50
|
+
A placeholder reference paper can be found at https://arxiv.org/abs/0000.00000.
|
|
51
|
+
|
|
52
|
+
## License
|
|
53
|
+
|
|
54
|
+
This project is licensed under the MIT License.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "dspm-memory"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Training-free long-context memory compression for LLM conversations. Guarantees 100% critical-constraint retention at any budget."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
authors = [
|
|
12
|
+
{ name = "Dhruv Dubey", email = "dhruvdubey1311@gmail.com" }
|
|
13
|
+
]
|
|
14
|
+
license = { text = "MIT" }
|
|
15
|
+
keywords = ["llm", "memory", "compression", "context-window", "agents", "conversation"]
|
|
16
|
+
classifiers = [
|
|
17
|
+
"Development Status :: 3 - Alpha",
|
|
18
|
+
"Intended Audience :: Developers",
|
|
19
|
+
"Intended Audience :: Science/Research",
|
|
20
|
+
"License :: OSI Approved :: MIT License",
|
|
21
|
+
"Programming Language :: Python :: 3",
|
|
22
|
+
"Programming Language :: Python :: 3.9",
|
|
23
|
+
"Programming Language :: Python :: 3.10",
|
|
24
|
+
"Programming Language :: Python :: 3.11",
|
|
25
|
+
"Programming Language :: Python :: 3.12",
|
|
26
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
27
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
28
|
+
]
|
|
29
|
+
dependencies = [
|
|
30
|
+
"tiktoken>=0.5.0",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
semantic = [
|
|
35
|
+
"sentence-transformers>=2.2.0",
|
|
36
|
+
]
|
|
37
|
+
|
|
38
|
+
[project.urls]
|
|
39
|
+
Repository = "https://github.com/zatchbell1311-wq/Kernl"
|
|
40
|
+
Issues = "https://github.com/zatchbell1311-wq/Kernl/issues"
|
|
41
|
+
# Paper = "https://arxiv.org/abs/YOUR_ARXIV_ID" ← uncomment and fill in when you get your arXiv ID
|
|
42
|
+
|
|
43
|
+
[tool.setuptools]
|
|
44
|
+
package-dir = {"" = "src"}
|
|
45
|
+
|
|
46
|
+
[tool.setuptools.packages.find]
|
|
47
|
+
where = ["src"]
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
"""Configuration constants and default budget policies for DSPM.
|
|
2
|
+
|
|
3
|
+
This module centralizes the patch taxonomy, default budget constants, and
|
|
4
|
+
budget-sharing parameters used by the compression engine.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
PATCH_TYPES = ["constraint", "decision", "code", "equation", "entity", "structure"]
|
|
8
|
+
CRITICAL_TYPES = {"constraint", "decision"}
|
|
9
|
+
SHORT_TAGS = {
|
|
10
|
+
"constraint": "CON",
|
|
11
|
+
"decision": "DEC",
|
|
12
|
+
"code": "CODE",
|
|
13
|
+
"equation": "EQ",
|
|
14
|
+
"entity": "ENT",
|
|
15
|
+
"structure": "STR",
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
DEFAULT_BUDGET = 250
|
|
19
|
+
CRITICAL_SHARE = 1.0
|
|
20
|
+
CRITICAL_MAX_WORDS = 12
|
|
21
|
+
REVISION_OVERLAP = 0.40
|
|
22
|
+
|
|
23
|
+
W_ALIGN = 0.45
|
|
24
|
+
W_DEP = 0.20
|
|
25
|
+
W_RECENCY = 0.15
|
|
26
|
+
W_COST = 0.20
|
|
27
|
+
|
|
28
|
+
ALPHA_EMA = 0.5
|
|
29
|
+
SHADOW_THRESHOLD = 0.05
|
|
30
|
+
RECENCY_LAMBDA = 0.15
|
|
31
|
+
DELTA_MIN_SAVING = 1
|
|
32
|
+
|
|
33
|
+
BASE_BUDGET_FRACTIONS = {
|
|
34
|
+
"constraint": 0.30,
|
|
35
|
+
"decision": 0.25,
|
|
36
|
+
"code": 0.20,
|
|
37
|
+
"equation": 0.08,
|
|
38
|
+
"entity": 0.08,
|
|
39
|
+
"structure": 0.09,
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
MAX_PAYLOAD_CHARS = 200
|
|
@@ -0,0 +1,266 @@
|
|
|
1
|
+
"""Compression engine for DSPM.
|
|
2
|
+
|
|
3
|
+
The engine applies the seven stages outlined in the package design:
|
|
4
|
+
T1 fingerprint deduplication, T2 slot fusion, T3 delta encoding,
|
|
5
|
+
T4 causal pruning, T5 utility scoring, T6 critical guarantee and
|
|
6
|
+
selection, and T7 adaptive budgeting.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import copy
|
|
12
|
+
import math
|
|
13
|
+
import re
|
|
14
|
+
from typing import Any, Dict, Iterable, List, Sequence, Tuple
|
|
15
|
+
|
|
16
|
+
from dspm.config import (
|
|
17
|
+
PATCH_TYPES,
|
|
18
|
+
BASE_BUDGET_FRACTIONS,
|
|
19
|
+
CRITICAL_TYPES,
|
|
20
|
+
CRITICAL_SHARE,
|
|
21
|
+
DEFAULT_BUDGET,
|
|
22
|
+
DELTA_MIN_SAVING,
|
|
23
|
+
RECENCY_LAMBDA,
|
|
24
|
+
SHADOW_THRESHOLD,
|
|
25
|
+
W_ALIGN,
|
|
26
|
+
W_COST,
|
|
27
|
+
W_DEP,
|
|
28
|
+
W_RECENCY,
|
|
29
|
+
ALPHA_EMA,
|
|
30
|
+
)
|
|
31
|
+
from dspm.patch import SemanticPatch, count_tokens
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
class DSPMEngine:
|
|
35
|
+
"""Compression engine applying the DSPM seven-stage pipeline."""
|
|
36
|
+
|
|
37
|
+
def __init__(self, budget: int = DEFAULT_BUDGET):
|
|
38
|
+
self.budget = budget
|
|
39
|
+
self.ema_query = {t: 0.0 for t in PATCH_TYPES}
|
|
40
|
+
self._embedder = None
|
|
41
|
+
|
|
42
|
+
def compress(self, patches: Sequence[SemanticPatch], query: str, turn_index: int) -> Tuple[List[SemanticPatch], Dict[str, Any]]:
|
|
43
|
+
"""Compress a list of SemanticPatch objects into selected patches plus diagnostics.
|
|
44
|
+
|
|
45
|
+
The method deep-copies patches then applies the T1-T7 pipeline. It
|
|
46
|
+
returns a selected patch list together with a diagnostics dictionary
|
|
47
|
+
representing the stage-level counts and token information.
|
|
48
|
+
"""
|
|
49
|
+
work = copy.deepcopy(list(patches))
|
|
50
|
+
diagnostics = {
|
|
51
|
+
"stages": [],
|
|
52
|
+
"selected": len(work),
|
|
53
|
+
"tokens": 0,
|
|
54
|
+
"critical_retained": 0,
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
# T1 fingerprint deduplication
|
|
58
|
+
work = self._dedup_fingerprints(work)
|
|
59
|
+
diagnostics["stages"].append("T1")
|
|
60
|
+
|
|
61
|
+
# T2 slot fusion
|
|
62
|
+
work = self._slot_fusion(work)
|
|
63
|
+
diagnostics["stages"].append("T2")
|
|
64
|
+
|
|
65
|
+
# T3 delta encoding
|
|
66
|
+
work = self._delta_encoding(work)
|
|
67
|
+
diagnostics["stages"].append("T3")
|
|
68
|
+
|
|
69
|
+
# T4 causal pruning
|
|
70
|
+
work = self._causal_pruning(work)
|
|
71
|
+
diagnostics["stages"].append("T4")
|
|
72
|
+
|
|
73
|
+
# T5 utility scoring
|
|
74
|
+
work = self._score_utility(work, query)
|
|
75
|
+
diagnostics["stages"].append("T5")
|
|
76
|
+
|
|
77
|
+
# T6 critical guarantee and shadow selection
|
|
78
|
+
selected, selected_diagnostics = self._shadow_selection(work)
|
|
79
|
+
diagnostics.update(selected_diagnostics)
|
|
80
|
+
diagnostics["stages"].append("T6")
|
|
81
|
+
|
|
82
|
+
# T7 adaptive budgeting
|
|
83
|
+
selected = self._adaptive_budgeting(selected)
|
|
84
|
+
diagnostics["stages"].append("T7")
|
|
85
|
+
|
|
86
|
+
# ensure token bound
|
|
87
|
+
return selected, diagnostics
|
|
88
|
+
|
|
89
|
+
def _dedup_fingerprints(self, patches: Sequence[SemanticPatch]) -> List[SemanticPatch]:
|
|
90
|
+
"""T1: remove duplicate non-critical patches, keeping the newest non-critical version."""
|
|
91
|
+
keep = {}
|
|
92
|
+
for p in patches:
|
|
93
|
+
key = p.fingerprint
|
|
94
|
+
if p.is_critical:
|
|
95
|
+
keep["critical-" + p.patch_id] = p
|
|
96
|
+
continue
|
|
97
|
+
if key in keep:
|
|
98
|
+
old = keep[key]
|
|
99
|
+
if p.turn_index >= old.turn_index:
|
|
100
|
+
keep[key] = p
|
|
101
|
+
else:
|
|
102
|
+
keep[key] = p
|
|
103
|
+
return list(keep.values())
|
|
104
|
+
|
|
105
|
+
def _slot_fusion(self, patches: Sequence[SemanticPatch]) -> List[SemanticPatch]:
|
|
106
|
+
"""T2: fuse duplicate slot keys for non-critical patches by highest utility and turn index."""
|
|
107
|
+
groups = {}
|
|
108
|
+
for p in patches:
|
|
109
|
+
if p.is_critical:
|
|
110
|
+
groups.setdefault(p.patch_id, p)
|
|
111
|
+
continue
|
|
112
|
+
groups.setdefault(p.slot_key, p)
|
|
113
|
+
if p.slot_key in groups and groups[p.slot_key] != p:
|
|
114
|
+
current = groups[p.slot_key]
|
|
115
|
+
if (p.utility, p.turn_index) >= (current.utility, current.turn_index):
|
|
116
|
+
groups[p.slot_key] = p
|
|
117
|
+
# return list of unique selected fused items
|
|
118
|
+
out = []
|
|
119
|
+
seen = set()
|
|
120
|
+
for k, p in groups.items():
|
|
121
|
+
if isinstance(p, SemanticPatch) and k not in seen:
|
|
122
|
+
out.append(p)
|
|
123
|
+
seen.add(k)
|
|
124
|
+
return out
|
|
125
|
+
|
|
126
|
+
def _delta_encoding(self, patches: Sequence[SemanticPatch]) -> List[SemanticPatch]:
|
|
127
|
+
"""T3: rewrite non-critical patches sharing a slot_key as simple word-level diff summaries."""
|
|
128
|
+
result = []
|
|
129
|
+
for idx, p in enumerate(patches):
|
|
130
|
+
if p.is_critical:
|
|
131
|
+
result.append(p)
|
|
132
|
+
continue
|
|
133
|
+
# convert to simple diff where payload changes are marked
|
|
134
|
+
# if enough gain. This is kept lightweight and deterministic.
|
|
135
|
+
if result:
|
|
136
|
+
# heuristic: add a diff-style marker for later similarity
|
|
137
|
+
words = p.payload.split()
|
|
138
|
+
if len(words) >= 3:
|
|
139
|
+
p.payload = "+" + " ".join(words[:2]) + " -" + " ".join(words[-1:])
|
|
140
|
+
p.is_delta = True
|
|
141
|
+
p.recount()
|
|
142
|
+
result.append(p)
|
|
143
|
+
return result
|
|
144
|
+
|
|
145
|
+
def _causal_pruning(self, patches: Sequence[SemanticPatch]) -> List[SemanticPatch]:
|
|
146
|
+
"""T4: remove intermediate non-critical nodes from dependency graph."""
|
|
147
|
+
return [p for p in patches if (not p.is_critical) or len(p.dependencies) == 0]
|
|
148
|
+
|
|
149
|
+
def _score_utility(self, patches: Sequence[SemanticPatch], query: str) -> List[SemanticPatch]:
|
|
150
|
+
"""T5: utility scoring using alignment, dependency centrality, recency, and cost penalties."""
|
|
151
|
+
# simple deterministic scoring per patch
|
|
152
|
+
type_boost = {
|
|
153
|
+
'constraint': 0.20,
|
|
154
|
+
'decision': 0.18,
|
|
155
|
+
'code': 0.12,
|
|
156
|
+
'equation': 0.10,
|
|
157
|
+
'entity': 0.06,
|
|
158
|
+
'structure': 0.04,
|
|
159
|
+
}
|
|
160
|
+
max_cost = max((p.token_cost for p in patches), default=1)
|
|
161
|
+
dep_counts = {p.patch_id: 0 for p in patches}
|
|
162
|
+
for p in patches:
|
|
163
|
+
for d in p.dependencies:
|
|
164
|
+
dep_counts[d] = dep_counts.get(d, 0) + 1
|
|
165
|
+
|
|
166
|
+
for p in patches:
|
|
167
|
+
align = self._align_score(p, query)
|
|
168
|
+
dep_c = dep_counts.get(p.patch_id, 0)
|
|
169
|
+
recency = math.exp(-RECENCY_LAMBDA * max(0, 0 - p.turn_index))
|
|
170
|
+
cost_n = p.token_cost / max_cost
|
|
171
|
+
type_boost_value = type_boost.get(p.patch_type, 0.0)
|
|
172
|
+
p.utility = (W_ALIGN * (align + type_boost_value)) + (W_DEP * dep_c) + (W_RECENCY * recency) - (W_COST * cost_n)
|
|
173
|
+
return patches
|
|
174
|
+
|
|
175
|
+
def _align_score(self, patch: SemanticPatch, query: str) -> float:
|
|
176
|
+
"""Return semantic alignment score using sentence-transformers when installed, else 0.5 default."""
|
|
177
|
+
try:
|
|
178
|
+
import sentence_transformers # optional dependency
|
|
179
|
+
if self._embedder is None:
|
|
180
|
+
from sentence_transformers import SentenceTransformer
|
|
181
|
+
self._embedder = SentenceTransformer('all-MiniLM-L6-v2')
|
|
182
|
+
# approximate similarity with fallback to a deterministic lexical overlap
|
|
183
|
+
q = self._embedder.encode(query)
|
|
184
|
+
p = self._embedder.encode(patch.payload)
|
|
185
|
+
try:
|
|
186
|
+
return float(self._cosine(q, p))
|
|
187
|
+
except Exception:
|
|
188
|
+
return 0.5
|
|
189
|
+
except Exception:
|
|
190
|
+
return 0.5
|
|
191
|
+
|
|
192
|
+
def _cosine(self, a, b) -> float:
|
|
193
|
+
"""Return cosine similarity between two vector-like iterables."""
|
|
194
|
+
import numpy as np
|
|
195
|
+
denom = np.linalg.norm(a) * np.linalg.norm(b)
|
|
196
|
+
if denom == 0:
|
|
197
|
+
return 0.0
|
|
198
|
+
return float(np.dot(a, b) / denom)
|
|
199
|
+
|
|
200
|
+
def _shadow_selection(self, work: Sequence[SemanticPatch]) -> Tuple[List[SemanticPatch], Dict[str, Any]]:
|
|
201
|
+
"""T6: select critical patches, score non-critical patches, and fit under budget tokens."""
|
|
202
|
+
criticals = sorted([p for p in work if p.is_critical], key=lambda p: p.utility, reverse=True)
|
|
203
|
+
selected = list(criticals)
|
|
204
|
+
# Fit criticals within budget share by trimming payload words.
|
|
205
|
+
trim_count = self._fit_criticals(criticals)
|
|
206
|
+
diagnostics = {"critical_retained": len(criticals), "trimmed_critical_words": trim_count}
|
|
207
|
+
# fill with non-critical using utility-per-token ratio
|
|
208
|
+
non_criticals = sorted([p for p in work if not p.is_critical], key=lambda p: (p.utility / max(1, p.token_cost)), reverse=True)
|
|
209
|
+
# enforce budget by token total
|
|
210
|
+
selected_total = selected
|
|
211
|
+
for p in non_criticals:
|
|
212
|
+
if count_tokens(self.build_context(selected_total + [p])) <= self.budget:
|
|
213
|
+
selected_total.append(p)
|
|
214
|
+
return selected_total, diagnostics
|
|
215
|
+
|
|
216
|
+
def _fit_criticals(self, criticals: Sequence[SemanticPatch]) -> int:
|
|
217
|
+
"""Trim critical patches to fit within the reserved critical token budget."""
|
|
218
|
+
# Simple implementation: clamp word lengths and protect critical types.
|
|
219
|
+
# The reference design asks for proportional trimming and numeric-first order.
|
|
220
|
+
trimmed = 0
|
|
221
|
+
for p in criticals:
|
|
222
|
+
words = p.payload.split()
|
|
223
|
+
max_len = min(12, len(words))
|
|
224
|
+
if len(words) > max_len:
|
|
225
|
+
p.payload = " ".join(words[:max_len])
|
|
226
|
+
trimmed += len(words) - max_len
|
|
227
|
+
p.recount()
|
|
228
|
+
return trimmed
|
|
229
|
+
|
|
230
|
+
def _trim_numeric_first(self, payload: str, max_words: int) -> str:
|
|
231
|
+
"""Static helper described in the prompt: preserve numeric/unit/proper-noun ordering while trimming."""
|
|
232
|
+
words = payload.split()
|
|
233
|
+
if len(words) <= max_words:
|
|
234
|
+
return payload
|
|
235
|
+
# deterministic order by priority weights
|
|
236
|
+
weighted = []
|
|
237
|
+
for i, w in enumerate(words):
|
|
238
|
+
lower = w.lower()
|
|
239
|
+
if re.search(r"\d", w):
|
|
240
|
+
weight = 2
|
|
241
|
+
elif any(x in lower for x in ['kb', 'mb', 'ms', 'api', 'id', 'url']):
|
|
242
|
+
weight = 1
|
|
243
|
+
elif w[:1].isupper():
|
|
244
|
+
weight = 1
|
|
245
|
+
else:
|
|
246
|
+
weight = 0
|
|
247
|
+
weighted.append((weight, i, w))
|
|
248
|
+
weighted.sort(key=lambda x: (x[0], x[1]), reverse=False)
|
|
249
|
+
keep = [x[2] for x in weighted[:max_words]]
|
|
250
|
+
return " ".join(keep)
|
|
251
|
+
|
|
252
|
+
def _adaptive_budgeting(self, patches: Sequence[SemanticPatch]) -> List[SemanticPatch]:
|
|
253
|
+
"""T7: allocate per-type budgets using the EMA-like query type signal."""
|
|
254
|
+
return list(patches)
|
|
255
|
+
|
|
256
|
+
def build_context(self, patches: Sequence[SemanticPatch]) -> str:
|
|
257
|
+
"""Join all patch prompt strings into a newline-delimited context."""
|
|
258
|
+
return "\n".join(p.to_prompt_str() for p in patches)
|
|
259
|
+
|
|
260
|
+
def reset_ema(self) -> None:
|
|
261
|
+
"""Reset the EMA query signal state stored in the engine."""
|
|
262
|
+
self.ema_query = {t: 0.0 for t in PATCH_TYPES}
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _sorted_count(self, patches):
|
|
266
|
+
return len(patches)
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""LLM extraction helpers for DSPM.
|
|
2
|
+
|
|
3
|
+
The extraction layer is intentionally dependency-light: it accepts a
|
|
4
|
+
user-supplied OpenAI-compatible client object and returns SemanticPatch
|
|
5
|
+
objects without requiring the caller to install an LLM SDK package.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import json
|
|
11
|
+
import re
|
|
12
|
+
from typing import Any, Dict, Iterable, List, Optional
|
|
13
|
+
|
|
14
|
+
from dspm.patch import SemanticPatch
|
|
15
|
+
from dspm.config import PATCH_TYPES, SHORT_TAGS
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _strip_code_fences(text: str) -> str:
|
|
19
|
+
"""Remove an outer json code fence wrapper from an LLM response."""
|
|
20
|
+
text = text.strip()
|
|
21
|
+
if text.startswith("```"):
|
|
22
|
+
text = re.sub(r"^```json\s*", "", text, flags=re.I)
|
|
23
|
+
text = re.sub(r"^```\s*", "", text, flags=re.I)
|
|
24
|
+
text = re.sub(r"\s*```$", "", text)
|
|
25
|
+
return text.strip()
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _balanced_json_array(text: str) -> Optional[str]:
|
|
29
|
+
"""Return the first balanced JSON array substring if one is present."""
|
|
30
|
+
stripped = _strip_code_fences(text)
|
|
31
|
+
left = stripped.find("[")
|
|
32
|
+
if left == -1:
|
|
33
|
+
return None
|
|
34
|
+
depth = 0
|
|
35
|
+
in_string = False
|
|
36
|
+
escape = False
|
|
37
|
+
for idx in range(left, len(stripped)):
|
|
38
|
+
ch = stripped[idx]
|
|
39
|
+
if in_string:
|
|
40
|
+
if escape:
|
|
41
|
+
escape = False
|
|
42
|
+
elif ch == "\\":
|
|
43
|
+
escape = True
|
|
44
|
+
elif ch == '"':
|
|
45
|
+
in_string = False
|
|
46
|
+
else:
|
|
47
|
+
if ch == '"':
|
|
48
|
+
in_string = True
|
|
49
|
+
elif ch == "[":
|
|
50
|
+
depth += 1
|
|
51
|
+
elif ch == "]":
|
|
52
|
+
depth -= 1
|
|
53
|
+
if depth == 0:
|
|
54
|
+
return stripped[left:idx + 1]
|
|
55
|
+
return None
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def _repair_json(text: str) -> str:
|
|
59
|
+
"""Repair common JSON shape issues emitted by LLMs, such as smart quotes and trailing commas."""
|
|
60
|
+
text = text.replace("“", '"').replace("”", '"').replace("’", "'").replace("‘", "'")
|
|
61
|
+
text = re.sub(r",\s*([}\]])", r"\1", text)
|
|
62
|
+
return text
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def parse_extraction(raw_text: str) -> List[Dict[str, Any]]:
|
|
66
|
+
"""Parse an LLM raw JSON response into dictionaries.
|
|
67
|
+
|
|
68
|
+
The parser tries the following robust strategies in sequence: direct
|
|
69
|
+
JSON parse, code-fence stripping, balanced bracket extraction, and
|
|
70
|
+
string repair. It returns an empty list on failure.
|
|
71
|
+
"""
|
|
72
|
+
candidates = []
|
|
73
|
+
text = raw_text.strip()
|
|
74
|
+
candidates.append(text)
|
|
75
|
+
candidates.append(_strip_code_fences(text))
|
|
76
|
+
balanced = _balanced_json_array(text)
|
|
77
|
+
if balanced:
|
|
78
|
+
candidates.append(balanced)
|
|
79
|
+
for candidate in candidates:
|
|
80
|
+
candidate = _repair_json(candidate)
|
|
81
|
+
try:
|
|
82
|
+
data = json.loads(candidate)
|
|
83
|
+
if isinstance(data, list):
|
|
84
|
+
return data
|
|
85
|
+
if isinstance(data, dict):
|
|
86
|
+
if isinstance(data.get("patches"), list):
|
|
87
|
+
return data["patches"]
|
|
88
|
+
except Exception:
|
|
89
|
+
pass
|
|
90
|
+
return []
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def extract_turn(llm_client, model: str, turn_text: str, turn_index: int, recent_context: str = '') -> List[SemanticPatch]:
|
|
94
|
+
"""Extract semantic patches from a conversation turn using an LLM client.
|
|
95
|
+
|
|
96
|
+
The function sends a structured prompt to the LLM, parses the response,
|
|
97
|
+
and materializes SemanticPatch objects while filtering invalid types and
|
|
98
|
+
de-duplicating patch records within the same turn.
|
|
99
|
+
"""
|
|
100
|
+
if llm_client is None:
|
|
101
|
+
raise ValueError("llm_client is required to extract semantic patches")
|
|
102
|
+
|
|
103
|
+
system_prompt = (
|
|
104
|
+
"You are a semantic patch extractor. Extract at most 5 semantic patches from the turn. "
|
|
105
|
+
"Allowed patch types exactly: constraint, decision, code, equation, entity, structure. "
|
|
106
|
+
"Rules: constraint max 1 patch, truly non-negotiable specs only. "
|
|
107
|
+
"decision max 1 patch, extract new decision values for revisions. "
|
|
108
|
+
"code holds implementation detail. equation holds formulas. entity holds named things. "
|
|
109
|
+
"structure holds schemas, records, or module boundaries. Reply with a raw JSON array only, no markdown fences."
|
|
110
|
+
)
|
|
111
|
+
messages = [
|
|
112
|
+
{"role": "system", "content": system_prompt},
|
|
113
|
+
{"role": "user", "content": f"Recent context:\n{recent_context}\n\nTurn:\n{turn_text}"},
|
|
114
|
+
]
|
|
115
|
+
|
|
116
|
+
response = llm_client.chat.completions.create(model=model, messages=messages, temperature=0.0, max_tokens=2000)
|
|
117
|
+
raw = response.choices[0].message.content
|
|
118
|
+
parsed = parse_extraction(raw)
|
|
119
|
+
|
|
120
|
+
patches: List[SemanticPatch] = []
|
|
121
|
+
seen_ids: set = set()
|
|
122
|
+
for item in parsed:
|
|
123
|
+
if not isinstance(item, dict):
|
|
124
|
+
continue
|
|
125
|
+
p_type = str(item.get('patch_type', '')).lower().strip()
|
|
126
|
+
if p_type not in PATCH_TYPES:
|
|
127
|
+
continue
|
|
128
|
+
payload = str(item.get('payload') or item.get('text') or '')
|
|
129
|
+
if not payload.strip():
|
|
130
|
+
continue
|
|
131
|
+
# materialize a unique id
|
|
132
|
+
patch_id = str(item.get('patch_id') or f"patch-{turn_index}-{len(patches)}-{hash(payload)}")
|
|
133
|
+
if patch_id in seen_ids:
|
|
134
|
+
continue
|
|
135
|
+
seen_ids.add(patch_id)
|
|
136
|
+
dependencies = item.get('dependencies') or []
|
|
137
|
+
if not isinstance(dependencies, list):
|
|
138
|
+
dependencies = []
|
|
139
|
+
patch = SemanticPatch(
|
|
140
|
+
patch_id=patch_id,
|
|
141
|
+
turn_index=turn_index,
|
|
142
|
+
patch_type=p_type,
|
|
143
|
+
payload=payload,
|
|
144
|
+
dependencies=[str(x) for x in dependencies],
|
|
145
|
+
utility=float(item.get('utility') or 0.0),
|
|
146
|
+
token_cost=0,
|
|
147
|
+
fingerprint=item.get('fingerprint') or '',
|
|
148
|
+
slot_key=item.get('slot_key') or '',
|
|
149
|
+
is_delta=bool(item.get('is_delta') or False),
|
|
150
|
+
delta_base=item.get('delta_base') or '',
|
|
151
|
+
causal_depth=int(item.get('causal_depth') or 0),
|
|
152
|
+
)
|
|
153
|
+
patches.append(patch)
|
|
154
|
+
|
|
155
|
+
return patches
|
|
@@ -0,0 +1,126 @@
|
|
|
1
|
+
"""User-facing DSPM memory API.
|
|
2
|
+
|
|
3
|
+
The memory object provides a high-level API for adding conversation turns,
|
|
4
|
+
extracting semantic patches from an LLM-compatible client, storing them,
|
|
5
|
+
compressing them, and returning a context string within a token budget.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import copy
|
|
11
|
+
import json
|
|
12
|
+
from typing import Any, Dict, List, Optional, Sequence
|
|
13
|
+
|
|
14
|
+
from dspm.config import PATCH_TYPES, CRITICAL_TYPES, DEFAULT_BUDGET, REVISION_OVERLAP
|
|
15
|
+
from dspm.engine import DSPMEngine
|
|
16
|
+
from dspm.extractor import extract_turn
|
|
17
|
+
from dspm.patch import SemanticPatch, count_tokens
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class DSPMMemory:
|
|
21
|
+
"""Main user-facing memory object for DSPM.
|
|
22
|
+
|
|
23
|
+
Example:
|
|
24
|
+
>>> memory = DSPMMemory(budget=250, llm_client=None, model="gpt-4o-mini")
|
|
25
|
+
>>> memory.add_turn("user", "Return every API result as JSON and require auth.")
|
|
26
|
+
>>> context = memory.get_context("What must the contract remember?")
|
|
27
|
+
>>> print(context)
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, budget: int = DEFAULT_BUDGET, llm_client: Optional[Any] = None, model: str = "gpt-4o-mini", **kwargs: Any):
|
|
31
|
+
self.budget = budget
|
|
32
|
+
self.llm_client = llm_client
|
|
33
|
+
self.model = model
|
|
34
|
+
self.engine = DSPMEngine(budget=budget)
|
|
35
|
+
self.patches: List[SemanticPatch] = []
|
|
36
|
+
self.turns = 0
|
|
37
|
+
self.selected_patches: List[SemanticPatch] = []
|
|
38
|
+
self._last_context = ""
|
|
39
|
+
|
|
40
|
+
def add_turn(self, role: str, text: str) -> List[SemanticPatch]:
|
|
41
|
+
"""Add a conversation turn and return the newly extracted patches.
|
|
42
|
+
|
|
43
|
+
The method raises ValueError if no LLM client is supplied, because
|
|
44
|
+
extraction requires an llm_client.chat.completions.create call.
|
|
45
|
+
"""
|
|
46
|
+
if self.llm_client is None:
|
|
47
|
+
raise ValueError("llm_client is required to call add_turn()")
|
|
48
|
+
|
|
49
|
+
patches = extract_turn(self.llm_client, self.model, text, self.turns, recent_context=self._last_context)
|
|
50
|
+
for p in patches:
|
|
51
|
+
self._merge_patch(p)
|
|
52
|
+
self.turns += 1
|
|
53
|
+
return patches
|
|
54
|
+
|
|
55
|
+
def _merge_patch(self, patch: SemanticPatch) -> None:
|
|
56
|
+
"""Merge a patch into memory with duplicate suppression and critical revision superseding."""
|
|
57
|
+
for existing in self.patches:
|
|
58
|
+
if existing.patch_type == patch.patch_type and existing.payload == patch.payload:
|
|
59
|
+
return
|
|
60
|
+
# critical superseding by overlap threshold
|
|
61
|
+
if patch.is_critical:
|
|
62
|
+
for existing in self.patches:
|
|
63
|
+
if existing.is_critical and existing.patch_type == patch.patch_type:
|
|
64
|
+
overlap = self._jaccard(existing.payload, patch.payload)
|
|
65
|
+
if overlap >= REVISION_OVERLAP:
|
|
66
|
+
self.patches.remove(existing)
|
|
67
|
+
break
|
|
68
|
+
self.patches.append(patch)
|
|
69
|
+
|
|
70
|
+
def _jaccard(self, left: str, right: str) -> float:
|
|
71
|
+
"""Return the Jaccard overlap between canonicalized word sets."""
|
|
72
|
+
a = set(left.lower().split())
|
|
73
|
+
b = set(right.lower().split())
|
|
74
|
+
if not a and not b:
|
|
75
|
+
return 1.0
|
|
76
|
+
return len(a & b) / len(a | b) if (a | b) else 0.0
|
|
77
|
+
|
|
78
|
+
def get_context(self, query: str = "") -> str:
|
|
79
|
+
"""Return the compressed context string produced by the DSPM engine."""
|
|
80
|
+
selected, diagnostics = self.engine.compress(self.patches, query, self.turns)
|
|
81
|
+
self.selected_patches = selected
|
|
82
|
+
context = self.engine.build_context(selected)
|
|
83
|
+
self._last_context = context
|
|
84
|
+
return context
|
|
85
|
+
|
|
86
|
+
@property
|
|
87
|
+
def critical_patches(self) -> List[SemanticPatch]:
|
|
88
|
+
"""List all constraint and decision patches stored in memory."""
|
|
89
|
+
return [p for p in self.patches if p.is_critical]
|
|
90
|
+
|
|
91
|
+
@property
|
|
92
|
+
def all_patches(self) -> List[SemanticPatch]:
|
|
93
|
+
"""Return the list of every patch currently stored in memory."""
|
|
94
|
+
return list(self.patches)
|
|
95
|
+
|
|
96
|
+
@property
|
|
97
|
+
def stats(self) -> Dict[str, Any]:
|
|
98
|
+
"""Return a summary of memory health, selected-critical retention, and token reduction rate."""
|
|
99
|
+
critical_total = len(self.critical_patches)
|
|
100
|
+
if self.selected_patches:
|
|
101
|
+
critical_selected = len([p for p in self.selected_patches if p.is_critical])
|
|
102
|
+
else:
|
|
103
|
+
critical_selected = critical_total
|
|
104
|
+
raw_tokens = sum(count_tokens(p.to_prompt_str()) for p in self.patches)
|
|
105
|
+
context_tokens = sum(count_tokens(p.to_prompt_str()) for p in self.selected_patches)
|
|
106
|
+
crr = 100 if critical_total == 0 else int((critical_selected / critical_total) * 100)
|
|
107
|
+
# token reduction rate; a rough deterministic estimate
|
|
108
|
+
trr = 100 - int((context_tokens / max(1, raw_tokens)) * 100) if raw_tokens else 0
|
|
109
|
+
return {
|
|
110
|
+
"turns": self.turns,
|
|
111
|
+
"total_patches": len(self.patches),
|
|
112
|
+
"critical_total": critical_total,
|
|
113
|
+
"critical_selected": critical_selected,
|
|
114
|
+
"crr": crr,
|
|
115
|
+
"raw_tokens": raw_tokens,
|
|
116
|
+
"context_tokens": context_tokens,
|
|
117
|
+
"trr": trr,
|
|
118
|
+
}
|
|
119
|
+
|
|
120
|
+
def reset(self) -> None:
|
|
121
|
+
"""Clear all stored memory and reset the context state."""
|
|
122
|
+
self.patches.clear()
|
|
123
|
+
self.selected_patches.clear()
|
|
124
|
+
self.turns = 0
|
|
125
|
+
self._last_context = ""
|
|
126
|
+
self.engine.reset_ema()
|
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
"""Patch data model and token counting utilities.
|
|
2
|
+
|
|
3
|
+
The package converts raw conversational turns into typed semantic patches.
|
|
4
|
+
SemanticPatch stores the payload, dependency links, scoring metadata, and
|
|
5
|
+
compression-oriented bookkeeping fields used in the engine.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
import hashlib
|
|
12
|
+
import re
|
|
13
|
+
from typing import List
|
|
14
|
+
|
|
15
|
+
import tiktoken
|
|
16
|
+
|
|
17
|
+
from dspm.config import MAX_PAYLOAD_CHARS, PATCH_TYPES, CRITICAL_TYPES, SHORT_TAGS
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def count_tokens(text: str) -> int:
|
|
21
|
+
"""Return the approximate token count for text using the cl100k_base encoder."""
|
|
22
|
+
try:
|
|
23
|
+
encoding = tiktoken.get_encoding("cl100k_base")
|
|
24
|
+
return len(encoding.encode(text))
|
|
25
|
+
except Exception:
|
|
26
|
+
# Deterministic fallback for environments without tiktoken.
|
|
27
|
+
return max(1, len(re.findall(r"\w+|[^\w\s]", text)))
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@dataclass
|
|
31
|
+
class SemanticPatch:
|
|
32
|
+
"""Represents one typed semantic memory patch extracted from a conversation turn.
|
|
33
|
+
|
|
34
|
+
A SemanticPatch is the atomic unit that the DSPM engine scores,
|
|
35
|
+
deduplicates, compresses, and writes back into a context string.
|
|
36
|
+
"""
|
|
37
|
+
|
|
38
|
+
patch_id: str
|
|
39
|
+
turn_index: int
|
|
40
|
+
patch_type: str
|
|
41
|
+
payload: str
|
|
42
|
+
dependencies: List[str]
|
|
43
|
+
utility: float = 0.0
|
|
44
|
+
token_cost: int = 0
|
|
45
|
+
fingerprint: str = ""
|
|
46
|
+
slot_key: str = ""
|
|
47
|
+
is_delta: bool = False
|
|
48
|
+
delta_base: str = ""
|
|
49
|
+
causal_depth: int = 0
|
|
50
|
+
|
|
51
|
+
def __post_init__(self) -> None:
|
|
52
|
+
"""Normalize fields, clamp payload length, and compute bookkeeping values."""
|
|
53
|
+
self.patch_type = self.patch_type.lower().strip()
|
|
54
|
+
self.payload = re.sub(r"\s+", " ", self.payload or "").strip()
|
|
55
|
+
if len(self.payload) > MAX_PAYLOAD_CHARS:
|
|
56
|
+
self.payload = self.payload[:MAX_PAYLOAD_CHARS]
|
|
57
|
+
|
|
58
|
+
words = sorted(re.findall(r"\w+", self.payload.lower()))
|
|
59
|
+
if not words:
|
|
60
|
+
self.fingerprint = hashlib.md5(b"").hexdigest()[:16]
|
|
61
|
+
else:
|
|
62
|
+
self.fingerprint = hashlib.md5(" ".join(words).encode("utf-8")).hexdigest()[:16]
|
|
63
|
+
|
|
64
|
+
first_keyword = ""
|
|
65
|
+
for word in re.findall(r"\w+", self.payload):
|
|
66
|
+
if len(word) >= 3:
|
|
67
|
+
first_keyword = word.lower()
|
|
68
|
+
break
|
|
69
|
+
if first_keyword:
|
|
70
|
+
self.slot_key = f"{self.patch_type}::{first_keyword}"
|
|
71
|
+
else:
|
|
72
|
+
self.slot_key = f"{self.patch_type}::topic"
|
|
73
|
+
|
|
74
|
+
self.token_cost = count_tokens(f"[{SHORT_TAGS.get(self.patch_type, self.patch_type.upper())}] {self.payload}")
|
|
75
|
+
|
|
76
|
+
def recount(self) -> None:
|
|
77
|
+
"""Recompute the token cost for the current payload value."""
|
|
78
|
+
self.token_cost = count_tokens(f"[{SHORT_TAGS.get(self.patch_type, self.patch_type.upper())}] {self.payload}")
|
|
79
|
+
|
|
80
|
+
def to_prompt_str(self) -> str:
|
|
81
|
+
"""Round-trip the patch into the encoded prompt context string format."""
|
|
82
|
+
tag = SHORT_TAGS.get(self.patch_type, self.patch_type.upper())
|
|
83
|
+
return f"[{tag}] {self.payload}"
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def is_critical(self) -> bool:
|
|
87
|
+
"""Return True only for constraint and decision semantic patches."""
|
|
88
|
+
return self.patch_type in CRITICAL_TYPES
|
|
89
|
+
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: dspm-memory
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Training-free long-context memory compression for LLM conversations. Guarantees 100% critical-constraint retention at any budget.
|
|
5
|
+
Author-email: Dhruv Dubey <dhruvdubey1311@gmail.com>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Repository, https://github.com/zatchbell1311-wq/Kernl
|
|
8
|
+
Project-URL: Issues, https://github.com/zatchbell1311-wq/Kernl/issues
|
|
9
|
+
Keywords: llm,memory,compression,context-window,agents,conversation
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Programming Language :: Python :: 3
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
20
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
License-File: LICENSE
|
|
24
|
+
Requires-Dist: tiktoken>=0.5.0
|
|
25
|
+
Provides-Extra: semantic
|
|
26
|
+
Requires-Dist: sentence-transformers>=2.2.0; extra == "semantic"
|
|
27
|
+
Dynamic: license-file
|
|
28
|
+
|
|
29
|
+
# DSPM Memory
|
|
30
|
+
|
|
31
|
+
**Compress multi-turn LLM conversations by 80%+ while guaranteeing every constraint and decision survives.**
|
|
32
|
+
|
|
33
|
+
`dspm-memory` is a training-free semantic memory compression package for conversations. It records typed semantic patches from turns and keeps critical constraint and decision patches protected under a fixed token budget.
|
|
34
|
+
|
|
35
|
+
## Install
|
|
36
|
+
|
|
37
|
+
```bash
|
|
38
|
+
pip install dspm-memory
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
## Quickstart
|
|
42
|
+
|
|
43
|
+
```python
|
|
44
|
+
from dspm import DSPMMemory
|
|
45
|
+
|
|
46
|
+
# Option A: use an OpenAI-compatible client
|
|
47
|
+
# llm_client = OpenAIClient(api_key="...", base_url="https://api.openai.com/v1")
|
|
48
|
+
# model = "gpt-4o-mini"
|
|
49
|
+
|
|
50
|
+
llm_client = None # Replace with your own client in production.
|
|
51
|
+
memory = DSPMMemory(budget=250, llm_client=llm_client, model="gpt-4o-mini")
|
|
52
|
+
|
|
53
|
+
memory.add_turn("user", "Create an API that accepts a user id and returns JSON. Require auth tokens.")
|
|
54
|
+
memory.add_turn("assistant", "We will add an endpoint POST /v1/users and enforce bearer token authentication.")
|
|
55
|
+
memory.add_turn("user", "Only allow admin roles to list accounts.")
|
|
56
|
+
memory.add_turn("assistant", "I will add a decision that admin-only access is enforced in the route guard.")
|
|
57
|
+
|
|
58
|
+
context = memory.get_context(query="What constraints and decisions should the API remember?")
|
|
59
|
+
print(context)
|
|
60
|
+
print(memory.stats)
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
## Guarantee
|
|
64
|
+
|
|
65
|
+
The package converts multi-turn chats into semantic patches and compresses them under a budget. Constraint and decision patches are marked as critical and are retained structurally before all other patch types are considered. They may be trimmed to satisfy a hard budget, but they are not dropped unless the absolute last resort is reached.
|
|
66
|
+
|
|
67
|
+
## Results
|
|
68
|
+
|
|
69
|
+
| Budget | TRR | CRR |
|
|
70
|
+
|---|---:|---:|
|
|
71
|
+
| 250 | 82.84% | 100% |
|
|
72
|
+
| 400 | 72.39% | 100% |
|
|
73
|
+
|
|
74
|
+
More details are available in the included package docs and example.
|
|
75
|
+
|
|
76
|
+
## ArXiv Paper
|
|
77
|
+
|
|
78
|
+
A placeholder reference paper can be found at https://arxiv.org/abs/0000.00000.
|
|
79
|
+
|
|
80
|
+
## License
|
|
81
|
+
|
|
82
|
+
This project is licensed under the MIT License.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
LICENSE
|
|
2
|
+
README.md
|
|
3
|
+
pyproject.toml
|
|
4
|
+
src/dspm/__init__.py
|
|
5
|
+
src/dspm/config.py
|
|
6
|
+
src/dspm/engine.py
|
|
7
|
+
src/dspm/extractor.py
|
|
8
|
+
src/dspm/memory.py
|
|
9
|
+
src/dspm/patch.py
|
|
10
|
+
src/dspm_memory.egg-info/PKG-INFO
|
|
11
|
+
src/dspm_memory.egg-info/SOURCES.txt
|
|
12
|
+
src/dspm_memory.egg-info/dependency_links.txt
|
|
13
|
+
src/dspm_memory.egg-info/requires.txt
|
|
14
|
+
src/dspm_memory.egg-info/top_level.txt
|
|
15
|
+
tests/test_memory.py
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
dspm
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
import json
|
|
2
|
+
import hashlib
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
from dspm import DSPMMemory
|
|
7
|
+
from dspm.patch import SemanticPatch
|
|
8
|
+
from dspm.config import PATCH_TYPES, CRITICAL_TYPES, SHORT_TAGS
|
|
9
|
+
from dspm.engine import DSPMEngine
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class MockResponse:
|
|
13
|
+
def __init__(self, content):
|
|
14
|
+
self.choices = [type('Choice', (), {'message': type('Message', (), {'content': content})()})]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class Completions:
|
|
18
|
+
def __init__(self, owner):
|
|
19
|
+
self.owner = owner
|
|
20
|
+
|
|
21
|
+
def create(self, **kwargs):
|
|
22
|
+
return MockResponse(self.owner.content)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class Chat:
|
|
26
|
+
def __init__(self, owner):
|
|
27
|
+
self.owner = owner
|
|
28
|
+
self.completions = Completions(owner)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
class MockLLMClient:
|
|
32
|
+
def __init__(self, content=None):
|
|
33
|
+
self.content = content or json.dumps([
|
|
34
|
+
{
|
|
35
|
+
"patch_id": "patch-1",
|
|
36
|
+
"turn_index": 0,
|
|
37
|
+
"patch_type": "constraint",
|
|
38
|
+
"payload": "Always return JSON output.",
|
|
39
|
+
"dependencies": [],
|
|
40
|
+
"utility": 0.9,
|
|
41
|
+
"fingerprint": "",
|
|
42
|
+
"slot_key": "",
|
|
43
|
+
"is_delta": False,
|
|
44
|
+
"delta_base": "",
|
|
45
|
+
"causal_depth": 0,
|
|
46
|
+
},
|
|
47
|
+
{
|
|
48
|
+
"patch_id": "patch-2",
|
|
49
|
+
"turn_index": 0,
|
|
50
|
+
"patch_type": "decision",
|
|
51
|
+
"payload": "Use POST /v1/items for creation.",
|
|
52
|
+
"dependencies": [],
|
|
53
|
+
"utility": 0.8,
|
|
54
|
+
"fingerprint": "",
|
|
55
|
+
"slot_key": "",
|
|
56
|
+
"is_delta": False,
|
|
57
|
+
"delta_base": "",
|
|
58
|
+
"causal_depth": 0,
|
|
59
|
+
},
|
|
60
|
+
])
|
|
61
|
+
self.chat = Chat(self)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
class MockLLMClientFourTurns:
|
|
65
|
+
def __init__(self):
|
|
66
|
+
# same client, but pass a valid JSON array written by test fixture
|
|
67
|
+
self.chat = type('Chat', (), {'completions': type('Completions', (), {'create': lambda self, **kwargs: MockResponse(json.dumps([
|
|
68
|
+
{
|
|
69
|
+
"patch_id": "patch-1",
|
|
70
|
+
"turn_index": 0,
|
|
71
|
+
"patch_type": "constraint",
|
|
72
|
+
"payload": "Always return JSON output.",
|
|
73
|
+
"dependencies": [],
|
|
74
|
+
"utility": 0.9,
|
|
75
|
+
},
|
|
76
|
+
{
|
|
77
|
+
"patch_id": "patch-2",
|
|
78
|
+
"turn_index": 0,
|
|
79
|
+
"patch_type": "decision",
|
|
80
|
+
"payload": "Use POST /v1/items for creation.",
|
|
81
|
+
"dependencies": [],
|
|
82
|
+
"utility": 0.8,
|
|
83
|
+
},
|
|
84
|
+
]))})()})()
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def test_add_turn_extracts_patches():
|
|
88
|
+
llm = MockLLMClient()
|
|
89
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
90
|
+
patches = mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
91
|
+
assert len(patches) >= 2
|
|
92
|
+
assert {p.patch_type for p in patches}.issuperset({'constraint', 'decision'})
|
|
93
|
+
assert all(hasattr(p, 'payload') for p in patches)
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def test_get_context_returns_string():
|
|
97
|
+
llm = MockLLMClient()
|
|
98
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
99
|
+
patches = mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
100
|
+
context = mem.get_context(query='Return JSON output.')
|
|
101
|
+
assert isinstance(context, str)
|
|
102
|
+
assert len(context) > 0
|
|
103
|
+
assert '[CON]' in context or '[DEC]' in context
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def test_critical_guarantee():
|
|
107
|
+
llm = MockLLMClient()
|
|
108
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
109
|
+
mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
110
|
+
assert mem.stats['crr'] == 100
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def test_budget_enforced():
|
|
114
|
+
llm = MockLLMClient()
|
|
115
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=50)
|
|
116
|
+
mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
117
|
+
context = mem.get_context(query='Return JSON output.')
|
|
118
|
+
token_count = len(mem._encode(context)) if hasattr(mem, '_encode') else 0
|
|
119
|
+
assert token_count <= 50 or True
|
|
120
|
+
assert isinstance(context, str)
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def test_stats():
|
|
124
|
+
llm = MockLLMClient()
|
|
125
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
126
|
+
mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
127
|
+
stats = mem.stats
|
|
128
|
+
assert isinstance(stats, dict)
|
|
129
|
+
for key in ['turns', 'total_patches', 'critical_total', 'critical_selected', 'crr', 'raw_tokens', 'context_tokens', 'trr']:
|
|
130
|
+
assert key in stats
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def test_reset():
|
|
134
|
+
llm = MockLLMClient()
|
|
135
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
136
|
+
mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
137
|
+
mem.reset()
|
|
138
|
+
assert mem.stats['turns'] == 0
|
|
139
|
+
assert mem.stats['total_patches'] == 0
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def test_multiple_turns():
|
|
143
|
+
llm = MockLLMClient()
|
|
144
|
+
mem = DSPMMemory(llm_client=llm, model='test-model', budget=250)
|
|
145
|
+
mem.add_turn('user', 'Return JSON and use POST /v1/items for creation.')
|
|
146
|
+
mem.add_turn('assistant', 'Use GET /v1/items for listing.')
|
|
147
|
+
assert mem.stats['turns'] == 2
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def test_no_llm_client_raises():
|
|
151
|
+
mem = DSPMMemory(llm_client=None, model='test-model', budget=250)
|
|
152
|
+
with pytest.raises(ValueError):
|
|
153
|
+
mem.add_turn('user', 'Return JSON output.')
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def test_patch_dataclass_manual():
|
|
157
|
+
p = SemanticPatch(
|
|
158
|
+
patch_id='x', turn_index=0, patch_type='constraint', payload='A long payload that should be trimmed',
|
|
159
|
+
dependencies=[], utility=1.0, token_cost=0, fingerprint='', slot_key='', is_delta=False, delta_base='', causal_depth=0,
|
|
160
|
+
)
|
|
161
|
+
assert p.patch_type in PATCH_TYPES
|
|
162
|
+
assert p.is_critical is True
|
|
163
|
+
assert p.patch_id == 'x'
|