superlocalmemory 4.1.7 → 4.1.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +2 -2
- package/CHANGELOG.md +46 -0
- package/README.md +3 -3
- package/package.json +1 -1
- package/plugin/.claude-plugin/plugin.json +1 -1
- package/plugin/CLAUDE.md +3 -3
- package/plugin/agents/slm-governance-advisor.md +1 -1
- package/plugin/agents/slm-loop-runner.md +1 -1
- package/plugin/agents/slm-memory-advisor.md +1 -1
- package/plugin/agents/slm-optimize-advisor.md +1 -1
- package/plugin/requirements.txt +1 -1
- package/plugin/skills/slm-cache/SKILL.md +1 -1
- package/plugin/skills/slm-compress/SKILL.md +1 -1
- package/plugin/skills/slm-governance/SKILL.md +1 -1
- package/plugin/skills/slm-graph/SKILL.md +1 -1
- package/plugin/skills/slm-loop/SKILL.md +1 -1
- package/plugin/skills/slm-mesh/SKILL.md +1 -1
- package/plugin/skills/slm-profile/SKILL.md +1 -1
- package/plugin/skills/slm-recall/SKILL.md +1 -1
- package/plugin/skills/slm-remember/SKILL.md +1 -1
- package/plugin/skills/slm-scope/SKILL.md +1 -1
- package/plugin/skills/slm-session/SKILL.md +1 -1
- package/plugin/skills/slm-status/SKILL.md +1 -1
- package/plugin-src/agents/slm-memory-advisor.md +1 -1
- package/plugin-src/agents/slm-optimize-advisor.md +1 -1
- package/plugin-src/rules/AGENTS.md +1 -1
- package/plugin-src/skills/slm-cache/SKILL.md +1 -1
- package/plugin-src/skills/slm-compress/SKILL.md +1 -1
- package/plugin-src/skills/slm-governance/SKILL.md +1 -1
- package/plugin-src/skills/slm-graph/SKILL.md +1 -1
- package/plugin-src/skills/slm-loop/SKILL.md +1 -1
- package/plugin-src/skills/slm-mesh/SKILL.md +1 -1
- package/plugin-src/skills/slm-profile/SKILL.md +1 -1
- package/plugin-src/skills/slm-recall/SKILL.md +1 -1
- package/plugin-src/skills/slm-remember/SKILL.md +1 -1
- package/plugin-src/skills/slm-scope/SKILL.md +1 -1
- package/plugin-src/skills/slm-session/SKILL.md +1 -1
- package/plugin-src/skills/slm-status/SKILL.md +1 -1
- package/pyproject.toml +1 -1
- package/src/superlocalmemory/__init__.py +1 -1
- package/src/superlocalmemory/core/recall_pipeline.py +16 -13
- package/src/superlocalmemory/hooks/post_tool_outcome_hook.py +122 -0
- package/src/superlocalmemory/learning/reward.py +39 -4
- package/src/superlocalmemory/storage/migrations/M048_upcoming_holds_only_what_is_upcoming.py +22 -0
|
@@ -21,7 +21,7 @@
|
|
|
21
21
|
"license": "AGPL-3.0-or-later",
|
|
22
22
|
"name": "superlocalmemory",
|
|
23
23
|
"source": "./plugin",
|
|
24
|
-
"version": "4.1.
|
|
24
|
+
"version": "4.1.9"
|
|
25
25
|
},
|
|
26
26
|
{
|
|
27
27
|
"author": {
|
|
@@ -39,7 +39,7 @@
|
|
|
39
39
|
"license": "AGPL-3.0-or-later",
|
|
40
40
|
"name": "superlocalmemory-codex",
|
|
41
41
|
"source": "./codex-plugin",
|
|
42
|
-
"version": "4.1.
|
|
42
|
+
"version": "4.1.9"
|
|
43
43
|
}
|
|
44
44
|
]
|
|
45
45
|
}
|
package/CHANGELOG.md
CHANGED
|
@@ -5,6 +5,52 @@ All notable changes to SuperLocalMemory will be documented in this file.
|
|
|
5
5
|
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.0.0/),
|
|
6
6
|
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
7
|
|
|
8
|
+
## [4.1.9] — Silence is not a verdict
|
|
9
|
+
|
|
10
|
+
### Fixed
|
|
11
|
+
- **A recall nobody engaged with was scored as an average one.** How useful a recall
|
|
12
|
+
was is worked out from what happened next — whether a memory was quoted, edited, or
|
|
13
|
+
asked about again. When nothing at all had happened, that calculation still landed
|
|
14
|
+
on a middling score, and it was saved as if it had been measured. That is not a
|
|
15
|
+
blank: downstream it reads as mild approval, nudging results toward whatever was
|
|
16
|
+
shown and making the system steadily more certain it has no preference either way.
|
|
17
|
+
Recalls with nothing observed are now closed without a score, which is what the
|
|
18
|
+
batch pass already did — the two paths simply disagreed. On a store in daily use,
|
|
19
|
+
every one of 492 saved scores turned out to be this value, so nothing recorded there
|
|
20
|
+
reflected real use.
|
|
21
|
+
- **Signals recorded as explicitly absent were counted as present.** A recall whose
|
|
22
|
+
signals were all recorded as "did not happen" was treated as having something to
|
|
23
|
+
say, and scored, because the record existed even though everything in it was empty.
|
|
24
|
+
Both paths now ask whether anything actually happened rather than whether a record
|
|
25
|
+
was written.
|
|
26
|
+
|
|
27
|
+
## [4.1.8] — The same question, answered the same way
|
|
28
|
+
|
|
29
|
+
### Fixed
|
|
30
|
+
- **Recall could return the same memories in a different order each time it was
|
|
31
|
+
asked.** Adaptive ranking became the default in 4.1.7, which let a contextual
|
|
32
|
+
bandit choose how much to weigh each retrieval channel per query. That choice
|
|
33
|
+
is sampled from what an arm has learned, and arms that have settled no rewards
|
|
34
|
+
yet hold only their starting assumption — so the sample varied, and two
|
|
35
|
+
identical questions could weigh semantic against keyword evidence quite
|
|
36
|
+
differently for no reason drawn from anything observed. Adaptive ranking is
|
|
37
|
+
opt-in again: set `SLM_RANKING=v2-ensemble` to enable it. Nothing else about
|
|
38
|
+
retrieval changed, and no stored memory is affected.
|
|
39
|
+
- **One out-of-date label could make the whole store refuse to answer.** A
|
|
40
|
+
memory filed as a plan stops reading as a plan once its date passes — ordinary
|
|
41
|
+
use, and the maintenance pass re-reads those on its own. That check was
|
|
42
|
+
treated like a missing table: every route answered `503 Service unavailable`
|
|
43
|
+
until the daemon was restarted, and because the readiness answer was decided
|
|
44
|
+
once at startup, the repair that had already run could not lift it. The check
|
|
45
|
+
now reports what it is, a data quality signal, and the store keeps serving
|
|
46
|
+
while the next maintenance pass settles it.
|
|
47
|
+
- **Tool usage stopped being recorded.** Behavioural patterns, per-skill
|
|
48
|
+
performance, and the engagement measures that tell a recall whether it helped
|
|
49
|
+
all read the same record of which tools ran. Nothing on a normal install was
|
|
50
|
+
writing it, so those readings kept reporting from progressively older activity
|
|
51
|
+
and eventually stood still. Each completed tool call is recorded again, with
|
|
52
|
+
credential-shaped text removed and summaries capped before they are stored.
|
|
53
|
+
|
|
8
54
|
## [4.1.7] — Recall that improves with use
|
|
9
55
|
|
|
10
56
|
### Added
|
package/README.md
CHANGED
|
@@ -5,7 +5,7 @@
|
|
|
5
5
|
</picture>
|
|
6
6
|
</p>
|
|
7
7
|
|
|
8
|
-
<h1 align="center">SuperLocalMemory V4.1.
|
|
8
|
+
<h1 align="center">SuperLocalMemory V4.1.9</h1>
|
|
9
9
|
|
|
10
10
|
<h2 align="center">Rent the LLM. Own the memory.</h2>
|
|
11
11
|
|
|
@@ -27,12 +27,12 @@ guarantee here is stated as a falsifiable invariant, tested under an adversarial
|
|
|
27
27
|
negative control, and shipped with the harness that regenerates the evidence:
|
|
28
28
|
<code>python benchmark/run_all.py --trials 200 --output-dir results/</code>. What each experiment
|
|
29
29
|
does <em>not</em> exercise is stated too.</p>
|
|
30
|
-
<p align="center"><code>v4.1.
|
|
30
|
+
<p align="center"><code>v4.1.9</code> — one control plane: <strong>SLM-Mesh</strong> peer coordination · multi-scope memory (personal / shared / global) · profiles · Cache · Compress · 7-layer retrieval · code graph · Entity Explorer · skill evolution · Modes A/B/C · GDPR retention & audit chain · bounded loops — across CLI, MCP, dashboard, the <strong>Claude plugin</strong>, the <strong>Codex add-on</strong>, and documented IDE integrations.<br/>
|
|
31
31
|
Proxy: <code>slm wrap claude</code> · MCP: add <code>slm_compress</code> to your config · Skill: zero-config</p>
|
|
32
32
|
<p align="center"><strong>Four public arXiv preprints</strong> · V4: <a href="https://arxiv.org/abs/2608.08253">arXiv:2608.08253</a> · companion archive: <a href="https://zenodo.org/records/21853302">Zenodo 21853302</a> (<a href="https://doi.org/10.5281/zenodo.21853302">DOI 10.5281/zenodo.21853302</a>) · prior preprints: <a href="https://arxiv.org/abs/2603.02240">2603.02240</a> · <a href="https://arxiv.org/abs/2603.14588">2603.14588</a> · <a href="https://arxiv.org/abs/2604.04514">2604.04514</a>.</p>
|
|
33
33
|
|
|
34
34
|
<p align="center">
|
|
35
|
-
<a href="CHANGELOG.md"><img src="https://img.shields.io/badge/v4.1.
|
|
35
|
+
<a href="CHANGELOG.md"><img src="https://img.shields.io/badge/v4.1.9-Current_Release-2ea44f?style=for-the-badge&logo=checkmarx&logoColor=white" alt="v4.1.9 — Current Release"/></a>
|
|
36
36
|
<a href="https://arxiv.org/abs/2608.08253"><img src="https://img.shields.io/badge/arXiv-2608.08253-b31b1b?style=for-the-badge&logo=arxiv&logoColor=white" alt="SuperLocalMemory 4.0 paper on arXiv:2608.08253"/></a>
|
|
37
37
|
<a href="https://zenodo.org/records/21853302"><img src="https://img.shields.io/badge/Zenodo-10.5281%2Fzenodo.21853302-1682D4?style=for-the-badge&logo=zenodo&logoColor=white" alt="V4 paper on Zenodo: 10.5281/zenodo.21853302"/></a>
|
|
38
38
|
<a href="https://arxiv.org/abs/2603.14588"><img src="https://img.shields.io/badge/arXiv-2603.14588-b31b1b?style=for-the-badge&logo=arxiv&logoColor=white" alt="arXiv Paper"/></a>
|
package/package.json
CHANGED
package/plugin/CLAUDE.md
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
<!-- BEGIN SuperLocalMemory v4.1.
|
|
1
|
+
<!-- BEGIN SuperLocalMemory v4.1.9 -->
|
|
2
2
|
|
|
3
3
|
## SuperLocalMemory (SLM) — Agent Rules
|
|
4
4
|
|
|
@@ -39,6 +39,6 @@ slm-recall · slm-remember · slm-session · slm-status · slm-cache · slm-comp
|
|
|
39
39
|
### Subagents
|
|
40
40
|
slm-memory-advisor (memory decisions, session hygiene, scope/profile guidance) · slm-optimize-advisor (context compression + KV cache) · slm-governance-advisor (scope/roles/compliance/GDPR)
|
|
41
41
|
|
|
42
|
-
<!-- END SuperLocalMemory v4.1.
|
|
42
|
+
<!-- END SuperLocalMemory v4.1.9 -->
|
|
43
43
|
|
|
44
|
-
SuperLocalMemory v4.1.
|
|
44
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
|
@@ -77,4 +77,4 @@ slm-scope · slm-governance · slm-profile · slm-remember · slm-recall
|
|
|
77
77
|
# What NOT to do
|
|
78
78
|
Never session_init twice; never forget without dry-run preview; never store secrets; never bypass role checks; never claim an erasure succeeded without verifying via recall.
|
|
79
79
|
|
|
80
|
-
SuperLocalMemory v4.1.
|
|
80
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
|
@@ -46,4 +46,4 @@ slm-recall · slm-remember · slm-session · slm-scope · slm-profile · slm-gov
|
|
|
46
46
|
# What NOT to do
|
|
47
47
|
Never session_init twice; never forget dry_run=False without reporting preview; never dump a whole file into remember; never invent a memory; never claim "saved" without success:true / clean CLI exit; never bypass scope or governance restrictions.
|
|
48
48
|
|
|
49
|
-
SuperLocalMemory v4.1.
|
|
49
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
|
@@ -41,4 +41,4 @@ slm-compress · slm-cache · slm-status · slm-profile
|
|
|
41
41
|
# What NOT to do
|
|
42
42
|
Never compress code-for-edit/JSON-to-parse/<500 chars; never store secrets/ccr_ids; never let optimize failure block/alter the task; never claim a specific savings %; never carry ccr_ids across profile switches.
|
|
43
43
|
|
|
44
|
-
SuperLocalMemory v4.1.
|
|
44
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
package/plugin/requirements.txt
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
superlocalmemory==4.1.
|
|
1
|
+
superlocalmemory==4.1.9
|
|
@@ -46,4 +46,4 @@ slm-recall · slm-remember · slm-session · slm-scope · slm-profile · slm-gov
|
|
|
46
46
|
# What NOT to do
|
|
47
47
|
Never session_init twice; never forget dry_run=False without reporting preview; never dump a whole file into remember; never invent a memory; never claim "saved" without success:true / clean CLI exit; never bypass scope or governance restrictions.
|
|
48
48
|
|
|
49
|
-
SuperLocalMemory v4.1.
|
|
49
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
|
@@ -41,4 +41,4 @@ slm-compress · slm-cache · slm-status · slm-profile
|
|
|
41
41
|
# What NOT to do
|
|
42
42
|
Never compress code-for-edit/JSON-to-parse/<500 chars; never store secrets/ccr_ids; never let optimize failure block/alter the task; never claim a specific savings %; never carry ccr_ids across profile switches.
|
|
43
43
|
|
|
44
|
-
SuperLocalMemory v4.1.
|
|
44
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
|
@@ -137,4 +137,4 @@ When the SLM MCP server is unavailable, use these CLI equivalents:
|
|
|
137
137
|
- **slm-optimize-advisor** — context compression and KV cache
|
|
138
138
|
- **slm-governance-advisor** — scope/role compliance, retention policies, GDPR
|
|
139
139
|
|
|
140
|
-
SuperLocalMemory v4.1.
|
|
140
|
+
SuperLocalMemory v4.1.9 · Qualixar · AGPL-3.0-or-later
|
package/pyproject.toml
CHANGED
|
@@ -32,7 +32,7 @@ if "OMP_NUM_THREADS" not in os.environ:
|
|
|
32
32
|
os.environ["OMP_NUM_THREADS"] = "2"
|
|
33
33
|
# ---------------------------------------------------------------------------
|
|
34
34
|
|
|
35
|
-
__version__ = "4.1.
|
|
35
|
+
__version__ = "4.1.9"
|
|
36
36
|
|
|
37
37
|
_REQUIRED_VERSIONS = {
|
|
38
38
|
"sentence_transformers": "5.3.0",
|
|
@@ -340,23 +340,26 @@ class _ReadOnlyLearningView:
|
|
|
340
340
|
def _resolve_ranking_mode(env: "dict[str, str] | os._Environ[str]") -> str:
|
|
341
341
|
"""Map the ``SLM_RANKING`` env var to a canonical mode.
|
|
342
342
|
|
|
343
|
-
``SLM_RANKING`` is an explicit operator policy
|
|
344
|
-
means
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
343
|
+
``SLM_RANKING`` is an explicit operator policy, and the absence of one
|
|
344
|
+
means ``off``: an upgrade must not start reordering results on its own.
|
|
345
|
+
|
|
346
|
+
Fixing the settlement path was necessary for the ensemble to be worth
|
|
347
|
+
enabling, but it is not sufficient. Settlement decides what an arm learns
|
|
348
|
+
*after* a query; the channel weights come from sampling that arm's
|
|
349
|
+
posterior *before* it. Until arms have posteriors shaped by real
|
|
350
|
+
observations, that sample is drawn from the prior, so two identical
|
|
351
|
+
queries can be weighted differently for no reason the store can justify —
|
|
352
|
+
variance with nothing bought by it. A memory system's answers should move
|
|
353
|
+
because the evidence moved.
|
|
354
|
+
|
|
355
|
+
So the ensemble stays opt-in until arms carry settled rewards. Set
|
|
356
|
+
``SLM_RANKING=v2-ensemble`` to enable it, or ``v1``/``v2`` for the
|
|
357
|
+
narrower passes.
|
|
355
358
|
"""
|
|
356
359
|
raw = (env.get("SLM_RANKING", "") or "").strip().lower()
|
|
357
360
|
if raw in _RANKING_MODES:
|
|
358
361
|
return raw
|
|
359
|
-
return "
|
|
362
|
+
return "off"
|
|
360
363
|
|
|
361
364
|
|
|
362
365
|
def apply_ranking(
|
|
@@ -83,6 +83,122 @@ def _validate(marker: str) -> str | None:
|
|
|
83
83
|
return None
|
|
84
84
|
|
|
85
85
|
|
|
86
|
+
# --- Behavioural telemetry -------------------------------------------------
|
|
87
|
+
# ``tool_events`` is the record of what the agent did, and several readers
|
|
88
|
+
# depend on it: assertion mining, skill-performance mining, and the engagement
|
|
89
|
+
# features that settle a recall against the actions which followed it. Each of
|
|
90
|
+
# those reads the table; none of them writes it. The only writers were an
|
|
91
|
+
# explicit ``log_tool_event`` call and a bulk importer, so on an ordinary
|
|
92
|
+
# install the table simply stopped receiving invocations and every reader
|
|
93
|
+
# quietly aged out with it.
|
|
94
|
+
#
|
|
95
|
+
# This hook already runs on exactly the right edge -- after each tool completes,
|
|
96
|
+
# with the session that ran it -- so it records the invocation here. It happens
|
|
97
|
+
# before the marker scan and independently of it: whether a recalled memory was
|
|
98
|
+
# named in the output decides what a *reward* is worth, not whether the action
|
|
99
|
+
# occurred at all.
|
|
100
|
+
_MAX_SUMMARY_LEN = 500
|
|
101
|
+
|
|
102
|
+
# Compiled once at import. This path now runs on every tool call rather than
|
|
103
|
+
# only on the ~1-in-5 that carry a marker, so per-call ``re`` compilation or a
|
|
104
|
+
# module import would be paid on the hot path for no benefit.
|
|
105
|
+
_SECRET_PATTERNS = (
|
|
106
|
+
(re.compile(r"\b(?:sk-|pk-|api[_-]?key[_-]?)[A-Za-z0-9_-]{10,}\b"),
|
|
107
|
+
"[REDACTED]"),
|
|
108
|
+
(re.compile(r"\b[A-Za-z0-9+/]{40,}={0,2}\b"), "[REDACTED]"),
|
|
109
|
+
(re.compile(r"password\s*[=:]\s*\S+", re.IGNORECASE),
|
|
110
|
+
"password=[REDACTED]"),
|
|
111
|
+
)
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def _scrub(text: str) -> str:
|
|
115
|
+
"""Strip credential-shaped substrings from telemetry text."""
|
|
116
|
+
for pattern, replacement in _SECRET_PATTERNS:
|
|
117
|
+
text = pattern.sub(replacement, text)
|
|
118
|
+
return text
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def _summary(raw: object) -> str:
|
|
122
|
+
"""Truncate-then-scrub ``raw`` for a summary column.
|
|
123
|
+
|
|
124
|
+
Truncation comes first so the regex cost is bounded by
|
|
125
|
+
``_MAX_SUMMARY_LEN`` and not by the size of a tool response.
|
|
126
|
+
"""
|
|
127
|
+
if raw is None:
|
|
128
|
+
return ""
|
|
129
|
+
if not isinstance(raw, str):
|
|
130
|
+
try:
|
|
131
|
+
import json as _json
|
|
132
|
+
raw = _json.dumps(raw, default=str)
|
|
133
|
+
except Exception:
|
|
134
|
+
try:
|
|
135
|
+
raw = str(raw)
|
|
136
|
+
except Exception:
|
|
137
|
+
return ""
|
|
138
|
+
return _scrub(raw[:_MAX_SUMMARY_LEN])
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def _record_tool_event(
|
|
142
|
+
session_id: str,
|
|
143
|
+
tool_name: str,
|
|
144
|
+
payload: dict,
|
|
145
|
+
response_text: str,
|
|
146
|
+
) -> bool:
|
|
147
|
+
"""Append one ``tool_events`` row. Never raises; returns whether it wrote.
|
|
148
|
+
|
|
149
|
+
Failure is silent by design: telemetry must never be the reason a tool call
|
|
150
|
+
reports a problem to the user, and the hook contract is to exit 0 whatever
|
|
151
|
+
happens.
|
|
152
|
+
"""
|
|
153
|
+
if not tool_name:
|
|
154
|
+
return False
|
|
155
|
+
try:
|
|
156
|
+
import sqlite3
|
|
157
|
+
from datetime import datetime, timezone
|
|
158
|
+
|
|
159
|
+
try:
|
|
160
|
+
from superlocalmemory.hooks.session_registry import (
|
|
161
|
+
resolve_active_profile,
|
|
162
|
+
)
|
|
163
|
+
profile_id = resolve_active_profile() or "default"
|
|
164
|
+
except Exception:
|
|
165
|
+
profile_id = "default"
|
|
166
|
+
|
|
167
|
+
project_path = ""
|
|
168
|
+
raw_cwd = payload.get("cwd")
|
|
169
|
+
if isinstance(raw_cwd, str):
|
|
170
|
+
project_path = raw_cwd[:_MAX_SUMMARY_LEN]
|
|
171
|
+
|
|
172
|
+
conn = sqlite3.connect(str(_memory_db_path()), timeout=0.05)
|
|
173
|
+
try:
|
|
174
|
+
conn.execute("PRAGMA busy_timeout=50")
|
|
175
|
+
conn.execute(
|
|
176
|
+
"INSERT INTO tool_events "
|
|
177
|
+
"(session_id, profile_id, project_path, tool_name, event_type,"
|
|
178
|
+
" input_summary, output_summary, duration_ms, metadata,"
|
|
179
|
+
" created_at) "
|
|
180
|
+
"VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
181
|
+
(
|
|
182
|
+
session_id,
|
|
183
|
+
profile_id,
|
|
184
|
+
project_path,
|
|
185
|
+
tool_name,
|
|
186
|
+
"complete",
|
|
187
|
+
_summary(payload.get("tool_input")),
|
|
188
|
+
_summary(response_text),
|
|
189
|
+
0,
|
|
190
|
+
"{}",
|
|
191
|
+
datetime.now(timezone.utc).isoformat(),
|
|
192
|
+
),
|
|
193
|
+
)
|
|
194
|
+
conn.commit()
|
|
195
|
+
finally:
|
|
196
|
+
conn.close()
|
|
197
|
+
return True
|
|
198
|
+
except Exception:
|
|
199
|
+
return False
|
|
200
|
+
|
|
201
|
+
|
|
86
202
|
def _inner_main() -> str:
|
|
87
203
|
"""Return an ``outcome`` string (for perf log); never raises."""
|
|
88
204
|
payload = read_stdin_json()
|
|
@@ -119,6 +235,12 @@ def _inner_main() -> str:
|
|
|
119
235
|
|
|
120
236
|
# Response scan — capped BEFORE regex (bound O(cap)).
|
|
121
237
|
response_text = summarize_response(payload.get("tool_response"))
|
|
238
|
+
|
|
239
|
+
# The invocation is recorded before anything is decided about markers. An
|
|
240
|
+
# action the agent took is a fact about the session on its own terms; the
|
|
241
|
+
# marker only decides whether it also settles a pending recall.
|
|
242
|
+
_record_tool_event(session_id, tool_name, payload, response_text)
|
|
243
|
+
|
|
122
244
|
if not response_text:
|
|
123
245
|
return "no_response"
|
|
124
246
|
|
|
@@ -113,6 +113,30 @@ _MAX_CLOCK_SKEW_MS: Final[int] = 60_000
|
|
|
113
113
|
# ---------------------------------------------------------------------------
|
|
114
114
|
|
|
115
115
|
|
|
116
|
+
#: Dwell below this is not engagement; the label formula uses the same bound.
|
|
117
|
+
_DWELL_THRESHOLD_MS: Final[int] = 2000
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
def _carries_evidence(signals: Mapping[str, object]) -> bool:
|
|
121
|
+
"""Did anything actually happen after this recall?
|
|
122
|
+
|
|
123
|
+
``_compute_label`` is built as ``0.5 + bonuses - penalties``, so a signal
|
|
124
|
+
set with nothing in it evaluates to exactly one half. That number is not an
|
|
125
|
+
absence of judgement, it is a confident one: it becomes a mid-strength
|
|
126
|
+
positive training label and a Beta update that tightens a posterior around
|
|
127
|
+
its prior. An empty dict and a dict of falsy values mean the same thing here
|
|
128
|
+
and must both be refused.
|
|
129
|
+
"""
|
|
130
|
+
if not signals:
|
|
131
|
+
return False
|
|
132
|
+
if signals.get("cite") or signals.get("edit") or signals.get("requery"):
|
|
133
|
+
return True
|
|
134
|
+
try:
|
|
135
|
+
return int(signals.get("dwell_ms", 0) or 0) >= _DWELL_THRESHOLD_MS
|
|
136
|
+
except (TypeError, ValueError):
|
|
137
|
+
return False
|
|
138
|
+
|
|
139
|
+
|
|
116
140
|
def _compute_label(signals: Mapping[str, object]) -> float:
|
|
117
141
|
"""Deterministic label in ``[0.0, 1.0]`` per the manifest A.1 formula.
|
|
118
142
|
|
|
@@ -141,8 +165,9 @@ def _compute_label(signals: Mapping[str, object]) -> float:
|
|
|
141
165
|
dwell_int = 0
|
|
142
166
|
|
|
143
167
|
dwell_bonus = 0.0
|
|
144
|
-
if dwell_int >=
|
|
145
|
-
dwell_bonus = min(
|
|
168
|
+
if dwell_int >= _DWELL_THRESHOLD_MS:
|
|
169
|
+
dwell_bonus = min(
|
|
170
|
+
0.15, 0.05 + (dwell_int - _DWELL_THRESHOLD_MS) / 80_000.0)
|
|
146
171
|
|
|
147
172
|
label = (
|
|
148
173
|
0.5
|
|
@@ -585,9 +610,19 @@ class EngagementRewardModel:
|
|
|
585
610
|
signals = json.loads(pending["signals_json"] or "{}")
|
|
586
611
|
except json.JSONDecodeError: # pragma: no cover — defensive
|
|
587
612
|
signals = {}
|
|
588
|
-
reward = _compute_label(signals)
|
|
589
613
|
now_ms = self._clock_ms()
|
|
590
614
|
timestamp_iso = _iso_from_ms(now_ms)
|
|
615
|
+
if not _carries_evidence(signals):
|
|
616
|
+
# Finalize so it stops being rescanned, but record no
|
|
617
|
+
# outcome: nothing was observed, and a score here would
|
|
618
|
+
# be invented rather than measured. Mirrors reap_stale.
|
|
619
|
+
conn.execute(
|
|
620
|
+
"UPDATE pending_outcomes "
|
|
621
|
+
"SET status = 'settled' WHERE outcome_id = ?",
|
|
622
|
+
(outcome_id,),
|
|
623
|
+
)
|
|
624
|
+
return _FALLBACK_REWARD
|
|
625
|
+
reward = _compute_label(signals)
|
|
591
626
|
|
|
592
627
|
# NOTE: Split INSERT across lines so the Stage-5b CI
|
|
593
628
|
# gate's single-line regex (LLD-00 §13) does not fire.
|
|
@@ -746,7 +781,7 @@ class EngagementRewardModel:
|
|
|
746
781
|
#
|
|
747
782
|
# The row is still finalized, so it stops being rescanned; it is
|
|
748
783
|
# simply not turned into evidence it never was.
|
|
749
|
-
if not signals:
|
|
784
|
+
if not _carries_evidence(signals):
|
|
750
785
|
settle_ids.append(row["outcome_id"])
|
|
751
786
|
continue
|
|
752
787
|
reward = _compute_label(signals)
|
package/src/superlocalmemory/storage/migrations/M048_upcoming_holds_only_what_is_upcoming.py
CHANGED
|
@@ -205,3 +205,25 @@ def verify(conn: sqlite3.Connection) -> bool:
|
|
|
205
205
|
)
|
|
206
206
|
return False
|
|
207
207
|
return True
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def blocks_serving(conn: sqlite3.Connection) -> bool:
|
|
211
|
+
"""Should a daemon refuse to serve while this check does not hold? No.
|
|
212
|
+
|
|
213
|
+
Nothing here is about schema. ``verify()`` reads ``content`` and compares
|
|
214
|
+
today's reading of it against the ``fact_type`` already stored, so a False
|
|
215
|
+
means some memories are filed as plans that no longer read as plans. Every
|
|
216
|
+
table and column a query needs is present either way; the store answers
|
|
217
|
+
normally, and at worst a handful of memories carry a stale label until the
|
|
218
|
+
next maintenance pass re-reads them.
|
|
219
|
+
|
|
220
|
+
The distinction matters because this is a standing guard over data that
|
|
221
|
+
ordinary use re-violates by design: a plan whose date passes stops reading
|
|
222
|
+
as upcoming, which is the rule working, not a fault. Treating that like a
|
|
223
|
+
missing table let one drifted row answer 503 on every route for as long as
|
|
224
|
+
the process lived — and because the readiness snapshot is taken once at
|
|
225
|
+
startup, the background pass that repairs the data could not lift the
|
|
226
|
+
refusal it caused. An outage produced by a quality check is worse than the
|
|
227
|
+
thing the check is for. Same reasoning as ``M043.blocks_serving``.
|
|
228
|
+
"""
|
|
229
|
+
return False
|