stillvalid 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- stillvalid/__init__.py +24 -0
- stillvalid/backfill.py +146 -0
- stillvalid/cascade.py +170 -0
- stillvalid/doc.py +27 -0
- stillvalid/history.py +142 -0
- stillvalid/integrations/__init__.py +5 -0
- stillvalid/integrations/common.py +185 -0
- stillvalid/integrations/langchain.py +86 -0
- stillvalid/integrations/llamaindex.py +81 -0
- stillvalid/mcp_server.py +247 -0
- stillvalid/probe.py +250 -0
- stillvalid/py.typed +0 -0
- stillvalid/refresh.py +124 -0
- stillvalid/survival.py +71 -0
- stillvalid/verdict.py +96 -0
- stillvalid-0.1.0.dist-info/METADATA +369 -0
- stillvalid-0.1.0.dist-info/RECORD +21 -0
- stillvalid-0.1.0.dist-info/WHEEL +5 -0
- stillvalid-0.1.0.dist-info/entry_points.txt +2 -0
- stillvalid-0.1.0.dist-info/licenses/LICENSE +21 -0
- stillvalid-0.1.0.dist-info/top_level.txt +1 -0
stillvalid/__init__.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""stillvalid — is this information still valid?
|
|
2
|
+
|
|
3
|
+
A validity layer between retrieval and action. Feed it what you know about
|
|
4
|
+
a document; it answers VALID / LIKELY_VALID / VERIFY / STALE / UNKNOWN with
|
|
5
|
+
evidence, via a cascade that prefers cheap deterministic checks and only
|
|
6
|
+
falls back to learned probability where nothing deterministic exists.
|
|
7
|
+
|
|
8
|
+
from stillvalid import Checker, Doc
|
|
9
|
+
|
|
10
|
+
sv = Checker(history_db="observations.db")
|
|
11
|
+
v = sv.check(Doc(id="policy/refund-de",
|
|
12
|
+
last_verified_at=1790000000,
|
|
13
|
+
last_changed_at=1767225600))
|
|
14
|
+
if not v.usable:
|
|
15
|
+
... # re-fetch the source before acting on it
|
|
16
|
+
|
|
17
|
+
Standard library only. Experimental (v0.1) — API may change.
|
|
18
|
+
"""
|
|
19
|
+
from .cascade import Checker
|
|
20
|
+
from .doc import Doc
|
|
21
|
+
from .verdict import Policy, State, Verdict
|
|
22
|
+
|
|
23
|
+
__all__ = ["Checker", "Doc", "Policy", "State", "Verdict"]
|
|
24
|
+
__version__ = "0.1.0"
|
stillvalid/backfill.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Backfill — import a corpus's past, instead of waiting to observe it.
|
|
2
|
+
|
|
3
|
+
The learned layers need change history. Waiting weeks to accumulate it is
|
|
4
|
+
the single worst thing about installing a freshness layer, and it is also
|
|
5
|
+
unnecessary: most knowledge bases already keep that history. Git has it in
|
|
6
|
+
the commit log; wikis keep page revisions; CMSes keep audit trails; a
|
|
7
|
+
warehouse keeps `updated_at` snapshots. None of it has to be observed
|
|
8
|
+
going forward — it can be read today.
|
|
9
|
+
|
|
10
|
+
from stillvalid import Checker
|
|
11
|
+
from stillvalid.backfill import from_git
|
|
12
|
+
|
|
13
|
+
with Checker(history_db="observations.db") as sv:
|
|
14
|
+
n = from_git(sv, r"C:\\work\\handbook") # seconds, not weeks
|
|
15
|
+
|
|
16
|
+
What is imported is a sighting per (document, change time) — the same
|
|
17
|
+
shape `record()` writes — so the change-rate layer lights up immediately
|
|
18
|
+
and the calibrated trainer has something to chew on.
|
|
19
|
+
"""
|
|
20
|
+
from __future__ import annotations
|
|
21
|
+
|
|
22
|
+
import csv
|
|
23
|
+
import subprocess
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Callable, Iterable, Iterator
|
|
26
|
+
|
|
27
|
+
#: file suffixes worth treating as documents when scanning a repository
|
|
28
|
+
DOC_SUFFIXES = (".md", ".mdx", ".rst", ".txt", ".adoc", ".html", ".htm",
|
|
29
|
+
".yaml", ".yml", ".json", ".csv", ".org", ".tex")
|
|
30
|
+
|
|
31
|
+
_SKIP_PARTS = {".git", "node_modules", "__pycache__", ".venv", "dist",
|
|
32
|
+
"build", "vendor", ".next", "target"}
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
def _import(checker, rows: Iterable[tuple[str, int, str]],
|
|
36
|
+
prefix: str = "") -> int:
|
|
37
|
+
"""Write (doc_id, ts, value_hash) sightings. Returns rows written."""
|
|
38
|
+
if checker.history is None:
|
|
39
|
+
raise RuntimeError("Checker was created without history_db")
|
|
40
|
+
n = 0
|
|
41
|
+
for doc_id, ts, value in rows:
|
|
42
|
+
if checker.history.record(f"{prefix}{doc_id}", int(ts), value):
|
|
43
|
+
n += 1
|
|
44
|
+
return n
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def iter_git_changes(repo: str | Path, *, suffixes: tuple[str, ...] | None
|
|
48
|
+
= DOC_SUFFIXES, max_commits: int = 0,
|
|
49
|
+
paths: Iterable[str] | None = None
|
|
50
|
+
) -> Iterator[tuple[str, int, str]]:
|
|
51
|
+
"""Yield (path, commit_time, commit_sha) for every file change.
|
|
52
|
+
|
|
53
|
+
`suffixes=None` imports every tracked file. Renames break a file's
|
|
54
|
+
chain (git reports the new path), which understates history for moved
|
|
55
|
+
documents — the same limitation the GitHub API has.
|
|
56
|
+
"""
|
|
57
|
+
repo = str(repo)
|
|
58
|
+
cmd = ["git", "-C", repo, "log", "--name-only", "--format=%x01%H %ct"]
|
|
59
|
+
if max_commits:
|
|
60
|
+
cmd.insert(4, f"-n{max_commits}")
|
|
61
|
+
if paths:
|
|
62
|
+
cmd += ["--", *paths]
|
|
63
|
+
out = subprocess.run(cmd, capture_output=True, timeout=600,
|
|
64
|
+
check=True).stdout.decode("utf-8", "replace")
|
|
65
|
+
sha = ts = None
|
|
66
|
+
for line in out.splitlines():
|
|
67
|
+
line = line.rstrip()
|
|
68
|
+
if line.startswith("\x01"):
|
|
69
|
+
try:
|
|
70
|
+
sha, ts = line[1:].split()
|
|
71
|
+
except ValueError:
|
|
72
|
+
sha = ts = None
|
|
73
|
+
elif line and ts:
|
|
74
|
+
p = Path(line)
|
|
75
|
+
if _SKIP_PARTS & set(p.parts):
|
|
76
|
+
continue
|
|
77
|
+
if suffixes and p.suffix.lower() not in suffixes:
|
|
78
|
+
continue
|
|
79
|
+
yield line, int(ts), sha
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def from_git(checker, repo: str | Path, *, prefix: str = "",
|
|
83
|
+
suffixes: tuple[str, ...] | None = DOC_SUFFIXES,
|
|
84
|
+
max_commits: int = 0) -> int:
|
|
85
|
+
"""Import a git repository's change history. Returns sightings written.
|
|
86
|
+
|
|
87
|
+
Fast enough to be interactive: thousands of changes in seconds, versus
|
|
88
|
+
weeks of waiting for the same signal to arrive on its own.
|
|
89
|
+
"""
|
|
90
|
+
return _import(checker, iter_git_changes(
|
|
91
|
+
repo, suffixes=suffixes, max_commits=max_commits), prefix)
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
def from_rows(checker, rows: Iterable[tuple[str, float, str]], *,
|
|
95
|
+
prefix: str = "") -> int:
|
|
96
|
+
"""Import (doc_id, timestamp, revision) triples from anywhere.
|
|
97
|
+
|
|
98
|
+
The general entry point: a wiki's revision API, a CMS audit table, a
|
|
99
|
+
warehouse query. `revision` is any value that changes when the content
|
|
100
|
+
changes (version number, hash, etag); consecutive equal values read as
|
|
101
|
+
"still the same", which is exactly the censoring information the
|
|
102
|
+
learned layers want.
|
|
103
|
+
"""
|
|
104
|
+
return _import(checker, ((d, int(t), str(r)) for d, t, r in rows), prefix)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def from_csv(checker, path: str | Path, *, prefix: str = "") -> int:
|
|
108
|
+
"""Import a CSV with columns doc_id, ts, rev (header required).
|
|
109
|
+
|
|
110
|
+
`ts` may be epoch seconds or an ISO-8601 string.
|
|
111
|
+
"""
|
|
112
|
+
from .integrations.common import to_epoch
|
|
113
|
+
|
|
114
|
+
with open(path, encoding="utf-8-sig", newline="") as f:
|
|
115
|
+
reader = csv.DictReader(f)
|
|
116
|
+
missing = {"doc_id", "ts"} - set(reader.fieldnames or ())
|
|
117
|
+
if missing:
|
|
118
|
+
raise ValueError(f"CSV is missing columns: {sorted(missing)} "
|
|
119
|
+
"(expected doc_id, ts[, rev])")
|
|
120
|
+
rows = []
|
|
121
|
+
for r in reader:
|
|
122
|
+
ts = to_epoch(r["ts"])
|
|
123
|
+
if ts is None:
|
|
124
|
+
continue
|
|
125
|
+
rows.append((r["doc_id"], ts, r.get("rev") or str(int(ts))))
|
|
126
|
+
return from_rows(checker, rows, prefix=prefix)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def summarize(checker, doc_ids: Iterable[str] | None = None) -> dict:
|
|
130
|
+
"""What did the import actually buy? Counts per readiness tier."""
|
|
131
|
+
h = checker.history
|
|
132
|
+
if h is None:
|
|
133
|
+
raise RuntimeError("Checker was created without history_db")
|
|
134
|
+
with h._lock:
|
|
135
|
+
rows = h.conn.execute(
|
|
136
|
+
"SELECT doc_id, COUNT(*) FROM observations GROUP BY doc_id"
|
|
137
|
+
).fetchall()
|
|
138
|
+
if doc_ids is not None:
|
|
139
|
+
want = set(doc_ids)
|
|
140
|
+
rows = [r for r in rows if r[0] in want]
|
|
141
|
+
ready = sum(1 for _, c in rows if c >= 2)
|
|
142
|
+
trainable = sum(1 for _, c in rows if c >= 8)
|
|
143
|
+
return {"documents": len(rows),
|
|
144
|
+
"sightings": sum(c for _, c in rows),
|
|
145
|
+
"change_rate_ready": ready, # layer 5 can estimate
|
|
146
|
+
"training_ready": trainable} # enough events to train on
|
stillvalid/cascade.py
ADDED
|
@@ -0,0 +1,170 @@
|
|
|
1
|
+
"""The cascade — try cheap, certain judgments first; stop at the first
|
|
2
|
+
layer that can decide; abstain honestly when none can.
|
|
3
|
+
|
|
4
|
+
Order:
|
|
5
|
+
1. explicit expiry (deterministic, zero history)
|
|
6
|
+
2. live content hash compare (deterministic, zero history)
|
|
7
|
+
3. live Last-Modified compare (deterministic, zero history)
|
|
8
|
+
6. calibrated survival model (needs gate-passed params covering the doc)
|
|
9
|
+
5. crude change-rate estimate (needs >= 2 observed changes; uncalibrated)
|
|
10
|
+
7. UNKNOWN (honest abstention)
|
|
11
|
+
|
|
12
|
+
Layer 6 runs before layer 5 on purpose: when a calibrated model covers the
|
|
13
|
+
document it beats the crude estimate. The numbering follows capability,
|
|
14
|
+
not execution order. Cross-source verification (layer 4 in the design
|
|
15
|
+
memo) is on the roadmap, not in v0.1.
|
|
16
|
+
"""
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import math
|
|
20
|
+
import time
|
|
21
|
+
from pathlib import Path
|
|
22
|
+
from typing import Iterable
|
|
23
|
+
|
|
24
|
+
from .doc import Doc
|
|
25
|
+
from .history import History
|
|
26
|
+
from .survival import SurvivalParams
|
|
27
|
+
from .verdict import DEFAULT_ACTIONS, Policy, State, Verdict
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class Checker:
|
|
31
|
+
def __init__(self, *, survival_params: str | None = None,
|
|
32
|
+
history_db: str | Path | None = None,
|
|
33
|
+
policy: Policy | None = None):
|
|
34
|
+
self.params = SurvivalParams.load(survival_params) \
|
|
35
|
+
if survival_params else None
|
|
36
|
+
self.history = History(history_db) if history_db else None
|
|
37
|
+
self.policy = policy or Policy()
|
|
38
|
+
|
|
39
|
+
def close(self) -> None:
|
|
40
|
+
if self.history is not None:
|
|
41
|
+
self.history.close()
|
|
42
|
+
|
|
43
|
+
def __enter__(self):
|
|
44
|
+
return self
|
|
45
|
+
|
|
46
|
+
def __exit__(self, *exc):
|
|
47
|
+
self.close()
|
|
48
|
+
|
|
49
|
+
# ── public API ────────────────────────────────────────────
|
|
50
|
+
|
|
51
|
+
def check(self, doc: Doc, now: float | None = None) -> Verdict:
|
|
52
|
+
now = time.time() if now is None else now
|
|
53
|
+
for layer in (self._expiry, self._hash_compare, self._modified_compare,
|
|
54
|
+
self._survival, self._change_rate):
|
|
55
|
+
v = layer(doc, now)
|
|
56
|
+
if v is not None:
|
|
57
|
+
return v
|
|
58
|
+
return self._verdict(doc, State.UNKNOWN, "none",
|
|
59
|
+
"no layer could judge this document")
|
|
60
|
+
|
|
61
|
+
def check_many(self, docs: Iterable[Doc],
|
|
62
|
+
now: float | None = None) -> list[Verdict]:
|
|
63
|
+
now = time.time() if now is None else now # one clock per batch
|
|
64
|
+
return [self.check(d, now=now) for d in docs]
|
|
65
|
+
|
|
66
|
+
def record(self, doc_id: str, value_hash: str,
|
|
67
|
+
ts: float | None = None, dedup_s: float = 0.0) -> bool:
|
|
68
|
+
"""Log a sighting — call whenever you fetch/reindex a document."""
|
|
69
|
+
if self.history is None:
|
|
70
|
+
raise RuntimeError("Checker was created without history_db")
|
|
71
|
+
return self.history.record(
|
|
72
|
+
doc_id, int(time.time() if ts is None else ts),
|
|
73
|
+
value_hash, dedup_s=dedup_s)
|
|
74
|
+
|
|
75
|
+
# ── layers ────────────────────────────────────────────────
|
|
76
|
+
|
|
77
|
+
def _expiry(self, doc: Doc, now: float) -> Verdict | None:
|
|
78
|
+
if doc.expires_at is None or math.isnan(doc.expires_at):
|
|
79
|
+
return None # NaN compares False to everything — that is not
|
|
80
|
+
# evidence of expiry, it is a broken input
|
|
81
|
+
if now < doc.expires_at:
|
|
82
|
+
return self._verdict(doc, State.VALID, "expiry",
|
|
83
|
+
"explicit expiry is "
|
|
84
|
+
f"{_dur(doc.expires_at - now)} away")
|
|
85
|
+
return self._verdict(doc, State.STALE, "expiry",
|
|
86
|
+
"explicitly expired "
|
|
87
|
+
f"{_dur(now - doc.expires_at)} ago")
|
|
88
|
+
|
|
89
|
+
def _hash_compare(self, doc: Doc, now: float) -> Verdict | None:
|
|
90
|
+
if not doc.source_hash or not doc.verified_hash:
|
|
91
|
+
return None # empty strings are missing data, not a match
|
|
92
|
+
if doc.source_hash == doc.verified_hash:
|
|
93
|
+
return self._verdict(doc, State.VALID, "hash",
|
|
94
|
+
"live content hash equals verified hash")
|
|
95
|
+
return self._verdict(doc, State.STALE, "hash",
|
|
96
|
+
"live content hash differs from verified hash")
|
|
97
|
+
|
|
98
|
+
def _modified_compare(self, doc: Doc, now: float) -> Verdict | None:
|
|
99
|
+
if doc.source_modified_at is None or doc.last_verified_at is None \
|
|
100
|
+
or math.isnan(doc.source_modified_at) \
|
|
101
|
+
or math.isnan(doc.last_verified_at):
|
|
102
|
+
return None
|
|
103
|
+
if doc.source_modified_at <= doc.last_verified_at:
|
|
104
|
+
return self._verdict(doc, State.VALID, "modified",
|
|
105
|
+
"source unmodified since your verification "
|
|
106
|
+
f"{_dur(now - doc.last_verified_at)} ago")
|
|
107
|
+
return self._verdict(doc, State.STALE, "modified",
|
|
108
|
+
"source was modified "
|
|
109
|
+
f"{_dur(doc.source_modified_at - doc.last_verified_at)}"
|
|
110
|
+
" after your last verification")
|
|
111
|
+
|
|
112
|
+
def _survival(self, doc: Doc, now: float) -> Verdict | None:
|
|
113
|
+
if (self.params is None or doc.last_changed_at is None
|
|
114
|
+
or doc.last_verified_at is None):
|
|
115
|
+
return None
|
|
116
|
+
if self.params.status(doc.id) != "ok":
|
|
117
|
+
return None # observing/unknown → let later layers try
|
|
118
|
+
p = self.params.p_valid(doc.id, doc.last_changed_at,
|
|
119
|
+
doc.last_verified_at, now)
|
|
120
|
+
if p is None:
|
|
121
|
+
return None # params declined to judge (broken input)
|
|
122
|
+
state = self.policy.state_for(p)
|
|
123
|
+
return self._verdict(doc, state, "survival",
|
|
124
|
+
f"calibrated {p:.0%} probability it is unchanged "
|
|
125
|
+
f"{_dur(now - doc.last_verified_at)} after "
|
|
126
|
+
"verification, from its learned update habits",
|
|
127
|
+
probability=p, calibrated=True)
|
|
128
|
+
|
|
129
|
+
def _change_rate(self, doc: Doc, now: float) -> Verdict | None:
|
|
130
|
+
if self.history is None or doc.last_verified_at is None:
|
|
131
|
+
return None
|
|
132
|
+
n, mean_gap = self.history.change_stats(doc.id)
|
|
133
|
+
if mean_gap is None or mean_gap <= 0:
|
|
134
|
+
return None
|
|
135
|
+
# Exponential (memoryless) survival since last verification —
|
|
136
|
+
# a crude, UNCALIBRATED estimate. It exists so the cascade is useful
|
|
137
|
+
# before the survival model's gate opens; treat it as a hint.
|
|
138
|
+
p = math.exp(-max(now - doc.last_verified_at, 0.0) / mean_gap)
|
|
139
|
+
state = self.policy.state_for(p)
|
|
140
|
+
return self._verdict(doc, state, "change_rate",
|
|
141
|
+
f"rough {p:.0%} estimate (uncalibrated) - it has "
|
|
142
|
+
f"changed {n} times, roughly every "
|
|
143
|
+
f"{_dur(mean_gap)}, and is "
|
|
144
|
+
f"{_dur(now - doc.last_verified_at)} unverified",
|
|
145
|
+
probability=p, calibrated=False)
|
|
146
|
+
|
|
147
|
+
# ── helpers ───────────────────────────────────────────────
|
|
148
|
+
|
|
149
|
+
def _verdict(self, doc: Doc, state: State, layer: str, reason: str,
|
|
150
|
+
probability: float | None = None,
|
|
151
|
+
calibrated: bool = False) -> Verdict:
|
|
152
|
+
return Verdict(doc_id=doc.id, state=state,
|
|
153
|
+
action=DEFAULT_ACTIONS[state], layer=layer,
|
|
154
|
+
reason=reason, probability=probability,
|
|
155
|
+
calibrated=calibrated,
|
|
156
|
+
last_verified_at=doc.last_verified_at)
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def _dur(seconds: float) -> str:
|
|
160
|
+
"""Humanize a duration — receipts read better than epoch math."""
|
|
161
|
+
s = abs(seconds)
|
|
162
|
+
if s < 120:
|
|
163
|
+
return f"{s:.0f}s"
|
|
164
|
+
if s < 7200:
|
|
165
|
+
return f"{s / 60:.0f}m"
|
|
166
|
+
if s < 2 * 86400:
|
|
167
|
+
return f"{s / 3600:.0f}h"
|
|
168
|
+
if s < 2 * 365.25 * 86400:
|
|
169
|
+
return f"{s / 86400:.0f}d"
|
|
170
|
+
return f"{s / (365.25 * 86400):.1f}y"
|
stillvalid/doc.py
ADDED
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
"""Input shape — what the caller knows about a retrieved document."""
|
|
2
|
+
from __future__ import annotations
|
|
3
|
+
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
@dataclass(frozen=True)
|
|
8
|
+
class Doc:
|
|
9
|
+
"""Everything is optional except id; each field unlocks a cascade layer.
|
|
10
|
+
|
|
11
|
+
Times are Unix epoch seconds.
|
|
12
|
+
|
|
13
|
+
last_verified_at : when YOU last confirmed the content (e.g., index time).
|
|
14
|
+
last_changed_at : last modification time known as of that verification.
|
|
15
|
+
expires_at : explicit expiry, if the source declares one (layer 1).
|
|
16
|
+
source_hash : live content hash fetched just now, if you have it
|
|
17
|
+
(layer 2 — compare against verified_hash).
|
|
18
|
+
verified_hash : content hash recorded at verification time.
|
|
19
|
+
source_modified_at: live Last-Modified fetched just now (layer 3).
|
|
20
|
+
"""
|
|
21
|
+
id: str
|
|
22
|
+
last_verified_at: float | None = None
|
|
23
|
+
last_changed_at: float | None = None
|
|
24
|
+
expires_at: float | None = None
|
|
25
|
+
source_hash: str | None = None
|
|
26
|
+
verified_hash: str | None = None
|
|
27
|
+
source_modified_at: float | None = None
|
stillvalid/history.py
ADDED
|
@@ -0,0 +1,142 @@
|
|
|
1
|
+
"""Local observation log — fuels layers 5 (change rate) and, with enough
|
|
2
|
+
events, training data for layer 6.
|
|
3
|
+
|
|
4
|
+
One SQLite file, standard library only. Record what you saw whenever you
|
|
5
|
+
fetch or reindex a document; the cascade gets smarter as this accrues.
|
|
6
|
+
"""
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import sqlite3
|
|
10
|
+
import threading
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
_SCHEMA_BASE = """
|
|
14
|
+
CREATE TABLE IF NOT EXISTS observations (
|
|
15
|
+
id INTEGER PRIMARY KEY AUTOINCREMENT,
|
|
16
|
+
doc_id TEXT NOT NULL,
|
|
17
|
+
ts INTEGER NOT NULL,
|
|
18
|
+
value_hash TEXT NOT NULL
|
|
19
|
+
);
|
|
20
|
+
CREATE INDEX IF NOT EXISTS idx_obs_doc_ts ON observations (doc_id, ts);
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
#: One sighting per (document, instant, value): re-importing the same
|
|
24
|
+
#: history is an ordinary thing to do (a second import_history call, a
|
|
25
|
+
#: retried job) and must not double-count a document's change rate.
|
|
26
|
+
_UNIQUE_INDEX = ("CREATE UNIQUE INDEX IF NOT EXISTS idx_obs_unique "
|
|
27
|
+
"ON observations (doc_id, ts, value_hash)")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
class History:
|
|
31
|
+
"""Thread- and process-safe observation log.
|
|
32
|
+
|
|
33
|
+
A single Checker is routinely shared across request threads (a web
|
|
34
|
+
server with the LangChain wrapper, an MCP server), so the connection
|
|
35
|
+
is opened with check_same_thread=False and every statement runs under
|
|
36
|
+
one lock. WAL lets a second process (an indexing job, another server)
|
|
37
|
+
read and write the same file without blocking.
|
|
38
|
+
"""
|
|
39
|
+
|
|
40
|
+
def __init__(self, path: str | Path, timeout: float = 10.0):
|
|
41
|
+
self.conn = sqlite3.connect(str(path), check_same_thread=False,
|
|
42
|
+
timeout=timeout)
|
|
43
|
+
self._lock = threading.Lock()
|
|
44
|
+
with self._lock:
|
|
45
|
+
try:
|
|
46
|
+
self.conn.execute("PRAGMA journal_mode=WAL")
|
|
47
|
+
# Default autocheckpoint let the -wal file reach ~4 MB for a
|
|
48
|
+
# 0.8 MB database in a write-heavy soak; a long-lived server
|
|
49
|
+
# never closes, so pin a tighter budget.
|
|
50
|
+
self.conn.execute("PRAGMA wal_autocheckpoint=256")
|
|
51
|
+
except sqlite3.DatabaseError:
|
|
52
|
+
pass # e.g. a network filesystem that refuses WAL
|
|
53
|
+
self.conn.executescript(_SCHEMA_BASE)
|
|
54
|
+
try:
|
|
55
|
+
self.conn.execute(_UNIQUE_INDEX)
|
|
56
|
+
except sqlite3.IntegrityError:
|
|
57
|
+
# a log written before the constraint existed may hold
|
|
58
|
+
# duplicates; collapse them, then enforce it from now on
|
|
59
|
+
self.conn.execute(
|
|
60
|
+
"DELETE FROM observations WHERE id NOT IN ("
|
|
61
|
+
" SELECT MIN(id) FROM observations"
|
|
62
|
+
" GROUP BY doc_id, ts, value_hash)")
|
|
63
|
+
self.conn.execute(_UNIQUE_INDEX)
|
|
64
|
+
self.conn.commit()
|
|
65
|
+
|
|
66
|
+
def close(self):
|
|
67
|
+
with self._lock:
|
|
68
|
+
self.conn.close()
|
|
69
|
+
|
|
70
|
+
def __enter__(self):
|
|
71
|
+
return self
|
|
72
|
+
|
|
73
|
+
def __exit__(self, *exc):
|
|
74
|
+
self.close()
|
|
75
|
+
|
|
76
|
+
def record(self, doc_id: str, ts: int, value_hash: str,
|
|
77
|
+
dedup_s: float = 0.0) -> bool:
|
|
78
|
+
"""Log one sighting of a document's content hash.
|
|
79
|
+
|
|
80
|
+
An identical (doc_id, ts, value_hash) row is never stored twice, so
|
|
81
|
+
re-importing a history is harmless. With dedup_s > 0, a sighting
|
|
82
|
+
with the SAME hash as the latest row is additionally skipped unless
|
|
83
|
+
that row is older than dedup_s — keeps high-traffic auto-recording
|
|
84
|
+
from flooding the log while still capturing every change
|
|
85
|
+
immediately. Returns True if a row was written.
|
|
86
|
+
"""
|
|
87
|
+
with self._lock:
|
|
88
|
+
if dedup_s > 0:
|
|
89
|
+
prev = self._last(doc_id)
|
|
90
|
+
if prev is not None and prev[1] == value_hash \
|
|
91
|
+
and ts - prev[0] < dedup_s:
|
|
92
|
+
return False
|
|
93
|
+
cur = self.conn.execute(
|
|
94
|
+
"INSERT OR IGNORE INTO observations (doc_id, ts, value_hash) "
|
|
95
|
+
"VALUES (?,?,?)", (doc_id, int(ts), value_hash))
|
|
96
|
+
self.conn.commit()
|
|
97
|
+
return cur.rowcount > 0
|
|
98
|
+
|
|
99
|
+
def last(self, doc_id: str) -> tuple[int, str] | None:
|
|
100
|
+
"""(ts, value_hash) of the most recent sighting, or None."""
|
|
101
|
+
with self._lock:
|
|
102
|
+
return self._last(doc_id)
|
|
103
|
+
|
|
104
|
+
def _last(self, doc_id: str) -> tuple[int, str] | None:
|
|
105
|
+
"""Caller must hold the lock."""
|
|
106
|
+
row = self.conn.execute(
|
|
107
|
+
"SELECT ts, value_hash FROM observations WHERE doc_id=? "
|
|
108
|
+
"ORDER BY ts DESC, id DESC LIMIT 1", (doc_id,)).fetchone()
|
|
109
|
+
return (row[0], row[1]) if row else None
|
|
110
|
+
|
|
111
|
+
#: how many recent sightings the change-rate estimate looks at. Reading
|
|
112
|
+
#: a document's whole history made judging cost grow linearly with it
|
|
113
|
+
#: (12 ms at 20k sightings) — and recent behavior is the better
|
|
114
|
+
#: predictor anyway, so the window is both faster and more honest.
|
|
115
|
+
RECENT_WINDOW = 500
|
|
116
|
+
|
|
117
|
+
def change_stats(self, doc_id: str,
|
|
118
|
+
window: int | None = None) -> tuple[int, float | None]:
|
|
119
|
+
"""(number of observed changes, mean seconds between changes).
|
|
120
|
+
|
|
121
|
+
A change is a transition between consecutive distinct hashes; the
|
|
122
|
+
interval is measured between the observation times of those
|
|
123
|
+
transitions. Needs >= 2 changes for a mean. Only the most recent
|
|
124
|
+
`window` sightings are considered (RECENT_WINDOW by default).
|
|
125
|
+
"""
|
|
126
|
+
with self._lock:
|
|
127
|
+
rows = self.conn.execute(
|
|
128
|
+
"SELECT ts, value_hash FROM ("
|
|
129
|
+
" SELECT ts, value_hash, id FROM observations WHERE doc_id=?"
|
|
130
|
+
" ORDER BY ts DESC, id DESC LIMIT ?"
|
|
131
|
+
") ORDER BY ts ASC, id ASC",
|
|
132
|
+
(doc_id, window or self.RECENT_WINDOW)).fetchall()
|
|
133
|
+
change_ts = []
|
|
134
|
+
prev_hash = None
|
|
135
|
+
for ts, h in rows:
|
|
136
|
+
if prev_hash is not None and h != prev_hash:
|
|
137
|
+
change_ts.append(ts)
|
|
138
|
+
prev_hash = h
|
|
139
|
+
if len(change_ts) < 2:
|
|
140
|
+
return len(change_ts), None
|
|
141
|
+
gaps = [b - a for a, b in zip(change_ts, change_ts[1:])]
|
|
142
|
+
return len(change_ts), sum(gaps) / len(gaps)
|