stillvalid 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
stillvalid/__init__.py ADDED
@@ -0,0 +1,24 @@
1
+ """stillvalid — is this information still valid?
2
+
3
+ A validity layer between retrieval and action. Feed it what you know about
4
+ a document; it answers VALID / LIKELY_VALID / VERIFY / STALE / UNKNOWN with
5
+ evidence, via a cascade that prefers cheap deterministic checks and only
6
+ falls back to learned probability where nothing deterministic exists.
7
+
8
+ from stillvalid import Checker, Doc
9
+
10
+ sv = Checker(history_db="observations.db")
11
+ v = sv.check(Doc(id="policy/refund-de",
12
+ last_verified_at=1790000000,
13
+ last_changed_at=1767225600))
14
+ if not v.usable:
15
+ ... # re-fetch the source before acting on it
16
+
17
+ Standard library only. Experimental (v0.1) — API may change.
18
+ """
19
+ from .cascade import Checker
20
+ from .doc import Doc
21
+ from .verdict import Policy, State, Verdict
22
+
23
+ __all__ = ["Checker", "Doc", "Policy", "State", "Verdict"]
24
+ __version__ = "0.1.0"
stillvalid/backfill.py ADDED
@@ -0,0 +1,146 @@
1
+ """Backfill — import a corpus's past, instead of waiting to observe it.
2
+
3
+ The learned layers need change history. Waiting weeks to accumulate it is
4
+ the single worst thing about installing a freshness layer, and it is also
5
+ unnecessary: most knowledge bases already keep that history. Git has it in
6
+ the commit log; wikis keep page revisions; CMSes keep audit trails; a
7
+ warehouse keeps `updated_at` snapshots. None of it has to be observed
8
+ going forward — it can be read today.
9
+
10
+ from stillvalid import Checker
11
+ from stillvalid.backfill import from_git
12
+
13
+ with Checker(history_db="observations.db") as sv:
14
+ n = from_git(sv, r"C:\\work\\handbook") # seconds, not weeks
15
+
16
+ What is imported is a sighting per (document, change time) — the same
17
+ shape `record()` writes — so the change-rate layer lights up immediately
18
+ and the calibrated trainer has something to chew on.
19
+ """
20
+ from __future__ import annotations
21
+
22
+ import csv
23
+ import subprocess
24
+ from pathlib import Path
25
+ from typing import Callable, Iterable, Iterator
26
+
27
+ #: file suffixes worth treating as documents when scanning a repository
28
+ DOC_SUFFIXES = (".md", ".mdx", ".rst", ".txt", ".adoc", ".html", ".htm",
29
+ ".yaml", ".yml", ".json", ".csv", ".org", ".tex")
30
+
31
+ _SKIP_PARTS = {".git", "node_modules", "__pycache__", ".venv", "dist",
32
+ "build", "vendor", ".next", "target"}
33
+
34
+
35
+ def _import(checker, rows: Iterable[tuple[str, int, str]],
36
+ prefix: str = "") -> int:
37
+ """Write (doc_id, ts, value_hash) sightings. Returns rows written."""
38
+ if checker.history is None:
39
+ raise RuntimeError("Checker was created without history_db")
40
+ n = 0
41
+ for doc_id, ts, value in rows:
42
+ if checker.history.record(f"{prefix}{doc_id}", int(ts), value):
43
+ n += 1
44
+ return n
45
+
46
+
47
+ def iter_git_changes(repo: str | Path, *, suffixes: tuple[str, ...] | None
48
+ = DOC_SUFFIXES, max_commits: int = 0,
49
+ paths: Iterable[str] | None = None
50
+ ) -> Iterator[tuple[str, int, str]]:
51
+ """Yield (path, commit_time, commit_sha) for every file change.
52
+
53
+ `suffixes=None` imports every tracked file. Renames break a file's
54
+ chain (git reports the new path), which understates history for moved
55
+ documents — the same limitation the GitHub API has.
56
+ """
57
+ repo = str(repo)
58
+ cmd = ["git", "-C", repo, "log", "--name-only", "--format=%x01%H %ct"]
59
+ if max_commits:
60
+ cmd.insert(4, f"-n{max_commits}")
61
+ if paths:
62
+ cmd += ["--", *paths]
63
+ out = subprocess.run(cmd, capture_output=True, timeout=600,
64
+ check=True).stdout.decode("utf-8", "replace")
65
+ sha = ts = None
66
+ for line in out.splitlines():
67
+ line = line.rstrip()
68
+ if line.startswith("\x01"):
69
+ try:
70
+ sha, ts = line[1:].split()
71
+ except ValueError:
72
+ sha = ts = None
73
+ elif line and ts:
74
+ p = Path(line)
75
+ if _SKIP_PARTS & set(p.parts):
76
+ continue
77
+ if suffixes and p.suffix.lower() not in suffixes:
78
+ continue
79
+ yield line, int(ts), sha
80
+
81
+
82
+ def from_git(checker, repo: str | Path, *, prefix: str = "",
83
+ suffixes: tuple[str, ...] | None = DOC_SUFFIXES,
84
+ max_commits: int = 0) -> int:
85
+ """Import a git repository's change history. Returns sightings written.
86
+
87
+ Fast enough to be interactive: thousands of changes in seconds, versus
88
+ weeks of waiting for the same signal to arrive on its own.
89
+ """
90
+ return _import(checker, iter_git_changes(
91
+ repo, suffixes=suffixes, max_commits=max_commits), prefix)
92
+
93
+
94
+ def from_rows(checker, rows: Iterable[tuple[str, float, str]], *,
95
+ prefix: str = "") -> int:
96
+ """Import (doc_id, timestamp, revision) triples from anywhere.
97
+
98
+ The general entry point: a wiki's revision API, a CMS audit table, a
99
+ warehouse query. `revision` is any value that changes when the content
100
+ changes (version number, hash, etag); consecutive equal values read as
101
+ "still the same", which is exactly the censoring information the
102
+ learned layers want.
103
+ """
104
+ return _import(checker, ((d, int(t), str(r)) for d, t, r in rows), prefix)
105
+
106
+
107
+ def from_csv(checker, path: str | Path, *, prefix: str = "") -> int:
108
+ """Import a CSV with columns doc_id, ts, rev (header required).
109
+
110
+ `ts` may be epoch seconds or an ISO-8601 string.
111
+ """
112
+ from .integrations.common import to_epoch
113
+
114
+ with open(path, encoding="utf-8-sig", newline="") as f:
115
+ reader = csv.DictReader(f)
116
+ missing = {"doc_id", "ts"} - set(reader.fieldnames or ())
117
+ if missing:
118
+ raise ValueError(f"CSV is missing columns: {sorted(missing)} "
119
+ "(expected doc_id, ts[, rev])")
120
+ rows = []
121
+ for r in reader:
122
+ ts = to_epoch(r["ts"])
123
+ if ts is None:
124
+ continue
125
+ rows.append((r["doc_id"], ts, r.get("rev") or str(int(ts))))
126
+ return from_rows(checker, rows, prefix=prefix)
127
+
128
+
129
+ def summarize(checker, doc_ids: Iterable[str] | None = None) -> dict:
130
+ """What did the import actually buy? Counts per readiness tier."""
131
+ h = checker.history
132
+ if h is None:
133
+ raise RuntimeError("Checker was created without history_db")
134
+ with h._lock:
135
+ rows = h.conn.execute(
136
+ "SELECT doc_id, COUNT(*) FROM observations GROUP BY doc_id"
137
+ ).fetchall()
138
+ if doc_ids is not None:
139
+ want = set(doc_ids)
140
+ rows = [r for r in rows if r[0] in want]
141
+ ready = sum(1 for _, c in rows if c >= 2)
142
+ trainable = sum(1 for _, c in rows if c >= 8)
143
+ return {"documents": len(rows),
144
+ "sightings": sum(c for _, c in rows),
145
+ "change_rate_ready": ready, # layer 5 can estimate
146
+ "training_ready": trainable} # enough events to train on
stillvalid/cascade.py ADDED
@@ -0,0 +1,170 @@
1
+ """The cascade — try cheap, certain judgments first; stop at the first
2
+ layer that can decide; abstain honestly when none can.
3
+
4
+ Order:
5
+ 1. explicit expiry (deterministic, zero history)
6
+ 2. live content hash compare (deterministic, zero history)
7
+ 3. live Last-Modified compare (deterministic, zero history)
8
+ 6. calibrated survival model (needs gate-passed params covering the doc)
9
+ 5. crude change-rate estimate (needs >= 2 observed changes; uncalibrated)
10
+ 7. UNKNOWN (honest abstention)
11
+
12
+ Layer 6 runs before layer 5 on purpose: when a calibrated model covers the
13
+ document it beats the crude estimate. The numbering follows capability,
14
+ not execution order. Cross-source verification (layer 4 in the design
15
+ memo) is on the roadmap, not in v0.1.
16
+ """
17
+ from __future__ import annotations
18
+
19
+ import math
20
+ import time
21
+ from pathlib import Path
22
+ from typing import Iterable
23
+
24
+ from .doc import Doc
25
+ from .history import History
26
+ from .survival import SurvivalParams
27
+ from .verdict import DEFAULT_ACTIONS, Policy, State, Verdict
28
+
29
+
30
+ class Checker:
31
+ def __init__(self, *, survival_params: str | None = None,
32
+ history_db: str | Path | None = None,
33
+ policy: Policy | None = None):
34
+ self.params = SurvivalParams.load(survival_params) \
35
+ if survival_params else None
36
+ self.history = History(history_db) if history_db else None
37
+ self.policy = policy or Policy()
38
+
39
+ def close(self) -> None:
40
+ if self.history is not None:
41
+ self.history.close()
42
+
43
+ def __enter__(self):
44
+ return self
45
+
46
+ def __exit__(self, *exc):
47
+ self.close()
48
+
49
+ # ── public API ────────────────────────────────────────────
50
+
51
+ def check(self, doc: Doc, now: float | None = None) -> Verdict:
52
+ now = time.time() if now is None else now
53
+ for layer in (self._expiry, self._hash_compare, self._modified_compare,
54
+ self._survival, self._change_rate):
55
+ v = layer(doc, now)
56
+ if v is not None:
57
+ return v
58
+ return self._verdict(doc, State.UNKNOWN, "none",
59
+ "no layer could judge this document")
60
+
61
+ def check_many(self, docs: Iterable[Doc],
62
+ now: float | None = None) -> list[Verdict]:
63
+ now = time.time() if now is None else now # one clock per batch
64
+ return [self.check(d, now=now) for d in docs]
65
+
66
+ def record(self, doc_id: str, value_hash: str,
67
+ ts: float | None = None, dedup_s: float = 0.0) -> bool:
68
+ """Log a sighting — call whenever you fetch/reindex a document."""
69
+ if self.history is None:
70
+ raise RuntimeError("Checker was created without history_db")
71
+ return self.history.record(
72
+ doc_id, int(time.time() if ts is None else ts),
73
+ value_hash, dedup_s=dedup_s)
74
+
75
+ # ── layers ────────────────────────────────────────────────
76
+
77
+ def _expiry(self, doc: Doc, now: float) -> Verdict | None:
78
+ if doc.expires_at is None or math.isnan(doc.expires_at):
79
+ return None # NaN compares False to everything — that is not
80
+ # evidence of expiry, it is a broken input
81
+ if now < doc.expires_at:
82
+ return self._verdict(doc, State.VALID, "expiry",
83
+ "explicit expiry is "
84
+ f"{_dur(doc.expires_at - now)} away")
85
+ return self._verdict(doc, State.STALE, "expiry",
86
+ "explicitly expired "
87
+ f"{_dur(now - doc.expires_at)} ago")
88
+
89
+ def _hash_compare(self, doc: Doc, now: float) -> Verdict | None:
90
+ if not doc.source_hash or not doc.verified_hash:
91
+ return None # empty strings are missing data, not a match
92
+ if doc.source_hash == doc.verified_hash:
93
+ return self._verdict(doc, State.VALID, "hash",
94
+ "live content hash equals verified hash")
95
+ return self._verdict(doc, State.STALE, "hash",
96
+ "live content hash differs from verified hash")
97
+
98
+ def _modified_compare(self, doc: Doc, now: float) -> Verdict | None:
99
+ if doc.source_modified_at is None or doc.last_verified_at is None \
100
+ or math.isnan(doc.source_modified_at) \
101
+ or math.isnan(doc.last_verified_at):
102
+ return None
103
+ if doc.source_modified_at <= doc.last_verified_at:
104
+ return self._verdict(doc, State.VALID, "modified",
105
+ "source unmodified since your verification "
106
+ f"{_dur(now - doc.last_verified_at)} ago")
107
+ return self._verdict(doc, State.STALE, "modified",
108
+ "source was modified "
109
+ f"{_dur(doc.source_modified_at - doc.last_verified_at)}"
110
+ " after your last verification")
111
+
112
+ def _survival(self, doc: Doc, now: float) -> Verdict | None:
113
+ if (self.params is None or doc.last_changed_at is None
114
+ or doc.last_verified_at is None):
115
+ return None
116
+ if self.params.status(doc.id) != "ok":
117
+ return None # observing/unknown → let later layers try
118
+ p = self.params.p_valid(doc.id, doc.last_changed_at,
119
+ doc.last_verified_at, now)
120
+ if p is None:
121
+ return None # params declined to judge (broken input)
122
+ state = self.policy.state_for(p)
123
+ return self._verdict(doc, state, "survival",
124
+ f"calibrated {p:.0%} probability it is unchanged "
125
+ f"{_dur(now - doc.last_verified_at)} after "
126
+ "verification, from its learned update habits",
127
+ probability=p, calibrated=True)
128
+
129
+ def _change_rate(self, doc: Doc, now: float) -> Verdict | None:
130
+ if self.history is None or doc.last_verified_at is None:
131
+ return None
132
+ n, mean_gap = self.history.change_stats(doc.id)
133
+ if mean_gap is None or mean_gap <= 0:
134
+ return None
135
+ # Exponential (memoryless) survival since last verification —
136
+ # a crude, UNCALIBRATED estimate. It exists so the cascade is useful
137
+ # before the survival model's gate opens; treat it as a hint.
138
+ p = math.exp(-max(now - doc.last_verified_at, 0.0) / mean_gap)
139
+ state = self.policy.state_for(p)
140
+ return self._verdict(doc, state, "change_rate",
141
+ f"rough {p:.0%} estimate (uncalibrated) - it has "
142
+ f"changed {n} times, roughly every "
143
+ f"{_dur(mean_gap)}, and is "
144
+ f"{_dur(now - doc.last_verified_at)} unverified",
145
+ probability=p, calibrated=False)
146
+
147
+ # ── helpers ───────────────────────────────────────────────
148
+
149
+ def _verdict(self, doc: Doc, state: State, layer: str, reason: str,
150
+ probability: float | None = None,
151
+ calibrated: bool = False) -> Verdict:
152
+ return Verdict(doc_id=doc.id, state=state,
153
+ action=DEFAULT_ACTIONS[state], layer=layer,
154
+ reason=reason, probability=probability,
155
+ calibrated=calibrated,
156
+ last_verified_at=doc.last_verified_at)
157
+
158
+
159
+ def _dur(seconds: float) -> str:
160
+ """Humanize a duration — receipts read better than epoch math."""
161
+ s = abs(seconds)
162
+ if s < 120:
163
+ return f"{s:.0f}s"
164
+ if s < 7200:
165
+ return f"{s / 60:.0f}m"
166
+ if s < 2 * 86400:
167
+ return f"{s / 3600:.0f}h"
168
+ if s < 2 * 365.25 * 86400:
169
+ return f"{s / 86400:.0f}d"
170
+ return f"{s / (365.25 * 86400):.1f}y"
stillvalid/doc.py ADDED
@@ -0,0 +1,27 @@
1
+ """Input shape — what the caller knows about a retrieved document."""
2
+ from __future__ import annotations
3
+
4
+ from dataclasses import dataclass
5
+
6
+
7
+ @dataclass(frozen=True)
8
+ class Doc:
9
+ """Everything is optional except id; each field unlocks a cascade layer.
10
+
11
+ Times are Unix epoch seconds.
12
+
13
+ last_verified_at : when YOU last confirmed the content (e.g., index time).
14
+ last_changed_at : last modification time known as of that verification.
15
+ expires_at : explicit expiry, if the source declares one (layer 1).
16
+ source_hash : live content hash fetched just now, if you have it
17
+ (layer 2 — compare against verified_hash).
18
+ verified_hash : content hash recorded at verification time.
19
+ source_modified_at: live Last-Modified fetched just now (layer 3).
20
+ """
21
+ id: str
22
+ last_verified_at: float | None = None
23
+ last_changed_at: float | None = None
24
+ expires_at: float | None = None
25
+ source_hash: str | None = None
26
+ verified_hash: str | None = None
27
+ source_modified_at: float | None = None
stillvalid/history.py ADDED
@@ -0,0 +1,142 @@
1
+ """Local observation log — fuels layers 5 (change rate) and, with enough
2
+ events, training data for layer 6.
3
+
4
+ One SQLite file, standard library only. Record what you saw whenever you
5
+ fetch or reindex a document; the cascade gets smarter as this accrues.
6
+ """
7
+ from __future__ import annotations
8
+
9
+ import sqlite3
10
+ import threading
11
+ from pathlib import Path
12
+
13
+ _SCHEMA_BASE = """
14
+ CREATE TABLE IF NOT EXISTS observations (
15
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
16
+ doc_id TEXT NOT NULL,
17
+ ts INTEGER NOT NULL,
18
+ value_hash TEXT NOT NULL
19
+ );
20
+ CREATE INDEX IF NOT EXISTS idx_obs_doc_ts ON observations (doc_id, ts);
21
+ """
22
+
23
+ #: One sighting per (document, instant, value): re-importing the same
24
+ #: history is an ordinary thing to do (a second import_history call, a
25
+ #: retried job) and must not double-count a document's change rate.
26
+ _UNIQUE_INDEX = ("CREATE UNIQUE INDEX IF NOT EXISTS idx_obs_unique "
27
+ "ON observations (doc_id, ts, value_hash)")
28
+
29
+
30
+ class History:
31
+ """Thread- and process-safe observation log.
32
+
33
+ A single Checker is routinely shared across request threads (a web
34
+ server with the LangChain wrapper, an MCP server), so the connection
35
+ is opened with check_same_thread=False and every statement runs under
36
+ one lock. WAL lets a second process (an indexing job, another server)
37
+ read and write the same file without blocking.
38
+ """
39
+
40
+ def __init__(self, path: str | Path, timeout: float = 10.0):
41
+ self.conn = sqlite3.connect(str(path), check_same_thread=False,
42
+ timeout=timeout)
43
+ self._lock = threading.Lock()
44
+ with self._lock:
45
+ try:
46
+ self.conn.execute("PRAGMA journal_mode=WAL")
47
+ # Default autocheckpoint let the -wal file reach ~4 MB for a
48
+ # 0.8 MB database in a write-heavy soak; a long-lived server
49
+ # never closes, so pin a tighter budget.
50
+ self.conn.execute("PRAGMA wal_autocheckpoint=256")
51
+ except sqlite3.DatabaseError:
52
+ pass # e.g. a network filesystem that refuses WAL
53
+ self.conn.executescript(_SCHEMA_BASE)
54
+ try:
55
+ self.conn.execute(_UNIQUE_INDEX)
56
+ except sqlite3.IntegrityError:
57
+ # a log written before the constraint existed may hold
58
+ # duplicates; collapse them, then enforce it from now on
59
+ self.conn.execute(
60
+ "DELETE FROM observations WHERE id NOT IN ("
61
+ " SELECT MIN(id) FROM observations"
62
+ " GROUP BY doc_id, ts, value_hash)")
63
+ self.conn.execute(_UNIQUE_INDEX)
64
+ self.conn.commit()
65
+
66
+ def close(self):
67
+ with self._lock:
68
+ self.conn.close()
69
+
70
+ def __enter__(self):
71
+ return self
72
+
73
+ def __exit__(self, *exc):
74
+ self.close()
75
+
76
+ def record(self, doc_id: str, ts: int, value_hash: str,
77
+ dedup_s: float = 0.0) -> bool:
78
+ """Log one sighting of a document's content hash.
79
+
80
+ An identical (doc_id, ts, value_hash) row is never stored twice, so
81
+ re-importing a history is harmless. With dedup_s > 0, a sighting
82
+ with the SAME hash as the latest row is additionally skipped unless
83
+ that row is older than dedup_s — keeps high-traffic auto-recording
84
+ from flooding the log while still capturing every change
85
+ immediately. Returns True if a row was written.
86
+ """
87
+ with self._lock:
88
+ if dedup_s > 0:
89
+ prev = self._last(doc_id)
90
+ if prev is not None and prev[1] == value_hash \
91
+ and ts - prev[0] < dedup_s:
92
+ return False
93
+ cur = self.conn.execute(
94
+ "INSERT OR IGNORE INTO observations (doc_id, ts, value_hash) "
95
+ "VALUES (?,?,?)", (doc_id, int(ts), value_hash))
96
+ self.conn.commit()
97
+ return cur.rowcount > 0
98
+
99
+ def last(self, doc_id: str) -> tuple[int, str] | None:
100
+ """(ts, value_hash) of the most recent sighting, or None."""
101
+ with self._lock:
102
+ return self._last(doc_id)
103
+
104
+ def _last(self, doc_id: str) -> tuple[int, str] | None:
105
+ """Caller must hold the lock."""
106
+ row = self.conn.execute(
107
+ "SELECT ts, value_hash FROM observations WHERE doc_id=? "
108
+ "ORDER BY ts DESC, id DESC LIMIT 1", (doc_id,)).fetchone()
109
+ return (row[0], row[1]) if row else None
110
+
111
+ #: how many recent sightings the change-rate estimate looks at. Reading
112
+ #: a document's whole history made judging cost grow linearly with it
113
+ #: (12 ms at 20k sightings) — and recent behavior is the better
114
+ #: predictor anyway, so the window is both faster and more honest.
115
+ RECENT_WINDOW = 500
116
+
117
+ def change_stats(self, doc_id: str,
118
+ window: int | None = None) -> tuple[int, float | None]:
119
+ """(number of observed changes, mean seconds between changes).
120
+
121
+ A change is a transition between consecutive distinct hashes; the
122
+ interval is measured between the observation times of those
123
+ transitions. Needs >= 2 changes for a mean. Only the most recent
124
+ `window` sightings are considered (RECENT_WINDOW by default).
125
+ """
126
+ with self._lock:
127
+ rows = self.conn.execute(
128
+ "SELECT ts, value_hash FROM ("
129
+ " SELECT ts, value_hash, id FROM observations WHERE doc_id=?"
130
+ " ORDER BY ts DESC, id DESC LIMIT ?"
131
+ ") ORDER BY ts ASC, id ASC",
132
+ (doc_id, window or self.RECENT_WINDOW)).fetchall()
133
+ change_ts = []
134
+ prev_hash = None
135
+ for ts, h in rows:
136
+ if prev_hash is not None and h != prev_hash:
137
+ change_ts.append(ts)
138
+ prev_hash = h
139
+ if len(change_ts) < 2:
140
+ return len(change_ts), None
141
+ gaps = [b - a for a, b in zip(change_ts, change_ts[1:])]
142
+ return len(change_ts), sum(gaps) / len(gaps)
@@ -0,0 +1,5 @@
1
+ """Framework integrations — each module requires its framework installed.
2
+
3
+ stillvalid.integrations.langchain — document compressor (retriever stage)
4
+ stillvalid.integrations.llamaindex — node postprocessor
5
+ """