stillvalid 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,185 @@
1
+ """Shared glue — build a stillvalid Doc from a framework metadata dict.
2
+
3
+ Frameworks disagree on metadata key names and time formats; this module
4
+ normalizes both so the two wrappers (and any future one) stay thin.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import hashlib
9
+ from datetime import datetime, timezone
10
+ from typing import Any, Mapping, Sequence
11
+
12
+ from ..doc import Doc
13
+
14
+ #: metadata keys tried in order, per Doc field. "stillvalid_*" prefixed keys
15
+ #: always win so callers can be explicit without renaming their pipeline.
16
+ #: The rest are what real loaders actually write — verified against
17
+ #: LangChain's TextLoader/DirectoryLoader and LlamaIndex's
18
+ #: SimpleDirectoryReader rather than guessed.
19
+ DEFAULT_KEYS: dict[str, Sequence[str]] = {
20
+ "last_verified_at": ("stillvalid_last_verified_at", "last_verified_at",
21
+ "indexed_at", "index_time"),
22
+ "last_changed_at": ("stillvalid_last_changed_at", "last_changed_at",
23
+ "last_modified", "modified_at", "updated_at",
24
+ "last_modified_date"), # LlamaIndex
25
+ "expires_at": ("stillvalid_expires_at", "expires_at", "expiry"),
26
+ "source_hash": ("stillvalid_source_hash", "source_hash"),
27
+ "verified_hash": ("stillvalid_verified_hash", "verified_hash",
28
+ "content_hash"),
29
+ "source_modified_at": ("stillvalid_source_modified_at",
30
+ "source_modified_at"),
31
+ }
32
+
33
+ #: metadata keys tried in order to identify a document. A stable id is not
34
+ #: cosmetic: observations are keyed by it, so falling back to a per-run UUID
35
+ #: would scatter one document's history across many ids and the change-rate
36
+ #: and survival layers could never light up.
37
+ DEFAULT_ID_KEYS: Sequence[str] = (
38
+ "stillvalid_doc_id", "source", "file_path", "url", "id", "doc_id",
39
+ "file_name",
40
+ )
41
+
42
+ _TIME_FIELDS = {"last_verified_at", "last_changed_at", "expires_at",
43
+ "source_modified_at"}
44
+
45
+
46
+ def doc_id_from_metadata(metadata: Mapping[str, Any],
47
+ id_key: str | None = None,
48
+ *fallbacks: Any) -> str | None:
49
+ """Pick a stable document id: explicit key, then common loader keys,
50
+ then framework-supplied fallbacks (ref_doc_id, node id …)."""
51
+ keys = (id_key, *DEFAULT_ID_KEYS) if id_key else DEFAULT_ID_KEYS
52
+ for k in keys:
53
+ v = metadata.get(k)
54
+ if v:
55
+ return str(v)
56
+ for f in fallbacks:
57
+ if f:
58
+ return str(f)
59
+ return None
60
+
61
+
62
+ def to_epoch(value: Any) -> float | None:
63
+ """Accept epoch numbers, ISO-8601 strings, or datetime objects.
64
+
65
+ Naive values (no offset) are read as UTC. That is the only defensible
66
+ default — guessing the machine's zone would make the same metadata mean
67
+ different things on different hosts — but it is a real trap: a naive
68
+ local timestamp from, say, UTC+9 lands nine hours off, which is enough
69
+ to turn "just edited" into "edited this morning" and flip a verdict.
70
+ Pass offsets (or epoch seconds) when the source has a zone.
71
+ """
72
+ if value is None:
73
+ return None
74
+ if isinstance(value, (int, float)):
75
+ return float(value)
76
+ if isinstance(value, datetime):
77
+ if value.tzinfo is None:
78
+ value = value.replace(tzinfo=timezone.utc)
79
+ return value.timestamp()
80
+ if isinstance(value, str):
81
+ try:
82
+ return float(value)
83
+ except ValueError:
84
+ pass
85
+ try:
86
+ dt = datetime.fromisoformat(value.replace("Z", "+00:00"))
87
+ except ValueError:
88
+ return None
89
+ if dt.tzinfo is None:
90
+ dt = dt.replace(tzinfo=timezone.utc)
91
+ return dt.timestamp()
92
+ return None
93
+
94
+
95
+ def doc_from_metadata(doc_id: str, metadata: Mapping[str, Any],
96
+ keys: Mapping[str, Sequence[str]] | None = None) -> Doc:
97
+ keys = keys or DEFAULT_KEYS
98
+ fields: dict[str, Any] = {}
99
+ for field, candidates in keys.items():
100
+ for k in candidates:
101
+ if k in metadata and metadata[k] is not None:
102
+ v = metadata[k]
103
+ fields[field] = to_epoch(v) if field in _TIME_FIELDS else str(v)
104
+ break
105
+ return Doc(id=doc_id, **fields)
106
+
107
+
108
+ def live_signals(ids: Sequence[str], *, timeout: float = 5.0,
109
+ allow_private: bool = False) -> dict[str, dict]:
110
+ """Fetch CURRENT source signals for document ids, cheaply.
111
+
112
+ Index metadata is always a past snapshot, so on its own it can never
113
+ tell you whether the source has moved since — without this, a freshly
114
+ installed wrapper can only answer UNKNOWN. Two sources of truth are
115
+ cheap enough to consult at query time:
116
+
117
+ local path → os.stat().st_mtime (no I/O on the content)
118
+ http(s) URL → conditional HEAD request (no body, batched)
119
+
120
+ Returns {id: {"source_modified_at": …, "source_hash": …}} for the ids
121
+ it could resolve; unresolvable ids are simply absent.
122
+ """
123
+ import os
124
+
125
+ out: dict[str, dict] = {}
126
+ urls = []
127
+ for ident in ids:
128
+ if not ident:
129
+ continue
130
+ low = ident.lower()
131
+ if low.startswith(("http://", "https://")):
132
+ urls.append(ident)
133
+ continue
134
+ try: # local file: current mtime is the signal
135
+ st = os.stat(ident)
136
+ except (OSError, ValueError):
137
+ continue
138
+ out[ident] = {"source_modified_at": st.st_mtime}
139
+ if urls:
140
+ from ..probe import probe_many
141
+ for ident, r in zip(urls, probe_many(urls, timeout=timeout,
142
+ allow_private=allow_private)):
143
+ if not r.ok:
144
+ continue
145
+ sig = {}
146
+ if r.last_modified is not None:
147
+ sig["source_modified_at"] = r.last_modified
148
+ if r.etag:
149
+ sig["source_hash"] = r.etag
150
+ if sig:
151
+ out[ident] = sig
152
+ return out
153
+
154
+
155
+ def verdict_metadata(v) -> dict:
156
+ """Verdict → JSON-safe dict to attach to framework metadata."""
157
+ return {
158
+ "state": v.state.value,
159
+ "action": v.action,
160
+ "usable": v.usable,
161
+ "layer": v.layer,
162
+ "reason": v.reason,
163
+ "summary": v.summary,
164
+ "probability": v.probability,
165
+ "calibrated": v.calibrated,
166
+ }
167
+
168
+
169
+ #: auto-record: identical-hash sightings within this window are not re-logged
170
+ AUTO_RECORD_DEDUP_S = 3600.0
171
+
172
+
173
+ def auto_record(checker, doc_id: str, content: str, now: float) -> None:
174
+ """Log a sighting of retrieved content, if the checker keeps history.
175
+
176
+ Semantics worth knowing: this observes YOUR INDEX's copy, so a change
177
+ becomes visible only after a reindex — timestamps carry that lag. That
178
+ is exactly the fidelity the crude change-rate layer claims (uncalibrated
179
+ hint), so it feeds layer 5 legitimately; calibrated training should
180
+ prefer source-side timestamps when available.
181
+ """
182
+ if checker.history is None or not doc_id or content is None:
183
+ return
184
+ h = hashlib.sha256(content.encode("utf-8", "replace")).hexdigest()
185
+ checker.record(doc_id, h, ts=now, dedup_s=AUTO_RECORD_DEDUP_S)
@@ -0,0 +1,86 @@
1
+ """LangChain integration — a document compressor that judges validity.
2
+
3
+ Use with ContextualCompressionRetriever so every retrieved document is
4
+ checked between retrieval and the LLM:
5
+
6
+ from stillvalid import Checker
7
+ from stillvalid.integrations.langchain import StillValidCompressor
8
+ from langchain.retrievers import ContextualCompressionRetriever
9
+
10
+ retriever = ContextualCompressionRetriever(
11
+ base_compressor=StillValidCompressor(checker=Checker(...)),
12
+ base_retriever=vectorstore.as_retriever(),
13
+ )
14
+
15
+ mode="annotate" (default) attaches the verdict to doc.metadata["stillvalid"]
16
+ so your chain/agent can decide; mode="filter" drops non-usable documents
17
+ outright (UNKNOWN kept unless drop_unknown=True — absence of evidence is
18
+ not evidence of staleness).
19
+
20
+ Requires langchain-core: pip install "stillvalid[langchain]".
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import time
25
+ from typing import Any, Optional, Sequence
26
+
27
+ try:
28
+ from langchain_core.callbacks import Callbacks
29
+ from langchain_core.documents import Document
30
+ from langchain_core.documents.compressor import BaseDocumentCompressor
31
+ except ImportError as exc: # pragma: no cover
32
+ raise ImportError(
33
+ "stillvalid.integrations.langchain requires langchain-core - "
34
+ "pip install 'stillvalid[langchain]'") from exc
35
+
36
+ from ..cascade import Checker
37
+ from .common import (DEFAULT_KEYS, auto_record, doc_from_metadata,
38
+ doc_id_from_metadata, live_signals, verdict_metadata)
39
+
40
+
41
+ class StillValidCompressor(BaseDocumentCompressor):
42
+ """Judge each retrieved document's validity; annotate or filter."""
43
+
44
+ model_config = {"arbitrary_types_allowed": True}
45
+
46
+ checker: Any # stillvalid.Checker
47
+ mode: str = "annotate" # "annotate" | "filter"
48
+ drop_unknown: bool = False # filter mode: drop UNKNOWN too?
49
+ id_key: str | None = None # metadata key holding the document id
50
+ metadata_keys: dict = dict(DEFAULT_KEYS)
51
+ record: bool = True # auto-log sightings when history exists
52
+ live: bool = False # consult the source for CURRENT signals
53
+ live_timeout: float = 5.0
54
+ allow_private: bool = False # live: permit internal addresses
55
+
56
+ def compress_documents(
57
+ self, documents: Sequence[Document], query: str,
58
+ callbacks: Optional[Callbacks] = None,
59
+ ) -> Sequence[Document]:
60
+ now = time.time() # one clock for the batch
61
+ checker: Checker = self.checker
62
+ ids = [doc_id_from_metadata(d.metadata, self.id_key, d.id)
63
+ for d in documents]
64
+ signals = live_signals([i for i in ids if i],
65
+ timeout=self.live_timeout,
66
+ allow_private=self.allow_private) \
67
+ if self.live else {}
68
+ out: list[Document] = []
69
+ for d, doc_id in zip(documents, ids):
70
+ if doc_id is None:
71
+ out.append(d) # unidentifiable — nothing to judge or learn
72
+ continue
73
+ if self.record:
74
+ auto_record(checker, doc_id, d.page_content, now)
75
+ meta = d.metadata
76
+ if doc_id in signals:
77
+ meta = {**meta, **signals[doc_id]}
78
+ v = checker.check(
79
+ doc_from_metadata(doc_id, meta, self.metadata_keys),
80
+ now=now)
81
+ if self.mode == "filter" and not v.usable:
82
+ if v.state.value != "UNKNOWN" or self.drop_unknown:
83
+ continue
84
+ d.metadata["stillvalid"] = verdict_metadata(v)
85
+ out.append(d)
86
+ return out
@@ -0,0 +1,81 @@
1
+ """LlamaIndex integration — a node postprocessor that judges validity.
2
+
3
+ from stillvalid import Checker
4
+ from stillvalid.integrations.llamaindex import StillValidPostprocessor
5
+
6
+ engine = index.as_query_engine(
7
+ node_postprocessors=[StillValidPostprocessor(checker=Checker(...))])
8
+
9
+ mode="annotate" (default) writes the verdict into node.metadata["stillvalid"];
10
+ mode="filter" drops non-usable nodes (UNKNOWN kept unless drop_unknown=True).
11
+
12
+ Requires llama-index-core: pip install "stillvalid[llamaindex]".
13
+ """
14
+ from __future__ import annotations
15
+
16
+ import time
17
+ from typing import Any, List, Optional
18
+
19
+ try:
20
+ from llama_index.core.postprocessor.types import BaseNodePostprocessor
21
+ from llama_index.core.schema import NodeWithScore, QueryBundle
22
+ except ImportError as exc: # pragma: no cover
23
+ raise ImportError(
24
+ "stillvalid.integrations.llamaindex requires llama-index-core - "
25
+ "pip install 'stillvalid[llamaindex]'") from exc
26
+
27
+ from .common import (DEFAULT_KEYS, auto_record, doc_from_metadata,
28
+ doc_id_from_metadata, live_signals, verdict_metadata)
29
+
30
+
31
+ class StillValidPostprocessor(BaseNodePostprocessor):
32
+ """Judge each retrieved node's validity; annotate or filter."""
33
+
34
+ model_config = {"arbitrary_types_allowed": True}
35
+
36
+ checker: Any # stillvalid.Checker
37
+ mode: str = "annotate" # "annotate" | "filter"
38
+ drop_unknown: bool = False
39
+ id_key: str | None = None # metadata key holding the document id
40
+ metadata_keys: dict = dict(DEFAULT_KEYS)
41
+ record: bool = True # auto-log sightings when history exists
42
+ live: bool = False # consult the source for CURRENT signals
43
+ live_timeout: float = 5.0
44
+ allow_private: bool = False # live: permit internal addresses
45
+
46
+ @classmethod
47
+ def class_name(cls) -> str:
48
+ return "StillValidPostprocessor"
49
+
50
+ def _postprocess_nodes(
51
+ self, nodes: List[NodeWithScore],
52
+ query_bundle: Optional[QueryBundle] = None,
53
+ ) -> List[NodeWithScore]:
54
+ now = time.time() # one clock for the batch
55
+ # node_id is a per-ingest UUID — only a last resort, and never a
56
+ # stable key to learn from
57
+ ids = [doc_id_from_metadata(n.node.metadata, self.id_key,
58
+ getattr(n.node, "ref_doc_id", None),
59
+ n.node.node_id) for n in nodes]
60
+ signals = live_signals([i for i in ids if i],
61
+ timeout=self.live_timeout,
62
+ allow_private=self.allow_private) \
63
+ if self.live else {}
64
+ out: List[NodeWithScore] = []
65
+ for nws, doc_id in zip(nodes, ids):
66
+ node = nws.node
67
+ if self.record:
68
+ auto_record(self.checker, doc_id,
69
+ node.get_content(), now)
70
+ meta = node.metadata
71
+ if doc_id in signals:
72
+ meta = {**meta, **signals[doc_id]}
73
+ v = self.checker.check(
74
+ doc_from_metadata(doc_id, meta, self.metadata_keys),
75
+ now=now)
76
+ if self.mode == "filter" and not v.usable:
77
+ if v.state.value != "UNKNOWN" or self.drop_unknown:
78
+ continue
79
+ node.metadata["stillvalid"] = verdict_metadata(v)
80
+ out.append(nws)
81
+ return out
@@ -0,0 +1,247 @@
1
+ """MCP server — expose validity checking as tools an agent can call.
2
+
3
+ An agent that is about to act on retrieved evidence asks this server
4
+ whether the evidence is still trustworthy, and records sightings so the
5
+ cascade learns. Run over stdio:
6
+
7
+ python -m stillvalid.mcp_server
8
+
9
+ Configuration via environment variables:
10
+ STILLVALID_HISTORY_DB path to the observation log (created if absent;
11
+ omit to run without layers 5/record)
12
+ STILLVALID_PARAMS path to gate-passed survival params JSON (layer 6)
13
+
14
+ Claude Desktop / MCP client config example:
15
+
16
+ {"mcpServers": {"stillvalid": {
17
+ "command": "python", "args": ["-m", "stillvalid.mcp_server"],
18
+ "env": {"STILLVALID_HISTORY_DB": "C:/data/observations.db"}}}}
19
+
20
+ Requires the mcp package: pip install "stillvalid[mcp]".
21
+ """
22
+ from __future__ import annotations
23
+
24
+ import os
25
+ import time
26
+
27
+ from .cascade import Checker
28
+ from .doc import Doc
29
+ from .integrations.common import to_epoch, verdict_metadata
30
+
31
+ _checker: Checker | None = None
32
+
33
+
34
+ def get_checker() -> Checker:
35
+ global _checker
36
+ if _checker is None:
37
+ _checker = Checker(
38
+ survival_params=os.environ.get("STILLVALID_PARAMS") or None,
39
+ history_db=os.environ.get("STILLVALID_HISTORY_DB") or None)
40
+ return _checker
41
+
42
+
43
+ # ── tool implementations (plain functions — unit-testable without MCP) ──
44
+
45
+ def check_validity_impl(doc_id: str,
46
+ last_verified_at: float | str | None = None,
47
+ last_changed_at: float | str | None = None,
48
+ expires_at: float | str | None = None,
49
+ source_modified_at: float | str | None = None,
50
+ source_hash: str | None = None,
51
+ verified_hash: str | None = None) -> dict:
52
+ v = get_checker().check(Doc(
53
+ id=doc_id,
54
+ last_verified_at=to_epoch(last_verified_at),
55
+ last_changed_at=to_epoch(last_changed_at),
56
+ expires_at=to_epoch(expires_at),
57
+ source_modified_at=to_epoch(source_modified_at),
58
+ source_hash=source_hash, verified_hash=verified_hash))
59
+ return {"doc_id": doc_id, **verdict_metadata(v)}
60
+
61
+
62
+ def check_validity_batch_impl(docs: list[dict]) -> dict:
63
+ # Agents assemble these dicts from whatever metadata they scraped, so a
64
+ # malformed entry is routine. Isolate it: one bad item must not discard
65
+ # the verdicts for the good ones.
66
+ verdicts = []
67
+ for d in docs:
68
+ try:
69
+ verdicts.append(check_validity_impl(**d))
70
+ except Exception as e:
71
+ doc_id = str(d.get("doc_id", "?")) if isinstance(d, dict) else "?"
72
+ verdicts.append({
73
+ "doc_id": doc_id, "state": "UNKNOWN", "action": "verify",
74
+ "usable": False, "layer": "none",
75
+ "reason": f"could not be read: {type(e).__name__}",
76
+ "summary": f"'{doc_id}' could not be judged - its input was "
77
+ f"malformed ({type(e).__name__}). Treat it as "
78
+ "unverified.",
79
+ "probability": None, "calibrated": False})
80
+ n = len(verdicts)
81
+ # stale ("known to have moved") and unknown ("no evidence either way")
82
+ # call for the same caution but are not the same claim — saying so keeps
83
+ # the relayed sentence honest.
84
+ stale = [v["doc_id"] for v in verdicts if v["state"] in ("STALE", "VERIFY")]
85
+ unknown = [v["doc_id"] for v in verdicts if v["state"] == "UNKNOWN"]
86
+ if not stale and not unknown:
87
+ digest = f"All {n} documents are safe to use - no re-verification needed."
88
+ else:
89
+ parts = []
90
+ if stale:
91
+ parts.append(f"{len(stale)} changed or uncertain "
92
+ f"({', '.join(stale)}) - re-fetch before quoting")
93
+ if unknown:
94
+ parts.append(f"{len(unknown)} unjudgeable for lack of evidence "
95
+ f"({', '.join(unknown)}) - treat as unverified")
96
+ digest = f"Of {n} documents: " + "; ".join(parts) + "."
97
+ return {"digest": digest, "verdicts": verdicts}
98
+
99
+
100
+ def import_history_impl(git_repo: str | None = None,
101
+ csv_path: str | None = None,
102
+ prefix: str = "") -> dict:
103
+ from . import backfill
104
+
105
+ ck = get_checker()
106
+ if git_repo:
107
+ n = backfill.from_git(ck, git_repo, prefix=prefix)
108
+ src = f"git repository {git_repo}"
109
+ elif csv_path:
110
+ n = backfill.from_csv(ck, csv_path, prefix=prefix)
111
+ src = f"CSV {csv_path}"
112
+ else:
113
+ raise ValueError("pass git_repo or csv_path")
114
+ s = backfill.summarize(ck)
115
+ return {
116
+ "imported": n, "source": src, **s,
117
+ "summary": (f"Imported {n:,} past changes from {src}. "
118
+ f"{s['change_rate_ready']:,} of {s['documents']:,} "
119
+ "documents now have enough history to be judged on "
120
+ "their update behavior, with no waiting."),
121
+ }
122
+
123
+
124
+ def refresh_evidence_impl(doc_id: str, source: str | None = None,
125
+ known_hash: str | None = None,
126
+ allow_private: bool = False) -> dict:
127
+ from .refresh import refresh
128
+
129
+ return refresh(get_checker(), doc_id, source, known_hash=known_hash,
130
+ allow_private=allow_private)
131
+
132
+
133
+ def record_observation_impl(doc_id: str, value_hash: str,
134
+ ts: float | str | None = None) -> dict:
135
+ get_checker().record(doc_id, value_hash, ts=to_epoch(ts))
136
+ return {"recorded": True, "doc_id": doc_id,
137
+ "ts": to_epoch(ts) or time.time(),
138
+ "summary": f"Recorded a sighting of '{doc_id}'. Its update "
139
+ "behavior is being learned from these observations."}
140
+
141
+
142
+ # ── MCP wiring ───────────────────────────────────────────────
143
+
144
+ def build_server():
145
+ try:
146
+ from mcp.server.mcpserver import MCPServer as _Server # mcp >= 2
147
+ except ImportError:
148
+ try:
149
+ from mcp.server.fastmcp import FastMCP as _Server # mcp 1.x
150
+ except ImportError as exc: # pragma: no cover
151
+ raise ImportError("the MCP server requires the mcp package - "
152
+ "pip install 'stillvalid[mcp]'") from exc
153
+
154
+ from . import __version__
155
+ try:
156
+ mcp = _Server("stillvalid", version=__version__)
157
+ except TypeError: # mcp 1.x FastMCP takes no version
158
+ mcp = _Server("stillvalid")
159
+
160
+ @mcp.tool()
161
+ def check_validity(doc_id: str,
162
+ last_verified_at: float | str | None = None,
163
+ last_changed_at: float | str | None = None,
164
+ expires_at: float | str | None = None,
165
+ source_modified_at: float | str | None = None,
166
+ source_hash: str | None = None,
167
+ verified_hash: str | None = None) -> dict:
168
+ """Judge whether a piece of retrieved evidence is still valid.
169
+
170
+ Call BEFORE acting on a document (quoting a policy, using a price,
171
+ executing a workflow step). Returns state VALID / LIKELY_VALID /
172
+ VERIFY / STALE / UNKNOWN plus a one-sentence `summary` — relay that
173
+ summary to the user so they can see why evidence was trusted or
174
+ re-fetched. When the verdict is STALE or VERIFY, do not quote the
175
+ cached copy: call refresh_evidence to get the current content and
176
+ answer from that instead. Timestamps are epoch seconds or ISO-8601
177
+ strings.
178
+ """
179
+ return check_validity_impl(
180
+ doc_id, last_verified_at, last_changed_at, expires_at,
181
+ source_modified_at, source_hash, verified_hash)
182
+
183
+ @mcp.tool()
184
+ def check_validity_batch(docs: list[dict]) -> dict:
185
+ """Judge several documents at once (one shared clock).
186
+
187
+ Each item: {"doc_id": ..., "last_verified_at": ...,
188
+ "last_changed_at": ..., optionally expires_at /
189
+ source_modified_at / source_hash / verified_hash}.
190
+ Returns {"digest": one-sentence overview to relay to the user,
191
+ "verdicts": [...]}. Act on the digest: re-fetch the listed
192
+ documents before quoting them.
193
+ """
194
+ return check_validity_batch_impl(docs)
195
+
196
+ @mcp.tool()
197
+ def import_history(git_repo: str | None = None,
198
+ csv_path: str | None = None,
199
+ prefix: str = "") -> dict:
200
+ """Load a corpus's PAST changes so judging works immediately.
201
+
202
+ Use this first, whenever the user points at a knowledge base that
203
+ already keeps history — a git repository of docs, or an export of
204
+ wiki revisions / CMS audit rows as CSV (columns: doc_id, ts[, rev]).
205
+ It takes seconds and removes the need to observe anything for
206
+ weeks before the update-behavior layers can say anything.
207
+ Relay the returned summary so the user sees what became judgeable.
208
+ """
209
+ return import_history_impl(git_repo, csv_path, prefix)
210
+
211
+ @mcp.tool()
212
+ def refresh_evidence(doc_id: str, source: str | None = None,
213
+ known_hash: str | None = None,
214
+ allow_private: bool = False) -> dict:
215
+ """Fetch the CURRENT content of a stale document and use it.
216
+
217
+ Call this when check_validity says STALE or VERIFY: it retrieves
218
+ the live content (local path or http(s) URL; `source` defaults to
219
+ doc_id), logs the sighting so future judgments improve, and — when
220
+ you pass the hash of your cached copy as known_hash — tells you
221
+ whether the content actually changed. Answer the user from the
222
+ returned `content`, not from your cached copy, and relay the
223
+ `summary`. Set allow_private=True only for internal addresses.
224
+ """
225
+ return refresh_evidence_impl(doc_id, source, known_hash,
226
+ allow_private)
227
+
228
+ @mcp.tool()
229
+ def record_observation(doc_id: str, value_hash: str,
230
+ ts: float | str | None = None) -> dict:
231
+ """Log that you saw a document's current content hash.
232
+
233
+ Call whenever evidence is fetched or re-verified — the validity
234
+ cascade learns each document's change behavior from this log.
235
+ Requires STILLVALID_HISTORY_DB to be configured.
236
+ """
237
+ return record_observation_impl(doc_id, value_hash, ts)
238
+
239
+ return mcp
240
+
241
+
242
+ def main() -> None:
243
+ build_server().run()
244
+
245
+
246
+ if __name__ == "__main__":
247
+ main()