provena 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- provena/__init__.py +26 -0
- provena/__main__.py +3 -0
- provena/cli/__init__.py +0 -0
- provena/cli/main.py +315 -0
- provena/exporters/__init__.py +5 -0
- provena/exporters/otel.py +70 -0
- provena/hasher.py +84 -0
- provena/integrations/__init__.py +0 -0
- provena/integrations/langchain.py +90 -0
- provena/integrations/llamaindex.py +79 -0
- provena/models.py +287 -0
- provena/py.typed +0 -0
- provena/storage.py +340 -0
- provena/trail.py +619 -0
- provena/validators/__init__.py +6 -0
- provena/validators/freshness.py +257 -0
- provena/validators/provenance.py +64 -0
- provena-0.6.0.dist-info/METADATA +230 -0
- provena-0.6.0.dist-info/RECORD +22 -0
- provena-0.6.0.dist-info/WHEEL +4 -0
- provena-0.6.0.dist-info/entry_points.txt +2 -0
- provena-0.6.0.dist-info/licenses/LICENSE +161 -0
provena/__init__.py
ADDED
|
@@ -0,0 +1,26 @@
|
|
|
1
|
+
"""Provena — Context governance for agentic AI systems."""
|
|
2
|
+
|
|
3
|
+
from provena.models import (
|
|
4
|
+
ChainVerdict,
|
|
5
|
+
ContextEntry,
|
|
6
|
+
ContextSource,
|
|
7
|
+
FreshnessResult,
|
|
8
|
+
ProvenanceMetadata,
|
|
9
|
+
TrailRecord,
|
|
10
|
+
ValidationResult,
|
|
11
|
+
)
|
|
12
|
+
from provena.trail import ContextTrail
|
|
13
|
+
|
|
14
|
+
__version__ = "0.6.0"
|
|
15
|
+
|
|
16
|
+
__all__ = [
|
|
17
|
+
"ChainVerdict",
|
|
18
|
+
"ContextEntry",
|
|
19
|
+
"ContextSource",
|
|
20
|
+
"ContextTrail",
|
|
21
|
+
"FreshnessResult",
|
|
22
|
+
"ProvenanceMetadata",
|
|
23
|
+
"TrailRecord",
|
|
24
|
+
"ValidationResult",
|
|
25
|
+
"__version__",
|
|
26
|
+
]
|
provena/__main__.py
ADDED
provena/cli/__init__.py
ADDED
|
File without changes
|
provena/cli/main.py
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from datetime import datetime, timezone
|
|
6
|
+
from typing import Any
|
|
7
|
+
|
|
8
|
+
import click
|
|
9
|
+
|
|
10
|
+
from provena.trail import ContextTrail
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@click.group()
|
|
14
|
+
@click.option(
|
|
15
|
+
"--db",
|
|
16
|
+
default="provena.db",
|
|
17
|
+
envvar="PROVENA_DB",
|
|
18
|
+
help="Path to Provena database file.",
|
|
19
|
+
type=click.Path(),
|
|
20
|
+
)
|
|
21
|
+
@click.option(
|
|
22
|
+
"--signing-key",
|
|
23
|
+
default=None,
|
|
24
|
+
envvar="PROVENA_SIGNING_KEY",
|
|
25
|
+
help="HMAC signing key for chain verification.",
|
|
26
|
+
)
|
|
27
|
+
@click.version_option(package_name="provena")
|
|
28
|
+
@click.pass_context
|
|
29
|
+
def cli(ctx: click.Context, db: str, signing_key: str | None) -> None:
|
|
30
|
+
"""Provena — Context governance for agentic AI systems."""
|
|
31
|
+
ctx.ensure_object(dict)
|
|
32
|
+
ctx.obj["db"] = db
|
|
33
|
+
ctx.obj["signing_key"] = signing_key
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@cli.command()
|
|
37
|
+
@click.option("--source", "-s", default=None, help="Filter by source type.")
|
|
38
|
+
@click.option("--limit", "-n", default=20, type=int, help="Max records to show.")
|
|
39
|
+
@click.option(
|
|
40
|
+
"--from",
|
|
41
|
+
"start",
|
|
42
|
+
type=click.DateTime(formats=["%Y-%m-%d"]),
|
|
43
|
+
default=None,
|
|
44
|
+
help="Filter records from this date (YYYY-MM-DD).",
|
|
45
|
+
)
|
|
46
|
+
@click.option(
|
|
47
|
+
"--to",
|
|
48
|
+
"end",
|
|
49
|
+
type=click.DateTime(formats=["%Y-%m-%d"]),
|
|
50
|
+
default=None,
|
|
51
|
+
help="Filter records up to this date (YYYY-MM-DD).",
|
|
52
|
+
)
|
|
53
|
+
@click.option(
|
|
54
|
+
"--format",
|
|
55
|
+
"fmt",
|
|
56
|
+
default="table",
|
|
57
|
+
type=click.Choice(["table", "json"]),
|
|
58
|
+
help="Output format.",
|
|
59
|
+
)
|
|
60
|
+
@click.pass_context
|
|
61
|
+
def audit(
|
|
62
|
+
ctx: click.Context,
|
|
63
|
+
source: str | None,
|
|
64
|
+
limit: int,
|
|
65
|
+
start: datetime | None,
|
|
66
|
+
end: datetime | None,
|
|
67
|
+
fmt: str,
|
|
68
|
+
) -> None:
|
|
69
|
+
"""Query the context governance audit log."""
|
|
70
|
+
db_path = ctx.obj["db"]
|
|
71
|
+
if not os.path.exists(db_path):
|
|
72
|
+
click.echo(f"Database not found: {db_path}", err=True)
|
|
73
|
+
ctx.exit(1)
|
|
74
|
+
return
|
|
75
|
+
|
|
76
|
+
trail = ContextTrail(storage_path=db_path, signing_key=ctx.obj.get("signing_key"))
|
|
77
|
+
try:
|
|
78
|
+
records = trail.query(source=source, limit=limit, start=start, end=end)
|
|
79
|
+
|
|
80
|
+
if not records:
|
|
81
|
+
click.echo("No records found.")
|
|
82
|
+
return
|
|
83
|
+
|
|
84
|
+
if fmt == "json":
|
|
85
|
+
click.echo(json.dumps(records, indent=2, default=str))
|
|
86
|
+
else:
|
|
87
|
+
_print_table(records)
|
|
88
|
+
finally:
|
|
89
|
+
trail.close()
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
@cli.command()
|
|
93
|
+
@click.pass_context
|
|
94
|
+
def verify(ctx: click.Context) -> None:
|
|
95
|
+
"""Verify the integrity of the hash-chained audit trail."""
|
|
96
|
+
db_path = ctx.obj["db"]
|
|
97
|
+
if not os.path.exists(db_path):
|
|
98
|
+
click.echo(f"Database not found: {db_path}", err=True)
|
|
99
|
+
ctx.exit(1)
|
|
100
|
+
return
|
|
101
|
+
|
|
102
|
+
trail = ContextTrail(storage_path=db_path, signing_key=ctx.obj.get("signing_key"))
|
|
103
|
+
try:
|
|
104
|
+
verdict = trail.verify_chain()
|
|
105
|
+
|
|
106
|
+
if verdict.total_records == 0:
|
|
107
|
+
click.echo("EMPTY — No records in the audit trail.")
|
|
108
|
+
return
|
|
109
|
+
|
|
110
|
+
if verdict.intact:
|
|
111
|
+
click.echo(
|
|
112
|
+
click.style("PASS", fg="green", bold=True)
|
|
113
|
+
+ f" — Chain intact ({verdict.total_records} records verified)"
|
|
114
|
+
)
|
|
115
|
+
else:
|
|
116
|
+
click.echo(
|
|
117
|
+
click.style("FAIL", fg="red", bold=True) + f" — {verdict.details}"
|
|
118
|
+
)
|
|
119
|
+
ctx.exit(1)
|
|
120
|
+
finally:
|
|
121
|
+
trail.close()
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
@cli.command()
|
|
125
|
+
@click.option(
|
|
126
|
+
"--format",
|
|
127
|
+
"fmt",
|
|
128
|
+
default="json",
|
|
129
|
+
type=click.Choice(["json", "text", "csv"]),
|
|
130
|
+
help="Output format.",
|
|
131
|
+
)
|
|
132
|
+
@click.option("--output", "-o", default=None, type=click.Path(), help="Write to file.")
|
|
133
|
+
@click.pass_context
|
|
134
|
+
def report(ctx: click.Context, fmt: str, output: str | None) -> None:
|
|
135
|
+
"""Generate a context governance compliance report."""
|
|
136
|
+
db_path = ctx.obj["db"]
|
|
137
|
+
if not os.path.exists(db_path):
|
|
138
|
+
click.echo(f"Database not found: {db_path}", err=True)
|
|
139
|
+
ctx.exit(1)
|
|
140
|
+
return
|
|
141
|
+
|
|
142
|
+
trail = ContextTrail(storage_path=db_path, signing_key=ctx.obj.get("signing_key"))
|
|
143
|
+
try:
|
|
144
|
+
summary = trail.summary()
|
|
145
|
+
verdict = trail.verify_chain()
|
|
146
|
+
|
|
147
|
+
report_data = {
|
|
148
|
+
"generated_at": datetime.now(timezone.utc).isoformat(),
|
|
149
|
+
"database": db_path,
|
|
150
|
+
"total_records": summary["total"],
|
|
151
|
+
"chain_integrity": {
|
|
152
|
+
"status": "INTACT" if verdict.intact else "BROKEN",
|
|
153
|
+
"records_verified": verdict.total_records,
|
|
154
|
+
"broken_at": verdict.broken_at,
|
|
155
|
+
},
|
|
156
|
+
"provenance": summary.get("provenance", {}),
|
|
157
|
+
"freshness": summary.get("freshness", {}),
|
|
158
|
+
"sources": summary.get("sources", {}),
|
|
159
|
+
"signed": summary.get("signed", False),
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
if fmt == "json":
|
|
163
|
+
content = json.dumps(report_data, indent=2)
|
|
164
|
+
elif fmt == "csv":
|
|
165
|
+
content = trail.export(format="csv")
|
|
166
|
+
else:
|
|
167
|
+
content = _format_text_report(report_data)
|
|
168
|
+
|
|
169
|
+
if output:
|
|
170
|
+
with open(output, "w") as f:
|
|
171
|
+
f.write(content)
|
|
172
|
+
click.echo(f"Report written to {output}")
|
|
173
|
+
else:
|
|
174
|
+
click.echo(content)
|
|
175
|
+
finally:
|
|
176
|
+
trail.close()
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
@cli.command()
|
|
180
|
+
@click.pass_context
|
|
181
|
+
def summary(ctx: click.Context) -> None:
|
|
182
|
+
"""Show a quick summary of the audit trail."""
|
|
183
|
+
db_path = ctx.obj["db"]
|
|
184
|
+
if not os.path.exists(db_path):
|
|
185
|
+
click.echo(f"Database not found: {db_path}", err=True)
|
|
186
|
+
ctx.exit(1)
|
|
187
|
+
return
|
|
188
|
+
|
|
189
|
+
trail = ContextTrail(storage_path=db_path, signing_key=ctx.obj.get("signing_key"))
|
|
190
|
+
try:
|
|
191
|
+
s = trail.summary()
|
|
192
|
+
h = trail.health()
|
|
193
|
+
|
|
194
|
+
click.echo(f"Records: {s['total']}")
|
|
195
|
+
click.echo(f"Backend: {h.get('backend', 'unknown')}")
|
|
196
|
+
click.echo(f"Signed: {'Yes' if s.get('signed') else 'No'}")
|
|
197
|
+
|
|
198
|
+
if s["total"] > 0:
|
|
199
|
+
click.echo("")
|
|
200
|
+
click.echo("Provenance:")
|
|
201
|
+
for status, count in sorted(s.get("provenance", {}).items()):
|
|
202
|
+
click.echo(f" {status:12s} {count}")
|
|
203
|
+
|
|
204
|
+
click.echo("")
|
|
205
|
+
click.echo("Freshness:")
|
|
206
|
+
for status, count in sorted(s.get("freshness", {}).items()):
|
|
207
|
+
click.echo(f" {status:12s} {count}")
|
|
208
|
+
|
|
209
|
+
click.echo("")
|
|
210
|
+
click.echo("Sources:")
|
|
211
|
+
for src, count in sorted(s.get("sources", {}).items()):
|
|
212
|
+
click.echo(f" {src:12s} {count}")
|
|
213
|
+
finally:
|
|
214
|
+
trail.close()
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
def _print_table(records: list[dict[str, Any]]) -> None:
|
|
218
|
+
try:
|
|
219
|
+
import rich # noqa: F401
|
|
220
|
+
|
|
221
|
+
_print_rich_table(records)
|
|
222
|
+
except ImportError:
|
|
223
|
+
_print_plain_table(records)
|
|
224
|
+
|
|
225
|
+
|
|
226
|
+
def _print_rich_table(records: list[dict[str, Any]]) -> None:
|
|
227
|
+
from rich.console import Console
|
|
228
|
+
from rich.table import Table
|
|
229
|
+
|
|
230
|
+
table = Table(title="Provena Audit Trail")
|
|
231
|
+
table.add_column("ID", style="dim", width=5)
|
|
232
|
+
table.add_column("Timestamp", width=20)
|
|
233
|
+
table.add_column("Source", style="cyan")
|
|
234
|
+
table.add_column("Name")
|
|
235
|
+
table.add_column("Hash", style="dim", width=12)
|
|
236
|
+
table.add_column("Provenance")
|
|
237
|
+
table.add_column("Freshness")
|
|
238
|
+
|
|
239
|
+
for r in records:
|
|
240
|
+
ts = r.get("timestamp", "")[:19]
|
|
241
|
+
ch = r.get("content_hash", "")[:12]
|
|
242
|
+
prov = r.get("provenance_status", "?")
|
|
243
|
+
fresh = r.get("freshness_status", "?")
|
|
244
|
+
|
|
245
|
+
prov_style = {"VALID": "green", "MISSING": "red", "INCOMPLETE": "yellow"}.get(
|
|
246
|
+
prov, ""
|
|
247
|
+
)
|
|
248
|
+
fresh_style = {"FRESH": "green", "STALE": "red", "UNKNOWN": "dim"}.get(
|
|
249
|
+
fresh, ""
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
table.add_row(
|
|
253
|
+
str(r.get("id", "")),
|
|
254
|
+
ts,
|
|
255
|
+
r.get("source", ""),
|
|
256
|
+
r.get("source_name", ""),
|
|
257
|
+
ch,
|
|
258
|
+
f"[{prov_style}]{prov}[/{prov_style}]" if prov_style else prov,
|
|
259
|
+
f"[{fresh_style}]{fresh}[/{fresh_style}]" if fresh_style else fresh,
|
|
260
|
+
)
|
|
261
|
+
|
|
262
|
+
Console().print(table)
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _print_plain_table(records: list[dict[str, Any]]) -> None:
|
|
266
|
+
header = f"{'ID':>5} {'Timestamp':20s} {'Source':12s} {'Name':15s} {'Hash':12s} {'Prov':12s} {'Fresh':7s}"
|
|
267
|
+
click.echo(header)
|
|
268
|
+
click.echo("-" * len(header))
|
|
269
|
+
for r in records:
|
|
270
|
+
click.echo(
|
|
271
|
+
f"{r.get('id', ''):>5} "
|
|
272
|
+
f"{r.get('timestamp', '')[:19]:20s} "
|
|
273
|
+
f"{r.get('source', ''):12s} "
|
|
274
|
+
f"{r.get('source_name', ''):15s} "
|
|
275
|
+
f"{r.get('content_hash', '')[:12]:12s} "
|
|
276
|
+
f"{r.get('provenance_status', '?'):12s} "
|
|
277
|
+
f"{r.get('freshness_status', '?'):7s}"
|
|
278
|
+
)
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def _format_text_report(data: dict[str, Any]) -> str:
|
|
282
|
+
lines = [
|
|
283
|
+
"=" * 50,
|
|
284
|
+
"PROVENA GOVERNANCE REPORT",
|
|
285
|
+
"=" * 50,
|
|
286
|
+
f"Generated: {data['generated_at']}",
|
|
287
|
+
f"Database: {data['database']}",
|
|
288
|
+
f"Records: {data['total_records']}",
|
|
289
|
+
f"Signed: {'Yes' if data.get('signed') else 'No'}",
|
|
290
|
+
"",
|
|
291
|
+
"Chain Integrity:",
|
|
292
|
+
f" Status: {data['chain_integrity']['status']}",
|
|
293
|
+
f" Verified: {data['chain_integrity']['records_verified']} records",
|
|
294
|
+
]
|
|
295
|
+
|
|
296
|
+
if data["chain_integrity"]["broken_at"] is not None:
|
|
297
|
+
lines.append(f" Broken at record: {data['chain_integrity']['broken_at']}")
|
|
298
|
+
|
|
299
|
+
lines.append("")
|
|
300
|
+
lines.append("Provenance:")
|
|
301
|
+
for status, count in sorted(data.get("provenance", {}).items()):
|
|
302
|
+
lines.append(f" {status:12s} {count}")
|
|
303
|
+
|
|
304
|
+
lines.append("")
|
|
305
|
+
lines.append("Freshness:")
|
|
306
|
+
for status, count in sorted(data.get("freshness", {}).items()):
|
|
307
|
+
lines.append(f" {status:12s} {count}")
|
|
308
|
+
|
|
309
|
+
lines.append("")
|
|
310
|
+
lines.append("Sources:")
|
|
311
|
+
for src, count in sorted(data.get("sources", {}).items()):
|
|
312
|
+
lines.append(f" {src:12s} {count}")
|
|
313
|
+
|
|
314
|
+
lines.append("=" * 50)
|
|
315
|
+
return "\n".join(lines)
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""OpenTelemetry span exporter for governance events."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import TYPE_CHECKING, Any
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from provena.models import TrailRecord
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from opentelemetry import trace
|
|
12
|
+
|
|
13
|
+
_HAS_OTEL = True
|
|
14
|
+
except ImportError:
|
|
15
|
+
_HAS_OTEL = False
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
class OTelExporter:
|
|
19
|
+
"""Emits OpenTelemetry spans for each context governance event.
|
|
20
|
+
|
|
21
|
+
Requires the ``opentelemetry-api`` package. When disabled or when OTel
|
|
22
|
+
is not installed, all methods are safe no-ops.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
def __init__(
|
|
26
|
+
self,
|
|
27
|
+
enabled: bool = True,
|
|
28
|
+
service_name: str = "provena",
|
|
29
|
+
version: str = "",
|
|
30
|
+
tracer: Any = None,
|
|
31
|
+
) -> None:
|
|
32
|
+
self._enabled = enabled and _HAS_OTEL
|
|
33
|
+
self._tracer: Any = tracer
|
|
34
|
+
if self._enabled and self._tracer is None:
|
|
35
|
+
self._tracer = trace.get_tracer(service_name, version or None)
|
|
36
|
+
|
|
37
|
+
@property
|
|
38
|
+
def enabled(self) -> bool:
|
|
39
|
+
"""Whether OTel export is active."""
|
|
40
|
+
return self._enabled
|
|
41
|
+
|
|
42
|
+
def emit(self, record: TrailRecord) -> None:
|
|
43
|
+
"""Emit a span for the given trail record."""
|
|
44
|
+
if not self._enabled or self._tracer is None:
|
|
45
|
+
return
|
|
46
|
+
|
|
47
|
+
entry = record.entry
|
|
48
|
+
attributes: dict[str, str | int | bool] = {
|
|
49
|
+
"provena.source": entry.source.value,
|
|
50
|
+
"provena.source_name": entry.source_name,
|
|
51
|
+
"provena.content_hash": entry.content_hash,
|
|
52
|
+
"provena.chain_hash": record.chain_hash,
|
|
53
|
+
"provena.timestamp": entry.timestamp.isoformat(),
|
|
54
|
+
"provena.content_type": entry.content_type,
|
|
55
|
+
"provena.truncated": entry.truncated,
|
|
56
|
+
}
|
|
57
|
+
|
|
58
|
+
if record.provenance_result:
|
|
59
|
+
attributes["provena.provenance_status"] = record.provenance_result.status
|
|
60
|
+
if record.freshness_result:
|
|
61
|
+
attributes["provena.freshness_status"] = record.freshness_result.status
|
|
62
|
+
|
|
63
|
+
span = self._tracer.start_span(
|
|
64
|
+
name=f"provena.track.{entry.source_name}",
|
|
65
|
+
attributes=attributes,
|
|
66
|
+
)
|
|
67
|
+
import contextlib
|
|
68
|
+
|
|
69
|
+
with contextlib.suppress(Exception):
|
|
70
|
+
span.end()
|
provena/hasher.py
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
1
|
+
"""SHA-256 hash chain computation and verification."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import hashlib
|
|
6
|
+
import hmac
|
|
7
|
+
|
|
8
|
+
GENESIS_HASH = hashlib.sha256(b"provena:genesis").hexdigest()
|
|
9
|
+
HASH_ALGORITHM = "sha256"
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
class ChainHasher:
|
|
13
|
+
"""Computes and verifies SHA-256 hash chain links.
|
|
14
|
+
|
|
15
|
+
Supports optional HMAC-SHA256 signing when a signing key is provided.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
def __init__(self, signing_key: bytes | None = None) -> None:
|
|
19
|
+
"""Initialize the hasher with an optional HMAC signing key."""
|
|
20
|
+
self._signing_key = signing_key
|
|
21
|
+
|
|
22
|
+
@property
|
|
23
|
+
def is_signed(self) -> bool:
|
|
24
|
+
"""Whether this hasher uses HMAC signing."""
|
|
25
|
+
return self._signing_key is not None
|
|
26
|
+
|
|
27
|
+
def compute_chain_hash(
|
|
28
|
+
self,
|
|
29
|
+
previous_hash: str,
|
|
30
|
+
content_hash: str,
|
|
31
|
+
source: str,
|
|
32
|
+
timestamp: str,
|
|
33
|
+
) -> str:
|
|
34
|
+
"""Compute the hash for a chain link.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
previous_hash: The chain hash of the preceding record.
|
|
38
|
+
content_hash: SHA-256 hex digest of the content.
|
|
39
|
+
source: Source type string (e.g. ``"retriever"``).
|
|
40
|
+
timestamp: ISO-format timestamp string.
|
|
41
|
+
|
|
42
|
+
Returns:
|
|
43
|
+
Hex-encoded SHA-256 or HMAC-SHA256 digest of the link.
|
|
44
|
+
"""
|
|
45
|
+
payload = f"{previous_hash}:{content_hash}:{source}:{timestamp}"
|
|
46
|
+
if self._signing_key is not None:
|
|
47
|
+
return hmac.new(
|
|
48
|
+
self._signing_key,
|
|
49
|
+
payload.encode("utf-8"),
|
|
50
|
+
hashlib.sha256,
|
|
51
|
+
).hexdigest()
|
|
52
|
+
return hashlib.sha256(payload.encode("utf-8")).hexdigest()
|
|
53
|
+
|
|
54
|
+
def verify_link(
|
|
55
|
+
self,
|
|
56
|
+
previous_hash: str,
|
|
57
|
+
content_hash: str,
|
|
58
|
+
source: str,
|
|
59
|
+
timestamp: str,
|
|
60
|
+
expected_hash: str,
|
|
61
|
+
) -> bool:
|
|
62
|
+
"""Verify that a chain link matches the expected hash.
|
|
63
|
+
|
|
64
|
+
Uses constant-time comparison to prevent timing attacks.
|
|
65
|
+
|
|
66
|
+
Args:
|
|
67
|
+
previous_hash: The chain hash of the preceding record.
|
|
68
|
+
content_hash: SHA-256 hex digest of the content.
|
|
69
|
+
source: Source type string.
|
|
70
|
+
timestamp: ISO-format timestamp string.
|
|
71
|
+
expected_hash: The hash value to verify against.
|
|
72
|
+
|
|
73
|
+
Returns:
|
|
74
|
+
True if the recomputed hash matches expected_hash.
|
|
75
|
+
"""
|
|
76
|
+
computed = self.compute_chain_hash(
|
|
77
|
+
previous_hash, content_hash, source, timestamp
|
|
78
|
+
)
|
|
79
|
+
return hmac.compare_digest(computed, expected_hash)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def content_hash(content: bytes) -> str:
|
|
83
|
+
"""Compute the SHA-256 hex digest of raw content bytes."""
|
|
84
|
+
return hashlib.sha256(content).hexdigest()
|
|
File without changes
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING, Any
|
|
4
|
+
from uuid import UUID
|
|
5
|
+
|
|
6
|
+
from provena.models import ContextSource, ProvenanceMetadata
|
|
7
|
+
|
|
8
|
+
if TYPE_CHECKING:
|
|
9
|
+
from provena.trail import ContextTrail
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
from langchain_core.callbacks.base import BaseCallbackHandler
|
|
13
|
+
|
|
14
|
+
class ProvenaCallback(BaseCallbackHandler):
|
|
15
|
+
"""LangChain callback that logs retriever results and tool outputs to a Provena trail.
|
|
16
|
+
|
|
17
|
+
Usage::
|
|
18
|
+
|
|
19
|
+
from provena import ContextTrail
|
|
20
|
+
from provena.integrations.langchain import ProvenaCallback
|
|
21
|
+
|
|
22
|
+
trail = ContextTrail()
|
|
23
|
+
chain = RetrievalQA.from_chain_type(
|
|
24
|
+
llm=llm,
|
|
25
|
+
retriever=retriever,
|
|
26
|
+
callbacks=[ProvenaCallback(trail=trail)],
|
|
27
|
+
)
|
|
28
|
+
"""
|
|
29
|
+
|
|
30
|
+
def __init__(self, trail: ContextTrail, **kwargs: Any) -> None:
|
|
31
|
+
super().__init__(**kwargs)
|
|
32
|
+
self._trail = trail
|
|
33
|
+
|
|
34
|
+
def on_retriever_end(
|
|
35
|
+
self,
|
|
36
|
+
documents: Any,
|
|
37
|
+
*,
|
|
38
|
+
run_id: UUID,
|
|
39
|
+
parent_run_id: UUID | None = None,
|
|
40
|
+
**kwargs: Any,
|
|
41
|
+
) -> None:
|
|
42
|
+
for doc in documents or ():
|
|
43
|
+
content = getattr(doc, "page_content", None)
|
|
44
|
+
if content is None:
|
|
45
|
+
content = str(doc)
|
|
46
|
+
provenance = _extract_langchain_provenance(doc)
|
|
47
|
+
self._trail.log(
|
|
48
|
+
content=content,
|
|
49
|
+
source=ContextSource.RETRIEVER,
|
|
50
|
+
source_name="langchain",
|
|
51
|
+
provenance=provenance,
|
|
52
|
+
metadata={"run_id": str(run_id)},
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
def on_tool_end(
|
|
56
|
+
self,
|
|
57
|
+
output: Any,
|
|
58
|
+
*,
|
|
59
|
+
run_id: UUID,
|
|
60
|
+
parent_run_id: UUID | None = None,
|
|
61
|
+
**kwargs: Any,
|
|
62
|
+
) -> None:
|
|
63
|
+
content = str(output)
|
|
64
|
+
self._trail.log(
|
|
65
|
+
content=content,
|
|
66
|
+
source=ContextSource.TOOL,
|
|
67
|
+
source_name="langchain",
|
|
68
|
+
metadata={"run_id": str(run_id)},
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
except ImportError:
|
|
72
|
+
|
|
73
|
+
class ProvenaCallback: # type: ignore[no-redef]
|
|
74
|
+
"""Placeholder when langchain-core is not installed."""
|
|
75
|
+
|
|
76
|
+
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
|
77
|
+
raise ImportError(
|
|
78
|
+
"langchain-core is required for LangChain integration. "
|
|
79
|
+
"Install with: pip install provena[langchain]"
|
|
80
|
+
)
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def _extract_langchain_provenance(doc: Any) -> ProvenanceMetadata | None:
|
|
84
|
+
meta = getattr(doc, "metadata", None)
|
|
85
|
+
if not isinstance(meta, dict):
|
|
86
|
+
return None
|
|
87
|
+
return ProvenanceMetadata(
|
|
88
|
+
source_url=meta.get("source") or meta.get("source_url"),
|
|
89
|
+
author=meta.get("author"),
|
|
90
|
+
)
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from typing import TYPE_CHECKING, Any
|
|
4
|
+
|
|
5
|
+
from provena.models import ContextSource, ProvenanceMetadata
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from provena.trail import ContextTrail
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from llama_index.core.postprocessor.types import BaseNodePostprocessor
|
|
12
|
+
from llama_index.core.schema import NodeWithScore, QueryBundle
|
|
13
|
+
from pydantic import ConfigDict
|
|
14
|
+
|
|
15
|
+
class ProvenaPostprocessor(BaseNodePostprocessor):
|
|
16
|
+
"""LlamaIndex postprocessor that logs retrieved nodes to a Provena trail.
|
|
17
|
+
|
|
18
|
+
Usage::
|
|
19
|
+
|
|
20
|
+
from provena import ContextTrail
|
|
21
|
+
from provena.integrations.llamaindex import ProvenaPostprocessor
|
|
22
|
+
|
|
23
|
+
trail = ContextTrail()
|
|
24
|
+
query_engine = index.as_query_engine(
|
|
25
|
+
node_postprocessors=[ProvenaPostprocessor(trail=trail)]
|
|
26
|
+
)
|
|
27
|
+
"""
|
|
28
|
+
|
|
29
|
+
model_config = ConfigDict(arbitrary_types_allowed=True)
|
|
30
|
+
trail: Any
|
|
31
|
+
|
|
32
|
+
def __init__(self, trail: ContextTrail, **kwargs: Any) -> None:
|
|
33
|
+
super().__init__(trail=trail, **kwargs)
|
|
34
|
+
|
|
35
|
+
def _postprocess_nodes(
|
|
36
|
+
self,
|
|
37
|
+
nodes: list[NodeWithScore],
|
|
38
|
+
query_bundle: QueryBundle | None = None,
|
|
39
|
+
) -> list[NodeWithScore]:
|
|
40
|
+
for node_with_score in nodes:
|
|
41
|
+
node = node_with_score.node
|
|
42
|
+
content = getattr(node, "text", None)
|
|
43
|
+
if content is None:
|
|
44
|
+
content = str(node)
|
|
45
|
+
provenance = _extract_llamaindex_provenance(node)
|
|
46
|
+
metadata: dict[str, Any] = {}
|
|
47
|
+
if node_with_score.score is not None:
|
|
48
|
+
metadata["score"] = node_with_score.score
|
|
49
|
+
if query_bundle:
|
|
50
|
+
metadata["query"] = query_bundle.query_str
|
|
51
|
+
self.trail.log(
|
|
52
|
+
content=content,
|
|
53
|
+
source=ContextSource.RETRIEVER,
|
|
54
|
+
source_name="llamaindex",
|
|
55
|
+
provenance=provenance,
|
|
56
|
+
metadata=metadata,
|
|
57
|
+
)
|
|
58
|
+
return nodes
|
|
59
|
+
|
|
60
|
+
except ImportError:
|
|
61
|
+
|
|
62
|
+
class ProvenaPostprocessor: # type: ignore[no-redef]
|
|
63
|
+
"""Placeholder when llama-index-core is not installed."""
|
|
64
|
+
|
|
65
|
+
def __init__(self, *args: Any, **kwargs: Any) -> None:
|
|
66
|
+
raise ImportError(
|
|
67
|
+
"llama-index-core is required for LlamaIndex integration. "
|
|
68
|
+
"Install with: pip install provena[llamaindex]"
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _extract_llamaindex_provenance(node: Any) -> ProvenanceMetadata | None:
|
|
73
|
+
meta = getattr(node, "metadata", None)
|
|
74
|
+
if not isinstance(meta, dict):
|
|
75
|
+
return None
|
|
76
|
+
return ProvenanceMetadata(
|
|
77
|
+
source_url=meta.get("source") or meta.get("file_path"),
|
|
78
|
+
author=meta.get("author"),
|
|
79
|
+
)
|