contexttrace 0.8.0__tar.gz → 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {contexttrace-0.8.0 → contexttrace-1.0.0}/PKG-INFO +83 -11
- {contexttrace-0.8.0 → contexttrace-1.0.0}/README.md +255 -192
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/__init__.py +6 -0
- contexttrace-1.0.0/contexttrace/_version.py +1 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/cli.py +862 -682
- contexttrace-1.0.0/contexttrace/diagnose.py +611 -0
- contexttrace-1.0.0/contexttrace/diagnose_report.py +297 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/__init__.py +53 -1
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/abstention.py +81 -70
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/benchmark.py +631 -580
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/citations.py +128 -99
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/claims.py +125 -4
- contexttrace-1.0.0/contexttrace/verify/evidence.py +996 -0
- contexttrace-1.0.0/contexttrace/verify/facts.py +3883 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/judges.py +5 -2
- contexttrace-1.0.0/contexttrace/verify/local_nli.py +418 -0
- contexttrace-1.0.0/contexttrace/verify/nli_calibration.py +468 -0
- contexttrace-1.0.0/contexttrace/verify/public_holdout_cases.json +3538 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/qa_report.py +41 -21
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/report.py +98 -4
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/root_cause.py +284 -218
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/runner.py +413 -259
- contexttrace-1.0.0/contexttrace/verify/source_trust.py +340 -0
- contexttrace-1.0.0/contexttrace/verify/spans.py +246 -0
- contexttrace-1.0.0/contexttrace/verify/statuses.py +122 -0
- contexttrace-1.0.0/contexttrace/verify/verdicts.py +578 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace.egg-info/SOURCES.txt +7 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/pyproject.toml +16 -5
- contexttrace-0.8.0/contexttrace/_version.py +0 -1
- contexttrace-0.8.0/contexttrace/verify/evidence.py +0 -453
- contexttrace-0.8.0/contexttrace/verify/facts.py +0 -606
- contexttrace-0.8.0/contexttrace/verify/spans.py +0 -103
- contexttrace-0.8.0/contexttrace/verify/verdicts.py +0 -304
- {contexttrace-0.8.0 → contexttrace-1.0.0}/MANIFEST.in +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/capture.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/capture_endpoint.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/client.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/config.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/demo.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/demo_data.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/endpoint_eval.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/errors.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/evaluator.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/__init__.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/fastapi.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/langchain.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/langgraph.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/llamaindex.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/integrations/opentelemetry.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/local.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/py.typed +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/regression.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/reliability.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/storage/__init__.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/storage/sqlite_store.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/thresholds.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/transport.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/audit.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/audit_benchmark.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/audit_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/calibration.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/compare.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/compare_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/demos.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/local_ml.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/qa.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/schema.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/suite.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/suite_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/verify/trace_inspect.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/contexttrace/viewer.py +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/setup.cfg +0 -0
- {contexttrace-0.8.0 → contexttrace-1.0.0}/setup.py +0 -0
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: contexttrace
|
|
3
|
-
Version: 0.
|
|
4
|
-
Summary: Local-first
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: Local-first evidence-chain debugger for RAG and AI agent claim grounding, citation checks, root-cause diagnosis, and regression tests.
|
|
5
5
|
Author: ContextTrace contributors
|
|
6
6
|
License: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
|
|
@@ -10,7 +10,7 @@ Project-URL: Repository, https://github.com/samarth1412/Context-Trace
|
|
|
10
10
|
Project-URL: Issues, https://github.com/samarth1412/Context-Trace/issues
|
|
11
11
|
Project-URL: Changelog, https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
|
|
12
12
|
Keywords: rag,llm,retrieval-augmented-generation,citations,evaluation,observability,agents,cli,sqlite
|
|
13
|
-
Classifier: Development Status ::
|
|
13
|
+
Classifier: Development Status :: 5 - Production/Stable
|
|
14
14
|
Classifier: Intended Audience :: Developers
|
|
15
15
|
Classifier: License :: OSI Approved :: MIT License
|
|
16
16
|
Classifier: Programming Language :: Python :: 3
|
|
@@ -34,6 +34,12 @@ Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
|
|
|
34
34
|
Provides-Extra: local
|
|
35
35
|
Provides-Extra: local-ml
|
|
36
36
|
Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
|
|
37
|
+
Provides-Extra: nli
|
|
38
|
+
Requires-Dist: torch>=2.0; extra == "nli"
|
|
39
|
+
Requires-Dist: transformers>=4.41; extra == "nli"
|
|
40
|
+
Provides-Extra: nli-onnx
|
|
41
|
+
Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
|
|
42
|
+
Requires-Dist: transformers>=4.41; extra == "nli-onnx"
|
|
37
43
|
Provides-Extra: fastapi
|
|
38
44
|
Requires-Dist: fastapi>=0.110; extra == "fastapi"
|
|
39
45
|
Provides-Extra: langgraph
|
|
@@ -55,20 +61,29 @@ Requires-Dist: langgraph>=0.2; extra == "all"
|
|
|
55
61
|
Requires-Dist: llama-index-core>=0.10; extra == "all"
|
|
56
62
|
Requires-Dist: opentelemetry-api>=1.24; extra == "all"
|
|
57
63
|
Requires-Dist: sentence-transformers>=2.7; extra == "all"
|
|
64
|
+
Requires-Dist: torch>=2.0; extra == "all"
|
|
65
|
+
Requires-Dist: transformers>=4.41; extra == "all"
|
|
66
|
+
Requires-Dist: onnxruntime>=1.17; extra == "all"
|
|
58
67
|
Provides-Extra: test
|
|
59
68
|
Requires-Dist: pytest>=8.0; extra == "test"
|
|
60
69
|
|
|
61
70
|
# ContextTrace
|
|
62
71
|
|
|
63
|
-
**Local-first evidence-chain
|
|
72
|
+
**Local-first evidence-chain forensics for RAG and AI agents.**
|
|
64
73
|
|
|
65
|
-
ContextTrace
|
|
74
|
+
ContextTrace is a Python SDK and CLI for tracing a failed answer from the user
|
|
75
|
+
query through retrieved context, answer claims, citations, verdicts, root cause,
|
|
76
|
+
repair guidance, and CI regression tests.
|
|
66
77
|
|
|
67
78
|
```text
|
|
68
|
-
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
79
|
+
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause -> regression test
|
|
69
80
|
```
|
|
70
81
|
|
|
71
|
-
|
|
82
|
+
Use it when a RAG or agent score is not enough: ContextTrace points at the
|
|
83
|
+
unsupported or contradicted claim, the evidence span and citation involved, why
|
|
84
|
+
the failure likely happened, and how to keep it from coming back. It is not a
|
|
85
|
+
hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by
|
|
86
|
+
default.
|
|
72
87
|
|
|
73
88
|
## Install
|
|
74
89
|
|
|
@@ -113,10 +128,53 @@ Run local evidence checks:
|
|
|
113
128
|
```bash
|
|
114
129
|
contexttrace inspect trace.json
|
|
115
130
|
contexttrace verify trace.json --report
|
|
131
|
+
contexttrace diagnose trace.json --report
|
|
116
132
|
contexttrace qa trace.json --corpus docs/ --report
|
|
117
133
|
```
|
|
118
134
|
|
|
119
|
-
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then
|
|
135
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
|
|
136
|
+
|
|
137
|
+
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
138
|
+
|
|
139
|
+
## Diagnose An Agent Trace
|
|
140
|
+
|
|
141
|
+
`diagnose` also accepts agent step traces and localizes tool/final-answer
|
|
142
|
+
failures:
|
|
143
|
+
|
|
144
|
+
```json
|
|
145
|
+
{
|
|
146
|
+
"goal": "Book a meeting with Alex",
|
|
147
|
+
"steps": [
|
|
148
|
+
{
|
|
149
|
+
"type": "tool_call",
|
|
150
|
+
"tool": "calendar.search",
|
|
151
|
+
"args": {"date": "Friday"},
|
|
152
|
+
"result": "No availability"
|
|
153
|
+
},
|
|
154
|
+
{
|
|
155
|
+
"type": "final_answer",
|
|
156
|
+
"content": "I booked it for Friday."
|
|
157
|
+
}
|
|
158
|
+
]
|
|
159
|
+
}
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
```bash
|
|
163
|
+
contexttrace diagnose examples/diagnose_agent_trace.json --report --fail-on high_risk
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
The diagnosis flags `tool_result_contradicted_by_final_answer` and suggests
|
|
167
|
+
gating final-answer generation on tool-result status.
|
|
168
|
+
|
|
169
|
+
Turn that diagnosis into a CI regression test:
|
|
170
|
+
|
|
171
|
+
```bash
|
|
172
|
+
contexttrace diagnose examples/diagnose_agent_trace.json \
|
|
173
|
+
--generate-test \
|
|
174
|
+
--test-out tests/contexttrace/test_calendar_agent_diagnosis.py
|
|
175
|
+
|
|
176
|
+
pytest tests/contexttrace/test_calendar_agent_diagnosis.py
|
|
177
|
+
```
|
|
120
178
|
|
|
121
179
|
## Local Verification Modes
|
|
122
180
|
|
|
@@ -125,7 +183,8 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
|
|
|
125
183
|
| `lexical` | Fast default checks with no optional dependencies. |
|
|
126
184
|
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
127
185
|
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
128
|
-
| `
|
|
186
|
+
| `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
|
|
187
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
|
|
129
188
|
|
|
130
189
|
Run the stronger local non-LLM verifier:
|
|
131
190
|
|
|
@@ -138,7 +197,16 @@ Optional neural local-ML support never downloads models automatically:
|
|
|
138
197
|
|
|
139
198
|
```bash
|
|
140
199
|
pip install "contexttrace[local-ml]"
|
|
141
|
-
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C
|
|
200
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
|
|
201
|
+
```
|
|
202
|
+
|
|
203
|
+
Run local NLI when you want mechanical claim-versus-span entailment:
|
|
204
|
+
|
|
205
|
+
```bash
|
|
206
|
+
pip install "contexttrace[nli]"
|
|
207
|
+
set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
|
|
208
|
+
contexttrace verify trace.json --mode nli --report
|
|
209
|
+
contexttrace nli-calibrate --case-set all --report
|
|
142
210
|
```
|
|
143
211
|
|
|
144
212
|
Run a local judge with Ollama:
|
|
@@ -169,6 +237,10 @@ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/
|
|
|
169
237
|
|
|
170
238
|
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
171
239
|
|
|
240
|
+
`support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
|
|
241
|
+
|
|
242
|
+
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
243
|
+
|
|
172
244
|
## Capture Existing Systems
|
|
173
245
|
|
|
174
246
|
Capture one live endpoint response:
|
|
@@ -247,7 +319,7 @@ ContextTrace makes no network calls unless you point it at an endpoint or config
|
|
|
247
319
|
|
|
248
320
|
## Limits
|
|
249
321
|
|
|
250
|
-
ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
322
|
+
ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
251
323
|
|
|
252
324
|
## Links
|
|
253
325
|
|
|
@@ -1,197 +1,260 @@
|
|
|
1
|
-
# ContextTrace
|
|
1
|
+
# ContextTrace
|
|
2
|
+
|
|
3
|
+
**Local-first evidence-chain forensics for RAG and AI agents.**
|
|
2
4
|
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
```text
|
|
8
|
-
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
9
|
-
```
|
|
10
|
-
|
|
11
|
-
It is a Python SDK and CLI, not a hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by default.
|
|
12
|
-
|
|
13
|
-
## Install
|
|
14
|
-
|
|
15
|
-
```bash
|
|
16
|
-
pip install contexttrace
|
|
17
|
-
contexttrace init
|
|
18
|
-
```
|
|
19
|
-
|
|
20
|
-
## Quickstart
|
|
21
|
-
|
|
22
|
-
```bash
|
|
23
|
-
contexttrace verify-demo unsupported_claim --report
|
|
24
|
-
contexttrace demo --dataset refund_policy
|
|
25
|
-
contexttrace report --last --open
|
|
26
|
-
```
|
|
27
|
-
|
|
28
|
-
Default local storage:
|
|
5
|
+
ContextTrace is a Python SDK and CLI for tracing a failed answer from the user
|
|
6
|
+
query through retrieved context, answer claims, citations, verdicts, root cause,
|
|
7
|
+
repair guidance, and CI regression tests.
|
|
29
8
|
|
|
30
9
|
```text
|
|
31
|
-
|
|
32
|
-
```
|
|
33
|
-
|
|
34
|
-
## Verify A RAG Trace
|
|
35
|
-
|
|
36
|
-
Create a portable trace with a query, answer, retrieved contexts, and optional citations:
|
|
37
|
-
|
|
38
|
-
```json
|
|
39
|
-
{
|
|
40
|
-
"query": "How long does refund processing take?",
|
|
41
|
-
"answer": "Refunds are processed within 5 business days.",
|
|
42
|
-
"contexts": [
|
|
43
|
-
{
|
|
44
|
-
"id": "policy",
|
|
45
|
-
"text": "Customers may request refunds within 30 days of purchase."
|
|
46
|
-
}
|
|
47
|
-
]
|
|
48
|
-
}
|
|
49
|
-
```
|
|
50
|
-
|
|
51
|
-
Run local evidence checks:
|
|
52
|
-
|
|
53
|
-
```bash
|
|
54
|
-
contexttrace inspect trace.json
|
|
55
|
-
contexttrace verify trace.json --report
|
|
56
|
-
contexttrace qa trace.json --corpus docs/ --report
|
|
57
|
-
```
|
|
58
|
-
|
|
59
|
-
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then explains the likely fix.
|
|
60
|
-
|
|
61
|
-
## Local Verification Modes
|
|
62
|
-
|
|
63
|
-
| Mode | Use When |
|
|
64
|
-
| --- | --- |
|
|
65
|
-
| `lexical` | Fast default checks with no optional dependencies. |
|
|
66
|
-
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
67
|
-
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
68
|
-
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. |
|
|
69
|
-
|
|
70
|
-
Run the stronger local non-LLM verifier:
|
|
71
|
-
|
|
72
|
-
```bash
|
|
73
|
-
contexttrace verify trace.json --mode local_ml --report
|
|
74
|
-
contexttrace verify-benchmark --mode local_ml --case-set all
|
|
75
|
-
```
|
|
76
|
-
|
|
77
|
-
Optional neural local-ML support never downloads models automatically:
|
|
78
|
-
|
|
79
|
-
```bash
|
|
80
|
-
pip install "contexttrace[local-ml]"
|
|
81
|
-
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:\models\bge-small-en-v1.5
|
|
82
|
-
```
|
|
83
|
-
|
|
84
|
-
Run a local judge with Ollama:
|
|
85
|
-
|
|
86
|
-
```bash
|
|
87
|
-
set CONTEXTTRACE_JUDGE_PROVIDER=ollama
|
|
88
|
-
set CONTEXTTRACE_JUDGE_MODEL=llama3.1
|
|
89
|
-
|
|
90
|
-
contexttrace verify trace.json --mode judge --report
|
|
91
|
-
contexttrace judge-calibrate --case-set all --report
|
|
10
|
+
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause -> regression test
|
|
92
11
|
```
|
|
93
12
|
|
|
94
|
-
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
|
|
116
|
-
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
|
|
132
|
-
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
```
|
|
141
|
-
|
|
142
|
-
|
|
143
|
-
|
|
144
|
-
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
|
|
148
|
-
|
|
149
|
-
|
|
150
|
-
|
|
151
|
-
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
163
|
-
|
|
164
|
-
|
|
165
|
-
|
|
166
|
-
|
|
167
|
-
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
|
|
184
|
-
|
|
185
|
-
-
|
|
186
|
-
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
-
|
|
197
|
-
-
|
|
13
|
+
Use it when a RAG or agent score is not enough: ContextTrace points at the
|
|
14
|
+
unsupported or contradicted claim, the evidence span and citation involved, why
|
|
15
|
+
the failure likely happened, and how to keep it from coming back. It is not a
|
|
16
|
+
hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by
|
|
17
|
+
default.
|
|
18
|
+
|
|
19
|
+
## Install
|
|
20
|
+
|
|
21
|
+
```bash
|
|
22
|
+
pip install contexttrace
|
|
23
|
+
contexttrace init
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
## Quickstart
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
contexttrace verify-demo unsupported_claim --report
|
|
30
|
+
contexttrace demo --dataset refund_policy
|
|
31
|
+
contexttrace report --last --open
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
Default local storage:
|
|
35
|
+
|
|
36
|
+
```text
|
|
37
|
+
.contexttrace/contexttrace.db
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
## Verify A RAG Trace
|
|
41
|
+
|
|
42
|
+
Create a portable trace with a query, answer, retrieved contexts, and optional citations:
|
|
43
|
+
|
|
44
|
+
```json
|
|
45
|
+
{
|
|
46
|
+
"query": "How long does refund processing take?",
|
|
47
|
+
"answer": "Refunds are processed within 5 business days.",
|
|
48
|
+
"contexts": [
|
|
49
|
+
{
|
|
50
|
+
"id": "policy",
|
|
51
|
+
"text": "Customers may request refunds within 30 days of purchase."
|
|
52
|
+
}
|
|
53
|
+
]
|
|
54
|
+
}
|
|
55
|
+
```
|
|
56
|
+
|
|
57
|
+
Run local evidence checks:
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
contexttrace inspect trace.json
|
|
61
|
+
contexttrace verify trace.json --report
|
|
62
|
+
contexttrace diagnose trace.json --report
|
|
63
|
+
contexttrace qa trace.json --corpus docs/ --report
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
|
|
67
|
+
|
|
68
|
+
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
69
|
+
|
|
70
|
+
## Diagnose An Agent Trace
|
|
71
|
+
|
|
72
|
+
`diagnose` also accepts agent step traces and localizes tool/final-answer
|
|
73
|
+
failures:
|
|
74
|
+
|
|
75
|
+
```json
|
|
76
|
+
{
|
|
77
|
+
"goal": "Book a meeting with Alex",
|
|
78
|
+
"steps": [
|
|
79
|
+
{
|
|
80
|
+
"type": "tool_call",
|
|
81
|
+
"tool": "calendar.search",
|
|
82
|
+
"args": {"date": "Friday"},
|
|
83
|
+
"result": "No availability"
|
|
84
|
+
},
|
|
85
|
+
{
|
|
86
|
+
"type": "final_answer",
|
|
87
|
+
"content": "I booked it for Friday."
|
|
88
|
+
}
|
|
89
|
+
]
|
|
90
|
+
}
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
contexttrace diagnose examples/diagnose_agent_trace.json --report --fail-on high_risk
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
The diagnosis flags `tool_result_contradicted_by_final_answer` and suggests
|
|
98
|
+
gating final-answer generation on tool-result status.
|
|
99
|
+
|
|
100
|
+
Turn that diagnosis into a CI regression test:
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
contexttrace diagnose examples/diagnose_agent_trace.json \
|
|
104
|
+
--generate-test \
|
|
105
|
+
--test-out tests/contexttrace/test_calendar_agent_diagnosis.py
|
|
106
|
+
|
|
107
|
+
pytest tests/contexttrace/test_calendar_agent_diagnosis.py
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
## Local Verification Modes
|
|
111
|
+
|
|
112
|
+
| Mode | Use When |
|
|
113
|
+
| --- | --- |
|
|
114
|
+
| `lexical` | Fast default checks with no optional dependencies. |
|
|
115
|
+
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
116
|
+
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
117
|
+
| `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
|
|
118
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
|
|
119
|
+
|
|
120
|
+
Run the stronger local non-LLM verifier:
|
|
121
|
+
|
|
122
|
+
```bash
|
|
123
|
+
contexttrace verify trace.json --mode local_ml --report
|
|
124
|
+
contexttrace verify-benchmark --mode local_ml --case-set all
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Optional neural local-ML support never downloads models automatically:
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
pip install "contexttrace[local-ml]"
|
|
131
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
Run local NLI when you want mechanical claim-versus-span entailment:
|
|
135
|
+
|
|
136
|
+
```bash
|
|
137
|
+
pip install "contexttrace[nli]"
|
|
138
|
+
set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
|
|
139
|
+
contexttrace verify trace.json --mode nli --report
|
|
140
|
+
contexttrace nli-calibrate --case-set all --report
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Run a local judge with Ollama:
|
|
144
|
+
|
|
145
|
+
```bash
|
|
146
|
+
set CONTEXTTRACE_JUDGE_PROVIDER=ollama
|
|
147
|
+
set CONTEXTTRACE_JUDGE_MODEL=llama3.1
|
|
148
|
+
|
|
149
|
+
contexttrace verify trace.json --mode judge --report
|
|
150
|
+
contexttrace judge-calibrate --case-set all --report
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
|
|
154
|
+
|
|
155
|
+
## Diagnose And Regression-Test
|
|
156
|
+
|
|
157
|
+
```bash
|
|
158
|
+
# Find whether support existed elsewhere in the corpus.
|
|
159
|
+
contexttrace audit trace.json --corpus docs/ --report
|
|
160
|
+
|
|
161
|
+
# Compare a baseline and current answer after a prompt, model, or retriever change.
|
|
162
|
+
contexttrace compare baseline.json current.json --report
|
|
163
|
+
|
|
164
|
+
# Turn saved failures into replayable endpoint tests.
|
|
165
|
+
contexttrace suite create traces/failure.json --out contexttrace-suite.json
|
|
166
|
+
contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
170
|
+
|
|
171
|
+
`support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
|
|
172
|
+
|
|
173
|
+
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
174
|
+
|
|
175
|
+
## Capture Existing Systems
|
|
176
|
+
|
|
177
|
+
Capture one live endpoint response:
|
|
178
|
+
|
|
179
|
+
```bash
|
|
180
|
+
contexttrace capture endpoint \
|
|
181
|
+
--endpoint http://localhost:8000/query \
|
|
182
|
+
--query "What is the refund policy?" \
|
|
183
|
+
--answer-path $.answer \
|
|
184
|
+
--contexts-path $.contexts \
|
|
185
|
+
--citations-path $.citations \
|
|
186
|
+
--out traces/refund_trace.json \
|
|
187
|
+
--verify \
|
|
188
|
+
--report
|
|
189
|
+
```
|
|
190
|
+
|
|
191
|
+
Or capture artifacts from Python:
|
|
192
|
+
|
|
193
|
+
```python
|
|
194
|
+
from contexttrace import capture_rag_trace, write_rag_trace
|
|
195
|
+
|
|
196
|
+
trace = capture_rag_trace(
|
|
197
|
+
query=question,
|
|
198
|
+
answer=answer,
|
|
199
|
+
contexts=retrieved_docs,
|
|
200
|
+
metadata={"system": "support-rag"},
|
|
201
|
+
)
|
|
202
|
+
write_rag_trace(trace, "trace.json")
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
## SDK Example
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
from contexttrace import ContextTrace
|
|
209
|
+
|
|
210
|
+
ct = ContextTrace(project="support-rag")
|
|
211
|
+
|
|
212
|
+
with ct.trace(query="What is the refund policy?") as trace:
|
|
213
|
+
chunks = retriever.search("What is the refund policy?")
|
|
214
|
+
trace.log_retrieval(chunks)
|
|
215
|
+
trace.log_context(chunks[:5])
|
|
216
|
+
|
|
217
|
+
answer = llm.generate("What is the refund policy?", chunks[:5])
|
|
218
|
+
trace.log_answer(answer, usage={"total_tokens": 1200})
|
|
219
|
+
trace.log_citations([
|
|
220
|
+
{"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
|
|
221
|
+
])
|
|
222
|
+
|
|
223
|
+
result = trace.evaluate()
|
|
224
|
+
print(result["failure"]["failure_type"])
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
## Integrations
|
|
228
|
+
|
|
229
|
+
```bash
|
|
230
|
+
pip install "contexttrace[langchain]"
|
|
231
|
+
pip install "contexttrace[llamaindex]"
|
|
232
|
+
pip install "contexttrace[fastapi]"
|
|
233
|
+
pip install "contexttrace[langgraph]"
|
|
234
|
+
pip install "contexttrace[otel]"
|
|
235
|
+
pip install "contexttrace[all]"
|
|
236
|
+
```
|
|
237
|
+
|
|
238
|
+
Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
|
|
239
|
+
|
|
240
|
+
## Privacy
|
|
241
|
+
|
|
242
|
+
ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
|
|
243
|
+
|
|
244
|
+
- `local_only: true`
|
|
245
|
+
- `log_chunk_text: false`
|
|
246
|
+
- `log_answer_text: false`
|
|
247
|
+
- `storage_path`
|
|
248
|
+
- `judge_cache_enabled: true`
|
|
249
|
+
- `judge_cache_path: .contexttrace/judge_cache.json`
|
|
250
|
+
|
|
251
|
+
## Limits
|
|
252
|
+
|
|
253
|
+
ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
254
|
+
|
|
255
|
+
## Links
|
|
256
|
+
|
|
257
|
+
- Repository: https://github.com/samarth1412/Context-Trace
|
|
258
|
+
- Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
|
|
259
|
+
- Issues: https://github.com/samarth1412/Context-Trace/issues
|
|
260
|
+
- Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
|
|
@@ -3,6 +3,8 @@ from contexttrace.capture import capture_rag_trace, langchain_documents_to_conte
|
|
|
3
3
|
from contexttrace.capture_endpoint import EndpointCapture, capture_endpoint_trace, capture_response_trace
|
|
4
4
|
from contexttrace.client import AsyncContextTrace, ContextTrace
|
|
5
5
|
from contexttrace.config import ContextTraceConfig
|
|
6
|
+
from contexttrace.diagnose import diagnose_payload, diagnose_trace_file, write_diagnosis_regression_test
|
|
7
|
+
from contexttrace.diagnose_report import DiagnoseReportGenerator
|
|
6
8
|
from contexttrace.errors import (
|
|
7
9
|
ContextTraceConfigError,
|
|
8
10
|
ContextTraceError,
|
|
@@ -29,6 +31,7 @@ __all__ = [
|
|
|
29
31
|
"ContextTraceLocalError",
|
|
30
32
|
"ContextTraceLangGraphTracer",
|
|
31
33
|
"ContextTraceLlamaIndexCallbackHandler",
|
|
34
|
+
"DiagnoseReportGenerator",
|
|
32
35
|
"EndpointCapture",
|
|
33
36
|
"OpenTelemetryExporter",
|
|
34
37
|
"ReliabilityScore",
|
|
@@ -37,6 +40,9 @@ __all__ = [
|
|
|
37
40
|
"capture_rag_trace",
|
|
38
41
|
"capture_endpoint_trace",
|
|
39
42
|
"capture_response_trace",
|
|
43
|
+
"diagnose_payload",
|
|
44
|
+
"diagnose_trace_file",
|
|
45
|
+
"write_diagnosis_regression_test",
|
|
40
46
|
"export_contexttrace_trace",
|
|
41
47
|
"langchain_documents_to_contexts",
|
|
42
48
|
"write_rag_trace",
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.0.0"
|