contexttrace 0.6.0__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contexttrace-0.8.0/PKG-INFO +257 -0
- contexttrace-0.8.0/README.md +197 -0
- contexttrace-0.8.0/contexttrace/_version.py +1 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/cli.py +682 -126
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/config.py +52 -0
- contexttrace-0.8.0/contexttrace/integrations/opentelemetry.py +200 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/__init__.py +54 -23
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/audit_benchmark_cases.json +2 -2
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/benchmark.py +8 -2
- contexttrace-0.8.0/contexttrace/verify/calibration.py +284 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/evidence.py +28 -10
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/facts.py +157 -4
- contexttrace-0.8.0/contexttrace/verify/judges.py +573 -0
- contexttrace-0.8.0/contexttrace/verify/local_ml.py +110 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/real_benchmark_cases.json +102 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/runner.py +118 -10
- contexttrace-0.8.0/contexttrace/verify/suite.py +662 -0
- contexttrace-0.8.0/contexttrace/verify/suite_report.py +316 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/verdicts.py +49 -2
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace.egg-info/SOURCES.txt +5 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/pyproject.toml +14 -10
- contexttrace-0.6.0/PKG-INFO +0 -231
- contexttrace-0.6.0/README.md +0 -174
- contexttrace-0.6.0/contexttrace/_version.py +0 -1
- contexttrace-0.6.0/contexttrace/integrations/opentelemetry.py +0 -111
- {contexttrace-0.6.0 → contexttrace-0.8.0}/MANIFEST.in +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/__init__.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/capture.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/capture_endpoint.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/client.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/demo.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/demo_data.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/endpoint_eval.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/errors.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/evaluator.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/integrations/__init__.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/integrations/fastapi.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/integrations/langchain.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/integrations/langgraph.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/integrations/llamaindex.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/local.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/py.typed +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/regression.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/reliability.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/report.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/storage/__init__.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/storage/sqlite_store.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/thresholds.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/transport.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/abstention.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/audit.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/audit_benchmark.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/audit_report.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/citations.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/claims.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/compare.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/compare_report.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/demos.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/qa.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/qa_report.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/report.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/root_cause.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/schema.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/spans.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/verify/trace_inspect.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/contexttrace/viewer.py +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/setup.cfg +0 -0
- {contexttrace-0.6.0 → contexttrace-0.8.0}/setup.py +0 -0
|
@@ -0,0 +1,257 @@
|
|
|
1
|
+
Metadata-Version: 2.1
|
|
2
|
+
Name: contexttrace
|
|
3
|
+
Version: 0.8.0
|
|
4
|
+
Summary: Local-first SDK and CLI for RAG and agent reliability tracing, citation checks, and failure diagnosis.
|
|
5
|
+
Author: ContextTrace contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
|
|
8
|
+
Project-URL: Documentation, https://github.com/samarth1412/Context-Trace/tree/main/docs
|
|
9
|
+
Project-URL: Repository, https://github.com/samarth1412/Context-Trace
|
|
10
|
+
Project-URL: Issues, https://github.com/samarth1412/Context-Trace/issues
|
|
11
|
+
Project-URL: Changelog, https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
|
|
12
|
+
Keywords: rag,llm,retrieval-augmented-generation,citations,evaluation,observability,agents,cli,sqlite
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.8
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
23
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: >=3.8
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
Requires-Dist: click>=8.1
|
|
28
|
+
Requires-Dist: httpx>=0.27
|
|
29
|
+
Requires-Dist: typing-extensions>=4.9
|
|
30
|
+
Provides-Extra: langchain
|
|
31
|
+
Requires-Dist: langchain-core>=0.2; extra == "langchain"
|
|
32
|
+
Provides-Extra: llamaindex
|
|
33
|
+
Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
|
|
34
|
+
Provides-Extra: local
|
|
35
|
+
Provides-Extra: local-ml
|
|
36
|
+
Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
|
|
37
|
+
Provides-Extra: fastapi
|
|
38
|
+
Requires-Dist: fastapi>=0.110; extra == "fastapi"
|
|
39
|
+
Provides-Extra: langgraph
|
|
40
|
+
Requires-Dist: langgraph>=0.2; extra == "langgraph"
|
|
41
|
+
Provides-Extra: otel
|
|
42
|
+
Requires-Dist: opentelemetry-api>=1.24; extra == "otel"
|
|
43
|
+
Provides-Extra: opentelemetry
|
|
44
|
+
Requires-Dist: opentelemetry-api>=1.24; extra == "opentelemetry"
|
|
45
|
+
Provides-Extra: integrations
|
|
46
|
+
Requires-Dist: fastapi>=0.110; extra == "integrations"
|
|
47
|
+
Requires-Dist: langchain-core>=0.2; extra == "integrations"
|
|
48
|
+
Requires-Dist: langgraph>=0.2; extra == "integrations"
|
|
49
|
+
Requires-Dist: llama-index-core>=0.10; extra == "integrations"
|
|
50
|
+
Requires-Dist: opentelemetry-api>=1.24; extra == "integrations"
|
|
51
|
+
Provides-Extra: all
|
|
52
|
+
Requires-Dist: fastapi>=0.110; extra == "all"
|
|
53
|
+
Requires-Dist: langchain-core>=0.2; extra == "all"
|
|
54
|
+
Requires-Dist: langgraph>=0.2; extra == "all"
|
|
55
|
+
Requires-Dist: llama-index-core>=0.10; extra == "all"
|
|
56
|
+
Requires-Dist: opentelemetry-api>=1.24; extra == "all"
|
|
57
|
+
Requires-Dist: sentence-transformers>=2.7; extra == "all"
|
|
58
|
+
Provides-Extra: test
|
|
59
|
+
Requires-Dist: pytest>=8.0; extra == "test"
|
|
60
|
+
|
|
61
|
+
# ContextTrace
|
|
62
|
+
|
|
63
|
+
**Local-first evidence-chain debugging for RAG and AI agents.**
|
|
64
|
+
|
|
65
|
+
ContextTrace shows where an answer stopped being grounded:
|
|
66
|
+
|
|
67
|
+
```text
|
|
68
|
+
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
It is a Python SDK and CLI, not a hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by default.
|
|
72
|
+
|
|
73
|
+
## Install
|
|
74
|
+
|
|
75
|
+
```bash
|
|
76
|
+
pip install contexttrace
|
|
77
|
+
contexttrace init
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
## Quickstart
|
|
81
|
+
|
|
82
|
+
```bash
|
|
83
|
+
contexttrace verify-demo unsupported_claim --report
|
|
84
|
+
contexttrace demo --dataset refund_policy
|
|
85
|
+
contexttrace report --last --open
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
Default local storage:
|
|
89
|
+
|
|
90
|
+
```text
|
|
91
|
+
.contexttrace/contexttrace.db
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## Verify A RAG Trace
|
|
95
|
+
|
|
96
|
+
Create a portable trace with a query, answer, retrieved contexts, and optional citations:
|
|
97
|
+
|
|
98
|
+
```json
|
|
99
|
+
{
|
|
100
|
+
"query": "How long does refund processing take?",
|
|
101
|
+
"answer": "Refunds are processed within 5 business days.",
|
|
102
|
+
"contexts": [
|
|
103
|
+
{
|
|
104
|
+
"id": "policy",
|
|
105
|
+
"text": "Customers may request refunds within 30 days of purchase."
|
|
106
|
+
}
|
|
107
|
+
]
|
|
108
|
+
}
|
|
109
|
+
```
|
|
110
|
+
|
|
111
|
+
Run local evidence checks:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
contexttrace inspect trace.json
|
|
115
|
+
contexttrace verify trace.json --report
|
|
116
|
+
contexttrace qa trace.json --corpus docs/ --report
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then explains the likely fix.
|
|
120
|
+
|
|
121
|
+
## Local Verification Modes
|
|
122
|
+
|
|
123
|
+
| Mode | Use When |
|
|
124
|
+
| --- | --- |
|
|
125
|
+
| `lexical` | Fast default checks with no optional dependencies. |
|
|
126
|
+
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
127
|
+
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
128
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. |
|
|
129
|
+
|
|
130
|
+
Run the stronger local non-LLM verifier:
|
|
131
|
+
|
|
132
|
+
```bash
|
|
133
|
+
contexttrace verify trace.json --mode local_ml --report
|
|
134
|
+
contexttrace verify-benchmark --mode local_ml --case-set all
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
Optional neural local-ML support never downloads models automatically:
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
pip install "contexttrace[local-ml]"
|
|
141
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:\models\bge-small-en-v1.5
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
Run a local judge with Ollama:
|
|
145
|
+
|
|
146
|
+
```bash
|
|
147
|
+
set CONTEXTTRACE_JUDGE_PROVIDER=ollama
|
|
148
|
+
set CONTEXTTRACE_JUDGE_MODEL=llama3.1
|
|
149
|
+
|
|
150
|
+
contexttrace verify trace.json --mode judge --report
|
|
151
|
+
contexttrace judge-calibrate --case-set all --report
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
|
|
155
|
+
|
|
156
|
+
## Diagnose And Regression-Test
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
# Find whether support existed elsewhere in the corpus.
|
|
160
|
+
contexttrace audit trace.json --corpus docs/ --report
|
|
161
|
+
|
|
162
|
+
# Compare a baseline and current answer after a prompt, model, or retriever change.
|
|
163
|
+
contexttrace compare baseline.json current.json --report
|
|
164
|
+
|
|
165
|
+
# Turn saved failures into replayable endpoint tests.
|
|
166
|
+
contexttrace suite create traces/failure.json --out contexttrace-suite.json
|
|
167
|
+
contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
171
|
+
|
|
172
|
+
## Capture Existing Systems
|
|
173
|
+
|
|
174
|
+
Capture one live endpoint response:
|
|
175
|
+
|
|
176
|
+
```bash
|
|
177
|
+
contexttrace capture endpoint \
|
|
178
|
+
--endpoint http://localhost:8000/query \
|
|
179
|
+
--query "What is the refund policy?" \
|
|
180
|
+
--answer-path $.answer \
|
|
181
|
+
--contexts-path $.contexts \
|
|
182
|
+
--citations-path $.citations \
|
|
183
|
+
--out traces/refund_trace.json \
|
|
184
|
+
--verify \
|
|
185
|
+
--report
|
|
186
|
+
```
|
|
187
|
+
|
|
188
|
+
Or capture artifacts from Python:
|
|
189
|
+
|
|
190
|
+
```python
|
|
191
|
+
from contexttrace import capture_rag_trace, write_rag_trace
|
|
192
|
+
|
|
193
|
+
trace = capture_rag_trace(
|
|
194
|
+
query=question,
|
|
195
|
+
answer=answer,
|
|
196
|
+
contexts=retrieved_docs,
|
|
197
|
+
metadata={"system": "support-rag"},
|
|
198
|
+
)
|
|
199
|
+
write_rag_trace(trace, "trace.json")
|
|
200
|
+
```
|
|
201
|
+
|
|
202
|
+
## SDK Example
|
|
203
|
+
|
|
204
|
+
```python
|
|
205
|
+
from contexttrace import ContextTrace
|
|
206
|
+
|
|
207
|
+
ct = ContextTrace(project="support-rag")
|
|
208
|
+
|
|
209
|
+
with ct.trace(query="What is the refund policy?") as trace:
|
|
210
|
+
chunks = retriever.search("What is the refund policy?")
|
|
211
|
+
trace.log_retrieval(chunks)
|
|
212
|
+
trace.log_context(chunks[:5])
|
|
213
|
+
|
|
214
|
+
answer = llm.generate("What is the refund policy?", chunks[:5])
|
|
215
|
+
trace.log_answer(answer, usage={"total_tokens": 1200})
|
|
216
|
+
trace.log_citations([
|
|
217
|
+
{"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
|
|
218
|
+
])
|
|
219
|
+
|
|
220
|
+
result = trace.evaluate()
|
|
221
|
+
print(result["failure"]["failure_type"])
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
## Integrations
|
|
225
|
+
|
|
226
|
+
```bash
|
|
227
|
+
pip install "contexttrace[langchain]"
|
|
228
|
+
pip install "contexttrace[llamaindex]"
|
|
229
|
+
pip install "contexttrace[fastapi]"
|
|
230
|
+
pip install "contexttrace[langgraph]"
|
|
231
|
+
pip install "contexttrace[otel]"
|
|
232
|
+
pip install "contexttrace[all]"
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
|
|
236
|
+
|
|
237
|
+
## Privacy
|
|
238
|
+
|
|
239
|
+
ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
|
|
240
|
+
|
|
241
|
+
- `local_only: true`
|
|
242
|
+
- `log_chunk_text: false`
|
|
243
|
+
- `log_answer_text: false`
|
|
244
|
+
- `storage_path`
|
|
245
|
+
- `judge_cache_enabled: true`
|
|
246
|
+
- `judge_cache_path: .contexttrace/judge_cache.json`
|
|
247
|
+
|
|
248
|
+
## Limits
|
|
249
|
+
|
|
250
|
+
ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
251
|
+
|
|
252
|
+
## Links
|
|
253
|
+
|
|
254
|
+
- Repository: https://github.com/samarth1412/Context-Trace
|
|
255
|
+
- Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
|
|
256
|
+
- Issues: https://github.com/samarth1412/Context-Trace/issues
|
|
257
|
+
- Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
|
|
@@ -0,0 +1,197 @@
|
|
|
1
|
+
# ContextTrace
|
|
2
|
+
|
|
3
|
+
**Local-first evidence-chain debugging for RAG and AI agents.**
|
|
4
|
+
|
|
5
|
+
ContextTrace shows where an answer stopped being grounded:
|
|
6
|
+
|
|
7
|
+
```text
|
|
8
|
+
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
It is a Python SDK and CLI, not a hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by default.
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
```bash
|
|
16
|
+
pip install contexttrace
|
|
17
|
+
contexttrace init
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
## Quickstart
|
|
21
|
+
|
|
22
|
+
```bash
|
|
23
|
+
contexttrace verify-demo unsupported_claim --report
|
|
24
|
+
contexttrace demo --dataset refund_policy
|
|
25
|
+
contexttrace report --last --open
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
Default local storage:
|
|
29
|
+
|
|
30
|
+
```text
|
|
31
|
+
.contexttrace/contexttrace.db
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
## Verify A RAG Trace
|
|
35
|
+
|
|
36
|
+
Create a portable trace with a query, answer, retrieved contexts, and optional citations:
|
|
37
|
+
|
|
38
|
+
```json
|
|
39
|
+
{
|
|
40
|
+
"query": "How long does refund processing take?",
|
|
41
|
+
"answer": "Refunds are processed within 5 business days.",
|
|
42
|
+
"contexts": [
|
|
43
|
+
{
|
|
44
|
+
"id": "policy",
|
|
45
|
+
"text": "Customers may request refunds within 30 days of purchase."
|
|
46
|
+
}
|
|
47
|
+
]
|
|
48
|
+
}
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Run local evidence checks:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
contexttrace inspect trace.json
|
|
55
|
+
contexttrace verify trace.json --report
|
|
56
|
+
contexttrace qa trace.json --corpus docs/ --report
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then explains the likely fix.
|
|
60
|
+
|
|
61
|
+
## Local Verification Modes
|
|
62
|
+
|
|
63
|
+
| Mode | Use When |
|
|
64
|
+
| --- | --- |
|
|
65
|
+
| `lexical` | Fast default checks with no optional dependencies. |
|
|
66
|
+
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
67
|
+
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
68
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. |
|
|
69
|
+
|
|
70
|
+
Run the stronger local non-LLM verifier:
|
|
71
|
+
|
|
72
|
+
```bash
|
|
73
|
+
contexttrace verify trace.json --mode local_ml --report
|
|
74
|
+
contexttrace verify-benchmark --mode local_ml --case-set all
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
Optional neural local-ML support never downloads models automatically:
|
|
78
|
+
|
|
79
|
+
```bash
|
|
80
|
+
pip install "contexttrace[local-ml]"
|
|
81
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:\models\bge-small-en-v1.5
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
Run a local judge with Ollama:
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
set CONTEXTTRACE_JUDGE_PROVIDER=ollama
|
|
88
|
+
set CONTEXTTRACE_JUDGE_MODEL=llama3.1
|
|
89
|
+
|
|
90
|
+
contexttrace verify trace.json --mode judge --report
|
|
91
|
+
contexttrace judge-calibrate --case-set all --report
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
|
|
95
|
+
|
|
96
|
+
## Diagnose And Regression-Test
|
|
97
|
+
|
|
98
|
+
```bash
|
|
99
|
+
# Find whether support existed elsewhere in the corpus.
|
|
100
|
+
contexttrace audit trace.json --corpus docs/ --report
|
|
101
|
+
|
|
102
|
+
# Compare a baseline and current answer after a prompt, model, or retriever change.
|
|
103
|
+
contexttrace compare baseline.json current.json --report
|
|
104
|
+
|
|
105
|
+
# Turn saved failures into replayable endpoint tests.
|
|
106
|
+
contexttrace suite create traces/failure.json --out contexttrace-suite.json
|
|
107
|
+
contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
111
|
+
|
|
112
|
+
## Capture Existing Systems
|
|
113
|
+
|
|
114
|
+
Capture one live endpoint response:
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
contexttrace capture endpoint \
|
|
118
|
+
--endpoint http://localhost:8000/query \
|
|
119
|
+
--query "What is the refund policy?" \
|
|
120
|
+
--answer-path $.answer \
|
|
121
|
+
--contexts-path $.contexts \
|
|
122
|
+
--citations-path $.citations \
|
|
123
|
+
--out traces/refund_trace.json \
|
|
124
|
+
--verify \
|
|
125
|
+
--report
|
|
126
|
+
```
|
|
127
|
+
|
|
128
|
+
Or capture artifacts from Python:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
from contexttrace import capture_rag_trace, write_rag_trace
|
|
132
|
+
|
|
133
|
+
trace = capture_rag_trace(
|
|
134
|
+
query=question,
|
|
135
|
+
answer=answer,
|
|
136
|
+
contexts=retrieved_docs,
|
|
137
|
+
metadata={"system": "support-rag"},
|
|
138
|
+
)
|
|
139
|
+
write_rag_trace(trace, "trace.json")
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
## SDK Example
|
|
143
|
+
|
|
144
|
+
```python
|
|
145
|
+
from contexttrace import ContextTrace
|
|
146
|
+
|
|
147
|
+
ct = ContextTrace(project="support-rag")
|
|
148
|
+
|
|
149
|
+
with ct.trace(query="What is the refund policy?") as trace:
|
|
150
|
+
chunks = retriever.search("What is the refund policy?")
|
|
151
|
+
trace.log_retrieval(chunks)
|
|
152
|
+
trace.log_context(chunks[:5])
|
|
153
|
+
|
|
154
|
+
answer = llm.generate("What is the refund policy?", chunks[:5])
|
|
155
|
+
trace.log_answer(answer, usage={"total_tokens": 1200})
|
|
156
|
+
trace.log_citations([
|
|
157
|
+
{"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
|
|
158
|
+
])
|
|
159
|
+
|
|
160
|
+
result = trace.evaluate()
|
|
161
|
+
print(result["failure"]["failure_type"])
|
|
162
|
+
```
|
|
163
|
+
|
|
164
|
+
## Integrations
|
|
165
|
+
|
|
166
|
+
```bash
|
|
167
|
+
pip install "contexttrace[langchain]"
|
|
168
|
+
pip install "contexttrace[llamaindex]"
|
|
169
|
+
pip install "contexttrace[fastapi]"
|
|
170
|
+
pip install "contexttrace[langgraph]"
|
|
171
|
+
pip install "contexttrace[otel]"
|
|
172
|
+
pip install "contexttrace[all]"
|
|
173
|
+
```
|
|
174
|
+
|
|
175
|
+
Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
|
|
176
|
+
|
|
177
|
+
## Privacy
|
|
178
|
+
|
|
179
|
+
ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
|
|
180
|
+
|
|
181
|
+
- `local_only: true`
|
|
182
|
+
- `log_chunk_text: false`
|
|
183
|
+
- `log_answer_text: false`
|
|
184
|
+
- `storage_path`
|
|
185
|
+
- `judge_cache_enabled: true`
|
|
186
|
+
- `judge_cache_path: .contexttrace/judge_cache.json`
|
|
187
|
+
|
|
188
|
+
## Limits
|
|
189
|
+
|
|
190
|
+
ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
191
|
+
|
|
192
|
+
## Links
|
|
193
|
+
|
|
194
|
+
- Repository: https://github.com/samarth1412/Context-Trace
|
|
195
|
+
- Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
|
|
196
|
+
- Issues: https://github.com/samarth1412/Context-Trace/issues
|
|
197
|
+
- Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.8.0"
|