contexttrace 0.7.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (75) hide show
  1. contexttrace-0.9.0/PKG-INFO +282 -0
  2. contexttrace-0.9.0/README.md +213 -0
  3. contexttrace-0.9.0/contexttrace/_version.py +1 -0
  4. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/cli.py +422 -108
  5. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/config.py +52 -0
  6. contexttrace-0.9.0/contexttrace/integrations/opentelemetry.py +200 -0
  7. contexttrace-0.9.0/contexttrace/verify/__init__.py +121 -0
  8. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark_cases.json +2 -2
  9. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/benchmark.py +14 -2
  10. contexttrace-0.9.0/contexttrace/verify/calibration.py +284 -0
  11. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/citations.py +2 -1
  12. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/evidence.py +33 -14
  13. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/facts.py +157 -4
  14. contexttrace-0.9.0/contexttrace/verify/judges.py +576 -0
  15. contexttrace-0.9.0/contexttrace/verify/local_ml.py +110 -0
  16. contexttrace-0.9.0/contexttrace/verify/local_nli.py +418 -0
  17. contexttrace-0.9.0/contexttrace/verify/nli_calibration.py +468 -0
  18. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/qa_report.py +41 -21
  19. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/real_benchmark_cases.json +102 -0
  20. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/report.py +98 -4
  21. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/root_cause.py +52 -0
  22. contexttrace-0.9.0/contexttrace/verify/runner.py +412 -0
  23. contexttrace-0.9.0/contexttrace/verify/source_trust.py +340 -0
  24. contexttrace-0.9.0/contexttrace/verify/statuses.py +122 -0
  25. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/verdicts.py +49 -2
  26. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace.egg-info/SOURCES.txt +7 -0
  27. {contexttrace-0.7.0 → contexttrace-0.9.0}/pyproject.toml +27 -12
  28. contexttrace-0.7.0/PKG-INFO +0 -239
  29. contexttrace-0.7.0/README.md +0 -182
  30. contexttrace-0.7.0/contexttrace/_version.py +0 -1
  31. contexttrace-0.7.0/contexttrace/integrations/opentelemetry.py +0 -111
  32. contexttrace-0.7.0/contexttrace/verify/__init__.py +0 -38
  33. contexttrace-0.7.0/contexttrace/verify/runner.py +0 -151
  34. {contexttrace-0.7.0 → contexttrace-0.9.0}/MANIFEST.in +0 -0
  35. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/__init__.py +0 -0
  36. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/capture.py +0 -0
  37. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/capture_endpoint.py +0 -0
  38. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/client.py +0 -0
  39. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/demo.py +0 -0
  40. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/demo_data.py +0 -0
  41. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/endpoint_eval.py +0 -0
  42. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/errors.py +0 -0
  43. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/evaluator.py +0 -0
  44. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/integrations/__init__.py +0 -0
  45. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/integrations/fastapi.py +0 -0
  46. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/integrations/langchain.py +0 -0
  47. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/integrations/langgraph.py +0 -0
  48. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/integrations/llamaindex.py +0 -0
  49. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/local.py +0 -0
  50. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/py.typed +0 -0
  51. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/regression.py +0 -0
  52. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/reliability.py +0 -0
  53. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/report.py +0 -0
  54. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/storage/__init__.py +0 -0
  55. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/storage/sqlite_store.py +0 -0
  56. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/thresholds.py +0 -0
  57. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/transport.py +0 -0
  58. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/abstention.py +0 -0
  59. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/audit.py +0 -0
  60. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark.py +0 -0
  61. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/audit_report.py +0 -0
  62. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/claims.py +0 -0
  63. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/compare.py +0 -0
  64. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/compare_report.py +0 -0
  65. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/demos.py +0 -0
  66. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
  67. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/qa.py +0 -0
  68. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/schema.py +0 -0
  69. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/spans.py +0 -0
  70. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/suite.py +0 -0
  71. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/suite_report.py +0 -0
  72. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/verify/trace_inspect.py +0 -0
  73. {contexttrace-0.7.0 → contexttrace-0.9.0}/contexttrace/viewer.py +0 -0
  74. {contexttrace-0.7.0 → contexttrace-0.9.0}/setup.cfg +0 -0
  75. {contexttrace-0.7.0 → contexttrace-0.9.0}/setup.py +0 -0
@@ -0,0 +1,282 @@
1
+ Metadata-Version: 2.1
2
+ Name: contexttrace
3
+ Version: 0.9.0
4
+ Summary: Local-first SDK and CLI for RAG and agent reliability tracing, citation checks, and failure diagnosis.
5
+ Author: ContextTrace contributors
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
8
+ Project-URL: Documentation, https://github.com/samarth1412/Context-Trace/tree/main/docs
9
+ Project-URL: Repository, https://github.com/samarth1412/Context-Trace
10
+ Project-URL: Issues, https://github.com/samarth1412/Context-Trace/issues
11
+ Project-URL: Changelog, https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
12
+ Keywords: rag,llm,retrieval-augmented-generation,citations,evaluation,observability,agents,cli,sqlite
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Programming Language :: Python :: 3.8
18
+ Classifier: Programming Language :: Python :: 3.9
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Programming Language :: Python :: 3.13
23
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
24
+ Classifier: Typing :: Typed
25
+ Requires-Python: >=3.8
26
+ Description-Content-Type: text/markdown
27
+ Requires-Dist: click>=8.1
28
+ Requires-Dist: httpx>=0.27
29
+ Requires-Dist: typing-extensions>=4.9
30
+ Provides-Extra: langchain
31
+ Requires-Dist: langchain-core>=0.2; extra == "langchain"
32
+ Provides-Extra: llamaindex
33
+ Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
34
+ Provides-Extra: local
35
+ Provides-Extra: local-ml
36
+ Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
37
+ Provides-Extra: nli
38
+ Requires-Dist: torch>=2.0; extra == "nli"
39
+ Requires-Dist: transformers>=4.41; extra == "nli"
40
+ Provides-Extra: nli-onnx
41
+ Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
42
+ Requires-Dist: transformers>=4.41; extra == "nli-onnx"
43
+ Provides-Extra: fastapi
44
+ Requires-Dist: fastapi>=0.110; extra == "fastapi"
45
+ Provides-Extra: langgraph
46
+ Requires-Dist: langgraph>=0.2; extra == "langgraph"
47
+ Provides-Extra: otel
48
+ Requires-Dist: opentelemetry-api>=1.24; extra == "otel"
49
+ Provides-Extra: opentelemetry
50
+ Requires-Dist: opentelemetry-api>=1.24; extra == "opentelemetry"
51
+ Provides-Extra: integrations
52
+ Requires-Dist: fastapi>=0.110; extra == "integrations"
53
+ Requires-Dist: langchain-core>=0.2; extra == "integrations"
54
+ Requires-Dist: langgraph>=0.2; extra == "integrations"
55
+ Requires-Dist: llama-index-core>=0.10; extra == "integrations"
56
+ Requires-Dist: opentelemetry-api>=1.24; extra == "integrations"
57
+ Provides-Extra: all
58
+ Requires-Dist: fastapi>=0.110; extra == "all"
59
+ Requires-Dist: langchain-core>=0.2; extra == "all"
60
+ Requires-Dist: langgraph>=0.2; extra == "all"
61
+ Requires-Dist: llama-index-core>=0.10; extra == "all"
62
+ Requires-Dist: opentelemetry-api>=1.24; extra == "all"
63
+ Requires-Dist: sentence-transformers>=2.7; extra == "all"
64
+ Requires-Dist: torch>=2.0; extra == "all"
65
+ Requires-Dist: transformers>=4.41; extra == "all"
66
+ Requires-Dist: onnxruntime>=1.17; extra == "all"
67
+ Provides-Extra: test
68
+ Requires-Dist: pytest>=8.0; extra == "test"
69
+
70
+ # ContextTrace
71
+
72
+ **Local-first evidence-chain debugging for RAG and AI agents.**
73
+
74
+ ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
75
+
76
+ ```text
77
+ query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
78
+ ```
79
+
80
+ It is a Python SDK and CLI, not a hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by default.
81
+
82
+ ## Install
83
+
84
+ ```bash
85
+ pip install contexttrace
86
+ contexttrace init
87
+ ```
88
+
89
+ ## Quickstart
90
+
91
+ ```bash
92
+ contexttrace verify-demo unsupported_claim --report
93
+ contexttrace demo --dataset refund_policy
94
+ contexttrace report --last --open
95
+ ```
96
+
97
+ Default local storage:
98
+
99
+ ```text
100
+ .contexttrace/contexttrace.db
101
+ ```
102
+
103
+ ## Verify A RAG Trace
104
+
105
+ Create a portable trace with a query, answer, retrieved contexts, and optional citations:
106
+
107
+ ```json
108
+ {
109
+ "query": "How long does refund processing take?",
110
+ "answer": "Refunds are processed within 5 business days.",
111
+ "contexts": [
112
+ {
113
+ "id": "policy",
114
+ "text": "Customers may request refunds within 30 days of purchase."
115
+ }
116
+ ]
117
+ }
118
+ ```
119
+
120
+ Run local evidence checks:
121
+
122
+ ```bash
123
+ contexttrace inspect trace.json
124
+ contexttrace verify trace.json --report
125
+ contexttrace qa trace.json --corpus docs/ --report
126
+ ```
127
+
128
+ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
129
+
130
+ Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
131
+
132
+ ## Local Verification Modes
133
+
134
+ | Mode | Use When |
135
+ | --- | --- |
136
+ | `lexical` | Fast default checks with no optional dependencies. |
137
+ | `semantic` | Local paraphrase and role-aware contradiction checks. |
138
+ | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
139
+ | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
140
+ | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
141
+
142
+ Run the stronger local non-LLM verifier:
143
+
144
+ ```bash
145
+ contexttrace verify trace.json --mode local_ml --report
146
+ contexttrace verify-benchmark --mode local_ml --case-set all
147
+ ```
148
+
149
+ Optional neural local-ML support never downloads models automatically:
150
+
151
+ ```bash
152
+ pip install "contexttrace[local-ml]"
153
+ set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
154
+ ```
155
+
156
+ Run local NLI when you want mechanical claim-versus-span entailment:
157
+
158
+ ```bash
159
+ pip install "contexttrace[nli]"
160
+ set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
161
+ contexttrace verify trace.json --mode nli --report
162
+ contexttrace nli-calibrate --case-set all --report
163
+ ```
164
+
165
+ Run a local judge with Ollama:
166
+
167
+ ```bash
168
+ set CONTEXTTRACE_JUDGE_PROVIDER=ollama
169
+ set CONTEXTTRACE_JUDGE_MODEL=llama3.1
170
+
171
+ contexttrace verify trace.json --mode judge --report
172
+ contexttrace judge-calibrate --case-set all --report
173
+ ```
174
+
175
+ Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
176
+
177
+ ## Diagnose And Regression-Test
178
+
179
+ ```bash
180
+ # Find whether support existed elsewhere in the corpus.
181
+ contexttrace audit trace.json --corpus docs/ --report
182
+
183
+ # Compare a baseline and current answer after a prompt, model, or retriever change.
184
+ contexttrace compare baseline.json current.json --report
185
+
186
+ # Turn saved failures into replayable endpoint tests.
187
+ contexttrace suite create traces/failure.json --out contexttrace-suite.json
188
+ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
189
+ ```
190
+
191
+ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
192
+
193
+ `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
194
+
195
+ Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
196
+
197
+ ## Capture Existing Systems
198
+
199
+ Capture one live endpoint response:
200
+
201
+ ```bash
202
+ contexttrace capture endpoint \
203
+ --endpoint http://localhost:8000/query \
204
+ --query "What is the refund policy?" \
205
+ --answer-path $.answer \
206
+ --contexts-path $.contexts \
207
+ --citations-path $.citations \
208
+ --out traces/refund_trace.json \
209
+ --verify \
210
+ --report
211
+ ```
212
+
213
+ Or capture artifacts from Python:
214
+
215
+ ```python
216
+ from contexttrace import capture_rag_trace, write_rag_trace
217
+
218
+ trace = capture_rag_trace(
219
+ query=question,
220
+ answer=answer,
221
+ contexts=retrieved_docs,
222
+ metadata={"system": "support-rag"},
223
+ )
224
+ write_rag_trace(trace, "trace.json")
225
+ ```
226
+
227
+ ## SDK Example
228
+
229
+ ```python
230
+ from contexttrace import ContextTrace
231
+
232
+ ct = ContextTrace(project="support-rag")
233
+
234
+ with ct.trace(query="What is the refund policy?") as trace:
235
+ chunks = retriever.search("What is the refund policy?")
236
+ trace.log_retrieval(chunks)
237
+ trace.log_context(chunks[:5])
238
+
239
+ answer = llm.generate("What is the refund policy?", chunks[:5])
240
+ trace.log_answer(answer, usage={"total_tokens": 1200})
241
+ trace.log_citations([
242
+ {"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
243
+ ])
244
+
245
+ result = trace.evaluate()
246
+ print(result["failure"]["failure_type"])
247
+ ```
248
+
249
+ ## Integrations
250
+
251
+ ```bash
252
+ pip install "contexttrace[langchain]"
253
+ pip install "contexttrace[llamaindex]"
254
+ pip install "contexttrace[fastapi]"
255
+ pip install "contexttrace[langgraph]"
256
+ pip install "contexttrace[otel]"
257
+ pip install "contexttrace[all]"
258
+ ```
259
+
260
+ Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
261
+
262
+ ## Privacy
263
+
264
+ ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
265
+
266
+ - `local_only: true`
267
+ - `log_chunk_text: false`
268
+ - `log_answer_text: false`
269
+ - `storage_path`
270
+ - `judge_cache_enabled: true`
271
+ - `judge_cache_path: .contexttrace/judge_cache.json`
272
+
273
+ ## Limits
274
+
275
+ ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
276
+
277
+ ## Links
278
+
279
+ - Repository: https://github.com/samarth1412/Context-Trace
280
+ - Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
281
+ - Issues: https://github.com/samarth1412/Context-Trace/issues
282
+ - Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
@@ -0,0 +1,213 @@
1
+ # ContextTrace
2
+
3
+ **Local-first evidence-chain debugging for RAG and AI agents.**
4
+
5
+ ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
6
+
7
+ ```text
8
+ query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
9
+ ```
10
+
11
+ It is a Python SDK and CLI, not a hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by default.
12
+
13
+ ## Install
14
+
15
+ ```bash
16
+ pip install contexttrace
17
+ contexttrace init
18
+ ```
19
+
20
+ ## Quickstart
21
+
22
+ ```bash
23
+ contexttrace verify-demo unsupported_claim --report
24
+ contexttrace demo --dataset refund_policy
25
+ contexttrace report --last --open
26
+ ```
27
+
28
+ Default local storage:
29
+
30
+ ```text
31
+ .contexttrace/contexttrace.db
32
+ ```
33
+
34
+ ## Verify A RAG Trace
35
+
36
+ Create a portable trace with a query, answer, retrieved contexts, and optional citations:
37
+
38
+ ```json
39
+ {
40
+ "query": "How long does refund processing take?",
41
+ "answer": "Refunds are processed within 5 business days.",
42
+ "contexts": [
43
+ {
44
+ "id": "policy",
45
+ "text": "Customers may request refunds within 30 days of purchase."
46
+ }
47
+ ]
48
+ }
49
+ ```
50
+
51
+ Run local evidence checks:
52
+
53
+ ```bash
54
+ contexttrace inspect trace.json
55
+ contexttrace verify trace.json --report
56
+ contexttrace qa trace.json --corpus docs/ --report
57
+ ```
58
+
59
+ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
60
+
61
+ Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
62
+
63
+ ## Local Verification Modes
64
+
65
+ | Mode | Use When |
66
+ | --- | --- |
67
+ | `lexical` | Fast default checks with no optional dependencies. |
68
+ | `semantic` | Local paraphrase and role-aware contradiction checks. |
69
+ | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
70
+ | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
71
+ | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
72
+
73
+ Run the stronger local non-LLM verifier:
74
+
75
+ ```bash
76
+ contexttrace verify trace.json --mode local_ml --report
77
+ contexttrace verify-benchmark --mode local_ml --case-set all
78
+ ```
79
+
80
+ Optional neural local-ML support never downloads models automatically:
81
+
82
+ ```bash
83
+ pip install "contexttrace[local-ml]"
84
+ set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
85
+ ```
86
+
87
+ Run local NLI when you want mechanical claim-versus-span entailment:
88
+
89
+ ```bash
90
+ pip install "contexttrace[nli]"
91
+ set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
92
+ contexttrace verify trace.json --mode nli --report
93
+ contexttrace nli-calibrate --case-set all --report
94
+ ```
95
+
96
+ Run a local judge with Ollama:
97
+
98
+ ```bash
99
+ set CONTEXTTRACE_JUDGE_PROVIDER=ollama
100
+ set CONTEXTTRACE_JUDGE_MODEL=llama3.1
101
+
102
+ contexttrace verify trace.json --mode judge --report
103
+ contexttrace judge-calibrate --case-set all --report
104
+ ```
105
+
106
+ Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
107
+
108
+ ## Diagnose And Regression-Test
109
+
110
+ ```bash
111
+ # Find whether support existed elsewhere in the corpus.
112
+ contexttrace audit trace.json --corpus docs/ --report
113
+
114
+ # Compare a baseline and current answer after a prompt, model, or retriever change.
115
+ contexttrace compare baseline.json current.json --report
116
+
117
+ # Turn saved failures into replayable endpoint tests.
118
+ contexttrace suite create traces/failure.json --out contexttrace-suite.json
119
+ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
120
+ ```
121
+
122
+ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
123
+
124
+ `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
125
+
126
+ Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
127
+
128
+ ## Capture Existing Systems
129
+
130
+ Capture one live endpoint response:
131
+
132
+ ```bash
133
+ contexttrace capture endpoint \
134
+ --endpoint http://localhost:8000/query \
135
+ --query "What is the refund policy?" \
136
+ --answer-path $.answer \
137
+ --contexts-path $.contexts \
138
+ --citations-path $.citations \
139
+ --out traces/refund_trace.json \
140
+ --verify \
141
+ --report
142
+ ```
143
+
144
+ Or capture artifacts from Python:
145
+
146
+ ```python
147
+ from contexttrace import capture_rag_trace, write_rag_trace
148
+
149
+ trace = capture_rag_trace(
150
+ query=question,
151
+ answer=answer,
152
+ contexts=retrieved_docs,
153
+ metadata={"system": "support-rag"},
154
+ )
155
+ write_rag_trace(trace, "trace.json")
156
+ ```
157
+
158
+ ## SDK Example
159
+
160
+ ```python
161
+ from contexttrace import ContextTrace
162
+
163
+ ct = ContextTrace(project="support-rag")
164
+
165
+ with ct.trace(query="What is the refund policy?") as trace:
166
+ chunks = retriever.search("What is the refund policy?")
167
+ trace.log_retrieval(chunks)
168
+ trace.log_context(chunks[:5])
169
+
170
+ answer = llm.generate("What is the refund policy?", chunks[:5])
171
+ trace.log_answer(answer, usage={"total_tokens": 1200})
172
+ trace.log_citations([
173
+ {"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
174
+ ])
175
+
176
+ result = trace.evaluate()
177
+ print(result["failure"]["failure_type"])
178
+ ```
179
+
180
+ ## Integrations
181
+
182
+ ```bash
183
+ pip install "contexttrace[langchain]"
184
+ pip install "contexttrace[llamaindex]"
185
+ pip install "contexttrace[fastapi]"
186
+ pip install "contexttrace[langgraph]"
187
+ pip install "contexttrace[otel]"
188
+ pip install "contexttrace[all]"
189
+ ```
190
+
191
+ Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
192
+
193
+ ## Privacy
194
+
195
+ ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
196
+
197
+ - `local_only: true`
198
+ - `log_chunk_text: false`
199
+ - `log_answer_text: false`
200
+ - `storage_path`
201
+ - `judge_cache_enabled: true`
202
+ - `judge_cache_path: .contexttrace/judge_cache.json`
203
+
204
+ ## Limits
205
+
206
+ ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
207
+
208
+ ## Links
209
+
210
+ - Repository: https://github.com/samarth1412/Context-Trace
211
+ - Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
212
+ - Issues: https://github.com/samarth1412/Context-Trace/issues
213
+ - Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
@@ -0,0 +1 @@
1
+ __version__ = "0.9.0"