contexttrace 1.0.0__tar.gz → 1.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (88) hide show
  1. {contexttrace-1.0.0 → contexttrace-1.1.0}/PKG-INFO +351 -329
  2. {contexttrace-1.0.0 → contexttrace-1.1.0}/README.md +263 -245
  3. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/__init__.py +60 -50
  4. contexttrace-1.1.0/contexttrace/_version.py +1 -0
  5. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/capture_endpoint.py +174 -174
  6. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/cli.py +2000 -1959
  7. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/client.py +65 -2
  8. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/config.py +60 -0
  9. contexttrace-1.1.0/contexttrace/contracts.py +84 -0
  10. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/diagnose.py +614 -611
  11. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/diagnose_report.py +297 -297
  12. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/endpoint_eval.py +315 -315
  13. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/fastapi.py +169 -58
  14. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/langchain.py +130 -10
  15. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/langgraph.py +93 -6
  16. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/llamaindex.py +104 -7
  17. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/local.py +8 -1
  18. contexttrace-1.1.0/contexttrace/privacy.py +221 -0
  19. contexttrace-1.1.0/contexttrace/repair.py +481 -0
  20. contexttrace-1.1.0/contexttrace/schemas/__init__.py +1 -0
  21. contexttrace-1.1.0/contexttrace/schemas/claim-verification-v1.schema.json +22 -0
  22. contexttrace-1.1.0/contexttrace/schemas/diagnosis-v1.schema.json +21 -0
  23. contexttrace-1.1.0/contexttrace/schemas/regression-case-v1.schema.json +17 -0
  24. contexttrace-1.1.0/contexttrace/schemas/repair-plan-v1.schema.json +22 -0
  25. contexttrace-1.1.0/contexttrace/schemas/trace-v1.schema.json +43 -0
  26. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/storage/sqlite_store.py +55 -3
  27. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/__init__.py +16 -14
  28. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/abstention.py +87 -81
  29. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/audit.py +688 -688
  30. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/audit_report.py +415 -415
  31. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/benchmark.py +638 -631
  32. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/citations.py +143 -128
  33. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/claims.py +4 -2
  34. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/evidence.py +481 -463
  35. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/facts.py +697 -648
  36. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/public_holdout_cases.json +3538 -3538
  37. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/qa.py +268 -268
  38. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/qa_report.py +340 -340
  39. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/root_cause.py +300 -284
  40. contexttrace-1.1.0/contexttrace/verify/rulepacks/generic_v1.yaml +12 -0
  41. contexttrace-1.1.0/contexttrace/verify/rulepacks/legacy_ragtruth_calibrated.yaml +21 -0
  42. contexttrace-1.1.0/contexttrace/verify/rulepacks/policy.yaml +8 -0
  43. contexttrace-1.1.0/contexttrace/verify/rulepacks/temporal.yaml +8 -0
  44. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/runner.py +688 -413
  45. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/schema.py +14 -0
  46. contexttrace-1.1.0/contexttrace/verify/semantic_core/__init__.py +6 -0
  47. contexttrace-1.1.0/contexttrace/verify/semantic_normalization.py +103 -0
  48. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/source_trust.py +72 -4
  49. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/spans.py +9 -0
  50. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/trace_inspect.py +92 -92
  51. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/verdicts.py +406 -284
  52. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace.egg-info/SOURCES.txt +16 -1
  53. {contexttrace-1.0.0 → contexttrace-1.1.0}/pyproject.toml +94 -86
  54. {contexttrace-1.0.0 → contexttrace-1.1.0}/setup.cfg +4 -4
  55. contexttrace-1.0.0/contexttrace/_version.py +0 -1
  56. {contexttrace-1.0.0 → contexttrace-1.1.0}/MANIFEST.in +0 -0
  57. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/capture.py +0 -0
  58. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/demo.py +0 -0
  59. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/demo_data.py +0 -0
  60. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/errors.py +0 -0
  61. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/evaluator.py +0 -0
  62. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/__init__.py +0 -0
  63. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/integrations/opentelemetry.py +0 -0
  64. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/py.typed +0 -0
  65. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/regression.py +0 -0
  66. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/reliability.py +0 -0
  67. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/report.py +0 -0
  68. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/storage/__init__.py +0 -0
  69. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/thresholds.py +0 -0
  70. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/transport.py +0 -0
  71. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/audit_benchmark.py +0 -0
  72. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
  73. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/calibration.py +0 -0
  74. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/compare.py +0 -0
  75. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/compare_report.py +0 -0
  76. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/demos.py +0 -0
  77. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
  78. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/judges.py +0 -0
  79. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/local_ml.py +0 -0
  80. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/local_nli.py +0 -0
  81. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/nli_calibration.py +0 -0
  82. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
  83. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/report.py +0 -0
  84. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/statuses.py +0 -0
  85. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/suite.py +0 -0
  86. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/verify/suite_report.py +0 -0
  87. {contexttrace-1.0.0 → contexttrace-1.1.0}/contexttrace/viewer.py +0 -0
  88. {contexttrace-1.0.0 → contexttrace-1.1.0}/setup.py +0 -0
@@ -1,329 +1,351 @@
1
- Metadata-Version: 2.1
2
- Name: contexttrace
3
- Version: 1.0.0
4
- Summary: Local-first evidence-chain debugger for RAG and AI agent claim grounding, citation checks, root-cause diagnosis, and regression tests.
5
- Author: ContextTrace contributors
6
- License: MIT
7
- Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
8
- Project-URL: Documentation, https://github.com/samarth1412/Context-Trace/tree/main/docs
9
- Project-URL: Repository, https://github.com/samarth1412/Context-Trace
10
- Project-URL: Issues, https://github.com/samarth1412/Context-Trace/issues
11
- Project-URL: Changelog, https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
12
- Keywords: rag,llm,retrieval-augmented-generation,citations,evaluation,observability,agents,cli,sqlite
13
- Classifier: Development Status :: 5 - Production/Stable
14
- Classifier: Intended Audience :: Developers
15
- Classifier: License :: OSI Approved :: MIT License
16
- Classifier: Programming Language :: Python :: 3
17
- Classifier: Programming Language :: Python :: 3.8
18
- Classifier: Programming Language :: Python :: 3.9
19
- Classifier: Programming Language :: Python :: 3.10
20
- Classifier: Programming Language :: Python :: 3.11
21
- Classifier: Programming Language :: Python :: 3.12
22
- Classifier: Programming Language :: Python :: 3.13
23
- Classifier: Topic :: Software Development :: Libraries :: Python Modules
24
- Classifier: Typing :: Typed
25
- Requires-Python: >=3.8
26
- Description-Content-Type: text/markdown
27
- Requires-Dist: click>=8.1
28
- Requires-Dist: httpx>=0.27
29
- Requires-Dist: typing-extensions>=4.9
30
- Provides-Extra: langchain
31
- Requires-Dist: langchain-core>=0.2; extra == "langchain"
32
- Provides-Extra: llamaindex
33
- Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
34
- Provides-Extra: local
35
- Provides-Extra: local-ml
36
- Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
37
- Provides-Extra: nli
38
- Requires-Dist: torch>=2.0; extra == "nli"
39
- Requires-Dist: transformers>=4.41; extra == "nli"
40
- Provides-Extra: nli-onnx
41
- Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
42
- Requires-Dist: transformers>=4.41; extra == "nli-onnx"
43
- Provides-Extra: fastapi
44
- Requires-Dist: fastapi>=0.110; extra == "fastapi"
45
- Provides-Extra: langgraph
46
- Requires-Dist: langgraph>=0.2; extra == "langgraph"
47
- Provides-Extra: otel
48
- Requires-Dist: opentelemetry-api>=1.24; extra == "otel"
49
- Provides-Extra: opentelemetry
50
- Requires-Dist: opentelemetry-api>=1.24; extra == "opentelemetry"
51
- Provides-Extra: integrations
52
- Requires-Dist: fastapi>=0.110; extra == "integrations"
53
- Requires-Dist: langchain-core>=0.2; extra == "integrations"
54
- Requires-Dist: langgraph>=0.2; extra == "integrations"
55
- Requires-Dist: llama-index-core>=0.10; extra == "integrations"
56
- Requires-Dist: opentelemetry-api>=1.24; extra == "integrations"
57
- Provides-Extra: all
58
- Requires-Dist: fastapi>=0.110; extra == "all"
59
- Requires-Dist: langchain-core>=0.2; extra == "all"
60
- Requires-Dist: langgraph>=0.2; extra == "all"
61
- Requires-Dist: llama-index-core>=0.10; extra == "all"
62
- Requires-Dist: opentelemetry-api>=1.24; extra == "all"
63
- Requires-Dist: sentence-transformers>=2.7; extra == "all"
64
- Requires-Dist: torch>=2.0; extra == "all"
65
- Requires-Dist: transformers>=4.41; extra == "all"
66
- Requires-Dist: onnxruntime>=1.17; extra == "all"
67
- Provides-Extra: test
68
- Requires-Dist: pytest>=8.0; extra == "test"
69
-
70
- # ContextTrace
71
-
72
- **Local-first evidence-chain forensics for RAG and AI agents.**
73
-
74
- ContextTrace is a Python SDK and CLI for tracing a failed answer from the user
75
- query through retrieved context, answer claims, citations, verdicts, root cause,
76
- repair guidance, and CI regression tests.
77
-
78
- ```text
79
- query -> retrieved context -> answer claims -> citations -> verdicts -> root cause -> regression test
80
- ```
81
-
82
- Use it when a RAG or agent score is not enough: ContextTrace points at the
83
- unsupported or contradicted claim, the evidence span and citation involved, why
84
- the failure likely happened, and how to keep it from coming back. It is not a
85
- hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by
86
- default.
87
-
88
- ## Install
89
-
90
- ```bash
91
- pip install contexttrace
92
- contexttrace init
93
- ```
94
-
95
- ## Quickstart
96
-
97
- ```bash
98
- contexttrace verify-demo unsupported_claim --report
99
- contexttrace demo --dataset refund_policy
100
- contexttrace report --last --open
101
- ```
102
-
103
- Default local storage:
104
-
105
- ```text
106
- .contexttrace/contexttrace.db
107
- ```
108
-
109
- ## Verify A RAG Trace
110
-
111
- Create a portable trace with a query, answer, retrieved contexts, and optional citations:
112
-
113
- ```json
114
- {
115
- "query": "How long does refund processing take?",
116
- "answer": "Refunds are processed within 5 business days.",
117
- "contexts": [
118
- {
119
- "id": "policy",
120
- "text": "Customers may request refunds within 30 days of purchase."
121
- }
122
- ]
123
- }
124
- ```
125
-
126
- Run local evidence checks:
127
-
128
- ```bash
129
- contexttrace inspect trace.json
130
- contexttrace verify trace.json --report
131
- contexttrace diagnose trace.json --report
132
- contexttrace qa trace.json --corpus docs/ --report
133
- ```
134
-
135
- ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
136
-
137
- Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
138
-
139
- ## Diagnose An Agent Trace
140
-
141
- `diagnose` also accepts agent step traces and localizes tool/final-answer
142
- failures:
143
-
144
- ```json
145
- {
146
- "goal": "Book a meeting with Alex",
147
- "steps": [
148
- {
149
- "type": "tool_call",
150
- "tool": "calendar.search",
151
- "args": {"date": "Friday"},
152
- "result": "No availability"
153
- },
154
- {
155
- "type": "final_answer",
156
- "content": "I booked it for Friday."
157
- }
158
- ]
159
- }
160
- ```
161
-
162
- ```bash
163
- contexttrace diagnose examples/diagnose_agent_trace.json --report --fail-on high_risk
164
- ```
165
-
166
- The diagnosis flags `tool_result_contradicted_by_final_answer` and suggests
167
- gating final-answer generation on tool-result status.
168
-
169
- Turn that diagnosis into a CI regression test:
170
-
171
- ```bash
172
- contexttrace diagnose examples/diagnose_agent_trace.json \
173
- --generate-test \
174
- --test-out tests/contexttrace/test_calendar_agent_diagnosis.py
175
-
176
- pytest tests/contexttrace/test_calendar_agent_diagnosis.py
177
- ```
178
-
179
- ## Local Verification Modes
180
-
181
- | Mode | Use When |
182
- | --- | --- |
183
- | `lexical` | Fast default checks with no optional dependencies. |
184
- | `semantic` | Local paraphrase and role-aware contradiction checks. |
185
- | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
186
- | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
187
- | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
188
-
189
- Run the stronger local non-LLM verifier:
190
-
191
- ```bash
192
- contexttrace verify trace.json --mode local_ml --report
193
- contexttrace verify-benchmark --mode local_ml --case-set all
194
- ```
195
-
196
- Optional neural local-ML support never downloads models automatically:
197
-
198
- ```bash
199
- pip install "contexttrace[local-ml]"
200
- set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
201
- ```
202
-
203
- Run local NLI when you want mechanical claim-versus-span entailment:
204
-
205
- ```bash
206
- pip install "contexttrace[nli]"
207
- set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
208
- contexttrace verify trace.json --mode nli --report
209
- contexttrace nli-calibrate --case-set all --report
210
- ```
211
-
212
- Run a local judge with Ollama:
213
-
214
- ```bash
215
- set CONTEXTTRACE_JUDGE_PROVIDER=ollama
216
- set CONTEXTTRACE_JUDGE_MODEL=llama3.1
217
-
218
- contexttrace verify trace.json --mode judge --report
219
- contexttrace judge-calibrate --case-set all --report
220
- ```
221
-
222
- Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
223
-
224
- ## Diagnose And Regression-Test
225
-
226
- ```bash
227
- # Find whether support existed elsewhere in the corpus.
228
- contexttrace audit trace.json --corpus docs/ --report
229
-
230
- # Compare a baseline and current answer after a prompt, model, or retriever change.
231
- contexttrace compare baseline.json current.json --report
232
-
233
- # Turn saved failures into replayable endpoint tests.
234
- contexttrace suite create traces/failure.json --out contexttrace-suite.json
235
- contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
236
- ```
237
-
238
- Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
239
-
240
- `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
241
-
242
- Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
243
-
244
- ## Capture Existing Systems
245
-
246
- Capture one live endpoint response:
247
-
248
- ```bash
249
- contexttrace capture endpoint \
250
- --endpoint http://localhost:8000/query \
251
- --query "What is the refund policy?" \
252
- --answer-path $.answer \
253
- --contexts-path $.contexts \
254
- --citations-path $.citations \
255
- --out traces/refund_trace.json \
256
- --verify \
257
- --report
258
- ```
259
-
260
- Or capture artifacts from Python:
261
-
262
- ```python
263
- from contexttrace import capture_rag_trace, write_rag_trace
264
-
265
- trace = capture_rag_trace(
266
- query=question,
267
- answer=answer,
268
- contexts=retrieved_docs,
269
- metadata={"system": "support-rag"},
270
- )
271
- write_rag_trace(trace, "trace.json")
272
- ```
273
-
274
- ## SDK Example
275
-
276
- ```python
277
- from contexttrace import ContextTrace
278
-
279
- ct = ContextTrace(project="support-rag")
280
-
281
- with ct.trace(query="What is the refund policy?") as trace:
282
- chunks = retriever.search("What is the refund policy?")
283
- trace.log_retrieval(chunks)
284
- trace.log_context(chunks[:5])
285
-
286
- answer = llm.generate("What is the refund policy?", chunks[:5])
287
- trace.log_answer(answer, usage={"total_tokens": 1200})
288
- trace.log_citations([
289
- {"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
290
- ])
291
-
292
- result = trace.evaluate()
293
- print(result["failure"]["failure_type"])
294
- ```
295
-
296
- ## Integrations
297
-
298
- ```bash
299
- pip install "contexttrace[langchain]"
300
- pip install "contexttrace[llamaindex]"
301
- pip install "contexttrace[fastapi]"
302
- pip install "contexttrace[langgraph]"
303
- pip install "contexttrace[otel]"
304
- pip install "contexttrace[all]"
305
- ```
306
-
307
- Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
308
-
309
- ## Privacy
310
-
311
- ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
312
-
313
- - `local_only: true`
314
- - `log_chunk_text: false`
315
- - `log_answer_text: false`
316
- - `storage_path`
317
- - `judge_cache_enabled: true`
318
- - `judge_cache_path: .contexttrace/judge_cache.json`
319
-
320
- ## Limits
321
-
322
- ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
323
-
324
- ## Links
325
-
326
- - Repository: https://github.com/samarth1412/Context-Trace
327
- - Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
328
- - Issues: https://github.com/samarth1412/Context-Trace/issues
329
- - Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
1
+ Metadata-Version: 2.4
2
+ Name: contexttrace
3
+ Version: 1.1.0
4
+ Summary: Local-first evidence-chain debugger for RAG and AI agent claim grounding, citation checks, root-cause diagnosis, and regression tests.
5
+ Author: ContextTrace contributors
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
8
+ Project-URL: Documentation, https://github.com/samarth1412/Context-Trace/tree/main/docs
9
+ Project-URL: Repository, https://github.com/samarth1412/Context-Trace
10
+ Project-URL: Issues, https://github.com/samarth1412/Context-Trace/issues
11
+ Project-URL: Changelog, https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md
12
+ Keywords: rag,llm,retrieval-augmented-generation,citations,evaluation,observability,agents,cli,sqlite
13
+ Classifier: Development Status :: 5 - Production/Stable
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.10
17
+ Classifier: Programming Language :: Python :: 3.11
18
+ Classifier: Programming Language :: Python :: 3.12
19
+ Classifier: Programming Language :: Python :: 3.13
20
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
21
+ Classifier: Typing :: Typed
22
+ Requires-Python: >=3.10
23
+ Description-Content-Type: text/markdown
24
+ Requires-Dist: click>=8.1
25
+ Requires-Dist: httpx>=0.27
26
+ Requires-Dist: typing-extensions>=4.9
27
+ Provides-Extra: langchain
28
+ Requires-Dist: langchain-core>=0.2; extra == "langchain"
29
+ Provides-Extra: llamaindex
30
+ Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
31
+ Provides-Extra: local
32
+ Provides-Extra: local-ml
33
+ Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
34
+ Provides-Extra: nli
35
+ Requires-Dist: torch>=2.0; extra == "nli"
36
+ Requires-Dist: transformers>=4.41; extra == "nli"
37
+ Provides-Extra: nli-onnx
38
+ Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
39
+ Requires-Dist: transformers>=4.41; extra == "nli-onnx"
40
+ Provides-Extra: fastapi
41
+ Requires-Dist: fastapi>=0.110; extra == "fastapi"
42
+ Provides-Extra: langgraph
43
+ Requires-Dist: langgraph>=0.2; extra == "langgraph"
44
+ Provides-Extra: otel
45
+ Requires-Dist: opentelemetry-api>=1.24; extra == "otel"
46
+ Provides-Extra: opentelemetry
47
+ Requires-Dist: opentelemetry-api>=1.24; extra == "opentelemetry"
48
+ Provides-Extra: integrations
49
+ Requires-Dist: fastapi>=0.110; extra == "integrations"
50
+ Requires-Dist: langchain-core>=0.2; extra == "integrations"
51
+ Requires-Dist: langgraph>=0.2; extra == "integrations"
52
+ Requires-Dist: llama-index-core>=0.10; extra == "integrations"
53
+ Requires-Dist: opentelemetry-api>=1.24; extra == "integrations"
54
+ Provides-Extra: all
55
+ Requires-Dist: fastapi>=0.110; extra == "all"
56
+ Requires-Dist: langchain-core>=0.2; extra == "all"
57
+ Requires-Dist: langgraph>=0.2; extra == "all"
58
+ Requires-Dist: llama-index-core>=0.10; extra == "all"
59
+ Requires-Dist: opentelemetry-api>=1.24; extra == "all"
60
+ Requires-Dist: sentence-transformers>=2.7; extra == "all"
61
+ Requires-Dist: torch>=2.0; extra == "all"
62
+ Requires-Dist: transformers>=4.41; extra == "all"
63
+ Requires-Dist: onnxruntime>=1.17; extra == "all"
64
+ Provides-Extra: test
65
+ Requires-Dist: jsonschema>=4.22; extra == "test"
66
+ Requires-Dist: pytest>=8.0; extra == "test"
67
+ Provides-Extra: quality
68
+ Requires-Dist: build>=1.2; extra == "quality"
69
+ Requires-Dist: mypy>=1.10; extra == "quality"
70
+ Requires-Dist: pip-audit>=2.7; extra == "quality"
71
+ Requires-Dist: pytest-cov>=5.0; extra == "quality"
72
+ Requires-Dist: ruff>=0.5; extra == "quality"
73
+
74
+ # ContextTrace
75
+
76
+ **Local-first evidence-chain forensics for RAG and AI agents.**
77
+
78
+ ContextTrace is a Python SDK and CLI for tracing a failed answer from the user
79
+ query through retrieved context, answer claims, citations, verdicts, root cause,
80
+ repair guidance, and CI regression tests.
81
+
82
+ ```text
83
+ query -> retrieved context -> answer claims -> citations -> verdicts -> root cause -> regression test
84
+ ```
85
+
86
+ Use it when a RAG or agent score is not enough: ContextTrace points at the
87
+ unsupported or contradicted claim, the evidence span and citation involved, why
88
+ the failure likely happened, and how to keep it from coming back. It is not a
89
+ hosted dashboard. Traces, reports, judge cache, and SQLite state stay local by
90
+ default.
91
+
92
+ ## Install
93
+
94
+ ```bash
95
+ pip install contexttrace
96
+ contexttrace init
97
+ ```
98
+
99
+ ## Quickstart
100
+
101
+ ```bash
102
+ contexttrace verify-demo unsupported_claim --report
103
+ contexttrace demo --dataset refund_policy
104
+ contexttrace report --last --open
105
+ ```
106
+
107
+ Default local storage:
108
+
109
+ ```text
110
+ .contexttrace/contexttrace.db
111
+ ```
112
+
113
+ ## Verify A RAG Trace
114
+
115
+ Create a portable trace with a query, answer, retrieved contexts, and optional citations:
116
+
117
+ ```json
118
+ {
119
+ "query": "How long does refund processing take?",
120
+ "answer": "Refunds are processed within 5 business days.",
121
+ "contexts": [
122
+ {
123
+ "id": "policy",
124
+ "text": "Customers may request refunds within 30 days of purchase."
125
+ }
126
+ ]
127
+ }
128
+ ```
129
+
130
+ Run local evidence checks:
131
+
132
+ ```bash
133
+ contexttrace inspect trace.json
134
+ contexttrace verify trace.json --report
135
+ contexttrace diagnose trace.json --report
136
+ contexttrace qa trace.json --corpus docs/ --report
137
+ contexttrace repair trace.json --corpus docs/ --out repair_plan.md
138
+ ```
139
+
140
+ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
141
+
142
+ Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
143
+
144
+ ## Diagnose An Agent Trace
145
+
146
+ `diagnose` also accepts agent step traces and localizes tool/final-answer
147
+ failures:
148
+
149
+ ```json
150
+ {
151
+ "goal": "Book a meeting with Alex",
152
+ "steps": [
153
+ {
154
+ "type": "tool_call",
155
+ "tool": "calendar.search",
156
+ "args": {"date": "Friday"},
157
+ "result": "No availability"
158
+ },
159
+ {
160
+ "type": "final_answer",
161
+ "content": "I booked it for Friday."
162
+ }
163
+ ]
164
+ }
165
+ ```
166
+
167
+ ```bash
168
+ contexttrace diagnose examples/diagnose_agent_trace.json --report --fail-on high_risk
169
+ ```
170
+
171
+ The diagnosis flags `tool_result_contradicted_by_final_answer` and suggests
172
+ gating final-answer generation on tool-result status.
173
+
174
+ Turn that diagnosis into a CI regression test:
175
+
176
+ ```bash
177
+ contexttrace diagnose examples/diagnose_agent_trace.json \
178
+ --generate-test \
179
+ --test-out tests/contexttrace/test_calendar_agent_diagnosis.py
180
+
181
+ pytest tests/contexttrace/test_calendar_agent_diagnosis.py
182
+ ```
183
+
184
+ ## Build A Repair Plan
185
+
186
+ `repair` turns diagnosis into an evidence-backed implementation plan. With a
187
+ local corpus, it distinguishes retrieval miss, reranking failure, chunking
188
+ issue, corpus gap, answer overreach, and stale or conflicting evidence:
189
+
190
+ ```bash
191
+ contexttrace repair trace.json \
192
+ --corpus docs/ \
193
+ --out repair_plan.md \
194
+ --json-out repair_plan.json
195
+ ```
196
+
197
+ The plan records the failed claim, retrieved and corpus evidence, prioritized
198
+ root-cause-specific changes, and commands to verify the fix. Add only the
199
+ recaptured passing trace to the generated must-pass regression command.
200
+
201
+ ## Local Verification Modes
202
+
203
+ | Mode | Use When |
204
+ | --- | --- |
205
+ | `lexical` | Fast default checks with no optional dependencies. |
206
+ | `semantic` | Local paraphrase and role-aware contradiction checks. |
207
+ | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
208
+ | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
209
+ | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
210
+
211
+ Run the stronger local non-LLM verifier:
212
+
213
+ ```bash
214
+ contexttrace verify trace.json --mode local_ml --report
215
+ contexttrace verify-benchmark --mode local_ml --case-set all
216
+ ```
217
+
218
+ Optional neural local-ML support never downloads models automatically:
219
+
220
+ ```bash
221
+ pip install "contexttrace[local-ml]"
222
+ set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
223
+ ```
224
+
225
+ Run local NLI when you want mechanical claim-versus-span entailment:
226
+
227
+ ```bash
228
+ pip install "contexttrace[nli]"
229
+ set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
230
+ contexttrace verify trace.json --mode nli --report
231
+ contexttrace nli-calibrate --case-set all --report
232
+ ```
233
+
234
+ Run a local judge with Ollama:
235
+
236
+ ```bash
237
+ set CONTEXTTRACE_JUDGE_PROVIDER=ollama
238
+ set CONTEXTTRACE_JUDGE_MODEL=llama3.1
239
+
240
+ contexttrace verify trace.json --mode judge --report
241
+ contexttrace judge-calibrate --case-set all --report
242
+ ```
243
+
244
+ Remote judges are blocked while `local_only: true` is active. To use a remote judge, explicitly disable local-only mode and configure the provider/API key.
245
+
246
+ ## Diagnose And Regression-Test
247
+
248
+ ```bash
249
+ # Find whether support existed elsewhere in the corpus.
250
+ contexttrace audit trace.json --corpus docs/ --report
251
+
252
+ # Compare a baseline and current answer after a prompt, model, or retriever change.
253
+ contexttrace compare baseline.json current.json --report
254
+
255
+ # Turn saved failures into replayable endpoint tests.
256
+ contexttrace suite create traces/failure.json --out contexttrace-suite.json
257
+ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/query --report
258
+ ```
259
+
260
+ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
261
+
262
+ `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
263
+
264
+ Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
265
+
266
+ ## Capture Existing Systems
267
+
268
+ Capture one live endpoint response:
269
+
270
+ ```bash
271
+ contexttrace capture endpoint \
272
+ --endpoint http://localhost:8000/query \
273
+ --query "What is the refund policy?" \
274
+ --answer-path $.answer \
275
+ --contexts-path $.contexts \
276
+ --citations-path $.citations \
277
+ --out traces/refund_trace.json \
278
+ --verify \
279
+ --report
280
+ ```
281
+
282
+ Or capture artifacts from Python:
283
+
284
+ ```python
285
+ from contexttrace import capture_rag_trace, write_rag_trace
286
+
287
+ trace = capture_rag_trace(
288
+ query=question,
289
+ answer=answer,
290
+ contexts=retrieved_docs,
291
+ metadata={"system": "support-rag"},
292
+ )
293
+ write_rag_trace(trace, "trace.json")
294
+ ```
295
+
296
+ ## SDK Example
297
+
298
+ ```python
299
+ from contexttrace import ContextTrace
300
+
301
+ ct = ContextTrace(project="support-rag")
302
+
303
+ with ct.trace(query="What is the refund policy?") as trace:
304
+ chunks = retriever.search("What is the refund policy?")
305
+ trace.log_retrieval(chunks)
306
+ trace.log_context(chunks[:5])
307
+
308
+ answer = llm.generate("What is the refund policy?", chunks[:5])
309
+ trace.log_answer(answer, usage={"total_tokens": 1200})
310
+ trace.log_citations([
311
+ {"claim": "Refunds are available within 30 days.", "source_chunk_id": "chunk_12"}
312
+ ])
313
+
314
+ result = trace.evaluate()
315
+ print(result["failure"]["failure_type"])
316
+ ```
317
+
318
+ ## Integrations
319
+
320
+ ```bash
321
+ pip install "contexttrace[langchain]"
322
+ pip install "contexttrace[llamaindex]"
323
+ pip install "contexttrace[fastapi]"
324
+ pip install "contexttrace[langgraph]"
325
+ pip install "contexttrace[otel]"
326
+ pip install "contexttrace[all]"
327
+ ```
328
+
329
+ Includes LangChain, LlamaIndex, FastAPI, LangGraph, and OpenTelemetry hooks.
330
+
331
+ ## Privacy
332
+
333
+ ContextTrace makes no network calls unless you point it at an endpoint or configure a judge provider. Local controls include:
334
+
335
+ - `local_only: true`
336
+ - `log_chunk_text: false`
337
+ - `log_answer_text: false`
338
+ - `storage_path`
339
+ - `judge_cache_enabled: true`
340
+ - `judge_cache_path: .contexttrace/judge_cache.json`
341
+
342
+ ## Limits
343
+
344
+ ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
345
+
346
+ ## Links
347
+
348
+ - Repository: https://github.com/samarth1412/Context-Trace
349
+ - Documentation: https://github.com/samarth1412/Context-Trace/tree/main/docs
350
+ - Issues: https://github.com/samarth1412/Context-Trace/issues
351
+ - Changelog: https://github.com/samarth1412/Context-Trace/blob/main/CHANGELOG.md