contexttrace 0.8.0__tar.gz → 0.9.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. {contexttrace-0.8.0 → contexttrace-0.9.0}/PKG-INFO +31 -6
  2. {contexttrace-0.8.0 → contexttrace-0.9.0}/README.md +21 -5
  3. contexttrace-0.9.0/contexttrace/_version.py +1 -0
  4. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/cli.py +111 -19
  5. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/__init__.py +53 -1
  6. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/benchmark.py +7 -1
  7. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/citations.py +2 -1
  8. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/evidence.py +6 -5
  9. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/judges.py +5 -2
  10. contexttrace-0.9.0/contexttrace/verify/local_nli.py +418 -0
  11. contexttrace-0.9.0/contexttrace/verify/nli_calibration.py +468 -0
  12. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/qa_report.py +41 -21
  13. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/report.py +98 -4
  14. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/root_cause.py +52 -0
  15. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/runner.py +163 -10
  16. contexttrace-0.9.0/contexttrace/verify/source_trust.py +340 -0
  17. contexttrace-0.9.0/contexttrace/verify/statuses.py +122 -0
  18. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/verdicts.py +1 -1
  19. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace.egg-info/SOURCES.txt +4 -0
  20. {contexttrace-0.8.0 → contexttrace-0.9.0}/pyproject.toml +14 -3
  21. contexttrace-0.8.0/contexttrace/_version.py +0 -1
  22. {contexttrace-0.8.0 → contexttrace-0.9.0}/MANIFEST.in +0 -0
  23. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/__init__.py +0 -0
  24. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/capture.py +0 -0
  25. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/capture_endpoint.py +0 -0
  26. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/client.py +0 -0
  27. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/config.py +0 -0
  28. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/demo.py +0 -0
  29. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/demo_data.py +0 -0
  30. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/endpoint_eval.py +0 -0
  31. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/errors.py +0 -0
  32. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/evaluator.py +0 -0
  33. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/__init__.py +0 -0
  34. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/fastapi.py +0 -0
  35. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/langchain.py +0 -0
  36. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/langgraph.py +0 -0
  37. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/llamaindex.py +0 -0
  38. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/opentelemetry.py +0 -0
  39. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/local.py +0 -0
  40. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/py.typed +0 -0
  41. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/regression.py +0 -0
  42. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/reliability.py +0 -0
  43. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/report.py +0 -0
  44. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/storage/__init__.py +0 -0
  45. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/storage/sqlite_store.py +0 -0
  46. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/thresholds.py +0 -0
  47. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/transport.py +0 -0
  48. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/abstention.py +0 -0
  49. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit.py +0 -0
  50. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark.py +0 -0
  51. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
  52. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_report.py +0 -0
  53. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/calibration.py +0 -0
  54. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/claims.py +0 -0
  55. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/compare.py +0 -0
  56. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/compare_report.py +0 -0
  57. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/demos.py +0 -0
  58. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
  59. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/facts.py +0 -0
  60. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/local_ml.py +0 -0
  61. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/qa.py +0 -0
  62. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
  63. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/schema.py +0 -0
  64. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/spans.py +0 -0
  65. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/suite.py +0 -0
  66. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/suite_report.py +0 -0
  67. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/trace_inspect.py +0 -0
  68. {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/viewer.py +0 -0
  69. {contexttrace-0.8.0 → contexttrace-0.9.0}/setup.cfg +0 -0
  70. {contexttrace-0.8.0 → contexttrace-0.9.0}/setup.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.1
2
2
  Name: contexttrace
3
- Version: 0.8.0
3
+ Version: 0.9.0
4
4
  Summary: Local-first SDK and CLI for RAG and agent reliability tracing, citation checks, and failure diagnosis.
5
5
  Author: ContextTrace contributors
6
6
  License: MIT
@@ -34,6 +34,12 @@ Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
34
34
  Provides-Extra: local
35
35
  Provides-Extra: local-ml
36
36
  Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
37
+ Provides-Extra: nli
38
+ Requires-Dist: torch>=2.0; extra == "nli"
39
+ Requires-Dist: transformers>=4.41; extra == "nli"
40
+ Provides-Extra: nli-onnx
41
+ Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
42
+ Requires-Dist: transformers>=4.41; extra == "nli-onnx"
37
43
  Provides-Extra: fastapi
38
44
  Requires-Dist: fastapi>=0.110; extra == "fastapi"
39
45
  Provides-Extra: langgraph
@@ -55,6 +61,9 @@ Requires-Dist: langgraph>=0.2; extra == "all"
55
61
  Requires-Dist: llama-index-core>=0.10; extra == "all"
56
62
  Requires-Dist: opentelemetry-api>=1.24; extra == "all"
57
63
  Requires-Dist: sentence-transformers>=2.7; extra == "all"
64
+ Requires-Dist: torch>=2.0; extra == "all"
65
+ Requires-Dist: transformers>=4.41; extra == "all"
66
+ Requires-Dist: onnxruntime>=1.17; extra == "all"
58
67
  Provides-Extra: test
59
68
  Requires-Dist: pytest>=8.0; extra == "test"
60
69
 
@@ -62,7 +71,7 @@ Requires-Dist: pytest>=8.0; extra == "test"
62
71
 
63
72
  **Local-first evidence-chain debugging for RAG and AI agents.**
64
73
 
65
- ContextTrace shows where an answer stopped being grounded:
74
+ ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
66
75
 
67
76
  ```text
68
77
  query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
@@ -116,7 +125,9 @@ contexttrace verify trace.json --report
116
125
  contexttrace qa trace.json --corpus docs/ --report
117
126
  ```
118
127
 
119
- ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then explains the likely fix.
128
+ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
129
+
130
+ Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
120
131
 
121
132
  ## Local Verification Modes
122
133
 
@@ -125,7 +136,8 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
125
136
  | `lexical` | Fast default checks with no optional dependencies. |
126
137
  | `semantic` | Local paraphrase and role-aware contradiction checks. |
127
138
  | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
128
- | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. |
139
+ | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
140
+ | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
129
141
 
130
142
  Run the stronger local non-LLM verifier:
131
143
 
@@ -138,7 +150,16 @@ Optional neural local-ML support never downloads models automatically:
138
150
 
139
151
  ```bash
140
152
  pip install "contexttrace[local-ml]"
141
- set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:\models\bge-small-en-v1.5
153
+ set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
154
+ ```
155
+
156
+ Run local NLI when you want mechanical claim-versus-span entailment:
157
+
158
+ ```bash
159
+ pip install "contexttrace[nli]"
160
+ set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
161
+ contexttrace verify trace.json --mode nli --report
162
+ contexttrace nli-calibrate --case-set all --report
142
163
  ```
143
164
 
144
165
  Run a local judge with Ollama:
@@ -169,6 +190,10 @@ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/
169
190
 
170
191
  Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
171
192
 
193
+ `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
194
+
195
+ Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
196
+
172
197
  ## Capture Existing Systems
173
198
 
174
199
  Capture one live endpoint response:
@@ -247,7 +272,7 @@ ContextTrace makes no network calls unless you point it at an endpoint or config
247
272
 
248
273
  ## Limits
249
274
 
250
- ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
275
+ ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
251
276
 
252
277
  ## Links
253
278
 
@@ -2,7 +2,7 @@
2
2
 
3
3
  **Local-first evidence-chain debugging for RAG and AI agents.**
4
4
 
5
- ContextTrace shows where an answer stopped being grounded:
5
+ ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
6
6
 
7
7
  ```text
8
8
  query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
@@ -56,7 +56,9 @@ contexttrace verify trace.json --report
56
56
  contexttrace qa trace.json --corpus docs/ --report
57
57
  ```
58
58
 
59
- ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then explains the likely fix.
59
+ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
60
+
61
+ Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
60
62
 
61
63
  ## Local Verification Modes
62
64
 
@@ -65,7 +67,8 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
65
67
  | `lexical` | Fast default checks with no optional dependencies. |
66
68
  | `semantic` | Local paraphrase and role-aware contradiction checks. |
67
69
  | `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
68
- | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. |
70
+ | `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
71
+ | `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
69
72
 
70
73
  Run the stronger local non-LLM verifier:
71
74
 
@@ -78,7 +81,16 @@ Optional neural local-ML support never downloads models automatically:
78
81
 
79
82
  ```bash
80
83
  pip install "contexttrace[local-ml]"
81
- set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:\models\bge-small-en-v1.5
84
+ set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
85
+ ```
86
+
87
+ Run local NLI when you want mechanical claim-versus-span entailment:
88
+
89
+ ```bash
90
+ pip install "contexttrace[nli]"
91
+ set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
92
+ contexttrace verify trace.json --mode nli --report
93
+ contexttrace nli-calibrate --case-set all --report
82
94
  ```
83
95
 
84
96
  Run a local judge with Ollama:
@@ -109,6 +121,10 @@ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/
109
121
 
110
122
  Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
111
123
 
124
+ `support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
125
+
126
+ Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
127
+
112
128
  ## Capture Existing Systems
113
129
 
114
130
  Capture one live endpoint response:
@@ -187,7 +203,7 @@ ContextTrace makes no network calls unless you point it at an endpoint or config
187
203
 
188
204
  ## Limits
189
205
 
190
- ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
206
+ ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
191
207
 
192
208
  ## Links
193
209
 
@@ -0,0 +1 @@
1
+ __version__ = "0.9.0"
@@ -31,6 +31,7 @@ from contexttrace.verify import (
31
31
  audit_trace,
32
32
  build_judge_provider,
33
33
  run_judge_calibration,
34
+ run_nli_calibration,
34
35
  compare_failures,
35
36
  compare_trace_files,
36
37
  list_verify_demos,
@@ -38,6 +39,7 @@ from contexttrace.verify import (
38
39
  load_verify_demo,
39
40
  verify_trace,
40
41
  write_judge_calibration_report,
42
+ write_nli_calibration_report,
41
43
  )
42
44
  from contexttrace.verify.benchmark import run_verify_benchmark, write_verify_benchmark_report
43
45
  from contexttrace.verify.audit_benchmark import run_audit_benchmark, write_audit_benchmark_report
@@ -64,13 +66,16 @@ from contexttrace.verify.trace_inspect import inspect_trace
64
66
  from contexttrace.viewer import serve_viewer
65
67
 
66
68
 
67
- SAMPLE_QUESTIONS = [
68
- {
69
- "id": "refund_policy",
70
- "query": "What is the refund policy?",
71
- "expected_sources": ["refund_policy.md"],
72
- }
73
- ]
69
+ SAMPLE_QUESTIONS = [
70
+ {
71
+ "id": "refund_policy",
72
+ "query": "What is the refund policy?",
73
+ "expected_sources": ["refund_policy.md"],
74
+ }
75
+ ]
76
+
77
+ BASIC_VERIFY_MODES = ["lexical", "semantic", "local_ml", "local-ml", "nli"]
78
+ JUDGE_VERIFY_MODES = [*BASIC_VERIFY_MODES, "judge"]
74
79
 
75
80
 
76
81
  @click.group(context_settings={"help_option_names": ["-h", "--help"]})
@@ -244,7 +249,7 @@ def report(
244
249
  @click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
245
250
  @click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
246
251
  @click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
247
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml", "judge"]), help="Evidence scoring mode.")
252
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
248
253
  @click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
249
254
  @click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
250
255
  @click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
@@ -354,7 +359,7 @@ def inspect_command(trace_json: str, json_output: bool) -> int:
354
359
  @click.option("--json", "json_output", is_flag=True, help="Print the full QA result as JSON.")
355
360
  @click.option("--report", is_flag=True, help="Generate a local HTML evidence QA report.")
356
361
  @click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
357
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode.")
362
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
358
363
  @click.option("--fail-on", multiple=True, help="Fail on high_risk, medium_risk, any_risk, unsupported, should_abstain, audit_failure, or inspect_warning.")
359
364
  def qa_command(
360
365
  trace_json: str,
@@ -429,7 +434,7 @@ def suite_group() -> None:
429
434
  @click.argument("trace_json", nargs=-1, required=True)
430
435
  @click.option("--out", default="contexttrace-suite.json", show_default=True, help="Suite JSON file to write.")
431
436
  @click.option("--name", default=None, help="Suite name.")
432
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode for baseline QA.")
437
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for baseline QA.")
433
438
  @click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
434
439
  def suite_create_command(
435
440
  trace_json: tuple[str, ...],
@@ -461,7 +466,7 @@ def suite_create_command(
461
466
  @click.argument("suite_json")
462
467
  @click.argument("trace_json", nargs=-1, required=True)
463
468
  @click.option("--out", default=None, help="Suite JSON file to write. Defaults to overwriting suite_json.")
464
- @click.option("--mode", default=None, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode for added baselines. Defaults to the suite mode.")
469
+ @click.option("--mode", default=None, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for added baselines. Defaults to the suite mode.")
465
470
  @click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
466
471
  @click.option("--replace", is_flag=True, help="Replace existing cases with the same generated case IDs.")
467
472
  def suite_add_command(
@@ -600,7 +605,7 @@ def suite_prune_command(
600
605
  @click.option("--json", "json_output", is_flag=True, help="Print the full suite result as JSON.")
601
606
  @click.option("--report", is_flag=True, help="Generate a local HTML suite report.")
602
607
  @click.option("--report-out", default=None, help="HTML report path. Implies --report when provided.")
603
- @click.option("--mode", default=None, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode. Defaults to the suite mode.")
608
+ @click.option("--mode", default=None, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode. Defaults to the suite mode.")
604
609
  @click.option("--fail-on", multiple=True, help="Fail on failed_case, regression, unsupported, should_abstain, high_risk, medium_risk, error, or any_failure.")
605
610
  @click.pass_context
606
611
  def suite_run_command(
@@ -711,7 +716,7 @@ def suite_report_command(results_json: str, out: Optional[str]) -> int:
711
716
  @click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
712
717
  @click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
713
718
  @click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
714
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml", "judge"]), help="Evidence scoring mode.")
719
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
715
720
  @click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
716
721
  @click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
717
722
  @click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
@@ -769,7 +774,7 @@ def verify_demo_command(
769
774
 
770
775
 
771
776
  @cli.command("verify-benchmark")
772
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml", "judge"]), help="Evidence scoring mode.")
777
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
773
778
  @click.option("--case-set", default="contexttrace", show_default=True, type=click.Choice(["contexttrace", "external", "all"]), help="Benchmark case set to run.")
774
779
  @click.option("--json", "json_output", is_flag=True, help="Print benchmark results as JSON.")
775
780
  @click.option("--report", is_flag=True, help="Generate a local HTML benchmark report.")
@@ -926,8 +931,95 @@ def judge_calibrate_command(
926
931
  return 1 if result.get("failures") else 0
927
932
 
928
933
 
934
+ @cli.command("nli-calibrate")
935
+ @click.option("--case-set", default="all", show_default=True, type=click.Choice(["contexttrace", "external", "all"]), help="Golden benchmark case set to run.")
936
+ @click.option("--model-path", default=None, help="Local NLI model directory or .onnx file. Defaults to CONTEXTTRACE_NLI_MODEL_PATH.")
937
+ @click.option("--tokenizer-path", default=None, help="Optional local tokenizer directory. Defaults to model path.")
938
+ @click.option("--backend", default="auto", show_default=True, type=click.Choice(["auto", "transformers", "onnx"]), help="Local NLI backend.")
939
+ @click.option("--max-length", default=512, show_default=True, type=int, help="Maximum NLI input token length.")
940
+ @click.option("--json", "json_output", is_flag=True, help="Print calibration result as JSON.")
941
+ @click.option("--report", is_flag=True, help="Generate a local HTML calibration report.")
942
+ @click.option("--out", default=None, help="HTML calibration report path. Implies --report when provided.")
943
+ @click.option("--min-exact-match-rate", default=0.8, show_default=True, type=float, help="Minimum acceptable failure-label exact match rate.")
944
+ @click.option("--min-entailment-precision", default=0.85, show_default=True, type=float, help="Minimum acceptable supported-claim precision.")
945
+ @click.option("--min-contradiction-recall", default=0.7, show_default=True, type=float, help="Minimum acceptable contradiction recall.")
946
+ @click.option("--max-dangerous-miss-rate", default=0.05, show_default=True, type=float, help="Maximum allowed rate of risky cases predicted as no failure.")
947
+ @click.option("--max-p95-latency-ms", default=0.0, show_default=True, type=float, help="Maximum p95 case latency in milliseconds. Use 0 to disable.")
948
+ def nli_calibrate_command(
949
+ case_set: str,
950
+ model_path: Optional[str],
951
+ tokenizer_path: Optional[str],
952
+ backend: str,
953
+ max_length: int,
954
+ json_output: bool,
955
+ report: bool,
956
+ out: Optional[str],
957
+ min_exact_match_rate: float,
958
+ min_entailment_precision: float,
959
+ min_contradiction_recall: float,
960
+ max_dangerous_miss_rate: float,
961
+ max_p95_latency_ms: float,
962
+ ) -> int:
963
+ """Calibrate a local NLI model against golden RAG failure cases."""
964
+
965
+ result = run_nli_calibration(
966
+ model_path=model_path,
967
+ tokenizer_path=tokenizer_path,
968
+ backend=backend,
969
+ max_length=max_length,
970
+ case_set=case_set,
971
+ min_exact_match_rate=min_exact_match_rate,
972
+ min_entailment_precision=min_entailment_precision,
973
+ min_contradiction_recall=min_contradiction_recall,
974
+ max_dangerous_miss_rate=max_dangerous_miss_rate,
975
+ max_p95_latency_ms=max_p95_latency_ms,
976
+ )
977
+ written_report = None
978
+ if report or out:
979
+ nli_name = _safe_filename("%s_%s" % (
980
+ (result.get("nli") or {}).get("backend") or "nli",
981
+ (result.get("nli") or {}).get("model") or "model",
982
+ ))
983
+ output_path = out or str(Path(".contexttrace") / "reports" / ("nli_calibration_%s.html" % nli_name))
984
+ written_report = write_nli_calibration_report(result, path=output_path)
985
+ if json_output:
986
+ if written_report:
987
+ click.echo("Report: %s" % written_report, err=True)
988
+ click.echo(json.dumps(result, indent=2))
989
+ for failure in result.get("failures") or []:
990
+ click.echo("NLI calibration failed: %s" % failure, err=True)
991
+ return 1 if result.get("failures") else 0
992
+
993
+ scorecard = result["scorecard"]
994
+ click.echo("Status: %s" % result["status"])
995
+ click.echo("NLI: %s" % json.dumps(result.get("nli") or {}, sort_keys=True))
996
+ click.echo("Cases: %s" % result["cases"])
997
+ click.echo("Exact match rate: %.3f" % float(scorecard["exact_match_rate"]))
998
+ click.echo("Verdict match rate: %.3f" % float(scorecard["verdict_match_rate"]))
999
+ click.echo("Citation match rate: %.3f" % float(scorecard["citation_match_rate"]))
1000
+ click.echo("Entailment precision: %.3f" % float(scorecard["entailment_precision"]))
1001
+ click.echo("Entailment recall: %.3f" % float(scorecard["entailment_recall"]))
1002
+ click.echo("Contradiction recall: %.3f" % float(scorecard["contradiction_recall"]))
1003
+ click.echo("Unsupported-like recall: %.3f" % float(scorecard["unsupported_like_recall"]))
1004
+ click.echo("Dangerous miss rate: %.3f" % float(scorecard["dangerous_miss_rate"]))
1005
+ click.echo("Unsupported marked supported: %s" % int(scorecard["unsupported_as_supported_cases"]))
1006
+ click.echo("Case latency p50/p95 ms: %.3f / %.3f" % (
1007
+ float(scorecard["case_latency_p50_ms"]),
1008
+ float(scorecard["case_latency_p95_ms"]),
1009
+ ))
1010
+ click.echo("NLI call latency p50/p95 ms: %.3f / %.3f" % (
1011
+ float(scorecard["nli_call_latency_p50_ms"]),
1012
+ float(scorecard["nli_call_latency_p95_ms"]),
1013
+ ))
1014
+ if written_report:
1015
+ click.echo("Report: %s" % written_report)
1016
+ for failure in result.get("failures") or []:
1017
+ click.echo("NLI calibration failed: %s" % failure, err=True)
1018
+ return 1 if result.get("failures") else 0
1019
+
1020
+
929
1021
  @cli.command("audit-benchmark")
930
- @click.option("--mode", default="semantic", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode.")
1022
+ @click.option("--mode", default="semantic", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
931
1023
  @click.option("--case-set", default="real", show_default=True, type=click.Choice(["real"]), help="Benchmark case set to run.")
932
1024
  @click.option("--json", "json_output", is_flag=True, help="Print audit benchmark results as JSON.")
933
1025
  @click.option("--report", is_flag=True, help="Generate a local HTML audit benchmark report.")
@@ -987,7 +1079,7 @@ def audit_benchmark_command(mode: str, case_set: str, json_output: bool, report:
987
1079
  @click.option("--json", "json_output", is_flag=True, help="Print the full comparison result as JSON.")
988
1080
  @click.option("--report", is_flag=True, help="Generate a local HTML regression report.")
989
1081
  @click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
990
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode for raw trace inputs.")
1082
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for raw trace inputs.")
991
1083
  @click.option("--fail-on", multiple=True, help="Fail on new_failure, new_unsupported, new_citation_mismatch, should_abstain_flip, support_rate_drop, new_root_cause, or any_regression.")
992
1084
  def compare_command(
993
1085
  baseline_json: str,
@@ -1048,7 +1140,7 @@ def compare_command(
1048
1140
  @click.option("--json", "json_output", is_flag=True, help="Print the full audit result as JSON.")
1049
1141
  @click.option("--report", is_flag=True, help="Generate a local HTML retrieval audit report.")
1050
1142
  @click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
1051
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml"]), help="Evidence scoring mode.")
1143
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
1052
1144
  @click.option("--fail-on", multiple=True, help="Fail on retrieval_miss, reranking_failure, chunking_issue, corpus_gap, answer_overreach, stale_source, insufficient_context, or any_failure.")
1053
1145
  def audit_command(
1054
1146
  trace_json: str,
@@ -1132,7 +1224,7 @@ def capture_group() -> None:
1132
1224
  @click.option("--json", "json_output", is_flag=True, help="Print capture output as JSON.")
1133
1225
  @click.option("--report", is_flag=True, help="Generate a local HTML verification report. Implies --verify.")
1134
1226
  @click.option("--report-out", default=None, help="HTML report path. Implies --report and --verify when provided.")
1135
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml", "judge"]), help="Evidence scoring mode when verifying.")
1227
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode when verifying.")
1136
1228
  @click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
1137
1229
  @click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
1138
1230
  @click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
@@ -1221,7 +1313,7 @@ def capture_endpoint_command(
1221
1313
  @click.option("--json", "json_output", is_flag=True, help="Print capture output as JSON.")
1222
1314
  @click.option("--report", is_flag=True, help="Generate a local HTML verification report. Implies --verify.")
1223
1315
  @click.option("--report-out", default=None, help="HTML report path. Implies --report and --verify when provided.")
1224
- @click.option("--mode", default="lexical", show_default=True, type=click.Choice(["lexical", "semantic", "local_ml", "local-ml", "judge"]), help="Evidence scoring mode when verifying.")
1316
+ @click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode when verifying.")
1225
1317
  @click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
1226
1318
  @click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
1227
1319
  @click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
@@ -28,7 +28,37 @@ from contexttrace.verify.calibration import (
28
28
  run_judge_calibration,
29
29
  write_judge_calibration_report,
30
30
  )
31
+ from contexttrace.verify.nli_calibration import (
32
+ nli_calibration_failures,
33
+ run_nli_calibration,
34
+ write_nli_calibration_report,
35
+ )
31
36
  from contexttrace.verify.local_ml import LocalMLError, local_ml_similarity
37
+ from contexttrace.verify.local_nli import (
38
+ LocalNLIError,
39
+ LocalNLIJudge,
40
+ NLIResult,
41
+ build_nli_provider,
42
+ local_nli_entailment,
43
+ )
44
+ from contexttrace.verify.statuses import (
45
+ SOURCE_FRESHNESS_UNKNOWN,
46
+ TRUTH_NOT_ASSESSED,
47
+ attach_grounding_statuses,
48
+ source_status,
49
+ status_note,
50
+ support_status,
51
+ truth_status,
52
+ )
53
+ from contexttrace.verify.source_trust import (
54
+ GROUNDED_BUT_CONFLICTED,
55
+ GROUNDED_BUT_STALE,
56
+ GROUNDED_BY_LOW_AUTHORITY_SOURCE,
57
+ SUPPORTED_BY_CANONICAL_SOURCE,
58
+ attach_source_assessments,
59
+ source_assessment,
60
+ source_status_from_assessment,
61
+ )
32
62
 
33
63
  __all__ = [
34
64
  "CachedJudge",
@@ -37,17 +67,29 @@ __all__ = [
37
67
  "JudgeError",
38
68
  "JudgeVerdict",
39
69
  "LocalMLError",
70
+ "LocalNLIError",
71
+ "LocalNLIJudge",
72
+ "NLIResult",
40
73
  "OllamaJudge",
41
74
  "OpenAICompatibleJudge",
42
75
  "RAGTrace",
76
+ "SOURCE_FRESHNESS_UNKNOWN",
77
+ "GROUNDED_BUT_CONFLICTED",
78
+ "GROUNDED_BUT_STALE",
79
+ "GROUNDED_BY_LOW_AUTHORITY_SOURCE",
80
+ "SUPPORTED_BY_CANONICAL_SOURCE",
81
+ "TRUTH_NOT_ASSESSED",
43
82
  "TraceCitation",
44
83
  "TraceContext",
45
84
  "VerificationInputError",
46
85
  "audit_failures",
47
- "audit_trace",
86
+ "audit_trace",
48
87
  "audit_trace_file",
49
88
  "audit_trace_with_corpus",
89
+ "attach_grounding_statuses",
90
+ "attach_source_assessments",
50
91
  "build_judge_provider",
92
+ "build_nli_provider",
51
93
  "calibration_failures",
52
94
  "compare_failures",
53
95
  "compare_trace_files",
@@ -59,11 +101,21 @@ __all__ = [
59
101
  "load_trace_file",
60
102
  "load_verify_demo",
61
103
  "local_ml_similarity",
104
+ "local_nli_entailment",
62
105
  "qa_failures",
63
106
  "qa_trace",
64
107
  "run_judge_calibration",
108
+ "run_nli_calibration",
65
109
  "run_audit_benchmark",
110
+ "nli_calibration_failures",
111
+ "source_status",
112
+ "source_assessment",
113
+ "source_status_from_assessment",
114
+ "status_note",
115
+ "support_status",
116
+ "truth_status",
66
117
  "verify_trace",
67
118
  "verify_trace_file",
68
119
  "write_judge_calibration_report",
120
+ "write_nli_calibration_report",
69
121
  ]
@@ -1,6 +1,7 @@
1
1
  from __future__ import annotations
2
2
 
3
3
  import json
4
+ import time
4
5
  from dataclasses import dataclass
5
6
  from html import escape
6
7
  from pathlib import Path
@@ -47,6 +48,8 @@ def run_verify_benchmark(
47
48
  mode: str = "lexical",
48
49
  case_set: str = "contexttrace",
49
50
  judge: ClaimJudge | None = None,
51
+ nli: ClaimJudge | None = None,
52
+ time_cases: bool = False,
50
53
  ) -> dict[str, Any]:
51
54
  rows = []
52
55
  labels = set()
@@ -58,7 +61,9 @@ def run_verify_benchmark(
58
61
  abstention_expected = 0
59
62
 
60
63
  for case in benchmark_cases(case_set=case_set):
61
- result = verify_trace(case.trace, mode=mode, judge=judge)
64
+ started = time.perf_counter()
65
+ result = verify_trace(case.trace, mode=mode, judge=judge, nli=nli)
66
+ latency_ms = round((time.perf_counter() - started) * 1000, 3)
62
67
  predicted = _predicted_labels(result)
63
68
  expected_verdict_counts = dict(case.expected_verdict_counts)
64
69
  predicted_verdict_counts = {
@@ -112,6 +117,7 @@ def run_verify_benchmark(
112
117
  "summary": result.get("summary") or {},
113
118
  "claims": result.get("claims") or [],
114
119
  "abstention": result.get("abstention") or {},
120
+ **({"latency_ms": latency_ms} if time_cases else {}),
115
121
  }
116
122
  )
117
123
 
@@ -82,7 +82,8 @@ def find_citation_for_claim(
82
82
 
83
83
 
84
84
  def _source_fully_supports_claim(claim_text: str, match: object, *, mode: str) -> bool:
85
- fact_mode = "semantic" if mode == "semantic" else "lexical"
85
+ normalized_mode = str(mode or "").strip().lower().replace("-", "_")
86
+ fact_mode = "semantic" if normalized_mode in {"semantic", "local_ml", "nli"} else "lexical"
86
87
  fact_match = compare_facts(
87
88
  claim_text,
88
89
  str(getattr(match, "supporting_text", "") or getattr(match, "snippet", "")),
@@ -209,10 +209,11 @@ def score_claim_against_context(
209
209
  spans = split_context_spans(context)
210
210
  best_score = 0.0
211
211
  best_terms: list[str] = []
212
- best_snippet = context.text.strip()
213
- best_start: int | None = None
214
- best_end: int | None = None
215
- best_hash: str | None = None
212
+ fallback_span = spans[0] if spans else None
213
+ best_snippet = fallback_span.text.strip() if fallback_span is not None else context.text.strip()
214
+ best_start: int | None = fallback_span.start_char if fallback_span is not None else None
215
+ best_end: int | None = fallback_span.end_char if fallback_span is not None else None
216
+ best_hash: str | None = fallback_span.span_hash if fallback_span is not None else None
216
217
  span_candidates: list[dict[str, object]] = []
217
218
  for span in spans:
218
219
  score, terms = lexical_score(claim_text, span.text, mode=mode)
@@ -395,7 +396,7 @@ def _local_ml_enabled(mode: str) -> bool:
395
396
 
396
397
 
397
398
  def _semantic_enabled(mode: str) -> bool:
398
- return str(mode or "").strip().lower().replace("-", "_") in {"semantic", "local_ml"}
399
+ return str(mode or "").strip().lower().replace("-", "_") in {"semantic", "local_ml", "nli"}
399
400
 
400
401
 
401
402
  def _token_mode(mode: str) -> str:
@@ -35,11 +35,13 @@ DEFAULT_OPENAI_MODEL = "gpt-4.1-mini"
35
35
  DEFAULT_JUDGE_CACHE_PATH = ".contexttrace/judge_cache.json"
36
36
  JUDGE_SYSTEM_PROMPT = (
37
37
  "You are a strict RAG evidence judge. Given a user query, one generated claim, "
38
- "and retrieved evidence contexts, return only JSON with this schema: "
38
+ "and selected evidence spans, return only JSON with this schema: "
39
39
  "{\"verdict\":\"supported|partially_supported|unsupported|contradicted|unverifiable\","
40
40
  "\"confidence\":0.0,\"reason\":\"brief explanation\","
41
41
  "\"matched_facts\":[],\"missing_facts\":[],\"conflicting_facts\":[]}. "
42
42
  "Use supported only when the evidence directly entails every material part of the claim. "
43
+ "Treat supplied contexts as minimal evidence spans, not the full source document. "
44
+ "Do not infer support from omitted surrounding context or from fluent answer wording. "
43
45
  "Use contradicted when the evidence conflicts with the claim, including wrong entities, "
44
46
  "wrong dates, wrong numbers, negation conflicts, or reversed causal/attribution roles. "
45
47
  "Use partially_supported when some material facts are supported but others are missing. "
@@ -331,7 +333,7 @@ def judge_cache_key(
331
333
  payload: dict[str, Any],
332
334
  ) -> str:
333
335
  cache_payload = {
334
- "version": 1,
336
+ "version": 2,
335
337
  "provider": provider,
336
338
  "model": model or "",
337
339
  "system_prompt": JUDGE_SYSTEM_PROMPT,
@@ -417,6 +419,7 @@ def build_judge_provider(
417
419
 
418
420
  def _claim_payload(*, query: str, claim: str, contexts: list[TraceContext]) -> dict[str, Any]:
419
421
  return {
422
+ "evidence_scope": "selected_evidence_spans_only",
420
423
  "query": query,
421
424
  "claim": claim,
422
425
  "contexts": [