contexttrace 0.8.0__tar.gz → 0.9.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {contexttrace-0.8.0 → contexttrace-0.9.0}/PKG-INFO +31 -6
- {contexttrace-0.8.0 → contexttrace-0.9.0}/README.md +21 -5
- contexttrace-0.9.0/contexttrace/_version.py +1 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/cli.py +111 -19
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/__init__.py +53 -1
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/benchmark.py +7 -1
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/citations.py +2 -1
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/evidence.py +6 -5
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/judges.py +5 -2
- contexttrace-0.9.0/contexttrace/verify/local_nli.py +418 -0
- contexttrace-0.9.0/contexttrace/verify/nli_calibration.py +468 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/qa_report.py +41 -21
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/report.py +98 -4
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/root_cause.py +52 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/runner.py +163 -10
- contexttrace-0.9.0/contexttrace/verify/source_trust.py +340 -0
- contexttrace-0.9.0/contexttrace/verify/statuses.py +122 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/verdicts.py +1 -1
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace.egg-info/SOURCES.txt +4 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/pyproject.toml +14 -3
- contexttrace-0.8.0/contexttrace/_version.py +0 -1
- {contexttrace-0.8.0 → contexttrace-0.9.0}/MANIFEST.in +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/__init__.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/capture.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/capture_endpoint.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/client.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/config.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/demo.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/demo_data.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/endpoint_eval.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/errors.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/evaluator.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/__init__.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/fastapi.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/langchain.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/langgraph.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/llamaindex.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/integrations/opentelemetry.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/local.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/py.typed +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/regression.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/reliability.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/storage/__init__.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/storage/sqlite_store.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/thresholds.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/transport.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/abstention.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/audit_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/calibration.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/claims.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/compare.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/compare_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/demos.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/facts.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/local_ml.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/qa.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/schema.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/spans.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/suite.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/suite_report.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/verify/trace_inspect.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/contexttrace/viewer.py +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/setup.cfg +0 -0
- {contexttrace-0.8.0 → contexttrace-0.9.0}/setup.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.1
|
|
2
2
|
Name: contexttrace
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.9.0
|
|
4
4
|
Summary: Local-first SDK and CLI for RAG and agent reliability tracing, citation checks, and failure diagnosis.
|
|
5
5
|
Author: ContextTrace contributors
|
|
6
6
|
License: MIT
|
|
@@ -34,6 +34,12 @@ Requires-Dist: llama-index-core>=0.10; extra == "llamaindex"
|
|
|
34
34
|
Provides-Extra: local
|
|
35
35
|
Provides-Extra: local-ml
|
|
36
36
|
Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
|
|
37
|
+
Provides-Extra: nli
|
|
38
|
+
Requires-Dist: torch>=2.0; extra == "nli"
|
|
39
|
+
Requires-Dist: transformers>=4.41; extra == "nli"
|
|
40
|
+
Provides-Extra: nli-onnx
|
|
41
|
+
Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
|
|
42
|
+
Requires-Dist: transformers>=4.41; extra == "nli-onnx"
|
|
37
43
|
Provides-Extra: fastapi
|
|
38
44
|
Requires-Dist: fastapi>=0.110; extra == "fastapi"
|
|
39
45
|
Provides-Extra: langgraph
|
|
@@ -55,6 +61,9 @@ Requires-Dist: langgraph>=0.2; extra == "all"
|
|
|
55
61
|
Requires-Dist: llama-index-core>=0.10; extra == "all"
|
|
56
62
|
Requires-Dist: opentelemetry-api>=1.24; extra == "all"
|
|
57
63
|
Requires-Dist: sentence-transformers>=2.7; extra == "all"
|
|
64
|
+
Requires-Dist: torch>=2.0; extra == "all"
|
|
65
|
+
Requires-Dist: transformers>=4.41; extra == "all"
|
|
66
|
+
Requires-Dist: onnxruntime>=1.17; extra == "all"
|
|
58
67
|
Provides-Extra: test
|
|
59
68
|
Requires-Dist: pytest>=8.0; extra == "test"
|
|
60
69
|
|
|
@@ -62,7 +71,7 @@ Requires-Dist: pytest>=8.0; extra == "test"
|
|
|
62
71
|
|
|
63
72
|
**Local-first evidence-chain debugging for RAG and AI agents.**
|
|
64
73
|
|
|
65
|
-
ContextTrace shows where an answer stopped being grounded:
|
|
74
|
+
ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
|
|
66
75
|
|
|
67
76
|
```text
|
|
68
77
|
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
@@ -116,7 +125,9 @@ contexttrace verify trace.json --report
|
|
|
116
125
|
contexttrace qa trace.json --corpus docs/ --report
|
|
117
126
|
```
|
|
118
127
|
|
|
119
|
-
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then
|
|
128
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
|
|
129
|
+
|
|
130
|
+
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
120
131
|
|
|
121
132
|
## Local Verification Modes
|
|
122
133
|
|
|
@@ -125,7 +136,8 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
|
|
|
125
136
|
| `lexical` | Fast default checks with no optional dependencies. |
|
|
126
137
|
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
127
138
|
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
128
|
-
| `
|
|
139
|
+
| `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
|
|
140
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
|
|
129
141
|
|
|
130
142
|
Run the stronger local non-LLM verifier:
|
|
131
143
|
|
|
@@ -138,7 +150,16 @@ Optional neural local-ML support never downloads models automatically:
|
|
|
138
150
|
|
|
139
151
|
```bash
|
|
140
152
|
pip install "contexttrace[local-ml]"
|
|
141
|
-
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C
|
|
153
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
|
|
154
|
+
```
|
|
155
|
+
|
|
156
|
+
Run local NLI when you want mechanical claim-versus-span entailment:
|
|
157
|
+
|
|
158
|
+
```bash
|
|
159
|
+
pip install "contexttrace[nli]"
|
|
160
|
+
set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
|
|
161
|
+
contexttrace verify trace.json --mode nli --report
|
|
162
|
+
contexttrace nli-calibrate --case-set all --report
|
|
142
163
|
```
|
|
143
164
|
|
|
144
165
|
Run a local judge with Ollama:
|
|
@@ -169,6 +190,10 @@ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/
|
|
|
169
190
|
|
|
170
191
|
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
171
192
|
|
|
193
|
+
`support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
|
|
194
|
+
|
|
195
|
+
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
196
|
+
|
|
172
197
|
## Capture Existing Systems
|
|
173
198
|
|
|
174
199
|
Capture one live endpoint response:
|
|
@@ -247,7 +272,7 @@ ContextTrace makes no network calls unless you point it at an endpoint or config
|
|
|
247
272
|
|
|
248
273
|
## Limits
|
|
249
274
|
|
|
250
|
-
ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
275
|
+
ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
251
276
|
|
|
252
277
|
## Links
|
|
253
278
|
|
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
**Local-first evidence-chain debugging for RAG and AI agents.**
|
|
4
4
|
|
|
5
|
-
ContextTrace shows where an answer stopped being grounded:
|
|
5
|
+
ContextTrace shows where an answer stopped being grounded in the evidence you gave it:
|
|
6
6
|
|
|
7
7
|
```text
|
|
8
8
|
query -> retrieved context -> answer claims -> citations -> verdicts -> root cause
|
|
@@ -56,7 +56,9 @@ contexttrace verify trace.json --report
|
|
|
56
56
|
contexttrace qa trace.json --corpus docs/ --report
|
|
57
57
|
```
|
|
58
58
|
|
|
59
|
-
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then
|
|
59
|
+
ContextTrace classifies each claim as `supported`, `partially_supported`, `unsupported`, `unverifiable`, or `contradicted`, then exposes separate statuses for support, truth, source freshness, citation quality, and likely fix.
|
|
60
|
+
|
|
61
|
+
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
60
62
|
|
|
61
63
|
## Local Verification Modes
|
|
62
64
|
|
|
@@ -65,7 +67,8 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
|
|
|
65
67
|
| `lexical` | Fast default checks with no optional dependencies. |
|
|
66
68
|
| `semantic` | Local paraphrase and role-aware contradiction checks. |
|
|
67
69
|
| `local_ml` | Offline hash-embedding similarity, optionally backed by a local SentenceTransformers model. |
|
|
68
|
-
| `
|
|
70
|
+
| `nli` | Local claim+span entailment or contradiction with a local Transformers or ONNX NLI model. |
|
|
71
|
+
| `judge` | Higher-accuracy local LLM judging through Ollama, LM Studio, vLLM, or a local OpenAI-compatible server. The judge sees selected evidence spans, not the full answer prose. |
|
|
69
72
|
|
|
70
73
|
Run the stronger local non-LLM verifier:
|
|
71
74
|
|
|
@@ -78,7 +81,16 @@ Optional neural local-ML support never downloads models automatically:
|
|
|
78
81
|
|
|
79
82
|
```bash
|
|
80
83
|
pip install "contexttrace[local-ml]"
|
|
81
|
-
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C
|
|
84
|
+
set CONTEXTTRACE_LOCAL_ML_MODEL_PATH=C:/models/bge-small-en-v1.5
|
|
85
|
+
```
|
|
86
|
+
|
|
87
|
+
Run local NLI when you want mechanical claim-versus-span entailment:
|
|
88
|
+
|
|
89
|
+
```bash
|
|
90
|
+
pip install "contexttrace[nli]"
|
|
91
|
+
set CONTEXTTRACE_NLI_MODEL_PATH=C:/models/deberta-v3-nli
|
|
92
|
+
contexttrace verify trace.json --mode nli --report
|
|
93
|
+
contexttrace nli-calibrate --case-set all --report
|
|
82
94
|
```
|
|
83
95
|
|
|
84
96
|
Run a local judge with Ollama:
|
|
@@ -109,6 +121,10 @@ contexttrace suite run contexttrace-suite.json --endpoint http://localhost:8000/
|
|
|
109
121
|
|
|
110
122
|
Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issue`, `corpus_gap`, `answer_overreach`, `stale_source`, `citation_mismatch`, and `should_have_abstained`.
|
|
111
123
|
|
|
124
|
+
`support_status`, `truth_status`, and `source_status` stay separate so a claim can be grounded by a source while the source itself remains stale, wrong, or unassessed.
|
|
125
|
+
|
|
126
|
+
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
127
|
+
|
|
112
128
|
## Capture Existing Systems
|
|
113
129
|
|
|
114
130
|
Capture one live endpoint response:
|
|
@@ -187,7 +203,7 @@ ContextTrace makes no network calls unless you point it at an endpoint or config
|
|
|
187
203
|
|
|
188
204
|
## Limits
|
|
189
205
|
|
|
190
|
-
ContextTrace is a diagnostic tool, not a correctness proof. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
206
|
+
ContextTrace is a diagnostic tool, not a correctness proof. It verifies grounding against provided evidence; it does not certify real-world truth. Claim extraction is rule-based, contradiction detection is conservative, and high-stakes outputs still need human review.
|
|
191
207
|
|
|
192
208
|
## Links
|
|
193
209
|
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.9.0"
|
|
@@ -31,6 +31,7 @@ from contexttrace.verify import (
|
|
|
31
31
|
audit_trace,
|
|
32
32
|
build_judge_provider,
|
|
33
33
|
run_judge_calibration,
|
|
34
|
+
run_nli_calibration,
|
|
34
35
|
compare_failures,
|
|
35
36
|
compare_trace_files,
|
|
36
37
|
list_verify_demos,
|
|
@@ -38,6 +39,7 @@ from contexttrace.verify import (
|
|
|
38
39
|
load_verify_demo,
|
|
39
40
|
verify_trace,
|
|
40
41
|
write_judge_calibration_report,
|
|
42
|
+
write_nli_calibration_report,
|
|
41
43
|
)
|
|
42
44
|
from contexttrace.verify.benchmark import run_verify_benchmark, write_verify_benchmark_report
|
|
43
45
|
from contexttrace.verify.audit_benchmark import run_audit_benchmark, write_audit_benchmark_report
|
|
@@ -64,13 +66,16 @@ from contexttrace.verify.trace_inspect import inspect_trace
|
|
|
64
66
|
from contexttrace.viewer import serve_viewer
|
|
65
67
|
|
|
66
68
|
|
|
67
|
-
SAMPLE_QUESTIONS = [
|
|
68
|
-
{
|
|
69
|
-
"id": "refund_policy",
|
|
70
|
-
"query": "What is the refund policy?",
|
|
71
|
-
"expected_sources": ["refund_policy.md"],
|
|
72
|
-
}
|
|
73
|
-
]
|
|
69
|
+
SAMPLE_QUESTIONS = [
|
|
70
|
+
{
|
|
71
|
+
"id": "refund_policy",
|
|
72
|
+
"query": "What is the refund policy?",
|
|
73
|
+
"expected_sources": ["refund_policy.md"],
|
|
74
|
+
}
|
|
75
|
+
]
|
|
76
|
+
|
|
77
|
+
BASIC_VERIFY_MODES = ["lexical", "semantic", "local_ml", "local-ml", "nli"]
|
|
78
|
+
JUDGE_VERIFY_MODES = [*BASIC_VERIFY_MODES, "judge"]
|
|
74
79
|
|
|
75
80
|
|
|
76
81
|
@click.group(context_settings={"help_option_names": ["-h", "--help"]})
|
|
@@ -244,7 +249,7 @@ def report(
|
|
|
244
249
|
@click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
|
|
245
250
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
|
|
246
251
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
247
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
252
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
|
|
248
253
|
@click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
|
|
249
254
|
@click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
|
|
250
255
|
@click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
|
|
@@ -354,7 +359,7 @@ def inspect_command(trace_json: str, json_output: bool) -> int:
|
|
|
354
359
|
@click.option("--json", "json_output", is_flag=True, help="Print the full QA result as JSON.")
|
|
355
360
|
@click.option("--report", is_flag=True, help="Generate a local HTML evidence QA report.")
|
|
356
361
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
357
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
362
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
|
|
358
363
|
@click.option("--fail-on", multiple=True, help="Fail on high_risk, medium_risk, any_risk, unsupported, should_abstain, audit_failure, or inspect_warning.")
|
|
359
364
|
def qa_command(
|
|
360
365
|
trace_json: str,
|
|
@@ -429,7 +434,7 @@ def suite_group() -> None:
|
|
|
429
434
|
@click.argument("trace_json", nargs=-1, required=True)
|
|
430
435
|
@click.option("--out", default="contexttrace-suite.json", show_default=True, help="Suite JSON file to write.")
|
|
431
436
|
@click.option("--name", default=None, help="Suite name.")
|
|
432
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
437
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for baseline QA.")
|
|
433
438
|
@click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
|
|
434
439
|
def suite_create_command(
|
|
435
440
|
trace_json: tuple[str, ...],
|
|
@@ -461,7 +466,7 @@ def suite_create_command(
|
|
|
461
466
|
@click.argument("suite_json")
|
|
462
467
|
@click.argument("trace_json", nargs=-1, required=True)
|
|
463
468
|
@click.option("--out", default=None, help="Suite JSON file to write. Defaults to overwriting suite_json.")
|
|
464
|
-
@click.option("--mode", default=None, type=click.Choice(
|
|
469
|
+
@click.option("--mode", default=None, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for added baselines. Defaults to the suite mode.")
|
|
465
470
|
@click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
|
|
466
471
|
@click.option("--replace", is_flag=True, help="Replace existing cases with the same generated case IDs.")
|
|
467
472
|
def suite_add_command(
|
|
@@ -600,7 +605,7 @@ def suite_prune_command(
|
|
|
600
605
|
@click.option("--json", "json_output", is_flag=True, help="Print the full suite result as JSON.")
|
|
601
606
|
@click.option("--report", is_flag=True, help="Generate a local HTML suite report.")
|
|
602
607
|
@click.option("--report-out", default=None, help="HTML report path. Implies --report when provided.")
|
|
603
|
-
@click.option("--mode", default=None, type=click.Choice(
|
|
608
|
+
@click.option("--mode", default=None, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode. Defaults to the suite mode.")
|
|
604
609
|
@click.option("--fail-on", multiple=True, help="Fail on failed_case, regression, unsupported, should_abstain, high_risk, medium_risk, error, or any_failure.")
|
|
605
610
|
@click.pass_context
|
|
606
611
|
def suite_run_command(
|
|
@@ -711,7 +716,7 @@ def suite_report_command(results_json: str, out: Optional[str]) -> int:
|
|
|
711
716
|
@click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
|
|
712
717
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
|
|
713
718
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
714
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
719
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
|
|
715
720
|
@click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
|
|
716
721
|
@click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
|
|
717
722
|
@click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
|
|
@@ -769,7 +774,7 @@ def verify_demo_command(
|
|
|
769
774
|
|
|
770
775
|
|
|
771
776
|
@cli.command("verify-benchmark")
|
|
772
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
777
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
|
|
773
778
|
@click.option("--case-set", default="contexttrace", show_default=True, type=click.Choice(["contexttrace", "external", "all"]), help="Benchmark case set to run.")
|
|
774
779
|
@click.option("--json", "json_output", is_flag=True, help="Print benchmark results as JSON.")
|
|
775
780
|
@click.option("--report", is_flag=True, help="Generate a local HTML benchmark report.")
|
|
@@ -926,8 +931,95 @@ def judge_calibrate_command(
|
|
|
926
931
|
return 1 if result.get("failures") else 0
|
|
927
932
|
|
|
928
933
|
|
|
934
|
+
@cli.command("nli-calibrate")
|
|
935
|
+
@click.option("--case-set", default="all", show_default=True, type=click.Choice(["contexttrace", "external", "all"]), help="Golden benchmark case set to run.")
|
|
936
|
+
@click.option("--model-path", default=None, help="Local NLI model directory or .onnx file. Defaults to CONTEXTTRACE_NLI_MODEL_PATH.")
|
|
937
|
+
@click.option("--tokenizer-path", default=None, help="Optional local tokenizer directory. Defaults to model path.")
|
|
938
|
+
@click.option("--backend", default="auto", show_default=True, type=click.Choice(["auto", "transformers", "onnx"]), help="Local NLI backend.")
|
|
939
|
+
@click.option("--max-length", default=512, show_default=True, type=int, help="Maximum NLI input token length.")
|
|
940
|
+
@click.option("--json", "json_output", is_flag=True, help="Print calibration result as JSON.")
|
|
941
|
+
@click.option("--report", is_flag=True, help="Generate a local HTML calibration report.")
|
|
942
|
+
@click.option("--out", default=None, help="HTML calibration report path. Implies --report when provided.")
|
|
943
|
+
@click.option("--min-exact-match-rate", default=0.8, show_default=True, type=float, help="Minimum acceptable failure-label exact match rate.")
|
|
944
|
+
@click.option("--min-entailment-precision", default=0.85, show_default=True, type=float, help="Minimum acceptable supported-claim precision.")
|
|
945
|
+
@click.option("--min-contradiction-recall", default=0.7, show_default=True, type=float, help="Minimum acceptable contradiction recall.")
|
|
946
|
+
@click.option("--max-dangerous-miss-rate", default=0.05, show_default=True, type=float, help="Maximum allowed rate of risky cases predicted as no failure.")
|
|
947
|
+
@click.option("--max-p95-latency-ms", default=0.0, show_default=True, type=float, help="Maximum p95 case latency in milliseconds. Use 0 to disable.")
|
|
948
|
+
def nli_calibrate_command(
|
|
949
|
+
case_set: str,
|
|
950
|
+
model_path: Optional[str],
|
|
951
|
+
tokenizer_path: Optional[str],
|
|
952
|
+
backend: str,
|
|
953
|
+
max_length: int,
|
|
954
|
+
json_output: bool,
|
|
955
|
+
report: bool,
|
|
956
|
+
out: Optional[str],
|
|
957
|
+
min_exact_match_rate: float,
|
|
958
|
+
min_entailment_precision: float,
|
|
959
|
+
min_contradiction_recall: float,
|
|
960
|
+
max_dangerous_miss_rate: float,
|
|
961
|
+
max_p95_latency_ms: float,
|
|
962
|
+
) -> int:
|
|
963
|
+
"""Calibrate a local NLI model against golden RAG failure cases."""
|
|
964
|
+
|
|
965
|
+
result = run_nli_calibration(
|
|
966
|
+
model_path=model_path,
|
|
967
|
+
tokenizer_path=tokenizer_path,
|
|
968
|
+
backend=backend,
|
|
969
|
+
max_length=max_length,
|
|
970
|
+
case_set=case_set,
|
|
971
|
+
min_exact_match_rate=min_exact_match_rate,
|
|
972
|
+
min_entailment_precision=min_entailment_precision,
|
|
973
|
+
min_contradiction_recall=min_contradiction_recall,
|
|
974
|
+
max_dangerous_miss_rate=max_dangerous_miss_rate,
|
|
975
|
+
max_p95_latency_ms=max_p95_latency_ms,
|
|
976
|
+
)
|
|
977
|
+
written_report = None
|
|
978
|
+
if report or out:
|
|
979
|
+
nli_name = _safe_filename("%s_%s" % (
|
|
980
|
+
(result.get("nli") or {}).get("backend") or "nli",
|
|
981
|
+
(result.get("nli") or {}).get("model") or "model",
|
|
982
|
+
))
|
|
983
|
+
output_path = out or str(Path(".contexttrace") / "reports" / ("nli_calibration_%s.html" % nli_name))
|
|
984
|
+
written_report = write_nli_calibration_report(result, path=output_path)
|
|
985
|
+
if json_output:
|
|
986
|
+
if written_report:
|
|
987
|
+
click.echo("Report: %s" % written_report, err=True)
|
|
988
|
+
click.echo(json.dumps(result, indent=2))
|
|
989
|
+
for failure in result.get("failures") or []:
|
|
990
|
+
click.echo("NLI calibration failed: %s" % failure, err=True)
|
|
991
|
+
return 1 if result.get("failures") else 0
|
|
992
|
+
|
|
993
|
+
scorecard = result["scorecard"]
|
|
994
|
+
click.echo("Status: %s" % result["status"])
|
|
995
|
+
click.echo("NLI: %s" % json.dumps(result.get("nli") or {}, sort_keys=True))
|
|
996
|
+
click.echo("Cases: %s" % result["cases"])
|
|
997
|
+
click.echo("Exact match rate: %.3f" % float(scorecard["exact_match_rate"]))
|
|
998
|
+
click.echo("Verdict match rate: %.3f" % float(scorecard["verdict_match_rate"]))
|
|
999
|
+
click.echo("Citation match rate: %.3f" % float(scorecard["citation_match_rate"]))
|
|
1000
|
+
click.echo("Entailment precision: %.3f" % float(scorecard["entailment_precision"]))
|
|
1001
|
+
click.echo("Entailment recall: %.3f" % float(scorecard["entailment_recall"]))
|
|
1002
|
+
click.echo("Contradiction recall: %.3f" % float(scorecard["contradiction_recall"]))
|
|
1003
|
+
click.echo("Unsupported-like recall: %.3f" % float(scorecard["unsupported_like_recall"]))
|
|
1004
|
+
click.echo("Dangerous miss rate: %.3f" % float(scorecard["dangerous_miss_rate"]))
|
|
1005
|
+
click.echo("Unsupported marked supported: %s" % int(scorecard["unsupported_as_supported_cases"]))
|
|
1006
|
+
click.echo("Case latency p50/p95 ms: %.3f / %.3f" % (
|
|
1007
|
+
float(scorecard["case_latency_p50_ms"]),
|
|
1008
|
+
float(scorecard["case_latency_p95_ms"]),
|
|
1009
|
+
))
|
|
1010
|
+
click.echo("NLI call latency p50/p95 ms: %.3f / %.3f" % (
|
|
1011
|
+
float(scorecard["nli_call_latency_p50_ms"]),
|
|
1012
|
+
float(scorecard["nli_call_latency_p95_ms"]),
|
|
1013
|
+
))
|
|
1014
|
+
if written_report:
|
|
1015
|
+
click.echo("Report: %s" % written_report)
|
|
1016
|
+
for failure in result.get("failures") or []:
|
|
1017
|
+
click.echo("NLI calibration failed: %s" % failure, err=True)
|
|
1018
|
+
return 1 if result.get("failures") else 0
|
|
1019
|
+
|
|
1020
|
+
|
|
929
1021
|
@cli.command("audit-benchmark")
|
|
930
|
-
@click.option("--mode", default="semantic", show_default=True, type=click.Choice(
|
|
1022
|
+
@click.option("--mode", default="semantic", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
|
|
931
1023
|
@click.option("--case-set", default="real", show_default=True, type=click.Choice(["real"]), help="Benchmark case set to run.")
|
|
932
1024
|
@click.option("--json", "json_output", is_flag=True, help="Print audit benchmark results as JSON.")
|
|
933
1025
|
@click.option("--report", is_flag=True, help="Generate a local HTML audit benchmark report.")
|
|
@@ -987,7 +1079,7 @@ def audit_benchmark_command(mode: str, case_set: str, json_output: bool, report:
|
|
|
987
1079
|
@click.option("--json", "json_output", is_flag=True, help="Print the full comparison result as JSON.")
|
|
988
1080
|
@click.option("--report", is_flag=True, help="Generate a local HTML regression report.")
|
|
989
1081
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
990
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
1082
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for raw trace inputs.")
|
|
991
1083
|
@click.option("--fail-on", multiple=True, help="Fail on new_failure, new_unsupported, new_citation_mismatch, should_abstain_flip, support_rate_drop, new_root_cause, or any_regression.")
|
|
992
1084
|
def compare_command(
|
|
993
1085
|
baseline_json: str,
|
|
@@ -1048,7 +1140,7 @@ def compare_command(
|
|
|
1048
1140
|
@click.option("--json", "json_output", is_flag=True, help="Print the full audit result as JSON.")
|
|
1049
1141
|
@click.option("--report", is_flag=True, help="Generate a local HTML retrieval audit report.")
|
|
1050
1142
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
1051
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
1143
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode.")
|
|
1052
1144
|
@click.option("--fail-on", multiple=True, help="Fail on retrieval_miss, reranking_failure, chunking_issue, corpus_gap, answer_overreach, stale_source, insufficient_context, or any_failure.")
|
|
1053
1145
|
def audit_command(
|
|
1054
1146
|
trace_json: str,
|
|
@@ -1132,7 +1224,7 @@ def capture_group() -> None:
|
|
|
1132
1224
|
@click.option("--json", "json_output", is_flag=True, help="Print capture output as JSON.")
|
|
1133
1225
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report. Implies --verify.")
|
|
1134
1226
|
@click.option("--report-out", default=None, help="HTML report path. Implies --report and --verify when provided.")
|
|
1135
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
1227
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode when verifying.")
|
|
1136
1228
|
@click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
|
|
1137
1229
|
@click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
|
|
1138
1230
|
@click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
|
|
@@ -1221,7 +1313,7 @@ def capture_endpoint_command(
|
|
|
1221
1313
|
@click.option("--json", "json_output", is_flag=True, help="Print capture output as JSON.")
|
|
1222
1314
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report. Implies --verify.")
|
|
1223
1315
|
@click.option("--report-out", default=None, help="HTML report path. Implies --report and --verify when provided.")
|
|
1224
|
-
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(
|
|
1316
|
+
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode when verifying.")
|
|
1225
1317
|
@click.option("--judge-provider", default=None, help="Judge provider for --mode judge, for example ollama or local_openai.")
|
|
1226
1318
|
@click.option("--judge-base-url", default=None, help="Judge base URL. Defaults are local for ollama/local_openai/lmstudio/vllm.")
|
|
1227
1319
|
@click.option("--judge-api-key", default=None, help="Judge API key. Optional for local providers; prefer CONTEXTTRACE_JUDGE_API_KEY in CI.")
|
|
@@ -28,7 +28,37 @@ from contexttrace.verify.calibration import (
|
|
|
28
28
|
run_judge_calibration,
|
|
29
29
|
write_judge_calibration_report,
|
|
30
30
|
)
|
|
31
|
+
from contexttrace.verify.nli_calibration import (
|
|
32
|
+
nli_calibration_failures,
|
|
33
|
+
run_nli_calibration,
|
|
34
|
+
write_nli_calibration_report,
|
|
35
|
+
)
|
|
31
36
|
from contexttrace.verify.local_ml import LocalMLError, local_ml_similarity
|
|
37
|
+
from contexttrace.verify.local_nli import (
|
|
38
|
+
LocalNLIError,
|
|
39
|
+
LocalNLIJudge,
|
|
40
|
+
NLIResult,
|
|
41
|
+
build_nli_provider,
|
|
42
|
+
local_nli_entailment,
|
|
43
|
+
)
|
|
44
|
+
from contexttrace.verify.statuses import (
|
|
45
|
+
SOURCE_FRESHNESS_UNKNOWN,
|
|
46
|
+
TRUTH_NOT_ASSESSED,
|
|
47
|
+
attach_grounding_statuses,
|
|
48
|
+
source_status,
|
|
49
|
+
status_note,
|
|
50
|
+
support_status,
|
|
51
|
+
truth_status,
|
|
52
|
+
)
|
|
53
|
+
from contexttrace.verify.source_trust import (
|
|
54
|
+
GROUNDED_BUT_CONFLICTED,
|
|
55
|
+
GROUNDED_BUT_STALE,
|
|
56
|
+
GROUNDED_BY_LOW_AUTHORITY_SOURCE,
|
|
57
|
+
SUPPORTED_BY_CANONICAL_SOURCE,
|
|
58
|
+
attach_source_assessments,
|
|
59
|
+
source_assessment,
|
|
60
|
+
source_status_from_assessment,
|
|
61
|
+
)
|
|
32
62
|
|
|
33
63
|
__all__ = [
|
|
34
64
|
"CachedJudge",
|
|
@@ -37,17 +67,29 @@ __all__ = [
|
|
|
37
67
|
"JudgeError",
|
|
38
68
|
"JudgeVerdict",
|
|
39
69
|
"LocalMLError",
|
|
70
|
+
"LocalNLIError",
|
|
71
|
+
"LocalNLIJudge",
|
|
72
|
+
"NLIResult",
|
|
40
73
|
"OllamaJudge",
|
|
41
74
|
"OpenAICompatibleJudge",
|
|
42
75
|
"RAGTrace",
|
|
76
|
+
"SOURCE_FRESHNESS_UNKNOWN",
|
|
77
|
+
"GROUNDED_BUT_CONFLICTED",
|
|
78
|
+
"GROUNDED_BUT_STALE",
|
|
79
|
+
"GROUNDED_BY_LOW_AUTHORITY_SOURCE",
|
|
80
|
+
"SUPPORTED_BY_CANONICAL_SOURCE",
|
|
81
|
+
"TRUTH_NOT_ASSESSED",
|
|
43
82
|
"TraceCitation",
|
|
44
83
|
"TraceContext",
|
|
45
84
|
"VerificationInputError",
|
|
46
85
|
"audit_failures",
|
|
47
|
-
"audit_trace",
|
|
86
|
+
"audit_trace",
|
|
48
87
|
"audit_trace_file",
|
|
49
88
|
"audit_trace_with_corpus",
|
|
89
|
+
"attach_grounding_statuses",
|
|
90
|
+
"attach_source_assessments",
|
|
50
91
|
"build_judge_provider",
|
|
92
|
+
"build_nli_provider",
|
|
51
93
|
"calibration_failures",
|
|
52
94
|
"compare_failures",
|
|
53
95
|
"compare_trace_files",
|
|
@@ -59,11 +101,21 @@ __all__ = [
|
|
|
59
101
|
"load_trace_file",
|
|
60
102
|
"load_verify_demo",
|
|
61
103
|
"local_ml_similarity",
|
|
104
|
+
"local_nli_entailment",
|
|
62
105
|
"qa_failures",
|
|
63
106
|
"qa_trace",
|
|
64
107
|
"run_judge_calibration",
|
|
108
|
+
"run_nli_calibration",
|
|
65
109
|
"run_audit_benchmark",
|
|
110
|
+
"nli_calibration_failures",
|
|
111
|
+
"source_status",
|
|
112
|
+
"source_assessment",
|
|
113
|
+
"source_status_from_assessment",
|
|
114
|
+
"status_note",
|
|
115
|
+
"support_status",
|
|
116
|
+
"truth_status",
|
|
66
117
|
"verify_trace",
|
|
67
118
|
"verify_trace_file",
|
|
68
119
|
"write_judge_calibration_report",
|
|
120
|
+
"write_nli_calibration_report",
|
|
69
121
|
]
|
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
3
|
import json
|
|
4
|
+
import time
|
|
4
5
|
from dataclasses import dataclass
|
|
5
6
|
from html import escape
|
|
6
7
|
from pathlib import Path
|
|
@@ -47,6 +48,8 @@ def run_verify_benchmark(
|
|
|
47
48
|
mode: str = "lexical",
|
|
48
49
|
case_set: str = "contexttrace",
|
|
49
50
|
judge: ClaimJudge | None = None,
|
|
51
|
+
nli: ClaimJudge | None = None,
|
|
52
|
+
time_cases: bool = False,
|
|
50
53
|
) -> dict[str, Any]:
|
|
51
54
|
rows = []
|
|
52
55
|
labels = set()
|
|
@@ -58,7 +61,9 @@ def run_verify_benchmark(
|
|
|
58
61
|
abstention_expected = 0
|
|
59
62
|
|
|
60
63
|
for case in benchmark_cases(case_set=case_set):
|
|
61
|
-
|
|
64
|
+
started = time.perf_counter()
|
|
65
|
+
result = verify_trace(case.trace, mode=mode, judge=judge, nli=nli)
|
|
66
|
+
latency_ms = round((time.perf_counter() - started) * 1000, 3)
|
|
62
67
|
predicted = _predicted_labels(result)
|
|
63
68
|
expected_verdict_counts = dict(case.expected_verdict_counts)
|
|
64
69
|
predicted_verdict_counts = {
|
|
@@ -112,6 +117,7 @@ def run_verify_benchmark(
|
|
|
112
117
|
"summary": result.get("summary") or {},
|
|
113
118
|
"claims": result.get("claims") or [],
|
|
114
119
|
"abstention": result.get("abstention") or {},
|
|
120
|
+
**({"latency_ms": latency_ms} if time_cases else {}),
|
|
115
121
|
}
|
|
116
122
|
)
|
|
117
123
|
|
|
@@ -82,7 +82,8 @@ def find_citation_for_claim(
|
|
|
82
82
|
|
|
83
83
|
|
|
84
84
|
def _source_fully_supports_claim(claim_text: str, match: object, *, mode: str) -> bool:
|
|
85
|
-
|
|
85
|
+
normalized_mode = str(mode or "").strip().lower().replace("-", "_")
|
|
86
|
+
fact_mode = "semantic" if normalized_mode in {"semantic", "local_ml", "nli"} else "lexical"
|
|
86
87
|
fact_match = compare_facts(
|
|
87
88
|
claim_text,
|
|
88
89
|
str(getattr(match, "supporting_text", "") or getattr(match, "snippet", "")),
|
|
@@ -209,10 +209,11 @@ def score_claim_against_context(
|
|
|
209
209
|
spans = split_context_spans(context)
|
|
210
210
|
best_score = 0.0
|
|
211
211
|
best_terms: list[str] = []
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
212
|
+
fallback_span = spans[0] if spans else None
|
|
213
|
+
best_snippet = fallback_span.text.strip() if fallback_span is not None else context.text.strip()
|
|
214
|
+
best_start: int | None = fallback_span.start_char if fallback_span is not None else None
|
|
215
|
+
best_end: int | None = fallback_span.end_char if fallback_span is not None else None
|
|
216
|
+
best_hash: str | None = fallback_span.span_hash if fallback_span is not None else None
|
|
216
217
|
span_candidates: list[dict[str, object]] = []
|
|
217
218
|
for span in spans:
|
|
218
219
|
score, terms = lexical_score(claim_text, span.text, mode=mode)
|
|
@@ -395,7 +396,7 @@ def _local_ml_enabled(mode: str) -> bool:
|
|
|
395
396
|
|
|
396
397
|
|
|
397
398
|
def _semantic_enabled(mode: str) -> bool:
|
|
398
|
-
return str(mode or "").strip().lower().replace("-", "_") in {"semantic", "local_ml"}
|
|
399
|
+
return str(mode or "").strip().lower().replace("-", "_") in {"semantic", "local_ml", "nli"}
|
|
399
400
|
|
|
400
401
|
|
|
401
402
|
def _token_mode(mode: str) -> str:
|
|
@@ -35,11 +35,13 @@ DEFAULT_OPENAI_MODEL = "gpt-4.1-mini"
|
|
|
35
35
|
DEFAULT_JUDGE_CACHE_PATH = ".contexttrace/judge_cache.json"
|
|
36
36
|
JUDGE_SYSTEM_PROMPT = (
|
|
37
37
|
"You are a strict RAG evidence judge. Given a user query, one generated claim, "
|
|
38
|
-
"and
|
|
38
|
+
"and selected evidence spans, return only JSON with this schema: "
|
|
39
39
|
"{\"verdict\":\"supported|partially_supported|unsupported|contradicted|unverifiable\","
|
|
40
40
|
"\"confidence\":0.0,\"reason\":\"brief explanation\","
|
|
41
41
|
"\"matched_facts\":[],\"missing_facts\":[],\"conflicting_facts\":[]}. "
|
|
42
42
|
"Use supported only when the evidence directly entails every material part of the claim. "
|
|
43
|
+
"Treat supplied contexts as minimal evidence spans, not the full source document. "
|
|
44
|
+
"Do not infer support from omitted surrounding context or from fluent answer wording. "
|
|
43
45
|
"Use contradicted when the evidence conflicts with the claim, including wrong entities, "
|
|
44
46
|
"wrong dates, wrong numbers, negation conflicts, or reversed causal/attribution roles. "
|
|
45
47
|
"Use partially_supported when some material facts are supported but others are missing. "
|
|
@@ -331,7 +333,7 @@ def judge_cache_key(
|
|
|
331
333
|
payload: dict[str, Any],
|
|
332
334
|
) -> str:
|
|
333
335
|
cache_payload = {
|
|
334
|
-
"version":
|
|
336
|
+
"version": 2,
|
|
335
337
|
"provider": provider,
|
|
336
338
|
"model": model or "",
|
|
337
339
|
"system_prompt": JUDGE_SYSTEM_PROMPT,
|
|
@@ -417,6 +419,7 @@ def build_judge_provider(
|
|
|
417
419
|
|
|
418
420
|
def _claim_payload(*, query: str, claim: str, contexts: list[TraceContext]) -> dict[str, Any]:
|
|
419
421
|
return {
|
|
422
|
+
"evidence_scope": "selected_evidence_spans_only",
|
|
420
423
|
"query": query,
|
|
421
424
|
"claim": claim,
|
|
422
425
|
"contexts": [
|