contexttrace 1.1.0__tar.gz → 1.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contexttrace-1.3.0/LICENSE +21 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/MANIFEST.in +1 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/PKG-INFO +92 -7
- {contexttrace-1.1.0 → contexttrace-1.3.0}/README.md +83 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/__init__.py +13 -2
- contexttrace-1.3.0/contexttrace/_version.py +1 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/cli.py +178 -7
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/contracts.py +1 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/demo.py +67 -0
- contexttrace-1.3.0/contexttrace/evidence_integrity.py +257 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/__init__.py +10 -2
- contexttrace-1.3.0/contexttrace/integrations/_lineage.py +34 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/langchain.py +46 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/llamaindex.py +50 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/repair.py +77 -6
- contexttrace-1.3.0/contexttrace/schemas/claim-verification-hybrid-v2.schema.json +43 -0
- contexttrace-1.3.0/contexttrace/schemas/claim-verification-v2.schema.json +497 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/__init__.py +8 -0
- contexttrace-1.3.0/contexttrace/verify/atomic_coverage.py +312 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/__init__.py +31 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/constants.py +24 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/development_manifest.json +28 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/relations.py +236 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/runner.py +118 -0
- contexttrace-1.3.0/contexttrace/verify/hybrid_v2/schema.py +11 -0
- contexttrace-1.3.0/contexttrace/verify/local_quality.py +884 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/qa.py +22 -2
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/rulepacks/generic_v1.yaml +3 -2
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/runner.py +211 -7
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/__init__.py +46 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/cascade.py +390 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/citations.py +85 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/claims.py +138 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/constants.py +105 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/diagnosis.py +143 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/limits.py +79 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/nli.py +83 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/profile.py +111 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/rulepacks/generic_v2.json +18 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/rulepacks/source_condition_v2.json +20 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/rulepacks.py +26 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/runner.py +381 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/schema.py +18 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2/source.py +207 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/__init__.py +72 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/artifacts/__init__.py +1 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/artifacts/support-risk-v1.json +54 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/attribution.py +264 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/bounded.py +85 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/checker.py +432 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/claims.py +371 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/grouping.py +235 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/profile.py +58 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/risk_model.py +233 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/runner.py +359 -0
- contexttrace-1.3.0/contexttrace/verify/semantic_core_v2_1/source.py +449 -0
- contexttrace-1.3.0/contexttrace/verify/source_trust.py +881 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/suite.py +57 -8
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/suite_report.py +5 -1
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/trace_inspect.py +3 -0
- contexttrace-1.3.0/contexttrace/verify/triage.py +59 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace.egg-info/SOURCES.txt +42 -1
- {contexttrace-1.1.0 → contexttrace-1.3.0}/pyproject.toml +20 -9
- contexttrace-1.1.0/contexttrace/_version.py +0 -1
- contexttrace-1.1.0/contexttrace/verify/source_trust.py +0 -408
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/capture.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/capture_endpoint.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/client.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/config.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/demo_data.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/diagnose.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/diagnose_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/endpoint_eval.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/errors.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/evaluator.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/fastapi.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/langgraph.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/integrations/opentelemetry.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/local.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/privacy.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/py.typed +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/regression.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/reliability.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/claim-verification-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/diagnosis-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/regression-case-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/repair-plan-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/schemas/trace-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/storage/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/storage/sqlite_store.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/thresholds.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/transport.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/abstention.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/audit.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/audit_benchmark.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/audit_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/benchmark.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/calibration.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/citations.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/claims.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/compare.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/compare_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/demos.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/evidence.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/facts.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/judges.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/local_ml.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/local_nli.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/nli_calibration.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/public_holdout_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/qa_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/root_cause.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/rulepacks/legacy_ragtruth_calibrated.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/rulepacks/policy.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/rulepacks/temporal.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/schema.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/semantic_core/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/semantic_normalization.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/spans.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/statuses.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/verify/verdicts.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/contexttrace/viewer.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/setup.cfg +0 -0
- {contexttrace-1.1.0 → contexttrace-1.3.0}/setup.py +0 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ContextTrace contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,8 +1,8 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: contexttrace
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.3.0
|
|
4
4
|
Summary: Local-first evidence-chain debugger for RAG and AI agent claim grounding, citation checks, root-cause diagnosis, and regression tests.
|
|
5
|
-
Author:
|
|
5
|
+
Author: Samarth Vinayaka
|
|
6
6
|
License-Expression: MIT
|
|
7
7
|
Project-URL: Homepage, https://github.com/samarth1412/Context-Trace
|
|
8
8
|
Project-URL: Documentation, https://github.com/samarth1412/Context-Trace/tree/main/docs
|
|
@@ -21,6 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
|
21
21
|
Classifier: Typing :: Typed
|
|
22
22
|
Requires-Python: >=3.10
|
|
23
23
|
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
24
25
|
Requires-Dist: click>=8.1
|
|
25
26
|
Requires-Dist: httpx>=0.27
|
|
26
27
|
Requires-Dist: typing-extensions>=4.9
|
|
@@ -32,11 +33,11 @@ Provides-Extra: local
|
|
|
32
33
|
Provides-Extra: local-ml
|
|
33
34
|
Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
|
|
34
35
|
Provides-Extra: nli
|
|
35
|
-
Requires-Dist: torch
|
|
36
|
-
Requires-Dist: transformers
|
|
36
|
+
Requires-Dist: torch<3,>=2.0; extra == "nli"
|
|
37
|
+
Requires-Dist: transformers<6,>=4.41; extra == "nli"
|
|
37
38
|
Provides-Extra: nli-onnx
|
|
38
39
|
Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
|
|
39
|
-
Requires-Dist: transformers
|
|
40
|
+
Requires-Dist: transformers<6,>=4.41; extra == "nli-onnx"
|
|
40
41
|
Provides-Extra: fastapi
|
|
41
42
|
Requires-Dist: fastapi>=0.110; extra == "fastapi"
|
|
42
43
|
Provides-Extra: langgraph
|
|
@@ -58,8 +59,8 @@ Requires-Dist: langgraph>=0.2; extra == "all"
|
|
|
58
59
|
Requires-Dist: llama-index-core>=0.10; extra == "all"
|
|
59
60
|
Requires-Dist: opentelemetry-api>=1.24; extra == "all"
|
|
60
61
|
Requires-Dist: sentence-transformers>=2.7; extra == "all"
|
|
61
|
-
Requires-Dist: torch
|
|
62
|
-
Requires-Dist: transformers
|
|
62
|
+
Requires-Dist: torch<3,>=2.0; extra == "all"
|
|
63
|
+
Requires-Dist: transformers<6,>=4.41; extra == "all"
|
|
63
64
|
Requires-Dist: onnxruntime>=1.17; extra == "all"
|
|
64
65
|
Provides-Extra: test
|
|
65
66
|
Requires-Dist: jsonschema>=4.22; extra == "test"
|
|
@@ -70,6 +71,7 @@ Requires-Dist: mypy>=1.10; extra == "quality"
|
|
|
70
71
|
Requires-Dist: pip-audit>=2.7; extra == "quality"
|
|
71
72
|
Requires-Dist: pytest-cov>=5.0; extra == "quality"
|
|
72
73
|
Requires-Dist: ruff>=0.5; extra == "quality"
|
|
74
|
+
Dynamic: license-file
|
|
73
75
|
|
|
74
76
|
# ContextTrace
|
|
75
77
|
|
|
@@ -104,6 +106,12 @@ contexttrace demo --dataset refund_policy
|
|
|
104
106
|
contexttrace report --last --open
|
|
105
107
|
```
|
|
106
108
|
|
|
109
|
+
Learn the workflow through
|
|
110
|
+
[six reproducible failure investigations](https://github.com/samarth1412/Context-Trace/tree/main/examples/investigations),
|
|
111
|
+
then run the
|
|
112
|
+
[LangChain and LlamaIndex regression gates](https://github.com/samarth1412/Context-Trace/tree/main/examples/integrations).
|
|
113
|
+
The public examples use fictional data and need no model API.
|
|
114
|
+
|
|
107
115
|
Default local storage:
|
|
108
116
|
|
|
109
117
|
```text
|
|
@@ -141,6 +149,63 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
|
|
|
141
149
|
|
|
142
150
|
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
143
151
|
|
|
152
|
+
### Audit evidence transformations
|
|
153
|
+
|
|
154
|
+
Instrument a selector or chunker with captured source lineage, then check
|
|
155
|
+
whether it dropped a linked answer, declared qualifier, condition, or value:
|
|
156
|
+
|
|
157
|
+
```python
|
|
158
|
+
from contexttrace import build_evidence_lineage, capture_rag_trace
|
|
159
|
+
|
|
160
|
+
source = "Question: When? Answer: After approval."
|
|
161
|
+
lineage = build_evidence_lineage(
|
|
162
|
+
source_unit_id="refund_qa",
|
|
163
|
+
source_text=source,
|
|
164
|
+
linked_parts=[
|
|
165
|
+
{"id": "question", "role": "question", "text": "Question: When?"},
|
|
166
|
+
{"id": "answer", "role": "answer", "text": "Answer: After approval."},
|
|
167
|
+
],
|
|
168
|
+
)
|
|
169
|
+
trace = capture_rag_trace(
|
|
170
|
+
query="When?",
|
|
171
|
+
answer="After approval.",
|
|
172
|
+
contexts=[{"id": "selected", "text": "Question: When?", "metadata": lineage}],
|
|
173
|
+
)
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
At a framework selection boundary, clone and bind the selected object directly:
|
|
177
|
+
|
|
178
|
+
```python
|
|
179
|
+
from contexttrace import (
|
|
180
|
+
bind_langchain_evidence_lineage,
|
|
181
|
+
bind_llamaindex_evidence_lineage,
|
|
182
|
+
)
|
|
183
|
+
|
|
184
|
+
selected_document = bind_langchain_evidence_lineage(
|
|
185
|
+
selected_document,
|
|
186
|
+
source_document=parent_document,
|
|
187
|
+
material_spans=declared_material_spans,
|
|
188
|
+
)
|
|
189
|
+
selected_node = bind_llamaindex_evidence_lineage(
|
|
190
|
+
selected_node,
|
|
191
|
+
source_node=parent_node,
|
|
192
|
+
material_spans=declared_material_spans,
|
|
193
|
+
)
|
|
194
|
+
```
|
|
195
|
+
|
|
196
|
+
Both helpers preserve the original object and existing callback paths carry the
|
|
197
|
+
lineage metadata into ContextTrace.
|
|
198
|
+
|
|
199
|
+
```bash
|
|
200
|
+
contexttrace inspect trace.json --fail-on evidence_integrity
|
|
201
|
+
contexttrace repair trace.json --out repair.md
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
The audit runs locally with no model or network calls. Missing lineage stays
|
|
205
|
+
unknown, and the stable claim verifier remains unchanged. See the
|
|
206
|
+
[evidence-integrity contract](../../docs/evidence-integrity-v1.3.md) and
|
|
207
|
+
[offline examples](../../examples/evidence_integrity/README.md).
|
|
208
|
+
|
|
144
209
|
## Diagnose An Agent Trace
|
|
145
210
|
|
|
146
211
|
`diagnose` also accepts agent step traces and localizes tool/final-answer
|
|
@@ -263,6 +328,26 @@ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issu
|
|
|
263
328
|
|
|
264
329
|
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
265
330
|
|
|
331
|
+
The experimental, opt-in `hybrid_v2` verifier can also infer dated/versioned
|
|
332
|
+
source relationships from observable document text when those metadata fields
|
|
333
|
+
are absent. It reports the inference basis, distinguishes historical questions
|
|
334
|
+
from current operational guidance, preserves unresolved source disagreement,
|
|
335
|
+
and marks relevant evidence that lacks the requested fact as `unverifiable`
|
|
336
|
+
instead of treating absence as proof. The default `verify_trace` path remains
|
|
337
|
+
the frozen `semantic_v1_calibrated` verifier.
|
|
338
|
+
|
|
339
|
+
```python
|
|
340
|
+
from contexttrace.verify import verify_trace_hybrid_v2
|
|
341
|
+
|
|
342
|
+
result = verify_trace_hybrid_v2(trace, mode="semantic")
|
|
343
|
+
```
|
|
344
|
+
|
|
345
|
+
`hybrid_v2` is an experimental API. In a frozen controlled study it caught more
|
|
346
|
+
faults but produced substantially more false alarms and underperformed the stable
|
|
347
|
+
default on the balanced release-gate measure. Calibrate it on your target system
|
|
348
|
+
before using it as a blocking gate. Its diagnostics do not certify real-world
|
|
349
|
+
truth.
|
|
350
|
+
|
|
266
351
|
## Capture Existing Systems
|
|
267
352
|
|
|
268
353
|
Capture one live endpoint response:
|
|
@@ -31,6 +31,12 @@ contexttrace demo --dataset refund_policy
|
|
|
31
31
|
contexttrace report --last --open
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
+
Learn the workflow through
|
|
35
|
+
[six reproducible failure investigations](https://github.com/samarth1412/Context-Trace/tree/main/examples/investigations),
|
|
36
|
+
then run the
|
|
37
|
+
[LangChain and LlamaIndex regression gates](https://github.com/samarth1412/Context-Trace/tree/main/examples/integrations).
|
|
38
|
+
The public examples use fictional data and need no model API.
|
|
39
|
+
|
|
34
40
|
Default local storage:
|
|
35
41
|
|
|
36
42
|
```text
|
|
@@ -68,6 +74,63 @@ ContextTrace classifies each claim as `supported`, `partially_supported`, `unsup
|
|
|
68
74
|
|
|
69
75
|
Important: `supported` means grounded by the selected evidence span. It does not mean independently true, current, or authoritative.
|
|
70
76
|
|
|
77
|
+
### Audit evidence transformations
|
|
78
|
+
|
|
79
|
+
Instrument a selector or chunker with captured source lineage, then check
|
|
80
|
+
whether it dropped a linked answer, declared qualifier, condition, or value:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
from contexttrace import build_evidence_lineage, capture_rag_trace
|
|
84
|
+
|
|
85
|
+
source = "Question: When? Answer: After approval."
|
|
86
|
+
lineage = build_evidence_lineage(
|
|
87
|
+
source_unit_id="refund_qa",
|
|
88
|
+
source_text=source,
|
|
89
|
+
linked_parts=[
|
|
90
|
+
{"id": "question", "role": "question", "text": "Question: When?"},
|
|
91
|
+
{"id": "answer", "role": "answer", "text": "Answer: After approval."},
|
|
92
|
+
],
|
|
93
|
+
)
|
|
94
|
+
trace = capture_rag_trace(
|
|
95
|
+
query="When?",
|
|
96
|
+
answer="After approval.",
|
|
97
|
+
contexts=[{"id": "selected", "text": "Question: When?", "metadata": lineage}],
|
|
98
|
+
)
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
At a framework selection boundary, clone and bind the selected object directly:
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
from contexttrace import (
|
|
105
|
+
bind_langchain_evidence_lineage,
|
|
106
|
+
bind_llamaindex_evidence_lineage,
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
selected_document = bind_langchain_evidence_lineage(
|
|
110
|
+
selected_document,
|
|
111
|
+
source_document=parent_document,
|
|
112
|
+
material_spans=declared_material_spans,
|
|
113
|
+
)
|
|
114
|
+
selected_node = bind_llamaindex_evidence_lineage(
|
|
115
|
+
selected_node,
|
|
116
|
+
source_node=parent_node,
|
|
117
|
+
material_spans=declared_material_spans,
|
|
118
|
+
)
|
|
119
|
+
```
|
|
120
|
+
|
|
121
|
+
Both helpers preserve the original object and existing callback paths carry the
|
|
122
|
+
lineage metadata into ContextTrace.
|
|
123
|
+
|
|
124
|
+
```bash
|
|
125
|
+
contexttrace inspect trace.json --fail-on evidence_integrity
|
|
126
|
+
contexttrace repair trace.json --out repair.md
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
The audit runs locally with no model or network calls. Missing lineage stays
|
|
130
|
+
unknown, and the stable claim verifier remains unchanged. See the
|
|
131
|
+
[evidence-integrity contract](../../docs/evidence-integrity-v1.3.md) and
|
|
132
|
+
[offline examples](../../examples/evidence_integrity/README.md).
|
|
133
|
+
|
|
71
134
|
## Diagnose An Agent Trace
|
|
72
135
|
|
|
73
136
|
`diagnose` also accepts agent step traces and localizes tool/final-answer
|
|
@@ -190,6 +253,26 @@ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issu
|
|
|
190
253
|
|
|
191
254
|
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
192
255
|
|
|
256
|
+
The experimental, opt-in `hybrid_v2` verifier can also infer dated/versioned
|
|
257
|
+
source relationships from observable document text when those metadata fields
|
|
258
|
+
are absent. It reports the inference basis, distinguishes historical questions
|
|
259
|
+
from current operational guidance, preserves unresolved source disagreement,
|
|
260
|
+
and marks relevant evidence that lacks the requested fact as `unverifiable`
|
|
261
|
+
instead of treating absence as proof. The default `verify_trace` path remains
|
|
262
|
+
the frozen `semantic_v1_calibrated` verifier.
|
|
263
|
+
|
|
264
|
+
```python
|
|
265
|
+
from contexttrace.verify import verify_trace_hybrid_v2
|
|
266
|
+
|
|
267
|
+
result = verify_trace_hybrid_v2(trace, mode="semantic")
|
|
268
|
+
```
|
|
269
|
+
|
|
270
|
+
`hybrid_v2` is an experimental API. In a frozen controlled study it caught more
|
|
271
|
+
faults but produced substantially more false alarms and underperformed the stable
|
|
272
|
+
default on the balanced release-gate measure. Calibrate it on your target system
|
|
273
|
+
before using it as a blocking gate. Its diagnostics do not certify real-world
|
|
274
|
+
truth.
|
|
275
|
+
|
|
193
276
|
## Capture Existing Systems
|
|
194
277
|
|
|
195
278
|
Capture one live endpoint response:
|
|
@@ -12,10 +12,17 @@ from contexttrace.errors import (
|
|
|
12
12
|
ContextTraceHTTPError,
|
|
13
13
|
ContextTraceLocalError,
|
|
14
14
|
)
|
|
15
|
+
from contexttrace.evidence_integrity import audit_evidence_integrity, build_evidence_lineage
|
|
15
16
|
from contexttrace.integrations.fastapi import ContextTraceFastAPIMiddleware
|
|
16
|
-
from contexttrace.integrations.langchain import
|
|
17
|
+
from contexttrace.integrations.langchain import (
|
|
18
|
+
ContextTraceCallbackHandler,
|
|
19
|
+
bind_langchain_evidence_lineage,
|
|
20
|
+
)
|
|
17
21
|
from contexttrace.integrations.langgraph import ContextTraceLangGraphTracer
|
|
18
|
-
from contexttrace.integrations.llamaindex import
|
|
22
|
+
from contexttrace.integrations.llamaindex import (
|
|
23
|
+
ContextTraceLlamaIndexCallbackHandler,
|
|
24
|
+
bind_llamaindex_evidence_lineage,
|
|
25
|
+
)
|
|
19
26
|
from contexttrace.integrations.opentelemetry import OpenTelemetryExporter, export_contexttrace_trace
|
|
20
27
|
from contexttrace.privacy import PrivacyPolicy, TextCipher
|
|
21
28
|
from contexttrace.reliability import ReliabilityScore, ReliabilityScorer
|
|
@@ -47,7 +54,11 @@ __all__ = [
|
|
|
47
54
|
"capture_response_trace",
|
|
48
55
|
"build_repair_plan",
|
|
49
56
|
"build_regression_case",
|
|
57
|
+
"build_evidence_lineage",
|
|
58
|
+
"bind_langchain_evidence_lineage",
|
|
59
|
+
"bind_llamaindex_evidence_lineage",
|
|
50
60
|
"diagnose_payload",
|
|
61
|
+
"audit_evidence_integrity",
|
|
51
62
|
"diagnose_trace_file",
|
|
52
63
|
"write_diagnosis_regression_test",
|
|
53
64
|
"export_contexttrace_trace",
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.3.0"
|
|
@@ -17,7 +17,7 @@ from contexttrace.capture import write_rag_trace
|
|
|
17
17
|
from contexttrace.capture_endpoint import capture_endpoint_trace, capture_response_trace
|
|
18
18
|
from contexttrace.client import ContextTrace
|
|
19
19
|
from contexttrace.config import ContextTraceConfig, load_config, write_default_config
|
|
20
|
-
from contexttrace.demo import run_demo_dataset
|
|
20
|
+
from contexttrace.demo import run_demo_dataset, run_groundedness_gap_demo
|
|
21
21
|
from contexttrace.demo_data import list_demo_datasets
|
|
22
22
|
from contexttrace.diagnose import (
|
|
23
23
|
DiagnoseInputError,
|
|
@@ -53,7 +53,7 @@ from contexttrace.verify.benchmark import run_verify_benchmark, write_verify_ben
|
|
|
53
53
|
from contexttrace.verify.audit_benchmark import run_audit_benchmark, write_audit_benchmark_report
|
|
54
54
|
from contexttrace.verify.audit_report import AuditReportGenerator
|
|
55
55
|
from contexttrace.verify.compare_report import CompareReportGenerator
|
|
56
|
-
from contexttrace.verify.qa import qa_failures, qa_trace
|
|
56
|
+
from contexttrace.verify.qa import QA_VERIFIERS, qa_failures, qa_trace
|
|
57
57
|
from contexttrace.verify.qa_report import QAReportGenerator
|
|
58
58
|
from contexttrace.verify.report import VerifyReportGenerator
|
|
59
59
|
from contexttrace.verify.suite import (
|
|
@@ -70,7 +70,14 @@ from contexttrace.verify.suite import (
|
|
|
70
70
|
write_suite_result,
|
|
71
71
|
)
|
|
72
72
|
from contexttrace.verify.suite_report import SuiteReportGenerator
|
|
73
|
+
from contexttrace.verify.semantic_core_v2 import build_pinned_nli
|
|
74
|
+
from contexttrace.verify.semantic_core_v2_1 import (
|
|
75
|
+
DETERMINISTIC_ONLY_V2_1_PROFILE,
|
|
76
|
+
SELECTIVE_V2_1_PROFILE,
|
|
77
|
+
verify_trace_v2_1,
|
|
78
|
+
)
|
|
73
79
|
from contexttrace.verify.trace_inspect import inspect_trace
|
|
80
|
+
from contexttrace.verify.triage import render_triage_summary
|
|
74
81
|
from contexttrace.viewer import serve_viewer
|
|
75
82
|
|
|
76
83
|
|
|
@@ -255,6 +262,7 @@ def report(
|
|
|
255
262
|
@cli.command("verify")
|
|
256
263
|
@click.argument("trace_json")
|
|
257
264
|
@click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
|
|
265
|
+
@click.option("--format", "output_format", default="text", show_default=True, type=click.Choice(["text", "json", "triage"]), help="Console output format.")
|
|
258
266
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
|
|
259
267
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
260
268
|
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
|
|
@@ -268,6 +276,7 @@ def verify_command(
|
|
|
268
276
|
ctx: click.Context,
|
|
269
277
|
trace_json: str,
|
|
270
278
|
json_output: bool,
|
|
279
|
+
output_format: str,
|
|
271
280
|
report: bool,
|
|
272
281
|
out: Optional[str],
|
|
273
282
|
mode: str,
|
|
@@ -305,16 +314,104 @@ def verify_command(
|
|
|
305
314
|
)
|
|
306
315
|
return _print_verify_result(
|
|
307
316
|
result,
|
|
308
|
-
json_output=json_output,
|
|
317
|
+
json_output=json_output or output_format == "json",
|
|
318
|
+
triage_output=output_format == "triage",
|
|
309
319
|
written_report=written_report,
|
|
310
320
|
fail_on=fail_on,
|
|
311
321
|
)
|
|
312
322
|
|
|
313
323
|
|
|
324
|
+
@cli.command("verify-v2")
|
|
325
|
+
@click.argument("trace_json")
|
|
326
|
+
@click.option(
|
|
327
|
+
"--profile",
|
|
328
|
+
"profile_name",
|
|
329
|
+
default="deterministic",
|
|
330
|
+
show_default=True,
|
|
331
|
+
type=click.Choice(["deterministic", "selective"]),
|
|
332
|
+
help="Use deterministic-only verification or the pinned selective NLI cascade.",
|
|
333
|
+
)
|
|
334
|
+
@click.option(
|
|
335
|
+
"--model-path",
|
|
336
|
+
default=None,
|
|
337
|
+
envvar="CONTEXTTRACE_NLI_MODEL_PATH",
|
|
338
|
+
help="Local pinned NLI directory; required for --profile selective.",
|
|
339
|
+
)
|
|
340
|
+
@click.option("--json", "json_output", is_flag=True, help="Print the full v2 result as JSON.")
|
|
341
|
+
@click.option("--output", default=None, help="Optional path for the v2 result JSON.")
|
|
342
|
+
@click.option(
|
|
343
|
+
"--fail-on",
|
|
344
|
+
multiple=True,
|
|
345
|
+
type=click.Choice(["warning", "abstained", "non_green", "any_failure"]),
|
|
346
|
+
help="Return non-zero for selected result states.",
|
|
347
|
+
)
|
|
348
|
+
def verify_v2_command(
|
|
349
|
+
trace_json: str,
|
|
350
|
+
profile_name: str,
|
|
351
|
+
model_path: Optional[str],
|
|
352
|
+
json_output: bool,
|
|
353
|
+
output: Optional[str],
|
|
354
|
+
fail_on: tuple[str, ...],
|
|
355
|
+
) -> int:
|
|
356
|
+
"""Run the opt-in safety verifier without changing stable v1 behavior."""
|
|
357
|
+
|
|
358
|
+
try:
|
|
359
|
+
trace = load_trace_file(trace_json)
|
|
360
|
+
if profile_name == "selective":
|
|
361
|
+
if not model_path:
|
|
362
|
+
raise click.ClickException(
|
|
363
|
+
"--model-path or CONTEXTTRACE_NLI_MODEL_PATH is required for the selective profile."
|
|
364
|
+
)
|
|
365
|
+
nli = build_pinned_nli(model_path)
|
|
366
|
+
profile = SELECTIVE_V2_1_PROFILE
|
|
367
|
+
else:
|
|
368
|
+
nli = None
|
|
369
|
+
profile = DETERMINISTIC_ONLY_V2_1_PROFILE
|
|
370
|
+
result = verify_trace_v2_1(trace, profile=profile, nli=nli)
|
|
371
|
+
except VerificationInputError as exc:
|
|
372
|
+
raise click.ClickException(str(exc)) from exc
|
|
373
|
+
except (OSError, RuntimeError, ValueError) as exc:
|
|
374
|
+
raise click.ClickException(str(exc)) from exc
|
|
375
|
+
|
|
376
|
+
rendered = json.dumps(result, indent=2, sort_keys=True) + "\n"
|
|
377
|
+
if output:
|
|
378
|
+
output_path = Path(output)
|
|
379
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
380
|
+
output_path.write_text(rendered, encoding="utf-8")
|
|
381
|
+
|
|
382
|
+
if json_output:
|
|
383
|
+
click.echo(rendered, nl=False)
|
|
384
|
+
else:
|
|
385
|
+
summary = result["summary"]
|
|
386
|
+
click.echo("Profile: %s" % result["profile_id"])
|
|
387
|
+
click.echo("Claims: %s" % summary["total_claims"])
|
|
388
|
+
click.echo("Status: %s" % summary["overall_status"])
|
|
389
|
+
click.echo("Green claims: %s" % summary["green_claims"])
|
|
390
|
+
click.echo("Diagnostic abstentions: %s" % summary["diagnostic_abstentions"])
|
|
391
|
+
click.echo("NLI invocations: %s" % summary["nli_invocations"])
|
|
392
|
+
if output:
|
|
393
|
+
click.echo("Result: %s" % output)
|
|
394
|
+
|
|
395
|
+
status = str(result["summary"]["overall_status"])
|
|
396
|
+
should_fail = bool(
|
|
397
|
+
"any_failure" in fail_on and status != "green"
|
|
398
|
+
or "non_green" in fail_on and status != "green"
|
|
399
|
+
or "warning" in fail_on and status == "warning"
|
|
400
|
+
or "abstained" in fail_on and status == "abstained"
|
|
401
|
+
)
|
|
402
|
+
return 1 if should_fail else 0
|
|
403
|
+
|
|
404
|
+
|
|
314
405
|
@cli.command("inspect")
|
|
315
406
|
@click.argument("trace_json")
|
|
316
407
|
@click.option("--json", "json_output", is_flag=True, help="Print trace inspection as JSON.")
|
|
317
|
-
|
|
408
|
+
@click.option(
|
|
409
|
+
"--fail-on",
|
|
410
|
+
multiple=True,
|
|
411
|
+
type=click.Choice(["warning", "evidence_integrity", "unknown_integrity"]),
|
|
412
|
+
help="Exit nonzero on trace warnings, observed evidence-integrity issues, or uncaptured lineage.",
|
|
413
|
+
)
|
|
414
|
+
def inspect_command(trace_json: str, json_output: bool, fail_on: tuple[str, ...]) -> int:
|
|
318
415
|
"""Inspect portable RAG trace shape before verification."""
|
|
319
416
|
|
|
320
417
|
try:
|
|
@@ -325,7 +422,7 @@ def inspect_command(trace_json: str, json_output: bool) -> int:
|
|
|
325
422
|
result = inspect_trace(trace, trace_path=trace_json)
|
|
326
423
|
if json_output:
|
|
327
424
|
click.echo(json.dumps(result, indent=2))
|
|
328
|
-
return
|
|
425
|
+
return _inspect_exit_code(result, fail_on)
|
|
329
426
|
|
|
330
427
|
click.echo("Trace: %s" % trace_json)
|
|
331
428
|
click.echo("Query: %s" % result["query"])
|
|
@@ -349,6 +446,21 @@ def inspect_command(trace_json: str, json_output: bool) -> int:
|
|
|
349
446
|
if result["metadata_keys"]:
|
|
350
447
|
click.echo("Metadata keys: %s" % ", ".join(result["metadata_keys"]))
|
|
351
448
|
|
|
449
|
+
integrity = result["evidence_integrity"]
|
|
450
|
+
integrity_summary = integrity["summary"]
|
|
451
|
+
click.echo(
|
|
452
|
+
"Evidence integrity: %s (%s issues, %s assessed, %s unknown)"
|
|
453
|
+
% (
|
|
454
|
+
integrity["status"],
|
|
455
|
+
integrity_summary["issue_count"],
|
|
456
|
+
integrity_summary["assessed_contexts"],
|
|
457
|
+
integrity_summary["unknown_contexts"],
|
|
458
|
+
)
|
|
459
|
+
)
|
|
460
|
+
for issue in integrity["issues"]:
|
|
461
|
+
detail = issue.get("item_role") or issue.get("observed")
|
|
462
|
+
click.echo("- %s [%s]: %s" % (issue["type"], issue["selected_context_id"], detail))
|
|
463
|
+
|
|
352
464
|
warnings = result["warnings"]
|
|
353
465
|
if warnings:
|
|
354
466
|
click.echo("Warnings:")
|
|
@@ -358,7 +470,20 @@ def inspect_command(trace_json: str, json_output: bool) -> int:
|
|
|
358
470
|
click.echo("Suggested next commands:")
|
|
359
471
|
for command in result["suggested_next_commands"]:
|
|
360
472
|
click.echo("- %s" % command)
|
|
361
|
-
return
|
|
473
|
+
return _inspect_exit_code(result, fail_on)
|
|
474
|
+
|
|
475
|
+
|
|
476
|
+
def _inspect_exit_code(result: dict[str, object], fail_on: tuple[str, ...]) -> int:
|
|
477
|
+
integrity = result.get("evidence_integrity") or {}
|
|
478
|
+
summary = (integrity.get("summary") or {}) if isinstance(integrity, dict) else {}
|
|
479
|
+
issue_count = int(summary.get("issue_count") or 0) if isinstance(summary, dict) else 0
|
|
480
|
+
unknown_count = int(summary.get("unknown_contexts") or 0) if isinstance(summary, dict) else 0
|
|
481
|
+
should_fail = (
|
|
482
|
+
"warning" in fail_on and bool(result.get("warnings"))
|
|
483
|
+
or "evidence_integrity" in fail_on and issue_count > 0
|
|
484
|
+
or "unknown_integrity" in fail_on and unknown_count > 0
|
|
485
|
+
)
|
|
486
|
+
return 1 if should_fail else 0
|
|
362
487
|
|
|
363
488
|
|
|
364
489
|
@cli.command("diagnose")
|
|
@@ -563,12 +688,14 @@ def suite_group() -> None:
|
|
|
563
688
|
@click.argument("trace_json", nargs=-1, required=True)
|
|
564
689
|
@click.option("--out", default="contexttrace-suite.json", show_default=True, help="Suite JSON file to write.")
|
|
565
690
|
@click.option("--name", default=None, help="Suite name.")
|
|
691
|
+
@click.option("--verifier", default="semantic_v1_calibrated", show_default=True, type=click.Choice(QA_VERIFIERS), help="Verifier saved in the suite. hybrid_v2 is experimental and opt-in.")
|
|
566
692
|
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for baseline QA.")
|
|
567
693
|
@click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
|
|
568
694
|
def suite_create_command(
|
|
569
695
|
trace_json: tuple[str, ...],
|
|
570
696
|
out: str,
|
|
571
697
|
name: Optional[str],
|
|
698
|
+
verifier: str,
|
|
572
699
|
mode: str,
|
|
573
700
|
corpus_path: Optional[str],
|
|
574
701
|
) -> int:
|
|
@@ -578,6 +705,7 @@ def suite_create_command(
|
|
|
578
705
|
suite = create_suite_from_trace_files(
|
|
579
706
|
trace_json,
|
|
580
707
|
name=name,
|
|
708
|
+
verifier=verifier,
|
|
581
709
|
mode=mode,
|
|
582
710
|
corpus_path=corpus_path,
|
|
583
711
|
)
|
|
@@ -587,6 +715,8 @@ def suite_create_command(
|
|
|
587
715
|
|
|
588
716
|
click.echo("Suite: %s" % written)
|
|
589
717
|
click.echo("Cases: %s" % len(suite.get("cases") or []))
|
|
718
|
+
if suite.get("verifier") == "hybrid_v2":
|
|
719
|
+
click.echo("Verifier: hybrid_v2 (experimental; retained for add and run)")
|
|
590
720
|
click.echo("Policy: saved cases must pass on replay")
|
|
591
721
|
return 0
|
|
592
722
|
|
|
@@ -1582,10 +1712,18 @@ def _print_verify_result(
|
|
|
1582
1712
|
result: dict,
|
|
1583
1713
|
*,
|
|
1584
1714
|
json_output: bool,
|
|
1715
|
+
triage_output: bool = False,
|
|
1585
1716
|
written_report: Optional[str],
|
|
1586
1717
|
fail_on: tuple[str, ...] = (),
|
|
1587
1718
|
) -> int:
|
|
1588
1719
|
fail_messages = _verify_failures(result, fail_on)
|
|
1720
|
+
if triage_output:
|
|
1721
|
+
click.echo(render_triage_summary(result))
|
|
1722
|
+
if written_report:
|
|
1723
|
+
click.echo("Report: %s" % written_report)
|
|
1724
|
+
for message in fail_messages:
|
|
1725
|
+
click.echo("Verification failed: %s" % message, err=True)
|
|
1726
|
+
return 1 if fail_messages else 0
|
|
1589
1727
|
if json_output:
|
|
1590
1728
|
if written_report:
|
|
1591
1729
|
click.echo("Report: %s" % written_report, err=True)
|
|
@@ -1755,10 +1893,41 @@ def eval_command(
|
|
|
1755
1893
|
|
|
1756
1894
|
|
|
1757
1895
|
@cli.command()
|
|
1896
|
+
@click.argument("scenario", required=False)
|
|
1758
1897
|
@click.option("--dataset", default="refund_policy", show_default=True, help="Demo dataset name or path.")
|
|
1759
1898
|
@click.option("--strategy", default="adaptive", show_default=True, help="Demo retrieval strategy.")
|
|
1760
1899
|
@click.pass_context
|
|
1761
|
-
def demo(ctx: click.Context, dataset: str, strategy: str) -> None:
|
|
1900
|
+
def demo(ctx: click.Context, scenario: Optional[str], dataset: str, strategy: str) -> None:
|
|
1901
|
+
if scenario:
|
|
1902
|
+
normalized = scenario.strip().lower().replace("_", "-")
|
|
1903
|
+
if normalized != "groundedness-gap":
|
|
1904
|
+
raise click.UsageError("Unknown demo scenario %r. Use 'groundedness-gap'." % scenario)
|
|
1905
|
+
groundedness_demo = run_groundedness_gap_demo()
|
|
1906
|
+
rag = groundedness_demo.result.get("rag") or {}
|
|
1907
|
+
claim = ((rag.get("claims") or [{}])[0])
|
|
1908
|
+
summary = rag.get("summary") or {}
|
|
1909
|
+
abstention = rag.get("abstention") or {}
|
|
1910
|
+
root = claim.get("root_cause") or {}
|
|
1911
|
+
trace_payload = json.loads(Path(groundedness_demo.trace_path).read_text(encoding="utf-8"))
|
|
1912
|
+
click.echo("Query: %s" % rag.get("query"))
|
|
1913
|
+
click.echo("Answer: %s" % rag.get("answer"))
|
|
1914
|
+
for context in trace_payload.get("contexts") or []:
|
|
1915
|
+
metadata = context.get("metadata") or {}
|
|
1916
|
+
condition = metadata.get("freshness") or "unknown"
|
|
1917
|
+
click.echo(
|
|
1918
|
+
"Retrieved source [%s, %s]: %s"
|
|
1919
|
+
% (context.get("id"), condition, context.get("text"))
|
|
1920
|
+
)
|
|
1921
|
+
click.echo("Citation: atlas_policy_2024")
|
|
1922
|
+
click.echo("Support verdict: %s" % claim.get("verdict"))
|
|
1923
|
+
click.echo("Citation status: %s" % claim.get("citation_status"))
|
|
1924
|
+
click.echo("Source condition: %s" % claim.get("source_status"))
|
|
1925
|
+
click.echo("Abstain: %s" % str(bool(abstention.get("should_abstain"))).lower())
|
|
1926
|
+
click.echo("Root cause: %s" % root.get("label"))
|
|
1927
|
+
click.echo("Repair: %s" % (root.get("suggested_fix") or summary.get("suggested_fix")))
|
|
1928
|
+
click.echo("Trace: %s" % groundedness_demo.trace_path)
|
|
1929
|
+
click.echo("Regression test: %s" % groundedness_demo.regression_test_path)
|
|
1930
|
+
return
|
|
1762
1931
|
client = _client(ctx)
|
|
1763
1932
|
config = _load(ctx)
|
|
1764
1933
|
report_path = Path(config.local_store_dir) / "reports" / ("%s_demo.html" % Path(dataset).name)
|
|
@@ -1878,6 +2047,8 @@ def _print_suite_result(
|
|
|
1878
2047
|
) -> None:
|
|
1879
2048
|
summary = result.get("summary") or {}
|
|
1880
2049
|
click.echo("Suite: %s" % result.get("suite_name"))
|
|
2050
|
+
if result.get("verifier") == "hybrid_v2":
|
|
2051
|
+
click.echo("Verifier: hybrid_v2 (experimental)")
|
|
1881
2052
|
click.echo("Status: %s" % summary.get("status"))
|
|
1882
2053
|
click.echo("Cases: %s" % summary.get("total_cases"))
|
|
1883
2054
|
click.echo("Passed: %s" % summary.get("passed"))
|
|
@@ -19,6 +19,7 @@ DEFAULT_PROFILE_ID = "full_v1"
|
|
|
19
19
|
SCHEMA_FILES = {
|
|
20
20
|
"TraceV1": "trace-v1.schema.json",
|
|
21
21
|
"ClaimVerificationV1": "claim-verification-v1.schema.json",
|
|
22
|
+
"ClaimVerificationHybridV2": "claim-verification-hybrid-v2.schema.json",
|
|
22
23
|
"DiagnosisV1": "diagnosis-v1.schema.json",
|
|
23
24
|
"RepairPlanV1": "repair-plan-v1.schema.json",
|
|
24
25
|
"RegressionCaseV1": "regression-case-v1.schema.json",
|