contexttrace 1.1.0__tar.gz → 1.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- contexttrace-1.2.0/LICENSE +21 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/MANIFEST.in +1 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/PKG-INFO +34 -6
- {contexttrace-1.1.0 → contexttrace-1.2.0}/README.md +26 -0
- contexttrace-1.2.0/contexttrace/_version.py +1 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/cli.py +141 -4
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/contracts.py +1 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/demo.py +67 -0
- contexttrace-1.2.0/contexttrace/schemas/claim-verification-hybrid-v2.schema.json +43 -0
- contexttrace-1.2.0/contexttrace/schemas/claim-verification-v2.schema.json +497 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/__init__.py +8 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/__init__.py +31 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/constants.py +24 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/development_manifest.json +28 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/relations.py +236 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/runner.py +118 -0
- contexttrace-1.2.0/contexttrace/verify/hybrid_v2/schema.py +11 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/qa.py +22 -2
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/rulepacks/generic_v1.yaml +3 -2
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/runner.py +211 -7
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/__init__.py +46 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/cascade.py +390 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/citations.py +85 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/claims.py +138 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/constants.py +105 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/diagnosis.py +143 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/limits.py +79 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/nli.py +83 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/profile.py +111 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/rulepacks/generic_v2.json +18 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/rulepacks/source_condition_v2.json +20 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/rulepacks.py +26 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/runner.py +381 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/schema.py +18 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2/source.py +207 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/__init__.py +72 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/artifacts/__init__.py +1 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/artifacts/support-risk-v1.json +54 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/attribution.py +264 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/bounded.py +85 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/checker.py +432 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/claims.py +371 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/grouping.py +235 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/profile.py +58 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/risk_model.py +233 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/runner.py +359 -0
- contexttrace-1.2.0/contexttrace/verify/semantic_core_v2_1/source.py +449 -0
- contexttrace-1.2.0/contexttrace/verify/source_trust.py +881 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/suite.py +57 -8
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/suite_report.py +5 -1
- contexttrace-1.2.0/contexttrace/verify/triage.py +59 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace.egg-info/SOURCES.txt +38 -1
- {contexttrace-1.1.0 → contexttrace-1.2.0}/pyproject.toml +19 -8
- contexttrace-1.1.0/contexttrace/_version.py +0 -1
- contexttrace-1.1.0/contexttrace/verify/source_trust.py +0 -408
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/capture.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/capture_endpoint.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/client.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/config.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/demo_data.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/diagnose.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/diagnose_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/endpoint_eval.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/errors.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/evaluator.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/fastapi.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/langchain.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/langgraph.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/llamaindex.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/integrations/opentelemetry.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/local.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/privacy.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/py.typed +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/regression.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/reliability.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/repair.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/claim-verification-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/diagnosis-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/regression-case-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/repair-plan-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/schemas/trace-v1.schema.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/storage/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/storage/sqlite_store.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/thresholds.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/transport.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/abstention.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/audit.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/audit_benchmark.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/audit_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/audit_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/benchmark.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/calibration.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/citations.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/claims.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/compare.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/compare_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/demos.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/evidence.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/external_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/facts.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/judges.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/local_ml.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/local_nli.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/nli_calibration.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/public_holdout_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/qa_report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/real_benchmark_cases.json +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/report.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/root_cause.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/rulepacks/legacy_ragtruth_calibrated.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/rulepacks/policy.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/rulepacks/temporal.yaml +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/schema.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/semantic_core/__init__.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/semantic_normalization.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/spans.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/statuses.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/trace_inspect.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/verify/verdicts.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/contexttrace/viewer.py +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/setup.cfg +0 -0
- {contexttrace-1.1.0 → contexttrace-1.2.0}/setup.py +0 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 ContextTrace contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: contexttrace
|
|
3
|
-
Version: 1.
|
|
3
|
+
Version: 1.2.0
|
|
4
4
|
Summary: Local-first evidence-chain debugger for RAG and AI agent claim grounding, citation checks, root-cause diagnosis, and regression tests.
|
|
5
5
|
Author: ContextTrace contributors
|
|
6
6
|
License-Expression: MIT
|
|
@@ -21,6 +21,7 @@ Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
|
21
21
|
Classifier: Typing :: Typed
|
|
22
22
|
Requires-Python: >=3.10
|
|
23
23
|
Description-Content-Type: text/markdown
|
|
24
|
+
License-File: LICENSE
|
|
24
25
|
Requires-Dist: click>=8.1
|
|
25
26
|
Requires-Dist: httpx>=0.27
|
|
26
27
|
Requires-Dist: typing-extensions>=4.9
|
|
@@ -32,11 +33,11 @@ Provides-Extra: local
|
|
|
32
33
|
Provides-Extra: local-ml
|
|
33
34
|
Requires-Dist: sentence-transformers>=2.7; extra == "local-ml"
|
|
34
35
|
Provides-Extra: nli
|
|
35
|
-
Requires-Dist: torch
|
|
36
|
-
Requires-Dist: transformers
|
|
36
|
+
Requires-Dist: torch<3,>=2.0; extra == "nli"
|
|
37
|
+
Requires-Dist: transformers<6,>=4.41; extra == "nli"
|
|
37
38
|
Provides-Extra: nli-onnx
|
|
38
39
|
Requires-Dist: onnxruntime>=1.17; extra == "nli-onnx"
|
|
39
|
-
Requires-Dist: transformers
|
|
40
|
+
Requires-Dist: transformers<6,>=4.41; extra == "nli-onnx"
|
|
40
41
|
Provides-Extra: fastapi
|
|
41
42
|
Requires-Dist: fastapi>=0.110; extra == "fastapi"
|
|
42
43
|
Provides-Extra: langgraph
|
|
@@ -58,8 +59,8 @@ Requires-Dist: langgraph>=0.2; extra == "all"
|
|
|
58
59
|
Requires-Dist: llama-index-core>=0.10; extra == "all"
|
|
59
60
|
Requires-Dist: opentelemetry-api>=1.24; extra == "all"
|
|
60
61
|
Requires-Dist: sentence-transformers>=2.7; extra == "all"
|
|
61
|
-
Requires-Dist: torch
|
|
62
|
-
Requires-Dist: transformers
|
|
62
|
+
Requires-Dist: torch<3,>=2.0; extra == "all"
|
|
63
|
+
Requires-Dist: transformers<6,>=4.41; extra == "all"
|
|
63
64
|
Requires-Dist: onnxruntime>=1.17; extra == "all"
|
|
64
65
|
Provides-Extra: test
|
|
65
66
|
Requires-Dist: jsonschema>=4.22; extra == "test"
|
|
@@ -70,6 +71,7 @@ Requires-Dist: mypy>=1.10; extra == "quality"
|
|
|
70
71
|
Requires-Dist: pip-audit>=2.7; extra == "quality"
|
|
71
72
|
Requires-Dist: pytest-cov>=5.0; extra == "quality"
|
|
72
73
|
Requires-Dist: ruff>=0.5; extra == "quality"
|
|
74
|
+
Dynamic: license-file
|
|
73
75
|
|
|
74
76
|
# ContextTrace
|
|
75
77
|
|
|
@@ -104,6 +106,12 @@ contexttrace demo --dataset refund_policy
|
|
|
104
106
|
contexttrace report --last --open
|
|
105
107
|
```
|
|
106
108
|
|
|
109
|
+
Learn the workflow through
|
|
110
|
+
[six reproducible failure investigations](https://github.com/samarth1412/Context-Trace/tree/main/examples/investigations),
|
|
111
|
+
then run the
|
|
112
|
+
[LangChain and LlamaIndex regression gates](https://github.com/samarth1412/Context-Trace/tree/main/examples/integrations).
|
|
113
|
+
The public examples use fictional data and need no model API.
|
|
114
|
+
|
|
107
115
|
Default local storage:
|
|
108
116
|
|
|
109
117
|
```text
|
|
@@ -263,6 +271,26 @@ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issu
|
|
|
263
271
|
|
|
264
272
|
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
265
273
|
|
|
274
|
+
The experimental, opt-in `hybrid_v2` verifier can also infer dated/versioned
|
|
275
|
+
source relationships from observable document text when those metadata fields
|
|
276
|
+
are absent. It reports the inference basis, distinguishes historical questions
|
|
277
|
+
from current operational guidance, preserves unresolved source disagreement,
|
|
278
|
+
and marks relevant evidence that lacks the requested fact as `unverifiable`
|
|
279
|
+
instead of treating absence as proof. The default `verify_trace` path remains
|
|
280
|
+
the frozen `semantic_v1_calibrated` verifier.
|
|
281
|
+
|
|
282
|
+
```python
|
|
283
|
+
from contexttrace.verify import verify_trace_hybrid_v2
|
|
284
|
+
|
|
285
|
+
result = verify_trace_hybrid_v2(trace, mode="semantic")
|
|
286
|
+
```
|
|
287
|
+
|
|
288
|
+
`hybrid_v2` is an experimental API. In a frozen controlled study it caught more
|
|
289
|
+
faults but produced substantially more false alarms and underperformed the stable
|
|
290
|
+
default on the balanced release-gate measure. Calibrate it on your target system
|
|
291
|
+
before using it as a blocking gate. Its diagnostics do not certify real-world
|
|
292
|
+
truth.
|
|
293
|
+
|
|
266
294
|
## Capture Existing Systems
|
|
267
295
|
|
|
268
296
|
Capture one live endpoint response:
|
|
@@ -31,6 +31,12 @@ contexttrace demo --dataset refund_policy
|
|
|
31
31
|
contexttrace report --last --open
|
|
32
32
|
```
|
|
33
33
|
|
|
34
|
+
Learn the workflow through
|
|
35
|
+
[six reproducible failure investigations](https://github.com/samarth1412/Context-Trace/tree/main/examples/investigations),
|
|
36
|
+
then run the
|
|
37
|
+
[LangChain and LlamaIndex regression gates](https://github.com/samarth1412/Context-Trace/tree/main/examples/integrations).
|
|
38
|
+
The public examples use fictional data and need no model API.
|
|
39
|
+
|
|
34
40
|
Default local storage:
|
|
35
41
|
|
|
36
42
|
```text
|
|
@@ -190,6 +196,26 @@ Common root causes include `retrieval_miss`, `reranking_failure`, `chunking_issu
|
|
|
190
196
|
|
|
191
197
|
Source metadata can include `source_authority`, `source_timestamp`, `source_version`, `canonical`, or `canonical_source`. ContextTrace uses those local fields to flag `grounded_but_stale`, `grounded_but_conflicted`, `grounded_by_low_authority_source`, or `supported_by_canonical_source`.
|
|
192
198
|
|
|
199
|
+
The experimental, opt-in `hybrid_v2` verifier can also infer dated/versioned
|
|
200
|
+
source relationships from observable document text when those metadata fields
|
|
201
|
+
are absent. It reports the inference basis, distinguishes historical questions
|
|
202
|
+
from current operational guidance, preserves unresolved source disagreement,
|
|
203
|
+
and marks relevant evidence that lacks the requested fact as `unverifiable`
|
|
204
|
+
instead of treating absence as proof. The default `verify_trace` path remains
|
|
205
|
+
the frozen `semantic_v1_calibrated` verifier.
|
|
206
|
+
|
|
207
|
+
```python
|
|
208
|
+
from contexttrace.verify import verify_trace_hybrid_v2
|
|
209
|
+
|
|
210
|
+
result = verify_trace_hybrid_v2(trace, mode="semantic")
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
`hybrid_v2` is an experimental API. In a frozen controlled study it caught more
|
|
214
|
+
faults but produced substantially more false alarms and underperformed the stable
|
|
215
|
+
default on the balanced release-gate measure. Calibrate it on your target system
|
|
216
|
+
before using it as a blocking gate. Its diagnostics do not certify real-world
|
|
217
|
+
truth.
|
|
218
|
+
|
|
193
219
|
## Capture Existing Systems
|
|
194
220
|
|
|
195
221
|
Capture one live endpoint response:
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "1.2.0"
|
|
@@ -17,7 +17,7 @@ from contexttrace.capture import write_rag_trace
|
|
|
17
17
|
from contexttrace.capture_endpoint import capture_endpoint_trace, capture_response_trace
|
|
18
18
|
from contexttrace.client import ContextTrace
|
|
19
19
|
from contexttrace.config import ContextTraceConfig, load_config, write_default_config
|
|
20
|
-
from contexttrace.demo import run_demo_dataset
|
|
20
|
+
from contexttrace.demo import run_demo_dataset, run_groundedness_gap_demo
|
|
21
21
|
from contexttrace.demo_data import list_demo_datasets
|
|
22
22
|
from contexttrace.diagnose import (
|
|
23
23
|
DiagnoseInputError,
|
|
@@ -53,7 +53,7 @@ from contexttrace.verify.benchmark import run_verify_benchmark, write_verify_ben
|
|
|
53
53
|
from contexttrace.verify.audit_benchmark import run_audit_benchmark, write_audit_benchmark_report
|
|
54
54
|
from contexttrace.verify.audit_report import AuditReportGenerator
|
|
55
55
|
from contexttrace.verify.compare_report import CompareReportGenerator
|
|
56
|
-
from contexttrace.verify.qa import qa_failures, qa_trace
|
|
56
|
+
from contexttrace.verify.qa import QA_VERIFIERS, qa_failures, qa_trace
|
|
57
57
|
from contexttrace.verify.qa_report import QAReportGenerator
|
|
58
58
|
from contexttrace.verify.report import VerifyReportGenerator
|
|
59
59
|
from contexttrace.verify.suite import (
|
|
@@ -70,7 +70,14 @@ from contexttrace.verify.suite import (
|
|
|
70
70
|
write_suite_result,
|
|
71
71
|
)
|
|
72
72
|
from contexttrace.verify.suite_report import SuiteReportGenerator
|
|
73
|
+
from contexttrace.verify.semantic_core_v2 import build_pinned_nli
|
|
74
|
+
from contexttrace.verify.semantic_core_v2_1 import (
|
|
75
|
+
DETERMINISTIC_ONLY_V2_1_PROFILE,
|
|
76
|
+
SELECTIVE_V2_1_PROFILE,
|
|
77
|
+
verify_trace_v2_1,
|
|
78
|
+
)
|
|
73
79
|
from contexttrace.verify.trace_inspect import inspect_trace
|
|
80
|
+
from contexttrace.verify.triage import render_triage_summary
|
|
74
81
|
from contexttrace.viewer import serve_viewer
|
|
75
82
|
|
|
76
83
|
|
|
@@ -255,6 +262,7 @@ def report(
|
|
|
255
262
|
@cli.command("verify")
|
|
256
263
|
@click.argument("trace_json")
|
|
257
264
|
@click.option("--json", "json_output", is_flag=True, help="Print the full verification result as JSON.")
|
|
265
|
+
@click.option("--format", "output_format", default="text", show_default=True, type=click.Choice(["text", "json", "triage"]), help="Console output format.")
|
|
258
266
|
@click.option("--report", is_flag=True, help="Generate a local HTML verification report.")
|
|
259
267
|
@click.option("--out", default=None, help="HTML report path. Implies --report when provided.")
|
|
260
268
|
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(JUDGE_VERIFY_MODES), help="Evidence scoring mode.")
|
|
@@ -268,6 +276,7 @@ def verify_command(
|
|
|
268
276
|
ctx: click.Context,
|
|
269
277
|
trace_json: str,
|
|
270
278
|
json_output: bool,
|
|
279
|
+
output_format: str,
|
|
271
280
|
report: bool,
|
|
272
281
|
out: Optional[str],
|
|
273
282
|
mode: str,
|
|
@@ -305,12 +314,94 @@ def verify_command(
|
|
|
305
314
|
)
|
|
306
315
|
return _print_verify_result(
|
|
307
316
|
result,
|
|
308
|
-
json_output=json_output,
|
|
317
|
+
json_output=json_output or output_format == "json",
|
|
318
|
+
triage_output=output_format == "triage",
|
|
309
319
|
written_report=written_report,
|
|
310
320
|
fail_on=fail_on,
|
|
311
321
|
)
|
|
312
322
|
|
|
313
323
|
|
|
324
|
+
@cli.command("verify-v2")
|
|
325
|
+
@click.argument("trace_json")
|
|
326
|
+
@click.option(
|
|
327
|
+
"--profile",
|
|
328
|
+
"profile_name",
|
|
329
|
+
default="deterministic",
|
|
330
|
+
show_default=True,
|
|
331
|
+
type=click.Choice(["deterministic", "selective"]),
|
|
332
|
+
help="Use deterministic-only verification or the pinned selective NLI cascade.",
|
|
333
|
+
)
|
|
334
|
+
@click.option(
|
|
335
|
+
"--model-path",
|
|
336
|
+
default=None,
|
|
337
|
+
envvar="CONTEXTTRACE_NLI_MODEL_PATH",
|
|
338
|
+
help="Local pinned NLI directory; required for --profile selective.",
|
|
339
|
+
)
|
|
340
|
+
@click.option("--json", "json_output", is_flag=True, help="Print the full v2 result as JSON.")
|
|
341
|
+
@click.option("--output", default=None, help="Optional path for the v2 result JSON.")
|
|
342
|
+
@click.option(
|
|
343
|
+
"--fail-on",
|
|
344
|
+
multiple=True,
|
|
345
|
+
type=click.Choice(["warning", "abstained", "non_green", "any_failure"]),
|
|
346
|
+
help="Return non-zero for selected result states.",
|
|
347
|
+
)
|
|
348
|
+
def verify_v2_command(
|
|
349
|
+
trace_json: str,
|
|
350
|
+
profile_name: str,
|
|
351
|
+
model_path: Optional[str],
|
|
352
|
+
json_output: bool,
|
|
353
|
+
output: Optional[str],
|
|
354
|
+
fail_on: tuple[str, ...],
|
|
355
|
+
) -> int:
|
|
356
|
+
"""Run the opt-in safety verifier without changing stable v1 behavior."""
|
|
357
|
+
|
|
358
|
+
try:
|
|
359
|
+
trace = load_trace_file(trace_json)
|
|
360
|
+
if profile_name == "selective":
|
|
361
|
+
if not model_path:
|
|
362
|
+
raise click.ClickException(
|
|
363
|
+
"--model-path or CONTEXTTRACE_NLI_MODEL_PATH is required for the selective profile."
|
|
364
|
+
)
|
|
365
|
+
nli = build_pinned_nli(model_path)
|
|
366
|
+
profile = SELECTIVE_V2_1_PROFILE
|
|
367
|
+
else:
|
|
368
|
+
nli = None
|
|
369
|
+
profile = DETERMINISTIC_ONLY_V2_1_PROFILE
|
|
370
|
+
result = verify_trace_v2_1(trace, profile=profile, nli=nli)
|
|
371
|
+
except VerificationInputError as exc:
|
|
372
|
+
raise click.ClickException(str(exc)) from exc
|
|
373
|
+
except (OSError, RuntimeError, ValueError) as exc:
|
|
374
|
+
raise click.ClickException(str(exc)) from exc
|
|
375
|
+
|
|
376
|
+
rendered = json.dumps(result, indent=2, sort_keys=True) + "\n"
|
|
377
|
+
if output:
|
|
378
|
+
output_path = Path(output)
|
|
379
|
+
output_path.parent.mkdir(parents=True, exist_ok=True)
|
|
380
|
+
output_path.write_text(rendered, encoding="utf-8")
|
|
381
|
+
|
|
382
|
+
if json_output:
|
|
383
|
+
click.echo(rendered, nl=False)
|
|
384
|
+
else:
|
|
385
|
+
summary = result["summary"]
|
|
386
|
+
click.echo("Profile: %s" % result["profile_id"])
|
|
387
|
+
click.echo("Claims: %s" % summary["total_claims"])
|
|
388
|
+
click.echo("Status: %s" % summary["overall_status"])
|
|
389
|
+
click.echo("Green claims: %s" % summary["green_claims"])
|
|
390
|
+
click.echo("Diagnostic abstentions: %s" % summary["diagnostic_abstentions"])
|
|
391
|
+
click.echo("NLI invocations: %s" % summary["nli_invocations"])
|
|
392
|
+
if output:
|
|
393
|
+
click.echo("Result: %s" % output)
|
|
394
|
+
|
|
395
|
+
status = str(result["summary"]["overall_status"])
|
|
396
|
+
should_fail = bool(
|
|
397
|
+
"any_failure" in fail_on and status != "green"
|
|
398
|
+
or "non_green" in fail_on and status != "green"
|
|
399
|
+
or "warning" in fail_on and status == "warning"
|
|
400
|
+
or "abstained" in fail_on and status == "abstained"
|
|
401
|
+
)
|
|
402
|
+
return 1 if should_fail else 0
|
|
403
|
+
|
|
404
|
+
|
|
314
405
|
@cli.command("inspect")
|
|
315
406
|
@click.argument("trace_json")
|
|
316
407
|
@click.option("--json", "json_output", is_flag=True, help="Print trace inspection as JSON.")
|
|
@@ -563,12 +654,14 @@ def suite_group() -> None:
|
|
|
563
654
|
@click.argument("trace_json", nargs=-1, required=True)
|
|
564
655
|
@click.option("--out", default="contexttrace-suite.json", show_default=True, help="Suite JSON file to write.")
|
|
565
656
|
@click.option("--name", default=None, help="Suite name.")
|
|
657
|
+
@click.option("--verifier", default="semantic_v1_calibrated", show_default=True, type=click.Choice(QA_VERIFIERS), help="Verifier saved in the suite. hybrid_v2 is experimental and opt-in.")
|
|
566
658
|
@click.option("--mode", default="lexical", show_default=True, type=click.Choice(BASIC_VERIFY_MODES), help="Evidence scoring mode for baseline QA.")
|
|
567
659
|
@click.option("--corpus", "corpus_path", default=None, help="Optional local corpus directory or file for baseline retrieval/corpus audit.")
|
|
568
660
|
def suite_create_command(
|
|
569
661
|
trace_json: tuple[str, ...],
|
|
570
662
|
out: str,
|
|
571
663
|
name: Optional[str],
|
|
664
|
+
verifier: str,
|
|
572
665
|
mode: str,
|
|
573
666
|
corpus_path: Optional[str],
|
|
574
667
|
) -> int:
|
|
@@ -578,6 +671,7 @@ def suite_create_command(
|
|
|
578
671
|
suite = create_suite_from_trace_files(
|
|
579
672
|
trace_json,
|
|
580
673
|
name=name,
|
|
674
|
+
verifier=verifier,
|
|
581
675
|
mode=mode,
|
|
582
676
|
corpus_path=corpus_path,
|
|
583
677
|
)
|
|
@@ -587,6 +681,8 @@ def suite_create_command(
|
|
|
587
681
|
|
|
588
682
|
click.echo("Suite: %s" % written)
|
|
589
683
|
click.echo("Cases: %s" % len(suite.get("cases") or []))
|
|
684
|
+
if suite.get("verifier") == "hybrid_v2":
|
|
685
|
+
click.echo("Verifier: hybrid_v2 (experimental; retained for add and run)")
|
|
590
686
|
click.echo("Policy: saved cases must pass on replay")
|
|
591
687
|
return 0
|
|
592
688
|
|
|
@@ -1582,10 +1678,18 @@ def _print_verify_result(
|
|
|
1582
1678
|
result: dict,
|
|
1583
1679
|
*,
|
|
1584
1680
|
json_output: bool,
|
|
1681
|
+
triage_output: bool = False,
|
|
1585
1682
|
written_report: Optional[str],
|
|
1586
1683
|
fail_on: tuple[str, ...] = (),
|
|
1587
1684
|
) -> int:
|
|
1588
1685
|
fail_messages = _verify_failures(result, fail_on)
|
|
1686
|
+
if triage_output:
|
|
1687
|
+
click.echo(render_triage_summary(result))
|
|
1688
|
+
if written_report:
|
|
1689
|
+
click.echo("Report: %s" % written_report)
|
|
1690
|
+
for message in fail_messages:
|
|
1691
|
+
click.echo("Verification failed: %s" % message, err=True)
|
|
1692
|
+
return 1 if fail_messages else 0
|
|
1589
1693
|
if json_output:
|
|
1590
1694
|
if written_report:
|
|
1591
1695
|
click.echo("Report: %s" % written_report, err=True)
|
|
@@ -1755,10 +1859,41 @@ def eval_command(
|
|
|
1755
1859
|
|
|
1756
1860
|
|
|
1757
1861
|
@cli.command()
|
|
1862
|
+
@click.argument("scenario", required=False)
|
|
1758
1863
|
@click.option("--dataset", default="refund_policy", show_default=True, help="Demo dataset name or path.")
|
|
1759
1864
|
@click.option("--strategy", default="adaptive", show_default=True, help="Demo retrieval strategy.")
|
|
1760
1865
|
@click.pass_context
|
|
1761
|
-
def demo(ctx: click.Context, dataset: str, strategy: str) -> None:
|
|
1866
|
+
def demo(ctx: click.Context, scenario: Optional[str], dataset: str, strategy: str) -> None:
|
|
1867
|
+
if scenario:
|
|
1868
|
+
normalized = scenario.strip().lower().replace("_", "-")
|
|
1869
|
+
if normalized != "groundedness-gap":
|
|
1870
|
+
raise click.UsageError("Unknown demo scenario %r. Use 'groundedness-gap'." % scenario)
|
|
1871
|
+
groundedness_demo = run_groundedness_gap_demo()
|
|
1872
|
+
rag = groundedness_demo.result.get("rag") or {}
|
|
1873
|
+
claim = ((rag.get("claims") or [{}])[0])
|
|
1874
|
+
summary = rag.get("summary") or {}
|
|
1875
|
+
abstention = rag.get("abstention") or {}
|
|
1876
|
+
root = claim.get("root_cause") or {}
|
|
1877
|
+
trace_payload = json.loads(Path(groundedness_demo.trace_path).read_text(encoding="utf-8"))
|
|
1878
|
+
click.echo("Query: %s" % rag.get("query"))
|
|
1879
|
+
click.echo("Answer: %s" % rag.get("answer"))
|
|
1880
|
+
for context in trace_payload.get("contexts") or []:
|
|
1881
|
+
metadata = context.get("metadata") or {}
|
|
1882
|
+
condition = metadata.get("freshness") or "unknown"
|
|
1883
|
+
click.echo(
|
|
1884
|
+
"Retrieved source [%s, %s]: %s"
|
|
1885
|
+
% (context.get("id"), condition, context.get("text"))
|
|
1886
|
+
)
|
|
1887
|
+
click.echo("Citation: atlas_policy_2024")
|
|
1888
|
+
click.echo("Support verdict: %s" % claim.get("verdict"))
|
|
1889
|
+
click.echo("Citation status: %s" % claim.get("citation_status"))
|
|
1890
|
+
click.echo("Source condition: %s" % claim.get("source_status"))
|
|
1891
|
+
click.echo("Abstain: %s" % str(bool(abstention.get("should_abstain"))).lower())
|
|
1892
|
+
click.echo("Root cause: %s" % root.get("label"))
|
|
1893
|
+
click.echo("Repair: %s" % (root.get("suggested_fix") or summary.get("suggested_fix")))
|
|
1894
|
+
click.echo("Trace: %s" % groundedness_demo.trace_path)
|
|
1895
|
+
click.echo("Regression test: %s" % groundedness_demo.regression_test_path)
|
|
1896
|
+
return
|
|
1762
1897
|
client = _client(ctx)
|
|
1763
1898
|
config = _load(ctx)
|
|
1764
1899
|
report_path = Path(config.local_store_dir) / "reports" / ("%s_demo.html" % Path(dataset).name)
|
|
@@ -1878,6 +2013,8 @@ def _print_suite_result(
|
|
|
1878
2013
|
) -> None:
|
|
1879
2014
|
summary = result.get("summary") or {}
|
|
1880
2015
|
click.echo("Suite: %s" % result.get("suite_name"))
|
|
2016
|
+
if result.get("verifier") == "hybrid_v2":
|
|
2017
|
+
click.echo("Verifier: hybrid_v2 (experimental)")
|
|
1881
2018
|
click.echo("Status: %s" % summary.get("status"))
|
|
1882
2019
|
click.echo("Cases: %s" % summary.get("total_cases"))
|
|
1883
2020
|
click.echo("Passed: %s" % summary.get("passed"))
|
|
@@ -19,6 +19,7 @@ DEFAULT_PROFILE_ID = "full_v1"
|
|
|
19
19
|
SCHEMA_FILES = {
|
|
20
20
|
"TraceV1": "trace-v1.schema.json",
|
|
21
21
|
"ClaimVerificationV1": "claim-verification-v1.schema.json",
|
|
22
|
+
"ClaimVerificationHybridV2": "claim-verification-hybrid-v2.schema.json",
|
|
22
23
|
"DiagnosisV1": "diagnosis-v1.schema.json",
|
|
23
24
|
"RepairPlanV1": "repair-plan-v1.schema.json",
|
|
24
25
|
"RegressionCaseV1": "regression-case-v1.schema.json",
|
|
@@ -1,5 +1,6 @@
|
|
|
1
1
|
from __future__ import annotations
|
|
2
2
|
|
|
3
|
+
import json
|
|
3
4
|
import re
|
|
4
5
|
from dataclasses import dataclass
|
|
5
6
|
from pathlib import Path
|
|
@@ -8,6 +9,7 @@ from typing import Any, Iterable
|
|
|
8
9
|
from contexttrace.client import ContextTrace
|
|
9
10
|
from contexttrace.demo_data import load_demo_dataset
|
|
10
11
|
from contexttrace.report import ReportGenerator
|
|
12
|
+
from contexttrace.diagnose import diagnose_payload, write_diagnosis_regression_test
|
|
11
13
|
|
|
12
14
|
|
|
13
15
|
STRATEGY_TOP_K = {
|
|
@@ -32,6 +34,71 @@ class DemoRun:
|
|
|
32
34
|
summary: dict[str, Any]
|
|
33
35
|
|
|
34
36
|
|
|
37
|
+
@dataclass(frozen=True)
|
|
38
|
+
class GroundednessGapDemo:
|
|
39
|
+
trace_path: str
|
|
40
|
+
regression_test_path: str
|
|
41
|
+
result: dict[str, Any]
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def run_groundedness_gap_demo(
|
|
45
|
+
*, output_dir: str | Path = ".contexttrace/demo/groundedness-gap"
|
|
46
|
+
) -> GroundednessGapDemo:
|
|
47
|
+
"""Run the paper's grounded-but-stale example without network access."""
|
|
48
|
+
output = Path(output_dir)
|
|
49
|
+
output.mkdir(parents=True, exist_ok=True)
|
|
50
|
+
trace = {
|
|
51
|
+
"query": "What is Atlas's current cancellation window?",
|
|
52
|
+
"answer": "Atlas allows cancellation within 43 days.",
|
|
53
|
+
"contexts": [
|
|
54
|
+
{
|
|
55
|
+
"id": "atlas_policy_2024",
|
|
56
|
+
"text": "Atlas allows cancellation within 43 days.",
|
|
57
|
+
"metadata": {
|
|
58
|
+
"canonical": False,
|
|
59
|
+
"freshness": "stale",
|
|
60
|
+
"source_group": "atlas_cancellation_policy",
|
|
61
|
+
"source_version": "2024.1",
|
|
62
|
+
},
|
|
63
|
+
},
|
|
64
|
+
{
|
|
65
|
+
"id": "atlas_policy_2026",
|
|
66
|
+
"text": "The current Atlas policy allows cancellation within 42 days.",
|
|
67
|
+
"metadata": {
|
|
68
|
+
"canonical": True,
|
|
69
|
+
"freshness": "current",
|
|
70
|
+
"source_authority": "official",
|
|
71
|
+
"source_group": "atlas_cancellation_policy",
|
|
72
|
+
"source_version": "2026.1",
|
|
73
|
+
},
|
|
74
|
+
},
|
|
75
|
+
],
|
|
76
|
+
"citations": [
|
|
77
|
+
{
|
|
78
|
+
"claim": "Atlas allows cancellation within 43 days.",
|
|
79
|
+
"source_id": "atlas_policy_2024",
|
|
80
|
+
}
|
|
81
|
+
],
|
|
82
|
+
"metadata": {"demo": "groundedness-gap", "synthetic": True},
|
|
83
|
+
}
|
|
84
|
+
trace_path = output / "groundedness_gap_trace.json"
|
|
85
|
+
trace_path.write_text(json.dumps(trace, indent=2) + "\n", encoding="utf-8")
|
|
86
|
+
result = diagnose_payload(trace, mode="semantic", trace_path=str(trace_path))
|
|
87
|
+
test_path = output / "test_groundedness_gap_diagnosis.py"
|
|
88
|
+
write_diagnosis_regression_test(
|
|
89
|
+
trace_path,
|
|
90
|
+
result,
|
|
91
|
+
output_path=test_path,
|
|
92
|
+
mode="semantic",
|
|
93
|
+
overwrite=True,
|
|
94
|
+
)
|
|
95
|
+
return GroundednessGapDemo(
|
|
96
|
+
trace_path=str(trace_path),
|
|
97
|
+
regression_test_path=str(test_path),
|
|
98
|
+
result=result,
|
|
99
|
+
)
|
|
100
|
+
|
|
101
|
+
|
|
35
102
|
def run_demo_dataset(
|
|
36
103
|
*,
|
|
37
104
|
dataset: str,
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "https://json-schema.org/draft/2020-12/schema",
|
|
3
|
+
"$id": "https://contexttrace.dev/schemas/claim-verification-hybrid-v2.schema.json",
|
|
4
|
+
"title": "ClaimVerificationHybridV2",
|
|
5
|
+
"type": "object",
|
|
6
|
+
"required": [
|
|
7
|
+
"schema_version",
|
|
8
|
+
"taxonomy_version",
|
|
9
|
+
"verifier_version",
|
|
10
|
+
"profile_id",
|
|
11
|
+
"diagnostic_reasoner_version",
|
|
12
|
+
"experimental",
|
|
13
|
+
"query",
|
|
14
|
+
"answer",
|
|
15
|
+
"summary",
|
|
16
|
+
"claims",
|
|
17
|
+
"abstention",
|
|
18
|
+
"diagnostics",
|
|
19
|
+
"capabilities",
|
|
20
|
+
"limitations",
|
|
21
|
+
"truncation"
|
|
22
|
+
],
|
|
23
|
+
"properties": {
|
|
24
|
+
"schema_version": {"const": "2.0"},
|
|
25
|
+
"taxonomy_version": {"const": "contexttrace-hybrid-v2.0"},
|
|
26
|
+
"verifier_version": {"const": "hybrid_v2"},
|
|
27
|
+
"profile_id": {"type": "string", "pattern": "^hybrid_v2"},
|
|
28
|
+
"diagnostic_reasoner_version": {"const": "evidence_chain_v2"},
|
|
29
|
+
"experimental": {"const": true},
|
|
30
|
+
"query": {"type": "string"},
|
|
31
|
+
"answer": {"type": "string"},
|
|
32
|
+
"summary": {"type": "object"},
|
|
33
|
+
"claims": {"type": "array", "items": {"type": "object"}},
|
|
34
|
+
"abstention": {"type": "object"},
|
|
35
|
+
"diagnostics": {"type": "object"},
|
|
36
|
+
"metadata": {"type": "object"},
|
|
37
|
+
"verification_profile": {"type": "object"},
|
|
38
|
+
"capabilities": {"type": "object"},
|
|
39
|
+
"limitations": {"type": "array", "items": {"type": "string"}},
|
|
40
|
+
"truncation": {"type": "object"}
|
|
41
|
+
},
|
|
42
|
+
"additionalProperties": false
|
|
43
|
+
}
|