verdictkit 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verdictkit-0.1.0/PKG-INFO +191 -0
- verdictkit-0.1.0/README.md +163 -0
- verdictkit-0.1.0/pyproject.toml +47 -0
- verdictkit-0.1.0/setup.cfg +4 -0
- verdictkit-0.1.0/tests/test_checks.py +157 -0
- verdictkit-0.1.0/tests/test_cli.py +87 -0
- verdictkit-0.1.0/tests/test_client.py +160 -0
- verdictkit-0.1.0/tests/test_contracts.py +59 -0
- verdictkit-0.1.0/tests/test_evidence.py +155 -0
- verdictkit-0.1.0/tests/test_verdict.py +97 -0
- verdictkit-0.1.0/verdictkit/__init__.py +40 -0
- verdictkit-0.1.0/verdictkit/checks.py +92 -0
- verdictkit-0.1.0/verdictkit/cli.py +54 -0
- verdictkit-0.1.0/verdictkit/client.py +124 -0
- verdictkit-0.1.0/verdictkit/contracts.py +37 -0
- verdictkit-0.1.0/verdictkit/evidence.py +166 -0
- verdictkit-0.1.0/verdictkit/verdict.py +63 -0
- verdictkit-0.1.0/verdictkit.egg-info/PKG-INFO +191 -0
- verdictkit-0.1.0/verdictkit.egg-info/SOURCES.txt +21 -0
- verdictkit-0.1.0/verdictkit.egg-info/dependency_links.txt +1 -0
- verdictkit-0.1.0/verdictkit.egg-info/entry_points.txt +2 -0
- verdictkit-0.1.0/verdictkit.egg-info/requires.txt +5 -0
- verdictkit-0.1.0/verdictkit.egg-info/top_level.txt +1 -0
|
@@ -0,0 +1,191 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: verdictkit
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Non-custodial verification for agent work: did the agent do what it said?
|
|
5
|
+
Author: agentbuilt
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/mattedwardseo/verdictkit
|
|
8
|
+
Project-URL: Documentation, https://github.com/mattedwardseo/verdictkit#readme
|
|
9
|
+
Project-URL: Repository, https://github.com/mattedwardseo/verdictkit
|
|
10
|
+
Project-URL: Issues, https://github.com/mattedwardseo/verdictkit/issues
|
|
11
|
+
Keywords: agents,verification,ai-safety,auditing,deterministic-checks
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Software Development :: Testing
|
|
21
|
+
Classifier: Topic :: Security
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Description-Content-Type: text/markdown
|
|
24
|
+
Requires-Dist: requests>=2.28
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
27
|
+
Requires-Dist: build; extra == "dev"
|
|
28
|
+
|
|
29
|
+
# verdictkit
|
|
30
|
+
|
|
31
|
+
Non-custodial verification for agent work. Answers one question: **did the
|
|
32
|
+
agent do what it said it would?**
|
|
33
|
+
|
|
34
|
+
The product is the *verification layer*, not custody: a machine-readable
|
|
35
|
+
contract (intended action + checkable success criteria), deterministic
|
|
36
|
+
checks against the world, and a rubric verdict — `PASS` / `FLAG` / `FAIL`
|
|
37
|
+
— with evidence. No money moves through it, so there is no money-transmitter
|
|
38
|
+
licensing wall. That is the whole strategic point.
|
|
39
|
+
|
|
40
|
+
**Status: 0.1.0, launch-ready, NOT published to PyPI.** Do not run the
|
|
41
|
+
publish command without approval.
|
|
42
|
+
|
|
43
|
+
## Quickstart
|
|
44
|
+
|
|
45
|
+
```bash
|
|
46
|
+
pip install verdictkit # not on PyPI yet; use: pip install .
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
### Local verification (free, unlimited, no key, no network)
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
import verdictkit as vk
|
|
53
|
+
|
|
54
|
+
contract = vk.define_contract(
|
|
55
|
+
task_id="research-042",
|
|
56
|
+
agent_id="agent-7f3a",
|
|
57
|
+
intended_action="Summarize the Q3 incident reports into incidents.md",
|
|
58
|
+
success_criteria=[
|
|
59
|
+
{"id": "c1", "type": "file_exists",
|
|
60
|
+
"path": "incidents.md",
|
|
61
|
+
"description": "summary file was written"},
|
|
62
|
+
{"id": "c2", "type": "pattern_count_gte",
|
|
63
|
+
"path": "incidents.md",
|
|
64
|
+
"pattern": r"^## Incident",
|
|
65
|
+
"min_count": 3,
|
|
66
|
+
"description": "covers at least 3 incidents"},
|
|
67
|
+
],
|
|
68
|
+
unverifiable_claims=["tone is executive-appropriate"], # -> FLAG, honestly
|
|
69
|
+
)
|
|
70
|
+
|
|
71
|
+
result = vk.verify_run(contract, base_path="./output")
|
|
72
|
+
print(result["verdict"]) # PASS | FLAG | FAIL
|
|
73
|
+
print(result["reason"]) # evidence, always
|
|
74
|
+
```
|
|
75
|
+
|
|
76
|
+
Or from the shell:
|
|
77
|
+
|
|
78
|
+
```bash
|
|
79
|
+
verdictkit verify contract.json --base-path ./output -v
|
|
80
|
+
# exit code: 0 = PASS, 1 = FAIL, 2 = FLAG, 3 = usage/IO error
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### Hosted verification (reputation ledger + re-verification)
|
|
84
|
+
|
|
85
|
+
Local mode checks *your* machine. The hosted tier lets a third party trust
|
|
86
|
+
the verdict: you capture evidence locally, the server **re-runs the
|
|
87
|
+
deterministic checks against the evidence** — it never trusts the agent's
|
|
88
|
+
claims alone.
|
|
89
|
+
|
|
90
|
+
```python
|
|
91
|
+
import verdictkit as vk
|
|
92
|
+
|
|
93
|
+
client = vk.Client(api_key="vk_live_...") # base_url=... to override
|
|
94
|
+
contract_id = client.create_contract(contract) # -> "vc_..."
|
|
95
|
+
bundle = vk.collect_evidence(contract, base_path="./output")
|
|
96
|
+
verdict = client.submit_run(contract_id, bundle) # server-side verdict dict
|
|
97
|
+
|
|
98
|
+
client.get_verdict(contract_id) # latest verdict
|
|
99
|
+
client.list_verdicts(agent_id="agent-7f3a") # reputation ledger
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Errors are clean: `vk.AuthError` on bad credentials (401/403),
|
|
103
|
+
`vk.VerdictkitError` on network trouble, timeouts, or bad responses.
|
|
104
|
+
|
|
105
|
+
## API reference
|
|
106
|
+
|
|
107
|
+
| Symbol | Kind | Description |
|
|
108
|
+
|---|---|---|
|
|
109
|
+
| `vk.define_contract(task_id, agent_id, intended_action, success_criteria, scope=None, unverifiable_claims=None, principal="unknown")` | function | Build a validated contract dict. Raises `ValueError` on unknown check types or criteria missing `id`/`type`. |
|
|
110
|
+
| `vk.verify_run(contract, base_path=None)` | function | Run all checks locally, return verdict dict. No key, no network. `base_path` resolves relative criterion paths. |
|
|
111
|
+
| `vk.Verdict` | class | Constants `PASS`, `FLAG`, `FAIL`. |
|
|
112
|
+
| `vk.collect_evidence(contract, base_path=None)` | function | Capture a content-addressed evidence bundle (file bytes + sha256, dir listings, parsed JSON, sqlite row counts). Local only, no network. |
|
|
113
|
+
| `vk.verify_bundle_hash(bundle)` | function | Recompute `bundle_hash`; `False` means the bundle was tampered with. |
|
|
114
|
+
| `vk.Client(api_key, base_url=..., timeout=30.0)` | class | Hosted API client. `create_contract(contract) -> contract_id`, `submit_run(contract_id, evidence_bundle) -> verdict dict`, `get_verdict(contract_id) -> dict`, `list_verdicts(agent_id=None, limit=100) -> list`. Raises `ValueError` without an API key. |
|
|
115
|
+
| `vk.VerdictkitError` | exception | Base error: network, timeout, bad server response. |
|
|
116
|
+
| `vk.AuthError(VerdictkitError)` | exception | 401/403 from the API. |
|
|
117
|
+
|
|
118
|
+
### Verdict shape
|
|
119
|
+
|
|
120
|
+
```python
|
|
121
|
+
{"contract_id": ..., "task_id": ..., "agent_id": ...,
|
|
122
|
+
"verdict": "PASS" | "FLAG" | "FAIL",
|
|
123
|
+
"reason": "human-readable evidence summary",
|
|
124
|
+
"check_results": [{"check_id": ..., "type": ..., "status": "pass"|"fail"|"error",
|
|
125
|
+
"evidence": ...}, ...],
|
|
126
|
+
"unverifiable_claims": [...]}
|
|
127
|
+
```
|
|
128
|
+
|
|
129
|
+
### Check types
|
|
130
|
+
|
|
131
|
+
`file_exists` · `file_nonempty` · `file_contains` (regex) ·
|
|
132
|
+
`pattern_count_gte` · `dir_exists` · `json_valid` · `line_count_gte` ·
|
|
133
|
+
`sqlite_table_has_rows` · `path_absent` (scope guard)
|
|
134
|
+
|
|
135
|
+
## The rubric
|
|
136
|
+
|
|
137
|
+
- **FAIL** — any check fails. Fail dominates everything else.
|
|
138
|
+
- **FLAG** — no failures, but something couldn't be verified: a check
|
|
139
|
+
errored, a claim was declared unverifiable, or there was nothing
|
|
140
|
+
checkable at all.
|
|
141
|
+
- **PASS** — every check passed cleanly, and nothing was left unverifiable.
|
|
142
|
+
|
|
143
|
+
Design rule: a check that cannot run is an *error*, never a pass. A run
|
|
144
|
+
with nothing checkable can at best be *flagged*. Absence of evidence is
|
|
145
|
+
never evidence of success.
|
|
146
|
+
|
|
147
|
+
## Trust model
|
|
148
|
+
|
|
149
|
+
1. **Contracts are machine-readable promises.** What the agent intended to
|
|
150
|
+
do, written before the run, with deterministic success criteria.
|
|
151
|
+
2. **Evidence is content-addressed.** `collect_evidence` captures file
|
|
152
|
+
bytes with their sha256, directory listings, and query results. The
|
|
153
|
+
bundle carries a `bundle_hash` over canonical JSON — the server
|
|
154
|
+
recomputes every hash; tampering breaks them.
|
|
155
|
+
3. **The server re-runs checks; it never trusts claims.** Hosted
|
|
156
|
+
verification executes the same deterministic check engine against the
|
|
157
|
+
bundle's content. Claims without evidence stay in
|
|
158
|
+
`unverifiable_claims` and can only produce FLAG.
|
|
159
|
+
4. **Truncation is honest.** Artifacts are capped (256 KiB file content,
|
|
160
|
+
2000 dir entries). Truncated artifacts are marked, and checks the
|
|
161
|
+
server cannot fully re-run become FLAG — never PASS.
|
|
162
|
+
|
|
163
|
+
## What verification is NOT
|
|
164
|
+
|
|
165
|
+
- **Not a guarantee of quality.** PASS means the deterministic criteria
|
|
166
|
+
were met — the file exists, has ≥3 sections, parses as JSON. It says
|
|
167
|
+
nothing about whether the writing is good, the code is correct, or the
|
|
168
|
+
summary is accurate. Put anything subjective in `unverifiable_claims`
|
|
169
|
+
and accept the FLAG honestly.
|
|
170
|
+
- **Not an oracle.** Checks only see what they can read: files, dirs,
|
|
171
|
+
sqlite. They cannot verify "the user is happy" or "the bug is fixed"
|
|
172
|
+
unless that was operationalized into a checkable artifact.
|
|
173
|
+
- **Not custody, not escrow.** No money moves through verdictkit; it
|
|
174
|
+
attests to work done, it does not hold funds or release payment.
|
|
175
|
+
- **Not tamper-proof on a compromised machine.** Local verification trusts
|
|
176
|
+
the local filesystem. The hosted tier raises the bar (content-addressed
|
|
177
|
+
bundles, server-side re-runs, reputation ledger) but a fully
|
|
178
|
+
compromised agent host can still fake the inputs. Verification makes
|
|
179
|
+
lying *auditable*, not impossible.
|
|
180
|
+
|
|
181
|
+
## Development
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
pip install -e ".[dev]" # dev extras: pytest, build
|
|
185
|
+
python3 -m pytest # 59 tests, all local, no network
|
|
186
|
+
python3 -m build # wheel/sdist -- DO NOT upload without approval
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
## License
|
|
190
|
+
|
|
191
|
+
MIT. See `LICENSE`.
|
|
@@ -0,0 +1,163 @@
|
|
|
1
|
+
# verdictkit
|
|
2
|
+
|
|
3
|
+
Non-custodial verification for agent work. Answers one question: **did the
|
|
4
|
+
agent do what it said it would?**
|
|
5
|
+
|
|
6
|
+
The product is the *verification layer*, not custody: a machine-readable
|
|
7
|
+
contract (intended action + checkable success criteria), deterministic
|
|
8
|
+
checks against the world, and a rubric verdict — `PASS` / `FLAG` / `FAIL`
|
|
9
|
+
— with evidence. No money moves through it, so there is no money-transmitter
|
|
10
|
+
licensing wall. That is the whole strategic point.
|
|
11
|
+
|
|
12
|
+
**Status: 0.1.0, launch-ready, NOT published to PyPI.** Do not run the
|
|
13
|
+
publish command without approval.
|
|
14
|
+
|
|
15
|
+
## Quickstart
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
pip install verdictkit # not on PyPI yet; use: pip install .
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
### Local verification (free, unlimited, no key, no network)
|
|
22
|
+
|
|
23
|
+
```python
|
|
24
|
+
import verdictkit as vk
|
|
25
|
+
|
|
26
|
+
contract = vk.define_contract(
|
|
27
|
+
task_id="research-042",
|
|
28
|
+
agent_id="agent-7f3a",
|
|
29
|
+
intended_action="Summarize the Q3 incident reports into incidents.md",
|
|
30
|
+
success_criteria=[
|
|
31
|
+
{"id": "c1", "type": "file_exists",
|
|
32
|
+
"path": "incidents.md",
|
|
33
|
+
"description": "summary file was written"},
|
|
34
|
+
{"id": "c2", "type": "pattern_count_gte",
|
|
35
|
+
"path": "incidents.md",
|
|
36
|
+
"pattern": r"^## Incident",
|
|
37
|
+
"min_count": 3,
|
|
38
|
+
"description": "covers at least 3 incidents"},
|
|
39
|
+
],
|
|
40
|
+
unverifiable_claims=["tone is executive-appropriate"], # -> FLAG, honestly
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
result = vk.verify_run(contract, base_path="./output")
|
|
44
|
+
print(result["verdict"]) # PASS | FLAG | FAIL
|
|
45
|
+
print(result["reason"]) # evidence, always
|
|
46
|
+
```
|
|
47
|
+
|
|
48
|
+
Or from the shell:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
verdictkit verify contract.json --base-path ./output -v
|
|
52
|
+
# exit code: 0 = PASS, 1 = FAIL, 2 = FLAG, 3 = usage/IO error
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
### Hosted verification (reputation ledger + re-verification)
|
|
56
|
+
|
|
57
|
+
Local mode checks *your* machine. The hosted tier lets a third party trust
|
|
58
|
+
the verdict: you capture evidence locally, the server **re-runs the
|
|
59
|
+
deterministic checks against the evidence** — it never trusts the agent's
|
|
60
|
+
claims alone.
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
import verdictkit as vk
|
|
64
|
+
|
|
65
|
+
client = vk.Client(api_key="vk_live_...") # base_url=... to override
|
|
66
|
+
contract_id = client.create_contract(contract) # -> "vc_..."
|
|
67
|
+
bundle = vk.collect_evidence(contract, base_path="./output")
|
|
68
|
+
verdict = client.submit_run(contract_id, bundle) # server-side verdict dict
|
|
69
|
+
|
|
70
|
+
client.get_verdict(contract_id) # latest verdict
|
|
71
|
+
client.list_verdicts(agent_id="agent-7f3a") # reputation ledger
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
Errors are clean: `vk.AuthError` on bad credentials (401/403),
|
|
75
|
+
`vk.VerdictkitError` on network trouble, timeouts, or bad responses.
|
|
76
|
+
|
|
77
|
+
## API reference
|
|
78
|
+
|
|
79
|
+
| Symbol | Kind | Description |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| `vk.define_contract(task_id, agent_id, intended_action, success_criteria, scope=None, unverifiable_claims=None, principal="unknown")` | function | Build a validated contract dict. Raises `ValueError` on unknown check types or criteria missing `id`/`type`. |
|
|
82
|
+
| `vk.verify_run(contract, base_path=None)` | function | Run all checks locally, return verdict dict. No key, no network. `base_path` resolves relative criterion paths. |
|
|
83
|
+
| `vk.Verdict` | class | Constants `PASS`, `FLAG`, `FAIL`. |
|
|
84
|
+
| `vk.collect_evidence(contract, base_path=None)` | function | Capture a content-addressed evidence bundle (file bytes + sha256, dir listings, parsed JSON, sqlite row counts). Local only, no network. |
|
|
85
|
+
| `vk.verify_bundle_hash(bundle)` | function | Recompute `bundle_hash`; `False` means the bundle was tampered with. |
|
|
86
|
+
| `vk.Client(api_key, base_url=..., timeout=30.0)` | class | Hosted API client. `create_contract(contract) -> contract_id`, `submit_run(contract_id, evidence_bundle) -> verdict dict`, `get_verdict(contract_id) -> dict`, `list_verdicts(agent_id=None, limit=100) -> list`. Raises `ValueError` without an API key. |
|
|
87
|
+
| `vk.VerdictkitError` | exception | Base error: network, timeout, bad server response. |
|
|
88
|
+
| `vk.AuthError(VerdictkitError)` | exception | 401/403 from the API. |
|
|
89
|
+
|
|
90
|
+
### Verdict shape
|
|
91
|
+
|
|
92
|
+
```python
|
|
93
|
+
{"contract_id": ..., "task_id": ..., "agent_id": ...,
|
|
94
|
+
"verdict": "PASS" | "FLAG" | "FAIL",
|
|
95
|
+
"reason": "human-readable evidence summary",
|
|
96
|
+
"check_results": [{"check_id": ..., "type": ..., "status": "pass"|"fail"|"error",
|
|
97
|
+
"evidence": ...}, ...],
|
|
98
|
+
"unverifiable_claims": [...]}
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Check types
|
|
102
|
+
|
|
103
|
+
`file_exists` · `file_nonempty` · `file_contains` (regex) ·
|
|
104
|
+
`pattern_count_gte` · `dir_exists` · `json_valid` · `line_count_gte` ·
|
|
105
|
+
`sqlite_table_has_rows` · `path_absent` (scope guard)
|
|
106
|
+
|
|
107
|
+
## The rubric
|
|
108
|
+
|
|
109
|
+
- **FAIL** — any check fails. Fail dominates everything else.
|
|
110
|
+
- **FLAG** — no failures, but something couldn't be verified: a check
|
|
111
|
+
errored, a claim was declared unverifiable, or there was nothing
|
|
112
|
+
checkable at all.
|
|
113
|
+
- **PASS** — every check passed cleanly, and nothing was left unverifiable.
|
|
114
|
+
|
|
115
|
+
Design rule: a check that cannot run is an *error*, never a pass. A run
|
|
116
|
+
with nothing checkable can at best be *flagged*. Absence of evidence is
|
|
117
|
+
never evidence of success.
|
|
118
|
+
|
|
119
|
+
## Trust model
|
|
120
|
+
|
|
121
|
+
1. **Contracts are machine-readable promises.** What the agent intended to
|
|
122
|
+
do, written before the run, with deterministic success criteria.
|
|
123
|
+
2. **Evidence is content-addressed.** `collect_evidence` captures file
|
|
124
|
+
bytes with their sha256, directory listings, and query results. The
|
|
125
|
+
bundle carries a `bundle_hash` over canonical JSON — the server
|
|
126
|
+
recomputes every hash; tampering breaks them.
|
|
127
|
+
3. **The server re-runs checks; it never trusts claims.** Hosted
|
|
128
|
+
verification executes the same deterministic check engine against the
|
|
129
|
+
bundle's content. Claims without evidence stay in
|
|
130
|
+
`unverifiable_claims` and can only produce FLAG.
|
|
131
|
+
4. **Truncation is honest.** Artifacts are capped (256 KiB file content,
|
|
132
|
+
2000 dir entries). Truncated artifacts are marked, and checks the
|
|
133
|
+
server cannot fully re-run become FLAG — never PASS.
|
|
134
|
+
|
|
135
|
+
## What verification is NOT
|
|
136
|
+
|
|
137
|
+
- **Not a guarantee of quality.** PASS means the deterministic criteria
|
|
138
|
+
were met — the file exists, has ≥3 sections, parses as JSON. It says
|
|
139
|
+
nothing about whether the writing is good, the code is correct, or the
|
|
140
|
+
summary is accurate. Put anything subjective in `unverifiable_claims`
|
|
141
|
+
and accept the FLAG honestly.
|
|
142
|
+
- **Not an oracle.** Checks only see what they can read: files, dirs,
|
|
143
|
+
sqlite. They cannot verify "the user is happy" or "the bug is fixed"
|
|
144
|
+
unless that was operationalized into a checkable artifact.
|
|
145
|
+
- **Not custody, not escrow.** No money moves through verdictkit; it
|
|
146
|
+
attests to work done, it does not hold funds or release payment.
|
|
147
|
+
- **Not tamper-proof on a compromised machine.** Local verification trusts
|
|
148
|
+
the local filesystem. The hosted tier raises the bar (content-addressed
|
|
149
|
+
bundles, server-side re-runs, reputation ledger) but a fully
|
|
150
|
+
compromised agent host can still fake the inputs. Verification makes
|
|
151
|
+
lying *auditable*, not impossible.
|
|
152
|
+
|
|
153
|
+
## Development
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
pip install -e ".[dev]" # dev extras: pytest, build
|
|
157
|
+
python3 -m pytest # 59 tests, all local, no network
|
|
158
|
+
python3 -m build # wheel/sdist -- DO NOT upload without approval
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
## License
|
|
162
|
+
|
|
163
|
+
MIT. See `LICENSE`.
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "verdictkit"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Non-custodial verification for agent work: did the agent do what it said?"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.10"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "agentbuilt" }]
|
|
13
|
+
keywords = ["agents", "verification", "ai-safety", "auditing", "deterministic-checks"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Topic :: Software Development :: Testing",
|
|
24
|
+
"Topic :: Security",
|
|
25
|
+
]
|
|
26
|
+
dependencies = [
|
|
27
|
+
"requests>=2.28",
|
|
28
|
+
]
|
|
29
|
+
|
|
30
|
+
[project.optional-dependencies]
|
|
31
|
+
dev = ["pytest>=7", "build"]
|
|
32
|
+
# PLACEHOLDER URLs -- replace with the real repo/docs once they exist.
|
|
33
|
+
[project.urls]
|
|
34
|
+
Homepage = "https://github.com/mattedwardseo/verdictkit"
|
|
35
|
+
Documentation = "https://github.com/mattedwardseo/verdictkit#readme"
|
|
36
|
+
Repository = "https://github.com/mattedwardseo/verdictkit"
|
|
37
|
+
Issues = "https://github.com/mattedwardseo/verdictkit/issues"
|
|
38
|
+
|
|
39
|
+
[project.scripts]
|
|
40
|
+
verdictkit = "verdictkit.cli:main"
|
|
41
|
+
|
|
42
|
+
[tool.setuptools.packages.find]
|
|
43
|
+
where = ["."]
|
|
44
|
+
include = ["verdictkit*"]
|
|
45
|
+
|
|
46
|
+
[tool.pytest.ini_options]
|
|
47
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,157 @@
|
|
|
1
|
+
"""All 9 check types: pass, fail, and error paths."""
|
|
2
|
+
import json
|
|
3
|
+
import os
|
|
4
|
+
import sqlite3
|
|
5
|
+
|
|
6
|
+
import pytest
|
|
7
|
+
|
|
8
|
+
from verdictkit.checks import run_check
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def crit(cid, ctype, **kw):
|
|
12
|
+
return {"id": cid, "type": ctype, **kw}
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
# -- file_exists -----------------------------------------------------------
|
|
16
|
+
def test_file_exists_pass_fail(tmp_path):
|
|
17
|
+
f = tmp_path / "a.txt"
|
|
18
|
+
f.write_text("hi")
|
|
19
|
+
assert run_check(crit("c", "file_exists", path=str(f)))["status"] == "pass"
|
|
20
|
+
assert run_check(crit("c", "file_exists", path=str(tmp_path / "nope")))["status"] == "fail"
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
# -- file_nonempty ---------------------------------------------------------
|
|
24
|
+
def test_file_nonempty_pass_fail(tmp_path):
|
|
25
|
+
full = tmp_path / "full.txt"
|
|
26
|
+
full.write_text("x")
|
|
27
|
+
assert run_check(crit("c", "file_nonempty", path=str(full)))["status"] == "pass"
|
|
28
|
+
empty = tmp_path / "empty.txt"
|
|
29
|
+
empty.write_text("")
|
|
30
|
+
assert run_check(crit("c", "file_nonempty", path=str(empty)))["status"] == "fail"
|
|
31
|
+
assert run_check(crit("c", "file_nonempty", path=str(tmp_path / "nope")))["status"] == "fail"
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
# -- file_contains ---------------------------------------------------------
|
|
35
|
+
def test_file_contains_pass_fail_and_bad_regex(tmp_path):
|
|
36
|
+
f = tmp_path / "doc.md"
|
|
37
|
+
f.write_text("## Incident one\nbody\n")
|
|
38
|
+
assert run_check(crit("c", "file_contains", path=str(f),
|
|
39
|
+
pattern=r"^## Incident"))["status"] == "pass"
|
|
40
|
+
assert run_check(crit("c", "file_contains", path=str(f),
|
|
41
|
+
pattern=r"nope"))["status"] == "fail"
|
|
42
|
+
r = run_check(crit("c", "file_contains", path=str(f), pattern=r"([unclosed"))
|
|
43
|
+
assert r["status"] == "error" and "bad regex" in r["evidence"]
|
|
44
|
+
r = run_check(crit("c", "file_contains", path=str(tmp_path / "nope"),
|
|
45
|
+
pattern=r"x"))
|
|
46
|
+
assert r["status"] == "fail" # unreadable file is fail, not pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# -- dir_exists ------------------------------------------------------------
|
|
50
|
+
def test_dir_exists_pass_fail(tmp_path):
|
|
51
|
+
d = tmp_path / "sub"
|
|
52
|
+
d.mkdir()
|
|
53
|
+
assert run_check(crit("c", "dir_exists", path=str(d)))["status"] == "pass"
|
|
54
|
+
assert run_check(crit("c", "dir_exists", path=str(tmp_path / "nope")))["status"] == "fail"
|
|
55
|
+
f = tmp_path / "file.txt"
|
|
56
|
+
f.write_text("x")
|
|
57
|
+
assert run_check(crit("c", "dir_exists", path=str(f)))["status"] == "fail" # file != dir
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
# -- json_valid ------------------------------------------------------------
|
|
61
|
+
def test_json_valid_pass_fail(tmp_path):
|
|
62
|
+
good = tmp_path / "good.json"
|
|
63
|
+
good.write_text(json.dumps({"a": 1}))
|
|
64
|
+
assert run_check(crit("c", "json_valid", path=str(good)))["status"] == "pass"
|
|
65
|
+
bad = tmp_path / "bad.json"
|
|
66
|
+
bad.write_text("{not json")
|
|
67
|
+
assert run_check(crit("c", "json_valid", path=str(bad)))["status"] == "fail"
|
|
68
|
+
assert run_check(crit("c", "json_valid", path=str(tmp_path / "nope")))["status"] == "fail"
|
|
69
|
+
|
|
70
|
+
|
|
71
|
+
# -- line_count_gte --------------------------------------------------------
|
|
72
|
+
def test_line_count_gte_pass_fail(tmp_path):
|
|
73
|
+
f = tmp_path / "lines.txt"
|
|
74
|
+
f.write_text("one\ntwo\nthree\n")
|
|
75
|
+
assert run_check(crit("c", "line_count_gte", path=str(f), min_lines=3))["status"] == "pass"
|
|
76
|
+
assert run_check(crit("c", "line_count_gte", path=str(f), min_lines=4))["status"] == "fail"
|
|
77
|
+
assert run_check(crit("c", "line_count_gte", path=str(tmp_path / "nope")))["status"] == "fail"
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
# -- sqlite_table_has_rows --------------------------------------------------
|
|
81
|
+
@pytest.fixture
|
|
82
|
+
def db_path(tmp_path):
|
|
83
|
+
p = str(tmp_path / "test.db")
|
|
84
|
+
con = sqlite3.connect(p)
|
|
85
|
+
con.execute("CREATE TABLE events (id INTEGER)")
|
|
86
|
+
con.executemany("INSERT INTO events VALUES (?)", [(1,), (2,)])
|
|
87
|
+
con.execute("CREATE TABLE empty_t (id INTEGER)")
|
|
88
|
+
con.commit()
|
|
89
|
+
con.close()
|
|
90
|
+
return p
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def test_sqlite_table_has_rows_pass_fail(db_path):
|
|
94
|
+
assert run_check(crit("c", "sqlite_table_has_rows", db=db_path,
|
|
95
|
+
table="events", min_rows=2))["status"] == "pass"
|
|
96
|
+
assert run_check(crit("c", "sqlite_table_has_rows", db=db_path,
|
|
97
|
+
table="events", min_rows=3))["status"] == "fail"
|
|
98
|
+
assert run_check(crit("c", "sqlite_table_has_rows", db=db_path,
|
|
99
|
+
table="empty_t"))["status"] == "fail"
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def test_sqlite_errors(tmp_path):
|
|
103
|
+
r = run_check(crit("c", "sqlite_table_has_rows", db=str(tmp_path / "x.db"),
|
|
104
|
+
table="events"))
|
|
105
|
+
assert r["status"] == "error" # missing db -> error, never pass
|
|
106
|
+
r = run_check(crit("c", "sqlite_table_has_rows", db="x",
|
|
107
|
+
table="events; DROP TABLE events;--"))
|
|
108
|
+
assert r["status"] == "error" and "unsafe table name" in r["evidence"]
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
# -- path_absent -----------------------------------------------------------
|
|
112
|
+
def test_path_absent_pass_fail(tmp_path):
|
|
113
|
+
f = tmp_path / "present.txt"
|
|
114
|
+
f.write_text("x")
|
|
115
|
+
assert run_check(crit("c", "path_absent", path=str(f)))["status"] == "fail"
|
|
116
|
+
assert run_check(crit("c", "path_absent", path=str(tmp_path / "gone")))["status"] == "pass"
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# -- pattern_count_gte -----------------------------------------------------
|
|
120
|
+
def test_pattern_count_gte_pass_fail_and_bad_regex(tmp_path):
|
|
121
|
+
f = tmp_path / "doc.md"
|
|
122
|
+
f.write_text("## A\n## B\n## C\n")
|
|
123
|
+
assert run_check(crit("c", "pattern_count_gte", path=str(f),
|
|
124
|
+
pattern=r"^## ", min_count=3))["status"] == "pass"
|
|
125
|
+
assert run_check(crit("c", "pattern_count_gte", path=str(f),
|
|
126
|
+
pattern=r"^## ", min_count=4))["status"] == "fail"
|
|
127
|
+
r = run_check(crit("c", "pattern_count_gte", path=str(f),
|
|
128
|
+
pattern=r"([bad", min_count=1))
|
|
129
|
+
assert r["status"] == "error" and "bad regex" in r["evidence"]
|
|
130
|
+
r = run_check(crit("c", "pattern_count_gte", path=str(tmp_path / "nope"),
|
|
131
|
+
pattern=r"x"))
|
|
132
|
+
assert r["status"] == "fail"
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# -- unknown type ----------------------------------------------------------
|
|
136
|
+
def test_unknown_type_is_error():
|
|
137
|
+
r = run_check({"id": "c", "type": "does_not_exist"})
|
|
138
|
+
assert r["status"] == "error"
|
|
139
|
+
|
|
140
|
+
|
|
141
|
+
def test_check_result_shape(tmp_path):
|
|
142
|
+
f = tmp_path / "a.txt"
|
|
143
|
+
f.write_text("hi")
|
|
144
|
+
r = run_check(crit("c1", "file_exists", path=str(f)))
|
|
145
|
+
assert set(r) == {"check_id", "type", "status", "evidence"}
|
|
146
|
+
assert r["check_id"] == "c1" and r["type"] == "file_exists"
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def test_env_var_expansion(tmp_path):
|
|
150
|
+
os.environ["VK_TEST_DIR"] = str(tmp_path)
|
|
151
|
+
try:
|
|
152
|
+
f = tmp_path / "env.txt"
|
|
153
|
+
f.write_text("hi")
|
|
154
|
+
r = run_check(crit("c", "file_exists", path="$VK_TEST_DIR/env.txt"))
|
|
155
|
+
assert r["status"] == "pass"
|
|
156
|
+
finally:
|
|
157
|
+
del os.environ["VK_TEST_DIR"]
|
|
@@ -0,0 +1,87 @@
|
|
|
1
|
+
"""CLI and public API surface."""
|
|
2
|
+
import json
|
|
3
|
+
|
|
4
|
+
import pytest
|
|
5
|
+
|
|
6
|
+
import verdictkit as vk
|
|
7
|
+
from verdictkit.cli import main
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
def test_public_surface():
|
|
11
|
+
for name in ["define_contract", "verify_run", "Verdict",
|
|
12
|
+
"collect_evidence", "verify_bundle_hash",
|
|
13
|
+
"Client", "VerdictkitError", "AuthError"]:
|
|
14
|
+
assert hasattr(vk, name), name
|
|
15
|
+
assert name in vk.__all__
|
|
16
|
+
assert vk.__version__ == "0.1.0"
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
def write_contract(path, tmp_path, filename="contract.json"):
|
|
20
|
+
p = tmp_path / filename
|
|
21
|
+
p.write_text(json.dumps(path))
|
|
22
|
+
return str(p)
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _contract_dict(out_file):
|
|
26
|
+
return {
|
|
27
|
+
"contract_id": "vc_test",
|
|
28
|
+
"task_id": "t",
|
|
29
|
+
"agent_id": "a",
|
|
30
|
+
"created_at": "2026-09-29T00:00:00+00:00",
|
|
31
|
+
"intended_action": "write file",
|
|
32
|
+
"scope": {},
|
|
33
|
+
"success_criteria": [
|
|
34
|
+
{"id": "c1", "type": "file_nonempty", "path": out_file,
|
|
35
|
+
"description": "written"}
|
|
36
|
+
],
|
|
37
|
+
"unverifiable_claims": [],
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def test_cli_verify_pass(tmp_path, capsys):
|
|
42
|
+
f = tmp_path / "out.txt"
|
|
43
|
+
f.write_text("hello")
|
|
44
|
+
p = write_contract(_contract_dict(str(f)), tmp_path)
|
|
45
|
+
code = main(["verify", p])
|
|
46
|
+
assert code == 0
|
|
47
|
+
out = capsys.readouterr().out
|
|
48
|
+
assert "verdict: PASS" in out
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def test_cli_verify_fail(tmp_path, capsys):
|
|
52
|
+
p = write_contract(_contract_dict(str(tmp_path / "missing.txt")), tmp_path)
|
|
53
|
+
code = main(["verify", p])
|
|
54
|
+
assert code == 1
|
|
55
|
+
assert "verdict: FAIL" in capsys.readouterr().out
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def test_cli_verify_flag(tmp_path, capsys):
|
|
59
|
+
c = _contract_dict(str(tmp_path / "missing.txt"))
|
|
60
|
+
c["success_criteria"] = []
|
|
61
|
+
p = write_contract(c, tmp_path)
|
|
62
|
+
code = main(["verify", p])
|
|
63
|
+
assert code == 2
|
|
64
|
+
assert "verdict: FLAG" in capsys.readouterr().out
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def test_cli_bad_file(tmp_path, capsys):
|
|
68
|
+
code = main(["verify", str(tmp_path / "nope.json")])
|
|
69
|
+
assert code == 3
|
|
70
|
+
assert "error" in capsys.readouterr().err
|
|
71
|
+
|
|
72
|
+
|
|
73
|
+
def test_cli_verbose(tmp_path, capsys):
|
|
74
|
+
f = tmp_path / "out.txt"
|
|
75
|
+
f.write_text("hello")
|
|
76
|
+
p = write_contract(_contract_dict(str(f)), tmp_path)
|
|
77
|
+
code = main(["verify", p, "--base-path", str(tmp_path), "-v"])
|
|
78
|
+
assert code == 0
|
|
79
|
+
out = capsys.readouterr().out
|
|
80
|
+
assert "[ pass]" in out and "c1" in out
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def test_cli_version(capsys):
|
|
84
|
+
with pytest.raises(SystemExit) as e:
|
|
85
|
+
main(["--version"])
|
|
86
|
+
assert e.value.code == 0
|
|
87
|
+
assert "0.1.0" in capsys.readouterr().out
|