aer1-strands 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- aer1_strands-0.1.0/PKG-INFO +246 -0
- aer1_strands-0.1.0/README.md +227 -0
- aer1_strands-0.1.0/aer1_strands/__init__.py +58 -0
- aer1_strands-0.1.0/aer1_strands/anchor.py +660 -0
- aer1_strands-0.1.0/aer1_strands/collector.py +514 -0
- aer1_strands-0.1.0/aer1_strands/verdicts.py +256 -0
- aer1_strands-0.1.0/aer1_strands.egg-info/PKG-INFO +246 -0
- aer1_strands-0.1.0/aer1_strands.egg-info/SOURCES.txt +13 -0
- aer1_strands-0.1.0/aer1_strands.egg-info/dependency_links.txt +1 -0
- aer1_strands-0.1.0/aer1_strands.egg-info/requires.txt +5 -0
- aer1_strands-0.1.0/aer1_strands.egg-info/top_level.txt +1 -0
- aer1_strands-0.1.0/pyproject.toml +30 -0
- aer1_strands-0.1.0/setup.cfg +4 -0
- aer1_strands-0.1.0/tests/test_collector.py +316 -0
- aer1_strands-0.1.0/tests/test_verdicts.py +205 -0
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: aer1-strands
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: One-line verifiable execution receipts for Strands Agents. Free, no API key, offline verification, chain-head anchoring to Nostr and Bitcoin.
|
|
5
|
+
License: Apache-2.0
|
|
6
|
+
Project-URL: Homepage, https://zambo.dev
|
|
7
|
+
Project-URL: IETF Draft, https://datatracker.ietf.org/doc/draft-zambo-aer1/
|
|
8
|
+
Keywords: aer-1,strands,ai-agents,verifiable-receipts,audit,tracing,mcp,zambo,audit-trail
|
|
9
|
+
Classifier: Development Status :: 3 - Alpha
|
|
10
|
+
Classifier: Intended Audience :: Developers
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
13
|
+
Requires-Python: >=3.10
|
|
14
|
+
Description-Content-Type: text/markdown
|
|
15
|
+
Requires-Dist: strands-agents>=1.0
|
|
16
|
+
Provides-Extra: anchor
|
|
17
|
+
Requires-Dist: coincurve>=18.0; extra == "anchor"
|
|
18
|
+
Requires-Dist: websocket-client>=1.7; extra == "anchor"
|
|
19
|
+
|
|
20
|
+
# aer1-strands
|
|
21
|
+
|
|
22
|
+
One-line install, no API key, free forever: attach this collector to an `Agent` and every [Strands Agents](https://strandsagents.com/) run emits a hash-chained, offline-verifiable **verifiable workflow receipt** (AER-1, an IETF Internet-Draft, Section 8). What the agent did, in what order, with per-step hashes and a Merkle root over the whole run. No network calls, no behavior changes: every hook callback returns `None`, so Strands executes exactly as it would without us. The collector only observes. Receipt chain heads can anchor to Nostr and Bitcoin, so anyone can later confirm the record was not changed, without trusting any server.
|
|
23
|
+
|
|
24
|
+
## The AER-1 framework collector family
|
|
25
|
+
|
|
26
|
+
The Strands collector in the AER-1 framework collector family. Any agent running on these frameworks can emit verifiable AER-1 execution receipts: every step recorded, hash-chained, one Merkle root over the whole run.
|
|
27
|
+
|
|
28
|
+
- [aer1-strands](https://pypi.org/project/aer1-strands/), Strands Agents via the typed hook system
|
|
29
|
+
|
|
30
|
+
See the [AER-1 implementation registry](https://rambozambodotdev.gitlab.io/registry.html) for every implementation.
|
|
31
|
+
- [aer1-langchain](https://pypi.org/project/aer1-langchain/), LangChain chains, agents, tools, and retrievers via callback handler
|
|
32
|
+
- [aer1-llamaindex](https://pypi.org/project/aer1-llamaindex/), LlamaIndex agents via callback handler
|
|
33
|
+
- [aer1-smolagents](https://pypi.org/project/aer1-smolagents/), SmolAgents agents via step callbacks
|
|
34
|
+
- [aer1-openai-agents](https://pypi.org/project/aer1-openai-agents/), the OpenAI Agents SDK via its TracingProcessor
|
|
35
|
+
- [aer1-crewai](https://pypi.org/project/aer1-crewai/), CrewAI crews via the event bus
|
|
36
|
+
- [aer1-langgraph](https://pypi.org/project/aer1-langgraph/), LangGraph swarms via callback handler
|
|
37
|
+
- [aer1-autogen](https://pypi.org/project/aer1-autogen/), AutoGen multi-agent chats
|
|
38
|
+
- [aer1-haystack](https://pypi.org/project/aer1-haystack/), Haystack 2.x pipelines via run wrapping
|
|
39
|
+
- [aer1-pydantic](https://pypi.org/project/aer1-pydantic/), Pydantic AI agents via run wrapping and manual tool-call recording
|
|
40
|
+
|
|
41
|
+
## Install
|
|
42
|
+
|
|
43
|
+
```bash
|
|
44
|
+
pip install aer1-strands
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## Use it (copy, paste, run, no API keys needed)
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
# pip install aer1-strands
|
|
51
|
+
from aer1_strands import AER1StrandsCollector
|
|
52
|
+
from strands.hooks import HookRegistry
|
|
53
|
+
from strands.hooks.events import (
|
|
54
|
+
BeforeModelCallEvent, AfterModelCallEvent,
|
|
55
|
+
BeforeToolCallEvent, AfterToolCallEvent,
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
class FakeAgent:
|
|
59
|
+
"""Stand-in for strands.Agent: a name plus a real HookRegistry."""
|
|
60
|
+
def __init__(self, name="assistant"):
|
|
61
|
+
self.name = name
|
|
62
|
+
self.hooks = HookRegistry()
|
|
63
|
+
|
|
64
|
+
def get_weather(location: str) -> str:
|
|
65
|
+
"""Local mock tool: executed on this machine, no network."""
|
|
66
|
+
return f"Sunny, 22C in {location}"
|
|
67
|
+
|
|
68
|
+
collector = AER1StrandsCollector(goal="check the Paris weather")
|
|
69
|
+
agent = FakeAgent("assistant")
|
|
70
|
+
collector.attach(agent) # line 1: register the hooks
|
|
71
|
+
|
|
72
|
+
# A two-step flow: model reasoning, then a tool call. In production
|
|
73
|
+
# Strands fires these events itself as your real agent runs, so you
|
|
74
|
+
# call collector.attach(your_real_agent) instead of this script.
|
|
75
|
+
agent.hooks.invoke_callbacks(BeforeModelCallEvent(
|
|
76
|
+
agent=agent, invocation_state={}, projected_input_tokens=96))
|
|
77
|
+
agent.hooks.invoke_callbacks(AfterModelCallEvent(
|
|
78
|
+
agent=agent, invocation_state={},
|
|
79
|
+
stop_response={"stopReason": "tool_use"}, exception=None))
|
|
80
|
+
|
|
81
|
+
tool_use = {"toolUseId": "t1", "name": "get_weather",
|
|
82
|
+
"input": {"location": "Paris"}}
|
|
83
|
+
agent.hooks.invoke_callbacks(BeforeToolCallEvent(
|
|
84
|
+
agent=agent, selected_tool=None, tool_use=tool_use,
|
|
85
|
+
invocation_state={}))
|
|
86
|
+
observation = get_weather("Paris")
|
|
87
|
+
agent.hooks.invoke_callbacks(AfterToolCallEvent(
|
|
88
|
+
agent=agent, selected_tool=None, tool_use=tool_use,
|
|
89
|
+
invocation_state={},
|
|
90
|
+
result={"toolUseId": "t1", "status": "success",
|
|
91
|
+
"content": [{"text": observation}]},
|
|
92
|
+
exception=None))
|
|
93
|
+
|
|
94
|
+
receipt = collector.finalize(final_answer=observation) # line 2
|
|
95
|
+
assert collector.verify(receipt) == [] # VALID
|
|
96
|
+
collector.save("receipt.json", workflow=receipt)
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
That is the whole integration: attach the collector, run, finalize.
|
|
100
|
+
The receipt is a plain JSON object you can store, ship to an auditor,
|
|
101
|
+
or render in a UI. In production, attach to your real `strands.Agent`
|
|
102
|
+
(the typed hook events fire automatically) instead of the scripted
|
|
103
|
+
stand-in above.
|
|
104
|
+
|
|
105
|
+
The hooks used are Strands' typed lifecycle events
|
|
106
|
+
(`BeforeToolCallEvent` / `AfterToolCallEvent` for tools,
|
|
107
|
+
`BeforeModelCallEvent` / `AfterModelCallEvent` for model calls),
|
|
108
|
+
registered on the agent's own `HookRegistry` without disturbing any
|
|
109
|
+
hooks you already have. For a multi-agent graph, call
|
|
110
|
+
`collector.attach(...)` on each agent; every step carries its agent's
|
|
111
|
+
name.
|
|
112
|
+
|
|
113
|
+
## What the receipt contains
|
|
114
|
+
|
|
115
|
+
Workflow level (AER-1 Section 8, Table 2):
|
|
116
|
+
|
|
117
|
+
- `type`, `version`, `workflow_id`, `receipt_id`, `session_id`
|
|
118
|
+
- `goal`, `status`
|
|
119
|
+
- `steps`: one record per tool/model call, seq 1..n in order
|
|
120
|
+
- `merkle_root`: Section 8.1 root over the ordered step receipt ids
|
|
121
|
+
- `output_hash`: SHA-256 of the final answer
|
|
122
|
+
- `verify_url`: where the verification procedure is documented
|
|
123
|
+
|
|
124
|
+
Step level (AER-1 Section 8, Table 3):
|
|
125
|
+
|
|
126
|
+
- `seq`, `receipt_id`, `tool`, `receipt_hash`, `started_at`, `ended_at`, `status`
|
|
127
|
+
|
|
128
|
+
Each tool step's `tool` is the tool name (for example `get_weather`);
|
|
129
|
+
each model step's `tool` is `model`. The `receipt_hash` is SHA-256 over
|
|
130
|
+
the canonical JSON of the observed content (tool name, model-supplied
|
|
131
|
+
input, tool result or stop response), so the hash commits to what
|
|
132
|
+
actually happened. The receipt stays compact while the hash commits to
|
|
133
|
+
the content. Steps also carry the agent name when the collector is
|
|
134
|
+
attached to more than one agent.
|
|
135
|
+
|
|
136
|
+
## Verification
|
|
137
|
+
|
|
138
|
+
`collector.verify(receipt)` runs the full offline check and returns a
|
|
139
|
+
list of failure reasons, empty when valid:
|
|
140
|
+
|
|
141
|
+
- all Table 2 / Table 3 members present and well-formed
|
|
142
|
+
- `seq` values exactly 1..n in order, no gaps
|
|
143
|
+
- no two steps share a `receipt_id` (MM-1)
|
|
144
|
+
- `merkle_root` matches the recomputed Section 8.1 root
|
|
145
|
+
- strict RFC 3339 timestamps, lowercase UUIDs, 64-char hex digests
|
|
146
|
+
|
|
147
|
+
Tamper with any field and verification fails. Try it:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
receipt["steps"][0]["tool"] = ""
|
|
151
|
+
assert collector.verify(receipt) != [] # fails, as it should
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
### Separate verification verdicts
|
|
155
|
+
|
|
156
|
+
For honest reporting, use `verify_receipt_verdicts()` instead of a
|
|
157
|
+
single boolean. Each dimension gets its own verdict; they are never
|
|
158
|
+
conflated:
|
|
159
|
+
|
|
160
|
+
```python
|
|
161
|
+
from aer1_strands import verify_receipt_verdicts, verdicts_summary
|
|
162
|
+
|
|
163
|
+
verdicts = verify_receipt_verdicts(receipt)
|
|
164
|
+
print(verdicts_summary(verdicts))
|
|
165
|
+
# byte_integrity=pass schema_validity=pass issuer_authenticity=not_checked
|
|
166
|
+
# evidence_linkage=not_checked anchor_verification=not_checked
|
|
167
|
+
# chain_integrity=pass
|
|
168
|
+
```
|
|
169
|
+
|
|
170
|
+
| Dimension | What it checks | Collector behavior |
|
|
171
|
+
|---|---|---|
|
|
172
|
+
| `byte_integrity` | Merkle root matches the recomputed root | `pass` / `fail` |
|
|
173
|
+
| `schema_validity` | All required fields present and well-formed | `pass` / `fail` |
|
|
174
|
+
| `issuer_authenticity` | Cryptographic proof of who issued the receipt | Always `not_checked`; the collector does not verify signatures |
|
|
175
|
+
| `evidence_linkage` | Upstream evidence bytes match their digests | Always `not_checked`; the collector does not capture upstream evidence |
|
|
176
|
+
| `anchor_verification` | Anchor proof is valid and binds the root | `pass` / `fail` / `not_checked` (no proof provided) |
|
|
177
|
+
| `chain_integrity` | `seq` values are 1..n in order, no duplicate receipt ids | `pass` / `fail` |
|
|
178
|
+
|
|
179
|
+
Verdict values are `pass`, `fail`, `not_checked`, and `unavailable`.
|
|
180
|
+
`not_checked` is neither a successful check nor evidence of failure.
|
|
181
|
+
|
|
182
|
+
### What the commitment covers (and what it does not)
|
|
183
|
+
|
|
184
|
+
The `merkle_root` commits to the ordered list of step `receipt_id`
|
|
185
|
+
values. Each step `receipt_hash` commits to that step's tool name,
|
|
186
|
+
arguments, and observations. The `output_hash` commits to the final
|
|
187
|
+
answer bytes.
|
|
188
|
+
|
|
189
|
+
What this does **not** cover:
|
|
190
|
+
|
|
191
|
+
- The commitment does not prove the bytes are unchanged since issuance.
|
|
192
|
+
An attacker able to replace both bytes and digest creates another
|
|
193
|
+
matching pair. Continuity requires comparing against a digest from
|
|
194
|
+
an independently trusted path (a retained receipt, a verified
|
|
195
|
+
signature with a trusted key, or a separately verified anchor).
|
|
196
|
+
- A matching hash does not prove the action ran, the data is true, or
|
|
197
|
+
any external outcome occurred. It proves the supplied bytes agree
|
|
198
|
+
with the supplied digest.
|
|
199
|
+
- The receipt records the collector's observations. It does not prove
|
|
200
|
+
provider truth, correct reasoning, authorization, or business success.
|
|
201
|
+
- Without an anchor, truncation (deleting steps from the end and
|
|
202
|
+
recomputing the root) is not detectable.
|
|
203
|
+
|
|
204
|
+
### Construction versions
|
|
205
|
+
|
|
206
|
+
This package uses these exact constructions (see `CONSTRUCTION_VERSIONS`
|
|
207
|
+
in the `verdicts` module):
|
|
208
|
+
|
|
209
|
+
- Merkle tree: `section-8.1-binary-merkle-v1`
|
|
210
|
+
- Step digest: `sha256-canonical-json-v1`
|
|
211
|
+
- Output commitment: `sha256-utf8-v1`
|
|
212
|
+
- Anchor proof: `aer1-anchor-proof-v1`
|
|
213
|
+
|
|
214
|
+
## Notes
|
|
215
|
+
|
|
216
|
+
- `attach(agent)` appends to the agent's existing hook callbacks; call
|
|
217
|
+
it once per agent (repeat calls are no-ops).
|
|
218
|
+
- A tool result with `status: "error"`, or a model call that raised, is
|
|
219
|
+
recorded with `status: "error"`, which marks the whole receipt error.
|
|
220
|
+
- Tool before/after events are correlated by `toolUseId`; an
|
|
221
|
+
after-event that arrives without its before-event still becomes a
|
|
222
|
+
step, so no observation is lost.
|
|
223
|
+
- `session_id` defaults to a fresh UUID per collector; pass your own to
|
|
224
|
+
correlate receipts across runs.
|
|
225
|
+
- `verify_url` defaults to the AER-1 specification page; point it at
|
|
226
|
+
your own verifier in production.
|
|
227
|
+
|
|
228
|
+
## Spec
|
|
229
|
+
|
|
230
|
+
AER-1: Agent Execution Receipts is an IETF Internet-Draft,
|
|
231
|
+
`draft-zambo-aer1`,
|
|
232
|
+
https://datatracker.ietf.org/doc/draft-zambo-aer1/
|
|
233
|
+
|
|
234
|
+
## See it live
|
|
235
|
+
|
|
236
|
+
Your receipt is offline-verifiable, but you can also check it on the live verifier:
|
|
237
|
+
|
|
238
|
+
1. Copy the receipt JSON your code produced
|
|
239
|
+
2. Paste it at https://zambo.dev/verify
|
|
240
|
+
3. See the verification result with the Merkle root and step hashes
|
|
241
|
+
|
|
242
|
+
Or mint a live receipt directly: run any call at https://zambo.dev/demo and get a shareable receipt URL like https://zambo.dev/run/<id>.
|
|
243
|
+
|
|
244
|
+
## License
|
|
245
|
+
|
|
246
|
+
Apache-2.0
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
# aer1-strands
|
|
2
|
+
|
|
3
|
+
One-line install, no API key, free forever: attach this collector to an `Agent` and every [Strands Agents](https://strandsagents.com/) run emits a hash-chained, offline-verifiable **verifiable workflow receipt** (AER-1, an IETF Internet-Draft, Section 8). What the agent did, in what order, with per-step hashes and a Merkle root over the whole run. No network calls, no behavior changes: every hook callback returns `None`, so Strands executes exactly as it would without us. The collector only observes. Receipt chain heads can anchor to Nostr and Bitcoin, so anyone can later confirm the record was not changed, without trusting any server.
|
|
4
|
+
|
|
5
|
+
## The AER-1 framework collector family
|
|
6
|
+
|
|
7
|
+
The Strands collector in the AER-1 framework collector family. Any agent running on these frameworks can emit verifiable AER-1 execution receipts: every step recorded, hash-chained, one Merkle root over the whole run.
|
|
8
|
+
|
|
9
|
+
- [aer1-strands](https://pypi.org/project/aer1-strands/), Strands Agents via the typed hook system
|
|
10
|
+
|
|
11
|
+
See the [AER-1 implementation registry](https://rambozambodotdev.gitlab.io/registry.html) for every implementation.
|
|
12
|
+
- [aer1-langchain](https://pypi.org/project/aer1-langchain/), LangChain chains, agents, tools, and retrievers via callback handler
|
|
13
|
+
- [aer1-llamaindex](https://pypi.org/project/aer1-llamaindex/), LlamaIndex agents via callback handler
|
|
14
|
+
- [aer1-smolagents](https://pypi.org/project/aer1-smolagents/), SmolAgents agents via step callbacks
|
|
15
|
+
- [aer1-openai-agents](https://pypi.org/project/aer1-openai-agents/), the OpenAI Agents SDK via its TracingProcessor
|
|
16
|
+
- [aer1-crewai](https://pypi.org/project/aer1-crewai/), CrewAI crews via the event bus
|
|
17
|
+
- [aer1-langgraph](https://pypi.org/project/aer1-langgraph/), LangGraph swarms via callback handler
|
|
18
|
+
- [aer1-autogen](https://pypi.org/project/aer1-autogen/), AutoGen multi-agent chats
|
|
19
|
+
- [aer1-haystack](https://pypi.org/project/aer1-haystack/), Haystack 2.x pipelines via run wrapping
|
|
20
|
+
- [aer1-pydantic](https://pypi.org/project/aer1-pydantic/), Pydantic AI agents via run wrapping and manual tool-call recording
|
|
21
|
+
|
|
22
|
+
## Install
|
|
23
|
+
|
|
24
|
+
```bash
|
|
25
|
+
pip install aer1-strands
|
|
26
|
+
```
|
|
27
|
+
|
|
28
|
+
## Use it (copy, paste, run, no API keys needed)
|
|
29
|
+
|
|
30
|
+
```python
|
|
31
|
+
# pip install aer1-strands
|
|
32
|
+
from aer1_strands import AER1StrandsCollector
|
|
33
|
+
from strands.hooks import HookRegistry
|
|
34
|
+
from strands.hooks.events import (
|
|
35
|
+
BeforeModelCallEvent, AfterModelCallEvent,
|
|
36
|
+
BeforeToolCallEvent, AfterToolCallEvent,
|
|
37
|
+
)
|
|
38
|
+
|
|
39
|
+
class FakeAgent:
|
|
40
|
+
"""Stand-in for strands.Agent: a name plus a real HookRegistry."""
|
|
41
|
+
def __init__(self, name="assistant"):
|
|
42
|
+
self.name = name
|
|
43
|
+
self.hooks = HookRegistry()
|
|
44
|
+
|
|
45
|
+
def get_weather(location: str) -> str:
|
|
46
|
+
"""Local mock tool: executed on this machine, no network."""
|
|
47
|
+
return f"Sunny, 22C in {location}"
|
|
48
|
+
|
|
49
|
+
collector = AER1StrandsCollector(goal="check the Paris weather")
|
|
50
|
+
agent = FakeAgent("assistant")
|
|
51
|
+
collector.attach(agent) # line 1: register the hooks
|
|
52
|
+
|
|
53
|
+
# A two-step flow: model reasoning, then a tool call. In production
|
|
54
|
+
# Strands fires these events itself as your real agent runs, so you
|
|
55
|
+
# call collector.attach(your_real_agent) instead of this script.
|
|
56
|
+
agent.hooks.invoke_callbacks(BeforeModelCallEvent(
|
|
57
|
+
agent=agent, invocation_state={}, projected_input_tokens=96))
|
|
58
|
+
agent.hooks.invoke_callbacks(AfterModelCallEvent(
|
|
59
|
+
agent=agent, invocation_state={},
|
|
60
|
+
stop_response={"stopReason": "tool_use"}, exception=None))
|
|
61
|
+
|
|
62
|
+
tool_use = {"toolUseId": "t1", "name": "get_weather",
|
|
63
|
+
"input": {"location": "Paris"}}
|
|
64
|
+
agent.hooks.invoke_callbacks(BeforeToolCallEvent(
|
|
65
|
+
agent=agent, selected_tool=None, tool_use=tool_use,
|
|
66
|
+
invocation_state={}))
|
|
67
|
+
observation = get_weather("Paris")
|
|
68
|
+
agent.hooks.invoke_callbacks(AfterToolCallEvent(
|
|
69
|
+
agent=agent, selected_tool=None, tool_use=tool_use,
|
|
70
|
+
invocation_state={},
|
|
71
|
+
result={"toolUseId": "t1", "status": "success",
|
|
72
|
+
"content": [{"text": observation}]},
|
|
73
|
+
exception=None))
|
|
74
|
+
|
|
75
|
+
receipt = collector.finalize(final_answer=observation) # line 2
|
|
76
|
+
assert collector.verify(receipt) == [] # VALID
|
|
77
|
+
collector.save("receipt.json", workflow=receipt)
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
That is the whole integration: attach the collector, run, finalize.
|
|
81
|
+
The receipt is a plain JSON object you can store, ship to an auditor,
|
|
82
|
+
or render in a UI. In production, attach to your real `strands.Agent`
|
|
83
|
+
(the typed hook events fire automatically) instead of the scripted
|
|
84
|
+
stand-in above.
|
|
85
|
+
|
|
86
|
+
The hooks used are Strands' typed lifecycle events
|
|
87
|
+
(`BeforeToolCallEvent` / `AfterToolCallEvent` for tools,
|
|
88
|
+
`BeforeModelCallEvent` / `AfterModelCallEvent` for model calls),
|
|
89
|
+
registered on the agent's own `HookRegistry` without disturbing any
|
|
90
|
+
hooks you already have. For a multi-agent graph, call
|
|
91
|
+
`collector.attach(...)` on each agent; every step carries its agent's
|
|
92
|
+
name.
|
|
93
|
+
|
|
94
|
+
## What the receipt contains
|
|
95
|
+
|
|
96
|
+
Workflow level (AER-1 Section 8, Table 2):
|
|
97
|
+
|
|
98
|
+
- `type`, `version`, `workflow_id`, `receipt_id`, `session_id`
|
|
99
|
+
- `goal`, `status`
|
|
100
|
+
- `steps`: one record per tool/model call, seq 1..n in order
|
|
101
|
+
- `merkle_root`: Section 8.1 root over the ordered step receipt ids
|
|
102
|
+
- `output_hash`: SHA-256 of the final answer
|
|
103
|
+
- `verify_url`: where the verification procedure is documented
|
|
104
|
+
|
|
105
|
+
Step level (AER-1 Section 8, Table 3):
|
|
106
|
+
|
|
107
|
+
- `seq`, `receipt_id`, `tool`, `receipt_hash`, `started_at`, `ended_at`, `status`
|
|
108
|
+
|
|
109
|
+
Each tool step's `tool` is the tool name (for example `get_weather`);
|
|
110
|
+
each model step's `tool` is `model`. The `receipt_hash` is SHA-256 over
|
|
111
|
+
the canonical JSON of the observed content (tool name, model-supplied
|
|
112
|
+
input, tool result or stop response), so the hash commits to what
|
|
113
|
+
actually happened. The receipt stays compact while the hash commits to
|
|
114
|
+
the content. Steps also carry the agent name when the collector is
|
|
115
|
+
attached to more than one agent.
|
|
116
|
+
|
|
117
|
+
## Verification
|
|
118
|
+
|
|
119
|
+
`collector.verify(receipt)` runs the full offline check and returns a
|
|
120
|
+
list of failure reasons, empty when valid:
|
|
121
|
+
|
|
122
|
+
- all Table 2 / Table 3 members present and well-formed
|
|
123
|
+
- `seq` values exactly 1..n in order, no gaps
|
|
124
|
+
- no two steps share a `receipt_id` (MM-1)
|
|
125
|
+
- `merkle_root` matches the recomputed Section 8.1 root
|
|
126
|
+
- strict RFC 3339 timestamps, lowercase UUIDs, 64-char hex digests
|
|
127
|
+
|
|
128
|
+
Tamper with any field and verification fails. Try it:
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
receipt["steps"][0]["tool"] = ""
|
|
132
|
+
assert collector.verify(receipt) != [] # fails, as it should
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
### Separate verification verdicts
|
|
136
|
+
|
|
137
|
+
For honest reporting, use `verify_receipt_verdicts()` instead of a
|
|
138
|
+
single boolean. Each dimension gets its own verdict; they are never
|
|
139
|
+
conflated:
|
|
140
|
+
|
|
141
|
+
```python
|
|
142
|
+
from aer1_strands import verify_receipt_verdicts, verdicts_summary
|
|
143
|
+
|
|
144
|
+
verdicts = verify_receipt_verdicts(receipt)
|
|
145
|
+
print(verdicts_summary(verdicts))
|
|
146
|
+
# byte_integrity=pass schema_validity=pass issuer_authenticity=not_checked
|
|
147
|
+
# evidence_linkage=not_checked anchor_verification=not_checked
|
|
148
|
+
# chain_integrity=pass
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
| Dimension | What it checks | Collector behavior |
|
|
152
|
+
|---|---|---|
|
|
153
|
+
| `byte_integrity` | Merkle root matches the recomputed root | `pass` / `fail` |
|
|
154
|
+
| `schema_validity` | All required fields present and well-formed | `pass` / `fail` |
|
|
155
|
+
| `issuer_authenticity` | Cryptographic proof of who issued the receipt | Always `not_checked`; the collector does not verify signatures |
|
|
156
|
+
| `evidence_linkage` | Upstream evidence bytes match their digests | Always `not_checked`; the collector does not capture upstream evidence |
|
|
157
|
+
| `anchor_verification` | Anchor proof is valid and binds the root | `pass` / `fail` / `not_checked` (no proof provided) |
|
|
158
|
+
| `chain_integrity` | `seq` values are 1..n in order, no duplicate receipt ids | `pass` / `fail` |
|
|
159
|
+
|
|
160
|
+
Verdict values are `pass`, `fail`, `not_checked`, and `unavailable`.
|
|
161
|
+
`not_checked` is neither a successful check nor evidence of failure.
|
|
162
|
+
|
|
163
|
+
### What the commitment covers (and what it does not)
|
|
164
|
+
|
|
165
|
+
The `merkle_root` commits to the ordered list of step `receipt_id`
|
|
166
|
+
values. Each step `receipt_hash` commits to that step's tool name,
|
|
167
|
+
arguments, and observations. The `output_hash` commits to the final
|
|
168
|
+
answer bytes.
|
|
169
|
+
|
|
170
|
+
What this does **not** cover:
|
|
171
|
+
|
|
172
|
+
- The commitment does not prove the bytes are unchanged since issuance.
|
|
173
|
+
An attacker able to replace both bytes and digest creates another
|
|
174
|
+
matching pair. Continuity requires comparing against a digest from
|
|
175
|
+
an independently trusted path (a retained receipt, a verified
|
|
176
|
+
signature with a trusted key, or a separately verified anchor).
|
|
177
|
+
- A matching hash does not prove the action ran, the data is true, or
|
|
178
|
+
any external outcome occurred. It proves the supplied bytes agree
|
|
179
|
+
with the supplied digest.
|
|
180
|
+
- The receipt records the collector's observations. It does not prove
|
|
181
|
+
provider truth, correct reasoning, authorization, or business success.
|
|
182
|
+
- Without an anchor, truncation (deleting steps from the end and
|
|
183
|
+
recomputing the root) is not detectable.
|
|
184
|
+
|
|
185
|
+
### Construction versions
|
|
186
|
+
|
|
187
|
+
This package uses these exact constructions (see `CONSTRUCTION_VERSIONS`
|
|
188
|
+
in the `verdicts` module):
|
|
189
|
+
|
|
190
|
+
- Merkle tree: `section-8.1-binary-merkle-v1`
|
|
191
|
+
- Step digest: `sha256-canonical-json-v1`
|
|
192
|
+
- Output commitment: `sha256-utf8-v1`
|
|
193
|
+
- Anchor proof: `aer1-anchor-proof-v1`
|
|
194
|
+
|
|
195
|
+
## Notes
|
|
196
|
+
|
|
197
|
+
- `attach(agent)` appends to the agent's existing hook callbacks; call
|
|
198
|
+
it once per agent (repeat calls are no-ops).
|
|
199
|
+
- A tool result with `status: "error"`, or a model call that raised, is
|
|
200
|
+
recorded with `status: "error"`, which marks the whole receipt error.
|
|
201
|
+
- Tool before/after events are correlated by `toolUseId`; an
|
|
202
|
+
after-event that arrives without its before-event still becomes a
|
|
203
|
+
step, so no observation is lost.
|
|
204
|
+
- `session_id` defaults to a fresh UUID per collector; pass your own to
|
|
205
|
+
correlate receipts across runs.
|
|
206
|
+
- `verify_url` defaults to the AER-1 specification page; point it at
|
|
207
|
+
your own verifier in production.
|
|
208
|
+
|
|
209
|
+
## Spec
|
|
210
|
+
|
|
211
|
+
AER-1: Agent Execution Receipts is an IETF Internet-Draft,
|
|
212
|
+
`draft-zambo-aer1`,
|
|
213
|
+
https://datatracker.ietf.org/doc/draft-zambo-aer1/
|
|
214
|
+
|
|
215
|
+
## See it live
|
|
216
|
+
|
|
217
|
+
Your receipt is offline-verifiable, but you can also check it on the live verifier:
|
|
218
|
+
|
|
219
|
+
1. Copy the receipt JSON your code produced
|
|
220
|
+
2. Paste it at https://zambo.dev/verify
|
|
221
|
+
3. See the verification result with the Merkle root and step hashes
|
|
222
|
+
|
|
223
|
+
Or mint a live receipt directly: run any call at https://zambo.dev/demo and get a shareable receipt URL like https://zambo.dev/run/<id>.
|
|
224
|
+
|
|
225
|
+
## License
|
|
226
|
+
|
|
227
|
+
Apache-2.0
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
"""aer1-strands: AER-1 verifiable workflow receipts for Strands Agents."""
|
|
2
|
+
|
|
3
|
+
from .anchor import (
|
|
4
|
+
AnchorError,
|
|
5
|
+
NostrBackend,
|
|
6
|
+
OTSBackend,
|
|
7
|
+
anchor_chain_head,
|
|
8
|
+
bip340_verify,
|
|
9
|
+
proof_from_json,
|
|
10
|
+
proof_to_json,
|
|
11
|
+
verify_anchor,
|
|
12
|
+
)
|
|
13
|
+
from .collector import (
|
|
14
|
+
AER1StrandsCollector,
|
|
15
|
+
DEFAULT_VERIFY_URL,
|
|
16
|
+
WORKFLOW_TYPE,
|
|
17
|
+
WORKFLOW_VERSION,
|
|
18
|
+
merkle_root,
|
|
19
|
+
verify_workflow_receipt,
|
|
20
|
+
)
|
|
21
|
+
from .verdicts import (
|
|
22
|
+
CONSTRUCTION_VERSIONS,
|
|
23
|
+
VERDICT_FAIL,
|
|
24
|
+
VERDICT_NOT_CHECKED,
|
|
25
|
+
VERDICT_PASS,
|
|
26
|
+
VERDICT_UNAVAILABLE,
|
|
27
|
+
all_verdicts_pass,
|
|
28
|
+
verdicts_summary,
|
|
29
|
+
verify_receipt_verdicts,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"AER1StrandsCollector",
|
|
35
|
+
"DEFAULT_VERIFY_URL",
|
|
36
|
+
"WORKFLOW_TYPE",
|
|
37
|
+
"WORKFLOW_VERSION",
|
|
38
|
+
"AnchorError",
|
|
39
|
+
"NostrBackend",
|
|
40
|
+
"OTSBackend",
|
|
41
|
+
"anchor_chain_head",
|
|
42
|
+
"bip340_verify",
|
|
43
|
+
"merkle_root",
|
|
44
|
+
"proof_from_json",
|
|
45
|
+
"proof_to_json",
|
|
46
|
+
"verify_anchor",
|
|
47
|
+
"verify_workflow_receipt",
|
|
48
|
+
"CONSTRUCTION_VERSIONS",
|
|
49
|
+
"VERDICT_FAIL",
|
|
50
|
+
"VERDICT_NOT_CHECKED",
|
|
51
|
+
"VERDICT_PASS",
|
|
52
|
+
"VERDICT_UNAVAILABLE",
|
|
53
|
+
"all_verdicts_pass",
|
|
54
|
+
"verdicts_summary",
|
|
55
|
+
"verify_receipt_verdicts",
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
__version__ = "0.1.0"
|