solystopia-agent-preflight 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- solystopia_agent_preflight-0.1.0/LICENSE +21 -0
- solystopia_agent_preflight-0.1.0/PKG-INFO +167 -0
- solystopia_agent_preflight-0.1.0/README.md +146 -0
- solystopia_agent_preflight-0.1.0/pyproject.toml +39 -0
- solystopia_agent_preflight-0.1.0/scripts/preflight.py +873 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/PKG-INFO +167 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/SOURCES.txt +10 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/dependency_links.txt +1 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/entry_points.txt +2 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/requires.txt +1 -0
- solystopia_agent_preflight-0.1.0/scripts/solystopia_agent_preflight.egg-info/top_level.txt +1 -0
- solystopia_agent_preflight-0.1.0/setup.cfg +4 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Arthur Silas
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: solystopia-agent-preflight
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Deterministic pre-deployment checks for AI agent systems. Blocks unsafe designs before they ship.
|
|
5
|
+
Author-email: Arthur Silas <arthur-silas@agentmail.to>
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/arthursilas-ai/agent-preflight
|
|
8
|
+
Project-URL: Repository, https://github.com/arthursilas-ai/agent-preflight
|
|
9
|
+
Project-URL: Changelog, https://github.com/arthursilas-ai/agent-preflight/blob/main/CHANGELOG.md
|
|
10
|
+
Keywords: ai-agents,ai-safety,ci-cd,deployment,preflight
|
|
11
|
+
Classifier: Programming Language :: Python :: 3
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Operating System :: OS Independent
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
16
|
+
Requires-Python: >=3.9
|
|
17
|
+
Description-Content-Type: text/markdown
|
|
18
|
+
License-File: LICENSE
|
|
19
|
+
Requires-Dist: pyyaml>=6.0
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+

|
|
23
|
+
|
|
24
|
+
# agent-preflight
|
|
25
|
+
|
|
26
|
+
**Deterministic pre-deployment checks for AI agent systems.**
|
|
27
|
+
|
|
28
|
+
[](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml)
|
|
29
|
+
[](LICENSE)
|
|
30
|
+
[](https://www.python.org/)
|
|
31
|
+
|
|
32
|
+
Your agent works in the demo. Will it pass review?
|
|
33
|
+
|
|
34
|
+
## Install
|
|
35
|
+
|
|
36
|
+
As an agent skill (works in Claude Code, Cursor, Copilot, Codex, Gemini, Zed and others):
|
|
37
|
+
|
|
38
|
+
```bash
|
|
39
|
+
npx skills add arthursilas-ai/agent-preflight
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
As a CLI, straight from GitHub — no PyPI account needed, no repo to clone:
|
|
43
|
+
|
|
44
|
+
```bash
|
|
45
|
+
pip install git+https://github.com/arthursilas-ai/agent-preflight.git
|
|
46
|
+
agent-preflight --init # write a starter spec
|
|
47
|
+
agent-preflight agent-spec.yaml # check it
|
|
48
|
+
agent-preflight --explain ops.liveness # why a check exists
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
Or standalone — one file, no install, no account:
|
|
52
|
+
|
|
53
|
+
```bash
|
|
54
|
+
curl -O https://raw.githubusercontent.com/arthursilas-ai/agent-preflight/main/scripts/preflight.py
|
|
55
|
+
pip install pyyaml
|
|
56
|
+
|
|
57
|
+
python3 preflight.py --init # write a starter spec
|
|
58
|
+
python3 preflight.py agent-spec.yaml # check it
|
|
59
|
+
python3 preflight.py --explain ops.liveness # why a check exists
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
Exit codes: `0` passed · `1` blocked · `2` spec unreadable. Add `--json` for
|
|
63
|
+
CI, `--strict` to fail on warnings too.
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Why
|
|
68
|
+
|
|
69
|
+
Around **88% of enterprise agent pilots never reach production**. The
|
|
70
|
+
reported blockers are evaluation gaps, governance friction and reliability —
|
|
71
|
+
not model quality.
|
|
72
|
+
|
|
73
|
+
There are excellent tools for watching an agent at runtime. There is very
|
|
74
|
+
little for the question that actually stalls a pilot:
|
|
75
|
+
|
|
76
|
+
> *Is this safe to ship, and can I show someone why?*
|
|
77
|
+
|
|
78
|
+
`agent-preflight` answers that with a deterministic verdict. No model calls,
|
|
79
|
+
no network, no vendor. The same spec always yields the same result, which is
|
|
80
|
+
what makes it reviewable — and what makes it usable in CI.
|
|
81
|
+
|
|
82
|
+
New here? [**Walkthrough: from blank spec to shippable**](docs/walkthrough.md) — a real pass over a small agent, about ten minutes.
|
|
83
|
+
|
|
84
|
+
Second example, different failure profile: [**a research agent**](docs/research-agent.md) that reads untrusted content but never touches money.
|
|
85
|
+
|
|
86
|
+
## What it catches
|
|
87
|
+
|
|
88
|
+
Run against a realistic refunds agent that demos perfectly:
|
|
89
|
+
|
|
90
|
+
```
|
|
91
|
+
BLOCKING (22)
|
|
92
|
+
x [tenancy.rls] Multi-tenant system without row-level security.
|
|
93
|
+
x [credentials.exposure] Privileged credentials are not restricted to server-side only.
|
|
94
|
+
x [injection.gating] Consequential actions are reachable from untrusted content.
|
|
95
|
+
x [tool.approval] Tool issue_refund: irreversible tool without an approval gate.
|
|
96
|
+
x [tool.idempotency] Tool issue_refund: irreversible tool has no idempotency strategy.
|
|
97
|
+
x [agent.step_limit] Agent refund_agent: no step_limit.
|
|
98
|
+
x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
|
|
99
|
+
...
|
|
100
|
+
|
|
101
|
+
VERDICT: BLOCKED — do not deploy until the above are resolved.
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Every finding carries a fix, not just a complaint.
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
## Use it in CI
|
|
108
|
+
|
|
109
|
+
```yaml
|
|
110
|
+
- uses: arthursilas-ai/agent-preflight@main
|
|
111
|
+
with:
|
|
112
|
+
spec: agent-spec.yaml
|
|
113
|
+
strict: "false"
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
Fails the build on blocking findings, writes them to the job summary, and
|
|
117
|
+
comments on the pull request. Outputs `passed` and `blocking-count` if you
|
|
118
|
+
want to gate a deploy step on them.
|
|
119
|
+
|
|
120
|
+
## The check areas
|
|
121
|
+
|
|
122
|
+
Purpose · Architecture shape · Tenancy isolation · Credential exposure ·
|
|
123
|
+
Prompt-injection gating · Tool contracts (idempotency, approval, scope) ·
|
|
124
|
+
Agent bounds (step limit, cost budget, stop conditions) · Evaluation
|
|
125
|
+
(including adversarial cases) · Operations (logging, alerting, rollback) ·
|
|
126
|
+
Liveness · Billing and fulfilment.
|
|
127
|
+
|
|
128
|
+
Full rationale and threat model: [references/checks.md](references/checks.md).
|
|
129
|
+
|
|
130
|
+
## One check that exists because it bit us
|
|
131
|
+
|
|
132
|
+
```
|
|
133
|
+
x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
|
|
134
|
+
```
|
|
135
|
+
|
|
136
|
+
We built an agent with scheduled daily routines. The jobs were declared
|
|
137
|
+
correctly and registered correctly on the platform. They simply never fired
|
|
138
|
+
— for two days — and nothing alerted, because *nothing had gone wrong*.
|
|
139
|
+
There were no errors to catch. The absence of runs looked exactly like a
|
|
140
|
+
quiet day.
|
|
141
|
+
|
|
142
|
+
We only found it by querying the database and noticing that every event had
|
|
143
|
+
come from a manual test.
|
|
144
|
+
|
|
145
|
+
Most agent outages are not crashes. They are things that quietly stopped
|
|
146
|
+
happening. If your monitoring only watches for errors, it cannot see this
|
|
147
|
+
class of failure at all.
|
|
148
|
+
|
|
149
|
+
## Honest limits
|
|
150
|
+
|
|
151
|
+
This validates **declared design**, not running behaviour.
|
|
152
|
+
|
|
153
|
+
A passing verdict means the stated design contains no known unsafe patterns.
|
|
154
|
+
It is **evidence for a human reviewer** — not a guarantee, not a security
|
|
155
|
+
certification, and not a compliance attestation. A spec that lies still
|
|
156
|
+
passes. It complements runtime observability; it does not replace it.
|
|
157
|
+
|
|
158
|
+
## Contributing
|
|
159
|
+
|
|
160
|
+
New checks are welcome, with one requirement: a check must encode a failure
|
|
161
|
+
that has actually happened to someone, and must ship with a fix line telling
|
|
162
|
+
the reader what to do. Speculative checks make the tool noisy and get it
|
|
163
|
+
ignored.
|
|
164
|
+
|
|
165
|
+
## Licence
|
|
166
|
+
|
|
167
|
+
MIT © [Arthur](https://github.com/arthursilas-ai)
|
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+

|
|
2
|
+
|
|
3
|
+
# agent-preflight
|
|
4
|
+
|
|
5
|
+
**Deterministic pre-deployment checks for AI agent systems.**
|
|
6
|
+
|
|
7
|
+
[](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml)
|
|
8
|
+
[](LICENSE)
|
|
9
|
+
[](https://www.python.org/)
|
|
10
|
+
|
|
11
|
+
Your agent works in the demo. Will it pass review?
|
|
12
|
+
|
|
13
|
+
## Install
|
|
14
|
+
|
|
15
|
+
As an agent skill (works in Claude Code, Cursor, Copilot, Codex, Gemini, Zed and others):
|
|
16
|
+
|
|
17
|
+
```bash
|
|
18
|
+
npx skills add arthursilas-ai/agent-preflight
|
|
19
|
+
```
|
|
20
|
+
|
|
21
|
+
As a CLI, straight from GitHub — no PyPI account needed, no repo to clone:
|
|
22
|
+
|
|
23
|
+
```bash
|
|
24
|
+
pip install git+https://github.com/arthursilas-ai/agent-preflight.git
|
|
25
|
+
agent-preflight --init # write a starter spec
|
|
26
|
+
agent-preflight agent-spec.yaml # check it
|
|
27
|
+
agent-preflight --explain ops.liveness # why a check exists
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Or standalone — one file, no install, no account:
|
|
31
|
+
|
|
32
|
+
```bash
|
|
33
|
+
curl -O https://raw.githubusercontent.com/arthursilas-ai/agent-preflight/main/scripts/preflight.py
|
|
34
|
+
pip install pyyaml
|
|
35
|
+
|
|
36
|
+
python3 preflight.py --init # write a starter spec
|
|
37
|
+
python3 preflight.py agent-spec.yaml # check it
|
|
38
|
+
python3 preflight.py --explain ops.liveness # why a check exists
|
|
39
|
+
```
|
|
40
|
+
|
|
41
|
+
Exit codes: `0` passed · `1` blocked · `2` spec unreadable. Add `--json` for
|
|
42
|
+
CI, `--strict` to fail on warnings too.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## Why
|
|
47
|
+
|
|
48
|
+
Around **88% of enterprise agent pilots never reach production**. The
|
|
49
|
+
reported blockers are evaluation gaps, governance friction and reliability —
|
|
50
|
+
not model quality.
|
|
51
|
+
|
|
52
|
+
There are excellent tools for watching an agent at runtime. There is very
|
|
53
|
+
little for the question that actually stalls a pilot:
|
|
54
|
+
|
|
55
|
+
> *Is this safe to ship, and can I show someone why?*
|
|
56
|
+
|
|
57
|
+
`agent-preflight` answers that with a deterministic verdict. No model calls,
|
|
58
|
+
no network, no vendor. The same spec always yields the same result, which is
|
|
59
|
+
what makes it reviewable — and what makes it usable in CI.
|
|
60
|
+
|
|
61
|
+
New here? [**Walkthrough: from blank spec to shippable**](docs/walkthrough.md) — a real pass over a small agent, about ten minutes.
|
|
62
|
+
|
|
63
|
+
Second example, different failure profile: [**a research agent**](docs/research-agent.md) that reads untrusted content but never touches money.
|
|
64
|
+
|
|
65
|
+
## What it catches
|
|
66
|
+
|
|
67
|
+
Run against a realistic refunds agent that demos perfectly:
|
|
68
|
+
|
|
69
|
+
```
|
|
70
|
+
BLOCKING (22)
|
|
71
|
+
x [tenancy.rls] Multi-tenant system without row-level security.
|
|
72
|
+
x [credentials.exposure] Privileged credentials are not restricted to server-side only.
|
|
73
|
+
x [injection.gating] Consequential actions are reachable from untrusted content.
|
|
74
|
+
x [tool.approval] Tool issue_refund: irreversible tool without an approval gate.
|
|
75
|
+
x [tool.idempotency] Tool issue_refund: irreversible tool has no idempotency strategy.
|
|
76
|
+
x [agent.step_limit] Agent refund_agent: no step_limit.
|
|
77
|
+
x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
|
|
78
|
+
...
|
|
79
|
+
|
|
80
|
+
VERDICT: BLOCKED — do not deploy until the above are resolved.
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
Every finding carries a fix, not just a complaint.
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
## Use it in CI
|
|
87
|
+
|
|
88
|
+
```yaml
|
|
89
|
+
- uses: arthursilas-ai/agent-preflight@main
|
|
90
|
+
with:
|
|
91
|
+
spec: agent-spec.yaml
|
|
92
|
+
strict: "false"
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
Fails the build on blocking findings, writes them to the job summary, and
|
|
96
|
+
comments on the pull request. Outputs `passed` and `blocking-count` if you
|
|
97
|
+
want to gate a deploy step on them.
|
|
98
|
+
|
|
99
|
+
## The check areas
|
|
100
|
+
|
|
101
|
+
Purpose · Architecture shape · Tenancy isolation · Credential exposure ·
|
|
102
|
+
Prompt-injection gating · Tool contracts (idempotency, approval, scope) ·
|
|
103
|
+
Agent bounds (step limit, cost budget, stop conditions) · Evaluation
|
|
104
|
+
(including adversarial cases) · Operations (logging, alerting, rollback) ·
|
|
105
|
+
Liveness · Billing and fulfilment.
|
|
106
|
+
|
|
107
|
+
Full rationale and threat model: [references/checks.md](references/checks.md).
|
|
108
|
+
|
|
109
|
+
## One check that exists because it bit us
|
|
110
|
+
|
|
111
|
+
```
|
|
112
|
+
x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
We built an agent with scheduled daily routines. The jobs were declared
|
|
116
|
+
correctly and registered correctly on the platform. They simply never fired
|
|
117
|
+
— for two days — and nothing alerted, because *nothing had gone wrong*.
|
|
118
|
+
There were no errors to catch. The absence of runs looked exactly like a
|
|
119
|
+
quiet day.
|
|
120
|
+
|
|
121
|
+
We only found it by querying the database and noticing that every event had
|
|
122
|
+
come from a manual test.
|
|
123
|
+
|
|
124
|
+
Most agent outages are not crashes. They are things that quietly stopped
|
|
125
|
+
happening. If your monitoring only watches for errors, it cannot see this
|
|
126
|
+
class of failure at all.
|
|
127
|
+
|
|
128
|
+
## Honest limits
|
|
129
|
+
|
|
130
|
+
This validates **declared design**, not running behaviour.
|
|
131
|
+
|
|
132
|
+
A passing verdict means the stated design contains no known unsafe patterns.
|
|
133
|
+
It is **evidence for a human reviewer** — not a guarantee, not a security
|
|
134
|
+
certification, and not a compliance attestation. A spec that lies still
|
|
135
|
+
passes. It complements runtime observability; it does not replace it.
|
|
136
|
+
|
|
137
|
+
## Contributing
|
|
138
|
+
|
|
139
|
+
New checks are welcome, with one requirement: a check must encode a failure
|
|
140
|
+
that has actually happened to someone, and must ship with a fix line telling
|
|
141
|
+
the reader what to do. Speculative checks make the tool noisy and get it
|
|
142
|
+
ignored.
|
|
143
|
+
|
|
144
|
+
## Licence
|
|
145
|
+
|
|
146
|
+
MIT © [Arthur](https://github.com/arthursilas-ai)
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "solystopia-agent-preflight"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Deterministic pre-deployment checks for AI agent systems. Blocks unsafe designs before they ship."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
dependencies = ["pyyaml>=6.0"]
|
|
12
|
+
license = { text = "MIT" }
|
|
13
|
+
authors = [{ name = "Arthur Silas", email = "arthur-silas@agentmail.to" }]
|
|
14
|
+
keywords = ["ai-agents", "ai-safety", "ci-cd", "deployment", "preflight"]
|
|
15
|
+
classifiers = [
|
|
16
|
+
"Programming Language :: Python :: 3",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Operating System :: OS Independent",
|
|
19
|
+
"Intended Audience :: Developers",
|
|
20
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.urls]
|
|
24
|
+
Homepage = "https://github.com/arthursilas-ai/agent-preflight"
|
|
25
|
+
Repository = "https://github.com/arthursilas-ai/agent-preflight"
|
|
26
|
+
Changelog = "https://github.com/arthursilas-ai/agent-preflight/blob/main/CHANGELOG.md"
|
|
27
|
+
|
|
28
|
+
# The distribution name is prefixed solystopia- because "agent-preflight" is
|
|
29
|
+
# already taken on PyPI by an unrelated project. The installed command stays
|
|
30
|
+
# short — `agent-preflight` — since command names don't need to be globally
|
|
31
|
+
# unique the way distribution names do.
|
|
32
|
+
[project.scripts]
|
|
33
|
+
agent-preflight = "preflight:cli"
|
|
34
|
+
|
|
35
|
+
[tool.setuptools]
|
|
36
|
+
py-modules = ["preflight"]
|
|
37
|
+
|
|
38
|
+
[tool.setuptools.package-dir]
|
|
39
|
+
"" = "scripts"
|