solystopia-agent-preflight 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Arthur Silas
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,167 @@
1
+ Metadata-Version: 2.4
2
+ Name: solystopia-agent-preflight
3
+ Version: 0.1.0
4
+ Summary: Deterministic pre-deployment checks for AI agent systems. Blocks unsafe designs before they ship.
5
+ Author-email: Arthur Silas <arthur-silas@agentmail.to>
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/arthursilas-ai/agent-preflight
8
+ Project-URL: Repository, https://github.com/arthursilas-ai/agent-preflight
9
+ Project-URL: Changelog, https://github.com/arthursilas-ai/agent-preflight/blob/main/CHANGELOG.md
10
+ Keywords: ai-agents,ai-safety,ci-cd,deployment,preflight
11
+ Classifier: Programming Language :: Python :: 3
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Operating System :: OS Independent
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Topic :: Software Development :: Quality Assurance
16
+ Requires-Python: >=3.9
17
+ Description-Content-Type: text/markdown
18
+ License-File: LICENSE
19
+ Requires-Dist: pyyaml>=6.0
20
+ Dynamic: license-file
21
+
22
+ ![agent-preflight](assets/banner.svg)
23
+
24
+ # agent-preflight
25
+
26
+ **Deterministic pre-deployment checks for AI agent systems.**
27
+
28
+ [![ci](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml/badge.svg)](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml)
29
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
30
+ [![python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue.svg)](https://www.python.org/)
31
+
32
+ Your agent works in the demo. Will it pass review?
33
+
34
+ ## Install
35
+
36
+ As an agent skill (works in Claude Code, Cursor, Copilot, Codex, Gemini, Zed and others):
37
+
38
+ ```bash
39
+ npx skills add arthursilas-ai/agent-preflight
40
+ ```
41
+
42
+ As a CLI, straight from GitHub — no PyPI account needed, no repo to clone:
43
+
44
+ ```bash
45
+ pip install git+https://github.com/arthursilas-ai/agent-preflight.git
46
+ agent-preflight --init # write a starter spec
47
+ agent-preflight agent-spec.yaml # check it
48
+ agent-preflight --explain ops.liveness # why a check exists
49
+ ```
50
+
51
+ Or standalone — one file, no install, no account:
52
+
53
+ ```bash
54
+ curl -O https://raw.githubusercontent.com/arthursilas-ai/agent-preflight/main/scripts/preflight.py
55
+ pip install pyyaml
56
+
57
+ python3 preflight.py --init # write a starter spec
58
+ python3 preflight.py agent-spec.yaml # check it
59
+ python3 preflight.py --explain ops.liveness # why a check exists
60
+ ```
61
+
62
+ Exit codes: `0` passed · `1` blocked · `2` spec unreadable. Add `--json` for
63
+ CI, `--strict` to fail on warnings too.
64
+
65
+ ---
66
+
67
+ ## Why
68
+
69
+ Around **88% of enterprise agent pilots never reach production**. The
70
+ reported blockers are evaluation gaps, governance friction and reliability —
71
+ not model quality.
72
+
73
+ There are excellent tools for watching an agent at runtime. There is very
74
+ little for the question that actually stalls a pilot:
75
+
76
+ > *Is this safe to ship, and can I show someone why?*
77
+
78
+ `agent-preflight` answers that with a deterministic verdict. No model calls,
79
+ no network, no vendor. The same spec always yields the same result, which is
80
+ what makes it reviewable — and what makes it usable in CI.
81
+
82
+ New here? [**Walkthrough: from blank spec to shippable**](docs/walkthrough.md) — a real pass over a small agent, about ten minutes.
83
+
84
+ Second example, different failure profile: [**a research agent**](docs/research-agent.md) that reads untrusted content but never touches money.
85
+
86
+ ## What it catches
87
+
88
+ Run against a realistic refunds agent that demos perfectly:
89
+
90
+ ```
91
+ BLOCKING (22)
92
+ x [tenancy.rls] Multi-tenant system without row-level security.
93
+ x [credentials.exposure] Privileged credentials are not restricted to server-side only.
94
+ x [injection.gating] Consequential actions are reachable from untrusted content.
95
+ x [tool.approval] Tool issue_refund: irreversible tool without an approval gate.
96
+ x [tool.idempotency] Tool issue_refund: irreversible tool has no idempotency strategy.
97
+ x [agent.step_limit] Agent refund_agent: no step_limit.
98
+ x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
99
+ ...
100
+
101
+ VERDICT: BLOCKED — do not deploy until the above are resolved.
102
+ ```
103
+
104
+ Every finding carries a fix, not just a complaint.
105
+
106
+
107
+ ## Use it in CI
108
+
109
+ ```yaml
110
+ - uses: arthursilas-ai/agent-preflight@main
111
+ with:
112
+ spec: agent-spec.yaml
113
+ strict: "false"
114
+ ```
115
+
116
+ Fails the build on blocking findings, writes them to the job summary, and
117
+ comments on the pull request. Outputs `passed` and `blocking-count` if you
118
+ want to gate a deploy step on them.
119
+
120
+ ## The check areas
121
+
122
+ Purpose · Architecture shape · Tenancy isolation · Credential exposure ·
123
+ Prompt-injection gating · Tool contracts (idempotency, approval, scope) ·
124
+ Agent bounds (step limit, cost budget, stop conditions) · Evaluation
125
+ (including adversarial cases) · Operations (logging, alerting, rollback) ·
126
+ Liveness · Billing and fulfilment.
127
+
128
+ Full rationale and threat model: [references/checks.md](references/checks.md).
129
+
130
+ ## One check that exists because it bit us
131
+
132
+ ```
133
+ x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
134
+ ```
135
+
136
+ We built an agent with scheduled daily routines. The jobs were declared
137
+ correctly and registered correctly on the platform. They simply never fired
138
+ — for two days — and nothing alerted, because *nothing had gone wrong*.
139
+ There were no errors to catch. The absence of runs looked exactly like a
140
+ quiet day.
141
+
142
+ We only found it by querying the database and noticing that every event had
143
+ come from a manual test.
144
+
145
+ Most agent outages are not crashes. They are things that quietly stopped
146
+ happening. If your monitoring only watches for errors, it cannot see this
147
+ class of failure at all.
148
+
149
+ ## Honest limits
150
+
151
+ This validates **declared design**, not running behaviour.
152
+
153
+ A passing verdict means the stated design contains no known unsafe patterns.
154
+ It is **evidence for a human reviewer** — not a guarantee, not a security
155
+ certification, and not a compliance attestation. A spec that lies still
156
+ passes. It complements runtime observability; it does not replace it.
157
+
158
+ ## Contributing
159
+
160
+ New checks are welcome, with one requirement: a check must encode a failure
161
+ that has actually happened to someone, and must ship with a fix line telling
162
+ the reader what to do. Speculative checks make the tool noisy and get it
163
+ ignored.
164
+
165
+ ## Licence
166
+
167
+ MIT © [Arthur](https://github.com/arthursilas-ai)
@@ -0,0 +1,146 @@
1
+ ![agent-preflight](assets/banner.svg)
2
+
3
+ # agent-preflight
4
+
5
+ **Deterministic pre-deployment checks for AI agent systems.**
6
+
7
+ [![ci](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml/badge.svg)](https://github.com/arthursilas-ai/agent-preflight/actions/workflows/ci.yml)
8
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
9
+ [![python 3.9+](https://img.shields.io/badge/python-3.9%2B-blue.svg)](https://www.python.org/)
10
+
11
+ Your agent works in the demo. Will it pass review?
12
+
13
+ ## Install
14
+
15
+ As an agent skill (works in Claude Code, Cursor, Copilot, Codex, Gemini, Zed and others):
16
+
17
+ ```bash
18
+ npx skills add arthursilas-ai/agent-preflight
19
+ ```
20
+
21
+ As a CLI, straight from GitHub — no PyPI account needed, no repo to clone:
22
+
23
+ ```bash
24
+ pip install git+https://github.com/arthursilas-ai/agent-preflight.git
25
+ agent-preflight --init # write a starter spec
26
+ agent-preflight agent-spec.yaml # check it
27
+ agent-preflight --explain ops.liveness # why a check exists
28
+ ```
29
+
30
+ Or standalone — one file, no install, no account:
31
+
32
+ ```bash
33
+ curl -O https://raw.githubusercontent.com/arthursilas-ai/agent-preflight/main/scripts/preflight.py
34
+ pip install pyyaml
35
+
36
+ python3 preflight.py --init # write a starter spec
37
+ python3 preflight.py agent-spec.yaml # check it
38
+ python3 preflight.py --explain ops.liveness # why a check exists
39
+ ```
40
+
41
+ Exit codes: `0` passed · `1` blocked · `2` spec unreadable. Add `--json` for
42
+ CI, `--strict` to fail on warnings too.
43
+
44
+ ---
45
+
46
+ ## Why
47
+
48
+ Around **88% of enterprise agent pilots never reach production**. The
49
+ reported blockers are evaluation gaps, governance friction and reliability —
50
+ not model quality.
51
+
52
+ There are excellent tools for watching an agent at runtime. There is very
53
+ little for the question that actually stalls a pilot:
54
+
55
+ > *Is this safe to ship, and can I show someone why?*
56
+
57
+ `agent-preflight` answers that with a deterministic verdict. No model calls,
58
+ no network, no vendor. The same spec always yields the same result, which is
59
+ what makes it reviewable — and what makes it usable in CI.
60
+
61
+ New here? [**Walkthrough: from blank spec to shippable**](docs/walkthrough.md) — a real pass over a small agent, about ten minutes.
62
+
63
+ Second example, different failure profile: [**a research agent**](docs/research-agent.md) that reads untrusted content but never touches money.
64
+
65
+ ## What it catches
66
+
67
+ Run against a realistic refunds agent that demos perfectly:
68
+
69
+ ```
70
+ BLOCKING (22)
71
+ x [tenancy.rls] Multi-tenant system without row-level security.
72
+ x [credentials.exposure] Privileged credentials are not restricted to server-side only.
73
+ x [injection.gating] Consequential actions are reachable from untrusted content.
74
+ x [tool.approval] Tool issue_refund: irreversible tool without an approval gate.
75
+ x [tool.idempotency] Tool issue_refund: irreversible tool has no idempotency strategy.
76
+ x [agent.step_limit] Agent refund_agent: no step_limit.
77
+ x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
78
+ ...
79
+
80
+ VERDICT: BLOCKED — do not deploy until the above are resolved.
81
+ ```
82
+
83
+ Every finding carries a fix, not just a complaint.
84
+
85
+
86
+ ## Use it in CI
87
+
88
+ ```yaml
89
+ - uses: arthursilas-ai/agent-preflight@main
90
+ with:
91
+ spec: agent-spec.yaml
92
+ strict: "false"
93
+ ```
94
+
95
+ Fails the build on blocking findings, writes them to the job summary, and
96
+ comments on the pull request. Outputs `passed` and `blocking-count` if you
97
+ want to gate a deploy step on them.
98
+
99
+ ## The check areas
100
+
101
+ Purpose · Architecture shape · Tenancy isolation · Credential exposure ·
102
+ Prompt-injection gating · Tool contracts (idempotency, approval, scope) ·
103
+ Agent bounds (step limit, cost budget, stop conditions) · Evaluation
104
+ (including adversarial cases) · Operations (logging, alerting, rollback) ·
105
+ Liveness · Billing and fulfilment.
106
+
107
+ Full rationale and threat model: [references/checks.md](references/checks.md).
108
+
109
+ ## One check that exists because it bit us
110
+
111
+ ```
112
+ x [ops.liveness] Scheduled system has no liveness alert for a run that never happens.
113
+ ```
114
+
115
+ We built an agent with scheduled daily routines. The jobs were declared
116
+ correctly and registered correctly on the platform. They simply never fired
117
+ — for two days — and nothing alerted, because *nothing had gone wrong*.
118
+ There were no errors to catch. The absence of runs looked exactly like a
119
+ quiet day.
120
+
121
+ We only found it by querying the database and noticing that every event had
122
+ come from a manual test.
123
+
124
+ Most agent outages are not crashes. They are things that quietly stopped
125
+ happening. If your monitoring only watches for errors, it cannot see this
126
+ class of failure at all.
127
+
128
+ ## Honest limits
129
+
130
+ This validates **declared design**, not running behaviour.
131
+
132
+ A passing verdict means the stated design contains no known unsafe patterns.
133
+ It is **evidence for a human reviewer** — not a guarantee, not a security
134
+ certification, and not a compliance attestation. A spec that lies still
135
+ passes. It complements runtime observability; it does not replace it.
136
+
137
+ ## Contributing
138
+
139
+ New checks are welcome, with one requirement: a check must encode a failure
140
+ that has actually happened to someone, and must ship with a fix line telling
141
+ the reader what to do. Speculative checks make the tool noisy and get it
142
+ ignored.
143
+
144
+ ## Licence
145
+
146
+ MIT © [Arthur](https://github.com/arthursilas-ai)
@@ -0,0 +1,39 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "solystopia-agent-preflight"
7
+ version = "0.1.0"
8
+ description = "Deterministic pre-deployment checks for AI agent systems. Blocks unsafe designs before they ship."
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ dependencies = ["pyyaml>=6.0"]
12
+ license = { text = "MIT" }
13
+ authors = [{ name = "Arthur Silas", email = "arthur-silas@agentmail.to" }]
14
+ keywords = ["ai-agents", "ai-safety", "ci-cd", "deployment", "preflight"]
15
+ classifiers = [
16
+ "Programming Language :: Python :: 3",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Operating System :: OS Independent",
19
+ "Intended Audience :: Developers",
20
+ "Topic :: Software Development :: Quality Assurance",
21
+ ]
22
+
23
+ [project.urls]
24
+ Homepage = "https://github.com/arthursilas-ai/agent-preflight"
25
+ Repository = "https://github.com/arthursilas-ai/agent-preflight"
26
+ Changelog = "https://github.com/arthursilas-ai/agent-preflight/blob/main/CHANGELOG.md"
27
+
28
+ # The distribution name is prefixed solystopia- because "agent-preflight" is
29
+ # already taken on PyPI by an unrelated project. The installed command stays
30
+ # short — `agent-preflight` — since command names don't need to be globally
31
+ # unique the way distribution names do.
32
+ [project.scripts]
33
+ agent-preflight = "preflight:cli"
34
+
35
+ [tool.setuptools]
36
+ py-modules = ["preflight"]
37
+
38
+ [tool.setuptools.package-dir]
39
+ "" = "scripts"