verifygate 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifygate-0.1.0/LICENSE +21 -0
- verifygate-0.1.0/PKG-INFO +301 -0
- verifygate-0.1.0/README.md +280 -0
- verifygate-0.1.0/pyproject.toml +37 -0
- verifygate-0.1.0/setup.cfg +4 -0
- verifygate-0.1.0/tests/test_comments.py +99 -0
- verifygate-0.1.0/tests/test_guards.py +146 -0
- verifygate-0.1.0/tests/test_json_output.py +117 -0
- verifygate-0.1.0/tests/test_runner.py +197 -0
- verifygate-0.1.0/tests/test_spec.py +91 -0
- verifygate-0.1.0/verifygate/__init__.py +29 -0
- verifygate-0.1.0/verifygate/cli.py +310 -0
- verifygate-0.1.0/verifygate/comments.py +180 -0
- verifygate-0.1.0/verifygate/gitutil.py +162 -0
- verifygate-0.1.0/verifygate/guards.py +218 -0
- verifygate-0.1.0/verifygate/runner.py +294 -0
- verifygate-0.1.0/verifygate/spec.py +168 -0
- verifygate-0.1.0/verifygate.egg-info/PKG-INFO +301 -0
- verifygate-0.1.0/verifygate.egg-info/SOURCES.txt +21 -0
- verifygate-0.1.0/verifygate.egg-info/dependency_links.txt +1 -0
- verifygate-0.1.0/verifygate.egg-info/entry_points.txt +2 -0
- verifygate-0.1.0/verifygate.egg-info/requires.txt +3 -0
- verifygate-0.1.0/verifygate.egg-info/top_level.txt +1 -0
verifygate-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Tsuruta Lab
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: verifygate
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Delegate work only when a machine can tell you it was done: a verification gate for coding agents.
|
|
5
|
+
Author: Tsuruta Lab
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/tsurutanmen/verifygate
|
|
8
|
+
Project-URL: Issues, https://github.com/tsurutanmen/verifygate/issues
|
|
9
|
+
Keywords: codex,claude,coding agent,verification,code review,delegation,git worktree
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
15
|
+
Requires-Python: >=3.9
|
|
16
|
+
Description-Content-Type: text/markdown
|
|
17
|
+
License-File: LICENSE
|
|
18
|
+
Provides-Extra: test
|
|
19
|
+
Requires-Dist: pytest>=7; extra == "test"
|
|
20
|
+
Dynamic: license-file
|
|
21
|
+
|
|
22
|
+
# verifygate
|
|
23
|
+
|
|
24
|
+
**Delegate work to a coding agent only when a machine can tell you it was done.**
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
pip install verifygate
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
```
|
|
31
|
+
verifygate run order.json
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
```
|
|
35
|
+
run i18n-20260913-203914
|
|
36
|
+
baseline failed, as it should in 0.1s
|
|
37
|
+
verify PASS in 0.1s
|
|
38
|
+
diff +2 / -1 across 1 file(s)
|
|
39
|
+
|
|
40
|
+
guards
|
|
41
|
+
[ok ] vacuous check was red before the work, as it should be
|
|
42
|
+
[ok ] head_moved HEAD unchanged
|
|
43
|
+
[FAIL] comment_stash 1 fragment(s) left code and appeared in comments
|
|
44
|
+
app.js: string '使い方を見る'
|
|
45
|
+
This is how a presence check is satisfied without the work.
|
|
46
|
+
[FAIL] must_keep 1 of 1 required string(s) not in real code
|
|
47
|
+
[ok ] diff_shape diff shape matches a 'add' job (+2 / -1)
|
|
48
|
+
[ok ] outside_cwd all changes inside the scope
|
|
49
|
+
|
|
50
|
+
NOT ACCEPTED. Nothing has been committed.
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
The verification command passed. The work was not done. That gap is what this
|
|
54
|
+
tool is about.
|
|
55
|
+
|
|
56
|
+
## The one rule
|
|
57
|
+
|
|
58
|
+
**An order without a verification command is rejected.** Not warned about —
|
|
59
|
+
rejected, before anything runs.
|
|
60
|
+
|
|
61
|
+
```
|
|
62
|
+
$ verifygate check order.json
|
|
63
|
+
rejected.
|
|
64
|
+
|
|
65
|
+
order.json: no verify command.
|
|
66
|
+
|
|
67
|
+
A work order without a verification command is not accepted.
|
|
68
|
+
If you cannot write a command that tells you the job was done,
|
|
69
|
+
the job is not ready to delegate - do it yourself, or find the
|
|
70
|
+
check first.
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
This is not bureaucracy. Whether you can write that command *is* the test of
|
|
74
|
+
whether the job can be handed over at all. If a machine cannot tell you the
|
|
75
|
+
work was done, you will read every line of the result yourself, and reading
|
|
76
|
+
someone else's code takes longer than writing your own. The question "what
|
|
77
|
+
command would prove this?" is the same question as "should I delegate this?"
|
|
78
|
+
|
|
79
|
+
| | how you would check it | verdict |
|
|
80
|
+
|---|---|---|
|
|
81
|
+
| remove every decorative emoji from 83 files | count them, and count what must survive | delegate |
|
|
82
|
+
| generate 240 quiz items | parse the JSON, check the answer index, check for duplicates | delegate |
|
|
83
|
+
| wrap UI strings in a translation call | the call appears, and the original text is still in code | delegate |
|
|
84
|
+
| decide which paragraph to cut | you have to read it | keep |
|
|
85
|
+
| make the page feel less generic | there is no command | keep |
|
|
86
|
+
| click through the SDK flow | you cannot run it | keep |
|
|
87
|
+
|
|
88
|
+
## What the check cannot see
|
|
89
|
+
|
|
90
|
+
An exit code of zero tells you one thing: that command exited zero. Six guards
|
|
91
|
+
ask what it does not. Each one is here because it happened.
|
|
92
|
+
|
|
93
|
+
### vacuous — the check was already green
|
|
94
|
+
|
|
95
|
+
If the verification command passes *before* the work starts, passing after it
|
|
96
|
+
proves nothing. Either the job was already done, or the check does not measure
|
|
97
|
+
it. `verifygate` runs the check on the untouched starting state first, and
|
|
98
|
+
fails the run if it was green both times.
|
|
99
|
+
|
|
100
|
+
This is the most commonly skipped step and the one that invalidates the most
|
|
101
|
+
results. A control that cannot come out differently is not a control.
|
|
102
|
+
|
|
103
|
+
### comment_stash — the text moved into a comment
|
|
104
|
+
|
|
105
|
+
Asked to restructure some UI while keeping the wording, an agent once parked
|
|
106
|
+
the deleted fragments in a block comment headed `source markers for former
|
|
107
|
+
fragments`, and the "is the wording still there?" check went green across
|
|
108
|
+
twenty-three files. Four of them reached production before anyone noticed.
|
|
109
|
+
|
|
110
|
+
`verifygate` strips comments before comparing, and separately looks for text
|
|
111
|
+
that left the code and reappeared in a comment in the same file — by whole line
|
|
112
|
+
and by quoted string, because it is usually a quoted string that gets parked.
|
|
113
|
+
|
|
114
|
+
The same check is available on its own, to put inside your own verify command:
|
|
115
|
+
|
|
116
|
+
```
|
|
117
|
+
verifygate guard --keep "使い方を見る" --keep "保存する" -- src/*.js
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
### must_keep — what had to survive
|
|
121
|
+
|
|
122
|
+
Removal jobs need a list of things that must *not* be removed. A check that
|
|
123
|
+
says "the emoji are gone" is satisfied by deleting the close button, because
|
|
124
|
+
`✕` is in the emoji range. Declare what must remain, and it is checked in real
|
|
125
|
+
code — a comment does not count.
|
|
126
|
+
|
|
127
|
+
```json
|
|
128
|
+
"must_keep": ["✕", "❤", "保存する"]
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
### diff_shape — the diff contradicts the job
|
|
132
|
+
|
|
133
|
+
An adding job that deletes more than it adds is a regression, and you want to
|
|
134
|
+
know before you start reading. Declare the direction with `"expect": "add"` or
|
|
135
|
+
`"expect": "remove"`.
|
|
136
|
+
|
|
137
|
+
### head_moved — the agent committed
|
|
138
|
+
|
|
139
|
+
If HEAD moved during the run, the diff you are about to review is not the
|
|
140
|
+
change. The order forbids committing; this notices when it happened anyway.
|
|
141
|
+
|
|
142
|
+
### outside_cwd — files written out of scope
|
|
143
|
+
|
|
144
|
+
Anything changed outside the directory the job was scoped to.
|
|
145
|
+
|
|
146
|
+
## Isolation
|
|
147
|
+
|
|
148
|
+
Each run gets **its own git worktree on its own branch**. Three consequences:
|
|
149
|
+
|
|
150
|
+
- the working copy you are using is untouched, so an agent and a person can
|
|
151
|
+
work at the same time
|
|
152
|
+
- two runs cannot collide, so write-heavy jobs can go in parallel — without
|
|
153
|
+
this you have to serialise them
|
|
154
|
+
- nothing lands anywhere by accident; `accept` commits on the run's branch and
|
|
155
|
+
a person merges
|
|
156
|
+
|
|
157
|
+
A run **refuses to start on `main`, `master`, `production`, `prod` or
|
|
158
|
+
`release`**. Somewhere there is a repository where pushing to `main` publishes
|
|
159
|
+
a website within seconds. Use `--allow-branch main` if yours is not one.
|
|
160
|
+
|
|
161
|
+
`git add -A` is never used; `accept` stages exactly the paths the run changed.
|
|
162
|
+
Working copies get shared, and a blanket add sweeps someone else's unfinished
|
|
163
|
+
work into your commit.
|
|
164
|
+
|
|
165
|
+
## The order
|
|
166
|
+
|
|
167
|
+
```json
|
|
168
|
+
{
|
|
169
|
+
"id": "i18n-en-1",
|
|
170
|
+
"objective": "Wrap the UI strings in t(). Do not change any Japanese wording.",
|
|
171
|
+
"cwd": "C:/work/portal-site",
|
|
172
|
+
"verify": "php -l includes/i18n.php && python check_i18n.py",
|
|
173
|
+
"notes": "php is at C:/tools/php/php.exe. There is no database locally.",
|
|
174
|
+
"must_keep": ["使い方を見る", "保存する"],
|
|
175
|
+
"expect": "add",
|
|
176
|
+
"forbid": ["Do not add npm packages"],
|
|
177
|
+
"model": "gpt-5.6-luna",
|
|
178
|
+
"effort": "medium",
|
|
179
|
+
"timeout_sec": 2400,
|
|
180
|
+
"web": false
|
|
181
|
+
}
|
|
182
|
+
```
|
|
183
|
+
|
|
184
|
+
Required: `id`, `objective`, `cwd`, `verify`. Everything else is optional.
|
|
185
|
+
Write paths with forward slashes — a JSON file full of doubled backslashes
|
|
186
|
+
gets mangled by whatever sits between you and the agent.
|
|
187
|
+
|
|
188
|
+
Unknown fields are rejected rather than ignored, so `"verfy"` is caught at load
|
|
189
|
+
time instead of silently disabling the one rule.
|
|
190
|
+
|
|
191
|
+
Every order carries house rules the agent is told about up front, including
|
|
192
|
+
*do not move text into comments to satisfy the check*. Guards are the backstop,
|
|
193
|
+
not the first line — say it first, then check.
|
|
194
|
+
|
|
195
|
+
See `verifygate check order.json --print-order` for the exact text sent.
|
|
196
|
+
|
|
197
|
+
## Commands
|
|
198
|
+
|
|
199
|
+
```
|
|
200
|
+
verifygate check order.json is this order acceptable at all
|
|
201
|
+
verifygate run order.json baseline, delegate, verify, guard
|
|
202
|
+
verifygate run order.json --dry-run baseline only, do not delegate
|
|
203
|
+
verifygate list runs on record
|
|
204
|
+
verifygate show <run-id> --diff --output
|
|
205
|
+
verifygate accept <run-id> commit it; refuses anything that failed
|
|
206
|
+
verifygate discard <run-id> throw the branch and worktree away
|
|
207
|
+
verifygate guard --keep STR -- files
|
|
208
|
+
```
|
|
209
|
+
|
|
210
|
+
Exit codes: `0` passed, `1` held, `2` the order or the repository was refused.
|
|
211
|
+
|
|
212
|
+
`list --json` and `show <run-id> --json` print the record and nothing else, for
|
|
213
|
+
a pipeline that needs to read the verdict rather than a table.
|
|
214
|
+
|
|
215
|
+
```
|
|
216
|
+
verifygate show "$RUN" --json | jq -r '.findings[] | select(.status=="fail") | .name'
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
## The implementer
|
|
220
|
+
|
|
221
|
+
By default `verifygate` invokes the **Codex CLI** (`codex exec`), passing the
|
|
222
|
+
order on stdin. Any command works:
|
|
223
|
+
|
|
224
|
+
```
|
|
225
|
+
verifygate run order.json --implementer "claude -p"
|
|
226
|
+
verifygate run order.json --implementer "python my_agent.py"
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
The implementer is a command that reads an order on stdin and edits files in
|
|
230
|
+
its working directory. That is the whole contract — which is why the tests can
|
|
231
|
+
use a three-line script for the honest agent, the lazy one, and the one that
|
|
232
|
+
games the check.
|
|
233
|
+
|
|
234
|
+
## Before you delegate, run your own check
|
|
235
|
+
|
|
236
|
+
Run the verification command by hand on the current state first. A check that
|
|
237
|
+
is broken looks exactly like an agent that failed, and you will spend the
|
|
238
|
+
afternoon debugging the wrong one. `--dry-run` does this for you, and tells you
|
|
239
|
+
whether the check is red on the starting state.
|
|
240
|
+
|
|
241
|
+
One specific trap: on Windows, a PowerShell script with non-ASCII comments and
|
|
242
|
+
no BOM is read as the legacy code page. The same script then returns a
|
|
243
|
+
different number. That once turned 343 into 5466.
|
|
244
|
+
|
|
245
|
+
## Install
|
|
246
|
+
|
|
247
|
+
```
|
|
248
|
+
pip install verifygate
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
Python 3.9+. No dependencies beyond the standard library and `git` on PATH.
|
|
252
|
+
|
|
253
|
+
## Claude Code skill
|
|
254
|
+
|
|
255
|
+
`skill/verifygate/SKILL.md` teaches Claude Code when to reach for this. Copy it
|
|
256
|
+
to `~/.claude/skills/verifygate/`.
|
|
257
|
+
|
|
258
|
+
## What happened the first time this was pointed at a real agent
|
|
259
|
+
|
|
260
|
+
The tool gated a change to itself: the tests in `tests/test_json_output.py`
|
|
261
|
+
were written first, confirmed red, committed, and then the job was delegated.
|
|
262
|
+
|
|
263
|
+
Run one — the implementer exited 127. `codex` is on PATH and runs from a
|
|
264
|
+
terminal, but npm installs it on Windows as `codex.CMD`, and `subprocess`
|
|
265
|
+
without a shell does not apply `PATHEXT`. It looked exactly like the CLI was
|
|
266
|
+
not installed. Fixed by resolving the executable through `shutil.which`, with a
|
|
267
|
+
regression test.
|
|
268
|
+
|
|
269
|
+
Run two — the implementer exited 1: the account had hit its usage limit.
|
|
270
|
+
|
|
271
|
+
Neither run produced a line of code, and that is the part worth reporting: both
|
|
272
|
+
were held, nothing was committed, the working copy was untouched, and the
|
|
273
|
+
`vacuous` guard confirmed on both runs that the check was red to begin with. A
|
|
274
|
+
gate earns its keep on the runs that fail, and most of them fail for reasons
|
|
275
|
+
that have nothing to do with the code.
|
|
276
|
+
|
|
277
|
+
## Limits
|
|
278
|
+
|
|
279
|
+
- Comment stripping is a scanner, not a parser. It tracks string literals well
|
|
280
|
+
enough that a URL is not mistaken for a comment, and it knows the syntax for
|
|
281
|
+
about forty extensions, but it will not be right about every corner of every
|
|
282
|
+
grammar. `comments.strip(..., keep_strings=False)` gives the stricter
|
|
283
|
+
reading; run both if it matters.
|
|
284
|
+
- Guards look at what changed. A job that changes nothing passes every guard
|
|
285
|
+
and fails the verification command, which is the right outcome by a slightly
|
|
286
|
+
indirect route.
|
|
287
|
+
- `comment_stash` compares against the run's base commit. It sees the agent's
|
|
288
|
+
work, not what you had uncommitted before it started.
|
|
289
|
+
- There is no sandbox here. `verifygate` isolates *the repository*, not the
|
|
290
|
+
machine. Sandboxing is the agent's job — and worth confirming, since at least
|
|
291
|
+
one CLI has silently ignored its own sandbox setting on Windows.
|
|
292
|
+
|
|
293
|
+
## Related
|
|
294
|
+
|
|
295
|
+
[strictnull](https://github.com/tsurutanmen/strictnull) — build the control
|
|
296
|
+
before you compare. The same idea one floor down: a comparison without a
|
|
297
|
+
control that could have come out differently is not a measurement.
|
|
298
|
+
|
|
299
|
+
## License
|
|
300
|
+
|
|
301
|
+
MIT
|
|
@@ -0,0 +1,280 @@
|
|
|
1
|
+
# verifygate
|
|
2
|
+
|
|
3
|
+
**Delegate work to a coding agent only when a machine can tell you it was done.**
|
|
4
|
+
|
|
5
|
+
```
|
|
6
|
+
pip install verifygate
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
```
|
|
10
|
+
verifygate run order.json
|
|
11
|
+
```
|
|
12
|
+
|
|
13
|
+
```
|
|
14
|
+
run i18n-20260913-203914
|
|
15
|
+
baseline failed, as it should in 0.1s
|
|
16
|
+
verify PASS in 0.1s
|
|
17
|
+
diff +2 / -1 across 1 file(s)
|
|
18
|
+
|
|
19
|
+
guards
|
|
20
|
+
[ok ] vacuous check was red before the work, as it should be
|
|
21
|
+
[ok ] head_moved HEAD unchanged
|
|
22
|
+
[FAIL] comment_stash 1 fragment(s) left code and appeared in comments
|
|
23
|
+
app.js: string '使い方を見る'
|
|
24
|
+
This is how a presence check is satisfied without the work.
|
|
25
|
+
[FAIL] must_keep 1 of 1 required string(s) not in real code
|
|
26
|
+
[ok ] diff_shape diff shape matches a 'add' job (+2 / -1)
|
|
27
|
+
[ok ] outside_cwd all changes inside the scope
|
|
28
|
+
|
|
29
|
+
NOT ACCEPTED. Nothing has been committed.
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
The verification command passed. The work was not done. That gap is what this
|
|
33
|
+
tool is about.
|
|
34
|
+
|
|
35
|
+
## The one rule
|
|
36
|
+
|
|
37
|
+
**An order without a verification command is rejected.** Not warned about —
|
|
38
|
+
rejected, before anything runs.
|
|
39
|
+
|
|
40
|
+
```
|
|
41
|
+
$ verifygate check order.json
|
|
42
|
+
rejected.
|
|
43
|
+
|
|
44
|
+
order.json: no verify command.
|
|
45
|
+
|
|
46
|
+
A work order without a verification command is not accepted.
|
|
47
|
+
If you cannot write a command that tells you the job was done,
|
|
48
|
+
the job is not ready to delegate - do it yourself, or find the
|
|
49
|
+
check first.
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
This is not bureaucracy. Whether you can write that command *is* the test of
|
|
53
|
+
whether the job can be handed over at all. If a machine cannot tell you the
|
|
54
|
+
work was done, you will read every line of the result yourself, and reading
|
|
55
|
+
someone else's code takes longer than writing your own. The question "what
|
|
56
|
+
command would prove this?" is the same question as "should I delegate this?"
|
|
57
|
+
|
|
58
|
+
| | how you would check it | verdict |
|
|
59
|
+
|---|---|---|
|
|
60
|
+
| remove every decorative emoji from 83 files | count them, and count what must survive | delegate |
|
|
61
|
+
| generate 240 quiz items | parse the JSON, check the answer index, check for duplicates | delegate |
|
|
62
|
+
| wrap UI strings in a translation call | the call appears, and the original text is still in code | delegate |
|
|
63
|
+
| decide which paragraph to cut | you have to read it | keep |
|
|
64
|
+
| make the page feel less generic | there is no command | keep |
|
|
65
|
+
| click through the SDK flow | you cannot run it | keep |
|
|
66
|
+
|
|
67
|
+
## What the check cannot see
|
|
68
|
+
|
|
69
|
+
An exit code of zero tells you one thing: that command exited zero. Six guards
|
|
70
|
+
ask what it does not. Each one is here because it happened.
|
|
71
|
+
|
|
72
|
+
### vacuous — the check was already green
|
|
73
|
+
|
|
74
|
+
If the verification command passes *before* the work starts, passing after it
|
|
75
|
+
proves nothing. Either the job was already done, or the check does not measure
|
|
76
|
+
it. `verifygate` runs the check on the untouched starting state first, and
|
|
77
|
+
fails the run if it was green both times.
|
|
78
|
+
|
|
79
|
+
This is the most commonly skipped step and the one that invalidates the most
|
|
80
|
+
results. A control that cannot come out differently is not a control.
|
|
81
|
+
|
|
82
|
+
### comment_stash — the text moved into a comment
|
|
83
|
+
|
|
84
|
+
Asked to restructure some UI while keeping the wording, an agent once parked
|
|
85
|
+
the deleted fragments in a block comment headed `source markers for former
|
|
86
|
+
fragments`, and the "is the wording still there?" check went green across
|
|
87
|
+
twenty-three files. Four of them reached production before anyone noticed.
|
|
88
|
+
|
|
89
|
+
`verifygate` strips comments before comparing, and separately looks for text
|
|
90
|
+
that left the code and reappeared in a comment in the same file — by whole line
|
|
91
|
+
and by quoted string, because it is usually a quoted string that gets parked.
|
|
92
|
+
|
|
93
|
+
The same check is available on its own, to put inside your own verify command:
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
verifygate guard --keep "使い方を見る" --keep "保存する" -- src/*.js
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
### must_keep — what had to survive
|
|
100
|
+
|
|
101
|
+
Removal jobs need a list of things that must *not* be removed. A check that
|
|
102
|
+
says "the emoji are gone" is satisfied by deleting the close button, because
|
|
103
|
+
`✕` is in the emoji range. Declare what must remain, and it is checked in real
|
|
104
|
+
code — a comment does not count.
|
|
105
|
+
|
|
106
|
+
```json
|
|
107
|
+
"must_keep": ["✕", "❤", "保存する"]
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
### diff_shape — the diff contradicts the job
|
|
111
|
+
|
|
112
|
+
An adding job that deletes more than it adds is a regression, and you want to
|
|
113
|
+
know before you start reading. Declare the direction with `"expect": "add"` or
|
|
114
|
+
`"expect": "remove"`.
|
|
115
|
+
|
|
116
|
+
### head_moved — the agent committed
|
|
117
|
+
|
|
118
|
+
If HEAD moved during the run, the diff you are about to review is not the
|
|
119
|
+
change. The order forbids committing; this notices when it happened anyway.
|
|
120
|
+
|
|
121
|
+
### outside_cwd — files written out of scope
|
|
122
|
+
|
|
123
|
+
Anything changed outside the directory the job was scoped to.
|
|
124
|
+
|
|
125
|
+
## Isolation
|
|
126
|
+
|
|
127
|
+
Each run gets **its own git worktree on its own branch**. Three consequences:
|
|
128
|
+
|
|
129
|
+
- the working copy you are using is untouched, so an agent and a person can
|
|
130
|
+
work at the same time
|
|
131
|
+
- two runs cannot collide, so write-heavy jobs can go in parallel — without
|
|
132
|
+
this you have to serialise them
|
|
133
|
+
- nothing lands anywhere by accident; `accept` commits on the run's branch and
|
|
134
|
+
a person merges
|
|
135
|
+
|
|
136
|
+
A run **refuses to start on `main`, `master`, `production`, `prod` or
|
|
137
|
+
`release`**. Somewhere there is a repository where pushing to `main` publishes
|
|
138
|
+
a website within seconds. Use `--allow-branch main` if yours is not one.
|
|
139
|
+
|
|
140
|
+
`git add -A` is never used; `accept` stages exactly the paths the run changed.
|
|
141
|
+
Working copies get shared, and a blanket add sweeps someone else's unfinished
|
|
142
|
+
work into your commit.
|
|
143
|
+
|
|
144
|
+
## The order
|
|
145
|
+
|
|
146
|
+
```json
|
|
147
|
+
{
|
|
148
|
+
"id": "i18n-en-1",
|
|
149
|
+
"objective": "Wrap the UI strings in t(). Do not change any Japanese wording.",
|
|
150
|
+
"cwd": "C:/work/portal-site",
|
|
151
|
+
"verify": "php -l includes/i18n.php && python check_i18n.py",
|
|
152
|
+
"notes": "php is at C:/tools/php/php.exe. There is no database locally.",
|
|
153
|
+
"must_keep": ["使い方を見る", "保存する"],
|
|
154
|
+
"expect": "add",
|
|
155
|
+
"forbid": ["Do not add npm packages"],
|
|
156
|
+
"model": "gpt-5.6-luna",
|
|
157
|
+
"effort": "medium",
|
|
158
|
+
"timeout_sec": 2400,
|
|
159
|
+
"web": false
|
|
160
|
+
}
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Required: `id`, `objective`, `cwd`, `verify`. Everything else is optional.
|
|
164
|
+
Write paths with forward slashes — a JSON file full of doubled backslashes
|
|
165
|
+
gets mangled by whatever sits between you and the agent.
|
|
166
|
+
|
|
167
|
+
Unknown fields are rejected rather than ignored, so `"verfy"` is caught at load
|
|
168
|
+
time instead of silently disabling the one rule.
|
|
169
|
+
|
|
170
|
+
Every order carries house rules the agent is told about up front, including
|
|
171
|
+
*do not move text into comments to satisfy the check*. Guards are the backstop,
|
|
172
|
+
not the first line — say it first, then check.
|
|
173
|
+
|
|
174
|
+
See `verifygate check order.json --print-order` for the exact text sent.
|
|
175
|
+
|
|
176
|
+
## Commands
|
|
177
|
+
|
|
178
|
+
```
|
|
179
|
+
verifygate check order.json is this order acceptable at all
|
|
180
|
+
verifygate run order.json baseline, delegate, verify, guard
|
|
181
|
+
verifygate run order.json --dry-run baseline only, do not delegate
|
|
182
|
+
verifygate list runs on record
|
|
183
|
+
verifygate show <run-id> --diff --output
|
|
184
|
+
verifygate accept <run-id> commit it; refuses anything that failed
|
|
185
|
+
verifygate discard <run-id> throw the branch and worktree away
|
|
186
|
+
verifygate guard --keep STR -- files
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
Exit codes: `0` passed, `1` held, `2` the order or the repository was refused.
|
|
190
|
+
|
|
191
|
+
`list --json` and `show <run-id> --json` print the record and nothing else, for
|
|
192
|
+
a pipeline that needs to read the verdict rather than a table.
|
|
193
|
+
|
|
194
|
+
```
|
|
195
|
+
verifygate show "$RUN" --json | jq -r '.findings[] | select(.status=="fail") | .name'
|
|
196
|
+
```
|
|
197
|
+
|
|
198
|
+
## The implementer
|
|
199
|
+
|
|
200
|
+
By default `verifygate` invokes the **Codex CLI** (`codex exec`), passing the
|
|
201
|
+
order on stdin. Any command works:
|
|
202
|
+
|
|
203
|
+
```
|
|
204
|
+
verifygate run order.json --implementer "claude -p"
|
|
205
|
+
verifygate run order.json --implementer "python my_agent.py"
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
The implementer is a command that reads an order on stdin and edits files in
|
|
209
|
+
its working directory. That is the whole contract — which is why the tests can
|
|
210
|
+
use a three-line script for the honest agent, the lazy one, and the one that
|
|
211
|
+
games the check.
|
|
212
|
+
|
|
213
|
+
## Before you delegate, run your own check
|
|
214
|
+
|
|
215
|
+
Run the verification command by hand on the current state first. A check that
|
|
216
|
+
is broken looks exactly like an agent that failed, and you will spend the
|
|
217
|
+
afternoon debugging the wrong one. `--dry-run` does this for you, and tells you
|
|
218
|
+
whether the check is red on the starting state.
|
|
219
|
+
|
|
220
|
+
One specific trap: on Windows, a PowerShell script with non-ASCII comments and
|
|
221
|
+
no BOM is read as the legacy code page. The same script then returns a
|
|
222
|
+
different number. That once turned 343 into 5466.
|
|
223
|
+
|
|
224
|
+
## Install
|
|
225
|
+
|
|
226
|
+
```
|
|
227
|
+
pip install verifygate
|
|
228
|
+
```
|
|
229
|
+
|
|
230
|
+
Python 3.9+. No dependencies beyond the standard library and `git` on PATH.
|
|
231
|
+
|
|
232
|
+
## Claude Code skill
|
|
233
|
+
|
|
234
|
+
`skill/verifygate/SKILL.md` teaches Claude Code when to reach for this. Copy it
|
|
235
|
+
to `~/.claude/skills/verifygate/`.
|
|
236
|
+
|
|
237
|
+
## What happened the first time this was pointed at a real agent
|
|
238
|
+
|
|
239
|
+
The tool gated a change to itself: the tests in `tests/test_json_output.py`
|
|
240
|
+
were written first, confirmed red, committed, and then the job was delegated.
|
|
241
|
+
|
|
242
|
+
Run one — the implementer exited 127. `codex` is on PATH and runs from a
|
|
243
|
+
terminal, but npm installs it on Windows as `codex.CMD`, and `subprocess`
|
|
244
|
+
without a shell does not apply `PATHEXT`. It looked exactly like the CLI was
|
|
245
|
+
not installed. Fixed by resolving the executable through `shutil.which`, with a
|
|
246
|
+
regression test.
|
|
247
|
+
|
|
248
|
+
Run two — the implementer exited 1: the account had hit its usage limit.
|
|
249
|
+
|
|
250
|
+
Neither run produced a line of code, and that is the part worth reporting: both
|
|
251
|
+
were held, nothing was committed, the working copy was untouched, and the
|
|
252
|
+
`vacuous` guard confirmed on both runs that the check was red to begin with. A
|
|
253
|
+
gate earns its keep on the runs that fail, and most of them fail for reasons
|
|
254
|
+
that have nothing to do with the code.
|
|
255
|
+
|
|
256
|
+
## Limits
|
|
257
|
+
|
|
258
|
+
- Comment stripping is a scanner, not a parser. It tracks string literals well
|
|
259
|
+
enough that a URL is not mistaken for a comment, and it knows the syntax for
|
|
260
|
+
about forty extensions, but it will not be right about every corner of every
|
|
261
|
+
grammar. `comments.strip(..., keep_strings=False)` gives the stricter
|
|
262
|
+
reading; run both if it matters.
|
|
263
|
+
- Guards look at what changed. A job that changes nothing passes every guard
|
|
264
|
+
and fails the verification command, which is the right outcome by a slightly
|
|
265
|
+
indirect route.
|
|
266
|
+
- `comment_stash` compares against the run's base commit. It sees the agent's
|
|
267
|
+
work, not what you had uncommitted before it started.
|
|
268
|
+
- There is no sandbox here. `verifygate` isolates *the repository*, not the
|
|
269
|
+
machine. Sandboxing is the agent's job — and worth confirming, since at least
|
|
270
|
+
one CLI has silently ignored its own sandbox setting on Windows.
|
|
271
|
+
|
|
272
|
+
## Related
|
|
273
|
+
|
|
274
|
+
[strictnull](https://github.com/tsurutanmen/strictnull) — build the control
|
|
275
|
+
before you compare. The same idea one floor down: a comparison without a
|
|
276
|
+
control that could have come out differently is not a measurement.
|
|
277
|
+
|
|
278
|
+
## License
|
|
279
|
+
|
|
280
|
+
MIT
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "verifygate"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Delegate work only when a machine can tell you it was done: a verification gate for coding agents."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Tsuruta Lab" }]
|
|
13
|
+
keywords = ["codex", "claude", "coding agent", "verification", "code review", "delegation", "git worktree"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 4 - Beta",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
20
|
+
]
|
|
21
|
+
dependencies = []
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
test = ["pytest>=7"]
|
|
25
|
+
|
|
26
|
+
[project.urls]
|
|
27
|
+
Homepage = "https://github.com/tsurutanmen/verifygate"
|
|
28
|
+
Issues = "https://github.com/tsurutanmen/verifygate/issues"
|
|
29
|
+
|
|
30
|
+
[project.scripts]
|
|
31
|
+
verifygate = "verifygate.cli:main"
|
|
32
|
+
|
|
33
|
+
[tool.setuptools.packages.find]
|
|
34
|
+
include = ["verifygate*"]
|
|
35
|
+
|
|
36
|
+
[tool.pytest.ini_options]
|
|
37
|
+
testpaths = ["tests"]
|