verifygate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- verifygate/__init__.py +29 -0
- verifygate/cli.py +310 -0
- verifygate/comments.py +180 -0
- verifygate/gitutil.py +162 -0
- verifygate/guards.py +218 -0
- verifygate/runner.py +294 -0
- verifygate/spec.py +168 -0
- verifygate-0.1.0.dist-info/METADATA +301 -0
- verifygate-0.1.0.dist-info/RECORD +13 -0
- verifygate-0.1.0.dist-info/WHEEL +5 -0
- verifygate-0.1.0.dist-info/entry_points.txt +2 -0
- verifygate-0.1.0.dist-info/licenses/LICENSE +21 -0
- verifygate-0.1.0.dist-info/top_level.txt +1 -0
verifygate/__init__.py
ADDED
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"""verifygate - delegate work only when a machine can tell you it was done.
|
|
2
|
+
|
|
3
|
+
The rule the tool exists to enforce: an order without a verification command is
|
|
4
|
+
not accepted. Whether you can write that command is the test of whether the
|
|
5
|
+
job can be handed over at all.
|
|
6
|
+
|
|
7
|
+
Everything else here is about the gap between "the command exited zero" and
|
|
8
|
+
"the work was done":
|
|
9
|
+
|
|
10
|
+
* the baseline - a check that was already green proves nothing
|
|
11
|
+
* the comment check - text parked in a comment satisfies a presence check
|
|
12
|
+
without satisfying the job
|
|
13
|
+
* the diff shape - an adding job that mostly deletes is a regression
|
|
14
|
+
* isolation - each run gets its own worktree and branch, so nothing lands
|
|
15
|
+
anywhere by accident and two runs cannot collide
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
__version__ = "0.1.0"
|
|
19
|
+
|
|
20
|
+
from .spec import Spec, SpecError, HOUSE_RULES # noqa: E402,F401
|
|
21
|
+
from .runner import Runner, RunResult, CommandResult # noqa: E402,F401
|
|
22
|
+
from .guards import Finding # noqa: E402,F401
|
|
23
|
+
from . import comments, guards, gitutil # noqa: E402,F401
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"Spec", "SpecError", "HOUSE_RULES",
|
|
27
|
+
"Runner", "RunResult", "CommandResult", "Finding",
|
|
28
|
+
"comments", "guards", "gitutil", "__version__",
|
|
29
|
+
]
|
verifygate/cli.py
ADDED
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Command line.
|
|
2
|
+
|
|
3
|
+
verifygate check order.json is this order acceptable at all?
|
|
4
|
+
verifygate run order.json baseline, delegate, verify, guard
|
|
5
|
+
verifygate list runs on record
|
|
6
|
+
verifygate show <run-id> what happened, and what the guards saw
|
|
7
|
+
verifygate accept <run-id> commit it - refuses anything that failed
|
|
8
|
+
verifygate discard <run-id> throw the branch and the worktree away
|
|
9
|
+
verifygate guard --keep X -- files the comment-stash check on its own
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
import argparse
|
|
15
|
+
import json
|
|
16
|
+
import os
|
|
17
|
+
import sys
|
|
18
|
+
|
|
19
|
+
from . import __version__, comments, gitutil, guards
|
|
20
|
+
from .runner import Runner
|
|
21
|
+
from .spec import Spec, SpecError
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _store(args) -> str:
|
|
25
|
+
return args.store or os.path.join(os.path.expanduser("~"), ".verifygate", "runs")
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def _p(s: str = "") -> None:
|
|
29
|
+
try:
|
|
30
|
+
sys.stdout.write(s + "\n")
|
|
31
|
+
except UnicodeEncodeError:
|
|
32
|
+
enc = sys.stdout.encoding or "ascii"
|
|
33
|
+
sys.stdout.write(s.encode(enc, "replace").decode(enc) + "\n")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def _utf8_stdout() -> None:
|
|
37
|
+
"""Findings quote source text, which is often not ASCII.
|
|
38
|
+
|
|
39
|
+
A Windows console defaults to a legacy code page and turns a Japanese label
|
|
40
|
+
into mojibake, which makes a real finding look like a bug in the tool.
|
|
41
|
+
"""
|
|
42
|
+
for stream in (sys.stdout, sys.stderr):
|
|
43
|
+
try:
|
|
44
|
+
stream.reconfigure(encoding="utf-8", errors="replace") # type: ignore[attr-defined]
|
|
45
|
+
except (AttributeError, ValueError, OSError):
|
|
46
|
+
pass
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
# ---------------------------------------------------------------- subcommands
|
|
50
|
+
def cmd_check(args) -> int:
|
|
51
|
+
try:
|
|
52
|
+
spec = Spec.load(args.spec)
|
|
53
|
+
except SpecError as e:
|
|
54
|
+
_p("rejected.\n\n{}".format(e))
|
|
55
|
+
return 2
|
|
56
|
+
_p("accepted: {}".format(spec.id))
|
|
57
|
+
_p(" scope {}".format(spec.cwd))
|
|
58
|
+
_p(" verify {}".format(spec.verify))
|
|
59
|
+
if spec.must_keep:
|
|
60
|
+
_p(" keeps {} string(s)".format(len(spec.must_keep)))
|
|
61
|
+
if spec.expect:
|
|
62
|
+
_p(" shape expected to {}".format(spec.expect))
|
|
63
|
+
if args.print_order:
|
|
64
|
+
_p("")
|
|
65
|
+
_p(spec.prompt())
|
|
66
|
+
return 0
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def cmd_run(args) -> int:
|
|
70
|
+
try:
|
|
71
|
+
spec = Spec.load(args.spec)
|
|
72
|
+
except SpecError as e:
|
|
73
|
+
_p("rejected.\n\n{}".format(e))
|
|
74
|
+
return 2
|
|
75
|
+
|
|
76
|
+
runner = Runner(_store(args),
|
|
77
|
+
implementer=(args.implementer.split() if args.implementer else None),
|
|
78
|
+
allow_branch=tuple(args.allow_branch or ()),
|
|
79
|
+
keep_worktree=args.keep)
|
|
80
|
+
try:
|
|
81
|
+
res = runner.run(spec, dry_run=args.dry_run)
|
|
82
|
+
except RuntimeError as e:
|
|
83
|
+
_p("refusing to start.\n\n {}".format(e))
|
|
84
|
+
return 2
|
|
85
|
+
except gitutil.GitError as e:
|
|
86
|
+
_p("git: {}".format(e))
|
|
87
|
+
return 2
|
|
88
|
+
|
|
89
|
+
_p("run {}".format(res.run_id))
|
|
90
|
+
_p(" branch {}".format(res.branch))
|
|
91
|
+
_p(" worktree {}".format(res.worktree))
|
|
92
|
+
if res.baseline:
|
|
93
|
+
_p(" baseline {} in {:.1f}s".format(
|
|
94
|
+
"passed (this check cannot show the work)" if res.baseline.ok else "failed, as it should",
|
|
95
|
+
res.baseline.seconds))
|
|
96
|
+
if args.dry_run:
|
|
97
|
+
_p("\ndry run: the implementer was not invoked.")
|
|
98
|
+
_p("The order is at {}".format(os.path.join(runner.store, res.run_id, "order.md")))
|
|
99
|
+
return 0
|
|
100
|
+
if res.implementer and not res.implementer.ok:
|
|
101
|
+
_p(" implementer exited {}".format(res.implementer.code))
|
|
102
|
+
if res.implementer.stderr.strip():
|
|
103
|
+
_p(" " + res.implementer.stderr.strip().splitlines()[-1])
|
|
104
|
+
if res.after:
|
|
105
|
+
_p(" verify {} in {:.1f}s".format("PASS" if res.after.ok else "FAIL",
|
|
106
|
+
res.after.seconds))
|
|
107
|
+
_p(" diff +{} / -{} across {} file(s)".format(
|
|
108
|
+
res.insertions, res.deletions, len(res.changed)))
|
|
109
|
+
_p("")
|
|
110
|
+
_p("guards")
|
|
111
|
+
for f in res.findings:
|
|
112
|
+
_p(f.render())
|
|
113
|
+
_p("")
|
|
114
|
+
if res.passed:
|
|
115
|
+
_p("PASSED. verifygate accept {}".format(res.run_id))
|
|
116
|
+
return 0
|
|
117
|
+
_p("NOT ACCEPTED. Nothing has been committed.")
|
|
118
|
+
_p(" verifygate show {} to read the diff and the output".format(res.run_id))
|
|
119
|
+
return 1
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def cmd_list(args) -> int:
|
|
123
|
+
runner = Runner(_store(args))
|
|
124
|
+
ids = runner.list_runs()
|
|
125
|
+
if getattr(args, "json", False):
|
|
126
|
+
rows = []
|
|
127
|
+
for rid in ids:
|
|
128
|
+
try:
|
|
129
|
+
d = runner.load(rid)
|
|
130
|
+
except (OSError, ValueError):
|
|
131
|
+
continue
|
|
132
|
+
rows.append({
|
|
133
|
+
"run_id": d["run_id"], "passed": d["passed"],
|
|
134
|
+
"verified": d["verified"], "guards_passed": d["guards_passed"],
|
|
135
|
+
"insertions": d["insertions"], "deletions": d["deletions"],
|
|
136
|
+
"branch": d["branch"], "verify": d["spec"]["verify"],
|
|
137
|
+
})
|
|
138
|
+
_p(json.dumps(rows, ensure_ascii=False, indent=2))
|
|
139
|
+
return 0
|
|
140
|
+
if not ids:
|
|
141
|
+
_p("no runs in {}".format(runner.store))
|
|
142
|
+
return 0
|
|
143
|
+
for rid in ids:
|
|
144
|
+
try:
|
|
145
|
+
d = runner.load(rid)
|
|
146
|
+
except (OSError, ValueError):
|
|
147
|
+
continue
|
|
148
|
+
_p("{:<34} {:<7} +{}/-{} {}".format(
|
|
149
|
+
rid, "PASSED" if d["passed"] else "held",
|
|
150
|
+
d["insertions"], d["deletions"], d["spec"]["verify"][:44]))
|
|
151
|
+
return 0
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def cmd_show(args) -> int:
|
|
155
|
+
runner = Runner(_store(args))
|
|
156
|
+
as_json = getattr(args, "json", False)
|
|
157
|
+
try:
|
|
158
|
+
d = runner.load(args.run_id)
|
|
159
|
+
except OSError:
|
|
160
|
+
if as_json:
|
|
161
|
+
_p(json.dumps({"error": "no such run", "run_id": args.run_id}))
|
|
162
|
+
else:
|
|
163
|
+
_p("no such run: {}".format(args.run_id))
|
|
164
|
+
return 2
|
|
165
|
+
if as_json:
|
|
166
|
+
# the stored record is the contract; do not rebuild it here
|
|
167
|
+
_p(json.dumps(d, ensure_ascii=False, indent=2))
|
|
168
|
+
return 0
|
|
169
|
+
_p("{} {}".format(d["run_id"], "PASSED" if d["passed"] else "held"))
|
|
170
|
+
_p(" objective {}".format(d["spec"]["objective"].splitlines()[0][:90]))
|
|
171
|
+
_p(" verify {}".format(d["spec"]["verify"]))
|
|
172
|
+
_p(" branch {}".format(d["branch"]))
|
|
173
|
+
_p(" diff +{} / -{} across {} file(s)".format(
|
|
174
|
+
d["insertions"], d["deletions"], len(d["changed"])))
|
|
175
|
+
_p("")
|
|
176
|
+
for f in d["findings"]:
|
|
177
|
+
_p(guards.Finding(f["name"], f["status"], f["summary"], f["detail"]).render())
|
|
178
|
+
for key in ("baseline", "after"):
|
|
179
|
+
c = d.get(key)
|
|
180
|
+
if c and args.output:
|
|
181
|
+
_p("")
|
|
182
|
+
_p("--- {} (exit {}) ---".format(key, c["code"]))
|
|
183
|
+
_p((c["stdout"] or "").strip()[-4000:])
|
|
184
|
+
if (c["stderr"] or "").strip():
|
|
185
|
+
_p((c["stderr"] or "").strip()[-2000:])
|
|
186
|
+
patch = os.path.join(runner.store, args.run_id, "diff.patch")
|
|
187
|
+
if args.diff and os.path.isfile(patch):
|
|
188
|
+
_p("")
|
|
189
|
+
with open(patch, "r", encoding="utf-8", errors="replace") as f:
|
|
190
|
+
_p(f.read())
|
|
191
|
+
return 0
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
def cmd_accept(args) -> int:
|
|
195
|
+
runner = Runner(_store(args))
|
|
196
|
+
try:
|
|
197
|
+
d = runner.load(args.run_id)
|
|
198
|
+
sha = runner.accept(args.run_id, gitutil.toplevel(d["spec"]["cwd"]), args.message or "")
|
|
199
|
+
except OSError:
|
|
200
|
+
_p("no such run: {}".format(args.run_id))
|
|
201
|
+
return 2
|
|
202
|
+
except RuntimeError as e:
|
|
203
|
+
_p(str(e))
|
|
204
|
+
return 1
|
|
205
|
+
_p("committed {} on {}".format(sha[:12], d["branch"]))
|
|
206
|
+
_p("A person merges it from here.")
|
|
207
|
+
return 0
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def cmd_discard(args) -> int:
|
|
211
|
+
runner = Runner(_store(args))
|
|
212
|
+
try:
|
|
213
|
+
d = runner.load(args.run_id)
|
|
214
|
+
runner.discard(args.run_id, gitutil.toplevel(d["spec"]["cwd"]))
|
|
215
|
+
except OSError:
|
|
216
|
+
_p("no such run: {}".format(args.run_id))
|
|
217
|
+
return 2
|
|
218
|
+
_p("discarded {}".format(args.run_id))
|
|
219
|
+
return 0
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def cmd_guard(args) -> int:
|
|
223
|
+
"""The comment check on its own, for use inside a verify command."""
|
|
224
|
+
bad = 0
|
|
225
|
+
for path in args.files:
|
|
226
|
+
try:
|
|
227
|
+
with open(path, "r", encoding="utf-8", errors="replace") as f:
|
|
228
|
+
text = f.read()
|
|
229
|
+
except OSError as e:
|
|
230
|
+
_p("{}: {}".format(path, e))
|
|
231
|
+
bad += 1
|
|
232
|
+
continue
|
|
233
|
+
for needle in args.keep:
|
|
234
|
+
in_code = comments.count_in_code(text, needle, path)
|
|
235
|
+
in_comment = comments.count_in_comments(text, needle, path)
|
|
236
|
+
if in_code == 0 and in_comment > 0:
|
|
237
|
+
_p("{}: {!r} only inside a comment".format(path, needle))
|
|
238
|
+
bad += 1
|
|
239
|
+
elif in_code == 0:
|
|
240
|
+
_p("{}: {!r} not found".format(path, needle))
|
|
241
|
+
bad += 1
|
|
242
|
+
if bad:
|
|
243
|
+
_p("{} problem(s)".format(bad))
|
|
244
|
+
return 1
|
|
245
|
+
_p("ok: every string present in code")
|
|
246
|
+
return 0
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
# --------------------------------------------------------------------- parser
|
|
250
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
251
|
+
ap = argparse.ArgumentParser(
|
|
252
|
+
prog="verifygate",
|
|
253
|
+
description="Delegate work only when a machine can tell you it was done.")
|
|
254
|
+
ap.add_argument("--version", action="version", version="verifygate " + __version__)
|
|
255
|
+
ap.add_argument("--store", help="where runs are kept (default ~/.verifygate/runs)")
|
|
256
|
+
sub = ap.add_subparsers(dest="cmd")
|
|
257
|
+
|
|
258
|
+
c = sub.add_parser("check", help="validate an order without running it")
|
|
259
|
+
c.add_argument("spec")
|
|
260
|
+
c.add_argument("--print-order", action="store_true", help="show the text the implementer gets")
|
|
261
|
+
c.set_defaults(func=cmd_check)
|
|
262
|
+
|
|
263
|
+
r = sub.add_parser("run", help="baseline, delegate, verify, guard")
|
|
264
|
+
r.add_argument("spec")
|
|
265
|
+
r.add_argument("--dry-run", action="store_true", help="baseline only; do not delegate")
|
|
266
|
+
r.add_argument("--implementer", help="command to run instead of the Codex CLI")
|
|
267
|
+
r.add_argument("--allow-branch", action="append",
|
|
268
|
+
help="permit a branch normally treated as deploying")
|
|
269
|
+
r.add_argument("--keep", action="store_true", help="keep the worktree after the run")
|
|
270
|
+
r.set_defaults(func=cmd_run)
|
|
271
|
+
|
|
272
|
+
li = sub.add_parser("list", help="runs on record")
|
|
273
|
+
li.add_argument("--json", action="store_true", help="print the runs as a JSON array")
|
|
274
|
+
li.set_defaults(func=cmd_list)
|
|
275
|
+
|
|
276
|
+
s = sub.add_parser("show", help="what happened in a run")
|
|
277
|
+
s.add_argument("run_id")
|
|
278
|
+
s.add_argument("--diff", action="store_true")
|
|
279
|
+
s.add_argument("--output", action="store_true", help="include the verify output")
|
|
280
|
+
s.add_argument("--json", action="store_true", help="print the run record as JSON and nothing else")
|
|
281
|
+
s.set_defaults(func=cmd_show)
|
|
282
|
+
|
|
283
|
+
a = sub.add_parser("accept", help="commit a run that passed")
|
|
284
|
+
a.add_argument("run_id")
|
|
285
|
+
a.add_argument("-m", "--message")
|
|
286
|
+
a.set_defaults(func=cmd_accept)
|
|
287
|
+
|
|
288
|
+
d = sub.add_parser("discard", help="delete a run's branch and worktree")
|
|
289
|
+
d.add_argument("run_id")
|
|
290
|
+
d.set_defaults(func=cmd_discard)
|
|
291
|
+
|
|
292
|
+
g = sub.add_parser("guard", help="check strings survive in code, not comments")
|
|
293
|
+
g.add_argument("--keep", action="append", required=True, metavar="STRING")
|
|
294
|
+
g.add_argument("files", nargs="+")
|
|
295
|
+
g.set_defaults(func=cmd_guard)
|
|
296
|
+
return ap
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
def main(argv=None) -> int:
|
|
300
|
+
_utf8_stdout()
|
|
301
|
+
ap = build_parser()
|
|
302
|
+
args = ap.parse_args(argv)
|
|
303
|
+
if not getattr(args, "func", None):
|
|
304
|
+
ap.print_help()
|
|
305
|
+
return 0
|
|
306
|
+
return args.func(args)
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
if __name__ == "__main__":
|
|
310
|
+
sys.exit(main())
|
verifygate/comments.py
ADDED
|
@@ -0,0 +1,180 @@
|
|
|
1
|
+
"""Strip comments so a check counts code, not commentary.
|
|
2
|
+
|
|
3
|
+
A check that asks "is this string still in the file?" is satisfied by a comment
|
|
4
|
+
containing the string. That is not a hypothetical: an implementer asked to
|
|
5
|
+
keep some text while restructuring it once parked the deleted fragments in a
|
|
6
|
+
block comment headed "source markers for former fragments", and the check went
|
|
7
|
+
green across twenty-three files.
|
|
8
|
+
|
|
9
|
+
So before counting, take the comments out. This is a scanner, not a parser: it
|
|
10
|
+
tracks string literals well enough that a URL inside a quoted string is not
|
|
11
|
+
mistaken for a line comment, and it knows which comment syntaxes belong to
|
|
12
|
+
which extension. It does not understand every corner of every grammar, and it
|
|
13
|
+
says so - :func:`strip` is documented as approximate, and both the strict and
|
|
14
|
+
the loose reading are available so a caller can require agreement.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import os
|
|
20
|
+
from typing import Dict, List, Sequence, Tuple
|
|
21
|
+
|
|
22
|
+
#: extension -> (line comment markers, block comment pairs, string quote chars)
|
|
23
|
+
_LANGS: Dict[str, Tuple[Sequence[str], Sequence[Tuple[str, str]], Sequence[str]]] = {}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _reg(exts, line, block, quotes):
|
|
27
|
+
for e in exts:
|
|
28
|
+
_LANGS[e] = (tuple(line), tuple(block), tuple(quotes))
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
_reg([".py", ".pyi"], ["#"], [], ['"""', "'''", '"', "'"])
|
|
32
|
+
_reg([".js", ".mjs", ".cjs", ".jsx", ".ts", ".tsx"], ["//"], [("/*", "*/")], ['"', "'", "`"])
|
|
33
|
+
_reg([".java", ".c", ".h", ".cc", ".cpp", ".hpp", ".cs", ".go", ".rs", ".swift", ".kt"],
|
|
34
|
+
["//"], [("/*", "*/")], ['"', "'"])
|
|
35
|
+
_reg([".php"], ["//", "#"], [("/*", "*/"), ("<!--", "-->")], ['"', "'"])
|
|
36
|
+
_reg([".css", ".scss", ".less"], [], [("/*", "*/")], ['"', "'"])
|
|
37
|
+
_reg([".html", ".htm", ".xml", ".svg", ".vue", ".xhtml"], [], [("<!--", "-->")], ['"', "'"])
|
|
38
|
+
_reg([".sh", ".bash", ".zsh", ".rb", ".yml", ".yaml", ".toml", ".ini", ".cfg", ".conf"],
|
|
39
|
+
["#"], [], ['"', "'"])
|
|
40
|
+
_reg([".sql"], ["--"], [("/*", "*/")], ["'", '"'])
|
|
41
|
+
_reg([".ps1", ".psm1"], ["#"], [("<#", "#>")], ['"', "'"])
|
|
42
|
+
|
|
43
|
+
#: used when the extension is unknown: everything plausible is treated as a comment
|
|
44
|
+
_FALLBACK = (("//", "#", "--"), (("/*", "*/"), ("<!--", "-->")), ('"', "'", "`"))
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def rules_for(path: str):
|
|
48
|
+
"""Comment and string rules for a path. Unknown extensions get a loose set."""
|
|
49
|
+
return _LANGS.get(os.path.splitext(path)[1].lower(), _FALLBACK)
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def strip(text: str, path: str = "", keep_strings: bool = True) -> str:
|
|
53
|
+
"""Return ``text`` with comments replaced by spaces of the same length.
|
|
54
|
+
|
|
55
|
+
Lengths and line breaks are preserved so that line numbers and offsets
|
|
56
|
+
still line up with the original.
|
|
57
|
+
|
|
58
|
+
``keep_strings=False`` also blanks string literals, which is the stricter
|
|
59
|
+
reading: it catches text parked in an unused constant, and it also blanks
|
|
60
|
+
text that is legitimately a user-visible string. Callers that care about
|
|
61
|
+
the difference should run both and compare.
|
|
62
|
+
|
|
63
|
+
This is approximate by construction; see the module docstring.
|
|
64
|
+
"""
|
|
65
|
+
line_marks, block_pairs, quotes = rules_for(path)
|
|
66
|
+
# longest first so "///" and '"""' win over their prefixes
|
|
67
|
+
line_marks = tuple(sorted(line_marks, key=len, reverse=True))
|
|
68
|
+
quotes = tuple(sorted(quotes, key=len, reverse=True))
|
|
69
|
+
block_pairs = tuple(block_pairs)
|
|
70
|
+
|
|
71
|
+
out: List[str] = []
|
|
72
|
+
i = 0
|
|
73
|
+
n = len(text)
|
|
74
|
+
while i < n:
|
|
75
|
+
ch = text[i]
|
|
76
|
+
|
|
77
|
+
# inside a string literal: copy it out (or blank it) and skip to the end
|
|
78
|
+
q = _starts_with_any(text, i, quotes)
|
|
79
|
+
if q:
|
|
80
|
+
j = _end_of_string(text, i + len(q), q)
|
|
81
|
+
chunk = text[i:j]
|
|
82
|
+
out.append(chunk if keep_strings else _blank(chunk))
|
|
83
|
+
i = j
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
pair = _starts_with_block(text, i, block_pairs)
|
|
87
|
+
if pair:
|
|
88
|
+
open_s, close_s = pair
|
|
89
|
+
end = text.find(close_s, i + len(open_s))
|
|
90
|
+
end = n if end == -1 else end + len(close_s)
|
|
91
|
+
out.append(_blank(text[i:end]))
|
|
92
|
+
i = end
|
|
93
|
+
continue
|
|
94
|
+
|
|
95
|
+
m = _starts_with_any(text, i, line_marks)
|
|
96
|
+
if m:
|
|
97
|
+
end = text.find("\n", i)
|
|
98
|
+
end = n if end == -1 else end
|
|
99
|
+
out.append(_blank(text[i:end]))
|
|
100
|
+
i = end
|
|
101
|
+
continue
|
|
102
|
+
|
|
103
|
+
out.append(ch)
|
|
104
|
+
i += 1
|
|
105
|
+
return "".join(out)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _blank(s: str) -> str:
|
|
109
|
+
return "".join("\n" if c == "\n" else " " for c in s)
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _starts_with_any(text: str, i: int, marks: Sequence[str]):
|
|
113
|
+
for m in marks:
|
|
114
|
+
if m and text.startswith(m, i):
|
|
115
|
+
return m
|
|
116
|
+
return None
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
def _starts_with_block(text: str, i: int, pairs):
|
|
120
|
+
for open_s, close_s in pairs:
|
|
121
|
+
if text.startswith(open_s, i):
|
|
122
|
+
return (open_s, close_s)
|
|
123
|
+
return None
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _end_of_string(text: str, i: int, quote: str) -> int:
|
|
127
|
+
"""Index just past the closing quote (or end of text)."""
|
|
128
|
+
n = len(text)
|
|
129
|
+
while i < n:
|
|
130
|
+
c = text[i]
|
|
131
|
+
if c == "\\":
|
|
132
|
+
i += 2
|
|
133
|
+
continue
|
|
134
|
+
if text.startswith(quote, i):
|
|
135
|
+
return i + len(quote)
|
|
136
|
+
# a single-character quote does not survive a newline in most languages
|
|
137
|
+
if len(quote) == 1 and c == "\n" and quote in ("'", '"'):
|
|
138
|
+
return i
|
|
139
|
+
i += 1
|
|
140
|
+
return n
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def literals(text: str, path: str = "", min_len: int = 3) -> List[str]:
|
|
144
|
+
"""Contents of the string literals in ``text``, outside comments.
|
|
145
|
+
|
|
146
|
+
Quoted text is what actually gets parked: a label is lifted out of the code
|
|
147
|
+
and dropped into a comment so that a "the wording is still there" check
|
|
148
|
+
stays green. Comparing literals before and after catches that even when
|
|
149
|
+
the surrounding line was rewritten.
|
|
150
|
+
"""
|
|
151
|
+
code = strip(text, path)
|
|
152
|
+
_, _, quotes = rules_for(path)
|
|
153
|
+
quotes = tuple(sorted(quotes, key=len, reverse=True))
|
|
154
|
+
out: List[str] = []
|
|
155
|
+
i, n = 0, len(code)
|
|
156
|
+
while i < n:
|
|
157
|
+
q = _starts_with_any(code, i, quotes)
|
|
158
|
+
if not q:
|
|
159
|
+
i += 1
|
|
160
|
+
continue
|
|
161
|
+
j = _end_of_string(code, i + len(q), q)
|
|
162
|
+
body = code[i + len(q): max(i + len(q), j - len(q))]
|
|
163
|
+
if len(body.strip()) >= min_len:
|
|
164
|
+
out.append(body)
|
|
165
|
+
i = j
|
|
166
|
+
return out
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def count_in_code(text: str, needle: str, path: str = "") -> int:
|
|
170
|
+
"""How many times ``needle`` appears outside comments."""
|
|
171
|
+
if not needle:
|
|
172
|
+
return 0
|
|
173
|
+
return strip(text, path).count(needle)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def count_in_comments(text: str, needle: str, path: str = "") -> int:
|
|
177
|
+
"""How many times ``needle`` appears *only* inside comments."""
|
|
178
|
+
if not needle:
|
|
179
|
+
return 0
|
|
180
|
+
return text.count(needle) - count_in_code(text, needle, path)
|
verifygate/gitutil.py
ADDED
|
@@ -0,0 +1,162 @@
|
|
|
1
|
+
"""The small amount of git this needs, with the sharp edges filed off.
|
|
2
|
+
|
|
3
|
+
Two habits are enforced here rather than left to the caller:
|
|
4
|
+
|
|
5
|
+
* ``git add -A`` is never used. Working copies get shared - with another
|
|
6
|
+
agent, another session, a person - and a blanket add sweeps someone else's
|
|
7
|
+
unfinished work into your commit. Paths are always named.
|
|
8
|
+
* A run never starts on a branch that deploys. Somewhere there is a repository
|
|
9
|
+
where pushing to ``main`` publishes a website within seconds, and the cost of
|
|
10
|
+
finding that out during an automated run is a broken site.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import os
|
|
16
|
+
import subprocess
|
|
17
|
+
from typing import List, Optional, Sequence, Tuple
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class GitError(RuntimeError):
|
|
21
|
+
pass
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def run(args: Sequence[str], cwd: str, check: bool = True, timeout: int = 120) -> str:
|
|
25
|
+
p = subprocess.run(
|
|
26
|
+
["git"] + list(args), cwd=cwd, capture_output=True, text=True,
|
|
27
|
+
encoding="utf-8", errors="replace", timeout=timeout,
|
|
28
|
+
)
|
|
29
|
+
if check and p.returncode != 0:
|
|
30
|
+
raise GitError("git {}: {}".format(" ".join(args), (p.stderr or p.stdout).strip()))
|
|
31
|
+
return p.stdout
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def is_repo(cwd: str) -> bool:
|
|
35
|
+
try:
|
|
36
|
+
return run(["rev-parse", "--is-inside-work-tree"], cwd).strip() == "true"
|
|
37
|
+
except (GitError, OSError):
|
|
38
|
+
return False
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def current_branch(cwd: str) -> str:
|
|
42
|
+
return run(["rev-parse", "--abbrev-ref", "HEAD"], cwd).strip()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def head(cwd: str) -> str:
|
|
46
|
+
return run(["rev-parse", "HEAD"], cwd).strip()
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def toplevel(cwd: str) -> str:
|
|
50
|
+
return run(["rev-parse", "--show-toplevel"], cwd).strip()
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def is_clean(cwd: str) -> bool:
|
|
54
|
+
return run(["status", "--porcelain"], cwd).strip() == ""
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def status_paths(cwd: str) -> List[Tuple[str, str]]:
|
|
58
|
+
"""[(status, path)] for every changed or untracked path."""
|
|
59
|
+
out = []
|
|
60
|
+
for line in run(["status", "--porcelain"], cwd).splitlines():
|
|
61
|
+
if len(line) < 4:
|
|
62
|
+
continue
|
|
63
|
+
code, path = line[:2].strip() or "??", line[3:].strip()
|
|
64
|
+
if " -> " in path: # renames
|
|
65
|
+
path = path.split(" -> ", 1)[1]
|
|
66
|
+
out.append((code, path.strip('"')))
|
|
67
|
+
return out
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def diff(cwd: str, ref: Optional[str] = None, paths: Optional[Sequence[str]] = None) -> str:
|
|
71
|
+
args = ["diff"]
|
|
72
|
+
if ref:
|
|
73
|
+
args.append(ref)
|
|
74
|
+
args += ["--no-color"]
|
|
75
|
+
if paths:
|
|
76
|
+
args += ["--"] + list(paths)
|
|
77
|
+
return run(args, cwd, check=False)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def diff_numstat(cwd: str, ref: Optional[str] = None) -> List[Tuple[int, int, str]]:
|
|
81
|
+
"""[(insertions, deletions, path)]. Binary files report -1, -1."""
|
|
82
|
+
args = ["diff", "--numstat"]
|
|
83
|
+
if ref:
|
|
84
|
+
args.append(ref)
|
|
85
|
+
out = []
|
|
86
|
+
for line in run(args, cwd, check=False).splitlines():
|
|
87
|
+
parts = line.split("\t")
|
|
88
|
+
if len(parts) != 3:
|
|
89
|
+
continue
|
|
90
|
+
a, d, path = parts
|
|
91
|
+
out.append((-1 if a == "-" else int(a), -1 if d == "-" else int(d), path.strip()))
|
|
92
|
+
return out
|
|
93
|
+
|
|
94
|
+
|
|
95
|
+
def file_at(cwd: str, ref: str, path: str) -> Optional[str]:
|
|
96
|
+
"""Contents of ``path`` at ``ref``, or None if it did not exist there."""
|
|
97
|
+
p = subprocess.run(
|
|
98
|
+
["git", "show", "{}:{}".format(ref, path)], cwd=cwd,
|
|
99
|
+
capture_output=True, text=True, encoding="utf-8", errors="replace",
|
|
100
|
+
)
|
|
101
|
+
return None if p.returncode != 0 else p.stdout
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def add_paths(cwd: str, paths: Sequence[str]) -> None:
|
|
105
|
+
"""Stage exactly these paths. There is deliberately no 'add everything'."""
|
|
106
|
+
if not paths:
|
|
107
|
+
return
|
|
108
|
+
run(["add", "--"] + list(paths), cwd)
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def stash_push(cwd: str, message: str) -> bool:
|
|
112
|
+
"""Set the working tree aside. True if something was actually stashed."""
|
|
113
|
+
before = run(["stash", "list"], cwd).count("\n")
|
|
114
|
+
run(["stash", "push", "--include-untracked", "-m", message], cwd, check=False)
|
|
115
|
+
return run(["stash", "list"], cwd).count("\n") > before
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def stash_pop(cwd: str) -> None:
|
|
119
|
+
run(["stash", "pop"], cwd, check=False)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
DEPLOYING_BRANCHES = ("main", "master", "production", "prod", "release")
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
def refuses_to_run_here(cwd: str, allow: Sequence[str] = ()) -> Optional[str]:
|
|
126
|
+
"""Why this working copy must not be worked in directly, or None."""
|
|
127
|
+
if not is_repo(cwd):
|
|
128
|
+
return "not a git repository: {}".format(cwd)
|
|
129
|
+
branch = current_branch(cwd)
|
|
130
|
+
if branch in DEPLOYING_BRANCHES and branch not in allow:
|
|
131
|
+
return (
|
|
132
|
+
"on branch '{}'. Work is not done on a branch that deploys.\n"
|
|
133
|
+
" Check out a working branch first, or pass --allow-branch {} if this\n"
|
|
134
|
+
" repository really does not publish from it.".format(branch, branch)
|
|
135
|
+
)
|
|
136
|
+
return None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def worktree_add(repo: str, path: str, branch: str, base: str) -> None:
|
|
140
|
+
run(["worktree", "add", "-b", branch, path, base], repo)
|
|
141
|
+
|
|
142
|
+
|
|
143
|
+
def worktree_remove(repo: str, path: str, force: bool = True) -> None:
|
|
144
|
+
args = ["worktree", "remove"]
|
|
145
|
+
if force:
|
|
146
|
+
args.append("--force")
|
|
147
|
+
args.append(path)
|
|
148
|
+
run(args, repo, check=False)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def worktree_list(repo: str) -> List[str]:
|
|
152
|
+
out = []
|
|
153
|
+
for line in run(["worktree", "list", "--porcelain"], repo, check=False).splitlines():
|
|
154
|
+
if line.startswith("worktree "):
|
|
155
|
+
out.append(line[len("worktree "):].strip())
|
|
156
|
+
return out
|
|
157
|
+
|
|
158
|
+
|
|
159
|
+
def rel_to_top(cwd: str, path: str) -> str:
|
|
160
|
+
top = os.path.abspath(toplevel(cwd))
|
|
161
|
+
ap = os.path.abspath(path)
|
|
162
|
+
return os.path.relpath(ap, top).replace("\\", "/")
|