loki-mode 9.8.0 → 9.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. package/README.md +19 -14
  2. package/SKILL.md +3 -2
  3. package/VERSION +1 -1
  4. package/autonomy/loki +122 -1
  5. package/autonomy/run.sh +49 -2
  6. package/dashboard/__init__.py +1 -1
  7. package/dashboard/api_evidence.py +411 -0
  8. package/dashboard/api_operator.py +283 -0
  9. package/dashboard/api_phases.py +262 -0
  10. package/dashboard/api_releases.py +242 -0
  11. package/dashboard/api_runs.py +477 -0
  12. package/dashboard/api_tests.py +444 -0
  13. package/dashboard/api_v2.py +47 -1
  14. package/dashboard/server.py +54 -0
  15. package/dashboard/static/index.html +246 -135
  16. package/docs/ARCHITECTURE-OVERVIEW.md +5 -3
  17. package/docs/CAPABILITY-BACKLOG.md +53 -0
  18. package/docs/COMPARISON.md +2 -2
  19. package/docs/COMPETITIVE-ANALYSIS.md +1 -1
  20. package/docs/COMPETITIVE-SCORECARD.md +422 -0
  21. package/docs/DASHBOARD-9.12-EVIDENCE.md +97 -0
  22. package/docs/DASHBOARD-ARCHITECTURE.md +423 -0
  23. package/docs/DEMOS.md +21 -23
  24. package/docs/HANDOFF-2026-08-03.md +439 -0
  25. package/docs/INSTALLATION.md +17 -10
  26. package/docs/OUTCOME-FRONTIER.md +536 -0
  27. package/docs/PROMPT-ABLATION-RESULT.md +97 -0
  28. package/docs/TOOLS.md +800 -0
  29. package/docs/alternative-installations.md +2 -3
  30. package/docs/audit-logging.md +44 -35
  31. package/docs/authentication.md +13 -2
  32. package/docs/authorization.md +87 -81
  33. package/docs/git-workflow.md +6 -3
  34. package/docs/metrics.md +15 -16
  35. package/docs/network-security.md +16 -13
  36. package/docs/openclaw-integration.md +36 -556
  37. package/docs/show-hn-post.md +2 -2
  38. package/docs/siem-integration.md +39 -36
  39. package/loki-ts/dist/loki.js +18 -18
  40. package/mcp/__init__.py +1 -1
  41. package/package.json +2 -2
  42. package/plugins/loki-mode/.claude-plugin/plugin.json +1 -1
  43. package/references/confidence-routing.md +18 -1
  44. package/references/invariant-checks.md +13 -8
  45. package/references/magic-rarv-integration.md +0 -1
  46. package/references/multi-provider.md +27 -5
  47. package/skills/healing.md +4 -2
  48. package/tools/audit-docs.py +488 -0
  49. package/tools/baseline-pin.py +19 -1
  50. package/tools/calibration-audit.py +523 -0
  51. package/tools/ci-gate.py +19 -1
  52. package/tools/cost-forecast.py +344 -0
  53. package/tools/cost-guard.py +19 -1
  54. package/tools/cost-history.py +19 -1
  55. package/tools/cost-per-outcome.py +394 -0
  56. package/tools/estimate-run.py +19 -1
  57. package/tools/evidence-freshness.py +307 -0
  58. package/tools/gate-init.py +19 -1
  59. package/tools/gate-report.py +19 -1
  60. package/tools/gate-simulate.py +570 -0
  61. package/tools/gate-trend.py +354 -0
  62. package/tools/model-advisor.py +52 -1
  63. package/tools/policy-load.py +19 -1
  64. package/tools/prompt-cost.py +363 -0
  65. package/tools/prompt-diff.py +448 -0
  66. package/tools/prompt-lint.py +448 -0
  67. package/tools/receipt-bundle.py +72 -2
  68. package/tools/receipt-diff.py +19 -1
  69. package/tools/receipt-find.py +19 -1
  70. package/tools/receipt-stats.py +380 -0
  71. package/tools/receipt-timeline.py +478 -0
  72. package/tools/receipt-verify-batch.py +291 -0
  73. package/tools/run-replay.py +19 -1
  74. package/tools/signing-status.py +19 -1
  75. package/tools/token-guard.py +19 -1
  76. package/tools/token-tax.py +375 -0
  77. package/tools/tool-index.py +19 -1
  78. package/tools/verification-tax.py +277 -0
  79. package/tools/verify-chain.py +361 -0
@@ -0,0 +1,354 @@
1
+ #!/usr/bin/env python3
2
+ """Is the merge gate getting better, getting worse, or going BLIND?
3
+
4
+ WHY THIS EXISTS. tools/gate-log.py made the verdicts durable and reports the
5
+ totals. Totals answer "how have we done", which is the wrong question for a
6
+ gate. The question an engineer asks before trusting a merge is directional:
7
+
8
+ is this gate improving, or is it quietly degrading?
9
+
10
+ gate-log's own trend field compares two halves on one axis, the pass rate. That
11
+ is enough to say "healthier" or "worse", and it is deliberately not enough to
12
+ say WHY, because on one axis the two ways a gate degrades are indistinguishable.
13
+ This file separates them, because they call for opposite responses.
14
+
15
+ THE DISTINCTION THIS TOOL IS FOR:
16
+
17
+ a rising FAIL rate = the gate is doing its job, and the code is worse
18
+ a rising UNEVALUABLE = the gate is going BLIND, and knows nothing at all
19
+
20
+ The second is more urgent and reads as less urgent, which is how it survives. A
21
+ failing gate is loud: someone is blocked, someone investigates. A blind gate is
22
+ silent and its numbers look BETTER the longer it stays blind, because an axis
23
+ that cannot be evaluated can never fail. Three weeks of blindness on the receipt
24
+ axis renders as a falling failure rate, which is indistinguishable from three
25
+ weeks of improvement. So GOING BLIND outranks WORSENING in the verdict here: an
26
+ operator told only "worsening" fixes the failures, re-runs, sees the failure
27
+ rate drop, and is still blind.
28
+
29
+ THE ARITHMETIC THAT KILLS THIS, and the reason there is no subtraction below:
30
+
31
+ unevaluable = total - passes - failures
32
+ pass_rate = 1 - fail_rate
33
+
34
+ Both are natural, both fold the third category into one of the other two, and
35
+ both are the exact defect gate-log.py's header names. Every rate here is counted
36
+ from its OWN bucket over the shared denominator. Nothing is derived by
37
+ complement, so no category can absorb another's mass.
38
+
39
+ FEWER THAN 2 RECORDS IS NOT "FLAT". Flat is a claim about change over time and
40
+ needs two points to make. One record reports INSUFFICIENT DATA and exits
41
+ non-zero, because this is a gate: saying "I cannot establish a trend" while
42
+ exiting 0 tells CI the trend axis was checked and was fine.
43
+
44
+ A CORRUPT LINE IS COUNTED, NEVER SKIPPED. Skipping is how a log half-eaten by a
45
+ crashed writer reports a clean history of whatever survived. Corrupt lines get
46
+ their own count, are included in the window and the denominator, and force the
47
+ blind exit code, since a line we could not read is a verdict we do not have.
48
+
49
+ AN EMPTY LOG EXITS 3, NOT 0. "No verdicts recorded" and "no verdict ever
50
+ blocked" are opposite facts. Absent is not zero.
51
+
52
+ WHAT IS DELIBERATELY NOT HERE: any cost figure, and therefore no use of
53
+ record_is_measured() from autonomy/lib/efficiency_cost.py. gate-log rows carry
54
+ no measured cost by design (see its "WHAT IS DELIBERATELY NOT HERE"), and
55
+ recovering one by parsing a policy's human-readable reason string would restate
56
+ a predicate efficiency_cost.py owns. That drift is what this repo fixed across
57
+ four surfaces. No cost surface beats a re-derived one, so this tool reports
58
+ categories only.
59
+
60
+ Usage:
61
+ tools/gate-trend.py [--file .loki/gate-log.jsonl] [--window N] [--json]
62
+
63
+ Exit: 0 improving or steady with every record readable and passing-or-failing
64
+ cleanly, 1 a genuine regression in the FAIL rate, 2 could not evaluate (the
65
+ gate is blind: unevaluable or corrupt records present, or fewer than 2 records
66
+ to compare), 3 the log exists but holds no records, 64 usage error, 66 no log
67
+ file at that path.
68
+ """
69
+
70
+ import argparse
71
+ import json
72
+ import os
73
+ import sys
74
+
75
+ sys.dont_write_bytecode = True
76
+
77
+ PASSED, FAILED, COULD_NOT_CHECK, NOTHING, USAGE, MISSING = 0, 1, 2, 3, 64, 66
78
+
79
+ DEFAULT_LOG = os.path.join(".loki", "gate-log.jsonl")
80
+
81
+ # THE ONE MAPPING from a recorded verdict word to a bucket. Three buckets,
82
+ # never two. Folding the blind bucket into the passing one is the whole defect
83
+ # this file exists to prevent, and it is a one-word edit -- which is exactly
84
+ # what tests/test_gate_trend.py mutates.
85
+ _CATEGORY = {"PASS": "pass", "FAIL": "fail", "UNEVALUABLE": "unevaluable"}
86
+
87
+ # Not a verdict any gate can emit; what a damaged line becomes. Kept out of
88
+ # _CATEGORY so no recorded state word can ever map into it.
89
+ _CORRUPT = "corrupt"
90
+
91
+ _ORDER = ["pass", "fail", "unevaluable", _CORRUPT]
92
+
93
+ # A category that means the history itself is unreadable on that record. Both
94
+ # force the blind exit: a verdict we could not read is a verdict we do not have.
95
+ _BLIND = ("unevaluable", _CORRUPT)
96
+
97
+
98
+ class _Parser(argparse.ArgumentParser):
99
+ """argparse exits 2 on a usage error. Here 2 means "could not check".
100
+
101
+ A mistyped flag would otherwise be indistinguishable from this gate
102
+ reporting that it was blind, and a CI job branching on the code would treat
103
+ an operator's typo as a real finding about the merge. 64 is the convention.
104
+ It also means a stray positional is an ERROR rather than something read as
105
+ a log path: receipt-attest once read `--help` as a proof path and issued a
106
+ verdict about it.
107
+ """
108
+
109
+ def error(self, message):
110
+ self.print_usage(sys.stderr)
111
+ sys.stderr.write("gate-trend: %s\n" % message)
112
+ raise SystemExit(USAGE)
113
+
114
+
115
+ def classify(verdict):
116
+ """Bucket ONE recorded verdict. Anything unrecognised is unevaluable.
117
+
118
+ Never defaults to "pass". A default of pass is how an aggregator launders
119
+ every shape it did not anticipate into green, and the shapes it did not
120
+ anticipate are precisely the broken ones.
121
+ """
122
+ if not isinstance(verdict, dict):
123
+ return _CATEGORY["UNEVALUABLE"]
124
+ state = verdict.get("state")
125
+ if isinstance(state, str) and state.upper() in _CATEGORY:
126
+ return _CATEGORY[state.upper()]
127
+ return _CATEGORY["UNEVALUABLE"]
128
+
129
+
130
+ def category_of(entry):
131
+ """A log entry's bucket, re-derived from its verdict when it lacks one.
132
+
133
+ Trusting a stored "category" blindly would let a hand-edited log assert
134
+ anything; the embedded verdict stays authoritative. An entry with neither is
135
+ unevaluable, not a pass. Deliberately checks _CATEGORY.values() and not
136
+ _ORDER, so a stored "corrupt" cannot inflate the count of lines this reader
137
+ actually failed to parse.
138
+ """
139
+ if not isinstance(entry, dict):
140
+ return _CATEGORY["UNEVALUABLE"]
141
+ stored = entry.get("category")
142
+ if isinstance(stored, str) and stored in _CATEGORY.values():
143
+ return stored
144
+ return classify(entry.get("verdict"))
145
+
146
+
147
+ def read_categories(path):
148
+ """Every line in order, as a list of buckets. Corrupt lines INCLUDED.
149
+
150
+ A corrupt line that is merely skipped is a verdict deleted from the record
151
+ by the reader, and the resulting trend describes a history that never
152
+ happened. Returning it in sequence also keeps it inside --window, so a
153
+ window cannot launder corruption by sliding past it.
154
+ """
155
+ out = []
156
+ with open(path, "r", encoding="utf-8") as fh:
157
+ for raw in fh:
158
+ if not raw.strip():
159
+ continue # a trailing newline is not a damaged record
160
+ try:
161
+ entry = json.loads(raw)
162
+ except ValueError:
163
+ out.append(_CORRUPT)
164
+ continue
165
+ if not isinstance(entry, dict):
166
+ out.append(_CORRUPT)
167
+ continue
168
+ out.append(category_of(entry))
169
+ return out
170
+
171
+
172
+ def _rates(seq):
173
+ """One rate per bucket, each counted from its OWN members.
174
+
175
+ No complement, no subtraction from the total. `unevaluable = 1 - pass` is
176
+ the one line that would make a blind gate look healthy, so every rate is
177
+ an independent count over the shared denominator.
178
+ """
179
+ n = float(len(seq))
180
+ return dict((name, sum(1 for c in seq if c == name) / n) for name in _ORDER)
181
+
182
+
183
+ def analyse(seq):
184
+ """Split the window in half and compare each category's rate independently.
185
+
186
+ Returns a summary whose "direction" is None -- UNKNOWN, never "flat" --
187
+ when there are fewer than 2 records. Flat is a claim about change, and one
188
+ point cannot support it.
189
+ """
190
+ counts = dict((name, sum(1 for c in seq if c == name)) for name in _ORDER)
191
+ blind = sum(counts[name] for name in _BLIND)
192
+ summary = {
193
+ "records": len(seq),
194
+ "counts": counts,
195
+ "blind_records": blind,
196
+ "direction": None,
197
+ "older": None,
198
+ "newer": None,
199
+ "window": None,
200
+ }
201
+ if len(seq) < 2:
202
+ return summary
203
+
204
+ half = len(seq) // 2
205
+ older, newer = _rates(seq[:half]), _rates(seq[half:])
206
+ summary["older"] = older
207
+ summary["newer"] = newer
208
+ summary["window"] = [half, len(seq) - half]
209
+
210
+ # PRECEDENCE. Blindness first: a gate that cannot evaluate has not passed,
211
+ # and its failure rate falls precisely because it is blind. Reporting
212
+ # "improving" off a falling failure rate while the blind rate climbs is the
213
+ # inversion this tool exists to make impossible.
214
+ blind_older = older["unevaluable"] + older[_CORRUPT]
215
+ blind_newer = newer["unevaluable"] + newer[_CORRUPT]
216
+ if blind_newer > blind_older:
217
+ summary["direction"] = "going_blind"
218
+ elif blind_newer:
219
+ # Blind and getting less blind is STILL BLIND, never "improving". A
220
+ # gate 75% unable to evaluate, printing "IMPROVING -- fewer blocked or
221
+ # blind runs than before", is the same inversion as the rising case
222
+ # wearing a recovery story: the headline an operator reads says the
223
+ # thing is getting better while most of its axes report nothing. The
224
+ # exit code alone is not enough here, because the human reads the word.
225
+ summary["direction"] = "still_blind"
226
+ elif newer["fail"] > older["fail"]:
227
+ summary["direction"] = "worsening"
228
+ elif newer["fail"] < older["fail"] or blind_newer < blind_older:
229
+ summary["direction"] = "improving"
230
+ else:
231
+ summary["direction"] = "steady"
232
+ summary["blind_rate_older"] = blind_older
233
+ summary["blind_rate_newer"] = blind_newer
234
+ return summary
235
+
236
+
237
+ def exit_code(summary):
238
+ """Weakest link. Blind outranks failed, and INSUFFICIENT DATA is blind.
239
+
240
+ A gate exiting 0 while its own output says it could not establish a trend
241
+ tells CI the axis was checked and was fine. There is no reading of "fewer
242
+ than 2 records" that earns a green.
243
+ """
244
+ if summary["blind_records"]:
245
+ return COULD_NOT_CHECK
246
+ if summary["direction"] is None:
247
+ return COULD_NOT_CHECK
248
+ if summary["direction"] == "worsening":
249
+ return FAILED
250
+ return PASSED
251
+
252
+
253
+ _HEADLINE = {
254
+ "going_blind": "GOING BLIND -- the unevaluable rate is RISING. This is "
255
+ "more urgent than a rising failure rate: a gate that "
256
+ "cannot evaluate can never fail, so blindness renders as "
257
+ "improvement.",
258
+ "still_blind": "STILL BLIND -- the unevaluable rate fell but is NOT zero. "
259
+ "Less blind is not sighted: the remaining blind runs "
260
+ "checked nothing, so this is not an improvement to report.",
261
+ "worsening": "WORSENING -- the failure rate is rising. The gate is "
262
+ "working; the code getting to it is worse.",
263
+ "improving": "IMPROVING -- fewer blocked or blind runs than before.",
264
+ "steady": "STEADY -- no measured change in either rate.",
265
+ }
266
+
267
+
268
+ def render(summary):
269
+ counts = summary["counts"]
270
+ lines = ["gate-trend: %d record(s) in window" % summary["records"], ""]
271
+ for name in _ORDER:
272
+ label = {"unevaluable": "unevaluable (NOT a pass)",
273
+ _CORRUPT: "corrupt (counted, not skipped)"}.get(name, name)
274
+ lines.append(" %-30s %d" % (label, counts[name]))
275
+ lines.append("")
276
+ if summary["direction"] is None:
277
+ lines.append(
278
+ "trend: INSUFFICIENT DATA -- %d record(s), 2 needed to compare. "
279
+ "Not flat: flat is a claim about change." % summary["records"])
280
+ return "\n".join(lines)
281
+ lines.append("trend: %s" % _HEADLINE[summary["direction"]])
282
+ lines.append("")
283
+ older, newer = summary["older"], summary["newer"]
284
+ lines.append(" %-14s %8s -> %8s" % ("rate", "older", "newer"))
285
+ for name in _ORDER:
286
+ lines.append(" %-14s %7.0f%% -> %7.0f%%"
287
+ % (name, older[name] * 100, newer[name] * 100))
288
+ lines.append("")
289
+ lines.append(" window: %d older then %d newer record(s)"
290
+ % (summary["window"][0], summary["window"][1]))
291
+ if summary["blind_records"]:
292
+ lines.append(
293
+ " NOTE: %d record(s) could not be evaluated or read. Those are "
294
+ "counted here and are neither passes nor failures."
295
+ % summary["blind_records"])
296
+ return "\n".join(lines)
297
+
298
+
299
+ def main(argv=None):
300
+ ap = _Parser(
301
+ description="Report whether the merge gate is improving, worsening, "
302
+ "or going blind over its recorded verdicts.")
303
+ # No positional argument, deliberately. A stray path then lands on
304
+ # _Parser.error() -> 64, so a mistyped invocation can never be read as a
305
+ # log to judge.
306
+ ap.add_argument("--file", default=DEFAULT_LOG,
307
+ help="JSONL log written by gate-log.py (default: %s)"
308
+ % DEFAULT_LOG)
309
+ ap.add_argument("--window", type=int, default=None,
310
+ help="compare only the most recent N records")
311
+ ap.add_argument("--json", action="store_true", dest="as_json",
312
+ help="emit the summary as JSON")
313
+ args = ap.parse_args(argv)
314
+
315
+ # `is None` and never falsy: --window 0 is a value an operator typed, and
316
+ # swallowing it as "not given" would silently analyse the whole log while
317
+ # the operator believes a window is in force.
318
+ if args.window is not None and args.window < 2:
319
+ ap.error("--window must be at least 2: fewer than 2 records cannot "
320
+ "establish a trend")
321
+
322
+ if not os.path.exists(args.file):
323
+ sys.stderr.write(
324
+ "gate-trend: no log at %s -- UNKNOWN, not a clean history. A gate "
325
+ "whose verdicts were never recorded is not a gate that never "
326
+ "blocked.\n" % args.file)
327
+ return MISSING
328
+ try:
329
+ seq = read_categories(args.file)
330
+ except OSError as exc:
331
+ # An unreachable log is "could not check", never "checked and failed".
332
+ sys.stderr.write("gate-trend: could not read %s: %s\n"
333
+ % (args.file, exc))
334
+ return COULD_NOT_CHECK
335
+
336
+ if not seq:
337
+ sys.stderr.write(
338
+ "gate-trend: %s holds no records -- nothing to trend. An empty log "
339
+ "must never read as 'never blocked'.\n" % args.file)
340
+ return NOTHING
341
+
342
+ if args.window is not None:
343
+ seq = seq[-args.window:]
344
+
345
+ summary = analyse(seq)
346
+ if args.as_json:
347
+ print(json.dumps(summary, indent=2, sort_keys=True))
348
+ else:
349
+ print(render(summary))
350
+ return exit_code(summary)
351
+
352
+
353
+ if __name__ == "__main__":
354
+ sys.exit(main())
@@ -138,6 +138,24 @@ SWEBENCH_CITATION = {
138
138
  }
139
139
 
140
140
 
141
+ class _Parser(argparse.ArgumentParser):
142
+ """Usage errors exit 64, not argparse's default 2.
143
+
144
+ In this repo's convention 2 means "could NOT be checked" -- a real
145
+ answer about the subject. A mistyped flag is not that: it is an error
146
+ about the INVOCATION, and nothing about the subject was examined. The
147
+ two call for opposite responses, since retrying cannot fix a typo.
148
+
149
+ argparse exits 2 for every usage error unless this is overridden, so
150
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
151
+ """
152
+
153
+ def error(self, message):
154
+ self.print_usage(sys.stderr)
155
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
156
+ raise SystemExit(64)
157
+
158
+
141
159
  def _num(v):
142
160
  """Non-bool int/float, else None. Never coerces junk to 0."""
143
161
  if isinstance(v, bool) or not isinstance(v, (int, float)):
@@ -460,6 +478,39 @@ def render(adv):
460
478
  lines.append(" Cheapest rate: none cheaper than the model "
461
479
  "already in use")
462
480
 
481
+ # LOCAL CALIBRATION, shown as a CAVEAT and never as a ranking input.
482
+ #
483
+ # This module's own docstring says quality "is not something any cost
484
+ # record can answer", and that stands: nothing below re-ranks a candidate
485
+ # or weights a saving. advise() is untouched, so the recommendation is
486
+ # provably identical with and without this block.
487
+ #
488
+ # What it adds is one honest local fact. tools/calibration-audit.py scores
489
+ # council votes against council outcomes on THIS workload, so unlike the
490
+ # SWE-bench citation it is not borrowed from someone else's task. It is
491
+ # surfaced ABOVE that citation for exactly that reason: local evidence
492
+ # first, external evidence second.
493
+ #
494
+ # THE LIMIT, restated here because a reader arriving at a cost tool will
495
+ # not have read the audit's header: the audit measures AGREEMENT WITH THE
496
+ # MAJORITY, not correctness. The council outcome is derived from the votes,
497
+ # so a voter partly causes its own label. Treating that as a quality score
498
+ # would be the fabricated authority this tool exists to refuse -- so it is
499
+ # printed as a pointer, with the caveat attached, and never as a number
500
+ # that moves a recommendation.
501
+ lines.append("")
502
+ lines.append(" Local calibration -- a CAVEAT, not a ranking input:")
503
+ lines.append(" Cheaper is not better if the cheaper model agrees with "
504
+ "your council less often.")
505
+ lines.append(" This tool does NOT measure that and does not pretend to. "
506
+ "For the local signal:")
507
+ lines.append(" python3 tools/calibration-audit.py <workspace>")
508
+ lines.append(" Read its header first: it scores agreement with the "
509
+ "majority, NOT accuracy.")
510
+ lines.append(" No artifact records whether the council was right, so no "
511
+ "quality claim is")
512
+ lines.append(" available from any tool in this repo today.")
513
+
463
514
  cite = adv["swebench_citation"]
464
515
  lines.append("")
465
516
  lines.append(" Cited external benchmark (%s) -- NOT a measurement of your "
@@ -477,7 +528,7 @@ def render(adv):
477
528
 
478
529
 
479
530
  def main(argv=None):
480
- ap = argparse.ArgumentParser(
531
+ ap = _Parser(
481
532
  description="Recommend a cheaper model from this workspace's measured "
482
533
  "cost history, and quantify the saving.")
483
534
  ap.add_argument("workspace", nargs="?", default=".")
@@ -63,6 +63,24 @@ class PolicyError(Exception):
63
63
  """A policy that must not be handed to a gate."""
64
64
 
65
65
 
66
+ class _Parser(argparse.ArgumentParser):
67
+ """Usage errors exit 64, not argparse's default 2.
68
+
69
+ In this repo's convention 2 means "could NOT be checked" -- a real
70
+ answer about the subject. A mistyped flag is not that: it is an error
71
+ about the INVOCATION, and nothing about the subject was examined. The
72
+ two call for opposite responses, since retrying cannot fix a typo.
73
+
74
+ argparse exits 2 for every usage error unless this is overridden, so
75
+ every tool needs it. tests/test_tool_exit_contract.py asserts it.
76
+ """
77
+
78
+ def error(self, message):
79
+ self.print_usage(sys.stderr)
80
+ sys.stderr.write("%s: error: %s\n" % (self.prog, message))
81
+ raise SystemExit(64)
82
+
83
+
66
84
  def _check_max_usd(value):
67
85
  # bool is a subclass of int: `true` would otherwise become a $1.00 ceiling.
68
86
  if isinstance(value, bool) or not isinstance(value, (int, float)):
@@ -157,7 +175,7 @@ def as_args(policy):
157
175
 
158
176
 
159
177
  def main(argv=None):
160
- ap = argparse.ArgumentParser(
178
+ ap = _Parser(
161
179
  description="Load and validate a merge policy file for ci-gate.py.")
162
180
  ap.add_argument("--file", default=DEFAULT_FILE,
163
181
  help="policy file to load; default {}".format(DEFAULT_FILE))