@andresmassello/uscha 2.0.0 → 2.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +21 -4
- package/package.json +1 -1
- package/uscha-kit/.claude/skills/uscha-adr-refine/SKILL.md +2 -0
- package/uscha-kit/.claude/skills/uscha-characterize/SKILL.md +2 -0
- package/uscha-kit/.claude/skills/uscha-devloop/SKILL.md +171 -15
- package/uscha-kit/.claude/skills/uscha-devloop/qa_ledger.py +2089 -97
- package/uscha-kit/.claude/skills/uscha-discovery/SKILL.md +58 -4
- package/uscha-kit/.claude/skills/uscha-mirador/SKILL.md +2 -0
- package/uscha-kit/.claude/skills/uscha-reverse-discovery/SKILL.md +2 -0
- package/uscha-kit/.claude/skills/uscha-rubric/SKILL.md +2 -0
- package/uscha-kit/.claude/skills/uscha-status/SKILL.md +40 -0
- package/uscha-kit/.claude/skills/uscha-sysdoc/SKILL.md +2 -0
- package/uscha-kit/.claude-plugin/plugin.json +2 -2
- package/uscha-kit/.codex-plugin/plugin.json +1 -1
- package/uscha-kit/README.md +249 -11
- package/uscha-kit/VERSION +1 -1
- package/uscha-kit/install-uscha.py +18 -2
- package/uscha-kit/skills/uscha-adr-refine/SKILL.md +2 -0
- package/uscha-kit/skills/uscha-characterize/SKILL.md +2 -0
- package/uscha-kit/skills/uscha-devloop/SKILL.md +171 -15
- package/uscha-kit/skills/uscha-devloop/qa_ledger.py +2089 -97
- package/uscha-kit/skills/uscha-discovery/SKILL.md +58 -4
- package/uscha-kit/skills/uscha-mirador/SKILL.md +2 -0
- package/uscha-kit/skills/uscha-reverse-discovery/SKILL.md +2 -0
- package/uscha-kit/skills/uscha-rubric/SKILL.md +2 -0
- package/uscha-kit/skills/uscha-status/SKILL.md +40 -0
- package/uscha-kit/skills/uscha-sysdoc/SKILL.md +2 -0
- package/uscha-kit/templates/CLAUDE.md +16 -0
- package/uscha-kit/templates/CONSTITUTION.md +55 -3
- package/uscha-kit/templates/docs/adr/README.md +15 -0
- package/uscha-kit/templates/scripts/smoke-report-example.json +24 -0
- package/uscha-kit/uscha.config.json +8 -2
|
@@ -36,6 +36,9 @@ Usage (see `--help` on each subcommand):
|
|
|
36
36
|
[--ruff reports/ruff.json --mypy reports/mypy.txt]
|
|
37
37
|
qa_ledger.py log-gate --repo backend-api --iteration 1 --kind golden-diff \
|
|
38
38
|
--verdict pass|fail|not-run [--count N] [--note "..."]
|
|
39
|
+
qa_ledger.py corpus-run --repo backend-api --corpus corpus.jsonl \
|
|
40
|
+
--command "python -m parser" [--threshold 99] [--ac AC-FIELD-01]
|
|
41
|
+
qa_ledger.py smoke-ingest --repo backend-api --report reports/smoke.json [--json]
|
|
39
42
|
qa_ledger.py flag-blocker --repo backend-api --kind constitution --note "INV-XX breached" \
|
|
40
43
|
[--resolve]
|
|
41
44
|
qa_ledger.py production-finding --repo backend-api --severity HIGH --title "..." --evidence "..."
|
|
@@ -44,11 +47,13 @@ Usage (see `--help` on each subcommand):
|
|
|
44
47
|
qa_ledger.py rebuild --mode baseline --config uscha.config.json [--out REBUILD-BASELINE.json]
|
|
45
48
|
qa_ledger.py rebuild --mode compare --baseline REBUILD-BASELINE.json [--json]
|
|
46
49
|
qa_ledger.py simplicity-check --diff changes.diff [--config uscha.config.json] [--json]
|
|
47
|
-
qa_ledger.py simplicity-check --from-git --base main
|
|
50
|
+
qa_ledger.py simplicity-check --from-git --base main (advisory: exit 0)
|
|
51
|
+
qa_ledger.py simplicity-check --from-git --base main --max-lines-added 400 --gate
|
|
48
52
|
qa_ledger.py pit-check --report target/pit-reports/*/mutations.xml [--min-score 60] [--json]
|
|
49
53
|
qa_ledger.py gate-check --from-git --base main [--strict] [--json]
|
|
50
54
|
qa_ledger.py spec-check --spec SPEC.md [--spec ACCEPTANCE.md] [--strict] [--json]
|
|
51
55
|
qa_ledger.py golden-diff [--dir .] [--labels golden-labels.json] [--json]
|
|
56
|
+
qa_ledger.py operability --repo backend-api [--json] (ADR-048: exit 0 always)
|
|
52
57
|
"""
|
|
53
58
|
|
|
54
59
|
import argparse
|
|
@@ -1883,14 +1888,55 @@ GOLDEN_REQUIRED_CEILING = 49 # ADR-002: no approved golden -> NOT READY (does n
|
|
|
1883
1888
|
RISK_PROFILES = {
|
|
1884
1889
|
"A": {"qa_tools_order": ["code-review"]},
|
|
1885
1890
|
"B": {"qa_tools_order": ["code-review", "improve"]},
|
|
1886
|
-
"C": {"qa_tools_order": ["code-review", "judgment-day", "improve"]
|
|
1891
|
+
"C": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
|
|
1892
|
+
"operability.gate": True},
|
|
1887
1893
|
"D": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
|
|
1888
|
-
"coverage_threshold": 70, "golden_required": True
|
|
1894
|
+
"coverage_threshold": 70, "golden_required": True,
|
|
1895
|
+
"operability.gate": True},
|
|
1889
1896
|
"E": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
|
|
1890
|
-
"coverage_threshold": 80, "golden_required": True
|
|
1897
|
+
"coverage_threshold": 80, "golden_required": True,
|
|
1898
|
+
"operability.gate": True},
|
|
1891
1899
|
}
|
|
1892
1900
|
|
|
1893
1901
|
|
|
1902
|
+
def _knob_get(mapping, key):
|
|
1903
|
+
"""Read a knob out of a defaults-shaped mapping by its (possibly DOTTED) name ->
|
|
1904
|
+
(found, value). Dotted since 2.2.0 (ADR-048): the first profile-owned knob that lives one
|
|
1905
|
+
level down is `defaults.operability.gate`, in the same block shape the config already uses
|
|
1906
|
+
for `simplicity.gate` and `waste.gate`. A non-dict on the way down is NOT a hit --
|
|
1907
|
+
`operability: true` declares nothing this ladder can read, and guessing what it meant is
|
|
1908
|
+
exactly how a preset goes silently inert (the defect ADR-001 was amended for)."""
|
|
1909
|
+
cur = mapping
|
|
1910
|
+
for part in key.split("."):
|
|
1911
|
+
if not isinstance(cur, dict) or part not in cur:
|
|
1912
|
+
return False, None
|
|
1913
|
+
cur = cur[part]
|
|
1914
|
+
return True, cur
|
|
1915
|
+
|
|
1916
|
+
|
|
1917
|
+
def _knob_set(mapping, key, value):
|
|
1918
|
+
"""Write a (possibly dotted) knob, creating the intermediate objects. Intermediates are
|
|
1919
|
+
COPIED on the way down: callers hand this shallow copies of `defaults`, and mutating a
|
|
1920
|
+
nested dict in place would reach back into the config the caller promised not to touch.
|
|
1921
|
+
Returns False -- writing nothing -- when something that is not an object already sits on
|
|
1922
|
+
the path: an explicit declaration always wins, even a malformed one, and naming a
|
|
1923
|
+
malformed one is `_validate_init_config`'s job, not this function's."""
|
|
1924
|
+
parts = key.split(".")
|
|
1925
|
+
cur = mapping
|
|
1926
|
+
for part in parts[:-1]:
|
|
1927
|
+
nxt = cur.get(part)
|
|
1928
|
+
if nxt is None:
|
|
1929
|
+
nxt = {}
|
|
1930
|
+
elif isinstance(nxt, dict):
|
|
1931
|
+
nxt = dict(nxt)
|
|
1932
|
+
else:
|
|
1933
|
+
return False
|
|
1934
|
+
cur[part] = nxt
|
|
1935
|
+
cur = nxt
|
|
1936
|
+
cur[parts[-1]] = value
|
|
1937
|
+
return True
|
|
1938
|
+
|
|
1939
|
+
|
|
1894
1940
|
def _apply_risk_profile(defaults):
|
|
1895
1941
|
"""Expand defaults['risk_profile'] into concrete knobs UNDER any explicit value (explicit
|
|
1896
1942
|
wins per key). Records the profile-provided keys in defaults['_risk_profile_keys'] so a cap
|
|
@@ -1904,8 +1950,9 @@ def _apply_risk_profile(defaults):
|
|
|
1904
1950
|
+ ", ".join(sorted(RISK_PROFILES)))
|
|
1905
1951
|
provided = []
|
|
1906
1952
|
for key, value in RISK_PROFILES[profile].items():
|
|
1907
|
-
if key
|
|
1908
|
-
|
|
1953
|
+
if _knob_get(defaults, key)[0]: # an explicit declaration always wins
|
|
1954
|
+
continue
|
|
1955
|
+
if _knob_set(defaults, key, list(value) if isinstance(value, list) else value):
|
|
1909
1956
|
provided.append(key)
|
|
1910
1957
|
defaults["_risk_profile_keys"] = provided
|
|
1911
1958
|
return defaults
|
|
@@ -1930,6 +1977,11 @@ ENGINE_DEFAULTS = {
|
|
|
1930
1977
|
"qa_tools_order": None,
|
|
1931
1978
|
"coverage_threshold": 60,
|
|
1932
1979
|
"golden_required": False,
|
|
1980
|
+
# ADR-048. The engine's own posture on operability is ADVISORY: `operability` measures and
|
|
1981
|
+
# records on every profile, but only C/D/E turn the record into a gate. A kit that gated
|
|
1982
|
+
# release + reset + RUNBOOK on every project would be the kit's opinion wearing an exit
|
|
1983
|
+
# code -- the same thing ADR-043 refused for the simplicity budget.
|
|
1984
|
+
"operability.gate": False,
|
|
1933
1985
|
}
|
|
1934
1986
|
ENGINE_DEFAULT_NOTES = {
|
|
1935
1987
|
"qa_tools_order": "not declared - convergence uses a window of --tools-per-cycle "
|
|
@@ -1940,17 +1992,35 @@ ENGINE_DEFAULT_NOTES = {
|
|
|
1940
1992
|
def _resolved_defaults(cfg):
|
|
1941
1993
|
"""(resolved defaults, origin per profile-owned key) for a RAW config -- the effective
|
|
1942
1994
|
settings `doctor` reports. Origin is `override` (declared in defaults), `profile <X>`, or
|
|
1943
|
-
`default`. Read-only: never mutates cfg and never writes anything back.
|
|
1995
|
+
`default`. Read-only: never mutates cfg and never writes anything back.
|
|
1996
|
+
|
|
1997
|
+
Works over a RAW config (what `doctor` reads) and over the FROZEN one inside a ledger
|
|
1998
|
+
alike: `init` expands the profile before freezing, so on the frozen copy a profile-supplied
|
|
1999
|
+
knob is already sitting in `defaults` and would otherwise read as a human `override`.
|
|
2000
|
+
`_risk_profile_keys` -- written by `_apply_risk_profile` for exactly this reason, and
|
|
2001
|
+
already read this way by the golden cap -- is what tells the two apart (ADR-048).
|
|
2002
|
+
|
|
2003
|
+
It tells them apart only while the value still IS the profile's, and that is the case the
|
|
2004
|
+
fresh review found: a human who edits the FROZEN copy leaves a declaration the profile
|
|
2005
|
+
never made, `_risk_profile_keys` still names the key, and `operability.gate: false` under
|
|
2006
|
+
profile E reported `origin profile E` -- crediting the preset with the opposite of what it
|
|
2007
|
+
supplies. The DECLARATION is therefore read first: a raw value that disagrees with what the
|
|
2008
|
+
profile would have written is an `override`, whoever typed it and whenever."""
|
|
1944
2009
|
raw = cfg.get("defaults") if isinstance(cfg, dict) else None
|
|
1945
2010
|
raw = dict(raw) if isinstance(raw, dict) else {}
|
|
1946
|
-
declared = set(raw)
|
|
1947
2011
|
profile = raw.get("risk_profile")
|
|
2012
|
+
from_profile = set(raw.get("_risk_profile_keys") or [])
|
|
1948
2013
|
expanded = _apply_risk_profile(dict(raw))
|
|
1949
2014
|
resolved, origin = {}, {}
|
|
1950
2015
|
for key, fallback in ENGINE_DEFAULTS.items():
|
|
1951
|
-
|
|
1952
|
-
if
|
|
2016
|
+
found, value = _knob_get(expanded, key)
|
|
2017
|
+
resolved[key] = value if found else fallback
|
|
2018
|
+
raw_found, raw_value = _knob_get(raw, key)
|
|
2019
|
+
profile_value = RISK_PROFILES.get(profile, {}).get(key) if profile else None
|
|
2020
|
+
if raw_found and not (key in from_profile and raw_value == profile_value):
|
|
1953
2021
|
origin[key] = "override"
|
|
2022
|
+
elif key in from_profile and profile:
|
|
2023
|
+
origin[key] = "profile %s" % profile
|
|
1954
2024
|
elif profile and key in RISK_PROFILES.get(profile, {}):
|
|
1955
2025
|
origin[key] = "profile %s" % profile
|
|
1956
2026
|
else:
|
|
@@ -2107,7 +2177,96 @@ def _validate_log_step_counts(args):
|
|
|
2107
2177
|
raise SystemExit("[qa_ledger] gated-reported cannot exceed reported")
|
|
2108
2178
|
|
|
2109
2179
|
|
|
2180
|
+
def _add_repo_to_config_file(path, entry, frozen_names):
|
|
2181
|
+
"""Mirror the new repo into uscha.config.json when that file IS the source the ledger was
|
|
2182
|
+
frozen from -- same repo names, in the same order. A file that has DRIFTED from the frozen
|
|
2183
|
+
config is somebody else's edit, and it is left alone with the divergence named: silently
|
|
2184
|
+
rewriting a config that no longer matches the ledger would be resolving a conflict the human
|
|
2185
|
+
has not seen. Returns the sentence to print, never raises: the ledger is the source of truth
|
|
2186
|
+
and its write already happened."""
|
|
2187
|
+
if not os.path.isfile(path):
|
|
2188
|
+
return "%s not found -- the ledger's frozen config is the only copy that changed" % path
|
|
2189
|
+
try:
|
|
2190
|
+
with open(path, "r", encoding="utf-8") as fh:
|
|
2191
|
+
cfg = json.load(fh)
|
|
2192
|
+
names = [r.get("name") for r in cfg.get("repos", [])]
|
|
2193
|
+
except (OSError, ValueError, AttributeError, TypeError) as exc:
|
|
2194
|
+
return "%s left untouched (unreadable: %s)" % (path, exc)
|
|
2195
|
+
if names != frozen_names:
|
|
2196
|
+
return ("%s left untouched: its repos %s differ from the ledger's frozen %s, so it is "
|
|
2197
|
+
"not the source this ledger was built from" % (path, names, frozen_names))
|
|
2198
|
+
cfg["repos"].append(dict(entry))
|
|
2199
|
+
tmp = path + ".tmp"
|
|
2200
|
+
try:
|
|
2201
|
+
with open(tmp, "w", encoding="utf-8") as fh:
|
|
2202
|
+
json.dump(cfg, fh, indent=2, ensure_ascii=False)
|
|
2203
|
+
fh.write("\n")
|
|
2204
|
+
os.replace(tmp, path)
|
|
2205
|
+
except OSError as exc:
|
|
2206
|
+
return "%s left untouched (not writable: %s)" % (path, exc)
|
|
2207
|
+
return "%s updated too (it is the source this ledger was frozen from)" % path
|
|
2208
|
+
|
|
2209
|
+
|
|
2210
|
+
def _init_add_repo(args):
|
|
2211
|
+
"""Append ONE repo to an EXISTING ledger without resetting it (2.2.0 field fix).
|
|
2212
|
+
|
|
2213
|
+
Until now the only door was re-running `init --config`, which builds a NEW ledger: the step
|
|
2214
|
+
counter, every repo's iterations and every snapshot went back to zero, so adding a second
|
|
2215
|
+
service to a live loop cost the evidence of the first. Editing QA-LEDGER.json by hand is not
|
|
2216
|
+
the workaround either -- `_load` verifies the sha256 that `_save` writes, so a hand edit
|
|
2217
|
+
turns the file into a refusal. That refusal is correct and stays: it is what makes 'measured
|
|
2218
|
+
beats narrated' worth anything. What was missing was a supported door, and this is it.
|
|
2219
|
+
|
|
2220
|
+
What this deliberately does NOT do: touch any existing repo's steps, snapshots or
|
|
2221
|
+
iterations, and re-freeze `defaults`. The config a project started under stays the config it
|
|
2222
|
+
ran under -- only the repo list grows. `_save` re-seals the checksum over the result.
|
|
2223
|
+
|
|
2224
|
+
The new repo starts with NO evidence, and readiness says so rather than hiding it: it enters
|
|
2225
|
+
`facts.static_unmeasured_repos`, and every aggregate that averages over repos reads LOWER
|
|
2226
|
+
until its first snapshot or gate lands. That is an unmeasured repo, never a regression --
|
|
2227
|
+
nothing already measured changed, and `by_repo` proves it entry by entry. Excluding it from
|
|
2228
|
+
the average instead would be the opposite mistake: an aggregate produced by silence."""
|
|
2229
|
+
name = args.add_repo
|
|
2230
|
+
if not _has_text(args.path) or not _has_text(args.type):
|
|
2231
|
+
raise SystemExit("[qa_ledger] init --add-repo needs --path and --type: the engine never "
|
|
2232
|
+
"guesses where a repo lives or how it is built")
|
|
2233
|
+
if name == "integration":
|
|
2234
|
+
raise SystemExit("[qa_ledger] repo name 'integration' is reserved")
|
|
2235
|
+
ledger = _load(args.out)
|
|
2236
|
+
cfg = ledger.get("config") or {}
|
|
2237
|
+
frozen_names = [r.get("name") for r in cfg.get("repos", [])]
|
|
2238
|
+
if name in frozen_names or name in ledger.get("repos", {}):
|
|
2239
|
+
raise SystemExit(
|
|
2240
|
+
"[qa_ledger] repo '%s' already exists in %s -- nothing was written. Adding it twice "
|
|
2241
|
+
"would either duplicate the config entry or reset that repo's steps, which is the "
|
|
2242
|
+
"very loss this flag exists to prevent." % (name, args.out))
|
|
2243
|
+
entry = {"name": name, "type": args.type, "path": args.path}
|
|
2244
|
+
if _has_text(args.test_command):
|
|
2245
|
+
entry["test_command"] = args.test_command
|
|
2246
|
+
cfg.setdefault("repos", []).append(entry)
|
|
2247
|
+
ledger["config"] = cfg
|
|
2248
|
+
# the same node shape `init` writes, so nothing downstream can tell an appended repo from
|
|
2249
|
+
# one that was there since the first run.
|
|
2250
|
+
ledger["repos"][name] = {"type": args.type, "path": args.path,
|
|
2251
|
+
"snapshots": [], "iterations": []}
|
|
2252
|
+
_save(args.out, ledger)
|
|
2253
|
+
cfg_path = args.config or os.path.join(
|
|
2254
|
+
os.path.dirname(os.path.abspath(args.out)), "uscha.config.json")
|
|
2255
|
+
print("[qa_ledger] %s: repo '%s' added (%s at %s) -- %d repos, every existing repo's steps "
|
|
2256
|
+
"untouched, checksum re-sealed"
|
|
2257
|
+
% (args.out, name, args.type, args.path, len(ledger["repos"])))
|
|
2258
|
+
print("[qa_ledger] %s" % _add_repo_to_config_file(cfg_path, entry, frozen_names))
|
|
2259
|
+
print("[qa_ledger] '%s' starts with NO evidence: it reads as UNMEASURED (readiness lists it "
|
|
2260
|
+
"under facts.static_unmeasured_repos) and every repo average reads lower until its "
|
|
2261
|
+
"first snapshot or gate lands. That is an unmeasured repo, not a regression." % name)
|
|
2262
|
+
|
|
2263
|
+
|
|
2110
2264
|
def cmd_init(args):
|
|
2265
|
+
if getattr(args, "add_repo", None):
|
|
2266
|
+
return _init_add_repo(args)
|
|
2267
|
+
if not _has_text(getattr(args, "config", None)):
|
|
2268
|
+
raise SystemExit("[qa_ledger] init needs --config to create a ledger, or --add-repo "
|
|
2269
|
+
"NAME (with --path and --type) to append one repo to an existing one")
|
|
2111
2270
|
cfg = _load(args.config, what="config", flag="--config")
|
|
2112
2271
|
_validate_init_config(cfg)
|
|
2113
2272
|
defaults = cfg.get("defaults", {})
|
|
@@ -2497,7 +2656,16 @@ def _gate_rollup(ledger):
|
|
|
2497
2656
|
for tool, rec in _latest_static_by_tool(rnode).items():
|
|
2498
2657
|
gates.append({"repo": rname, "tool": tool,
|
|
2499
2658
|
"blocking": rec.get("gated_reported", 0) > 0,
|
|
2659
|
+
# ADR-043: an advisory record is non-blocking BY CONSTRUCTION, which
|
|
2660
|
+
# is not the same fact as a gate that ran and came back clean. It
|
|
2661
|
+
# travels so no consumer has to guess which of the two it is looking
|
|
2662
|
+
# at -- the false-clean is the failure mode, not the absence.
|
|
2663
|
+
"advisory": bool(rec.get("advisory")),
|
|
2500
2664
|
"gated": rec.get("gated_reported", 0),
|
|
2665
|
+
# WHERE it was measured (2.2.0, log-gate --ref): a CI verdict without
|
|
2666
|
+
# its run is a claim, and the rollup is where a reader looks first.
|
|
2667
|
+
# Absent on every record that carries none -- never invented.
|
|
2668
|
+
**({"ref": rec["ref"]} if rec.get("ref") else {}),
|
|
2501
2669
|
"note": rec.get("note")})
|
|
2502
2670
|
return sorted(gates, key=lambda g: (g["repo"], g["tool"]))
|
|
2503
2671
|
|
|
@@ -2724,11 +2892,18 @@ def cmd_spec_change_request(args):
|
|
|
2724
2892
|
(row["id"], args.repo, args.source, args.requested_change))
|
|
2725
2893
|
|
|
2726
2894
|
|
|
2727
|
-
def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, note
|
|
2895
|
+
def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, note,
|
|
2896
|
+
advisory=False, ref=None):
|
|
2728
2897
|
"""Append a static-gate-shaped record for a FACT gate so the EXISTING plumbing
|
|
2729
2898
|
sees it: _gate_open_and_sev feeds the BLOCKER/CRITICAL readiness cap (<=65) and
|
|
2730
2899
|
_converged refuses while the latest record for the tool is failing. A later
|
|
2731
|
-
clean record for the same tool clears it (latest-per-tool wins).
|
|
2900
|
+
clean record for the same tool clears it (latest-per-tool wins).
|
|
2901
|
+
|
|
2902
|
+
advisory=True (kit 2.1.0, ADR-043) records a MEASUREMENT that is not a gate: the record
|
|
2903
|
+
carries zero gated findings, so it can neither cap readiness nor block convergence, and it
|
|
2904
|
+
is stamped so no surface can render it as a clean gate either. That distinction is the whole
|
|
2905
|
+
point -- a check running advisory is not the same fact as a check running green, and a
|
|
2906
|
+
ledger that cannot tell them apart is the false-clean this flag exists to refuse."""
|
|
2732
2907
|
ledger["step_counter"] += 1
|
|
2733
2908
|
n = max(1, count) if failing else 0
|
|
2734
2909
|
rec = {
|
|
@@ -2740,24 +2915,70 @@ def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, not
|
|
|
2740
2915
|
"tests_passed": None, "files_changed": 0,
|
|
2741
2916
|
"fingerprint": None, "finding_ids": None, "note": note,
|
|
2742
2917
|
}
|
|
2918
|
+
if advisory:
|
|
2919
|
+
rec["advisory"] = True
|
|
2920
|
+
if ref:
|
|
2921
|
+
# WHERE the verdict was measured (2.2.0). A CI verdict typed by hand is a claim; the
|
|
2922
|
+
# run id or URL beside it is the receipt, and the ledger is the only place it survives
|
|
2923
|
+
# the conversation that produced it.
|
|
2924
|
+
rec["ref"] = ref
|
|
2743
2925
|
node["iterations"].append(rec)
|
|
2744
2926
|
ledger["steps"].append({"n": rec["n"], "at": rec["at"], "kind": "static-gate",
|
|
2745
2927
|
"repo": repo, "tool": tool, "iteration": iteration})
|
|
2746
2928
|
return rec
|
|
2747
2929
|
|
|
2748
2930
|
|
|
2931
|
+
# The only --kind values log-gate accepts with --verdict advisory (ADR-043): the checks whose
|
|
2932
|
+
# default mode IS advisory. Every other kind is a FACT gate and records pass/fail/not-run only.
|
|
2933
|
+
#
|
|
2934
|
+
# `corpus` (2.2.0, ADR-046) is admitted because `corpus-run` ITSELF runs advisory whenever no
|
|
2935
|
+
# threshold is declared -- the percentage is measured against no adopted budget, so it gates
|
|
2936
|
+
# nothing. Refusing the verdict on this door while the check emits it on the other would give
|
|
2937
|
+
# one fact two incompatible records: the parity door could only spell an unbudgeted run as
|
|
2938
|
+
# `pass`, which is the false clean ADR-043 exists to refuse. Admitting it costs nothing that
|
|
2939
|
+
# matters: with a threshold declared, `corpus` is a FACT gate like any other.
|
|
2940
|
+
#
|
|
2941
|
+
# `operability` (ADR-048) joins them because its posture is PROFILE-DEPENDENT: on A, B or no
|
|
2942
|
+
# profile the four checks are measured and recorded advisory, and only `defaults.operability.gate`
|
|
2943
|
+
# (which C, D and E own) turns the same measurement into a gate. It is a FACT either way -- a
|
|
2944
|
+
# workflow file exists or it does not -- so admitting it here widens WHEN it gates, never WHAT
|
|
2945
|
+
# counts as evidence, which is the line INV-ADVISORY-01 draws.
|
|
2946
|
+
ADVISORY_CAPABLE_KINDS = ("simplicity", "waste", "corpus", "operability")
|
|
2947
|
+
|
|
2948
|
+
|
|
2749
2949
|
def cmd_log_gate(args):
|
|
2750
|
-
"""Persist a FACT-gate verdict (golden-diff / gate-check / pit-check / simplicity /
|
|
2751
|
-
into the ledger, so 'facts may block' is enforced by the engine, not by
|
|
2752
|
-
|
|
2753
|
-
|
|
2754
|
-
|
|
2755
|
-
|
|
2950
|
+
"""Persist a FACT-gate verdict (golden-diff / gate-check / pit-check / simplicity /
|
|
2951
|
+
regression / ci / smoke) into the ledger, so 'facts may block' is enforced by the engine, not by
|
|
2952
|
+
goodwill.
|
|
2953
|
+
fail -> BLOCKER record: trips the <=65 readiness cap AND blocks convergence.
|
|
2954
|
+
pass -> clean record for the same tool: credits the fix, convergence sees clean.
|
|
2955
|
+
advisory -> a MEASURED, non-gating record (kit 2.1.0, ADR-043): zero gated findings, so
|
|
2956
|
+
it can never cap readiness nor block convergence, and stamped `advisory` so
|
|
2957
|
+
no surface counts it as an `ok` gate. Use it for a check the project has not
|
|
2958
|
+
declared as a gate -- `simplicity-check` in its default advisory mode above
|
|
2959
|
+
all. The alternative (persisting an advisory as `pass`) is the false clean
|
|
2960
|
+
this verdict exists to refuse: a reader cannot tell "the gate was green"
|
|
2961
|
+
from "there was no gate", and the second is what actually happened.
|
|
2962
|
+
not-run -> a steps event ONLY, never an iterations record: absence is not
|
|
2963
|
+
evidence — it neither reads as clean nor fakes a red (last state stands).
|
|
2964
|
+
|
|
2965
|
+
The engine cannot observe which mode a SEPARATE `simplicity-check` process ran in, so this
|
|
2966
|
+
is a named verdict rather than a refusal: refusing would only be enforceable on trust,
|
|
2967
|
+
while a third verdict is enforceable on the ledger. The caller declares the mode; every
|
|
2968
|
+
reader downstream then sees it as a fact instead of inferring it.
|
|
2756
2969
|
"""
|
|
2757
2970
|
# INV-ADVISORY-01 note (ADR-014): --kind is a CLOSED vocabulary (argparse choices), so
|
|
2758
2971
|
# an advisory-class dimension (e.g. "semantic") cannot be registered as a gate through
|
|
2759
2972
|
# this door at all -- the refusal is structural. The smoke suite measures that the
|
|
2760
2973
|
# vocabulary stays closed; widening it to admit an advisory kind is a red build.
|
|
2974
|
+
#
|
|
2975
|
+
# `ci` and `smoke` (2.2.0) are the additions since ADR-014, and both are FACTS: a pipeline
|
|
2976
|
+
# either went green on a commit or it did not, and the engine can be TOLD that fact with
|
|
2977
|
+
# the run id or URL beside it (--ref); a smoke check either answered or it did not
|
|
2978
|
+
# (ADR-047), which is why neither may run advisory. They are admitted here because they
|
|
2979
|
+
# are measurable, not because they are useful -- an LLM judgment does not become a gate by
|
|
2980
|
+
# being important. Adding a FACT kind is DECLARED in the CONSTITUTION template, which is
|
|
2981
|
+
# where the closed vocabulary is stated to the project rather than only to this parser.
|
|
2761
2982
|
ledger = _load(args.ledger)
|
|
2762
2983
|
node = _repo_node(ledger, args.repo)
|
|
2763
2984
|
tool = f"gate:{args.kind}"
|
|
@@ -2774,12 +2995,575 @@ def cmd_log_gate(args):
|
|
|
2774
2995
|
f"(no evidence — last logged state stands, absence is never green)")
|
|
2775
2996
|
return
|
|
2776
2997
|
failing = args.verdict == "fail"
|
|
2998
|
+
advisory = args.verdict == "advisory"
|
|
2999
|
+
if advisory and args.kind not in ADVISORY_CAPABLE_KINDS:
|
|
3000
|
+
# ADR-043 widens --verdict for the checks that RUN advisory by default. A FACT gate
|
|
3001
|
+
# (deleted tests, a lowered threshold, a golden drift) recorded as advisory would be
|
|
3002
|
+
# a mandatory gate cleared by goodwill -- the exact thing this ledger exists to refuse.
|
|
3003
|
+
print(f"[qa_ledger] log-gate: --verdict advisory is not accepted for --kind {args.kind}: "
|
|
3004
|
+
f"only {', '.join(ADVISORY_CAPABLE_KINDS)} run in an advisory mode; a FACT gate "
|
|
3005
|
+
f"records pass, fail or not-run", file=sys.stderr)
|
|
3006
|
+
sys.exit(2)
|
|
2777
3007
|
rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
|
|
2778
|
-
failing, args.count, args.note
|
|
3008
|
+
failing, args.count, args.note, advisory=advisory,
|
|
3009
|
+
ref=getattr(args, "ref", None))
|
|
2779
3010
|
_save(args.ledger, ledger)
|
|
2780
|
-
|
|
2781
|
-
|
|
2782
|
-
|
|
3011
|
+
if advisory:
|
|
3012
|
+
state, effect = "ADVISORY (measured, not gating)", (
|
|
3013
|
+
"reported everywhere as advisory, never as ok; caps nothing, blocks nothing")
|
|
3014
|
+
elif failing:
|
|
3015
|
+
state, effect = (f"FAIL (BLOCKER x{rec['gated_reported']})",
|
|
3016
|
+
"caps readiness <=65 and blocks convergence")
|
|
3017
|
+
else:
|
|
3018
|
+
state, effect = "PASS (clean)", "clears the gate for convergence"
|
|
3019
|
+
print(f"[qa_ledger] {args.repo}/{tool}: {state} logged — {effect}"
|
|
3020
|
+
+ (f" [ref {rec['ref']}]" if rec.get("ref") else ""))
|
|
3021
|
+
|
|
3022
|
+
|
|
3023
|
+
# --------------------------------------------------------------------------- #
|
|
3024
|
+
# corpus-run (kit 2.2.0, ADR-046): FIELD TRUTH for greenfield work.
|
|
3025
|
+
#
|
|
3026
|
+
# `characterize`/`golden-diff` answer the brownfield question -- "does the new code still do
|
|
3027
|
+
# what the OLD code did?" -- and they have no answer at all for a system that never had an old
|
|
3028
|
+
# code. In a greenfield project every test payload was INVENTED by the agent that wrote the
|
|
3029
|
+
# code, so a suite can be green over inputs the world never produces. The field report this
|
|
3030
|
+
# subcommand comes from is exactly that shape: a parser passed every test its author wrote and
|
|
3031
|
+
# was wrong; running the REAL corpus moved it from 96.96 % to 99.645 %.
|
|
3032
|
+
#
|
|
3033
|
+
# So the corpus is the missing evidence class: real inputs with their real expected outputs,
|
|
3034
|
+
# run through the real command, scored as a percentage the ledger persists like any other FACT.
|
|
3035
|
+
# It is deterministic (stable case order, per-case timeout) and it refuses rather than guessing:
|
|
3036
|
+
# a corpus that is missing, empty or malformed is exit 2 naming the line, never a silent 0 %.
|
|
3037
|
+
#
|
|
3038
|
+
# The 2.1.0 posture holds (ADR-043): a gate needs an ADOPTED budget. With no threshold declared
|
|
3039
|
+
# anywhere the run is ADVISORY -- the percentage is measured and persisted, and it gates nothing.
|
|
3040
|
+
# The WEIGHT of field truth in the readiness score is deliberately NOT here: adding a `field`
|
|
3041
|
+
# dimension moves every existing project's number, and that is its own ADR.
|
|
3042
|
+
# --------------------------------------------------------------------------- #
|
|
3043
|
+
CORPUS_DEFAULT_TIMEOUT = 30
|
|
3044
|
+
CORPUS_DEFAULT_MAX_MISSES = 5
|
|
3045
|
+
|
|
3046
|
+
|
|
3047
|
+
def _corpus_refuse(msg):
|
|
3048
|
+
"""Every corpus refusal is exit 2 and NAMES what it could not read. The failure mode this
|
|
3049
|
+
exists to prevent is a corpus the runner could not parse being scored as 0 % -- an
|
|
3050
|
+
unmeasurable input reported as a measured catastrophe (or, with the arithmetic the other
|
|
3051
|
+
way, as a clean 100 % over zero cases)."""
|
|
3052
|
+
print("[qa_ledger] corpus-run: " + msg, file=sys.stderr)
|
|
3053
|
+
sys.exit(2)
|
|
3054
|
+
|
|
3055
|
+
|
|
3056
|
+
def _corpus_cases(path):
|
|
3057
|
+
"""Read a JSONL corpus into an ORDERED list of cases: file order IS run order, so two runs
|
|
3058
|
+
over the same file report the same misses in the same places. One JSON object per line,
|
|
3059
|
+
`input` and `expected` required, `id` optional (a positional id is derived when absent)."""
|
|
3060
|
+
if not os.path.isfile(path):
|
|
3061
|
+
_corpus_refuse("corpus not found: %s -- a corpus that is not there is UNMEASURED, "
|
|
3062
|
+
"never a measured 0 %%" % path)
|
|
3063
|
+
try:
|
|
3064
|
+
with open(path, encoding="utf-8") as fh:
|
|
3065
|
+
raw = fh.read().splitlines()
|
|
3066
|
+
except OSError as exc:
|
|
3067
|
+
_corpus_refuse("corpus unreadable: %s" % exc)
|
|
3068
|
+
cases = []
|
|
3069
|
+
for n, line in enumerate(raw, 1):
|
|
3070
|
+
if not line.strip():
|
|
3071
|
+
continue
|
|
3072
|
+
try:
|
|
3073
|
+
row = json.loads(line)
|
|
3074
|
+
except ValueError as exc:
|
|
3075
|
+
_corpus_refuse("%s line %d is not valid JSON (%s) -- a malformed corpus is "
|
|
3076
|
+
"refused, never scored" % (path, n, exc))
|
|
3077
|
+
if not isinstance(row, dict):
|
|
3078
|
+
_corpus_refuse("%s line %d is a %s, not a JSON object with `input` and `expected`"
|
|
3079
|
+
% (path, n, type(row).__name__))
|
|
3080
|
+
missing = [k for k in ("input", "expected") if k not in row]
|
|
3081
|
+
if missing:
|
|
3082
|
+
_corpus_refuse("%s line %d has no %s key" % (path, n, " and no ".join(missing)))
|
|
3083
|
+
cases.append({"id": str(row["id"]) if row.get("id") is not None
|
|
3084
|
+
else "case-%03d" % (len(cases) + 1),
|
|
3085
|
+
"input": row["input"], "expected": row["expected"]})
|
|
3086
|
+
if not cases:
|
|
3087
|
+
_corpus_refuse("%s holds 0 cases -- an empty corpus is refused: 0/0 is not 100 %%, "
|
|
3088
|
+
"and it is not 0 %% either" % path)
|
|
3089
|
+
return cases
|
|
3090
|
+
|
|
3091
|
+
|
|
3092
|
+
def _corpus_text(value):
|
|
3093
|
+
"""A JSON scalar string travels as itself; anything else travels as its JSON encoding.
|
|
3094
|
+
sort_keys so an object payload is byte-stable across runs."""
|
|
3095
|
+
return value if isinstance(value, str) else json.dumps(value, ensure_ascii=False,
|
|
3096
|
+
sort_keys=True)
|
|
3097
|
+
|
|
3098
|
+
|
|
3099
|
+
def _corpus_match(actual, expected):
|
|
3100
|
+
"""Trimmed string compare first, then JSON-equal when BOTH sides parse as JSON -- so a
|
|
3101
|
+
command that reorders an object's keys or prints 1.0 for 1 is not a false miss, while a
|
|
3102
|
+
command that prints prose is compared as prose."""
|
|
3103
|
+
exp, act = _corpus_text(expected).strip(), actual.strip()
|
|
3104
|
+
if act == exp:
|
|
3105
|
+
return True
|
|
3106
|
+
try:
|
|
3107
|
+
return json.loads(act) == json.loads(exp)
|
|
3108
|
+
except ValueError:
|
|
3109
|
+
return False
|
|
3110
|
+
|
|
3111
|
+
|
|
3112
|
+
def _corpus_case(command, case, timeout):
|
|
3113
|
+
"""Run ONE case: the input on stdin, the trimmed stdout compared to `expected`.
|
|
3114
|
+
Returns (hit, reason, actual). A non-zero exit is a miss (the command did not answer);
|
|
3115
|
+
a case that outruns the timeout is a miss NAMED `timeout`, never an ambiguous hang."""
|
|
3116
|
+
try:
|
|
3117
|
+
p = subprocess.run(command, shell=True, input=_corpus_text(case["input"]),
|
|
3118
|
+
stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True,
|
|
3119
|
+
encoding="utf-8", errors="replace", timeout=timeout)
|
|
3120
|
+
except subprocess.TimeoutExpired:
|
|
3121
|
+
return False, "timeout", ""
|
|
3122
|
+
except OSError as exc:
|
|
3123
|
+
return False, "command failed to start: %s" % exc, ""
|
|
3124
|
+
actual = (p.stdout or "").strip()
|
|
3125
|
+
if p.returncode != 0:
|
|
3126
|
+
return False, "exit %d" % p.returncode, actual
|
|
3127
|
+
if _corpus_match(actual, case["expected"]):
|
|
3128
|
+
return True, None, actual
|
|
3129
|
+
return False, "mismatch", actual
|
|
3130
|
+
|
|
3131
|
+
|
|
3132
|
+
def _corpus_num(value):
|
|
3133
|
+
"""90.0 reads back as 90 and 99.5 as 99.5 -- the declared budget renders as the human
|
|
3134
|
+
typed it, in the ledger and on every surface that reads the record."""
|
|
3135
|
+
if value is None:
|
|
3136
|
+
return None
|
|
3137
|
+
return int(value) if float(value).is_integer() else float(value)
|
|
3138
|
+
|
|
3139
|
+
|
|
3140
|
+
def _corpus_clip(text, n=120):
|
|
3141
|
+
text = str(text).replace("\n", "\\n")
|
|
3142
|
+
return text if len(text) <= n else text[:n] + "..."
|
|
3143
|
+
|
|
3144
|
+
|
|
3145
|
+
def cmd_corpus_run(args):
|
|
3146
|
+
"""Run a REAL-INPUT corpus against a command and persist the result as `gate:corpus`.
|
|
3147
|
+
|
|
3148
|
+
The threshold is the project's, never the kit's: --threshold, else
|
|
3149
|
+
repos[R].corpus_threshold, else defaults.corpus_threshold. With none of the three the run
|
|
3150
|
+
is ADVISORY -- measured, persisted, and gating nothing (ADR-043: a gate needs an adopted
|
|
3151
|
+
budget). With one, `pass` is a clean gate and `fail` is a BLOCKER through the SAME record
|
|
3152
|
+
shape gate-check uses: readiness capped <=65, convergence blocked, and a later green run
|
|
3153
|
+
clears it (latest-per-tool wins).
|
|
3154
|
+
|
|
3155
|
+
--ac stamps criterion ids on the record. A criterion whose only evidence is a corpus record
|
|
3156
|
+
closes MEASURED iff that record passed -- the same rule a green JUnit testcase already
|
|
3157
|
+
obeys, and for the same reason: red or unbudgeted evidence closes nothing."""
|
|
3158
|
+
ledger = _load(args.ledger)
|
|
3159
|
+
node = _repo_node(ledger, args.repo)
|
|
3160
|
+
tool = "gate:corpus"
|
|
3161
|
+
_validate_iteration(node, tool, args.iteration)
|
|
3162
|
+
cfg = next((r for r in ledger["config"].get("repos", [])
|
|
3163
|
+
if r.get("name") == args.repo), {})
|
|
3164
|
+
defaults = ledger["config"].get("defaults", {})
|
|
3165
|
+
|
|
3166
|
+
corpus_path = args.corpus or cfg.get("corpus")
|
|
3167
|
+
if not corpus_path:
|
|
3168
|
+
_corpus_refuse("no corpus: pass --corpus PATH or declare repos[%s].corpus in the "
|
|
3169
|
+
"config. Where the real inputs live is the project's fact, not a "
|
|
3170
|
+
"default this engine may invent." % args.repo)
|
|
3171
|
+
cases = _corpus_cases(corpus_path)
|
|
3172
|
+
|
|
3173
|
+
threshold, source = args.threshold, "--threshold"
|
|
3174
|
+
if threshold is None:
|
|
3175
|
+
threshold, source = cfg.get("corpus_threshold"), "repos[%s].corpus_threshold" % args.repo
|
|
3176
|
+
if threshold is None:
|
|
3177
|
+
threshold, source = defaults.get("corpus_threshold"), "defaults.corpus_threshold"
|
|
3178
|
+
if threshold is None:
|
|
3179
|
+
source = None
|
|
3180
|
+
threshold = _corpus_num(threshold)
|
|
3181
|
+
|
|
3182
|
+
acs = []
|
|
3183
|
+
for raw in (args.ac or []):
|
|
3184
|
+
cid, _end = _ac_id_of(raw.strip())
|
|
3185
|
+
if cid is None:
|
|
3186
|
+
_corpus_refuse("--ac %r is not a criterion id ('AC-01' or 'AC-FIELD-01')" % raw)
|
|
3187
|
+
if cid not in acs:
|
|
3188
|
+
acs.append(cid)
|
|
3189
|
+
|
|
3190
|
+
hits, misses = 0, []
|
|
3191
|
+
for case in cases:
|
|
3192
|
+
ok, reason, actual = _corpus_case(args.command, case, args.timeout)
|
|
3193
|
+
if ok:
|
|
3194
|
+
hits += 1
|
|
3195
|
+
else:
|
|
3196
|
+
misses.append({"id": case["id"], "reason": reason,
|
|
3197
|
+
"expected": _corpus_clip(_corpus_text(case["expected"])),
|
|
3198
|
+
"actual": _corpus_clip(actual)})
|
|
3199
|
+
total = len(cases)
|
|
3200
|
+
pct = round(100.0 * hits / total, 1)
|
|
3201
|
+
|
|
3202
|
+
advisory = source is None
|
|
3203
|
+
failing = (not advisory) and pct < threshold
|
|
3204
|
+
if advisory:
|
|
3205
|
+
note = "corpus %s: %d/%d (%s %%), no threshold declared" % (
|
|
3206
|
+
os.path.basename(corpus_path), hits, total, pct)
|
|
3207
|
+
else:
|
|
3208
|
+
note = "corpus %s: %d/%d (%s %%) %s %s %%" % (
|
|
3209
|
+
os.path.basename(corpus_path), hits, total, pct,
|
|
3210
|
+
"<" if failing else ">=", threshold)
|
|
3211
|
+
rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
|
|
3212
|
+
failing, len(misses), note, advisory=advisory)
|
|
3213
|
+
# the MEASUREMENT travels on the record, not only its verdict: readiness reads the
|
|
3214
|
+
# percentage back for the field line, and a reader six months later can see WHAT was
|
|
3215
|
+
# measured against WHICH budget instead of a bare pass.
|
|
3216
|
+
rec["corpus"] = {"path": corpus_path.replace("\\", "/"), "hits": hits, "total": total,
|
|
3217
|
+
"percent": pct, "threshold": threshold, "threshold_source": source,
|
|
3218
|
+
"timeout_s": args.timeout,
|
|
3219
|
+
"misses": misses[:max(0, args.max_misses)]}
|
|
3220
|
+
if acs:
|
|
3221
|
+
rec["ac"] = acs
|
|
3222
|
+
_save(args.ledger, ledger)
|
|
3223
|
+
|
|
3224
|
+
state = ("ADVISORY (measured, not gating)" if advisory
|
|
3225
|
+
else "FAIL" if failing else "PASS")
|
|
3226
|
+
out = {"repo": args.repo, "tool": tool, "verdict": state.split()[0].lower(),
|
|
3227
|
+
"advisory": advisory, "corpus": rec["corpus"], "ac": acs, "note": note}
|
|
3228
|
+
if args.json:
|
|
3229
|
+
print(json.dumps(out, indent=2, ensure_ascii=False))
|
|
3230
|
+
else:
|
|
3231
|
+
if advisory:
|
|
3232
|
+
budget = ("no threshold declared — the percentage is recorded and nothing gates "
|
|
3233
|
+
"(a gate needs an adopted budget, ADR-043)")
|
|
3234
|
+
else:
|
|
3235
|
+
budget = "threshold %s %% from %s" % (threshold, source)
|
|
3236
|
+
print("[qa_ledger] %s/%s: %s %s %% (%d/%d) — %s"
|
|
3237
|
+
% (args.repo, tool, state, pct, hits, total, budget))
|
|
3238
|
+
for m in rec["corpus"]["misses"]:
|
|
3239
|
+
print(" miss %s (%s): expected %r, got %r"
|
|
3240
|
+
% (m["id"], m["reason"], m["expected"], m["actual"]))
|
|
3241
|
+
if len(misses) > len(rec["corpus"]["misses"]):
|
|
3242
|
+
print(" ... +%d more miss(es) not listed (--max-misses)"
|
|
3243
|
+
% (len(misses) - len(rec["corpus"]["misses"])))
|
|
3244
|
+
if failing:
|
|
3245
|
+
print(" caps readiness <=65 and blocks convergence until a green run")
|
|
3246
|
+
sys.exit(1 if failing else 0)
|
|
3247
|
+
|
|
3248
|
+
|
|
3249
|
+
def _corpus_records(ledger):
|
|
3250
|
+
"""The LATEST corpus record per repo -- the same latest-per-tool rule the gate rollup and
|
|
3251
|
+
convergence already use, so the three cannot disagree about which run is current. A record
|
|
3252
|
+
logged through the `log-gate --kind corpus` parity door carries no `corpus` block and is
|
|
3253
|
+
deliberately not here: it gates (it is a gate record like any other), but it has no run to
|
|
3254
|
+
read back, and rendering it anyway printed the honest-looking nonsense
|
|
3255
|
+
`corpus None % (None/None) >= None % PASS`. Same exclusion, same reason, as
|
|
3256
|
+
`_smoke_records`."""
|
|
3257
|
+
out = {}
|
|
3258
|
+
for rname, rnode in ledger.get("repos", {}).items():
|
|
3259
|
+
rec = _latest_static_by_tool(rnode).get("gate:corpus")
|
|
3260
|
+
if rec is not None and rec.get("corpus"):
|
|
3261
|
+
out[rname] = rec
|
|
3262
|
+
return out
|
|
3263
|
+
|
|
3264
|
+
|
|
3265
|
+
def _corpus_ac_verdicts(ledger):
|
|
3266
|
+
"""(closed, vetoed) criterion ids from the latest corpus record of every repo (ADR-046).
|
|
3267
|
+
|
|
3268
|
+
A corpus record CLOSES an AC iff it PASSED. An **advisory** run measured a percentage
|
|
3269
|
+
against no adopted budget: it is not a green gate (ADR-043 refuses to let it read as one)
|
|
3270
|
+
and it is not red evidence either, so it neither closes nor vetoes. A **failing** run is
|
|
3271
|
+
evidence AGAINST, which ADR-046 said from the start -- and the engine used to merely skip
|
|
3272
|
+
it, so a green testcase went on closing a criterion the field had just refuted. It now
|
|
3273
|
+
VETOES the ids it carries, the same rule a red JUnit testcase and a failed tagged smoke
|
|
3274
|
+
check already obey, and for the same reason: the cheapest way to fake a closed criterion is
|
|
3275
|
+
to put a green beside a red."""
|
|
3276
|
+
closed, vetoed = set(), set()
|
|
3277
|
+
for rec in _corpus_records(ledger).values():
|
|
3278
|
+
if rec.get("advisory"):
|
|
3279
|
+
continue
|
|
3280
|
+
if (rec.get("gated_reported") or 0) > 0:
|
|
3281
|
+
vetoed.update(rec.get("ac") or [])
|
|
3282
|
+
else:
|
|
3283
|
+
closed.update(rec.get("ac") or [])
|
|
3284
|
+
return closed - vetoed, vetoed
|
|
3285
|
+
|
|
3286
|
+
|
|
3287
|
+
def _corpus_field(ledger):
|
|
3288
|
+
"""Per-repo FIELD readout -- ONE derivation, read by both the readiness text and its JSON.
|
|
3289
|
+
CONDITIONAL like lifecycle and agent-origin: a repo that neither declares a corpus nor ever
|
|
3290
|
+
ran one is absent from it, so a project without field truth prints and emits exactly what
|
|
3291
|
+
it printed and emitted before this existed. No weight, no cap, no gate of its own -- a
|
|
3292
|
+
failing corpus already blocks through its `gate:corpus` record, and the `field` DIMENSION
|
|
3293
|
+
(its own ADR) is deliberately not here."""
|
|
3294
|
+
recs = _corpus_records(ledger)
|
|
3295
|
+
out = {}
|
|
3296
|
+
for rname in ledger.get("repos", {}):
|
|
3297
|
+
cfg = next((r for r in ledger.get("config", {}).get("repos", [])
|
|
3298
|
+
if r.get("name") == rname), {})
|
|
3299
|
+
rec, declared = recs.get(rname), cfg.get("corpus")
|
|
3300
|
+
if rec is None and not declared:
|
|
3301
|
+
continue
|
|
3302
|
+
if rec is None:
|
|
3303
|
+
out[rname] = {"state": "UNMEASURED", "corpus": declared, "percent": None,
|
|
3304
|
+
"hits": None, "total": None, "threshold": None,
|
|
3305
|
+
"reason": "a corpus is declared and was never run"}
|
|
3306
|
+
continue
|
|
3307
|
+
c = rec.get("corpus") or {}
|
|
3308
|
+
out[rname] = {"state": ("ADVISORY" if rec.get("advisory")
|
|
3309
|
+
else "FAIL" if (rec.get("gated_reported") or 0) > 0
|
|
3310
|
+
else "PASS"),
|
|
3311
|
+
"corpus": c.get("path", declared),
|
|
3312
|
+
"percent": c.get("percent"), "hits": c.get("hits"),
|
|
3313
|
+
"total": c.get("total"), "threshold": c.get("threshold"),
|
|
3314
|
+
"threshold_source": c.get("threshold_source"),
|
|
3315
|
+
"ac": rec.get("ac") or []}
|
|
3316
|
+
return out
|
|
3317
|
+
|
|
3318
|
+
|
|
3319
|
+
def _corpus_field_line(rname, f):
|
|
3320
|
+
"""The one-line rendering of a repo's field state. Text lives beside the derivation so the
|
|
3321
|
+
JSON and the human readout can never drift apart."""
|
|
3322
|
+
if f["state"] == "UNMEASURED":
|
|
3323
|
+
return ("--- field %s: corpus UNMEASURED — %s (%s)"
|
|
3324
|
+
% (rname, f["reason"], f["corpus"]))
|
|
3325
|
+
head = "--- field %s: corpus %s %% (%s/%s)" % (rname, f["percent"], f["hits"], f["total"])
|
|
3326
|
+
if f["state"] == "ADVISORY":
|
|
3327
|
+
return head + " — no threshold declared, ADVISORY (measured, not gating)"
|
|
3328
|
+
return "%s %s %s %% %s" % (head, "<" if f["state"] == "FAIL" else ">=",
|
|
3329
|
+
f["threshold"], f["state"])
|
|
3330
|
+
|
|
3331
|
+
|
|
3332
|
+
# --------------------------------------------------------------------------- #
|
|
3333
|
+
# smoke-ingest (kit 2.2.0, ADR-047): the SMOKE RUN as measured evidence.
|
|
3334
|
+
#
|
|
3335
|
+
# The ledger already ingests the evidence a machine produces on its own -- JUnit, coverage,
|
|
3336
|
+
# linters, a static gate's XML, a CI verdict. The smoke list was the hole: "the jar served
|
|
3337
|
+
# /admin", "the simulator answered 200 in 6 ms" arrived as a sub-agent's NARRATION, believed
|
|
3338
|
+
# because it was written confidently. The field report this subcommand comes from is exactly
|
|
3339
|
+
# that shape: every simulator run returned an empty list because the database had no rows, and
|
|
3340
|
+
# a smoke narrated as "verified" would have hidden it behind a sentence.
|
|
3341
|
+
#
|
|
3342
|
+
# So the smoke run stops being prose and becomes a REPORT the project's own tool writes --
|
|
3343
|
+
# {"checks": [{"name", "ok", "status"?, "latency_ms"?, "evidence"?}, ...]} -- and the engine
|
|
3344
|
+
# ingests it like any other fact. `smoke` is a FACT kind and is never advisory: a smoke check is
|
|
3345
|
+
# binary. It either answered or it did not; there is no "measured against no adopted budget"
|
|
3346
|
+
# reading of `ok`, which is why `corpus` may run advisory and this may not.
|
|
3347
|
+
#
|
|
3348
|
+
# Evidence is EXECUTED, not narrated -- and a report the engine cannot read is refused (exit 2)
|
|
3349
|
+
# rather than scored, for the same reason a malformed corpus is: an unreadable smoke reported as
|
|
3350
|
+
# "0 checks ok" would be an unmeasurable run rendered as a measured catastrophe, and an EMPTY
|
|
3351
|
+
# `checks` list read as a clean gate would be the false clean in the other direction.
|
|
3352
|
+
# --------------------------------------------------------------------------- #
|
|
3353
|
+
# How many checks (and failed names) travel on the record. A receipt cites evidence, it is not
|
|
3354
|
+
# a dump -- the same cap `_ac_tags` puts on its testcase receipts, for the same reason.
|
|
3355
|
+
SMOKE_MAX_PERSISTED = 20
|
|
3356
|
+
|
|
3357
|
+
|
|
3358
|
+
def _smoke_refuse(msg):
|
|
3359
|
+
"""Every smoke-ingest refusal is exit 2 and NAMES the offending check or field. The report
|
|
3360
|
+
is written by the PROJECT's own tool, so what this guards against is a contract the project
|
|
3361
|
+
got subtly wrong -- and a wrong contract that scored anyway would be worse than one that
|
|
3362
|
+
refused, because the number would look like a measurement."""
|
|
3363
|
+
print("[qa_ledger] smoke-ingest: " + msg, file=sys.stderr)
|
|
3364
|
+
sys.exit(2)
|
|
3365
|
+
|
|
3366
|
+
|
|
3367
|
+
def _smoke_checks(path):
|
|
3368
|
+
"""Read a smoke report into an ORDERED list of checks. The contract is deliberately the
|
|
3369
|
+
smallest thing a shell script can emit:
|
|
3370
|
+
|
|
3371
|
+
{"checks": [{"name": "...", "ok": true|false,
|
|
3372
|
+
"status": <int|string, optional>,
|
|
3373
|
+
"latency_ms": <number, optional>,
|
|
3374
|
+
"evidence": "<string, optional>"}]}
|
|
3375
|
+
|
|
3376
|
+
`name` and a BOOLEAN `ok` are the whole mandatory surface. `ok: "true"` and `ok: 1` are
|
|
3377
|
+
refused: a string and a number are not verdicts, and a contract that coerces is a contract
|
|
3378
|
+
that cannot say what it measured. The optional fields are validated when PRESENT and never
|
|
3379
|
+
invented when absent -- a check with no `latency_ms` measured no latency, which is a
|
|
3380
|
+
different fact from a latency of 0."""
|
|
3381
|
+
if not os.path.isfile(path):
|
|
3382
|
+
_smoke_refuse("report not found: %s -- a smoke run that left no report is UNMEASURED, "
|
|
3383
|
+
"never a clean gate" % path)
|
|
3384
|
+
try:
|
|
3385
|
+
with open(path, encoding="utf-8") as fh:
|
|
3386
|
+
doc = json.load(fh)
|
|
3387
|
+
except OSError as exc:
|
|
3388
|
+
_smoke_refuse("report unreadable: %s" % exc)
|
|
3389
|
+
except ValueError as exc:
|
|
3390
|
+
_smoke_refuse("%s is not valid JSON (%s) -- a malformed report is refused, never "
|
|
3391
|
+
"scored" % (path, exc))
|
|
3392
|
+
if not isinstance(doc, dict):
|
|
3393
|
+
_smoke_refuse("%s is a %s, not a JSON object carrying a `checks` list"
|
|
3394
|
+
% (path, type(doc).__name__))
|
|
3395
|
+
if "checks" not in doc:
|
|
3396
|
+
_smoke_refuse('%s has no `checks` key -- the contract is {"checks": [{"name": ..., '
|
|
3397
|
+
'"ok": true|false}, ...]}' % path)
|
|
3398
|
+
raw = doc["checks"]
|
|
3399
|
+
if not isinstance(raw, list):
|
|
3400
|
+
_smoke_refuse("%s: `checks` is a %s, not a list" % (path, type(raw).__name__))
|
|
3401
|
+
if not raw:
|
|
3402
|
+
_smoke_refuse("%s holds 0 checks -- an EMPTY smoke is not evidence: it is a run that "
|
|
3403
|
+
"verified nothing, and it does not read as a clean gate" % path)
|
|
3404
|
+
checks = []
|
|
3405
|
+
for i, row in enumerate(raw):
|
|
3406
|
+
where = "check %d" % (i + 1)
|
|
3407
|
+
if not isinstance(row, dict):
|
|
3408
|
+
_smoke_refuse("%s: %s is a %s, not an object" % (path, where, type(row).__name__))
|
|
3409
|
+
name = row.get("name")
|
|
3410
|
+
if not isinstance(name, str) or not name.strip():
|
|
3411
|
+
_smoke_refuse("%s: %s has no `name` -- a check nobody can name is a check nobody "
|
|
3412
|
+
"can act on" % (path, where))
|
|
3413
|
+
where = "%s (%s)" % (where, name.strip())
|
|
3414
|
+
if not isinstance(row.get("ok"), bool):
|
|
3415
|
+
_smoke_refuse("%s: %s has no boolean `ok` (got %r) -- a smoke check is a binary "
|
|
3416
|
+
"fact, and a string, a number or a missing key is not a verdict"
|
|
3417
|
+
% (path, where, row.get("ok")))
|
|
3418
|
+
status = row.get("status")
|
|
3419
|
+
if status is not None and (isinstance(status, bool)
|
|
3420
|
+
or not isinstance(status, (int, str))):
|
|
3421
|
+
_smoke_refuse("%s: %s has a `status` that is neither an integer nor a string (%r)"
|
|
3422
|
+
% (path, where, status))
|
|
3423
|
+
lat = row.get("latency_ms")
|
|
3424
|
+
if lat is not None and (isinstance(lat, bool) or not isinstance(lat, (int, float))):
|
|
3425
|
+
_smoke_refuse("%s: %s has a `latency_ms` that is not a number (%r)"
|
|
3426
|
+
% (path, where, lat))
|
|
3427
|
+
ev = row.get("evidence")
|
|
3428
|
+
if ev is not None and not isinstance(ev, str):
|
|
3429
|
+
_smoke_refuse("%s: %s has an `evidence` that is not a string (%r)"
|
|
3430
|
+
% (path, where, ev))
|
|
3431
|
+
checks.append({"name": name.strip(), "ok": row["ok"], "status": status,
|
|
3432
|
+
"latency_ms": lat, "evidence": ev})
|
|
3433
|
+
return checks
|
|
3434
|
+
|
|
3435
|
+
|
|
3436
|
+
def _smoke_ac_of(checks, want_ok):
|
|
3437
|
+
"""Criterion ids tagged on check NAMES, in the SAME grammar a JUnit testcase name uses
|
|
3438
|
+
(`_ac_tag_ids`, ADR-036): a check named "AC-28 the admin page serves" tags AC-28. One
|
|
3439
|
+
extractor, so a name cannot close a criterion in the suite and fail to close it here."""
|
|
3440
|
+
ids = []
|
|
3441
|
+
for c in checks:
|
|
3442
|
+
if bool(c["ok"]) != want_ok:
|
|
3443
|
+
continue
|
|
3444
|
+
for cid in _ac_tag_ids(c["name"]):
|
|
3445
|
+
if cid not in ids:
|
|
3446
|
+
ids.append(cid)
|
|
3447
|
+
return sorted(ids, key=_top_ac_key)
|
|
3448
|
+
|
|
3449
|
+
|
|
3450
|
+
def cmd_smoke_ingest(args):
|
|
3451
|
+
"""Ingest a smoke report as `gate:smoke` -- the smoke run MEASURED instead of narrated.
|
|
3452
|
+
|
|
3453
|
+
A failed check is a BLOCKER through the SAME record shape `gate-check` and `corpus-run`
|
|
3454
|
+
write: readiness capped <=65, convergence blocked, and a later clean report clears it
|
|
3455
|
+
(latest-per-tool wins). `smoke` is a FACT kind and is NOT advisory-capable: `ok` is binary,
|
|
3456
|
+
so there is no budget it could be measured against and no honest advisory reading of it.
|
|
3457
|
+
|
|
3458
|
+
A check whose NAME carries an AC tag closes that criterion MEASURED iff the check is `ok`
|
|
3459
|
+
AND the report as a whole passed -- and a FAILED tagged check is red evidence that vetoes,
|
|
3460
|
+
exactly like a red JUnit testcase. Fail-closed in both directions: a green check inside a
|
|
3461
|
+
failing smoke closes nothing, because the run it belongs to did not hold."""
|
|
3462
|
+
ledger = _load(args.ledger)
|
|
3463
|
+
node = _repo_node(ledger, args.repo)
|
|
3464
|
+
tool = "gate:smoke"
|
|
3465
|
+
_validate_iteration(node, tool, args.iteration)
|
|
3466
|
+
checks = _smoke_checks(args.report)
|
|
3467
|
+
|
|
3468
|
+
n_ok = sum(1 for c in checks if c["ok"])
|
|
3469
|
+
failed = [c for c in checks if not c["ok"]]
|
|
3470
|
+
failing = bool(failed)
|
|
3471
|
+
failed_names = [c["name"] for c in failed]
|
|
3472
|
+
note = "smoke %s: %d/%d checks ok%s" % (
|
|
3473
|
+
os.path.basename(args.report), n_ok, len(checks),
|
|
3474
|
+
(", %d failed (%s)" % (len(failed), ", ".join(failed_names[:SMOKE_MAX_PERSISTED])))
|
|
3475
|
+
if failed else "")
|
|
3476
|
+
rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
|
|
3477
|
+
failing, len(failed), note)
|
|
3478
|
+
# the MEASUREMENT travels on the record, not only its verdict: a reader six months later
|
|
3479
|
+
# sees WHICH checks ran, what each answered and how long it took -- the facts the narration
|
|
3480
|
+
# used to carry and lose.
|
|
3481
|
+
rec["smoke"] = {
|
|
3482
|
+
"report": args.report.replace("\\", "/"), "ok": n_ok, "failed": len(failed),
|
|
3483
|
+
"checks": [{"name": c["name"], "ok": c["ok"], "status": c["status"],
|
|
3484
|
+
"latency_ms": c["latency_ms"]} for c in checks[:SMOKE_MAX_PERSISTED]],
|
|
3485
|
+
"failed_names": failed_names[:SMOKE_MAX_PERSISTED],
|
|
3486
|
+
# computed over EVERY check, never over the truncated receipt above: the cap is a
|
|
3487
|
+
# display budget, and a criterion's fate must not depend on where the list was cut.
|
|
3488
|
+
"ac": _smoke_ac_of(checks, True),
|
|
3489
|
+
"ac_red": _smoke_ac_of(checks, False),
|
|
3490
|
+
}
|
|
3491
|
+
_save(args.ledger, ledger)
|
|
3492
|
+
|
|
3493
|
+
out = {"repo": args.repo, "tool": tool, "verdict": "fail" if failing else "pass",
|
|
3494
|
+
"smoke": rec["smoke"], "note": note}
|
|
3495
|
+
if args.json:
|
|
3496
|
+
print(json.dumps(out, indent=2, ensure_ascii=False))
|
|
3497
|
+
else:
|
|
3498
|
+
print("[qa_ledger] %s/%s: %s — %d/%d checks ok"
|
|
3499
|
+
% (args.repo, tool, "FAIL" if failing else "PASS", n_ok, len(checks)))
|
|
3500
|
+
for c in checks[:SMOKE_MAX_PERSISTED]:
|
|
3501
|
+
bits = [b for b in ("status %s" % c["status"] if c["status"] is not None else None,
|
|
3502
|
+
"%s ms" % c["latency_ms"] if c["latency_ms"] is not None
|
|
3503
|
+
else None) if b]
|
|
3504
|
+
print(" %s %s%s" % ("ok " if c["ok"] else "FAIL", c["name"],
|
|
3505
|
+
(" (%s)" % ", ".join(bits)) if bits else ""))
|
|
3506
|
+
if len(checks) > SMOKE_MAX_PERSISTED:
|
|
3507
|
+
print(" ... +%d more check(s) not listed or persisted"
|
|
3508
|
+
% (len(checks) - SMOKE_MAX_PERSISTED))
|
|
3509
|
+
if failing:
|
|
3510
|
+
print(" caps readiness <=65 and blocks convergence until a clean smoke")
|
|
3511
|
+
sys.exit(1 if failing else 0)
|
|
3512
|
+
|
|
3513
|
+
|
|
3514
|
+
def _smoke_records(ledger):
|
|
3515
|
+
"""The LATEST smoke record per repo -- the same latest-per-tool rule the gate rollup and
|
|
3516
|
+
convergence already use, so the three cannot disagree about which run is current. A record
|
|
3517
|
+
logged through the `log-gate` parity door carries no `smoke` block and is deliberately not
|
|
3518
|
+
here: it gates (it is a gate record like any other), but it has no checks to read back."""
|
|
3519
|
+
out = {}
|
|
3520
|
+
for rname, rnode in ledger.get("repos", {}).items():
|
|
3521
|
+
rec = _latest_static_by_tool(rnode).get("gate:smoke")
|
|
3522
|
+
if rec is not None and rec.get("smoke"):
|
|
3523
|
+
out[rname] = rec
|
|
3524
|
+
return out
|
|
3525
|
+
|
|
3526
|
+
|
|
3527
|
+
def _smoke_ac_verdicts(ledger):
|
|
3528
|
+
"""(closed, vetoed) criterion ids from the latest smoke record of every repo (ADR-047).
|
|
3529
|
+
|
|
3530
|
+
A tagged check closes MEASURED iff it is `ok` AND its report's gate PASSED: a green check
|
|
3531
|
+
inside a failing smoke is a green light on a run that did not hold, and the ledger refuses
|
|
3532
|
+
to read it as one. A FAILED tagged check is red evidence and vetoes wherever it appears --
|
|
3533
|
+
the same rule as a red JUnit testcase, and it outranks every green."""
|
|
3534
|
+
closed, vetoed = set(), set()
|
|
3535
|
+
for rec in _smoke_records(ledger).values():
|
|
3536
|
+
s = rec.get("smoke") or {}
|
|
3537
|
+
vetoed.update(s.get("ac_red") or [])
|
|
3538
|
+
if not (rec.get("gated_reported") or 0):
|
|
3539
|
+
closed.update(s.get("ac") or [])
|
|
3540
|
+
return closed - vetoed, vetoed
|
|
3541
|
+
|
|
3542
|
+
|
|
3543
|
+
def _smoke_report(ledger):
|
|
3544
|
+
"""Per-repo SMOKE readout -- ONE derivation, read by both the readiness text and its JSON.
|
|
3545
|
+
CONDITIONAL like lifecycle, agent-origin and field: a repo that never ingested a smoke
|
|
3546
|
+
report is absent from it, so a project that never ran one prints and emits exactly what it
|
|
3547
|
+
printed and emitted before this existed."""
|
|
3548
|
+
out = {}
|
|
3549
|
+
for rname, rec in _smoke_records(ledger).items():
|
|
3550
|
+
s = rec["smoke"]
|
|
3551
|
+
out[rname] = {"state": "FAIL" if (rec.get("gated_reported") or 0) else "PASS",
|
|
3552
|
+
"report": s.get("report"), "ok": s.get("ok"),
|
|
3553
|
+
"failed": s.get("failed"),
|
|
3554
|
+
"total": (s.get("ok") or 0) + (s.get("failed") or 0),
|
|
3555
|
+
"failed_names": s.get("failed_names") or [],
|
|
3556
|
+
"ac": s.get("ac") or [], "ac_red": s.get("ac_red") or []}
|
|
3557
|
+
return out
|
|
3558
|
+
|
|
3559
|
+
|
|
3560
|
+
def _smoke_line(rname, s):
|
|
3561
|
+
"""The one-line rendering of a repo's smoke state. Text lives beside the derivation so the
|
|
3562
|
+
JSON and the human readout can never drift apart."""
|
|
3563
|
+
head = "--- smoke %s: %d/%d checks ok" % (rname, s["ok"], s["total"])
|
|
3564
|
+
if s["failed"]:
|
|
3565
|
+
head += ", %d failed (%s)" % (s["failed"], ", ".join(s["failed_names"]))
|
|
3566
|
+
return "%s %s" % (head, s["state"])
|
|
2783
3567
|
|
|
2784
3568
|
|
|
2785
3569
|
def cmd_flag_blocker(args):
|
|
@@ -2966,6 +3750,16 @@ def _derive_phase(ledger, name, node, k, qa_order):
|
|
|
2966
3750
|
if len(_dl["uncurated"]) > 3 else "")
|
|
2967
3751
|
+ " -- INV-CURATION-01: sin juicio no hay promocion")
|
|
2968
3752
|
conv = False
|
|
3753
|
+
# operability gate (ADR-048). NAMING only: a failing record is already a BLOCKER through
|
|
3754
|
+
# _gate_open_and_sev and already vetoes convergence through _converged, so nothing new is
|
|
3755
|
+
# gated here. What is added is WHICH check is missing -- "static-gate gated=1
|
|
3756
|
+
# (gate:operability:1)" tells a human the gate is red without telling them whether to write
|
|
3757
|
+
# a workflow or a RUNBOOK, and phase --require pr-ready is where that answer is needed.
|
|
3758
|
+
_op = _latest_static_by_tool(node).get("gate:operability")
|
|
3759
|
+
if _op and (_op.get("gated_reported") or 0) > 0:
|
|
3760
|
+
reasons.append("operability: %s -- release, reset and the RUNBOOK are part of done, "
|
|
3761
|
+
"not of the last week (ADR-048)"
|
|
3762
|
+
% (_op.get("note") or "checks missing"))
|
|
2969
3763
|
_cr = _cr_cfg(ledger)
|
|
2970
3764
|
if _cr and _cr.get("mode") == "final":
|
|
2971
3765
|
_head = None
|
|
@@ -3455,15 +4249,35 @@ def cmd_spec_drift(args):
|
|
|
3455
4249
|
lag_days = int(args.max_lag_days if args.max_lag_days is not None
|
|
3456
4250
|
else cfg.get("max_lag_days", 30))
|
|
3457
4251
|
repo_path = _scope_path(ledger, args.repo)
|
|
4252
|
+
root_path = os.path.dirname(os.path.abspath(args.ledger)) or "."
|
|
3458
4253
|
|
|
3459
4254
|
# The spec surface is fixed by ADR-005: the repo SPEC.md plus every ADR.
|
|
3460
|
-
|
|
3461
|
-
|
|
3462
|
-
|
|
3463
|
-
|
|
3464
|
-
|
|
3465
|
-
|
|
3466
|
-
|
|
4255
|
+
def _specs_at(base):
|
|
4256
|
+
found = []
|
|
4257
|
+
if os.path.isfile(os.path.join(base, "SPEC.md")):
|
|
4258
|
+
found.append("SPEC.md")
|
|
4259
|
+
adr_dir = os.path.join(base, "docs", "adr")
|
|
4260
|
+
if os.path.isdir(adr_dir):
|
|
4261
|
+
found += sorted("docs/adr/" + f for f in os.listdir(adr_dir)
|
|
4262
|
+
if f.lower().endswith(".md"))
|
|
4263
|
+
return found
|
|
4264
|
+
|
|
4265
|
+
# 2.2.0 field fix: a MONOREPO keeps ONE SPEC.md and one docs/adr/ at the root, next to
|
|
4266
|
+
# uscha.config.json, while repos[R].path points at a subdirectory -- and this command read
|
|
4267
|
+
# only that subdirectory, so `spec-drift --repo backend-api` answered "no spec documents"
|
|
4268
|
+
# about a project whose spec was one level up. A spec found at the root is the MONOREPO's
|
|
4269
|
+
# spec and governs every repo. The repo's own path still WINS when it has one (a repo that
|
|
4270
|
+
# carries its own SPEC is describing itself); the config root -- taken as the ledger's
|
|
4271
|
+
# directory, which is where `init --config uscha.config.json` is run -- is the fallback.
|
|
4272
|
+
# The answer NAMES which of the two it read: "no drift" and "read the wrong tree" produced
|
|
4273
|
+
# the same silence, and that is what made the field report take a week to notice.
|
|
4274
|
+
base_path, spec_source = repo_path, "repo"
|
|
4275
|
+
spec_files = _specs_at(repo_path)
|
|
4276
|
+
if not spec_files and os.path.realpath(root_path) != os.path.realpath(repo_path):
|
|
4277
|
+
root_specs = _specs_at(root_path)
|
|
4278
|
+
if root_specs:
|
|
4279
|
+
base_path, spec_source, spec_files = root_path, "root", root_specs
|
|
4280
|
+
repo_path = base_path
|
|
3467
4281
|
|
|
3468
4282
|
tracked = []
|
|
3469
4283
|
ls = subprocess.run(["git", "ls-files"], cwd=repo_path, capture_output=True,
|
|
@@ -3535,20 +4349,26 @@ def cmd_spec_drift(args):
|
|
|
3535
4349
|
results.append(row)
|
|
3536
4350
|
|
|
3537
4351
|
out = {"repo": args.repo, "max_lag_days": lag_days, "results": results,
|
|
3538
|
-
"advisory": True
|
|
4352
|
+
"advisory": True, "spec_source": spec_source if spec_files else None,
|
|
4353
|
+
"spec_base": repo_path}
|
|
3539
4354
|
|
|
3540
4355
|
# Latest-state record so the mirador can surface an advisory row. Advisory data,
|
|
3541
4356
|
# not a step in the loop: no step_counter, no gate record, no readiness input.
|
|
3542
4357
|
ledger["spec_drift"] = {"repo": args.repo, "at": _now(), "max_lag_days": lag_days,
|
|
3543
|
-
"results": results
|
|
4358
|
+
"results": results,
|
|
4359
|
+
"spec_source": spec_source if spec_files else None}
|
|
3544
4360
|
_save(args.ledger, ledger)
|
|
3545
4361
|
|
|
3546
4362
|
if args.json:
|
|
3547
4363
|
print(json.dumps(out, indent=2, ensure_ascii=False))
|
|
3548
4364
|
else:
|
|
3549
4365
|
print("SPEC-DRIFT %s (advisory, lag > %dd):" % (args.repo, lag_days))
|
|
4366
|
+
if spec_source == "root" and spec_files:
|
|
4367
|
+
print(" specs read from the CONFIG ROOT (%s): the monorepo's SPEC governs "
|
|
4368
|
+
"every repo" % repo_path)
|
|
3550
4369
|
if not results:
|
|
3551
|
-
print(" no spec documents found (SPEC.md / docs/adr/*.md)"
|
|
4370
|
+
print(" no spec documents found (SPEC.md / docs/adr/*.md) in the repo path "
|
|
4371
|
+
"nor at the config root")
|
|
3552
4372
|
mark = {"SPEC_STALE": "!!", "CLEAN": "ok", "UNMAPPED": "--", "UNTRACKED": "--",
|
|
3553
4373
|
"NO-CODE": "ok"}
|
|
3554
4374
|
for r_ in results:
|
|
@@ -3563,6 +4383,330 @@ def cmd_spec_drift(args):
|
|
|
3563
4383
|
sys.exit(0)
|
|
3564
4384
|
|
|
3565
4385
|
|
|
4386
|
+
# --------------------------------------------------------------------------- #
|
|
4387
|
+
# operability (ADR-048: release, reset and the RUNBOOK are MEASURED, not narrated)
|
|
4388
|
+
# --------------------------------------------------------------------------- #
|
|
4389
|
+
# The field finding, twice in a row: release-by-CI, the reset/seed script and the RUNBOOK
|
|
4390
|
+
# arrived in the last week of two projects. The devloop NAMED them in phase 8 prose and
|
|
4391
|
+
# nothing measured them, so "we'll do it at the end" survived every gate the kit has -- the
|
|
4392
|
+
# same shape as every other narrated dimension this engine has replaced with a fact.
|
|
4393
|
+
#
|
|
4394
|
+
# What this command reads is FILES IN THE TREE, never prose and never a claim: a workflow that
|
|
4395
|
+
# runs the repo's own test command, a workflow that publishes something, a RUNBOOK with the
|
|
4396
|
+
# four headings an operator needs at 3am, and a seed/reset command the config declares whose
|
|
4397
|
+
# script is actually on disk. It cannot read whether the RUNBOOK is GOOD -- that is a human
|
|
4398
|
+
# judgment and it is not pretended here. It can read that the file exists and that the
|
|
4399
|
+
# sections are named, which is the difference between a project that thought about rollback
|
|
4400
|
+
# and one that has not yet.
|
|
4401
|
+
#
|
|
4402
|
+
# It never executes anything (ADR-008: the engine is not an executor of config-supplied
|
|
4403
|
+
# shell) and its own exit code is always 0. Whether the verdict GATES is the project's
|
|
4404
|
+
# declaration -- `defaults.operability.gate`, owned by risk profiles C, D and E.
|
|
4405
|
+
OPERABILITY_CHECKS = ("ci", "release", "runbook", "seed")
|
|
4406
|
+
|
|
4407
|
+
# GitHub Actions is the only CI this reads. Another system is NAMED as unknown rather than
|
|
4408
|
+
# failed: the engine cannot open a pipeline it does not understand, and a red invented for a
|
|
4409
|
+
# pipeline nobody read would be exactly the manufactured verdict the kit refuses elsewhere.
|
|
4410
|
+
OPERABILITY_CI_DIR = (".github", "workflows")
|
|
4411
|
+
OPERABILITY_OTHER_CI = (
|
|
4412
|
+
".gitlab-ci.yml", ".travis.yml", "azure-pipelines.yml", "bitbucket-pipelines.yml",
|
|
4413
|
+
"Jenkinsfile", ".circleci", ".drone.yml", ".teamcity",
|
|
4414
|
+
)
|
|
4415
|
+
# The publish recognisers, listed rather than guessed: a workflow step that creates or
|
|
4416
|
+
# attaches a release asset. Short and documented on purpose -- a regex over "release" would
|
|
4417
|
+
# match a branch name, a job title and a comment, and a gate that matches prose is a gate that
|
|
4418
|
+
# certifies prose.
|
|
4419
|
+
OPERABILITY_RELEASE_MARKERS = (
|
|
4420
|
+
"softprops/action-gh-release",
|
|
4421
|
+
"actions/upload-release-asset",
|
|
4422
|
+
"gh release create",
|
|
4423
|
+
"gh release upload",
|
|
4424
|
+
"gh release",
|
|
4425
|
+
"npm publish",
|
|
4426
|
+
"twine upload",
|
|
4427
|
+
)
|
|
4428
|
+
# A test-ish subcommand, for the case where the workflow does not repeat the configured
|
|
4429
|
+
# command verbatim (a Makefile target, an extra flag, a matrix variable in the middle).
|
|
4430
|
+
OPERABILITY_TEST_WORDS = ("test", "tests", "pytest", "nextest", "jest", "check", "verify")
|
|
4431
|
+
# The four headings an operator needs, matched case-insensitively in EN and ES. What is
|
|
4432
|
+
# matched is the HEADING, not the body: the engine can see that rollback was thought about,
|
|
4433
|
+
# never that the procedure is correct.
|
|
4434
|
+
OPERABILITY_RUNBOOK_SECTIONS = (
|
|
4435
|
+
("start", r"arranque|start|boot"),
|
|
4436
|
+
("config", r"config"),
|
|
4437
|
+
("rollback", r"rollback|reversi"),
|
|
4438
|
+
("smoke", r"smoke|humo"),
|
|
4439
|
+
)
|
|
4440
|
+
OPERABILITY_RUNBOOK_PATHS = ("docs/RUNBOOK.md", "RUNBOOK.md")
|
|
4441
|
+
OPERABILITY_SCRIPT_EXT = (".sh", ".py", ".ps1", ".bat", ".sql", ".js", ".ts", ".rb")
|
|
4442
|
+
|
|
4443
|
+
|
|
4444
|
+
def _op_bases(ledger, repo, ledger_path):
|
|
4445
|
+
"""(repo path, config root) with realpath on BOTH sides -- the Windows 8.3 lesson, and the
|
|
4446
|
+
monorepo lesson spec-drift paid for in 2.2.0: the workflows and the RUNBOOK of a monorepo
|
|
4447
|
+
live at the config root while repos[R].path points at a subdirectory. The repo's own tree
|
|
4448
|
+
WINS when it has the artifact; the root is the fallback, and the answer NAMES which one it
|
|
4449
|
+
read, because "no CI" and "read the wrong tree" produced the same silence."""
|
|
4450
|
+
repo_path = _scope_path(ledger, repo)
|
|
4451
|
+
root_path = os.path.dirname(os.path.abspath(ledger_path)) or "."
|
|
4452
|
+
bases = [(repo_path, "repo")]
|
|
4453
|
+
if os.path.realpath(root_path) != os.path.realpath(repo_path):
|
|
4454
|
+
bases.append((root_path, "root"))
|
|
4455
|
+
return bases
|
|
4456
|
+
|
|
4457
|
+
|
|
4458
|
+
def _op_workflows(bases):
|
|
4459
|
+
"""(list of (relative name, text), base, where) for the first base that HAS a GitHub
|
|
4460
|
+
Actions directory, else ([], None, None)."""
|
|
4461
|
+
for base, where in bases:
|
|
4462
|
+
wdir = os.path.join(base, *OPERABILITY_CI_DIR)
|
|
4463
|
+
if not os.path.isdir(wdir):
|
|
4464
|
+
continue
|
|
4465
|
+
files = []
|
|
4466
|
+
for name in sorted(os.listdir(wdir)):
|
|
4467
|
+
if not name.lower().endswith((".yml", ".yaml")):
|
|
4468
|
+
continue
|
|
4469
|
+
try:
|
|
4470
|
+
with open(os.path.join(wdir, name), encoding="utf-8",
|
|
4471
|
+
errors="replace") as fh:
|
|
4472
|
+
files.append((name, fh.read()))
|
|
4473
|
+
except OSError:
|
|
4474
|
+
continue
|
|
4475
|
+
if files:
|
|
4476
|
+
return files, base, where
|
|
4477
|
+
return [], None, None
|
|
4478
|
+
|
|
4479
|
+
|
|
4480
|
+
def _op_other_ci(bases):
|
|
4481
|
+
"""The first non-Actions CI marker found, as `name (where)`, or None."""
|
|
4482
|
+
for base, where in bases:
|
|
4483
|
+
for name in OPERABILITY_OTHER_CI:
|
|
4484
|
+
if os.path.exists(os.path.join(base, name)):
|
|
4485
|
+
return "%s (%s)" % (name, where)
|
|
4486
|
+
return None
|
|
4487
|
+
|
|
4488
|
+
|
|
4489
|
+
def _op_test_command(ledger, repo):
|
|
4490
|
+
"""The repo's CONFIGURED test command: repos[R].test_command, else the per-type
|
|
4491
|
+
defaults.test_command_<type>. Read, never run."""
|
|
4492
|
+
cfg = ledger.get("config") or {}
|
|
4493
|
+
defaults = cfg.get("defaults") or {}
|
|
4494
|
+
entry = {}
|
|
4495
|
+
for r in cfg.get("repos", []):
|
|
4496
|
+
if r.get("name") == repo:
|
|
4497
|
+
entry = r
|
|
4498
|
+
break
|
|
4499
|
+
if _has_text(entry.get("test_command")):
|
|
4500
|
+
return entry["test_command"].strip()
|
|
4501
|
+
rtype = entry.get("type") or (ledger.get("repos", {}).get(repo) or {}).get("type")
|
|
4502
|
+
value = defaults.get("test_command_%s" % rtype) if rtype else None
|
|
4503
|
+
return value.strip() if _has_text(value) else None
|
|
4504
|
+
|
|
4505
|
+
|
|
4506
|
+
def _op_ci(workflows, command):
|
|
4507
|
+
"""(status, detail). `ok` when a workflow step runs the configured command -- verbatim, or
|
|
4508
|
+
its first token beside a test-ish subcommand on the same line. The detail SAYS what
|
|
4509
|
+
matched, so the human can disagree with the match instead of with a boolean."""
|
|
4510
|
+
if not command:
|
|
4511
|
+
return "missing", "no test command configured for this repo (nothing to look for)"
|
|
4512
|
+
head = command.split()[0]
|
|
4513
|
+
tail = os.path.basename(head)
|
|
4514
|
+
for name, body in workflows:
|
|
4515
|
+
for line in body.splitlines():
|
|
4516
|
+
stripped = line.strip()
|
|
4517
|
+
if command in line:
|
|
4518
|
+
return "ok", '%s runs "%s"' % (name, command)
|
|
4519
|
+
if head not in line and tail not in line:
|
|
4520
|
+
continue
|
|
4521
|
+
words = re.split(r"[^A-Za-z0-9_.-]+", stripped.lower())
|
|
4522
|
+
if any(w in OPERABILITY_TEST_WORDS for w in words):
|
|
4523
|
+
shown = re.sub(r"^-\s*", "", stripped)
|
|
4524
|
+
shown = re.sub(r"^run:\s*", "", shown)
|
|
4525
|
+
return "ok", '%s runs "%s"' % (name, shown[:70])
|
|
4526
|
+
return "missing", ("no workflow step runs the configured test command (%s)"
|
|
4527
|
+
% command)
|
|
4528
|
+
|
|
4529
|
+
|
|
4530
|
+
def _op_release(workflows):
|
|
4531
|
+
for name, body in workflows:
|
|
4532
|
+
for marker in OPERABILITY_RELEASE_MARKERS:
|
|
4533
|
+
if marker in body:
|
|
4534
|
+
return "ok", '%s runs "%s"' % (name, marker)
|
|
4535
|
+
return "missing", ("no workflow publishes or attaches a release asset (looked for: %s)"
|
|
4536
|
+
% ", ".join(OPERABILITY_RELEASE_MARKERS))
|
|
4537
|
+
|
|
4538
|
+
|
|
4539
|
+
def _op_runbook(bases, declared):
|
|
4540
|
+
"""(status, detail, path). `declared` is defaults.operability.runbook when the project set
|
|
4541
|
+
one; otherwise docs/RUNBOOK.md then RUNBOOK.md, repo tree before config root."""
|
|
4542
|
+
candidates = [declared] if _has_text(declared) else list(OPERABILITY_RUNBOOK_PATHS)
|
|
4543
|
+
for base, where in bases:
|
|
4544
|
+
for rel in candidates:
|
|
4545
|
+
path = os.path.join(base, rel.replace("/", os.sep))
|
|
4546
|
+
if not os.path.isfile(path):
|
|
4547
|
+
continue
|
|
4548
|
+
try:
|
|
4549
|
+
with open(path, encoding="utf-8", errors="replace") as fh:
|
|
4550
|
+
body = fh.read()
|
|
4551
|
+
except OSError as exc:
|
|
4552
|
+
return "missing", "%s unreadable: %s" % (rel, exc), rel
|
|
4553
|
+
heads = [row.lstrip("#").strip().lower()
|
|
4554
|
+
for row in body.splitlines() if row.lstrip().startswith("#")]
|
|
4555
|
+
blob = "\n".join(heads)
|
|
4556
|
+
absent = [label for label, pattern in OPERABILITY_RUNBOOK_SECTIONS
|
|
4557
|
+
if not re.search(pattern, blob, re.I)]
|
|
4558
|
+
if absent:
|
|
4559
|
+
return ("missing", "missing sections (%s) in %s [%s]"
|
|
4560
|
+
% (", ".join(absent), rel, where), rel)
|
|
4561
|
+
return "ok", "%s [%s], all four sections named" % (rel, where), rel
|
|
4562
|
+
return ("missing", "no RUNBOOK found (looked for %s)" % ", ".join(candidates), None)
|
|
4563
|
+
|
|
4564
|
+
|
|
4565
|
+
def _op_seed(ledger, repo, bases, defaults_op):
|
|
4566
|
+
"""(status, detail). The seed/reset command is DECLARED, never sniffed: an engine that
|
|
4567
|
+
guessed which script resets the database would be guessing about the most destructive
|
|
4568
|
+
command in the project."""
|
|
4569
|
+
entry = {}
|
|
4570
|
+
for r in (ledger.get("config") or {}).get("repos", []):
|
|
4571
|
+
if r.get("name") == repo:
|
|
4572
|
+
entry = r
|
|
4573
|
+
break
|
|
4574
|
+
repo_op = entry.get("operability") if isinstance(entry.get("operability"), dict) else {}
|
|
4575
|
+
command = repo_op.get("seed_command")
|
|
4576
|
+
if not _has_text(command):
|
|
4577
|
+
command = defaults_op.get("seed_command")
|
|
4578
|
+
if not _has_text(command):
|
|
4579
|
+
return "missing", ("no seed/reset command declared "
|
|
4580
|
+
"(repos[R].operability.seed_command or "
|
|
4581
|
+
"defaults.operability.seed_command)")
|
|
4582
|
+
command = command.strip()
|
|
4583
|
+
script = None
|
|
4584
|
+
for token in re.split(r"\s+", command):
|
|
4585
|
+
bare = token.strip("\"'")
|
|
4586
|
+
if "/" in bare or "\\" in bare or bare.lower().endswith(OPERABILITY_SCRIPT_EXT):
|
|
4587
|
+
script = bare
|
|
4588
|
+
break
|
|
4589
|
+
if script is None:
|
|
4590
|
+
# A command with no path in it (`make seed`, `npm run reset`) names a target this
|
|
4591
|
+
# engine cannot resolve without running something. Declared is what is measurable.
|
|
4592
|
+
return "ok", 'declared: "%s" (no script path to verify)' % command
|
|
4593
|
+
for base, _where in bases:
|
|
4594
|
+
if os.path.exists(os.path.join(base, script.replace("/", os.sep))):
|
|
4595
|
+
return "ok", 'declared: "%s"' % command
|
|
4596
|
+
return "missing", "script not found: %s" % script
|
|
4597
|
+
|
|
4598
|
+
|
|
4599
|
+
def _operability_report(ledger, repo, ledger_path):
|
|
4600
|
+
"""The four FACTS, with the base each was read from. Pure measurement: no persistence,
|
|
4601
|
+
no exit code, no gate decision -- the caller owns all three."""
|
|
4602
|
+
defaults = (ledger.get("config") or {}).get("defaults") or {}
|
|
4603
|
+
op_cfg = defaults.get("operability") if isinstance(defaults.get("operability"), dict) else {}
|
|
4604
|
+
bases = _op_bases(ledger, repo, ledger_path)
|
|
4605
|
+
workflows, wf_base, wf_where = _op_workflows(bases)
|
|
4606
|
+
checks = {}
|
|
4607
|
+
if workflows:
|
|
4608
|
+
ci_status, ci_detail = _op_ci(workflows, _op_test_command(ledger, repo))
|
|
4609
|
+
rel_status, rel_detail = _op_release(workflows)
|
|
4610
|
+
ci_source = "%s [%s]" % ("/".join(OPERABILITY_CI_DIR), wf_where)
|
|
4611
|
+
else:
|
|
4612
|
+
other = _op_other_ci(bases)
|
|
4613
|
+
ci_source = None
|
|
4614
|
+
if other:
|
|
4615
|
+
ci_status = rel_status = "unknown"
|
|
4616
|
+
ci_detail = rel_detail = ("unknown ci system: %s -- this reads GitHub Actions "
|
|
4617
|
+
"only, so it neither certifies nor condemns it" % other)
|
|
4618
|
+
else:
|
|
4619
|
+
ci_status = rel_status = "missing"
|
|
4620
|
+
ci_detail = rel_detail = ("no %s in the repo tree nor at the config root"
|
|
4621
|
+
% "/".join(OPERABILITY_CI_DIR))
|
|
4622
|
+
checks["ci"] = {"status": ci_status, "detail": ci_detail}
|
|
4623
|
+
checks["release"] = {"status": rel_status, "detail": rel_detail}
|
|
4624
|
+
rb_status, rb_detail, rb_path = _op_runbook(bases, op_cfg.get("runbook"))
|
|
4625
|
+
checks["runbook"] = {"status": rb_status, "detail": rb_detail, "path": rb_path}
|
|
4626
|
+
sd_status, sd_detail = _op_seed(ledger, repo, bases, op_cfg)
|
|
4627
|
+
checks["seed"] = {"status": sd_status, "detail": sd_detail}
|
|
4628
|
+
return {"repo": repo, "checks": checks, "ci_source": ci_source,
|
|
4629
|
+
"bases": [{"path": b, "where": w} for b, w in bases]}
|
|
4630
|
+
|
|
4631
|
+
|
|
4632
|
+
def _operability_verdict(report, gate):
|
|
4633
|
+
"""(verdict, summary note). Three states, because there are three facts:
|
|
4634
|
+
- every check `ok` -> pass (a gate that ran clean)
|
|
4635
|
+
- any check `missing` -> fail under a declared gate, advisory without
|
|
4636
|
+
- none missing, some `unknown` -> advisory even under a declared gate
|
|
4637
|
+
The third is the one worth spelling out: a pipeline this engine cannot read is UNMEASURED,
|
|
4638
|
+
and recording UNMEASURED as `pass` would hand a declared gate a green nobody measured --
|
|
4639
|
+
the false clean ADR-043 exists to refuse, arriving through a different door."""
|
|
4640
|
+
statuses = [report["checks"][k]["status"] for k in OPERABILITY_CHECKS]
|
|
4641
|
+
note = " · ".join("%s %s" % (k, report["checks"][k]["status"])
|
|
4642
|
+
for k in OPERABILITY_CHECKS)
|
|
4643
|
+
if not gate:
|
|
4644
|
+
return "advisory", note
|
|
4645
|
+
if "missing" in statuses:
|
|
4646
|
+
return "fail", note
|
|
4647
|
+
if "unknown" in statuses:
|
|
4648
|
+
return "advisory", note
|
|
4649
|
+
return "pass", note
|
|
4650
|
+
|
|
4651
|
+
|
|
4652
|
+
def cmd_operability(args):
|
|
4653
|
+
"""Measure the operability of a repo: CI, release, RUNBOOK, seed (ADR-048).
|
|
4654
|
+
|
|
4655
|
+
Exit code is ALWAYS 0 for the check itself -- the gate decision belongs to the profile,
|
|
4656
|
+
and it is enforced where every other FACT gate is enforced: the persisted record caps
|
|
4657
|
+
readiness <=65 and blocks convergence when it is a `fail`."""
|
|
4658
|
+
ledger = _load(args.ledger)
|
|
4659
|
+
node = _repo_node(ledger, args.repo)
|
|
4660
|
+
resolved, origin = _resolved_defaults(ledger.get("config") or {})
|
|
4661
|
+
gate = bool(resolved.get("operability.gate"))
|
|
4662
|
+
report = _operability_report(ledger, args.repo, args.ledger)
|
|
4663
|
+
verdict, note = _operability_verdict(report, gate)
|
|
4664
|
+
missing = [k for k in OPERABILITY_CHECKS
|
|
4665
|
+
if report["checks"][k]["status"] == "missing"]
|
|
4666
|
+
report.update({"gate": gate, "gate_origin": origin.get("operability.gate"),
|
|
4667
|
+
"verdict": verdict, "note": note, "missing": missing})
|
|
4668
|
+
|
|
4669
|
+
rec = _append_gate_record(ledger, node, args.repo, "gate:operability", args.iteration,
|
|
4670
|
+
verdict == "fail", len(missing) or 1, note,
|
|
4671
|
+
advisory=(verdict == "advisory"))
|
|
4672
|
+
_save(args.ledger, ledger)
|
|
4673
|
+
report["step"] = rec["n"]
|
|
4674
|
+
|
|
4675
|
+
if args.json:
|
|
4676
|
+
print(json.dumps(report, indent=2, ensure_ascii=False))
|
|
4677
|
+
sys.exit(0)
|
|
4678
|
+
# "not declared" and "declared false" are DIFFERENT facts: the first is a project that
|
|
4679
|
+
# never mentioned the knob, the second a human who turned the gate off on purpose. Saying
|
|
4680
|
+
# "not declared" for both erased the decision (fresh review, 2.2.0).
|
|
4681
|
+
gate_origin = origin.get("operability.gate")
|
|
4682
|
+
gate_shown = ("declared" if gate
|
|
4683
|
+
else "declared false" if gate_origin == "override" else "not declared")
|
|
4684
|
+
print("OPERABILITY %s (gate: %s, origin %s)" % (args.repo, gate_shown, gate_origin))
|
|
4685
|
+
if report["ci_source"]:
|
|
4686
|
+
print(" read: %s" % report["ci_source"])
|
|
4687
|
+
for key in OPERABILITY_CHECKS:
|
|
4688
|
+
c = report["checks"][key]
|
|
4689
|
+
mark = {"ok": "ok", "missing": "!!", "unknown": "--"}.get(c["status"], "??")
|
|
4690
|
+
detail = c["detail"]
|
|
4691
|
+
print(" %s %s: %s%s" % (mark, key, c["status"],
|
|
4692
|
+
(" (%s)" % detail) if detail else ""))
|
|
4693
|
+
present = sum(1 for k in OPERABILITY_CHECKS
|
|
4694
|
+
if report["checks"][k]["status"] == "ok")
|
|
4695
|
+
if verdict == "fail":
|
|
4696
|
+
print(" -- %d of %d present -- FAIL: caps readiness <=65 and blocks convergence "
|
|
4697
|
+
"until %s exist(s)" % (present, len(OPERABILITY_CHECKS), ", ".join(missing)))
|
|
4698
|
+
elif verdict == "pass":
|
|
4699
|
+
print(" -- %d of %d present -- PASS: the declared operability gate ran clean"
|
|
4700
|
+
% (present, len(OPERABILITY_CHECKS)))
|
|
4701
|
+
else:
|
|
4702
|
+
print(" -- %d of %d present -- ADVISORY (measured, not gating): caps nothing, "
|
|
4703
|
+
"blocks nothing%s" % (present, len(OPERABILITY_CHECKS),
|
|
4704
|
+
("; declare defaults.operability.gate: true (or a risk "
|
|
4705
|
+
"profile C/D/E) to make it a gate" if not gate else
|
|
4706
|
+
"; a check this engine cannot read is UNMEASURED, "
|
|
4707
|
+
"never a green")))
|
|
4708
|
+
sys.exit(0)
|
|
4709
|
+
|
|
3566
4710
|
|
|
3567
4711
|
# --------------------------------------------------------------------------- #
|
|
3568
4712
|
# golden-coverage (ADR-006: the golden<->source mapping, DERIVED BY MEASUREMENT)
|
|
@@ -7319,27 +8463,147 @@ def _render_lang_md(r, has_js=False):
|
|
|
7319
8463
|
# --------------------------------------------------------------------------- #
|
|
7320
8464
|
|
|
7321
8465
|
FACTS_FILE = "SYSTEM-FACTS.json"
|
|
8466
|
+
# The NAMESPACED name `install-uscha.py` writes the kit's version under, beside the installed
|
|
8467
|
+
# skills (2.2.0). It is not `VERSION`: that bare name is shared with whatever else the agent's
|
|
8468
|
+
# skills directory holds, and an installer that writes it owns a file it did not create.
|
|
8469
|
+
KIT_VERSION_COPY = ".uscha-kit-VERSION"
|
|
8470
|
+
|
|
8471
|
+
|
|
8472
|
+
def _kit_root():
|
|
8473
|
+
"""The kit directory this engine belongs to, or None.
|
|
8474
|
+
|
|
8475
|
+
By MARKER, not by fixed depth: the canonical engine sits 4 levels deep
|
|
8476
|
+
(.claude/skills/uscha-devloop/) and the Codex twin 3 (skills/uscha-devloop/). A fixed
|
|
8477
|
+
dirname walk made the twin silently derive the OUTER repo root -- version None,
|
|
8478
|
+
0 skills, no error (fresh-review HIGH, reproduced by running both copies).
|
|
8479
|
+
|
|
8480
|
+
The marker is a VERSION file AND a skills tree beside it (2.2.0). VERSION alone stopped
|
|
8481
|
+
being sufficient the moment `install-uscha.py` began dropping one beside the installed
|
|
8482
|
+
skills so `doctor` could date them: that root carries a version and no kit, and answering
|
|
8483
|
+
it here would make `facts` derive `0 skills` from a directory that never held any --
|
|
8484
|
+
a manufactured fact, which is worse than the honest refusal this returns instead.
|
|
8485
|
+
An engine that only needs the VERSION asks _engine_kit_version().
|
|
8486
|
+
|
|
8487
|
+
The walk starts at the engine's REALPATH: a `--mode link` install puts a link to the kit's
|
|
8488
|
+
skill directory under the agent's skills root, and an abspath walk-up from there climbs the
|
|
8489
|
+
agent's tree instead of the checkout the link points into (fresh review, 2.2.0). Same lesson
|
|
8490
|
+
the Windows 8.3 gotcha teaches: resolve before you compare, resolve before you walk."""
|
|
8491
|
+
cur = os.path.dirname(os.path.realpath(__file__))
|
|
8492
|
+
for _ in range(6):
|
|
8493
|
+
if (os.path.isfile(os.path.join(cur, "VERSION"))
|
|
8494
|
+
and (os.path.isdir(os.path.join(cur, ".claude", "skills"))
|
|
8495
|
+
or os.path.isdir(os.path.join(cur, "skills")))):
|
|
8496
|
+
return cur
|
|
8497
|
+
nxt = os.path.dirname(cur)
|
|
8498
|
+
if nxt == cur:
|
|
8499
|
+
break
|
|
8500
|
+
cur = nxt
|
|
8501
|
+
return None
|
|
7322
8502
|
|
|
7323
8503
|
|
|
7324
|
-
def
|
|
7325
|
-
"""
|
|
7326
|
-
|
|
7327
|
-
skill
|
|
7328
|
-
|
|
7329
|
-
|
|
7330
|
-
|
|
7331
|
-
|
|
7332
|
-
|
|
7333
|
-
|
|
7334
|
-
|
|
8504
|
+
def _engine_kit_version():
|
|
8505
|
+
"""(version, where) for the kit THIS engine came from, or (None, [dirs it looked in]).
|
|
8506
|
+
|
|
8507
|
+
Which kit an installed skill came from is a different question from which kit tree this
|
|
8508
|
+
engine sits in, and an INSTALLED engine has no kit tree at all: `install-uscha.py` copies
|
|
8509
|
+
the kit's version beside the installed skills precisely so the question stays answerable
|
|
8510
|
+
there -- from a checkout this walk lands on the kit root like _kit_root() does, and from an
|
|
8511
|
+
install it lands on the install root.
|
|
8512
|
+
|
|
8513
|
+
THREE sources, in this order at every level (2.2.0, fresh review):
|
|
8514
|
+
|
|
8515
|
+
1. `.uscha-kit-VERSION` -- the NAMESPACED copy the installer writes. The copy used to be
|
|
8516
|
+
called `VERSION`, a bare shared name dropped into directories the kit does not own
|
|
8517
|
+
(`~/.claude/skills/`, `~/.cursor/skills/`): whatever else lived under that name was
|
|
8518
|
+
overwritten by an install and deleted by an uninstall. The kit owns its prefix and
|
|
8519
|
+
nothing else.
|
|
8520
|
+
2. `VERSION` -- the KIT ROOT's own file, which is the checkout case and is never written by
|
|
8521
|
+
an install any more. A foreign file under that name is read, not written: reading is what
|
|
8522
|
+
this walk is for, and a wrong version is reported as a version, never as damage.
|
|
8523
|
+
3. `uscha-install.json`'s `version` -- the install marker, already on the walk, so an
|
|
8524
|
+
install whose copy was removed by hand still answers instead of going UNMEASURED.
|
|
8525
|
+
|
|
8526
|
+
The walk starts at the engine's REALPATH: under `--mode link` the installed skill directory
|
|
8527
|
+
is a link into the kit checkout, and an abspath walk-up climbs the agent's tree rather than
|
|
8528
|
+
the kit's -- so a link install could not resolve its own kit without the copy.
|
|
8529
|
+
|
|
8530
|
+
Never guesses: with no version anywhere it returns the directories it READ, so the report
|
|
8531
|
+
can say where it looked instead of only that it failed."""
|
|
8532
|
+
cur = os.path.dirname(os.path.realpath(__file__))
|
|
8533
|
+
looked = []
|
|
7335
8534
|
for _ in range(6):
|
|
7336
|
-
|
|
7337
|
-
|
|
7338
|
-
|
|
8535
|
+
looked.append(cur)
|
|
8536
|
+
for name in (KIT_VERSION_COPY, "VERSION"):
|
|
8537
|
+
try:
|
|
8538
|
+
with open(os.path.join(cur, name), encoding="utf-8") as fh:
|
|
8539
|
+
return fh.read().strip().split()[-1], cur
|
|
8540
|
+
except (OSError, IndexError):
|
|
8541
|
+
pass
|
|
8542
|
+
try:
|
|
8543
|
+
with open(os.path.join(cur, "uscha-install.json"), encoding="utf-8") as fh:
|
|
8544
|
+
declared = (json.load(fh) or {}).get("version")
|
|
8545
|
+
if isinstance(declared, str) and declared.strip():
|
|
8546
|
+
return declared.strip().split()[-1], cur
|
|
8547
|
+
except (OSError, ValueError, AttributeError):
|
|
8548
|
+
pass
|
|
7339
8549
|
nxt = os.path.dirname(cur)
|
|
7340
8550
|
if nxt == cur:
|
|
7341
8551
|
break
|
|
7342
8552
|
cur = nxt
|
|
8553
|
+
return None, looked
|
|
8554
|
+
|
|
8555
|
+
|
|
8556
|
+
BENCH_DOC = "DIAMOND-BENCH.md"
|
|
8557
|
+
# One generated table row per archetype: `| crud-store | PASS | M1 12/12, ... |`. Anchored at the
|
|
8558
|
+
# line start and on the closing pipe so the per-entry prose below the table ("### guard -- PARTIAL")
|
|
8559
|
+
# cannot be counted twice.
|
|
8560
|
+
_BENCH_ROW = re.compile(r"^\|\s*([A-Za-z0-9][\w.-]*)\s*\|\s*(PASS|PARTIAL|FAIL|PENDING)\s*\|")
|
|
8561
|
+
|
|
8562
|
+
|
|
8563
|
+
def _derive_bench(kit):
|
|
8564
|
+
"""The Diamond Bench headline, COUNTED out of the bench's own generated report, or None.
|
|
8565
|
+
|
|
8566
|
+
`DIAMOND-BENCH.md` is written by `qa_ledger.py bench` over the committed fixture and carries
|
|
8567
|
+
the "do not hand-edit" banner: every row in it is a measured run. Re-running the bench here
|
|
8568
|
+
would be the honest derivation and is not affordable -- a full pass is ~650 child processes,
|
|
8569
|
+
and `facts` runs on every suite, every deploy and twice per release. So the fact is counted
|
|
8570
|
+
from the RECORDED verdicts, per archetype, never from the summary sentence beside them and
|
|
8571
|
+
never from a number typed into a document.
|
|
8572
|
+
|
|
8573
|
+
The report lives at the REPO root, one level above the kit: an installed kit has no bench,
|
|
8574
|
+
which is why this returns None instead of guessing. `--check` then reports any claim about it
|
|
8575
|
+
as UNMEASURED rather than letting it pass unexamined."""
|
|
8576
|
+
if not kit:
|
|
8577
|
+
return None
|
|
8578
|
+
for cand in (os.path.join(os.path.dirname(kit), BENCH_DOC),
|
|
8579
|
+
os.path.join(kit, BENCH_DOC)):
|
|
8580
|
+
if not os.path.isfile(cand):
|
|
8581
|
+
continue
|
|
8582
|
+
try:
|
|
8583
|
+
with open(cand, encoding="utf-8-sig") as fh:
|
|
8584
|
+
body = fh.read()
|
|
8585
|
+
except OSError:
|
|
8586
|
+
continue
|
|
8587
|
+
verdicts = []
|
|
8588
|
+
for line in body.split("\n"):
|
|
8589
|
+
m = _BENCH_ROW.match(line.strip())
|
|
8590
|
+
if m:
|
|
8591
|
+
verdicts.append(m.group(2))
|
|
8592
|
+
if not verdicts:
|
|
8593
|
+
continue
|
|
8594
|
+
return {"entries": len(verdicts), "pass": verdicts.count("PASS"),
|
|
8595
|
+
"partial": verdicts.count("PARTIAL"), "fail": verdicts.count("FAIL"),
|
|
8596
|
+
"pending": verdicts.count("PENDING")}
|
|
8597
|
+
return None
|
|
8598
|
+
|
|
8599
|
+
|
|
8600
|
+
def _derive_facts():
|
|
8601
|
+
"""Facts derived from the ARTIFACTS themselves, never from prose and never from greps
|
|
8602
|
+
over documentation: the subcommand list comes from introspecting the REAL parser, the
|
|
8603
|
+
skill list from the REAL kit tree, the version from the kit VERSION file, the Diamond
|
|
8604
|
+
headline from the bench's own generated report. No timestamp on purpose: regeneration
|
|
8605
|
+
over an unchanged repo must be byte-identical (AC-SF-01)."""
|
|
8606
|
+
kit = _kit_root()
|
|
7343
8607
|
if kit is None:
|
|
7344
8608
|
print("[qa_ledger] facts: no VERSION file found walking up from the engine -- "
|
|
7345
8609
|
"facts that cannot locate their own kit are not facts.", file=sys.stderr)
|
|
@@ -7356,14 +8620,24 @@ def _derive_facts():
|
|
|
7356
8620
|
skills = sorted(d for d in os.listdir(sdir)
|
|
7357
8621
|
if os.path.isfile(os.path.join(sdir, d, "SKILL.md")))
|
|
7358
8622
|
break
|
|
8623
|
+
bench = _derive_bench(kit)
|
|
7359
8624
|
return {
|
|
7360
8625
|
"version": version,
|
|
7361
8626
|
"subcommands": {"count": len(subs), "list": subs},
|
|
7362
8627
|
"skills": {"count": len(skills), "list": skills},
|
|
8628
|
+
# null, never a zero: an installed kit ships no bench report, and "0 PASS" would be a
|
|
8629
|
+
# measured-looking answer to a question this tree cannot answer.
|
|
8630
|
+
"diamond": bench,
|
|
7363
8631
|
"_derivation": {
|
|
7364
8632
|
"version": "uscha-kit/VERSION",
|
|
7365
8633
|
"subcommands": "argparse introspection of build_parser()",
|
|
7366
8634
|
"skills": "SKILL.md inventory under uscha-kit/.claude/skills/",
|
|
8635
|
+
"diamond": ("verdict rows of DIAMOND-BENCH.md, the report `bench` generates over "
|
|
8636
|
+
"uscha-kit/tests/fixtures/diamond-bench (regenerate it with: qa_ledger.py "
|
|
8637
|
+
"bench --dir uscha-kit/tests/fixtures/diamond-bench --out "
|
|
8638
|
+
"DIAMOND-BENCH.md)" if bench else
|
|
8639
|
+
"UNMEASURED: no DIAMOND-BENCH.md beside the kit -- claims about the "
|
|
8640
|
+
"bench headline are reported as undecidable, never as green"),
|
|
7367
8641
|
"omitted": "stack matrix and REAL/VISION registry: no mechanical "
|
|
7368
8642
|
"source exists yet -- omitted, not guessed (ADR-012)",
|
|
7369
8643
|
},
|
|
@@ -7393,7 +8667,18 @@ _SPELLED = dict((_spell(n), n) for n in range(1, 100))
|
|
|
7393
8667
|
# longest alternative first: an alternation offering "six" before "sixty-three" matches the prefix
|
|
7394
8668
|
_NUM_ALT = "|".join(sorted(_SPELLED, key=len, reverse=True))
|
|
7395
8669
|
# the leading \b so that "someone skills" cannot be read as the claim "one skills"
|
|
7396
|
-
|
|
8670
|
+
_NUM = r"\b(\d+|" + _NUM_ALT + r")"
|
|
8671
|
+
# HTML splits a claim across elements: the homepage's stat tile reads
|
|
8672
|
+
# `<div class="v">8<small>/12</small></div><div class="k">archetypes regenerate</div>`, so the
|
|
8673
|
+
# count and the noun that gives it meaning are separated by markup rather than by a space. The
|
|
8674
|
+
# gap is therefore whitespace OR tags; `[^<>]` stops one tag from swallowing the rest of the
|
|
8675
|
+
# line, and the repetition is bounded so the gap can never run from one sentence into the next.
|
|
8676
|
+
_GAP = r"(?:\s|<[^<>]{0,80}>){0,8}"
|
|
8677
|
+
_ARCH = r"(?:archetypes?|arquetipos?)"
|
|
8678
|
+
# a claimed count that is NOT the one being rewritten (the other half of a verdict pair)
|
|
8679
|
+
_ANYNUM = r"(?:\d+|" + _NUM_ALT + r")"
|
|
8680
|
+
# what separates `8 PASS` from `4 PARTIAL`: a comma, a middle dot, a slash, spaces, markup
|
|
8681
|
+
_VSEP = r"[^A-Za-z0-9<>]{0,8}" + _GAP
|
|
7397
8682
|
|
|
7398
8683
|
|
|
7399
8684
|
def _unspell(token):
|
|
@@ -7405,17 +8690,57 @@ _CLAIM_PATTERNS = (
|
|
|
7405
8690
|
# (fact key path, regex over one line, needs-context substring or None)
|
|
7406
8691
|
("version", r"v(\d+\.\d+\.\d+)", "kit"),
|
|
7407
8692
|
("version", r"uscha-kit\s+v?(\d+\.\d+\.\d+)", None),
|
|
7408
|
-
|
|
7409
|
-
("
|
|
8693
|
+
# `_GAP` rather than a plain space since 2.2.0: the site's stat tiles put the count in one
|
|
8694
|
+
# element and its noun in the next (`<div class="v">52</div><div class="k">engine
|
|
8695
|
+
# subcommands</div>`), so a gate that demanded whitespace between them read the homepage's
|
|
8696
|
+
# headline numbers as prose. It had `52 engine subcommands` and `9/12 archetypes` on one
|
|
8697
|
+
# screen, both stale, both invisible to a green release.
|
|
8698
|
+
("subcommands.count", _NUM + _GAP + r"(?:engine\s+)?sub-?comm?ands", None),
|
|
8699
|
+
("subcommands.count", r"\b(\d+)" + _GAP + r"subcomandos", None),
|
|
7410
8700
|
# "agent skills" is the kit's own noun phrase and the paper's; nothing wider is let in,
|
|
7411
8701
|
# because a WRITER that guessed at "two other skills" would corrupt the sentence it fixed.
|
|
7412
|
-
("skills.count",
|
|
8702
|
+
("skills.count", _NUM + _GAP + r"(?:agent\s+)?skills", None),
|
|
8703
|
+
# The Diamond Bench headline (2.2.0). The homepage said 9/12 for nine releases after ADR-042
|
|
8704
|
+
# moved `transformer` to PARTIAL, because no gate could see the claim: the number sat in one
|
|
8705
|
+
# HTML element and its noun in the next.
|
|
8706
|
+
#
|
|
8707
|
+
# Two shapes only, and both are narrow BY MEASUREMENT -- each wider draft was tried against
|
|
8708
|
+
# the gated set first and rejected by what it caught:
|
|
8709
|
+
#
|
|
8710
|
+
# `<n>/12 archetypes`, `<n> of 12 archetypes`, `<n> de 12 arquetipos` -- the count must be
|
|
8711
|
+
# followed, across markup but never across prose, by the noun it counts, and the
|
|
8712
|
+
# denominator must be the DIGITS 12. That is what keeps the repo's own historical sentence
|
|
8713
|
+
# ("It was 9 of 12 until ADR-042 moved transformer") and the paper's "ten of twelve
|
|
8714
|
+
# archetypes" out of a writer that would have silently rewritten both.
|
|
8715
|
+
#
|
|
8716
|
+
# `<n> PASS <sep> <n> PARTIAL` as ONE shape, verdicts case-sensitive (`(?-i:...)` against
|
|
8717
|
+
# the module-wide re.I). Reading the two numbers independently caught the paper's snapshot
|
|
8718
|
+
# of the July-August arm -- "nine archetypes PASS ... three PARTIAL", a sentence about a
|
|
8719
|
+
# different experiment -- and would have offered to rewrite it. The bench headline always
|
|
8720
|
+
# writes the pair together, so adjacency is both narrower and closer to the real claim.
|
|
8721
|
+
#
|
|
8722
|
+
# `<n> archetypes` alone is deliberately NOT a claim: across the gated set it names subsets
|
|
8723
|
+
# far more often than the bench ("five archetypes", "two archetypes have no second run"), so
|
|
8724
|
+
# the entries count stays a derived fact with no recognised published shape.
|
|
8725
|
+
("diamond.pass", _NUM + _GAP + r"(?:/|of|de)" + _GAP + r"12" + _GAP + _ARCH, None),
|
|
8726
|
+
("diamond.pass", _NUM + _GAP + r"(?-i:PASS)" + _VSEP + _ANYNUM + _GAP + r"(?-i:PARTIAL)",
|
|
8727
|
+
None),
|
|
8728
|
+
("diamond.partial",
|
|
8729
|
+
r"\b" + _ANYNUM + _GAP + r"(?-i:PASS)" + _VSEP + _NUM + _GAP + r"(?-i:PARTIAL)", None),
|
|
7413
8730
|
)
|
|
7414
8731
|
|
|
7415
8732
|
|
|
7416
8733
|
def _fact_value(facts, dotted):
|
|
8734
|
+
"""The derived fact behind a claim key, or None when THIS tree cannot derive it.
|
|
8735
|
+
|
|
8736
|
+
Not every fact exists everywhere: an installed kit has no `DIAMOND-BENCH.md`, so
|
|
8737
|
+
`diamond.*` is null there. None is the honest answer and both consumers act on it --
|
|
8738
|
+
`--write` leaves the claim alone (there is nothing to write it to) and `--check` reports it
|
|
8739
|
+
as UNMEASURED. Neither treats an underivable fact as agreement."""
|
|
7417
8740
|
cur = facts
|
|
7418
8741
|
for part in dotted.split("."):
|
|
8742
|
+
if not isinstance(cur, dict) or cur.get(part) is None:
|
|
8743
|
+
return None
|
|
7419
8744
|
cur = cur[part]
|
|
7420
8745
|
return str(cur)
|
|
7421
8746
|
|
|
@@ -7506,7 +8831,10 @@ def _write_claims(facts, paths):
|
|
|
7506
8831
|
# right to left: an earlier rewrite must not move the offsets of a later one
|
|
7507
8832
|
for key, start, end, token in sorted(claims, key=lambda c: c[1], reverse=True):
|
|
7508
8833
|
actual = _fact_value(facts, key)
|
|
7509
|
-
if _claim_norm(token) == actual:
|
|
8834
|
+
if actual is None or _claim_norm(token) == actual:
|
|
8835
|
+
# None = this tree cannot derive the fact; there is nothing to rewrite the
|
|
8836
|
+
# claim TO, and inventing one would be the opposite of the gate. The --check
|
|
8837
|
+
# that follows reports it as UNMEASURED.
|
|
7510
8838
|
continue
|
|
7511
8839
|
line = line[:start] + _claim_rewrite(token, actual) + line[end:]
|
|
7512
8840
|
n += 1
|
|
@@ -7579,7 +8907,11 @@ def cmd_facts(args):
|
|
|
7579
8907
|
# comment spans -- a comment is not a published claim
|
|
7580
8908
|
for key, _s, _e, claimed in _iter_claims(line):
|
|
7581
8909
|
actual = _fact_value(facts, key)
|
|
7582
|
-
if
|
|
8910
|
+
if actual is None:
|
|
8911
|
+
problems.append((path, n, key, claimed,
|
|
8912
|
+
"UNMEASURED -- this tree derives no such fact "
|
|
8913
|
+
"(see SYSTEM-FACTS _derivation)"))
|
|
8914
|
+
elif _claim_norm(claimed) != actual:
|
|
7583
8915
|
problems.append((path, n, key, claimed, actual))
|
|
7584
8916
|
# the parser-surface table (Subcommand/Subcomando header, one `<td class="t">`
|
|
7585
8917
|
# row per subcommand) is a claim too, just not a numeric one -- a row can go
|
|
@@ -7619,9 +8951,12 @@ def cmd_facts(args):
|
|
|
7619
8951
|
body = json.dumps(facts, indent=2, ensure_ascii=False, sort_keys=True) + "\n"
|
|
7620
8952
|
with open(args.out, "w", encoding="utf-8", newline="\n") as fh:
|
|
7621
8953
|
fh.write(body)
|
|
7622
|
-
|
|
8954
|
+
dia = facts.get("diamond")
|
|
8955
|
+
print("FACTS -> %s: version %s · %d subcommands · %d skills · diamond %s"
|
|
7623
8956
|
% (args.out, facts["version"], facts["subcommands"]["count"],
|
|
7624
|
-
facts["skills"]["count"]
|
|
8957
|
+
facts["skills"]["count"],
|
|
8958
|
+
("%d PASS · %d PARTIAL of %d" % (dia["pass"], dia["partial"], dia["entries"])
|
|
8959
|
+
if dia else "UNMEASURED (no %s beside the kit)" % BENCH_DOC)))
|
|
7625
8960
|
|
|
7626
8961
|
|
|
7627
8962
|
|
|
@@ -8399,17 +9734,21 @@ def cmd_dashboard(args):
|
|
|
8399
9734
|
subscores = [{"k": "coverage",
|
|
8400
9735
|
"val": round(covp) if isinstance(covp, (int, float)) else None,
|
|
8401
9736
|
"bd": (f"{round(covp)}%" if isinstance(covp, (int, float)) else None)}]
|
|
8402
|
-
gate_block, gate_note = {}, {}
|
|
9737
|
+
gate_block, gate_note, gate_adv = {}, {}, {}
|
|
8403
9738
|
for g in rd.get("gates", []):
|
|
8404
9739
|
kind = (g.get("tool") or "").replace("gate:", "")
|
|
8405
9740
|
key = "golden" if kind.startswith("golden") else kind
|
|
8406
9741
|
gate_block[key] = gate_block.get(key, False) or bool(g.get("blocking"))
|
|
9742
|
+
gate_adv[key] = gate_adv.get(key, False) or bool(g.get("advisory"))
|
|
8407
9743
|
if g.get("note") and key not in gate_note:
|
|
8408
9744
|
gate_note[key] = g.get("note")
|
|
8409
9745
|
for key in ("simplicity", "waste", "golden"):
|
|
8410
9746
|
if key in gate_block:
|
|
9747
|
+
# ADR-043: a non-blocking ADVISORY is not "OK" — OK means a declared gate ran clean.
|
|
9748
|
+
_bd = ("FAIL" if gate_block[key]
|
|
9749
|
+
else "ADVISORY" if gate_adv.get(key) else "OK")
|
|
8411
9750
|
subscores.append({"k": key, "val": None,
|
|
8412
|
-
"bd": gate_note.get(key) or
|
|
9751
|
+
"bd": gate_note.get(key) or _bd})
|
|
8413
9752
|
|
|
8414
9753
|
# loops: iters + estado por repo (escalated > converged > active). max sin fuente.
|
|
8415
9754
|
# El estado se deriva ENTERO con _derive_phase (kit 1.48.1) — la MISMA funcion que
|
|
@@ -9033,6 +10372,10 @@ def _top_events(ledger, limit=TOP_EVENTS_TAIL):
|
|
|
9033
10372
|
gated = it.get("gated_reported")
|
|
9034
10373
|
if kind == "gate-not-run":
|
|
9035
10374
|
tail = "not run — nobody measured it"
|
|
10375
|
+
elif kind == "static-gate" and it.get("advisory"):
|
|
10376
|
+
# ADR-043: measured but not gating. `info` (never green, never red) is the
|
|
10377
|
+
# honest level -- rendering it `pass`/`clean` is the false clean again.
|
|
10378
|
+
level, tail = "info", "advisory — measured, not gating"
|
|
9036
10379
|
elif kind == "static-gate" and isinstance(gated, int):
|
|
9037
10380
|
level = "fail" if gated >= 1 else "pass"
|
|
9038
10381
|
tail = "%d gated finding(s)" % gated if gated else "clean"
|
|
@@ -9583,10 +10926,33 @@ def cmd_readiness(args):
|
|
|
9583
10926
|
# ingeridos (y 0 rojos). El checkbox es RELATO; el testcase es HECHO.
|
|
9584
10927
|
ac_ids = [i for i in acc_items if i["id"]]
|
|
9585
10928
|
ac_tags, stale_reports = _sum_ac_tags(ledger)
|
|
10929
|
+
# ADR-046: a green corpus run is the OTHER way a criterion closes measured. Greenfield has
|
|
10930
|
+
# no old code to characterize, so for the criteria that are about real-world input the only
|
|
10931
|
+
# field evidence there can be is a corpus run over real inputs -- and it closes exactly like
|
|
10932
|
+
# a green testcase does, with the same fail-closed rule below. `corpus_red` is a FAILING
|
|
10933
|
+
# tagged run -- evidence AGAINST, in ADR-046's own words -- and it vetoes like a red
|
|
10934
|
+
# testcase; an ADVISORY run is neither and does neither.
|
|
10935
|
+
corpus_closed, corpus_red = _corpus_ac_verdicts(ledger)
|
|
10936
|
+
# ADR-047: and a green SMOKE check is the third way. "the jar served /admin" used to
|
|
10937
|
+
# arrive as a sub-agent's sentence; now it arrives as a check in a report the engine
|
|
10938
|
+
# READ, and a check named "AC-28 ..." closes AC-28 exactly as a green testcase named
|
|
10939
|
+
# "AC-28 ..." does. `smoke_red` is a FAILED tagged check -- red evidence, and it vetoes
|
|
10940
|
+
# like a red testcase.
|
|
10941
|
+
smoke_closed, smoke_red = _smoke_ac_verdicts(ledger)
|
|
9586
10942
|
|
|
9587
10943
|
def _ac_closed(cid):
|
|
9588
10944
|
d = ac_tags.get(cid)
|
|
9589
|
-
|
|
10945
|
+
# fail-closed FIRST, always: red evidence of ANY kind outranks every green one,
|
|
10946
|
+
# because the cheapest way to fake a closed criterion is to add a green beside a red.
|
|
10947
|
+
if d and d["red"]:
|
|
10948
|
+
return False # red evidence vetoes, whatever else says (fail-closed)
|
|
10949
|
+
if cid in smoke_red:
|
|
10950
|
+
return False # a FAILED tagged smoke check is red evidence too
|
|
10951
|
+
if cid in corpus_red:
|
|
10952
|
+
return False # and so is a FAILING tagged corpus run (ADR-046)
|
|
10953
|
+
if d and d["green"] >= 1:
|
|
10954
|
+
return True
|
|
10955
|
+
return cid in corpus_closed or cid in smoke_closed
|
|
9590
10956
|
|
|
9591
10957
|
# IDs duplicados (ACCEPTANCE mal numerado) cuentan UNA sola vez — si no,
|
|
9592
10958
|
# un solo test verde cierra "medido" tantos criterios como copias del ID.
|
|
@@ -9832,6 +11198,17 @@ def cmd_readiness(args):
|
|
|
9832
11198
|
if (acc_traceable and total) else None),
|
|
9833
11199
|
"narrated_only": narrated_only,
|
|
9834
11200
|
"measured_unchecked": measured_unchecked,
|
|
11201
|
+
# ADR-046: WHICH ids a green corpus run closed, so a reader can tell
|
|
11202
|
+
# field evidence from suite evidence instead of inferring it -- and
|
|
11203
|
+
# `corpus_vetoed` for the ids a FAILING run holds open, the half a
|
|
11204
|
+
# reader cannot infer from the closed list.
|
|
11205
|
+
"corpus_closed": sorted(corpus_closed, key=_top_ac_key),
|
|
11206
|
+
"corpus_vetoed": sorted(corpus_red, key=_top_ac_key),
|
|
11207
|
+
# ADR-047: the same for the ids a green smoke check closed -- and
|
|
11208
|
+
# `smoke_vetoed` for the ids a FAILED one holds open, which is the
|
|
11209
|
+
# half a reader cannot infer from the closed list.
|
|
11210
|
+
"smoke_closed": sorted(smoke_closed, key=_top_ac_key),
|
|
11211
|
+
"smoke_vetoed": sorted(smoke_red, key=_top_ac_key),
|
|
9835
11212
|
"stale_reports": stale_reports},
|
|
9836
11213
|
"facts": {"coverage_pct": round(agg_cov_pct, 2), "coverage_threshold": threshold,
|
|
9837
11214
|
"gated_open": total_open, "severity": agg_sev,
|
|
@@ -9854,9 +11231,32 @@ def cmd_readiness(args):
|
|
|
9854
11231
|
# lifecycle (ADR-040): advisory, and CONDITIONAL like fast_path/spec_drift -- a project
|
|
9855
11232
|
# that declares no lifecycle: block keeps the exact prior payload and the exact prior
|
|
9856
11233
|
# text. Speaking only when it matters is the anti-ceremony rule applied to itself.
|
|
9857
|
-
|
|
11234
|
+
_ready_root = os.path.dirname(os.path.abspath(args.ledger)) or os.getcwd()
|
|
11235
|
+
_lc = _lifecycle_for(_ready_root)
|
|
9858
11236
|
if _lc["declared"]:
|
|
9859
11237
|
out["lifecycle"] = _lc
|
|
11238
|
+
# agent-origin (ADR-044): advisory and CONDITIONAL for the same reason -- a project
|
|
11239
|
+
# that tags nothing keeps the exact prior payload and the exact prior text. It never
|
|
11240
|
+
# enters the gates line, never caps the score, never blocks convergence.
|
|
11241
|
+
# field truth (ADR-046): advisory and CONDITIONAL for the same reason -- a project that
|
|
11242
|
+
# declares no corpus and ran none keeps the exact prior payload and the exact prior text.
|
|
11243
|
+
# The `field` DIMENSION and its weight are deliberately NOT here: adding one moves every
|
|
11244
|
+
# existing project's score, and that is its own ADR.
|
|
11245
|
+
_field = _corpus_field(ledger)
|
|
11246
|
+
if _field:
|
|
11247
|
+
out["field"] = _field
|
|
11248
|
+
# smoke (ADR-047): conditional for the same reason. A failing smoke ALREADY blocks
|
|
11249
|
+
# through its `gate:smoke` record; this block adds what the rollup cannot carry --
|
|
11250
|
+
# which checks ran, and which of them answered wrong.
|
|
11251
|
+
_smoke = _smoke_report(ledger)
|
|
11252
|
+
if _smoke:
|
|
11253
|
+
out["smoke"] = _smoke
|
|
11254
|
+
_ao = _agent_origin_report(_ready_root,
|
|
11255
|
+
acc_path if acc_found else None)
|
|
11256
|
+
if _ao["n_unconfirmed"] or _ao["confirmed"]:
|
|
11257
|
+
out["agent_origin"] = {"unconfirmed": _ao["unconfirmed"],
|
|
11258
|
+
"confirmed": _ao["confirmed"],
|
|
11259
|
+
"files_scanned": _ao["files_scanned"]}
|
|
9860
11260
|
if args.json:
|
|
9861
11261
|
print(json.dumps(out, indent=2, ensure_ascii=False))
|
|
9862
11262
|
return
|
|
@@ -9890,8 +11290,9 @@ def cmd_readiness(args):
|
|
|
9890
11290
|
"or the explicit weight in config.defaults.readiness_weights")
|
|
9891
11291
|
if narrated_only:
|
|
9892
11292
|
print(f" ! narrated-only: {', '.join(narrated_only)} — checkbox ticked "
|
|
9893
|
-
f"WITHOUT a green 'AC-n' testcase in the reports
|
|
9894
|
-
f"
|
|
11293
|
+
f"WITHOUT a green 'AC-n' testcase in the reports, without a green "
|
|
11294
|
+
f"corpus run carrying it and without a green 'AC-n' smoke check "
|
|
11295
|
+
f"(measured beats narrated: does NOT close)")
|
|
9895
11296
|
if measured_unchecked:
|
|
9896
11297
|
print(f" · measured but unticked: {', '.join(measured_unchecked)} — there is "
|
|
9897
11298
|
f"a green testcase; tick the checkbox if the criterion is done")
|
|
@@ -9957,13 +11358,52 @@ def cmd_readiness(args):
|
|
|
9957
11358
|
gate_roll = out["gates"]
|
|
9958
11359
|
if gate_roll:
|
|
9959
11360
|
blocking = [g for g in gate_roll if g["blocking"]]
|
|
9960
|
-
|
|
11361
|
+
# ADR-043: an advisory NEVER joins the ok count. "3 ok" must mean three gates ran and
|
|
11362
|
+
# came back clean; folding a check the project never declared as a gate into that number
|
|
11363
|
+
# is the false clean the advisory verdict exists to refuse. The segment is conditional,
|
|
11364
|
+
# so a ledger with no advisory record prints exactly what it printed before.
|
|
11365
|
+
advisory = [g for g in gate_roll if g.get("advisory") and not g["blocking"]]
|
|
11366
|
+
n_ok = len(gate_roll) - len(blocking) - len(advisory)
|
|
11367
|
+
adv_str = (f" · {len(advisory)} advisory ("
|
|
11368
|
+
+ ", ".join(f"{g['repo']}/{g['tool']}" for g in advisory) + ")"
|
|
11369
|
+
if advisory else "")
|
|
9961
11370
|
hint = "" if args.verbose else " (readiness --verbose for the detail)"
|
|
9962
11371
|
if blocking:
|
|
9963
11372
|
names = ", ".join(f"{g['repo']}/{g['tool']}" for g in blocking)
|
|
9964
|
-
print(f"--- gates: {n_ok} ok · {len(blocking)} blocking ({names}){hint}")
|
|
11373
|
+
print(f"--- gates: {n_ok} ok{adv_str} · {len(blocking)} blocking ({names}){hint}")
|
|
9965
11374
|
else:
|
|
9966
|
-
print(f"--- gates: {n_ok} ok, none blocking{hint}")
|
|
11375
|
+
print(f"--- gates: {n_ok} ok{adv_str}, none blocking{hint}")
|
|
11376
|
+
# ADR-046: the FIELD line, one per repo that declares a corpus or has run one. A failing
|
|
11377
|
+
# corpus ALREADY appears in the gates rollup above (it is a gate:corpus record like any
|
|
11378
|
+
# other); this line adds the number the rollup cannot carry -- what percentage of REAL
|
|
11379
|
+
# inputs the system gets right, against which declared budget.
|
|
11380
|
+
for _rname in sorted(_field):
|
|
11381
|
+
print(_corpus_field_line(_rname, _field[_rname]))
|
|
11382
|
+
# ADR-047: the SMOKE line, one per repo that ingested a report. Like the field line it
|
|
11383
|
+
# adds the numbers the gates rollup cannot carry -- how many checks ran, how many
|
|
11384
|
+
# answered wrong, and WHICH ones, so the failure is named instead of counted.
|
|
11385
|
+
for _rname in sorted(_smoke):
|
|
11386
|
+
print(_smoke_line(_rname, _smoke[_rname]))
|
|
11387
|
+
# ADR-044: its OWN line, deliberately outside the gates rollup. An unconfirmed
|
|
11388
|
+
# agent-origin item is a decision still owed to the human, not a gate that ran --
|
|
11389
|
+
# folding it into "N ok" or into "N blocking" would be the false clean ADR-043
|
|
11390
|
+
# refused, in the other direction. It caps nothing and blocks nothing.
|
|
11391
|
+
if _ao["n_unconfirmed"]:
|
|
11392
|
+
print(f"--- origin: {_ao['n_unconfirmed']} agent-origin item(s) unconfirmed"
|
|
11393
|
+
f" (spec-check names them)")
|
|
11394
|
+
# ADR-048: operability gets its own line, CONDITIONAL on a record existing -- a ledger that
|
|
11395
|
+
# never ran the check prints exactly what it printed before. The gates rollup above already
|
|
11396
|
+
# counts the record correctly (ok / blocking / advisory); what it cannot say is WHICH of the
|
|
11397
|
+
# four is the one to go and build, and "1 blocking (backend-api/gate:operability)" is
|
|
11398
|
+
# precisely the message that sends a human to read the source.
|
|
11399
|
+
_ops = [(rname, _latest_static_by_tool(rnode).get("gate:operability"))
|
|
11400
|
+
for rname, rnode in ledger["repos"].items()]
|
|
11401
|
+
_ops = [(rname, rec) for rname, rec in _ops if rec and rec.get("note")]
|
|
11402
|
+
for _rname, _op in _ops:
|
|
11403
|
+
_label = "operability" if len(_ops) == 1 else "operability %s" % _rname
|
|
11404
|
+
_state = (" (advisory)" if _op.get("advisory")
|
|
11405
|
+
else " (gate: FAIL)" if (_op.get("gated_reported") or 0) else " (gate)")
|
|
11406
|
+
print(f"--- {_label}: {_op['note']}{_state}")
|
|
9967
11407
|
if not args.verbose:
|
|
9968
11408
|
return
|
|
9969
11409
|
print("--- dimensions (weight | raw | contribution) ---")
|
|
@@ -10023,6 +11463,14 @@ DEFAULT_COVERAGE_TOLERANCE = 5.0 # pct points the rebuilt coverage may drop
|
|
|
10023
11463
|
# abstraction is INTENTIONALLY not weighted: the "new types" regex is a prose/AST proxy
|
|
10024
11464
|
# that false-positives on Java records/DTOs, so it must not gate the band. It stays as an
|
|
10025
11465
|
# advisory metric + flag only (distilled: hard caps gate, guessy proxies advise).
|
|
11466
|
+
#
|
|
11467
|
+
# ADVISORY BY DEFAULT (kit 2.1.0, ADR-043). Every budget below is the KIT'S OPINION, not the
|
|
11468
|
+
# project's requirement, and an opinion that exits 1 is a gate nobody declared. Until a project
|
|
11469
|
+
# declares at least one numeric budget AND `defaults.simplicity.gate: true`, the verdict is
|
|
11470
|
+
# reported and the exit code is 0. This is NOT INV-ADVISORY-01 (that invariant quarantines
|
|
11471
|
+
# LLM-class JUDGMENT; these proxies are deterministic and may gate the moment a human says so)
|
|
11472
|
+
# -- it is the provenance rule of 1.17.0 applied to an exit code: a default is an opinion, and
|
|
11473
|
+
# only a declaration is a requirement.
|
|
10026
11474
|
SIMPLICITY_WEIGHTS = {
|
|
10027
11475
|
"diff_size": 35, "nesting": 30, "net_growth": 20, "fan_out": 8, "blob": 7,
|
|
10028
11476
|
}
|
|
@@ -10037,6 +11485,17 @@ SIMPLICITY_DEFAULTS = {
|
|
|
10037
11485
|
"max_abstraction_density": 3.0, # new *types* per 100 added LOC
|
|
10038
11486
|
"indent_width": 4,
|
|
10039
11487
|
}
|
|
11488
|
+
# Keys under defaults.simplicity that are NOT budgets, so declaring one never satisfies the
|
|
11489
|
+
# "a gate needs a budget" rule: `indent_width` is a PARSING parameter and `gate` is the switch
|
|
11490
|
+
# itself. `gate: true` with nothing but these declared is a refusal, not a gate (ADR-043).
|
|
11491
|
+
_SIMPLICITY_NON_BUDGET = ("indent_width", "gate")
|
|
11492
|
+
# What `max_nesting` actually measures, said once and reused by every surface that prints it.
|
|
11493
|
+
# It is INDENTATION DEPTH over added lines, not AST nesting: a wrapped call argument, JSX, a
|
|
11494
|
+
# multi-line Java string or any deep continuation raises it without any control flow existing.
|
|
11495
|
+
# The kit does NOT make it language-aware (that needs a parser per stack, which this stdlib
|
|
11496
|
+
# engine will not have) -- it names the proxy instead, so a reader can discount it.
|
|
11497
|
+
_NESTING_PROXY_NOTE = ("indentation depth over added lines, NOT AST nesting -- continuation "
|
|
11498
|
+
"lines, JSX and multi-line literals inflate it")
|
|
10040
11499
|
# code files only — docs, config, resources and generated trees are noise for a
|
|
10041
11500
|
# code-simplicity gate. Broader than SOURCE_EXT (which is repo-typed for rebuild).
|
|
10042
11501
|
_SIMPLICITY_CODE_EXT = {
|
|
@@ -10235,18 +11694,25 @@ def _rebuild_compare(args):
|
|
|
10235
11694
|
# --------------------------------------------------------------------------- #
|
|
10236
11695
|
# simplicity-check (the "Reduce" gate)
|
|
10237
11696
|
# --------------------------------------------------------------------------- #
|
|
10238
|
-
def _read_diff(args):
|
|
10239
|
-
"""Unified-diff text from --diff, --from-git, or stdin.
|
|
11697
|
+
def _read_diff(args, detect_renames=False):
|
|
11698
|
+
"""Unified-diff text from --diff, --from-git, or stdin.
|
|
11699
|
+
|
|
11700
|
+
detect_renames (2.2.0) adds `-M` to the `--from-git` command so git reports a rename AS a
|
|
11701
|
+
rename, whatever the caller's `diff.renames` config says. It is OPT-IN because the other
|
|
11702
|
+
readers of this helper count LINES (simplicity, waste, regression): collapsing a rename into
|
|
11703
|
+
a header would silently change the numbers they have been measuring for releases. gate-check
|
|
11704
|
+
is the caller that needs it -- a rename read as a delete/add pair is what made
|
|
11705
|
+
`git mv tests/a_test.py tests/b_test.py` block as a deleted test in the field."""
|
|
10240
11706
|
if getattr(args, "diff", None):
|
|
10241
11707
|
with open(args.diff, "r", encoding="utf-8", errors="replace") as fh:
|
|
10242
11708
|
return fh.read()
|
|
10243
11709
|
if getattr(args, "from_git", False):
|
|
10244
11710
|
import subprocess
|
|
10245
11711
|
base = args.base or "HEAD"
|
|
11712
|
+
cmd = ["git", "diff", "--unified=0"] + (["-M"] if detect_renames else []) + [base]
|
|
10246
11713
|
try:
|
|
10247
11714
|
return subprocess.run(
|
|
10248
|
-
|
|
10249
|
-
check=True, capture_output=True, text=True,
|
|
11715
|
+
cmd, check=True, capture_output=True, text=True,
|
|
10250
11716
|
encoding="utf-8", errors="replace").stdout
|
|
10251
11717
|
except Exception as exc: # noqa: BLE001
|
|
10252
11718
|
print(f"[qa_ledger] git diff failed: {exc}", file=sys.stderr)
|
|
@@ -10449,8 +11915,9 @@ def _simplicity_score(m, b):
|
|
|
10449
11915
|
def _simplicity_flags(m, b):
|
|
10450
11916
|
f = []
|
|
10451
11917
|
if m["max_nesting"] > b["max_nesting_depth"]:
|
|
10452
|
-
f.append(f"
|
|
10453
|
-
f"aplanar: guard clauses / extraer función
|
|
11918
|
+
f.append(f"max_nesting (indentation proxy) {m['max_nesting']} > "
|
|
11919
|
+
f"{b['max_nesting_depth']} — aplanar: guard clauses / extraer función "
|
|
11920
|
+
f"(CWE-1124). Proxy: {_NESTING_PROXY_NOTE}")
|
|
10454
11921
|
if m["new_abstractions"] > b["max_new_abstractions"]:
|
|
10455
11922
|
f.append(f"{m['new_abstractions']} tipos/capas nuevos > "
|
|
10456
11923
|
f"{b['max_new_abstractions']} — ¿todos pedidos? "
|
|
@@ -10476,10 +11943,12 @@ def _simplicity_flags(m, b):
|
|
|
10476
11943
|
def cmd_simplicity_check(args):
|
|
10477
11944
|
b = dict(SIMPLICITY_DEFAULTS)
|
|
10478
11945
|
declared = set() # presupuestos declarados por el humano (config o CLI)
|
|
11946
|
+
gate = False # ADR-043: solo lo enciende una DECLARACION, nunca un default
|
|
10479
11947
|
if args.config and os.path.exists(args.config):
|
|
10480
11948
|
cfg = _load(args.config).get("defaults", {}).get("simplicity", {})
|
|
10481
11949
|
b.update({k: cfg[k] for k in b if k in cfg})
|
|
10482
|
-
declared |= {k for k in b if k in cfg and k
|
|
11950
|
+
declared |= {k for k in b if k in cfg and k not in _SIMPLICITY_NON_BUDGET}
|
|
11951
|
+
gate = bool(cfg.get("gate"))
|
|
10483
11952
|
for k in ("max_lines_added", "max_net_lines", "max_files_changed",
|
|
10484
11953
|
"max_nesting_depth", "max_hunk_added", "max_new_abstractions",
|
|
10485
11954
|
"indent_width"):
|
|
@@ -10491,34 +11960,55 @@ def cmd_simplicity_check(args):
|
|
|
10491
11960
|
if args.max_abstraction_density is not None:
|
|
10492
11961
|
b["max_abstraction_density"] = args.max_abstraction_density
|
|
10493
11962
|
declared.add("max_abstraction_density")
|
|
11963
|
+
if getattr(args, "gate", False):
|
|
11964
|
+
gate = True
|
|
11965
|
+
# A gate with no budget is not a gate: it is the kit's opinion wearing an exit code, which
|
|
11966
|
+
# is exactly the defect ADR-043 exists to remove. Refuse BEFORE reading the diff -- a
|
|
11967
|
+
# misconfigured gate must not produce a score anyone could quote.
|
|
11968
|
+
if gate and not declared:
|
|
11969
|
+
print("[qa_ledger] invalid config: defaults.simplicity.gate is true (or --gate was "
|
|
11970
|
+
"passed) but no simplicity budget is declared — a gate with no budget is not a "
|
|
11971
|
+
"gate, only the kit's opinion with an exit code. Declare at least one of "
|
|
11972
|
+
"max_lines_added, max_net_lines, max_files_changed, max_nesting_depth, "
|
|
11973
|
+
"max_hunk_added, max_new_abstractions, max_abstraction_density in "
|
|
11974
|
+
"defaults.simplicity (or pass the matching --max-... flag), or set gate to false.",
|
|
11975
|
+
file=sys.stderr)
|
|
11976
|
+
sys.exit(2)
|
|
11977
|
+
mode = "gate" if gate else "advisory"
|
|
10494
11978
|
|
|
10495
11979
|
m = _simplicity_metrics(_read_diff(args), b["indent_width"])
|
|
10496
11980
|
score, dims = _simplicity_score(m, b)
|
|
10497
11981
|
verdict = _simplicity_band(score)
|
|
10498
11982
|
flags = _simplicity_flags(m, b)
|
|
11983
|
+
exit_code = 1 if (verdict == "OVERBUILT" and gate) else 0
|
|
10499
11984
|
|
|
10500
|
-
out = {"score": score, "verdict": verdict, "
|
|
11985
|
+
out = {"score": score, "verdict": verdict, "mode": mode, "gate": gate,
|
|
11986
|
+
"weights": SIMPLICITY_WEIGHTS,
|
|
10501
11987
|
"dimensions": {k: round(v, 3) for k, v in dims.items()},
|
|
10502
|
-
"metrics": m, "
|
|
11988
|
+
"metrics": m, "metrics_notes": {"max_nesting": _NESTING_PROXY_NOTE},
|
|
11989
|
+
"budgets": b, "budgets_declared": sorted(declared),
|
|
10503
11990
|
"flags": flags}
|
|
10504
11991
|
if args.json:
|
|
10505
11992
|
print(json.dumps(out, indent=2, ensure_ascii=False))
|
|
10506
|
-
sys.exit(
|
|
11993
|
+
sys.exit(exit_code)
|
|
10507
11994
|
|
|
10508
|
-
|
|
11995
|
+
mode_str = ("declared gate" if gate else
|
|
11996
|
+
"advisory (declare budgets + defaults.simplicity.gate to make it block)")
|
|
11997
|
+
print(f"SIMPLICITY: {score}/100 — {verdict} ({mode_str})")
|
|
10509
11998
|
print("--- metrics (value / budget · * = declared by the human) ---")
|
|
10510
11999
|
rows = [
|
|
10511
12000
|
("lines_added", m["lines_added"], b["max_lines_added"], "max_lines_added"),
|
|
10512
12001
|
("net_lines", m["net_lines"], b["max_net_lines"], "max_net_lines"),
|
|
10513
12002
|
("files_changed", m["files_changed"], b["max_files_changed"], "max_files_changed"),
|
|
10514
|
-
("max_nesting", m["max_nesting"], b["max_nesting_depth"], "max_nesting_depth"),
|
|
12003
|
+
("max_nesting (indentation proxy)", m["max_nesting"], b["max_nesting_depth"], "max_nesting_depth"),
|
|
10515
12004
|
("new_abstractions", m["new_abstractions"], b["max_new_abstractions"], "max_new_abstractions"),
|
|
10516
12005
|
("abstraction/100", m["abstraction_density"], b["max_abstraction_density"], "max_abstraction_density"),
|
|
10517
12006
|
("max_hunk_added", m["max_hunk_added"], b["max_hunk_added"], "max_hunk_added"),
|
|
10518
12007
|
]
|
|
10519
12008
|
for name, val, bud, key in rows:
|
|
10520
12009
|
mark = "*" if key in declared else ""
|
|
10521
|
-
print(f" {name:
|
|
12010
|
+
print(f" {name:31s} {str(val):>7s} / {bud}{mark}")
|
|
12011
|
+
print(f" (max_nesting is a PROXY: {_NESTING_PROXY_NOTE})")
|
|
10522
12012
|
if not declared:
|
|
10523
12013
|
print(" (every budget is a kit default — an opinion, not a "
|
|
10524
12014
|
"requirement: declare yours in config.defaults.simplicity)")
|
|
@@ -10535,7 +12025,11 @@ def cmd_simplicity_check(args):
|
|
|
10535
12025
|
print(f" ! {fl}")
|
|
10536
12026
|
else:
|
|
10537
12027
|
print(" within budget — nothing to cut")
|
|
10538
|
-
|
|
12028
|
+
if verdict == "OVERBUILT" and not gate:
|
|
12029
|
+
print("--- advisory: OVERBUILT is REPORTED, not enforced (exit 0). Cut what is cheap, "
|
|
12030
|
+
"say so in the PR body, and do not let it block the loop. To make it block, "
|
|
12031
|
+
"declare your budgets AND defaults.simplicity.gate: true ---")
|
|
12032
|
+
sys.exit(exit_code)
|
|
10539
12033
|
|
|
10540
12034
|
|
|
10541
12035
|
# --------------------------------------------------------------------------- #
|
|
@@ -10978,8 +12472,125 @@ def _gc_new_dep(path, body):
|
|
|
10978
12472
|
return bool(rx.search(body)) if rx else False
|
|
10979
12473
|
|
|
10980
12474
|
|
|
12475
|
+
def _gc_moves(diff):
|
|
12476
|
+
"""Renames read as MOVES, never as deletions (2.2.0 field fix).
|
|
12477
|
+
|
|
12478
|
+
`git mv tests/a_test.py tests/b_test.py` used to be reported as a deleted test -- a BLOCKER
|
|
12479
|
+
and exit 1 for a change that deleted nothing. A rename reaches this parser in one of two
|
|
12480
|
+
shapes, and both are read here:
|
|
12481
|
+
|
|
12482
|
+
* git's own `rename from` / `rename to` headers, present when the producer detected
|
|
12483
|
+
renames (`--from-git` now forces `-M`, so the caller's `diff.renames` config can no
|
|
12484
|
+
longer hide one); and
|
|
12485
|
+
* an EXACT delete/add pair -- the same file content leaving one path and arriving at
|
|
12486
|
+
another inside the same diff. That is what a producer with rename detection OFF emits,
|
|
12487
|
+
and it is the shape that actually blocked in the field.
|
|
12488
|
+
|
|
12489
|
+
Returns (moves, paired). `moves` is the informational report. `paired` holds the paths of
|
|
12490
|
+
the exact pairs ONLY: their hunks say nothing about the change and are skipped. A rename
|
|
12491
|
+
WITH edits keeps its hunks, because moving a file is not a deletion but deleting a test out
|
|
12492
|
+
of a moved file still is -- and that verdict must not change.
|
|
12493
|
+
|
|
12494
|
+
The pairing is deliberately EXACT and one-to-one: same content, one file losing it, one file
|
|
12495
|
+
gaining it. Two deleted files with identical bodies are ambiguous, so neither is paired --
|
|
12496
|
+
guessing which moved where would be inventing a fact to clear a gate, which is the one thing
|
|
12497
|
+
this gate exists to refuse."""
|
|
12498
|
+
moves = []
|
|
12499
|
+
deleted, added = {}, {} # path -> tuple of line bodies
|
|
12500
|
+
path = None # the whole-file side currently being collected
|
|
12501
|
+
side = None # "-" while inside a deletion, "+" inside an addition
|
|
12502
|
+
bodies = []
|
|
12503
|
+
minus_path = None
|
|
12504
|
+
|
|
12505
|
+
def _flush():
|
|
12506
|
+
if path is not None and bodies:
|
|
12507
|
+
(deleted if side == "-" else added)[path] = tuple(bodies)
|
|
12508
|
+
|
|
12509
|
+
for raw in diff.splitlines():
|
|
12510
|
+
if raw.startswith("diff --git"):
|
|
12511
|
+
_flush()
|
|
12512
|
+
path, side, bodies, minus_path = None, None, [], None
|
|
12513
|
+
continue
|
|
12514
|
+
if raw.startswith("rename from "):
|
|
12515
|
+
minus_path = raw[len("rename from "):].strip()
|
|
12516
|
+
continue
|
|
12517
|
+
if raw.startswith("rename to "):
|
|
12518
|
+
if minus_path:
|
|
12519
|
+
moves.append("%s -> %s" % (minus_path, raw[len("rename to "):].strip()))
|
|
12520
|
+
minus_path = None
|
|
12521
|
+
continue
|
|
12522
|
+
if raw.startswith("--- "):
|
|
12523
|
+
p = raw[4:].strip().split("\t")[0]
|
|
12524
|
+
if p == "/dev/null":
|
|
12525
|
+
side = "+"
|
|
12526
|
+
else:
|
|
12527
|
+
minus_path = p[2:] if p[:2] in ("a/", "b/") else p
|
|
12528
|
+
continue
|
|
12529
|
+
if raw.startswith("+++ "):
|
|
12530
|
+
p = raw[4:].strip().split("\t")[0]
|
|
12531
|
+
if p == "/dev/null":
|
|
12532
|
+
side, path = "-", minus_path
|
|
12533
|
+
elif side == "+":
|
|
12534
|
+
path = p[2:] if p[:2] in ("a/", "b/") else p
|
|
12535
|
+
else:
|
|
12536
|
+
path, side = None, None # an ordinary edit: neither half of a move
|
|
12537
|
+
bodies = []
|
|
12538
|
+
continue
|
|
12539
|
+
if path is not None and side and raw.startswith(side):
|
|
12540
|
+
bodies.append(raw[1:])
|
|
12541
|
+
_flush()
|
|
12542
|
+
|
|
12543
|
+
paired = set()
|
|
12544
|
+
for dpath, content in deleted.items():
|
|
12545
|
+
hits = [a for a, c in added.items() if c == content]
|
|
12546
|
+
if len(hits) != 1:
|
|
12547
|
+
continue
|
|
12548
|
+
if sum(1 for c in deleted.values() if c == content) != 1:
|
|
12549
|
+
continue
|
|
12550
|
+
moves.append("%s -> %s" % (dpath, hits[0]))
|
|
12551
|
+
paired.add(dpath)
|
|
12552
|
+
paired.add(hits[0])
|
|
12553
|
+
return sorted(set(moves)), paired
|
|
12554
|
+
|
|
12555
|
+
|
|
12556
|
+
def _gc_scope(args, ledger):
|
|
12557
|
+
"""`--repo R` SCOPES the diff to the files under repos[R].path (2.2.0 field fix).
|
|
12558
|
+
|
|
12559
|
+
In a monorepo one `git diff` carries every repo's hunks, and gate-check reported all of them
|
|
12560
|
+
under whichever repo was named: a fact about someone ELSE's code, attributed to yours, with
|
|
12561
|
+
your exit code behind it. Returns (base, scope) as absolute directories, or None when there
|
|
12562
|
+
is nothing to scope by (no --repo, or a scope that is the whole tree).
|
|
12563
|
+
|
|
12564
|
+
realpath on BOTH sides before comparing. On Windows a path under a username longer than 8
|
|
12565
|
+
characters comes back short-formed (`RUNNER~1`) from one API and long-formed from another,
|
|
12566
|
+
and a file INSIDE the tree is then judged outside it -- the CI-only failure this repo has
|
|
12567
|
+
already paid for once. The scope directory is realpath'd; the diff path is joined onto an
|
|
12568
|
+
already-realpath'd base rather than realpath'd itself, because a DELETED file no longer
|
|
12569
|
+
exists and would resolve inconsistently."""
|
|
12570
|
+
if ledger is None or not getattr(args, "repo", None):
|
|
12571
|
+
return None
|
|
12572
|
+
base = os.path.realpath(os.path.dirname(os.path.abspath(args.ledger)) or ".")
|
|
12573
|
+
scope = os.path.realpath(os.path.join(base, _scope_path(ledger, args.repo)))
|
|
12574
|
+
return None if scope == base else (base, scope)
|
|
12575
|
+
|
|
12576
|
+
|
|
12577
|
+
def _gc_in_scope(path, scope):
|
|
12578
|
+
if scope is None:
|
|
12579
|
+
return True
|
|
12580
|
+
base, root = scope
|
|
12581
|
+
full = os.path.normpath(os.path.join(base, path.replace("/", os.sep)))
|
|
12582
|
+
return full == root or full.startswith(root + os.sep)
|
|
12583
|
+
|
|
12584
|
+
|
|
10981
12585
|
def cmd_gate_check(args):
|
|
10982
|
-
diff
|
|
12586
|
+
# --repo now does TWO things, and both need the ledger: it scopes the diff to that repo's
|
|
12587
|
+
# path (_gc_scope) and it adds the measured snapshot cross-check below. Loading it here
|
|
12588
|
+
# keeps the existing behaviour of an unreadable ledger or an unknown repo name exiting 2
|
|
12589
|
+
# rather than being scoped to nothing in silence.
|
|
12590
|
+
ledger = _load(args.ledger) if getattr(args, "repo", None) else None
|
|
12591
|
+
scope = _gc_scope(args, ledger)
|
|
12592
|
+
diff = _read_diff(args, detect_renames=True)
|
|
12593
|
+
moves, paired = _gc_moves(diff)
|
|
10983
12594
|
removed_tests, disabled_tests, suppressions, thresholds = [], [], [], []
|
|
10984
12595
|
secrets, secret_literals, scrub_edits, new_deps = [], [], [], []
|
|
10985
12596
|
assertions_removed = 0
|
|
@@ -11010,12 +12621,16 @@ def cmd_gate_check(args):
|
|
|
11010
12621
|
path = minus_path
|
|
11011
12622
|
else:
|
|
11012
12623
|
path = p[2:] if p[:2] in ("a/", "b/") else p
|
|
11013
|
-
|
|
11014
|
-
|
|
12624
|
+
if path is not None and (path in paired or not _gc_in_scope(path, scope)):
|
|
12625
|
+
# a MOVED half (one side of an exact delete/add pair) and a file outside
|
|
12626
|
+
# --repo's scope are not this run's business: their hunks are skipped whole.
|
|
12627
|
+
path = None
|
|
12628
|
+
elif path is not None and p != "/dev/null" and _GC_KEYFILE.search(path):
|
|
12629
|
+
secrets.append(f"{path}: contenedor de claves agregado/modificado")
|
|
11015
12630
|
elif raw.startswith("Binary files "):
|
|
11016
12631
|
# los .p12/.jks binarios no traen +++ — el lado b/ vive en esta linea
|
|
11017
12632
|
m = re.search(r" and b/(.+) differ$", raw)
|
|
11018
|
-
if m and _GC_KEYFILE.search(m.group(1)):
|
|
12633
|
+
if m and _GC_KEYFILE.search(m.group(1)) and _gc_in_scope(m.group(1), scope):
|
|
11019
12634
|
secrets.append(f"{m.group(1)}: contenedor de claves agregado/modificado (binario)")
|
|
11020
12635
|
continue
|
|
11021
12636
|
if not path:
|
|
@@ -11076,9 +12691,8 @@ def cmd_gate_check(args):
|
|
|
11076
12691
|
# optional MEASURED cross-check (heuristic-independent): with --repo, compare the
|
|
11077
12692
|
# last two snapshots' executed-test totals — a drop is a fact no regex can miss.
|
|
11078
12693
|
test_count_drop = None
|
|
11079
|
-
if
|
|
12694
|
+
if ledger is not None:
|
|
11080
12695
|
try:
|
|
11081
|
-
ledger = _load(args.ledger)
|
|
11082
12696
|
node = _repo_node(ledger, args.repo)
|
|
11083
12697
|
snaps = node.get("snapshots", [])
|
|
11084
12698
|
if len(snaps) >= 2:
|
|
@@ -11112,10 +12726,15 @@ def cmd_gate_check(args):
|
|
|
11112
12726
|
"new_dependencies": sorted(set(new_deps)),
|
|
11113
12727
|
"assertions_removed": assertions_removed,
|
|
11114
12728
|
"test_count_drop": test_count_drop,
|
|
12729
|
+
# informational, and deliberately OUTSIDE hard/soft: a move is neither a finding
|
|
12730
|
+
# nor an absolution, it is the reason a deletion is not being reported.
|
|
12731
|
+
"moved": moves,
|
|
12732
|
+
"scope": (os.path.basename(scope[1]) if scope else None),
|
|
11115
12733
|
}, indent=2, ensure_ascii=False))
|
|
11116
12734
|
sys.exit(1 if blocker else 0)
|
|
11117
12735
|
|
|
11118
|
-
print(f"GATE-INTEGRITY: {verdict}"
|
|
12736
|
+
print(f"GATE-INTEGRITY: {verdict}"
|
|
12737
|
+
+ (f" (scoped to {args.repo})" if scope else ""))
|
|
11119
12738
|
|
|
11120
12739
|
def _show(label, items):
|
|
11121
12740
|
if items:
|
|
@@ -11137,6 +12756,10 @@ def cmd_gate_check(args):
|
|
|
11137
12756
|
print(f" ~ asserts removed from tests: {assertions_removed} (review)")
|
|
11138
12757
|
if test_count_drop:
|
|
11139
12758
|
print(f" ~ executed-test count dropped: {test_count_drop} (measured in snapshots — review)")
|
|
12759
|
+
if moves:
|
|
12760
|
+
tail = " ..." if len(moves) > 5 else ""
|
|
12761
|
+
print(f" . files moved: {len(moves)} — {'; '.join(moves[:5])}{tail} "
|
|
12762
|
+
f"(informational: a rename is not a deletion)")
|
|
11140
12763
|
if verdict == "CLEAN":
|
|
11141
12764
|
print(" the change does not weaken the measuring apparatus")
|
|
11142
12765
|
elif not blocker:
|
|
@@ -11529,6 +13152,142 @@ def _lifecycle_for(root, adr_dir=None, spec_text=None, fallback=True):
|
|
|
11529
13152
|
return _lifecycle_report(adr_dir or os.path.join(root, "docs", "adr"), spec_text)
|
|
11530
13153
|
|
|
11531
13154
|
|
|
13155
|
+
# --------------------------------------------------------------------------- #
|
|
13156
|
+
# agent-origin markers (ADR-044): the agent asks for DECISIONS, never for INFORMATION
|
|
13157
|
+
# --------------------------------------------------------------------------- #
|
|
13158
|
+
# A decision the human never made must not enter scope by rebound. Every acceptance
|
|
13159
|
+
# criterion, ADR decision item or HANDOFF rule the AGENT introduced carries one trailing
|
|
13160
|
+
# marker on its own line:
|
|
13161
|
+
#
|
|
13162
|
+
# (origin: agent) introduced by the agent, NOT confirmed
|
|
13163
|
+
# (origin: agent, confirmed: YYYY-MM-DD) a human confirmed THIS item, on that day
|
|
13164
|
+
#
|
|
13165
|
+
# No marker = human origin. That is the default on purpose: nothing existing is
|
|
13166
|
+
# retro-tagged, so the absence of a marker never has to be re-audited.
|
|
13167
|
+
#
|
|
13168
|
+
# ADVISORY, always. This section never changes an exit code and never caps readiness --
|
|
13169
|
+
# it reports what has not been confirmed yet, and the human decides. A gate here would
|
|
13170
|
+
# need an adopted budget (the 2.1.0 posture, ADR-043), and nobody has declared one.
|
|
13171
|
+
#
|
|
13172
|
+
# A marker inside a fenced block or an inline code span is DOCUMENTATION, not a decision:
|
|
13173
|
+
# the ADR and the ACCEPTANCE section that DEFINE this grammar quote it, and a scanner that
|
|
13174
|
+
# read its own definition as a finding would be measuring its own prose.
|
|
13175
|
+
_AO_MARK_RX = re.compile(r"origin:\s*agent\b(?P<rest>[^)\n]*)", re.I)
|
|
13176
|
+
_AO_CONF_RX = re.compile(r"confirmed:\s*(?P<value>[^,)\s]*)", re.I)
|
|
13177
|
+
_AO_ID_RX = re.compile(r"^[\s>*+-]*(?:\[[ xX]\]\s*)?(?:\d+[.)]\s*)?[*_]*"
|
|
13178
|
+
r"(?P<id>[A-Z][A-Z0-9]*(?:-[A-Z0-9]+)+)")
|
|
13179
|
+
_AO_INLINE_CODE_RX = re.compile(r"`[^`]*`")
|
|
13180
|
+
_AO_HTML_COMMENT_RX = re.compile(r"<!--.*?-->")
|
|
13181
|
+
|
|
13182
|
+
|
|
13183
|
+
def _agent_origin_scan_text(text, label):
|
|
13184
|
+
"""Every agent-origin marker in one markdown file. Returns (unconfirmed, confirmed):
|
|
13185
|
+
a list of {id, file, line, malformed, detail} and a count. A `confirmed:` that is not
|
|
13186
|
+
a real YYYY-MM-DD date counts as UNCONFIRMED and is NAMED -- a typo must never read as
|
|
13187
|
+
a human's approval, which is the one failure this marker exists to prevent."""
|
|
13188
|
+
lines = text.split("\n")
|
|
13189
|
+
# A fence that never closes is a typo, not a decision to hide the rest of the file:
|
|
13190
|
+
# the lines after an unmatched opener are scanned as prose. Every OTHER fence still
|
|
13191
|
+
# hides its body, so the grammar's own definitions stay documentation.
|
|
13192
|
+
fenced, in_fence, opener = [False] * len(lines), False, None
|
|
13193
|
+
for idx, raw in enumerate(lines):
|
|
13194
|
+
st = raw.strip()
|
|
13195
|
+
if st.startswith("```") or st.startswith("~~~"):
|
|
13196
|
+
in_fence = not in_fence
|
|
13197
|
+
opener = idx if in_fence else None
|
|
13198
|
+
fenced[idx] = True
|
|
13199
|
+
continue
|
|
13200
|
+
fenced[idx] = in_fence
|
|
13201
|
+
if in_fence and opener is not None:
|
|
13202
|
+
for idx in range(opener, len(lines)):
|
|
13203
|
+
fenced[idx] = False
|
|
13204
|
+
unconfirmed, confirmed = [], 0
|
|
13205
|
+
for i, raw in enumerate(lines, 1):
|
|
13206
|
+
if fenced[i - 1]:
|
|
13207
|
+
continue
|
|
13208
|
+
line = _AO_HTML_COMMENT_RX.sub("", _AO_INLINE_CODE_RX.sub("", raw))
|
|
13209
|
+
idm = _AO_ID_RX.match(line)
|
|
13210
|
+
item = idm.group("id") if idm else "line %d" % i
|
|
13211
|
+
# every marker on the line, not the first: a confirmation appended after the
|
|
13212
|
+
# original tag must be read, never dropped in silence
|
|
13213
|
+
for m in _AO_MARK_RX.finditer(line):
|
|
13214
|
+
cm = _AO_CONF_RX.search(m.group("rest") or "")
|
|
13215
|
+
if cm is None:
|
|
13216
|
+
unconfirmed.append({"id": item, "file": label, "line": i,
|
|
13217
|
+
"malformed": False, "detail": None})
|
|
13218
|
+
elif _lc_valid_date(cm.group("value")):
|
|
13219
|
+
confirmed += 1
|
|
13220
|
+
else:
|
|
13221
|
+
unconfirmed.append({"id": item, "file": label, "line": i, "malformed": True,
|
|
13222
|
+
"detail": "malformed confirmed: %s"
|
|
13223
|
+
% (cm.group("value") or "(empty)")})
|
|
13224
|
+
return unconfirmed, confirmed
|
|
13225
|
+
|
|
13226
|
+
|
|
13227
|
+
def _agent_origin_report(root, acceptance=None, adr_dir=None, extra=()):
|
|
13228
|
+
"""The advisory dimension over the files that hold decisions: the ACCEPTANCE file
|
|
13229
|
+
(the one named, else `<root>/ACCEPTANCE.md`), every ADR under `adr_dir`, `HANDOFF.md`
|
|
13230
|
+
when present, and whatever the caller already had open (`extra`). Absent files are
|
|
13231
|
+
simply not scanned -- there is nothing to report about a file that does not exist."""
|
|
13232
|
+
paths, seen = [], set()
|
|
13233
|
+
|
|
13234
|
+
def add(p):
|
|
13235
|
+
if not p:
|
|
13236
|
+
return
|
|
13237
|
+
try:
|
|
13238
|
+
key = os.path.realpath(p)
|
|
13239
|
+
except OSError:
|
|
13240
|
+
key = os.path.abspath(p)
|
|
13241
|
+
if key in seen or not os.path.isfile(p):
|
|
13242
|
+
return
|
|
13243
|
+
seen.add(key)
|
|
13244
|
+
paths.append(p)
|
|
13245
|
+
|
|
13246
|
+
for p in extra:
|
|
13247
|
+
add(p)
|
|
13248
|
+
add(acceptance or os.path.join(root, "ACCEPTANCE.md"))
|
|
13249
|
+
add(os.path.join(root, "HANDOFF.md"))
|
|
13250
|
+
adr = adr_dir or os.path.join(root, "docs", "adr")
|
|
13251
|
+
if os.path.isdir(adr):
|
|
13252
|
+
for f in sorted(glob.glob(os.path.join(adr, "*.md"))):
|
|
13253
|
+
add(f)
|
|
13254
|
+
unconfirmed, confirmed, scanned = [], 0, []
|
|
13255
|
+
for p in paths:
|
|
13256
|
+
try:
|
|
13257
|
+
with open(p, "r", encoding="utf-8", errors="replace") as fh:
|
|
13258
|
+
body = fh.read()
|
|
13259
|
+
except OSError:
|
|
13260
|
+
continue # unreadable is not a finding; it is simply not scanned
|
|
13261
|
+
# forward slashes always: the same tree must name the same file the same way on
|
|
13262
|
+
# Windows and on the CI cells, or a pinned line differs by separator alone.
|
|
13263
|
+
label = _lc_short(p).replace("\\", "/")
|
|
13264
|
+
scanned.append(label)
|
|
13265
|
+
u, c = _agent_origin_scan_text(body, label)
|
|
13266
|
+
unconfirmed += u
|
|
13267
|
+
confirmed += c
|
|
13268
|
+
return {"unconfirmed": unconfirmed, "confirmed": confirmed,
|
|
13269
|
+
"n_unconfirmed": len(unconfirmed), "files_scanned": scanned}
|
|
13270
|
+
|
|
13271
|
+
|
|
13272
|
+
def _agent_origin_names(ao, limit=6):
|
|
13273
|
+
"""`AC-07 (ACCEPTANCE.md:41), D-03 (docs/adr/ADR-002-x.md:57)` -- the id, where it is,
|
|
13274
|
+
and for a malformed marker WHY it did not count as confirmed."""
|
|
13275
|
+
items = ao["unconfirmed"]
|
|
13276
|
+
out = ", ".join("%s (%s:%d%s)" % (x["id"], x["file"], x["line"],
|
|
13277
|
+
", " + x["detail"] if x["detail"] else "")
|
|
13278
|
+
for x in items[:limit])
|
|
13279
|
+
if len(items) > limit:
|
|
13280
|
+
out += " +%d more" % (len(items) - limit)
|
|
13281
|
+
return out
|
|
13282
|
+
|
|
13283
|
+
|
|
13284
|
+
def _agent_origin_line(ao):
|
|
13285
|
+
"""The one advisory line, shared by spec-check and readiness so the two surfaces
|
|
13286
|
+
cannot drift apart."""
|
|
13287
|
+
return ("origin: %d agent-origin item(s) unconfirmed -- %s"
|
|
13288
|
+
% (ao["n_unconfirmed"], _agent_origin_names(ao)))
|
|
13289
|
+
|
|
13290
|
+
|
|
11532
13291
|
def _spec_check_text(text):
|
|
11533
13292
|
lines = text.split("\n")
|
|
11534
13293
|
n = len(lines)
|
|
@@ -11824,10 +13583,14 @@ def cmd_spec_check(args):
|
|
|
11824
13583
|
else {"blockers": [], "untestable": [], "stack_hits": [],
|
|
11825
13584
|
"non_ears": 0, "n_criteria": 0})
|
|
11826
13585
|
# lifecycle (ADR-040): read-only and advisory -- it never touches `fail` below.
|
|
11827
|
-
|
|
11828
|
-
|
|
11829
|
-
getattr(args, "adr_dir", None), text,
|
|
13586
|
+
_root = _lifecycle_root(args.spec[0] if args.spec else None, args.acceptance)
|
|
13587
|
+
lc = _lifecycle_for(_root, getattr(args, "adr_dir", None), text,
|
|
11830
13588
|
fallback=not args.spec)
|
|
13589
|
+
# agent-origin (ADR-044): read-only and advisory -- like lifecycle above, it never
|
|
13590
|
+
# touches `fail` below. An unconfirmed item is not a defect; it is a decision still
|
|
13591
|
+
# owed to the human.
|
|
13592
|
+
ao = _agent_origin_report(_root, args.acceptance,
|
|
13593
|
+
getattr(args, "adr_dir", None), extra=args.spec or ())
|
|
11831
13594
|
structural = len(m["blockers"]) + len(acc_block) # estructura = FACT -> bloquea
|
|
11832
13595
|
soft_find = len(m["untestable"]) + len(m["stack_hits"]) + len(acc_adv)
|
|
11833
13596
|
fail = structural > 0 or (args.strict and soft_find > 0)
|
|
@@ -11836,7 +13599,9 @@ def cmd_spec_check(args):
|
|
|
11836
13599
|
if args.json:
|
|
11837
13600
|
print(json.dumps({"verdict": verdict, "advisory": structural == 0,
|
|
11838
13601
|
"acceptance_blockers": acc_block,
|
|
11839
|
-
"acceptance_advisory": acc_adv, "lifecycle": lc,
|
|
13602
|
+
"acceptance_advisory": acc_adv, "lifecycle": lc,
|
|
13603
|
+
"agent_origin": {"unconfirmed": ao["unconfirmed"],
|
|
13604
|
+
"confirmed": ao["confirmed"]}, **m},
|
|
11840
13605
|
indent=2, ensure_ascii=False))
|
|
11841
13606
|
sys.exit(1 if fail else 0)
|
|
11842
13607
|
|
|
@@ -11862,6 +13627,10 @@ def cmd_spec_check(args):
|
|
|
11862
13627
|
print(" %s %s %s - %s (%s)"
|
|
11863
13628
|
% ("!" if c["status"] == "expires before go-live" else "~",
|
|
11864
13629
|
c["component"], c.get("version") or "?", c["status"], c["detail"]))
|
|
13630
|
+
# conditional, like every other advisory line here: a tree with no marker prints
|
|
13631
|
+
# exactly what it printed before this release.
|
|
13632
|
+
if ao["n_unconfirmed"]:
|
|
13633
|
+
print(" ~ " + _agent_origin_line(ao))
|
|
11865
13634
|
print(" i consistency: INFERENTIAL (an uncorrelated checker), not this lint · "
|
|
11866
13635
|
"structure = FACT (blocks) · prose = advisory (--strict to gate)")
|
|
11867
13636
|
if verdict == "OK":
|
|
@@ -12216,6 +13985,91 @@ def _doctor_hook_registered(settings_path):
|
|
|
12216
13985
|
return next((n for n in HOOK_NAMES if n in blob), None)
|
|
12217
13986
|
|
|
12218
13987
|
|
|
13988
|
+
# --------------------------------------------------------------------------- #
|
|
13989
|
+
# installed-skill freshness (2.2.0)
|
|
13990
|
+
# --------------------------------------------------------------------------- #
|
|
13991
|
+
# The field case: skills sat under ~/.claude/skills/uscha-* dated before 1.54.0 while the kit
|
|
13992
|
+
# was 1.97.0. A whole discovery ran on the old prose and nothing said a word, because a SKILL.md
|
|
13993
|
+
# carried no version to compare. Since 2.2.0 the GENERATED orientation block opens with
|
|
13994
|
+
# `<!-- uscha kit: X.Y.Z ... -->` (tools/skill-blocks/, rendered by tools/gen-skill-blocks.py and
|
|
13995
|
+
# re-rendered by tools/release.py at every bump), so the comparison is mechanical.
|
|
13996
|
+
#
|
|
13997
|
+
# ADVISORY, always: this reports, it never gates. `doctor` already exits 1 only on errors, and an
|
|
13998
|
+
# outdated install is a WARN -- the operator may be pinning a version on purpose.
|
|
13999
|
+
SKILL_KIT_MARK = re.compile(r"uscha kit:\s*(\d+\.\d+\.\d+)")
|
|
14000
|
+
# Where install-uscha.py puts the skills: TARGETS = ("codex", "claude") + SKILL_ROOTS. Retyped
|
|
14001
|
+
# here rather than imported because install-uscha.py lives at the KIT ROOT and is not installed
|
|
14002
|
+
# alongside the engine -- an installed engine could not import it. Kept in one place so the
|
|
14003
|
+
# drift, if it ever happens, is one table against one table.
|
|
14004
|
+
SKILL_INSTALL_ROOTS = (
|
|
14005
|
+
("claude", (".claude", "skills")),
|
|
14006
|
+
("codex", ("plugins", "uscha", "skills")),
|
|
14007
|
+
("pi", (".agents", "skills")),
|
|
14008
|
+
("cursor", (".cursor", "skills")),
|
|
14009
|
+
("copilot", (".copilot", "skills")),
|
|
14010
|
+
("gemini", (".gemini", "skills")),
|
|
14011
|
+
("cline", (".cline", "skills")),
|
|
14012
|
+
)
|
|
14013
|
+
|
|
14014
|
+
|
|
14015
|
+
def _semver_tuple(text):
|
|
14016
|
+
"""(major, minor, patch) for comparison, or None when the string is not one."""
|
|
14017
|
+
m = re.match(r"^(\d+)\.(\d+)\.(\d+)$", (text or "").strip())
|
|
14018
|
+
return tuple(int(g) for g in m.groups()) if m else None
|
|
14019
|
+
|
|
14020
|
+
|
|
14021
|
+
def _installed_skill_report(root, kit_version):
|
|
14022
|
+
"""One install root's uscha-* skills, each with the kit version its block was stamped with.
|
|
14023
|
+
|
|
14024
|
+
Returns None when the root holds no uscha skill at all -- "not installed" is a state, not a
|
|
14025
|
+
fault, and reporting it as an error would make `doctor` red on every machine that installed
|
|
14026
|
+
for one agent out of seven."""
|
|
14027
|
+
if not os.path.isdir(root):
|
|
14028
|
+
return None
|
|
14029
|
+
want = _semver_tuple(kit_version)
|
|
14030
|
+
found, oldest = [], None
|
|
14031
|
+
for name in USCHA_SKILLS:
|
|
14032
|
+
smd = os.path.join(root, name, "SKILL.md")
|
|
14033
|
+
if not os.path.isfile(smd):
|
|
14034
|
+
continue
|
|
14035
|
+
try:
|
|
14036
|
+
with open(smd, encoding="utf-8", errors="replace") as fh:
|
|
14037
|
+
head = fh.read(8192)
|
|
14038
|
+
except OSError:
|
|
14039
|
+
head = ""
|
|
14040
|
+
m = SKILL_KIT_MARK.search(head)
|
|
14041
|
+
seen = m.group(1) if m else None
|
|
14042
|
+
found.append({"skill": name, "installed": seen})
|
|
14043
|
+
got = _semver_tuple(seen)
|
|
14044
|
+
if got is not None and (oldest is None or got < oldest):
|
|
14045
|
+
oldest = got
|
|
14046
|
+
if not found:
|
|
14047
|
+
return None
|
|
14048
|
+
unmarked = [f["skill"] for f in found if f["installed"] is None]
|
|
14049
|
+
if unmarked:
|
|
14050
|
+
# no marker at all = a block rendered before 2.2.0. Older than anything that carries
|
|
14051
|
+
# one, and said that way rather than as a version nobody wrote.
|
|
14052
|
+
status = "outdated"
|
|
14053
|
+
installed = None
|
|
14054
|
+
elif want is None or oldest is None:
|
|
14055
|
+
status = "unknown"
|
|
14056
|
+
installed = ".".join(str(p) for p in oldest) if oldest else None
|
|
14057
|
+
else:
|
|
14058
|
+
installed = ".".join(str(p) for p in oldest)
|
|
14059
|
+
status = "outdated" if oldest < want else "current"
|
|
14060
|
+
return {"root": root, "status": status, "installed": installed, "kit": kit_version,
|
|
14061
|
+
"skills": found, "unmarked": unmarked}
|
|
14062
|
+
|
|
14063
|
+
|
|
14064
|
+
def _installed_skill_roots(args):
|
|
14065
|
+
"""(target, root) pairs to inspect: the caller's `--installed` when given, else every root
|
|
14066
|
+
the installer knows, under the user's home."""
|
|
14067
|
+
if getattr(args, "installed", None):
|
|
14068
|
+
return [("--installed", d) for d in args.installed]
|
|
14069
|
+
home = os.path.expanduser("~")
|
|
14070
|
+
return [(t, os.path.join(home, *parts)) for t, parts in SKILL_INSTALL_ROOTS]
|
|
14071
|
+
|
|
14072
|
+
|
|
12219
14073
|
def cmd_doctor(args):
|
|
12220
14074
|
checks = [] # (nivel 'ok'|'warn'|'error', titulo, detalle)
|
|
12221
14075
|
|
|
@@ -12285,6 +14139,45 @@ def cmd_doctor(args):
|
|
|
12285
14139
|
if mismatched:
|
|
12286
14140
|
err(f"SKILL.md con frontmatter name distinto al directorio: {', '.join(mismatched)}")
|
|
12287
14141
|
|
|
14142
|
+
# --- installed skills vs the kit's VERSION (2.2.0) ----------------------
|
|
14143
|
+
# A discovery once ran on prose from 1.54.0 while the kit was 1.97.0, and nothing said so.
|
|
14144
|
+
# Advisory: reported as a warning, never an error, so it can never fail an installation
|
|
14145
|
+
# someone pinned on purpose.
|
|
14146
|
+
kit_version, version_src = _engine_kit_version()
|
|
14147
|
+
skills_installed = []
|
|
14148
|
+
absent = []
|
|
14149
|
+
for target, sroot in _installed_skill_roots(args):
|
|
14150
|
+
rep = _installed_skill_report(sroot, kit_version)
|
|
14151
|
+
if rep is None:
|
|
14152
|
+
absent.append((target, sroot))
|
|
14153
|
+
skills_installed.append({"target": target, "root": sroot,
|
|
14154
|
+
"status": "not installed", "installed": None,
|
|
14155
|
+
"kit": kit_version, "skills": [], "unmarked": []})
|
|
14156
|
+
continue
|
|
14157
|
+
rep["target"] = target
|
|
14158
|
+
skills_installed.append(rep)
|
|
14159
|
+
fix = (f"re-install: python install-uscha.py install --target {target} "
|
|
14160
|
+
f"(or `uscha install`)")
|
|
14161
|
+
if rep["status"] == "outdated":
|
|
14162
|
+
shown = rep["installed"] or "no `kit:` marker (block rendered before 2.2.0)"
|
|
14163
|
+
warn(f"SKILLS OUTDATED at {sroot}: installed {shown} < kit {kit_version}", fix)
|
|
14164
|
+
elif rep["status"] == "current":
|
|
14165
|
+
ok(f"skills {target}: current (kit {kit_version})", sroot)
|
|
14166
|
+
else:
|
|
14167
|
+
warn(f"skills {target}: version UNMEASURED at {sroot}",
|
|
14168
|
+
"the installed blocks carry a marker this engine cannot compare "
|
|
14169
|
+
f"(installed {rep['installed']!r}, kit {kit_version!r})")
|
|
14170
|
+
if absent:
|
|
14171
|
+
ok("skills not installed for: " + ", ".join(t for t, _ in absent),
|
|
14172
|
+
"not an error -- the kit installs per agent, one target at a time")
|
|
14173
|
+
if kit_version is None:
|
|
14174
|
+
warn("kit version not readable: installed-skill freshness is UNMEASURED",
|
|
14175
|
+
"the comparison needs a `%s`, a `VERSION` or an `uscha-install.json` at or "
|
|
14176
|
+
"above the engine; looked in: " % KIT_VERSION_COPY
|
|
14177
|
+
+ ", ".join(version_src or []) + " -- re-install with "
|
|
14178
|
+
"`python install-uscha.py install` (2.2.0 and later copy it beside the "
|
|
14179
|
+
"installed skills)")
|
|
14180
|
+
|
|
12288
14181
|
# --- hook INV-GOLDEN-01 -------------------------------------------------
|
|
12289
14182
|
kit_root = os.path.abspath(os.path.join(engine_dir, "..", "..", ".."))
|
|
12290
14183
|
hook_dirs = [os.path.join(os.path.expanduser("~"), ".claude", "hooks"),
|
|
@@ -12461,6 +14354,10 @@ def cmd_doctor(args):
|
|
|
12461
14354
|
# effective settings + origin per knob (2.0.0); null when there is
|
|
12462
14355
|
# no project config here to resolve them from
|
|
12463
14356
|
"risk_profile": risk_profile, "effective": effective,
|
|
14357
|
+
# installed-skill freshness (2.2.0): one row per install root the
|
|
14358
|
+
# installer knows, with BOTH versions -- advisory, never in the verdict
|
|
14359
|
+
"kit_version": kit_version,
|
|
14360
|
+
"skills_installed": skills_installed,
|
|
12464
14361
|
"checks": [{"level": lv, "title": t, "detail": d}
|
|
12465
14362
|
for lv, t, d in checks]},
|
|
12466
14363
|
indent=2, ensure_ascii=True))
|
|
@@ -12489,6 +14386,11 @@ def build_parser():
|
|
|
12489
14386
|
"python/git, skills, hook, project config, per-repo toolchains")
|
|
12490
14387
|
pdoc.add_argument("--config", default=None,
|
|
12491
14388
|
help="project config to inspect (default: ./uscha.config.json)")
|
|
14389
|
+
pdoc.add_argument("--installed", action="append", default=None, metavar="DIR",
|
|
14390
|
+
help="skill install root to compare against the kit's VERSION "
|
|
14391
|
+
"(repeatable; default: every root install-uscha.py writes to, "
|
|
14392
|
+
"under ~). Advisory -- an outdated install is a warning, never "
|
|
14393
|
+
"an error")
|
|
12492
14394
|
pdoc.add_argument("--ledger", default=DEFAULT_LEDGER)
|
|
12493
14395
|
pdoc.add_argument("--json", action="store_true")
|
|
12494
14396
|
pdoc.set_defaults(func=cmd_doctor)
|
|
@@ -12510,9 +14412,21 @@ def build_parser():
|
|
|
12510
14412
|
pri.add_argument("--json", action="store_true")
|
|
12511
14413
|
pri.set_defaults(func=cmd_rubric_ingest)
|
|
12512
14414
|
|
|
12513
|
-
pi = sub.add_parser("init", help="create the ledger from a config file"
|
|
12514
|
-
|
|
14415
|
+
pi = sub.add_parser("init", help="create the ledger from a config file, or --add-repo one "
|
|
14416
|
+
"repo into an existing ledger")
|
|
14417
|
+
pi.add_argument("--config", default=None,
|
|
14418
|
+
help="the uscha.config.json to freeze into a NEW ledger (required unless "
|
|
14419
|
+
"--add-repo); with --add-repo it is the file to mirror the new repo "
|
|
14420
|
+
"into (default: uscha.config.json next to --out)")
|
|
12515
14421
|
pi.add_argument("--out", default=DEFAULT_LEDGER)
|
|
14422
|
+
pi.add_argument("--add-repo", dest="add_repo", default=None, metavar="NAME",
|
|
14423
|
+
help="append ONE repo to the EXISTING ledger at --out instead of creating a "
|
|
14424
|
+
"new one: every existing repo's steps, snapshots and iterations are "
|
|
14425
|
+
"left untouched and the checksum is re-sealed")
|
|
14426
|
+
pi.add_argument("--path", default=None, help="(with --add-repo) the new repo's path")
|
|
14427
|
+
pi.add_argument("--type", default=None, help="(with --add-repo) the new repo's type")
|
|
14428
|
+
pi.add_argument("--test-command", dest="test_command", default=None,
|
|
14429
|
+
help="(with --add-repo) the new repo's test command")
|
|
12516
14430
|
pi.set_defaults(func=cmd_init)
|
|
12517
14431
|
|
|
12518
14432
|
def add_ledger(sp):
|
|
@@ -12858,6 +14772,15 @@ def build_parser():
|
|
|
12858
14772
|
help="override defaults.spec_drift.max_lag_days (default 30)")
|
|
12859
14773
|
psd.add_argument("--json", action="store_true")
|
|
12860
14774
|
psd.set_defaults(func=cmd_spec_drift)
|
|
14775
|
+
|
|
14776
|
+
pop = sub.add_parser("operability",
|
|
14777
|
+
help="measure CI / release / RUNBOOK / seed as FACTS in the tree "
|
|
14778
|
+
"(ADR-048); exit 0 always -- the gate is the profile's")
|
|
14779
|
+
add_ledger(pop)
|
|
14780
|
+
pop.add_argument("--repo", required=True)
|
|
14781
|
+
pop.add_argument("--iteration", type=int, default=1)
|
|
14782
|
+
pop.add_argument("--json", action="store_true")
|
|
14783
|
+
pop.set_defaults(func=cmd_operability)
|
|
12861
14784
|
pre = sub.add_parser("resolve-escalation",
|
|
12862
14785
|
help="close open escalations for a repo (recorded event; "
|
|
12863
14786
|
"lifts the readiness cap)")
|
|
@@ -12868,20 +14791,82 @@ def build_parser():
|
|
|
12868
14791
|
|
|
12869
14792
|
plg = sub.add_parser(
|
|
12870
14793
|
"log-gate",
|
|
12871
|
-
help="persist a FACT-gate verdict (golden-diff/gate-check/pit-check/simplicity/
|
|
12872
|
-
"so converged and readiness actually see it")
|
|
14794
|
+
help="persist a FACT-gate verdict (golden-diff/gate-check/pit-check/simplicity/"
|
|
14795
|
+
"regression/ci) so converged and readiness actually see it")
|
|
12873
14796
|
add_ledger(plg)
|
|
12874
14797
|
plg.add_argument("--repo", required=True)
|
|
12875
14798
|
plg.add_argument("--iteration", type=int, required=True)
|
|
12876
14799
|
plg.add_argument("--kind", required=True,
|
|
12877
14800
|
choices=["golden-diff", "gate-check", "pit-check", "simplicity",
|
|
12878
|
-
"regression", "rubric", "waste"
|
|
12879
|
-
|
|
14801
|
+
"regression", "rubric", "waste", "ci", "corpus", "smoke",
|
|
14802
|
+
"operability"],
|
|
14803
|
+
help="ci (2.2.0) records a pipeline run as the FACT it is: a fail caps "
|
|
14804
|
+
"readiness <=65 and blocks convergence exactly like gate-check. "
|
|
14805
|
+
"corpus (ADR-046) is the parity door for a field-truth run measured "
|
|
14806
|
+
"elsewhere; corpus-run writes the same record with the evidence on "
|
|
14807
|
+
"it. smoke (ADR-047) is the same door for a smoke run measured "
|
|
14808
|
+
"elsewhere -- a FACT kind, so advisory is refused on it. "
|
|
14809
|
+
"operability (ADR-048) is accepted for parity with the "
|
|
14810
|
+
"`operability` subcommand, which is what normally writes it")
|
|
14811
|
+
plg.add_argument("--verdict", required=True,
|
|
14812
|
+
choices=["pass", "fail", "advisory", "not-run"],
|
|
14813
|
+
help="advisory (ADR-043) records a measured, non-gating run: it never "
|
|
14814
|
+
"caps readiness, never blocks convergence, and never reads as ok; "
|
|
14815
|
+
"accepted only for --kind simplicity|waste|corpus|operability, a FACT "
|
|
14816
|
+
"gate refuses it")
|
|
12880
14817
|
plg.add_argument("--count", type=int, default=1,
|
|
12881
14818
|
help="failing finding count (fail only; default 1)")
|
|
12882
14819
|
plg.add_argument("--note", default=None)
|
|
14820
|
+
plg.add_argument("--ref", default=None,
|
|
14821
|
+
help="where the verdict was measured (a CI run URL or id), stored on the "
|
|
14822
|
+
"record so the evidence outlives the conversation")
|
|
12883
14823
|
plg.set_defaults(func=cmd_log_gate)
|
|
12884
14824
|
|
|
14825
|
+
pcr = sub.add_parser(
|
|
14826
|
+
"corpus-run",
|
|
14827
|
+
help="run a REAL-INPUT corpus against a command and persist gate:corpus (ADR-046): "
|
|
14828
|
+
"field truth for greenfield, where every test payload was invented by the agent")
|
|
14829
|
+
add_ledger(pcr)
|
|
14830
|
+
pcr.add_argument("--repo", required=True)
|
|
14831
|
+
pcr.add_argument("--corpus", default=None,
|
|
14832
|
+
help="JSONL corpus: one {\"input\": ..., \"expected\": ..., \"id\": ...} "
|
|
14833
|
+
"per line (default: repos[R].corpus from the config)")
|
|
14834
|
+
pcr.add_argument("--command", required=True,
|
|
14835
|
+
help="the command under test; each case's input arrives on ITS stdin "
|
|
14836
|
+
"(JSON-encoded when it is not a string)")
|
|
14837
|
+
pcr.add_argument("--threshold", type=float, default=None,
|
|
14838
|
+
help="hit percentage the run must reach to PASS (default: "
|
|
14839
|
+
"repos[R].corpus_threshold, else defaults.corpus_threshold; with "
|
|
14840
|
+
"NONE declared the run is advisory -- a gate needs an adopted budget)")
|
|
14841
|
+
pcr.add_argument("--ac", action="append", default=None,
|
|
14842
|
+
help="criterion id this run is evidence for (repeatable); a criterion "
|
|
14843
|
+
"whose only evidence is a corpus record closes MEASURED iff it passed")
|
|
14844
|
+
pcr.add_argument("--timeout", type=float, default=CORPUS_DEFAULT_TIMEOUT,
|
|
14845
|
+
help="per-case seconds before the case is a miss named `timeout` "
|
|
14846
|
+
"(default %d)" % CORPUS_DEFAULT_TIMEOUT)
|
|
14847
|
+
pcr.add_argument("--max-misses", type=int, default=CORPUS_DEFAULT_MAX_MISSES,
|
|
14848
|
+
help="how many misses to report and persist (default %d)"
|
|
14849
|
+
% CORPUS_DEFAULT_MAX_MISSES)
|
|
14850
|
+
pcr.add_argument("--iteration", type=int, default=1)
|
|
14851
|
+
pcr.add_argument("--json", action="store_true")
|
|
14852
|
+
pcr.set_defaults(func=cmd_corpus_run)
|
|
14853
|
+
|
|
14854
|
+
psi = sub.add_parser(
|
|
14855
|
+
"smoke-ingest",
|
|
14856
|
+
help="ingest a smoke report as gate:smoke (ADR-047): the smoke run MEASURED "
|
|
14857
|
+
"instead of narrated -- evidence is executed, not written down")
|
|
14858
|
+
add_ledger(psi)
|
|
14859
|
+
psi.add_argument("--repo", required=True)
|
|
14860
|
+
psi.add_argument("--report", required=True,
|
|
14861
|
+
help='JSON the PROJECT writes with its own smoke tool: '
|
|
14862
|
+
'{"checks": [{"name": ..., "ok": true|false, "status": ..., '
|
|
14863
|
+
'"latency_ms": ..., "evidence": ...}, ...]}. A missing `checks`, '
|
|
14864
|
+
'an empty list, or a check without a name or a boolean `ok` is '
|
|
14865
|
+
'exit 2 naming it -- never a scored run')
|
|
14866
|
+
psi.add_argument("--iteration", type=int, default=1)
|
|
14867
|
+
psi.add_argument("--json", action="store_true")
|
|
14868
|
+
psi.set_defaults(func=cmd_smoke_ingest)
|
|
14869
|
+
|
|
12885
14870
|
pfb = sub.add_parser(
|
|
12886
14871
|
"flag-blocker",
|
|
12887
14872
|
help="record (or --resolve) a CONSTITUTION/invariant breach as a BLOCKER "
|
|
@@ -13057,7 +15042,9 @@ def build_parser():
|
|
|
13057
15042
|
|
|
13058
15043
|
ps2 = sub.add_parser(
|
|
13059
15044
|
"simplicity-check",
|
|
13060
|
-
help="Reduce gate: score diff minimality/complexity over a unified diff"
|
|
15045
|
+
help="Reduce gate: score diff minimality/complexity over a unified diff. Advisory "
|
|
15046
|
+
"by default; gates only with --gate or defaults.simplicity.gate AND at least "
|
|
15047
|
+
"one declared budget")
|
|
13061
15048
|
ps2.add_argument("--diff", help="path to a unified diff (else --from-git or stdin)")
|
|
13062
15049
|
ps2.add_argument("--from-git", action="store_true",
|
|
13063
15050
|
help="run `git diff --unified=0 <base>` for the diff")
|
|
@@ -13073,6 +15060,10 @@ def build_parser():
|
|
|
13073
15060
|
ps2.add_argument("--max-abstraction-density", dest="max_abstraction_density",
|
|
13074
15061
|
type=float, default=None)
|
|
13075
15062
|
ps2.add_argument("--indent-width", dest="indent_width", type=int)
|
|
15063
|
+
ps2.add_argument("--gate", action="store_true",
|
|
15064
|
+
help="make an OVERBUILT verdict exit 1 (same switch as "
|
|
15065
|
+
"defaults.simplicity.gate). Requires at least one declared budget: "
|
|
15066
|
+
"without one the run REFUSES with exit 2 (ADR-043)")
|
|
13076
15067
|
ps2.add_argument("--json", action="store_true")
|
|
13077
15068
|
ps2.set_defaults(func=cmd_simplicity_check)
|
|
13078
15069
|
|
|
@@ -13121,8 +15112,9 @@ def build_parser():
|
|
|
13121
15112
|
pgc.add_argument("--ledger", default=DEFAULT_LEDGER,
|
|
13122
15113
|
help="(with --repo) ledger for the measured test-count cross-check")
|
|
13123
15114
|
pgc.add_argument("--repo", default=None,
|
|
13124
|
-
help="
|
|
13125
|
-
"
|
|
15115
|
+
help="SCOPE the diff to repos[R].path (a monorepo sibling's hunks are not "
|
|
15116
|
+
"this repo's findings) and compare the last two snapshots' "
|
|
15117
|
+
"executed-test totals (a measured drop flags REVIEW)")
|
|
13126
15118
|
pgc.add_argument("--json", action="store_true")
|
|
13127
15119
|
pgc.set_defaults(func=cmd_gate_check)
|
|
13128
15120
|
|