@andresmassello/uscha 2.0.0 → 2.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/README.md +21 -4
  2. package/package.json +1 -1
  3. package/uscha-kit/.claude/skills/uscha-adr-refine/SKILL.md +2 -0
  4. package/uscha-kit/.claude/skills/uscha-characterize/SKILL.md +2 -0
  5. package/uscha-kit/.claude/skills/uscha-devloop/SKILL.md +171 -15
  6. package/uscha-kit/.claude/skills/uscha-devloop/qa_ledger.py +2089 -97
  7. package/uscha-kit/.claude/skills/uscha-discovery/SKILL.md +58 -4
  8. package/uscha-kit/.claude/skills/uscha-mirador/SKILL.md +2 -0
  9. package/uscha-kit/.claude/skills/uscha-reverse-discovery/SKILL.md +2 -0
  10. package/uscha-kit/.claude/skills/uscha-rubric/SKILL.md +2 -0
  11. package/uscha-kit/.claude/skills/uscha-status/SKILL.md +40 -0
  12. package/uscha-kit/.claude/skills/uscha-sysdoc/SKILL.md +2 -0
  13. package/uscha-kit/.claude-plugin/plugin.json +2 -2
  14. package/uscha-kit/.codex-plugin/plugin.json +1 -1
  15. package/uscha-kit/README.md +249 -11
  16. package/uscha-kit/VERSION +1 -1
  17. package/uscha-kit/install-uscha.py +18 -2
  18. package/uscha-kit/skills/uscha-adr-refine/SKILL.md +2 -0
  19. package/uscha-kit/skills/uscha-characterize/SKILL.md +2 -0
  20. package/uscha-kit/skills/uscha-devloop/SKILL.md +171 -15
  21. package/uscha-kit/skills/uscha-devloop/qa_ledger.py +2089 -97
  22. package/uscha-kit/skills/uscha-discovery/SKILL.md +58 -4
  23. package/uscha-kit/skills/uscha-mirador/SKILL.md +2 -0
  24. package/uscha-kit/skills/uscha-reverse-discovery/SKILL.md +2 -0
  25. package/uscha-kit/skills/uscha-rubric/SKILL.md +2 -0
  26. package/uscha-kit/skills/uscha-status/SKILL.md +40 -0
  27. package/uscha-kit/skills/uscha-sysdoc/SKILL.md +2 -0
  28. package/uscha-kit/templates/CLAUDE.md +16 -0
  29. package/uscha-kit/templates/CONSTITUTION.md +55 -3
  30. package/uscha-kit/templates/docs/adr/README.md +15 -0
  31. package/uscha-kit/templates/scripts/smoke-report-example.json +24 -0
  32. package/uscha-kit/uscha.config.json +8 -2
@@ -36,6 +36,9 @@ Usage (see `--help` on each subcommand):
36
36
  [--ruff reports/ruff.json --mypy reports/mypy.txt]
37
37
  qa_ledger.py log-gate --repo backend-api --iteration 1 --kind golden-diff \
38
38
  --verdict pass|fail|not-run [--count N] [--note "..."]
39
+ qa_ledger.py corpus-run --repo backend-api --corpus corpus.jsonl \
40
+ --command "python -m parser" [--threshold 99] [--ac AC-FIELD-01]
41
+ qa_ledger.py smoke-ingest --repo backend-api --report reports/smoke.json [--json]
39
42
  qa_ledger.py flag-blocker --repo backend-api --kind constitution --note "INV-XX breached" \
40
43
  [--resolve]
41
44
  qa_ledger.py production-finding --repo backend-api --severity HIGH --title "..." --evidence "..."
@@ -44,11 +47,13 @@ Usage (see `--help` on each subcommand):
44
47
  qa_ledger.py rebuild --mode baseline --config uscha.config.json [--out REBUILD-BASELINE.json]
45
48
  qa_ledger.py rebuild --mode compare --baseline REBUILD-BASELINE.json [--json]
46
49
  qa_ledger.py simplicity-check --diff changes.diff [--config uscha.config.json] [--json]
47
- qa_ledger.py simplicity-check --from-git --base main
50
+ qa_ledger.py simplicity-check --from-git --base main (advisory: exit 0)
51
+ qa_ledger.py simplicity-check --from-git --base main --max-lines-added 400 --gate
48
52
  qa_ledger.py pit-check --report target/pit-reports/*/mutations.xml [--min-score 60] [--json]
49
53
  qa_ledger.py gate-check --from-git --base main [--strict] [--json]
50
54
  qa_ledger.py spec-check --spec SPEC.md [--spec ACCEPTANCE.md] [--strict] [--json]
51
55
  qa_ledger.py golden-diff [--dir .] [--labels golden-labels.json] [--json]
56
+ qa_ledger.py operability --repo backend-api [--json] (ADR-048: exit 0 always)
52
57
  """
53
58
 
54
59
  import argparse
@@ -1883,14 +1888,55 @@ GOLDEN_REQUIRED_CEILING = 49 # ADR-002: no approved golden -> NOT READY (does n
1883
1888
  RISK_PROFILES = {
1884
1889
  "A": {"qa_tools_order": ["code-review"]},
1885
1890
  "B": {"qa_tools_order": ["code-review", "improve"]},
1886
- "C": {"qa_tools_order": ["code-review", "judgment-day", "improve"]},
1891
+ "C": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
1892
+ "operability.gate": True},
1887
1893
  "D": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
1888
- "coverage_threshold": 70, "golden_required": True},
1894
+ "coverage_threshold": 70, "golden_required": True,
1895
+ "operability.gate": True},
1889
1896
  "E": {"qa_tools_order": ["code-review", "judgment-day", "improve"],
1890
- "coverage_threshold": 80, "golden_required": True},
1897
+ "coverage_threshold": 80, "golden_required": True,
1898
+ "operability.gate": True},
1891
1899
  }
1892
1900
 
1893
1901
 
1902
+ def _knob_get(mapping, key):
1903
+ """Read a knob out of a defaults-shaped mapping by its (possibly DOTTED) name ->
1904
+ (found, value). Dotted since 2.2.0 (ADR-048): the first profile-owned knob that lives one
1905
+ level down is `defaults.operability.gate`, in the same block shape the config already uses
1906
+ for `simplicity.gate` and `waste.gate`. A non-dict on the way down is NOT a hit --
1907
+ `operability: true` declares nothing this ladder can read, and guessing what it meant is
1908
+ exactly how a preset goes silently inert (the defect ADR-001 was amended for)."""
1909
+ cur = mapping
1910
+ for part in key.split("."):
1911
+ if not isinstance(cur, dict) or part not in cur:
1912
+ return False, None
1913
+ cur = cur[part]
1914
+ return True, cur
1915
+
1916
+
1917
+ def _knob_set(mapping, key, value):
1918
+ """Write a (possibly dotted) knob, creating the intermediate objects. Intermediates are
1919
+ COPIED on the way down: callers hand this shallow copies of `defaults`, and mutating a
1920
+ nested dict in place would reach back into the config the caller promised not to touch.
1921
+ Returns False -- writing nothing -- when something that is not an object already sits on
1922
+ the path: an explicit declaration always wins, even a malformed one, and naming a
1923
+ malformed one is `_validate_init_config`'s job, not this function's."""
1924
+ parts = key.split(".")
1925
+ cur = mapping
1926
+ for part in parts[:-1]:
1927
+ nxt = cur.get(part)
1928
+ if nxt is None:
1929
+ nxt = {}
1930
+ elif isinstance(nxt, dict):
1931
+ nxt = dict(nxt)
1932
+ else:
1933
+ return False
1934
+ cur[part] = nxt
1935
+ cur = nxt
1936
+ cur[parts[-1]] = value
1937
+ return True
1938
+
1939
+
1894
1940
  def _apply_risk_profile(defaults):
1895
1941
  """Expand defaults['risk_profile'] into concrete knobs UNDER any explicit value (explicit
1896
1942
  wins per key). Records the profile-provided keys in defaults['_risk_profile_keys'] so a cap
@@ -1904,8 +1950,9 @@ def _apply_risk_profile(defaults):
1904
1950
  + ", ".join(sorted(RISK_PROFILES)))
1905
1951
  provided = []
1906
1952
  for key, value in RISK_PROFILES[profile].items():
1907
- if key not in defaults: # an explicit declaration always wins
1908
- defaults[key] = list(value) if isinstance(value, list) else value
1953
+ if _knob_get(defaults, key)[0]: # an explicit declaration always wins
1954
+ continue
1955
+ if _knob_set(defaults, key, list(value) if isinstance(value, list) else value):
1909
1956
  provided.append(key)
1910
1957
  defaults["_risk_profile_keys"] = provided
1911
1958
  return defaults
@@ -1930,6 +1977,11 @@ ENGINE_DEFAULTS = {
1930
1977
  "qa_tools_order": None,
1931
1978
  "coverage_threshold": 60,
1932
1979
  "golden_required": False,
1980
+ # ADR-048. The engine's own posture on operability is ADVISORY: `operability` measures and
1981
+ # records on every profile, but only C/D/E turn the record into a gate. A kit that gated
1982
+ # release + reset + RUNBOOK on every project would be the kit's opinion wearing an exit
1983
+ # code -- the same thing ADR-043 refused for the simplicity budget.
1984
+ "operability.gate": False,
1933
1985
  }
1934
1986
  ENGINE_DEFAULT_NOTES = {
1935
1987
  "qa_tools_order": "not declared - convergence uses a window of --tools-per-cycle "
@@ -1940,17 +1992,35 @@ ENGINE_DEFAULT_NOTES = {
1940
1992
  def _resolved_defaults(cfg):
1941
1993
  """(resolved defaults, origin per profile-owned key) for a RAW config -- the effective
1942
1994
  settings `doctor` reports. Origin is `override` (declared in defaults), `profile <X>`, or
1943
- `default`. Read-only: never mutates cfg and never writes anything back."""
1995
+ `default`. Read-only: never mutates cfg and never writes anything back.
1996
+
1997
+ Works over a RAW config (what `doctor` reads) and over the FROZEN one inside a ledger
1998
+ alike: `init` expands the profile before freezing, so on the frozen copy a profile-supplied
1999
+ knob is already sitting in `defaults` and would otherwise read as a human `override`.
2000
+ `_risk_profile_keys` -- written by `_apply_risk_profile` for exactly this reason, and
2001
+ already read this way by the golden cap -- is what tells the two apart (ADR-048).
2002
+
2003
+ It tells them apart only while the value still IS the profile's, and that is the case the
2004
+ fresh review found: a human who edits the FROZEN copy leaves a declaration the profile
2005
+ never made, `_risk_profile_keys` still names the key, and `operability.gate: false` under
2006
+ profile E reported `origin profile E` -- crediting the preset with the opposite of what it
2007
+ supplies. The DECLARATION is therefore read first: a raw value that disagrees with what the
2008
+ profile would have written is an `override`, whoever typed it and whenever."""
1944
2009
  raw = cfg.get("defaults") if isinstance(cfg, dict) else None
1945
2010
  raw = dict(raw) if isinstance(raw, dict) else {}
1946
- declared = set(raw)
1947
2011
  profile = raw.get("risk_profile")
2012
+ from_profile = set(raw.get("_risk_profile_keys") or [])
1948
2013
  expanded = _apply_risk_profile(dict(raw))
1949
2014
  resolved, origin = {}, {}
1950
2015
  for key, fallback in ENGINE_DEFAULTS.items():
1951
- resolved[key] = expanded[key] if key in expanded else fallback
1952
- if key in declared:
2016
+ found, value = _knob_get(expanded, key)
2017
+ resolved[key] = value if found else fallback
2018
+ raw_found, raw_value = _knob_get(raw, key)
2019
+ profile_value = RISK_PROFILES.get(profile, {}).get(key) if profile else None
2020
+ if raw_found and not (key in from_profile and raw_value == profile_value):
1953
2021
  origin[key] = "override"
2022
+ elif key in from_profile and profile:
2023
+ origin[key] = "profile %s" % profile
1954
2024
  elif profile and key in RISK_PROFILES.get(profile, {}):
1955
2025
  origin[key] = "profile %s" % profile
1956
2026
  else:
@@ -2107,7 +2177,96 @@ def _validate_log_step_counts(args):
2107
2177
  raise SystemExit("[qa_ledger] gated-reported cannot exceed reported")
2108
2178
 
2109
2179
 
2180
+ def _add_repo_to_config_file(path, entry, frozen_names):
2181
+ """Mirror the new repo into uscha.config.json when that file IS the source the ledger was
2182
+ frozen from -- same repo names, in the same order. A file that has DRIFTED from the frozen
2183
+ config is somebody else's edit, and it is left alone with the divergence named: silently
2184
+ rewriting a config that no longer matches the ledger would be resolving a conflict the human
2185
+ has not seen. Returns the sentence to print, never raises: the ledger is the source of truth
2186
+ and its write already happened."""
2187
+ if not os.path.isfile(path):
2188
+ return "%s not found -- the ledger's frozen config is the only copy that changed" % path
2189
+ try:
2190
+ with open(path, "r", encoding="utf-8") as fh:
2191
+ cfg = json.load(fh)
2192
+ names = [r.get("name") for r in cfg.get("repos", [])]
2193
+ except (OSError, ValueError, AttributeError, TypeError) as exc:
2194
+ return "%s left untouched (unreadable: %s)" % (path, exc)
2195
+ if names != frozen_names:
2196
+ return ("%s left untouched: its repos %s differ from the ledger's frozen %s, so it is "
2197
+ "not the source this ledger was built from" % (path, names, frozen_names))
2198
+ cfg["repos"].append(dict(entry))
2199
+ tmp = path + ".tmp"
2200
+ try:
2201
+ with open(tmp, "w", encoding="utf-8") as fh:
2202
+ json.dump(cfg, fh, indent=2, ensure_ascii=False)
2203
+ fh.write("\n")
2204
+ os.replace(tmp, path)
2205
+ except OSError as exc:
2206
+ return "%s left untouched (not writable: %s)" % (path, exc)
2207
+ return "%s updated too (it is the source this ledger was frozen from)" % path
2208
+
2209
+
2210
+ def _init_add_repo(args):
2211
+ """Append ONE repo to an EXISTING ledger without resetting it (2.2.0 field fix).
2212
+
2213
+ Until now the only door was re-running `init --config`, which builds a NEW ledger: the step
2214
+ counter, every repo's iterations and every snapshot went back to zero, so adding a second
2215
+ service to a live loop cost the evidence of the first. Editing QA-LEDGER.json by hand is not
2216
+ the workaround either -- `_load` verifies the sha256 that `_save` writes, so a hand edit
2217
+ turns the file into a refusal. That refusal is correct and stays: it is what makes 'measured
2218
+ beats narrated' worth anything. What was missing was a supported door, and this is it.
2219
+
2220
+ What this deliberately does NOT do: touch any existing repo's steps, snapshots or
2221
+ iterations, and re-freeze `defaults`. The config a project started under stays the config it
2222
+ ran under -- only the repo list grows. `_save` re-seals the checksum over the result.
2223
+
2224
+ The new repo starts with NO evidence, and readiness says so rather than hiding it: it enters
2225
+ `facts.static_unmeasured_repos`, and every aggregate that averages over repos reads LOWER
2226
+ until its first snapshot or gate lands. That is an unmeasured repo, never a regression --
2227
+ nothing already measured changed, and `by_repo` proves it entry by entry. Excluding it from
2228
+ the average instead would be the opposite mistake: an aggregate produced by silence."""
2229
+ name = args.add_repo
2230
+ if not _has_text(args.path) or not _has_text(args.type):
2231
+ raise SystemExit("[qa_ledger] init --add-repo needs --path and --type: the engine never "
2232
+ "guesses where a repo lives or how it is built")
2233
+ if name == "integration":
2234
+ raise SystemExit("[qa_ledger] repo name 'integration' is reserved")
2235
+ ledger = _load(args.out)
2236
+ cfg = ledger.get("config") or {}
2237
+ frozen_names = [r.get("name") for r in cfg.get("repos", [])]
2238
+ if name in frozen_names or name in ledger.get("repos", {}):
2239
+ raise SystemExit(
2240
+ "[qa_ledger] repo '%s' already exists in %s -- nothing was written. Adding it twice "
2241
+ "would either duplicate the config entry or reset that repo's steps, which is the "
2242
+ "very loss this flag exists to prevent." % (name, args.out))
2243
+ entry = {"name": name, "type": args.type, "path": args.path}
2244
+ if _has_text(args.test_command):
2245
+ entry["test_command"] = args.test_command
2246
+ cfg.setdefault("repos", []).append(entry)
2247
+ ledger["config"] = cfg
2248
+ # the same node shape `init` writes, so nothing downstream can tell an appended repo from
2249
+ # one that was there since the first run.
2250
+ ledger["repos"][name] = {"type": args.type, "path": args.path,
2251
+ "snapshots": [], "iterations": []}
2252
+ _save(args.out, ledger)
2253
+ cfg_path = args.config or os.path.join(
2254
+ os.path.dirname(os.path.abspath(args.out)), "uscha.config.json")
2255
+ print("[qa_ledger] %s: repo '%s' added (%s at %s) -- %d repos, every existing repo's steps "
2256
+ "untouched, checksum re-sealed"
2257
+ % (args.out, name, args.type, args.path, len(ledger["repos"])))
2258
+ print("[qa_ledger] %s" % _add_repo_to_config_file(cfg_path, entry, frozen_names))
2259
+ print("[qa_ledger] '%s' starts with NO evidence: it reads as UNMEASURED (readiness lists it "
2260
+ "under facts.static_unmeasured_repos) and every repo average reads lower until its "
2261
+ "first snapshot or gate lands. That is an unmeasured repo, not a regression." % name)
2262
+
2263
+
2110
2264
  def cmd_init(args):
2265
+ if getattr(args, "add_repo", None):
2266
+ return _init_add_repo(args)
2267
+ if not _has_text(getattr(args, "config", None)):
2268
+ raise SystemExit("[qa_ledger] init needs --config to create a ledger, or --add-repo "
2269
+ "NAME (with --path and --type) to append one repo to an existing one")
2111
2270
  cfg = _load(args.config, what="config", flag="--config")
2112
2271
  _validate_init_config(cfg)
2113
2272
  defaults = cfg.get("defaults", {})
@@ -2497,7 +2656,16 @@ def _gate_rollup(ledger):
2497
2656
  for tool, rec in _latest_static_by_tool(rnode).items():
2498
2657
  gates.append({"repo": rname, "tool": tool,
2499
2658
  "blocking": rec.get("gated_reported", 0) > 0,
2659
+ # ADR-043: an advisory record is non-blocking BY CONSTRUCTION, which
2660
+ # is not the same fact as a gate that ran and came back clean. It
2661
+ # travels so no consumer has to guess which of the two it is looking
2662
+ # at -- the false-clean is the failure mode, not the absence.
2663
+ "advisory": bool(rec.get("advisory")),
2500
2664
  "gated": rec.get("gated_reported", 0),
2665
+ # WHERE it was measured (2.2.0, log-gate --ref): a CI verdict without
2666
+ # its run is a claim, and the rollup is where a reader looks first.
2667
+ # Absent on every record that carries none -- never invented.
2668
+ **({"ref": rec["ref"]} if rec.get("ref") else {}),
2501
2669
  "note": rec.get("note")})
2502
2670
  return sorted(gates, key=lambda g: (g["repo"], g["tool"]))
2503
2671
 
@@ -2724,11 +2892,18 @@ def cmd_spec_change_request(args):
2724
2892
  (row["id"], args.repo, args.source, args.requested_change))
2725
2893
 
2726
2894
 
2727
- def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, note):
2895
+ def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, note,
2896
+ advisory=False, ref=None):
2728
2897
  """Append a static-gate-shaped record for a FACT gate so the EXISTING plumbing
2729
2898
  sees it: _gate_open_and_sev feeds the BLOCKER/CRITICAL readiness cap (<=65) and
2730
2899
  _converged refuses while the latest record for the tool is failing. A later
2731
- clean record for the same tool clears it (latest-per-tool wins)."""
2900
+ clean record for the same tool clears it (latest-per-tool wins).
2901
+
2902
+ advisory=True (kit 2.1.0, ADR-043) records a MEASUREMENT that is not a gate: the record
2903
+ carries zero gated findings, so it can neither cap readiness nor block convergence, and it
2904
+ is stamped so no surface can render it as a clean gate either. That distinction is the whole
2905
+ point -- a check running advisory is not the same fact as a check running green, and a
2906
+ ledger that cannot tell them apart is the false-clean this flag exists to refuse."""
2732
2907
  ledger["step_counter"] += 1
2733
2908
  n = max(1, count) if failing else 0
2734
2909
  rec = {
@@ -2740,24 +2915,70 @@ def _append_gate_record(ledger, node, repo, tool, iteration, failing, count, not
2740
2915
  "tests_passed": None, "files_changed": 0,
2741
2916
  "fingerprint": None, "finding_ids": None, "note": note,
2742
2917
  }
2918
+ if advisory:
2919
+ rec["advisory"] = True
2920
+ if ref:
2921
+ # WHERE the verdict was measured (2.2.0). A CI verdict typed by hand is a claim; the
2922
+ # run id or URL beside it is the receipt, and the ledger is the only place it survives
2923
+ # the conversation that produced it.
2924
+ rec["ref"] = ref
2743
2925
  node["iterations"].append(rec)
2744
2926
  ledger["steps"].append({"n": rec["n"], "at": rec["at"], "kind": "static-gate",
2745
2927
  "repo": repo, "tool": tool, "iteration": iteration})
2746
2928
  return rec
2747
2929
 
2748
2930
 
2931
+ # The only --kind values log-gate accepts with --verdict advisory (ADR-043): the checks whose
2932
+ # default mode IS advisory. Every other kind is a FACT gate and records pass/fail/not-run only.
2933
+ #
2934
+ # `corpus` (2.2.0, ADR-046) is admitted because `corpus-run` ITSELF runs advisory whenever no
2935
+ # threshold is declared -- the percentage is measured against no adopted budget, so it gates
2936
+ # nothing. Refusing the verdict on this door while the check emits it on the other would give
2937
+ # one fact two incompatible records: the parity door could only spell an unbudgeted run as
2938
+ # `pass`, which is the false clean ADR-043 exists to refuse. Admitting it costs nothing that
2939
+ # matters: with a threshold declared, `corpus` is a FACT gate like any other.
2940
+ #
2941
+ # `operability` (ADR-048) joins them because its posture is PROFILE-DEPENDENT: on A, B or no
2942
+ # profile the four checks are measured and recorded advisory, and only `defaults.operability.gate`
2943
+ # (which C, D and E own) turns the same measurement into a gate. It is a FACT either way -- a
2944
+ # workflow file exists or it does not -- so admitting it here widens WHEN it gates, never WHAT
2945
+ # counts as evidence, which is the line INV-ADVISORY-01 draws.
2946
+ ADVISORY_CAPABLE_KINDS = ("simplicity", "waste", "corpus", "operability")
2947
+
2948
+
2749
2949
  def cmd_log_gate(args):
2750
- """Persist a FACT-gate verdict (golden-diff / gate-check / pit-check / simplicity / regression)
2751
- into the ledger, so 'facts may block' is enforced by the engine, not by goodwill.
2752
- fail -> BLOCKER record: trips the <=65 readiness cap AND blocks convergence.
2753
- pass -> clean record for the same tool: credits the fix, convergence sees clean.
2754
- not-run -> a steps event ONLY, never an iterations record: absence is not
2755
- evidence it neither reads as clean nor fakes a red (last state stands).
2950
+ """Persist a FACT-gate verdict (golden-diff / gate-check / pit-check / simplicity /
2951
+ regression / ci / smoke) into the ledger, so 'facts may block' is enforced by the engine, not by
2952
+ goodwill.
2953
+ fail -> BLOCKER record: trips the <=65 readiness cap AND blocks convergence.
2954
+ pass -> clean record for the same tool: credits the fix, convergence sees clean.
2955
+ advisory -> a MEASURED, non-gating record (kit 2.1.0, ADR-043): zero gated findings, so
2956
+ it can never cap readiness nor block convergence, and stamped `advisory` so
2957
+ no surface counts it as an `ok` gate. Use it for a check the project has not
2958
+ declared as a gate -- `simplicity-check` in its default advisory mode above
2959
+ all. The alternative (persisting an advisory as `pass`) is the false clean
2960
+ this verdict exists to refuse: a reader cannot tell "the gate was green"
2961
+ from "there was no gate", and the second is what actually happened.
2962
+ not-run -> a steps event ONLY, never an iterations record: absence is not
2963
+ evidence — it neither reads as clean nor fakes a red (last state stands).
2964
+
2965
+ The engine cannot observe which mode a SEPARATE `simplicity-check` process ran in, so this
2966
+ is a named verdict rather than a refusal: refusing would only be enforceable on trust,
2967
+ while a third verdict is enforceable on the ledger. The caller declares the mode; every
2968
+ reader downstream then sees it as a fact instead of inferring it.
2756
2969
  """
2757
2970
  # INV-ADVISORY-01 note (ADR-014): --kind is a CLOSED vocabulary (argparse choices), so
2758
2971
  # an advisory-class dimension (e.g. "semantic") cannot be registered as a gate through
2759
2972
  # this door at all -- the refusal is structural. The smoke suite measures that the
2760
2973
  # vocabulary stays closed; widening it to admit an advisory kind is a red build.
2974
+ #
2975
+ # `ci` and `smoke` (2.2.0) are the additions since ADR-014, and both are FACTS: a pipeline
2976
+ # either went green on a commit or it did not, and the engine can be TOLD that fact with
2977
+ # the run id or URL beside it (--ref); a smoke check either answered or it did not
2978
+ # (ADR-047), which is why neither may run advisory. They are admitted here because they
2979
+ # are measurable, not because they are useful -- an LLM judgment does not become a gate by
2980
+ # being important. Adding a FACT kind is DECLARED in the CONSTITUTION template, which is
2981
+ # where the closed vocabulary is stated to the project rather than only to this parser.
2761
2982
  ledger = _load(args.ledger)
2762
2983
  node = _repo_node(ledger, args.repo)
2763
2984
  tool = f"gate:{args.kind}"
@@ -2774,12 +2995,575 @@ def cmd_log_gate(args):
2774
2995
  f"(no evidence — last logged state stands, absence is never green)")
2775
2996
  return
2776
2997
  failing = args.verdict == "fail"
2998
+ advisory = args.verdict == "advisory"
2999
+ if advisory and args.kind not in ADVISORY_CAPABLE_KINDS:
3000
+ # ADR-043 widens --verdict for the checks that RUN advisory by default. A FACT gate
3001
+ # (deleted tests, a lowered threshold, a golden drift) recorded as advisory would be
3002
+ # a mandatory gate cleared by goodwill -- the exact thing this ledger exists to refuse.
3003
+ print(f"[qa_ledger] log-gate: --verdict advisory is not accepted for --kind {args.kind}: "
3004
+ f"only {', '.join(ADVISORY_CAPABLE_KINDS)} run in an advisory mode; a FACT gate "
3005
+ f"records pass, fail or not-run", file=sys.stderr)
3006
+ sys.exit(2)
2777
3007
  rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
2778
- failing, args.count, args.note)
3008
+ failing, args.count, args.note, advisory=advisory,
3009
+ ref=getattr(args, "ref", None))
2779
3010
  _save(args.ledger, ledger)
2780
- state = f"FAIL (BLOCKER x{rec['gated_reported']})" if failing else "PASS (clean)"
2781
- print(f"[qa_ledger] {args.repo}/{tool}: {state} logged "
2782
- f"{'caps readiness <=65 and blocks convergence' if failing else 'clears the gate for convergence'}")
3011
+ if advisory:
3012
+ state, effect = "ADVISORY (measured, not gating)", (
3013
+ "reported everywhere as advisory, never as ok; caps nothing, blocks nothing")
3014
+ elif failing:
3015
+ state, effect = (f"FAIL (BLOCKER x{rec['gated_reported']})",
3016
+ "caps readiness <=65 and blocks convergence")
3017
+ else:
3018
+ state, effect = "PASS (clean)", "clears the gate for convergence"
3019
+ print(f"[qa_ledger] {args.repo}/{tool}: {state} logged — {effect}"
3020
+ + (f" [ref {rec['ref']}]" if rec.get("ref") else ""))
3021
+
3022
+
3023
+ # --------------------------------------------------------------------------- #
3024
+ # corpus-run (kit 2.2.0, ADR-046): FIELD TRUTH for greenfield work.
3025
+ #
3026
+ # `characterize`/`golden-diff` answer the brownfield question -- "does the new code still do
3027
+ # what the OLD code did?" -- and they have no answer at all for a system that never had an old
3028
+ # code. In a greenfield project every test payload was INVENTED by the agent that wrote the
3029
+ # code, so a suite can be green over inputs the world never produces. The field report this
3030
+ # subcommand comes from is exactly that shape: a parser passed every test its author wrote and
3031
+ # was wrong; running the REAL corpus moved it from 96.96 % to 99.645 %.
3032
+ #
3033
+ # So the corpus is the missing evidence class: real inputs with their real expected outputs,
3034
+ # run through the real command, scored as a percentage the ledger persists like any other FACT.
3035
+ # It is deterministic (stable case order, per-case timeout) and it refuses rather than guessing:
3036
+ # a corpus that is missing, empty or malformed is exit 2 naming the line, never a silent 0 %.
3037
+ #
3038
+ # The 2.1.0 posture holds (ADR-043): a gate needs an ADOPTED budget. With no threshold declared
3039
+ # anywhere the run is ADVISORY -- the percentage is measured and persisted, and it gates nothing.
3040
+ # The WEIGHT of field truth in the readiness score is deliberately NOT here: adding a `field`
3041
+ # dimension moves every existing project's number, and that is its own ADR.
3042
+ # --------------------------------------------------------------------------- #
3043
+ CORPUS_DEFAULT_TIMEOUT = 30
3044
+ CORPUS_DEFAULT_MAX_MISSES = 5
3045
+
3046
+
3047
+ def _corpus_refuse(msg):
3048
+ """Every corpus refusal is exit 2 and NAMES what it could not read. The failure mode this
3049
+ exists to prevent is a corpus the runner could not parse being scored as 0 % -- an
3050
+ unmeasurable input reported as a measured catastrophe (or, with the arithmetic the other
3051
+ way, as a clean 100 % over zero cases)."""
3052
+ print("[qa_ledger] corpus-run: " + msg, file=sys.stderr)
3053
+ sys.exit(2)
3054
+
3055
+
3056
+ def _corpus_cases(path):
3057
+ """Read a JSONL corpus into an ORDERED list of cases: file order IS run order, so two runs
3058
+ over the same file report the same misses in the same places. One JSON object per line,
3059
+ `input` and `expected` required, `id` optional (a positional id is derived when absent)."""
3060
+ if not os.path.isfile(path):
3061
+ _corpus_refuse("corpus not found: %s -- a corpus that is not there is UNMEASURED, "
3062
+ "never a measured 0 %%" % path)
3063
+ try:
3064
+ with open(path, encoding="utf-8") as fh:
3065
+ raw = fh.read().splitlines()
3066
+ except OSError as exc:
3067
+ _corpus_refuse("corpus unreadable: %s" % exc)
3068
+ cases = []
3069
+ for n, line in enumerate(raw, 1):
3070
+ if not line.strip():
3071
+ continue
3072
+ try:
3073
+ row = json.loads(line)
3074
+ except ValueError as exc:
3075
+ _corpus_refuse("%s line %d is not valid JSON (%s) -- a malformed corpus is "
3076
+ "refused, never scored" % (path, n, exc))
3077
+ if not isinstance(row, dict):
3078
+ _corpus_refuse("%s line %d is a %s, not a JSON object with `input` and `expected`"
3079
+ % (path, n, type(row).__name__))
3080
+ missing = [k for k in ("input", "expected") if k not in row]
3081
+ if missing:
3082
+ _corpus_refuse("%s line %d has no %s key" % (path, n, " and no ".join(missing)))
3083
+ cases.append({"id": str(row["id"]) if row.get("id") is not None
3084
+ else "case-%03d" % (len(cases) + 1),
3085
+ "input": row["input"], "expected": row["expected"]})
3086
+ if not cases:
3087
+ _corpus_refuse("%s holds 0 cases -- an empty corpus is refused: 0/0 is not 100 %%, "
3088
+ "and it is not 0 %% either" % path)
3089
+ return cases
3090
+
3091
+
3092
+ def _corpus_text(value):
3093
+ """A JSON scalar string travels as itself; anything else travels as its JSON encoding.
3094
+ sort_keys so an object payload is byte-stable across runs."""
3095
+ return value if isinstance(value, str) else json.dumps(value, ensure_ascii=False,
3096
+ sort_keys=True)
3097
+
3098
+
3099
+ def _corpus_match(actual, expected):
3100
+ """Trimmed string compare first, then JSON-equal when BOTH sides parse as JSON -- so a
3101
+ command that reorders an object's keys or prints 1.0 for 1 is not a false miss, while a
3102
+ command that prints prose is compared as prose."""
3103
+ exp, act = _corpus_text(expected).strip(), actual.strip()
3104
+ if act == exp:
3105
+ return True
3106
+ try:
3107
+ return json.loads(act) == json.loads(exp)
3108
+ except ValueError:
3109
+ return False
3110
+
3111
+
3112
+ def _corpus_case(command, case, timeout):
3113
+ """Run ONE case: the input on stdin, the trimmed stdout compared to `expected`.
3114
+ Returns (hit, reason, actual). A non-zero exit is a miss (the command did not answer);
3115
+ a case that outruns the timeout is a miss NAMED `timeout`, never an ambiguous hang."""
3116
+ try:
3117
+ p = subprocess.run(command, shell=True, input=_corpus_text(case["input"]),
3118
+ stdout=subprocess.PIPE, stderr=subprocess.PIPE, text=True,
3119
+ encoding="utf-8", errors="replace", timeout=timeout)
3120
+ except subprocess.TimeoutExpired:
3121
+ return False, "timeout", ""
3122
+ except OSError as exc:
3123
+ return False, "command failed to start: %s" % exc, ""
3124
+ actual = (p.stdout or "").strip()
3125
+ if p.returncode != 0:
3126
+ return False, "exit %d" % p.returncode, actual
3127
+ if _corpus_match(actual, case["expected"]):
3128
+ return True, None, actual
3129
+ return False, "mismatch", actual
3130
+
3131
+
3132
+ def _corpus_num(value):
3133
+ """90.0 reads back as 90 and 99.5 as 99.5 -- the declared budget renders as the human
3134
+ typed it, in the ledger and on every surface that reads the record."""
3135
+ if value is None:
3136
+ return None
3137
+ return int(value) if float(value).is_integer() else float(value)
3138
+
3139
+
3140
+ def _corpus_clip(text, n=120):
3141
+ text = str(text).replace("\n", "\\n")
3142
+ return text if len(text) <= n else text[:n] + "..."
3143
+
3144
+
3145
+ def cmd_corpus_run(args):
3146
+ """Run a REAL-INPUT corpus against a command and persist the result as `gate:corpus`.
3147
+
3148
+ The threshold is the project's, never the kit's: --threshold, else
3149
+ repos[R].corpus_threshold, else defaults.corpus_threshold. With none of the three the run
3150
+ is ADVISORY -- measured, persisted, and gating nothing (ADR-043: a gate needs an adopted
3151
+ budget). With one, `pass` is a clean gate and `fail` is a BLOCKER through the SAME record
3152
+ shape gate-check uses: readiness capped <=65, convergence blocked, and a later green run
3153
+ clears it (latest-per-tool wins).
3154
+
3155
+ --ac stamps criterion ids on the record. A criterion whose only evidence is a corpus record
3156
+ closes MEASURED iff that record passed -- the same rule a green JUnit testcase already
3157
+ obeys, and for the same reason: red or unbudgeted evidence closes nothing."""
3158
+ ledger = _load(args.ledger)
3159
+ node = _repo_node(ledger, args.repo)
3160
+ tool = "gate:corpus"
3161
+ _validate_iteration(node, tool, args.iteration)
3162
+ cfg = next((r for r in ledger["config"].get("repos", [])
3163
+ if r.get("name") == args.repo), {})
3164
+ defaults = ledger["config"].get("defaults", {})
3165
+
3166
+ corpus_path = args.corpus or cfg.get("corpus")
3167
+ if not corpus_path:
3168
+ _corpus_refuse("no corpus: pass --corpus PATH or declare repos[%s].corpus in the "
3169
+ "config. Where the real inputs live is the project's fact, not a "
3170
+ "default this engine may invent." % args.repo)
3171
+ cases = _corpus_cases(corpus_path)
3172
+
3173
+ threshold, source = args.threshold, "--threshold"
3174
+ if threshold is None:
3175
+ threshold, source = cfg.get("corpus_threshold"), "repos[%s].corpus_threshold" % args.repo
3176
+ if threshold is None:
3177
+ threshold, source = defaults.get("corpus_threshold"), "defaults.corpus_threshold"
3178
+ if threshold is None:
3179
+ source = None
3180
+ threshold = _corpus_num(threshold)
3181
+
3182
+ acs = []
3183
+ for raw in (args.ac or []):
3184
+ cid, _end = _ac_id_of(raw.strip())
3185
+ if cid is None:
3186
+ _corpus_refuse("--ac %r is not a criterion id ('AC-01' or 'AC-FIELD-01')" % raw)
3187
+ if cid not in acs:
3188
+ acs.append(cid)
3189
+
3190
+ hits, misses = 0, []
3191
+ for case in cases:
3192
+ ok, reason, actual = _corpus_case(args.command, case, args.timeout)
3193
+ if ok:
3194
+ hits += 1
3195
+ else:
3196
+ misses.append({"id": case["id"], "reason": reason,
3197
+ "expected": _corpus_clip(_corpus_text(case["expected"])),
3198
+ "actual": _corpus_clip(actual)})
3199
+ total = len(cases)
3200
+ pct = round(100.0 * hits / total, 1)
3201
+
3202
+ advisory = source is None
3203
+ failing = (not advisory) and pct < threshold
3204
+ if advisory:
3205
+ note = "corpus %s: %d/%d (%s %%), no threshold declared" % (
3206
+ os.path.basename(corpus_path), hits, total, pct)
3207
+ else:
3208
+ note = "corpus %s: %d/%d (%s %%) %s %s %%" % (
3209
+ os.path.basename(corpus_path), hits, total, pct,
3210
+ "<" if failing else ">=", threshold)
3211
+ rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
3212
+ failing, len(misses), note, advisory=advisory)
3213
+ # the MEASUREMENT travels on the record, not only its verdict: readiness reads the
3214
+ # percentage back for the field line, and a reader six months later can see WHAT was
3215
+ # measured against WHICH budget instead of a bare pass.
3216
+ rec["corpus"] = {"path": corpus_path.replace("\\", "/"), "hits": hits, "total": total,
3217
+ "percent": pct, "threshold": threshold, "threshold_source": source,
3218
+ "timeout_s": args.timeout,
3219
+ "misses": misses[:max(0, args.max_misses)]}
3220
+ if acs:
3221
+ rec["ac"] = acs
3222
+ _save(args.ledger, ledger)
3223
+
3224
+ state = ("ADVISORY (measured, not gating)" if advisory
3225
+ else "FAIL" if failing else "PASS")
3226
+ out = {"repo": args.repo, "tool": tool, "verdict": state.split()[0].lower(),
3227
+ "advisory": advisory, "corpus": rec["corpus"], "ac": acs, "note": note}
3228
+ if args.json:
3229
+ print(json.dumps(out, indent=2, ensure_ascii=False))
3230
+ else:
3231
+ if advisory:
3232
+ budget = ("no threshold declared — the percentage is recorded and nothing gates "
3233
+ "(a gate needs an adopted budget, ADR-043)")
3234
+ else:
3235
+ budget = "threshold %s %% from %s" % (threshold, source)
3236
+ print("[qa_ledger] %s/%s: %s %s %% (%d/%d) — %s"
3237
+ % (args.repo, tool, state, pct, hits, total, budget))
3238
+ for m in rec["corpus"]["misses"]:
3239
+ print(" miss %s (%s): expected %r, got %r"
3240
+ % (m["id"], m["reason"], m["expected"], m["actual"]))
3241
+ if len(misses) > len(rec["corpus"]["misses"]):
3242
+ print(" ... +%d more miss(es) not listed (--max-misses)"
3243
+ % (len(misses) - len(rec["corpus"]["misses"])))
3244
+ if failing:
3245
+ print(" caps readiness <=65 and blocks convergence until a green run")
3246
+ sys.exit(1 if failing else 0)
3247
+
3248
+
3249
+ def _corpus_records(ledger):
3250
+ """The LATEST corpus record per repo -- the same latest-per-tool rule the gate rollup and
3251
+ convergence already use, so the three cannot disagree about which run is current. A record
3252
+ logged through the `log-gate --kind corpus` parity door carries no `corpus` block and is
3253
+ deliberately not here: it gates (it is a gate record like any other), but it has no run to
3254
+ read back, and rendering it anyway printed the honest-looking nonsense
3255
+ `corpus None % (None/None) >= None % PASS`. Same exclusion, same reason, as
3256
+ `_smoke_records`."""
3257
+ out = {}
3258
+ for rname, rnode in ledger.get("repos", {}).items():
3259
+ rec = _latest_static_by_tool(rnode).get("gate:corpus")
3260
+ if rec is not None and rec.get("corpus"):
3261
+ out[rname] = rec
3262
+ return out
3263
+
3264
+
3265
+ def _corpus_ac_verdicts(ledger):
3266
+ """(closed, vetoed) criterion ids from the latest corpus record of every repo (ADR-046).
3267
+
3268
+ A corpus record CLOSES an AC iff it PASSED. An **advisory** run measured a percentage
3269
+ against no adopted budget: it is not a green gate (ADR-043 refuses to let it read as one)
3270
+ and it is not red evidence either, so it neither closes nor vetoes. A **failing** run is
3271
+ evidence AGAINST, which ADR-046 said from the start -- and the engine used to merely skip
3272
+ it, so a green testcase went on closing a criterion the field had just refuted. It now
3273
+ VETOES the ids it carries, the same rule a red JUnit testcase and a failed tagged smoke
3274
+ check already obey, and for the same reason: the cheapest way to fake a closed criterion is
3275
+ to put a green beside a red."""
3276
+ closed, vetoed = set(), set()
3277
+ for rec in _corpus_records(ledger).values():
3278
+ if rec.get("advisory"):
3279
+ continue
3280
+ if (rec.get("gated_reported") or 0) > 0:
3281
+ vetoed.update(rec.get("ac") or [])
3282
+ else:
3283
+ closed.update(rec.get("ac") or [])
3284
+ return closed - vetoed, vetoed
3285
+
3286
+
3287
+ def _corpus_field(ledger):
3288
+ """Per-repo FIELD readout -- ONE derivation, read by both the readiness text and its JSON.
3289
+ CONDITIONAL like lifecycle and agent-origin: a repo that neither declares a corpus nor ever
3290
+ ran one is absent from it, so a project without field truth prints and emits exactly what
3291
+ it printed and emitted before this existed. No weight, no cap, no gate of its own -- a
3292
+ failing corpus already blocks through its `gate:corpus` record, and the `field` DIMENSION
3293
+ (its own ADR) is deliberately not here."""
3294
+ recs = _corpus_records(ledger)
3295
+ out = {}
3296
+ for rname in ledger.get("repos", {}):
3297
+ cfg = next((r for r in ledger.get("config", {}).get("repos", [])
3298
+ if r.get("name") == rname), {})
3299
+ rec, declared = recs.get(rname), cfg.get("corpus")
3300
+ if rec is None and not declared:
3301
+ continue
3302
+ if rec is None:
3303
+ out[rname] = {"state": "UNMEASURED", "corpus": declared, "percent": None,
3304
+ "hits": None, "total": None, "threshold": None,
3305
+ "reason": "a corpus is declared and was never run"}
3306
+ continue
3307
+ c = rec.get("corpus") or {}
3308
+ out[rname] = {"state": ("ADVISORY" if rec.get("advisory")
3309
+ else "FAIL" if (rec.get("gated_reported") or 0) > 0
3310
+ else "PASS"),
3311
+ "corpus": c.get("path", declared),
3312
+ "percent": c.get("percent"), "hits": c.get("hits"),
3313
+ "total": c.get("total"), "threshold": c.get("threshold"),
3314
+ "threshold_source": c.get("threshold_source"),
3315
+ "ac": rec.get("ac") or []}
3316
+ return out
3317
+
3318
+
3319
+ def _corpus_field_line(rname, f):
3320
+ """The one-line rendering of a repo's field state. Text lives beside the derivation so the
3321
+ JSON and the human readout can never drift apart."""
3322
+ if f["state"] == "UNMEASURED":
3323
+ return ("--- field %s: corpus UNMEASURED — %s (%s)"
3324
+ % (rname, f["reason"], f["corpus"]))
3325
+ head = "--- field %s: corpus %s %% (%s/%s)" % (rname, f["percent"], f["hits"], f["total"])
3326
+ if f["state"] == "ADVISORY":
3327
+ return head + " — no threshold declared, ADVISORY (measured, not gating)"
3328
+ return "%s %s %s %% %s" % (head, "<" if f["state"] == "FAIL" else ">=",
3329
+ f["threshold"], f["state"])
3330
+
3331
+
3332
+ # --------------------------------------------------------------------------- #
3333
+ # smoke-ingest (kit 2.2.0, ADR-047): the SMOKE RUN as measured evidence.
3334
+ #
3335
+ # The ledger already ingests the evidence a machine produces on its own -- JUnit, coverage,
3336
+ # linters, a static gate's XML, a CI verdict. The smoke list was the hole: "the jar served
3337
+ # /admin", "the simulator answered 200 in 6 ms" arrived as a sub-agent's NARRATION, believed
3338
+ # because it was written confidently. The field report this subcommand comes from is exactly
3339
+ # that shape: every simulator run returned an empty list because the database had no rows, and
3340
+ # a smoke narrated as "verified" would have hidden it behind a sentence.
3341
+ #
3342
+ # So the smoke run stops being prose and becomes a REPORT the project's own tool writes --
3343
+ # {"checks": [{"name", "ok", "status"?, "latency_ms"?, "evidence"?}, ...]} -- and the engine
3344
+ # ingests it like any other fact. `smoke` is a FACT kind and is never advisory: a smoke check is
3345
+ # binary. It either answered or it did not; there is no "measured against no adopted budget"
3346
+ # reading of `ok`, which is why `corpus` may run advisory and this may not.
3347
+ #
3348
+ # Evidence is EXECUTED, not narrated -- and a report the engine cannot read is refused (exit 2)
3349
+ # rather than scored, for the same reason a malformed corpus is: an unreadable smoke reported as
3350
+ # "0 checks ok" would be an unmeasurable run rendered as a measured catastrophe, and an EMPTY
3351
+ # `checks` list read as a clean gate would be the false clean in the other direction.
3352
+ # --------------------------------------------------------------------------- #
3353
+ # How many checks (and failed names) travel on the record. A receipt cites evidence, it is not
3354
+ # a dump -- the same cap `_ac_tags` puts on its testcase receipts, for the same reason.
3355
+ SMOKE_MAX_PERSISTED = 20
3356
+
3357
+
3358
+ def _smoke_refuse(msg):
3359
+ """Every smoke-ingest refusal is exit 2 and NAMES the offending check or field. The report
3360
+ is written by the PROJECT's own tool, so what this guards against is a contract the project
3361
+ got subtly wrong -- and a wrong contract that scored anyway would be worse than one that
3362
+ refused, because the number would look like a measurement."""
3363
+ print("[qa_ledger] smoke-ingest: " + msg, file=sys.stderr)
3364
+ sys.exit(2)
3365
+
3366
+
3367
+ def _smoke_checks(path):
3368
+ """Read a smoke report into an ORDERED list of checks. The contract is deliberately the
3369
+ smallest thing a shell script can emit:
3370
+
3371
+ {"checks": [{"name": "...", "ok": true|false,
3372
+ "status": <int|string, optional>,
3373
+ "latency_ms": <number, optional>,
3374
+ "evidence": "<string, optional>"}]}
3375
+
3376
+ `name` and a BOOLEAN `ok` are the whole mandatory surface. `ok: "true"` and `ok: 1` are
3377
+ refused: a string and a number are not verdicts, and a contract that coerces is a contract
3378
+ that cannot say what it measured. The optional fields are validated when PRESENT and never
3379
+ invented when absent -- a check with no `latency_ms` measured no latency, which is a
3380
+ different fact from a latency of 0."""
3381
+ if not os.path.isfile(path):
3382
+ _smoke_refuse("report not found: %s -- a smoke run that left no report is UNMEASURED, "
3383
+ "never a clean gate" % path)
3384
+ try:
3385
+ with open(path, encoding="utf-8") as fh:
3386
+ doc = json.load(fh)
3387
+ except OSError as exc:
3388
+ _smoke_refuse("report unreadable: %s" % exc)
3389
+ except ValueError as exc:
3390
+ _smoke_refuse("%s is not valid JSON (%s) -- a malformed report is refused, never "
3391
+ "scored" % (path, exc))
3392
+ if not isinstance(doc, dict):
3393
+ _smoke_refuse("%s is a %s, not a JSON object carrying a `checks` list"
3394
+ % (path, type(doc).__name__))
3395
+ if "checks" not in doc:
3396
+ _smoke_refuse('%s has no `checks` key -- the contract is {"checks": [{"name": ..., '
3397
+ '"ok": true|false}, ...]}' % path)
3398
+ raw = doc["checks"]
3399
+ if not isinstance(raw, list):
3400
+ _smoke_refuse("%s: `checks` is a %s, not a list" % (path, type(raw).__name__))
3401
+ if not raw:
3402
+ _smoke_refuse("%s holds 0 checks -- an EMPTY smoke is not evidence: it is a run that "
3403
+ "verified nothing, and it does not read as a clean gate" % path)
3404
+ checks = []
3405
+ for i, row in enumerate(raw):
3406
+ where = "check %d" % (i + 1)
3407
+ if not isinstance(row, dict):
3408
+ _smoke_refuse("%s: %s is a %s, not an object" % (path, where, type(row).__name__))
3409
+ name = row.get("name")
3410
+ if not isinstance(name, str) or not name.strip():
3411
+ _smoke_refuse("%s: %s has no `name` -- a check nobody can name is a check nobody "
3412
+ "can act on" % (path, where))
3413
+ where = "%s (%s)" % (where, name.strip())
3414
+ if not isinstance(row.get("ok"), bool):
3415
+ _smoke_refuse("%s: %s has no boolean `ok` (got %r) -- a smoke check is a binary "
3416
+ "fact, and a string, a number or a missing key is not a verdict"
3417
+ % (path, where, row.get("ok")))
3418
+ status = row.get("status")
3419
+ if status is not None and (isinstance(status, bool)
3420
+ or not isinstance(status, (int, str))):
3421
+ _smoke_refuse("%s: %s has a `status` that is neither an integer nor a string (%r)"
3422
+ % (path, where, status))
3423
+ lat = row.get("latency_ms")
3424
+ if lat is not None and (isinstance(lat, bool) or not isinstance(lat, (int, float))):
3425
+ _smoke_refuse("%s: %s has a `latency_ms` that is not a number (%r)"
3426
+ % (path, where, lat))
3427
+ ev = row.get("evidence")
3428
+ if ev is not None and not isinstance(ev, str):
3429
+ _smoke_refuse("%s: %s has an `evidence` that is not a string (%r)"
3430
+ % (path, where, ev))
3431
+ checks.append({"name": name.strip(), "ok": row["ok"], "status": status,
3432
+ "latency_ms": lat, "evidence": ev})
3433
+ return checks
3434
+
3435
+
3436
+ def _smoke_ac_of(checks, want_ok):
3437
+ """Criterion ids tagged on check NAMES, in the SAME grammar a JUnit testcase name uses
3438
+ (`_ac_tag_ids`, ADR-036): a check named "AC-28 the admin page serves" tags AC-28. One
3439
+ extractor, so a name cannot close a criterion in the suite and fail to close it here."""
3440
+ ids = []
3441
+ for c in checks:
3442
+ if bool(c["ok"]) != want_ok:
3443
+ continue
3444
+ for cid in _ac_tag_ids(c["name"]):
3445
+ if cid not in ids:
3446
+ ids.append(cid)
3447
+ return sorted(ids, key=_top_ac_key)
3448
+
3449
+
3450
+ def cmd_smoke_ingest(args):
3451
+ """Ingest a smoke report as `gate:smoke` -- the smoke run MEASURED instead of narrated.
3452
+
3453
+ A failed check is a BLOCKER through the SAME record shape `gate-check` and `corpus-run`
3454
+ write: readiness capped <=65, convergence blocked, and a later clean report clears it
3455
+ (latest-per-tool wins). `smoke` is a FACT kind and is NOT advisory-capable: `ok` is binary,
3456
+ so there is no budget it could be measured against and no honest advisory reading of it.
3457
+
3458
+ A check whose NAME carries an AC tag closes that criterion MEASURED iff the check is `ok`
3459
+ AND the report as a whole passed -- and a FAILED tagged check is red evidence that vetoes,
3460
+ exactly like a red JUnit testcase. Fail-closed in both directions: a green check inside a
3461
+ failing smoke closes nothing, because the run it belongs to did not hold."""
3462
+ ledger = _load(args.ledger)
3463
+ node = _repo_node(ledger, args.repo)
3464
+ tool = "gate:smoke"
3465
+ _validate_iteration(node, tool, args.iteration)
3466
+ checks = _smoke_checks(args.report)
3467
+
3468
+ n_ok = sum(1 for c in checks if c["ok"])
3469
+ failed = [c for c in checks if not c["ok"]]
3470
+ failing = bool(failed)
3471
+ failed_names = [c["name"] for c in failed]
3472
+ note = "smoke %s: %d/%d checks ok%s" % (
3473
+ os.path.basename(args.report), n_ok, len(checks),
3474
+ (", %d failed (%s)" % (len(failed), ", ".join(failed_names[:SMOKE_MAX_PERSISTED])))
3475
+ if failed else "")
3476
+ rec = _append_gate_record(ledger, node, args.repo, tool, args.iteration,
3477
+ failing, len(failed), note)
3478
+ # the MEASUREMENT travels on the record, not only its verdict: a reader six months later
3479
+ # sees WHICH checks ran, what each answered and how long it took -- the facts the narration
3480
+ # used to carry and lose.
3481
+ rec["smoke"] = {
3482
+ "report": args.report.replace("\\", "/"), "ok": n_ok, "failed": len(failed),
3483
+ "checks": [{"name": c["name"], "ok": c["ok"], "status": c["status"],
3484
+ "latency_ms": c["latency_ms"]} for c in checks[:SMOKE_MAX_PERSISTED]],
3485
+ "failed_names": failed_names[:SMOKE_MAX_PERSISTED],
3486
+ # computed over EVERY check, never over the truncated receipt above: the cap is a
3487
+ # display budget, and a criterion's fate must not depend on where the list was cut.
3488
+ "ac": _smoke_ac_of(checks, True),
3489
+ "ac_red": _smoke_ac_of(checks, False),
3490
+ }
3491
+ _save(args.ledger, ledger)
3492
+
3493
+ out = {"repo": args.repo, "tool": tool, "verdict": "fail" if failing else "pass",
3494
+ "smoke": rec["smoke"], "note": note}
3495
+ if args.json:
3496
+ print(json.dumps(out, indent=2, ensure_ascii=False))
3497
+ else:
3498
+ print("[qa_ledger] %s/%s: %s — %d/%d checks ok"
3499
+ % (args.repo, tool, "FAIL" if failing else "PASS", n_ok, len(checks)))
3500
+ for c in checks[:SMOKE_MAX_PERSISTED]:
3501
+ bits = [b for b in ("status %s" % c["status"] if c["status"] is not None else None,
3502
+ "%s ms" % c["latency_ms"] if c["latency_ms"] is not None
3503
+ else None) if b]
3504
+ print(" %s %s%s" % ("ok " if c["ok"] else "FAIL", c["name"],
3505
+ (" (%s)" % ", ".join(bits)) if bits else ""))
3506
+ if len(checks) > SMOKE_MAX_PERSISTED:
3507
+ print(" ... +%d more check(s) not listed or persisted"
3508
+ % (len(checks) - SMOKE_MAX_PERSISTED))
3509
+ if failing:
3510
+ print(" caps readiness <=65 and blocks convergence until a clean smoke")
3511
+ sys.exit(1 if failing else 0)
3512
+
3513
+
3514
+ def _smoke_records(ledger):
3515
+ """The LATEST smoke record per repo -- the same latest-per-tool rule the gate rollup and
3516
+ convergence already use, so the three cannot disagree about which run is current. A record
3517
+ logged through the `log-gate` parity door carries no `smoke` block and is deliberately not
3518
+ here: it gates (it is a gate record like any other), but it has no checks to read back."""
3519
+ out = {}
3520
+ for rname, rnode in ledger.get("repos", {}).items():
3521
+ rec = _latest_static_by_tool(rnode).get("gate:smoke")
3522
+ if rec is not None and rec.get("smoke"):
3523
+ out[rname] = rec
3524
+ return out
3525
+
3526
+
3527
+ def _smoke_ac_verdicts(ledger):
3528
+ """(closed, vetoed) criterion ids from the latest smoke record of every repo (ADR-047).
3529
+
3530
+ A tagged check closes MEASURED iff it is `ok` AND its report's gate PASSED: a green check
3531
+ inside a failing smoke is a green light on a run that did not hold, and the ledger refuses
3532
+ to read it as one. A FAILED tagged check is red evidence and vetoes wherever it appears --
3533
+ the same rule as a red JUnit testcase, and it outranks every green."""
3534
+ closed, vetoed = set(), set()
3535
+ for rec in _smoke_records(ledger).values():
3536
+ s = rec.get("smoke") or {}
3537
+ vetoed.update(s.get("ac_red") or [])
3538
+ if not (rec.get("gated_reported") or 0):
3539
+ closed.update(s.get("ac") or [])
3540
+ return closed - vetoed, vetoed
3541
+
3542
+
3543
+ def _smoke_report(ledger):
3544
+ """Per-repo SMOKE readout -- ONE derivation, read by both the readiness text and its JSON.
3545
+ CONDITIONAL like lifecycle, agent-origin and field: a repo that never ingested a smoke
3546
+ report is absent from it, so a project that never ran one prints and emits exactly what it
3547
+ printed and emitted before this existed."""
3548
+ out = {}
3549
+ for rname, rec in _smoke_records(ledger).items():
3550
+ s = rec["smoke"]
3551
+ out[rname] = {"state": "FAIL" if (rec.get("gated_reported") or 0) else "PASS",
3552
+ "report": s.get("report"), "ok": s.get("ok"),
3553
+ "failed": s.get("failed"),
3554
+ "total": (s.get("ok") or 0) + (s.get("failed") or 0),
3555
+ "failed_names": s.get("failed_names") or [],
3556
+ "ac": s.get("ac") or [], "ac_red": s.get("ac_red") or []}
3557
+ return out
3558
+
3559
+
3560
+ def _smoke_line(rname, s):
3561
+ """The one-line rendering of a repo's smoke state. Text lives beside the derivation so the
3562
+ JSON and the human readout can never drift apart."""
3563
+ head = "--- smoke %s: %d/%d checks ok" % (rname, s["ok"], s["total"])
3564
+ if s["failed"]:
3565
+ head += ", %d failed (%s)" % (s["failed"], ", ".join(s["failed_names"]))
3566
+ return "%s %s" % (head, s["state"])
2783
3567
 
2784
3568
 
2785
3569
  def cmd_flag_blocker(args):
@@ -2966,6 +3750,16 @@ def _derive_phase(ledger, name, node, k, qa_order):
2966
3750
  if len(_dl["uncurated"]) > 3 else "")
2967
3751
  + " -- INV-CURATION-01: sin juicio no hay promocion")
2968
3752
  conv = False
3753
+ # operability gate (ADR-048). NAMING only: a failing record is already a BLOCKER through
3754
+ # _gate_open_and_sev and already vetoes convergence through _converged, so nothing new is
3755
+ # gated here. What is added is WHICH check is missing -- "static-gate gated=1
3756
+ # (gate:operability:1)" tells a human the gate is red without telling them whether to write
3757
+ # a workflow or a RUNBOOK, and phase --require pr-ready is where that answer is needed.
3758
+ _op = _latest_static_by_tool(node).get("gate:operability")
3759
+ if _op and (_op.get("gated_reported") or 0) > 0:
3760
+ reasons.append("operability: %s -- release, reset and the RUNBOOK are part of done, "
3761
+ "not of the last week (ADR-048)"
3762
+ % (_op.get("note") or "checks missing"))
2969
3763
  _cr = _cr_cfg(ledger)
2970
3764
  if _cr and _cr.get("mode") == "final":
2971
3765
  _head = None
@@ -3455,15 +4249,35 @@ def cmd_spec_drift(args):
3455
4249
  lag_days = int(args.max_lag_days if args.max_lag_days is not None
3456
4250
  else cfg.get("max_lag_days", 30))
3457
4251
  repo_path = _scope_path(ledger, args.repo)
4252
+ root_path = os.path.dirname(os.path.abspath(args.ledger)) or "."
3458
4253
 
3459
4254
  # The spec surface is fixed by ADR-005: the repo SPEC.md plus every ADR.
3460
- spec_files = []
3461
- if os.path.isfile(os.path.join(repo_path, "SPEC.md")):
3462
- spec_files.append("SPEC.md")
3463
- adr_dir = os.path.join(repo_path, "docs", "adr")
3464
- if os.path.isdir(adr_dir):
3465
- spec_files += sorted("docs/adr/" + f for f in os.listdir(adr_dir)
3466
- if f.lower().endswith(".md"))
4255
+ def _specs_at(base):
4256
+ found = []
4257
+ if os.path.isfile(os.path.join(base, "SPEC.md")):
4258
+ found.append("SPEC.md")
4259
+ adr_dir = os.path.join(base, "docs", "adr")
4260
+ if os.path.isdir(adr_dir):
4261
+ found += sorted("docs/adr/" + f for f in os.listdir(adr_dir)
4262
+ if f.lower().endswith(".md"))
4263
+ return found
4264
+
4265
+ # 2.2.0 field fix: a MONOREPO keeps ONE SPEC.md and one docs/adr/ at the root, next to
4266
+ # uscha.config.json, while repos[R].path points at a subdirectory -- and this command read
4267
+ # only that subdirectory, so `spec-drift --repo backend-api` answered "no spec documents"
4268
+ # about a project whose spec was one level up. A spec found at the root is the MONOREPO's
4269
+ # spec and governs every repo. The repo's own path still WINS when it has one (a repo that
4270
+ # carries its own SPEC is describing itself); the config root -- taken as the ledger's
4271
+ # directory, which is where `init --config uscha.config.json` is run -- is the fallback.
4272
+ # The answer NAMES which of the two it read: "no drift" and "read the wrong tree" produced
4273
+ # the same silence, and that is what made the field report take a week to notice.
4274
+ base_path, spec_source = repo_path, "repo"
4275
+ spec_files = _specs_at(repo_path)
4276
+ if not spec_files and os.path.realpath(root_path) != os.path.realpath(repo_path):
4277
+ root_specs = _specs_at(root_path)
4278
+ if root_specs:
4279
+ base_path, spec_source, spec_files = root_path, "root", root_specs
4280
+ repo_path = base_path
3467
4281
 
3468
4282
  tracked = []
3469
4283
  ls = subprocess.run(["git", "ls-files"], cwd=repo_path, capture_output=True,
@@ -3535,20 +4349,26 @@ def cmd_spec_drift(args):
3535
4349
  results.append(row)
3536
4350
 
3537
4351
  out = {"repo": args.repo, "max_lag_days": lag_days, "results": results,
3538
- "advisory": True}
4352
+ "advisory": True, "spec_source": spec_source if spec_files else None,
4353
+ "spec_base": repo_path}
3539
4354
 
3540
4355
  # Latest-state record so the mirador can surface an advisory row. Advisory data,
3541
4356
  # not a step in the loop: no step_counter, no gate record, no readiness input.
3542
4357
  ledger["spec_drift"] = {"repo": args.repo, "at": _now(), "max_lag_days": lag_days,
3543
- "results": results}
4358
+ "results": results,
4359
+ "spec_source": spec_source if spec_files else None}
3544
4360
  _save(args.ledger, ledger)
3545
4361
 
3546
4362
  if args.json:
3547
4363
  print(json.dumps(out, indent=2, ensure_ascii=False))
3548
4364
  else:
3549
4365
  print("SPEC-DRIFT %s (advisory, lag > %dd):" % (args.repo, lag_days))
4366
+ if spec_source == "root" and spec_files:
4367
+ print(" specs read from the CONFIG ROOT (%s): the monorepo's SPEC governs "
4368
+ "every repo" % repo_path)
3550
4369
  if not results:
3551
- print(" no spec documents found (SPEC.md / docs/adr/*.md)")
4370
+ print(" no spec documents found (SPEC.md / docs/adr/*.md) in the repo path "
4371
+ "nor at the config root")
3552
4372
  mark = {"SPEC_STALE": "!!", "CLEAN": "ok", "UNMAPPED": "--", "UNTRACKED": "--",
3553
4373
  "NO-CODE": "ok"}
3554
4374
  for r_ in results:
@@ -3563,6 +4383,330 @@ def cmd_spec_drift(args):
3563
4383
  sys.exit(0)
3564
4384
 
3565
4385
 
4386
+ # --------------------------------------------------------------------------- #
4387
+ # operability (ADR-048: release, reset and the RUNBOOK are MEASURED, not narrated)
4388
+ # --------------------------------------------------------------------------- #
4389
+ # The field finding, twice in a row: release-by-CI, the reset/seed script and the RUNBOOK
4390
+ # arrived in the last week of two projects. The devloop NAMED them in phase 8 prose and
4391
+ # nothing measured them, so "we'll do it at the end" survived every gate the kit has -- the
4392
+ # same shape as every other narrated dimension this engine has replaced with a fact.
4393
+ #
4394
+ # What this command reads is FILES IN THE TREE, never prose and never a claim: a workflow that
4395
+ # runs the repo's own test command, a workflow that publishes something, a RUNBOOK with the
4396
+ # four headings an operator needs at 3am, and a seed/reset command the config declares whose
4397
+ # script is actually on disk. It cannot read whether the RUNBOOK is GOOD -- that is a human
4398
+ # judgment and it is not pretended here. It can read that the file exists and that the
4399
+ # sections are named, which is the difference between a project that thought about rollback
4400
+ # and one that has not yet.
4401
+ #
4402
+ # It never executes anything (ADR-008: the engine is not an executor of config-supplied
4403
+ # shell) and its own exit code is always 0. Whether the verdict GATES is the project's
4404
+ # declaration -- `defaults.operability.gate`, owned by risk profiles C, D and E.
4405
+ OPERABILITY_CHECKS = ("ci", "release", "runbook", "seed")
4406
+
4407
+ # GitHub Actions is the only CI this reads. Another system is NAMED as unknown rather than
4408
+ # failed: the engine cannot open a pipeline it does not understand, and a red invented for a
4409
+ # pipeline nobody read would be exactly the manufactured verdict the kit refuses elsewhere.
4410
+ OPERABILITY_CI_DIR = (".github", "workflows")
4411
+ OPERABILITY_OTHER_CI = (
4412
+ ".gitlab-ci.yml", ".travis.yml", "azure-pipelines.yml", "bitbucket-pipelines.yml",
4413
+ "Jenkinsfile", ".circleci", ".drone.yml", ".teamcity",
4414
+ )
4415
+ # The publish recognisers, listed rather than guessed: a workflow step that creates or
4416
+ # attaches a release asset. Short and documented on purpose -- a regex over "release" would
4417
+ # match a branch name, a job title and a comment, and a gate that matches prose is a gate that
4418
+ # certifies prose.
4419
+ OPERABILITY_RELEASE_MARKERS = (
4420
+ "softprops/action-gh-release",
4421
+ "actions/upload-release-asset",
4422
+ "gh release create",
4423
+ "gh release upload",
4424
+ "gh release",
4425
+ "npm publish",
4426
+ "twine upload",
4427
+ )
4428
+ # A test-ish subcommand, for the case where the workflow does not repeat the configured
4429
+ # command verbatim (a Makefile target, an extra flag, a matrix variable in the middle).
4430
+ OPERABILITY_TEST_WORDS = ("test", "tests", "pytest", "nextest", "jest", "check", "verify")
4431
+ # The four headings an operator needs, matched case-insensitively in EN and ES. What is
4432
+ # matched is the HEADING, not the body: the engine can see that rollback was thought about,
4433
+ # never that the procedure is correct.
4434
+ OPERABILITY_RUNBOOK_SECTIONS = (
4435
+ ("start", r"arranque|start|boot"),
4436
+ ("config", r"config"),
4437
+ ("rollback", r"rollback|reversi"),
4438
+ ("smoke", r"smoke|humo"),
4439
+ )
4440
+ OPERABILITY_RUNBOOK_PATHS = ("docs/RUNBOOK.md", "RUNBOOK.md")
4441
+ OPERABILITY_SCRIPT_EXT = (".sh", ".py", ".ps1", ".bat", ".sql", ".js", ".ts", ".rb")
4442
+
4443
+
4444
+ def _op_bases(ledger, repo, ledger_path):
4445
+ """(repo path, config root) with realpath on BOTH sides -- the Windows 8.3 lesson, and the
4446
+ monorepo lesson spec-drift paid for in 2.2.0: the workflows and the RUNBOOK of a monorepo
4447
+ live at the config root while repos[R].path points at a subdirectory. The repo's own tree
4448
+ WINS when it has the artifact; the root is the fallback, and the answer NAMES which one it
4449
+ read, because "no CI" and "read the wrong tree" produced the same silence."""
4450
+ repo_path = _scope_path(ledger, repo)
4451
+ root_path = os.path.dirname(os.path.abspath(ledger_path)) or "."
4452
+ bases = [(repo_path, "repo")]
4453
+ if os.path.realpath(root_path) != os.path.realpath(repo_path):
4454
+ bases.append((root_path, "root"))
4455
+ return bases
4456
+
4457
+
4458
+ def _op_workflows(bases):
4459
+ """(list of (relative name, text), base, where) for the first base that HAS a GitHub
4460
+ Actions directory, else ([], None, None)."""
4461
+ for base, where in bases:
4462
+ wdir = os.path.join(base, *OPERABILITY_CI_DIR)
4463
+ if not os.path.isdir(wdir):
4464
+ continue
4465
+ files = []
4466
+ for name in sorted(os.listdir(wdir)):
4467
+ if not name.lower().endswith((".yml", ".yaml")):
4468
+ continue
4469
+ try:
4470
+ with open(os.path.join(wdir, name), encoding="utf-8",
4471
+ errors="replace") as fh:
4472
+ files.append((name, fh.read()))
4473
+ except OSError:
4474
+ continue
4475
+ if files:
4476
+ return files, base, where
4477
+ return [], None, None
4478
+
4479
+
4480
+ def _op_other_ci(bases):
4481
+ """The first non-Actions CI marker found, as `name (where)`, or None."""
4482
+ for base, where in bases:
4483
+ for name in OPERABILITY_OTHER_CI:
4484
+ if os.path.exists(os.path.join(base, name)):
4485
+ return "%s (%s)" % (name, where)
4486
+ return None
4487
+
4488
+
4489
+ def _op_test_command(ledger, repo):
4490
+ """The repo's CONFIGURED test command: repos[R].test_command, else the per-type
4491
+ defaults.test_command_<type>. Read, never run."""
4492
+ cfg = ledger.get("config") or {}
4493
+ defaults = cfg.get("defaults") or {}
4494
+ entry = {}
4495
+ for r in cfg.get("repos", []):
4496
+ if r.get("name") == repo:
4497
+ entry = r
4498
+ break
4499
+ if _has_text(entry.get("test_command")):
4500
+ return entry["test_command"].strip()
4501
+ rtype = entry.get("type") or (ledger.get("repos", {}).get(repo) or {}).get("type")
4502
+ value = defaults.get("test_command_%s" % rtype) if rtype else None
4503
+ return value.strip() if _has_text(value) else None
4504
+
4505
+
4506
+ def _op_ci(workflows, command):
4507
+ """(status, detail). `ok` when a workflow step runs the configured command -- verbatim, or
4508
+ its first token beside a test-ish subcommand on the same line. The detail SAYS what
4509
+ matched, so the human can disagree with the match instead of with a boolean."""
4510
+ if not command:
4511
+ return "missing", "no test command configured for this repo (nothing to look for)"
4512
+ head = command.split()[0]
4513
+ tail = os.path.basename(head)
4514
+ for name, body in workflows:
4515
+ for line in body.splitlines():
4516
+ stripped = line.strip()
4517
+ if command in line:
4518
+ return "ok", '%s runs "%s"' % (name, command)
4519
+ if head not in line and tail not in line:
4520
+ continue
4521
+ words = re.split(r"[^A-Za-z0-9_.-]+", stripped.lower())
4522
+ if any(w in OPERABILITY_TEST_WORDS for w in words):
4523
+ shown = re.sub(r"^-\s*", "", stripped)
4524
+ shown = re.sub(r"^run:\s*", "", shown)
4525
+ return "ok", '%s runs "%s"' % (name, shown[:70])
4526
+ return "missing", ("no workflow step runs the configured test command (%s)"
4527
+ % command)
4528
+
4529
+
4530
+ def _op_release(workflows):
4531
+ for name, body in workflows:
4532
+ for marker in OPERABILITY_RELEASE_MARKERS:
4533
+ if marker in body:
4534
+ return "ok", '%s runs "%s"' % (name, marker)
4535
+ return "missing", ("no workflow publishes or attaches a release asset (looked for: %s)"
4536
+ % ", ".join(OPERABILITY_RELEASE_MARKERS))
4537
+
4538
+
4539
+ def _op_runbook(bases, declared):
4540
+ """(status, detail, path). `declared` is defaults.operability.runbook when the project set
4541
+ one; otherwise docs/RUNBOOK.md then RUNBOOK.md, repo tree before config root."""
4542
+ candidates = [declared] if _has_text(declared) else list(OPERABILITY_RUNBOOK_PATHS)
4543
+ for base, where in bases:
4544
+ for rel in candidates:
4545
+ path = os.path.join(base, rel.replace("/", os.sep))
4546
+ if not os.path.isfile(path):
4547
+ continue
4548
+ try:
4549
+ with open(path, encoding="utf-8", errors="replace") as fh:
4550
+ body = fh.read()
4551
+ except OSError as exc:
4552
+ return "missing", "%s unreadable: %s" % (rel, exc), rel
4553
+ heads = [row.lstrip("#").strip().lower()
4554
+ for row in body.splitlines() if row.lstrip().startswith("#")]
4555
+ blob = "\n".join(heads)
4556
+ absent = [label for label, pattern in OPERABILITY_RUNBOOK_SECTIONS
4557
+ if not re.search(pattern, blob, re.I)]
4558
+ if absent:
4559
+ return ("missing", "missing sections (%s) in %s [%s]"
4560
+ % (", ".join(absent), rel, where), rel)
4561
+ return "ok", "%s [%s], all four sections named" % (rel, where), rel
4562
+ return ("missing", "no RUNBOOK found (looked for %s)" % ", ".join(candidates), None)
4563
+
4564
+
4565
+ def _op_seed(ledger, repo, bases, defaults_op):
4566
+ """(status, detail). The seed/reset command is DECLARED, never sniffed: an engine that
4567
+ guessed which script resets the database would be guessing about the most destructive
4568
+ command in the project."""
4569
+ entry = {}
4570
+ for r in (ledger.get("config") or {}).get("repos", []):
4571
+ if r.get("name") == repo:
4572
+ entry = r
4573
+ break
4574
+ repo_op = entry.get("operability") if isinstance(entry.get("operability"), dict) else {}
4575
+ command = repo_op.get("seed_command")
4576
+ if not _has_text(command):
4577
+ command = defaults_op.get("seed_command")
4578
+ if not _has_text(command):
4579
+ return "missing", ("no seed/reset command declared "
4580
+ "(repos[R].operability.seed_command or "
4581
+ "defaults.operability.seed_command)")
4582
+ command = command.strip()
4583
+ script = None
4584
+ for token in re.split(r"\s+", command):
4585
+ bare = token.strip("\"'")
4586
+ if "/" in bare or "\\" in bare or bare.lower().endswith(OPERABILITY_SCRIPT_EXT):
4587
+ script = bare
4588
+ break
4589
+ if script is None:
4590
+ # A command with no path in it (`make seed`, `npm run reset`) names a target this
4591
+ # engine cannot resolve without running something. Declared is what is measurable.
4592
+ return "ok", 'declared: "%s" (no script path to verify)' % command
4593
+ for base, _where in bases:
4594
+ if os.path.exists(os.path.join(base, script.replace("/", os.sep))):
4595
+ return "ok", 'declared: "%s"' % command
4596
+ return "missing", "script not found: %s" % script
4597
+
4598
+
4599
+ def _operability_report(ledger, repo, ledger_path):
4600
+ """The four FACTS, with the base each was read from. Pure measurement: no persistence,
4601
+ no exit code, no gate decision -- the caller owns all three."""
4602
+ defaults = (ledger.get("config") or {}).get("defaults") or {}
4603
+ op_cfg = defaults.get("operability") if isinstance(defaults.get("operability"), dict) else {}
4604
+ bases = _op_bases(ledger, repo, ledger_path)
4605
+ workflows, wf_base, wf_where = _op_workflows(bases)
4606
+ checks = {}
4607
+ if workflows:
4608
+ ci_status, ci_detail = _op_ci(workflows, _op_test_command(ledger, repo))
4609
+ rel_status, rel_detail = _op_release(workflows)
4610
+ ci_source = "%s [%s]" % ("/".join(OPERABILITY_CI_DIR), wf_where)
4611
+ else:
4612
+ other = _op_other_ci(bases)
4613
+ ci_source = None
4614
+ if other:
4615
+ ci_status = rel_status = "unknown"
4616
+ ci_detail = rel_detail = ("unknown ci system: %s -- this reads GitHub Actions "
4617
+ "only, so it neither certifies nor condemns it" % other)
4618
+ else:
4619
+ ci_status = rel_status = "missing"
4620
+ ci_detail = rel_detail = ("no %s in the repo tree nor at the config root"
4621
+ % "/".join(OPERABILITY_CI_DIR))
4622
+ checks["ci"] = {"status": ci_status, "detail": ci_detail}
4623
+ checks["release"] = {"status": rel_status, "detail": rel_detail}
4624
+ rb_status, rb_detail, rb_path = _op_runbook(bases, op_cfg.get("runbook"))
4625
+ checks["runbook"] = {"status": rb_status, "detail": rb_detail, "path": rb_path}
4626
+ sd_status, sd_detail = _op_seed(ledger, repo, bases, op_cfg)
4627
+ checks["seed"] = {"status": sd_status, "detail": sd_detail}
4628
+ return {"repo": repo, "checks": checks, "ci_source": ci_source,
4629
+ "bases": [{"path": b, "where": w} for b, w in bases]}
4630
+
4631
+
4632
+ def _operability_verdict(report, gate):
4633
+ """(verdict, summary note). Three states, because there are three facts:
4634
+ - every check `ok` -> pass (a gate that ran clean)
4635
+ - any check `missing` -> fail under a declared gate, advisory without
4636
+ - none missing, some `unknown` -> advisory even under a declared gate
4637
+ The third is the one worth spelling out: a pipeline this engine cannot read is UNMEASURED,
4638
+ and recording UNMEASURED as `pass` would hand a declared gate a green nobody measured --
4639
+ the false clean ADR-043 exists to refuse, arriving through a different door."""
4640
+ statuses = [report["checks"][k]["status"] for k in OPERABILITY_CHECKS]
4641
+ note = " · ".join("%s %s" % (k, report["checks"][k]["status"])
4642
+ for k in OPERABILITY_CHECKS)
4643
+ if not gate:
4644
+ return "advisory", note
4645
+ if "missing" in statuses:
4646
+ return "fail", note
4647
+ if "unknown" in statuses:
4648
+ return "advisory", note
4649
+ return "pass", note
4650
+
4651
+
4652
+ def cmd_operability(args):
4653
+ """Measure the operability of a repo: CI, release, RUNBOOK, seed (ADR-048).
4654
+
4655
+ Exit code is ALWAYS 0 for the check itself -- the gate decision belongs to the profile,
4656
+ and it is enforced where every other FACT gate is enforced: the persisted record caps
4657
+ readiness <=65 and blocks convergence when it is a `fail`."""
4658
+ ledger = _load(args.ledger)
4659
+ node = _repo_node(ledger, args.repo)
4660
+ resolved, origin = _resolved_defaults(ledger.get("config") or {})
4661
+ gate = bool(resolved.get("operability.gate"))
4662
+ report = _operability_report(ledger, args.repo, args.ledger)
4663
+ verdict, note = _operability_verdict(report, gate)
4664
+ missing = [k for k in OPERABILITY_CHECKS
4665
+ if report["checks"][k]["status"] == "missing"]
4666
+ report.update({"gate": gate, "gate_origin": origin.get("operability.gate"),
4667
+ "verdict": verdict, "note": note, "missing": missing})
4668
+
4669
+ rec = _append_gate_record(ledger, node, args.repo, "gate:operability", args.iteration,
4670
+ verdict == "fail", len(missing) or 1, note,
4671
+ advisory=(verdict == "advisory"))
4672
+ _save(args.ledger, ledger)
4673
+ report["step"] = rec["n"]
4674
+
4675
+ if args.json:
4676
+ print(json.dumps(report, indent=2, ensure_ascii=False))
4677
+ sys.exit(0)
4678
+ # "not declared" and "declared false" are DIFFERENT facts: the first is a project that
4679
+ # never mentioned the knob, the second a human who turned the gate off on purpose. Saying
4680
+ # "not declared" for both erased the decision (fresh review, 2.2.0).
4681
+ gate_origin = origin.get("operability.gate")
4682
+ gate_shown = ("declared" if gate
4683
+ else "declared false" if gate_origin == "override" else "not declared")
4684
+ print("OPERABILITY %s (gate: %s, origin %s)" % (args.repo, gate_shown, gate_origin))
4685
+ if report["ci_source"]:
4686
+ print(" read: %s" % report["ci_source"])
4687
+ for key in OPERABILITY_CHECKS:
4688
+ c = report["checks"][key]
4689
+ mark = {"ok": "ok", "missing": "!!", "unknown": "--"}.get(c["status"], "??")
4690
+ detail = c["detail"]
4691
+ print(" %s %s: %s%s" % (mark, key, c["status"],
4692
+ (" (%s)" % detail) if detail else ""))
4693
+ present = sum(1 for k in OPERABILITY_CHECKS
4694
+ if report["checks"][k]["status"] == "ok")
4695
+ if verdict == "fail":
4696
+ print(" -- %d of %d present -- FAIL: caps readiness <=65 and blocks convergence "
4697
+ "until %s exist(s)" % (present, len(OPERABILITY_CHECKS), ", ".join(missing)))
4698
+ elif verdict == "pass":
4699
+ print(" -- %d of %d present -- PASS: the declared operability gate ran clean"
4700
+ % (present, len(OPERABILITY_CHECKS)))
4701
+ else:
4702
+ print(" -- %d of %d present -- ADVISORY (measured, not gating): caps nothing, "
4703
+ "blocks nothing%s" % (present, len(OPERABILITY_CHECKS),
4704
+ ("; declare defaults.operability.gate: true (or a risk "
4705
+ "profile C/D/E) to make it a gate" if not gate else
4706
+ "; a check this engine cannot read is UNMEASURED, "
4707
+ "never a green")))
4708
+ sys.exit(0)
4709
+
3566
4710
 
3567
4711
  # --------------------------------------------------------------------------- #
3568
4712
  # golden-coverage (ADR-006: the golden<->source mapping, DERIVED BY MEASUREMENT)
@@ -7319,27 +8463,147 @@ def _render_lang_md(r, has_js=False):
7319
8463
  # --------------------------------------------------------------------------- #
7320
8464
 
7321
8465
  FACTS_FILE = "SYSTEM-FACTS.json"
8466
+ # The NAMESPACED name `install-uscha.py` writes the kit's version under, beside the installed
8467
+ # skills (2.2.0). It is not `VERSION`: that bare name is shared with whatever else the agent's
8468
+ # skills directory holds, and an installer that writes it owns a file it did not create.
8469
+ KIT_VERSION_COPY = ".uscha-kit-VERSION"
8470
+
8471
+
8472
+ def _kit_root():
8473
+ """The kit directory this engine belongs to, or None.
8474
+
8475
+ By MARKER, not by fixed depth: the canonical engine sits 4 levels deep
8476
+ (.claude/skills/uscha-devloop/) and the Codex twin 3 (skills/uscha-devloop/). A fixed
8477
+ dirname walk made the twin silently derive the OUTER repo root -- version None,
8478
+ 0 skills, no error (fresh-review HIGH, reproduced by running both copies).
8479
+
8480
+ The marker is a VERSION file AND a skills tree beside it (2.2.0). VERSION alone stopped
8481
+ being sufficient the moment `install-uscha.py` began dropping one beside the installed
8482
+ skills so `doctor` could date them: that root carries a version and no kit, and answering
8483
+ it here would make `facts` derive `0 skills` from a directory that never held any --
8484
+ a manufactured fact, which is worse than the honest refusal this returns instead.
8485
+ An engine that only needs the VERSION asks _engine_kit_version().
8486
+
8487
+ The walk starts at the engine's REALPATH: a `--mode link` install puts a link to the kit's
8488
+ skill directory under the agent's skills root, and an abspath walk-up from there climbs the
8489
+ agent's tree instead of the checkout the link points into (fresh review, 2.2.0). Same lesson
8490
+ the Windows 8.3 gotcha teaches: resolve before you compare, resolve before you walk."""
8491
+ cur = os.path.dirname(os.path.realpath(__file__))
8492
+ for _ in range(6):
8493
+ if (os.path.isfile(os.path.join(cur, "VERSION"))
8494
+ and (os.path.isdir(os.path.join(cur, ".claude", "skills"))
8495
+ or os.path.isdir(os.path.join(cur, "skills")))):
8496
+ return cur
8497
+ nxt = os.path.dirname(cur)
8498
+ if nxt == cur:
8499
+ break
8500
+ cur = nxt
8501
+ return None
7322
8502
 
7323
8503
 
7324
- def _derive_facts():
7325
- """Facts derived from the ARTIFACTS themselves, never from prose and never from greps
7326
- over documentation: the subcommand list comes from introspecting the REAL parser, the
7327
- skill list from the REAL kit tree, the version from the kit VERSION file. No timestamp
7328
- on purpose: regeneration over an unchanged repo must be byte-identical (AC-SF-01)."""
7329
- here = os.path.abspath(__file__)
7330
- # kit root by MARKER, not by fixed depth: the canonical engine sits 4 levels deep
7331
- # (.claude/skills/uscha-devloop/) and the Codex twin 3 (skills/uscha-devloop/). A fixed
7332
- # dirname walk made the twin silently derive the OUTER repo root -- version None,
7333
- # 0 skills, no error (fresh-review HIGH, reproduced by running both copies).
7334
- kit, cur = None, os.path.dirname(here)
8504
+ def _engine_kit_version():
8505
+ """(version, where) for the kit THIS engine came from, or (None, [dirs it looked in]).
8506
+
8507
+ Which kit an installed skill came from is a different question from which kit tree this
8508
+ engine sits in, and an INSTALLED engine has no kit tree at all: `install-uscha.py` copies
8509
+ the kit's version beside the installed skills precisely so the question stays answerable
8510
+ there -- from a checkout this walk lands on the kit root like _kit_root() does, and from an
8511
+ install it lands on the install root.
8512
+
8513
+ THREE sources, in this order at every level (2.2.0, fresh review):
8514
+
8515
+ 1. `.uscha-kit-VERSION` -- the NAMESPACED copy the installer writes. The copy used to be
8516
+ called `VERSION`, a bare shared name dropped into directories the kit does not own
8517
+ (`~/.claude/skills/`, `~/.cursor/skills/`): whatever else lived under that name was
8518
+ overwritten by an install and deleted by an uninstall. The kit owns its prefix and
8519
+ nothing else.
8520
+ 2. `VERSION` -- the KIT ROOT's own file, which is the checkout case and is never written by
8521
+ an install any more. A foreign file under that name is read, not written: reading is what
8522
+ this walk is for, and a wrong version is reported as a version, never as damage.
8523
+ 3. `uscha-install.json`'s `version` -- the install marker, already on the walk, so an
8524
+ install whose copy was removed by hand still answers instead of going UNMEASURED.
8525
+
8526
+ The walk starts at the engine's REALPATH: under `--mode link` the installed skill directory
8527
+ is a link into the kit checkout, and an abspath walk-up climbs the agent's tree rather than
8528
+ the kit's -- so a link install could not resolve its own kit without the copy.
8529
+
8530
+ Never guesses: with no version anywhere it returns the directories it READ, so the report
8531
+ can say where it looked instead of only that it failed."""
8532
+ cur = os.path.dirname(os.path.realpath(__file__))
8533
+ looked = []
7335
8534
  for _ in range(6):
7336
- if os.path.isfile(os.path.join(cur, "VERSION")):
7337
- kit = cur
7338
- break
8535
+ looked.append(cur)
8536
+ for name in (KIT_VERSION_COPY, "VERSION"):
8537
+ try:
8538
+ with open(os.path.join(cur, name), encoding="utf-8") as fh:
8539
+ return fh.read().strip().split()[-1], cur
8540
+ except (OSError, IndexError):
8541
+ pass
8542
+ try:
8543
+ with open(os.path.join(cur, "uscha-install.json"), encoding="utf-8") as fh:
8544
+ declared = (json.load(fh) or {}).get("version")
8545
+ if isinstance(declared, str) and declared.strip():
8546
+ return declared.strip().split()[-1], cur
8547
+ except (OSError, ValueError, AttributeError):
8548
+ pass
7339
8549
  nxt = os.path.dirname(cur)
7340
8550
  if nxt == cur:
7341
8551
  break
7342
8552
  cur = nxt
8553
+ return None, looked
8554
+
8555
+
8556
+ BENCH_DOC = "DIAMOND-BENCH.md"
8557
+ # One generated table row per archetype: `| crud-store | PASS | M1 12/12, ... |`. Anchored at the
8558
+ # line start and on the closing pipe so the per-entry prose below the table ("### guard -- PARTIAL")
8559
+ # cannot be counted twice.
8560
+ _BENCH_ROW = re.compile(r"^\|\s*([A-Za-z0-9][\w.-]*)\s*\|\s*(PASS|PARTIAL|FAIL|PENDING)\s*\|")
8561
+
8562
+
8563
+ def _derive_bench(kit):
8564
+ """The Diamond Bench headline, COUNTED out of the bench's own generated report, or None.
8565
+
8566
+ `DIAMOND-BENCH.md` is written by `qa_ledger.py bench` over the committed fixture and carries
8567
+ the "do not hand-edit" banner: every row in it is a measured run. Re-running the bench here
8568
+ would be the honest derivation and is not affordable -- a full pass is ~650 child processes,
8569
+ and `facts` runs on every suite, every deploy and twice per release. So the fact is counted
8570
+ from the RECORDED verdicts, per archetype, never from the summary sentence beside them and
8571
+ never from a number typed into a document.
8572
+
8573
+ The report lives at the REPO root, one level above the kit: an installed kit has no bench,
8574
+ which is why this returns None instead of guessing. `--check` then reports any claim about it
8575
+ as UNMEASURED rather than letting it pass unexamined."""
8576
+ if not kit:
8577
+ return None
8578
+ for cand in (os.path.join(os.path.dirname(kit), BENCH_DOC),
8579
+ os.path.join(kit, BENCH_DOC)):
8580
+ if not os.path.isfile(cand):
8581
+ continue
8582
+ try:
8583
+ with open(cand, encoding="utf-8-sig") as fh:
8584
+ body = fh.read()
8585
+ except OSError:
8586
+ continue
8587
+ verdicts = []
8588
+ for line in body.split("\n"):
8589
+ m = _BENCH_ROW.match(line.strip())
8590
+ if m:
8591
+ verdicts.append(m.group(2))
8592
+ if not verdicts:
8593
+ continue
8594
+ return {"entries": len(verdicts), "pass": verdicts.count("PASS"),
8595
+ "partial": verdicts.count("PARTIAL"), "fail": verdicts.count("FAIL"),
8596
+ "pending": verdicts.count("PENDING")}
8597
+ return None
8598
+
8599
+
8600
+ def _derive_facts():
8601
+ """Facts derived from the ARTIFACTS themselves, never from prose and never from greps
8602
+ over documentation: the subcommand list comes from introspecting the REAL parser, the
8603
+ skill list from the REAL kit tree, the version from the kit VERSION file, the Diamond
8604
+ headline from the bench's own generated report. No timestamp on purpose: regeneration
8605
+ over an unchanged repo must be byte-identical (AC-SF-01)."""
8606
+ kit = _kit_root()
7343
8607
  if kit is None:
7344
8608
  print("[qa_ledger] facts: no VERSION file found walking up from the engine -- "
7345
8609
  "facts that cannot locate their own kit are not facts.", file=sys.stderr)
@@ -7356,14 +8620,24 @@ def _derive_facts():
7356
8620
  skills = sorted(d for d in os.listdir(sdir)
7357
8621
  if os.path.isfile(os.path.join(sdir, d, "SKILL.md")))
7358
8622
  break
8623
+ bench = _derive_bench(kit)
7359
8624
  return {
7360
8625
  "version": version,
7361
8626
  "subcommands": {"count": len(subs), "list": subs},
7362
8627
  "skills": {"count": len(skills), "list": skills},
8628
+ # null, never a zero: an installed kit ships no bench report, and "0 PASS" would be a
8629
+ # measured-looking answer to a question this tree cannot answer.
8630
+ "diamond": bench,
7363
8631
  "_derivation": {
7364
8632
  "version": "uscha-kit/VERSION",
7365
8633
  "subcommands": "argparse introspection of build_parser()",
7366
8634
  "skills": "SKILL.md inventory under uscha-kit/.claude/skills/",
8635
+ "diamond": ("verdict rows of DIAMOND-BENCH.md, the report `bench` generates over "
8636
+ "uscha-kit/tests/fixtures/diamond-bench (regenerate it with: qa_ledger.py "
8637
+ "bench --dir uscha-kit/tests/fixtures/diamond-bench --out "
8638
+ "DIAMOND-BENCH.md)" if bench else
8639
+ "UNMEASURED: no DIAMOND-BENCH.md beside the kit -- claims about the "
8640
+ "bench headline are reported as undecidable, never as green"),
7367
8641
  "omitted": "stack matrix and REAL/VISION registry: no mechanical "
7368
8642
  "source exists yet -- omitted, not guessed (ADR-012)",
7369
8643
  },
@@ -7393,7 +8667,18 @@ _SPELLED = dict((_spell(n), n) for n in range(1, 100))
7393
8667
  # longest alternative first: an alternation offering "six" before "sixty-three" matches the prefix
7394
8668
  _NUM_ALT = "|".join(sorted(_SPELLED, key=len, reverse=True))
7395
8669
  # the leading \b so that "someone skills" cannot be read as the claim "one skills"
7396
- _COUNT = r"\b(\d+|" + _NUM_ALT + r")\s+"
8670
+ _NUM = r"\b(\d+|" + _NUM_ALT + r")"
8671
+ # HTML splits a claim across elements: the homepage's stat tile reads
8672
+ # `<div class="v">8<small>/12</small></div><div class="k">archetypes regenerate</div>`, so the
8673
+ # count and the noun that gives it meaning are separated by markup rather than by a space. The
8674
+ # gap is therefore whitespace OR tags; `[^<>]` stops one tag from swallowing the rest of the
8675
+ # line, and the repetition is bounded so the gap can never run from one sentence into the next.
8676
+ _GAP = r"(?:\s|<[^<>]{0,80}>){0,8}"
8677
+ _ARCH = r"(?:archetypes?|arquetipos?)"
8678
+ # a claimed count that is NOT the one being rewritten (the other half of a verdict pair)
8679
+ _ANYNUM = r"(?:\d+|" + _NUM_ALT + r")"
8680
+ # what separates `8 PASS` from `4 PARTIAL`: a comma, a middle dot, a slash, spaces, markup
8681
+ _VSEP = r"[^A-Za-z0-9<>]{0,8}" + _GAP
7397
8682
 
7398
8683
 
7399
8684
  def _unspell(token):
@@ -7405,17 +8690,57 @@ _CLAIM_PATTERNS = (
7405
8690
  # (fact key path, regex over one line, needs-context substring or None)
7406
8691
  ("version", r"v(\d+\.\d+\.\d+)", "kit"),
7407
8692
  ("version", r"uscha-kit\s+v?(\d+\.\d+\.\d+)", None),
7408
- ("subcommands.count", _COUNT + r"sub-?comm?ands", None),
7409
- ("subcommands.count", r"(\d+)\s+subcomandos", None),
8693
+ # `_GAP` rather than a plain space since 2.2.0: the site's stat tiles put the count in one
8694
+ # element and its noun in the next (`<div class="v">52</div><div class="k">engine
8695
+ # subcommands</div>`), so a gate that demanded whitespace between them read the homepage's
8696
+ # headline numbers as prose. It had `52 engine subcommands` and `9/12 archetypes` on one
8697
+ # screen, both stale, both invisible to a green release.
8698
+ ("subcommands.count", _NUM + _GAP + r"(?:engine\s+)?sub-?comm?ands", None),
8699
+ ("subcommands.count", r"\b(\d+)" + _GAP + r"subcomandos", None),
7410
8700
  # "agent skills" is the kit's own noun phrase and the paper's; nothing wider is let in,
7411
8701
  # because a WRITER that guessed at "two other skills" would corrupt the sentence it fixed.
7412
- ("skills.count", _COUNT + r"(?:agent\s+)?skills", None),
8702
+ ("skills.count", _NUM + _GAP + r"(?:agent\s+)?skills", None),
8703
+ # The Diamond Bench headline (2.2.0). The homepage said 9/12 for nine releases after ADR-042
8704
+ # moved `transformer` to PARTIAL, because no gate could see the claim: the number sat in one
8705
+ # HTML element and its noun in the next.
8706
+ #
8707
+ # Two shapes only, and both are narrow BY MEASUREMENT -- each wider draft was tried against
8708
+ # the gated set first and rejected by what it caught:
8709
+ #
8710
+ # `<n>/12 archetypes`, `<n> of 12 archetypes`, `<n> de 12 arquetipos` -- the count must be
8711
+ # followed, across markup but never across prose, by the noun it counts, and the
8712
+ # denominator must be the DIGITS 12. That is what keeps the repo's own historical sentence
8713
+ # ("It was 9 of 12 until ADR-042 moved transformer") and the paper's "ten of twelve
8714
+ # archetypes" out of a writer that would have silently rewritten both.
8715
+ #
8716
+ # `<n> PASS <sep> <n> PARTIAL` as ONE shape, verdicts case-sensitive (`(?-i:...)` against
8717
+ # the module-wide re.I). Reading the two numbers independently caught the paper's snapshot
8718
+ # of the July-August arm -- "nine archetypes PASS ... three PARTIAL", a sentence about a
8719
+ # different experiment -- and would have offered to rewrite it. The bench headline always
8720
+ # writes the pair together, so adjacency is both narrower and closer to the real claim.
8721
+ #
8722
+ # `<n> archetypes` alone is deliberately NOT a claim: across the gated set it names subsets
8723
+ # far more often than the bench ("five archetypes", "two archetypes have no second run"), so
8724
+ # the entries count stays a derived fact with no recognised published shape.
8725
+ ("diamond.pass", _NUM + _GAP + r"(?:/|of|de)" + _GAP + r"12" + _GAP + _ARCH, None),
8726
+ ("diamond.pass", _NUM + _GAP + r"(?-i:PASS)" + _VSEP + _ANYNUM + _GAP + r"(?-i:PARTIAL)",
8727
+ None),
8728
+ ("diamond.partial",
8729
+ r"\b" + _ANYNUM + _GAP + r"(?-i:PASS)" + _VSEP + _NUM + _GAP + r"(?-i:PARTIAL)", None),
7413
8730
  )
7414
8731
 
7415
8732
 
7416
8733
  def _fact_value(facts, dotted):
8734
+ """The derived fact behind a claim key, or None when THIS tree cannot derive it.
8735
+
8736
+ Not every fact exists everywhere: an installed kit has no `DIAMOND-BENCH.md`, so
8737
+ `diamond.*` is null there. None is the honest answer and both consumers act on it --
8738
+ `--write` leaves the claim alone (there is nothing to write it to) and `--check` reports it
8739
+ as UNMEASURED. Neither treats an underivable fact as agreement."""
7417
8740
  cur = facts
7418
8741
  for part in dotted.split("."):
8742
+ if not isinstance(cur, dict) or cur.get(part) is None:
8743
+ return None
7419
8744
  cur = cur[part]
7420
8745
  return str(cur)
7421
8746
 
@@ -7506,7 +8831,10 @@ def _write_claims(facts, paths):
7506
8831
  # right to left: an earlier rewrite must not move the offsets of a later one
7507
8832
  for key, start, end, token in sorted(claims, key=lambda c: c[1], reverse=True):
7508
8833
  actual = _fact_value(facts, key)
7509
- if _claim_norm(token) == actual:
8834
+ if actual is None or _claim_norm(token) == actual:
8835
+ # None = this tree cannot derive the fact; there is nothing to rewrite the
8836
+ # claim TO, and inventing one would be the opposite of the gate. The --check
8837
+ # that follows reports it as UNMEASURED.
7510
8838
  continue
7511
8839
  line = line[:start] + _claim_rewrite(token, actual) + line[end:]
7512
8840
  n += 1
@@ -7579,7 +8907,11 @@ def cmd_facts(args):
7579
8907
  # comment spans -- a comment is not a published claim
7580
8908
  for key, _s, _e, claimed in _iter_claims(line):
7581
8909
  actual = _fact_value(facts, key)
7582
- if _claim_norm(claimed) != actual:
8910
+ if actual is None:
8911
+ problems.append((path, n, key, claimed,
8912
+ "UNMEASURED -- this tree derives no such fact "
8913
+ "(see SYSTEM-FACTS _derivation)"))
8914
+ elif _claim_norm(claimed) != actual:
7583
8915
  problems.append((path, n, key, claimed, actual))
7584
8916
  # the parser-surface table (Subcommand/Subcomando header, one `<td class="t">`
7585
8917
  # row per subcommand) is a claim too, just not a numeric one -- a row can go
@@ -7619,9 +8951,12 @@ def cmd_facts(args):
7619
8951
  body = json.dumps(facts, indent=2, ensure_ascii=False, sort_keys=True) + "\n"
7620
8952
  with open(args.out, "w", encoding="utf-8", newline="\n") as fh:
7621
8953
  fh.write(body)
7622
- print("FACTS -> %s: version %s · %d subcommands · %d skills"
8954
+ dia = facts.get("diamond")
8955
+ print("FACTS -> %s: version %s · %d subcommands · %d skills · diamond %s"
7623
8956
  % (args.out, facts["version"], facts["subcommands"]["count"],
7624
- facts["skills"]["count"]))
8957
+ facts["skills"]["count"],
8958
+ ("%d PASS · %d PARTIAL of %d" % (dia["pass"], dia["partial"], dia["entries"])
8959
+ if dia else "UNMEASURED (no %s beside the kit)" % BENCH_DOC)))
7625
8960
 
7626
8961
 
7627
8962
 
@@ -8399,17 +9734,21 @@ def cmd_dashboard(args):
8399
9734
  subscores = [{"k": "coverage",
8400
9735
  "val": round(covp) if isinstance(covp, (int, float)) else None,
8401
9736
  "bd": (f"{round(covp)}%" if isinstance(covp, (int, float)) else None)}]
8402
- gate_block, gate_note = {}, {}
9737
+ gate_block, gate_note, gate_adv = {}, {}, {}
8403
9738
  for g in rd.get("gates", []):
8404
9739
  kind = (g.get("tool") or "").replace("gate:", "")
8405
9740
  key = "golden" if kind.startswith("golden") else kind
8406
9741
  gate_block[key] = gate_block.get(key, False) or bool(g.get("blocking"))
9742
+ gate_adv[key] = gate_adv.get(key, False) or bool(g.get("advisory"))
8407
9743
  if g.get("note") and key not in gate_note:
8408
9744
  gate_note[key] = g.get("note")
8409
9745
  for key in ("simplicity", "waste", "golden"):
8410
9746
  if key in gate_block:
9747
+ # ADR-043: a non-blocking ADVISORY is not "OK" — OK means a declared gate ran clean.
9748
+ _bd = ("FAIL" if gate_block[key]
9749
+ else "ADVISORY" if gate_adv.get(key) else "OK")
8411
9750
  subscores.append({"k": key, "val": None,
8412
- "bd": gate_note.get(key) or ("FAIL" if gate_block[key] else "OK")})
9751
+ "bd": gate_note.get(key) or _bd})
8413
9752
 
8414
9753
  # loops: iters + estado por repo (escalated > converged > active). max sin fuente.
8415
9754
  # El estado se deriva ENTERO con _derive_phase (kit 1.48.1) — la MISMA funcion que
@@ -9033,6 +10372,10 @@ def _top_events(ledger, limit=TOP_EVENTS_TAIL):
9033
10372
  gated = it.get("gated_reported")
9034
10373
  if kind == "gate-not-run":
9035
10374
  tail = "not run — nobody measured it"
10375
+ elif kind == "static-gate" and it.get("advisory"):
10376
+ # ADR-043: measured but not gating. `info` (never green, never red) is the
10377
+ # honest level -- rendering it `pass`/`clean` is the false clean again.
10378
+ level, tail = "info", "advisory — measured, not gating"
9036
10379
  elif kind == "static-gate" and isinstance(gated, int):
9037
10380
  level = "fail" if gated >= 1 else "pass"
9038
10381
  tail = "%d gated finding(s)" % gated if gated else "clean"
@@ -9583,10 +10926,33 @@ def cmd_readiness(args):
9583
10926
  # ingeridos (y 0 rojos). El checkbox es RELATO; el testcase es HECHO.
9584
10927
  ac_ids = [i for i in acc_items if i["id"]]
9585
10928
  ac_tags, stale_reports = _sum_ac_tags(ledger)
10929
+ # ADR-046: a green corpus run is the OTHER way a criterion closes measured. Greenfield has
10930
+ # no old code to characterize, so for the criteria that are about real-world input the only
10931
+ # field evidence there can be is a corpus run over real inputs -- and it closes exactly like
10932
+ # a green testcase does, with the same fail-closed rule below. `corpus_red` is a FAILING
10933
+ # tagged run -- evidence AGAINST, in ADR-046's own words -- and it vetoes like a red
10934
+ # testcase; an ADVISORY run is neither and does neither.
10935
+ corpus_closed, corpus_red = _corpus_ac_verdicts(ledger)
10936
+ # ADR-047: and a green SMOKE check is the third way. "the jar served /admin" used to
10937
+ # arrive as a sub-agent's sentence; now it arrives as a check in a report the engine
10938
+ # READ, and a check named "AC-28 ..." closes AC-28 exactly as a green testcase named
10939
+ # "AC-28 ..." does. `smoke_red` is a FAILED tagged check -- red evidence, and it vetoes
10940
+ # like a red testcase.
10941
+ smoke_closed, smoke_red = _smoke_ac_verdicts(ledger)
9586
10942
 
9587
10943
  def _ac_closed(cid):
9588
10944
  d = ac_tags.get(cid)
9589
- return bool(d and d["green"] >= 1 and d["red"] == 0)
10945
+ # fail-closed FIRST, always: red evidence of ANY kind outranks every green one,
10946
+ # because the cheapest way to fake a closed criterion is to add a green beside a red.
10947
+ if d and d["red"]:
10948
+ return False # red evidence vetoes, whatever else says (fail-closed)
10949
+ if cid in smoke_red:
10950
+ return False # a FAILED tagged smoke check is red evidence too
10951
+ if cid in corpus_red:
10952
+ return False # and so is a FAILING tagged corpus run (ADR-046)
10953
+ if d and d["green"] >= 1:
10954
+ return True
10955
+ return cid in corpus_closed or cid in smoke_closed
9590
10956
 
9591
10957
  # IDs duplicados (ACCEPTANCE mal numerado) cuentan UNA sola vez — si no,
9592
10958
  # un solo test verde cierra "medido" tantos criterios como copias del ID.
@@ -9832,6 +11198,17 @@ def cmd_readiness(args):
9832
11198
  if (acc_traceable and total) else None),
9833
11199
  "narrated_only": narrated_only,
9834
11200
  "measured_unchecked": measured_unchecked,
11201
+ # ADR-046: WHICH ids a green corpus run closed, so a reader can tell
11202
+ # field evidence from suite evidence instead of inferring it -- and
11203
+ # `corpus_vetoed` for the ids a FAILING run holds open, the half a
11204
+ # reader cannot infer from the closed list.
11205
+ "corpus_closed": sorted(corpus_closed, key=_top_ac_key),
11206
+ "corpus_vetoed": sorted(corpus_red, key=_top_ac_key),
11207
+ # ADR-047: the same for the ids a green smoke check closed -- and
11208
+ # `smoke_vetoed` for the ids a FAILED one holds open, which is the
11209
+ # half a reader cannot infer from the closed list.
11210
+ "smoke_closed": sorted(smoke_closed, key=_top_ac_key),
11211
+ "smoke_vetoed": sorted(smoke_red, key=_top_ac_key),
9835
11212
  "stale_reports": stale_reports},
9836
11213
  "facts": {"coverage_pct": round(agg_cov_pct, 2), "coverage_threshold": threshold,
9837
11214
  "gated_open": total_open, "severity": agg_sev,
@@ -9854,9 +11231,32 @@ def cmd_readiness(args):
9854
11231
  # lifecycle (ADR-040): advisory, and CONDITIONAL like fast_path/spec_drift -- a project
9855
11232
  # that declares no lifecycle: block keeps the exact prior payload and the exact prior
9856
11233
  # text. Speaking only when it matters is the anti-ceremony rule applied to itself.
9857
- _lc = _lifecycle_for(os.path.dirname(os.path.abspath(args.ledger)) or os.getcwd())
11234
+ _ready_root = os.path.dirname(os.path.abspath(args.ledger)) or os.getcwd()
11235
+ _lc = _lifecycle_for(_ready_root)
9858
11236
  if _lc["declared"]:
9859
11237
  out["lifecycle"] = _lc
11238
+ # agent-origin (ADR-044): advisory and CONDITIONAL for the same reason -- a project
11239
+ # that tags nothing keeps the exact prior payload and the exact prior text. It never
11240
+ # enters the gates line, never caps the score, never blocks convergence.
11241
+ # field truth (ADR-046): advisory and CONDITIONAL for the same reason -- a project that
11242
+ # declares no corpus and ran none keeps the exact prior payload and the exact prior text.
11243
+ # The `field` DIMENSION and its weight are deliberately NOT here: adding one moves every
11244
+ # existing project's score, and that is its own ADR.
11245
+ _field = _corpus_field(ledger)
11246
+ if _field:
11247
+ out["field"] = _field
11248
+ # smoke (ADR-047): conditional for the same reason. A failing smoke ALREADY blocks
11249
+ # through its `gate:smoke` record; this block adds what the rollup cannot carry --
11250
+ # which checks ran, and which of them answered wrong.
11251
+ _smoke = _smoke_report(ledger)
11252
+ if _smoke:
11253
+ out["smoke"] = _smoke
11254
+ _ao = _agent_origin_report(_ready_root,
11255
+ acc_path if acc_found else None)
11256
+ if _ao["n_unconfirmed"] or _ao["confirmed"]:
11257
+ out["agent_origin"] = {"unconfirmed": _ao["unconfirmed"],
11258
+ "confirmed": _ao["confirmed"],
11259
+ "files_scanned": _ao["files_scanned"]}
9860
11260
  if args.json:
9861
11261
  print(json.dumps(out, indent=2, ensure_ascii=False))
9862
11262
  return
@@ -9890,8 +11290,9 @@ def cmd_readiness(args):
9890
11290
  "or the explicit weight in config.defaults.readiness_weights")
9891
11291
  if narrated_only:
9892
11292
  print(f" ! narrated-only: {', '.join(narrated_only)} — checkbox ticked "
9893
- f"WITHOUT a green 'AC-n' testcase in the reports (measured beats "
9894
- f"narrated: does NOT close)")
11293
+ f"WITHOUT a green 'AC-n' testcase in the reports, without a green "
11294
+ f"corpus run carrying it and without a green 'AC-n' smoke check "
11295
+ f"(measured beats narrated: does NOT close)")
9895
11296
  if measured_unchecked:
9896
11297
  print(f" · measured but unticked: {', '.join(measured_unchecked)} — there is "
9897
11298
  f"a green testcase; tick the checkbox if the criterion is done")
@@ -9957,13 +11358,52 @@ def cmd_readiness(args):
9957
11358
  gate_roll = out["gates"]
9958
11359
  if gate_roll:
9959
11360
  blocking = [g for g in gate_roll if g["blocking"]]
9960
- n_ok = len(gate_roll) - len(blocking)
11361
+ # ADR-043: an advisory NEVER joins the ok count. "3 ok" must mean three gates ran and
11362
+ # came back clean; folding a check the project never declared as a gate into that number
11363
+ # is the false clean the advisory verdict exists to refuse. The segment is conditional,
11364
+ # so a ledger with no advisory record prints exactly what it printed before.
11365
+ advisory = [g for g in gate_roll if g.get("advisory") and not g["blocking"]]
11366
+ n_ok = len(gate_roll) - len(blocking) - len(advisory)
11367
+ adv_str = (f" · {len(advisory)} advisory ("
11368
+ + ", ".join(f"{g['repo']}/{g['tool']}" for g in advisory) + ")"
11369
+ if advisory else "")
9961
11370
  hint = "" if args.verbose else " (readiness --verbose for the detail)"
9962
11371
  if blocking:
9963
11372
  names = ", ".join(f"{g['repo']}/{g['tool']}" for g in blocking)
9964
- print(f"--- gates: {n_ok} ok · {len(blocking)} blocking ({names}){hint}")
11373
+ print(f"--- gates: {n_ok} ok{adv_str} · {len(blocking)} blocking ({names}){hint}")
9965
11374
  else:
9966
- print(f"--- gates: {n_ok} ok, none blocking{hint}")
11375
+ print(f"--- gates: {n_ok} ok{adv_str}, none blocking{hint}")
11376
+ # ADR-046: the FIELD line, one per repo that declares a corpus or has run one. A failing
11377
+ # corpus ALREADY appears in the gates rollup above (it is a gate:corpus record like any
11378
+ # other); this line adds the number the rollup cannot carry -- what percentage of REAL
11379
+ # inputs the system gets right, against which declared budget.
11380
+ for _rname in sorted(_field):
11381
+ print(_corpus_field_line(_rname, _field[_rname]))
11382
+ # ADR-047: the SMOKE line, one per repo that ingested a report. Like the field line it
11383
+ # adds the numbers the gates rollup cannot carry -- how many checks ran, how many
11384
+ # answered wrong, and WHICH ones, so the failure is named instead of counted.
11385
+ for _rname in sorted(_smoke):
11386
+ print(_smoke_line(_rname, _smoke[_rname]))
11387
+ # ADR-044: its OWN line, deliberately outside the gates rollup. An unconfirmed
11388
+ # agent-origin item is a decision still owed to the human, not a gate that ran --
11389
+ # folding it into "N ok" or into "N blocking" would be the false clean ADR-043
11390
+ # refused, in the other direction. It caps nothing and blocks nothing.
11391
+ if _ao["n_unconfirmed"]:
11392
+ print(f"--- origin: {_ao['n_unconfirmed']} agent-origin item(s) unconfirmed"
11393
+ f" (spec-check names them)")
11394
+ # ADR-048: operability gets its own line, CONDITIONAL on a record existing -- a ledger that
11395
+ # never ran the check prints exactly what it printed before. The gates rollup above already
11396
+ # counts the record correctly (ok / blocking / advisory); what it cannot say is WHICH of the
11397
+ # four is the one to go and build, and "1 blocking (backend-api/gate:operability)" is
11398
+ # precisely the message that sends a human to read the source.
11399
+ _ops = [(rname, _latest_static_by_tool(rnode).get("gate:operability"))
11400
+ for rname, rnode in ledger["repos"].items()]
11401
+ _ops = [(rname, rec) for rname, rec in _ops if rec and rec.get("note")]
11402
+ for _rname, _op in _ops:
11403
+ _label = "operability" if len(_ops) == 1 else "operability %s" % _rname
11404
+ _state = (" (advisory)" if _op.get("advisory")
11405
+ else " (gate: FAIL)" if (_op.get("gated_reported") or 0) else " (gate)")
11406
+ print(f"--- {_label}: {_op['note']}{_state}")
9967
11407
  if not args.verbose:
9968
11408
  return
9969
11409
  print("--- dimensions (weight | raw | contribution) ---")
@@ -10023,6 +11463,14 @@ DEFAULT_COVERAGE_TOLERANCE = 5.0 # pct points the rebuilt coverage may drop
10023
11463
  # abstraction is INTENTIONALLY not weighted: the "new types" regex is a prose/AST proxy
10024
11464
  # that false-positives on Java records/DTOs, so it must not gate the band. It stays as an
10025
11465
  # advisory metric + flag only (distilled: hard caps gate, guessy proxies advise).
11466
+ #
11467
+ # ADVISORY BY DEFAULT (kit 2.1.0, ADR-043). Every budget below is the KIT'S OPINION, not the
11468
+ # project's requirement, and an opinion that exits 1 is a gate nobody declared. Until a project
11469
+ # declares at least one numeric budget AND `defaults.simplicity.gate: true`, the verdict is
11470
+ # reported and the exit code is 0. This is NOT INV-ADVISORY-01 (that invariant quarantines
11471
+ # LLM-class JUDGMENT; these proxies are deterministic and may gate the moment a human says so)
11472
+ # -- it is the provenance rule of 1.17.0 applied to an exit code: a default is an opinion, and
11473
+ # only a declaration is a requirement.
10026
11474
  SIMPLICITY_WEIGHTS = {
10027
11475
  "diff_size": 35, "nesting": 30, "net_growth": 20, "fan_out": 8, "blob": 7,
10028
11476
  }
@@ -10037,6 +11485,17 @@ SIMPLICITY_DEFAULTS = {
10037
11485
  "max_abstraction_density": 3.0, # new *types* per 100 added LOC
10038
11486
  "indent_width": 4,
10039
11487
  }
11488
+ # Keys under defaults.simplicity that are NOT budgets, so declaring one never satisfies the
11489
+ # "a gate needs a budget" rule: `indent_width` is a PARSING parameter and `gate` is the switch
11490
+ # itself. `gate: true` with nothing but these declared is a refusal, not a gate (ADR-043).
11491
+ _SIMPLICITY_NON_BUDGET = ("indent_width", "gate")
11492
+ # What `max_nesting` actually measures, said once and reused by every surface that prints it.
11493
+ # It is INDENTATION DEPTH over added lines, not AST nesting: a wrapped call argument, JSX, a
11494
+ # multi-line Java string or any deep continuation raises it without any control flow existing.
11495
+ # The kit does NOT make it language-aware (that needs a parser per stack, which this stdlib
11496
+ # engine will not have) -- it names the proxy instead, so a reader can discount it.
11497
+ _NESTING_PROXY_NOTE = ("indentation depth over added lines, NOT AST nesting -- continuation "
11498
+ "lines, JSX and multi-line literals inflate it")
10040
11499
  # code files only — docs, config, resources and generated trees are noise for a
10041
11500
  # code-simplicity gate. Broader than SOURCE_EXT (which is repo-typed for rebuild).
10042
11501
  _SIMPLICITY_CODE_EXT = {
@@ -10235,18 +11694,25 @@ def _rebuild_compare(args):
10235
11694
  # --------------------------------------------------------------------------- #
10236
11695
  # simplicity-check (the "Reduce" gate)
10237
11696
  # --------------------------------------------------------------------------- #
10238
- def _read_diff(args):
10239
- """Unified-diff text from --diff, --from-git, or stdin."""
11697
+ def _read_diff(args, detect_renames=False):
11698
+ """Unified-diff text from --diff, --from-git, or stdin.
11699
+
11700
+ detect_renames (2.2.0) adds `-M` to the `--from-git` command so git reports a rename AS a
11701
+ rename, whatever the caller's `diff.renames` config says. It is OPT-IN because the other
11702
+ readers of this helper count LINES (simplicity, waste, regression): collapsing a rename into
11703
+ a header would silently change the numbers they have been measuring for releases. gate-check
11704
+ is the caller that needs it -- a rename read as a delete/add pair is what made
11705
+ `git mv tests/a_test.py tests/b_test.py` block as a deleted test in the field."""
10240
11706
  if getattr(args, "diff", None):
10241
11707
  with open(args.diff, "r", encoding="utf-8", errors="replace") as fh:
10242
11708
  return fh.read()
10243
11709
  if getattr(args, "from_git", False):
10244
11710
  import subprocess
10245
11711
  base = args.base or "HEAD"
11712
+ cmd = ["git", "diff", "--unified=0"] + (["-M"] if detect_renames else []) + [base]
10246
11713
  try:
10247
11714
  return subprocess.run(
10248
- ["git", "diff", "--unified=0", base],
10249
- check=True, capture_output=True, text=True,
11715
+ cmd, check=True, capture_output=True, text=True,
10250
11716
  encoding="utf-8", errors="replace").stdout
10251
11717
  except Exception as exc: # noqa: BLE001
10252
11718
  print(f"[qa_ledger] git diff failed: {exc}", file=sys.stderr)
@@ -10449,8 +11915,9 @@ def _simplicity_score(m, b):
10449
11915
  def _simplicity_flags(m, b):
10450
11916
  f = []
10451
11917
  if m["max_nesting"] > b["max_nesting_depth"]:
10452
- f.append(f"nesting {m['max_nesting']} > {b['max_nesting_depth']} — "
10453
- f"aplanar: guard clauses / extraer función (CWE-1124)")
11918
+ f.append(f"max_nesting (indentation proxy) {m['max_nesting']} > "
11919
+ f"{b['max_nesting_depth']} — aplanar: guard clauses / extraer función "
11920
+ f"(CWE-1124). Proxy: {_NESTING_PROXY_NOTE}")
10454
11921
  if m["new_abstractions"] > b["max_new_abstractions"]:
10455
11922
  f.append(f"{m['new_abstractions']} tipos/capas nuevos > "
10456
11923
  f"{b['max_new_abstractions']} — ¿todos pedidos? "
@@ -10476,10 +11943,12 @@ def _simplicity_flags(m, b):
10476
11943
  def cmd_simplicity_check(args):
10477
11944
  b = dict(SIMPLICITY_DEFAULTS)
10478
11945
  declared = set() # presupuestos declarados por el humano (config o CLI)
11946
+ gate = False # ADR-043: solo lo enciende una DECLARACION, nunca un default
10479
11947
  if args.config and os.path.exists(args.config):
10480
11948
  cfg = _load(args.config).get("defaults", {}).get("simplicity", {})
10481
11949
  b.update({k: cfg[k] for k in b if k in cfg})
10482
- declared |= {k for k in b if k in cfg and k != "indent_width"}
11950
+ declared |= {k for k in b if k in cfg and k not in _SIMPLICITY_NON_BUDGET}
11951
+ gate = bool(cfg.get("gate"))
10483
11952
  for k in ("max_lines_added", "max_net_lines", "max_files_changed",
10484
11953
  "max_nesting_depth", "max_hunk_added", "max_new_abstractions",
10485
11954
  "indent_width"):
@@ -10491,34 +11960,55 @@ def cmd_simplicity_check(args):
10491
11960
  if args.max_abstraction_density is not None:
10492
11961
  b["max_abstraction_density"] = args.max_abstraction_density
10493
11962
  declared.add("max_abstraction_density")
11963
+ if getattr(args, "gate", False):
11964
+ gate = True
11965
+ # A gate with no budget is not a gate: it is the kit's opinion wearing an exit code, which
11966
+ # is exactly the defect ADR-043 exists to remove. Refuse BEFORE reading the diff -- a
11967
+ # misconfigured gate must not produce a score anyone could quote.
11968
+ if gate and not declared:
11969
+ print("[qa_ledger] invalid config: defaults.simplicity.gate is true (or --gate was "
11970
+ "passed) but no simplicity budget is declared — a gate with no budget is not a "
11971
+ "gate, only the kit's opinion with an exit code. Declare at least one of "
11972
+ "max_lines_added, max_net_lines, max_files_changed, max_nesting_depth, "
11973
+ "max_hunk_added, max_new_abstractions, max_abstraction_density in "
11974
+ "defaults.simplicity (or pass the matching --max-... flag), or set gate to false.",
11975
+ file=sys.stderr)
11976
+ sys.exit(2)
11977
+ mode = "gate" if gate else "advisory"
10494
11978
 
10495
11979
  m = _simplicity_metrics(_read_diff(args), b["indent_width"])
10496
11980
  score, dims = _simplicity_score(m, b)
10497
11981
  verdict = _simplicity_band(score)
10498
11982
  flags = _simplicity_flags(m, b)
11983
+ exit_code = 1 if (verdict == "OVERBUILT" and gate) else 0
10499
11984
 
10500
- out = {"score": score, "verdict": verdict, "weights": SIMPLICITY_WEIGHTS,
11985
+ out = {"score": score, "verdict": verdict, "mode": mode, "gate": gate,
11986
+ "weights": SIMPLICITY_WEIGHTS,
10501
11987
  "dimensions": {k: round(v, 3) for k, v in dims.items()},
10502
- "metrics": m, "budgets": b, "budgets_declared": sorted(declared),
11988
+ "metrics": m, "metrics_notes": {"max_nesting": _NESTING_PROXY_NOTE},
11989
+ "budgets": b, "budgets_declared": sorted(declared),
10503
11990
  "flags": flags}
10504
11991
  if args.json:
10505
11992
  print(json.dumps(out, indent=2, ensure_ascii=False))
10506
- sys.exit(0 if verdict != "OVERBUILT" else 1)
11993
+ sys.exit(exit_code)
10507
11994
 
10508
- print(f"SIMPLICITY: {score}/100 {verdict}")
11995
+ mode_str = ("declared gate" if gate else
11996
+ "advisory (declare budgets + defaults.simplicity.gate to make it block)")
11997
+ print(f"SIMPLICITY: {score}/100 — {verdict} ({mode_str})")
10509
11998
  print("--- metrics (value / budget · * = declared by the human) ---")
10510
11999
  rows = [
10511
12000
  ("lines_added", m["lines_added"], b["max_lines_added"], "max_lines_added"),
10512
12001
  ("net_lines", m["net_lines"], b["max_net_lines"], "max_net_lines"),
10513
12002
  ("files_changed", m["files_changed"], b["max_files_changed"], "max_files_changed"),
10514
- ("max_nesting", m["max_nesting"], b["max_nesting_depth"], "max_nesting_depth"),
12003
+ ("max_nesting (indentation proxy)", m["max_nesting"], b["max_nesting_depth"], "max_nesting_depth"),
10515
12004
  ("new_abstractions", m["new_abstractions"], b["max_new_abstractions"], "max_new_abstractions"),
10516
12005
  ("abstraction/100", m["abstraction_density"], b["max_abstraction_density"], "max_abstraction_density"),
10517
12006
  ("max_hunk_added", m["max_hunk_added"], b["max_hunk_added"], "max_hunk_added"),
10518
12007
  ]
10519
12008
  for name, val, bud, key in rows:
10520
12009
  mark = "*" if key in declared else ""
10521
- print(f" {name:17s} {str(val):>7s} / {bud}{mark}")
12010
+ print(f" {name:31s} {str(val):>7s} / {bud}{mark}")
12011
+ print(f" (max_nesting is a PROXY: {_NESTING_PROXY_NOTE})")
10522
12012
  if not declared:
10523
12013
  print(" (every budget is a kit default — an opinion, not a "
10524
12014
  "requirement: declare yours in config.defaults.simplicity)")
@@ -10535,7 +12025,11 @@ def cmd_simplicity_check(args):
10535
12025
  print(f" ! {fl}")
10536
12026
  else:
10537
12027
  print(" within budget — nothing to cut")
10538
- sys.exit(0 if verdict != "OVERBUILT" else 1)
12028
+ if verdict == "OVERBUILT" and not gate:
12029
+ print("--- advisory: OVERBUILT is REPORTED, not enforced (exit 0). Cut what is cheap, "
12030
+ "say so in the PR body, and do not let it block the loop. To make it block, "
12031
+ "declare your budgets AND defaults.simplicity.gate: true ---")
12032
+ sys.exit(exit_code)
10539
12033
 
10540
12034
 
10541
12035
  # --------------------------------------------------------------------------- #
@@ -10978,8 +12472,125 @@ def _gc_new_dep(path, body):
10978
12472
  return bool(rx.search(body)) if rx else False
10979
12473
 
10980
12474
 
12475
+ def _gc_moves(diff):
12476
+ """Renames read as MOVES, never as deletions (2.2.0 field fix).
12477
+
12478
+ `git mv tests/a_test.py tests/b_test.py` used to be reported as a deleted test -- a BLOCKER
12479
+ and exit 1 for a change that deleted nothing. A rename reaches this parser in one of two
12480
+ shapes, and both are read here:
12481
+
12482
+ * git's own `rename from` / `rename to` headers, present when the producer detected
12483
+ renames (`--from-git` now forces `-M`, so the caller's `diff.renames` config can no
12484
+ longer hide one); and
12485
+ * an EXACT delete/add pair -- the same file content leaving one path and arriving at
12486
+ another inside the same diff. That is what a producer with rename detection OFF emits,
12487
+ and it is the shape that actually blocked in the field.
12488
+
12489
+ Returns (moves, paired). `moves` is the informational report. `paired` holds the paths of
12490
+ the exact pairs ONLY: their hunks say nothing about the change and are skipped. A rename
12491
+ WITH edits keeps its hunks, because moving a file is not a deletion but deleting a test out
12492
+ of a moved file still is -- and that verdict must not change.
12493
+
12494
+ The pairing is deliberately EXACT and one-to-one: same content, one file losing it, one file
12495
+ gaining it. Two deleted files with identical bodies are ambiguous, so neither is paired --
12496
+ guessing which moved where would be inventing a fact to clear a gate, which is the one thing
12497
+ this gate exists to refuse."""
12498
+ moves = []
12499
+ deleted, added = {}, {} # path -> tuple of line bodies
12500
+ path = None # the whole-file side currently being collected
12501
+ side = None # "-" while inside a deletion, "+" inside an addition
12502
+ bodies = []
12503
+ minus_path = None
12504
+
12505
+ def _flush():
12506
+ if path is not None and bodies:
12507
+ (deleted if side == "-" else added)[path] = tuple(bodies)
12508
+
12509
+ for raw in diff.splitlines():
12510
+ if raw.startswith("diff --git"):
12511
+ _flush()
12512
+ path, side, bodies, minus_path = None, None, [], None
12513
+ continue
12514
+ if raw.startswith("rename from "):
12515
+ minus_path = raw[len("rename from "):].strip()
12516
+ continue
12517
+ if raw.startswith("rename to "):
12518
+ if minus_path:
12519
+ moves.append("%s -> %s" % (minus_path, raw[len("rename to "):].strip()))
12520
+ minus_path = None
12521
+ continue
12522
+ if raw.startswith("--- "):
12523
+ p = raw[4:].strip().split("\t")[0]
12524
+ if p == "/dev/null":
12525
+ side = "+"
12526
+ else:
12527
+ minus_path = p[2:] if p[:2] in ("a/", "b/") else p
12528
+ continue
12529
+ if raw.startswith("+++ "):
12530
+ p = raw[4:].strip().split("\t")[0]
12531
+ if p == "/dev/null":
12532
+ side, path = "-", minus_path
12533
+ elif side == "+":
12534
+ path = p[2:] if p[:2] in ("a/", "b/") else p
12535
+ else:
12536
+ path, side = None, None # an ordinary edit: neither half of a move
12537
+ bodies = []
12538
+ continue
12539
+ if path is not None and side and raw.startswith(side):
12540
+ bodies.append(raw[1:])
12541
+ _flush()
12542
+
12543
+ paired = set()
12544
+ for dpath, content in deleted.items():
12545
+ hits = [a for a, c in added.items() if c == content]
12546
+ if len(hits) != 1:
12547
+ continue
12548
+ if sum(1 for c in deleted.values() if c == content) != 1:
12549
+ continue
12550
+ moves.append("%s -> %s" % (dpath, hits[0]))
12551
+ paired.add(dpath)
12552
+ paired.add(hits[0])
12553
+ return sorted(set(moves)), paired
12554
+
12555
+
12556
+ def _gc_scope(args, ledger):
12557
+ """`--repo R` SCOPES the diff to the files under repos[R].path (2.2.0 field fix).
12558
+
12559
+ In a monorepo one `git diff` carries every repo's hunks, and gate-check reported all of them
12560
+ under whichever repo was named: a fact about someone ELSE's code, attributed to yours, with
12561
+ your exit code behind it. Returns (base, scope) as absolute directories, or None when there
12562
+ is nothing to scope by (no --repo, or a scope that is the whole tree).
12563
+
12564
+ realpath on BOTH sides before comparing. On Windows a path under a username longer than 8
12565
+ characters comes back short-formed (`RUNNER~1`) from one API and long-formed from another,
12566
+ and a file INSIDE the tree is then judged outside it -- the CI-only failure this repo has
12567
+ already paid for once. The scope directory is realpath'd; the diff path is joined onto an
12568
+ already-realpath'd base rather than realpath'd itself, because a DELETED file no longer
12569
+ exists and would resolve inconsistently."""
12570
+ if ledger is None or not getattr(args, "repo", None):
12571
+ return None
12572
+ base = os.path.realpath(os.path.dirname(os.path.abspath(args.ledger)) or ".")
12573
+ scope = os.path.realpath(os.path.join(base, _scope_path(ledger, args.repo)))
12574
+ return None if scope == base else (base, scope)
12575
+
12576
+
12577
+ def _gc_in_scope(path, scope):
12578
+ if scope is None:
12579
+ return True
12580
+ base, root = scope
12581
+ full = os.path.normpath(os.path.join(base, path.replace("/", os.sep)))
12582
+ return full == root or full.startswith(root + os.sep)
12583
+
12584
+
10981
12585
  def cmd_gate_check(args):
10982
- diff = _read_diff(args)
12586
+ # --repo now does TWO things, and both need the ledger: it scopes the diff to that repo's
12587
+ # path (_gc_scope) and it adds the measured snapshot cross-check below. Loading it here
12588
+ # keeps the existing behaviour of an unreadable ledger or an unknown repo name exiting 2
12589
+ # rather than being scoped to nothing in silence.
12590
+ ledger = _load(args.ledger) if getattr(args, "repo", None) else None
12591
+ scope = _gc_scope(args, ledger)
12592
+ diff = _read_diff(args, detect_renames=True)
12593
+ moves, paired = _gc_moves(diff)
10983
12594
  removed_tests, disabled_tests, suppressions, thresholds = [], [], [], []
10984
12595
  secrets, secret_literals, scrub_edits, new_deps = [], [], [], []
10985
12596
  assertions_removed = 0
@@ -11010,12 +12621,16 @@ def cmd_gate_check(args):
11010
12621
  path = minus_path
11011
12622
  else:
11012
12623
  path = p[2:] if p[:2] in ("a/", "b/") else p
11013
- if _GC_KEYFILE.search(path):
11014
- secrets.append(f"{path}: contenedor de claves agregado/modificado")
12624
+ if path is not None and (path in paired or not _gc_in_scope(path, scope)):
12625
+ # a MOVED half (one side of an exact delete/add pair) and a file outside
12626
+ # --repo's scope are not this run's business: their hunks are skipped whole.
12627
+ path = None
12628
+ elif path is not None and p != "/dev/null" and _GC_KEYFILE.search(path):
12629
+ secrets.append(f"{path}: contenedor de claves agregado/modificado")
11015
12630
  elif raw.startswith("Binary files "):
11016
12631
  # los .p12/.jks binarios no traen +++ — el lado b/ vive en esta linea
11017
12632
  m = re.search(r" and b/(.+) differ$", raw)
11018
- if m and _GC_KEYFILE.search(m.group(1)):
12633
+ if m and _GC_KEYFILE.search(m.group(1)) and _gc_in_scope(m.group(1), scope):
11019
12634
  secrets.append(f"{m.group(1)}: contenedor de claves agregado/modificado (binario)")
11020
12635
  continue
11021
12636
  if not path:
@@ -11076,9 +12691,8 @@ def cmd_gate_check(args):
11076
12691
  # optional MEASURED cross-check (heuristic-independent): with --repo, compare the
11077
12692
  # last two snapshots' executed-test totals — a drop is a fact no regex can miss.
11078
12693
  test_count_drop = None
11079
- if getattr(args, "repo", None):
12694
+ if ledger is not None:
11080
12695
  try:
11081
- ledger = _load(args.ledger)
11082
12696
  node = _repo_node(ledger, args.repo)
11083
12697
  snaps = node.get("snapshots", [])
11084
12698
  if len(snaps) >= 2:
@@ -11112,10 +12726,15 @@ def cmd_gate_check(args):
11112
12726
  "new_dependencies": sorted(set(new_deps)),
11113
12727
  "assertions_removed": assertions_removed,
11114
12728
  "test_count_drop": test_count_drop,
12729
+ # informational, and deliberately OUTSIDE hard/soft: a move is neither a finding
12730
+ # nor an absolution, it is the reason a deletion is not being reported.
12731
+ "moved": moves,
12732
+ "scope": (os.path.basename(scope[1]) if scope else None),
11115
12733
  }, indent=2, ensure_ascii=False))
11116
12734
  sys.exit(1 if blocker else 0)
11117
12735
 
11118
- print(f"GATE-INTEGRITY: {verdict}")
12736
+ print(f"GATE-INTEGRITY: {verdict}"
12737
+ + (f" (scoped to {args.repo})" if scope else ""))
11119
12738
 
11120
12739
  def _show(label, items):
11121
12740
  if items:
@@ -11137,6 +12756,10 @@ def cmd_gate_check(args):
11137
12756
  print(f" ~ asserts removed from tests: {assertions_removed} (review)")
11138
12757
  if test_count_drop:
11139
12758
  print(f" ~ executed-test count dropped: {test_count_drop} (measured in snapshots — review)")
12759
+ if moves:
12760
+ tail = " ..." if len(moves) > 5 else ""
12761
+ print(f" . files moved: {len(moves)} — {'; '.join(moves[:5])}{tail} "
12762
+ f"(informational: a rename is not a deletion)")
11140
12763
  if verdict == "CLEAN":
11141
12764
  print(" the change does not weaken the measuring apparatus")
11142
12765
  elif not blocker:
@@ -11529,6 +13152,142 @@ def _lifecycle_for(root, adr_dir=None, spec_text=None, fallback=True):
11529
13152
  return _lifecycle_report(adr_dir or os.path.join(root, "docs", "adr"), spec_text)
11530
13153
 
11531
13154
 
13155
+ # --------------------------------------------------------------------------- #
13156
+ # agent-origin markers (ADR-044): the agent asks for DECISIONS, never for INFORMATION
13157
+ # --------------------------------------------------------------------------- #
13158
+ # A decision the human never made must not enter scope by rebound. Every acceptance
13159
+ # criterion, ADR decision item or HANDOFF rule the AGENT introduced carries one trailing
13160
+ # marker on its own line:
13161
+ #
13162
+ # (origin: agent) introduced by the agent, NOT confirmed
13163
+ # (origin: agent, confirmed: YYYY-MM-DD) a human confirmed THIS item, on that day
13164
+ #
13165
+ # No marker = human origin. That is the default on purpose: nothing existing is
13166
+ # retro-tagged, so the absence of a marker never has to be re-audited.
13167
+ #
13168
+ # ADVISORY, always. This section never changes an exit code and never caps readiness --
13169
+ # it reports what has not been confirmed yet, and the human decides. A gate here would
13170
+ # need an adopted budget (the 2.1.0 posture, ADR-043), and nobody has declared one.
13171
+ #
13172
+ # A marker inside a fenced block or an inline code span is DOCUMENTATION, not a decision:
13173
+ # the ADR and the ACCEPTANCE section that DEFINE this grammar quote it, and a scanner that
13174
+ # read its own definition as a finding would be measuring its own prose.
13175
+ _AO_MARK_RX = re.compile(r"origin:\s*agent\b(?P<rest>[^)\n]*)", re.I)
13176
+ _AO_CONF_RX = re.compile(r"confirmed:\s*(?P<value>[^,)\s]*)", re.I)
13177
+ _AO_ID_RX = re.compile(r"^[\s>*+-]*(?:\[[ xX]\]\s*)?(?:\d+[.)]\s*)?[*_]*"
13178
+ r"(?P<id>[A-Z][A-Z0-9]*(?:-[A-Z0-9]+)+)")
13179
+ _AO_INLINE_CODE_RX = re.compile(r"`[^`]*`")
13180
+ _AO_HTML_COMMENT_RX = re.compile(r"<!--.*?-->")
13181
+
13182
+
13183
+ def _agent_origin_scan_text(text, label):
13184
+ """Every agent-origin marker in one markdown file. Returns (unconfirmed, confirmed):
13185
+ a list of {id, file, line, malformed, detail} and a count. A `confirmed:` that is not
13186
+ a real YYYY-MM-DD date counts as UNCONFIRMED and is NAMED -- a typo must never read as
13187
+ a human's approval, which is the one failure this marker exists to prevent."""
13188
+ lines = text.split("\n")
13189
+ # A fence that never closes is a typo, not a decision to hide the rest of the file:
13190
+ # the lines after an unmatched opener are scanned as prose. Every OTHER fence still
13191
+ # hides its body, so the grammar's own definitions stay documentation.
13192
+ fenced, in_fence, opener = [False] * len(lines), False, None
13193
+ for idx, raw in enumerate(lines):
13194
+ st = raw.strip()
13195
+ if st.startswith("```") or st.startswith("~~~"):
13196
+ in_fence = not in_fence
13197
+ opener = idx if in_fence else None
13198
+ fenced[idx] = True
13199
+ continue
13200
+ fenced[idx] = in_fence
13201
+ if in_fence and opener is not None:
13202
+ for idx in range(opener, len(lines)):
13203
+ fenced[idx] = False
13204
+ unconfirmed, confirmed = [], 0
13205
+ for i, raw in enumerate(lines, 1):
13206
+ if fenced[i - 1]:
13207
+ continue
13208
+ line = _AO_HTML_COMMENT_RX.sub("", _AO_INLINE_CODE_RX.sub("", raw))
13209
+ idm = _AO_ID_RX.match(line)
13210
+ item = idm.group("id") if idm else "line %d" % i
13211
+ # every marker on the line, not the first: a confirmation appended after the
13212
+ # original tag must be read, never dropped in silence
13213
+ for m in _AO_MARK_RX.finditer(line):
13214
+ cm = _AO_CONF_RX.search(m.group("rest") or "")
13215
+ if cm is None:
13216
+ unconfirmed.append({"id": item, "file": label, "line": i,
13217
+ "malformed": False, "detail": None})
13218
+ elif _lc_valid_date(cm.group("value")):
13219
+ confirmed += 1
13220
+ else:
13221
+ unconfirmed.append({"id": item, "file": label, "line": i, "malformed": True,
13222
+ "detail": "malformed confirmed: %s"
13223
+ % (cm.group("value") or "(empty)")})
13224
+ return unconfirmed, confirmed
13225
+
13226
+
13227
+ def _agent_origin_report(root, acceptance=None, adr_dir=None, extra=()):
13228
+ """The advisory dimension over the files that hold decisions: the ACCEPTANCE file
13229
+ (the one named, else `<root>/ACCEPTANCE.md`), every ADR under `adr_dir`, `HANDOFF.md`
13230
+ when present, and whatever the caller already had open (`extra`). Absent files are
13231
+ simply not scanned -- there is nothing to report about a file that does not exist."""
13232
+ paths, seen = [], set()
13233
+
13234
+ def add(p):
13235
+ if not p:
13236
+ return
13237
+ try:
13238
+ key = os.path.realpath(p)
13239
+ except OSError:
13240
+ key = os.path.abspath(p)
13241
+ if key in seen or not os.path.isfile(p):
13242
+ return
13243
+ seen.add(key)
13244
+ paths.append(p)
13245
+
13246
+ for p in extra:
13247
+ add(p)
13248
+ add(acceptance or os.path.join(root, "ACCEPTANCE.md"))
13249
+ add(os.path.join(root, "HANDOFF.md"))
13250
+ adr = adr_dir or os.path.join(root, "docs", "adr")
13251
+ if os.path.isdir(adr):
13252
+ for f in sorted(glob.glob(os.path.join(adr, "*.md"))):
13253
+ add(f)
13254
+ unconfirmed, confirmed, scanned = [], 0, []
13255
+ for p in paths:
13256
+ try:
13257
+ with open(p, "r", encoding="utf-8", errors="replace") as fh:
13258
+ body = fh.read()
13259
+ except OSError:
13260
+ continue # unreadable is not a finding; it is simply not scanned
13261
+ # forward slashes always: the same tree must name the same file the same way on
13262
+ # Windows and on the CI cells, or a pinned line differs by separator alone.
13263
+ label = _lc_short(p).replace("\\", "/")
13264
+ scanned.append(label)
13265
+ u, c = _agent_origin_scan_text(body, label)
13266
+ unconfirmed += u
13267
+ confirmed += c
13268
+ return {"unconfirmed": unconfirmed, "confirmed": confirmed,
13269
+ "n_unconfirmed": len(unconfirmed), "files_scanned": scanned}
13270
+
13271
+
13272
+ def _agent_origin_names(ao, limit=6):
13273
+ """`AC-07 (ACCEPTANCE.md:41), D-03 (docs/adr/ADR-002-x.md:57)` -- the id, where it is,
13274
+ and for a malformed marker WHY it did not count as confirmed."""
13275
+ items = ao["unconfirmed"]
13276
+ out = ", ".join("%s (%s:%d%s)" % (x["id"], x["file"], x["line"],
13277
+ ", " + x["detail"] if x["detail"] else "")
13278
+ for x in items[:limit])
13279
+ if len(items) > limit:
13280
+ out += " +%d more" % (len(items) - limit)
13281
+ return out
13282
+
13283
+
13284
+ def _agent_origin_line(ao):
13285
+ """The one advisory line, shared by spec-check and readiness so the two surfaces
13286
+ cannot drift apart."""
13287
+ return ("origin: %d agent-origin item(s) unconfirmed -- %s"
13288
+ % (ao["n_unconfirmed"], _agent_origin_names(ao)))
13289
+
13290
+
11532
13291
  def _spec_check_text(text):
11533
13292
  lines = text.split("\n")
11534
13293
  n = len(lines)
@@ -11824,10 +13583,14 @@ def cmd_spec_check(args):
11824
13583
  else {"blockers": [], "untestable": [], "stack_hits": [],
11825
13584
  "non_ears": 0, "n_criteria": 0})
11826
13585
  # lifecycle (ADR-040): read-only and advisory -- it never touches `fail` below.
11827
- lc = _lifecycle_for(_lifecycle_root(args.spec[0] if args.spec else None,
11828
- args.acceptance),
11829
- getattr(args, "adr_dir", None), text,
13586
+ _root = _lifecycle_root(args.spec[0] if args.spec else None, args.acceptance)
13587
+ lc = _lifecycle_for(_root, getattr(args, "adr_dir", None), text,
11830
13588
  fallback=not args.spec)
13589
+ # agent-origin (ADR-044): read-only and advisory -- like lifecycle above, it never
13590
+ # touches `fail` below. An unconfirmed item is not a defect; it is a decision still
13591
+ # owed to the human.
13592
+ ao = _agent_origin_report(_root, args.acceptance,
13593
+ getattr(args, "adr_dir", None), extra=args.spec or ())
11831
13594
  structural = len(m["blockers"]) + len(acc_block) # estructura = FACT -> bloquea
11832
13595
  soft_find = len(m["untestable"]) + len(m["stack_hits"]) + len(acc_adv)
11833
13596
  fail = structural > 0 or (args.strict and soft_find > 0)
@@ -11836,7 +13599,9 @@ def cmd_spec_check(args):
11836
13599
  if args.json:
11837
13600
  print(json.dumps({"verdict": verdict, "advisory": structural == 0,
11838
13601
  "acceptance_blockers": acc_block,
11839
- "acceptance_advisory": acc_adv, "lifecycle": lc, **m},
13602
+ "acceptance_advisory": acc_adv, "lifecycle": lc,
13603
+ "agent_origin": {"unconfirmed": ao["unconfirmed"],
13604
+ "confirmed": ao["confirmed"]}, **m},
11840
13605
  indent=2, ensure_ascii=False))
11841
13606
  sys.exit(1 if fail else 0)
11842
13607
 
@@ -11862,6 +13627,10 @@ def cmd_spec_check(args):
11862
13627
  print(" %s %s %s - %s (%s)"
11863
13628
  % ("!" if c["status"] == "expires before go-live" else "~",
11864
13629
  c["component"], c.get("version") or "?", c["status"], c["detail"]))
13630
+ # conditional, like every other advisory line here: a tree with no marker prints
13631
+ # exactly what it printed before this release.
13632
+ if ao["n_unconfirmed"]:
13633
+ print(" ~ " + _agent_origin_line(ao))
11865
13634
  print(" i consistency: INFERENTIAL (an uncorrelated checker), not this lint · "
11866
13635
  "structure = FACT (blocks) · prose = advisory (--strict to gate)")
11867
13636
  if verdict == "OK":
@@ -12216,6 +13985,91 @@ def _doctor_hook_registered(settings_path):
12216
13985
  return next((n for n in HOOK_NAMES if n in blob), None)
12217
13986
 
12218
13987
 
13988
+ # --------------------------------------------------------------------------- #
13989
+ # installed-skill freshness (2.2.0)
13990
+ # --------------------------------------------------------------------------- #
13991
+ # The field case: skills sat under ~/.claude/skills/uscha-* dated before 1.54.0 while the kit
13992
+ # was 1.97.0. A whole discovery ran on the old prose and nothing said a word, because a SKILL.md
13993
+ # carried no version to compare. Since 2.2.0 the GENERATED orientation block opens with
13994
+ # `<!-- uscha kit: X.Y.Z ... -->` (tools/skill-blocks/, rendered by tools/gen-skill-blocks.py and
13995
+ # re-rendered by tools/release.py at every bump), so the comparison is mechanical.
13996
+ #
13997
+ # ADVISORY, always: this reports, it never gates. `doctor` already exits 1 only on errors, and an
13998
+ # outdated install is a WARN -- the operator may be pinning a version on purpose.
13999
+ SKILL_KIT_MARK = re.compile(r"uscha kit:\s*(\d+\.\d+\.\d+)")
14000
+ # Where install-uscha.py puts the skills: TARGETS = ("codex", "claude") + SKILL_ROOTS. Retyped
14001
+ # here rather than imported because install-uscha.py lives at the KIT ROOT and is not installed
14002
+ # alongside the engine -- an installed engine could not import it. Kept in one place so the
14003
+ # drift, if it ever happens, is one table against one table.
14004
+ SKILL_INSTALL_ROOTS = (
14005
+ ("claude", (".claude", "skills")),
14006
+ ("codex", ("plugins", "uscha", "skills")),
14007
+ ("pi", (".agents", "skills")),
14008
+ ("cursor", (".cursor", "skills")),
14009
+ ("copilot", (".copilot", "skills")),
14010
+ ("gemini", (".gemini", "skills")),
14011
+ ("cline", (".cline", "skills")),
14012
+ )
14013
+
14014
+
14015
+ def _semver_tuple(text):
14016
+ """(major, minor, patch) for comparison, or None when the string is not one."""
14017
+ m = re.match(r"^(\d+)\.(\d+)\.(\d+)$", (text or "").strip())
14018
+ return tuple(int(g) for g in m.groups()) if m else None
14019
+
14020
+
14021
+ def _installed_skill_report(root, kit_version):
14022
+ """One install root's uscha-* skills, each with the kit version its block was stamped with.
14023
+
14024
+ Returns None when the root holds no uscha skill at all -- "not installed" is a state, not a
14025
+ fault, and reporting it as an error would make `doctor` red on every machine that installed
14026
+ for one agent out of seven."""
14027
+ if not os.path.isdir(root):
14028
+ return None
14029
+ want = _semver_tuple(kit_version)
14030
+ found, oldest = [], None
14031
+ for name in USCHA_SKILLS:
14032
+ smd = os.path.join(root, name, "SKILL.md")
14033
+ if not os.path.isfile(smd):
14034
+ continue
14035
+ try:
14036
+ with open(smd, encoding="utf-8", errors="replace") as fh:
14037
+ head = fh.read(8192)
14038
+ except OSError:
14039
+ head = ""
14040
+ m = SKILL_KIT_MARK.search(head)
14041
+ seen = m.group(1) if m else None
14042
+ found.append({"skill": name, "installed": seen})
14043
+ got = _semver_tuple(seen)
14044
+ if got is not None and (oldest is None or got < oldest):
14045
+ oldest = got
14046
+ if not found:
14047
+ return None
14048
+ unmarked = [f["skill"] for f in found if f["installed"] is None]
14049
+ if unmarked:
14050
+ # no marker at all = a block rendered before 2.2.0. Older than anything that carries
14051
+ # one, and said that way rather than as a version nobody wrote.
14052
+ status = "outdated"
14053
+ installed = None
14054
+ elif want is None or oldest is None:
14055
+ status = "unknown"
14056
+ installed = ".".join(str(p) for p in oldest) if oldest else None
14057
+ else:
14058
+ installed = ".".join(str(p) for p in oldest)
14059
+ status = "outdated" if oldest < want else "current"
14060
+ return {"root": root, "status": status, "installed": installed, "kit": kit_version,
14061
+ "skills": found, "unmarked": unmarked}
14062
+
14063
+
14064
+ def _installed_skill_roots(args):
14065
+ """(target, root) pairs to inspect: the caller's `--installed` when given, else every root
14066
+ the installer knows, under the user's home."""
14067
+ if getattr(args, "installed", None):
14068
+ return [("--installed", d) for d in args.installed]
14069
+ home = os.path.expanduser("~")
14070
+ return [(t, os.path.join(home, *parts)) for t, parts in SKILL_INSTALL_ROOTS]
14071
+
14072
+
12219
14073
  def cmd_doctor(args):
12220
14074
  checks = [] # (nivel 'ok'|'warn'|'error', titulo, detalle)
12221
14075
 
@@ -12285,6 +14139,45 @@ def cmd_doctor(args):
12285
14139
  if mismatched:
12286
14140
  err(f"SKILL.md con frontmatter name distinto al directorio: {', '.join(mismatched)}")
12287
14141
 
14142
+ # --- installed skills vs the kit's VERSION (2.2.0) ----------------------
14143
+ # A discovery once ran on prose from 1.54.0 while the kit was 1.97.0, and nothing said so.
14144
+ # Advisory: reported as a warning, never an error, so it can never fail an installation
14145
+ # someone pinned on purpose.
14146
+ kit_version, version_src = _engine_kit_version()
14147
+ skills_installed = []
14148
+ absent = []
14149
+ for target, sroot in _installed_skill_roots(args):
14150
+ rep = _installed_skill_report(sroot, kit_version)
14151
+ if rep is None:
14152
+ absent.append((target, sroot))
14153
+ skills_installed.append({"target": target, "root": sroot,
14154
+ "status": "not installed", "installed": None,
14155
+ "kit": kit_version, "skills": [], "unmarked": []})
14156
+ continue
14157
+ rep["target"] = target
14158
+ skills_installed.append(rep)
14159
+ fix = (f"re-install: python install-uscha.py install --target {target} "
14160
+ f"(or `uscha install`)")
14161
+ if rep["status"] == "outdated":
14162
+ shown = rep["installed"] or "no `kit:` marker (block rendered before 2.2.0)"
14163
+ warn(f"SKILLS OUTDATED at {sroot}: installed {shown} < kit {kit_version}", fix)
14164
+ elif rep["status"] == "current":
14165
+ ok(f"skills {target}: current (kit {kit_version})", sroot)
14166
+ else:
14167
+ warn(f"skills {target}: version UNMEASURED at {sroot}",
14168
+ "the installed blocks carry a marker this engine cannot compare "
14169
+ f"(installed {rep['installed']!r}, kit {kit_version!r})")
14170
+ if absent:
14171
+ ok("skills not installed for: " + ", ".join(t for t, _ in absent),
14172
+ "not an error -- the kit installs per agent, one target at a time")
14173
+ if kit_version is None:
14174
+ warn("kit version not readable: installed-skill freshness is UNMEASURED",
14175
+ "the comparison needs a `%s`, a `VERSION` or an `uscha-install.json` at or "
14176
+ "above the engine; looked in: " % KIT_VERSION_COPY
14177
+ + ", ".join(version_src or []) + " -- re-install with "
14178
+ "`python install-uscha.py install` (2.2.0 and later copy it beside the "
14179
+ "installed skills)")
14180
+
12288
14181
  # --- hook INV-GOLDEN-01 -------------------------------------------------
12289
14182
  kit_root = os.path.abspath(os.path.join(engine_dir, "..", "..", ".."))
12290
14183
  hook_dirs = [os.path.join(os.path.expanduser("~"), ".claude", "hooks"),
@@ -12461,6 +14354,10 @@ def cmd_doctor(args):
12461
14354
  # effective settings + origin per knob (2.0.0); null when there is
12462
14355
  # no project config here to resolve them from
12463
14356
  "risk_profile": risk_profile, "effective": effective,
14357
+ # installed-skill freshness (2.2.0): one row per install root the
14358
+ # installer knows, with BOTH versions -- advisory, never in the verdict
14359
+ "kit_version": kit_version,
14360
+ "skills_installed": skills_installed,
12464
14361
  "checks": [{"level": lv, "title": t, "detail": d}
12465
14362
  for lv, t, d in checks]},
12466
14363
  indent=2, ensure_ascii=True))
@@ -12489,6 +14386,11 @@ def build_parser():
12489
14386
  "python/git, skills, hook, project config, per-repo toolchains")
12490
14387
  pdoc.add_argument("--config", default=None,
12491
14388
  help="project config to inspect (default: ./uscha.config.json)")
14389
+ pdoc.add_argument("--installed", action="append", default=None, metavar="DIR",
14390
+ help="skill install root to compare against the kit's VERSION "
14391
+ "(repeatable; default: every root install-uscha.py writes to, "
14392
+ "under ~). Advisory -- an outdated install is a warning, never "
14393
+ "an error")
12492
14394
  pdoc.add_argument("--ledger", default=DEFAULT_LEDGER)
12493
14395
  pdoc.add_argument("--json", action="store_true")
12494
14396
  pdoc.set_defaults(func=cmd_doctor)
@@ -12510,9 +14412,21 @@ def build_parser():
12510
14412
  pri.add_argument("--json", action="store_true")
12511
14413
  pri.set_defaults(func=cmd_rubric_ingest)
12512
14414
 
12513
- pi = sub.add_parser("init", help="create the ledger from a config file")
12514
- pi.add_argument("--config", required=True)
14415
+ pi = sub.add_parser("init", help="create the ledger from a config file, or --add-repo one "
14416
+ "repo into an existing ledger")
14417
+ pi.add_argument("--config", default=None,
14418
+ help="the uscha.config.json to freeze into a NEW ledger (required unless "
14419
+ "--add-repo); with --add-repo it is the file to mirror the new repo "
14420
+ "into (default: uscha.config.json next to --out)")
12515
14421
  pi.add_argument("--out", default=DEFAULT_LEDGER)
14422
+ pi.add_argument("--add-repo", dest="add_repo", default=None, metavar="NAME",
14423
+ help="append ONE repo to the EXISTING ledger at --out instead of creating a "
14424
+ "new one: every existing repo's steps, snapshots and iterations are "
14425
+ "left untouched and the checksum is re-sealed")
14426
+ pi.add_argument("--path", default=None, help="(with --add-repo) the new repo's path")
14427
+ pi.add_argument("--type", default=None, help="(with --add-repo) the new repo's type")
14428
+ pi.add_argument("--test-command", dest="test_command", default=None,
14429
+ help="(with --add-repo) the new repo's test command")
12516
14430
  pi.set_defaults(func=cmd_init)
12517
14431
 
12518
14432
  def add_ledger(sp):
@@ -12858,6 +14772,15 @@ def build_parser():
12858
14772
  help="override defaults.spec_drift.max_lag_days (default 30)")
12859
14773
  psd.add_argument("--json", action="store_true")
12860
14774
  psd.set_defaults(func=cmd_spec_drift)
14775
+
14776
+ pop = sub.add_parser("operability",
14777
+ help="measure CI / release / RUNBOOK / seed as FACTS in the tree "
14778
+ "(ADR-048); exit 0 always -- the gate is the profile's")
14779
+ add_ledger(pop)
14780
+ pop.add_argument("--repo", required=True)
14781
+ pop.add_argument("--iteration", type=int, default=1)
14782
+ pop.add_argument("--json", action="store_true")
14783
+ pop.set_defaults(func=cmd_operability)
12861
14784
  pre = sub.add_parser("resolve-escalation",
12862
14785
  help="close open escalations for a repo (recorded event; "
12863
14786
  "lifts the readiness cap)")
@@ -12868,20 +14791,82 @@ def build_parser():
12868
14791
 
12869
14792
  plg = sub.add_parser(
12870
14793
  "log-gate",
12871
- help="persist a FACT-gate verdict (golden-diff/gate-check/pit-check/simplicity/regression) "
12872
- "so converged and readiness actually see it")
14794
+ help="persist a FACT-gate verdict (golden-diff/gate-check/pit-check/simplicity/"
14795
+ "regression/ci) so converged and readiness actually see it")
12873
14796
  add_ledger(plg)
12874
14797
  plg.add_argument("--repo", required=True)
12875
14798
  plg.add_argument("--iteration", type=int, required=True)
12876
14799
  plg.add_argument("--kind", required=True,
12877
14800
  choices=["golden-diff", "gate-check", "pit-check", "simplicity",
12878
- "regression", "rubric", "waste"])
12879
- plg.add_argument("--verdict", required=True, choices=["pass", "fail", "not-run"])
14801
+ "regression", "rubric", "waste", "ci", "corpus", "smoke",
14802
+ "operability"],
14803
+ help="ci (2.2.0) records a pipeline run as the FACT it is: a fail caps "
14804
+ "readiness <=65 and blocks convergence exactly like gate-check. "
14805
+ "corpus (ADR-046) is the parity door for a field-truth run measured "
14806
+ "elsewhere; corpus-run writes the same record with the evidence on "
14807
+ "it. smoke (ADR-047) is the same door for a smoke run measured "
14808
+ "elsewhere -- a FACT kind, so advisory is refused on it. "
14809
+ "operability (ADR-048) is accepted for parity with the "
14810
+ "`operability` subcommand, which is what normally writes it")
14811
+ plg.add_argument("--verdict", required=True,
14812
+ choices=["pass", "fail", "advisory", "not-run"],
14813
+ help="advisory (ADR-043) records a measured, non-gating run: it never "
14814
+ "caps readiness, never blocks convergence, and never reads as ok; "
14815
+ "accepted only for --kind simplicity|waste|corpus|operability, a FACT "
14816
+ "gate refuses it")
12880
14817
  plg.add_argument("--count", type=int, default=1,
12881
14818
  help="failing finding count (fail only; default 1)")
12882
14819
  plg.add_argument("--note", default=None)
14820
+ plg.add_argument("--ref", default=None,
14821
+ help="where the verdict was measured (a CI run URL or id), stored on the "
14822
+ "record so the evidence outlives the conversation")
12883
14823
  plg.set_defaults(func=cmd_log_gate)
12884
14824
 
14825
+ pcr = sub.add_parser(
14826
+ "corpus-run",
14827
+ help="run a REAL-INPUT corpus against a command and persist gate:corpus (ADR-046): "
14828
+ "field truth for greenfield, where every test payload was invented by the agent")
14829
+ add_ledger(pcr)
14830
+ pcr.add_argument("--repo", required=True)
14831
+ pcr.add_argument("--corpus", default=None,
14832
+ help="JSONL corpus: one {\"input\": ..., \"expected\": ..., \"id\": ...} "
14833
+ "per line (default: repos[R].corpus from the config)")
14834
+ pcr.add_argument("--command", required=True,
14835
+ help="the command under test; each case's input arrives on ITS stdin "
14836
+ "(JSON-encoded when it is not a string)")
14837
+ pcr.add_argument("--threshold", type=float, default=None,
14838
+ help="hit percentage the run must reach to PASS (default: "
14839
+ "repos[R].corpus_threshold, else defaults.corpus_threshold; with "
14840
+ "NONE declared the run is advisory -- a gate needs an adopted budget)")
14841
+ pcr.add_argument("--ac", action="append", default=None,
14842
+ help="criterion id this run is evidence for (repeatable); a criterion "
14843
+ "whose only evidence is a corpus record closes MEASURED iff it passed")
14844
+ pcr.add_argument("--timeout", type=float, default=CORPUS_DEFAULT_TIMEOUT,
14845
+ help="per-case seconds before the case is a miss named `timeout` "
14846
+ "(default %d)" % CORPUS_DEFAULT_TIMEOUT)
14847
+ pcr.add_argument("--max-misses", type=int, default=CORPUS_DEFAULT_MAX_MISSES,
14848
+ help="how many misses to report and persist (default %d)"
14849
+ % CORPUS_DEFAULT_MAX_MISSES)
14850
+ pcr.add_argument("--iteration", type=int, default=1)
14851
+ pcr.add_argument("--json", action="store_true")
14852
+ pcr.set_defaults(func=cmd_corpus_run)
14853
+
14854
+ psi = sub.add_parser(
14855
+ "smoke-ingest",
14856
+ help="ingest a smoke report as gate:smoke (ADR-047): the smoke run MEASURED "
14857
+ "instead of narrated -- evidence is executed, not written down")
14858
+ add_ledger(psi)
14859
+ psi.add_argument("--repo", required=True)
14860
+ psi.add_argument("--report", required=True,
14861
+ help='JSON the PROJECT writes with its own smoke tool: '
14862
+ '{"checks": [{"name": ..., "ok": true|false, "status": ..., '
14863
+ '"latency_ms": ..., "evidence": ...}, ...]}. A missing `checks`, '
14864
+ 'an empty list, or a check without a name or a boolean `ok` is '
14865
+ 'exit 2 naming it -- never a scored run')
14866
+ psi.add_argument("--iteration", type=int, default=1)
14867
+ psi.add_argument("--json", action="store_true")
14868
+ psi.set_defaults(func=cmd_smoke_ingest)
14869
+
12885
14870
  pfb = sub.add_parser(
12886
14871
  "flag-blocker",
12887
14872
  help="record (or --resolve) a CONSTITUTION/invariant breach as a BLOCKER "
@@ -13057,7 +15042,9 @@ def build_parser():
13057
15042
 
13058
15043
  ps2 = sub.add_parser(
13059
15044
  "simplicity-check",
13060
- help="Reduce gate: score diff minimality/complexity over a unified diff")
15045
+ help="Reduce gate: score diff minimality/complexity over a unified diff. Advisory "
15046
+ "by default; gates only with --gate or defaults.simplicity.gate AND at least "
15047
+ "one declared budget")
13061
15048
  ps2.add_argument("--diff", help="path to a unified diff (else --from-git or stdin)")
13062
15049
  ps2.add_argument("--from-git", action="store_true",
13063
15050
  help="run `git diff --unified=0 <base>` for the diff")
@@ -13073,6 +15060,10 @@ def build_parser():
13073
15060
  ps2.add_argument("--max-abstraction-density", dest="max_abstraction_density",
13074
15061
  type=float, default=None)
13075
15062
  ps2.add_argument("--indent-width", dest="indent_width", type=int)
15063
+ ps2.add_argument("--gate", action="store_true",
15064
+ help="make an OVERBUILT verdict exit 1 (same switch as "
15065
+ "defaults.simplicity.gate). Requires at least one declared budget: "
15066
+ "without one the run REFUSES with exit 2 (ADR-043)")
13076
15067
  ps2.add_argument("--json", action="store_true")
13077
15068
  ps2.set_defaults(func=cmd_simplicity_check)
13078
15069
 
@@ -13121,8 +15112,9 @@ def build_parser():
13121
15112
  pgc.add_argument("--ledger", default=DEFAULT_LEDGER,
13122
15113
  help="(with --repo) ledger for the measured test-count cross-check")
13123
15114
  pgc.add_argument("--repo", default=None,
13124
- help="optional: compare the last two snapshots' executed-test totals "
13125
- "(a measured drop flags REVIEW no regex can miss it)")
15115
+ help="SCOPE the diff to repos[R].path (a monorepo sibling's hunks are not "
15116
+ "this repo's findings) and compare the last two snapshots' "
15117
+ "executed-test totals (a measured drop flags REVIEW)")
13126
15118
  pgc.add_argument("--json", action="store_true")
13127
15119
  pgc.set_defaults(func=cmd_gate_check)
13128
15120