greencheck 0.3.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- greencheck/__init__.py +70 -0
- greencheck/cli.py +262 -0
- greencheck/core.py +671 -0
- greencheck/mutate.py +430 -0
- greencheck/report.py +75 -0
- greencheck/skills/README.md +30 -0
- greencheck/skills/bare-zero/SKILL.md +80 -0
- greencheck/skills/dead-check/SKILL.md +74 -0
- greencheck/skills/dimension-scope/SKILL.md +63 -0
- greencheck/skills/measurement-or-decoration/SKILL.md +76 -0
- greencheck/skills/positive-control/SKILL.md +64 -0
- greencheck-0.3.0.dist-info/METADATA +379 -0
- greencheck-0.3.0.dist-info/RECORD +17 -0
- greencheck-0.3.0.dist-info/WHEEL +5 -0
- greencheck-0.3.0.dist-info/entry_points.txt +2 -0
- greencheck-0.3.0.dist-info/licenses/LICENSE +21 -0
- greencheck-0.3.0.dist-info/top_level.txt +1 -0
greencheck/__init__.py
ADDED
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
"""greencheck — discriminability testing for self-reported metrics."""
|
|
2
|
+
|
|
3
|
+
from .core import (
|
|
4
|
+
ALL_VERDICTS,
|
|
5
|
+
BARE_ZERO,
|
|
6
|
+
CONSTANT,
|
|
7
|
+
DEAD_GATE,
|
|
8
|
+
IDENTITY,
|
|
9
|
+
LIVENESS_BIT,
|
|
10
|
+
NO_DATA,
|
|
11
|
+
PASS,
|
|
12
|
+
AuditResult,
|
|
13
|
+
Finding,
|
|
14
|
+
ProbePair,
|
|
15
|
+
audit_gate,
|
|
16
|
+
audit_gate_counts,
|
|
17
|
+
audit_ledger,
|
|
18
|
+
discriminate,
|
|
19
|
+
get_path,
|
|
20
|
+
load_jsonl,
|
|
21
|
+
)
|
|
22
|
+
|
|
23
|
+
def _resolve_version() -> str:
|
|
24
|
+
"""Single source of truth for the version: pyproject.toml.
|
|
25
|
+
|
|
26
|
+
A second copy hardcoded here would drift from it sooner or later, and a
|
|
27
|
+
package that reports a version different from the one it was built as is
|
|
28
|
+
exactly the failure this whole tool is about. So there is one copy.
|
|
29
|
+
"""
|
|
30
|
+
try: # installed: read back the metadata the installer wrote
|
|
31
|
+
from importlib.metadata import version as _dist_version
|
|
32
|
+
|
|
33
|
+
return _dist_version("greencheck")
|
|
34
|
+
except Exception:
|
|
35
|
+
pass
|
|
36
|
+
try: # running from a source checkout
|
|
37
|
+
import pathlib
|
|
38
|
+
import re
|
|
39
|
+
|
|
40
|
+
pyproject = pathlib.Path(__file__).resolve().parent.parent / "pyproject.toml"
|
|
41
|
+
m = re.search(r'^version\s*=\s*"([^"]+)"', pyproject.read_text(encoding="utf-8"), re.M)
|
|
42
|
+
if m:
|
|
43
|
+
return m.group(1)
|
|
44
|
+
except Exception:
|
|
45
|
+
pass
|
|
46
|
+
return "0.0.0+unknown"
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
__version__ = _resolve_version()
|
|
50
|
+
|
|
51
|
+
__all__ = [
|
|
52
|
+
"ALL_VERDICTS",
|
|
53
|
+
"BARE_ZERO",
|
|
54
|
+
"CONSTANT",
|
|
55
|
+
"DEAD_GATE",
|
|
56
|
+
"IDENTITY",
|
|
57
|
+
"LIVENESS_BIT",
|
|
58
|
+
"NO_DATA",
|
|
59
|
+
"PASS",
|
|
60
|
+
"AuditResult",
|
|
61
|
+
"Finding",
|
|
62
|
+
"ProbePair",
|
|
63
|
+
"audit_gate",
|
|
64
|
+
"audit_gate_counts",
|
|
65
|
+
"audit_ledger",
|
|
66
|
+
"discriminate",
|
|
67
|
+
"get_path",
|
|
68
|
+
"load_jsonl",
|
|
69
|
+
"__version__",
|
|
70
|
+
]
|
greencheck/cli.py
ADDED
|
@@ -0,0 +1,262 @@
|
|
|
1
|
+
"""greencheck.cli — command line entry point.
|
|
2
|
+
|
|
3
|
+
greencheck audit examples/ledger.jsonl --field identity.identity_score --count memory_recall.sampled
|
|
4
|
+
greencheck gate examples/gate_events.jsonl --fire-event block_issued
|
|
5
|
+
greencheck demo
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import argparse
|
|
11
|
+
import json
|
|
12
|
+
import pathlib
|
|
13
|
+
import sys
|
|
14
|
+
|
|
15
|
+
from . import __version__
|
|
16
|
+
from .core import (
|
|
17
|
+
audit_gate,
|
|
18
|
+
audit_gate_daily,
|
|
19
|
+
audit_ledger,
|
|
20
|
+
get_path,
|
|
21
|
+
load_jsonl,
|
|
22
|
+
looks_like_daily_aggregate,
|
|
23
|
+
)
|
|
24
|
+
from .report import render_markdown, render_text, to_dicts
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _project(rows, field, count_field=None, fresh_field=None):
|
|
28
|
+
"""Flatten dotted paths into the flat keys core.audit_ledger expects."""
|
|
29
|
+
out = []
|
|
30
|
+
for r in rows:
|
|
31
|
+
item = {"value": get_path(r, field)}
|
|
32
|
+
if count_field:
|
|
33
|
+
item["_count"] = get_path(r, count_field)
|
|
34
|
+
if fresh_field:
|
|
35
|
+
item["_fresh"] = get_path(r, fresh_field)
|
|
36
|
+
out.append(item)
|
|
37
|
+
return out
|
|
38
|
+
|
|
39
|
+
|
|
40
|
+
def cmd_audit(args) -> int:
|
|
41
|
+
rows = load_jsonl(args.ledger)
|
|
42
|
+
if not rows:
|
|
43
|
+
print(f"no records in {args.ledger}", file=sys.stderr)
|
|
44
|
+
return 2
|
|
45
|
+
projected = _project(rows, args.field, args.count, args.fresh)
|
|
46
|
+
result = audit_ledger(
|
|
47
|
+
projected,
|
|
48
|
+
field="value",
|
|
49
|
+
subject=args.field,
|
|
50
|
+
count_field="_count" if args.count else None,
|
|
51
|
+
fresh_field="_fresh" if args.fresh else None,
|
|
52
|
+
)
|
|
53
|
+
if args.json:
|
|
54
|
+
print(json.dumps([result.to_dict()], indent=2, ensure_ascii=False))
|
|
55
|
+
else:
|
|
56
|
+
print(render_text([result], title=f"ledger audit — {args.ledger}"))
|
|
57
|
+
return 0 if result.ok else 1
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def cmd_gate(args) -> int:
|
|
61
|
+
events = load_jsonl(args.events)
|
|
62
|
+
if not events:
|
|
63
|
+
print(f"no events in {args.events}", file=sys.stderr)
|
|
64
|
+
return 2
|
|
65
|
+
|
|
66
|
+
# A per-day aggregate counts days; an event log counts events. Reading one
|
|
67
|
+
# as the other produces a confident number that means nothing — a sixteen
|
|
68
|
+
# day window reported as "16 samples". The dataset published in this
|
|
69
|
+
# repository is daily, so the distinction is checked before the audit runs.
|
|
70
|
+
if looks_like_daily_aggregate(events):
|
|
71
|
+
before, whole = audit_gate_daily(
|
|
72
|
+
events, fire_event=args.fire_event, subject=args.events
|
|
73
|
+
)
|
|
74
|
+
results = [before, whole]
|
|
75
|
+
title = f"gate audit — {args.events} (per-day aggregates)"
|
|
76
|
+
else:
|
|
77
|
+
results = [audit_gate(events, fire_event=args.fire_event, subject=args.events)]
|
|
78
|
+
title = f"gate audit — {args.events}"
|
|
79
|
+
|
|
80
|
+
if args.json:
|
|
81
|
+
print(json.dumps(to_dicts(results), indent=2, ensure_ascii=False))
|
|
82
|
+
else:
|
|
83
|
+
print(render_text(results, title=title))
|
|
84
|
+
return 0 if all(r.ok for r in results) else 1
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def cmd_demo(args) -> int:
|
|
88
|
+
"""Self-contained demonstration: one clean instrument, two broken ones."""
|
|
89
|
+
from .core import ProbePair, discriminate
|
|
90
|
+
|
|
91
|
+
# --- three instruments that all look plausible in a dashboard -----------
|
|
92
|
+
def content_length(x):
|
|
93
|
+
"""Honest: responds to the property it is named after."""
|
|
94
|
+
return len(str(x))
|
|
95
|
+
|
|
96
|
+
def content_presence_score(x):
|
|
97
|
+
"""Dishonest: named for content, responds only to whether input exists."""
|
|
98
|
+
return 1.0 if x else 0.0
|
|
99
|
+
|
|
100
|
+
def always_one(x):
|
|
101
|
+
"""The degenerate control: no input affects it."""
|
|
102
|
+
return 1.0
|
|
103
|
+
|
|
104
|
+
# --- probe pairs, tagged by the dimension they vary ---------------------
|
|
105
|
+
pairs = [
|
|
106
|
+
ProbePair(
|
|
107
|
+
"rich text vs degenerate text",
|
|
108
|
+
"the quick brown fox jumps",
|
|
109
|
+
"aaaa",
|
|
110
|
+
dimension="content",
|
|
111
|
+
note="both non-empty; content differs",
|
|
112
|
+
),
|
|
113
|
+
ProbePair(
|
|
114
|
+
"text present vs absent",
|
|
115
|
+
"hello",
|
|
116
|
+
"",
|
|
117
|
+
dimension="presence",
|
|
118
|
+
note="existence differs; content dimension held out",
|
|
119
|
+
),
|
|
120
|
+
]
|
|
121
|
+
|
|
122
|
+
results = [
|
|
123
|
+
discriminate(content_length, pairs, subject="content_length", claimed_dimensions=["content"]),
|
|
124
|
+
discriminate(content_presence_score, pairs, subject="content_presence_score", claimed_dimensions=["content"]),
|
|
125
|
+
discriminate(always_one, pairs, subject="always_one (control)", claimed_dimensions=["content"]),
|
|
126
|
+
]
|
|
127
|
+
print(render_text(results, title="greencheck demo — discriminate()"))
|
|
128
|
+
print()
|
|
129
|
+
print("Note: content_presence_score separates the presence pair (1/2 pairs")
|
|
130
|
+
print("separated overall) yet is still reported FAIL, because it collapses")
|
|
131
|
+
print("every pair in the dimension it claims to measure.")
|
|
132
|
+
print()
|
|
133
|
+
ledger = [
|
|
134
|
+
{"value": 1.0, "n": 5}, {"value": 1.0, "n": 5}, {"value": 1.0, "n": 5},
|
|
135
|
+
{"value": 1.0, "n": 0}, {"value": 1.0, "n": 0},
|
|
136
|
+
]
|
|
137
|
+
r = audit_ledger(ledger, field="value", count_field="n", subject="demo.identity_score")
|
|
138
|
+
print(render_text([r], title="greencheck demo — audit_ledger()"))
|
|
139
|
+
return 0
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _skills_root() -> pathlib.Path:
|
|
143
|
+
"""Locate the bundled skills directory.
|
|
144
|
+
|
|
145
|
+
Sits next to the package in a checkout. The search is forgiving so that an
|
|
146
|
+
installed copy still finds it wherever the data files landed.
|
|
147
|
+
"""
|
|
148
|
+
here = pathlib.Path(__file__).resolve().parent
|
|
149
|
+
for cand in (here.parent / "skills", here / "skills"):
|
|
150
|
+
if cand.is_dir():
|
|
151
|
+
return cand
|
|
152
|
+
return here.parent / "skills"
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def _frontmatter_description(path: pathlib.Path) -> str:
|
|
156
|
+
"""Pull `description:` out of a SKILL.md YAML frontmatter block."""
|
|
157
|
+
text = path.read_text(encoding="utf-8")
|
|
158
|
+
if not text.startswith("---"):
|
|
159
|
+
return ""
|
|
160
|
+
end = text.find("\n---", 3)
|
|
161
|
+
block = text[3:end] if end != -1 else text[3:]
|
|
162
|
+
for line in block.splitlines():
|
|
163
|
+
if line.lower().startswith("description:"):
|
|
164
|
+
return line.split(":", 1)[1].strip()
|
|
165
|
+
return ""
|
|
166
|
+
|
|
167
|
+
|
|
168
|
+
def cmd_skills(args) -> int:
|
|
169
|
+
"""List the bundled skills, or print one in full so an agent can read it."""
|
|
170
|
+
root = _skills_root()
|
|
171
|
+
if not root.is_dir():
|
|
172
|
+
print(f"no skills directory at {root}", file=sys.stderr)
|
|
173
|
+
return 2
|
|
174
|
+
entries = sorted(p for p in root.iterdir() if (p / "SKILL.md").is_file())
|
|
175
|
+
if not entries:
|
|
176
|
+
print(f"no skills found in {root}", file=sys.stderr)
|
|
177
|
+
return 2
|
|
178
|
+
|
|
179
|
+
if args.name:
|
|
180
|
+
target = root / args.name / "SKILL.md"
|
|
181
|
+
if not target.is_file():
|
|
182
|
+
print(f"no skill named {args.name!r}", file=sys.stderr)
|
|
183
|
+
print("available: " + ", ".join(p.name for p in entries), file=sys.stderr)
|
|
184
|
+
return 2
|
|
185
|
+
print(target.read_text(encoding="utf-8"))
|
|
186
|
+
return 0
|
|
187
|
+
|
|
188
|
+
if args.json:
|
|
189
|
+
print(json.dumps(
|
|
190
|
+
[{"name": p.name, "description": _frontmatter_description(p / "SKILL.md")}
|
|
191
|
+
for p in entries],
|
|
192
|
+
indent=2, ensure_ascii=False))
|
|
193
|
+
return 0
|
|
194
|
+
|
|
195
|
+
print(f"greencheck skills — {len(entries)} available\n")
|
|
196
|
+
width = max(len(p.name) for p in entries)
|
|
197
|
+
for p in entries:
|
|
198
|
+
desc = _frontmatter_description(p / "SKILL.md")
|
|
199
|
+
if len(desc) > 96:
|
|
200
|
+
desc = desc[:93] + "..."
|
|
201
|
+
print(f" {p.name:<{width}} {desc}")
|
|
202
|
+
print()
|
|
203
|
+
print("Read one in full: greencheck skills <name>")
|
|
204
|
+
print("These are plain SKILL.md files with YAML frontmatter, the format used")
|
|
205
|
+
print("by most agent harnesses. Point your agent at the skills/ directory.")
|
|
206
|
+
return 0
|
|
207
|
+
|
|
208
|
+
|
|
209
|
+
def cmd_mutate(args) -> int:
|
|
210
|
+
"""Delegate to the mutation engine, which owns its own argument parser.
|
|
211
|
+
|
|
212
|
+
`mutate` takes a gate command plus its own flags, so the subparser collects
|
|
213
|
+
everything after it verbatim rather than trying to re-declare those flags.
|
|
214
|
+
"""
|
|
215
|
+
from .mutate import main as mutate_main
|
|
216
|
+
|
|
217
|
+
return mutate_main(args.args)
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
def main(argv=None) -> int:
|
|
221
|
+
ap = argparse.ArgumentParser(prog="greencheck", description=__doc__.splitlines()[0])
|
|
222
|
+
ap.add_argument("--version", action="version", version=f"greencheck {__version__}")
|
|
223
|
+
sub = ap.add_subparsers(dest="cmd", required=True)
|
|
224
|
+
|
|
225
|
+
a = sub.add_parser("audit", help="audit a recorded metric ledger")
|
|
226
|
+
a.add_argument("ledger")
|
|
227
|
+
a.add_argument("--field", required=True, help="dotted path to the value")
|
|
228
|
+
a.add_argument("--count", help="dotted path to the number of items examined")
|
|
229
|
+
a.add_argument("--fresh", help="dotted path to a boolean input to attribute variance to")
|
|
230
|
+
a.add_argument("--json", action="store_true")
|
|
231
|
+
a.set_defaults(func=cmd_audit)
|
|
232
|
+
|
|
233
|
+
g = sub.add_parser("gate", help="check whether an assertion has ever fired")
|
|
234
|
+
g.add_argument("events")
|
|
235
|
+
g.add_argument("--fire-event", default="block_issued")
|
|
236
|
+
g.add_argument("--json", action="store_true")
|
|
237
|
+
g.set_defaults(func=cmd_gate)
|
|
238
|
+
|
|
239
|
+
d = sub.add_parser("demo", help="run the built-in demonstration")
|
|
240
|
+
d.set_defaults(func=cmd_demo)
|
|
241
|
+
|
|
242
|
+
s = sub.add_parser("skills", help="list the bundled agent skills, or print one in full")
|
|
243
|
+
s.add_argument("name", nargs="?", help="skill name to print in full")
|
|
244
|
+
s.add_argument("--json", action="store_true")
|
|
245
|
+
s.set_defaults(func=cmd_skills)
|
|
246
|
+
|
|
247
|
+
m = sub.add_parser("mutate", help="does your gate actually say no? mutate the input, run the gate")
|
|
248
|
+
m.set_defaults(func=None)
|
|
249
|
+
|
|
250
|
+
# `mutate` takes its own flags (--gate, --target, ...). REMAINDER does not
|
|
251
|
+
# capture option-likes, so parse known args here and hand the rest to the
|
|
252
|
+
# mutation engine, which owns that argument surface.
|
|
253
|
+
args, rest = ap.parse_known_args(argv)
|
|
254
|
+
if getattr(args, "cmd", None) == "mutate":
|
|
255
|
+
from .mutate import main as mutate_main
|
|
256
|
+
|
|
257
|
+
return mutate_main(rest)
|
|
258
|
+
return args.func(args)
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
if __name__ == "__main__":
|
|
262
|
+
raise SystemExit(main())
|