hedgemony 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hedgemony/__init__.py ADDED
@@ -0,0 +1,15 @@
1
+ """hedgemony -- find the things an AI made up.
2
+
3
+ A fabrication is a claim about the world that is false: a package that was never published, a
4
+ method that does not exist, a call that cannot be made. Every check here is decided by the
5
+ interpreter or by a package registry, never by a language model, so a finding is a fact rather
6
+ than a confidence score.
7
+ """
8
+ __version__ = "1.0.0"
9
+
10
+ from .scan import scan, FabricationClass # noqa: F401
11
+ from .contracts import check_contracts # noqa: F401
12
+ from .sandbox import Limits, DEFAULT_LIMITS # noqa: F401
13
+
14
+ __all__ = ["scan", "check_contracts", "Limits", "DEFAULT_LIMITS", "FabricationClass",
15
+ "__version__"]
hedgemony/__main__.py ADDED
@@ -0,0 +1,7 @@
1
+ """Lets `python -m hedgemony` work identically to the `hedgemony` command."""
2
+ import sys
3
+
4
+ from .cli import main
5
+
6
+ if __name__ == "__main__":
7
+ sys.exit(main())
hedgemony/_runner.py ADDED
@@ -0,0 +1,276 @@
1
+ """The child process. Runs inside the sandbox and never imported by the tool.
2
+
3
+ This file is executed as a script in a separate interpreter under kernel limits. It is the
4
+ only place where code under test is ever imported, and it exists as its own file so that
5
+ nothing here can be reached by accident from the parent.
6
+
7
+ The guards below are installed BEFORE the target is loaded, because importing a module runs
8
+ its top-level statements. By the time a target's own code is reachable, every guard is already
9
+ in place.
10
+
11
+ One line of output matters: a single `__HEDGEMONY__` line carrying JSON. Everything the target
12
+ prints is left alone on the surrounding lines, so a target that writes to stdout cannot
13
+ corrupt the result by printing something that looks like a report.
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import json
18
+ import sys
19
+
20
+
21
+ # Exit code the child uses when it stops itself for exceeding the memory ceiling. Distinct so
22
+ # the parent can tell "this run was cut short by memory" from any status the target produced.
23
+ MEMORY_EXIT = 93
24
+
25
+
26
+ class NetworkBlocked(OSError):
27
+ """Raised in place of any outbound connection."""
28
+
29
+
30
+ def _watch_memory(limit_mb):
31
+ """Stop this process if it goes over the ceiling, without waiting to be asked.
32
+
33
+ WHY THE CHILD WATCHES ITSELF AS WELL AS THE PARENT. No kernel memory limit is enforced on
34
+ every platform -- on macOS RLIMIT_AS, RLIMIT_DATA and RLIMIT_RSS were all measured taking a
35
+ 200 MB allocation under a 64 MB cap without complaint -- so the ceiling has to be enforced
36
+ in software. The parent samples the whole process group, which is what catches a run that
37
+ spawns children or stops responding, but sampling from outside costs a process per look and
38
+ so cannot be done very often. From inside, the same question is a single library call, so
39
+ it can be asked far more frequently and closes most of the gap between the parent's looks.
40
+
41
+ Neither guard replaces the other: this one cannot see memory held by child processes, and
42
+ the parent's cannot look often. Together the window in which a runaway goes unnoticed is
43
+ small, and it is still a window rather than a hard ceiling -- a real one needs a container.
44
+ """
45
+ if not limit_mb or limit_mb <= 0:
46
+ return
47
+ import os
48
+ import resource
49
+ import threading
50
+ import time
51
+
52
+ # `ru_maxrss` is bytes on macOS and kilobytes on Linux. Getting this backwards would make
53
+ # the guard either fire instantly or never, so it is chosen from the platform rather than
54
+ # assumed.
55
+ to_mb = (1.0 / (1024 * 1024)) if sys.platform == "darwin" else (1.0 / 1024)
56
+
57
+ def loop():
58
+ while True:
59
+ peak = resource.getrusage(resource.RUSAGE_SELF).ru_maxrss * to_mb
60
+ if peak > limit_mb:
61
+ sys.stderr.write(f"hedgemony: stopped at {peak:.0f} MB, over the "
62
+ f"{limit_mb:.0f} MB limit\n")
63
+ sys.stderr.flush()
64
+ os._exit(MEMORY_EXIT)
65
+ time.sleep(0.02)
66
+
67
+ threading.Thread(target=loop, daemon=True).start()
68
+
69
+
70
+ def _block_network():
71
+ """Refuse sockets before the target is loaded.
72
+
73
+ A contract example has no business opening a connection, and code that reaches for the
74
+ network during a check is either wrong or doing something the person running this tool did
75
+ not ask for. Refusing at the socket layer covers everything built on top of it, so no
76
+ separate rule is needed per library.
77
+
78
+ This is a guard against accident, not against a determined escape: code that is trying to
79
+ get out can reach the syscall by other means. Containment for hostile input is the
80
+ container, documented as such.
81
+ """
82
+ import socket
83
+
84
+ def deny(*_a, **_k):
85
+ raise NetworkBlocked("network access is disabled during checking")
86
+
87
+ for name in ("socket", "create_connection", "create_server", "socketpair",
88
+ "getaddrinfo", "gethostbyname", "gethostbyname_ex", "gethostbyaddr"):
89
+ if hasattr(socket, name):
90
+ setattr(socket, name, deny)
91
+
92
+ # `socket` is a thin wrapper over the built-in `_socket`, so leaving that reachable would
93
+ # let anything import it directly and open a connection with the wrapper untouched.
94
+ try:
95
+ import _socket
96
+ for name in ("socket", "getaddrinfo", "gethostbyname"):
97
+ if hasattr(_socket, name):
98
+ setattr(_socket, name, deny)
99
+ except ImportError:
100
+ pass
101
+
102
+ # The higher-level clients are built on the above and are already covered, but they cache
103
+ # references at import time, so any imported before now would keep working. Blocking them
104
+ # by name closes that.
105
+ for module_name, attributes in (("urllib.request", ("urlopen",)),
106
+ ("http.client", ("HTTPConnection", "HTTPSConnection"))):
107
+ module = sys.modules.get(module_name)
108
+ if module is not None:
109
+ for name in attributes:
110
+ if hasattr(module, name):
111
+ setattr(module, name, deny)
112
+
113
+
114
+ def _apply_limits():
115
+ """Set this process's resource limits, before the target is loaded.
116
+
117
+ WHY THESE ARE SET HERE RATHER THAN BY THE PARENT AT FORK TIME. `subprocess` offers a hook
118
+ that runs between fork and exec, and it is the obvious place for this -- but that hook runs
119
+ in a just-forked process and is documented as unsafe in any program with threads. This one
120
+ has threads: output is drained on them, because a target that fills a pipe would otherwise
121
+ deadlock. Rather than rely on the two never overlapping, the limits are applied here, in
122
+ ordinary code, before anything of the target's has run. Same effect, no fork-safety
123
+ question, and it works on platforms with no such hook at all.
124
+
125
+ Each limit is applied on its own. A platform that refuses one -- macOS refuses every memory
126
+ limit -- must never prevent the rest from being applied.
127
+ """
128
+ import os
129
+ import resource
130
+
131
+ try:
132
+ wanted = json.loads(os.environ.get("HEDGEMONY_LIMITS") or "{}")
133
+ except ValueError:
134
+ return
135
+ for name, value in (
136
+ ("RLIMIT_CPU", (wanted.get("cpu"), wanted.get("cpu", 0) + 1)),
137
+ ("RLIMIT_FSIZE", (wanted.get("fsize"),) * 2),
138
+ ("RLIMIT_NPROC", (wanted.get("nproc"),) * 2),
139
+ ("RLIMIT_NOFILE", (wanted.get("nofile"),) * 2),
140
+ ("RLIMIT_AS", (wanted.get("mem"),) * 2),
141
+ ):
142
+ limit = getattr(resource, name, None)
143
+ if limit is None or value[0] is None:
144
+ continue
145
+ try:
146
+ resource.setrlimit(limit, (int(value[0]), int(value[1])))
147
+ except (ValueError, OSError):
148
+ pass
149
+ os.umask(0o077)
150
+
151
+
152
+ def _emit(payload):
153
+ """Send the result on its own channel, never on stdout.
154
+
155
+ WHY NOT STDOUT. The target prints to stdout, and anything it prints sits in the same stream
156
+ the result would. A file that happens to print a line looking like a result -- by accident
157
+ or on purpose -- could otherwise be read as one. The parent opens a separate pipe and
158
+ passes its number in; nothing the target writes to stdout or stderr can reach it, so the
159
+ two can never be confused.
160
+ """
161
+ import os
162
+ body = json.dumps(payload).encode()
163
+ fd = os.environ.get("HEDGEMONY_RESULT_FD")
164
+ if fd is not None:
165
+ try:
166
+ os.write(int(fd), body)
167
+ os.close(int(fd))
168
+ return
169
+ except (OSError, ValueError):
170
+ pass # channel unusable; fall through rather than lose it
171
+ sys.stdout.write("\n__HEDGEMONY__" + body.decode() + "\n")
172
+ sys.stdout.flush()
173
+
174
+
175
+ def _load(path):
176
+ """Import the target file as a module under a name that cannot collide."""
177
+ import importlib.util
178
+ spec = importlib.util.spec_from_file_location("_hedgemony_target", path)
179
+ if spec is None or spec.loader is None:
180
+ raise ImportError(f"cannot load {path}")
181
+ mod = importlib.util.module_from_spec(spec)
182
+ sys.modules["_hedgemony_target"] = mod
183
+ spec.loader.exec_module(mod)
184
+ return mod
185
+
186
+
187
+ def _run_contracts(path):
188
+ """Execute every stated example in the file and report each one that did not hold.
189
+
190
+ The report is per example rather than per file. A function with four examples where one
191
+ fails is a different fact from a function that fails everywhere, and collapsing them would
192
+ throw away the part that tells you where to look.
193
+ """
194
+ import doctest
195
+
196
+ mod = _load(path)
197
+ finder = doctest.DocTestFinder(exclude_empty=True)
198
+ runner = doctest.DocTestRunner(verbose=False, optionflags=doctest.ELLIPSIS)
199
+
200
+ failures = []
201
+ examples = 0
202
+
203
+ class Collect(doctest.OutputChecker):
204
+ pass
205
+
206
+ for test in finder.find(mod):
207
+ if not test.examples:
208
+ continue
209
+ examples += len(test.examples)
210
+ # ONE NAMESPACE FOR THE WHOLE DOCSTRING, not one per example. Examples build on each
211
+ # other -- `>>> value = 2` and then `>>> value + 1` -- which is how doctest has always
212
+ # worked and how people ordinarily write them. Giving each example its own copy of the
213
+ # globals turned that into a NameError and reported correct code as a broken contract:
214
+ # a false alarm, on the most common shape there is. Each example is still RUN
215
+ # separately, because that is what locates a failure to one line; only the namespace is
216
+ # shared. `DocTest.__init__` copies whatever it is handed, so what each example bound
217
+ # is carried forward explicitly afterwards.
218
+ globs = test.globs.copy()
219
+ for example in test.examples:
220
+ single = doctest.DocTest([example], globs, test.name,
221
+ test.filename, test.lineno, test.docstring)
222
+ out = []
223
+ runner.run(single, out=out.append, clear_globs=False)
224
+ globs.update(single.globs)
225
+ if runner.failures:
226
+ runner.failures = 0
227
+ # `example.lineno` is relative to the docstring; `test.lineno` locates the
228
+ # docstring in the file. Reporting an absolute line is what makes the finding
229
+ # usable without the reader counting lines by hand.
230
+ base = (test.lineno or 0) + 1
231
+ failures.append({
232
+ "name": test.name,
233
+ "line": base + example.lineno,
234
+ "statement": example.source.strip(),
235
+ "expected": example.want.strip(),
236
+ "detail": "".join(out).strip()[-800:],
237
+ })
238
+ return {"kind": "contracts", "examples": examples, "failures": failures}
239
+
240
+
241
+ def main():
242
+ if len(sys.argv) < 3:
243
+ _emit({"error": "usage: _runner.py <target> <mode>"})
244
+ return 2
245
+ path, mode = sys.argv[1], sys.argv[2]
246
+
247
+ import os
248
+ _apply_limits()
249
+ if os.environ.get("HEDGEMONY_ALLOW_NETWORK") != "1":
250
+ _block_network()
251
+
252
+ # Started before the target is loaded, because importing a module runs its top-level code
253
+ # and that is as capable of running away as anything in a function.
254
+ try:
255
+ _watch_memory(float(os.environ.get("HEDGEMONY_MEMORY_MB") or 0))
256
+ except (TypeError, ValueError):
257
+ pass
258
+
259
+ try:
260
+ if mode == "contracts":
261
+ _emit(_run_contracts(path))
262
+ else:
263
+ _emit({"error": f"unknown mode {mode!r}"})
264
+ return 2
265
+ except BaseException as exc: # noqa: BLE001
266
+ # An import that raises is a fact about the file worth reporting, not a crash to hide.
267
+ # It is reported as an error rather than as a failed contract, because the file never
268
+ # got far enough for its contracts to be tested at all.
269
+ _emit({"kind": "contracts", "error": f"{type(exc).__name__}: {exc}"[:400],
270
+ "examples": 0, "failures": []})
271
+ return 1
272
+ return 0
273
+
274
+
275
+ if __name__ == "__main__":
276
+ sys.exit(main())
hedgemony/board.py ADDED
@@ -0,0 +1,113 @@
1
+ """Rank sources of code by how much of it does not exist.
2
+
3
+ WHY A RATE AND NOT A COUNT. One caught fabrication is an anecdote. Two hundred lines with four
4
+ fabrications and eight hundred lines with four are not the same thing, and only a rate says so.
5
+ Fabrications per hundred lines is what makes two directories -- two models, two versions, two
6
+ prompting strategies, last week and this week -- comparable at all.
7
+
8
+ WHAT A SOURCE IS. A directory. Point this at several and each is scored on its own, labelled by
9
+ its directory name. If the folders hold output from different models, this ranks models. If
10
+ they hold last month and this month, it ranks a change over time. If they hold one team's code
11
+ and a vendor's, it ranks those. Nothing here knows or cares which, because the measurement is
12
+ the same in every case.
13
+
14
+ NO MODEL IS CONTACTED, AND NOTHING IS GENERATED. This reads code that already exists on disk.
15
+ That is deliberate: a ranking tool that had to call an endpoint would need credentials, network
16
+ access and a configured provider before it could tell anybody anything, and it would only be
17
+ able to score models the person running it happened to have. Save output into folders and this
18
+ scores anything.
19
+
20
+ THE HONEST LIMITATION, WHICH THE OUTPUT ALSO STATES. A low rate can mean a source is careful,
21
+ or it can mean a source wrote less and attempted less. The rate measures fabrication, not
22
+ capability, and it will reward timidity if read alone. Line counts sit next to every rate so
23
+ that a source which scored well by saying very little is visible rather than hidden.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import os
28
+
29
+ from .cli import collect
30
+ from .contracts import check_contracts
31
+ from .sandbox import Limits
32
+ from .scan import scan
33
+
34
+ __all__ = ["rank", "Source"]
35
+
36
+
37
+ class Source:
38
+ """One scored collection of code."""
39
+
40
+ __slots__ = ("label", "path", "files", "lines", "fabrications", "rate",
41
+ "contracts_checked", "contracts_broken", "examples", "top",
42
+ "defects", "defect_rate")
43
+
44
+ def __init__(self, label, path):
45
+ self.label = label
46
+ self.path = path
47
+ self.files = self.lines = self.fabrications = 0
48
+ self.contracts_checked = self.contracts_broken = self.examples = 0
49
+ self.rate = 0.0
50
+ self.defects = 0
51
+ self.defect_rate = 0.0
52
+ self.top = []
53
+
54
+ def as_dict(self):
55
+ return {s: getattr(self, s) for s in self.__slots__}
56
+
57
+
58
+ def rank(paths, run_contracts=True, limits: Limits = None, interpreter=None, scans=None):
59
+ """Score each path and return the sources ordered best first.
60
+
61
+ Ties are broken by line count, longest first: two sources at zero are not equally
62
+ informative, and the one that produced more code while staying clean is the stronger
63
+ result. Without that rule an empty directory would win every board.
64
+
65
+ `interpreter` and `scans` carry the environment the code actually belongs to: `scans` holds
66
+ results already obtained there, and `interpreter` runs the contracts. They matter more here
67
+ than anywhere else. A board scored against the wrong interpreter cannot see a project's
68
+ libraries, so every name reached through one goes unexamined and the directory ranks CLEAN
69
+ for the reason that should have disqualified it -- and unlike a single file report, a
70
+ ranking gives the reader no obvious place to notice.
71
+ """
72
+ limits = limits or Limits()
73
+ scans = scans or {}
74
+ sources = []
75
+
76
+ for path in paths:
77
+ label = os.path.basename(os.path.abspath(path.rstrip(os.sep))) or path
78
+ source = Source(label, path)
79
+ counted = {}
80
+
81
+ for filename in collect([path]):
82
+ try:
83
+ result = scans.get(filename) or scan(filename)
84
+ except OSError:
85
+ continue
86
+ if "error" in result:
87
+ continue
88
+ source.files += 1
89
+ source.lines += result["lines"]
90
+ source.fabrications += len(result["findings"])
91
+ for finding in result["findings"]:
92
+ counted[finding["detail"]] = counted.get(finding["detail"], 0) + 1
93
+
94
+ if run_contracts and result["language"] == "python":
95
+ contract = check_contracts(filename, limits=limits, interpreter=interpreter)
96
+ if contract.checked:
97
+ source.contracts_checked += 1
98
+ source.examples += contract.examples
99
+ source.contracts_broken += len(contract.findings)
100
+
101
+ source.rate = (100.0 * source.fabrications / source.lines) if source.lines else 0.0
102
+ # A BROKEN CONTRACT IS A DEFECT FOUND, and the ordering has to say so. Ranking on
103
+ # fabrications alone put a source with two failing examples above one whose four
104
+ # examples all held, purely because neither had invented a name -- which reads as an
105
+ # endorsement of the worse code. The rank is therefore over everything this tool
106
+ # actually found, and both components stay visible in their own columns so the reason
107
+ # for a position is never hidden inside a single number.
108
+ source.defects = source.fabrications + source.contracts_broken
109
+ source.defect_rate = (100.0 * source.defects / source.lines) if source.lines else 0.0
110
+ source.top = sorted(counted.items(), key=lambda kv: -kv[1])[:3]
111
+ sources.append(source)
112
+
113
+ return sorted(sources, key=lambda s: (s.defect_rate, -s.lines))