vouch-paper 0.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
vouch/__init__.py ADDED
@@ -0,0 +1,82 @@
1
+ """vouch: every number in your paper, vouched for by the code that produced it.
2
+
3
+ Record what an experiment function returns, keyed by its arguments::
4
+
5
+ import vouch
6
+
7
+ @vouch.track(over="seed")
8
+ def evaluate(dataset, model, seed=0):
9
+ ...
10
+ return {"acc": acc, "loss": loss}
11
+
12
+ or record values directly::
13
+
14
+ vouch.record("cifar.resnet.acc", acc, desc="top-1 test accuracy")
15
+ vouch.record_all(metrics, prefix="cifar.resnet")
16
+
17
+ Compute numbers *from* results in ``vouch_values.py``; ``vouch build`` evaluates it::
18
+
19
+ @vouch.derive("cifar.gap", fmt=".1f", desc="ResNet minus ViT, points")
20
+ def gap(v):
21
+ return 100 * (v["cifar.resnet.acc.mean"] - v["cifar.vit.acc.mean"])
22
+
23
+ then cite them in LaTeX as ``\\vouch{cifar.gap}``. See SPEC.md.
24
+
25
+ The public functions load on first use, so ``import vouch`` stays cheap: the
26
+ command line (and the Claude Code hook, run after every edit) pays only for what
27
+ it uses.
28
+ """
29
+
30
+ import os as _os
31
+ import sys as _sys
32
+
33
+ try:
34
+ from ._version import __version__
35
+ except ImportError:
36
+ __version__ = "0+unknown"
37
+
38
+ _PUBLIC = {
39
+ "Run": "api", "active_run": "api", "artifact": "api", "claim": "api", "input": "api",
40
+ "params": "api", "record": "api", "record_all": "api", "run": "api", "table": "api",
41
+ "alias": "derived", "derive": "derived", "expect": "derived", "track": "tracked",
42
+ "Stat": "values", "Verdict": "verdict", "all_of": "verdict", "any_of": "verdict",
43
+ "approx": "verdict", "between": "verdict", "ge": "verdict", "gt": "verdict",
44
+ "le": "verdict", "lt": "verdict",
45
+ }
46
+ __all__ = sorted(_PUBLIC) + ["__version__"]
47
+
48
+
49
+ def __getattr__(name: str):
50
+ mod = _PUBLIC.get(name)
51
+ if mod is None:
52
+ raise AttributeError(f"module 'vouch' has no attribute {name!r}")
53
+ from importlib import import_module
54
+ value = getattr(import_module(f"{__name__}.{mod}"), name)
55
+ globals()[name] = value
56
+ return value
57
+
58
+
59
+ def __dir__():
60
+ return sorted(set(globals()) | set(_PUBLIC))
61
+
62
+
63
+ def _is_cli() -> bool:
64
+ """True when this process is the ``vouch`` command itself, which runs no experiment."""
65
+ prog = _os.path.basename(_sys.argv[0] if _sys.argv else "").lower()
66
+ if prog in ("vouch", "vouch.exe", "vouch-script.py"):
67
+ return True
68
+ argv = list(getattr(_sys, "orig_argv", []))
69
+ return "-m" in argv and argv.index("-m") + 1 < len(argv) and \
70
+ argv[argv.index("-m") + 1] in ("vouch", "vouch.cli")
71
+
72
+
73
+ if not _is_cli():
74
+ # Track which functions the experiment executes, from the moment vouch is
75
+ # imported (SPEC 8.2), and record figures saved with matplotlib (SPEC 4.7).
76
+ from .tracing import tracker as _tracker
77
+
78
+ _tracker.start()
79
+
80
+ from . import figures as _figures
81
+
82
+ _figures.install()
vouch/__main__.py ADDED
@@ -0,0 +1,4 @@
1
+ from ._entry import main
2
+ import sys
3
+
4
+ sys.exit(main())
vouch/_entry.py ADDED
@@ -0,0 +1,48 @@
1
+ """The ``vouch`` command's entry point.
2
+
3
+ Claude Code runs ``vouch hook claude`` after *every* file edit, most of them not to
4
+ a ``.tex`` file. Those return here, having imported nothing but ``json``; only a
5
+ tex edit (and every other command) loads the real command line.
6
+ """
7
+
8
+ import sys
9
+
10
+
11
+ def main() -> int:
12
+ argv = sys.argv[1:]
13
+ if argv == ["hook", "claude"]:
14
+ data = "" if sys.stdin is None or sys.stdin.isatty() else sys.stdin.read()
15
+ if ".tex" not in data: # not a tex edit: nothing to check
16
+ return 0
17
+ from .edithook import run_edit_hook
18
+ code, text = run_edit_hook(data)
19
+ if text:
20
+ sys.stderr.write(text)
21
+ return code
22
+ from .cli import main as cli_main
23
+ try:
24
+ return cli_main(argv)
25
+ except OSError as exc:
26
+ if not _stdout_closed(exc):
27
+ raise
28
+ import os # `vouch ls | head`: the reader went away
29
+ os.dup2(os.open(os.devnull, os.O_WRONLY), sys.stdout.fileno())
30
+ return 1
31
+
32
+
33
+ def _stdout_closed(exc: OSError) -> bool:
34
+ """A write to a pipe whose reader exited: BrokenPipeError on POSIX, EINVAL on Windows."""
35
+ if isinstance(exc, BrokenPipeError):
36
+ return True
37
+ import errno
38
+ if exc.errno != errno.EINVAL:
39
+ return False
40
+ try:
41
+ sys.stdout.flush()
42
+ except OSError:
43
+ return True
44
+ return False
45
+
46
+
47
+ if __name__ == "__main__": # pragma: no cover
48
+ sys.exit(main())
vouch/_version.py ADDED
@@ -0,0 +1,24 @@
1
+ # file generated by vcs-versioning
2
+ # don't change, don't track in version control
3
+ from __future__ import annotations
4
+
5
+ __all__ = [
6
+ "__version__",
7
+ "__version_tuple__",
8
+ "version",
9
+ "version_tuple",
10
+ "__commit_id__",
11
+ "commit_id",
12
+ ]
13
+
14
+ version: str
15
+ __version__: str
16
+ __version_tuple__: tuple[int | str, ...]
17
+ version_tuple: tuple[int | str, ...]
18
+ commit_id: str | None
19
+ __commit_id__: str | None
20
+
21
+ __version__ = version = '0.1'
22
+ __version_tuple__ = version_tuple = (0, 1)
23
+
24
+ __commit_id__ = commit_id = None
vouch/agents.py ADDED
@@ -0,0 +1,214 @@
1
+ """``vouch init --agents``: teach Claude Code (or any agent) to use vouch (SPEC §13.5).
2
+
3
+ Three optional parts, each shown as a diff before anything is written:
4
+
5
+ * **skill** -- ``.claude/skills/vouch/SKILL.md``: workflows for recording results,
6
+ writing a results paragraph, handling changed values and converting a paper. It
7
+ loads only when relevant, so it costs no context otherwise.
8
+ * **rules** -- a short block in ``CLAUDE.md`` (or ``AGENTS.md``) between
9
+ ``<!-- vouch -->`` markers, updated in place.
10
+ * **hook** -- a PostToolUse hook in ``.claude/settings.json`` running ``vouch hook
11
+ claude`` after every edit, so a typed number or a mistyped key is caught in the
12
+ same turn. ``--stop-gate`` adds a Stop hook running ``vouch check --strict``.
13
+ * **mcp** (only when asked for) -- ``.mcp.json`` registering ``vouch mcp``, the same
14
+ lookups as tools for any MCP client.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import difflib
20
+ import json
21
+ import shutil
22
+ import sys
23
+ from pathlib import Path
24
+
25
+ PARTS = ("skill", "rules", "hook") # the default set
26
+ OPTIONAL = ("mcp",) # .mcp.json: the MCP server, for any MCP client
27
+ MARK_START, MARK_END = "<!-- vouch -->", "<!-- /vouch -->"
28
+
29
+ SKILL = r"""---
30
+ name: vouch
31
+ description: Use when writing or editing LaTeX in a project that has vouch.toml (a paper whose numbers come from experiments), when writing experiment code whose results go into a paper, or when the user mentions results, numbers, tables, claims, figures or vouch. Every number is cited from a recorded run, never typed.
32
+ ---
33
+
34
+ # vouch: every number in the paper comes from a run
35
+
36
+ The paper never contains a typed number. Experiments record values; the paper
37
+ cites them as `\vouch{key}`; `vouch check` proves each one exists and is fresh.
38
+ Numbers computed from other numbers live in `vouch_values.py`. A number that
39
+ doesn't exist yet is a placeholder (`vouch.expect`), never a guess.
40
+
41
+ Start by reading `.vouch/CATALOG.md`: every citable key, one per line.
42
+
43
+ ## Record results in an experiment
44
+
45
+ Decorate the function that computes the result; every call is recorded, keyed by
46
+ the function and its arguments:
47
+
48
+ ```python
49
+ import vouch
50
+
51
+ @vouch.track(over="seed") # seeds -> mean ± std, per call results kept
52
+ def evaluate(dataset: str, model: str, seed: int = 0) -> dict:
53
+ ...
54
+ return {"acc": acc, "loss": loss} # -> evaluate.<dataset>.<model>.acc / .loss / .time
55
+ ```
56
+
57
+ - Descriptions and formats for families of keys go in `vouch.toml` `[metrics]`
58
+ (`"*.acc" = { fmt = ".1pct", better = "higher", desc = "top-1 test accuracy" }`),
59
+ not repeated in code. Every new metric needs a desc and `better=`.
60
+ - A single number: `vouch.record(key, value, desc=...)`; a dict: `vouch.record_all(d, prefix=...)`.
61
+ - Multi-seed results: `over="seed"` or `vouch.Stat.of(per_seed)` -- never a hand-computed mean.
62
+ - Data read: `vouch.input(path)`; files written: `vouch.artifact(path)`; figures saved
63
+ with matplotlib inside the run are tracked automatically.
64
+ - Run the script normally. `vouch status` lists stale runs with their re-run command.
65
+
66
+ ## Write a results paragraph
67
+
68
+ 1. Find each number: `vouch search "vit accuracy cifar"` or the catalog.
69
+ 2. Paste exactly what `vouch cite KEY` prints (`\vouch{...}`, or `\vouch[.2pct]{...}`
70
+ for another precision; `.mean`, `.std`, `.n` for parts of a mean ± std).
71
+ 3. Differences, ratios, "2x": never compute them. `vouch compare A B --write` adds a
72
+ `@vouch.derive` and a `@vouch.claim` to `vouch_values.py`; then `vouch build` and
73
+ cite the derived key.
74
+ 4. "Outperforms", "in every seed", "significantly": back them with a claim and wrap
75
+ the prose: `\vouchclaim{key}{ResNet outperforms ViT}`. Write "significantly" only
76
+ if `vouch compare` reports p < 0.05.
77
+ 5. A number no run has produced yet: cite `\vouch{new.key}` anyway, add
78
+ `vouch.expect("new.key", desc=..., producer="python ...")` to `vouch_values.py`,
79
+ and tell the user the experiment is owed (`vouch todo`). Never invent it.
80
+ 6. `vouch build`, then `vouch check --strict` must pass.
81
+
82
+ ## Handle changed values
83
+
84
+ `vouch changes` lists cited values that moved since someone last read their prose,
85
+ with each citing sentence. Re-read every sentence; fix any the new value makes
86
+ wrong ("the best", "roughly doubles"); report SUSPICIOUS changes to the user (they
87
+ may be bugs). Only the user acknowledges: never run `vouch ack` or `vouch accept`
88
+ yourself.
89
+
90
+ ## Convert an existing paper
91
+
92
+ `vouch suggest` classifies every typed number: replaceable (exactly one recorded
93
+ value prints like it), ambiguous, or NO SOURCE (nothing recorded prints like it).
94
+ `vouch suggest --apply` rewrites only the replaceable ones. Show the user every NO
95
+ SOURCE number: each needs a run that records it, or removal.
96
+
97
+ ## Cheat sheet
98
+
99
+ | | |
100
+ |---|---|
101
+ | `vouch search WORDS` / `vouch cite KEY` | find a key / the exact snippet |
102
+ | `vouch compare A B [--write]` | arithmetic, significance, derive + claim code |
103
+ | `vouch todo` | values the paper cites that no run recorded yet |
104
+ | `vouch build` | evaluate `vouch_values.py`, regenerate the LaTeX and the catalog |
105
+ | `vouch check --strict [--json]` | the gate; `--json` lists issues in the order to fix them |
106
+ | `vouch changes` | cited values that moved (the user acks, not you) |
107
+ | `vouch trace KEY` | where a value came from |
108
+ | `vouch explore` | browse everything in a web page |
109
+
110
+ Never edit `.vouch/` or the generated `vouch-values.tex` / `vouch-tables/` by hand.
111
+ """
112
+
113
+ RULES = """<!-- vouch -->
114
+ ## Numbers in the paper (vouch)
115
+ - Never type an empirical number into LaTeX. Find it (`vouch search`, `.vouch/CATALOG.md`), then paste what `vouch cite KEY` prints.
116
+ - If it doesn't exist: record it in the experiment, `@vouch.derive` it, or `vouch.expect()` it and tell the user it is owed.
117
+ - Never compute with numbers in prose (differences, ratios, "2x"): `vouch compare A B --write`, then cite the derived key.
118
+ - Qualitative comparisons ("outperforms", "all seeds") go in `\\vouchclaim` backed by a claim.
119
+ - Before finishing: `vouch build` and `vouch check --strict` must pass.
120
+ - If `vouch changes` lists anything: re-read each cited sentence, fix wrong text, and report SUSPICIOUS changes to the user.
121
+ - Never run `vouch ack` or `vouch accept` without the user's approval. Never edit `.vouch/` or generated files.
122
+ <!-- /vouch -->
123
+ """
124
+
125
+
126
+ def _hook_command(sub: str) -> tuple[str, bool]:
127
+ """(command, portable): ``vouch hook ...`` if vouch is on PATH, else this Python."""
128
+ if shutil.which("vouch"):
129
+ return f"vouch hook {sub}", True
130
+ return f'"{sys.executable}" -m vouch hook {sub}', False
131
+
132
+
133
+ def _with_hook(settings: dict, event: str, matcher: str | None, command: str) -> dict:
134
+ hooks = settings.setdefault("hooks", {})
135
+ groups = hooks.setdefault(event, [])
136
+ for g in groups:
137
+ if any(h.get("command") == command for h in g.get("hooks", [])):
138
+ return settings
139
+ group: dict = {"hooks": [{"type": "command", "command": command}]}
140
+ if matcher:
141
+ group = {"matcher": matcher, **group}
142
+ groups.append(group)
143
+ return settings
144
+
145
+
146
+ def planned(root: Path, parts: list[str], stop_gate: bool = False) -> dict[Path, tuple[str, str]]:
147
+ """{path: (old text, new text)} for the files ``init --agents`` would write."""
148
+ out: dict[Path, tuple[str, str]] = {}
149
+
150
+ def old(p: Path) -> str:
151
+ try:
152
+ return p.read_text(encoding="utf-8")
153
+ except OSError:
154
+ return ""
155
+ if "skill" in parts:
156
+ p = root / ".claude" / "skills" / "vouch" / "SKILL.md"
157
+ out[p] = (old(p), SKILL)
158
+ if "rules" in parts:
159
+ p = root / "CLAUDE.md"
160
+ if not p.exists() and (root / "AGENTS.md").exists():
161
+ p = root / "AGENTS.md"
162
+ text = old(p)
163
+ if MARK_START in text and MARK_END in text:
164
+ a, rest = text.split(MARK_START, 1)
165
+ _, b = rest.split(MARK_END, 1)
166
+ new = a + RULES.rstrip("\n") + b
167
+ else:
168
+ new = (text.rstrip("\n") + "\n\n" if text.strip() else "") + RULES
169
+ out[p] = (text, new)
170
+ if "hook" in parts:
171
+ cmd, portable = _hook_command("claude")
172
+ p = root / ".claude" / ("settings.json" if portable else "settings.local.json")
173
+ text = old(p)
174
+ try:
175
+ settings = json.loads(text) if text.strip() else {}
176
+ except ValueError:
177
+ settings = None
178
+ if settings is not None:
179
+ settings = _with_hook(settings, "PostToolUse", "Edit|Write|MultiEdit", cmd)
180
+ if stop_gate:
181
+ settings = _with_hook(settings, "Stop", None, _hook_command("stop")[0])
182
+ out[p] = (text, json.dumps(settings, indent=2) + "\n")
183
+ if "mcp" in parts:
184
+ p = root / ".mcp.json"
185
+ text = old(p)
186
+ try:
187
+ cfg = json.loads(text) if text.strip() else {}
188
+ except ValueError:
189
+ cfg = None
190
+ if cfg is not None:
191
+ cmd, portable = _hook_command("")
192
+ entry = ({"command": "vouch", "args": ["mcp"]} if portable else
193
+ {"command": sys.executable, "args": ["-m", "vouch", "mcp"]})
194
+ servers = cfg.setdefault("mcpServers", {})
195
+ if servers.get("vouch") != entry:
196
+ servers["vouch"] = entry
197
+ out[p] = (text, json.dumps(cfg, indent=2) + "\n")
198
+ return {p: (a, b) for p, (a, b) in out.items() if a != b}
199
+
200
+
201
+ def diff(root: Path, changes: dict[Path, tuple[str, str]]) -> str:
202
+ chunks = []
203
+ for p, (a, b) in changes.items():
204
+ rel = p.relative_to(root).as_posix()
205
+ chunks += difflib.unified_diff(a.splitlines(keepends=True), b.splitlines(keepends=True),
206
+ fromfile=f"a/{rel}" if a else "/dev/null",
207
+ tofile=f"b/{rel}")
208
+ return "".join(chunks)
209
+
210
+
211
+ def write(changes: dict[Path, tuple[str, str]]) -> None:
212
+ for p, (_, text) in changes.items():
213
+ p.parent.mkdir(parents=True, exist_ok=True)
214
+ p.write_text(text, encoding="utf-8", newline="\n")