cairnmap 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cairn/__init__.py +3 -0
- cairn/__main__.py +3 -0
- cairn/authored_store.py +135 -0
- cairn/bench/__init__.py +0 -0
- cairn/bench/conditions.py +60 -0
- cairn/bench/grading.py +119 -0
- cairn/bench/report.py +76 -0
- cairn/bench/run.py +77 -0
- cairn/bench/runner.py +176 -0
- cairn/bench/suite.py +34 -0
- cairn/bench/workspace.py +29 -0
- cairn/cli.py +445 -0
- cairn/config.py +16 -0
- cairn/detectors/__init__.py +21 -0
- cairn/detectors/base.py +120 -0
- cairn/detectors/database.py +304 -0
- cairn/detectors/docs.py +74 -0
- cairn/detectors/identity.py +147 -0
- cairn/detectors/manifests.py +124 -0
- cairn/detectors/packages.py +102 -0
- cairn/detectors/pathrefs.py +54 -0
- cairn/detectors/profile.py +215 -0
- cairn/discover/__init__.py +0 -0
- cairn/discover/files.py +162 -0
- cairn/discover/git.py +173 -0
- cairn/discover/proc.py +69 -0
- cairn/discover/repos.py +143 -0
- cairn/emit.py +43 -0
- cairn/errors.py +18 -0
- cairn/integrations/__init__.py +0 -0
- cairn/integrations/claude.py +84 -0
- cairn/integrations/config_files.py +187 -0
- cairn/integrations/content.py +54 -0
- cairn/integrations/git_hooks.py +153 -0
- cairn/integrations/harnesses.py +230 -0
- cairn/integrations/homes.py +31 -0
- cairn/integrations/registry.py +65 -0
- cairn/integrations/server_command.py +48 -0
- cairn/load.py +54 -0
- cairn/match/__init__.py +0 -0
- cairn/match/matcher.py +323 -0
- cairn/match/overrides.py +106 -0
- cairn/match/scoring.py +47 -0
- cairn/mcp_server/__init__.py +0 -0
- cairn/mcp_server/server.py +113 -0
- cairn/mcp_server/tools.py +190 -0
- cairn/model/__init__.py +0 -0
- cairn/model/graph.py +163 -0
- cairn/model/overrides.py +38 -0
- cairn/paths.py +45 -0
- cairn/py.typed +0 -0
- cairn/render/__init__.py +0 -0
- cairn/render/card.py +176 -0
- cairn/render/index.py +113 -0
- cairn/render/markers.py +58 -0
- cairn/render/tokens.py +7 -0
- cairn/resolve.py +102 -0
- cairn/scan.py +253 -0
- cairn/scan_cache.py +99 -0
- cairn/scan_log.py +24 -0
- cairn/security/__init__.py +0 -0
- cairn/security/policy.py +52 -0
- cairn/security/redact.py +78 -0
- cairn/security/text.py +29 -0
- cairn/store/__init__.py +0 -0
- cairn/store/atomic.py +68 -0
- cairn/store/lock.py +115 -0
- cairn/store/workspace_store.py +29 -0
- cairnmap-0.1.0.dist-info/METADATA +352 -0
- cairnmap-0.1.0.dist-info/RECORD +74 -0
- cairnmap-0.1.0.dist-info/WHEEL +4 -0
- cairnmap-0.1.0.dist-info/entry_points.txt +2 -0
- cairnmap-0.1.0.dist-info/licenses/LICENSE +202 -0
- cairnmap-0.1.0.dist-info/licenses/NOTICE +6 -0
cairn/__init__.py
ADDED
cairn/__main__.py
ADDED
cairn/authored_store.py
ADDED
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
"""Write harness/human-authored content (.cairn/authored/<repo>.yaml) without clobbering it."""
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Iterable, Mapping
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Literal
|
|
7
|
+
|
|
8
|
+
import yaml
|
|
9
|
+
|
|
10
|
+
from cairn.errors import CairnError, CairnInputError
|
|
11
|
+
from cairn.load import load_authored
|
|
12
|
+
from cairn.model.graph import SYMMETRIC_TYPES, Edge, EdgeType
|
|
13
|
+
from cairn.model.overrides import Authored
|
|
14
|
+
from cairn.paths import authored_dir
|
|
15
|
+
from cairn.resolve import resolve_repo
|
|
16
|
+
from cairn.store.atomic import atomic_write_text
|
|
17
|
+
from cairn.store.workspace_store import load_workspace
|
|
18
|
+
|
|
19
|
+
_KEY = re.compile(r"^(?P<source>.+?)->(?P<target>.+):(?P<type>[a-z_]+)$")
|
|
20
|
+
Review = Literal["confirmed", "rejected"]
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
def parse_edge_key(key: str) -> tuple[str, str, EdgeType]:
|
|
24
|
+
match = _KEY.match(key.strip())
|
|
25
|
+
if not match:
|
|
26
|
+
raise CairnInputError(key, "expected an edge key like 'source->target:shares_db'")
|
|
27
|
+
try:
|
|
28
|
+
edge_type = EdgeType(match["type"])
|
|
29
|
+
except ValueError as exc:
|
|
30
|
+
raise CairnInputError(key, f"unknown edge type '{match['type']}'") from exc
|
|
31
|
+
return match["source"], match["target"], edge_type
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def save_authored(ws_root: Path, repo_id: str, authored: Authored) -> Path:
|
|
35
|
+
path = authored_dir(ws_root) / f"{repo_id}.yaml"
|
|
36
|
+
data = authored.model_dump(mode="json", exclude_defaults=True)
|
|
37
|
+
atomic_write_text(path, yaml.safe_dump(data, sort_keys=False, allow_unicode=True))
|
|
38
|
+
return path
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def annotate_edge(ws_root: Path, key: str, *, review: Review | None, why: str | None) -> Path:
|
|
42
|
+
"""Record a review and/or explanation for an edge, keeping every other authored field.
|
|
43
|
+
|
|
44
|
+
The decision is stored under the edge's canonical key (its current direction), and any
|
|
45
|
+
stale copy under the reversed key is removed, so the newest decision always wins.
|
|
46
|
+
An edge the user rejected earlier is no longer in the map but can still be re-decided.
|
|
47
|
+
"""
|
|
48
|
+
_, _, edge_type = parse_edge_key(key)
|
|
49
|
+
workspace = load_workspace(ws_root)
|
|
50
|
+
if workspace is None:
|
|
51
|
+
raise CairnError("No map found. Run `cairn scan` first.")
|
|
52
|
+
authored = load_authored(ws_root)
|
|
53
|
+
variants = _key_variants(key, edge_type)
|
|
54
|
+
canonical = _canonical_key(variants, edge_type, workspace.edges, authored)
|
|
55
|
+
if canonical is None:
|
|
56
|
+
raise CairnError(f"No edge '{key}' in the current map. `cairn status` lists edge keys.")
|
|
57
|
+
carried = _drop_variants(ws_root, authored, set(variants) - {canonical})
|
|
58
|
+
source = parse_edge_key(canonical)[0]
|
|
59
|
+
current = load_authored(ws_root).get(source, Authored())
|
|
60
|
+
reviews = {**current.edge_reviews}
|
|
61
|
+
whys = {**current.edge_whys}
|
|
62
|
+
if carried.edge_whys and canonical not in whys:
|
|
63
|
+
whys[canonical] = next(iter(carried.edge_whys.values())) # keep the old explanation
|
|
64
|
+
if review is not None:
|
|
65
|
+
reviews[canonical] = review
|
|
66
|
+
if why:
|
|
67
|
+
whys[canonical] = why
|
|
68
|
+
update = {"edge_reviews": reviews, "edge_whys": whys}
|
|
69
|
+
return save_authored(ws_root, source, current.model_copy(update=update))
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def _key_variants(key: str, edge_type: EdgeType) -> tuple[str, ...]:
|
|
73
|
+
source, target, _ = parse_edge_key(key)
|
|
74
|
+
flipped = f"{target}->{source}:{edge_type.value}"
|
|
75
|
+
return (key, flipped) if edge_type in SYMMETRIC_TYPES else (key,)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def _canonical_key(
|
|
79
|
+
variants: tuple[str, ...],
|
|
80
|
+
edge_type: EdgeType,
|
|
81
|
+
edges: Iterable[Edge],
|
|
82
|
+
authored: Mapping[str, Authored],
|
|
83
|
+
) -> str | None:
|
|
84
|
+
live = next((e.key for e in edges if e.type is edge_type and e.key in variants), None)
|
|
85
|
+
if live is not None:
|
|
86
|
+
return live
|
|
87
|
+
decided = [k for a in authored.values() for k in (*a.edge_reviews, *a.edge_whys)]
|
|
88
|
+
return next((k for k in decided if k in variants), None)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _drop_variants(ws_root: Path, authored: Mapping[str, Authored], stale: set[str]) -> Authored:
|
|
92
|
+
"""Remove stale keys from every authored file; return their whys so they can be carried over."""
|
|
93
|
+
carried: dict[str, str] = {}
|
|
94
|
+
for repo_id, entry in authored.items():
|
|
95
|
+
if not stale & {*entry.edge_reviews, *entry.edge_whys}:
|
|
96
|
+
continue
|
|
97
|
+
carried.update({k: v for k, v in entry.edge_whys.items() if k in stale})
|
|
98
|
+
cleaned = entry.model_copy(
|
|
99
|
+
update={
|
|
100
|
+
"edge_reviews": {k: v for k, v in entry.edge_reviews.items() if k not in stale},
|
|
101
|
+
"edge_whys": {k: v for k, v in entry.edge_whys.items() if k not in stale},
|
|
102
|
+
}
|
|
103
|
+
)
|
|
104
|
+
save_authored(ws_root, repo_id, cleaned)
|
|
105
|
+
return Authored(edge_whys=carried)
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
SUMMARY_LIMIT = 500
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def set_summary(ws_root: Path, repo_id: str, summary: str, *, aliases: Iterable[str] = ()) -> Path:
|
|
112
|
+
"""Store a harness/human-written summary (spec §9.2), stamped with the repo's current HEAD."""
|
|
113
|
+
text = " ".join(summary.split())
|
|
114
|
+
if not text:
|
|
115
|
+
raise CairnInputError("summary", "must not be empty")
|
|
116
|
+
if len(text) > SUMMARY_LIMIT:
|
|
117
|
+
raise CairnInputError(
|
|
118
|
+
"summary", f"is {len(text)} characters; keep it under {SUMMARY_LIMIT}"
|
|
119
|
+
)
|
|
120
|
+
workspace = load_workspace(ws_root)
|
|
121
|
+
if workspace is None:
|
|
122
|
+
raise CairnError("No map found. Run `cairn scan` first.")
|
|
123
|
+
repo = workspace.repo(repo_id)
|
|
124
|
+
if repo is None:
|
|
125
|
+
matches = resolve_repo(workspace, load_authored(ws_root), repo_id, limit=3)
|
|
126
|
+
hint = ", ".join(m.repo_id for m in matches) or "none"
|
|
127
|
+
raise CairnError(f"No repo '{repo_id}'. Did you mean: {hint}?")
|
|
128
|
+
current = load_authored(ws_root).get(repo_id, Authored())
|
|
129
|
+
extra = (a.strip().lower() for a in aliases if a.strip())
|
|
130
|
+
update = {
|
|
131
|
+
"summary": text,
|
|
132
|
+
"summary_sha": repo.head_sha,
|
|
133
|
+
"aliases": tuple(dict.fromkeys((*current.aliases, *extra))),
|
|
134
|
+
}
|
|
135
|
+
return save_authored(ws_root, repo_id, current.model_copy(update=update))
|
cairn/bench/__init__.py
ADDED
|
File without changes
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
"""Set up one isolated workspace copy per benchmark condition (spec §11 E2).
|
|
2
|
+
|
|
3
|
+
A: cold (no map, no CLAUDE.md) B: hand-written RELATED_REPOS-style doc as CLAUDE.md
|
|
4
|
+
C: cairn INDEX only (no cards/graph) D: INDEX + cards E: D + the cairn MCP server
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
import shutil
|
|
9
|
+
from dataclasses import dataclass
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
|
|
12
|
+
from cairn.bench.suite import Suite
|
|
13
|
+
from cairn.bench.workspace import materialize
|
|
14
|
+
from cairn.emit import write_outputs
|
|
15
|
+
from cairn.integrations.claude import install_claude
|
|
16
|
+
from cairn.integrations.server_command import server_command
|
|
17
|
+
from cairn.paths import cairn_dir, index_file
|
|
18
|
+
from cairn.render.index import render_index
|
|
19
|
+
from cairn.scan import scan_workspace
|
|
20
|
+
from cairn.store.atomic import atomic_write_text
|
|
21
|
+
|
|
22
|
+
CONDITIONS = ("A", "B", "C", "D", "E")
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass(frozen=True)
|
|
26
|
+
class Prepared:
|
|
27
|
+
ws: Path
|
|
28
|
+
mcp_config: Path | None
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def prepare(condition: str, suite: Suite, suite_dir: Path, run_dir: Path) -> Prepared:
|
|
32
|
+
ws = materialize(suite_dir / suite.workspace, run_dir / "ws")
|
|
33
|
+
if condition in ("A", "B"):
|
|
34
|
+
shutil.rmtree(cairn_dir(ws), ignore_errors=True) # no authored summaries either
|
|
35
|
+
if condition == "B":
|
|
36
|
+
doc = (suite_dir / suite.related_repos_doc).read_text(encoding="utf-8")
|
|
37
|
+
atomic_write_text(ws / "CLAUDE.md", doc)
|
|
38
|
+
return Prepared(ws, None)
|
|
39
|
+
result = scan_workspace(ws)
|
|
40
|
+
write_outputs(ws, result)
|
|
41
|
+
if condition == "C":
|
|
42
|
+
# INDEX only: no card pointer in the text, and no cards or graph JSON to read instead.
|
|
43
|
+
text = render_index(
|
|
44
|
+
result.workspace,
|
|
45
|
+
result.authored,
|
|
46
|
+
threshold=result.config.index_threshold,
|
|
47
|
+
with_cards=False,
|
|
48
|
+
)
|
|
49
|
+
atomic_write_text(index_file(ws), text)
|
|
50
|
+
install_claude(ws)
|
|
51
|
+
shutil.rmtree(cairn_dir(ws))
|
|
52
|
+
return Prepared(ws, None)
|
|
53
|
+
install_claude(ws)
|
|
54
|
+
if condition != "E":
|
|
55
|
+
return Prepared(ws, None)
|
|
56
|
+
config = run_dir / "mcp.json"
|
|
57
|
+
command = [*server_command(), "--workspace", ws.as_posix()]
|
|
58
|
+
entry = {"command": command[0], "args": command[1:]}
|
|
59
|
+
atomic_write_text(config, json.dumps({"mcpServers": {"cairn": entry}}, indent=2))
|
|
60
|
+
return Prepared(ws, config)
|
cairn/bench/grading.py
ADDED
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
"""Deterministic grading of benchmark answers (spec §11): file recall or required keywords."""
|
|
2
|
+
|
|
3
|
+
import posixpath
|
|
4
|
+
import re
|
|
5
|
+
from dataclasses import dataclass
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from cairn.bench.suite import Task
|
|
9
|
+
|
|
10
|
+
_SEGMENT = r"[\w@.\-\[\]]+"
|
|
11
|
+
# Windows absolute, POSIX absolute, then relative `a/b/c` paths (either slash).
|
|
12
|
+
_PATH = re.compile(rf"[A-Za-z]:[\\/][^\s`'\"()]+|/[^\s`'\"()]+|{_SEGMENT}(?:[\\/]{_SEGMENT})+")
|
|
13
|
+
_TRAILING = ".,;:)"
|
|
14
|
+
_DRIVE = re.compile(r"([A-Za-z]):/(.*)")
|
|
15
|
+
_MSYS = re.compile(r"/([A-Za-z])/(.*)")
|
|
16
|
+
PASS_RECALL = 0.8
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@dataclass(frozen=True)
|
|
20
|
+
class Grade:
|
|
21
|
+
success: bool
|
|
22
|
+
recall: float
|
|
23
|
+
precision: float
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _root_spellings(ws_root: Path) -> tuple[str, ...]:
|
|
27
|
+
"""The workspace root as an agent might print it: as given, resolved, Git Bash style."""
|
|
28
|
+
spellings: set[str] = set()
|
|
29
|
+
for root in {ws_root.as_posix(), ws_root.resolve().as_posix()}:
|
|
30
|
+
root = root.rstrip("/") + "/"
|
|
31
|
+
spellings.add(root)
|
|
32
|
+
drive = _DRIVE.fullmatch(root)
|
|
33
|
+
if drive:
|
|
34
|
+
spellings.add(f"/{drive.group(1).lower()}/{drive.group(2)}")
|
|
35
|
+
return tuple(sorted(spellings, key=len, reverse=True))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _inside(path: str, roots: tuple[str, ...]) -> str | None:
|
|
39
|
+
"""`path` relative to the workspace, or None when it points elsewhere."""
|
|
40
|
+
for root in roots:
|
|
41
|
+
if path.casefold().startswith(root.casefold()):
|
|
42
|
+
return path[len(root) :]
|
|
43
|
+
windows = any(_DRIVE.match(root) for root in roots)
|
|
44
|
+
try: # 8.3 short names, symlinked temp dirs (macOS /var -> /private/var)
|
|
45
|
+
msys = _MSYS.fullmatch(path) if windows else None
|
|
46
|
+
real = Path(f"{msys.group(1)}:/{msys.group(2)}" if msys else path)
|
|
47
|
+
resolved = real.resolve().as_posix()
|
|
48
|
+
except (OSError, ValueError):
|
|
49
|
+
return None
|
|
50
|
+
for root in roots:
|
|
51
|
+
if resolved.casefold().startswith(root.casefold()):
|
|
52
|
+
return resolved[len(root) :]
|
|
53
|
+
return None
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def _heading_repo(line: str, repo_ids: set[str]) -> str | None | bool:
|
|
57
|
+
"""For a heading-like line with no paths: the one repo it names, or None.
|
|
58
|
+
|
|
59
|
+
Returns False when the line isn't a heading (the current context carries on).
|
|
60
|
+
"""
|
|
61
|
+
stripped = line.strip().strip("-* ").strip()
|
|
62
|
+
heading = line.lstrip().startswith("#") or stripped.endswith(":") or line.strip().endswith("**")
|
|
63
|
+
if not stripped or not heading or _PATH.search(line):
|
|
64
|
+
return False
|
|
65
|
+
named = [r for r in repo_ids if re.search(rf"(?<![\w-]){re.escape(r)}(?![\w-])", line)]
|
|
66
|
+
return named[0] if len(named) == 1 else None
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _strip_root_tail(path: str, ws_parts: tuple[str, ...], repo_ids: set[str]) -> str:
|
|
70
|
+
"""`ws/orders-svc/x` (relative to the workspace's parent) -> `orders-svc/x`."""
|
|
71
|
+
segments = path.split("/")
|
|
72
|
+
for i in range(1, min(len(segments), len(ws_parts) + 1)):
|
|
73
|
+
if segments[i] in repo_ids and tuple(segments[:i]) == ws_parts[-i:]:
|
|
74
|
+
return "/".join(segments[i:])
|
|
75
|
+
return path
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def mentioned_files(text: str, ws_root: Path, working_repo: str, repo_ids: set[str]) -> set[str]:
|
|
79
|
+
"""Workspace paths (`repo/path`) named in `text`.
|
|
80
|
+
|
|
81
|
+
A bare path belongs to the repo named by the heading it sits under (`**orders-svc:**`),
|
|
82
|
+
else to the working repo.
|
|
83
|
+
"""
|
|
84
|
+
roots = _root_spellings(ws_root)
|
|
85
|
+
ws_parts = tuple(p for p in ws_root.as_posix().split("/") if p)
|
|
86
|
+
found: set[str] = set()
|
|
87
|
+
context = working_repo
|
|
88
|
+
for line in text.splitlines():
|
|
89
|
+
heading = _heading_repo(line, repo_ids)
|
|
90
|
+
if heading is not False:
|
|
91
|
+
context = heading or working_repo
|
|
92
|
+
continue
|
|
93
|
+
for token in _PATH.findall(line):
|
|
94
|
+
path = token.replace("\\", "/").rstrip(_TRAILING)
|
|
95
|
+
if path.startswith("/") or _DRIVE.match(path):
|
|
96
|
+
inside = _inside(path, roots)
|
|
97
|
+
if inside is None:
|
|
98
|
+
continue
|
|
99
|
+
path = inside
|
|
100
|
+
else:
|
|
101
|
+
path = _strip_root_tail(path, ws_parts, repo_ids)
|
|
102
|
+
if path.split("/", 1)[0] not in repo_ids:
|
|
103
|
+
path = f"{context}/{path}"
|
|
104
|
+
path = posixpath.normpath(path)
|
|
105
|
+
if not path.startswith(".."):
|
|
106
|
+
found.add(path)
|
|
107
|
+
return found
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def grade(task: Task, text: str, ws_root: Path, repo_ids: set[str]) -> Grade:
|
|
111
|
+
keywords_ok = all(k.lower() in text.lower() for k in task.expect_keywords)
|
|
112
|
+
if not task.expect_files:
|
|
113
|
+
return Grade(success=keywords_ok, recall=1.0 if keywords_ok else 0.0, precision=1.0)
|
|
114
|
+
found = mentioned_files(text, ws_root, task.repo, repo_ids)
|
|
115
|
+
expected = set(task.expect_files)
|
|
116
|
+
hits = len(found & expected)
|
|
117
|
+
recall = hits / len(expected)
|
|
118
|
+
precision = hits / len(found) if found else 0.0
|
|
119
|
+
return Grade(success=keywords_ok and recall >= PASS_RECALL, recall=recall, precision=precision)
|
cairn/bench/report.py
ADDED
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
"""Markdown summary of a benchmark run: per-condition totals, then condition × task."""
|
|
2
|
+
|
|
3
|
+
from collections.abc import Mapping, Sequence
|
|
4
|
+
from statistics import fmean
|
|
5
|
+
from typing import TYPE_CHECKING
|
|
6
|
+
|
|
7
|
+
if TYPE_CHECKING:
|
|
8
|
+
from cairn.bench.run import RunRecord
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
def _mean(values: Sequence[float]) -> float:
|
|
12
|
+
return fmean(values) if values else 0.0
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def _summary(records: Sequence["RunRecord"]) -> list[str]:
|
|
16
|
+
"""Success counts every run; token/cost/turn means use only runs that didn't error,
|
|
17
|
+
since an errored run reports zeros that would flatter its condition."""
|
|
18
|
+
lines = [
|
|
19
|
+
"| Condition | Runs | Success | Fresh tokens | Cache-read tokens | Cost (USD) | Turns "
|
|
20
|
+
"| Errors |",
|
|
21
|
+
"|---|---|---|---|---|---|---|---|",
|
|
22
|
+
]
|
|
23
|
+
for condition in dict.fromkeys(r.condition for r in records):
|
|
24
|
+
rows = [r for r in records if r.condition == condition]
|
|
25
|
+
ok = [r for r in rows if not r.result.is_error]
|
|
26
|
+
success = sum(r.grade.success for r in rows) / len(rows)
|
|
27
|
+
lines.append(
|
|
28
|
+
f"| {condition} | {len(rows)} | {success:.0%} "
|
|
29
|
+
f"| {_mean([r.result.fresh_tokens for r in ok]):,.0f} "
|
|
30
|
+
f"| {_mean([r.result.cache_read_tokens for r in ok]):,.0f} "
|
|
31
|
+
f"| {_mean([r.result.cost_usd for r in ok]):.4f} "
|
|
32
|
+
f"| {_mean([r.result.num_turns for r in ok]):.1f} "
|
|
33
|
+
f"| {len(rows) - len(ok)} |"
|
|
34
|
+
)
|
|
35
|
+
return lines
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def _per_task(records: Sequence["RunRecord"]) -> list[str]:
|
|
39
|
+
lines = [
|
|
40
|
+
"| Condition | Task | Success | Recall | Cost (USD) |",
|
|
41
|
+
"|---|---|---|---|---|",
|
|
42
|
+
]
|
|
43
|
+
for condition in dict.fromkeys(r.condition for r in records):
|
|
44
|
+
for task_id in dict.fromkeys(r.task_id for r in records):
|
|
45
|
+
rows = [r for r in records if r.condition == condition and r.task_id == task_id]
|
|
46
|
+
if not rows:
|
|
47
|
+
continue
|
|
48
|
+
passed = sum(r.grade.success for r in rows)
|
|
49
|
+
lines.append(
|
|
50
|
+
f"| {condition} | {task_id} | {passed}/{len(rows)} "
|
|
51
|
+
f"| {_mean([r.grade.recall for r in rows]):.2f} "
|
|
52
|
+
f"| {_mean([r.result.cost_usd for r in rows if not r.result.is_error]):.4f} |"
|
|
53
|
+
)
|
|
54
|
+
return lines
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def render_markdown(
|
|
58
|
+
records: Sequence["RunRecord"], *, meta: Mapping[str, object] | None = None
|
|
59
|
+
) -> str:
|
|
60
|
+
errors = sum(r.result.is_error for r in records)
|
|
61
|
+
about = [f"{key}: {value}" for key, value in (meta or {}).items() if key != "tasks"]
|
|
62
|
+
parts = [
|
|
63
|
+
"# cairn benchmark",
|
|
64
|
+
"",
|
|
65
|
+
*([", ".join(about), ""] if about else []),
|
|
66
|
+
"Conditions: A cold, B hand-written doc, C cairn INDEX, D INDEX + cards, E D + MCP",
|
|
67
|
+
"",
|
|
68
|
+
*_summary(records),
|
|
69
|
+
"",
|
|
70
|
+
"## Per task",
|
|
71
|
+
"",
|
|
72
|
+
*_per_task(records),
|
|
73
|
+
]
|
|
74
|
+
if errors:
|
|
75
|
+
parts += ["", f"{errors} run(s) ended in an agent error; see the JSON for details."]
|
|
76
|
+
return "\n".join(parts) + "\n"
|
cairn/bench/run.py
ADDED
|
@@ -0,0 +1,77 @@
|
|
|
1
|
+
"""Run a benchmark suite across conditions and write a report (spec §11 E2)."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import tempfile
|
|
5
|
+
from collections.abc import Mapping, Sequence
|
|
6
|
+
from dataclasses import asdict, dataclass
|
|
7
|
+
from datetime import UTC, datetime
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
import cairn
|
|
11
|
+
from cairn.bench.conditions import prepare
|
|
12
|
+
from cairn.bench.grading import Grade, grade
|
|
13
|
+
from cairn.bench.report import render_markdown
|
|
14
|
+
from cairn.bench.runner import Runner, RunResult
|
|
15
|
+
from cairn.bench.suite import load_suite
|
|
16
|
+
from cairn.store.atomic import atomic_write_text
|
|
17
|
+
|
|
18
|
+
__all__ = ["RunRecord", "render_markdown", "run_bench"]
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass(frozen=True)
|
|
22
|
+
class RunRecord:
|
|
23
|
+
condition: str
|
|
24
|
+
task_id: str
|
|
25
|
+
run: int
|
|
26
|
+
grade: Grade
|
|
27
|
+
result: RunResult
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def run_bench(
|
|
31
|
+
suite_dir: Path,
|
|
32
|
+
*,
|
|
33
|
+
conditions: Sequence[str],
|
|
34
|
+
runs: int,
|
|
35
|
+
task_ids: Sequence[str] | None,
|
|
36
|
+
runner: Runner,
|
|
37
|
+
out_dir: Path,
|
|
38
|
+
now: str,
|
|
39
|
+
meta: Mapping[str, object] | None = None,
|
|
40
|
+
) -> tuple[RunRecord, ...]:
|
|
41
|
+
"""Each finished run is appended to `<now>.jsonl` at once, so an interrupted (paid)
|
|
42
|
+
benchmark keeps what it already measured; `<now>.json` and `.md` are written at the end."""
|
|
43
|
+
suite = load_suite(suite_dir)
|
|
44
|
+
tasks = [t for t in suite.tasks if not task_ids or t.id in task_ids]
|
|
45
|
+
header = {
|
|
46
|
+
"cairn_version": cairn.__version__,
|
|
47
|
+
"suite": suite.name,
|
|
48
|
+
"conditions": list(conditions),
|
|
49
|
+
"runs": runs,
|
|
50
|
+
"tasks": [t.id for t in tasks],
|
|
51
|
+
"started_at": datetime.now(UTC).isoformat(timespec="seconds"),
|
|
52
|
+
**(meta or {}),
|
|
53
|
+
}
|
|
54
|
+
out_dir.mkdir(parents=True, exist_ok=True)
|
|
55
|
+
log = out_dir / f"{now}.jsonl"
|
|
56
|
+
records: list[RunRecord] = []
|
|
57
|
+
for condition in conditions:
|
|
58
|
+
for task in tasks:
|
|
59
|
+
for index in range(runs):
|
|
60
|
+
# The MCP server of condition E may still hold files for a moment on Windows.
|
|
61
|
+
with tempfile.TemporaryDirectory(
|
|
62
|
+
prefix="cairn-bench-", ignore_cleanup_errors=True
|
|
63
|
+
) as tmp:
|
|
64
|
+
prepared = prepare(condition, suite, suite_dir, Path(tmp))
|
|
65
|
+
ws = prepared.ws
|
|
66
|
+
repo_ids = {p.name for p in ws.iterdir() if (p / ".git").exists()}
|
|
67
|
+
result = runner.run(task.prompt, ws / task.repo, ws, prepared.mcp_config)
|
|
68
|
+
graded = grade(task, result.result_text, ws, repo_ids)
|
|
69
|
+
record = RunRecord(condition, task.id, index, graded, result)
|
|
70
|
+
records.append(record)
|
|
71
|
+
with log.open("a", encoding="utf-8") as handle:
|
|
72
|
+
handle.write(json.dumps(asdict(record)) + "\n")
|
|
73
|
+
out = tuple(records)
|
|
74
|
+
report = {"meta": header, "records": [asdict(r) for r in out]}
|
|
75
|
+
atomic_write_text(out_dir / f"{now}.json", json.dumps(report, indent=2))
|
|
76
|
+
atomic_write_text(out_dir / f"{now}.md", render_markdown(out, meta=header))
|
|
77
|
+
return out
|
cairn/bench/runner.py
ADDED
|
@@ -0,0 +1,176 @@
|
|
|
1
|
+
"""Run one benchmark prompt through a headless agent (spec §11 E2).
|
|
2
|
+
|
|
3
|
+
ClaudeRunner drives `claude -p` in an isolated, read-only session: user settings and
|
|
4
|
+
memory are not loaded (`--setting-sources project,local`), only the MCP servers in the
|
|
5
|
+
condition's config are visible (`--strict-mcp-config`), and only read tools are allowed
|
|
6
|
+
(no shell: even `find` can delete or run programs through `-delete`/`-exec`).
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
import re
|
|
12
|
+
import shutil
|
|
13
|
+
import subprocess
|
|
14
|
+
from collections.abc import Callable
|
|
15
|
+
from dataclasses import dataclass
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
from typing import Protocol
|
|
18
|
+
|
|
19
|
+
READ_ONLY_TOOLS = ("Read", "Grep", "Glob", "mcp__cairn")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class RunResult:
|
|
24
|
+
result_text: str
|
|
25
|
+
num_turns: int = 0
|
|
26
|
+
cost_usd: float = 0.0
|
|
27
|
+
input_tokens: int = 0
|
|
28
|
+
cache_creation_tokens: int = 0
|
|
29
|
+
cache_read_tokens: int = 0
|
|
30
|
+
output_tokens: int = 0
|
|
31
|
+
duration_ms: int = 0
|
|
32
|
+
is_error: bool = False
|
|
33
|
+
models: tuple[str, ...] = ()
|
|
34
|
+
|
|
35
|
+
@property
|
|
36
|
+
def fresh_tokens(self) -> int:
|
|
37
|
+
"""Tokens the model actually processed fresh (cache reads are near-free)."""
|
|
38
|
+
return self.input_tokens + self.cache_creation_tokens + self.output_tokens
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
def parse_result(raw: str) -> RunResult:
|
|
42
|
+
try:
|
|
43
|
+
data = json.loads(raw)
|
|
44
|
+
except json.JSONDecodeError:
|
|
45
|
+
return RunResult(result_text=raw[-2000:], is_error=True)
|
|
46
|
+
if not isinstance(data, dict):
|
|
47
|
+
return RunResult(result_text=raw[-2000:], is_error=True)
|
|
48
|
+
usage = data.get("usage") or {}
|
|
49
|
+
model_usage = data.get("modelUsage")
|
|
50
|
+
return RunResult(
|
|
51
|
+
result_text=str(data.get("result") or ""),
|
|
52
|
+
num_turns=int(data.get("num_turns") or 0),
|
|
53
|
+
cost_usd=float(data.get("total_cost_usd") or 0.0),
|
|
54
|
+
input_tokens=int(usage.get("input_tokens") or 0),
|
|
55
|
+
cache_creation_tokens=int(usage.get("cache_creation_input_tokens") or 0),
|
|
56
|
+
cache_read_tokens=int(usage.get("cache_read_input_tokens") or 0),
|
|
57
|
+
output_tokens=int(usage.get("output_tokens") or 0),
|
|
58
|
+
duration_ms=int(data.get("duration_ms") or 0),
|
|
59
|
+
is_error=bool(data.get("is_error")),
|
|
60
|
+
models=tuple(sorted(model_usage)) if isinstance(model_usage, dict) else (),
|
|
61
|
+
)
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def claude_home() -> Path:
|
|
65
|
+
"""Claude Code's config folder (the user's real one: benchmarks only clean up after runs)."""
|
|
66
|
+
override = os.environ.get("CLAUDE_CONFIG_DIR")
|
|
67
|
+
return Path(override) if override else Path.home() / ".claude"
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def project_slug(cwd: Path) -> str:
|
|
71
|
+
"""How Claude Code names a working directory's folder under `projects/`."""
|
|
72
|
+
return re.sub(r"[^A-Za-z0-9]", "-", str(cwd))
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def forget_project(home: Path, cwd: Path) -> None:
|
|
76
|
+
"""Remove the per-cwd folder a headless run leaves behind, if it holds no files."""
|
|
77
|
+
target = home / "projects" / project_slug(cwd)
|
|
78
|
+
if target.is_dir() and not any(p.is_file() for p in target.rglob("*")):
|
|
79
|
+
shutil.rmtree(target, ignore_errors=True)
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class Runner(Protocol):
|
|
83
|
+
def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult: ...
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
@dataclass(frozen=True)
|
|
87
|
+
class ClaudeRunner:
|
|
88
|
+
model: str = "haiku"
|
|
89
|
+
timeout: float = 900.0
|
|
90
|
+
home: Path | None = None # Claude Code's config folder; default claude_home()
|
|
91
|
+
|
|
92
|
+
def isolation_settings(self, ws: Path) -> dict[str, object]:
|
|
93
|
+
"""Keep the benchmarking user's own instructions out of every run.
|
|
94
|
+
|
|
95
|
+
`--setting-sources project,local` doesn't cover memory files: ~/.claude/CLAUDE.md
|
|
96
|
+
and ~/.claude/rules still load, as would a CLAUDE.md in any folder above the
|
|
97
|
+
temporary workspace. Only the condition's workspace CLAUDE.md may load.
|
|
98
|
+
"""
|
|
99
|
+
home = (self.home or claude_home()).as_posix().rstrip("/")
|
|
100
|
+
above = [p.as_posix().rstrip("/") for p in ws.parents]
|
|
101
|
+
excludes = [f"{home}/**"]
|
|
102
|
+
excludes += [f"{d}/{name}" for d in above for name in ("CLAUDE.md", "CLAUDE.local.md")]
|
|
103
|
+
excludes += [f"{d}/.claude/**" for d in above]
|
|
104
|
+
return {"claudeMdExcludes": excludes, "autoMemoryEnabled": False}
|
|
105
|
+
|
|
106
|
+
def command(self, prompt: str, ws: Path, mcp_config: Path | None) -> list[str]:
|
|
107
|
+
# The resolved path, so npm's `claude.cmd` shim launches on Windows too.
|
|
108
|
+
cmd = [
|
|
109
|
+
shutil.which("claude") or "claude",
|
|
110
|
+
"-p",
|
|
111
|
+
prompt,
|
|
112
|
+
"--output-format",
|
|
113
|
+
"json",
|
|
114
|
+
"--model",
|
|
115
|
+
self.model,
|
|
116
|
+
"--setting-sources",
|
|
117
|
+
"project,local",
|
|
118
|
+
"--settings",
|
|
119
|
+
json.dumps(self.isolation_settings(ws)),
|
|
120
|
+
"--no-session-persistence",
|
|
121
|
+
"--permission-mode",
|
|
122
|
+
"dontAsk",
|
|
123
|
+
"--allowedTools",
|
|
124
|
+
*READ_ONLY_TOOLS,
|
|
125
|
+
"--add-dir",
|
|
126
|
+
str(ws),
|
|
127
|
+
"--strict-mcp-config",
|
|
128
|
+
]
|
|
129
|
+
if mcp_config is not None:
|
|
130
|
+
cmd += ["--mcp-config", str(mcp_config)]
|
|
131
|
+
return cmd
|
|
132
|
+
|
|
133
|
+
def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult:
|
|
134
|
+
try:
|
|
135
|
+
proc = subprocess.run(
|
|
136
|
+
self.command(prompt, ws, mcp_config),
|
|
137
|
+
cwd=cwd,
|
|
138
|
+
capture_output=True,
|
|
139
|
+
text=True,
|
|
140
|
+
encoding="utf-8",
|
|
141
|
+
errors="replace",
|
|
142
|
+
timeout=self.timeout,
|
|
143
|
+
check=False,
|
|
144
|
+
)
|
|
145
|
+
except (OSError, subprocess.TimeoutExpired) as exc:
|
|
146
|
+
return RunResult(result_text=f"runner error: {exc}", is_error=True)
|
|
147
|
+
finally:
|
|
148
|
+
home = self.home or claude_home()
|
|
149
|
+
for path in {cwd, cwd.resolve()}:
|
|
150
|
+
forget_project(home, path)
|
|
151
|
+
return parse_result(proc.stdout or proc.stderr)
|
|
152
|
+
|
|
153
|
+
def version(self) -> str:
|
|
154
|
+
try:
|
|
155
|
+
done = subprocess.run(
|
|
156
|
+
[shutil.which("claude") or "claude", "--version"],
|
|
157
|
+
capture_output=True,
|
|
158
|
+
text=True,
|
|
159
|
+
encoding="utf-8",
|
|
160
|
+
errors="replace",
|
|
161
|
+
timeout=60,
|
|
162
|
+
check=False,
|
|
163
|
+
)
|
|
164
|
+
except (OSError, subprocess.TimeoutExpired):
|
|
165
|
+
return "unknown"
|
|
166
|
+
return done.stdout.strip() or "unknown"
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
@dataclass(frozen=True)
|
|
170
|
+
class FakeRunner:
|
|
171
|
+
"""Test double: `reply(prompt, cwd)` returns the raw JSON claude would print."""
|
|
172
|
+
|
|
173
|
+
reply: Callable[[str, Path], str]
|
|
174
|
+
|
|
175
|
+
def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult:
|
|
176
|
+
return parse_result(self.reply(prompt, cwd))
|