cairnmap 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (74) hide show
  1. cairn/__init__.py +3 -0
  2. cairn/__main__.py +3 -0
  3. cairn/authored_store.py +135 -0
  4. cairn/bench/__init__.py +0 -0
  5. cairn/bench/conditions.py +60 -0
  6. cairn/bench/grading.py +119 -0
  7. cairn/bench/report.py +76 -0
  8. cairn/bench/run.py +77 -0
  9. cairn/bench/runner.py +176 -0
  10. cairn/bench/suite.py +34 -0
  11. cairn/bench/workspace.py +29 -0
  12. cairn/cli.py +445 -0
  13. cairn/config.py +16 -0
  14. cairn/detectors/__init__.py +21 -0
  15. cairn/detectors/base.py +120 -0
  16. cairn/detectors/database.py +304 -0
  17. cairn/detectors/docs.py +74 -0
  18. cairn/detectors/identity.py +147 -0
  19. cairn/detectors/manifests.py +124 -0
  20. cairn/detectors/packages.py +102 -0
  21. cairn/detectors/pathrefs.py +54 -0
  22. cairn/detectors/profile.py +215 -0
  23. cairn/discover/__init__.py +0 -0
  24. cairn/discover/files.py +162 -0
  25. cairn/discover/git.py +173 -0
  26. cairn/discover/proc.py +69 -0
  27. cairn/discover/repos.py +143 -0
  28. cairn/emit.py +43 -0
  29. cairn/errors.py +18 -0
  30. cairn/integrations/__init__.py +0 -0
  31. cairn/integrations/claude.py +84 -0
  32. cairn/integrations/config_files.py +187 -0
  33. cairn/integrations/content.py +54 -0
  34. cairn/integrations/git_hooks.py +153 -0
  35. cairn/integrations/harnesses.py +230 -0
  36. cairn/integrations/homes.py +31 -0
  37. cairn/integrations/registry.py +65 -0
  38. cairn/integrations/server_command.py +48 -0
  39. cairn/load.py +54 -0
  40. cairn/match/__init__.py +0 -0
  41. cairn/match/matcher.py +323 -0
  42. cairn/match/overrides.py +106 -0
  43. cairn/match/scoring.py +47 -0
  44. cairn/mcp_server/__init__.py +0 -0
  45. cairn/mcp_server/server.py +113 -0
  46. cairn/mcp_server/tools.py +190 -0
  47. cairn/model/__init__.py +0 -0
  48. cairn/model/graph.py +163 -0
  49. cairn/model/overrides.py +38 -0
  50. cairn/paths.py +45 -0
  51. cairn/py.typed +0 -0
  52. cairn/render/__init__.py +0 -0
  53. cairn/render/card.py +176 -0
  54. cairn/render/index.py +113 -0
  55. cairn/render/markers.py +58 -0
  56. cairn/render/tokens.py +7 -0
  57. cairn/resolve.py +102 -0
  58. cairn/scan.py +253 -0
  59. cairn/scan_cache.py +99 -0
  60. cairn/scan_log.py +24 -0
  61. cairn/security/__init__.py +0 -0
  62. cairn/security/policy.py +52 -0
  63. cairn/security/redact.py +78 -0
  64. cairn/security/text.py +29 -0
  65. cairn/store/__init__.py +0 -0
  66. cairn/store/atomic.py +68 -0
  67. cairn/store/lock.py +115 -0
  68. cairn/store/workspace_store.py +29 -0
  69. cairnmap-0.1.0.dist-info/METADATA +352 -0
  70. cairnmap-0.1.0.dist-info/RECORD +74 -0
  71. cairnmap-0.1.0.dist-info/WHEEL +4 -0
  72. cairnmap-0.1.0.dist-info/entry_points.txt +2 -0
  73. cairnmap-0.1.0.dist-info/licenses/LICENSE +202 -0
  74. cairnmap-0.1.0.dist-info/licenses/NOTICE +6 -0
cairn/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """cairn: a workspace-level repo map and router for coding agents."""
2
+
3
+ __version__ = "0.1.0"
cairn/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from cairn.cli import app
2
+
3
+ app()
@@ -0,0 +1,135 @@
1
+ """Write harness/human-authored content (.cairn/authored/<repo>.yaml) without clobbering it."""
2
+
3
+ import re
4
+ from collections.abc import Iterable, Mapping
5
+ from pathlib import Path
6
+ from typing import Literal
7
+
8
+ import yaml
9
+
10
+ from cairn.errors import CairnError, CairnInputError
11
+ from cairn.load import load_authored
12
+ from cairn.model.graph import SYMMETRIC_TYPES, Edge, EdgeType
13
+ from cairn.model.overrides import Authored
14
+ from cairn.paths import authored_dir
15
+ from cairn.resolve import resolve_repo
16
+ from cairn.store.atomic import atomic_write_text
17
+ from cairn.store.workspace_store import load_workspace
18
+
19
+ _KEY = re.compile(r"^(?P<source>.+?)->(?P<target>.+):(?P<type>[a-z_]+)$")
20
+ Review = Literal["confirmed", "rejected"]
21
+
22
+
23
+ def parse_edge_key(key: str) -> tuple[str, str, EdgeType]:
24
+ match = _KEY.match(key.strip())
25
+ if not match:
26
+ raise CairnInputError(key, "expected an edge key like 'source->target:shares_db'")
27
+ try:
28
+ edge_type = EdgeType(match["type"])
29
+ except ValueError as exc:
30
+ raise CairnInputError(key, f"unknown edge type '{match['type']}'") from exc
31
+ return match["source"], match["target"], edge_type
32
+
33
+
34
+ def save_authored(ws_root: Path, repo_id: str, authored: Authored) -> Path:
35
+ path = authored_dir(ws_root) / f"{repo_id}.yaml"
36
+ data = authored.model_dump(mode="json", exclude_defaults=True)
37
+ atomic_write_text(path, yaml.safe_dump(data, sort_keys=False, allow_unicode=True))
38
+ return path
39
+
40
+
41
+ def annotate_edge(ws_root: Path, key: str, *, review: Review | None, why: str | None) -> Path:
42
+ """Record a review and/or explanation for an edge, keeping every other authored field.
43
+
44
+ The decision is stored under the edge's canonical key (its current direction), and any
45
+ stale copy under the reversed key is removed, so the newest decision always wins.
46
+ An edge the user rejected earlier is no longer in the map but can still be re-decided.
47
+ """
48
+ _, _, edge_type = parse_edge_key(key)
49
+ workspace = load_workspace(ws_root)
50
+ if workspace is None:
51
+ raise CairnError("No map found. Run `cairn scan` first.")
52
+ authored = load_authored(ws_root)
53
+ variants = _key_variants(key, edge_type)
54
+ canonical = _canonical_key(variants, edge_type, workspace.edges, authored)
55
+ if canonical is None:
56
+ raise CairnError(f"No edge '{key}' in the current map. `cairn status` lists edge keys.")
57
+ carried = _drop_variants(ws_root, authored, set(variants) - {canonical})
58
+ source = parse_edge_key(canonical)[0]
59
+ current = load_authored(ws_root).get(source, Authored())
60
+ reviews = {**current.edge_reviews}
61
+ whys = {**current.edge_whys}
62
+ if carried.edge_whys and canonical not in whys:
63
+ whys[canonical] = next(iter(carried.edge_whys.values())) # keep the old explanation
64
+ if review is not None:
65
+ reviews[canonical] = review
66
+ if why:
67
+ whys[canonical] = why
68
+ update = {"edge_reviews": reviews, "edge_whys": whys}
69
+ return save_authored(ws_root, source, current.model_copy(update=update))
70
+
71
+
72
+ def _key_variants(key: str, edge_type: EdgeType) -> tuple[str, ...]:
73
+ source, target, _ = parse_edge_key(key)
74
+ flipped = f"{target}->{source}:{edge_type.value}"
75
+ return (key, flipped) if edge_type in SYMMETRIC_TYPES else (key,)
76
+
77
+
78
+ def _canonical_key(
79
+ variants: tuple[str, ...],
80
+ edge_type: EdgeType,
81
+ edges: Iterable[Edge],
82
+ authored: Mapping[str, Authored],
83
+ ) -> str | None:
84
+ live = next((e.key for e in edges if e.type is edge_type and e.key in variants), None)
85
+ if live is not None:
86
+ return live
87
+ decided = [k for a in authored.values() for k in (*a.edge_reviews, *a.edge_whys)]
88
+ return next((k for k in decided if k in variants), None)
89
+
90
+
91
+ def _drop_variants(ws_root: Path, authored: Mapping[str, Authored], stale: set[str]) -> Authored:
92
+ """Remove stale keys from every authored file; return their whys so they can be carried over."""
93
+ carried: dict[str, str] = {}
94
+ for repo_id, entry in authored.items():
95
+ if not stale & {*entry.edge_reviews, *entry.edge_whys}:
96
+ continue
97
+ carried.update({k: v for k, v in entry.edge_whys.items() if k in stale})
98
+ cleaned = entry.model_copy(
99
+ update={
100
+ "edge_reviews": {k: v for k, v in entry.edge_reviews.items() if k not in stale},
101
+ "edge_whys": {k: v for k, v in entry.edge_whys.items() if k not in stale},
102
+ }
103
+ )
104
+ save_authored(ws_root, repo_id, cleaned)
105
+ return Authored(edge_whys=carried)
106
+
107
+
108
+ SUMMARY_LIMIT = 500
109
+
110
+
111
+ def set_summary(ws_root: Path, repo_id: str, summary: str, *, aliases: Iterable[str] = ()) -> Path:
112
+ """Store a harness/human-written summary (spec §9.2), stamped with the repo's current HEAD."""
113
+ text = " ".join(summary.split())
114
+ if not text:
115
+ raise CairnInputError("summary", "must not be empty")
116
+ if len(text) > SUMMARY_LIMIT:
117
+ raise CairnInputError(
118
+ "summary", f"is {len(text)} characters; keep it under {SUMMARY_LIMIT}"
119
+ )
120
+ workspace = load_workspace(ws_root)
121
+ if workspace is None:
122
+ raise CairnError("No map found. Run `cairn scan` first.")
123
+ repo = workspace.repo(repo_id)
124
+ if repo is None:
125
+ matches = resolve_repo(workspace, load_authored(ws_root), repo_id, limit=3)
126
+ hint = ", ".join(m.repo_id for m in matches) or "none"
127
+ raise CairnError(f"No repo '{repo_id}'. Did you mean: {hint}?")
128
+ current = load_authored(ws_root).get(repo_id, Authored())
129
+ extra = (a.strip().lower() for a in aliases if a.strip())
130
+ update = {
131
+ "summary": text,
132
+ "summary_sha": repo.head_sha,
133
+ "aliases": tuple(dict.fromkeys((*current.aliases, *extra))),
134
+ }
135
+ return save_authored(ws_root, repo_id, current.model_copy(update=update))
File without changes
@@ -0,0 +1,60 @@
1
+ """Set up one isolated workspace copy per benchmark condition (spec §11 E2).
2
+
3
+ A: cold (no map, no CLAUDE.md) B: hand-written RELATED_REPOS-style doc as CLAUDE.md
4
+ C: cairn INDEX only (no cards/graph) D: INDEX + cards E: D + the cairn MCP server
5
+ """
6
+
7
+ import json
8
+ import shutil
9
+ from dataclasses import dataclass
10
+ from pathlib import Path
11
+
12
+ from cairn.bench.suite import Suite
13
+ from cairn.bench.workspace import materialize
14
+ from cairn.emit import write_outputs
15
+ from cairn.integrations.claude import install_claude
16
+ from cairn.integrations.server_command import server_command
17
+ from cairn.paths import cairn_dir, index_file
18
+ from cairn.render.index import render_index
19
+ from cairn.scan import scan_workspace
20
+ from cairn.store.atomic import atomic_write_text
21
+
22
+ CONDITIONS = ("A", "B", "C", "D", "E")
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class Prepared:
27
+ ws: Path
28
+ mcp_config: Path | None
29
+
30
+
31
+ def prepare(condition: str, suite: Suite, suite_dir: Path, run_dir: Path) -> Prepared:
32
+ ws = materialize(suite_dir / suite.workspace, run_dir / "ws")
33
+ if condition in ("A", "B"):
34
+ shutil.rmtree(cairn_dir(ws), ignore_errors=True) # no authored summaries either
35
+ if condition == "B":
36
+ doc = (suite_dir / suite.related_repos_doc).read_text(encoding="utf-8")
37
+ atomic_write_text(ws / "CLAUDE.md", doc)
38
+ return Prepared(ws, None)
39
+ result = scan_workspace(ws)
40
+ write_outputs(ws, result)
41
+ if condition == "C":
42
+ # INDEX only: no card pointer in the text, and no cards or graph JSON to read instead.
43
+ text = render_index(
44
+ result.workspace,
45
+ result.authored,
46
+ threshold=result.config.index_threshold,
47
+ with_cards=False,
48
+ )
49
+ atomic_write_text(index_file(ws), text)
50
+ install_claude(ws)
51
+ shutil.rmtree(cairn_dir(ws))
52
+ return Prepared(ws, None)
53
+ install_claude(ws)
54
+ if condition != "E":
55
+ return Prepared(ws, None)
56
+ config = run_dir / "mcp.json"
57
+ command = [*server_command(), "--workspace", ws.as_posix()]
58
+ entry = {"command": command[0], "args": command[1:]}
59
+ atomic_write_text(config, json.dumps({"mcpServers": {"cairn": entry}}, indent=2))
60
+ return Prepared(ws, config)
cairn/bench/grading.py ADDED
@@ -0,0 +1,119 @@
1
+ """Deterministic grading of benchmark answers (spec §11): file recall or required keywords."""
2
+
3
+ import posixpath
4
+ import re
5
+ from dataclasses import dataclass
6
+ from pathlib import Path
7
+
8
+ from cairn.bench.suite import Task
9
+
10
+ _SEGMENT = r"[\w@.\-\[\]]+"
11
+ # Windows absolute, POSIX absolute, then relative `a/b/c` paths (either slash).
12
+ _PATH = re.compile(rf"[A-Za-z]:[\\/][^\s`'\"()]+|/[^\s`'\"()]+|{_SEGMENT}(?:[\\/]{_SEGMENT})+")
13
+ _TRAILING = ".,;:)"
14
+ _DRIVE = re.compile(r"([A-Za-z]):/(.*)")
15
+ _MSYS = re.compile(r"/([A-Za-z])/(.*)")
16
+ PASS_RECALL = 0.8
17
+
18
+
19
+ @dataclass(frozen=True)
20
+ class Grade:
21
+ success: bool
22
+ recall: float
23
+ precision: float
24
+
25
+
26
+ def _root_spellings(ws_root: Path) -> tuple[str, ...]:
27
+ """The workspace root as an agent might print it: as given, resolved, Git Bash style."""
28
+ spellings: set[str] = set()
29
+ for root in {ws_root.as_posix(), ws_root.resolve().as_posix()}:
30
+ root = root.rstrip("/") + "/"
31
+ spellings.add(root)
32
+ drive = _DRIVE.fullmatch(root)
33
+ if drive:
34
+ spellings.add(f"/{drive.group(1).lower()}/{drive.group(2)}")
35
+ return tuple(sorted(spellings, key=len, reverse=True))
36
+
37
+
38
+ def _inside(path: str, roots: tuple[str, ...]) -> str | None:
39
+ """`path` relative to the workspace, or None when it points elsewhere."""
40
+ for root in roots:
41
+ if path.casefold().startswith(root.casefold()):
42
+ return path[len(root) :]
43
+ windows = any(_DRIVE.match(root) for root in roots)
44
+ try: # 8.3 short names, symlinked temp dirs (macOS /var -> /private/var)
45
+ msys = _MSYS.fullmatch(path) if windows else None
46
+ real = Path(f"{msys.group(1)}:/{msys.group(2)}" if msys else path)
47
+ resolved = real.resolve().as_posix()
48
+ except (OSError, ValueError):
49
+ return None
50
+ for root in roots:
51
+ if resolved.casefold().startswith(root.casefold()):
52
+ return resolved[len(root) :]
53
+ return None
54
+
55
+
56
+ def _heading_repo(line: str, repo_ids: set[str]) -> str | None | bool:
57
+ """For a heading-like line with no paths: the one repo it names, or None.
58
+
59
+ Returns False when the line isn't a heading (the current context carries on).
60
+ """
61
+ stripped = line.strip().strip("-* ").strip()
62
+ heading = line.lstrip().startswith("#") or stripped.endswith(":") or line.strip().endswith("**")
63
+ if not stripped or not heading or _PATH.search(line):
64
+ return False
65
+ named = [r for r in repo_ids if re.search(rf"(?<![\w-]){re.escape(r)}(?![\w-])", line)]
66
+ return named[0] if len(named) == 1 else None
67
+
68
+
69
+ def _strip_root_tail(path: str, ws_parts: tuple[str, ...], repo_ids: set[str]) -> str:
70
+ """`ws/orders-svc/x` (relative to the workspace's parent) -> `orders-svc/x`."""
71
+ segments = path.split("/")
72
+ for i in range(1, min(len(segments), len(ws_parts) + 1)):
73
+ if segments[i] in repo_ids and tuple(segments[:i]) == ws_parts[-i:]:
74
+ return "/".join(segments[i:])
75
+ return path
76
+
77
+
78
+ def mentioned_files(text: str, ws_root: Path, working_repo: str, repo_ids: set[str]) -> set[str]:
79
+ """Workspace paths (`repo/path`) named in `text`.
80
+
81
+ A bare path belongs to the repo named by the heading it sits under (`**orders-svc:**`),
82
+ else to the working repo.
83
+ """
84
+ roots = _root_spellings(ws_root)
85
+ ws_parts = tuple(p for p in ws_root.as_posix().split("/") if p)
86
+ found: set[str] = set()
87
+ context = working_repo
88
+ for line in text.splitlines():
89
+ heading = _heading_repo(line, repo_ids)
90
+ if heading is not False:
91
+ context = heading or working_repo
92
+ continue
93
+ for token in _PATH.findall(line):
94
+ path = token.replace("\\", "/").rstrip(_TRAILING)
95
+ if path.startswith("/") or _DRIVE.match(path):
96
+ inside = _inside(path, roots)
97
+ if inside is None:
98
+ continue
99
+ path = inside
100
+ else:
101
+ path = _strip_root_tail(path, ws_parts, repo_ids)
102
+ if path.split("/", 1)[0] not in repo_ids:
103
+ path = f"{context}/{path}"
104
+ path = posixpath.normpath(path)
105
+ if not path.startswith(".."):
106
+ found.add(path)
107
+ return found
108
+
109
+
110
+ def grade(task: Task, text: str, ws_root: Path, repo_ids: set[str]) -> Grade:
111
+ keywords_ok = all(k.lower() in text.lower() for k in task.expect_keywords)
112
+ if not task.expect_files:
113
+ return Grade(success=keywords_ok, recall=1.0 if keywords_ok else 0.0, precision=1.0)
114
+ found = mentioned_files(text, ws_root, task.repo, repo_ids)
115
+ expected = set(task.expect_files)
116
+ hits = len(found & expected)
117
+ recall = hits / len(expected)
118
+ precision = hits / len(found) if found else 0.0
119
+ return Grade(success=keywords_ok and recall >= PASS_RECALL, recall=recall, precision=precision)
cairn/bench/report.py ADDED
@@ -0,0 +1,76 @@
1
+ """Markdown summary of a benchmark run: per-condition totals, then condition × task."""
2
+
3
+ from collections.abc import Mapping, Sequence
4
+ from statistics import fmean
5
+ from typing import TYPE_CHECKING
6
+
7
+ if TYPE_CHECKING:
8
+ from cairn.bench.run import RunRecord
9
+
10
+
11
+ def _mean(values: Sequence[float]) -> float:
12
+ return fmean(values) if values else 0.0
13
+
14
+
15
+ def _summary(records: Sequence["RunRecord"]) -> list[str]:
16
+ """Success counts every run; token/cost/turn means use only runs that didn't error,
17
+ since an errored run reports zeros that would flatter its condition."""
18
+ lines = [
19
+ "| Condition | Runs | Success | Fresh tokens | Cache-read tokens | Cost (USD) | Turns "
20
+ "| Errors |",
21
+ "|---|---|---|---|---|---|---|---|",
22
+ ]
23
+ for condition in dict.fromkeys(r.condition for r in records):
24
+ rows = [r for r in records if r.condition == condition]
25
+ ok = [r for r in rows if not r.result.is_error]
26
+ success = sum(r.grade.success for r in rows) / len(rows)
27
+ lines.append(
28
+ f"| {condition} | {len(rows)} | {success:.0%} "
29
+ f"| {_mean([r.result.fresh_tokens for r in ok]):,.0f} "
30
+ f"| {_mean([r.result.cache_read_tokens for r in ok]):,.0f} "
31
+ f"| {_mean([r.result.cost_usd for r in ok]):.4f} "
32
+ f"| {_mean([r.result.num_turns for r in ok]):.1f} "
33
+ f"| {len(rows) - len(ok)} |"
34
+ )
35
+ return lines
36
+
37
+
38
+ def _per_task(records: Sequence["RunRecord"]) -> list[str]:
39
+ lines = [
40
+ "| Condition | Task | Success | Recall | Cost (USD) |",
41
+ "|---|---|---|---|---|",
42
+ ]
43
+ for condition in dict.fromkeys(r.condition for r in records):
44
+ for task_id in dict.fromkeys(r.task_id for r in records):
45
+ rows = [r for r in records if r.condition == condition and r.task_id == task_id]
46
+ if not rows:
47
+ continue
48
+ passed = sum(r.grade.success for r in rows)
49
+ lines.append(
50
+ f"| {condition} | {task_id} | {passed}/{len(rows)} "
51
+ f"| {_mean([r.grade.recall for r in rows]):.2f} "
52
+ f"| {_mean([r.result.cost_usd for r in rows if not r.result.is_error]):.4f} |"
53
+ )
54
+ return lines
55
+
56
+
57
+ def render_markdown(
58
+ records: Sequence["RunRecord"], *, meta: Mapping[str, object] | None = None
59
+ ) -> str:
60
+ errors = sum(r.result.is_error for r in records)
61
+ about = [f"{key}: {value}" for key, value in (meta or {}).items() if key != "tasks"]
62
+ parts = [
63
+ "# cairn benchmark",
64
+ "",
65
+ *([", ".join(about), ""] if about else []),
66
+ "Conditions: A cold, B hand-written doc, C cairn INDEX, D INDEX + cards, E D + MCP",
67
+ "",
68
+ *_summary(records),
69
+ "",
70
+ "## Per task",
71
+ "",
72
+ *_per_task(records),
73
+ ]
74
+ if errors:
75
+ parts += ["", f"{errors} run(s) ended in an agent error; see the JSON for details."]
76
+ return "\n".join(parts) + "\n"
cairn/bench/run.py ADDED
@@ -0,0 +1,77 @@
1
+ """Run a benchmark suite across conditions and write a report (spec §11 E2)."""
2
+
3
+ import json
4
+ import tempfile
5
+ from collections.abc import Mapping, Sequence
6
+ from dataclasses import asdict, dataclass
7
+ from datetime import UTC, datetime
8
+ from pathlib import Path
9
+
10
+ import cairn
11
+ from cairn.bench.conditions import prepare
12
+ from cairn.bench.grading import Grade, grade
13
+ from cairn.bench.report import render_markdown
14
+ from cairn.bench.runner import Runner, RunResult
15
+ from cairn.bench.suite import load_suite
16
+ from cairn.store.atomic import atomic_write_text
17
+
18
+ __all__ = ["RunRecord", "render_markdown", "run_bench"]
19
+
20
+
21
+ @dataclass(frozen=True)
22
+ class RunRecord:
23
+ condition: str
24
+ task_id: str
25
+ run: int
26
+ grade: Grade
27
+ result: RunResult
28
+
29
+
30
+ def run_bench(
31
+ suite_dir: Path,
32
+ *,
33
+ conditions: Sequence[str],
34
+ runs: int,
35
+ task_ids: Sequence[str] | None,
36
+ runner: Runner,
37
+ out_dir: Path,
38
+ now: str,
39
+ meta: Mapping[str, object] | None = None,
40
+ ) -> tuple[RunRecord, ...]:
41
+ """Each finished run is appended to `<now>.jsonl` at once, so an interrupted (paid)
42
+ benchmark keeps what it already measured; `<now>.json` and `.md` are written at the end."""
43
+ suite = load_suite(suite_dir)
44
+ tasks = [t for t in suite.tasks if not task_ids or t.id in task_ids]
45
+ header = {
46
+ "cairn_version": cairn.__version__,
47
+ "suite": suite.name,
48
+ "conditions": list(conditions),
49
+ "runs": runs,
50
+ "tasks": [t.id for t in tasks],
51
+ "started_at": datetime.now(UTC).isoformat(timespec="seconds"),
52
+ **(meta or {}),
53
+ }
54
+ out_dir.mkdir(parents=True, exist_ok=True)
55
+ log = out_dir / f"{now}.jsonl"
56
+ records: list[RunRecord] = []
57
+ for condition in conditions:
58
+ for task in tasks:
59
+ for index in range(runs):
60
+ # The MCP server of condition E may still hold files for a moment on Windows.
61
+ with tempfile.TemporaryDirectory(
62
+ prefix="cairn-bench-", ignore_cleanup_errors=True
63
+ ) as tmp:
64
+ prepared = prepare(condition, suite, suite_dir, Path(tmp))
65
+ ws = prepared.ws
66
+ repo_ids = {p.name for p in ws.iterdir() if (p / ".git").exists()}
67
+ result = runner.run(task.prompt, ws / task.repo, ws, prepared.mcp_config)
68
+ graded = grade(task, result.result_text, ws, repo_ids)
69
+ record = RunRecord(condition, task.id, index, graded, result)
70
+ records.append(record)
71
+ with log.open("a", encoding="utf-8") as handle:
72
+ handle.write(json.dumps(asdict(record)) + "\n")
73
+ out = tuple(records)
74
+ report = {"meta": header, "records": [asdict(r) for r in out]}
75
+ atomic_write_text(out_dir / f"{now}.json", json.dumps(report, indent=2))
76
+ atomic_write_text(out_dir / f"{now}.md", render_markdown(out, meta=header))
77
+ return out
cairn/bench/runner.py ADDED
@@ -0,0 +1,176 @@
1
+ """Run one benchmark prompt through a headless agent (spec §11 E2).
2
+
3
+ ClaudeRunner drives `claude -p` in an isolated, read-only session: user settings and
4
+ memory are not loaded (`--setting-sources project,local`), only the MCP servers in the
5
+ condition's config are visible (`--strict-mcp-config`), and only read tools are allowed
6
+ (no shell: even `find` can delete or run programs through `-delete`/`-exec`).
7
+ """
8
+
9
+ import json
10
+ import os
11
+ import re
12
+ import shutil
13
+ import subprocess
14
+ from collections.abc import Callable
15
+ from dataclasses import dataclass
16
+ from pathlib import Path
17
+ from typing import Protocol
18
+
19
+ READ_ONLY_TOOLS = ("Read", "Grep", "Glob", "mcp__cairn")
20
+
21
+
22
+ @dataclass(frozen=True)
23
+ class RunResult:
24
+ result_text: str
25
+ num_turns: int = 0
26
+ cost_usd: float = 0.0
27
+ input_tokens: int = 0
28
+ cache_creation_tokens: int = 0
29
+ cache_read_tokens: int = 0
30
+ output_tokens: int = 0
31
+ duration_ms: int = 0
32
+ is_error: bool = False
33
+ models: tuple[str, ...] = ()
34
+
35
+ @property
36
+ def fresh_tokens(self) -> int:
37
+ """Tokens the model actually processed fresh (cache reads are near-free)."""
38
+ return self.input_tokens + self.cache_creation_tokens + self.output_tokens
39
+
40
+
41
+ def parse_result(raw: str) -> RunResult:
42
+ try:
43
+ data = json.loads(raw)
44
+ except json.JSONDecodeError:
45
+ return RunResult(result_text=raw[-2000:], is_error=True)
46
+ if not isinstance(data, dict):
47
+ return RunResult(result_text=raw[-2000:], is_error=True)
48
+ usage = data.get("usage") or {}
49
+ model_usage = data.get("modelUsage")
50
+ return RunResult(
51
+ result_text=str(data.get("result") or ""),
52
+ num_turns=int(data.get("num_turns") or 0),
53
+ cost_usd=float(data.get("total_cost_usd") or 0.0),
54
+ input_tokens=int(usage.get("input_tokens") or 0),
55
+ cache_creation_tokens=int(usage.get("cache_creation_input_tokens") or 0),
56
+ cache_read_tokens=int(usage.get("cache_read_input_tokens") or 0),
57
+ output_tokens=int(usage.get("output_tokens") or 0),
58
+ duration_ms=int(data.get("duration_ms") or 0),
59
+ is_error=bool(data.get("is_error")),
60
+ models=tuple(sorted(model_usage)) if isinstance(model_usage, dict) else (),
61
+ )
62
+
63
+
64
+ def claude_home() -> Path:
65
+ """Claude Code's config folder (the user's real one: benchmarks only clean up after runs)."""
66
+ override = os.environ.get("CLAUDE_CONFIG_DIR")
67
+ return Path(override) if override else Path.home() / ".claude"
68
+
69
+
70
+ def project_slug(cwd: Path) -> str:
71
+ """How Claude Code names a working directory's folder under `projects/`."""
72
+ return re.sub(r"[^A-Za-z0-9]", "-", str(cwd))
73
+
74
+
75
+ def forget_project(home: Path, cwd: Path) -> None:
76
+ """Remove the per-cwd folder a headless run leaves behind, if it holds no files."""
77
+ target = home / "projects" / project_slug(cwd)
78
+ if target.is_dir() and not any(p.is_file() for p in target.rglob("*")):
79
+ shutil.rmtree(target, ignore_errors=True)
80
+
81
+
82
+ class Runner(Protocol):
83
+ def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult: ...
84
+
85
+
86
+ @dataclass(frozen=True)
87
+ class ClaudeRunner:
88
+ model: str = "haiku"
89
+ timeout: float = 900.0
90
+ home: Path | None = None # Claude Code's config folder; default claude_home()
91
+
92
+ def isolation_settings(self, ws: Path) -> dict[str, object]:
93
+ """Keep the benchmarking user's own instructions out of every run.
94
+
95
+ `--setting-sources project,local` doesn't cover memory files: ~/.claude/CLAUDE.md
96
+ and ~/.claude/rules still load, as would a CLAUDE.md in any folder above the
97
+ temporary workspace. Only the condition's workspace CLAUDE.md may load.
98
+ """
99
+ home = (self.home or claude_home()).as_posix().rstrip("/")
100
+ above = [p.as_posix().rstrip("/") for p in ws.parents]
101
+ excludes = [f"{home}/**"]
102
+ excludes += [f"{d}/{name}" for d in above for name in ("CLAUDE.md", "CLAUDE.local.md")]
103
+ excludes += [f"{d}/.claude/**" for d in above]
104
+ return {"claudeMdExcludes": excludes, "autoMemoryEnabled": False}
105
+
106
+ def command(self, prompt: str, ws: Path, mcp_config: Path | None) -> list[str]:
107
+ # The resolved path, so npm's `claude.cmd` shim launches on Windows too.
108
+ cmd = [
109
+ shutil.which("claude") or "claude",
110
+ "-p",
111
+ prompt,
112
+ "--output-format",
113
+ "json",
114
+ "--model",
115
+ self.model,
116
+ "--setting-sources",
117
+ "project,local",
118
+ "--settings",
119
+ json.dumps(self.isolation_settings(ws)),
120
+ "--no-session-persistence",
121
+ "--permission-mode",
122
+ "dontAsk",
123
+ "--allowedTools",
124
+ *READ_ONLY_TOOLS,
125
+ "--add-dir",
126
+ str(ws),
127
+ "--strict-mcp-config",
128
+ ]
129
+ if mcp_config is not None:
130
+ cmd += ["--mcp-config", str(mcp_config)]
131
+ return cmd
132
+
133
+ def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult:
134
+ try:
135
+ proc = subprocess.run(
136
+ self.command(prompt, ws, mcp_config),
137
+ cwd=cwd,
138
+ capture_output=True,
139
+ text=True,
140
+ encoding="utf-8",
141
+ errors="replace",
142
+ timeout=self.timeout,
143
+ check=False,
144
+ )
145
+ except (OSError, subprocess.TimeoutExpired) as exc:
146
+ return RunResult(result_text=f"runner error: {exc}", is_error=True)
147
+ finally:
148
+ home = self.home or claude_home()
149
+ for path in {cwd, cwd.resolve()}:
150
+ forget_project(home, path)
151
+ return parse_result(proc.stdout or proc.stderr)
152
+
153
+ def version(self) -> str:
154
+ try:
155
+ done = subprocess.run(
156
+ [shutil.which("claude") or "claude", "--version"],
157
+ capture_output=True,
158
+ text=True,
159
+ encoding="utf-8",
160
+ errors="replace",
161
+ timeout=60,
162
+ check=False,
163
+ )
164
+ except (OSError, subprocess.TimeoutExpired):
165
+ return "unknown"
166
+ return done.stdout.strip() or "unknown"
167
+
168
+
169
+ @dataclass(frozen=True)
170
+ class FakeRunner:
171
+ """Test double: `reply(prompt, cwd)` returns the raw JSON claude would print."""
172
+
173
+ reply: Callable[[str, Path], str]
174
+
175
+ def run(self, prompt: str, cwd: Path, ws: Path, mcp_config: Path | None) -> RunResult:
176
+ return parse_result(self.reply(prompt, cwd))