cairn-browser-mcp 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- cairn_browser_mcp-0.1.0/.gitignore +75 -0
- cairn_browser_mcp-0.1.0/CLAUDE.md +64 -0
- cairn_browser_mcp-0.1.0/PKG-INFO +60 -0
- cairn_browser_mcp-0.1.0/PLAN.md +46 -0
- cairn_browser_mcp-0.1.0/PROGRESS.md +201 -0
- cairn_browser_mcp-0.1.0/README.md +34 -0
- cairn_browser_mcp-0.1.0/pyproject.toml +67 -0
- cairn_browser_mcp-0.1.0/src/cairn_mcp/__init__.py +6 -0
- cairn_browser_mcp-0.1.0/src/cairn_mcp/__main__.py +11 -0
- cairn_browser_mcp-0.1.0/src/cairn_mcp/server.py +1064 -0
- cairn_browser_mcp-0.1.0/tests/conftest.py +96 -0
- cairn_browser_mcp-0.1.0/tests/helpers.py +81 -0
- cairn_browser_mcp-0.1.0/tests/test_commons.py +249 -0
- cairn_browser_mcp-0.1.0/tests/test_market.py +226 -0
- cairn_browser_mcp-0.1.0/tests/test_server.py +403 -0
- cairn_browser_mcp-0.1.0/tests/test_surface.py +300 -0
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
# --- Python (package/, mcp/, backend/) ---
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.egg-info/
|
|
6
|
+
.eggs/
|
|
7
|
+
dist/
|
|
8
|
+
build/
|
|
9
|
+
.venv/
|
|
10
|
+
venv/
|
|
11
|
+
env/
|
|
12
|
+
.python-version
|
|
13
|
+
|
|
14
|
+
# test + type + lint caches
|
|
15
|
+
.pytest_cache/
|
|
16
|
+
.mypy_cache/
|
|
17
|
+
.ruff_cache/
|
|
18
|
+
.coverage
|
|
19
|
+
coverage.xml
|
|
20
|
+
htmlcov/
|
|
21
|
+
|
|
22
|
+
# --- Node / Next.js (frontend/) ---
|
|
23
|
+
node_modules/
|
|
24
|
+
.next/
|
|
25
|
+
out/
|
|
26
|
+
.vercel/
|
|
27
|
+
*.tsbuildinfo
|
|
28
|
+
next-env.d.ts
|
|
29
|
+
npm-debug.log*
|
|
30
|
+
yarn-debug.log*
|
|
31
|
+
yarn-error.log*
|
|
32
|
+
.pnpm-debug.log*
|
|
33
|
+
|
|
34
|
+
# --- Secrets ---
|
|
35
|
+
# Cairn needs NO API key to run. If one ever appears (OpenRouter, x402), it stays out of git.
|
|
36
|
+
.env
|
|
37
|
+
.env.*
|
|
38
|
+
!.env.example
|
|
39
|
+
*.pem
|
|
40
|
+
|
|
41
|
+
# --- Cairn runtime state ---
|
|
42
|
+
# Memory lives in ~/.sibyl-memory by default. If it is ever written next to the
|
|
43
|
+
# code, it must never be committed: the deletion-gate test needs a real wipe.
|
|
44
|
+
*.db
|
|
45
|
+
*.db-journal
|
|
46
|
+
*.sqlite
|
|
47
|
+
*.sqlite3
|
|
48
|
+
.cairn/
|
|
49
|
+
runs/
|
|
50
|
+
|
|
51
|
+
# --- Browser automation output ---
|
|
52
|
+
.playwright-mcp/
|
|
53
|
+
playwright-report/
|
|
54
|
+
test-results/
|
|
55
|
+
blob-report/
|
|
56
|
+
.cache/ms-playwright/
|
|
57
|
+
|
|
58
|
+
# --- Agent tooling ---
|
|
59
|
+
# Third-party skills are cloned in, not authored here. Reinstall with:
|
|
60
|
+
# git clone --depth 1 https://github.com/greensock/gsap-skills
|
|
61
|
+
# cp -r gsap-skills/skills/* .claude/skills/
|
|
62
|
+
.claude/skills/
|
|
63
|
+
.claude/settings.local.json
|
|
64
|
+
|
|
65
|
+
# --- Scratch ---
|
|
66
|
+
*.b64.txt
|
|
67
|
+
scratch/
|
|
68
|
+
tmp/
|
|
69
|
+
|
|
70
|
+
# --- OS / editor ---
|
|
71
|
+
.DS_Store
|
|
72
|
+
Thumbs.db
|
|
73
|
+
desktop.ini
|
|
74
|
+
.idea/
|
|
75
|
+
*.swp
|
|
@@ -0,0 +1,64 @@
|
|
|
1
|
+
# CLAUDE.md — mcp/ (THE PRODUCT)
|
|
2
|
+
|
|
3
|
+
**This folder is the main deliverable.** Cairn as an MCP server, so Claude Code, Codex or
|
|
4
|
+
Cursor becomes the brain and gets a browser that remembers websites.
|
|
5
|
+
|
|
6
|
+
Phase 2. Read `PLAN.md` for what to build, `PROGRESS.md` for where we are. Root rules in
|
|
7
|
+
`../CLAUDE.md` apply. Depends on `package/` (Phase 1) only.
|
|
8
|
+
|
|
9
|
+
## Why this shape (do not drift back)
|
|
10
|
+
|
|
11
|
+
The host AI does the thinking. Cairn supplies the browser and the memory. That means: no API
|
|
12
|
+
key for us or the user, one-command install, and a judge can try it inside their own Claude
|
|
13
|
+
Code. Warm replay is deterministic — zero model calls.
|
|
14
|
+
|
|
15
|
+
## Stack (versions verified 2026-08-31)
|
|
16
|
+
|
|
17
|
+
Python 3.11+ · official `mcp` SDK 2.1.1 **or** `fastmcp` 3.4.7 (decide at build time after
|
|
18
|
+
opening both docs — RESEARCH rule) · the cairn package installed editable.
|
|
19
|
+
|
|
20
|
+
## The one architecture rule
|
|
21
|
+
|
|
22
|
+
Thin wrapper. Imports `package/` ONLY — never backend/ or frontend/. No browser logic, no
|
|
23
|
+
Sibyl calls, no model calls here. Each tool is a few lines calling into `cairn`. If a tool
|
|
24
|
+
needs real logic, that logic belongs in the package.
|
|
25
|
+
|
|
26
|
+
## Tools to expose
|
|
27
|
+
|
|
28
|
+
**Cold path (the host AI drives these to learn a new site):**
|
|
29
|
+
|
|
30
|
+
| tool | does |
|
|
31
|
+
|---|---|
|
|
32
|
+
| `cairn_act(intent, action, ref?, value?, to?)` | ONE tool for all 31 actions, chosen by the `action` argument |
|
|
33
|
+
| `cairn_read(kind, ref?, attribute?)` | `kind="page"` lists the controls; the other 12 kinds read one element |
|
|
34
|
+
| `cairn_save(task)` | distil this session's trace into a playbook, store it |
|
|
35
|
+
|
|
36
|
+
**One tool per verb, never one tool per action (locked 2026-09-01).** Tool choice is the
|
|
37
|
+
most fragile part of this system — on the first live test a host AI ignored Cairn and used
|
|
38
|
+
`curl`. Thirty-one tool names to choose between makes that worse, not better. Both tool
|
|
39
|
+
descriptions are GENERATED from `actions.ACTIONS` and `reads.READS`, so an action can never
|
|
40
|
+
exist without being discoverable.
|
|
41
|
+
|
|
42
|
+
**Warm path (one call, no thinking):**
|
|
43
|
+
|
|
44
|
+
| tool | does |
|
|
45
|
+
|---|---|
|
|
46
|
+
| `cairn_run(task, site)` | replay the playbook deterministically; repairs a broken step by asking the host AI only for that step |
|
|
47
|
+
| `cairn_sites()` | learned sites + playbook health |
|
|
48
|
+
| `cairn_show(domain)` | the playbook, human-readable |
|
|
49
|
+
| `cairn_forget(domain)` | wipe a site's memory — the gate test, from inside any MCP client |
|
|
50
|
+
|
|
51
|
+
Tool descriptions ARE the UX — the host AI decides what to call from them alone. One dedicated
|
|
52
|
+
revision pass on the wording is mandatory before this phase is done.
|
|
53
|
+
|
|
54
|
+
## Clean code rules
|
|
55
|
+
|
|
56
|
+
- One file if it fits (`server.py`), two at most. Type hints, ruff clean.
|
|
57
|
+
- Long operations must report progress, never hang the client silently.
|
|
58
|
+
- Errors return readable messages, never stack traces.
|
|
59
|
+
- Never print to stdout — it corrupts stdio MCP transport. Log to stderr only.
|
|
60
|
+
|
|
61
|
+
## Definition of done
|
|
62
|
+
|
|
63
|
+
Works from a CLEAN Claude Code session in a DIFFERENT folder + install instructions tested by
|
|
64
|
+
following them literally + PROGRESS.md updated.
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: cairn-browser-mcp
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Cairn as MCP tools: a browser with a memory, for Claude Code, Cursor and Codex.
|
|
5
|
+
Project-URL: Homepage, https://github.com/rohit-jsfreaky/cairn
|
|
6
|
+
Project-URL: Repository, https://github.com/rohit-jsfreaky/cairn
|
|
7
|
+
Project-URL: Issues, https://github.com/rohit-jsfreaky/cairn/issues
|
|
8
|
+
Author: Rohit Kashyap
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
16
|
+
Classifier: Topic :: Internet :: WWW/HTTP :: Browsers
|
|
17
|
+
Classifier: Topic :: Software Development :: Testing :: Acceptance
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Requires-Dist: cairn-browser
|
|
20
|
+
Requires-Dist: mcp<2,>=1.29.1
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
23
|
+
Provides-Extra: market
|
|
24
|
+
Requires-Dist: cairn-browser[market]; extra == 'market'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# Cairn as MCP tools
|
|
28
|
+
|
|
29
|
+
**Your AI can use websites. But it forgets how, every single time. Cairn makes it remember.**
|
|
30
|
+
|
|
31
|
+
This gives Claude Code, Cursor, Codex or any MCP client a browser that remembers. Your AI
|
|
32
|
+
walks a site once and Cairn writes down the route; every run after that is **one tool call,
|
|
33
|
+
no page reading, and zero model calls**. When the site changes, Cairn hands back the single
|
|
34
|
+
step that moved rather than the whole task.
|
|
35
|
+
|
|
36
|
+
```bash
|
|
37
|
+
pip install cairn-browser-mcp
|
|
38
|
+
playwright install chromium # the browser is a separate download
|
|
39
|
+
|
|
40
|
+
claude mcp add cairn -- cairn-mcp
|
|
41
|
+
```
|
|
42
|
+
|
|
43
|
+
Then just ask, in your own words: *"go to github.com/microsoft/playwright and tell me how
|
|
44
|
+
many open issues it has."* The first time it explores. After that it is one call.
|
|
45
|
+
|
|
46
|
+
**No API key.** There is no model call anywhere in this package or anything it imports —
|
|
47
|
+
your AI does the thinking and Cairn supplies the browser and the memory.
|
|
48
|
+
|
|
49
|
+
Optional extra, so one agent can buy a trail from another over x402 on Base:
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
pip install "cairn-browser-mcp[market]"
|
|
53
|
+
```
|
|
54
|
+
|
|
55
|
+
The engine on its own is [`cairn-browser`](https://pypi.org/project/cairn-browser/).
|
|
56
|
+
|
|
57
|
+
Full documentation, the tool list, and the deletion test that proves the memory is
|
|
58
|
+
load-bearing: **https://github.com/rohit-jsfreaky/cairn**
|
|
59
|
+
|
|
60
|
+
MIT licensed.
|
|
@@ -0,0 +1,46 @@
|
|
|
1
|
+
# PLAN — mcp/ (Phase 2, Sep 4–5)
|
|
2
|
+
|
|
3
|
+
Finish line in `../MASTER-PLAN.md`. Start after Phase 1 passes. This is the main deliverable —
|
|
4
|
+
it is NOT cuttable.
|
|
5
|
+
|
|
6
|
+
### 2a. Research (timebox 30 min)
|
|
7
|
+
- Open the `mcp` (2.1.1) and `fastmcp` (3.4.7) docs with Playwright. Pick one, note the choice
|
|
8
|
+
and reason in `../RESEARCH.md`.
|
|
9
|
+
- Look at how `sibyl-memory-mcp` structures its server — same sponsor, good reference.
|
|
10
|
+
|
|
11
|
+
### 2b. Cold-path tools
|
|
12
|
+
- `cairn_look`, `cairn_act`, `cairn_save` wired to `cairn.operations` + `cairn.distill`.
|
|
13
|
+
- Session handling: one browser session per MCP connection, cleaned up properly on exit.
|
|
14
|
+
- ✅ in Claude Code, asking it to do the demo task works: it looks, acts, and saves a playbook.
|
|
15
|
+
|
|
16
|
+
### 2c. Warm-path tools
|
|
17
|
+
- `cairn_run`, `cairn_sites`, `cairn_show`, `cairn_forget`.
|
|
18
|
+
- Repair inside `cairn_run`: when a step breaks, return a precise repair request so the host AI
|
|
19
|
+
fixes only that step, then persist it.
|
|
20
|
+
- ✅ the whole four-beat flow works from Claude Code with no terminal.
|
|
21
|
+
|
|
22
|
+
### 2d. Tool description pass
|
|
23
|
+
- Rewrite every description so the host AI reliably picks `cairn_run` (warm) over the cold
|
|
24
|
+
tools when a playbook already exists. Test by asking vaguely: "get my invoice from that
|
|
25
|
+
portal" — it should go straight to `cairn_run`.
|
|
26
|
+
- ✅ three vague prompts in a row route correctly.
|
|
27
|
+
|
|
28
|
+
### 2e. Install path + adoption
|
|
29
|
+
- `claude mcp add` line, plus config snippets for Codex and Cursor.
|
|
30
|
+
- **OpenClaw (388k★) — confirmed: docs.openclaw.ai lists "Connect MCP servers", so our server
|
|
31
|
+
plugs in with install docs only, no extra code.** Note their users ALREADY have browser
|
|
32
|
+
tools (Browser control API, Chrome Extension, etc.), so the message there is not "here is a
|
|
33
|
+
browser" but "your browser stops re-learning the same site". They also run ClawHub + a
|
|
34
|
+
5,400-skill registry — a good place to be seen. See ../RESEARCH.md.
|
|
35
|
+
- **Hermes Agent (Nous Research) — check this before writing the install docs.** Sibyl
|
|
36
|
+
officially supports Hermes, so it is inside the sponsor's own world. First find out whether
|
|
37
|
+
Hermes loads MCP servers directly: if yes, we get it free and only need install docs; if no,
|
|
38
|
+
a thin plugin adapter (mirror how `sibyl-memory-hermes` installs into
|
|
39
|
+
`$HERMES_HOME/plugins/`). Hermes is the strongest home for Cairn because it runs
|
|
40
|
+
**unattended on a cron schedule** — exactly where deterministic replay beats an improvising
|
|
41
|
+
AI. See ../RESEARCH.md.
|
|
42
|
+
- README install section, tested by following it literally in a clean folder.
|
|
43
|
+
- Share in the hackathon Discord — one non-Rohit person installs and runs it. That is real,
|
|
44
|
+
verifiable PMF evidence (and the rules require evidence a judge can check in 5 minutes).
|
|
45
|
+
- ✅ Phase 2 finish line: clean Claude Code session learns a site, quit, fresh session replays
|
|
46
|
+
it in seconds, `cairn_forget` makes it slow again.
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
# PROGRESS — mcp/ (THE PRODUCT)
|
|
2
|
+
|
|
3
|
+
> Working memory for this folder. Read first, update before ending every session.
|
|
4
|
+
|
|
5
|
+
## Current state — 2026-09-01
|
|
6
|
+
|
|
7
|
+
**PHASE 2 IS DONE. The finish line passed on a live Claude Code session, 2026-09-01.**
|
|
8
|
+
|
|
9
|
+
24 tests pass, ruff clean, real stdio transport verified, and all four beats ran in a
|
|
10
|
+
CLEAN Claude Code session in a DIFFERENT folder (`D:\my_projects\project_research`).
|
|
11
|
+
|
|
12
|
+
```
|
|
13
|
+
pytest mcp 22 passed
|
|
14
|
+
ruff check mcp All checks passed
|
|
15
|
+
stdio handshake server: cairn, 9 tools listed, structured results returned
|
|
16
|
+
```
|
|
17
|
+
|
|
18
|
+
## What exists
|
|
19
|
+
|
|
20
|
+
```
|
|
21
|
+
mcp/
|
|
22
|
+
pyproject.toml distribution `cairn-mcp`, console script `cairn-mcp`
|
|
23
|
+
README.md install for Claude Code / Cursor / Codex + the 4-beat demo
|
|
24
|
+
src/cairn_mcp/
|
|
25
|
+
server.py all 9 tools, thin wrappers over the engine
|
|
26
|
+
__main__.py `python -m cairn_mcp`
|
|
27
|
+
tests/
|
|
28
|
+
conftest.py demo site + a server with its own memory db
|
|
29
|
+
helpers.py call() a tool the way a host AI would
|
|
30
|
+
test_server.py 22 tests
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Plus `../.mcp.json` at the repo root, so Claude Code opened in this repo offers the server
|
|
34
|
+
with a **relative** command — clone, make the venv, approve, done. Verified with
|
|
35
|
+
`claude mcp list`: `cairn: .venv/Scripts/python.exe -m cairn_mcp - Pending approval`.
|
|
36
|
+
|
|
37
|
+
## The 9 tools
|
|
38
|
+
|
|
39
|
+
| tool | path | note |
|
|
40
|
+
|---|---|---|
|
|
41
|
+
| `cairn_run` | warm | the product. One call, 0 model calls, 0 pages read. |
|
|
42
|
+
| `cairn_repair` | warm | applies the host AI's fix to ONE step |
|
|
43
|
+
| `cairn_sites` `cairn_show` `cairn_forget` | warm | inspect and wipe |
|
|
44
|
+
| `cairn_open` `cairn_look` `cairn_act` `cairn_save` | cold | only for a site never seen |
|
|
45
|
+
|
|
46
|
+
Two departures from the table in CLAUDE.md, both deliberate:
|
|
47
|
+
|
|
48
|
+
- **`cairn_open` added.** The plan routed navigation through `cairn_act(action="goto")`.
|
|
49
|
+
That is awkward to describe and easy for a host AI to get wrong, and tool descriptions
|
|
50
|
+
are the entire UX here. A named tool for "start here" is clearer.
|
|
51
|
+
- **`cairn_repair` added.** `cairn_run` can detect a break and describe it, but applying
|
|
52
|
+
the fix needs a second call, because the host AI is the thing that decides what the new
|
|
53
|
+
control is. Without this tool the repair loop cannot close.
|
|
54
|
+
|
|
55
|
+
## Decisions
|
|
56
|
+
|
|
57
|
+
- **Official `mcp` SDK, not the `fastmcp` package.** Reasons in `../RESEARCH.md`; the short
|
|
58
|
+
version is that FastMCP ships inside the official SDK, `mcp` was already installed, and
|
|
59
|
+
Sibyl's own MCP server uses exactly this shape.
|
|
60
|
+
- **The browser runs on its own thread** (`cairn/worker.py` in the engine). Playwright's
|
|
61
|
+
sync API binds objects to their creating thread and refuses to run inside an asyncio
|
|
62
|
+
loop; an MCP server breaks both rules. Marshalling every call onto one thread is the
|
|
63
|
+
proper fix, not a workaround.
|
|
64
|
+
- **Every tool returns a `next` string** when the AI has more to do. The result is the only
|
|
65
|
+
thing it sees, so telling it what to do next is not decoration, it is the control flow.
|
|
66
|
+
- **Tool descriptions are tested.** `TestToolDescriptions` asserts `cairn_run` says
|
|
67
|
+
"TRY THIS FIRST" and appears before `cairn_open` in the server instructions. A host AI
|
|
68
|
+
that reaches for the cold tools on a known site would quietly destroy the whole point of
|
|
69
|
+
the project, so the wording that prevents it cannot be allowed to rot silently.
|
|
70
|
+
|
|
71
|
+
## The finish line — PASSED 2026-09-01, observed live
|
|
72
|
+
|
|
73
|
+
Clean Claude Code (v2.1.252, Opus 5), folder `D:\my_projects\project_research`, which has
|
|
74
|
+
never contained any Cairn code.
|
|
75
|
+
|
|
76
|
+
| beat | asked | what happened |
|
|
77
|
+
|---|---|---|
|
|
78
|
+
| recall | "Download this month's invoice from http://127.0.0.1:8787" | **one** cairn call. 4 steps replayed, ~2s, no page reading. Reported the saved path. |
|
|
79
|
+
| repair | "same thing but the site is now at ...?variant=b" | 2 cairn calls. 3 steps replayed, step 4 broke, it chose `#get-pdf` ("Get PDF"), repaired, re-ran. Said the fix is saved so the next run needs no repair. |
|
|
80
|
+
| gate | "Forget 127.0.0.1:8787" | forgotten, and it explained the consequence unprompted: the next download "will need full exploring again". |
|
|
81
|
+
|
|
82
|
+
**Discovery worked with no prompting.** It went straight to `cairn_run` without being told
|
|
83
|
+
to use Cairn — the description rewrite (below) is what fixed that.
|
|
84
|
+
|
|
85
|
+
### The two failures that got us here, both real
|
|
86
|
+
|
|
87
|
+
1. **It used `curl` and ignored Cairn entirely.** Cause: `cairn_run`'s summary line read
|
|
88
|
+
"a website **that Cairn already knows**" — a condition the AI cannot evaluate while
|
|
89
|
+
scanning the tool list, so it skipped the tool. And "TRY THIS FIRST, before any other
|
|
90
|
+
**Cairn** tool" only ranked our own tools against each other; it never mentioned the
|
|
91
|
+
shell, which is what it actually chose. Fixed by opening unconditionally with "USE THIS
|
|
92
|
+
FOR ANY WEBSITE TASK" and naming curl/wget/fetch/shell explicitly. Pinned by tests.
|
|
93
|
+
|
|
94
|
+
2. **Downloads never reached disk.** Found by the host AI itself, not by our tests.
|
|
95
|
+
`Browser` accepted a `downloads` path that nobody passed, so Playwright deleted the file
|
|
96
|
+
when the context closed. Our test only asserted the download *event* fired — it was
|
|
97
|
+
green while proving the wrong thing. Fixed with a default `~/.cairn/downloads`, deferred
|
|
98
|
+
saving (saving inside Playwright's event callback fails with "Download.save_as:
|
|
99
|
+
canceled"), and `saved_files` reported through run/CLI/MCP. Tests now assert a real
|
|
100
|
+
non-empty file.
|
|
101
|
+
|
|
102
|
+
**Lesson worth keeping:** a real host AI on a real task found a bug that 94 passing tests
|
|
103
|
+
did not. Assert the user-visible outcome, not the internal event.
|
|
104
|
+
|
|
105
|
+
**Phase 1g is DONE (2026-09-02).** Proven through these tools on 8 real websites, two of
|
|
106
|
+
them signed in. Nothing here is demo-site-only any more.
|
|
107
|
+
|
|
108
|
+
## Session log
|
|
109
|
+
|
|
110
|
+
- **2026-08-31** — folder created, plan written. No code.
|
|
111
|
+
- **2026-09-01** — Phase 2 built in one pass. SDK decision closed by reading Sibyl's own
|
|
112
|
+
installed server rather than docs. Found and fixed the Playwright-threading problem with a
|
|
113
|
+
dedicated browser thread in the engine. `call_tool` turned out to return a
|
|
114
|
+
`(blocks, structured)` tuple, which cost one debugging round. Install docs written and the
|
|
115
|
+
project-scoped `.mcp.json` verified with `claude mcp list`.
|
|
116
|
+
- **2026-09-02/03** — surface collapsed to `cairn_act` + `cairn_read`, both descriptions
|
|
117
|
+
GENERATED from the engine registries so a capability cannot exist without being
|
|
118
|
+
discoverable. Three commons tools added (`cairn_share`, `cairn_borrow`, `cairn_commons`)
|
|
119
|
+
and the `cairn_run` miss branch rewritten so `next` is REPLACED, not appended — left as it
|
|
120
|
+
was it said "explore this site" and "do NOT explore this site" in the same message.
|
|
121
|
+
`run_stdio()` now reads `CAIRN_AGENT`, `CAIRN_PROFILE` and `CAIRN_DB`, so a second agent
|
|
122
|
+
can be configured from `.mcp.json` at all — before this there was no way to.
|
|
123
|
+
`cairn_forget` now reports what it withdrew from the commons and what it cannot reach.
|
|
124
|
+
**80 tests, ruff clean.**
|
|
125
|
+
|
|
126
|
+
- **2026-09-03 (Phase 5b — Base x402 — BUILT)** — a trail you can sell. The original plan
|
|
127
|
+
assumed payment could be bolted onto the local commons; it could not, because x402 is
|
|
128
|
+
defined by an HTTP 402 exchange and the commons is two Sibyl tenants in one local file with
|
|
129
|
+
no network anywhere. So the phase grew an HTTP boundary: `cairn sell` serves this agent's
|
|
130
|
+
shared trails, `cairn buy` (and the `cairn_buy` MCP tool) pays for one. That also closes a
|
|
131
|
+
real gap — two agents could previously only share memory by sharing a database file.
|
|
132
|
+
Design rules held to: browsing the catalogue is FREE and carries no steps or locators (it
|
|
133
|
+
reuses `describe_offer`, a shape with none in it to leak); the trail is genuinely
|
|
134
|
+
unreachable without a settled payment; **the trail never goes on chain**, only the payment
|
|
135
|
+
does; and the local commons stays free, because charging your own second agent on your own
|
|
136
|
+
laptop is theatre. All x402 lives in ONE file, `payments.py`, mirroring the `store.py` rule
|
|
137
|
+
— `shop.py` goes through `payments.gate()` rather than importing the SDK, and a test walks
|
|
138
|
+
the source to keep it that way.
|
|
139
|
+
Borrowing and buying now share `_import_offer`, so a bought trail gets the same provenance,
|
|
140
|
+
the same protection over a repaired trail and the same note merging. Two import paths would
|
|
141
|
+
have drifted, and the paid one is the one nobody exercises by accident.
|
|
142
|
+
**518 engine + 98 MCP tests, ruff clean.** Four new deletion-gate tests are the ones that
|
|
143
|
+
matter: a bought trail can still be forgotten, the transaction cannot bring it back, the
|
|
144
|
+
seller's shelf empties when the seller forgets, and the buyer keeps what it paid for when
|
|
145
|
+
the seller forgets.
|
|
146
|
+
Facts were read off the INSTALLED SDK, not its docs, which were wrong twice: `ResourceConfig`
|
|
147
|
+
takes `payTo` (camelCase) while `PaymentOption` takes `pay_to`, and the ASGI middleware needs
|
|
148
|
+
the async resource server. Also found: the middleware skips settlement on any 4xx, so a
|
|
149
|
+
buyer who pays for a trail the shop does not have gets a 404 and an untouched wallet.
|
|
150
|
+
No new `events.py` types: share and borrow do not emit any either, the cold journal is the
|
|
151
|
+
record, and three event classes nothing subscribes to would be dead code.
|
|
152
|
+
**Still needed from Rohit: a funded wallet.** Everything up to the signature is verified —
|
|
153
|
+
a live shop answering a real `HTTP/1.1 402 Payment Required`, the challenge naming Base
|
|
154
|
+
Sepolia and the real USDC contract `0x036CbD…F7e`, and a purchase attempt that reached the
|
|
155
|
+
facilitator and failed only on funds.
|
|
156
|
+
|
|
157
|
+
- **2026-09-04 (Phase 6 — hardening, part one)** — an audit before touching anything, then
|
|
158
|
+
the fixes. Five real bugs, not tidying:
|
|
159
|
+
1. **`steps_repaired` had never once been true.** `executor.py` hardcoded it to 0 and
|
|
160
|
+
nothing incremented it, because a run cannot repair anything — it stops at the broken
|
|
161
|
+
step and the fix arrives as a separate call. The CLI printed "0 repaired" after every
|
|
162
|
+
run regardless, including runs of a trail that HAD been repaired, and `benchmark.py`
|
|
163
|
+
faked its own repair count by hand to make the README table read right. Replaced with
|
|
164
|
+
`trail_repairs`, which is the trail's real repair history, mentioned only when it is
|
|
165
|
+
non-zero. Two tests now hold it in place.
|
|
166
|
+
2. **Three blind `except Exception`** — including `except (PWTimeout, Exception)` in
|
|
167
|
+
`resolve()`, where the second clause swallowed the first along with any real bug in
|
|
168
|
+
`_to_playwright` and reported it as ordinary site drift. All narrowed to
|
|
169
|
+
`PlaywrightError`, so only the browser's own failures count as drift and a fault of
|
|
170
|
+
ours surfaces as itself. **`BLE` added to ruff's `select`** in both packages so this
|
|
171
|
+
cannot come back — it had already caused two incidents here.
|
|
172
|
+
3. **A machine with no browser was told its profile was broken** and invited to delete it.
|
|
173
|
+
`_is_missing_browser` existed but was only consulted on the clean-mode path, and
|
|
174
|
+
profile mode is the default. It now says `playwright install chromium` and explicitly
|
|
175
|
+
that nothing is wrong with the profile.
|
|
176
|
+
4. **The front door was Windows-only.** `.mcp.json` named `.venv/Scripts/cairn-mcp.exe`,
|
|
177
|
+
and Claude Code reads that file the moment anyone opens the repo — so a judge on a Mac
|
|
178
|
+
got a broken server before reading a word. Now it runs `mcp-server.py`, a launcher that
|
|
179
|
+
finds the venv on either layout. The README shows both `claude mcp add` commands, and
|
|
180
|
+
the demo site's busy-port help no longer prints `netstat`/`taskkill` to Linux users.
|
|
181
|
+
5. **`mcp>=1.29.1` let a fresh install pick up mcp 2.x, where `FastMCP` was renamed to
|
|
182
|
+
`MCPServer`.** The server raised ModuleNotFoundError on import and never started. This
|
|
183
|
+
venv held 1.29.1 from an earlier install, so all 616 tests passed while a stranger's
|
|
184
|
+
`pip install` was completely broken. **This is the one that would have hit every
|
|
185
|
+
judge.** Found by building the wheels and installing them into a clean Python 3.11
|
|
186
|
+
virtualenv. Pinned to `<2`; migrating to 2.x is post-deadline work.
|
|
187
|
+
|
|
188
|
+
Also: the 47 payment tests are `importorskip`-gated, so the README's Development block now
|
|
189
|
+
installs `[market]` first — otherwise "all tests passed" can be true while none of the
|
|
190
|
+
Base code ran. CI does the same.
|
|
191
|
+
|
|
192
|
+
**Packaging.** `cairn` and `cairn-mcp` are both taken on PyPI by unrelated projects, so
|
|
193
|
+
the distributions are now **`cairn-browser`** and **`cairn-browser-mcp`**. Only the
|
|
194
|
+
distribution names changed: the import package is still `cairn` and the commands are still
|
|
195
|
+
`cairn` and `cairn-mcp`. Added readmes, classifiers and project URLs so the PyPI pages are
|
|
196
|
+
not blank. All four artefacts pass `twine check`, and both wheels install and run from a
|
|
197
|
+
clean Python 3.11 venv.
|
|
198
|
+
|
|
199
|
+
**CI.** `.github/workflows/test.yml` installs from scratch on Ubuntu against Python 3.11
|
|
200
|
+
and 3.13, runs both suites with `[market]`, and checks ruff. It cannot run until Rohit
|
|
201
|
+
pushes. The lint commands were verified locally from the repo root first.
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
# Cairn as MCP tools
|
|
2
|
+
|
|
3
|
+
**Your AI can use websites. But it forgets how, every single time. Cairn makes it remember.**
|
|
4
|
+
|
|
5
|
+
This gives Claude Code, Cursor, Codex or any MCP client a browser that remembers. Your AI
|
|
6
|
+
walks a site once and Cairn writes down the route; every run after that is **one tool call,
|
|
7
|
+
no page reading, and zero model calls**. When the site changes, Cairn hands back the single
|
|
8
|
+
step that moved rather than the whole task.
|
|
9
|
+
|
|
10
|
+
```bash
|
|
11
|
+
pip install cairn-browser-mcp
|
|
12
|
+
playwright install chromium # the browser is a separate download
|
|
13
|
+
|
|
14
|
+
claude mcp add cairn -- cairn-mcp
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
Then just ask, in your own words: *"go to github.com/microsoft/playwright and tell me how
|
|
18
|
+
many open issues it has."* The first time it explores. After that it is one call.
|
|
19
|
+
|
|
20
|
+
**No API key.** There is no model call anywhere in this package or anything it imports —
|
|
21
|
+
your AI does the thinking and Cairn supplies the browser and the memory.
|
|
22
|
+
|
|
23
|
+
Optional extra, so one agent can buy a trail from another over x402 on Base:
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install "cairn-browser-mcp[market]"
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
The engine on its own is [`cairn-browser`](https://pypi.org/project/cairn-browser/).
|
|
30
|
+
|
|
31
|
+
Full documentation, the tool list, and the deletion test that proves the memory is
|
|
32
|
+
load-bearing: **https://github.com/rohit-jsfreaky/cairn**
|
|
33
|
+
|
|
34
|
+
MIT licensed.
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "cairn-browser-mcp"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Cairn as MCP tools: a browser with a memory, for Claude Code, Cursor and Codex."
|
|
9
|
+
requires-python = ">=3.11"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
readme = "README.md"
|
|
12
|
+
classifiers = [
|
|
13
|
+
"Development Status :: 4 - Beta",
|
|
14
|
+
"Intended Audience :: Developers",
|
|
15
|
+
"License :: OSI Approved :: MIT License",
|
|
16
|
+
"Programming Language :: Python :: 3.11",
|
|
17
|
+
"Programming Language :: Python :: 3.12",
|
|
18
|
+
"Programming Language :: Python :: 3.13",
|
|
19
|
+
"Topic :: Software Development :: Testing :: Acceptance",
|
|
20
|
+
"Topic :: Internet :: WWW/HTTP :: Browsers",
|
|
21
|
+
]
|
|
22
|
+
authors = [{ name = "Rohit Kashyap" }]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"cairn-browser",
|
|
25
|
+
# Capped at <2 deliberately, and it is not cosmetic. In mcp 2.x `FastMCP` was renamed
|
|
26
|
+
# to `MCPServer` and other APIs changed with it, so `from mcp.server.fastmcp import
|
|
27
|
+
# FastMCP` raises ModuleNotFoundError on import and the server never starts.
|
|
28
|
+
# This venv happened to hold 1.29.1 from an earlier install, so every test passed while
|
|
29
|
+
# a fresh `pip install` was completely broken — found by installing the built wheels
|
|
30
|
+
# into a clean virtualenv, 2026-09-04. Migrating to 2.x is post-deadline work.
|
|
31
|
+
"mcp>=1.29.1,<2",
|
|
32
|
+
]
|
|
33
|
+
|
|
34
|
+
[project.optional-dependencies]
|
|
35
|
+
dev = ["pytest>=8.0"]
|
|
36
|
+
# `cairn_buy` needs the payment code, which lives in the engine's optional extra.
|
|
37
|
+
# Declaring it here means one install is enough, instead of relying on somebody having
|
|
38
|
+
# put the engine's extra into the same virtualenv by hand.
|
|
39
|
+
market = ["cairn-browser[market]"]
|
|
40
|
+
|
|
41
|
+
[project.urls]
|
|
42
|
+
Homepage = "https://github.com/rohit-jsfreaky/cairn"
|
|
43
|
+
Repository = "https://github.com/rohit-jsfreaky/cairn"
|
|
44
|
+
Issues = "https://github.com/rohit-jsfreaky/cairn/issues"
|
|
45
|
+
|
|
46
|
+
[project.scripts]
|
|
47
|
+
cairn-mcp = "cairn_mcp.__main__:main"
|
|
48
|
+
|
|
49
|
+
[tool.hatch.build.targets.wheel]
|
|
50
|
+
packages = ["src/cairn_mcp"]
|
|
51
|
+
|
|
52
|
+
[tool.hatch.metadata]
|
|
53
|
+
allow-direct-references = true
|
|
54
|
+
|
|
55
|
+
[tool.pytest.ini_options]
|
|
56
|
+
testpaths = ["tests"]
|
|
57
|
+
|
|
58
|
+
[tool.ruff]
|
|
59
|
+
line-length = 100
|
|
60
|
+
src = ["src", "tests"]
|
|
61
|
+
|
|
62
|
+
[tool.ruff.lint]
|
|
63
|
+
# BLE catches `except Exception`. It is here because that exact shape has already
|
|
64
|
+
# caused two incidents in this project: a silent sign-out, and an error that blamed
|
|
65
|
+
# the browser profile for every possible failure. Where a blind catch is genuinely
|
|
66
|
+
# right, mark it `# noqa: BLE001` with the reason.
|
|
67
|
+
select = ["E", "F", "I", "UP", "B", "SIM", "BLE"]
|