kairoseki 0.1.2__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- kairoseki-0.1.2/.github/ISSUE_TEMPLATE/attack-scenario.yml +27 -0
- kairoseki-0.1.2/.github/ISSUE_TEMPLATE/config.yml +5 -0
- kairoseki-0.1.2/.github/ISSUE_TEMPLATE/false-positive.yml +21 -0
- kairoseki-0.1.2/.github/workflows/ci.yml +54 -0
- kairoseki-0.1.2/.github/workflows/release.yml +18 -0
- kairoseki-0.1.2/.gitignore +9 -0
- kairoseki-0.1.2/CHANGELOG.md +45 -0
- kairoseki-0.1.2/CONTRIBUTING.md +38 -0
- kairoseki-0.1.2/LICENSE +21 -0
- kairoseki-0.1.2/PKG-INFO +301 -0
- kairoseki-0.1.2/README.es.md +282 -0
- kairoseki-0.1.2/README.md +275 -0
- kairoseki-0.1.2/SECURITY.md +12 -0
- kairoseki-0.1.2/docs/assets/attack.svg +331 -0
- kairoseki-0.1.2/docs/assets/banner.svg +1 -0
- kairoseki-0.1.2/docs/assets/scan.svg +175 -0
- kairoseki-0.1.2/docs/assets/score.svg +1 -0
- kairoseki-0.1.2/docs/assets/social-preview.png +0 -0
- kairoseki-0.1.2/docs/how-it-works.md +136 -0
- kairoseki-0.1.2/examples/kairoseki.yaml +28 -0
- kairoseki-0.1.2/pyproject.toml +63 -0
- kairoseki-0.1.2/src/kairoseki/__init__.py +3 -0
- kairoseki-0.1.2/src/kairoseki/__main__.py +3 -0
- kairoseki-0.1.2/src/kairoseki/cli.py +459 -0
- kairoseki-0.1.2/src/kairoseki/client.py +144 -0
- kairoseki-0.1.2/src/kairoseki/configs.py +188 -0
- kairoseki-0.1.2/src/kairoseki/detect.py +215 -0
- kairoseki-0.1.2/src/kairoseki/engine.py +336 -0
- kairoseki-0.1.2/src/kairoseki/lab/__init__.py +1 -0
- kairoseki-0.1.2/src/kairoseki/lab/attack.py +220 -0
- kairoseki-0.1.2/src/kairoseki/lab/badge.py +30 -0
- kairoseki-0.1.2/src/kairoseki/lab/scenarios.py +386 -0
- kairoseki-0.1.2/src/kairoseki/lab/server.py +233 -0
- kairoseki-0.1.2/src/kairoseki/labels.py +479 -0
- kairoseki-0.1.2/src/kairoseki/policy.py +158 -0
- kairoseki-0.1.2/src/kairoseki/proxy.py +400 -0
- kairoseki-0.1.2/src/kairoseki/session_id.py +262 -0
- kairoseki-0.1.2/src/kairoseki/store.py +444 -0
- kairoseki-0.1.2/tests/__init__.py +0 -0
- kairoseki-0.1.2/tests/conftest.py +23 -0
- kairoseki-0.1.2/tests/servers/sdk_server.py +53 -0
- kairoseki-0.1.2/tests/test_detect.py +128 -0
- kairoseki-0.1.2/tests/test_engine.py +358 -0
- kairoseki-0.1.2/tests/test_lab.py +51 -0
- kairoseki-0.1.2/tests/test_labels.py +102 -0
- kairoseki-0.1.2/tests/test_misc.py +241 -0
- kairoseki-0.1.2/tests/test_proxy.py +129 -0
- kairoseki-0.1.2/tests/test_sdk_e2e.py +190 -0
- kairoseki-0.1.2/tests/test_session_e2e.py +90 -0
- kairoseki-0.1.2/tests/test_session_id.py +110 -0
- kairoseki-0.1.2/uv.lock +1493 -0
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
name: ⚔️ New attack scenario
|
|
2
|
+
description: Propose a real-world MCP / prompt-injection attack for the lab
|
|
3
|
+
labels: ["attack-lab", "good first issue"]
|
|
4
|
+
body:
|
|
5
|
+
- type: input
|
|
6
|
+
id: reference
|
|
7
|
+
attributes:
|
|
8
|
+
label: Reference
|
|
9
|
+
description: Link to the write-up, CVE or paper describing the attack
|
|
10
|
+
validations:
|
|
11
|
+
required: true
|
|
12
|
+
- type: textarea
|
|
13
|
+
id: flow
|
|
14
|
+
attributes:
|
|
15
|
+
label: Attack flow
|
|
16
|
+
description: Which tools does the hijacked agent call, in order, and where does the data leak?
|
|
17
|
+
placeholder: |
|
|
18
|
+
1. get_issue (untrusted content with the injection)
|
|
19
|
+
2. get_file_contents (private data)
|
|
20
|
+
3. create_pull_request (exfiltration)
|
|
21
|
+
validations:
|
|
22
|
+
required: true
|
|
23
|
+
- type: dropdown
|
|
24
|
+
id: blocked
|
|
25
|
+
attributes:
|
|
26
|
+
label: Does Kairoseki block it today?
|
|
27
|
+
options: ["Not sure", "Yes", "No - this is a bypass"]
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
name: 🧭 False positive
|
|
2
|
+
description: Kairoseki blocked or asked about something that was safe
|
|
3
|
+
labels: ["false-positive"]
|
|
4
|
+
body:
|
|
5
|
+
- type: input
|
|
6
|
+
id: server
|
|
7
|
+
attributes:
|
|
8
|
+
label: MCP server and tool
|
|
9
|
+
placeholder: "@modelcontextprotocol/server-github / create_pull_request"
|
|
10
|
+
validations:
|
|
11
|
+
required: true
|
|
12
|
+
- type: textarea
|
|
13
|
+
id: log
|
|
14
|
+
attributes:
|
|
15
|
+
label: Output of `kairoseki log -n 20`
|
|
16
|
+
description: Please remove anything private
|
|
17
|
+
render: text
|
|
18
|
+
- type: textarea
|
|
19
|
+
id: expected
|
|
20
|
+
attributes:
|
|
21
|
+
label: What did you expect?
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
lint:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: astral-sh/setup-uv@v6
|
|
17
|
+
- run: uv sync --group dev
|
|
18
|
+
- run: uv run ruff check src tests
|
|
19
|
+
- run: uv run ruff format --check src tests
|
|
20
|
+
- run: uv run mypy
|
|
21
|
+
|
|
22
|
+
test:
|
|
23
|
+
name: test (${{ matrix.os }}, py${{ matrix.python }}, ${{ matrix.mcp }})
|
|
24
|
+
runs-on: ${{ matrix.os }}
|
|
25
|
+
strategy:
|
|
26
|
+
fail-fast: false
|
|
27
|
+
matrix:
|
|
28
|
+
os: [ubuntu-latest, macos-latest, windows-latest]
|
|
29
|
+
python: ["3.10", "3.13"]
|
|
30
|
+
mcp: ["mcp<2", "mcp>=2"]
|
|
31
|
+
exclude:
|
|
32
|
+
- python: "3.10"
|
|
33
|
+
mcp: "mcp>=2"
|
|
34
|
+
steps:
|
|
35
|
+
- uses: actions/checkout@v4
|
|
36
|
+
- uses: astral-sh/setup-uv@v6
|
|
37
|
+
with:
|
|
38
|
+
python-version: ${{ matrix.python }}
|
|
39
|
+
- run: uv venv
|
|
40
|
+
- run: uv pip install -e . "${{ matrix.mcp }}" pytest pytest-asyncio
|
|
41
|
+
- run: uv run --no-sync pytest -q
|
|
42
|
+
|
|
43
|
+
attack-lab:
|
|
44
|
+
runs-on: ubuntu-latest
|
|
45
|
+
steps:
|
|
46
|
+
- uses: actions/checkout@v4
|
|
47
|
+
- uses: astral-sh/setup-uv@v6
|
|
48
|
+
- run: uv sync
|
|
49
|
+
- name: Every real-world attack must be blocked
|
|
50
|
+
run: uv run kairoseki attack --badge score.svg
|
|
51
|
+
- uses: actions/upload-artifact@v4
|
|
52
|
+
with:
|
|
53
|
+
name: kairoseki-score
|
|
54
|
+
path: score.svg
|
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
name: Release
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags: ["v*"]
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
pypi:
|
|
9
|
+
runs-on: ubuntu-latest
|
|
10
|
+
environment: pypi
|
|
11
|
+
permissions:
|
|
12
|
+
id-token: write # PyPI trusted publishing, no API token stored in the repo
|
|
13
|
+
contents: read
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: astral-sh/setup-uv@v6
|
|
17
|
+
- run: uv build
|
|
18
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.2
|
|
4
|
+
|
|
5
|
+
* Fix false positives from fragment fingerprints: vendor prefixes shared by every key (`sk-ant-api03-` is exactly
|
|
6
|
+
12 characters, `github_pat_`, `sk-proj-`, `xoxb-`...) are no longer fingerprinted as fragments, so another key of
|
|
7
|
+
the same vendor, or docs that mention the prefix, are not mistaken for a leak. Fragments of the random part are
|
|
8
|
+
still caught.
|
|
9
|
+
|
|
10
|
+
## 0.1.1
|
|
11
|
+
|
|
12
|
+
Fixes and improvements from real-world testing on Windows.
|
|
13
|
+
|
|
14
|
+
* **Cross-server taint now works on every OS.** The session is keyed by the MCP client process, found by walking
|
|
15
|
+
the process tree past Kairoseki's own launchers (`kairoseki.exe`, the venv `python.exe` redirector, `uv`/`uvx`,
|
|
16
|
+
re-exec'd framework Pythons on macOS). Before, `uv tool` installs on Windows gave every server its own session,
|
|
17
|
+
so the lethal-trifecta rule could not fire across servers.
|
|
18
|
+
* New `kairoseki session --explain` shows the process-tree walk.
|
|
19
|
+
* `scan` and `wrap` now see Claude Code *local*-scope servers (`~/.claude.json` → `projects[...].mcpServers`),
|
|
20
|
+
with Windows path normalization. `--all-projects` covers every project.
|
|
21
|
+
* `kairoseki status` lists protected vs unprotected servers, shows live sessions first and hides ended ones
|
|
22
|
+
(`--all` to show them). `wrap` ends with the same summary.
|
|
23
|
+
* Secrets split into pieces of 12+ characters are recognized (fragment fingerprints), even when a policy `allow:`
|
|
24
|
+
turns the trifecta rule off for that tool.
|
|
25
|
+
* Tool labels ignore a server-name prefix (`codegraph_search` → `search`) and know code-intelligence objects
|
|
26
|
+
(symbols, callers, definitions...).
|
|
27
|
+
* Secret scanning is one linear pass with no length limit: ~200x faster, and padding an argument past the old
|
|
28
|
+
200k-character cut-off no longer hides a secret or private data.
|
|
29
|
+
* No more false positives on `PATH`/`PATHEXT`/`PSModulePath`/`PWD`, drive-letter paths, or tool confirmations like
|
|
30
|
+
"Email sent to bob@acme.com". npm tokens are detected.
|
|
31
|
+
* Windows: lock-file and file-replace races between concurrent proxies are handled.
|
|
32
|
+
* The injection warning is separated from the tool output by a blank line.
|
|
33
|
+
* Docs: Windows upgrade note (close the client first), `server-filesystem` roots, and the limits of fingerprints.
|
|
34
|
+
|
|
35
|
+
## 0.1.0
|
|
36
|
+
|
|
37
|
+
First release.
|
|
38
|
+
|
|
39
|
+
* stdio MCP proxy (`kairoseki run`) for both protocol eras (handshake and 2026-07-28)
|
|
40
|
+
* Lethal-trifecta rule with taint shared across servers in one session
|
|
41
|
+
* Private-data-flow rule against disguised exfiltration tools
|
|
42
|
+
* Secret fingerprints with encoding-aware exfiltration detection, plus redaction of secrets and optional PII
|
|
43
|
+
* Poisoned tool detection (injection text, ANSI escapes, invisible Unicode) and rug-pull pinning
|
|
44
|
+
* Approvals via MCP elicitation, `input_required` (SEP-2322), or `kairoseki approve`
|
|
45
|
+
* `kairoseki scan`, `wrap`, `attack` (9 real-world attacks, 7 everyday tasks), `log`, `status`, `pins`, `init`
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
Thanks for helping make agents harder to hijack! 🪨
|
|
4
|
+
|
|
5
|
+
## Setup
|
|
6
|
+
|
|
7
|
+
```bash
|
|
8
|
+
git clone https://github.com/AnthonyRiveraI/kairoseki
|
|
9
|
+
cd kairoseki
|
|
10
|
+
uv sync --group dev
|
|
11
|
+
uv run pytest
|
|
12
|
+
uv run ruff check src tests && uv run mypy
|
|
13
|
+
```
|
|
14
|
+
|
|
15
|
+
## Add an attack scenario (the best first contribution)
|
|
16
|
+
|
|
17
|
+
1. Find a real, public write-up of an MCP or prompt-injection attack.
|
|
18
|
+
2. Add a `Scenario` to `src/kairoseki/lab/scenarios.py`:
|
|
19
|
+
* `world`: the content the lab servers will serve (issues, pages, files, inbox, notes, shell outputs)
|
|
20
|
+
* `steps`: the calls a fully hijacked agent would make. Use `{out:N}`, `{grep:N:REGEX}` or `{b64grep:N:REGEX}`
|
|
21
|
+
to reuse what the agent saw at step N.
|
|
22
|
+
* `canary` and `exfil_points`: what must not reach which tool
|
|
23
|
+
* `reference`: credit the original researchers
|
|
24
|
+
3. If you need a new kind of server, add a role in `src/kairoseki/lab/server.py`.
|
|
25
|
+
4. Run `uv run kairoseki attack --only <your-id>`. The baseline (without Kairoseki) **must leak**, otherwise the
|
|
26
|
+
scenario proves nothing. `tests/test_lab.py` enforces this.
|
|
27
|
+
|
|
28
|
+
If your scenario leaks *with* Kairoseki, you found a bypass. Please report it privately first (see `SECURITY.md`).
|
|
29
|
+
|
|
30
|
+
## Fix a false positive
|
|
31
|
+
|
|
32
|
+
Add the tool name to `tests/test_labels.py` with the labels you expect, then adjust `src/kairoseki/labels.py`.
|
|
33
|
+
|
|
34
|
+
## Pull requests
|
|
35
|
+
|
|
36
|
+
* Keep PRs focused, with tests.
|
|
37
|
+
* `ruff`, `mypy --strict` and `pytest` must pass. CI runs Linux, macOS and Windows, Python 3.10 and 3.13,
|
|
38
|
+
and MCP SDK 1.x and 2.x.
|
kairoseki-0.1.2/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Anthony Rivera
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
kairoseki-0.1.2/PKG-INFO
ADDED
|
@@ -0,0 +1,301 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: kairoseki
|
|
3
|
+
Version: 0.1.2
|
|
4
|
+
Summary: Seastone for your AI agents: an MCP firewall that breaks the lethal trifecta with taint tracking.
|
|
5
|
+
Project-URL: Homepage, https://github.com/AnthonyRiveraI/kairoseki
|
|
6
|
+
Project-URL: Issues, https://github.com/AnthonyRiveraI/kairoseki/issues
|
|
7
|
+
Author-email: Anthony Rivera <anthony.g.rivera.i@gmail.com>
|
|
8
|
+
License-Expression: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: ai-agents,claude,cursor,firewall,llm,mcp,model-context-protocol,prompt-injection,security
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Environment :: Console
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
20
|
+
Classifier: Topic :: Security
|
|
21
|
+
Classifier: Topic :: Software Development :: Libraries
|
|
22
|
+
Requires-Python: >=3.10
|
|
23
|
+
Requires-Dist: pyyaml>=6.0
|
|
24
|
+
Requires-Dist: rich>=13.0
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
<p align="center">
|
|
28
|
+
<img src="docs/assets/banner.svg" width="100%" alt="Kairoseki - seastone for your AI agents" />
|
|
29
|
+
</p>
|
|
30
|
+
|
|
31
|
+
<p align="center"><b>English</b> · <a href="README.es.md">Español</a></p>
|
|
32
|
+
|
|
33
|
+
<p align="center">
|
|
34
|
+
<a href="https://github.com/AnthonyRiveraI/kairoseki/actions/workflows/ci.yml"><img src="https://github.com/AnthonyRiveraI/kairoseki/actions/workflows/ci.yml/badge.svg" alt="CI" /></a>
|
|
35
|
+
<img src="https://img.shields.io/badge/version-0.1.2-0b1026?labelColor=0b1026" alt="Version 0.1.2" />
|
|
36
|
+
<img src="https://img.shields.io/badge/python-3.10%20%E2%86%92%203.13-0b1026?labelColor=0b1026" alt="Python" />
|
|
37
|
+
<img src="https://img.shields.io/badge/MCP-2024--11%20%E2%86%92%202026--07-0b1026?labelColor=0b1026" alt="MCP versions" />
|
|
38
|
+
<img src="docs/assets/score.svg" alt="Kairoseki score" />
|
|
39
|
+
</p>
|
|
40
|
+
|
|
41
|
+
<p align="center">
|
|
42
|
+
<b>An MCP firewall that stops prompt-injection data theft by tracking <i>where data came from</i>,<br/>
|
|
43
|
+
not by guessing what attacks look like.</b>
|
|
44
|
+
</p>
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
In One Piece, **kairoseki** (seastone) cancels Devil Fruit powers. Kairoseki does the same for your agent's
|
|
49
|
+
most dangerous power: reading something an attacker wrote, and then quietly sending your data somewhere.
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
uv tool install git+https://github.com/AnthonyRiveraI/kairoseki
|
|
53
|
+
kairoseki scan # what can a single prompt injection do with your MCP setup?
|
|
54
|
+
kairoseki wrap # put every MCP server in your Claude / Cursor / VS Code config behind Kairoseki
|
|
55
|
+
kairoseki attack # replay 9 real-world attacks against your setup and get a grade
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
<p align="center"><img src="docs/assets/attack.svg" width="92%" alt="kairoseki attack: 9 of 9 real-world attacks blocked, 7 of 7 normal tasks uninterrupted" /></p>
|
|
59
|
+
|
|
60
|
+
## Why
|
|
61
|
+
|
|
62
|
+
An agent is exploitable **by design** when it has all three legs of the
|
|
63
|
+
[lethal trifecta](https://simonwillison.net/2025/Jun/16/the-lethal-trifecta/):
|
|
64
|
+
|
|
65
|
+
1. **access to private data** (your files, repos, inbox),
|
|
66
|
+
2. **exposure to untrusted content** (a web page, a GitHub issue, an email), and
|
|
67
|
+
3. **a way to send data out** (HTTP, email, a comment, a pull request).
|
|
68
|
+
|
|
69
|
+
Plug a filesystem server and a fetch server into Claude Code, Cursor or Claude Desktop and you have all three.
|
|
70
|
+
This keeps happening in the real world:
|
|
71
|
+
|
|
72
|
+
| Incident | What happened |
|
|
73
|
+
| :-- | :-- |
|
|
74
|
+
| [GitHub MCP exploit](https://invariantlabs.ai/blog/mcp-github-vulnerability) (May 2025) | A malicious public issue made an agent copy private repo data into a public pull request |
|
|
75
|
+
| [Tool poisoning](https://invariantlabs.ai/blog/mcp-security-notification-tool-poisoning-attacks) (Apr 2025) | Hidden instructions in a tool *description* stole `~/.cursor/mcp.json` |
|
|
76
|
+
| [MCPoison, CVE-2025-54136](https://nvd.nist.gov/vuln/detail/CVE-2025-54136) (Jul 2025) | An approved MCP config was silently swapped later (rug pull) |
|
|
77
|
+
| [Comment and Control](https://oddguan.com/blog/comment-and-control-prompt-injection-credential-theft-claude-code-gemini-cli-github-copilot/) (Apr 2026) | Injections in PR titles and comments made Claude Code, Gemini CLI and Copilot agents leak their own secrets |
|
|
78
|
+
|
|
79
|
+
Most defenses scan text for attack patterns, and attackers just rephrase. Kairoseki breaks the trifecta instead.
|
|
80
|
+
The idea is inspired by [CaMeL](https://arxiv.org/abs/2503.18813) (Google DeepMind): track taint across the
|
|
81
|
+
whole session and step in exactly when untrusted content, private data and an outbound channel meet.
|
|
82
|
+
|
|
83
|
+
## What it does
|
|
84
|
+
|
|
85
|
+
| | |
|
|
86
|
+
| :-- | :-- |
|
|
87
|
+
| 🔗 **Session taint across servers** | Every `kairoseki run` in one agent session shares state. The web page comes from the `fetch` server, the secret from `filesystem`, the leak goes through `github`: Kairoseki still sees one chain. |
|
|
88
|
+
| 🧪 **Secret fingerprints** | Secrets seen in tool output or in a server's environment are fingerprinted (never stored). If one shows up in a later tool call, even base64, hex, URL-encoded or reversed, the call is denied. |
|
|
89
|
+
| 🫥 **Redaction** | API keys, tokens and private keys are replaced with `[REDACTED:kind]` before they reach the model. The model can't leak what it never saw. |
|
|
90
|
+
| ☠️ **Poisoned tool detection** | Tool descriptions and schemas with injection text, ANSI escapes or invisible Unicode are neutralized before the model reads them, and the tool is blocked. |
|
|
91
|
+
| 📌 **Rug-pull pins** | Tool definitions are pinned on first use. If a server changes one later, that tool is blocked until you re-approve it. |
|
|
92
|
+
| ✋ **Approvals that fit your client** | When the trifecta closes, Kairoseki asks *you*: an in-client prompt (MCP elicitation, in both protocol eras), or a one-time `kairoseki approve K-1A2B3C` from any terminal. |
|
|
93
|
+
| ⚔️ **Attack lab and badge** | `kairoseki attack` replays real attacks against a *fully hijacked* agent and checks, on the attacker's side, whether the canary leaked. |
|
|
94
|
+
| 🔍 **Scanner** | `kairoseki scan` labels every tool in your config and tells you if you already have the lethal trifecta. |
|
|
95
|
+
|
|
96
|
+
It is a transparent stdio proxy that speaks raw JSON-RPC, so it works with any MCP server and client, in
|
|
97
|
+
both the handshake era (`initialize`, 2024-11-05 → 2025-11-25) and the modern era (`server/discover`,
|
|
98
|
+
2026-07-28). Two dependencies: `pyyaml` and `rich`.
|
|
99
|
+
|
|
100
|
+
<p align="center"><img src="docs/assets/scan.svg" width="80%" alt="kairoseki scan finds a poisoned tool and the lethal trifecta in a real config" /></p>
|
|
101
|
+
|
|
102
|
+
## Quickstart
|
|
103
|
+
|
|
104
|
+
### 0. Prerequisites
|
|
105
|
+
|
|
106
|
+
| You need | Why | How to get it |
|
|
107
|
+
| :-- | :-- | :-- |
|
|
108
|
+
| **[uv](https://docs.astral.sh/uv/)** (recommended) or **pipx** | Installs Kairoseki as an isolated command-line tool | macOS / Linux: `curl -LsSf https://astral.sh/uv/install.sh \| sh`<br/>Windows (PowerShell): `powershell -ExecutionPolicy ByPass -c "irm https://astral.sh/uv/install.ps1 \| iex"` |
|
|
109
|
+
| **Python 3.10+** | Kairoseki is written in Python | uv downloads a suitable Python automatically if you don't have one. With pipx, install Python yourself. |
|
|
110
|
+
| **Git** | To install straight from GitHub | [git-scm.com](https://git-scm.com/downloads) |
|
|
111
|
+
| **An MCP client** | Something to protect | Claude Code, Claude Desktop, Cursor, VS Code, Windsurf... |
|
|
112
|
+
| **Node.js** *(optional)* | Only if your MCP servers start with `npx` | [nodejs.org](https://nodejs.org/) |
|
|
113
|
+
|
|
114
|
+
### 1. Install
|
|
115
|
+
|
|
116
|
+
Kairoseki is not on PyPI yet, so install it from GitHub:
|
|
117
|
+
|
|
118
|
+
```bash
|
|
119
|
+
uv tool install git+https://github.com/AnthonyRiveraI/kairoseki
|
|
120
|
+
# or with pipx:
|
|
121
|
+
pipx install git+https://github.com/AnthonyRiveraI/kairoseki
|
|
122
|
+
```
|
|
123
|
+
|
|
124
|
+
Then make sure the `kairoseki` command is on your `PATH`, and open a **new** terminal:
|
|
125
|
+
|
|
126
|
+
```bash
|
|
127
|
+
uv tool update-shell # or: pipx ensurepath
|
|
128
|
+
kairoseki --version # should print: kairoseki 0.1.2
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
> **`kairoseki: command not found`, or your MCP client can't start it?** The tool lives in `~/.local/bin`
|
|
132
|
+
> (`%USERPROFILE%\.local\bin` on Windows). Run `uv tool update-shell`, then fully restart your terminal **and**
|
|
133
|
+
> your MCP client so they pick up the new `PATH`. `kairoseki wrap` also writes the absolute path into your config,
|
|
134
|
+
> which avoids the problem entirely.
|
|
135
|
+
|
|
136
|
+
To update later: `uv tool upgrade kairoseki` (or `pipx upgrade kairoseki`).
|
|
137
|
+
|
|
138
|
+
> **Windows:** close your MCP client (or disable its Kairoseki-wrapped servers) before upgrading. While a client is
|
|
139
|
+
> running `kairoseki.exe`, Windows locks the file and `uv tool upgrade` fails with *os error 32*.
|
|
140
|
+
|
|
141
|
+
### 2. See your exposure
|
|
142
|
+
|
|
143
|
+
```bash
|
|
144
|
+
kairoseki scan
|
|
145
|
+
```
|
|
146
|
+
|
|
147
|
+
It reads the MCP configs of Claude Desktop, Claude Code, Cursor, Windsurf and VS Code, starts each server, and
|
|
148
|
+
labels every tool as `private`, `untrusted`, `sink` and/or `destructive`. For Claude Code that includes project
|
|
149
|
+
servers (`.mcp.json`), user servers and the current project's *local* servers in `~/.claude.json`; add
|
|
150
|
+
`--all-projects` to include every project's local servers.
|
|
151
|
+
|
|
152
|
+
### 3. Wrap your servers
|
|
153
|
+
|
|
154
|
+
```bash
|
|
155
|
+
kairoseki wrap # all detected configs (a .kairoseki.bak backup is written first)
|
|
156
|
+
kairoseki wrap --undo # restore
|
|
157
|
+
kairoseki status # which servers are protected, and what each live session has seen
|
|
158
|
+
```
|
|
159
|
+
|
|
160
|
+
Or wrap a single server by hand. Every client uses the same pattern, `kairoseki run --name <name> -- <original command>`:
|
|
161
|
+
|
|
162
|
+
```jsonc
|
|
163
|
+
{
|
|
164
|
+
"mcpServers": {
|
|
165
|
+
"fetch": {
|
|
166
|
+
"command": "kairoseki",
|
|
167
|
+
"args": ["run", "--name", "fetch", "--", "uvx", "mcp-server-fetch"]
|
|
168
|
+
}
|
|
169
|
+
}
|
|
170
|
+
}
|
|
171
|
+
```
|
|
172
|
+
|
|
173
|
+
With Claude Code:
|
|
174
|
+
|
|
175
|
+
```bash
|
|
176
|
+
claude mcp add fetch -- kairoseki run --name fetch -- uvx mcp-server-fetch
|
|
177
|
+
```
|
|
178
|
+
|
|
179
|
+
Restart your client and check that the server connects (with Claude Code: `claude mcp get fetch`). That's it.
|
|
180
|
+
|
|
181
|
+
> **Testing with `@modelcontextprotocol/server-filesystem`?** It replaces the directories you pass on its command
|
|
182
|
+
> line with the client's *roots*. Claude Code sends the project directory, so the server will serve that folder,
|
|
183
|
+
> not the one in your config.
|
|
184
|
+
|
|
185
|
+
### 4. Try to break it
|
|
186
|
+
|
|
187
|
+
```bash
|
|
188
|
+
kairoseki attack # grade your current policy
|
|
189
|
+
kairoseki attack --badge kairoseki.svg # and get a README badge
|
|
190
|
+
```
|
|
191
|
+
|
|
192
|
+
## How decisions are made
|
|
193
|
+
|
|
194
|
+
```mermaid
|
|
195
|
+
flowchart LR
|
|
196
|
+
A[tool call] --> B{denied by policy,<br/>poisoned or rug-pulled?}
|
|
197
|
+
B -- yes --> X[⛔ deny]
|
|
198
|
+
B -- no --> C{arguments contain a<br/>secret seen this session?}
|
|
199
|
+
C -- yes --> X
|
|
200
|
+
C -- no --> D{session saw untrusted content<br/>AND private data<br/>AND this tool is a sink?}
|
|
201
|
+
D -- yes --> Q[✋ ask the user]
|
|
202
|
+
D -- no --> E{untrusted content seen AND private data<br/>flows into an unclassified tool?}
|
|
203
|
+
E -- yes --> Q
|
|
204
|
+
E -- no --> OK[✅ forward to the server]
|
|
205
|
+
OK --> R[result: sanitize, redact,<br/>fingerprint secrets, update taint]
|
|
206
|
+
```
|
|
207
|
+
|
|
208
|
+
| Mode | Behaviour |
|
|
209
|
+
| :-- | :-- |
|
|
210
|
+
| `monitor` | Never blocks. Logs what would have happened (redaction still applies). Good for your first week. |
|
|
211
|
+
| `balanced` *(default)* | Denies exfiltration, poisoned tools and rug pulls. Asks before the lethal trifecta closes. |
|
|
212
|
+
| `strict` | Also asks before **any** sink or destructive tool once untrusted content entered the session. |
|
|
213
|
+
|
|
214
|
+
When Kairoseki asks:
|
|
215
|
+
|
|
216
|
+
* **In-client prompt.** If your client supports MCP elicitation, you get an "Allow this call once?" form.
|
|
217
|
+
Kairoseki sends `elicitation/create` in the handshake era and an `input_required` result (SEP-2322) in the 2026-07-28 era.
|
|
218
|
+
* **Terminal.** Otherwise the agent gets a clear refusal with an id. Run `kairoseki approve K-1A2B3C`, then ask the
|
|
219
|
+
agent to retry. Approvals are single-use, bound to the exact arguments, and expire after 10 minutes.
|
|
220
|
+
|
|
221
|
+
Learn more in [docs/how-it-works.md](docs/how-it-works.md).
|
|
222
|
+
|
|
223
|
+
## Policy
|
|
224
|
+
|
|
225
|
+
```bash
|
|
226
|
+
kairoseki init # writes a commented kairoseki.yaml
|
|
227
|
+
```
|
|
228
|
+
|
|
229
|
+
```yaml
|
|
230
|
+
mode: balanced
|
|
231
|
+
redact:
|
|
232
|
+
secrets: true
|
|
233
|
+
pii: false
|
|
234
|
+
servers:
|
|
235
|
+
github:
|
|
236
|
+
tools:
|
|
237
|
+
create_or_update_file: [sink, destructive] # explicit labels replace the heuristics
|
|
238
|
+
allow: [search_repositories] # never ask (redaction still applies)
|
|
239
|
+
deny: [delete_repository] # always block
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
Kairoseki looks for `--policy`, then `$KAIROSEKI_POLICY`, then `./kairoseki.yaml`, then `~/.kairoseki/kairoseki.yaml`.
|
|
243
|
+
|
|
244
|
+
## Other commands
|
|
245
|
+
|
|
246
|
+
```bash
|
|
247
|
+
kairoseki log # recent decisions, redactions and detections
|
|
248
|
+
kairoseki status # protected vs unprotected servers, and live sessions (--all for ended ones)
|
|
249
|
+
kairoseki session --explain # which session this process joins, and why
|
|
250
|
+
kairoseki approve # list pending approvals
|
|
251
|
+
kairoseki pins list # servers whose tools changed since you pinned them
|
|
252
|
+
kairoseki pins approve github
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
## How it compares
|
|
256
|
+
|
|
257
|
+
| | Kairoseki | Pattern scanners / guardrail models | [mcp-context-protector](https://github.com/trailofbits/mcp-context-protector) | Enterprise MCP gateways |
|
|
258
|
+
| :-- | :--: | :--: | :--: | :--: |
|
|
259
|
+
| Blocks rephrased or novel injections (data-flow based) | ✅ | ❌ | ❌ | ➖ |
|
|
260
|
+
| Taint shared across servers in one session | ✅ | ❌ | ❌ | ➖ |
|
|
261
|
+
| Detects encoded secret exfiltration | ✅ | ➖ | ❌ | ➖ |
|
|
262
|
+
| Rug-pull pinning | ✅ | ❌ | ✅ | ✅ |
|
|
263
|
+
| Poisoned descriptions, ANSI, invisible Unicode | ✅ | ✅ | ✅ | ➖ |
|
|
264
|
+
| Runs locally, no service or API key | ✅ | ➖ | ✅ | ❌ |
|
|
265
|
+
| Measurable attack lab with a grade | ✅ | ❌ | ❌ | ❌ |
|
|
266
|
+
|
|
267
|
+
➖ = depends on the product. Kairoseki borrows trust-on-first-use pinning and ANSI sanitization from Trail of Bits'
|
|
268
|
+
excellent mcp-context-protector. The two are complementary.
|
|
269
|
+
|
|
270
|
+
## Limitations (please read)
|
|
271
|
+
|
|
272
|
+
* **It only sees MCP traffic.** Built-in client tools (for example Claude Code's own `Bash` or `WebFetch`) never go
|
|
273
|
+
through MCP. Pair Kairoseki with your client's permission rules.
|
|
274
|
+
* **Labels are heuristics.** Tool names and descriptions are read as verb + object (`get_issue`, `send_email`).
|
|
275
|
+
They can be wrong, and the [policy](#policy) lets you fix them. Server annotations can only *add* risk, never remove it.
|
|
276
|
+
* **Taint is per session and coarse on purpose.** Once untrusted content is in the context, Kairoseki assumes it may
|
|
277
|
+
have influenced everything after it. That is what makes it robust, and it is why `strict` mode asks more often.
|
|
278
|
+
* **Secret fingerprints are exact matching.** They catch a secret copied whole, base64/hex/URL-encoded, reversed, or
|
|
279
|
+
split into pieces of 12+ characters, but not one interleaved character by character or run through a custom
|
|
280
|
+
cipher. The lethal-trifecta rule is the safety net that does not need to recognize the data, so be careful with
|
|
281
|
+
`monitor` mode and policy `allow:` entries, which turn it off.
|
|
282
|
+
* **stdio servers only** in v0.1. Streamable HTTP servers are on the roadmap.
|
|
283
|
+
* **It is not a sandbox.** A malicious server binary can still do anything your user account can. Kairoseki protects
|
|
284
|
+
against malicious *content*, not malicious *code*.
|
|
285
|
+
|
|
286
|
+
## Roadmap
|
|
287
|
+
|
|
288
|
+
- [ ] Publish on PyPI (`pipx install kairoseki`)
|
|
289
|
+
- [ ] Streamable HTTP transport
|
|
290
|
+
- [ ] 🐌 Den Den Mushi: Kairoseki *calls your phone* to approve risky actions
|
|
291
|
+
- [ ] OpenTelemetry export of decisions
|
|
292
|
+
- [ ] More attack scenarios. [Propose one!](https://github.com/AnthonyRiveraI/kairoseki/issues/new?template=attack-scenario.yml)
|
|
293
|
+
|
|
294
|
+
## Contributing
|
|
295
|
+
|
|
296
|
+
The most valuable contribution is a new **attack scenario**: a real write-up turned into a replayable test.
|
|
297
|
+
See [CONTRIBUTING.md](CONTRIBUTING.md). Found a bypass? Please [report it privately](SECURITY.md).
|
|
298
|
+
|
|
299
|
+
## License
|
|
300
|
+
|
|
301
|
+
MIT © Anthony Rivera
|