siftscan 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
siftscan-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Baran Ayaztaş
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,125 @@
1
+ Metadata-Version: 2.4
2
+ Name: siftscan
3
+ Version: 0.1.0
4
+ Summary: Scan AI agent instruction files (CLAUDE.md, .cursorrules, AGENTS.md, mcp.json) for hidden or planted instructions.
5
+ Author: Baran Ayaztas
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/ReazGan/sift
8
+ Project-URL: Issues, https://github.com/ReazGan/sift/issues
9
+ Keywords: security,prompt-injection,supply-chain,llm,ai-agents,cursor,claude,copilot,mcp,static-analysis,cli
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Environment :: Console
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Information Technology
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Operating System :: OS Independent
16
+ Classifier: Programming Language :: Python :: 3
17
+ Classifier: Topic :: Security
18
+ Requires-Python: >=3.10
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Requires-Dist: click>=8.1
22
+ Requires-Dist: rich>=13.7
23
+ Dynamic: license-file
24
+
25
+ # sift
26
+
27
+ [![CI](https://github.com/ReazGan/sift/actions/workflows/ci.yml/badge.svg)](https://github.com/ReazGan/sift/actions/workflows/ci.yml)
28
+ [![PyPI](https://img.shields.io/pypi/v/siftscan)](https://pypi.org/project/siftscan/)
29
+
30
+ Scan AI agent instruction files for hidden or planted instructions.
31
+
32
+ Your repo now ships files that tell an assistant what to do: `CLAUDE.md`,
33
+ `.cursorrules`, `AGENTS.md`, Copilot instructions, `mcp.json`. A human skims
34
+ them in a diff; the model obeys every byte. That gap is where an attacker hides
35
+ a command, pulled in through a cloned starter, a dependency, or a pull request.
36
+ sift reads those files the way the model does and flags what a reviewer can't
37
+ see: invisible characters, instructions buried in comments, "ignore previous
38
+ instructions" payloads, data-exfiltration steps, and risky MCP configs.
39
+
40
+ Runs offline. No network calls, nothing leaves your machine.
41
+
42
+ ![sift flagging invisible smuggled text, a hidden comment and a risky MCP config](https://raw.githubusercontent.com/ReazGan/sift/main/docs/screenshot.svg)
43
+
44
+ ## Install
45
+
46
+ ```
47
+ pip install siftscan
48
+ ```
49
+
50
+ or, to keep it isolated:
51
+
52
+ ```
53
+ pipx install siftscan
54
+ ```
55
+
56
+ The command is `sift`.
57
+
58
+ ## Usage
59
+
60
+ ```
61
+ sift scan the current directory
62
+ sift path/to/repo scan a directory or a single file
63
+ sift --min high only show high and critical findings
64
+ sift --json machine-readable output
65
+ sift --quiet no output, just the exit code (for hooks/CI)
66
+ ```
67
+
68
+ Exit status is `0` when clean, `1` when there is a finding at or above the fail
69
+ level (`--fail-on`, default `high`), and `2` on error, so it drops straight
70
+ into a hook or a pipeline.
71
+
72
+ ### pre-commit
73
+
74
+ ```yaml
75
+ # .pre-commit-config.yaml
76
+ repos:
77
+ - repo: https://github.com/ReazGan/sift
78
+ rev: v0.1.0
79
+ hooks:
80
+ - id: sift
81
+ ```
82
+
83
+ ### GitHub Action
84
+
85
+ ```yaml
86
+ - uses: actions/checkout@v4
87
+ - uses: ReazGan/sift@v0.1.0
88
+ with:
89
+ fail-on: high
90
+ ```
91
+
92
+ ## What it checks
93
+
94
+ | Check | Severity | What it finds |
95
+ |-------|----------|---------------|
96
+ | `invisible-chars` | critical / high | Unicode tag characters (ASCII smuggling), variation-selector stego, zero-width and other invisible characters. Decodes and shows the hidden text. |
97
+ | `bidi-override` | critical | Bidirectional control characters (Trojan Source) that reorder how a line is displayed. |
98
+ | `unusual-encoding` | medium | An instruction file that is not plain UTF-8 (e.g. UTF-16), a way to hide a payload from UTF-8 tools. |
99
+ | `hidden-comment` | high | Instruction-like text inside an HTML comment, invisible in rendered Markdown. |
100
+ | `padded-line` | medium | Text pushed off-screen by a long run of spaces. |
101
+ | `invisible-html` | high | Text colored to blend into the background. |
102
+ | `instruction-override` | high | "Ignore previous instructions", re-role attempts, "don't tell the user" (English and Turkish). Also runs on MCP tool descriptions (tool poisoning). |
103
+ | `data-exfiltration` | critical / high | A webhook endpoint, or a secret file (`.env`, `id_rsa`, ...) named next to a "send ... to" step. |
104
+ | `mcp-auto-approve` | high | An MCP server set to approve its own tool calls. |
105
+ | `mcp-secret` | high | A credential hard-coded in an MCP config. |
106
+ | `mcp-remote` | medium / low | An MCP server reached over the network, including `npx mcp-remote <url>` bridges. |
107
+
108
+ Files scanned: `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, `.cursorrules` and
109
+ `.cursor/rules/*`, `.github/copilot-instructions.md`, `.windsurfrules`,
110
+ `.clinerules`, Qwen/Roo/Aider/IDX instruction files, and MCP configs
111
+ (`.mcp.json`, `.cursor/mcp.json`, `.vscode/mcp.json`). `node_modules`, `.git`
112
+ and build folders are skipped.
113
+
114
+ ## False positives
115
+
116
+ sift is tuned to stay quiet on real instruction files. The checks are narrow on
117
+ purpose: ordinary advice like "never commit secrets" or "always run the tests"
118
+ is not flagged, only wording that overrides, hides, or exfiltrates. If sift
119
+ flags something you wrote on purpose, it is pointing at a line worth a second
120
+ look, but you are the judge. Found a false positive? Open an issue with the
121
+ line.
122
+
123
+ ## License
124
+
125
+ MIT
@@ -0,0 +1,101 @@
1
+ # sift
2
+
3
+ [![CI](https://github.com/ReazGan/sift/actions/workflows/ci.yml/badge.svg)](https://github.com/ReazGan/sift/actions/workflows/ci.yml)
4
+ [![PyPI](https://img.shields.io/pypi/v/siftscan)](https://pypi.org/project/siftscan/)
5
+
6
+ Scan AI agent instruction files for hidden or planted instructions.
7
+
8
+ Your repo now ships files that tell an assistant what to do: `CLAUDE.md`,
9
+ `.cursorrules`, `AGENTS.md`, Copilot instructions, `mcp.json`. A human skims
10
+ them in a diff; the model obeys every byte. That gap is where an attacker hides
11
+ a command, pulled in through a cloned starter, a dependency, or a pull request.
12
+ sift reads those files the way the model does and flags what a reviewer can't
13
+ see: invisible characters, instructions buried in comments, "ignore previous
14
+ instructions" payloads, data-exfiltration steps, and risky MCP configs.
15
+
16
+ Runs offline. No network calls, nothing leaves your machine.
17
+
18
+ ![sift flagging invisible smuggled text, a hidden comment and a risky MCP config](https://raw.githubusercontent.com/ReazGan/sift/main/docs/screenshot.svg)
19
+
20
+ ## Install
21
+
22
+ ```
23
+ pip install siftscan
24
+ ```
25
+
26
+ or, to keep it isolated:
27
+
28
+ ```
29
+ pipx install siftscan
30
+ ```
31
+
32
+ The command is `sift`.
33
+
34
+ ## Usage
35
+
36
+ ```
37
+ sift scan the current directory
38
+ sift path/to/repo scan a directory or a single file
39
+ sift --min high only show high and critical findings
40
+ sift --json machine-readable output
41
+ sift --quiet no output, just the exit code (for hooks/CI)
42
+ ```
43
+
44
+ Exit status is `0` when clean, `1` when there is a finding at or above the fail
45
+ level (`--fail-on`, default `high`), and `2` on error, so it drops straight
46
+ into a hook or a pipeline.
47
+
48
+ ### pre-commit
49
+
50
+ ```yaml
51
+ # .pre-commit-config.yaml
52
+ repos:
53
+ - repo: https://github.com/ReazGan/sift
54
+ rev: v0.1.0
55
+ hooks:
56
+ - id: sift
57
+ ```
58
+
59
+ ### GitHub Action
60
+
61
+ ```yaml
62
+ - uses: actions/checkout@v4
63
+ - uses: ReazGan/sift@v0.1.0
64
+ with:
65
+ fail-on: high
66
+ ```
67
+
68
+ ## What it checks
69
+
70
+ | Check | Severity | What it finds |
71
+ |-------|----------|---------------|
72
+ | `invisible-chars` | critical / high | Unicode tag characters (ASCII smuggling), variation-selector stego, zero-width and other invisible characters. Decodes and shows the hidden text. |
73
+ | `bidi-override` | critical | Bidirectional control characters (Trojan Source) that reorder how a line is displayed. |
74
+ | `unusual-encoding` | medium | An instruction file that is not plain UTF-8 (e.g. UTF-16), a way to hide a payload from UTF-8 tools. |
75
+ | `hidden-comment` | high | Instruction-like text inside an HTML comment, invisible in rendered Markdown. |
76
+ | `padded-line` | medium | Text pushed off-screen by a long run of spaces. |
77
+ | `invisible-html` | high | Text colored to blend into the background. |
78
+ | `instruction-override` | high | "Ignore previous instructions", re-role attempts, "don't tell the user" (English and Turkish). Also runs on MCP tool descriptions (tool poisoning). |
79
+ | `data-exfiltration` | critical / high | A webhook endpoint, or a secret file (`.env`, `id_rsa`, ...) named next to a "send ... to" step. |
80
+ | `mcp-auto-approve` | high | An MCP server set to approve its own tool calls. |
81
+ | `mcp-secret` | high | A credential hard-coded in an MCP config. |
82
+ | `mcp-remote` | medium / low | An MCP server reached over the network, including `npx mcp-remote <url>` bridges. |
83
+
84
+ Files scanned: `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, `.cursorrules` and
85
+ `.cursor/rules/*`, `.github/copilot-instructions.md`, `.windsurfrules`,
86
+ `.clinerules`, Qwen/Roo/Aider/IDX instruction files, and MCP configs
87
+ (`.mcp.json`, `.cursor/mcp.json`, `.vscode/mcp.json`). `node_modules`, `.git`
88
+ and build folders are skipped.
89
+
90
+ ## False positives
91
+
92
+ sift is tuned to stay quiet on real instruction files. The checks are narrow on
93
+ purpose: ordinary advice like "never commit secrets" or "always run the tests"
94
+ is not flagged, only wording that overrides, hides, or exfiltrates. If sift
95
+ flags something you wrote on purpose, it is pointing at a line worth a second
96
+ look, but you are the judge. Found a false positive? Open an issue with the
97
+ line.
98
+
99
+ ## License
100
+
101
+ MIT
@@ -0,0 +1,40 @@
1
+ [project]
2
+ name = "siftscan"
3
+ version = "0.1.0"
4
+ description = "Scan AI agent instruction files (CLAUDE.md, .cursorrules, AGENTS.md, mcp.json) for hidden or planted instructions."
5
+ readme = "README.md"
6
+ requires-python = ">=3.10"
7
+ license = { text = "MIT" }
8
+ authors = [{ name = "Baran Ayaztas" }]
9
+ keywords = [
10
+ "security", "prompt-injection", "supply-chain", "llm", "ai-agents",
11
+ "cursor", "claude", "copilot", "mcp", "static-analysis", "cli",
12
+ ]
13
+ classifiers = [
14
+ "Development Status :: 4 - Beta",
15
+ "Environment :: Console",
16
+ "Intended Audience :: Developers",
17
+ "Intended Audience :: Information Technology",
18
+ "License :: OSI Approved :: MIT License",
19
+ "Operating System :: OS Independent",
20
+ "Programming Language :: Python :: 3",
21
+ "Topic :: Security",
22
+ ]
23
+ dependencies = [
24
+ "click>=8.1",
25
+ "rich>=13.7",
26
+ ]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/ReazGan/sift"
30
+ Issues = "https://github.com/ReazGan/sift/issues"
31
+
32
+ [project.scripts]
33
+ sift = "sift.cli:main"
34
+
35
+ [build-system]
36
+ requires = ["setuptools>=68"]
37
+ build-backend = "setuptools.build_meta"
38
+
39
+ [tool.setuptools.packages.find]
40
+ include = ["sift*"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,23 @@
1
+ """sift: find hidden instructions in AI agent instruction files."""
2
+
3
+ from __future__ import annotations
4
+
5
+ __all__ = ["scan", "__version__"]
6
+
7
+ try: # populated from package metadata once installed
8
+ from importlib.metadata import version as _version
9
+
10
+ __version__ = _version("siftscan")
11
+ except Exception: # running from a source checkout
12
+ __version__ = "0.0.0+dev"
13
+
14
+
15
+ def scan(root: str):
16
+ """Scan a path and return a list of Finding objects. Library entry point."""
17
+ from sift.checks import run_all
18
+ from sift.discover import discover
19
+
20
+ findings = []
21
+ for target in discover(root):
22
+ findings.extend(run_all(target))
23
+ return findings
@@ -0,0 +1,20 @@
1
+ """Check registry.
2
+
3
+ Every module exposes `check(ctx: ScanTarget) -> list[Finding]`. run_all runs
4
+ them over one target and returns the findings, worst severity first.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ from sift.checks import encoding, exfil, hidden, injection, invisible, mcp
10
+ from sift.checks.base import SEVERITY_ORDER, Finding, ScanTarget
11
+
12
+ _CHECKS = (invisible, encoding, hidden, injection, exfil, mcp)
13
+
14
+
15
+ def run_all(ctx: ScanTarget) -> list[Finding]:
16
+ findings: list[Finding] = []
17
+ for module in _CHECKS:
18
+ findings.extend(module.check(ctx))
19
+ findings.sort(key=lambda f: (SEVERITY_ORDER[f.severity], f.line or 0))
20
+ return findings
@@ -0,0 +1,56 @@
1
+ """Shared types for sift checks.
2
+
3
+ Each check module exposes `check(ctx) -> list[Finding]`, where ctx is a
4
+ ScanTarget (one AI-agent instruction or config file). Findings are later
5
+ serialized with `Finding.to_dict()`.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass, field
11
+ from typing import Any, Literal
12
+
13
+ Severity = Literal["low", "medium", "high", "critical"]
14
+
15
+ SEVERITY_ORDER: dict[str, int] = {"critical": 0, "high": 1, "medium": 2, "low": 3}
16
+
17
+
18
+ @dataclass
19
+ class ScanTarget:
20
+ """One file sift looks at."""
21
+
22
+ path: str # path as shown to the user, forward-slash separated
23
+ kind: str # which agent it belongs to, e.g. "claude", "cursor", "mcp"
24
+ text: str # decoded file contents
25
+ is_config: bool = False # True for JSON configs (mcp.json etc.)
26
+ encoding: str = "utf-8" # how the bytes were decoded
27
+
28
+
29
+ @dataclass
30
+ class Finding:
31
+ check: str
32
+ title: str
33
+ description: str
34
+ severity: Severity
35
+ path: str
36
+ line: int | None = None
37
+ evidence: dict[str, Any] = field(default_factory=dict)
38
+
39
+ def to_dict(self) -> dict[str, Any]:
40
+ out: dict[str, Any] = {
41
+ "check": self.check,
42
+ "title": self.title,
43
+ "severity": self.severity,
44
+ "path": self.path,
45
+ }
46
+ if self.line is not None:
47
+ out["line"] = self.line
48
+ out["description"] = self.description
49
+ if self.evidence:
50
+ out["evidence"] = self.evidence
51
+ return out
52
+
53
+
54
+ def line_of(text: str, index: int) -> int:
55
+ """1-based line number of the character at `index`."""
56
+ return text.count("\n", 0, index) + 1
@@ -0,0 +1,38 @@
1
+ """Unusual file encoding.
2
+
3
+ Instruction files are plain UTF-8 text. One saved as UTF-16, or that only
4
+ decodes with byte replacement, is a cheap way to hide a payload from a scanner
5
+ that assumes UTF-8 while the editor/agent still reads it. sift decodes it anyway
6
+ and flags the anomaly itself.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from sift.checks.base import Finding, ScanTarget
12
+
13
+ CHECK = "unusual-encoding"
14
+
15
+ _OK = {"utf-8", "ascii"}
16
+
17
+
18
+ def check(ctx: ScanTarget) -> list[Finding]:
19
+ if ctx.encoding in _OK:
20
+ return []
21
+ if ctx.encoding == "utf-8-sig":
22
+ # A BOM is common on Windows; not worth a finding on its own.
23
+ return []
24
+ return [
25
+ Finding(
26
+ check=CHECK,
27
+ title=f"Instruction file is not plain UTF-8 ({ctx.encoding})",
28
+ description=(
29
+ "This file is not plain UTF-8. Instruction files normally are, and "
30
+ "a non-UTF-8 encoding can hide text from tools that assume UTF-8 "
31
+ "while the assistant still reads it. sift decoded and scanned it "
32
+ "anyway; re-save it as UTF-8 and review the contents."
33
+ ),
34
+ severity="medium",
35
+ path=ctx.path,
36
+ evidence={"encoding": ctx.encoding},
37
+ )
38
+ ]
@@ -0,0 +1,104 @@
1
+ """Data-theft instructions planted in an instruction file.
2
+
3
+ The payoff of a backdoored instruction file is usually to make the assistant
4
+ read something secret and send it somewhere. That reads as a plausible "setup
5
+ step" to a skimming reviewer, so it is worth calling out.
6
+
7
+ - SIFT030 data-exfiltration: a webhook/collector endpoint, or a concrete secret
8
+ file (.env, id_rsa, ...) named next to an outward send verb.
9
+
10
+ Note: a bare "curl ... | bash" is NOT flagged - it is the exact shape of nearly
11
+ every legitimate installer, and flagging it would fire on most setup docs. A
12
+ hidden fetch-and-run is still caught, because to hide it an attacker has to use
13
+ the invisible-character, comment, or override tricks the other checks cover.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ import re
19
+
20
+ from sift.checks.base import Finding, ScanTarget, line_of
21
+
22
+ CHECK_EXFIL = "data-exfiltration"
23
+
24
+ # Concrete secret-bearing artifacts. Deliberately NOT the generic words
25
+ # "credentials/secrets/password/api keys/environment variables" - real
26
+ # instruction files mention those constantly as advice.
27
+ _SECRET_RE = re.compile(
28
+ r"\.env(?:\.\w+)?\b|\bid_rsa\b|\.ssh/|\.aws/credentials\b|\.npmrc\b|"
29
+ r"\.git-credentials\b|\.pem\b|\.p12\b|\bprivate\s+key\b|"
30
+ r"\bSECRET_ACCESS_KEY\b|\bprivate[_\s-]?key\b",
31
+ re.IGNORECASE,
32
+ )
33
+
34
+ # An outward send verb followed (within a short span) by "to" - "send/upload/
35
+ # post X to ...". Requiring "to" keeps ordinary "load the .env" / "fetch config"
36
+ # from matching.
37
+ _SEND_TO_RE = re.compile(
38
+ r"\b(?:send|upload|post|exfiltrat\w*|leak|ship|transmit|beacon|email)\b"
39
+ r"[^\n]{0,40}\bto\b",
40
+ re.IGNORECASE,
41
+ )
42
+
43
+ _URL_RE = re.compile(r"https?://[^\s'\"<>)\]]+", re.IGNORECASE)
44
+
45
+ _WEBHOOK_RE = re.compile(
46
+ r"https?://(?:[\w-]+\.)?(?:discord(?:app)?\.com/api/webhooks|hooks\.slack\.com|"
47
+ r"webhook\.site|requestbin|pipedream\.net|ngrok\.io|interact\.sh|oast\.\w+|"
48
+ r"burpcollaborator\.net)/?\S*",
49
+ re.IGNORECASE,
50
+ )
51
+
52
+
53
+ def _window(text: str, index: int, radius: int = 200) -> str:
54
+ return text[max(0, index - radius): index + radius]
55
+
56
+
57
+ def check(ctx: ScanTarget) -> list[Finding]:
58
+ findings: list[Finding] = []
59
+ text = ctx.text
60
+
61
+ wm = _WEBHOOK_RE.search(text)
62
+ if wm:
63
+ findings.append(
64
+ Finding(
65
+ check=CHECK_EXFIL,
66
+ title="Webhook endpoint in an instruction file",
67
+ description=(
68
+ "This file points at a webhook / request-collector URL. "
69
+ "Instruction files rarely need one; combined with any 'read X "
70
+ "and send it here' wording it is a data-theft channel. Confirm "
71
+ "why it is here."
72
+ ),
73
+ severity="critical",
74
+ path=ctx.path,
75
+ line=line_of(text, wm.start()),
76
+ evidence={"url": wm.group(0)[:160]},
77
+ )
78
+ )
79
+
80
+ secret = _SECRET_RE.search(text)
81
+ if secret and not (wm and secret.start() == wm.start()):
82
+ window = _window(text, secret.start())
83
+ if _SEND_TO_RE.search(window):
84
+ findings.append(
85
+ Finding(
86
+ check=CHECK_EXFIL,
87
+ title="Instruction to read a secret and send it out",
88
+ description=(
89
+ "A secret file is named next to an outward 'send ... to' "
90
+ "step. That is the shape of a data-theft instruction. A "
91
+ "normal instruction file does not tell the assistant to "
92
+ "ship secret files anywhere."
93
+ ),
94
+ severity="high",
95
+ path=ctx.path,
96
+ line=line_of(text, secret.start()),
97
+ evidence={
98
+ "secret": secret.group(0),
99
+ "has_external_url": bool(_URL_RE.search(window)),
100
+ },
101
+ )
102
+ )
103
+
104
+ return findings