siftscan 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- siftscan-0.1.0/LICENSE +21 -0
- siftscan-0.1.0/PKG-INFO +125 -0
- siftscan-0.1.0/README.md +101 -0
- siftscan-0.1.0/pyproject.toml +40 -0
- siftscan-0.1.0/setup.cfg +4 -0
- siftscan-0.1.0/sift/__init__.py +23 -0
- siftscan-0.1.0/sift/checks/__init__.py +20 -0
- siftscan-0.1.0/sift/checks/base.py +56 -0
- siftscan-0.1.0/sift/checks/encoding.py +38 -0
- siftscan-0.1.0/sift/checks/exfil.py +104 -0
- siftscan-0.1.0/sift/checks/hidden.py +147 -0
- siftscan-0.1.0/sift/checks/injection.py +85 -0
- siftscan-0.1.0/sift/checks/invisible.py +200 -0
- siftscan-0.1.0/sift/checks/mcp.py +254 -0
- siftscan-0.1.0/sift/cli.py +185 -0
- siftscan-0.1.0/sift/discover.py +163 -0
- siftscan-0.1.0/siftscan.egg-info/PKG-INFO +125 -0
- siftscan-0.1.0/siftscan.egg-info/SOURCES.txt +27 -0
- siftscan-0.1.0/siftscan.egg-info/dependency_links.txt +1 -0
- siftscan-0.1.0/siftscan.egg-info/entry_points.txt +2 -0
- siftscan-0.1.0/siftscan.egg-info/requires.txt +2 -0
- siftscan-0.1.0/siftscan.egg-info/top_level.txt +1 -0
- siftscan-0.1.0/tests/test_cli.py +73 -0
- siftscan-0.1.0/tests/test_discover.py +76 -0
- siftscan-0.1.0/tests/test_exfil.py +44 -0
- siftscan-0.1.0/tests/test_hidden.py +48 -0
- siftscan-0.1.0/tests/test_injection.py +32 -0
- siftscan-0.1.0/tests/test_invisible.py +60 -0
- siftscan-0.1.0/tests/test_mcp.py +78 -0
siftscan-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Baran Ayaztaş
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
siftscan-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: siftscan
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Scan AI agent instruction files (CLAUDE.md, .cursorrules, AGENTS.md, mcp.json) for hidden or planted instructions.
|
|
5
|
+
Author: Baran Ayaztas
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/ReazGan/sift
|
|
8
|
+
Project-URL: Issues, https://github.com/ReazGan/sift/issues
|
|
9
|
+
Keywords: security,prompt-injection,supply-chain,llm,ai-agents,cursor,claude,copilot,mcp,static-analysis,cli
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Environment :: Console
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Information Technology
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Operating System :: OS Independent
|
|
16
|
+
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Topic :: Security
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Requires-Dist: click>=8.1
|
|
22
|
+
Requires-Dist: rich>=13.7
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# sift
|
|
26
|
+
|
|
27
|
+
[](https://github.com/ReazGan/sift/actions/workflows/ci.yml)
|
|
28
|
+
[](https://pypi.org/project/siftscan/)
|
|
29
|
+
|
|
30
|
+
Scan AI agent instruction files for hidden or planted instructions.
|
|
31
|
+
|
|
32
|
+
Your repo now ships files that tell an assistant what to do: `CLAUDE.md`,
|
|
33
|
+
`.cursorrules`, `AGENTS.md`, Copilot instructions, `mcp.json`. A human skims
|
|
34
|
+
them in a diff; the model obeys every byte. That gap is where an attacker hides
|
|
35
|
+
a command, pulled in through a cloned starter, a dependency, or a pull request.
|
|
36
|
+
sift reads those files the way the model does and flags what a reviewer can't
|
|
37
|
+
see: invisible characters, instructions buried in comments, "ignore previous
|
|
38
|
+
instructions" payloads, data-exfiltration steps, and risky MCP configs.
|
|
39
|
+
|
|
40
|
+
Runs offline. No network calls, nothing leaves your machine.
|
|
41
|
+
|
|
42
|
+

|
|
43
|
+
|
|
44
|
+
## Install
|
|
45
|
+
|
|
46
|
+
```
|
|
47
|
+
pip install siftscan
|
|
48
|
+
```
|
|
49
|
+
|
|
50
|
+
or, to keep it isolated:
|
|
51
|
+
|
|
52
|
+
```
|
|
53
|
+
pipx install siftscan
|
|
54
|
+
```
|
|
55
|
+
|
|
56
|
+
The command is `sift`.
|
|
57
|
+
|
|
58
|
+
## Usage
|
|
59
|
+
|
|
60
|
+
```
|
|
61
|
+
sift scan the current directory
|
|
62
|
+
sift path/to/repo scan a directory or a single file
|
|
63
|
+
sift --min high only show high and critical findings
|
|
64
|
+
sift --json machine-readable output
|
|
65
|
+
sift --quiet no output, just the exit code (for hooks/CI)
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
Exit status is `0` when clean, `1` when there is a finding at or above the fail
|
|
69
|
+
level (`--fail-on`, default `high`), and `2` on error, so it drops straight
|
|
70
|
+
into a hook or a pipeline.
|
|
71
|
+
|
|
72
|
+
### pre-commit
|
|
73
|
+
|
|
74
|
+
```yaml
|
|
75
|
+
# .pre-commit-config.yaml
|
|
76
|
+
repos:
|
|
77
|
+
- repo: https://github.com/ReazGan/sift
|
|
78
|
+
rev: v0.1.0
|
|
79
|
+
hooks:
|
|
80
|
+
- id: sift
|
|
81
|
+
```
|
|
82
|
+
|
|
83
|
+
### GitHub Action
|
|
84
|
+
|
|
85
|
+
```yaml
|
|
86
|
+
- uses: actions/checkout@v4
|
|
87
|
+
- uses: ReazGan/sift@v0.1.0
|
|
88
|
+
with:
|
|
89
|
+
fail-on: high
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
## What it checks
|
|
93
|
+
|
|
94
|
+
| Check | Severity | What it finds |
|
|
95
|
+
|-------|----------|---------------|
|
|
96
|
+
| `invisible-chars` | critical / high | Unicode tag characters (ASCII smuggling), variation-selector stego, zero-width and other invisible characters. Decodes and shows the hidden text. |
|
|
97
|
+
| `bidi-override` | critical | Bidirectional control characters (Trojan Source) that reorder how a line is displayed. |
|
|
98
|
+
| `unusual-encoding` | medium | An instruction file that is not plain UTF-8 (e.g. UTF-16), a way to hide a payload from UTF-8 tools. |
|
|
99
|
+
| `hidden-comment` | high | Instruction-like text inside an HTML comment, invisible in rendered Markdown. |
|
|
100
|
+
| `padded-line` | medium | Text pushed off-screen by a long run of spaces. |
|
|
101
|
+
| `invisible-html` | high | Text colored to blend into the background. |
|
|
102
|
+
| `instruction-override` | high | "Ignore previous instructions", re-role attempts, "don't tell the user" (English and Turkish). Also runs on MCP tool descriptions (tool poisoning). |
|
|
103
|
+
| `data-exfiltration` | critical / high | A webhook endpoint, or a secret file (`.env`, `id_rsa`, ...) named next to a "send ... to" step. |
|
|
104
|
+
| `mcp-auto-approve` | high | An MCP server set to approve its own tool calls. |
|
|
105
|
+
| `mcp-secret` | high | A credential hard-coded in an MCP config. |
|
|
106
|
+
| `mcp-remote` | medium / low | An MCP server reached over the network, including `npx mcp-remote <url>` bridges. |
|
|
107
|
+
|
|
108
|
+
Files scanned: `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, `.cursorrules` and
|
|
109
|
+
`.cursor/rules/*`, `.github/copilot-instructions.md`, `.windsurfrules`,
|
|
110
|
+
`.clinerules`, Qwen/Roo/Aider/IDX instruction files, and MCP configs
|
|
111
|
+
(`.mcp.json`, `.cursor/mcp.json`, `.vscode/mcp.json`). `node_modules`, `.git`
|
|
112
|
+
and build folders are skipped.
|
|
113
|
+
|
|
114
|
+
## False positives
|
|
115
|
+
|
|
116
|
+
sift is tuned to stay quiet on real instruction files. The checks are narrow on
|
|
117
|
+
purpose: ordinary advice like "never commit secrets" or "always run the tests"
|
|
118
|
+
is not flagged, only wording that overrides, hides, or exfiltrates. If sift
|
|
119
|
+
flags something you wrote on purpose, it is pointing at a line worth a second
|
|
120
|
+
look, but you are the judge. Found a false positive? Open an issue with the
|
|
121
|
+
line.
|
|
122
|
+
|
|
123
|
+
## License
|
|
124
|
+
|
|
125
|
+
MIT
|
siftscan-0.1.0/README.md
ADDED
|
@@ -0,0 +1,101 @@
|
|
|
1
|
+
# sift
|
|
2
|
+
|
|
3
|
+
[](https://github.com/ReazGan/sift/actions/workflows/ci.yml)
|
|
4
|
+
[](https://pypi.org/project/siftscan/)
|
|
5
|
+
|
|
6
|
+
Scan AI agent instruction files for hidden or planted instructions.
|
|
7
|
+
|
|
8
|
+
Your repo now ships files that tell an assistant what to do: `CLAUDE.md`,
|
|
9
|
+
`.cursorrules`, `AGENTS.md`, Copilot instructions, `mcp.json`. A human skims
|
|
10
|
+
them in a diff; the model obeys every byte. That gap is where an attacker hides
|
|
11
|
+
a command, pulled in through a cloned starter, a dependency, or a pull request.
|
|
12
|
+
sift reads those files the way the model does and flags what a reviewer can't
|
|
13
|
+
see: invisible characters, instructions buried in comments, "ignore previous
|
|
14
|
+
instructions" payloads, data-exfiltration steps, and risky MCP configs.
|
|
15
|
+
|
|
16
|
+
Runs offline. No network calls, nothing leaves your machine.
|
|
17
|
+
|
|
18
|
+

|
|
19
|
+
|
|
20
|
+
## Install
|
|
21
|
+
|
|
22
|
+
```
|
|
23
|
+
pip install siftscan
|
|
24
|
+
```
|
|
25
|
+
|
|
26
|
+
or, to keep it isolated:
|
|
27
|
+
|
|
28
|
+
```
|
|
29
|
+
pipx install siftscan
|
|
30
|
+
```
|
|
31
|
+
|
|
32
|
+
The command is `sift`.
|
|
33
|
+
|
|
34
|
+
## Usage
|
|
35
|
+
|
|
36
|
+
```
|
|
37
|
+
sift scan the current directory
|
|
38
|
+
sift path/to/repo scan a directory or a single file
|
|
39
|
+
sift --min high only show high and critical findings
|
|
40
|
+
sift --json machine-readable output
|
|
41
|
+
sift --quiet no output, just the exit code (for hooks/CI)
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
Exit status is `0` when clean, `1` when there is a finding at or above the fail
|
|
45
|
+
level (`--fail-on`, default `high`), and `2` on error, so it drops straight
|
|
46
|
+
into a hook or a pipeline.
|
|
47
|
+
|
|
48
|
+
### pre-commit
|
|
49
|
+
|
|
50
|
+
```yaml
|
|
51
|
+
# .pre-commit-config.yaml
|
|
52
|
+
repos:
|
|
53
|
+
- repo: https://github.com/ReazGan/sift
|
|
54
|
+
rev: v0.1.0
|
|
55
|
+
hooks:
|
|
56
|
+
- id: sift
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
### GitHub Action
|
|
60
|
+
|
|
61
|
+
```yaml
|
|
62
|
+
- uses: actions/checkout@v4
|
|
63
|
+
- uses: ReazGan/sift@v0.1.0
|
|
64
|
+
with:
|
|
65
|
+
fail-on: high
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
## What it checks
|
|
69
|
+
|
|
70
|
+
| Check | Severity | What it finds |
|
|
71
|
+
|-------|----------|---------------|
|
|
72
|
+
| `invisible-chars` | critical / high | Unicode tag characters (ASCII smuggling), variation-selector stego, zero-width and other invisible characters. Decodes and shows the hidden text. |
|
|
73
|
+
| `bidi-override` | critical | Bidirectional control characters (Trojan Source) that reorder how a line is displayed. |
|
|
74
|
+
| `unusual-encoding` | medium | An instruction file that is not plain UTF-8 (e.g. UTF-16), a way to hide a payload from UTF-8 tools. |
|
|
75
|
+
| `hidden-comment` | high | Instruction-like text inside an HTML comment, invisible in rendered Markdown. |
|
|
76
|
+
| `padded-line` | medium | Text pushed off-screen by a long run of spaces. |
|
|
77
|
+
| `invisible-html` | high | Text colored to blend into the background. |
|
|
78
|
+
| `instruction-override` | high | "Ignore previous instructions", re-role attempts, "don't tell the user" (English and Turkish). Also runs on MCP tool descriptions (tool poisoning). |
|
|
79
|
+
| `data-exfiltration` | critical / high | A webhook endpoint, or a secret file (`.env`, `id_rsa`, ...) named next to a "send ... to" step. |
|
|
80
|
+
| `mcp-auto-approve` | high | An MCP server set to approve its own tool calls. |
|
|
81
|
+
| `mcp-secret` | high | A credential hard-coded in an MCP config. |
|
|
82
|
+
| `mcp-remote` | medium / low | An MCP server reached over the network, including `npx mcp-remote <url>` bridges. |
|
|
83
|
+
|
|
84
|
+
Files scanned: `CLAUDE.md`, `AGENTS.md`, `GEMINI.md`, `.cursorrules` and
|
|
85
|
+
`.cursor/rules/*`, `.github/copilot-instructions.md`, `.windsurfrules`,
|
|
86
|
+
`.clinerules`, Qwen/Roo/Aider/IDX instruction files, and MCP configs
|
|
87
|
+
(`.mcp.json`, `.cursor/mcp.json`, `.vscode/mcp.json`). `node_modules`, `.git`
|
|
88
|
+
and build folders are skipped.
|
|
89
|
+
|
|
90
|
+
## False positives
|
|
91
|
+
|
|
92
|
+
sift is tuned to stay quiet on real instruction files. The checks are narrow on
|
|
93
|
+
purpose: ordinary advice like "never commit secrets" or "always run the tests"
|
|
94
|
+
is not flagged, only wording that overrides, hides, or exfiltrates. If sift
|
|
95
|
+
flags something you wrote on purpose, it is pointing at a line worth a second
|
|
96
|
+
look, but you are the judge. Found a false positive? Open an issue with the
|
|
97
|
+
line.
|
|
98
|
+
|
|
99
|
+
## License
|
|
100
|
+
|
|
101
|
+
MIT
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "siftscan"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Scan AI agent instruction files (CLAUDE.md, .cursorrules, AGENTS.md, mcp.json) for hidden or planted instructions."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
requires-python = ">=3.10"
|
|
7
|
+
license = { text = "MIT" }
|
|
8
|
+
authors = [{ name = "Baran Ayaztas" }]
|
|
9
|
+
keywords = [
|
|
10
|
+
"security", "prompt-injection", "supply-chain", "llm", "ai-agents",
|
|
11
|
+
"cursor", "claude", "copilot", "mcp", "static-analysis", "cli",
|
|
12
|
+
]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Environment :: Console",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"Intended Audience :: Information Technology",
|
|
18
|
+
"License :: OSI Approved :: MIT License",
|
|
19
|
+
"Operating System :: OS Independent",
|
|
20
|
+
"Programming Language :: Python :: 3",
|
|
21
|
+
"Topic :: Security",
|
|
22
|
+
]
|
|
23
|
+
dependencies = [
|
|
24
|
+
"click>=8.1",
|
|
25
|
+
"rich>=13.7",
|
|
26
|
+
]
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://github.com/ReazGan/sift"
|
|
30
|
+
Issues = "https://github.com/ReazGan/sift/issues"
|
|
31
|
+
|
|
32
|
+
[project.scripts]
|
|
33
|
+
sift = "sift.cli:main"
|
|
34
|
+
|
|
35
|
+
[build-system]
|
|
36
|
+
requires = ["setuptools>=68"]
|
|
37
|
+
build-backend = "setuptools.build_meta"
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.packages.find]
|
|
40
|
+
include = ["sift*"]
|
siftscan-0.1.0/setup.cfg
ADDED
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""sift: find hidden instructions in AI agent instruction files."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
__all__ = ["scan", "__version__"]
|
|
6
|
+
|
|
7
|
+
try: # populated from package metadata once installed
|
|
8
|
+
from importlib.metadata import version as _version
|
|
9
|
+
|
|
10
|
+
__version__ = _version("siftscan")
|
|
11
|
+
except Exception: # running from a source checkout
|
|
12
|
+
__version__ = "0.0.0+dev"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def scan(root: str):
|
|
16
|
+
"""Scan a path and return a list of Finding objects. Library entry point."""
|
|
17
|
+
from sift.checks import run_all
|
|
18
|
+
from sift.discover import discover
|
|
19
|
+
|
|
20
|
+
findings = []
|
|
21
|
+
for target in discover(root):
|
|
22
|
+
findings.extend(run_all(target))
|
|
23
|
+
return findings
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
"""Check registry.
|
|
2
|
+
|
|
3
|
+
Every module exposes `check(ctx: ScanTarget) -> list[Finding]`. run_all runs
|
|
4
|
+
them over one target and returns the findings, worst severity first.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
from sift.checks import encoding, exfil, hidden, injection, invisible, mcp
|
|
10
|
+
from sift.checks.base import SEVERITY_ORDER, Finding, ScanTarget
|
|
11
|
+
|
|
12
|
+
_CHECKS = (invisible, encoding, hidden, injection, exfil, mcp)
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def run_all(ctx: ScanTarget) -> list[Finding]:
|
|
16
|
+
findings: list[Finding] = []
|
|
17
|
+
for module in _CHECKS:
|
|
18
|
+
findings.extend(module.check(ctx))
|
|
19
|
+
findings.sort(key=lambda f: (SEVERITY_ORDER[f.severity], f.line or 0))
|
|
20
|
+
return findings
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
"""Shared types for sift checks.
|
|
2
|
+
|
|
3
|
+
Each check module exposes `check(ctx) -> list[Finding]`, where ctx is a
|
|
4
|
+
ScanTarget (one AI-agent instruction or config file). Findings are later
|
|
5
|
+
serialized with `Finding.to_dict()`.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass, field
|
|
11
|
+
from typing import Any, Literal
|
|
12
|
+
|
|
13
|
+
Severity = Literal["low", "medium", "high", "critical"]
|
|
14
|
+
|
|
15
|
+
SEVERITY_ORDER: dict[str, int] = {"critical": 0, "high": 1, "medium": 2, "low": 3}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
@dataclass
|
|
19
|
+
class ScanTarget:
|
|
20
|
+
"""One file sift looks at."""
|
|
21
|
+
|
|
22
|
+
path: str # path as shown to the user, forward-slash separated
|
|
23
|
+
kind: str # which agent it belongs to, e.g. "claude", "cursor", "mcp"
|
|
24
|
+
text: str # decoded file contents
|
|
25
|
+
is_config: bool = False # True for JSON configs (mcp.json etc.)
|
|
26
|
+
encoding: str = "utf-8" # how the bytes were decoded
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Finding:
|
|
31
|
+
check: str
|
|
32
|
+
title: str
|
|
33
|
+
description: str
|
|
34
|
+
severity: Severity
|
|
35
|
+
path: str
|
|
36
|
+
line: int | None = None
|
|
37
|
+
evidence: dict[str, Any] = field(default_factory=dict)
|
|
38
|
+
|
|
39
|
+
def to_dict(self) -> dict[str, Any]:
|
|
40
|
+
out: dict[str, Any] = {
|
|
41
|
+
"check": self.check,
|
|
42
|
+
"title": self.title,
|
|
43
|
+
"severity": self.severity,
|
|
44
|
+
"path": self.path,
|
|
45
|
+
}
|
|
46
|
+
if self.line is not None:
|
|
47
|
+
out["line"] = self.line
|
|
48
|
+
out["description"] = self.description
|
|
49
|
+
if self.evidence:
|
|
50
|
+
out["evidence"] = self.evidence
|
|
51
|
+
return out
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def line_of(text: str, index: int) -> int:
|
|
55
|
+
"""1-based line number of the character at `index`."""
|
|
56
|
+
return text.count("\n", 0, index) + 1
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Unusual file encoding.
|
|
2
|
+
|
|
3
|
+
Instruction files are plain UTF-8 text. One saved as UTF-16, or that only
|
|
4
|
+
decodes with byte replacement, is a cheap way to hide a payload from a scanner
|
|
5
|
+
that assumes UTF-8 while the editor/agent still reads it. sift decodes it anyway
|
|
6
|
+
and flags the anomaly itself.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from sift.checks.base import Finding, ScanTarget
|
|
12
|
+
|
|
13
|
+
CHECK = "unusual-encoding"
|
|
14
|
+
|
|
15
|
+
_OK = {"utf-8", "ascii"}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def check(ctx: ScanTarget) -> list[Finding]:
|
|
19
|
+
if ctx.encoding in _OK:
|
|
20
|
+
return []
|
|
21
|
+
if ctx.encoding == "utf-8-sig":
|
|
22
|
+
# A BOM is common on Windows; not worth a finding on its own.
|
|
23
|
+
return []
|
|
24
|
+
return [
|
|
25
|
+
Finding(
|
|
26
|
+
check=CHECK,
|
|
27
|
+
title=f"Instruction file is not plain UTF-8 ({ctx.encoding})",
|
|
28
|
+
description=(
|
|
29
|
+
"This file is not plain UTF-8. Instruction files normally are, and "
|
|
30
|
+
"a non-UTF-8 encoding can hide text from tools that assume UTF-8 "
|
|
31
|
+
"while the assistant still reads it. sift decoded and scanned it "
|
|
32
|
+
"anyway; re-save it as UTF-8 and review the contents."
|
|
33
|
+
),
|
|
34
|
+
severity="medium",
|
|
35
|
+
path=ctx.path,
|
|
36
|
+
evidence={"encoding": ctx.encoding},
|
|
37
|
+
)
|
|
38
|
+
]
|
|
@@ -0,0 +1,104 @@
|
|
|
1
|
+
"""Data-theft instructions planted in an instruction file.
|
|
2
|
+
|
|
3
|
+
The payoff of a backdoored instruction file is usually to make the assistant
|
|
4
|
+
read something secret and send it somewhere. That reads as a plausible "setup
|
|
5
|
+
step" to a skimming reviewer, so it is worth calling out.
|
|
6
|
+
|
|
7
|
+
- SIFT030 data-exfiltration: a webhook/collector endpoint, or a concrete secret
|
|
8
|
+
file (.env, id_rsa, ...) named next to an outward send verb.
|
|
9
|
+
|
|
10
|
+
Note: a bare "curl ... | bash" is NOT flagged - it is the exact shape of nearly
|
|
11
|
+
every legitimate installer, and flagging it would fire on most setup docs. A
|
|
12
|
+
hidden fetch-and-run is still caught, because to hide it an attacker has to use
|
|
13
|
+
the invisible-character, comment, or override tricks the other checks cover.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
import re
|
|
19
|
+
|
|
20
|
+
from sift.checks.base import Finding, ScanTarget, line_of
|
|
21
|
+
|
|
22
|
+
CHECK_EXFIL = "data-exfiltration"
|
|
23
|
+
|
|
24
|
+
# Concrete secret-bearing artifacts. Deliberately NOT the generic words
|
|
25
|
+
# "credentials/secrets/password/api keys/environment variables" - real
|
|
26
|
+
# instruction files mention those constantly as advice.
|
|
27
|
+
_SECRET_RE = re.compile(
|
|
28
|
+
r"\.env(?:\.\w+)?\b|\bid_rsa\b|\.ssh/|\.aws/credentials\b|\.npmrc\b|"
|
|
29
|
+
r"\.git-credentials\b|\.pem\b|\.p12\b|\bprivate\s+key\b|"
|
|
30
|
+
r"\bSECRET_ACCESS_KEY\b|\bprivate[_\s-]?key\b",
|
|
31
|
+
re.IGNORECASE,
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
# An outward send verb followed (within a short span) by "to" - "send/upload/
|
|
35
|
+
# post X to ...". Requiring "to" keeps ordinary "load the .env" / "fetch config"
|
|
36
|
+
# from matching.
|
|
37
|
+
_SEND_TO_RE = re.compile(
|
|
38
|
+
r"\b(?:send|upload|post|exfiltrat\w*|leak|ship|transmit|beacon|email)\b"
|
|
39
|
+
r"[^\n]{0,40}\bto\b",
|
|
40
|
+
re.IGNORECASE,
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
_URL_RE = re.compile(r"https?://[^\s'\"<>)\]]+", re.IGNORECASE)
|
|
44
|
+
|
|
45
|
+
_WEBHOOK_RE = re.compile(
|
|
46
|
+
r"https?://(?:[\w-]+\.)?(?:discord(?:app)?\.com/api/webhooks|hooks\.slack\.com|"
|
|
47
|
+
r"webhook\.site|requestbin|pipedream\.net|ngrok\.io|interact\.sh|oast\.\w+|"
|
|
48
|
+
r"burpcollaborator\.net)/?\S*",
|
|
49
|
+
re.IGNORECASE,
|
|
50
|
+
)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _window(text: str, index: int, radius: int = 200) -> str:
|
|
54
|
+
return text[max(0, index - radius): index + radius]
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def check(ctx: ScanTarget) -> list[Finding]:
|
|
58
|
+
findings: list[Finding] = []
|
|
59
|
+
text = ctx.text
|
|
60
|
+
|
|
61
|
+
wm = _WEBHOOK_RE.search(text)
|
|
62
|
+
if wm:
|
|
63
|
+
findings.append(
|
|
64
|
+
Finding(
|
|
65
|
+
check=CHECK_EXFIL,
|
|
66
|
+
title="Webhook endpoint in an instruction file",
|
|
67
|
+
description=(
|
|
68
|
+
"This file points at a webhook / request-collector URL. "
|
|
69
|
+
"Instruction files rarely need one; combined with any 'read X "
|
|
70
|
+
"and send it here' wording it is a data-theft channel. Confirm "
|
|
71
|
+
"why it is here."
|
|
72
|
+
),
|
|
73
|
+
severity="critical",
|
|
74
|
+
path=ctx.path,
|
|
75
|
+
line=line_of(text, wm.start()),
|
|
76
|
+
evidence={"url": wm.group(0)[:160]},
|
|
77
|
+
)
|
|
78
|
+
)
|
|
79
|
+
|
|
80
|
+
secret = _SECRET_RE.search(text)
|
|
81
|
+
if secret and not (wm and secret.start() == wm.start()):
|
|
82
|
+
window = _window(text, secret.start())
|
|
83
|
+
if _SEND_TO_RE.search(window):
|
|
84
|
+
findings.append(
|
|
85
|
+
Finding(
|
|
86
|
+
check=CHECK_EXFIL,
|
|
87
|
+
title="Instruction to read a secret and send it out",
|
|
88
|
+
description=(
|
|
89
|
+
"A secret file is named next to an outward 'send ... to' "
|
|
90
|
+
"step. That is the shape of a data-theft instruction. A "
|
|
91
|
+
"normal instruction file does not tell the assistant to "
|
|
92
|
+
"ship secret files anywhere."
|
|
93
|
+
),
|
|
94
|
+
severity="high",
|
|
95
|
+
path=ctx.path,
|
|
96
|
+
line=line_of(text, secret.start()),
|
|
97
|
+
evidence={
|
|
98
|
+
"secret": secret.group(0),
|
|
99
|
+
"has_external_url": bool(_URL_RE.search(window)),
|
|
100
|
+
},
|
|
101
|
+
)
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
return findings
|