antipluto 1.0.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- antipluto-1.0.0/LICENSE +17 -0
- antipluto-1.0.0/PKG-INFO +256 -0
- antipluto-1.0.0/README.md +210 -0
- antipluto-1.0.0/antipluto/__init__.py +34 -0
- antipluto-1.0.0/antipluto/__main__.py +1232 -0
- antipluto-1.0.0/antipluto/api/__init__.py +32 -0
- antipluto-1.0.0/antipluto/api/app.py +239 -0
- antipluto-1.0.0/antipluto/api/parser.py +89 -0
- antipluto-1.0.0/antipluto/api/pipeline.py +102 -0
- antipluto-1.0.0/antipluto/api/schemas.py +73 -0
- antipluto-1.0.0/antipluto/api/service.py +218 -0
- antipluto-1.0.0/antipluto/baseline/__init__.py +39 -0
- antipluto-1.0.0/antipluto/baseline/clients.py +237 -0
- antipluto-1.0.0/antipluto/baseline/evaluator.py +351 -0
- antipluto-1.0.0/antipluto/baseline/mime_envelope.py +78 -0
- antipluto-1.0.0/antipluto/baseline/unmasked_loader.py +174 -0
- antipluto-1.0.0/antipluto/classifier/__init__.py +76 -0
- antipluto-1.0.0/antipluto/classifier/benchmark.py +376 -0
- antipluto-1.0.0/antipluto/classifier/data.py +214 -0
- antipluto-1.0.0/antipluto/classifier/evaluator.py +178 -0
- antipluto-1.0.0/antipluto/classifier/exporter.py +207 -0
- antipluto-1.0.0/antipluto/classifier/heads/__init__.py +54 -0
- antipluto-1.0.0/antipluto/classifier/heads/base.py +84 -0
- antipluto-1.0.0/antipluto/classifier/heads/cnn_head.py +373 -0
- antipluto-1.0.0/antipluto/classifier/heads/mlp_head.py +355 -0
- antipluto-1.0.0/antipluto/classifier/heads/rf_head.py +159 -0
- antipluto-1.0.0/antipluto/classifier/heads/xgboost_head.py +239 -0
- antipluto-1.0.0/antipluto/classifier/semantic.py +286 -0
- antipluto-1.0.0/antipluto/classifier/stylometric.py +136 -0
- antipluto-1.0.0/antipluto/classifier/trainer.py +247 -0
- antipluto-1.0.0/antipluto/configs/__init__.py +1 -0
- antipluto-1.0.0/antipluto/configs/default.example.yaml +288 -0
- antipluto-1.0.0/antipluto/generation/__init__.py +25 -0
- antipluto-1.0.0/antipluto/generation/client.py +674 -0
- antipluto-1.0.0/antipluto/generation/generator.py +588 -0
- antipluto-1.0.0/antipluto/generation/prompts.py +285 -0
- antipluto-1.0.0/antipluto/generation/provenance.py +249 -0
- antipluto-1.0.0/antipluto/masking/__init__.py +1 -0
- antipluto-1.0.0/antipluto/masking/masker.py +668 -0
- antipluto-1.0.0/antipluto/preprocessing/__init__.py +9 -0
- antipluto-1.0.0/antipluto/preprocessing/cleaners/__init__.py +1 -0
- antipluto-1.0.0/antipluto/preprocessing/cleaners/feature_extractors.py +194 -0
- antipluto-1.0.0/antipluto/preprocessing/cleaners/header_slicer.py +303 -0
- antipluto-1.0.0/antipluto/preprocessing/cleaners/html_stripper.py +168 -0
- antipluto-1.0.0/antipluto/preprocessing/cleaners/unicode_normalizer.py +172 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/__init__.py +1 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/base.py +361 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/enron.py +247 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/nazario.py +270 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/nigerian_fraud.py +174 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/spamassassin.py +294 -0
- antipluto-1.0.0/antipluto/preprocessing/parsers/trec.py +327 -0
- antipluto-1.0.0/antipluto/preprocessing/pipeline.py +380 -0
- antipluto-1.0.0/antipluto/preprocessing/validators/__init__.py +1 -0
- antipluto-1.0.0/antipluto/preprocessing/validators/schema.py +167 -0
- antipluto-1.0.0/antipluto/utils/__init__.py +1 -0
- antipluto-1.0.0/antipluto/utils/io.py +155 -0
- antipluto-1.0.0/antipluto.egg-info/PKG-INFO +256 -0
- antipluto-1.0.0/antipluto.egg-info/SOURCES.txt +63 -0
- antipluto-1.0.0/antipluto.egg-info/dependency_links.txt +1 -0
- antipluto-1.0.0/antipluto.egg-info/entry_points.txt +2 -0
- antipluto-1.0.0/antipluto.egg-info/requires.txt +34 -0
- antipluto-1.0.0/antipluto.egg-info/top_level.txt +1 -0
- antipluto-1.0.0/pyproject.toml +66 -0
- antipluto-1.0.0/setup.cfg +4 -0
antipluto-1.0.0/LICENSE
ADDED
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
Creative Commons Attribution-NonCommercial 4.0 International Public License
|
|
2
|
+
|
|
3
|
+
By exercising the Licensed Rights (defined below), You accept and agree to be bound by the terms and conditions of this Creative Commons Attribution-NonCommercial 4.0 International Public License ("Public License"). To the extent this Public License may be interpreted as a contract, You are granted the Licensed Rights in consideration of Your acceptance of these terms and conditions, and the Licensor grants You such rights in consideration of benefits the Licensor receives from making the Licensed Material available under these terms and conditions.
|
|
4
|
+
|
|
5
|
+
For the full legal text of this license, please visit:
|
|
6
|
+
https://creativecommons.org/licenses/by-nc/4.0/legalcode
|
|
7
|
+
|
|
8
|
+
Summary of the License:
|
|
9
|
+
You are free to:
|
|
10
|
+
- Share: copy and redistribute the material in any medium or format
|
|
11
|
+
- Adapt: remix, transform, and build upon the material
|
|
12
|
+
|
|
13
|
+
Under the following terms:
|
|
14
|
+
- Attribution: You must give appropriate credit, provide a link to the license, and indicate if changes were made. You may do so in any reasonable manner, but not in any way that suggests the licensor endorses you or your use.
|
|
15
|
+
- NonCommercial: You may not use the material for commercial purposes.
|
|
16
|
+
|
|
17
|
+
No additional restrictions: You may not apply legal terms or technological measures that legally restrict others from doing anything the license permits.
|
antipluto-1.0.0/PKG-INFO
ADDED
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: antipluto
|
|
3
|
+
Version: 1.0.0
|
|
4
|
+
Summary: A unified framework for email preprocessing, PII masking, LLM cohort generation, and dual-branch semantic-stylometric phishing classification.
|
|
5
|
+
License: CC BY-NC 4.0
|
|
6
|
+
Classifier: Programming Language :: Python :: 3
|
|
7
|
+
Classifier: License :: Free for non-commercial use
|
|
8
|
+
Classifier: Operating System :: OS Independent
|
|
9
|
+
Classifier: Intended Audience :: Science/Research
|
|
10
|
+
Classifier: Topic :: Security
|
|
11
|
+
Requires-Python: >=3.10
|
|
12
|
+
Description-Content-Type: text/markdown
|
|
13
|
+
License-File: LICENSE
|
|
14
|
+
Requires-Dist: click>=8.0
|
|
15
|
+
Requires-Dist: pydantic>=2.0
|
|
16
|
+
Requires-Dist: PyYAML>=6.0
|
|
17
|
+
Requires-Dist: beautifulsoup4>=4.12
|
|
18
|
+
Requires-Dist: spacy>=3.7
|
|
19
|
+
Requires-Dist: langdetect>=1.0.9
|
|
20
|
+
Requires-Dist: openai>=1.0
|
|
21
|
+
Requires-Dist: anthropic>=0.30
|
|
22
|
+
Requires-Dist: pandas>=2.0
|
|
23
|
+
Provides-Extra: classifier
|
|
24
|
+
Requires-Dist: scikit-learn>=1.4; extra == "classifier"
|
|
25
|
+
Requires-Dist: scipy>=1.12; extra == "classifier"
|
|
26
|
+
Requires-Dist: xgboost>=2.0; extra == "classifier"
|
|
27
|
+
Requires-Dist: accelerate>=0.30; extra == "classifier"
|
|
28
|
+
Requires-Dist: datasets>=2.0; extra == "classifier"
|
|
29
|
+
Requires-Dist: sentencepiece>=0.2.0; extra == "classifier"
|
|
30
|
+
Requires-Dist: tiktoken>=0.7.0; extra == "classifier"
|
|
31
|
+
Requires-Dist: onnx>=1.16; extra == "classifier"
|
|
32
|
+
Requires-Dist: onnxruntime>=1.18; extra == "classifier"
|
|
33
|
+
Requires-Dist: onnxscript>=0.1.0; extra == "classifier"
|
|
34
|
+
Requires-Dist: skl2onnx>=1.17; extra == "classifier"
|
|
35
|
+
Requires-Dist: onnxmltools>=1.12; extra == "classifier"
|
|
36
|
+
Requires-Dist: optimum>=1.20; extra == "classifier"
|
|
37
|
+
Requires-Dist: matplotlib>=3.8; extra == "classifier"
|
|
38
|
+
Provides-Extra: api
|
|
39
|
+
Requires-Dist: fastapi>=0.110.0; extra == "api"
|
|
40
|
+
Requires-Dist: uvicorn[standard]>=0.28.0; extra == "api"
|
|
41
|
+
Requires-Dist: python-multipart>=0.0.9; extra == "api"
|
|
42
|
+
Provides-Extra: dev
|
|
43
|
+
Requires-Dist: dvc[gdrive]; extra == "dev"
|
|
44
|
+
Requires-Dist: spacy-transformers>=1.3; extra == "dev"
|
|
45
|
+
Dynamic: license-file
|
|
46
|
+
|
|
47
|
+
<div align="center">
|
|
48
|
+
<img src="antipluto.svg" alt="Anti-PLUTO Logo" width="120" />
|
|
49
|
+
<h1>Anti-PLUTO</h1>
|
|
50
|
+
<p><strong>Anti-Phishing Lexical Utilities and Threat Observation</strong></p>
|
|
51
|
+
<p>
|
|
52
|
+
<em>A unified framework for email dataset preprocessing, PII masking, synthetic data generation, and semantic-stylometric phishing classification.</em>
|
|
53
|
+
</p>
|
|
54
|
+
<p>
|
|
55
|
+
<a href="https://github.com/mmaarij/antipluto/blob/main/LICENSE">
|
|
56
|
+
<img src="https://img.shields.io/badge/License-CC%20BY--NC%204.0-mistyrose.svg" alt="License: CC BY-NC 4.0" />
|
|
57
|
+
</a>
|
|
58
|
+
<a href="https://python.org">
|
|
59
|
+
<img src="https://img.shields.io/badge/Python-3.10%2B-lightblue" alt="Python 3.10+" />
|
|
60
|
+
</a>
|
|
61
|
+
</p>
|
|
62
|
+
</div>
|
|
63
|
+
|
|
64
|
+
<hr />
|
|
65
|
+
|
|
66
|
+
## Overview
|
|
67
|
+
|
|
68
|
+
With the rapid proliferation of Large Language Models (LLMs) enabling threat actors to generate contextually fluent spear-phishing emails that evade traditional rule-based filters (like SpamAssassin and Rspamd), defensive systems must adapt. **Anti-PLUTO** addresses this emerging threat by utilizing a state-of-the-art dual-branch feature fusion architecture.
|
|
69
|
+
|
|
70
|
+
## Key Components & Architecture
|
|
71
|
+
|
|
72
|
+
### 1. Dual-Branch Feature Fusion
|
|
73
|
+
- **High-Dimensional Stylometric Branch (100,000 dimensions)**: Extracts joint word- and character-level $n$-gram TF-IDF representations. It preserves functional stop-words, capitalization patterns, and punctuation marks to isolate the subtle, subconscious stylistic fingerprints of human writers vs. LLM token sampling distributions.
|
|
74
|
+
- **Deep Contextual Semantic Branch (768 dimensions)**: Utilizes a dense continuous vector manifold generated by fine-tuning DeBERTa-v3-small (Decoding-enhanced BERT with Disentangled Attention). It extracts contextual intent, emotional manipulation vectors, and social engineering semantics.
|
|
75
|
+
|
|
76
|
+
### 2. Unified 4-Cohort Classification
|
|
77
|
+
Anti-PLUTO is designed to effectively differentiate between four email cohorts:
|
|
78
|
+
- **Human Benign (HB)**: Legitimate, human-authored corporate correspondence.
|
|
79
|
+
- **Human Phishing (HP)**: Malicious, human-authored attacks (e.g., credential harvesting, advance-fee fraud).
|
|
80
|
+
- **LLM Benign (LB)**: Legitimate corporate communications written or polished by generative writing assistants.
|
|
81
|
+
- **LLM Phishing (LP)**: Malicious communications synthesized or polished by generative AI to execute social engineering.
|
|
82
|
+
|
|
83
|
+
### 3. Privacy-Preserving Lexical Processing Pipeline
|
|
84
|
+
Features an extensive 5-stage preprocessing pipeline designed to prevent target leakage:
|
|
85
|
+
- **RFC 5322 MIME Extraction**: Parses complex multipart email structures.
|
|
86
|
+
- **HTML Stripping & Thread Slicing**: Cleans markup and removes historic reply chains.
|
|
87
|
+
- **Unicode Normalization**: Mitigates homoglyph attacks.
|
|
88
|
+
- **Two-Stage PII Masking**: Employs deterministic regex tokenization and spaCy NER to mask URLs, IPs, email addresses, names, and organizations into standard tokens (e.g., `[NAME]`, `[URL]`).
|
|
89
|
+
|
|
90
|
+
### 4. Production Serving
|
|
91
|
+
Includes an asynchronous FastAPI REST server supporting inference on pre-computed feature vectors at high throughput (up to 7,497 msgs/s), with ONNX model export support.
|
|
92
|
+
|
|
93
|
+
---
|
|
94
|
+
|
|
95
|
+
## Installation
|
|
96
|
+
|
|
97
|
+
Anti-PLUTO is built with modularity in mind. You can install the core framework or include optional components like the classifier and REST API depending on your needs.
|
|
98
|
+
|
|
99
|
+
### Quick Setup (Recommended)
|
|
100
|
+
For the most streamlined installation experience, we have provided a setup script that installs the core framework, all optional dependencies, and the required language models in an editable state.
|
|
101
|
+
|
|
102
|
+
```bash
|
|
103
|
+
git clone https://github.com/mmaarij/antipluto.git
|
|
104
|
+
cd antipluto
|
|
105
|
+
chmod +x setup_env.sh
|
|
106
|
+
./setup_env.sh
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
### Standard Installation
|
|
110
|
+
|
|
111
|
+
Once published to PyPI, you can install the core framework directly:
|
|
112
|
+
|
|
113
|
+
```bash
|
|
114
|
+
pip install antipluto
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
To install directly from the source repository:
|
|
118
|
+
|
|
119
|
+
```bash
|
|
120
|
+
git clone https://github.com/mmaarij/antipluto.git
|
|
121
|
+
cd antipluto
|
|
122
|
+
pip install -e .
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
#### Optional Dependencies
|
|
126
|
+
|
|
127
|
+
You can install specific components of the framework depending on your use case:
|
|
128
|
+
|
|
129
|
+
- **Classifier Mode:** `pip install -e ".[classifier]"` (Installs ML dependencies)
|
|
130
|
+
- **API Mode:** `pip install -e ".[api]"` (Installs FastAPI and Uvicorn for REST endpoints)
|
|
131
|
+
- **Development/Full:** `pip install -e ".[dev,classifier,api]"`
|
|
132
|
+
|
|
133
|
+
#### Language Models (Required for Masking)
|
|
134
|
+
|
|
135
|
+
Because direct URL dependencies are restricted on PyPI, you must install the required spaCy language models manually after installing the framework:
|
|
136
|
+
|
|
137
|
+
```bash
|
|
138
|
+
# Standard model (Fast, recommended for general usage)
|
|
139
|
+
pip install https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl
|
|
140
|
+
|
|
141
|
+
# Transformer model (Highest accuracy, requires 'dev' extra dependencies)
|
|
142
|
+
pip install https://github.com/explosion/spacy-models/releases/download/en_core_web_trf-3.8.0/en_core_web_trf-3.8.0-py3-none-any.whl
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
### GPU Acceleration (Optional)
|
|
146
|
+
|
|
147
|
+
To enable GPU acceleration for the transformer-based masking and classification, you must install PyTorch with CUDA support manually. Ensure you have the CUDA Toolkit 13.x installed, then run:
|
|
148
|
+
|
|
149
|
+
```bash
|
|
150
|
+
pip install cupy-cuda13x
|
|
151
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu132 --force-reinstall
|
|
152
|
+
```
|
|
153
|
+
|
|
154
|
+
---
|
|
155
|
+
|
|
156
|
+
## Configuration
|
|
157
|
+
|
|
158
|
+
Before running the framework, you must configure your API keys (if you are generating datasets) and other system settings.
|
|
159
|
+
|
|
160
|
+
1. Copy the example configuration file:
|
|
161
|
+
```bash
|
|
162
|
+
cp antipluto/configs/default.example.yaml antipluto/configs/default.yaml
|
|
163
|
+
```
|
|
164
|
+
2. Open `antipluto/configs/default.yaml` and fill in your respective keys (e.g., `OPENROUTER_API_KEY`, `FOUNDRY_API_KEY`).
|
|
165
|
+
|
|
166
|
+
*Note: The `default.yaml` file is intentionally ignored by git to prevent sensitive data leakage.*
|
|
167
|
+
|
|
168
|
+
---
|
|
169
|
+
|
|
170
|
+
## Usage
|
|
171
|
+
|
|
172
|
+
Anti-PLUTO provides a robust CLI interface. You can invoke it after installation as `antipluto` or via python as `python -m antipluto`.
|
|
173
|
+
|
|
174
|
+
### 1. Preprocessing Data
|
|
175
|
+
Run the full ingestion pipeline to clean and process your raw corpora:
|
|
176
|
+
```bash
|
|
177
|
+
# Run with default configuration:
|
|
178
|
+
antipluto preprocess
|
|
179
|
+
|
|
180
|
+
# Run a specific dataset source (e.g., nazario):
|
|
181
|
+
antipluto preprocess --source nazario
|
|
182
|
+
|
|
183
|
+
# Override the export mode (full = 20 features, minimal = core NLP):
|
|
184
|
+
antipluto preprocess --mode full
|
|
185
|
+
```
|
|
186
|
+
|
|
187
|
+
### 2. PII Masking
|
|
188
|
+
Apply Named Entity Recognition (NER) and regex masking to anonymize sensitive information and prevent target leakage.
|
|
189
|
+
|
|
190
|
+
```bash
|
|
191
|
+
# Mask all preprocessed cohorts:
|
|
192
|
+
antipluto mask --input datasets/datasets_preprocessed/
|
|
193
|
+
|
|
194
|
+
# Mask a specific file:
|
|
195
|
+
antipluto mask --input datasets/datasets_preprocessed/human_written_phishing.jsonl
|
|
196
|
+
|
|
197
|
+
# High-accuracy transformer masking (requires GPU and en_core_web_trf):
|
|
198
|
+
antipluto mask --input datasets/datasets_preprocessed/ --spacy-model en_core_web_trf --batch-size 64
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
### 3. LLM Cohort Generation
|
|
202
|
+
Generate synthetic LLM cohorts using Azure OpenAI, Anthropic, or OpenRouter models (configurable via `default.yaml`).
|
|
203
|
+
```bash
|
|
204
|
+
# Generate benign and phishing cohorts:
|
|
205
|
+
antipluto generate --cohort benign
|
|
206
|
+
antipluto generate --cohort phishing
|
|
207
|
+
|
|
208
|
+
# Trace a synthetic email back to its original human seed email:
|
|
209
|
+
antipluto trace --record '{"source": "gpt-5-mini", "seed_idx": 42, "label": 1}'
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
### 4. Training the Classifiers
|
|
213
|
+
Train the dual-branch feature fusion classifier with a chosen fusion head (XGBoost, Random Forest, PyTorch MLP, 1D-CNN).
|
|
214
|
+
```bash
|
|
215
|
+
# Train the default XGBoost model on all cohorts:
|
|
216
|
+
antipluto train --classifier xgboost
|
|
217
|
+
|
|
218
|
+
# Compare all available models (XGBoost, RF, MLP, CNN) to find the best configuration:
|
|
219
|
+
antipluto compare --classifier all
|
|
220
|
+
```
|
|
221
|
+
|
|
222
|
+
### 5. Prediction & Exporting
|
|
223
|
+
Classify individual emails and export trained models to ONNX for portable deployment.
|
|
224
|
+
```bash
|
|
225
|
+
# Predict a single email's cohort and phishing probability:
|
|
226
|
+
antipluto predict --subject "Urgent Account Update" --body "Click here to secure your account."
|
|
227
|
+
|
|
228
|
+
# Predict from a raw email file using a specific fusion head:
|
|
229
|
+
antipluto predict --file suspicious_email.eml --classifier rf --mode fusion
|
|
230
|
+
|
|
231
|
+
# Export trained DeBERTa and classifier heads to ONNX format:
|
|
232
|
+
antipluto export --classifier all
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
### 6. Baseline Benchmarking
|
|
236
|
+
Evaluate Anti-PLUTO against industry-standard legacy engines (SpamAssassin & Rspamd) on unseen test emails.
|
|
237
|
+
```bash
|
|
238
|
+
antipluto baseline --engine all
|
|
239
|
+
```
|
|
240
|
+
|
|
241
|
+
### 7. Production API Serving
|
|
242
|
+
Launch the asynchronous FastAPI inference server to expose REST endpoints for real-time email security prediction.
|
|
243
|
+
```bash
|
|
244
|
+
# Start the API server on 127.0.0.1:8000
|
|
245
|
+
antipluto serve --classifier xgboost --mode fusion
|
|
246
|
+
|
|
247
|
+
# Enable auto-reload for local development:
|
|
248
|
+
antipluto serve --reload
|
|
249
|
+
```
|
|
250
|
+
|
|
251
|
+
### 8. Validation & Telemetry
|
|
252
|
+
Evaluate the statistics and schema integrity of your generated cohorts.
|
|
253
|
+
```bash
|
|
254
|
+
antipluto validate --input datasets/datasets_preprocessed/human_written_benign.jsonl
|
|
255
|
+
antipluto stats --input datasets/datasets_preprocessed/human_written_phishing.jsonl
|
|
256
|
+
```
|
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
<div align="center">
|
|
2
|
+
<img src="antipluto.svg" alt="Anti-PLUTO Logo" width="120" />
|
|
3
|
+
<h1>Anti-PLUTO</h1>
|
|
4
|
+
<p><strong>Anti-Phishing Lexical Utilities and Threat Observation</strong></p>
|
|
5
|
+
<p>
|
|
6
|
+
<em>A unified framework for email dataset preprocessing, PII masking, synthetic data generation, and semantic-stylometric phishing classification.</em>
|
|
7
|
+
</p>
|
|
8
|
+
<p>
|
|
9
|
+
<a href="https://github.com/mmaarij/antipluto/blob/main/LICENSE">
|
|
10
|
+
<img src="https://img.shields.io/badge/License-CC%20BY--NC%204.0-mistyrose.svg" alt="License: CC BY-NC 4.0" />
|
|
11
|
+
</a>
|
|
12
|
+
<a href="https://python.org">
|
|
13
|
+
<img src="https://img.shields.io/badge/Python-3.10%2B-lightblue" alt="Python 3.10+" />
|
|
14
|
+
</a>
|
|
15
|
+
</p>
|
|
16
|
+
</div>
|
|
17
|
+
|
|
18
|
+
<hr />
|
|
19
|
+
|
|
20
|
+
## Overview
|
|
21
|
+
|
|
22
|
+
With the rapid proliferation of Large Language Models (LLMs) enabling threat actors to generate contextually fluent spear-phishing emails that evade traditional rule-based filters (like SpamAssassin and Rspamd), defensive systems must adapt. **Anti-PLUTO** addresses this emerging threat by utilizing a state-of-the-art dual-branch feature fusion architecture.
|
|
23
|
+
|
|
24
|
+
## Key Components & Architecture
|
|
25
|
+
|
|
26
|
+
### 1. Dual-Branch Feature Fusion
|
|
27
|
+
- **High-Dimensional Stylometric Branch (100,000 dimensions)**: Extracts joint word- and character-level $n$-gram TF-IDF representations. It preserves functional stop-words, capitalization patterns, and punctuation marks to isolate the subtle, subconscious stylistic fingerprints of human writers vs. LLM token sampling distributions.
|
|
28
|
+
- **Deep Contextual Semantic Branch (768 dimensions)**: Utilizes a dense continuous vector manifold generated by fine-tuning DeBERTa-v3-small (Decoding-enhanced BERT with Disentangled Attention). It extracts contextual intent, emotional manipulation vectors, and social engineering semantics.
|
|
29
|
+
|
|
30
|
+
### 2. Unified 4-Cohort Classification
|
|
31
|
+
Anti-PLUTO is designed to effectively differentiate between four email cohorts:
|
|
32
|
+
- **Human Benign (HB)**: Legitimate, human-authored corporate correspondence.
|
|
33
|
+
- **Human Phishing (HP)**: Malicious, human-authored attacks (e.g., credential harvesting, advance-fee fraud).
|
|
34
|
+
- **LLM Benign (LB)**: Legitimate corporate communications written or polished by generative writing assistants.
|
|
35
|
+
- **LLM Phishing (LP)**: Malicious communications synthesized or polished by generative AI to execute social engineering.
|
|
36
|
+
|
|
37
|
+
### 3. Privacy-Preserving Lexical Processing Pipeline
|
|
38
|
+
Features an extensive 5-stage preprocessing pipeline designed to prevent target leakage:
|
|
39
|
+
- **RFC 5322 MIME Extraction**: Parses complex multipart email structures.
|
|
40
|
+
- **HTML Stripping & Thread Slicing**: Cleans markup and removes historic reply chains.
|
|
41
|
+
- **Unicode Normalization**: Mitigates homoglyph attacks.
|
|
42
|
+
- **Two-Stage PII Masking**: Employs deterministic regex tokenization and spaCy NER to mask URLs, IPs, email addresses, names, and organizations into standard tokens (e.g., `[NAME]`, `[URL]`).
|
|
43
|
+
|
|
44
|
+
### 4. Production Serving
|
|
45
|
+
Includes an asynchronous FastAPI REST server supporting inference on pre-computed feature vectors at high throughput (up to 7,497 msgs/s), with ONNX model export support.
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Installation
|
|
50
|
+
|
|
51
|
+
Anti-PLUTO is built with modularity in mind. You can install the core framework or include optional components like the classifier and REST API depending on your needs.
|
|
52
|
+
|
|
53
|
+
### Quick Setup (Recommended)
|
|
54
|
+
For the most streamlined installation experience, we have provided a setup script that installs the core framework, all optional dependencies, and the required language models in an editable state.
|
|
55
|
+
|
|
56
|
+
```bash
|
|
57
|
+
git clone https://github.com/mmaarij/antipluto.git
|
|
58
|
+
cd antipluto
|
|
59
|
+
chmod +x setup_env.sh
|
|
60
|
+
./setup_env.sh
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
### Standard Installation
|
|
64
|
+
|
|
65
|
+
Once published to PyPI, you can install the core framework directly:
|
|
66
|
+
|
|
67
|
+
```bash
|
|
68
|
+
pip install antipluto
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
To install directly from the source repository:
|
|
72
|
+
|
|
73
|
+
```bash
|
|
74
|
+
git clone https://github.com/mmaarij/antipluto.git
|
|
75
|
+
cd antipluto
|
|
76
|
+
pip install -e .
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
#### Optional Dependencies
|
|
80
|
+
|
|
81
|
+
You can install specific components of the framework depending on your use case:
|
|
82
|
+
|
|
83
|
+
- **Classifier Mode:** `pip install -e ".[classifier]"` (Installs ML dependencies)
|
|
84
|
+
- **API Mode:** `pip install -e ".[api]"` (Installs FastAPI and Uvicorn for REST endpoints)
|
|
85
|
+
- **Development/Full:** `pip install -e ".[dev,classifier,api]"`
|
|
86
|
+
|
|
87
|
+
#### Language Models (Required for Masking)
|
|
88
|
+
|
|
89
|
+
Because direct URL dependencies are restricted on PyPI, you must install the required spaCy language models manually after installing the framework:
|
|
90
|
+
|
|
91
|
+
```bash
|
|
92
|
+
# Standard model (Fast, recommended for general usage)
|
|
93
|
+
pip install https://github.com/explosion/spacy-models/releases/download/en_core_web_sm-3.8.0/en_core_web_sm-3.8.0-py3-none-any.whl
|
|
94
|
+
|
|
95
|
+
# Transformer model (Highest accuracy, requires 'dev' extra dependencies)
|
|
96
|
+
pip install https://github.com/explosion/spacy-models/releases/download/en_core_web_trf-3.8.0/en_core_web_trf-3.8.0-py3-none-any.whl
|
|
97
|
+
```
|
|
98
|
+
|
|
99
|
+
### GPU Acceleration (Optional)
|
|
100
|
+
|
|
101
|
+
To enable GPU acceleration for the transformer-based masking and classification, you must install PyTorch with CUDA support manually. Ensure you have the CUDA Toolkit 13.x installed, then run:
|
|
102
|
+
|
|
103
|
+
```bash
|
|
104
|
+
pip install cupy-cuda13x
|
|
105
|
+
pip install torch --index-url https://download.pytorch.org/whl/cu132 --force-reinstall
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Configuration
|
|
111
|
+
|
|
112
|
+
Before running the framework, you must configure your API keys (if you are generating datasets) and other system settings.
|
|
113
|
+
|
|
114
|
+
1. Copy the example configuration file:
|
|
115
|
+
```bash
|
|
116
|
+
cp antipluto/configs/default.example.yaml antipluto/configs/default.yaml
|
|
117
|
+
```
|
|
118
|
+
2. Open `antipluto/configs/default.yaml` and fill in your respective keys (e.g., `OPENROUTER_API_KEY`, `FOUNDRY_API_KEY`).
|
|
119
|
+
|
|
120
|
+
*Note: The `default.yaml` file is intentionally ignored by git to prevent sensitive data leakage.*
|
|
121
|
+
|
|
122
|
+
---
|
|
123
|
+
|
|
124
|
+
## Usage
|
|
125
|
+
|
|
126
|
+
Anti-PLUTO provides a robust CLI interface. You can invoke it after installation as `antipluto` or via python as `python -m antipluto`.
|
|
127
|
+
|
|
128
|
+
### 1. Preprocessing Data
|
|
129
|
+
Run the full ingestion pipeline to clean and process your raw corpora:
|
|
130
|
+
```bash
|
|
131
|
+
# Run with default configuration:
|
|
132
|
+
antipluto preprocess
|
|
133
|
+
|
|
134
|
+
# Run a specific dataset source (e.g., nazario):
|
|
135
|
+
antipluto preprocess --source nazario
|
|
136
|
+
|
|
137
|
+
# Override the export mode (full = 20 features, minimal = core NLP):
|
|
138
|
+
antipluto preprocess --mode full
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### 2. PII Masking
|
|
142
|
+
Apply Named Entity Recognition (NER) and regex masking to anonymize sensitive information and prevent target leakage.
|
|
143
|
+
|
|
144
|
+
```bash
|
|
145
|
+
# Mask all preprocessed cohorts:
|
|
146
|
+
antipluto mask --input datasets/datasets_preprocessed/
|
|
147
|
+
|
|
148
|
+
# Mask a specific file:
|
|
149
|
+
antipluto mask --input datasets/datasets_preprocessed/human_written_phishing.jsonl
|
|
150
|
+
|
|
151
|
+
# High-accuracy transformer masking (requires GPU and en_core_web_trf):
|
|
152
|
+
antipluto mask --input datasets/datasets_preprocessed/ --spacy-model en_core_web_trf --batch-size 64
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
### 3. LLM Cohort Generation
|
|
156
|
+
Generate synthetic LLM cohorts using Azure OpenAI, Anthropic, or OpenRouter models (configurable via `default.yaml`).
|
|
157
|
+
```bash
|
|
158
|
+
# Generate benign and phishing cohorts:
|
|
159
|
+
antipluto generate --cohort benign
|
|
160
|
+
antipluto generate --cohort phishing
|
|
161
|
+
|
|
162
|
+
# Trace a synthetic email back to its original human seed email:
|
|
163
|
+
antipluto trace --record '{"source": "gpt-5-mini", "seed_idx": 42, "label": 1}'
|
|
164
|
+
```
|
|
165
|
+
|
|
166
|
+
### 4. Training the Classifiers
|
|
167
|
+
Train the dual-branch feature fusion classifier with a chosen fusion head (XGBoost, Random Forest, PyTorch MLP, 1D-CNN).
|
|
168
|
+
```bash
|
|
169
|
+
# Train the default XGBoost model on all cohorts:
|
|
170
|
+
antipluto train --classifier xgboost
|
|
171
|
+
|
|
172
|
+
# Compare all available models (XGBoost, RF, MLP, CNN) to find the best configuration:
|
|
173
|
+
antipluto compare --classifier all
|
|
174
|
+
```
|
|
175
|
+
|
|
176
|
+
### 5. Prediction & Exporting
|
|
177
|
+
Classify individual emails and export trained models to ONNX for portable deployment.
|
|
178
|
+
```bash
|
|
179
|
+
# Predict a single email's cohort and phishing probability:
|
|
180
|
+
antipluto predict --subject "Urgent Account Update" --body "Click here to secure your account."
|
|
181
|
+
|
|
182
|
+
# Predict from a raw email file using a specific fusion head:
|
|
183
|
+
antipluto predict --file suspicious_email.eml --classifier rf --mode fusion
|
|
184
|
+
|
|
185
|
+
# Export trained DeBERTa and classifier heads to ONNX format:
|
|
186
|
+
antipluto export --classifier all
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
### 6. Baseline Benchmarking
|
|
190
|
+
Evaluate Anti-PLUTO against industry-standard legacy engines (SpamAssassin & Rspamd) on unseen test emails.
|
|
191
|
+
```bash
|
|
192
|
+
antipluto baseline --engine all
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
### 7. Production API Serving
|
|
196
|
+
Launch the asynchronous FastAPI inference server to expose REST endpoints for real-time email security prediction.
|
|
197
|
+
```bash
|
|
198
|
+
# Start the API server on 127.0.0.1:8000
|
|
199
|
+
antipluto serve --classifier xgboost --mode fusion
|
|
200
|
+
|
|
201
|
+
# Enable auto-reload for local development:
|
|
202
|
+
antipluto serve --reload
|
|
203
|
+
```
|
|
204
|
+
|
|
205
|
+
### 8. Validation & Telemetry
|
|
206
|
+
Evaluate the statistics and schema integrity of your generated cohorts.
|
|
207
|
+
```bash
|
|
208
|
+
antipluto validate --input datasets/datasets_preprocessed/human_written_benign.jsonl
|
|
209
|
+
antipluto stats --input datasets/datasets_preprocessed/human_written_phishing.jsonl
|
|
210
|
+
```
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""
|
|
2
|
+
antipluto
|
|
3
|
+
=======
|
|
4
|
+
Anti-PLUTO — Anti-Phishing Lexical Utilities and Threat Observation.
|
|
5
|
+
|
|
6
|
+
A modular, end-to-end Python framework for:
|
|
7
|
+
|
|
8
|
+
- Preprocessing raw email corpora into clean, schema-validated JSONL datasets.
|
|
9
|
+
- Anonymising PII via a two-stage MeAJOR-compatible masking pipeline.
|
|
10
|
+
- Generating LLM-assisted and LLM-generated email cohorts via OpenRouter.
|
|
11
|
+
- Training a dual-branch stylometric + semantic phishing classifier.
|
|
12
|
+
- Evaluating classifier performance against SpamAssassin / Rspamd baselines.
|
|
13
|
+
- Serving real-time phishing predictions via a FastAPI endpoint.
|
|
14
|
+
|
|
15
|
+
Academic Context
|
|
16
|
+
----------------
|
|
17
|
+
MSc Computing Thesis — Dual-branch detection of LLM-generated phishing email.
|
|
18
|
+
(Stylometric TF-IDF + Semantic DeBERTaV3 → XGBoost)
|
|
19
|
+
|
|
20
|
+
Subpackages
|
|
21
|
+
-----------
|
|
22
|
+
antipluto.preprocessing Corpus ingestion, cleaning, and schema validation.
|
|
23
|
+
antipluto.masking Two-stage PII anonymisation (spaCy NER + regex).
|
|
24
|
+
antipluto.generation LLM cohort generation via OpenRouter API.
|
|
25
|
+
antipluto.classifier Dual-branch stylometric + semantic classifier.
|
|
26
|
+
antipluto.baseline SpamAssassin / Rspamd comparison harness.
|
|
27
|
+
antipluto.api FastAPI real-world prediction endpoint.
|
|
28
|
+
antipluto.utils Shared JSONL I/O utilities.
|
|
29
|
+
|
|
30
|
+
License: MIT
|
|
31
|
+
"""
|
|
32
|
+
|
|
33
|
+
__version__ = "2.0.0"
|
|
34
|
+
__author__ = "Maarij"
|