saber-risk 0.2.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- saber_risk-0.2.0/.github/dependabot.yml +6 -0
- saber_risk-0.2.0/.github/workflows/codeql.yml +35 -0
- saber_risk-0.2.0/.gitignore +116 -0
- saber_risk-0.2.0/LICENSE +21 -0
- saber_risk-0.2.0/PKG-INFO +165 -0
- saber_risk-0.2.0/README.md +136 -0
- saber_risk-0.2.0/SECURITY.md +14 -0
- saber_risk-0.2.0/docs/Makefile +14 -0
- saber_risk-0.2.0/docs/advanced_usage.rst +124 -0
- saber_risk-0.2.0/docs/api_reference.rst +259 -0
- saber_risk-0.2.0/docs/conf.py +58 -0
- saber_risk-0.2.0/docs/index.rst +47 -0
- saber_risk-0.2.0/docs/quickstart.rst +84 -0
- saber_risk-0.2.0/docs/requirements.txt +3 -0
- saber_risk-0.2.0/examples/basic_usage.py +189 -0
- saber_risk-0.2.0/figures/saber_method.png +0 -0
- saber_risk-0.2.0/pyproject.toml +63 -0
- saber_risk-0.2.0/saber/__init__.py +67 -0
- saber_risk-0.2.0/saber/_types.py +115 -0
- saber_risk-0.2.0/saber/_version.py +3 -0
- saber_risk-0.2.0/saber/estimator.py +573 -0
- saber_risk-0.2.0/saber/fitting.py +298 -0
- saber_risk-0.2.0/saber/prediction.py +269 -0
- saber_risk-0.2.0/saber/uncertainty.py +272 -0
- saber_risk-0.2.0/tests/__init__.py +0 -0
- saber_risk-0.2.0/tests/test_saber.py +629 -0
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
name: "CodeQL"
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [ main ]
|
|
6
|
+
pull_request:
|
|
7
|
+
branches: [ main ]
|
|
8
|
+
schedule:
|
|
9
|
+
- cron: '0 0 * * 0' # Weekly on Sunday
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
analyze:
|
|
13
|
+
name: Analyze
|
|
14
|
+
runs-on: ubuntu-latest
|
|
15
|
+
permissions:
|
|
16
|
+
actions: read
|
|
17
|
+
contents: read
|
|
18
|
+
security-events: write
|
|
19
|
+
|
|
20
|
+
strategy:
|
|
21
|
+
fail-fast: false
|
|
22
|
+
matrix:
|
|
23
|
+
language: [ 'python' ]
|
|
24
|
+
|
|
25
|
+
steps:
|
|
26
|
+
- name: Checkout repository
|
|
27
|
+
uses: actions/checkout@v4
|
|
28
|
+
|
|
29
|
+
- name: Initialize CodeQL
|
|
30
|
+
uses: github/codeql-action/init@v3
|
|
31
|
+
with:
|
|
32
|
+
languages: ${{ matrix.language }}
|
|
33
|
+
|
|
34
|
+
- name: Perform CodeQL Analysis
|
|
35
|
+
uses: github/codeql-action/analyze@v3
|
|
@@ -0,0 +1,116 @@
|
|
|
1
|
+
# Byte-compiled / optimized / DLL files
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
|
|
6
|
+
# C extensions
|
|
7
|
+
*.so
|
|
8
|
+
|
|
9
|
+
# Distribution / packaging
|
|
10
|
+
.Python
|
|
11
|
+
build/
|
|
12
|
+
develop-eggs/
|
|
13
|
+
dist/
|
|
14
|
+
downloads/
|
|
15
|
+
eggs/
|
|
16
|
+
.eggs/
|
|
17
|
+
lib/
|
|
18
|
+
lib64/
|
|
19
|
+
parts/
|
|
20
|
+
sdist/
|
|
21
|
+
var/
|
|
22
|
+
wheels/
|
|
23
|
+
*.egg-info/
|
|
24
|
+
.installed.cfg
|
|
25
|
+
*.egg
|
|
26
|
+
MANIFEST
|
|
27
|
+
.DS_Store
|
|
28
|
+
|
|
29
|
+
# PyInstaller
|
|
30
|
+
# Usually these files are written by a python script from a template
|
|
31
|
+
# before PyInstaller builds the exe, so as to inject date/other infos into it.
|
|
32
|
+
*.manifest
|
|
33
|
+
*.spec
|
|
34
|
+
|
|
35
|
+
# Installer logs
|
|
36
|
+
pip-log.txt
|
|
37
|
+
pip-delete-this-directory.txt
|
|
38
|
+
|
|
39
|
+
# Unit test / coverage reports
|
|
40
|
+
htmlcov/
|
|
41
|
+
.tox/
|
|
42
|
+
.coverage
|
|
43
|
+
.coverage.*
|
|
44
|
+
.cache
|
|
45
|
+
nosetests.xml
|
|
46
|
+
coverage.xml
|
|
47
|
+
*.cover
|
|
48
|
+
.hypothesis/
|
|
49
|
+
.pytest_cache/
|
|
50
|
+
|
|
51
|
+
# Translations
|
|
52
|
+
*.mo
|
|
53
|
+
*.pot
|
|
54
|
+
|
|
55
|
+
# Django stuff:
|
|
56
|
+
*.log
|
|
57
|
+
local_settings.py
|
|
58
|
+
db.sqlite3
|
|
59
|
+
|
|
60
|
+
# Flask stuff:
|
|
61
|
+
instance/
|
|
62
|
+
.webassets-cache
|
|
63
|
+
|
|
64
|
+
# Scrapy stuff:
|
|
65
|
+
.scrapy
|
|
66
|
+
|
|
67
|
+
# Sphinx documentation
|
|
68
|
+
docs/_build/
|
|
69
|
+
|
|
70
|
+
# PyBuilder
|
|
71
|
+
target/
|
|
72
|
+
|
|
73
|
+
# Jupyter Notebook
|
|
74
|
+
.ipynb_checkpoints
|
|
75
|
+
|
|
76
|
+
# pyenv
|
|
77
|
+
.python-version
|
|
78
|
+
|
|
79
|
+
# celery beat schedule file
|
|
80
|
+
celerybeat-schedule
|
|
81
|
+
|
|
82
|
+
# SageMath parsed files
|
|
83
|
+
*.sage.py
|
|
84
|
+
|
|
85
|
+
# Environments
|
|
86
|
+
.env
|
|
87
|
+
.venv
|
|
88
|
+
env/
|
|
89
|
+
venv/
|
|
90
|
+
ENV/
|
|
91
|
+
env.bak/
|
|
92
|
+
venv.bak/
|
|
93
|
+
|
|
94
|
+
# Spyder project settings
|
|
95
|
+
.spyderproject
|
|
96
|
+
.spyproject
|
|
97
|
+
|
|
98
|
+
# Rope project settings
|
|
99
|
+
.ropeproject
|
|
100
|
+
|
|
101
|
+
# mkdocs documentation
|
|
102
|
+
/site
|
|
103
|
+
|
|
104
|
+
# mypy
|
|
105
|
+
.mypy_cache/
|
|
106
|
+
|
|
107
|
+
# IDE pycharm
|
|
108
|
+
.idea/
|
|
109
|
+
|
|
110
|
+
.pt_description_history
|
|
111
|
+
.git-credentials
|
|
112
|
+
pt_bert/philly
|
|
113
|
+
.vs
|
|
114
|
+
*.pyproj
|
|
115
|
+
*pyc
|
|
116
|
+
/.vscode
|
saber_risk-0.2.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) Microsoft Corporation.
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE
|
|
@@ -0,0 +1,165 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: saber-risk
|
|
3
|
+
Version: 0.2.0
|
|
4
|
+
Summary: SABER: Scaling-Aware Best-of-N Estimation of Risk for LLM safety evaluation
|
|
5
|
+
Project-URL: Homepage, https://github.com/microsoft/saber
|
|
6
|
+
Project-URL: Documentation, https://github.com/microsoft/saber#readme
|
|
7
|
+
Project-URL: Repository, https://github.com/microsoft/saber
|
|
8
|
+
Author: Xiaodong Liu, Mingqian Feng
|
|
9
|
+
License: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: adversarial,best-of-n,jailbreak,llm,risk-estimation,safety
|
|
12
|
+
Classifier: Development Status :: 4 - Beta
|
|
13
|
+
Classifier: Intended Audience :: Science/Research
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: numpy>=1.20.0
|
|
23
|
+
Requires-Dist: scipy>=1.7.0
|
|
24
|
+
Provides-Extra: dev
|
|
25
|
+
Requires-Dist: mypy>=1.0.0; extra == 'dev'
|
|
26
|
+
Requires-Dist: pytest-cov>=4.0.0; extra == 'dev'
|
|
27
|
+
Requires-Dist: pytest>=7.0.0; extra == 'dev'
|
|
28
|
+
Description-Content-Type: text/markdown
|
|
29
|
+
|
|
30
|
+
# SABER
|
|
31
|
+
|
|
32
|
+
**S**caling-**A**ware **B**est-of-N **E**stimation of **R**isk
|
|
33
|
+
|
|
34
|
+
A Python package for predicting large-scale adversarial risk in Large Language Models under Best-of-N sampling.
|
|
35
|
+
|
|
36
|
+
**Paper**: https://arxiv.org/pdf/2601.22636
|
|
37
|
+
|
|
38
|
+
[](https://www.python.org/downloads/)
|
|
39
|
+
[](https://opensource.org/licenses/MIT)
|
|
40
|
+
|
|
41
|
+
## Overview
|
|
42
|
+
|
|
43
|
+
Standard LLM safety evaluations use single-shot (ASR@1) metrics, but real attackers can exploit parallel sampling to repeatedly probe models. SABER provides a principled statistical framework to:
|
|
44
|
+
|
|
45
|
+
- **Predict** ASR@N at large budgets from small measurements
|
|
46
|
+
- **Estimate** how many attempts are needed to reach a target success rate
|
|
47
|
+
- **Quantify** uncertainty in adversarial risk predictions
|
|
48
|
+
|
|
49
|
+

|
|
50
|
+
|
|
51
|
+
### Key Insight
|
|
52
|
+
|
|
53
|
+
Attack success rates scale according to a power law governed by the Beta distribution of per-query vulnerabilities:
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
ASR@N ≈ 1 - Γ(α+β)/Γ(β) · N^(-α)
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
The parameter **α** controls how fast risk amplifies with more attempts.
|
|
60
|
+
|
|
61
|
+
## Installation
|
|
62
|
+
|
|
63
|
+
```bash
|
|
64
|
+
pip install saber-risk
|
|
65
|
+
```
|
|
66
|
+
|
|
67
|
+
Or from source:
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
git clone https://github.com/microsoft/saber
|
|
71
|
+
cd saber
|
|
72
|
+
pip install -e .
|
|
73
|
+
```
|
|
74
|
+
|
|
75
|
+
## Quick Start
|
|
76
|
+
|
|
77
|
+
```python
|
|
78
|
+
import numpy as np
|
|
79
|
+
from saber import SABER
|
|
80
|
+
|
|
81
|
+
# Your jailbreak evaluation data:
|
|
82
|
+
# k[i] = number of successful jailbreaks for query i
|
|
83
|
+
# n[i] = number of attempts for query i
|
|
84
|
+
k = np.array([3, 5, 0, 2, 8, 1, 4, 0, 6, 2])
|
|
85
|
+
n = 100 # 100 attempts per query
|
|
86
|
+
|
|
87
|
+
# Fit and predict
|
|
88
|
+
model = SABER()
|
|
89
|
+
model.fit(k, n)
|
|
90
|
+
|
|
91
|
+
# Predict ASR at N=1000 attempts
|
|
92
|
+
result = model.predict(N=1000)
|
|
93
|
+
print(f"ASR@1000 = {result.asr:.2%}")
|
|
94
|
+
|
|
95
|
+
# With confidence interval
|
|
96
|
+
result = model.predict(N=1000, confidence=0.95)
|
|
97
|
+
print(f"ASR@1000 = {result.asr:.2%} [{result.ci_lower:.2%}, {result.ci_upper:.2%}]")
|
|
98
|
+
```
|
|
99
|
+
|
|
100
|
+
## Core Usage
|
|
101
|
+
|
|
102
|
+
```python
|
|
103
|
+
from saber import SABER
|
|
104
|
+
|
|
105
|
+
# 1. Collect jailbreak data
|
|
106
|
+
# Run n attempts per query, count successes k
|
|
107
|
+
k = [...] # successes per query
|
|
108
|
+
n = 100 # trials per query (or array for heterogeneous budgets)
|
|
109
|
+
|
|
110
|
+
# 2. Fit the model
|
|
111
|
+
model = SABER()
|
|
112
|
+
model.fit(k, n)
|
|
113
|
+
|
|
114
|
+
# 3. Predict ASR at target budget
|
|
115
|
+
asr_1000 = model.predict(N=1000).asr
|
|
116
|
+
|
|
117
|
+
# Budget estimation
|
|
118
|
+
result = model.budget_for_asr(target=0.95)
|
|
119
|
+
print(f"Need {result.budget:.0f} attempts for 95% ASR")
|
|
120
|
+
|
|
121
|
+
# Fluent API
|
|
122
|
+
asr = SABER().fit(k, n).predict(1000).asr
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
## Documentation
|
|
126
|
+
|
|
127
|
+
Full documentation is available in the `docs/` directory. To build:
|
|
128
|
+
|
|
129
|
+
```bash
|
|
130
|
+
cd docs
|
|
131
|
+
pip install -r requirements.txt
|
|
132
|
+
make html
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
- **[Quick Start](https://github.com/microsoft/saber/blob/main/docs/quickstart.rst)** - Getting started guide
|
|
136
|
+
- **[API Reference](https://github.com/microsoft/saber/blob/main/docs/api_reference.rst)** - Complete API documentation
|
|
137
|
+
- **[Advanced Usage](https://github.com/microsoft/saber/blob/main/docs/advanced_usage.rst)** - Model selection, scaling curves, low-level API
|
|
138
|
+
|
|
139
|
+
## Citation
|
|
140
|
+
|
|
141
|
+
If you use SABER in your research, please cite:
|
|
142
|
+
|
|
143
|
+
```bibtex
|
|
144
|
+
@misc{feng2026statisticalestimationadversarialrisk,
|
|
145
|
+
title={Statistical Estimation of Adversarial Risk in Large Language Models under Best-of-N Sampling},
|
|
146
|
+
author={Mingqian Feng and Xiaodong Liu and Weiwei Yang and Chenliang Xu and Christopher White and Jianfeng Gao},
|
|
147
|
+
year={2026},
|
|
148
|
+
eprint={2601.22636},
|
|
149
|
+
archivePrefix={arXiv},
|
|
150
|
+
primaryClass={cs.AI},
|
|
151
|
+
url={https://arxiv.org/abs/2601.22636},
|
|
152
|
+
}
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
## Contact
|
|
156
|
+
|
|
157
|
+
For any questions regarding the package or paper, feel free to reach out to:
|
|
158
|
+
|
|
159
|
+
- **Mingqian Feng** - mfeng7@ur.rochester.edu
|
|
160
|
+
- **Xiaodong Liu** - xiaodl@microsoft.com
|
|
161
|
+
- **Weiwei Yang** - weiwei.yang@microsoft.com
|
|
162
|
+
|
|
163
|
+
## License
|
|
164
|
+
|
|
165
|
+
MIT License - see [LICENSE](https://github.com/microsoft/saber/blob/main/LICENSE) for details.
|
|
@@ -0,0 +1,136 @@
|
|
|
1
|
+
# SABER
|
|
2
|
+
|
|
3
|
+
**S**caling-**A**ware **B**est-of-N **E**stimation of **R**isk
|
|
4
|
+
|
|
5
|
+
A Python package for predicting large-scale adversarial risk in Large Language Models under Best-of-N sampling.
|
|
6
|
+
|
|
7
|
+
**Paper**: https://arxiv.org/pdf/2601.22636
|
|
8
|
+
|
|
9
|
+
[](https://www.python.org/downloads/)
|
|
10
|
+
[](https://opensource.org/licenses/MIT)
|
|
11
|
+
|
|
12
|
+
## Overview
|
|
13
|
+
|
|
14
|
+
Standard LLM safety evaluations use single-shot (ASR@1) metrics, but real attackers can exploit parallel sampling to repeatedly probe models. SABER provides a principled statistical framework to:
|
|
15
|
+
|
|
16
|
+
- **Predict** ASR@N at large budgets from small measurements
|
|
17
|
+
- **Estimate** how many attempts are needed to reach a target success rate
|
|
18
|
+
- **Quantify** uncertainty in adversarial risk predictions
|
|
19
|
+
|
|
20
|
+

|
|
21
|
+
|
|
22
|
+
### Key Insight
|
|
23
|
+
|
|
24
|
+
Attack success rates scale according to a power law governed by the Beta distribution of per-query vulnerabilities:
|
|
25
|
+
|
|
26
|
+
```
|
|
27
|
+
ASR@N ≈ 1 - Γ(α+β)/Γ(β) · N^(-α)
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
The parameter **α** controls how fast risk amplifies with more attempts.
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install saber-risk
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
Or from source:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
git clone https://github.com/microsoft/saber
|
|
42
|
+
cd saber
|
|
43
|
+
pip install -e .
|
|
44
|
+
```
|
|
45
|
+
|
|
46
|
+
## Quick Start
|
|
47
|
+
|
|
48
|
+
```python
|
|
49
|
+
import numpy as np
|
|
50
|
+
from saber import SABER
|
|
51
|
+
|
|
52
|
+
# Your jailbreak evaluation data:
|
|
53
|
+
# k[i] = number of successful jailbreaks for query i
|
|
54
|
+
# n[i] = number of attempts for query i
|
|
55
|
+
k = np.array([3, 5, 0, 2, 8, 1, 4, 0, 6, 2])
|
|
56
|
+
n = 100 # 100 attempts per query
|
|
57
|
+
|
|
58
|
+
# Fit and predict
|
|
59
|
+
model = SABER()
|
|
60
|
+
model.fit(k, n)
|
|
61
|
+
|
|
62
|
+
# Predict ASR at N=1000 attempts
|
|
63
|
+
result = model.predict(N=1000)
|
|
64
|
+
print(f"ASR@1000 = {result.asr:.2%}")
|
|
65
|
+
|
|
66
|
+
# With confidence interval
|
|
67
|
+
result = model.predict(N=1000, confidence=0.95)
|
|
68
|
+
print(f"ASR@1000 = {result.asr:.2%} [{result.ci_lower:.2%}, {result.ci_upper:.2%}]")
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## Core Usage
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from saber import SABER
|
|
75
|
+
|
|
76
|
+
# 1. Collect jailbreak data
|
|
77
|
+
# Run n attempts per query, count successes k
|
|
78
|
+
k = [...] # successes per query
|
|
79
|
+
n = 100 # trials per query (or array for heterogeneous budgets)
|
|
80
|
+
|
|
81
|
+
# 2. Fit the model
|
|
82
|
+
model = SABER()
|
|
83
|
+
model.fit(k, n)
|
|
84
|
+
|
|
85
|
+
# 3. Predict ASR at target budget
|
|
86
|
+
asr_1000 = model.predict(N=1000).asr
|
|
87
|
+
|
|
88
|
+
# Budget estimation
|
|
89
|
+
result = model.budget_for_asr(target=0.95)
|
|
90
|
+
print(f"Need {result.budget:.0f} attempts for 95% ASR")
|
|
91
|
+
|
|
92
|
+
# Fluent API
|
|
93
|
+
asr = SABER().fit(k, n).predict(1000).asr
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
## Documentation
|
|
97
|
+
|
|
98
|
+
Full documentation is available in the `docs/` directory. To build:
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
cd docs
|
|
102
|
+
pip install -r requirements.txt
|
|
103
|
+
make html
|
|
104
|
+
```
|
|
105
|
+
|
|
106
|
+
- **[Quick Start](https://github.com/microsoft/saber/blob/main/docs/quickstart.rst)** - Getting started guide
|
|
107
|
+
- **[API Reference](https://github.com/microsoft/saber/blob/main/docs/api_reference.rst)** - Complete API documentation
|
|
108
|
+
- **[Advanced Usage](https://github.com/microsoft/saber/blob/main/docs/advanced_usage.rst)** - Model selection, scaling curves, low-level API
|
|
109
|
+
|
|
110
|
+
## Citation
|
|
111
|
+
|
|
112
|
+
If you use SABER in your research, please cite:
|
|
113
|
+
|
|
114
|
+
```bibtex
|
|
115
|
+
@misc{feng2026statisticalestimationadversarialrisk,
|
|
116
|
+
title={Statistical Estimation of Adversarial Risk in Large Language Models under Best-of-N Sampling},
|
|
117
|
+
author={Mingqian Feng and Xiaodong Liu and Weiwei Yang and Chenliang Xu and Christopher White and Jianfeng Gao},
|
|
118
|
+
year={2026},
|
|
119
|
+
eprint={2601.22636},
|
|
120
|
+
archivePrefix={arXiv},
|
|
121
|
+
primaryClass={cs.AI},
|
|
122
|
+
url={https://arxiv.org/abs/2601.22636},
|
|
123
|
+
}
|
|
124
|
+
```
|
|
125
|
+
|
|
126
|
+
## Contact
|
|
127
|
+
|
|
128
|
+
For any questions regarding the package or paper, feel free to reach out to:
|
|
129
|
+
|
|
130
|
+
- **Mingqian Feng** - mfeng7@ur.rochester.edu
|
|
131
|
+
- **Xiaodong Liu** - xiaodl@microsoft.com
|
|
132
|
+
- **Weiwei Yang** - weiwei.yang@microsoft.com
|
|
133
|
+
|
|
134
|
+
## License
|
|
135
|
+
|
|
136
|
+
MIT License - see [LICENSE](https://github.com/microsoft/saber/blob/main/LICENSE) for details.
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
<!-- BEGIN MICROSOFT SECURITY.MD V1.0.0 BLOCK -->
|
|
2
|
+
|
|
3
|
+
## Security
|
|
4
|
+
|
|
5
|
+
Microsoft takes the security of our software products and services seriously, which
|
|
6
|
+
includes all source code repositories in our GitHub organizations.
|
|
7
|
+
|
|
8
|
+
**Please do not report security vulnerabilities through public GitHub issues.**
|
|
9
|
+
|
|
10
|
+
For security reporting information, locations, contact information, and policies,
|
|
11
|
+
please review the latest guidance for Microsoft repositories at
|
|
12
|
+
[https://aka.ms/SECURITY.md](https://aka.ms/SECURITY.md).
|
|
13
|
+
|
|
14
|
+
<!-- END MICROSOFT SECURITY.MD BLOCK -->
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Minimal makefile for Sphinx documentation
|
|
2
|
+
|
|
3
|
+
SPHINXOPTS ?=
|
|
4
|
+
SPHINXBUILD ?= sphinx-build
|
|
5
|
+
SOURCEDIR = .
|
|
6
|
+
BUILDDIR = _build
|
|
7
|
+
|
|
8
|
+
help:
|
|
9
|
+
@$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
|
10
|
+
|
|
11
|
+
.PHONY: help Makefile
|
|
12
|
+
|
|
13
|
+
%: Makefile
|
|
14
|
+
@$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
Advanced Usage
|
|
2
|
+
==============
|
|
3
|
+
|
|
4
|
+
Heterogeneous Budgets
|
|
5
|
+
---------------------
|
|
6
|
+
|
|
7
|
+
SABER supports different trial counts per query:
|
|
8
|
+
|
|
9
|
+
.. code-block:: python
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
from saber import SABER
|
|
13
|
+
|
|
14
|
+
k = np.array([3, 5, 0, 2, 8])
|
|
15
|
+
n = np.array([50, 100, 75, 100, 200]) # Different budgets
|
|
16
|
+
|
|
17
|
+
model = SABER()
|
|
18
|
+
model.fit(k, n)
|
|
19
|
+
|
|
20
|
+
Model Selection
|
|
21
|
+
---------------
|
|
22
|
+
|
|
23
|
+
.. code-block:: python
|
|
24
|
+
|
|
25
|
+
# SABER-Anchored (default, recommended)
|
|
26
|
+
model = SABER(method="anchored")
|
|
27
|
+
|
|
28
|
+
# SABER-Plugin (uses β directly)
|
|
29
|
+
model = SABER(method="plugin")
|
|
30
|
+
|
|
31
|
+
# Override for specific prediction
|
|
32
|
+
result = model.predict(N=1000, method="naive")
|
|
33
|
+
result = model.predict(N=1000, method="combinatorial") # N must be <= min(n)
|
|
34
|
+
result = model.predict(N=1000, method="log_linear_curve") # auto-fits curve if not fitted
|
|
35
|
+
|
|
36
|
+
# Small-N Correction
|
|
37
|
+
result = model.predict(N=1000, method="anchored", correction=True)
|
|
38
|
+
|
|
39
|
+
Log-Linear Curve Fit
|
|
40
|
+
--------------------
|
|
41
|
+
|
|
42
|
+
The default fit is always the Beta distribution (α, β). You can additionally fit a log-linear curve:
|
|
43
|
+
|
|
44
|
+
.. code-block:: python
|
|
45
|
+
|
|
46
|
+
# Fit both distribution and curve
|
|
47
|
+
model = SABER()
|
|
48
|
+
model.fit(k, n, also_fit_curve=True)
|
|
49
|
+
|
|
50
|
+
# Or: fit only distribution; curve will be auto-fitted on first use
|
|
51
|
+
model.fit(k, n)
|
|
52
|
+
result = model.predict(N=1000, method="log_linear_curve")
|
|
53
|
+
|
|
54
|
+
Low-Level API
|
|
55
|
+
-------------
|
|
56
|
+
|
|
57
|
+
For advanced usage, access the underlying functions directly:
|
|
58
|
+
|
|
59
|
+
.. code-block:: python
|
|
60
|
+
|
|
61
|
+
from saber import fit_beta_binomial, predict_anchored, estimate_budget
|
|
62
|
+
|
|
63
|
+
# Fit distribution
|
|
64
|
+
alpha, beta = fit_beta_binomial(k, n)
|
|
65
|
+
|
|
66
|
+
# Predict ASR
|
|
67
|
+
asr = predict_anchored(N=1000, n=100, asr_at_n=0.73, alpha=alpha)
|
|
68
|
+
|
|
69
|
+
# Estimate budget
|
|
70
|
+
budget = estimate_budget(target_asr=0.95, n=100, asr_at_n=0.73, alpha=alpha)
|
|
71
|
+
|
|
72
|
+
Scaling Curves
|
|
73
|
+
--------------
|
|
74
|
+
|
|
75
|
+
Generate data for plotting:
|
|
76
|
+
|
|
77
|
+
.. code-block:: python
|
|
78
|
+
|
|
79
|
+
import matplotlib.pyplot as plt
|
|
80
|
+
|
|
81
|
+
curve = model.scaling_curve(
|
|
82
|
+
include_naive=True,
|
|
83
|
+
include_combinatorial=True,
|
|
84
|
+
correction=False
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
plt.figure(figsize=(8, 5))
|
|
88
|
+
plt.plot(curve['N'], curve['asr_anchored'] * 100, label='SABER-Anchored')
|
|
89
|
+
plt.plot(curve['N'], curve['asr_plugin'] * 100, label='SABER-Plugin')
|
|
90
|
+
if 'asr_log_linear_curve' in curve:
|
|
91
|
+
plt.plot(curve['N'], curve['asr_log_linear_curve'] * 100, label='Log-linear curve')
|
|
92
|
+
plt.plot(curve['N'], curve['asr_naive'] * 100, '--', label='Naive Baseline')
|
|
93
|
+
plt.xscale('log')
|
|
94
|
+
plt.xlabel('Number of Attempts (N)')
|
|
95
|
+
plt.ylabel('ASR@N (%)')
|
|
96
|
+
plt.legend()
|
|
97
|
+
plt.title('ASR Scaling Curve')
|
|
98
|
+
plt.show()
|
|
99
|
+
|
|
100
|
+
Estimator Variants
|
|
101
|
+
------------------
|
|
102
|
+
|
|
103
|
+
.. list-table::
|
|
104
|
+
:widths: 25 40 35
|
|
105
|
+
:header-rows: 1
|
|
106
|
+
|
|
107
|
+
* - Variant
|
|
108
|
+
- Equation
|
|
109
|
+
- Best For
|
|
110
|
+
* - **SABER-Anchored**
|
|
111
|
+
- ``1 - (1-ASR@n)·(n/N)^α``
|
|
112
|
+
- Most cases (default)
|
|
113
|
+
* - **SABER-Plugin**
|
|
114
|
+
- ``1 - Γ(α+β)/Γ(β)·N^(-α)``
|
|
115
|
+
- When ASR@n unavailable
|
|
116
|
+
* - **SABER-Fitting**
|
|
117
|
+
- ``exp(-a·N^(-b))``
|
|
118
|
+
- Alternative scaling model
|
|
119
|
+
* - **Naive Baseline**
|
|
120
|
+
- ``(1/K) Σ (1-(1-θ)^N)``
|
|
121
|
+
- Comparison only
|
|
122
|
+
* - **Combinatorial**
|
|
123
|
+
- ``1 - (1/K) Σ (C(n-k,N)/C(n,N))``
|
|
124
|
+
- N ≤ min(n) only
|