saber-risk 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ version: 2
2
+ updates:
3
+ - package-ecosystem: "pip"
4
+ directory: "/"
5
+ schedule:
6
+ interval: "weekly"
@@ -0,0 +1,35 @@
1
+ name: "CodeQL"
2
+
3
+ on:
4
+ push:
5
+ branches: [ main ]
6
+ pull_request:
7
+ branches: [ main ]
8
+ schedule:
9
+ - cron: '0 0 * * 0' # Weekly on Sunday
10
+
11
+ jobs:
12
+ analyze:
13
+ name: Analyze
14
+ runs-on: ubuntu-latest
15
+ permissions:
16
+ actions: read
17
+ contents: read
18
+ security-events: write
19
+
20
+ strategy:
21
+ fail-fast: false
22
+ matrix:
23
+ language: [ 'python' ]
24
+
25
+ steps:
26
+ - name: Checkout repository
27
+ uses: actions/checkout@v4
28
+
29
+ - name: Initialize CodeQL
30
+ uses: github/codeql-action/init@v3
31
+ with:
32
+ languages: ${{ matrix.language }}
33
+
34
+ - name: Perform CodeQL Analysis
35
+ uses: github/codeql-action/analyze@v3
@@ -0,0 +1,116 @@
1
+ # Byte-compiled / optimized / DLL files
2
+ __pycache__/
3
+ *.py[cod]
4
+ *$py.class
5
+
6
+ # C extensions
7
+ *.so
8
+
9
+ # Distribution / packaging
10
+ .Python
11
+ build/
12
+ develop-eggs/
13
+ dist/
14
+ downloads/
15
+ eggs/
16
+ .eggs/
17
+ lib/
18
+ lib64/
19
+ parts/
20
+ sdist/
21
+ var/
22
+ wheels/
23
+ *.egg-info/
24
+ .installed.cfg
25
+ *.egg
26
+ MANIFEST
27
+ .DS_Store
28
+
29
+ # PyInstaller
30
+ # Usually these files are written by a python script from a template
31
+ # before PyInstaller builds the exe, so as to inject date/other infos into it.
32
+ *.manifest
33
+ *.spec
34
+
35
+ # Installer logs
36
+ pip-log.txt
37
+ pip-delete-this-directory.txt
38
+
39
+ # Unit test / coverage reports
40
+ htmlcov/
41
+ .tox/
42
+ .coverage
43
+ .coverage.*
44
+ .cache
45
+ nosetests.xml
46
+ coverage.xml
47
+ *.cover
48
+ .hypothesis/
49
+ .pytest_cache/
50
+
51
+ # Translations
52
+ *.mo
53
+ *.pot
54
+
55
+ # Django stuff:
56
+ *.log
57
+ local_settings.py
58
+ db.sqlite3
59
+
60
+ # Flask stuff:
61
+ instance/
62
+ .webassets-cache
63
+
64
+ # Scrapy stuff:
65
+ .scrapy
66
+
67
+ # Sphinx documentation
68
+ docs/_build/
69
+
70
+ # PyBuilder
71
+ target/
72
+
73
+ # Jupyter Notebook
74
+ .ipynb_checkpoints
75
+
76
+ # pyenv
77
+ .python-version
78
+
79
+ # celery beat schedule file
80
+ celerybeat-schedule
81
+
82
+ # SageMath parsed files
83
+ *.sage.py
84
+
85
+ # Environments
86
+ .env
87
+ .venv
88
+ env/
89
+ venv/
90
+ ENV/
91
+ env.bak/
92
+ venv.bak/
93
+
94
+ # Spyder project settings
95
+ .spyderproject
96
+ .spyproject
97
+
98
+ # Rope project settings
99
+ .ropeproject
100
+
101
+ # mkdocs documentation
102
+ /site
103
+
104
+ # mypy
105
+ .mypy_cache/
106
+
107
+ # IDE pycharm
108
+ .idea/
109
+
110
+ .pt_description_history
111
+ .git-credentials
112
+ pt_bert/philly
113
+ .vs
114
+ *.pyproj
115
+ *pyc
116
+ /.vscode
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) Microsoft Corporation.
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE
@@ -0,0 +1,165 @@
1
+ Metadata-Version: 2.5
2
+ Name: saber-risk
3
+ Version: 0.2.0
4
+ Summary: SABER: Scaling-Aware Best-of-N Estimation of Risk for LLM safety evaluation
5
+ Project-URL: Homepage, https://github.com/microsoft/saber
6
+ Project-URL: Documentation, https://github.com/microsoft/saber#readme
7
+ Project-URL: Repository, https://github.com/microsoft/saber
8
+ Author: Xiaodong Liu, Mingqian Feng
9
+ License: MIT
10
+ License-File: LICENSE
11
+ Keywords: adversarial,best-of-n,jailbreak,llm,risk-estimation,safety
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Intended Audience :: Science/Research
14
+ Classifier: License :: OSI Approved :: MIT License
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
21
+ Requires-Python: >=3.9
22
+ Requires-Dist: numpy>=1.20.0
23
+ Requires-Dist: scipy>=1.7.0
24
+ Provides-Extra: dev
25
+ Requires-Dist: mypy>=1.0.0; extra == 'dev'
26
+ Requires-Dist: pytest-cov>=4.0.0; extra == 'dev'
27
+ Requires-Dist: pytest>=7.0.0; extra == 'dev'
28
+ Description-Content-Type: text/markdown
29
+
30
+ # SABER
31
+
32
+ **S**caling-**A**ware **B**est-of-N **E**stimation of **R**isk
33
+
34
+ A Python package for predicting large-scale adversarial risk in Large Language Models under Best-of-N sampling.
35
+
36
+ **Paper**: https://arxiv.org/pdf/2601.22636
37
+
38
+ [![Python 3.9+](https://img.shields.io/badge/python-3.9+-blue.svg)](https://www.python.org/downloads/)
39
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
40
+
41
+ ## Overview
42
+
43
+ Standard LLM safety evaluations use single-shot (ASR@1) metrics, but real attackers can exploit parallel sampling to repeatedly probe models. SABER provides a principled statistical framework to:
44
+
45
+ - **Predict** ASR@N at large budgets from small measurements
46
+ - **Estimate** how many attempts are needed to reach a target success rate
47
+ - **Quantify** uncertainty in adversarial risk predictions
48
+
49
+ ![SABER Method Overview](https://raw.githubusercontent.com/microsoft/saber/main/figures/saber_method.png)
50
+
51
+ ### Key Insight
52
+
53
+ Attack success rates scale according to a power law governed by the Beta distribution of per-query vulnerabilities:
54
+
55
+ ```
56
+ ASR@N ≈ 1 - Γ(α+β)/Γ(β) · N^(-α)
57
+ ```
58
+
59
+ The parameter **α** controls how fast risk amplifies with more attempts.
60
+
61
+ ## Installation
62
+
63
+ ```bash
64
+ pip install saber-risk
65
+ ```
66
+
67
+ Or from source:
68
+
69
+ ```bash
70
+ git clone https://github.com/microsoft/saber
71
+ cd saber
72
+ pip install -e .
73
+ ```
74
+
75
+ ## Quick Start
76
+
77
+ ```python
78
+ import numpy as np
79
+ from saber import SABER
80
+
81
+ # Your jailbreak evaluation data:
82
+ # k[i] = number of successful jailbreaks for query i
83
+ # n[i] = number of attempts for query i
84
+ k = np.array([3, 5, 0, 2, 8, 1, 4, 0, 6, 2])
85
+ n = 100 # 100 attempts per query
86
+
87
+ # Fit and predict
88
+ model = SABER()
89
+ model.fit(k, n)
90
+
91
+ # Predict ASR at N=1000 attempts
92
+ result = model.predict(N=1000)
93
+ print(f"ASR@1000 = {result.asr:.2%}")
94
+
95
+ # With confidence interval
96
+ result = model.predict(N=1000, confidence=0.95)
97
+ print(f"ASR@1000 = {result.asr:.2%} [{result.ci_lower:.2%}, {result.ci_upper:.2%}]")
98
+ ```
99
+
100
+ ## Core Usage
101
+
102
+ ```python
103
+ from saber import SABER
104
+
105
+ # 1. Collect jailbreak data
106
+ # Run n attempts per query, count successes k
107
+ k = [...] # successes per query
108
+ n = 100 # trials per query (or array for heterogeneous budgets)
109
+
110
+ # 2. Fit the model
111
+ model = SABER()
112
+ model.fit(k, n)
113
+
114
+ # 3. Predict ASR at target budget
115
+ asr_1000 = model.predict(N=1000).asr
116
+
117
+ # Budget estimation
118
+ result = model.budget_for_asr(target=0.95)
119
+ print(f"Need {result.budget:.0f} attempts for 95% ASR")
120
+
121
+ # Fluent API
122
+ asr = SABER().fit(k, n).predict(1000).asr
123
+ ```
124
+
125
+ ## Documentation
126
+
127
+ Full documentation is available in the `docs/` directory. To build:
128
+
129
+ ```bash
130
+ cd docs
131
+ pip install -r requirements.txt
132
+ make html
133
+ ```
134
+
135
+ - **[Quick Start](https://github.com/microsoft/saber/blob/main/docs/quickstart.rst)** - Getting started guide
136
+ - **[API Reference](https://github.com/microsoft/saber/blob/main/docs/api_reference.rst)** - Complete API documentation
137
+ - **[Advanced Usage](https://github.com/microsoft/saber/blob/main/docs/advanced_usage.rst)** - Model selection, scaling curves, low-level API
138
+
139
+ ## Citation
140
+
141
+ If you use SABER in your research, please cite:
142
+
143
+ ```bibtex
144
+ @misc{feng2026statisticalestimationadversarialrisk,
145
+ title={Statistical Estimation of Adversarial Risk in Large Language Models under Best-of-N Sampling},
146
+ author={Mingqian Feng and Xiaodong Liu and Weiwei Yang and Chenliang Xu and Christopher White and Jianfeng Gao},
147
+ year={2026},
148
+ eprint={2601.22636},
149
+ archivePrefix={arXiv},
150
+ primaryClass={cs.AI},
151
+ url={https://arxiv.org/abs/2601.22636},
152
+ }
153
+ ```
154
+
155
+ ## Contact
156
+
157
+ For any questions regarding the package or paper, feel free to reach out to:
158
+
159
+ - **Mingqian Feng** - mfeng7@ur.rochester.edu
160
+ - **Xiaodong Liu** - xiaodl@microsoft.com
161
+ - **Weiwei Yang** - weiwei.yang@microsoft.com
162
+
163
+ ## License
164
+
165
+ MIT License - see [LICENSE](https://github.com/microsoft/saber/blob/main/LICENSE) for details.
@@ -0,0 +1,136 @@
1
+ # SABER
2
+
3
+ **S**caling-**A**ware **B**est-of-N **E**stimation of **R**isk
4
+
5
+ A Python package for predicting large-scale adversarial risk in Large Language Models under Best-of-N sampling.
6
+
7
+ **Paper**: https://arxiv.org/pdf/2601.22636
8
+
9
+ [![Python 3.9+](https://img.shields.io/badge/python-3.9+-blue.svg)](https://www.python.org/downloads/)
10
+ [![License: MIT](https://img.shields.io/badge/License-MIT-yellow.svg)](https://opensource.org/licenses/MIT)
11
+
12
+ ## Overview
13
+
14
+ Standard LLM safety evaluations use single-shot (ASR@1) metrics, but real attackers can exploit parallel sampling to repeatedly probe models. SABER provides a principled statistical framework to:
15
+
16
+ - **Predict** ASR@N at large budgets from small measurements
17
+ - **Estimate** how many attempts are needed to reach a target success rate
18
+ - **Quantify** uncertainty in adversarial risk predictions
19
+
20
+ ![SABER Method Overview](https://raw.githubusercontent.com/microsoft/saber/main/figures/saber_method.png)
21
+
22
+ ### Key Insight
23
+
24
+ Attack success rates scale according to a power law governed by the Beta distribution of per-query vulnerabilities:
25
+
26
+ ```
27
+ ASR@N ≈ 1 - Γ(α+β)/Γ(β) · N^(-α)
28
+ ```
29
+
30
+ The parameter **α** controls how fast risk amplifies with more attempts.
31
+
32
+ ## Installation
33
+
34
+ ```bash
35
+ pip install saber-risk
36
+ ```
37
+
38
+ Or from source:
39
+
40
+ ```bash
41
+ git clone https://github.com/microsoft/saber
42
+ cd saber
43
+ pip install -e .
44
+ ```
45
+
46
+ ## Quick Start
47
+
48
+ ```python
49
+ import numpy as np
50
+ from saber import SABER
51
+
52
+ # Your jailbreak evaluation data:
53
+ # k[i] = number of successful jailbreaks for query i
54
+ # n[i] = number of attempts for query i
55
+ k = np.array([3, 5, 0, 2, 8, 1, 4, 0, 6, 2])
56
+ n = 100 # 100 attempts per query
57
+
58
+ # Fit and predict
59
+ model = SABER()
60
+ model.fit(k, n)
61
+
62
+ # Predict ASR at N=1000 attempts
63
+ result = model.predict(N=1000)
64
+ print(f"ASR@1000 = {result.asr:.2%}")
65
+
66
+ # With confidence interval
67
+ result = model.predict(N=1000, confidence=0.95)
68
+ print(f"ASR@1000 = {result.asr:.2%} [{result.ci_lower:.2%}, {result.ci_upper:.2%}]")
69
+ ```
70
+
71
+ ## Core Usage
72
+
73
+ ```python
74
+ from saber import SABER
75
+
76
+ # 1. Collect jailbreak data
77
+ # Run n attempts per query, count successes k
78
+ k = [...] # successes per query
79
+ n = 100 # trials per query (or array for heterogeneous budgets)
80
+
81
+ # 2. Fit the model
82
+ model = SABER()
83
+ model.fit(k, n)
84
+
85
+ # 3. Predict ASR at target budget
86
+ asr_1000 = model.predict(N=1000).asr
87
+
88
+ # Budget estimation
89
+ result = model.budget_for_asr(target=0.95)
90
+ print(f"Need {result.budget:.0f} attempts for 95% ASR")
91
+
92
+ # Fluent API
93
+ asr = SABER().fit(k, n).predict(1000).asr
94
+ ```
95
+
96
+ ## Documentation
97
+
98
+ Full documentation is available in the `docs/` directory. To build:
99
+
100
+ ```bash
101
+ cd docs
102
+ pip install -r requirements.txt
103
+ make html
104
+ ```
105
+
106
+ - **[Quick Start](https://github.com/microsoft/saber/blob/main/docs/quickstart.rst)** - Getting started guide
107
+ - **[API Reference](https://github.com/microsoft/saber/blob/main/docs/api_reference.rst)** - Complete API documentation
108
+ - **[Advanced Usage](https://github.com/microsoft/saber/blob/main/docs/advanced_usage.rst)** - Model selection, scaling curves, low-level API
109
+
110
+ ## Citation
111
+
112
+ If you use SABER in your research, please cite:
113
+
114
+ ```bibtex
115
+ @misc{feng2026statisticalestimationadversarialrisk,
116
+ title={Statistical Estimation of Adversarial Risk in Large Language Models under Best-of-N Sampling},
117
+ author={Mingqian Feng and Xiaodong Liu and Weiwei Yang and Chenliang Xu and Christopher White and Jianfeng Gao},
118
+ year={2026},
119
+ eprint={2601.22636},
120
+ archivePrefix={arXiv},
121
+ primaryClass={cs.AI},
122
+ url={https://arxiv.org/abs/2601.22636},
123
+ }
124
+ ```
125
+
126
+ ## Contact
127
+
128
+ For any questions regarding the package or paper, feel free to reach out to:
129
+
130
+ - **Mingqian Feng** - mfeng7@ur.rochester.edu
131
+ - **Xiaodong Liu** - xiaodl@microsoft.com
132
+ - **Weiwei Yang** - weiwei.yang@microsoft.com
133
+
134
+ ## License
135
+
136
+ MIT License - see [LICENSE](https://github.com/microsoft/saber/blob/main/LICENSE) for details.
@@ -0,0 +1,14 @@
1
+ <!-- BEGIN MICROSOFT SECURITY.MD V1.0.0 BLOCK -->
2
+
3
+ ## Security
4
+
5
+ Microsoft takes the security of our software products and services seriously, which
6
+ includes all source code repositories in our GitHub organizations.
7
+
8
+ **Please do not report security vulnerabilities through public GitHub issues.**
9
+
10
+ For security reporting information, locations, contact information, and policies,
11
+ please review the latest guidance for Microsoft repositories at
12
+ [https://aka.ms/SECURITY.md](https://aka.ms/SECURITY.md).
13
+
14
+ <!-- END MICROSOFT SECURITY.MD BLOCK -->
@@ -0,0 +1,14 @@
1
+ # Minimal makefile for Sphinx documentation
2
+
3
+ SPHINXOPTS ?=
4
+ SPHINXBUILD ?= sphinx-build
5
+ SOURCEDIR = .
6
+ BUILDDIR = _build
7
+
8
+ help:
9
+ @$(SPHINXBUILD) -M help "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
10
+
11
+ .PHONY: help Makefile
12
+
13
+ %: Makefile
14
+ @$(SPHINXBUILD) -M $@ "$(SOURCEDIR)" "$(BUILDDIR)" $(SPHINXOPTS) $(O)
@@ -0,0 +1,124 @@
1
+ Advanced Usage
2
+ ==============
3
+
4
+ Heterogeneous Budgets
5
+ ---------------------
6
+
7
+ SABER supports different trial counts per query:
8
+
9
+ .. code-block:: python
10
+
11
+ import numpy as np
12
+ from saber import SABER
13
+
14
+ k = np.array([3, 5, 0, 2, 8])
15
+ n = np.array([50, 100, 75, 100, 200]) # Different budgets
16
+
17
+ model = SABER()
18
+ model.fit(k, n)
19
+
20
+ Model Selection
21
+ ---------------
22
+
23
+ .. code-block:: python
24
+
25
+ # SABER-Anchored (default, recommended)
26
+ model = SABER(method="anchored")
27
+
28
+ # SABER-Plugin (uses β directly)
29
+ model = SABER(method="plugin")
30
+
31
+ # Override for specific prediction
32
+ result = model.predict(N=1000, method="naive")
33
+ result = model.predict(N=1000, method="combinatorial") # N must be <= min(n)
34
+ result = model.predict(N=1000, method="log_linear_curve") # auto-fits curve if not fitted
35
+
36
+ # Small-N Correction
37
+ result = model.predict(N=1000, method="anchored", correction=True)
38
+
39
+ Log-Linear Curve Fit
40
+ --------------------
41
+
42
+ The default fit is always the Beta distribution (α, β). You can additionally fit a log-linear curve:
43
+
44
+ .. code-block:: python
45
+
46
+ # Fit both distribution and curve
47
+ model = SABER()
48
+ model.fit(k, n, also_fit_curve=True)
49
+
50
+ # Or: fit only distribution; curve will be auto-fitted on first use
51
+ model.fit(k, n)
52
+ result = model.predict(N=1000, method="log_linear_curve")
53
+
54
+ Low-Level API
55
+ -------------
56
+
57
+ For advanced usage, access the underlying functions directly:
58
+
59
+ .. code-block:: python
60
+
61
+ from saber import fit_beta_binomial, predict_anchored, estimate_budget
62
+
63
+ # Fit distribution
64
+ alpha, beta = fit_beta_binomial(k, n)
65
+
66
+ # Predict ASR
67
+ asr = predict_anchored(N=1000, n=100, asr_at_n=0.73, alpha=alpha)
68
+
69
+ # Estimate budget
70
+ budget = estimate_budget(target_asr=0.95, n=100, asr_at_n=0.73, alpha=alpha)
71
+
72
+ Scaling Curves
73
+ --------------
74
+
75
+ Generate data for plotting:
76
+
77
+ .. code-block:: python
78
+
79
+ import matplotlib.pyplot as plt
80
+
81
+ curve = model.scaling_curve(
82
+ include_naive=True,
83
+ include_combinatorial=True,
84
+ correction=False
85
+ )
86
+
87
+ plt.figure(figsize=(8, 5))
88
+ plt.plot(curve['N'], curve['asr_anchored'] * 100, label='SABER-Anchored')
89
+ plt.plot(curve['N'], curve['asr_plugin'] * 100, label='SABER-Plugin')
90
+ if 'asr_log_linear_curve' in curve:
91
+ plt.plot(curve['N'], curve['asr_log_linear_curve'] * 100, label='Log-linear curve')
92
+ plt.plot(curve['N'], curve['asr_naive'] * 100, '--', label='Naive Baseline')
93
+ plt.xscale('log')
94
+ plt.xlabel('Number of Attempts (N)')
95
+ plt.ylabel('ASR@N (%)')
96
+ plt.legend()
97
+ plt.title('ASR Scaling Curve')
98
+ plt.show()
99
+
100
+ Estimator Variants
101
+ ------------------
102
+
103
+ .. list-table::
104
+ :widths: 25 40 35
105
+ :header-rows: 1
106
+
107
+ * - Variant
108
+ - Equation
109
+ - Best For
110
+ * - **SABER-Anchored**
111
+ - ``1 - (1-ASR@n)·(n/N)^α``
112
+ - Most cases (default)
113
+ * - **SABER-Plugin**
114
+ - ``1 - Γ(α+β)/Γ(β)·N^(-α)``
115
+ - When ASR@n unavailable
116
+ * - **SABER-Fitting**
117
+ - ``exp(-a·N^(-b))``
118
+ - Alternative scaling model
119
+ * - **Naive Baseline**
120
+ - ``(1/K) Σ (1-(1-θ)^N)``
121
+ - Comparison only
122
+ * - **Combinatorial**
123
+ - ``1 - (1/K) Σ (C(n-k,N)/C(n,N))``
124
+ - N ≤ min(n) only