haloguard 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- haloguard-0.1.1/.github/dependabot.yml +10 -0
- haloguard-0.1.1/.github/workflows/ci.yml +56 -0
- haloguard-0.1.1/.github/workflows/publish.yml +82 -0
- haloguard-0.1.1/.gitignore +15 -0
- haloguard-0.1.1/CHANGELOG.md +44 -0
- haloguard-0.1.1/LICENSE +21 -0
- haloguard-0.1.1/MANIFEST.in +6 -0
- haloguard-0.1.1/PKG-INFO +153 -0
- haloguard-0.1.1/README.md +113 -0
- haloguard-0.1.1/pyproject.toml +83 -0
- haloguard-0.1.1/scripts/export_onnx.py +115 -0
- haloguard-0.1.1/scripts/validate_scoring.py +84 -0
- haloguard-0.1.1/src/haloguard/__init__.py +29 -0
- haloguard-0.1.1/src/haloguard/__main__.py +6 -0
- haloguard-0.1.1/src/haloguard/cli/__init__.py +1 -0
- haloguard-0.1.1/src/haloguard/cli/__main__.py +6 -0
- haloguard-0.1.1/src/haloguard/cli/commands.py +81 -0
- haloguard-0.1.1/src/haloguard/core/__init__.py +22 -0
- haloguard-0.1.1/src/haloguard/core/config.py +33 -0
- haloguard-0.1.1/src/haloguard/core/exceptions.py +23 -0
- haloguard-0.1.1/src/haloguard/core/firewall.py +157 -0
- haloguard-0.1.1/src/haloguard/core/result.py +45 -0
- haloguard-0.1.1/src/haloguard/integrations/__init__.py +1 -0
- haloguard-0.1.1/src/haloguard/integrations/langchain_handler.py +72 -0
- haloguard-0.1.1/src/haloguard/integrations/llamaindex_handler.py +60 -0
- haloguard-0.1.1/src/haloguard/integrations/raw_wrappers.py +77 -0
- haloguard-0.1.1/src/haloguard/models/__init__.py +5 -0
- haloguard-0.1.1/src/haloguard/models/loader.py +84 -0
- haloguard-0.1.1/src/haloguard/models/registry.py +24 -0
- haloguard-0.1.1/src/haloguard/scorers/__init__.py +8 -0
- haloguard-0.1.1/src/haloguard/scorers/aggregator.py +36 -0
- haloguard-0.1.1/src/haloguard/scorers/base.py +34 -0
- haloguard-0.1.1/src/haloguard/scorers/consistency.py +54 -0
- haloguard-0.1.1/src/haloguard/scorers/entailment.py +87 -0
- haloguard-0.1.1/tests/conftest.py +23 -0
- haloguard-0.1.1/tests/golden_dataset/labeled_pairs.jsonl +15 -0
- haloguard-0.1.1/tests/integration/test_batch_async.py +39 -0
- haloguard-0.1.1/tests/integration/test_cli.py +71 -0
- haloguard-0.1.1/tests/integration/test_consistency.py +50 -0
- haloguard-0.1.1/tests/integration/test_entailment_golden.py +39 -0
- haloguard-0.1.1/tests/integration/test_integrations.py +89 -0
- haloguard-0.1.1/tests/unit/test_aggregator.py +46 -0
- haloguard-0.1.1/tests/unit/test_config.py +34 -0
- haloguard-0.1.1/tests/unit/test_firewall.py +23 -0
- haloguard-0.1.1/tests/unit/test_loader.py +29 -0
- haloguard-0.1.1/tests/unit/test_sentence_splitting.py +21 -0
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
branches: [main]
|
|
6
|
+
pull_request:
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
lint:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
- name: Install
|
|
20
|
+
run: pip install -e ".[dev]"
|
|
21
|
+
- name: Ruff
|
|
22
|
+
run: ruff check src tests scripts
|
|
23
|
+
- name: Mypy
|
|
24
|
+
run: mypy
|
|
25
|
+
- name: pip-audit
|
|
26
|
+
run: pip-audit
|
|
27
|
+
|
|
28
|
+
test:
|
|
29
|
+
strategy:
|
|
30
|
+
fail-fast: false
|
|
31
|
+
matrix:
|
|
32
|
+
os: [ubuntu-latest, macos-latest, windows-latest]
|
|
33
|
+
python-version: ["3.9", "3.10", "3.11", "3.12"]
|
|
34
|
+
runs-on: ${{ matrix.os }}
|
|
35
|
+
steps:
|
|
36
|
+
- uses: actions/checkout@v4
|
|
37
|
+
- uses: actions/setup-python@v5
|
|
38
|
+
with:
|
|
39
|
+
python-version: ${{ matrix.python-version }}
|
|
40
|
+
- name: Install
|
|
41
|
+
run: pip install -e ".[dev]"
|
|
42
|
+
- name: Test
|
|
43
|
+
run: pytest tests -v
|
|
44
|
+
|
|
45
|
+
install-smoke:
|
|
46
|
+
runs-on: ubuntu-latest
|
|
47
|
+
steps:
|
|
48
|
+
- uses: actions/checkout@v4
|
|
49
|
+
- uses: actions/setup-python@v5
|
|
50
|
+
with:
|
|
51
|
+
python-version: "3.12"
|
|
52
|
+
- name: Fresh-venv install smoke test
|
|
53
|
+
run: |
|
|
54
|
+
python -m venv /tmp/smoke
|
|
55
|
+
/tmp/smoke/bin/pip install .
|
|
56
|
+
/tmp/smoke/bin/haloguard version
|
|
@@ -0,0 +1,82 @@
|
|
|
1
|
+
name: Publish
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
tags:
|
|
6
|
+
- "v*"
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: actions/setup-python@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
- name: Build wheel and sdist
|
|
20
|
+
run: |
|
|
21
|
+
pip install build
|
|
22
|
+
python -m build
|
|
23
|
+
- uses: actions/upload-artifact@v4
|
|
24
|
+
with:
|
|
25
|
+
name: dist
|
|
26
|
+
path: dist/
|
|
27
|
+
|
|
28
|
+
test:
|
|
29
|
+
runs-on: ubuntu-latest
|
|
30
|
+
steps:
|
|
31
|
+
- uses: actions/checkout@v4
|
|
32
|
+
- uses: actions/setup-python@v5
|
|
33
|
+
with:
|
|
34
|
+
python-version: "3.12"
|
|
35
|
+
- name: Run test suite before publishing
|
|
36
|
+
run: |
|
|
37
|
+
pip install -e ".[dev]"
|
|
38
|
+
pytest tests -v
|
|
39
|
+
|
|
40
|
+
publish:
|
|
41
|
+
needs: [build, test]
|
|
42
|
+
runs-on: ubuntu-latest
|
|
43
|
+
environment: pypi
|
|
44
|
+
permissions:
|
|
45
|
+
id-token: write
|
|
46
|
+
steps:
|
|
47
|
+
- uses: actions/download-artifact@v4
|
|
48
|
+
with:
|
|
49
|
+
name: dist
|
|
50
|
+
path: dist/
|
|
51
|
+
- name: Publish to PyPI (trusted publishing)
|
|
52
|
+
uses: pypa/gh-action-pypi-publish@release/v1
|
|
53
|
+
|
|
54
|
+
github-release:
|
|
55
|
+
needs: [publish]
|
|
56
|
+
runs-on: ubuntu-latest
|
|
57
|
+
permissions:
|
|
58
|
+
contents: write
|
|
59
|
+
steps:
|
|
60
|
+
- uses: actions/checkout@v4
|
|
61
|
+
- uses: actions/download-artifact@v4
|
|
62
|
+
with:
|
|
63
|
+
name: dist
|
|
64
|
+
path: dist/
|
|
65
|
+
- name: Extract changelog section for this version
|
|
66
|
+
id: changelog
|
|
67
|
+
run: |
|
|
68
|
+
python - <<'EOF'
|
|
69
|
+
import re
|
|
70
|
+
from pathlib import Path
|
|
71
|
+
text = Path("CHANGELOG.md").read_text(encoding="utf-8")
|
|
72
|
+
version = "${{ github.ref_name }}".lstrip("v")
|
|
73
|
+
pattern = rf"## \[{re.escape(version)}\]\n(.*?)(?=\n## \[|\Z)"
|
|
74
|
+
match = re.search(pattern, text, flags=re.DOTALL)
|
|
75
|
+
body = match.group(1).strip() if match else "See CHANGELOG.md"
|
|
76
|
+
Path("release_notes.md").write_text(body + "\n", encoding="utf-8")
|
|
77
|
+
EOF
|
|
78
|
+
- name: Create GitHub Release
|
|
79
|
+
uses: softprops/action-gh-release@v2
|
|
80
|
+
with:
|
|
81
|
+
body_path: release_notes.md
|
|
82
|
+
files: dist/*
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes to this project will be documented in this file.
|
|
4
|
+
|
|
5
|
+
The format is based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
6
|
+
and this project adheres to [Semantic Versioning](https://semver.org/spec/v2.0.0.html).
|
|
7
|
+
|
|
8
|
+
## [Unreleased]
|
|
9
|
+
|
|
10
|
+
### Added
|
|
11
|
+
- Initial scaffold: `src/haloguard/` layout with `core/` and `scorers/` subpackages.
|
|
12
|
+
- Exception hierarchy (`HaloGuardError`, `ModelLoadError`, `InputTooLargeError`,
|
|
13
|
+
`InferenceTimeoutError`, `ConfigError`).
|
|
14
|
+
- `FirewallResult` frozen dataclass contract.
|
|
15
|
+
- Golden dataset of 15 hand-labeled (context, response) pairs used to validate
|
|
16
|
+
entailment scoring (`tests/golden_dataset/labeled_pairs.jsonl`).
|
|
17
|
+
- `Firewall.check()` / `acheck()` / `check_batch()` backed by ONNX Runtime (CPU),
|
|
18
|
+
with fail-open/fail-closed behaviour, input size caps, and an inference timeout.
|
|
19
|
+
- Consistency mode: context-free scoring by cross-checking the response's own
|
|
20
|
+
claims pairwise with the NLI model (batched inference).
|
|
21
|
+
- `FirewallInput` / `FirewallResult` contract shared by SDK, CLI, and hooks.
|
|
22
|
+
- Typer CLI (`haloguard check`, `haloguard version`) with exit codes
|
|
23
|
+
0 PASS / 1 FLAG / 2 BLOCK / 3 internal error and `--json` output.
|
|
24
|
+
- Integrations: LangChain callback handler, LlamaIndex-style query hook, and
|
|
25
|
+
raw client adapters (OpenAI / Anthropic / Ollama) via `guarded_call`.
|
|
26
|
+
- `models/` subpackage: pinned SHA256 manifest verified on every load, plus a
|
|
27
|
+
local cache dir (`platformdirs`) with `HALOGUARD_MODEL_DIR` override.
|
|
28
|
+
- `scripts/export_onnx.py` to rebuild the ONNX artifact from the source model.
|
|
29
|
+
- GitHub Actions CI (lint, type-check, 3 OS x Python 3.9-3.12 test matrix,
|
|
30
|
+
fresh-venv install smoke test, pip-audit) and PyPI trusted-publishing
|
|
31
|
+
release workflow gated on version tags.
|
|
32
|
+
|
|
33
|
+
### Changed
|
|
34
|
+
- Quantization strategy: naive dynamic INT8 quantization measurably hurt accuracy
|
|
35
|
+
(a borderline hallucination's risk dropped 0.84 -> 0.33, flipping it to PASS).
|
|
36
|
+
Switched to static uint8 quantization with graph preprocessing and calibration
|
|
37
|
+
data, which keeps golden-set verdicts identical to the torch reference.
|
|
38
|
+
- Runtime inference uses `onnxruntime` + `tokenizers` only; `sentence-transformers`
|
|
39
|
+
/ torch moved to the optional `export` extra.
|
|
40
|
+
|
|
41
|
+
### Notes
|
|
42
|
+
- The quantized artifact is ~161MB (fp32 export ~568MB), larger than the design
|
|
43
|
+
doc's ~35-90MB estimate: DeBERTa-v3's 128k-vocabulary embedding dominates the
|
|
44
|
+
file size. Install-size expectations in the doc should be revised accordingly.
|
haloguard-0.1.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Kunal Soyane
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
haloguard-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: haloguard
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Local-first hallucination firewall for LLM applications
|
|
5
|
+
Project-URL: Homepage, https://github.com/KunalSoyane/Halogaurd
|
|
6
|
+
Project-URL: Issues, https://github.com/KunalSoyane/Halogaurd/issues
|
|
7
|
+
Project-URL: Changelog, https://github.com/KunalSoyane/Halogaurd/blob/main/CHANGELOG.md
|
|
8
|
+
Author: Kunal Soyane
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: hallucination,llm,nli,onnx,rag,safety
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
21
|
+
Requires-Python: >=3.9
|
|
22
|
+
Requires-Dist: onnxruntime>=1.17
|
|
23
|
+
Requires-Dist: platformdirs>=4.0
|
|
24
|
+
Requires-Dist: pydantic>=2.7
|
|
25
|
+
Requires-Dist: tokenizers>=0.19
|
|
26
|
+
Requires-Dist: typer>=0.12
|
|
27
|
+
Provides-Extra: dev
|
|
28
|
+
Requires-Dist: mypy>=1.10; extra == 'dev'
|
|
29
|
+
Requires-Dist: pip-audit>=2.7; extra == 'dev'
|
|
30
|
+
Requires-Dist: pytest>=8.0; extra == 'dev'
|
|
31
|
+
Requires-Dist: ruff>=0.5; extra == 'dev'
|
|
32
|
+
Provides-Extra: export
|
|
33
|
+
Requires-Dist: optimum[onnxruntime]>=1.21; extra == 'export'
|
|
34
|
+
Requires-Dist: sentence-transformers<6,>=3.0; extra == 'export'
|
|
35
|
+
Provides-Extra: langchain
|
|
36
|
+
Requires-Dist: langchain-core>=0.2; extra == 'langchain'
|
|
37
|
+
Provides-Extra: llamaindex
|
|
38
|
+
Requires-Dist: llama-index-core>=0.10; extra == 'llamaindex'
|
|
39
|
+
Description-Content-Type: text/markdown
|
|
40
|
+
|
|
41
|
+
# HaloGuard
|
|
42
|
+
|
|
43
|
+
A local-first hallucination firewall for LLM applications. HaloGuard sits between an
|
|
44
|
+
LLM and the application consuming its output, scoring every response for hallucination
|
|
45
|
+
risk before it reaches a user. Everything runs on the caller's machine -- no prompt,
|
|
46
|
+
response, or context ever leaves the device.
|
|
47
|
+
|
|
48
|
+
## Scoring modes
|
|
49
|
+
|
|
50
|
+
- **Entailment mode** (RAG-style): scores whether the response is supported by supplied
|
|
51
|
+
source context, using an NLI cross-encoder (`DeBERTa-v3-small`, ONNX, quantized).
|
|
52
|
+
- **Consistency mode** (no context): scores whether the response is internally
|
|
53
|
+
consistent, cross-checking its own claims with the same NLI model.
|
|
54
|
+
|
|
55
|
+
Mode is selected automatically by input shape (`auto`), or set explicitly.
|
|
56
|
+
|
|
57
|
+
## Install
|
|
58
|
+
|
|
59
|
+
```bash
|
|
60
|
+
pip install haloguard
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
One-time model setup (builds the ONNX artifact into your local cache):
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install "haloguard[export]"
|
|
67
|
+
python scripts/export_onnx.py
|
|
68
|
+
```
|
|
69
|
+
|
|
70
|
+
## SDK quickstart
|
|
71
|
+
|
|
72
|
+
```python
|
|
73
|
+
from haloguard import Firewall
|
|
74
|
+
|
|
75
|
+
fw = Firewall() # threshold=0.7, block_threshold=0.9, mode="auto"
|
|
76
|
+
|
|
77
|
+
# Entailment mode: context supplied
|
|
78
|
+
result = fw.check(
|
|
79
|
+
prompt="Where is the Eiffel Tower?",
|
|
80
|
+
response="The Eiffel Tower is in Paris.",
|
|
81
|
+
context="The Eiffel Tower is located in Paris and was completed in 1889.",
|
|
82
|
+
)
|
|
83
|
+
print(result.verdict, result.score, result.reason)
|
|
84
|
+
|
|
85
|
+
# Consistency mode: no context
|
|
86
|
+
result = fw.check(
|
|
87
|
+
prompt="When is the meeting?",
|
|
88
|
+
response="The meeting is on Tuesday. The meeting is on Friday.",
|
|
89
|
+
)
|
|
90
|
+
```
|
|
91
|
+
|
|
92
|
+
Every check returns a `FirewallResult`. Also available: `acheck()` (async),
|
|
93
|
+
`check_batch()` (many items over one loaded session).
|
|
94
|
+
|
|
95
|
+
## CLI
|
|
96
|
+
|
|
97
|
+
```bash
|
|
98
|
+
haloguard check --prompt "..." --response "..." # consistency mode
|
|
99
|
+
haloguard check --prompt "..." --response "..." --context FILE # entailment mode
|
|
100
|
+
haloguard check ... --json # machine-readable
|
|
101
|
+
haloguard version
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
Exit codes: `0` PASS / `1` FLAG / `2` BLOCK / `3` internal error (incl. UNKNOWN).
|
|
105
|
+
|
|
106
|
+
## Framework hooks
|
|
107
|
+
|
|
108
|
+
```python
|
|
109
|
+
# LangChain: pip install "haloguard[langchain]"
|
|
110
|
+
from haloguard.integrations.langchain_handler import HaloGuardCallbackHandler
|
|
111
|
+
handler = HaloGuardCallbackHandler(context_provider=lambda _text: retrieved_context)
|
|
112
|
+
|
|
113
|
+
# LlamaIndex-style query responses (duck-typed, no hard dependency)
|
|
114
|
+
from haloguard.integrations.llamaindex_handler import HaloGuardQueryHook
|
|
115
|
+
result = HaloGuardQueryHook().check_response(query_response)
|
|
116
|
+
|
|
117
|
+
# Any client SDK: adapt to generate(prompt) -> str, then score
|
|
118
|
+
from haloguard.integrations.raw_wrappers import guarded_call, openai_generate
|
|
119
|
+
response, result = guarded_call(openai_generate(client), prompt, context=context)
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
## Verdicts
|
|
123
|
+
|
|
124
|
+
- `score` -- 0.0-1.0 hallucination risk (higher = more likely hallucinated)
|
|
125
|
+
- `verdict` -- `PASS` (risk < threshold), `FLAG` (threshold <= risk < block_threshold),
|
|
126
|
+
`BLOCK` (risk >= block_threshold), `UNKNOWN` (scoring failed, fail-open)
|
|
127
|
+
- `reason`, `mode_used`, `latency_ms`
|
|
128
|
+
|
|
129
|
+
`UNKNOWN` is the fail-open verdict returned when scoring itself fails and
|
|
130
|
+
`strict_mode=False` (the default). Set `strict_mode=True` to fail closed instead.
|
|
131
|
+
|
|
132
|
+
## Honest limitations
|
|
133
|
+
|
|
134
|
+
HaloGuard is **defense-in-depth, not a guarantee**. An adversarially crafted response
|
|
135
|
+
can read as entailed/consistent to any NLI model while still being false. The measured
|
|
136
|
+
false-negative rate on the golden benchmark is the real accuracy statement; treat
|
|
137
|
+
HaloGuard as one layer in a safety stack, not the only one.
|
|
138
|
+
|
|
139
|
+
## Development
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
pip install -e ".[dev]"
|
|
143
|
+
pytest tests -v # unit tests always; integration tests need model artifacts
|
|
144
|
+
ruff check src tests scripts
|
|
145
|
+
mypy
|
|
146
|
+
```
|
|
147
|
+
|
|
148
|
+
Integration tests run real inference against `tests/golden_dataset/labeled_pairs.jsonl`
|
|
149
|
+
and are skipped automatically when model artifacts are absent.
|
|
150
|
+
|
|
151
|
+
## License
|
|
152
|
+
|
|
153
|
+
MIT
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
# HaloGuard
|
|
2
|
+
|
|
3
|
+
A local-first hallucination firewall for LLM applications. HaloGuard sits between an
|
|
4
|
+
LLM and the application consuming its output, scoring every response for hallucination
|
|
5
|
+
risk before it reaches a user. Everything runs on the caller's machine -- no prompt,
|
|
6
|
+
response, or context ever leaves the device.
|
|
7
|
+
|
|
8
|
+
## Scoring modes
|
|
9
|
+
|
|
10
|
+
- **Entailment mode** (RAG-style): scores whether the response is supported by supplied
|
|
11
|
+
source context, using an NLI cross-encoder (`DeBERTa-v3-small`, ONNX, quantized).
|
|
12
|
+
- **Consistency mode** (no context): scores whether the response is internally
|
|
13
|
+
consistent, cross-checking its own claims with the same NLI model.
|
|
14
|
+
|
|
15
|
+
Mode is selected automatically by input shape (`auto`), or set explicitly.
|
|
16
|
+
|
|
17
|
+
## Install
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install haloguard
|
|
21
|
+
```
|
|
22
|
+
|
|
23
|
+
One-time model setup (builds the ONNX artifact into your local cache):
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install "haloguard[export]"
|
|
27
|
+
python scripts/export_onnx.py
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
## SDK quickstart
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
from haloguard import Firewall
|
|
34
|
+
|
|
35
|
+
fw = Firewall() # threshold=0.7, block_threshold=0.9, mode="auto"
|
|
36
|
+
|
|
37
|
+
# Entailment mode: context supplied
|
|
38
|
+
result = fw.check(
|
|
39
|
+
prompt="Where is the Eiffel Tower?",
|
|
40
|
+
response="The Eiffel Tower is in Paris.",
|
|
41
|
+
context="The Eiffel Tower is located in Paris and was completed in 1889.",
|
|
42
|
+
)
|
|
43
|
+
print(result.verdict, result.score, result.reason)
|
|
44
|
+
|
|
45
|
+
# Consistency mode: no context
|
|
46
|
+
result = fw.check(
|
|
47
|
+
prompt="When is the meeting?",
|
|
48
|
+
response="The meeting is on Tuesday. The meeting is on Friday.",
|
|
49
|
+
)
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
Every check returns a `FirewallResult`. Also available: `acheck()` (async),
|
|
53
|
+
`check_batch()` (many items over one loaded session).
|
|
54
|
+
|
|
55
|
+
## CLI
|
|
56
|
+
|
|
57
|
+
```bash
|
|
58
|
+
haloguard check --prompt "..." --response "..." # consistency mode
|
|
59
|
+
haloguard check --prompt "..." --response "..." --context FILE # entailment mode
|
|
60
|
+
haloguard check ... --json # machine-readable
|
|
61
|
+
haloguard version
|
|
62
|
+
```
|
|
63
|
+
|
|
64
|
+
Exit codes: `0` PASS / `1` FLAG / `2` BLOCK / `3` internal error (incl. UNKNOWN).
|
|
65
|
+
|
|
66
|
+
## Framework hooks
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
# LangChain: pip install "haloguard[langchain]"
|
|
70
|
+
from haloguard.integrations.langchain_handler import HaloGuardCallbackHandler
|
|
71
|
+
handler = HaloGuardCallbackHandler(context_provider=lambda _text: retrieved_context)
|
|
72
|
+
|
|
73
|
+
# LlamaIndex-style query responses (duck-typed, no hard dependency)
|
|
74
|
+
from haloguard.integrations.llamaindex_handler import HaloGuardQueryHook
|
|
75
|
+
result = HaloGuardQueryHook().check_response(query_response)
|
|
76
|
+
|
|
77
|
+
# Any client SDK: adapt to generate(prompt) -> str, then score
|
|
78
|
+
from haloguard.integrations.raw_wrappers import guarded_call, openai_generate
|
|
79
|
+
response, result = guarded_call(openai_generate(client), prompt, context=context)
|
|
80
|
+
```
|
|
81
|
+
|
|
82
|
+
## Verdicts
|
|
83
|
+
|
|
84
|
+
- `score` -- 0.0-1.0 hallucination risk (higher = more likely hallucinated)
|
|
85
|
+
- `verdict` -- `PASS` (risk < threshold), `FLAG` (threshold <= risk < block_threshold),
|
|
86
|
+
`BLOCK` (risk >= block_threshold), `UNKNOWN` (scoring failed, fail-open)
|
|
87
|
+
- `reason`, `mode_used`, `latency_ms`
|
|
88
|
+
|
|
89
|
+
`UNKNOWN` is the fail-open verdict returned when scoring itself fails and
|
|
90
|
+
`strict_mode=False` (the default). Set `strict_mode=True` to fail closed instead.
|
|
91
|
+
|
|
92
|
+
## Honest limitations
|
|
93
|
+
|
|
94
|
+
HaloGuard is **defense-in-depth, not a guarantee**. An adversarially crafted response
|
|
95
|
+
can read as entailed/consistent to any NLI model while still being false. The measured
|
|
96
|
+
false-negative rate on the golden benchmark is the real accuracy statement; treat
|
|
97
|
+
HaloGuard as one layer in a safety stack, not the only one.
|
|
98
|
+
|
|
99
|
+
## Development
|
|
100
|
+
|
|
101
|
+
```bash
|
|
102
|
+
pip install -e ".[dev]"
|
|
103
|
+
pytest tests -v # unit tests always; integration tests need model artifacts
|
|
104
|
+
ruff check src tests scripts
|
|
105
|
+
mypy
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Integration tests run real inference against `tests/golden_dataset/labeled_pairs.jsonl`
|
|
109
|
+
and are skipped automatically when model artifacts are absent.
|
|
110
|
+
|
|
111
|
+
## License
|
|
112
|
+
|
|
113
|
+
MIT
|
|
@@ -0,0 +1,83 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "haloguard"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "Local-first hallucination firewall for LLM applications"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
requires-python = ">=3.9"
|
|
12
|
+
authors = [{ name = "Kunal Soyane" }]
|
|
13
|
+
keywords = ["llm", "hallucination", "nli", "rag", "safety", "onnx"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Programming Language :: Python :: 3",
|
|
19
|
+
"Programming Language :: Python :: 3.9",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
24
|
+
]
|
|
25
|
+
dependencies = [
|
|
26
|
+
"onnxruntime>=1.17",
|
|
27
|
+
"tokenizers>=0.19",
|
|
28
|
+
"platformdirs>=4.0",
|
|
29
|
+
"pydantic>=2.7",
|
|
30
|
+
"typer>=0.12",
|
|
31
|
+
]
|
|
32
|
+
|
|
33
|
+
[project.optional-dependencies]
|
|
34
|
+
export = [
|
|
35
|
+
"optimum[onnxruntime]>=1.21",
|
|
36
|
+
"sentence-transformers>=3.0,<6",
|
|
37
|
+
]
|
|
38
|
+
langchain = ["langchain-core>=0.2"]
|
|
39
|
+
llamaindex = ["llama-index-core>=0.10"]
|
|
40
|
+
dev = [
|
|
41
|
+
"pytest>=8.0",
|
|
42
|
+
"ruff>=0.5",
|
|
43
|
+
"mypy>=1.10",
|
|
44
|
+
"pip-audit>=2.7",
|
|
45
|
+
]
|
|
46
|
+
|
|
47
|
+
[project.scripts]
|
|
48
|
+
haloguard = "haloguard.cli.commands:app"
|
|
49
|
+
|
|
50
|
+
[project.urls]
|
|
51
|
+
Homepage = "https://github.com/KunalSoyane/Halogaurd"
|
|
52
|
+
Issues = "https://github.com/KunalSoyane/Halogaurd/issues"
|
|
53
|
+
Changelog = "https://github.com/KunalSoyane/Halogaurd/blob/main/CHANGELOG.md"
|
|
54
|
+
|
|
55
|
+
[tool.hatch.build.targets.wheel]
|
|
56
|
+
packages = ["src/haloguard"]
|
|
57
|
+
|
|
58
|
+
[tool.pytest.ini_options]
|
|
59
|
+
testpaths = ["tests"]
|
|
60
|
+
markers = [
|
|
61
|
+
"integration: tests that run real model inference against the golden dataset",
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
[tool.ruff]
|
|
65
|
+
src = ["src", "tests"]
|
|
66
|
+
line-length = 100
|
|
67
|
+
|
|
68
|
+
[tool.ruff.lint]
|
|
69
|
+
select = ["E", "F", "I", "UP", "B"]
|
|
70
|
+
|
|
71
|
+
[tool.ruff.lint.flake8-bugbear]
|
|
72
|
+
extend-immutable-calls = ["typer.Option", "typer.Argument"]
|
|
73
|
+
|
|
74
|
+
[tool.ruff.lint.per-file-ignores]
|
|
75
|
+
"scripts/*" = ["E501"]
|
|
76
|
+
|
|
77
|
+
[tool.mypy]
|
|
78
|
+
python_version = "3.12"
|
|
79
|
+
mypy_path = "src"
|
|
80
|
+
packages = ["haloguard"]
|
|
81
|
+
warn_unused_ignores = true
|
|
82
|
+
check_untyped_defs = true
|
|
83
|
+
ignore_missing_imports = true
|