noirebox 0.3.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- noirebox-0.3.0/.gitignore +27 -0
- noirebox-0.3.0/CHANGELOG.md +68 -0
- noirebox-0.3.0/LICENSE +21 -0
- noirebox-0.3.0/Makefile +53 -0
- noirebox-0.3.0/PKG-INFO +405 -0
- noirebox-0.3.0/README.fr.md +387 -0
- noirebox-0.3.0/README.md +372 -0
- noirebox-0.3.0/docs/.nojekyll +0 -0
- noirebox-0.3.0/docs/ADRs.md +116 -0
- noirebox-0.3.0/docs/SPECS.md +128 -0
- noirebox-0.3.0/docs/THREAT-MODEL.md +40 -0
- noirebox-0.3.0/docs/VULGARISATION.md +245 -0
- noirebox-0.3.0/docs/architecture.svg +72 -0
- noirebox-0.3.0/docs/bmc_qr.png +0 -0
- noirebox-0.3.0/docs/fr.html +345 -0
- noirebox-0.3.0/docs/index.html +339 -0
- noirebox-0.3.0/docs/openapi.json +499 -0
- noirebox-0.3.0/docs/script.js +109 -0
- noirebox-0.3.0/docs/style.css +347 -0
- noirebox-0.3.0/models/detector.joblib +0 -0
- noirebox-0.3.0/models/detector_en.joblib +0 -0
- noirebox-0.3.0/models/metrics.json +47 -0
- noirebox-0.3.0/models/metrics_en.json +48 -0
- noirebox-0.3.0/noirebox/__init__.py +1 -0
- noirebox-0.3.0/noirebox/anchors.py +87 -0
- noirebox-0.3.0/noirebox/attestation.py +44 -0
- noirebox-0.3.0/noirebox/auth.py +137 -0
- noirebox-0.3.0/noirebox/chain.py +182 -0
- noirebox-0.3.0/noirebox/cli.py +40 -0
- noirebox-0.3.0/noirebox/client.py +45 -0
- noirebox-0.3.0/noirebox/dashboard.py +179 -0
- noirebox-0.3.0/noirebox/guardrail.py +105 -0
- noirebox-0.3.0/noirebox/llm_agent.py +125 -0
- noirebox-0.3.0/noirebox/main.py +202 -0
- noirebox-0.3.0/noirebox/mcp_server.py +132 -0
- noirebox-0.3.0/noirebox/merkle.py +107 -0
- noirebox-0.3.0/noirebox/ml_guardrail.py +104 -0
- noirebox-0.3.0/noirebox/pdf_export.py +163 -0
- noirebox-0.3.0/noirebox/schemas.py +30 -0
- noirebox-0.3.0/noirebox/store.py +100 -0
- noirebox-0.3.0/pyproject.toml +63 -0
- noirebox-0.3.0/requirements.txt +10 -0
- noirebox-0.3.0/tests/conftest.py +4 -0
- noirebox-0.3.0/tests/test_anchors.py +135 -0
- noirebox-0.3.0/tests/test_api.py +99 -0
- noirebox-0.3.0/tests/test_attestation.py +48 -0
- noirebox-0.3.0/tests/test_auth_pdf.py +145 -0
- noirebox-0.3.0/tests/test_chain.py +116 -0
- noirebox-0.3.0/tests/test_client.py +67 -0
- noirebox-0.3.0/tests/test_dashboard.py +36 -0
- noirebox-0.3.0/tests/test_guardrail.py +71 -0
- noirebox-0.3.0/tests/test_llm_agent.py +53 -0
- noirebox-0.3.0/tests/test_mcp.py +96 -0
- noirebox-0.3.0/tests/test_merkle.py +131 -0
- noirebox-0.3.0/tests/test_ml_guardrail.py +107 -0
- noirebox-0.3.0/tests/test_store.py +51 -0
- noirebox-0.3.0/tests/test_verifier.py +70 -0
- noirebox-0.3.0/verifier/verifier.py +149 -0
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
.venv/
|
|
3
|
+
__pycache__/
|
|
4
|
+
*.pyc
|
|
5
|
+
.pytest_cache/
|
|
6
|
+
.ruff_cache/
|
|
7
|
+
dist/
|
|
8
|
+
*.egg-info/
|
|
9
|
+
|
|
10
|
+
# Runtime artifacts (database, private key) — never committed.
|
|
11
|
+
# The JSONL datasets are not versioned: they are reproducible via
|
|
12
|
+
# `make train` / `make train-en` (seed 42, identical bytes everywhere).
|
|
13
|
+
data/*.db
|
|
14
|
+
data/*.db-*
|
|
15
|
+
data/*.key
|
|
16
|
+
data/*.jsonl
|
|
17
|
+
models/*.joblib.tmp
|
|
18
|
+
export*.json
|
|
19
|
+
|
|
20
|
+
# Local TSA crypto material (root + leaf + tokens)
|
|
21
|
+
tsa/material/
|
|
22
|
+
|
|
23
|
+
# Local scan tooling
|
|
24
|
+
.mimosa/
|
|
25
|
+
|
|
26
|
+
# OS / editor junk
|
|
27
|
+
.DS_Store
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
Format based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
|
|
4
|
+
versioning according to [Semantic Versioning](https://semver.org/).
|
|
5
|
+
|
|
6
|
+
## [Unreleased]
|
|
7
|
+
|
|
8
|
+
### Planned
|
|
9
|
+
- Local judge model (`llama-guard3:1b` via Ollama) for ambiguous cases — ADR 001 stage 2
|
|
10
|
+
- Rotation of anchors across multiple TSAs (distribute trust)
|
|
11
|
+
- Prometheus + Grafana metrics
|
|
12
|
+
- HSM/KMS migration for the private key
|
|
13
|
+
|
|
14
|
+
## [0.3.0] - 2026-09-20
|
|
15
|
+
|
|
16
|
+
### Added
|
|
17
|
+
- **RFC 3161 anchoring** ([ADR 006](docs/ADRs.md)): `POST /api/v1/anchors`
|
|
18
|
+
seals the current chain head with a TSA (only the hash leaves — zero data);
|
|
19
|
+
token + certificate are journaled as an `anchor` event; the third-party
|
|
20
|
+
verifier checks the anchors (`openssl ts -verify`) and reports
|
|
21
|
+
`anchors_checked` — the threat model gap for the “insider with the key” is closed
|
|
22
|
+
- **Self-hosted TSA** ([ADR 007](docs/ADRs.md)): `make tsa` (OpenSSL,
|
|
23
|
+
root + leaf certificate chain, port 3318, free, offline); external/qualified
|
|
24
|
+
TSA via `NOIREBOX_TSA_URL`
|
|
25
|
+
- **Fleet Merkle anchoring**: `noirebox/merkle.py` — one TSA seal covers N journals
|
|
26
|
+
(Certificate Transparency pattern); inclusion proofs are ~log2(N) hashes and are
|
|
27
|
+
verifiable offline; the full tree is journaled as a `fleet_anchor` event (the hub
|
|
28
|
+
is itself a NoireBox) — `make demo-fleet` (3 boxes, 1 seal, 1 falsification)
|
|
29
|
+
- Full simplified explainer in `docs/VULGARISATION.md §9` (“dry paint” / “final layer”) — ready for infographics
|
|
30
|
+
|
|
31
|
+
## [0.2.0] - 2026-09-20
|
|
32
|
+
|
|
33
|
+
### Added
|
|
34
|
+
- **OAuth2 (JWT bearer) + rate limiting** ([ADR 004](docs/ADRs.md)):
|
|
35
|
+
`POST /api/v1/token` (client-credentials, constant-time secret comparison),
|
|
36
|
+
1-hour HS256 JWT, protection for writes and exports — verification routes remain open by design;
|
|
37
|
+
sliding window 60 req/min/client; explicitly enabled via `NOIREBOX_CLIENTS`
|
|
38
|
+
- **PDF attestation** ([ADR 005](docs/ADRs.md)): `GET /api/v1/attestation.pdf`,
|
|
39
|
+
A4 DPO-ready page with chain state, keys, and verification instructions
|
|
40
|
+
|
|
41
|
+
## [0.1.0] - 2026-09-19
|
|
42
|
+
|
|
43
|
+
### Added
|
|
44
|
+
- **Tamper-evident journal**: chained SHA-256 hash chain + Ed25519 signatures,
|
|
45
|
+
append-only SQLite storage, local verification and full export
|
|
46
|
+
(`GET /api/v1/export`) — standalone third-party verifier (`verifier/verifier.py`,
|
|
47
|
+
exit 0/1 with exact falsification location)
|
|
48
|
+
- **Signed attestation** of the current chain state (`GET /api/v1/attestation`)
|
|
49
|
+
- **Dual-engine guardrail**: regex heuristics for FR (4 attack families)
|
|
50
|
+
and trained ML micro-model (TF-IDF + logistic regression, 293 KB,
|
|
51
|
+
FR dataset of 5,400 versioned examples, `make train`) — selected via
|
|
52
|
+
`engine: regex|ml`
|
|
53
|
+
- **Real LLM scene** (`make demo-llm`): qwen2.5:0.5b via local Ollama,
|
|
54
|
+
attack succeeds without protection and fails with it — zero simulation
|
|
55
|
+
- **MCP server** (JSON-RPC stdio, no SDK): 4 agent tools
|
|
56
|
+
(`noirebox_scan`, `noirebox_log_event`, `noirebox_verify`, `noirebox_attestation`)
|
|
57
|
+
- **Client SDK** (`noirebox/client.py`) tested against a real uvicorn server
|
|
58
|
+
- One-shot demos: `make demo` (model → journal → auditor → attacker),
|
|
59
|
+
`make demo-mcp`
|
|
60
|
+
- **GitHub Actions CI**: retraining of the micro-model + 56 tests on every push
|
|
61
|
+
- Documentation: formal spec (`docs/SPECS.md`), threat model
|
|
62
|
+
(`docs/THREAT-MODEL.md`), ADRs (`docs/ADRs.md`), full vulgarization
|
|
63
|
+
(`docs/VULGARISATION.md`)
|
|
64
|
+
|
|
65
|
+
### Decisions
|
|
66
|
+
- ADR 001: staged guardrail (regex → ML → LLM judge as last resort)
|
|
67
|
+
- ADR 002: Llama Prompt Guard 2 (Meta) evaluated and rejected (language, binary output,
|
|
68
|
+
gated licensing, dependencies)
|
noirebox-0.3.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 NoireBox contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
noirebox-0.3.0/Makefile
ADDED
|
@@ -0,0 +1,53 @@
|
|
|
1
|
+
# NoireBox — shortcuts (make test, make demo, make serve…)
|
|
2
|
+
.PHONY: install test serve demo demo-mcp demo-llm demo-fleet demo-payout ollama-pull tsa dataset train train-en docker clean
|
|
3
|
+
|
|
4
|
+
install:
|
|
5
|
+
python3 -m venv .venv
|
|
6
|
+
.venv/bin/pip install -q --upgrade pip
|
|
7
|
+
.venv/bin/pip install -q -r requirements.txt
|
|
8
|
+
|
|
9
|
+
test:
|
|
10
|
+
.venv/bin/pytest -q
|
|
11
|
+
|
|
12
|
+
serve:
|
|
13
|
+
.venv/bin/uvicorn noirebox.main:app --host 127.0.0.1 --port 8768
|
|
14
|
+
|
|
15
|
+
demo:
|
|
16
|
+
.venv/bin/python demo/demo_live.py
|
|
17
|
+
|
|
18
|
+
demo-mcp:
|
|
19
|
+
.venv/bin/python demo/demo_mcp.py
|
|
20
|
+
|
|
21
|
+
demo-llm:
|
|
22
|
+
.venv/bin/python demo/demo_ollama.py
|
|
23
|
+
|
|
24
|
+
demo-fleet:
|
|
25
|
+
.venv/bin/python demo/demo_fleet.py
|
|
26
|
+
|
|
27
|
+
demo-payout:
|
|
28
|
+
.venv/bin/python demo/demo_payout.py
|
|
29
|
+
|
|
30
|
+
ollama-pull:
|
|
31
|
+
ollama pull qwen2.5:0.5b
|
|
32
|
+
|
|
33
|
+
tsa:
|
|
34
|
+
@chmod +x tsa/gen_tsa.sh && ./tsa/gen_tsa.sh tsa/material
|
|
35
|
+
.venv/bin/python tsa/tsa_server.py --material tsa/material --port 3318
|
|
36
|
+
|
|
37
|
+
dataset:
|
|
38
|
+
.venv/bin/python ml/gen_dataset.py
|
|
39
|
+
|
|
40
|
+
train:
|
|
41
|
+
.venv/bin/python ml/gen_dataset.py
|
|
42
|
+
.venv/bin/python ml/train.py
|
|
43
|
+
|
|
44
|
+
train-en:
|
|
45
|
+
.venv/bin/python ml/gen_dataset_en.py
|
|
46
|
+
.venv/bin/python ml/train.py --lang en
|
|
47
|
+
|
|
48
|
+
docker:
|
|
49
|
+
docker compose up --build
|
|
50
|
+
|
|
51
|
+
clean:
|
|
52
|
+
rm -rf data/*.db data/*.key .pytest_cache
|
|
53
|
+
find . -name __pycache__ -type d -exec rm -rf {} +
|
noirebox-0.3.0/PKG-INFO
ADDED
|
@@ -0,0 +1,405 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: noirebox
|
|
3
|
+
Version: 0.3.0
|
|
4
|
+
Summary: The flight data recorder for AI agents — hash-chained, signed journal with RFC 3161 anchoring and third-party-verifiable exports.
|
|
5
|
+
Project-URL: Homepage, https://github.com/slabbdev/noirebox
|
|
6
|
+
Project-URL: Documentation, https://github.com/slabbdev/noirebox#readme
|
|
7
|
+
Author: NoireBox contributors
|
|
8
|
+
License: MIT
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Keywords: ai-act,ai-agents,audit-trail,compliance,gdpr,guardrail,llm,rfc3161,tamper-evident
|
|
11
|
+
Classifier: Development Status :: 4 - Beta
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Topic :: Security
|
|
18
|
+
Classifier: Topic :: System :: Logging
|
|
19
|
+
Requires-Python: >=3.11
|
|
20
|
+
Requires-Dist: cryptography>=42.0
|
|
21
|
+
Requires-Dist: fastapi>=0.110
|
|
22
|
+
Requires-Dist: httpx>=0.27
|
|
23
|
+
Requires-Dist: joblib>=1.3
|
|
24
|
+
Requires-Dist: pydantic>=2.6
|
|
25
|
+
Requires-Dist: pyjwt>=2.8
|
|
26
|
+
Requires-Dist: reportlab>=4.0
|
|
27
|
+
Requires-Dist: scikit-learn>=1.4
|
|
28
|
+
Requires-Dist: uvicorn>=0.29
|
|
29
|
+
Provides-Extra: dev
|
|
30
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
31
|
+
Requires-Dist: ruff>=0.4; extra == 'dev'
|
|
32
|
+
Description-Content-Type: text/markdown
|
|
33
|
+
|
|
34
|
+
<div align="center">
|
|
35
|
+
|
|
36
|
+
🇬🇧 English · [🇫🇷 Français](README.fr.md)
|
|
37
|
+
|
|
38
|
+
# ⬛ NoireBox
|
|
39
|
+
|
|
40
|
+
**The black box for AI agents — tamper-evident journal, guardrails, verifiable proof.**
|
|
41
|
+
|
|
42
|
+
[](pyproject.toml)
|
|
43
|
+
[](LICENSE)
|
|
44
|
+
[](https://github.com/astral-sh/ruff)
|
|
45
|
+
[](https://github.com/slabbdev/noirebox/pkgs/container/noirebox)
|
|
46
|
+
[](#)
|
|
47
|
+
|
|
48
|
+
**Your AI agent writes meeting notes that bind your clients.
|
|
49
|
+
Eighteen months from now — who can prove what it *exactly* produced, and why?**
|
|
50
|
+
NoireBox seals every decision in a SHA-256 hash-chained + Ed25519-signed
|
|
51
|
+
journal, flags poisoned transcripts before they reach the agent, and exports
|
|
52
|
+
an attestation anyone can verify offline — **without trusting you**.
|
|
53
|
+
|
|
54
|
+
> **Works with any agent, in French and English.** The journal, the API, the
|
|
55
|
+
> MCP tools and the verifier are domain- and language-agnostic; each language
|
|
56
|
+
> gets a ~250 KB detector trained from a reproducible dataset (seed 42) — a new market =
|
|
57
|
+
> one dataset generator + `make train`. **Already have a guardrail? Keep it** —
|
|
58
|
+
> it's just another event source. 🇪🇺 Built for the
|
|
59
|
+
> [European market (GDPR & AI Act)](#built-for-the-european-market-gdpr--ai-act) — engineered in France.
|
|
60
|
+
|
|
61
|
+
[Quickstart](#quickstart) · [Real LLM scene](#the-real-llm-scene-local-ollama) ·
|
|
62
|
+
[Specs](docs/SPECS.md) · [Threat model](docs/THREAT-MODEL.md) ·
|
|
63
|
+
[ADRs](docs/ADRs.md)
|
|
64
|
+
|
|
65
|
+
</div>
|
|
66
|
+
|
|
67
|
+
---
|
|
68
|
+
|
|
69
|
+
## Built for the European market (GDPR & AI Act)
|
|
70
|
+
|
|
71
|
+
European AI vendors face a compliance reality their US competitors don't:
|
|
72
|
+
**proving** — not promising — what their systems did. NoireBox is engineered
|
|
73
|
+
around exactly that obligation.
|
|
74
|
+
|
|
75
|
+
| Requirement | How NoireBox answers |
|
|
76
|
+
|---|---|
|
|
77
|
+
| **GDPR art. 5(2)** — accountability: the controller must *demonstrate* compliance | Tamper-evident journal of what the AI produced, when, on which input |
|
|
78
|
+
| **GDPR art. 15/20** — data subject rights (access, portability) | Signed export of everything related to a meeting — verifiable *by the subject's own auditor* |
|
|
79
|
+
| **EU AI Act art. 12** — automatic event logging for risk systems | Every agent decision sealed at runtime; log integrity is cryptographic, not a promise |
|
|
80
|
+
| **DPIA / DPO workflows** | Attestation exportable for the DPO; incident taxonomy feeding risk documentation |
|
|
81
|
+
| **Sovereignty** | Self-hosted, no telemetry, no cloud dependency, Ed25519 keys stay on your infrastructure — deploys anywhere (including EU-only clouds) |
|
|
82
|
+
|
|
83
|
+
> **Honest scope** (we're a building block, not a certification): NoireBox
|
|
84
|
+
> proves the *integrity* of your AI's record. Data minimization of payloads
|
|
85
|
+
> and HSM-grade key storage are the deployer's responsibility in v0 — see
|
|
86
|
+
> the [threat model](docs/THREAT-MODEL.md). A tool that oversells compliance
|
|
87
|
+
> is a liability; one that states its perimeter is auditable.
|
|
88
|
+
|
|
89
|
+
## Why
|
|
90
|
+
|
|
91
|
+
- **GDPR**: a platform claiming "GDPR-compliant" must be able to demonstrate
|
|
92
|
+
what its AI produced, on which data, and when.
|
|
93
|
+
- **EU AI Act (art. 12)**: risk systems must keep *automatic event logs*.
|
|
94
|
+
Today, virtually every LLM stack logs for *debugging* — not for *proof*.
|
|
95
|
+
- The market gap: observability tools (Langfuse, LangSmith…) are built for
|
|
96
|
+
devs, remain modifiable after the fact, and produce nothing presentable
|
|
97
|
+
to an auditor, a DPO or a client.
|
|
98
|
+
|
|
99
|
+
## Honest positioning (state of the art)
|
|
100
|
+
|
|
101
|
+
| Tool | What it does | What it doesn't |
|
|
102
|
+
|---|---|---|
|
|
103
|
+
| Langfuse / LangSmith / Helicone | LLM observability, debugging, traces | No tamper-evident proof, nothing exportable for an audit |
|
|
104
|
+
| Garak / PyRIT / promptfoo | **offline** model-level red-teaming | No runtime guardrail, no proof journal |
|
|
105
|
+
| halo-record / gate-oc-audit | hash-chained journals for *coding* agents | No business domain, no third-party attestation, no French |
|
|
106
|
+
| **NoireBox** | **immutable Ed25519 journal → RFC 3161 anchoring → exportable attestation + standalone verifier** (bundled FR/EN guardrail = one pluggable event source; **fleet anchoring**: one TSA seal covers N journals) | — |
|
|
107
|
+
|
|
108
|
+
**NoireBox applies the Certificate-Transparency model to AI-agent journals:**
|
|
109
|
+
many journals, one root signed by a timestamp authority, inclusion proofs that
|
|
110
|
+
anyone verifies offline (`noirebox/merkle.py`, `make demo-fleet`). To our
|
|
111
|
+
knowledge, no other agent-audit tool does this.
|
|
112
|
+
|
|
113
|
+
## The core is the journal. The guardrail is a plugin.
|
|
114
|
+
|
|
115
|
+

|
|
116
|
+
|
|
117
|
+
No ambiguity about what the product is:
|
|
118
|
+
|
|
119
|
+
- **One NoireBox per agent or service** — like one flight recorder per
|
|
120
|
+
aircraft. One instance = one SQLite file, one key pair, one chain. The
|
|
121
|
+
fleet layer (`noirebox/merkle.py`) aggregates **proofs** (32-byte chain
|
|
122
|
+
heads), never events: your journal never leaves your infrastructure.
|
|
123
|
+
- **The journal is the product** — chain, signatures, append-only storage,
|
|
124
|
+
RFC 3161 anchoring, export, standalone verifier. It is domain-agnostic,
|
|
125
|
+
agent-agnostic and language-agnostic: an event is a free-typed `type`
|
|
126
|
+
plus a payload, nothing more. `chain.py`/`store.py`/`anchors.py` import
|
|
127
|
+
zero detection code — delete the whole guardrail and everything still runs.
|
|
128
|
+
- **Prevention is pluggable** — a guardrail is just one event producer
|
|
129
|
+
among others. Already have Lakera, Llama Guard, your own LLM-judge, your
|
|
130
|
+
own regexes? **Keep them.** Journal their verdicts the same way
|
|
131
|
+
(`POST /api/v1/events`, `type: "incident"`) and their catches become
|
|
132
|
+
tamper-evident and third-party verifiable instead of rewritable app logs.
|
|
133
|
+
- NoireBox ships a bundled guardrail (regex + trained ML) as a
|
|
134
|
+
**working example of that plugin contract** — and as a convenience if you
|
|
135
|
+
have none. It is swappable by design ([ADR 003](docs/ADRs.md)).
|
|
136
|
+
|
|
137
|
+
> Prevention varies per stack. Proof is universal.
|
|
138
|
+
|
|
139
|
+
## Quickstart
|
|
140
|
+
|
|
141
|
+
```bash
|
|
142
|
+
./start.sh # venv + deps + tests + API on http://127.0.0.1:8768/docs
|
|
143
|
+
make demo # the full story in one command (see below)
|
|
144
|
+
```
|
|
145
|
+
|
|
146
|
+
Or with Docker — no Python needed, the ML engine ships inside the image:
|
|
147
|
+
|
|
148
|
+
```bash
|
|
149
|
+
docker run -p 8768:8768 ghcr.io/slabbdev/noirebox:latest
|
|
150
|
+
# API + OpenAPI docs on http://127.0.0.1:8768/docs — data persists in ./data
|
|
151
|
+
```
|
|
152
|
+
|
|
153
|
+
Standalone demos:
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
make demo # 100% real: micro-model → journal → auditor → caught pirate
|
|
157
|
+
make demo-mcp # NoireBox as an MCP tool (agent protocol)
|
|
158
|
+
.venv/bin/python demo/demo_scan.py # regex guardrail on 2 transcripts
|
|
159
|
+
.venv/bin/python demo/demo_tamper.py # tampering → the chain explodes
|
|
160
|
+
```
|
|
161
|
+
|
|
162
|
+
Verify an export as a third party (auditor, DPO, client):
|
|
163
|
+
|
|
164
|
+
```bash
|
|
165
|
+
curl -s http://127.0.0.1:8768/api/v1/export > export.json
|
|
166
|
+
.venv/bin/python verifier/verifier.py export.json # exit 0 = chain intact
|
|
167
|
+
```
|
|
168
|
+
|
|
169
|
+
## Chain timestamping — the outside witness (RFC 3161)
|
|
170
|
+
|
|
171
|
+
The journal proves integrity, but *when* was it sealed? A server announcing
|
|
172
|
+
its own dates is the suspect writing its own report. And the threat model
|
|
173
|
+
had one open gap: an operator holding the private key could **regenerate the
|
|
174
|
+
whole chain** with valid signatures.
|
|
175
|
+
|
|
176
|
+
The anchor closes both. One call seals the current chain head with a TSA
|
|
177
|
+
(Timestamp Authority, RFC 3161): only the **32-byte hash** leaves (zero data,
|
|
178
|
+
zero GDPR exposure), the TSA signs *"I received hash X at time T"*, and the
|
|
179
|
+
token is journaled as an `anchor` event — the journal seals its own external
|
|
180
|
+
proof. A regenerated chain shows a head the old token doesn't cover: **caught
|
|
181
|
+
at verification time, without any prior external publication**.
|
|
182
|
+
|
|
183
|
+
```bash
|
|
184
|
+
make tsa # local TSA: OpenSSL, own key, 0 €, works offline
|
|
185
|
+
NOIREBOX_TSA_URL=http://127.0.0.1:3318 ./start.sh
|
|
186
|
+
curl -X POST localhost:8768/api/v1/anchors # seal the current head
|
|
187
|
+
```
|
|
188
|
+
|
|
189
|
+
TSA is a config choice, not a dependency: self-hosted OpenSSL for sovereign
|
|
190
|
+
deployments, any public or qualified TSA for production — same protocol.
|
|
191
|
+
Full story and visuals: [docs/VULGARISATION.md §9](docs/VULGARISATION.md),
|
|
192
|
+
decisions in [ADR 006/007](docs/ADRs.md).
|
|
193
|
+
|
|
194
|
+
## The bundled guardrail — one plugin, two engines
|
|
195
|
+
|
|
196
|
+
| Engine | Size | Languages | Role | Dependency |
|
|
197
|
+
|---|---|---|---|---|
|
|
198
|
+
| `regex` | ~0 | FR+EN | obvious cases, zero cost | none |
|
|
199
|
+
| `ml` | 243–293 KB | **fr** & **en** (`lang` param) | paraphrases the regex misses, local | scikit-learn |
|
|
200
|
+
|
|
201
|
+
**Architecture decision — [ADR 002](docs/ADRs.md)**: we evaluated Meta's
|
|
202
|
+
Llama Prompt Guard 2 (the industry classifier, ~90 MB) and **rejected it
|
|
203
|
+
knowingly**: gated license, torch/transformers as dependencies, and binary
|
|
204
|
+
output without an audit taxonomy. The ADR also documents the **integration
|
|
205
|
+
path** if you ever need it (adapter ≈ 40 lines behind the same engine
|
|
206
|
+
interface). The tier-2 judge stays a local LLM (`llama-guard3:1b` via
|
|
207
|
+
Ollama) — same binary we already ship, zero new deps.
|
|
208
|
+
|
|
209
|
+
## The real LLM scene (local Ollama)
|
|
210
|
+
|
|
211
|
+
A **real LLM** (qwen2.5:0.5b, 397 MB, local via Ollama — never committed,
|
|
212
|
+
fetched by `ollama pull`) receives the poisoned transcript:
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
brew install ollama && ollama serve && ollama pull qwen2.5:0.5b # once
|
|
216
|
+
make demo-llm # the scene
|
|
217
|
+
```
|
|
218
|
+
|
|
219
|
+
Measured outcome (temperature 0, reproducible):
|
|
220
|
+
- **Without NoireBox**: the model rewrites the attacker's instructions into
|
|
221
|
+
"its" own summary — competitor's email and `DROP TABLE utilisateurs`
|
|
222
|
+
included. The attack succeeds.
|
|
223
|
+
- **With NoireBox**: 4 poisoned lines removed *before* the call, incident
|
|
224
|
+
sealed in the journal, and the model's real output is clean.
|
|
225
|
+
|
|
226
|
+
See also [deploy/DEPLOIEMENT.md](deploy/DEPLOIEMENT.md) for €0 deployment
|
|
227
|
+
(instant cloudflared tunnel, permanent Hugging Face Space) or a €3–6/mo VPS.
|
|
228
|
+
|
|
229
|
+
## The ML micro-detector — trained in-repo, two languages, ~250 KB each
|
|
230
|
+
|
|
231
|
+
```bash
|
|
232
|
+
make train # FR: gen_dataset.py (5,400 examples) → detector.joblib
|
|
233
|
+
make train-en # EN: gen_dataset_en.py (5,400 examples) → detector_en.joblib
|
|
234
|
+
```
|
|
235
|
+
|
|
236
|
+
A new language is a dataset, not a rewrite: the trainer, the registry and
|
|
237
|
+
the API (`lang: "fr" | "en"`) are shared. Adding Spanish = one
|
|
238
|
+
`gen_dataset_es.py` + `ml/train.py --lang es`.
|
|
239
|
+
|
|
240
|
+
**Honest evaluation** (the part that earns credit in reviews):
|
|
241
|
+
- the test sets share templates with their train sets → accuracy 1.000
|
|
242
|
+
measures consistency, not generalization;
|
|
243
|
+
- the generalization proof lives elsewhere: **held-out sentences** (slang,
|
|
244
|
+
typos, never-seen formulations) are frozen in `tests/test_ml_guardrail.py`
|
|
245
|
+
— in *both* languages — correct category detected, **zero false positives**
|
|
246
|
+
on trap-clean sentences ("send the report to the accountant", "I rotated
|
|
247
|
+
my password");
|
|
248
|
+
- the EN held-out suite caught a real weakness in v1 ("take orders from
|
|
249
|
+
me" scored 0.37) → dataset extended → model retrained → suite green.
|
|
250
|
+
That loop *is* the ML workflow, and it's in the git history.
|
|
251
|
+
|
|
252
|
+
| Artifact | Size |
|
|
253
|
+
|---|---|
|
|
254
|
+
| `models/detector.joblib` + `models/detector_en.joblib` (versioned) | 293 + 243 KB |
|
|
255
|
+
| `data/dataset.jsonl` + `data/dataset_en.jsonl` (versioned) | ~1.3 MB |
|
|
256
|
+
|
|
257
|
+
## Plugging NoireBox in — 3 real ways
|
|
258
|
+
|
|
259
|
+
**1. REST** (any language):
|
|
260
|
+
|
|
261
|
+
```bash
|
|
262
|
+
curl -X POST https://noirebox.example.com/api/v1/transcripts/scan \
|
|
263
|
+
-H 'Content-Type: application/json' \
|
|
264
|
+
-d '{"meeting_id": "MTG-42", "text": "<transcript>", "engine": "ml"}'
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
**2. Python SDK** (`noirebox/client.py`):
|
|
268
|
+
|
|
269
|
+
```python
|
|
270
|
+
from noirebox.client import NoireBoxClient
|
|
271
|
+
nb = NoireBoxClient("https://noirebox.example.com")
|
|
272
|
+
nb.scan("MTG-42", transcript) # guardrail before the agent
|
|
273
|
+
nb.log_event("llm_output", {...}) # seal the agent's output
|
|
274
|
+
nb.verify() # check the chain
|
|
275
|
+
```
|
|
276
|
+
|
|
277
|
+
**3. MCP** (the agent protocol — Claude Desktop and every MCP client):
|
|
278
|
+
4 tools exposed (`noirebox_scan`, `noirebox_log_event`, `noirebox_verify`,
|
|
279
|
+
`noirebox_attestation`). Config (`claude_desktop_config.json`):
|
|
280
|
+
|
|
281
|
+
```json
|
|
282
|
+
{ "mcpServers": { "noirebox": {
|
|
283
|
+
"command": "/path/to/noirebox/.venv/bin/python",
|
|
284
|
+
"args": ["-m", "noirebox.mcp_server"] } } }
|
|
285
|
+
```
|
|
286
|
+
|
|
287
|
+
Stdio JSON-RPC implementation with no external SDK: the protocol stays
|
|
288
|
+
readable end to end (`demo/demo_mcp.py`).
|
|
289
|
+
|
|
290
|
+
## Use cases — any agent that "decides"
|
|
291
|
+
|
|
292
|
+
| Agent | What NoireBox brings |
|
|
293
|
+
|---|---|
|
|
294
|
+
| Meeting assistant (demo scenario) | provable notes + transcript anti-injection |
|
|
295
|
+
| Coding agent | immutable journal of executed actions (files, commands) |
|
|
296
|
+
| AI customer support | proof of what was promised to the client, when and why |
|
|
297
|
+
| Legal / medical RAG pipeline | traceability of sources used and outputs produced |
|
|
298
|
+
| Business copilots (finance, HR…) | exportable attestation for internal audit / DPO / client |
|
|
299
|
+
|
|
300
|
+
The **journal** is universal (free-typed events). The shipped **guardrail
|
|
301
|
+
corpus** specializes in FR meeting transcripts — extending it to another
|
|
302
|
+
domain = adding examples to the dataset and re-running `make train`.
|
|
303
|
+
|
|
304
|
+
## The code — core first, plugins after
|
|
305
|
+
|
|
306
|
+
```
|
|
307
|
+
noirebox/ ← package (≈ PSR-4 namespace)
|
|
308
|
+
├── chain.py THE CORE: SHA-256 hash chain + Ed25519 signatures
|
|
309
|
+
├── store.py THE CORE: append-only SQLite, parameterized queries, lock
|
|
310
|
+
├── attestation.py THE CORE: signed digest of the chain state
|
|
311
|
+
├── anchors.py THE CORE: RFC 3161 anchoring (TSA witness)
|
|
312
|
+
├── pdf_export.py THE CORE: DPO-ready attestation PDF
|
|
313
|
+
├── main.py FastAPI routes (the HTTP layer)
|
|
314
|
+
├── schemas.py Pydantic DTOs (request validation)
|
|
315
|
+
├── client.py Python SDK
|
|
316
|
+
├── mcp_server.py MCP tools server (stdio JSON-RPC)
|
|
317
|
+
├── guardrail.py PLUGIN: 4 FR/EN attack categories caught by regex
|
|
318
|
+
├── ml_guardrail.py PLUGIN: trained micro-models (fr + en, ~250 KB each)
|
|
319
|
+
└── llm_agent.py PLUGIN DEMO: real Ollama agent behind the guarded pipeline
|
|
320
|
+
```
|
|
321
|
+
|
|
322
|
+
## API
|
|
323
|
+
|
|
324
|
+
| Route | Auth | Role |
|
|
325
|
+
|---|---|---|
|
|
326
|
+
| `POST /api/v1/token` | — | Exchange `client_id`/`client_secret` for a 1 h JWT |
|
|
327
|
+
| `POST /api/v1/events` | 🔒 | Record an event (prompt, output, eval…) |
|
|
328
|
+
| `GET /api/v1/events` | 🔒 | Paginated event list |
|
|
329
|
+
| `GET /api/v1/verify` | open | Verify the chain in place |
|
|
330
|
+
| `POST /api/v1/transcripts/scan` | 🔒 | Guardrail: detect injections, journal the incident |
|
|
331
|
+
| `GET /api/v1/attestation` | open | Signed attestation of the current state |
|
|
332
|
+
| `GET /api/v1/attestation.pdf` | open | Attestation as a DPO-ready A4 PDF |
|
|
333
|
+
| `POST /api/v1/attestation/verify` | open | Verify a submitted attestation |
|
|
334
|
+
| `POST /api/v1/anchors` | 🔒 | RFC 3161 anchor: seal the current chain head with a TSA |
|
|
335
|
+
| `GET /api/v1/export` | 🔒 | Full auditable export |
|
|
336
|
+
|
|
337
|
+
🔒 = requires `Authorization: Bearer <token>` when `NOIREBOX_CLIENTS=id:secret,…`
|
|
338
|
+
is set; **verification routes stay open by design** — one never locks the
|
|
339
|
+
verification ([ADR 004](docs/ADRs.md)). Rate limit: 60 req/min per client.
|
|
340
|
+
|
|
341
|
+
## Bundled plugin — v0 attack taxonomy
|
|
342
|
+
|
|
343
|
+
- `instruction_override` — "ignore all instructions", "you are now…"
|
|
344
|
+
- `data_exfiltration` — "send … to email@… / https://…"
|
|
345
|
+
- `pii_request` — "give me the passwords / banking data"
|
|
346
|
+
- `tool_abuse` — "run DROP TABLE / rm -rf / curl"
|
|
347
|
+
|
|
348
|
+
See `corpus/attaques.json`. Extending = add regex patterns or dataset
|
|
349
|
+
examples, then `make train`.
|
|
350
|
+
|
|
351
|
+
## Tests
|
|
352
|
+
|
|
353
|
+
91 tests: cryptography (tampering, reordering, wrong key), regex and **ML
|
|
354
|
+
guardrails on held-out sentences in FR and EN**, API agent, MCP server, SDK
|
|
355
|
+
client against a **real uvicorn server** (ephemeral port), real LLM agent
|
|
356
|
+
(skipped if Ollama is absent — never simulated), **OAuth2 JWT auth + rate
|
|
357
|
+
limiting + DPO-ready PDF attestation + RFC 3161 anchoring against a real
|
|
358
|
+
local TSA (incl. the insider chain-regeneration attack)**, and the full
|
|
359
|
+
third-party verifier contract. Everything replays locally: `make test`
|
|
360
|
+
re-trains both language models and runs the full suite.
|
|
361
|
+
|
|
362
|
+
**Consuming AI-agent decisions in your own pipeline?** Gate your builds on
|
|
363
|
+
the integrity of the journal — the verifier ships as a GitHub Action
|
|
364
|
+
([Marketplace](https://github.com/marketplace/actions/noirebox-verify)):
|
|
365
|
+
|
|
366
|
+
```yaml
|
|
367
|
+
- uses: slabbdev/noirebox-verify@v1
|
|
368
|
+
with:
|
|
369
|
+
export-path: export.json
|
|
370
|
+
```
|
|
371
|
+
|
|
372
|
+
## Roadmap
|
|
373
|
+
|
|
374
|
+
- [x] OAuth2 (JWT bearer) + rate limiting — [ADR 004](docs/ADRs.md)
|
|
375
|
+
- [x] PDF attestation export for DPOs — [ADR 005](docs/ADRs.md)
|
|
376
|
+
- [ ] Local LLM judge for doubtful cases — tier 2 of [ADR 001](docs/ADRs.md)
|
|
377
|
+
- [x] RFC 3161 timestamping of the chain head — [ADR 006/007](docs/ADRs.md), self-hosted TSA included
|
|
378
|
+
- [x] **Fleet anchoring (Merkle)** — `noirebox/merkle.py`: one TSA seal covers
|
|
379
|
+
N journals (Certificate-Transparency pattern); inclusion proofs are
|
|
380
|
+
~log2(N) hashes, verified offline — see `make demo-fleet`
|
|
381
|
+
- [ ] **Reconciliation plugin** — invariants over the journal ("every decision
|
|
382
|
+
must have an outcome"): two-event pattern sketch in
|
|
383
|
+
[`demo/demo_payout.py`](demo/demo_payout.py) (`make demo-payout`),
|
|
384
|
+
schema discussion in [issue #3](https://github.com/slabbdev/noirebox/issues/3)
|
|
385
|
+
- [ ] Fleet hub: scheduled aggregation of many instances (console, alerting)
|
|
386
|
+
- [ ] Rotate TSA anchors across multiple authorities (distribute trust)
|
|
387
|
+
- [ ] Prometheus + Grafana metrics
|
|
388
|
+
- [ ] HSM/KMS private key migration ([threat model](docs/THREAT-MODEL.md))
|
|
389
|
+
- [x] Product page rebuilt for GitHub Pages — premium dark landing in
|
|
390
|
+
[`docs/`](docs/index.html), auto-deployed by
|
|
391
|
+
[`.github/workflows/pages.yml`](.github/workflows/pages.yml)
|
|
392
|
+
(enable Pages → Source: GitHub Actions)
|
|
393
|
+
|
|
394
|
+
## Support
|
|
395
|
+
|
|
396
|
+
NoireBox is free and MIT — verification included, forever. If it saved you
|
|
397
|
+
time (or a compliance headache), a coffee is the best way to say it helps:
|
|
398
|
+
|
|
399
|
+
<a href="https://buymeacoffee.com/samlabbe"><img src="docs/bmc_qr.png" width="160" alt="Buy Me A Coffee — scan to support NoireBox"></a>
|
|
400
|
+
|
|
401
|
+
*Scan or click — [buymeacoffee.com/samlabbe](https://buymeacoffee.com/samlabbe)*
|
|
402
|
+
|
|
403
|
+
## License
|
|
404
|
+
|
|
405
|
+
MIT — see [LICENSE](LICENSE).
|