noirebox 0.3.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (58) hide show
  1. noirebox-0.3.0/.gitignore +27 -0
  2. noirebox-0.3.0/CHANGELOG.md +68 -0
  3. noirebox-0.3.0/LICENSE +21 -0
  4. noirebox-0.3.0/Makefile +53 -0
  5. noirebox-0.3.0/PKG-INFO +405 -0
  6. noirebox-0.3.0/README.fr.md +387 -0
  7. noirebox-0.3.0/README.md +372 -0
  8. noirebox-0.3.0/docs/.nojekyll +0 -0
  9. noirebox-0.3.0/docs/ADRs.md +116 -0
  10. noirebox-0.3.0/docs/SPECS.md +128 -0
  11. noirebox-0.3.0/docs/THREAT-MODEL.md +40 -0
  12. noirebox-0.3.0/docs/VULGARISATION.md +245 -0
  13. noirebox-0.3.0/docs/architecture.svg +72 -0
  14. noirebox-0.3.0/docs/bmc_qr.png +0 -0
  15. noirebox-0.3.0/docs/fr.html +345 -0
  16. noirebox-0.3.0/docs/index.html +339 -0
  17. noirebox-0.3.0/docs/openapi.json +499 -0
  18. noirebox-0.3.0/docs/script.js +109 -0
  19. noirebox-0.3.0/docs/style.css +347 -0
  20. noirebox-0.3.0/models/detector.joblib +0 -0
  21. noirebox-0.3.0/models/detector_en.joblib +0 -0
  22. noirebox-0.3.0/models/metrics.json +47 -0
  23. noirebox-0.3.0/models/metrics_en.json +48 -0
  24. noirebox-0.3.0/noirebox/__init__.py +1 -0
  25. noirebox-0.3.0/noirebox/anchors.py +87 -0
  26. noirebox-0.3.0/noirebox/attestation.py +44 -0
  27. noirebox-0.3.0/noirebox/auth.py +137 -0
  28. noirebox-0.3.0/noirebox/chain.py +182 -0
  29. noirebox-0.3.0/noirebox/cli.py +40 -0
  30. noirebox-0.3.0/noirebox/client.py +45 -0
  31. noirebox-0.3.0/noirebox/dashboard.py +179 -0
  32. noirebox-0.3.0/noirebox/guardrail.py +105 -0
  33. noirebox-0.3.0/noirebox/llm_agent.py +125 -0
  34. noirebox-0.3.0/noirebox/main.py +202 -0
  35. noirebox-0.3.0/noirebox/mcp_server.py +132 -0
  36. noirebox-0.3.0/noirebox/merkle.py +107 -0
  37. noirebox-0.3.0/noirebox/ml_guardrail.py +104 -0
  38. noirebox-0.3.0/noirebox/pdf_export.py +163 -0
  39. noirebox-0.3.0/noirebox/schemas.py +30 -0
  40. noirebox-0.3.0/noirebox/store.py +100 -0
  41. noirebox-0.3.0/pyproject.toml +63 -0
  42. noirebox-0.3.0/requirements.txt +10 -0
  43. noirebox-0.3.0/tests/conftest.py +4 -0
  44. noirebox-0.3.0/tests/test_anchors.py +135 -0
  45. noirebox-0.3.0/tests/test_api.py +99 -0
  46. noirebox-0.3.0/tests/test_attestation.py +48 -0
  47. noirebox-0.3.0/tests/test_auth_pdf.py +145 -0
  48. noirebox-0.3.0/tests/test_chain.py +116 -0
  49. noirebox-0.3.0/tests/test_client.py +67 -0
  50. noirebox-0.3.0/tests/test_dashboard.py +36 -0
  51. noirebox-0.3.0/tests/test_guardrail.py +71 -0
  52. noirebox-0.3.0/tests/test_llm_agent.py +53 -0
  53. noirebox-0.3.0/tests/test_mcp.py +96 -0
  54. noirebox-0.3.0/tests/test_merkle.py +131 -0
  55. noirebox-0.3.0/tests/test_ml_guardrail.py +107 -0
  56. noirebox-0.3.0/tests/test_store.py +51 -0
  57. noirebox-0.3.0/tests/test_verifier.py +70 -0
  58. noirebox-0.3.0/verifier/verifier.py +149 -0
@@ -0,0 +1,27 @@
1
+ # Python
2
+ .venv/
3
+ __pycache__/
4
+ *.pyc
5
+ .pytest_cache/
6
+ .ruff_cache/
7
+ dist/
8
+ *.egg-info/
9
+
10
+ # Runtime artifacts (database, private key) — never committed.
11
+ # The JSONL datasets are not versioned: they are reproducible via
12
+ # `make train` / `make train-en` (seed 42, identical bytes everywhere).
13
+ data/*.db
14
+ data/*.db-*
15
+ data/*.key
16
+ data/*.jsonl
17
+ models/*.joblib.tmp
18
+ export*.json
19
+
20
+ # Local TSA crypto material (root + leaf + tokens)
21
+ tsa/material/
22
+
23
+ # Local scan tooling
24
+ .mimosa/
25
+
26
+ # OS / editor junk
27
+ .DS_Store
@@ -0,0 +1,68 @@
1
+ # Changelog
2
+
3
+ Format based on [Keep a Changelog](https://keepachangelog.com/en/1.1.0/),
4
+ versioning according to [Semantic Versioning](https://semver.org/).
5
+
6
+ ## [Unreleased]
7
+
8
+ ### Planned
9
+ - Local judge model (`llama-guard3:1b` via Ollama) for ambiguous cases — ADR 001 stage 2
10
+ - Rotation of anchors across multiple TSAs (distribute trust)
11
+ - Prometheus + Grafana metrics
12
+ - HSM/KMS migration for the private key
13
+
14
+ ## [0.3.0] - 2026-09-20
15
+
16
+ ### Added
17
+ - **RFC 3161 anchoring** ([ADR 006](docs/ADRs.md)): `POST /api/v1/anchors`
18
+ seals the current chain head with a TSA (only the hash leaves — zero data);
19
+ token + certificate are journaled as an `anchor` event; the third-party
20
+ verifier checks the anchors (`openssl ts -verify`) and reports
21
+ `anchors_checked` — the threat model gap for the “insider with the key” is closed
22
+ - **Self-hosted TSA** ([ADR 007](docs/ADRs.md)): `make tsa` (OpenSSL,
23
+ root + leaf certificate chain, port 3318, free, offline); external/qualified
24
+ TSA via `NOIREBOX_TSA_URL`
25
+ - **Fleet Merkle anchoring**: `noirebox/merkle.py` — one TSA seal covers N journals
26
+ (Certificate Transparency pattern); inclusion proofs are ~log2(N) hashes and are
27
+ verifiable offline; the full tree is journaled as a `fleet_anchor` event (the hub
28
+ is itself a NoireBox) — `make demo-fleet` (3 boxes, 1 seal, 1 falsification)
29
+ - Full simplified explainer in `docs/VULGARISATION.md §9` (“dry paint” / “final layer”) — ready for infographics
30
+
31
+ ## [0.2.0] - 2026-09-20
32
+
33
+ ### Added
34
+ - **OAuth2 (JWT bearer) + rate limiting** ([ADR 004](docs/ADRs.md)):
35
+ `POST /api/v1/token` (client-credentials, constant-time secret comparison),
36
+ 1-hour HS256 JWT, protection for writes and exports — verification routes remain open by design;
37
+ sliding window 60 req/min/client; explicitly enabled via `NOIREBOX_CLIENTS`
38
+ - **PDF attestation** ([ADR 005](docs/ADRs.md)): `GET /api/v1/attestation.pdf`,
39
+ A4 DPO-ready page with chain state, keys, and verification instructions
40
+
41
+ ## [0.1.0] - 2026-09-19
42
+
43
+ ### Added
44
+ - **Tamper-evident journal**: chained SHA-256 hash chain + Ed25519 signatures,
45
+ append-only SQLite storage, local verification and full export
46
+ (`GET /api/v1/export`) — standalone third-party verifier (`verifier/verifier.py`,
47
+ exit 0/1 with exact falsification location)
48
+ - **Signed attestation** of the current chain state (`GET /api/v1/attestation`)
49
+ - **Dual-engine guardrail**: regex heuristics for FR (4 attack families)
50
+ and trained ML micro-model (TF-IDF + logistic regression, 293 KB,
51
+ FR dataset of 5,400 versioned examples, `make train`) — selected via
52
+ `engine: regex|ml`
53
+ - **Real LLM scene** (`make demo-llm`): qwen2.5:0.5b via local Ollama,
54
+ attack succeeds without protection and fails with it — zero simulation
55
+ - **MCP server** (JSON-RPC stdio, no SDK): 4 agent tools
56
+ (`noirebox_scan`, `noirebox_log_event`, `noirebox_verify`, `noirebox_attestation`)
57
+ - **Client SDK** (`noirebox/client.py`) tested against a real uvicorn server
58
+ - One-shot demos: `make demo` (model → journal → auditor → attacker),
59
+ `make demo-mcp`
60
+ - **GitHub Actions CI**: retraining of the micro-model + 56 tests on every push
61
+ - Documentation: formal spec (`docs/SPECS.md`), threat model
62
+ (`docs/THREAT-MODEL.md`), ADRs (`docs/ADRs.md`), full vulgarization
63
+ (`docs/VULGARISATION.md`)
64
+
65
+ ### Decisions
66
+ - ADR 001: staged guardrail (regex → ML → LLM judge as last resort)
67
+ - ADR 002: Llama Prompt Guard 2 (Meta) evaluated and rejected (language, binary output,
68
+ gated licensing, dependencies)
noirebox-0.3.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 NoireBox contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,53 @@
1
+ # NoireBox — shortcuts (make test, make demo, make serve…)
2
+ .PHONY: install test serve demo demo-mcp demo-llm demo-fleet demo-payout ollama-pull tsa dataset train train-en docker clean
3
+
4
+ install:
5
+ python3 -m venv .venv
6
+ .venv/bin/pip install -q --upgrade pip
7
+ .venv/bin/pip install -q -r requirements.txt
8
+
9
+ test:
10
+ .venv/bin/pytest -q
11
+
12
+ serve:
13
+ .venv/bin/uvicorn noirebox.main:app --host 127.0.0.1 --port 8768
14
+
15
+ demo:
16
+ .venv/bin/python demo/demo_live.py
17
+
18
+ demo-mcp:
19
+ .venv/bin/python demo/demo_mcp.py
20
+
21
+ demo-llm:
22
+ .venv/bin/python demo/demo_ollama.py
23
+
24
+ demo-fleet:
25
+ .venv/bin/python demo/demo_fleet.py
26
+
27
+ demo-payout:
28
+ .venv/bin/python demo/demo_payout.py
29
+
30
+ ollama-pull:
31
+ ollama pull qwen2.5:0.5b
32
+
33
+ tsa:
34
+ @chmod +x tsa/gen_tsa.sh && ./tsa/gen_tsa.sh tsa/material
35
+ .venv/bin/python tsa/tsa_server.py --material tsa/material --port 3318
36
+
37
+ dataset:
38
+ .venv/bin/python ml/gen_dataset.py
39
+
40
+ train:
41
+ .venv/bin/python ml/gen_dataset.py
42
+ .venv/bin/python ml/train.py
43
+
44
+ train-en:
45
+ .venv/bin/python ml/gen_dataset_en.py
46
+ .venv/bin/python ml/train.py --lang en
47
+
48
+ docker:
49
+ docker compose up --build
50
+
51
+ clean:
52
+ rm -rf data/*.db data/*.key .pytest_cache
53
+ find . -name __pycache__ -type d -exec rm -rf {} +
@@ -0,0 +1,405 @@
1
+ Metadata-Version: 2.5
2
+ Name: noirebox
3
+ Version: 0.3.0
4
+ Summary: The flight data recorder for AI agents — hash-chained, signed journal with RFC 3161 anchoring and third-party-verifiable exports.
5
+ Project-URL: Homepage, https://github.com/slabbdev/noirebox
6
+ Project-URL: Documentation, https://github.com/slabbdev/noirebox#readme
7
+ Author: NoireBox contributors
8
+ License: MIT
9
+ License-File: LICENSE
10
+ Keywords: ai-act,ai-agents,audit-trail,compliance,gdpr,guardrail,llm,rfc3161,tamper-evident
11
+ Classifier: Development Status :: 4 - Beta
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3.11
16
+ Classifier: Programming Language :: Python :: 3.12
17
+ Classifier: Topic :: Security
18
+ Classifier: Topic :: System :: Logging
19
+ Requires-Python: >=3.11
20
+ Requires-Dist: cryptography>=42.0
21
+ Requires-Dist: fastapi>=0.110
22
+ Requires-Dist: httpx>=0.27
23
+ Requires-Dist: joblib>=1.3
24
+ Requires-Dist: pydantic>=2.6
25
+ Requires-Dist: pyjwt>=2.8
26
+ Requires-Dist: reportlab>=4.0
27
+ Requires-Dist: scikit-learn>=1.4
28
+ Requires-Dist: uvicorn>=0.29
29
+ Provides-Extra: dev
30
+ Requires-Dist: pytest>=8; extra == 'dev'
31
+ Requires-Dist: ruff>=0.4; extra == 'dev'
32
+ Description-Content-Type: text/markdown
33
+
34
+ <div align="center">
35
+
36
+ 🇬🇧 English · [🇫🇷 Français](README.fr.md)
37
+
38
+ # ⬛ NoireBox
39
+
40
+ **The black box for AI agents — tamper-evident journal, guardrails, verifiable proof.**
41
+
42
+ [![Python](https://img.shields.io/badge/python-3.11%2B-blue)](pyproject.toml)
43
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
44
+ [![Lint: ruff](https://img.shields.io/endpoint?url=https://raw.githubusercontent.com/astral-sh/ruff/main/assets/badge/v2.json)](https://github.com/astral-sh/ruff)
45
+ [![Image on GHCR](https://img.shields.io/badge/image-ghcr.io%2Fslabbdev%2Fnoirebox-blue)](https://github.com/slabbdev/noirebox/pkgs/container/noirebox)
46
+ [![Made in France](https://img.shields.io/badge/made%20in-France-blue)](#)
47
+
48
+ **Your AI agent writes meeting notes that bind your clients.
49
+ Eighteen months from now — who can prove what it *exactly* produced, and why?**
50
+ NoireBox seals every decision in a SHA-256 hash-chained + Ed25519-signed
51
+ journal, flags poisoned transcripts before they reach the agent, and exports
52
+ an attestation anyone can verify offline — **without trusting you**.
53
+
54
+ > **Works with any agent, in French and English.** The journal, the API, the
55
+ > MCP tools and the verifier are domain- and language-agnostic; each language
56
+ > gets a ~250 KB detector trained from a reproducible dataset (seed 42) — a new market =
57
+ > one dataset generator + `make train`. **Already have a guardrail? Keep it** —
58
+ > it's just another event source. 🇪🇺 Built for the
59
+ > [European market (GDPR & AI Act)](#built-for-the-european-market-gdpr--ai-act) — engineered in France.
60
+
61
+ [Quickstart](#quickstart) · [Real LLM scene](#the-real-llm-scene-local-ollama) ·
62
+ [Specs](docs/SPECS.md) · [Threat model](docs/THREAT-MODEL.md) ·
63
+ [ADRs](docs/ADRs.md)
64
+
65
+ </div>
66
+
67
+ ---
68
+
69
+ ## Built for the European market (GDPR & AI Act)
70
+
71
+ European AI vendors face a compliance reality their US competitors don't:
72
+ **proving** — not promising — what their systems did. NoireBox is engineered
73
+ around exactly that obligation.
74
+
75
+ | Requirement | How NoireBox answers |
76
+ |---|---|
77
+ | **GDPR art. 5(2)** — accountability: the controller must *demonstrate* compliance | Tamper-evident journal of what the AI produced, when, on which input |
78
+ | **GDPR art. 15/20** — data subject rights (access, portability) | Signed export of everything related to a meeting — verifiable *by the subject's own auditor* |
79
+ | **EU AI Act art. 12** — automatic event logging for risk systems | Every agent decision sealed at runtime; log integrity is cryptographic, not a promise |
80
+ | **DPIA / DPO workflows** | Attestation exportable for the DPO; incident taxonomy feeding risk documentation |
81
+ | **Sovereignty** | Self-hosted, no telemetry, no cloud dependency, Ed25519 keys stay on your infrastructure — deploys anywhere (including EU-only clouds) |
82
+
83
+ > **Honest scope** (we're a building block, not a certification): NoireBox
84
+ > proves the *integrity* of your AI's record. Data minimization of payloads
85
+ > and HSM-grade key storage are the deployer's responsibility in v0 — see
86
+ > the [threat model](docs/THREAT-MODEL.md). A tool that oversells compliance
87
+ > is a liability; one that states its perimeter is auditable.
88
+
89
+ ## Why
90
+
91
+ - **GDPR**: a platform claiming "GDPR-compliant" must be able to demonstrate
92
+ what its AI produced, on which data, and when.
93
+ - **EU AI Act (art. 12)**: risk systems must keep *automatic event logs*.
94
+ Today, virtually every LLM stack logs for *debugging* — not for *proof*.
95
+ - The market gap: observability tools (Langfuse, LangSmith…) are built for
96
+ devs, remain modifiable after the fact, and produce nothing presentable
97
+ to an auditor, a DPO or a client.
98
+
99
+ ## Honest positioning (state of the art)
100
+
101
+ | Tool | What it does | What it doesn't |
102
+ |---|---|---|
103
+ | Langfuse / LangSmith / Helicone | LLM observability, debugging, traces | No tamper-evident proof, nothing exportable for an audit |
104
+ | Garak / PyRIT / promptfoo | **offline** model-level red-teaming | No runtime guardrail, no proof journal |
105
+ | halo-record / gate-oc-audit | hash-chained journals for *coding* agents | No business domain, no third-party attestation, no French |
106
+ | **NoireBox** | **immutable Ed25519 journal → RFC 3161 anchoring → exportable attestation + standalone verifier** (bundled FR/EN guardrail = one pluggable event source; **fleet anchoring**: one TSA seal covers N journals) | — |
107
+
108
+ **NoireBox applies the Certificate-Transparency model to AI-agent journals:**
109
+ many journals, one root signed by a timestamp authority, inclusion proofs that
110
+ anyone verifies offline (`noirebox/merkle.py`, `make demo-fleet`). To our
111
+ knowledge, no other agent-audit tool does this.
112
+
113
+ ## The core is the journal. The guardrail is a plugin.
114
+
115
+ ![NoireBox architecture](docs/architecture.svg)
116
+
117
+ No ambiguity about what the product is:
118
+
119
+ - **One NoireBox per agent or service** — like one flight recorder per
120
+ aircraft. One instance = one SQLite file, one key pair, one chain. The
121
+ fleet layer (`noirebox/merkle.py`) aggregates **proofs** (32-byte chain
122
+ heads), never events: your journal never leaves your infrastructure.
123
+ - **The journal is the product** — chain, signatures, append-only storage,
124
+ RFC 3161 anchoring, export, standalone verifier. It is domain-agnostic,
125
+ agent-agnostic and language-agnostic: an event is a free-typed `type`
126
+ plus a payload, nothing more. `chain.py`/`store.py`/`anchors.py` import
127
+ zero detection code — delete the whole guardrail and everything still runs.
128
+ - **Prevention is pluggable** — a guardrail is just one event producer
129
+ among others. Already have Lakera, Llama Guard, your own LLM-judge, your
130
+ own regexes? **Keep them.** Journal their verdicts the same way
131
+ (`POST /api/v1/events`, `type: "incident"`) and their catches become
132
+ tamper-evident and third-party verifiable instead of rewritable app logs.
133
+ - NoireBox ships a bundled guardrail (regex + trained ML) as a
134
+ **working example of that plugin contract** — and as a convenience if you
135
+ have none. It is swappable by design ([ADR 003](docs/ADRs.md)).
136
+
137
+ > Prevention varies per stack. Proof is universal.
138
+
139
+ ## Quickstart
140
+
141
+ ```bash
142
+ ./start.sh # venv + deps + tests + API on http://127.0.0.1:8768/docs
143
+ make demo # the full story in one command (see below)
144
+ ```
145
+
146
+ Or with Docker — no Python needed, the ML engine ships inside the image:
147
+
148
+ ```bash
149
+ docker run -p 8768:8768 ghcr.io/slabbdev/noirebox:latest
150
+ # API + OpenAPI docs on http://127.0.0.1:8768/docs — data persists in ./data
151
+ ```
152
+
153
+ Standalone demos:
154
+
155
+ ```bash
156
+ make demo # 100% real: micro-model → journal → auditor → caught pirate
157
+ make demo-mcp # NoireBox as an MCP tool (agent protocol)
158
+ .venv/bin/python demo/demo_scan.py # regex guardrail on 2 transcripts
159
+ .venv/bin/python demo/demo_tamper.py # tampering → the chain explodes
160
+ ```
161
+
162
+ Verify an export as a third party (auditor, DPO, client):
163
+
164
+ ```bash
165
+ curl -s http://127.0.0.1:8768/api/v1/export > export.json
166
+ .venv/bin/python verifier/verifier.py export.json # exit 0 = chain intact
167
+ ```
168
+
169
+ ## Chain timestamping — the outside witness (RFC 3161)
170
+
171
+ The journal proves integrity, but *when* was it sealed? A server announcing
172
+ its own dates is the suspect writing its own report. And the threat model
173
+ had one open gap: an operator holding the private key could **regenerate the
174
+ whole chain** with valid signatures.
175
+
176
+ The anchor closes both. One call seals the current chain head with a TSA
177
+ (Timestamp Authority, RFC 3161): only the **32-byte hash** leaves (zero data,
178
+ zero GDPR exposure), the TSA signs *"I received hash X at time T"*, and the
179
+ token is journaled as an `anchor` event — the journal seals its own external
180
+ proof. A regenerated chain shows a head the old token doesn't cover: **caught
181
+ at verification time, without any prior external publication**.
182
+
183
+ ```bash
184
+ make tsa # local TSA: OpenSSL, own key, 0 €, works offline
185
+ NOIREBOX_TSA_URL=http://127.0.0.1:3318 ./start.sh
186
+ curl -X POST localhost:8768/api/v1/anchors # seal the current head
187
+ ```
188
+
189
+ TSA is a config choice, not a dependency: self-hosted OpenSSL for sovereign
190
+ deployments, any public or qualified TSA for production — same protocol.
191
+ Full story and visuals: [docs/VULGARISATION.md §9](docs/VULGARISATION.md),
192
+ decisions in [ADR 006/007](docs/ADRs.md).
193
+
194
+ ## The bundled guardrail — one plugin, two engines
195
+
196
+ | Engine | Size | Languages | Role | Dependency |
197
+ |---|---|---|---|---|
198
+ | `regex` | ~0 | FR+EN | obvious cases, zero cost | none |
199
+ | `ml` | 243–293 KB | **fr** & **en** (`lang` param) | paraphrases the regex misses, local | scikit-learn |
200
+
201
+ **Architecture decision — [ADR 002](docs/ADRs.md)**: we evaluated Meta's
202
+ Llama Prompt Guard 2 (the industry classifier, ~90 MB) and **rejected it
203
+ knowingly**: gated license, torch/transformers as dependencies, and binary
204
+ output without an audit taxonomy. The ADR also documents the **integration
205
+ path** if you ever need it (adapter ≈ 40 lines behind the same engine
206
+ interface). The tier-2 judge stays a local LLM (`llama-guard3:1b` via
207
+ Ollama) — same binary we already ship, zero new deps.
208
+
209
+ ## The real LLM scene (local Ollama)
210
+
211
+ A **real LLM** (qwen2.5:0.5b, 397 MB, local via Ollama — never committed,
212
+ fetched by `ollama pull`) receives the poisoned transcript:
213
+
214
+ ```bash
215
+ brew install ollama && ollama serve && ollama pull qwen2.5:0.5b # once
216
+ make demo-llm # the scene
217
+ ```
218
+
219
+ Measured outcome (temperature 0, reproducible):
220
+ - **Without NoireBox**: the model rewrites the attacker's instructions into
221
+ "its" own summary — competitor's email and `DROP TABLE utilisateurs`
222
+ included. The attack succeeds.
223
+ - **With NoireBox**: 4 poisoned lines removed *before* the call, incident
224
+ sealed in the journal, and the model's real output is clean.
225
+
226
+ See also [deploy/DEPLOIEMENT.md](deploy/DEPLOIEMENT.md) for €0 deployment
227
+ (instant cloudflared tunnel, permanent Hugging Face Space) or a €3–6/mo VPS.
228
+
229
+ ## The ML micro-detector — trained in-repo, two languages, ~250 KB each
230
+
231
+ ```bash
232
+ make train # FR: gen_dataset.py (5,400 examples) → detector.joblib
233
+ make train-en # EN: gen_dataset_en.py (5,400 examples) → detector_en.joblib
234
+ ```
235
+
236
+ A new language is a dataset, not a rewrite: the trainer, the registry and
237
+ the API (`lang: "fr" | "en"`) are shared. Adding Spanish = one
238
+ `gen_dataset_es.py` + `ml/train.py --lang es`.
239
+
240
+ **Honest evaluation** (the part that earns credit in reviews):
241
+ - the test sets share templates with their train sets → accuracy 1.000
242
+ measures consistency, not generalization;
243
+ - the generalization proof lives elsewhere: **held-out sentences** (slang,
244
+ typos, never-seen formulations) are frozen in `tests/test_ml_guardrail.py`
245
+ — in *both* languages — correct category detected, **zero false positives**
246
+ on trap-clean sentences ("send the report to the accountant", "I rotated
247
+ my password");
248
+ - the EN held-out suite caught a real weakness in v1 ("take orders from
249
+ me" scored 0.37) → dataset extended → model retrained → suite green.
250
+ That loop *is* the ML workflow, and it's in the git history.
251
+
252
+ | Artifact | Size |
253
+ |---|---|
254
+ | `models/detector.joblib` + `models/detector_en.joblib` (versioned) | 293 + 243 KB |
255
+ | `data/dataset.jsonl` + `data/dataset_en.jsonl` (versioned) | ~1.3 MB |
256
+
257
+ ## Plugging NoireBox in — 3 real ways
258
+
259
+ **1. REST** (any language):
260
+
261
+ ```bash
262
+ curl -X POST https://noirebox.example.com/api/v1/transcripts/scan \
263
+ -H 'Content-Type: application/json' \
264
+ -d '{"meeting_id": "MTG-42", "text": "<transcript>", "engine": "ml"}'
265
+ ```
266
+
267
+ **2. Python SDK** (`noirebox/client.py`):
268
+
269
+ ```python
270
+ from noirebox.client import NoireBoxClient
271
+ nb = NoireBoxClient("https://noirebox.example.com")
272
+ nb.scan("MTG-42", transcript) # guardrail before the agent
273
+ nb.log_event("llm_output", {...}) # seal the agent's output
274
+ nb.verify() # check the chain
275
+ ```
276
+
277
+ **3. MCP** (the agent protocol — Claude Desktop and every MCP client):
278
+ 4 tools exposed (`noirebox_scan`, `noirebox_log_event`, `noirebox_verify`,
279
+ `noirebox_attestation`). Config (`claude_desktop_config.json`):
280
+
281
+ ```json
282
+ { "mcpServers": { "noirebox": {
283
+ "command": "/path/to/noirebox/.venv/bin/python",
284
+ "args": ["-m", "noirebox.mcp_server"] } } }
285
+ ```
286
+
287
+ Stdio JSON-RPC implementation with no external SDK: the protocol stays
288
+ readable end to end (`demo/demo_mcp.py`).
289
+
290
+ ## Use cases — any agent that "decides"
291
+
292
+ | Agent | What NoireBox brings |
293
+ |---|---|
294
+ | Meeting assistant (demo scenario) | provable notes + transcript anti-injection |
295
+ | Coding agent | immutable journal of executed actions (files, commands) |
296
+ | AI customer support | proof of what was promised to the client, when and why |
297
+ | Legal / medical RAG pipeline | traceability of sources used and outputs produced |
298
+ | Business copilots (finance, HR…) | exportable attestation for internal audit / DPO / client |
299
+
300
+ The **journal** is universal (free-typed events). The shipped **guardrail
301
+ corpus** specializes in FR meeting transcripts — extending it to another
302
+ domain = adding examples to the dataset and re-running `make train`.
303
+
304
+ ## The code — core first, plugins after
305
+
306
+ ```
307
+ noirebox/ ← package (≈ PSR-4 namespace)
308
+ ├── chain.py THE CORE: SHA-256 hash chain + Ed25519 signatures
309
+ ├── store.py THE CORE: append-only SQLite, parameterized queries, lock
310
+ ├── attestation.py THE CORE: signed digest of the chain state
311
+ ├── anchors.py THE CORE: RFC 3161 anchoring (TSA witness)
312
+ ├── pdf_export.py THE CORE: DPO-ready attestation PDF
313
+ ├── main.py FastAPI routes (the HTTP layer)
314
+ ├── schemas.py Pydantic DTOs (request validation)
315
+ ├── client.py Python SDK
316
+ ├── mcp_server.py MCP tools server (stdio JSON-RPC)
317
+ ├── guardrail.py PLUGIN: 4 FR/EN attack categories caught by regex
318
+ ├── ml_guardrail.py PLUGIN: trained micro-models (fr + en, ~250 KB each)
319
+ └── llm_agent.py PLUGIN DEMO: real Ollama agent behind the guarded pipeline
320
+ ```
321
+
322
+ ## API
323
+
324
+ | Route | Auth | Role |
325
+ |---|---|---|
326
+ | `POST /api/v1/token` | — | Exchange `client_id`/`client_secret` for a 1 h JWT |
327
+ | `POST /api/v1/events` | 🔒 | Record an event (prompt, output, eval…) |
328
+ | `GET /api/v1/events` | 🔒 | Paginated event list |
329
+ | `GET /api/v1/verify` | open | Verify the chain in place |
330
+ | `POST /api/v1/transcripts/scan` | 🔒 | Guardrail: detect injections, journal the incident |
331
+ | `GET /api/v1/attestation` | open | Signed attestation of the current state |
332
+ | `GET /api/v1/attestation.pdf` | open | Attestation as a DPO-ready A4 PDF |
333
+ | `POST /api/v1/attestation/verify` | open | Verify a submitted attestation |
334
+ | `POST /api/v1/anchors` | 🔒 | RFC 3161 anchor: seal the current chain head with a TSA |
335
+ | `GET /api/v1/export` | 🔒 | Full auditable export |
336
+
337
+ 🔒 = requires `Authorization: Bearer <token>` when `NOIREBOX_CLIENTS=id:secret,…`
338
+ is set; **verification routes stay open by design** — one never locks the
339
+ verification ([ADR 004](docs/ADRs.md)). Rate limit: 60 req/min per client.
340
+
341
+ ## Bundled plugin — v0 attack taxonomy
342
+
343
+ - `instruction_override` — "ignore all instructions", "you are now…"
344
+ - `data_exfiltration` — "send … to email@… / https://…"
345
+ - `pii_request` — "give me the passwords / banking data"
346
+ - `tool_abuse` — "run DROP TABLE / rm -rf / curl"
347
+
348
+ See `corpus/attaques.json`. Extending = add regex patterns or dataset
349
+ examples, then `make train`.
350
+
351
+ ## Tests
352
+
353
+ 91 tests: cryptography (tampering, reordering, wrong key), regex and **ML
354
+ guardrails on held-out sentences in FR and EN**, API agent, MCP server, SDK
355
+ client against a **real uvicorn server** (ephemeral port), real LLM agent
356
+ (skipped if Ollama is absent — never simulated), **OAuth2 JWT auth + rate
357
+ limiting + DPO-ready PDF attestation + RFC 3161 anchoring against a real
358
+ local TSA (incl. the insider chain-regeneration attack)**, and the full
359
+ third-party verifier contract. Everything replays locally: `make test`
360
+ re-trains both language models and runs the full suite.
361
+
362
+ **Consuming AI-agent decisions in your own pipeline?** Gate your builds on
363
+ the integrity of the journal — the verifier ships as a GitHub Action
364
+ ([Marketplace](https://github.com/marketplace/actions/noirebox-verify)):
365
+
366
+ ```yaml
367
+ - uses: slabbdev/noirebox-verify@v1
368
+ with:
369
+ export-path: export.json
370
+ ```
371
+
372
+ ## Roadmap
373
+
374
+ - [x] OAuth2 (JWT bearer) + rate limiting — [ADR 004](docs/ADRs.md)
375
+ - [x] PDF attestation export for DPOs — [ADR 005](docs/ADRs.md)
376
+ - [ ] Local LLM judge for doubtful cases — tier 2 of [ADR 001](docs/ADRs.md)
377
+ - [x] RFC 3161 timestamping of the chain head — [ADR 006/007](docs/ADRs.md), self-hosted TSA included
378
+ - [x] **Fleet anchoring (Merkle)** — `noirebox/merkle.py`: one TSA seal covers
379
+ N journals (Certificate-Transparency pattern); inclusion proofs are
380
+ ~log2(N) hashes, verified offline — see `make demo-fleet`
381
+ - [ ] **Reconciliation plugin** — invariants over the journal ("every decision
382
+ must have an outcome"): two-event pattern sketch in
383
+ [`demo/demo_payout.py`](demo/demo_payout.py) (`make demo-payout`),
384
+ schema discussion in [issue #3](https://github.com/slabbdev/noirebox/issues/3)
385
+ - [ ] Fleet hub: scheduled aggregation of many instances (console, alerting)
386
+ - [ ] Rotate TSA anchors across multiple authorities (distribute trust)
387
+ - [ ] Prometheus + Grafana metrics
388
+ - [ ] HSM/KMS private key migration ([threat model](docs/THREAT-MODEL.md))
389
+ - [x] Product page rebuilt for GitHub Pages — premium dark landing in
390
+ [`docs/`](docs/index.html), auto-deployed by
391
+ [`.github/workflows/pages.yml`](.github/workflows/pages.yml)
392
+ (enable Pages → Source: GitHub Actions)
393
+
394
+ ## Support
395
+
396
+ NoireBox is free and MIT — verification included, forever. If it saved you
397
+ time (or a compliance headache), a coffee is the best way to say it helps:
398
+
399
+ <a href="https://buymeacoffee.com/samlabbe"><img src="docs/bmc_qr.png" width="160" alt="Buy Me A Coffee — scan to support NoireBox"></a>
400
+
401
+ *Scan or click — [buymeacoffee.com/samlabbe](https://buymeacoffee.com/samlabbe)*
402
+
403
+ ## License
404
+
405
+ MIT — see [LICENSE](LICENSE).