quorumgate-llm 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- quorumgate_llm-0.1.0/LICENSE +21 -0
- quorumgate_llm-0.1.0/PKG-INFO +246 -0
- quorumgate_llm-0.1.0/README.md +220 -0
- quorumgate_llm-0.1.0/pyproject.toml +47 -0
- quorumgate_llm-0.1.0/setup.cfg +4 -0
- quorumgate_llm-0.1.0/src/quorumgate/__init__.py +52 -0
- quorumgate_llm-0.1.0/src/quorumgate/breaker.py +55 -0
- quorumgate_llm-0.1.0/src/quorumgate/council.py +222 -0
- quorumgate_llm-0.1.0/src/quorumgate/gate.py +200 -0
- quorumgate_llm-0.1.0/src/quorumgate/jsonx.py +27 -0
- quorumgate_llm-0.1.0/src/quorumgate/pipeline.py +54 -0
- quorumgate_llm-0.1.0/src/quorumgate/py.typed +0 -0
- quorumgate_llm-0.1.0/src/quorumgate/types.py +98 -0
- quorumgate_llm-0.1.0/src/quorumgate_llm.egg-info/PKG-INFO +246 -0
- quorumgate_llm-0.1.0/src/quorumgate_llm.egg-info/SOURCES.txt +21 -0
- quorumgate_llm-0.1.0/src/quorumgate_llm.egg-info/dependency_links.txt +1 -0
- quorumgate_llm-0.1.0/src/quorumgate_llm.egg-info/requires.txt +3 -0
- quorumgate_llm-0.1.0/src/quorumgate_llm.egg-info/top_level.txt +1 -0
- quorumgate_llm-0.1.0/tests/test_breaker.py +39 -0
- quorumgate_llm-0.1.0/tests/test_council.py +159 -0
- quorumgate_llm-0.1.0/tests/test_gate.py +159 -0
- quorumgate_llm-0.1.0/tests/test_jsonx.py +27 -0
- quorumgate_llm-0.1.0/tests/test_pipeline.py +78 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Delvin Chang
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,246 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: quorumgate-llm
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Never ship unverified LLM output: multi-model councils, severity-graded audit gates, bounded retries, deterministic fallbacks. Zero dependencies.
|
|
5
|
+
Author: Delvin Chang
|
|
6
|
+
License: MIT
|
|
7
|
+
Keywords: llm,reliability,validation,ensemble,guardrails,fallback,multi-model
|
|
8
|
+
Classifier: Development Status :: 4 - Beta
|
|
9
|
+
Classifier: Intended Audience :: Developers
|
|
10
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
11
|
+
Classifier: Operating System :: OS Independent
|
|
12
|
+
Classifier: Programming Language :: Python :: 3
|
|
13
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
18
|
+
Classifier: Topic :: Software Development :: Libraries :: Python Modules
|
|
19
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
20
|
+
Requires-Python: >=3.9
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
License-File: LICENSE
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
25
|
+
Dynamic: license-file
|
|
26
|
+
|
|
27
|
+
# quorumgate
|
|
28
|
+
|
|
29
|
+
**Never ship unverified LLM output.**
|
|
30
|
+
|
|
31
|
+
`quorumgate` is a zero-dependency reliability layer for LLM pipelines. It was
|
|
32
|
+
distilled from a production system that emails AI-generated reports to real
|
|
33
|
+
subscribers every day, where a hallucinated number or a half-rendered template
|
|
34
|
+
is not a bug ticket — it lands in someone's inbox. The rules that system lives
|
|
35
|
+
by are the rules this library encodes:
|
|
36
|
+
|
|
37
|
+
1. **Every output is audited before it ships.** Checks are graded
|
|
38
|
+
HIGH / MED / LOW. HIGH means *do not ship this*.
|
|
39
|
+
2. **Failure earns a bounded retry, not a shrug.** The retry sees exactly
|
|
40
|
+
which checks failed, so it can switch to a stronger model or a tighter
|
|
41
|
+
prompt instead of rolling the same dice again.
|
|
42
|
+
3. **The last line of defense is deterministic.** When generation cannot pass
|
|
43
|
+
the gate, a template built without any LLM ships instead — degraded, but
|
|
44
|
+
never wrong, and never silent.
|
|
45
|
+
4. **Ensembles must degrade gracefully.** A multi-model council is an
|
|
46
|
+
enhancement, not a dependency: dead seats get circuit-broken, quorum
|
|
47
|
+
failures are reported instead of raised, and a wall-clock budget stops a
|
|
48
|
+
sick provider from blowing your deadline.
|
|
49
|
+
|
|
50
|
+
No LLM SDK is imported anywhere. A "model" is any `Callable[[str], str]` you
|
|
51
|
+
provide — OpenAI, Anthropic, Gemini, Ollama, a local server, or a lambda in a
|
|
52
|
+
test. The framework supplies structure; you supply models, checks, and
|
|
53
|
+
fallbacks.
|
|
54
|
+
|
|
55
|
+
```
|
|
56
|
+
pip install quorumgate # stdlib only, Python >= 3.9
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
## Quickstart
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
from quorumgate import AuditGate, check, Severity
|
|
63
|
+
|
|
64
|
+
@check("no_invented_links")
|
|
65
|
+
def no_invented_links(output, context):
|
|
66
|
+
if "http" in output and "http" not in context["source"]:
|
|
67
|
+
return "summary contains a URL that is not in the source"
|
|
68
|
+
|
|
69
|
+
@check("long_enough")
|
|
70
|
+
def long_enough(output, context):
|
|
71
|
+
return len(output.split()) < 15 and "summary under 15 words"
|
|
72
|
+
|
|
73
|
+
def summarize(attempt, context):
|
|
74
|
+
model = strong_model if attempt.is_retry else cheap_model # your callables
|
|
75
|
+
return model(f"Summarize:\n{context['source']}")
|
|
76
|
+
|
|
77
|
+
def first_sentences(context): # deterministic: never wrong
|
|
78
|
+
return " ".join(context["source"].split(". ")[:2]) + "."
|
|
79
|
+
|
|
80
|
+
gate = AuditGate([no_invented_links, long_enough], max_retries=1, cooldown=60)
|
|
81
|
+
result = gate.run(summarize, fallback=first_sentences, context={"source": text})
|
|
82
|
+
|
|
83
|
+
result.output # what ships — always
|
|
84
|
+
result.source # "primary" | "retry" | "soft-pass" | "fallback"
|
|
85
|
+
result.verified # True unless HIGH failures survived (fallback is audited too)
|
|
86
|
+
```
|
|
87
|
+
|
|
88
|
+
With a fallback, `gate.run` **never raises** — the deadline philosophy is
|
|
89
|
+
that shipping the deterministic version beats shipping nothing, and both beat
|
|
90
|
+
shipping something wrong. Without a fallback, unresolvable HIGH failures
|
|
91
|
+
raise `GateError`, because silence is the one thing the gate will not do.
|
|
92
|
+
|
|
93
|
+
### A multi-model council
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
import json
|
|
97
|
+
from quorumgate import Council, Seat, JudgeArbiter, CircuitBreaker
|
|
98
|
+
|
|
99
|
+
council = Council(
|
|
100
|
+
seats=[
|
|
101
|
+
Seat("gemini-flash", call=my_gemini), # each: prompt -> raw text
|
|
102
|
+
Seat("gpt-mini", call=my_openai),
|
|
103
|
+
Seat("local-qwen", call=my_ollama),
|
|
104
|
+
],
|
|
105
|
+
arbiter=JudgeArbiter( # an LLM judge synthesizes...
|
|
106
|
+
call=my_judge_model,
|
|
107
|
+
build_prompt=lambda ops: "Synthesize one verdict from:\n"
|
|
108
|
+
+ json.dumps([o.content for o in ops]),
|
|
109
|
+
), # ...and if the judge dies,
|
|
110
|
+
quorum=2, # highest-confidence wins
|
|
111
|
+
breaker=CircuitBreaker(max_strikes=3, dead_markers=("quota", "402")),
|
|
112
|
+
time_budget=600, # whole batch, wall-clock
|
|
113
|
+
opposed=[("long", "short")], # head-on conflict = dissent 2
|
|
114
|
+
)
|
|
115
|
+
|
|
116
|
+
verdict = council.convene("Assess the 1-2 week outlook for ACME.")
|
|
117
|
+
verdict.quorum_met # False -> use your single-model path, don't crash
|
|
118
|
+
verdict.dissent # 0 unanimous / 1 mixed / 2 head-on conflict
|
|
119
|
+
verdict.verdict # the arbiter's synthesis
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
Seats parse their own replies (default: tolerant JSON extraction that strips
|
|
123
|
+
markdown fences); arbitration is a strategy object (`JudgeArbiter`,
|
|
124
|
+
`MajorityArbiter`, `HighestConfidenceArbiter`, or your own). Disagreement is
|
|
125
|
+
a first-class signal: pass `dissent` downstream so a split council produces a
|
|
126
|
+
hedged output, not false confidence.
|
|
127
|
+
|
|
128
|
+
### Composed
|
|
129
|
+
|
|
130
|
+
```python
|
|
131
|
+
from quorumgate import Pipeline
|
|
132
|
+
|
|
133
|
+
pipeline = Pipeline(
|
|
134
|
+
generate=write_report, # (attempt, context, council_result) -> str
|
|
135
|
+
gate=gate,
|
|
136
|
+
council=council,
|
|
137
|
+
council_prompt=lambda ctx: f"Debate the outlook for {ctx['topic']}.",
|
|
138
|
+
fallback=deterministic_report,
|
|
139
|
+
)
|
|
140
|
+
result = pipeline.run({"topic": "solar"})
|
|
141
|
+
```
|
|
142
|
+
|
|
143
|
+
Two runnable, key-free demos live in [`examples/`](examples/):
|
|
144
|
+
[`daily_brief.py`](examples/daily_brief.py) (council + gate + fallback +
|
|
145
|
+
systemic alerting across a subscriber batch) and
|
|
146
|
+
[`summarize.py`](examples/summarize.py) (escalate-on-retry summarization).
|
|
147
|
+
|
|
148
|
+
## Architecture
|
|
149
|
+
|
|
150
|
+
```
|
|
151
|
+
┌─────────────────────────────────────────┐
|
|
152
|
+
│ Pipeline │
|
|
153
|
+
│ │
|
|
154
|
+
context ──────► Council (optional) │
|
|
155
|
+
│ seats (parallel) ──► parse ──► quorum? │
|
|
156
|
+
│ │ per-seat CircuitBreaker │
|
|
157
|
+
│ │ wall-clock budget │
|
|
158
|
+
│ ▼ │
|
|
159
|
+
│ Arbiter (judge / majority / custom) │
|
|
160
|
+
│ │ judge fails -> fallback arbiter │
|
|
161
|
+
│ ▼ verdict + dissent │
|
|
162
|
+
│ │
|
|
163
|
+
│ generate(attempt, ctx, verdict) │
|
|
164
|
+
│ │ │
|
|
165
|
+
│ ▼ │
|
|
166
|
+
│ AuditGate ── checks (HIGH/MED/LOW) │
|
|
167
|
+
│ │ HIGH? -> cooldown -> retry │
|
|
168
|
+
│ │ (attempt carries failures) │
|
|
169
|
+
│ │ still HIGH? │
|
|
170
|
+
│ │ ├─ all soft + policy ok -> ship │
|
|
171
|
+
│ │ ├─ fallback() -> audit -> ship │
|
|
172
|
+
│ │ └─ no fallback -> GateError │
|
|
173
|
+
│ ▼ │
|
|
174
|
+
│ GateResult(output, source, verified) │
|
|
175
|
+
└─────────────────────────────────────────┘
|
|
176
|
+
```
|
|
177
|
+
|
|
178
|
+
### Production extras you will eventually want
|
|
179
|
+
|
|
180
|
+
- **Soft HIGH checks** — some checks gate *quality* ("reasoning too
|
|
181
|
+
shallow"), not *correctness* ("price is fabricated"). Mark them
|
|
182
|
+
`soft_checks` and supply a `soft_policy` to decide when a below-bar-but-
|
|
183
|
+
not-wrong output may ship (e.g. to an internal audience) instead of
|
|
184
|
+
degrading to the fallback.
|
|
185
|
+
- **Systemic escalation** — when the same HIGH check fails across
|
|
186
|
+
`systemic_threshold` independent `run()` calls, that is not per-item bad
|
|
187
|
+
luck; it is a template or prompt bug hitting your whole batch. `on_systemic`
|
|
188
|
+
fires exactly once per check so a human is paged *before* the batch
|
|
189
|
+
finishes, not after.
|
|
190
|
+
- **Retry cooldown** — free-tier rate limits are usually per-minute windows.
|
|
191
|
+
An instant retry lands in the same window and fails the same way;
|
|
192
|
+
`cooldown=60` makes the retry actually mean something.
|
|
193
|
+
- **Circuit breaker dead-markers** — "quota exceeded" and "payment required"
|
|
194
|
+
are not transient. Substring markers open the seat on the first strike
|
|
195
|
+
instead of paying for three.
|
|
196
|
+
|
|
197
|
+
## What this is not
|
|
198
|
+
|
|
199
|
+
`quorumgate` is a **reliability layer**, not an orchestration framework.
|
|
200
|
+
|
|
201
|
+
| | LangChain / LlamaIndex / DSPy | quorumgate |
|
|
202
|
+
|---|---|---|
|
|
203
|
+
| Core question | *How do I chain LLM calls together?* | *How do I make sure a bad LLM output never reaches a user?* |
|
|
204
|
+
| LLM clients | Bundled integrations | None — you inject callables |
|
|
205
|
+
| Dependencies | Many | Zero (stdlib only) |
|
|
206
|
+
| Prompting | Templates, optimizers | Your problem, on purpose |
|
|
207
|
+
| Failure model | Exceptions / callbacks | Graded checks → retry → deterministic fallback, never silent |
|
|
208
|
+
| Ensembles | Chains/graphs of calls | Council with quorum, circuit breaking, budget, dissent signal |
|
|
209
|
+
|
|
210
|
+
Use both if you like: build your chain in anything, then put its final output
|
|
211
|
+
behind an `AuditGate`. The gate does not care who generated the text.
|
|
212
|
+
|
|
213
|
+
Similarly, this is not a guardrails DSL — there is no YAML, no built-in
|
|
214
|
+
toxicity classifier, no schema language. A check is a Python function over
|
|
215
|
+
`(output, context)`, because in practice the checks that save you are
|
|
216
|
+
domain-specific ones nobody could have shipped in a library.
|
|
217
|
+
|
|
218
|
+
## API surface
|
|
219
|
+
|
|
220
|
+
| Object | Role |
|
|
221
|
+
|---|---|
|
|
222
|
+
| `Seat(name, call, parse?)` | One model voice: prompt → raw text → parsed opinion |
|
|
223
|
+
| `Council(seats, arbiter, quorum, breaker, time_budget, opposed)` | Parallel collection + arbitration + graceful decay |
|
|
224
|
+
| `Arbiter` / `JudgeArbiter` / `MajorityArbiter` / `HighestConfidenceArbiter` | Pluggable arbitration strategies |
|
|
225
|
+
| `CircuitBreaker(max_strikes, dead_markers)` | Skip dead providers for the rest of a run |
|
|
226
|
+
| `@check(name, severity)` | Predicate → graded check |
|
|
227
|
+
| `AuditGate(checks, max_retries, cooldown, soft_checks, soft_policy, systemic_threshold, on_systemic)` | Verify → retry → fallback; never ship unverified |
|
|
228
|
+
| `Pipeline(generate, gate, council?, fallback?)` | The common composition in one call |
|
|
229
|
+
| `extract_json(text)` | Tolerant JSON-from-LLM-text helper |
|
|
230
|
+
| `GateResult` / `CouncilResult` / `Failure` / `Attempt` / `Severity` | Typed results end to end |
|
|
231
|
+
|
|
232
|
+
## Testing
|
|
233
|
+
|
|
234
|
+
```
|
|
235
|
+
pip install -e ".[dev]"
|
|
236
|
+
pytest
|
|
237
|
+
```
|
|
238
|
+
|
|
239
|
+
The suite covers council arbitration (majority, judge, judge-death fallback,
|
|
240
|
+
quorum, dissent, budget exhaustion), gate behavior (retry, fallback, soft
|
|
241
|
+
policy, systemic escalation, crashing checks), and the circuit breaker —
|
|
242
|
+
all with plain callables, no network.
|
|
243
|
+
|
|
244
|
+
## License
|
|
245
|
+
|
|
246
|
+
MIT © Delvin Chang
|
|
@@ -0,0 +1,220 @@
|
|
|
1
|
+
# quorumgate
|
|
2
|
+
|
|
3
|
+
**Never ship unverified LLM output.**
|
|
4
|
+
|
|
5
|
+
`quorumgate` is a zero-dependency reliability layer for LLM pipelines. It was
|
|
6
|
+
distilled from a production system that emails AI-generated reports to real
|
|
7
|
+
subscribers every day, where a hallucinated number or a half-rendered template
|
|
8
|
+
is not a bug ticket — it lands in someone's inbox. The rules that system lives
|
|
9
|
+
by are the rules this library encodes:
|
|
10
|
+
|
|
11
|
+
1. **Every output is audited before it ships.** Checks are graded
|
|
12
|
+
HIGH / MED / LOW. HIGH means *do not ship this*.
|
|
13
|
+
2. **Failure earns a bounded retry, not a shrug.** The retry sees exactly
|
|
14
|
+
which checks failed, so it can switch to a stronger model or a tighter
|
|
15
|
+
prompt instead of rolling the same dice again.
|
|
16
|
+
3. **The last line of defense is deterministic.** When generation cannot pass
|
|
17
|
+
the gate, a template built without any LLM ships instead — degraded, but
|
|
18
|
+
never wrong, and never silent.
|
|
19
|
+
4. **Ensembles must degrade gracefully.** A multi-model council is an
|
|
20
|
+
enhancement, not a dependency: dead seats get circuit-broken, quorum
|
|
21
|
+
failures are reported instead of raised, and a wall-clock budget stops a
|
|
22
|
+
sick provider from blowing your deadline.
|
|
23
|
+
|
|
24
|
+
No LLM SDK is imported anywhere. A "model" is any `Callable[[str], str]` you
|
|
25
|
+
provide — OpenAI, Anthropic, Gemini, Ollama, a local server, or a lambda in a
|
|
26
|
+
test. The framework supplies structure; you supply models, checks, and
|
|
27
|
+
fallbacks.
|
|
28
|
+
|
|
29
|
+
```
|
|
30
|
+
pip install quorumgate # stdlib only, Python >= 3.9
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
## Quickstart
|
|
34
|
+
|
|
35
|
+
```python
|
|
36
|
+
from quorumgate import AuditGate, check, Severity
|
|
37
|
+
|
|
38
|
+
@check("no_invented_links")
|
|
39
|
+
def no_invented_links(output, context):
|
|
40
|
+
if "http" in output and "http" not in context["source"]:
|
|
41
|
+
return "summary contains a URL that is not in the source"
|
|
42
|
+
|
|
43
|
+
@check("long_enough")
|
|
44
|
+
def long_enough(output, context):
|
|
45
|
+
return len(output.split()) < 15 and "summary under 15 words"
|
|
46
|
+
|
|
47
|
+
def summarize(attempt, context):
|
|
48
|
+
model = strong_model if attempt.is_retry else cheap_model # your callables
|
|
49
|
+
return model(f"Summarize:\n{context['source']}")
|
|
50
|
+
|
|
51
|
+
def first_sentences(context): # deterministic: never wrong
|
|
52
|
+
return " ".join(context["source"].split(". ")[:2]) + "."
|
|
53
|
+
|
|
54
|
+
gate = AuditGate([no_invented_links, long_enough], max_retries=1, cooldown=60)
|
|
55
|
+
result = gate.run(summarize, fallback=first_sentences, context={"source": text})
|
|
56
|
+
|
|
57
|
+
result.output # what ships — always
|
|
58
|
+
result.source # "primary" | "retry" | "soft-pass" | "fallback"
|
|
59
|
+
result.verified # True unless HIGH failures survived (fallback is audited too)
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
With a fallback, `gate.run` **never raises** — the deadline philosophy is
|
|
63
|
+
that shipping the deterministic version beats shipping nothing, and both beat
|
|
64
|
+
shipping something wrong. Without a fallback, unresolvable HIGH failures
|
|
65
|
+
raise `GateError`, because silence is the one thing the gate will not do.
|
|
66
|
+
|
|
67
|
+
### A multi-model council
|
|
68
|
+
|
|
69
|
+
```python
|
|
70
|
+
import json
|
|
71
|
+
from quorumgate import Council, Seat, JudgeArbiter, CircuitBreaker
|
|
72
|
+
|
|
73
|
+
council = Council(
|
|
74
|
+
seats=[
|
|
75
|
+
Seat("gemini-flash", call=my_gemini), # each: prompt -> raw text
|
|
76
|
+
Seat("gpt-mini", call=my_openai),
|
|
77
|
+
Seat("local-qwen", call=my_ollama),
|
|
78
|
+
],
|
|
79
|
+
arbiter=JudgeArbiter( # an LLM judge synthesizes...
|
|
80
|
+
call=my_judge_model,
|
|
81
|
+
build_prompt=lambda ops: "Synthesize one verdict from:\n"
|
|
82
|
+
+ json.dumps([o.content for o in ops]),
|
|
83
|
+
), # ...and if the judge dies,
|
|
84
|
+
quorum=2, # highest-confidence wins
|
|
85
|
+
breaker=CircuitBreaker(max_strikes=3, dead_markers=("quota", "402")),
|
|
86
|
+
time_budget=600, # whole batch, wall-clock
|
|
87
|
+
opposed=[("long", "short")], # head-on conflict = dissent 2
|
|
88
|
+
)
|
|
89
|
+
|
|
90
|
+
verdict = council.convene("Assess the 1-2 week outlook for ACME.")
|
|
91
|
+
verdict.quorum_met # False -> use your single-model path, don't crash
|
|
92
|
+
verdict.dissent # 0 unanimous / 1 mixed / 2 head-on conflict
|
|
93
|
+
verdict.verdict # the arbiter's synthesis
|
|
94
|
+
```
|
|
95
|
+
|
|
96
|
+
Seats parse their own replies (default: tolerant JSON extraction that strips
|
|
97
|
+
markdown fences); arbitration is a strategy object (`JudgeArbiter`,
|
|
98
|
+
`MajorityArbiter`, `HighestConfidenceArbiter`, or your own). Disagreement is
|
|
99
|
+
a first-class signal: pass `dissent` downstream so a split council produces a
|
|
100
|
+
hedged output, not false confidence.
|
|
101
|
+
|
|
102
|
+
### Composed
|
|
103
|
+
|
|
104
|
+
```python
|
|
105
|
+
from quorumgate import Pipeline
|
|
106
|
+
|
|
107
|
+
pipeline = Pipeline(
|
|
108
|
+
generate=write_report, # (attempt, context, council_result) -> str
|
|
109
|
+
gate=gate,
|
|
110
|
+
council=council,
|
|
111
|
+
council_prompt=lambda ctx: f"Debate the outlook for {ctx['topic']}.",
|
|
112
|
+
fallback=deterministic_report,
|
|
113
|
+
)
|
|
114
|
+
result = pipeline.run({"topic": "solar"})
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Two runnable, key-free demos live in [`examples/`](examples/):
|
|
118
|
+
[`daily_brief.py`](examples/daily_brief.py) (council + gate + fallback +
|
|
119
|
+
systemic alerting across a subscriber batch) and
|
|
120
|
+
[`summarize.py`](examples/summarize.py) (escalate-on-retry summarization).
|
|
121
|
+
|
|
122
|
+
## Architecture
|
|
123
|
+
|
|
124
|
+
```
|
|
125
|
+
┌─────────────────────────────────────────┐
|
|
126
|
+
│ Pipeline │
|
|
127
|
+
│ │
|
|
128
|
+
context ──────► Council (optional) │
|
|
129
|
+
│ seats (parallel) ──► parse ──► quorum? │
|
|
130
|
+
│ │ per-seat CircuitBreaker │
|
|
131
|
+
│ │ wall-clock budget │
|
|
132
|
+
│ ▼ │
|
|
133
|
+
│ Arbiter (judge / majority / custom) │
|
|
134
|
+
│ │ judge fails -> fallback arbiter │
|
|
135
|
+
│ ▼ verdict + dissent │
|
|
136
|
+
│ │
|
|
137
|
+
│ generate(attempt, ctx, verdict) │
|
|
138
|
+
│ │ │
|
|
139
|
+
│ ▼ │
|
|
140
|
+
│ AuditGate ── checks (HIGH/MED/LOW) │
|
|
141
|
+
│ │ HIGH? -> cooldown -> retry │
|
|
142
|
+
│ │ (attempt carries failures) │
|
|
143
|
+
│ │ still HIGH? │
|
|
144
|
+
│ │ ├─ all soft + policy ok -> ship │
|
|
145
|
+
│ │ ├─ fallback() -> audit -> ship │
|
|
146
|
+
│ │ └─ no fallback -> GateError │
|
|
147
|
+
│ ▼ │
|
|
148
|
+
│ GateResult(output, source, verified) │
|
|
149
|
+
└─────────────────────────────────────────┘
|
|
150
|
+
```
|
|
151
|
+
|
|
152
|
+
### Production extras you will eventually want
|
|
153
|
+
|
|
154
|
+
- **Soft HIGH checks** — some checks gate *quality* ("reasoning too
|
|
155
|
+
shallow"), not *correctness* ("price is fabricated"). Mark them
|
|
156
|
+
`soft_checks` and supply a `soft_policy` to decide when a below-bar-but-
|
|
157
|
+
not-wrong output may ship (e.g. to an internal audience) instead of
|
|
158
|
+
degrading to the fallback.
|
|
159
|
+
- **Systemic escalation** — when the same HIGH check fails across
|
|
160
|
+
`systemic_threshold` independent `run()` calls, that is not per-item bad
|
|
161
|
+
luck; it is a template or prompt bug hitting your whole batch. `on_systemic`
|
|
162
|
+
fires exactly once per check so a human is paged *before* the batch
|
|
163
|
+
finishes, not after.
|
|
164
|
+
- **Retry cooldown** — free-tier rate limits are usually per-minute windows.
|
|
165
|
+
An instant retry lands in the same window and fails the same way;
|
|
166
|
+
`cooldown=60` makes the retry actually mean something.
|
|
167
|
+
- **Circuit breaker dead-markers** — "quota exceeded" and "payment required"
|
|
168
|
+
are not transient. Substring markers open the seat on the first strike
|
|
169
|
+
instead of paying for three.
|
|
170
|
+
|
|
171
|
+
## What this is not
|
|
172
|
+
|
|
173
|
+
`quorumgate` is a **reliability layer**, not an orchestration framework.
|
|
174
|
+
|
|
175
|
+
| | LangChain / LlamaIndex / DSPy | quorumgate |
|
|
176
|
+
|---|---|---|
|
|
177
|
+
| Core question | *How do I chain LLM calls together?* | *How do I make sure a bad LLM output never reaches a user?* |
|
|
178
|
+
| LLM clients | Bundled integrations | None — you inject callables |
|
|
179
|
+
| Dependencies | Many | Zero (stdlib only) |
|
|
180
|
+
| Prompting | Templates, optimizers | Your problem, on purpose |
|
|
181
|
+
| Failure model | Exceptions / callbacks | Graded checks → retry → deterministic fallback, never silent |
|
|
182
|
+
| Ensembles | Chains/graphs of calls | Council with quorum, circuit breaking, budget, dissent signal |
|
|
183
|
+
|
|
184
|
+
Use both if you like: build your chain in anything, then put its final output
|
|
185
|
+
behind an `AuditGate`. The gate does not care who generated the text.
|
|
186
|
+
|
|
187
|
+
Similarly, this is not a guardrails DSL — there is no YAML, no built-in
|
|
188
|
+
toxicity classifier, no schema language. A check is a Python function over
|
|
189
|
+
`(output, context)`, because in practice the checks that save you are
|
|
190
|
+
domain-specific ones nobody could have shipped in a library.
|
|
191
|
+
|
|
192
|
+
## API surface
|
|
193
|
+
|
|
194
|
+
| Object | Role |
|
|
195
|
+
|---|---|
|
|
196
|
+
| `Seat(name, call, parse?)` | One model voice: prompt → raw text → parsed opinion |
|
|
197
|
+
| `Council(seats, arbiter, quorum, breaker, time_budget, opposed)` | Parallel collection + arbitration + graceful decay |
|
|
198
|
+
| `Arbiter` / `JudgeArbiter` / `MajorityArbiter` / `HighestConfidenceArbiter` | Pluggable arbitration strategies |
|
|
199
|
+
| `CircuitBreaker(max_strikes, dead_markers)` | Skip dead providers for the rest of a run |
|
|
200
|
+
| `@check(name, severity)` | Predicate → graded check |
|
|
201
|
+
| `AuditGate(checks, max_retries, cooldown, soft_checks, soft_policy, systemic_threshold, on_systemic)` | Verify → retry → fallback; never ship unverified |
|
|
202
|
+
| `Pipeline(generate, gate, council?, fallback?)` | The common composition in one call |
|
|
203
|
+
| `extract_json(text)` | Tolerant JSON-from-LLM-text helper |
|
|
204
|
+
| `GateResult` / `CouncilResult` / `Failure` / `Attempt` / `Severity` | Typed results end to end |
|
|
205
|
+
|
|
206
|
+
## Testing
|
|
207
|
+
|
|
208
|
+
```
|
|
209
|
+
pip install -e ".[dev]"
|
|
210
|
+
pytest
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
The suite covers council arbitration (majority, judge, judge-death fallback,
|
|
214
|
+
quorum, dissent, budget exhaustion), gate behavior (retry, fallback, soft
|
|
215
|
+
policy, systemic escalation, crashing checks), and the circuit breaker —
|
|
216
|
+
all with plain callables, no network.
|
|
217
|
+
|
|
218
|
+
## License
|
|
219
|
+
|
|
220
|
+
MIT © Delvin Chang
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "quorumgate-llm"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Never ship unverified LLM output: multi-model councils, severity-graded audit gates, bounded retries, deterministic fallbacks. Zero dependencies."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
authors = [{ name = "Delvin Chang" }]
|
|
13
|
+
keywords = [
|
|
14
|
+
"llm",
|
|
15
|
+
"reliability",
|
|
16
|
+
"validation",
|
|
17
|
+
"ensemble",
|
|
18
|
+
"guardrails",
|
|
19
|
+
"fallback",
|
|
20
|
+
"multi-model",
|
|
21
|
+
]
|
|
22
|
+
classifiers = [
|
|
23
|
+
"Development Status :: 4 - Beta",
|
|
24
|
+
"Intended Audience :: Developers",
|
|
25
|
+
"License :: OSI Approved :: MIT License",
|
|
26
|
+
"Operating System :: OS Independent",
|
|
27
|
+
"Programming Language :: Python :: 3",
|
|
28
|
+
"Programming Language :: Python :: 3.9",
|
|
29
|
+
"Programming Language :: Python :: 3.10",
|
|
30
|
+
"Programming Language :: Python :: 3.11",
|
|
31
|
+
"Programming Language :: Python :: 3.12",
|
|
32
|
+
"Programming Language :: Python :: 3.13",
|
|
33
|
+
"Topic :: Software Development :: Libraries :: Python Modules",
|
|
34
|
+
"Topic :: Software Development :: Quality Assurance",
|
|
35
|
+
]
|
|
36
|
+
|
|
37
|
+
[project.optional-dependencies]
|
|
38
|
+
dev = ["pytest>=7"]
|
|
39
|
+
|
|
40
|
+
[tool.setuptools.packages.find]
|
|
41
|
+
where = ["src"]
|
|
42
|
+
|
|
43
|
+
[tool.setuptools.package-data]
|
|
44
|
+
quorumgate = ["py.typed"]
|
|
45
|
+
|
|
46
|
+
[tool.pytest.ini_options]
|
|
47
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
"""quorumgate -- never ship unverified LLM output.
|
|
2
|
+
|
|
3
|
+
A zero-dependency reliability layer for LLM pipelines: multi-model councils
|
|
4
|
+
with pluggable arbitration, severity-graded audit gates with bounded retries,
|
|
5
|
+
and deterministic fallbacks. Bring your own models; the framework provides
|
|
6
|
+
the structure.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from .breaker import CircuitBreaker
|
|
10
|
+
from .council import (
|
|
11
|
+
Arbiter,
|
|
12
|
+
Council,
|
|
13
|
+
HighestConfidenceArbiter,
|
|
14
|
+
JudgeArbiter,
|
|
15
|
+
MajorityArbiter,
|
|
16
|
+
Seat,
|
|
17
|
+
)
|
|
18
|
+
from .gate import AuditGate, check
|
|
19
|
+
from .jsonx import extract_json
|
|
20
|
+
from .pipeline import Pipeline
|
|
21
|
+
from .types import (
|
|
22
|
+
Attempt,
|
|
23
|
+
CouncilResult,
|
|
24
|
+
Failure,
|
|
25
|
+
GateError,
|
|
26
|
+
GateResult,
|
|
27
|
+
Opinion,
|
|
28
|
+
Severity,
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
__version__ = "0.1.0"
|
|
32
|
+
|
|
33
|
+
__all__ = [
|
|
34
|
+
"Arbiter",
|
|
35
|
+
"Attempt",
|
|
36
|
+
"AuditGate",
|
|
37
|
+
"CircuitBreaker",
|
|
38
|
+
"Council",
|
|
39
|
+
"CouncilResult",
|
|
40
|
+
"Failure",
|
|
41
|
+
"GateError",
|
|
42
|
+
"GateResult",
|
|
43
|
+
"HighestConfidenceArbiter",
|
|
44
|
+
"JudgeArbiter",
|
|
45
|
+
"MajorityArbiter",
|
|
46
|
+
"Opinion",
|
|
47
|
+
"Pipeline",
|
|
48
|
+
"Seat",
|
|
49
|
+
"Severity",
|
|
50
|
+
"check",
|
|
51
|
+
"extract_json",
|
|
52
|
+
]
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
"""Per-seat circuit breaker.
|
|
2
|
+
|
|
3
|
+
In a long batch run, a dead model endpoint (exhausted quota, missing key,
|
|
4
|
+
billing wall) fails identically on every call. Without a breaker each item in
|
|
5
|
+
the batch pays the latency and log noise of the same doomed call. The breaker
|
|
6
|
+
opens a seat after ``max_strikes`` consecutive failures -- or immediately when
|
|
7
|
+
the error message contains a *dead marker*, a substring that identifies a
|
|
8
|
+
non-transient failure (e.g. ``"quota exceeded"``, ``"402"``, ``"invalid api
|
|
9
|
+
key"``). Open seats are skipped for the rest of the run.
|
|
10
|
+
"""
|
|
11
|
+
from __future__ import annotations
|
|
12
|
+
|
|
13
|
+
import threading
|
|
14
|
+
from typing import Dict, Iterable, Tuple
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class CircuitBreaker:
|
|
18
|
+
def __init__(self, max_strikes: int = 3, dead_markers: Iterable[str] = ()):
|
|
19
|
+
if max_strikes < 1:
|
|
20
|
+
raise ValueError("max_strikes must be >= 1")
|
|
21
|
+
self.max_strikes = max_strikes
|
|
22
|
+
self.dead_markers: Tuple[str, ...] = tuple(dead_markers)
|
|
23
|
+
self._strikes: Dict[str, int] = {}
|
|
24
|
+
self._open: Dict[str, str] = {}
|
|
25
|
+
self._lock = threading.Lock()
|
|
26
|
+
|
|
27
|
+
def is_open(self, name: str) -> bool:
|
|
28
|
+
return name in self._open
|
|
29
|
+
|
|
30
|
+
@property
|
|
31
|
+
def open_reasons(self) -> Dict[str, str]:
|
|
32
|
+
"""Mapping of open seat name -> truncated reason it was opened."""
|
|
33
|
+
return dict(self._open)
|
|
34
|
+
|
|
35
|
+
def record_success(self, name: str) -> None:
|
|
36
|
+
with self._lock:
|
|
37
|
+
self._strikes[name] = 0
|
|
38
|
+
|
|
39
|
+
def record_failure(self, name: str, error: Exception) -> bool:
|
|
40
|
+
"""Record a failure; return True if the seat is now open."""
|
|
41
|
+
msg = str(error)
|
|
42
|
+
with self._lock:
|
|
43
|
+
if name in self._open:
|
|
44
|
+
return True
|
|
45
|
+
self._strikes[name] = self._strikes.get(name, 0) + 1
|
|
46
|
+
fatal = any(marker in msg for marker in self.dead_markers)
|
|
47
|
+
if fatal or self._strikes[name] >= self.max_strikes:
|
|
48
|
+
self._open[name] = msg[:120]
|
|
49
|
+
return True
|
|
50
|
+
return False
|
|
51
|
+
|
|
52
|
+
def reset(self) -> None:
|
|
53
|
+
with self._lock:
|
|
54
|
+
self._strikes.clear()
|
|
55
|
+
self._open.clear()
|