dutygate 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- dutygate-0.1.0/.gitignore +19 -0
- dutygate-0.1.0/CHANGELOG.md +44 -0
- dutygate-0.1.0/LICENSE +21 -0
- dutygate-0.1.0/PKG-INFO +240 -0
- dutygate-0.1.0/README.md +203 -0
- dutygate-0.1.0/SPEC.md +253 -0
- dutygate-0.1.0/conformance/cases.json +1905 -0
- dutygate-0.1.0/conformance/packs/compound.yaml +24 -0
- dutygate-0.1.0/evals/README.md +56 -0
- dutygate-0.1.0/evals/legal-triggers/dataset.jsonl +159 -0
- dutygate-0.1.0/evals/legal-triggers/keywords.yaml +58 -0
- dutygate-0.1.0/evals/outbound-claims/dataset.jsonl +66 -0
- dutygate-0.1.0/evals/outbound-claims/keywords.yaml +27 -0
- dutygate-0.1.0/packs/legal-triggers.yaml +150 -0
- dutygate-0.1.0/packs/outbound-claims.yaml +107 -0
- dutygate-0.1.0/pyproject.toml +115 -0
- dutygate-0.1.0/src/dutygate/__init__.py +41 -0
- dutygate-0.1.0/src/dutygate/_version.py +1 -0
- dutygate-0.1.0/src/dutygate/adapters/__init__.py +1 -0
- dutygate-0.1.0/src/dutygate/adapters/langchain.py +193 -0
- dutygate-0.1.0/src/dutygate/adapters/langgraph.py +160 -0
- dutygate-0.1.0/src/dutygate/adapters/webhook.py +163 -0
- dutygate-0.1.0/src/dutygate/backends/__init__.py +0 -0
- dutygate-0.1.0/src/dutygate/backends/base.py +51 -0
- dutygate-0.1.0/src/dutygate/backends/jev.py +286 -0
- dutygate-0.1.0/src/dutygate/backends/keyword.py +59 -0
- dutygate-0.1.0/src/dutygate/backends/replay.py +86 -0
- dutygate-0.1.0/src/dutygate/bundled.py +53 -0
- dutygate-0.1.0/src/dutygate/cli.py +444 -0
- dutygate-0.1.0/src/dutygate/decision.py +101 -0
- dutygate-0.1.0/src/dutygate/engine.py +226 -0
- dutygate-0.1.0/src/dutygate/errors.py +42 -0
- dutygate-0.1.0/src/dutygate/evaluation/__init__.py +19 -0
- dutygate-0.1.0/src/dutygate/evaluation/dataset.py +68 -0
- dutygate-0.1.0/src/dutygate/evaluation/harness.py +108 -0
- dutygate-0.1.0/src/dutygate/evaluation/metrics.py +126 -0
- dutygate-0.1.0/src/dutygate/evaluation/report.py +78 -0
- dutygate-0.1.0/src/dutygate/evaluation/sweep.py +116 -0
- dutygate-0.1.0/src/dutygate/gate.py +121 -0
- dutygate-0.1.0/src/dutygate/holding.py +13 -0
- dutygate-0.1.0/src/dutygate/py.typed +0 -0
- dutygate-0.1.0/src/dutygate/redact.py +63 -0
- dutygate-0.1.0/src/dutygate/schema.py +191 -0
- dutygate-0.1.0/src/dutygate/server/__init__.py +5 -0
- dutygate-0.1.0/src/dutygate/server/app.py +268 -0
- dutygate-0.1.0/src/dutygate/server/audit.py +40 -0
- dutygate-0.1.0/src/dutygate/server/logs.py +44 -0
- dutygate-0.1.0/src/dutygate/server/metrics.py +45 -0
- dutygate-0.1.0/src/dutygate/server/run.py +25 -0
- dutygate-0.1.0/src/dutygate/validate.py +78 -0
- dutygate-0.1.0/tests/__init__.py +0 -0
- dutygate-0.1.0/tests/adapters/__init__.py +0 -0
- dutygate-0.1.0/tests/adapters/test_langchain.py +209 -0
- dutygate-0.1.0/tests/adapters/test_langgraph.py +276 -0
- dutygate-0.1.0/tests/adapters/test_webhook.py +244 -0
- dutygate-0.1.0/tests/backends/__init__.py +0 -0
- dutygate-0.1.0/tests/backends/test_jev.py +399 -0
- dutygate-0.1.0/tests/backends/test_keyword.py +64 -0
- dutygate-0.1.0/tests/backends/test_replay.py +135 -0
- dutygate-0.1.0/tests/conformance/__init__.py +0 -0
- dutygate-0.1.0/tests/conformance/loader.py +59 -0
- dutygate-0.1.0/tests/conformance/test_engine_conformance.py +67 -0
- dutygate-0.1.0/tests/conformance/test_sidecar_conformance.py +35 -0
- dutygate-0.1.0/tests/evaluation/__init__.py +0 -0
- dutygate-0.1.0/tests/evaluation/test_harness.py +107 -0
- dutygate-0.1.0/tests/evaluation/test_metrics.py +145 -0
- dutygate-0.1.0/tests/evaluation/test_sweep.py +91 -0
- dutygate-0.1.0/tests/helpers.py +89 -0
- dutygate-0.1.0/tests/live/__init__.py +0 -0
- dutygate-0.1.0/tests/live/test_jev_live.py +40 -0
- dutygate-0.1.0/tests/server/__init__.py +0 -0
- dutygate-0.1.0/tests/server/conftest.py +43 -0
- dutygate-0.1.0/tests/server/test_app.py +370 -0
- dutygate-0.1.0/tests/server/test_serve_cli.py +163 -0
- dutygate-0.1.0/tests/test_bundled.py +143 -0
- dutygate-0.1.0/tests/test_cli.py +240 -0
- dutygate-0.1.0/tests/test_cli_eval.py +176 -0
- dutygate-0.1.0/tests/test_datasets.py +76 -0
- dutygate-0.1.0/tests/test_decision.py +101 -0
- dutygate-0.1.0/tests/test_docs.py +72 -0
- dutygate-0.1.0/tests/test_engine.py +210 -0
- dutygate-0.1.0/tests/test_engine_properties.py +37 -0
- dutygate-0.1.0/tests/test_examples.py +78 -0
- dutygate-0.1.0/tests/test_gate.py +167 -0
- dutygate-0.1.0/tests/test_redact.py +114 -0
- dutygate-0.1.0/tests/test_schema.py +274 -0
- dutygate-0.1.0/tests/test_smoke.py +5 -0
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
__pycache__/
|
|
2
|
+
*.py[cod]
|
|
3
|
+
*.egg-info/
|
|
4
|
+
.venv/
|
|
5
|
+
dist/
|
|
6
|
+
build/
|
|
7
|
+
.coverage
|
|
8
|
+
coverage.xml
|
|
9
|
+
htmlcov/
|
|
10
|
+
.pytest_cache/
|
|
11
|
+
.mypy_cache/
|
|
12
|
+
.ruff_cache/
|
|
13
|
+
.hypothesis/
|
|
14
|
+
node_modules/
|
|
15
|
+
clients/typescript/dist/
|
|
16
|
+
.env
|
|
17
|
+
.env.*
|
|
18
|
+
*.log
|
|
19
|
+
.DS_Store
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
All notable changes are documented here. The format follows
|
|
4
|
+
[Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
|
|
5
|
+
[Semantic Versioning](https://semver.org/).
|
|
6
|
+
|
|
7
|
+
## [Unreleased]
|
|
8
|
+
|
|
9
|
+
## [0.1.0] - 2026-09-28
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
- Policy pack format (`schema_version: 1`) with validation that reports every error at once,
|
|
13
|
+
plus warnings.
|
|
14
|
+
- Engine with one backend call per message; `match: any|all`; per-rule thresholds; confident
|
|
15
|
+
and grey flags; fail-safe error decisions with machine-readable codes.
|
|
16
|
+
- Backends:
|
|
17
|
+
- TypeSafe Jev: total time budget, retries on 429/529/5xx
|
|
18
|
+
- replay: fixtures bound to a questions digest
|
|
19
|
+
- keyword: evaluation baseline only
|
|
20
|
+
- Redaction before any backend call: emails, and card numbers only when they pass a Luhn check.
|
|
21
|
+
- `Gate` facade (sync and async).
|
|
22
|
+
- CLI: `validate`, `run`, `eval` and `serve`.
|
|
23
|
+
- Evaluation harness: metrics per category and per tag, `--save-answers`, `--replay`,
|
|
24
|
+
threshold `--sweep`, keyword baseline comparison, and `--fail-under` gates for CI.
|
|
25
|
+
- HTTP sidecar:
|
|
26
|
+
- bearer auth with key rotation, multi-pack routing, a body size limit
|
|
27
|
+
- Prometheus metrics and an opt-in audit log
|
|
28
|
+
- JSON logs without message text
|
|
29
|
+
- Docker image
|
|
30
|
+
- TypeScript client `dutygate-client`: zero dependencies, fail-safe `review` fallback,
|
|
31
|
+
ESM and CJS builds.
|
|
32
|
+
- `GatedHandler` for any web framework, the LangChain `with_legal_gate` adapter, and LangGraph
|
|
33
|
+
`gate_node` / `outbound_node` (extra `langgraph`).
|
|
34
|
+
- Packs: `legal-triggers` (inbound) and `outbound-claims` (the bot's replies), bundled in the
|
|
35
|
+
package with their sample datasets and keyword baselines, and loadable by name.
|
|
36
|
+
- `dutygate init <pack>` copies a bundled pack, its dataset and its keywords into your project
|
|
37
|
+
to customize. `dutygate eval <pack>` and `--backend keyword` default to the bundled files.
|
|
38
|
+
- Conformance suite (55 cases), run through the engine, the sidecar and the TypeScript client.
|
|
39
|
+
- Synthetic evaluation datasets (159 and 66 rows) with keyword baselines.
|
|
40
|
+
|
|
41
|
+
### Known limitations
|
|
42
|
+
- The TypeSafe Jev integration follows the published API reference but has not yet been
|
|
43
|
+
verified against the live API.
|
|
44
|
+
- Question wording and thresholds are untuned.
|
dutygate-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Taimoor Rashid
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
dutygate-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,240 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: dutygate
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: DutyGate: a chatbot-agnostic gate that flags legal and compliance triggers in inbound messages before the bot replies.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Taimoor2500/Dutygate
|
|
6
|
+
Project-URL: Documentation, https://github.com/Taimoor2500/Dutygate#readme
|
|
7
|
+
Project-URL: Issues, https://github.com/Taimoor2500/Dutygate/issues
|
|
8
|
+
Author: Taimoor Rashid
|
|
9
|
+
License-Expression: MIT
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Keywords: chatbot,compliance,guardrail,legal,policy,typesafe
|
|
12
|
+
Classifier: Development Status :: 3 - Alpha
|
|
13
|
+
Classifier: Intended Audience :: Developers
|
|
14
|
+
Classifier: Operating System :: OS Independent
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3 :: Only
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Communications :: Chat
|
|
22
|
+
Classifier: Typing :: Typed
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Requires-Dist: click>=8.1
|
|
25
|
+
Requires-Dist: httpx>=0.27
|
|
26
|
+
Requires-Dist: pydantic>=2.6
|
|
27
|
+
Requires-Dist: pyyaml>=6
|
|
28
|
+
Provides-Extra: langchain
|
|
29
|
+
Requires-Dist: langchain-core>=0.3; extra == 'langchain'
|
|
30
|
+
Provides-Extra: langgraph
|
|
31
|
+
Requires-Dist: langgraph>=1.0; extra == 'langgraph'
|
|
32
|
+
Provides-Extra: server
|
|
33
|
+
Requires-Dist: fastapi>=0.110; extra == 'server'
|
|
34
|
+
Requires-Dist: prometheus-client>=0.20; extra == 'server'
|
|
35
|
+
Requires-Dist: uvicorn[standard]>=0.29; extra == 'server'
|
|
36
|
+
Description-Content-Type: text/markdown
|
|
37
|
+
|
|
38
|
+
# DutyGate
|
|
39
|
+
|
|
40
|
+
**Catch the legal obligations hiding in ordinary support messages before your chatbot replies.**
|
|
41
|
+
|
|
42
|
+

|
|
43
|
+
|
|
44
|
+
"pls stop texting me" is an opt-out. "I want everything you have on me" is a privacy access
|
|
45
|
+
request. "My lawyer will hear about this" is a legal threat. Topic triage files these under
|
|
46
|
+
*billing* or *angry customer*, keyword lists miss the way people actually write, and an LLM bot
|
|
47
|
+
that answers "Done, you're unsubscribed!" when nothing happened creates the liability itself.
|
|
48
|
+
|
|
49
|
+
DutyGate is a small, chatbot-agnostic gate placed in front of the reply step. It reads every
|
|
50
|
+
inbound message, asks [TypeSafe Jev](https://docs.typesafe.ai) a batch of yes/no questions in
|
|
51
|
+
**one** call, applies versioned rules from a YAML **policy pack**, and tells your code what to do:
|
|
52
|
+
|
|
53
|
+
```
|
|
54
|
+
inbound message ──► DUTYGATE ──► continue ──► bot replies normally
|
|
55
|
+
├── route ──► holding reply + case in the right queue
|
|
56
|
+
└── review ──► bot may reply; a human takes a look
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
- **Flags and routes, never acts.** It does not unsubscribe, delete, refund or reply on its
|
|
60
|
+
own. Your people and workflows do that.
|
|
61
|
+
- **Recall-first.** A miss can cost a statutory deadline; a false alarm costs a reviewer a few
|
|
62
|
+
seconds. Low-confidence signals go to `review`, not to `continue`.
|
|
63
|
+
- **Fails safe.** If TypeSafe is down, slow, or returns something malformed, you get a `review`
|
|
64
|
+
decision with an error code, never an exception or a dropped message.
|
|
65
|
+
- **Auditable.** Rules are versioned YAML, testable offline, with a published evaluation
|
|
66
|
+
harness and a cross-language conformance suite.
|
|
67
|
+
|
|
68
|
+
> **Not legal advice.** Which categories matter, their deadlines, and how they are routed depend
|
|
69
|
+
> on your jurisdiction and business. Have counsel review the packs and your holding replies
|
|
70
|
+
> before production use.
|
|
71
|
+
|
|
72
|
+
## Quickstart
|
|
73
|
+
|
|
74
|
+
Offline, with no API key. The reference packs, their sample datasets and keyword lists ship
|
|
75
|
+
inside the package, so they can be used by name. The `keyword` backend here is the naive
|
|
76
|
+
baseline, which exists only for evaluation and demos.
|
|
77
|
+
|
|
78
|
+
```console
|
|
79
|
+
pip install 'dutygate[server]'
|
|
80
|
+
|
|
81
|
+
dutygate validate legal-triggers outbound-claims
|
|
82
|
+
dutygate run legal-triggers --backend keyword --state "pls stop texting me"
|
|
83
|
+
dutygate eval legal-triggers --backend keyword
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
With TypeSafe Jev, the real backend:
|
|
87
|
+
|
|
88
|
+
```console
|
|
89
|
+
export TYPESAFE_API_KEY=... # from typesafe.ai
|
|
90
|
+
export DUTYGATE_JEV_MODEL=jev-1.13.0 # pin the model in production
|
|
91
|
+
dutygate run legal-triggers --state "I want everything you have on me"
|
|
92
|
+
dutygate eval legal-triggers # scores Jev on the bundled sample dataset
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
## Make it yours
|
|
96
|
+
|
|
97
|
+
Copy a reference pack, with its sample dataset and keyword list, into your project and edit
|
|
98
|
+
it. No code is involved: questions, thresholds, queues and new categories are all YAML.
|
|
99
|
+
|
|
100
|
+
```console
|
|
101
|
+
dutygate init legal-triggers # writes legal-triggers.yaml, .dataset.jsonl, .keywords.yaml
|
|
102
|
+
# edit legal-triggers.yaml (and give it your own `name`), add rows to the dataset
|
|
103
|
+
dutygate validate legal-triggers.yaml
|
|
104
|
+
dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --save-answers answers.jsonl
|
|
105
|
+
dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --replay answers.jsonl --sweep
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
Then point your bot at your file: `Gate.from_pack("legal-triggers.yaml")`. To measure it on
|
|
109
|
+
real traffic, add your own labeled messages to the dataset; see
|
|
110
|
+
[docs/labeling.md](docs/labeling.md). The format is in [SPEC.md](SPEC.md).
|
|
111
|
+
|
|
112
|
+
## Use it from your bot
|
|
113
|
+
|
|
114
|
+
### Python
|
|
115
|
+
|
|
116
|
+
```python
|
|
117
|
+
from dutygate import Gate, default_holding_reply
|
|
118
|
+
|
|
119
|
+
gate = Gate.from_pack("packs/legal-triggers.yaml") # TypeSafe Jev, configured from env
|
|
120
|
+
|
|
121
|
+
decision = await gate.check_async(message, conversation_id=cid, channel="sms")
|
|
122
|
+
if decision.action == "route":
|
|
123
|
+
await create_case(decision.flags, message) # your ticketing / queue
|
|
124
|
+
reply = default_holding_reply(decision) # neutral; no bot improvisation
|
|
125
|
+
elif decision.action == "review":
|
|
126
|
+
await create_review_task(decision.flags, message)
|
|
127
|
+
if decision.error: # the gate could not judge it
|
|
128
|
+
await enqueue_rescan(message)
|
|
129
|
+
reply = await bot.reply(message)
|
|
130
|
+
else:
|
|
131
|
+
reply = await bot.reply(message)
|
|
132
|
+
```
|
|
133
|
+
|
|
134
|
+
`GatedHandler` (in `dutygate.adapters.webhook`) wraps exactly this logic for any web
|
|
135
|
+
framework, `with_legal_gate` does the same for LangChain runnables, and
|
|
136
|
+
`gate_node` / `outbound_node` add DutyGate to a LangGraph graph. See
|
|
137
|
+
[docs/adapters.md](docs/adapters.md).
|
|
138
|
+
|
|
139
|
+
### Any language: the HTTP sidecar
|
|
140
|
+
|
|
141
|
+
```console
|
|
142
|
+
DUTYGATE_SIDECAR_KEYS=change-me TYPESAFE_API_KEY=... \
|
|
143
|
+
dutygate serve packs/legal-triggers.yaml packs/outbound-claims.yaml --host 0.0.0.0
|
|
144
|
+
|
|
145
|
+
curl -H 'Authorization: Bearer change-me' -H 'content-type: application/json' \
|
|
146
|
+
-d '{"message": "pls stop texting me", "channel": "sms"}' localhost:8080/v1/gate
|
|
147
|
+
```
|
|
148
|
+
|
|
149
|
+
Or use the Docker image, the [TypeScript client](clients/typescript) (`dutygate-client`), or
|
|
150
|
+
any no-code tool ([n8n / Zapier recipe](examples/n8n/README.md)). See
|
|
151
|
+
[docs/sidecar.md](docs/sidecar.md).
|
|
152
|
+
|
|
153
|
+
## The decision
|
|
154
|
+
|
|
155
|
+
```json
|
|
156
|
+
{
|
|
157
|
+
"id": "dec_4f1c2a9e8b7d4c3fa1e2b3c4d5e6f708",
|
|
158
|
+
"action": "route",
|
|
159
|
+
"primary": "privacy_request",
|
|
160
|
+
"flags": [
|
|
161
|
+
{"category": "privacy_request", "rule_id": "privacy-request", "queue": "privacy",
|
|
162
|
+
"priority": "high", "confidence": 0.93, "level": "confident"},
|
|
163
|
+
{"category": "opt_out", "rule_id": "opt-out", "queue": "compliance",
|
|
164
|
+
"priority": "high", "confidence": 0.31, "level": "grey"}
|
|
165
|
+
],
|
|
166
|
+
"error": null,
|
|
167
|
+
"policy": {"name": "legal-triggers", "version": "0.1.0"},
|
|
168
|
+
"backend": {"name": "typesafe-jev", "model": "jev-1.13.0"},
|
|
169
|
+
"conversation_id": "c_123",
|
|
170
|
+
"latency_ms": 212
|
|
171
|
+
}
|
|
172
|
+
```
|
|
173
|
+
|
|
174
|
+
| `action` | meaning | your bot should |
|
|
175
|
+
|---|---|---|
|
|
176
|
+
| `continue` | no trigger | reply normally |
|
|
177
|
+
| `route` | at least one confident trigger | create a case, send a neutral holding reply, and not improvise about the trigger |
|
|
178
|
+
| `review` | a grey-zone signal, or the gate could not judge (`error` is set) | reply if you like, queue a human review, never auto-close; re-scan later if `error` is set |
|
|
179
|
+
|
|
180
|
+
`primary` is the first confident flag in the pack's rule order (or the first grey one).
|
|
181
|
+
Error codes and every field are specified in [SPEC.md](SPEC.md).
|
|
182
|
+
|
|
183
|
+
## What it detects
|
|
184
|
+
|
|
185
|
+
| pack | categories |
|
|
186
|
+
|---|---|
|
|
187
|
+
| `legal-triggers` (inbound) | `legal_threat`, `regulator_complaint`, `privacy_request`, `chargeback_dispute`, `opt_out`, `accessibility_request` |
|
|
188
|
+
| `outbound-claims` (the bot's own reply) | `admits_liability`, `claims_action_completed`, `legal_statement`, `promises_outcome` |
|
|
189
|
+
|
|
190
|
+
One message can raise several flags ("erase my data and stop the emails"). Queues and
|
|
191
|
+
priorities are examples: map them to your own systems. Packs are plain YAML, so you can add
|
|
192
|
+
categories or tighten questions without touching code ([SPEC.md](SPEC.md)).
|
|
193
|
+
|
|
194
|
+
## Surfaces
|
|
195
|
+
|
|
196
|
+
| surface | status |
|
|
197
|
+
|---|---|
|
|
198
|
+
| Python library (`Gate`, `evaluate`) | ✅ |
|
|
199
|
+
| CLI: `validate`, `run`, `eval`, `serve` | ✅ |
|
|
200
|
+
| HTTP sidecar + Docker image | ✅ |
|
|
201
|
+
| TypeScript client `dutygate-client` | ✅ passes the conformance suite end to end |
|
|
202
|
+
| `GatedHandler` for any web framework (FastAPI, Flask examples) | ✅ |
|
|
203
|
+
| LangChain `with_legal_gate` | ✅ |
|
|
204
|
+
| LangGraph gate and outbound nodes | ✅ |
|
|
205
|
+
|
|
206
|
+
Every surface must reproduce [`conformance/cases.json`](conformance/cases.json).
|
|
207
|
+
|
|
208
|
+
## Evaluation
|
|
209
|
+
|
|
210
|
+
The repo ships synthetic labeled datasets, a deliberately naive keyword baseline, and a
|
|
211
|
+
harness that scores any backend, saves raw answers, and sweeps thresholds offline.
|
|
212
|
+
|
|
213
|
+
On the bundled `legal-triggers` set, the keyword baseline catches **48%** of triggers
|
|
214
|
+
(0% of non-English ones) and false-alarms on **13%** of clean messages. The datasets were
|
|
215
|
+
written by the same author as the baseline, so they are a sanity check, not a benchmark:
|
|
216
|
+
measure on your own traffic before you rely on this. See [evals/README.md](evals/README.md).
|
|
217
|
+
|
|
218
|
+
**If Jev does not beat a keyword list on your data, do not pay for it.**
|
|
219
|
+
|
|
220
|
+
## Status
|
|
221
|
+
|
|
222
|
+
| item | state |
|
|
223
|
+
|---|---|
|
|
224
|
+
| Packs, engine, validator, CLI, eval harness, sidecar, TS client, adapters | built and tested offline (Python coverage ≥ 90%) |
|
|
225
|
+
| Live TypeSafe Jev integration | built against the published API reference and tested with mocked HTTP; **not yet verified against the live API** (`pytest -m live`) |
|
|
226
|
+
| Question wording and thresholds | untuned; tune with `dutygate eval --save-answers` then `--replay --sweep` |
|
|
227
|
+
|
|
228
|
+
## Docs
|
|
229
|
+
|
|
230
|
+
- [SPEC.md](SPEC.md): pack format, evaluation algorithm, decision contract (normative)
|
|
231
|
+
- [docs/integration.md](docs/integration.md): host contract, holding replies, failure behavior, latency patterns
|
|
232
|
+
- [docs/sidecar.md](docs/sidecar.md): HTTP API, auth, metrics, audit log, Docker
|
|
233
|
+
- [docs/adapters.md](docs/adapters.md): `GatedHandler`, LangChain, LangGraph
|
|
234
|
+
- [docs/privacy.md](docs/privacy.md): what leaves your system, redaction, prompt injection
|
|
235
|
+
- [docs/labeling.md](docs/labeling.md): building a real evaluation set
|
|
236
|
+
- [CONTRIBUTING.md](CONTRIBUTING.md) · [SECURITY.md](SECURITY.md) · [CHANGELOG.md](CHANGELOG.md)
|
|
237
|
+
|
|
238
|
+
## License
|
|
239
|
+
|
|
240
|
+
MIT. See [LICENSE](LICENSE).
|
dutygate-0.1.0/README.md
ADDED
|
@@ -0,0 +1,203 @@
|
|
|
1
|
+
# DutyGate
|
|
2
|
+
|
|
3
|
+
**Catch the legal obligations hiding in ordinary support messages before your chatbot replies.**
|
|
4
|
+
|
|
5
|
+

|
|
6
|
+
|
|
7
|
+
"pls stop texting me" is an opt-out. "I want everything you have on me" is a privacy access
|
|
8
|
+
request. "My lawyer will hear about this" is a legal threat. Topic triage files these under
|
|
9
|
+
*billing* or *angry customer*, keyword lists miss the way people actually write, and an LLM bot
|
|
10
|
+
that answers "Done, you're unsubscribed!" when nothing happened creates the liability itself.
|
|
11
|
+
|
|
12
|
+
DutyGate is a small, chatbot-agnostic gate placed in front of the reply step. It reads every
|
|
13
|
+
inbound message, asks [TypeSafe Jev](https://docs.typesafe.ai) a batch of yes/no questions in
|
|
14
|
+
**one** call, applies versioned rules from a YAML **policy pack**, and tells your code what to do:
|
|
15
|
+
|
|
16
|
+
```
|
|
17
|
+
inbound message ──► DUTYGATE ──► continue ──► bot replies normally
|
|
18
|
+
├── route ──► holding reply + case in the right queue
|
|
19
|
+
└── review ──► bot may reply; a human takes a look
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
- **Flags and routes, never acts.** It does not unsubscribe, delete, refund or reply on its
|
|
23
|
+
own. Your people and workflows do that.
|
|
24
|
+
- **Recall-first.** A miss can cost a statutory deadline; a false alarm costs a reviewer a few
|
|
25
|
+
seconds. Low-confidence signals go to `review`, not to `continue`.
|
|
26
|
+
- **Fails safe.** If TypeSafe is down, slow, or returns something malformed, you get a `review`
|
|
27
|
+
decision with an error code, never an exception or a dropped message.
|
|
28
|
+
- **Auditable.** Rules are versioned YAML, testable offline, with a published evaluation
|
|
29
|
+
harness and a cross-language conformance suite.
|
|
30
|
+
|
|
31
|
+
> **Not legal advice.** Which categories matter, their deadlines, and how they are routed depend
|
|
32
|
+
> on your jurisdiction and business. Have counsel review the packs and your holding replies
|
|
33
|
+
> before production use.
|
|
34
|
+
|
|
35
|
+
## Quickstart
|
|
36
|
+
|
|
37
|
+
Offline, with no API key. The reference packs, their sample datasets and keyword lists ship
|
|
38
|
+
inside the package, so they can be used by name. The `keyword` backend here is the naive
|
|
39
|
+
baseline, which exists only for evaluation and demos.
|
|
40
|
+
|
|
41
|
+
```console
|
|
42
|
+
pip install 'dutygate[server]'
|
|
43
|
+
|
|
44
|
+
dutygate validate legal-triggers outbound-claims
|
|
45
|
+
dutygate run legal-triggers --backend keyword --state "pls stop texting me"
|
|
46
|
+
dutygate eval legal-triggers --backend keyword
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
With TypeSafe Jev, the real backend:
|
|
50
|
+
|
|
51
|
+
```console
|
|
52
|
+
export TYPESAFE_API_KEY=... # from typesafe.ai
|
|
53
|
+
export DUTYGATE_JEV_MODEL=jev-1.13.0 # pin the model in production
|
|
54
|
+
dutygate run legal-triggers --state "I want everything you have on me"
|
|
55
|
+
dutygate eval legal-triggers # scores Jev on the bundled sample dataset
|
|
56
|
+
```
|
|
57
|
+
|
|
58
|
+
## Make it yours
|
|
59
|
+
|
|
60
|
+
Copy a reference pack, with its sample dataset and keyword list, into your project and edit
|
|
61
|
+
it. No code is involved: questions, thresholds, queues and new categories are all YAML.
|
|
62
|
+
|
|
63
|
+
```console
|
|
64
|
+
dutygate init legal-triggers # writes legal-triggers.yaml, .dataset.jsonl, .keywords.yaml
|
|
65
|
+
# edit legal-triggers.yaml (and give it your own `name`), add rows to the dataset
|
|
66
|
+
dutygate validate legal-triggers.yaml
|
|
67
|
+
dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --save-answers answers.jsonl
|
|
68
|
+
dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --replay answers.jsonl --sweep
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
Then point your bot at your file: `Gate.from_pack("legal-triggers.yaml")`. To measure it on
|
|
72
|
+
real traffic, add your own labeled messages to the dataset; see
|
|
73
|
+
[docs/labeling.md](docs/labeling.md). The format is in [SPEC.md](SPEC.md).
|
|
74
|
+
|
|
75
|
+
## Use it from your bot
|
|
76
|
+
|
|
77
|
+
### Python
|
|
78
|
+
|
|
79
|
+
```python
|
|
80
|
+
from dutygate import Gate, default_holding_reply
|
|
81
|
+
|
|
82
|
+
gate = Gate.from_pack("packs/legal-triggers.yaml") # TypeSafe Jev, configured from env
|
|
83
|
+
|
|
84
|
+
decision = await gate.check_async(message, conversation_id=cid, channel="sms")
|
|
85
|
+
if decision.action == "route":
|
|
86
|
+
await create_case(decision.flags, message) # your ticketing / queue
|
|
87
|
+
reply = default_holding_reply(decision) # neutral; no bot improvisation
|
|
88
|
+
elif decision.action == "review":
|
|
89
|
+
await create_review_task(decision.flags, message)
|
|
90
|
+
if decision.error: # the gate could not judge it
|
|
91
|
+
await enqueue_rescan(message)
|
|
92
|
+
reply = await bot.reply(message)
|
|
93
|
+
else:
|
|
94
|
+
reply = await bot.reply(message)
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
`GatedHandler` (in `dutygate.adapters.webhook`) wraps exactly this logic for any web
|
|
98
|
+
framework, `with_legal_gate` does the same for LangChain runnables, and
|
|
99
|
+
`gate_node` / `outbound_node` add DutyGate to a LangGraph graph. See
|
|
100
|
+
[docs/adapters.md](docs/adapters.md).
|
|
101
|
+
|
|
102
|
+
### Any language: the HTTP sidecar
|
|
103
|
+
|
|
104
|
+
```console
|
|
105
|
+
DUTYGATE_SIDECAR_KEYS=change-me TYPESAFE_API_KEY=... \
|
|
106
|
+
dutygate serve packs/legal-triggers.yaml packs/outbound-claims.yaml --host 0.0.0.0
|
|
107
|
+
|
|
108
|
+
curl -H 'Authorization: Bearer change-me' -H 'content-type: application/json' \
|
|
109
|
+
-d '{"message": "pls stop texting me", "channel": "sms"}' localhost:8080/v1/gate
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
Or use the Docker image, the [TypeScript client](clients/typescript) (`dutygate-client`), or
|
|
113
|
+
any no-code tool ([n8n / Zapier recipe](examples/n8n/README.md)). See
|
|
114
|
+
[docs/sidecar.md](docs/sidecar.md).
|
|
115
|
+
|
|
116
|
+
## The decision
|
|
117
|
+
|
|
118
|
+
```json
|
|
119
|
+
{
|
|
120
|
+
"id": "dec_4f1c2a9e8b7d4c3fa1e2b3c4d5e6f708",
|
|
121
|
+
"action": "route",
|
|
122
|
+
"primary": "privacy_request",
|
|
123
|
+
"flags": [
|
|
124
|
+
{"category": "privacy_request", "rule_id": "privacy-request", "queue": "privacy",
|
|
125
|
+
"priority": "high", "confidence": 0.93, "level": "confident"},
|
|
126
|
+
{"category": "opt_out", "rule_id": "opt-out", "queue": "compliance",
|
|
127
|
+
"priority": "high", "confidence": 0.31, "level": "grey"}
|
|
128
|
+
],
|
|
129
|
+
"error": null,
|
|
130
|
+
"policy": {"name": "legal-triggers", "version": "0.1.0"},
|
|
131
|
+
"backend": {"name": "typesafe-jev", "model": "jev-1.13.0"},
|
|
132
|
+
"conversation_id": "c_123",
|
|
133
|
+
"latency_ms": 212
|
|
134
|
+
}
|
|
135
|
+
```
|
|
136
|
+
|
|
137
|
+
| `action` | meaning | your bot should |
|
|
138
|
+
|---|---|---|
|
|
139
|
+
| `continue` | no trigger | reply normally |
|
|
140
|
+
| `route` | at least one confident trigger | create a case, send a neutral holding reply, and not improvise about the trigger |
|
|
141
|
+
| `review` | a grey-zone signal, or the gate could not judge (`error` is set) | reply if you like, queue a human review, never auto-close; re-scan later if `error` is set |
|
|
142
|
+
|
|
143
|
+
`primary` is the first confident flag in the pack's rule order (or the first grey one).
|
|
144
|
+
Error codes and every field are specified in [SPEC.md](SPEC.md).
|
|
145
|
+
|
|
146
|
+
## What it detects
|
|
147
|
+
|
|
148
|
+
| pack | categories |
|
|
149
|
+
|---|---|
|
|
150
|
+
| `legal-triggers` (inbound) | `legal_threat`, `regulator_complaint`, `privacy_request`, `chargeback_dispute`, `opt_out`, `accessibility_request` |
|
|
151
|
+
| `outbound-claims` (the bot's own reply) | `admits_liability`, `claims_action_completed`, `legal_statement`, `promises_outcome` |
|
|
152
|
+
|
|
153
|
+
One message can raise several flags ("erase my data and stop the emails"). Queues and
|
|
154
|
+
priorities are examples: map them to your own systems. Packs are plain YAML, so you can add
|
|
155
|
+
categories or tighten questions without touching code ([SPEC.md](SPEC.md)).
|
|
156
|
+
|
|
157
|
+
## Surfaces
|
|
158
|
+
|
|
159
|
+
| surface | status |
|
|
160
|
+
|---|---|
|
|
161
|
+
| Python library (`Gate`, `evaluate`) | ✅ |
|
|
162
|
+
| CLI: `validate`, `run`, `eval`, `serve` | ✅ |
|
|
163
|
+
| HTTP sidecar + Docker image | ✅ |
|
|
164
|
+
| TypeScript client `dutygate-client` | ✅ passes the conformance suite end to end |
|
|
165
|
+
| `GatedHandler` for any web framework (FastAPI, Flask examples) | ✅ |
|
|
166
|
+
| LangChain `with_legal_gate` | ✅ |
|
|
167
|
+
| LangGraph gate and outbound nodes | ✅ |
|
|
168
|
+
|
|
169
|
+
Every surface must reproduce [`conformance/cases.json`](conformance/cases.json).
|
|
170
|
+
|
|
171
|
+
## Evaluation
|
|
172
|
+
|
|
173
|
+
The repo ships synthetic labeled datasets, a deliberately naive keyword baseline, and a
|
|
174
|
+
harness that scores any backend, saves raw answers, and sweeps thresholds offline.
|
|
175
|
+
|
|
176
|
+
On the bundled `legal-triggers` set, the keyword baseline catches **48%** of triggers
|
|
177
|
+
(0% of non-English ones) and false-alarms on **13%** of clean messages. The datasets were
|
|
178
|
+
written by the same author as the baseline, so they are a sanity check, not a benchmark:
|
|
179
|
+
measure on your own traffic before you rely on this. See [evals/README.md](evals/README.md).
|
|
180
|
+
|
|
181
|
+
**If Jev does not beat a keyword list on your data, do not pay for it.**
|
|
182
|
+
|
|
183
|
+
## Status
|
|
184
|
+
|
|
185
|
+
| item | state |
|
|
186
|
+
|---|---|
|
|
187
|
+
| Packs, engine, validator, CLI, eval harness, sidecar, TS client, adapters | built and tested offline (Python coverage ≥ 90%) |
|
|
188
|
+
| Live TypeSafe Jev integration | built against the published API reference and tested with mocked HTTP; **not yet verified against the live API** (`pytest -m live`) |
|
|
189
|
+
| Question wording and thresholds | untuned; tune with `dutygate eval --save-answers` then `--replay --sweep` |
|
|
190
|
+
|
|
191
|
+
## Docs
|
|
192
|
+
|
|
193
|
+
- [SPEC.md](SPEC.md): pack format, evaluation algorithm, decision contract (normative)
|
|
194
|
+
- [docs/integration.md](docs/integration.md): host contract, holding replies, failure behavior, latency patterns
|
|
195
|
+
- [docs/sidecar.md](docs/sidecar.md): HTTP API, auth, metrics, audit log, Docker
|
|
196
|
+
- [docs/adapters.md](docs/adapters.md): `GatedHandler`, LangChain, LangGraph
|
|
197
|
+
- [docs/privacy.md](docs/privacy.md): what leaves your system, redaction, prompt injection
|
|
198
|
+
- [docs/labeling.md](docs/labeling.md): building a real evaluation set
|
|
199
|
+
- [CONTRIBUTING.md](CONTRIBUTING.md) · [SECURITY.md](SECURITY.md) · [CHANGELOG.md](CHANGELOG.md)
|
|
200
|
+
|
|
201
|
+
## License
|
|
202
|
+
|
|
203
|
+
MIT. See [LICENSE](LICENSE).
|