dutygate 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. dutygate-0.1.0/.gitignore +19 -0
  2. dutygate-0.1.0/CHANGELOG.md +44 -0
  3. dutygate-0.1.0/LICENSE +21 -0
  4. dutygate-0.1.0/PKG-INFO +240 -0
  5. dutygate-0.1.0/README.md +203 -0
  6. dutygate-0.1.0/SPEC.md +253 -0
  7. dutygate-0.1.0/conformance/cases.json +1905 -0
  8. dutygate-0.1.0/conformance/packs/compound.yaml +24 -0
  9. dutygate-0.1.0/evals/README.md +56 -0
  10. dutygate-0.1.0/evals/legal-triggers/dataset.jsonl +159 -0
  11. dutygate-0.1.0/evals/legal-triggers/keywords.yaml +58 -0
  12. dutygate-0.1.0/evals/outbound-claims/dataset.jsonl +66 -0
  13. dutygate-0.1.0/evals/outbound-claims/keywords.yaml +27 -0
  14. dutygate-0.1.0/packs/legal-triggers.yaml +150 -0
  15. dutygate-0.1.0/packs/outbound-claims.yaml +107 -0
  16. dutygate-0.1.0/pyproject.toml +115 -0
  17. dutygate-0.1.0/src/dutygate/__init__.py +41 -0
  18. dutygate-0.1.0/src/dutygate/_version.py +1 -0
  19. dutygate-0.1.0/src/dutygate/adapters/__init__.py +1 -0
  20. dutygate-0.1.0/src/dutygate/adapters/langchain.py +193 -0
  21. dutygate-0.1.0/src/dutygate/adapters/langgraph.py +160 -0
  22. dutygate-0.1.0/src/dutygate/adapters/webhook.py +163 -0
  23. dutygate-0.1.0/src/dutygate/backends/__init__.py +0 -0
  24. dutygate-0.1.0/src/dutygate/backends/base.py +51 -0
  25. dutygate-0.1.0/src/dutygate/backends/jev.py +286 -0
  26. dutygate-0.1.0/src/dutygate/backends/keyword.py +59 -0
  27. dutygate-0.1.0/src/dutygate/backends/replay.py +86 -0
  28. dutygate-0.1.0/src/dutygate/bundled.py +53 -0
  29. dutygate-0.1.0/src/dutygate/cli.py +444 -0
  30. dutygate-0.1.0/src/dutygate/decision.py +101 -0
  31. dutygate-0.1.0/src/dutygate/engine.py +226 -0
  32. dutygate-0.1.0/src/dutygate/errors.py +42 -0
  33. dutygate-0.1.0/src/dutygate/evaluation/__init__.py +19 -0
  34. dutygate-0.1.0/src/dutygate/evaluation/dataset.py +68 -0
  35. dutygate-0.1.0/src/dutygate/evaluation/harness.py +108 -0
  36. dutygate-0.1.0/src/dutygate/evaluation/metrics.py +126 -0
  37. dutygate-0.1.0/src/dutygate/evaluation/report.py +78 -0
  38. dutygate-0.1.0/src/dutygate/evaluation/sweep.py +116 -0
  39. dutygate-0.1.0/src/dutygate/gate.py +121 -0
  40. dutygate-0.1.0/src/dutygate/holding.py +13 -0
  41. dutygate-0.1.0/src/dutygate/py.typed +0 -0
  42. dutygate-0.1.0/src/dutygate/redact.py +63 -0
  43. dutygate-0.1.0/src/dutygate/schema.py +191 -0
  44. dutygate-0.1.0/src/dutygate/server/__init__.py +5 -0
  45. dutygate-0.1.0/src/dutygate/server/app.py +268 -0
  46. dutygate-0.1.0/src/dutygate/server/audit.py +40 -0
  47. dutygate-0.1.0/src/dutygate/server/logs.py +44 -0
  48. dutygate-0.1.0/src/dutygate/server/metrics.py +45 -0
  49. dutygate-0.1.0/src/dutygate/server/run.py +25 -0
  50. dutygate-0.1.0/src/dutygate/validate.py +78 -0
  51. dutygate-0.1.0/tests/__init__.py +0 -0
  52. dutygate-0.1.0/tests/adapters/__init__.py +0 -0
  53. dutygate-0.1.0/tests/adapters/test_langchain.py +209 -0
  54. dutygate-0.1.0/tests/adapters/test_langgraph.py +276 -0
  55. dutygate-0.1.0/tests/adapters/test_webhook.py +244 -0
  56. dutygate-0.1.0/tests/backends/__init__.py +0 -0
  57. dutygate-0.1.0/tests/backends/test_jev.py +399 -0
  58. dutygate-0.1.0/tests/backends/test_keyword.py +64 -0
  59. dutygate-0.1.0/tests/backends/test_replay.py +135 -0
  60. dutygate-0.1.0/tests/conformance/__init__.py +0 -0
  61. dutygate-0.1.0/tests/conformance/loader.py +59 -0
  62. dutygate-0.1.0/tests/conformance/test_engine_conformance.py +67 -0
  63. dutygate-0.1.0/tests/conformance/test_sidecar_conformance.py +35 -0
  64. dutygate-0.1.0/tests/evaluation/__init__.py +0 -0
  65. dutygate-0.1.0/tests/evaluation/test_harness.py +107 -0
  66. dutygate-0.1.0/tests/evaluation/test_metrics.py +145 -0
  67. dutygate-0.1.0/tests/evaluation/test_sweep.py +91 -0
  68. dutygate-0.1.0/tests/helpers.py +89 -0
  69. dutygate-0.1.0/tests/live/__init__.py +0 -0
  70. dutygate-0.1.0/tests/live/test_jev_live.py +40 -0
  71. dutygate-0.1.0/tests/server/__init__.py +0 -0
  72. dutygate-0.1.0/tests/server/conftest.py +43 -0
  73. dutygate-0.1.0/tests/server/test_app.py +370 -0
  74. dutygate-0.1.0/tests/server/test_serve_cli.py +163 -0
  75. dutygate-0.1.0/tests/test_bundled.py +143 -0
  76. dutygate-0.1.0/tests/test_cli.py +240 -0
  77. dutygate-0.1.0/tests/test_cli_eval.py +176 -0
  78. dutygate-0.1.0/tests/test_datasets.py +76 -0
  79. dutygate-0.1.0/tests/test_decision.py +101 -0
  80. dutygate-0.1.0/tests/test_docs.py +72 -0
  81. dutygate-0.1.0/tests/test_engine.py +210 -0
  82. dutygate-0.1.0/tests/test_engine_properties.py +37 -0
  83. dutygate-0.1.0/tests/test_examples.py +78 -0
  84. dutygate-0.1.0/tests/test_gate.py +167 -0
  85. dutygate-0.1.0/tests/test_redact.py +114 -0
  86. dutygate-0.1.0/tests/test_schema.py +274 -0
  87. dutygate-0.1.0/tests/test_smoke.py +5 -0
@@ -0,0 +1,19 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ *.egg-info/
4
+ .venv/
5
+ dist/
6
+ build/
7
+ .coverage
8
+ coverage.xml
9
+ htmlcov/
10
+ .pytest_cache/
11
+ .mypy_cache/
12
+ .ruff_cache/
13
+ .hypothesis/
14
+ node_modules/
15
+ clients/typescript/dist/
16
+ .env
17
+ .env.*
18
+ *.log
19
+ .DS_Store
@@ -0,0 +1,44 @@
1
+ # Changelog
2
+
3
+ All notable changes are documented here. The format follows
4
+ [Keep a Changelog](https://keepachangelog.com/en/1.1.0/), and the project uses
5
+ [Semantic Versioning](https://semver.org/).
6
+
7
+ ## [Unreleased]
8
+
9
+ ## [0.1.0] - 2026-09-28
10
+
11
+ ### Added
12
+ - Policy pack format (`schema_version: 1`) with validation that reports every error at once,
13
+ plus warnings.
14
+ - Engine with one backend call per message; `match: any|all`; per-rule thresholds; confident
15
+ and grey flags; fail-safe error decisions with machine-readable codes.
16
+ - Backends:
17
+ - TypeSafe Jev: total time budget, retries on 429/529/5xx
18
+ - replay: fixtures bound to a questions digest
19
+ - keyword: evaluation baseline only
20
+ - Redaction before any backend call: emails, and card numbers only when they pass a Luhn check.
21
+ - `Gate` facade (sync and async).
22
+ - CLI: `validate`, `run`, `eval` and `serve`.
23
+ - Evaluation harness: metrics per category and per tag, `--save-answers`, `--replay`,
24
+ threshold `--sweep`, keyword baseline comparison, and `--fail-under` gates for CI.
25
+ - HTTP sidecar:
26
+ - bearer auth with key rotation, multi-pack routing, a body size limit
27
+ - Prometheus metrics and an opt-in audit log
28
+ - JSON logs without message text
29
+ - Docker image
30
+ - TypeScript client `dutygate-client`: zero dependencies, fail-safe `review` fallback,
31
+ ESM and CJS builds.
32
+ - `GatedHandler` for any web framework, the LangChain `with_legal_gate` adapter, and LangGraph
33
+ `gate_node` / `outbound_node` (extra `langgraph`).
34
+ - Packs: `legal-triggers` (inbound) and `outbound-claims` (the bot's replies), bundled in the
35
+ package with their sample datasets and keyword baselines, and loadable by name.
36
+ - `dutygate init <pack>` copies a bundled pack, its dataset and its keywords into your project
37
+ to customize. `dutygate eval <pack>` and `--backend keyword` default to the bundled files.
38
+ - Conformance suite (55 cases), run through the engine, the sidecar and the TypeScript client.
39
+ - Synthetic evaluation datasets (159 and 66 rows) with keyword baselines.
40
+
41
+ ### Known limitations
42
+ - The TypeSafe Jev integration follows the published API reference but has not yet been
43
+ verified against the live API.
44
+ - Question wording and thresholds are untuned.
dutygate-0.1.0/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Taimoor Rashid
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,240 @@
1
+ Metadata-Version: 2.5
2
+ Name: dutygate
3
+ Version: 0.1.0
4
+ Summary: DutyGate: a chatbot-agnostic gate that flags legal and compliance triggers in inbound messages before the bot replies.
5
+ Project-URL: Homepage, https://github.com/Taimoor2500/Dutygate
6
+ Project-URL: Documentation, https://github.com/Taimoor2500/Dutygate#readme
7
+ Project-URL: Issues, https://github.com/Taimoor2500/Dutygate/issues
8
+ Author: Taimoor Rashid
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: chatbot,compliance,guardrail,legal,policy,typesafe
12
+ Classifier: Development Status :: 3 - Alpha
13
+ Classifier: Intended Audience :: Developers
14
+ Classifier: Operating System :: OS Independent
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3 :: Only
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Communications :: Chat
22
+ Classifier: Typing :: Typed
23
+ Requires-Python: >=3.10
24
+ Requires-Dist: click>=8.1
25
+ Requires-Dist: httpx>=0.27
26
+ Requires-Dist: pydantic>=2.6
27
+ Requires-Dist: pyyaml>=6
28
+ Provides-Extra: langchain
29
+ Requires-Dist: langchain-core>=0.3; extra == 'langchain'
30
+ Provides-Extra: langgraph
31
+ Requires-Dist: langgraph>=1.0; extra == 'langgraph'
32
+ Provides-Extra: server
33
+ Requires-Dist: fastapi>=0.110; extra == 'server'
34
+ Requires-Dist: prometheus-client>=0.20; extra == 'server'
35
+ Requires-Dist: uvicorn[standard]>=0.29; extra == 'server'
36
+ Description-Content-Type: text/markdown
37
+
38
+ # DutyGate
39
+
40
+ **Catch the legal obligations hiding in ordinary support messages before your chatbot replies.**
41
+
42
+ ![DutyGate demo: legal triggers in customer messages are held before the bot replies](https://raw.githubusercontent.com/Taimoor2500/Dutygate/main/docs/assets/dutygate-demo.gif)
43
+
44
+ "pls stop texting me" is an opt-out. "I want everything you have on me" is a privacy access
45
+ request. "My lawyer will hear about this" is a legal threat. Topic triage files these under
46
+ *billing* or *angry customer*, keyword lists miss the way people actually write, and an LLM bot
47
+ that answers "Done, you're unsubscribed!" when nothing happened creates the liability itself.
48
+
49
+ DutyGate is a small, chatbot-agnostic gate placed in front of the reply step. It reads every
50
+ inbound message, asks [TypeSafe Jev](https://docs.typesafe.ai) a batch of yes/no questions in
51
+ **one** call, applies versioned rules from a YAML **policy pack**, and tells your code what to do:
52
+
53
+ ```
54
+ inbound message ──► DUTYGATE ──► continue ──► bot replies normally
55
+ ├── route ──► holding reply + case in the right queue
56
+ └── review ──► bot may reply; a human takes a look
57
+ ```
58
+
59
+ - **Flags and routes, never acts.** It does not unsubscribe, delete, refund or reply on its
60
+ own. Your people and workflows do that.
61
+ - **Recall-first.** A miss can cost a statutory deadline; a false alarm costs a reviewer a few
62
+ seconds. Low-confidence signals go to `review`, not to `continue`.
63
+ - **Fails safe.** If TypeSafe is down, slow, or returns something malformed, you get a `review`
64
+ decision with an error code, never an exception or a dropped message.
65
+ - **Auditable.** Rules are versioned YAML, testable offline, with a published evaluation
66
+ harness and a cross-language conformance suite.
67
+
68
+ > **Not legal advice.** Which categories matter, their deadlines, and how they are routed depend
69
+ > on your jurisdiction and business. Have counsel review the packs and your holding replies
70
+ > before production use.
71
+
72
+ ## Quickstart
73
+
74
+ Offline, with no API key. The reference packs, their sample datasets and keyword lists ship
75
+ inside the package, so they can be used by name. The `keyword` backend here is the naive
76
+ baseline, which exists only for evaluation and demos.
77
+
78
+ ```console
79
+ pip install 'dutygate[server]'
80
+
81
+ dutygate validate legal-triggers outbound-claims
82
+ dutygate run legal-triggers --backend keyword --state "pls stop texting me"
83
+ dutygate eval legal-triggers --backend keyword
84
+ ```
85
+
86
+ With TypeSafe Jev, the real backend:
87
+
88
+ ```console
89
+ export TYPESAFE_API_KEY=... # from typesafe.ai
90
+ export DUTYGATE_JEV_MODEL=jev-1.13.0 # pin the model in production
91
+ dutygate run legal-triggers --state "I want everything you have on me"
92
+ dutygate eval legal-triggers # scores Jev on the bundled sample dataset
93
+ ```
94
+
95
+ ## Make it yours
96
+
97
+ Copy a reference pack, with its sample dataset and keyword list, into your project and edit
98
+ it. No code is involved: questions, thresholds, queues and new categories are all YAML.
99
+
100
+ ```console
101
+ dutygate init legal-triggers # writes legal-triggers.yaml, .dataset.jsonl, .keywords.yaml
102
+ # edit legal-triggers.yaml (and give it your own `name`), add rows to the dataset
103
+ dutygate validate legal-triggers.yaml
104
+ dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --save-answers answers.jsonl
105
+ dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --replay answers.jsonl --sweep
106
+ ```
107
+
108
+ Then point your bot at your file: `Gate.from_pack("legal-triggers.yaml")`. To measure it on
109
+ real traffic, add your own labeled messages to the dataset; see
110
+ [docs/labeling.md](docs/labeling.md). The format is in [SPEC.md](SPEC.md).
111
+
112
+ ## Use it from your bot
113
+
114
+ ### Python
115
+
116
+ ```python
117
+ from dutygate import Gate, default_holding_reply
118
+
119
+ gate = Gate.from_pack("packs/legal-triggers.yaml") # TypeSafe Jev, configured from env
120
+
121
+ decision = await gate.check_async(message, conversation_id=cid, channel="sms")
122
+ if decision.action == "route":
123
+ await create_case(decision.flags, message) # your ticketing / queue
124
+ reply = default_holding_reply(decision) # neutral; no bot improvisation
125
+ elif decision.action == "review":
126
+ await create_review_task(decision.flags, message)
127
+ if decision.error: # the gate could not judge it
128
+ await enqueue_rescan(message)
129
+ reply = await bot.reply(message)
130
+ else:
131
+ reply = await bot.reply(message)
132
+ ```
133
+
134
+ `GatedHandler` (in `dutygate.adapters.webhook`) wraps exactly this logic for any web
135
+ framework, `with_legal_gate` does the same for LangChain runnables, and
136
+ `gate_node` / `outbound_node` add DutyGate to a LangGraph graph. See
137
+ [docs/adapters.md](docs/adapters.md).
138
+
139
+ ### Any language: the HTTP sidecar
140
+
141
+ ```console
142
+ DUTYGATE_SIDECAR_KEYS=change-me TYPESAFE_API_KEY=... \
143
+ dutygate serve packs/legal-triggers.yaml packs/outbound-claims.yaml --host 0.0.0.0
144
+
145
+ curl -H 'Authorization: Bearer change-me' -H 'content-type: application/json' \
146
+ -d '{"message": "pls stop texting me", "channel": "sms"}' localhost:8080/v1/gate
147
+ ```
148
+
149
+ Or use the Docker image, the [TypeScript client](clients/typescript) (`dutygate-client`), or
150
+ any no-code tool ([n8n / Zapier recipe](examples/n8n/README.md)). See
151
+ [docs/sidecar.md](docs/sidecar.md).
152
+
153
+ ## The decision
154
+
155
+ ```json
156
+ {
157
+ "id": "dec_4f1c2a9e8b7d4c3fa1e2b3c4d5e6f708",
158
+ "action": "route",
159
+ "primary": "privacy_request",
160
+ "flags": [
161
+ {"category": "privacy_request", "rule_id": "privacy-request", "queue": "privacy",
162
+ "priority": "high", "confidence": 0.93, "level": "confident"},
163
+ {"category": "opt_out", "rule_id": "opt-out", "queue": "compliance",
164
+ "priority": "high", "confidence": 0.31, "level": "grey"}
165
+ ],
166
+ "error": null,
167
+ "policy": {"name": "legal-triggers", "version": "0.1.0"},
168
+ "backend": {"name": "typesafe-jev", "model": "jev-1.13.0"},
169
+ "conversation_id": "c_123",
170
+ "latency_ms": 212
171
+ }
172
+ ```
173
+
174
+ | `action` | meaning | your bot should |
175
+ |---|---|---|
176
+ | `continue` | no trigger | reply normally |
177
+ | `route` | at least one confident trigger | create a case, send a neutral holding reply, and not improvise about the trigger |
178
+ | `review` | a grey-zone signal, or the gate could not judge (`error` is set) | reply if you like, queue a human review, never auto-close; re-scan later if `error` is set |
179
+
180
+ `primary` is the first confident flag in the pack's rule order (or the first grey one).
181
+ Error codes and every field are specified in [SPEC.md](SPEC.md).
182
+
183
+ ## What it detects
184
+
185
+ | pack | categories |
186
+ |---|---|
187
+ | `legal-triggers` (inbound) | `legal_threat`, `regulator_complaint`, `privacy_request`, `chargeback_dispute`, `opt_out`, `accessibility_request` |
188
+ | `outbound-claims` (the bot's own reply) | `admits_liability`, `claims_action_completed`, `legal_statement`, `promises_outcome` |
189
+
190
+ One message can raise several flags ("erase my data and stop the emails"). Queues and
191
+ priorities are examples: map them to your own systems. Packs are plain YAML, so you can add
192
+ categories or tighten questions without touching code ([SPEC.md](SPEC.md)).
193
+
194
+ ## Surfaces
195
+
196
+ | surface | status |
197
+ |---|---|
198
+ | Python library (`Gate`, `evaluate`) | ✅ |
199
+ | CLI: `validate`, `run`, `eval`, `serve` | ✅ |
200
+ | HTTP sidecar + Docker image | ✅ |
201
+ | TypeScript client `dutygate-client` | ✅ passes the conformance suite end to end |
202
+ | `GatedHandler` for any web framework (FastAPI, Flask examples) | ✅ |
203
+ | LangChain `with_legal_gate` | ✅ |
204
+ | LangGraph gate and outbound nodes | ✅ |
205
+
206
+ Every surface must reproduce [`conformance/cases.json`](conformance/cases.json).
207
+
208
+ ## Evaluation
209
+
210
+ The repo ships synthetic labeled datasets, a deliberately naive keyword baseline, and a
211
+ harness that scores any backend, saves raw answers, and sweeps thresholds offline.
212
+
213
+ On the bundled `legal-triggers` set, the keyword baseline catches **48%** of triggers
214
+ (0% of non-English ones) and false-alarms on **13%** of clean messages. The datasets were
215
+ written by the same author as the baseline, so they are a sanity check, not a benchmark:
216
+ measure on your own traffic before you rely on this. See [evals/README.md](evals/README.md).
217
+
218
+ **If Jev does not beat a keyword list on your data, do not pay for it.**
219
+
220
+ ## Status
221
+
222
+ | item | state |
223
+ |---|---|
224
+ | Packs, engine, validator, CLI, eval harness, sidecar, TS client, adapters | built and tested offline (Python coverage ≥ 90%) |
225
+ | Live TypeSafe Jev integration | built against the published API reference and tested with mocked HTTP; **not yet verified against the live API** (`pytest -m live`) |
226
+ | Question wording and thresholds | untuned; tune with `dutygate eval --save-answers` then `--replay --sweep` |
227
+
228
+ ## Docs
229
+
230
+ - [SPEC.md](SPEC.md): pack format, evaluation algorithm, decision contract (normative)
231
+ - [docs/integration.md](docs/integration.md): host contract, holding replies, failure behavior, latency patterns
232
+ - [docs/sidecar.md](docs/sidecar.md): HTTP API, auth, metrics, audit log, Docker
233
+ - [docs/adapters.md](docs/adapters.md): `GatedHandler`, LangChain, LangGraph
234
+ - [docs/privacy.md](docs/privacy.md): what leaves your system, redaction, prompt injection
235
+ - [docs/labeling.md](docs/labeling.md): building a real evaluation set
236
+ - [CONTRIBUTING.md](CONTRIBUTING.md) · [SECURITY.md](SECURITY.md) · [CHANGELOG.md](CHANGELOG.md)
237
+
238
+ ## License
239
+
240
+ MIT. See [LICENSE](LICENSE).
@@ -0,0 +1,203 @@
1
+ # DutyGate
2
+
3
+ **Catch the legal obligations hiding in ordinary support messages before your chatbot replies.**
4
+
5
+ ![DutyGate demo: legal triggers in customer messages are held before the bot replies](https://raw.githubusercontent.com/Taimoor2500/Dutygate/main/docs/assets/dutygate-demo.gif)
6
+
7
+ "pls stop texting me" is an opt-out. "I want everything you have on me" is a privacy access
8
+ request. "My lawyer will hear about this" is a legal threat. Topic triage files these under
9
+ *billing* or *angry customer*, keyword lists miss the way people actually write, and an LLM bot
10
+ that answers "Done, you're unsubscribed!" when nothing happened creates the liability itself.
11
+
12
+ DutyGate is a small, chatbot-agnostic gate placed in front of the reply step. It reads every
13
+ inbound message, asks [TypeSafe Jev](https://docs.typesafe.ai) a batch of yes/no questions in
14
+ **one** call, applies versioned rules from a YAML **policy pack**, and tells your code what to do:
15
+
16
+ ```
17
+ inbound message ──► DUTYGATE ──► continue ──► bot replies normally
18
+ ├── route ──► holding reply + case in the right queue
19
+ └── review ──► bot may reply; a human takes a look
20
+ ```
21
+
22
+ - **Flags and routes, never acts.** It does not unsubscribe, delete, refund or reply on its
23
+ own. Your people and workflows do that.
24
+ - **Recall-first.** A miss can cost a statutory deadline; a false alarm costs a reviewer a few
25
+ seconds. Low-confidence signals go to `review`, not to `continue`.
26
+ - **Fails safe.** If TypeSafe is down, slow, or returns something malformed, you get a `review`
27
+ decision with an error code, never an exception or a dropped message.
28
+ - **Auditable.** Rules are versioned YAML, testable offline, with a published evaluation
29
+ harness and a cross-language conformance suite.
30
+
31
+ > **Not legal advice.** Which categories matter, their deadlines, and how they are routed depend
32
+ > on your jurisdiction and business. Have counsel review the packs and your holding replies
33
+ > before production use.
34
+
35
+ ## Quickstart
36
+
37
+ Offline, with no API key. The reference packs, their sample datasets and keyword lists ship
38
+ inside the package, so they can be used by name. The `keyword` backend here is the naive
39
+ baseline, which exists only for evaluation and demos.
40
+
41
+ ```console
42
+ pip install 'dutygate[server]'
43
+
44
+ dutygate validate legal-triggers outbound-claims
45
+ dutygate run legal-triggers --backend keyword --state "pls stop texting me"
46
+ dutygate eval legal-triggers --backend keyword
47
+ ```
48
+
49
+ With TypeSafe Jev, the real backend:
50
+
51
+ ```console
52
+ export TYPESAFE_API_KEY=... # from typesafe.ai
53
+ export DUTYGATE_JEV_MODEL=jev-1.13.0 # pin the model in production
54
+ dutygate run legal-triggers --state "I want everything you have on me"
55
+ dutygate eval legal-triggers # scores Jev on the bundled sample dataset
56
+ ```
57
+
58
+ ## Make it yours
59
+
60
+ Copy a reference pack, with its sample dataset and keyword list, into your project and edit
61
+ it. No code is involved: questions, thresholds, queues and new categories are all YAML.
62
+
63
+ ```console
64
+ dutygate init legal-triggers # writes legal-triggers.yaml, .dataset.jsonl, .keywords.yaml
65
+ # edit legal-triggers.yaml (and give it your own `name`), add rows to the dataset
66
+ dutygate validate legal-triggers.yaml
67
+ dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --save-answers answers.jsonl
68
+ dutygate eval legal-triggers.yaml legal-triggers.dataset.jsonl --replay answers.jsonl --sweep
69
+ ```
70
+
71
+ Then point your bot at your file: `Gate.from_pack("legal-triggers.yaml")`. To measure it on
72
+ real traffic, add your own labeled messages to the dataset; see
73
+ [docs/labeling.md](docs/labeling.md). The format is in [SPEC.md](SPEC.md).
74
+
75
+ ## Use it from your bot
76
+
77
+ ### Python
78
+
79
+ ```python
80
+ from dutygate import Gate, default_holding_reply
81
+
82
+ gate = Gate.from_pack("packs/legal-triggers.yaml") # TypeSafe Jev, configured from env
83
+
84
+ decision = await gate.check_async(message, conversation_id=cid, channel="sms")
85
+ if decision.action == "route":
86
+ await create_case(decision.flags, message) # your ticketing / queue
87
+ reply = default_holding_reply(decision) # neutral; no bot improvisation
88
+ elif decision.action == "review":
89
+ await create_review_task(decision.flags, message)
90
+ if decision.error: # the gate could not judge it
91
+ await enqueue_rescan(message)
92
+ reply = await bot.reply(message)
93
+ else:
94
+ reply = await bot.reply(message)
95
+ ```
96
+
97
+ `GatedHandler` (in `dutygate.adapters.webhook`) wraps exactly this logic for any web
98
+ framework, `with_legal_gate` does the same for LangChain runnables, and
99
+ `gate_node` / `outbound_node` add DutyGate to a LangGraph graph. See
100
+ [docs/adapters.md](docs/adapters.md).
101
+
102
+ ### Any language: the HTTP sidecar
103
+
104
+ ```console
105
+ DUTYGATE_SIDECAR_KEYS=change-me TYPESAFE_API_KEY=... \
106
+ dutygate serve packs/legal-triggers.yaml packs/outbound-claims.yaml --host 0.0.0.0
107
+
108
+ curl -H 'Authorization: Bearer change-me' -H 'content-type: application/json' \
109
+ -d '{"message": "pls stop texting me", "channel": "sms"}' localhost:8080/v1/gate
110
+ ```
111
+
112
+ Or use the Docker image, the [TypeScript client](clients/typescript) (`dutygate-client`), or
113
+ any no-code tool ([n8n / Zapier recipe](examples/n8n/README.md)). See
114
+ [docs/sidecar.md](docs/sidecar.md).
115
+
116
+ ## The decision
117
+
118
+ ```json
119
+ {
120
+ "id": "dec_4f1c2a9e8b7d4c3fa1e2b3c4d5e6f708",
121
+ "action": "route",
122
+ "primary": "privacy_request",
123
+ "flags": [
124
+ {"category": "privacy_request", "rule_id": "privacy-request", "queue": "privacy",
125
+ "priority": "high", "confidence": 0.93, "level": "confident"},
126
+ {"category": "opt_out", "rule_id": "opt-out", "queue": "compliance",
127
+ "priority": "high", "confidence": 0.31, "level": "grey"}
128
+ ],
129
+ "error": null,
130
+ "policy": {"name": "legal-triggers", "version": "0.1.0"},
131
+ "backend": {"name": "typesafe-jev", "model": "jev-1.13.0"},
132
+ "conversation_id": "c_123",
133
+ "latency_ms": 212
134
+ }
135
+ ```
136
+
137
+ | `action` | meaning | your bot should |
138
+ |---|---|---|
139
+ | `continue` | no trigger | reply normally |
140
+ | `route` | at least one confident trigger | create a case, send a neutral holding reply, and not improvise about the trigger |
141
+ | `review` | a grey-zone signal, or the gate could not judge (`error` is set) | reply if you like, queue a human review, never auto-close; re-scan later if `error` is set |
142
+
143
+ `primary` is the first confident flag in the pack's rule order (or the first grey one).
144
+ Error codes and every field are specified in [SPEC.md](SPEC.md).
145
+
146
+ ## What it detects
147
+
148
+ | pack | categories |
149
+ |---|---|
150
+ | `legal-triggers` (inbound) | `legal_threat`, `regulator_complaint`, `privacy_request`, `chargeback_dispute`, `opt_out`, `accessibility_request` |
151
+ | `outbound-claims` (the bot's own reply) | `admits_liability`, `claims_action_completed`, `legal_statement`, `promises_outcome` |
152
+
153
+ One message can raise several flags ("erase my data and stop the emails"). Queues and
154
+ priorities are examples: map them to your own systems. Packs are plain YAML, so you can add
155
+ categories or tighten questions without touching code ([SPEC.md](SPEC.md)).
156
+
157
+ ## Surfaces
158
+
159
+ | surface | status |
160
+ |---|---|
161
+ | Python library (`Gate`, `evaluate`) | ✅ |
162
+ | CLI: `validate`, `run`, `eval`, `serve` | ✅ |
163
+ | HTTP sidecar + Docker image | ✅ |
164
+ | TypeScript client `dutygate-client` | ✅ passes the conformance suite end to end |
165
+ | `GatedHandler` for any web framework (FastAPI, Flask examples) | ✅ |
166
+ | LangChain `with_legal_gate` | ✅ |
167
+ | LangGraph gate and outbound nodes | ✅ |
168
+
169
+ Every surface must reproduce [`conformance/cases.json`](conformance/cases.json).
170
+
171
+ ## Evaluation
172
+
173
+ The repo ships synthetic labeled datasets, a deliberately naive keyword baseline, and a
174
+ harness that scores any backend, saves raw answers, and sweeps thresholds offline.
175
+
176
+ On the bundled `legal-triggers` set, the keyword baseline catches **48%** of triggers
177
+ (0% of non-English ones) and false-alarms on **13%** of clean messages. The datasets were
178
+ written by the same author as the baseline, so they are a sanity check, not a benchmark:
179
+ measure on your own traffic before you rely on this. See [evals/README.md](evals/README.md).
180
+
181
+ **If Jev does not beat a keyword list on your data, do not pay for it.**
182
+
183
+ ## Status
184
+
185
+ | item | state |
186
+ |---|---|
187
+ | Packs, engine, validator, CLI, eval harness, sidecar, TS client, adapters | built and tested offline (Python coverage ≥ 90%) |
188
+ | Live TypeSafe Jev integration | built against the published API reference and tested with mocked HTTP; **not yet verified against the live API** (`pytest -m live`) |
189
+ | Question wording and thresholds | untuned; tune with `dutygate eval --save-answers` then `--replay --sweep` |
190
+
191
+ ## Docs
192
+
193
+ - [SPEC.md](SPEC.md): pack format, evaluation algorithm, decision contract (normative)
194
+ - [docs/integration.md](docs/integration.md): host contract, holding replies, failure behavior, latency patterns
195
+ - [docs/sidecar.md](docs/sidecar.md): HTTP API, auth, metrics, audit log, Docker
196
+ - [docs/adapters.md](docs/adapters.md): `GatedHandler`, LangChain, LangGraph
197
+ - [docs/privacy.md](docs/privacy.md): what leaves your system, redaction, prompt injection
198
+ - [docs/labeling.md](docs/labeling.md): building a real evaluation set
199
+ - [CONTRIBUTING.md](CONTRIBUTING.md) · [SECURITY.md](SECURITY.md) · [CHANGELOG.md](CHANGELOG.md)
200
+
201
+ ## License
202
+
203
+ MIT. See [LICENSE](LICENSE).