grounding-gate 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- grounding_gate/__init__.py +30 -0
- grounding_gate/boundary.py +103 -0
- grounding_gate/classifier.py +45 -0
- grounding_gate/state.py +67 -0
- grounding_gate-0.1.0.dist-info/METADATA +194 -0
- grounding_gate-0.1.0.dist-info/RECORD +8 -0
- grounding_gate-0.1.0.dist-info/WHEEL +4 -0
- grounding_gate-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,30 @@
|
|
|
1
|
+
"""Grounding Gate — zero-token structural verifier for agent loops.
|
|
2
|
+
|
|
3
|
+
One choke point at the submit boundary enforces:
|
|
4
|
+
G (grounding): terminal claims require a qualifying observation this turn
|
|
5
|
+
(novel AND relevant AND consequence-tier-correct).
|
|
6
|
+
B (budget): grounded steps refill rope, pure reasoning decrements; exhaustion
|
|
7
|
+
halts to {qualifying call | typed `unverified` terminal}.
|
|
8
|
+
|
|
9
|
+
Hash/set/integer operations only — no LLM calls anywhere in the gate.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from .state import PRESETS, GateState, extract_identifiers, normalize
|
|
13
|
+
from .classifier import classify_observation
|
|
14
|
+
from .boundary import ACCEPT, LEGAL_NEXT, REJECT, boundary_check, turn_loop
|
|
15
|
+
|
|
16
|
+
__version__ = "0.1.0"
|
|
17
|
+
|
|
18
|
+
__all__ = [
|
|
19
|
+
"ACCEPT",
|
|
20
|
+
"LEGAL_NEXT",
|
|
21
|
+
"PRESETS",
|
|
22
|
+
"REJECT",
|
|
23
|
+
"GateState",
|
|
24
|
+
"boundary_check",
|
|
25
|
+
"classify_observation",
|
|
26
|
+
"extract_identifiers",
|
|
27
|
+
"normalize",
|
|
28
|
+
"turn_loop",
|
|
29
|
+
"__version__",
|
|
30
|
+
]
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
"""Submit-boundary choke point and reference turn loop — spec module 4.
|
|
2
|
+
|
|
3
|
+
``boundary_check`` must be the SOLE path to any terminal output. ``turn_loop``
|
|
4
|
+
is the reference wiring that drives a scripted agent through the gate — use it
|
|
5
|
+
as the template for integrating the gate into a real agent loop (and in tests
|
|
6
|
+
and demos, where scripted steps make behavior deterministic).
|
|
7
|
+
|
|
8
|
+
Four corrections survived adversarial review of the original draft — all in
|
|
9
|
+
``turn_loop``, between the individually-passing acceptance cases:
|
|
10
|
+
(C1) grounding flags are LATCHES within a turn, not last-call assignments —
|
|
11
|
+
otherwise a later non-completion-grade call overwrites a valid
|
|
12
|
+
verification back to false (false-rejects legitimate work).
|
|
13
|
+
(C2) halt is cleared ONLY by a QUALIFYING observation — clearing on any tool
|
|
14
|
+
call lets a halted model escape via a novelty-defeated no-op read.
|
|
15
|
+
(C4) refused reasoning still decrements budget and surfaces ``legal_next``
|
|
16
|
+
(re-prompt) — otherwise a halted spinner livelocks for free.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from .classifier import classify_observation
|
|
20
|
+
|
|
21
|
+
ACCEPT, REJECT = "ACCEPT", "REJECT"
|
|
22
|
+
LEGAL_NEXT = ["qualifying_tool_call", "unverified_terminal"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def boundary_check(terminal_attempt, state):
|
|
26
|
+
"""THE choke point — must be the sole path to any terminal output."""
|
|
27
|
+
ct = terminal_attempt["claim_type"] # none|assertion|completion|unverified
|
|
28
|
+
|
|
29
|
+
if ct == "unverified": # universal escape hatch (typed)
|
|
30
|
+
state.halted = False
|
|
31
|
+
return {"verdict": ACCEPT, "legal_next": []}
|
|
32
|
+
|
|
33
|
+
claim_bearing = ct in ("assertion", "completion")
|
|
34
|
+
|
|
35
|
+
if claim_bearing and state.budget <= 0:
|
|
36
|
+
state.halted = True
|
|
37
|
+
return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
|
|
38
|
+
|
|
39
|
+
if ct == "completion":
|
|
40
|
+
if not state.verified_this_turn or any(
|
|
41
|
+
s not in state.verified_signals for s in state.goal_predicates):
|
|
42
|
+
state.halted = True
|
|
43
|
+
return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
|
|
44
|
+
state.halted = False
|
|
45
|
+
return {"verdict": ACCEPT, "legal_next": []}
|
|
46
|
+
|
|
47
|
+
if ct == "assertion":
|
|
48
|
+
# strict G (skipper preset): observed tier is not enough — even an
|
|
49
|
+
# assertion needs verified-tier grounding (spec module 6)
|
|
50
|
+
grounded = state.verified_this_turn if state.strict_g else state.grounded_this_turn
|
|
51
|
+
if not grounded:
|
|
52
|
+
state.halted = True
|
|
53
|
+
return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
|
|
54
|
+
state.halted = False
|
|
55
|
+
return {"verdict": ACCEPT, "legal_next": []}
|
|
56
|
+
|
|
57
|
+
return {"verdict": ACCEPT, "legal_next": []} # ct == none: exempt
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def turn_loop(script, state):
|
|
61
|
+
"""Drive a scripted agent through the gate. Returns ``(emitted, trace)``.
|
|
62
|
+
|
|
63
|
+
Script steps:
|
|
64
|
+
{"type": "reasoning"}
|
|
65
|
+
{"type": "tool_call", "tool", "args", "result",
|
|
66
|
+
"mutating": bool = False, "signals": list = None, "exit_ok": bool = True}
|
|
67
|
+
{"type": "terminal", "attempt": {"claim_type", "content"}}
|
|
68
|
+
"""
|
|
69
|
+
trace = []
|
|
70
|
+
for step in script:
|
|
71
|
+
if step["type"] == "reasoning":
|
|
72
|
+
state.budget -= 1 # (C4) refusals starve too
|
|
73
|
+
if state.halted:
|
|
74
|
+
trace.append(("refused_reasoning", LEGAL_NEXT))
|
|
75
|
+
continue
|
|
76
|
+
trace.append(("reasoning", None))
|
|
77
|
+
continue
|
|
78
|
+
|
|
79
|
+
if step["type"] == "tool_call":
|
|
80
|
+
state.current_step += 1
|
|
81
|
+
obs = classify_observation(step["tool"], step["args"], step["result"],
|
|
82
|
+
state, read_only=not step.get("mutating", False))
|
|
83
|
+
qualifying = obs["grounds_assertion"] or obs["grounds_completion"]
|
|
84
|
+
state.grounded_this_turn |= obs["grounds_assertion"] # (C1) latch
|
|
85
|
+
state.verified_this_turn |= obs["grounds_completion"] # (C1) latch
|
|
86
|
+
if step.get("signals") and step.get("exit_ok", True):
|
|
87
|
+
# declarative-rails signal mapper (reference: script-declared)
|
|
88
|
+
state.verified_signals |= set(step["signals"])
|
|
89
|
+
if qualifying:
|
|
90
|
+
state.budget = min(state.budget + state.refill, state.cap)
|
|
91
|
+
state.halted = False # (C2) qualifying only
|
|
92
|
+
if step.get("mutating"):
|
|
93
|
+
state.last_mutation_step = state.current_step
|
|
94
|
+
trace.append(("tool_call", obs))
|
|
95
|
+
continue
|
|
96
|
+
|
|
97
|
+
if step["type"] == "terminal":
|
|
98
|
+
v = boundary_check(step["attempt"], state)
|
|
99
|
+
trace.append(("terminal", v["verdict"]))
|
|
100
|
+
if v["verdict"] == ACCEPT:
|
|
101
|
+
return step["attempt"]["content"], trace # SOLE terminal emit
|
|
102
|
+
continue
|
|
103
|
+
return None, trace
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Observation classifier — spec module 2, the hard part.
|
|
2
|
+
|
|
3
|
+
A completed tool call grounds a claim only if it is
|
|
4
|
+
novel AND relevant AND consequence-tier-correct.
|
|
5
|
+
|
|
6
|
+
Two corrections survived adversarial review of the original draft (the drafting
|
|
7
|
+
model had also authored the test that ratified its own bug — acceptance cases
|
|
8
|
+
here are authored by the reviewer, never the generator):
|
|
9
|
+
(C1) a completion requires a mutation to have OCCURRED
|
|
10
|
+
(``last_mutation_step > 0``) — otherwise a plain read grounds a
|
|
11
|
+
"I changed X" claim when nothing was ever changed.
|
|
12
|
+
(C3) the novelty hash is recorded only AFTER the relevance gate passes —
|
|
13
|
+
otherwise a novel-but-irrelevant read burns its hash and is wrongly
|
|
14
|
+
denied if it later becomes relevant.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from .state import extract_identifiers, normalize
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def classify_observation(tool_name, args, result, state, read_only):
|
|
21
|
+
"""Decide whether one completed tool call flips grounding.
|
|
22
|
+
|
|
23
|
+
Returns ``{"grounds_assertion": bool, "grounds_completion": bool}``.
|
|
24
|
+
Mutates ``state.recent_result_hashes`` for qualifying novel calls.
|
|
25
|
+
"""
|
|
26
|
+
ret = {"grounds_assertion": False, "grounds_completion": False}
|
|
27
|
+
|
|
28
|
+
h = None
|
|
29
|
+
if tool_name not in state.novelty_exempt: # 1. NOVELTY
|
|
30
|
+
h = hash((tool_name, normalize(args), normalize(result)))
|
|
31
|
+
if h in state.recent_result_hashes:
|
|
32
|
+
return ret
|
|
33
|
+
|
|
34
|
+
idents = extract_identifiers(args, result) # 2. RELEVANCE
|
|
35
|
+
if not (idents & state.claim_surface):
|
|
36
|
+
return ret
|
|
37
|
+
|
|
38
|
+
if h is not None: # (C3)
|
|
39
|
+
state.recent_result_hashes.add(h)
|
|
40
|
+
|
|
41
|
+
if read_only: # 3. CONSEQUENCE
|
|
42
|
+
ret["grounds_assertion"] = True
|
|
43
|
+
if state.last_mutation_step > 0 and state.current_step > state.last_mutation_step:
|
|
44
|
+
ret["grounds_completion"] = True # (C1)
|
|
45
|
+
return ret
|
grounding_gate/state.py
ADDED
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
"""Gate state, model-class presets, and the normalization helpers.
|
|
2
|
+
|
|
3
|
+
Spec modules 1 (state container) and 6 (per-model-class presets): fleet
|
|
4
|
+
variance is absorbed as integers, not prose.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
|
|
10
|
+
# Per-model-class presets. Two documented agent failure modes get their own
|
|
11
|
+
# tuning: "skipper" models emit confident terminals without observing reality
|
|
12
|
+
# (strict grounding), "diverger" models reason in closed context until a
|
|
13
|
+
# confident wrong answer ships (small budget, small refill — starves loops).
|
|
14
|
+
PRESETS = {
|
|
15
|
+
"skipper": {"CAP": 5, "REFILL": 2, "strict_g": True},
|
|
16
|
+
"diverger": {"CAP": 4, "REFILL": 1, "strict_g": False},
|
|
17
|
+
"default": {"CAP": 6, "REFILL": 2, "strict_g": False},
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class GateState:
|
|
23
|
+
budget: int
|
|
24
|
+
cap: int
|
|
25
|
+
refill: int
|
|
26
|
+
# strict G (skipper preset): even assertions require the verified tier
|
|
27
|
+
strict_g: bool = False
|
|
28
|
+
# novelty
|
|
29
|
+
recent_result_hashes: set = field(default_factory=set)
|
|
30
|
+
novelty_exempt: set = field(default_factory=set)
|
|
31
|
+
# relevance
|
|
32
|
+
claim_surface: set = field(default_factory=set)
|
|
33
|
+
# consequence
|
|
34
|
+
last_mutation_step: int = 0 # 0 = no mutation has EVER occurred
|
|
35
|
+
current_step: int = 0
|
|
36
|
+
# per-turn latches
|
|
37
|
+
grounded_this_turn: bool = False
|
|
38
|
+
verified_this_turn: bool = False
|
|
39
|
+
# declarative rails
|
|
40
|
+
verified_signals: set = field(default_factory=set)
|
|
41
|
+
goal_predicates: list = field(default_factory=list)
|
|
42
|
+
halted: bool = False
|
|
43
|
+
|
|
44
|
+
@classmethod
|
|
45
|
+
def for_model_class(cls, model_class="default", **kw):
|
|
46
|
+
p = PRESETS[model_class]
|
|
47
|
+
kw.setdefault("strict_g", p["strict_g"])
|
|
48
|
+
return cls(budget=p["CAP"], cap=p["CAP"], refill=p["REFILL"], **kw)
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def normalize(text):
|
|
52
|
+
"""Strip nondeterminism before hashing. Reference: ISO timestamps + hex ids.
|
|
53
|
+
|
|
54
|
+
Real deployments extend this per-tool; too-weak normalization means novelty
|
|
55
|
+
never fires on noisy tools (the no-op defense weakens).
|
|
56
|
+
"""
|
|
57
|
+
text = re.sub(r"\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}\S*", "<TS>", str(text))
|
|
58
|
+
return re.sub(r"\b[0-9a-f]{8,}\b", "<HEX>", text)
|
|
59
|
+
|
|
60
|
+
|
|
61
|
+
def extract_identifiers(args, result):
|
|
62
|
+
"""Conservative token extraction for the relevance check.
|
|
63
|
+
|
|
64
|
+
Over-extraction leaks relevance; under-extraction false-rejects
|
|
65
|
+
cross-cutting work (open risk, flagged in docs/module-2-classifier.md).
|
|
66
|
+
"""
|
|
67
|
+
return set(re.findall(r"[\w.\-/]+", f"{args} {result}"))
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: grounding-gate
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Zero-token structural verifier for agent loops: one choke point at the submit boundary enforcing grounding and budget invariants. No LLM calls, stdlib only.
|
|
5
|
+
Project-URL: Homepage, https://github.com/CiphemonJY/grounding-gate
|
|
6
|
+
Project-URL: Repository, https://github.com/CiphemonJY/grounding-gate
|
|
7
|
+
Project-URL: Issues, https://github.com/CiphemonJY/grounding-gate/issues
|
|
8
|
+
Project-URL: Changelog, https://github.com/CiphemonJY/grounding-gate/blob/main/CHANGELOG.md
|
|
9
|
+
Author: Ciphemon
|
|
10
|
+
License-Expression: MIT
|
|
11
|
+
License-File: LICENSE
|
|
12
|
+
Keywords: agent-loop,agents,grounding,guardrails,hallucination,llm,verification
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
21
|
+
Classifier: Topic :: Software Development :: Quality Assurance
|
|
22
|
+
Requires-Python: >=3.9
|
|
23
|
+
Provides-Extra: dev
|
|
24
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
25
|
+
Description-Content-Type: text/markdown
|
|
26
|
+
|
|
27
|
+
# grounding-gate
|
|
28
|
+
|
|
29
|
+
[](https://github.com/CiphemonJY/grounding-gate/actions/workflows/ci.yml)
|
|
30
|
+
|
|
31
|
+
**Zero-token structural verifier for agent loops.** One choke point at the
|
|
32
|
+
submit boundary decides whether an agent is allowed to say "X is true" or
|
|
33
|
+
"I did X" — using hash, set, and integer operations only. No LLM calls, no
|
|
34
|
+
per-turn prompt injection, no dependencies.
|
|
35
|
+
|
|
36
|
+
```
|
|
37
|
+
pip install grounding-gate # stdlib only, Python >= 3.9
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
The demo ships in the repo (not the wheel):
|
|
41
|
+
|
|
42
|
+
```
|
|
43
|
+
git clone https://github.com/CiphemonJY/grounding-gate && cd grounding-gate
|
|
44
|
+
python examples/demo.py # the whole idea in 30 seconds
|
|
45
|
+
```
|
|
46
|
+
|
|
47
|
+
## The problem
|
|
48
|
+
|
|
49
|
+
Agents fail in two characteristic ways, and both ship *confident* wrong answers:
|
|
50
|
+
|
|
51
|
+
- **Skip-and-hallucinate** — emit a terminal claim ("done, config fixed")
|
|
52
|
+
without ever observing reality after acting on it.
|
|
53
|
+
- **Reason-and-diverge** — loop in closed context, burning steps on
|
|
54
|
+
reasoning about stale beliefs, until a confident wrong answer ships.
|
|
55
|
+
|
|
56
|
+
The standard fix is prose: "remember to verify your work" injected into every
|
|
57
|
+
turn. Prose costs tokens on every turn, behaves differently per model, and —
|
|
58
|
+
critically — is *skippable*. A reminder is not an invariant.
|
|
59
|
+
|
|
60
|
+
## The idea
|
|
61
|
+
|
|
62
|
+
Move enforcement out of the prompt and into **control flow**. A single gate
|
|
63
|
+
wraps the submit/conclude boundary, and a terminal output is emitted only if
|
|
64
|
+
both invariants hold:
|
|
65
|
+
|
|
66
|
+
- **G (grounding)** — a *qualifying* observation happened this turn, or the
|
|
67
|
+
output makes no factual claim. Qualifying means **novel** (result hash not
|
|
68
|
+
seen before, after stripping timestamps/ids) **∧ relevant** (touches the
|
|
69
|
+
identifiers the claim is about) **∧ consequence-tier-correct** (see below).
|
|
70
|
+
- **B (budget)** — reasoning rope remains. Qualifying observations *refill*
|
|
71
|
+
the budget (up to a cap); pure reasoning steps decrement it. Grounded work
|
|
72
|
+
runs effectively unbounded; closed-loop reasoning hits a hard floor.
|
|
73
|
+
|
|
74
|
+
Fail either → the terminal is **rejected** and the agent is told its only
|
|
75
|
+
legal moves: make a qualifying tool call, or exit with a typed **`unverified`**
|
|
76
|
+
terminal. `unverified` is a first-class, always-legal escape hatch — the gate
|
|
77
|
+
never traps an agent, it only forbids *confident* ungrounded claims.
|
|
78
|
+
|
|
79
|
+
### Consequence tiers
|
|
80
|
+
|
|
81
|
+
The gate distinguishes what kind of claim an observation can support:
|
|
82
|
+
|
|
83
|
+
| Claim type | Example | Requires |
|
|
84
|
+
|--------------|--------------------------|----------|
|
|
85
|
+
| `assertion` | "X is true" | a novel, relevant, read-only observation this turn |
|
|
86
|
+
| `completion` | "I changed X" | a novel, relevant read taken **after** the mutation — a mutating call never self-grounds its own effect |
|
|
87
|
+
| `unverified` | "couldn't confirm X" | nothing — always legal |
|
|
88
|
+
| `none` | no factual claim | nothing — exempt |
|
|
89
|
+
|
|
90
|
+
That second row is the heart of it: *writing a file and claiming success is
|
|
91
|
+
not verification; reading it back afterwards is.*
|
|
92
|
+
|
|
93
|
+
## Quickstart
|
|
94
|
+
|
|
95
|
+
```python
|
|
96
|
+
from grounding_gate import GateState, classify_observation, boundary_check
|
|
97
|
+
|
|
98
|
+
state = GateState.for_model_class("default", claim_surface={"app.cfg"})
|
|
99
|
+
|
|
100
|
+
# after EVERY tool call in your agent loop:
|
|
101
|
+
state.current_step += 1
|
|
102
|
+
obs = classify_observation(tool, args, result, state, read_only=not mutating)
|
|
103
|
+
state.grounded_this_turn |= obs["grounds_assertion"]
|
|
104
|
+
state.verified_this_turn |= obs["grounds_completion"]
|
|
105
|
+
if mutating:
|
|
106
|
+
state.last_mutation_step = state.current_step # a completion now needs a read AFTER this
|
|
107
|
+
|
|
108
|
+
# at every submit/conclude attempt — this must be the ONLY path to output:
|
|
109
|
+
verdict = boundary_check({"claim_type": "completion", "content": answer}, state)
|
|
110
|
+
if verdict["verdict"] == "REJECT":
|
|
111
|
+
... # surface verdict["legal_next"] to the model and continue the loop
|
|
112
|
+
```
|
|
113
|
+
|
|
114
|
+
Note the mutation bookkeeping: without `last_mutation_step` ever being set, no
|
|
115
|
+
read can reach the verified tier and a `completion` can never be accepted —
|
|
116
|
+
that is the gate working as designed, not a bug.
|
|
117
|
+
|
|
118
|
+
`turn_loop` in [boundary.py](https://github.com/CiphemonJY/grounding-gate/blob/main/src/grounding_gate/boundary.py)
|
|
119
|
+
is the complete reference wiring (budget refill, mutation tracking, halt
|
|
120
|
+
semantics, signal mapping) — use it as the integration template. The
|
|
121
|
+
[demo](https://github.com/CiphemonJY/grounding-gate/blob/main/examples/demo.py)
|
|
122
|
+
runs the same scripted agent through an ungated and a gated loop, side by side.
|
|
123
|
+
|
|
124
|
+
## Model-class presets
|
|
125
|
+
|
|
126
|
+
Fleet variance is absorbed as integers, not prose. Pick the preset matching
|
|
127
|
+
how your model fails:
|
|
128
|
+
|
|
129
|
+
| Preset | CAP | REFILL | Strict G | For |
|
|
130
|
+
|------------|-----|--------|----------|-----|
|
|
131
|
+
| `skipper` | 5 | 2 | yes | models that hallucinate-and-skip |
|
|
132
|
+
| `diverger` | 4 | 1 | no | models that reason forever |
|
|
133
|
+
| `default` | 6 | 2 | no | everything else |
|
|
134
|
+
|
|
135
|
+
Strict G means even plain *assertions* require verified-tier grounding (a
|
|
136
|
+
post-mutation observation) — an observed-tier read is not enough. In a task
|
|
137
|
+
that never mutates anything, a strict-G agent can only exit via the typed
|
|
138
|
+
`unverified` terminal; that hard line is the point of the skipper preset, so
|
|
139
|
+
pick `default` for read-only/Q&A workloads.
|
|
140
|
+
|
|
141
|
+
## Declarative rails
|
|
142
|
+
|
|
143
|
+
A task can declare signals that must be verified before any completion is
|
|
144
|
+
accepted (`state.goal_predicates = ["tests_passed"]`). The gate never
|
|
145
|
+
interprets meaning — it only checks that a signal named `tests_passed` was
|
|
146
|
+
registered by a mapped, real command outcome. Semantic judgment stays out of
|
|
147
|
+
the floor by design.
|
|
148
|
+
|
|
149
|
+
## What the gate does NOT do
|
|
150
|
+
|
|
151
|
+
Honest scope, from the design's leak audit:
|
|
152
|
+
|
|
153
|
+
- **No semantic correctness.** A grounded claim can still be wrong (the model
|
|
154
|
+
can misread a real result). That is punted to a declared verifier tier
|
|
155
|
+
(`verify_with`), not smuggled into the floor.
|
|
156
|
+
- **Relevance can be spoofed** by a model that deliberately mentions the right
|
|
157
|
+
identifiers in an irrelevant call. The floor defends against *lazy*
|
|
158
|
+
ungroundedness, which is the overwhelmingly common failure; adversarial
|
|
159
|
+
self-deception needs the verifier tier.
|
|
160
|
+
- **Nondeterministic tools** need per-tool `normalize()` extensions or the
|
|
161
|
+
audited `novelty_exempt` allowlist — too-weak normalization quietly disables
|
|
162
|
+
the no-op defense.
|
|
163
|
+
|
|
164
|
+
## How this was built
|
|
165
|
+
|
|
166
|
+
The modules were drafted by different LLMs and adversarially reviewed before
|
|
167
|
+
assembly; the final behavior is pinned by a 19-case acceptance suite
|
|
168
|
+
([tests/test_gate.py](https://github.com/CiphemonJY/grounding-gate/blob/main/tests/test_gate.py))
|
|
169
|
+
that runs on bare Python with zero dependencies. Two review findings shaped
|
|
170
|
+
the method and are preserved in the docstrings:
|
|
171
|
+
|
|
172
|
+
- A drafting model shipped a consequence-tier bug **and authored the test that
|
|
173
|
+
ratified it** — since then, expected outcomes are authored by the reviewer,
|
|
174
|
+
never by the generator
|
|
175
|
+
([docs/module-2-classifier.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/module-2-classifier.md)).
|
|
176
|
+
- The remaining leaks lived *between* individually-passing test cases —
|
|
177
|
+
latch-vs-assignment, halt cleared by non-qualifying calls
|
|
178
|
+
([docs/module-4-boundary.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/module-4-boundary.md)).
|
|
179
|
+
|
|
180
|
+
Full design spec:
|
|
181
|
+
[docs/spec.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/spec.md).
|
|
182
|
+
|
|
183
|
+
## Status & roadmap
|
|
184
|
+
|
|
185
|
+
This is the reference implementation — correct, minimal, and framework-free.
|
|
186
|
+
Planned next:
|
|
187
|
+
|
|
188
|
+
- Adapters: Claude Agent SDK hook, LangGraph middleware, OpenAI Agents SDK.
|
|
189
|
+
- A real signal-mapper module (command exit code → declared signal).
|
|
190
|
+
- Empirical preset tuning across model classes.
|
|
191
|
+
|
|
192
|
+
## License
|
|
193
|
+
|
|
194
|
+
MIT
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
grounding_gate/__init__.py,sha256=Zm7VbfHRzhVfmg8Vtt520Pqm3CvAy7Uz5ua_9fGQn_Y,910
|
|
2
|
+
grounding_gate/boundary.py,sha256=JU4ge0mNaqorau4iEFmgy2UkmMFArKoon3_qR8j1Z3Q,4717
|
|
3
|
+
grounding_gate/classifier.py,sha256=KgldOBV9fkzsCyBTdYcSDwX__37T28sH4GBa645Le04,1968
|
|
4
|
+
grounding_gate/state.py,sha256=2xXDSPPBtw5O012oGmVe2lYz0nwy2WCOsyElBcKzF0E,2425
|
|
5
|
+
grounding_gate-0.1.0.dist-info/METADATA,sha256=FmruvXaYWLsLGPSlqAzOERHgEWWEbAdNH7Ygg0cZRtE,8946
|
|
6
|
+
grounding_gate-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
7
|
+
grounding_gate-0.1.0.dist-info/licenses/LICENSE,sha256=76kJpx8qwE8s8qdKjsnngPtznV3WasZ54ed4WWW2y7o,1065
|
|
8
|
+
grounding_gate-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Ciphemon
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|