grounding-gate 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,30 @@
1
+ """Grounding Gate — zero-token structural verifier for agent loops.
2
+
3
+ One choke point at the submit boundary enforces:
4
+ G (grounding): terminal claims require a qualifying observation this turn
5
+ (novel AND relevant AND consequence-tier-correct).
6
+ B (budget): grounded steps refill rope, pure reasoning decrements; exhaustion
7
+ halts to {qualifying call | typed `unverified` terminal}.
8
+
9
+ Hash/set/integer operations only — no LLM calls anywhere in the gate.
10
+ """
11
+
12
+ from .state import PRESETS, GateState, extract_identifiers, normalize
13
+ from .classifier import classify_observation
14
+ from .boundary import ACCEPT, LEGAL_NEXT, REJECT, boundary_check, turn_loop
15
+
16
+ __version__ = "0.1.0"
17
+
18
+ __all__ = [
19
+ "ACCEPT",
20
+ "LEGAL_NEXT",
21
+ "PRESETS",
22
+ "REJECT",
23
+ "GateState",
24
+ "boundary_check",
25
+ "classify_observation",
26
+ "extract_identifiers",
27
+ "normalize",
28
+ "turn_loop",
29
+ "__version__",
30
+ ]
@@ -0,0 +1,103 @@
1
+ """Submit-boundary choke point and reference turn loop — spec module 4.
2
+
3
+ ``boundary_check`` must be the SOLE path to any terminal output. ``turn_loop``
4
+ is the reference wiring that drives a scripted agent through the gate — use it
5
+ as the template for integrating the gate into a real agent loop (and in tests
6
+ and demos, where scripted steps make behavior deterministic).
7
+
8
+ Four corrections survived adversarial review of the original draft — all in
9
+ ``turn_loop``, between the individually-passing acceptance cases:
10
+ (C1) grounding flags are LATCHES within a turn, not last-call assignments —
11
+ otherwise a later non-completion-grade call overwrites a valid
12
+ verification back to false (false-rejects legitimate work).
13
+ (C2) halt is cleared ONLY by a QUALIFYING observation — clearing on any tool
14
+ call lets a halted model escape via a novelty-defeated no-op read.
15
+ (C4) refused reasoning still decrements budget and surfaces ``legal_next``
16
+ (re-prompt) — otherwise a halted spinner livelocks for free.
17
+ """
18
+
19
+ from .classifier import classify_observation
20
+
21
+ ACCEPT, REJECT = "ACCEPT", "REJECT"
22
+ LEGAL_NEXT = ["qualifying_tool_call", "unverified_terminal"]
23
+
24
+
25
+ def boundary_check(terminal_attempt, state):
26
+ """THE choke point — must be the sole path to any terminal output."""
27
+ ct = terminal_attempt["claim_type"] # none|assertion|completion|unverified
28
+
29
+ if ct == "unverified": # universal escape hatch (typed)
30
+ state.halted = False
31
+ return {"verdict": ACCEPT, "legal_next": []}
32
+
33
+ claim_bearing = ct in ("assertion", "completion")
34
+
35
+ if claim_bearing and state.budget <= 0:
36
+ state.halted = True
37
+ return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
38
+
39
+ if ct == "completion":
40
+ if not state.verified_this_turn or any(
41
+ s not in state.verified_signals for s in state.goal_predicates):
42
+ state.halted = True
43
+ return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
44
+ state.halted = False
45
+ return {"verdict": ACCEPT, "legal_next": []}
46
+
47
+ if ct == "assertion":
48
+ # strict G (skipper preset): observed tier is not enough — even an
49
+ # assertion needs verified-tier grounding (spec module 6)
50
+ grounded = state.verified_this_turn if state.strict_g else state.grounded_this_turn
51
+ if not grounded:
52
+ state.halted = True
53
+ return {"verdict": REJECT, "legal_next": LEGAL_NEXT}
54
+ state.halted = False
55
+ return {"verdict": ACCEPT, "legal_next": []}
56
+
57
+ return {"verdict": ACCEPT, "legal_next": []} # ct == none: exempt
58
+
59
+
60
+ def turn_loop(script, state):
61
+ """Drive a scripted agent through the gate. Returns ``(emitted, trace)``.
62
+
63
+ Script steps:
64
+ {"type": "reasoning"}
65
+ {"type": "tool_call", "tool", "args", "result",
66
+ "mutating": bool = False, "signals": list = None, "exit_ok": bool = True}
67
+ {"type": "terminal", "attempt": {"claim_type", "content"}}
68
+ """
69
+ trace = []
70
+ for step in script:
71
+ if step["type"] == "reasoning":
72
+ state.budget -= 1 # (C4) refusals starve too
73
+ if state.halted:
74
+ trace.append(("refused_reasoning", LEGAL_NEXT))
75
+ continue
76
+ trace.append(("reasoning", None))
77
+ continue
78
+
79
+ if step["type"] == "tool_call":
80
+ state.current_step += 1
81
+ obs = classify_observation(step["tool"], step["args"], step["result"],
82
+ state, read_only=not step.get("mutating", False))
83
+ qualifying = obs["grounds_assertion"] or obs["grounds_completion"]
84
+ state.grounded_this_turn |= obs["grounds_assertion"] # (C1) latch
85
+ state.verified_this_turn |= obs["grounds_completion"] # (C1) latch
86
+ if step.get("signals") and step.get("exit_ok", True):
87
+ # declarative-rails signal mapper (reference: script-declared)
88
+ state.verified_signals |= set(step["signals"])
89
+ if qualifying:
90
+ state.budget = min(state.budget + state.refill, state.cap)
91
+ state.halted = False # (C2) qualifying only
92
+ if step.get("mutating"):
93
+ state.last_mutation_step = state.current_step
94
+ trace.append(("tool_call", obs))
95
+ continue
96
+
97
+ if step["type"] == "terminal":
98
+ v = boundary_check(step["attempt"], state)
99
+ trace.append(("terminal", v["verdict"]))
100
+ if v["verdict"] == ACCEPT:
101
+ return step["attempt"]["content"], trace # SOLE terminal emit
102
+ continue
103
+ return None, trace
@@ -0,0 +1,45 @@
1
+ """Observation classifier — spec module 2, the hard part.
2
+
3
+ A completed tool call grounds a claim only if it is
4
+ novel AND relevant AND consequence-tier-correct.
5
+
6
+ Two corrections survived adversarial review of the original draft (the drafting
7
+ model had also authored the test that ratified its own bug — acceptance cases
8
+ here are authored by the reviewer, never the generator):
9
+ (C1) a completion requires a mutation to have OCCURRED
10
+ (``last_mutation_step > 0``) — otherwise a plain read grounds a
11
+ "I changed X" claim when nothing was ever changed.
12
+ (C3) the novelty hash is recorded only AFTER the relevance gate passes —
13
+ otherwise a novel-but-irrelevant read burns its hash and is wrongly
14
+ denied if it later becomes relevant.
15
+ """
16
+
17
+ from .state import extract_identifiers, normalize
18
+
19
+
20
+ def classify_observation(tool_name, args, result, state, read_only):
21
+ """Decide whether one completed tool call flips grounding.
22
+
23
+ Returns ``{"grounds_assertion": bool, "grounds_completion": bool}``.
24
+ Mutates ``state.recent_result_hashes`` for qualifying novel calls.
25
+ """
26
+ ret = {"grounds_assertion": False, "grounds_completion": False}
27
+
28
+ h = None
29
+ if tool_name not in state.novelty_exempt: # 1. NOVELTY
30
+ h = hash((tool_name, normalize(args), normalize(result)))
31
+ if h in state.recent_result_hashes:
32
+ return ret
33
+
34
+ idents = extract_identifiers(args, result) # 2. RELEVANCE
35
+ if not (idents & state.claim_surface):
36
+ return ret
37
+
38
+ if h is not None: # (C3)
39
+ state.recent_result_hashes.add(h)
40
+
41
+ if read_only: # 3. CONSEQUENCE
42
+ ret["grounds_assertion"] = True
43
+ if state.last_mutation_step > 0 and state.current_step > state.last_mutation_step:
44
+ ret["grounds_completion"] = True # (C1)
45
+ return ret
@@ -0,0 +1,67 @@
1
+ """Gate state, model-class presets, and the normalization helpers.
2
+
3
+ Spec modules 1 (state container) and 6 (per-model-class presets): fleet
4
+ variance is absorbed as integers, not prose.
5
+ """
6
+
7
+ import re
8
+ from dataclasses import dataclass, field
9
+
10
+ # Per-model-class presets. Two documented agent failure modes get their own
11
+ # tuning: "skipper" models emit confident terminals without observing reality
12
+ # (strict grounding), "diverger" models reason in closed context until a
13
+ # confident wrong answer ships (small budget, small refill — starves loops).
14
+ PRESETS = {
15
+ "skipper": {"CAP": 5, "REFILL": 2, "strict_g": True},
16
+ "diverger": {"CAP": 4, "REFILL": 1, "strict_g": False},
17
+ "default": {"CAP": 6, "REFILL": 2, "strict_g": False},
18
+ }
19
+
20
+
21
+ @dataclass
22
+ class GateState:
23
+ budget: int
24
+ cap: int
25
+ refill: int
26
+ # strict G (skipper preset): even assertions require the verified tier
27
+ strict_g: bool = False
28
+ # novelty
29
+ recent_result_hashes: set = field(default_factory=set)
30
+ novelty_exempt: set = field(default_factory=set)
31
+ # relevance
32
+ claim_surface: set = field(default_factory=set)
33
+ # consequence
34
+ last_mutation_step: int = 0 # 0 = no mutation has EVER occurred
35
+ current_step: int = 0
36
+ # per-turn latches
37
+ grounded_this_turn: bool = False
38
+ verified_this_turn: bool = False
39
+ # declarative rails
40
+ verified_signals: set = field(default_factory=set)
41
+ goal_predicates: list = field(default_factory=list)
42
+ halted: bool = False
43
+
44
+ @classmethod
45
+ def for_model_class(cls, model_class="default", **kw):
46
+ p = PRESETS[model_class]
47
+ kw.setdefault("strict_g", p["strict_g"])
48
+ return cls(budget=p["CAP"], cap=p["CAP"], refill=p["REFILL"], **kw)
49
+
50
+
51
+ def normalize(text):
52
+ """Strip nondeterminism before hashing. Reference: ISO timestamps + hex ids.
53
+
54
+ Real deployments extend this per-tool; too-weak normalization means novelty
55
+ never fires on noisy tools (the no-op defense weakens).
56
+ """
57
+ text = re.sub(r"\d{4}-\d{2}-\d{2}[T ]\d{2}:\d{2}:\d{2}\S*", "<TS>", str(text))
58
+ return re.sub(r"\b[0-9a-f]{8,}\b", "<HEX>", text)
59
+
60
+
61
+ def extract_identifiers(args, result):
62
+ """Conservative token extraction for the relevance check.
63
+
64
+ Over-extraction leaks relevance; under-extraction false-rejects
65
+ cross-cutting work (open risk, flagged in docs/module-2-classifier.md).
66
+ """
67
+ return set(re.findall(r"[\w.\-/]+", f"{args} {result}"))
@@ -0,0 +1,194 @@
1
+ Metadata-Version: 2.4
2
+ Name: grounding-gate
3
+ Version: 0.1.0
4
+ Summary: Zero-token structural verifier for agent loops: one choke point at the submit boundary enforcing grounding and budget invariants. No LLM calls, stdlib only.
5
+ Project-URL: Homepage, https://github.com/CiphemonJY/grounding-gate
6
+ Project-URL: Repository, https://github.com/CiphemonJY/grounding-gate
7
+ Project-URL: Issues, https://github.com/CiphemonJY/grounding-gate/issues
8
+ Project-URL: Changelog, https://github.com/CiphemonJY/grounding-gate/blob/main/CHANGELOG.md
9
+ Author: Ciphemon
10
+ License-Expression: MIT
11
+ License-File: LICENSE
12
+ Keywords: agent-loop,agents,grounding,guardrails,hallucination,llm,verification
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Programming Language :: Python :: 3.10
18
+ Classifier: Programming Language :: Python :: 3.11
19
+ Classifier: Programming Language :: Python :: 3.12
20
+ Classifier: Programming Language :: Python :: 3.13
21
+ Classifier: Topic :: Software Development :: Quality Assurance
22
+ Requires-Python: >=3.9
23
+ Provides-Extra: dev
24
+ Requires-Dist: pytest>=7; extra == 'dev'
25
+ Description-Content-Type: text/markdown
26
+
27
+ # grounding-gate
28
+
29
+ [![ci](https://github.com/CiphemonJY/grounding-gate/actions/workflows/ci.yml/badge.svg)](https://github.com/CiphemonJY/grounding-gate/actions/workflows/ci.yml)
30
+
31
+ **Zero-token structural verifier for agent loops.** One choke point at the
32
+ submit boundary decides whether an agent is allowed to say "X is true" or
33
+ "I did X" — using hash, set, and integer operations only. No LLM calls, no
34
+ per-turn prompt injection, no dependencies.
35
+
36
+ ```
37
+ pip install grounding-gate # stdlib only, Python >= 3.9
38
+ ```
39
+
40
+ The demo ships in the repo (not the wheel):
41
+
42
+ ```
43
+ git clone https://github.com/CiphemonJY/grounding-gate && cd grounding-gate
44
+ python examples/demo.py # the whole idea in 30 seconds
45
+ ```
46
+
47
+ ## The problem
48
+
49
+ Agents fail in two characteristic ways, and both ship *confident* wrong answers:
50
+
51
+ - **Skip-and-hallucinate** — emit a terminal claim ("done, config fixed")
52
+ without ever observing reality after acting on it.
53
+ - **Reason-and-diverge** — loop in closed context, burning steps on
54
+ reasoning about stale beliefs, until a confident wrong answer ships.
55
+
56
+ The standard fix is prose: "remember to verify your work" injected into every
57
+ turn. Prose costs tokens on every turn, behaves differently per model, and —
58
+ critically — is *skippable*. A reminder is not an invariant.
59
+
60
+ ## The idea
61
+
62
+ Move enforcement out of the prompt and into **control flow**. A single gate
63
+ wraps the submit/conclude boundary, and a terminal output is emitted only if
64
+ both invariants hold:
65
+
66
+ - **G (grounding)** — a *qualifying* observation happened this turn, or the
67
+ output makes no factual claim. Qualifying means **novel** (result hash not
68
+ seen before, after stripping timestamps/ids) **∧ relevant** (touches the
69
+ identifiers the claim is about) **∧ consequence-tier-correct** (see below).
70
+ - **B (budget)** — reasoning rope remains. Qualifying observations *refill*
71
+ the budget (up to a cap); pure reasoning steps decrement it. Grounded work
72
+ runs effectively unbounded; closed-loop reasoning hits a hard floor.
73
+
74
+ Fail either → the terminal is **rejected** and the agent is told its only
75
+ legal moves: make a qualifying tool call, or exit with a typed **`unverified`**
76
+ terminal. `unverified` is a first-class, always-legal escape hatch — the gate
77
+ never traps an agent, it only forbids *confident* ungrounded claims.
78
+
79
+ ### Consequence tiers
80
+
81
+ The gate distinguishes what kind of claim an observation can support:
82
+
83
+ | Claim type | Example | Requires |
84
+ |--------------|--------------------------|----------|
85
+ | `assertion` | "X is true" | a novel, relevant, read-only observation this turn |
86
+ | `completion` | "I changed X" | a novel, relevant read taken **after** the mutation — a mutating call never self-grounds its own effect |
87
+ | `unverified` | "couldn't confirm X" | nothing — always legal |
88
+ | `none` | no factual claim | nothing — exempt |
89
+
90
+ That second row is the heart of it: *writing a file and claiming success is
91
+ not verification; reading it back afterwards is.*
92
+
93
+ ## Quickstart
94
+
95
+ ```python
96
+ from grounding_gate import GateState, classify_observation, boundary_check
97
+
98
+ state = GateState.for_model_class("default", claim_surface={"app.cfg"})
99
+
100
+ # after EVERY tool call in your agent loop:
101
+ state.current_step += 1
102
+ obs = classify_observation(tool, args, result, state, read_only=not mutating)
103
+ state.grounded_this_turn |= obs["grounds_assertion"]
104
+ state.verified_this_turn |= obs["grounds_completion"]
105
+ if mutating:
106
+ state.last_mutation_step = state.current_step # a completion now needs a read AFTER this
107
+
108
+ # at every submit/conclude attempt — this must be the ONLY path to output:
109
+ verdict = boundary_check({"claim_type": "completion", "content": answer}, state)
110
+ if verdict["verdict"] == "REJECT":
111
+ ... # surface verdict["legal_next"] to the model and continue the loop
112
+ ```
113
+
114
+ Note the mutation bookkeeping: without `last_mutation_step` ever being set, no
115
+ read can reach the verified tier and a `completion` can never be accepted —
116
+ that is the gate working as designed, not a bug.
117
+
118
+ `turn_loop` in [boundary.py](https://github.com/CiphemonJY/grounding-gate/blob/main/src/grounding_gate/boundary.py)
119
+ is the complete reference wiring (budget refill, mutation tracking, halt
120
+ semantics, signal mapping) — use it as the integration template. The
121
+ [demo](https://github.com/CiphemonJY/grounding-gate/blob/main/examples/demo.py)
122
+ runs the same scripted agent through an ungated and a gated loop, side by side.
123
+
124
+ ## Model-class presets
125
+
126
+ Fleet variance is absorbed as integers, not prose. Pick the preset matching
127
+ how your model fails:
128
+
129
+ | Preset | CAP | REFILL | Strict G | For |
130
+ |------------|-----|--------|----------|-----|
131
+ | `skipper` | 5 | 2 | yes | models that hallucinate-and-skip |
132
+ | `diverger` | 4 | 1 | no | models that reason forever |
133
+ | `default` | 6 | 2 | no | everything else |
134
+
135
+ Strict G means even plain *assertions* require verified-tier grounding (a
136
+ post-mutation observation) — an observed-tier read is not enough. In a task
137
+ that never mutates anything, a strict-G agent can only exit via the typed
138
+ `unverified` terminal; that hard line is the point of the skipper preset, so
139
+ pick `default` for read-only/Q&A workloads.
140
+
141
+ ## Declarative rails
142
+
143
+ A task can declare signals that must be verified before any completion is
144
+ accepted (`state.goal_predicates = ["tests_passed"]`). The gate never
145
+ interprets meaning — it only checks that a signal named `tests_passed` was
146
+ registered by a mapped, real command outcome. Semantic judgment stays out of
147
+ the floor by design.
148
+
149
+ ## What the gate does NOT do
150
+
151
+ Honest scope, from the design's leak audit:
152
+
153
+ - **No semantic correctness.** A grounded claim can still be wrong (the model
154
+ can misread a real result). That is punted to a declared verifier tier
155
+ (`verify_with`), not smuggled into the floor.
156
+ - **Relevance can be spoofed** by a model that deliberately mentions the right
157
+ identifiers in an irrelevant call. The floor defends against *lazy*
158
+ ungroundedness, which is the overwhelmingly common failure; adversarial
159
+ self-deception needs the verifier tier.
160
+ - **Nondeterministic tools** need per-tool `normalize()` extensions or the
161
+ audited `novelty_exempt` allowlist — too-weak normalization quietly disables
162
+ the no-op defense.
163
+
164
+ ## How this was built
165
+
166
+ The modules were drafted by different LLMs and adversarially reviewed before
167
+ assembly; the final behavior is pinned by a 19-case acceptance suite
168
+ ([tests/test_gate.py](https://github.com/CiphemonJY/grounding-gate/blob/main/tests/test_gate.py))
169
+ that runs on bare Python with zero dependencies. Two review findings shaped
170
+ the method and are preserved in the docstrings:
171
+
172
+ - A drafting model shipped a consequence-tier bug **and authored the test that
173
+ ratified it** — since then, expected outcomes are authored by the reviewer,
174
+ never by the generator
175
+ ([docs/module-2-classifier.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/module-2-classifier.md)).
176
+ - The remaining leaks lived *between* individually-passing test cases —
177
+ latch-vs-assignment, halt cleared by non-qualifying calls
178
+ ([docs/module-4-boundary.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/module-4-boundary.md)).
179
+
180
+ Full design spec:
181
+ [docs/spec.md](https://github.com/CiphemonJY/grounding-gate/blob/main/docs/spec.md).
182
+
183
+ ## Status & roadmap
184
+
185
+ This is the reference implementation — correct, minimal, and framework-free.
186
+ Planned next:
187
+
188
+ - Adapters: Claude Agent SDK hook, LangGraph middleware, OpenAI Agents SDK.
189
+ - A real signal-mapper module (command exit code → declared signal).
190
+ - Empirical preset tuning across model classes.
191
+
192
+ ## License
193
+
194
+ MIT
@@ -0,0 +1,8 @@
1
+ grounding_gate/__init__.py,sha256=Zm7VbfHRzhVfmg8Vtt520Pqm3CvAy7Uz5ua_9fGQn_Y,910
2
+ grounding_gate/boundary.py,sha256=JU4ge0mNaqorau4iEFmgy2UkmMFArKoon3_qR8j1Z3Q,4717
3
+ grounding_gate/classifier.py,sha256=KgldOBV9fkzsCyBTdYcSDwX__37T28sH4GBa645Le04,1968
4
+ grounding_gate/state.py,sha256=2xXDSPPBtw5O012oGmVe2lYz0nwy2WCOsyElBcKzF0E,2425
5
+ grounding_gate-0.1.0.dist-info/METADATA,sha256=FmruvXaYWLsLGPSlqAzOERHgEWWEbAdNH7Ygg0cZRtE,8946
6
+ grounding_gate-0.1.0.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
7
+ grounding_gate-0.1.0.dist-info/licenses/LICENSE,sha256=76kJpx8qwE8s8qdKjsnngPtznV3WasZ54ed4WWW2y7o,1065
8
+ grounding_gate-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,4 @@
1
+ Wheel-Version: 1.0
2
+ Generator: hatchling 1.31.0
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Ciphemon
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.