chad-code 1.12.0__tar.gz → 1.13.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {chad_code-1.12.0 → chad_code-1.13.0}/PKG-INFO +1 -1
- {chad_code-1.12.0 → chad_code-1.13.0}/pyproject.toml +1 -1
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/__init__.py +1 -1
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/agent.py +45 -5
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/guardrails.py +61 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/levers.py +8 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/PKG-INFO +1 -1
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent_guards.py +31 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lever_bite.py +10 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/LICENSE +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/README.md +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/setup.cfg +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/ambient.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/atif.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/base_engine.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/bench.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/checkpoint.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/cli.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/compaction.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/completion_engine.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/config.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/diag.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/engine.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/ignore.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lsp.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lspclient.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lspservers.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mcp.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mcp_oauth.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_fastpath.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_moe_fused.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_qsdpa.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/LICENSE +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/__init__.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/alignment.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/attention.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/audio.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/cache.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/conformer.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/ctc.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/parakeet.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/rnnt.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/tokenizer.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/utils.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/profiles.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/prompt.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/prove.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/render.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/repomap.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/seatbelt.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/serve.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/session.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/skills.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/speech.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/symbols.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/syntaxgate.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/toolcall_parse.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/tools.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/tui.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/validate.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/SOURCES.txt +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/dependency_links.txt +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/entry_points.txt +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/requires.txt +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/top_level.txt +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent_e2e.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_ambient.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_atif.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_bench.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_checkpoint.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_cli.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_compact_notice.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_compaction.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_completion_engine.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_config.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_confirm_preview.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_done_audit.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_drift_warn.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_edit.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_edit_corruption.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine_kvquant.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine_pld_hybrid.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_feel_pack.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_garble_invariant.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_gate.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_ignore.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_intent.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lever_instrumentation.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_levers.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_log_redaction.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lsp.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lsp_live.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lspclient.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mcp.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mcp_oauth.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_fastpath.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_moe_fused.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_qsdpa.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_plan_review.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_prove.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_render.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_replace_lines.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_repomap.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_repomap_polyglot.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_seatbelt.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_serve.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_session.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_skills.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_speech.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_speech_tui.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_subagent.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_symbols.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_syntaxgate.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_toolcall_parse.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_tools.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_tui.py +0 -0
- {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_validate.py +0 -0
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
# import name, and command name are independent. `uvx chad-code` runs the alias
|
|
5
5
|
# script added under [project.scripts].
|
|
6
6
|
name = "chad-code"
|
|
7
|
-
version = "1.
|
|
7
|
+
version = "1.13.0"
|
|
8
8
|
description = "Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)"
|
|
9
9
|
readme = "README.md"
|
|
10
10
|
license = "MIT"
|
|
@@ -4,7 +4,7 @@ A flat collection of cooperating modules behind one console script (``chad``):
|
|
|
4
4
|
the inference engine, the tool layer, the agent loop, and the terminal UI.
|
|
5
5
|
"""
|
|
6
6
|
|
|
7
|
-
__version__ = "1.
|
|
7
|
+
__version__ = "1.13.0"
|
|
8
8
|
|
|
9
9
|
# chad sets no MLX_* runtime vars. MLX_METAL_FAST_SYNCH, MLX_MAX_OPS_PER_BUFFER
|
|
10
10
|
# and MLX_MAX_MB_PER_BUFFER were each measured end-to-end on the 35B and every
|
|
@@ -337,7 +337,7 @@ class Agent:
|
|
|
337
337
|
ctx_limit: int = 24000, mode: str = None, emit=None,
|
|
338
338
|
confirm=None, should_stop=None, drain_steering=None,
|
|
339
339
|
thinking: bool = True,
|
|
340
|
-
max_gen_tokens: int =
|
|
340
|
+
max_gen_tokens: int = None, resume: list = None, persist: bool = False,
|
|
341
341
|
think_budget: int = None, think_ceiling: int = None,
|
|
342
342
|
turn_budget_tokens: int = None,
|
|
343
343
|
turn_budget_s: float = None, subagent: bool = False,
|
|
@@ -377,8 +377,21 @@ class Agent:
|
|
|
377
377
|
# Per-step generation cap. The old 2048 default truncated legitimate work —
|
|
378
378
|
# a reasoning turn that thinks, then emits a `write` of a whole test file can
|
|
379
379
|
# exceed 2048 tokens, and the cut-off was being misread as a final answer
|
|
380
|
-
# (the "answers on paper then stops" bug). 8192
|
|
381
|
-
#
|
|
380
|
+
# (the "answers on paper then stops" bug). 8192 fixed that but had its own
|
|
381
|
+
# loss class: on hard problems a long chain of thought can pin the cap while
|
|
382
|
+
# still inside <think>, so the step ends as discarded reasoning with no
|
|
383
|
+
# action and the next step re-derives from scratch — trace measurement shows
|
|
384
|
+
# these mid-think pins cluster on exactly the tasks that fail, and that an
|
|
385
|
+
# uncapped run of the same model almost never exceeds 8192 on its own
|
|
386
|
+
# (p99 ~4.7k) yet the rare 15-20k thought that does run long completes real
|
|
387
|
+
# work when allowed to finish. So the cap buys little in the common case and
|
|
388
|
+
# converts the long-thought tail into zero-progress dead steps. 32768 keeps
|
|
389
|
+
# a backstop against non-repetitive runaway garble (literal decode loops are
|
|
390
|
+
# the repeat guard's job, default ON) while fitting the longest observed
|
|
391
|
+
# legitimate thought with room for the action. Env knob for A/B without a
|
|
392
|
+
# code change (CHAD_* family).
|
|
393
|
+
if max_gen_tokens is None:
|
|
394
|
+
max_gen_tokens = config.env_int("CHAD_MAX_GEN_TOKENS", 32768)
|
|
382
395
|
self.max_gen_tokens = max_gen_tokens
|
|
383
396
|
# Soft think-cap base. None => the mechanism is OFF and generation is
|
|
384
397
|
# byte-identical to before. Falls back to the CHAD_THINK_BUDGET env knob so the
|
|
@@ -849,6 +862,10 @@ class Agent:
|
|
|
849
862
|
break_nudges = 0 # times we escalated a stuck edit this turn
|
|
850
863
|
readonly_streak = 0 # consecutive steps with substantive tools but no landed edit
|
|
851
864
|
gate_nudges = 0 # times the investigation->edit gate fired this turn
|
|
865
|
+
explore_streak = 0 # consecutive bash steps since the last landed edit (counts
|
|
866
|
+
# AFTER an edit too — investigation_gate can't; see
|
|
867
|
+
# guardrails.verification_matrix)
|
|
868
|
+
explore_gate_fires = 0 # times the verification-matrix gate fired this turn
|
|
852
869
|
subagent_sigs = set() # (description, prompt) of sub-agents already spawned this turn
|
|
853
870
|
truncation_nudges = 0 # times we pushed past a token-cap truncation this turn
|
|
854
871
|
garble_nudges = 0 # malformed-tool-call re-nudges (own counter, so a step-0
|
|
@@ -1105,7 +1122,7 @@ class Agent:
|
|
|
1105
1122
|
return n >= _cap and "</think>" not in text_so_far
|
|
1106
1123
|
|
|
1107
1124
|
# Degenerate-repetition stop (default ON): greedy decode can lock into
|
|
1108
|
-
# repeating one short string until the
|
|
1125
|
+
# repeating one short string until the max_gen_tokens cap — minutes of dead
|
|
1109
1126
|
# generation per occurrence at 9B decode speed. Checked every 16 tokens on
|
|
1110
1127
|
# the generation's tail; a hit stops the step (rep_fired tells this stopper
|
|
1111
1128
|
# apart from the think-cap, which shares stats.stop_condition_fired) and the
|
|
@@ -1368,7 +1385,7 @@ class Agent:
|
|
|
1368
1385
|
# inside <think> — a distinct sub-bucket from a THINK-CAP/ceiling stop
|
|
1369
1386
|
# (those are agent-side stop_conditions; this is max_gen_tokens itself),
|
|
1370
1387
|
# so future autopsies can size it directly instead of inferring from
|
|
1371
|
-
# max_gen
|
|
1388
|
+
# the max_gen value. A salvaged step already injected the closing tag, so
|
|
1372
1389
|
# this is naturally False for it.
|
|
1373
1390
|
"reasoning_length_stop": hit_cap and "</think>" not in text,
|
|
1374
1391
|
"deadline_abort": deadline_fired[0], # 103: wall-clock hard wrap-up cut
|
|
@@ -2170,6 +2187,29 @@ class Agent:
|
|
|
2170
2187
|
step, gate_nudges)
|
|
2171
2188
|
self.messages.append({"role": "tool", "name": "edit", "content": gate})
|
|
2172
2189
|
|
|
2190
|
+
# Convergence gate (verification-matrix design): count exploratory-
|
|
2191
|
+
# bash steps since the last landed edit — unlike readonly_streak/
|
|
2192
|
+
# investigation_gate this keeps counting AFTER an edit lands, which is where
|
|
2193
|
+
# the demonstrated thrash lives (write early, then probe 60 more times
|
|
2194
|
+
# without calling done). A NEW edit this step resets it; a bash-only step
|
|
2195
|
+
# advances it. When it bites, the model gets the task's real requirements as
|
|
2196
|
+
# a matrix to close by evidence-or-unverified, not an open-ended "keep going".
|
|
2197
|
+
if made_edit and not _gov_prev_made:
|
|
2198
|
+
explore_streak = 0
|
|
2199
|
+
elif any(n == "bash" for n, _ in calls):
|
|
2200
|
+
explore_streak += 1
|
|
2201
|
+
if not read_only_intent:
|
|
2202
|
+
matrix = guardrails.verification_matrix(
|
|
2203
|
+
guardrails.audit_task_text(user_text),
|
|
2204
|
+
explore_streak, made_edit, explore_gate_fires)
|
|
2205
|
+
if matrix:
|
|
2206
|
+
explore_gate_fires += 1
|
|
2207
|
+
explore_streak = 0
|
|
2208
|
+
log.info("VERIFICATION-MATRIX gate at step %d (bash-streak, no "
|
|
2209
|
+
"change) -> nudge #%d", step, explore_gate_fires)
|
|
2210
|
+
self.messages.append(
|
|
2211
|
+
{"role": "tool", "name": "bash", "content": matrix})
|
|
2212
|
+
|
|
2173
2213
|
# Break a flailing-probe run (e.g. guessing the test runner, repeated
|
|
2174
2214
|
# `python -c import` checks) that the exact-call loop guard can't see because
|
|
2175
2215
|
# each failing command differs by a few characters.
|
|
@@ -782,6 +782,67 @@ def investigation_gate(readonly_streak, made_edit, gate_nudges, threshold=6):
|
|
|
782
782
|
"verify it. If you already know the fix, apply it this step.]")
|
|
783
783
|
|
|
784
784
|
|
|
785
|
+
def verification_matrix(task_text, explore_streak, made_edit, gate_fires,
|
|
786
|
+
threshold=8, cap=6):
|
|
787
|
+
"""Pull a thrashing turn into a bounded verification phase.
|
|
788
|
+
|
|
789
|
+
The demonstrated failure (measured on the bare full-89 run against a reference
|
|
790
|
+
harness on the same model/box): chad's #1 loss class is not wrong answers, it is
|
|
791
|
+
non-convergence. 69% of failing trials never called `done` — they ran a long tail
|
|
792
|
+
of successful, exploratory bash (median 39 commands vs 12 on passes) with the SAME
|
|
793
|
+
two edits a passing trial makes, and the step cap killed them mid-poke. On
|
|
794
|
+
break-filter the failing trial ran 72 bash probes / 2 writes across 75 steps while
|
|
795
|
+
chad's OWN passing trial found the identical exploit in 13; the model knows the
|
|
796
|
+
answer and researches past the budget. A reference harness on the same weights
|
|
797
|
+
sits at chad's PASSING step count.
|
|
798
|
+
|
|
799
|
+
The design ported here reframes completion as a VERIFICATION MATRIX: every
|
|
800
|
+
requirement the task states must be closed by ONE of (a) causal evidence from a
|
|
801
|
+
real run through the public path, or (b) an explicit "unverified" note. That
|
|
802
|
+
single rule both BOUNDS the loop (a finite checklist has a terminal state, so the
|
|
803
|
+
turn knows when to stop) and blocks false-done (a row is not filled by a
|
|
804
|
+
self-authored PASS label, a snapshot, "no error", or an oracle built from the same
|
|
805
|
+
assumption as the code). The honest-unverified escape is load-bearing: it lets a
|
|
806
|
+
turn finish a genuinely-unverifiable requirement instead of thrashing on it
|
|
807
|
+
forever.
|
|
808
|
+
|
|
809
|
+
Reuses chad's own requirement extractor (`audit_requirement_lines`, the same engine
|
|
810
|
+
`done_audit` runs) so the matrix rows are the task's real predicates, not a
|
|
811
|
+
paraphrase. Fires from the THRASH entry point — the exploratory-bash streak — which
|
|
812
|
+
`done_audit` cannot reach (a ran-out turn never calls `done`) and `investigation_
|
|
813
|
+
gate` cannot reach (it freezes the moment any edit lands; this thrash is all
|
|
814
|
+
post-edit). Re-arms up to `cap`, never bare-"call done" (so it will not inflate the
|
|
815
|
+
done-but-wrong bucket that `done_audit` still guards at the actual `done`). Returns
|
|
816
|
+
nudge text or None; caller resets the streak after a firing."""
|
|
817
|
+
if not levers.enabled("verification_matrix"):
|
|
818
|
+
return None
|
|
819
|
+
if explore_streak < threshold or gate_fires >= cap:
|
|
820
|
+
return None
|
|
821
|
+
req_lines = audit_requirement_lines(task_text, audit_extract_paths(task_text))
|
|
822
|
+
levers.fired("verification_matrix", streak=explore_streak,
|
|
823
|
+
reqs=len(req_lines), made_edit=made_edit, fires=gate_fires)
|
|
824
|
+
parts = [
|
|
825
|
+
f"[you've run ~{explore_streak} commands since your last change without "
|
|
826
|
+
"finishing. Stop exploring and close this out as a verification matrix. The "
|
|
827
|
+
"hidden grader checks the task's OWN requirements — for EACH one below, you "
|
|
828
|
+
"need exactly ONE of two things:",
|
|
829
|
+
" (a) EVIDENCE: a real run through the actual public path that shows the "
|
|
830
|
+
"required outcome. A snapshot, your own \"PASS\"/\"OK\" text, \"no error\", or "
|
|
831
|
+
"a check you built from the same assumption as the code do NOT count.",
|
|
832
|
+
" (b) UNVERIFIED: a plain note that you could not verify it. An honest "
|
|
833
|
+
"\"unverified\" is allowed and lets you finish; a false \"it works\" is not.",
|
|
834
|
+
]
|
|
835
|
+
if req_lines:
|
|
836
|
+
parts.append("Requirements from the task statement:")
|
|
837
|
+
parts += [f" > {ln}" for ln in req_lines]
|
|
838
|
+
else:
|
|
839
|
+
parts.append("Re-read the task statement and list exactly what it requires, "
|
|
840
|
+
"then close each item as (a) or (b).")
|
|
841
|
+
parts.append("When every requirement is (a) or (b), apply any fix still needed, "
|
|
842
|
+
"then call `done`. Do not run more exploratory probes.]")
|
|
843
|
+
return "\n".join(parts)
|
|
844
|
+
|
|
845
|
+
|
|
785
846
|
def edit_failed_to_land(result: str) -> bool:
|
|
786
847
|
"""True when an edit/symbolic-edit tool result means the change did NOT apply — a
|
|
787
848
|
no-op (old==new / replacement leaves file unchanged), an unmatched `old` string, or
|
|
@@ -91,6 +91,14 @@ LEVERS: dict[str, Lever] = {
|
|
|
91
91
|
"After ~6 read-only steps with no landed edit, steer the model to act before the "
|
|
92
92
|
"step cap kills the turn with an empty patch.",
|
|
93
93
|
"iter2"),
|
|
94
|
+
"verification_matrix": Lever(
|
|
95
|
+
"After ~8 exploratory-bash steps since the last change (even AFTER an edit has "
|
|
96
|
+
"landed — the case investigation_gate can't see), pull the turn into a bounded "
|
|
97
|
+
"verification phase: for each task requirement, close it with a real causal run "
|
|
98
|
+
"OR an explicit 'unverified', then `done`. Reuses done_audit's requirement "
|
|
99
|
+
"extractor; targets the 69%-of-fails non-convergence (ran-out) bucket while the "
|
|
100
|
+
"honest-unverified escape keeps it from thrashing on the genuinely-unverifiable.",
|
|
101
|
+
"iter16"),
|
|
94
102
|
"edit_loop_break": Lever(
|
|
95
103
|
"After 2 consecutive edits that failed to land, stop the model re-trying "
|
|
96
104
|
"variations and tell it to read the real lines / replace the whole symbol.",
|
|
@@ -885,6 +885,36 @@ def test_investigation_gate():
|
|
|
885
885
|
check("gate bounded by gate_nudges", investigation_gate(20, made_edit=False, gate_nudges=2) is None)
|
|
886
886
|
|
|
887
887
|
|
|
888
|
+
def test_verification_matrix():
|
|
889
|
+
from chad.guardrails import verification_matrix
|
|
890
|
+
task = (
|
|
891
|
+
"Write /app/out.html so it still triggers alert() after /app/filter.py runs.\n"
|
|
892
|
+
"The output must be valid HTML and must not require user interaction.\n")
|
|
893
|
+
# Below threshold: silent.
|
|
894
|
+
check("no matrix below threshold",
|
|
895
|
+
verification_matrix(task, 4, made_edit=False, gate_fires=0) is None)
|
|
896
|
+
# At/over threshold: fires with the evidence-or-unverified matrix.
|
|
897
|
+
g = verification_matrix(task, 8, made_edit=False, gate_fires=0)
|
|
898
|
+
check("matrix fires at threshold", g is not None)
|
|
899
|
+
check("matrix offers the evidence path", g is not None and "EVIDENCE" in g)
|
|
900
|
+
check("matrix offers the honest-unverified escape",
|
|
901
|
+
g is not None and "UNVERIFIED" in g)
|
|
902
|
+
check("matrix ends at done", g is not None and "`done`" in g)
|
|
903
|
+
# Reuses the requirement extractor: the task's own predicate lines appear as rows.
|
|
904
|
+
check("matrix quotes a real requirement line",
|
|
905
|
+
g is not None and "out.html" in g)
|
|
906
|
+
# The whole point vs investigation_gate: it STILL fires after an edit landed
|
|
907
|
+
# (the break-filter thrash shape — write early, probe forever).
|
|
908
|
+
check("matrix fires even after an edit landed",
|
|
909
|
+
verification_matrix(task, 12, made_edit=True, gate_fires=0) is not None)
|
|
910
|
+
# A task with no extractable requirement lines still fires (generic close-out).
|
|
911
|
+
check("matrix fires with no extractable requirements",
|
|
912
|
+
verification_matrix("do the thing", 8, made_edit=False, gate_fires=0) is not None)
|
|
913
|
+
# Re-armable but bounded by the firing cap.
|
|
914
|
+
check("matrix bounded by cap",
|
|
915
|
+
verification_matrix(task, 30, made_edit=True, gate_fires=6) is None)
|
|
916
|
+
|
|
917
|
+
|
|
888
918
|
def test_edit_failed_to_land():
|
|
889
919
|
check("no-op edit failed to land", edit_failed_to_land("[no-op edit: old and new are identical]"))
|
|
890
920
|
check("not-found failed to land", edit_failed_to_land("[old string not found; no change made.]"))
|
|
@@ -1200,6 +1230,7 @@ if __name__ == "__main__":
|
|
|
1200
1230
|
test_bash_result_verifies_ignores_trivial_checks()
|
|
1201
1231
|
test_bash_result_verifies_requires_executing_command()
|
|
1202
1232
|
test_investigation_gate()
|
|
1233
|
+
test_verification_matrix()
|
|
1203
1234
|
test_edit_failed_to_land()
|
|
1204
1235
|
test_edit_loop_break()
|
|
1205
1236
|
test_done_rejection()
|
|
@@ -169,6 +169,16 @@ def test_investigation_gate(monkeypatch):
|
|
|
169
169
|
assert guardrails.investigation_gate(8, made_edit=False, gate_nudges=0) is None
|
|
170
170
|
|
|
171
171
|
|
|
172
|
+
def test_verification_matrix(monkeypatch):
|
|
173
|
+
n = bite("verification_matrix")
|
|
174
|
+
on(monkeypatch)
|
|
175
|
+
assert guardrails.verification_matrix("write /app/out.txt", 8,
|
|
176
|
+
made_edit=True, gate_fires=0)
|
|
177
|
+
off(monkeypatch, n)
|
|
178
|
+
assert guardrails.verification_matrix("write /app/out.txt", 8,
|
|
179
|
+
made_edit=True, gate_fires=0) is None
|
|
180
|
+
|
|
181
|
+
|
|
172
182
|
def test_edit_loop_break(monkeypatch):
|
|
173
183
|
n = bite("edit_loop_break")
|
|
174
184
|
on(monkeypatch)
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|