chad-code 1.12.0__tar.gz → 1.13.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. {chad_code-1.12.0 → chad_code-1.13.0}/PKG-INFO +1 -1
  2. {chad_code-1.12.0 → chad_code-1.13.0}/pyproject.toml +1 -1
  3. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/__init__.py +1 -1
  4. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/agent.py +45 -5
  5. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/guardrails.py +61 -0
  6. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/levers.py +8 -0
  7. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/PKG-INFO +1 -1
  8. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent_guards.py +31 -0
  9. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lever_bite.py +10 -0
  10. {chad_code-1.12.0 → chad_code-1.13.0}/LICENSE +0 -0
  11. {chad_code-1.12.0 → chad_code-1.13.0}/README.md +0 -0
  12. {chad_code-1.12.0 → chad_code-1.13.0}/setup.cfg +0 -0
  13. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/ambient.py +0 -0
  14. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/atif.py +0 -0
  15. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/base_engine.py +0 -0
  16. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/bench.py +0 -0
  17. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/checkpoint.py +0 -0
  18. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/cli.py +0 -0
  19. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/compaction.py +0 -0
  20. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/completion_engine.py +0 -0
  21. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/config.py +0 -0
  22. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/diag.py +0 -0
  23. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/engine.py +0 -0
  24. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/ignore.py +0 -0
  25. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lsp.py +0 -0
  26. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lspclient.py +0 -0
  27. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/lspservers.py +0 -0
  28. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mcp.py +0 -0
  29. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mcp_oauth.py +0 -0
  30. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_fastpath.py +0 -0
  31. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_moe_fused.py +0 -0
  32. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/mlx_qsdpa.py +0 -0
  33. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/LICENSE +0 -0
  34. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/__init__.py +0 -0
  35. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/alignment.py +0 -0
  36. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/attention.py +0 -0
  37. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/audio.py +0 -0
  38. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/cache.py +0 -0
  39. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/conformer.py +0 -0
  40. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/ctc.py +0 -0
  41. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/parakeet.py +0 -0
  42. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/rnnt.py +0 -0
  43. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/tokenizer.py +0 -0
  44. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/parakeet/utils.py +0 -0
  45. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/profiles.py +0 -0
  46. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/prompt.py +0 -0
  47. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/prove.py +0 -0
  48. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/render.py +0 -0
  49. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/repomap.py +0 -0
  50. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/seatbelt.py +0 -0
  51. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/serve.py +0 -0
  52. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/session.py +0 -0
  53. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/skills.py +0 -0
  54. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/speech.py +0 -0
  55. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/symbols.py +0 -0
  56. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/syntaxgate.py +0 -0
  57. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/toolcall_parse.py +0 -0
  58. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/tools.py +0 -0
  59. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/tui.py +0 -0
  60. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad/validate.py +0 -0
  61. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/SOURCES.txt +0 -0
  62. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/dependency_links.txt +0 -0
  63. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/entry_points.txt +0 -0
  64. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/requires.txt +0 -0
  65. {chad_code-1.12.0 → chad_code-1.13.0}/src/chad_code.egg-info/top_level.txt +0 -0
  66. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent.py +0 -0
  67. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_agent_e2e.py +0 -0
  68. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_ambient.py +0 -0
  69. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_atif.py +0 -0
  70. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_bench.py +0 -0
  71. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_checkpoint.py +0 -0
  72. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_cli.py +0 -0
  73. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_compact_notice.py +0 -0
  74. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_compaction.py +0 -0
  75. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_completion_engine.py +0 -0
  76. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_config.py +0 -0
  77. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_confirm_preview.py +0 -0
  78. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_done_audit.py +0 -0
  79. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_drift_warn.py +0 -0
  80. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_edit.py +0 -0
  81. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_edit_corruption.py +0 -0
  82. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine.py +0 -0
  83. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine_kvquant.py +0 -0
  84. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_engine_pld_hybrid.py +0 -0
  85. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_feel_pack.py +0 -0
  86. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_garble_invariant.py +0 -0
  87. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_gate.py +0 -0
  88. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_ignore.py +0 -0
  89. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_intent.py +0 -0
  90. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lever_instrumentation.py +0 -0
  91. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_levers.py +0 -0
  92. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_log_redaction.py +0 -0
  93. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lsp.py +0 -0
  94. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lsp_live.py +0 -0
  95. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_lspclient.py +0 -0
  96. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mcp.py +0 -0
  97. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mcp_oauth.py +0 -0
  98. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_fastpath.py +0 -0
  99. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_moe_fused.py +0 -0
  100. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_mlx_qsdpa.py +0 -0
  101. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_plan_review.py +0 -0
  102. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_prove.py +0 -0
  103. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_render.py +0 -0
  104. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_replace_lines.py +0 -0
  105. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_repomap.py +0 -0
  106. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_repomap_polyglot.py +0 -0
  107. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_seatbelt.py +0 -0
  108. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_serve.py +0 -0
  109. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_session.py +0 -0
  110. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_skills.py +0 -0
  111. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_speech.py +0 -0
  112. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_speech_tui.py +0 -0
  113. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_subagent.py +0 -0
  114. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_symbols.py +0 -0
  115. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_syntaxgate.py +0 -0
  116. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_toolcall_parse.py +0 -0
  117. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_tools.py +0 -0
  118. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_tui.py +0 -0
  119. {chad_code-1.12.0 → chad_code-1.13.0}/tests/test_validate.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: chad-code
3
- Version: 1.12.0
3
+ Version: 1.13.0
4
4
  Summary: Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)
5
5
  License-Expression: MIT
6
6
  Project-URL: Repository, https://github.com/nathansutton/chad
@@ -4,7 +4,7 @@
4
4
  # import name, and command name are independent. `uvx chad-code` runs the alias
5
5
  # script added under [project.scripts].
6
6
  name = "chad-code"
7
- version = "1.12.0"
7
+ version = "1.13.0"
8
8
  description = "Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -4,7 +4,7 @@ A flat collection of cooperating modules behind one console script (``chad``):
4
4
  the inference engine, the tool layer, the agent loop, and the terminal UI.
5
5
  """
6
6
 
7
- __version__ = "1.12.0"
7
+ __version__ = "1.13.0"
8
8
 
9
9
  # chad sets no MLX_* runtime vars. MLX_METAL_FAST_SYNCH, MLX_MAX_OPS_PER_BUFFER
10
10
  # and MLX_MAX_MB_PER_BUFFER were each measured end-to-end on the 35B and every
@@ -337,7 +337,7 @@ class Agent:
337
337
  ctx_limit: int = 24000, mode: str = None, emit=None,
338
338
  confirm=None, should_stop=None, drain_steering=None,
339
339
  thinking: bool = True,
340
- max_gen_tokens: int = 8192, resume: list = None, persist: bool = False,
340
+ max_gen_tokens: int = None, resume: list = None, persist: bool = False,
341
341
  think_budget: int = None, think_ceiling: int = None,
342
342
  turn_budget_tokens: int = None,
343
343
  turn_budget_s: float = None, subagent: bool = False,
@@ -377,8 +377,21 @@ class Agent:
377
377
  # Per-step generation cap. The old 2048 default truncated legitimate work —
378
378
  # a reasoning turn that thinks, then emits a `write` of a whole test file can
379
379
  # exceed 2048 tokens, and the cut-off was being misread as a final answer
380
- # (the "answers on paper then stops" bug). 8192 leaves room for think + a
381
- # full-file write; truncation past it is now detected and nudged, not accepted.
380
+ # (the "answers on paper then stops" bug). 8192 fixed that but had its own
381
+ # loss class: on hard problems a long chain of thought can pin the cap while
382
+ # still inside <think>, so the step ends as discarded reasoning with no
383
+ # action and the next step re-derives from scratch — trace measurement shows
384
+ # these mid-think pins cluster on exactly the tasks that fail, and that an
385
+ # uncapped run of the same model almost never exceeds 8192 on its own
386
+ # (p99 ~4.7k) yet the rare 15-20k thought that does run long completes real
387
+ # work when allowed to finish. So the cap buys little in the common case and
388
+ # converts the long-thought tail into zero-progress dead steps. 32768 keeps
389
+ # a backstop against non-repetitive runaway garble (literal decode loops are
390
+ # the repeat guard's job, default ON) while fitting the longest observed
391
+ # legitimate thought with room for the action. Env knob for A/B without a
392
+ # code change (CHAD_* family).
393
+ if max_gen_tokens is None:
394
+ max_gen_tokens = config.env_int("CHAD_MAX_GEN_TOKENS", 32768)
382
395
  self.max_gen_tokens = max_gen_tokens
383
396
  # Soft think-cap base. None => the mechanism is OFF and generation is
384
397
  # byte-identical to before. Falls back to the CHAD_THINK_BUDGET env knob so the
@@ -849,6 +862,10 @@ class Agent:
849
862
  break_nudges = 0 # times we escalated a stuck edit this turn
850
863
  readonly_streak = 0 # consecutive steps with substantive tools but no landed edit
851
864
  gate_nudges = 0 # times the investigation->edit gate fired this turn
865
+ explore_streak = 0 # consecutive bash steps since the last landed edit (counts
866
+ # AFTER an edit too — investigation_gate can't; see
867
+ # guardrails.verification_matrix)
868
+ explore_gate_fires = 0 # times the verification-matrix gate fired this turn
852
869
  subagent_sigs = set() # (description, prompt) of sub-agents already spawned this turn
853
870
  truncation_nudges = 0 # times we pushed past a token-cap truncation this turn
854
871
  garble_nudges = 0 # malformed-tool-call re-nudges (own counter, so a step-0
@@ -1105,7 +1122,7 @@ class Agent:
1105
1122
  return n >= _cap and "</think>" not in text_so_far
1106
1123
 
1107
1124
  # Degenerate-repetition stop (default ON): greedy decode can lock into
1108
- # repeating one short string until the 8192-token cap — ~4 minutes of dead
1125
+ # repeating one short string until the max_gen_tokens cap — minutes of dead
1109
1126
  # generation per occurrence at 9B decode speed. Checked every 16 tokens on
1110
1127
  # the generation's tail; a hit stops the step (rep_fired tells this stopper
1111
1128
  # apart from the think-cap, which shares stats.stop_condition_fired) and the
@@ -1368,7 +1385,7 @@ class Agent:
1368
1385
  # inside <think> — a distinct sub-bucket from a THINK-CAP/ceiling stop
1369
1386
  # (those are agent-side stop_conditions; this is max_gen_tokens itself),
1370
1387
  # so future autopsies can size it directly instead of inferring from
1371
- # max_gen==8192. A salvaged step already injected the closing tag, so
1388
+ # the max_gen value. A salvaged step already injected the closing tag, so
1372
1389
  # this is naturally False for it.
1373
1390
  "reasoning_length_stop": hit_cap and "</think>" not in text,
1374
1391
  "deadline_abort": deadline_fired[0], # 103: wall-clock hard wrap-up cut
@@ -2170,6 +2187,29 @@ class Agent:
2170
2187
  step, gate_nudges)
2171
2188
  self.messages.append({"role": "tool", "name": "edit", "content": gate})
2172
2189
 
2190
+ # Convergence gate (verification-matrix design): count exploratory-
2191
+ # bash steps since the last landed edit — unlike readonly_streak/
2192
+ # investigation_gate this keeps counting AFTER an edit lands, which is where
2193
+ # the demonstrated thrash lives (write early, then probe 60 more times
2194
+ # without calling done). A NEW edit this step resets it; a bash-only step
2195
+ # advances it. When it bites, the model gets the task's real requirements as
2196
+ # a matrix to close by evidence-or-unverified, not an open-ended "keep going".
2197
+ if made_edit and not _gov_prev_made:
2198
+ explore_streak = 0
2199
+ elif any(n == "bash" for n, _ in calls):
2200
+ explore_streak += 1
2201
+ if not read_only_intent:
2202
+ matrix = guardrails.verification_matrix(
2203
+ guardrails.audit_task_text(user_text),
2204
+ explore_streak, made_edit, explore_gate_fires)
2205
+ if matrix:
2206
+ explore_gate_fires += 1
2207
+ explore_streak = 0
2208
+ log.info("VERIFICATION-MATRIX gate at step %d (bash-streak, no "
2209
+ "change) -> nudge #%d", step, explore_gate_fires)
2210
+ self.messages.append(
2211
+ {"role": "tool", "name": "bash", "content": matrix})
2212
+
2173
2213
  # Break a flailing-probe run (e.g. guessing the test runner, repeated
2174
2214
  # `python -c import` checks) that the exact-call loop guard can't see because
2175
2215
  # each failing command differs by a few characters.
@@ -782,6 +782,67 @@ def investigation_gate(readonly_streak, made_edit, gate_nudges, threshold=6):
782
782
  "verify it. If you already know the fix, apply it this step.]")
783
783
 
784
784
 
785
+ def verification_matrix(task_text, explore_streak, made_edit, gate_fires,
786
+ threshold=8, cap=6):
787
+ """Pull a thrashing turn into a bounded verification phase.
788
+
789
+ The demonstrated failure (measured on the bare full-89 run against a reference
790
+ harness on the same model/box): chad's #1 loss class is not wrong answers, it is
791
+ non-convergence. 69% of failing trials never called `done` — they ran a long tail
792
+ of successful, exploratory bash (median 39 commands vs 12 on passes) with the SAME
793
+ two edits a passing trial makes, and the step cap killed them mid-poke. On
794
+ break-filter the failing trial ran 72 bash probes / 2 writes across 75 steps while
795
+ chad's OWN passing trial found the identical exploit in 13; the model knows the
796
+ answer and researches past the budget. A reference harness on the same weights
797
+ sits at chad's PASSING step count.
798
+
799
+ The design ported here reframes completion as a VERIFICATION MATRIX: every
800
+ requirement the task states must be closed by ONE of (a) causal evidence from a
801
+ real run through the public path, or (b) an explicit "unverified" note. That
802
+ single rule both BOUNDS the loop (a finite checklist has a terminal state, so the
803
+ turn knows when to stop) and blocks false-done (a row is not filled by a
804
+ self-authored PASS label, a snapshot, "no error", or an oracle built from the same
805
+ assumption as the code). The honest-unverified escape is load-bearing: it lets a
806
+ turn finish a genuinely-unverifiable requirement instead of thrashing on it
807
+ forever.
808
+
809
+ Reuses chad's own requirement extractor (`audit_requirement_lines`, the same engine
810
+ `done_audit` runs) so the matrix rows are the task's real predicates, not a
811
+ paraphrase. Fires from the THRASH entry point — the exploratory-bash streak — which
812
+ `done_audit` cannot reach (a ran-out turn never calls `done`) and `investigation_
813
+ gate` cannot reach (it freezes the moment any edit lands; this thrash is all
814
+ post-edit). Re-arms up to `cap`, never bare-"call done" (so it will not inflate the
815
+ done-but-wrong bucket that `done_audit` still guards at the actual `done`). Returns
816
+ nudge text or None; caller resets the streak after a firing."""
817
+ if not levers.enabled("verification_matrix"):
818
+ return None
819
+ if explore_streak < threshold or gate_fires >= cap:
820
+ return None
821
+ req_lines = audit_requirement_lines(task_text, audit_extract_paths(task_text))
822
+ levers.fired("verification_matrix", streak=explore_streak,
823
+ reqs=len(req_lines), made_edit=made_edit, fires=gate_fires)
824
+ parts = [
825
+ f"[you've run ~{explore_streak} commands since your last change without "
826
+ "finishing. Stop exploring and close this out as a verification matrix. The "
827
+ "hidden grader checks the task's OWN requirements — for EACH one below, you "
828
+ "need exactly ONE of two things:",
829
+ " (a) EVIDENCE: a real run through the actual public path that shows the "
830
+ "required outcome. A snapshot, your own \"PASS\"/\"OK\" text, \"no error\", or "
831
+ "a check you built from the same assumption as the code do NOT count.",
832
+ " (b) UNVERIFIED: a plain note that you could not verify it. An honest "
833
+ "\"unverified\" is allowed and lets you finish; a false \"it works\" is not.",
834
+ ]
835
+ if req_lines:
836
+ parts.append("Requirements from the task statement:")
837
+ parts += [f" > {ln}" for ln in req_lines]
838
+ else:
839
+ parts.append("Re-read the task statement and list exactly what it requires, "
840
+ "then close each item as (a) or (b).")
841
+ parts.append("When every requirement is (a) or (b), apply any fix still needed, "
842
+ "then call `done`. Do not run more exploratory probes.]")
843
+ return "\n".join(parts)
844
+
845
+
785
846
  def edit_failed_to_land(result: str) -> bool:
786
847
  """True when an edit/symbolic-edit tool result means the change did NOT apply — a
787
848
  no-op (old==new / replacement leaves file unchanged), an unmatched `old` string, or
@@ -91,6 +91,14 @@ LEVERS: dict[str, Lever] = {
91
91
  "After ~6 read-only steps with no landed edit, steer the model to act before the "
92
92
  "step cap kills the turn with an empty patch.",
93
93
  "iter2"),
94
+ "verification_matrix": Lever(
95
+ "After ~8 exploratory-bash steps since the last change (even AFTER an edit has "
96
+ "landed — the case investigation_gate can't see), pull the turn into a bounded "
97
+ "verification phase: for each task requirement, close it with a real causal run "
98
+ "OR an explicit 'unverified', then `done`. Reuses done_audit's requirement "
99
+ "extractor; targets the 69%-of-fails non-convergence (ran-out) bucket while the "
100
+ "honest-unverified escape keeps it from thrashing on the genuinely-unverifiable.",
101
+ "iter16"),
94
102
  "edit_loop_break": Lever(
95
103
  "After 2 consecutive edits that failed to land, stop the model re-trying "
96
104
  "variations and tell it to read the real lines / replace the whole symbol.",
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: chad-code
3
- Version: 1.12.0
3
+ Version: 1.13.0
4
4
  Summary: Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)
5
5
  License-Expression: MIT
6
6
  Project-URL: Repository, https://github.com/nathansutton/chad
@@ -885,6 +885,36 @@ def test_investigation_gate():
885
885
  check("gate bounded by gate_nudges", investigation_gate(20, made_edit=False, gate_nudges=2) is None)
886
886
 
887
887
 
888
+ def test_verification_matrix():
889
+ from chad.guardrails import verification_matrix
890
+ task = (
891
+ "Write /app/out.html so it still triggers alert() after /app/filter.py runs.\n"
892
+ "The output must be valid HTML and must not require user interaction.\n")
893
+ # Below threshold: silent.
894
+ check("no matrix below threshold",
895
+ verification_matrix(task, 4, made_edit=False, gate_fires=0) is None)
896
+ # At/over threshold: fires with the evidence-or-unverified matrix.
897
+ g = verification_matrix(task, 8, made_edit=False, gate_fires=0)
898
+ check("matrix fires at threshold", g is not None)
899
+ check("matrix offers the evidence path", g is not None and "EVIDENCE" in g)
900
+ check("matrix offers the honest-unverified escape",
901
+ g is not None and "UNVERIFIED" in g)
902
+ check("matrix ends at done", g is not None and "`done`" in g)
903
+ # Reuses the requirement extractor: the task's own predicate lines appear as rows.
904
+ check("matrix quotes a real requirement line",
905
+ g is not None and "out.html" in g)
906
+ # The whole point vs investigation_gate: it STILL fires after an edit landed
907
+ # (the break-filter thrash shape — write early, probe forever).
908
+ check("matrix fires even after an edit landed",
909
+ verification_matrix(task, 12, made_edit=True, gate_fires=0) is not None)
910
+ # A task with no extractable requirement lines still fires (generic close-out).
911
+ check("matrix fires with no extractable requirements",
912
+ verification_matrix("do the thing", 8, made_edit=False, gate_fires=0) is not None)
913
+ # Re-armable but bounded by the firing cap.
914
+ check("matrix bounded by cap",
915
+ verification_matrix(task, 30, made_edit=True, gate_fires=6) is None)
916
+
917
+
888
918
  def test_edit_failed_to_land():
889
919
  check("no-op edit failed to land", edit_failed_to_land("[no-op edit: old and new are identical]"))
890
920
  check("not-found failed to land", edit_failed_to_land("[old string not found; no change made.]"))
@@ -1200,6 +1230,7 @@ if __name__ == "__main__":
1200
1230
  test_bash_result_verifies_ignores_trivial_checks()
1201
1231
  test_bash_result_verifies_requires_executing_command()
1202
1232
  test_investigation_gate()
1233
+ test_verification_matrix()
1203
1234
  test_edit_failed_to_land()
1204
1235
  test_edit_loop_break()
1205
1236
  test_done_rejection()
@@ -169,6 +169,16 @@ def test_investigation_gate(monkeypatch):
169
169
  assert guardrails.investigation_gate(8, made_edit=False, gate_nudges=0) is None
170
170
 
171
171
 
172
+ def test_verification_matrix(monkeypatch):
173
+ n = bite("verification_matrix")
174
+ on(monkeypatch)
175
+ assert guardrails.verification_matrix("write /app/out.txt", 8,
176
+ made_edit=True, gate_fires=0)
177
+ off(monkeypatch, n)
178
+ assert guardrails.verification_matrix("write /app/out.txt", 8,
179
+ made_edit=True, gate_fires=0) is None
180
+
181
+
172
182
  def test_edit_loop_break(monkeypatch):
173
183
  n = bite("edit_loop_break")
174
184
  on(monkeypatch)
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes