chad-code 1.11.0__tar.gz → 1.13.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (119) hide show
  1. {chad_code-1.11.0 → chad_code-1.13.0}/PKG-INFO +1 -1
  2. {chad_code-1.11.0 → chad_code-1.13.0}/pyproject.toml +1 -1
  3. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/__init__.py +1 -1
  4. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/agent.py +72 -6
  5. chad_code-1.13.0/src/chad/checkpoint.py +150 -0
  6. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/guardrails.py +61 -0
  7. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/levers.py +52 -0
  8. chad_code-1.13.0/src/chad/seatbelt.py +255 -0
  9. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/tools.py +46 -5
  10. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/tui.py +30 -1
  11. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/PKG-INFO +1 -1
  12. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/SOURCES.txt +4 -0
  13. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_agent_guards.py +31 -0
  14. chad_code-1.13.0/tests/test_checkpoint.py +132 -0
  15. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_lever_bite.py +81 -0
  16. chad_code-1.13.0/tests/test_seatbelt.py +336 -0
  17. {chad_code-1.11.0 → chad_code-1.13.0}/LICENSE +0 -0
  18. {chad_code-1.11.0 → chad_code-1.13.0}/README.md +0 -0
  19. {chad_code-1.11.0 → chad_code-1.13.0}/setup.cfg +0 -0
  20. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/ambient.py +0 -0
  21. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/atif.py +0 -0
  22. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/base_engine.py +0 -0
  23. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/bench.py +0 -0
  24. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/cli.py +0 -0
  25. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/compaction.py +0 -0
  26. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/completion_engine.py +0 -0
  27. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/config.py +0 -0
  28. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/diag.py +0 -0
  29. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/engine.py +0 -0
  30. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/ignore.py +0 -0
  31. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/lsp.py +0 -0
  32. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/lspclient.py +0 -0
  33. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/lspservers.py +0 -0
  34. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/mcp.py +0 -0
  35. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/mcp_oauth.py +0 -0
  36. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/mlx_fastpath.py +0 -0
  37. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/mlx_moe_fused.py +0 -0
  38. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/mlx_qsdpa.py +0 -0
  39. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/LICENSE +0 -0
  40. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/__init__.py +0 -0
  41. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/alignment.py +0 -0
  42. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/attention.py +0 -0
  43. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/audio.py +0 -0
  44. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/cache.py +0 -0
  45. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/conformer.py +0 -0
  46. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/ctc.py +0 -0
  47. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/parakeet.py +0 -0
  48. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/rnnt.py +0 -0
  49. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/tokenizer.py +0 -0
  50. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/parakeet/utils.py +0 -0
  51. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/profiles.py +0 -0
  52. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/prompt.py +0 -0
  53. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/prove.py +0 -0
  54. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/render.py +0 -0
  55. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/repomap.py +0 -0
  56. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/serve.py +0 -0
  57. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/session.py +0 -0
  58. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/skills.py +0 -0
  59. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/speech.py +0 -0
  60. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/symbols.py +0 -0
  61. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/syntaxgate.py +0 -0
  62. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/toolcall_parse.py +0 -0
  63. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad/validate.py +0 -0
  64. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/dependency_links.txt +0 -0
  65. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/entry_points.txt +0 -0
  66. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/requires.txt +0 -0
  67. {chad_code-1.11.0 → chad_code-1.13.0}/src/chad_code.egg-info/top_level.txt +0 -0
  68. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_agent.py +0 -0
  69. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_agent_e2e.py +0 -0
  70. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_ambient.py +0 -0
  71. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_atif.py +0 -0
  72. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_bench.py +0 -0
  73. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_cli.py +0 -0
  74. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_compact_notice.py +0 -0
  75. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_compaction.py +0 -0
  76. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_completion_engine.py +0 -0
  77. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_config.py +0 -0
  78. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_confirm_preview.py +0 -0
  79. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_done_audit.py +0 -0
  80. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_drift_warn.py +0 -0
  81. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_edit.py +0 -0
  82. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_edit_corruption.py +0 -0
  83. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_engine.py +0 -0
  84. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_engine_kvquant.py +0 -0
  85. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_engine_pld_hybrid.py +0 -0
  86. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_feel_pack.py +0 -0
  87. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_garble_invariant.py +0 -0
  88. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_gate.py +0 -0
  89. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_ignore.py +0 -0
  90. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_intent.py +0 -0
  91. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_lever_instrumentation.py +0 -0
  92. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_levers.py +0 -0
  93. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_log_redaction.py +0 -0
  94. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_lsp.py +0 -0
  95. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_lsp_live.py +0 -0
  96. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_lspclient.py +0 -0
  97. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_mcp.py +0 -0
  98. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_mcp_oauth.py +0 -0
  99. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_mlx_fastpath.py +0 -0
  100. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_mlx_moe_fused.py +0 -0
  101. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_mlx_qsdpa.py +0 -0
  102. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_plan_review.py +0 -0
  103. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_prove.py +0 -0
  104. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_render.py +0 -0
  105. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_replace_lines.py +0 -0
  106. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_repomap.py +0 -0
  107. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_repomap_polyglot.py +0 -0
  108. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_serve.py +0 -0
  109. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_session.py +0 -0
  110. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_skills.py +0 -0
  111. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_speech.py +0 -0
  112. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_speech_tui.py +0 -0
  113. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_subagent.py +0 -0
  114. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_symbols.py +0 -0
  115. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_syntaxgate.py +0 -0
  116. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_toolcall_parse.py +0 -0
  117. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_tools.py +0 -0
  118. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_tui.py +0 -0
  119. {chad_code-1.11.0 → chad_code-1.13.0}/tests/test_validate.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: chad-code
3
- Version: 1.11.0
3
+ Version: 1.13.0
4
4
  Summary: Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)
5
5
  License-Expression: MIT
6
6
  Project-URL: Repository, https://github.com/nathansutton/chad
@@ -4,7 +4,7 @@
4
4
  # import name, and command name are independent. `uvx chad-code` runs the alias
5
5
  # script added under [project.scripts].
6
6
  name = "chad-code"
7
- version = "1.11.0"
7
+ version = "1.13.0"
8
8
  description = "Local MLX-backed, Claude-Code-style coding agent (Apple Silicon, Ornith 35B/9B)"
9
9
  readme = "README.md"
10
10
  license = "MIT"
@@ -4,7 +4,7 @@ A flat collection of cooperating modules behind one console script (``chad``):
4
4
  the inference engine, the tool layer, the agent loop, and the terminal UI.
5
5
  """
6
6
 
7
- __version__ = "1.11.0"
7
+ __version__ = "1.13.0"
8
8
 
9
9
  # chad sets no MLX_* runtime vars. MLX_METAL_FAST_SYNCH, MLX_MAX_OPS_PER_BUFFER
10
10
  # and MLX_MAX_MB_PER_BUFFER were each measured end-to-end on the 35B and every
@@ -14,7 +14,18 @@ import re
14
14
  import sys
15
15
  import time
16
16
 
17
- from . import ambient, atif, compaction, config, guardrails, levers, session, syntaxgate
17
+ from . import (
18
+ ambient,
19
+ atif,
20
+ checkpoint,
21
+ compaction,
22
+ config,
23
+ guardrails,
24
+ levers,
25
+ seatbelt,
26
+ session,
27
+ syntaxgate,
28
+ )
18
29
  from .base_engine import BackendError, BaseEngine
19
30
  from .diag import args_preview, log, redact, result_preview
20
31
  from .prompt import build_subagent_prompt, build_system_prompt, classify_intent
@@ -326,7 +337,7 @@ class Agent:
326
337
  ctx_limit: int = 24000, mode: str = None, emit=None,
327
338
  confirm=None, should_stop=None, drain_steering=None,
328
339
  thinking: bool = True,
329
- max_gen_tokens: int = 8192, resume: list = None, persist: bool = False,
340
+ max_gen_tokens: int = None, resume: list = None, persist: bool = False,
330
341
  think_budget: int = None, think_ceiling: int = None,
331
342
  turn_budget_tokens: int = None,
332
343
  turn_budget_s: float = None, subagent: bool = False,
@@ -366,8 +377,21 @@ class Agent:
366
377
  # Per-step generation cap. The old 2048 default truncated legitimate work —
367
378
  # a reasoning turn that thinks, then emits a `write` of a whole test file can
368
379
  # exceed 2048 tokens, and the cut-off was being misread as a final answer
369
- # (the "answers on paper then stops" bug). 8192 leaves room for think + a
370
- # full-file write; truncation past it is now detected and nudged, not accepted.
380
+ # (the "answers on paper then stops" bug). 8192 fixed that but had its own
381
+ # loss class: on hard problems a long chain of thought can pin the cap while
382
+ # still inside <think>, so the step ends as discarded reasoning with no
383
+ # action and the next step re-derives from scratch — trace measurement shows
384
+ # these mid-think pins cluster on exactly the tasks that fail, and that an
385
+ # uncapped run of the same model almost never exceeds 8192 on its own
386
+ # (p99 ~4.7k) yet the rare 15-20k thought that does run long completes real
387
+ # work when allowed to finish. So the cap buys little in the common case and
388
+ # converts the long-thought tail into zero-progress dead steps. 32768 keeps
389
+ # a backstop against non-repetitive runaway garble (literal decode loops are
390
+ # the repeat guard's job, default ON) while fitting the longest observed
391
+ # legitimate thought with room for the action. Env knob for A/B without a
392
+ # code change (CHAD_* family).
393
+ if max_gen_tokens is None:
394
+ max_gen_tokens = config.env_int("CHAD_MAX_GEN_TOKENS", 32768)
371
395
  self.max_gen_tokens = max_gen_tokens
372
396
  # Soft think-cap base. None => the mechanism is OFF and generation is
373
397
  # byte-identical to before. Falls back to the CHAD_THINK_BUDGET env knob so the
@@ -838,6 +862,10 @@ class Agent:
838
862
  break_nudges = 0 # times we escalated a stuck edit this turn
839
863
  readonly_streak = 0 # consecutive steps with substantive tools but no landed edit
840
864
  gate_nudges = 0 # times the investigation->edit gate fired this turn
865
+ explore_streak = 0 # consecutive bash steps since the last landed edit (counts
866
+ # AFTER an edit too — investigation_gate can't; see
867
+ # guardrails.verification_matrix)
868
+ explore_gate_fires = 0 # times the verification-matrix gate fired this turn
841
869
  subagent_sigs = set() # (description, prompt) of sub-agents already spawned this turn
842
870
  truncation_nudges = 0 # times we pushed past a token-cap truncation this turn
843
871
  garble_nudges = 0 # malformed-tool-call re-nudges (own counter, so a step-0
@@ -1094,7 +1122,7 @@ class Agent:
1094
1122
  return n >= _cap and "</think>" not in text_so_far
1095
1123
 
1096
1124
  # Degenerate-repetition stop (default ON): greedy decode can lock into
1097
- # repeating one short string until the 8192-token cap — ~4 minutes of dead
1125
+ # repeating one short string until the max_gen_tokens cap — minutes of dead
1098
1126
  # generation per occurrence at 9B decode speed. Checked every 16 tokens on
1099
1127
  # the generation's tail; a hit stops the step (rep_fired tells this stopper
1100
1128
  # apart from the think-cap, which shares stats.stop_condition_fired) and the
@@ -1357,7 +1385,7 @@ class Agent:
1357
1385
  # inside <think> — a distinct sub-bucket from a THINK-CAP/ceiling stop
1358
1386
  # (those are agent-side stop_conditions; this is max_gen_tokens itself),
1359
1387
  # so future autopsies can size it directly instead of inferring from
1360
- # max_gen==8192. A salvaged step already injected the closing tag, so
1388
+ # the max_gen value. A salvaged step already injected the closing tag, so
1361
1389
  # this is naturally False for it.
1362
1390
  "reasoning_length_stop": hit_cap and "</think>" not in text,
1363
1391
  "deadline_abort": deadline_fired[0], # 103: wall-clock hard wrap-up cut
@@ -1971,12 +1999,27 @@ class Agent:
1971
1999
  else:
1972
2000
  _t0 = time.perf_counter()
1973
2001
  self.tool_dispatches += 1
2002
+ # A snapshot happens AFTER approval, immediately before the tool
2003
+ # runs — a denied edit must not leave a checkpoint claiming it ran.
2004
+ if (name in AUTO_EDIT_TOOLS
2005
+ and levers.enabled("edit_checkpoint")):
2006
+ _ref = checkpoint.snapshot(
2007
+ os.getcwd(), f"before {name} {args.get('path', '')}".strip())
2008
+ if _ref:
2009
+ levers.fired("edit_checkpoint", step=step, ref=_ref)
2010
+ # Seatbelt context is scoped to exactly this dispatch: `active`
2011
+ # reflects the mode of the agent actually executing, so a
2012
+ # sub-agent's yolo loop is confined inside a normal-mode parent
2013
+ # turn, and the context can't leak to the `!cmd` passthrough.
2014
+ seatbelt.set_context(self.mode == "yolo", os.getcwd())
1974
2015
  try:
1975
2016
  result = fn(args, self._should_stop)
1976
2017
  if plan_write and result.startswith("[wrote"):
1977
2018
  self.last_plan_path = os.path.abspath(args["path"])
1978
2019
  except Exception as e: # noqa: BLE001 - surface tool errors to model
1979
2020
  result = f"[tool error: {type(e).__name__}: {e}]"
2021
+ finally:
2022
+ seatbelt.set_context(False, None)
1980
2023
  _tool_s = time.perf_counter() - _t0
1981
2024
  # Backstop: bound the prefill from any tool, AND from the step as a
1982
2025
  # whole — several calls in one step stack into one prefill, so later
@@ -2144,6 +2187,29 @@ class Agent:
2144
2187
  step, gate_nudges)
2145
2188
  self.messages.append({"role": "tool", "name": "edit", "content": gate})
2146
2189
 
2190
+ # Convergence gate (verification-matrix design): count exploratory-
2191
+ # bash steps since the last landed edit — unlike readonly_streak/
2192
+ # investigation_gate this keeps counting AFTER an edit lands, which is where
2193
+ # the demonstrated thrash lives (write early, then probe 60 more times
2194
+ # without calling done). A NEW edit this step resets it; a bash-only step
2195
+ # advances it. When it bites, the model gets the task's real requirements as
2196
+ # a matrix to close by evidence-or-unverified, not an open-ended "keep going".
2197
+ if made_edit and not _gov_prev_made:
2198
+ explore_streak = 0
2199
+ elif any(n == "bash" for n, _ in calls):
2200
+ explore_streak += 1
2201
+ if not read_only_intent:
2202
+ matrix = guardrails.verification_matrix(
2203
+ guardrails.audit_task_text(user_text),
2204
+ explore_streak, made_edit, explore_gate_fires)
2205
+ if matrix:
2206
+ explore_gate_fires += 1
2207
+ explore_streak = 0
2208
+ log.info("VERIFICATION-MATRIX gate at step %d (bash-streak, no "
2209
+ "change) -> nudge #%d", step, explore_gate_fires)
2210
+ self.messages.append(
2211
+ {"role": "tool", "name": "bash", "content": matrix})
2212
+
2147
2213
  # Break a flailing-probe run (e.g. guessing the test runner, repeated
2148
2214
  # `python -c import` checks) that the exact-call loop guard can't see because
2149
2215
  # each failing command differs by a few characters.
@@ -0,0 +1,150 @@
1
+ """Shadow-git checkpoints for file edits (lever: edit_checkpoint).
2
+
3
+ Auto-approved edits are justified as "a diff you can read and revert" — this
4
+ module supplies the revert. Before each file-mutating tool lands, the workspace
5
+ is committed to a *shadow* git repository under ~/.chad/checkpoints/<hash>. The
6
+ user's own .git is never opened, never written, never required to exist: the
7
+ shadow repo has its own GIT_DIR and treats the workspace purely as a work tree.
8
+ /undo and /restore in the TUI check files back out of it.
9
+
10
+ Failure policy: never raise. An edit must not die because the checkpoint
11
+ machinery hiccuped — snapshot() returns None on any failure and the tool call
12
+ proceeds unprotected (logged, and visible in `chad levers` firing counts as the
13
+ absence of a fire).
14
+
15
+ Restore policy (deliberately conservative): `git restore --source=<ref>` puts
16
+ every snapshotted file back to its snapshotted content. Files *created after*
17
+ the snapshot are left in place — deleting files is exactly the blast radius this
18
+ lever exists to contain, so the restore path never does it.
19
+ """
20
+
21
+ import hashlib
22
+ import logging
23
+ import os
24
+ import subprocess
25
+ from typing import Optional
26
+
27
+ log = logging.getLogger("chad")
28
+
29
+ _GIT_TIMEOUT_S = 120 # first snapshot of a big tree is seconds; never hang a turn
30
+
31
+ # Junk that snapshots must not swallow when the workspace has no .gitignore of its
32
+ # own (a non-git project): written to the shadow repo's info/exclude, which
33
+ # composes with any workspace .gitignore rather than replacing it.
34
+ _DEFAULT_EXCLUDES = (".venv/\nvenv/\nnode_modules/\n__pycache__/\n*.pyc\n"
35
+ ".mypy_cache/\n.ruff_cache/\n.pytest_cache/\n.DS_Store\n")
36
+
37
+
38
+ def _history_root() -> str:
39
+ # CHAD_CHECKPOINT_DIR: test/e2e override so suites never write real home state.
40
+ # NOT ~/.chad/history — that name is taken by the TUI's prompt-history FILE.
41
+ env = os.environ.get("CHAD_CHECKPOINT_DIR")
42
+ if env:
43
+ return env
44
+ return os.path.join(os.path.expanduser("~"), ".chad", "checkpoints")
45
+
46
+
47
+ def shadow_dir(workspace: str) -> str:
48
+ ws = os.path.realpath(workspace)
49
+ tag = hashlib.sha1(ws.encode()).hexdigest()[:16]
50
+ return os.path.join(_history_root(), tag, "shadow.git")
51
+
52
+
53
+ def _git(workspace: str, *args: str) -> subprocess.CompletedProcess:
54
+ """Run git against the shadow repo with the workspace as work tree. Identity
55
+ and signing come from -c flags so the user's global config is never consulted
56
+ for authorship and never mutated."""
57
+ cmd = ["git", "--git-dir", shadow_dir(workspace), "--work-tree",
58
+ os.path.realpath(workspace),
59
+ "-c", "user.email=chad@localhost", "-c", "user.name=chad",
60
+ "-c", "commit.gpgsign=false", "-c", "core.hooksPath=/dev/null",
61
+ *args]
62
+ return subprocess.run(cmd, capture_output=True, text=True,
63
+ timeout=_GIT_TIMEOUT_S, check=False)
64
+
65
+
66
+ def _ensure_shadow(workspace: str) -> bool:
67
+ sd = shadow_dir(workspace)
68
+ if os.path.isdir(sd):
69
+ return True
70
+ try:
71
+ os.makedirs(os.path.dirname(sd), exist_ok=True)
72
+ r = subprocess.run(["git", "init", "-q", "--bare", sd],
73
+ capture_output=True, text=True,
74
+ timeout=_GIT_TIMEOUT_S, check=False)
75
+ if r.returncode != 0:
76
+ log.warning("CHECKPOINT shadow init failed: %s", r.stderr.strip())
77
+ return False
78
+ info = os.path.join(sd, "info")
79
+ os.makedirs(info, exist_ok=True)
80
+ with open(os.path.join(info, "exclude"), "w", encoding="utf-8") as fh:
81
+ fh.write(_DEFAULT_EXCLUDES)
82
+ return True
83
+ except (OSError, subprocess.SubprocessError) as e:
84
+ log.warning("CHECKPOINT shadow init failed: %s", e)
85
+ return False
86
+
87
+
88
+ def snapshot(workspace: str, label: str) -> Optional[str]:
89
+ """Commit the workspace state to the shadow repo; return the short hash, or
90
+ None if no checkpoint exists after the attempt. An unchanged tree is not a
91
+ failure — the previous snapshot already covers it, so its hash is returned."""
92
+ try:
93
+ if not _ensure_shadow(workspace):
94
+ return None
95
+ add = _git(workspace, "add", "-A")
96
+ if add.returncode != 0:
97
+ log.warning("CHECKPOINT add failed: %s", add.stderr.strip())
98
+ return None
99
+ commit = _git(workspace, "commit", "-q", "-m", label)
100
+ head = _git(workspace, "rev-parse", "--short", "HEAD")
101
+ if head.returncode != 0:
102
+ # Nothing staged AND no prior snapshot — an empty workspace's first
103
+ # checkpoint. Record the empty state anyway so the timeline exists.
104
+ commit = _git(workspace, "commit", "-q", "--allow-empty", "-m", label)
105
+ head = _git(workspace, "rev-parse", "--short", "HEAD")
106
+ if head.returncode != 0:
107
+ log.warning("CHECKPOINT commit failed: %s", commit.stderr.strip())
108
+ return None
109
+ return head.stdout.strip()
110
+ except (OSError, subprocess.SubprocessError) as e:
111
+ log.warning("CHECKPOINT snapshot failed: %s", e)
112
+ return None
113
+
114
+
115
+ def snapshots(workspace: str, limit: int = 10) -> list:
116
+ """Newest-first [(short_hash, iso_time, label)] — empty on any failure."""
117
+ try:
118
+ r = _git(workspace, "log", f"-{limit}", "--format=%h%x09%cI%x09%s")
119
+ if r.returncode != 0:
120
+ return []
121
+ rows = []
122
+ for line in r.stdout.splitlines():
123
+ parts = line.split("\t", 2)
124
+ if len(parts) == 3:
125
+ rows.append((parts[0], parts[1], parts[2]))
126
+ return rows
127
+ except (OSError, subprocess.SubprocessError):
128
+ return []
129
+
130
+
131
+ def restore(workspace: str, ref: str = "HEAD") -> str:
132
+ """Check snapshotted files back out at `ref`. Returns a human-readable
133
+ one-liner (also used verbatim by the TUI)."""
134
+ try:
135
+ ok = _git(workspace, "rev-parse", "--verify", "--quiet", f"{ref}^{{commit}}")
136
+ if ok.returncode != 0:
137
+ have = snapshots(workspace, limit=1)
138
+ return (f"no checkpoint named {ref!r}" if have else
139
+ "no checkpoints exist for this workspace yet — snapshots are "
140
+ "taken before file edits when the edit_checkpoint lever is on")
141
+ changed = _git(workspace, "diff", "--name-only", ref)
142
+ n = len([ln for ln in changed.stdout.splitlines() if ln.strip()])
143
+ r = _git(workspace, "restore", "--worktree", f"--source={ref}", "--", ".")
144
+ if r.returncode != 0:
145
+ return f"restore failed: {r.stderr.strip() or 'unknown git error'}"
146
+ label = _git(workspace, "log", "-1", "--format=%s", ref).stdout.strip()
147
+ return (f"restored {n} file(s) to checkpoint {ref} ({label}). Files created "
148
+ f"since that snapshot were left in place.")
149
+ except (OSError, subprocess.SubprocessError) as e:
150
+ return f"restore failed: {e}"
@@ -782,6 +782,67 @@ def investigation_gate(readonly_streak, made_edit, gate_nudges, threshold=6):
782
782
  "verify it. If you already know the fix, apply it this step.]")
783
783
 
784
784
 
785
+ def verification_matrix(task_text, explore_streak, made_edit, gate_fires,
786
+ threshold=8, cap=6):
787
+ """Pull a thrashing turn into a bounded verification phase.
788
+
789
+ The demonstrated failure (measured on the bare full-89 run against a reference
790
+ harness on the same model/box): chad's #1 loss class is not wrong answers, it is
791
+ non-convergence. 69% of failing trials never called `done` — they ran a long tail
792
+ of successful, exploratory bash (median 39 commands vs 12 on passes) with the SAME
793
+ two edits a passing trial makes, and the step cap killed them mid-poke. On
794
+ break-filter the failing trial ran 72 bash probes / 2 writes across 75 steps while
795
+ chad's OWN passing trial found the identical exploit in 13; the model knows the
796
+ answer and researches past the budget. A reference harness on the same weights
797
+ sits at chad's PASSING step count.
798
+
799
+ The design ported here reframes completion as a VERIFICATION MATRIX: every
800
+ requirement the task states must be closed by ONE of (a) causal evidence from a
801
+ real run through the public path, or (b) an explicit "unverified" note. That
802
+ single rule both BOUNDS the loop (a finite checklist has a terminal state, so the
803
+ turn knows when to stop) and blocks false-done (a row is not filled by a
804
+ self-authored PASS label, a snapshot, "no error", or an oracle built from the same
805
+ assumption as the code). The honest-unverified escape is load-bearing: it lets a
806
+ turn finish a genuinely-unverifiable requirement instead of thrashing on it
807
+ forever.
808
+
809
+ Reuses chad's own requirement extractor (`audit_requirement_lines`, the same engine
810
+ `done_audit` runs) so the matrix rows are the task's real predicates, not a
811
+ paraphrase. Fires from the THRASH entry point — the exploratory-bash streak — which
812
+ `done_audit` cannot reach (a ran-out turn never calls `done`) and `investigation_
813
+ gate` cannot reach (it freezes the moment any edit lands; this thrash is all
814
+ post-edit). Re-arms up to `cap`, never bare-"call done" (so it will not inflate the
815
+ done-but-wrong bucket that `done_audit` still guards at the actual `done`). Returns
816
+ nudge text or None; caller resets the streak after a firing."""
817
+ if not levers.enabled("verification_matrix"):
818
+ return None
819
+ if explore_streak < threshold or gate_fires >= cap:
820
+ return None
821
+ req_lines = audit_requirement_lines(task_text, audit_extract_paths(task_text))
822
+ levers.fired("verification_matrix", streak=explore_streak,
823
+ reqs=len(req_lines), made_edit=made_edit, fires=gate_fires)
824
+ parts = [
825
+ f"[you've run ~{explore_streak} commands since your last change without "
826
+ "finishing. Stop exploring and close this out as a verification matrix. The "
827
+ "hidden grader checks the task's OWN requirements — for EACH one below, you "
828
+ "need exactly ONE of two things:",
829
+ " (a) EVIDENCE: a real run through the actual public path that shows the "
830
+ "required outcome. A snapshot, your own \"PASS\"/\"OK\" text, \"no error\", or "
831
+ "a check you built from the same assumption as the code do NOT count.",
832
+ " (b) UNVERIFIED: a plain note that you could not verify it. An honest "
833
+ "\"unverified\" is allowed and lets you finish; a false \"it works\" is not.",
834
+ ]
835
+ if req_lines:
836
+ parts.append("Requirements from the task statement:")
837
+ parts += [f" > {ln}" for ln in req_lines]
838
+ else:
839
+ parts.append("Re-read the task statement and list exactly what it requires, "
840
+ "then close each item as (a) or (b).")
841
+ parts.append("When every requirement is (a) or (b), apply any fix still needed, "
842
+ "then call `done`. Do not run more exploratory probes.]")
843
+ return "\n".join(parts)
844
+
845
+
785
846
  def edit_failed_to_land(result: str) -> bool:
786
847
  """True when an edit/symbolic-edit tool result means the change did NOT apply — a
787
848
  no-op (old==new / replacement leaves file unchanged), an unmatched `old` string, or
@@ -91,6 +91,14 @@ LEVERS: dict[str, Lever] = {
91
91
  "After ~6 read-only steps with no landed edit, steer the model to act before the "
92
92
  "step cap kills the turn with an empty patch.",
93
93
  "iter2"),
94
+ "verification_matrix": Lever(
95
+ "After ~8 exploratory-bash steps since the last change (even AFTER an edit has "
96
+ "landed — the case investigation_gate can't see), pull the turn into a bounded "
97
+ "verification phase: for each task requirement, close it with a real causal run "
98
+ "OR an explicit 'unverified', then `done`. Reuses done_audit's requirement "
99
+ "extractor; targets the 69%-of-fails non-convergence (ran-out) bucket while the "
100
+ "honest-unverified escape keeps it from thrashing on the genuinely-unverifiable.",
101
+ "iter16"),
94
102
  "edit_loop_break": Lever(
95
103
  "After 2 consecutive edits that failed to land, stop the model re-trying "
96
104
  "variations and tell it to read the real lines / replace the whole symbol.",
@@ -514,6 +522,50 @@ LEVERS: dict[str, Lever] = {
514
522
  "only a server already warmed by find_refs/hover/disambiguation is "
515
523
  "consulted; absent one, silence.",
516
524
  "ctxengine"),
525
+
526
+ # --- group "safety": blast-radius containment for unattended mutation. ---------
527
+ "yolo_seatbelt": Lever(
528
+ "In yolo mode on macOS, each bash command's shell child runs under a "
529
+ "Seatbelt profile (sandbox-exec) that denies file writes outside the "
530
+ "workspace, temp dirs, tool caches, and ~/.chad — reads, network, and exec "
531
+ "stay open, and ~/.chad/checkpoints (the undo history) is carved back out "
532
+ "so a sandboxed command cannot delete it. The model process is never "
533
+ "sandboxed, only the spawned shell. The startup probe proves the profile "
534
+ "ENFORCES (an allowed write must land, a denied write must not) before any "
535
+ "confinement is claimed; either failure runs unconfined with a loud log. "
536
+ "Fires when a denial is detected in command output: each fire is a write "
537
+ "the command-pattern denylist did not catch.",
538
+ "safety"),
539
+ "edit_checkpoint": Lever(
540
+ "Before a file-mutating tool lands, commit a workspace snapshot to a "
541
+ "shadow git repo under ~/.chad/checkpoints (the project's own .git is "
542
+ "never touched or required); /undo and /restore in the TUI revert from "
543
+ "it. Fires once per snapshot taken and once per restore performed.",
544
+ "safety"),
545
+ "seatbelt_protect_git": Lever(
546
+ "Opt-in tier on top of yolo_seatbelt: the workspace's git metadata "
547
+ "(.git, and a worktree's external gitdir + common dir) is write-DENIED "
548
+ "inside the otherwise-writable workspace, so an unreviewed command "
549
+ "cannot destroy project history (`rm -rf .git`) — and with "
550
+ "edit_checkpoint on, the undo snapshots stay tamper-proof from inside "
551
+ "the sandbox too. The cost is real: every .git-writing git command "
552
+ "(commit, add, checkout) EPERMs — sized at <=2.65% of 91,910 real "
553
+ "session commands, concentrated in 14% of sessions — which is why this "
554
+ "is its own lever, not part of the base profile. Inert unless "
555
+ "yolo_seatbelt is also enabled; a denial it causes surfaces through "
556
+ "yolo_seatbelt's own firing (the profile cannot say which rule bit).",
557
+ "safety", fires=PASSIVE),
558
+ "bash_env_guard": Lever(
559
+ "Spawned bash children get a FILTERED copy of the environment: variable "
560
+ "names shaped like credentials (…_TOKEN, …_SECRET, …_PASSWORD, "
561
+ "…_API_KEY, …_ACCESS_KEY(_ID), …_PRIVATE_KEY, …_CREDENTIALS) are "
562
+ "dropped, closing the gap where a filesystem sandbox still hands every "
563
+ "command the operator's cloud keys. Name-pattern only — values are "
564
+ "never read or logged. OFF (the default) inherits the parent env "
565
+ "untouched, today's behavior; commands that legitimately need a "
566
+ "credential need the lever off. Fires per spawn that actually dropped "
567
+ "something, with the count.",
568
+ "safety"),
517
569
  }
518
570
 
519
571