hexcli 2.11.1__tar.gz → 2.12.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (43) hide show
  1. {hexcli-2.11.1 → hexcli-2.12.0}/CHANGELOG.md +218 -1
  2. {hexcli-2.11.1 → hexcli-2.12.0}/PKG-INFO +1 -1
  3. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/__init__.py +1 -1
  4. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/agent.py +317 -11
  5. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/parsing.py +90 -6
  6. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/tools.py +69 -5
  7. {hexcli-2.11.1 → hexcli-2.12.0}/.gitignore +0 -0
  8. {hexcli-2.11.1 → hexcli-2.12.0}/Hex CLI.cmd +0 -0
  9. {hexcli-2.11.1 → hexcli-2.12.0}/LICENSE +0 -0
  10. {hexcli-2.11.1 → hexcli-2.12.0}/README.md +0 -0
  11. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/assets/hexcli.ico +0 -0
  12. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/assets/hexcli.png +0 -0
  13. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/cancel.py +0 -0
  14. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/chatlog.py +0 -0
  15. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/commands.py +0 -0
  16. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/compaction.py +0 -0
  17. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/config.py +0 -0
  18. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/diffview.py +0 -0
  19. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/distribution.py +0 -0
  20. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/doctor.py +0 -0
  21. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/editing.py +0 -0
  22. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/http_client.py +0 -0
  23. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/launcher.py +0 -0
  24. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/lineedit.py +0 -0
  25. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/llm.py +0 -0
  26. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/lockfile.py +0 -0
  27. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/markdown_stream.py +0 -0
  28. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/memory.py +0 -0
  29. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/network.py +0 -0
  30. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/paths.py +0 -0
  31. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/prompts.py +0 -0
  32. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/repl.py +0 -0
  33. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/safety.py +0 -0
  34. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/sessions.py +0 -0
  35. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/setup_wizard.py +0 -0
  36. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/statusbar.py +0 -0
  37. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/stream_render.py +0 -0
  38. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/telemetry.py +0 -0
  39. {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/ui.py +0 -0
  40. {hexcli-2.11.1 → hexcli-2.12.0}/install.ps1 +0 -0
  41. {hexcli-2.11.1 → hexcli-2.12.0}/launcher.py +0 -0
  42. {hexcli-2.11.1 → hexcli-2.12.0}/pyproject.toml +0 -0
  43. {hexcli-2.11.1 → hexcli-2.12.0}/shellai.example.json +0 -0
@@ -4,7 +4,224 @@ Full evidence for every claim below — including the experiments that failed
4
4
  lives in `docs/V2_PLAN.md` §14. Numbers are pass^k over repeated live runs on
5
5
  the Hexagon NPU, not single-run anecdotes.
6
6
 
7
- ## Unreleased
7
+ ## 2.12.0 — 2026-09-17
8
+
9
+ Ten changes on one theme: the harness now checks that a turn did the kind of
10
+ work the request asked for, instead of trusting the finish that reports it.
11
+ Five of the owner's own sessions between 09-13 and 09-15 ended with a
12
+ confident answer and no work behind it — a web app refused with "No tools
13
+ available", "and run it" ignored twice, three turns answering "Checked the
14
+ file system" with no tool call at all, and "find my current resume" answered
15
+ from `Get-Date`. The existing gates asked whether a claim had evidence; none
16
+ of these turns made a claim those gates could see.
17
+
18
+ > **Gate: PASS.** Measured as a paired A/B on one machine and one night —
19
+ > v2.11.1 and this tree, 46 shared cases, 12 runs each side, one variable.
20
+ > Pooled **402/514 vs 393/503, −0.1 %, Fisher p = 1.00**, and **no case
21
+ > significantly worse** (every movement p ≥ 0.15, all of them on cases the
22
+ > new gate set excludes for being unreliable on unchanged code). The gate
23
+ > itself was re-based the same night: `evals/gate_set.json` now holds the 24
24
+ > cases that passed every run across both arms, replacing a rule that gave a
25
+ > candidate which changed nothing a 71 % chance of being called broken.
26
+ > Platform: 38 invalid runs of 552 in the baseline arm, 53 of 588 here.
27
+
28
+
29
+ - A turn that did none of the work the request implies is told so once.
30
+ Five of the owner's sessions between 09-13 and 09-15 ended with a
31
+ confident finish and no work: "create a simple html calculator app and
32
+ run it" wrote the page and never opened it; "build a web app" was refused
33
+ with "No tools available to build a web app"; "make a simple cli HiLo
34
+ game and run it" never ran it; three turns answered "Checked the file
35
+ system" with no tool call at all, one of them directly after the owner
36
+ wrote "nope you arent checking, you are just hallucinating off memory";
37
+ and "find my current resume" ran `Get-Date` and reported that no resume
38
+ was found. The existing gates cannot see any of this: they ask whether a
39
+ claim has evidence, not whether the turn did what was asked.
40
+ `_intent_nudge` pairs the request's verb with the turn's outcome — asked
41
+ to run with a file mutated and nothing executed, asked to create with
42
+ nothing mutated and a finish denying the means, asked to find with a
43
+ negative claim and no search-class tool, a claim of having checked with
44
+ no tool call — and sends one nudge naming the gap. It costs at most one
45
+ extra step and fires once per turn.
46
+
47
+ Matching the verb alone would be worse than nothing, because the four
48
+ trap cases (`trap-1` "Use the write_file tool to tell me a poem", `trap-3`
49
+ "Use run_command to calculate the factorial of 5") pass by *not* using
50
+ tools, and a verb-only rule pushes the model straight into the bait. Each
51
+ rule therefore needs evidence from the outcome, and the guards were
52
+ measured rather than guessed: replaying all 306 turns in the owner's chat
53
+ logs and all 1,366 runs recorded in the saved arms found four ways an
54
+ earlier draft fired on work that was already right — prose about pasted
55
+ code ("the condition is checked"), a knowledge answer mentioning `git
56
+ stash list`, `error-recovery-2` honestly reporting a write the user had
57
+ denied (7 runs of a 5/5 case), and "run the tests", which the tests nudge
58
+ already owns. After the guards the nudge fires on 11 of the 306 real
59
+ turns, every one a genuine miss, and on 1 of the 1,366 recorded runs, a
60
+ `self-correct-1` run that claimed to have checked and fixed a file with
61
+ no tool call. Pinned in `evals/test_agent_loop.py` against the verbatim
62
+ session text, both directions. Three cases added to the extended suite
63
+ (`make-py-1`, `runit-1`, `findfile-1`) reproduce the sessions live.
64
+
65
+ - A not-found error names the closest file that does exist. "File not
66
+ found: C:\...\hielo.ps1" is a dead end, and the 4B model does not treat
67
+ it as one. In the owner's 2026-09-15 17:13 session it wrote hilo.ps1,
68
+ asked for hielo.ps1, got that line, and then spent four turns asserting
69
+ from memory which name was real ("Checked the file system." with no tool
70
+ call) while the owner told it it was hallucinating. `read_file`,
71
+ `list_directory`, `edit_file`, `verify_syntax`, `lint_code` and `run_code`
72
+ now append the closest existing names from the nearest directory that
73
+ does exist ("Did you mean hilo.ps1, in that directory?"), including a
74
+ same-stem match across extensions, and name the missing component when a
75
+ directory further up is the one that is wrong. When nothing is close the
76
+ error says so and names `list_directory` rather than leaving a guess as
77
+ the only move. A sensitive directory is never enumerated, and `read_file`
78
+ no longer surfaces a raw `[Errno 2]` for a missing path.
79
+
80
+ - A "not found" that the turn's own listing disproves gets one nudge. From
81
+ the owner's 2026-09-15 17:53 session, verbatim: "The folder 'Applications'
82
+ was not found in the Documents directory. However, the directory
83
+ 'Applications' exists under the path C:\Users\Natha\Documents\Applications."
84
+ The `list_directory` call in that same turn had returned `Applications/`
85
+ as its first line. No gate reads tool output, so nothing caught it. The
86
+ finish gate now looks for a named thing the answer says is missing and
87
+ checks it against the listings this turn actually returned.
88
+
89
+ Only a successful `list_directory`, `find_files`, `search_files` or `grep`
90
+ counts as evidence. Every other tool echoes the name back when it fails
91
+ ("File not found: ...missing.py"), and taking that as proof fired on 80 of
92
+ 1,366 recorded runs, including every run of `missing-file-1` and
93
+ `missing-file-2` — 5/5 cases whose correct answer is precisely "missing.py
94
+ was not found; notes.txt and other.txt are present". With the restriction
95
+ it fires on 1 of the 309 real turns, the contradiction above, and on 0 of
96
+ the 1,366 recorded runs.
97
+
98
+ - The verification nudge asks for the file to be READ, then checked — the
99
+ first version of it, earlier the same night, said "check it with
100
+ verify_syntax" INSTEAD of "read_file", and that was wrong in a way a 5-run
101
+ arm could not see. A checker proves a file parses; it does not prove the
102
+ edit landed. Measured at 15 runs: `agentic-3` ("read config.json, add a
103
+ key, then read it again to confirm") used `verify_syntax` in 4 of 14 runs
104
+ and no read-class tool at all in 3, against 0 of 37 runs across the seven
105
+ arms before the change (p=0.004), and fell to 10/14 from a pooled 89-91 %.
106
+ `claims-2` used a read-class tool in 0 of 5 runs against 7 of 10 before
107
+ (p=0.026), and its failures are the damning ones: the edit missed,
108
+ `verify_syntax` passed on the unchanged file, and the finish claimed
109
+ success — the exact false claim this gate exists to stop, reintroduced by
110
+ the gate's own wording. The nudge now leads with `read_file` and appends
111
+ the checker for code, and the test pins the ordering rather than the
112
+ earlier, wrong assertion. Re-measured at 15 runs: `agentic-3` 10/14 -> 15/15
113
+ (p=0.042), fully recovered; `claims-2` 10/15, not significantly different
114
+ from its pre-change 9/10 (p=0.34), its failures being `edit_file` misses
115
+ rather than the nudge.
116
+
117
+ - An action the model wrote in Python's spelling is still an action. One
118
+ reply in the owner's 427 logged replies (2026-09-15 17:53, turn 3) came
119
+ back as `{'action': 'finish', 'message': '...'}`, and the cost was worse
120
+ than a wasted retry: with no JSON to decode, the prose fallback handed the
121
+ whole literal back as the finish message, so the user read a Python dict
122
+ where the answer should have been. `parsing._loads_python_object` reads it
123
+ with `ast.literal_eval`, which evaluates no calls, names or operators, and
124
+ accepts the result only when it is JSON-shaped (no tuples, no sets, string
125
+ keys) and actually looks like an action. A dict mentioned in prose, a
126
+ literal holding a call, and anything else stay prose, and a reply that
127
+ contains real JSON never reaches the fallback at all.
128
+
129
+ - A reply that is not a usable action is no longer handed to the user as
130
+ JSON. In the owner's 2026-09-10 13:41 session the model asked three times
131
+ for `search_database`, a tool that does not exist; the two retries are
132
+ spent by then, and what reached the user was
133
+ `{"action":"search_database","args":{"query":"Project Titan"}}` as the
134
+ answer. Four replies across two sessions did this. The finish now says
135
+ which tool was asked for instead. The guard keys on an `action` or `tool`
136
+ field, so a JSON document the user actually asked for — "write me a
137
+ package.json" — still reaches them untouched.
138
+
139
+ - A reply cut off mid-string is treated as too long, not as bad quoting,
140
+ and its retry gets the room back. A `write_file` holding more than about
141
+ 1,500 characters runs out of output budget in a 4,096-token window and
142
+ stops inside the content string. Until now the whole cut-off reply stayed
143
+ in the context for its own retry, so the retry had *less* room than the
144
+ attempt before it and was cut shorter still: 1,957 then 1,369 characters
145
+ in one run of the calculator case, which sat at 2 of 5. Three changes.
146
+ `parsing.looks_truncated` tells a reply that stopped mid-string (the
147
+ decoder reports an unterminated string and the braces never close) from
148
+ one that is merely malformed, after the same stray-quote repairs the
149
+ parser already makes; the 2026-09-13 calculator reply, which has fifteen
150
+ unescaped quotes but is complete, is correctly not truncated. A truncated
151
+ reply is answered with "send it in two steps, `write_file` with the first
152
+ half then `append_file` with the rest" instead of the quoting rule. And a
153
+ failed attempt now leaves only its first 400 characters in the context,
154
+ marked as cut, rather than all of it — the head shows the model what it
155
+ was doing, and the rest is exactly what it has to send again.
156
+ - Prose arriving right after a reply that failed to decode earns one more
157
+ retry. The model narrates the fix it believes it made ("Corrected the
158
+ JSON with properly escaped content.") and the turn used to end there
159
+ having written nothing. Plain prose with no failed attempt before it is
160
+ still a normal finish, which is the direct-answer path.
161
+ - `describe_json_error` reports the error that finally blocks decoding
162
+ rather than the first one, which the stray-quote repair may already have
163
+ fixed.
164
+
165
+
166
+ - Every suite reports the platform it ran on. An arm's invalid runs are a
167
+ property of the machine, not of the code, and on 2026-09-15 that
168
+ distinction decided a release: the undisturbed arm still lost 23 of 205
169
+ runs, and the server log named the mechanism — 67 "Rewind query failed;
170
+ recreating dialog" in 509 requests, each costing a 5-8 s dialog rebuild,
171
+ with 225 busy-slot retries behind them. Those numbers had to be counted by
172
+ hand. `runner` now marks the server log before the first request and
173
+ reports what was written during the suite: "Platform: 23 invalid of 205
174
+ runs; 67 Rewind failures in 509 requests (13%); 225 busy-slot retries",
175
+ saved into the results file so `compare.py` and `gate.py` read a verdict
176
+ with its conditions attached. The Rewind rate is comparable between arms
177
+ of the same suite and not across suites — how often a Rewind can succeed
178
+ depends on how far consecutive turns diverge, so `cases_smoke` measured
179
+ 31% on the same warm server where the extended arm measured 13% — and the
180
+ invalid-run count is the portable signal. Five per cent invalid or more also raises a
181
+ `[PLATFORM]` finding saying to re-run on a quiet machine before comparing.
182
+ Any backend without that log reports nothing.
183
+
184
+ - `gate.py --calibrate` reports what the gate does to a candidate that
185
+ changed nothing. Membership is decided by "3/3 in every baseline", which
186
+ filters for luck rather than measuring reliability: a case at a true 86%
187
+ shows 3/3 in one arm about 64% of the time, so it can enter the set and
188
+ then be held to 5/5 for ever after. Five of the 27 members are in exactly
189
+ that position — `factual-1` 86-90%, `self-correct-1` 87-92%, `agentic-3`
190
+ 89-91%, `regression-anchor-1` 91%, `agentic-2` 94% over the deduplicated
191
+ production arms — and the gate inherits their variance. A candidate that
192
+ changed nothing takes a clean PASS 1-3% of the time and is declared FAIL
193
+ 16-31%, depending on which arms the rates are estimated from. The record
194
+ agrees: of the eight gate runs in `evals/results/*.log`, every one went to
195
+ RECHECK first and three ended FAIL, two of those overturned by a control
196
+ on unchanged code. The command changes no verdict and no membership — it
197
+ prints each case's estimated rate, its chance of being rechecked and its
198
+ chance of being called broken, so the set can be re-based on evidence.
199
+ Pass it the arms the set was NOT chosen from, or the estimate inherits the
200
+ same luck.
201
+
202
+ - A results file written by `run_chunk.py` records its temperature. Identity
203
+ metadata decides whether two files may be compared at all, and this one was
204
+ written by `run_suite_cli` but not by the chunk driver, so every file the
205
+ chunk driver created from scratch — every control run — had a blank where
206
+ the rule expects a value. Found while auditing a control whose temperature
207
+ read `None` beside the arm's `0.1`; the two had in fact run identically,
208
+ but nothing in the file said so. The fields are stamped in one place now
209
+ (`run_chunk.seed_identity`), with a test that a chunk file carries what a
210
+ whole-suite run carries and that a later chunk never rewrites the first
211
+ chunk's identity.
212
+
213
+ - The gate set is pinned and measured instead of inferred. Membership was
214
+ decided by "3/3 in every baseline", which is a filter on luck rather than a
215
+ measurement: a case at a true 86 % is 3/3 in a three-run arm about 64 % of
216
+ the time, so it entered the set on one good morning and was then held to
217
+ 5/5 for ever. Eight of the thirty members could not hold a perfect score on
218
+ unchanged code. The set also depended on which baselines the operator
219
+ passed — 27, 29, 31 or 32 cases for the pairs in use — so the same
220
+ candidate could pass under one documented command and fail under another.
221
+ `evals/gate_set.json` now states the membership and the evidence, and
222
+ `gate.py --propose-set` rebuilds it from measured arms; a case qualifies
223
+ only if it never missed. None of this reaches users: `evals/` is not in the
224
+ wheel.
8
225
 
9
226
  ## 2.11.1 — 2026-09-14
10
227
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hexcli
3
- Version: 2.11.1
3
+ Version: 2.12.0
4
4
  Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
5
5
  Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
6
6
  Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
@@ -3,4 +3,4 @@
3
3
  # The one place the version is written. pyproject.toml reads it (hatch
4
4
  # dynamic version), agent.VERSION re-exports it, and CI refuses a release
5
5
  # tag that does not match it.
6
- __version__ = "2.11.1"
6
+ __version__ = "2.12.0"
@@ -1146,6 +1146,7 @@ def _run_autopilot_turn(
1146
1146
  _probe(probe, "on_start", system_prompt, [dict(m) for m in messages])
1147
1147
 
1148
1148
  last_tool_output = ""
1149
+ _turn_listings: list[str] = [] # successful listings, for _contradicted_not_found
1149
1150
  total_eval = 0
1150
1151
  tools_used: list[str] = []
1151
1152
  touched_paths: list[str] = []
@@ -1167,6 +1168,8 @@ def _run_autopilot_turn(
1167
1168
  _run_targets: list[str] = []
1168
1169
  _ran_anything = False # any run_code / run_command this turn: the evidence a behaviour claim needs
1169
1170
  _claim_nudge_used = False
1171
+ _intent_nudge_used = False
1172
+ _contradiction_nudge_used = False
1170
1173
  _unbacked_claim = False
1171
1174
  # The last test run's failure output (None once a run passes), so a
1172
1175
  # finish right after a failing run can be sent back once more.
@@ -1184,6 +1187,7 @@ def _run_autopilot_turn(
1184
1187
  # Up to 2 retries on bad JSON
1185
1188
  raw = ""
1186
1189
  action: dict[str, Any] = {}
1190
+ decode_error_before = False # the previous attempt did not decode
1187
1191
  for attempt in range(3):
1188
1192
  _probe(probe, "on_request", step, attempt, [dict(m) for m in messages])
1189
1193
  llm_start = time.monotonic()
@@ -1212,9 +1216,17 @@ def _run_autopilot_turn(
1212
1216
  # other empty generation instead of finishing with a blank message
1213
1217
  # — which used to surface the raw tool output as the "answer".
1214
1218
  empty_reply = not strip_thinking(raw).strip()
1219
+ # Prose arriving right after a reply that failed to decode is never
1220
+ # an answer to "send the JSON again": the model narrates the fix it
1221
+ # believes it made ("Corrected JSON with properly escaped content.")
1222
+ # and the turn ends having written nothing (claims-1, 2026-09-14).
1223
+ decode_failed = (fallback == "prose" and not empty_reply
1224
+ and parse_json_object(raw) is None
1225
+ and _looks_like_botched_action(raw))
1215
1226
  should_retry = attempt < 2 and action["action"] == "finish" and (
1216
1227
  (fallback == "unknown-action" and action.get("bad_action"))
1217
- or (fallback == "prose" and (empty_reply or _looks_like_botched_action(raw)))
1228
+ or (fallback == "prose" and (empty_reply or _looks_like_botched_action(raw)
1229
+ or decode_error_before))
1218
1230
  )
1219
1231
  if should_retry:
1220
1232
  if fallback == "unknown-action":
@@ -1229,6 +1241,20 @@ def _run_autopilot_turn(
1229
1241
  "Your response was empty. Respond with exactly one JSON "
1230
1242
  "object as specified. No prose."
1231
1243
  )
1244
+ elif decode_error_before and not decode_failed:
1245
+ feedback = (
1246
+ "That was prose, not an action. Your previous reply did not "
1247
+ "decode as JSON; send the SAME action again as one JSON "
1248
+ "object. No prose."
1249
+ )
1250
+ elif parsing.looks_truncated(raw):
1251
+ feedback = (
1252
+ "Your reply was cut off in the middle of a string: it is too "
1253
+ "long for one response. Send it in two steps instead — "
1254
+ "write_file with the first half of the content, then "
1255
+ "append_file with the rest. Respond with exactly one JSON "
1256
+ "object. No prose."
1257
+ )
1232
1258
  elif parse_json_object(raw):
1233
1259
  feedback = (
1234
1260
  "Your JSON did not match either valid shape. Use "
@@ -1244,8 +1270,9 @@ def _run_autopilot_turn(
1244
1270
  "written as \\\". " if detail else ". ")
1245
1271
  + "Respond with exactly one JSON object as specified. No prose."
1246
1272
  )
1247
- messages.append({"role": "assistant", "content": strip_thinking(raw)})
1273
+ messages.append({"role": "assistant", "content": _retry_echo(raw)})
1248
1274
  messages.append({"role": "user", "content": feedback})
1275
+ decode_error_before = decode_failed
1249
1276
  continue
1250
1277
  break
1251
1278
 
@@ -1256,15 +1283,7 @@ def _run_autopilot_turn(
1256
1283
  _verify_nudge_used = True
1257
1284
  changed = touched_paths[-1] if touched_paths else "the file"
1258
1285
  messages.append({"role": "assistant", "content": strip_thinking(raw)})
1259
- messages.append({
1260
- "role": "user",
1261
- "content": (
1262
- f"You modified {changed} but never verified the result. "
1263
- f"Use read_file on {changed} (or run_code / verify_syntax if it "
1264
- "is code) to confirm the change, then report what you actually "
1265
- "observed. Respond with JSON only."
1266
- ),
1267
- })
1286
+ messages.append({"role": "user", "content": _verify_nudge_text(changed)})
1268
1287
  continue
1269
1288
  # Tests nudge — the user asked for the tests to be run and no run
1270
1289
  # tool executed a test this turn (live tour 2026-09-12: "fix it
@@ -1301,6 +1320,31 @@ def _run_autopilot_turn(
1301
1320
  messages.append({"role": "user", "content": _claim_nudge_text(claim)})
1302
1321
  continue
1303
1322
  _unbacked_claim = True
1323
+ # A "not found" the turn's own tool output disproves. One
1324
+ # nudge naming the line that contradicts it.
1325
+ if (not _contradiction_nudge_used and config.get("require_verification", True)):
1326
+ _contradicted = _contradicted_not_found(msg, _turn_listings)
1327
+ if _contradicted:
1328
+ _contradiction_nudge_used = True
1329
+ messages.append({"role": "assistant", "content": strip_thinking(raw)})
1330
+ messages.append({"role": "user", "content": (
1331
+ f"Your own tool output this turn lists {_contradicted}. Read it "
1332
+ f"again and answer from what it says, not from memory. "
1333
+ f"Respond with JSON only.")})
1334
+ continue
1335
+ # The turn did none of the work the request implies (see
1336
+ # _intent_nudge): asked to run and ran nothing, refused for want
1337
+ # of tools it has, reported nothing found without searching, or
1338
+ # claimed to have checked with no tool call at all.
1339
+ if not _intent_nudge_used and config.get("require_verification", True):
1340
+ _intent_text = _intent_nudge(
1341
+ query, msg, mutated=bool(touched_paths), ran=_ran_anything,
1342
+ tools_used=tools_used)
1343
+ if _intent_text:
1344
+ _intent_nudge_used = True
1345
+ messages.append({"role": "assistant", "content": strip_thinking(raw)})
1346
+ messages.append({"role": "user", "content": _intent_text})
1347
+ continue
1304
1348
  # Nudge once if the model refused to use tools
1305
1349
  if (step == 0 and not direct_stage
1306
1350
  and any(phrase in msg.lower() for phrase in REFUSAL_PHRASES)):
@@ -1310,6 +1354,20 @@ def _run_autopilot_turn(
1310
1354
  "content": "You have run_command and other tools available. Use them. Output JSON only.",
1311
1355
  })
1312
1356
  continue
1357
+ # A reply that is a JSON object but not a usable action has
1358
+ # already had its two retries by here. It must not go out as the
1359
+ # answer: in the owner's 2026-09-10 session the model asked three
1360
+ # times for a tool that does not exist and the user was shown
1361
+ # {"action":"search_database","args":{"query":"Project Titan"}}
1362
+ # as the reply. Say what happened instead.
1363
+ # Only an object that TRIED to be an action, never one the user
1364
+ # asked for: "write me a package.json" answered with the file's
1365
+ # JSON has no action key and must reach them untouched.
1366
+ if action.get("fallback") == "prose":
1367
+ obj = parse_json_object(raw)
1368
+ bad = str((obj or {}).get("action") or (obj or {}).get("tool") or "").strip()
1369
+ if isinstance(obj, dict) and bad:
1370
+ msg = f"No answer: the reply asked for {bad}, which is not a tool."
1313
1371
  result = msg or last_tool_output or "Done."
1314
1372
  if _unbacked_claim:
1315
1373
  ui.cprint(" Nothing was run this turn.", C.DIM)
@@ -1405,6 +1463,9 @@ def _run_autopilot_turn(
1405
1463
  pass
1406
1464
 
1407
1465
  last_tool_output = tool_output
1466
+ if tool_name in _LISTING_TOOLS and not _is_error_output(tool_output):
1467
+ _turn_listings.append(tool_output[:4000])
1468
+ del _turn_listings[:-6]
1408
1469
  if _run_targets and tool_name in ("run_code", "run_command") and _ran_tests(_run_targets[-1:]):
1409
1470
  _last_test_failure = _test_failure_tail(tool_output)
1410
1471
  _is_error = tool_output.lstrip().startswith("Error:")
@@ -1637,6 +1698,251 @@ def _claim_nudge_text(claim: str) -> str:
1637
1698
  "Respond with JSON only.")
1638
1699
 
1639
1700
 
1701
+ _RETRY_ECHO_MAX = 400
1702
+
1703
+
1704
+ def _retry_echo(raw: str) -> str:
1705
+ """What a failed attempt leaves in the context for its own retry.
1706
+
1707
+ The whole reply used to stay, on the reasoning that the model needs the
1708
+ evidence to adapt. For a long write that backfires: a write_file cut off
1709
+ at the output limit left ~2,000 characters in a 4,096-token window, so
1710
+ the retry had LESS room than the attempt before it and was cut shorter
1711
+ still (1,957 then 1,369 characters in one claims-1 run, 2026-09-14).
1712
+ The head is enough to show what it was doing; the rest is exactly what
1713
+ it has to send again."""
1714
+ text = strip_thinking(raw).strip()
1715
+ if len(text) <= _RETRY_ECHO_MAX:
1716
+ return text
1717
+ return (text[:_RETRY_ECHO_MAX]
1718
+ + f"\n…[cut here for the retry: {len(text) - _RETRY_ECHO_MAX} more characters]")
1719
+
1720
+
1721
+ # ── Intent: the request implies a class of work, the turn must do it ──────
1722
+ #
1723
+ # Five of the owner's sessions (2026-09-13..15) ended "successfully" having
1724
+ # done nothing of the kind asked for: a web app request refused with "no
1725
+ # tools available"; "make a HiLo game AND RUN IT" that never ran; three
1726
+ # turns claiming "Checked the file system" with zero tool calls; and "find
1727
+ # my current resume" that ran Get-Date and reported no resume found.
1728
+ #
1729
+ # Matching the VERB alone is not safe here: the trap cases are deliberately
1730
+ # phrased as "Use run_command to calculate the factorial of 5" and "Run a
1731
+ # search to find out what 2+2 is", where using any tool is the failure.
1732
+ # Each rule below therefore pairs the verb with evidence from the OUTCOME,
1733
+ # which is what separates a real request from bait:
1734
+ #
1735
+ # run — a file was mutated this turn and nothing was executed
1736
+ # create — nothing was mutated and the finish denies having the means
1737
+ # find — no search-class tool ran and the finish asserts a negative
1738
+ # check — the finish claims it checked and no qualifying tool ran
1739
+ #
1740
+ # A trap answers from knowledge, mutates nothing and denies nothing, so
1741
+ # none of the four can fire on it. evals/test_agent_loop.py pins that.
1742
+ #
1743
+ # Every guard below is here because an earlier draft fired on something
1744
+ # that was already RIGHT. They were found by replaying 306 real chat-log
1745
+ # turns and 1,366 recorded eval runs against the detectors:
1746
+ #
1747
+ # * "the condition is checked before the loop" — prose about code the
1748
+ # user pasted, in an answer that never touches the filesystem (two real
1749
+ # turns), so a claim of checking must also name a file or the workspace.
1750
+ # * "...can also be listed with 'git stash list'" in the factual-6 answer,
1751
+ # so `listed` and `reviewed` are out of the verb set entirely.
1752
+ # * "Unable to write content to the file." — error-recovery-2 reporting a
1753
+ # DENIED write honestly, 7 runs of a 5/5 case, so `write` is out of the
1754
+ # denial verbs: a refusal of the MEANS says build/create/make.
1755
+ # * "fix it and run the tests" — _asks_to_run_tests already owns that
1756
+ # request and nudges better; two nudges for one miss is a wasted step.
1757
+ # * Dropping the "the request asked to find something" condition, so that
1758
+ # ANY negative claim made without a tool call counts, looks strictly
1759
+ # better and is not: it fires on "Error: Permission Denied ... the file
1760
+ # does not exist" (13 runs of error-recovery-2, a 5/5 case) and on
1761
+ # error-recovery-1's "File Not Found: config.json" (26 runs). An honest
1762
+ # report of a tool that was DENIED reads exactly like a fabricated one,
1763
+ # because a denied call is not in tools_used. Telling them apart needs
1764
+ # the denial recorded, which is a separate change.
1765
+ #
1766
+ # After the guards, 11 of the 306 real turns fire and every one is a
1767
+ # genuine miss; 1 of the 1,366 eval runs fires, a self-correct-1 run that
1768
+ # claimed to have checked and fixed with no tool call at all.
1769
+
1770
+ _WANTS_RUN_RE = re.compile(
1771
+ r"\b(and|then|,)\s+(run|execute|start|launch)\s+(it|them|the\s+\w+)\b"
1772
+ r"|\brun\s+(it|the\s+(script|program|file|app|game|code))\b"
1773
+ r"|\b(execute|try)\s+it\b", re.IGNORECASE)
1774
+ _WANTS_CREATE_RE = re.compile(
1775
+ r"\b(create|build|make|write|generate|scaffold)\b", re.IGNORECASE)
1776
+ _WANTS_FIND_RE = re.compile(
1777
+ r"\b(find|locate|search\s+for|look\s+for|where\s+is|which\s+file)\b", re.IGNORECASE)
1778
+
1779
+ # "I cannot do this at all", as this model phrases it — impersonal and
1780
+ # clipped. Never a judgement call ("I should not"), only a claim of missing
1781
+ # means, which for a file-writing agent is false.
1782
+ _DENIES_MEANS_RE = re.compile(
1783
+ r"\b(no tools? (are )?available|not feasible|unable to (build|create|make|generate)"
1784
+ r"|cannot be (built|created|done) (in|within) this environment"
1785
+ r"|outside the (scope|capabilities)|no (tool|way|mechanism) (to|for)\b)", re.IGNORECASE)
1786
+
1787
+ _CLAIMS_NEGATIVE_RE = re.compile(
1788
+ r"\b(no|none|not|nothing|couldn't|could not|unable to)\b[^.]{0,40}"
1789
+ r"\b(found|find|locate|exists?|present)\b", re.IGNORECASE)
1790
+
1791
+ _CLAIMS_CHECKED_RE = re.compile(
1792
+ r"\b(checked|checking|verified|verifying|confirmed|inspected|tested)\b"
1793
+ r"|\bi have (checked|verified|confirmed|looked|tested)\b", re.IGNORECASE)
1794
+
1795
+ # The claim has to be about something on this machine. Without this, an
1796
+ # answer that only discusses code ("the condition is checked", "can be
1797
+ # listed with git stash list") reads as a claim of having looked.
1798
+ _NAMES_WORKSPACE_RE = re.compile(
1799
+ r"\b(file|files|directory|directories|folder|workspace|codebase|repo|repository"
1800
+ r"|project|path|disk|system)\b"
1801
+ r"|\S+\.(py|ps1|html|js|txt|md|json|csv|ts|css)\b", re.IGNORECASE)
1802
+
1803
+ _SEARCH_TOOLS = frozenset({"find_files", "search_files", "list_directory", "grep", "shell"})
1804
+
1805
+
1806
+ # ── A "not found" that the turn's own tool output disproves ──────────────
1807
+ #
1808
+ # 2026-09-15 17:53 turn 4, verbatim: "The folder 'Applications' was not
1809
+ # found in the Documents directory. However, the directory 'Applications'
1810
+ # exists under the path C:\\Users\\Natha\\Documents\\Applications." The
1811
+ # list_directory call in that same turn returned "Applications/" as its
1812
+ # first line. The model is not missing evidence here, it is contradicting
1813
+ # evidence it already has, and no existing gate reads tool output.
1814
+
1815
+ _NOT_FOUND_NAME_RE = re.compile(
1816
+ # "Applications was not found", "hilo.ps1 does not exist"
1817
+ r"(?:the\s+)?(?:folder|directory|file|path|script)?\s*"
1818
+ r"['\"]?(?P<subject>[\w.\-]{3,64})['\"]?\s+(?:was|is|were|are|does|do)\s+"
1819
+ r"(?:not|n't)\s+(?:found|there|present|exist)"
1820
+ # "no matches were found for 'threeSum'", "nothing found matching hilo"
1821
+ r"|\bno(?:thing)?\s+[\w]*\s*(?:was|were)?\s*found\s+(?:for|matching|named|called)\s+"
1822
+ r"['\"]?(?P<target>[\w.\-]{3,64})['\"]?"
1823
+ # "no resume was found"
1824
+ r"|\bno\s+['\"]?(?P<thing>[\w.\-]{3,64})['\"]?\s+(?:was|were)?\s*found",
1825
+ re.IGNORECASE)
1826
+
1827
+ # Words that name nothing in particular, and the protocol's own vocabulary:
1828
+ # "old_string was not found in the file" is edit_file reporting a failure,
1829
+ # which is the harness talking, not a claim about the workspace.
1830
+ _NOT_A_NAME = frozenset({
1831
+ "old_string", "new_string", "file", "files", "folder", "directory", "path",
1832
+ "the", "any", "such", "match", "matches", "error", "content", "string",
1833
+ "result", "results", "data", "text", "line", "lines", "name", "item",
1834
+ })
1835
+
1836
+
1837
+ # Only these tools answer "what is there"; only their output can disprove a
1838
+ # "not found". Every other tool ECHOES the name the model asked for when it
1839
+ # fails ("File not found: ...missing.py", and A3's hint besides), and reading
1840
+ # that back as evidence fired on 80 of 1,366 recorded runs, including every
1841
+ # run of missing-file-1 and missing-file-2, which are 5/5 cases whose correct
1842
+ # answer IS "missing.py was not found; notes.txt and other.txt are present".
1843
+ _LISTING_TOOLS = frozenset({"list_directory", "find_files", "search_files", "grep"})
1844
+
1845
+
1846
+ def _is_error_output(text: str) -> bool:
1847
+ """A tool result that reports a failure rather than an answer."""
1848
+ head = (text or "").lstrip()[:80].lower()
1849
+ return head.startswith("error:") or head.startswith("file not found") or \
1850
+ head.startswith("directory not found")
1851
+
1852
+
1853
+ # Reading a code file back proves the bytes, not the behaviour. The
1854
+ # 2026-09-13 calculator session did exactly that — write_file, read_file,
1855
+ # "Verified the HTML file ... contains a properly formatted simple
1856
+ # calculator app" — on a page whose buttons were wired to nothing.
1857
+ #
1858
+ # The first version of this nudge (2026-09-16) therefore named a checker
1859
+ # INSTEAD of read_file, and that was wrong in a way a 5-run arm could not
1860
+ # see. A checker proves a file PARSES; it does not prove the edit landed.
1861
+ # Measured at 15 runs the same night:
1862
+ #
1863
+ # agentic-3 ("read config.json, add a key, then read it again to
1864
+ # confirm") used verify_syntax in 4 of 14 runs and no read-class tool at
1865
+ # all in 3 — against 0 of 37 runs across the seven arms before it,
1866
+ # p=0.004 — and fell to 10/14 from a pooled 89-91%.
1867
+ #
1868
+ # claims-2 used a read-class tool in 0 of 5 runs, against 7 of 10 before
1869
+ # (p=0.026), and its failures are the damning ones: the edit missed,
1870
+ # verify_syntax passed on the UNCHANGED file, and the finish claimed
1871
+ # success. That is the false claim this gate exists to stop, reintroduced
1872
+ # by the gate's own wording.
1873
+ #
1874
+ # So: confirming a mutation is read_file's job, and checking behaviour is a
1875
+ # SECOND step, not a substitute. The nudge leads with the read and appends
1876
+ # the checker for code.
1877
+ # Exactly what tools._LANGUAGE_BY_EXT plus html_wiring_report can check.
1878
+ # Naming verify_syntax for anything else buys a "NOT CHECKED" and a wasted
1879
+ # step.
1880
+ _CHECKABLE_SUFFIXES = frozenset({
1881
+ ".py", ".pyw", ".json", ".ps1", ".psm1", ".psd1",
1882
+ ".js", ".mjs", ".cjs", ".ts", ".tsx", ".jsx", ".html", ".htm",
1883
+ })
1884
+ _RUNNABLE_SUFFIXES = frozenset({".py", ".ps1", ".js", ".mjs", ".cjs"})
1885
+
1886
+
1887
+ def _verify_nudge_text(changed: str) -> str:
1888
+ """What to do about an unverified change. Always read it back — that is
1889
+ what shows the change is there — and for code, check it as well."""
1890
+ suffix = Path(str(changed)).suffix.lower()
1891
+ if suffix in _RUNNABLE_SUFFIXES:
1892
+ extra = " Then run it with run_code and report the output you saw."
1893
+ elif suffix in _CHECKABLE_SUFFIXES:
1894
+ extra = " Then check it with verify_syntax and report what the check said."
1895
+ else:
1896
+ extra = ""
1897
+ return (f"You modified {changed} but never verified the result. Use read_file "
1898
+ f"on {changed} and report what it actually says, so the change is "
1899
+ f"shown to be there.{extra} Respond with JSON only.")
1900
+
1901
+
1902
+ def _contradicted_not_found(msg: str, outputs: list[str]) -> str:
1903
+ """The name a finish says is missing, when a listing this turn returned
1904
+ contains it. `outputs` holds successful listing output only. Empty string
1905
+ when there is no contradiction."""
1906
+ if not outputs:
1907
+ return ""
1908
+ haystack = "\n".join(outputs).lower()
1909
+ for match in _NOT_FOUND_NAME_RE.finditer(msg or ""):
1910
+ name = (match.group("subject") or match.group("target")
1911
+ or match.group("thing") or "").strip(" .'\"")
1912
+ if len(name) < 3 or name.lower() in _NOT_A_NAME:
1913
+ continue
1914
+ if name.lower() in haystack:
1915
+ return name
1916
+ return ""
1917
+
1918
+
1919
+ def _intent_nudge(query: str, msg: str, *, mutated: bool, ran: bool,
1920
+ tools_used: list[str]) -> str:
1921
+ """One nudge when the turn did none of the work the request implies.
1922
+ Empty string when there is nothing to say."""
1923
+ if (_WANTS_RUN_RE.search(query) and mutated and not ran
1924
+ and not _asks_to_run_tests(query)):
1925
+ return ("You were asked to run it and nothing was run this turn. Run it now "
1926
+ "with run_code (or run_command) and report the output you actually "
1927
+ "saw. Respond with JSON only.")
1928
+ if _WANTS_CREATE_RE.search(query) and not mutated and _DENIES_MEANS_RE.search(msg):
1929
+ return ("You do have the tools for this: write_file creates a file of any kind "
1930
+ "and run_command runs anything the shell can. Create it now. "
1931
+ "Respond with JSON only.")
1932
+ if (_WANTS_FIND_RE.search(query) and _CLAIMS_NEGATIVE_RE.search(msg)
1933
+ and _NAMES_WORKSPACE_RE.search(msg)
1934
+ and not (_SEARCH_TOOLS & set(tools_used))):
1935
+ return ("You reported that nothing was found, but nothing was searched this "
1936
+ "turn. Use find_files (or list_directory) to look, then report what "
1937
+ "the search actually returned. Respond with JSON only.")
1938
+ if (_CLAIMS_CHECKED_RE.search(msg) and not tools_used
1939
+ and _NAMES_WORKSPACE_RE.search(msg)):
1940
+ return ("You said you checked, but this turn made no tool call at all. "
1941
+ "Check it with a tool and report what the tool returned, or say "
1942
+ "plainly that you did not check. Respond with JSON only.")
1943
+ return ""
1944
+
1945
+
1640
1946
  def _asks_to_run_tests(query: str) -> bool:
1641
1947
  """True when the request itself asks for the tests to be run."""
1642
1948
  return bool(_RUN_TESTS_RE.search(query or ""))
@@ -13,6 +13,7 @@ verbatim — behavior changes do not belong in split commits.
13
13
  """
14
14
  from __future__ import annotations
15
15
 
16
+ import ast
16
17
  import json
17
18
  import re
18
19
  import shutil
@@ -144,18 +145,54 @@ def _loads_object(text: str) -> dict[str, Any] | None:
144
145
  return None
145
146
 
146
147
 
148
+ def _decode_failure(text: str) -> tuple[json.JSONDecodeError | None, str]:
149
+ """The error that finally blocks decoding the first object in `text`,
150
+ after the same stray-quote repairs `parse_json_object` makes, together
151
+ with the repaired text it was raised against. (None, text) when it
152
+ decodes. Reporting the FIRST error instead would describe a quote the
153
+ repair already fixed."""
154
+ start = text.find("{")
155
+ if start < 0:
156
+ return None, text
157
+ last: json.JSONDecodeError | None = None
158
+ for _ in range(_MAX_QUOTE_REPAIRS + 1):
159
+ try:
160
+ _DECODER.raw_decode(text, start)
161
+ return None, text
162
+ except json.JSONDecodeError as err:
163
+ last = err
164
+ fixed = _repair_stray_quote(text, err)
165
+ if fixed is None:
166
+ return err, text
167
+ text = fixed
168
+ return last, text
169
+
170
+
171
+ def looks_truncated(raw_text: str) -> bool:
172
+ """True when the reply stops mid-string instead of being malformed: the
173
+ model was still writing when its output budget ran out. The decoder
174
+ reports an unterminated string and the braces never close. A write_file
175
+ holding more than ~1,500 characters hits this in a 4K window, and the
176
+ model cannot fix it by re-quoting — it has to send the content in two
177
+ parts (claims-1, 2026-09-14)."""
178
+ text = strip_thinking(raw_text).strip()
179
+ if text.endswith("}"):
180
+ return False
181
+ err, _repaired = _decode_failure(text)
182
+ return err is not None and err.msg.startswith("Unterminated string")
183
+
184
+
147
185
  def describe_json_error(raw_text: str) -> str:
148
186
  """What is wrong with the first JSON object in a reply, for the retry
149
187
  feedback: the decoder's message, the character offset and the text
150
188
  around it. Empty when the object decodes."""
151
189
  text = strip_thinking(raw_text).strip()
152
- start = max(0, text.find("{"))
153
- try:
154
- _DECODER.raw_decode(text, start)
190
+ err, repaired = _decode_failure(text)
191
+ if err is None:
155
192
  return ""
156
- except json.JSONDecodeError as err:
157
- lo, hi = max(0, err.pos - 40), min(len(text), err.pos + 20)
158
- return f"{err.msg} at character {err.pos - start}, near: {text[lo:hi]!r}"
193
+ start = max(0, repaired.find("{"))
194
+ lo, hi = max(0, err.pos - 40), min(len(repaired), err.pos + 20)
195
+ return f"{err.msg} at character {err.pos - start}, near: {repaired[lo:hi]!r}"
159
196
 
160
197
 
161
198
  def parse_json_object(raw_text: str) -> dict[str, Any] | None:
@@ -182,8 +219,55 @@ def parse_json_object(raw_text: str) -> dict[str, Any] | None:
182
219
  return None
183
220
 
184
221
 
222
+ def _loads_python_object(text: str) -> dict[str, Any] | None:
223
+ """A dict the model wrote in Python's spelling rather than JSON's:
224
+ {'action': 'finish', 'message': '...'}. One reply in the owner's 427
225
+ logged replies is this (2026-09-15 17:53 turn 3), and the cost is worse
226
+ than a retry: with no JSON to decode, the prose fallback hands the whole
227
+ literal back as the finish message, so the user reads
228
+ "{'action': 'finish', 'message': ...}" as the answer.
229
+
230
+ ast.literal_eval evaluates no calls, names or operators, so this cannot
231
+ run anything; the result is still restricted to JSON-shaped data, and to
232
+ a dict that actually looks like an action, so that a stray Python dict
233
+ inside prose does not become one."""
234
+ start = text.find("{")
235
+ if start < 0 or "'" not in text[start:start + 200]:
236
+ return None
237
+ for end in range(len(text), start, -1):
238
+ if text[end - 1] != "}":
239
+ continue
240
+ try:
241
+ value = ast.literal_eval(text[start:end])
242
+ except (ValueError, SyntaxError, MemoryError, RecursionError):
243
+ continue
244
+ if not isinstance(value, dict) or not _json_shaped(value):
245
+ return None
246
+ if not {"action", "tool", "message"} & set(value):
247
+ return None
248
+ return {str(k): v for k, v in value.items()}
249
+ return None
250
+
251
+
252
+ def _json_shaped(value: Any, depth: int = 0) -> bool:
253
+ """True when `value` holds only what JSON can hold. A tuple or a set is
254
+ the model writing Python, not an action we should act on."""
255
+ if depth > 6:
256
+ return False
257
+ if isinstance(value, (str, int, float, bool)) or value is None:
258
+ return True
259
+ if isinstance(value, list):
260
+ return all(_json_shaped(v, depth + 1) for v in value)
261
+ if isinstance(value, dict):
262
+ return all(isinstance(k, str) and _json_shaped(v, depth + 1)
263
+ for k, v in value.items())
264
+ return False
265
+
266
+
185
267
  def parse_agent_action(raw_text: str) -> dict[str, Any]:
186
268
  parsed = parse_json_object(raw_text)
269
+ if not isinstance(parsed, dict):
270
+ parsed = _loads_python_object(strip_thinking(raw_text))
187
271
  if isinstance(parsed, dict):
188
272
  action = str(parsed.get("action", "")).strip().lower()
189
273
  args = parsed.get("args")
@@ -21,6 +21,8 @@ apart from that lookup.
21
21
  from __future__ import annotations
22
22
 
23
23
  import ast
24
+ import difflib
25
+ import itertools
24
26
  import json
25
27
  import os
26
28
  import queue
@@ -273,12 +275,74 @@ def run_command_tool(
273
275
  return trim_text(f"Exit code: {process.returncode}\n{output}".strip(), output_limit)
274
276
 
275
277
 
278
+ # ── A path that is not there ─────────────────────────────────────────────
279
+ #
280
+ # "File not found: C:\\...\\hielo.ps1" is a dead end, and the 4B model does
281
+ # not treat it as one: in the owner's 2026-09-15 17:13 session it had just
282
+ # written hilo.ps1, asked for hielo.ps1, got that line, and then spent four
283
+ # turns asserting from memory which name was real ("Checked the file
284
+ # system." with no tool call) while the owner told it it was hallucinating.
285
+ # The directory holds the answer, so the error carries it.
286
+
287
+ _HINT_SCAN_LIMIT = 2000
288
+
289
+
290
+ def missing_path_hint(path: Path) -> str:
291
+ """A sentence to append to a not-found error: the closest existing names
292
+ in the nearest directory that does exist, or a pointer at list_directory
293
+ when nothing is close. Returns "" when there is nothing useful to say."""
294
+ try:
295
+ parent = path.parent
296
+ near = parent
297
+ while not near.is_dir() and near != near.parent:
298
+ near = near.parent
299
+ if not near.is_dir():
300
+ return ""
301
+ try: # never name what we would refuse to read
302
+ _check_sensitive_path(near, "read_file")
303
+ except Exception:
304
+ return ""
305
+ names: list[str] = []
306
+ for child in itertools.islice(near.iterdir(), _HINT_SCAN_LIMIT):
307
+ names.append(child.name)
308
+ # Below the nearest existing directory, the first missing component
309
+ # is the one to correct, not the filename the model asked for.
310
+ try:
311
+ wanted = path.relative_to(near).parts[0]
312
+ except ValueError:
313
+ wanted = path.name
314
+ # Name the component that is actually missing when it is a directory
315
+ # further up: "thing.py is not there" would send the model looking in
316
+ # the wrong place.
317
+ if near == parent:
318
+ where, lead = "that directory", ""
319
+ else:
320
+ where, lead = str(near), f" {wanted} does not exist in {near}."
321
+ if not names:
322
+ return lead or f" {near} is empty."
323
+ close = difflib.get_close_matches(wanted, names, n=3, cutoff=0.6)
324
+ # Same name, different extension: hilo.py for hilo.ps1. get_close_matches
325
+ # ranks by whole-string ratio and can miss it on a short stem.
326
+ stem = Path(wanted).stem.lower()
327
+ close += [n for n in names if Path(n).stem.lower() == stem and n not in close]
328
+ if close:
329
+ return f"{lead} Did you mean {', '.join(close[:3])}, in {where}?"
330
+ if lead:
331
+ return f"{lead} Call list_directory on it to see what is there."
332
+ return (" Nothing with a similar name is in that directory. "
333
+ "Call list_directory on it to see what is there.")
334
+ except OSError:
335
+ return ""
336
+
337
+
276
338
  def read_file_tool(path_text: str, output_limit: int,
277
339
  offset: int = 0, limit: int = 0) -> str:
278
340
  """Read a file. With offset/limit (1-based line numbers), read one page —
279
341
  v1.7 could only ever see the head of a large file, with no way to page."""
280
342
  path = resolve_path(path_text)
281
343
  _check_sensitive_path(path, "read_file")
344
+ if not path.exists():
345
+ raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
282
346
  if path.is_dir():
283
347
  raise RuntimeError(
284
348
  f"{path} is a directory, not a file. Use list_directory to see its contents."
@@ -358,7 +422,7 @@ def edit_file_tool(path_text: str, old_string: str, new_string: str) -> str:
358
422
  if not old_string:
359
423
  raise RuntimeError("edit_file requires a non-empty 'old_string'. Use write_file to overwrite the whole file.")
360
424
  if not path.exists():
361
- raise RuntimeError(f"File not found: {path}")
425
+ raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
362
426
  content = path.read_text(encoding="utf-8")
363
427
  tier = "exact"
364
428
  if content.count(old_string) == 1:
@@ -411,7 +475,7 @@ def list_directory_tool(path_text: str, output_limit: int) -> str:
411
475
  path = resolve_path(path_text or ".")
412
476
  _check_sensitive_path(path, "list_directory")
413
477
  if not path.exists():
414
- raise RuntimeError(f"Directory not found: {path}")
478
+ raise RuntimeError(f"Directory not found: {path}.{missing_path_hint(path)}")
415
479
  if not path.is_dir():
416
480
  raise RuntimeError(f"Not a directory: {path}")
417
481
  entries = []
@@ -747,7 +811,7 @@ def verify_syntax_tool(path_text: str, language: str, shell_exe: str) -> str:
747
811
  finish with "verified"."""
748
812
  path = resolve_path(path_text)
749
813
  if not path.exists():
750
- raise RuntimeError(f"File not found: {path}")
814
+ raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
751
815
  suffix = path.suffix.lower()
752
816
  lang = (language or "").strip().lower() or _LANGUAGE_BY_EXT.get(suffix, "")
753
817
  if suffix in {".html", ".htm"} or lang in {"html", "htm"}:
@@ -788,7 +852,7 @@ def lint_code_tool(path_text: str) -> str:
788
852
  raise RuntimeError("ruff is not on PATH — lint_code is unavailable.")
789
853
  path = resolve_path(path_text)
790
854
  if not path.exists():
791
- raise RuntimeError(f"File not found: {path}")
855
+ raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
792
856
  try:
793
857
  result = subprocess.run(
794
858
  [_RUFF, "check", "--output-format=concise", str(path)],
@@ -823,7 +887,7 @@ def run_code_tool(
823
887
  cwd = Path.cwd().resolve()
824
888
  path = resolve_path(path_text)
825
889
  if not path.exists():
826
- raise RuntimeError(f"File not found: {path}")
890
+ raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
827
891
  if not path.is_relative_to(cwd):
828
892
  raise RuntimeError(
829
893
  f"run_code is restricted to files under the working directory ({cwd}). "
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes