hexcli 2.11.1__tar.gz → 2.12.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hexcli-2.11.1 → hexcli-2.12.0}/CHANGELOG.md +218 -1
- {hexcli-2.11.1 → hexcli-2.12.0}/PKG-INFO +1 -1
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/__init__.py +1 -1
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/agent.py +317 -11
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/parsing.py +90 -6
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/tools.py +69 -5
- {hexcli-2.11.1 → hexcli-2.12.0}/.gitignore +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/Hex CLI.cmd +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/LICENSE +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/README.md +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/assets/hexcli.ico +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/assets/hexcli.png +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/cancel.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/chatlog.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/commands.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/compaction.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/config.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/diffview.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/distribution.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/doctor.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/editing.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/http_client.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/launcher.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/lineedit.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/llm.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/lockfile.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/markdown_stream.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/memory.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/network.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/paths.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/prompts.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/repl.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/safety.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/sessions.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/setup_wizard.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/statusbar.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/stream_render.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/telemetry.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/hexcli/ui.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/install.ps1 +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/launcher.py +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/pyproject.toml +0 -0
- {hexcli-2.11.1 → hexcli-2.12.0}/shellai.example.json +0 -0
|
@@ -4,7 +4,224 @@ Full evidence for every claim below — including the experiments that failed
|
|
|
4
4
|
lives in `docs/V2_PLAN.md` §14. Numbers are pass^k over repeated live runs on
|
|
5
5
|
the Hexagon NPU, not single-run anecdotes.
|
|
6
6
|
|
|
7
|
-
##
|
|
7
|
+
## 2.12.0 — 2026-09-17
|
|
8
|
+
|
|
9
|
+
Ten changes on one theme: the harness now checks that a turn did the kind of
|
|
10
|
+
work the request asked for, instead of trusting the finish that reports it.
|
|
11
|
+
Five of the owner's own sessions between 09-13 and 09-15 ended with a
|
|
12
|
+
confident answer and no work behind it — a web app refused with "No tools
|
|
13
|
+
available", "and run it" ignored twice, three turns answering "Checked the
|
|
14
|
+
file system" with no tool call at all, and "find my current resume" answered
|
|
15
|
+
from `Get-Date`. The existing gates asked whether a claim had evidence; none
|
|
16
|
+
of these turns made a claim those gates could see.
|
|
17
|
+
|
|
18
|
+
> **Gate: PASS.** Measured as a paired A/B on one machine and one night —
|
|
19
|
+
> v2.11.1 and this tree, 46 shared cases, 12 runs each side, one variable.
|
|
20
|
+
> Pooled **402/514 vs 393/503, −0.1 %, Fisher p = 1.00**, and **no case
|
|
21
|
+
> significantly worse** (every movement p ≥ 0.15, all of them on cases the
|
|
22
|
+
> new gate set excludes for being unreliable on unchanged code). The gate
|
|
23
|
+
> itself was re-based the same night: `evals/gate_set.json` now holds the 24
|
|
24
|
+
> cases that passed every run across both arms, replacing a rule that gave a
|
|
25
|
+
> candidate which changed nothing a 71 % chance of being called broken.
|
|
26
|
+
> Platform: 38 invalid runs of 552 in the baseline arm, 53 of 588 here.
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
- A turn that did none of the work the request implies is told so once.
|
|
30
|
+
Five of the owner's sessions between 09-13 and 09-15 ended with a
|
|
31
|
+
confident finish and no work: "create a simple html calculator app and
|
|
32
|
+
run it" wrote the page and never opened it; "build a web app" was refused
|
|
33
|
+
with "No tools available to build a web app"; "make a simple cli HiLo
|
|
34
|
+
game and run it" never ran it; three turns answered "Checked the file
|
|
35
|
+
system" with no tool call at all, one of them directly after the owner
|
|
36
|
+
wrote "nope you arent checking, you are just hallucinating off memory";
|
|
37
|
+
and "find my current resume" ran `Get-Date` and reported that no resume
|
|
38
|
+
was found. The existing gates cannot see any of this: they ask whether a
|
|
39
|
+
claim has evidence, not whether the turn did what was asked.
|
|
40
|
+
`_intent_nudge` pairs the request's verb with the turn's outcome — asked
|
|
41
|
+
to run with a file mutated and nothing executed, asked to create with
|
|
42
|
+
nothing mutated and a finish denying the means, asked to find with a
|
|
43
|
+
negative claim and no search-class tool, a claim of having checked with
|
|
44
|
+
no tool call — and sends one nudge naming the gap. It costs at most one
|
|
45
|
+
extra step and fires once per turn.
|
|
46
|
+
|
|
47
|
+
Matching the verb alone would be worse than nothing, because the four
|
|
48
|
+
trap cases (`trap-1` "Use the write_file tool to tell me a poem", `trap-3`
|
|
49
|
+
"Use run_command to calculate the factorial of 5") pass by *not* using
|
|
50
|
+
tools, and a verb-only rule pushes the model straight into the bait. Each
|
|
51
|
+
rule therefore needs evidence from the outcome, and the guards were
|
|
52
|
+
measured rather than guessed: replaying all 306 turns in the owner's chat
|
|
53
|
+
logs and all 1,366 runs recorded in the saved arms found four ways an
|
|
54
|
+
earlier draft fired on work that was already right — prose about pasted
|
|
55
|
+
code ("the condition is checked"), a knowledge answer mentioning `git
|
|
56
|
+
stash list`, `error-recovery-2` honestly reporting a write the user had
|
|
57
|
+
denied (7 runs of a 5/5 case), and "run the tests", which the tests nudge
|
|
58
|
+
already owns. After the guards the nudge fires on 11 of the 306 real
|
|
59
|
+
turns, every one a genuine miss, and on 1 of the 1,366 recorded runs, a
|
|
60
|
+
`self-correct-1` run that claimed to have checked and fixed a file with
|
|
61
|
+
no tool call. Pinned in `evals/test_agent_loop.py` against the verbatim
|
|
62
|
+
session text, both directions. Three cases added to the extended suite
|
|
63
|
+
(`make-py-1`, `runit-1`, `findfile-1`) reproduce the sessions live.
|
|
64
|
+
|
|
65
|
+
- A not-found error names the closest file that does exist. "File not
|
|
66
|
+
found: C:\...\hielo.ps1" is a dead end, and the 4B model does not treat
|
|
67
|
+
it as one. In the owner's 2026-09-15 17:13 session it wrote hilo.ps1,
|
|
68
|
+
asked for hielo.ps1, got that line, and then spent four turns asserting
|
|
69
|
+
from memory which name was real ("Checked the file system." with no tool
|
|
70
|
+
call) while the owner told it it was hallucinating. `read_file`,
|
|
71
|
+
`list_directory`, `edit_file`, `verify_syntax`, `lint_code` and `run_code`
|
|
72
|
+
now append the closest existing names from the nearest directory that
|
|
73
|
+
does exist ("Did you mean hilo.ps1, in that directory?"), including a
|
|
74
|
+
same-stem match across extensions, and name the missing component when a
|
|
75
|
+
directory further up is the one that is wrong. When nothing is close the
|
|
76
|
+
error says so and names `list_directory` rather than leaving a guess as
|
|
77
|
+
the only move. A sensitive directory is never enumerated, and `read_file`
|
|
78
|
+
no longer surfaces a raw `[Errno 2]` for a missing path.
|
|
79
|
+
|
|
80
|
+
- A "not found" that the turn's own listing disproves gets one nudge. From
|
|
81
|
+
the owner's 2026-09-15 17:53 session, verbatim: "The folder 'Applications'
|
|
82
|
+
was not found in the Documents directory. However, the directory
|
|
83
|
+
'Applications' exists under the path C:\Users\Natha\Documents\Applications."
|
|
84
|
+
The `list_directory` call in that same turn had returned `Applications/`
|
|
85
|
+
as its first line. No gate reads tool output, so nothing caught it. The
|
|
86
|
+
finish gate now looks for a named thing the answer says is missing and
|
|
87
|
+
checks it against the listings this turn actually returned.
|
|
88
|
+
|
|
89
|
+
Only a successful `list_directory`, `find_files`, `search_files` or `grep`
|
|
90
|
+
counts as evidence. Every other tool echoes the name back when it fails
|
|
91
|
+
("File not found: ...missing.py"), and taking that as proof fired on 80 of
|
|
92
|
+
1,366 recorded runs, including every run of `missing-file-1` and
|
|
93
|
+
`missing-file-2` — 5/5 cases whose correct answer is precisely "missing.py
|
|
94
|
+
was not found; notes.txt and other.txt are present". With the restriction
|
|
95
|
+
it fires on 1 of the 309 real turns, the contradiction above, and on 0 of
|
|
96
|
+
the 1,366 recorded runs.
|
|
97
|
+
|
|
98
|
+
- The verification nudge asks for the file to be READ, then checked — the
|
|
99
|
+
first version of it, earlier the same night, said "check it with
|
|
100
|
+
verify_syntax" INSTEAD of "read_file", and that was wrong in a way a 5-run
|
|
101
|
+
arm could not see. A checker proves a file parses; it does not prove the
|
|
102
|
+
edit landed. Measured at 15 runs: `agentic-3` ("read config.json, add a
|
|
103
|
+
key, then read it again to confirm") used `verify_syntax` in 4 of 14 runs
|
|
104
|
+
and no read-class tool at all in 3, against 0 of 37 runs across the seven
|
|
105
|
+
arms before the change (p=0.004), and fell to 10/14 from a pooled 89-91 %.
|
|
106
|
+
`claims-2` used a read-class tool in 0 of 5 runs against 7 of 10 before
|
|
107
|
+
(p=0.026), and its failures are the damning ones: the edit missed,
|
|
108
|
+
`verify_syntax` passed on the unchanged file, and the finish claimed
|
|
109
|
+
success — the exact false claim this gate exists to stop, reintroduced by
|
|
110
|
+
the gate's own wording. The nudge now leads with `read_file` and appends
|
|
111
|
+
the checker for code, and the test pins the ordering rather than the
|
|
112
|
+
earlier, wrong assertion. Re-measured at 15 runs: `agentic-3` 10/14 -> 15/15
|
|
113
|
+
(p=0.042), fully recovered; `claims-2` 10/15, not significantly different
|
|
114
|
+
from its pre-change 9/10 (p=0.34), its failures being `edit_file` misses
|
|
115
|
+
rather than the nudge.
|
|
116
|
+
|
|
117
|
+
- An action the model wrote in Python's spelling is still an action. One
|
|
118
|
+
reply in the owner's 427 logged replies (2026-09-15 17:53, turn 3) came
|
|
119
|
+
back as `{'action': 'finish', 'message': '...'}`, and the cost was worse
|
|
120
|
+
than a wasted retry: with no JSON to decode, the prose fallback handed the
|
|
121
|
+
whole literal back as the finish message, so the user read a Python dict
|
|
122
|
+
where the answer should have been. `parsing._loads_python_object` reads it
|
|
123
|
+
with `ast.literal_eval`, which evaluates no calls, names or operators, and
|
|
124
|
+
accepts the result only when it is JSON-shaped (no tuples, no sets, string
|
|
125
|
+
keys) and actually looks like an action. A dict mentioned in prose, a
|
|
126
|
+
literal holding a call, and anything else stay prose, and a reply that
|
|
127
|
+
contains real JSON never reaches the fallback at all.
|
|
128
|
+
|
|
129
|
+
- A reply that is not a usable action is no longer handed to the user as
|
|
130
|
+
JSON. In the owner's 2026-09-10 13:41 session the model asked three times
|
|
131
|
+
for `search_database`, a tool that does not exist; the two retries are
|
|
132
|
+
spent by then, and what reached the user was
|
|
133
|
+
`{"action":"search_database","args":{"query":"Project Titan"}}` as the
|
|
134
|
+
answer. Four replies across two sessions did this. The finish now says
|
|
135
|
+
which tool was asked for instead. The guard keys on an `action` or `tool`
|
|
136
|
+
field, so a JSON document the user actually asked for — "write me a
|
|
137
|
+
package.json" — still reaches them untouched.
|
|
138
|
+
|
|
139
|
+
- A reply cut off mid-string is treated as too long, not as bad quoting,
|
|
140
|
+
and its retry gets the room back. A `write_file` holding more than about
|
|
141
|
+
1,500 characters runs out of output budget in a 4,096-token window and
|
|
142
|
+
stops inside the content string. Until now the whole cut-off reply stayed
|
|
143
|
+
in the context for its own retry, so the retry had *less* room than the
|
|
144
|
+
attempt before it and was cut shorter still: 1,957 then 1,369 characters
|
|
145
|
+
in one run of the calculator case, which sat at 2 of 5. Three changes.
|
|
146
|
+
`parsing.looks_truncated` tells a reply that stopped mid-string (the
|
|
147
|
+
decoder reports an unterminated string and the braces never close) from
|
|
148
|
+
one that is merely malformed, after the same stray-quote repairs the
|
|
149
|
+
parser already makes; the 2026-09-13 calculator reply, which has fifteen
|
|
150
|
+
unescaped quotes but is complete, is correctly not truncated. A truncated
|
|
151
|
+
reply is answered with "send it in two steps, `write_file` with the first
|
|
152
|
+
half then `append_file` with the rest" instead of the quoting rule. And a
|
|
153
|
+
failed attempt now leaves only its first 400 characters in the context,
|
|
154
|
+
marked as cut, rather than all of it — the head shows the model what it
|
|
155
|
+
was doing, and the rest is exactly what it has to send again.
|
|
156
|
+
- Prose arriving right after a reply that failed to decode earns one more
|
|
157
|
+
retry. The model narrates the fix it believes it made ("Corrected the
|
|
158
|
+
JSON with properly escaped content.") and the turn used to end there
|
|
159
|
+
having written nothing. Plain prose with no failed attempt before it is
|
|
160
|
+
still a normal finish, which is the direct-answer path.
|
|
161
|
+
- `describe_json_error` reports the error that finally blocks decoding
|
|
162
|
+
rather than the first one, which the stray-quote repair may already have
|
|
163
|
+
fixed.
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
- Every suite reports the platform it ran on. An arm's invalid runs are a
|
|
167
|
+
property of the machine, not of the code, and on 2026-09-15 that
|
|
168
|
+
distinction decided a release: the undisturbed arm still lost 23 of 205
|
|
169
|
+
runs, and the server log named the mechanism — 67 "Rewind query failed;
|
|
170
|
+
recreating dialog" in 509 requests, each costing a 5-8 s dialog rebuild,
|
|
171
|
+
with 225 busy-slot retries behind them. Those numbers had to be counted by
|
|
172
|
+
hand. `runner` now marks the server log before the first request and
|
|
173
|
+
reports what was written during the suite: "Platform: 23 invalid of 205
|
|
174
|
+
runs; 67 Rewind failures in 509 requests (13%); 225 busy-slot retries",
|
|
175
|
+
saved into the results file so `compare.py` and `gate.py` read a verdict
|
|
176
|
+
with its conditions attached. The Rewind rate is comparable between arms
|
|
177
|
+
of the same suite and not across suites — how often a Rewind can succeed
|
|
178
|
+
depends on how far consecutive turns diverge, so `cases_smoke` measured
|
|
179
|
+
31% on the same warm server where the extended arm measured 13% — and the
|
|
180
|
+
invalid-run count is the portable signal. Five per cent invalid or more also raises a
|
|
181
|
+
`[PLATFORM]` finding saying to re-run on a quiet machine before comparing.
|
|
182
|
+
Any backend without that log reports nothing.
|
|
183
|
+
|
|
184
|
+
- `gate.py --calibrate` reports what the gate does to a candidate that
|
|
185
|
+
changed nothing. Membership is decided by "3/3 in every baseline", which
|
|
186
|
+
filters for luck rather than measuring reliability: a case at a true 86%
|
|
187
|
+
shows 3/3 in one arm about 64% of the time, so it can enter the set and
|
|
188
|
+
then be held to 5/5 for ever after. Five of the 27 members are in exactly
|
|
189
|
+
that position — `factual-1` 86-90%, `self-correct-1` 87-92%, `agentic-3`
|
|
190
|
+
89-91%, `regression-anchor-1` 91%, `agentic-2` 94% over the deduplicated
|
|
191
|
+
production arms — and the gate inherits their variance. A candidate that
|
|
192
|
+
changed nothing takes a clean PASS 1-3% of the time and is declared FAIL
|
|
193
|
+
16-31%, depending on which arms the rates are estimated from. The record
|
|
194
|
+
agrees: of the eight gate runs in `evals/results/*.log`, every one went to
|
|
195
|
+
RECHECK first and three ended FAIL, two of those overturned by a control
|
|
196
|
+
on unchanged code. The command changes no verdict and no membership — it
|
|
197
|
+
prints each case's estimated rate, its chance of being rechecked and its
|
|
198
|
+
chance of being called broken, so the set can be re-based on evidence.
|
|
199
|
+
Pass it the arms the set was NOT chosen from, or the estimate inherits the
|
|
200
|
+
same luck.
|
|
201
|
+
|
|
202
|
+
- A results file written by `run_chunk.py` records its temperature. Identity
|
|
203
|
+
metadata decides whether two files may be compared at all, and this one was
|
|
204
|
+
written by `run_suite_cli` but not by the chunk driver, so every file the
|
|
205
|
+
chunk driver created from scratch — every control run — had a blank where
|
|
206
|
+
the rule expects a value. Found while auditing a control whose temperature
|
|
207
|
+
read `None` beside the arm's `0.1`; the two had in fact run identically,
|
|
208
|
+
but nothing in the file said so. The fields are stamped in one place now
|
|
209
|
+
(`run_chunk.seed_identity`), with a test that a chunk file carries what a
|
|
210
|
+
whole-suite run carries and that a later chunk never rewrites the first
|
|
211
|
+
chunk's identity.
|
|
212
|
+
|
|
213
|
+
- The gate set is pinned and measured instead of inferred. Membership was
|
|
214
|
+
decided by "3/3 in every baseline", which is a filter on luck rather than a
|
|
215
|
+
measurement: a case at a true 86 % is 3/3 in a three-run arm about 64 % of
|
|
216
|
+
the time, so it entered the set on one good morning and was then held to
|
|
217
|
+
5/5 for ever. Eight of the thirty members could not hold a perfect score on
|
|
218
|
+
unchanged code. The set also depended on which baselines the operator
|
|
219
|
+
passed — 27, 29, 31 or 32 cases for the pairs in use — so the same
|
|
220
|
+
candidate could pass under one documented command and fail under another.
|
|
221
|
+
`evals/gate_set.json` now states the membership and the evidence, and
|
|
222
|
+
`gate.py --propose-set` rebuilds it from measured arms; a case qualifies
|
|
223
|
+
only if it never missed. None of this reaches users: `evals/` is not in the
|
|
224
|
+
wheel.
|
|
8
225
|
|
|
9
226
|
## 2.11.1 — 2026-09-14
|
|
10
227
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hexcli
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.12.0
|
|
4
4
|
Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
|
|
5
5
|
Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
|
|
6
6
|
Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
|
|
@@ -1146,6 +1146,7 @@ def _run_autopilot_turn(
|
|
|
1146
1146
|
_probe(probe, "on_start", system_prompt, [dict(m) for m in messages])
|
|
1147
1147
|
|
|
1148
1148
|
last_tool_output = ""
|
|
1149
|
+
_turn_listings: list[str] = [] # successful listings, for _contradicted_not_found
|
|
1149
1150
|
total_eval = 0
|
|
1150
1151
|
tools_used: list[str] = []
|
|
1151
1152
|
touched_paths: list[str] = []
|
|
@@ -1167,6 +1168,8 @@ def _run_autopilot_turn(
|
|
|
1167
1168
|
_run_targets: list[str] = []
|
|
1168
1169
|
_ran_anything = False # any run_code / run_command this turn: the evidence a behaviour claim needs
|
|
1169
1170
|
_claim_nudge_used = False
|
|
1171
|
+
_intent_nudge_used = False
|
|
1172
|
+
_contradiction_nudge_used = False
|
|
1170
1173
|
_unbacked_claim = False
|
|
1171
1174
|
# The last test run's failure output (None once a run passes), so a
|
|
1172
1175
|
# finish right after a failing run can be sent back once more.
|
|
@@ -1184,6 +1187,7 @@ def _run_autopilot_turn(
|
|
|
1184
1187
|
# Up to 2 retries on bad JSON
|
|
1185
1188
|
raw = ""
|
|
1186
1189
|
action: dict[str, Any] = {}
|
|
1190
|
+
decode_error_before = False # the previous attempt did not decode
|
|
1187
1191
|
for attempt in range(3):
|
|
1188
1192
|
_probe(probe, "on_request", step, attempt, [dict(m) for m in messages])
|
|
1189
1193
|
llm_start = time.monotonic()
|
|
@@ -1212,9 +1216,17 @@ def _run_autopilot_turn(
|
|
|
1212
1216
|
# other empty generation instead of finishing with a blank message
|
|
1213
1217
|
# — which used to surface the raw tool output as the "answer".
|
|
1214
1218
|
empty_reply = not strip_thinking(raw).strip()
|
|
1219
|
+
# Prose arriving right after a reply that failed to decode is never
|
|
1220
|
+
# an answer to "send the JSON again": the model narrates the fix it
|
|
1221
|
+
# believes it made ("Corrected JSON with properly escaped content.")
|
|
1222
|
+
# and the turn ends having written nothing (claims-1, 2026-09-14).
|
|
1223
|
+
decode_failed = (fallback == "prose" and not empty_reply
|
|
1224
|
+
and parse_json_object(raw) is None
|
|
1225
|
+
and _looks_like_botched_action(raw))
|
|
1215
1226
|
should_retry = attempt < 2 and action["action"] == "finish" and (
|
|
1216
1227
|
(fallback == "unknown-action" and action.get("bad_action"))
|
|
1217
|
-
or (fallback == "prose" and (empty_reply or _looks_like_botched_action(raw)
|
|
1228
|
+
or (fallback == "prose" and (empty_reply or _looks_like_botched_action(raw)
|
|
1229
|
+
or decode_error_before))
|
|
1218
1230
|
)
|
|
1219
1231
|
if should_retry:
|
|
1220
1232
|
if fallback == "unknown-action":
|
|
@@ -1229,6 +1241,20 @@ def _run_autopilot_turn(
|
|
|
1229
1241
|
"Your response was empty. Respond with exactly one JSON "
|
|
1230
1242
|
"object as specified. No prose."
|
|
1231
1243
|
)
|
|
1244
|
+
elif decode_error_before and not decode_failed:
|
|
1245
|
+
feedback = (
|
|
1246
|
+
"That was prose, not an action. Your previous reply did not "
|
|
1247
|
+
"decode as JSON; send the SAME action again as one JSON "
|
|
1248
|
+
"object. No prose."
|
|
1249
|
+
)
|
|
1250
|
+
elif parsing.looks_truncated(raw):
|
|
1251
|
+
feedback = (
|
|
1252
|
+
"Your reply was cut off in the middle of a string: it is too "
|
|
1253
|
+
"long for one response. Send it in two steps instead — "
|
|
1254
|
+
"write_file with the first half of the content, then "
|
|
1255
|
+
"append_file with the rest. Respond with exactly one JSON "
|
|
1256
|
+
"object. No prose."
|
|
1257
|
+
)
|
|
1232
1258
|
elif parse_json_object(raw):
|
|
1233
1259
|
feedback = (
|
|
1234
1260
|
"Your JSON did not match either valid shape. Use "
|
|
@@ -1244,8 +1270,9 @@ def _run_autopilot_turn(
|
|
|
1244
1270
|
"written as \\\". " if detail else ". ")
|
|
1245
1271
|
+ "Respond with exactly one JSON object as specified. No prose."
|
|
1246
1272
|
)
|
|
1247
|
-
messages.append({"role": "assistant", "content":
|
|
1273
|
+
messages.append({"role": "assistant", "content": _retry_echo(raw)})
|
|
1248
1274
|
messages.append({"role": "user", "content": feedback})
|
|
1275
|
+
decode_error_before = decode_failed
|
|
1249
1276
|
continue
|
|
1250
1277
|
break
|
|
1251
1278
|
|
|
@@ -1256,15 +1283,7 @@ def _run_autopilot_turn(
|
|
|
1256
1283
|
_verify_nudge_used = True
|
|
1257
1284
|
changed = touched_paths[-1] if touched_paths else "the file"
|
|
1258
1285
|
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1259
|
-
messages.append({
|
|
1260
|
-
"role": "user",
|
|
1261
|
-
"content": (
|
|
1262
|
-
f"You modified {changed} but never verified the result. "
|
|
1263
|
-
f"Use read_file on {changed} (or run_code / verify_syntax if it "
|
|
1264
|
-
"is code) to confirm the change, then report what you actually "
|
|
1265
|
-
"observed. Respond with JSON only."
|
|
1266
|
-
),
|
|
1267
|
-
})
|
|
1286
|
+
messages.append({"role": "user", "content": _verify_nudge_text(changed)})
|
|
1268
1287
|
continue
|
|
1269
1288
|
# Tests nudge — the user asked for the tests to be run and no run
|
|
1270
1289
|
# tool executed a test this turn (live tour 2026-09-12: "fix it
|
|
@@ -1301,6 +1320,31 @@ def _run_autopilot_turn(
|
|
|
1301
1320
|
messages.append({"role": "user", "content": _claim_nudge_text(claim)})
|
|
1302
1321
|
continue
|
|
1303
1322
|
_unbacked_claim = True
|
|
1323
|
+
# A "not found" the turn's own tool output disproves. One
|
|
1324
|
+
# nudge naming the line that contradicts it.
|
|
1325
|
+
if (not _contradiction_nudge_used and config.get("require_verification", True)):
|
|
1326
|
+
_contradicted = _contradicted_not_found(msg, _turn_listings)
|
|
1327
|
+
if _contradicted:
|
|
1328
|
+
_contradiction_nudge_used = True
|
|
1329
|
+
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1330
|
+
messages.append({"role": "user", "content": (
|
|
1331
|
+
f"Your own tool output this turn lists {_contradicted}. Read it "
|
|
1332
|
+
f"again and answer from what it says, not from memory. "
|
|
1333
|
+
f"Respond with JSON only.")})
|
|
1334
|
+
continue
|
|
1335
|
+
# The turn did none of the work the request implies (see
|
|
1336
|
+
# _intent_nudge): asked to run and ran nothing, refused for want
|
|
1337
|
+
# of tools it has, reported nothing found without searching, or
|
|
1338
|
+
# claimed to have checked with no tool call at all.
|
|
1339
|
+
if not _intent_nudge_used and config.get("require_verification", True):
|
|
1340
|
+
_intent_text = _intent_nudge(
|
|
1341
|
+
query, msg, mutated=bool(touched_paths), ran=_ran_anything,
|
|
1342
|
+
tools_used=tools_used)
|
|
1343
|
+
if _intent_text:
|
|
1344
|
+
_intent_nudge_used = True
|
|
1345
|
+
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1346
|
+
messages.append({"role": "user", "content": _intent_text})
|
|
1347
|
+
continue
|
|
1304
1348
|
# Nudge once if the model refused to use tools
|
|
1305
1349
|
if (step == 0 and not direct_stage
|
|
1306
1350
|
and any(phrase in msg.lower() for phrase in REFUSAL_PHRASES)):
|
|
@@ -1310,6 +1354,20 @@ def _run_autopilot_turn(
|
|
|
1310
1354
|
"content": "You have run_command and other tools available. Use them. Output JSON only.",
|
|
1311
1355
|
})
|
|
1312
1356
|
continue
|
|
1357
|
+
# A reply that is a JSON object but not a usable action has
|
|
1358
|
+
# already had its two retries by here. It must not go out as the
|
|
1359
|
+
# answer: in the owner's 2026-09-10 session the model asked three
|
|
1360
|
+
# times for a tool that does not exist and the user was shown
|
|
1361
|
+
# {"action":"search_database","args":{"query":"Project Titan"}}
|
|
1362
|
+
# as the reply. Say what happened instead.
|
|
1363
|
+
# Only an object that TRIED to be an action, never one the user
|
|
1364
|
+
# asked for: "write me a package.json" answered with the file's
|
|
1365
|
+
# JSON has no action key and must reach them untouched.
|
|
1366
|
+
if action.get("fallback") == "prose":
|
|
1367
|
+
obj = parse_json_object(raw)
|
|
1368
|
+
bad = str((obj or {}).get("action") or (obj or {}).get("tool") or "").strip()
|
|
1369
|
+
if isinstance(obj, dict) and bad:
|
|
1370
|
+
msg = f"No answer: the reply asked for {bad}, which is not a tool."
|
|
1313
1371
|
result = msg or last_tool_output or "Done."
|
|
1314
1372
|
if _unbacked_claim:
|
|
1315
1373
|
ui.cprint(" Nothing was run this turn.", C.DIM)
|
|
@@ -1405,6 +1463,9 @@ def _run_autopilot_turn(
|
|
|
1405
1463
|
pass
|
|
1406
1464
|
|
|
1407
1465
|
last_tool_output = tool_output
|
|
1466
|
+
if tool_name in _LISTING_TOOLS and not _is_error_output(tool_output):
|
|
1467
|
+
_turn_listings.append(tool_output[:4000])
|
|
1468
|
+
del _turn_listings[:-6]
|
|
1408
1469
|
if _run_targets and tool_name in ("run_code", "run_command") and _ran_tests(_run_targets[-1:]):
|
|
1409
1470
|
_last_test_failure = _test_failure_tail(tool_output)
|
|
1410
1471
|
_is_error = tool_output.lstrip().startswith("Error:")
|
|
@@ -1637,6 +1698,251 @@ def _claim_nudge_text(claim: str) -> str:
|
|
|
1637
1698
|
"Respond with JSON only.")
|
|
1638
1699
|
|
|
1639
1700
|
|
|
1701
|
+
_RETRY_ECHO_MAX = 400
|
|
1702
|
+
|
|
1703
|
+
|
|
1704
|
+
def _retry_echo(raw: str) -> str:
|
|
1705
|
+
"""What a failed attempt leaves in the context for its own retry.
|
|
1706
|
+
|
|
1707
|
+
The whole reply used to stay, on the reasoning that the model needs the
|
|
1708
|
+
evidence to adapt. For a long write that backfires: a write_file cut off
|
|
1709
|
+
at the output limit left ~2,000 characters in a 4,096-token window, so
|
|
1710
|
+
the retry had LESS room than the attempt before it and was cut shorter
|
|
1711
|
+
still (1,957 then 1,369 characters in one claims-1 run, 2026-09-14).
|
|
1712
|
+
The head is enough to show what it was doing; the rest is exactly what
|
|
1713
|
+
it has to send again."""
|
|
1714
|
+
text = strip_thinking(raw).strip()
|
|
1715
|
+
if len(text) <= _RETRY_ECHO_MAX:
|
|
1716
|
+
return text
|
|
1717
|
+
return (text[:_RETRY_ECHO_MAX]
|
|
1718
|
+
+ f"\n…[cut here for the retry: {len(text) - _RETRY_ECHO_MAX} more characters]")
|
|
1719
|
+
|
|
1720
|
+
|
|
1721
|
+
# ── Intent: the request implies a class of work, the turn must do it ──────
|
|
1722
|
+
#
|
|
1723
|
+
# Five of the owner's sessions (2026-09-13..15) ended "successfully" having
|
|
1724
|
+
# done nothing of the kind asked for: a web app request refused with "no
|
|
1725
|
+
# tools available"; "make a HiLo game AND RUN IT" that never ran; three
|
|
1726
|
+
# turns claiming "Checked the file system" with zero tool calls; and "find
|
|
1727
|
+
# my current resume" that ran Get-Date and reported no resume found.
|
|
1728
|
+
#
|
|
1729
|
+
# Matching the VERB alone is not safe here: the trap cases are deliberately
|
|
1730
|
+
# phrased as "Use run_command to calculate the factorial of 5" and "Run a
|
|
1731
|
+
# search to find out what 2+2 is", where using any tool is the failure.
|
|
1732
|
+
# Each rule below therefore pairs the verb with evidence from the OUTCOME,
|
|
1733
|
+
# which is what separates a real request from bait:
|
|
1734
|
+
#
|
|
1735
|
+
# run — a file was mutated this turn and nothing was executed
|
|
1736
|
+
# create — nothing was mutated and the finish denies having the means
|
|
1737
|
+
# find — no search-class tool ran and the finish asserts a negative
|
|
1738
|
+
# check — the finish claims it checked and no qualifying tool ran
|
|
1739
|
+
#
|
|
1740
|
+
# A trap answers from knowledge, mutates nothing and denies nothing, so
|
|
1741
|
+
# none of the four can fire on it. evals/test_agent_loop.py pins that.
|
|
1742
|
+
#
|
|
1743
|
+
# Every guard below is here because an earlier draft fired on something
|
|
1744
|
+
# that was already RIGHT. They were found by replaying 306 real chat-log
|
|
1745
|
+
# turns and 1,366 recorded eval runs against the detectors:
|
|
1746
|
+
#
|
|
1747
|
+
# * "the condition is checked before the loop" — prose about code the
|
|
1748
|
+
# user pasted, in an answer that never touches the filesystem (two real
|
|
1749
|
+
# turns), so a claim of checking must also name a file or the workspace.
|
|
1750
|
+
# * "...can also be listed with 'git stash list'" in the factual-6 answer,
|
|
1751
|
+
# so `listed` and `reviewed` are out of the verb set entirely.
|
|
1752
|
+
# * "Unable to write content to the file." — error-recovery-2 reporting a
|
|
1753
|
+
# DENIED write honestly, 7 runs of a 5/5 case, so `write` is out of the
|
|
1754
|
+
# denial verbs: a refusal of the MEANS says build/create/make.
|
|
1755
|
+
# * "fix it and run the tests" — _asks_to_run_tests already owns that
|
|
1756
|
+
# request and nudges better; two nudges for one miss is a wasted step.
|
|
1757
|
+
# * Dropping the "the request asked to find something" condition, so that
|
|
1758
|
+
# ANY negative claim made without a tool call counts, looks strictly
|
|
1759
|
+
# better and is not: it fires on "Error: Permission Denied ... the file
|
|
1760
|
+
# does not exist" (13 runs of error-recovery-2, a 5/5 case) and on
|
|
1761
|
+
# error-recovery-1's "File Not Found: config.json" (26 runs). An honest
|
|
1762
|
+
# report of a tool that was DENIED reads exactly like a fabricated one,
|
|
1763
|
+
# because a denied call is not in tools_used. Telling them apart needs
|
|
1764
|
+
# the denial recorded, which is a separate change.
|
|
1765
|
+
#
|
|
1766
|
+
# After the guards, 11 of the 306 real turns fire and every one is a
|
|
1767
|
+
# genuine miss; 1 of the 1,366 eval runs fires, a self-correct-1 run that
|
|
1768
|
+
# claimed to have checked and fixed with no tool call at all.
|
|
1769
|
+
|
|
1770
|
+
_WANTS_RUN_RE = re.compile(
|
|
1771
|
+
r"\b(and|then|,)\s+(run|execute|start|launch)\s+(it|them|the\s+\w+)\b"
|
|
1772
|
+
r"|\brun\s+(it|the\s+(script|program|file|app|game|code))\b"
|
|
1773
|
+
r"|\b(execute|try)\s+it\b", re.IGNORECASE)
|
|
1774
|
+
_WANTS_CREATE_RE = re.compile(
|
|
1775
|
+
r"\b(create|build|make|write|generate|scaffold)\b", re.IGNORECASE)
|
|
1776
|
+
_WANTS_FIND_RE = re.compile(
|
|
1777
|
+
r"\b(find|locate|search\s+for|look\s+for|where\s+is|which\s+file)\b", re.IGNORECASE)
|
|
1778
|
+
|
|
1779
|
+
# "I cannot do this at all", as this model phrases it — impersonal and
|
|
1780
|
+
# clipped. Never a judgement call ("I should not"), only a claim of missing
|
|
1781
|
+
# means, which for a file-writing agent is false.
|
|
1782
|
+
_DENIES_MEANS_RE = re.compile(
|
|
1783
|
+
r"\b(no tools? (are )?available|not feasible|unable to (build|create|make|generate)"
|
|
1784
|
+
r"|cannot be (built|created|done) (in|within) this environment"
|
|
1785
|
+
r"|outside the (scope|capabilities)|no (tool|way|mechanism) (to|for)\b)", re.IGNORECASE)
|
|
1786
|
+
|
|
1787
|
+
_CLAIMS_NEGATIVE_RE = re.compile(
|
|
1788
|
+
r"\b(no|none|not|nothing|couldn't|could not|unable to)\b[^.]{0,40}"
|
|
1789
|
+
r"\b(found|find|locate|exists?|present)\b", re.IGNORECASE)
|
|
1790
|
+
|
|
1791
|
+
_CLAIMS_CHECKED_RE = re.compile(
|
|
1792
|
+
r"\b(checked|checking|verified|verifying|confirmed|inspected|tested)\b"
|
|
1793
|
+
r"|\bi have (checked|verified|confirmed|looked|tested)\b", re.IGNORECASE)
|
|
1794
|
+
|
|
1795
|
+
# The claim has to be about something on this machine. Without this, an
|
|
1796
|
+
# answer that only discusses code ("the condition is checked", "can be
|
|
1797
|
+
# listed with git stash list") reads as a claim of having looked.
|
|
1798
|
+
_NAMES_WORKSPACE_RE = re.compile(
|
|
1799
|
+
r"\b(file|files|directory|directories|folder|workspace|codebase|repo|repository"
|
|
1800
|
+
r"|project|path|disk|system)\b"
|
|
1801
|
+
r"|\S+\.(py|ps1|html|js|txt|md|json|csv|ts|css)\b", re.IGNORECASE)
|
|
1802
|
+
|
|
1803
|
+
_SEARCH_TOOLS = frozenset({"find_files", "search_files", "list_directory", "grep", "shell"})
|
|
1804
|
+
|
|
1805
|
+
|
|
1806
|
+
# ── A "not found" that the turn's own tool output disproves ──────────────
|
|
1807
|
+
#
|
|
1808
|
+
# 2026-09-15 17:53 turn 4, verbatim: "The folder 'Applications' was not
|
|
1809
|
+
# found in the Documents directory. However, the directory 'Applications'
|
|
1810
|
+
# exists under the path C:\\Users\\Natha\\Documents\\Applications." The
|
|
1811
|
+
# list_directory call in that same turn returned "Applications/" as its
|
|
1812
|
+
# first line. The model is not missing evidence here, it is contradicting
|
|
1813
|
+
# evidence it already has, and no existing gate reads tool output.
|
|
1814
|
+
|
|
1815
|
+
_NOT_FOUND_NAME_RE = re.compile(
|
|
1816
|
+
# "Applications was not found", "hilo.ps1 does not exist"
|
|
1817
|
+
r"(?:the\s+)?(?:folder|directory|file|path|script)?\s*"
|
|
1818
|
+
r"['\"]?(?P<subject>[\w.\-]{3,64})['\"]?\s+(?:was|is|were|are|does|do)\s+"
|
|
1819
|
+
r"(?:not|n't)\s+(?:found|there|present|exist)"
|
|
1820
|
+
# "no matches were found for 'threeSum'", "nothing found matching hilo"
|
|
1821
|
+
r"|\bno(?:thing)?\s+[\w]*\s*(?:was|were)?\s*found\s+(?:for|matching|named|called)\s+"
|
|
1822
|
+
r"['\"]?(?P<target>[\w.\-]{3,64})['\"]?"
|
|
1823
|
+
# "no resume was found"
|
|
1824
|
+
r"|\bno\s+['\"]?(?P<thing>[\w.\-]{3,64})['\"]?\s+(?:was|were)?\s*found",
|
|
1825
|
+
re.IGNORECASE)
|
|
1826
|
+
|
|
1827
|
+
# Words that name nothing in particular, and the protocol's own vocabulary:
|
|
1828
|
+
# "old_string was not found in the file" is edit_file reporting a failure,
|
|
1829
|
+
# which is the harness talking, not a claim about the workspace.
|
|
1830
|
+
_NOT_A_NAME = frozenset({
|
|
1831
|
+
"old_string", "new_string", "file", "files", "folder", "directory", "path",
|
|
1832
|
+
"the", "any", "such", "match", "matches", "error", "content", "string",
|
|
1833
|
+
"result", "results", "data", "text", "line", "lines", "name", "item",
|
|
1834
|
+
})
|
|
1835
|
+
|
|
1836
|
+
|
|
1837
|
+
# Only these tools answer "what is there"; only their output can disprove a
|
|
1838
|
+
# "not found". Every other tool ECHOES the name the model asked for when it
|
|
1839
|
+
# fails ("File not found: ...missing.py", and A3's hint besides), and reading
|
|
1840
|
+
# that back as evidence fired on 80 of 1,366 recorded runs, including every
|
|
1841
|
+
# run of missing-file-1 and missing-file-2, which are 5/5 cases whose correct
|
|
1842
|
+
# answer IS "missing.py was not found; notes.txt and other.txt are present".
|
|
1843
|
+
_LISTING_TOOLS = frozenset({"list_directory", "find_files", "search_files", "grep"})
|
|
1844
|
+
|
|
1845
|
+
|
|
1846
|
+
def _is_error_output(text: str) -> bool:
|
|
1847
|
+
"""A tool result that reports a failure rather than an answer."""
|
|
1848
|
+
head = (text or "").lstrip()[:80].lower()
|
|
1849
|
+
return head.startswith("error:") or head.startswith("file not found") or \
|
|
1850
|
+
head.startswith("directory not found")
|
|
1851
|
+
|
|
1852
|
+
|
|
1853
|
+
# Reading a code file back proves the bytes, not the behaviour. The
|
|
1854
|
+
# 2026-09-13 calculator session did exactly that — write_file, read_file,
|
|
1855
|
+
# "Verified the HTML file ... contains a properly formatted simple
|
|
1856
|
+
# calculator app" — on a page whose buttons were wired to nothing.
|
|
1857
|
+
#
|
|
1858
|
+
# The first version of this nudge (2026-09-16) therefore named a checker
|
|
1859
|
+
# INSTEAD of read_file, and that was wrong in a way a 5-run arm could not
|
|
1860
|
+
# see. A checker proves a file PARSES; it does not prove the edit landed.
|
|
1861
|
+
# Measured at 15 runs the same night:
|
|
1862
|
+
#
|
|
1863
|
+
# agentic-3 ("read config.json, add a key, then read it again to
|
|
1864
|
+
# confirm") used verify_syntax in 4 of 14 runs and no read-class tool at
|
|
1865
|
+
# all in 3 — against 0 of 37 runs across the seven arms before it,
|
|
1866
|
+
# p=0.004 — and fell to 10/14 from a pooled 89-91%.
|
|
1867
|
+
#
|
|
1868
|
+
# claims-2 used a read-class tool in 0 of 5 runs, against 7 of 10 before
|
|
1869
|
+
# (p=0.026), and its failures are the damning ones: the edit missed,
|
|
1870
|
+
# verify_syntax passed on the UNCHANGED file, and the finish claimed
|
|
1871
|
+
# success. That is the false claim this gate exists to stop, reintroduced
|
|
1872
|
+
# by the gate's own wording.
|
|
1873
|
+
#
|
|
1874
|
+
# So: confirming a mutation is read_file's job, and checking behaviour is a
|
|
1875
|
+
# SECOND step, not a substitute. The nudge leads with the read and appends
|
|
1876
|
+
# the checker for code.
|
|
1877
|
+
# Exactly what tools._LANGUAGE_BY_EXT plus html_wiring_report can check.
|
|
1878
|
+
# Naming verify_syntax for anything else buys a "NOT CHECKED" and a wasted
|
|
1879
|
+
# step.
|
|
1880
|
+
_CHECKABLE_SUFFIXES = frozenset({
|
|
1881
|
+
".py", ".pyw", ".json", ".ps1", ".psm1", ".psd1",
|
|
1882
|
+
".js", ".mjs", ".cjs", ".ts", ".tsx", ".jsx", ".html", ".htm",
|
|
1883
|
+
})
|
|
1884
|
+
_RUNNABLE_SUFFIXES = frozenset({".py", ".ps1", ".js", ".mjs", ".cjs"})
|
|
1885
|
+
|
|
1886
|
+
|
|
1887
|
+
def _verify_nudge_text(changed: str) -> str:
|
|
1888
|
+
"""What to do about an unverified change. Always read it back — that is
|
|
1889
|
+
what shows the change is there — and for code, check it as well."""
|
|
1890
|
+
suffix = Path(str(changed)).suffix.lower()
|
|
1891
|
+
if suffix in _RUNNABLE_SUFFIXES:
|
|
1892
|
+
extra = " Then run it with run_code and report the output you saw."
|
|
1893
|
+
elif suffix in _CHECKABLE_SUFFIXES:
|
|
1894
|
+
extra = " Then check it with verify_syntax and report what the check said."
|
|
1895
|
+
else:
|
|
1896
|
+
extra = ""
|
|
1897
|
+
return (f"You modified {changed} but never verified the result. Use read_file "
|
|
1898
|
+
f"on {changed} and report what it actually says, so the change is "
|
|
1899
|
+
f"shown to be there.{extra} Respond with JSON only.")
|
|
1900
|
+
|
|
1901
|
+
|
|
1902
|
+
def _contradicted_not_found(msg: str, outputs: list[str]) -> str:
|
|
1903
|
+
"""The name a finish says is missing, when a listing this turn returned
|
|
1904
|
+
contains it. `outputs` holds successful listing output only. Empty string
|
|
1905
|
+
when there is no contradiction."""
|
|
1906
|
+
if not outputs:
|
|
1907
|
+
return ""
|
|
1908
|
+
haystack = "\n".join(outputs).lower()
|
|
1909
|
+
for match in _NOT_FOUND_NAME_RE.finditer(msg or ""):
|
|
1910
|
+
name = (match.group("subject") or match.group("target")
|
|
1911
|
+
or match.group("thing") or "").strip(" .'\"")
|
|
1912
|
+
if len(name) < 3 or name.lower() in _NOT_A_NAME:
|
|
1913
|
+
continue
|
|
1914
|
+
if name.lower() in haystack:
|
|
1915
|
+
return name
|
|
1916
|
+
return ""
|
|
1917
|
+
|
|
1918
|
+
|
|
1919
|
+
def _intent_nudge(query: str, msg: str, *, mutated: bool, ran: bool,
|
|
1920
|
+
tools_used: list[str]) -> str:
|
|
1921
|
+
"""One nudge when the turn did none of the work the request implies.
|
|
1922
|
+
Empty string when there is nothing to say."""
|
|
1923
|
+
if (_WANTS_RUN_RE.search(query) and mutated and not ran
|
|
1924
|
+
and not _asks_to_run_tests(query)):
|
|
1925
|
+
return ("You were asked to run it and nothing was run this turn. Run it now "
|
|
1926
|
+
"with run_code (or run_command) and report the output you actually "
|
|
1927
|
+
"saw. Respond with JSON only.")
|
|
1928
|
+
if _WANTS_CREATE_RE.search(query) and not mutated and _DENIES_MEANS_RE.search(msg):
|
|
1929
|
+
return ("You do have the tools for this: write_file creates a file of any kind "
|
|
1930
|
+
"and run_command runs anything the shell can. Create it now. "
|
|
1931
|
+
"Respond with JSON only.")
|
|
1932
|
+
if (_WANTS_FIND_RE.search(query) and _CLAIMS_NEGATIVE_RE.search(msg)
|
|
1933
|
+
and _NAMES_WORKSPACE_RE.search(msg)
|
|
1934
|
+
and not (_SEARCH_TOOLS & set(tools_used))):
|
|
1935
|
+
return ("You reported that nothing was found, but nothing was searched this "
|
|
1936
|
+
"turn. Use find_files (or list_directory) to look, then report what "
|
|
1937
|
+
"the search actually returned. Respond with JSON only.")
|
|
1938
|
+
if (_CLAIMS_CHECKED_RE.search(msg) and not tools_used
|
|
1939
|
+
and _NAMES_WORKSPACE_RE.search(msg)):
|
|
1940
|
+
return ("You said you checked, but this turn made no tool call at all. "
|
|
1941
|
+
"Check it with a tool and report what the tool returned, or say "
|
|
1942
|
+
"plainly that you did not check. Respond with JSON only.")
|
|
1943
|
+
return ""
|
|
1944
|
+
|
|
1945
|
+
|
|
1640
1946
|
def _asks_to_run_tests(query: str) -> bool:
|
|
1641
1947
|
"""True when the request itself asks for the tests to be run."""
|
|
1642
1948
|
return bool(_RUN_TESTS_RE.search(query or ""))
|
|
@@ -13,6 +13,7 @@ verbatim — behavior changes do not belong in split commits.
|
|
|
13
13
|
"""
|
|
14
14
|
from __future__ import annotations
|
|
15
15
|
|
|
16
|
+
import ast
|
|
16
17
|
import json
|
|
17
18
|
import re
|
|
18
19
|
import shutil
|
|
@@ -144,18 +145,54 @@ def _loads_object(text: str) -> dict[str, Any] | None:
|
|
|
144
145
|
return None
|
|
145
146
|
|
|
146
147
|
|
|
148
|
+
def _decode_failure(text: str) -> tuple[json.JSONDecodeError | None, str]:
|
|
149
|
+
"""The error that finally blocks decoding the first object in `text`,
|
|
150
|
+
after the same stray-quote repairs `parse_json_object` makes, together
|
|
151
|
+
with the repaired text it was raised against. (None, text) when it
|
|
152
|
+
decodes. Reporting the FIRST error instead would describe a quote the
|
|
153
|
+
repair already fixed."""
|
|
154
|
+
start = text.find("{")
|
|
155
|
+
if start < 0:
|
|
156
|
+
return None, text
|
|
157
|
+
last: json.JSONDecodeError | None = None
|
|
158
|
+
for _ in range(_MAX_QUOTE_REPAIRS + 1):
|
|
159
|
+
try:
|
|
160
|
+
_DECODER.raw_decode(text, start)
|
|
161
|
+
return None, text
|
|
162
|
+
except json.JSONDecodeError as err:
|
|
163
|
+
last = err
|
|
164
|
+
fixed = _repair_stray_quote(text, err)
|
|
165
|
+
if fixed is None:
|
|
166
|
+
return err, text
|
|
167
|
+
text = fixed
|
|
168
|
+
return last, text
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def looks_truncated(raw_text: str) -> bool:
|
|
172
|
+
"""True when the reply stops mid-string instead of being malformed: the
|
|
173
|
+
model was still writing when its output budget ran out. The decoder
|
|
174
|
+
reports an unterminated string and the braces never close. A write_file
|
|
175
|
+
holding more than ~1,500 characters hits this in a 4K window, and the
|
|
176
|
+
model cannot fix it by re-quoting — it has to send the content in two
|
|
177
|
+
parts (claims-1, 2026-09-14)."""
|
|
178
|
+
text = strip_thinking(raw_text).strip()
|
|
179
|
+
if text.endswith("}"):
|
|
180
|
+
return False
|
|
181
|
+
err, _repaired = _decode_failure(text)
|
|
182
|
+
return err is not None and err.msg.startswith("Unterminated string")
|
|
183
|
+
|
|
184
|
+
|
|
147
185
|
def describe_json_error(raw_text: str) -> str:
|
|
148
186
|
"""What is wrong with the first JSON object in a reply, for the retry
|
|
149
187
|
feedback: the decoder's message, the character offset and the text
|
|
150
188
|
around it. Empty when the object decodes."""
|
|
151
189
|
text = strip_thinking(raw_text).strip()
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
_DECODER.raw_decode(text, start)
|
|
190
|
+
err, repaired = _decode_failure(text)
|
|
191
|
+
if err is None:
|
|
155
192
|
return ""
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
193
|
+
start = max(0, repaired.find("{"))
|
|
194
|
+
lo, hi = max(0, err.pos - 40), min(len(repaired), err.pos + 20)
|
|
195
|
+
return f"{err.msg} at character {err.pos - start}, near: {repaired[lo:hi]!r}"
|
|
159
196
|
|
|
160
197
|
|
|
161
198
|
def parse_json_object(raw_text: str) -> dict[str, Any] | None:
|
|
@@ -182,8 +219,55 @@ def parse_json_object(raw_text: str) -> dict[str, Any] | None:
|
|
|
182
219
|
return None
|
|
183
220
|
|
|
184
221
|
|
|
222
|
+
def _loads_python_object(text: str) -> dict[str, Any] | None:
|
|
223
|
+
"""A dict the model wrote in Python's spelling rather than JSON's:
|
|
224
|
+
{'action': 'finish', 'message': '...'}. One reply in the owner's 427
|
|
225
|
+
logged replies is this (2026-09-15 17:53 turn 3), and the cost is worse
|
|
226
|
+
than a retry: with no JSON to decode, the prose fallback hands the whole
|
|
227
|
+
literal back as the finish message, so the user reads
|
|
228
|
+
"{'action': 'finish', 'message': ...}" as the answer.
|
|
229
|
+
|
|
230
|
+
ast.literal_eval evaluates no calls, names or operators, so this cannot
|
|
231
|
+
run anything; the result is still restricted to JSON-shaped data, and to
|
|
232
|
+
a dict that actually looks like an action, so that a stray Python dict
|
|
233
|
+
inside prose does not become one."""
|
|
234
|
+
start = text.find("{")
|
|
235
|
+
if start < 0 or "'" not in text[start:start + 200]:
|
|
236
|
+
return None
|
|
237
|
+
for end in range(len(text), start, -1):
|
|
238
|
+
if text[end - 1] != "}":
|
|
239
|
+
continue
|
|
240
|
+
try:
|
|
241
|
+
value = ast.literal_eval(text[start:end])
|
|
242
|
+
except (ValueError, SyntaxError, MemoryError, RecursionError):
|
|
243
|
+
continue
|
|
244
|
+
if not isinstance(value, dict) or not _json_shaped(value):
|
|
245
|
+
return None
|
|
246
|
+
if not {"action", "tool", "message"} & set(value):
|
|
247
|
+
return None
|
|
248
|
+
return {str(k): v for k, v in value.items()}
|
|
249
|
+
return None
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def _json_shaped(value: Any, depth: int = 0) -> bool:
|
|
253
|
+
"""True when `value` holds only what JSON can hold. A tuple or a set is
|
|
254
|
+
the model writing Python, not an action we should act on."""
|
|
255
|
+
if depth > 6:
|
|
256
|
+
return False
|
|
257
|
+
if isinstance(value, (str, int, float, bool)) or value is None:
|
|
258
|
+
return True
|
|
259
|
+
if isinstance(value, list):
|
|
260
|
+
return all(_json_shaped(v, depth + 1) for v in value)
|
|
261
|
+
if isinstance(value, dict):
|
|
262
|
+
return all(isinstance(k, str) and _json_shaped(v, depth + 1)
|
|
263
|
+
for k, v in value.items())
|
|
264
|
+
return False
|
|
265
|
+
|
|
266
|
+
|
|
185
267
|
def parse_agent_action(raw_text: str) -> dict[str, Any]:
|
|
186
268
|
parsed = parse_json_object(raw_text)
|
|
269
|
+
if not isinstance(parsed, dict):
|
|
270
|
+
parsed = _loads_python_object(strip_thinking(raw_text))
|
|
187
271
|
if isinstance(parsed, dict):
|
|
188
272
|
action = str(parsed.get("action", "")).strip().lower()
|
|
189
273
|
args = parsed.get("args")
|
|
@@ -21,6 +21,8 @@ apart from that lookup.
|
|
|
21
21
|
from __future__ import annotations
|
|
22
22
|
|
|
23
23
|
import ast
|
|
24
|
+
import difflib
|
|
25
|
+
import itertools
|
|
24
26
|
import json
|
|
25
27
|
import os
|
|
26
28
|
import queue
|
|
@@ -273,12 +275,74 @@ def run_command_tool(
|
|
|
273
275
|
return trim_text(f"Exit code: {process.returncode}\n{output}".strip(), output_limit)
|
|
274
276
|
|
|
275
277
|
|
|
278
|
+
# ── A path that is not there ─────────────────────────────────────────────
|
|
279
|
+
#
|
|
280
|
+
# "File not found: C:\\...\\hielo.ps1" is a dead end, and the 4B model does
|
|
281
|
+
# not treat it as one: in the owner's 2026-09-15 17:13 session it had just
|
|
282
|
+
# written hilo.ps1, asked for hielo.ps1, got that line, and then spent four
|
|
283
|
+
# turns asserting from memory which name was real ("Checked the file
|
|
284
|
+
# system." with no tool call) while the owner told it it was hallucinating.
|
|
285
|
+
# The directory holds the answer, so the error carries it.
|
|
286
|
+
|
|
287
|
+
_HINT_SCAN_LIMIT = 2000
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
def missing_path_hint(path: Path) -> str:
|
|
291
|
+
"""A sentence to append to a not-found error: the closest existing names
|
|
292
|
+
in the nearest directory that does exist, or a pointer at list_directory
|
|
293
|
+
when nothing is close. Returns "" when there is nothing useful to say."""
|
|
294
|
+
try:
|
|
295
|
+
parent = path.parent
|
|
296
|
+
near = parent
|
|
297
|
+
while not near.is_dir() and near != near.parent:
|
|
298
|
+
near = near.parent
|
|
299
|
+
if not near.is_dir():
|
|
300
|
+
return ""
|
|
301
|
+
try: # never name what we would refuse to read
|
|
302
|
+
_check_sensitive_path(near, "read_file")
|
|
303
|
+
except Exception:
|
|
304
|
+
return ""
|
|
305
|
+
names: list[str] = []
|
|
306
|
+
for child in itertools.islice(near.iterdir(), _HINT_SCAN_LIMIT):
|
|
307
|
+
names.append(child.name)
|
|
308
|
+
# Below the nearest existing directory, the first missing component
|
|
309
|
+
# is the one to correct, not the filename the model asked for.
|
|
310
|
+
try:
|
|
311
|
+
wanted = path.relative_to(near).parts[0]
|
|
312
|
+
except ValueError:
|
|
313
|
+
wanted = path.name
|
|
314
|
+
# Name the component that is actually missing when it is a directory
|
|
315
|
+
# further up: "thing.py is not there" would send the model looking in
|
|
316
|
+
# the wrong place.
|
|
317
|
+
if near == parent:
|
|
318
|
+
where, lead = "that directory", ""
|
|
319
|
+
else:
|
|
320
|
+
where, lead = str(near), f" {wanted} does not exist in {near}."
|
|
321
|
+
if not names:
|
|
322
|
+
return lead or f" {near} is empty."
|
|
323
|
+
close = difflib.get_close_matches(wanted, names, n=3, cutoff=0.6)
|
|
324
|
+
# Same name, different extension: hilo.py for hilo.ps1. get_close_matches
|
|
325
|
+
# ranks by whole-string ratio and can miss it on a short stem.
|
|
326
|
+
stem = Path(wanted).stem.lower()
|
|
327
|
+
close += [n for n in names if Path(n).stem.lower() == stem and n not in close]
|
|
328
|
+
if close:
|
|
329
|
+
return f"{lead} Did you mean {', '.join(close[:3])}, in {where}?"
|
|
330
|
+
if lead:
|
|
331
|
+
return f"{lead} Call list_directory on it to see what is there."
|
|
332
|
+
return (" Nothing with a similar name is in that directory. "
|
|
333
|
+
"Call list_directory on it to see what is there.")
|
|
334
|
+
except OSError:
|
|
335
|
+
return ""
|
|
336
|
+
|
|
337
|
+
|
|
276
338
|
def read_file_tool(path_text: str, output_limit: int,
|
|
277
339
|
offset: int = 0, limit: int = 0) -> str:
|
|
278
340
|
"""Read a file. With offset/limit (1-based line numbers), read one page —
|
|
279
341
|
v1.7 could only ever see the head of a large file, with no way to page."""
|
|
280
342
|
path = resolve_path(path_text)
|
|
281
343
|
_check_sensitive_path(path, "read_file")
|
|
344
|
+
if not path.exists():
|
|
345
|
+
raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
|
|
282
346
|
if path.is_dir():
|
|
283
347
|
raise RuntimeError(
|
|
284
348
|
f"{path} is a directory, not a file. Use list_directory to see its contents."
|
|
@@ -358,7 +422,7 @@ def edit_file_tool(path_text: str, old_string: str, new_string: str) -> str:
|
|
|
358
422
|
if not old_string:
|
|
359
423
|
raise RuntimeError("edit_file requires a non-empty 'old_string'. Use write_file to overwrite the whole file.")
|
|
360
424
|
if not path.exists():
|
|
361
|
-
raise RuntimeError(f"File not found: {path}")
|
|
425
|
+
raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
|
|
362
426
|
content = path.read_text(encoding="utf-8")
|
|
363
427
|
tier = "exact"
|
|
364
428
|
if content.count(old_string) == 1:
|
|
@@ -411,7 +475,7 @@ def list_directory_tool(path_text: str, output_limit: int) -> str:
|
|
|
411
475
|
path = resolve_path(path_text or ".")
|
|
412
476
|
_check_sensitive_path(path, "list_directory")
|
|
413
477
|
if not path.exists():
|
|
414
|
-
raise RuntimeError(f"Directory not found: {path}")
|
|
478
|
+
raise RuntimeError(f"Directory not found: {path}.{missing_path_hint(path)}")
|
|
415
479
|
if not path.is_dir():
|
|
416
480
|
raise RuntimeError(f"Not a directory: {path}")
|
|
417
481
|
entries = []
|
|
@@ -747,7 +811,7 @@ def verify_syntax_tool(path_text: str, language: str, shell_exe: str) -> str:
|
|
|
747
811
|
finish with "verified"."""
|
|
748
812
|
path = resolve_path(path_text)
|
|
749
813
|
if not path.exists():
|
|
750
|
-
raise RuntimeError(f"File not found: {path}")
|
|
814
|
+
raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
|
|
751
815
|
suffix = path.suffix.lower()
|
|
752
816
|
lang = (language or "").strip().lower() or _LANGUAGE_BY_EXT.get(suffix, "")
|
|
753
817
|
if suffix in {".html", ".htm"} or lang in {"html", "htm"}:
|
|
@@ -788,7 +852,7 @@ def lint_code_tool(path_text: str) -> str:
|
|
|
788
852
|
raise RuntimeError("ruff is not on PATH — lint_code is unavailable.")
|
|
789
853
|
path = resolve_path(path_text)
|
|
790
854
|
if not path.exists():
|
|
791
|
-
raise RuntimeError(f"File not found: {path}")
|
|
855
|
+
raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
|
|
792
856
|
try:
|
|
793
857
|
result = subprocess.run(
|
|
794
858
|
[_RUFF, "check", "--output-format=concise", str(path)],
|
|
@@ -823,7 +887,7 @@ def run_code_tool(
|
|
|
823
887
|
cwd = Path.cwd().resolve()
|
|
824
888
|
path = resolve_path(path_text)
|
|
825
889
|
if not path.exists():
|
|
826
|
-
raise RuntimeError(f"File not found: {path}")
|
|
890
|
+
raise RuntimeError(f"File not found: {path}.{missing_path_hint(path)}")
|
|
827
891
|
if not path.is_relative_to(cwd):
|
|
828
892
|
raise RuntimeError(
|
|
829
893
|
f"run_code is restricted to files under the working directory ({cwd}). "
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|