hexcli 2.9.1__tar.gz → 2.11.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hexcli-2.9.1 → hexcli-2.11.0}/.gitignore +4 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/CHANGELOG.md +71 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/PKG-INFO +1 -1
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/__init__.py +1 -1
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/agent.py +50 -2
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/lineedit.py +7 -3
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/parsing.py +79 -30
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/tools.py +188 -5
- {hexcli-2.9.1 → hexcli-2.11.0}/Hex CLI.cmd +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/LICENSE +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/README.md +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/assets/hexcli.ico +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/assets/hexcli.png +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/cancel.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/chatlog.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/commands.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/compaction.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/config.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/diffview.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/distribution.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/doctor.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/escalate.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/http_client.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/launcher.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/llm.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/local_escalation.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/lockfile.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/loop_v2.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/markdown_stream.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/memory.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/network.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/paths.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/prompts.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/protocol_v2.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/repl.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/safety.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/sessions.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/setup_wizard.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/shell_session.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/statusbar.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/stream_render.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/telemetry.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/hexcli/ui.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/install.ps1 +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/launcher.py +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/pyproject.toml +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/shellai.cmd +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/shellai.example.json +0 -0
- {hexcli-2.9.1 → hexcli-2.11.0}/shellai.py +0 -0
|
@@ -56,3 +56,7 @@ tools/backend_bench/.battery_probe.ps1
|
|
|
56
56
|
|
|
57
57
|
# python -m build output
|
|
58
58
|
dist/
|
|
59
|
+
|
|
60
|
+
# Local-only notes: research surveys and working documents that are not user-facing.
|
|
61
|
+
# Only the paper (docs/paper) and user-facing docs are committed (owner rule, 2026-09-14).
|
|
62
|
+
docs/local/
|
|
@@ -6,6 +6,77 @@ the Hexagon NPU, not single-run anecdotes.
|
|
|
6
6
|
|
|
7
7
|
## Unreleased
|
|
8
8
|
|
|
9
|
+
## 2.11.0 — 2026-09-14
|
|
10
|
+
|
|
11
|
+
A minor release: tool result text the model reads changed (`verify_syntax`
|
|
12
|
+
reports its rung) and a new nudge was added. Gate: extended suite at 5 runs, seed 20260914, gate PASS after a 6-run recheck (five cases missed once, all 5/6 or better on recheck; run-level 156/205 vs 165/208, p=0.48); multi-turn at 3 runs on a fresh server with 0 invalid runs, no 3/3 case lost, uc3-t7 gained (38/48 vs 33/44, p=0.80); smoke 10/10; stall probe clean; CI green on main and the tag.
|
|
13
|
+
|
|
14
|
+
- Claims need evidence. A finish that says it ran something, that it
|
|
15
|
+
works, that the buttons respond or that the tests pass, needs a run this
|
|
16
|
+
turn (run_code or run_command); reading the file back proves the bytes,
|
|
17
|
+
not the behaviour. The owner's 2026-09-13 calculator session made three
|
|
18
|
+
such claims after read_file, with nothing ever run and nothing wired.
|
|
19
|
+
One nudge names the claim and asks for a run or a plain statement that
|
|
20
|
+
it was not run; a second unbacked claim goes out with a dim "Nothing was
|
|
21
|
+
run this turn." under it. The tests-claim nudge is the same rule for one
|
|
22
|
+
phrase; this is the class.
|
|
23
|
+
- `verify_syntax` is a ladder that reports the rung it reached. A page is
|
|
24
|
+
parsed and cross-referenced: handlers the markup calls must be functions
|
|
25
|
+
a script defines, ids the scripts look up must be elements the markup
|
|
26
|
+
has, buttons must be wired to something, and a script after `</body>` is
|
|
27
|
+
noted (the calculator: no handler on any button, an id no element had, a
|
|
28
|
+
function nothing called). Python is parsed and cross-referenced for
|
|
29
|
+
names it reads but never binds (a star import turns that off). JSON and
|
|
30
|
+
PowerShell say "parsed". A file with no checker is `NOT CHECKED`, never
|
|
31
|
+
`OK: skipped`, and says so, so the claim above cannot rest on it.
|
|
32
|
+
- Two eval cases: `claims-1`, the calculator prompt graded on the page
|
|
33
|
+
being wired and the finish not claiming a run that never happened, and
|
|
34
|
+
`claims-2`, a Python edit-and-claim variant. The 2026-09-13 page is a
|
|
35
|
+
fixture.
|
|
36
|
+
- The eval runner waits for the inference slot after a client timeout.
|
|
37
|
+
An abandoned request keeps the server's one slot until it finishes; the
|
|
38
|
+
next run queued behind it, timed out too, and took a whole scenario
|
|
39
|
+
down as invalid (uc3, 2026-09-13, twice). After a timed-out run the
|
|
40
|
+
runner probes until the backend answers again, up to ten minutes, and
|
|
41
|
+
the timeout no longer counts toward the abort streak once the slot is
|
|
42
|
+
back.
|
|
43
|
+
|
|
44
|
+
## 2.10.0 — 2026-09-14
|
|
45
|
+
|
|
46
|
+
A minor release: the retry feedback the model reads changed. Gate: extended
|
|
47
|
+
suite at 5 runs, seed 20260914, 32/44 pass^5 (the 2.7.x baseline is 32/44), run-level
|
|
48
|
+
159/205 vs 165/208 (p=0.72); one gate case missed once and passed its 6-run
|
|
49
|
+
recheck; CI green on main and the tag.
|
|
50
|
+
|
|
51
|
+
- A reply whose first JSON object does not decode is never "finished" by a
|
|
52
|
+
later object in the same reply. The owner's 2026-09-13 session: asked for
|
|
53
|
+
a calculator page, the model answered with a `write_file` holding 1.6K of
|
|
54
|
+
HTML and, behind it, a `finish` saying the file was created. Fifteen
|
|
55
|
+
attribute quotes (`onclick=\"input('7')">`) were unescaped, the string
|
|
56
|
+
closed early, the object failed to decode, the parser moved on to the
|
|
57
|
+
next complete object, accepted the finish, and the turn claimed a file it
|
|
58
|
+
never wrote; the next turn found nothing to open. `parse_json_object` now
|
|
59
|
+
decodes the first object from the first brace with `raw_decode` (batched
|
|
60
|
+
actions still take the first and let the loop drive the rest), repairs a
|
|
61
|
+
stray quote inside a string value up to sixty-four times by escaping the
|
|
62
|
+
last unescaped quote before the decoder's error (a truncated string is
|
|
63
|
+
left alone), accepts raw control characters inside strings, and returns
|
|
64
|
+
nothing when the first object still fails, so the loop's existing retry
|
|
65
|
+
fires. That retry now tells the model the decoder's complaint, the
|
|
66
|
+
character offset, the text around it and the quoting rule instead of
|
|
67
|
+
"not valid JSON". The session's reply, verbatim, is a fixture: it decodes
|
|
68
|
+
to the write with all 1,466 characters of HTML.
|
|
69
|
+
- The `/` command menu sits above the input row, between the top rule and
|
|
70
|
+
the prompt. The box is pinned to the window's last rows, so the rows
|
|
71
|
+
2.9.1 added below the input pushed the input row and the caret up
|
|
72
|
+
whenever the menu appeared or changed height; rows above the input grow
|
|
73
|
+
the box upward and the caret stays where it is.
|
|
74
|
+
- `run_arm.cmd` stops when the suite exits non-zero. An aborted suite (six
|
|
75
|
+
consecutive backend timeouts on a starved machine, 2026-09-13) left the
|
|
76
|
+
previous arm's results file in place, and the script copied it as the
|
|
77
|
+
candidate and gated it, which read as a RECHECK verdict against stale
|
|
78
|
+
data. It now logs the abort and writes no candidate.
|
|
79
|
+
|
|
9
80
|
## 2.9.1 — 2026-09-13
|
|
10
81
|
|
|
11
82
|
A patch release: the input line and the status bar; nothing model-facing
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hexcli
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.11.0
|
|
4
4
|
Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
|
|
5
5
|
Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
|
|
6
6
|
Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
|
|
@@ -1193,6 +1193,9 @@ def _run_autopilot_turn(
|
|
|
1193
1193
|
_tests_requested = _asks_to_run_tests(query)
|
|
1194
1194
|
_tests_nudge_used = False
|
|
1195
1195
|
_run_targets: list[str] = []
|
|
1196
|
+
_ran_anything = False # any run_code / run_command this turn: the evidence a behaviour claim needs
|
|
1197
|
+
_claim_nudge_used = False
|
|
1198
|
+
_unbacked_claim = False
|
|
1196
1199
|
# The last test run's failure output (None once a run passes), so a
|
|
1197
1200
|
# finish right after a failing run can be sent back once more.
|
|
1198
1201
|
_last_test_failure: str | None = None
|
|
@@ -1289,9 +1292,12 @@ def _run_autopilot_turn(
|
|
|
1289
1292
|
"one JSON object. No prose."
|
|
1290
1293
|
)
|
|
1291
1294
|
else:
|
|
1295
|
+
detail = parsing.describe_json_error(raw)
|
|
1292
1296
|
feedback = (
|
|
1293
|
-
"Your response was not valid JSON
|
|
1294
|
-
"
|
|
1297
|
+
"Your response was not valid JSON"
|
|
1298
|
+
+ (f": {detail}. Inside a string value every double quote must be "
|
|
1299
|
+
"written as \\\". " if detail else ". ")
|
|
1300
|
+
+ "Respond with exactly one JSON object as specified. No prose."
|
|
1295
1301
|
)
|
|
1296
1302
|
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1297
1303
|
messages.append({"role": "user", "content": feedback})
|
|
@@ -1335,6 +1341,21 @@ def _run_autopilot_turn(
|
|
|
1335
1341
|
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1336
1342
|
messages.append({"role": "user", "content": _retest_nudge_text(_last_test_failure)})
|
|
1337
1343
|
continue
|
|
1344
|
+
# Claims need evidence. "Ran it successfully", "the buttons now
|
|
1345
|
+
# respond", "tests pass": a behaviour claim in the finish needs
|
|
1346
|
+
# a run this turn; reading a file back proves the bytes, not the
|
|
1347
|
+
# behaviour (the calculator session, 2026-09-13: three such
|
|
1348
|
+
# claims, nothing ever run, nothing wired). One nudge: run it, or
|
|
1349
|
+
# say plainly that it was not run. A second unbacked claim goes
|
|
1350
|
+
# out with a one-line notice under it, so the user knows.
|
|
1351
|
+
claim = _behaviour_claim(msg)
|
|
1352
|
+
if (claim and not _ran_anything and config.get("require_verification", True)):
|
|
1353
|
+
if not _claim_nudge_used:
|
|
1354
|
+
_claim_nudge_used = True
|
|
1355
|
+
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1356
|
+
messages.append({"role": "user", "content": _claim_nudge_text(claim)})
|
|
1357
|
+
continue
|
|
1358
|
+
_unbacked_claim = True
|
|
1338
1359
|
# Escalation trigger B — the verification nudge was ignored: the
|
|
1339
1360
|
# model finished a second time without checking its own mutation.
|
|
1340
1361
|
if (_unverified_mutation and _verify_nudge_used
|
|
@@ -1364,6 +1385,8 @@ def _run_autopilot_turn(
|
|
|
1364
1385
|
})
|
|
1365
1386
|
continue
|
|
1366
1387
|
result = msg or last_tool_output or "Done."
|
|
1388
|
+
if _unbacked_claim:
|
|
1389
|
+
ui.cprint(" Nothing was run this turn.", C.DIM)
|
|
1367
1390
|
memory.maybe_index_turn(config, query, tools_used, touched_paths, outcome="completed")
|
|
1368
1391
|
if session:
|
|
1369
1392
|
_record_undo_snapshots(session, _turn_snapshots)
|
|
@@ -1395,6 +1418,7 @@ def _run_autopilot_turn(
|
|
|
1395
1418
|
touched_paths.append(str(tool_path))
|
|
1396
1419
|
if tool_name in ("run_code", "run_command") and isinstance(action.get("args"), dict):
|
|
1397
1420
|
_run_targets.append(str(action["args"].get("path") or action["args"].get("command") or ""))
|
|
1421
|
+
_ran_anything = True
|
|
1398
1422
|
|
|
1399
1423
|
# Capture file state before first mutation so /undo can restore it.
|
|
1400
1424
|
if tool_name in {"edit_file", "write_file", "append_file"} and tool_path:
|
|
@@ -1682,6 +1706,30 @@ _RUN_TESTS_RE = re.compile(
|
|
|
1682
1706
|
)
|
|
1683
1707
|
|
|
1684
1708
|
|
|
1709
|
+
_BEHAVIOUR_CLAIM_RE = re.compile(
|
|
1710
|
+
r"\b(ran (?:it|the [a-z]+|successfully)(?: successfully)?|runs (?:correctly|fine|successfully|as expected)|"
|
|
1711
|
+
r"(?:it|this|that|everything|the [a-z]+) (?:now |all )?works\b|works (?:as expected|correctly|now|fine)|"
|
|
1712
|
+
r"(?:it|this|that|everything|the [a-z]+) is (?:now )?working(?: correctly| as expected)?|"
|
|
1713
|
+
r"(?:buttons?|it|the app|the page|the script) (?:now )?responds?|"
|
|
1714
|
+
r"calculates? correctly|opens? correctly|executed successfully|"
|
|
1715
|
+
r"tests? (?:pass|passed|passing|succeed)|all tests pass|verified (?:that )?it (?:works|runs)|"
|
|
1716
|
+
r"confirmed (?:that )?it (?:works|runs))\b", re.IGNORECASE)
|
|
1717
|
+
|
|
1718
|
+
|
|
1719
|
+
def _behaviour_claim(message: str) -> str:
|
|
1720
|
+
"""The first behaviour claim in a finish message, or ''. Deliberately a
|
|
1721
|
+
list of ways to say "I ran it and it works", not a list of verbs: "the
|
|
1722
|
+
function returns the sum" is a description, "it works" is a claim."""
|
|
1723
|
+
m = _BEHAVIOUR_CLAIM_RE.search(message or "")
|
|
1724
|
+
return m.group(0) if m else ""
|
|
1725
|
+
|
|
1726
|
+
|
|
1727
|
+
def _claim_nudge_text(claim: str) -> str:
|
|
1728
|
+
return (f"You said \"{claim}\" but nothing was run this turn. Run it with run_code or "
|
|
1729
|
+
"run_command and report what you observed, or say plainly that it was not run. "
|
|
1730
|
+
"Respond with JSON only.")
|
|
1731
|
+
|
|
1732
|
+
|
|
1685
1733
|
def _asks_to_run_tests(query: str) -> bool:
|
|
1686
1734
|
"""True when the request itself asks for the tests to be run."""
|
|
1687
1735
|
return bool(_RUN_TESTS_RE.search(query or ""))
|
|
@@ -629,6 +629,12 @@ class LineEditor:
|
|
|
629
629
|
logical: list[tuple[str, int]] = [] # (rendered text, visible width)
|
|
630
630
|
for line in above:
|
|
631
631
|
logical.append((line, visible_len(line)))
|
|
632
|
+
# The / menu sits between the top rule and the input row. The box is
|
|
633
|
+
# pinned to the window's last rows, so rows added BELOW the input
|
|
634
|
+
# pushed the input row and the caret up when the menu appeared; rows
|
|
635
|
+
# above it grow the box upward and the caret stays put.
|
|
636
|
+
menu = self._menu_rows() if chrome else []
|
|
637
|
+
logical.extend(menu)
|
|
632
638
|
for line in prompt_lines[:-1]:
|
|
633
639
|
logical.append((line, visible_len(line)))
|
|
634
640
|
last_prompt = prompt_lines[-1]
|
|
@@ -647,8 +653,6 @@ class LineEditor:
|
|
|
647
653
|
text, vis = logical[-1]
|
|
648
654
|
styled = f"\033[2m{ghost}\033[0m" if self.styled else ghost
|
|
649
655
|
logical[-1] = (text + styled, vis + len(ghost))
|
|
650
|
-
if chrome:
|
|
651
|
-
logical.extend(self._menu_rows())
|
|
652
656
|
for line in below:
|
|
653
657
|
logical.append((line, visible_len(line)))
|
|
654
658
|
|
|
@@ -656,7 +660,7 @@ class LineEditor:
|
|
|
656
660
|
before = self.buffer[:self.pos]
|
|
657
661
|
cur_line = before.count("\n")
|
|
658
662
|
col_in_line = len(before) - (before.rfind("\n") + 1)
|
|
659
|
-
cursor_logical = len(above) + len(prompt_lines) - 1 + cur_line
|
|
663
|
+
cursor_logical = len(above) + len(menu) + len(prompt_lines) - 1 + cur_line
|
|
660
664
|
cursor_vis = visible_len(prefixes[cur_line]) + col_in_line
|
|
661
665
|
|
|
662
666
|
pieces: list[str] = []
|
|
@@ -93,43 +93,92 @@ def strip_thinking(text: str) -> str:
|
|
|
93
93
|
# Parsing
|
|
94
94
|
# ---------------------------------------------------------------------------
|
|
95
95
|
|
|
96
|
+
_STRAY_QUOTE_MSGS = ("Expecting ',' delimiter", "Expecting ':' delimiter",
|
|
97
|
+
"Expecting property name")
|
|
98
|
+
_MAX_QUOTE_REPAIRS = 64
|
|
99
|
+
_DECODER = json.JSONDecoder(strict=False) # raw newlines inside a written file are fine
|
|
100
|
+
|
|
101
|
+
|
|
102
|
+
def _escaped_at(text: str, i: int) -> bool:
|
|
103
|
+
"""Is the character at i preceded by an odd run of backslashes?"""
|
|
104
|
+
n = 0
|
|
105
|
+
while i - 1 - n >= 0 and text[i - 1 - n] == "\\":
|
|
106
|
+
n += 1
|
|
107
|
+
return n % 2 == 1
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _repair_stray_quote(text: str, err: json.JSONDecodeError) -> str | None:
|
|
111
|
+
"""Escape the quote that ended a string early. A 4B writing 1.6K of HTML
|
|
112
|
+
inside a JSON string forgets the backslash on a closing attribute quote
|
|
113
|
+
(`onclick=\\"input('7')">`, sixteen times in one reply, 2026-09-13): the
|
|
114
|
+
string closes there and the decoder trips on the next token. The last
|
|
115
|
+
unescaped quote before the error is that closer; escape it and let the
|
|
116
|
+
caller decode again. A truncated string (\"Unterminated string\") is not
|
|
117
|
+
repairable and is left alone."""
|
|
118
|
+
if not any(err.msg.startswith(m) for m in _STRAY_QUOTE_MSGS):
|
|
119
|
+
return None
|
|
120
|
+
q = text.rfind('"', 0, err.pos)
|
|
121
|
+
while q > 0 and _escaped_at(text, q):
|
|
122
|
+
q = text.rfind('"', 0, q)
|
|
123
|
+
if q <= 0:
|
|
124
|
+
return None
|
|
125
|
+
return text[:q] + "\\" + text[q:]
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _loads_object(text: str) -> dict[str, Any] | None:
|
|
129
|
+
"""The first JSON object in `text`, decoded from its first brace and
|
|
130
|
+
ignoring whatever follows it (a second batched action, prose). Stray
|
|
131
|
+
quotes inside string values are repaired a bounded number of times."""
|
|
132
|
+
start = text.find("{")
|
|
133
|
+
if start < 0:
|
|
134
|
+
return None
|
|
135
|
+
for _ in range(_MAX_QUOTE_REPAIRS + 1):
|
|
136
|
+
try:
|
|
137
|
+
parsed, _end = _DECODER.raw_decode(text, start)
|
|
138
|
+
return parsed if isinstance(parsed, dict) else None
|
|
139
|
+
except json.JSONDecodeError as err:
|
|
140
|
+
fixed = _repair_stray_quote(text, err)
|
|
141
|
+
if fixed is None:
|
|
142
|
+
return None
|
|
143
|
+
text = fixed
|
|
144
|
+
return None
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def describe_json_error(raw_text: str) -> str:
|
|
148
|
+
"""What is wrong with the first JSON object in a reply, for the retry
|
|
149
|
+
feedback: the decoder's message, the character offset and the text
|
|
150
|
+
around it. Empty when the object decodes."""
|
|
151
|
+
text = strip_thinking(raw_text).strip()
|
|
152
|
+
start = max(0, text.find("{"))
|
|
153
|
+
try:
|
|
154
|
+
_DECODER.raw_decode(text, start)
|
|
155
|
+
return ""
|
|
156
|
+
except json.JSONDecodeError as err:
|
|
157
|
+
lo, hi = max(0, err.pos - 40), min(len(text), err.pos + 20)
|
|
158
|
+
return f"{err.msg} at character {err.pos - start}, near: {text[lo:hi]!r}"
|
|
159
|
+
|
|
160
|
+
|
|
96
161
|
def parse_json_object(raw_text: str) -> dict[str, Any] | None:
|
|
97
162
|
text = strip_thinking(raw_text).strip()
|
|
98
163
|
if not text:
|
|
99
164
|
return None
|
|
100
165
|
# Direct parse
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
return parsed
|
|
105
|
-
except json.JSONDecodeError:
|
|
106
|
-
pass
|
|
166
|
+
parsed = _loads_object(text)
|
|
167
|
+
if parsed is not None:
|
|
168
|
+
return parsed
|
|
107
169
|
# Strip markdown fences
|
|
108
170
|
stripped = re.sub(r"^```[a-zA-Z]*\s*|```\s*$", "", text, flags=re.MULTILINE).strip()
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
114
|
-
|
|
115
|
-
#
|
|
116
|
-
#
|
|
117
|
-
#
|
|
118
|
-
#
|
|
119
|
-
#
|
|
120
|
-
# — that match is not valid JSON, so the whole response was discarded, the
|
|
121
|
-
# identical retry was issued up to 3×, and the turn ended with no tool call
|
|
122
|
-
# at all. Measured 2026-07-30: this is what actually killed uc1-t4/t5/t6
|
|
123
|
-
# (0/3 each), NOT a context-length cliff. Batching is a natural response to
|
|
124
|
-
# rules that prescribe an edit→verify→run sequence, so take the first
|
|
125
|
-
# action and let the loop drive the rest.
|
|
126
|
-
for candidate in _iter_json_objects(text):
|
|
127
|
-
try:
|
|
128
|
-
parsed = json.loads(candidate)
|
|
129
|
-
if isinstance(parsed, dict):
|
|
130
|
-
return parsed
|
|
131
|
-
except json.JSONDecodeError:
|
|
132
|
-
continue
|
|
171
|
+
parsed = _loads_object(stripped)
|
|
172
|
+
if parsed is not None:
|
|
173
|
+
return parsed
|
|
174
|
+
# Nothing decodable from the first brace. Batched actions ({edit}{verify}
|
|
175
|
+
# {run}) are handled above: raw_decode takes the FIRST object and ignores
|
|
176
|
+
# the rest, and the loop drives the next step (measured 2026-07-30: the
|
|
177
|
+
# old greedy match discarded such replies and killed uc1-t4/t5/t6). A
|
|
178
|
+
# broken first object is a malformed reply, never skipped for a later
|
|
179
|
+
# one: on 2026-09-13 the finish behind an unparseable write_file was
|
|
180
|
+
# accepted and the turn claimed a file it had not written. The loop
|
|
181
|
+
# retries with the decoder's complaint (describe_json_error).
|
|
133
182
|
return None
|
|
134
183
|
|
|
135
184
|
|
|
@@ -581,22 +581,205 @@ def _verify_node_syntax(path: Path) -> tuple[bool, str]:
|
|
|
581
581
|
return False, f"FAIL: {result.stderr.strip() or result.stdout.strip()}"
|
|
582
582
|
|
|
583
583
|
|
|
584
|
+
_PY_IMPLICIT = {"__name__", "__file__", "__doc__", "__builtins__", "__spec__", "__loader__",
|
|
585
|
+
"__package__", "__path__", "__annotations__", "__debug__"}
|
|
586
|
+
|
|
587
|
+
|
|
588
|
+
def undefined_python_names(source: str) -> list[str]:
|
|
589
|
+
"""Names the module reads but never binds anywhere and Python does not
|
|
590
|
+
provide: the cross-reference rung for Python. Scopes are collapsed to
|
|
591
|
+
the module (a name bound anywhere counts), so this never flags a name
|
|
592
|
+
that some function defines; it only catches the calculator-class
|
|
593
|
+
mistake of calling something that does not exist. A star import turns
|
|
594
|
+
the check off."""
|
|
595
|
+
import builtins
|
|
596
|
+
try:
|
|
597
|
+
tree = ast.parse(source)
|
|
598
|
+
except SyntaxError:
|
|
599
|
+
return []
|
|
600
|
+
bound: set[str] = set(dir(builtins)) | _PY_IMPLICIT
|
|
601
|
+
loads: dict[str, int] = {}
|
|
602
|
+
for node in ast.walk(tree):
|
|
603
|
+
if isinstance(node, ast.ImportFrom) and any(a.name == "*" for a in node.names):
|
|
604
|
+
return []
|
|
605
|
+
if isinstance(node, (ast.Import, ast.ImportFrom)):
|
|
606
|
+
for a in node.names:
|
|
607
|
+
bound.add((a.asname or a.name).split(".")[0])
|
|
608
|
+
elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef, ast.ClassDef)):
|
|
609
|
+
bound.add(node.name)
|
|
610
|
+
if not isinstance(node, ast.ClassDef):
|
|
611
|
+
args = node.args
|
|
612
|
+
for a in [*args.posonlyargs, *args.args, *args.kwonlyargs]:
|
|
613
|
+
bound.add(a.arg)
|
|
614
|
+
for a in (args.vararg, args.kwarg):
|
|
615
|
+
if a is not None:
|
|
616
|
+
bound.add(a.arg)
|
|
617
|
+
elif isinstance(node, ast.Lambda):
|
|
618
|
+
args = node.args
|
|
619
|
+
for a in [*args.posonlyargs, *args.args, *args.kwonlyargs]:
|
|
620
|
+
bound.add(a.arg)
|
|
621
|
+
for a in (args.vararg, args.kwarg):
|
|
622
|
+
if a is not None:
|
|
623
|
+
bound.add(a.arg)
|
|
624
|
+
elif isinstance(node, ast.ExceptHandler) and node.name:
|
|
625
|
+
bound.add(node.name)
|
|
626
|
+
elif isinstance(node, (ast.Global, ast.Nonlocal)):
|
|
627
|
+
bound.update(node.names)
|
|
628
|
+
elif isinstance(node, ast.MatchAs) and node.name:
|
|
629
|
+
bound.add(node.name)
|
|
630
|
+
elif isinstance(node, ast.MatchStar) and node.name:
|
|
631
|
+
bound.add(node.name)
|
|
632
|
+
elif isinstance(node, ast.MatchMapping) and node.rest:
|
|
633
|
+
bound.add(node.rest)
|
|
634
|
+
elif isinstance(node, ast.Name):
|
|
635
|
+
if isinstance(node.ctx, (ast.Store, ast.Del)):
|
|
636
|
+
bound.add(node.id)
|
|
637
|
+
else:
|
|
638
|
+
loads.setdefault(node.id, node.lineno)
|
|
639
|
+
return [f"{name} (line {line})" for name, line in sorted(loads.items(), key=lambda kv: kv[1])
|
|
640
|
+
if name not in bound]
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
_JS_BUILTIN_CALLS = {"alert", "confirm", "prompt", "eval", "parseInt", "parseFloat", "String",
|
|
644
|
+
"Number", "Boolean", "Math", "setTimeout", "setInterval", "console",
|
|
645
|
+
"document", "window", "event", "this", "if", "for", "while", "return",
|
|
646
|
+
"function", "new", "Array", "Object", "JSON", "isNaN", "Date", "fetch",
|
|
647
|
+
"requestAnimationFrame", "clearTimeout", "clearInterval", "Promise"}
|
|
648
|
+
|
|
649
|
+
|
|
650
|
+
def html_wiring_report(path: Path) -> tuple[bool, str]:
|
|
651
|
+
"""Cross-reference rung for a page: the handlers the markup calls must be
|
|
652
|
+
functions the scripts define, the ids the scripts look up must be
|
|
653
|
+
elements the markup has, and buttons must be wired to something. Reading
|
|
654
|
+
a page back proves the bytes; this proves the parts are connected
|
|
655
|
+
(2026-09-14: a calculator with buttons that called nothing, a script that
|
|
656
|
+
looked up an id no element had, and a function nothing called was
|
|
657
|
+
"verified" three times by read_file)."""
|
|
658
|
+
from html.parser import HTMLParser
|
|
659
|
+
|
|
660
|
+
class _Walk(HTMLParser):
|
|
661
|
+
def __init__(self) -> None:
|
|
662
|
+
super().__init__()
|
|
663
|
+
self.ids: set[str] = set()
|
|
664
|
+
self.handlers: list[str] = []
|
|
665
|
+
self.buttons = 0
|
|
666
|
+
self.scripts: list[str] = []
|
|
667
|
+
self.script_srcs: list[str] = []
|
|
668
|
+
self._in_script = False
|
|
669
|
+
self._body_closed = False
|
|
670
|
+
self.script_after_body = False
|
|
671
|
+
self._buf: list[str] = []
|
|
672
|
+
|
|
673
|
+
def handle_starttag(self, tag: str, attrs: list[tuple[str, str | None]]) -> None:
|
|
674
|
+
a = {k.lower(): (v or "") for k, v in attrs}
|
|
675
|
+
if a.get("id"):
|
|
676
|
+
self.ids.add(a["id"])
|
|
677
|
+
for k, v in a.items():
|
|
678
|
+
if k.startswith("on") and v:
|
|
679
|
+
self.handlers.extend(m for m in re.findall(r"([A-Za-z_$][\w$]*)\s*\(", v))
|
|
680
|
+
if tag == "button" or (tag == "input" and a.get("type", "").lower() in {"button", "submit"}):
|
|
681
|
+
self.buttons += 1
|
|
682
|
+
if tag == "script":
|
|
683
|
+
self._in_script = True
|
|
684
|
+
self._buf = []
|
|
685
|
+
if a.get("src"):
|
|
686
|
+
self.script_srcs.append(a["src"])
|
|
687
|
+
if self._body_closed:
|
|
688
|
+
self.script_after_body = True
|
|
689
|
+
|
|
690
|
+
def handle_endtag(self, tag: str) -> None:
|
|
691
|
+
if tag == "script" and self._in_script:
|
|
692
|
+
self._in_script = False
|
|
693
|
+
self.scripts.append("".join(self._buf))
|
|
694
|
+
if tag in {"body", "html"}:
|
|
695
|
+
self._body_closed = True
|
|
696
|
+
|
|
697
|
+
def handle_data(self, data: str) -> None:
|
|
698
|
+
if self._in_script:
|
|
699
|
+
self._buf.append(data)
|
|
700
|
+
|
|
701
|
+
source = path.read_text(encoding="utf-8", errors="replace")
|
|
702
|
+
w = _Walk()
|
|
703
|
+
try:
|
|
704
|
+
w.feed(source)
|
|
705
|
+
w.close()
|
|
706
|
+
except Exception as exc: # noqa: BLE001 — html.parser is lenient; anything else is a report
|
|
707
|
+
return False, f"FAIL: could not parse the page: {exc}"
|
|
708
|
+
script = "\n".join(w.scripts)
|
|
709
|
+
for src in w.script_srcs:
|
|
710
|
+
try:
|
|
711
|
+
script += "\n" + (path.parent / src).read_text(encoding="utf-8", errors="replace")
|
|
712
|
+
except OSError:
|
|
713
|
+
pass
|
|
714
|
+
defined = set(re.findall(r"\bfunction\s+([A-Za-z_$][\w$]*)\s*\(", script))
|
|
715
|
+
defined |= set(re.findall(r"\b(?:const|let|var)\s+([A-Za-z_$][\w$]*)\s*=\s*(?:async\s+)?(?:function\b|\(|[A-Za-z_$][\w$]*\s*=>)", script))
|
|
716
|
+
defined |= set(re.findall(r"\b(?:window\.)?([A-Za-z_$][\w$]*)\s*=\s*(?:async\s+)?function\b", script))
|
|
717
|
+
looked_up = set(re.findall(r"getElementById\(\s*['\"]([^'\"]+)['\"]", script))
|
|
718
|
+
looked_up |= set(re.findall(r"querySelector(?:All)?\(\s*['\"]#([\w-]+)", script))
|
|
719
|
+
listeners = bool(re.search(r"addEventListener\s*\(|\.on[a-z]+\s*=", script))
|
|
720
|
+
problems: list[str] = []
|
|
721
|
+
undefined = sorted({h for h in w.handlers if h not in defined and h not in _JS_BUILTIN_CALLS})
|
|
722
|
+
if undefined:
|
|
723
|
+
problems.append("handlers the markup calls but no script defines: " + ", ".join(undefined))
|
|
724
|
+
missing = sorted(i for i in looked_up if i not in w.ids)
|
|
725
|
+
if missing:
|
|
726
|
+
problems.append("ids the script looks up but no element has: " + ", ".join(missing))
|
|
727
|
+
if w.buttons and not w.handlers and not listeners:
|
|
728
|
+
problems.append(f"{w.buttons} button(s) and none is wired to a script (no on* attribute, no addEventListener)")
|
|
729
|
+
notes: list[str] = []
|
|
730
|
+
if w.script_after_body:
|
|
731
|
+
notes.append("a <script> sits after </body>")
|
|
732
|
+
if defined and not w.handlers and not listeners:
|
|
733
|
+
notes.append("functions defined but nothing calls them: " + ", ".join(sorted(defined)))
|
|
734
|
+
if problems:
|
|
735
|
+
return False, "FAIL: parsed, cross-referenced: " + "; ".join(problems + notes)
|
|
736
|
+
summary = f"OK: parsed, cross-referenced ({len(w.handlers)} handler call(s), {len(w.ids)} id(s), {w.buttons} button(s))"
|
|
737
|
+
if notes:
|
|
738
|
+
summary += "; " + "; ".join(notes)
|
|
739
|
+
return True, summary
|
|
740
|
+
|
|
741
|
+
|
|
584
742
|
def verify_syntax_tool(path_text: str, language: str, shell_exe: str) -> str:
|
|
743
|
+
"""The verification ladder. Every result says which rung was reached:
|
|
744
|
+
executed (not here: run_code does that), parsed and cross-referenced,
|
|
745
|
+
parsed only, or NOT CHECKED. A file this tool cannot check is reported
|
|
746
|
+
as unchecked, never as OK: the model used to read "OK: skipped" and
|
|
747
|
+
finish with "verified"."""
|
|
585
748
|
path = resolve_path(path_text)
|
|
586
749
|
if not path.exists():
|
|
587
750
|
raise RuntimeError(f"File not found: {path}")
|
|
588
|
-
|
|
589
|
-
|
|
751
|
+
suffix = path.suffix.lower()
|
|
752
|
+
lang = (language or "").strip().lower() or _LANGUAGE_BY_EXT.get(suffix, "")
|
|
753
|
+
if suffix in {".html", ".htm"} or lang in {"html", "htm"}:
|
|
754
|
+
ok, detail = html_wiring_report(path)
|
|
755
|
+
elif lang == "python":
|
|
590
756
|
ok, detail = _verify_python_syntax(path)
|
|
757
|
+
if ok and not detail.startswith("OK: skipped"):
|
|
758
|
+
undefined = undefined_python_names(path.read_text(encoding="utf-8", errors="replace"))
|
|
759
|
+
if undefined:
|
|
760
|
+
ok, detail = False, "FAIL: parsed, cross-referenced: references undefined name(s): " + ", ".join(undefined)
|
|
761
|
+
else:
|
|
762
|
+
detail = "OK: parsed, cross-referenced: no syntax errors, every name it reads is defined"
|
|
591
763
|
elif lang == "json":
|
|
592
764
|
ok, detail = _verify_json_syntax(path)
|
|
765
|
+
if ok:
|
|
766
|
+
detail = detail.replace("OK: valid JSON", "OK: parsed: valid JSON")
|
|
593
767
|
elif lang == "powershell":
|
|
594
768
|
ok, detail = _verify_powershell_syntax(path, shell_exe)
|
|
595
|
-
|
|
769
|
+
if ok and detail.startswith("OK: skipped"):
|
|
770
|
+
ok, detail = True, "NOT CHECKED: " + detail[len("OK: skipped ("):].rstrip(")")
|
|
771
|
+
elif ok:
|
|
772
|
+
detail = "OK: parsed: no syntax errors"
|
|
773
|
+
elif lang == "node" or suffix in {".js", ".mjs", ".cjs", ".ts", ".tsx", ".jsx"}:
|
|
596
774
|
ok, detail = _verify_node_syntax(path)
|
|
775
|
+
if ok and detail.startswith("OK: skipped"):
|
|
776
|
+
ok, detail = True, "NOT CHECKED: " + detail[len("OK: skipped ("):].rstrip(")")
|
|
777
|
+
elif ok:
|
|
778
|
+
detail = "OK: parsed: no syntax errors"
|
|
597
779
|
else:
|
|
598
|
-
ok, detail = True, f"
|
|
599
|
-
|
|
780
|
+
ok, detail = True, (f"NOT CHECKED: no checker for '{suffix or language or 'this file'}'. "
|
|
781
|
+
"The file was not verified; say so if you report on it.")
|
|
782
|
+
ui.tool_event("verify", f"{path} ({'fail' if not ok else ('unchecked' if detail.startswith('NOT CHECKED') else 'pass')})")
|
|
600
783
|
return detail
|
|
601
784
|
|
|
602
785
|
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|