hexcli 2.10.0__tar.gz → 2.11.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hexcli-2.10.0 → hexcli-2.11.1}/.gitignore +1 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/CHANGELOG.md +62 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/PKG-INFO +2 -2
- {hexcli-2.10.0 → hexcli-2.11.1}/README.md +1 -1
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/__init__.py +1 -1
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/agent.py +45 -93
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/compaction.py +1 -1
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/config.py +0 -29
- hexcli-2.11.1/hexcli/editing.py +456 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/llm.py +1 -1
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/memory.py +1 -70
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/parsing.py +0 -32
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/repl.py +1 -8
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/tools.py +191 -8
- {hexcli-2.10.0 → hexcli-2.11.1}/pyproject.toml +0 -2
- {hexcli-2.10.0 → hexcli-2.11.1}/shellai.example.json +1 -9
- hexcli-2.10.0/hexcli/escalate.py +0 -192
- hexcli-2.10.0/hexcli/local_escalation.py +0 -191
- hexcli-2.10.0/hexcli/loop_v2.py +0 -396
- hexcli-2.10.0/hexcli/protocol_v2.py +0 -819
- hexcli-2.10.0/hexcli/shell_session.py +0 -186
- hexcli-2.10.0/shellai.cmd +0 -2
- hexcli-2.10.0/shellai.py +0 -15
- {hexcli-2.10.0 → hexcli-2.11.1}/Hex CLI.cmd +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/LICENSE +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/assets/hexcli.ico +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/assets/hexcli.png +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/cancel.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/chatlog.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/commands.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/diffview.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/distribution.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/doctor.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/http_client.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/launcher.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/lineedit.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/lockfile.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/markdown_stream.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/network.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/paths.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/prompts.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/safety.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/sessions.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/setup_wizard.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/statusbar.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/stream_render.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/telemetry.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/hexcli/ui.py +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/install.ps1 +0 -0
- {hexcli-2.10.0 → hexcli-2.11.1}/launcher.py +0 -0
|
@@ -6,6 +6,68 @@ the Hexagon NPU, not single-run anecdotes.
|
|
|
6
6
|
|
|
7
7
|
## Unreleased
|
|
8
8
|
|
|
9
|
+
## 2.11.1 — 2026-09-14
|
|
10
|
+
|
|
11
|
+
A patch release: nothing model-facing and nothing the launcher hands the
|
|
12
|
+
server; code and documents nobody used are gone. Gate: every remaining
|
|
13
|
+
suite green (27 suites, 739 tests; 29 suites and 798 tests before), smoke on a fresh server
|
|
14
|
+
9/10 then 10/10 (the miss was factual-1, a no-tool knowledge answer that
|
|
15
|
+
missed once in every arm today), CI green on main and the tag.
|
|
16
|
+
|
|
17
|
+
- The prune. Nothing here was used: protocol v2 (`loop_v2.py`,
|
|
18
|
+
`shell_session.py`, the v2 parser and prompt in `protocol_v2.py`, its
|
|
19
|
+
suite, the `protocol` config key), which lost its A/B at 13/36 vs 22/35
|
|
20
|
+
and doubled every safety and file-tool change; the local escalation
|
|
21
|
+
ladder (no viable bigger model on this hardware) and the cloud
|
|
22
|
+
escalation path (never configured), with their six config keys, so "no
|
|
23
|
+
code leaves the machine" is now structural rather than a default; the
|
|
24
|
+
memory dreaming daemon, off since it fabricated hardware facts; the
|
|
25
|
+
unused brace scanner in the parser; the root `shellai.py` / `shellai.cmd`
|
|
26
|
+
shims. The SEARCH/REPLACE applier that `edit_file` uses moved out of
|
|
27
|
+
`protocol_v2.py` into `hexcli/editing.py` unchanged, with its tests in
|
|
28
|
+
`evals/test_editing.py`. Internal documents (the V2 plan and roadmap, the
|
|
29
|
+
levers memo, the backend study, ARCHITECTURE.md) and the study-only bench
|
|
30
|
+
probes are no longer tracked; they live in git-ignored `docs/local/` and
|
|
31
|
+
`tools/local/` (owner rule: only the paper and user-facing docs are
|
|
32
|
+
committed). `tools/backend_bench/` keeps `stall_rate.py` and its two
|
|
33
|
+
imports, which the release gate runs. About 3,300 lines and three suites
|
|
34
|
+
gone; every remaining suite green.
|
|
35
|
+
|
|
36
|
+
## 2.11.0 — 2026-09-14
|
|
37
|
+
|
|
38
|
+
A minor release: tool result text the model reads changed (`verify_syntax`
|
|
39
|
+
reports its rung) and a new nudge was added. Gate: extended suite at 5 runs, seed 20260914, gate PASS after a 6-run recheck (five cases missed once, all 5/6 or better on recheck; run-level 156/205 vs 165/208, p=0.48); multi-turn at 3 runs on a fresh server with 0 invalid runs, no 3/3 case lost, uc3-t7 gained (38/48 vs 33/44, p=0.80); smoke 10/10; stall probe clean; CI green on main and the tag.
|
|
40
|
+
|
|
41
|
+
- Claims need evidence. A finish that says it ran something, that it
|
|
42
|
+
works, that the buttons respond or that the tests pass, needs a run this
|
|
43
|
+
turn (run_code or run_command); reading the file back proves the bytes,
|
|
44
|
+
not the behaviour. The owner's 2026-09-13 calculator session made three
|
|
45
|
+
such claims after read_file, with nothing ever run and nothing wired.
|
|
46
|
+
One nudge names the claim and asks for a run or a plain statement that
|
|
47
|
+
it was not run; a second unbacked claim goes out with a dim "Nothing was
|
|
48
|
+
run this turn." under it. The tests-claim nudge is the same rule for one
|
|
49
|
+
phrase; this is the class.
|
|
50
|
+
- `verify_syntax` is a ladder that reports the rung it reached. A page is
|
|
51
|
+
parsed and cross-referenced: handlers the markup calls must be functions
|
|
52
|
+
a script defines, ids the scripts look up must be elements the markup
|
|
53
|
+
has, buttons must be wired to something, and a script after `</body>` is
|
|
54
|
+
noted (the calculator: no handler on any button, an id no element had, a
|
|
55
|
+
function nothing called). Python is parsed and cross-referenced for
|
|
56
|
+
names it reads but never binds (a star import turns that off). JSON and
|
|
57
|
+
PowerShell say "parsed". A file with no checker is `NOT CHECKED`, never
|
|
58
|
+
`OK: skipped`, and says so, so the claim above cannot rest on it.
|
|
59
|
+
- Two eval cases: `claims-1`, the calculator prompt graded on the page
|
|
60
|
+
being wired and the finish not claiming a run that never happened, and
|
|
61
|
+
`claims-2`, a Python edit-and-claim variant. The 2026-09-13 page is a
|
|
62
|
+
fixture.
|
|
63
|
+
- The eval runner waits for the inference slot after a client timeout.
|
|
64
|
+
An abandoned request keeps the server's one slot until it finishes; the
|
|
65
|
+
next run queued behind it, timed out too, and took a whole scenario
|
|
66
|
+
down as invalid (uc3, 2026-09-13, twice). After a timed-out run the
|
|
67
|
+
runner probes until the backend answers again, up to ten minutes, and
|
|
68
|
+
the timeout no longer counts toward the abort streak once the slot is
|
|
69
|
+
back.
|
|
70
|
+
|
|
9
71
|
## 2.10.0 — 2026-09-14
|
|
10
72
|
|
|
11
73
|
A minor release: the retry feedback the model reads changed. Gate: extended
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hexcli
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.11.1
|
|
4
4
|
Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
|
|
5
5
|
Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
|
|
6
6
|
Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
|
|
@@ -353,7 +353,7 @@ Restart the NPU server before each suite. After an hour or two of steady
|
|
|
353
353
|
use it starts returning errors for everything, which looks like a model
|
|
354
354
|
regression. The runner detects this and marks those runs invalid.
|
|
355
355
|
|
|
356
|
-
`
|
|
356
|
+
`CLAUDE.md` describes the module layout. The paper in `docs/paper/` has the
|
|
357
357
|
hardware measurements, the eval method, and the reasoning behind each
|
|
358
358
|
safety layer. `RELEASING.md` covers how a release is cut.
|
|
359
359
|
|
|
@@ -326,7 +326,7 @@ Restart the NPU server before each suite. After an hour or two of steady
|
|
|
326
326
|
use it starts returning errors for everything, which looks like a model
|
|
327
327
|
regression. The runner detects this and marks those runs invalid.
|
|
328
328
|
|
|
329
|
-
`
|
|
329
|
+
`CLAUDE.md` describes the module layout. The paper in `docs/paper/` has the
|
|
330
330
|
hardware measurements, the eval method, and the reasoning behind each
|
|
331
331
|
safety layer. `RELEASING.md` covers how a release is cut.
|
|
332
332
|
|
|
@@ -28,10 +28,8 @@ from hexcli import (
|
|
|
28
28
|
compaction,
|
|
29
29
|
diffview,
|
|
30
30
|
distribution,
|
|
31
|
-
escalate,
|
|
32
31
|
http_client,
|
|
33
32
|
llm,
|
|
34
|
-
local_escalation,
|
|
35
33
|
lockfile,
|
|
36
34
|
memory,
|
|
37
35
|
network,
|
|
@@ -381,7 +379,6 @@ is_small_talk = parsing.is_small_talk
|
|
|
381
379
|
local_meta_response = parsing.local_meta_response
|
|
382
380
|
strip_thinking = parsing.strip_thinking
|
|
383
381
|
parse_json_object = parsing.parse_json_object
|
|
384
|
-
_iter_json_objects = parsing._iter_json_objects
|
|
385
382
|
parse_agent_action = parsing.parse_agent_action
|
|
386
383
|
_looks_like_botched_action = parsing._looks_like_botched_action
|
|
387
384
|
|
|
@@ -444,23 +441,6 @@ def set_active_config(config: dict[str, Any] | None) -> None:
|
|
|
444
441
|
|
|
445
442
|
|
|
446
443
|
|
|
447
|
-
# One escalation server per (model, bind) for the process lifetime — spawning
|
|
448
|
-
# a fresh 4.6 GB bundle load per consult would make escalation useless.
|
|
449
|
-
_ESCALATORS: dict[str, local_escalation.LocalEscalator] = {}
|
|
450
|
-
|
|
451
|
-
|
|
452
|
-
def _get_escalator(config: dict[str, Any]) -> local_escalation.LocalEscalator | None:
|
|
453
|
-
model = str(config.get("escalation_local_model", "") or "")
|
|
454
|
-
if not model:
|
|
455
|
-
return None
|
|
456
|
-
key = f"{model}@{config.get('escalation_local_bind', '127.0.0.1:11436')}"
|
|
457
|
-
esc = _ESCALATORS.get(key)
|
|
458
|
-
if esc is None:
|
|
459
|
-
esc = local_escalation.LocalEscalator(config)
|
|
460
|
-
_ESCALATORS[key] = esc
|
|
461
|
-
return esc if esc.enabled else None
|
|
462
|
-
|
|
463
|
-
|
|
464
444
|
class AutopilotProbe:
|
|
465
445
|
"""Optional instrumentation hook for run_autopilot, used by evals/ to
|
|
466
446
|
observe the production agent loop without reimplementing it. Every
|
|
@@ -1120,14 +1100,6 @@ def _run_autopilot_turn(
|
|
|
1120
1100
|
|
|
1121
1101
|
_sync_context_window(config)
|
|
1122
1102
|
|
|
1123
|
-
if str(config.get("protocol", "v1")).lower() == "v2":
|
|
1124
|
-
from . import loop_v2
|
|
1125
|
-
_CURRENT_SESSION_ID = str(uuid4())
|
|
1126
|
-
return loop_v2.run(
|
|
1127
|
-
config, history, query, shell_exe,
|
|
1128
|
-
session=session, turn=turn, probe=probe,
|
|
1129
|
-
)
|
|
1130
|
-
|
|
1131
1103
|
# Fresh UUID for this agent loop: lets the npurun server detect
|
|
1132
1104
|
# continuation turns (messages only appended) and skip reset_dialog(),
|
|
1133
1105
|
# so Genie re-prefills only the new tokens via SentenceCode::Rewind.
|
|
@@ -1193,6 +1165,9 @@ def _run_autopilot_turn(
|
|
|
1193
1165
|
_tests_requested = _asks_to_run_tests(query)
|
|
1194
1166
|
_tests_nudge_used = False
|
|
1195
1167
|
_run_targets: list[str] = []
|
|
1168
|
+
_ran_anything = False # any run_code / run_command this turn: the evidence a behaviour claim needs
|
|
1169
|
+
_claim_nudge_used = False
|
|
1170
|
+
_unbacked_claim = False
|
|
1196
1171
|
# The last test run's failure output (None once a run passes), so a
|
|
1197
1172
|
# finish right after a failing run can be sent back once more.
|
|
1198
1173
|
_last_test_failure: str | None = None
|
|
@@ -1200,33 +1175,6 @@ def _run_autopilot_turn(
|
|
|
1200
1175
|
# Local escalation (docs/V2_PLAN.md §4 ladder): consult the bigger local
|
|
1201
1176
|
# model at hard moments. At most one consult per turn; every failure path
|
|
1202
1177
|
# degrades to the pre-escalation behaviour.
|
|
1203
|
-
_escalator = _get_escalator(config)
|
|
1204
|
-
_escalation_used = False
|
|
1205
|
-
_turn_events: list[str] = []
|
|
1206
|
-
|
|
1207
|
-
def _consult_and_inject(problem: str, raw_response: str) -> bool:
|
|
1208
|
-
"""Ask the local escalation model for advice and inject it as the next
|
|
1209
|
-
user message. Returns True when advice was injected."""
|
|
1210
|
-
nonlocal _escalation_used
|
|
1211
|
-
if _escalator is None or _escalation_used:
|
|
1212
|
-
return False
|
|
1213
|
-
cprint("\n Consulting the escalation model.", C.DIM, file=sys.stderr)
|
|
1214
|
-
advice = _escalator.consult(
|
|
1215
|
-
local_escalation.build_situation(query, _turn_events, problem))
|
|
1216
|
-
if not advice:
|
|
1217
|
-
return False
|
|
1218
|
-
_escalation_used = True
|
|
1219
|
-
if raw_response:
|
|
1220
|
-
messages.append({"role": "assistant", "content": strip_thinking(raw_response)})
|
|
1221
|
-
messages.append({
|
|
1222
|
-
"role": "user",
|
|
1223
|
-
"content": (
|
|
1224
|
-
"A senior engineer reviewed the situation and advises:\n"
|
|
1225
|
-
f"{advice}\n"
|
|
1226
|
-
"Apply this advice now using the tools. Respond with JSON only."
|
|
1227
|
-
),
|
|
1228
|
-
})
|
|
1229
|
-
return True
|
|
1230
1178
|
|
|
1231
1179
|
for step in range(max_steps):
|
|
1232
1180
|
step_label = "thinking" if step == 0 else f"step {step + 1}/{max_steps}"
|
|
@@ -1338,25 +1286,21 @@ def _run_autopilot_turn(
|
|
|
1338
1286
|
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1339
1287
|
messages.append({"role": "user", "content": _retest_nudge_text(_last_test_failure)})
|
|
1340
1288
|
continue
|
|
1341
|
-
#
|
|
1342
|
-
#
|
|
1343
|
-
|
|
1344
|
-
|
|
1345
|
-
|
|
1346
|
-
|
|
1347
|
-
|
|
1348
|
-
|
|
1349
|
-
|
|
1350
|
-
|
|
1351
|
-
|
|
1352
|
-
|
|
1353
|
-
|
|
1354
|
-
|
|
1355
|
-
|
|
1356
|
-
and _consult_and_inject(
|
|
1357
|
-
"The task asks for a file change, but the agent is finishing "
|
|
1358
|
-
f"without having modified any file. Its answer was: {msg[:300]}", raw)):
|
|
1359
|
-
continue
|
|
1289
|
+
# Claims need evidence. "Ran it successfully", "the buttons now
|
|
1290
|
+
# respond", "tests pass": a behaviour claim in the finish needs
|
|
1291
|
+
# a run this turn; reading a file back proves the bytes, not the
|
|
1292
|
+
# behaviour (the calculator session, 2026-09-13: three such
|
|
1293
|
+
# claims, nothing ever run, nothing wired). One nudge: run it, or
|
|
1294
|
+
# say plainly that it was not run. A second unbacked claim goes
|
|
1295
|
+
# out with a one-line notice under it, so the user knows.
|
|
1296
|
+
claim = _behaviour_claim(msg)
|
|
1297
|
+
if (claim and not _ran_anything and config.get("require_verification", True)):
|
|
1298
|
+
if not _claim_nudge_used:
|
|
1299
|
+
_claim_nudge_used = True
|
|
1300
|
+
messages.append({"role": "assistant", "content": strip_thinking(raw)})
|
|
1301
|
+
messages.append({"role": "user", "content": _claim_nudge_text(claim)})
|
|
1302
|
+
continue
|
|
1303
|
+
_unbacked_claim = True
|
|
1360
1304
|
# Nudge once if the model refused to use tools
|
|
1361
1305
|
if (step == 0 and not direct_stage
|
|
1362
1306
|
and any(phrase in msg.lower() for phrase in REFUSAL_PHRASES)):
|
|
@@ -1367,6 +1311,8 @@ def _run_autopilot_turn(
|
|
|
1367
1311
|
})
|
|
1368
1312
|
continue
|
|
1369
1313
|
result = msg or last_tool_output or "Done."
|
|
1314
|
+
if _unbacked_claim:
|
|
1315
|
+
ui.cprint(" Nothing was run this turn.", C.DIM)
|
|
1370
1316
|
memory.maybe_index_turn(config, query, tools_used, touched_paths, outcome="completed")
|
|
1371
1317
|
if session:
|
|
1372
1318
|
_record_undo_snapshots(session, _turn_snapshots)
|
|
@@ -1398,6 +1344,7 @@ def _run_autopilot_turn(
|
|
|
1398
1344
|
touched_paths.append(str(tool_path))
|
|
1399
1345
|
if tool_name in ("run_code", "run_command") and isinstance(action.get("args"), dict):
|
|
1400
1346
|
_run_targets.append(str(action["args"].get("path") or action["args"].get("command") or ""))
|
|
1347
|
+
_ran_anything = True
|
|
1401
1348
|
|
|
1402
1349
|
# Capture file state before first mutation so /undo can restore it.
|
|
1403
1350
|
if tool_name in {"edit_file", "write_file", "append_file"} and tool_path:
|
|
@@ -1458,7 +1405,6 @@ def _run_autopilot_turn(
|
|
|
1458
1405
|
pass
|
|
1459
1406
|
|
|
1460
1407
|
last_tool_output = tool_output
|
|
1461
|
-
_turn_events.append(f"{tool_name}: {tool_output[:220]}")
|
|
1462
1408
|
if _run_targets and tool_name in ("run_code", "run_command") and _ran_tests(_run_targets[-1:]):
|
|
1463
1409
|
_last_test_failure = _test_failure_tail(tool_output)
|
|
1464
1410
|
_is_error = tool_output.lstrip().startswith("Error:")
|
|
@@ -1483,28 +1429,10 @@ def _run_autopilot_turn(
|
|
|
1483
1429
|
and all(err for _t, _tgt, err, _out in _loop_tracker)
|
|
1484
1430
|
and len({(t, tgt) for t, tgt, _e, _out in _loop_tracker}) == 1)
|
|
1485
1431
|
if _identical_trip or _failure_trip:
|
|
1486
|
-
# Escalation trigger A — the loop detector: consult the local
|
|
1487
|
-
# model BEFORE giving up (the cloud path stays as the fallback).
|
|
1488
|
-
if _consult_and_inject(
|
|
1489
|
-
f"The agent repeated the same failing call 3 times: {tool_name} "
|
|
1490
|
-
f"kept returning:\n{tool_output[:400]}", raw):
|
|
1491
|
-
_loop_tracker.clear()
|
|
1492
|
-
continue
|
|
1493
1432
|
what = f"{tool_name} returned the same result" if _identical_trip else f"{tool_name} failed"
|
|
1494
1433
|
if not _in_delegate: # a sub-agent's stop is its own result, not the turn's
|
|
1495
1434
|
cprint(f"\n ⚠ Stopped: {what} three times in a row.", C.BYELLOW)
|
|
1496
1435
|
_mark_turn_stopped("loop")
|
|
1497
|
-
if escalate.get_api_key(config):
|
|
1498
|
-
# Same non-interactive hazard as the safety confirms: this sits in
|
|
1499
|
-
# the autopilot path, so an unattended run must not stall here.
|
|
1500
|
-
escalated = ui.confirm_or_deny(" Escalate to the cloud model? [y/N] ")
|
|
1501
|
-
if escalated:
|
|
1502
|
-
tool_seq = [entry[0] for entry in _loop_tracker]
|
|
1503
|
-
suggestion = escalate.escalate(config, messages, tool_seq)
|
|
1504
|
-
print()
|
|
1505
|
-
cprint(" Cloud suggestion", C.BOLD)
|
|
1506
|
-
print(suggestion)
|
|
1507
|
-
print()
|
|
1508
1436
|
memory.maybe_index_turn(config, query, tools_used, touched_paths, outcome="error_loop")
|
|
1509
1437
|
if session:
|
|
1510
1438
|
_record_undo_snapshots(session, _turn_snapshots)
|
|
@@ -1685,6 +1613,30 @@ _RUN_TESTS_RE = re.compile(
|
|
|
1685
1613
|
)
|
|
1686
1614
|
|
|
1687
1615
|
|
|
1616
|
+
_BEHAVIOUR_CLAIM_RE = re.compile(
|
|
1617
|
+
r"\b(ran (?:it|the [a-z]+|successfully)(?: successfully)?|runs (?:correctly|fine|successfully|as expected)|"
|
|
1618
|
+
r"(?:it|this|that|everything|the [a-z]+) (?:now |all )?works\b|works (?:as expected|correctly|now|fine)|"
|
|
1619
|
+
r"(?:it|this|that|everything|the [a-z]+) is (?:now )?working(?: correctly| as expected)?|"
|
|
1620
|
+
r"(?:buttons?|it|the app|the page|the script) (?:now )?responds?|"
|
|
1621
|
+
r"calculates? correctly|opens? correctly|executed successfully|"
|
|
1622
|
+
r"tests? (?:pass|passed|passing|succeed)|all tests pass|verified (?:that )?it (?:works|runs)|"
|
|
1623
|
+
r"confirmed (?:that )?it (?:works|runs))\b", re.IGNORECASE)
|
|
1624
|
+
|
|
1625
|
+
|
|
1626
|
+
def _behaviour_claim(message: str) -> str:
|
|
1627
|
+
"""The first behaviour claim in a finish message, or ''. Deliberately a
|
|
1628
|
+
list of ways to say "I ran it and it works", not a list of verbs: "the
|
|
1629
|
+
function returns the sum" is a description, "it works" is a claim."""
|
|
1630
|
+
m = _BEHAVIOUR_CLAIM_RE.search(message or "")
|
|
1631
|
+
return m.group(0) if m else ""
|
|
1632
|
+
|
|
1633
|
+
|
|
1634
|
+
def _claim_nudge_text(claim: str) -> str:
|
|
1635
|
+
return (f"You said \"{claim}\" but nothing was run this turn. Run it with run_code or "
|
|
1636
|
+
"run_command and report what you observed, or say plainly that it was not run. "
|
|
1637
|
+
"Respond with JSON only.")
|
|
1638
|
+
|
|
1639
|
+
|
|
1688
1640
|
def _asks_to_run_tests(query: str) -> bool:
|
|
1689
1641
|
"""True when the request itself asks for the tests to be run."""
|
|
1690
1642
|
return bool(_RUN_TESTS_RE.search(query or ""))
|
|
@@ -8,7 +8,7 @@ The deterministic merge-aware compactor, the LLM summarizer behind explicit
|
|
|
8
8
|
Cross-cutting names (call_llm, build_autopilot_prompt, estimate_tokens,
|
|
9
9
|
sync_session_store, the token estimator, and compact_history itself when
|
|
10
10
|
auto-compact fires it) are resolved through the agent module AT CALL TIME —
|
|
11
|
-
the same idiom
|
|
11
|
+
the same idiom the loop uses — so every existing sa.<name> patch site keeps
|
|
12
12
|
intercepting. Module-local calls stay module-local only when nothing patches
|
|
13
13
|
them.
|
|
14
14
|
|
|
@@ -78,22 +78,11 @@ DEFAULT_CONFIG: dict[str, Any] = {
|
|
|
78
78
|
"chat_log_enabled": True,
|
|
79
79
|
"chat_log_dir": "",
|
|
80
80
|
"memory_enabled": True,
|
|
81
|
-
# The dreaming consolidation daemon is OFF by default: measured 2026-08-16
|
|
82
|
-
# writing the same five fabricated machine "facts" (wrong CPU, wrong RAM,
|
|
83
|
-
# an invented temperature) into memory_rules.md every idle cycle, which
|
|
84
|
-
# workspace_snapshot then injected as "Prior knowledge" — locking the
|
|
85
|
-
# model's hardware confabulations in permanently. V2X_ROADMAP already
|
|
86
|
-
# ruled it ships only with a quality eval; the eval now exists and it
|
|
87
|
-
# failed it. Re-enable only with new evidence.
|
|
88
|
-
"memory_dreaming": False,
|
|
89
81
|
"autopilot_confirm_destructive": True,
|
|
90
82
|
# Sensitive-data command gate (ssh keys, credential stores, security
|
|
91
83
|
# files, obfuscated execution). Separate from the destructive flag so
|
|
92
84
|
# injection defense holds even when destructive confirms are disabled.
|
|
93
85
|
"autopilot_confirm_sensitive": True,
|
|
94
|
-
# Agent protocol: "v1" (JSON action loop) or "v2" (native tool-call format,
|
|
95
|
-
# payload-block edits, persistent shell — see docs/V2_PLAN.md §5).
|
|
96
|
-
"protocol": "v1",
|
|
97
86
|
# Auto-compact is deterministic (no LLM call) by default: summarising via
|
|
98
87
|
# the same model that is already at its context cliff produced unverified
|
|
99
88
|
# summaries and cost a full extra re-prefill. Set true to restore the
|
|
@@ -102,21 +91,11 @@ DEFAULT_CONFIG: dict[str, Any] = {
|
|
|
102
91
|
# Override the derived history budget (tokens). Empty = derive from the
|
|
103
92
|
# measured system-prompt size.
|
|
104
93
|
"context_warn_tokens": 0,
|
|
105
|
-
# Local escalation ladder (docs/V2_PLAN.md §4): name of a bigger local
|
|
106
|
-
# npurun model to consult at hard moments (loop trips, ignored
|
|
107
|
-
# verification, prose-instead-of-edit). Empty = disabled. The server is
|
|
108
|
-
# spawned lazily on the bind address below and reused for the session.
|
|
109
|
-
"escalation_local_model": "",
|
|
110
|
-
"escalation_local_bind": "127.0.0.1:11436",
|
|
111
|
-
"escalation_max_output_tokens": 900,
|
|
112
|
-
"escalation_timeout_seconds": 240,
|
|
113
94
|
"ollama": {"host": "http://127.0.0.1:11434"},
|
|
114
95
|
"openai_compatible": {
|
|
115
96
|
"base_url": "http://127.0.0.1:8000/v1",
|
|
116
97
|
"api_key": "local",
|
|
117
98
|
},
|
|
118
|
-
"anthropic_api_key": "",
|
|
119
|
-
"escalation_model": "claude-haiku-4-5-20251001",
|
|
120
99
|
}
|
|
121
100
|
|
|
122
101
|
|
|
@@ -187,18 +166,10 @@ _CONFIG_SETTABLE: dict[str, str] = {
|
|
|
187
166
|
"chat_log_enabled": "bool",
|
|
188
167
|
"chat_log_dir": "str",
|
|
189
168
|
"memory_enabled": "bool",
|
|
190
|
-
"memory_dreaming": "bool",
|
|
191
169
|
"autopilot_confirm_destructive": "bool",
|
|
192
170
|
"autopilot_confirm_sensitive": "bool",
|
|
193
|
-
"protocol": "str",
|
|
194
171
|
"auto_compact_uses_llm": "bool",
|
|
195
172
|
"context_warn_tokens": "int",
|
|
196
|
-
"escalation_local_model": "str",
|
|
197
|
-
"escalation_local_bind": "str",
|
|
198
|
-
"escalation_max_output_tokens": "int",
|
|
199
|
-
"escalation_timeout_seconds": "int",
|
|
200
|
-
"anthropic_api_key": "str",
|
|
201
|
-
"escalation_model": "str",
|
|
202
173
|
}
|
|
203
174
|
|
|
204
175
|
|