hexcli 2.11.0__tar.gz → 2.11.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (50) hide show
  1. {hexcli-2.11.0 → hexcli-2.11.1}/.gitignore +1 -0
  2. {hexcli-2.11.0 → hexcli-2.11.1}/CHANGELOG.md +27 -0
  3. {hexcli-2.11.0 → hexcli-2.11.1}/PKG-INFO +2 -2
  4. {hexcli-2.11.0 → hexcli-2.11.1}/README.md +1 -1
  5. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/__init__.py +1 -1
  6. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/agent.py +0 -93
  7. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/compaction.py +1 -1
  8. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/config.py +0 -29
  9. hexcli-2.11.1/hexcli/editing.py +456 -0
  10. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/llm.py +1 -1
  11. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/memory.py +1 -70
  12. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/parsing.py +0 -32
  13. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/repl.py +1 -8
  14. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/tools.py +3 -3
  15. {hexcli-2.11.0 → hexcli-2.11.1}/pyproject.toml +0 -2
  16. {hexcli-2.11.0 → hexcli-2.11.1}/shellai.example.json +1 -9
  17. hexcli-2.11.0/hexcli/escalate.py +0 -192
  18. hexcli-2.11.0/hexcli/local_escalation.py +0 -191
  19. hexcli-2.11.0/hexcli/loop_v2.py +0 -396
  20. hexcli-2.11.0/hexcli/protocol_v2.py +0 -819
  21. hexcli-2.11.0/hexcli/shell_session.py +0 -186
  22. hexcli-2.11.0/shellai.cmd +0 -2
  23. hexcli-2.11.0/shellai.py +0 -15
  24. {hexcli-2.11.0 → hexcli-2.11.1}/Hex CLI.cmd +0 -0
  25. {hexcli-2.11.0 → hexcli-2.11.1}/LICENSE +0 -0
  26. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/assets/hexcli.ico +0 -0
  27. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/assets/hexcli.png +0 -0
  28. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/cancel.py +0 -0
  29. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/chatlog.py +0 -0
  30. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/commands.py +0 -0
  31. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/diffview.py +0 -0
  32. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/distribution.py +0 -0
  33. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/doctor.py +0 -0
  34. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/http_client.py +0 -0
  35. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/launcher.py +0 -0
  36. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/lineedit.py +0 -0
  37. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/lockfile.py +0 -0
  38. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/markdown_stream.py +0 -0
  39. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/network.py +0 -0
  40. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/paths.py +0 -0
  41. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/prompts.py +0 -0
  42. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/safety.py +0 -0
  43. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/sessions.py +0 -0
  44. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/setup_wizard.py +0 -0
  45. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/statusbar.py +0 -0
  46. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/stream_render.py +0 -0
  47. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/telemetry.py +0 -0
  48. {hexcli-2.11.0 → hexcli-2.11.1}/hexcli/ui.py +0 -0
  49. {hexcli-2.11.0 → hexcli-2.11.1}/install.ps1 +0 -0
  50. {hexcli-2.11.0 → hexcli-2.11.1}/launcher.py +0 -0
@@ -60,3 +60,4 @@ dist/
60
60
  # Local-only notes: research surveys and working documents that are not user-facing.
61
61
  # Only the paper (docs/paper) and user-facing docs are committed (owner rule, 2026-09-14).
62
62
  docs/local/
63
+ tools/local/
@@ -6,6 +6,33 @@ the Hexagon NPU, not single-run anecdotes.
6
6
 
7
7
  ## Unreleased
8
8
 
9
+ ## 2.11.1 — 2026-09-14
10
+
11
+ A patch release: nothing model-facing and nothing the launcher hands the
12
+ server; code and documents nobody used are gone. Gate: every remaining
13
+ suite green (27 suites, 739 tests; 29 suites and 798 tests before), smoke on a fresh server
14
+ 9/10 then 10/10 (the miss was factual-1, a no-tool knowledge answer that
15
+ missed once in every arm today), CI green on main and the tag.
16
+
17
+ - The prune. Nothing here was used: protocol v2 (`loop_v2.py`,
18
+ `shell_session.py`, the v2 parser and prompt in `protocol_v2.py`, its
19
+ suite, the `protocol` config key), which lost its A/B at 13/36 vs 22/35
20
+ and doubled every safety and file-tool change; the local escalation
21
+ ladder (no viable bigger model on this hardware) and the cloud
22
+ escalation path (never configured), with their six config keys, so "no
23
+ code leaves the machine" is now structural rather than a default; the
24
+ memory dreaming daemon, off since it fabricated hardware facts; the
25
+ unused brace scanner in the parser; the root `shellai.py` / `shellai.cmd`
26
+ shims. The SEARCH/REPLACE applier that `edit_file` uses moved out of
27
+ `protocol_v2.py` into `hexcli/editing.py` unchanged, with its tests in
28
+ `evals/test_editing.py`. Internal documents (the V2 plan and roadmap, the
29
+ levers memo, the backend study, ARCHITECTURE.md) and the study-only bench
30
+ probes are no longer tracked; they live in git-ignored `docs/local/` and
31
+ `tools/local/` (owner rule: only the paper and user-facing docs are
32
+ committed). `tools/backend_bench/` keeps `stall_rate.py` and its two
33
+ imports, which the release gate runs. About 3,300 lines and three suites
34
+ gone; every remaining suite green.
35
+
9
36
  ## 2.11.0 — 2026-09-14
10
37
 
11
38
  A minor release: tool result text the model reads changed (`verify_syntax`
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: hexcli
3
- Version: 2.11.0
3
+ Version: 2.11.1
4
4
  Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
5
5
  Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
6
6
  Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
@@ -353,7 +353,7 @@ Restart the NPU server before each suite. After an hour or two of steady
353
353
  use it starts returning errors for everything, which looks like a model
354
354
  regression. The runner detects this and marks those runs invalid.
355
355
 
356
- `ARCHITECTURE.md` describes the module layout. `docs/V2_PLAN.md` has the
356
+ `CLAUDE.md` describes the module layout. The paper in `docs/paper/` has the
357
357
  hardware measurements, the eval method, and the reasoning behind each
358
358
  safety layer. `RELEASING.md` covers how a release is cut.
359
359
 
@@ -326,7 +326,7 @@ Restart the NPU server before each suite. After an hour or two of steady
326
326
  use it starts returning errors for everything, which looks like a model
327
327
  regression. The runner detects this and marks those runs invalid.
328
328
 
329
- `ARCHITECTURE.md` describes the module layout. `docs/V2_PLAN.md` has the
329
+ `CLAUDE.md` describes the module layout. The paper in `docs/paper/` has the
330
330
  hardware measurements, the eval method, and the reasoning behind each
331
331
  safety layer. `RELEASING.md` covers how a release is cut.
332
332
 
@@ -3,4 +3,4 @@
3
3
  # The one place the version is written. pyproject.toml reads it (hatch
4
4
  # dynamic version), agent.VERSION re-exports it, and CI refuses a release
5
5
  # tag that does not match it.
6
- __version__ = "2.11.0"
6
+ __version__ = "2.11.1"
@@ -28,10 +28,8 @@ from hexcli import (
28
28
  compaction,
29
29
  diffview,
30
30
  distribution,
31
- escalate,
32
31
  http_client,
33
32
  llm,
34
- local_escalation,
35
33
  lockfile,
36
34
  memory,
37
35
  network,
@@ -381,7 +379,6 @@ is_small_talk = parsing.is_small_talk
381
379
  local_meta_response = parsing.local_meta_response
382
380
  strip_thinking = parsing.strip_thinking
383
381
  parse_json_object = parsing.parse_json_object
384
- _iter_json_objects = parsing._iter_json_objects
385
382
  parse_agent_action = parsing.parse_agent_action
386
383
  _looks_like_botched_action = parsing._looks_like_botched_action
387
384
 
@@ -444,23 +441,6 @@ def set_active_config(config: dict[str, Any] | None) -> None:
444
441
 
445
442
 
446
443
 
447
- # One escalation server per (model, bind) for the process lifetime — spawning
448
- # a fresh 4.6 GB bundle load per consult would make escalation useless.
449
- _ESCALATORS: dict[str, local_escalation.LocalEscalator] = {}
450
-
451
-
452
- def _get_escalator(config: dict[str, Any]) -> local_escalation.LocalEscalator | None:
453
- model = str(config.get("escalation_local_model", "") or "")
454
- if not model:
455
- return None
456
- key = f"{model}@{config.get('escalation_local_bind', '127.0.0.1:11436')}"
457
- esc = _ESCALATORS.get(key)
458
- if esc is None:
459
- esc = local_escalation.LocalEscalator(config)
460
- _ESCALATORS[key] = esc
461
- return esc if esc.enabled else None
462
-
463
-
464
444
  class AutopilotProbe:
465
445
  """Optional instrumentation hook for run_autopilot, used by evals/ to
466
446
  observe the production agent loop without reimplementing it. Every
@@ -1120,14 +1100,6 @@ def _run_autopilot_turn(
1120
1100
 
1121
1101
  _sync_context_window(config)
1122
1102
 
1123
- if str(config.get("protocol", "v1")).lower() == "v2":
1124
- from . import loop_v2
1125
- _CURRENT_SESSION_ID = str(uuid4())
1126
- return loop_v2.run(
1127
- config, history, query, shell_exe,
1128
- session=session, turn=turn, probe=probe,
1129
- )
1130
-
1131
1103
  # Fresh UUID for this agent loop: lets the npurun server detect
1132
1104
  # continuation turns (messages only appended) and skip reset_dialog(),
1133
1105
  # so Genie re-prefills only the new tokens via SentenceCode::Rewind.
@@ -1203,33 +1175,6 @@ def _run_autopilot_turn(
1203
1175
  # Local escalation (docs/V2_PLAN.md §4 ladder): consult the bigger local
1204
1176
  # model at hard moments. At most one consult per turn; every failure path
1205
1177
  # degrades to the pre-escalation behaviour.
1206
- _escalator = _get_escalator(config)
1207
- _escalation_used = False
1208
- _turn_events: list[str] = []
1209
-
1210
- def _consult_and_inject(problem: str, raw_response: str) -> bool:
1211
- """Ask the local escalation model for advice and inject it as the next
1212
- user message. Returns True when advice was injected."""
1213
- nonlocal _escalation_used
1214
- if _escalator is None or _escalation_used:
1215
- return False
1216
- cprint("\n Consulting the escalation model.", C.DIM, file=sys.stderr)
1217
- advice = _escalator.consult(
1218
- local_escalation.build_situation(query, _turn_events, problem))
1219
- if not advice:
1220
- return False
1221
- _escalation_used = True
1222
- if raw_response:
1223
- messages.append({"role": "assistant", "content": strip_thinking(raw_response)})
1224
- messages.append({
1225
- "role": "user",
1226
- "content": (
1227
- "A senior engineer reviewed the situation and advises:\n"
1228
- f"{advice}\n"
1229
- "Apply this advice now using the tools. Respond with JSON only."
1230
- ),
1231
- })
1232
- return True
1233
1178
 
1234
1179
  for step in range(max_steps):
1235
1180
  step_label = "thinking" if step == 0 else f"step {step + 1}/{max_steps}"
@@ -1356,25 +1301,6 @@ def _run_autopilot_turn(
1356
1301
  messages.append({"role": "user", "content": _claim_nudge_text(claim)})
1357
1302
  continue
1358
1303
  _unbacked_claim = True
1359
- # Escalation trigger B — the verification nudge was ignored: the
1360
- # model finished a second time without checking its own mutation.
1361
- if (_unverified_mutation and _verify_nudge_used
1362
- and config.get("require_verification", True)
1363
- and _consult_and_inject(
1364
- "The agent modified a file but is finishing WITHOUT verifying "
1365
- "the change, even after being asked to verify.", raw)):
1366
- continue
1367
- # Escalation trigger C — prose instead of action: the task asks
1368
- # for a file change, nothing was mutated, and the finish is not a
1369
- # clarifying question (questions are the CORRECT outcome for
1370
- # ambiguous requests — never escalate those).
1371
- if (local_escalation.looks_like_edit_request(query)
1372
- and not local_escalation.turn_mutated(tools_used)
1373
- and not msg.rstrip().endswith("?")
1374
- and _consult_and_inject(
1375
- "The task asks for a file change, but the agent is finishing "
1376
- f"without having modified any file. Its answer was: {msg[:300]}", raw)):
1377
- continue
1378
1304
  # Nudge once if the model refused to use tools
1379
1305
  if (step == 0 and not direct_stage
1380
1306
  and any(phrase in msg.lower() for phrase in REFUSAL_PHRASES)):
@@ -1479,7 +1405,6 @@ def _run_autopilot_turn(
1479
1405
  pass
1480
1406
 
1481
1407
  last_tool_output = tool_output
1482
- _turn_events.append(f"{tool_name}: {tool_output[:220]}")
1483
1408
  if _run_targets and tool_name in ("run_code", "run_command") and _ran_tests(_run_targets[-1:]):
1484
1409
  _last_test_failure = _test_failure_tail(tool_output)
1485
1410
  _is_error = tool_output.lstrip().startswith("Error:")
@@ -1504,28 +1429,10 @@ def _run_autopilot_turn(
1504
1429
  and all(err for _t, _tgt, err, _out in _loop_tracker)
1505
1430
  and len({(t, tgt) for t, tgt, _e, _out in _loop_tracker}) == 1)
1506
1431
  if _identical_trip or _failure_trip:
1507
- # Escalation trigger A — the loop detector: consult the local
1508
- # model BEFORE giving up (the cloud path stays as the fallback).
1509
- if _consult_and_inject(
1510
- f"The agent repeated the same failing call 3 times: {tool_name} "
1511
- f"kept returning:\n{tool_output[:400]}", raw):
1512
- _loop_tracker.clear()
1513
- continue
1514
1432
  what = f"{tool_name} returned the same result" if _identical_trip else f"{tool_name} failed"
1515
1433
  if not _in_delegate: # a sub-agent's stop is its own result, not the turn's
1516
1434
  cprint(f"\n ⚠ Stopped: {what} three times in a row.", C.BYELLOW)
1517
1435
  _mark_turn_stopped("loop")
1518
- if escalate.get_api_key(config):
1519
- # Same non-interactive hazard as the safety confirms: this sits in
1520
- # the autopilot path, so an unattended run must not stall here.
1521
- escalated = ui.confirm_or_deny(" Escalate to the cloud model? [y/N] ")
1522
- if escalated:
1523
- tool_seq = [entry[0] for entry in _loop_tracker]
1524
- suggestion = escalate.escalate(config, messages, tool_seq)
1525
- print()
1526
- cprint(" Cloud suggestion", C.BOLD)
1527
- print(suggestion)
1528
- print()
1529
1436
  memory.maybe_index_turn(config, query, tools_used, touched_paths, outcome="error_loop")
1530
1437
  if session:
1531
1438
  _record_undo_snapshots(session, _turn_snapshots)
@@ -8,7 +8,7 @@ The deterministic merge-aware compactor, the LLM summarizer behind explicit
8
8
  Cross-cutting names (call_llm, build_autopilot_prompt, estimate_tokens,
9
9
  sync_session_store, the token estimator, and compact_history itself when
10
10
  auto-compact fires it) are resolved through the agent module AT CALL TIME —
11
- the same idiom loop_v2 uses — so every existing sa.<name> patch site keeps
11
+ the same idiom the loop uses — so every existing sa.<name> patch site keeps
12
12
  intercepting. Module-local calls stay module-local only when nothing patches
13
13
  them.
14
14
 
@@ -78,22 +78,11 @@ DEFAULT_CONFIG: dict[str, Any] = {
78
78
  "chat_log_enabled": True,
79
79
  "chat_log_dir": "",
80
80
  "memory_enabled": True,
81
- # The dreaming consolidation daemon is OFF by default: measured 2026-08-16
82
- # writing the same five fabricated machine "facts" (wrong CPU, wrong RAM,
83
- # an invented temperature) into memory_rules.md every idle cycle, which
84
- # workspace_snapshot then injected as "Prior knowledge" — locking the
85
- # model's hardware confabulations in permanently. V2X_ROADMAP already
86
- # ruled it ships only with a quality eval; the eval now exists and it
87
- # failed it. Re-enable only with new evidence.
88
- "memory_dreaming": False,
89
81
  "autopilot_confirm_destructive": True,
90
82
  # Sensitive-data command gate (ssh keys, credential stores, security
91
83
  # files, obfuscated execution). Separate from the destructive flag so
92
84
  # injection defense holds even when destructive confirms are disabled.
93
85
  "autopilot_confirm_sensitive": True,
94
- # Agent protocol: "v1" (JSON action loop) or "v2" (native tool-call format,
95
- # payload-block edits, persistent shell — see docs/V2_PLAN.md §5).
96
- "protocol": "v1",
97
86
  # Auto-compact is deterministic (no LLM call) by default: summarising via
98
87
  # the same model that is already at its context cliff produced unverified
99
88
  # summaries and cost a full extra re-prefill. Set true to restore the
@@ -102,21 +91,11 @@ DEFAULT_CONFIG: dict[str, Any] = {
102
91
  # Override the derived history budget (tokens). Empty = derive from the
103
92
  # measured system-prompt size.
104
93
  "context_warn_tokens": 0,
105
- # Local escalation ladder (docs/V2_PLAN.md §4): name of a bigger local
106
- # npurun model to consult at hard moments (loop trips, ignored
107
- # verification, prose-instead-of-edit). Empty = disabled. The server is
108
- # spawned lazily on the bind address below and reused for the session.
109
- "escalation_local_model": "",
110
- "escalation_local_bind": "127.0.0.1:11436",
111
- "escalation_max_output_tokens": 900,
112
- "escalation_timeout_seconds": 240,
113
94
  "ollama": {"host": "http://127.0.0.1:11434"},
114
95
  "openai_compatible": {
115
96
  "base_url": "http://127.0.0.1:8000/v1",
116
97
  "api_key": "local",
117
98
  },
118
- "anthropic_api_key": "",
119
- "escalation_model": "claude-haiku-4-5-20251001",
120
99
  }
121
100
 
122
101
 
@@ -187,18 +166,10 @@ _CONFIG_SETTABLE: dict[str, str] = {
187
166
  "chat_log_enabled": "bool",
188
167
  "chat_log_dir": "str",
189
168
  "memory_enabled": "bool",
190
- "memory_dreaming": "bool",
191
169
  "autopilot_confirm_destructive": "bool",
192
170
  "autopilot_confirm_sensitive": "bool",
193
- "protocol": "str",
194
171
  "auto_compact_uses_llm": "bool",
195
172
  "context_warn_tokens": "int",
196
- "escalation_local_model": "str",
197
- "escalation_local_bind": "str",
198
- "escalation_max_output_tokens": "int",
199
- "escalation_timeout_seconds": "int",
200
- "anthropic_api_key": "str",
201
- "escalation_model": "str",
202
173
  }
203
174
 
204
175