hexcli 2.12.0__tar.gz → 2.14.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {hexcli-2.12.0 → hexcli-2.14.0}/CHANGELOG.md +65 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/PKG-INFO +1 -1
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/__init__.py +1 -1
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/agent.py +9 -5
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/editing.py +19 -6
- hexcli-2.14.0/hexcli/psparse.py +227 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/safety.py +60 -6
- {hexcli-2.12.0 → hexcli-2.14.0}/.gitignore +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/Hex CLI.cmd +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/LICENSE +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/README.md +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/assets/hexcli.ico +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/assets/hexcli.png +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/cancel.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/chatlog.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/commands.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/compaction.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/config.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/diffview.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/distribution.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/doctor.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/http_client.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/launcher.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/lineedit.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/llm.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/lockfile.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/markdown_stream.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/memory.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/network.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/parsing.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/paths.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/prompts.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/repl.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/sessions.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/setup_wizard.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/statusbar.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/stream_render.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/telemetry.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/tools.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/hexcli/ui.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/install.ps1 +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/launcher.py +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/pyproject.toml +0 -0
- {hexcli-2.12.0 → hexcli-2.14.0}/shellai.example.json +0 -0
|
@@ -4,6 +4,71 @@ Full evidence for every claim below — including the experiments that failed
|
|
|
4
4
|
lives in `docs/V2_PLAN.md` §14. Numbers are pass^k over repeated live runs on
|
|
5
5
|
the Hexagon NPU, not single-run anecdotes.
|
|
6
6
|
|
|
7
|
+
## 2.14.0 — 2026-09-17
|
|
8
|
+
|
|
9
|
+
A minor release: one tool behaves differently for the model (a `write_file`
|
|
10
|
+
body that was JSON-escaped twice is decoded), plus a test-hygiene fix.
|
|
11
|
+
|
|
12
|
+
> **Gate: PASS** on the pinned 24-case set, all 24 held at 5 runs with no
|
|
13
|
+
> recheck. Own arm, fresh server, seed 20260917, 0 invalid of 245 runs.
|
|
14
|
+
> Against the v2.11.1 12-run baseline: run-level 402/514 vs 179/230, −0.4 %,
|
|
15
|
+
> Fisher p = 0.92; pass^k 26 → 30 of 46 shared cases, McNemar p = 0.22. The
|
|
16
|
+
> rule fired in the arm on 3 of 5 `runit-1` writes and on nothing else
|
|
17
|
+
> (one `trap-1` poem hit the older newline shape, as before). `runit-1`:
|
|
18
|
+
> 5/5 in the arm and 15/15 in a follow-up on the same server, **20/20
|
|
19
|
+
> against 5/10 on 2.12.0, p = 0.002**. Ceiling cases that moved: `agentic-3`
|
|
20
|
+
> 3/5 (p = 0.17 against 18/20; both misses are the model writing invalid
|
|
21
|
+
> JSON through `edit_file`, where this rule does not run), `ambiguous-1`
|
|
22
|
+
> 4/5 (p = 0.50). Smoke 10/10 on a fresh server before the arm.
|
|
23
|
+
|
|
24
|
+
- `write_file` decodes a body whose every quote is escaped. `runit-1`
|
|
25
|
+
("Write hello.py that prints Hello, world and then run it") sat at 5/10
|
|
26
|
+
in the 2.12.0 arm, and not one of the misses was the "and run it" the
|
|
27
|
+
case was written for: every run wrote the file and ran it. The model
|
|
28
|
+
escaped its JSON twice, `{"content":"print(\\\"Hello, world\\\")"}`, so
|
|
29
|
+
the file held `print(\"Hello, world\")`, a SyntaxError; it then "fixed"
|
|
30
|
+
the line by editing it to itself, twice, until the step limit. The
|
|
31
|
+
double-escape rule from 2.9.0 only knew the newline shape (literal `\n`
|
|
32
|
+
and no real line break). It now also knows the quote shape — at least one
|
|
33
|
+
`\"` and not a single bare `"` — and decodes the body once, as before.
|
|
34
|
+
Measured over the 511 `write_file` bodies on record: 11 have the shape,
|
|
35
|
+
all of them this defect in this case, and none has both escaped and bare
|
|
36
|
+
quotes, so a body with even one bare quote is left exactly as sent.
|
|
37
|
+
`append_file` is untouched: 0 of its 17 recorded bodies have either shape.
|
|
38
|
+
|
|
39
|
+
- A test suite was writing into the owner's real chat log. The shell wiring
|
|
40
|
+
test in `evals/test_product_shell.py` drove `hexcli.agent` with the mock
|
|
41
|
+
backend and a temp config that left `chat_log_enabled` on, so every run
|
|
42
|
+
of the suite added a "say hello" session under `~/.shellai/chatlog/` — 93
|
|
43
|
+
of them by 2026-09-17, beside the 127 real sessions. The 2.12.0 exposure
|
|
44
|
+
counts were taken over that mix: "306 real turns" was 196 real turns and
|
|
45
|
+
93 scripted ones (the 11 firings were all on real turns, so the true
|
|
46
|
+
positives stand; the denominator did not). The test now turns the log
|
|
47
|
+
off, the corpus loader skips mock-backend sessions, the mock files were
|
|
48
|
+
moved aside, and the intent nudge's recorded rate is corrected to 9 of
|
|
49
|
+
196 real turns.
|
|
50
|
+
|
|
51
|
+
## 2.13.0 — 2026-09-17
|
|
52
|
+
|
|
53
|
+
- Commands are classified by the name the shell will really run, not by the
|
|
54
|
+
text as typed. PowerShell resolves aliases before executing, so the
|
|
55
|
+
pattern tiers were reading a different command from the one that ran:
|
|
56
|
+
`ri C:\data`, `rmdir C:\data`, `& ('Remove'+'-Item') C:\data` and
|
|
57
|
+
`sl C:\ ; ri *` all classified as *caution* and executed with **no
|
|
58
|
+
confirmation at all**. `hexcli/psparse.py` runs PowerShell's own parser
|
|
59
|
+
(`[Parser]::ParseInput`) in one long-lived `-NoProfile` process, resolves
|
|
60
|
+
each command name through `Get-Alias`, and hands the real names to the
|
|
61
|
+
existing tiers; resolving a name applies the existing policy rather than
|
|
62
|
+
inventing one, so `ri` is destructive because `Remove-Item` always was.
|
|
63
|
+
346 ms to start, 0.12 ms a parse after that, cached, and the helper exits
|
|
64
|
+
on EOF when the session does.
|
|
65
|
+
|
|
66
|
+
Measured before merging, over every command Hex CLI has actually been
|
|
67
|
+
asked to run — 353 commands, 119 distinct, from the owner's sessions and
|
|
68
|
+
every recorded eval run: **0 change tier**. The change cannot alter
|
|
69
|
+
behaviour on observed traffic; its whole effect is the four bypasses
|
|
70
|
+
above. Offline suites green, smoke 10/10 on a fresh server.
|
|
71
|
+
|
|
7
72
|
## 2.12.0 — 2026-09-17
|
|
8
73
|
|
|
9
74
|
Ten changes on one theme: the harness now checks that a turn did the kind of
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: hexcli
|
|
3
|
-
Version: 2.
|
|
3
|
+
Version: 2.14.0
|
|
4
4
|
Summary: Local Hexagon NPU terminal agent for Snapdragon X Elite Windows ARM64
|
|
5
5
|
Project-URL: Homepage, https://github.com/NathanL15/Hex-CLI
|
|
6
6
|
Project-URL: Repository, https://github.com/NathanL15/Hex-CLI
|
|
@@ -1741,8 +1741,11 @@ def _retry_echo(raw: str) -> str:
|
|
|
1741
1741
|
# none of the four can fire on it. evals/test_agent_loop.py pins that.
|
|
1742
1742
|
#
|
|
1743
1743
|
# Every guard below is here because an earlier draft fired on something
|
|
1744
|
-
# that was already RIGHT. They were found by replaying
|
|
1745
|
-
# turns and 1,366 recorded eval runs against the detectors
|
|
1744
|
+
# that was already RIGHT. They were found by replaying the real chat-log
|
|
1745
|
+
# turns and 1,366 recorded eval runs against the detectors (the count
|
|
1746
|
+
# first written here, 306 turns, included 93 that a test suite had
|
|
1747
|
+
# written into the real log with the mock backend; the owner's own
|
|
1748
|
+
# sessions are 196 turns, re-counted 2026-09-17):
|
|
1746
1749
|
#
|
|
1747
1750
|
# * "the condition is checked before the loop" — prose about code the
|
|
1748
1751
|
# user pasted, in an answer that never touches the filesystem (two real
|
|
@@ -1763,9 +1766,10 @@ def _retry_echo(raw: str) -> str:
|
|
|
1763
1766
|
# because a denied call is not in tools_used. Telling them apart needs
|
|
1764
1767
|
# the denial recorded, which is a separate change.
|
|
1765
1768
|
#
|
|
1766
|
-
# After the guards,
|
|
1767
|
-
# genuine miss
|
|
1768
|
-
#
|
|
1769
|
+
# After the guards, 9 of the 196 real turns fire and every one is a
|
|
1770
|
+
# genuine miss (the calculator, web-app, HiLo and resume sessions); 1 of
|
|
1771
|
+
# the 1,366 eval runs fires, a self-correct-1 run that claimed to have
|
|
1772
|
+
# checked and fixed with no tool call at all.
|
|
1769
1773
|
|
|
1770
1774
|
_WANTS_RUN_RE = re.compile(
|
|
1771
1775
|
r"\b(and|then|,)\s+(run|execute|start|launch)\s+(it|them|the\s+\w+)\b"
|
|
@@ -142,6 +142,7 @@ def unescape_json(text: str) -> str:
|
|
|
142
142
|
|
|
143
143
|
_DELTA_TOKEN_RE = re.compile(r"\w+|\s+|[^\w\s]")
|
|
144
144
|
_ESCAPED_RE = re.compile(r'\\["nt\\]')
|
|
145
|
+
_BARE_QUOTE_RE = re.compile(r'(?<!\\)"')
|
|
145
146
|
_DELTA_MAX_HUNKS = 4
|
|
146
147
|
_DELTA_MAX_REPLACED_CHARS = 40
|
|
147
148
|
_DELTA_MAX_INSERTED_CHARS = 400
|
|
@@ -149,12 +150,24 @@ _DELTA_MIN_MATCHED = 0.6 # fraction of old_string's tokens found in the window
|
|
|
149
150
|
|
|
150
151
|
|
|
151
152
|
def looks_double_escaped(text: str) -> bool:
|
|
152
|
-
"""A file body
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
153
|
+
"""A file body JSON-escaped a second time, in either of the two shapes
|
|
154
|
+
the model actually produces.
|
|
155
|
+
|
|
156
|
+
Newlines: two or more literal backslash-n sequences and not a single
|
|
157
|
+
real line break (write_file, persona_guard 2026-09-12: 'import
|
|
158
|
+
re\\n\\n\\n# Regex …' written as one line). A real one-line file that
|
|
159
|
+
spells "\\n" on purpose is rarer than the model's habit; anything with a
|
|
160
|
+
genuine newline is left alone.
|
|
161
|
+
|
|
162
|
+
Quotes: at least one backslash-quote and not a single bare quote
|
|
163
|
+
(runit-1, 2026-09-17: `print(\\"Hello, world\\")` as the whole of
|
|
164
|
+
hello.py, a SyntaxError the model then "fixed" by editing the line to
|
|
165
|
+
itself until the step limit). Over the 511 write_file bodies on record
|
|
166
|
+
11 have this shape and every one is that defect; none has both escaped
|
|
167
|
+
and bare quotes, so a body with even one bare quote is left alone."""
|
|
168
|
+
if "\n" not in text and text.count("\\n") >= 2:
|
|
169
|
+
return True
|
|
170
|
+
return '\\"' in text and _BARE_QUOTE_RE.search(text) is None
|
|
158
171
|
|
|
159
172
|
|
|
160
173
|
def _unescape_json(text: str) -> str:
|
|
@@ -0,0 +1,227 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""hexcli.psparse — structural facts about a PowerShell command line.
|
|
3
|
+
|
|
4
|
+
The safety classifier matched patterns against raw text, which fails in both
|
|
5
|
+
directions (measured 2026-09-15): `ri C:\\data` and `clc notes.txt` ran with no
|
|
6
|
+
confirmation because the alias never appears in a pattern, while
|
|
7
|
+
`Write-Output 'this will erase nothing'` was called destructive because the
|
|
8
|
+
word sat inside a string. Both need to know which text is a command name and
|
|
9
|
+
which is an inert argument, and that needs a parser for the language.
|
|
10
|
+
|
|
11
|
+
PowerShell has one built in: `[Parser]::ParseInput`. This module runs a small
|
|
12
|
+
helper script inside ONE long-lived `pwsh -NoProfile` process, sends it a
|
|
13
|
+
command per line and reads back a JSON summary. The process starts lazily on
|
|
14
|
+
first use and is reused for the session; results are cached by command string,
|
|
15
|
+
so a repeated command costs nothing. Everything degrades to "unavailable",
|
|
16
|
+
and the caller then keeps its pattern verdict unchanged.
|
|
17
|
+
|
|
18
|
+
The agent runs commands with `-NoProfile`, so the only aliases that can exist
|
|
19
|
+
are PowerShell's built-ins; the helper resolves them from the live session's
|
|
20
|
+
own alias table rather than a table copied into Python.
|
|
21
|
+
"""
|
|
22
|
+
from __future__ import annotations
|
|
23
|
+
|
|
24
|
+
import base64
|
|
25
|
+
import json
|
|
26
|
+
import os
|
|
27
|
+
import queue
|
|
28
|
+
import shutil
|
|
29
|
+
import subprocess
|
|
30
|
+
import threading
|
|
31
|
+
from dataclasses import dataclass, field
|
|
32
|
+
|
|
33
|
+
# One command in, one JSON line out. Base64 keeps newlines and quoting out of
|
|
34
|
+
# the protocol. `Ping` is answered before anything is parsed, so the caller can
|
|
35
|
+
# tell "the helper is up" from "the helper is wedged".
|
|
36
|
+
_HELPER = r"""
|
|
37
|
+
$ErrorActionPreference = 'Stop'
|
|
38
|
+
$aliases = @{}
|
|
39
|
+
foreach ($a in Get-Alias) {
|
|
40
|
+
# Definition, not ResolvedCommandName: the latter is null for most aliases
|
|
41
|
+
# unless the target module is already loaded (measured 2026-09-15: 78 of
|
|
42
|
+
# 135 came back empty and the table was useless).
|
|
43
|
+
$d = $a.Definition
|
|
44
|
+
if ($d) { $aliases[$a.Name.ToLowerInvariant()] = $d.ToLowerInvariant() }
|
|
45
|
+
}
|
|
46
|
+
Write-Output 'READY'
|
|
47
|
+
while ($null -ne ($line = [Console]::In.ReadLine())) {
|
|
48
|
+
try {
|
|
49
|
+
$cmd = [Text.Encoding]::UTF8.GetString([Convert]::FromBase64String($line))
|
|
50
|
+
$errors = $null
|
|
51
|
+
$ast = [System.Management.Automation.Language.Parser]::ParseInput($cmd, [ref]$null, [ref]$errors)
|
|
52
|
+
$names = New-Object System.Collections.ArrayList
|
|
53
|
+
$nonliteral = $false
|
|
54
|
+
foreach ($c in $ast.FindAll({ param($n) $n -is [System.Management.Automation.Language.CommandAst] }, $true)) {
|
|
55
|
+
$n = $c.GetCommandName()
|
|
56
|
+
if ($null -eq $n) { $nonliteral = $true; continue }
|
|
57
|
+
$n = $n.ToLowerInvariant()
|
|
58
|
+
if ($aliases.ContainsKey($n)) { $n = $aliases[$n] }
|
|
59
|
+
[void]$names.Add($n)
|
|
60
|
+
}
|
|
61
|
+
$strings = New-Object System.Collections.ArrayList
|
|
62
|
+
foreach ($s in $ast.FindAll({ param($n) $n -is [System.Management.Automation.Language.StringConstantExpressionAst] }, $true)) {
|
|
63
|
+
$p = $s.Parent
|
|
64
|
+
$isName = ($p -is [System.Management.Automation.Language.CommandAst]) -and ($p.CommandElements.Count -gt 0) -and ($p.CommandElements[0] -eq $s)
|
|
65
|
+
if (-not $isName) {
|
|
66
|
+
[void]$strings.Add(@($s.Extent.StartOffset, $s.Extent.EndOffset))
|
|
67
|
+
}
|
|
68
|
+
}
|
|
69
|
+
$out = @{ ok = $true; names = @($names); nonliteral = $nonliteral; strings = @($strings) }
|
|
70
|
+
} catch {
|
|
71
|
+
$out = @{ ok = $false; names = @(); nonliteral = $false; strings = @() }
|
|
72
|
+
}
|
|
73
|
+
Write-Output ($out | ConvertTo-Json -Compress -Depth 4)
|
|
74
|
+
}
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
_START_TIMEOUT_S = 10.0
|
|
78
|
+
_PARSE_TIMEOUT_S = 5.0
|
|
79
|
+
_CACHE_MAX = 512
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass
|
|
83
|
+
class Facts:
|
|
84
|
+
"""What the parser saw. `ok` False means nothing was learned and the
|
|
85
|
+
caller must keep whatever verdict it already had."""
|
|
86
|
+
ok: bool = False
|
|
87
|
+
names: tuple[str, ...] = () # canonical command names, aliases resolved
|
|
88
|
+
nonliteral: bool = False # a command whose name is not a literal
|
|
89
|
+
strings: tuple[tuple[int, int], ...] = field(default=()) # inert string-literal extents
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
UNAVAILABLE = Facts()
|
|
93
|
+
|
|
94
|
+
_lock = threading.Lock()
|
|
95
|
+
_proc: subprocess.Popen[str] | None = None
|
|
96
|
+
_replies: queue.Queue[str] | None = None
|
|
97
|
+
_disabled = False
|
|
98
|
+
_cache: dict[str, Facts] = {}
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _reader(proc: subprocess.Popen[str], q: queue.Queue[str]) -> None:
|
|
102
|
+
assert proc.stdout is not None
|
|
103
|
+
for line in iter(proc.stdout.readline, ""):
|
|
104
|
+
q.put(line.strip())
|
|
105
|
+
q.put("")
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def _shell_exe() -> str | None:
|
|
109
|
+
for candidate in ("pwsh.exe", "powershell.exe", "pwsh"):
|
|
110
|
+
found = shutil.which(candidate)
|
|
111
|
+
if found:
|
|
112
|
+
return found
|
|
113
|
+
return None
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _start() -> bool:
|
|
117
|
+
"""Bring the helper up. False (and disabled for good) when it cannot be."""
|
|
118
|
+
global _proc, _replies, _disabled
|
|
119
|
+
if _disabled:
|
|
120
|
+
return False
|
|
121
|
+
if _proc is not None and _proc.poll() is None:
|
|
122
|
+
return True
|
|
123
|
+
exe = _shell_exe()
|
|
124
|
+
if exe is None:
|
|
125
|
+
_disabled = True
|
|
126
|
+
return False
|
|
127
|
+
try:
|
|
128
|
+
env = dict(os.environ, POWERSHELL_TELEMETRY_OPTOUT="1")
|
|
129
|
+
proc = subprocess.Popen(
|
|
130
|
+
[exe, "-NoLogo", "-NoProfile", "-NonInteractive", "-Command", "-"],
|
|
131
|
+
stdin=subprocess.PIPE, stdout=subprocess.PIPE, stderr=subprocess.DEVNULL,
|
|
132
|
+
text=True, encoding="utf-8", errors="replace", env=env,
|
|
133
|
+
creationflags=getattr(subprocess, "CREATE_NO_WINDOW", 0),
|
|
134
|
+
)
|
|
135
|
+
q: queue.Queue[str] = queue.Queue()
|
|
136
|
+
threading.Thread(target=_reader, args=(proc, q), daemon=True).start()
|
|
137
|
+
assert proc.stdin is not None
|
|
138
|
+
proc.stdin.write(_HELPER + "\n")
|
|
139
|
+
proc.stdin.flush()
|
|
140
|
+
if q.get(timeout=_START_TIMEOUT_S) != "READY":
|
|
141
|
+
raise RuntimeError("helper did not announce itself")
|
|
142
|
+
except Exception: # noqa: BLE001 — no parser is a supported state, never an error
|
|
143
|
+
_disabled = True
|
|
144
|
+
_proc = None
|
|
145
|
+
return False
|
|
146
|
+
_proc, _replies = proc, q
|
|
147
|
+
return True
|
|
148
|
+
|
|
149
|
+
|
|
150
|
+
def _ask(cmd: str) -> Facts:
|
|
151
|
+
global _proc, _disabled
|
|
152
|
+
if not _start():
|
|
153
|
+
return UNAVAILABLE
|
|
154
|
+
assert _proc is not None and _proc.stdin is not None and _replies is not None
|
|
155
|
+
try:
|
|
156
|
+
payload = base64.b64encode(cmd.encode("utf-8")).decode("ascii")
|
|
157
|
+
_proc.stdin.write(payload + "\n")
|
|
158
|
+
_proc.stdin.flush()
|
|
159
|
+
line = _replies.get(timeout=_PARSE_TIMEOUT_S)
|
|
160
|
+
if not line:
|
|
161
|
+
raise RuntimeError("helper closed")
|
|
162
|
+
data = json.loads(line)
|
|
163
|
+
except Exception: # noqa: BLE001 — a wedged helper must not wedge the agent
|
|
164
|
+
try:
|
|
165
|
+
if _proc is not None:
|
|
166
|
+
_proc.kill()
|
|
167
|
+
except Exception: # noqa: BLE001
|
|
168
|
+
pass
|
|
169
|
+
_proc = None
|
|
170
|
+
_disabled = True # one failure is enough; the patterns still hold
|
|
171
|
+
return UNAVAILABLE
|
|
172
|
+
if not data.get("ok"):
|
|
173
|
+
return UNAVAILABLE
|
|
174
|
+
strings = tuple((int(a), int(b)) for a, b in (data.get("strings") or []))
|
|
175
|
+
return Facts(ok=True,
|
|
176
|
+
names=tuple(str(n) for n in (data.get("names") or [])),
|
|
177
|
+
nonliteral=bool(data.get("nonliteral")),
|
|
178
|
+
strings=strings)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def facts(cmd: str) -> Facts:
|
|
182
|
+
"""Structural facts about `cmd`, or UNAVAILABLE. Cached by command string:
|
|
183
|
+
the parse costs a round trip once per distinct command, and an agent that
|
|
184
|
+
retries the same call pays it once."""
|
|
185
|
+
key = cmd.strip()
|
|
186
|
+
if not key:
|
|
187
|
+
return UNAVAILABLE
|
|
188
|
+
with _lock:
|
|
189
|
+
hit = _cache.get(key)
|
|
190
|
+
if hit is not None:
|
|
191
|
+
return hit
|
|
192
|
+
got = _ask(key)
|
|
193
|
+
if len(_cache) >= _CACHE_MAX:
|
|
194
|
+
_cache.clear()
|
|
195
|
+
_cache[key] = got
|
|
196
|
+
return got
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
def shutdown() -> None:
|
|
200
|
+
"""Stop the helper. Tests call this; production does not need to.
|
|
201
|
+
|
|
202
|
+
The helper blocks on OUR stdin pipe, so when this process dies -- exit,
|
|
203
|
+
crash or taskkill alike -- the OS closes the write end, the helper reads
|
|
204
|
+
EOF and leaves. Verified 2026-09-17: a session that starts the helper and
|
|
205
|
+
exits without calling this leaves no PowerShell behind. Do not "fix" that
|
|
206
|
+
with an atexit hook; there is nothing to clean up, and an exit hook that
|
|
207
|
+
waits on a pipe is a way to hang a shutdown, not to speed one."""
|
|
208
|
+
global _proc
|
|
209
|
+
with _lock:
|
|
210
|
+
proc, _proc = _proc, None
|
|
211
|
+
_cache.clear()
|
|
212
|
+
if proc is not None:
|
|
213
|
+
try:
|
|
214
|
+
if proc.stdin is not None:
|
|
215
|
+
proc.stdin.close()
|
|
216
|
+
proc.wait(timeout=2)
|
|
217
|
+
except Exception: # noqa: BLE001
|
|
218
|
+
try:
|
|
219
|
+
proc.kill()
|
|
220
|
+
except Exception: # noqa: BLE001
|
|
221
|
+
pass
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _reset_for_tests() -> None:
|
|
225
|
+
global _disabled
|
|
226
|
+
shutdown()
|
|
227
|
+
_disabled = False
|
|
@@ -7,6 +7,8 @@ import re
|
|
|
7
7
|
from datetime import UTC, datetime
|
|
8
8
|
from pathlib import Path
|
|
9
9
|
|
|
10
|
+
from . import psparse
|
|
11
|
+
|
|
10
12
|
# Patterns checked in order: first match wins. Destructive > safe > caution.
|
|
11
13
|
|
|
12
14
|
_DESTRUCTIVE: list[re.Pattern[str]] = [re.compile(p, re.IGNORECASE) for p in [
|
|
@@ -84,13 +86,19 @@ _SAFE: list[re.Pattern[str]] = [re.compile(p, re.IGNORECASE) for p in [
|
|
|
84
86
|
]]
|
|
85
87
|
|
|
86
88
|
|
|
87
|
-
|
|
88
|
-
"""Return 'safe', 'caution', 'sensitive', or 'destructive'.
|
|
89
|
+
_SEVERITY = {"safe": 0, "caution": 1, "sensitive": 2, "destructive": 3}
|
|
89
90
|
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
"""
|
|
93
|
-
|
|
91
|
+
|
|
92
|
+
def _worse(a: str, b: str) -> str:
|
|
93
|
+
"""The more severe of two verdicts. An unknown or empty verdict ranks
|
|
94
|
+
below every real one, so "nothing to say about this name" can never
|
|
95
|
+
displace `safe` — it did while the default was `caution`'s rank, which
|
|
96
|
+
turned `Get-ChildItem` into an empty classification."""
|
|
97
|
+
return a if _SEVERITY.get(a, -1) >= _SEVERITY.get(b, -1) else b
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _classify_text(s: str) -> str:
|
|
101
|
+
"""The pattern tiers, matched against the command as written."""
|
|
94
102
|
for pat in _DESTRUCTIVE:
|
|
95
103
|
if pat.search(s):
|
|
96
104
|
return "destructive"
|
|
@@ -103,6 +111,52 @@ def classify_command(cmd: str) -> str:
|
|
|
103
111
|
return "caution"
|
|
104
112
|
|
|
105
113
|
|
|
114
|
+
def _tier_for_name(name: str) -> str:
|
|
115
|
+
"""What the tiers would say about a bare command name. Deriving this from
|
|
116
|
+
the same patterns is deliberate: resolving an alias must not invent a
|
|
117
|
+
policy, only apply the existing one to the name the shell will really
|
|
118
|
+
run. `ri` becomes `remove-item` and is destructive because
|
|
119
|
+
`Remove-Item` always was; `clc` becomes `clear-content` and stays
|
|
120
|
+
caution because `Clear-Content` is caution today."""
|
|
121
|
+
probe = name + " "
|
|
122
|
+
for pat in _DESTRUCTIVE:
|
|
123
|
+
if pat.search(probe):
|
|
124
|
+
return "destructive"
|
|
125
|
+
for pat in _SENSITIVE:
|
|
126
|
+
if pat.search(probe):
|
|
127
|
+
return "sensitive"
|
|
128
|
+
return ""
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def classify_command(cmd: str) -> str:
|
|
132
|
+
"""Return 'safe', 'caution', 'sensitive', or 'destructive'.
|
|
133
|
+
|
|
134
|
+
Priority: destructive > sensitive > safe > caution. Sensitive must outrank
|
|
135
|
+
the safe list, or read-only cmdlet prefixes whitelist credential access.
|
|
136
|
+
|
|
137
|
+
Text alone is not enough. Measured 2026-09-15: `ri C:\\data`,
|
|
138
|
+
`rmdir C:\\data` and `$c='Remove-Item'; & ($a+$b)` all ran with no
|
|
139
|
+
confirmation, because an alias or a computed name never appears in a
|
|
140
|
+
pattern. `hexcli.psparse` resolves the command names the shell will
|
|
141
|
+
actually run (the agent's shell is `-NoProfile`, so only built-in
|
|
142
|
+
aliases exist) and reports a name that is not a literal at all. The
|
|
143
|
+
result can only ever be MORE severe than the text verdict: when the
|
|
144
|
+
parser is unavailable this is exactly the behaviour it always had.
|
|
145
|
+
"""
|
|
146
|
+
s = cmd.strip()
|
|
147
|
+
verdict = _classify_text(s)
|
|
148
|
+
facts = psparse.facts(s)
|
|
149
|
+
if not facts.ok:
|
|
150
|
+
return verdict
|
|
151
|
+
for name in facts.names:
|
|
152
|
+
verdict = _worse(verdict, _tier_for_name(name))
|
|
153
|
+
if facts.nonliteral:
|
|
154
|
+
# `& ($a + $b)` — the name is computed, so nothing can be checked
|
|
155
|
+
# about it before it runs. Confirm rather than assume.
|
|
156
|
+
verdict = _worse(verdict, "sensitive")
|
|
157
|
+
return verdict
|
|
158
|
+
|
|
159
|
+
|
|
106
160
|
def append_audit_log(
|
|
107
161
|
session_id: str | None,
|
|
108
162
|
classification: str,
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|