sift-cli 1.0.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- sift/__init__.py +18 -0
- sift/answers.py +175 -0
- sift/background.py +444 -0
- sift/capture.py +240 -0
- sift/cli.py +820 -0
- sift/digest.py +101 -0
- sift/distill.py +670 -0
- sift/fallback.py +98 -0
- sift/hook.py +275 -0
- sift/lines.py +37 -0
- sift/many.py +51 -0
- sift/memory.py +94 -0
- sift/model.py +433 -0
- sift/outline.py +117 -0
- sift/peek.py +161 -0
- sift/privacy.py +145 -0
- sift/records.py +95 -0
- sift/server.py +552 -0
- sift/store.py +499 -0
- sift/tools.py +76 -0
- sift/view.py +317 -0
- sift/watch.py +166 -0
- sift_cli-1.0.0.dist-info/METADATA +326 -0
- sift_cli-1.0.0.dist-info/RECORD +27 -0
- sift_cli-1.0.0.dist-info/WHEEL +4 -0
- sift_cli-1.0.0.dist-info/entry_points.txt +3 -0
- sift_cli-1.0.0.dist-info/licenses/LICENSE +21 -0
sift/fallback.py
ADDED
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
"""What to show when nobody could be asked.
|
|
2
|
+
|
|
3
|
+
There is no key, or no network, or the endpoint is busy, or the reply made no
|
|
4
|
+
sense. The command has already run and its output is already on disk. Something
|
|
5
|
+
has to be shown, and it has to be shown without a model.
|
|
6
|
+
|
|
7
|
+
The temptation here is to be clever: look for the word "error", recognise a
|
|
8
|
+
stack trace, score lines by how alarming they look. That is exactly the design
|
|
9
|
+
this project was rewritten to get rid of. Patterns like those are a list of
|
|
10
|
+
languages wearing a disguise -- they work for English and for the half-dozen
|
|
11
|
+
formats whoever wrote them happened to know, and they quietly fail for Turkish,
|
|
12
|
+
for Japanese, for a tool that shipped last week. Worse, they fail while looking
|
|
13
|
+
confident.
|
|
14
|
+
|
|
15
|
+
So this makes no claim about meaning at all. It shows the beginning and the end
|
|
16
|
+
and marks what it skipped. The beginning says what was run; the end says how it
|
|
17
|
+
came out. That is true of a compiler, a test runner, an installer, a shell
|
|
18
|
+
script, in every language a person or a machine writes in, and it needs to know
|
|
19
|
+
nothing about any of them.
|
|
20
|
+
|
|
21
|
+
It is worse than a model. It is meant to be. What it must never be is wrong, and
|
|
22
|
+
a rule that makes no claim cannot make a false one.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
from sift import lines as text_lines
|
|
28
|
+
from sift import records
|
|
29
|
+
from sift.capture import Capture
|
|
30
|
+
from sift.distill import View, render
|
|
31
|
+
|
|
32
|
+
# The first lines are usually the invocation and the first thing to go wrong.
|
|
33
|
+
HEAD = 10
|
|
34
|
+
|
|
35
|
+
# The last lines are the result: the summary, the exit, the reason. In every
|
|
36
|
+
# command-line tradition, the ending is where the answer lives.
|
|
37
|
+
TAIL = 40
|
|
38
|
+
|
|
39
|
+
# A run that failed says why near the end, and tends to say more of it: a
|
|
40
|
+
# traceback, a diagnostic, a list of what did not build.
|
|
41
|
+
TAIL_WHEN_FAILED = 80
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def ends(total: int, *, head: int = HEAD, tail: int = TAIL) -> set[int]:
|
|
45
|
+
"""The first `head` and the last `tail` lines, or all of them if that is fewer."""
|
|
46
|
+
if total <= head + tail:
|
|
47
|
+
return set(range(1, total + 1))
|
|
48
|
+
return set(range(1, head + 1)) | set(range(total - tail + 1, total + 1))
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def from_lines(
|
|
52
|
+
lines: list[str],
|
|
53
|
+
handle: str,
|
|
54
|
+
*,
|
|
55
|
+
tail: int = TAIL,
|
|
56
|
+
first: int = 1,
|
|
57
|
+
unit: str = "line",
|
|
58
|
+
) -> View:
|
|
59
|
+
"""The ends of any numbered text, marked with what lies between them.
|
|
60
|
+
|
|
61
|
+
A source file gets the same treatment as a capture, and for the same reason:
|
|
62
|
+
the moment this function starts telling the two apart it has begun keeping a
|
|
63
|
+
list of what things are, which is the list this project was rewritten to be
|
|
64
|
+
rid of. The beginning and the end of a file are a poor outline. They are not
|
|
65
|
+
a wrong one.
|
|
66
|
+
"""
|
|
67
|
+
total = len(lines)
|
|
68
|
+
if not total:
|
|
69
|
+
return View(handle, "", 0, 0, None, 0, unit=unit)
|
|
70
|
+
|
|
71
|
+
# `ends` counts from one because it is arithmetic about a length, not about
|
|
72
|
+
# a capture. Where these lines sit in the run is the caller's business, and
|
|
73
|
+
# it is added here, once, rather than taught to the arithmetic.
|
|
74
|
+
chosen = {number + first - 1 for number in ends(total, head=HEAD, tail=tail)}
|
|
75
|
+
return View(
|
|
76
|
+
handle=handle,
|
|
77
|
+
text=render(lines, chosen, handle, first, unit),
|
|
78
|
+
kept=len(chosen),
|
|
79
|
+
total=total,
|
|
80
|
+
model=None,
|
|
81
|
+
asks=0,
|
|
82
|
+
unit=unit,
|
|
83
|
+
)
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def fallback(capture: Capture) -> View:
|
|
87
|
+
"""A view built without asking anything, and honest about being one.
|
|
88
|
+
|
|
89
|
+
`model` is left empty, which is how the caller can tell this apart from a
|
|
90
|
+
view a model chose. A reader who is told which lines were picked, and by
|
|
91
|
+
what, can decide whether to go and read the rest.
|
|
92
|
+
"""
|
|
93
|
+
tail = TAIL_WHEN_FAILED if capture.meta.failed else TAIL
|
|
94
|
+
text = capture.text()
|
|
95
|
+
found = records.of(text)
|
|
96
|
+
if found is not None:
|
|
97
|
+
return from_lines(found, capture.handle, tail=tail, unit="record")
|
|
98
|
+
return from_lines(text_lines.of(text), capture.handle, tail=tail)
|
sift/hook.py
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
"""Catching the shell commands a client runs on its own.
|
|
2
|
+
|
|
3
|
+
The MCP server can only distil what it was asked to distil. Everything else a
|
|
4
|
+
client does with a shell -- and a coding agent does a great deal -- lands in the
|
|
5
|
+
conversation whole. This closes that gap without a proxy: the client is asked to
|
|
6
|
+
send its shell commands here first, `sift` runs them, and what comes back is a
|
|
7
|
+
view.
|
|
8
|
+
|
|
9
|
+
Two decisions, and the first is the one worth arguing about.
|
|
10
|
+
|
|
11
|
+
**Everything is routed. Nothing decides whether a command "looks noisy".**
|
|
12
|
+
|
|
13
|
+
That rule was tempting and it is exactly the mistake this project was rewritten
|
|
14
|
+
to avoid. A list of commands worth intercepting is a list of tools wearing a
|
|
15
|
+
disguise: it would know `pytest` and `cargo` and `npm`, be wrong about the
|
|
16
|
+
in-house script, and be confidently silent about the one that printed forty
|
|
17
|
+
thousand lines. And it cannot be right in principle -- how much a command prints
|
|
18
|
+
is not knowable before it runs.
|
|
19
|
+
|
|
20
|
+
Routing everything costs nothing, because a view of a short output *is* that
|
|
21
|
+
output: the budget only takes hold when there is more than the budget. Twelve
|
|
22
|
+
lines in, twelve lines out.
|
|
23
|
+
|
|
24
|
+
**It fails open, and that is the third rule again.** A bug here, an unreadable
|
|
25
|
+
event, a command that could not be started -- every one of them ends with the
|
|
26
|
+
client running the command itself, exactly as it would have. A gate that breaks
|
|
27
|
+
a shell is worse than no gate, and this one is allowed to break.
|
|
28
|
+
|
|
29
|
+
`SIFT_HOOK=0` turns it off for someone who wants their shell back untouched.
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import contextlib
|
|
35
|
+
import json
|
|
36
|
+
import os
|
|
37
|
+
from pathlib import Path
|
|
38
|
+
|
|
39
|
+
from sift.capture import run
|
|
40
|
+
from sift.view import best_view, footer
|
|
41
|
+
|
|
42
|
+
# What this returns when it has nothing to say. An empty answer means "carry on
|
|
43
|
+
# as you were", which is the only safe thing to say when something went wrong.
|
|
44
|
+
PASS = {}
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def wanted() -> bool:
|
|
48
|
+
"""Whether shell commands should be routed here at all."""
|
|
49
|
+
return (os.environ.get("SIFT_HOOK") or "1").strip().lower() not in {
|
|
50
|
+
"0",
|
|
51
|
+
"false",
|
|
52
|
+
"no",
|
|
53
|
+
"off",
|
|
54
|
+
"",
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def answer(event: dict) -> dict:
|
|
59
|
+
"""What to tell the client about one shell command it is about to run.
|
|
60
|
+
|
|
61
|
+
The event is whatever the client put on stdin, and it is treated as data
|
|
62
|
+
from somewhere else: every field is checked before it is used, and anything
|
|
63
|
+
unexpected means *carry on*, never an error. A hook that raises on a shape
|
|
64
|
+
it did not expect takes the shell down with it.
|
|
65
|
+
"""
|
|
66
|
+
if not wanted():
|
|
67
|
+
return PASS
|
|
68
|
+
if not isinstance(event, dict):
|
|
69
|
+
return PASS
|
|
70
|
+
if event.get("tool_name") != "Bash":
|
|
71
|
+
return PASS
|
|
72
|
+
|
|
73
|
+
given = event.get("tool_input")
|
|
74
|
+
command = given.get("command") if isinstance(given, dict) else None
|
|
75
|
+
if not isinstance(command, str) or not command.strip():
|
|
76
|
+
return PASS
|
|
77
|
+
|
|
78
|
+
try:
|
|
79
|
+
capture = run([command], shell=True)
|
|
80
|
+
view, who = best_view(capture)
|
|
81
|
+
except Exception: # a bug here must not cost the caller their shell
|
|
82
|
+
return PASS
|
|
83
|
+
|
|
84
|
+
said = view.text + "\n\n" + footer(capture, view, who) if view.text else footer(
|
|
85
|
+
capture, view, who
|
|
86
|
+
)
|
|
87
|
+
return {
|
|
88
|
+
"hookSpecificOutput": {
|
|
89
|
+
"hookEventName": "PreToolUse",
|
|
90
|
+
"permissionDecision": "deny",
|
|
91
|
+
"permissionDecisionReason": said,
|
|
92
|
+
}
|
|
93
|
+
}
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
# Where the client keeps the settings this would be written into, and the one
|
|
97
|
+
# line that would be written. `SIFT_SETTINGS` moves it, which is how this is
|
|
98
|
+
# tested without touching the file somebody actually uses.
|
|
99
|
+
SETTINGS = "~/.claude/settings.json"
|
|
100
|
+
COMMAND = "sift hook"
|
|
101
|
+
EVENT = "PreToolUse"
|
|
102
|
+
MATCHER = "Bash"
|
|
103
|
+
|
|
104
|
+
# What somebody is told before they are asked. Everything it gives and
|
|
105
|
+
# everything it costs, in the order somebody deciding would want them.
|
|
106
|
+
OFFER = """\
|
|
107
|
+
sift can also catch the shell commands the client runs on its own.
|
|
108
|
+
|
|
109
|
+
Right now sift only sees what you or the client explicitly hand it. A coding
|
|
110
|
+
agent runs a great deal of shell besides that, and all of it lands in the
|
|
111
|
+
conversation whole -- and is re-sent on every turn after.
|
|
112
|
+
|
|
113
|
+
With this on, every shell command the client runs goes through sift first: it
|
|
114
|
+
runs the command, keeps every byte, and hands back the lines that mattered.
|
|
115
|
+
Nothing decides which commands are "worth" catching, because how much a command
|
|
116
|
+
prints is not knowable before it runs. Twelve lines in, twelve lines out.
|
|
117
|
+
|
|
118
|
+
What it costs, honestly:
|
|
119
|
+
|
|
120
|
+
* A command that outruns the client's hook timeout is killed there, and the
|
|
121
|
+
client then runs it itself -- so a very long command can run twice. Keep
|
|
122
|
+
this in mind for anything that should not happen twice.
|
|
123
|
+
* Every caught command costs one model request.
|
|
124
|
+
|
|
125
|
+
It fails open: a bug in it leaves your shell exactly as it was, and
|
|
126
|
+
`SIFT_HOOK=0` switches it off without touching your settings again.
|
|
127
|
+
|
|
128
|
+
This would add one line to {where}:
|
|
129
|
+
|
|
130
|
+
{event} / {matcher} -> {command}
|
|
131
|
+
|
|
132
|
+
Nothing already in that file is changed or removed."""
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
def settings_file() -> Path:
|
|
136
|
+
"""The settings file this writes into, with `SIFT_SETTINGS` overriding."""
|
|
137
|
+
return Path(os.environ.get("SIFT_SETTINGS") or SETTINGS).expanduser()
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def offer() -> str:
|
|
141
|
+
"""The notice, with the real paths filled in."""
|
|
142
|
+
return OFFER.format(
|
|
143
|
+
where=settings_file(), event=EVENT, matcher=MATCHER, command=COMMAND
|
|
144
|
+
)
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _entries(settings: dict) -> list | None:
|
|
148
|
+
"""The list this would be added to, or None if the file is not that shape.
|
|
149
|
+
|
|
150
|
+
Refusing an unfamiliar shape rather than reshaping it is the whole of the
|
|
151
|
+
safety here. This writes into a file somebody else owns, which may hold
|
|
152
|
+
hooks they depend on; a merge that is not certain what it is merging into
|
|
153
|
+
should not merge.
|
|
154
|
+
"""
|
|
155
|
+
hooks = settings.get("hooks", {})
|
|
156
|
+
if not isinstance(hooks, dict):
|
|
157
|
+
return None
|
|
158
|
+
found = hooks.setdefault(EVENT, [])
|
|
159
|
+
return found if isinstance(found, list) else None
|
|
160
|
+
|
|
161
|
+
|
|
162
|
+
def installed(settings: dict | None = None) -> bool:
|
|
163
|
+
"""Whether the client is already routing its shell commands here."""
|
|
164
|
+
if settings is None:
|
|
165
|
+
settings = read_settings() or {}
|
|
166
|
+
entries = _entries(dict(settings))
|
|
167
|
+
if entries is None:
|
|
168
|
+
return False
|
|
169
|
+
for entry in entries:
|
|
170
|
+
if not isinstance(entry, dict):
|
|
171
|
+
continue
|
|
172
|
+
for one in entry.get("hooks", []) or []:
|
|
173
|
+
if isinstance(one, dict) and COMMAND in str(one.get("command", "")):
|
|
174
|
+
return True
|
|
175
|
+
return False
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def read_settings() -> dict | None:
|
|
179
|
+
"""What is in the settings file, {} if there is none, None if it is not JSON."""
|
|
180
|
+
path = settings_file()
|
|
181
|
+
if not path.is_file():
|
|
182
|
+
return {}
|
|
183
|
+
try:
|
|
184
|
+
found = json.loads(path.read_text(encoding="utf-8"))
|
|
185
|
+
except (OSError, json.JSONDecodeError):
|
|
186
|
+
return None
|
|
187
|
+
return found if isinstance(found, dict) else None
|
|
188
|
+
|
|
189
|
+
|
|
190
|
+
def install() -> tuple[bool, str]:
|
|
191
|
+
"""Add the routing, without disturbing anything already there.
|
|
192
|
+
|
|
193
|
+
A copy of the original is kept beside it the first time, because this edits
|
|
194
|
+
a file this tool does not own and did not write.
|
|
195
|
+
"""
|
|
196
|
+
settings = read_settings()
|
|
197
|
+
if settings is None:
|
|
198
|
+
return False, f"sift: {settings_file()} is not JSON this can add to safely."
|
|
199
|
+
if installed(settings):
|
|
200
|
+
return True, "sift: the shell is already routed here. Nothing to do."
|
|
201
|
+
|
|
202
|
+
entries = _entries(settings)
|
|
203
|
+
if entries is None:
|
|
204
|
+
return False, f"sift: {settings_file()} has hooks in a shape this cannot merge."
|
|
205
|
+
|
|
206
|
+
path = settings_file()
|
|
207
|
+
backup = path.with_suffix(path.suffix + ".before-sift")
|
|
208
|
+
if path.is_file() and not backup.exists():
|
|
209
|
+
with contextlib.suppress(OSError):
|
|
210
|
+
backup.write_text(path.read_text(encoding="utf-8"), encoding="utf-8")
|
|
211
|
+
|
|
212
|
+
for entry in entries:
|
|
213
|
+
if isinstance(entry, dict) and entry.get("matcher") == MATCHER:
|
|
214
|
+
mine = entry.setdefault("hooks", [])
|
|
215
|
+
if isinstance(mine, list):
|
|
216
|
+
mine.append({"type": "command", "command": COMMAND})
|
|
217
|
+
break
|
|
218
|
+
return False, f"sift: the {MATCHER} entry has hooks this cannot merge."
|
|
219
|
+
else:
|
|
220
|
+
entries.append(
|
|
221
|
+
{"matcher": MATCHER, "hooks": [{"type": "command", "command": COMMAND}]}
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
settings.setdefault("hooks", {})[EVENT] = entries
|
|
225
|
+
try:
|
|
226
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
227
|
+
path.write_text(
|
|
228
|
+
json.dumps(settings, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
229
|
+
)
|
|
230
|
+
except OSError as exc:
|
|
231
|
+
return False, f"sift: could not write {path} ({exc})"
|
|
232
|
+
|
|
233
|
+
kept = f" The file as it was is in {backup.name}." if backup.exists() else ""
|
|
234
|
+
return True, (
|
|
235
|
+
f"sift: the shell is routed here now. Restart the client for it to take"
|
|
236
|
+
f" effect.{kept}\n"
|
|
237
|
+
f" Undo with `sift hook --uninstall`, or switch it off for one"
|
|
238
|
+
f" session with SIFT_HOOK=0."
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
|
|
242
|
+
def uninstall() -> tuple[bool, str]:
|
|
243
|
+
"""Take the routing out again, and leave everything else exactly as it was."""
|
|
244
|
+
settings = read_settings()
|
|
245
|
+
if settings is None:
|
|
246
|
+
return False, f"sift: {settings_file()} is not JSON this can edit safely."
|
|
247
|
+
if not installed(settings):
|
|
248
|
+
return True, "sift: the shell was not routed here. Nothing to do."
|
|
249
|
+
|
|
250
|
+
entries = _entries(settings) or []
|
|
251
|
+
for entry in entries:
|
|
252
|
+
if not isinstance(entry, dict):
|
|
253
|
+
continue
|
|
254
|
+
mine = entry.get("hooks")
|
|
255
|
+
if isinstance(mine, list):
|
|
256
|
+
entry["hooks"] = [
|
|
257
|
+
one
|
|
258
|
+
for one in mine
|
|
259
|
+
if not (isinstance(one, dict) and COMMAND in str(one.get("command", "")))
|
|
260
|
+
]
|
|
261
|
+
# An entry whose only hook was this one is removed; one that had others keeps
|
|
262
|
+
# them. Leaving an empty matcher behind would be leaving litter in somebody
|
|
263
|
+
# else's file.
|
|
264
|
+
settings["hooks"][EVENT] = [
|
|
265
|
+
entry
|
|
266
|
+
for entry in entries
|
|
267
|
+
if not (isinstance(entry, dict) and entry.get("hooks") == [])
|
|
268
|
+
]
|
|
269
|
+
try:
|
|
270
|
+
settings_file().write_text(
|
|
271
|
+
json.dumps(settings, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
|
|
272
|
+
)
|
|
273
|
+
except OSError as exc:
|
|
274
|
+
return False, f"sift: could not write {settings_file()} ({exc})"
|
|
275
|
+
return True, "sift: the shell is no longer routed here."
|
sift/lines.py
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""What counts as a line.
|
|
2
|
+
|
|
3
|
+
`str.splitlines()` splits on more than anyone else does: form feed, vertical
|
|
4
|
+
tab, the file and group separators, NEL (U+0085), and the Unicode line and
|
|
5
|
+
paragraph separators (U+2028, U+2029). Every one of those turns up in real
|
|
6
|
+
command output. U+0085 comes out of EBCDIC conversions, which is how a mainframe
|
|
7
|
+
job log reaches a terminal. U+2028 comes out of tooling that pasted a string it
|
|
8
|
+
read from somewhere else. A form feed is how a good deal of older software
|
|
9
|
+
starts a new page.
|
|
10
|
+
|
|
11
|
+
When they turn up, a capture gains lines that nothing else agrees exist. `wc -l`
|
|
12
|
+
would not count them, the terminal did not show them, and `sift peek 40 50`
|
|
13
|
+
would answer with different text than the editor open beside it. For a tool
|
|
14
|
+
whose whole promise is that the line you were shown is the line that is there,
|
|
15
|
+
that is not a small disagreement.
|
|
16
|
+
|
|
17
|
+
So a line here ends at a newline, and at nothing else. A carriage return before
|
|
18
|
+
it belongs to the ending rather than to the content -- which is what a terminal
|
|
19
|
+
does with CRLF, and what every other tool means by a line.
|
|
20
|
+
|
|
21
|
+
A carriage return anywhere *else* is left exactly where it is. A progress bar
|
|
22
|
+
that redraws itself nine hundred times is one line in the terminal, and it is
|
|
23
|
+
one line here too, rather than nine hundred lines of a number counting up.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def of(text: str) -> list[str]:
|
|
30
|
+
"""The lines of `text`, counted the way the rest of the world counts them."""
|
|
31
|
+
if not text:
|
|
32
|
+
return []
|
|
33
|
+
found = text.split("\n")
|
|
34
|
+
if found[-1] == "":
|
|
35
|
+
# A trailing newline ends the last line; it does not begin another one.
|
|
36
|
+
found.pop()
|
|
37
|
+
return [line[:-1] if line.endswith("\r") else line for line in found]
|
sift/many.py
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
"""Several questions at once, and the one number that keeps them honest.
|
|
2
|
+
|
|
3
|
+
Everything else in this tool answers one question about one thing. This runs
|
|
4
|
+
several of those side by side: four log files digested together, three running
|
|
5
|
+
commands read in one call, a directory outlined in the time one file used to
|
|
6
|
+
take.
|
|
7
|
+
|
|
8
|
+
The whole module is a dozen lines, because the hard part is not here. Asking in
|
|
9
|
+
parallel is easy; asking in parallel *without multiplying the load* is what took
|
|
10
|
+
a day to learn, and that lesson lives one layer down. Every ask in this project
|
|
11
|
+
passes through `distill.in_flight`, a single gate sized by `SIFT_WORKERS`. So
|
|
12
|
+
this file may start as many jobs as it likes and the number of requests actually
|
|
13
|
+
in the air is still the number the machine was told to allow.
|
|
14
|
+
|
|
15
|
+
That division of labour is deliberate. A ceiling that each caller has to work
|
|
16
|
+
out for itself is a ceiling that the next caller will get wrong -- and the next
|
|
17
|
+
caller is always the one written six months later by someone who never read the
|
|
18
|
+
note. The gate cannot be got wrong by arithmetic, because there is none to do.
|
|
19
|
+
|
|
20
|
+
What this file does own is the promise that **the answers come back in the order
|
|
21
|
+
the questions were asked**. Which of them finishes first is a fact about the
|
|
22
|
+
network, and a result that reordered itself by network timing would be a
|
|
23
|
+
different answer to the same question every time it was asked.
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
from collections.abc import Callable, Sequence
|
|
29
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
30
|
+
|
|
31
|
+
from sift.distill import workers
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def together[T](jobs: Sequence[Callable[[], T]]) -> list[T]:
|
|
35
|
+
"""Run every job at the same time and return what they returned, in order.
|
|
36
|
+
|
|
37
|
+
One job is run directly rather than through a pool, because a pool for one
|
|
38
|
+
job is a thread and a queue doing what a call does.
|
|
39
|
+
|
|
40
|
+
An exception from a job comes back out of here, at the position that job
|
|
41
|
+
would have had. Swallowing it would leave the caller a list with a hole in
|
|
42
|
+
it and nothing to say what happened -- and every caller above this one
|
|
43
|
+
already has a net of its own that turns a failure into a view.
|
|
44
|
+
"""
|
|
45
|
+
if not jobs:
|
|
46
|
+
return []
|
|
47
|
+
if len(jobs) == 1:
|
|
48
|
+
return [jobs[0]()]
|
|
49
|
+
|
|
50
|
+
with ThreadPoolExecutor(max_workers=min(workers(), len(jobs))) as pool:
|
|
51
|
+
return list(pool.map(lambda job: job(), jobs))
|
sift/memory.py
ADDED
|
@@ -0,0 +1,94 @@
|
|
|
1
|
+
"""What this machine already knows about the commands run on it.
|
|
2
|
+
|
|
3
|
+
Every other module here asks a model something. This one asks nothing, and that
|
|
4
|
+
is the design rather than an omission.
|
|
5
|
+
|
|
6
|
+
The question is *how often was this run here, how did it go, and which failures
|
|
7
|
+
keep coming back* -- and every part of that is counting. A model asked to count
|
|
8
|
+
would be slower, cost a request, and be wrong sometimes; there is no judgement
|
|
9
|
+
in the question for it to supply. Reaching for it anyway would be the same
|
|
10
|
+
mistake as a rules engine deciding which lines matter: using the wrong tool
|
|
11
|
+
because it is the one the project is proud of.
|
|
12
|
+
|
|
13
|
+
So the rule this file stands on is the mirror of the rest of the project: **the
|
|
14
|
+
model decides what cannot be computed, and nothing else.**
|
|
15
|
+
|
|
16
|
+
What it can answer is bounded by what is still on disk. A capture that was swept
|
|
17
|
+
up took its record with it, because the record lives beside the bytes rather
|
|
18
|
+
than in a ledger of its own -- which is what makes deleting a capture actually
|
|
19
|
+
delete it. This module says how far back it can see rather than implying it
|
|
20
|
+
remembers everything.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
from collections.abc import Sequence
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
|
|
28
|
+
from sift import store
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True)
|
|
32
|
+
class Habit:
|
|
33
|
+
"""One command, and everything this machine has seen of it."""
|
|
34
|
+
|
|
35
|
+
command: str
|
|
36
|
+
runs: int
|
|
37
|
+
failures: int
|
|
38
|
+
last_ending: str
|
|
39
|
+
last_at: float
|
|
40
|
+
cwds: tuple[str, ...]
|
|
41
|
+
|
|
42
|
+
@property
|
|
43
|
+
def never_worked(self) -> bool:
|
|
44
|
+
"""True when every run of this command here has failed.
|
|
45
|
+
|
|
46
|
+
Worth its own name because it is the one shape that answers a question
|
|
47
|
+
somebody actually asks: *is this thing broken, or is it me?* A command
|
|
48
|
+
that has failed four times out of four in this directory is not a flaky
|
|
49
|
+
test.
|
|
50
|
+
"""
|
|
51
|
+
return self.runs > 0 and self.failures == self.runs
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
def habits(term: str | None = None, cwd: str | None = None, limit: int = 200) -> list[Habit]:
|
|
55
|
+
"""What has been run here, most recent first.
|
|
56
|
+
|
|
57
|
+
`term` narrows to commands containing it, `cwd` to a directory. Both are
|
|
58
|
+
plain text rather than patterns: this is a memory, not a search engine, and
|
|
59
|
+
somebody typing `pytest` should not have to think about what the letters
|
|
60
|
+
mean to a regular expression.
|
|
61
|
+
"""
|
|
62
|
+
seen: dict[str, list[store.Meta]] = {}
|
|
63
|
+
for meta in store.recent(limit):
|
|
64
|
+
written = " ".join(meta.command)
|
|
65
|
+
if term and term not in written:
|
|
66
|
+
continue
|
|
67
|
+
if cwd and meta.cwd != cwd:
|
|
68
|
+
continue
|
|
69
|
+
seen.setdefault(written, []).append(meta)
|
|
70
|
+
|
|
71
|
+
found = [_habit(written, runs) for written, runs in seen.items()]
|
|
72
|
+
found.sort(key=lambda one: one.last_at, reverse=True)
|
|
73
|
+
return found
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def _habit(written: str, runs: Sequence[store.Meta]) -> Habit:
|
|
77
|
+
newest = max(runs, key=lambda meta: meta.started_at)
|
|
78
|
+
return Habit(
|
|
79
|
+
command=written,
|
|
80
|
+
runs=len(runs),
|
|
81
|
+
failures=sum(1 for meta in runs if meta.failed),
|
|
82
|
+
last_ending="timed out" if newest.timed_out else f"exit {newest.exit_code}",
|
|
83
|
+
last_at=newest.started_at,
|
|
84
|
+
cwds=tuple(sorted({meta.cwd for meta in runs})),
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def reach(limit: int = 200) -> int:
|
|
89
|
+
"""How many runs are still on disk to remember at all.
|
|
90
|
+
|
|
91
|
+
Printed with the answer, because "this has never failed here" means one thing
|
|
92
|
+
after four hundred runs and nothing at all after two.
|
|
93
|
+
"""
|
|
94
|
+
return len(store.recent(limit))
|