sift-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sift/fallback.py ADDED
@@ -0,0 +1,98 @@
1
+ """What to show when nobody could be asked.
2
+
3
+ There is no key, or no network, or the endpoint is busy, or the reply made no
4
+ sense. The command has already run and its output is already on disk. Something
5
+ has to be shown, and it has to be shown without a model.
6
+
7
+ The temptation here is to be clever: look for the word "error", recognise a
8
+ stack trace, score lines by how alarming they look. That is exactly the design
9
+ this project was rewritten to get rid of. Patterns like those are a list of
10
+ languages wearing a disguise -- they work for English and for the half-dozen
11
+ formats whoever wrote them happened to know, and they quietly fail for Turkish,
12
+ for Japanese, for a tool that shipped last week. Worse, they fail while looking
13
+ confident.
14
+
15
+ So this makes no claim about meaning at all. It shows the beginning and the end
16
+ and marks what it skipped. The beginning says what was run; the end says how it
17
+ came out. That is true of a compiler, a test runner, an installer, a shell
18
+ script, in every language a person or a machine writes in, and it needs to know
19
+ nothing about any of them.
20
+
21
+ It is worse than a model. It is meant to be. What it must never be is wrong, and
22
+ a rule that makes no claim cannot make a false one.
23
+ """
24
+
25
+ from __future__ import annotations
26
+
27
+ from sift import lines as text_lines
28
+ from sift import records
29
+ from sift.capture import Capture
30
+ from sift.distill import View, render
31
+
32
+ # The first lines are usually the invocation and the first thing to go wrong.
33
+ HEAD = 10
34
+
35
+ # The last lines are the result: the summary, the exit, the reason. In every
36
+ # command-line tradition, the ending is where the answer lives.
37
+ TAIL = 40
38
+
39
+ # A run that failed says why near the end, and tends to say more of it: a
40
+ # traceback, a diagnostic, a list of what did not build.
41
+ TAIL_WHEN_FAILED = 80
42
+
43
+
44
+ def ends(total: int, *, head: int = HEAD, tail: int = TAIL) -> set[int]:
45
+ """The first `head` and the last `tail` lines, or all of them if that is fewer."""
46
+ if total <= head + tail:
47
+ return set(range(1, total + 1))
48
+ return set(range(1, head + 1)) | set(range(total - tail + 1, total + 1))
49
+
50
+
51
+ def from_lines(
52
+ lines: list[str],
53
+ handle: str,
54
+ *,
55
+ tail: int = TAIL,
56
+ first: int = 1,
57
+ unit: str = "line",
58
+ ) -> View:
59
+ """The ends of any numbered text, marked with what lies between them.
60
+
61
+ A source file gets the same treatment as a capture, and for the same reason:
62
+ the moment this function starts telling the two apart it has begun keeping a
63
+ list of what things are, which is the list this project was rewritten to be
64
+ rid of. The beginning and the end of a file are a poor outline. They are not
65
+ a wrong one.
66
+ """
67
+ total = len(lines)
68
+ if not total:
69
+ return View(handle, "", 0, 0, None, 0, unit=unit)
70
+
71
+ # `ends` counts from one because it is arithmetic about a length, not about
72
+ # a capture. Where these lines sit in the run is the caller's business, and
73
+ # it is added here, once, rather than taught to the arithmetic.
74
+ chosen = {number + first - 1 for number in ends(total, head=HEAD, tail=tail)}
75
+ return View(
76
+ handle=handle,
77
+ text=render(lines, chosen, handle, first, unit),
78
+ kept=len(chosen),
79
+ total=total,
80
+ model=None,
81
+ asks=0,
82
+ unit=unit,
83
+ )
84
+
85
+
86
+ def fallback(capture: Capture) -> View:
87
+ """A view built without asking anything, and honest about being one.
88
+
89
+ `model` is left empty, which is how the caller can tell this apart from a
90
+ view a model chose. A reader who is told which lines were picked, and by
91
+ what, can decide whether to go and read the rest.
92
+ """
93
+ tail = TAIL_WHEN_FAILED if capture.meta.failed else TAIL
94
+ text = capture.text()
95
+ found = records.of(text)
96
+ if found is not None:
97
+ return from_lines(found, capture.handle, tail=tail, unit="record")
98
+ return from_lines(text_lines.of(text), capture.handle, tail=tail)
sift/hook.py ADDED
@@ -0,0 +1,275 @@
1
+ """Catching the shell commands a client runs on its own.
2
+
3
+ The MCP server can only distil what it was asked to distil. Everything else a
4
+ client does with a shell -- and a coding agent does a great deal -- lands in the
5
+ conversation whole. This closes that gap without a proxy: the client is asked to
6
+ send its shell commands here first, `sift` runs them, and what comes back is a
7
+ view.
8
+
9
+ Two decisions, and the first is the one worth arguing about.
10
+
11
+ **Everything is routed. Nothing decides whether a command "looks noisy".**
12
+
13
+ That rule was tempting and it is exactly the mistake this project was rewritten
14
+ to avoid. A list of commands worth intercepting is a list of tools wearing a
15
+ disguise: it would know `pytest` and `cargo` and `npm`, be wrong about the
16
+ in-house script, and be confidently silent about the one that printed forty
17
+ thousand lines. And it cannot be right in principle -- how much a command prints
18
+ is not knowable before it runs.
19
+
20
+ Routing everything costs nothing, because a view of a short output *is* that
21
+ output: the budget only takes hold when there is more than the budget. Twelve
22
+ lines in, twelve lines out.
23
+
24
+ **It fails open, and that is the third rule again.** A bug here, an unreadable
25
+ event, a command that could not be started -- every one of them ends with the
26
+ client running the command itself, exactly as it would have. A gate that breaks
27
+ a shell is worse than no gate, and this one is allowed to break.
28
+
29
+ `SIFT_HOOK=0` turns it off for someone who wants their shell back untouched.
30
+ """
31
+
32
+ from __future__ import annotations
33
+
34
+ import contextlib
35
+ import json
36
+ import os
37
+ from pathlib import Path
38
+
39
+ from sift.capture import run
40
+ from sift.view import best_view, footer
41
+
42
+ # What this returns when it has nothing to say. An empty answer means "carry on
43
+ # as you were", which is the only safe thing to say when something went wrong.
44
+ PASS = {}
45
+
46
+
47
+ def wanted() -> bool:
48
+ """Whether shell commands should be routed here at all."""
49
+ return (os.environ.get("SIFT_HOOK") or "1").strip().lower() not in {
50
+ "0",
51
+ "false",
52
+ "no",
53
+ "off",
54
+ "",
55
+ }
56
+
57
+
58
+ def answer(event: dict) -> dict:
59
+ """What to tell the client about one shell command it is about to run.
60
+
61
+ The event is whatever the client put on stdin, and it is treated as data
62
+ from somewhere else: every field is checked before it is used, and anything
63
+ unexpected means *carry on*, never an error. A hook that raises on a shape
64
+ it did not expect takes the shell down with it.
65
+ """
66
+ if not wanted():
67
+ return PASS
68
+ if not isinstance(event, dict):
69
+ return PASS
70
+ if event.get("tool_name") != "Bash":
71
+ return PASS
72
+
73
+ given = event.get("tool_input")
74
+ command = given.get("command") if isinstance(given, dict) else None
75
+ if not isinstance(command, str) or not command.strip():
76
+ return PASS
77
+
78
+ try:
79
+ capture = run([command], shell=True)
80
+ view, who = best_view(capture)
81
+ except Exception: # a bug here must not cost the caller their shell
82
+ return PASS
83
+
84
+ said = view.text + "\n\n" + footer(capture, view, who) if view.text else footer(
85
+ capture, view, who
86
+ )
87
+ return {
88
+ "hookSpecificOutput": {
89
+ "hookEventName": "PreToolUse",
90
+ "permissionDecision": "deny",
91
+ "permissionDecisionReason": said,
92
+ }
93
+ }
94
+
95
+
96
+ # Where the client keeps the settings this would be written into, and the one
97
+ # line that would be written. `SIFT_SETTINGS` moves it, which is how this is
98
+ # tested without touching the file somebody actually uses.
99
+ SETTINGS = "~/.claude/settings.json"
100
+ COMMAND = "sift hook"
101
+ EVENT = "PreToolUse"
102
+ MATCHER = "Bash"
103
+
104
+ # What somebody is told before they are asked. Everything it gives and
105
+ # everything it costs, in the order somebody deciding would want them.
106
+ OFFER = """\
107
+ sift can also catch the shell commands the client runs on its own.
108
+
109
+ Right now sift only sees what you or the client explicitly hand it. A coding
110
+ agent runs a great deal of shell besides that, and all of it lands in the
111
+ conversation whole -- and is re-sent on every turn after.
112
+
113
+ With this on, every shell command the client runs goes through sift first: it
114
+ runs the command, keeps every byte, and hands back the lines that mattered.
115
+ Nothing decides which commands are "worth" catching, because how much a command
116
+ prints is not knowable before it runs. Twelve lines in, twelve lines out.
117
+
118
+ What it costs, honestly:
119
+
120
+ * A command that outruns the client's hook timeout is killed there, and the
121
+ client then runs it itself -- so a very long command can run twice. Keep
122
+ this in mind for anything that should not happen twice.
123
+ * Every caught command costs one model request.
124
+
125
+ It fails open: a bug in it leaves your shell exactly as it was, and
126
+ `SIFT_HOOK=0` switches it off without touching your settings again.
127
+
128
+ This would add one line to {where}:
129
+
130
+ {event} / {matcher} -> {command}
131
+
132
+ Nothing already in that file is changed or removed."""
133
+
134
+
135
+ def settings_file() -> Path:
136
+ """The settings file this writes into, with `SIFT_SETTINGS` overriding."""
137
+ return Path(os.environ.get("SIFT_SETTINGS") or SETTINGS).expanduser()
138
+
139
+
140
+ def offer() -> str:
141
+ """The notice, with the real paths filled in."""
142
+ return OFFER.format(
143
+ where=settings_file(), event=EVENT, matcher=MATCHER, command=COMMAND
144
+ )
145
+
146
+
147
+ def _entries(settings: dict) -> list | None:
148
+ """The list this would be added to, or None if the file is not that shape.
149
+
150
+ Refusing an unfamiliar shape rather than reshaping it is the whole of the
151
+ safety here. This writes into a file somebody else owns, which may hold
152
+ hooks they depend on; a merge that is not certain what it is merging into
153
+ should not merge.
154
+ """
155
+ hooks = settings.get("hooks", {})
156
+ if not isinstance(hooks, dict):
157
+ return None
158
+ found = hooks.setdefault(EVENT, [])
159
+ return found if isinstance(found, list) else None
160
+
161
+
162
+ def installed(settings: dict | None = None) -> bool:
163
+ """Whether the client is already routing its shell commands here."""
164
+ if settings is None:
165
+ settings = read_settings() or {}
166
+ entries = _entries(dict(settings))
167
+ if entries is None:
168
+ return False
169
+ for entry in entries:
170
+ if not isinstance(entry, dict):
171
+ continue
172
+ for one in entry.get("hooks", []) or []:
173
+ if isinstance(one, dict) and COMMAND in str(one.get("command", "")):
174
+ return True
175
+ return False
176
+
177
+
178
+ def read_settings() -> dict | None:
179
+ """What is in the settings file, {} if there is none, None if it is not JSON."""
180
+ path = settings_file()
181
+ if not path.is_file():
182
+ return {}
183
+ try:
184
+ found = json.loads(path.read_text(encoding="utf-8"))
185
+ except (OSError, json.JSONDecodeError):
186
+ return None
187
+ return found if isinstance(found, dict) else None
188
+
189
+
190
+ def install() -> tuple[bool, str]:
191
+ """Add the routing, without disturbing anything already there.
192
+
193
+ A copy of the original is kept beside it the first time, because this edits
194
+ a file this tool does not own and did not write.
195
+ """
196
+ settings = read_settings()
197
+ if settings is None:
198
+ return False, f"sift: {settings_file()} is not JSON this can add to safely."
199
+ if installed(settings):
200
+ return True, "sift: the shell is already routed here. Nothing to do."
201
+
202
+ entries = _entries(settings)
203
+ if entries is None:
204
+ return False, f"sift: {settings_file()} has hooks in a shape this cannot merge."
205
+
206
+ path = settings_file()
207
+ backup = path.with_suffix(path.suffix + ".before-sift")
208
+ if path.is_file() and not backup.exists():
209
+ with contextlib.suppress(OSError):
210
+ backup.write_text(path.read_text(encoding="utf-8"), encoding="utf-8")
211
+
212
+ for entry in entries:
213
+ if isinstance(entry, dict) and entry.get("matcher") == MATCHER:
214
+ mine = entry.setdefault("hooks", [])
215
+ if isinstance(mine, list):
216
+ mine.append({"type": "command", "command": COMMAND})
217
+ break
218
+ return False, f"sift: the {MATCHER} entry has hooks this cannot merge."
219
+ else:
220
+ entries.append(
221
+ {"matcher": MATCHER, "hooks": [{"type": "command", "command": COMMAND}]}
222
+ )
223
+
224
+ settings.setdefault("hooks", {})[EVENT] = entries
225
+ try:
226
+ path.parent.mkdir(parents=True, exist_ok=True)
227
+ path.write_text(
228
+ json.dumps(settings, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
229
+ )
230
+ except OSError as exc:
231
+ return False, f"sift: could not write {path} ({exc})"
232
+
233
+ kept = f" The file as it was is in {backup.name}." if backup.exists() else ""
234
+ return True, (
235
+ f"sift: the shell is routed here now. Restart the client for it to take"
236
+ f" effect.{kept}\n"
237
+ f" Undo with `sift hook --uninstall`, or switch it off for one"
238
+ f" session with SIFT_HOOK=0."
239
+ )
240
+
241
+
242
+ def uninstall() -> tuple[bool, str]:
243
+ """Take the routing out again, and leave everything else exactly as it was."""
244
+ settings = read_settings()
245
+ if settings is None:
246
+ return False, f"sift: {settings_file()} is not JSON this can edit safely."
247
+ if not installed(settings):
248
+ return True, "sift: the shell was not routed here. Nothing to do."
249
+
250
+ entries = _entries(settings) or []
251
+ for entry in entries:
252
+ if not isinstance(entry, dict):
253
+ continue
254
+ mine = entry.get("hooks")
255
+ if isinstance(mine, list):
256
+ entry["hooks"] = [
257
+ one
258
+ for one in mine
259
+ if not (isinstance(one, dict) and COMMAND in str(one.get("command", "")))
260
+ ]
261
+ # An entry whose only hook was this one is removed; one that had others keeps
262
+ # them. Leaving an empty matcher behind would be leaving litter in somebody
263
+ # else's file.
264
+ settings["hooks"][EVENT] = [
265
+ entry
266
+ for entry in entries
267
+ if not (isinstance(entry, dict) and entry.get("hooks") == [])
268
+ ]
269
+ try:
270
+ settings_file().write_text(
271
+ json.dumps(settings, ensure_ascii=False, indent=2) + "\n", encoding="utf-8"
272
+ )
273
+ except OSError as exc:
274
+ return False, f"sift: could not write {settings_file()} ({exc})"
275
+ return True, "sift: the shell is no longer routed here."
sift/lines.py ADDED
@@ -0,0 +1,37 @@
1
+ """What counts as a line.
2
+
3
+ `str.splitlines()` splits on more than anyone else does: form feed, vertical
4
+ tab, the file and group separators, NEL (U+0085), and the Unicode line and
5
+ paragraph separators (U+2028, U+2029). Every one of those turns up in real
6
+ command output. U+0085 comes out of EBCDIC conversions, which is how a mainframe
7
+ job log reaches a terminal. U+2028 comes out of tooling that pasted a string it
8
+ read from somewhere else. A form feed is how a good deal of older software
9
+ starts a new page.
10
+
11
+ When they turn up, a capture gains lines that nothing else agrees exist. `wc -l`
12
+ would not count them, the terminal did not show them, and `sift peek 40 50`
13
+ would answer with different text than the editor open beside it. For a tool
14
+ whose whole promise is that the line you were shown is the line that is there,
15
+ that is not a small disagreement.
16
+
17
+ So a line here ends at a newline, and at nothing else. A carriage return before
18
+ it belongs to the ending rather than to the content -- which is what a terminal
19
+ does with CRLF, and what every other tool means by a line.
20
+
21
+ A carriage return anywhere *else* is left exactly where it is. A progress bar
22
+ that redraws itself nine hundred times is one line in the terminal, and it is
23
+ one line here too, rather than nine hundred lines of a number counting up.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+
29
+ def of(text: str) -> list[str]:
30
+ """The lines of `text`, counted the way the rest of the world counts them."""
31
+ if not text:
32
+ return []
33
+ found = text.split("\n")
34
+ if found[-1] == "":
35
+ # A trailing newline ends the last line; it does not begin another one.
36
+ found.pop()
37
+ return [line[:-1] if line.endswith("\r") else line for line in found]
sift/many.py ADDED
@@ -0,0 +1,51 @@
1
+ """Several questions at once, and the one number that keeps them honest.
2
+
3
+ Everything else in this tool answers one question about one thing. This runs
4
+ several of those side by side: four log files digested together, three running
5
+ commands read in one call, a directory outlined in the time one file used to
6
+ take.
7
+
8
+ The whole module is a dozen lines, because the hard part is not here. Asking in
9
+ parallel is easy; asking in parallel *without multiplying the load* is what took
10
+ a day to learn, and that lesson lives one layer down. Every ask in this project
11
+ passes through `distill.in_flight`, a single gate sized by `SIFT_WORKERS`. So
12
+ this file may start as many jobs as it likes and the number of requests actually
13
+ in the air is still the number the machine was told to allow.
14
+
15
+ That division of labour is deliberate. A ceiling that each caller has to work
16
+ out for itself is a ceiling that the next caller will get wrong -- and the next
17
+ caller is always the one written six months later by someone who never read the
18
+ note. The gate cannot be got wrong by arithmetic, because there is none to do.
19
+
20
+ What this file does own is the promise that **the answers come back in the order
21
+ the questions were asked**. Which of them finishes first is a fact about the
22
+ network, and a result that reordered itself by network timing would be a
23
+ different answer to the same question every time it was asked.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ from collections.abc import Callable, Sequence
29
+ from concurrent.futures import ThreadPoolExecutor
30
+
31
+ from sift.distill import workers
32
+
33
+
34
+ def together[T](jobs: Sequence[Callable[[], T]]) -> list[T]:
35
+ """Run every job at the same time and return what they returned, in order.
36
+
37
+ One job is run directly rather than through a pool, because a pool for one
38
+ job is a thread and a queue doing what a call does.
39
+
40
+ An exception from a job comes back out of here, at the position that job
41
+ would have had. Swallowing it would leave the caller a list with a hole in
42
+ it and nothing to say what happened -- and every caller above this one
43
+ already has a net of its own that turns a failure into a view.
44
+ """
45
+ if not jobs:
46
+ return []
47
+ if len(jobs) == 1:
48
+ return [jobs[0]()]
49
+
50
+ with ThreadPoolExecutor(max_workers=min(workers(), len(jobs))) as pool:
51
+ return list(pool.map(lambda job: job(), jobs))
sift/memory.py ADDED
@@ -0,0 +1,94 @@
1
+ """What this machine already knows about the commands run on it.
2
+
3
+ Every other module here asks a model something. This one asks nothing, and that
4
+ is the design rather than an omission.
5
+
6
+ The question is *how often was this run here, how did it go, and which failures
7
+ keep coming back* -- and every part of that is counting. A model asked to count
8
+ would be slower, cost a request, and be wrong sometimes; there is no judgement
9
+ in the question for it to supply. Reaching for it anyway would be the same
10
+ mistake as a rules engine deciding which lines matter: using the wrong tool
11
+ because it is the one the project is proud of.
12
+
13
+ So the rule this file stands on is the mirror of the rest of the project: **the
14
+ model decides what cannot be computed, and nothing else.**
15
+
16
+ What it can answer is bounded by what is still on disk. A capture that was swept
17
+ up took its record with it, because the record lives beside the bytes rather
18
+ than in a ledger of its own -- which is what makes deleting a capture actually
19
+ delete it. This module says how far back it can see rather than implying it
20
+ remembers everything.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ from collections.abc import Sequence
26
+ from dataclasses import dataclass
27
+
28
+ from sift import store
29
+
30
+
31
+ @dataclass(frozen=True)
32
+ class Habit:
33
+ """One command, and everything this machine has seen of it."""
34
+
35
+ command: str
36
+ runs: int
37
+ failures: int
38
+ last_ending: str
39
+ last_at: float
40
+ cwds: tuple[str, ...]
41
+
42
+ @property
43
+ def never_worked(self) -> bool:
44
+ """True when every run of this command here has failed.
45
+
46
+ Worth its own name because it is the one shape that answers a question
47
+ somebody actually asks: *is this thing broken, or is it me?* A command
48
+ that has failed four times out of four in this directory is not a flaky
49
+ test.
50
+ """
51
+ return self.runs > 0 and self.failures == self.runs
52
+
53
+
54
+ def habits(term: str | None = None, cwd: str | None = None, limit: int = 200) -> list[Habit]:
55
+ """What has been run here, most recent first.
56
+
57
+ `term` narrows to commands containing it, `cwd` to a directory. Both are
58
+ plain text rather than patterns: this is a memory, not a search engine, and
59
+ somebody typing `pytest` should not have to think about what the letters
60
+ mean to a regular expression.
61
+ """
62
+ seen: dict[str, list[store.Meta]] = {}
63
+ for meta in store.recent(limit):
64
+ written = " ".join(meta.command)
65
+ if term and term not in written:
66
+ continue
67
+ if cwd and meta.cwd != cwd:
68
+ continue
69
+ seen.setdefault(written, []).append(meta)
70
+
71
+ found = [_habit(written, runs) for written, runs in seen.items()]
72
+ found.sort(key=lambda one: one.last_at, reverse=True)
73
+ return found
74
+
75
+
76
+ def _habit(written: str, runs: Sequence[store.Meta]) -> Habit:
77
+ newest = max(runs, key=lambda meta: meta.started_at)
78
+ return Habit(
79
+ command=written,
80
+ runs=len(runs),
81
+ failures=sum(1 for meta in runs if meta.failed),
82
+ last_ending="timed out" if newest.timed_out else f"exit {newest.exit_code}",
83
+ last_at=newest.started_at,
84
+ cwds=tuple(sorted({meta.cwd for meta in runs})),
85
+ )
86
+
87
+
88
+ def reach(limit: int = 200) -> int:
89
+ """How many runs are still on disk to remember at all.
90
+
91
+ Printed with the answer, because "this has never failed here" means one thing
92
+ after four hundred runs and nothing at all after two.
93
+ """
94
+ return len(store.recent(limit))