sift-cli 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
sift/view.py ADDED
@@ -0,0 +1,317 @@
1
+ """What you are looking at, and where it came from.
2
+
3
+ Two front ends now ask the same question. The command line prints the view for a
4
+ person; the MCP server hands it to a model. Neither may keep its own copy of the
5
+ ladder that gets there -- a fallback that exists in one and not the other would
6
+ mean the third rule holds at the terminal and not over the wire, which is the
7
+ same as not holding at all.
8
+
9
+ So the ladder lives here, once, and both callers climb it. Every way of failing
10
+ to reach a model ends at the ends of the text rather than at an error, and there
11
+ is no path out of any function below that does not return something to read.
12
+
13
+ The lines that report on a view live here for the same reason. A view that does
14
+ not say a model was never reached is a view that claims one was, and that claim
15
+ has to be identical wherever it is made.
16
+
17
+ And what a view cost is written down here, on the way past, for a third time the
18
+ same reason: a saving measured at the terminal and not over the wire would be a
19
+ number about one caller published as a number about the tool.
20
+ """
21
+
22
+ from __future__ import annotations
23
+
24
+ from collections.abc import Sequence
25
+
26
+ from sift import store
27
+ from sift.background import alive
28
+ from sift.capture import Capture
29
+ from sift.digest import digest
30
+ from sift.digest import ends_of as digest_ends
31
+ from sift.distill import BUDGET, View, distill, follow, render
32
+ from sift.fallback import fallback, from_lines
33
+ from sift.many import together
34
+ from sift.model import Bridge
35
+ from sift.outline import ends_of, outline
36
+ from sift.peek import Peek
37
+
38
+
39
+ def best_view(
40
+ capture: Capture, budget: int | None = BUDGET, keep: str | None = None
41
+ ) -> tuple[View, str]:
42
+ """The best view of a capture, a word about where it came from, and the bill.
43
+
44
+ A view built without a model is recorded too. It cost the caller whatever it
45
+ cost them, and a report that counted only the good runs would be measuring
46
+ the tool on its best days.
47
+ """
48
+ view, who = _choose(capture, budget, keep)
49
+ store.record(
50
+ store.Saving(
51
+ handle=capture.handle,
52
+ raw_bytes=capture.meta.byte_count,
53
+ shown_bytes=len(view.text.encode("utf-8")),
54
+ kept=view.kept,
55
+ total=view.total,
56
+ model=view.model,
57
+ asks=view.asks,
58
+ unanswered=view.unanswered,
59
+ tokens=view.tokens,
60
+ )
61
+ )
62
+ return view, who
63
+
64
+
65
+ def _choose(
66
+ capture: Capture, budget: int | None = BUDGET, keep: str | None = None
67
+ ) -> tuple[View, str]:
68
+ """Every failure below lands in the same place: the ends of the capture,
69
+ shown without a model. That is the whole safety net."""
70
+ bridge = Bridge()
71
+ reason = "no model"
72
+ try:
73
+ chosen = distill(capture, bridge, budget, keep)
74
+ if chosen is not None:
75
+ return chosen, chose(chosen)
76
+ reason = bridge.last_error or "no lines chosen"
77
+ except Exception as exc: # a bug here must not cost the caller their output
78
+ reason = f"{type(exc).__name__}: {exc}"
79
+ return fallback(capture), f"no model ({reason})"
80
+
81
+
82
+ def best_follow(handle: str, lines: list[str], first: int) -> tuple[View, str]:
83
+ """The best view of what a running command has said since the last look.
84
+
85
+ Shaped like `best_view`, and different in the one way that matters: here an
86
+ empty answer is an answer. A finished capture always has something worth
87
+ showing -- what was run, how it ended -- so `distill` coming back with
88
+ nothing means something went wrong and the ends are shown instead. But a
89
+ minute of a build that printed nothing except progress genuinely has nothing
90
+ in it worth a reader's attention, and saying so plainly is more use than
91
+ showing forty lines of it to prove the tool is awake.
92
+
93
+ The two are told apart by asking the bridge. A model that answered and named
94
+ no lines leaves no error behind; a model that was never reached does.
95
+
96
+ Nothing is recorded. A saving is a claim about a capture, and this is one
97
+ look at a run still going: the run gets recorded when it ends, and counting
98
+ every look as a capture of its own would inflate the only numbers this
99
+ project publishes about itself.
100
+ """
101
+ if not lines:
102
+ return View(handle, "", 0, 0, None, 0), "nothing new"
103
+
104
+ bridge = Bridge()
105
+ reason = "no model"
106
+ try:
107
+ chosen = follow(lines, handle, first, bridge)
108
+ if chosen is not None:
109
+ return chosen, chose(chosen)
110
+ if bridge.last_error is None:
111
+ return _quiet(lines, handle, first), "model (nothing new worth showing)"
112
+ reason = bridge.last_error
113
+ except Exception as exc: # a bug here must not cost the caller their output
114
+ reason = f"{type(exc).__name__}: {exc}"
115
+ return from_lines(lines, handle, first=first), f"no model ({reason})"
116
+
117
+
118
+ def _quiet(lines: list[str], handle: str, first: int) -> View:
119
+ """Lines a model read and chose none of: folded away, and counted as read.
120
+
121
+ `model` is left empty even though a model was reached, because that field
122
+ says which model's choice built this text and no text was built from one.
123
+ What happened is told to the reader by the word that comes back beside it.
124
+ """
125
+ return View(
126
+ handle=handle,
127
+ text=render(lines, set(), handle, first),
128
+ kept=0,
129
+ total=len(lines),
130
+ model=None,
131
+ asks=1,
132
+ )
133
+
134
+
135
+ def best_outline(
136
+ path: str, budget: int | None = None, keep: str | None = None
137
+ ) -> tuple[View, str]:
138
+ """The best outline of a file, and a word about where it came from.
139
+
140
+ Shaped like `best_view` and for the same reason. The one failure allowed
141
+ through is the file being unreadable, because then there is nothing to show
142
+ and saying so is the only honest answer.
143
+ """
144
+ bridge = Bridge()
145
+ reason = "no model"
146
+ try:
147
+ chosen = outline(path, bridge, budget, keep)
148
+ if chosen is not None:
149
+ return chosen, chose(chosen)
150
+ reason = bridge.last_error or "no lines chosen"
151
+ except OSError:
152
+ raise
153
+ except Exception as exc: # a bug here must not cost the caller their outline
154
+ reason = f"{type(exc).__name__}: {exc}"
155
+ return ends_of(path), f"no model ({reason})"
156
+
157
+
158
+ def footer(capture: Capture, view: View, who: str) -> str:
159
+ """One line telling the reader what they are looking at, and what they are not."""
160
+ meta = capture.meta
161
+ return (
162
+ f"sift {capture.handle} · {ending(meta)} · {counted(view)}"
163
+ f" · {who} · {meta.duration_s:.1f}s" + silence(view)
164
+ )
165
+
166
+
167
+ def best_digest(
168
+ path: str, budget: int | None = None, keep: str | None = None
169
+ ) -> tuple[View, str]:
170
+ """The best view of a file somebody else produced, and where it came from.
171
+
172
+ The third of the same shape, and deliberately identical to `best_outline`
173
+ except for the question underneath. Two ladders that differed would mean a
174
+ file read one way behaved differently from a file read the other, which is
175
+ the sort of difference nobody finds until it matters.
176
+ """
177
+ bridge = Bridge()
178
+ reason = "no model"
179
+ try:
180
+ chosen = digest(path, bridge, budget, keep)
181
+ if chosen is not None:
182
+ return chosen, chose(chosen)
183
+ reason = bridge.last_error or "no lines chosen"
184
+ except OSError:
185
+ raise
186
+ except Exception as exc: # a bug here must not cost the caller their file
187
+ reason = f"{type(exc).__name__}: {exc}"
188
+ return digest_ends(path), f"no model ({reason})"
189
+
190
+
191
+ def best_digests(
192
+ paths: Sequence[str], budget: int | None = None, keep: str | None = None
193
+ ) -> list[tuple[str, View, str]]:
194
+ """Several files, asked about at the same time, answered in the order given.
195
+
196
+ The parallelism is worth having because the waiting is the whole cost: four
197
+ files are four requests that sit in a socket, and sending them together
198
+ turns four waits into one. What it does *not* do is send more at once than
199
+ the machine was told to allow -- `distill.in_flight` sees to that, and this
200
+ function deliberately does no arithmetic of its own about it.
201
+
202
+ A file that could not be read comes back as the sentence about it rather
203
+ than taking the other three down with it. One unreadable path in a list of
204
+ four is not a reason to answer nothing.
205
+ """
206
+
207
+ def one(path: str):
208
+ def ask() -> tuple[str, View, str]:
209
+ try:
210
+ built, who = best_digest(path, budget, keep)
211
+ except OSError as exc:
212
+ return path, View(path, "", 0, 0, None, 0), f"unreadable ({exc})"
213
+ return path, built, who
214
+
215
+ return ask
216
+
217
+ return together([one(path) for path in paths])
218
+
219
+
220
+ def ending(meta: store.Meta) -> str:
221
+ """How a run came out, in the few words every part of this tool uses for it.
222
+
223
+ A timeout has no exit code, so it is named rather than printed as an empty
224
+ one. Said in one place because a run that reads `timed out` at the terminal
225
+ and `exit None` in a listing is two tools wearing one name.
226
+ """
227
+ return "timed out" if meta.timed_out else f"exit {meta.exit_code}"
228
+
229
+
230
+ def state(running: store.Running | None, meta: store.Meta | None) -> str:
231
+ """What to call a run right now, in the one or two words both front ends use.
232
+
233
+ `lost` is a word of its own on purpose. A supervisor that is gone without
234
+ having written an ending is not a command that finished: nobody knows how
235
+ that one came out, and printing `exit None` would be claiming otherwise.
236
+ """
237
+ if running is not None:
238
+ return "running" if alive(running) else "lost"
239
+ return ending(meta) if meta is not None else "unknown"
240
+
241
+
242
+ def follow_footer(handle: str, view: View, who: str, first: int, state: str) -> str:
243
+ """Where in a running command this look sits, and whether more is coming.
244
+
245
+ The numbers are the ones the whole capture uses, so `sift peek` on any of
246
+ them lands on the line the reader was actually shown. That is the entire
247
+ reason a run is followed by offset rather than by copying each look into a
248
+ capture of its own.
249
+ """
250
+ if not view.total:
251
+ return f"sift {handle} · {state} · nothing new since the last look"
252
+ last = first + view.total - 1
253
+ return (
254
+ f"sift {handle} · {state} · new lines {first:,}-{last:,}"
255
+ f" · {view.kept:,} shown · {who}" + silence(view)
256
+ )
257
+
258
+
259
+ def outline_footer(path: str, view: View, who: str) -> str:
260
+ """The same line for a file: how much of it is here, and who left the rest out."""
261
+ return f"sift {path} · {counted(view)} · {who}" + silence(view)
262
+
263
+
264
+ def chose(view: View) -> str:
265
+ """Which model's judgement this is, and whether it was reached just now.
266
+
267
+ A remembered answer is the same model's answer and the footer keeps saying
268
+ whose it is. What it must not do is imply the model was asked: a reader
269
+ watching a tool spend requests deserves to know which of these views cost
270
+ one. A view naming a model and costing no asks can only have come from the
271
+ cache, so nothing has to be carried alongside it to know.
272
+ """
273
+ name = view.model or "model"
274
+ return f"{name} (remembered)" if view.model and view.asks == 0 else name
275
+
276
+
277
+ def counted(view: View) -> str:
278
+ """How much of it is here, in whatever the view was counting.
279
+
280
+ The word comes off the view rather than being written into each footer,
281
+ because a footer that says "lines" about a count of records is not a wording
282
+ slip: it is the one number this tool publishes about itself, described as
283
+ something it is not.
284
+ """
285
+ return f"{view.kept:,}/{view.total:,} {view.unit}s"
286
+
287
+
288
+ def silence(view: View) -> str:
289
+ """What to add when part of the capture was never actually looked at.
290
+
291
+ A short view and an incomplete view read the same. Both say a small number
292
+ out of a large one, in the same words, with the same confidence. The
293
+ difference is that one of them is a choice and the other is a gap, and only
294
+ this line tells the reader which one they are holding.
295
+ """
296
+ if not view.unanswered:
297
+ return ""
298
+ asked = "question" if view.unanswered == 1 else "questions"
299
+ return f" · {view.unanswered} {asked} unanswered"
300
+
301
+
302
+ def peek_footer(found: Peek) -> str:
303
+ """No model in this one, so it says only where in the whole this piece sits.
304
+
305
+ A search says how many lines it matched as well as which it shows. Those are
306
+ different numbers -- context is added around each match and the answer is
307
+ capped -- and a reader who is told only the second cannot tell whether they
308
+ have seen everything their pattern found.
309
+ """
310
+ where = (
311
+ f"sift {found.handle} · lines {found.first_line:,}-{found.last_line:,}"
312
+ f" of {found.total_lines:,}"
313
+ )
314
+ if not found.matched:
315
+ return where
316
+ line = "line" if found.matched == 1 else "lines"
317
+ return f"{where} · {found.matched:,} {line} matched"
sift/watch.py ADDED
@@ -0,0 +1,166 @@
1
+ """The process that stays behind when a command is left running.
2
+
3
+ `sift run` waits for the command and then hands back a view. `sift run
4
+ --background` cannot: the point of it is that the caller gets their prompt back.
5
+ Something still has to be there when the command ends, because the one thing a
6
+ capture cannot be asked afterwards is how it ended. A process that has already
7
+ exited does not tell a stranger its exit code.
8
+
9
+ So the launcher starts this, and this starts the command. It is the smallest
10
+ program that can honestly close a run: it waits, then writes `meta.json`. No
11
+ model, no network, no judgement -- those happen later, in whichever process
12
+ asks to see the output.
13
+
14
+ The command's bytes go straight to the capture file and not through here. That
15
+ is worth saying out loud: if this process is killed, the output keeps arriving.
16
+ What is lost is the ending, which is why `background.stop` is prepared to write
17
+ one itself.
18
+
19
+ Stopping works because of where this sits. The launcher puts it in a session of
20
+ its own and the command, started without one, joins its group -- so ending the
21
+ group ends both, a build and the four compilers it spawned included. A SIGTERM
22
+ arriving here is passed on rather than taken personally, so a stopped run still
23
+ gets a `meta.json` instead of looking, forever, like something still going.
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import contextlib
29
+ import os
30
+ import signal
31
+ import subprocess
32
+ import sys
33
+
34
+ from sift import store
35
+
36
+ _USAGE = "usage: python -m sift.watch <handle> <started_at> <shell|noshell> <cwd> -- command..."
37
+
38
+ # What a command that could not be started is reported as. It is what a shell
39
+ # reports for the same failure, so a command that does not exist ends the same
40
+ # way whether or not `shell=True` was asked for.
41
+ _CANNOT_START = 127
42
+
43
+
44
+ def main(argv: list[str] | None = None) -> int:
45
+ """Wait for one command and write down how it ended.
46
+
47
+ The whole plan arrives in argv rather than through a file. Both would work,
48
+ but argv arrives with the process: there is no moment where this is running
49
+ and does not yet know what it is watching, and nothing left behind to clean
50
+ up if the launcher dies between writing the plan and spawning this.
51
+ """
52
+ words = list(sys.argv[1:] if argv is None else argv)
53
+ if len(words) < 6 or words[4] != "--":
54
+ print(_USAGE, file=sys.stderr)
55
+ return 2
56
+ handle, when, how, cwd = words[:4]
57
+ try:
58
+ started_at = float(when)
59
+ except ValueError:
60
+ print(_USAGE, file=sys.stderr)
61
+ return 2
62
+ return watch(handle, words[5:], started_at=started_at, shell=how == "shell", cwd=cwd)
63
+
64
+
65
+ def watch(
66
+ handle: str,
67
+ command: list[str],
68
+ *,
69
+ started_at: float,
70
+ shell: bool = False,
71
+ cwd: str = "",
72
+ ) -> int:
73
+ """Run the command, then record the ending nobody was there to see."""
74
+ target = store.begin(handle)
75
+
76
+ # Opened for appending, not writing: append is the mode that survives two
77
+ # writers and a restart. The command inherits this handle and does its own
78
+ # writing through it, so the bytes land whether or not this process is still
79
+ # alive to care.
80
+ sink = open(target, "ab") # noqa: SIM115 -- the command writes through this; closed below
81
+ try:
82
+ proc = subprocess.Popen(
83
+ " ".join(command) if shell else command,
84
+ stdout=sink,
85
+ stderr=subprocess.STDOUT,
86
+ stdin=subprocess.DEVNULL,
87
+ cwd=cwd or None,
88
+ shell=shell,
89
+ )
90
+ except OSError:
91
+ # Nobody to raise at. This process is the only one that knows the
92
+ # command never started, and a handle left marked running is worse than
93
+ # a handle marked failed.
94
+ sink.close()
95
+ _record(handle, command, shell, cwd, started_at, _CANNOT_START)
96
+ return _CANNOT_START
97
+
98
+ with sink:
99
+ exit_code = _wait(proc)
100
+ _record(handle, command, shell, cwd, started_at, exit_code)
101
+ return exit_code
102
+
103
+
104
+ def _wait(proc: subprocess.Popen) -> int:
105
+ """Wait for the command, passing on any request to stop rather than dying of it.
106
+
107
+ A supervisor that took SIGTERM personally would leave the command running
108
+ with nothing watching it and no `meta.json` ever written: the run would look
109
+ busy for as long as its directory survives. So the signal is forwarded and
110
+ the wait goes on. Whatever the command decides to do about it, this process
111
+ is still here afterwards to write down what happened.
112
+
113
+ Forwarding is belt and braces -- a signal sent to the group has already
114
+ reached the command -- and it is what makes `kill <pid>` by hand behave the
115
+ same as `sift stop`.
116
+ """
117
+
118
+ def relay(number, _frame):
119
+ with contextlib.suppress(OSError, ProcessLookupError):
120
+ proc.send_signal(number)
121
+
122
+ for name in ("SIGTERM", "SIGINT", "SIGHUP"):
123
+ number = getattr(signal, name, None)
124
+ if number is not None:
125
+ with contextlib.suppress(OSError, ValueError):
126
+ signal.signal(number, relay)
127
+
128
+ return proc.wait()
129
+
130
+
131
+ def _record(
132
+ handle: str,
133
+ command: list[str],
134
+ shell: bool,
135
+ cwd: str,
136
+ started_at: float,
137
+ exit_code: int,
138
+ ) -> None:
139
+ """Close the run: write the ending first, then take down the marker.
140
+
141
+ In that order, and the order is the whole point. A reader that arrives
142
+ between the two sees a capture that is finished and still marked running,
143
+ which `store.load_running` already answers correctly -- `meta.json` wins.
144
+ The other order has no such answer: for that moment the run is neither
145
+ running nor finished, and a caller waiting on it is told there is nothing
146
+ left to wait for.
147
+ """
148
+ target = store.raw_path(handle)
149
+ store.finish(
150
+ store.Meta(
151
+ handle=handle,
152
+ command=list(command),
153
+ shell=shell,
154
+ exit_code=exit_code,
155
+ timed_out=False,
156
+ started_at=started_at,
157
+ duration_s=round(store.now() - started_at, 3),
158
+ byte_count=target.stat().st_size if target.is_file() else 0,
159
+ cwd=cwd or os.getcwd(),
160
+ )
161
+ )
162
+ store.clear_running(handle)
163
+
164
+
165
+ if __name__ == "__main__": # pragma: no cover -- the entry point itself
166
+ raise SystemExit(main())