sift-cli 1.0.2__tar.gz → 1.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {sift_cli-1.0.2 → sift_cli-1.1.0}/.github/workflows/ci.yml +5 -2
- sift_cli-1.1.0/.github/workflows/registry.yml +72 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/.github/workflows/release.yml +9 -6
- {sift_cli-1.0.2 → sift_cli-1.1.0}/.gitignore +0 -1
- sift_cli-1.1.0/CONTRIBUTING.md +98 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/PKG-INFO +106 -17
- {sift_cli-1.0.2 → sift_cli-1.1.0}/README.md +105 -16
- sift_cli-1.1.0/SECURITY.md +75 -0
- sift_cli-1.1.0/gelistirme/00-PLAN.md +157 -0
- sift_cli-1.1.0/gelistirme/BULGULAR.md +178 -0
- sift_cli-1.1.0/gelistirme/G1-KENDI-MODELIN.md +86 -0
- sift_cli-1.1.0/gelistirme/G10-BATARYA.md +98 -0
- sift_cli-1.1.0/gelistirme/G10-batarya.log +200 -0
- sift_cli-1.1.0/gelistirme/G10-yeniden-kosum.log +50 -0
- sift_cli-1.1.0/gelistirme/G2-KUNYE.md +63 -0
- sift_cli-1.1.0/gelistirme/G3-YAPISAL-SONUC.md +63 -0
- sift_cli-1.1.0/gelistirme/G4-HALA-CALISIYOR.md +75 -0
- sift_cli-1.1.0/gelistirme/G5-BORU-VE-MODUL.md +61 -0
- sift_cli-1.1.0/gelistirme/G6-TAVAN.md +58 -0
- sift_cli-1.1.0/gelistirme/G7-WINDOWS-AGAC.md +83 -0
- sift_cli-1.1.0/gelistirme/G8-KAZANC.md +71 -0
- sift_cli-1.1.0/gelistirme/G9-DEPO-SAGLIGI.md +82 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/pyproject.toml +1 -1
- sift_cli-1.1.0/server.json +48 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/__init__.py +1 -1
- sift_cli-1.1.0/src/sift/__main__.py +13 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/background.py +17 -5
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/capture.py +154 -11
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/cli.py +72 -12
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/digest.py +14 -5
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/distill.py +10 -0
- sift_cli-1.1.0/src/sift/jobs.py +122 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/model.py +61 -4
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/outline.py +4 -3
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/server.py +112 -9
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/store.py +9 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/view.py +28 -7
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/watch.py +6 -1
- {sift_cli-1.0.2 → sift_cli-1.1.0}/tanitim/00-PLAN.md +14 -10
- sift_cli-1.1.0/tanitim/ARASTIRMA.md +305 -0
- sift_cli-1.1.0/tanitim/T2-KAYIT-DEFTERI.md +93 -0
- sift_cli-1.1.0/tanitim/T3-GITHUB.md +48 -0
- sift_cli-1.1.0/tanitim/T4-DIZINLER.md +43 -0
- sift_cli-1.1.0/tanitim/T5-AWESOME.md +36 -0
- sift_cli-1.1.0/tanitim/T6-KONUMLANDIRMA.md +61 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/conftest.py +5 -1
- sift_cli-1.1.0/test/kazanc.py +114 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/mutations.py +177 -31
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_background.py +75 -2
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_capture.py +161 -8
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_cli.py +23 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_digest.py +78 -1
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_distill.py +60 -3
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_model.py +70 -1
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_package.py +51 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_privacy.py +28 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_server.py +169 -18
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_setup.py +53 -8
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_tokens.py +76 -1
- sift_cli-1.1.0/uv.lock +765 -0
- sift_cli-1.0.2/tanitim/ARASTIRMA.md +0 -139
- {sift_cli-1.0.2 → sift_cli-1.1.0}/.gitattributes +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/LICENSE +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/00-PLAN.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/01-YAKALAMA.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/02-MODEL-KOPRUSU.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/03-DAMITMA.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/04-GUVENLIK-AGI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/05-DIL-BAGIMSIZLIGI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/06-DOSYA-TASLAGI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/07-MCP-SUNUCUSU.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/08-BUTCE-VE-OLCUM.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/09-ARKA-PLAN-KOMUTLARI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/10-GIZLILIK.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/11-YAYIN.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/12-CAGIRANIN-SOZ-HAKKI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/13-DOSYA-DAMITMA.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/14-ARAYAN-PEEK.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/15-COKLU-IS.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/16-HAFIZA.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/17-YOGUN-ARACLAR.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/18-KAPSAM.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/19-KALINTI-MUTASYON.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/20-KAYIT-BIRIMI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/21-SAKLAMA.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/22-YANIT-ONBELLEGI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/23-OLCUM-BIRIMI.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/24-ANAHTAR.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/25-TEKLIF.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/26-BEKLEMEK.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/27-WINDOWS.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/notlar/28-KURULUM.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/answers.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/fallback.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/hook.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/lines.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/many.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/memory.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/peek.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/privacy.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/records.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/src/sift/tools.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/tanitim/T1-SURUM-1-0-1.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/budget.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/README.md +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/ansi-progress.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/ansi-progress.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cargo-de.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cargo-de.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cmake-vi.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cmake-vi.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cobol-mainframe.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/cobol-mainframe.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/docker-ar.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/docker-ar.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/dotnet-fr.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/dotnet-fr.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/ffmpeg-fa.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/ffmpeg-fa.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gcc-ru.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gcc-ru.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/git-quiet.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/git-quiet.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gotest-zh.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gotest-zh.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gradle-ko.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/gradle-ko.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/journalctl-it.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/journalctl-it.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/kubectl-hi.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/kubectl-hi.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/latex-pl.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/latex-pl.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/maven-tr.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/maven-tr.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/mix-he.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/mix-he.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/psql-es.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/psql-es.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/pytest-en.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/pytest-en.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/rspec-id.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/rspec-id.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/terraform-pt.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/terraform-pt.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/vite-ja.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/vite-ja.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/zig-sw.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus/zig-sw.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/clojure-rapor.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/clojure-rapor.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/cpp-baslik.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/cpp-baslik.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/elixir-onbellek.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/elixir-onbellek.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/go-kuyruk.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/go-kuyruk.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/haskell-ayristirici.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/haskell-ayristirici.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/hcl-altyapi.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/hcl-altyapi.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/java-depo.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/java-depo.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/makefile-yapi.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/makefile-yapi.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/python-akis.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/python-akis.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/qqzz-uydurma.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/qqzz-uydurma.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/ruby-fatura.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/ruby-fatura.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/rust-matris.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/rust-matris.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/shell-dagitim.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/shell-dagitim.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/sql-sema.json +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus-kaynak/sql-sema.txt +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/korpus_reader.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/languages.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/outlines.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_answers.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_fallback.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_gc.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_hook.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_hook_setup.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_kaynak.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_korpus.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_lines.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_many.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_outline.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/test_records.py +0 -0
- {sift_cli-1.0.2 → sift_cli-1.1.0}/test/wire.py +0 -0
|
@@ -22,9 +22,12 @@ jobs:
|
|
|
22
22
|
python: ["3.12", "3.13"]
|
|
23
23
|
|
|
24
24
|
steps:
|
|
25
|
-
- uses: actions/checkout@
|
|
25
|
+
- uses: actions/checkout@v7
|
|
26
26
|
|
|
27
|
-
|
|
27
|
+
# Pinned to a full version, not a major alias: this action does not
|
|
28
|
+
# publish one. `@v10` resolves to nothing and every leg of CI fails
|
|
29
|
+
# before it runs a line -- which is how this was found.
|
|
30
|
+
- uses: astral-sh/setup-uv@v10.0.1
|
|
28
31
|
with:
|
|
29
32
|
enable-cache: true
|
|
30
33
|
|
|
@@ -0,0 +1,72 @@
|
|
|
1
|
+
name: registry
|
|
2
|
+
|
|
3
|
+
# The MCP registry is the one channel that reaches a model too new to have been
|
|
4
|
+
# trained on this package: clients query it live. Publishing there is a separate
|
|
5
|
+
# act from publishing to PyPI because it depends on it -- the registry proves
|
|
6
|
+
# that we own `sift-cli` by reading the description of the exact version named
|
|
7
|
+
# in server.json, which only exists once PyPI has it.
|
|
8
|
+
#
|
|
9
|
+
# So this runs on the tag as well, and waits rather than assuming.
|
|
10
|
+
on:
|
|
11
|
+
push:
|
|
12
|
+
tags: ["v*"]
|
|
13
|
+
workflow_dispatch:
|
|
14
|
+
|
|
15
|
+
permissions:
|
|
16
|
+
contents: read
|
|
17
|
+
|
|
18
|
+
jobs:
|
|
19
|
+
publish:
|
|
20
|
+
runs-on: ubuntu-latest
|
|
21
|
+
permissions:
|
|
22
|
+
# Trusted publishing again, and for the same reason: GitHub signs a
|
|
23
|
+
# short-lived token saying which repository this is, and the registry
|
|
24
|
+
# believes that rather than a secret we would have to store and rotate.
|
|
25
|
+
id-token: write
|
|
26
|
+
contents: read
|
|
27
|
+
steps:
|
|
28
|
+
- uses: actions/checkout@v7
|
|
29
|
+
|
|
30
|
+
# Four places say the version and they are written by hand. The test suite
|
|
31
|
+
# holds three of them together; the tag is the fourth and only exists here.
|
|
32
|
+
- name: The tag and server.json agree
|
|
33
|
+
if: github.ref_type == 'tag'
|
|
34
|
+
run: |
|
|
35
|
+
tag="${GITHUB_REF_NAME#v}"
|
|
36
|
+
entry=$(python -c "import json;print(json.load(open('server.json'))['version'])")
|
|
37
|
+
echo "tag=$tag server.json=$entry"
|
|
38
|
+
test "$tag" = "$entry"
|
|
39
|
+
|
|
40
|
+
# PyPI is the source of truth for ownership, and the release workflow is
|
|
41
|
+
# still finishing when this starts. Ten minutes is generous; the wait is
|
|
42
|
+
# cheap and the failure it prevents is a publish that says the package
|
|
43
|
+
# does not exist.
|
|
44
|
+
- name: Wait until PyPI has that version
|
|
45
|
+
run: |
|
|
46
|
+
version=$(python -c "import json;print(json.load(open('server.json'))['version'])")
|
|
47
|
+
for _ in $(seq 1 60); do
|
|
48
|
+
code=$(curl -s -o /dev/null -w '%{http_code}' \
|
|
49
|
+
"https://pypi.org/pypi/sift-cli/$version/json")
|
|
50
|
+
if [ "$code" = "200" ]; then
|
|
51
|
+
echo "PyPI has $version"
|
|
52
|
+
exit 0
|
|
53
|
+
fi
|
|
54
|
+
echo "PyPI does not have $version yet ($code); waiting"
|
|
55
|
+
sleep 10
|
|
56
|
+
done
|
|
57
|
+
echo "PyPI never showed $version" >&2
|
|
58
|
+
exit 1
|
|
59
|
+
|
|
60
|
+
# Pinned, not `latest`: what publishes a release should not change under
|
|
61
|
+
# it between one release and the next.
|
|
62
|
+
- name: Install mcp-publisher
|
|
63
|
+
run: |
|
|
64
|
+
curl -fsSL "https://github.com/modelcontextprotocol/registry/releases/download/v1.8.1/mcp-publisher_linux_amd64.tar.gz" \
|
|
65
|
+
| tar xz mcp-publisher
|
|
66
|
+
./mcp-publisher --help > /dev/null
|
|
67
|
+
|
|
68
|
+
- name: Log in with GitHub OIDC
|
|
69
|
+
run: ./mcp-publisher login github-oidc
|
|
70
|
+
|
|
71
|
+
- name: Publish
|
|
72
|
+
run: ./mcp-publisher publish
|
|
@@ -23,8 +23,11 @@ jobs:
|
|
|
23
23
|
os: [ubuntu-latest, macos-latest, windows-latest]
|
|
24
24
|
python: ["3.12", "3.13"]
|
|
25
25
|
steps:
|
|
26
|
-
- uses: actions/checkout@
|
|
27
|
-
|
|
26
|
+
- uses: actions/checkout@v7
|
|
27
|
+
# Pinned to a full version, not a major alias: this action does not
|
|
28
|
+
# publish one. `@v10` resolves to nothing and every leg of CI fails
|
|
29
|
+
# before it runs a line -- which is how this was found.
|
|
30
|
+
- uses: astral-sh/setup-uv@v10.0.1
|
|
28
31
|
with:
|
|
29
32
|
enable-cache: true
|
|
30
33
|
- name: Install
|
|
@@ -38,8 +41,8 @@ jobs:
|
|
|
38
41
|
needs: check
|
|
39
42
|
runs-on: ubuntu-latest
|
|
40
43
|
steps:
|
|
41
|
-
- uses: actions/checkout@
|
|
42
|
-
- uses: astral-sh/setup-uv@
|
|
44
|
+
- uses: actions/checkout@v7
|
|
45
|
+
- uses: astral-sh/setup-uv@v10.0.1
|
|
43
46
|
|
|
44
47
|
# The tag and the version in the package have to be the same thing. They
|
|
45
48
|
# are written in two places by two people at two times, so they are
|
|
@@ -55,7 +58,7 @@ jobs:
|
|
|
55
58
|
- name: Build
|
|
56
59
|
run: uv build
|
|
57
60
|
|
|
58
|
-
- uses: actions/upload-artifact@
|
|
61
|
+
- uses: actions/upload-artifact@v7
|
|
59
62
|
with:
|
|
60
63
|
name: dist
|
|
61
64
|
path: dist/
|
|
@@ -70,7 +73,7 @@ jobs:
|
|
|
70
73
|
permissions:
|
|
71
74
|
id-token: write
|
|
72
75
|
steps:
|
|
73
|
-
- uses: actions/download-artifact@
|
|
76
|
+
- uses: actions/download-artifact@v8
|
|
74
77
|
with:
|
|
75
78
|
name: dist
|
|
76
79
|
path: dist/
|
|
@@ -0,0 +1,98 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
The unusual things about this repository are worth knowing before you spend an
|
|
4
|
+
hour on it. None of them are preferences; each one is there because something
|
|
5
|
+
went wrong without it.
|
|
6
|
+
|
|
7
|
+
## Two languages, on purpose
|
|
8
|
+
|
|
9
|
+
**The notes are in Turkish. The code, the comments and the tests are in
|
|
10
|
+
English.** `notlar/`, `tanitim/` and `gelistirme/` are written for the person
|
|
11
|
+
whose project this is; everything a stranger reads while working — module
|
|
12
|
+
docstrings, comments, test names, commit subjects' meaning — is in English.
|
|
13
|
+
|
|
14
|
+
You do not need Turkish to contribute. Read `README.md` and the docstrings; they
|
|
15
|
+
are the design document.
|
|
16
|
+
|
|
17
|
+
## Three rules that do not bend
|
|
18
|
+
|
|
19
|
+
Every change is measured against these. A pull request that improves something
|
|
20
|
+
by weakening one of them will be declined, and the reason will be this list.
|
|
21
|
+
|
|
22
|
+
1. **Nothing shown is invented.** The model returns line numbers. Text always
|
|
23
|
+
comes from the local file, byte for byte.
|
|
24
|
+
2. **Nothing is thrown away.** `peek` returns the raw capture. Every gap says
|
|
25
|
+
how many lines it stands for.
|
|
26
|
+
3. **Nothing can break the command.** No key, no network, a nonsense reply, a
|
|
27
|
+
bug in the distiller — each falls back to something that needs none of them.
|
|
28
|
+
The command still runs, and its exit code is still its own.
|
|
29
|
+
|
|
30
|
+
## Every rule gets a mutation
|
|
31
|
+
|
|
32
|
+
A green test suite says the tests did not object to this version of the code. It
|
|
33
|
+
does not say they would object to a worse one. Twice in this project a test
|
|
34
|
+
passed while proving nothing, and both times looked exactly like the tests
|
|
35
|
+
around them.
|
|
36
|
+
|
|
37
|
+
So `test/mutations.py` breaks each rule deliberately, one at a time, and the
|
|
38
|
+
suite has to notice:
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
uv run python test/mutations.py # all of them, ~3.5 hours
|
|
42
|
+
uv run python test/mutations.py "some words" # only the rules that mention them
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
**A change that adds a rule adds a line there.** If your mutation escapes, the
|
|
46
|
+
rule is unguarded — that is a finding, not a formality, and the fix is usually a
|
|
47
|
+
test rather than an argument.
|
|
48
|
+
|
|
49
|
+
Three times during this project the battery found that a *test* proved nothing:
|
|
50
|
+
a command that finished before the thing being tested happened, a query that
|
|
51
|
+
matched itself, an input too small to reach the branch. Expect it to find yours.
|
|
52
|
+
|
|
53
|
+
If a rule cannot be broken from here — a Windows-only path, say — do not add a
|
|
54
|
+
mutation that will escape every run. Say in a comment which leg of CI guards it.
|
|
55
|
+
|
|
56
|
+
## Running things
|
|
57
|
+
|
|
58
|
+
```bash
|
|
59
|
+
uv sync --all-extras --dev
|
|
60
|
+
uv run pytest -q # about a minute
|
|
61
|
+
uv run ruff check . # must be clean
|
|
62
|
+
uv run python test/budget.py # what a ceiling costs, against the corpus (spends requests)
|
|
63
|
+
uv run python test/kazanc.py # what a view saves, by the endpoint's own count
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The suite reaches no network and finds no key: `test/conftest.py` moves `HOME`,
|
|
67
|
+
deletes the key variables and sets `SIFT_NO_MODEL=1`. That is not tidiness —
|
|
68
|
+
before it existed, tests were spending the developer's own quota and one of them
|
|
69
|
+
swept the real capture store.
|
|
70
|
+
|
|
71
|
+
The measurement scripts do spend requests, which is why they are scripts and not
|
|
72
|
+
tests.
|
|
73
|
+
|
|
74
|
+
## Numbers
|
|
75
|
+
|
|
76
|
+
**Do not write a number you did not measure.** Not in the README, not in a
|
|
77
|
+
comment, not in a note. Where a number appears, say what measured it: the
|
|
78
|
+
endpoint's own token count, this tool's arithmetic over bytes it holds, a run of
|
|
79
|
+
`test/budget.py` against the corpus.
|
|
80
|
+
|
|
81
|
+
Bytes divided by four is not a token count. This tool is pointed at Japanese, at
|
|
82
|
+
Turkish, at base64 and at stack traces, where that rule of thumb is wrong by a
|
|
83
|
+
factor rather than a margin — and `stats` refuses it explicitly.
|
|
84
|
+
|
|
85
|
+
## Commits and phases
|
|
86
|
+
|
|
87
|
+
Work lands one phase at a time: code, tests, a mutation, a note in the matching
|
|
88
|
+
folder, and one commit. A commit message here says *what was wrong and why the
|
|
89
|
+
change is the answer* — not what the diff already shows.
|
|
90
|
+
|
|
91
|
+
`notlar/` is the tool's own history, `tanitim/` is about being findable,
|
|
92
|
+
`gelistirme/` is about being better. A phase that decided **not** to do
|
|
93
|
+
something gets a note too, with the measurement that decided it.
|
|
94
|
+
|
|
95
|
+
## Reporting
|
|
96
|
+
|
|
97
|
+
Bugs and ideas: <https://github.com/slymnysr/sift/issues>.
|
|
98
|
+
Anything security-shaped: `SECURITY.md`.
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: sift-cli
|
|
3
|
-
Version: 1.0
|
|
3
|
+
Version: 1.1.0
|
|
4
4
|
Summary: Runs your command, then gives the model only the lines that matter.
|
|
5
5
|
Project-URL: Homepage, https://github.com/slymnysr/sift
|
|
6
6
|
Author: slmnys
|
|
@@ -26,12 +26,20 @@ Runs your command, then gives the model only the lines that matter.
|
|
|
26
26
|
|
|
27
27
|
A test suite prints 4,000 lines and eleven of them are the failure. A build
|
|
28
28
|
prints a progress bar that redraws 900 times. An install lists every package it
|
|
29
|
-
touched. All of it lands in the
|
|
30
|
-
— it is re-sent in full on every turn that follows.
|
|
29
|
+
touched. All of it lands in the context window, and — this is the part that
|
|
30
|
+
costs — it is re-sent in full on every turn that follows.
|
|
31
31
|
|
|
32
|
-
`sift`
|
|
33
|
-
|
|
34
|
-
rest is marked, not
|
|
32
|
+
`sift` is an MCP server and a command line for that problem. It runs the command
|
|
33
|
+
itself, keeps every byte on disk, and hands the model a view: the failures, the
|
|
34
|
+
summary, the lines a reader would actually stop on. The rest is marked, not
|
|
35
|
+
deleted.
|
|
36
|
+
|
|
37
|
+
Two numbers, both measured, both reproducible from this repository. Running this
|
|
38
|
+
project's own test suite prints 652 lines — 21,392 tokens, as the model's own
|
|
39
|
+
tokenizer counts them. What comes back is 8 lines and 208 tokens: **99% fewer**
|
|
40
|
+
(`python test/kazanc.py`). Removing almost everything is the easy half. Over a
|
|
41
|
+
22-sample corpus of real build and test output, the default budget keeps **138
|
|
42
|
+
of the 140 lines a reader could not do without** (`python test/budget.py`).
|
|
35
43
|
|
|
36
44
|
```
|
|
37
45
|
$ sift run -- pytest
|
|
@@ -135,6 +143,9 @@ is still shown to you in full**.
|
|
|
135
143
|
Three switches:
|
|
136
144
|
|
|
137
145
|
```bash
|
|
146
|
+
SIFT_MAX_CAPTURE=0 # keep everything a command writes, however much that is
|
|
147
|
+
SIFT_BASE_URL=... # ask your own endpoint instead, and no key is wanted
|
|
148
|
+
SIFT_MODELS=a,b # which models to ask there, best first
|
|
138
149
|
SIFT_NO_MODEL=1 # never send anything; use the deterministic view
|
|
139
150
|
SIFT_MASK=0 # send unmasked
|
|
140
151
|
SIFT_CACHE=0 # ask again, even about text already answered
|
|
@@ -163,6 +174,15 @@ Captured bytes never leave `$SIFT_HOME` (`~/.cache/sift` by default). Nothing is
|
|
|
163
174
|
uploaded, nothing is logged elsewhere, and removing a capture directory removes
|
|
164
175
|
everything that was ever kept about it.
|
|
165
176
|
|
|
177
|
+
One capture keeps at most a gigabyte. That is far past any real build log and
|
|
178
|
+
a few seconds of a command stuck in a loop, which is the case it exists for:
|
|
179
|
+
nothing is thrown away, and the disk somebody else needs is not filled either.
|
|
180
|
+
Reading never stops — a pipe nobody drains would stop the command, and that is
|
|
181
|
+
the one thing this will not do — so the command finishes, its exit code is its
|
|
182
|
+
own, and the footer says `kept the first 1,073,741,824 bytes of it` rather than
|
|
183
|
+
letting you believe you have the whole run. `SIFT_MAX_CAPTURE=0` turns the
|
|
184
|
+
ceiling off for anyone who would rather have the disk.
|
|
185
|
+
|
|
166
186
|
They also never go away on their own. Nothing here sweeps, expires or tidies in
|
|
167
187
|
the background: `sift gc [DAYS]` is the only thing that deletes a capture, and
|
|
168
188
|
it deletes when you type it and not before. What it leaves is one line per
|
|
@@ -178,8 +198,8 @@ sift run [--timeout SECONDS] [--shell] [--background] [--cwd DIR]
|
|
|
178
198
|
sift follow [HANDLE] [--all] [--wait N]
|
|
179
199
|
what a background run has said since you last looked
|
|
180
200
|
sift stop [HANDLE] end it, and everything it started
|
|
181
|
-
sift outline PATH
|
|
182
|
-
sift digest PATH
|
|
201
|
+
sift outline PATH|- what a file declares, without its bodies
|
|
202
|
+
sift digest PATH...|- what is in files somebody else produced
|
|
183
203
|
sift peek HANDLE|PATH [FIRST] [LAST]
|
|
184
204
|
sift hook answer one shell-command event on stdin
|
|
185
205
|
sift mcp speak the protocol on stdin, for a client
|
|
@@ -191,6 +211,21 @@ sift stats [COUNT] what the shortening cost, and what it saved
|
|
|
191
211
|
sift gc [DAYS] remove captures older than that, and say what went
|
|
192
212
|
```
|
|
193
213
|
|
|
214
|
+
A path of `-` reads standard input, which is the other way somebody else's
|
|
215
|
+
output turns up:
|
|
216
|
+
|
|
217
|
+
```
|
|
218
|
+
$ journalctl -u nginx --since yesterday | sift digest -
|
|
219
|
+
Sep 08 04:11:07 nginx[2114]: worker process 2119 exited on signal 11
|
|
220
|
+
─ 8,204 lines not shown · sift peek 7c1a04e9 for any of them ─
|
|
221
|
+
Sep 09 01:02:55 nginx[2114]: signal 15 (SIGTERM) received, exiting
|
|
222
|
+
```
|
|
223
|
+
|
|
224
|
+
What arrives on a pipe has no path anybody could type again, so it is kept as a
|
|
225
|
+
capture of its own and the gap marker names that instead. The second rule is
|
|
226
|
+
why: a view that left lines out and pointed at a scratch file would be pointing
|
|
227
|
+
at nothing by the time somebody read it.
|
|
228
|
+
|
|
194
229
|
Several paths given to `digest` are asked about at the same time, and `--all`
|
|
195
230
|
follows every running command in one go. Both are the same idea: the waiting is
|
|
196
231
|
the cost, so do it once. `--wait N` holds until a run actually says something
|
|
@@ -229,10 +264,28 @@ slower, costs a request and is sometimes wrong. The model decides what cannot be
|
|
|
229
264
|
computed, and nothing else.
|
|
230
265
|
|
|
231
266
|
`sift stats` says what the shortening saved and what it cost, and keeps those
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
|
|
235
|
-
|
|
267
|
+
apart because they are not the same kind of number. The share is this tool's own
|
|
268
|
+
arithmetic over bytes it holds, so it is exact. The cost is the endpoint's count
|
|
269
|
+
of its own tokens, so it is measured. And beside it now sits the number the
|
|
270
|
+
question is really about — what the output weighed, counted by the same
|
|
271
|
+
tokenizer at the moment it was sent:
|
|
272
|
+
|
|
273
|
+
```
|
|
274
|
+
$ sift stats
|
|
275
|
+
handle captured shown part asks weighed cost command
|
|
276
|
+
9f2c41ab 62,003 B 450 B 0.7% 1 21,392 5,120 pytest -v
|
|
277
|
+
|
|
278
|
+
1 run · 62,003 B captured · 450 B shown · 0.7% of it
|
|
279
|
+
cost 5,120 tokens, as the endpoint counted them
|
|
280
|
+
the output put to a model weighed 21,392 tokens -- numbering and question included
|
|
281
|
+
```
|
|
282
|
+
|
|
283
|
+
That last line is a little more than the capture alone, because the lines went
|
|
284
|
+
out numbered with a question in front of them, and it says so rather than
|
|
285
|
+
subtracting a guess. It also lets the report be unflattering: on a small
|
|
286
|
+
command the asking costs more than the output ever weighed, and this is where
|
|
287
|
+
you would see that. A run the endpoint did not count is a dash, left out of both
|
|
288
|
+
totals rather than filled in with bytes divided by four.
|
|
236
289
|
|
|
237
290
|
`--keep PATTERN` shows every line matching it whatever else was chosen and
|
|
238
291
|
whatever the budget says. It is your pattern, not one this tool guessed at —
|
|
@@ -304,11 +357,15 @@ nothing there is this.
|
|
|
304
357
|
|
|
305
358
|
Python 3.12 or newer, and no dependencies for the command line.
|
|
306
359
|
|
|
307
|
-
### You need your own
|
|
360
|
+
### You need a model — a key, or one of your own
|
|
361
|
+
|
|
362
|
+
`sift` asks a model which lines matter. There are two ways to give it one and
|
|
363
|
+
you need exactly one of them.
|
|
308
364
|
|
|
309
|
-
|
|
310
|
-
|
|
311
|
-
|
|
365
|
+
#### A free key
|
|
366
|
+
|
|
367
|
+
**Not shipped, and not shareable.** The key has to be yours. It is free, and it
|
|
368
|
+
takes a minute:
|
|
312
369
|
|
|
313
370
|
1. Get a key at **<https://build.nvidia.com>**
|
|
314
371
|
2. Put it anywhere `sift` looks:
|
|
@@ -319,7 +376,33 @@ export SIFT_API_KEY=nvapi-... # or NVIDIA_API_KEY
|
|
|
319
376
|
mkdir -p ~/.config/nvidia && echo 'nvapi-...' > ~/.config/nvidia/api_key
|
|
320
377
|
```
|
|
321
378
|
|
|
322
|
-
|
|
379
|
+
#### Or a model of your own, and no key at all
|
|
380
|
+
|
|
381
|
+
Point `sift` somewhere and it asks there instead. Nothing about the question
|
|
382
|
+
changes; the endpoint is asked the ordinary OpenAI-shaped way, and no
|
|
383
|
+
`Authorization` header is sent when there is no key to put in it.
|
|
384
|
+
|
|
385
|
+
```bash
|
|
386
|
+
export SIFT_BASE_URL=http://localhost:11434/v1 # Ollama
|
|
387
|
+
export SIFT_MODELS=qwen3:8b # what to ask, best first
|
|
388
|
+
```
|
|
389
|
+
|
|
390
|
+
The same two lines fit llama.cpp (`--api`), vLLM, LM Studio, LocalAI, a company
|
|
391
|
+
gateway, or any other endpoint that speaks `POST /v1/chat/completions`.
|
|
392
|
+
`SIFT_MODELS` takes a comma-separated ladder and is asked in order.
|
|
393
|
+
|
|
394
|
+
An address you typed is treated as a decision: nothing warns you about a missing
|
|
395
|
+
key, and the MCP server does not decline. What that endpoint wants for
|
|
396
|
+
credentials is between you and it.
|
|
397
|
+
|
|
398
|
+
> Tested here as a shape rather than as a list: the suite proves that an
|
|
399
|
+
> endpoint of your own is asked, and asked without a key. Which local servers
|
|
400
|
+
> answer *well* is a question about the model you run, and the corpus in
|
|
401
|
+
> `test/budget.py` is how you can settle it for yours.
|
|
402
|
+
|
|
403
|
+
#### Without either of them
|
|
404
|
+
|
|
405
|
+
The two callers are answered differently, on purpose:
|
|
323
406
|
|
|
324
407
|
- **At a terminal** everything still runs — the command, the bytes, the exit
|
|
325
408
|
code, the third rule — and a loud banner says no model chose these lines and
|
|
@@ -332,6 +415,12 @@ Without it the two callers are answered differently, on purpose:
|
|
|
332
415
|
If you *want* to run without a model, say so with `SIFT_NO_MODEL=1`. That is a
|
|
333
416
|
decision rather than an oversight, everything works, and nothing lectures you.
|
|
334
417
|
|
|
418
|
+
## Contributing, and reporting something
|
|
419
|
+
|
|
420
|
+
`CONTRIBUTING.md` is the map: two languages on purpose, three rules that do not
|
|
421
|
+
bend, and a mutation for every rule. `SECURITY.md` says what runs, what leaves
|
|
422
|
+
the machine, and where to send a finding privately.
|
|
423
|
+
|
|
335
424
|
## How it was built
|
|
336
425
|
|
|
337
426
|
Twenty-four phases, each one closed before the next began, each with a note in
|
|
@@ -6,12 +6,20 @@ Runs your command, then gives the model only the lines that matter.
|
|
|
6
6
|
|
|
7
7
|
A test suite prints 4,000 lines and eleven of them are the failure. A build
|
|
8
8
|
prints a progress bar that redraws 900 times. An install lists every package it
|
|
9
|
-
touched. All of it lands in the
|
|
10
|
-
— it is re-sent in full on every turn that follows.
|
|
9
|
+
touched. All of it lands in the context window, and — this is the part that
|
|
10
|
+
costs — it is re-sent in full on every turn that follows.
|
|
11
11
|
|
|
12
|
-
`sift`
|
|
13
|
-
|
|
14
|
-
rest is marked, not
|
|
12
|
+
`sift` is an MCP server and a command line for that problem. It runs the command
|
|
13
|
+
itself, keeps every byte on disk, and hands the model a view: the failures, the
|
|
14
|
+
summary, the lines a reader would actually stop on. The rest is marked, not
|
|
15
|
+
deleted.
|
|
16
|
+
|
|
17
|
+
Two numbers, both measured, both reproducible from this repository. Running this
|
|
18
|
+
project's own test suite prints 652 lines — 21,392 tokens, as the model's own
|
|
19
|
+
tokenizer counts them. What comes back is 8 lines and 208 tokens: **99% fewer**
|
|
20
|
+
(`python test/kazanc.py`). Removing almost everything is the easy half. Over a
|
|
21
|
+
22-sample corpus of real build and test output, the default budget keeps **138
|
|
22
|
+
of the 140 lines a reader could not do without** (`python test/budget.py`).
|
|
15
23
|
|
|
16
24
|
```
|
|
17
25
|
$ sift run -- pytest
|
|
@@ -115,6 +123,9 @@ is still shown to you in full**.
|
|
|
115
123
|
Three switches:
|
|
116
124
|
|
|
117
125
|
```bash
|
|
126
|
+
SIFT_MAX_CAPTURE=0 # keep everything a command writes, however much that is
|
|
127
|
+
SIFT_BASE_URL=... # ask your own endpoint instead, and no key is wanted
|
|
128
|
+
SIFT_MODELS=a,b # which models to ask there, best first
|
|
118
129
|
SIFT_NO_MODEL=1 # never send anything; use the deterministic view
|
|
119
130
|
SIFT_MASK=0 # send unmasked
|
|
120
131
|
SIFT_CACHE=0 # ask again, even about text already answered
|
|
@@ -143,6 +154,15 @@ Captured bytes never leave `$SIFT_HOME` (`~/.cache/sift` by default). Nothing is
|
|
|
143
154
|
uploaded, nothing is logged elsewhere, and removing a capture directory removes
|
|
144
155
|
everything that was ever kept about it.
|
|
145
156
|
|
|
157
|
+
One capture keeps at most a gigabyte. That is far past any real build log and
|
|
158
|
+
a few seconds of a command stuck in a loop, which is the case it exists for:
|
|
159
|
+
nothing is thrown away, and the disk somebody else needs is not filled either.
|
|
160
|
+
Reading never stops — a pipe nobody drains would stop the command, and that is
|
|
161
|
+
the one thing this will not do — so the command finishes, its exit code is its
|
|
162
|
+
own, and the footer says `kept the first 1,073,741,824 bytes of it` rather than
|
|
163
|
+
letting you believe you have the whole run. `SIFT_MAX_CAPTURE=0` turns the
|
|
164
|
+
ceiling off for anyone who would rather have the disk.
|
|
165
|
+
|
|
146
166
|
They also never go away on their own. Nothing here sweeps, expires or tidies in
|
|
147
167
|
the background: `sift gc [DAYS]` is the only thing that deletes a capture, and
|
|
148
168
|
it deletes when you type it and not before. What it leaves is one line per
|
|
@@ -158,8 +178,8 @@ sift run [--timeout SECONDS] [--shell] [--background] [--cwd DIR]
|
|
|
158
178
|
sift follow [HANDLE] [--all] [--wait N]
|
|
159
179
|
what a background run has said since you last looked
|
|
160
180
|
sift stop [HANDLE] end it, and everything it started
|
|
161
|
-
sift outline PATH
|
|
162
|
-
sift digest PATH
|
|
181
|
+
sift outline PATH|- what a file declares, without its bodies
|
|
182
|
+
sift digest PATH...|- what is in files somebody else produced
|
|
163
183
|
sift peek HANDLE|PATH [FIRST] [LAST]
|
|
164
184
|
sift hook answer one shell-command event on stdin
|
|
165
185
|
sift mcp speak the protocol on stdin, for a client
|
|
@@ -171,6 +191,21 @@ sift stats [COUNT] what the shortening cost, and what it saved
|
|
|
171
191
|
sift gc [DAYS] remove captures older than that, and say what went
|
|
172
192
|
```
|
|
173
193
|
|
|
194
|
+
A path of `-` reads standard input, which is the other way somebody else's
|
|
195
|
+
output turns up:
|
|
196
|
+
|
|
197
|
+
```
|
|
198
|
+
$ journalctl -u nginx --since yesterday | sift digest -
|
|
199
|
+
Sep 08 04:11:07 nginx[2114]: worker process 2119 exited on signal 11
|
|
200
|
+
─ 8,204 lines not shown · sift peek 7c1a04e9 for any of them ─
|
|
201
|
+
Sep 09 01:02:55 nginx[2114]: signal 15 (SIGTERM) received, exiting
|
|
202
|
+
```
|
|
203
|
+
|
|
204
|
+
What arrives on a pipe has no path anybody could type again, so it is kept as a
|
|
205
|
+
capture of its own and the gap marker names that instead. The second rule is
|
|
206
|
+
why: a view that left lines out and pointed at a scratch file would be pointing
|
|
207
|
+
at nothing by the time somebody read it.
|
|
208
|
+
|
|
174
209
|
Several paths given to `digest` are asked about at the same time, and `--all`
|
|
175
210
|
follows every running command in one go. Both are the same idea: the waiting is
|
|
176
211
|
the cost, so do it once. `--wait N` holds until a run actually says something
|
|
@@ -209,10 +244,28 @@ slower, costs a request and is sometimes wrong. The model decides what cannot be
|
|
|
209
244
|
computed, and nothing else.
|
|
210
245
|
|
|
211
246
|
`sift stats` says what the shortening saved and what it cost, and keeps those
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
247
|
+
apart because they are not the same kind of number. The share is this tool's own
|
|
248
|
+
arithmetic over bytes it holds, so it is exact. The cost is the endpoint's count
|
|
249
|
+
of its own tokens, so it is measured. And beside it now sits the number the
|
|
250
|
+
question is really about — what the output weighed, counted by the same
|
|
251
|
+
tokenizer at the moment it was sent:
|
|
252
|
+
|
|
253
|
+
```
|
|
254
|
+
$ sift stats
|
|
255
|
+
handle captured shown part asks weighed cost command
|
|
256
|
+
9f2c41ab 62,003 B 450 B 0.7% 1 21,392 5,120 pytest -v
|
|
257
|
+
|
|
258
|
+
1 run · 62,003 B captured · 450 B shown · 0.7% of it
|
|
259
|
+
cost 5,120 tokens, as the endpoint counted them
|
|
260
|
+
the output put to a model weighed 21,392 tokens -- numbering and question included
|
|
261
|
+
```
|
|
262
|
+
|
|
263
|
+
That last line is a little more than the capture alone, because the lines went
|
|
264
|
+
out numbered with a question in front of them, and it says so rather than
|
|
265
|
+
subtracting a guess. It also lets the report be unflattering: on a small
|
|
266
|
+
command the asking costs more than the output ever weighed, and this is where
|
|
267
|
+
you would see that. A run the endpoint did not count is a dash, left out of both
|
|
268
|
+
totals rather than filled in with bytes divided by four.
|
|
216
269
|
|
|
217
270
|
`--keep PATTERN` shows every line matching it whatever else was chosen and
|
|
218
271
|
whatever the budget says. It is your pattern, not one this tool guessed at —
|
|
@@ -284,11 +337,15 @@ nothing there is this.
|
|
|
284
337
|
|
|
285
338
|
Python 3.12 or newer, and no dependencies for the command line.
|
|
286
339
|
|
|
287
|
-
### You need your own
|
|
340
|
+
### You need a model — a key, or one of your own
|
|
341
|
+
|
|
342
|
+
`sift` asks a model which lines matter. There are two ways to give it one and
|
|
343
|
+
you need exactly one of them.
|
|
288
344
|
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
345
|
+
#### A free key
|
|
346
|
+
|
|
347
|
+
**Not shipped, and not shareable.** The key has to be yours. It is free, and it
|
|
348
|
+
takes a minute:
|
|
292
349
|
|
|
293
350
|
1. Get a key at **<https://build.nvidia.com>**
|
|
294
351
|
2. Put it anywhere `sift` looks:
|
|
@@ -299,7 +356,33 @@ export SIFT_API_KEY=nvapi-... # or NVIDIA_API_KEY
|
|
|
299
356
|
mkdir -p ~/.config/nvidia && echo 'nvapi-...' > ~/.config/nvidia/api_key
|
|
300
357
|
```
|
|
301
358
|
|
|
302
|
-
|
|
359
|
+
#### Or a model of your own, and no key at all
|
|
360
|
+
|
|
361
|
+
Point `sift` somewhere and it asks there instead. Nothing about the question
|
|
362
|
+
changes; the endpoint is asked the ordinary OpenAI-shaped way, and no
|
|
363
|
+
`Authorization` header is sent when there is no key to put in it.
|
|
364
|
+
|
|
365
|
+
```bash
|
|
366
|
+
export SIFT_BASE_URL=http://localhost:11434/v1 # Ollama
|
|
367
|
+
export SIFT_MODELS=qwen3:8b # what to ask, best first
|
|
368
|
+
```
|
|
369
|
+
|
|
370
|
+
The same two lines fit llama.cpp (`--api`), vLLM, LM Studio, LocalAI, a company
|
|
371
|
+
gateway, or any other endpoint that speaks `POST /v1/chat/completions`.
|
|
372
|
+
`SIFT_MODELS` takes a comma-separated ladder and is asked in order.
|
|
373
|
+
|
|
374
|
+
An address you typed is treated as a decision: nothing warns you about a missing
|
|
375
|
+
key, and the MCP server does not decline. What that endpoint wants for
|
|
376
|
+
credentials is between you and it.
|
|
377
|
+
|
|
378
|
+
> Tested here as a shape rather than as a list: the suite proves that an
|
|
379
|
+
> endpoint of your own is asked, and asked without a key. Which local servers
|
|
380
|
+
> answer *well* is a question about the model you run, and the corpus in
|
|
381
|
+
> `test/budget.py` is how you can settle it for yours.
|
|
382
|
+
|
|
383
|
+
#### Without either of them
|
|
384
|
+
|
|
385
|
+
The two callers are answered differently, on purpose:
|
|
303
386
|
|
|
304
387
|
- **At a terminal** everything still runs — the command, the bytes, the exit
|
|
305
388
|
code, the third rule — and a loud banner says no model chose these lines and
|
|
@@ -312,6 +395,12 @@ Without it the two callers are answered differently, on purpose:
|
|
|
312
395
|
If you *want* to run without a model, say so with `SIFT_NO_MODEL=1`. That is a
|
|
313
396
|
decision rather than an oversight, everything works, and nothing lectures you.
|
|
314
397
|
|
|
398
|
+
## Contributing, and reporting something
|
|
399
|
+
|
|
400
|
+
`CONTRIBUTING.md` is the map: two languages on purpose, three rules that do not
|
|
401
|
+
bend, and a mutation for every rule. `SECURITY.md` says what runs, what leaves
|
|
402
|
+
the machine, and where to send a finding privately.
|
|
403
|
+
|
|
315
404
|
## How it was built
|
|
316
405
|
|
|
317
406
|
Twenty-four phases, each one closed before the next began, each with a note in
|