clef-compactor 0.2.0__tar.gz → 0.2.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (53) hide show
  1. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/workflows/publish.yml +1 -1
  2. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.gitignore +2 -0
  3. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/PKG-INFO +34 -16
  4. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/README.md +33 -15
  5. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/index.html +10 -9
  6. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/README.md +14 -1
  7. clef_compactor-0.2.1/evals/backends.py +159 -0
  8. clef_compactor-0.2.1/evals/results/results-local-t4x2.json +274 -0
  9. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/results/results.json +25 -24
  10. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/run_eval.py +34 -5
  11. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/examples/demo.ipynb +18 -2
  12. clef_compactor-0.2.1/kaggle-kernel/kernel-metadata.json +13 -0
  13. clef_compactor-0.2.1/kaggle-kernel/script.py +47 -0
  14. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/pyproject.toml +1 -1
  15. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/__init__.py +1 -1
  16. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_smoke.py +3 -1
  17. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/actions/clef-evals/action.yml +0 -0
  18. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/actions/clef-evals/check_regression.py +0 -0
  19. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/workflows/test.yml +0 -0
  20. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/LICENSE +0 -0
  21. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/Makefile +0 -0
  22. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/assets/logo.png +0 -0
  23. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/assets/social-preview.png +0 -0
  24. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/brag.jpg +0 -0
  25. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/brag.mp4 +0 -0
  26. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/how-it-works.svg +0 -0
  27. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/data/compaction_suite.jsonl +0 -0
  28. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/scripts/make_assets.py +0 -0
  29. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/cli.py +0 -0
  30. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/client.py +0 -0
  31. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/compat/__init__.py +0 -0
  32. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/compat/openai.py +0 -0
  33. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/config.py +0 -0
  34. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/core.py +0 -0
  35. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/exceptions.py +0 -0
  36. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/__init__.py +0 -0
  37. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/langchain.py +0 -0
  38. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/llamaindex.py +0 -0
  39. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/models.py +0 -0
  40. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/py.typed +0 -0
  41. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/conftest.py +0 -0
  42. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/fixtures/error_envelope.json +0 -0
  43. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/fixtures/noul_score_response.json +0 -0
  44. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_cli.py +0 -0
  45. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_client.py +0 -0
  46. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_compat_openai.py +0 -0
  47. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_config.py +0 -0
  48. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_core.py +0 -0
  49. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_exceptions.py +0 -0
  50. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integration_live.py +0 -0
  51. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integrations_langchain.py +0 -0
  52. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integrations_llamaindex.py +0 -0
  53. {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_models.py +0 -0
@@ -17,4 +17,4 @@ jobs:
17
17
  with:
18
18
  python-version: "3.12"
19
19
  - run: uv build
20
- - run: uv publish
20
+ - run: uv publish --check-url https://pypi.org/simple
@@ -39,3 +39,5 @@ Thumbs.db
39
39
  .env
40
40
  .env.*
41
41
  *.pem
42
+ kaggle-out/
43
+ kaggle-out/
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: clef-compactor
3
- Version: 0.2.0
3
+ Version: 0.2.1
4
4
  Summary: Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise.
5
5
  Project-URL: Homepage, https://github.com/Gjusev/clef-compactor
6
6
  Project-URL: Repository, https://github.com/Gjusev/clef-compactor
@@ -194,21 +194,38 @@ query_engine = RetrieverQueryEngine(
194
194
 
195
195
  ## Measured results
196
196
 
197
- Replay run (deterministic simulated scorer, 20 cases, 104 chunks, seed
198
- 20261001, reproducible with `python evals/run_eval.py`). This validates the
199
- pipeline, not model quality; see [Limitations](#limitations).
197
+ Three measurement sources, each labelled for what it actually proves.
198
+
199
+ **Real model, modest hardware** — open-weights `Cloudflare/clef-flash` (9B),
200
+ float16 sharded across 2x Kaggle T4, 20 cases, 104 chunks. Produced by
201
+ `python evals/run_eval.py --mode local`; results committed in
202
+ `evals/results/results-local-t4x2.json` and reproducible from the
203
+ [`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
204
+ kernel.
200
205
 
201
206
  | metric | value |
202
207
  |---|---|
203
- | chunk accuracy | 0.990 |
204
- | kept precision | 1.000 |
205
- | relevant recall | 0.984 |
206
- | kept F1 | 0.992 |
207
- | context tokens saved | 34.0% |
208
- | cost per 1k calls | $0.019 (input tokens at $0.24/M) |
209
-
210
- Live-API numbers are pending credentials and will replace these once a run
211
- lands. The harness is ready: `python evals/run_eval.py --mode live`.
208
+ | chunk accuracy | 0.712 |
209
+ | kept precision | 0.771 |
210
+ | relevant recall | 0.746 |
211
+ | kept F1 | 0.758 |
212
+ | context tokens saved | 38.4% |
213
+ | scoring latency p50 / p95 | 1,274 / 1,407 ms |
214
+ | cost per 1k calls | $0.00 (self-hosted weights) |
215
+
216
+ **Pipeline validation** — deterministic simulated scorer, same dataset, seed
217
+ 20261001 (`python evals/run_eval.py`). Proves the ranking and budget
218
+ machinery, not model quality: chunk accuracy 0.990, kept F1 0.992, 34.0%
219
+ tokens saved.
220
+
221
+ **Hosted API** — pending credentials; `--mode live` produces it.
222
+
223
+ Reading the real numbers plainly: on a T4 the 9B model makes the right
224
+ keep/drop call 71% of the time, keeps three quarters of the relevant chunks,
225
+ and still removes 38% of the context tokens. Local latency is dominated by
226
+ the modest GPU and the torch fallback for Qwen3.5's linear attention (the
227
+ fast-path kernels were not installed in the kernel); the hosted endpoint
228
+ reports 38.8 ms median for clef-flash on Cloudflare's own hardware.
212
229
 
213
230
  ## clef vs laya
214
231
 
@@ -248,9 +265,10 @@ and honor `Retry-After`.
248
265
 
249
266
  ## Limitations
250
267
 
251
- - **The committed accuracy numbers come from a simulated scorer.** They prove
252
- the pipeline ranks and cuts correctly under noisy probabilities. They do not
253
- measure Clef. Run `--mode live` before quoting quality.
268
+ - **Real-model numbers are from a 9B model on a T4 pair, not from the hosted
269
+ endpoint.** The hosted API (and the 27B model, which needs ~54 GB) may
270
+ score better. Measuring the hosted endpoint is one command away:
271
+ `--mode live` with credentials.
254
272
  - **Latency.** clef's median decision latency is 209 ms (38.8 ms for
255
273
  clef-flash) plus network. laya keeps the whole job local at 5.8 ms. If you
256
274
  need sub-10 ms compaction on every request, see
@@ -153,21 +153,38 @@ query_engine = RetrieverQueryEngine(
153
153
 
154
154
  ## Measured results
155
155
 
156
- Replay run (deterministic simulated scorer, 20 cases, 104 chunks, seed
157
- 20261001, reproducible with `python evals/run_eval.py`). This validates the
158
- pipeline, not model quality; see [Limitations](#limitations).
156
+ Three measurement sources, each labelled for what it actually proves.
157
+
158
+ **Real model, modest hardware** — open-weights `Cloudflare/clef-flash` (9B),
159
+ float16 sharded across 2x Kaggle T4, 20 cases, 104 chunks. Produced by
160
+ `python evals/run_eval.py --mode local`; results committed in
161
+ `evals/results/results-local-t4x2.json` and reproducible from the
162
+ [`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
163
+ kernel.
159
164
 
160
165
  | metric | value |
161
166
  |---|---|
162
- | chunk accuracy | 0.990 |
163
- | kept precision | 1.000 |
164
- | relevant recall | 0.984 |
165
- | kept F1 | 0.992 |
166
- | context tokens saved | 34.0% |
167
- | cost per 1k calls | $0.019 (input tokens at $0.24/M) |
168
-
169
- Live-API numbers are pending credentials and will replace these once a run
170
- lands. The harness is ready: `python evals/run_eval.py --mode live`.
167
+ | chunk accuracy | 0.712 |
168
+ | kept precision | 0.771 |
169
+ | relevant recall | 0.746 |
170
+ | kept F1 | 0.758 |
171
+ | context tokens saved | 38.4% |
172
+ | scoring latency p50 / p95 | 1,274 / 1,407 ms |
173
+ | cost per 1k calls | $0.00 (self-hosted weights) |
174
+
175
+ **Pipeline validation** — deterministic simulated scorer, same dataset, seed
176
+ 20261001 (`python evals/run_eval.py`). Proves the ranking and budget
177
+ machinery, not model quality: chunk accuracy 0.990, kept F1 0.992, 34.0%
178
+ tokens saved.
179
+
180
+ **Hosted API** — pending credentials; `--mode live` produces it.
181
+
182
+ Reading the real numbers plainly: on a T4 the 9B model makes the right
183
+ keep/drop call 71% of the time, keeps three quarters of the relevant chunks,
184
+ and still removes 38% of the context tokens. Local latency is dominated by
185
+ the modest GPU and the torch fallback for Qwen3.5's linear attention (the
186
+ fast-path kernels were not installed in the kernel); the hosted endpoint
187
+ reports 38.8 ms median for clef-flash on Cloudflare's own hardware.
171
188
 
172
189
  ## clef vs laya
173
190
 
@@ -207,9 +224,10 @@ and honor `Retry-After`.
207
224
 
208
225
  ## Limitations
209
226
 
210
- - **The committed accuracy numbers come from a simulated scorer.** They prove
211
- the pipeline ranks and cuts correctly under noisy probabilities. They do not
212
- measure Clef. Run `--mode live` before quoting quality.
227
+ - **Real-model numbers are from a 9B model on a T4 pair, not from the hosted
228
+ endpoint.** The hosted API (and the 27B model, which needs ~54 GB) may
229
+ score better. Measuring the hosted endpoint is one command away:
230
+ `--mode live` with credentials.
213
231
  - **Latency.** clef's median decision latency is 209 ms (38.8 ms for
214
232
  clef-flash) plus network. laya keeps the whole job local at 5.8 ms. If you
215
233
  need sub-10 ms compaction on every request, see
@@ -164,16 +164,17 @@
164
164
 
165
165
  <section id="results" class="wrap">
166
166
  <div class="kicker">measured results</div>
167
- <h2 class="serif">Replay baseline, committed and reproducible</h2>
168
- <p class="lede">Deterministic simulated scorer, 20 cases, 104 chunks, seed 20261001. This validates the pipeline; it does not measure Clef. Live numbers land via <span class="mono">evals/run_eval.py --mode live</span>.</p>
167
+ <h2 class="serif">Real model, measured on real hardware</h2>
168
+ <p class="lede">Open-weights clef-flash 9B, float16 on 2x Kaggle T4, 20 hand-labelled cases, 104 chunks. Reproducible from the <a href="https://www.kaggle.com/code/gjusev/clef-compactor-evals">public kernel</a>. The replay baseline (simulated scorer) and a hosted-API run are documented in the repo.</p>
169
169
  <table class="reveal" style="--i:1">
170
- <tr><th>metric</th><th>value</th></tr>
171
- <tr><td>chunk accuracy</td><td class="num">0.990</td></tr>
172
- <tr><td>kept precision</td><td class="num">1.000</td></tr>
173
- <tr><td>relevant recall</td><td class="num">0.984</td></tr>
174
- <tr><td>kept F1</td><td class="num">0.992</td></tr>
175
- <tr><td>context tokens saved</td><td class="num best">34.0%</td></tr>
176
- <tr><td>cost per 1k calls</td><td class="num">$0.019</td></tr>
170
+ <tr><th>metric</th><th>open-weights 9B on 2xT4</th><th>replay baseline</th></tr>
171
+ <tr><td>chunk accuracy</td><td class="num">0.712</td><td class="num">0.990</td></tr>
172
+ <tr><td>kept precision</td><td class="num">0.771</td><td class="num">1.000</td></tr>
173
+ <tr><td>relevant recall</td><td class="num">0.746</td><td class="num">0.984</td></tr>
174
+ <tr><td>kept F1</td><td class="num">0.758</td><td class="num">0.992</td></tr>
175
+ <tr><td>context tokens saved</td><td class="num best">38.4%</td><td class="num">34.0%</td></tr>
176
+ <tr><td>scoring latency p50</td><td class="num">1,274 ms</td><td class="num">&lt;1 ms</td></tr>
177
+ <tr><td>cost per 1k calls</td><td class="num best">$0.00 self-hosted</td><td class="num">$0.019</td></tr>
177
178
  </table>
178
179
  </section>
179
180
 
@@ -16,7 +16,10 @@ by hand against the query; they are the ground truth for every metric.
16
16
  # Pipeline validation: real compaction code, deterministic simulated scorer
17
17
  python evals/run_eval.py
18
18
 
19
- # Real quality numbers: needs CLEF_ACCOUNT_ID and CLEF_API_TOKEN
19
+ # Real model weights, no Cloudflare account: 9B on a local GPU (T4 pair tested)
20
+ python evals/run_eval.py --mode local --model-path Cloudflare/clef-flash
21
+
22
+ # Hosted endpoint: the published prices and latencies apply
20
23
  python evals/run_eval.py --mode live
21
24
 
22
25
  # CI gate: exit code 1 when chunk accuracy drops below the threshold
@@ -26,6 +29,16 @@ python evals/run_eval.py --min-accuracy 0.9
26
29
  python evals/run_eval.py --compare-laya
27
30
  ```
28
31
 
32
+ ## Measured so far
33
+
34
+ `results/results.json` is the replay baseline (simulated scorer).
35
+ `results/results-local-t4x2.json` is the real open-weights clef-flash 9B
36
+ measured on a Kaggle T4 pair via the
37
+ [`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
38
+ kernel: chunk accuracy 0.712, kept F1 0.758, 38.4% of context tokens removed,
39
+ scoring latency ~1.27 s p50 (T4, torch attention fallback). The 27B model
40
+ needs ~54 GB and does not fit on that hardware.
41
+
29
42
  Every run writes `results/results.json`: aggregate metrics plus one entry per
30
43
  case. The committed file was produced by `--mode replay`, so it is
31
44
  reproducible byte for byte given the same seed.
@@ -0,0 +1,159 @@
1
+ """Local inference backend: run the open-weights Clef models without the API.
2
+
3
+ Wraps ``joint_schema_model.systemone`` from the Hugging Face release
4
+ (https://huggingface.co/Cloudflare/clef) in the same ``ask()`` interface the
5
+ sync compactor expects, so the whole eval pipeline runs unmodified against
6
+ weights loaded on a local GPU.
7
+
8
+ The stock ``load_release_model`` pins the whole backbone to one device via
9
+ ``device_map={"": device}``. That fits clef-flash (9B, ~18 GB in fp16) on a
10
+ single H200 but not on a single 16 GB T4, so :class:`LocalClefClient`
11
+ replicates the loader with ``device_map="auto"`` and places the joint schema
12
+ head on the device that holds the backbone's final layers.
13
+
14
+ Requirements (not package dependencies; installed by the Kaggle kernel):
15
+ ``torch``, ``transformers>=5.10``, ``accelerate``, ``safetensors``,
16
+ ``huggingface_hub``, ``pillow``.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import sys
22
+ import time
23
+ from pathlib import Path
24
+ from typing import Any
25
+
26
+ from clef_compactor.config import Settings
27
+ from clef_compactor.models import ClefReply, TokenUsage
28
+
29
+ __all__ = ["LocalClefClient"]
30
+
31
+
32
+ class LocalClefClient:
33
+ """Drop-in ``ask()`` client backed by locally loaded Clef weights.
34
+
35
+ Args:
36
+ settings: Validated settings; ``model`` labels the replies.
37
+ model_path: HF repo id (e.g. ``"Cloudflare/clef-flash"``) or a local
38
+ snapshot directory.
39
+ dtype: Torch dtype name for the backbone, ``"float16"`` for Turing GPUs
40
+ (T4), ``"bfloat16"`` on Ampere or newer.
41
+ device_map: Accelerate device map, ``"auto"`` to shard across GPUs.
42
+ max_memory_gib: Per-GPU memory ceiling used with ``device_map="auto"``.
43
+ max_length: Token bound forwarded to ``encode_record`` (default 16384).
44
+ """
45
+
46
+ def __init__(
47
+ self,
48
+ settings: Settings,
49
+ model_path: str,
50
+ *,
51
+ dtype: str = "float16",
52
+ device_map: str = "auto",
53
+ max_memory_gib: int = 15,
54
+ max_length: int = 16384,
55
+ ) -> None:
56
+ import torch # heavy: imported lazily so import cost stays off the API path
57
+
58
+ self.torch = torch
59
+ self.settings = settings
60
+ self.max_length = max_length
61
+
62
+ path = Path(model_path)
63
+ if not path.is_dir():
64
+ from huggingface_hub import snapshot_download
65
+
66
+ path = Path(snapshot_download(model_path))
67
+
68
+ joint_schema_dir = str(path)
69
+ if joint_schema_dir not in sys.path:
70
+ sys.path.insert(0, joint_schema_dir)
71
+ import json
72
+
73
+ from joint_schema_model import ClefModel, JointSchemaHead # type: ignore[import-not-found]
74
+ from safetensors.torch import load_file
75
+ from transformers import AutoProcessor, Qwen3_5ForConditionalGeneration
76
+
77
+ torch_dtype = getattr(torch, dtype)
78
+ backbone = Qwen3_5ForConditionalGeneration.from_pretrained(
79
+ path,
80
+ dtype=torch_dtype,
81
+ device_map=device_map,
82
+ max_memory={index: f"{max_memory_gib}GiB" for index in range(torch.cuda.device_count())},
83
+ )
84
+ backbone.config.use_cache = False
85
+
86
+ head_config = json.loads((path / "joint_head_config.json").read_text())
87
+ head = JointSchemaHead(**head_config)
88
+ head.load_state_dict(load_file(path / "joint_head.safetensors"), strict=True)
89
+ head = head.to(device=self._last_device(backbone), dtype=torch_dtype)
90
+
91
+ self.model = ClefModel(backbone, head).eval()
92
+ self.processor = AutoProcessor.from_pretrained(path)
93
+
94
+ @staticmethod
95
+ def _last_device(backbone: Any) -> Any:
96
+ """The device holding the backbone's final layers (head must match).
97
+
98
+ ``hf_device_map`` values may be ints (``0``, ``1``) or strings
99
+ (``"cuda:1"``); normalise both.
100
+ """
101
+ import torch
102
+
103
+ devices = list(getattr(backbone, "hf_device_map", {}).values())
104
+ if not devices:
105
+ return torch.device("cuda" if torch.cuda.is_available() else "cpu")
106
+ indexes = [int(str(value).split(":")[-1]) for value in devices]
107
+ return torch.device("cuda", max(indexes))
108
+
109
+ def ask(
110
+ self,
111
+ state: str | dict[str, Any],
112
+ questions: dict[str, dict[str, Any]],
113
+ *,
114
+ images: list[str] | None = None,
115
+ model: str | None = None,
116
+ ) -> ClefReply:
117
+ """Answer one scoring request with a single local forward pass."""
118
+ from joint_schema_model import systemone # type: ignore[import-not-found]
119
+
120
+ if not questions:
121
+ raise ValueError("questions must contain at least one entry")
122
+ request = {
123
+ "model": model or self.settings.model,
124
+ "state": state,
125
+ "questions": questions,
126
+ }
127
+ started = time.perf_counter()
128
+ response = systemone(self._collate_on_head_device(), self.processor, request, max_length=self.max_length)
129
+ latency_ms = (time.perf_counter() - started) * 1000
130
+ return ClefReply(
131
+ model=str(response["model"]),
132
+ answers=response["answers"],
133
+ usage=TokenUsage.from_api(response.get("usage")),
134
+ request_id="local-weights",
135
+ latency_ms=latency_ms,
136
+ )
137
+
138
+ def _collate_on_head_device(self) -> Any:
139
+ """View of the model whose first parameter lives where the head does.
140
+
141
+ ``systemone`` collates the batch onto ``next(model.parameters()).device``.
142
+ Across a sharded load, the joint head reads ``token_ids`` from the batch
143
+ and indexes tensors on its own device, so the batch must be collated on
144
+ the head's device; accelerate's hooks move the backbone inputs to the
145
+ first shard as usual.
146
+ """
147
+ inner = self.model
148
+
149
+ class _CollateOnHeadDevice:
150
+ def parameters(self, *args: Any, **kwargs: Any) -> Any:
151
+ return inner.head.parameters()
152
+
153
+ def __call__(self, batch: dict[str, Any]) -> Any:
154
+ return inner(batch)
155
+
156
+ def __getattr__(self, name: str) -> Any:
157
+ return getattr(inner, name)
158
+
159
+ return _CollateOnHeadDevice()
@@ -0,0 +1,274 @@
1
+ {
2
+ "meta": {
3
+ "mode": "local",
4
+ "scorer": "open-weights Cloudflare/clef-flash (float16, local GPU)",
5
+ "transport": "local torch forward pass",
6
+ "dataset": [
7
+ "/kaggle/working/clef-compactor/evals/data/compaction_suite.jsonl"
8
+ ],
9
+ "relevance_threshold": 0.5,
10
+ "cost_model": "self-hosted open weights: no per-token API cost",
11
+ "price_note": "Output-token pricing is unpublished; only input tokens are ever costed."
12
+ },
13
+ "metrics": {
14
+ "cases": 20,
15
+ "chunks": 104,
16
+ "chunk_accuracy": 0.7115,
17
+ "kept_precision": 0.7705,
18
+ "relevant_recall": 0.746,
19
+ "kept_f1": 0.7581,
20
+ "input_tokens": 1582,
21
+ "output_tokens": 975,
22
+ "avg_saved_fraction": 0.3837,
23
+ "latency_ms": {
24
+ "p50": 1273.86,
25
+ "p95": 1406.64,
26
+ "p99": 3404.95
27
+ },
28
+ "api_input_tokens": 16229,
29
+ "cost_usd_per_1k_calls": 0.194748,
30
+ "cost_model": "self-hosted open weights: no per-token API cost"
31
+ },
32
+ "per_case": [
33
+ {
34
+ "case_id": "support-refund-001",
35
+ "correct": 4,
36
+ "total": 5,
37
+ "kept_relevant": 3,
38
+ "total_relevant": 3,
39
+ "kept_irrelevant": 1,
40
+ "input_tokens": 74,
41
+ "output_tokens": 58,
42
+ "latency_ms": 3404.9475499999744,
43
+ "api_input_tokens": 778
44
+ },
45
+ {
46
+ "case_id": "support-shipping-002",
47
+ "correct": 4,
48
+ "total": 5,
49
+ "kept_relevant": 2,
50
+ "total_relevant": 2,
51
+ "kept_irrelevant": 1,
52
+ "input_tokens": 67,
53
+ "output_tokens": 45,
54
+ "latency_ms": 1406.6402149999249,
55
+ "api_input_tokens": 771
56
+ },
57
+ {
58
+ "case_id": "tech-python-003",
59
+ "correct": 5,
60
+ "total": 6,
61
+ "kept_relevant": 3,
62
+ "total_relevant": 3,
63
+ "kept_irrelevant": 1,
64
+ "input_tokens": 101,
65
+ "output_tokens": 76,
66
+ "latency_ms": 1362.4956160000465,
67
+ "api_input_tokens": 935
68
+ },
69
+ {
70
+ "case_id": "tech-db-004",
71
+ "correct": 2,
72
+ "total": 5,
73
+ "kept_relevant": 1,
74
+ "total_relevant": 3,
75
+ "kept_irrelevant": 1,
76
+ "input_tokens": 73,
77
+ "output_tokens": 32,
78
+ "latency_ms": 1252.008700000033,
79
+ "api_input_tokens": 783
80
+ },
81
+ {
82
+ "case_id": "finance-fees-005",
83
+ "correct": 5,
84
+ "total": 5,
85
+ "kept_relevant": 3,
86
+ "total_relevant": 3,
87
+ "kept_irrelevant": 0,
88
+ "input_tokens": 68,
89
+ "output_tokens": 44,
90
+ "latency_ms": 1249.5986689999654,
91
+ "api_input_tokens": 779
92
+ },
93
+ {
94
+ "case_id": "travel-visa-006",
95
+ "correct": 3,
96
+ "total": 5,
97
+ "kept_relevant": 2,
98
+ "total_relevant": 3,
99
+ "kept_irrelevant": 1,
100
+ "input_tokens": 81,
101
+ "output_tokens": 57,
102
+ "latency_ms": 1261.9110020000335,
103
+ "api_input_tokens": 791
104
+ },
105
+ {
106
+ "case_id": "health-insurance-007",
107
+ "correct": 3,
108
+ "total": 5,
109
+ "kept_relevant": 3,
110
+ "total_relevant": 3,
111
+ "kept_irrelevant": 2,
112
+ "input_tokens": 87,
113
+ "output_tokens": 87,
114
+ "latency_ms": 1253.8356179999255,
115
+ "api_input_tokens": 795
116
+ },
117
+ {
118
+ "case_id": "devops-k8s-008",
119
+ "correct": 4,
120
+ "total": 5,
121
+ "kept_relevant": 2,
122
+ "total_relevant": 3,
123
+ "kept_irrelevant": 0,
124
+ "input_tokens": 78,
125
+ "output_tokens": 33,
126
+ "latency_ms": 831.2345519999553,
127
+ "api_input_tokens": 783
128
+ },
129
+ {
130
+ "case_id": "legal-privacy-009",
131
+ "correct": 4,
132
+ "total": 6,
133
+ "kept_relevant": 2,
134
+ "total_relevant": 4,
135
+ "kept_irrelevant": 0,
136
+ "input_tokens": 90,
137
+ "output_tokens": 26,
138
+ "latency_ms": 1363.2799339999337,
139
+ "api_input_tokens": 924
140
+ },
141
+ {
142
+ "case_id": "product-pricing-010",
143
+ "correct": 3,
144
+ "total": 5,
145
+ "kept_relevant": 3,
146
+ "total_relevant": 3,
147
+ "kept_irrelevant": 2,
148
+ "input_tokens": 69,
149
+ "output_tokens": 69,
150
+ "latency_ms": 1254.592740000021,
151
+ "api_input_tokens": 777
152
+ },
153
+ {
154
+ "case_id": "science-space-011",
155
+ "correct": 4,
156
+ "total": 5,
157
+ "kept_relevant": 3,
158
+ "total_relevant": 3,
159
+ "kept_irrelevant": 1,
160
+ "input_tokens": 80,
161
+ "output_tokens": 66,
162
+ "latency_ms": 1273.8597500000424,
163
+ "api_input_tokens": 781
164
+ },
165
+ {
166
+ "case_id": "education-math-012",
167
+ "correct": 2,
168
+ "total": 5,
169
+ "kept_relevant": 1,
170
+ "total_relevant": 4,
171
+ "kept_irrelevant": 0,
172
+ "input_tokens": 103,
173
+ "output_tokens": 17,
174
+ "latency_ms": 1268.813750999925,
175
+ "api_input_tokens": 808
176
+ },
177
+ {
178
+ "case_id": "support-account-013",
179
+ "correct": 3,
180
+ "total": 5,
181
+ "kept_relevant": 1,
182
+ "total_relevant": 3,
183
+ "kept_irrelevant": 0,
184
+ "input_tokens": 79,
185
+ "output_tokens": 21,
186
+ "latency_ms": 1265.0558409999348,
187
+ "api_input_tokens": 787
188
+ },
189
+ {
190
+ "case_id": "ops-oncall-014",
191
+ "correct": 6,
192
+ "total": 6,
193
+ "kept_relevant": 4,
194
+ "total_relevant": 4,
195
+ "kept_irrelevant": 0,
196
+ "input_tokens": 79,
197
+ "output_tokens": 57,
198
+ "latency_ms": 1367.6004610000518,
199
+ "api_input_tokens": 914
200
+ },
201
+ {
202
+ "case_id": "hr-leave-015",
203
+ "correct": 3,
204
+ "total": 5,
205
+ "kept_relevant": 3,
206
+ "total_relevant": 3,
207
+ "kept_irrelevant": 2,
208
+ "input_tokens": 76,
209
+ "output_tokens": 76,
210
+ "latency_ms": 1306.5989709999712,
211
+ "api_input_tokens": 784
212
+ },
213
+ {
214
+ "case_id": "ml-training-016",
215
+ "correct": 5,
216
+ "total": 6,
217
+ "kept_relevant": 3,
218
+ "total_relevant": 4,
219
+ "kept_irrelevant": 0,
220
+ "input_tokens": 83,
221
+ "output_tokens": 42,
222
+ "latency_ms": 1375.9791559999712,
223
+ "api_input_tokens": 918
224
+ },
225
+ {
226
+ "case_id": "ecommerce-inventory-017",
227
+ "correct": 5,
228
+ "total": 5,
229
+ "kept_relevant": 3,
230
+ "total_relevant": 3,
231
+ "kept_irrelevant": 0,
232
+ "input_tokens": 70,
233
+ "output_tokens": 49,
234
+ "latency_ms": 834.4232719999809,
235
+ "api_input_tokens": 778
236
+ },
237
+ {
238
+ "case_id": "security-auth-018",
239
+ "correct": 3,
240
+ "total": 5,
241
+ "kept_relevant": 2,
242
+ "total_relevant": 3,
243
+ "kept_irrelevant": 1,
244
+ "input_tokens": 67,
245
+ "output_tokens": 41,
246
+ "latency_ms": 1267.2506279999425,
247
+ "api_input_tokens": 772
248
+ },
249
+ {
250
+ "case_id": "nutrition-diet-019",
251
+ "correct": 4,
252
+ "total": 5,
253
+ "kept_relevant": 2,
254
+ "total_relevant": 3,
255
+ "kept_irrelevant": 0,
256
+ "input_tokens": 81,
257
+ "output_tokens": 44,
258
+ "latency_ms": 1292.3661680000578,
259
+ "api_input_tokens": 785
260
+ },
261
+ {
262
+ "case_id": "tools-git-020",
263
+ "correct": 2,
264
+ "total": 5,
265
+ "kept_relevant": 1,
266
+ "total_relevant": 3,
267
+ "kept_irrelevant": 1,
268
+ "input_tokens": 76,
269
+ "output_tokens": 35,
270
+ "latency_ms": 1304.9001580000095,
271
+ "api_input_tokens": 786
272
+ }
273
+ ]
274
+ }
@@ -7,7 +7,8 @@
7
7
  "C:\\Code Main\\clef-compactor\\evals\\data\\compaction_suite.jsonl"
8
8
  ],
9
9
  "relevance_threshold": 0.5,
10
- "price_note": "Clef input tokens cost $0.24/M; output pricing is unpublished."
10
+ "cost_model": "input tokens x $0.24/M (Cloudflare published price)",
11
+ "price_note": "Output-token pricing is unpublished; only input tokens are ever costed."
11
12
  },
12
13
  "metrics": {
13
14
  "cases": 20,
@@ -20,9 +21,9 @@
20
21
  "output_tokens": 1044,
21
22
  "avg_saved_fraction": 0.3401,
22
23
  "latency_ms": {
23
- "p50": 0.11,
24
- "p95": 0.16,
25
- "p99": 150.94
24
+ "p50": 0.12,
25
+ "p95": 0.17,
26
+ "p99": 156.81
26
27
  },
27
28
  "api_input_tokens": null,
28
29
  "cost_usd_per_1k_calls": 0.018984,
@@ -38,7 +39,7 @@
38
39
  "kept_irrelevant": 0,
39
40
  "input_tokens": 74,
40
41
  "output_tokens": 43,
41
- "latency_ms": 150.935500000287,
42
+ "latency_ms": 156.8135000006805,
42
43
  "api_input_tokens": null
43
44
  },
44
45
  {
@@ -50,7 +51,7 @@
50
51
  "kept_irrelevant": 0,
51
52
  "input_tokens": 67,
52
53
  "output_tokens": 29,
53
- "latency_ms": 0.1609999999345746,
54
+ "latency_ms": 0.16309999955410603,
54
55
  "api_input_tokens": null
55
56
  },
56
57
  {
@@ -62,7 +63,7 @@
62
63
  "kept_irrelevant": 0,
63
64
  "input_tokens": 101,
64
65
  "output_tokens": 64,
65
- "latency_ms": 0.15769999981785077,
66
+ "latency_ms": 0.1578999999765074,
66
67
  "api_input_tokens": null
67
68
  },
68
69
  {
@@ -74,7 +75,7 @@
74
75
  "kept_irrelevant": 0,
75
76
  "input_tokens": 73,
76
77
  "output_tokens": 48,
77
- "latency_ms": 0.1133000000663742,
78
+ "latency_ms": 0.1140000003942987,
78
79
  "api_input_tokens": null
79
80
  },
80
81
  {
@@ -86,7 +87,7 @@
86
87
  "kept_irrelevant": 0,
87
88
  "input_tokens": 68,
88
89
  "output_tokens": 44,
89
- "latency_ms": 0.10729999985414906,
90
+ "latency_ms": 0.10680000013962854,
90
91
  "api_input_tokens": null
91
92
  },
92
93
  {
@@ -98,7 +99,7 @@
98
99
  "kept_irrelevant": 0,
99
100
  "input_tokens": 81,
100
101
  "output_tokens": 57,
101
- "latency_ms": 0.11080000012952951,
102
+ "latency_ms": 0.1368000002912595,
102
103
  "api_input_tokens": null
103
104
  },
104
105
  {
@@ -110,7 +111,7 @@
110
111
  "kept_irrelevant": 0,
111
112
  "input_tokens": 87,
112
113
  "output_tokens": 62,
113
- "latency_ms": 0.11819999963336159,
114
+ "latency_ms": 0.17209999987244373,
114
115
  "api_input_tokens": null
115
116
  },
116
117
  {
@@ -122,7 +123,7 @@
122
123
  "kept_irrelevant": 0,
123
124
  "input_tokens": 78,
124
125
  "output_tokens": 53,
125
- "latency_ms": 0.11480000011943048,
126
+ "latency_ms": 0.13159999980416615,
126
127
  "api_input_tokens": null
127
128
  },
128
129
  {
@@ -134,7 +135,7 @@
134
135
  "kept_irrelevant": 0,
135
136
  "input_tokens": 90,
136
137
  "output_tokens": 56,
137
- "latency_ms": 0.12699999979304266,
138
+ "latency_ms": 0.13320000016392441,
138
139
  "api_input_tokens": null
139
140
  },
140
141
  {
@@ -146,7 +147,7 @@
146
147
  "kept_irrelevant": 0,
147
148
  "input_tokens": 69,
148
149
  "output_tokens": 45,
149
- "latency_ms": 0.10059999976874678,
150
+ "latency_ms": 0.11009999980160501,
150
151
  "api_input_tokens": null
151
152
  },
152
153
  {
@@ -158,7 +159,7 @@
158
159
  "kept_irrelevant": 0,
159
160
  "input_tokens": 80,
160
161
  "output_tokens": 52,
161
- "latency_ms": 0.10849999989659409,
162
+ "latency_ms": 0.11050000011891825,
162
163
  "api_input_tokens": null
163
164
  },
164
165
  {
@@ -170,7 +171,7 @@
170
171
  "kept_irrelevant": 0,
171
172
  "input_tokens": 103,
172
173
  "output_tokens": 90,
173
- "latency_ms": 0.12549999973998638,
174
+ "latency_ms": 0.12269999933778308,
174
175
  "api_input_tokens": null
175
176
  },
176
177
  {
@@ -182,7 +183,7 @@
182
183
  "kept_irrelevant": 0,
183
184
  "input_tokens": 79,
184
185
  "output_tokens": 53,
185
- "latency_ms": 0.10330000031899544,
186
+ "latency_ms": 0.10440000005473848,
186
187
  "api_input_tokens": null
187
188
  },
188
189
  {
@@ -194,7 +195,7 @@
194
195
  "kept_irrelevant": 0,
195
196
  "input_tokens": 79,
196
197
  "output_tokens": 57,
197
- "latency_ms": 0.11709999989761855,
198
+ "latency_ms": 0.11859999995067483,
198
199
  "api_input_tokens": null
199
200
  },
200
201
  {
@@ -206,7 +207,7 @@
206
207
  "kept_irrelevant": 0,
207
208
  "input_tokens": 76,
208
209
  "output_tokens": 40,
209
- "latency_ms": 0.10100000008606003,
210
+ "latency_ms": 0.12630000037461286,
210
211
  "api_input_tokens": null
211
212
  },
212
213
  {
@@ -218,7 +219,7 @@
218
219
  "kept_irrelevant": 0,
219
220
  "input_tokens": 83,
220
221
  "output_tokens": 58,
221
- "latency_ms": 0.12579999975059764,
222
+ "latency_ms": 0.1384999995934777,
222
223
  "api_input_tokens": null
223
224
  },
224
225
  {
@@ -230,7 +231,7 @@
230
231
  "kept_irrelevant": 0,
231
232
  "input_tokens": 70,
232
233
  "output_tokens": 49,
233
- "latency_ms": 0.10109999993801466,
234
+ "latency_ms": 0.1022000005832524,
234
235
  "api_input_tokens": null
235
236
  },
236
237
  {
@@ -242,7 +243,7 @@
242
243
  "kept_irrelevant": 0,
243
244
  "input_tokens": 67,
244
245
  "output_tokens": 39,
245
- "latency_ms": 0.09580000005371403,
246
+ "latency_ms": 0.09659999977884581,
246
247
  "api_input_tokens": null
247
248
  },
248
249
  {
@@ -254,7 +255,7 @@
254
255
  "kept_irrelevant": 0,
255
256
  "input_tokens": 81,
256
257
  "output_tokens": 54,
257
- "latency_ms": 0.10539999993852689,
258
+ "latency_ms": 0.10689999999158317,
258
259
  "api_input_tokens": null
259
260
  },
260
261
  {
@@ -266,7 +267,7 @@
266
267
  "kept_irrelevant": 0,
267
268
  "input_tokens": 76,
268
269
  "output_tokens": 51,
269
- "latency_ms": 0.10370000018156134,
270
+ "latency_ms": 0.10449999990669312,
270
271
  "api_input_tokens": null
271
272
  }
272
273
  ]
@@ -181,7 +181,7 @@ def run_case(engine: ClefCompactor, case: EvalCase) -> CaseOutcome:
181
181
  )
182
182
 
183
183
 
184
- def aggregate(outcomes: Sequence[CaseOutcome]) -> dict[str, Any]:
184
+ def aggregate(outcomes: Sequence[CaseOutcome], cost_model: str) -> dict[str, Any]:
185
185
  """Reduce per-case outcomes into the reported metric block."""
186
186
  total_chunks = sum(outcome.total for outcome in outcomes)
187
187
  correct = sum(outcome.correct for outcome in outcomes)
@@ -219,7 +219,7 @@ def aggregate(outcomes: Sequence[CaseOutcome]) -> dict[str, Any]:
219
219
  "latency_ms": {"p50": percentile_share(0.50), "p95": percentile_share(0.95), "p99": percentile_share(0.99)},
220
220
  "api_input_tokens": api_input_tokens or None,
221
221
  "cost_usd_per_1k_calls": round(cost_per_1k, 6),
222
- "cost_model": f"input tokens x ${INPUT_PRICE_PER_M}/M (Cloudflare published price)",
222
+ "cost_model": cost_model,
223
223
  }
224
224
 
225
225
 
@@ -239,7 +239,12 @@ def print_laya_comparison() -> None:
239
239
  def main(argv: Sequence[str] | None = None) -> int:
240
240
  """Entry point; returns a process exit code (1 below --min-accuracy)."""
241
241
  parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
242
- parser.add_argument("--mode", choices=("replay", "live"), default="replay")
242
+ parser.add_argument("--mode", choices=("replay", "live", "local"), default="replay")
243
+ parser.add_argument("--model-path", default="Cloudflare/clef-flash",
244
+ help="HF repo id or snapshot dir for --mode local")
245
+ parser.add_argument("--dtype", default="float16", choices=("float16", "bfloat16", "float32"),
246
+ help="Torch dtype for --mode local (float16 on T4)")
247
+ parser.add_argument("--device-map", default="auto", help="Accelerate device map for --mode local")
243
248
  parser.add_argument("--data", default=DATA_GLOB, help="Glob of dataset .jsonl files")
244
249
  parser.add_argument("--out", default=str(DEFAULT_OUT), help="Path of the results JSON")
245
250
  parser.add_argument("--seed", type=int, default=20261001, help="Replay noise seed")
@@ -253,6 +258,28 @@ def main(argv: Sequence[str] | None = None) -> int:
253
258
  engine = ClefCompactor()
254
259
  scorer = "cloudflare-clef (live)"
255
260
  transport = "https://api.cloudflare.com"
261
+ cost_model = "input tokens x $0.24/M (Cloudflare published price)"
262
+ elif args.mode == "local":
263
+ sys.path.insert(0, str(Path(__file__).resolve().parent))
264
+ from backends import LocalClefClient
265
+
266
+ model_label = args.model_path.split("/")[-1]
267
+ client = LocalClefClient(
268
+ Settings(account_id="local", api_token="local", model=model_label, max_retries=0),
269
+ args.model_path,
270
+ dtype=args.dtype,
271
+ device_map=args.device_map,
272
+ )
273
+ engine = ClefCompactor(
274
+ account_id="local",
275
+ api_token="local",
276
+ model=model_label,
277
+ relevance_threshold=args.threshold,
278
+ client=client, # type: ignore[arg-type]
279
+ )
280
+ scorer = f"open-weights {args.model_path} ({args.dtype}, local GPU)"
281
+ transport = "local torch forward pass"
282
+ cost_model = "self-hosted open weights: no per-token API cost"
256
283
  else:
257
284
  labels_by_query = {case.query: case.labels for case in cases}
258
285
  engine = ClefCompactor(
@@ -263,9 +290,10 @@ def main(argv: Sequence[str] | None = None) -> int:
263
290
  )
264
291
  scorer = f"simulated-clef seed={args.seed} (pipeline validation, not model quality)"
265
292
  transport = "in-process mock"
293
+ cost_model = "input tokens x $0.24/M (Cloudflare published price)"
266
294
 
267
295
  outcomes = [run_case(engine, case) for case in cases]
268
- metrics = aggregate(outcomes)
296
+ metrics = aggregate(outcomes, cost_model)
269
297
  payload = {
270
298
  "meta": {
271
299
  "mode": args.mode,
@@ -273,7 +301,8 @@ def main(argv: Sequence[str] | None = None) -> int:
273
301
  "transport": transport,
274
302
  "dataset": sorted(glob_module.glob(args.data)),
275
303
  "relevance_threshold": args.threshold,
276
- "price_note": "Clef input tokens cost $0.24/M; output pricing is unpublished.",
304
+ "cost_model": cost_model,
305
+ "price_note": "Output-token pricing is unpublished; only input tokens are ever costed.",
277
306
  },
278
307
  "metrics": metrics,
279
308
  "per_case": [vars(outcome) for outcome in outcomes],
@@ -2,6 +2,7 @@
2
2
  "cells": [
3
3
  {
4
4
  "cell_type": "markdown",
5
+ "id": "7fb27b941602401d91542211134fc71a",
5
6
  "metadata": {},
6
7
  "source": [
7
8
  "# clef-compactor demo\n",
@@ -12,10 +13,12 @@
12
13
  {
13
14
  "cell_type": "code",
14
15
  "execution_count": null,
16
+ "id": "acae54e37e7d407bbb7b55eff062a284",
15
17
  "metadata": {},
16
18
  "outputs": [],
17
19
  "source": [
18
20
  "import httpx\n",
21
+ "\n",
19
22
  "from clef_compactor import ClefCompactor\n",
20
23
  "from clef_compactor.client import ClefClient\n",
21
24
  "from clef_compactor.config import Settings\n",
@@ -50,6 +53,7 @@
50
53
  {
51
54
  "cell_type": "code",
52
55
  "execution_count": null,
56
+ "id": "9a63283cbaf04dbcab1f6479b197f3a8",
53
57
  "metadata": {},
54
58
  "outputs": [],
55
59
  "source": [
@@ -71,6 +75,7 @@
71
75
  {
72
76
  "cell_type": "code",
73
77
  "execution_count": null,
78
+ "id": "8dd0d8092fe74a7c96281538738b07e2",
74
79
  "metadata": {},
75
80
  "outputs": [],
76
81
  "source": [
@@ -83,6 +88,7 @@
83
88
  },
84
89
  {
85
90
  "cell_type": "markdown",
91
+ "id": "72eea5119410473aa328ad9291626812",
86
92
  "metadata": {},
87
93
  "source": [
88
94
  "## The compacted context\n",
@@ -93,6 +99,7 @@
93
99
  {
94
100
  "cell_type": "code",
95
101
  "execution_count": null,
102
+ "id": "8edb47106e1a46a883d545849b8ab81b",
96
103
  "metadata": {},
97
104
  "outputs": [],
98
105
  "source": [
@@ -103,6 +110,7 @@
103
110
  {
104
111
  "cell_type": "code",
105
112
  "execution_count": null,
113
+ "id": "10185d26023b46108eb7d9f57d49d2b3",
106
114
  "metadata": {},
107
115
  "outputs": [],
108
116
  "source": [
@@ -116,6 +124,7 @@
116
124
  },
117
125
  {
118
126
  "cell_type": "markdown",
127
+ "id": "8763a12b2bbd4a93a75aff182afb95dc",
119
128
  "metadata": {},
120
129
  "source": [
121
130
  "## Going live\n",
@@ -130,8 +139,15 @@
130
139
  }
131
140
  ],
132
141
  "metadata": {
133
- "kernelspec": {"display_name": "Python 3", "language": "python", "name": "python3"},
134
- "language_info": {"name": "python", "version": "3.10"}
142
+ "kernelspec": {
143
+ "display_name": "Python 3",
144
+ "language": "python",
145
+ "name": "python3"
146
+ },
147
+ "language_info": {
148
+ "name": "python",
149
+ "version": "3.10"
150
+ }
135
151
  },
136
152
  "nbformat": 4,
137
153
  "nbformat_minor": 5
@@ -0,0 +1,13 @@
1
+ {
2
+ "id": "gjusev/clef-compactor-evals",
3
+ "title": "clef-compactor-evals",
4
+ "code_file": "script.py",
5
+ "language": "python",
6
+ "kernel_type": "script",
7
+ "is_private": true,
8
+ "enable_gpu": true,
9
+ "enable_internet": true,
10
+ "dataset_sources": [],
11
+ "kernel_sources": [],
12
+ "model_sources": []
13
+ }
@@ -0,0 +1,47 @@
1
+ """Kaggle kernel: measure clef-compactor against the open-weights Clef model.
2
+
3
+ Follows the verified laya-evals kernel pattern (github.com/Gjusev/laya-evals):
4
+ clone the public repo, pip install it, run the measurement, print the verdict.
5
+ Outputs land in /kaggle/working/ and come back with `kaggle kernels output`.
6
+
7
+ Hardware: Kaggle T4 x2 (2 x 16 GB). We run Cloudflare/clef-flash (9B) in
8
+ float16 sharded across both GPUs with device_map="auto"; the 27B model does
9
+ not fit on this hardware. Credentials are never needed and never embedded:
10
+ the local backend loads weights from Hugging Face.
11
+ """
12
+
13
+ import subprocess
14
+ import sys
15
+
16
+
17
+ def run(cmd: str) -> None:
18
+ print(f"$ {cmd}", flush=True)
19
+ subprocess.run(cmd, shell=True, check=True)
20
+
21
+
22
+ # 0. GPU sanity: Kaggle can silently boot without the GPU you asked for.
23
+ print("=== GPU check ===", flush=True)
24
+ run("nvidia-smi")
25
+
26
+ # 1. Repo + dependencies. transformers 5.10.2 is the version the Clef release
27
+ # was tested with; Qwen3.5 support needs the 5.x line.
28
+ run("git clone --depth 1 https://github.com/Gjusev/clef-compactor.git")
29
+ run(f"{sys.executable} -m pip install -q --no-input ./clef-compactor")
30
+ run(f"{sys.executable} -m pip install -q --no-input 'transformers==5.10.2' accelerate safetensors")
31
+
32
+ run(f"{sys.executable} -c 'import torch, transformers; "
33
+ "print(torch.__version__, torch.cuda.is_available(), transformers.__version__)'")
34
+
35
+ # 2. The measurement: full pipeline (batching, ranking, budget) against the
36
+ # locally loaded clef-flash weights, scored on the committed gold dataset.
37
+ run(
38
+ f"{sys.executable} clef-compactor/evals/run_eval.py "
39
+ "--mode local --model-path Cloudflare/clef-flash --dtype float16 "
40
+ "--out /kaggle/working/results-local.json --compare-laya"
41
+ )
42
+
43
+ # 3. Optional stretch: the 27B model in float16 needs ~54 GB and does not fit
44
+ # on 2 x T4; skipped on purpose. If Kaggle ever offers bigger GPUs, run:
45
+ # run_eval.py --mode local --model-path Cloudflare/clef
46
+
47
+ print("KERNEL COMPLETE", flush=True)
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "clef-compactor"
3
- version = "0.2.0"
3
+ version = "0.2.1"
4
4
  description = "Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise."
5
5
  readme = "README.md"
6
6
  license = { text = "Apache-2.0" }
@@ -29,7 +29,7 @@ from .exceptions import (
29
29
  )
30
30
  from .models import ClefReply, CompactResult, DropReason, ScoredChunk, TokenUsage
31
31
 
32
- __version__ = "0.2.0"
32
+ __version__ = "0.2.1"
33
33
 
34
34
  __all__ = [
35
35
  "AsyncClefCompactor",
@@ -8,7 +8,9 @@ from conftest import envelope, noul_answer
8
8
 
9
9
 
10
10
  def test_package_exposes_version() -> None:
11
- assert clef_compactor.__version__ == "0.2.0"
11
+ from importlib.metadata import version
12
+
13
+ assert clef_compactor.__version__ == version("clef-compactor")
12
14
 
13
15
 
14
16
  def test_public_api_surface() -> None:
File without changes
File without changes