clef-compactor 0.2.0__tar.gz → 0.2.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/workflows/publish.yml +1 -1
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.gitignore +2 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/PKG-INFO +34 -16
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/README.md +33 -15
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/index.html +10 -9
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/README.md +14 -1
- clef_compactor-0.2.1/evals/backends.py +159 -0
- clef_compactor-0.2.1/evals/results/results-local-t4x2.json +274 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/results/results.json +25 -24
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/run_eval.py +34 -5
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/examples/demo.ipynb +18 -2
- clef_compactor-0.2.1/kaggle-kernel/kernel-metadata.json +13 -0
- clef_compactor-0.2.1/kaggle-kernel/script.py +47 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/pyproject.toml +1 -1
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/__init__.py +1 -1
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_smoke.py +3 -1
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/actions/clef-evals/action.yml +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/actions/clef-evals/check_regression.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/workflows/test.yml +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/LICENSE +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/Makefile +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/assets/logo.png +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/assets/social-preview.png +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/brag.jpg +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/brag.mp4 +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/docs/how-it-works.svg +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/evals/data/compaction_suite.jsonl +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/scripts/make_assets.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/cli.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/client.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/compat/__init__.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/compat/openai.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/config.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/core.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/exceptions.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/__init__.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/langchain.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/integrations/llamaindex.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/models.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/src/clef_compactor/py.typed +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/conftest.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/fixtures/error_envelope.json +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/fixtures/noul_score_response.json +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_cli.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_client.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_compat_openai.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_config.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_core.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_exceptions.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integration_live.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integrations_langchain.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_integrations_llamaindex.py +0 -0
- {clef_compactor-0.2.0 → clef_compactor-0.2.1}/tests/test_models.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.5
|
|
2
2
|
Name: clef-compactor
|
|
3
|
-
Version: 0.2.
|
|
3
|
+
Version: 0.2.1
|
|
4
4
|
Summary: Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise.
|
|
5
5
|
Project-URL: Homepage, https://github.com/Gjusev/clef-compactor
|
|
6
6
|
Project-URL: Repository, https://github.com/Gjusev/clef-compactor
|
|
@@ -194,21 +194,38 @@ query_engine = RetrieverQueryEngine(
|
|
|
194
194
|
|
|
195
195
|
## Measured results
|
|
196
196
|
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
197
|
+
Three measurement sources, each labelled for what it actually proves.
|
|
198
|
+
|
|
199
|
+
**Real model, modest hardware** — open-weights `Cloudflare/clef-flash` (9B),
|
|
200
|
+
float16 sharded across 2x Kaggle T4, 20 cases, 104 chunks. Produced by
|
|
201
|
+
`python evals/run_eval.py --mode local`; results committed in
|
|
202
|
+
`evals/results/results-local-t4x2.json` and reproducible from the
|
|
203
|
+
[`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
|
|
204
|
+
kernel.
|
|
200
205
|
|
|
201
206
|
| metric | value |
|
|
202
207
|
|---|---|
|
|
203
|
-
| chunk accuracy | 0.
|
|
204
|
-
| kept precision |
|
|
205
|
-
| relevant recall | 0.
|
|
206
|
-
| kept F1 | 0.
|
|
207
|
-
| context tokens saved |
|
|
208
|
-
|
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
208
|
+
| chunk accuracy | 0.712 |
|
|
209
|
+
| kept precision | 0.771 |
|
|
210
|
+
| relevant recall | 0.746 |
|
|
211
|
+
| kept F1 | 0.758 |
|
|
212
|
+
| context tokens saved | 38.4% |
|
|
213
|
+
| scoring latency p50 / p95 | 1,274 / 1,407 ms |
|
|
214
|
+
| cost per 1k calls | $0.00 (self-hosted weights) |
|
|
215
|
+
|
|
216
|
+
**Pipeline validation** — deterministic simulated scorer, same dataset, seed
|
|
217
|
+
20261001 (`python evals/run_eval.py`). Proves the ranking and budget
|
|
218
|
+
machinery, not model quality: chunk accuracy 0.990, kept F1 0.992, 34.0%
|
|
219
|
+
tokens saved.
|
|
220
|
+
|
|
221
|
+
**Hosted API** — pending credentials; `--mode live` produces it.
|
|
222
|
+
|
|
223
|
+
Reading the real numbers plainly: on a T4 the 9B model makes the right
|
|
224
|
+
keep/drop call 71% of the time, keeps three quarters of the relevant chunks,
|
|
225
|
+
and still removes 38% of the context tokens. Local latency is dominated by
|
|
226
|
+
the modest GPU and the torch fallback for Qwen3.5's linear attention (the
|
|
227
|
+
fast-path kernels were not installed in the kernel); the hosted endpoint
|
|
228
|
+
reports 38.8 ms median for clef-flash on Cloudflare's own hardware.
|
|
212
229
|
|
|
213
230
|
## clef vs laya
|
|
214
231
|
|
|
@@ -248,9 +265,10 @@ and honor `Retry-After`.
|
|
|
248
265
|
|
|
249
266
|
## Limitations
|
|
250
267
|
|
|
251
|
-
- **
|
|
252
|
-
|
|
253
|
-
|
|
268
|
+
- **Real-model numbers are from a 9B model on a T4 pair, not from the hosted
|
|
269
|
+
endpoint.** The hosted API (and the 27B model, which needs ~54 GB) may
|
|
270
|
+
score better. Measuring the hosted endpoint is one command away:
|
|
271
|
+
`--mode live` with credentials.
|
|
254
272
|
- **Latency.** clef's median decision latency is 209 ms (38.8 ms for
|
|
255
273
|
clef-flash) plus network. laya keeps the whole job local at 5.8 ms. If you
|
|
256
274
|
need sub-10 ms compaction on every request, see
|
|
@@ -153,21 +153,38 @@ query_engine = RetrieverQueryEngine(
|
|
|
153
153
|
|
|
154
154
|
## Measured results
|
|
155
155
|
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
156
|
+
Three measurement sources, each labelled for what it actually proves.
|
|
157
|
+
|
|
158
|
+
**Real model, modest hardware** — open-weights `Cloudflare/clef-flash` (9B),
|
|
159
|
+
float16 sharded across 2x Kaggle T4, 20 cases, 104 chunks. Produced by
|
|
160
|
+
`python evals/run_eval.py --mode local`; results committed in
|
|
161
|
+
`evals/results/results-local-t4x2.json` and reproducible from the
|
|
162
|
+
[`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
|
|
163
|
+
kernel.
|
|
159
164
|
|
|
160
165
|
| metric | value |
|
|
161
166
|
|---|---|
|
|
162
|
-
| chunk accuracy | 0.
|
|
163
|
-
| kept precision |
|
|
164
|
-
| relevant recall | 0.
|
|
165
|
-
| kept F1 | 0.
|
|
166
|
-
| context tokens saved |
|
|
167
|
-
|
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
167
|
+
| chunk accuracy | 0.712 |
|
|
168
|
+
| kept precision | 0.771 |
|
|
169
|
+
| relevant recall | 0.746 |
|
|
170
|
+
| kept F1 | 0.758 |
|
|
171
|
+
| context tokens saved | 38.4% |
|
|
172
|
+
| scoring latency p50 / p95 | 1,274 / 1,407 ms |
|
|
173
|
+
| cost per 1k calls | $0.00 (self-hosted weights) |
|
|
174
|
+
|
|
175
|
+
**Pipeline validation** — deterministic simulated scorer, same dataset, seed
|
|
176
|
+
20261001 (`python evals/run_eval.py`). Proves the ranking and budget
|
|
177
|
+
machinery, not model quality: chunk accuracy 0.990, kept F1 0.992, 34.0%
|
|
178
|
+
tokens saved.
|
|
179
|
+
|
|
180
|
+
**Hosted API** — pending credentials; `--mode live` produces it.
|
|
181
|
+
|
|
182
|
+
Reading the real numbers plainly: on a T4 the 9B model makes the right
|
|
183
|
+
keep/drop call 71% of the time, keeps three quarters of the relevant chunks,
|
|
184
|
+
and still removes 38% of the context tokens. Local latency is dominated by
|
|
185
|
+
the modest GPU and the torch fallback for Qwen3.5's linear attention (the
|
|
186
|
+
fast-path kernels were not installed in the kernel); the hosted endpoint
|
|
187
|
+
reports 38.8 ms median for clef-flash on Cloudflare's own hardware.
|
|
171
188
|
|
|
172
189
|
## clef vs laya
|
|
173
190
|
|
|
@@ -207,9 +224,10 @@ and honor `Retry-After`.
|
|
|
207
224
|
|
|
208
225
|
## Limitations
|
|
209
226
|
|
|
210
|
-
- **
|
|
211
|
-
|
|
212
|
-
|
|
227
|
+
- **Real-model numbers are from a 9B model on a T4 pair, not from the hosted
|
|
228
|
+
endpoint.** The hosted API (and the 27B model, which needs ~54 GB) may
|
|
229
|
+
score better. Measuring the hosted endpoint is one command away:
|
|
230
|
+
`--mode live` with credentials.
|
|
213
231
|
- **Latency.** clef's median decision latency is 209 ms (38.8 ms for
|
|
214
232
|
clef-flash) plus network. laya keeps the whole job local at 5.8 ms. If you
|
|
215
233
|
need sub-10 ms compaction on every request, see
|
|
@@ -164,16 +164,17 @@
|
|
|
164
164
|
|
|
165
165
|
<section id="results" class="wrap">
|
|
166
166
|
<div class="kicker">measured results</div>
|
|
167
|
-
<h2 class="serif">
|
|
168
|
-
<p class="lede">
|
|
167
|
+
<h2 class="serif">Real model, measured on real hardware</h2>
|
|
168
|
+
<p class="lede">Open-weights clef-flash 9B, float16 on 2x Kaggle T4, 20 hand-labelled cases, 104 chunks. Reproducible from the <a href="https://www.kaggle.com/code/gjusev/clef-compactor-evals">public kernel</a>. The replay baseline (simulated scorer) and a hosted-API run are documented in the repo.</p>
|
|
169
169
|
<table class="reveal" style="--i:1">
|
|
170
|
-
<tr><th>metric</th><th>
|
|
171
|
-
<tr><td>chunk accuracy</td><td class="num">0.990</td></tr>
|
|
172
|
-
<tr><td>kept precision</td><td class="num">1.000</td></tr>
|
|
173
|
-
<tr><td>relevant recall</td><td class="num">0.984</td></tr>
|
|
174
|
-
<tr><td>kept F1</td><td class="num">0.992</td></tr>
|
|
175
|
-
<tr><td>context tokens saved</td><td class="num best">34.0%</td></tr>
|
|
176
|
-
<tr><td>
|
|
170
|
+
<tr><th>metric</th><th>open-weights 9B on 2xT4</th><th>replay baseline</th></tr>
|
|
171
|
+
<tr><td>chunk accuracy</td><td class="num">0.712</td><td class="num">0.990</td></tr>
|
|
172
|
+
<tr><td>kept precision</td><td class="num">0.771</td><td class="num">1.000</td></tr>
|
|
173
|
+
<tr><td>relevant recall</td><td class="num">0.746</td><td class="num">0.984</td></tr>
|
|
174
|
+
<tr><td>kept F1</td><td class="num">0.758</td><td class="num">0.992</td></tr>
|
|
175
|
+
<tr><td>context tokens saved</td><td class="num best">38.4%</td><td class="num">34.0%</td></tr>
|
|
176
|
+
<tr><td>scoring latency p50</td><td class="num">1,274 ms</td><td class="num"><1 ms</td></tr>
|
|
177
|
+
<tr><td>cost per 1k calls</td><td class="num best">$0.00 self-hosted</td><td class="num">$0.019</td></tr>
|
|
177
178
|
</table>
|
|
178
179
|
</section>
|
|
179
180
|
|
|
@@ -16,7 +16,10 @@ by hand against the query; they are the ground truth for every metric.
|
|
|
16
16
|
# Pipeline validation: real compaction code, deterministic simulated scorer
|
|
17
17
|
python evals/run_eval.py
|
|
18
18
|
|
|
19
|
-
# Real
|
|
19
|
+
# Real model weights, no Cloudflare account: 9B on a local GPU (T4 pair tested)
|
|
20
|
+
python evals/run_eval.py --mode local --model-path Cloudflare/clef-flash
|
|
21
|
+
|
|
22
|
+
# Hosted endpoint: the published prices and latencies apply
|
|
20
23
|
python evals/run_eval.py --mode live
|
|
21
24
|
|
|
22
25
|
# CI gate: exit code 1 when chunk accuracy drops below the threshold
|
|
@@ -26,6 +29,16 @@ python evals/run_eval.py --min-accuracy 0.9
|
|
|
26
29
|
python evals/run_eval.py --compare-laya
|
|
27
30
|
```
|
|
28
31
|
|
|
32
|
+
## Measured so far
|
|
33
|
+
|
|
34
|
+
`results/results.json` is the replay baseline (simulated scorer).
|
|
35
|
+
`results/results-local-t4x2.json` is the real open-weights clef-flash 9B
|
|
36
|
+
measured on a Kaggle T4 pair via the
|
|
37
|
+
[`clef-compactor-evals`](https://www.kaggle.com/code/gjusev/clef-compactor-evals)
|
|
38
|
+
kernel: chunk accuracy 0.712, kept F1 0.758, 38.4% of context tokens removed,
|
|
39
|
+
scoring latency ~1.27 s p50 (T4, torch attention fallback). The 27B model
|
|
40
|
+
needs ~54 GB and does not fit on that hardware.
|
|
41
|
+
|
|
29
42
|
Every run writes `results/results.json`: aggregate metrics plus one entry per
|
|
30
43
|
case. The committed file was produced by `--mode replay`, so it is
|
|
31
44
|
reproducible byte for byte given the same seed.
|
|
@@ -0,0 +1,159 @@
|
|
|
1
|
+
"""Local inference backend: run the open-weights Clef models without the API.
|
|
2
|
+
|
|
3
|
+
Wraps ``joint_schema_model.systemone`` from the Hugging Face release
|
|
4
|
+
(https://huggingface.co/Cloudflare/clef) in the same ``ask()`` interface the
|
|
5
|
+
sync compactor expects, so the whole eval pipeline runs unmodified against
|
|
6
|
+
weights loaded on a local GPU.
|
|
7
|
+
|
|
8
|
+
The stock ``load_release_model`` pins the whole backbone to one device via
|
|
9
|
+
``device_map={"": device}``. That fits clef-flash (9B, ~18 GB in fp16) on a
|
|
10
|
+
single H200 but not on a single 16 GB T4, so :class:`LocalClefClient`
|
|
11
|
+
replicates the loader with ``device_map="auto"`` and places the joint schema
|
|
12
|
+
head on the device that holds the backbone's final layers.
|
|
13
|
+
|
|
14
|
+
Requirements (not package dependencies; installed by the Kaggle kernel):
|
|
15
|
+
``torch``, ``transformers>=5.10``, ``accelerate``, ``safetensors``,
|
|
16
|
+
``huggingface_hub``, ``pillow``.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import sys
|
|
22
|
+
import time
|
|
23
|
+
from pathlib import Path
|
|
24
|
+
from typing import Any
|
|
25
|
+
|
|
26
|
+
from clef_compactor.config import Settings
|
|
27
|
+
from clef_compactor.models import ClefReply, TokenUsage
|
|
28
|
+
|
|
29
|
+
__all__ = ["LocalClefClient"]
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
class LocalClefClient:
|
|
33
|
+
"""Drop-in ``ask()`` client backed by locally loaded Clef weights.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
settings: Validated settings; ``model`` labels the replies.
|
|
37
|
+
model_path: HF repo id (e.g. ``"Cloudflare/clef-flash"``) or a local
|
|
38
|
+
snapshot directory.
|
|
39
|
+
dtype: Torch dtype name for the backbone, ``"float16"`` for Turing GPUs
|
|
40
|
+
(T4), ``"bfloat16"`` on Ampere or newer.
|
|
41
|
+
device_map: Accelerate device map, ``"auto"`` to shard across GPUs.
|
|
42
|
+
max_memory_gib: Per-GPU memory ceiling used with ``device_map="auto"``.
|
|
43
|
+
max_length: Token bound forwarded to ``encode_record`` (default 16384).
|
|
44
|
+
"""
|
|
45
|
+
|
|
46
|
+
def __init__(
|
|
47
|
+
self,
|
|
48
|
+
settings: Settings,
|
|
49
|
+
model_path: str,
|
|
50
|
+
*,
|
|
51
|
+
dtype: str = "float16",
|
|
52
|
+
device_map: str = "auto",
|
|
53
|
+
max_memory_gib: int = 15,
|
|
54
|
+
max_length: int = 16384,
|
|
55
|
+
) -> None:
|
|
56
|
+
import torch # heavy: imported lazily so import cost stays off the API path
|
|
57
|
+
|
|
58
|
+
self.torch = torch
|
|
59
|
+
self.settings = settings
|
|
60
|
+
self.max_length = max_length
|
|
61
|
+
|
|
62
|
+
path = Path(model_path)
|
|
63
|
+
if not path.is_dir():
|
|
64
|
+
from huggingface_hub import snapshot_download
|
|
65
|
+
|
|
66
|
+
path = Path(snapshot_download(model_path))
|
|
67
|
+
|
|
68
|
+
joint_schema_dir = str(path)
|
|
69
|
+
if joint_schema_dir not in sys.path:
|
|
70
|
+
sys.path.insert(0, joint_schema_dir)
|
|
71
|
+
import json
|
|
72
|
+
|
|
73
|
+
from joint_schema_model import ClefModel, JointSchemaHead # type: ignore[import-not-found]
|
|
74
|
+
from safetensors.torch import load_file
|
|
75
|
+
from transformers import AutoProcessor, Qwen3_5ForConditionalGeneration
|
|
76
|
+
|
|
77
|
+
torch_dtype = getattr(torch, dtype)
|
|
78
|
+
backbone = Qwen3_5ForConditionalGeneration.from_pretrained(
|
|
79
|
+
path,
|
|
80
|
+
dtype=torch_dtype,
|
|
81
|
+
device_map=device_map,
|
|
82
|
+
max_memory={index: f"{max_memory_gib}GiB" for index in range(torch.cuda.device_count())},
|
|
83
|
+
)
|
|
84
|
+
backbone.config.use_cache = False
|
|
85
|
+
|
|
86
|
+
head_config = json.loads((path / "joint_head_config.json").read_text())
|
|
87
|
+
head = JointSchemaHead(**head_config)
|
|
88
|
+
head.load_state_dict(load_file(path / "joint_head.safetensors"), strict=True)
|
|
89
|
+
head = head.to(device=self._last_device(backbone), dtype=torch_dtype)
|
|
90
|
+
|
|
91
|
+
self.model = ClefModel(backbone, head).eval()
|
|
92
|
+
self.processor = AutoProcessor.from_pretrained(path)
|
|
93
|
+
|
|
94
|
+
@staticmethod
|
|
95
|
+
def _last_device(backbone: Any) -> Any:
|
|
96
|
+
"""The device holding the backbone's final layers (head must match).
|
|
97
|
+
|
|
98
|
+
``hf_device_map`` values may be ints (``0``, ``1``) or strings
|
|
99
|
+
(``"cuda:1"``); normalise both.
|
|
100
|
+
"""
|
|
101
|
+
import torch
|
|
102
|
+
|
|
103
|
+
devices = list(getattr(backbone, "hf_device_map", {}).values())
|
|
104
|
+
if not devices:
|
|
105
|
+
return torch.device("cuda" if torch.cuda.is_available() else "cpu")
|
|
106
|
+
indexes = [int(str(value).split(":")[-1]) for value in devices]
|
|
107
|
+
return torch.device("cuda", max(indexes))
|
|
108
|
+
|
|
109
|
+
def ask(
|
|
110
|
+
self,
|
|
111
|
+
state: str | dict[str, Any],
|
|
112
|
+
questions: dict[str, dict[str, Any]],
|
|
113
|
+
*,
|
|
114
|
+
images: list[str] | None = None,
|
|
115
|
+
model: str | None = None,
|
|
116
|
+
) -> ClefReply:
|
|
117
|
+
"""Answer one scoring request with a single local forward pass."""
|
|
118
|
+
from joint_schema_model import systemone # type: ignore[import-not-found]
|
|
119
|
+
|
|
120
|
+
if not questions:
|
|
121
|
+
raise ValueError("questions must contain at least one entry")
|
|
122
|
+
request = {
|
|
123
|
+
"model": model or self.settings.model,
|
|
124
|
+
"state": state,
|
|
125
|
+
"questions": questions,
|
|
126
|
+
}
|
|
127
|
+
started = time.perf_counter()
|
|
128
|
+
response = systemone(self._collate_on_head_device(), self.processor, request, max_length=self.max_length)
|
|
129
|
+
latency_ms = (time.perf_counter() - started) * 1000
|
|
130
|
+
return ClefReply(
|
|
131
|
+
model=str(response["model"]),
|
|
132
|
+
answers=response["answers"],
|
|
133
|
+
usage=TokenUsage.from_api(response.get("usage")),
|
|
134
|
+
request_id="local-weights",
|
|
135
|
+
latency_ms=latency_ms,
|
|
136
|
+
)
|
|
137
|
+
|
|
138
|
+
def _collate_on_head_device(self) -> Any:
|
|
139
|
+
"""View of the model whose first parameter lives where the head does.
|
|
140
|
+
|
|
141
|
+
``systemone`` collates the batch onto ``next(model.parameters()).device``.
|
|
142
|
+
Across a sharded load, the joint head reads ``token_ids`` from the batch
|
|
143
|
+
and indexes tensors on its own device, so the batch must be collated on
|
|
144
|
+
the head's device; accelerate's hooks move the backbone inputs to the
|
|
145
|
+
first shard as usual.
|
|
146
|
+
"""
|
|
147
|
+
inner = self.model
|
|
148
|
+
|
|
149
|
+
class _CollateOnHeadDevice:
|
|
150
|
+
def parameters(self, *args: Any, **kwargs: Any) -> Any:
|
|
151
|
+
return inner.head.parameters()
|
|
152
|
+
|
|
153
|
+
def __call__(self, batch: dict[str, Any]) -> Any:
|
|
154
|
+
return inner(batch)
|
|
155
|
+
|
|
156
|
+
def __getattr__(self, name: str) -> Any:
|
|
157
|
+
return getattr(inner, name)
|
|
158
|
+
|
|
159
|
+
return _CollateOnHeadDevice()
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
{
|
|
2
|
+
"meta": {
|
|
3
|
+
"mode": "local",
|
|
4
|
+
"scorer": "open-weights Cloudflare/clef-flash (float16, local GPU)",
|
|
5
|
+
"transport": "local torch forward pass",
|
|
6
|
+
"dataset": [
|
|
7
|
+
"/kaggle/working/clef-compactor/evals/data/compaction_suite.jsonl"
|
|
8
|
+
],
|
|
9
|
+
"relevance_threshold": 0.5,
|
|
10
|
+
"cost_model": "self-hosted open weights: no per-token API cost",
|
|
11
|
+
"price_note": "Output-token pricing is unpublished; only input tokens are ever costed."
|
|
12
|
+
},
|
|
13
|
+
"metrics": {
|
|
14
|
+
"cases": 20,
|
|
15
|
+
"chunks": 104,
|
|
16
|
+
"chunk_accuracy": 0.7115,
|
|
17
|
+
"kept_precision": 0.7705,
|
|
18
|
+
"relevant_recall": 0.746,
|
|
19
|
+
"kept_f1": 0.7581,
|
|
20
|
+
"input_tokens": 1582,
|
|
21
|
+
"output_tokens": 975,
|
|
22
|
+
"avg_saved_fraction": 0.3837,
|
|
23
|
+
"latency_ms": {
|
|
24
|
+
"p50": 1273.86,
|
|
25
|
+
"p95": 1406.64,
|
|
26
|
+
"p99": 3404.95
|
|
27
|
+
},
|
|
28
|
+
"api_input_tokens": 16229,
|
|
29
|
+
"cost_usd_per_1k_calls": 0.194748,
|
|
30
|
+
"cost_model": "self-hosted open weights: no per-token API cost"
|
|
31
|
+
},
|
|
32
|
+
"per_case": [
|
|
33
|
+
{
|
|
34
|
+
"case_id": "support-refund-001",
|
|
35
|
+
"correct": 4,
|
|
36
|
+
"total": 5,
|
|
37
|
+
"kept_relevant": 3,
|
|
38
|
+
"total_relevant": 3,
|
|
39
|
+
"kept_irrelevant": 1,
|
|
40
|
+
"input_tokens": 74,
|
|
41
|
+
"output_tokens": 58,
|
|
42
|
+
"latency_ms": 3404.9475499999744,
|
|
43
|
+
"api_input_tokens": 778
|
|
44
|
+
},
|
|
45
|
+
{
|
|
46
|
+
"case_id": "support-shipping-002",
|
|
47
|
+
"correct": 4,
|
|
48
|
+
"total": 5,
|
|
49
|
+
"kept_relevant": 2,
|
|
50
|
+
"total_relevant": 2,
|
|
51
|
+
"kept_irrelevant": 1,
|
|
52
|
+
"input_tokens": 67,
|
|
53
|
+
"output_tokens": 45,
|
|
54
|
+
"latency_ms": 1406.6402149999249,
|
|
55
|
+
"api_input_tokens": 771
|
|
56
|
+
},
|
|
57
|
+
{
|
|
58
|
+
"case_id": "tech-python-003",
|
|
59
|
+
"correct": 5,
|
|
60
|
+
"total": 6,
|
|
61
|
+
"kept_relevant": 3,
|
|
62
|
+
"total_relevant": 3,
|
|
63
|
+
"kept_irrelevant": 1,
|
|
64
|
+
"input_tokens": 101,
|
|
65
|
+
"output_tokens": 76,
|
|
66
|
+
"latency_ms": 1362.4956160000465,
|
|
67
|
+
"api_input_tokens": 935
|
|
68
|
+
},
|
|
69
|
+
{
|
|
70
|
+
"case_id": "tech-db-004",
|
|
71
|
+
"correct": 2,
|
|
72
|
+
"total": 5,
|
|
73
|
+
"kept_relevant": 1,
|
|
74
|
+
"total_relevant": 3,
|
|
75
|
+
"kept_irrelevant": 1,
|
|
76
|
+
"input_tokens": 73,
|
|
77
|
+
"output_tokens": 32,
|
|
78
|
+
"latency_ms": 1252.008700000033,
|
|
79
|
+
"api_input_tokens": 783
|
|
80
|
+
},
|
|
81
|
+
{
|
|
82
|
+
"case_id": "finance-fees-005",
|
|
83
|
+
"correct": 5,
|
|
84
|
+
"total": 5,
|
|
85
|
+
"kept_relevant": 3,
|
|
86
|
+
"total_relevant": 3,
|
|
87
|
+
"kept_irrelevant": 0,
|
|
88
|
+
"input_tokens": 68,
|
|
89
|
+
"output_tokens": 44,
|
|
90
|
+
"latency_ms": 1249.5986689999654,
|
|
91
|
+
"api_input_tokens": 779
|
|
92
|
+
},
|
|
93
|
+
{
|
|
94
|
+
"case_id": "travel-visa-006",
|
|
95
|
+
"correct": 3,
|
|
96
|
+
"total": 5,
|
|
97
|
+
"kept_relevant": 2,
|
|
98
|
+
"total_relevant": 3,
|
|
99
|
+
"kept_irrelevant": 1,
|
|
100
|
+
"input_tokens": 81,
|
|
101
|
+
"output_tokens": 57,
|
|
102
|
+
"latency_ms": 1261.9110020000335,
|
|
103
|
+
"api_input_tokens": 791
|
|
104
|
+
},
|
|
105
|
+
{
|
|
106
|
+
"case_id": "health-insurance-007",
|
|
107
|
+
"correct": 3,
|
|
108
|
+
"total": 5,
|
|
109
|
+
"kept_relevant": 3,
|
|
110
|
+
"total_relevant": 3,
|
|
111
|
+
"kept_irrelevant": 2,
|
|
112
|
+
"input_tokens": 87,
|
|
113
|
+
"output_tokens": 87,
|
|
114
|
+
"latency_ms": 1253.8356179999255,
|
|
115
|
+
"api_input_tokens": 795
|
|
116
|
+
},
|
|
117
|
+
{
|
|
118
|
+
"case_id": "devops-k8s-008",
|
|
119
|
+
"correct": 4,
|
|
120
|
+
"total": 5,
|
|
121
|
+
"kept_relevant": 2,
|
|
122
|
+
"total_relevant": 3,
|
|
123
|
+
"kept_irrelevant": 0,
|
|
124
|
+
"input_tokens": 78,
|
|
125
|
+
"output_tokens": 33,
|
|
126
|
+
"latency_ms": 831.2345519999553,
|
|
127
|
+
"api_input_tokens": 783
|
|
128
|
+
},
|
|
129
|
+
{
|
|
130
|
+
"case_id": "legal-privacy-009",
|
|
131
|
+
"correct": 4,
|
|
132
|
+
"total": 6,
|
|
133
|
+
"kept_relevant": 2,
|
|
134
|
+
"total_relevant": 4,
|
|
135
|
+
"kept_irrelevant": 0,
|
|
136
|
+
"input_tokens": 90,
|
|
137
|
+
"output_tokens": 26,
|
|
138
|
+
"latency_ms": 1363.2799339999337,
|
|
139
|
+
"api_input_tokens": 924
|
|
140
|
+
},
|
|
141
|
+
{
|
|
142
|
+
"case_id": "product-pricing-010",
|
|
143
|
+
"correct": 3,
|
|
144
|
+
"total": 5,
|
|
145
|
+
"kept_relevant": 3,
|
|
146
|
+
"total_relevant": 3,
|
|
147
|
+
"kept_irrelevant": 2,
|
|
148
|
+
"input_tokens": 69,
|
|
149
|
+
"output_tokens": 69,
|
|
150
|
+
"latency_ms": 1254.592740000021,
|
|
151
|
+
"api_input_tokens": 777
|
|
152
|
+
},
|
|
153
|
+
{
|
|
154
|
+
"case_id": "science-space-011",
|
|
155
|
+
"correct": 4,
|
|
156
|
+
"total": 5,
|
|
157
|
+
"kept_relevant": 3,
|
|
158
|
+
"total_relevant": 3,
|
|
159
|
+
"kept_irrelevant": 1,
|
|
160
|
+
"input_tokens": 80,
|
|
161
|
+
"output_tokens": 66,
|
|
162
|
+
"latency_ms": 1273.8597500000424,
|
|
163
|
+
"api_input_tokens": 781
|
|
164
|
+
},
|
|
165
|
+
{
|
|
166
|
+
"case_id": "education-math-012",
|
|
167
|
+
"correct": 2,
|
|
168
|
+
"total": 5,
|
|
169
|
+
"kept_relevant": 1,
|
|
170
|
+
"total_relevant": 4,
|
|
171
|
+
"kept_irrelevant": 0,
|
|
172
|
+
"input_tokens": 103,
|
|
173
|
+
"output_tokens": 17,
|
|
174
|
+
"latency_ms": 1268.813750999925,
|
|
175
|
+
"api_input_tokens": 808
|
|
176
|
+
},
|
|
177
|
+
{
|
|
178
|
+
"case_id": "support-account-013",
|
|
179
|
+
"correct": 3,
|
|
180
|
+
"total": 5,
|
|
181
|
+
"kept_relevant": 1,
|
|
182
|
+
"total_relevant": 3,
|
|
183
|
+
"kept_irrelevant": 0,
|
|
184
|
+
"input_tokens": 79,
|
|
185
|
+
"output_tokens": 21,
|
|
186
|
+
"latency_ms": 1265.0558409999348,
|
|
187
|
+
"api_input_tokens": 787
|
|
188
|
+
},
|
|
189
|
+
{
|
|
190
|
+
"case_id": "ops-oncall-014",
|
|
191
|
+
"correct": 6,
|
|
192
|
+
"total": 6,
|
|
193
|
+
"kept_relevant": 4,
|
|
194
|
+
"total_relevant": 4,
|
|
195
|
+
"kept_irrelevant": 0,
|
|
196
|
+
"input_tokens": 79,
|
|
197
|
+
"output_tokens": 57,
|
|
198
|
+
"latency_ms": 1367.6004610000518,
|
|
199
|
+
"api_input_tokens": 914
|
|
200
|
+
},
|
|
201
|
+
{
|
|
202
|
+
"case_id": "hr-leave-015",
|
|
203
|
+
"correct": 3,
|
|
204
|
+
"total": 5,
|
|
205
|
+
"kept_relevant": 3,
|
|
206
|
+
"total_relevant": 3,
|
|
207
|
+
"kept_irrelevant": 2,
|
|
208
|
+
"input_tokens": 76,
|
|
209
|
+
"output_tokens": 76,
|
|
210
|
+
"latency_ms": 1306.5989709999712,
|
|
211
|
+
"api_input_tokens": 784
|
|
212
|
+
},
|
|
213
|
+
{
|
|
214
|
+
"case_id": "ml-training-016",
|
|
215
|
+
"correct": 5,
|
|
216
|
+
"total": 6,
|
|
217
|
+
"kept_relevant": 3,
|
|
218
|
+
"total_relevant": 4,
|
|
219
|
+
"kept_irrelevant": 0,
|
|
220
|
+
"input_tokens": 83,
|
|
221
|
+
"output_tokens": 42,
|
|
222
|
+
"latency_ms": 1375.9791559999712,
|
|
223
|
+
"api_input_tokens": 918
|
|
224
|
+
},
|
|
225
|
+
{
|
|
226
|
+
"case_id": "ecommerce-inventory-017",
|
|
227
|
+
"correct": 5,
|
|
228
|
+
"total": 5,
|
|
229
|
+
"kept_relevant": 3,
|
|
230
|
+
"total_relevant": 3,
|
|
231
|
+
"kept_irrelevant": 0,
|
|
232
|
+
"input_tokens": 70,
|
|
233
|
+
"output_tokens": 49,
|
|
234
|
+
"latency_ms": 834.4232719999809,
|
|
235
|
+
"api_input_tokens": 778
|
|
236
|
+
},
|
|
237
|
+
{
|
|
238
|
+
"case_id": "security-auth-018",
|
|
239
|
+
"correct": 3,
|
|
240
|
+
"total": 5,
|
|
241
|
+
"kept_relevant": 2,
|
|
242
|
+
"total_relevant": 3,
|
|
243
|
+
"kept_irrelevant": 1,
|
|
244
|
+
"input_tokens": 67,
|
|
245
|
+
"output_tokens": 41,
|
|
246
|
+
"latency_ms": 1267.2506279999425,
|
|
247
|
+
"api_input_tokens": 772
|
|
248
|
+
},
|
|
249
|
+
{
|
|
250
|
+
"case_id": "nutrition-diet-019",
|
|
251
|
+
"correct": 4,
|
|
252
|
+
"total": 5,
|
|
253
|
+
"kept_relevant": 2,
|
|
254
|
+
"total_relevant": 3,
|
|
255
|
+
"kept_irrelevant": 0,
|
|
256
|
+
"input_tokens": 81,
|
|
257
|
+
"output_tokens": 44,
|
|
258
|
+
"latency_ms": 1292.3661680000578,
|
|
259
|
+
"api_input_tokens": 785
|
|
260
|
+
},
|
|
261
|
+
{
|
|
262
|
+
"case_id": "tools-git-020",
|
|
263
|
+
"correct": 2,
|
|
264
|
+
"total": 5,
|
|
265
|
+
"kept_relevant": 1,
|
|
266
|
+
"total_relevant": 3,
|
|
267
|
+
"kept_irrelevant": 1,
|
|
268
|
+
"input_tokens": 76,
|
|
269
|
+
"output_tokens": 35,
|
|
270
|
+
"latency_ms": 1304.9001580000095,
|
|
271
|
+
"api_input_tokens": 786
|
|
272
|
+
}
|
|
273
|
+
]
|
|
274
|
+
}
|
|
@@ -7,7 +7,8 @@
|
|
|
7
7
|
"C:\\Code Main\\clef-compactor\\evals\\data\\compaction_suite.jsonl"
|
|
8
8
|
],
|
|
9
9
|
"relevance_threshold": 0.5,
|
|
10
|
-
"
|
|
10
|
+
"cost_model": "input tokens x $0.24/M (Cloudflare published price)",
|
|
11
|
+
"price_note": "Output-token pricing is unpublished; only input tokens are ever costed."
|
|
11
12
|
},
|
|
12
13
|
"metrics": {
|
|
13
14
|
"cases": 20,
|
|
@@ -20,9 +21,9 @@
|
|
|
20
21
|
"output_tokens": 1044,
|
|
21
22
|
"avg_saved_fraction": 0.3401,
|
|
22
23
|
"latency_ms": {
|
|
23
|
-
"p50": 0.
|
|
24
|
-
"p95": 0.
|
|
25
|
-
"p99":
|
|
24
|
+
"p50": 0.12,
|
|
25
|
+
"p95": 0.17,
|
|
26
|
+
"p99": 156.81
|
|
26
27
|
},
|
|
27
28
|
"api_input_tokens": null,
|
|
28
29
|
"cost_usd_per_1k_calls": 0.018984,
|
|
@@ -38,7 +39,7 @@
|
|
|
38
39
|
"kept_irrelevant": 0,
|
|
39
40
|
"input_tokens": 74,
|
|
40
41
|
"output_tokens": 43,
|
|
41
|
-
"latency_ms":
|
|
42
|
+
"latency_ms": 156.8135000006805,
|
|
42
43
|
"api_input_tokens": null
|
|
43
44
|
},
|
|
44
45
|
{
|
|
@@ -50,7 +51,7 @@
|
|
|
50
51
|
"kept_irrelevant": 0,
|
|
51
52
|
"input_tokens": 67,
|
|
52
53
|
"output_tokens": 29,
|
|
53
|
-
"latency_ms": 0.
|
|
54
|
+
"latency_ms": 0.16309999955410603,
|
|
54
55
|
"api_input_tokens": null
|
|
55
56
|
},
|
|
56
57
|
{
|
|
@@ -62,7 +63,7 @@
|
|
|
62
63
|
"kept_irrelevant": 0,
|
|
63
64
|
"input_tokens": 101,
|
|
64
65
|
"output_tokens": 64,
|
|
65
|
-
"latency_ms": 0.
|
|
66
|
+
"latency_ms": 0.1578999999765074,
|
|
66
67
|
"api_input_tokens": null
|
|
67
68
|
},
|
|
68
69
|
{
|
|
@@ -74,7 +75,7 @@
|
|
|
74
75
|
"kept_irrelevant": 0,
|
|
75
76
|
"input_tokens": 73,
|
|
76
77
|
"output_tokens": 48,
|
|
77
|
-
"latency_ms": 0.
|
|
78
|
+
"latency_ms": 0.1140000003942987,
|
|
78
79
|
"api_input_tokens": null
|
|
79
80
|
},
|
|
80
81
|
{
|
|
@@ -86,7 +87,7 @@
|
|
|
86
87
|
"kept_irrelevant": 0,
|
|
87
88
|
"input_tokens": 68,
|
|
88
89
|
"output_tokens": 44,
|
|
89
|
-
"latency_ms": 0.
|
|
90
|
+
"latency_ms": 0.10680000013962854,
|
|
90
91
|
"api_input_tokens": null
|
|
91
92
|
},
|
|
92
93
|
{
|
|
@@ -98,7 +99,7 @@
|
|
|
98
99
|
"kept_irrelevant": 0,
|
|
99
100
|
"input_tokens": 81,
|
|
100
101
|
"output_tokens": 57,
|
|
101
|
-
"latency_ms": 0.
|
|
102
|
+
"latency_ms": 0.1368000002912595,
|
|
102
103
|
"api_input_tokens": null
|
|
103
104
|
},
|
|
104
105
|
{
|
|
@@ -110,7 +111,7 @@
|
|
|
110
111
|
"kept_irrelevant": 0,
|
|
111
112
|
"input_tokens": 87,
|
|
112
113
|
"output_tokens": 62,
|
|
113
|
-
"latency_ms": 0.
|
|
114
|
+
"latency_ms": 0.17209999987244373,
|
|
114
115
|
"api_input_tokens": null
|
|
115
116
|
},
|
|
116
117
|
{
|
|
@@ -122,7 +123,7 @@
|
|
|
122
123
|
"kept_irrelevant": 0,
|
|
123
124
|
"input_tokens": 78,
|
|
124
125
|
"output_tokens": 53,
|
|
125
|
-
"latency_ms": 0.
|
|
126
|
+
"latency_ms": 0.13159999980416615,
|
|
126
127
|
"api_input_tokens": null
|
|
127
128
|
},
|
|
128
129
|
{
|
|
@@ -134,7 +135,7 @@
|
|
|
134
135
|
"kept_irrelevant": 0,
|
|
135
136
|
"input_tokens": 90,
|
|
136
137
|
"output_tokens": 56,
|
|
137
|
-
"latency_ms": 0.
|
|
138
|
+
"latency_ms": 0.13320000016392441,
|
|
138
139
|
"api_input_tokens": null
|
|
139
140
|
},
|
|
140
141
|
{
|
|
@@ -146,7 +147,7 @@
|
|
|
146
147
|
"kept_irrelevant": 0,
|
|
147
148
|
"input_tokens": 69,
|
|
148
149
|
"output_tokens": 45,
|
|
149
|
-
"latency_ms": 0.
|
|
150
|
+
"latency_ms": 0.11009999980160501,
|
|
150
151
|
"api_input_tokens": null
|
|
151
152
|
},
|
|
152
153
|
{
|
|
@@ -158,7 +159,7 @@
|
|
|
158
159
|
"kept_irrelevant": 0,
|
|
159
160
|
"input_tokens": 80,
|
|
160
161
|
"output_tokens": 52,
|
|
161
|
-
"latency_ms": 0.
|
|
162
|
+
"latency_ms": 0.11050000011891825,
|
|
162
163
|
"api_input_tokens": null
|
|
163
164
|
},
|
|
164
165
|
{
|
|
@@ -170,7 +171,7 @@
|
|
|
170
171
|
"kept_irrelevant": 0,
|
|
171
172
|
"input_tokens": 103,
|
|
172
173
|
"output_tokens": 90,
|
|
173
|
-
"latency_ms": 0.
|
|
174
|
+
"latency_ms": 0.12269999933778308,
|
|
174
175
|
"api_input_tokens": null
|
|
175
176
|
},
|
|
176
177
|
{
|
|
@@ -182,7 +183,7 @@
|
|
|
182
183
|
"kept_irrelevant": 0,
|
|
183
184
|
"input_tokens": 79,
|
|
184
185
|
"output_tokens": 53,
|
|
185
|
-
"latency_ms": 0.
|
|
186
|
+
"latency_ms": 0.10440000005473848,
|
|
186
187
|
"api_input_tokens": null
|
|
187
188
|
},
|
|
188
189
|
{
|
|
@@ -194,7 +195,7 @@
|
|
|
194
195
|
"kept_irrelevant": 0,
|
|
195
196
|
"input_tokens": 79,
|
|
196
197
|
"output_tokens": 57,
|
|
197
|
-
"latency_ms": 0.
|
|
198
|
+
"latency_ms": 0.11859999995067483,
|
|
198
199
|
"api_input_tokens": null
|
|
199
200
|
},
|
|
200
201
|
{
|
|
@@ -206,7 +207,7 @@
|
|
|
206
207
|
"kept_irrelevant": 0,
|
|
207
208
|
"input_tokens": 76,
|
|
208
209
|
"output_tokens": 40,
|
|
209
|
-
"latency_ms": 0.
|
|
210
|
+
"latency_ms": 0.12630000037461286,
|
|
210
211
|
"api_input_tokens": null
|
|
211
212
|
},
|
|
212
213
|
{
|
|
@@ -218,7 +219,7 @@
|
|
|
218
219
|
"kept_irrelevant": 0,
|
|
219
220
|
"input_tokens": 83,
|
|
220
221
|
"output_tokens": 58,
|
|
221
|
-
"latency_ms": 0.
|
|
222
|
+
"latency_ms": 0.1384999995934777,
|
|
222
223
|
"api_input_tokens": null
|
|
223
224
|
},
|
|
224
225
|
{
|
|
@@ -230,7 +231,7 @@
|
|
|
230
231
|
"kept_irrelevant": 0,
|
|
231
232
|
"input_tokens": 70,
|
|
232
233
|
"output_tokens": 49,
|
|
233
|
-
"latency_ms": 0.
|
|
234
|
+
"latency_ms": 0.1022000005832524,
|
|
234
235
|
"api_input_tokens": null
|
|
235
236
|
},
|
|
236
237
|
{
|
|
@@ -242,7 +243,7 @@
|
|
|
242
243
|
"kept_irrelevant": 0,
|
|
243
244
|
"input_tokens": 67,
|
|
244
245
|
"output_tokens": 39,
|
|
245
|
-
"latency_ms": 0.
|
|
246
|
+
"latency_ms": 0.09659999977884581,
|
|
246
247
|
"api_input_tokens": null
|
|
247
248
|
},
|
|
248
249
|
{
|
|
@@ -254,7 +255,7 @@
|
|
|
254
255
|
"kept_irrelevant": 0,
|
|
255
256
|
"input_tokens": 81,
|
|
256
257
|
"output_tokens": 54,
|
|
257
|
-
"latency_ms": 0.
|
|
258
|
+
"latency_ms": 0.10689999999158317,
|
|
258
259
|
"api_input_tokens": null
|
|
259
260
|
},
|
|
260
261
|
{
|
|
@@ -266,7 +267,7 @@
|
|
|
266
267
|
"kept_irrelevant": 0,
|
|
267
268
|
"input_tokens": 76,
|
|
268
269
|
"output_tokens": 51,
|
|
269
|
-
"latency_ms": 0.
|
|
270
|
+
"latency_ms": 0.10449999990669312,
|
|
270
271
|
"api_input_tokens": null
|
|
271
272
|
}
|
|
272
273
|
]
|
|
@@ -181,7 +181,7 @@ def run_case(engine: ClefCompactor, case: EvalCase) -> CaseOutcome:
|
|
|
181
181
|
)
|
|
182
182
|
|
|
183
183
|
|
|
184
|
-
def aggregate(outcomes: Sequence[CaseOutcome]) -> dict[str, Any]:
|
|
184
|
+
def aggregate(outcomes: Sequence[CaseOutcome], cost_model: str) -> dict[str, Any]:
|
|
185
185
|
"""Reduce per-case outcomes into the reported metric block."""
|
|
186
186
|
total_chunks = sum(outcome.total for outcome in outcomes)
|
|
187
187
|
correct = sum(outcome.correct for outcome in outcomes)
|
|
@@ -219,7 +219,7 @@ def aggregate(outcomes: Sequence[CaseOutcome]) -> dict[str, Any]:
|
|
|
219
219
|
"latency_ms": {"p50": percentile_share(0.50), "p95": percentile_share(0.95), "p99": percentile_share(0.99)},
|
|
220
220
|
"api_input_tokens": api_input_tokens or None,
|
|
221
221
|
"cost_usd_per_1k_calls": round(cost_per_1k, 6),
|
|
222
|
-
"cost_model":
|
|
222
|
+
"cost_model": cost_model,
|
|
223
223
|
}
|
|
224
224
|
|
|
225
225
|
|
|
@@ -239,7 +239,12 @@ def print_laya_comparison() -> None:
|
|
|
239
239
|
def main(argv: Sequence[str] | None = None) -> int:
|
|
240
240
|
"""Entry point; returns a process exit code (1 below --min-accuracy)."""
|
|
241
241
|
parser = argparse.ArgumentParser(description=__doc__.splitlines()[0])
|
|
242
|
-
parser.add_argument("--mode", choices=("replay", "live"), default="replay")
|
|
242
|
+
parser.add_argument("--mode", choices=("replay", "live", "local"), default="replay")
|
|
243
|
+
parser.add_argument("--model-path", default="Cloudflare/clef-flash",
|
|
244
|
+
help="HF repo id or snapshot dir for --mode local")
|
|
245
|
+
parser.add_argument("--dtype", default="float16", choices=("float16", "bfloat16", "float32"),
|
|
246
|
+
help="Torch dtype for --mode local (float16 on T4)")
|
|
247
|
+
parser.add_argument("--device-map", default="auto", help="Accelerate device map for --mode local")
|
|
243
248
|
parser.add_argument("--data", default=DATA_GLOB, help="Glob of dataset .jsonl files")
|
|
244
249
|
parser.add_argument("--out", default=str(DEFAULT_OUT), help="Path of the results JSON")
|
|
245
250
|
parser.add_argument("--seed", type=int, default=20261001, help="Replay noise seed")
|
|
@@ -253,6 +258,28 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
253
258
|
engine = ClefCompactor()
|
|
254
259
|
scorer = "cloudflare-clef (live)"
|
|
255
260
|
transport = "https://api.cloudflare.com"
|
|
261
|
+
cost_model = "input tokens x $0.24/M (Cloudflare published price)"
|
|
262
|
+
elif args.mode == "local":
|
|
263
|
+
sys.path.insert(0, str(Path(__file__).resolve().parent))
|
|
264
|
+
from backends import LocalClefClient
|
|
265
|
+
|
|
266
|
+
model_label = args.model_path.split("/")[-1]
|
|
267
|
+
client = LocalClefClient(
|
|
268
|
+
Settings(account_id="local", api_token="local", model=model_label, max_retries=0),
|
|
269
|
+
args.model_path,
|
|
270
|
+
dtype=args.dtype,
|
|
271
|
+
device_map=args.device_map,
|
|
272
|
+
)
|
|
273
|
+
engine = ClefCompactor(
|
|
274
|
+
account_id="local",
|
|
275
|
+
api_token="local",
|
|
276
|
+
model=model_label,
|
|
277
|
+
relevance_threshold=args.threshold,
|
|
278
|
+
client=client, # type: ignore[arg-type]
|
|
279
|
+
)
|
|
280
|
+
scorer = f"open-weights {args.model_path} ({args.dtype}, local GPU)"
|
|
281
|
+
transport = "local torch forward pass"
|
|
282
|
+
cost_model = "self-hosted open weights: no per-token API cost"
|
|
256
283
|
else:
|
|
257
284
|
labels_by_query = {case.query: case.labels for case in cases}
|
|
258
285
|
engine = ClefCompactor(
|
|
@@ -263,9 +290,10 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
263
290
|
)
|
|
264
291
|
scorer = f"simulated-clef seed={args.seed} (pipeline validation, not model quality)"
|
|
265
292
|
transport = "in-process mock"
|
|
293
|
+
cost_model = "input tokens x $0.24/M (Cloudflare published price)"
|
|
266
294
|
|
|
267
295
|
outcomes = [run_case(engine, case) for case in cases]
|
|
268
|
-
metrics = aggregate(outcomes)
|
|
296
|
+
metrics = aggregate(outcomes, cost_model)
|
|
269
297
|
payload = {
|
|
270
298
|
"meta": {
|
|
271
299
|
"mode": args.mode,
|
|
@@ -273,7 +301,8 @@ def main(argv: Sequence[str] | None = None) -> int:
|
|
|
273
301
|
"transport": transport,
|
|
274
302
|
"dataset": sorted(glob_module.glob(args.data)),
|
|
275
303
|
"relevance_threshold": args.threshold,
|
|
276
|
-
"
|
|
304
|
+
"cost_model": cost_model,
|
|
305
|
+
"price_note": "Output-token pricing is unpublished; only input tokens are ever costed.",
|
|
277
306
|
},
|
|
278
307
|
"metrics": metrics,
|
|
279
308
|
"per_case": [vars(outcome) for outcome in outcomes],
|
|
@@ -2,6 +2,7 @@
|
|
|
2
2
|
"cells": [
|
|
3
3
|
{
|
|
4
4
|
"cell_type": "markdown",
|
|
5
|
+
"id": "7fb27b941602401d91542211134fc71a",
|
|
5
6
|
"metadata": {},
|
|
6
7
|
"source": [
|
|
7
8
|
"# clef-compactor demo\n",
|
|
@@ -12,10 +13,12 @@
|
|
|
12
13
|
{
|
|
13
14
|
"cell_type": "code",
|
|
14
15
|
"execution_count": null,
|
|
16
|
+
"id": "acae54e37e7d407bbb7b55eff062a284",
|
|
15
17
|
"metadata": {},
|
|
16
18
|
"outputs": [],
|
|
17
19
|
"source": [
|
|
18
20
|
"import httpx\n",
|
|
21
|
+
"\n",
|
|
19
22
|
"from clef_compactor import ClefCompactor\n",
|
|
20
23
|
"from clef_compactor.client import ClefClient\n",
|
|
21
24
|
"from clef_compactor.config import Settings\n",
|
|
@@ -50,6 +53,7 @@
|
|
|
50
53
|
{
|
|
51
54
|
"cell_type": "code",
|
|
52
55
|
"execution_count": null,
|
|
56
|
+
"id": "9a63283cbaf04dbcab1f6479b197f3a8",
|
|
53
57
|
"metadata": {},
|
|
54
58
|
"outputs": [],
|
|
55
59
|
"source": [
|
|
@@ -71,6 +75,7 @@
|
|
|
71
75
|
{
|
|
72
76
|
"cell_type": "code",
|
|
73
77
|
"execution_count": null,
|
|
78
|
+
"id": "8dd0d8092fe74a7c96281538738b07e2",
|
|
74
79
|
"metadata": {},
|
|
75
80
|
"outputs": [],
|
|
76
81
|
"source": [
|
|
@@ -83,6 +88,7 @@
|
|
|
83
88
|
},
|
|
84
89
|
{
|
|
85
90
|
"cell_type": "markdown",
|
|
91
|
+
"id": "72eea5119410473aa328ad9291626812",
|
|
86
92
|
"metadata": {},
|
|
87
93
|
"source": [
|
|
88
94
|
"## The compacted context\n",
|
|
@@ -93,6 +99,7 @@
|
|
|
93
99
|
{
|
|
94
100
|
"cell_type": "code",
|
|
95
101
|
"execution_count": null,
|
|
102
|
+
"id": "8edb47106e1a46a883d545849b8ab81b",
|
|
96
103
|
"metadata": {},
|
|
97
104
|
"outputs": [],
|
|
98
105
|
"source": [
|
|
@@ -103,6 +110,7 @@
|
|
|
103
110
|
{
|
|
104
111
|
"cell_type": "code",
|
|
105
112
|
"execution_count": null,
|
|
113
|
+
"id": "10185d26023b46108eb7d9f57d49d2b3",
|
|
106
114
|
"metadata": {},
|
|
107
115
|
"outputs": [],
|
|
108
116
|
"source": [
|
|
@@ -116,6 +124,7 @@
|
|
|
116
124
|
},
|
|
117
125
|
{
|
|
118
126
|
"cell_type": "markdown",
|
|
127
|
+
"id": "8763a12b2bbd4a93a75aff182afb95dc",
|
|
119
128
|
"metadata": {},
|
|
120
129
|
"source": [
|
|
121
130
|
"## Going live\n",
|
|
@@ -130,8 +139,15 @@
|
|
|
130
139
|
}
|
|
131
140
|
],
|
|
132
141
|
"metadata": {
|
|
133
|
-
"kernelspec": {
|
|
134
|
-
|
|
142
|
+
"kernelspec": {
|
|
143
|
+
"display_name": "Python 3",
|
|
144
|
+
"language": "python",
|
|
145
|
+
"name": "python3"
|
|
146
|
+
},
|
|
147
|
+
"language_info": {
|
|
148
|
+
"name": "python",
|
|
149
|
+
"version": "3.10"
|
|
150
|
+
}
|
|
135
151
|
},
|
|
136
152
|
"nbformat": 4,
|
|
137
153
|
"nbformat_minor": 5
|
|
@@ -0,0 +1,13 @@
|
|
|
1
|
+
{
|
|
2
|
+
"id": "gjusev/clef-compactor-evals",
|
|
3
|
+
"title": "clef-compactor-evals",
|
|
4
|
+
"code_file": "script.py",
|
|
5
|
+
"language": "python",
|
|
6
|
+
"kernel_type": "script",
|
|
7
|
+
"is_private": true,
|
|
8
|
+
"enable_gpu": true,
|
|
9
|
+
"enable_internet": true,
|
|
10
|
+
"dataset_sources": [],
|
|
11
|
+
"kernel_sources": [],
|
|
12
|
+
"model_sources": []
|
|
13
|
+
}
|
|
@@ -0,0 +1,47 @@
|
|
|
1
|
+
"""Kaggle kernel: measure clef-compactor against the open-weights Clef model.
|
|
2
|
+
|
|
3
|
+
Follows the verified laya-evals kernel pattern (github.com/Gjusev/laya-evals):
|
|
4
|
+
clone the public repo, pip install it, run the measurement, print the verdict.
|
|
5
|
+
Outputs land in /kaggle/working/ and come back with `kaggle kernels output`.
|
|
6
|
+
|
|
7
|
+
Hardware: Kaggle T4 x2 (2 x 16 GB). We run Cloudflare/clef-flash (9B) in
|
|
8
|
+
float16 sharded across both GPUs with device_map="auto"; the 27B model does
|
|
9
|
+
not fit on this hardware. Credentials are never needed and never embedded:
|
|
10
|
+
the local backend loads weights from Hugging Face.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import subprocess
|
|
14
|
+
import sys
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
def run(cmd: str) -> None:
|
|
18
|
+
print(f"$ {cmd}", flush=True)
|
|
19
|
+
subprocess.run(cmd, shell=True, check=True)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
# 0. GPU sanity: Kaggle can silently boot without the GPU you asked for.
|
|
23
|
+
print("=== GPU check ===", flush=True)
|
|
24
|
+
run("nvidia-smi")
|
|
25
|
+
|
|
26
|
+
# 1. Repo + dependencies. transformers 5.10.2 is the version the Clef release
|
|
27
|
+
# was tested with; Qwen3.5 support needs the 5.x line.
|
|
28
|
+
run("git clone --depth 1 https://github.com/Gjusev/clef-compactor.git")
|
|
29
|
+
run(f"{sys.executable} -m pip install -q --no-input ./clef-compactor")
|
|
30
|
+
run(f"{sys.executable} -m pip install -q --no-input 'transformers==5.10.2' accelerate safetensors")
|
|
31
|
+
|
|
32
|
+
run(f"{sys.executable} -c 'import torch, transformers; "
|
|
33
|
+
"print(torch.__version__, torch.cuda.is_available(), transformers.__version__)'")
|
|
34
|
+
|
|
35
|
+
# 2. The measurement: full pipeline (batching, ranking, budget) against the
|
|
36
|
+
# locally loaded clef-flash weights, scored on the committed gold dataset.
|
|
37
|
+
run(
|
|
38
|
+
f"{sys.executable} clef-compactor/evals/run_eval.py "
|
|
39
|
+
"--mode local --model-path Cloudflare/clef-flash --dtype float16 "
|
|
40
|
+
"--out /kaggle/working/results-local.json --compare-laya"
|
|
41
|
+
)
|
|
42
|
+
|
|
43
|
+
# 3. Optional stretch: the 27B model in float16 needs ~54 GB and does not fit
|
|
44
|
+
# on 2 x T4; skipped on purpose. If Kaggle ever offers bigger GPUs, run:
|
|
45
|
+
# run_eval.py --mode local --model-path Cloudflare/clef
|
|
46
|
+
|
|
47
|
+
print("KERNEL COMPLETE", flush=True)
|
|
@@ -8,7 +8,9 @@ from conftest import envelope, noul_answer
|
|
|
8
8
|
|
|
9
9
|
|
|
10
10
|
def test_package_exposes_version() -> None:
|
|
11
|
-
|
|
11
|
+
from importlib.metadata import version
|
|
12
|
+
|
|
13
|
+
assert clef_compactor.__version__ == version("clef-compactor")
|
|
12
14
|
|
|
13
15
|
|
|
14
16
|
def test_public_api_surface() -> None:
|
|
File without changes
|
{clef_compactor-0.2.0 → clef_compactor-0.2.1}/.github/actions/clef-evals/check_regression.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|