clef-compactor 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
name: Publish to PyPI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
release:
|
|
5
|
+
types: [published]
|
|
6
|
+
workflow_dispatch:
|
|
7
|
+
|
|
8
|
+
jobs:
|
|
9
|
+
publish:
|
|
10
|
+
runs-on: ubuntu-latest
|
|
11
|
+
environment: pypi
|
|
12
|
+
permissions:
|
|
13
|
+
id-token: write
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: astral-sh/setup-uv@v5
|
|
17
|
+
with:
|
|
18
|
+
python-version: "3.12"
|
|
19
|
+
- run: uv build
|
|
20
|
+
- run: uv publish
|
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: clef-compactor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise.
|
|
5
|
+
Author: Youssef Ouhaghi Ahmian
|
|
6
|
+
License: Apache-2.0
|
|
7
|
+
Keywords: clef,cloudflare,compaction,context,llm,rag,tokens
|
|
8
|
+
Requires-Python: >=3.10
|
|
9
|
+
Requires-Dist: httpx>=0.25
|
|
10
|
+
Requires-Dist: tiktoken>=0.7
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pytest>=7; extra == 'dev'
|
|
13
|
+
Description-Content-Type: text/markdown
|
|
14
|
+
|
|
15
|
+
# clef-compactor
|
|
16
|
+
|
|
17
|
+
Query-aware RAG context compaction using Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/).
|
|
18
|
+
|
|
19
|
+
Your retrieval pipeline returns 12 chunks. Your LLM reads all of them. You pay for all of them. clef-compactor scores each document against the query using Clef's 64k context window and cuts the noise.
|
|
20
|
+
|
|
21
|
+
**Delete, do not rewrite.** Documents are kept verbatim or removed with an auditable reason. Never summarized.
|
|
22
|
+
|
|
23
|
+
## Quick start
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install clef-compactor
|
|
27
|
+
export CLEF_ACCOUNT_ID=your_id
|
|
28
|
+
export CLEF_API_TOKEN=your_token
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
```python
|
|
32
|
+
from clef_compactor import ClefCompactor
|
|
33
|
+
|
|
34
|
+
compactor = ClefCompactor()
|
|
35
|
+
result = compactor.compact(
|
|
36
|
+
question="What is the refund policy?",
|
|
37
|
+
documents=retrieved_chunks,
|
|
38
|
+
token_budget=1000,
|
|
39
|
+
)
|
|
40
|
+
# result.kept: documents that survived
|
|
41
|
+
# result.dropped: [{doc, score, reason}] for the ones cut
|
|
42
|
+
# result.total_output_tokens < result.total_input_tokens
|
|
43
|
+
```
|
|
44
|
+
|
|
45
|
+
## CLI
|
|
46
|
+
|
|
47
|
+
```bash
|
|
48
|
+
clef-compact -q "refund policy" -d "doc1..." "doc2..." --budget 1000
|
|
49
|
+
```
|
|
50
|
+
|
|
51
|
+
## Why Clef for compaction?
|
|
52
|
+
|
|
53
|
+
Clef's 64k context window means you can score more documents in a single pass than with laya (32k). The quality is higher (BFCL 98.76 vs laya's 38.13), which means fewer false positives when scoring document relevance.
|
|
54
|
+
|
|
55
|
+
For sub-10ms local compaction, see [laya-compactor](https://github.com/Gjusev/laya-compactor).
|
|
56
|
+
|
|
57
|
+
## License
|
|
58
|
+
|
|
59
|
+
Apache 2.0
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
# clef-compactor
|
|
2
|
+
|
|
3
|
+
Query-aware RAG context compaction using Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/).
|
|
4
|
+
|
|
5
|
+
Your retrieval pipeline returns 12 chunks. Your LLM reads all of them. You pay for all of them. clef-compactor scores each document against the query using Clef's 64k context window and cuts the noise.
|
|
6
|
+
|
|
7
|
+
**Delete, do not rewrite.** Documents are kept verbatim or removed with an auditable reason. Never summarized.
|
|
8
|
+
|
|
9
|
+
## Quick start
|
|
10
|
+
|
|
11
|
+
```bash
|
|
12
|
+
pip install clef-compactor
|
|
13
|
+
export CLEF_ACCOUNT_ID=your_id
|
|
14
|
+
export CLEF_API_TOKEN=your_token
|
|
15
|
+
```
|
|
16
|
+
|
|
17
|
+
```python
|
|
18
|
+
from clef_compactor import ClefCompactor
|
|
19
|
+
|
|
20
|
+
compactor = ClefCompactor()
|
|
21
|
+
result = compactor.compact(
|
|
22
|
+
question="What is the refund policy?",
|
|
23
|
+
documents=retrieved_chunks,
|
|
24
|
+
token_budget=1000,
|
|
25
|
+
)
|
|
26
|
+
# result.kept: documents that survived
|
|
27
|
+
# result.dropped: [{doc, score, reason}] for the ones cut
|
|
28
|
+
# result.total_output_tokens < result.total_input_tokens
|
|
29
|
+
```
|
|
30
|
+
|
|
31
|
+
## CLI
|
|
32
|
+
|
|
33
|
+
```bash
|
|
34
|
+
clef-compact -q "refund policy" -d "doc1..." "doc2..." --budget 1000
|
|
35
|
+
```
|
|
36
|
+
|
|
37
|
+
## Why Clef for compaction?
|
|
38
|
+
|
|
39
|
+
Clef's 64k context window means you can score more documents in a single pass than with laya (32k). The quality is higher (BFCL 98.76 vs laya's 38.13), which means fewer false positives when scoring document relevance.
|
|
40
|
+
|
|
41
|
+
For sub-10ms local compaction, see [laya-compactor](https://github.com/Gjusev/laya-compactor).
|
|
42
|
+
|
|
43
|
+
## License
|
|
44
|
+
|
|
45
|
+
Apache 2.0
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
[project]
|
|
2
|
+
name = "clef-compactor"
|
|
3
|
+
version = "0.1.0"
|
|
4
|
+
description = "Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise."
|
|
5
|
+
readme = "README.md"
|
|
6
|
+
license = { text = "Apache-2.0" }
|
|
7
|
+
requires-python = ">=3.10"
|
|
8
|
+
authors = [{ name = "Youssef Ouhaghi Ahmian" }]
|
|
9
|
+
keywords = ["rag", "context", "compaction", "clef", "cloudflare", "llm", "tokens"]
|
|
10
|
+
dependencies = ["httpx>=0.25", "tiktoken>=0.7"]
|
|
11
|
+
|
|
12
|
+
[project.scripts]
|
|
13
|
+
clef-compact = "clef_compactor.cli:main"
|
|
14
|
+
|
|
15
|
+
[project.optional-dependencies]
|
|
16
|
+
dev = ["pytest>=7"]
|
|
17
|
+
|
|
18
|
+
[build-system]
|
|
19
|
+
requires = ["hatchling"]
|
|
20
|
+
build-backend = "hatchling.build"
|
|
21
|
+
|
|
22
|
+
[tool.hatch.build.targets.wheel]
|
|
23
|
+
packages = ["src/clef_compactor"]
|
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
"""clef-compactor: Query-aware RAG context compaction using Cloudflare's Clef."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from dataclasses import dataclass, field
|
|
5
|
+
from typing import Optional
|
|
6
|
+
|
|
7
|
+
import httpx
|
|
8
|
+
import tiktoken
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
@dataclass
|
|
12
|
+
class CompactResult:
|
|
13
|
+
"""Result of compacting a retrieval batch."""
|
|
14
|
+
kept: list[dict] # documents that survived, with their scores
|
|
15
|
+
dropped: list[dict] # documents cut, with score and reason
|
|
16
|
+
total_input_tokens: int
|
|
17
|
+
total_output_tokens: int
|
|
18
|
+
token_budget: int
|
|
19
|
+
raw_response: dict = field(default_factory=dict, repr=False)
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass
|
|
23
|
+
class ClefCompactor:
|
|
24
|
+
"""Score and compact RAG retrieval batches using Cloudflare's Clef."""
|
|
25
|
+
|
|
26
|
+
account_id: str = field(default_factory=lambda: os.environ.get("CLEF_ACCOUNT_ID", ""))
|
|
27
|
+
api_token: str = field(default_factory=lambda: os.environ.get("CLEF_API_TOKEN", ""))
|
|
28
|
+
model: str = "@cf/cloudflare/clef"
|
|
29
|
+
base_url: str = "https://api.cloudflare.com/client/v4"
|
|
30
|
+
timeout: float = 60.0
|
|
31
|
+
encoding_model: str = "cl100k_base" # for token counting
|
|
32
|
+
|
|
33
|
+
def __post_init__(self):
|
|
34
|
+
if not self.account_id:
|
|
35
|
+
raise ValueError("Set CLEF_ACCOUNT_ID env var or pass account_id")
|
|
36
|
+
if not self.api_token:
|
|
37
|
+
raise ValueError("Set CLEF_API_TOKEN env var or pass api_token")
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def _endpoint(self) -> str:
|
|
41
|
+
return f"{self.base_url}/accounts/{self.account_id}/ai/run/{self.model}"
|
|
42
|
+
|
|
43
|
+
def _headers(self) -> dict:
|
|
44
|
+
return {"Authorization": f"Bearer {self.api_token}", "Content-Type": "application/json"}
|
|
45
|
+
|
|
46
|
+
def _count_tokens(self, text: str) -> int:
|
|
47
|
+
enc = tiktoken.get_encoding(self.encoding_model)
|
|
48
|
+
return len(enc.encode(text))
|
|
49
|
+
|
|
50
|
+
def compact(
|
|
51
|
+
self,
|
|
52
|
+
question: str,
|
|
53
|
+
documents: list[str],
|
|
54
|
+
token_budget: int = 1000,
|
|
55
|
+
) -> CompactResult:
|
|
56
|
+
"""Score each document against the question and keep the best within budget."""
|
|
57
|
+
if not documents:
|
|
58
|
+
return CompactResult(kept=[], dropped=[], total_input_tokens=0,
|
|
59
|
+
total_output_tokens=0, token_budget=token_budget)
|
|
60
|
+
|
|
61
|
+
state = f"Question: {question}\n\nDocuments to score:\n"
|
|
62
|
+
questions = {}
|
|
63
|
+
for i, doc in enumerate(documents):
|
|
64
|
+
snippet = doc[:300] # truncate for context efficiency
|
|
65
|
+
state += f"\n[Doc {i+1}]: {snippet}...\n"
|
|
66
|
+
questions[f"relevance_{i+1}"] = {
|
|
67
|
+
"type": "score",
|
|
68
|
+
"context": f"How relevant is document {i+1} to answering the question? "
|
|
69
|
+
"essential = directly answers the question. "
|
|
70
|
+
"relevant = provides supporting information. "
|
|
71
|
+
"background = tangentially related. "
|
|
72
|
+
"irrelevant = not useful.",
|
|
73
|
+
"levels": ["irrelevant", "background", "relevant", "essential"],
|
|
74
|
+
}
|
|
75
|
+
|
|
76
|
+
payload = {"state": state, "questions": questions}
|
|
77
|
+
resp = httpx.post(
|
|
78
|
+
self._endpoint, json=payload, headers=self._headers(), timeout=self.timeout,
|
|
79
|
+
)
|
|
80
|
+
resp.raise_for_status()
|
|
81
|
+
data = resp.json().get("result", {})
|
|
82
|
+
|
|
83
|
+
score_map = {"irrelevant": 0, "background": 1, "relevant": 2, "essential": 3}
|
|
84
|
+
scored = []
|
|
85
|
+
for i, doc in enumerate(documents):
|
|
86
|
+
raw_score = data.get(f"relevance_{i+1}", "irrelevant")
|
|
87
|
+
score = score_map.get(raw_score, 0)
|
|
88
|
+
tokens = self._count_tokens(doc)
|
|
89
|
+
scored.append({"doc": doc, "score": score, "raw": raw_score, "tokens": tokens})
|
|
90
|
+
|
|
91
|
+
# Sort by score descending, keep within budget
|
|
92
|
+
scored.sort(key=lambda x: x["score"], reverse=True)
|
|
93
|
+
kept, dropped = [], []
|
|
94
|
+
running_tokens = 0
|
|
95
|
+
for item in scored:
|
|
96
|
+
if item["score"] <= 0:
|
|
97
|
+
dropped.append({**item, "reason": "irrelevant"})
|
|
98
|
+
elif running_tokens + item["tokens"] <= token_budget:
|
|
99
|
+
kept.append(item)
|
|
100
|
+
running_tokens += item["tokens"]
|
|
101
|
+
else:
|
|
102
|
+
dropped.append({**item, "reason": "budget_exhausted"})
|
|
103
|
+
|
|
104
|
+
total_input = sum(item["tokens"] for item in scored)
|
|
105
|
+
return CompactResult(
|
|
106
|
+
kept=kept, dropped=dropped,
|
|
107
|
+
total_input_tokens=total_input,
|
|
108
|
+
total_output_tokens=running_tokens,
|
|
109
|
+
token_budget=token_budget,
|
|
110
|
+
)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
__version__ = "0.1.0"
|
|
114
|
+
__all__ = ["ClefCompactor", "CompactResult"]
|
|
@@ -0,0 +1,37 @@
|
|
|
1
|
+
"""CLI for clef-compactor."""
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
import json
|
|
5
|
+
|
|
6
|
+
from . import ClefCompactor
|
|
7
|
+
|
|
8
|
+
|
|
9
|
+
def main():
|
|
10
|
+
p = argparse.ArgumentParser(prog="clef-compact", description="Compact RAG context using Clef")
|
|
11
|
+
p.add_argument("--question", "-q", required=True)
|
|
12
|
+
p.add_argument("--docs", "-d", nargs="+", required=True, help="Document texts")
|
|
13
|
+
p.add_argument("--budget", "-b", type=int, default=1000)
|
|
14
|
+
p.add_argument("--json", action="store_true")
|
|
15
|
+
args = p.parse_args()
|
|
16
|
+
|
|
17
|
+
compactor = ClefCompactor()
|
|
18
|
+
result = compactor.compact(args.question, args.docs, token_budget=args.budget)
|
|
19
|
+
|
|
20
|
+
if args.json:
|
|
21
|
+
print(json.dumps({
|
|
22
|
+
"kept": len(result.kept),
|
|
23
|
+
"dropped": len(result.dropped),
|
|
24
|
+
"input_tokens": result.total_input_tokens,
|
|
25
|
+
"output_tokens": result.total_output_tokens,
|
|
26
|
+
"saved_pct": round(100 * (1 - result.total_output_tokens / max(result.total_input_tokens, 1)), 1),
|
|
27
|
+
}, indent=2))
|
|
28
|
+
else:
|
|
29
|
+
print(f"kept: {len(result.kept)} docs")
|
|
30
|
+
print(f"dropped: {len(result.dropped)} docs")
|
|
31
|
+
print(f"input tokens: {result.total_input_tokens}")
|
|
32
|
+
print(f"output tokens: {result.total_output_tokens}")
|
|
33
|
+
print(f"saved: {100 * (1 - result.total_output_tokens / max(result.total_input_tokens, 1)):.1f}%")
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
if __name__ == "__main__":
|
|
37
|
+
main()
|