clef-compactor 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,20 @@
1
+ name: Publish to PyPI
2
+
3
+ on:
4
+ release:
5
+ types: [published]
6
+ workflow_dispatch:
7
+
8
+ jobs:
9
+ publish:
10
+ runs-on: ubuntu-latest
11
+ environment: pypi
12
+ permissions:
13
+ id-token: write
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: astral-sh/setup-uv@v5
17
+ with:
18
+ python-version: "3.12"
19
+ - run: uv build
20
+ - run: uv publish
@@ -0,0 +1,59 @@
1
+ Metadata-Version: 2.5
2
+ Name: clef-compactor
3
+ Version: 0.1.0
4
+ Summary: Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise.
5
+ Author: Youssef Ouhaghi Ahmian
6
+ License: Apache-2.0
7
+ Keywords: clef,cloudflare,compaction,context,llm,rag,tokens
8
+ Requires-Python: >=3.10
9
+ Requires-Dist: httpx>=0.25
10
+ Requires-Dist: tiktoken>=0.7
11
+ Provides-Extra: dev
12
+ Requires-Dist: pytest>=7; extra == 'dev'
13
+ Description-Content-Type: text/markdown
14
+
15
+ # clef-compactor
16
+
17
+ Query-aware RAG context compaction using Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/).
18
+
19
+ Your retrieval pipeline returns 12 chunks. Your LLM reads all of them. You pay for all of them. clef-compactor scores each document against the query using Clef's 64k context window and cuts the noise.
20
+
21
+ **Delete, do not rewrite.** Documents are kept verbatim or removed with an auditable reason. Never summarized.
22
+
23
+ ## Quick start
24
+
25
+ ```bash
26
+ pip install clef-compactor
27
+ export CLEF_ACCOUNT_ID=your_id
28
+ export CLEF_API_TOKEN=your_token
29
+ ```
30
+
31
+ ```python
32
+ from clef_compactor import ClefCompactor
33
+
34
+ compactor = ClefCompactor()
35
+ result = compactor.compact(
36
+ question="What is the refund policy?",
37
+ documents=retrieved_chunks,
38
+ token_budget=1000,
39
+ )
40
+ # result.kept: documents that survived
41
+ # result.dropped: [{doc, score, reason}] for the ones cut
42
+ # result.total_output_tokens < result.total_input_tokens
43
+ ```
44
+
45
+ ## CLI
46
+
47
+ ```bash
48
+ clef-compact -q "refund policy" -d "doc1..." "doc2..." --budget 1000
49
+ ```
50
+
51
+ ## Why Clef for compaction?
52
+
53
+ Clef's 64k context window means you can score more documents in a single pass than with laya (32k). The quality is higher (BFCL 98.76 vs laya's 38.13), which means fewer false positives when scoring document relevance.
54
+
55
+ For sub-10ms local compaction, see [laya-compactor](https://github.com/Gjusev/laya-compactor).
56
+
57
+ ## License
58
+
59
+ Apache 2.0
@@ -0,0 +1,45 @@
1
+ # clef-compactor
2
+
3
+ Query-aware RAG context compaction using Cloudflare's [Clef](https://blog.cloudflare.com/clef-decision-models/).
4
+
5
+ Your retrieval pipeline returns 12 chunks. Your LLM reads all of them. You pay for all of them. clef-compactor scores each document against the query using Clef's 64k context window and cuts the noise.
6
+
7
+ **Delete, do not rewrite.** Documents are kept verbatim or removed with an auditable reason. Never summarized.
8
+
9
+ ## Quick start
10
+
11
+ ```bash
12
+ pip install clef-compactor
13
+ export CLEF_ACCOUNT_ID=your_id
14
+ export CLEF_API_TOKEN=your_token
15
+ ```
16
+
17
+ ```python
18
+ from clef_compactor import ClefCompactor
19
+
20
+ compactor = ClefCompactor()
21
+ result = compactor.compact(
22
+ question="What is the refund policy?",
23
+ documents=retrieved_chunks,
24
+ token_budget=1000,
25
+ )
26
+ # result.kept: documents that survived
27
+ # result.dropped: [{doc, score, reason}] for the ones cut
28
+ # result.total_output_tokens < result.total_input_tokens
29
+ ```
30
+
31
+ ## CLI
32
+
33
+ ```bash
34
+ clef-compact -q "refund policy" -d "doc1..." "doc2..." --budget 1000
35
+ ```
36
+
37
+ ## Why Clef for compaction?
38
+
39
+ Clef's 64k context window means you can score more documents in a single pass than with laya (32k). The quality is higher (BFCL 98.76 vs laya's 38.13), which means fewer false positives when scoring document relevance.
40
+
41
+ For sub-10ms local compaction, see [laya-compactor](https://github.com/Gjusev/laya-compactor).
42
+
43
+ ## License
44
+
45
+ Apache 2.0
@@ -0,0 +1,23 @@
1
+ [project]
2
+ name = "clef-compactor"
3
+ version = "0.1.0"
4
+ description = "Query-aware RAG context compaction using Cloudflare's Clef. Keep the evidence, cut the noise."
5
+ readme = "README.md"
6
+ license = { text = "Apache-2.0" }
7
+ requires-python = ">=3.10"
8
+ authors = [{ name = "Youssef Ouhaghi Ahmian" }]
9
+ keywords = ["rag", "context", "compaction", "clef", "cloudflare", "llm", "tokens"]
10
+ dependencies = ["httpx>=0.25", "tiktoken>=0.7"]
11
+
12
+ [project.scripts]
13
+ clef-compact = "clef_compactor.cli:main"
14
+
15
+ [project.optional-dependencies]
16
+ dev = ["pytest>=7"]
17
+
18
+ [build-system]
19
+ requires = ["hatchling"]
20
+ build-backend = "hatchling.build"
21
+
22
+ [tool.hatch.build.targets.wheel]
23
+ packages = ["src/clef_compactor"]
@@ -0,0 +1,114 @@
1
+ """clef-compactor: Query-aware RAG context compaction using Cloudflare's Clef."""
2
+
3
+ import os
4
+ from dataclasses import dataclass, field
5
+ from typing import Optional
6
+
7
+ import httpx
8
+ import tiktoken
9
+
10
+
11
+ @dataclass
12
+ class CompactResult:
13
+ """Result of compacting a retrieval batch."""
14
+ kept: list[dict] # documents that survived, with their scores
15
+ dropped: list[dict] # documents cut, with score and reason
16
+ total_input_tokens: int
17
+ total_output_tokens: int
18
+ token_budget: int
19
+ raw_response: dict = field(default_factory=dict, repr=False)
20
+
21
+
22
+ @dataclass
23
+ class ClefCompactor:
24
+ """Score and compact RAG retrieval batches using Cloudflare's Clef."""
25
+
26
+ account_id: str = field(default_factory=lambda: os.environ.get("CLEF_ACCOUNT_ID", ""))
27
+ api_token: str = field(default_factory=lambda: os.environ.get("CLEF_API_TOKEN", ""))
28
+ model: str = "@cf/cloudflare/clef"
29
+ base_url: str = "https://api.cloudflare.com/client/v4"
30
+ timeout: float = 60.0
31
+ encoding_model: str = "cl100k_base" # for token counting
32
+
33
+ def __post_init__(self):
34
+ if not self.account_id:
35
+ raise ValueError("Set CLEF_ACCOUNT_ID env var or pass account_id")
36
+ if not self.api_token:
37
+ raise ValueError("Set CLEF_API_TOKEN env var or pass api_token")
38
+
39
+ @property
40
+ def _endpoint(self) -> str:
41
+ return f"{self.base_url}/accounts/{self.account_id}/ai/run/{self.model}"
42
+
43
+ def _headers(self) -> dict:
44
+ return {"Authorization": f"Bearer {self.api_token}", "Content-Type": "application/json"}
45
+
46
+ def _count_tokens(self, text: str) -> int:
47
+ enc = tiktoken.get_encoding(self.encoding_model)
48
+ return len(enc.encode(text))
49
+
50
+ def compact(
51
+ self,
52
+ question: str,
53
+ documents: list[str],
54
+ token_budget: int = 1000,
55
+ ) -> CompactResult:
56
+ """Score each document against the question and keep the best within budget."""
57
+ if not documents:
58
+ return CompactResult(kept=[], dropped=[], total_input_tokens=0,
59
+ total_output_tokens=0, token_budget=token_budget)
60
+
61
+ state = f"Question: {question}\n\nDocuments to score:\n"
62
+ questions = {}
63
+ for i, doc in enumerate(documents):
64
+ snippet = doc[:300] # truncate for context efficiency
65
+ state += f"\n[Doc {i+1}]: {snippet}...\n"
66
+ questions[f"relevance_{i+1}"] = {
67
+ "type": "score",
68
+ "context": f"How relevant is document {i+1} to answering the question? "
69
+ "essential = directly answers the question. "
70
+ "relevant = provides supporting information. "
71
+ "background = tangentially related. "
72
+ "irrelevant = not useful.",
73
+ "levels": ["irrelevant", "background", "relevant", "essential"],
74
+ }
75
+
76
+ payload = {"state": state, "questions": questions}
77
+ resp = httpx.post(
78
+ self._endpoint, json=payload, headers=self._headers(), timeout=self.timeout,
79
+ )
80
+ resp.raise_for_status()
81
+ data = resp.json().get("result", {})
82
+
83
+ score_map = {"irrelevant": 0, "background": 1, "relevant": 2, "essential": 3}
84
+ scored = []
85
+ for i, doc in enumerate(documents):
86
+ raw_score = data.get(f"relevance_{i+1}", "irrelevant")
87
+ score = score_map.get(raw_score, 0)
88
+ tokens = self._count_tokens(doc)
89
+ scored.append({"doc": doc, "score": score, "raw": raw_score, "tokens": tokens})
90
+
91
+ # Sort by score descending, keep within budget
92
+ scored.sort(key=lambda x: x["score"], reverse=True)
93
+ kept, dropped = [], []
94
+ running_tokens = 0
95
+ for item in scored:
96
+ if item["score"] <= 0:
97
+ dropped.append({**item, "reason": "irrelevant"})
98
+ elif running_tokens + item["tokens"] <= token_budget:
99
+ kept.append(item)
100
+ running_tokens += item["tokens"]
101
+ else:
102
+ dropped.append({**item, "reason": "budget_exhausted"})
103
+
104
+ total_input = sum(item["tokens"] for item in scored)
105
+ return CompactResult(
106
+ kept=kept, dropped=dropped,
107
+ total_input_tokens=total_input,
108
+ total_output_tokens=running_tokens,
109
+ token_budget=token_budget,
110
+ )
111
+
112
+
113
+ __version__ = "0.1.0"
114
+ __all__ = ["ClefCompactor", "CompactResult"]
@@ -0,0 +1,37 @@
1
+ """CLI for clef-compactor."""
2
+
3
+ import argparse
4
+ import json
5
+
6
+ from . import ClefCompactor
7
+
8
+
9
+ def main():
10
+ p = argparse.ArgumentParser(prog="clef-compact", description="Compact RAG context using Clef")
11
+ p.add_argument("--question", "-q", required=True)
12
+ p.add_argument("--docs", "-d", nargs="+", required=True, help="Document texts")
13
+ p.add_argument("--budget", "-b", type=int, default=1000)
14
+ p.add_argument("--json", action="store_true")
15
+ args = p.parse_args()
16
+
17
+ compactor = ClefCompactor()
18
+ result = compactor.compact(args.question, args.docs, token_budget=args.budget)
19
+
20
+ if args.json:
21
+ print(json.dumps({
22
+ "kept": len(result.kept),
23
+ "dropped": len(result.dropped),
24
+ "input_tokens": result.total_input_tokens,
25
+ "output_tokens": result.total_output_tokens,
26
+ "saved_pct": round(100 * (1 - result.total_output_tokens / max(result.total_input_tokens, 1)), 1),
27
+ }, indent=2))
28
+ else:
29
+ print(f"kept: {len(result.kept)} docs")
30
+ print(f"dropped: {len(result.dropped)} docs")
31
+ print(f"input tokens: {result.total_input_tokens}")
32
+ print(f"output tokens: {result.total_output_tokens}")
33
+ print(f"saved: {100 * (1 - result.total_output_tokens / max(result.total_input_tokens, 1)):.1f}%")
34
+
35
+
36
+ if __name__ == "__main__":
37
+ main()