edgenote 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- edgenote-0.1.0/.gitignore +33 -0
- edgenote-0.1.0/LICENSE +18 -0
- edgenote-0.1.0/PKG-INFO +275 -0
- edgenote-0.1.0/README.md +245 -0
- edgenote-0.1.0/benchmarks/lost_in_middle.py +492 -0
- edgenote-0.1.0/benchmarks/requirements.txt +2 -0
- edgenote-0.1.0/dist_new/edgenote-0.3.0-py3-none-any.whl +0 -0
- edgenote-0.1.0/dist_new/edgenote-0.3.0.tar.gz +0 -0
- edgenote-0.1.0/dist_v1/edgenote-1.0.0-py3-none-any.whl +0 -0
- edgenote-0.1.0/dist_v1/edgenote-1.0.0.tar.gz +0 -0
- edgenote-0.1.0/pyproject.toml +63 -0
- edgenote-0.1.0/src/edgenote/__init__.py +113 -0
- edgenote-0.1.0/src/edgenote/__main__.py +144 -0
- edgenote-0.1.0/src/edgenote/chunker.py +565 -0
- edgenote-0.1.0/src/edgenote/compressor.py +220 -0
- edgenote-0.1.0/src/edgenote/downloader.py +278 -0
- edgenote-0.1.0/src/edgenote/integrations.py +240 -0
- edgenote-0.1.0/src/edgenote/reranker.py +516 -0
- edgenote-0.1.0/src/edgenote/session.py +402 -0
- edgenote-0.1.0/src/edgenote/style.py +45 -0
- edgenote-0.1.0/src/edgenote/tokenizers.py +350 -0
- edgenote-0.1.0/src/edgenote/tokens.py +17 -0
- edgenote-0.1.0/tests/test_chunker.py +365 -0
- edgenote-0.1.0/tests/test_compressor.py +166 -0
- edgenote-0.1.0/tests/test_downloader.py +368 -0
- edgenote-0.1.0/tests/test_integrations.py +160 -0
- edgenote-0.1.0/tests/test_large_prompt.py +212 -0
- edgenote-0.1.0/tests/test_reranker.py +295 -0
- edgenote-0.1.0/tests/test_session.py +240 -0
- edgenote-0.1.0/tests/test_tokenizers.py +247 -0
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Python artifacts
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*$py.class
|
|
5
|
+
*.so
|
|
6
|
+
.Python
|
|
7
|
+
build/
|
|
8
|
+
develop-eggs/
|
|
9
|
+
dist/
|
|
10
|
+
downloads/
|
|
11
|
+
eggs/
|
|
12
|
+
.eggs/
|
|
13
|
+
lib/
|
|
14
|
+
lib64/
|
|
15
|
+
parts/
|
|
16
|
+
sdist/
|
|
17
|
+
var/
|
|
18
|
+
wheels/
|
|
19
|
+
*.egg-info/
|
|
20
|
+
.installed.cfg
|
|
21
|
+
*.egg
|
|
22
|
+
|
|
23
|
+
# Testing
|
|
24
|
+
.pytest_cache/
|
|
25
|
+
.coverage
|
|
26
|
+
htmlcov/
|
|
27
|
+
|
|
28
|
+
# Environments
|
|
29
|
+
.env
|
|
30
|
+
.venv
|
|
31
|
+
env/
|
|
32
|
+
venv/
|
|
33
|
+
ENV/
|
edgenote-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
Proprietary License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Md Tareq Shah Alam. All rights reserved.
|
|
4
|
+
|
|
5
|
+
This software, source code, and associated documentation files (the "Software")
|
|
6
|
+
are proprietary and confidential.
|
|
7
|
+
|
|
8
|
+
Unauthorized copying, modification, reverse engineering, redistribution,
|
|
9
|
+
sublicensing, or creation of derivative works of this Software, in whole or
|
|
10
|
+
in part, via any medium, is strictly prohibited without prior explicit
|
|
11
|
+
written permission from the copyright owner (Md Tareq Shah Alam).
|
|
12
|
+
|
|
13
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
14
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
15
|
+
FITNESS FOR A PARTICULAR PURPOSE, AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
16
|
+
COPYRIGHT HOLDER BE LIABLE FOR ANY CLAIM, DAMAGES, OR OTHER LIABILITY, WHETHER
|
|
17
|
+
IN AN ACTION OF CONTRACT, TORT, OR OTHERWISE, ARISING FROM, OUT OF, OR IN
|
|
18
|
+
CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE SOFTWARE.
|
edgenote-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,275 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: edgenote
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Place important notes at the edges of your LLM prompts to fight the lost-in-the-middle problem.
|
|
5
|
+
Project-URL: LinkedIn, https://www.linkedin.com/in/md-tareq-shah-alam/
|
|
6
|
+
Author-email: Md Tareq Shah Alam <tareqshah.027@gmail.com>
|
|
7
|
+
License: Proprietary
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: bm25,chunking,context-window,llm,lost-in-the-middle,prompt-engineering,prompt-optimization,rag,reranking,token-counting
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Developers
|
|
12
|
+
Classifier: License :: Other/Proprietary License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
18
|
+
Requires-Python: >=3.10
|
|
19
|
+
Requires-Dist: regex>=2023.0
|
|
20
|
+
Requires-Dist: sentence-transformers>=3.0
|
|
21
|
+
Requires-Dist: tiktoken>=0.7
|
|
22
|
+
Requires-Dist: transformers>=4.40
|
|
23
|
+
Provides-Extra: all
|
|
24
|
+
Provides-Extra: chunker
|
|
25
|
+
Provides-Extra: dev
|
|
26
|
+
Requires-Dist: pytest>=8; extra == 'dev'
|
|
27
|
+
Provides-Extra: regex
|
|
28
|
+
Provides-Extra: rerank
|
|
29
|
+
Description-Content-Type: text/markdown
|
|
30
|
+
|
|
31
|
+
# edgenote
|
|
32
|
+
|
|
33
|
+
Prompt edge-pinning framework for LLMs. Mitigates the "Lost in the Middle" attention degradation problem by anchoring critical constraints and high-scoring context at prompt boundaries.
|
|
34
|
+
|
|
35
|
+
---
|
|
36
|
+
|
|
37
|
+
## The Problem
|
|
38
|
+
|
|
39
|
+
Transformer architectures exhibit U-shaped attention distributions ([Liu et al., 2023](https://arxiv.org/abs/2307.03172)). Information positioned at the **beginning (primacy)** and **end (recency)** of the context window is retrieved reliably, while critical data in the **middle** suffers severe recall degradation.
|
|
40
|
+
|
|
41
|
+
`edgenote` structures prompts defensively:
|
|
42
|
+
1. **Edge Pinning:** Anchors high-priority instructions, constraints, and facts to both the head and tail.
|
|
43
|
+
2. **U-Shaped Interleaving:** Organises ranked documents so that top-scoring chunks occupy the high-attention edges while lower-ranked data remains in the middle.
|
|
44
|
+
3. **Exact Token Budgeting:** Measures real token lengths and evicts or compresses low-priority middle context when limits are exceeded.
|
|
45
|
+
|
|
46
|
+
---
|
|
47
|
+
|
|
48
|
+
## Empirical Benchmark
|
|
49
|
+
|
|
50
|
+
Multi-document needle-in-a-haystack retrieval evaluation across context positions:
|
|
51
|
+
|
|
52
|
+
| Position in Context | Baseline Prompt | `edgenote` | Overhead |
|
|
53
|
+
| :--- | :---: | :---: | :---: |
|
|
54
|
+
| **Start** (0.0) | CORRECT | CORRECT | +44 tokens |
|
|
55
|
+
| **25%** | WRONG | CORRECT | +44 tokens |
|
|
56
|
+
| **Middle** (0.5) | WRONG | CORRECT | +44 tokens |
|
|
57
|
+
| **75%** | WRONG | CORRECT | +44 tokens |
|
|
58
|
+
| **End** (1.0) | CORRECT | CORRECT | +44 tokens |
|
|
59
|
+
|
|
60
|
+
---
|
|
61
|
+
|
|
62
|
+
## Installation
|
|
63
|
+
|
|
64
|
+
```bash
|
|
65
|
+
pip install edgenote
|
|
66
|
+
```
|
|
67
|
+
|
|
68
|
+
`edgenote` ships fully equipped with native tokenisation and neural ranking runtimes:
|
|
69
|
+
|
|
70
|
+
| Subsystem | Engine | Supported Models |
|
|
71
|
+
| :--- | :--- | :--- |
|
|
72
|
+
| **OpenAI Tokeniser** | `tiktoken` | `gpt-4o`, `o1`, `o3-mini`, `cl100k_base`, `o200k_base` |
|
|
73
|
+
| **Open-Weights Tokeniser** | `transformers` | Llama 3, Mistral, Gemma, Qwen, DeepSeek |
|
|
74
|
+
| **Neural Re-ranking** | `sentence-transformers` | Cross-encoder relevance scoring & bi-encoder embeddings |
|
|
75
|
+
| **Pattern Engine** | `regex` | Unicode-aware BPE tokenisation |
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Quick Start
|
|
80
|
+
|
|
81
|
+
```python
|
|
82
|
+
from edgenote import Session
|
|
83
|
+
|
|
84
|
+
s = Session()
|
|
85
|
+
s.pin("Hardware budget is strictly capped at $185,000.", label="Constraint")
|
|
86
|
+
s.add("Proposal A: Liquid cooling loop overhaul ($240,000)", relevance=0.7)
|
|
87
|
+
s.add("Proposal B: High-density compute cluster ($180,000)", relevance=0.92)
|
|
88
|
+
|
|
89
|
+
result = s.render("Which proposal satisfies our constraints?")
|
|
90
|
+
print(result.text)
|
|
91
|
+
|
|
92
|
+
messages = result.to_messages()
|
|
93
|
+
```
|
|
94
|
+
|
|
95
|
+
---
|
|
96
|
+
|
|
97
|
+
## Core Capabilities
|
|
98
|
+
|
|
99
|
+
### 1. Model-Accurate Token Counting
|
|
100
|
+
|
|
101
|
+
Pass any standard model identifier to bind the exact tokenizer backend:
|
|
102
|
+
|
|
103
|
+
```python
|
|
104
|
+
from edgenote import Session
|
|
105
|
+
|
|
106
|
+
s = Session(token_counter="gpt-4o")
|
|
107
|
+
s = Session(token_counter="meta-llama/Meta-Llama-3-8B-Instruct")
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Custom counting callables are also accepted:
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
import tiktoken
|
|
114
|
+
from edgenote import Session
|
|
115
|
+
|
|
116
|
+
enc = tiktoken.encoding_for_model("gpt-4o")
|
|
117
|
+
s = Session(token_counter=enc.encode)
|
|
118
|
+
```
|
|
119
|
+
|
|
120
|
+
### 2. Automated Re-ranking
|
|
121
|
+
|
|
122
|
+
Automatically score and reorder retrieved documents at render time without manual relevance labels:
|
|
123
|
+
|
|
124
|
+
```python
|
|
125
|
+
from edgenote import Session
|
|
126
|
+
|
|
127
|
+
s = Session(reranker="cross-encoder")
|
|
128
|
+
s.add("Quarterly financial filings")
|
|
129
|
+
s.add("Personnel roster and team structure")
|
|
130
|
+
s.add("Enterprise procurement guidelines")
|
|
131
|
+
|
|
132
|
+
result = s.render("What were the fourth-quarter operating expenditures?")
|
|
133
|
+
```
|
|
134
|
+
|
|
135
|
+
Pure-Python zero-overhead rankers are also available:
|
|
136
|
+
|
|
137
|
+
```python
|
|
138
|
+
from edgenote import Session
|
|
139
|
+
from edgenote.reranker import BM25Reranker, TFIDFReranker
|
|
140
|
+
|
|
141
|
+
s = Session(reranker=BM25Reranker())
|
|
142
|
+
s = Session(reranker=TFIDFReranker())
|
|
143
|
+
```
|
|
144
|
+
|
|
145
|
+
### 3. Context Compression
|
|
146
|
+
|
|
147
|
+
Summarize low-priority context when token limits are reached instead of dropping text:
|
|
148
|
+
|
|
149
|
+
```python
|
|
150
|
+
from edgenote import Session
|
|
151
|
+
from groq import Groq
|
|
152
|
+
|
|
153
|
+
client = Groq()
|
|
154
|
+
|
|
155
|
+
def summarize(text: str, max_tokens: int) -> str:
|
|
156
|
+
response = client.chat.completions.create(
|
|
157
|
+
model="llama3-8b-8192",
|
|
158
|
+
messages=[
|
|
159
|
+
{"role": "system", "content": f"Summarize in {max_tokens} tokens or fewer."},
|
|
160
|
+
{"role": "user", "content": text},
|
|
161
|
+
],
|
|
162
|
+
max_tokens=max_tokens,
|
|
163
|
+
)
|
|
164
|
+
return response.choices[0].message.content
|
|
165
|
+
|
|
166
|
+
s = Session(compressor=summarize)
|
|
167
|
+
s.add("Long technical specification...")
|
|
168
|
+
result = s.render("Summarize system latency limits", budget=2048)
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
### 4. Text Chunking
|
|
172
|
+
|
|
173
|
+
Split large documents before ingestion:
|
|
174
|
+
|
|
175
|
+
```python
|
|
176
|
+
from edgenote.chunker import StructuralChunker, SemanticChunker
|
|
177
|
+
|
|
178
|
+
chunker = StructuralChunker(max_tokens=512, overlap_tokens=64)
|
|
179
|
+
chunks = chunker.chunk(document_text)
|
|
180
|
+
|
|
181
|
+
semantic_chunker = SemanticChunker(threshold=0.5)
|
|
182
|
+
semantic_chunks = semantic_chunker.chunk(document_text)
|
|
183
|
+
```
|
|
184
|
+
|
|
185
|
+
### 5. Framework Integrations
|
|
186
|
+
|
|
187
|
+
#### LangChain
|
|
188
|
+
|
|
189
|
+
```python
|
|
190
|
+
from edgenote.integrations import from_langchain
|
|
191
|
+
|
|
192
|
+
result = from_langchain(
|
|
193
|
+
docs,
|
|
194
|
+
query="What is the operating budget?",
|
|
195
|
+
pins=["Strictly cite sources using document headers."],
|
|
196
|
+
budget=4000,
|
|
197
|
+
)
|
|
198
|
+
messages = result.to_messages()
|
|
199
|
+
```
|
|
200
|
+
|
|
201
|
+
#### LlamaIndex
|
|
202
|
+
|
|
203
|
+
```python
|
|
204
|
+
from edgenote.integrations import from_llamaindex
|
|
205
|
+
|
|
206
|
+
result = from_llamaindex(
|
|
207
|
+
nodes,
|
|
208
|
+
query="Synthesize quarterly performance metrics.",
|
|
209
|
+
pins=["Output format: Markdown table."],
|
|
210
|
+
)
|
|
211
|
+
```
|
|
212
|
+
|
|
213
|
+
#### Generic Dictionaries
|
|
214
|
+
|
|
215
|
+
```python
|
|
216
|
+
from edgenote.integrations import from_dicts
|
|
217
|
+
|
|
218
|
+
docs = [
|
|
219
|
+
{"text": "Annual recurring revenue reached $12M", "relevance": 0.95, "source": "Finance"},
|
|
220
|
+
{"text": "Total headcount expanded to 120", "relevance": 0.40, "source": "HR"},
|
|
221
|
+
]
|
|
222
|
+
result = from_dicts(docs, query="Provide financial summary")
|
|
223
|
+
```
|
|
224
|
+
|
|
225
|
+
---
|
|
226
|
+
|
|
227
|
+
## CLI & Model Cache
|
|
228
|
+
|
|
229
|
+
`edgenote` provides a command-line interface for verification and pre-caching neural weights:
|
|
230
|
+
|
|
231
|
+
```bash
|
|
232
|
+
# Verify installation and active backends
|
|
233
|
+
edgenote
|
|
234
|
+
|
|
235
|
+
# Run live prompt assembly demonstration
|
|
236
|
+
edgenote --demo
|
|
237
|
+
|
|
238
|
+
# Pre-cache weights for air-gapped environments
|
|
239
|
+
edgenote download bge-small
|
|
240
|
+
```
|
|
241
|
+
|
|
242
|
+
---
|
|
243
|
+
|
|
244
|
+
## API Reference
|
|
245
|
+
|
|
246
|
+
### `Session` Parameters
|
|
247
|
+
|
|
248
|
+
| Parameter | Type | Default | Description |
|
|
249
|
+
| :--- | :--- | :--- | :--- |
|
|
250
|
+
| `style` | `Style` | `None` | Layout and section formatting configuration |
|
|
251
|
+
| `token_counter` | `Union[str, TokenCounter, Callable]` | `None` | Model name or tokenizer callable for token measurement |
|
|
252
|
+
| `orderer` | `Union[str, Callable]` | `"edges"` | Context layout strategy (`"edges"`, `"none"`, or custom) |
|
|
253
|
+
| `reranker` | `Union[str, Reranker, Callable]` | `None` | Document scoring engine (`"cross-encoder"`, `"embedding"`, etc.) |
|
|
254
|
+
| `compressor` | `Union[str, Compressor, Callable]` | `None` | Eviction or summarization strategy for budget fitting |
|
|
255
|
+
| `unranked_relevance`| `float` | `0.0` | Default score assigned to unranked chunks |
|
|
256
|
+
|
|
257
|
+
### `Rendered` Attributes
|
|
258
|
+
|
|
259
|
+
| Attribute | Type | Description |
|
|
260
|
+
| :--- | :--- | :--- |
|
|
261
|
+
| `text` / `user_text` | `str` | Fully assembled user prompt |
|
|
262
|
+
| `system` | `str` | System prompt text |
|
|
263
|
+
| `tokens` | `int` | Total measured token consumption |
|
|
264
|
+
| `pin_overhead` | `int` | Token cost incurred by tail-edge pin repetition |
|
|
265
|
+
| `dropped` | `List[str]` | Chunk texts evicted or summarized to fit budget |
|
|
266
|
+
| `to_messages()` | `List[Dict]` | Returns OpenAI-compatible payload `[{"role": ..., "content": ...}]` |
|
|
267
|
+
|
|
268
|
+
---
|
|
269
|
+
|
|
270
|
+
## Author
|
|
271
|
+
|
|
272
|
+
[**Md Tareq Shah Alam**](https://tareqshahalam.is-a.dev/)
|
|
273
|
+
- LinkedIn: [Md Tareq Shah Alam](https://www.linkedin.com/in/md-tareq-shah-alam/)
|
|
274
|
+
- Email: [tareqshah.027@gmail.com](mailto:tareqshah.027@gmail.com)
|
|
275
|
+
- License: Proprietary (All Rights Reserved)
|
edgenote-0.1.0/README.md
ADDED
|
@@ -0,0 +1,245 @@
|
|
|
1
|
+
# edgenote
|
|
2
|
+
|
|
3
|
+
Prompt edge-pinning framework for LLMs. Mitigates the "Lost in the Middle" attention degradation problem by anchoring critical constraints and high-scoring context at prompt boundaries.
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## The Problem
|
|
8
|
+
|
|
9
|
+
Transformer architectures exhibit U-shaped attention distributions ([Liu et al., 2023](https://arxiv.org/abs/2307.03172)). Information positioned at the **beginning (primacy)** and **end (recency)** of the context window is retrieved reliably, while critical data in the **middle** suffers severe recall degradation.
|
|
10
|
+
|
|
11
|
+
`edgenote` structures prompts defensively:
|
|
12
|
+
1. **Edge Pinning:** Anchors high-priority instructions, constraints, and facts to both the head and tail.
|
|
13
|
+
2. **U-Shaped Interleaving:** Organises ranked documents so that top-scoring chunks occupy the high-attention edges while lower-ranked data remains in the middle.
|
|
14
|
+
3. **Exact Token Budgeting:** Measures real token lengths and evicts or compresses low-priority middle context when limits are exceeded.
|
|
15
|
+
|
|
16
|
+
---
|
|
17
|
+
|
|
18
|
+
## Empirical Benchmark
|
|
19
|
+
|
|
20
|
+
Multi-document needle-in-a-haystack retrieval evaluation across context positions:
|
|
21
|
+
|
|
22
|
+
| Position in Context | Baseline Prompt | `edgenote` | Overhead |
|
|
23
|
+
| :--- | :---: | :---: | :---: |
|
|
24
|
+
| **Start** (0.0) | CORRECT | CORRECT | +44 tokens |
|
|
25
|
+
| **25%** | WRONG | CORRECT | +44 tokens |
|
|
26
|
+
| **Middle** (0.5) | WRONG | CORRECT | +44 tokens |
|
|
27
|
+
| **75%** | WRONG | CORRECT | +44 tokens |
|
|
28
|
+
| **End** (1.0) | CORRECT | CORRECT | +44 tokens |
|
|
29
|
+
|
|
30
|
+
---
|
|
31
|
+
|
|
32
|
+
## Installation
|
|
33
|
+
|
|
34
|
+
```bash
|
|
35
|
+
pip install edgenote
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
`edgenote` ships fully equipped with native tokenisation and neural ranking runtimes:
|
|
39
|
+
|
|
40
|
+
| Subsystem | Engine | Supported Models |
|
|
41
|
+
| :--- | :--- | :--- |
|
|
42
|
+
| **OpenAI Tokeniser** | `tiktoken` | `gpt-4o`, `o1`, `o3-mini`, `cl100k_base`, `o200k_base` |
|
|
43
|
+
| **Open-Weights Tokeniser** | `transformers` | Llama 3, Mistral, Gemma, Qwen, DeepSeek |
|
|
44
|
+
| **Neural Re-ranking** | `sentence-transformers` | Cross-encoder relevance scoring & bi-encoder embeddings |
|
|
45
|
+
| **Pattern Engine** | `regex` | Unicode-aware BPE tokenisation |
|
|
46
|
+
|
|
47
|
+
---
|
|
48
|
+
|
|
49
|
+
## Quick Start
|
|
50
|
+
|
|
51
|
+
```python
|
|
52
|
+
from edgenote import Session
|
|
53
|
+
|
|
54
|
+
s = Session()
|
|
55
|
+
s.pin("Hardware budget is strictly capped at $185,000.", label="Constraint")
|
|
56
|
+
s.add("Proposal A: Liquid cooling loop overhaul ($240,000)", relevance=0.7)
|
|
57
|
+
s.add("Proposal B: High-density compute cluster ($180,000)", relevance=0.92)
|
|
58
|
+
|
|
59
|
+
result = s.render("Which proposal satisfies our constraints?")
|
|
60
|
+
print(result.text)
|
|
61
|
+
|
|
62
|
+
messages = result.to_messages()
|
|
63
|
+
```
|
|
64
|
+
|
|
65
|
+
---
|
|
66
|
+
|
|
67
|
+
## Core Capabilities
|
|
68
|
+
|
|
69
|
+
### 1. Model-Accurate Token Counting
|
|
70
|
+
|
|
71
|
+
Pass any standard model identifier to bind the exact tokenizer backend:
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from edgenote import Session
|
|
75
|
+
|
|
76
|
+
s = Session(token_counter="gpt-4o")
|
|
77
|
+
s = Session(token_counter="meta-llama/Meta-Llama-3-8B-Instruct")
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
Custom counting callables are also accepted:
|
|
81
|
+
|
|
82
|
+
```python
|
|
83
|
+
import tiktoken
|
|
84
|
+
from edgenote import Session
|
|
85
|
+
|
|
86
|
+
enc = tiktoken.encoding_for_model("gpt-4o")
|
|
87
|
+
s = Session(token_counter=enc.encode)
|
|
88
|
+
```
|
|
89
|
+
|
|
90
|
+
### 2. Automated Re-ranking
|
|
91
|
+
|
|
92
|
+
Automatically score and reorder retrieved documents at render time without manual relevance labels:
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
from edgenote import Session
|
|
96
|
+
|
|
97
|
+
s = Session(reranker="cross-encoder")
|
|
98
|
+
s.add("Quarterly financial filings")
|
|
99
|
+
s.add("Personnel roster and team structure")
|
|
100
|
+
s.add("Enterprise procurement guidelines")
|
|
101
|
+
|
|
102
|
+
result = s.render("What were the fourth-quarter operating expenditures?")
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Pure-Python zero-overhead rankers are also available:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
from edgenote import Session
|
|
109
|
+
from edgenote.reranker import BM25Reranker, TFIDFReranker
|
|
110
|
+
|
|
111
|
+
s = Session(reranker=BM25Reranker())
|
|
112
|
+
s = Session(reranker=TFIDFReranker())
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
### 3. Context Compression
|
|
116
|
+
|
|
117
|
+
Summarize low-priority context when token limits are reached instead of dropping text:
|
|
118
|
+
|
|
119
|
+
```python
|
|
120
|
+
from edgenote import Session
|
|
121
|
+
from groq import Groq
|
|
122
|
+
|
|
123
|
+
client = Groq()
|
|
124
|
+
|
|
125
|
+
def summarize(text: str, max_tokens: int) -> str:
|
|
126
|
+
response = client.chat.completions.create(
|
|
127
|
+
model="llama3-8b-8192",
|
|
128
|
+
messages=[
|
|
129
|
+
{"role": "system", "content": f"Summarize in {max_tokens} tokens or fewer."},
|
|
130
|
+
{"role": "user", "content": text},
|
|
131
|
+
],
|
|
132
|
+
max_tokens=max_tokens,
|
|
133
|
+
)
|
|
134
|
+
return response.choices[0].message.content
|
|
135
|
+
|
|
136
|
+
s = Session(compressor=summarize)
|
|
137
|
+
s.add("Long technical specification...")
|
|
138
|
+
result = s.render("Summarize system latency limits", budget=2048)
|
|
139
|
+
```
|
|
140
|
+
|
|
141
|
+
### 4. Text Chunking
|
|
142
|
+
|
|
143
|
+
Split large documents before ingestion:
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
from edgenote.chunker import StructuralChunker, SemanticChunker
|
|
147
|
+
|
|
148
|
+
chunker = StructuralChunker(max_tokens=512, overlap_tokens=64)
|
|
149
|
+
chunks = chunker.chunk(document_text)
|
|
150
|
+
|
|
151
|
+
semantic_chunker = SemanticChunker(threshold=0.5)
|
|
152
|
+
semantic_chunks = semantic_chunker.chunk(document_text)
|
|
153
|
+
```
|
|
154
|
+
|
|
155
|
+
### 5. Framework Integrations
|
|
156
|
+
|
|
157
|
+
#### LangChain
|
|
158
|
+
|
|
159
|
+
```python
|
|
160
|
+
from edgenote.integrations import from_langchain
|
|
161
|
+
|
|
162
|
+
result = from_langchain(
|
|
163
|
+
docs,
|
|
164
|
+
query="What is the operating budget?",
|
|
165
|
+
pins=["Strictly cite sources using document headers."],
|
|
166
|
+
budget=4000,
|
|
167
|
+
)
|
|
168
|
+
messages = result.to_messages()
|
|
169
|
+
```
|
|
170
|
+
|
|
171
|
+
#### LlamaIndex
|
|
172
|
+
|
|
173
|
+
```python
|
|
174
|
+
from edgenote.integrations import from_llamaindex
|
|
175
|
+
|
|
176
|
+
result = from_llamaindex(
|
|
177
|
+
nodes,
|
|
178
|
+
query="Synthesize quarterly performance metrics.",
|
|
179
|
+
pins=["Output format: Markdown table."],
|
|
180
|
+
)
|
|
181
|
+
```
|
|
182
|
+
|
|
183
|
+
#### Generic Dictionaries
|
|
184
|
+
|
|
185
|
+
```python
|
|
186
|
+
from edgenote.integrations import from_dicts
|
|
187
|
+
|
|
188
|
+
docs = [
|
|
189
|
+
{"text": "Annual recurring revenue reached $12M", "relevance": 0.95, "source": "Finance"},
|
|
190
|
+
{"text": "Total headcount expanded to 120", "relevance": 0.40, "source": "HR"},
|
|
191
|
+
]
|
|
192
|
+
result = from_dicts(docs, query="Provide financial summary")
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
---
|
|
196
|
+
|
|
197
|
+
## CLI & Model Cache
|
|
198
|
+
|
|
199
|
+
`edgenote` provides a command-line interface for verification and pre-caching neural weights:
|
|
200
|
+
|
|
201
|
+
```bash
|
|
202
|
+
# Verify installation and active backends
|
|
203
|
+
edgenote
|
|
204
|
+
|
|
205
|
+
# Run live prompt assembly demonstration
|
|
206
|
+
edgenote --demo
|
|
207
|
+
|
|
208
|
+
# Pre-cache weights for air-gapped environments
|
|
209
|
+
edgenote download bge-small
|
|
210
|
+
```
|
|
211
|
+
|
|
212
|
+
---
|
|
213
|
+
|
|
214
|
+
## API Reference
|
|
215
|
+
|
|
216
|
+
### `Session` Parameters
|
|
217
|
+
|
|
218
|
+
| Parameter | Type | Default | Description |
|
|
219
|
+
| :--- | :--- | :--- | :--- |
|
|
220
|
+
| `style` | `Style` | `None` | Layout and section formatting configuration |
|
|
221
|
+
| `token_counter` | `Union[str, TokenCounter, Callable]` | `None` | Model name or tokenizer callable for token measurement |
|
|
222
|
+
| `orderer` | `Union[str, Callable]` | `"edges"` | Context layout strategy (`"edges"`, `"none"`, or custom) |
|
|
223
|
+
| `reranker` | `Union[str, Reranker, Callable]` | `None` | Document scoring engine (`"cross-encoder"`, `"embedding"`, etc.) |
|
|
224
|
+
| `compressor` | `Union[str, Compressor, Callable]` | `None` | Eviction or summarization strategy for budget fitting |
|
|
225
|
+
| `unranked_relevance`| `float` | `0.0` | Default score assigned to unranked chunks |
|
|
226
|
+
|
|
227
|
+
### `Rendered` Attributes
|
|
228
|
+
|
|
229
|
+
| Attribute | Type | Description |
|
|
230
|
+
| :--- | :--- | :--- |
|
|
231
|
+
| `text` / `user_text` | `str` | Fully assembled user prompt |
|
|
232
|
+
| `system` | `str` | System prompt text |
|
|
233
|
+
| `tokens` | `int` | Total measured token consumption |
|
|
234
|
+
| `pin_overhead` | `int` | Token cost incurred by tail-edge pin repetition |
|
|
235
|
+
| `dropped` | `List[str]` | Chunk texts evicted or summarized to fit budget |
|
|
236
|
+
| `to_messages()` | `List[Dict]` | Returns OpenAI-compatible payload `[{"role": ..., "content": ...}]` |
|
|
237
|
+
|
|
238
|
+
---
|
|
239
|
+
|
|
240
|
+
## Author
|
|
241
|
+
|
|
242
|
+
[**Md Tareq Shah Alam**](https://tareqshahalam.is-a.dev/)
|
|
243
|
+
- LinkedIn: [Md Tareq Shah Alam](https://www.linkedin.com/in/md-tareq-shah-alam/)
|
|
244
|
+
- Email: [tareqshah.027@gmail.com](mailto:tareqshah.027@gmail.com)
|
|
245
|
+
- License: Proprietary (All Rights Reserved)
|