flense 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- flense-0.1.0/.gitignore +36 -0
- flense-0.1.0/LICENSE +6 -0
- flense-0.1.0/PKG-INFO +148 -0
- flense-0.1.0/README.md +125 -0
- flense-0.1.0/pyproject.toml +42 -0
- flense-0.1.0/spec.md +382 -0
- flense-0.1.0/src/flense/__init__.py +1 -0
- flense-0.1.0/src/flense/app.py +54 -0
- flense-0.1.0/src/flense/cli.py +124 -0
- flense-0.1.0/src/flense/code_writer.py +230 -0
- flense-0.1.0/src/flense/compression/__init__.py +244 -0
- flense-0.1.0/src/flense/compression/classifier.py +136 -0
- flense-0.1.0/src/flense/compression/ctags.py +102 -0
- flense-0.1.0/src/flense/compression/estimator.py +50 -0
- flense-0.1.0/src/flense/compression/grammar.py +108 -0
- flense-0.1.0/src/flense/compression/language.py +112 -0
- flense-0.1.0/src/flense/compression/regex.py +58 -0
- flense-0.1.0/src/flense/compression/treesitter.py +214 -0
- flense-0.1.0/src/flense/compression/types.py +65 -0
- flense-0.1.0/src/flense/config.py +118 -0
- flense-0.1.0/src/flense/daemon.py +82 -0
- flense-0.1.0/src/flense/pricing.py +83 -0
- flense-0.1.0/src/flense/providers/__init__.py +31 -0
- flense-0.1.0/src/flense/providers/anthropic.py +23 -0
- flense-0.1.0/src/flense/providers/base.py +66 -0
- flense-0.1.0/src/flense/providers/openai.py +28 -0
- flense-0.1.0/src/flense/proxy.py +198 -0
- flense-0.1.0/src/flense/server.py +31 -0
- flense-0.1.0/src/flense/telemetry.py +169 -0
- flense-0.1.0/src/flense/tui.py +156 -0
- flense-0.1.0/tests/conftest.py +14 -0
- flense-0.1.0/tests/test_classifier.py +110 -0
- flense-0.1.0/tests/test_code_writer.py +361 -0
- flense-0.1.0/tests/test_config.py +75 -0
- flense-0.1.0/tests/test_config_telemetry.py +60 -0
- flense-0.1.0/tests/test_estimator.py +65 -0
- flense-0.1.0/tests/test_language.py +58 -0
- flense-0.1.0/tests/test_pipeline.py +103 -0
- flense-0.1.0/tests/test_pricing.py +60 -0
- flense-0.1.0/tests/test_proxy_integration.py +198 -0
- flense-0.1.0/tests/test_regex.py +62 -0
- flense-0.1.0/tests/test_telemetry.py +180 -0
- flense-0.1.0/tests/test_treesitter.py +85 -0
- flense-0.1.0/work-handover.md +169 -0
flense-0.1.0/.gitignore
ADDED
|
@@ -0,0 +1,36 @@
|
|
|
1
|
+
# Python
|
|
2
|
+
__pycache__/
|
|
3
|
+
*.py[cod]
|
|
4
|
+
*.so
|
|
5
|
+
*.egg-info/
|
|
6
|
+
dist/
|
|
7
|
+
build/
|
|
8
|
+
.eggs/
|
|
9
|
+
|
|
10
|
+
# Virtual environments
|
|
11
|
+
.venv/
|
|
12
|
+
venv/
|
|
13
|
+
env/
|
|
14
|
+
|
|
15
|
+
# Environment variables
|
|
16
|
+
.env
|
|
17
|
+
.env.*
|
|
18
|
+
|
|
19
|
+
# Testing
|
|
20
|
+
.pytest_cache/
|
|
21
|
+
.coverage
|
|
22
|
+
htmlcov/
|
|
23
|
+
|
|
24
|
+
# Type checking
|
|
25
|
+
.mypy_cache/
|
|
26
|
+
.ruff_cache/
|
|
27
|
+
|
|
28
|
+
# IDE
|
|
29
|
+
.vscode/
|
|
30
|
+
.idea/
|
|
31
|
+
|
|
32
|
+
# macOS
|
|
33
|
+
.DS_Store
|
|
34
|
+
|
|
35
|
+
# Guppy runtime
|
|
36
|
+
*.log
|
flense-0.1.0/LICENSE
ADDED
|
@@ -0,0 +1,6 @@
|
|
|
1
|
+
Copyright (c) 2026 Anand. All rights reserved.
|
|
2
|
+
|
|
3
|
+
This software and its source code are proprietary and confidential.
|
|
4
|
+
No part of this software may be reproduced, distributed, modified,
|
|
5
|
+
or transmitted in any form or by any means without the prior written
|
|
6
|
+
permission of the copyright owner.
|
flense-0.1.0/PKG-INFO
ADDED
|
@@ -0,0 +1,148 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: flense
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Reverse proxy that compresses AI API payloads
|
|
5
|
+
Project-URL: Homepage, https://github.com/anandvmp-pintlab/flense
|
|
6
|
+
Project-URL: Repository, https://github.com/anandvmp-pintlab/flense.git
|
|
7
|
+
Project-URL: Issues, https://github.com/anandvmp-pintlab/flense/issues
|
|
8
|
+
Author-email: Anand <anand.v@pintlab.com>
|
|
9
|
+
License-Expression: AGPL-3.0-only
|
|
10
|
+
License-File: LICENSE
|
|
11
|
+
Requires-Python: >=3.11
|
|
12
|
+
Requires-Dist: fastapi>=0.115
|
|
13
|
+
Requires-Dist: httpx[http2]>=0.27
|
|
14
|
+
Requires-Dist: textual>=0.80
|
|
15
|
+
Requires-Dist: tiktoken>=0.7
|
|
16
|
+
Requires-Dist: tree-sitter>=0.23
|
|
17
|
+
Requires-Dist: typer>=0.12
|
|
18
|
+
Requires-Dist: uvicorn[standard]>=0.30
|
|
19
|
+
Provides-Extra: test
|
|
20
|
+
Requires-Dist: pytest-asyncio>=0.24; extra == 'test'
|
|
21
|
+
Requires-Dist: pytest>=8; extra == 'test'
|
|
22
|
+
Description-Content-Type: text/markdown
|
|
23
|
+
|
|
24
|
+
# flense
|
|
25
|
+
|
|
26
|
+
A lightweight reverse proxy that cuts your AI API costs by compressing large payloads before they reach expensive frontier models, with no changes to your application code.
|
|
27
|
+
|
|
28
|
+
---
|
|
29
|
+
|
|
30
|
+
## The Problem
|
|
31
|
+
|
|
32
|
+
Frontier AI models (Claude Opus, GPT-4, Gemini Ultra) are billed per token. Most large requests are padded with stuff the model does not need to see in full: entire source files, verbose docs, repetitive boilerplate. You pay full price for all of it.
|
|
33
|
+
|
|
34
|
+
## How Flense Helps
|
|
35
|
+
|
|
36
|
+
Flense sits between your application and the AI API. It intercepts outgoing requests, compresses heavy payloads using static code analysis, and forwards an optimised version to the frontier model. Your app sees no difference. Just a cheaper bill.
|
|
37
|
+
|
|
38
|
+
```
|
|
39
|
+
Your App -> localhost:2912 (flense) -> api.anthropic.com
|
|
40
|
+
```
|
|
41
|
+
|
|
42
|
+
No model calls. No quality trade-off for compression. Just fewer tokens.
|
|
43
|
+
|
|
44
|
+
---
|
|
45
|
+
|
|
46
|
+
## Quick Start
|
|
47
|
+
|
|
48
|
+
```bash
|
|
49
|
+
pip install flense
|
|
50
|
+
flense start
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
Then point your SDK at flense with the provider prefix:
|
|
54
|
+
|
|
55
|
+
```python
|
|
56
|
+
# Anthropic
|
|
57
|
+
client = Anthropic(base_url="http://localhost:2912/anthropic")
|
|
58
|
+
|
|
59
|
+
# OpenAI
|
|
60
|
+
client = OpenAI(base_url="http://localhost:2912/openai")
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
---
|
|
64
|
+
|
|
65
|
+
## How It Works
|
|
66
|
+
|
|
67
|
+
### Bulk-Reader
|
|
68
|
+
When a payload contains large code files or documents, flense compresses them using [Tree-sitter](https://tree-sitter.github.io/tree-sitter/), a fast, error-tolerant code parser that supports 100+ languages.
|
|
69
|
+
|
|
70
|
+
Instead of sending the full file, flense extracts:
|
|
71
|
+
- Class and function signatures
|
|
72
|
+
- Method names, parameters, return types
|
|
73
|
+
- Line number anchors so the model knows where everything came from
|
|
74
|
+
|
|
75
|
+
Function bodies, comments, and docstrings are stripped. The frontier model gets a skeleton. Enough to reason accurately, at a fraction of the token cost.
|
|
76
|
+
|
|
77
|
+
**No secondary model. No tokens spent on compression. Tree-sitter runs locally.**
|
|
78
|
+
|
|
79
|
+
Fallback chain: Tree-sitter -> Universal Ctags -> regex heuristics
|
|
80
|
+
|
|
81
|
+
### Code-Writer Bypass
|
|
82
|
+
For repetitive code generation tasks (scaffolding tests, mapping schemas), flense can route the request directly to a cheaper cloud model and write the output straight to disk, bypassing the frontier model entirely.
|
|
83
|
+
|
|
84
|
+
Triggered explicitly via a request header:
|
|
85
|
+
```python
|
|
86
|
+
headers={"X-Flense-Strategy": "code-writer"}
|
|
87
|
+
```
|
|
88
|
+
|
|
89
|
+
---
|
|
90
|
+
|
|
91
|
+
## Telemetry
|
|
92
|
+
|
|
93
|
+
Flense appends savings data to every response header. Readable by your app, or visible in the terminal dashboard:
|
|
94
|
+
|
|
95
|
+
```
|
|
96
|
+
X-Flense-Tokens-Saved: 18400
|
|
97
|
+
X-Flense-Est-Savings: $0.552
|
|
98
|
+
X-Flense-Compression-Time: 42ms
|
|
99
|
+
X-Flense-Strategy: bulk-reader
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
Run `flense tui` for a live terminal dashboard showing real-time savings across your session.
|
|
103
|
+
|
|
104
|
+
---
|
|
105
|
+
|
|
106
|
+
## Configuration
|
|
107
|
+
|
|
108
|
+
```toml
|
|
109
|
+
# flense.toml — all fields optional
|
|
110
|
+
|
|
111
|
+
[server]
|
|
112
|
+
port = 2912
|
|
113
|
+
headless = false
|
|
114
|
+
|
|
115
|
+
[compression]
|
|
116
|
+
threshold = 5000 # compress payloads above this token count
|
|
117
|
+
strategy = "auto" # auto | ast | ctags | passthrough
|
|
118
|
+
|
|
119
|
+
[providers.anthropic]
|
|
120
|
+
upstream = "https://api.anthropic.com"
|
|
121
|
+
|
|
122
|
+
[providers.openai]
|
|
123
|
+
upstream = "https://api.openai.com"
|
|
124
|
+
|
|
125
|
+
[code_writer]
|
|
126
|
+
model = "claude-haiku-4-5"
|
|
127
|
+
fallback = "gpt-4o-mini"
|
|
128
|
+
output_dir = "./generated"
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
---
|
|
132
|
+
|
|
133
|
+
## Provider Support
|
|
134
|
+
|
|
135
|
+
Flense supports **Anthropic** and **OpenAI** at v1. Route by URL prefix — each provider gets its own adapter with correct token counting and pricing:
|
|
136
|
+
|
|
137
|
+
```
|
|
138
|
+
localhost:2912/anthropic/v1/messages → api.anthropic.com
|
|
139
|
+
localhost:2912/openai/v1/chat/completions → api.openai.com
|
|
140
|
+
```
|
|
141
|
+
|
|
142
|
+
Adding a new provider is a single adapter file and route registration — nothing else changes.
|
|
143
|
+
|
|
144
|
+
---
|
|
145
|
+
|
|
146
|
+
## License
|
|
147
|
+
|
|
148
|
+
Copyright (c) 2026 Anand. All rights reserved. See [LICENSE](LICENSE).
|
flense-0.1.0/README.md
ADDED
|
@@ -0,0 +1,125 @@
|
|
|
1
|
+
# flense
|
|
2
|
+
|
|
3
|
+
A lightweight reverse proxy that cuts your AI API costs by compressing large payloads before they reach expensive frontier models, with no changes to your application code.
|
|
4
|
+
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
## The Problem
|
|
8
|
+
|
|
9
|
+
Frontier AI models (Claude Opus, GPT-4, Gemini Ultra) are billed per token. Most large requests are padded with stuff the model does not need to see in full: entire source files, verbose docs, repetitive boilerplate. You pay full price for all of it.
|
|
10
|
+
|
|
11
|
+
## How Flense Helps
|
|
12
|
+
|
|
13
|
+
Flense sits between your application and the AI API. It intercepts outgoing requests, compresses heavy payloads using static code analysis, and forwards an optimised version to the frontier model. Your app sees no difference. Just a cheaper bill.
|
|
14
|
+
|
|
15
|
+
```
|
|
16
|
+
Your App -> localhost:2912 (flense) -> api.anthropic.com
|
|
17
|
+
```
|
|
18
|
+
|
|
19
|
+
No model calls. No quality trade-off for compression. Just fewer tokens.
|
|
20
|
+
|
|
21
|
+
---
|
|
22
|
+
|
|
23
|
+
## Quick Start
|
|
24
|
+
|
|
25
|
+
```bash
|
|
26
|
+
pip install flense
|
|
27
|
+
flense start
|
|
28
|
+
```
|
|
29
|
+
|
|
30
|
+
Then point your SDK at flense with the provider prefix:
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
# Anthropic
|
|
34
|
+
client = Anthropic(base_url="http://localhost:2912/anthropic")
|
|
35
|
+
|
|
36
|
+
# OpenAI
|
|
37
|
+
client = OpenAI(base_url="http://localhost:2912/openai")
|
|
38
|
+
```
|
|
39
|
+
|
|
40
|
+
---
|
|
41
|
+
|
|
42
|
+
## How It Works
|
|
43
|
+
|
|
44
|
+
### Bulk-Reader
|
|
45
|
+
When a payload contains large code files or documents, flense compresses them using [Tree-sitter](https://tree-sitter.github.io/tree-sitter/), a fast, error-tolerant code parser that supports 100+ languages.
|
|
46
|
+
|
|
47
|
+
Instead of sending the full file, flense extracts:
|
|
48
|
+
- Class and function signatures
|
|
49
|
+
- Method names, parameters, return types
|
|
50
|
+
- Line number anchors so the model knows where everything came from
|
|
51
|
+
|
|
52
|
+
Function bodies, comments, and docstrings are stripped. The frontier model gets a skeleton. Enough to reason accurately, at a fraction of the token cost.
|
|
53
|
+
|
|
54
|
+
**No secondary model. No tokens spent on compression. Tree-sitter runs locally.**
|
|
55
|
+
|
|
56
|
+
Fallback chain: Tree-sitter -> Universal Ctags -> regex heuristics
|
|
57
|
+
|
|
58
|
+
### Code-Writer Bypass
|
|
59
|
+
For repetitive code generation tasks (scaffolding tests, mapping schemas), flense can route the request directly to a cheaper cloud model and write the output straight to disk, bypassing the frontier model entirely.
|
|
60
|
+
|
|
61
|
+
Triggered explicitly via a request header:
|
|
62
|
+
```python
|
|
63
|
+
headers={"X-Flense-Strategy": "code-writer"}
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## Telemetry
|
|
69
|
+
|
|
70
|
+
Flense appends savings data to every response header. Readable by your app, or visible in the terminal dashboard:
|
|
71
|
+
|
|
72
|
+
```
|
|
73
|
+
X-Flense-Tokens-Saved: 18400
|
|
74
|
+
X-Flense-Est-Savings: $0.552
|
|
75
|
+
X-Flense-Compression-Time: 42ms
|
|
76
|
+
X-Flense-Strategy: bulk-reader
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Run `flense tui` for a live terminal dashboard showing real-time savings across your session.
|
|
80
|
+
|
|
81
|
+
---
|
|
82
|
+
|
|
83
|
+
## Configuration
|
|
84
|
+
|
|
85
|
+
```toml
|
|
86
|
+
# flense.toml — all fields optional
|
|
87
|
+
|
|
88
|
+
[server]
|
|
89
|
+
port = 2912
|
|
90
|
+
headless = false
|
|
91
|
+
|
|
92
|
+
[compression]
|
|
93
|
+
threshold = 5000 # compress payloads above this token count
|
|
94
|
+
strategy = "auto" # auto | ast | ctags | passthrough
|
|
95
|
+
|
|
96
|
+
[providers.anthropic]
|
|
97
|
+
upstream = "https://api.anthropic.com"
|
|
98
|
+
|
|
99
|
+
[providers.openai]
|
|
100
|
+
upstream = "https://api.openai.com"
|
|
101
|
+
|
|
102
|
+
[code_writer]
|
|
103
|
+
model = "claude-haiku-4-5"
|
|
104
|
+
fallback = "gpt-4o-mini"
|
|
105
|
+
output_dir = "./generated"
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
---
|
|
109
|
+
|
|
110
|
+
## Provider Support
|
|
111
|
+
|
|
112
|
+
Flense supports **Anthropic** and **OpenAI** at v1. Route by URL prefix — each provider gets its own adapter with correct token counting and pricing:
|
|
113
|
+
|
|
114
|
+
```
|
|
115
|
+
localhost:2912/anthropic/v1/messages → api.anthropic.com
|
|
116
|
+
localhost:2912/openai/v1/chat/completions → api.openai.com
|
|
117
|
+
```
|
|
118
|
+
|
|
119
|
+
Adding a new provider is a single adapter file and route registration — nothing else changes.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## License
|
|
124
|
+
|
|
125
|
+
Copyright (c) 2026 Anand. All rights reserved. See [LICENSE](LICENSE).
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["hatchling"]
|
|
3
|
+
build-backend = "hatchling.build"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "flense"
|
|
7
|
+
version = "0.1.0"
|
|
8
|
+
description = "Reverse proxy that compresses AI API payloads"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.11"
|
|
11
|
+
license = "AGPL-3.0-only"
|
|
12
|
+
authors = [{ name = "Anand", email = "anand.v@pintlab.com" }]
|
|
13
|
+
dependencies = [
|
|
14
|
+
"fastapi>=0.115",
|
|
15
|
+
"httpx[http2]>=0.27",
|
|
16
|
+
"textual>=0.80",
|
|
17
|
+
"tiktoken>=0.7",
|
|
18
|
+
"tree-sitter>=0.23",
|
|
19
|
+
"typer>=0.12",
|
|
20
|
+
"uvicorn[standard]>=0.30",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
[project.optional-dependencies]
|
|
24
|
+
test = [
|
|
25
|
+
"pytest>=8",
|
|
26
|
+
"pytest-asyncio>=0.24",
|
|
27
|
+
]
|
|
28
|
+
|
|
29
|
+
[project.urls]
|
|
30
|
+
Homepage = "https://github.com/anandvmp-pintlab/flense"
|
|
31
|
+
Repository = "https://github.com/anandvmp-pintlab/flense.git"
|
|
32
|
+
Issues = "https://github.com/anandvmp-pintlab/flense/issues"
|
|
33
|
+
|
|
34
|
+
[project.scripts]
|
|
35
|
+
flense = "flense.cli:app"
|
|
36
|
+
|
|
37
|
+
[tool.hatch.build.targets.wheel]
|
|
38
|
+
packages = ["src/flense"]
|
|
39
|
+
|
|
40
|
+
[tool.pytest.ini_options]
|
|
41
|
+
asyncio_mode = "auto"
|
|
42
|
+
testpaths = ["tests"]
|