awecompress 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- awecompress/__init__.py +14 -0
- awecompress/cli.py +127 -0
- awecompress/compress.py +260 -0
- awecompress/config.py +161 -0
- awecompress/integrate.py +177 -0
- awecompress/protocols.py +421 -0
- awecompress/server.py +282 -0
- awecompress/store.py +110 -0
- awecompress/summarize.py +66 -0
- awecompress-0.2.0.dist-info/METADATA +206 -0
- awecompress-0.2.0.dist-info/RECORD +15 -0
- awecompress-0.2.0.dist-info/WHEEL +5 -0
- awecompress-0.2.0.dist-info/entry_points.txt +2 -0
- awecompress-0.2.0.dist-info/licenses/LICENSE +207 -0
- awecompress-0.2.0.dist-info/top_level.txt +1 -0
awecompress/__init__.py
ADDED
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
"""awecompress: local context compression for coding agents.
|
|
2
|
+
|
|
3
|
+
Sits between the harness (Claude Code, OpenCode, ...) and its upstream, or
|
|
4
|
+
runs inside awerouter beside odcp/rtk. When a session's history crosses a
|
|
5
|
+
token threshold, the oldest whole turns are replaced by one frozen LLM
|
|
6
|
+
summary — cached, so every later request reuses the same bytes.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from importlib.metadata import PackageNotFoundError, version
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
__version__ = version("awecompress")
|
|
13
|
+
except PackageNotFoundError: # running from a source checkout
|
|
14
|
+
__version__ = "0.2.0"
|
awecompress/cli.py
ADDED
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
"""CLI: serve / status / config / clear.
|
|
2
|
+
|
|
3
|
+
`serve` runs in the foreground (Ctrl-C stops it). Backgrounding is the
|
|
4
|
+
user's process manager — v1 ships no daemon machinery.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import asyncio
|
|
10
|
+
from dataclasses import replace
|
|
11
|
+
from pathlib import Path
|
|
12
|
+
|
|
13
|
+
import aiohttp
|
|
14
|
+
import click
|
|
15
|
+
|
|
16
|
+
from awecompress import __version__
|
|
17
|
+
from awecompress.config import config_path, load_config
|
|
18
|
+
from awecompress.integrate import Compressor
|
|
19
|
+
from awecompress.server import serve as serve_proxy
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@click.group(name="awecompress", context_settings={"help_option_names": ["-h", "--help"]})
|
|
23
|
+
@click.version_option(__version__, "-v", "--version", message="awecompress %(version)s")
|
|
24
|
+
def cli() -> None:
|
|
25
|
+
"""Compress long coding-agent context before it reaches your provider."""
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
@cli.command()
|
|
29
|
+
@click.option("--port", type=int, default=None, help="Listen port (default: config port or 8808).")
|
|
30
|
+
@click.option("--host", default=None, help="Listen address (default: 127.0.0.1).")
|
|
31
|
+
@click.option("--upstream", default=None, help="Upstream base URL (default: config upstream).")
|
|
32
|
+
@click.option("--config", "config_file", type=click.Path(), default=None,
|
|
33
|
+
help="Config file path (default: ~/.config/awecompress/config.json).")
|
|
34
|
+
def serve(port: int, host: str, upstream: str, config_file: str) -> None:
|
|
35
|
+
"""Run the compression proxy in the foreground."""
|
|
36
|
+
cfg = load_config(Path(config_file).expanduser() if config_file else None)
|
|
37
|
+
if upstream:
|
|
38
|
+
cfg = replace(cfg, upstream=upstream)
|
|
39
|
+
compressor = Compressor(cfg.db_path)
|
|
40
|
+
try:
|
|
41
|
+
asyncio.run(serve_proxy(cfg, compressor, port, host))
|
|
42
|
+
except KeyboardInterrupt:
|
|
43
|
+
pass
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@cli.command()
|
|
47
|
+
@click.option("--config", "config_file", type=click.Path(), default=None,
|
|
48
|
+
help="Config file path (default: ~/.config/awecompress/config.json).")
|
|
49
|
+
def status(config_file: str) -> None:
|
|
50
|
+
"""Show running state, config, and compression stats."""
|
|
51
|
+
cfg = load_config(Path(config_file).expanduser() if config_file else None)
|
|
52
|
+
url = f"http://{cfg.host}:{cfg.port}/"
|
|
53
|
+
running = False
|
|
54
|
+
try:
|
|
55
|
+
async def probe():
|
|
56
|
+
async with aiohttp.ClientSession() as session:
|
|
57
|
+
async with session.get(url, timeout=aiohttp.ClientTimeout(total=2)) as resp:
|
|
58
|
+
return await resp.json(content_type=None)
|
|
59
|
+
running = asyncio.run(probe()).get("service") == "awecompress"
|
|
60
|
+
except Exception:
|
|
61
|
+
pass
|
|
62
|
+
|
|
63
|
+
compressor = Compressor(cfg.db_path)
|
|
64
|
+
stats = compressor.stats()
|
|
65
|
+
compressor.close()
|
|
66
|
+
click.echo(f"awecompress {__version__}")
|
|
67
|
+
click.echo(f" proxy : {'running at ' + url if running else 'not running'}")
|
|
68
|
+
click.echo(f" upstream : {cfg.upstream}")
|
|
69
|
+
click.echo(f" compress : above {cfg.threshold_tokens} est. tokens, "
|
|
70
|
+
f"keep last {cfg.keep_recent_turns} turns, min span {cfg.min_span_tokens}")
|
|
71
|
+
click.echo(f" protected : {len(cfg.protected_tools)} tools"
|
|
72
|
+
+ (f", {len(cfg.protected_file_patterns)} file patterns"
|
|
73
|
+
if cfg.protected_file_patterns else ""))
|
|
74
|
+
click.echo(f" summaries : {stats['sessions']} sessions, {stats['calls']} summary calls, "
|
|
75
|
+
f"~{stats['saved_tokens']} tokens saved")
|
|
76
|
+
click.echo(f" store : {cfg.db_path}")
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
@cli.group()
|
|
80
|
+
def config() -> None:
|
|
81
|
+
"""Show the config file."""
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
@config.command("path")
|
|
85
|
+
def config_path_cmd() -> None:
|
|
86
|
+
"""Print the config file path."""
|
|
87
|
+
click.echo(str(config_path()))
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
@config.command("show")
|
|
91
|
+
def config_show() -> None:
|
|
92
|
+
"""Print the config file contents."""
|
|
93
|
+
path = config_path()
|
|
94
|
+
if not path.exists():
|
|
95
|
+
click.echo(f"(no config file at {path} — defaults apply; run 'awecompress serve' to write one)")
|
|
96
|
+
return
|
|
97
|
+
click.echo(path.read_text().rstrip())
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
@cli.command()
|
|
101
|
+
@click.option("--config", "config_file", type=click.Path(), default=None,
|
|
102
|
+
help="Config file path (default: ~/.config/awecompress/config.json).")
|
|
103
|
+
@click.option("--yes", is_flag=True, help="Delete without asking.")
|
|
104
|
+
def clear(yes: bool, config_file: str) -> None:
|
|
105
|
+
"""Delete stored summaries (sessions start uncompressed)."""
|
|
106
|
+
cfg = load_config(Path(config_file).expanduser() if config_file else None)
|
|
107
|
+
path = Path(cfg.db_path)
|
|
108
|
+
if not path.exists():
|
|
109
|
+
click.echo("nothing to clear — no summary store yet")
|
|
110
|
+
return
|
|
111
|
+
if not yes:
|
|
112
|
+
click.confirm(f"clear {path} (all frozen summaries)?", abort=True)
|
|
113
|
+
# Clear rows rather than unlink the file: through WAL this also empties
|
|
114
|
+
# the store a running proxy sees, instead of leaving it on a deleted file.
|
|
115
|
+
compressor = Compressor(path)
|
|
116
|
+
sessions = compressor.stats()["sessions"]
|
|
117
|
+
compressor.clear()
|
|
118
|
+
compressor.close()
|
|
119
|
+
click.echo(f"cleared {sessions} session(s) from {path}")
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def main() -> None:
|
|
123
|
+
cli()
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
if __name__ == "__main__":
|
|
127
|
+
main()
|
awecompress/compress.py
ADDED
|
@@ -0,0 +1,260 @@
|
|
|
1
|
+
"""Compression planning over a request body: pure functions, no I/O. The
|
|
2
|
+
integrate/server layers turn a Plan into one LLM call (summarize.py) and a
|
|
3
|
+
frozen replacement (store.py).
|
|
4
|
+
|
|
5
|
+
Model: a session's oldest whole turns are replaced by a single summary
|
|
6
|
+
message. A cut may only land on a turn boundary — a message a human actually
|
|
7
|
+
sent — so a tool call is never separated from its result and upstream pairing
|
|
8
|
+
validation never sees a half pair. History shapes are per-protocol
|
|
9
|
+
(protocols.py); this module only plans.
|
|
10
|
+
|
|
11
|
+
Protected content (DCP's Compress idea, proxy-shaped): tool calls whose name
|
|
12
|
+
is in protectedTools, or whose path-ish arguments match protectedFilePatterns,
|
|
13
|
+
render into the summarizer transcript uncapped and marked [protected]; the
|
|
14
|
+
summary prompt demands their content survive compression verbatim — todo
|
|
15
|
+
lists, plans, and task/skill outcomes are live planning state, not history
|
|
16
|
+
noise.
|
|
17
|
+
|
|
18
|
+
Summaries are frozen: every later request reuses the same stored bytes for
|
|
19
|
+
the same covered prefix, so the provider prompt cache sees a stable prefix.
|
|
20
|
+
Growing the covered span rewrites the summary once — a one-time cache miss.
|
|
21
|
+
"""
|
|
22
|
+
|
|
23
|
+
from __future__ import annotations
|
|
24
|
+
|
|
25
|
+
import hashlib
|
|
26
|
+
from dataclasses import dataclass
|
|
27
|
+
from fnmatch import fnmatch
|
|
28
|
+
|
|
29
|
+
from awecompress.protocols import (
|
|
30
|
+
ENDPOINT_PATHS, # noqa: F401 (re-export: standalone server relays by it)
|
|
31
|
+
PROTOCOLS,
|
|
32
|
+
canonical,
|
|
33
|
+
estimate_tokens, # noqa: F401 (re-export: callers import it from here)
|
|
34
|
+
)
|
|
35
|
+
|
|
36
|
+
# Marks the synthetic message we inject, in place of the covered prefix.
|
|
37
|
+
SUMMARY_MARKER = "[awecompress: summary of earlier turns — original messages removed]"
|
|
38
|
+
|
|
39
|
+
# Per-tool-input cap when flattening history for the summarizer: the summary
|
|
40
|
+
# needs what a call did, not every byte of its arguments.
|
|
41
|
+
TOOL_INPUT_CAP = 2000
|
|
42
|
+
|
|
43
|
+
# Protected by default: planning-state tools whose trace must survive
|
|
44
|
+
# compression intact (same convention as DCP's compress.protectedTools).
|
|
45
|
+
DEFAULT_PROTECTED_TOOLS = ("task", "skill", "todowrite", "todoread", "updateplan")
|
|
46
|
+
|
|
47
|
+
# Argument keys treated as file paths for protectedFilePatterns matching.
|
|
48
|
+
_PATH_KEYS = ("file_path", "path", "filepath", "notebook_path")
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _norm_tool(name: str) -> str:
|
|
52
|
+
"""Tool-name identity: lowercase with separators stripped, so TodoWrite
|
|
53
|
+
== todo_write == todowrite (same convention as awerouter's odcp)."""
|
|
54
|
+
return name.lower().replace("_", "").replace("-", "")
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def is_protected(name, args, protected_tools, patterns) -> bool:
|
|
58
|
+
"""One tool call's protection verdict. Name match is on the normalized
|
|
59
|
+
form; pattern match tests the call's path-ish argument values."""
|
|
60
|
+
if _norm_tool(name or "") in protected_tools:
|
|
61
|
+
return True
|
|
62
|
+
if patterns and isinstance(args, dict):
|
|
63
|
+
for key in _PATH_KEYS:
|
|
64
|
+
value = args.get(key)
|
|
65
|
+
if isinstance(value, str) and any(fnmatch(value, p) for p in patterns):
|
|
66
|
+
return True
|
|
67
|
+
return False
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
# ---------------------------------------------------------------------------
|
|
71
|
+
# Turn boundaries
|
|
72
|
+
# ---------------------------------------------------------------------------
|
|
73
|
+
|
|
74
|
+
def _is_turn_start(msg) -> bool:
|
|
75
|
+
"""Anthropic-shaped turn test (kept for direct callers/tests)."""
|
|
76
|
+
return PROTOCOLS["anthropic"].is_turn_start(msg)
|
|
77
|
+
|
|
78
|
+
|
|
79
|
+
def safe_cut(messages: list, keep_recent_turns: int, protocol: str = "anthropic") -> int:
|
|
80
|
+
"""Index where the compressed span may end: the start of the
|
|
81
|
+
keep_recent_turns-th-from-last genuine user turn, or 0 when the history
|
|
82
|
+
is too short to cut anything."""
|
|
83
|
+
adapter = PROTOCOLS[protocol]
|
|
84
|
+
boundaries = [i for i, m in enumerate(messages) if adapter.is_turn_start(m)]
|
|
85
|
+
if len(boundaries) <= keep_recent_turns:
|
|
86
|
+
return 0
|
|
87
|
+
return boundaries[-keep_recent_turns]
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
# ---------------------------------------------------------------------------
|
|
91
|
+
# Identity and change detection
|
|
92
|
+
# ---------------------------------------------------------------------------
|
|
93
|
+
|
|
94
|
+
def session_key(body: dict, protocol: str = "anthropic") -> str:
|
|
95
|
+
"""Stable identity across one session's requests: the protocol's standing
|
|
96
|
+
instructions plus the first message (coding agents resend both
|
|
97
|
+
byte-identical every turn). Empty string when there is nothing stable to
|
|
98
|
+
hold on to."""
|
|
99
|
+
adapter = PROTOCOLS[protocol]
|
|
100
|
+
messages = adapter.message_list(body)
|
|
101
|
+
if not messages:
|
|
102
|
+
return ""
|
|
103
|
+
return hashlib.sha256(canonical([adapter.system_identity(body),
|
|
104
|
+
messages[0]]).encode()).hexdigest()[:16]
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def prefix_hash(messages: list, upto: int) -> str:
|
|
108
|
+
"""Fingerprint of messages[:upto] — detects a session rewound to a
|
|
109
|
+
checkpoint or forked under a stored summary."""
|
|
110
|
+
return hashlib.sha256(canonical(messages[:upto]).encode()).hexdigest()[:16]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# ---------------------------------------------------------------------------
|
|
114
|
+
# Token estimates
|
|
115
|
+
# ---------------------------------------------------------------------------
|
|
116
|
+
|
|
117
|
+
def estimate_messages_tokens(messages: list, protocol: str = "anthropic") -> int:
|
|
118
|
+
adapter = PROTOCOLS[protocol]
|
|
119
|
+
return sum(adapter.item_tokens(m) for m in messages)
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def estimate_body_tokens(body: dict, messages: list, protocol: str = "anthropic") -> int:
|
|
123
|
+
"""Rough size of the request as sent: standing instructions plus messages.
|
|
124
|
+
Tool definitions are constant per session and deliberately excluded."""
|
|
125
|
+
return PROTOCOLS[protocol].system_tokens(body) \
|
|
126
|
+
+ estimate_messages_tokens(messages, protocol)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
# ---------------------------------------------------------------------------
|
|
130
|
+
# Transcript rendering (input to the summarizer)
|
|
131
|
+
# ---------------------------------------------------------------------------
|
|
132
|
+
|
|
133
|
+
def render_transcript(messages: list, cfg, protocol: str = "anthropic") -> str:
|
|
134
|
+
"""Flatten messages to compact text for the summarizer, applying the
|
|
135
|
+
protection rules. Protected calls render uncapped and carry the
|
|
136
|
+
[protected] marker; everything else is capped (result cap from cfg,
|
|
137
|
+
TOOL_INPUT_CAP for arguments). Thinking/reasoning never renders — the
|
|
138
|
+
assistant's visible text restates whatever mattered."""
|
|
139
|
+
adapter = PROTOCOLS[protocol]
|
|
140
|
+
tools = {_norm_tool(t) for t in (getattr(cfg, "protected_tools", None) or ())}
|
|
141
|
+
patterns = tuple(getattr(cfg, "protected_file_patterns", None) or ())
|
|
142
|
+
lines = []
|
|
143
|
+
for seg in adapter.segments(messages):
|
|
144
|
+
kind = seg[0]
|
|
145
|
+
if kind == "text":
|
|
146
|
+
_, role, text = seg
|
|
147
|
+
if text.strip():
|
|
148
|
+
lines.append(f"{role}: {text}")
|
|
149
|
+
elif kind == "call":
|
|
150
|
+
_, name, args = seg
|
|
151
|
+
protected = is_protected(name, args, tools, patterns)
|
|
152
|
+
text = canonical(args)
|
|
153
|
+
if not protected and len(text) > TOOL_INPUT_CAP:
|
|
154
|
+
text = text[:TOOL_INPUT_CAP] + f" [... {len(text) - TOOL_INPUT_CAP} chars truncated]"
|
|
155
|
+
mark = " [protected]" if protected else ""
|
|
156
|
+
lines.append(f"assistant calls {name}{mark}: {text}")
|
|
157
|
+
else: # result
|
|
158
|
+
_, name, args, text, errored = seg
|
|
159
|
+
protected = is_protected(name, args, tools, patterns)
|
|
160
|
+
cap = cfg.transcript_result_cap
|
|
161
|
+
if not protected and len(text) > cap:
|
|
162
|
+
text = text[:cap] + f" [... {len(text) - cap} chars truncated]"
|
|
163
|
+
if errored:
|
|
164
|
+
text = f"[error] {text}"
|
|
165
|
+
mark = " [protected]" if protected else ""
|
|
166
|
+
lines.append(f"tool (result){mark}: {text}")
|
|
167
|
+
return "\n".join(lines)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
# ---------------------------------------------------------------------------
|
|
171
|
+
# The synthetic summary message
|
|
172
|
+
# ---------------------------------------------------------------------------
|
|
173
|
+
|
|
174
|
+
def summary_message(summary: str, protocol: str = "anthropic") -> dict:
|
|
175
|
+
return PROTOCOLS[protocol].summary_message(f"{SUMMARY_MARKER}\n\n{summary}")
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def apply_summary(body: dict, summary: str, upto: int, protocol: str) -> list:
|
|
179
|
+
"""The rewritten history: protected preamble (openai-chat standing
|
|
180
|
+
instructions), the summary message, then everything from `upto` on."""
|
|
181
|
+
adapter = PROTOCOLS[protocol]
|
|
182
|
+
items = adapter.message_list(body) or []
|
|
183
|
+
return list(items[:adapter.preamble(items)]) \
|
|
184
|
+
+ [summary_message(summary, protocol)] + list(items[upto:])
|
|
185
|
+
|
|
186
|
+
|
|
187
|
+
# ---------------------------------------------------------------------------
|
|
188
|
+
# Planning
|
|
189
|
+
# ---------------------------------------------------------------------------
|
|
190
|
+
|
|
191
|
+
@dataclass
|
|
192
|
+
class Plan:
|
|
193
|
+
action: str # "passthrough" | "reuse" | "init" | "extend"
|
|
194
|
+
key: str = "" # session key (empty when unusable)
|
|
195
|
+
base_upto: int = 0 # messages already covered by a stored summary
|
|
196
|
+
upto: int = 0 # messages covered once the plan runs
|
|
197
|
+
prev_summary: str = "" # stored summary to merge into ("" when fresh)
|
|
198
|
+
span_tokens: int = 0 # estimate of the compressible new span
|
|
199
|
+
body_tokens: int = 0 # estimate of the body as it would be sent now
|
|
200
|
+
raw_tokens: int = 0 # estimate of the body with no compression
|
|
201
|
+
saved_tokens: int = 0 # raw_tokens - body_tokens (reuse path)
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def plan(body, stored, cfg, protocol: str = "anthropic") -> Plan:
|
|
205
|
+
"""Decide what to do with this request body.
|
|
206
|
+
|
|
207
|
+
passthrough — nothing to do (session small, or nothing new to cover).
|
|
208
|
+
reuse — stored summary still covers the prefix; apply it, no LLM call.
|
|
209
|
+
init/extend — over threshold with a new compressible span; summarize
|
|
210
|
+
messages[base_upto:cut] and freeze the result.
|
|
211
|
+
|
|
212
|
+
A stored record whose prefix no longer hashes right (session rewound to a
|
|
213
|
+
checkpoint, or forked) is ignored: the honest reading of a changed
|
|
214
|
+
history is to start over, never to splice an old summary onto it.
|
|
215
|
+
"""
|
|
216
|
+
if not isinstance(body, dict):
|
|
217
|
+
return Plan("passthrough")
|
|
218
|
+
adapter = PROTOCOLS[protocol]
|
|
219
|
+
messages = adapter.message_list(body)
|
|
220
|
+
if not messages:
|
|
221
|
+
return Plan("passthrough")
|
|
222
|
+
key = session_key(body, protocol)
|
|
223
|
+
if not key:
|
|
224
|
+
return Plan("passthrough")
|
|
225
|
+
start = adapter.preamble(messages) # standing instructions stay messages
|
|
226
|
+
|
|
227
|
+
if stored is not None and stored.prefix_hash != prefix_hash(messages, min(stored.upto, len(messages))):
|
|
228
|
+
stored = None
|
|
229
|
+
base = stored.upto if stored is not None else 0
|
|
230
|
+
|
|
231
|
+
raw_tokens = estimate_body_tokens(body, messages, protocol)
|
|
232
|
+
if stored is not None:
|
|
233
|
+
summary_item = summary_message(stored.summary, protocol)
|
|
234
|
+
body_tokens = adapter.system_tokens(body) \
|
|
235
|
+
+ estimate_messages_tokens(messages[:start], protocol) \
|
|
236
|
+
+ adapter.item_tokens(summary_item) \
|
|
237
|
+
+ estimate_messages_tokens(messages[base:], protocol)
|
|
238
|
+
else:
|
|
239
|
+
body_tokens = raw_tokens
|
|
240
|
+
|
|
241
|
+
def reuse() -> Plan:
|
|
242
|
+
return Plan("reuse", key, base, base, stored.summary, 0,
|
|
243
|
+
body_tokens, raw_tokens, max(0, raw_tokens - body_tokens))
|
|
244
|
+
|
|
245
|
+
cut = safe_cut(messages, cfg.keep_recent_turns, protocol)
|
|
246
|
+
if cut <= max(base, start):
|
|
247
|
+
return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
|
|
248
|
+
if body_tokens <= cfg.threshold_tokens:
|
|
249
|
+
return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
|
|
250
|
+
|
|
251
|
+
span_tokens = estimate_messages_tokens(messages[max(base, start):cut], protocol)
|
|
252
|
+
if span_tokens < cfg.min_span_tokens:
|
|
253
|
+
# Over threshold but the new span is crumbs — wait for more history
|
|
254
|
+
# rather than burn a summary call on nothing.
|
|
255
|
+
return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
|
|
256
|
+
|
|
257
|
+
action = "extend" if base > 0 else "init"
|
|
258
|
+
return Plan(action, key, base, cut,
|
|
259
|
+
stored.summary if stored is not None else "",
|
|
260
|
+
span_tokens, body_tokens, raw_tokens)
|
awecompress/config.py
ADDED
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
"""Config: one JSON file, written with defaults on first run.
|
|
2
|
+
|
|
3
|
+
Location: $AWECOMPRESS_CONFIG_DIR or ~/.config/awecompress/config.json
|
|
4
|
+
(same convention as awerouter). The summary store lives beside it.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from __future__ import annotations
|
|
8
|
+
|
|
9
|
+
import json
|
|
10
|
+
import os
|
|
11
|
+
from dataclasses import dataclass, replace
|
|
12
|
+
from pathlib import Path
|
|
13
|
+
from urllib.parse import urlparse
|
|
14
|
+
|
|
15
|
+
from awecompress.compress import DEFAULT_PROTECTED_TOOLS
|
|
16
|
+
|
|
17
|
+
DEFAULT_PORT = 8808
|
|
18
|
+
DEFAULT_UPSTREAM = "http://127.0.0.1:20128" # awerouter's default listen port
|
|
19
|
+
|
|
20
|
+
# Rough token estimates (chars/4 heuristic) — a 200k-token model's usable
|
|
21
|
+
# history comfortably crosses 60k long before the hard limit; compressing
|
|
22
|
+
# early keeps the summary small relative to what it replaces.
|
|
23
|
+
DEFAULT_THRESHOLD_TOKENS = 60000
|
|
24
|
+
DEFAULT_KEEP_RECENT_TURNS = 4
|
|
25
|
+
DEFAULT_MIN_SPAN_TOKENS = 8000
|
|
26
|
+
DEFAULT_TRANSCRIPT_RESULT_CAP = 4000
|
|
27
|
+
DEFAULT_SUMMARY_MAX_TOKENS = 2048
|
|
28
|
+
DEFAULT_SUMMARY_TIMEOUT_SECONDS = 60
|
|
29
|
+
|
|
30
|
+
def die(message: str) -> "SystemExit":
|
|
31
|
+
raise SystemExit(f"awecompress: {message}")
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def config_dir() -> Path:
|
|
35
|
+
return Path(os.environ.get("AWECOMPRESS_CONFIG_DIR", "~/.config/awecompress")).expanduser()
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def config_path() -> Path:
|
|
39
|
+
return config_dir() / "config.json"
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def db_path() -> Path:
|
|
43
|
+
return config_dir() / "summaries.db"
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass
|
|
47
|
+
class Config:
|
|
48
|
+
port: int = DEFAULT_PORT
|
|
49
|
+
host: str = "127.0.0.1"
|
|
50
|
+
upstream: str = DEFAULT_UPSTREAM
|
|
51
|
+
threshold_tokens: int = DEFAULT_THRESHOLD_TOKENS
|
|
52
|
+
keep_recent_turns: int = DEFAULT_KEEP_RECENT_TURNS
|
|
53
|
+
min_span_tokens: int = DEFAULT_MIN_SPAN_TOKENS
|
|
54
|
+
transcript_result_cap: int = DEFAULT_TRANSCRIPT_RESULT_CAP
|
|
55
|
+
# Empty = summarize with the request's own model, which the upstream
|
|
56
|
+
# (awerouter) then routes like any other request — usually flash.
|
|
57
|
+
summary_model: str = ""
|
|
58
|
+
summary_max_tokens: int = DEFAULT_SUMMARY_MAX_TOKENS
|
|
59
|
+
summary_timeout_seconds: int = DEFAULT_SUMMARY_TIMEOUT_SECONDS
|
|
60
|
+
# Protected content (see compress.py): these tools' calls and results
|
|
61
|
+
# render into the summarizer transcript uncapped and must survive the
|
|
62
|
+
# summary verbatim; file patterns protect path-matching calls the same way.
|
|
63
|
+
protected_tools: tuple = DEFAULT_PROTECTED_TOOLS
|
|
64
|
+
protected_file_patterns: tuple = ()
|
|
65
|
+
db_path: str = "" # empty = db_path() default
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def _default_file() -> dict:
|
|
69
|
+
return {
|
|
70
|
+
"port": DEFAULT_PORT,
|
|
71
|
+
"upstream": DEFAULT_UPSTREAM,
|
|
72
|
+
"thresholdTokens": DEFAULT_THRESHOLD_TOKENS,
|
|
73
|
+
"keepRecentTurns": DEFAULT_KEEP_RECENT_TURNS,
|
|
74
|
+
"minSpanTokens": DEFAULT_MIN_SPAN_TOKENS,
|
|
75
|
+
"summaryModel": "",
|
|
76
|
+
"summaryMaxTokens": DEFAULT_SUMMARY_MAX_TOKENS,
|
|
77
|
+
"protectedTools": list(DEFAULT_PROTECTED_TOOLS),
|
|
78
|
+
"protectedFilePatterns": [],
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
# File keys are camelCase (hand-edited like awerouter's routing.json);
|
|
83
|
+
# dataclass fields stay snake_case.
|
|
84
|
+
_KEY_MAP = {
|
|
85
|
+
"port": "port",
|
|
86
|
+
"host": "host",
|
|
87
|
+
"upstream": "upstream",
|
|
88
|
+
"thresholdTokens": "threshold_tokens",
|
|
89
|
+
"keepRecentTurns": "keep_recent_turns",
|
|
90
|
+
"minSpanTokens": "min_span_tokens",
|
|
91
|
+
"transcriptResultCap": "transcript_result_cap",
|
|
92
|
+
"summaryModel": "summary_model",
|
|
93
|
+
"summaryMaxTokens": "summary_max_tokens",
|
|
94
|
+
"summaryTimeoutSeconds": "summary_timeout_seconds",
|
|
95
|
+
"protectedTools": "protected_tools",
|
|
96
|
+
"protectedFilePatterns": "protected_file_patterns",
|
|
97
|
+
"dbPath": "db_path",
|
|
98
|
+
}
|
|
99
|
+
|
|
100
|
+
_INT_FIELDS = {
|
|
101
|
+
"port", "threshold_tokens", "keep_recent_turns", "min_span_tokens",
|
|
102
|
+
"transcript_result_cap", "summary_max_tokens", "summary_timeout_seconds",
|
|
103
|
+
}
|
|
104
|
+
_STR_FIELDS = {"host", "upstream", "summary_model", "db_path"}
|
|
105
|
+
_LIST_FIELDS = {"protected_tools", "protected_file_patterns"}
|
|
106
|
+
|
|
107
|
+
|
|
108
|
+
def load_config(path: "Path | None" = None) -> Config:
|
|
109
|
+
"""Load config, creating the default file on first run. Dies on unknown
|
|
110
|
+
keys or invalid values — a typo must not silently become a default."""
|
|
111
|
+
path = path or config_path()
|
|
112
|
+
if not path.exists():
|
|
113
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
114
|
+
path.write_text(json.dumps(_default_file(), indent=2) + "\n")
|
|
115
|
+
print(f"awecompress: wrote default config to {path}")
|
|
116
|
+
cfg = Config()
|
|
117
|
+
else:
|
|
118
|
+
try:
|
|
119
|
+
raw = json.loads(path.read_text())
|
|
120
|
+
except json.JSONDecodeError as exc:
|
|
121
|
+
die(f"{path}: invalid JSON ({exc})")
|
|
122
|
+
if not isinstance(raw, dict):
|
|
123
|
+
die(f"{path}: expected a JSON object")
|
|
124
|
+
|
|
125
|
+
cfg = Config()
|
|
126
|
+
for key, val in raw.items():
|
|
127
|
+
if key not in _KEY_MAP:
|
|
128
|
+
die(f"{path}: unknown key '{key}' (see README for the full list)")
|
|
129
|
+
field = _KEY_MAP[key]
|
|
130
|
+
if field in _INT_FIELDS:
|
|
131
|
+
if not isinstance(val, int) or isinstance(val, bool):
|
|
132
|
+
die(f"{path}: '{key}' must be an integer")
|
|
133
|
+
elif field in _STR_FIELDS:
|
|
134
|
+
if not isinstance(val, str):
|
|
135
|
+
die(f"{path}: '{key}' must be a string")
|
|
136
|
+
elif field in _LIST_FIELDS:
|
|
137
|
+
if not isinstance(val, list) or not all(isinstance(v, str) for v in val):
|
|
138
|
+
die(f"{path}: '{key}' must be an array of strings")
|
|
139
|
+
val = tuple(val)
|
|
140
|
+
setattr(cfg, field, val)
|
|
141
|
+
|
|
142
|
+
_validate(cfg)
|
|
143
|
+
return replace(cfg, db_path=cfg.db_path or str(db_path()))
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def _validate(cfg: Config) -> None:
|
|
147
|
+
if not (1 <= cfg.port <= 65535):
|
|
148
|
+
die(f"port must be 1..65535, got {cfg.port}")
|
|
149
|
+
parsed = urlparse(cfg.upstream)
|
|
150
|
+
if parsed.scheme not in ("http", "https") or not parsed.netloc:
|
|
151
|
+
die(f"upstream must be an http(s) URL, got '{cfg.upstream}'")
|
|
152
|
+
if cfg.threshold_tokens < 1000:
|
|
153
|
+
die(f"thresholdTokens must be >= 1000, got {cfg.threshold_tokens}")
|
|
154
|
+
if cfg.keep_recent_turns < 1:
|
|
155
|
+
die(f"keepRecentTurns must be >= 1, got {cfg.keep_recent_turns}")
|
|
156
|
+
if cfg.min_span_tokens < 1000:
|
|
157
|
+
die(f"minSpanTokens must be >= 1000, got {cfg.min_span_tokens}")
|
|
158
|
+
if cfg.summary_max_tokens < 256:
|
|
159
|
+
die(f"summaryMaxTokens must be >= 256, got {cfg.summary_max_tokens}")
|
|
160
|
+
if not (10 <= cfg.summary_timeout_seconds <= 600):
|
|
161
|
+
die(f"summaryTimeoutSeconds must be 10..600, got {cfg.summary_timeout_seconds}")
|