awecompress 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,14 @@
1
+ """awecompress: local context compression for coding agents.
2
+
3
+ Sits between the harness (Claude Code, OpenCode, ...) and its upstream, or
4
+ runs inside awerouter beside odcp/rtk. When a session's history crosses a
5
+ token threshold, the oldest whole turns are replaced by one frozen LLM
6
+ summary — cached, so every later request reuses the same bytes.
7
+ """
8
+
9
+ from importlib.metadata import PackageNotFoundError, version
10
+
11
+ try:
12
+ __version__ = version("awecompress")
13
+ except PackageNotFoundError: # running from a source checkout
14
+ __version__ = "0.2.0"
awecompress/cli.py ADDED
@@ -0,0 +1,127 @@
1
+ """CLI: serve / status / config / clear.
2
+
3
+ `serve` runs in the foreground (Ctrl-C stops it). Backgrounding is the
4
+ user's process manager — v1 ships no daemon machinery.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import asyncio
10
+ from dataclasses import replace
11
+ from pathlib import Path
12
+
13
+ import aiohttp
14
+ import click
15
+
16
+ from awecompress import __version__
17
+ from awecompress.config import config_path, load_config
18
+ from awecompress.integrate import Compressor
19
+ from awecompress.server import serve as serve_proxy
20
+
21
+
22
+ @click.group(name="awecompress", context_settings={"help_option_names": ["-h", "--help"]})
23
+ @click.version_option(__version__, "-v", "--version", message="awecompress %(version)s")
24
+ def cli() -> None:
25
+ """Compress long coding-agent context before it reaches your provider."""
26
+
27
+
28
+ @cli.command()
29
+ @click.option("--port", type=int, default=None, help="Listen port (default: config port or 8808).")
30
+ @click.option("--host", default=None, help="Listen address (default: 127.0.0.1).")
31
+ @click.option("--upstream", default=None, help="Upstream base URL (default: config upstream).")
32
+ @click.option("--config", "config_file", type=click.Path(), default=None,
33
+ help="Config file path (default: ~/.config/awecompress/config.json).")
34
+ def serve(port: int, host: str, upstream: str, config_file: str) -> None:
35
+ """Run the compression proxy in the foreground."""
36
+ cfg = load_config(Path(config_file).expanduser() if config_file else None)
37
+ if upstream:
38
+ cfg = replace(cfg, upstream=upstream)
39
+ compressor = Compressor(cfg.db_path)
40
+ try:
41
+ asyncio.run(serve_proxy(cfg, compressor, port, host))
42
+ except KeyboardInterrupt:
43
+ pass
44
+
45
+
46
+ @cli.command()
47
+ @click.option("--config", "config_file", type=click.Path(), default=None,
48
+ help="Config file path (default: ~/.config/awecompress/config.json).")
49
+ def status(config_file: str) -> None:
50
+ """Show running state, config, and compression stats."""
51
+ cfg = load_config(Path(config_file).expanduser() if config_file else None)
52
+ url = f"http://{cfg.host}:{cfg.port}/"
53
+ running = False
54
+ try:
55
+ async def probe():
56
+ async with aiohttp.ClientSession() as session:
57
+ async with session.get(url, timeout=aiohttp.ClientTimeout(total=2)) as resp:
58
+ return await resp.json(content_type=None)
59
+ running = asyncio.run(probe()).get("service") == "awecompress"
60
+ except Exception:
61
+ pass
62
+
63
+ compressor = Compressor(cfg.db_path)
64
+ stats = compressor.stats()
65
+ compressor.close()
66
+ click.echo(f"awecompress {__version__}")
67
+ click.echo(f" proxy : {'running at ' + url if running else 'not running'}")
68
+ click.echo(f" upstream : {cfg.upstream}")
69
+ click.echo(f" compress : above {cfg.threshold_tokens} est. tokens, "
70
+ f"keep last {cfg.keep_recent_turns} turns, min span {cfg.min_span_tokens}")
71
+ click.echo(f" protected : {len(cfg.protected_tools)} tools"
72
+ + (f", {len(cfg.protected_file_patterns)} file patterns"
73
+ if cfg.protected_file_patterns else ""))
74
+ click.echo(f" summaries : {stats['sessions']} sessions, {stats['calls']} summary calls, "
75
+ f"~{stats['saved_tokens']} tokens saved")
76
+ click.echo(f" store : {cfg.db_path}")
77
+
78
+
79
+ @cli.group()
80
+ def config() -> None:
81
+ """Show the config file."""
82
+
83
+
84
+ @config.command("path")
85
+ def config_path_cmd() -> None:
86
+ """Print the config file path."""
87
+ click.echo(str(config_path()))
88
+
89
+
90
+ @config.command("show")
91
+ def config_show() -> None:
92
+ """Print the config file contents."""
93
+ path = config_path()
94
+ if not path.exists():
95
+ click.echo(f"(no config file at {path} — defaults apply; run 'awecompress serve' to write one)")
96
+ return
97
+ click.echo(path.read_text().rstrip())
98
+
99
+
100
+ @cli.command()
101
+ @click.option("--config", "config_file", type=click.Path(), default=None,
102
+ help="Config file path (default: ~/.config/awecompress/config.json).")
103
+ @click.option("--yes", is_flag=True, help="Delete without asking.")
104
+ def clear(yes: bool, config_file: str) -> None:
105
+ """Delete stored summaries (sessions start uncompressed)."""
106
+ cfg = load_config(Path(config_file).expanduser() if config_file else None)
107
+ path = Path(cfg.db_path)
108
+ if not path.exists():
109
+ click.echo("nothing to clear — no summary store yet")
110
+ return
111
+ if not yes:
112
+ click.confirm(f"clear {path} (all frozen summaries)?", abort=True)
113
+ # Clear rows rather than unlink the file: through WAL this also empties
114
+ # the store a running proxy sees, instead of leaving it on a deleted file.
115
+ compressor = Compressor(path)
116
+ sessions = compressor.stats()["sessions"]
117
+ compressor.clear()
118
+ compressor.close()
119
+ click.echo(f"cleared {sessions} session(s) from {path}")
120
+
121
+
122
+ def main() -> None:
123
+ cli()
124
+
125
+
126
+ if __name__ == "__main__":
127
+ main()
@@ -0,0 +1,260 @@
1
+ """Compression planning over a request body: pure functions, no I/O. The
2
+ integrate/server layers turn a Plan into one LLM call (summarize.py) and a
3
+ frozen replacement (store.py).
4
+
5
+ Model: a session's oldest whole turns are replaced by a single summary
6
+ message. A cut may only land on a turn boundary — a message a human actually
7
+ sent — so a tool call is never separated from its result and upstream pairing
8
+ validation never sees a half pair. History shapes are per-protocol
9
+ (protocols.py); this module only plans.
10
+
11
+ Protected content (DCP's Compress idea, proxy-shaped): tool calls whose name
12
+ is in protectedTools, or whose path-ish arguments match protectedFilePatterns,
13
+ render into the summarizer transcript uncapped and marked [protected]; the
14
+ summary prompt demands their content survive compression verbatim — todo
15
+ lists, plans, and task/skill outcomes are live planning state, not history
16
+ noise.
17
+
18
+ Summaries are frozen: every later request reuses the same stored bytes for
19
+ the same covered prefix, so the provider prompt cache sees a stable prefix.
20
+ Growing the covered span rewrites the summary once — a one-time cache miss.
21
+ """
22
+
23
+ from __future__ import annotations
24
+
25
+ import hashlib
26
+ from dataclasses import dataclass
27
+ from fnmatch import fnmatch
28
+
29
+ from awecompress.protocols import (
30
+ ENDPOINT_PATHS, # noqa: F401 (re-export: standalone server relays by it)
31
+ PROTOCOLS,
32
+ canonical,
33
+ estimate_tokens, # noqa: F401 (re-export: callers import it from here)
34
+ )
35
+
36
+ # Marks the synthetic message we inject, in place of the covered prefix.
37
+ SUMMARY_MARKER = "[awecompress: summary of earlier turns — original messages removed]"
38
+
39
+ # Per-tool-input cap when flattening history for the summarizer: the summary
40
+ # needs what a call did, not every byte of its arguments.
41
+ TOOL_INPUT_CAP = 2000
42
+
43
+ # Protected by default: planning-state tools whose trace must survive
44
+ # compression intact (same convention as DCP's compress.protectedTools).
45
+ DEFAULT_PROTECTED_TOOLS = ("task", "skill", "todowrite", "todoread", "updateplan")
46
+
47
+ # Argument keys treated as file paths for protectedFilePatterns matching.
48
+ _PATH_KEYS = ("file_path", "path", "filepath", "notebook_path")
49
+
50
+
51
+ def _norm_tool(name: str) -> str:
52
+ """Tool-name identity: lowercase with separators stripped, so TodoWrite
53
+ == todo_write == todowrite (same convention as awerouter's odcp)."""
54
+ return name.lower().replace("_", "").replace("-", "")
55
+
56
+
57
+ def is_protected(name, args, protected_tools, patterns) -> bool:
58
+ """One tool call's protection verdict. Name match is on the normalized
59
+ form; pattern match tests the call's path-ish argument values."""
60
+ if _norm_tool(name or "") in protected_tools:
61
+ return True
62
+ if patterns and isinstance(args, dict):
63
+ for key in _PATH_KEYS:
64
+ value = args.get(key)
65
+ if isinstance(value, str) and any(fnmatch(value, p) for p in patterns):
66
+ return True
67
+ return False
68
+
69
+
70
+ # ---------------------------------------------------------------------------
71
+ # Turn boundaries
72
+ # ---------------------------------------------------------------------------
73
+
74
+ def _is_turn_start(msg) -> bool:
75
+ """Anthropic-shaped turn test (kept for direct callers/tests)."""
76
+ return PROTOCOLS["anthropic"].is_turn_start(msg)
77
+
78
+
79
+ def safe_cut(messages: list, keep_recent_turns: int, protocol: str = "anthropic") -> int:
80
+ """Index where the compressed span may end: the start of the
81
+ keep_recent_turns-th-from-last genuine user turn, or 0 when the history
82
+ is too short to cut anything."""
83
+ adapter = PROTOCOLS[protocol]
84
+ boundaries = [i for i, m in enumerate(messages) if adapter.is_turn_start(m)]
85
+ if len(boundaries) <= keep_recent_turns:
86
+ return 0
87
+ return boundaries[-keep_recent_turns]
88
+
89
+
90
+ # ---------------------------------------------------------------------------
91
+ # Identity and change detection
92
+ # ---------------------------------------------------------------------------
93
+
94
+ def session_key(body: dict, protocol: str = "anthropic") -> str:
95
+ """Stable identity across one session's requests: the protocol's standing
96
+ instructions plus the first message (coding agents resend both
97
+ byte-identical every turn). Empty string when there is nothing stable to
98
+ hold on to."""
99
+ adapter = PROTOCOLS[protocol]
100
+ messages = adapter.message_list(body)
101
+ if not messages:
102
+ return ""
103
+ return hashlib.sha256(canonical([adapter.system_identity(body),
104
+ messages[0]]).encode()).hexdigest()[:16]
105
+
106
+
107
+ def prefix_hash(messages: list, upto: int) -> str:
108
+ """Fingerprint of messages[:upto] — detects a session rewound to a
109
+ checkpoint or forked under a stored summary."""
110
+ return hashlib.sha256(canonical(messages[:upto]).encode()).hexdigest()[:16]
111
+
112
+
113
+ # ---------------------------------------------------------------------------
114
+ # Token estimates
115
+ # ---------------------------------------------------------------------------
116
+
117
+ def estimate_messages_tokens(messages: list, protocol: str = "anthropic") -> int:
118
+ adapter = PROTOCOLS[protocol]
119
+ return sum(adapter.item_tokens(m) for m in messages)
120
+
121
+
122
+ def estimate_body_tokens(body: dict, messages: list, protocol: str = "anthropic") -> int:
123
+ """Rough size of the request as sent: standing instructions plus messages.
124
+ Tool definitions are constant per session and deliberately excluded."""
125
+ return PROTOCOLS[protocol].system_tokens(body) \
126
+ + estimate_messages_tokens(messages, protocol)
127
+
128
+
129
+ # ---------------------------------------------------------------------------
130
+ # Transcript rendering (input to the summarizer)
131
+ # ---------------------------------------------------------------------------
132
+
133
+ def render_transcript(messages: list, cfg, protocol: str = "anthropic") -> str:
134
+ """Flatten messages to compact text for the summarizer, applying the
135
+ protection rules. Protected calls render uncapped and carry the
136
+ [protected] marker; everything else is capped (result cap from cfg,
137
+ TOOL_INPUT_CAP for arguments). Thinking/reasoning never renders — the
138
+ assistant's visible text restates whatever mattered."""
139
+ adapter = PROTOCOLS[protocol]
140
+ tools = {_norm_tool(t) for t in (getattr(cfg, "protected_tools", None) or ())}
141
+ patterns = tuple(getattr(cfg, "protected_file_patterns", None) or ())
142
+ lines = []
143
+ for seg in adapter.segments(messages):
144
+ kind = seg[0]
145
+ if kind == "text":
146
+ _, role, text = seg
147
+ if text.strip():
148
+ lines.append(f"{role}: {text}")
149
+ elif kind == "call":
150
+ _, name, args = seg
151
+ protected = is_protected(name, args, tools, patterns)
152
+ text = canonical(args)
153
+ if not protected and len(text) > TOOL_INPUT_CAP:
154
+ text = text[:TOOL_INPUT_CAP] + f" [... {len(text) - TOOL_INPUT_CAP} chars truncated]"
155
+ mark = " [protected]" if protected else ""
156
+ lines.append(f"assistant calls {name}{mark}: {text}")
157
+ else: # result
158
+ _, name, args, text, errored = seg
159
+ protected = is_protected(name, args, tools, patterns)
160
+ cap = cfg.transcript_result_cap
161
+ if not protected and len(text) > cap:
162
+ text = text[:cap] + f" [... {len(text) - cap} chars truncated]"
163
+ if errored:
164
+ text = f"[error] {text}"
165
+ mark = " [protected]" if protected else ""
166
+ lines.append(f"tool (result){mark}: {text}")
167
+ return "\n".join(lines)
168
+
169
+
170
+ # ---------------------------------------------------------------------------
171
+ # The synthetic summary message
172
+ # ---------------------------------------------------------------------------
173
+
174
+ def summary_message(summary: str, protocol: str = "anthropic") -> dict:
175
+ return PROTOCOLS[protocol].summary_message(f"{SUMMARY_MARKER}\n\n{summary}")
176
+
177
+
178
+ def apply_summary(body: dict, summary: str, upto: int, protocol: str) -> list:
179
+ """The rewritten history: protected preamble (openai-chat standing
180
+ instructions), the summary message, then everything from `upto` on."""
181
+ adapter = PROTOCOLS[protocol]
182
+ items = adapter.message_list(body) or []
183
+ return list(items[:adapter.preamble(items)]) \
184
+ + [summary_message(summary, protocol)] + list(items[upto:])
185
+
186
+
187
+ # ---------------------------------------------------------------------------
188
+ # Planning
189
+ # ---------------------------------------------------------------------------
190
+
191
+ @dataclass
192
+ class Plan:
193
+ action: str # "passthrough" | "reuse" | "init" | "extend"
194
+ key: str = "" # session key (empty when unusable)
195
+ base_upto: int = 0 # messages already covered by a stored summary
196
+ upto: int = 0 # messages covered once the plan runs
197
+ prev_summary: str = "" # stored summary to merge into ("" when fresh)
198
+ span_tokens: int = 0 # estimate of the compressible new span
199
+ body_tokens: int = 0 # estimate of the body as it would be sent now
200
+ raw_tokens: int = 0 # estimate of the body with no compression
201
+ saved_tokens: int = 0 # raw_tokens - body_tokens (reuse path)
202
+
203
+
204
+ def plan(body, stored, cfg, protocol: str = "anthropic") -> Plan:
205
+ """Decide what to do with this request body.
206
+
207
+ passthrough — nothing to do (session small, or nothing new to cover).
208
+ reuse — stored summary still covers the prefix; apply it, no LLM call.
209
+ init/extend — over threshold with a new compressible span; summarize
210
+ messages[base_upto:cut] and freeze the result.
211
+
212
+ A stored record whose prefix no longer hashes right (session rewound to a
213
+ checkpoint, or forked) is ignored: the honest reading of a changed
214
+ history is to start over, never to splice an old summary onto it.
215
+ """
216
+ if not isinstance(body, dict):
217
+ return Plan("passthrough")
218
+ adapter = PROTOCOLS[protocol]
219
+ messages = adapter.message_list(body)
220
+ if not messages:
221
+ return Plan("passthrough")
222
+ key = session_key(body, protocol)
223
+ if not key:
224
+ return Plan("passthrough")
225
+ start = adapter.preamble(messages) # standing instructions stay messages
226
+
227
+ if stored is not None and stored.prefix_hash != prefix_hash(messages, min(stored.upto, len(messages))):
228
+ stored = None
229
+ base = stored.upto if stored is not None else 0
230
+
231
+ raw_tokens = estimate_body_tokens(body, messages, protocol)
232
+ if stored is not None:
233
+ summary_item = summary_message(stored.summary, protocol)
234
+ body_tokens = adapter.system_tokens(body) \
235
+ + estimate_messages_tokens(messages[:start], protocol) \
236
+ + adapter.item_tokens(summary_item) \
237
+ + estimate_messages_tokens(messages[base:], protocol)
238
+ else:
239
+ body_tokens = raw_tokens
240
+
241
+ def reuse() -> Plan:
242
+ return Plan("reuse", key, base, base, stored.summary, 0,
243
+ body_tokens, raw_tokens, max(0, raw_tokens - body_tokens))
244
+
245
+ cut = safe_cut(messages, cfg.keep_recent_turns, protocol)
246
+ if cut <= max(base, start):
247
+ return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
248
+ if body_tokens <= cfg.threshold_tokens:
249
+ return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
250
+
251
+ span_tokens = estimate_messages_tokens(messages[max(base, start):cut], protocol)
252
+ if span_tokens < cfg.min_span_tokens:
253
+ # Over threshold but the new span is crumbs — wait for more history
254
+ # rather than burn a summary call on nothing.
255
+ return reuse() if stored is not None else Plan("passthrough", key, raw_tokens=raw_tokens)
256
+
257
+ action = "extend" if base > 0 else "init"
258
+ return Plan(action, key, base, cut,
259
+ stored.summary if stored is not None else "",
260
+ span_tokens, body_tokens, raw_tokens)
awecompress/config.py ADDED
@@ -0,0 +1,161 @@
1
+ """Config: one JSON file, written with defaults on first run.
2
+
3
+ Location: $AWECOMPRESS_CONFIG_DIR or ~/.config/awecompress/config.json
4
+ (same convention as awerouter). The summary store lives beside it.
5
+ """
6
+
7
+ from __future__ import annotations
8
+
9
+ import json
10
+ import os
11
+ from dataclasses import dataclass, replace
12
+ from pathlib import Path
13
+ from urllib.parse import urlparse
14
+
15
+ from awecompress.compress import DEFAULT_PROTECTED_TOOLS
16
+
17
+ DEFAULT_PORT = 8808
18
+ DEFAULT_UPSTREAM = "http://127.0.0.1:20128" # awerouter's default listen port
19
+
20
+ # Rough token estimates (chars/4 heuristic) — a 200k-token model's usable
21
+ # history comfortably crosses 60k long before the hard limit; compressing
22
+ # early keeps the summary small relative to what it replaces.
23
+ DEFAULT_THRESHOLD_TOKENS = 60000
24
+ DEFAULT_KEEP_RECENT_TURNS = 4
25
+ DEFAULT_MIN_SPAN_TOKENS = 8000
26
+ DEFAULT_TRANSCRIPT_RESULT_CAP = 4000
27
+ DEFAULT_SUMMARY_MAX_TOKENS = 2048
28
+ DEFAULT_SUMMARY_TIMEOUT_SECONDS = 60
29
+
30
+ def die(message: str) -> "SystemExit":
31
+ raise SystemExit(f"awecompress: {message}")
32
+
33
+
34
+ def config_dir() -> Path:
35
+ return Path(os.environ.get("AWECOMPRESS_CONFIG_DIR", "~/.config/awecompress")).expanduser()
36
+
37
+
38
+ def config_path() -> Path:
39
+ return config_dir() / "config.json"
40
+
41
+
42
+ def db_path() -> Path:
43
+ return config_dir() / "summaries.db"
44
+
45
+
46
+ @dataclass
47
+ class Config:
48
+ port: int = DEFAULT_PORT
49
+ host: str = "127.0.0.1"
50
+ upstream: str = DEFAULT_UPSTREAM
51
+ threshold_tokens: int = DEFAULT_THRESHOLD_TOKENS
52
+ keep_recent_turns: int = DEFAULT_KEEP_RECENT_TURNS
53
+ min_span_tokens: int = DEFAULT_MIN_SPAN_TOKENS
54
+ transcript_result_cap: int = DEFAULT_TRANSCRIPT_RESULT_CAP
55
+ # Empty = summarize with the request's own model, which the upstream
56
+ # (awerouter) then routes like any other request — usually flash.
57
+ summary_model: str = ""
58
+ summary_max_tokens: int = DEFAULT_SUMMARY_MAX_TOKENS
59
+ summary_timeout_seconds: int = DEFAULT_SUMMARY_TIMEOUT_SECONDS
60
+ # Protected content (see compress.py): these tools' calls and results
61
+ # render into the summarizer transcript uncapped and must survive the
62
+ # summary verbatim; file patterns protect path-matching calls the same way.
63
+ protected_tools: tuple = DEFAULT_PROTECTED_TOOLS
64
+ protected_file_patterns: tuple = ()
65
+ db_path: str = "" # empty = db_path() default
66
+
67
+
68
+ def _default_file() -> dict:
69
+ return {
70
+ "port": DEFAULT_PORT,
71
+ "upstream": DEFAULT_UPSTREAM,
72
+ "thresholdTokens": DEFAULT_THRESHOLD_TOKENS,
73
+ "keepRecentTurns": DEFAULT_KEEP_RECENT_TURNS,
74
+ "minSpanTokens": DEFAULT_MIN_SPAN_TOKENS,
75
+ "summaryModel": "",
76
+ "summaryMaxTokens": DEFAULT_SUMMARY_MAX_TOKENS,
77
+ "protectedTools": list(DEFAULT_PROTECTED_TOOLS),
78
+ "protectedFilePatterns": [],
79
+ }
80
+
81
+
82
+ # File keys are camelCase (hand-edited like awerouter's routing.json);
83
+ # dataclass fields stay snake_case.
84
+ _KEY_MAP = {
85
+ "port": "port",
86
+ "host": "host",
87
+ "upstream": "upstream",
88
+ "thresholdTokens": "threshold_tokens",
89
+ "keepRecentTurns": "keep_recent_turns",
90
+ "minSpanTokens": "min_span_tokens",
91
+ "transcriptResultCap": "transcript_result_cap",
92
+ "summaryModel": "summary_model",
93
+ "summaryMaxTokens": "summary_max_tokens",
94
+ "summaryTimeoutSeconds": "summary_timeout_seconds",
95
+ "protectedTools": "protected_tools",
96
+ "protectedFilePatterns": "protected_file_patterns",
97
+ "dbPath": "db_path",
98
+ }
99
+
100
+ _INT_FIELDS = {
101
+ "port", "threshold_tokens", "keep_recent_turns", "min_span_tokens",
102
+ "transcript_result_cap", "summary_max_tokens", "summary_timeout_seconds",
103
+ }
104
+ _STR_FIELDS = {"host", "upstream", "summary_model", "db_path"}
105
+ _LIST_FIELDS = {"protected_tools", "protected_file_patterns"}
106
+
107
+
108
+ def load_config(path: "Path | None" = None) -> Config:
109
+ """Load config, creating the default file on first run. Dies on unknown
110
+ keys or invalid values — a typo must not silently become a default."""
111
+ path = path or config_path()
112
+ if not path.exists():
113
+ path.parent.mkdir(parents=True, exist_ok=True)
114
+ path.write_text(json.dumps(_default_file(), indent=2) + "\n")
115
+ print(f"awecompress: wrote default config to {path}")
116
+ cfg = Config()
117
+ else:
118
+ try:
119
+ raw = json.loads(path.read_text())
120
+ except json.JSONDecodeError as exc:
121
+ die(f"{path}: invalid JSON ({exc})")
122
+ if not isinstance(raw, dict):
123
+ die(f"{path}: expected a JSON object")
124
+
125
+ cfg = Config()
126
+ for key, val in raw.items():
127
+ if key not in _KEY_MAP:
128
+ die(f"{path}: unknown key '{key}' (see README for the full list)")
129
+ field = _KEY_MAP[key]
130
+ if field in _INT_FIELDS:
131
+ if not isinstance(val, int) or isinstance(val, bool):
132
+ die(f"{path}: '{key}' must be an integer")
133
+ elif field in _STR_FIELDS:
134
+ if not isinstance(val, str):
135
+ die(f"{path}: '{key}' must be a string")
136
+ elif field in _LIST_FIELDS:
137
+ if not isinstance(val, list) or not all(isinstance(v, str) for v in val):
138
+ die(f"{path}: '{key}' must be an array of strings")
139
+ val = tuple(val)
140
+ setattr(cfg, field, val)
141
+
142
+ _validate(cfg)
143
+ return replace(cfg, db_path=cfg.db_path or str(db_path()))
144
+
145
+
146
+ def _validate(cfg: Config) -> None:
147
+ if not (1 <= cfg.port <= 65535):
148
+ die(f"port must be 1..65535, got {cfg.port}")
149
+ parsed = urlparse(cfg.upstream)
150
+ if parsed.scheme not in ("http", "https") or not parsed.netloc:
151
+ die(f"upstream must be an http(s) URL, got '{cfg.upstream}'")
152
+ if cfg.threshold_tokens < 1000:
153
+ die(f"thresholdTokens must be >= 1000, got {cfg.threshold_tokens}")
154
+ if cfg.keep_recent_turns < 1:
155
+ die(f"keepRecentTurns must be >= 1, got {cfg.keep_recent_turns}")
156
+ if cfg.min_span_tokens < 1000:
157
+ die(f"minSpanTokens must be >= 1000, got {cfg.min_span_tokens}")
158
+ if cfg.summary_max_tokens < 256:
159
+ die(f"summaryMaxTokens must be >= 256, got {cfg.summary_max_tokens}")
160
+ if not (10 <= cfg.summary_timeout_seconds <= 600):
161
+ die(f"summaryTimeoutSeconds must be 10..600, got {cfg.summary_timeout_seconds}")