token-runtime 0.1.0a2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- token_runtime/__init__.py +1 -0
- token_runtime/adapters.py +181 -0
- token_runtime/agent_integrations.py +114 -0
- token_runtime/anthropic_adapter.py +301 -0
- token_runtime/anthropic_conformance.py +336 -0
- token_runtime/anthropic_preserved_thinking_cert.py +213 -0
- token_runtime/benchmark.py +106 -0
- token_runtime/capabilities.py +111 -0
- token_runtime/cli.py +284 -0
- token_runtime/codex_recertification.py +115 -0
- token_runtime/compat_cert.py +462 -0
- token_runtime/compatibility.py +81 -0
- token_runtime/config.py +72 -0
- token_runtime/conformance.py +36 -0
- token_runtime/contracts.py +70 -0
- token_runtime/engine.py +136 -0
- token_runtime/feature_flags.py +78 -0
- token_runtime/gateway.py +151 -0
- token_runtime/gemini_adapter.py +260 -0
- token_runtime/gemini_conformance.py +351 -0
- token_runtime/integrations.py +272 -0
- token_runtime/metrics.py +79 -0
- token_runtime/model.py +39 -0
- token_runtime/openai_certification.py +310 -0
- token_runtime/planner.py +96 -0
- token_runtime/reducers.py +179 -0
- token_runtime/store.py +36 -0
- token_runtime/strategies.py +50 -0
- token_runtime/terminal_ui.py +96 -0
- token_runtime-0.1.0a2.dist-info/METADATA +238 -0
- token_runtime-0.1.0a2.dist-info/RECORD +35 -0
- token_runtime-0.1.0a2.dist-info/WHEEL +5 -0
- token_runtime-0.1.0a2.dist-info/entry_points.txt +2 -0
- token_runtime-0.1.0a2.dist-info/licenses/LICENSE +202 -0
- token_runtime-0.1.0a2.dist-info/top_level.txt +1 -0
token_runtime/metrics.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
import sqlite3
|
|
7
|
+
import time
|
|
8
|
+
from typing import Any, Mapping
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
_ALLOWED = {
|
|
12
|
+
"decision", "adapter", "provider", "model",
|
|
13
|
+
"before_input_tokens", "after_input_tokens",
|
|
14
|
+
"provider_input_tokens", "provider_output_tokens", "cached_tokens",
|
|
15
|
+
"latency_ms", "reducer_ids", "reason_codes",
|
|
16
|
+
}
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class MetricsStore:
|
|
20
|
+
def __init__(self, path: str | Path):
|
|
21
|
+
self.path = str(path)
|
|
22
|
+
Path(self.path).parent.mkdir(parents=True, exist_ok=True)
|
|
23
|
+
with sqlite3.connect(self.path) as conn:
|
|
24
|
+
conn.execute(
|
|
25
|
+
"CREATE TABLE IF NOT EXISTS events ("
|
|
26
|
+
"ts REAL NOT NULL, decision TEXT NOT NULL, adapter TEXT, provider TEXT, model TEXT, "
|
|
27
|
+
"before_input_tokens INTEGER, after_input_tokens INTEGER, provider_input_tokens INTEGER, "
|
|
28
|
+
"provider_output_tokens INTEGER, cached_tokens INTEGER, latency_ms REAL, "
|
|
29
|
+
"reducer_ids TEXT NOT NULL, reason_codes TEXT NOT NULL)"
|
|
30
|
+
)
|
|
31
|
+
if os.name == "posix":
|
|
32
|
+
os.chmod(self.path, 0o600)
|
|
33
|
+
|
|
34
|
+
def record(self, event: Mapping[str, Any]) -> None:
|
|
35
|
+
unknown = set(event) - _ALLOWED
|
|
36
|
+
if unknown:
|
|
37
|
+
raise ValueError(f"unsupported telemetry fields: {sorted(unknown)}")
|
|
38
|
+
decision = event.get("decision")
|
|
39
|
+
if decision not in {"optimize", "bypass"}:
|
|
40
|
+
raise ValueError("decision must be optimize or bypass")
|
|
41
|
+
|
|
42
|
+
values = (
|
|
43
|
+
time.time(), decision, event.get("adapter"), event.get("provider"), event.get("model"),
|
|
44
|
+
event.get("before_input_tokens"), event.get("after_input_tokens"),
|
|
45
|
+
event.get("provider_input_tokens"), event.get("provider_output_tokens"),
|
|
46
|
+
event.get("cached_tokens"), event.get("latency_ms"),
|
|
47
|
+
json.dumps(tuple(event.get("reducer_ids", ()))),
|
|
48
|
+
json.dumps(tuple(event.get("reason_codes", ()))),
|
|
49
|
+
)
|
|
50
|
+
with sqlite3.connect(self.path) as conn:
|
|
51
|
+
conn.execute(
|
|
52
|
+
"INSERT INTO events VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
|
|
53
|
+
values,
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
def summary(self) -> dict[str, int | float]:
|
|
57
|
+
with sqlite3.connect(self.path) as conn:
|
|
58
|
+
row = conn.execute(
|
|
59
|
+
"SELECT COUNT(*), "
|
|
60
|
+
"SUM(CASE WHEN decision='optimize' THEN 1 ELSE 0 END), "
|
|
61
|
+
"COALESCE(SUM(before_input_tokens),0), COALESCE(SUM(after_input_tokens),0), "
|
|
62
|
+
"COALESCE(SUM(provider_input_tokens),0), COALESCE(SUM(provider_output_tokens),0), "
|
|
63
|
+
"COALESCE(SUM(cached_tokens),0) FROM events"
|
|
64
|
+
).fetchone()
|
|
65
|
+
requests, optimized, before, after, provider_in, provider_out, cached = row
|
|
66
|
+
saved = before - after
|
|
67
|
+
pct = round(saved * 100 / before, 1) if before else 0.0
|
|
68
|
+
return {
|
|
69
|
+
"requests": requests,
|
|
70
|
+
"optimized": optimized or 0,
|
|
71
|
+
"bypassed": requests - (optimized or 0),
|
|
72
|
+
"estimated_input_before": before,
|
|
73
|
+
"estimated_input_after": after,
|
|
74
|
+
"estimated_input_saved": saved,
|
|
75
|
+
"estimated_input_saved_pct": pct,
|
|
76
|
+
"provider_input_tokens": provider_in,
|
|
77
|
+
"provider_output_tokens": provider_out,
|
|
78
|
+
"cached_tokens": cached,
|
|
79
|
+
}
|
token_runtime/model.py
ADDED
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, field
|
|
4
|
+
from enum import Enum
|
|
5
|
+
from typing import Any, Mapping
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
class OptimizationDecision(str, Enum):
|
|
9
|
+
BYPASS = "bypass"
|
|
10
|
+
OPTIMIZE = "optimize"
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
@dataclass(frozen=True, slots=True)
|
|
14
|
+
class ContextBlock:
|
|
15
|
+
id: str
|
|
16
|
+
kind: str
|
|
17
|
+
text: str
|
|
18
|
+
role: str | None = None
|
|
19
|
+
turn_index: int = 0
|
|
20
|
+
metadata: Mapping[str, Any] = field(default_factory=dict)
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
@dataclass(frozen=True, slots=True)
|
|
24
|
+
class RequestEnvelope:
|
|
25
|
+
blocks: tuple[ContextBlock, ...]
|
|
26
|
+
opaque: Mapping[str, Any] = field(default_factory=dict)
|
|
27
|
+
adapter: str = "normalized"
|
|
28
|
+
wire_safe: bool = True
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass(frozen=True, slots=True)
|
|
32
|
+
class OptimizationResult:
|
|
33
|
+
decision: OptimizationDecision
|
|
34
|
+
envelope: RequestEnvelope
|
|
35
|
+
reasons: tuple[str, ...]
|
|
36
|
+
before_estimated_tokens: int
|
|
37
|
+
after_estimated_tokens: int
|
|
38
|
+
changed_block_ids: tuple[str, ...] = ()
|
|
39
|
+
recovery_refs: tuple[str, ...] = ()
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable, Mapping
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
from .adapters import ChatCompletionsAdapter, ResponsesAdapter
|
|
7
|
+
from .capabilities import (
|
|
8
|
+
CapabilityDetector,
|
|
9
|
+
CapabilityKey,
|
|
10
|
+
CapabilityProfile,
|
|
11
|
+
CapabilityRegistry,
|
|
12
|
+
)
|
|
13
|
+
from .compatibility import (
|
|
14
|
+
CompatibilityMatrix,
|
|
15
|
+
CompatibilityRecord,
|
|
16
|
+
CompatibilityRegistry,
|
|
17
|
+
CompatibilityState,
|
|
18
|
+
)
|
|
19
|
+
from .conformance import ConformanceResult, require_conformance, run_conformance
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
|
|
23
|
+
CODEX_RESPONSES_KEY = CapabilityKey(
|
|
24
|
+
"codex",
|
|
25
|
+
"responses",
|
|
26
|
+
"openai-compatible",
|
|
27
|
+
client_version="0.153.4",
|
|
28
|
+
)
|
|
29
|
+
OPENAI_RESPONSES_KEY = CapabilityKey(
|
|
30
|
+
"generic-openai",
|
|
31
|
+
"responses",
|
|
32
|
+
"openai-compatible",
|
|
33
|
+
)
|
|
34
|
+
OPENAI_CHAT_COMPLETIONS_KEY = CapabilityKey(
|
|
35
|
+
"generic-openai",
|
|
36
|
+
"chat_completions",
|
|
37
|
+
"openai-compatible",
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True, slots=True)
|
|
42
|
+
class CertificationEvidence:
|
|
43
|
+
evidence_id: str
|
|
44
|
+
version_scope: str
|
|
45
|
+
scope: str
|
|
46
|
+
assertion_ids: tuple[str, ...]
|
|
47
|
+
|
|
48
|
+
def to_primitive(self) -> dict[str, object]:
|
|
49
|
+
return {
|
|
50
|
+
"evidence_id": self.evidence_id,
|
|
51
|
+
"version_scope": self.version_scope,
|
|
52
|
+
"scope": self.scope,
|
|
53
|
+
"assertion_ids": list(self.assertion_ids),
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
@dataclass(frozen=True, slots=True)
|
|
58
|
+
class OpenAICertificationBundle:
|
|
59
|
+
profiles: tuple[CapabilityProfile, ...]
|
|
60
|
+
records: tuple[CompatibilityRecord, ...]
|
|
61
|
+
evidence: tuple[CertificationEvidence, ...]
|
|
62
|
+
detector: CapabilityDetector
|
|
63
|
+
matrix: CompatibilityMatrix
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _profile(key: CapabilityKey, evidence_id: str) -> CapabilityProfile:
|
|
67
|
+
return CapabilityProfile(
|
|
68
|
+
key=key,
|
|
69
|
+
tool_calls=True,
|
|
70
|
+
reasoning_state="opaque-preserved",
|
|
71
|
+
streaming=True,
|
|
72
|
+
exact_byte_preservation=True,
|
|
73
|
+
evidence_id=evidence_id,
|
|
74
|
+
)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def build_openai_certification() -> OpenAICertificationBundle:
|
|
78
|
+
definitions = (
|
|
79
|
+
(
|
|
80
|
+
CODEX_RESPONSES_KEY,
|
|
81
|
+
CertificationEvidence(
|
|
82
|
+
evidence_id="token-openai-cert-1:codex-responses-0.153.4",
|
|
83
|
+
version_scope="codex-cli-0.153.4",
|
|
84
|
+
scope="codex-responses/request-boundary",
|
|
85
|
+
assertion_ids=(
|
|
86
|
+
"codex-live-smoke-v1",
|
|
87
|
+
"responses-exact-roundtrip",
|
|
88
|
+
"unsupported-exact-passthrough",
|
|
89
|
+
),
|
|
90
|
+
),
|
|
91
|
+
),
|
|
92
|
+
(
|
|
93
|
+
OPENAI_CHAT_COMPLETIONS_KEY,
|
|
94
|
+
CertificationEvidence(
|
|
95
|
+
evidence_id="token-openai-cert-1:chat-completions-text-v1",
|
|
96
|
+
version_scope="token-request-contract-v1",
|
|
97
|
+
scope="openai-chat-completions/request-boundary-text-v1",
|
|
98
|
+
assertion_ids=(
|
|
99
|
+
"chat-completions-exact-roundtrip",
|
|
100
|
+
"chat-completions-unsupported-exact-passthrough",
|
|
101
|
+
),
|
|
102
|
+
),
|
|
103
|
+
),
|
|
104
|
+
(
|
|
105
|
+
OPENAI_RESPONSES_KEY,
|
|
106
|
+
CertificationEvidence(
|
|
107
|
+
evidence_id="token-openai-cert-1:responses-text-v1",
|
|
108
|
+
version_scope="token-request-contract-v1",
|
|
109
|
+
scope="openai-responses/request-boundary-text-v1",
|
|
110
|
+
assertion_ids=(
|
|
111
|
+
"responses-exact-roundtrip",
|
|
112
|
+
"responses-unsupported-exact-passthrough",
|
|
113
|
+
),
|
|
114
|
+
),
|
|
115
|
+
),
|
|
116
|
+
)
|
|
117
|
+
capabilities = CapabilityRegistry()
|
|
118
|
+
compatibility = CompatibilityRegistry()
|
|
119
|
+
evidence: list[CertificationEvidence] = []
|
|
120
|
+
|
|
121
|
+
for key, item in definitions:
|
|
122
|
+
capabilities.register(_profile(key, item.evidence_id))
|
|
123
|
+
compatibility.register(
|
|
124
|
+
CompatibilityRecord(
|
|
125
|
+
key=key,
|
|
126
|
+
state=CompatibilityState.CERTIFIED,
|
|
127
|
+
evidence_id=item.evidence_id,
|
|
128
|
+
reason="request_boundary_certified",
|
|
129
|
+
)
|
|
130
|
+
)
|
|
131
|
+
evidence.append(item)
|
|
132
|
+
|
|
133
|
+
return OpenAICertificationBundle(
|
|
134
|
+
profiles=capabilities.snapshot(),
|
|
135
|
+
records=compatibility.snapshot(),
|
|
136
|
+
evidence=tuple(sorted(evidence, key=lambda item: item.evidence_id)),
|
|
137
|
+
detector=CapabilityDetector(capabilities, compatibility),
|
|
138
|
+
matrix=compatibility.matrix(),
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _responses_fixture() -> dict[str, object]:
|
|
143
|
+
return {
|
|
144
|
+
"model": "opaque-model",
|
|
145
|
+
"instructions": "Preserve exact constraints.",
|
|
146
|
+
"input": [
|
|
147
|
+
{"role": "developer", "content": "stable policy"},
|
|
148
|
+
{
|
|
149
|
+
"role": "user",
|
|
150
|
+
"content": [{"type": "input_text", "text": "current task"}],
|
|
151
|
+
},
|
|
152
|
+
{
|
|
153
|
+
"type": "function_call",
|
|
154
|
+
"call_id": "c1",
|
|
155
|
+
"name": "query",
|
|
156
|
+
"arguments": '{"q":"status"}',
|
|
157
|
+
},
|
|
158
|
+
{
|
|
159
|
+
"type": "function_call_output",
|
|
160
|
+
"call_id": "c1",
|
|
161
|
+
"output": "status=green",
|
|
162
|
+
},
|
|
163
|
+
],
|
|
164
|
+
"tools": [
|
|
165
|
+
{
|
|
166
|
+
"type": "function",
|
|
167
|
+
"name": "query",
|
|
168
|
+
"description": "fixture",
|
|
169
|
+
"parameters": {"type": "object"},
|
|
170
|
+
}
|
|
171
|
+
],
|
|
172
|
+
"reasoning": {"effort": "medium"},
|
|
173
|
+
"stream": True,
|
|
174
|
+
"metadata": {"fixture": "c3"},
|
|
175
|
+
"future_field": {"preserve": True},
|
|
176
|
+
}
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
def _chat_completions_fixture() -> dict[str, object]:
|
|
180
|
+
return {
|
|
181
|
+
"model": "opaque-model",
|
|
182
|
+
"messages": [
|
|
183
|
+
{"role": "system", "content": "system"},
|
|
184
|
+
{"role": "developer", "content": "developer"},
|
|
185
|
+
{
|
|
186
|
+
"role": "assistant",
|
|
187
|
+
"content": "old",
|
|
188
|
+
"tool_calls": [
|
|
189
|
+
{
|
|
190
|
+
"id": "c1",
|
|
191
|
+
"type": "function",
|
|
192
|
+
"function": {"name": "query", "arguments": "{}"},
|
|
193
|
+
}
|
|
194
|
+
],
|
|
195
|
+
},
|
|
196
|
+
{"role": "tool", "tool_call_id": "c1", "content": "result"},
|
|
197
|
+
{
|
|
198
|
+
"role": "user",
|
|
199
|
+
"content": [{"type": "text", "text": "continue"}],
|
|
200
|
+
},
|
|
201
|
+
],
|
|
202
|
+
"tools": [
|
|
203
|
+
{
|
|
204
|
+
"type": "function",
|
|
205
|
+
"function": {"name": "query", "parameters": {"type": "object"}},
|
|
206
|
+
}
|
|
207
|
+
],
|
|
208
|
+
"response_format": {"type": "json_object"},
|
|
209
|
+
"stream": True,
|
|
210
|
+
"future_field": {"preserve": True},
|
|
211
|
+
}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def _responses_roundtrip() -> bool:
|
|
215
|
+
payload = _responses_fixture()
|
|
216
|
+
adapter = ResponsesAdapter()
|
|
217
|
+
envelope = adapter.parse(payload)
|
|
218
|
+
return envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _chat_completions_roundtrip() -> bool:
|
|
222
|
+
payload = _chat_completions_fixture()
|
|
223
|
+
adapter = ChatCompletionsAdapter()
|
|
224
|
+
envelope = adapter.parse(payload)
|
|
225
|
+
return envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
226
|
+
|
|
227
|
+
|
|
228
|
+
def _responses_unsupported_passthrough() -> bool:
|
|
229
|
+
payload = {
|
|
230
|
+
"model": "opaque-model",
|
|
231
|
+
"input": [
|
|
232
|
+
{
|
|
233
|
+
"role": "user",
|
|
234
|
+
"content": [{"type": "input_image", "image_url": "fixture://image"}],
|
|
235
|
+
}
|
|
236
|
+
],
|
|
237
|
+
"future_field": {"preserve": True},
|
|
238
|
+
}
|
|
239
|
+
adapter = ResponsesAdapter()
|
|
240
|
+
envelope = adapter.parse(payload)
|
|
241
|
+
return not envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def _chat_completions_unsupported_passthrough() -> bool:
|
|
245
|
+
payload = {
|
|
246
|
+
"model": "opaque-model",
|
|
247
|
+
"messages": [
|
|
248
|
+
{
|
|
249
|
+
"role": "user",
|
|
250
|
+
"content": [
|
|
251
|
+
{
|
|
252
|
+
"type": "image_url",
|
|
253
|
+
"image_url": {"url": "fixture://image"},
|
|
254
|
+
}
|
|
255
|
+
],
|
|
256
|
+
}
|
|
257
|
+
],
|
|
258
|
+
"future_field": {"preserve": True},
|
|
259
|
+
}
|
|
260
|
+
adapter = ChatCompletionsAdapter()
|
|
261
|
+
envelope = adapter.parse(payload)
|
|
262
|
+
return not envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
263
|
+
|
|
264
|
+
|
|
265
|
+
def _certification_matrix_conforms() -> bool:
|
|
266
|
+
bundle = build_openai_certification()
|
|
267
|
+
expected = (
|
|
268
|
+
CODEX_RESPONSES_KEY,
|
|
269
|
+
OPENAI_CHAT_COMPLETIONS_KEY,
|
|
270
|
+
OPENAI_RESPONSES_KEY,
|
|
271
|
+
)
|
|
272
|
+
return (
|
|
273
|
+
tuple(profile.key for profile in bundle.profiles) == expected
|
|
274
|
+
and tuple(record.key for record in bundle.records) == expected
|
|
275
|
+
and all(record.state is CompatibilityState.CERTIFIED for record in bundle.records)
|
|
276
|
+
)
|
|
277
|
+
|
|
278
|
+
|
|
279
|
+
def _unknown_key_passthrough() -> bool:
|
|
280
|
+
bundle = build_openai_certification()
|
|
281
|
+
unknown = bundle.detector.detect(
|
|
282
|
+
CapabilityKey(
|
|
283
|
+
"codex",
|
|
284
|
+
"responses",
|
|
285
|
+
"openai-compatible",
|
|
286
|
+
"future-model",
|
|
287
|
+
)
|
|
288
|
+
)
|
|
289
|
+
return (
|
|
290
|
+
unknown.profile is None
|
|
291
|
+
and unknown.compatibility.state is CompatibilityState.PASSTHROUGH_ONLY
|
|
292
|
+
and unknown.compatibility.reason == "unknown_capability"
|
|
293
|
+
)
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def openai_conformance_cases() -> Mapping[str, Callable[[], bool]]:
|
|
297
|
+
return {
|
|
298
|
+
"certification_matrix": _certification_matrix_conforms,
|
|
299
|
+
"chat_completions_roundtrip": _chat_completions_roundtrip,
|
|
300
|
+
"chat_completions_unsupported_passthrough": _chat_completions_unsupported_passthrough,
|
|
301
|
+
"responses_roundtrip": _responses_roundtrip,
|
|
302
|
+
"responses_unsupported_passthrough": _responses_unsupported_passthrough,
|
|
303
|
+
"unknown_key_passthrough": _unknown_key_passthrough,
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
|
|
307
|
+
def run_openai_conformance() -> tuple[ConformanceResult, ...]:
|
|
308
|
+
results = run_conformance(openai_conformance_cases())
|
|
309
|
+
require_conformance(results)
|
|
310
|
+
return results
|
token_runtime/planner.py
ADDED
|
@@ -0,0 +1,96 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from .model import OptimizationDecision, RequestEnvelope
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
_DECISION_RE = re.compile(r"(?im)^\s*DECISION\s*:\s*(.+?)\s*$")
|
|
11
|
+
_CONSTRAINT_RE = re.compile(r"\b(must|never|do not|don't|constraint|required)\b", re.I)
|
|
12
|
+
_EXACT_EDIT_MARKERS = ("old_string=", "new_string=", "diff --git", "*** Begin Patch")
|
|
13
|
+
_PROTOCOL_KINDS = {
|
|
14
|
+
"system", "developer", "tools", "tool_schema", "tool_call", "function_call"
|
|
15
|
+
}
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _is_structured_json(text: str) -> bool:
|
|
19
|
+
try:
|
|
20
|
+
value = json.loads(text)
|
|
21
|
+
except (json.JSONDecodeError, TypeError):
|
|
22
|
+
return False
|
|
23
|
+
return isinstance(value, (dict, list))
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class Plan:
|
|
28
|
+
decision: OptimizationDecision
|
|
29
|
+
protected_ids: frozenset[str]
|
|
30
|
+
reasons: tuple[str, ...]
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class ContextPlanner:
|
|
34
|
+
def plan(self, envelope: RequestEnvelope) -> Plan:
|
|
35
|
+
protected: set[str] = set()
|
|
36
|
+
reasons: list[str] = []
|
|
37
|
+
latest_turn = max((block.turn_index for block in envelope.blocks), default=0)
|
|
38
|
+
|
|
39
|
+
for block in envelope.blocks:
|
|
40
|
+
if block.kind in _PROTOCOL_KINDS:
|
|
41
|
+
protected.add(block.id)
|
|
42
|
+
if _CONSTRAINT_RE.search(block.text):
|
|
43
|
+
protected.add(block.id)
|
|
44
|
+
if block.turn_index == latest_turn:
|
|
45
|
+
protected.add(block.id)
|
|
46
|
+
|
|
47
|
+
self._protect_latest(envelope, "user", protected)
|
|
48
|
+
self._protect_latest(envelope, "tool_output", protected)
|
|
49
|
+
|
|
50
|
+
active_decisions: list[str] = []
|
|
51
|
+
for block in envelope.blocks:
|
|
52
|
+
matches = _DECISION_RE.findall(block.text)
|
|
53
|
+
if matches:
|
|
54
|
+
protected.add(block.id)
|
|
55
|
+
if block.turn_index != latest_turn or block.kind in _PROTOCOL_KINDS:
|
|
56
|
+
continue
|
|
57
|
+
for match in matches:
|
|
58
|
+
active_decisions.append(" ".join(match.lower().split()))
|
|
59
|
+
|
|
60
|
+
if not envelope.wire_safe:
|
|
61
|
+
reasons.append("unsupported_wire_shape")
|
|
62
|
+
if any(
|
|
63
|
+
marker in block.text
|
|
64
|
+
for block in envelope.blocks
|
|
65
|
+
for marker in _EXACT_EDIT_MARKERS
|
|
66
|
+
):
|
|
67
|
+
reasons.append("exact_edit_context")
|
|
68
|
+
if len(set(active_decisions)) > 1:
|
|
69
|
+
reasons.append("competing_active_decisions")
|
|
70
|
+
has_decision_context = any(_DECISION_RE.search(block.text) for block in envelope.blocks)
|
|
71
|
+
has_structured_evidence = any(
|
|
72
|
+
block.kind == "tool_output" and _is_structured_json(block.text)
|
|
73
|
+
for block in envelope.blocks
|
|
74
|
+
)
|
|
75
|
+
if has_decision_context and has_structured_evidence:
|
|
76
|
+
reasons.append("structured_decision_context")
|
|
77
|
+
|
|
78
|
+
if reasons:
|
|
79
|
+
return Plan(OptimizationDecision.BYPASS, frozenset(protected), tuple(reasons))
|
|
80
|
+
return Plan(
|
|
81
|
+
OptimizationDecision.OPTIMIZE,
|
|
82
|
+
frozenset(protected),
|
|
83
|
+
("eligible_request",),
|
|
84
|
+
)
|
|
85
|
+
|
|
86
|
+
@staticmethod
|
|
87
|
+
def _protect_latest(
|
|
88
|
+
envelope: RequestEnvelope, kind: str, protected: set[str]
|
|
89
|
+
) -> None:
|
|
90
|
+
candidates = [
|
|
91
|
+
(block.turn_index, index, block.id)
|
|
92
|
+
for index, block in enumerate(envelope.blocks)
|
|
93
|
+
if block.kind == kind
|
|
94
|
+
]
|
|
95
|
+
if candidates:
|
|
96
|
+
protected.add(max(candidates)[2])
|
|
@@ -0,0 +1,179 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, replace
|
|
4
|
+
import json
|
|
5
|
+
import re
|
|
6
|
+
|
|
7
|
+
from .model import ContextBlock
|
|
8
|
+
from .store import RecoveryStore
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
_ANCHOR_RE = re.compile(
|
|
12
|
+
r"\b(error|failed|failure|exception|warning|must|never|constraint|decision|status|exit code)\b",
|
|
13
|
+
re.I,
|
|
14
|
+
)
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True, slots=True)
|
|
18
|
+
class Reduction:
|
|
19
|
+
block: ContextBlock
|
|
20
|
+
changed: bool
|
|
21
|
+
reducer_id: str | None = None
|
|
22
|
+
recovery_ref: str | None = None
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def _unchanged(block: ContextBlock) -> Reduction:
|
|
26
|
+
return Reduction(block=block, changed=False)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def _is_json_container(text: str) -> bool:
|
|
30
|
+
try:
|
|
31
|
+
value = json.loads(text)
|
|
32
|
+
except (json.JSONDecodeError, TypeError):
|
|
33
|
+
return False
|
|
34
|
+
return isinstance(value, (dict, list))
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class RepeatedLineReducer:
|
|
38
|
+
reducer_id = "repeat_runs_v1"
|
|
39
|
+
|
|
40
|
+
def reduce(self, block: ContextBlock, store: RecoveryStore, *, protected: bool) -> Reduction:
|
|
41
|
+
if protected or _is_json_container(block.text):
|
|
42
|
+
return _unchanged(block)
|
|
43
|
+
lines = block.text.splitlines()
|
|
44
|
+
if len(lines) < 2:
|
|
45
|
+
return _unchanged(block)
|
|
46
|
+
|
|
47
|
+
out: list[str] = []
|
|
48
|
+
current = lines[0]
|
|
49
|
+
count = 1
|
|
50
|
+
for line in lines[1:]:
|
|
51
|
+
if line == current:
|
|
52
|
+
count += 1
|
|
53
|
+
continue
|
|
54
|
+
out.append(current if count == 1 else f"{current} [repeated x{count}]")
|
|
55
|
+
current, count = line, 1
|
|
56
|
+
out.append(current if count == 1 else f"{current} [repeated x{count}]")
|
|
57
|
+
candidate = "\n".join(out)
|
|
58
|
+
if candidate == block.text or len(candidate) >= len(block.text):
|
|
59
|
+
return _unchanged(block)
|
|
60
|
+
ref = store.put(block.text.encode())
|
|
61
|
+
return Reduction(
|
|
62
|
+
block=replace(block, text=candidate),
|
|
63
|
+
changed=True,
|
|
64
|
+
reducer_id=self.reducer_id,
|
|
65
|
+
recovery_ref=ref,
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
class JsonToolOutputReducer:
|
|
70
|
+
reducer_id = "json_minify_v1"
|
|
71
|
+
|
|
72
|
+
def reduce(self, block: ContextBlock, store: RecoveryStore, *, protected: bool) -> Reduction:
|
|
73
|
+
if protected or block.kind != "tool_output" or not _is_json_container(block.text):
|
|
74
|
+
return _unchanged(block)
|
|
75
|
+
value = json.loads(block.text)
|
|
76
|
+
candidate = json.dumps(value, ensure_ascii=False, separators=(",", ":"))
|
|
77
|
+
if len(candidate) >= len(block.text):
|
|
78
|
+
return _unchanged(block)
|
|
79
|
+
ref = store.put(block.text.encode())
|
|
80
|
+
return Reduction(block=replace(block, text=candidate), changed=True,
|
|
81
|
+
reducer_id=self.reducer_id, recovery_ref=ref)
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
class ToolOutputReducer:
|
|
85
|
+
reducer_id = "tool_output_v1"
|
|
86
|
+
|
|
87
|
+
def __init__(self, min_chars: int = 1800, edge_lines: int = 8):
|
|
88
|
+
self.min_chars = min_chars
|
|
89
|
+
self.edge_lines = edge_lines
|
|
90
|
+
|
|
91
|
+
def reduce(self, block: ContextBlock, store: RecoveryStore, *, protected: bool) -> Reduction:
|
|
92
|
+
if (protected or block.kind != "tool_output" or len(block.text) < self.min_chars
|
|
93
|
+
or _is_json_container(block.text)):
|
|
94
|
+
return _unchanged(block)
|
|
95
|
+
|
|
96
|
+
lines = block.text.splitlines()
|
|
97
|
+
edge = self.edge_lines
|
|
98
|
+
if len(lines) <= edge * 2 + 1:
|
|
99
|
+
return _unchanged(block)
|
|
100
|
+
middle = lines[edge:-edge]
|
|
101
|
+
anchors = [line for line in middle if _ANCHOR_RE.search(line)]
|
|
102
|
+
ref = store.put(block.text.encode())
|
|
103
|
+
marker = f"[TOKEN_REF:{ref}]"
|
|
104
|
+
candidate_lines = lines[:edge] + anchors + [marker] + lines[-edge:]
|
|
105
|
+
candidate = "\n".join(dict.fromkeys(candidate_lines))
|
|
106
|
+
if len(candidate) >= len(block.text):
|
|
107
|
+
return _unchanged(block)
|
|
108
|
+
return Reduction(
|
|
109
|
+
block=replace(block, text=candidate),
|
|
110
|
+
changed=True,
|
|
111
|
+
reducer_id=self.reducer_id,
|
|
112
|
+
recovery_ref=ref,
|
|
113
|
+
)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
_EXPLICIT_GUARD_RE = re.compile(
|
|
117
|
+
r"(?im)^\s*DECISION\s*:|\b(must|never|do not|don't|required|constraint)\b"
|
|
118
|
+
)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
class RetrievedDuplicateReducer:
|
|
122
|
+
"""Remove long exact adjacent duplicate spans from old retrieval/tool evidence."""
|
|
123
|
+
|
|
124
|
+
reducer_id = "evidence_exact_duplicate_v1"
|
|
125
|
+
|
|
126
|
+
def __init__(self, min_words: int = 8):
|
|
127
|
+
self.min_words = min_words
|
|
128
|
+
|
|
129
|
+
def reduce(self, block: ContextBlock, store: RecoveryStore, *, protected: bool) -> Reduction:
|
|
130
|
+
if (protected or block.kind not in {"retrieved", "tool_output"}
|
|
131
|
+
or _EXPLICIT_GUARD_RE.search(block.text) or _is_json_container(block.text)):
|
|
132
|
+
return _unchanged(block)
|
|
133
|
+
candidate = self._collapse_text(block.text)
|
|
134
|
+
if candidate == block.text or len(candidate) >= len(block.text):
|
|
135
|
+
return _unchanged(block)
|
|
136
|
+
ref = store.put(block.text.encode())
|
|
137
|
+
return Reduction(
|
|
138
|
+
block=replace(block, text=candidate),
|
|
139
|
+
changed=True,
|
|
140
|
+
reducer_id=self.reducer_id,
|
|
141
|
+
recovery_ref=ref,
|
|
142
|
+
)
|
|
143
|
+
|
|
144
|
+
def _collapse_text(self, text: str) -> str:
|
|
145
|
+
lines = text.splitlines(keepends=True)
|
|
146
|
+
return "".join(self._collapse_line(line) for line in lines)
|
|
147
|
+
|
|
148
|
+
def _collapse_line(self, line: str) -> str:
|
|
149
|
+
newline = ""
|
|
150
|
+
body = line
|
|
151
|
+
if body.endswith("\r\n"):
|
|
152
|
+
body, newline = body[:-2], "\r\n"
|
|
153
|
+
elif body.endswith("\n"):
|
|
154
|
+
body, newline = body[:-1], "\n"
|
|
155
|
+
|
|
156
|
+
while True:
|
|
157
|
+
match = self._longest_adjacent_repeat(body)
|
|
158
|
+
if match is None:
|
|
159
|
+
return body + newline
|
|
160
|
+
start_second, end_second = match
|
|
161
|
+
left = body[:start_second].rstrip()
|
|
162
|
+
right = body[end_second:]
|
|
163
|
+
body = left + right
|
|
164
|
+
|
|
165
|
+
def _longest_adjacent_repeat(self, text: str) -> tuple[int, int] | None:
|
|
166
|
+
words = list(re.finditer(r"\S+", text))
|
|
167
|
+
tokens = [item.group(0) for item in words]
|
|
168
|
+
best: tuple[int, int, int] | None = None
|
|
169
|
+
for start in range(len(tokens)):
|
|
170
|
+
max_span = (len(tokens) - start) // 2
|
|
171
|
+
for span in range(max_span, self.min_words - 1, -1):
|
|
172
|
+
if tokens[start:start + span] != tokens[start + span:start + 2 * span]:
|
|
173
|
+
continue
|
|
174
|
+
second_start = words[start + span].start()
|
|
175
|
+
second_end = words[start + 2 * span - 1].end()
|
|
176
|
+
if best is None or span > best[0]:
|
|
177
|
+
best = (span, second_start, second_end)
|
|
178
|
+
break
|
|
179
|
+
return None if best is None else (best[1], best[2])
|