token-runtime 0.1.0a2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- token_runtime/__init__.py +1 -0
- token_runtime/adapters.py +181 -0
- token_runtime/agent_integrations.py +114 -0
- token_runtime/anthropic_adapter.py +301 -0
- token_runtime/anthropic_conformance.py +336 -0
- token_runtime/anthropic_preserved_thinking_cert.py +213 -0
- token_runtime/benchmark.py +106 -0
- token_runtime/capabilities.py +111 -0
- token_runtime/cli.py +284 -0
- token_runtime/codex_recertification.py +115 -0
- token_runtime/compat_cert.py +462 -0
- token_runtime/compatibility.py +81 -0
- token_runtime/config.py +72 -0
- token_runtime/conformance.py +36 -0
- token_runtime/contracts.py +70 -0
- token_runtime/engine.py +136 -0
- token_runtime/feature_flags.py +78 -0
- token_runtime/gateway.py +151 -0
- token_runtime/gemini_adapter.py +260 -0
- token_runtime/gemini_conformance.py +351 -0
- token_runtime/integrations.py +272 -0
- token_runtime/metrics.py +79 -0
- token_runtime/model.py +39 -0
- token_runtime/openai_certification.py +310 -0
- token_runtime/planner.py +96 -0
- token_runtime/reducers.py +179 -0
- token_runtime/store.py +36 -0
- token_runtime/strategies.py +50 -0
- token_runtime/terminal_ui.py +96 -0
- token_runtime-0.1.0a2.dist-info/METADATA +238 -0
- token_runtime-0.1.0a2.dist-info/RECORD +35 -0
- token_runtime-0.1.0a2.dist-info/WHEEL +5 -0
- token_runtime-0.1.0a2.dist-info/entry_points.txt +2 -0
- token_runtime-0.1.0a2.dist-info/licenses/LICENSE +202 -0
- token_runtime-0.1.0a2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,336 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from collections.abc import Callable, Mapping
|
|
4
|
+
from dataclasses import dataclass, replace
|
|
5
|
+
|
|
6
|
+
from .anthropic_adapter import AnthropicMessagesAdapter
|
|
7
|
+
from .capabilities import (
|
|
8
|
+
CapabilityDetector,
|
|
9
|
+
CapabilityKey,
|
|
10
|
+
CapabilityProfile,
|
|
11
|
+
CapabilityRegistry,
|
|
12
|
+
)
|
|
13
|
+
from .compatibility import (
|
|
14
|
+
CompatibilityRecord,
|
|
15
|
+
CompatibilityRegistry,
|
|
16
|
+
CompatibilityState,
|
|
17
|
+
)
|
|
18
|
+
from .conformance import ConformanceResult, require_conformance, run_conformance
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
ANTHROPIC_MESSAGES_KEY = CapabilityKey(
|
|
22
|
+
"generic-anthropic",
|
|
23
|
+
"anthropic_messages",
|
|
24
|
+
"anthropic",
|
|
25
|
+
)
|
|
26
|
+
EVIDENCE_ID = "token-anthropic-adapter-1:offline-request-boundary-v1"
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass(frozen=True, slots=True)
|
|
30
|
+
class AnthropicConformanceBundle:
|
|
31
|
+
profile: CapabilityProfile
|
|
32
|
+
record: CompatibilityRecord
|
|
33
|
+
detector: CapabilityDetector
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def build_anthropic_conformance() -> AnthropicConformanceBundle:
|
|
37
|
+
profile = CapabilityProfile(
|
|
38
|
+
key=ANTHROPIC_MESSAGES_KEY,
|
|
39
|
+
prompt_cache="request-markers-preserved",
|
|
40
|
+
tool_calls=True,
|
|
41
|
+
reasoning_state="exact-replay-required",
|
|
42
|
+
exact_byte_preservation=True,
|
|
43
|
+
evidence_id=EVIDENCE_ID,
|
|
44
|
+
)
|
|
45
|
+
record = CompatibilityRecord(
|
|
46
|
+
key=ANTHROPIC_MESSAGES_KEY,
|
|
47
|
+
state=CompatibilityState.PASSTHROUGH_ONLY,
|
|
48
|
+
evidence_id=EVIDENCE_ID,
|
|
49
|
+
reason="offline_conformance_only",
|
|
50
|
+
)
|
|
51
|
+
capabilities = CapabilityRegistry()
|
|
52
|
+
capabilities.register(profile)
|
|
53
|
+
compatibility = CompatibilityRegistry()
|
|
54
|
+
compatibility.register(record)
|
|
55
|
+
return AnthropicConformanceBundle(
|
|
56
|
+
profile=profile,
|
|
57
|
+
record=record,
|
|
58
|
+
detector=CapabilityDetector(capabilities, compatibility),
|
|
59
|
+
)
|
|
60
|
+
|
|
61
|
+
|
|
62
|
+
def _supported_fixture() -> dict[str, object]:
|
|
63
|
+
return {
|
|
64
|
+
"model": "opaque-model",
|
|
65
|
+
"max_tokens": 256,
|
|
66
|
+
"system": [
|
|
67
|
+
{
|
|
68
|
+
"type": "text",
|
|
69
|
+
"text": "stable policy",
|
|
70
|
+
"cache_control": {"type": "ephemeral"},
|
|
71
|
+
}
|
|
72
|
+
],
|
|
73
|
+
"messages": [
|
|
74
|
+
{"role": "user", "content": "current task"},
|
|
75
|
+
{
|
|
76
|
+
"role": "assistant",
|
|
77
|
+
"content": [
|
|
78
|
+
{
|
|
79
|
+
"type": "tool_use",
|
|
80
|
+
"id": "toolu_1",
|
|
81
|
+
"name": "query",
|
|
82
|
+
"input": {"q": "status"},
|
|
83
|
+
}
|
|
84
|
+
],
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"role": "user",
|
|
88
|
+
"content": [
|
|
89
|
+
{
|
|
90
|
+
"type": "tool_result",
|
|
91
|
+
"tool_use_id": "toolu_1",
|
|
92
|
+
"content": "status=green",
|
|
93
|
+
}
|
|
94
|
+
],
|
|
95
|
+
},
|
|
96
|
+
],
|
|
97
|
+
"tools": [
|
|
98
|
+
{
|
|
99
|
+
"name": "query",
|
|
100
|
+
"description": "fixture",
|
|
101
|
+
"input_schema": {"type": "object"},
|
|
102
|
+
"cache_control": {"type": "ephemeral"},
|
|
103
|
+
}
|
|
104
|
+
],
|
|
105
|
+
"stream": True,
|
|
106
|
+
"future_field": {"preserve": True},
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _supported_roundtrip() -> bool:
|
|
111
|
+
payload = _supported_fixture()
|
|
112
|
+
adapter = AnthropicMessagesAdapter()
|
|
113
|
+
envelope = adapter.parse(payload)
|
|
114
|
+
return envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _cache_markers_preserved() -> bool:
|
|
118
|
+
payload = _supported_fixture()
|
|
119
|
+
adapter = AnthropicMessagesAdapter()
|
|
120
|
+
envelope = adapter.parse(payload)
|
|
121
|
+
serialized = adapter.serialize(envelope)
|
|
122
|
+
return (
|
|
123
|
+
serialized == payload
|
|
124
|
+
and serialized["system"][0]["cache_control"] == {"type": "ephemeral"}
|
|
125
|
+
and serialized["tools"][0]["cache_control"] == {"type": "ephemeral"}
|
|
126
|
+
)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
|
|
130
|
+
def _text_path_mutation_preserves_protocol() -> bool:
|
|
131
|
+
payload = _supported_fixture()
|
|
132
|
+
adapter = AnthropicMessagesAdapter()
|
|
133
|
+
envelope = adapter.parse(payload)
|
|
134
|
+
target = next(block for block in envelope.blocks if block.text == "current task")
|
|
135
|
+
changed = replace(
|
|
136
|
+
envelope,
|
|
137
|
+
blocks=tuple(
|
|
138
|
+
replace(block, text="updated task") if block is target else block
|
|
139
|
+
for block in envelope.blocks
|
|
140
|
+
),
|
|
141
|
+
)
|
|
142
|
+
serialized = adapter.serialize(changed)
|
|
143
|
+
return (
|
|
144
|
+
serialized["messages"][0]["content"] == "updated task"
|
|
145
|
+
and serialized["messages"][1:] == payload["messages"][1:]
|
|
146
|
+
and serialized["system"] == payload["system"]
|
|
147
|
+
and serialized["tools"] == payload["tools"]
|
|
148
|
+
and serialized["future_field"] == payload["future_field"]
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def _tool_classification() -> bool:
|
|
153
|
+
adapter = AnthropicMessagesAdapter()
|
|
154
|
+
envelope = adapter.parse(_supported_fixture())
|
|
155
|
+
tool_call = next(block for block in envelope.blocks if block.kind == "tool_call")
|
|
156
|
+
tool_output = next(block for block in envelope.blocks if block.kind == "tool_output")
|
|
157
|
+
return (
|
|
158
|
+
envelope.wire_safe
|
|
159
|
+
and "path" not in tool_call.metadata
|
|
160
|
+
and tool_output.role == "tool"
|
|
161
|
+
and tool_output.metadata.get("path")
|
|
162
|
+
== ("messages", 2, "content", 0, "content")
|
|
163
|
+
)
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def _stream_preserved() -> bool:
|
|
167
|
+
payload = _supported_fixture()
|
|
168
|
+
adapter = AnthropicMessagesAdapter()
|
|
169
|
+
envelope = adapter.parse(payload)
|
|
170
|
+
serialized = adapter.serialize(envelope)
|
|
171
|
+
return envelope.wire_safe and serialized.get("stream") is True
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _near_match_non_inheritance() -> bool:
|
|
175
|
+
bundle = build_anthropic_conformance()
|
|
176
|
+
near_matches = (
|
|
177
|
+
CapabilityKey(
|
|
178
|
+
"generic-anthropic",
|
|
179
|
+
"anthropic_messages",
|
|
180
|
+
"anthropic",
|
|
181
|
+
model_family="claude-future",
|
|
182
|
+
),
|
|
183
|
+
CapabilityKey("generic-anthropic", "anthropic_messages", "anthropic-v2"),
|
|
184
|
+
)
|
|
185
|
+
for key in near_matches:
|
|
186
|
+
detected = bundle.detector.detect(key)
|
|
187
|
+
if detected.profile is not None:
|
|
188
|
+
return False
|
|
189
|
+
if detected.compatibility.state is not CompatibilityState.PASSTHROUGH_ONLY:
|
|
190
|
+
return False
|
|
191
|
+
if detected.compatibility.reason != "unknown_capability":
|
|
192
|
+
return False
|
|
193
|
+
return True
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _cited_text_exact_passthrough() -> bool:
|
|
197
|
+
payload = {
|
|
198
|
+
"model": "opaque-model",
|
|
199
|
+
"max_tokens": 128,
|
|
200
|
+
"system": [
|
|
201
|
+
{
|
|
202
|
+
"type": "text",
|
|
203
|
+
"text": "cited system",
|
|
204
|
+
"citations": [{"type": "char_location", "cited_text": "cited"}],
|
|
205
|
+
}
|
|
206
|
+
],
|
|
207
|
+
"messages": [
|
|
208
|
+
{
|
|
209
|
+
"role": "user",
|
|
210
|
+
"content": [
|
|
211
|
+
{
|
|
212
|
+
"type": "text",
|
|
213
|
+
"text": "cited message",
|
|
214
|
+
"citations": [
|
|
215
|
+
{"type": "char_location", "cited_text": "cited"}
|
|
216
|
+
],
|
|
217
|
+
}
|
|
218
|
+
],
|
|
219
|
+
},
|
|
220
|
+
{
|
|
221
|
+
"role": "assistant",
|
|
222
|
+
"content": [
|
|
223
|
+
{
|
|
224
|
+
"type": "tool_use",
|
|
225
|
+
"id": "toolu_cited",
|
|
226
|
+
"name": "query",
|
|
227
|
+
"input": {},
|
|
228
|
+
}
|
|
229
|
+
],
|
|
230
|
+
},
|
|
231
|
+
{
|
|
232
|
+
"role": "user",
|
|
233
|
+
"content": [
|
|
234
|
+
{
|
|
235
|
+
"type": "tool_result",
|
|
236
|
+
"tool_use_id": "toolu_cited",
|
|
237
|
+
"content": [
|
|
238
|
+
{
|
|
239
|
+
"type": "text",
|
|
240
|
+
"text": "cited tool result",
|
|
241
|
+
"citations": [
|
|
242
|
+
{
|
|
243
|
+
"type": "char_location",
|
|
244
|
+
"cited_text": "cited",
|
|
245
|
+
}
|
|
246
|
+
],
|
|
247
|
+
}
|
|
248
|
+
],
|
|
249
|
+
}
|
|
250
|
+
],
|
|
251
|
+
},
|
|
252
|
+
],
|
|
253
|
+
}
|
|
254
|
+
adapter = AnthropicMessagesAdapter()
|
|
255
|
+
envelope = adapter.parse(payload)
|
|
256
|
+
cited_blocks = [block for block in envelope.blocks if block.text.startswith("cited")]
|
|
257
|
+
return (
|
|
258
|
+
not envelope.wire_safe
|
|
259
|
+
and len(cited_blocks) == 3
|
|
260
|
+
and all("path" not in block.metadata for block in cited_blocks)
|
|
261
|
+
and adapter.serialize(envelope) == payload
|
|
262
|
+
)
|
|
263
|
+
|
|
264
|
+
def _thinking_exact_passthrough() -> bool:
|
|
265
|
+
payload = {
|
|
266
|
+
"model": "opaque-model",
|
|
267
|
+
"max_tokens": 64,
|
|
268
|
+
"messages": [
|
|
269
|
+
{
|
|
270
|
+
"role": "assistant",
|
|
271
|
+
"content": [
|
|
272
|
+
{"type": "thinking", "thinking": "opaque", "signature": "sig"}
|
|
273
|
+
],
|
|
274
|
+
},
|
|
275
|
+
{"role": "user", "content": "continue"},
|
|
276
|
+
],
|
|
277
|
+
}
|
|
278
|
+
adapter = AnthropicMessagesAdapter()
|
|
279
|
+
envelope = adapter.parse(payload)
|
|
280
|
+
return not envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
def _unknown_multimodal_passthrough() -> bool:
|
|
284
|
+
payload = {
|
|
285
|
+
"model": "opaque-model",
|
|
286
|
+
"max_tokens": 64,
|
|
287
|
+
"messages": [
|
|
288
|
+
{
|
|
289
|
+
"role": "user",
|
|
290
|
+
"content": [
|
|
291
|
+
{
|
|
292
|
+
"type": "image",
|
|
293
|
+
"source": {
|
|
294
|
+
"type": "base64",
|
|
295
|
+
"media_type": "image/png",
|
|
296
|
+
"data": "fixture",
|
|
297
|
+
},
|
|
298
|
+
}
|
|
299
|
+
],
|
|
300
|
+
}
|
|
301
|
+
],
|
|
302
|
+
"future_field": {"preserve": True},
|
|
303
|
+
}
|
|
304
|
+
adapter = AnthropicMessagesAdapter()
|
|
305
|
+
envelope = adapter.parse(payload)
|
|
306
|
+
return not envelope.wire_safe and adapter.serialize(envelope) == payload
|
|
307
|
+
|
|
308
|
+
|
|
309
|
+
def _certification_state_passthrough_only() -> bool:
|
|
310
|
+
bundle = build_anthropic_conformance()
|
|
311
|
+
return (
|
|
312
|
+
bundle.record.state is CompatibilityState.PASSTHROUGH_ONLY
|
|
313
|
+
and bundle.record.reason == "offline_conformance_only"
|
|
314
|
+
and bundle.profile.evidence_id == bundle.record.evidence_id
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def anthropic_conformance_cases() -> Mapping[str, Callable[[], bool]]:
|
|
319
|
+
return {
|
|
320
|
+
"cache_markers_preserved": _cache_markers_preserved,
|
|
321
|
+
"certification_state_passthrough_only": _certification_state_passthrough_only,
|
|
322
|
+
"cited_text_exact_passthrough": _cited_text_exact_passthrough,
|
|
323
|
+
"near_match_non_inheritance": _near_match_non_inheritance,
|
|
324
|
+
"stream_preserved": _stream_preserved,
|
|
325
|
+
"supported_roundtrip": _supported_roundtrip,
|
|
326
|
+
"text_path_mutation_preserves_protocol": _text_path_mutation_preserves_protocol,
|
|
327
|
+
"thinking_exact_passthrough": _thinking_exact_passthrough,
|
|
328
|
+
"tool_classification": _tool_classification,
|
|
329
|
+
"unknown_multimodal_passthrough": _unknown_multimodal_passthrough,
|
|
330
|
+
}
|
|
331
|
+
|
|
332
|
+
|
|
333
|
+
def run_anthropic_conformance() -> tuple[ConformanceResult, ...]:
|
|
334
|
+
results = run_conformance(anthropic_conformance_cases())
|
|
335
|
+
require_conformance(results)
|
|
336
|
+
return results
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass, fields
|
|
4
|
+
from decimal import Decimal, InvalidOperation
|
|
5
|
+
from hashlib import sha256
|
|
6
|
+
import json
|
|
7
|
+
|
|
8
|
+
from .capabilities import CapabilityKey
|
|
9
|
+
from .compat_cert import BenchmarkGeneration
|
|
10
|
+
from .compatibility import CompatibilityRecord, CompatibilityState
|
|
11
|
+
|
|
12
|
+
|
|
13
|
+
MODEL_ID = "claude-fable-5-1"
|
|
14
|
+
BINDING_BETA = "thinking-binding-controls-2026-08-01"
|
|
15
|
+
BINDING_MODE = "drop_block"
|
|
16
|
+
GENERATION_ID = "token-anthropic-preserved-thinking-cert-1:pt-v1"
|
|
17
|
+
CORPUS_VERSION = "anthropic-preserved-thinking-v1"
|
|
18
|
+
EVALUATOR_VERSION = "anthropic-pt-eval-v2"
|
|
19
|
+
POLICY_VERSION = "compat-policy-v1"
|
|
20
|
+
ACCOUNTING_STATES = frozenset({
|
|
21
|
+
"unavailable",
|
|
22
|
+
"estimated_from_response_usage",
|
|
23
|
+
"usage_reconciled",
|
|
24
|
+
"billed_workspace_delta_reconciled",
|
|
25
|
+
})
|
|
26
|
+
|
|
27
|
+
ANTHROPIC_PRESERVED_THINKING_KEY = CapabilityKey(
|
|
28
|
+
"generic-anthropic", "anthropic_messages", "anthropic", model_family=MODEL_ID
|
|
29
|
+
)
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
def _require_evidence_id(value: str) -> None:
|
|
33
|
+
if not isinstance(value, str) or not value.strip():
|
|
34
|
+
raise ValueError("evidence_id must be non-blank")
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _require_bool_fields(instance: object, names: tuple[str, ...]) -> None:
|
|
38
|
+
values = {item.name: getattr(instance, item.name) for item in fields(instance)}
|
|
39
|
+
for name in names:
|
|
40
|
+
if type(values[name]) is not bool:
|
|
41
|
+
raise TypeError(f"{name} must be bool")
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def _require_nonnegative_finite_decimal(value: str, name: str) -> Decimal:
|
|
45
|
+
if not isinstance(value, str):
|
|
46
|
+
raise ValueError(f"{name} must be a decimal string")
|
|
47
|
+
try:
|
|
48
|
+
parsed = Decimal(value)
|
|
49
|
+
except (InvalidOperation, ValueError) as exc:
|
|
50
|
+
raise ValueError(f"{name} must be a decimal string") from exc
|
|
51
|
+
if not parsed.is_finite() or parsed < 0:
|
|
52
|
+
raise ValueError(f"{name} must be finite and non-negative")
|
|
53
|
+
return parsed
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
@dataclass(frozen=True, slots=True)
|
|
57
|
+
class AnthropicPreservedThinkingOfflineEvidence:
|
|
58
|
+
evidence_id: str
|
|
59
|
+
inventory_complete: bool
|
|
60
|
+
model_scope_exact: bool
|
|
61
|
+
hard_bypass_complete: bool
|
|
62
|
+
thinking_stripped_control_reducible: bool
|
|
63
|
+
protected_prefix_exact: bool
|
|
64
|
+
token_off_on_wire_equivalent: bool
|
|
65
|
+
tool_history_fidelity: bool
|
|
66
|
+
cache_metadata_preserved: bool
|
|
67
|
+
resume_continuity: bool
|
|
68
|
+
unknown_native_passthrough: bool
|
|
69
|
+
|
|
70
|
+
def __post_init__(self) -> None:
|
|
71
|
+
_require_evidence_id(self.evidence_id)
|
|
72
|
+
_require_bool_fields(
|
|
73
|
+
self,
|
|
74
|
+
(
|
|
75
|
+
"inventory_complete",
|
|
76
|
+
"model_scope_exact",
|
|
77
|
+
"hard_bypass_complete",
|
|
78
|
+
"thinking_stripped_control_reducible",
|
|
79
|
+
"protected_prefix_exact",
|
|
80
|
+
"token_off_on_wire_equivalent",
|
|
81
|
+
"tool_history_fidelity",
|
|
82
|
+
"cache_metadata_preserved",
|
|
83
|
+
"resume_continuity",
|
|
84
|
+
"unknown_native_passthrough",
|
|
85
|
+
),
|
|
86
|
+
)
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def complete(self) -> bool:
|
|
90
|
+
return all(
|
|
91
|
+
(
|
|
92
|
+
self.inventory_complete,
|
|
93
|
+
self.model_scope_exact,
|
|
94
|
+
self.hard_bypass_complete,
|
|
95
|
+
self.thinking_stripped_control_reducible,
|
|
96
|
+
self.protected_prefix_exact,
|
|
97
|
+
self.token_off_on_wire_equivalent,
|
|
98
|
+
self.tool_history_fidelity,
|
|
99
|
+
self.cache_metadata_preserved,
|
|
100
|
+
self.resume_continuity,
|
|
101
|
+
self.unknown_native_passthrough,
|
|
102
|
+
)
|
|
103
|
+
)
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
@dataclass(frozen=True, slots=True)
|
|
107
|
+
class AnthropicPreservedThinkingLiveEvidence:
|
|
108
|
+
evidence_id: str
|
|
109
|
+
model_id: str
|
|
110
|
+
binding_beta: str
|
|
111
|
+
prefix_mismatch_behavior: str
|
|
112
|
+
cache_hit_observed: bool
|
|
113
|
+
tool_history_fidelity: bool
|
|
114
|
+
resume_continuity: bool
|
|
115
|
+
prefix_binding_observed: bool
|
|
116
|
+
input_transformations_clear: bool
|
|
117
|
+
usage_complete: bool
|
|
118
|
+
estimated_cost_usd: str
|
|
119
|
+
accounting_status: str
|
|
120
|
+
|
|
121
|
+
def __post_init__(self) -> None:
|
|
122
|
+
_require_evidence_id(self.evidence_id)
|
|
123
|
+
if not isinstance(self.model_id, str) or not self.model_id:
|
|
124
|
+
raise ValueError("model_id must be non-blank")
|
|
125
|
+
if self.binding_beta != BINDING_BETA:
|
|
126
|
+
raise ValueError("unsupported binding beta")
|
|
127
|
+
if self.prefix_mismatch_behavior != BINDING_MODE:
|
|
128
|
+
raise ValueError("unsupported prefix mismatch behavior")
|
|
129
|
+
_require_bool_fields(
|
|
130
|
+
self,
|
|
131
|
+
(
|
|
132
|
+
"cache_hit_observed",
|
|
133
|
+
"tool_history_fidelity",
|
|
134
|
+
"resume_continuity",
|
|
135
|
+
"prefix_binding_observed",
|
|
136
|
+
"input_transformations_clear",
|
|
137
|
+
"usage_complete",
|
|
138
|
+
),
|
|
139
|
+
)
|
|
140
|
+
_require_nonnegative_finite_decimal(self.estimated_cost_usd, "estimated_cost_usd")
|
|
141
|
+
if self.accounting_status not in ACCOUNTING_STATES:
|
|
142
|
+
raise ValueError("unsupported accounting_status")
|
|
143
|
+
|
|
144
|
+
@property
|
|
145
|
+
def complete(self) -> bool:
|
|
146
|
+
return (
|
|
147
|
+
self.model_id == MODEL_ID
|
|
148
|
+
and self.cache_hit_observed
|
|
149
|
+
and self.tool_history_fidelity
|
|
150
|
+
and self.resume_continuity
|
|
151
|
+
and self.prefix_binding_observed
|
|
152
|
+
and self.input_transformations_clear
|
|
153
|
+
and self.usage_complete
|
|
154
|
+
)
|
|
155
|
+
|
|
156
|
+
|
|
157
|
+
def _combined_evidence_id(*evidence_ids: str) -> str:
|
|
158
|
+
normalized = tuple(sorted(set(evidence_ids)))
|
|
159
|
+
encoded = json.dumps(list(normalized), separators=(",", ":")).encode("utf-8")
|
|
160
|
+
return f"anthropic-pt:{sha256(encoded).hexdigest()}"
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def build_anthropic_preserved_thinking_generation(
|
|
164
|
+
evidence_ids: tuple[str, ...],
|
|
165
|
+
) -> BenchmarkGeneration:
|
|
166
|
+
normalized = tuple(sorted(set(evidence_ids)))
|
|
167
|
+
if not normalized or any(not isinstance(item, str) or not item.strip() for item in normalized):
|
|
168
|
+
raise ValueError("evidence_ids must contain non-blank values")
|
|
169
|
+
encoded = json.dumps(list(normalized), separators=(",", ":")).encode("utf-8")
|
|
170
|
+
return BenchmarkGeneration(
|
|
171
|
+
generation_id=GENERATION_ID,
|
|
172
|
+
schema_version=1,
|
|
173
|
+
corpus_version=CORPUS_VERSION,
|
|
174
|
+
evaluator_version=EVALUATOR_VERSION,
|
|
175
|
+
policy_version=POLICY_VERSION,
|
|
176
|
+
evidence_digest=sha256(encoded).hexdigest(),
|
|
177
|
+
evidence_ids=normalized,
|
|
178
|
+
)
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
def classify_anthropic_preserved_thinking(
|
|
182
|
+
offline: AnthropicPreservedThinkingOfflineEvidence,
|
|
183
|
+
live: AnthropicPreservedThinkingLiveEvidence | None,
|
|
184
|
+
) -> CompatibilityRecord:
|
|
185
|
+
if not offline.complete:
|
|
186
|
+
return CompatibilityRecord(
|
|
187
|
+
key=ANTHROPIC_PRESERVED_THINKING_KEY,
|
|
188
|
+
state=CompatibilityState.PASSTHROUGH_ONLY,
|
|
189
|
+
evidence_id=offline.evidence_id,
|
|
190
|
+
reason="offline_contract_incomplete",
|
|
191
|
+
)
|
|
192
|
+
if live is None:
|
|
193
|
+
return CompatibilityRecord(
|
|
194
|
+
key=ANTHROPIC_PRESERVED_THINKING_KEY,
|
|
195
|
+
state=CompatibilityState.PASSTHROUGH_ONLY,
|
|
196
|
+
evidence_id=offline.evidence_id,
|
|
197
|
+
reason="offline_conformance_only",
|
|
198
|
+
)
|
|
199
|
+
|
|
200
|
+
evidence_id = _combined_evidence_id(offline.evidence_id, live.evidence_id)
|
|
201
|
+
if not live.complete:
|
|
202
|
+
return CompatibilityRecord(
|
|
203
|
+
key=ANTHROPIC_PRESERVED_THINKING_KEY,
|
|
204
|
+
state=CompatibilityState.PASSTHROUGH_ONLY,
|
|
205
|
+
evidence_id=evidence_id,
|
|
206
|
+
reason="live_contract_incomplete",
|
|
207
|
+
)
|
|
208
|
+
return CompatibilityRecord(
|
|
209
|
+
key=ANTHROPIC_PRESERVED_THINKING_KEY,
|
|
210
|
+
state=CompatibilityState.PASSTHROUGH_ONLY,
|
|
211
|
+
evidence_id=evidence_id,
|
|
212
|
+
reason="live_canary_candidate",
|
|
213
|
+
)
|
|
@@ -0,0 +1,106 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
from dataclasses import dataclass
|
|
4
|
+
|
|
5
|
+
from .model import OptimizationDecision, RequestEnvelope
|
|
6
|
+
|
|
7
|
+
|
|
8
|
+
@dataclass(frozen=True, slots=True)
|
|
9
|
+
class BenchmarkCase:
|
|
10
|
+
name: str
|
|
11
|
+
envelope: RequestEnvelope
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True, slots=True)
|
|
15
|
+
class BenchmarkCaseResult:
|
|
16
|
+
name: str
|
|
17
|
+
decision: str
|
|
18
|
+
before_estimated_tokens: int
|
|
19
|
+
after_estimated_tokens: int
|
|
20
|
+
saved_pct: float
|
|
21
|
+
reasons: tuple[str, ...]
|
|
22
|
+
protected_byte_fidelity: float
|
|
23
|
+
recoverable: bool
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
@dataclass(frozen=True, slots=True)
|
|
27
|
+
class BenchmarkReport:
|
|
28
|
+
cases: tuple[BenchmarkCaseResult, ...]
|
|
29
|
+
baseline_estimated_tokens: int
|
|
30
|
+
optimized_estimated_tokens: int
|
|
31
|
+
total_saved_pct: float
|
|
32
|
+
optimized_count: int
|
|
33
|
+
bypass_count: int
|
|
34
|
+
eligible_saved_pct: float
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _fidelity(original: RequestEnvelope, optimized: RequestEnvelope, protected_ids) -> float:
|
|
38
|
+
original_by_id = {block.id: block.text for block in original.blocks}
|
|
39
|
+
optimized_by_id = {block.id: block.text for block in optimized.blocks}
|
|
40
|
+
ids = tuple(protected_ids)
|
|
41
|
+
if not ids:
|
|
42
|
+
return 1.0
|
|
43
|
+
intact = sum(
|
|
44
|
+
1 for block_id in ids
|
|
45
|
+
if optimized_by_id.get(block_id) == original_by_id.get(block_id)
|
|
46
|
+
)
|
|
47
|
+
return round(intact / len(ids), 6)
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
def _recoverable(engine, refs: tuple[str, ...]) -> bool:
|
|
51
|
+
try:
|
|
52
|
+
for ref in refs:
|
|
53
|
+
engine.store.get(ref)
|
|
54
|
+
except Exception:
|
|
55
|
+
return False
|
|
56
|
+
return True
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def benchmark_cases(cases, engine) -> BenchmarkReport:
|
|
60
|
+
rows: list[BenchmarkCaseResult] = []
|
|
61
|
+
baseline_total = optimized_total = 0
|
|
62
|
+
eligible_before = eligible_after = 0
|
|
63
|
+
optimized_count = 0
|
|
64
|
+
|
|
65
|
+
for case in cases:
|
|
66
|
+
plan = engine.planner.plan(case.envelope)
|
|
67
|
+
result = engine.optimize(case.envelope)
|
|
68
|
+
before = result.before_estimated_tokens
|
|
69
|
+
after = result.after_estimated_tokens
|
|
70
|
+
baseline_total += before
|
|
71
|
+
optimized_total += after
|
|
72
|
+
saved_pct = round((before - after) * 100 / before, 2) if before else 0.0
|
|
73
|
+
if result.decision is OptimizationDecision.OPTIMIZE:
|
|
74
|
+
optimized_count += 1
|
|
75
|
+
eligible_before += before
|
|
76
|
+
eligible_after += after
|
|
77
|
+
rows.append(BenchmarkCaseResult(
|
|
78
|
+
name=case.name,
|
|
79
|
+
decision=result.decision.value,
|
|
80
|
+
before_estimated_tokens=before,
|
|
81
|
+
after_estimated_tokens=after,
|
|
82
|
+
saved_pct=saved_pct,
|
|
83
|
+
reasons=result.reasons,
|
|
84
|
+
protected_byte_fidelity=_fidelity(
|
|
85
|
+
case.envelope, result.envelope, plan.protected_ids
|
|
86
|
+
),
|
|
87
|
+
recoverable=_recoverable(engine, result.recovery_refs),
|
|
88
|
+
))
|
|
89
|
+
|
|
90
|
+
total_saved_pct = (
|
|
91
|
+
round((baseline_total - optimized_total) * 100 / baseline_total, 2)
|
|
92
|
+
if baseline_total else 0.0
|
|
93
|
+
)
|
|
94
|
+
eligible_saved_pct = (
|
|
95
|
+
round((eligible_before - eligible_after) * 100 / eligible_before, 2)
|
|
96
|
+
if eligible_before else 0.0
|
|
97
|
+
)
|
|
98
|
+
return BenchmarkReport(
|
|
99
|
+
cases=tuple(rows),
|
|
100
|
+
baseline_estimated_tokens=baseline_total,
|
|
101
|
+
optimized_estimated_tokens=optimized_total,
|
|
102
|
+
total_saved_pct=total_saved_pct,
|
|
103
|
+
optimized_count=optimized_count,
|
|
104
|
+
bypass_count=len(rows) - optimized_count,
|
|
105
|
+
eligible_saved_pct=eligible_saved_pct,
|
|
106
|
+
)
|