tokenecon 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tokenecon/__init__.py +22 -0
- tokenecon/classifier.py +38 -0
- tokenecon/cli.py +90 -0
- tokenecon/cost_model.py +54 -0
- tokenecon/mock.py +24 -0
- tokenecon/pricing.py +40 -0
- tokenecon/receipts.py +59 -0
- tokenecon/router.py +146 -0
- tokenecon-0.1.0.dist-info/METADATA +127 -0
- tokenecon-0.1.0.dist-info/RECORD +14 -0
- tokenecon-0.1.0.dist-info/WHEEL +5 -0
- tokenecon-0.1.0.dist-info/entry_points.txt +2 -0
- tokenecon-0.1.0.dist-info/licenses/LICENSE +21 -0
- tokenecon-0.1.0.dist-info/top_level.txt +1 -0
tokenecon/__init__.py
ADDED
|
@@ -0,0 +1,22 @@
|
|
|
1
|
+
"""tokenecon: token-economics cost model and tiered routing for agentic AI."""
|
|
2
|
+
|
|
3
|
+
from .classifier import difficulty_score
|
|
4
|
+
from .cost_model import CallRecord, task_cost, tiering_ratio
|
|
5
|
+
from .mock import MockModel
|
|
6
|
+
from .pricing import DEFAULT_TIERS, ModelTier
|
|
7
|
+
from .receipts import StepReceipt, TaskReceipt
|
|
8
|
+
from .router import TieredRouter
|
|
9
|
+
|
|
10
|
+
__version__ = "0.1.0"
|
|
11
|
+
__all__ = [
|
|
12
|
+
"ModelTier",
|
|
13
|
+
"DEFAULT_TIERS",
|
|
14
|
+
"CallRecord",
|
|
15
|
+
"task_cost",
|
|
16
|
+
"tiering_ratio",
|
|
17
|
+
"difficulty_score",
|
|
18
|
+
"TieredRouter",
|
|
19
|
+
"StepReceipt",
|
|
20
|
+
"TaskReceipt",
|
|
21
|
+
"MockModel",
|
|
22
|
+
]
|
tokenecon/classifier.py
ADDED
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
"""Heuristic request-difficulty scoring (paper, section 3.1).
|
|
2
|
+
|
|
3
|
+
Deliberately simple and deliberately replaceable: scores surface signals —
|
|
4
|
+
request length, analytical keywords, question marks, multi-step phrasing —
|
|
5
|
+
into a 0..1 difficulty number. Teams should start here and graduate to a
|
|
6
|
+
learned classifier once they have labeled traffic.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import re
|
|
12
|
+
|
|
13
|
+
_ANALYTICAL_KEYWORDS = frozenset(
|
|
14
|
+
{
|
|
15
|
+
"compare", "contrast", "design", "prove", "proof", "analyze",
|
|
16
|
+
"analyse", "evaluation", "evaluate", "synthesize", "architect",
|
|
17
|
+
"derive", "derivation", "optimize", "trade-off", "tradeoff",
|
|
18
|
+
}
|
|
19
|
+
)
|
|
20
|
+
_MULTI_STEP_PATTERNS = (
|
|
21
|
+
r"\bstep by step\b",
|
|
22
|
+
r"\bfirst\b.*\bthen\b",
|
|
23
|
+
r"\bmulti-?step\b",
|
|
24
|
+
)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def difficulty_score(request: str) -> float:
|
|
28
|
+
"""Score request difficulty in [0, 1] from surface signals."""
|
|
29
|
+
text = request.lower()
|
|
30
|
+
score = 0.15 # floor: every request costs attention
|
|
31
|
+
words = len(text.split())
|
|
32
|
+
score += min(words / 400.0, 0.25)
|
|
33
|
+
hits = sum(1 for kw in _ANALYTICAL_KEYWORDS if kw in text)
|
|
34
|
+
score += min(hits * 0.12, 0.36)
|
|
35
|
+
score += min(text.count("?") * 0.05, 0.10)
|
|
36
|
+
if any(re.search(p, text) for p in _MULTI_STEP_PATTERNS):
|
|
37
|
+
score += 0.15
|
|
38
|
+
return round(min(max(score, 0.0), 1.0), 3)
|
tokenecon/cli.py
ADDED
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
"""Command-line interface: demo the router, forecast spend, compare baselines."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import sys
|
|
8
|
+
|
|
9
|
+
from .mock import MockModel
|
|
10
|
+
from .pricing import DEFAULT_TIERS, ModelTier
|
|
11
|
+
from .router import TieredRouter
|
|
12
|
+
|
|
13
|
+
DEMO_TASKS = [
|
|
14
|
+
"What is the capital of France?",
|
|
15
|
+
"Summarize these three paragraphs about our Q3 cloud spend in two sentences.",
|
|
16
|
+
"Compare Kubernetes and ECS for our batch workload and recommend one, with trade-offs.",
|
|
17
|
+
"Debug this multi-step data pipeline failure. First reproduce the error, then trace it "
|
|
18
|
+
"through three services, then propose a fix.",
|
|
19
|
+
"Design a fault-tolerant agent orchestration layer. Prove it handles partial tool "
|
|
20
|
+
"failures. Show your reasoning step by step.",
|
|
21
|
+
]
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _tiers_from_json(path: str):
|
|
25
|
+
with open(path) as fh:
|
|
26
|
+
data = json.load(fh)
|
|
27
|
+
return [ModelTier(**t) for t in data["tiers"]]
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def cmd_demo(args: argparse.Namespace) -> int:
|
|
31
|
+
tiers = _tiers_from_json(args.tiers) if args.tiers else list(DEFAULT_TIERS)
|
|
32
|
+
router = TieredRouter(tiers=tiers, budget=args.budget)
|
|
33
|
+
model = MockModel()
|
|
34
|
+
total = 0.0
|
|
35
|
+
for task in DEMO_TASKS:
|
|
36
|
+
receipt = router.run(task, model)
|
|
37
|
+
total += receipt.total_cost
|
|
38
|
+
print(receipt.pretty())
|
|
39
|
+
print("-" * 60)
|
|
40
|
+
# Baseline: every task served by the strongest tier, one step each.
|
|
41
|
+
strongest = max(tiers, key=lambda t: t.capability)
|
|
42
|
+
baseline = sum(
|
|
43
|
+
strongest.call_cost(len(t) // 4 + 200, 340) for t in DEMO_TASKS
|
|
44
|
+
)
|
|
45
|
+
print(f"router total : ${total:.6f}")
|
|
46
|
+
print(f"all-large total: ${baseline:.6f}")
|
|
47
|
+
print(f"ratio : {total / baseline:.2%} of baseline")
|
|
48
|
+
return 0
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def cmd_estimate(args: argparse.Namespace) -> int:
|
|
52
|
+
tiers = _tiers_from_json(args.tiers) if args.tiers else list(DEFAULT_TIERS)
|
|
53
|
+
router = TieredRouter(tiers=tiers)
|
|
54
|
+
with open(args.requests) as fh:
|
|
55
|
+
requests = [line.strip() for line in fh if line.strip()]
|
|
56
|
+
total = 0.0
|
|
57
|
+
print(f"{'difficulty':>10} {'tier':>8} {'est_cost':>12} request")
|
|
58
|
+
for req in requests:
|
|
59
|
+
f = router.forecast(req)
|
|
60
|
+
total += f["est_cost"]
|
|
61
|
+
print(f"{f['difficulty']:>10.3f} {f['tier']:>8} ${f['est_cost']:>10.6f} {req[:60]}")
|
|
62
|
+
print(f"\nforecast total for {len(requests)} requests: ${total:.6f}")
|
|
63
|
+
return 0
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
67
|
+
p = argparse.ArgumentParser(
|
|
68
|
+
prog="tokenecon",
|
|
69
|
+
description="Token-economics cost model and tiered routing for agentic AI.",
|
|
70
|
+
)
|
|
71
|
+
p.add_argument("--tiers", help="JSON file with custom tier definitions")
|
|
72
|
+
sub = p.add_subparsers(dest="command", required=True)
|
|
73
|
+
|
|
74
|
+
d = sub.add_parser("demo", help="run 5 sample tasks through the router")
|
|
75
|
+
d.add_argument("--budget", type=float, default=None, help="per-task budget in USD")
|
|
76
|
+
d.set_defaults(func=cmd_demo)
|
|
77
|
+
|
|
78
|
+
e = sub.add_parser("estimate", help="forecast cost for a list of requests")
|
|
79
|
+
e.add_argument("requests", help="text file, one request per line")
|
|
80
|
+
e.set_defaults(func=cmd_estimate)
|
|
81
|
+
return p
|
|
82
|
+
|
|
83
|
+
|
|
84
|
+
def main(argv=None) -> int:
|
|
85
|
+
args = build_parser().parse_args(argv)
|
|
86
|
+
return args.func(args)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
if __name__ == "__main__":
|
|
90
|
+
sys.exit(main())
|
tokenecon/cost_model.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""Per-task cost decomposition across the four token streams.
|
|
2
|
+
|
|
3
|
+
Streams (paper, section 2.1): prompt tokens, completion tokens, tool-call
|
|
4
|
+
overhead, and retrieval tokens. For a task making N calls::
|
|
5
|
+
|
|
6
|
+
C_task = sum_i( t_in_i * p_in_m(i) + t_out_i * p_out_m(i) )
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from dataclasses import dataclass
|
|
12
|
+
from typing import Iterable
|
|
13
|
+
|
|
14
|
+
from .pricing import ModelTier
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass
|
|
18
|
+
class CallRecord:
|
|
19
|
+
"""One priced model call inside a task."""
|
|
20
|
+
|
|
21
|
+
tier: ModelTier
|
|
22
|
+
prompt_tokens: int = 0
|
|
23
|
+
completion_tokens: int = 0
|
|
24
|
+
tool_tokens: int = 0
|
|
25
|
+
retrieval_tokens: int = 0
|
|
26
|
+
|
|
27
|
+
@property
|
|
28
|
+
def input_tokens(self) -> int:
|
|
29
|
+
return self.prompt_tokens + self.tool_tokens + self.retrieval_tokens
|
|
30
|
+
|
|
31
|
+
@property
|
|
32
|
+
def output_tokens(self) -> int:
|
|
33
|
+
return self.completion_tokens
|
|
34
|
+
|
|
35
|
+
def cost(self) -> float:
|
|
36
|
+
return self.tier.call_cost(self.input_tokens, self.output_tokens)
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def task_cost(calls: Iterable[CallRecord]) -> float:
|
|
40
|
+
"""Total USD cost of a task from its per-call records."""
|
|
41
|
+
return sum(call.cost() for call in calls)
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
def tiering_ratio(planning_fraction: float, price_ratio: float) -> float:
|
|
45
|
+
"""C_tiered / C_all_strong ≈ α + (1 − α) / r (paper, section 2.3).
|
|
46
|
+
|
|
47
|
+
With planning fraction α ≈ 0.1 and price ratio r ≈ 30, tiered cost is
|
|
48
|
+
roughly 13% of the all-strong baseline.
|
|
49
|
+
"""
|
|
50
|
+
if price_ratio <= 0:
|
|
51
|
+
raise ValueError("price_ratio must be positive")
|
|
52
|
+
if not 0.0 <= planning_fraction <= 1.0:
|
|
53
|
+
raise ValueError("planning_fraction must be in [0, 1]")
|
|
54
|
+
return planning_fraction + (1.0 - planning_fraction) / price_ratio
|
tokenecon/mock.py
ADDED
|
@@ -0,0 +1,24 @@
|
|
|
1
|
+
"""Deterministic mock model for demos and tests.
|
|
2
|
+
|
|
3
|
+
Stands in for live provider calls so routing, escalation, receipts, and
|
|
4
|
+
guardrails can be exercised with no API keys. Confidence is a deterministic
|
|
5
|
+
function of tier capability vs. request difficulty — replace with a real
|
|
6
|
+
signal (logprobs, verifier, sampling agreement) in production.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from typing import Tuple
|
|
12
|
+
|
|
13
|
+
from .classifier import difficulty_score
|
|
14
|
+
from .pricing import ModelTier
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class MockModel:
|
|
18
|
+
"""model_fn(request, tier) -> (answer, confidence)."""
|
|
19
|
+
|
|
20
|
+
def __call__(self, request: str, tier: ModelTier) -> Tuple[str, float]:
|
|
21
|
+
difficulty = difficulty_score(request)
|
|
22
|
+
confidence = tier.capability - 0.25 * difficulty + 0.15
|
|
23
|
+
confidence = round(min(max(confidence, 0.0), 1.0), 3)
|
|
24
|
+
return f"[{tier.name}] answer to: {request[:60]}", confidence
|
tokenecon/pricing.py
ADDED
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
"""Model tier definitions and price handling.
|
|
2
|
+
|
|
3
|
+
A tier is a priced capability band: input/output price per million tokens
|
|
4
|
+
plus a capability score in [0, 1]. Prices here are ILLUSTRATIVE placeholders —
|
|
5
|
+
substitute your provider's actual price list before making decisions.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import dataclass
|
|
11
|
+
from typing import Tuple
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class ModelTier:
|
|
16
|
+
name: str
|
|
17
|
+
input_per_mtok: float # USD per million input tokens
|
|
18
|
+
output_per_mtok: float # USD per million output tokens
|
|
19
|
+
capability: float # 0..1, higher = more capable
|
|
20
|
+
|
|
21
|
+
def __post_init__(self) -> None:
|
|
22
|
+
if not 0.0 <= self.capability <= 1.0:
|
|
23
|
+
raise ValueError("capability must be in [0, 1]")
|
|
24
|
+
if self.input_per_mtok < 0 or self.output_per_mtok < 0:
|
|
25
|
+
raise ValueError("prices must be non-negative")
|
|
26
|
+
|
|
27
|
+
def call_cost(self, input_tokens: int, output_tokens: int) -> float:
|
|
28
|
+
"""Cost in USD of one call with the given token counts."""
|
|
29
|
+
return (
|
|
30
|
+
input_tokens / 1_000_000 * self.input_per_mtok
|
|
31
|
+
+ output_tokens / 1_000_000 * self.output_per_mtok
|
|
32
|
+
)
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
#: Illustrative three-tier ladder (price ratio large/small = 30, as in the paper).
|
|
36
|
+
DEFAULT_TIERS: Tuple[ModelTier, ...] = (
|
|
37
|
+
ModelTier("small", input_per_mtok=0.20, output_per_mtok=0.80, capability=0.35),
|
|
38
|
+
ModelTier("medium", input_per_mtok=1.00, output_per_mtok=4.00, capability=0.65),
|
|
39
|
+
ModelTier("large", input_per_mtok=6.00, output_per_mtok=24.00, capability=0.95),
|
|
40
|
+
)
|
tokenecon/receipts.py
ADDED
|
@@ -0,0 +1,59 @@
|
|
|
1
|
+
"""Per-step and per-task cost receipts (paper, section 4.2).
|
|
2
|
+
|
|
3
|
+
The receipt is the unit of cost observability: tier used, tokens, cost, and
|
|
4
|
+
confidence per step, plus totals. Accumulated over time, receipts become the
|
|
5
|
+
dataset from which the routing policy improves.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
from dataclasses import asdict, dataclass, field
|
|
11
|
+
from typing import Any, Dict, List, Optional
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class StepReceipt:
|
|
16
|
+
tier: str
|
|
17
|
+
input_tokens: int
|
|
18
|
+
output_tokens: int
|
|
19
|
+
cost: float
|
|
20
|
+
confidence: float
|
|
21
|
+
note: str = ""
|
|
22
|
+
|
|
23
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
24
|
+
return asdict(self)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class TaskReceipt:
|
|
29
|
+
request: str
|
|
30
|
+
difficulty: float
|
|
31
|
+
steps: List[StepReceipt] = field(default_factory=list)
|
|
32
|
+
answer: Optional[str] = None
|
|
33
|
+
total_cost: float = 0.0
|
|
34
|
+
outcome: str = "ok" # ok | degraded | degraded_cached
|
|
35
|
+
|
|
36
|
+
def to_dict(self) -> Dict[str, Any]:
|
|
37
|
+
return {
|
|
38
|
+
"request": self.request,
|
|
39
|
+
"difficulty": self.difficulty,
|
|
40
|
+
"outcome": self.outcome,
|
|
41
|
+
"total_cost": round(self.total_cost, 6),
|
|
42
|
+
"answer": self.answer,
|
|
43
|
+
"steps": [s.to_dict() for s in self.steps],
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
def pretty(self) -> str:
|
|
47
|
+
lines = [
|
|
48
|
+
f"request : {self.request[:80]}",
|
|
49
|
+
f"difficulty : {self.difficulty}",
|
|
50
|
+
f"outcome : {self.outcome}",
|
|
51
|
+
]
|
|
52
|
+
for i, s in enumerate(self.steps, 1):
|
|
53
|
+
lines.append(
|
|
54
|
+
f" step {i}: tier={s.tier} in={s.input_tokens} "
|
|
55
|
+
f"out={s.output_tokens} cost=${s.cost:.6f} "
|
|
56
|
+
f"conf={s.confidence:.2f} {s.note}".rstrip()
|
|
57
|
+
)
|
|
58
|
+
lines.append(f"total_cost : ${self.total_cost:.6f}")
|
|
59
|
+
return "\n".join(lines)
|
tokenecon/router.py
ADDED
|
@@ -0,0 +1,146 @@
|
|
|
1
|
+
"""Tiered routing: classify, select, escalate, budget, receipt (paper, section 3).
|
|
2
|
+
|
|
3
|
+
Design summary
|
|
4
|
+
--------------
|
|
5
|
+
1. Score the request's difficulty (0..1).
|
|
6
|
+
2. Select the cheapest tier whose capability clears difficulty + safety margin;
|
|
7
|
+
fall back to the strongest tier — the router never refuses work.
|
|
8
|
+
3. Run the model; if confidence is below the quality threshold, escalate to
|
|
9
|
+
the next tier (cascade). Confidence may come from logprobs, a verifier, or
|
|
10
|
+
sampling agreement — the router only needs a number and a threshold.
|
|
11
|
+
4. Guardrails (budgets degrade, never fail):
|
|
12
|
+
- Before spending: if even the cheapest first step exceeds the remaining
|
|
13
|
+
budget, serve a cached answer for $0 and mark the run degraded.
|
|
14
|
+
- Before escalating: if the next tier up would break the budget, keep the
|
|
15
|
+
current tier's answer and mark the run degraded.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
from dataclasses import dataclass, field
|
|
21
|
+
from typing import Callable, Dict, List, Optional, Tuple
|
|
22
|
+
|
|
23
|
+
from .classifier import difficulty_score
|
|
24
|
+
from .pricing import ModelTier
|
|
25
|
+
from .receipts import StepReceipt, TaskReceipt
|
|
26
|
+
|
|
27
|
+
#: model_fn(request, tier) -> (answer, confidence in [0, 1])
|
|
28
|
+
ModelFn = Callable[[str, ModelTier], Tuple[str, float]]
|
|
29
|
+
|
|
30
|
+
_SYSTEM_PROMPT_TOKENS = 200 # rough allowance for system instructions
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
@dataclass
|
|
34
|
+
class TieredRouter:
|
|
35
|
+
tiers: List[ModelTier]
|
|
36
|
+
safety_margin: float = 0.10
|
|
37
|
+
quality_threshold: float = 0.70
|
|
38
|
+
budget: Optional[float] = None # USD ceiling per task; None = unbounded
|
|
39
|
+
|
|
40
|
+
def __post_init__(self) -> None:
|
|
41
|
+
if len(self.tiers) < 2:
|
|
42
|
+
raise ValueError("need at least two tiers to route between")
|
|
43
|
+
# Ascending by price; capability is expected to rise with price.
|
|
44
|
+
self.tiers = sorted(self.tiers, key=lambda t: (t.input_per_mtok, t.capability))
|
|
45
|
+
self._cache: Dict[str, str] = {}
|
|
46
|
+
|
|
47
|
+
# -- policy ---------------------------------------------------------
|
|
48
|
+
|
|
49
|
+
def select_tier(self, difficulty: float) -> ModelTier:
|
|
50
|
+
"""Cheapest tier whose capability clears difficulty + margin."""
|
|
51
|
+
bar = min(difficulty + self.safety_margin, 1.0)
|
|
52
|
+
for tier in self.tiers:
|
|
53
|
+
if tier.capability >= bar:
|
|
54
|
+
return tier
|
|
55
|
+
return self.tiers[-1]
|
|
56
|
+
|
|
57
|
+
def _next_tier(self, tier: ModelTier) -> Optional[ModelTier]:
|
|
58
|
+
idx = self.tiers.index(tier)
|
|
59
|
+
return self.tiers[idx + 1] if idx + 1 < len(self.tiers) else None
|
|
60
|
+
|
|
61
|
+
# -- token estimation (heuristic; ~4 chars/token for English) --------
|
|
62
|
+
|
|
63
|
+
def _estimate_io(self, request: str, tier: ModelTier, step: int) -> Tuple[int, int]:
|
|
64
|
+
base_in = len(request) // 4 + _SYSTEM_PROMPT_TOKENS
|
|
65
|
+
# Context accumulates across steps: each prior step appends its
|
|
66
|
+
# output plus retrieval-ish overhead (paper, section 2.1).
|
|
67
|
+
input_tokens = base_in + step * 700
|
|
68
|
+
output_tokens = {"small": 120, "medium": 220}.get(tier.name, 340)
|
|
69
|
+
return input_tokens, output_tokens
|
|
70
|
+
|
|
71
|
+
# -- run ------------------------------------------------------------
|
|
72
|
+
|
|
73
|
+
def run(self, request: str, model_fn: ModelFn) -> TaskReceipt:
|
|
74
|
+
difficulty = difficulty_score(request)
|
|
75
|
+
receipt = TaskReceipt(request=request, difficulty=difficulty)
|
|
76
|
+
spent = 0.0
|
|
77
|
+
|
|
78
|
+
# Guardrail 1 — before spending: cheapest first step over budget
|
|
79
|
+
# and a cached answer exists -> serve it for $0, degraded.
|
|
80
|
+
cheapest = self.tiers[0]
|
|
81
|
+
in_tok, out_tok = self._estimate_io(request, cheapest, 0)
|
|
82
|
+
if (
|
|
83
|
+
self.budget is not None
|
|
84
|
+
and cheapest.call_cost(in_tok, out_tok) > self.budget
|
|
85
|
+
and request in self._cache
|
|
86
|
+
):
|
|
87
|
+
receipt.answer = self._cache[request]
|
|
88
|
+
receipt.outcome = "degraded_cached"
|
|
89
|
+
receipt.steps.append(
|
|
90
|
+
StepReceipt(
|
|
91
|
+
tier="cache", input_tokens=0, output_tokens=0,
|
|
92
|
+
cost=0.0, confidence=1.0,
|
|
93
|
+
note="budget exhausted; served cached answer",
|
|
94
|
+
)
|
|
95
|
+
)
|
|
96
|
+
return receipt
|
|
97
|
+
|
|
98
|
+
tier = self.select_tier(difficulty)
|
|
99
|
+
answer: Optional[str] = None
|
|
100
|
+
step = 0
|
|
101
|
+
while True:
|
|
102
|
+
in_tok, out_tok = self._estimate_io(request, tier, step)
|
|
103
|
+
step_cost = tier.call_cost(in_tok, out_tok)
|
|
104
|
+
answer, confidence = model_fn(request, tier)
|
|
105
|
+
spent += step_cost
|
|
106
|
+
receipt.steps.append(
|
|
107
|
+
StepReceipt(
|
|
108
|
+
tier=tier.name, input_tokens=in_tok,
|
|
109
|
+
output_tokens=out_tok, cost=step_cost,
|
|
110
|
+
confidence=round(confidence, 3),
|
|
111
|
+
)
|
|
112
|
+
)
|
|
113
|
+
if confidence >= self.quality_threshold:
|
|
114
|
+
break
|
|
115
|
+
nxt = self._next_tier(tier)
|
|
116
|
+
if nxt is None:
|
|
117
|
+
break
|
|
118
|
+
# Guardrail 2 — before escalating: next tier would break the
|
|
119
|
+
# budget -> keep this answer, mark degraded.
|
|
120
|
+
n_in, n_out = self._estimate_io(request, nxt, step + 1)
|
|
121
|
+
if self.budget is not None and spent + nxt.call_cost(n_in, n_out) > self.budget:
|
|
122
|
+
receipt.outcome = "degraded"
|
|
123
|
+
receipt.steps[-1].note = "escalation skipped: budget"
|
|
124
|
+
break
|
|
125
|
+
tier, step = nxt, step + 1
|
|
126
|
+
|
|
127
|
+
receipt.answer = answer
|
|
128
|
+
receipt.total_cost = spent
|
|
129
|
+
if receipt.outcome == "ok":
|
|
130
|
+
self._cache[request] = answer or ""
|
|
131
|
+
return receipt
|
|
132
|
+
|
|
133
|
+
# -- forecasting (no model calls) ------------------------------------
|
|
134
|
+
|
|
135
|
+
def forecast(self, request: str) -> Dict[str, float]:
|
|
136
|
+
"""Predicted tier and cost for a request without running any model."""
|
|
137
|
+
difficulty = difficulty_score(request)
|
|
138
|
+
tier = self.select_tier(difficulty)
|
|
139
|
+
in_tok, out_tok = self._estimate_io(request, tier, 0)
|
|
140
|
+
return {
|
|
141
|
+
"difficulty": difficulty,
|
|
142
|
+
"tier": tier.name,
|
|
143
|
+
"input_tokens": in_tok,
|
|
144
|
+
"output_tokens": out_tok,
|
|
145
|
+
"est_cost": tier.call_cost(in_tok, out_tok),
|
|
146
|
+
}
|
|
@@ -0,0 +1,127 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: tokenecon
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Token-economics cost model and tiered model routing for agentic AI workloads
|
|
5
|
+
Author: Karmendra Pandey
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/karmendra8386/agentic-ai-token-economics
|
|
8
|
+
Project-URL: Paper, https://zenodo.org/records/23195511
|
|
9
|
+
Keywords: llm,agents,cost,routing,token-economics,inference
|
|
10
|
+
Classifier: Programming Language :: Python :: 3
|
|
11
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
14
|
+
Requires-Python: >=3.9
|
|
15
|
+
Description-Content-Type: text/markdown
|
|
16
|
+
License-File: LICENSE
|
|
17
|
+
Dynamic: license-file
|
|
18
|
+
|
|
19
|
+
# tokenecon — Token-Economics for Agentic AI
|
|
20
|
+
|
|
21
|
+
A dependency-free Python implementation of the token-economics cost model and
|
|
22
|
+
tiered model routing design from the paper
|
|
23
|
+
[*Token-Economics for Agentic AI: A Cost Model and Tiered Routing Reference*](https://zenodo.org/records/23195511)
|
|
24
|
+
(K. Pandey, Oct 2026).
|
|
25
|
+
|
|
26
|
+
**The problem it solves:** a single model call has a predictable price; a looping
|
|
27
|
+
agent does not. Every tool call, observation, retry, and retrieved document adds
|
|
28
|
+
another round-trip through a priced model, and teams usually discover the total
|
|
29
|
+
from the cloud bill instead of the architecture. This package gives you two things:
|
|
30
|
+
|
|
31
|
+
1. **A cost model** — decompose per-task spend into prompt, completion, tool-call,
|
|
32
|
+
and retrieval tokens, priced per model tier, so you can forecast and budget
|
|
33
|
+
before deployment instead of reconciling after it.
|
|
34
|
+
2. **A tiered router** — classify each request by difficulty, send it to the
|
|
35
|
+
cheapest tier capable of handling it, escalate on low confidence, and enforce
|
|
36
|
+
per-task budgets that degrade gracefully instead of failing.
|
|
37
|
+
|
|
38
|
+
## Install
|
|
39
|
+
|
|
40
|
+
```bash
|
|
41
|
+
pip install tokenecon
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
No dependencies. Python 3.9+.
|
|
45
|
+
|
|
46
|
+
From source: `git clone https://github.com/karmendra8386/agentic-ai-token-economics.git && cd agentic-ai-token-economics && pip install .`
|
|
47
|
+
|
|
48
|
+
## Quickstart
|
|
49
|
+
|
|
50
|
+
```python
|
|
51
|
+
from tokenecon import TieredRouter, DEFAULT_TIERS, MockModel
|
|
52
|
+
|
|
53
|
+
router = TieredRouter(tiers=list(DEFAULT_TIERS), budget=0.05) # 5¢ per task
|
|
54
|
+
receipt = router.run("Summarize our Q3 cloud spend in two sentences.", MockModel())
|
|
55
|
+
print(receipt.pretty())
|
|
56
|
+
# request : Summarize our Q3 cloud spend in two sentences.
|
|
57
|
+
# difficulty : 0.213
|
|
58
|
+
# outcome : ok
|
|
59
|
+
# step 1: tier=small in=210 out=120 cost=$0.000138 conf=0.70
|
|
60
|
+
# total_cost : $0.000138
|
|
61
|
+
```
|
|
62
|
+
|
|
63
|
+
Every run produces a **receipt**: tier used, tokens, cost, and confidence per step.
|
|
64
|
+
Receipts are the unit of cost observability — accumulate them and they become the
|
|
65
|
+
dataset your routing policy improves from.
|
|
66
|
+
|
|
67
|
+
## CLI
|
|
68
|
+
|
|
69
|
+
```bash
|
|
70
|
+
# Run 5 sample tasks through the router and compare against the all-large baseline
|
|
71
|
+
tokenecon demo
|
|
72
|
+
|
|
73
|
+
# Same, under a 1¢ per-task budget (watch guardrails kick in)
|
|
74
|
+
tokenecon demo --budget 0.01
|
|
75
|
+
|
|
76
|
+
# Forecast spend for a list of requests (one per line, no model calls)
|
|
77
|
+
tokenecon estimate requests.txt
|
|
78
|
+
|
|
79
|
+
# Bring your own tiers (JSON: {"tiers": [{"name": ..., "input_per_mtok": ...,
|
|
80
|
+
# "output_per_mtok": ..., "capability": ...}]})
|
|
81
|
+
tokenecon demo --tiers my-tiers.json
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
## How the router works
|
|
85
|
+
|
|
86
|
+
1. **Classify** — each request gets a difficulty score (0–1) from surface signals:
|
|
87
|
+
length, analytical keywords ("compare", "design", "prove"), question marks,
|
|
88
|
+
multi-step phrasing. Simple and replaceable; graduate to a learned classifier
|
|
89
|
+
with labeled traffic.
|
|
90
|
+
2. **Select** — cheapest tier whose capability clears difficulty + safety margin.
|
|
91
|
+
Falls back to the strongest tier; the router never refuses work.
|
|
92
|
+
3. **Escalate** — every response carries a confidence signal; below the quality
|
|
93
|
+
threshold, the request cascades to the next tier (try cheap, escalate on doubt).
|
|
94
|
+
4. **Budget** — two guardrails, both degrade instead of failing:
|
|
95
|
+
- *Before spending:* if the cheapest first step exceeds the budget, serve a
|
|
96
|
+
cached answer for $0.
|
|
97
|
+
- *Before escalating:* if the next tier up would break the budget, keep the
|
|
98
|
+
current answer and mark the run degraded.
|
|
99
|
+
|
|
100
|
+
The tiering arithmetic: with planning fraction α ≈ 0.1 and price ratio r ≈ 30
|
|
101
|
+
between tiers, tiered cost is roughly 13% of the all-strong baseline —
|
|
102
|
+
`tokenecon.tiering_ratio(0.1, 30)`.
|
|
103
|
+
|
|
104
|
+
## Honest limitations
|
|
105
|
+
|
|
106
|
+
- **Pricing is illustrative.** Default tiers are placeholders. Substitute your
|
|
107
|
+
provider's actual price list before making decisions.
|
|
108
|
+
- **Token counting is heuristic** (~4 chars/token for English). Production use
|
|
109
|
+
needs the provider's tokenizer or usage API.
|
|
110
|
+
- **Confidence comes from the mock model** in demos. Production needs a real
|
|
111
|
+
signal: log-probabilities, a verifier model, or sampling agreement.
|
|
112
|
+
- **Single-request scope.** Multi-turn compounding (tool-call overhead, retrieval
|
|
113
|
+
accumulation) is modeled in the cost equations but not executed by the router.
|
|
114
|
+
- See the paper (§6) for the full limitations discussion.
|
|
115
|
+
|
|
116
|
+
## Cite
|
|
117
|
+
|
|
118
|
+
If you use this in your work, please cite the paper:
|
|
119
|
+
|
|
120
|
+
> K. Pandey, "Token-Economics for Agentic AI: A Cost Model and Tiered Routing
|
|
121
|
+
> Reference," Zenodo, DOI [10.5281/zenodo.23195511](https://zenodo.org/records/23195511), Oct. 2026.
|
|
122
|
+
|
|
123
|
+
A `CITATION.cff` is included for automated citation tooling.
|
|
124
|
+
|
|
125
|
+
## License
|
|
126
|
+
|
|
127
|
+
MIT — see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
tokenecon/__init__.py,sha256=yxOKUokNwOnnY2j6FD-FV3Svll0U_GnIh4of_PTvHdQ,566
|
|
2
|
+
tokenecon/classifier.py,sha256=pxoA2iva9nCuotJUNIz-5jpVcgu-1krbEe-gFPSyPZ0,1299
|
|
3
|
+
tokenecon/cli.py,sha256=tV8Q-QnJo7K9TjSrs9InbGVHkr9wgzlS60KSIAvUqJQ,3225
|
|
4
|
+
tokenecon/cost_model.py,sha256=-iZ6kQ3r876XfXacN2OnXNuKL4QWHmzK7-3zElAe6kg,1638
|
|
5
|
+
tokenecon/mock.py,sha256=TH76LRc8zpWrR85QJuAcDqPXLp9zevL4SGBrtGJXBW8,873
|
|
6
|
+
tokenecon/pricing.py,sha256=zZue3DoZcEzJTfSuUc7wDjVuLzHDHXBhk8AF7TQ2cT8,1547
|
|
7
|
+
tokenecon/receipts.py,sha256=Zi09mfJpg3brrCgYaD0Jm_fh0wvliFx4sgmI1iKy0C4,1761
|
|
8
|
+
tokenecon/router.py,sha256=C685V3zg0vBvEutTfE-zEMR3L3fnyLz8DEHsqWvhJCc,6037
|
|
9
|
+
tokenecon-0.1.0.dist-info/licenses/LICENSE,sha256=nlBEbSdzFQm0N1c7RgWKfW89WAZCcF39W6zQakVvdyE,1073
|
|
10
|
+
tokenecon-0.1.0.dist-info/METADATA,sha256=1qwic-p4DQwg9m7efYXIIqimcLZ-_wmjXOJGCT32808,5267
|
|
11
|
+
tokenecon-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
12
|
+
tokenecon-0.1.0.dist-info/entry_points.txt,sha256=jzy0_iVyWKmMDcodi1arMrEUUShjvdBsOb_8s7yT8-8,49
|
|
13
|
+
tokenecon-0.1.0.dist-info/top_level.txt,sha256=BqPju2pDQEbcM72iU3XRTFreHuDDJC7wLFjKFdlEuCw,10
|
|
14
|
+
tokenecon-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Karmendra Pandey
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
tokenecon
|