tokenecon 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
tokenecon/__init__.py ADDED
@@ -0,0 +1,22 @@
1
+ """tokenecon: token-economics cost model and tiered routing for agentic AI."""
2
+
3
+ from .classifier import difficulty_score
4
+ from .cost_model import CallRecord, task_cost, tiering_ratio
5
+ from .mock import MockModel
6
+ from .pricing import DEFAULT_TIERS, ModelTier
7
+ from .receipts import StepReceipt, TaskReceipt
8
+ from .router import TieredRouter
9
+
10
+ __version__ = "0.1.0"
11
+ __all__ = [
12
+ "ModelTier",
13
+ "DEFAULT_TIERS",
14
+ "CallRecord",
15
+ "task_cost",
16
+ "tiering_ratio",
17
+ "difficulty_score",
18
+ "TieredRouter",
19
+ "StepReceipt",
20
+ "TaskReceipt",
21
+ "MockModel",
22
+ ]
@@ -0,0 +1,38 @@
1
+ """Heuristic request-difficulty scoring (paper, section 3.1).
2
+
3
+ Deliberately simple and deliberately replaceable: scores surface signals —
4
+ request length, analytical keywords, question marks, multi-step phrasing —
5
+ into a 0..1 difficulty number. Teams should start here and graduate to a
6
+ learned classifier once they have labeled traffic.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import re
12
+
13
+ _ANALYTICAL_KEYWORDS = frozenset(
14
+ {
15
+ "compare", "contrast", "design", "prove", "proof", "analyze",
16
+ "analyse", "evaluation", "evaluate", "synthesize", "architect",
17
+ "derive", "derivation", "optimize", "trade-off", "tradeoff",
18
+ }
19
+ )
20
+ _MULTI_STEP_PATTERNS = (
21
+ r"\bstep by step\b",
22
+ r"\bfirst\b.*\bthen\b",
23
+ r"\bmulti-?step\b",
24
+ )
25
+
26
+
27
+ def difficulty_score(request: str) -> float:
28
+ """Score request difficulty in [0, 1] from surface signals."""
29
+ text = request.lower()
30
+ score = 0.15 # floor: every request costs attention
31
+ words = len(text.split())
32
+ score += min(words / 400.0, 0.25)
33
+ hits = sum(1 for kw in _ANALYTICAL_KEYWORDS if kw in text)
34
+ score += min(hits * 0.12, 0.36)
35
+ score += min(text.count("?") * 0.05, 0.10)
36
+ if any(re.search(p, text) for p in _MULTI_STEP_PATTERNS):
37
+ score += 0.15
38
+ return round(min(max(score, 0.0), 1.0), 3)
tokenecon/cli.py ADDED
@@ -0,0 +1,90 @@
1
+ """Command-line interface: demo the router, forecast spend, compare baselines."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import sys
8
+
9
+ from .mock import MockModel
10
+ from .pricing import DEFAULT_TIERS, ModelTier
11
+ from .router import TieredRouter
12
+
13
+ DEMO_TASKS = [
14
+ "What is the capital of France?",
15
+ "Summarize these three paragraphs about our Q3 cloud spend in two sentences.",
16
+ "Compare Kubernetes and ECS for our batch workload and recommend one, with trade-offs.",
17
+ "Debug this multi-step data pipeline failure. First reproduce the error, then trace it "
18
+ "through three services, then propose a fix.",
19
+ "Design a fault-tolerant agent orchestration layer. Prove it handles partial tool "
20
+ "failures. Show your reasoning step by step.",
21
+ ]
22
+
23
+
24
+ def _tiers_from_json(path: str):
25
+ with open(path) as fh:
26
+ data = json.load(fh)
27
+ return [ModelTier(**t) for t in data["tiers"]]
28
+
29
+
30
+ def cmd_demo(args: argparse.Namespace) -> int:
31
+ tiers = _tiers_from_json(args.tiers) if args.tiers else list(DEFAULT_TIERS)
32
+ router = TieredRouter(tiers=tiers, budget=args.budget)
33
+ model = MockModel()
34
+ total = 0.0
35
+ for task in DEMO_TASKS:
36
+ receipt = router.run(task, model)
37
+ total += receipt.total_cost
38
+ print(receipt.pretty())
39
+ print("-" * 60)
40
+ # Baseline: every task served by the strongest tier, one step each.
41
+ strongest = max(tiers, key=lambda t: t.capability)
42
+ baseline = sum(
43
+ strongest.call_cost(len(t) // 4 + 200, 340) for t in DEMO_TASKS
44
+ )
45
+ print(f"router total : ${total:.6f}")
46
+ print(f"all-large total: ${baseline:.6f}")
47
+ print(f"ratio : {total / baseline:.2%} of baseline")
48
+ return 0
49
+
50
+
51
+ def cmd_estimate(args: argparse.Namespace) -> int:
52
+ tiers = _tiers_from_json(args.tiers) if args.tiers else list(DEFAULT_TIERS)
53
+ router = TieredRouter(tiers=tiers)
54
+ with open(args.requests) as fh:
55
+ requests = [line.strip() for line in fh if line.strip()]
56
+ total = 0.0
57
+ print(f"{'difficulty':>10} {'tier':>8} {'est_cost':>12} request")
58
+ for req in requests:
59
+ f = router.forecast(req)
60
+ total += f["est_cost"]
61
+ print(f"{f['difficulty']:>10.3f} {f['tier']:>8} ${f['est_cost']:>10.6f} {req[:60]}")
62
+ print(f"\nforecast total for {len(requests)} requests: ${total:.6f}")
63
+ return 0
64
+
65
+
66
+ def build_parser() -> argparse.ArgumentParser:
67
+ p = argparse.ArgumentParser(
68
+ prog="tokenecon",
69
+ description="Token-economics cost model and tiered routing for agentic AI.",
70
+ )
71
+ p.add_argument("--tiers", help="JSON file with custom tier definitions")
72
+ sub = p.add_subparsers(dest="command", required=True)
73
+
74
+ d = sub.add_parser("demo", help="run 5 sample tasks through the router")
75
+ d.add_argument("--budget", type=float, default=None, help="per-task budget in USD")
76
+ d.set_defaults(func=cmd_demo)
77
+
78
+ e = sub.add_parser("estimate", help="forecast cost for a list of requests")
79
+ e.add_argument("requests", help="text file, one request per line")
80
+ e.set_defaults(func=cmd_estimate)
81
+ return p
82
+
83
+
84
+ def main(argv=None) -> int:
85
+ args = build_parser().parse_args(argv)
86
+ return args.func(args)
87
+
88
+
89
+ if __name__ == "__main__":
90
+ sys.exit(main())
@@ -0,0 +1,54 @@
1
+ """Per-task cost decomposition across the four token streams.
2
+
3
+ Streams (paper, section 2.1): prompt tokens, completion tokens, tool-call
4
+ overhead, and retrieval tokens. For a task making N calls::
5
+
6
+ C_task = sum_i( t_in_i * p_in_m(i) + t_out_i * p_out_m(i) )
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from dataclasses import dataclass
12
+ from typing import Iterable
13
+
14
+ from .pricing import ModelTier
15
+
16
+
17
+ @dataclass
18
+ class CallRecord:
19
+ """One priced model call inside a task."""
20
+
21
+ tier: ModelTier
22
+ prompt_tokens: int = 0
23
+ completion_tokens: int = 0
24
+ tool_tokens: int = 0
25
+ retrieval_tokens: int = 0
26
+
27
+ @property
28
+ def input_tokens(self) -> int:
29
+ return self.prompt_tokens + self.tool_tokens + self.retrieval_tokens
30
+
31
+ @property
32
+ def output_tokens(self) -> int:
33
+ return self.completion_tokens
34
+
35
+ def cost(self) -> float:
36
+ return self.tier.call_cost(self.input_tokens, self.output_tokens)
37
+
38
+
39
+ def task_cost(calls: Iterable[CallRecord]) -> float:
40
+ """Total USD cost of a task from its per-call records."""
41
+ return sum(call.cost() for call in calls)
42
+
43
+
44
+ def tiering_ratio(planning_fraction: float, price_ratio: float) -> float:
45
+ """C_tiered / C_all_strong ≈ α + (1 − α) / r (paper, section 2.3).
46
+
47
+ With planning fraction α ≈ 0.1 and price ratio r ≈ 30, tiered cost is
48
+ roughly 13% of the all-strong baseline.
49
+ """
50
+ if price_ratio <= 0:
51
+ raise ValueError("price_ratio must be positive")
52
+ if not 0.0 <= planning_fraction <= 1.0:
53
+ raise ValueError("planning_fraction must be in [0, 1]")
54
+ return planning_fraction + (1.0 - planning_fraction) / price_ratio
tokenecon/mock.py ADDED
@@ -0,0 +1,24 @@
1
+ """Deterministic mock model for demos and tests.
2
+
3
+ Stands in for live provider calls so routing, escalation, receipts, and
4
+ guardrails can be exercised with no API keys. Confidence is a deterministic
5
+ function of tier capability vs. request difficulty — replace with a real
6
+ signal (logprobs, verifier, sampling agreement) in production.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from typing import Tuple
12
+
13
+ from .classifier import difficulty_score
14
+ from .pricing import ModelTier
15
+
16
+
17
+ class MockModel:
18
+ """model_fn(request, tier) -> (answer, confidence)."""
19
+
20
+ def __call__(self, request: str, tier: ModelTier) -> Tuple[str, float]:
21
+ difficulty = difficulty_score(request)
22
+ confidence = tier.capability - 0.25 * difficulty + 0.15
23
+ confidence = round(min(max(confidence, 0.0), 1.0), 3)
24
+ return f"[{tier.name}] answer to: {request[:60]}", confidence
tokenecon/pricing.py ADDED
@@ -0,0 +1,40 @@
1
+ """Model tier definitions and price handling.
2
+
3
+ A tier is a priced capability band: input/output price per million tokens
4
+ plus a capability score in [0, 1]. Prices here are ILLUSTRATIVE placeholders —
5
+ substitute your provider's actual price list before making decisions.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import dataclass
11
+ from typing import Tuple
12
+
13
+
14
+ @dataclass(frozen=True)
15
+ class ModelTier:
16
+ name: str
17
+ input_per_mtok: float # USD per million input tokens
18
+ output_per_mtok: float # USD per million output tokens
19
+ capability: float # 0..1, higher = more capable
20
+
21
+ def __post_init__(self) -> None:
22
+ if not 0.0 <= self.capability <= 1.0:
23
+ raise ValueError("capability must be in [0, 1]")
24
+ if self.input_per_mtok < 0 or self.output_per_mtok < 0:
25
+ raise ValueError("prices must be non-negative")
26
+
27
+ def call_cost(self, input_tokens: int, output_tokens: int) -> float:
28
+ """Cost in USD of one call with the given token counts."""
29
+ return (
30
+ input_tokens / 1_000_000 * self.input_per_mtok
31
+ + output_tokens / 1_000_000 * self.output_per_mtok
32
+ )
33
+
34
+
35
+ #: Illustrative three-tier ladder (price ratio large/small = 30, as in the paper).
36
+ DEFAULT_TIERS: Tuple[ModelTier, ...] = (
37
+ ModelTier("small", input_per_mtok=0.20, output_per_mtok=0.80, capability=0.35),
38
+ ModelTier("medium", input_per_mtok=1.00, output_per_mtok=4.00, capability=0.65),
39
+ ModelTier("large", input_per_mtok=6.00, output_per_mtok=24.00, capability=0.95),
40
+ )
tokenecon/receipts.py ADDED
@@ -0,0 +1,59 @@
1
+ """Per-step and per-task cost receipts (paper, section 4.2).
2
+
3
+ The receipt is the unit of cost observability: tier used, tokens, cost, and
4
+ confidence per step, plus totals. Accumulated over time, receipts become the
5
+ dataset from which the routing policy improves.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ from dataclasses import asdict, dataclass, field
11
+ from typing import Any, Dict, List, Optional
12
+
13
+
14
+ @dataclass
15
+ class StepReceipt:
16
+ tier: str
17
+ input_tokens: int
18
+ output_tokens: int
19
+ cost: float
20
+ confidence: float
21
+ note: str = ""
22
+
23
+ def to_dict(self) -> Dict[str, Any]:
24
+ return asdict(self)
25
+
26
+
27
+ @dataclass
28
+ class TaskReceipt:
29
+ request: str
30
+ difficulty: float
31
+ steps: List[StepReceipt] = field(default_factory=list)
32
+ answer: Optional[str] = None
33
+ total_cost: float = 0.0
34
+ outcome: str = "ok" # ok | degraded | degraded_cached
35
+
36
+ def to_dict(self) -> Dict[str, Any]:
37
+ return {
38
+ "request": self.request,
39
+ "difficulty": self.difficulty,
40
+ "outcome": self.outcome,
41
+ "total_cost": round(self.total_cost, 6),
42
+ "answer": self.answer,
43
+ "steps": [s.to_dict() for s in self.steps],
44
+ }
45
+
46
+ def pretty(self) -> str:
47
+ lines = [
48
+ f"request : {self.request[:80]}",
49
+ f"difficulty : {self.difficulty}",
50
+ f"outcome : {self.outcome}",
51
+ ]
52
+ for i, s in enumerate(self.steps, 1):
53
+ lines.append(
54
+ f" step {i}: tier={s.tier} in={s.input_tokens} "
55
+ f"out={s.output_tokens} cost=${s.cost:.6f} "
56
+ f"conf={s.confidence:.2f} {s.note}".rstrip()
57
+ )
58
+ lines.append(f"total_cost : ${self.total_cost:.6f}")
59
+ return "\n".join(lines)
tokenecon/router.py ADDED
@@ -0,0 +1,146 @@
1
+ """Tiered routing: classify, select, escalate, budget, receipt (paper, section 3).
2
+
3
+ Design summary
4
+ --------------
5
+ 1. Score the request's difficulty (0..1).
6
+ 2. Select the cheapest tier whose capability clears difficulty + safety margin;
7
+ fall back to the strongest tier — the router never refuses work.
8
+ 3. Run the model; if confidence is below the quality threshold, escalate to
9
+ the next tier (cascade). Confidence may come from logprobs, a verifier, or
10
+ sampling agreement — the router only needs a number and a threshold.
11
+ 4. Guardrails (budgets degrade, never fail):
12
+ - Before spending: if even the cheapest first step exceeds the remaining
13
+ budget, serve a cached answer for $0 and mark the run degraded.
14
+ - Before escalating: if the next tier up would break the budget, keep the
15
+ current tier's answer and mark the run degraded.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ from dataclasses import dataclass, field
21
+ from typing import Callable, Dict, List, Optional, Tuple
22
+
23
+ from .classifier import difficulty_score
24
+ from .pricing import ModelTier
25
+ from .receipts import StepReceipt, TaskReceipt
26
+
27
+ #: model_fn(request, tier) -> (answer, confidence in [0, 1])
28
+ ModelFn = Callable[[str, ModelTier], Tuple[str, float]]
29
+
30
+ _SYSTEM_PROMPT_TOKENS = 200 # rough allowance for system instructions
31
+
32
+
33
+ @dataclass
34
+ class TieredRouter:
35
+ tiers: List[ModelTier]
36
+ safety_margin: float = 0.10
37
+ quality_threshold: float = 0.70
38
+ budget: Optional[float] = None # USD ceiling per task; None = unbounded
39
+
40
+ def __post_init__(self) -> None:
41
+ if len(self.tiers) < 2:
42
+ raise ValueError("need at least two tiers to route between")
43
+ # Ascending by price; capability is expected to rise with price.
44
+ self.tiers = sorted(self.tiers, key=lambda t: (t.input_per_mtok, t.capability))
45
+ self._cache: Dict[str, str] = {}
46
+
47
+ # -- policy ---------------------------------------------------------
48
+
49
+ def select_tier(self, difficulty: float) -> ModelTier:
50
+ """Cheapest tier whose capability clears difficulty + margin."""
51
+ bar = min(difficulty + self.safety_margin, 1.0)
52
+ for tier in self.tiers:
53
+ if tier.capability >= bar:
54
+ return tier
55
+ return self.tiers[-1]
56
+
57
+ def _next_tier(self, tier: ModelTier) -> Optional[ModelTier]:
58
+ idx = self.tiers.index(tier)
59
+ return self.tiers[idx + 1] if idx + 1 < len(self.tiers) else None
60
+
61
+ # -- token estimation (heuristic; ~4 chars/token for English) --------
62
+
63
+ def _estimate_io(self, request: str, tier: ModelTier, step: int) -> Tuple[int, int]:
64
+ base_in = len(request) // 4 + _SYSTEM_PROMPT_TOKENS
65
+ # Context accumulates across steps: each prior step appends its
66
+ # output plus retrieval-ish overhead (paper, section 2.1).
67
+ input_tokens = base_in + step * 700
68
+ output_tokens = {"small": 120, "medium": 220}.get(tier.name, 340)
69
+ return input_tokens, output_tokens
70
+
71
+ # -- run ------------------------------------------------------------
72
+
73
+ def run(self, request: str, model_fn: ModelFn) -> TaskReceipt:
74
+ difficulty = difficulty_score(request)
75
+ receipt = TaskReceipt(request=request, difficulty=difficulty)
76
+ spent = 0.0
77
+
78
+ # Guardrail 1 — before spending: cheapest first step over budget
79
+ # and a cached answer exists -> serve it for $0, degraded.
80
+ cheapest = self.tiers[0]
81
+ in_tok, out_tok = self._estimate_io(request, cheapest, 0)
82
+ if (
83
+ self.budget is not None
84
+ and cheapest.call_cost(in_tok, out_tok) > self.budget
85
+ and request in self._cache
86
+ ):
87
+ receipt.answer = self._cache[request]
88
+ receipt.outcome = "degraded_cached"
89
+ receipt.steps.append(
90
+ StepReceipt(
91
+ tier="cache", input_tokens=0, output_tokens=0,
92
+ cost=0.0, confidence=1.0,
93
+ note="budget exhausted; served cached answer",
94
+ )
95
+ )
96
+ return receipt
97
+
98
+ tier = self.select_tier(difficulty)
99
+ answer: Optional[str] = None
100
+ step = 0
101
+ while True:
102
+ in_tok, out_tok = self._estimate_io(request, tier, step)
103
+ step_cost = tier.call_cost(in_tok, out_tok)
104
+ answer, confidence = model_fn(request, tier)
105
+ spent += step_cost
106
+ receipt.steps.append(
107
+ StepReceipt(
108
+ tier=tier.name, input_tokens=in_tok,
109
+ output_tokens=out_tok, cost=step_cost,
110
+ confidence=round(confidence, 3),
111
+ )
112
+ )
113
+ if confidence >= self.quality_threshold:
114
+ break
115
+ nxt = self._next_tier(tier)
116
+ if nxt is None:
117
+ break
118
+ # Guardrail 2 — before escalating: next tier would break the
119
+ # budget -> keep this answer, mark degraded.
120
+ n_in, n_out = self._estimate_io(request, nxt, step + 1)
121
+ if self.budget is not None and spent + nxt.call_cost(n_in, n_out) > self.budget:
122
+ receipt.outcome = "degraded"
123
+ receipt.steps[-1].note = "escalation skipped: budget"
124
+ break
125
+ tier, step = nxt, step + 1
126
+
127
+ receipt.answer = answer
128
+ receipt.total_cost = spent
129
+ if receipt.outcome == "ok":
130
+ self._cache[request] = answer or ""
131
+ return receipt
132
+
133
+ # -- forecasting (no model calls) ------------------------------------
134
+
135
+ def forecast(self, request: str) -> Dict[str, float]:
136
+ """Predicted tier and cost for a request without running any model."""
137
+ difficulty = difficulty_score(request)
138
+ tier = self.select_tier(difficulty)
139
+ in_tok, out_tok = self._estimate_io(request, tier, 0)
140
+ return {
141
+ "difficulty": difficulty,
142
+ "tier": tier.name,
143
+ "input_tokens": in_tok,
144
+ "output_tokens": out_tok,
145
+ "est_cost": tier.call_cost(in_tok, out_tok),
146
+ }
@@ -0,0 +1,127 @@
1
+ Metadata-Version: 2.4
2
+ Name: tokenecon
3
+ Version: 0.1.0
4
+ Summary: Token-economics cost model and tiered model routing for agentic AI workloads
5
+ Author: Karmendra Pandey
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/karmendra8386/agentic-ai-token-economics
8
+ Project-URL: Paper, https://zenodo.org/records/23195511
9
+ Keywords: llm,agents,cost,routing,token-economics,inference
10
+ Classifier: Programming Language :: Python :: 3
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
14
+ Requires-Python: >=3.9
15
+ Description-Content-Type: text/markdown
16
+ License-File: LICENSE
17
+ Dynamic: license-file
18
+
19
+ # tokenecon — Token-Economics for Agentic AI
20
+
21
+ A dependency-free Python implementation of the token-economics cost model and
22
+ tiered model routing design from the paper
23
+ [*Token-Economics for Agentic AI: A Cost Model and Tiered Routing Reference*](https://zenodo.org/records/23195511)
24
+ (K. Pandey, Oct 2026).
25
+
26
+ **The problem it solves:** a single model call has a predictable price; a looping
27
+ agent does not. Every tool call, observation, retry, and retrieved document adds
28
+ another round-trip through a priced model, and teams usually discover the total
29
+ from the cloud bill instead of the architecture. This package gives you two things:
30
+
31
+ 1. **A cost model** — decompose per-task spend into prompt, completion, tool-call,
32
+ and retrieval tokens, priced per model tier, so you can forecast and budget
33
+ before deployment instead of reconciling after it.
34
+ 2. **A tiered router** — classify each request by difficulty, send it to the
35
+ cheapest tier capable of handling it, escalate on low confidence, and enforce
36
+ per-task budgets that degrade gracefully instead of failing.
37
+
38
+ ## Install
39
+
40
+ ```bash
41
+ pip install tokenecon
42
+ ```
43
+
44
+ No dependencies. Python 3.9+.
45
+
46
+ From source: `git clone https://github.com/karmendra8386/agentic-ai-token-economics.git && cd agentic-ai-token-economics && pip install .`
47
+
48
+ ## Quickstart
49
+
50
+ ```python
51
+ from tokenecon import TieredRouter, DEFAULT_TIERS, MockModel
52
+
53
+ router = TieredRouter(tiers=list(DEFAULT_TIERS), budget=0.05) # 5¢ per task
54
+ receipt = router.run("Summarize our Q3 cloud spend in two sentences.", MockModel())
55
+ print(receipt.pretty())
56
+ # request : Summarize our Q3 cloud spend in two sentences.
57
+ # difficulty : 0.213
58
+ # outcome : ok
59
+ # step 1: tier=small in=210 out=120 cost=$0.000138 conf=0.70
60
+ # total_cost : $0.000138
61
+ ```
62
+
63
+ Every run produces a **receipt**: tier used, tokens, cost, and confidence per step.
64
+ Receipts are the unit of cost observability — accumulate them and they become the
65
+ dataset your routing policy improves from.
66
+
67
+ ## CLI
68
+
69
+ ```bash
70
+ # Run 5 sample tasks through the router and compare against the all-large baseline
71
+ tokenecon demo
72
+
73
+ # Same, under a 1¢ per-task budget (watch guardrails kick in)
74
+ tokenecon demo --budget 0.01
75
+
76
+ # Forecast spend for a list of requests (one per line, no model calls)
77
+ tokenecon estimate requests.txt
78
+
79
+ # Bring your own tiers (JSON: {"tiers": [{"name": ..., "input_per_mtok": ...,
80
+ # "output_per_mtok": ..., "capability": ...}]})
81
+ tokenecon demo --tiers my-tiers.json
82
+ ```
83
+
84
+ ## How the router works
85
+
86
+ 1. **Classify** — each request gets a difficulty score (0–1) from surface signals:
87
+ length, analytical keywords ("compare", "design", "prove"), question marks,
88
+ multi-step phrasing. Simple and replaceable; graduate to a learned classifier
89
+ with labeled traffic.
90
+ 2. **Select** — cheapest tier whose capability clears difficulty + safety margin.
91
+ Falls back to the strongest tier; the router never refuses work.
92
+ 3. **Escalate** — every response carries a confidence signal; below the quality
93
+ threshold, the request cascades to the next tier (try cheap, escalate on doubt).
94
+ 4. **Budget** — two guardrails, both degrade instead of failing:
95
+ - *Before spending:* if the cheapest first step exceeds the budget, serve a
96
+ cached answer for $0.
97
+ - *Before escalating:* if the next tier up would break the budget, keep the
98
+ current answer and mark the run degraded.
99
+
100
+ The tiering arithmetic: with planning fraction α ≈ 0.1 and price ratio r ≈ 30
101
+ between tiers, tiered cost is roughly 13% of the all-strong baseline —
102
+ `tokenecon.tiering_ratio(0.1, 30)`.
103
+
104
+ ## Honest limitations
105
+
106
+ - **Pricing is illustrative.** Default tiers are placeholders. Substitute your
107
+ provider's actual price list before making decisions.
108
+ - **Token counting is heuristic** (~4 chars/token for English). Production use
109
+ needs the provider's tokenizer or usage API.
110
+ - **Confidence comes from the mock model** in demos. Production needs a real
111
+ signal: log-probabilities, a verifier model, or sampling agreement.
112
+ - **Single-request scope.** Multi-turn compounding (tool-call overhead, retrieval
113
+ accumulation) is modeled in the cost equations but not executed by the router.
114
+ - See the paper (§6) for the full limitations discussion.
115
+
116
+ ## Cite
117
+
118
+ If you use this in your work, please cite the paper:
119
+
120
+ > K. Pandey, "Token-Economics for Agentic AI: A Cost Model and Tiered Routing
121
+ > Reference," Zenodo, DOI [10.5281/zenodo.23195511](https://zenodo.org/records/23195511), Oct. 2026.
122
+
123
+ A `CITATION.cff` is included for automated citation tooling.
124
+
125
+ ## License
126
+
127
+ MIT — see [LICENSE](LICENSE).
@@ -0,0 +1,14 @@
1
+ tokenecon/__init__.py,sha256=yxOKUokNwOnnY2j6FD-FV3Svll0U_GnIh4of_PTvHdQ,566
2
+ tokenecon/classifier.py,sha256=pxoA2iva9nCuotJUNIz-5jpVcgu-1krbEe-gFPSyPZ0,1299
3
+ tokenecon/cli.py,sha256=tV8Q-QnJo7K9TjSrs9InbGVHkr9wgzlS60KSIAvUqJQ,3225
4
+ tokenecon/cost_model.py,sha256=-iZ6kQ3r876XfXacN2OnXNuKL4QWHmzK7-3zElAe6kg,1638
5
+ tokenecon/mock.py,sha256=TH76LRc8zpWrR85QJuAcDqPXLp9zevL4SGBrtGJXBW8,873
6
+ tokenecon/pricing.py,sha256=zZue3DoZcEzJTfSuUc7wDjVuLzHDHXBhk8AF7TQ2cT8,1547
7
+ tokenecon/receipts.py,sha256=Zi09mfJpg3brrCgYaD0Jm_fh0wvliFx4sgmI1iKy0C4,1761
8
+ tokenecon/router.py,sha256=C685V3zg0vBvEutTfE-zEMR3L3fnyLz8DEHsqWvhJCc,6037
9
+ tokenecon-0.1.0.dist-info/licenses/LICENSE,sha256=nlBEbSdzFQm0N1c7RgWKfW89WAZCcF39W6zQakVvdyE,1073
10
+ tokenecon-0.1.0.dist-info/METADATA,sha256=1qwic-p4DQwg9m7efYXIIqimcLZ-_wmjXOJGCT32808,5267
11
+ tokenecon-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
12
+ tokenecon-0.1.0.dist-info/entry_points.txt,sha256=jzy0_iVyWKmMDcodi1arMrEUUShjvdBsOb_8s7yT8-8,49
13
+ tokenecon-0.1.0.dist-info/top_level.txt,sha256=BqPju2pDQEbcM72iU3XRTFreHuDDJC7wLFjKFdlEuCw,10
14
+ tokenecon-0.1.0.dist-info/RECORD,,
@@ -0,0 +1,5 @@
1
+ Wheel-Version: 1.0
2
+ Generator: setuptools (84.0.0)
3
+ Root-Is-Purelib: true
4
+ Tag: py3-none-any
5
+
@@ -0,0 +1,2 @@
1
+ [console_scripts]
2
+ tokenecon = tokenecon.cli:main
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Karmendra Pandey
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ tokenecon