tanglebrain 0.16.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,99 @@
1
+ """Cheap local classifier gate — OFF by default.
2
+
3
+ The default routing strategy is frontier-first with orchestrator rotation. An optional escape valve:
4
+ put a **cheap local classifier in front** that does one narrow job — decide whether a request is
5
+ *trivial* (the free local backend can fully handle it) or *needs-frontier* (route to the
6
+ orchestrator). Trivial requests then skip the orchestrators entirely.
7
+
8
+ Two deliberate design rules:
9
+
10
+ - **Narrow classification, not self-judgement.** The classifier rates *task complexity*, it does not
11
+ ask the local model "can YOU do this?" — that framing is unreliable.
12
+ - **Fail safe toward the capable path.** Any ambiguity, parse miss, or classifier error resolves to
13
+ ``frontier``. The gate must never trap a hard task on the local tier because the classifier was
14
+ unsure or broke — at worst it falls back to normal frontier-first routing.
15
+
16
+ It is inert unless explicitly enabled (``classifier_gate_enabled`` in settings, or ``--gate`` on the
17
+ CLI).
18
+ """
19
+ from __future__ import annotations
20
+
21
+ import re
22
+ from typing import Callable
23
+
24
+ from tanglebrain.adapters import OpenAICompatAdapter
25
+ from tanglebrain.adapters.base import Adapter
26
+ from tanglebrain.roster import Roster, load_roster
27
+ from tanglebrain.selector import select_local
28
+
29
+ TRIVIAL = "trivial"
30
+ FRONTIER = "frontier"
31
+
32
+ # A local reasoning model spends part of its budget on internal reasoning, so give the classify call
33
+ # enough headroom to finish reasoning AND emit the verdict; a truncated (null) response just fails
34
+ # safe to FRONTIER. Kept modest because this runs in front of every gated request.
35
+ CLASSIFY_MAX_TOKENS = 1024
36
+
37
+ _INSTRUCTIONS = (
38
+ "You are a routing classifier. Decide how complex the USER REQUEST below is, so a dispatcher "
39
+ "can send simple work to a small local model and hard work to a frontier model.\n\n"
40
+ "Classify by the TASK's intrinsic complexity (not by who should do it):\n"
41
+ "- TRIVIAL: short, well-specified, single-step work — a factual lookup, a small/simple code "
42
+ "snippet, a quick rewrite or format, a direct question with a known answer.\n"
43
+ "- FRONTIER: anything needing multi-step reasoning, decomposition, architecture or design, "
44
+ "debugging across files, careful trade-offs, or ambiguous/open-ended judgement.\n\n"
45
+ "Decide quickly; do not overthink and do not attempt the task. When unsure, answer FRONTIER."
46
+ )
47
+
48
+
49
+ def _build_prompt(user_request: str) -> str:
50
+ """Fold the classify instructions and the request into one user message (the adapter sends one)."""
51
+ return (
52
+ f"{_INSTRUCTIONS}\n\n--- USER REQUEST ---\n{user_request}\n--- END REQUEST ---\n\n"
53
+ "Answer with exactly one word: TRIVIAL or FRONTIER."
54
+ )
55
+
56
+
57
+ def _parse_verdict(text: str) -> str:
58
+ """Map a classifier response to ``TRIVIAL`` or ``FRONTIER``, defaulting to ``FRONTIER``.
59
+
60
+ We instruct the model to answer with exactly one word, so the verdict is ``TRIVIAL`` **only** when
61
+ the response's first word token is exactly ``trivial`` and ``frontier`` appears nowhere. Anything
62
+ else — ``frontier``, a both-words answer, prose, a *negation* like "not trivial", junk, empty — is
63
+ ``FRONTIER`` (the safe default). The first-token rule is deliberately strict: free-form prose can't
64
+ leak a spurious ``TRIVIAL`` and strand a hard task on local.
65
+ """
66
+ words = re.findall(r"[a-z]+", (text or "").lower())
67
+ if "frontier" in words:
68
+ return FRONTIER
69
+ return TRIVIAL if words[:1] == ["trivial"] else FRONTIER
70
+
71
+
72
+ def classify(
73
+ prompt: str,
74
+ roster: Roster | None = None,
75
+ adapter_factory: Callable[..., Adapter] = OpenAICompatAdapter.from_entry,
76
+ max_tokens: int = CLASSIFY_MAX_TOKENS,
77
+ ) -> str:
78
+ """Classify ``prompt`` as :data:`TRIVIAL` or :data:`FRONTIER` using free local gpt-oss.
79
+
80
+ Never raises and never blocks routing: any failure (no local entry, transport error, truncated
81
+ or unparsable response) resolves to :data:`FRONTIER`, so a broken classifier degrades to today's
82
+ normal frontier-first routing rather than trapping a task on the local tier.
83
+
84
+ Args:
85
+ prompt: The user request to classify.
86
+ roster: The loaded roster (defaults to the packaged roster).
87
+ adapter_factory: Builds the local adapter from the selected entry (injectable for tests).
88
+ max_tokens: Budget for the classify call (gpt-oss needs reasoning headroom; see the constant).
89
+
90
+ Returns:
91
+ :data:`TRIVIAL` or :data:`FRONTIER`.
92
+ """
93
+ try:
94
+ entry = select_local(roster if roster is not None else load_roster())
95
+ adapter = adapter_factory(entry)
96
+ text = adapter.run(_build_prompt(prompt), {"max_tokens": max_tokens})
97
+ except Exception: # noqa: BLE001 — deliberately total: a broken classifier must fail safe, never block
98
+ return FRONTIER
99
+ return _parse_verdict(text)
tanglebrain/cli.py ADDED
@@ -0,0 +1,251 @@
1
+ """TangleBrain CLI — route one request and print the response.
2
+
3
+ Thin wiring over :func:`run_once`; the routing logic lives in the router/selector/adapters. The
4
+ path is chosen by flag precedence ``--model`` > ``--local`` > the frontier-first router (the
5
+ default): the router selects + rotates an orchestrator, fails over on errors, and gives it the
6
+ local-delegate tool so it offloads sub-tasks to the free local backend.
7
+
8
+ Usage::
9
+
10
+ tanglebrain "Refactor this module and add tests." # default: frontier-first router
11
+ tanglebrain --task code "..." # task-fit hint for the router
12
+ tanglebrain --local "Write a haiku about local inference." # force the free local tier
13
+ tanglebrain --model gemini "Summarize this long document." # pin a specific roster entry
14
+ """
15
+ from __future__ import annotations
16
+
17
+ import argparse
18
+ import sys
19
+ import uuid
20
+
21
+ from tanglebrain import __version__
22
+ from tanglebrain.adapters import AdapterError
23
+ from tanglebrain.classifier import TRIVIAL, classify
24
+ from tanglebrain.measurement import (
25
+ format_rollup,
26
+ load_pricing,
27
+ read_records,
28
+ record_task,
29
+ rollup,
30
+ )
31
+ from tanglebrain.roster import RosterError, load_roster
32
+ from tanglebrain.router import Router, RouterError
33
+ from tanglebrain.selector import SelectionError, build_adapter, select_by_id, select_local
34
+ from tanglebrain.settings import load_settings
35
+
36
+
37
+ def build_parser() -> argparse.ArgumentParser:
38
+ """Build the argument parser for the ``tanglebrain`` command.
39
+
40
+ Returns:
41
+ The configured :class:`argparse.ArgumentParser`.
42
+ """
43
+ parser = argparse.ArgumentParser(
44
+ prog="tanglebrain",
45
+ description=(
46
+ "Route one request to the cheapest capable tier (frontier-first by default), or "
47
+ "print the 'spend avoided' rollup with --stats."
48
+ ),
49
+ )
50
+ parser.add_argument(
51
+ "--version",
52
+ action="version",
53
+ version=f"%(prog)s {__version__}",
54
+ help="Print the TangleBrain version and exit.",
55
+ )
56
+ parser.add_argument(
57
+ "prompt",
58
+ nargs="?",
59
+ default=None,
60
+ help="The prompt to route. Optional only when --stats is given.",
61
+ )
62
+ parser.add_argument(
63
+ "--roster",
64
+ default=None,
65
+ help="Path to a roster YAML (defaults to the packaged tanglebrain/config/roster.yaml).",
66
+ )
67
+ parser.add_argument(
68
+ "--model",
69
+ default=None,
70
+ help=(
71
+ "Route to a specific roster entry by id (e.g. 'claude'). Without it, the default "
72
+ "local-first selection is used. This is an explicit override of routing."
73
+ ),
74
+ )
75
+ parser.add_argument(
76
+ "--local",
77
+ action="store_true",
78
+ help=(
79
+ "Force the free local tier (gpt-oss) instead of the default frontier-first router. "
80
+ "Use for a quick, $0, no-orchestration answer."
81
+ ),
82
+ )
83
+ parser.add_argument(
84
+ "--route",
85
+ action="store_true",
86
+ help="Deprecated/no-op: the frontier-first router is now the default. Kept for back-compat.",
87
+ )
88
+ parser.add_argument(
89
+ "--task",
90
+ default=None,
91
+ help="Task-fit hint for the router (a good_at tag, e.g. 'code', 'reasoning', 'long-context').",
92
+ )
93
+ gate_group = parser.add_mutually_exclusive_group()
94
+ gate_group.add_argument(
95
+ "--gate",
96
+ dest="gate",
97
+ action="store_true",
98
+ default=None,
99
+ help="Force the local classifier gate ON for this run: a cheap local classify sends "
100
+ "trivial requests straight to the free local backend, and only frontier ones to an "
101
+ "orchestrator.",
102
+ )
103
+ gate_group.add_argument(
104
+ "--no-gate",
105
+ dest="gate",
106
+ action="store_false",
107
+ help="Force the classifier gate OFF (always frontier-first router), ignoring the setting.",
108
+ )
109
+ parser.add_argument(
110
+ "--max-tokens",
111
+ type=int,
112
+ default=None,
113
+ help="Override the completion token cap (defaults to the adapter's 2048).",
114
+ )
115
+ parser.add_argument(
116
+ "--stats",
117
+ action="store_true",
118
+ help=(
119
+ "Print the 'spend avoided' rollup (cloud-equivalent cost of every routed task so far) "
120
+ "and exit. No prompt needed."
121
+ ),
122
+ )
123
+ return parser
124
+
125
+
126
+ def _served(path: str, entry) -> dict | None:
127
+ """Build the ``{path, tier, model}`` served-summary for a routed task, or ``None``."""
128
+ if entry is None:
129
+ return None
130
+ return {"path": path, "tier": entry.tier, "model": entry.id}
131
+
132
+
133
+ def run_once(
134
+ prompt: str,
135
+ roster_path: str | None = None,
136
+ max_tokens: int | None = None,
137
+ model: str | None = None,
138
+ local: bool = False,
139
+ task: str | None = None,
140
+ return_served: bool = False,
141
+ gate: bool | None = None,
142
+ ):
143
+ """Route a single prompt to a roster tier and return the response text.
144
+
145
+ Paths, in precedence order:
146
+
147
+ - ``model`` set → select that named entry explicitly (an override, not a routing decision).
148
+ - ``local`` true → the free local tier directly, no orchestration.
149
+ - otherwise → the default routing path. With the **classifier gate** off (the default), this
150
+ is **the frontier-first** :class:`~tanglebrain.router.Router`: task-fit orchestrator selection +
151
+ rotation + failover across the orchestrators, each given the local-delegate tool. With the gate
152
+ on, a cheap local classify runs first: a *trivial* request is handled directly on the free local
153
+ backend (path ``gate-local``, skipping the orchestrators), and everything else falls through to
154
+ the router.
155
+
156
+ Args:
157
+ prompt: The prompt to route.
158
+ roster_path: Optional roster YAML path (defaults to the packaged roster).
159
+ max_tokens: Optional completion token cap (honoured by the openai-compat adapter; the
160
+ CLI adapter ignores it, as each CLI controls its own limits).
161
+ model: Optional roster entry id to route to explicitly.
162
+ local: Force the free local tier instead of the frontier-first router.
163
+ task: Optional task-fit hint for the router (a ``good_at`` tag).
164
+ return_served: When ``True``, return ``(text, served)`` where ``served`` is
165
+ ``{path, tier, model}`` for the entry that served the task (or ``None`` if unknown).
166
+ The GUI uses this so it needn't re-read the usage log. Default ``False`` returns the
167
+ plain text string, so existing callers (``main``) are unchanged.
168
+ gate: Override for the classifier gate on the default path. ``None`` (default) uses the
169
+ ``classifier_gate_enabled`` setting; ``True``/``False`` force the gate on/off for this
170
+ call. Ignored when ``model`` or ``local`` is set.
171
+
172
+ Returns:
173
+ The response text (``str``), or ``(text, served)`` when ``return_served`` is ``True``.
174
+
175
+ Raises:
176
+ RosterError: If the roster cannot be loaded.
177
+ SelectionError: If ``model``/``local`` is used and no suitable entry is available.
178
+ RouterError: If the router runs and no orchestrator can serve the request.
179
+ AdapterError: If the adapter cannot produce text.
180
+ """
181
+ roster = load_roster(roster_path)
182
+ # Mint a task id for this routed task. It is recorded on the task and threaded through opts so
183
+ # the orchestrator-CLI adapter can propagate it to delegated sub-calls (see CliAdapter.run /
184
+ # PARENT_TASK_ID_ENV), linking the delegation tree back to this task. Cheap and side-effect-free
185
+ # to mint on every path; only the router path (orchestrators with the delegate tool) acts on it.
186
+ task_id = uuid.uuid4().hex
187
+ opts: dict = {"task_id": task_id}
188
+ if max_tokens is not None:
189
+ opts["max_tokens"] = max_tokens
190
+
191
+ if model is not None:
192
+ path, entry = "model", select_by_id(roster, model)
193
+ text = build_adapter(entry).run(prompt, opts)
194
+ elif local:
195
+ path, entry = "local", select_local(roster)
196
+ text = build_adapter(entry).run(prompt, opts)
197
+ else:
198
+ gate_on = load_settings().classifier_gate_enabled if gate is None else gate
199
+ if gate_on and classify(prompt, roster=roster) == TRIVIAL:
200
+ # classifier gate: a trivial request skips the orchestrators and is handled directly on
201
+ # the free local backend. Frontier (or any classifier failure) falls through to the router.
202
+ path, entry = "gate-local", select_local(roster)
203
+ text = build_adapter(entry).run(prompt, opts)
204
+ else:
205
+ path = "router"
206
+ router = Router(roster)
207
+ text = router.route(prompt, task=task, opts=opts)
208
+ entry = router.last_served
209
+
210
+ record_task(path=path, entry=entry, prompt=prompt, response=text, task_id=task_id)
211
+ return (text, _served(path, entry)) if return_served else text
212
+
213
+
214
+ def main(argv: list[str] | None = None) -> int:
215
+ """Console entry point.
216
+
217
+ Args:
218
+ argv: Optional argument list (defaults to ``sys.argv[1:]``).
219
+
220
+ Returns:
221
+ Process exit code: ``0`` on success, ``1`` on a known TangleBrain error.
222
+ """
223
+ parser = build_parser()
224
+ args = parser.parse_args(argv)
225
+
226
+ if args.stats:
227
+ print(format_rollup(rollup(read_records()), load_pricing()))
228
+ return 0
229
+
230
+ if args.prompt is None:
231
+ parser.error("prompt is required (unless --stats is given)")
232
+
233
+ try:
234
+ text = run_once(
235
+ args.prompt,
236
+ roster_path=args.roster,
237
+ max_tokens=args.max_tokens,
238
+ model=args.model,
239
+ local=args.local,
240
+ task=args.task,
241
+ gate=args.gate,
242
+ )
243
+ except (RosterError, SelectionError, RouterError, AdapterError) as exc:
244
+ print(f"tanglebrain: {exc}", file=sys.stderr)
245
+ return 1
246
+ print(text)
247
+ return 0
248
+
249
+
250
+ if __name__ == "__main__":
251
+ raise SystemExit(main())
@@ -0,0 +1,14 @@
1
+ # Cloud-equivalent reference pricing for the "spend avoided" rollup.
2
+ #
3
+ # Methodology: for each routed task, estimate what it WOULD have cost on a paid frontier API,
4
+ # valued at a reference frontier price — here Claude Sonnet, $3/MTok input, $15/MTok output. The
5
+ # rollup multiplies estimated tokens (a chars/4 heuristic over the visible prompt + response) by
6
+ # these rates. Values are US dollars per 1,000,000 tokens. Tune them to whatever frontier model you
7
+ # want to compare against.
8
+ #
9
+ # Set `placeholder: true` if your rates are rough/illustrative — the `tanglebrain --stats` rollup
10
+ # then flags every figure as PLACEHOLDER so no number is mistaken for a precise cost.
11
+ placeholder: false
12
+ reference_model: "Claude Sonnet ($3/$15 per MTok)"
13
+ input_per_mtok: 3.00
14
+ output_per_mtok: 15.00
@@ -0,0 +1,129 @@
1
+ # TangleBrain roster — a GENERIC EXAMPLE list of routable models. Edit it to your own backends.
2
+ #
3
+ # This shipped file is just a starting point. Your REAL roster lives OUTSIDE the repo and is
4
+ # auto-discovered, so a `git pull` never clobbers it. Resolution order (first hit wins):
5
+ # 1. $TANGLEBRAIN_ROSTER (explicit path)
6
+ # 2. ~/.config/tanglebrain/roster.yaml (your roster — recommended)
7
+ # 3. this packaged example (fallback)
8
+ # Copy this file to ~/.config/tanglebrain/roster.yaml and edit it there.
9
+ #
10
+ # Add / remove / reorganize entries freely — adding a model is a config edit, never a code change.
11
+ # Flag `can_orchestrate: true` to put an entry in the orchestrator rotation. Flag
12
+ # `can_delegate: true` to make an entry a delegate target an orchestrator can hand a sub-task to
13
+ # (via the `delegate` MCP tool), either by naming its id or by capability — `delegate(task: code)`
14
+ # picks the cheapest `can_delegate` entry whose `good_at` lists that tag (local before sub; paid
15
+ # `api` is never auto-selected). The free local model is always available as the default target.
16
+ #
17
+ # invoke.kind ∈ {openai-compat, cli, api}. key_ref ∈ {file:PATH, env:NAME, none} — a
18
+ # credential *reference*, never a raw secret.
19
+ #
20
+ # For `cli` entries: a literal `{prompt}` token in `cmd` is replaced with the prompt; with no
21
+ # token the prompt is appended as the final argument. `parse` names the output parser
22
+ # (claude-json | gemini-json | plain). `scrub_env` strips env vars from the subprocess.
23
+ #
24
+ # The paid-API tier (kind: api) is gated behind `api_billing_enabled` (config/settings.yaml,
25
+ # default false) AND each entry's own `enabled` flag — it parses but stays inert until both are
26
+ # on. A commented example is at the bottom of this file.
27
+
28
+ # Free local tier — any OpenAI-compatible local server. This example points at Ollama on its
29
+ # default port; run `ollama pull llama3.2` (or any model) first, or swap in your own endpoint.
30
+ - id: local-ollama
31
+ tier: local
32
+ invoke:
33
+ kind: openai-compat
34
+ base_url: "http://localhost:11434/v1"
35
+ model: "llama3.2"
36
+ # Local Ollama needs no auth, so no key_ref. For a keyed endpoint, add e.g.
37
+ # key_ref: "file:~/.config/tanglebrain/keys/local.key" (a reference, never the raw key).
38
+ cost: free
39
+ good_at: [grunt, code, tools]
40
+ can_delegate: true # an orchestrator may target this backend by id via the delegate tool
41
+
42
+ # --- OPT-IN delegate target (non-local) — COMMENTED OUT so it stays inert. ---
43
+ # A delegate target is a backend an orchestrator can hand a *sub-task* to (not the whole request) —
44
+ # e.g. a cheaper sub or a better-fit model. Flag any entry `can_delegate: true` and it joins the
45
+ # menu the `delegate` tool advertises; the model picks a target by its `good_at` fit. A target is
46
+ # invoked as a leaf (it never gets its own delegate tool — no recursion). `api` targets still obey
47
+ # the billing gate; delegated sub-calls are metered separately (see `tanglebrain --stats`). This
48
+ # example is an openai-compat backend; uncomment + point it at one you hold.
49
+ #
50
+ # - id: cheap-remote
51
+ # tier: sub
52
+ # invoke:
53
+ # kind: openai-compat
54
+ # base_url: "https://your-openai-compatible-endpoint/v1"
55
+ # model: "a-cheaper-model"
56
+ # key_ref: "file:~/.config/tanglebrain/keys/cheap.key" # a reference, never the raw key
57
+ # cost: cheap
58
+ # good_at: [grunt, summarization, extraction]
59
+ # can_delegate: true
60
+
61
+ # --- OPT-IN subscription / authenticated-CLI tier — COMMENTED OUT so it stays inert. ---
62
+ # These entries drive authenticated CLIs you already have installed and logged in (Claude Code /
63
+ # Codex / Gemini). They are OPT-IN: uncomment only the ones you use. Driving these CLIs is YOUR
64
+ # responsibility under each provider's terms of service — see DISCLAIMER.md. Once uncommented, an
65
+ # entry flagged `can_orchestrate: true` joins the orchestrator rotation (rotation + failover for
66
+ # resilience). Adapt the `cmd`/`parse`/`delegate_args` to your installed CLI's flags.
67
+ #
68
+ # - id: claude
69
+ # tier: sub
70
+ # invoke:
71
+ # kind: cli
72
+ # # --output-format json => one JSON object {"result": ..., "is_error": ...}; prompt appended.
73
+ # cmd: ["claude", "-p", "--output-format", "json"]
74
+ # parse: claude-json
75
+ # scrub_env: ["ANTHROPIC_API_KEY"] # use the CLI's own authenticated session, not an injected key
76
+ # # Per-invocation delegate injection: {delegate_mcp_json} is substituted with the local-delegate
77
+ # # MCP server config; the tool is mcp__tanglebrain-delegate__delegate_local.
78
+ # delegate_args: ["--mcp-config", "{delegate_mcp_json}", "--allowedTools", "mcp__tanglebrain-delegate__delegate_local", "--strict-mcp-config"]
79
+ # cost: subscription
80
+ # good_at: [reasoning, decomposition, review]
81
+ # can_orchestrate: true # joins the orchestrator rotation
82
+ #
83
+ # - id: codex
84
+ # tier: sub
85
+ # invoke:
86
+ # kind: cli
87
+ # # codex exec prints the answer to stdout (metadata to stderr); prompt appended.
88
+ # cmd: ["codex", "exec"]
89
+ # parse: plain
90
+ # # Register the delegate via -c TOML overrides (codex has no per-invocation --mcp-config), plus
91
+ # # the approval/sandbox bypass codex exec needs to auto-run a tool call headless.
92
+ # delegate_args: ["-c", "mcp_servers.tanglebrain_delegate.command={delegate_mcp_command}", "-c", "mcp_servers.tanglebrain_delegate.args=[\"-m\",\"tanglebrain.mcp_server\"]", "--dangerously-bypass-approvals-and-sandbox"]
93
+ # cost: subscription
94
+ # good_at: [code, agentic-code]
95
+ # can_orchestrate: true
96
+ #
97
+ # - id: gemini
98
+ # tier: sub
99
+ # invoke:
100
+ # kind: cli
101
+ # # -p requires the prompt as its value, so {prompt} marks the injection point.
102
+ # cmd: ["gemini", "-p", "{prompt}", "--output-format", "json"]
103
+ # parse: gemini-json
104
+ # # gemini has no per-invocation MCP-server config — it needs a one-time
105
+ # # gemini mcp add tanglebrain-delegate -- <python> -m tanglebrain.mcp_server
106
+ # # (see README); these per-invocation flags then allow + auto-approve the tool.
107
+ # delegate_args: ["--allowed-mcp-server-names", "tanglebrain-delegate", "--approval-mode", "yolo"]
108
+ # cost: subscription
109
+ # good_at: [long-context, structured-json]
110
+ # can_orchestrate: true
111
+
112
+ # --- Paid-API tier example (bring-your-own-key overflow) — COMMENTED OUT so it stays inert. ---
113
+ # Paid API is the genuine last resort. To enable: set api_billing_enabled: true in
114
+ # config/settings.yaml, then uncomment and point it at any OpenAI-compatible endpoint you hold a
115
+ # key for (OpenRouter, a self-hosted LiteLLM gateway, a provider directly, etc.). Prefer fronting
116
+ # it through a budget-capped gateway/virtual key. key_ref is a REFERENCE — never the raw key.
117
+ # budget_usd_month is recorded for visibility; enforce the hard cap at your gateway/provider.
118
+ #
119
+ # - id: paid-overflow
120
+ # tier: api
121
+ # invoke:
122
+ # kind: api
123
+ # base_url: "https://your-openai-compatible-endpoint/v1" # e.g. OpenRouter or your gateway
124
+ # model: "your-model" # the model id that endpoint exposes
125
+ # key_ref: "env:OPENAI_API_KEY" # or file:~/.config/tanglebrain/keys/paid.key
126
+ # cost: paid # real $ — last resort only
127
+ # good_at: [reasoning, hard]
128
+ # enabled: true # per-key kill-switch (false => never routable, even if billing on)
129
+ # budget_usd_month: 25 # display-only in v1; LiteLLM enforces the hard cap on the key
@@ -0,0 +1,39 @@
1
+ # TangleBrain global settings — knobs that are NOT per roster entry.
2
+ #
3
+ # The roster (roster.yaml) is a list of routable backends; per-entry policy lives there. The few
4
+ # truly global switches live here.
5
+ #
6
+ # api_billing_enabled — THE PAID-API BILLING GATE. This is the durable safety contract: paid
7
+ # billing is OFF unless this is explicitly `true`.
8
+ #
9
+ # false (default) → every `tier: api` roster entry still parses and is inspectable, but is
10
+ # NEVER routable — the adapter factory refuses to build it. Inert.
11
+ # true → enabled `tier: api` entries become routable (still last-resort), each
12
+ # fronted through a budget-scoped key (key_ref) on an OpenAI-compatible gateway.
13
+ #
14
+ # The durable rule: *no paid billing without this explicit toggle.* Keep it false unless you mean it.
15
+
16
+ api_billing_enabled: false
17
+
18
+ # classifier_gate_enabled — the LOCAL CLASSIFIER GATE (default false = normal frontier-first).
19
+ #
20
+ # false (default) → every request goes through the frontier-first router (orchestrator rotation).
21
+ # true → a cheap local classify runs FIRST: trivial requests are handled directly by
22
+ # the free local backend and never reach an orchestrator; only frontier requests
23
+ # do. Per CLI run, `--gate` / `--no-gate` overrides this.
24
+ #
25
+ # Classification fails safe to "frontier" — it never traps a hard task on local.
26
+
27
+ classifier_gate_enabled: false
28
+
29
+ # delegate_max_concurrency — cap on how many sub-tasks `delegate_many` fans out AT ONCE.
30
+ #
31
+ # unset (default) → TangleBrain derives the cap from this machine (os.cpu_count()).
32
+ # <positive int> → pin the cap. The TRUE limit is your backend's parallelism, which TangleBrain
33
+ # can't portably detect — set this to match it (e.g. your local model server's
34
+ # OLLAMA_NUM_PARALLEL, or your provider's safe concurrent-request budget).
35
+ #
36
+ # A per-call `max_concurrency` argument may LOWER this but never exceed it. Left unset below so the
37
+ # derived default applies out of the box.
38
+ #
39
+ # delegate_max_concurrency: 4