tanglebrain 0.16.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- tanglebrain/__init__.py +23 -0
- tanglebrain/adapters/__init__.py +15 -0
- tanglebrain/adapters/api.py +65 -0
- tanglebrain/adapters/base.py +46 -0
- tanglebrain/adapters/cli.py +341 -0
- tanglebrain/adapters/openai_compat.py +197 -0
- tanglebrain/classifier.py +99 -0
- tanglebrain/cli.py +251 -0
- tanglebrain/config/pricing.yaml +14 -0
- tanglebrain/config/roster.yaml +129 -0
- tanglebrain/config/settings.yaml +39 -0
- tanglebrain/delegate.py +485 -0
- tanglebrain/gui/__init__.py +10 -0
- tanglebrain/gui/server.py +162 -0
- tanglebrain/gui/static/index.html +295 -0
- tanglebrain/gui/static/logo.png +0 -0
- tanglebrain/gui/views.py +180 -0
- tanglebrain/mcp_server.py +208 -0
- tanglebrain/measurement.py +548 -0
- tanglebrain/roster.py +415 -0
- tanglebrain/roster_edit.py +201 -0
- tanglebrain/router.py +232 -0
- tanglebrain/selector.py +117 -0
- tanglebrain/settings.py +132 -0
- tanglebrain-0.16.0.dist-info/METADATA +369 -0
- tanglebrain-0.16.0.dist-info/RECORD +30 -0
- tanglebrain-0.16.0.dist-info/WHEEL +5 -0
- tanglebrain-0.16.0.dist-info/entry_points.txt +4 -0
- tanglebrain-0.16.0.dist-info/licenses/LICENSE +21 -0
- tanglebrain-0.16.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
"""Cheap local classifier gate — OFF by default.
|
|
2
|
+
|
|
3
|
+
The default routing strategy is frontier-first with orchestrator rotation. An optional escape valve:
|
|
4
|
+
put a **cheap local classifier in front** that does one narrow job — decide whether a request is
|
|
5
|
+
*trivial* (the free local backend can fully handle it) or *needs-frontier* (route to the
|
|
6
|
+
orchestrator). Trivial requests then skip the orchestrators entirely.
|
|
7
|
+
|
|
8
|
+
Two deliberate design rules:
|
|
9
|
+
|
|
10
|
+
- **Narrow classification, not self-judgement.** The classifier rates *task complexity*, it does not
|
|
11
|
+
ask the local model "can YOU do this?" — that framing is unreliable.
|
|
12
|
+
- **Fail safe toward the capable path.** Any ambiguity, parse miss, or classifier error resolves to
|
|
13
|
+
``frontier``. The gate must never trap a hard task on the local tier because the classifier was
|
|
14
|
+
unsure or broke — at worst it falls back to normal frontier-first routing.
|
|
15
|
+
|
|
16
|
+
It is inert unless explicitly enabled (``classifier_gate_enabled`` in settings, or ``--gate`` on the
|
|
17
|
+
CLI).
|
|
18
|
+
"""
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import re
|
|
22
|
+
from typing import Callable
|
|
23
|
+
|
|
24
|
+
from tanglebrain.adapters import OpenAICompatAdapter
|
|
25
|
+
from tanglebrain.adapters.base import Adapter
|
|
26
|
+
from tanglebrain.roster import Roster, load_roster
|
|
27
|
+
from tanglebrain.selector import select_local
|
|
28
|
+
|
|
29
|
+
TRIVIAL = "trivial"
|
|
30
|
+
FRONTIER = "frontier"
|
|
31
|
+
|
|
32
|
+
# A local reasoning model spends part of its budget on internal reasoning, so give the classify call
|
|
33
|
+
# enough headroom to finish reasoning AND emit the verdict; a truncated (null) response just fails
|
|
34
|
+
# safe to FRONTIER. Kept modest because this runs in front of every gated request.
|
|
35
|
+
CLASSIFY_MAX_TOKENS = 1024
|
|
36
|
+
|
|
37
|
+
_INSTRUCTIONS = (
|
|
38
|
+
"You are a routing classifier. Decide how complex the USER REQUEST below is, so a dispatcher "
|
|
39
|
+
"can send simple work to a small local model and hard work to a frontier model.\n\n"
|
|
40
|
+
"Classify by the TASK's intrinsic complexity (not by who should do it):\n"
|
|
41
|
+
"- TRIVIAL: short, well-specified, single-step work — a factual lookup, a small/simple code "
|
|
42
|
+
"snippet, a quick rewrite or format, a direct question with a known answer.\n"
|
|
43
|
+
"- FRONTIER: anything needing multi-step reasoning, decomposition, architecture or design, "
|
|
44
|
+
"debugging across files, careful trade-offs, or ambiguous/open-ended judgement.\n\n"
|
|
45
|
+
"Decide quickly; do not overthink and do not attempt the task. When unsure, answer FRONTIER."
|
|
46
|
+
)
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def _build_prompt(user_request: str) -> str:
|
|
50
|
+
"""Fold the classify instructions and the request into one user message (the adapter sends one)."""
|
|
51
|
+
return (
|
|
52
|
+
f"{_INSTRUCTIONS}\n\n--- USER REQUEST ---\n{user_request}\n--- END REQUEST ---\n\n"
|
|
53
|
+
"Answer with exactly one word: TRIVIAL or FRONTIER."
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _parse_verdict(text: str) -> str:
|
|
58
|
+
"""Map a classifier response to ``TRIVIAL`` or ``FRONTIER``, defaulting to ``FRONTIER``.
|
|
59
|
+
|
|
60
|
+
We instruct the model to answer with exactly one word, so the verdict is ``TRIVIAL`` **only** when
|
|
61
|
+
the response's first word token is exactly ``trivial`` and ``frontier`` appears nowhere. Anything
|
|
62
|
+
else — ``frontier``, a both-words answer, prose, a *negation* like "not trivial", junk, empty — is
|
|
63
|
+
``FRONTIER`` (the safe default). The first-token rule is deliberately strict: free-form prose can't
|
|
64
|
+
leak a spurious ``TRIVIAL`` and strand a hard task on local.
|
|
65
|
+
"""
|
|
66
|
+
words = re.findall(r"[a-z]+", (text or "").lower())
|
|
67
|
+
if "frontier" in words:
|
|
68
|
+
return FRONTIER
|
|
69
|
+
return TRIVIAL if words[:1] == ["trivial"] else FRONTIER
|
|
70
|
+
|
|
71
|
+
|
|
72
|
+
def classify(
|
|
73
|
+
prompt: str,
|
|
74
|
+
roster: Roster | None = None,
|
|
75
|
+
adapter_factory: Callable[..., Adapter] = OpenAICompatAdapter.from_entry,
|
|
76
|
+
max_tokens: int = CLASSIFY_MAX_TOKENS,
|
|
77
|
+
) -> str:
|
|
78
|
+
"""Classify ``prompt`` as :data:`TRIVIAL` or :data:`FRONTIER` using free local gpt-oss.
|
|
79
|
+
|
|
80
|
+
Never raises and never blocks routing: any failure (no local entry, transport error, truncated
|
|
81
|
+
or unparsable response) resolves to :data:`FRONTIER`, so a broken classifier degrades to today's
|
|
82
|
+
normal frontier-first routing rather than trapping a task on the local tier.
|
|
83
|
+
|
|
84
|
+
Args:
|
|
85
|
+
prompt: The user request to classify.
|
|
86
|
+
roster: The loaded roster (defaults to the packaged roster).
|
|
87
|
+
adapter_factory: Builds the local adapter from the selected entry (injectable for tests).
|
|
88
|
+
max_tokens: Budget for the classify call (gpt-oss needs reasoning headroom; see the constant).
|
|
89
|
+
|
|
90
|
+
Returns:
|
|
91
|
+
:data:`TRIVIAL` or :data:`FRONTIER`.
|
|
92
|
+
"""
|
|
93
|
+
try:
|
|
94
|
+
entry = select_local(roster if roster is not None else load_roster())
|
|
95
|
+
adapter = adapter_factory(entry)
|
|
96
|
+
text = adapter.run(_build_prompt(prompt), {"max_tokens": max_tokens})
|
|
97
|
+
except Exception: # noqa: BLE001 — deliberately total: a broken classifier must fail safe, never block
|
|
98
|
+
return FRONTIER
|
|
99
|
+
return _parse_verdict(text)
|
tanglebrain/cli.py
ADDED
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
"""TangleBrain CLI — route one request and print the response.
|
|
2
|
+
|
|
3
|
+
Thin wiring over :func:`run_once`; the routing logic lives in the router/selector/adapters. The
|
|
4
|
+
path is chosen by flag precedence ``--model`` > ``--local`` > the frontier-first router (the
|
|
5
|
+
default): the router selects + rotates an orchestrator, fails over on errors, and gives it the
|
|
6
|
+
local-delegate tool so it offloads sub-tasks to the free local backend.
|
|
7
|
+
|
|
8
|
+
Usage::
|
|
9
|
+
|
|
10
|
+
tanglebrain "Refactor this module and add tests." # default: frontier-first router
|
|
11
|
+
tanglebrain --task code "..." # task-fit hint for the router
|
|
12
|
+
tanglebrain --local "Write a haiku about local inference." # force the free local tier
|
|
13
|
+
tanglebrain --model gemini "Summarize this long document." # pin a specific roster entry
|
|
14
|
+
"""
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import argparse
|
|
18
|
+
import sys
|
|
19
|
+
import uuid
|
|
20
|
+
|
|
21
|
+
from tanglebrain import __version__
|
|
22
|
+
from tanglebrain.adapters import AdapterError
|
|
23
|
+
from tanglebrain.classifier import TRIVIAL, classify
|
|
24
|
+
from tanglebrain.measurement import (
|
|
25
|
+
format_rollup,
|
|
26
|
+
load_pricing,
|
|
27
|
+
read_records,
|
|
28
|
+
record_task,
|
|
29
|
+
rollup,
|
|
30
|
+
)
|
|
31
|
+
from tanglebrain.roster import RosterError, load_roster
|
|
32
|
+
from tanglebrain.router import Router, RouterError
|
|
33
|
+
from tanglebrain.selector import SelectionError, build_adapter, select_by_id, select_local
|
|
34
|
+
from tanglebrain.settings import load_settings
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
38
|
+
"""Build the argument parser for the ``tanglebrain`` command.
|
|
39
|
+
|
|
40
|
+
Returns:
|
|
41
|
+
The configured :class:`argparse.ArgumentParser`.
|
|
42
|
+
"""
|
|
43
|
+
parser = argparse.ArgumentParser(
|
|
44
|
+
prog="tanglebrain",
|
|
45
|
+
description=(
|
|
46
|
+
"Route one request to the cheapest capable tier (frontier-first by default), or "
|
|
47
|
+
"print the 'spend avoided' rollup with --stats."
|
|
48
|
+
),
|
|
49
|
+
)
|
|
50
|
+
parser.add_argument(
|
|
51
|
+
"--version",
|
|
52
|
+
action="version",
|
|
53
|
+
version=f"%(prog)s {__version__}",
|
|
54
|
+
help="Print the TangleBrain version and exit.",
|
|
55
|
+
)
|
|
56
|
+
parser.add_argument(
|
|
57
|
+
"prompt",
|
|
58
|
+
nargs="?",
|
|
59
|
+
default=None,
|
|
60
|
+
help="The prompt to route. Optional only when --stats is given.",
|
|
61
|
+
)
|
|
62
|
+
parser.add_argument(
|
|
63
|
+
"--roster",
|
|
64
|
+
default=None,
|
|
65
|
+
help="Path to a roster YAML (defaults to the packaged tanglebrain/config/roster.yaml).",
|
|
66
|
+
)
|
|
67
|
+
parser.add_argument(
|
|
68
|
+
"--model",
|
|
69
|
+
default=None,
|
|
70
|
+
help=(
|
|
71
|
+
"Route to a specific roster entry by id (e.g. 'claude'). Without it, the default "
|
|
72
|
+
"local-first selection is used. This is an explicit override of routing."
|
|
73
|
+
),
|
|
74
|
+
)
|
|
75
|
+
parser.add_argument(
|
|
76
|
+
"--local",
|
|
77
|
+
action="store_true",
|
|
78
|
+
help=(
|
|
79
|
+
"Force the free local tier (gpt-oss) instead of the default frontier-first router. "
|
|
80
|
+
"Use for a quick, $0, no-orchestration answer."
|
|
81
|
+
),
|
|
82
|
+
)
|
|
83
|
+
parser.add_argument(
|
|
84
|
+
"--route",
|
|
85
|
+
action="store_true",
|
|
86
|
+
help="Deprecated/no-op: the frontier-first router is now the default. Kept for back-compat.",
|
|
87
|
+
)
|
|
88
|
+
parser.add_argument(
|
|
89
|
+
"--task",
|
|
90
|
+
default=None,
|
|
91
|
+
help="Task-fit hint for the router (a good_at tag, e.g. 'code', 'reasoning', 'long-context').",
|
|
92
|
+
)
|
|
93
|
+
gate_group = parser.add_mutually_exclusive_group()
|
|
94
|
+
gate_group.add_argument(
|
|
95
|
+
"--gate",
|
|
96
|
+
dest="gate",
|
|
97
|
+
action="store_true",
|
|
98
|
+
default=None,
|
|
99
|
+
help="Force the local classifier gate ON for this run: a cheap local classify sends "
|
|
100
|
+
"trivial requests straight to the free local backend, and only frontier ones to an "
|
|
101
|
+
"orchestrator.",
|
|
102
|
+
)
|
|
103
|
+
gate_group.add_argument(
|
|
104
|
+
"--no-gate",
|
|
105
|
+
dest="gate",
|
|
106
|
+
action="store_false",
|
|
107
|
+
help="Force the classifier gate OFF (always frontier-first router), ignoring the setting.",
|
|
108
|
+
)
|
|
109
|
+
parser.add_argument(
|
|
110
|
+
"--max-tokens",
|
|
111
|
+
type=int,
|
|
112
|
+
default=None,
|
|
113
|
+
help="Override the completion token cap (defaults to the adapter's 2048).",
|
|
114
|
+
)
|
|
115
|
+
parser.add_argument(
|
|
116
|
+
"--stats",
|
|
117
|
+
action="store_true",
|
|
118
|
+
help=(
|
|
119
|
+
"Print the 'spend avoided' rollup (cloud-equivalent cost of every routed task so far) "
|
|
120
|
+
"and exit. No prompt needed."
|
|
121
|
+
),
|
|
122
|
+
)
|
|
123
|
+
return parser
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def _served(path: str, entry) -> dict | None:
|
|
127
|
+
"""Build the ``{path, tier, model}`` served-summary for a routed task, or ``None``."""
|
|
128
|
+
if entry is None:
|
|
129
|
+
return None
|
|
130
|
+
return {"path": path, "tier": entry.tier, "model": entry.id}
|
|
131
|
+
|
|
132
|
+
|
|
133
|
+
def run_once(
|
|
134
|
+
prompt: str,
|
|
135
|
+
roster_path: str | None = None,
|
|
136
|
+
max_tokens: int | None = None,
|
|
137
|
+
model: str | None = None,
|
|
138
|
+
local: bool = False,
|
|
139
|
+
task: str | None = None,
|
|
140
|
+
return_served: bool = False,
|
|
141
|
+
gate: bool | None = None,
|
|
142
|
+
):
|
|
143
|
+
"""Route a single prompt to a roster tier and return the response text.
|
|
144
|
+
|
|
145
|
+
Paths, in precedence order:
|
|
146
|
+
|
|
147
|
+
- ``model`` set → select that named entry explicitly (an override, not a routing decision).
|
|
148
|
+
- ``local`` true → the free local tier directly, no orchestration.
|
|
149
|
+
- otherwise → the default routing path. With the **classifier gate** off (the default), this
|
|
150
|
+
is **the frontier-first** :class:`~tanglebrain.router.Router`: task-fit orchestrator selection +
|
|
151
|
+
rotation + failover across the orchestrators, each given the local-delegate tool. With the gate
|
|
152
|
+
on, a cheap local classify runs first: a *trivial* request is handled directly on the free local
|
|
153
|
+
backend (path ``gate-local``, skipping the orchestrators), and everything else falls through to
|
|
154
|
+
the router.
|
|
155
|
+
|
|
156
|
+
Args:
|
|
157
|
+
prompt: The prompt to route.
|
|
158
|
+
roster_path: Optional roster YAML path (defaults to the packaged roster).
|
|
159
|
+
max_tokens: Optional completion token cap (honoured by the openai-compat adapter; the
|
|
160
|
+
CLI adapter ignores it, as each CLI controls its own limits).
|
|
161
|
+
model: Optional roster entry id to route to explicitly.
|
|
162
|
+
local: Force the free local tier instead of the frontier-first router.
|
|
163
|
+
task: Optional task-fit hint for the router (a ``good_at`` tag).
|
|
164
|
+
return_served: When ``True``, return ``(text, served)`` where ``served`` is
|
|
165
|
+
``{path, tier, model}`` for the entry that served the task (or ``None`` if unknown).
|
|
166
|
+
The GUI uses this so it needn't re-read the usage log. Default ``False`` returns the
|
|
167
|
+
plain text string, so existing callers (``main``) are unchanged.
|
|
168
|
+
gate: Override for the classifier gate on the default path. ``None`` (default) uses the
|
|
169
|
+
``classifier_gate_enabled`` setting; ``True``/``False`` force the gate on/off for this
|
|
170
|
+
call. Ignored when ``model`` or ``local`` is set.
|
|
171
|
+
|
|
172
|
+
Returns:
|
|
173
|
+
The response text (``str``), or ``(text, served)`` when ``return_served`` is ``True``.
|
|
174
|
+
|
|
175
|
+
Raises:
|
|
176
|
+
RosterError: If the roster cannot be loaded.
|
|
177
|
+
SelectionError: If ``model``/``local`` is used and no suitable entry is available.
|
|
178
|
+
RouterError: If the router runs and no orchestrator can serve the request.
|
|
179
|
+
AdapterError: If the adapter cannot produce text.
|
|
180
|
+
"""
|
|
181
|
+
roster = load_roster(roster_path)
|
|
182
|
+
# Mint a task id for this routed task. It is recorded on the task and threaded through opts so
|
|
183
|
+
# the orchestrator-CLI adapter can propagate it to delegated sub-calls (see CliAdapter.run /
|
|
184
|
+
# PARENT_TASK_ID_ENV), linking the delegation tree back to this task. Cheap and side-effect-free
|
|
185
|
+
# to mint on every path; only the router path (orchestrators with the delegate tool) acts on it.
|
|
186
|
+
task_id = uuid.uuid4().hex
|
|
187
|
+
opts: dict = {"task_id": task_id}
|
|
188
|
+
if max_tokens is not None:
|
|
189
|
+
opts["max_tokens"] = max_tokens
|
|
190
|
+
|
|
191
|
+
if model is not None:
|
|
192
|
+
path, entry = "model", select_by_id(roster, model)
|
|
193
|
+
text = build_adapter(entry).run(prompt, opts)
|
|
194
|
+
elif local:
|
|
195
|
+
path, entry = "local", select_local(roster)
|
|
196
|
+
text = build_adapter(entry).run(prompt, opts)
|
|
197
|
+
else:
|
|
198
|
+
gate_on = load_settings().classifier_gate_enabled if gate is None else gate
|
|
199
|
+
if gate_on and classify(prompt, roster=roster) == TRIVIAL:
|
|
200
|
+
# classifier gate: a trivial request skips the orchestrators and is handled directly on
|
|
201
|
+
# the free local backend. Frontier (or any classifier failure) falls through to the router.
|
|
202
|
+
path, entry = "gate-local", select_local(roster)
|
|
203
|
+
text = build_adapter(entry).run(prompt, opts)
|
|
204
|
+
else:
|
|
205
|
+
path = "router"
|
|
206
|
+
router = Router(roster)
|
|
207
|
+
text = router.route(prompt, task=task, opts=opts)
|
|
208
|
+
entry = router.last_served
|
|
209
|
+
|
|
210
|
+
record_task(path=path, entry=entry, prompt=prompt, response=text, task_id=task_id)
|
|
211
|
+
return (text, _served(path, entry)) if return_served else text
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
def main(argv: list[str] | None = None) -> int:
|
|
215
|
+
"""Console entry point.
|
|
216
|
+
|
|
217
|
+
Args:
|
|
218
|
+
argv: Optional argument list (defaults to ``sys.argv[1:]``).
|
|
219
|
+
|
|
220
|
+
Returns:
|
|
221
|
+
Process exit code: ``0`` on success, ``1`` on a known TangleBrain error.
|
|
222
|
+
"""
|
|
223
|
+
parser = build_parser()
|
|
224
|
+
args = parser.parse_args(argv)
|
|
225
|
+
|
|
226
|
+
if args.stats:
|
|
227
|
+
print(format_rollup(rollup(read_records()), load_pricing()))
|
|
228
|
+
return 0
|
|
229
|
+
|
|
230
|
+
if args.prompt is None:
|
|
231
|
+
parser.error("prompt is required (unless --stats is given)")
|
|
232
|
+
|
|
233
|
+
try:
|
|
234
|
+
text = run_once(
|
|
235
|
+
args.prompt,
|
|
236
|
+
roster_path=args.roster,
|
|
237
|
+
max_tokens=args.max_tokens,
|
|
238
|
+
model=args.model,
|
|
239
|
+
local=args.local,
|
|
240
|
+
task=args.task,
|
|
241
|
+
gate=args.gate,
|
|
242
|
+
)
|
|
243
|
+
except (RosterError, SelectionError, RouterError, AdapterError) as exc:
|
|
244
|
+
print(f"tanglebrain: {exc}", file=sys.stderr)
|
|
245
|
+
return 1
|
|
246
|
+
print(text)
|
|
247
|
+
return 0
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
if __name__ == "__main__":
|
|
251
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,14 @@
|
|
|
1
|
+
# Cloud-equivalent reference pricing for the "spend avoided" rollup.
|
|
2
|
+
#
|
|
3
|
+
# Methodology: for each routed task, estimate what it WOULD have cost on a paid frontier API,
|
|
4
|
+
# valued at a reference frontier price — here Claude Sonnet, $3/MTok input, $15/MTok output. The
|
|
5
|
+
# rollup multiplies estimated tokens (a chars/4 heuristic over the visible prompt + response) by
|
|
6
|
+
# these rates. Values are US dollars per 1,000,000 tokens. Tune them to whatever frontier model you
|
|
7
|
+
# want to compare against.
|
|
8
|
+
#
|
|
9
|
+
# Set `placeholder: true` if your rates are rough/illustrative — the `tanglebrain --stats` rollup
|
|
10
|
+
# then flags every figure as PLACEHOLDER so no number is mistaken for a precise cost.
|
|
11
|
+
placeholder: false
|
|
12
|
+
reference_model: "Claude Sonnet ($3/$15 per MTok)"
|
|
13
|
+
input_per_mtok: 3.00
|
|
14
|
+
output_per_mtok: 15.00
|
|
@@ -0,0 +1,129 @@
|
|
|
1
|
+
# TangleBrain roster — a GENERIC EXAMPLE list of routable models. Edit it to your own backends.
|
|
2
|
+
#
|
|
3
|
+
# This shipped file is just a starting point. Your REAL roster lives OUTSIDE the repo and is
|
|
4
|
+
# auto-discovered, so a `git pull` never clobbers it. Resolution order (first hit wins):
|
|
5
|
+
# 1. $TANGLEBRAIN_ROSTER (explicit path)
|
|
6
|
+
# 2. ~/.config/tanglebrain/roster.yaml (your roster — recommended)
|
|
7
|
+
# 3. this packaged example (fallback)
|
|
8
|
+
# Copy this file to ~/.config/tanglebrain/roster.yaml and edit it there.
|
|
9
|
+
#
|
|
10
|
+
# Add / remove / reorganize entries freely — adding a model is a config edit, never a code change.
|
|
11
|
+
# Flag `can_orchestrate: true` to put an entry in the orchestrator rotation. Flag
|
|
12
|
+
# `can_delegate: true` to make an entry a delegate target an orchestrator can hand a sub-task to
|
|
13
|
+
# (via the `delegate` MCP tool), either by naming its id or by capability — `delegate(task: code)`
|
|
14
|
+
# picks the cheapest `can_delegate` entry whose `good_at` lists that tag (local before sub; paid
|
|
15
|
+
# `api` is never auto-selected). The free local model is always available as the default target.
|
|
16
|
+
#
|
|
17
|
+
# invoke.kind ∈ {openai-compat, cli, api}. key_ref ∈ {file:PATH, env:NAME, none} — a
|
|
18
|
+
# credential *reference*, never a raw secret.
|
|
19
|
+
#
|
|
20
|
+
# For `cli` entries: a literal `{prompt}` token in `cmd` is replaced with the prompt; with no
|
|
21
|
+
# token the prompt is appended as the final argument. `parse` names the output parser
|
|
22
|
+
# (claude-json | gemini-json | plain). `scrub_env` strips env vars from the subprocess.
|
|
23
|
+
#
|
|
24
|
+
# The paid-API tier (kind: api) is gated behind `api_billing_enabled` (config/settings.yaml,
|
|
25
|
+
# default false) AND each entry's own `enabled` flag — it parses but stays inert until both are
|
|
26
|
+
# on. A commented example is at the bottom of this file.
|
|
27
|
+
|
|
28
|
+
# Free local tier — any OpenAI-compatible local server. This example points at Ollama on its
|
|
29
|
+
# default port; run `ollama pull llama3.2` (or any model) first, or swap in your own endpoint.
|
|
30
|
+
- id: local-ollama
|
|
31
|
+
tier: local
|
|
32
|
+
invoke:
|
|
33
|
+
kind: openai-compat
|
|
34
|
+
base_url: "http://localhost:11434/v1"
|
|
35
|
+
model: "llama3.2"
|
|
36
|
+
# Local Ollama needs no auth, so no key_ref. For a keyed endpoint, add e.g.
|
|
37
|
+
# key_ref: "file:~/.config/tanglebrain/keys/local.key" (a reference, never the raw key).
|
|
38
|
+
cost: free
|
|
39
|
+
good_at: [grunt, code, tools]
|
|
40
|
+
can_delegate: true # an orchestrator may target this backend by id via the delegate tool
|
|
41
|
+
|
|
42
|
+
# --- OPT-IN delegate target (non-local) — COMMENTED OUT so it stays inert. ---
|
|
43
|
+
# A delegate target is a backend an orchestrator can hand a *sub-task* to (not the whole request) —
|
|
44
|
+
# e.g. a cheaper sub or a better-fit model. Flag any entry `can_delegate: true` and it joins the
|
|
45
|
+
# menu the `delegate` tool advertises; the model picks a target by its `good_at` fit. A target is
|
|
46
|
+
# invoked as a leaf (it never gets its own delegate tool — no recursion). `api` targets still obey
|
|
47
|
+
# the billing gate; delegated sub-calls are metered separately (see `tanglebrain --stats`). This
|
|
48
|
+
# example is an openai-compat backend; uncomment + point it at one you hold.
|
|
49
|
+
#
|
|
50
|
+
# - id: cheap-remote
|
|
51
|
+
# tier: sub
|
|
52
|
+
# invoke:
|
|
53
|
+
# kind: openai-compat
|
|
54
|
+
# base_url: "https://your-openai-compatible-endpoint/v1"
|
|
55
|
+
# model: "a-cheaper-model"
|
|
56
|
+
# key_ref: "file:~/.config/tanglebrain/keys/cheap.key" # a reference, never the raw key
|
|
57
|
+
# cost: cheap
|
|
58
|
+
# good_at: [grunt, summarization, extraction]
|
|
59
|
+
# can_delegate: true
|
|
60
|
+
|
|
61
|
+
# --- OPT-IN subscription / authenticated-CLI tier — COMMENTED OUT so it stays inert. ---
|
|
62
|
+
# These entries drive authenticated CLIs you already have installed and logged in (Claude Code /
|
|
63
|
+
# Codex / Gemini). They are OPT-IN: uncomment only the ones you use. Driving these CLIs is YOUR
|
|
64
|
+
# responsibility under each provider's terms of service — see DISCLAIMER.md. Once uncommented, an
|
|
65
|
+
# entry flagged `can_orchestrate: true` joins the orchestrator rotation (rotation + failover for
|
|
66
|
+
# resilience). Adapt the `cmd`/`parse`/`delegate_args` to your installed CLI's flags.
|
|
67
|
+
#
|
|
68
|
+
# - id: claude
|
|
69
|
+
# tier: sub
|
|
70
|
+
# invoke:
|
|
71
|
+
# kind: cli
|
|
72
|
+
# # --output-format json => one JSON object {"result": ..., "is_error": ...}; prompt appended.
|
|
73
|
+
# cmd: ["claude", "-p", "--output-format", "json"]
|
|
74
|
+
# parse: claude-json
|
|
75
|
+
# scrub_env: ["ANTHROPIC_API_KEY"] # use the CLI's own authenticated session, not an injected key
|
|
76
|
+
# # Per-invocation delegate injection: {delegate_mcp_json} is substituted with the local-delegate
|
|
77
|
+
# # MCP server config; the tool is mcp__tanglebrain-delegate__delegate_local.
|
|
78
|
+
# delegate_args: ["--mcp-config", "{delegate_mcp_json}", "--allowedTools", "mcp__tanglebrain-delegate__delegate_local", "--strict-mcp-config"]
|
|
79
|
+
# cost: subscription
|
|
80
|
+
# good_at: [reasoning, decomposition, review]
|
|
81
|
+
# can_orchestrate: true # joins the orchestrator rotation
|
|
82
|
+
#
|
|
83
|
+
# - id: codex
|
|
84
|
+
# tier: sub
|
|
85
|
+
# invoke:
|
|
86
|
+
# kind: cli
|
|
87
|
+
# # codex exec prints the answer to stdout (metadata to stderr); prompt appended.
|
|
88
|
+
# cmd: ["codex", "exec"]
|
|
89
|
+
# parse: plain
|
|
90
|
+
# # Register the delegate via -c TOML overrides (codex has no per-invocation --mcp-config), plus
|
|
91
|
+
# # the approval/sandbox bypass codex exec needs to auto-run a tool call headless.
|
|
92
|
+
# delegate_args: ["-c", "mcp_servers.tanglebrain_delegate.command={delegate_mcp_command}", "-c", "mcp_servers.tanglebrain_delegate.args=[\"-m\",\"tanglebrain.mcp_server\"]", "--dangerously-bypass-approvals-and-sandbox"]
|
|
93
|
+
# cost: subscription
|
|
94
|
+
# good_at: [code, agentic-code]
|
|
95
|
+
# can_orchestrate: true
|
|
96
|
+
#
|
|
97
|
+
# - id: gemini
|
|
98
|
+
# tier: sub
|
|
99
|
+
# invoke:
|
|
100
|
+
# kind: cli
|
|
101
|
+
# # -p requires the prompt as its value, so {prompt} marks the injection point.
|
|
102
|
+
# cmd: ["gemini", "-p", "{prompt}", "--output-format", "json"]
|
|
103
|
+
# parse: gemini-json
|
|
104
|
+
# # gemini has no per-invocation MCP-server config — it needs a one-time
|
|
105
|
+
# # gemini mcp add tanglebrain-delegate -- <python> -m tanglebrain.mcp_server
|
|
106
|
+
# # (see README); these per-invocation flags then allow + auto-approve the tool.
|
|
107
|
+
# delegate_args: ["--allowed-mcp-server-names", "tanglebrain-delegate", "--approval-mode", "yolo"]
|
|
108
|
+
# cost: subscription
|
|
109
|
+
# good_at: [long-context, structured-json]
|
|
110
|
+
# can_orchestrate: true
|
|
111
|
+
|
|
112
|
+
# --- Paid-API tier example (bring-your-own-key overflow) — COMMENTED OUT so it stays inert. ---
|
|
113
|
+
# Paid API is the genuine last resort. To enable: set api_billing_enabled: true in
|
|
114
|
+
# config/settings.yaml, then uncomment and point it at any OpenAI-compatible endpoint you hold a
|
|
115
|
+
# key for (OpenRouter, a self-hosted LiteLLM gateway, a provider directly, etc.). Prefer fronting
|
|
116
|
+
# it through a budget-capped gateway/virtual key. key_ref is a REFERENCE — never the raw key.
|
|
117
|
+
# budget_usd_month is recorded for visibility; enforce the hard cap at your gateway/provider.
|
|
118
|
+
#
|
|
119
|
+
# - id: paid-overflow
|
|
120
|
+
# tier: api
|
|
121
|
+
# invoke:
|
|
122
|
+
# kind: api
|
|
123
|
+
# base_url: "https://your-openai-compatible-endpoint/v1" # e.g. OpenRouter or your gateway
|
|
124
|
+
# model: "your-model" # the model id that endpoint exposes
|
|
125
|
+
# key_ref: "env:OPENAI_API_KEY" # or file:~/.config/tanglebrain/keys/paid.key
|
|
126
|
+
# cost: paid # real $ — last resort only
|
|
127
|
+
# good_at: [reasoning, hard]
|
|
128
|
+
# enabled: true # per-key kill-switch (false => never routable, even if billing on)
|
|
129
|
+
# budget_usd_month: 25 # display-only in v1; LiteLLM enforces the hard cap on the key
|
|
@@ -0,0 +1,39 @@
|
|
|
1
|
+
# TangleBrain global settings — knobs that are NOT per roster entry.
|
|
2
|
+
#
|
|
3
|
+
# The roster (roster.yaml) is a list of routable backends; per-entry policy lives there. The few
|
|
4
|
+
# truly global switches live here.
|
|
5
|
+
#
|
|
6
|
+
# api_billing_enabled — THE PAID-API BILLING GATE. This is the durable safety contract: paid
|
|
7
|
+
# billing is OFF unless this is explicitly `true`.
|
|
8
|
+
#
|
|
9
|
+
# false (default) → every `tier: api` roster entry still parses and is inspectable, but is
|
|
10
|
+
# NEVER routable — the adapter factory refuses to build it. Inert.
|
|
11
|
+
# true → enabled `tier: api` entries become routable (still last-resort), each
|
|
12
|
+
# fronted through a budget-scoped key (key_ref) on an OpenAI-compatible gateway.
|
|
13
|
+
#
|
|
14
|
+
# The durable rule: *no paid billing without this explicit toggle.* Keep it false unless you mean it.
|
|
15
|
+
|
|
16
|
+
api_billing_enabled: false
|
|
17
|
+
|
|
18
|
+
# classifier_gate_enabled — the LOCAL CLASSIFIER GATE (default false = normal frontier-first).
|
|
19
|
+
#
|
|
20
|
+
# false (default) → every request goes through the frontier-first router (orchestrator rotation).
|
|
21
|
+
# true → a cheap local classify runs FIRST: trivial requests are handled directly by
|
|
22
|
+
# the free local backend and never reach an orchestrator; only frontier requests
|
|
23
|
+
# do. Per CLI run, `--gate` / `--no-gate` overrides this.
|
|
24
|
+
#
|
|
25
|
+
# Classification fails safe to "frontier" — it never traps a hard task on local.
|
|
26
|
+
|
|
27
|
+
classifier_gate_enabled: false
|
|
28
|
+
|
|
29
|
+
# delegate_max_concurrency — cap on how many sub-tasks `delegate_many` fans out AT ONCE.
|
|
30
|
+
#
|
|
31
|
+
# unset (default) → TangleBrain derives the cap from this machine (os.cpu_count()).
|
|
32
|
+
# <positive int> → pin the cap. The TRUE limit is your backend's parallelism, which TangleBrain
|
|
33
|
+
# can't portably detect — set this to match it (e.g. your local model server's
|
|
34
|
+
# OLLAMA_NUM_PARALLEL, or your provider's safe concurrent-request budget).
|
|
35
|
+
#
|
|
36
|
+
# A per-call `max_concurrency` argument may LOWER this but never exceed it. Left unset below so the
|
|
37
|
+
# derived default applies out of the box.
|
|
38
|
+
#
|
|
39
|
+
# delegate_max_concurrency: 4
|