evalroute 0.6.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
evalroute/__init__.py ADDED
@@ -0,0 +1,18 @@
1
+ """evalroute — model routing with a verified-success flywheel.
2
+
3
+ The library behind the Hermes plugin of the same name: a keyword-first
4
+ classifier that routes a task description to a (model, reasoning effort)
5
+ arm, a ledger that labels routes with pass/fail outcomes, and the Tier-A
6
+ harness that measures arms (cost per verified success) and feeds the
7
+ results back into the route table.
8
+
9
+ The Hermes plugin is a thin adapter: keppy/hermes-plugin-evalroute.
10
+ """
11
+
12
+ from __future__ import annotations
13
+
14
+ try:
15
+ from importlib.metadata import version as _version
16
+ __version__ = _version("evalroute")
17
+ except Exception: # pragma: no cover - running from a checkout, not installed
18
+ __version__ = "0.6.0"
@@ -0,0 +1,109 @@
1
+ """Gonogo adjudication for evalroute: is a route flip statistically real?
2
+
3
+ Reuses the merged gonogo plugin's stats (McNemar paired comparison, Wilson
4
+ intervals, INSUFFICIENT-EVIDENCE decisions) so evalroute never treats a
5
+ pilot-scale point-estimate gap as a measurement. Import is lazy and optional:
6
+ without gonogo installed, every stamp degrades to "unverified".
7
+
8
+ Three stamps:
9
+ distinguishable(winner, runner, per_task_outcomes) -> str|None
10
+ McNemar on paired per-task outcomes: "gap: +0.10 [95% CI -0.19,+0.39],
11
+ p=0.48 - NOT distinguishable at n=10" or "... distinguishable (p=0.03)".
12
+ observed_verdict(passes, trials) -> str|None
13
+ gonogo_decide semantics for flywheel observed rows: verdict + what
14
+ sample size would be needed.
15
+ route_stamp(lane, winner, runner, ...) -> str|None
16
+ The provenance line suffix for a measured lane.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import logging
22
+ from typing import Any, Optional
23
+
24
+ logger = logging.getLogger(__name__)
25
+
26
+
27
+ def _gonogo():
28
+ """Import gonogo lazily; None when not installed (stamps degrade)."""
29
+ try:
30
+ import gonogo as mod
31
+ return mod
32
+ except Exception as exc: # pragma: no cover - depends on environment
33
+ logger.debug("gonogo unavailable: %s", exc)
34
+ return None
35
+
36
+
37
+ def _paired_arm_outcomes(runs: list[dict[str, Any]], arm_a: str, arm_b: str
38
+ ) -> Optional[dict[tuple[str, str], tuple[bool, bool]]]:
39
+ """Per-task paired outcomes for two arms: {(task): (a_passed, b_passed)}.
40
+
41
+ Uses graded samples only; a task with k>1 samples is a task-pass iff any
42
+ sample passed (coverage semantics - same rule the report's cov uses).
43
+ Tasks not present in BOTH arms are dropped (McNemar needs pairs).
44
+ """
45
+ per: dict[str, dict[str, list[bool]]] = {}
46
+ for r in runs:
47
+ if "text" not in r or r.get("passed") is None:
48
+ continue
49
+ per.setdefault(r["task"], {}).setdefault(r["model"], []).append(bool(r["passed"]))
50
+ pairs = {t: (any(arms[arm_a]), any(arms[arm_b]))
51
+ for t, arms in per.items() if arm_a in arms and arm_b in arms}
52
+ return pairs or None
53
+
54
+
55
+ def distinguishable(runs: list[dict[str, Any]], arm_a: str, arm_b: str,
56
+ level: float = 0.95) -> Optional[str]:
57
+ """McNemar verdict on whether arm_a really beats arm_b, or None if
58
+ gonogo is unavailable or the arms share no task pairs."""
59
+ g = _gonogo()
60
+ if g is None:
61
+ return None
62
+ pairs = _paired_arm_outcomes(runs, arm_a, arm_b)
63
+ if pairs is None:
64
+ return None
65
+ try:
66
+ rep_a = {"cases": [{"id": t, "passed": a} for t, (a, _) in pairs.items()]}
67
+ rep_b = {"cases": [{"id": t, "passed": b} for t, (_, b) in pairs.items()]}
68
+ cmp = g.compare(rep_a, rep_b, level=level)
69
+ return str(cmp) # "B vs A (+x% [lo,hi], p=..) — verdict on N shared cases"
70
+ except Exception as exc:
71
+ logger.warning("gonogo compare failed: %s", exc)
72
+ return None
73
+
74
+
75
+ def observed_verdict(passes: int, trials: int, target: float = 0.9) -> Optional[str]:
76
+ """gonogo decide() semantics for an observed lane's pass rate.
77
+
78
+ decide() takes per-trial (confidence, passed) tuples; flywheel outcomes
79
+ carry no confidence signal, so every trial uses the same value and gonogo
80
+ itself will note no abstention threshold can be derived. Honest.
81
+ """
82
+ g = _gonogo()
83
+ if g is None or trials <= 0:
84
+ return None
85
+ results = [(0.5, bool(i < passes)) for i in range(trials)]
86
+ try:
87
+ d = g.decide(results=results, target=target, unit="tasks")
88
+ verdict = getattr(d, "verdict", None)
89
+ name = getattr(verdict, "name", str(verdict))
90
+ reason = getattr(d, "reason", "")
91
+ return f"gonogo {name}: {reason}" if reason else f"gonogo {name}"
92
+ except Exception as exc:
93
+ logger.warning("gonogo decide failed: %s", exc)
94
+ return None
95
+
96
+
97
+ def route_stamp(runs: list[dict[str, Any]], winner: str, runner_up: Optional[str],
98
+ level: float = 0.95) -> str:
99
+ """Provenance suffix for a measured lane: was the winner's edge real?
100
+
101
+ Never raises; degrades to "unverified" so provenance stays honest about
102
+ what was actually adjudicated rather than silently omitting the check.
103
+ """
104
+ if runner_up is None:
105
+ return "no runner-up arm to compare"
106
+ stamp = distinguishable(runs, winner, runner_up, level)
107
+ if stamp is None:
108
+ return "gap unverified (gonogo absent or arms share no tasks)"
109
+ return f"gap vs runner-up: {stamp}"
evalroute/cli.py ADDED
@@ -0,0 +1,140 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+
5
+ from . import dataset, dispatch, flywheel
6
+ from .routing import _route_for_args, _tool_result, install_routes
7
+
8
+ _WORKFLOW_EPILOG = """\
9
+ workflow (route -> arm -> rate, in the session that runs the task):
10
+ 1. /route <task> classify; prints the card (lane, model, effort)
11
+ 2. /model <model> set the arm from the card's "run:" line
12
+ (/reasoning <effort> too, unless install-routes
13
+ already wrote it into agent.reasoning_overrides)
14
+ 3. do the task in that session
15
+ 4. /rate pass|fail [--lane <lane-id>] [--note ...]
16
+ label the outcome; --lane files a correction
17
+ when the route got the lane wrong
18
+ --route-id <id> selects a pending route when overlapping
19
+ same flow from the terminal: hermes evalroute route "<task>" (step 1) and
20
+ hermes evalroute rate pass --note ... (step 4); steps 2-3 are chat commands.
21
+ hermes evalroute dispatch <brief.md> runs steps 1-3 on a subprocess worker and
22
+ prints the rate line for step 4.
23
+ routing data improves only when routes are rated: unrouted tasks cost the
24
+ same as ever, unrated routes teach nothing."""
25
+
26
+
27
+ def setup_cli(subparser) -> None:
28
+ """argparse wiring for `hermes evalroute` (register_cli_command setup_fn)."""
29
+ subs = subparser.add_subparsers(dest="evalroute_action")
30
+ route_p = subs.add_parser("route", help="Classify a task and print a route card",
31
+ epilog=_WORKFLOW_EPILOG,
32
+ formatter_class=argparse.RawDescriptionHelpFormatter)
33
+ route_p.add_argument("task", nargs="*", help="The task description")
34
+ route_p.add_argument("--lane", help="Pin a lane id instead of classifying")
35
+ route_p.add_argument("--replace-route-id", help="Replace a specific pending route (requires --lane)")
36
+ route_p.add_argument("--json", action="store_true",
37
+ help="Print the tool-result JSON envelope instead of the card")
38
+ rate_p = subs.add_parser("rate", help="Rate the last routed task: pass|fail",
39
+ epilog=_WORKFLOW_EPILOG,
40
+ formatter_class=argparse.RawDescriptionHelpFormatter)
41
+ rate_p.add_argument("verdict", nargs="?", choices=["pass", "fail", "skip"],
42
+ help="pass | fail | skip")
43
+ rate_p.add_argument("--lane", help="File a lane correction (the lane it should have been)")
44
+ rate_p.add_argument("--route-id", help="Select a pending route explicitly (profile-wide ledger)")
45
+ rate_p.add_argument("--model", help="Confirm the actual arm's model id (diagnostic; with --effort)")
46
+ rate_p.add_argument("--effort", help="Confirm the actual arm's effort (diagnostic; with --model)")
47
+ rate_p.add_argument("--note", help="Why — the highest-value part of the label")
48
+ install_p = subs.add_parser("install-routes", help="Write the route table's effort "
49
+ "column into agent.reasoning_overrides")
50
+ install_p.add_argument("--dry-run", action="store_true", help="Show the diff, write nothing")
51
+ status_p = subs.add_parser("sync", help="Pin the published route table "
52
+ "(keppy/evalroute-flywheel) under the Hermes home")
53
+ status_p.add_argument("--revision", help="Pin a specific dataset revision "
54
+ "(default: resolve 'main' to its commit sha)")
55
+ status_p.add_argument("--status", action="store_true",
56
+ help="Show which route table is active; no network")
57
+ status_p.add_argument("--clear", action="store_true",
58
+ help="Unpin the dataset table; route on the bundled table")
59
+ dispatch_p = subs.add_parser("dispatch",
60
+ help="Route a brief, spawn hermes chat on that arm, "
61
+ "print the rate line",
62
+ epilog=_WORKFLOW_EPILOG,
63
+ formatter_class=argparse.RawDescriptionHelpFormatter)
64
+ dispatch_p.add_argument("brief", help="Path to the brief markdown file")
65
+ dispatch_p.add_argument("--lane", help="Pin a lane id instead of classifying")
66
+ dispatch_p.add_argument("--in", dest="indir", help="Extra --in dir for the child session")
67
+ dispatch_p.add_argument("--task", help="Task description (default: the brief's first paragraph)")
68
+ dispatch_p.add_argument("--out", help="Report path (default: <brief stem>.report.md beside it)")
69
+ dispatch_p.add_argument("--timeout", type=float, help="Kill the child after SECONDS (exit 124)")
70
+ dispatch_p.add_argument("--rate-on-exit", choices=["fail"],
71
+ help="Auto-rate fail when the child exits non-zero (never auto-passes)")
72
+ dispatch_p.add_argument("--dry-run", action="store_true",
73
+ help="Route and print the argv; spawn nothing")
74
+ subparser.set_defaults(func=evalroute_cli)
75
+
76
+
77
+ def evalroute_cli(args) -> int:
78
+ """Handler for `hermes evalroute ...` (register_cli_command handler_fn)."""
79
+ action = getattr(args, "evalroute_action", None)
80
+ if action == "install-routes":
81
+ return install_routes(dry_run=bool(getattr(args, "dry_run", False)))
82
+ if action == "dispatch":
83
+ return dispatch.run(args)
84
+ if action == "sync":
85
+ return dataset.run(args)
86
+ if action == "rate":
87
+ parts = [getattr(args, "verdict", None) or ""]
88
+ if getattr(args, "lane", None):
89
+ parts.append(f"--lane {args.lane}")
90
+ if getattr(args, "route_id", None):
91
+ parts.append(f"--route-id {args.route_id}")
92
+ if getattr(args, "model", None):
93
+ parts.append(f"--model {args.model}")
94
+ if getattr(args, "effort", None):
95
+ parts.append(f"--effort {args.effort}")
96
+ if getattr(args, "note", None):
97
+ parts.append(f"--note {args.note}")
98
+ print(flywheel.handle_rate(" ".join(parts)))
99
+ return 0
100
+ if action == "route":
101
+ task = " ".join(getattr(args, "task", []) or [])
102
+ lane = getattr(args, "lane", None)
103
+ replace_id = getattr(args, "replace_route_id", None)
104
+ as_json = getattr(args, "json", False)
105
+ if not task and not lane:
106
+ print(_WORKFLOW_EPILOG)
107
+ return 2
108
+ if replace_id and not lane:
109
+ print("evalroute: --replace-route-id requires --lane <lane-id>")
110
+ return 2
111
+ try:
112
+ if lane:
113
+ raw = f"--lane {lane} {f'--replace-route-id {replace_id}' if replace_id else ''} {task}".strip()
114
+ else:
115
+ raw = task
116
+ card, lane_obj, conf, pinned, method, route_id = _route_for_args(raw)
117
+ except Exception as exc:
118
+ print(f"evalroute: {exc}")
119
+ return 1
120
+ if as_json:
121
+ print(_tool_result(card, lane_obj, conf, pinned, method=method, route_id=route_id))
122
+ else:
123
+ print(card)
124
+ return 0
125
+ print(_WORKFLOW_EPILOG)
126
+ return 2
127
+
128
+
129
+ def main() -> int:
130
+ """Standalone entry point (`evalroute ...`), byte-identical to `hermes evalroute ...`."""
131
+ import sys
132
+
133
+ parser = argparse.ArgumentParser(
134
+ prog="evalroute",
135
+ description="Route tasks to the right (model, reasoning effort) arm",
136
+ )
137
+ setup_cli(parser)
138
+ args = parser.parse_args()
139
+ rc = args.func(args)
140
+ sys.exit(rc)
evalroute/contract.py ADDED
@@ -0,0 +1,32 @@
1
+ """The plugin-facing contract of the evalroute library.
2
+
3
+ Everything the Hermes plugin (keppy/hermes-plugin-evalroute) relies on is
4
+ listed here. Those nine names are the public API; everything else in the
5
+ package is private to the library and may change without notice.
6
+ """
7
+
8
+ from __future__ import annotations
9
+
10
+ CONTRACT_VERSION = 1
11
+
12
+ #: The nine contract names, by module:
13
+ #: evalroute.routing.set_llm_facade (called at plugin register time)
14
+ #: evalroute.routing.evalroute_route (tool handler for evalroute_route)
15
+ #: evalroute.routing.handle_route_command (/route slash command)
16
+ #: evalroute.cli.setup_cli (register_cli_command setup_fn)
17
+ #: evalroute.cli.evalroute_cli (register_cli_command handler_fn)
18
+ #: evalroute.flywheel.handle_rate (/rate slash command + CLI rate)
19
+ #: evalroute.flywheel.on_pre_command (pre_command hook: /model /reasoning)
20
+ #: evalroute.flywheel.on_post_llm_call (post_llm_call hook)
21
+ #: evalroute.schemas.EVALROUTE_ROUTE (tool schema)
22
+ CONTRACT_NAMES = (
23
+ "routing.set_llm_facade",
24
+ "routing.evalroute_route",
25
+ "routing.handle_route_command",
26
+ "cli.setup_cli",
27
+ "cli.evalroute_cli",
28
+ "flywheel.handle_rate",
29
+ "flywheel.on_pre_command",
30
+ "flywheel.on_post_llm_call",
31
+ "schemas.EVALROUTE_ROUTE",
32
+ )
@@ -0,0 +1,56 @@
1
+ # Task facets: orthogonal dimensions a task can occupy simultaneously.
2
+ # Lanes are the ROUTING decision (one arm); facets are the LABEL — what the
3
+ # task actually is along each axis. A task may occupy one facet per axis.
4
+ # Facets are descriptive labels only; the lane classifier chooses the arm.
5
+ version: 1
6
+
7
+ facets:
8
+ - id: long-doc
9
+ axis: input-shape
10
+ description: input-heavy reading over many pages/files before answering
11
+ lanes: [long-doc-reading]
12
+
13
+ - id: interactive
14
+ axis: input-shape
15
+ description: conversational or small-context work; no bulk reading required
16
+ lanes: [] # default when no long-doc signal; never route ON this facet
17
+
18
+ - id: domain-dlml
19
+ axis: domain
20
+ description: training/eval harness engineering, RL on reasoning, ML research judgment
21
+ lanes: [dl-ml-research-engineering]
22
+
23
+ - id: domain-alignment
24
+ axis: domain
25
+ description: alignment/safety reasoning, paper-claim critique
26
+ lanes: [alignment-reasoning]
27
+
28
+ - id: domain-math
29
+ axis: domain
30
+ description: first-principles math, proofs, derivations
31
+ lanes: [math-first-principles]
32
+
33
+ - id: domain-prose
34
+ axis: domain
35
+ description: writing quality is the deliverable
36
+ lanes: [prose]
37
+
38
+ - id: domain-research
39
+ axis: domain
40
+ description: web/literature research, source gathering
41
+ lanes: [web-research]
42
+
43
+ - id: tier-routine
44
+ axis: demand-tier
45
+ description: self-contained, well-specified work; retries are cheap
46
+ lanes: [routine-coding]
47
+
48
+ - id: tier-hard
49
+ axis: demand-tier
50
+ description: multi-step, multi-file, or judgment-heavy work where a miss is expensive
51
+ lanes: [hard-agentic-coding]
52
+
53
+ - id: tier-orchestration
54
+ axis: demand-tier
55
+ description: the work IS planning and delegation
56
+ lanes: [orchestration]
@@ -0,0 +1,224 @@
1
+ version: 1
2
+ lanes:
3
+ - id: alignment-reasoning
4
+ label: Alignment reasoning, paper claims
5
+ model: z-ai/glm-5.3
6
+ effort: medium
7
+ provenance: 'measured 10 tasks, cov 1.0, all-in $0.0119/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
8
+ runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
9
+ on 10 shared cases'
10
+ keywords:
11
+ - alignment
12
+ - reward hacking
13
+ - goodhart
14
+ - sycophancy
15
+ - mesa-optimization
16
+ - mesaoptimization
17
+ - outer alignment
18
+ - inner alignment
19
+ - interpretability
20
+ - paper claim
21
+ - rlhf
22
+ - rlaif
23
+ - constitutional
24
+ match_hint: reasoning about failure modes and paper claims
25
+ notes: 'Four of five evaluated arms have 30 judged samples each; qwen3.8-max@medium has 30 pending, not graded. No checker audit. The p=1 and zero-width empirical gap describe this ten-task sample, not population equivalence.'
26
+ - id: routine-coding
27
+ label: Routine coding
28
+ model: z-ai/glm-5.3-flash
29
+ effort: medium
30
+ provenance: 'measured 10 tasks, cov 1.0, all-in $0.0001/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
31
+ runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
32
+ on 10 shared cases'
33
+ keywords:
34
+ - refactor
35
+ - lint
36
+ - unit test
37
+ - unit-test
38
+ - test
39
+ - typo
40
+ - bugfix
41
+ - bug fix
42
+ - script
43
+ - rename
44
+ - cleanup
45
+ - format
46
+ - boilerplate
47
+ - stub
48
+ - scaffold
49
+ - small function
50
+ - helper
51
+ match_hint: everyday code changes where a test or compiler checks the result
52
+ escalation: gpt-6-luna
53
+ notes: 'Verbose already; tests catch misses cheaply, so retries beat deep thinking. ds-v4.1-flash@medium had pass^k 1.00, but costs more; prefer when determinism matters. p=1 and a zero-width empirical gap at n=10 do not prove population equivalence.'
54
+ - id: dl-ml-research-engineering
55
+ label: DL / ML research engineering
56
+ model: z-ai/glm-5.3
57
+ effort: high
58
+ provenance: 'measured 10 tasks, cov 1.0, all-in $0.0024/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
59
+ runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
60
+ on 10 shared cases'
61
+ keywords:
62
+ - training run
63
+ - fine-tune
64
+ - finetune
65
+ - finetuning
66
+ - grpo
67
+ - trl
68
+ - lora
69
+ - sft
70
+ - rl
71
+ - reward model
72
+ - loss curve
73
+ - gradient
74
+ - ablation
75
+ - hyperparameter
76
+ - hyperparameter
77
+ - overfitting
78
+ - checkpoint
79
+ - dataloader
80
+ - tokenizer
81
+ - distillation
82
+ - colocation
83
+ - colocate
84
+ match_hint: training/eval harness engineering
85
+ escalation: gpt-6-sol
86
+ notes: 'The ten-task p=1, zero-width empirical gap vs runner-up is not proof of population equivalence; this route is a cost choice on measured coverage.'
87
+ - id: hard-agentic-coding
88
+ label: Hard agentic coding
89
+ keywords:
90
+ - hard agentic
91
+ - terminal-bench
92
+ - multi-file
93
+ - multi file
94
+ - repo-wide
95
+ - architecture
96
+ - migrate
97
+ - performance
98
+ - race condition
99
+ - deadlock
100
+ - memory leak
101
+ - integration
102
+ - e2e
103
+ - integration test
104
+ - flaky
105
+ - build failure
106
+ - build error
107
+ match_hint: multi-step repo work where a miss costs many tool-call rounds
108
+ model: z-ai/glm-5.3
109
+ effort: max
110
+ escalation: gpt-6-sol
111
+ provenance: priors 2026-09-26 (GLM 5.3 42% on TB 4.0 at max, Artificial Analysis)
112
+ notes: the 42% TB 4.0 figure was at max; escalate to GPT-6 Sol when coverage gaps
113
+ - id: long-doc-reading
114
+ label: Long-doc reading, lit review
115
+ keywords:
116
+ - lit review
117
+ - literature
118
+ - paper summary
119
+ - summarize paper
120
+ - read paper
121
+ - long document
122
+ - long doc
123
+ - pdf
124
+ - survey
125
+ - spec review
126
+ - rfp
127
+ - due diligence
128
+ - technical report
129
+ match_hint: input-heavy extraction over many pages
130
+ model: z-ai/glm-5.3-flash
131
+ effort: medium
132
+ escalation: null
133
+ provenance: priors 2026-09-26 (V4.1 Flash AA-LCR 84%, on par with GPT-5.6 Sol)
134
+ notes: input-heavy; gap to closed ~none
135
+ - id: web-research
136
+ label: Web research
137
+ keywords:
138
+ - web research
139
+ - browse
140
+ - search the web
141
+ - fact-check
142
+ - factcheck
143
+ - market research
144
+ - competitor
145
+ - news
146
+ - pricing
147
+ - sources
148
+ - citations
149
+ - open-source intelligence
150
+ - osint
151
+ match_hint: search-and-synthesize over live web content
152
+ model: moonshotai/kimi-k3
153
+ effort: medium
154
+ escalation: null
155
+ provenance: priors 2026-09-26 (K3 BrowseComp 91.2, vendor-reported, at max)
156
+ notes: input-heavy; max roughly triples output at $10.53/M so medium is the starting
157
+ arm
158
+ - id: math-first-principles
159
+ label: Math, first principles
160
+ keywords:
161
+ - proof
162
+ - prove
163
+ - theorem
164
+ - derivation
165
+ - derive
166
+ - combinatorics
167
+ - algebraic
168
+ - calculus
169
+ - probability bound
170
+ - asymptotic
171
+ - complexity bound
172
+ - lemma
173
+ - math
174
+ - inequality
175
+ match_hint: correctness comes from reasoning, not from tools
176
+ model: qwen/qwen3.8-max
177
+ effort: max
178
+ escalation: gpt-6-sol
179
+ provenance: priors 2026-09-26 (Qwen3.8-Max MathArena 56% vs GPT-6 Sol 85%, Opus
180
+ 5.5 83%)
181
+ notes: ~27-30 pts behind closed; skip open for one-shot math except Lean-verified
182
+ pipelines
183
+ - id: prose
184
+ label: Prose
185
+ keywords:
186
+ - blog post
187
+ - blog
188
+ - essay
189
+ - short story
190
+ - prose
191
+ - newsletter
192
+ - announcement
193
+ - launch post
194
+ - copywriting
195
+ - tagline
196
+ - bio
197
+ - readme rewrite
198
+ match_hint: human is the checker
199
+ model: moonshotai/kimi-k3
200
+ effort: medium
201
+ escalation: claude-opus-5-5
202
+ provenance: 'priors 2026-09-26 (K3 EQ-Bench creative writing #2 behind Opus 5; judged
203
+ by a Claude model)'
204
+ notes: untested prior; blind-test K3 vs Opus 5.5 at low-medium
205
+ - id: orchestration
206
+ label: Orchestrator / subagents
207
+ keywords:
208
+ - orchestrate
209
+ - orchestration
210
+ - orchestrator
211
+ - subagent
212
+ - multi-agent
213
+ - multiagent
214
+ - delegate
215
+ - fan out
216
+ - parallel agents
217
+ - crew
218
+ - workflow design
219
+ match_hint: the session plans and delegates, not does
220
+ model: claude-opus-5-5
221
+ effort: high
222
+ escalation: null
223
+ provenance: priors 2026-09-26 (image 2 only; not in README priors)
224
+ notes: subagents run at low-medium via delegation.reasoning_effort; set it separately