evalroute 0.6.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- evalroute/__init__.py +18 -0
- evalroute/adjudicate.py +109 -0
- evalroute/cli.py +140 -0
- evalroute/contract.py +32 -0
- evalroute/data/facets.yaml +56 -0
- evalroute/data/routes.yaml +224 -0
- evalroute/dataset.py +151 -0
- evalroute/dispatch.py +163 -0
- evalroute/flywheel.py +305 -0
- evalroute/harness/__init__.py +1 -0
- evalroute/harness/tier_a.py +501 -0
- evalroute/paths.py +19 -0
- evalroute/routes_from_labels.py +211 -0
- evalroute/routes_from_report.py +316 -0
- evalroute/routing.py +607 -0
- evalroute/runners/__init__.py +1 -0
- evalroute/runners/hermes_shim.py +157 -0
- evalroute/schemas.py +36 -0
- evalroute-0.6.0.dist-info/METADATA +351 -0
- evalroute-0.6.0.dist-info/RECORD +22 -0
- evalroute-0.6.0.dist-info/WHEEL +4 -0
- evalroute-0.6.0.dist-info/entry_points.txt +2 -0
evalroute/__init__.py
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
"""evalroute — model routing with a verified-success flywheel.
|
|
2
|
+
|
|
3
|
+
The library behind the Hermes plugin of the same name: a keyword-first
|
|
4
|
+
classifier that routes a task description to a (model, reasoning effort)
|
|
5
|
+
arm, a ledger that labels routes with pass/fail outcomes, and the Tier-A
|
|
6
|
+
harness that measures arms (cost per verified success) and feeds the
|
|
7
|
+
results back into the route table.
|
|
8
|
+
|
|
9
|
+
The Hermes plugin is a thin adapter: keppy/hermes-plugin-evalroute.
|
|
10
|
+
"""
|
|
11
|
+
|
|
12
|
+
from __future__ import annotations
|
|
13
|
+
|
|
14
|
+
try:
|
|
15
|
+
from importlib.metadata import version as _version
|
|
16
|
+
__version__ = _version("evalroute")
|
|
17
|
+
except Exception: # pragma: no cover - running from a checkout, not installed
|
|
18
|
+
__version__ = "0.6.0"
|
evalroute/adjudicate.py
ADDED
|
@@ -0,0 +1,109 @@
|
|
|
1
|
+
"""Gonogo adjudication for evalroute: is a route flip statistically real?
|
|
2
|
+
|
|
3
|
+
Reuses the merged gonogo plugin's stats (McNemar paired comparison, Wilson
|
|
4
|
+
intervals, INSUFFICIENT-EVIDENCE decisions) so evalroute never treats a
|
|
5
|
+
pilot-scale point-estimate gap as a measurement. Import is lazy and optional:
|
|
6
|
+
without gonogo installed, every stamp degrades to "unverified".
|
|
7
|
+
|
|
8
|
+
Three stamps:
|
|
9
|
+
distinguishable(winner, runner, per_task_outcomes) -> str|None
|
|
10
|
+
McNemar on paired per-task outcomes: "gap: +0.10 [95% CI -0.19,+0.39],
|
|
11
|
+
p=0.48 - NOT distinguishable at n=10" or "... distinguishable (p=0.03)".
|
|
12
|
+
observed_verdict(passes, trials) -> str|None
|
|
13
|
+
gonogo_decide semantics for flywheel observed rows: verdict + what
|
|
14
|
+
sample size would be needed.
|
|
15
|
+
route_stamp(lane, winner, runner, ...) -> str|None
|
|
16
|
+
The provenance line suffix for a measured lane.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
from __future__ import annotations
|
|
20
|
+
|
|
21
|
+
import logging
|
|
22
|
+
from typing import Any, Optional
|
|
23
|
+
|
|
24
|
+
logger = logging.getLogger(__name__)
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def _gonogo():
|
|
28
|
+
"""Import gonogo lazily; None when not installed (stamps degrade)."""
|
|
29
|
+
try:
|
|
30
|
+
import gonogo as mod
|
|
31
|
+
return mod
|
|
32
|
+
except Exception as exc: # pragma: no cover - depends on environment
|
|
33
|
+
logger.debug("gonogo unavailable: %s", exc)
|
|
34
|
+
return None
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _paired_arm_outcomes(runs: list[dict[str, Any]], arm_a: str, arm_b: str
|
|
38
|
+
) -> Optional[dict[tuple[str, str], tuple[bool, bool]]]:
|
|
39
|
+
"""Per-task paired outcomes for two arms: {(task): (a_passed, b_passed)}.
|
|
40
|
+
|
|
41
|
+
Uses graded samples only; a task with k>1 samples is a task-pass iff any
|
|
42
|
+
sample passed (coverage semantics - same rule the report's cov uses).
|
|
43
|
+
Tasks not present in BOTH arms are dropped (McNemar needs pairs).
|
|
44
|
+
"""
|
|
45
|
+
per: dict[str, dict[str, list[bool]]] = {}
|
|
46
|
+
for r in runs:
|
|
47
|
+
if "text" not in r or r.get("passed") is None:
|
|
48
|
+
continue
|
|
49
|
+
per.setdefault(r["task"], {}).setdefault(r["model"], []).append(bool(r["passed"]))
|
|
50
|
+
pairs = {t: (any(arms[arm_a]), any(arms[arm_b]))
|
|
51
|
+
for t, arms in per.items() if arm_a in arms and arm_b in arms}
|
|
52
|
+
return pairs or None
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def distinguishable(runs: list[dict[str, Any]], arm_a: str, arm_b: str,
|
|
56
|
+
level: float = 0.95) -> Optional[str]:
|
|
57
|
+
"""McNemar verdict on whether arm_a really beats arm_b, or None if
|
|
58
|
+
gonogo is unavailable or the arms share no task pairs."""
|
|
59
|
+
g = _gonogo()
|
|
60
|
+
if g is None:
|
|
61
|
+
return None
|
|
62
|
+
pairs = _paired_arm_outcomes(runs, arm_a, arm_b)
|
|
63
|
+
if pairs is None:
|
|
64
|
+
return None
|
|
65
|
+
try:
|
|
66
|
+
rep_a = {"cases": [{"id": t, "passed": a} for t, (a, _) in pairs.items()]}
|
|
67
|
+
rep_b = {"cases": [{"id": t, "passed": b} for t, (_, b) in pairs.items()]}
|
|
68
|
+
cmp = g.compare(rep_a, rep_b, level=level)
|
|
69
|
+
return str(cmp) # "B vs A (+x% [lo,hi], p=..) — verdict on N shared cases"
|
|
70
|
+
except Exception as exc:
|
|
71
|
+
logger.warning("gonogo compare failed: %s", exc)
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def observed_verdict(passes: int, trials: int, target: float = 0.9) -> Optional[str]:
|
|
76
|
+
"""gonogo decide() semantics for an observed lane's pass rate.
|
|
77
|
+
|
|
78
|
+
decide() takes per-trial (confidence, passed) tuples; flywheel outcomes
|
|
79
|
+
carry no confidence signal, so every trial uses the same value and gonogo
|
|
80
|
+
itself will note no abstention threshold can be derived. Honest.
|
|
81
|
+
"""
|
|
82
|
+
g = _gonogo()
|
|
83
|
+
if g is None or trials <= 0:
|
|
84
|
+
return None
|
|
85
|
+
results = [(0.5, bool(i < passes)) for i in range(trials)]
|
|
86
|
+
try:
|
|
87
|
+
d = g.decide(results=results, target=target, unit="tasks")
|
|
88
|
+
verdict = getattr(d, "verdict", None)
|
|
89
|
+
name = getattr(verdict, "name", str(verdict))
|
|
90
|
+
reason = getattr(d, "reason", "")
|
|
91
|
+
return f"gonogo {name}: {reason}" if reason else f"gonogo {name}"
|
|
92
|
+
except Exception as exc:
|
|
93
|
+
logger.warning("gonogo decide failed: %s", exc)
|
|
94
|
+
return None
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
def route_stamp(runs: list[dict[str, Any]], winner: str, runner_up: Optional[str],
|
|
98
|
+
level: float = 0.95) -> str:
|
|
99
|
+
"""Provenance suffix for a measured lane: was the winner's edge real?
|
|
100
|
+
|
|
101
|
+
Never raises; degrades to "unverified" so provenance stays honest about
|
|
102
|
+
what was actually adjudicated rather than silently omitting the check.
|
|
103
|
+
"""
|
|
104
|
+
if runner_up is None:
|
|
105
|
+
return "no runner-up arm to compare"
|
|
106
|
+
stamp = distinguishable(runs, winner, runner_up, level)
|
|
107
|
+
if stamp is None:
|
|
108
|
+
return "gap unverified (gonogo absent or arms share no tasks)"
|
|
109
|
+
return f"gap vs runner-up: {stamp}"
|
evalroute/cli.py
ADDED
|
@@ -0,0 +1,140 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import argparse
|
|
4
|
+
|
|
5
|
+
from . import dataset, dispatch, flywheel
|
|
6
|
+
from .routing import _route_for_args, _tool_result, install_routes
|
|
7
|
+
|
|
8
|
+
_WORKFLOW_EPILOG = """\
|
|
9
|
+
workflow (route -> arm -> rate, in the session that runs the task):
|
|
10
|
+
1. /route <task> classify; prints the card (lane, model, effort)
|
|
11
|
+
2. /model <model> set the arm from the card's "run:" line
|
|
12
|
+
(/reasoning <effort> too, unless install-routes
|
|
13
|
+
already wrote it into agent.reasoning_overrides)
|
|
14
|
+
3. do the task in that session
|
|
15
|
+
4. /rate pass|fail [--lane <lane-id>] [--note ...]
|
|
16
|
+
label the outcome; --lane files a correction
|
|
17
|
+
when the route got the lane wrong
|
|
18
|
+
--route-id <id> selects a pending route when overlapping
|
|
19
|
+
same flow from the terminal: hermes evalroute route "<task>" (step 1) and
|
|
20
|
+
hermes evalroute rate pass --note ... (step 4); steps 2-3 are chat commands.
|
|
21
|
+
hermes evalroute dispatch <brief.md> runs steps 1-3 on a subprocess worker and
|
|
22
|
+
prints the rate line for step 4.
|
|
23
|
+
routing data improves only when routes are rated: unrouted tasks cost the
|
|
24
|
+
same as ever, unrated routes teach nothing."""
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def setup_cli(subparser) -> None:
|
|
28
|
+
"""argparse wiring for `hermes evalroute` (register_cli_command setup_fn)."""
|
|
29
|
+
subs = subparser.add_subparsers(dest="evalroute_action")
|
|
30
|
+
route_p = subs.add_parser("route", help="Classify a task and print a route card",
|
|
31
|
+
epilog=_WORKFLOW_EPILOG,
|
|
32
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
33
|
+
route_p.add_argument("task", nargs="*", help="The task description")
|
|
34
|
+
route_p.add_argument("--lane", help="Pin a lane id instead of classifying")
|
|
35
|
+
route_p.add_argument("--replace-route-id", help="Replace a specific pending route (requires --lane)")
|
|
36
|
+
route_p.add_argument("--json", action="store_true",
|
|
37
|
+
help="Print the tool-result JSON envelope instead of the card")
|
|
38
|
+
rate_p = subs.add_parser("rate", help="Rate the last routed task: pass|fail",
|
|
39
|
+
epilog=_WORKFLOW_EPILOG,
|
|
40
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
41
|
+
rate_p.add_argument("verdict", nargs="?", choices=["pass", "fail", "skip"],
|
|
42
|
+
help="pass | fail | skip")
|
|
43
|
+
rate_p.add_argument("--lane", help="File a lane correction (the lane it should have been)")
|
|
44
|
+
rate_p.add_argument("--route-id", help="Select a pending route explicitly (profile-wide ledger)")
|
|
45
|
+
rate_p.add_argument("--model", help="Confirm the actual arm's model id (diagnostic; with --effort)")
|
|
46
|
+
rate_p.add_argument("--effort", help="Confirm the actual arm's effort (diagnostic; with --model)")
|
|
47
|
+
rate_p.add_argument("--note", help="Why — the highest-value part of the label")
|
|
48
|
+
install_p = subs.add_parser("install-routes", help="Write the route table's effort "
|
|
49
|
+
"column into agent.reasoning_overrides")
|
|
50
|
+
install_p.add_argument("--dry-run", action="store_true", help="Show the diff, write nothing")
|
|
51
|
+
status_p = subs.add_parser("sync", help="Pin the published route table "
|
|
52
|
+
"(keppy/evalroute-flywheel) under the Hermes home")
|
|
53
|
+
status_p.add_argument("--revision", help="Pin a specific dataset revision "
|
|
54
|
+
"(default: resolve 'main' to its commit sha)")
|
|
55
|
+
status_p.add_argument("--status", action="store_true",
|
|
56
|
+
help="Show which route table is active; no network")
|
|
57
|
+
status_p.add_argument("--clear", action="store_true",
|
|
58
|
+
help="Unpin the dataset table; route on the bundled table")
|
|
59
|
+
dispatch_p = subs.add_parser("dispatch",
|
|
60
|
+
help="Route a brief, spawn hermes chat on that arm, "
|
|
61
|
+
"print the rate line",
|
|
62
|
+
epilog=_WORKFLOW_EPILOG,
|
|
63
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
64
|
+
dispatch_p.add_argument("brief", help="Path to the brief markdown file")
|
|
65
|
+
dispatch_p.add_argument("--lane", help="Pin a lane id instead of classifying")
|
|
66
|
+
dispatch_p.add_argument("--in", dest="indir", help="Extra --in dir for the child session")
|
|
67
|
+
dispatch_p.add_argument("--task", help="Task description (default: the brief's first paragraph)")
|
|
68
|
+
dispatch_p.add_argument("--out", help="Report path (default: <brief stem>.report.md beside it)")
|
|
69
|
+
dispatch_p.add_argument("--timeout", type=float, help="Kill the child after SECONDS (exit 124)")
|
|
70
|
+
dispatch_p.add_argument("--rate-on-exit", choices=["fail"],
|
|
71
|
+
help="Auto-rate fail when the child exits non-zero (never auto-passes)")
|
|
72
|
+
dispatch_p.add_argument("--dry-run", action="store_true",
|
|
73
|
+
help="Route and print the argv; spawn nothing")
|
|
74
|
+
subparser.set_defaults(func=evalroute_cli)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def evalroute_cli(args) -> int:
|
|
78
|
+
"""Handler for `hermes evalroute ...` (register_cli_command handler_fn)."""
|
|
79
|
+
action = getattr(args, "evalroute_action", None)
|
|
80
|
+
if action == "install-routes":
|
|
81
|
+
return install_routes(dry_run=bool(getattr(args, "dry_run", False)))
|
|
82
|
+
if action == "dispatch":
|
|
83
|
+
return dispatch.run(args)
|
|
84
|
+
if action == "sync":
|
|
85
|
+
return dataset.run(args)
|
|
86
|
+
if action == "rate":
|
|
87
|
+
parts = [getattr(args, "verdict", None) or ""]
|
|
88
|
+
if getattr(args, "lane", None):
|
|
89
|
+
parts.append(f"--lane {args.lane}")
|
|
90
|
+
if getattr(args, "route_id", None):
|
|
91
|
+
parts.append(f"--route-id {args.route_id}")
|
|
92
|
+
if getattr(args, "model", None):
|
|
93
|
+
parts.append(f"--model {args.model}")
|
|
94
|
+
if getattr(args, "effort", None):
|
|
95
|
+
parts.append(f"--effort {args.effort}")
|
|
96
|
+
if getattr(args, "note", None):
|
|
97
|
+
parts.append(f"--note {args.note}")
|
|
98
|
+
print(flywheel.handle_rate(" ".join(parts)))
|
|
99
|
+
return 0
|
|
100
|
+
if action == "route":
|
|
101
|
+
task = " ".join(getattr(args, "task", []) or [])
|
|
102
|
+
lane = getattr(args, "lane", None)
|
|
103
|
+
replace_id = getattr(args, "replace_route_id", None)
|
|
104
|
+
as_json = getattr(args, "json", False)
|
|
105
|
+
if not task and not lane:
|
|
106
|
+
print(_WORKFLOW_EPILOG)
|
|
107
|
+
return 2
|
|
108
|
+
if replace_id and not lane:
|
|
109
|
+
print("evalroute: --replace-route-id requires --lane <lane-id>")
|
|
110
|
+
return 2
|
|
111
|
+
try:
|
|
112
|
+
if lane:
|
|
113
|
+
raw = f"--lane {lane} {f'--replace-route-id {replace_id}' if replace_id else ''} {task}".strip()
|
|
114
|
+
else:
|
|
115
|
+
raw = task
|
|
116
|
+
card, lane_obj, conf, pinned, method, route_id = _route_for_args(raw)
|
|
117
|
+
except Exception as exc:
|
|
118
|
+
print(f"evalroute: {exc}")
|
|
119
|
+
return 1
|
|
120
|
+
if as_json:
|
|
121
|
+
print(_tool_result(card, lane_obj, conf, pinned, method=method, route_id=route_id))
|
|
122
|
+
else:
|
|
123
|
+
print(card)
|
|
124
|
+
return 0
|
|
125
|
+
print(_WORKFLOW_EPILOG)
|
|
126
|
+
return 2
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def main() -> int:
|
|
130
|
+
"""Standalone entry point (`evalroute ...`), byte-identical to `hermes evalroute ...`."""
|
|
131
|
+
import sys
|
|
132
|
+
|
|
133
|
+
parser = argparse.ArgumentParser(
|
|
134
|
+
prog="evalroute",
|
|
135
|
+
description="Route tasks to the right (model, reasoning effort) arm",
|
|
136
|
+
)
|
|
137
|
+
setup_cli(parser)
|
|
138
|
+
args = parser.parse_args()
|
|
139
|
+
rc = args.func(args)
|
|
140
|
+
sys.exit(rc)
|
evalroute/contract.py
ADDED
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
"""The plugin-facing contract of the evalroute library.
|
|
2
|
+
|
|
3
|
+
Everything the Hermes plugin (keppy/hermes-plugin-evalroute) relies on is
|
|
4
|
+
listed here. Those nine names are the public API; everything else in the
|
|
5
|
+
package is private to the library and may change without notice.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
CONTRACT_VERSION = 1
|
|
11
|
+
|
|
12
|
+
#: The nine contract names, by module:
|
|
13
|
+
#: evalroute.routing.set_llm_facade (called at plugin register time)
|
|
14
|
+
#: evalroute.routing.evalroute_route (tool handler for evalroute_route)
|
|
15
|
+
#: evalroute.routing.handle_route_command (/route slash command)
|
|
16
|
+
#: evalroute.cli.setup_cli (register_cli_command setup_fn)
|
|
17
|
+
#: evalroute.cli.evalroute_cli (register_cli_command handler_fn)
|
|
18
|
+
#: evalroute.flywheel.handle_rate (/rate slash command + CLI rate)
|
|
19
|
+
#: evalroute.flywheel.on_pre_command (pre_command hook: /model /reasoning)
|
|
20
|
+
#: evalroute.flywheel.on_post_llm_call (post_llm_call hook)
|
|
21
|
+
#: evalroute.schemas.EVALROUTE_ROUTE (tool schema)
|
|
22
|
+
CONTRACT_NAMES = (
|
|
23
|
+
"routing.set_llm_facade",
|
|
24
|
+
"routing.evalroute_route",
|
|
25
|
+
"routing.handle_route_command",
|
|
26
|
+
"cli.setup_cli",
|
|
27
|
+
"cli.evalroute_cli",
|
|
28
|
+
"flywheel.handle_rate",
|
|
29
|
+
"flywheel.on_pre_command",
|
|
30
|
+
"flywheel.on_post_llm_call",
|
|
31
|
+
"schemas.EVALROUTE_ROUTE",
|
|
32
|
+
)
|
|
@@ -0,0 +1,56 @@
|
|
|
1
|
+
# Task facets: orthogonal dimensions a task can occupy simultaneously.
|
|
2
|
+
# Lanes are the ROUTING decision (one arm); facets are the LABEL — what the
|
|
3
|
+
# task actually is along each axis. A task may occupy one facet per axis.
|
|
4
|
+
# Facets are descriptive labels only; the lane classifier chooses the arm.
|
|
5
|
+
version: 1
|
|
6
|
+
|
|
7
|
+
facets:
|
|
8
|
+
- id: long-doc
|
|
9
|
+
axis: input-shape
|
|
10
|
+
description: input-heavy reading over many pages/files before answering
|
|
11
|
+
lanes: [long-doc-reading]
|
|
12
|
+
|
|
13
|
+
- id: interactive
|
|
14
|
+
axis: input-shape
|
|
15
|
+
description: conversational or small-context work; no bulk reading required
|
|
16
|
+
lanes: [] # default when no long-doc signal; never route ON this facet
|
|
17
|
+
|
|
18
|
+
- id: domain-dlml
|
|
19
|
+
axis: domain
|
|
20
|
+
description: training/eval harness engineering, RL on reasoning, ML research judgment
|
|
21
|
+
lanes: [dl-ml-research-engineering]
|
|
22
|
+
|
|
23
|
+
- id: domain-alignment
|
|
24
|
+
axis: domain
|
|
25
|
+
description: alignment/safety reasoning, paper-claim critique
|
|
26
|
+
lanes: [alignment-reasoning]
|
|
27
|
+
|
|
28
|
+
- id: domain-math
|
|
29
|
+
axis: domain
|
|
30
|
+
description: first-principles math, proofs, derivations
|
|
31
|
+
lanes: [math-first-principles]
|
|
32
|
+
|
|
33
|
+
- id: domain-prose
|
|
34
|
+
axis: domain
|
|
35
|
+
description: writing quality is the deliverable
|
|
36
|
+
lanes: [prose]
|
|
37
|
+
|
|
38
|
+
- id: domain-research
|
|
39
|
+
axis: domain
|
|
40
|
+
description: web/literature research, source gathering
|
|
41
|
+
lanes: [web-research]
|
|
42
|
+
|
|
43
|
+
- id: tier-routine
|
|
44
|
+
axis: demand-tier
|
|
45
|
+
description: self-contained, well-specified work; retries are cheap
|
|
46
|
+
lanes: [routine-coding]
|
|
47
|
+
|
|
48
|
+
- id: tier-hard
|
|
49
|
+
axis: demand-tier
|
|
50
|
+
description: multi-step, multi-file, or judgment-heavy work where a miss is expensive
|
|
51
|
+
lanes: [hard-agentic-coding]
|
|
52
|
+
|
|
53
|
+
- id: tier-orchestration
|
|
54
|
+
axis: demand-tier
|
|
55
|
+
description: the work IS planning and delegation
|
|
56
|
+
lanes: [orchestration]
|
|
@@ -0,0 +1,224 @@
|
|
|
1
|
+
version: 1
|
|
2
|
+
lanes:
|
|
3
|
+
- id: alignment-reasoning
|
|
4
|
+
label: Alignment reasoning, paper claims
|
|
5
|
+
model: z-ai/glm-5.3
|
|
6
|
+
effort: medium
|
|
7
|
+
provenance: 'measured 10 tasks, cov 1.0, all-in $0.0119/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
|
|
8
|
+
runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
|
|
9
|
+
on 10 shared cases'
|
|
10
|
+
keywords:
|
|
11
|
+
- alignment
|
|
12
|
+
- reward hacking
|
|
13
|
+
- goodhart
|
|
14
|
+
- sycophancy
|
|
15
|
+
- mesa-optimization
|
|
16
|
+
- mesaoptimization
|
|
17
|
+
- outer alignment
|
|
18
|
+
- inner alignment
|
|
19
|
+
- interpretability
|
|
20
|
+
- paper claim
|
|
21
|
+
- rlhf
|
|
22
|
+
- rlaif
|
|
23
|
+
- constitutional
|
|
24
|
+
match_hint: reasoning about failure modes and paper claims
|
|
25
|
+
notes: 'Four of five evaluated arms have 30 judged samples each; qwen3.8-max@medium has 30 pending, not graded. No checker audit. The p=1 and zero-width empirical gap describe this ten-task sample, not population equivalence.'
|
|
26
|
+
- id: routine-coding
|
|
27
|
+
label: Routine coding
|
|
28
|
+
model: z-ai/glm-5.3-flash
|
|
29
|
+
effort: medium
|
|
30
|
+
provenance: 'measured 10 tasks, cov 1.0, all-in $0.0001/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
|
|
31
|
+
runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
|
|
32
|
+
on 10 shared cases'
|
|
33
|
+
keywords:
|
|
34
|
+
- refactor
|
|
35
|
+
- lint
|
|
36
|
+
- unit test
|
|
37
|
+
- unit-test
|
|
38
|
+
- test
|
|
39
|
+
- typo
|
|
40
|
+
- bugfix
|
|
41
|
+
- bug fix
|
|
42
|
+
- script
|
|
43
|
+
- rename
|
|
44
|
+
- cleanup
|
|
45
|
+
- format
|
|
46
|
+
- boilerplate
|
|
47
|
+
- stub
|
|
48
|
+
- scaffold
|
|
49
|
+
- small function
|
|
50
|
+
- helper
|
|
51
|
+
match_hint: everyday code changes where a test or compiler checks the result
|
|
52
|
+
escalation: gpt-6-luna
|
|
53
|
+
notes: 'Verbose already; tests catch misses cheaply, so retries beat deep thinking. ds-v4.1-flash@medium had pass^k 1.00, but costs more; prefer when determinism matters. p=1 and a zero-width empirical gap at n=10 do not prove population equivalence.'
|
|
54
|
+
- id: dl-ml-research-engineering
|
|
55
|
+
label: DL / ML research engineering
|
|
56
|
+
model: z-ai/glm-5.3
|
|
57
|
+
effort: high
|
|
58
|
+
provenance: 'measured 10 tasks, cov 1.0, all-in $0.0024/succ, 2026-09-27; v1 runs omit model_id (mapped via models.json); gap vs
|
|
59
|
+
runner-up: 100.0% vs 100.0% (+0.0% [-33.4%, +33.4%], p=1.000) — not distinguishable
|
|
60
|
+
on 10 shared cases'
|
|
61
|
+
keywords:
|
|
62
|
+
- training run
|
|
63
|
+
- fine-tune
|
|
64
|
+
- finetune
|
|
65
|
+
- finetuning
|
|
66
|
+
- grpo
|
|
67
|
+
- trl
|
|
68
|
+
- lora
|
|
69
|
+
- sft
|
|
70
|
+
- rl
|
|
71
|
+
- reward model
|
|
72
|
+
- loss curve
|
|
73
|
+
- gradient
|
|
74
|
+
- ablation
|
|
75
|
+
- hyperparameter
|
|
76
|
+
- hyperparameter
|
|
77
|
+
- overfitting
|
|
78
|
+
- checkpoint
|
|
79
|
+
- dataloader
|
|
80
|
+
- tokenizer
|
|
81
|
+
- distillation
|
|
82
|
+
- colocation
|
|
83
|
+
- colocate
|
|
84
|
+
match_hint: training/eval harness engineering
|
|
85
|
+
escalation: gpt-6-sol
|
|
86
|
+
notes: 'The ten-task p=1, zero-width empirical gap vs runner-up is not proof of population equivalence; this route is a cost choice on measured coverage.'
|
|
87
|
+
- id: hard-agentic-coding
|
|
88
|
+
label: Hard agentic coding
|
|
89
|
+
keywords:
|
|
90
|
+
- hard agentic
|
|
91
|
+
- terminal-bench
|
|
92
|
+
- multi-file
|
|
93
|
+
- multi file
|
|
94
|
+
- repo-wide
|
|
95
|
+
- architecture
|
|
96
|
+
- migrate
|
|
97
|
+
- performance
|
|
98
|
+
- race condition
|
|
99
|
+
- deadlock
|
|
100
|
+
- memory leak
|
|
101
|
+
- integration
|
|
102
|
+
- e2e
|
|
103
|
+
- integration test
|
|
104
|
+
- flaky
|
|
105
|
+
- build failure
|
|
106
|
+
- build error
|
|
107
|
+
match_hint: multi-step repo work where a miss costs many tool-call rounds
|
|
108
|
+
model: z-ai/glm-5.3
|
|
109
|
+
effort: max
|
|
110
|
+
escalation: gpt-6-sol
|
|
111
|
+
provenance: priors 2026-09-26 (GLM 5.3 42% on TB 4.0 at max, Artificial Analysis)
|
|
112
|
+
notes: the 42% TB 4.0 figure was at max; escalate to GPT-6 Sol when coverage gaps
|
|
113
|
+
- id: long-doc-reading
|
|
114
|
+
label: Long-doc reading, lit review
|
|
115
|
+
keywords:
|
|
116
|
+
- lit review
|
|
117
|
+
- literature
|
|
118
|
+
- paper summary
|
|
119
|
+
- summarize paper
|
|
120
|
+
- read paper
|
|
121
|
+
- long document
|
|
122
|
+
- long doc
|
|
123
|
+
- pdf
|
|
124
|
+
- survey
|
|
125
|
+
- spec review
|
|
126
|
+
- rfp
|
|
127
|
+
- due diligence
|
|
128
|
+
- technical report
|
|
129
|
+
match_hint: input-heavy extraction over many pages
|
|
130
|
+
model: z-ai/glm-5.3-flash
|
|
131
|
+
effort: medium
|
|
132
|
+
escalation: null
|
|
133
|
+
provenance: priors 2026-09-26 (V4.1 Flash AA-LCR 84%, on par with GPT-5.6 Sol)
|
|
134
|
+
notes: input-heavy; gap to closed ~none
|
|
135
|
+
- id: web-research
|
|
136
|
+
label: Web research
|
|
137
|
+
keywords:
|
|
138
|
+
- web research
|
|
139
|
+
- browse
|
|
140
|
+
- search the web
|
|
141
|
+
- fact-check
|
|
142
|
+
- factcheck
|
|
143
|
+
- market research
|
|
144
|
+
- competitor
|
|
145
|
+
- news
|
|
146
|
+
- pricing
|
|
147
|
+
- sources
|
|
148
|
+
- citations
|
|
149
|
+
- open-source intelligence
|
|
150
|
+
- osint
|
|
151
|
+
match_hint: search-and-synthesize over live web content
|
|
152
|
+
model: moonshotai/kimi-k3
|
|
153
|
+
effort: medium
|
|
154
|
+
escalation: null
|
|
155
|
+
provenance: priors 2026-09-26 (K3 BrowseComp 91.2, vendor-reported, at max)
|
|
156
|
+
notes: input-heavy; max roughly triples output at $10.53/M so medium is the starting
|
|
157
|
+
arm
|
|
158
|
+
- id: math-first-principles
|
|
159
|
+
label: Math, first principles
|
|
160
|
+
keywords:
|
|
161
|
+
- proof
|
|
162
|
+
- prove
|
|
163
|
+
- theorem
|
|
164
|
+
- derivation
|
|
165
|
+
- derive
|
|
166
|
+
- combinatorics
|
|
167
|
+
- algebraic
|
|
168
|
+
- calculus
|
|
169
|
+
- probability bound
|
|
170
|
+
- asymptotic
|
|
171
|
+
- complexity bound
|
|
172
|
+
- lemma
|
|
173
|
+
- math
|
|
174
|
+
- inequality
|
|
175
|
+
match_hint: correctness comes from reasoning, not from tools
|
|
176
|
+
model: qwen/qwen3.8-max
|
|
177
|
+
effort: max
|
|
178
|
+
escalation: gpt-6-sol
|
|
179
|
+
provenance: priors 2026-09-26 (Qwen3.8-Max MathArena 56% vs GPT-6 Sol 85%, Opus
|
|
180
|
+
5.5 83%)
|
|
181
|
+
notes: ~27-30 pts behind closed; skip open for one-shot math except Lean-verified
|
|
182
|
+
pipelines
|
|
183
|
+
- id: prose
|
|
184
|
+
label: Prose
|
|
185
|
+
keywords:
|
|
186
|
+
- blog post
|
|
187
|
+
- blog
|
|
188
|
+
- essay
|
|
189
|
+
- short story
|
|
190
|
+
- prose
|
|
191
|
+
- newsletter
|
|
192
|
+
- announcement
|
|
193
|
+
- launch post
|
|
194
|
+
- copywriting
|
|
195
|
+
- tagline
|
|
196
|
+
- bio
|
|
197
|
+
- readme rewrite
|
|
198
|
+
match_hint: human is the checker
|
|
199
|
+
model: moonshotai/kimi-k3
|
|
200
|
+
effort: medium
|
|
201
|
+
escalation: claude-opus-5-5
|
|
202
|
+
provenance: 'priors 2026-09-26 (K3 EQ-Bench creative writing #2 behind Opus 5; judged
|
|
203
|
+
by a Claude model)'
|
|
204
|
+
notes: untested prior; blind-test K3 vs Opus 5.5 at low-medium
|
|
205
|
+
- id: orchestration
|
|
206
|
+
label: Orchestrator / subagents
|
|
207
|
+
keywords:
|
|
208
|
+
- orchestrate
|
|
209
|
+
- orchestration
|
|
210
|
+
- orchestrator
|
|
211
|
+
- subagent
|
|
212
|
+
- multi-agent
|
|
213
|
+
- multiagent
|
|
214
|
+
- delegate
|
|
215
|
+
- fan out
|
|
216
|
+
- parallel agents
|
|
217
|
+
- crew
|
|
218
|
+
- workflow design
|
|
219
|
+
match_hint: the session plans and delegates, not does
|
|
220
|
+
model: claude-opus-5-5
|
|
221
|
+
effort: high
|
|
222
|
+
escalation: null
|
|
223
|
+
provenance: priors 2026-09-26 (image 2 only; not in README priors)
|
|
224
|
+
notes: subagents run at low-medium via delegation.reasoning_effort; set it separately
|