loopmath 0.2.2__py3-none-any.whl → 0.2.3__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. loopmath/__init__.py +1 -1
  2. loopmath/belief/design.py +5 -3
  3. loopmath/belief/fit.py +5 -3
  4. loopmath/belief/priors.py +46 -7
  5. loopmath/belief/state.py +4 -1
  6. loopmath/builder/context.py +18 -2
  7. loopmath/cli_registry.py +6 -2
  8. loopmath/onboard/commands.py +5 -0
  9. loopmath/priors/__init__.py +31 -8
  10. loopmath/priors/benchmarks.py +6 -1
  11. loopmath/priors/benchmarks.toml +7 -2
  12. loopmath/priors/build.py +84 -33
  13. loopmath/priors/bundle/README.md +6 -2
  14. loopmath/priors/bundle/e0.jsonl.gz +0 -0
  15. loopmath/priors/bundle/lanes.jsonl.gz +0 -0
  16. loopmath/priors/bundle/manifest.json +133 -42
  17. loopmath/priors/bundle/rq1.jsonl.gz +0 -0
  18. loopmath/priors/bundle/sweep.jsonl.gz +0 -0
  19. loopmath/priors/commands.py +16 -10
  20. loopmath/priors/e0.py +8 -3
  21. loopmath/priors/lanes.py +15 -1
  22. loopmath/priors/ocpdoc.py +45 -0
  23. loopmath/priors/registry.py +3 -1
  24. loopmath/priors/rq1.py +13 -3
  25. loopmath/priors/show.py +111 -1
  26. loopmath/priors/sweep.py +34 -7
  27. loopmath/recommend/commands.py +18 -8
  28. loopmath/recommend/engine.py +18 -8
  29. loopmath/recommend/message.py +50 -4
  30. loopmath/recommend/storeread.py +8 -0
  31. loopmath/skill/skills/loopmath-import-runs/SKILL.md +2 -2
  32. loopmath/skill/skills/loopmath-onboard/SKILL.md +3 -1
  33. loopmath/skill/skills/loopmath-update-fit/SKILL.md +2 -2
  34. loopmath/skill/skills/reference.md +3 -3
  35. loopmath/store/config.py +5 -3
  36. loopmath/types.py +9 -1
  37. loopmath/views/assets/builder.css +4 -0
  38. loopmath/views/assets/builder.js +52 -16
  39. loopmath/views/assets/plans.css +1 -0
  40. loopmath/views/assets/plans.js +10 -1
  41. loopmath/views/assets/results.js +12 -3
  42. loopmath/views/posterior.py +14 -1
  43. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/METADATA +3 -1
  44. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/RECORD +51 -51
  45. {loopmath-0.2.2.data → loopmath-0.2.3.data}/data/spec/ocp-v0.2.schema.json +0 -0
  46. {loopmath-0.2.2.data → loopmath-0.2.3.data}/data/spec/ocp-v0.schema.json +0 -0
  47. {loopmath-0.2.2.data → loopmath-0.2.3.data}/data/spec/ocp_conformance.py +0 -0
  48. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/WHEEL +0 -0
  49. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/entry_points.txt +0 -0
  50. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/licenses/LICENSE +0 -0
  51. {loopmath-0.2.2.dist-info → loopmath-0.2.3.dist-info}/top_level.txt +0 -0
loopmath/__init__.py CHANGED
@@ -5,4 +5,4 @@ The command line (`loopmath`, or its alias `loop`) is the interface; see the
5
5
  README and `loopmath --help`.
6
6
  """
7
7
 
8
- __version__ = "0.2.2"
8
+ __version__ = "0.2.3"
loopmath/belief/design.py CHANGED
@@ -464,10 +464,12 @@ def source_label(doc: dict) -> str:
464
464
  ext = run.get("ext") or {}
465
465
  share = ext.get("dev.loopmath.share")
466
466
  if isinstance(share, dict):
467
- # the importer writes `source` and `task.org` as `shared:<org_hash>` already: never prefix twice
467
+ # the importer writes `source` and `task.org` as `shared:<org_hash>` already: never prefix twice. With no
468
+ # org at all the source is `shared`, the name `fit --without shared` takes (0.2.3, 22X review N1)
468
469
  org = (share.get("source") or share.get("org") or share.get("org_hash") or (run.get("task") or {}).get("org")
469
- or "shared")
470
- return "shared:" + str(org).removeprefix("shared:")
470
+ or "")
471
+ name = str(org).removeprefix("shared:")
472
+ return f"shared:{name}" if name and name != "shared" else "shared"
471
473
  src = (run.get("task") or {}).get("source")
472
474
  kind = src.get("kind") if isinstance(src, dict) else src
473
475
  kind = str(kind or "").strip().lower()
loopmath/belief/fit.py CHANGED
@@ -526,7 +526,8 @@ def fit(home: Path, *, no_prior: bool = False, without: tuple[str, ...] = (), fu
526
526
  with fit_lock(home, wait_s):
527
527
  remove_partials(home)
528
528
  config = _read_config(home)
529
- weight = float(config.get("benchmark_prior_weight", prior_data.BENCHMARK_PRIOR_WEIGHT))
529
+ raw_weight = config.get("benchmark_prior_weight") # unset: each benchmark's own weight (spec 04 section 5)
530
+ weight = None if raw_weight is None else float(raw_weight)
530
531
  try:
531
532
  features, features_error = FeatureSet.from_config(config.get("features")), None
532
533
  except FeatureConfigError as exc: # a hand-edited config.toml: fit with the built-ins, say why
@@ -550,9 +551,10 @@ def fit(home: Path, *, no_prior: bool = False, without: tuple[str, ...] = (), fu
550
551
  raise UnknownSource(unknown_source_message(unknown, known))
551
552
  if not any(hr.rows for hr in heads_rows.values()):
552
553
  raise NothingToFit(_nothing_to_fit(no_prior, without, dropped))
553
- specs = []
554
+ specs, bench_weights = [], {}
554
555
  if not no_prior and "benchmark" not in without:
555
556
  specs = prior_data.benchmark_factors(benchmarks, weight=weight)
557
+ bench_weights = prior_data.benchmark_weights(benchmarks, weight=weight)
556
558
  seed_key = input_key(heads_rows, specs, {"no_prior": no_prior, "without": sorted(without), "eb": eb,
557
559
  "benchmark_prior_weight": weight})
558
560
  forest = Forest()
@@ -604,7 +606,7 @@ def fit(home: Path, *, no_prior: bool = False, without: tuple[str, ...] = (), fu
604
606
  "seed_key": seed_key, "timebox_effort": design_rows.TIMEBOX_EFFORT_COST,
605
607
  "timebox_terms": list(design_rows.TIMEBOX_LEVELS),
606
608
  "options": {"no_prior": no_prior, "without": list(without), "full": full, "eb": eb,
607
- "benchmark_prior_weight": weight},
609
+ "benchmark_prior_weight": weight, "benchmark_weights": bench_weights},
608
610
  "runs_by_source": runs_by_source,
609
611
  "n_runs": {"prior": sum(v for k, v in runs_by_source.items() if k != "user"),
610
612
  "user": runs_by_source.get("user", 0)},
loopmath/belief/priors.py CHANGED
@@ -10,8 +10,11 @@
10
10
  version nodes: success head, the logit gap to the benchmark's reference model with variance
11
11
  `1 / (w p (1 - p))`; cost head, the log cost ratio with variance `1 / w`; tokens head
12
12
  (`kind = "tokens"`, published total tokens for the evaluation), the log token ratio
13
- with variance `1 / w`. `w` is config `benchmark_prior_weight` (default 5): one benchmark
14
- result counts like about 5 runs.
13
+ with variance `1 / w`. `w` is one model's share of the benchmark's weight `w_b` (0.2.3,
14
+ lane 23K): the `weight` of its `[[benchmark]]` table (default 5, about 5 runs), shared by
15
+ the model's k results on that benchmark, so each gets `w_b / k` and a model with five
16
+ efforts pulls as hard in total as a model with one. Config `benchmark_prior_weight`, when
17
+ set, is one `w_b` for every benchmark; 0 turns the factors off.
15
18
  """
16
19
 
17
20
  from __future__ import annotations
@@ -91,8 +94,30 @@ def _model_terms(model: str, effort: str | None, sign: float) -> list[tuple[str,
91
94
  return terms
92
95
 
93
96
 
94
- def benchmark_factors(path: Path | None = None, *, weight: float = BENCHMARK_PRIOR_WEIGHT) -> list[FactorSpec]:
95
- """Prior factors from `benchmarks.toml`. A missing file gives no factors."""
97
+ def benchmark_weights(path: Path | None = None, *, weight: float | None = None) -> dict[str, float]:
98
+ """Each benchmark's weight `w_b`: `weight` for every benchmark when it is given (config
99
+ `benchmark_prior_weight`), else the benchmark's own `weight` field, else 5. Empty for a missing file."""
100
+ path = path or benchmark_path()
101
+ if not path.is_file():
102
+ return {}
103
+ data = tomllib.loads(path.read_text(encoding="utf-8"))
104
+ return {str(b["id"]): _weight_of(b, weight) for b in data.get("benchmark") or [] if b.get("id")}
105
+
106
+
107
+ def _weight_of(bench: dict, weight: float | None) -> float:
108
+ if weight is not None:
109
+ return float(weight)
110
+ own = bench.get("weight")
111
+ return float(own) if isinstance(own, (int, float)) and not isinstance(own, bool) else BENCHMARK_PRIOR_WEIGHT
112
+
113
+
114
+ def benchmark_factors(path: Path | None = None, *, weight: float | None = None) -> list[FactorSpec]:
115
+ """Prior factors from `benchmarks.toml`. A missing file gives no factors.
116
+
117
+ `weight` (config `benchmark_prior_weight`) overrides every benchmark's own weight; None reads
118
+ each benchmark's `weight`. A model's k results on one benchmark share its weight: each factor
119
+ has weight `w_b / k` (B1). A benchmark with weight 0 or less gives no factors.
120
+ """
96
121
  path = path or benchmark_path()
97
122
  if not path.is_file():
98
123
  return []
@@ -103,11 +128,15 @@ def benchmark_factors(path: Path | None = None, *, weight: float = BENCHMARK_PRI
103
128
  for bid, bench in benches.items():
104
129
  kind = str(bench.get("kind") or "success")
105
130
  ref = canonical_model_id(str(bench.get("reference") or ""))
131
+ w_b = _weight_of(bench, weight)
132
+ if w_b <= 0:
133
+ continue
106
134
  rows = [r for r in results if r["benchmark"] == bid]
107
135
  ref_rows = [r for r in rows if canonical_model_id(str(r["model"])) == ref]
108
136
  if not ref_rows:
109
137
  continue
110
138
  ref_value = float(ref_rows[0]["value"])
139
+ by_model: dict[str, list[tuple[str, list, float, dict]]] = {} # model -> (head, terms, mean, row) per result
111
140
  for r in rows:
112
141
  model = canonical_model_id(str(r["model"]))
113
142
  if model == ref:
@@ -117,14 +146,24 @@ def benchmark_factors(path: Path | None = None, *, weight: float = BENCHMARK_PRI
117
146
  terms = _cancel(terms)
118
147
  if not terms:
119
148
  continue
120
- note = f"{bid}: {model} {value} against {ref} {ref_value} ({r.get('url', '')}, {r.get('date', '')})"
121
149
  if kind == "success":
122
150
  p = min(max(value, 0.01), 0.99)
123
151
  p_ref = min(max(ref_value, 0.01), 0.99)
124
152
  gap = math.log(p / (1 - p)) - math.log(p_ref / (1 - p_ref))
125
- out.append(FactorSpec("success", terms, gap, 1.0 / (weight * p * (1 - p)), note))
153
+ by_model.setdefault(model, []).append(("success", terms, gap, r))
126
154
  elif kind in ("cost", "tokens") and value > 0 and ref_value > 0: # Tokens feed the tokens head only
127
- out.append(FactorSpec(kind, terms, math.log(value / ref_value), 1.0 / weight, note))
155
+ by_model.setdefault(model, []).append((kind, terms, math.log(value / ref_value), r))
156
+ for model, items in by_model.items():
157
+ w = w_b / len(items) # one budget per model and benchmark (B1)
158
+ share = f"weight {w_b:g} over {len(items)}" if len(items) > 1 else f"weight {w_b:g}"
159
+ for head, terms, mean, r in items:
160
+ note = (f"{bid}: {model} {float(r['value'])} against {ref} {ref_value} ({share}; "
161
+ f"{r.get('url', '')}, {r.get('date', '')})")
162
+ if head == "success":
163
+ p = min(max(float(r["value"]), 0.01), 0.99)
164
+ out.append(FactorSpec("success", terms, mean, 1.0 / (w * p * (1 - p)), note))
165
+ else:
166
+ out.append(FactorSpec(head, terms, mean, 1.0 / w, note))
128
167
  return out
129
168
 
130
169
 
loopmath/belief/state.py CHANGED
@@ -792,10 +792,13 @@ class FitState:
792
792
  gate_pass=res[hg] if hg is not None else None, rounds=res[hr],
793
793
  cost_per_round=Money(usd=res[hpr[0]], tokens=res[hpr[1]]) if hpr else None)
794
794
  for piece, (hu, ht, hg, hr, hpr) in handles.items()}
795
+ # The typical run (0.2.3): the median of the same simulated runs, one per draw, the 80% range is read from
796
+ run_usd = dataclasses.replace(res[h_run[1]], median=float(np.median(rd.sim_usd)))
797
+ run_tokens = dataclasses.replace(res[h_run[2]], median=float(np.median(rd.sim_tokens)))
795
798
  pred = Prediction(
796
799
  config=cfg.id,
797
800
  p_success=res[h_run[0]],
798
- cost=Money(usd=res[h_run[1]], tokens=res[h_run[2]]),
801
+ cost=Money(usd=run_usd, tokens=run_tokens),
799
802
  ell=Money(usd=res[h_run[3]], tokens=res[h_run[4]]),
800
803
  rounds=res[h_run[5]],
801
804
  per_piece=per_piece,
@@ -137,7 +137,7 @@ def build_session(args: argparse.Namespace) -> tuple[Session | None, int]:
137
137
  rec = engine.recommend(belief, asked, rule, usual=usual, usual_from=usual_from, configs=configs,
138
138
  settings=settings, diff=C.diff_fn(), keep=[*(u.id for u in user),
139
139
  *(r.id for r in recorded)],
140
- **offered_kw(offered))
140
+ own_runs=storeread.own_runs(home), **offered_kw(offered))
141
141
  rec.task = task
142
142
  start = None
143
143
  if getattr(args, "start", None):
@@ -305,6 +305,17 @@ def candidate_entry(rec: Recommendation, c: Any) -> dict[str, Any]:
305
305
  "origin": c.origin, **rec.search_fields(c.config.id)}
306
306
 
307
307
 
308
+ def rescue_entry(session: Session, cfg: Configuration) -> dict[str, Any]:
309
+ """The rescue workflow as a candidate when the search did not offer it (22W N1, 0.2.3): predicted as a
310
+ candidate is, under the recommendation's task, rule and rescue, with origin `rescue`."""
311
+ rec = session.rec
312
+ bands: dict[str, Any] = {}
313
+ preds, medians = engine.predict_with_medians(session.belief, session.task, [cfg], session.rule, rec.rescue, bands)
314
+ view = with_numbers(rec, medians=medians, bands=bands)
315
+ return {"config": cfg.to_dict(), "label": rec.label(cfg), "numbers": numbers_of(view, preds[0]),
316
+ "origin": "rescue", **rec.search_fields(cfg.id)}
317
+
318
+
308
319
  def context_payload(session: Session) -> dict[str, Any]:
309
320
  """`GET /api/context` (spec 02, `builder`), built once per session."""
310
321
  if session.context is not None:
@@ -329,6 +340,10 @@ def context_payload(session: Session) -> dict[str, Any]:
329
340
  wanted.add(rec.rescue_config.id)
330
341
  top += [c for c in offer[TOP_CANDIDATES:]
331
342
  if c.config.id not in ids and (c.origin in KEEP_ORIGINS or c.config.id in wanted)]
343
+ candidates = [candidate_entry(rec, c) for c in top]
344
+ fix = rec.rescue_config
345
+ if fix is not None and fix.id not in retired and rec.by_id(fix.id) is None:
346
+ candidates.append(rescue_entry(session, fix))
332
347
  session.context = {
333
348
  "schema": SCHEMA,
334
349
  "task": C.task_block(rec.task, session.belief),
@@ -339,9 +354,10 @@ def context_payload(session: Session) -> dict[str, Any]:
339
354
  "rescue": core["rescue"],
340
355
  "reference": reference,
341
356
  "choices": choices,
342
- "candidates": [candidate_entry(rec, c) for c in top],
357
+ "candidates": candidates,
343
358
  "catalog": catalog(session),
344
359
  "start": session.start.to_dict() if session.start is not None else None,
360
+ "own_runs": rec.own_runs, # 0.2.3: the user's recorded runs; 0 leads each run cost with the typical run
345
361
  }
346
362
  return session.context
347
363
 
loopmath/cli_registry.py CHANGED
@@ -274,7 +274,11 @@ def _add_recording(sub) -> None:
274
274
  q.add_argument("key", nargs="?", default=None, metavar="KEY")
275
275
  _common(q)
276
276
  q.set_defaults(func=_lazy("loopmath.store.commands:config_get"))
277
- q = cs.add_parser("set", help="set one key")
277
+ q = cs.add_parser("set", help="set one key",
278
+ description="Sets one key in config.toml. benchmark_prior_weight is unset by default: each "
279
+ "benchmark in the shipped benchmarks.toml has its own weight, in runs, shared by "
280
+ "a model's results on it; a number here is one weight for every benchmark, and 0 "
281
+ "turns benchmark priors off.")
278
282
  q.add_argument("key", metavar="KEY")
279
283
  q.add_argument("value", metavar="VALUE")
280
284
  _common(q)
@@ -335,7 +339,7 @@ def _add_learning(sub) -> None:
335
339
  ps = p.add_subparsers(dest="prior_command", required=True)
336
340
  q = ps.add_parser("build", help="rebuild the packaged prior bundle from the sweep, E0 and RQ1 inputs")
337
341
  q.add_argument("--out", default=None, metavar="DIR", help="where to write the bundle (default: the packaged bundle folder, which it replaces)")
338
- q.add_argument("--sweep-dir", default=None, metavar="PATH", help="sweep results (default: LOOPMATH_SWEEP_DIR)")
342
+ q.add_argument("--sweep-dir", action="append", default=None, metavar="PATH", help="sweep results, one folder per batch; repeatable (default: LOOPMATH_SWEEP_DIR, one folder)")
339
343
  q.add_argument("--e0-corpus", default=None, metavar="PATH", help="E0 corpus (default: LOOPMATH_E0_CORPUS)")
340
344
  q.add_argument("--rq1-dir", default=None, metavar="PATH", help="RQ1 OCP documents (default: LOOPMATH_PRIOR_RQ1)")
341
345
  q.add_argument("--lanes-dir", default=None, metavar="PATH", help="build lane rows (default: LOOPMATH_PRIOR_LANES)")
@@ -31,6 +31,7 @@ from typing import Any, Callable
31
31
  from .. import output
32
32
  from ..output import EXIT_NOT_FOUND, EXIT_OK, EXIT_USER, emit_json, fail
33
33
  from .. import taskmodel
34
+ from ..priors.show import starting_prior
34
35
  from ..store.ids import since_days as since_window
35
36
  from ..taskmodel import LABEL_VERSION
36
37
  from . import history as H
@@ -124,6 +125,9 @@ def onboard(args: argparse.Namespace) -> int:
124
125
  # The one line printed even when stderr is captured: the wait that follows can be long.
125
126
  print(f"onboard: reading Claude Code and Codex history, {_window_words(since)}; "
126
127
  "a large history can take several minutes", file=sys.stderr, flush=True)
128
+ # Then, also when captured, what the answers start from, once (the skill shows this line to the user).
129
+ prior = starting_prior()
130
+ print(prior["line"], file=sys.stderr, flush=True)
127
131
  hist = deps.load_history(since_days, logs=deps.logs, progress=_progress, stage=_stage)
128
132
  now = deps.now()
129
133
  groups, before_window = H.in_window(H.group_sessions(hist.graph, hist.by_id), since_days, now=now)
@@ -160,6 +164,7 @@ def onboard(args: argparse.Namespace) -> int:
160
164
 
161
165
  payload: dict[str, Any] = {
162
166
  "dry_run": dry_run,
167
+ "prior": prior,
163
168
  "since": since,
164
169
  "since_days": round(since_days, 3),
165
170
  "window": _window(groups),
@@ -8,7 +8,9 @@ The read API for the belief model (lane 5):
8
8
  - `bundle_entries()`: the same documents as `(source, document)` pairs, so the fit
9
9
  knows where each run came from whatever its label says (spec 04 section 1).
10
10
  - `shipped_overlap(run_ids)` and `overlap_note(counts)`: how many of the user's runs
11
- are also in a shipped source; `run import`, `fit` and `status` say it once.
11
+ are also in a shipped source; `run import`, `fit` and `status` say it once. A
12
+ stored run and its shipped copy are matched on `run_id_of`, the run id less a
13
+ start stamp (`stable_run_id`), since the shipped rq1 ids lose theirs (23B).
12
14
  - `manifest()`, `sources()`, `bundle_dir()`: what the bundle holds and where.
13
15
 
14
16
  Reading never touches the network or the store; the bundle is package data.
@@ -18,6 +20,7 @@ from __future__ import annotations
18
20
 
19
21
  import gzip
20
22
  import json
23
+ import re
21
24
  from collections import Counter
22
25
  from pathlib import Path
23
26
  from typing import Iterable, Iterator
@@ -71,15 +74,33 @@ def bundle_docs(without: Iterable[str] = (), sources: Iterable[str] | None = Non
71
74
  yield doc
72
75
 
73
76
 
77
+ _STAMP = re.compile(r"-\d{8}-\d{6}(?=-|$)") # lane 10's run start, YYYYMMDD-HHMMSS local time
78
+
79
+
80
+ def stable_run_id(run_id: str) -> str:
81
+ """A run id less its `-YYYYMMDD-HHMMSS` start stamp; an id without one is unchanged.
82
+
83
+ The shipped rq1 runs carry lane 10's ids without the stamp (a real date and time
84
+ does not ship, 23B) while an RQ1 store keeps it, so a stored run and its shipped
85
+ copy compare on this key. It is the id itself, not a hash: the stamp space is
86
+ small enough that a hash of the stamped id would give the time away.
87
+ """
88
+ return _STAMP.sub("", run_id)
89
+
90
+
74
91
  def run_id_of(doc: dict) -> str:
75
- """The run id the fit keys on (`belief.design.parse_run`), read without parsing the run."""
92
+ """The key a stored run and its shipped copy are matched on, read without parsing the run.
93
+
94
+ The run id the fit keys on (`belief.design.parse_run`), less a start stamp
95
+ (`stable_run_id`); the fit and `shipped_overlap` both compare on it.
96
+ """
76
97
  run = doc.get("run") or {}
77
98
  task = run.get("task") or {}
78
- return str(run.get("id") or task.get("id") or (run.get("labels") or {}).get("task") or "unknown")
99
+ return stable_run_id(str(run.get("id") or task.get("id") or (run.get("labels") or {}).get("task") or "unknown"))
79
100
 
80
101
 
81
102
  def shipped_ids(directory: Path | None = None) -> dict[str, str]:
82
- """Run id to its shipped source, for every bundled run."""
103
+ """Run id (`run_id_of`) to its shipped source, for every bundled run."""
83
104
  out: dict[str, str] = {}
84
105
  for name, doc in bundle_entries(directory=directory):
85
106
  out.setdefault(run_id_of(doc), name)
@@ -89,10 +110,12 @@ def shipped_ids(directory: Path | None = None) -> dict[str, str]:
89
110
  def shipped_overlap(run_ids: Iterable[str], directory: Path | None = None) -> dict[str, int]:
90
111
  """How many of `run_ids` (the user's runs) are also in each shipped source.
91
112
 
92
- A fit uses the store's copy of such a run and leaves the shipped copy out.
113
+ A fit uses the store's copy of such a run and leaves the shipped copy out. The
114
+ ids compare as `run_id_of` does, without a start stamp.
93
115
  """
94
116
  ids = shipped_ids(directory)
95
- return dict(Counter(ids[r] for r in set(run_ids) if r in ids))
117
+ keys = [stable_run_id(r) for r in set(run_ids)]
118
+ return dict(Counter(ids[k] for k in keys if k in ids))
96
119
 
97
120
 
98
121
  def overlap_note(counts: dict[str, int] | None) -> str | None:
@@ -106,8 +129,8 @@ def overlap_note(counts: dict[str, int] | None) -> str | None:
106
129
  else:
107
130
  where = "the shipped prior (" + ", ".join(f"{k} {v}" for k, v in sorted(overlap.items(), key=lambda kv: -kv[1])) + ")"
108
131
  if total == 1:
109
- return f"1 of your runs is also in {where} (same run id): fits use your copy"
110
- return f"{total} of your runs are also in {where} (same run ids): fits use your copies"
132
+ return f"1 of your runs is also in {where} (same run): fits use your copy"
133
+ return f"{total} of your runs are also in {where} (same runs): fits use your copies"
111
134
 
112
135
 
113
136
  iter_runs = bundle_docs
@@ -1,11 +1,13 @@
1
1
  """Published benchmark results (lane 11): `priors/benchmarks.toml`.
2
2
 
3
3
  The file has `[[benchmark]]` tables (`id`, `title`, `metric`, `kind` success,
4
- cost or tokens, `reference` model, `reference_effort`, `harness`, `url`, `note`) and
4
+ cost or tokens, `reference` model, `reference_effort`, `weight`, `harness`, `url`, `note`) and
5
5
  `[[result]]` tables (`benchmark`, `model`, `effort`, `value`, `harness`, `date`,
6
6
  `url`, `note`). Lane 5's `belief.priors.benchmark_factors` turns them into prior
7
7
  factors on version nodes (spec 04 section 5); it reads the first result of the
8
8
  reference model as the reference, so that row is the one at `reference_effort`.
9
+ `weight` is the benchmark's weight in runs (5 when absent), shared by each model's
10
+ results on it (0.2.3, lane 23K; spec 04 section 5).
9
11
 
10
12
  This module loads the file and checks it: every value is a published number
11
13
  with its own source line, success values are fractions, the reference
@@ -49,6 +51,9 @@ def check_benchmarks(data: dict) -> list[str]:
49
51
  problems += [f"{where}: no {k}" for k in _BENCH_KEYS if not b.get(k)]
50
52
  if b.get("kind") and b["kind"] not in KINDS:
51
53
  problems.append(f"{where}: kind {b['kind']!r} is not one of {', '.join(KINDS)}")
54
+ if "weight" in b and (isinstance(b["weight"], bool) or not isinstance(b["weight"], (int, float))
55
+ or b["weight"] < 0):
56
+ problems.append(f"{where}: weight is not a number of 0 or more")
52
57
  if b.get("id") in benches:
53
58
  problems.append(f"{where}: duplicate id")
54
59
  benches[str(b.get("id"))] = b
@@ -3,8 +3,9 @@
3
3
  # design/0.1/data/benchmarks/artificialanalysis-2026-09-23.json (Terminal-Bench 4.0) and
4
4
  # artificialanalysis-2026-09-25.json (Terminal-Bench 2.1). Values are copied as published;
5
5
  # a missing cell stays missing. Lane 5's belief.priors.benchmark_factors reads this file: success
6
- # results become logit gaps to the reference model with variance 1 / (w p (1 - p)), w =
7
- # benchmark_prior_weight (default 5). The first result for the reference model is the reference.
6
+ # results become logit gaps to the reference model with variance 1 / (w p (1 - p)), w = the
7
+ # benchmark's weight over the model's k results on it (config benchmark_prior_weight, when set,
8
+ # replaces every weight). The first result for the reference model is the reference.
8
9
 
9
10
  [[benchmark]]
10
11
  id = "aa-terminal-bench-4.0"
@@ -13,6 +14,7 @@ metric = "pass@1 averaged over three repeats of 66 tasks, as a fraction"
13
14
  kind = "success"
14
15
  reference = "claude-opus-5"
15
16
  reference_effort = "max"
17
+ weight = 5.0
16
18
  harness = "mini-swe-agent (Artificial Analysis)"
17
19
  url = "https://artificialanalysis.ai/evaluations/terminalbench-4-0"
18
20
  note = "One harness for every model, so the gaps compare models rather than scaffolds. Variants without reasoning are left out (loopmath does not run them). Haiku 4.5 is the reasoning variant, which names no effort. Values are k of 198 task runs."
@@ -24,6 +26,7 @@ metric = "total tokens for the whole evaluation (66 tasks x 3 repeats): input (w
24
26
  kind = "tokens"
25
27
  reference = "claude-opus-5"
26
28
  reference_effort = "max"
29
+ weight = 5.0
27
30
  harness = "mini-swe-agent (Artificial Analysis)"
28
31
  url = "https://artificialanalysis.ai/evaluations/terminalbench-4-0"
29
32
  note = "D54: a log-ratio factor on the tokens head only, never converted to dollars. The same three streams for every model, so the ratio is like for like."
@@ -35,6 +38,7 @@ metric = "pass@1 averaged over three repeats of 89 tasks, as a fraction"
35
38
  kind = "success"
36
39
  reference = "claude-opus-5"
37
40
  reference_effort = "max"
41
+ weight = 5.0
38
42
  harness = "Terminus 2 (Artificial Analysis)"
39
43
  url = "https://artificialanalysis.ai/evaluations/terminalbench-2-1"
40
44
  note = "One harness for every model (Terminus 2, as the page says), so the gaps compare models rather than scaffolds. Variants without reasoning are left out. Haiku 4.5 is the reasoning variant, which names no effort. Values are k of 267 task runs. Artificial Analysis publishes no Terminal-Bench 2.1 result for claude-opus-5-5, gpt-6-sol or gpt-6-luna, and only the max effort for claude-sonnet-5 and claude-fable-5; those cells stay missing. The official Terminal-Bench 2.1 board mixes six harnesses and is left out (lane 22K)."
@@ -46,6 +50,7 @@ metric = "total tokens for the whole evaluation (89 tasks x 3 repeats): input (w
46
50
  kind = "tokens"
47
51
  reference = "claude-opus-5"
48
52
  reference_effort = "max"
53
+ weight = 5.0
49
54
  harness = "Terminus 2 (Artificial Analysis)"
50
55
  url = "https://artificialanalysis.ai/evaluations/terminalbench-2-1"
51
56
  note = "D54: a log-ratio factor on the tokens head only, never converted to dollars. The same three streams for every model, so the ratio is like for like. Only the variants the evaluation pages show carry token counts."
loopmath/priors/build.py CHANGED
@@ -4,7 +4,9 @@
4
4
  validates every document, and writes one gzipped JSON Lines file per source plus
5
5
  `manifest.json` with provenance. Output bytes are deterministic for the same
6
6
  inputs and salt (runs sorted by id, gzip mtime 0), so a rebuild diffs cleanly;
7
- only the manifest's `built_at` changes. The total must stay under 5 MB.
7
+ only the manifest's `built_at` changes. `built_at` is the build's UTC date
8
+ (`YYYY-MM-DD`), the prior's date shown to users; no clock time ships. The total
9
+ must stay under 5 MB.
8
10
  """
9
11
 
10
12
  from __future__ import annotations
@@ -18,7 +20,7 @@ import os
18
20
  import secrets
19
21
  from dataclasses import dataclass, field
20
22
  from pathlib import Path
21
- from typing import Callable, Iterable
23
+ from typing import Callable, Iterable, Sequence
22
24
 
23
25
  from .. import __version__
24
26
  from . import ocpdoc
@@ -48,8 +50,9 @@ class SourceResult:
48
50
  counts: dict = field(default_factory=dict)
49
51
 
50
52
 
51
- def now_local() -> str:
52
- return _dt.datetime.now().astimezone().isoformat(timespec="seconds")
53
+ def build_date() -> str:
54
+ """Today's date in UTC, `YYYY-MM-DD`: the bundle's `built_at`, with no clock time or offset."""
55
+ return _dt.datetime.now(_dt.timezone.utc).date().isoformat()
53
56
 
54
57
 
55
58
  def digest_files(paths: Iterable[Path]) -> dict:
@@ -94,7 +97,7 @@ def build_bundle(out_dir: Path, results: list[SourceResult], *, salt: str | None
94
97
  salt = salt or secrets.token_hex(32)
95
98
  manifest: dict = {
96
99
  "schema": MANIFEST_SCHEMA,
97
- "built_at": now_local(),
100
+ "built_at": build_date(),
98
101
  "loopmath_version": __version__,
99
102
  "ocp": ocpdoc.OCP_VERSION,
100
103
  "config_id_impl": ocpdoc.CONFIG_ID_IMPL,
@@ -160,29 +163,66 @@ def build_bundle(out_dir: Path, results: list[SourceResult], *, salt: str | None
160
163
 
161
164
 
162
165
  # ---------------------------------------------------------------- source runners
163
- def run_sweep(results_dir: Path) -> SourceResult:
166
+ _BATCH_NOTES = {
167
+ "sweep0925": "sweep0925: gpt-6-sol xhigh and gpt-6-luna low developers, reviewer claude-opus-5 xhigh, the "
168
+ "sweep0830 plans reused (its plan attempts are those shared plans)",
169
+ }
170
+
171
+
172
+ def run_sweep(results_dirs: Path | Sequence[Path]) -> SourceResult:
173
+ """The sweep source from one results folder per batch (`prior build --sweep-dir`, repeatable)."""
164
174
  from .sweep import CONVERTER_VERSION, iter_sweep, sweep_layout
165
175
 
166
- runs_dir, attempts = sweep_layout(results_dir)
167
- if not any(runs_dir.glob("*.run.json")):
168
- raise BundleError(f"sweep input {results_dir} has no run files (*.run.json), directly or in dagr/")
169
- docs, warnings = [], []
170
- counts = {"infra_error_runs": 0, "runs_with_infra_rounds": 0, "cost_incomplete_runs": 0,
171
- "superseded_rows": 0, "later_rows": 0}
172
- for _, doc, warn in iter_sweep(results_dir, producer_version=__version__):
173
- docs.append(doc)
174
- warnings += warn
175
- info = doc["run"]["ext"]["dev.loopmath.prior"]
176
- counts["infra_error_runs"] += bool(info["infra_error"])
177
- counts["runs_with_infra_rounds"] += info["infra_rounds"] > 0
178
- counts["cost_incomplete_runs"] += not info["cost_complete"]
179
- counts["superseded_rows"] += info["superseded_rows"]
180
- counts["later_rows"] += info["later_rows"]
181
- files = sorted(runs_dir.glob("*.run.json")) + [attempts]
182
- inputs = {"description": "loopmath internal sweep (sweep0830): the experiment repository's results, "
183
- "contract v3 run files and attempts.jsonl",
184
- **digest_files([f for f in files if f.exists()])}
176
+ dirs = [Path(results_dirs)] if isinstance(results_dirs, (str, Path)) else [Path(d) for d in results_dirs]
177
+ docs, warnings, files = [], [], []
178
+ counts: dict = {"infra_error_runs": 0, "runs_with_infra_rounds": 0, "cost_incomplete_runs": 0,
179
+ "superseded_rows": 0, "later_rows": 0, "batches": {}}
180
+ batch_inputs: dict = {}
181
+ for results_dir in dirs:
182
+ runs_dir, attempts = sweep_layout(results_dir)
183
+ if not any(runs_dir.glob("*.run.json")):
184
+ raise BundleError(f"sweep input {results_dir} has no run files (*.run.json), directly or in dagr/")
185
+ batches = set()
186
+ for _, doc, warn in iter_sweep(results_dir, producer_version=__version__):
187
+ docs.append(doc)
188
+ warnings += warn
189
+ run = doc["run"]
190
+ info = run["ext"]["dev.loopmath.prior"]
191
+ counts["infra_error_runs"] += bool(info["infra_error"])
192
+ counts["runs_with_infra_rounds"] += info["infra_rounds"] > 0
193
+ counts["cost_incomplete_runs"] += not info["cost_complete"]
194
+ counts["superseded_rows"] += info["superseded_rows"]
195
+ counts["later_rows"] += info["later_rows"]
196
+ batch = run["task"]["source"]["ref"]
197
+ batches.add(batch)
198
+ model = ((run["configuration"]["settings"].get("implement") or {}).get("model") or {}).get("id", "?")
199
+ row = counts["batches"].setdefault(batch, {}).setdefault(model, {
200
+ "runs": 0, "accepted": 0, "not_accepted": 0, "no_verdict": 0, "infra_error_runs": 0,
201
+ "cost_incomplete_runs": 0})
202
+ verdicts = {sig["name"]: sig["value"] for sig in run.get("signals") or [] if sig["kind"] == "verdict"}
203
+ acc = (False if {"fail", "reject"} & set(verdicts.values())
204
+ else True if (verdicts.get("tests"), verdicts.get("referee")) == ("pass", "accept") else None)
205
+ row["runs"] += 1
206
+ row["accepted"] += acc is True
207
+ row["not_accepted"] += acc is False
208
+ row["no_verdict"] += acc is None
209
+ row["infra_error_runs"] += bool(info["infra_error"])
210
+ row["cost_incomplete_runs"] += not info["cost_complete"]
211
+ if len(batches) != 1 or batches & set(batch_inputs):
212
+ raise BundleError(f"sweep input {results_dir} must hold exactly one batch not given before, "
213
+ f"found {', '.join(sorted(batches))}")
214
+ batch_files = [f for f in sorted(runs_dir.glob("*.run.json")) + [attempts] if f.exists()]
215
+ batch_inputs[batches.pop()] = digest_files(batch_files)
216
+ files += batch_files
217
+ counts["batches"] = {b: dict(sorted(m.items())) for b, m in sorted(counts["batches"].items())}
218
+ inputs = {"description": f"loopmath internal sweep batches ({', '.join(sorted(batch_inputs))}): the experiment "
219
+ "repository's results, contract v3 run files and attempts.jsonl, one folder per batch",
220
+ **digest_files(files), "batches": dict(sorted(batch_inputs.items()))}
185
221
  notes = [
222
+ "one results folder per batch; the batch id is the run id prefix and the source ref, and every batch "
223
+ "uses the sweep0830 task ids: the batches ran the same tasks and hidden gates, so they share a task node",
224
+ *[_BATCH_NOTES[b] for b in sorted(batch_inputs) if b in _BATCH_NOTES],
225
+ "no real date or clock time: each run starts at 1970-01-01T00:00:00Z and keeps its real elapsed seconds",
186
226
  "run files are authoritative; attempts.jsonl rows are matched by developer start time",
187
227
  "superseded_rows: earlier executions the harness re-ran after a crash; later_rows: executions after the "
188
228
  "run file was written; neither is in the bundle",
@@ -204,8 +244,8 @@ def run_e0(corpus_dir: Path) -> SourceResult:
204
244
  raise BundleError(f"E0 input {corpus_dir} has no sessions.jsonl")
205
245
  counts: dict = {}
206
246
  docs = list(iter_e0(corpus_dir, producer_version=__version__, counts=counts))
207
- inputs = {"description": "E0 corpus copy (parser-spec v1, built 2026-08-28): sessions.jsonl and "
208
- "session-dag-join.jsonl", **digest_files([f for f in files if f.is_file()])}
247
+ inputs = {"description": "E0 corpus copy (parser-spec v1): sessions.jsonl and session-dag-join.jsonl",
248
+ **digest_files([f for f in files if f.is_file()])}
209
249
  notes = [
210
250
  "one logged habit run per real main session (catalog solo shape: harness, primary model, dominant effort)",
211
251
  "no verdicts (D7): no signals and no acceptance rule; these runs feed the cost and tokens heads only",
@@ -214,6 +254,7 @@ def run_e0(corpus_dir: Path) -> SourceResult:
214
254
  "left out: fleet stubs (no model, an API error), sessions with no model or tokens, temporary folders",
215
255
  "Codex input tokens exclude the cached tokens (input minus cached, as ingest.codex does)",
216
256
  "dollars priced with the packaged prices.toml",
257
+ "no real date or clock time: each run starts at 1970-01-01T00:00:00Z and keeps its real elapsed seconds",
217
258
  ]
218
259
  return SourceResult("e0", docs, inputs, CONVERTER_VERSION, notes, [], counts)
219
260
 
@@ -230,15 +271,18 @@ def run_rq1(ocp_dir: Path) -> SourceResult:
230
271
  info = doc["run"]["ext"]["dev.loopmath.prior"]
231
272
  changed += info["lane10_config_id"] != doc["run"]["configuration"]["id"]
232
273
  manifest = ocp_dir / "MANIFEST.json"
233
- inputs = {"description": "loopmath internal RQ1 phase 1: lane 10's OCP v0.3 conversion (D15)",
274
+ inputs = {"description": "loopmath internal RQ1 phase 1: agent runs on ALE-Bench heuristic problems, as the "
275
+ "experiment wrote them in OCP v0.3 (D15)",
234
276
  **digest_files(files + ([manifest] if manifest.is_file() else []))}
235
277
  notes = [
236
278
  "D15 labels checked on every run: type feature, repo ale-bench, subtype ahc/<problem>, source rq1, "
237
279
  "rule heldout_perf>=2400",
238
- "configuration ids recomputed with the bundle's canonical form; lane 10's id kept as lane10_config_id",
280
+ "configuration ids recomputed with the bundle's canonical form; the experiment's id kept as lane10_config_id",
239
281
  "attempt roles written as their piece's role (D32)",
240
- "dollars repriced from tokens with the packaged prices.toml; lane 10's list-price total is recorded_usd",
241
- "lane 10's submission curve (dev.loopmath.rq1) is dropped by the share reduction",
282
+ "dollars repriced from tokens with the packaged prices.toml; the experiment's list-price total is recorded_usd",
283
+ "the experiment's submission curve (dev.loopmath.rq1) is dropped by the share reduction",
284
+ "no real date or clock time: run ids without the experiment's start stamp; each run starts at "
285
+ "1970-01-01T00:00:00Z and keeps its real elapsed seconds (the local offset is gone)",
242
286
  ]
243
287
  return SourceResult("rq1", docs, inputs, CONVERTER_VERSION, notes, [],
244
288
  {"runs": len(docs), "config_ids_changed": changed})
@@ -252,8 +296,11 @@ def run_lanes(input_dir: Path) -> SourceResult:
252
296
  raise BundleError(f"lanes input {input_dir} has no {INPUT_FILE}")
253
297
  counts: dict = {}
254
298
  docs = list(iter_lanes(input_dir, producer_version=__version__, counts=counts))
255
- inputs = {"description": "our own agent build lanes (lane 22L): metadata-only rows from the lane records, "
256
- "reviews and session logs", **digest_files([path])}
299
+ over = [(d["run"].get("ext") or {}).get("dev.loopmath.prior", {}).get("over_budget_rounds") or 0 for d in docs]
300
+ counts["over_budget_runs"] = sum(1 for n in over if n)
301
+ inputs = {"description": "our own agent build lanes, the agent sessions that built loopmath and others on "
302
+ "the same models: metadata-only rows from the lane records, reviews and session logs",
303
+ **digest_files([path])}
257
304
  notes = [
258
305
  "implementer lanes with reviews: catalog implement_review, one round per review up to the first merge, "
259
306
  "tokens measured per round from the implementer's log; follow-up work after the first merge not bundled",
@@ -263,6 +310,10 @@ def run_lanes(input_dir: Path) -> SourceResult:
263
310
  "integration lanes, lanes still running and lanes with no matching log are left out (counted by the "
264
311
  "extractor, not here)",
265
312
  "tokens from loopmath's own Claude Code and Codex parsers; dollars priced with the packaged prices.toml",
313
+ "rounds capped at the catalog budget (3): a lane merged after round 3 ends on round 3's rejection, as the "
314
+ "catalog configuration would have; its real round count and the rounds and tokens past the budget are in "
315
+ "dev.loopmath.prior (over_budget_runs counts these lanes)",
316
+ "no real date or clock time: each run starts at 1970-01-01T00:00:00Z and keeps its real elapsed seconds",
266
317
  ]
267
318
  return SourceResult("lanes", docs, inputs, CONVERTER_VERSION, notes, [], counts)
268
319
 
@@ -1,3 +1,7 @@
1
- OCP v0.3 runs, gzipped, with provenance (lane 11).
1
+ loopmath's shipped prior: OCP v0.3 runs, gzipped, with provenance in `manifest.json`.
2
2
 
3
- `lanes.jsonl.gz` (0.2.2) was built alone (`LOOPMATH_PRIOR_SOURCES=lanes loopmath prior build --lanes-dir PATH`) and merged in with `loopmath.priors.lanes.merge_into_bundle`, which leaves the other files' bytes alone: the other files were not rebuilt. A full `prior build` needs every source's input, `--lanes-dir` included.
3
+ Every file here and `manifest.json` come from one full `loopmath prior build` (0.2.3): `sweep`, `e0`, `rq1` and `lanes`, each with its inputs' digest, converter, notes and counts in the manifest. A full build needs every source's input: `--sweep-dir` (repeatable, one results folder per sweep batch), `--e0-corpus`, `--rq1-dir` and `--lanes-dir`, or their environment variables (`LOOPMATH_SWEEP_DIR` names one folder).
4
+
5
+ - Sweep batches: `sweep0830` and `sweep0925` (gpt-6-sol and gpt-6-luna developers). The batch is the run id prefix (`sweep0925/<run>`) and the source ref. Both batches use the task set's ids (`sweep0830/<task>`): they ran the same tasks and hidden gates, so a task is one node in the fit.
6
+ - No real date or clock time: every run starts at `1970-01-01T00:00:00Z` and keeps its real elapsed seconds, in UTC (`run.ext["dev.loopmath.prior"].clock`). RQ1 run ids carry no start stamp; a store that keeps the stamped ids still matches the shipped copies (`loopmath.priors.run_id_of`). The price table's date (`cost.tariff.date`) is the only date in a run, and `built_at` is the build's UTC date.
7
+ - `lanes`: rounds are capped at the catalog budget of 3 (see the manifest notes).
Binary file
Binary file