codeer-cli 0.1.6__py3-none-any.whl → 0.1.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,5 +62,7 @@ def print_json(value: Any) -> None:
62
62
  def write_json(path: str | None, value: Any) -> None:
63
63
  if not path:
64
64
  return
65
- Path(path).write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
65
+ out = Path(path)
66
+ out.parent.mkdir(parents=True, exist_ok=True)
67
+ out.write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
66
68
  log(f"wrote full detail to {path}")
@@ -154,7 +154,9 @@ def register(subparsers):
154
154
  g.add_argument("--latest", action="store_true",
155
155
  help="Auto-select the newest AgentHistory (default)")
156
156
  p.add_argument("--cases", default=None, help="Comma-separated case UUIDs (default: all)")
157
- p.add_argument("--evaluators", required=True, help="Comma-separated evaluator UUIDs")
157
+ g = p.add_mutually_exclusive_group()
158
+ g.add_argument("--evaluator", default=None, help="Evaluator UUID; common path for running many cases with one tester")
159
+ g.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
158
160
  p.add_argument("--poll-timeout", type=int, default=POLL_TIMEOUT)
159
161
  p.add_argument("--full", action="store_true",
160
162
  help="Use longer previews in stdout. Raw outputs/tool calls still require --out.")
@@ -195,10 +197,12 @@ def register(subparsers):
195
197
  p.set_defaults(func=run_cases_apply)
196
198
 
197
199
  # codeer eval rubrics
198
- p = sub.add_parser("rubrics", help="Read per-(case, evaluator) rubrics")
200
+ p = sub.add_parser("rubrics", help="Read assigned per-(case, evaluator) rubrics")
199
201
  p.add_argument("--agent", required=True)
200
202
  p.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
201
203
  p.add_argument("--cases", default=None, help="Comma-separated case UUIDs")
204
+ p.add_argument("--all-pairs", action="store_true",
205
+ help="With omitted --evaluators, scan every workspace evaluator instead of assigned pairs only.")
202
206
  p.add_argument("--full", action="store_true",
203
207
  help="Print complete rubric text. Default prints matrix summaries/previews.")
204
208
  p.add_argument("--out", default=None,
@@ -599,6 +603,114 @@ def run_evaluator_update(args, client) -> int:
599
603
  # eval run
600
604
  # ---------------------------------------------------------------------------
601
605
 
606
+ def _assigned_evaluators_by_case(info_rows: list[dict]) -> dict[str, dict[str, dict]]:
607
+ out: dict[str, dict[str, dict]] = {}
608
+ for row in info_rows:
609
+ case_id = row.get("case_id")
610
+ if not case_id:
611
+ continue
612
+ out[str(case_id)] = {
613
+ str(info.get("evaluator_id")): info
614
+ for info in (row.get("evaluators") or [])
615
+ if info.get("evaluator_id")
616
+ }
617
+ return out
618
+
619
+
620
+ def _planned_eval_pairs(
621
+ *,
622
+ case_ids: list[str],
623
+ assigned_by_case: dict[str, dict[str, dict]],
624
+ requested_evaluator_ids: list[str] | None,
625
+ ) -> tuple[list[dict[str, str]], list[dict[str, str]]]:
626
+ pairs: list[dict[str, str]] = []
627
+ skipped: list[dict[str, str]] = []
628
+
629
+ if requested_evaluator_ids:
630
+ for case_id in case_ids:
631
+ assigned = assigned_by_case.get(case_id, {})
632
+ for evaluator_id in requested_evaluator_ids:
633
+ if evaluator_id in assigned:
634
+ pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
635
+ else:
636
+ skipped.append({
637
+ "case_id": case_id,
638
+ "evaluator_id": evaluator_id,
639
+ "reason": "not_assigned",
640
+ })
641
+ return pairs, skipped
642
+
643
+ for case_id in case_ids:
644
+ for evaluator_id in assigned_by_case.get(case_id, {}):
645
+ pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
646
+ return pairs, skipped
647
+
648
+
649
+ def _group_case_ids_by_evaluator(pairs: list[dict[str, str]]) -> dict[str, list[str]]:
650
+ grouped: dict[str, list[str]] = defaultdict(list)
651
+ for pair in pairs:
652
+ grouped[pair["evaluator_id"]].append(pair["case_id"])
653
+ return dict(grouped)
654
+
655
+
656
+ def _pairs_from_rubric_batches(
657
+ client,
658
+ *,
659
+ case_ids: list[str],
660
+ evaluator_ids: list[str],
661
+ ) -> list[dict[str, str]]:
662
+ pairs: list[dict[str, str]] = []
663
+ for evaluator_id in evaluator_ids:
664
+ for row in eval_mod.get_rubrics_batch(client, case_ids=case_ids, evaluator_id=evaluator_id):
665
+ if row.get("rubric"):
666
+ case_id = row.get("case_id") or row.get("evaluation_case_id")
667
+ if case_id:
668
+ pairs.append({"case_id": str(case_id), "evaluator_id": evaluator_id})
669
+ return pairs
670
+
671
+
672
+ def _pair_key(pair: dict[str, str]) -> tuple[str, str]:
673
+ return pair["case_id"], pair["evaluator_id"]
674
+
675
+
676
+ def _skipped_pairs_from_trigger_response(response: Any) -> list[dict[str, str]]:
677
+ if not isinstance(response, dict):
678
+ return []
679
+ payload = response.get("data") if isinstance(response.get("data"), dict) else response
680
+ skipped = payload.get("skipped_pairs") if isinstance(payload, dict) else None
681
+ if not isinstance(skipped, list):
682
+ return []
683
+
684
+ out: list[dict[str, str]] = []
685
+ for row in skipped:
686
+ if not isinstance(row, dict):
687
+ continue
688
+ case_id = row.get("case_id")
689
+ evaluator_id = row.get("evaluator_id")
690
+ if not case_id or not evaluator_id:
691
+ continue
692
+ out.append({
693
+ "case_id": str(case_id),
694
+ "evaluator_id": str(evaluator_id),
695
+ "reason": str(row.get("reason") or "skipped"),
696
+ })
697
+ return out
698
+
699
+
700
+ def _remove_non_runnable_skipped_pairs(
701
+ pairs: list[dict[str, str]],
702
+ skipped_pairs: list[dict[str, str]],
703
+ ) -> list[dict[str, str]]:
704
+ non_runnable = {
705
+ _pair_key(pair)
706
+ for pair in skipped_pairs
707
+ if pair.get("reason") == "not_assigned"
708
+ }
709
+ if not non_runnable:
710
+ return pairs
711
+ return [pair for pair in pairs if _pair_key(pair) not in non_runnable]
712
+
713
+
602
714
  def run_run(args, client) -> int:
603
715
  workspace_id, _ = client.resolve_scope()
604
716
  if args.latest or not args.history:
@@ -625,41 +737,113 @@ def run_run(args, client) -> int:
625
737
  log("error: no cases to run")
626
738
  return 2
627
739
 
628
- evaluator_ids = _ids(args.evaluators) or []
629
- if not evaluator_ids:
630
- log("error: --evaluators is required")
740
+ evaluator_ids = [args.evaluator] if args.evaluator else (_ids(args.evaluators) or [])
741
+ requested_evaluator_ids = evaluator_ids or None
742
+
743
+ skipped_unassigned: list[dict[str, str]] = []
744
+ if requested_evaluator_ids:
745
+ pairs = [
746
+ {"case_id": case_id, "evaluator_id": evaluator_id}
747
+ for evaluator_id in requested_evaluator_ids
748
+ for case_id in case_ids
749
+ ]
750
+ else:
751
+ evaluator_ids = [e["id"] for e in eval_mod.list_evaluators(client, workspace_id)]
752
+ pairs = _pairs_from_rubric_batches(
753
+ client,
754
+ case_ids=case_ids,
755
+ evaluator_ids=evaluator_ids,
756
+ )
757
+ if not pairs:
758
+ log("error: no case/evaluator pairs to run")
759
+ print_json({
760
+ "agent_id": args.agent,
761
+ "history_id": args.history,
762
+ "requested_case_count": len(case_ids),
763
+ "requested_evaluator_count": len(evaluator_ids),
764
+ "triggered_pair_count": 0,
765
+ "skipped_unassigned": skipped_unassigned,
766
+ })
631
767
  return 2
768
+
769
+ evaluator_ids = _dedupe_preserve_order([pair["evaluator_id"] for pair in pairs])
632
770
  evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
633
771
 
634
772
  case_label_by_id = {c["id"]: truncate(c.get("input") or "", 60) for c in case_objs}
635
773
  evaluator_name_by_id = {e["id"]: e.get("name", e["id"]) for e in evaluators}
636
774
 
637
- log(f"triggering: {len(case_ids)} cases x {len(evaluator_ids)} evaluators on history {args.history}")
638
- eval_mod.trigger(client, case_ids=case_ids, evaluator_ids=evaluator_ids,
639
- agent_history_id=args.history)
775
+ requested_pairs = list(pairs)
776
+ requested_pair_count = len(requested_pairs)
777
+ log(f"triggering: {requested_pair_count} case/evaluator pairs on history {args.history}")
778
+ trigger_response: list[dict[str, Any]] = []
779
+ skipped_pairs: list[dict[str, str]] = []
780
+ for ev_id, ev_case_ids in _group_case_ids_by_evaluator(pairs).items():
781
+ response = eval_mod.trigger(
782
+ client,
783
+ case_ids=ev_case_ids,
784
+ evaluator_ids=[ev_id],
785
+ agent_history_id=args.history,
786
+ )
787
+ response_skipped = _skipped_pairs_from_trigger_response(response)
788
+ skipped_pairs.extend(response_skipped)
789
+ trigger_response.append({
790
+ "evaluator_id": ev_id,
791
+ "case_ids": ev_case_ids,
792
+ "response": response,
793
+ "skipped_pairs": response_skipped,
794
+ })
795
+
796
+ pairs = _remove_non_runnable_skipped_pairs(pairs, skipped_pairs)
797
+ skipped_unassigned = [pair for pair in skipped_pairs if pair.get("reason") == "not_assigned"]
798
+ if skipped_unassigned:
799
+ log(f"skipping {len(skipped_unassigned)} not-assigned pairs from polling")
800
+ if not pairs:
801
+ log("error: no runnable case/evaluator pairs after trigger response")
802
+ print_json({
803
+ "agent_id": args.agent,
804
+ "history_id": args.history,
805
+ "requested_case_count": len(case_ids),
806
+ "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
807
+ "requested_pair_count": requested_pair_count,
808
+ "triggered_pair_count": 0,
809
+ "skipped_pair_count": len(skipped_pairs),
810
+ "skipped_unassigned_count": len(skipped_unassigned),
811
+ "trigger_response": trigger_response,
812
+ "skipped_pairs": skipped_pairs,
813
+ "skipped_unassigned": skipped_unassigned,
814
+ })
815
+ return 2
640
816
 
641
817
  deadline = time.time() + args.poll_timeout
642
818
  results_by_eval: dict[str, list[dict]] = {}
819
+ case_ids_by_evaluator = _group_case_ids_by_evaluator(pairs)
820
+ target_pair_keys = {(pair["case_id"], pair["evaluator_id"]) for pair in pairs}
643
821
  while time.time() < deadline:
644
822
  results_by_eval = {}
645
- done = 0
646
- total = len(case_ids) * len(evaluator_ids)
647
- for ev_id in evaluator_ids:
823
+ done_pairs: set[tuple[str, str]] = set()
824
+ total = len(pairs)
825
+ for ev_id, ev_case_ids in case_ids_by_evaluator.items():
648
826
  rows = eval_mod.get_results(
649
- client, case_ids=case_ids, evaluator_id=ev_id,
827
+ client, case_ids=ev_case_ids, evaluator_id=ev_id,
650
828
  agent_history_id=args.history, workspace_id=workspace_id,
651
829
  include_output=True, include_reasoning_steps=True,
652
830
  )
653
831
  results_by_eval[ev_id] = rows
654
- done += sum(1 for r in rows if r.get("score") is not None)
655
- log(f" progress: {done}/{total}")
656
- if done >= total:
832
+ for row in rows:
833
+ key = (row.get("case_id") or row.get("evaluation_case_id"), ev_id)
834
+ if key in target_pair_keys and row.get("score") is not None:
835
+ done_pairs.add(key)
836
+ log(f" progress: {len(done_pairs)}/{total}")
837
+ if len(done_pairs) >= total:
657
838
  break
658
839
  time.sleep(POLL_INTERVAL)
659
840
 
660
841
  flat: list[dict] = []
661
842
  for ev_id, rows in results_by_eval.items():
662
843
  for r in rows:
844
+ row_case_id = r.get("case_id") or r.get("evaluation_case_id")
845
+ if (row_case_id, ev_id) not in target_pair_keys:
846
+ continue
663
847
  result_summary = parse_eval_result(r)
664
848
  tool_calls = parse_eval_tool_calls(r)
665
849
  total_tool_duration_ms = sum(
@@ -684,7 +868,15 @@ def run_run(args, client) -> int:
684
868
  "raw_result": r,
685
869
  })
686
870
 
687
- all_perfect = all((r.get("score") or 0.0) >= 1.0 for r in flat) if flat else False
871
+ scored_pair_keys = {
872
+ (r.get("case_id"), r.get("evaluator_id"))
873
+ for r in flat
874
+ if r.get("score") is not None
875
+ }
876
+ all_perfect = (
877
+ len(scored_pair_keys) == len(target_pair_keys)
878
+ and all((r.get("score") or 0.0) >= 1.0 for r in flat)
879
+ )
688
880
  log("\n" + "=" * 80)
689
881
  log(f"RESULTS agent={args.agent} history={args.history}")
690
882
  log("=" * 80)
@@ -726,15 +918,30 @@ def run_run(args, client) -> int:
726
918
  out = {
727
919
  "agent_id": args.agent,
728
920
  "history_id": args.history,
921
+ "requested_case_count": len(case_ids),
922
+ "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
923
+ "requested_pair_count": requested_pair_count,
924
+ "triggered_pair_count": len(pairs),
925
+ "scored_pair_count": len(scored_pair_keys),
926
+ "skipped_pair_count": len(skipped_pairs),
927
+ "skipped_unassigned_count": len(skipped_unassigned),
729
928
  "all_perfect": all_perfect,
730
929
  "result_count": len(result_summaries),
731
930
  "non_perfect_count": len(non_perfect),
732
931
  "wrote_full_detail": bool(args.out),
932
+ "trigger_response": trigger_response,
933
+ "skipped_pairs": skipped_pairs,
934
+ "skipped_unassigned": skipped_unassigned,
733
935
  "results": result_summaries,
734
936
  }
735
937
  full_out = {
736
938
  "agent_id": args.agent,
737
939
  "history_id": args.history,
940
+ "requested_pairs": requested_pairs,
941
+ "triggered_pairs": pairs,
942
+ "trigger_response": trigger_response,
943
+ "skipped_pairs": skipped_pairs,
944
+ "skipped_unassigned": skipped_unassigned,
738
945
  "all_perfect": all_perfect,
739
946
  "results": flat,
740
947
  }
@@ -1365,17 +1572,37 @@ def run_rubrics(args, client) -> int:
1365
1572
  else:
1366
1573
  evaluators = eval_mod.list_evaluators(client, workspace_id)
1367
1574
  evaluator_ids = [e["id"] for e in evaluators]
1368
- evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
1369
- if not evaluator_ids:
1370
- log("error: no evaluators in workspace")
1371
- return 2
1372
-
1373
- log(f"reading {len(case_ids)} cases x {len(evaluator_ids)} evaluators...")
1374
1575
 
1375
1576
  rubrics = eval_mod.get_case_rubrics(
1376
1577
  client, agent_id=args.agent, workspace_id=workspace_id,
1377
1578
  evaluator_ids=evaluator_ids, case_ids=case_ids,
1378
1579
  )
1580
+ assigned_by_case = {
1581
+ cid: {
1582
+ ev_id: {"evaluator_id": ev_id, "rubric": rubric_text}
1583
+ for ev_id, rubric_text in (rubrics.get(cid) or {}).items()
1584
+ if rubric_text
1585
+ }
1586
+ for cid in case_ids
1587
+ }
1588
+
1589
+ if args.evaluators or args.all_pairs:
1590
+ pass
1591
+ else:
1592
+ evaluator_ids = _dedupe_preserve_order([
1593
+ evaluator_id
1594
+ for case_id in case_ids
1595
+ for evaluator_id in assigned_by_case.get(case_id, {})
1596
+ ])
1597
+ evaluator_id_set = set(evaluator_ids)
1598
+ evaluators = [e for e in evaluators if e["id"] in evaluator_id_set]
1599
+ evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
1600
+ if not evaluator_ids:
1601
+ log("error: no evaluators with configured rubrics for these cases")
1602
+ return 2
1603
+
1604
+ mode = "all requested pairs" if args.evaluators or args.all_pairs else "pairs with configured rubrics"
1605
+ log(f"reading {mode}: {len(case_ids)} cases, {len(evaluator_ids)} evaluators...")
1379
1606
 
1380
1607
  if args.full:
1381
1608
  for cid in case_ids:
@@ -1383,8 +1610,14 @@ def run_rubrics(args, client) -> int:
1383
1610
  log(f"CASE {cid}")
1384
1611
  log(f" input: {truncate(case_input.get(cid, ''), 120)}")
1385
1612
  for ev_id in evaluator_ids:
1613
+ is_assigned = ev_id in assigned_by_case.get(cid, {})
1614
+ if not is_assigned and not (args.evaluators or args.all_pairs):
1615
+ continue
1386
1616
  ev_name = evaluator_name.get(ev_id, ev_id)
1387
1617
  rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
1618
+ if not is_assigned:
1619
+ log(f" [{ev_name}] (not assigned)")
1620
+ continue
1388
1621
  if not rubric_text:
1389
1622
  log(f" [{ev_name}] (rubric not set)")
1390
1623
  else:
@@ -1396,9 +1629,13 @@ def run_rubrics(args, client) -> int:
1396
1629
  for cid in case_ids:
1397
1630
  rubrics_summary = {}
1398
1631
  for ev_id in evaluator_ids:
1632
+ is_assigned = ev_id in assigned_by_case.get(cid, {})
1633
+ if not is_assigned and not (args.evaluators or args.all_pairs):
1634
+ continue
1399
1635
  rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
1400
1636
  rubrics_summary[ev_id] = {
1401
1637
  "evaluator_name": evaluator_name.get(ev_id, ev_id),
1638
+ "is_assigned": is_assigned,
1402
1639
  "is_set": bool(rubric_text),
1403
1640
  "chars": len(rubric_text),
1404
1641
  "preview": truncate(rubric_text, 240),
@@ -1412,6 +1649,7 @@ def run_rubrics(args, client) -> int:
1412
1649
  out = {
1413
1650
  "agent_id": args.agent,
1414
1651
  "workspace_id": workspace_id,
1652
+ "mode": mode,
1415
1653
  "evaluators": [{"id": e["id"], "name": e.get("name")} for e in evaluators],
1416
1654
  "cases": [
1417
1655
  {
@@ -1420,7 +1658,9 @@ def run_rubrics(args, client) -> int:
1420
1658
  "rubrics_by_evaluator": {
1421
1659
  ev_id: (rubrics.get(cid) or {}).get(ev_id)
1422
1660
  for ev_id in evaluator_ids
1661
+ if ev_id in assigned_by_case.get(cid, {}) or args.evaluators or args.all_pairs
1423
1662
  },
1663
+ "assigned_evaluator_ids": list(assigned_by_case.get(cid, {})),
1424
1664
  }
1425
1665
  for cid in case_ids
1426
1666
  ],
@@ -1432,6 +1672,7 @@ def run_rubrics(args, client) -> int:
1432
1672
  print_json({
1433
1673
  "agent_id": args.agent,
1434
1674
  "workspace_id": workspace_id,
1675
+ "mode": mode,
1435
1676
  "evaluator_count": len(evaluators),
1436
1677
  "case_count": len(case_ids),
1437
1678
  "wrote_full_detail": bool(args.out),
codeer_cli/eval_.py CHANGED
@@ -23,6 +23,7 @@ def create_case(
23
23
  rubric: Optional[str] = None,
24
24
  attachment_ids: Optional[List[str]] = None,
25
25
  label_ids: Optional[List[str]] = None,
26
+ evaluators: Optional[List[dict[str, Any]]] = None,
26
27
  meta: Optional[dict] = None,
27
28
  note: Optional[str] = None,
28
29
  ) -> dict:
@@ -44,6 +45,8 @@ def create_case(
44
45
  body["attachment_ids"] = attachment_ids
45
46
  if label_ids is not None:
46
47
  body["label_ids"] = label_ids
48
+ if evaluators is not None:
49
+ body["evaluators"] = evaluators
47
50
  if meta:
48
51
  body["meta"] = meta
49
52
  if note is not None:
@@ -51,8 +54,26 @@ def create_case(
51
54
  return client.post("/external/eval/cases", json=body)
52
55
 
53
56
 
57
+ def _unwrap_list_response(value: Any, *keys: str) -> list[dict]:
58
+ """Normalize list endpoints that may return either a bare list or envelope."""
59
+ if isinstance(value, list):
60
+ return value
61
+ if isinstance(value, dict):
62
+ for key in keys:
63
+ rows = value.get(key)
64
+ if isinstance(rows, list):
65
+ return rows
66
+ return []
67
+
68
+
54
69
  def list_cases(client: CodeerClient, agent_id: str) -> list[dict]:
55
- return client.get(f"/external/eval/agents/{agent_id}/cases")
70
+ return _unwrap_list_response(
71
+ client.get(f"/external/eval/agents/{agent_id}/cases"),
72
+ "cases",
73
+ "evaluation_cases",
74
+ "data",
75
+ "items",
76
+ )
56
77
 
57
78
 
58
79
  def get_case(client: CodeerClient, case_id: str) -> dict:
@@ -93,6 +114,31 @@ def delete_case(client: CodeerClient, case_id: str) -> dict:
93
114
  return client.delete(f"/external/eval/cases/{case_id}")
94
115
 
95
116
 
117
+ # --- case/evaluator assignments ------------------------------------------
118
+
119
+ def get_case_evaluator_infos(client: CodeerClient, *, case_ids: List[str]) -> list[dict]:
120
+ """Read assigned evaluator metadata for each case.
121
+
122
+ Returns rows shaped like ``{"case_id": str, "evaluators": [...]}``, where
123
+ each evaluator entry is the assigned ``{"evaluator_id", "rubric"}`` pair.
124
+ """
125
+ return client.post("/eval/case-evaluator-infos:batch", json={"case_ids": case_ids})
126
+
127
+
128
+ def replace_case_evaluator_infos(
129
+ client: CodeerClient,
130
+ *,
131
+ case_id: str,
132
+ evaluators: list[dict[str, Any]],
133
+ ) -> dict:
134
+ """Replace a case's assigned evaluators.
135
+
136
+ This is intentionally separate from rubric upsert: replacing removes
137
+ evaluator assignments that are not included in ``evaluators``.
138
+ """
139
+ return client.put(f"/eval/cases/{case_id}/case-evaluator-infos", json={"evaluators": evaluators})
140
+
141
+
96
142
  # --- case labels -----------------------------------------------------------
97
143
 
98
144
  def list_case_labels(client: CodeerClient, *, workspace_id: str) -> list[dict]:
@@ -202,6 +248,22 @@ def trigger(
202
248
  return client.post("/external/eval/runs", json=body)
203
249
 
204
250
 
251
+ def trigger_pairs(
252
+ client: CodeerClient,
253
+ *,
254
+ case_evaluator_pairs: list[dict[str, str]],
255
+ agent_history_id: str,
256
+ ) -> dict:
257
+ """Kick off evaluation for explicit assigned case/evaluator pairs."""
258
+ return client.post(
259
+ "/eval/trigger",
260
+ json={
261
+ "case_evaluator_pairs": case_evaluator_pairs,
262
+ "agent_history_id": agent_history_id,
263
+ },
264
+ )
265
+
266
+
205
267
  def stop(client: CodeerClient, *, case_id: str, evaluator_id: str) -> Any:
206
268
  return client.post("/external/eval/runs:stop", json={"case_id": case_id, "evaluator_id": evaluator_id})
207
269
 
@@ -451,9 +513,9 @@ def create_case_with_rubrics(
451
513
  is filled in for every evaluator it will be judged by.
452
514
 
453
515
  ``rubrics_by_evaluator`` maps ``evaluator_id → rubric_text``. Each entry
454
- becomes a ``POST /eval/rubric`` call after the case is created. Use
455
- different rubric wording per evaluator when the evaluators judge different
456
- aspects (e.g. Style/Tone vs Content Compliance).
516
+ becomes a case/evaluator assignment on create. Use different rubric wording
517
+ per evaluator when the evaluators judge different aspects (e.g. Style/Tone
518
+ vs Content Compliance).
457
519
  """
458
520
  case = create_case(
459
521
  client,
@@ -462,12 +524,11 @@ def create_case_with_rubrics(
462
524
  expected_output=expected_output,
463
525
  attachment_ids=attachment_ids,
464
526
  label_ids=label_ids,
527
+ evaluators=[
528
+ {"evaluator_id": ev_id, "rubric": rubric}
529
+ for ev_id, rubric in rubrics_by_evaluator.items()
530
+ ],
465
531
  meta=meta,
466
532
  note=note,
467
533
  )
468
- set_rubric_bulk(
469
- client,
470
- evaluation_case_id=case["id"],
471
- rubrics_by_evaluator=rubrics_by_evaluator,
472
- )
473
534
  return case
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codeer-cli
3
- Version: 0.1.6
3
+ Version: 0.1.8
4
4
  Summary: Command line tools for managing Codeer agents over the Codeer API.
5
5
  Project-URL: Homepage, https://www.codeer.ai
6
6
  Author: Codeer.AI
@@ -143,7 +143,7 @@ Use this pattern during agent lifecycle work:
143
143
  ```bash
144
144
  codeer agent list
145
145
  codeer history list --agent <agent-id> --limit 50
146
- codeer eval run --agent <agent-id> --evaluators <evaluator-id> --out .codeer/eval_run.json
146
+ codeer eval run --agent <agent-id> --cases <case-ids> --evaluator <evaluator-id> --out .codeer/eval_run.json
147
147
  ```
148
148
 
149
149
  Flags:
@@ -5,19 +5,19 @@ codeer_cli/chats.py,sha256=YVrZJhoa-d67o6tzX6riGXsbA-ehyhOxrZ8zRCcJNro,2675
5
5
  codeer_cli/cli.py,sha256=g-WR2D5MkaUdc13ZrpRCavXD1940CHE9eBELC034tic,4443
6
6
  codeer_cli/client.py,sha256=LpHVqf1IYNg1wFfIHnO9q4xg2h3IiGOitzCnvwB-Bcw,9809
7
7
  codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
8
- codeer_cli/eval_.py,sha256=z9RXFiYXOEHPKIh31nbooNe_WmlJ-L63GWBDw3CJhmM,15614
8
+ codeer_cli/eval_.py,sha256=4borDzuU7DOqWi1XidfaELmp7eYjV2FH90uKXgGIDuM,17541
9
9
  codeer_cli/histories.py,sha256=tk28git_peX4x703CIDU8u72JtlGaytyrtlHfxlK-7A,5979
10
10
  codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
11
11
  codeer_cli/parse.py,sha256=qrjZn0MUTjGfucp4cwxy8Pt7WS-0x15kK5F7kWTY8Ps,21818
12
12
  codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
- codeer_cli/commands/_util.py,sha256=VOB_HMWYzHFNY1ElLOED1HB6fpFpsniqH6Yx3VUlMrY,1644
13
+ codeer_cli/commands/_util.py,sha256=X9-9cYgnBYz94hVG4Pe9SsI97DHNv3HE5_7GlPMn93I,1708
14
14
  codeer_cli/commands/agent.py,sha256=amvfVVrbPOkbYKCGvA6EJB30C-aY6WSdfR7u7FklXbs,14793
15
15
  codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
16
- codeer_cli/commands/eval_cmd.py,sha256=3Fi_dwpoJ2hceitACunPtR2kVZ84WT2r_CyL-MaJaMw,62902
16
+ codeer_cli/commands/eval_cmd.py,sha256=V_6OptwWaQ8QLtzJeBqEtlHLUpYXWce1UneeMGPog14,72456
17
17
  codeer_cli/commands/history.py,sha256=Jv7t0GhSZcbZ8OuIXZT34CixXt7ECEVP3nZ-WW_Ya9E,12026
18
18
  codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
19
19
  codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
20
- codeer_cli-0.1.6.dist-info/METADATA,sha256=HDNq9yUPii8TA-KRxu3Mh0sScklObu3pl9OvUthgsdg,5981
21
- codeer_cli-0.1.6.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
- codeer_cli-0.1.6.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
- codeer_cli-0.1.6.dist-info/RECORD,,
20
+ codeer_cli-0.1.8.dist-info/METADATA,sha256=dj78KQR8zotgqs8U_VaO5hmIF3ib8XfdGYLDyG5kLOo,5999
21
+ codeer_cli-0.1.8.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
+ codeer_cli-0.1.8.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
+ codeer_cli-0.1.8.dist-info/RECORD,,