codeer-cli 0.1.6__py3-none-any.whl → 0.1.8__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codeer_cli/commands/_util.py +3 -1
- codeer_cli/commands/eval_cmd.py +263 -22
- codeer_cli/eval_.py +70 -9
- {codeer_cli-0.1.6.dist-info → codeer_cli-0.1.8.dist-info}/METADATA +2 -2
- {codeer_cli-0.1.6.dist-info → codeer_cli-0.1.8.dist-info}/RECORD +7 -7
- {codeer_cli-0.1.6.dist-info → codeer_cli-0.1.8.dist-info}/WHEEL +0 -0
- {codeer_cli-0.1.6.dist-info → codeer_cli-0.1.8.dist-info}/entry_points.txt +0 -0
codeer_cli/commands/_util.py
CHANGED
|
@@ -62,5 +62,7 @@ def print_json(value: Any) -> None:
|
|
|
62
62
|
def write_json(path: str | None, value: Any) -> None:
|
|
63
63
|
if not path:
|
|
64
64
|
return
|
|
65
|
-
Path(path)
|
|
65
|
+
out = Path(path)
|
|
66
|
+
out.parent.mkdir(parents=True, exist_ok=True)
|
|
67
|
+
out.write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
|
|
66
68
|
log(f"wrote full detail to {path}")
|
codeer_cli/commands/eval_cmd.py
CHANGED
|
@@ -154,7 +154,9 @@ def register(subparsers):
|
|
|
154
154
|
g.add_argument("--latest", action="store_true",
|
|
155
155
|
help="Auto-select the newest AgentHistory (default)")
|
|
156
156
|
p.add_argument("--cases", default=None, help="Comma-separated case UUIDs (default: all)")
|
|
157
|
-
p.
|
|
157
|
+
g = p.add_mutually_exclusive_group()
|
|
158
|
+
g.add_argument("--evaluator", default=None, help="Evaluator UUID; common path for running many cases with one tester")
|
|
159
|
+
g.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
|
|
158
160
|
p.add_argument("--poll-timeout", type=int, default=POLL_TIMEOUT)
|
|
159
161
|
p.add_argument("--full", action="store_true",
|
|
160
162
|
help="Use longer previews in stdout. Raw outputs/tool calls still require --out.")
|
|
@@ -195,10 +197,12 @@ def register(subparsers):
|
|
|
195
197
|
p.set_defaults(func=run_cases_apply)
|
|
196
198
|
|
|
197
199
|
# codeer eval rubrics
|
|
198
|
-
p = sub.add_parser("rubrics", help="Read per-(case, evaluator) rubrics")
|
|
200
|
+
p = sub.add_parser("rubrics", help="Read assigned per-(case, evaluator) rubrics")
|
|
199
201
|
p.add_argument("--agent", required=True)
|
|
200
202
|
p.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
|
|
201
203
|
p.add_argument("--cases", default=None, help="Comma-separated case UUIDs")
|
|
204
|
+
p.add_argument("--all-pairs", action="store_true",
|
|
205
|
+
help="With omitted --evaluators, scan every workspace evaluator instead of assigned pairs only.")
|
|
202
206
|
p.add_argument("--full", action="store_true",
|
|
203
207
|
help="Print complete rubric text. Default prints matrix summaries/previews.")
|
|
204
208
|
p.add_argument("--out", default=None,
|
|
@@ -599,6 +603,114 @@ def run_evaluator_update(args, client) -> int:
|
|
|
599
603
|
# eval run
|
|
600
604
|
# ---------------------------------------------------------------------------
|
|
601
605
|
|
|
606
|
+
def _assigned_evaluators_by_case(info_rows: list[dict]) -> dict[str, dict[str, dict]]:
|
|
607
|
+
out: dict[str, dict[str, dict]] = {}
|
|
608
|
+
for row in info_rows:
|
|
609
|
+
case_id = row.get("case_id")
|
|
610
|
+
if not case_id:
|
|
611
|
+
continue
|
|
612
|
+
out[str(case_id)] = {
|
|
613
|
+
str(info.get("evaluator_id")): info
|
|
614
|
+
for info in (row.get("evaluators") or [])
|
|
615
|
+
if info.get("evaluator_id")
|
|
616
|
+
}
|
|
617
|
+
return out
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def _planned_eval_pairs(
|
|
621
|
+
*,
|
|
622
|
+
case_ids: list[str],
|
|
623
|
+
assigned_by_case: dict[str, dict[str, dict]],
|
|
624
|
+
requested_evaluator_ids: list[str] | None,
|
|
625
|
+
) -> tuple[list[dict[str, str]], list[dict[str, str]]]:
|
|
626
|
+
pairs: list[dict[str, str]] = []
|
|
627
|
+
skipped: list[dict[str, str]] = []
|
|
628
|
+
|
|
629
|
+
if requested_evaluator_ids:
|
|
630
|
+
for case_id in case_ids:
|
|
631
|
+
assigned = assigned_by_case.get(case_id, {})
|
|
632
|
+
for evaluator_id in requested_evaluator_ids:
|
|
633
|
+
if evaluator_id in assigned:
|
|
634
|
+
pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
|
|
635
|
+
else:
|
|
636
|
+
skipped.append({
|
|
637
|
+
"case_id": case_id,
|
|
638
|
+
"evaluator_id": evaluator_id,
|
|
639
|
+
"reason": "not_assigned",
|
|
640
|
+
})
|
|
641
|
+
return pairs, skipped
|
|
642
|
+
|
|
643
|
+
for case_id in case_ids:
|
|
644
|
+
for evaluator_id in assigned_by_case.get(case_id, {}):
|
|
645
|
+
pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
|
|
646
|
+
return pairs, skipped
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def _group_case_ids_by_evaluator(pairs: list[dict[str, str]]) -> dict[str, list[str]]:
|
|
650
|
+
grouped: dict[str, list[str]] = defaultdict(list)
|
|
651
|
+
for pair in pairs:
|
|
652
|
+
grouped[pair["evaluator_id"]].append(pair["case_id"])
|
|
653
|
+
return dict(grouped)
|
|
654
|
+
|
|
655
|
+
|
|
656
|
+
def _pairs_from_rubric_batches(
|
|
657
|
+
client,
|
|
658
|
+
*,
|
|
659
|
+
case_ids: list[str],
|
|
660
|
+
evaluator_ids: list[str],
|
|
661
|
+
) -> list[dict[str, str]]:
|
|
662
|
+
pairs: list[dict[str, str]] = []
|
|
663
|
+
for evaluator_id in evaluator_ids:
|
|
664
|
+
for row in eval_mod.get_rubrics_batch(client, case_ids=case_ids, evaluator_id=evaluator_id):
|
|
665
|
+
if row.get("rubric"):
|
|
666
|
+
case_id = row.get("case_id") or row.get("evaluation_case_id")
|
|
667
|
+
if case_id:
|
|
668
|
+
pairs.append({"case_id": str(case_id), "evaluator_id": evaluator_id})
|
|
669
|
+
return pairs
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
def _pair_key(pair: dict[str, str]) -> tuple[str, str]:
|
|
673
|
+
return pair["case_id"], pair["evaluator_id"]
|
|
674
|
+
|
|
675
|
+
|
|
676
|
+
def _skipped_pairs_from_trigger_response(response: Any) -> list[dict[str, str]]:
|
|
677
|
+
if not isinstance(response, dict):
|
|
678
|
+
return []
|
|
679
|
+
payload = response.get("data") if isinstance(response.get("data"), dict) else response
|
|
680
|
+
skipped = payload.get("skipped_pairs") if isinstance(payload, dict) else None
|
|
681
|
+
if not isinstance(skipped, list):
|
|
682
|
+
return []
|
|
683
|
+
|
|
684
|
+
out: list[dict[str, str]] = []
|
|
685
|
+
for row in skipped:
|
|
686
|
+
if not isinstance(row, dict):
|
|
687
|
+
continue
|
|
688
|
+
case_id = row.get("case_id")
|
|
689
|
+
evaluator_id = row.get("evaluator_id")
|
|
690
|
+
if not case_id or not evaluator_id:
|
|
691
|
+
continue
|
|
692
|
+
out.append({
|
|
693
|
+
"case_id": str(case_id),
|
|
694
|
+
"evaluator_id": str(evaluator_id),
|
|
695
|
+
"reason": str(row.get("reason") or "skipped"),
|
|
696
|
+
})
|
|
697
|
+
return out
|
|
698
|
+
|
|
699
|
+
|
|
700
|
+
def _remove_non_runnable_skipped_pairs(
|
|
701
|
+
pairs: list[dict[str, str]],
|
|
702
|
+
skipped_pairs: list[dict[str, str]],
|
|
703
|
+
) -> list[dict[str, str]]:
|
|
704
|
+
non_runnable = {
|
|
705
|
+
_pair_key(pair)
|
|
706
|
+
for pair in skipped_pairs
|
|
707
|
+
if pair.get("reason") == "not_assigned"
|
|
708
|
+
}
|
|
709
|
+
if not non_runnable:
|
|
710
|
+
return pairs
|
|
711
|
+
return [pair for pair in pairs if _pair_key(pair) not in non_runnable]
|
|
712
|
+
|
|
713
|
+
|
|
602
714
|
def run_run(args, client) -> int:
|
|
603
715
|
workspace_id, _ = client.resolve_scope()
|
|
604
716
|
if args.latest or not args.history:
|
|
@@ -625,41 +737,113 @@ def run_run(args, client) -> int:
|
|
|
625
737
|
log("error: no cases to run")
|
|
626
738
|
return 2
|
|
627
739
|
|
|
628
|
-
evaluator_ids = _ids(args.evaluators) or []
|
|
629
|
-
|
|
630
|
-
|
|
740
|
+
evaluator_ids = [args.evaluator] if args.evaluator else (_ids(args.evaluators) or [])
|
|
741
|
+
requested_evaluator_ids = evaluator_ids or None
|
|
742
|
+
|
|
743
|
+
skipped_unassigned: list[dict[str, str]] = []
|
|
744
|
+
if requested_evaluator_ids:
|
|
745
|
+
pairs = [
|
|
746
|
+
{"case_id": case_id, "evaluator_id": evaluator_id}
|
|
747
|
+
for evaluator_id in requested_evaluator_ids
|
|
748
|
+
for case_id in case_ids
|
|
749
|
+
]
|
|
750
|
+
else:
|
|
751
|
+
evaluator_ids = [e["id"] for e in eval_mod.list_evaluators(client, workspace_id)]
|
|
752
|
+
pairs = _pairs_from_rubric_batches(
|
|
753
|
+
client,
|
|
754
|
+
case_ids=case_ids,
|
|
755
|
+
evaluator_ids=evaluator_ids,
|
|
756
|
+
)
|
|
757
|
+
if not pairs:
|
|
758
|
+
log("error: no case/evaluator pairs to run")
|
|
759
|
+
print_json({
|
|
760
|
+
"agent_id": args.agent,
|
|
761
|
+
"history_id": args.history,
|
|
762
|
+
"requested_case_count": len(case_ids),
|
|
763
|
+
"requested_evaluator_count": len(evaluator_ids),
|
|
764
|
+
"triggered_pair_count": 0,
|
|
765
|
+
"skipped_unassigned": skipped_unassigned,
|
|
766
|
+
})
|
|
631
767
|
return 2
|
|
768
|
+
|
|
769
|
+
evaluator_ids = _dedupe_preserve_order([pair["evaluator_id"] for pair in pairs])
|
|
632
770
|
evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
|
|
633
771
|
|
|
634
772
|
case_label_by_id = {c["id"]: truncate(c.get("input") or "", 60) for c in case_objs}
|
|
635
773
|
evaluator_name_by_id = {e["id"]: e.get("name", e["id"]) for e in evaluators}
|
|
636
774
|
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
775
|
+
requested_pairs = list(pairs)
|
|
776
|
+
requested_pair_count = len(requested_pairs)
|
|
777
|
+
log(f"triggering: {requested_pair_count} case/evaluator pairs on history {args.history}")
|
|
778
|
+
trigger_response: list[dict[str, Any]] = []
|
|
779
|
+
skipped_pairs: list[dict[str, str]] = []
|
|
780
|
+
for ev_id, ev_case_ids in _group_case_ids_by_evaluator(pairs).items():
|
|
781
|
+
response = eval_mod.trigger(
|
|
782
|
+
client,
|
|
783
|
+
case_ids=ev_case_ids,
|
|
784
|
+
evaluator_ids=[ev_id],
|
|
785
|
+
agent_history_id=args.history,
|
|
786
|
+
)
|
|
787
|
+
response_skipped = _skipped_pairs_from_trigger_response(response)
|
|
788
|
+
skipped_pairs.extend(response_skipped)
|
|
789
|
+
trigger_response.append({
|
|
790
|
+
"evaluator_id": ev_id,
|
|
791
|
+
"case_ids": ev_case_ids,
|
|
792
|
+
"response": response,
|
|
793
|
+
"skipped_pairs": response_skipped,
|
|
794
|
+
})
|
|
795
|
+
|
|
796
|
+
pairs = _remove_non_runnable_skipped_pairs(pairs, skipped_pairs)
|
|
797
|
+
skipped_unassigned = [pair for pair in skipped_pairs if pair.get("reason") == "not_assigned"]
|
|
798
|
+
if skipped_unassigned:
|
|
799
|
+
log(f"skipping {len(skipped_unassigned)} not-assigned pairs from polling")
|
|
800
|
+
if not pairs:
|
|
801
|
+
log("error: no runnable case/evaluator pairs after trigger response")
|
|
802
|
+
print_json({
|
|
803
|
+
"agent_id": args.agent,
|
|
804
|
+
"history_id": args.history,
|
|
805
|
+
"requested_case_count": len(case_ids),
|
|
806
|
+
"requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
|
|
807
|
+
"requested_pair_count": requested_pair_count,
|
|
808
|
+
"triggered_pair_count": 0,
|
|
809
|
+
"skipped_pair_count": len(skipped_pairs),
|
|
810
|
+
"skipped_unassigned_count": len(skipped_unassigned),
|
|
811
|
+
"trigger_response": trigger_response,
|
|
812
|
+
"skipped_pairs": skipped_pairs,
|
|
813
|
+
"skipped_unassigned": skipped_unassigned,
|
|
814
|
+
})
|
|
815
|
+
return 2
|
|
640
816
|
|
|
641
817
|
deadline = time.time() + args.poll_timeout
|
|
642
818
|
results_by_eval: dict[str, list[dict]] = {}
|
|
819
|
+
case_ids_by_evaluator = _group_case_ids_by_evaluator(pairs)
|
|
820
|
+
target_pair_keys = {(pair["case_id"], pair["evaluator_id"]) for pair in pairs}
|
|
643
821
|
while time.time() < deadline:
|
|
644
822
|
results_by_eval = {}
|
|
645
|
-
|
|
646
|
-
total = len(
|
|
647
|
-
for ev_id in
|
|
823
|
+
done_pairs: set[tuple[str, str]] = set()
|
|
824
|
+
total = len(pairs)
|
|
825
|
+
for ev_id, ev_case_ids in case_ids_by_evaluator.items():
|
|
648
826
|
rows = eval_mod.get_results(
|
|
649
|
-
client, case_ids=
|
|
827
|
+
client, case_ids=ev_case_ids, evaluator_id=ev_id,
|
|
650
828
|
agent_history_id=args.history, workspace_id=workspace_id,
|
|
651
829
|
include_output=True, include_reasoning_steps=True,
|
|
652
830
|
)
|
|
653
831
|
results_by_eval[ev_id] = rows
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
832
|
+
for row in rows:
|
|
833
|
+
key = (row.get("case_id") or row.get("evaluation_case_id"), ev_id)
|
|
834
|
+
if key in target_pair_keys and row.get("score") is not None:
|
|
835
|
+
done_pairs.add(key)
|
|
836
|
+
log(f" progress: {len(done_pairs)}/{total}")
|
|
837
|
+
if len(done_pairs) >= total:
|
|
657
838
|
break
|
|
658
839
|
time.sleep(POLL_INTERVAL)
|
|
659
840
|
|
|
660
841
|
flat: list[dict] = []
|
|
661
842
|
for ev_id, rows in results_by_eval.items():
|
|
662
843
|
for r in rows:
|
|
844
|
+
row_case_id = r.get("case_id") or r.get("evaluation_case_id")
|
|
845
|
+
if (row_case_id, ev_id) not in target_pair_keys:
|
|
846
|
+
continue
|
|
663
847
|
result_summary = parse_eval_result(r)
|
|
664
848
|
tool_calls = parse_eval_tool_calls(r)
|
|
665
849
|
total_tool_duration_ms = sum(
|
|
@@ -684,7 +868,15 @@ def run_run(args, client) -> int:
|
|
|
684
868
|
"raw_result": r,
|
|
685
869
|
})
|
|
686
870
|
|
|
687
|
-
|
|
871
|
+
scored_pair_keys = {
|
|
872
|
+
(r.get("case_id"), r.get("evaluator_id"))
|
|
873
|
+
for r in flat
|
|
874
|
+
if r.get("score") is not None
|
|
875
|
+
}
|
|
876
|
+
all_perfect = (
|
|
877
|
+
len(scored_pair_keys) == len(target_pair_keys)
|
|
878
|
+
and all((r.get("score") or 0.0) >= 1.0 for r in flat)
|
|
879
|
+
)
|
|
688
880
|
log("\n" + "=" * 80)
|
|
689
881
|
log(f"RESULTS agent={args.agent} history={args.history}")
|
|
690
882
|
log("=" * 80)
|
|
@@ -726,15 +918,30 @@ def run_run(args, client) -> int:
|
|
|
726
918
|
out = {
|
|
727
919
|
"agent_id": args.agent,
|
|
728
920
|
"history_id": args.history,
|
|
921
|
+
"requested_case_count": len(case_ids),
|
|
922
|
+
"requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
|
|
923
|
+
"requested_pair_count": requested_pair_count,
|
|
924
|
+
"triggered_pair_count": len(pairs),
|
|
925
|
+
"scored_pair_count": len(scored_pair_keys),
|
|
926
|
+
"skipped_pair_count": len(skipped_pairs),
|
|
927
|
+
"skipped_unassigned_count": len(skipped_unassigned),
|
|
729
928
|
"all_perfect": all_perfect,
|
|
730
929
|
"result_count": len(result_summaries),
|
|
731
930
|
"non_perfect_count": len(non_perfect),
|
|
732
931
|
"wrote_full_detail": bool(args.out),
|
|
932
|
+
"trigger_response": trigger_response,
|
|
933
|
+
"skipped_pairs": skipped_pairs,
|
|
934
|
+
"skipped_unassigned": skipped_unassigned,
|
|
733
935
|
"results": result_summaries,
|
|
734
936
|
}
|
|
735
937
|
full_out = {
|
|
736
938
|
"agent_id": args.agent,
|
|
737
939
|
"history_id": args.history,
|
|
940
|
+
"requested_pairs": requested_pairs,
|
|
941
|
+
"triggered_pairs": pairs,
|
|
942
|
+
"trigger_response": trigger_response,
|
|
943
|
+
"skipped_pairs": skipped_pairs,
|
|
944
|
+
"skipped_unassigned": skipped_unassigned,
|
|
738
945
|
"all_perfect": all_perfect,
|
|
739
946
|
"results": flat,
|
|
740
947
|
}
|
|
@@ -1365,17 +1572,37 @@ def run_rubrics(args, client) -> int:
|
|
|
1365
1572
|
else:
|
|
1366
1573
|
evaluators = eval_mod.list_evaluators(client, workspace_id)
|
|
1367
1574
|
evaluator_ids = [e["id"] for e in evaluators]
|
|
1368
|
-
evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
|
|
1369
|
-
if not evaluator_ids:
|
|
1370
|
-
log("error: no evaluators in workspace")
|
|
1371
|
-
return 2
|
|
1372
|
-
|
|
1373
|
-
log(f"reading {len(case_ids)} cases x {len(evaluator_ids)} evaluators...")
|
|
1374
1575
|
|
|
1375
1576
|
rubrics = eval_mod.get_case_rubrics(
|
|
1376
1577
|
client, agent_id=args.agent, workspace_id=workspace_id,
|
|
1377
1578
|
evaluator_ids=evaluator_ids, case_ids=case_ids,
|
|
1378
1579
|
)
|
|
1580
|
+
assigned_by_case = {
|
|
1581
|
+
cid: {
|
|
1582
|
+
ev_id: {"evaluator_id": ev_id, "rubric": rubric_text}
|
|
1583
|
+
for ev_id, rubric_text in (rubrics.get(cid) or {}).items()
|
|
1584
|
+
if rubric_text
|
|
1585
|
+
}
|
|
1586
|
+
for cid in case_ids
|
|
1587
|
+
}
|
|
1588
|
+
|
|
1589
|
+
if args.evaluators or args.all_pairs:
|
|
1590
|
+
pass
|
|
1591
|
+
else:
|
|
1592
|
+
evaluator_ids = _dedupe_preserve_order([
|
|
1593
|
+
evaluator_id
|
|
1594
|
+
for case_id in case_ids
|
|
1595
|
+
for evaluator_id in assigned_by_case.get(case_id, {})
|
|
1596
|
+
])
|
|
1597
|
+
evaluator_id_set = set(evaluator_ids)
|
|
1598
|
+
evaluators = [e for e in evaluators if e["id"] in evaluator_id_set]
|
|
1599
|
+
evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
|
|
1600
|
+
if not evaluator_ids:
|
|
1601
|
+
log("error: no evaluators with configured rubrics for these cases")
|
|
1602
|
+
return 2
|
|
1603
|
+
|
|
1604
|
+
mode = "all requested pairs" if args.evaluators or args.all_pairs else "pairs with configured rubrics"
|
|
1605
|
+
log(f"reading {mode}: {len(case_ids)} cases, {len(evaluator_ids)} evaluators...")
|
|
1379
1606
|
|
|
1380
1607
|
if args.full:
|
|
1381
1608
|
for cid in case_ids:
|
|
@@ -1383,8 +1610,14 @@ def run_rubrics(args, client) -> int:
|
|
|
1383
1610
|
log(f"CASE {cid}")
|
|
1384
1611
|
log(f" input: {truncate(case_input.get(cid, ''), 120)}")
|
|
1385
1612
|
for ev_id in evaluator_ids:
|
|
1613
|
+
is_assigned = ev_id in assigned_by_case.get(cid, {})
|
|
1614
|
+
if not is_assigned and not (args.evaluators or args.all_pairs):
|
|
1615
|
+
continue
|
|
1386
1616
|
ev_name = evaluator_name.get(ev_id, ev_id)
|
|
1387
1617
|
rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
|
|
1618
|
+
if not is_assigned:
|
|
1619
|
+
log(f" [{ev_name}] (not assigned)")
|
|
1620
|
+
continue
|
|
1388
1621
|
if not rubric_text:
|
|
1389
1622
|
log(f" [{ev_name}] (rubric not set)")
|
|
1390
1623
|
else:
|
|
@@ -1396,9 +1629,13 @@ def run_rubrics(args, client) -> int:
|
|
|
1396
1629
|
for cid in case_ids:
|
|
1397
1630
|
rubrics_summary = {}
|
|
1398
1631
|
for ev_id in evaluator_ids:
|
|
1632
|
+
is_assigned = ev_id in assigned_by_case.get(cid, {})
|
|
1633
|
+
if not is_assigned and not (args.evaluators or args.all_pairs):
|
|
1634
|
+
continue
|
|
1399
1635
|
rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
|
|
1400
1636
|
rubrics_summary[ev_id] = {
|
|
1401
1637
|
"evaluator_name": evaluator_name.get(ev_id, ev_id),
|
|
1638
|
+
"is_assigned": is_assigned,
|
|
1402
1639
|
"is_set": bool(rubric_text),
|
|
1403
1640
|
"chars": len(rubric_text),
|
|
1404
1641
|
"preview": truncate(rubric_text, 240),
|
|
@@ -1412,6 +1649,7 @@ def run_rubrics(args, client) -> int:
|
|
|
1412
1649
|
out = {
|
|
1413
1650
|
"agent_id": args.agent,
|
|
1414
1651
|
"workspace_id": workspace_id,
|
|
1652
|
+
"mode": mode,
|
|
1415
1653
|
"evaluators": [{"id": e["id"], "name": e.get("name")} for e in evaluators],
|
|
1416
1654
|
"cases": [
|
|
1417
1655
|
{
|
|
@@ -1420,7 +1658,9 @@ def run_rubrics(args, client) -> int:
|
|
|
1420
1658
|
"rubrics_by_evaluator": {
|
|
1421
1659
|
ev_id: (rubrics.get(cid) or {}).get(ev_id)
|
|
1422
1660
|
for ev_id in evaluator_ids
|
|
1661
|
+
if ev_id in assigned_by_case.get(cid, {}) or args.evaluators or args.all_pairs
|
|
1423
1662
|
},
|
|
1663
|
+
"assigned_evaluator_ids": list(assigned_by_case.get(cid, {})),
|
|
1424
1664
|
}
|
|
1425
1665
|
for cid in case_ids
|
|
1426
1666
|
],
|
|
@@ -1432,6 +1672,7 @@ def run_rubrics(args, client) -> int:
|
|
|
1432
1672
|
print_json({
|
|
1433
1673
|
"agent_id": args.agent,
|
|
1434
1674
|
"workspace_id": workspace_id,
|
|
1675
|
+
"mode": mode,
|
|
1435
1676
|
"evaluator_count": len(evaluators),
|
|
1436
1677
|
"case_count": len(case_ids),
|
|
1437
1678
|
"wrote_full_detail": bool(args.out),
|
codeer_cli/eval_.py
CHANGED
|
@@ -23,6 +23,7 @@ def create_case(
|
|
|
23
23
|
rubric: Optional[str] = None,
|
|
24
24
|
attachment_ids: Optional[List[str]] = None,
|
|
25
25
|
label_ids: Optional[List[str]] = None,
|
|
26
|
+
evaluators: Optional[List[dict[str, Any]]] = None,
|
|
26
27
|
meta: Optional[dict] = None,
|
|
27
28
|
note: Optional[str] = None,
|
|
28
29
|
) -> dict:
|
|
@@ -44,6 +45,8 @@ def create_case(
|
|
|
44
45
|
body["attachment_ids"] = attachment_ids
|
|
45
46
|
if label_ids is not None:
|
|
46
47
|
body["label_ids"] = label_ids
|
|
48
|
+
if evaluators is not None:
|
|
49
|
+
body["evaluators"] = evaluators
|
|
47
50
|
if meta:
|
|
48
51
|
body["meta"] = meta
|
|
49
52
|
if note is not None:
|
|
@@ -51,8 +54,26 @@ def create_case(
|
|
|
51
54
|
return client.post("/external/eval/cases", json=body)
|
|
52
55
|
|
|
53
56
|
|
|
57
|
+
def _unwrap_list_response(value: Any, *keys: str) -> list[dict]:
|
|
58
|
+
"""Normalize list endpoints that may return either a bare list or envelope."""
|
|
59
|
+
if isinstance(value, list):
|
|
60
|
+
return value
|
|
61
|
+
if isinstance(value, dict):
|
|
62
|
+
for key in keys:
|
|
63
|
+
rows = value.get(key)
|
|
64
|
+
if isinstance(rows, list):
|
|
65
|
+
return rows
|
|
66
|
+
return []
|
|
67
|
+
|
|
68
|
+
|
|
54
69
|
def list_cases(client: CodeerClient, agent_id: str) -> list[dict]:
|
|
55
|
-
return
|
|
70
|
+
return _unwrap_list_response(
|
|
71
|
+
client.get(f"/external/eval/agents/{agent_id}/cases"),
|
|
72
|
+
"cases",
|
|
73
|
+
"evaluation_cases",
|
|
74
|
+
"data",
|
|
75
|
+
"items",
|
|
76
|
+
)
|
|
56
77
|
|
|
57
78
|
|
|
58
79
|
def get_case(client: CodeerClient, case_id: str) -> dict:
|
|
@@ -93,6 +114,31 @@ def delete_case(client: CodeerClient, case_id: str) -> dict:
|
|
|
93
114
|
return client.delete(f"/external/eval/cases/{case_id}")
|
|
94
115
|
|
|
95
116
|
|
|
117
|
+
# --- case/evaluator assignments ------------------------------------------
|
|
118
|
+
|
|
119
|
+
def get_case_evaluator_infos(client: CodeerClient, *, case_ids: List[str]) -> list[dict]:
|
|
120
|
+
"""Read assigned evaluator metadata for each case.
|
|
121
|
+
|
|
122
|
+
Returns rows shaped like ``{"case_id": str, "evaluators": [...]}``, where
|
|
123
|
+
each evaluator entry is the assigned ``{"evaluator_id", "rubric"}`` pair.
|
|
124
|
+
"""
|
|
125
|
+
return client.post("/eval/case-evaluator-infos:batch", json={"case_ids": case_ids})
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def replace_case_evaluator_infos(
|
|
129
|
+
client: CodeerClient,
|
|
130
|
+
*,
|
|
131
|
+
case_id: str,
|
|
132
|
+
evaluators: list[dict[str, Any]],
|
|
133
|
+
) -> dict:
|
|
134
|
+
"""Replace a case's assigned evaluators.
|
|
135
|
+
|
|
136
|
+
This is intentionally separate from rubric upsert: replacing removes
|
|
137
|
+
evaluator assignments that are not included in ``evaluators``.
|
|
138
|
+
"""
|
|
139
|
+
return client.put(f"/eval/cases/{case_id}/case-evaluator-infos", json={"evaluators": evaluators})
|
|
140
|
+
|
|
141
|
+
|
|
96
142
|
# --- case labels -----------------------------------------------------------
|
|
97
143
|
|
|
98
144
|
def list_case_labels(client: CodeerClient, *, workspace_id: str) -> list[dict]:
|
|
@@ -202,6 +248,22 @@ def trigger(
|
|
|
202
248
|
return client.post("/external/eval/runs", json=body)
|
|
203
249
|
|
|
204
250
|
|
|
251
|
+
def trigger_pairs(
|
|
252
|
+
client: CodeerClient,
|
|
253
|
+
*,
|
|
254
|
+
case_evaluator_pairs: list[dict[str, str]],
|
|
255
|
+
agent_history_id: str,
|
|
256
|
+
) -> dict:
|
|
257
|
+
"""Kick off evaluation for explicit assigned case/evaluator pairs."""
|
|
258
|
+
return client.post(
|
|
259
|
+
"/eval/trigger",
|
|
260
|
+
json={
|
|
261
|
+
"case_evaluator_pairs": case_evaluator_pairs,
|
|
262
|
+
"agent_history_id": agent_history_id,
|
|
263
|
+
},
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
|
|
205
267
|
def stop(client: CodeerClient, *, case_id: str, evaluator_id: str) -> Any:
|
|
206
268
|
return client.post("/external/eval/runs:stop", json={"case_id": case_id, "evaluator_id": evaluator_id})
|
|
207
269
|
|
|
@@ -451,9 +513,9 @@ def create_case_with_rubrics(
|
|
|
451
513
|
is filled in for every evaluator it will be judged by.
|
|
452
514
|
|
|
453
515
|
``rubrics_by_evaluator`` maps ``evaluator_id → rubric_text``. Each entry
|
|
454
|
-
becomes a
|
|
455
|
-
|
|
456
|
-
|
|
516
|
+
becomes a case/evaluator assignment on create. Use different rubric wording
|
|
517
|
+
per evaluator when the evaluators judge different aspects (e.g. Style/Tone
|
|
518
|
+
vs Content Compliance).
|
|
457
519
|
"""
|
|
458
520
|
case = create_case(
|
|
459
521
|
client,
|
|
@@ -462,12 +524,11 @@ def create_case_with_rubrics(
|
|
|
462
524
|
expected_output=expected_output,
|
|
463
525
|
attachment_ids=attachment_ids,
|
|
464
526
|
label_ids=label_ids,
|
|
527
|
+
evaluators=[
|
|
528
|
+
{"evaluator_id": ev_id, "rubric": rubric}
|
|
529
|
+
for ev_id, rubric in rubrics_by_evaluator.items()
|
|
530
|
+
],
|
|
465
531
|
meta=meta,
|
|
466
532
|
note=note,
|
|
467
533
|
)
|
|
468
|
-
set_rubric_bulk(
|
|
469
|
-
client,
|
|
470
|
-
evaluation_case_id=case["id"],
|
|
471
|
-
rubrics_by_evaluator=rubrics_by_evaluator,
|
|
472
|
-
)
|
|
473
534
|
return case
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: codeer-cli
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.8
|
|
4
4
|
Summary: Command line tools for managing Codeer agents over the Codeer API.
|
|
5
5
|
Project-URL: Homepage, https://www.codeer.ai
|
|
6
6
|
Author: Codeer.AI
|
|
@@ -143,7 +143,7 @@ Use this pattern during agent lifecycle work:
|
|
|
143
143
|
```bash
|
|
144
144
|
codeer agent list
|
|
145
145
|
codeer history list --agent <agent-id> --limit 50
|
|
146
|
-
codeer eval run --agent <agent-id> --
|
|
146
|
+
codeer eval run --agent <agent-id> --cases <case-ids> --evaluator <evaluator-id> --out .codeer/eval_run.json
|
|
147
147
|
```
|
|
148
148
|
|
|
149
149
|
Flags:
|
|
@@ -5,19 +5,19 @@ codeer_cli/chats.py,sha256=YVrZJhoa-d67o6tzX6riGXsbA-ehyhOxrZ8zRCcJNro,2675
|
|
|
5
5
|
codeer_cli/cli.py,sha256=g-WR2D5MkaUdc13ZrpRCavXD1940CHE9eBELC034tic,4443
|
|
6
6
|
codeer_cli/client.py,sha256=LpHVqf1IYNg1wFfIHnO9q4xg2h3IiGOitzCnvwB-Bcw,9809
|
|
7
7
|
codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
|
|
8
|
-
codeer_cli/eval_.py,sha256=
|
|
8
|
+
codeer_cli/eval_.py,sha256=4borDzuU7DOqWi1XidfaELmp7eYjV2FH90uKXgGIDuM,17541
|
|
9
9
|
codeer_cli/histories.py,sha256=tk28git_peX4x703CIDU8u72JtlGaytyrtlHfxlK-7A,5979
|
|
10
10
|
codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
|
|
11
11
|
codeer_cli/parse.py,sha256=qrjZn0MUTjGfucp4cwxy8Pt7WS-0x15kK5F7kWTY8Ps,21818
|
|
12
12
|
codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
|
-
codeer_cli/commands/_util.py,sha256=
|
|
13
|
+
codeer_cli/commands/_util.py,sha256=X9-9cYgnBYz94hVG4Pe9SsI97DHNv3HE5_7GlPMn93I,1708
|
|
14
14
|
codeer_cli/commands/agent.py,sha256=amvfVVrbPOkbYKCGvA6EJB30C-aY6WSdfR7u7FklXbs,14793
|
|
15
15
|
codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
|
|
16
|
-
codeer_cli/commands/eval_cmd.py,sha256=
|
|
16
|
+
codeer_cli/commands/eval_cmd.py,sha256=V_6OptwWaQ8QLtzJeBqEtlHLUpYXWce1UneeMGPog14,72456
|
|
17
17
|
codeer_cli/commands/history.py,sha256=Jv7t0GhSZcbZ8OuIXZT34CixXt7ECEVP3nZ-WW_Ya9E,12026
|
|
18
18
|
codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
|
|
19
19
|
codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
|
|
20
|
-
codeer_cli-0.1.
|
|
21
|
-
codeer_cli-0.1.
|
|
22
|
-
codeer_cli-0.1.
|
|
23
|
-
codeer_cli-0.1.
|
|
20
|
+
codeer_cli-0.1.8.dist-info/METADATA,sha256=dj78KQR8zotgqs8U_VaO5hmIF3ib8XfdGYLDyG5kLOo,5999
|
|
21
|
+
codeer_cli-0.1.8.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
22
|
+
codeer_cli-0.1.8.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
|
|
23
|
+
codeer_cli-0.1.8.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|