codeer-cli 0.1.7__py3-none-any.whl → 0.1.8__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -62,5 +62,7 @@ def print_json(value: Any) -> None:
62
62
  def write_json(path: str | None, value: Any) -> None:
63
63
  if not path:
64
64
  return
65
- Path(path).write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
65
+ out = Path(path)
66
+ out.parent.mkdir(parents=True, exist_ok=True)
67
+ out.write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
66
68
  log(f"wrote full detail to {path}")
@@ -653,6 +653,64 @@ def _group_case_ids_by_evaluator(pairs: list[dict[str, str]]) -> dict[str, list[
653
653
  return dict(grouped)
654
654
 
655
655
 
656
+ def _pairs_from_rubric_batches(
657
+ client,
658
+ *,
659
+ case_ids: list[str],
660
+ evaluator_ids: list[str],
661
+ ) -> list[dict[str, str]]:
662
+ pairs: list[dict[str, str]] = []
663
+ for evaluator_id in evaluator_ids:
664
+ for row in eval_mod.get_rubrics_batch(client, case_ids=case_ids, evaluator_id=evaluator_id):
665
+ if row.get("rubric"):
666
+ case_id = row.get("case_id") or row.get("evaluation_case_id")
667
+ if case_id:
668
+ pairs.append({"case_id": str(case_id), "evaluator_id": evaluator_id})
669
+ return pairs
670
+
671
+
672
+ def _pair_key(pair: dict[str, str]) -> tuple[str, str]:
673
+ return pair["case_id"], pair["evaluator_id"]
674
+
675
+
676
+ def _skipped_pairs_from_trigger_response(response: Any) -> list[dict[str, str]]:
677
+ if not isinstance(response, dict):
678
+ return []
679
+ payload = response.get("data") if isinstance(response.get("data"), dict) else response
680
+ skipped = payload.get("skipped_pairs") if isinstance(payload, dict) else None
681
+ if not isinstance(skipped, list):
682
+ return []
683
+
684
+ out: list[dict[str, str]] = []
685
+ for row in skipped:
686
+ if not isinstance(row, dict):
687
+ continue
688
+ case_id = row.get("case_id")
689
+ evaluator_id = row.get("evaluator_id")
690
+ if not case_id or not evaluator_id:
691
+ continue
692
+ out.append({
693
+ "case_id": str(case_id),
694
+ "evaluator_id": str(evaluator_id),
695
+ "reason": str(row.get("reason") or "skipped"),
696
+ })
697
+ return out
698
+
699
+
700
+ def _remove_non_runnable_skipped_pairs(
701
+ pairs: list[dict[str, str]],
702
+ skipped_pairs: list[dict[str, str]],
703
+ ) -> list[dict[str, str]]:
704
+ non_runnable = {
705
+ _pair_key(pair)
706
+ for pair in skipped_pairs
707
+ if pair.get("reason") == "not_assigned"
708
+ }
709
+ if not non_runnable:
710
+ return pairs
711
+ return [pair for pair in pairs if _pair_key(pair) not in non_runnable]
712
+
713
+
656
714
  def run_run(args, client) -> int:
657
715
  workspace_id, _ = client.resolve_scope()
658
716
  if args.latest or not args.history:
@@ -682,18 +740,22 @@ def run_run(args, client) -> int:
682
740
  evaluator_ids = [args.evaluator] if args.evaluator else (_ids(args.evaluators) or [])
683
741
  requested_evaluator_ids = evaluator_ids or None
684
742
 
685
- assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
686
- assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
687
- pairs, skipped_unassigned = _planned_eval_pairs(
688
- case_ids=case_ids,
689
- assigned_by_case=assigned_by_case,
690
- requested_evaluator_ids=requested_evaluator_ids,
691
- )
743
+ skipped_unassigned: list[dict[str, str]] = []
744
+ if requested_evaluator_ids:
745
+ pairs = [
746
+ {"case_id": case_id, "evaluator_id": evaluator_id}
747
+ for evaluator_id in requested_evaluator_ids
748
+ for case_id in case_ids
749
+ ]
750
+ else:
751
+ evaluator_ids = [e["id"] for e in eval_mod.list_evaluators(client, workspace_id)]
752
+ pairs = _pairs_from_rubric_batches(
753
+ client,
754
+ case_ids=case_ids,
755
+ evaluator_ids=evaluator_ids,
756
+ )
692
757
  if not pairs:
693
- if skipped_unassigned:
694
- log("error: none of the requested case/evaluator pairs are assigned")
695
- else:
696
- log("error: no assigned case/evaluator pairs to run")
758
+ log("error: no case/evaluator pairs to run")
697
759
  print_json({
698
760
  "agent_id": args.agent,
699
761
  "history_id": args.history,
@@ -710,14 +772,47 @@ def run_run(args, client) -> int:
710
772
  case_label_by_id = {c["id"]: truncate(c.get("input") or "", 60) for c in case_objs}
711
773
  evaluator_name_by_id = {e["id"]: e.get("name", e["id"]) for e in evaluators}
712
774
 
775
+ requested_pairs = list(pairs)
776
+ requested_pair_count = len(requested_pairs)
777
+ log(f"triggering: {requested_pair_count} case/evaluator pairs on history {args.history}")
778
+ trigger_response: list[dict[str, Any]] = []
779
+ skipped_pairs: list[dict[str, str]] = []
780
+ for ev_id, ev_case_ids in _group_case_ids_by_evaluator(pairs).items():
781
+ response = eval_mod.trigger(
782
+ client,
783
+ case_ids=ev_case_ids,
784
+ evaluator_ids=[ev_id],
785
+ agent_history_id=args.history,
786
+ )
787
+ response_skipped = _skipped_pairs_from_trigger_response(response)
788
+ skipped_pairs.extend(response_skipped)
789
+ trigger_response.append({
790
+ "evaluator_id": ev_id,
791
+ "case_ids": ev_case_ids,
792
+ "response": response,
793
+ "skipped_pairs": response_skipped,
794
+ })
795
+
796
+ pairs = _remove_non_runnable_skipped_pairs(pairs, skipped_pairs)
797
+ skipped_unassigned = [pair for pair in skipped_pairs if pair.get("reason") == "not_assigned"]
713
798
  if skipped_unassigned:
714
- log(f"skipping {len(skipped_unassigned)} unassigned requested pairs")
715
- log(f"triggering: {len(pairs)} assigned case/evaluator pairs on history {args.history}")
716
- trigger_response = eval_mod.trigger_pairs(
717
- client,
718
- case_evaluator_pairs=pairs,
719
- agent_history_id=args.history,
720
- )
799
+ log(f"skipping {len(skipped_unassigned)} not-assigned pairs from polling")
800
+ if not pairs:
801
+ log("error: no runnable case/evaluator pairs after trigger response")
802
+ print_json({
803
+ "agent_id": args.agent,
804
+ "history_id": args.history,
805
+ "requested_case_count": len(case_ids),
806
+ "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
807
+ "requested_pair_count": requested_pair_count,
808
+ "triggered_pair_count": 0,
809
+ "skipped_pair_count": len(skipped_pairs),
810
+ "skipped_unassigned_count": len(skipped_unassigned),
811
+ "trigger_response": trigger_response,
812
+ "skipped_pairs": skipped_pairs,
813
+ "skipped_unassigned": skipped_unassigned,
814
+ })
815
+ return 2
721
816
 
722
817
  deadline = time.time() + args.poll_timeout
723
818
  results_by_eval: dict[str, list[dict]] = {}
@@ -825,22 +920,27 @@ def run_run(args, client) -> int:
825
920
  "history_id": args.history,
826
921
  "requested_case_count": len(case_ids),
827
922
  "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
923
+ "requested_pair_count": requested_pair_count,
828
924
  "triggered_pair_count": len(pairs),
829
925
  "scored_pair_count": len(scored_pair_keys),
926
+ "skipped_pair_count": len(skipped_pairs),
830
927
  "skipped_unassigned_count": len(skipped_unassigned),
831
928
  "all_perfect": all_perfect,
832
929
  "result_count": len(result_summaries),
833
930
  "non_perfect_count": len(non_perfect),
834
931
  "wrote_full_detail": bool(args.out),
835
932
  "trigger_response": trigger_response,
933
+ "skipped_pairs": skipped_pairs,
836
934
  "skipped_unassigned": skipped_unassigned,
837
935
  "results": result_summaries,
838
936
  }
839
937
  full_out = {
840
938
  "agent_id": args.agent,
841
939
  "history_id": args.history,
940
+ "requested_pairs": requested_pairs,
842
941
  "triggered_pairs": pairs,
843
942
  "trigger_response": trigger_response,
943
+ "skipped_pairs": skipped_pairs,
844
944
  "skipped_unassigned": skipped_unassigned,
845
945
  "all_perfect": all_perfect,
846
946
  "results": flat,
@@ -1466,35 +1566,44 @@ def run_rubrics(args, client) -> int:
1466
1566
  log("error: no cases for this agent")
1467
1567
  return 2
1468
1568
 
1469
- assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
1470
- assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
1471
-
1472
1569
  if args.evaluators:
1473
1570
  evaluator_ids = _ids(args.evaluators) or []
1474
1571
  evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
1475
- elif args.all_pairs:
1572
+ else:
1476
1573
  evaluators = eval_mod.list_evaluators(client, workspace_id)
1477
1574
  evaluator_ids = [e["id"] for e in evaluators]
1575
+
1576
+ rubrics = eval_mod.get_case_rubrics(
1577
+ client, agent_id=args.agent, workspace_id=workspace_id,
1578
+ evaluator_ids=evaluator_ids, case_ids=case_ids,
1579
+ )
1580
+ assigned_by_case = {
1581
+ cid: {
1582
+ ev_id: {"evaluator_id": ev_id, "rubric": rubric_text}
1583
+ for ev_id, rubric_text in (rubrics.get(cid) or {}).items()
1584
+ if rubric_text
1585
+ }
1586
+ for cid in case_ids
1587
+ }
1588
+
1589
+ if args.evaluators or args.all_pairs:
1590
+ pass
1478
1591
  else:
1479
1592
  evaluator_ids = _dedupe_preserve_order([
1480
1593
  evaluator_id
1481
1594
  for case_id in case_ids
1482
1595
  for evaluator_id in assigned_by_case.get(case_id, {})
1483
1596
  ])
1484
- evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
1597
+ evaluator_id_set = set(evaluator_ids)
1598
+ evaluators = [e for e in evaluators if e["id"] in evaluator_id_set]
1485
1599
  evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
1486
1600
  if not evaluator_ids:
1487
- log("error: no assigned evaluators for these cases")
1601
+ log("error: no evaluators with configured rubrics for these cases")
1488
1602
  return 2
1489
1603
 
1490
- mode = "all requested pairs" if args.evaluators or args.all_pairs else "assigned pairs"
1604
+ mode = "all requested pairs" if args.evaluators or args.all_pairs else "pairs with configured rubrics"
1491
1605
  log(f"reading {mode}: {len(case_ids)} cases, {len(evaluator_ids)} evaluators...")
1492
1606
 
1493
- rubrics = eval_mod.get_case_rubrics(
1494
- client, agent_id=args.agent, workspace_id=workspace_id,
1495
- evaluator_ids=evaluator_ids, case_ids=case_ids,
1496
- )
1497
-
1498
1607
  if args.full:
1499
1608
  for cid in case_ids:
1500
1609
  log("=" * 80)
codeer_cli/eval_.py CHANGED
@@ -54,8 +54,26 @@ def create_case(
54
54
  return client.post("/external/eval/cases", json=body)
55
55
 
56
56
 
57
+ def _unwrap_list_response(value: Any, *keys: str) -> list[dict]:
58
+ """Normalize list endpoints that may return either a bare list or envelope."""
59
+ if isinstance(value, list):
60
+ return value
61
+ if isinstance(value, dict):
62
+ for key in keys:
63
+ rows = value.get(key)
64
+ if isinstance(rows, list):
65
+ return rows
66
+ return []
67
+
68
+
57
69
  def list_cases(client: CodeerClient, agent_id: str) -> list[dict]:
58
- return client.get(f"/external/eval/agents/{agent_id}/cases")
70
+ return _unwrap_list_response(
71
+ client.get(f"/external/eval/agents/{agent_id}/cases"),
72
+ "cases",
73
+ "evaluation_cases",
74
+ "data",
75
+ "items",
76
+ )
59
77
 
60
78
 
61
79
  def get_case(client: CodeerClient, case_id: str) -> dict:
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codeer-cli
3
- Version: 0.1.7
3
+ Version: 0.1.8
4
4
  Summary: Command line tools for managing Codeer agents over the Codeer API.
5
5
  Project-URL: Homepage, https://www.codeer.ai
6
6
  Author: Codeer.AI
@@ -5,19 +5,19 @@ codeer_cli/chats.py,sha256=YVrZJhoa-d67o6tzX6riGXsbA-ehyhOxrZ8zRCcJNro,2675
5
5
  codeer_cli/cli.py,sha256=g-WR2D5MkaUdc13ZrpRCavXD1940CHE9eBELC034tic,4443
6
6
  codeer_cli/client.py,sha256=LpHVqf1IYNg1wFfIHnO9q4xg2h3IiGOitzCnvwB-Bcw,9809
7
7
  codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
8
- codeer_cli/eval_.py,sha256=XwmPxNOtxSyZq2EOae9FZe5NFp0YyVbIswTn22S8noc,17050
8
+ codeer_cli/eval_.py,sha256=4borDzuU7DOqWi1XidfaELmp7eYjV2FH90uKXgGIDuM,17541
9
9
  codeer_cli/histories.py,sha256=tk28git_peX4x703CIDU8u72JtlGaytyrtlHfxlK-7A,5979
10
10
  codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
11
11
  codeer_cli/parse.py,sha256=qrjZn0MUTjGfucp4cwxy8Pt7WS-0x15kK5F7kWTY8Ps,21818
12
12
  codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
- codeer_cli/commands/_util.py,sha256=VOB_HMWYzHFNY1ElLOED1HB6fpFpsniqH6Yx3VUlMrY,1644
13
+ codeer_cli/commands/_util.py,sha256=X9-9cYgnBYz94hVG4Pe9SsI97DHNv3HE5_7GlPMn93I,1708
14
14
  codeer_cli/commands/agent.py,sha256=amvfVVrbPOkbYKCGvA6EJB30C-aY6WSdfR7u7FklXbs,14793
15
15
  codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
16
- codeer_cli/commands/eval_cmd.py,sha256=fQu8ZRzGO7GWL_Og9NeZ0xYwHoN_iocmCnWw-kPNRkc,68599
16
+ codeer_cli/commands/eval_cmd.py,sha256=V_6OptwWaQ8QLtzJeBqEtlHLUpYXWce1UneeMGPog14,72456
17
17
  codeer_cli/commands/history.py,sha256=Jv7t0GhSZcbZ8OuIXZT34CixXt7ECEVP3nZ-WW_Ya9E,12026
18
18
  codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
19
19
  codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
20
- codeer_cli-0.1.7.dist-info/METADATA,sha256=z8yvV20PQ1ysK27LpsMuurXdAPWJ4Kz-o9-LuxrHczg,5999
21
- codeer_cli-0.1.7.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
- codeer_cli-0.1.7.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
- codeer_cli-0.1.7.dist-info/RECORD,,
20
+ codeer_cli-0.1.8.dist-info/METADATA,sha256=dj78KQR8zotgqs8U_VaO5hmIF3ib8XfdGYLDyG5kLOo,5999
21
+ codeer_cli-0.1.8.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
+ codeer_cli-0.1.8.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
+ codeer_cli-0.1.8.dist-info/RECORD,,