codeer-cli 0.1.5__py3-none-any.whl → 0.1.7__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codeer_cli/cli.py +4 -1
- codeer_cli/commands/eval_cmd.py +151 -19
- codeer_cli/commands/kb.py +81 -0
- codeer_cli/eval_.py +51 -8
- codeer_cli/kb.py +10 -0
- {codeer_cli-0.1.5.dist-info → codeer_cli-0.1.7.dist-info}/METADATA +16 -2
- {codeer_cli-0.1.5.dist-info → codeer_cli-0.1.7.dist-info}/RECORD +9 -9
- {codeer_cli-0.1.5.dist-info → codeer_cli-0.1.7.dist-info}/WHEEL +0 -0
- {codeer_cli-0.1.5.dist-info → codeer_cli-0.1.7.dist-info}/entry_points.txt +0 -0
codeer_cli/cli.py
CHANGED
|
@@ -2,7 +2,7 @@
|
|
|
2
2
|
|
|
3
3
|
codeer check
|
|
4
4
|
codeer agent list|get|apply|diff|versions
|
|
5
|
-
codeer kb list|files|upload|faq-list|faq-get|faq-create|faq-update|faq-delete
|
|
5
|
+
codeer kb list|files|upload|node-rename|node-delete|faq-list|faq-get|faq-create|faq-update|faq-delete
|
|
6
6
|
codeer eval list|label-list|label-create|label-update|label-delete|case-update|case-delete|evaluators|evaluator-create|evaluator-update|run|export|reconcile|cases-apply|rubrics|rubrics-apply
|
|
7
7
|
codeer history list|get|conversations|negative-feedback
|
|
8
8
|
"""
|
|
@@ -28,6 +28,7 @@ Safe workflow for coding agents:
|
|
|
28
28
|
codeer agent list
|
|
29
29
|
codeer agent get <agent-id> --full
|
|
30
30
|
codeer kb list
|
|
31
|
+
codeer kb files --kb-id <kb-id>
|
|
31
32
|
codeer eval list --agent <agent-id>
|
|
32
33
|
codeer eval label-list
|
|
33
34
|
codeer eval case-update --case <case-id> --input "..." --dry-run
|
|
@@ -43,6 +44,8 @@ Preview mutations before applying:
|
|
|
43
44
|
codeer eval cases-apply --agent <agent-id> --cases eval_cases.json --dry-run
|
|
44
45
|
codeer eval rubrics-apply --rubrics rubrics.json --dry-run
|
|
45
46
|
codeer kb upload --dir kb --name "Product KB" --dry-run
|
|
47
|
+
codeer kb node-rename --node-id <node-id> --name "New Name" --dry-run
|
|
48
|
+
codeer kb node-delete --node-id <node-id> --dry-run
|
|
46
49
|
codeer kb faq-create --context-object-id <snapshot-object-id> --question "..." --dry-run
|
|
47
50
|
|
|
48
51
|
Use --out <path> for large raw artifacts; stdout defaults to compact summaries.
|
codeer_cli/commands/eval_cmd.py
CHANGED
|
@@ -154,7 +154,9 @@ def register(subparsers):
|
|
|
154
154
|
g.add_argument("--latest", action="store_true",
|
|
155
155
|
help="Auto-select the newest AgentHistory (default)")
|
|
156
156
|
p.add_argument("--cases", default=None, help="Comma-separated case UUIDs (default: all)")
|
|
157
|
-
p.
|
|
157
|
+
g = p.add_mutually_exclusive_group()
|
|
158
|
+
g.add_argument("--evaluator", default=None, help="Evaluator UUID; common path for running many cases with one tester")
|
|
159
|
+
g.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
|
|
158
160
|
p.add_argument("--poll-timeout", type=int, default=POLL_TIMEOUT)
|
|
159
161
|
p.add_argument("--full", action="store_true",
|
|
160
162
|
help="Use longer previews in stdout. Raw outputs/tool calls still require --out.")
|
|
@@ -195,10 +197,12 @@ def register(subparsers):
|
|
|
195
197
|
p.set_defaults(func=run_cases_apply)
|
|
196
198
|
|
|
197
199
|
# codeer eval rubrics
|
|
198
|
-
p = sub.add_parser("rubrics", help="Read per-(case, evaluator) rubrics")
|
|
200
|
+
p = sub.add_parser("rubrics", help="Read assigned per-(case, evaluator) rubrics")
|
|
199
201
|
p.add_argument("--agent", required=True)
|
|
200
202
|
p.add_argument("--evaluators", default=None, help="Comma-separated evaluator UUIDs")
|
|
201
203
|
p.add_argument("--cases", default=None, help="Comma-separated case UUIDs")
|
|
204
|
+
p.add_argument("--all-pairs", action="store_true",
|
|
205
|
+
help="With omitted --evaluators, scan every workspace evaluator instead of assigned pairs only.")
|
|
202
206
|
p.add_argument("--full", action="store_true",
|
|
203
207
|
help="Print complete rubric text. Default prints matrix summaries/previews.")
|
|
204
208
|
p.add_argument("--out", default=None,
|
|
@@ -599,6 +603,56 @@ def run_evaluator_update(args, client) -> int:
|
|
|
599
603
|
# eval run
|
|
600
604
|
# ---------------------------------------------------------------------------
|
|
601
605
|
|
|
606
|
+
def _assigned_evaluators_by_case(info_rows: list[dict]) -> dict[str, dict[str, dict]]:
|
|
607
|
+
out: dict[str, dict[str, dict]] = {}
|
|
608
|
+
for row in info_rows:
|
|
609
|
+
case_id = row.get("case_id")
|
|
610
|
+
if not case_id:
|
|
611
|
+
continue
|
|
612
|
+
out[str(case_id)] = {
|
|
613
|
+
str(info.get("evaluator_id")): info
|
|
614
|
+
for info in (row.get("evaluators") or [])
|
|
615
|
+
if info.get("evaluator_id")
|
|
616
|
+
}
|
|
617
|
+
return out
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def _planned_eval_pairs(
|
|
621
|
+
*,
|
|
622
|
+
case_ids: list[str],
|
|
623
|
+
assigned_by_case: dict[str, dict[str, dict]],
|
|
624
|
+
requested_evaluator_ids: list[str] | None,
|
|
625
|
+
) -> tuple[list[dict[str, str]], list[dict[str, str]]]:
|
|
626
|
+
pairs: list[dict[str, str]] = []
|
|
627
|
+
skipped: list[dict[str, str]] = []
|
|
628
|
+
|
|
629
|
+
if requested_evaluator_ids:
|
|
630
|
+
for case_id in case_ids:
|
|
631
|
+
assigned = assigned_by_case.get(case_id, {})
|
|
632
|
+
for evaluator_id in requested_evaluator_ids:
|
|
633
|
+
if evaluator_id in assigned:
|
|
634
|
+
pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
|
|
635
|
+
else:
|
|
636
|
+
skipped.append({
|
|
637
|
+
"case_id": case_id,
|
|
638
|
+
"evaluator_id": evaluator_id,
|
|
639
|
+
"reason": "not_assigned",
|
|
640
|
+
})
|
|
641
|
+
return pairs, skipped
|
|
642
|
+
|
|
643
|
+
for case_id in case_ids:
|
|
644
|
+
for evaluator_id in assigned_by_case.get(case_id, {}):
|
|
645
|
+
pairs.append({"case_id": case_id, "evaluator_id": evaluator_id})
|
|
646
|
+
return pairs, skipped
|
|
647
|
+
|
|
648
|
+
|
|
649
|
+
def _group_case_ids_by_evaluator(pairs: list[dict[str, str]]) -> dict[str, list[str]]:
|
|
650
|
+
grouped: dict[str, list[str]] = defaultdict(list)
|
|
651
|
+
for pair in pairs:
|
|
652
|
+
grouped[pair["evaluator_id"]].append(pair["case_id"])
|
|
653
|
+
return dict(grouped)
|
|
654
|
+
|
|
655
|
+
|
|
602
656
|
def run_run(args, client) -> int:
|
|
603
657
|
workspace_id, _ = client.resolve_scope()
|
|
604
658
|
if args.latest or not args.history:
|
|
@@ -625,41 +679,76 @@ def run_run(args, client) -> int:
|
|
|
625
679
|
log("error: no cases to run")
|
|
626
680
|
return 2
|
|
627
681
|
|
|
628
|
-
evaluator_ids = _ids(args.evaluators) or []
|
|
629
|
-
|
|
630
|
-
|
|
682
|
+
evaluator_ids = [args.evaluator] if args.evaluator else (_ids(args.evaluators) or [])
|
|
683
|
+
requested_evaluator_ids = evaluator_ids or None
|
|
684
|
+
|
|
685
|
+
assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
|
|
686
|
+
assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
|
|
687
|
+
pairs, skipped_unassigned = _planned_eval_pairs(
|
|
688
|
+
case_ids=case_ids,
|
|
689
|
+
assigned_by_case=assigned_by_case,
|
|
690
|
+
requested_evaluator_ids=requested_evaluator_ids,
|
|
691
|
+
)
|
|
692
|
+
if not pairs:
|
|
693
|
+
if skipped_unassigned:
|
|
694
|
+
log("error: none of the requested case/evaluator pairs are assigned")
|
|
695
|
+
else:
|
|
696
|
+
log("error: no assigned case/evaluator pairs to run")
|
|
697
|
+
print_json({
|
|
698
|
+
"agent_id": args.agent,
|
|
699
|
+
"history_id": args.history,
|
|
700
|
+
"requested_case_count": len(case_ids),
|
|
701
|
+
"requested_evaluator_count": len(evaluator_ids),
|
|
702
|
+
"triggered_pair_count": 0,
|
|
703
|
+
"skipped_unassigned": skipped_unassigned,
|
|
704
|
+
})
|
|
631
705
|
return 2
|
|
706
|
+
|
|
707
|
+
evaluator_ids = _dedupe_preserve_order([pair["evaluator_id"] for pair in pairs])
|
|
632
708
|
evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
|
|
633
709
|
|
|
634
710
|
case_label_by_id = {c["id"]: truncate(c.get("input") or "", 60) for c in case_objs}
|
|
635
711
|
evaluator_name_by_id = {e["id"]: e.get("name", e["id"]) for e in evaluators}
|
|
636
712
|
|
|
637
|
-
|
|
638
|
-
|
|
639
|
-
|
|
713
|
+
if skipped_unassigned:
|
|
714
|
+
log(f"skipping {len(skipped_unassigned)} unassigned requested pairs")
|
|
715
|
+
log(f"triggering: {len(pairs)} assigned case/evaluator pairs on history {args.history}")
|
|
716
|
+
trigger_response = eval_mod.trigger_pairs(
|
|
717
|
+
client,
|
|
718
|
+
case_evaluator_pairs=pairs,
|
|
719
|
+
agent_history_id=args.history,
|
|
720
|
+
)
|
|
640
721
|
|
|
641
722
|
deadline = time.time() + args.poll_timeout
|
|
642
723
|
results_by_eval: dict[str, list[dict]] = {}
|
|
724
|
+
case_ids_by_evaluator = _group_case_ids_by_evaluator(pairs)
|
|
725
|
+
target_pair_keys = {(pair["case_id"], pair["evaluator_id"]) for pair in pairs}
|
|
643
726
|
while time.time() < deadline:
|
|
644
727
|
results_by_eval = {}
|
|
645
|
-
|
|
646
|
-
total = len(
|
|
647
|
-
for ev_id in
|
|
728
|
+
done_pairs: set[tuple[str, str]] = set()
|
|
729
|
+
total = len(pairs)
|
|
730
|
+
for ev_id, ev_case_ids in case_ids_by_evaluator.items():
|
|
648
731
|
rows = eval_mod.get_results(
|
|
649
|
-
client, case_ids=
|
|
732
|
+
client, case_ids=ev_case_ids, evaluator_id=ev_id,
|
|
650
733
|
agent_history_id=args.history, workspace_id=workspace_id,
|
|
651
734
|
include_output=True, include_reasoning_steps=True,
|
|
652
735
|
)
|
|
653
736
|
results_by_eval[ev_id] = rows
|
|
654
|
-
|
|
655
|
-
|
|
656
|
-
|
|
737
|
+
for row in rows:
|
|
738
|
+
key = (row.get("case_id") or row.get("evaluation_case_id"), ev_id)
|
|
739
|
+
if key in target_pair_keys and row.get("score") is not None:
|
|
740
|
+
done_pairs.add(key)
|
|
741
|
+
log(f" progress: {len(done_pairs)}/{total}")
|
|
742
|
+
if len(done_pairs) >= total:
|
|
657
743
|
break
|
|
658
744
|
time.sleep(POLL_INTERVAL)
|
|
659
745
|
|
|
660
746
|
flat: list[dict] = []
|
|
661
747
|
for ev_id, rows in results_by_eval.items():
|
|
662
748
|
for r in rows:
|
|
749
|
+
row_case_id = r.get("case_id") or r.get("evaluation_case_id")
|
|
750
|
+
if (row_case_id, ev_id) not in target_pair_keys:
|
|
751
|
+
continue
|
|
663
752
|
result_summary = parse_eval_result(r)
|
|
664
753
|
tool_calls = parse_eval_tool_calls(r)
|
|
665
754
|
total_tool_duration_ms = sum(
|
|
@@ -684,7 +773,15 @@ def run_run(args, client) -> int:
|
|
|
684
773
|
"raw_result": r,
|
|
685
774
|
})
|
|
686
775
|
|
|
687
|
-
|
|
776
|
+
scored_pair_keys = {
|
|
777
|
+
(r.get("case_id"), r.get("evaluator_id"))
|
|
778
|
+
for r in flat
|
|
779
|
+
if r.get("score") is not None
|
|
780
|
+
}
|
|
781
|
+
all_perfect = (
|
|
782
|
+
len(scored_pair_keys) == len(target_pair_keys)
|
|
783
|
+
and all((r.get("score") or 0.0) >= 1.0 for r in flat)
|
|
784
|
+
)
|
|
688
785
|
log("\n" + "=" * 80)
|
|
689
786
|
log(f"RESULTS agent={args.agent} history={args.history}")
|
|
690
787
|
log("=" * 80)
|
|
@@ -726,15 +823,25 @@ def run_run(args, client) -> int:
|
|
|
726
823
|
out = {
|
|
727
824
|
"agent_id": args.agent,
|
|
728
825
|
"history_id": args.history,
|
|
826
|
+
"requested_case_count": len(case_ids),
|
|
827
|
+
"requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
|
|
828
|
+
"triggered_pair_count": len(pairs),
|
|
829
|
+
"scored_pair_count": len(scored_pair_keys),
|
|
830
|
+
"skipped_unassigned_count": len(skipped_unassigned),
|
|
729
831
|
"all_perfect": all_perfect,
|
|
730
832
|
"result_count": len(result_summaries),
|
|
731
833
|
"non_perfect_count": len(non_perfect),
|
|
732
834
|
"wrote_full_detail": bool(args.out),
|
|
835
|
+
"trigger_response": trigger_response,
|
|
836
|
+
"skipped_unassigned": skipped_unassigned,
|
|
733
837
|
"results": result_summaries,
|
|
734
838
|
}
|
|
735
839
|
full_out = {
|
|
736
840
|
"agent_id": args.agent,
|
|
737
841
|
"history_id": args.history,
|
|
842
|
+
"triggered_pairs": pairs,
|
|
843
|
+
"trigger_response": trigger_response,
|
|
844
|
+
"skipped_unassigned": skipped_unassigned,
|
|
738
845
|
"all_perfect": all_perfect,
|
|
739
846
|
"results": flat,
|
|
740
847
|
}
|
|
@@ -1359,18 +1466,29 @@ def run_rubrics(args, client) -> int:
|
|
|
1359
1466
|
log("error: no cases for this agent")
|
|
1360
1467
|
return 2
|
|
1361
1468
|
|
|
1469
|
+
assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
|
|
1470
|
+
assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
|
|
1471
|
+
|
|
1362
1472
|
if args.evaluators:
|
|
1363
1473
|
evaluator_ids = _ids(args.evaluators) or []
|
|
1364
1474
|
evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
|
|
1365
|
-
|
|
1475
|
+
elif args.all_pairs:
|
|
1366
1476
|
evaluators = eval_mod.list_evaluators(client, workspace_id)
|
|
1367
1477
|
evaluator_ids = [e["id"] for e in evaluators]
|
|
1478
|
+
else:
|
|
1479
|
+
evaluator_ids = _dedupe_preserve_order([
|
|
1480
|
+
evaluator_id
|
|
1481
|
+
for case_id in case_ids
|
|
1482
|
+
for evaluator_id in assigned_by_case.get(case_id, {})
|
|
1483
|
+
])
|
|
1484
|
+
evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
|
|
1368
1485
|
evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
|
|
1369
1486
|
if not evaluator_ids:
|
|
1370
|
-
log("error: no evaluators
|
|
1487
|
+
log("error: no assigned evaluators for these cases")
|
|
1371
1488
|
return 2
|
|
1372
1489
|
|
|
1373
|
-
|
|
1490
|
+
mode = "all requested pairs" if args.evaluators or args.all_pairs else "assigned pairs"
|
|
1491
|
+
log(f"reading {mode}: {len(case_ids)} cases, {len(evaluator_ids)} evaluators...")
|
|
1374
1492
|
|
|
1375
1493
|
rubrics = eval_mod.get_case_rubrics(
|
|
1376
1494
|
client, agent_id=args.agent, workspace_id=workspace_id,
|
|
@@ -1383,8 +1501,14 @@ def run_rubrics(args, client) -> int:
|
|
|
1383
1501
|
log(f"CASE {cid}")
|
|
1384
1502
|
log(f" input: {truncate(case_input.get(cid, ''), 120)}")
|
|
1385
1503
|
for ev_id in evaluator_ids:
|
|
1504
|
+
is_assigned = ev_id in assigned_by_case.get(cid, {})
|
|
1505
|
+
if not is_assigned and not (args.evaluators or args.all_pairs):
|
|
1506
|
+
continue
|
|
1386
1507
|
ev_name = evaluator_name.get(ev_id, ev_id)
|
|
1387
1508
|
rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
|
|
1509
|
+
if not is_assigned:
|
|
1510
|
+
log(f" [{ev_name}] (not assigned)")
|
|
1511
|
+
continue
|
|
1388
1512
|
if not rubric_text:
|
|
1389
1513
|
log(f" [{ev_name}] (rubric not set)")
|
|
1390
1514
|
else:
|
|
@@ -1396,9 +1520,13 @@ def run_rubrics(args, client) -> int:
|
|
|
1396
1520
|
for cid in case_ids:
|
|
1397
1521
|
rubrics_summary = {}
|
|
1398
1522
|
for ev_id in evaluator_ids:
|
|
1523
|
+
is_assigned = ev_id in assigned_by_case.get(cid, {})
|
|
1524
|
+
if not is_assigned and not (args.evaluators or args.all_pairs):
|
|
1525
|
+
continue
|
|
1399
1526
|
rubric_text = (rubrics.get(cid) or {}).get(ev_id, "")
|
|
1400
1527
|
rubrics_summary[ev_id] = {
|
|
1401
1528
|
"evaluator_name": evaluator_name.get(ev_id, ev_id),
|
|
1529
|
+
"is_assigned": is_assigned,
|
|
1402
1530
|
"is_set": bool(rubric_text),
|
|
1403
1531
|
"chars": len(rubric_text),
|
|
1404
1532
|
"preview": truncate(rubric_text, 240),
|
|
@@ -1412,6 +1540,7 @@ def run_rubrics(args, client) -> int:
|
|
|
1412
1540
|
out = {
|
|
1413
1541
|
"agent_id": args.agent,
|
|
1414
1542
|
"workspace_id": workspace_id,
|
|
1543
|
+
"mode": mode,
|
|
1415
1544
|
"evaluators": [{"id": e["id"], "name": e.get("name")} for e in evaluators],
|
|
1416
1545
|
"cases": [
|
|
1417
1546
|
{
|
|
@@ -1420,7 +1549,9 @@ def run_rubrics(args, client) -> int:
|
|
|
1420
1549
|
"rubrics_by_evaluator": {
|
|
1421
1550
|
ev_id: (rubrics.get(cid) or {}).get(ev_id)
|
|
1422
1551
|
for ev_id in evaluator_ids
|
|
1552
|
+
if ev_id in assigned_by_case.get(cid, {}) or args.evaluators or args.all_pairs
|
|
1423
1553
|
},
|
|
1554
|
+
"assigned_evaluator_ids": list(assigned_by_case.get(cid, {})),
|
|
1424
1555
|
}
|
|
1425
1556
|
for cid in case_ids
|
|
1426
1557
|
],
|
|
@@ -1432,6 +1563,7 @@ def run_rubrics(args, client) -> int:
|
|
|
1432
1563
|
print_json({
|
|
1433
1564
|
"agent_id": args.agent,
|
|
1434
1565
|
"workspace_id": workspace_id,
|
|
1566
|
+
"mode": mode,
|
|
1435
1567
|
"evaluator_count": len(evaluators),
|
|
1436
1568
|
"case_count": len(case_ids),
|
|
1437
1569
|
"wrote_full_detail": bool(args.out),
|
codeer_cli/commands/kb.py
CHANGED
|
@@ -87,6 +87,21 @@ def register(subparsers):
|
|
|
87
87
|
p.add_argument("--poll-timeout", type=int, default=POLL_TIMEOUT)
|
|
88
88
|
p.set_defaults(func=run_upload)
|
|
89
89
|
|
|
90
|
+
p = sub.add_parser("node-rename", help="Rename a KB root, folder, or file node; run --dry-run first")
|
|
91
|
+
p.add_argument("--node-id", required=True, help="KnowledgeNode UUID")
|
|
92
|
+
p.add_argument("--name", required=True, help="New display name")
|
|
93
|
+
p.add_argument("--dry-run", action="store_true",
|
|
94
|
+
help="Print intended request without writing server state.")
|
|
95
|
+
p.add_argument("--out", default=None, help="Write result JSON to this file too")
|
|
96
|
+
p.set_defaults(func=run_node_rename)
|
|
97
|
+
|
|
98
|
+
p = sub.add_parser("node-delete", help="Delete a KB root, folder, or file node and descendants; run --dry-run first")
|
|
99
|
+
p.add_argument("--node-id", required=True, help="KnowledgeNode UUID")
|
|
100
|
+
p.add_argument("--dry-run", action="store_true",
|
|
101
|
+
help="Print intended request without writing server state.")
|
|
102
|
+
p.add_argument("--out", default=None, help="Write result JSON to this file too")
|
|
103
|
+
p.set_defaults(func=run_node_delete)
|
|
104
|
+
|
|
90
105
|
p = sub.add_parser("faq-list", help="List Context Object FAQ entries")
|
|
91
106
|
p.add_argument("--context-object-id", type=int, default=None,
|
|
92
107
|
help="Filter to a KB file snapshot_object_id")
|
|
@@ -413,6 +428,72 @@ def run_upload(args, client) -> int:
|
|
|
413
428
|
return 0 if not not_ready else 1
|
|
414
429
|
|
|
415
430
|
|
|
431
|
+
def run_node_rename(args, client) -> int:
|
|
432
|
+
workspace_id, organization_id = client.resolve_scope()
|
|
433
|
+
path = f"/external/knowledge-bases/nodes/{args.node_id}"
|
|
434
|
+
body = {"name": args.name}
|
|
435
|
+
if args.dry_run:
|
|
436
|
+
return _dry_run(
|
|
437
|
+
args.out,
|
|
438
|
+
{
|
|
439
|
+
"dry_run": True,
|
|
440
|
+
"operation": "kb_node_rename",
|
|
441
|
+
"method": "PATCH",
|
|
442
|
+
"path": path,
|
|
443
|
+
"workspace_id": workspace_id,
|
|
444
|
+
"organization_id": organization_id,
|
|
445
|
+
"node_id": args.node_id,
|
|
446
|
+
"body": body,
|
|
447
|
+
"would_write_server_state": True,
|
|
448
|
+
"next_step": "Review this summary, then rerun without --dry-run after approval.",
|
|
449
|
+
},
|
|
450
|
+
)
|
|
451
|
+
|
|
452
|
+
response = strip_noisy_fields(
|
|
453
|
+
kb_mod.update_node(
|
|
454
|
+
client,
|
|
455
|
+
organization_id=organization_id,
|
|
456
|
+
workspace_id=workspace_id,
|
|
457
|
+
node_id=args.node_id,
|
|
458
|
+
name=args.name,
|
|
459
|
+
)
|
|
460
|
+
)
|
|
461
|
+
_print_and_write(args.out, response)
|
|
462
|
+
return 0
|
|
463
|
+
|
|
464
|
+
|
|
465
|
+
def run_node_delete(args, client) -> int:
|
|
466
|
+
workspace_id, organization_id = client.resolve_scope()
|
|
467
|
+
path = f"/external/knowledge-bases/nodes/{args.node_id}"
|
|
468
|
+
if args.dry_run:
|
|
469
|
+
return _dry_run(
|
|
470
|
+
args.out,
|
|
471
|
+
{
|
|
472
|
+
"dry_run": True,
|
|
473
|
+
"operation": "kb_node_delete",
|
|
474
|
+
"method": "DELETE",
|
|
475
|
+
"path": path,
|
|
476
|
+
"workspace_id": workspace_id,
|
|
477
|
+
"organization_id": organization_id,
|
|
478
|
+
"node_id": args.node_id,
|
|
479
|
+
"deletes_descendants": True,
|
|
480
|
+
"would_write_server_state": True,
|
|
481
|
+
"next_step": "Review this summary, then rerun without --dry-run after approval.",
|
|
482
|
+
},
|
|
483
|
+
)
|
|
484
|
+
|
|
485
|
+
response = strip_noisy_fields(
|
|
486
|
+
kb_mod.delete_node(
|
|
487
|
+
client,
|
|
488
|
+
organization_id=organization_id,
|
|
489
|
+
workspace_id=workspace_id,
|
|
490
|
+
node_id=args.node_id,
|
|
491
|
+
)
|
|
492
|
+
)
|
|
493
|
+
_print_and_write(args.out, response)
|
|
494
|
+
return 0
|
|
495
|
+
|
|
496
|
+
|
|
416
497
|
def run_faq_list(args, client) -> int:
|
|
417
498
|
faqs = kb_mod.list_context_obj_faqs(
|
|
418
499
|
client,
|
codeer_cli/eval_.py
CHANGED
|
@@ -23,6 +23,7 @@ def create_case(
|
|
|
23
23
|
rubric: Optional[str] = None,
|
|
24
24
|
attachment_ids: Optional[List[str]] = None,
|
|
25
25
|
label_ids: Optional[List[str]] = None,
|
|
26
|
+
evaluators: Optional[List[dict[str, Any]]] = None,
|
|
26
27
|
meta: Optional[dict] = None,
|
|
27
28
|
note: Optional[str] = None,
|
|
28
29
|
) -> dict:
|
|
@@ -44,6 +45,8 @@ def create_case(
|
|
|
44
45
|
body["attachment_ids"] = attachment_ids
|
|
45
46
|
if label_ids is not None:
|
|
46
47
|
body["label_ids"] = label_ids
|
|
48
|
+
if evaluators is not None:
|
|
49
|
+
body["evaluators"] = evaluators
|
|
47
50
|
if meta:
|
|
48
51
|
body["meta"] = meta
|
|
49
52
|
if note is not None:
|
|
@@ -93,6 +96,31 @@ def delete_case(client: CodeerClient, case_id: str) -> dict:
|
|
|
93
96
|
return client.delete(f"/external/eval/cases/{case_id}")
|
|
94
97
|
|
|
95
98
|
|
|
99
|
+
# --- case/evaluator assignments ------------------------------------------
|
|
100
|
+
|
|
101
|
+
def get_case_evaluator_infos(client: CodeerClient, *, case_ids: List[str]) -> list[dict]:
|
|
102
|
+
"""Read assigned evaluator metadata for each case.
|
|
103
|
+
|
|
104
|
+
Returns rows shaped like ``{"case_id": str, "evaluators": [...]}``, where
|
|
105
|
+
each evaluator entry is the assigned ``{"evaluator_id", "rubric"}`` pair.
|
|
106
|
+
"""
|
|
107
|
+
return client.post("/eval/case-evaluator-infos:batch", json={"case_ids": case_ids})
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def replace_case_evaluator_infos(
|
|
111
|
+
client: CodeerClient,
|
|
112
|
+
*,
|
|
113
|
+
case_id: str,
|
|
114
|
+
evaluators: list[dict[str, Any]],
|
|
115
|
+
) -> dict:
|
|
116
|
+
"""Replace a case's assigned evaluators.
|
|
117
|
+
|
|
118
|
+
This is intentionally separate from rubric upsert: replacing removes
|
|
119
|
+
evaluator assignments that are not included in ``evaluators``.
|
|
120
|
+
"""
|
|
121
|
+
return client.put(f"/eval/cases/{case_id}/case-evaluator-infos", json={"evaluators": evaluators})
|
|
122
|
+
|
|
123
|
+
|
|
96
124
|
# --- case labels -----------------------------------------------------------
|
|
97
125
|
|
|
98
126
|
def list_case_labels(client: CodeerClient, *, workspace_id: str) -> list[dict]:
|
|
@@ -202,6 +230,22 @@ def trigger(
|
|
|
202
230
|
return client.post("/external/eval/runs", json=body)
|
|
203
231
|
|
|
204
232
|
|
|
233
|
+
def trigger_pairs(
|
|
234
|
+
client: CodeerClient,
|
|
235
|
+
*,
|
|
236
|
+
case_evaluator_pairs: list[dict[str, str]],
|
|
237
|
+
agent_history_id: str,
|
|
238
|
+
) -> dict:
|
|
239
|
+
"""Kick off evaluation for explicit assigned case/evaluator pairs."""
|
|
240
|
+
return client.post(
|
|
241
|
+
"/eval/trigger",
|
|
242
|
+
json={
|
|
243
|
+
"case_evaluator_pairs": case_evaluator_pairs,
|
|
244
|
+
"agent_history_id": agent_history_id,
|
|
245
|
+
},
|
|
246
|
+
)
|
|
247
|
+
|
|
248
|
+
|
|
205
249
|
def stop(client: CodeerClient, *, case_id: str, evaluator_id: str) -> Any:
|
|
206
250
|
return client.post("/external/eval/runs:stop", json={"case_id": case_id, "evaluator_id": evaluator_id})
|
|
207
251
|
|
|
@@ -451,9 +495,9 @@ def create_case_with_rubrics(
|
|
|
451
495
|
is filled in for every evaluator it will be judged by.
|
|
452
496
|
|
|
453
497
|
``rubrics_by_evaluator`` maps ``evaluator_id → rubric_text``. Each entry
|
|
454
|
-
becomes a
|
|
455
|
-
|
|
456
|
-
|
|
498
|
+
becomes a case/evaluator assignment on create. Use different rubric wording
|
|
499
|
+
per evaluator when the evaluators judge different aspects (e.g. Style/Tone
|
|
500
|
+
vs Content Compliance).
|
|
457
501
|
"""
|
|
458
502
|
case = create_case(
|
|
459
503
|
client,
|
|
@@ -462,12 +506,11 @@ def create_case_with_rubrics(
|
|
|
462
506
|
expected_output=expected_output,
|
|
463
507
|
attachment_ids=attachment_ids,
|
|
464
508
|
label_ids=label_ids,
|
|
509
|
+
evaluators=[
|
|
510
|
+
{"evaluator_id": ev_id, "rubric": rubric}
|
|
511
|
+
for ev_id, rubric in rubrics_by_evaluator.items()
|
|
512
|
+
],
|
|
465
513
|
meta=meta,
|
|
466
514
|
note=note,
|
|
467
515
|
)
|
|
468
|
-
set_rubric_bulk(
|
|
469
|
-
client,
|
|
470
|
-
evaluation_case_id=case["id"],
|
|
471
|
-
rubrics_by_evaluator=rubrics_by_evaluator,
|
|
472
|
-
)
|
|
473
516
|
return case
|
codeer_cli/kb.py
CHANGED
|
@@ -138,6 +138,16 @@ def update_node(
|
|
|
138
138
|
return client.patch(f"{_base(organization_id, workspace_id)}/nodes/{node_id}", json=body)
|
|
139
139
|
|
|
140
140
|
|
|
141
|
+
def delete_node(
|
|
142
|
+
client: CodeerClient,
|
|
143
|
+
*,
|
|
144
|
+
organization_id: str,
|
|
145
|
+
workspace_id: str,
|
|
146
|
+
node_id: str,
|
|
147
|
+
) -> dict:
|
|
148
|
+
return client.delete(f"{_base(organization_id, workspace_id)}/nodes/{node_id}")
|
|
149
|
+
|
|
150
|
+
|
|
141
151
|
def upload_file(
|
|
142
152
|
client: CodeerClient,
|
|
143
153
|
*,
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: codeer-cli
|
|
3
|
-
Version: 0.1.
|
|
3
|
+
Version: 0.1.7
|
|
4
4
|
Summary: Command line tools for managing Codeer agents over the Codeer API.
|
|
5
5
|
Project-URL: Homepage, https://www.codeer.ai
|
|
6
6
|
Author: Codeer.AI
|
|
@@ -143,7 +143,7 @@ Use this pattern during agent lifecycle work:
|
|
|
143
143
|
```bash
|
|
144
144
|
codeer agent list
|
|
145
145
|
codeer history list --agent <agent-id> --limit 50
|
|
146
|
-
codeer eval run --agent <agent-id> --
|
|
146
|
+
codeer eval run --agent <agent-id> --cases <case-ids> --evaluator <evaluator-id> --out .codeer/eval_run.json
|
|
147
147
|
```
|
|
148
148
|
|
|
149
149
|
Flags:
|
|
@@ -180,6 +180,20 @@ paths containing `*` so the shell passes the wildcard to the CLI. Advanced
|
|
|
180
180
|
settings can still be passed through `--config-json`; explicit crawler flags
|
|
181
181
|
override matching JSON keys.
|
|
182
182
|
|
|
183
|
+
## KB node rename and delete
|
|
184
|
+
|
|
185
|
+
Knowledge Base roots, folders, and files are all KnowledgeNodes. Use
|
|
186
|
+
`codeer kb list` and `codeer kb files` to find node IDs, then preview mutations
|
|
187
|
+
with `--dry-run`:
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
codeer kb node-rename --node-id <node-id> --name "New Name" --dry-run
|
|
191
|
+
codeer kb node-delete --node-id <node-id> --dry-run
|
|
192
|
+
```
|
|
193
|
+
|
|
194
|
+
`node-delete` deletes the target node and all descendants. Review the dry-run
|
|
195
|
+
output before rerunning without `--dry-run`.
|
|
196
|
+
|
|
183
197
|
## Context Object FAQ
|
|
184
198
|
|
|
185
199
|
Use Context Object FAQ entries to route high-value questions to a canonical KB
|
|
@@ -2,22 +2,22 @@ codeer_cli/__init__.py,sha256=-0gL8upoSsLAnXAfcRrwqZYJbwG0knzQoFf94O7Nc7c,1817
|
|
|
2
2
|
codeer_cli/_validate.py,sha256=pKUJa2TyTpERx5xmiYNZRn7tFqDxLZ2fF1rHAf1oz14,5415
|
|
3
3
|
codeer_cli/agents.py,sha256=diodgiGhXlowEi8sbCzcSK1qSeCLF2fBe6QBs3Sq_x8,5617
|
|
4
4
|
codeer_cli/chats.py,sha256=YVrZJhoa-d67o6tzX6riGXsbA-ehyhOxrZ8zRCcJNro,2675
|
|
5
|
-
codeer_cli/cli.py,sha256=
|
|
5
|
+
codeer_cli/cli.py,sha256=g-WR2D5MkaUdc13ZrpRCavXD1940CHE9eBELC034tic,4443
|
|
6
6
|
codeer_cli/client.py,sha256=LpHVqf1IYNg1wFfIHnO9q4xg2h3IiGOitzCnvwB-Bcw,9809
|
|
7
7
|
codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
|
|
8
|
-
codeer_cli/eval_.py,sha256=
|
|
8
|
+
codeer_cli/eval_.py,sha256=XwmPxNOtxSyZq2EOae9FZe5NFp0YyVbIswTn22S8noc,17050
|
|
9
9
|
codeer_cli/histories.py,sha256=tk28git_peX4x703CIDU8u72JtlGaytyrtlHfxlK-7A,5979
|
|
10
|
-
codeer_cli/kb.py,sha256
|
|
10
|
+
codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
|
|
11
11
|
codeer_cli/parse.py,sha256=qrjZn0MUTjGfucp4cwxy8Pt7WS-0x15kK5F7kWTY8Ps,21818
|
|
12
12
|
codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
|
|
13
13
|
codeer_cli/commands/_util.py,sha256=VOB_HMWYzHFNY1ElLOED1HB6fpFpsniqH6Yx3VUlMrY,1644
|
|
14
14
|
codeer_cli/commands/agent.py,sha256=amvfVVrbPOkbYKCGvA6EJB30C-aY6WSdfR7u7FklXbs,14793
|
|
15
15
|
codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
|
|
16
|
-
codeer_cli/commands/eval_cmd.py,sha256=
|
|
16
|
+
codeer_cli/commands/eval_cmd.py,sha256=fQu8ZRzGO7GWL_Og9NeZ0xYwHoN_iocmCnWw-kPNRkc,68599
|
|
17
17
|
codeer_cli/commands/history.py,sha256=Jv7t0GhSZcbZ8OuIXZT34CixXt7ECEVP3nZ-WW_Ya9E,12026
|
|
18
|
-
codeer_cli/commands/kb.py,sha256=
|
|
18
|
+
codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
|
|
19
19
|
codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
|
|
20
|
-
codeer_cli-0.1.
|
|
21
|
-
codeer_cli-0.1.
|
|
22
|
-
codeer_cli-0.1.
|
|
23
|
-
codeer_cli-0.1.
|
|
20
|
+
codeer_cli-0.1.7.dist-info/METADATA,sha256=z8yvV20PQ1ysK27LpsMuurXdAPWJ4Kz-o9-LuxrHczg,5999
|
|
21
|
+
codeer_cli-0.1.7.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
|
|
22
|
+
codeer_cli-0.1.7.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
|
|
23
|
+
codeer_cli-0.1.7.dist-info/RECORD,,
|
|
File without changes
|
|
File without changes
|