codeer-cli 0.1.13__py3-none-any.whl → 0.1.14__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -128,6 +128,12 @@ def register(subparsers):
128
128
  g.add_argument("--system-prompt-template", help="Evaluator system prompt template text")
129
129
  g.add_argument("--system-prompt-template-file", help="Path to evaluator system prompt template")
130
130
  p.add_argument("--description", default=None)
131
+ p.add_argument(
132
+ "--judge-model",
133
+ default=None,
134
+ metavar="MODEL_ID",
135
+ help="Judge LLM model ID (default: system default)",
136
+ )
131
137
  p.add_argument("--dry-run", action="store_true",
132
138
  help="Validate inputs and print intended mutation without writing server state.")
133
139
  p.set_defaults(func=run_evaluator_create)
@@ -140,6 +146,18 @@ def register(subparsers):
140
146
  g.add_argument("--system-prompt-template", help="Evaluator system prompt template text")
141
147
  g.add_argument("--system-prompt-template-file", help="Path to evaluator system prompt template")
142
148
  p.add_argument("--description", default=None)
149
+ g = p.add_mutually_exclusive_group()
150
+ g.add_argument(
151
+ "--judge-model",
152
+ default=None,
153
+ metavar="MODEL_ID",
154
+ help="Set the judge LLM model ID",
155
+ )
156
+ g.add_argument(
157
+ "--clear-judge-model",
158
+ action="store_true",
159
+ help="Clear the evaluator override and use the system default judge model",
160
+ )
143
161
  p.add_argument("--dry-run", action="store_true",
144
162
  help="Validate inputs and print intended mutation without writing server state.")
145
163
  p.set_defaults(func=run_evaluator_update)
@@ -249,6 +267,7 @@ def _evaluator_summary(evaluator: dict, *, full: bool = False) -> dict:
249
267
  "id": evaluator.get("id"),
250
268
  "name": evaluator.get("name"),
251
269
  "description": evaluator.get("description"),
270
+ "judge_llm_model_id": evaluator.get("judge_llm_model_id"),
252
271
  "system_prompt_template_chars": len(template),
253
272
  "has_tool_steps_placeholder": "{tool_steps}" in template,
254
273
  "has_output_placeholder": "{output}" in template,
@@ -534,6 +553,10 @@ def run_evaluator_create(args, client) -> int:
534
553
  "workspace_id": workspace_id,
535
554
  "name": args.name,
536
555
  "description": args.description,
556
+ "judge_model": {
557
+ "action": "set" if args.judge_model is not None else "use_system_default",
558
+ "model_id": args.judge_model,
559
+ },
537
560
  "system_prompt_template_chars": len(system_prompt_template or ""),
538
561
  "would_write_server_state": True,
539
562
  "next_step": "Review this summary, then rerun without --dry-run after approval.",
@@ -545,6 +568,7 @@ def run_evaluator_create(args, client) -> int:
545
568
  name=args.name,
546
569
  system_prompt_template=system_prompt_template,
547
570
  description=args.description,
571
+ judge_llm_model_id=args.judge_model,
548
572
  )
549
573
  print_json(_evaluator_summary(evaluator, full=True))
550
574
  return 0
@@ -560,13 +584,27 @@ def run_evaluator_update(args, client) -> int:
560
584
  else:
561
585
  system_prompt_template = args.system_prompt_template
562
586
 
563
- if args.name is None and args.description is None and system_prompt_template is None:
587
+ if (
588
+ args.name is None
589
+ and args.description is None
590
+ and system_prompt_template is None
591
+ and args.judge_model is None
592
+ and not args.clear_judge_model
593
+ ):
564
594
  log(
565
595
  "error: provide at least one of --name, --description, "
566
- "--system-prompt-template, --system-prompt-template-file"
596
+ "--system-prompt-template, --system-prompt-template-file, "
597
+ "--judge-model, --clear-judge-model"
567
598
  )
568
599
  return 2
569
600
 
601
+ if args.clear_judge_model:
602
+ judge_model = {"action": "clear_to_system_default", "model_id": None}
603
+ elif args.judge_model is not None:
604
+ judge_model = {"action": "set", "model_id": args.judge_model}
605
+ else:
606
+ judge_model = {"action": "unchanged"}
607
+
570
608
  if args.dry_run:
571
609
  print_json({
572
610
  "dry_run": True,
@@ -574,6 +612,7 @@ def run_evaluator_update(args, client) -> int:
574
612
  "evaluator_id": args.evaluator,
575
613
  "name": args.name,
576
614
  "description": args.description,
615
+ "judge_model": judge_model,
577
616
  "system_prompt_template_chars": (
578
617
  len(system_prompt_template) if system_prompt_template is not None else None
579
618
  ),
@@ -582,12 +621,20 @@ def run_evaluator_update(args, client) -> int:
582
621
  })
583
622
  return 0
584
623
 
624
+ evaluator_kwargs: dict[str, Any] = {
625
+ "name": args.name,
626
+ "system_prompt_template": system_prompt_template,
627
+ "description": args.description,
628
+ }
629
+ if args.clear_judge_model:
630
+ evaluator_kwargs["judge_llm_model_id"] = None
631
+ elif args.judge_model is not None:
632
+ evaluator_kwargs["judge_llm_model_id"] = args.judge_model
633
+
585
634
  evaluator = eval_mod.update_evaluator(
586
635
  client,
587
636
  evaluator_id=args.evaluator,
588
- name=args.name,
589
- system_prompt_template=system_prompt_template,
590
- description=args.description,
637
+ **evaluator_kwargs,
591
638
  )
592
639
  print_json(_evaluator_summary(evaluator, full=True))
593
640
  return 0
codeer_cli/eval_.py CHANGED
@@ -12,6 +12,13 @@ from typing import Any, List, Optional
12
12
  from .client import CodeerClient
13
13
 
14
14
 
15
+ class _UnsetType:
16
+ pass
17
+
18
+
19
+ _UNSET = _UnsetType()
20
+
21
+
15
22
  # --- cases ----------------------------------------------------------------
16
23
 
17
24
  def create_case(
@@ -186,6 +193,7 @@ def create_evaluator(
186
193
  name: str,
187
194
  system_prompt_template: str,
188
195
  description: Optional[str] = None,
196
+ judge_llm_model_id: Optional[str] = None,
189
197
  ) -> dict:
190
198
  body: dict[str, Any] = {
191
199
  "name": name,
@@ -193,6 +201,8 @@ def create_evaluator(
193
201
  }
194
202
  if description is not None:
195
203
  body["description"] = description
204
+ if judge_llm_model_id is not None:
205
+ body["judge_llm_model_id"] = judge_llm_model_id
196
206
  return client.post("/external/eval/evaluators", json=body)
197
207
 
198
208
 
@@ -211,6 +221,7 @@ def update_evaluator(
211
221
  name: Optional[str] = None,
212
222
  system_prompt_template: Optional[str] = None,
213
223
  description: Optional[str] = None,
224
+ judge_llm_model_id: str | None | _UnsetType = _UNSET,
214
225
  ) -> dict:
215
226
  body: dict[str, Any] = {}
216
227
  if name is not None:
@@ -219,6 +230,8 @@ def update_evaluator(
219
230
  body["system_prompt_template"] = system_prompt_template
220
231
  if description is not None:
221
232
  body["description"] = description
233
+ if judge_llm_model_id is not _UNSET:
234
+ body["judge_llm_model_id"] = judge_llm_model_id
222
235
  return client.put(f"/external/eval/evaluators/{evaluator_id}", json=body)
223
236
 
224
237
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.5
2
2
  Name: codeer-cli
3
- Version: 0.1.13
3
+ Version: 0.1.14
4
4
  Summary: Command line tools for managing Codeer agents over the Codeer API.
5
5
  Project-URL: Homepage, https://www.codeer.ai
6
6
  Author: Codeer.AI
@@ -123,6 +123,34 @@ List the active cloud models without opening the Codeer web app:
123
123
  codeer model list --type text
124
124
  ```
125
125
 
126
+ ## Custom evaluator judge models
127
+
128
+ Custom evaluator create/update commands can select a judge LLM model by ID:
129
+
130
+ ```bash
131
+ codeer eval evaluator-create \
132
+ --name "Correctness" \
133
+ --system-prompt-template-file evaluator-prompt.txt \
134
+ --judge-model <model-id> \
135
+ --dry-run
136
+
137
+ codeer eval evaluator-update \
138
+ --evaluator <evaluator-id> \
139
+ --judge-model <model-id> \
140
+ --dry-run
141
+ ```
142
+
143
+ Omit the judge-model flags on update to leave the current setting unchanged.
144
+ Use `--clear-judge-model` to explicitly clear the override and return to the
145
+ system default:
146
+
147
+ ```bash
148
+ codeer eval evaluator-update \
149
+ --evaluator <evaluator-id> \
150
+ --clear-judge-model \
151
+ --dry-run
152
+ ```
153
+
126
154
  ## Agent human handoff
127
155
 
128
156
  `codeer agent apply` accepts the same `human_handoff` object as the Agent API.
@@ -5,7 +5,7 @@ codeer_cli/chats.py,sha256=Vg1DIu8fT6mPyPpfqLpExKvXhdZEWCyJGPvalQy3BXg,7989
5
5
  codeer_cli/cli.py,sha256=ghZMn7LUxDDvb3f82mDYW5hIFLk7dtlrCrGGP40Gygw,4637
6
6
  codeer_cli/client.py,sha256=apUycjK-OvHEquAj58FqOy8FHmHjA6K9dScaYQ4M-VU,13175
7
7
  codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
8
- codeer_cli/eval_.py,sha256=2Tz4qokctocZ0fqLYL8Sk9UV8yJ7p-9zYiUFljsRRCs,17478
8
+ codeer_cli/eval_.py,sha256=4lDaT-gf_nCatnRoHA0L9MDBxCDgT0ORBWC5ku04pts,17827
9
9
  codeer_cli/histories.py,sha256=WvpqvuprPc1nJJCTElyM_eoUabyooACfmRCwYlcvPnI,6637
10
10
  codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
11
11
  codeer_cli/models.py,sha256=Ama9fx8hyRMdysiUNA3Ty0sYIYwzFDkOcAZg-id3-vQ,362
@@ -14,12 +14,12 @@ codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSu
14
14
  codeer_cli/commands/_util.py,sha256=X9-9cYgnBYz94hVG4Pe9SsI97DHNv3HE5_7GlPMn93I,1708
15
15
  codeer_cli/commands/agent.py,sha256=BjctOF0tflt8yVGzdc6320sWXVUAYrF39tNIiTxtQx4,16319
16
16
  codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
17
- codeer_cli/commands/eval_cmd.py,sha256=Zlt_YIA3D5-Y8WL1HZ6PxZgL-TcwFTDoXvO2gv7yfOM,71989
17
+ codeer_cli/commands/eval_cmd.py,sha256=S-ydzWZppPLMCu2i_HOK0OW5ZaoebxcLO--WDYOIuAk,73548
18
18
  codeer_cli/commands/history.py,sha256=2p9VuPwyexr7Q2FFIBSuv9hNzn7VdIMa-AtvtP-8LvI,17253
19
19
  codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
20
20
  codeer_cli/commands/model.py,sha256=91LDo_YeHi3POZV7KTCkoCcgtK5p87LwrHX1lZgfCps,1698
21
21
  codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
22
- codeer_cli-0.1.13.dist-info/METADATA,sha256=zslhdIB39lnbLxfGOM9sv8zrZdzDNzi6xjZWCDLrPQU,7885
23
- codeer_cli-0.1.13.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
24
- codeer_cli-0.1.13.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
25
- codeer_cli-0.1.13.dist-info/RECORD,,
22
+ codeer_cli-0.1.14.dist-info/METADATA,sha256=Dyw4s9TmhHF1Jsn9X0NSN9H-qouENxBsxIMRC4d_gb4,8547
23
+ codeer_cli-0.1.14.dist-info/WHEEL,sha256=zOwg4jB6zX2kU910N-cMawjivD6tO8NEWvE12je1bVk,87
24
+ codeer_cli-0.1.14.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
25
+ codeer_cli-0.1.14.dist-info/RECORD,,