codeer-cli 0.1.7__py3-none-any.whl → 0.1.9__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
codeer_cli/__init__.py CHANGED
@@ -19,7 +19,7 @@ workspace-local dotenv files or credential files.
19
19
  """
20
20
 
21
21
  from ._validate import ToolValidationError
22
- from .client import AuthError, CodeerClient, CodeerError
22
+ from .client import AuthError, CodeerClient, CodeerError, TransportError
23
23
  from .parse import (
24
24
  AgentSummary,
25
25
  ConversationTurn,
@@ -43,7 +43,7 @@ from .parse import (
43
43
  )
44
44
 
45
45
  __all__ = [
46
- "CodeerClient", "CodeerError", "AuthError", "ToolValidationError",
46
+ "CodeerClient", "CodeerError", "TransportError", "AuthError", "ToolValidationError",
47
47
  # parsers
48
48
  "AgentSummary", "ConversationTurn", "EvalResultSummary", "HistorySummary",
49
49
  "KBNode", "ToolCall", "EvalToolCall",
codeer_cli/chats.py CHANGED
@@ -36,6 +36,7 @@ def send_published_agent_message(
36
36
  external_user_id: Optional[str] = None,
37
37
  attachment_ids: Optional[List[str]] = None,
38
38
  stream: bool = False,
39
+ timeout: Optional[float] = None,
39
40
  ) -> Iterator[dict] | dict:
40
41
  """Send a user message through the API-key external chat flow.
41
42
 
@@ -51,7 +52,7 @@ def send_published_agent_message(
51
52
  path = f"/chats/{chat_id}/messages"
52
53
  if stream:
53
54
  return client.stream_sse("POST", path, json=body)
54
- return client.post(path, json=body)
55
+ return client.post(path, json=body, timeout=timeout)
55
56
 
56
57
 
57
58
  def send_message(
@@ -84,4 +85,3 @@ def list_messages(client: CodeerClient, chat_id: int) -> list[dict]:
84
85
 
85
86
  def list_chats(client: CodeerClient) -> list[dict]:
86
87
  return client.get("/chats")
87
-
codeer_cli/cli.py CHANGED
@@ -4,7 +4,7 @@
4
4
  codeer agent list|get|apply|diff|versions
5
5
  codeer kb list|files|upload|node-rename|node-delete|faq-list|faq-get|faq-create|faq-update|faq-delete
6
6
  codeer eval list|label-list|label-create|label-update|label-delete|case-update|case-delete|evaluators|evaluator-create|evaluator-update|run|export|reconcile|cases-apply|rubrics|rubrics-apply
7
- codeer history list|get|conversations|negative-feedback
7
+ codeer history list|get|conversations|negative-feedback|create|send
8
8
  """
9
9
 
10
10
  from __future__ import annotations
codeer_cli/client.py CHANGED
@@ -37,6 +37,16 @@ class ScopeResolutionError(CodeerError):
37
37
  """Raised when workspace or organization scope cannot be inferred."""
38
38
 
39
39
 
40
+ class TransportError(CodeerError):
41
+ """Raised when an HTTP request fails before a response is available."""
42
+
43
+ def __init__(self, message: str, body: Any = None):
44
+ RuntimeError.__init__(self, message)
45
+ self.status = 0
46
+ self.message = message
47
+ self.body = body
48
+
49
+
40
50
  @dataclass
41
51
  class CodeerClient:
42
52
  """Thin wrapper around httpx.Client with Codeer API-key auth.
@@ -148,9 +158,32 @@ class CodeerClient:
148
158
  json: Any = None,
149
159
  files: Any = None,
150
160
  data: Any = None,
161
+ timeout: Optional[float] = None,
151
162
  ) -> Any:
152
163
  url = path if path.startswith("http") else f"/api/v1{path if path.startswith('/') else '/' + path}"
153
- r = self._client.request(method, url, params=params, json=json, files=files, data=data)
164
+ request_kwargs: dict[str, Any] = {}
165
+ if timeout is not None:
166
+ request_kwargs["timeout"] = timeout
167
+ method_upper = method.upper()
168
+ try:
169
+ r = self._client.request(
170
+ method_upper,
171
+ url,
172
+ params=params,
173
+ json=json,
174
+ files=files,
175
+ data=data,
176
+ **request_kwargs,
177
+ )
178
+ except httpx.TimeoutException as exc:
179
+ raise self._transport_error(
180
+ method_upper,
181
+ path,
182
+ exc,
183
+ timeout_seconds=timeout if timeout is not None else self.timeout,
184
+ ) from exc
185
+ except httpx.RequestError as exc:
186
+ raise self._transport_error(method_upper, path, exc) from exc
154
187
  return self._parse(r)
155
188
 
156
189
  def get(self, path: str, **kwargs: Any) -> Any:
@@ -181,29 +214,68 @@ class CodeerClient:
181
214
  Each event is a dict like ``{"event": "message", "data": <parsed-json-or-str>}``.
182
215
  """
183
216
  url = path if path.startswith("http") else f"/api/v1{path if path.startswith('/') else '/' + path}"
184
- with self._client.stream(method, url, params=params, json=json) as r:
185
- if r.status_code >= 400:
186
- body = r.read().decode("utf-8", "replace")
187
- self._raise_for_error(r.status_code, body)
188
- event = "message"
189
- buf: list[str] = []
190
- for line in r.iter_lines():
191
- if line == "":
192
- if buf:
193
- raw = "\n".join(buf)
194
- yield {"event": event, "data": _maybe_json(raw)}
195
- buf = []
196
- event = "message"
197
- continue
198
- if line.startswith(":"):
199
- continue
200
- if line.startswith("event:"):
201
- event = line[len("event:"):].strip()
202
- continue
203
- if line.startswith("data:"):
204
- buf.append(line[len("data:"):].lstrip())
205
- if buf:
206
- yield {"event": event, "data": _maybe_json("\n".join(buf))}
217
+ method_upper = method.upper()
218
+ try:
219
+ with self._client.stream(method_upper, url, params=params, json=json) as r:
220
+ if r.status_code >= 400:
221
+ body = r.read().decode("utf-8", "replace")
222
+ self._raise_for_error(r.status_code, body)
223
+ event = "message"
224
+ buf: list[str] = []
225
+ for line in r.iter_lines():
226
+ if line == "":
227
+ if buf:
228
+ raw = "\n".join(buf)
229
+ yield {"event": event, "data": _maybe_json(raw)}
230
+ buf = []
231
+ event = "message"
232
+ continue
233
+ if line.startswith(":"):
234
+ continue
235
+ if line.startswith("event:"):
236
+ event = line[len("event:"):].strip()
237
+ continue
238
+ if line.startswith("data:"):
239
+ buf.append(line[len("data:"):].lstrip())
240
+ if buf:
241
+ yield {"event": event, "data": _maybe_json("\n".join(buf))}
242
+ except httpx.TimeoutException as exc:
243
+ raise self._transport_error(
244
+ method_upper,
245
+ path,
246
+ exc,
247
+ timeout_seconds=self.timeout,
248
+ ) from exc
249
+ except httpx.RequestError as exc:
250
+ raise self._transport_error(method_upper, path, exc) from exc
251
+
252
+ def _transport_error(
253
+ self,
254
+ method: str,
255
+ path: str,
256
+ exc: httpx.RequestError,
257
+ *,
258
+ timeout_seconds: float | None = None,
259
+ ) -> TransportError:
260
+ outcome_uncertain = (
261
+ method not in {"GET", "HEAD", "OPTIONS"}
262
+ and not isinstance(exc, httpx.ConnectError)
263
+ )
264
+ if isinstance(exc, httpx.TimeoutException):
265
+ timeout_value = timeout_seconds if timeout_seconds is not None else self.timeout
266
+ message = f"Request timed out after {timeout_value:g}s: {method} {path}."
267
+ if outcome_uncertain:
268
+ message += " The server may have completed the request; inspect current state before retrying."
269
+ else:
270
+ message = f"Request failed: {method} {path}: {exc}"
271
+ return TransportError(
272
+ message,
273
+ {
274
+ "method": method,
275
+ "path": path,
276
+ "outcome_uncertain": outcome_uncertain,
277
+ },
278
+ )
207
279
 
208
280
  def _parse(self, r: httpx.Response) -> Any:
209
281
  text = r.text
@@ -62,5 +62,7 @@ def print_json(value: Any) -> None:
62
62
  def write_json(path: str | None, value: Any) -> None:
63
63
  if not path:
64
64
  return
65
- Path(path).write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
65
+ out = Path(path)
66
+ out.parent.mkdir(parents=True, exist_ok=True)
67
+ out.write_text(json.dumps(value, ensure_ascii=False, indent=2, default=str) + "\n")
66
68
  log(f"wrote full detail to {path}")
@@ -48,14 +48,12 @@ def register(subparsers):
48
48
 
49
49
  # codeer eval label-list/create/update/delete
50
50
  p = sub.add_parser("label-list", help="List eval case labels in the workspace")
51
- p.add_argument("--workspace", default=None, help="Workspace UUID (default: active API-key workspace)")
52
51
  p.add_argument("--out", default=None)
53
52
  p.set_defaults(func=run_label_list)
54
53
 
55
54
  p = sub.add_parser("label-create", help="Create an eval case label; run --dry-run first")
56
55
  p.add_argument("--name", required=True)
57
56
  p.add_argument("--color", default=None, help="Hex color like #0969da (default: server default)")
58
- p.add_argument("--workspace", default=None, help="Workspace UUID (default: active API-key workspace)")
59
57
  p.add_argument("--dry-run", action="store_true")
60
58
  p.add_argument("--out", default=None)
61
59
  p.set_defaults(func=run_label_create)
@@ -288,9 +286,7 @@ def run_list(args, client) -> int:
288
286
  # eval case labels
289
287
  # ---------------------------------------------------------------------------
290
288
 
291
- def _workspace_arg_or_default(client, workspace_id: str | None) -> str:
292
- if workspace_id:
293
- return workspace_id
289
+ def _active_workspace(client) -> str:
294
290
  ws, _ = client.resolve_scope()
295
291
  return ws
296
292
 
@@ -305,8 +301,8 @@ def _label_summary(label: dict) -> dict:
305
301
 
306
302
 
307
303
  def run_label_list(args, client) -> int:
308
- workspace_id = _workspace_arg_or_default(client, args.workspace)
309
- labels = eval_mod.list_case_labels(client, workspace_id=workspace_id)
304
+ workspace_id = _active_workspace(client)
305
+ labels = eval_mod.list_case_labels(client)
310
306
  out = {
311
307
  "workspace_id": workspace_id,
312
308
  "label_count": len(labels),
@@ -318,13 +314,13 @@ def run_label_list(args, client) -> int:
318
314
 
319
315
 
320
316
  def run_label_create(args, client) -> int:
321
- workspace_id = _workspace_arg_or_default(client, args.workspace)
317
+ workspace_id = _active_workspace(client)
322
318
  if args.dry_run:
323
319
  out = {
324
320
  "dry_run": True,
325
321
  "operation": "label_create",
326
322
  "method": "POST",
327
- "path": f"/eval/workspaces/{workspace_id}/case-labels",
323
+ "path": "/external/eval/case-labels",
328
324
  "workspace_id": workspace_id,
329
325
  "name": args.name,
330
326
  "color": args.color,
@@ -335,9 +331,7 @@ def run_label_create(args, client) -> int:
335
331
  write_json(args.out, out)
336
332
  return 0
337
333
 
338
- label = eval_mod.create_case_label(
339
- client, workspace_id=workspace_id, name=args.name, color=args.color
340
- )
334
+ label = eval_mod.create_case_label(client, name=args.name, color=args.color)
341
335
  out = _label_summary(strip_noisy_fields(label))
342
336
  print_json(out)
343
337
  write_json(args.out, out)
@@ -354,7 +348,7 @@ def run_label_update(args, client) -> int:
354
348
  "dry_run": True,
355
349
  "operation": "label_update",
356
350
  "method": "PUT",
357
- "path": f"/eval/case-labels/{args.label_id}",
351
+ "path": f"/external/eval/case-labels/{args.label_id}",
358
352
  "label_id": args.label_id,
359
353
  "updates": {"name": args.name, "color": args.color},
360
354
  "would_write_server_state": True,
@@ -379,7 +373,7 @@ def run_label_delete(args, client) -> int:
379
373
  "dry_run": True,
380
374
  "operation": "label_delete",
381
375
  "method": "DELETE",
382
- "path": f"/eval/case-labels/{args.label_id}",
376
+ "path": f"/external/eval/case-labels/{args.label_id}",
383
377
  "label_id": args.label_id,
384
378
  "would_write_server_state": True,
385
379
  "next_step": "Review this summary, then rerun without --dry-run after approval.",
@@ -653,6 +647,64 @@ def _group_case_ids_by_evaluator(pairs: list[dict[str, str]]) -> dict[str, list[
653
647
  return dict(grouped)
654
648
 
655
649
 
650
+ def _pairs_from_rubric_batches(
651
+ client,
652
+ *,
653
+ case_ids: list[str],
654
+ evaluator_ids: list[str],
655
+ ) -> list[dict[str, str]]:
656
+ pairs: list[dict[str, str]] = []
657
+ for evaluator_id in evaluator_ids:
658
+ for row in eval_mod.get_rubrics_batch(client, case_ids=case_ids, evaluator_id=evaluator_id):
659
+ if row.get("rubric"):
660
+ case_id = row.get("case_id") or row.get("evaluation_case_id")
661
+ if case_id:
662
+ pairs.append({"case_id": str(case_id), "evaluator_id": evaluator_id})
663
+ return pairs
664
+
665
+
666
+ def _pair_key(pair: dict[str, str]) -> tuple[str, str]:
667
+ return pair["case_id"], pair["evaluator_id"]
668
+
669
+
670
+ def _skipped_pairs_from_trigger_response(response: Any) -> list[dict[str, str]]:
671
+ if not isinstance(response, dict):
672
+ return []
673
+ payload = response.get("data") if isinstance(response.get("data"), dict) else response
674
+ skipped = payload.get("skipped_pairs") if isinstance(payload, dict) else None
675
+ if not isinstance(skipped, list):
676
+ return []
677
+
678
+ out: list[dict[str, str]] = []
679
+ for row in skipped:
680
+ if not isinstance(row, dict):
681
+ continue
682
+ case_id = row.get("case_id")
683
+ evaluator_id = row.get("evaluator_id")
684
+ if not case_id or not evaluator_id:
685
+ continue
686
+ out.append({
687
+ "case_id": str(case_id),
688
+ "evaluator_id": str(evaluator_id),
689
+ "reason": str(row.get("reason") or "skipped"),
690
+ })
691
+ return out
692
+
693
+
694
+ def _remove_non_runnable_skipped_pairs(
695
+ pairs: list[dict[str, str]],
696
+ skipped_pairs: list[dict[str, str]],
697
+ ) -> list[dict[str, str]]:
698
+ non_runnable = {
699
+ _pair_key(pair)
700
+ for pair in skipped_pairs
701
+ if pair.get("reason") == "not_assigned"
702
+ }
703
+ if not non_runnable:
704
+ return pairs
705
+ return [pair for pair in pairs if _pair_key(pair) not in non_runnable]
706
+
707
+
656
708
  def run_run(args, client) -> int:
657
709
  workspace_id, _ = client.resolve_scope()
658
710
  if args.latest or not args.history:
@@ -682,18 +734,22 @@ def run_run(args, client) -> int:
682
734
  evaluator_ids = [args.evaluator] if args.evaluator else (_ids(args.evaluators) or [])
683
735
  requested_evaluator_ids = evaluator_ids or None
684
736
 
685
- assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
686
- assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
687
- pairs, skipped_unassigned = _planned_eval_pairs(
688
- case_ids=case_ids,
689
- assigned_by_case=assigned_by_case,
690
- requested_evaluator_ids=requested_evaluator_ids,
691
- )
737
+ skipped_unassigned: list[dict[str, str]] = []
738
+ if requested_evaluator_ids:
739
+ pairs = [
740
+ {"case_id": case_id, "evaluator_id": evaluator_id}
741
+ for evaluator_id in requested_evaluator_ids
742
+ for case_id in case_ids
743
+ ]
744
+ else:
745
+ evaluator_ids = [e["id"] for e in eval_mod.list_evaluators(client, workspace_id)]
746
+ pairs = _pairs_from_rubric_batches(
747
+ client,
748
+ case_ids=case_ids,
749
+ evaluator_ids=evaluator_ids,
750
+ )
692
751
  if not pairs:
693
- if skipped_unassigned:
694
- log("error: none of the requested case/evaluator pairs are assigned")
695
- else:
696
- log("error: no assigned case/evaluator pairs to run")
752
+ log("error: no case/evaluator pairs to run")
697
753
  print_json({
698
754
  "agent_id": args.agent,
699
755
  "history_id": args.history,
@@ -710,14 +766,47 @@ def run_run(args, client) -> int:
710
766
  case_label_by_id = {c["id"]: truncate(c.get("input") or "", 60) for c in case_objs}
711
767
  evaluator_name_by_id = {e["id"]: e.get("name", e["id"]) for e in evaluators}
712
768
 
769
+ requested_pairs = list(pairs)
770
+ requested_pair_count = len(requested_pairs)
771
+ log(f"triggering: {requested_pair_count} case/evaluator pairs on history {args.history}")
772
+ trigger_response: list[dict[str, Any]] = []
773
+ skipped_pairs: list[dict[str, str]] = []
774
+ for ev_id, ev_case_ids in _group_case_ids_by_evaluator(pairs).items():
775
+ response = eval_mod.trigger(
776
+ client,
777
+ case_ids=ev_case_ids,
778
+ evaluator_ids=[ev_id],
779
+ agent_history_id=args.history,
780
+ )
781
+ response_skipped = _skipped_pairs_from_trigger_response(response)
782
+ skipped_pairs.extend(response_skipped)
783
+ trigger_response.append({
784
+ "evaluator_id": ev_id,
785
+ "case_ids": ev_case_ids,
786
+ "response": response,
787
+ "skipped_pairs": response_skipped,
788
+ })
789
+
790
+ pairs = _remove_non_runnable_skipped_pairs(pairs, skipped_pairs)
791
+ skipped_unassigned = [pair for pair in skipped_pairs if pair.get("reason") == "not_assigned"]
713
792
  if skipped_unassigned:
714
- log(f"skipping {len(skipped_unassigned)} unassigned requested pairs")
715
- log(f"triggering: {len(pairs)} assigned case/evaluator pairs on history {args.history}")
716
- trigger_response = eval_mod.trigger_pairs(
717
- client,
718
- case_evaluator_pairs=pairs,
719
- agent_history_id=args.history,
720
- )
793
+ log(f"skipping {len(skipped_unassigned)} not-assigned pairs from polling")
794
+ if not pairs:
795
+ log("error: no runnable case/evaluator pairs after trigger response")
796
+ print_json({
797
+ "agent_id": args.agent,
798
+ "history_id": args.history,
799
+ "requested_case_count": len(case_ids),
800
+ "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
801
+ "requested_pair_count": requested_pair_count,
802
+ "triggered_pair_count": 0,
803
+ "skipped_pair_count": len(skipped_pairs),
804
+ "skipped_unassigned_count": len(skipped_unassigned),
805
+ "trigger_response": trigger_response,
806
+ "skipped_pairs": skipped_pairs,
807
+ "skipped_unassigned": skipped_unassigned,
808
+ })
809
+ return 2
721
810
 
722
811
  deadline = time.time() + args.poll_timeout
723
812
  results_by_eval: dict[str, list[dict]] = {}
@@ -825,22 +914,27 @@ def run_run(args, client) -> int:
825
914
  "history_id": args.history,
826
915
  "requested_case_count": len(case_ids),
827
916
  "requested_evaluator_count": len(requested_evaluator_ids or evaluator_ids),
917
+ "requested_pair_count": requested_pair_count,
828
918
  "triggered_pair_count": len(pairs),
829
919
  "scored_pair_count": len(scored_pair_keys),
920
+ "skipped_pair_count": len(skipped_pairs),
830
921
  "skipped_unassigned_count": len(skipped_unassigned),
831
922
  "all_perfect": all_perfect,
832
923
  "result_count": len(result_summaries),
833
924
  "non_perfect_count": len(non_perfect),
834
925
  "wrote_full_detail": bool(args.out),
835
926
  "trigger_response": trigger_response,
927
+ "skipped_pairs": skipped_pairs,
836
928
  "skipped_unassigned": skipped_unassigned,
837
929
  "results": result_summaries,
838
930
  }
839
931
  full_out = {
840
932
  "agent_id": args.agent,
841
933
  "history_id": args.history,
934
+ "requested_pairs": requested_pairs,
842
935
  "triggered_pairs": pairs,
843
936
  "trigger_response": trigger_response,
937
+ "skipped_pairs": skipped_pairs,
844
938
  "skipped_unassigned": skipped_unassigned,
845
939
  "all_perfect": all_perfect,
846
940
  "results": flat,
@@ -1284,7 +1378,7 @@ def run_cases_apply(args, client) -> int:
1284
1378
  if manifest_label_names:
1285
1379
  labels_by_name = {
1286
1380
  (label.get("name") or "").casefold(): label
1287
- for label in eval_mod.list_case_labels(client, workspace_id=workspace_id)
1381
+ for label in eval_mod.list_case_labels(client)
1288
1382
  if label.get("name")
1289
1383
  }
1290
1384
  missing_label_names = [
@@ -1310,7 +1404,7 @@ def run_cases_apply(args, client) -> int:
1310
1404
  else:
1311
1405
  for name in missing_label_names:
1312
1406
  log(f"creating label: {name}")
1313
- label = eval_mod.create_case_label(client, workspace_id=workspace_id, name=name)
1407
+ label = eval_mod.create_case_label(client, name=name)
1314
1408
  labels_by_name[name.casefold()] = label
1315
1409
  created_labels.append(_label_summary(label))
1316
1410
 
@@ -1466,35 +1560,44 @@ def run_rubrics(args, client) -> int:
1466
1560
  log("error: no cases for this agent")
1467
1561
  return 2
1468
1562
 
1469
- assignment_rows = eval_mod.get_case_evaluator_infos(client, case_ids=case_ids)
1470
- assigned_by_case = _assigned_evaluators_by_case(assignment_rows)
1471
-
1472
1563
  if args.evaluators:
1473
1564
  evaluator_ids = _ids(args.evaluators) or []
1474
1565
  evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
1475
- elif args.all_pairs:
1566
+ else:
1476
1567
  evaluators = eval_mod.list_evaluators(client, workspace_id)
1477
1568
  evaluator_ids = [e["id"] for e in evaluators]
1569
+
1570
+ rubrics = eval_mod.get_case_rubrics(
1571
+ client, agent_id=args.agent, workspace_id=workspace_id,
1572
+ evaluator_ids=evaluator_ids, case_ids=case_ids,
1573
+ )
1574
+ assigned_by_case = {
1575
+ cid: {
1576
+ ev_id: {"evaluator_id": ev_id, "rubric": rubric_text}
1577
+ for ev_id, rubric_text in (rubrics.get(cid) or {}).items()
1578
+ if rubric_text
1579
+ }
1580
+ for cid in case_ids
1581
+ }
1582
+
1583
+ if args.evaluators or args.all_pairs:
1584
+ pass
1478
1585
  else:
1479
1586
  evaluator_ids = _dedupe_preserve_order([
1480
1587
  evaluator_id
1481
1588
  for case_id in case_ids
1482
1589
  for evaluator_id in assigned_by_case.get(case_id, {})
1483
1590
  ])
1484
- evaluators = [eval_mod.get_evaluator(client, eid) for eid in evaluator_ids]
1591
+ evaluator_id_set = set(evaluator_ids)
1592
+ evaluators = [e for e in evaluators if e["id"] in evaluator_id_set]
1485
1593
  evaluator_name = {e["id"]: e.get("name", e["id"]) for e in evaluators}
1486
1594
  if not evaluator_ids:
1487
- log("error: no assigned evaluators for these cases")
1595
+ log("error: no evaluators with configured rubrics for these cases")
1488
1596
  return 2
1489
1597
 
1490
- mode = "all requested pairs" if args.evaluators or args.all_pairs else "assigned pairs"
1598
+ mode = "all requested pairs" if args.evaluators or args.all_pairs else "pairs with configured rubrics"
1491
1599
  log(f"reading {mode}: {len(case_ids)} cases, {len(evaluator_ids)} evaluators...")
1492
1600
 
1493
- rubrics = eval_mod.get_case_rubrics(
1494
- client, agent_id=args.agent, workspace_id=workspace_id,
1495
- evaluator_ids=evaluator_ids, case_ids=case_ids,
1496
- )
1497
-
1498
1601
  if args.full:
1499
1602
  for cid in case_ids:
1500
1603
  log("=" * 80)
@@ -1,10 +1,12 @@
1
1
  from __future__ import annotations
2
2
 
3
+ import math
3
4
  import os
4
5
 
5
6
  from .. import agents as agents_mod
6
7
  from .. import chats as chats_mod
7
8
  from .. import histories as hist_mod
9
+ from ..client import TransportError
8
10
  from ._util import log, print_json, strip_noisy_fields, truncate, write_json
9
11
 
10
12
 
@@ -74,10 +76,27 @@ def register(subparsers):
74
76
  p.add_argument("--user", default=None, help="external_user_id to associate with the history")
75
77
  p.add_argument("--message", action="append", required=True,
76
78
  help="User message to send. Repeat for multi-turn histories.")
79
+ p.add_argument("--timeout", type=float, default=120.0,
80
+ help="Per-message response timeout in seconds (default: 120).")
77
81
  p.add_argument("--out", default=None,
78
82
  help="Write complete create response/conversation artifact to this file; stdout stays compact.")
79
83
  p.set_defaults(func=run_create)
80
84
 
85
+ # codeer history send <id> --message ...
86
+ p = sub.add_parser("send", help="Continue an existing persisted conversation history")
87
+ p.add_argument("history_id", type=int)
88
+ p.add_argument("--agent", default=None,
89
+ help="Agent ID override (defaults to the history's agent or CODEER_AGENT_ID)")
90
+ p.add_argument("--user", default=None,
91
+ help="external_user_id override (defaults to the history's user)")
92
+ p.add_argument("--message", action="append", required=True,
93
+ help="User message to send. Repeat to append multiple turns.")
94
+ p.add_argument("--timeout", type=float, default=120.0,
95
+ help="Per-message response timeout in seconds (default: 120).")
96
+ p.add_argument("--out", default=None,
97
+ help="Write complete send response/conversation artifact to this file; stdout stays compact.")
98
+ p.set_defaults(func=run_send)
99
+
81
100
 
82
101
  def _parse_exclude(raw: str | None) -> list[str]:
83
102
  if not raw:
@@ -253,11 +272,51 @@ def _history_url(client, workspace_id: str, history_id: int) -> str:
253
272
  return f"{base}/workspaces/{workspace_id}/histories/{history_id}"
254
273
 
255
274
 
275
+ def _send_messages(
276
+ client,
277
+ *,
278
+ history_id: int,
279
+ agent_id: str,
280
+ external_user_id: str | None,
281
+ messages: list[str],
282
+ timeout: float,
283
+ ) -> list[dict]:
284
+ results = []
285
+ for idx, message in enumerate(messages, 1):
286
+ log(f"sending turn {idx}/{len(messages)}")
287
+ try:
288
+ result = chats_mod.send_published_agent_message(
289
+ client,
290
+ chat_id=history_id,
291
+ message=message,
292
+ agent_id=agent_id,
293
+ external_user_id=external_user_id,
294
+ stream=False,
295
+ timeout=timeout,
296
+ )
297
+ except TransportError as exc:
298
+ body = dict(exc.body) if isinstance(exc.body, dict) else {}
299
+ body.update({"history_id": history_id, "turn": idx, "turn_count": len(messages)})
300
+ raise TransportError(
301
+ f"Failed while sending turn {idx}/{len(messages)} to history {history_id}: {exc.message}",
302
+ body,
303
+ ) from exc
304
+ results.append(result)
305
+ return results
306
+
307
+
308
+ def _valid_timeout(timeout: float) -> bool:
309
+ return math.isfinite(timeout) and timeout > 0
310
+
311
+
256
312
  def run_create(args, client) -> int:
257
313
  agent_id = args.agent or os.environ.get("CODEER_AGENT_ID")
258
314
  if not agent_id:
259
315
  log("error: --agent is required or set CODEER_AGENT_ID")
260
316
  return 2
317
+ if not _valid_timeout(args.timeout):
318
+ log("error: --timeout must be a finite number greater than zero")
319
+ return 2
261
320
 
262
321
  workspace_id, _ = client.resolve_scope()
263
322
  title = args.title or (args.message[0].strip()[:80] if args.message else "CLI conversation")
@@ -269,19 +328,14 @@ def run_create(args, client) -> int:
269
328
  external_user_id=args.user,
270
329
  )
271
330
  history_id = chat["id"]
272
- message_results = []
273
-
274
- for idx, message in enumerate(args.message, 1):
275
- log(f"sending turn {idx}/{len(args.message)}")
276
- result = chats_mod.send_published_agent_message(
277
- client,
278
- chat_id=history_id,
279
- message=message,
280
- agent_id=agent_id,
281
- external_user_id=args.user,
282
- stream=False,
283
- )
284
- message_results.append(result)
331
+ message_results = _send_messages(
332
+ client,
333
+ history_id=history_id,
334
+ agent_id=agent_id,
335
+ external_user_id=args.user,
336
+ messages=args.message,
337
+ timeout=args.timeout,
338
+ )
285
339
 
286
340
  conversations = hist_mod.get_conversations(client, history_id)
287
341
  out = {
@@ -303,3 +357,61 @@ def run_create(args, client) -> int:
303
357
  "wrote_full_detail": bool(args.out),
304
358
  })
305
359
  return 0
360
+
361
+
362
+ def run_send(args, client) -> int:
363
+ if not _valid_timeout(args.timeout):
364
+ log("error: --timeout must be a finite number greater than zero")
365
+ return 2
366
+
367
+ history = hist_mod.get(client, args.history_id)
368
+ history_agent = history.get("agent") or {}
369
+ history_meta = history.get("meta") or {}
370
+ agent_id = (
371
+ args.agent
372
+ or history.get("agent_id")
373
+ or history_agent.get("id")
374
+ or history_meta.get("conversation_agent_id")
375
+ or os.environ.get("CODEER_AGENT_ID")
376
+ )
377
+ if not agent_id:
378
+ log("error: could not resolve agent from history; pass --agent or set CODEER_AGENT_ID")
379
+ return 2
380
+
381
+ external_user_id = (
382
+ args.user
383
+ if args.user is not None
384
+ else (
385
+ history.get("external_user_id")
386
+ or history_meta.get("external_user_id")
387
+ )
388
+ )
389
+ workspace_id, _ = client.resolve_scope()
390
+ message_results = _send_messages(
391
+ client,
392
+ history_id=args.history_id,
393
+ agent_id=agent_id,
394
+ external_user_id=external_user_id,
395
+ messages=args.message,
396
+ timeout=args.timeout,
397
+ )
398
+ conversations = hist_mod.get_conversations(client, args.history_id)
399
+ out = {
400
+ "agent_id": agent_id,
401
+ "history_id": args.history_id,
402
+ "external_user_id": external_user_id,
403
+ "url": _history_url(client, workspace_id, args.history_id),
404
+ "messages": message_results,
405
+ "conversations": conversations,
406
+ }
407
+ write_json(args.out, strip_noisy_fields(out))
408
+ print_json({
409
+ "agent_id": agent_id,
410
+ "history_id": args.history_id,
411
+ "external_user_id": external_user_id,
412
+ "url": out["url"],
413
+ "message_count": len(message_results),
414
+ "turn_count": len(conversations),
415
+ "wrote_full_detail": bool(args.out),
416
+ })
417
+ return 0
codeer_cli/eval_.py CHANGED
@@ -54,8 +54,26 @@ def create_case(
54
54
  return client.post("/external/eval/cases", json=body)
55
55
 
56
56
 
57
+ def _unwrap_list_response(value: Any, *keys: str) -> list[dict]:
58
+ """Normalize list endpoints that may return either a bare list or envelope."""
59
+ if isinstance(value, list):
60
+ return value
61
+ if isinstance(value, dict):
62
+ for key in keys:
63
+ rows = value.get(key)
64
+ if isinstance(rows, list):
65
+ return rows
66
+ return []
67
+
68
+
57
69
  def list_cases(client: CodeerClient, agent_id: str) -> list[dict]:
58
- return client.get(f"/external/eval/agents/{agent_id}/cases")
70
+ return _unwrap_list_response(
71
+ client.get(f"/external/eval/agents/{agent_id}/cases"),
72
+ "cases",
73
+ "evaluation_cases",
74
+ "data",
75
+ "items",
76
+ )
59
77
 
60
78
 
61
79
  def get_case(client: CodeerClient, case_id: str) -> dict:
@@ -123,21 +141,20 @@ def replace_case_evaluator_infos(
123
141
 
124
142
  # --- case labels -----------------------------------------------------------
125
143
 
126
- def list_case_labels(client: CodeerClient, *, workspace_id: str) -> list[dict]:
127
- return client.get(f"/eval/workspaces/{workspace_id}/case-labels")
144
+ def list_case_labels(client: CodeerClient) -> list[dict]:
145
+ return client.get("/external/eval/case-labels")
128
146
 
129
147
 
130
148
  def create_case_label(
131
149
  client: CodeerClient,
132
150
  *,
133
- workspace_id: str,
134
151
  name: str,
135
152
  color: Optional[str] = None,
136
153
  ) -> dict:
137
154
  body: dict[str, Any] = {"name": name}
138
155
  if color is not None:
139
156
  body["color"] = color
140
- return client.post(f"/eval/workspaces/{workspace_id}/case-labels", json=body)
157
+ return client.post("/external/eval/case-labels", json=body)
141
158
 
142
159
 
143
160
  def update_case_label(
@@ -152,11 +169,11 @@ def update_case_label(
152
169
  body["name"] = name
153
170
  if color is not None:
154
171
  body["color"] = color
155
- return client.put(f"/eval/case-labels/{label_id}", json=body)
172
+ return client.put(f"/external/eval/case-labels/{label_id}", json=body)
156
173
 
157
174
 
158
175
  def delete_case_label(client: CodeerClient, *, label_id: str) -> dict:
159
- return client.delete(f"/eval/case-labels/{label_id}")
176
+ return client.delete(f"/external/eval/case-labels/{label_id}")
160
177
 
161
178
 
162
179
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codeer-cli
3
- Version: 0.1.7
3
+ Version: 0.1.9
4
4
  Summary: Command line tools for managing Codeer agents over the Codeer API.
5
5
  Project-URL: Homepage, https://www.codeer.ai
6
6
  Author: Codeer.AI
@@ -143,9 +143,20 @@ Use this pattern during agent lifecycle work:
143
143
  ```bash
144
144
  codeer agent list
145
145
  codeer history list --agent <agent-id> --limit 50
146
+ codeer history create --agent <agent-id> --message "Review this plan" --timeout 120
147
+ codeer history send <history-id> --message "Use the recommended options" --timeout 120
146
148
  codeer eval run --agent <agent-id> --cases <case-ids> --evaluator <evaluator-id> --out .codeer/eval_run.json
147
149
  ```
148
150
 
151
+ `history create` and `history send` use the agent's current published version.
152
+ Their per-message timeout defaults to 120 seconds. If a write request times
153
+ out, inspect the history before retrying: the server may have completed the
154
+ turn after the client stopped waiting.
155
+
156
+ Eval case label commands always operate on the active API-key workspace. They
157
+ do not accept a workspace override; switch CLI profiles to target another
158
+ workspace.
159
+
149
160
  Flags:
150
161
 
151
162
  - `--full` prints bounded extra detail for human inspection. It is still
@@ -1,23 +1,23 @@
1
- codeer_cli/__init__.py,sha256=-0gL8upoSsLAnXAfcRrwqZYJbwG0knzQoFf94O7Nc7c,1817
1
+ codeer_cli/__init__.py,sha256=slzDQ-Le6vl5PrimV6ZKmcDcNsQ7my7yG-Is1JVlbi4,1851
2
2
  codeer_cli/_validate.py,sha256=pKUJa2TyTpERx5xmiYNZRn7tFqDxLZ2fF1rHAf1oz14,5415
3
3
  codeer_cli/agents.py,sha256=diodgiGhXlowEi8sbCzcSK1qSeCLF2fBe6QBs3Sq_x8,5617
4
- codeer_cli/chats.py,sha256=YVrZJhoa-d67o6tzX6riGXsbA-ehyhOxrZ8zRCcJNro,2675
5
- codeer_cli/cli.py,sha256=g-WR2D5MkaUdc13ZrpRCavXD1940CHE9eBELC034tic,4443
6
- codeer_cli/client.py,sha256=LpHVqf1IYNg1wFfIHnO9q4xg2h3IiGOitzCnvwB-Bcw,9809
4
+ codeer_cli/chats.py,sha256=pww_AVByTh_AT2Td0gj2OomJrcw7D-cjDIpQ0ZQOlrM,2728
5
+ codeer_cli/cli.py,sha256=tdHbAEqtfCWFVT1iecfWKh61P1IvjG59F9hfC4Srr0c,4455
6
+ codeer_cli/client.py,sha256=x-5sx06dgS6P8siKOTCVNM8JlBST5e6PsOD83KKkpN8,12361
7
7
  codeer_cli/constants.py,sha256=D1pV3wCoqYybrKGKeoupYjjFWLfaFviKp1yL7oh6Qso,2323
8
- codeer_cli/eval_.py,sha256=XwmPxNOtxSyZq2EOae9FZe5NFp0YyVbIswTn22S8noc,17050
8
+ codeer_cli/eval_.py,sha256=2Tz4qokctocZ0fqLYL8Sk9UV8yJ7p-9zYiUFljsRRCs,17478
9
9
  codeer_cli/histories.py,sha256=tk28git_peX4x703CIDU8u72JtlGaytyrtlHfxlK-7A,5979
10
10
  codeer_cli/kb.py,sha256=Ad4h65NByq5Rq5BTeMghLTKlWRhmOC2jxL0BaTGX3EM,10631
11
11
  codeer_cli/parse.py,sha256=qrjZn0MUTjGfucp4cwxy8Pt7WS-0x15kK5F7kWTY8Ps,21818
12
12
  codeer_cli/commands/__init__.py,sha256=47DEQpj8HBSa-_TImW-5JCeuQeRkm5NMpJWZG3hSuFU,0
13
- codeer_cli/commands/_util.py,sha256=VOB_HMWYzHFNY1ElLOED1HB6fpFpsniqH6Yx3VUlMrY,1644
13
+ codeer_cli/commands/_util.py,sha256=X9-9cYgnBYz94hVG4Pe9SsI97DHNv3HE5_7GlPMn93I,1708
14
14
  codeer_cli/commands/agent.py,sha256=amvfVVrbPOkbYKCGvA6EJB30C-aY6WSdfR7u7FklXbs,14793
15
15
  codeer_cli/commands/check.py,sha256=lTxolx1mIJ8jldPhJ5FXqie9nbCLVOO-sDPOHTSy1-w,3817
16
- codeer_cli/commands/eval_cmd.py,sha256=fQu8ZRzGO7GWL_Og9NeZ0xYwHoN_iocmCnWw-kPNRkc,68599
17
- codeer_cli/commands/history.py,sha256=Jv7t0GhSZcbZ8OuIXZT34CixXt7ECEVP3nZ-WW_Ya9E,12026
16
+ codeer_cli/commands/eval_cmd.py,sha256=Zlt_YIA3D5-Y8WL1HZ6PxZgL-TcwFTDoXvO2gv7yfOM,71989
17
+ codeer_cli/commands/history.py,sha256=bVBlCMStTvTxG23bggASHkHKuI0RMPdiXjcrs3U04yw,16081
18
18
  codeer_cli/commands/kb.py,sha256=kVEinBVM6NN8_0djOqIQErh46dLmArwFngXMNzvGeAI,28345
19
19
  codeer_cli/commands/profile.py,sha256=IdlXC_6cqobsfN3JRrAnt-1OgBUsIFneS9QtR4Un6Kc,6521
20
- codeer_cli-0.1.7.dist-info/METADATA,sha256=z8yvV20PQ1ysK27LpsMuurXdAPWJ4Kz-o9-LuxrHczg,5999
21
- codeer_cli-0.1.7.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
- codeer_cli-0.1.7.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
- codeer_cli-0.1.7.dist-info/RECORD,,
20
+ codeer_cli-0.1.9.dist-info/METADATA,sha256=XlAf4gazUUSZ8uk9YRBxOYCiJ-C13Tru02G548cqI5M,6605
21
+ codeer_cli-0.1.9.dist-info/WHEEL,sha256=lCkmxWfQsSc9CfIClYeavTdQeEX2toPqufh9gI35EQA,87
22
+ codeer_cli-0.1.9.dist-info/entry_points.txt,sha256=-nXIrlm5SR5r7gg3y8AS0tN66MwmvNHsrlwLNQNGD50,47
23
+ codeer_cli-0.1.9.dist-info/RECORD,,