codee-agent 0.6.5__tar.gz → 0.6.7__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {codee_agent-0.6.5 → codee_agent-0.6.7}/PKG-INFO +1 -1
  2. {codee_agent-0.6.5 → codee_agent-0.6.7}/pyproject.toml +1 -1
  3. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/admin.py +149 -6
  4. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/admin_service.py +4 -3
  5. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/runs_db.py +17 -9
  6. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/test_runs_db.py +47 -17
  7. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_admin_service.py +7 -3
  8. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_abstract/provider.py +11 -0
  9. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_github_copilot/provider.py +8 -3
  10. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_github_copilot/test.py +29 -1
  11. {codee_agent-0.6.5 → codee_agent-0.6.7}/LICENSE +0 -0
  12. {codee_agent-0.6.5 → codee_agent-0.6.7}/README.md +0 -0
  13. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/.gitignore +0 -0
  14. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/__init__.py +0 -0
  15. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/admin_api.py +0 -0
  16. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/admin_cli.py +0 -0
  17. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/agent_cli.py +0 -0
  18. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/coding_agents.py +0 -0
  19. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/executor.py +0 -0
  20. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/init_cli.py +0 -0
  21. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/__init__.py +0 -0
  22. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/claude_key_rotation.py +0 -0
  23. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/cron_describe.py +0 -0
  24. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/mcp_config.py +0 -0
  25. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/test_claude_key_rotation.py +0 -0
  26. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/test_mcp_config.py +0 -0
  27. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/test_trigger_cron_skills.py +0 -0
  28. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/test_trigger_issue_skills.py +0 -0
  29. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/trigger_aws_sqs_skills.py +0 -0
  30. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/trigger_cron_skills.py +0 -0
  31. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/trigger_email_skills.py +0 -0
  32. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/lib/trigger_issue_skills.py +0 -0
  33. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/mail_server.py +0 -0
  34. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/setup_wizard.py +0 -0
  35. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/start_cli.py +0 -0
  36. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/tasks_providers.py +0 -0
  37. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/AGENTS.md +0 -0
  38. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/CLAUDE.md +0 -0
  39. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/aws-sqs-alarm-response/SKILL.md +0 -0
  40. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/cron-research-5xx-errors/SKILL.md +0 -0
  41. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-code-reviewer/SKILL.md +0 -0
  42. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-developer/SKILL.md +0 -0
  43. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-planner/SKILL.md +0 -0
  44. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-planner/assets/readme-template.md +0 -0
  45. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-qa/SKILL.md +0 -0
  46. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/story-security-reviewer/SKILL.md +0 -0
  47. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/task-code-reviewer/SKILL.md +0 -0
  48. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/task-developer/SKILL.md +0 -0
  49. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/task-qa/SKILL.md +0 -0
  50. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/templates/skills/task-security-reviewer/SKILL.md +0 -0
  51. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_admin_api.py +0 -0
  52. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_admin_cli.py +0 -0
  53. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_agent_cli.py +0 -0
  54. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_coding_agents.py +0 -0
  55. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_executor.py +0 -0
  56. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_init_cli.py +0 -0
  57. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_memory_index.py +0 -0
  58. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_setup_wizard.py +0 -0
  59. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/test_start_cli.py +0 -0
  60. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee/workflow_graph.py +0 -0
  61. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_admin/__init__.py +0 -0
  62. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_admin/codee_admin.py +0 -0
  63. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_abstract/__init__.py +0 -0
  64. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/__init__.py +0 -0
  65. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/account.py +0 -0
  66. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/credentials.py +0 -0
  67. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/oauth.py +0 -0
  68. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/provider.py +0 -0
  69. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/test.py +0 -0
  70. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_claude_code/usage.py +0 -0
  71. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_codex/__init__.py +0 -0
  72. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_codex/provider.py +0 -0
  73. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_codex/test.py +0 -0
  74. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_agent_github_copilot/__init__.py +0 -0
  75. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_database/__init__.py +0 -0
  76. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_database/claude_code_accounts.py +0 -0
  77. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_database/database.py +0 -0
  78. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_database/oauth_tokens.py +0 -0
  79. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_main_context/__init__.py +0 -0
  80. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_main_context/context.py +0 -0
  81. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_main_context/logging.py +0 -0
  82. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_main_context/test_context.py +0 -0
  83. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_main_context/test_logging.py +0 -0
  84. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_abstract/__init__.py +0 -0
  85. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_abstract/provider.py +0 -0
  86. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_azure_devops/__init__.py +0 -0
  87. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_azure_devops/oauth.py +0 -0
  88. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_azure_devops/provider.py +0 -0
  89. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_azure_devops/test.py +0 -0
  90. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_jira/__init__.py +0 -0
  91. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_jira/provider.py +0 -0
  92. {codee_agent-0.6.5 → codee_agent-0.6.7}/src/codee_tasks_jira/test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codee-agent
3
- Version: 0.6.5
3
+ Version: 0.6.7
4
4
  Summary: A virtual co-worker that picks up Jira and Azure DevOps tasks and solves them with coding agents.
5
5
  Keywords: ai,agent,jira,azure-devops,claude-code,github-copilot,codex,automation
6
6
  Author: Denis Kibalko
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "codee-agent"
3
- version = "0.6.5"
3
+ version = "0.6.7"
4
4
  description = "A virtual co-worker that picks up Jira and Azure DevOps tasks and solves them with coding agents."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -273,6 +273,7 @@ class RunRecord(BaseModel):
273
273
  message: str
274
274
  user_message: str
275
275
  response: str
276
+ debug_logs: str
276
277
  preview: str
277
278
  viewer_url: str
278
279
 
@@ -367,6 +368,10 @@ class AdminState(rx.State):
367
368
  agent_test_messages: list[ConversationMessage] = []
368
369
  agent_test_sending: bool = False
369
370
  agent_test_generation: int = 0
371
+ agent_test_model: str = ""
372
+ agent_test_models: list[ModelOption] = []
373
+ agent_test_model_query: str = ""
374
+ agent_test_models_loading: bool = False
370
375
  # Whether the executor runs Claude Code on the connected accounts instead
371
376
  # of leaving ~/.claude/.credentials.json alone.
372
377
  claude_code_rotate_keys: bool = False
@@ -475,6 +480,32 @@ class AdminState(rx.State):
475
480
  return ""
476
481
  return query
477
482
 
483
+ @rx.var
484
+ def filtered_agent_test_models(self) -> list[ModelOption]:
485
+ query = self.agent_test_model_query.strip().lower()
486
+ return [
487
+ model for model in self.agent_test_models
488
+ if not query or query in f"{model.name} {model.id}".lower()
489
+ ]
490
+
491
+ @rx.var
492
+ def agent_test_model_label(self) -> str:
493
+ if not self.agent_test_model:
494
+ return "Agent default"
495
+ for model in self.agent_test_models:
496
+ if model.id == self.agent_test_model:
497
+ return model.name
498
+ return self.agent_test_model
499
+
500
+ @rx.var
501
+ def custom_agent_test_model_query(self) -> str:
502
+ query = self.agent_test_model_query.strip()
503
+ if not query or any(
504
+ model.id == query for model in self.agent_test_models
505
+ ):
506
+ return ""
507
+ return query
508
+
478
509
  @rx.var
479
510
  def active_route(self) -> str:
480
511
  return self.router.url.path.rstrip("/") or "/"
@@ -826,6 +857,7 @@ class AdminState(rx.State):
826
857
  message=message,
827
858
  user_message=(run.get("user_message") or message).strip(),
828
859
  response=(run.get("response") or "").strip(),
860
+ debug_logs=run.get("debug_logs") or "",
829
861
  preview=preview[:120] + ("..." if len(preview) > 120 else ""),
830
862
  viewer_url=(SERVICE.session_viewer.format(session_id=run["session_id"])
831
863
  if SERVICE.session_viewer and run.get("session_id") else ""),
@@ -1527,13 +1559,17 @@ class AdminState(rx.State):
1527
1559
  error = self._persist_settings()
1528
1560
  return rx.toast.error(error) if error else rx.toast.success("Settings saved")
1529
1561
 
1530
- def open_agent_test(self) -> None:
1562
+ def open_agent_test(self) -> Any:
1531
1563
  self.agent_test_input = ""
1532
1564
  self.agent_test_session_id = ""
1533
1565
  self.agent_test_messages = []
1534
1566
  self.agent_test_sending = False
1535
1567
  self.agent_test_generation += 1
1568
+ self.agent_test_model = ""
1569
+ self.agent_test_models = []
1570
+ self.agent_test_model_query = ""
1536
1571
  self.agent_test_open = True
1572
+ return AdminState.load_agent_test_models
1537
1573
 
1538
1574
  def set_agent_test_open(self, open_: bool) -> None:
1539
1575
  self.agent_test_open = open_
@@ -1544,6 +1580,28 @@ class AdminState(rx.State):
1544
1580
  def set_agent_test_input(self, value: str) -> None:
1545
1581
  self.agent_test_input = value
1546
1582
 
1583
+ def set_agent_test_model_query(self, value: str) -> None:
1584
+ self.agent_test_model_query = value
1585
+
1586
+ def choose_agent_test_model(self, model_id: str) -> None:
1587
+ self.agent_test_model = model_id.strip()
1588
+ self.agent_test_model_query = ""
1589
+
1590
+ @rx.event(background=True)
1591
+ async def load_agent_test_models(self) -> None:
1592
+ async with self:
1593
+ agent = self.coding_agent
1594
+ self.agent_test_models_loading = True
1595
+ try:
1596
+ models = await asyncio.to_thread(SERVICE.list_agent_models, agent)
1597
+ except Exception:
1598
+ models = []
1599
+ async with self:
1600
+ if self.coding_agent != agent:
1601
+ return
1602
+ self.agent_test_models = [ModelOption(**model) for model in models]
1603
+ self.agent_test_models_loading = False
1604
+
1547
1605
  @rx.event(background=True)
1548
1606
  async def send_agent_test_message(self) -> Any:
1549
1607
  async with self:
@@ -1552,6 +1610,7 @@ class AdminState(rx.State):
1552
1610
  return
1553
1611
  agent = self.coding_agent
1554
1612
  session_id = self.agent_test_session_id
1613
+ model = self.agent_test_model
1555
1614
  generation = self.agent_test_generation
1556
1615
  self.agent_test_input = ""
1557
1616
  self.agent_test_sending = True
@@ -1565,6 +1624,7 @@ class AdminState(rx.State):
1565
1624
  agent,
1566
1625
  message,
1567
1626
  session_id,
1627
+ model,
1568
1628
  )
1569
1629
  except Exception as error:
1570
1630
  async with self:
@@ -2131,7 +2191,7 @@ def model_menu_item(button: rx.Component) -> rx.Component:
2131
2191
  return rx.popover.close(rx.flex(button, width="100%"), width="100%")
2132
2192
 
2133
2193
 
2134
- def model_option_row(option: ModelOption) -> rx.Component:
2194
+ def _model_option_row(option: ModelOption, choose_model: Any) -> rx.Component:
2135
2195
  """One row of the model picker: friendly name left, model code right."""
2136
2196
  return model_menu_item(
2137
2197
  rx.button(
@@ -2143,7 +2203,15 @@ def model_option_row(option: ModelOption) -> rx.Component:
2143
2203
  align="center", spacing="2", width="100%"),
2144
2204
  variant="ghost", color_scheme="gray", width="100%",
2145
2205
  justify_content="start", padding="0.45rem 0.6rem",
2146
- on_click=AdminState.choose_model(option.id)))
2206
+ on_click=choose_model(option.id)))
2207
+
2208
+
2209
+ def model_option_row(option: ModelOption) -> rx.Component:
2210
+ return _model_option_row(option, AdminState.choose_model)
2211
+
2212
+
2213
+ def agent_test_model_option_row(option: ModelOption) -> rx.Component:
2214
+ return _model_option_row(option, AdminState.choose_agent_test_model)
2147
2215
 
2148
2216
 
2149
2217
  def agent_picker() -> rx.Component:
@@ -2481,7 +2549,7 @@ def run_row(run: RunRecord) -> rx.Component:
2481
2549
  rx.cond(run.viewer_url != "", rx.link(rx.icon("external-link", size=16), href=run.viewer_url,
2482
2550
  is_external=True, aria_label="View session", color=ACCENT)),
2483
2551
  gap="1rem", align="start", width="100%"),
2484
- rx.cond((run.user_message != "") | (run.response != ""), rx.accordion.root(rx.accordion.item(
2552
+ rx.cond((run.user_message != "") | (run.response != "") | (run.debug_logs != ""), rx.accordion.root(rx.accordion.item(
2485
2553
  header="Run info", content=rx.vstack(
2486
2554
  rx.text("User message", font_weight="600"),
2487
2555
  rx.text(run.user_message, white_space="pre-wrap"),
@@ -2489,6 +2557,12 @@ def run_row(run: RunRecord) -> rx.Component:
2489
2557
  rx.text("LLM response", font_weight="600",
2490
2558
  margin_top="0.75rem"),
2491
2559
  rx.text(run.response, white_space="pre-wrap"))),
2560
+ rx.cond(run.debug_logs != "", rx.fragment(
2561
+ rx.text("Debug logs", font_weight="600",
2562
+ margin_top="0.75rem"),
2563
+ rx.text(run.debug_logs, white_space="pre-wrap",
2564
+ font_family="IBM Plex Mono, monospace",
2565
+ font_size="0.8rem"))),
2492
2566
  spacing="2", align="start", width="100%"), value=run.started_at),
2493
2567
  collapsible=True, width="100%")),
2494
2568
  padding="1rem", background=SURFACE, border=BORDER, width="100%")
@@ -3309,6 +3383,72 @@ def agent_test_dialog() -> rx.Component:
3309
3383
  rx.dialog.description(
3310
3384
  "Chat with the selected coding agent in the Codee project.",
3311
3385
  color=MUTED),
3386
+ field(
3387
+ "Model",
3388
+ rx.popover.root(
3389
+ rx.popover.trigger(
3390
+ rx.button(
3391
+ rx.hstack(
3392
+ rx.text(AdminState.agent_test_model_label),
3393
+ rx.spacer(),
3394
+ rx.icon("chevrons-up-down", size=14),
3395
+ align="center", width="100%"),
3396
+ variant="surface", color_scheme="gray",
3397
+ width="100%", type="button")),
3398
+ rx.popover.content(
3399
+ rx.vstack(
3400
+ rx.input(
3401
+ placeholder=(
3402
+ "Search models, or type a model code"),
3403
+ value=AdminState.agent_test_model_query,
3404
+ on_change=AdminState.set_agent_test_model_query,
3405
+ auto_focus=True, width="100%"),
3406
+ rx.cond(
3407
+ AdminState.custom_agent_test_model_query != "",
3408
+ model_menu_item(
3409
+ rx.button(
3410
+ rx.hstack(
3411
+ rx.icon("plus", size=14),
3412
+ rx.text("Use "),
3413
+ rx.code(
3414
+ AdminState.custom_agent_test_model_query),
3415
+ align="center", spacing="2"),
3416
+ variant="soft", width="100%",
3417
+ justify_content="start",
3418
+ padding="0.45rem 0.6rem",
3419
+ on_click=AdminState.choose_agent_test_model(
3420
+ AdminState.custom_agent_test_model_query)))),
3421
+ rx.scroll_area(
3422
+ rx.vstack(
3423
+ model_menu_item(
3424
+ rx.button(
3425
+ "Agent default", variant="ghost",
3426
+ color_scheme="gray", width="100%",
3427
+ justify_content="start",
3428
+ padding="0.45rem 0.6rem",
3429
+ on_click=AdminState.choose_agent_test_model(""))),
3430
+ rx.foreach(
3431
+ AdminState.filtered_agent_test_models,
3432
+ agent_test_model_option_row),
3433
+ rx.cond(
3434
+ AdminState.agent_test_models_loading,
3435
+ rx.text(
3436
+ "Loading models from the coding agent…",
3437
+ color=MUTED, font_size="0.8rem",
3438
+ padding="0.5rem")),
3439
+ spacing="1", width="100%"),
3440
+ type="auto", scrollbars="vertical",
3441
+ max_height="15rem", width="100%"),
3442
+ spacing="2", width="100%"),
3443
+ width="24rem", max_width="calc(100vw - 3rem)")),
3444
+ rx.text(
3445
+ rx.cond(
3446
+ AdminState.agent_test_model == "",
3447
+ "Runs on whatever that agent defaults to.",
3448
+ rx.fragment("Uses ",
3449
+ rx.code(AdminState.agent_test_model),
3450
+ " for this conversation.")),
3451
+ color=MUTED, font_size="0.82rem")),
3312
3452
  rx.scroll_area(
3313
3453
  rx.vstack(
3314
3454
  rx.cond(
@@ -3374,7 +3514,7 @@ def settings_page() -> rx.Component:
3374
3514
  rx.heading("Coding agent", size="4", margin_bottom="1rem"),
3375
3515
  rx.vstack(
3376
3516
  field("Default agent",
3377
- rx.hstack(
3517
+ rx.grid(
3378
3518
  rx.select(
3379
3519
  ["claude_code", "github_copilot", "codex"],
3380
3520
  value=AdminState.coding_agent,
@@ -3384,7 +3524,10 @@ def settings_page() -> rx.Component:
3384
3524
  "Test conversation", variant="outline",
3385
3525
  white_space="nowrap",
3386
3526
  on_click=AdminState.open_agent_test),
3387
- width="100%", spacing="3")),
3527
+ grid_template_columns=rx.breakpoints(
3528
+ initial="minmax(0, 1fr)",
3529
+ md="minmax(0, 1fr) auto"),
3530
+ width="100%", gap="0.75rem")),
3388
3531
  field("Max parallel tasks",
3389
3532
  rx.input(value=AdminState.max_parallel_agents,
3390
3533
  on_change=AdminState.set_max_parallel_agents,
@@ -2444,7 +2444,8 @@ class AdminService:
2444
2444
  raise RuntimeError(str(exc)) from exc
2445
2445
 
2446
2446
  def test_agent_conversation(
2447
- self, agent_code: str, user_message: str, session_id: str = ""
2447
+ self, agent_code: str, user_message: str, session_id: str = "",
2448
+ model: str = "",
2448
2449
  ) -> tuple[str, str]:
2449
2450
  """Run one turn with the selected agent and return reply plus session id."""
2450
2451
  selected = resolve_agent_code(agent_code)
@@ -2459,10 +2460,10 @@ class AdminService:
2459
2460
 
2460
2461
  if session_id:
2461
2462
  response = agent.continue_conversation(
2462
- user_message, session_id, on_session_id=opened)
2463
+ user_message, session_id, model, on_session_id=opened)
2463
2464
  else:
2464
2465
  response = agent.run(
2465
- user_message, actual_session_id, on_session_id=opened)
2466
+ user_message, actual_session_id, model, on_session_id=opened)
2466
2467
  return response, actual_session_id
2467
2468
 
2468
2469
  def setup_tasks_mcp(
@@ -11,7 +11,7 @@ from codee_main_context.context import CodeeMainContext
11
11
  from codee_database.database import get_db_connection
12
12
 
13
13
  _COLUMNS = ("id", "skill_name", "trigger_type", "session_id", "status", "error",
14
- "started_at", "message", "user_message", "response")
14
+ "started_at", "message", "user_message", "response", "debug_logs")
15
15
 
16
16
  # Codee names a session before the agent runs, but not every agent runs under
17
17
  # the name it was given: Codex mints its own thread id and reports it back mid
@@ -52,7 +52,8 @@ def init(main_context: CodeeMainContext) -> None:
52
52
  started_at TEXT NOT NULL,
53
53
  message TEXT,
54
54
  user_message TEXT,
55
- response TEXT
55
+ response TEXT,
56
+ debug_logs TEXT
56
57
  )"""
57
58
  )
58
59
  conn.execute(
@@ -65,6 +66,8 @@ def init(main_context: CodeeMainContext) -> None:
65
66
  conn.execute("ALTER TABLE runs ADD COLUMN response TEXT")
66
67
  if "user_message" not in cols:
67
68
  conn.execute("ALTER TABLE runs ADD COLUMN user_message TEXT")
69
+ if "debug_logs" not in cols:
70
+ conn.execute("ALTER TABLE runs ADD COLUMN debug_logs TEXT")
68
71
  # In-flight claude runs; a row lives only while its subprocess is running.
69
72
  conn.execute(
70
73
  """CREATE TABLE IF NOT EXISTS active_jobs (
@@ -77,28 +80,33 @@ def init(main_context: CodeeMainContext) -> None:
77
80
  )"""
78
81
  )
79
82
  # Migrate DBs created before the dashboard named the agent behind a run.
80
- job_cols = {row[1] for row in conn.execute("PRAGMA table_info(active_jobs)")}
83
+ job_cols = {row[1] for row in conn.execute(
84
+ "PRAGMA table_info(active_jobs)")}
81
85
  for column in ("agent", "model"):
82
86
  if column not in job_cols:
83
- conn.execute(f"ALTER TABLE active_jobs ADD COLUMN {column} TEXT")
87
+ conn.execute(
88
+ f"ALTER TABLE active_jobs ADD COLUMN {column} TEXT")
84
89
 
85
90
 
86
91
  def record_run(skill_name, trigger_type, session_id, status, error=None, started_at=None,
87
- message=None, user_message=None, response=None,
92
+ message=None, user_message=None, response=None, debug_logs=None,
88
93
  *, main_context: CodeeMainContext) -> None:
89
94
  """Insert one run row. Never raises to the caller (FR-009)."""
90
95
  try:
91
96
  init(main_context)
92
97
  session_id = agent_session(session_id)
98
+ if debug_logs is None:
99
+ debug_logs = getattr(response, "debug_logs", None)
93
100
  if started_at is None:
94
101
  started_at = datetime.now(timezone.utc).isoformat()
95
102
  with get_db_connection(main_context) as conn:
96
103
  conn.execute(
97
104
  "INSERT INTO runs (skill_name, trigger_type, session_id, status, error,"
98
- " started_at, message, user_message, response)"
99
- " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)",
105
+ " started_at, message, user_message, response, debug_logs)"
106
+ " VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?)",
100
107
  (skill_name, trigger_type, session_id,
101
- status, error, started_at, message, user_message, response),
108
+ status, error, started_at, message, user_message, response,
109
+ debug_logs),
102
110
  )
103
111
  except Exception as exc: # ponytail: a logging miss must never abort the skill run
104
112
  print(f"[runs_db] Failed to record run for {skill_name}: {exc}")
@@ -112,7 +120,7 @@ def recent_runs(limit: int = 100, offset: int = 0, *,
112
120
  with get_db_connection(main_context) as conn:
113
121
  rows = conn.execute(
114
122
  "SELECT id, skill_name, trigger_type, session_id, status, error,"
115
- " started_at, message, user_message, response"
123
+ " started_at, message, user_message, response, debug_logs"
116
124
  " FROM runs ORDER BY started_at DESC, id DESC LIMIT ? OFFSET ?",
117
125
  (limit, max(offset, 0)),
118
126
  ).fetchall()
@@ -30,7 +30,8 @@ def test_round_trip_newest_first(tmp_path):
30
30
  started_at="2026-06-27T11:00:00+00:00", main_context=ctx)
31
31
 
32
32
  rows = runs_db.recent_runs(main_context=ctx)
33
- assert [r["skill_name"] for r in rows] == ["skill-b", "skill-a"] # newest first
33
+ assert [r["skill_name"] for r in rows] == [
34
+ "skill-b", "skill-a"] # newest first
34
35
  assert rows[0]["trigger_type"] == "email"
35
36
  assert rows[0]["status"] == "failed"
36
37
  assert rows[0]["error"] == "boom"
@@ -74,7 +75,8 @@ def test_main_context_is_required():
74
75
  call()
75
76
  except TypeError:
76
77
  continue
77
- raise AssertionError("main_context must be a required keyword argument")
78
+ raise AssertionError(
79
+ "main_context must be a required keyword argument")
78
80
 
79
81
 
80
82
  # ---------------------------------------------------------------- counts() (US1)
@@ -85,9 +87,12 @@ def _ago(hours):
85
87
  def test_counts_total_and_last_24h(tmp_path):
86
88
  ctx = _ctx(tmp_path)
87
89
  # Two inside the 24h window, one well outside, one exactly at the boundary (excluded, strict >).
88
- runs_db.record_run("a", "cron", "s1", "succeeded", started_at=_ago(1), main_context=ctx)
89
- runs_db.record_run("b", "cron", "s2", "succeeded", started_at=_ago(23), main_context=ctx)
90
- runs_db.record_run("c", "cron", "s3", "succeeded", started_at=_ago(48), main_context=ctx)
90
+ runs_db.record_run("a", "cron", "s1", "succeeded",
91
+ started_at=_ago(1), main_context=ctx)
92
+ runs_db.record_run("b", "cron", "s2", "succeeded",
93
+ started_at=_ago(23), main_context=ctx)
94
+ runs_db.record_run("c", "cron", "s3", "succeeded",
95
+ started_at=_ago(48), main_context=ctx)
91
96
  assert runs_db.counts(ctx) == {"total": 3, "last_24h": 2}
92
97
 
93
98
 
@@ -101,17 +106,22 @@ def test_counts_never_raises_on_bad_path(tmp_path):
101
106
 
102
107
  def test_runs_by_hour_buckets(tmp_path):
103
108
  ctx = _ctx(tmp_path)
104
- runs_db.record_run("a", "cron", "s1", "succeeded", started_at=_ago(2), main_context=ctx)
105
- runs_db.record_run("b", "cron", "s2", "succeeded", started_at=_ago(2.1), main_context=ctx)
106
- runs_db.record_run("c", "cron", "s3", "succeeded", started_at=_ago(48), main_context=ctx) # out of window
109
+ runs_db.record_run("a", "cron", "s1", "succeeded",
110
+ started_at=_ago(2), main_context=ctx)
111
+ runs_db.record_run("b", "cron", "s2", "succeeded",
112
+ started_at=_ago(2.1), main_context=ctx)
113
+ runs_db.record_run("c", "cron", "s3", "succeeded",
114
+ started_at=_ago(48), main_context=ctx) # out of window
107
115
 
108
116
  hourly = runs_db.runs_by_hour(ctx)
109
117
  assert len(hourly) == 24 # always 24 buckets
110
118
  # oldest-first: buckets are the last 24 hour labels in ascending time order
111
119
  now = datetime.now(timezone.utc).replace(minute=0, second=0, microsecond=0)
112
- expected = [(now - timedelta(hours=h)).strftime("%H:00") for h in range(23, -1, -1)]
120
+ expected = [(now - timedelta(hours=h)).strftime("%H:00")
121
+ for h in range(23, -1, -1)]
113
122
  assert [h["hour"] for h in hourly] == expected
114
- assert sum(h["runs"] for h in hourly) == 2 # 48h-old run excluded; the two ~2h-old runs counted
123
+ # 48h-old run excluded; the two ~2h-old runs counted
124
+ assert sum(h["runs"] for h in hourly) == 2
115
125
 
116
126
 
117
127
  def test_runs_by_hour_empty_and_bad_path(tmp_path):
@@ -123,10 +133,14 @@ def test_runs_by_hour_empty_and_bad_path(tmp_path):
123
133
  # ---------------------------------------------------------------- message column (US2)
124
134
  def test_message_round_trip(tmp_path):
125
135
  ctx = _ctx(tmp_path)
126
- runs_db.record_run("a", "cron", "s1", "succeeded", message="hello", main_context=ctx)
127
- runs_db.record_run("b", "email", "s2", "succeeded", message="", main_context=ctx)
128
- runs_db.record_run("c", "aws-sqs", "s3", "succeeded", main_context=ctx) # message defaults None
129
- by_skill = {r["skill_name"]: r["message"] for r in runs_db.recent_runs(main_context=ctx)}
136
+ runs_db.record_run("a", "cron", "s1", "succeeded",
137
+ message="hello", main_context=ctx)
138
+ runs_db.record_run("b", "email", "s2", "succeeded",
139
+ message="", main_context=ctx)
140
+ runs_db.record_run("c", "aws-sqs", "s3", "succeeded",
141
+ main_context=ctx) # message defaults None
142
+ by_skill = {r["skill_name"]: r["message"]
143
+ for r in runs_db.recent_runs(main_context=ctx)}
130
144
  assert by_skill == {"a": "hello", "b": "", "c": None}
131
145
 
132
146
 
@@ -141,6 +155,19 @@ def test_response_round_trip(tmp_path):
141
155
  assert run["response"] == "final answer"
142
156
 
143
157
 
158
+ def test_debug_logs_are_extracted_from_the_agent_response(tmp_path):
159
+ from codee_agent_abstract.provider import AgentResponse
160
+
161
+ ctx = _ctx(tmp_path)
162
+ response = AgentResponse("final answer", "debug one\ndebug two\n")
163
+ runs_db.record_run("a", "cron", "s1", "succeeded", response=response,
164
+ main_context=ctx)
165
+
166
+ run, = runs_db.recent_runs(main_context=ctx)
167
+ assert run["response"] == "final answer"
168
+ assert run["debug_logs"] == "debug one\ndebug two\n"
169
+
170
+
144
171
  # ---------------------------------------------------------------- active_jobs
145
172
  def test_active_job_lifecycle(tmp_path):
146
173
  ctx = _ctx(tmp_path)
@@ -195,7 +222,8 @@ def test_active_jobs_elapsed_and_order(tmp_path):
195
222
  runs_db.start_job("young", "b", started_at=_ago(0.01), main_context=ctx)
196
223
  runs_db.start_job("old", "a", started_at=_ago(1), main_context=ctx)
197
224
  jobs = runs_db.active_jobs(ctx)
198
- assert [j["session_id"] for j in jobs] == ["young", "old"] # youngest first
225
+ assert [j["session_id"]
226
+ for j in jobs] == ["young", "old"] # youngest first
199
227
  assert jobs[1]["elapsed"] >= 3500 # ~1h old
200
228
 
201
229
 
@@ -262,7 +290,8 @@ def test_set_job_session_repoints_a_live_job(tmp_path):
262
290
 
263
291
  runs_db.set_job_session(job_id, "codex-thread", main_context=ctx)
264
292
 
265
- assert runs_db.active_jobs(main_context=ctx)[0]["session_id"] == "codex-thread"
293
+ assert runs_db.active_jobs(main_context=ctx)[
294
+ 0]["session_id"] == "codex-thread"
266
295
 
267
296
 
268
297
  def test_set_job_session_is_a_no_op_without_a_job(tmp_path):
@@ -280,7 +309,8 @@ def test_a_noted_agent_session_is_what_the_run_records(tmp_path):
280
309
  runs_db.record_run("skill-a", "issue", "codee-sid", "succeeded",
281
310
  main_context=ctx)
282
311
 
283
- assert runs_db.recent_runs(main_context=ctx)[0]["session_id"] == "codex-thread"
312
+ assert runs_db.recent_runs(main_context=ctx)[
313
+ 0]["session_id"] == "codex-thread"
284
314
 
285
315
 
286
316
  def test_the_note_is_consumed_so_a_later_run_keeps_its_own_id(tmp_path):
@@ -182,23 +182,27 @@ class TestAgentConversationTest(unittest.TestCase):
182
182
  agent.run.side_effect = run
183
183
 
184
184
  response, session_id = self.service.test_agent_conversation(
185
- "codex", "Hello")
185
+ "codex", "Hello", model="gpt-6-astra")
186
186
 
187
187
  self.assertEqual(response, "Final response")
188
188
  self.assertEqual(session_id, "agent-thread")
189
189
  build_agent.assert_called_once_with(
190
190
  self.service.context.settings, Path("/repo"), CodingAgent.CODEX)
191
+ self.assertEqual(agent.run.call_args.args[2], "gpt-6-astra")
191
192
 
192
193
  @patch("codee.admin_service.build_coding_agent")
193
194
  def test_later_turn_resumes_the_same_session(self, build_agent) -> None:
194
195
  build_agent.return_value.continue_conversation.return_value = "Still here"
195
196
 
196
197
  response, session_id = self.service.test_agent_conversation(
197
- "github_copilot", "What did I ask?", "same-thread")
198
+ "github_copilot", "What did I ask?", "same-thread",
199
+ "claude-opus-5")
198
200
 
199
201
  self.assertEqual((response, session_id), ("Still here", "same-thread"))
200
202
  call = build_agent.return_value.continue_conversation.call_args
201
- self.assertEqual(call.args[:2], ("What did I ask?", "same-thread"))
203
+ self.assertEqual(
204
+ call.args[:3],
205
+ ("What did I ask?", "same-thread", "claude-opus-5"))
202
206
 
203
207
 
204
208
  class AdminServiceWorkItemsTest(unittest.TestCase):
@@ -19,6 +19,17 @@ class AgentModel:
19
19
  name: str
20
20
 
21
21
 
22
+ class AgentResponse(str):
23
+ """Agent text with optional diagnostic output for the run record."""
24
+
25
+ debug_logs: str
26
+
27
+ def __new__(cls, response: str, debug_logs: str = "") -> "AgentResponse":
28
+ value = super().__new__(cls, response)
29
+ value.debug_logs = debug_logs
30
+ return value
31
+
32
+
22
33
  class AbstractCodingAgent(ABC):
23
34
  """Base class every coding agent (e.g. Claude Code) inherits from.
24
35
 
@@ -7,7 +7,7 @@ import time
7
7
  from collections.abc import Callable
8
8
  from pathlib import Path
9
9
 
10
- from codee_agent_abstract.provider import AbstractCodingAgent, AgentModel
10
+ from codee_agent_abstract.provider import AbstractCodingAgent, AgentModel, AgentResponse
11
11
  from codee_main_context.context import Settings
12
12
  from codee_main_context.logging import get_logger
13
13
 
@@ -30,6 +30,7 @@ _SESSION_NEW_ID = 2
30
30
  # Ceiling on what one run may spend. AI credits bill at $0.04 each, so the
31
31
  # default is the same $20 cap the Claude Code agent puts on a run.
32
32
  MAX_AI_CREDITS_ENV_VAR = "CODEE_COPILOT_MAX_AI_CREDITS"
33
+ COPILOT_DEBUG_ENV_VAR = "COPILOT_DEBUG"
33
34
  DEFAULT_MAX_AI_CREDITS = 1000
34
35
  # The CLI rejects anything lower outright, which reads as a broken agent rather
35
36
  # than a misconfigured cap, so a smaller override is raised to it instead.
@@ -119,6 +120,10 @@ class GitHubCopilotAgent(AbstractCodingAgent):
119
120
  # body alone — so a skill's model only takes effect via this flag.
120
121
  if model:
121
122
  cmd += ["--model", model]
123
+ debug_enabled = os.environ.get(
124
+ COPILOT_DEBUG_ENV_VAR, "").strip().lower() == "true"
125
+ if debug_enabled:
126
+ cmd += ["--log-level", "debug"]
122
127
 
123
128
  log.info("Running copilot with message: %s", user_message)
124
129
  log.debug("cwd=%s cmd=%s", self._cwd, " ".join(cmd))
@@ -156,7 +161,7 @@ class GitHubCopilotAgent(AbstractCodingAgent):
156
161
  # rather than failing a run that the CLI itself called successful.
157
162
  log.warning(
158
163
  "copilot produced no result event; returning raw output")
159
- return stdout
164
+ return AgentResponse(stdout, stderr if debug_enabled else "")
160
165
  if outcome.get("exitCode"):
161
166
  raise RuntimeError(
162
167
  f"Copilot run errored (exit code {outcome['exitCode']}): "
@@ -166,7 +171,7 @@ class GitHubCopilotAgent(AbstractCodingAgent):
166
171
  raise RuntimeError(
167
172
  f"Copilot run produced no response: {_detail(errors, stderr)}"
168
173
  )
169
- return reply
174
+ return AgentResponse(reply, stderr if debug_enabled else "")
170
175
 
171
176
 
172
177
  def _relative(path: Path, cwd: Path) -> str:
@@ -7,7 +7,8 @@ from pathlib import Path
7
7
  from unittest.mock import Mock, patch
8
8
 
9
9
  from codee_agent_github_copilot.provider import (
10
- MAX_AI_CREDITS_ENV_VAR, GitHubCopilotAgent, _await_result)
10
+ COPILOT_DEBUG_ENV_VAR, MAX_AI_CREDITS_ENV_VAR, GitHubCopilotAgent,
11
+ _await_result)
11
12
  from codee_main_context.context import Settings
12
13
 
13
14
  SESSION = "82232f47-df60-4cb3-8c3a-de12074c9205"
@@ -153,6 +154,26 @@ class CopilotRunTest(unittest.TestCase):
153
154
 
154
155
  self.assertNotIn("--model", self.captured.call_args.args[0])
155
156
 
157
+ def test_debug_env_adds_log_level_and_preserves_stderr(self) -> None:
158
+ stdout = _stream(_event("assistant.message", {"content": "ok"}),
159
+ _event("result", exitCode=0))
160
+ with patch.dict(os.environ, {COPILOT_DEBUG_ENV_VAR: "true"}):
161
+ response = self._run(_completed(stdout, stderr="debug line\n"))
162
+
163
+ cmd = self.captured.call_args.args[0]
164
+ self.assertEqual(cmd[cmd.index("--log-level") + 1], "debug")
165
+ self.assertEqual(response, "ok")
166
+ self.assertEqual(response.debug_logs, "debug line\n")
167
+
168
+ def test_debug_env_must_be_true(self) -> None:
169
+ stdout = _stream(_event("assistant.message", {"content": "ok"}),
170
+ _event("result", exitCode=0))
171
+ with patch.dict(os.environ, {COPILOT_DEBUG_ENV_VAR: "1"}):
172
+ response = self._run(_completed(stdout, stderr="not retained"))
173
+
174
+ self.assertNotIn("--log-level", self.captured.call_args.args[0])
175
+ self.assertEqual(response.debug_logs, "")
176
+
156
177
  def test_the_best_model_is_an_anthropic_catalog_id(self) -> None:
157
178
  # Copilot has no latest-tier alias, so this id is pinned by hand and only
158
179
  # a real catalog id will be accepted by the CLI.
@@ -188,6 +209,13 @@ class CopilotRunTest(unittest.TestCase):
188
209
  self.assertEqual(self._run(_completed("plain text reply\n")),
189
210
  "plain text reply\n")
190
211
 
212
+ def test_raw_output_keeps_debug_logs(self) -> None:
213
+ with patch.dict(os.environ, {COPILOT_DEBUG_ENV_VAR: "TRUE"}):
214
+ response = self._run(_completed("plain text reply\n", "debug\n"))
215
+
216
+ self.assertEqual(response, "plain text reply\n")
217
+ self.assertEqual(response.debug_logs, "debug\n")
218
+
191
219
  def test_a_timeout_raises(self) -> None:
192
220
  with patch("subprocess.run",
193
221
  side_effect=subprocess.TimeoutExpired(cmd="copilot", timeout=7200)):
File without changes
File without changes