codee-agent 0.6.4__tar.gz → 0.6.6__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (92) hide show
  1. {codee_agent-0.6.4 → codee_agent-0.6.6}/PKG-INFO +1 -1
  2. {codee_agent-0.6.4 → codee_agent-0.6.6}/pyproject.toml +1 -1
  3. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/admin.py +280 -11
  4. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/admin_service.py +33 -5
  5. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_admin_service.py +64 -13
  6. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_abstract/provider.py +15 -0
  7. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/provider.py +17 -2
  8. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/test.py +12 -2
  9. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_codex/provider.py +30 -3
  10. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_codex/test.py +9 -0
  11. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_github_copilot/test.py +25 -8
  12. {codee_agent-0.6.4 → codee_agent-0.6.6}/LICENSE +0 -0
  13. {codee_agent-0.6.4 → codee_agent-0.6.6}/README.md +0 -0
  14. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/.gitignore +0 -0
  15. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/__init__.py +0 -0
  16. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/admin_api.py +0 -0
  17. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/admin_cli.py +0 -0
  18. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/agent_cli.py +0 -0
  19. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/coding_agents.py +0 -0
  20. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/executor.py +0 -0
  21. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/init_cli.py +0 -0
  22. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/__init__.py +0 -0
  23. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/claude_key_rotation.py +0 -0
  24. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/cron_describe.py +0 -0
  25. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/mcp_config.py +0 -0
  26. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/runs_db.py +0 -0
  27. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/test_claude_key_rotation.py +0 -0
  28. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/test_mcp_config.py +0 -0
  29. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/test_runs_db.py +0 -0
  30. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/test_trigger_cron_skills.py +0 -0
  31. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/test_trigger_issue_skills.py +0 -0
  32. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/trigger_aws_sqs_skills.py +0 -0
  33. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/trigger_cron_skills.py +0 -0
  34. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/trigger_email_skills.py +0 -0
  35. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/lib/trigger_issue_skills.py +0 -0
  36. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/mail_server.py +0 -0
  37. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/setup_wizard.py +0 -0
  38. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/start_cli.py +0 -0
  39. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/tasks_providers.py +0 -0
  40. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/AGENTS.md +0 -0
  41. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/CLAUDE.md +0 -0
  42. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/aws-sqs-alarm-response/SKILL.md +0 -0
  43. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/cron-research-5xx-errors/SKILL.md +0 -0
  44. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-code-reviewer/SKILL.md +0 -0
  45. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-developer/SKILL.md +0 -0
  46. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-planner/SKILL.md +0 -0
  47. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-planner/assets/readme-template.md +0 -0
  48. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-qa/SKILL.md +0 -0
  49. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/story-security-reviewer/SKILL.md +0 -0
  50. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/task-code-reviewer/SKILL.md +0 -0
  51. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/task-developer/SKILL.md +0 -0
  52. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/task-qa/SKILL.md +0 -0
  53. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/templates/skills/task-security-reviewer/SKILL.md +0 -0
  54. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_admin_api.py +0 -0
  55. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_admin_cli.py +0 -0
  56. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_agent_cli.py +0 -0
  57. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_coding_agents.py +0 -0
  58. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_executor.py +0 -0
  59. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_init_cli.py +0 -0
  60. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_memory_index.py +0 -0
  61. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_setup_wizard.py +0 -0
  62. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/test_start_cli.py +0 -0
  63. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee/workflow_graph.py +0 -0
  64. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_admin/__init__.py +0 -0
  65. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_admin/codee_admin.py +0 -0
  66. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_abstract/__init__.py +0 -0
  67. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/__init__.py +0 -0
  68. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/account.py +0 -0
  69. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/credentials.py +0 -0
  70. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/oauth.py +0 -0
  71. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_claude_code/usage.py +0 -0
  72. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_codex/__init__.py +0 -0
  73. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_github_copilot/__init__.py +0 -0
  74. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_agent_github_copilot/provider.py +0 -0
  75. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_database/__init__.py +0 -0
  76. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_database/claude_code_accounts.py +0 -0
  77. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_database/database.py +0 -0
  78. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_database/oauth_tokens.py +0 -0
  79. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_main_context/__init__.py +0 -0
  80. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_main_context/context.py +0 -0
  81. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_main_context/logging.py +0 -0
  82. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_main_context/test_context.py +0 -0
  83. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_main_context/test_logging.py +0 -0
  84. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_abstract/__init__.py +0 -0
  85. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_abstract/provider.py +0 -0
  86. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_azure_devops/__init__.py +0 -0
  87. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_azure_devops/oauth.py +0 -0
  88. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_azure_devops/provider.py +0 -0
  89. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_azure_devops/test.py +0 -0
  90. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_jira/__init__.py +0 -0
  91. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_jira/provider.py +0 -0
  92. {codee_agent-0.6.4 → codee_agent-0.6.6}/src/codee_tasks_jira/test.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: codee-agent
3
- Version: 0.6.4
3
+ Version: 0.6.6
4
4
  Summary: A virtual co-worker that picks up Jira and Azure DevOps tasks and solves them with coding agents.
5
5
  Keywords: ai,agent,jira,azure-devops,claude-code,github-copilot,codex,automation
6
6
  Author: Denis Kibalko
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "codee-agent"
3
- version = "0.6.4"
3
+ version = "0.6.6"
4
4
  description = "A virtual co-worker that picks up Jira and Azure DevOps tasks and solves them with coding agents."
5
5
  readme = "README.md"
6
6
  license = "MIT"
@@ -101,6 +101,11 @@ class ModelOption(BaseModel):
101
101
  name: str
102
102
 
103
103
 
104
+ class ConversationMessage(BaseModel):
105
+ role: str
106
+ content: str
107
+
108
+
104
109
  class ClaudeAccount(BaseModel):
105
110
  """One connected Claude account as the settings page lists it.
106
111
 
@@ -356,6 +361,16 @@ class AdminState(rx.State):
356
361
  tasks_provider: str = "jira"
357
362
  coding_agent: str = "claude_code"
358
363
  max_parallel_agents: str = "3"
364
+ agent_test_open: bool = False
365
+ agent_test_input: str = ""
366
+ agent_test_session_id: str = ""
367
+ agent_test_messages: list[ConversationMessage] = []
368
+ agent_test_sending: bool = False
369
+ agent_test_generation: int = 0
370
+ agent_test_model: str = ""
371
+ agent_test_models: list[ModelOption] = []
372
+ agent_test_model_query: str = ""
373
+ agent_test_models_loading: bool = False
359
374
  # Whether the executor runs Claude Code on the connected accounts instead
360
375
  # of leaving ~/.claude/.credentials.json alone.
361
376
  claude_code_rotate_keys: bool = False
@@ -464,6 +479,32 @@ class AdminState(rx.State):
464
479
  return ""
465
480
  return query
466
481
 
482
+ @rx.var
483
+ def filtered_agent_test_models(self) -> list[ModelOption]:
484
+ query = self.agent_test_model_query.strip().lower()
485
+ return [
486
+ model for model in self.agent_test_models
487
+ if not query or query in f"{model.name} {model.id}".lower()
488
+ ]
489
+
490
+ @rx.var
491
+ def agent_test_model_label(self) -> str:
492
+ if not self.agent_test_model:
493
+ return "Agent default"
494
+ for model in self.agent_test_models:
495
+ if model.id == self.agent_test_model:
496
+ return model.name
497
+ return self.agent_test_model
498
+
499
+ @rx.var
500
+ def custom_agent_test_model_query(self) -> str:
501
+ query = self.agent_test_model_query.strip()
502
+ if not query or any(
503
+ model.id == query for model in self.agent_test_models
504
+ ):
505
+ return ""
506
+ return query
507
+
467
508
  @rx.var
468
509
  def active_route(self) -> str:
469
510
  return self.router.url.path.rstrip("/") or "/"
@@ -1516,6 +1557,89 @@ class AdminState(rx.State):
1516
1557
  error = self._persist_settings()
1517
1558
  return rx.toast.error(error) if error else rx.toast.success("Settings saved")
1518
1559
 
1560
+ def open_agent_test(self) -> Any:
1561
+ self.agent_test_input = ""
1562
+ self.agent_test_session_id = ""
1563
+ self.agent_test_messages = []
1564
+ self.agent_test_sending = False
1565
+ self.agent_test_generation += 1
1566
+ self.agent_test_model = ""
1567
+ self.agent_test_models = []
1568
+ self.agent_test_model_query = ""
1569
+ self.agent_test_open = True
1570
+ return AdminState.load_agent_test_models
1571
+
1572
+ def set_agent_test_open(self, open_: bool) -> None:
1573
+ self.agent_test_open = open_
1574
+ if not open_:
1575
+ self.agent_test_generation += 1
1576
+ self.agent_test_sending = False
1577
+
1578
+ def set_agent_test_input(self, value: str) -> None:
1579
+ self.agent_test_input = value
1580
+
1581
+ def set_agent_test_model_query(self, value: str) -> None:
1582
+ self.agent_test_model_query = value
1583
+
1584
+ def choose_agent_test_model(self, model_id: str) -> None:
1585
+ self.agent_test_model = model_id.strip()
1586
+ self.agent_test_model_query = ""
1587
+
1588
+ @rx.event(background=True)
1589
+ async def load_agent_test_models(self) -> None:
1590
+ async with self:
1591
+ agent = self.coding_agent
1592
+ self.agent_test_models_loading = True
1593
+ try:
1594
+ models = await asyncio.to_thread(SERVICE.list_agent_models, agent)
1595
+ except Exception:
1596
+ models = []
1597
+ async with self:
1598
+ if self.coding_agent != agent:
1599
+ return
1600
+ self.agent_test_models = [ModelOption(**model) for model in models]
1601
+ self.agent_test_models_loading = False
1602
+
1603
+ @rx.event(background=True)
1604
+ async def send_agent_test_message(self) -> Any:
1605
+ async with self:
1606
+ message = self.agent_test_input.strip()
1607
+ if not message or self.agent_test_sending:
1608
+ return
1609
+ agent = self.coding_agent
1610
+ session_id = self.agent_test_session_id
1611
+ model = self.agent_test_model
1612
+ generation = self.agent_test_generation
1613
+ self.agent_test_input = ""
1614
+ self.agent_test_sending = True
1615
+ self.agent_test_messages = [
1616
+ *self.agent_test_messages,
1617
+ ConversationMessage(role="user", content=message),
1618
+ ]
1619
+ try:
1620
+ response, session_id = await asyncio.to_thread(
1621
+ SERVICE.test_agent_conversation,
1622
+ agent,
1623
+ message,
1624
+ session_id,
1625
+ model,
1626
+ )
1627
+ except Exception as error:
1628
+ async with self:
1629
+ if self.agent_test_generation == generation:
1630
+ self.agent_test_sending = False
1631
+ yield rx.toast.error(f"The coding agent failed: {error}")
1632
+ return
1633
+ async with self:
1634
+ if self.agent_test_generation != generation:
1635
+ return
1636
+ self.agent_test_session_id = session_id
1637
+ self.agent_test_messages = [
1638
+ *self.agent_test_messages,
1639
+ ConversationMessage(role="assistant", content=response),
1640
+ ]
1641
+ self.agent_test_sending = False
1642
+
1519
1643
 
1520
1644
  ACCENT = "var(--codee-accent)"
1521
1645
  ACCENT_DEEP = "var(--codee-accent-deep)"
@@ -2065,7 +2189,7 @@ def model_menu_item(button: rx.Component) -> rx.Component:
2065
2189
  return rx.popover.close(rx.flex(button, width="100%"), width="100%")
2066
2190
 
2067
2191
 
2068
- def model_option_row(option: ModelOption) -> rx.Component:
2192
+ def _model_option_row(option: ModelOption, choose_model: Any) -> rx.Component:
2069
2193
  """One row of the model picker: friendly name left, model code right."""
2070
2194
  return model_menu_item(
2071
2195
  rx.button(
@@ -2077,7 +2201,15 @@ def model_option_row(option: ModelOption) -> rx.Component:
2077
2201
  align="center", spacing="2", width="100%"),
2078
2202
  variant="ghost", color_scheme="gray", width="100%",
2079
2203
  justify_content="start", padding="0.45rem 0.6rem",
2080
- on_click=AdminState.choose_model(option.id)))
2204
+ on_click=choose_model(option.id)))
2205
+
2206
+
2207
+ def model_option_row(option: ModelOption) -> rx.Component:
2208
+ return _model_option_row(option, AdminState.choose_model)
2209
+
2210
+
2211
+ def agent_test_model_option_row(option: ModelOption) -> rx.Component:
2212
+ return _model_option_row(option, AdminState.choose_agent_test_model)
2081
2213
 
2082
2214
 
2083
2215
  def agent_picker() -> rx.Component:
@@ -2402,9 +2534,9 @@ def run_row(run: RunRecord) -> rx.Component:
2402
2534
  rx.vstack(rx.hstack(rx.text(run.skill_name, font_weight="600"),
2403
2535
  rx.badge(run.status, color_scheme=rx.cond(run.status == "succeeded", "green", "red"))),
2404
2536
  rx.text(local_datetime(run.started_at), color=MUTED,
2405
- font_size="0.8rem",
2537
+ font_size="0.8rem",
2406
2538
  font_family="IBM Plex Mono, monospace"),
2407
- rx.text("Thread ID: ", run.session_id, color=MUTED,
2539
+ rx.text("Thread ID: ", run.session_id, color=MUTED,
2408
2540
  font_size="0.8rem",
2409
2541
  font_family="IBM Plex Mono, monospace"),
2410
2542
  rx.text(run.preview, color=MUTED),
@@ -2418,9 +2550,10 @@ def run_row(run: RunRecord) -> rx.Component:
2418
2550
  rx.cond((run.user_message != "") | (run.response != ""), rx.accordion.root(rx.accordion.item(
2419
2551
  header="Run info", content=rx.vstack(
2420
2552
  rx.text("User message", font_weight="600"),
2421
- rx.text(run.user_message, white_space="pre-wrap"),
2553
+ rx.text(run.user_message, white_space="pre-wrap"),
2422
2554
  rx.cond(run.response != "", rx.fragment(
2423
- rx.text("LLM response", font_weight="600", margin_top="0.75rem"),
2555
+ rx.text("LLM response", font_weight="600",
2556
+ margin_top="0.75rem"),
2424
2557
  rx.text(run.response, white_space="pre-wrap"))),
2425
2558
  spacing="2", align="start", width="100%"), value=run.started_at),
2426
2559
  collapsible=True, width="100%")),
@@ -2644,7 +2777,8 @@ def workflow_page() -> rx.Component:
2644
2777
  rx.center(
2645
2778
  rx.vstack(
2646
2779
  rx.spinner(size="3"),
2647
- rx.foreach(AdminState.workflow_progress, workflow_progress_line),
2780
+ rx.foreach(AdminState.workflow_progress,
2781
+ workflow_progress_line),
2648
2782
  spacing="3",
2649
2783
  align="center",
2650
2784
  width="100%",
@@ -3221,6 +3355,130 @@ def claude_code_setting() -> rx.Component:
3221
3355
  padding="1.25rem", background=SURFACE, border=BORDER, width="100%")
3222
3356
 
3223
3357
 
3358
+ def conversation_message(message: ConversationMessage) -> rx.Component:
3359
+ is_user = message.role == "user"
3360
+ return rx.box(
3361
+ rx.text(message.content, white_space="pre-wrap"),
3362
+ align_self=rx.cond(is_user, "end", "start"),
3363
+ background=rx.cond(is_user, ACTIVE, SURFACE),
3364
+ border=rx.cond(is_user, "none", BORDER),
3365
+ border_radius="6px",
3366
+ padding="0.65rem 0.8rem",
3367
+ max_width="85%",
3368
+ )
3369
+
3370
+
3371
+ def agent_test_dialog() -> rx.Component:
3372
+ return rx.dialog.root(
3373
+ rx.dialog.content(
3374
+ rx.dialog.title("Test conversation"),
3375
+ rx.dialog.description(
3376
+ "Chat with the selected coding agent in the Codee project.",
3377
+ color=MUTED),
3378
+ field(
3379
+ "Model",
3380
+ rx.popover.root(
3381
+ rx.popover.trigger(
3382
+ rx.button(
3383
+ rx.hstack(
3384
+ rx.text(AdminState.agent_test_model_label),
3385
+ rx.spacer(),
3386
+ rx.icon("chevrons-up-down", size=14),
3387
+ align="center", width="100%"),
3388
+ variant="surface", color_scheme="gray",
3389
+ width="100%", type="button")),
3390
+ rx.popover.content(
3391
+ rx.vstack(
3392
+ rx.input(
3393
+ placeholder=(
3394
+ "Search models, or type a model code"),
3395
+ value=AdminState.agent_test_model_query,
3396
+ on_change=AdminState.set_agent_test_model_query,
3397
+ auto_focus=True, width="100%"),
3398
+ rx.cond(
3399
+ AdminState.custom_agent_test_model_query != "",
3400
+ model_menu_item(
3401
+ rx.button(
3402
+ rx.hstack(
3403
+ rx.icon("plus", size=14),
3404
+ rx.text("Use "),
3405
+ rx.code(
3406
+ AdminState.custom_agent_test_model_query),
3407
+ align="center", spacing="2"),
3408
+ variant="soft", width="100%",
3409
+ justify_content="start",
3410
+ padding="0.45rem 0.6rem",
3411
+ on_click=AdminState.choose_agent_test_model(
3412
+ AdminState.custom_agent_test_model_query)))),
3413
+ rx.scroll_area(
3414
+ rx.vstack(
3415
+ model_menu_item(
3416
+ rx.button(
3417
+ "Agent default", variant="ghost",
3418
+ color_scheme="gray", width="100%",
3419
+ justify_content="start",
3420
+ padding="0.45rem 0.6rem",
3421
+ on_click=AdminState.choose_agent_test_model(""))),
3422
+ rx.foreach(
3423
+ AdminState.filtered_agent_test_models,
3424
+ agent_test_model_option_row),
3425
+ rx.cond(
3426
+ AdminState.agent_test_models_loading,
3427
+ rx.text(
3428
+ "Loading models from the coding agent…",
3429
+ color=MUTED, font_size="0.8rem",
3430
+ padding="0.5rem")),
3431
+ spacing="1", width="100%"),
3432
+ type="auto", scrollbars="vertical",
3433
+ max_height="15rem", width="100%"),
3434
+ spacing="2", width="100%"),
3435
+ width="24rem", max_width="calc(100vw - 3rem)")),
3436
+ rx.text(
3437
+ rx.cond(
3438
+ AdminState.agent_test_model == "",
3439
+ "Runs on whatever that agent defaults to.",
3440
+ rx.fragment("Uses ",
3441
+ rx.code(AdminState.agent_test_model),
3442
+ " for this conversation.")),
3443
+ color=MUTED, font_size="0.82rem")),
3444
+ rx.scroll_area(
3445
+ rx.vstack(
3446
+ rx.cond(
3447
+ AdminState.agent_test_messages.length() == 0,
3448
+ rx.text("Send a message to start the conversation.",
3449
+ color=MUTED, font_size="0.9rem",
3450
+ align_self="center", margin_top="3rem"),
3451
+ rx.foreach(AdminState.agent_test_messages,
3452
+ conversation_message)),
3453
+ width="100%", spacing="3"),
3454
+ type="auto", scrollbars="vertical", height="22rem",
3455
+ width="100%", margin_top="1rem"),
3456
+ rx.form(
3457
+ rx.hstack(
3458
+ rx.input(
3459
+ value=AdminState.agent_test_input,
3460
+ on_change=AdminState.set_agent_test_input,
3461
+ placeholder="Message the agent",
3462
+ disabled=AdminState.agent_test_sending,
3463
+ auto_focus=True,
3464
+ width="100%"),
3465
+ rx.button(rx.icon("send", size=16), type="submit",
3466
+ loading=AdminState.agent_test_sending,
3467
+ disabled=AdminState.agent_test_input == ""),
3468
+ spacing="2", width="100%"),
3469
+ on_submit=AdminState.send_agent_test_message,
3470
+ reset_on_submit=False,
3471
+ width="100%", margin_top="1rem"),
3472
+ rx.flex(
3473
+ rx.dialog.close(rx.button("Close", variant="soft",
3474
+ color_scheme="gray")),
3475
+ justify="end", margin_top="1rem"),
3476
+ max_width="38rem"),
3477
+ open=AdminState.agent_test_open,
3478
+ on_open_change=AdminState.set_agent_test_open,
3479
+ )
3480
+
3481
+
3224
3482
  def settings_page() -> rx.Component:
3225
3483
  jira = TasksProvider.JIRA
3226
3484
  jira_fields = rx.vstack(
@@ -3248,10 +3506,20 @@ def settings_page() -> rx.Component:
3248
3506
  rx.heading("Coding agent", size="4", margin_bottom="1rem"),
3249
3507
  rx.vstack(
3250
3508
  field("Default agent",
3251
- rx.select(["claude_code", "github_copilot", "codex"],
3252
- value=AdminState.coding_agent,
3253
- on_change=AdminState.set_coding_agent,
3254
- width="100%")),
3509
+ rx.grid(
3510
+ rx.select(
3511
+ ["claude_code", "github_copilot", "codex"],
3512
+ value=AdminState.coding_agent,
3513
+ on_change=AdminState.set_coding_agent,
3514
+ width="100%"),
3515
+ rx.button(rx.icon("messages-square", size=16),
3516
+ "Test conversation", variant="outline",
3517
+ white_space="nowrap",
3518
+ on_click=AdminState.open_agent_test),
3519
+ grid_template_columns=rx.breakpoints(
3520
+ initial="minmax(0, 1fr)",
3521
+ md="minmax(0, 1fr) auto"),
3522
+ width="100%", gap="0.75rem")),
3255
3523
  field("Max parallel tasks",
3256
3524
  rx.input(value=AdminState.max_parallel_agents,
3257
3525
  on_change=AdminState.set_max_parallel_agents,
@@ -3277,6 +3545,7 @@ def settings_page() -> rx.Component:
3277
3545
  padding="1.25rem", background=SURFACE, border=BORDER, width="100%"),
3278
3546
  rx.button(rx.icon("save", size=16), "Save settings",
3279
3547
  on_click=AdminState.save_settings),
3548
+ agent_test_dialog(),
3280
3549
  spacing="5", align="start", width="100%"))
3281
3550
 
3282
3551
 
@@ -192,6 +192,7 @@ class WorkflowGeneration:
192
192
  workflow: dict[str, Any] | None = None
193
193
  error: str = ""
194
194
 
195
+
195
196
  # The checks the settings page runs against the tasks provider, in the order it
196
197
  # shows them: the second is only worth attempting once the first passes.
197
198
  TASKS_CHECK = "Tasks can be pulled"
@@ -893,7 +894,8 @@ class AdminService:
893
894
  # and how long it stands. Shared by every visitor to the dashboard,
894
895
  # because it is a property of the accounts rather than of whoever is
895
896
  # looking at them.
896
- self._usage_cache: dict[int, tuple[ConnectedAccount, float, float]] = {}
897
+ self._usage_cache: dict[int,
898
+ tuple[ConnectedAccount, float, float]] = {}
897
899
  self._usage_lock = threading.Lock()
898
900
 
899
901
  def _git_push(self, message: str) -> tuple[bool, str]:
@@ -1221,7 +1223,8 @@ class AdminService:
1221
1223
  name belongs on the other item's graph, so it is kept out of this one.
1222
1224
  """
1223
1225
  if not skills:
1224
- report(f"No issue-trigger skills for work item {issue_type.capitalize()}.")
1226
+ report(
1227
+ f"No issue-trigger skills for work item {issue_type.capitalize()}.")
1225
1228
  return {"nodes": [], "edges": [], "warnings": []}
1226
1229
 
1227
1230
  documents = []
@@ -1414,7 +1417,8 @@ class AdminService:
1414
1417
  for status in (transition["source"], transition["target"])}
1415
1418
  statuses, foreign_statuses = _own_work_item_statuses(
1416
1419
  statuses, scope, triggered | moved,
1417
- dict.fromkeys(document for _, document in skill_documents.values()),
1420
+ dict.fromkeys(document for _,
1421
+ document in skill_documents.values()),
1418
1422
  )
1419
1423
  if foreign_statuses:
1420
1424
  report(
@@ -1831,7 +1835,7 @@ class AdminService:
1831
1835
  existing, _ = parse_skill(current_path.read_text())
1832
1836
  extra = {key: value for key, value in existing.items()
1833
1837
  if key not in MANAGED}
1834
- action =f"rename {old_slug} -> {name}" if name != old_slug else f"update {name}"
1838
+ action = f"rename {old_slug} -> {name}" if name != old_slug else f"update {name}"
1835
1839
  saved, pushed, message = self._write_and_push(
1836
1840
  current_path,
1837
1841
  build_skill(frontmatter, extra, skill["body"]),
@@ -2439,6 +2443,29 @@ class AdminService:
2439
2443
  except ValueError as exc:
2440
2444
  raise RuntimeError(str(exc)) from exc
2441
2445
 
2446
+ def test_agent_conversation(
2447
+ self, agent_code: str, user_message: str, session_id: str = "",
2448
+ model: str = "",
2449
+ ) -> tuple[str, str]:
2450
+ """Run one turn with the selected agent and return reply plus session id."""
2451
+ selected = resolve_agent_code(agent_code)
2452
+ if selected is None:
2453
+ raise RuntimeError(f"Unsupported coding agent: {agent_code}")
2454
+ agent = build_coding_agent(self.context.settings, self.root, selected)
2455
+ actual_session_id = session_id or str(uuid.uuid4())
2456
+
2457
+ def opened(opened_session_id: str) -> None:
2458
+ nonlocal actual_session_id
2459
+ actual_session_id = opened_session_id
2460
+
2461
+ if session_id:
2462
+ response = agent.continue_conversation(
2463
+ user_message, session_id, model, on_session_id=opened)
2464
+ else:
2465
+ response = agent.run(
2466
+ user_message, actual_session_id, model, on_session_id=opened)
2467
+ return response, actual_session_id
2468
+
2442
2469
  def setup_tasks_mcp(
2443
2470
  self,
2444
2471
  tasks_provider: str,
@@ -2600,7 +2627,8 @@ def _format_token_expiry(expires_at: str | None) -> str:
2600
2627
  return "refreshes on next check"
2601
2628
  if deadline.tzinfo is None:
2602
2629
  deadline = deadline.replace(tzinfo=timezone.utc)
2603
- minutes = int((deadline - datetime.now(timezone.utc)).total_seconds() // 60)
2630
+ minutes = int(
2631
+ (deadline - datetime.now(timezone.utc)).total_seconds() // 60)
2604
2632
  if minutes < 1:
2605
2633
  return "refreshes on next check"
2606
2634
  if minutes < 60:
@@ -80,7 +80,8 @@ class NormalizeWorkItemsTest(unittest.TestCase):
80
80
  [("story", ["Story"], ""), ("task", ["Task", "Bug"], "")])
81
81
 
82
82
  self.assertEqual(error, "")
83
- self.assertEqual(mapping, {"story": ["Story"], "task": ["Task", "Bug"]})
83
+ self.assertEqual(
84
+ mapping, {"story": ["Story"], "task": ["Task", "Bug"]})
84
85
 
85
86
  def test_a_type_listed_twice_in_one_row_is_kept_once(self) -> None:
86
87
  mapping, _, error = normalize_work_items(
@@ -164,6 +165,46 @@ class NormalizeWorkItemsTest(unittest.TestCase):
164
165
  self.assertEqual(error, "'bug' is listed twice")
165
166
 
166
167
 
168
+ class TestAgentConversationTest(unittest.TestCase):
169
+ def setUp(self) -> None:
170
+ self.service = AdminService.__new__(AdminService)
171
+ self.service.root = Path("/repo")
172
+ self.service.context = Mock(settings=Settings())
173
+
174
+ @patch("codee.admin_service.build_coding_agent")
175
+ def test_first_turn_returns_the_agent_session_id(self, build_agent) -> None:
176
+ agent = build_agent.return_value
177
+
178
+ def run(message, session_id, model="", on_session_id=None):
179
+ on_session_id("agent-thread")
180
+ return "Final response"
181
+
182
+ agent.run.side_effect = run
183
+
184
+ response, session_id = self.service.test_agent_conversation(
185
+ "codex", "Hello", model="gpt-6-astra")
186
+
187
+ self.assertEqual(response, "Final response")
188
+ self.assertEqual(session_id, "agent-thread")
189
+ build_agent.assert_called_once_with(
190
+ self.service.context.settings, Path("/repo"), CodingAgent.CODEX)
191
+ self.assertEqual(agent.run.call_args.args[2], "gpt-6-astra")
192
+
193
+ @patch("codee.admin_service.build_coding_agent")
194
+ def test_later_turn_resumes_the_same_session(self, build_agent) -> None:
195
+ build_agent.return_value.continue_conversation.return_value = "Still here"
196
+
197
+ response, session_id = self.service.test_agent_conversation(
198
+ "github_copilot", "What did I ask?", "same-thread",
199
+ "claude-opus-5")
200
+
201
+ self.assertEqual((response, session_id), ("Still here", "same-thread"))
202
+ call = build_agent.return_value.continue_conversation.call_args
203
+ self.assertEqual(
204
+ call.args[:3],
205
+ ("What did I ask?", "same-thread", "claude-opus-5"))
206
+
207
+
167
208
  class AdminServiceWorkItemsTest(unittest.TestCase):
168
209
  def _service(self, directory: Path) -> AdminService:
169
210
  service = AdminService.__new__(AdminService)
@@ -408,7 +449,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
408
449
  self._connect("one@example.com")
409
450
  self._connect("two@example.com")
410
451
  second = self.service.claude_code_accounts()[1]
411
- claude_code_accounts.set_current_account(second.id, self.service.context)
452
+ claude_code_accounts.set_current_account(
453
+ second.id, self.service.context)
412
454
 
413
455
  self.service.disconnect_claude_code_account(second.id)
414
456
 
@@ -423,7 +465,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
423
465
 
424
466
  self.service.disconnect_claude_code_account(account.id)
425
467
 
426
- self.assertEqual(claude_code_accounts.accounts(self.service.context), [])
468
+ self.assertEqual(claude_code_accounts.accounts(
469
+ self.service.context), [])
427
470
 
428
471
  def test_switching_rotation_off_forgets_which_account_is_in_use(self) -> None:
429
472
  # So switching it back on later starts from the top rather than from
@@ -431,7 +474,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
431
474
  self._connect("one@example.com")
432
475
  self._connect("two@example.com")
433
476
  second = self.service.claude_code_accounts()[1]
434
- claude_code_accounts.set_current_account(second.id, self.service.context)
477
+ claude_code_accounts.set_current_account(
478
+ second.id, self.service.context)
435
479
 
436
480
  self.service.save_settings("jira", "claude_code", 3, {}, None, None,
437
481
  "", False)
@@ -465,14 +509,16 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
465
509
  "access-1", "refresh-1", expires_at=far_future,
466
510
  refresh_expires_at=far_future))
467
511
 
468
- self.assertFalse(self.service.claude_code_accounts()[0].needs_reconnect)
512
+ self.assertFalse(self.service.claude_code_accounts()
513
+ [0].needs_reconnect)
469
514
 
470
515
  def test_an_account_with_no_known_window_is_not_flagged(self) -> None:
471
516
  # Zero means the API never said, which is not the same as expired.
472
517
  self._connect("one@example.com", tokens=claude_oauth.Tokens(
473
518
  "access-1", "refresh-1", expires_at=1, refresh_expires_at=0))
474
519
 
475
- self.assertFalse(self.service.claude_code_accounts()[0].needs_reconnect)
520
+ self.assertFalse(self.service.claude_code_accounts()
521
+ [0].needs_reconnect)
476
522
 
477
523
  def test_each_account_reports_both_of_its_windows(self) -> None:
478
524
  self._connect("one@example.com")
@@ -487,7 +533,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
487
533
 
488
534
  self.assertEqual(measured[0].session_percent, 12.0)
489
535
  self.assertEqual(measured[0].weekly_percent, 70.0)
490
- self.assertEqual(measured[0].weekly_resets, "2026-09-16T09:00:00+00:00")
536
+ self.assertEqual(measured[0].weekly_resets,
537
+ "2026-09-16T09:00:00+00:00")
491
538
  self.assertEqual(measured[0].usage_error, "")
492
539
 
493
540
  def test_the_reading_is_cached_rather_than_taken_every_redraw(self) -> None:
@@ -555,7 +602,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
555
602
  measured[1].id, self.service.context)
556
603
  again = self.service.claude_code_account_usage()
557
604
 
558
- self.assertEqual([account.in_use for account in measured], [True, False])
605
+ self.assertEqual(
606
+ [account.in_use for account in measured], [True, False])
559
607
  self.assertEqual([account.in_use for account in again], [False, True])
560
608
  self.assertEqual(again[1].session_percent, 3.0)
561
609
 
@@ -594,7 +642,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
594
642
 
595
643
  self.assertEqual(measured[0].session_percent, 12.0)
596
644
  self.assertEqual(measured[0].weekly_percent, 70.0)
597
- self.assertEqual(measured[0].session_resets, "2026-09-13T14:10:00+00:00")
645
+ self.assertEqual(measured[0].session_resets,
646
+ "2026-09-13T14:10:00+00:00")
598
647
  self.assertEqual(measured[0].usage_error, "")
599
648
 
600
649
  def test_a_rate_limit_with_nothing_to_fall_back_on_says_so(self) -> None:
@@ -662,7 +711,8 @@ class AdminServiceClaudeCodeAccountsTest(unittest.TestCase):
662
711
  return_value=Usage(limited=False, windows={}, resets_at={})):
663
712
  measured = self.service.claude_code_account_usage()
664
713
 
665
- self.assertEqual([account.in_use for account in measured], [True, False])
714
+ self.assertEqual(
715
+ [account.in_use for account in measured], [True, False])
666
716
 
667
717
  def test_no_accounts_means_no_requests(self) -> None:
668
718
  with patch("codee.admin_service.fetch_usage") as fetch:
@@ -1927,7 +1977,6 @@ class AdminServiceIssueTriggerTest(unittest.TestCase):
1927
1977
 
1928
1978
  self.assertEqual(generate.call_count, 1)
1929
1979
 
1930
-
1931
1980
  def test_generate_workflow_regenerates_a_graph_from_an_older_version(self) -> None:
1932
1981
  with tempfile.TemporaryDirectory() as temporary_directory:
1933
1982
  root = Path(temporary_directory)
@@ -2046,7 +2095,8 @@ class AdminServiceWorkflowAgentTest(unittest.TestCase):
2046
2095
 
2047
2096
  nodes = self._story_nodes(service)
2048
2097
 
2049
- self.assertNotIn("workflow-node--agent", nodes["Review"]["className"])
2098
+ self.assertNotIn("workflow-node--agent",
2099
+ nodes["Review"]["className"])
2050
2100
  self.assertNotIn("style", nodes["Review"])
2051
2101
 
2052
2102
  def test_changing_the_agent_needs_no_second_inference(self) -> None:
@@ -2491,7 +2541,8 @@ class AdminServiceAgentModelsTest(unittest.TestCase):
2491
2541
  patch.object(ClaudeCodeAgent, "list_models") as claude_models:
2492
2542
  models = service.list_agent_models("codex")
2493
2543
 
2494
- self.assertEqual(models, [{"id": "gpt-6-astra", "name": "GPT-6 Astra"}])
2544
+ self.assertEqual(
2545
+ models, [{"id": "gpt-6-astra", "name": "GPT-6 Astra"}])
2495
2546
  claude_models.assert_not_called()
2496
2547
 
2497
2548
  def test_an_agent_codee_cannot_run_falls_back_to_the_default(self) -> None:
@@ -69,6 +69,21 @@ class AbstractCodingAgent(ABC):
69
69
  """
70
70
  ...
71
71
 
72
+ def continue_conversation(
73
+ self,
74
+ user_message: str,
75
+ session_id: str,
76
+ model: str = "",
77
+ on_session_id: Callable[[str], None] | None = None,
78
+ ) -> str:
79
+ """Continue an existing agent conversation and return its final reply.
80
+
81
+ Agents whose normal run command resumes when given an existing session
82
+ need no special implementation. Agents with a distinct resume command
83
+ override this method.
84
+ """
85
+ return self.run(user_message, session_id, model, on_session_id)
86
+
72
87
  def skill_prompt(self, slug: str, path: Path, argument: str = "",
73
88
  argument_name: str = "") -> str:
74
89
  """The message that makes this agent run the skill stored at ``path``.
@@ -58,10 +58,25 @@ class ClaudeCodeAgent(AbstractCodingAgent):
58
58
  # answer is known before the run starts.
59
59
  if on_session_id:
60
60
  on_session_id(session_id)
61
+ return self._run(user_message, session_id, model, False)
62
+
63
+ def continue_conversation(
64
+ self,
65
+ user_message: str,
66
+ session_id: str,
67
+ model: str = "",
68
+ on_session_id: Callable[[str], None] | None = None,
69
+ ) -> str:
70
+ if on_session_id:
71
+ on_session_id(session_id)
72
+ return self._run(user_message, session_id, model, True)
73
+
74
+ def _run(self, user_message: str, session_id: str, model: str,
75
+ resume: bool) -> str:
61
76
  cmd = [
62
77
  self.CLI_COMMAND,
63
78
  "-p", user_message,
64
- "--session-id", session_id,
79
+ "--resume" if resume else "--session-id", session_id,
65
80
  "--max-budget-usd", self.MAX_BUDGET_USD,
66
81
  "--output-format", "json",
67
82
  "--permission-mode", "bypassPermissions",
@@ -91,7 +106,7 @@ class ClaudeCodeAgent(AbstractCodingAgent):
91
106
  stdout = result.stdout or ""
92
107
  stderr = result.stderr or ""
93
108
  log.debug("claude exited %d (%d bytes stdout, %d bytes stderr)",
94
- result.returncode, len(stdout), len(stderr))
109
+ result.returncode, len(stdout), len(stderr))
95
110
 
96
111
  # Raise on any non-success so callers retry. Over-limit exits non-zero;
97
112
  # a completed-but-errored run sets is_error in the JSON.
@@ -34,7 +34,8 @@ def _response(status_code: int, text: str = "", json_body=None):
34
34
  response = Mock(spec=["status_code", "text", "json"])
35
35
  response.status_code = status_code
36
36
  response.text = text
37
- response.json = Mock(return_value=json_body if json_body is not None else {})
37
+ response.json = Mock(
38
+ return_value=json_body if json_body is not None else {})
38
39
  return response
39
40
 
40
41
 
@@ -56,6 +57,14 @@ class ClaudeCodeRunTest(unittest.TestCase):
56
57
 
57
58
  self.assertEqual(cmd[cmd.index("--model") + 1], "opus")
58
59
 
60
+ def test_continuing_uses_resume_instead_of_starting_a_session(self) -> None:
61
+ with patch("subprocess.run", return_value=_completed()) as run:
62
+ self.agent.continue_conversation("And now?", SESSION)
63
+
64
+ cmd = run.call_args.args[0]
65
+ self.assertEqual(cmd[cmd.index("--resume") + 1], SESSION)
66
+ self.assertNotIn("--session-id", cmd)
67
+
59
68
  def test_cli_output_is_decoded_as_utf8(self) -> None:
60
69
  with patch("subprocess.run", return_value=_completed()) as run:
61
70
  self.agent.run("/do-it CORE-1", SESSION)
@@ -129,7 +138,8 @@ class ClaudeCodeUsageTest(unittest.TestCase):
129
138
  # the executor to a key by looking like an error.
130
139
  self.assertFalse(read_usage({}).limited)
131
140
  self.assertFalse(read_usage({"five_hour": None}).limited)
132
- self.assertFalse(read_usage({"five_hour": {"utilization": None}}).limited)
141
+ self.assertFalse(read_usage(
142
+ {"five_hour": {"utilization": None}}).limited)
133
143
 
134
144
  def test_a_rejected_key_counts_as_spent(self) -> None:
135
145
  # Expired or revoked. As unusable as an exhausted one, and the same
@@ -101,12 +101,36 @@ class CodexAgent(AbstractCodingAgent):
101
101
  f"Codex run produced no response: {_detail(errors, result.stderr)}")
102
102
  return reply
103
103
 
104
+ def continue_conversation(
105
+ self,
106
+ user_message: str,
107
+ session_id: str,
108
+ model: str = "",
109
+ on_session_id: Callable[[str], None] | None = None,
110
+ ) -> str:
111
+ result = self._exec(user_message, model, on_session_id, session_id)
112
+ reply, completed, errors = _parse_events(result.stdout)
113
+ if result.returncode != 0:
114
+ raise RuntimeError(
115
+ f"Codex CLI exited {result.returncode}: "
116
+ f"{_detail(errors, result.stderr)}"
117
+ )
118
+ if not completed:
119
+ raise RuntimeError(
120
+ f"Codex run errored: {_detail(errors, result.stderr)}")
121
+ if not reply:
122
+ raise RuntimeError(
123
+ f"Codex run produced no response: {_detail(errors, result.stderr)}")
124
+ return reply
125
+
104
126
  def _exec(self, user_message: str, model: str,
105
- on_session_id: Callable[[str], None] | None = None
127
+ on_session_id: Callable[[str], None] | None = None,
128
+ resume_session_id: str = "",
106
129
  ) -> subprocess.CompletedProcess:
107
130
  """One headless ``codex exec`` in a thread of its own."""
108
131
  cmd = [
109
- self.CLI_COMMAND, "exec",
132
+ self.CLI_COMMAND, "exec", *
133
+ (["resume"] if resume_session_id else []),
110
134
  "--json",
111
135
  # Codee's project root is a repository in the normal case but need
112
136
  # not be one, and `codex exec` refuses to start outside git.
@@ -124,7 +148,10 @@ class CodexAgent(AbstractCodingAgent):
124
148
  cmd += _mcp_overrides(self._cwd)
125
149
  # The prompt goes last, behind `--`, so a skill whose slug collides with
126
150
  # a subcommand (`codex exec review`) still reaches the model.
127
- cmd += ["--", user_message]
151
+ cmd += ["--"]
152
+ if resume_session_id:
153
+ cmd += [resume_session_id]
154
+ cmd += [user_message]
128
155
 
129
156
  log.debug("cwd=%s cmd=%s", self._cwd, " ".join(cmd))
130
157
  # Streamed rather than collected at the end: the thread id is on the
@@ -105,6 +105,15 @@ class CodexRunTest(unittest.TestCase):
105
105
  self.assertNotIn(SESSION, self._cmd())
106
106
  self.assertNotIn("resume", self._cmd())
107
107
 
108
+ def test_continuing_resumes_the_thread(self) -> None:
109
+ with patch("subprocess.Popen", return_value=_completed(_turn())) as popen:
110
+ response = self.agent.continue_conversation("And now?", THREAD)
111
+
112
+ self.assertEqual(response, "Done, PR is up.")
113
+ cmd = popen.call_args.args[0]
114
+ self.assertEqual(cmd[:3], ["codex", "exec", "resume"])
115
+ self.assertEqual(cmd[-3:], ["--", THREAD, "And now?"])
116
+
108
117
  def test_the_thread_codex_opened_is_reported_back(self) -> None:
109
118
  # What the dashboard links its session viewer to while the run is live.
110
119
  self._run(_completed(_turn()))
@@ -65,7 +65,8 @@ class CopilotSkillPromptTest(unittest.TestCase):
65
65
  )
66
66
 
67
67
  def test_a_skill_outside_the_working_directory_keeps_its_full_path(self) -> None:
68
- prompt = self._prompt(Path("/elsewhere/skills/reviewer/SKILL.md"), "7", "ID")
68
+ prompt = self._prompt(
69
+ Path("/elsewhere/skills/reviewer/SKILL.md"), "7", "ID")
69
70
 
70
71
  self.assertEqual(
71
72
  prompt,
@@ -85,7 +86,8 @@ class CopilotRunTest(unittest.TestCase):
85
86
 
86
87
  def test_returns_the_last_assistant_message(self) -> None:
87
88
  stdout = _stream(
88
- _event("assistant.message", {"content": "Looking at it", "toolRequests": [{}]}),
89
+ _event("assistant.message", {
90
+ "content": "Looking at it", "toolRequests": [{}]}),
89
91
  _event("tool.execution_complete", {}),
90
92
  _event("assistant.message", {"content": "Done, PR is up.\n"}),
91
93
  _event("result", exitCode=0, sessionId=SESSION),
@@ -119,6 +121,15 @@ class CopilotRunTest(unittest.TestCase):
119
121
  self.assertEqual(self.captured.call_args.kwargs["encoding"], "utf-8")
120
122
  self.assertEqual(self.captured.call_args.kwargs["errors"], "replace")
121
123
 
124
+ def test_continuing_reuses_the_same_session_id(self) -> None:
125
+ stdout = _stream(_event("assistant.message", {"content": "ok"}),
126
+ _event("result", exitCode=0))
127
+ with patch("subprocess.run", return_value=_completed(stdout)) as run:
128
+ self.agent.continue_conversation("And now?", SESSION)
129
+
130
+ cmd = run.call_args.args[0]
131
+ self.assertEqual(cmd[cmd.index("--session-id") + 1], SESSION)
132
+
122
133
  def test_missing_captured_streams_do_not_raise_type_error(self) -> None:
123
134
  completed = subprocess.CompletedProcess(
124
135
  args=["copilot"], returncode=0, stdout=None, stderr=None)
@@ -159,7 +170,8 @@ class CopilotRunTest(unittest.TestCase):
159
170
 
160
171
  def test_a_failed_run_raises_with_the_session_error(self) -> None:
161
172
  stdout = _stream(
162
- _event("session.error", {"errorType": "quota", "message": "quota exceeded"}),
173
+ _event("session.error", {
174
+ "errorType": "quota", "message": "quota exceeded"}),
163
175
  _event("result", exitCode=1),
164
176
  )
165
177
 
@@ -193,7 +205,8 @@ class CopilotCreditCapTest(unittest.TestCase):
193
205
  else:
194
206
  os.environ[MAX_AI_CREDITS_ENV_VAR] = value
195
207
  with patch("subprocess.run", return_value=_completed(stdout)) as run:
196
- GitHubCopilotAgent(Settings(), Path("/repo")).run("/do-it", SESSION)
208
+ GitHubCopilotAgent(Settings(), Path(
209
+ "/repo")).run("/do-it", SESSION)
197
210
  cmd = run.call_args.args[0]
198
211
  return cmd[cmd.index("--max-ai-credits") + 1]
199
212
 
@@ -222,9 +235,12 @@ class CopilotModelCatalogTest(unittest.TestCase):
222
235
 
223
236
  def test_reads_the_session_new_result_past_other_traffic(self) -> None:
224
237
  lines = self._queue(
225
- json.dumps({"jsonrpc": "2.0", "id": 1, "result": {"protocolVersion": 1}}),
226
- json.dumps({"jsonrpc": "2.0", "method": "session/update", "params": {}}),
227
- json.dumps({"jsonrpc": "2.0", "id": 2, "result": {"sessionId": "s1"}}),
238
+ json.dumps({"jsonrpc": "2.0", "id": 1,
239
+ "result": {"protocolVersion": 1}}),
240
+ json.dumps(
241
+ {"jsonrpc": "2.0", "method": "session/update", "params": {}}),
242
+ json.dumps({"jsonrpc": "2.0", "id": 2,
243
+ "result": {"sessionId": "s1"}}),
228
244
  )
229
245
 
230
246
  result = _await_result(Mock(poll=Mock(return_value=None)), lines, 2)
@@ -243,7 +259,8 @@ class CopilotModelCatalogTest(unittest.TestCase):
243
259
  def test_a_catalog_becomes_id_and_name_pairs(self) -> None:
244
260
  result = {"models": {"availableModels": [
245
261
  {"modelId": "claude-opus-5", "name": "Claude Opus 5"},
246
- {"modelId": "gpt-5.4"}, # no display name: falls back to the id
262
+ # no display name: falls back to the id
263
+ {"modelId": "gpt-5.4"},
247
264
  {"name": "nameless"}, # no id at all: unusable, skipped
248
265
  ]}}
249
266
 
File without changes
File without changes