polymath-agent 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- polymath/__init__.py +2 -0
- polymath/adapters/__init__.py +7 -0
- polymath/adapters/base.py +175 -0
- polymath/adapters/claude.py +280 -0
- polymath/adapters/gemini.py +186 -0
- polymath/adapters/ollama.py +117 -0
- polymath/adapters/openai_adapter.py +168 -0
- polymath/bootstrap.py +159 -0
- polymath/command_registry.py +41 -0
- polymath/command_service.py +572 -0
- polymath/compressor.py +90 -0
- polymath/config.py +293 -0
- polymath/context_manager.py +76 -0
- polymath/context_store.py +336 -0
- polymath/detector.py +442 -0
- polymath/domain.py +78 -0
- polymath/execution_service.py +325 -0
- polymath/main.py +1293 -0
- polymath/memory/__init__.py +15 -0
- polymath/memory/chunker.py +6 -0
- polymath/memory/embedder.py +179 -0
- polymath/memory/migrate.py +2 -0
- polymath/memory/retriever.py +2 -0
- polymath/memory/store.py +9 -0
- polymath/memory/sync.py +2 -0
- polymath/memory/writer.py +9 -0
- polymath/model_policy.py +172 -0
- polymath/orchestrator/__init__.py +68 -0
- polymath/orchestrator/attempt_ledger.py +34 -0
- polymath/orchestrator/ensemble.py +229 -0
- polymath/orchestrator/fanout.py +322 -0
- polymath/orchestrator/output_policy.py +61 -0
- polymath/orchestrator/race.py +311 -0
- polymath/orchestrator/run_controller.py +91 -0
- polymath/orchestrator/speculative_review.py +120 -0
- polymath/orchestrator/state_responder.py +184 -0
- polymath/orchestrator/worker_pool.py +37 -0
- polymath/permissions.py +82 -0
- polymath/pipeline.py +700 -0
- polymath/project_config.py +229 -0
- polymath/project_runtime.py +109 -0
- polymath/router.py +127 -0
- polymath/setup_wizard.py +106 -0
- polymath/slash_commands.py +566 -0
- polymath/subagents.py +486 -0
- polymath/tools.py +333 -0
- polymath/ui_state.py +84 -0
- polymath/workspace.py +66 -0
- polymath_agent-0.4.0.dist-info/METADATA +693 -0
- polymath_agent-0.4.0.dist-info/RECORD +54 -0
- polymath_agent-0.4.0.dist-info/WHEEL +5 -0
- polymath_agent-0.4.0.dist-info/entry_points.txt +2 -0
- polymath_agent-0.4.0.dist-info/licenses/LICENSE +21 -0
- polymath_agent-0.4.0.dist-info/top_level.txt +1 -0
polymath/pipeline.py
ADDED
|
@@ -0,0 +1,700 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Task pipeline with agentic loop.
|
|
3
|
+
|
|
4
|
+
SIMPLE: PLAN → PLAN_APPROVAL → CLARIFY → EXECUTE → REVIEW → OUTPUT
|
|
5
|
+
MEDIUM: CLARIFY → PLAN → PLAN_APPROVAL → CLARIFY → EXECUTE → REVIEW → OUTPUT
|
|
6
|
+
COMPLEX: UNDERSTAND → CLARIFY → PLAN → PLAN_APPROVAL → CLARIFY
|
|
7
|
+
→ EXECUTE (REASON/ACT/OBSERVE) → REVIEW → CORRECTIONS → RE-REVIEW → OUTPUT
|
|
8
|
+
|
|
9
|
+
Plan mode is on for ALL complexities by default — this is Polymath's
|
|
10
|
+
headline UX. The only path that skips plan is `--ask` (direct one-shot),
|
|
11
|
+
handled outside this pipeline by run_ask_flow in main.py.
|
|
12
|
+
|
|
13
|
+
Override via cfg["plan_mode"]:
|
|
14
|
+
"always" → plan for every session (default)
|
|
15
|
+
"complexity" → plan only for MEDIUM + COMPLEX (legacy behaviour)
|
|
16
|
+
"off" → never plan (use sparingly; loses the moat)
|
|
17
|
+
"""
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import inspect
|
|
21
|
+
import json
|
|
22
|
+
import logging
|
|
23
|
+
import re
|
|
24
|
+
from dataclasses import dataclass, field
|
|
25
|
+
from enum import Enum
|
|
26
|
+
from typing import Any, Callable, AsyncIterator, Dict, List
|
|
27
|
+
|
|
28
|
+
from polymath.adapters.base import BaseAdapter, Message, ToolCall
|
|
29
|
+
|
|
30
|
+
logger = logging.getLogger(__name__)
|
|
31
|
+
from polymath.compressor import build_context
|
|
32
|
+
from polymath.config import ModelInfo, TaskComplexity, TaskType, memory_setting
|
|
33
|
+
from polymath.domain import PermissionRequest
|
|
34
|
+
from polymath.permissions import build_permission_request
|
|
35
|
+
from polymath.context_store import add_message, export_session_md, get_messages
|
|
36
|
+
from polymath.context_manager import build_context_injection, export_session_to_project
|
|
37
|
+
from polymath.memory.retriever import retrieve_or_fallback
|
|
38
|
+
from polymath.memory.writer import harvest_session as memory_write_back
|
|
39
|
+
from polymath.orchestrator.race import (
|
|
40
|
+
AllCandidatesFailed,
|
|
41
|
+
RaceCandidate,
|
|
42
|
+
StreamCandidate,
|
|
43
|
+
race_streams,
|
|
44
|
+
with_failover,
|
|
45
|
+
)
|
|
46
|
+
from polymath.orchestrator.speculative_review import SpeculativeReviewer
|
|
47
|
+
from polymath.tools import registry as tool_registry
|
|
48
|
+
from polymath.workspace import get_workspace_context
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
class Step(Enum):
|
|
52
|
+
UNDERSTAND = "understand"
|
|
53
|
+
CLARIFY_1 = "clarify_1"
|
|
54
|
+
PLAN = "plan"
|
|
55
|
+
CLARIFY_2 = "clarify_2"
|
|
56
|
+
EXECUTE = "execute"
|
|
57
|
+
REASON = "reason"
|
|
58
|
+
ACT = "act"
|
|
59
|
+
OBSERVE = "observe"
|
|
60
|
+
PREVIEW = "preview"
|
|
61
|
+
REVIEW = "review"
|
|
62
|
+
CORRECTIONS = "corrections"
|
|
63
|
+
RE_REVIEW = "re_review"
|
|
64
|
+
OUTPUT = "output"
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
@dataclass
|
|
68
|
+
class PipelineResult:
|
|
69
|
+
output: str = ""
|
|
70
|
+
plan: str = ""
|
|
71
|
+
review_notes: list[str] = field(default_factory=list)
|
|
72
|
+
primary_model: str = ""
|
|
73
|
+
reviewer_model: str = ""
|
|
74
|
+
steps_run: list[str] = field(default_factory=list)
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ── System prompts ────────────────────────────────────────────────────────────
|
|
78
|
+
|
|
79
|
+
UNDERSTAND_SYSTEM = (
|
|
80
|
+
"You are a thorough analyst. When given a task:\n"
|
|
81
|
+
"1. State your understanding of what's being asked.\n"
|
|
82
|
+
"2. Identify any ambiguities or missing information.\n"
|
|
83
|
+
"3. List any assumptions you're making.\n"
|
|
84
|
+
"Keep this concise — 3–5 sentences total."
|
|
85
|
+
)
|
|
86
|
+
|
|
87
|
+
CLARIFY_SYSTEM = (
|
|
88
|
+
"Based on your understanding, generate up to 3 clarifying questions that would "
|
|
89
|
+
"meaningfully improve your response. If no clarification is needed, say exactly: "
|
|
90
|
+
"NO_CLARIFICATION_NEEDED. Otherwise list questions numbered 1, 2, 3."
|
|
91
|
+
)
|
|
92
|
+
|
|
93
|
+
PLAN_SYSTEM = (
|
|
94
|
+
"Create a clear, numbered step-by-step plan to accomplish this task. "
|
|
95
|
+
"Be specific. Each step should be actionable. No commentary — just the plan."
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
AGENT_SYSTEM = (
|
|
99
|
+
"You are an autonomous agent capable of using tools to interact with your environment. "
|
|
100
|
+
"When a task requires information you don't have, use a tool to get it. "
|
|
101
|
+
"Always reason out loud before taking an action. "
|
|
102
|
+
"If you have all the information needed, provide a final answer. "
|
|
103
|
+
"Be precise and thorough."
|
|
104
|
+
)
|
|
105
|
+
|
|
106
|
+
REVIEW_SYSTEM = (
|
|
107
|
+
"You are a critical reviewer. You will receive a task and a response to it.\n"
|
|
108
|
+
"Review the response for:\n"
|
|
109
|
+
"- Correctness and completeness\n"
|
|
110
|
+
"- Missing edge cases or errors\n"
|
|
111
|
+
"- Clarity and quality\n\n"
|
|
112
|
+
"Format your review as:\n"
|
|
113
|
+
"VERDICT: APPROVED | NEEDS_WORK\n"
|
|
114
|
+
"ISSUES:\n- issue 1\n- issue 2\n"
|
|
115
|
+
"SUGGESTIONS:\n- suggestion 1\n"
|
|
116
|
+
"If APPROVED with no issues, just say: VERDICT: APPROVED"
|
|
117
|
+
)
|
|
118
|
+
|
|
119
|
+
CORRECTIONS_SYSTEM = (
|
|
120
|
+
"You are revising your previous response based on reviewer feedback. "
|
|
121
|
+
"Apply all suggested corrections. Output the complete corrected response only — "
|
|
122
|
+
"no meta-commentary about what changed."
|
|
123
|
+
)
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
# ── Pipeline ──────────────────────────────────────────────────────────────────
|
|
127
|
+
|
|
128
|
+
class Pipeline:
|
|
129
|
+
def __init__(
|
|
130
|
+
self,
|
|
131
|
+
session_id: str,
|
|
132
|
+
primary_adapter: BaseAdapter,
|
|
133
|
+
primary_model: ModelInfo,
|
|
134
|
+
reviewer_adapter: BaseAdapter | None,
|
|
135
|
+
reviewer_model: ModelInfo | None,
|
|
136
|
+
summarizer_adapter: BaseAdapter | None,
|
|
137
|
+
summarizer_model: str,
|
|
138
|
+
review_rounds: int = 2,
|
|
139
|
+
active_project: str = "",
|
|
140
|
+
cfg: dict | None = None,
|
|
141
|
+
failover_adapter: BaseAdapter | None = None,
|
|
142
|
+
failover_model: ModelInfo | None = None,
|
|
143
|
+
# Callbacks for UI
|
|
144
|
+
on_step: Callable[[Step, str], None] | None = None,
|
|
145
|
+
on_token: Callable[[str], None] | None = None,
|
|
146
|
+
on_ask: Callable[[str], Any] | None = None, # returns or awaits user answer
|
|
147
|
+
on_permission: Callable[[PermissionRequest], Any] | None = None,
|
|
148
|
+
on_plan_approval: Callable[[str], Any] | None = None,
|
|
149
|
+
) -> None:
|
|
150
|
+
self.session_id = session_id
|
|
151
|
+
self.primary = primary_adapter
|
|
152
|
+
self.primary_model = primary_model
|
|
153
|
+
self.reviewer = reviewer_adapter
|
|
154
|
+
self.reviewer_model = reviewer_model
|
|
155
|
+
self.summarizer = summarizer_adapter
|
|
156
|
+
self.summarizer_model = summarizer_model
|
|
157
|
+
self.review_rounds = review_rounds
|
|
158
|
+
self.active_project = active_project
|
|
159
|
+
self.cfg = cfg or {}
|
|
160
|
+
self.failover_adapter = failover_adapter
|
|
161
|
+
self.failover_model = failover_model
|
|
162
|
+
self._speculative_reviewer: SpeculativeReviewer | None = None
|
|
163
|
+
self.on_step = on_step or (lambda s, t: None)
|
|
164
|
+
self.on_token = on_token or (lambda t: None)
|
|
165
|
+
self.on_ask = on_ask or (lambda q: "")
|
|
166
|
+
self.on_permission = on_permission or (lambda q: True)
|
|
167
|
+
# Plan approval: receives plan text, returns "proceed"|"cancel"|"refine:<text>"
|
|
168
|
+
self.on_plan_approval = on_plan_approval or (lambda plan: "proceed")
|
|
169
|
+
|
|
170
|
+
async def _resolve_callback(self, callback_result: Any) -> Any:
|
|
171
|
+
if inspect.isawaitable(callback_result):
|
|
172
|
+
return await callback_result
|
|
173
|
+
return callback_result
|
|
174
|
+
|
|
175
|
+
async def run(
|
|
176
|
+
self,
|
|
177
|
+
user_input: str,
|
|
178
|
+
complexity: TaskComplexity,
|
|
179
|
+
task_type: TaskType,
|
|
180
|
+
active_project: str = "",
|
|
181
|
+
) -> PipelineResult:
|
|
182
|
+
# active_project may be passed directly; fall back to instance attribute
|
|
183
|
+
effective_project = active_project or self.active_project
|
|
184
|
+
result = PipelineResult(
|
|
185
|
+
primary_model=self.primary_model.display_name,
|
|
186
|
+
reviewer_model=self.reviewer_model.display_name if self.reviewer_model else "",
|
|
187
|
+
)
|
|
188
|
+
|
|
189
|
+
# Save user message
|
|
190
|
+
add_message(self.session_id, "user", user_input, step=Step.EXECUTE.value)
|
|
191
|
+
|
|
192
|
+
# Optional speculative reviewer — observes agent loop, primes final review
|
|
193
|
+
if (
|
|
194
|
+
self.reviewer is not None
|
|
195
|
+
and self.reviewer_model is not None
|
|
196
|
+
and memory_setting(self.cfg, "speculative_review")
|
|
197
|
+
):
|
|
198
|
+
self._speculative_reviewer = SpeculativeReviewer(
|
|
199
|
+
adapter=self.reviewer,
|
|
200
|
+
model_id=self.reviewer_model.id,
|
|
201
|
+
task=user_input,
|
|
202
|
+
)
|
|
203
|
+
|
|
204
|
+
# ── Pipeline logic ──────────────────────────────────────────────────
|
|
205
|
+
plan_mode = (self.cfg.get("plan_mode") or "always").lower()
|
|
206
|
+
plan_for_simple = plan_mode in ("always",)
|
|
207
|
+
plan_for_medium = plan_mode in ("always", "complexity")
|
|
208
|
+
plan_for_complex = plan_mode in ("always", "complexity")
|
|
209
|
+
|
|
210
|
+
if complexity == TaskComplexity.SIMPLE:
|
|
211
|
+
# Plan first (when on), approve, clarify, then execute
|
|
212
|
+
if plan_for_simple:
|
|
213
|
+
await self._plan(user_input, result)
|
|
214
|
+
if not await self._await_plan_approval(user_input, result):
|
|
215
|
+
return result
|
|
216
|
+
await self._clarify(user_input, result, round_num=1)
|
|
217
|
+
await self._agentic_execute(user_input, result, task_type=task_type, active_project=effective_project)
|
|
218
|
+
elif complexity == TaskComplexity.MEDIUM:
|
|
219
|
+
# Clarify intent → plan → approve → clarify again with refined context → execute
|
|
220
|
+
await self._clarify(user_input, result, round_num=1)
|
|
221
|
+
if plan_for_medium:
|
|
222
|
+
await self._plan(user_input, result)
|
|
223
|
+
if not await self._await_plan_approval(user_input, result):
|
|
224
|
+
return result
|
|
225
|
+
await self._clarify(user_input, result, round_num=2)
|
|
226
|
+
await self._agentic_execute(user_input, result, task_type=task_type, active_project=effective_project)
|
|
227
|
+
else: # COMPLEX
|
|
228
|
+
# Understand deeply → clarify → plan → approve → clarify again → execute
|
|
229
|
+
await self._understand(user_input, result)
|
|
230
|
+
await self._clarify(user_input, result, round_num=1)
|
|
231
|
+
if plan_for_complex:
|
|
232
|
+
await self._plan(user_input, result)
|
|
233
|
+
if not await self._await_plan_approval(user_input, result):
|
|
234
|
+
return result
|
|
235
|
+
await self._clarify(user_input, result, round_num=2)
|
|
236
|
+
await self._agentic_execute(user_input, result, task_type=task_type, active_project=effective_project)
|
|
237
|
+
|
|
238
|
+
# Review loop
|
|
239
|
+
if self.reviewer and self.reviewer_model:
|
|
240
|
+
await self._review_loop(user_input, result)
|
|
241
|
+
|
|
242
|
+
result.steps_run.append(Step.OUTPUT.value)
|
|
243
|
+
self.on_step(Step.OUTPUT, result.output)
|
|
244
|
+
|
|
245
|
+
# Export session for cross-model unified context persistence
|
|
246
|
+
try:
|
|
247
|
+
export_session_md(self.session_id)
|
|
248
|
+
except Exception as e:
|
|
249
|
+
logger.warning("export_session_md failed for session %s: %s", self.session_id, e)
|
|
250
|
+
|
|
251
|
+
# Export to project folder if a project is active
|
|
252
|
+
if effective_project:
|
|
253
|
+
try:
|
|
254
|
+
messages = get_messages(self.session_id)
|
|
255
|
+
export_session_to_project(
|
|
256
|
+
project=effective_project,
|
|
257
|
+
session_id=self.session_id,
|
|
258
|
+
messages=messages,
|
|
259
|
+
meta={"primary_model": result.primary_model, "steps_run": result.steps_run},
|
|
260
|
+
)
|
|
261
|
+
except Exception as e:
|
|
262
|
+
logger.warning("export_session_to_project failed for %s: %s", effective_project, e)
|
|
263
|
+
|
|
264
|
+
# Auto-append key decisions to decisions.md for COMPLEX tasks
|
|
265
|
+
if effective_project and complexity == TaskComplexity.COMPLEX:
|
|
266
|
+
try:
|
|
267
|
+
await self._maybe_append_decisions(result, effective_project)
|
|
268
|
+
except Exception as e:
|
|
269
|
+
logger.warning("decisions append failed for %s: %s", effective_project, e)
|
|
270
|
+
|
|
271
|
+
# Memory writeback: extract durable facts from this run as candidate chunks.
|
|
272
|
+
# OFF by default — prefer the deliberate `/learn-from <correction>` flow,
|
|
273
|
+
# which is Boris-style (manual, 100% reliable, no candidate→confirmed graveyard).
|
|
274
|
+
# Opt in by setting cfg["memory"]["auto_writeback"] = True.
|
|
275
|
+
if effective_project and self.cfg.get("memory", {}).get("auto_writeback", False):
|
|
276
|
+
try:
|
|
277
|
+
await memory_write_back(
|
|
278
|
+
session_id=self.session_id,
|
|
279
|
+
project=effective_project,
|
|
280
|
+
primary_adapter=self.primary,
|
|
281
|
+
primary_model_id=self.primary_model.id,
|
|
282
|
+
cfg=self.cfg,
|
|
283
|
+
)
|
|
284
|
+
except Exception as e:
|
|
285
|
+
logger.warning("memory writeback failed for %s: %s", effective_project, e)
|
|
286
|
+
|
|
287
|
+
return result
|
|
288
|
+
|
|
289
|
+
# ── Steps ─────────────────────────────────────────────────────────────────
|
|
290
|
+
|
|
291
|
+
async def _agentic_execute(
|
|
292
|
+
self,
|
|
293
|
+
user_input: str,
|
|
294
|
+
result: PipelineResult,
|
|
295
|
+
task_type: TaskType = TaskType.GENERAL,
|
|
296
|
+
active_project: str = "",
|
|
297
|
+
) -> None:
|
|
298
|
+
"""The main agentic loop: REASON -> ACT -> OBSERVE."""
|
|
299
|
+
self.on_step(Step.EXECUTE, "")
|
|
300
|
+
|
|
301
|
+
# Build initial context with workspace awareness + typed project context
|
|
302
|
+
context = await self._get_context()
|
|
303
|
+
ws_context = get_workspace_context()
|
|
304
|
+
effective_project = active_project or self.active_project
|
|
305
|
+
project_ctx = (
|
|
306
|
+
await retrieve_or_fallback(effective_project, user_input, task_type, self.cfg)
|
|
307
|
+
if effective_project else ""
|
|
308
|
+
)
|
|
309
|
+
system_prompt = AGENT_SYSTEM
|
|
310
|
+
if project_ctx:
|
|
311
|
+
system_prompt += f"\n\n{project_ctx}"
|
|
312
|
+
system_prompt += f"\n\n{ws_context}"
|
|
313
|
+
|
|
314
|
+
context.append(Message(role="user", content=user_input))
|
|
315
|
+
|
|
316
|
+
max_turns = 10
|
|
317
|
+
for turn in range(max_turns):
|
|
318
|
+
self.on_step(Step.REASON, f"Turn {turn + 1}")
|
|
319
|
+
|
|
320
|
+
# 1. Complete with tool support
|
|
321
|
+
response = await self.primary.complete(
|
|
322
|
+
messages=context,
|
|
323
|
+
model_id=self.primary_model.id,
|
|
324
|
+
system=system_prompt,
|
|
325
|
+
tools=tool_registry.list_tools()
|
|
326
|
+
)
|
|
327
|
+
|
|
328
|
+
if isinstance(response, str):
|
|
329
|
+
# Final answer
|
|
330
|
+
result.output = response
|
|
331
|
+
self.on_step(Step.PREVIEW, response)
|
|
332
|
+
add_message(
|
|
333
|
+
self.session_id, "assistant", response,
|
|
334
|
+
model=self.primary_model.id, step=Step.EXECUTE.value
|
|
335
|
+
)
|
|
336
|
+
if self._speculative_reviewer is not None:
|
|
337
|
+
self._speculative_reviewer.kick_off(response)
|
|
338
|
+
break
|
|
339
|
+
|
|
340
|
+
# Handle tool calls
|
|
341
|
+
assistant_msg = response
|
|
342
|
+
context.append(assistant_msg)
|
|
343
|
+
add_message(
|
|
344
|
+
self.session_id, "assistant", assistant_msg.content,
|
|
345
|
+
model=self.primary_model.id, step=Step.ACT.value
|
|
346
|
+
)
|
|
347
|
+
|
|
348
|
+
if assistant_msg.content:
|
|
349
|
+
self.on_step(Step.REASON, assistant_msg.content)
|
|
350
|
+
if self._speculative_reviewer is not None and len(assistant_msg.content) > 200:
|
|
351
|
+
self._speculative_reviewer.kick_off(assistant_msg.content)
|
|
352
|
+
|
|
353
|
+
for tc in assistant_msg.tool_calls:
|
|
354
|
+
self.on_step(Step.ACT, f"Calling {tc.name}({tc.arguments})")
|
|
355
|
+
|
|
356
|
+
# Check permission if needed
|
|
357
|
+
tool = tool_registry.get(tc.name)
|
|
358
|
+
if tool and tool.needs_permission(tc.arguments):
|
|
359
|
+
if not await self._resolve_callback(
|
|
360
|
+
self.on_permission(
|
|
361
|
+
build_permission_request(tc.name, tc.arguments)
|
|
362
|
+
)
|
|
363
|
+
):
|
|
364
|
+
obs = "Error: Permission denied by user."
|
|
365
|
+
else:
|
|
366
|
+
obs = await tool_registry.execute(tc.name, tc.arguments)
|
|
367
|
+
else:
|
|
368
|
+
obs = await tool_registry.execute(tc.name, tc.arguments)
|
|
369
|
+
|
|
370
|
+
self.on_step(Step.OBSERVE, obs[:500] + "..." if len(obs) > 500 else obs)
|
|
371
|
+
|
|
372
|
+
# Add observation to context
|
|
373
|
+
context.append(Message(role="tool", content=obs, model=tc.name, tool_call_id=tc.id))
|
|
374
|
+
add_message(self.session_id, "tool", obs, model=tc.name, step=Step.OBSERVE.value)
|
|
375
|
+
|
|
376
|
+
result.steps_run.append(Step.EXECUTE.value)
|
|
377
|
+
|
|
378
|
+
async def _understand(self, user_input: str, result: PipelineResult) -> None:
|
|
379
|
+
self.on_step(Step.UNDERSTAND, "")
|
|
380
|
+
response = await self._call_primary(
|
|
381
|
+
extra_user=f"Task: {user_input}",
|
|
382
|
+
system=UNDERSTAND_SYSTEM,
|
|
383
|
+
step=Step.UNDERSTAND,
|
|
384
|
+
)
|
|
385
|
+
result.steps_run.append(Step.UNDERSTAND.value)
|
|
386
|
+
|
|
387
|
+
async def _clarify(self, user_input: str, result: PipelineResult, round_num: int) -> None:
|
|
388
|
+
step = Step.CLARIFY_1 if round_num == 1 else Step.CLARIFY_2
|
|
389
|
+
self.on_step(step, "")
|
|
390
|
+
|
|
391
|
+
questions_raw = await self._call_primary(
|
|
392
|
+
extra_user=f"Original task: {user_input}\nDo you need clarification?",
|
|
393
|
+
system=CLARIFY_SYSTEM,
|
|
394
|
+
step=step,
|
|
395
|
+
)
|
|
396
|
+
|
|
397
|
+
if "NO_CLARIFICATION_NEEDED" in questions_raw.upper():
|
|
398
|
+
return
|
|
399
|
+
|
|
400
|
+
questions = _extract_questions(questions_raw)
|
|
401
|
+
if not questions:
|
|
402
|
+
return
|
|
403
|
+
|
|
404
|
+
for i, q in enumerate(questions[:3], 1):
|
|
405
|
+
answer = await self._resolve_callback(
|
|
406
|
+
self.on_ask(f"[Clarification {i}/{len(questions[:3])}] {q}")
|
|
407
|
+
)
|
|
408
|
+
if answer:
|
|
409
|
+
add_message(self.session_id, "user", answer, step=step.value)
|
|
410
|
+
|
|
411
|
+
result.steps_run.append(step.value)
|
|
412
|
+
|
|
413
|
+
async def _plan(self, user_input: str, result: PipelineResult) -> None:
|
|
414
|
+
self.on_step(Step.PLAN, "")
|
|
415
|
+
plan = await self._call_primary(
|
|
416
|
+
extra_user=f"Create a plan for: {user_input}",
|
|
417
|
+
system=PLAN_SYSTEM,
|
|
418
|
+
step=Step.PLAN,
|
|
419
|
+
)
|
|
420
|
+
result.plan = plan
|
|
421
|
+
result.steps_run.append(Step.PLAN.value)
|
|
422
|
+
|
|
423
|
+
async def _await_plan_approval(self, user_input: str, result: PipelineResult) -> bool:
|
|
424
|
+
"""
|
|
425
|
+
Show plan to user, loop until approved or cancelled.
|
|
426
|
+
Returns True to proceed, False to cancel.
|
|
427
|
+
Accepts up to 3 refinement rounds before auto-proceeding.
|
|
428
|
+
"""
|
|
429
|
+
for _ in range(3):
|
|
430
|
+
decision = await self._resolve_callback(self.on_plan_approval(result.plan))
|
|
431
|
+
d = decision.strip().lower()
|
|
432
|
+
if d in ("", "y", "yes", "proceed"):
|
|
433
|
+
return True
|
|
434
|
+
if d == "cancel":
|
|
435
|
+
result.steps_run.append("cancelled")
|
|
436
|
+
return False
|
|
437
|
+
if d.startswith("refine:"):
|
|
438
|
+
feedback = decision[7:].strip()
|
|
439
|
+
await self._plan(f"{user_input}\n\nFeedback on previous plan: {feedback}", result)
|
|
440
|
+
else:
|
|
441
|
+
# Unrecognised input — treat as proceed
|
|
442
|
+
return True
|
|
443
|
+
return True # auto-proceed after max refinements
|
|
444
|
+
|
|
445
|
+
async def _maybe_append_decisions(self, result: PipelineResult, project: str) -> None:
|
|
446
|
+
"""
|
|
447
|
+
Extract up to 5 key decisions from a COMPLEX task output and append to decisions.md.
|
|
448
|
+
Only runs when output is substantial (>200 chars).
|
|
449
|
+
"""
|
|
450
|
+
if len(result.output) < 200:
|
|
451
|
+
return
|
|
452
|
+
prompt = (
|
|
453
|
+
"Extract up to 5 key decisions or conclusions from this AI session output "
|
|
454
|
+
"as brief bullet points (start each with '- '). Only include genuinely "
|
|
455
|
+
"decision-worthy items (architectural choices, important findings, chosen "
|
|
456
|
+
"approaches, tradeoffs accepted). If none exist, respond with exactly: "
|
|
457
|
+
"NO_DECISIONS\n\n"
|
|
458
|
+
f"Output:\n{result.output[:4000]}"
|
|
459
|
+
)
|
|
460
|
+
resp = await self.primary.complete(
|
|
461
|
+
messages=[Message(role="user", content=prompt)],
|
|
462
|
+
model_id=self.primary_model.id,
|
|
463
|
+
system="Extract decisions only. Be terse.",
|
|
464
|
+
temperature=0.2,
|
|
465
|
+
)
|
|
466
|
+
text = resp if isinstance(resp, str) else resp.content
|
|
467
|
+
if "NO_DECISIONS" in text.upper():
|
|
468
|
+
return
|
|
469
|
+
from polymath.context_manager import read_context, write_context
|
|
470
|
+
from datetime import datetime, timezone
|
|
471
|
+
timestamp = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M UTC")
|
|
472
|
+
existing = read_context(project, "decisions")
|
|
473
|
+
if existing and not existing.endswith("\n"):
|
|
474
|
+
existing += "\n"
|
|
475
|
+
entry = f"\n### {timestamp}\n{text.strip()}\n"
|
|
476
|
+
write_context(project, "decisions", existing + entry)
|
|
477
|
+
|
|
478
|
+
async def _review_loop(self, user_input: str, result: PipelineResult) -> None:
|
|
479
|
+
for round_num in range(self.review_rounds):
|
|
480
|
+
step = Step.REVIEW if round_num == 0 else Step.RE_REVIEW
|
|
481
|
+
self.on_step(step, "")
|
|
482
|
+
|
|
483
|
+
if round_num == 0 and self._speculative_reviewer is not None:
|
|
484
|
+
review = await self._speculative_reviewer.finalize(result.output)
|
|
485
|
+
else:
|
|
486
|
+
review_prompt = (
|
|
487
|
+
f"ORIGINAL TASK:\n{user_input}\n\n"
|
|
488
|
+
f"RESPONSE TO REVIEW:\n{result.output}"
|
|
489
|
+
)
|
|
490
|
+
review = await self.reviewer.complete(
|
|
491
|
+
messages=[Message(role="user", content=review_prompt)],
|
|
492
|
+
model_id=self.reviewer_model.id,
|
|
493
|
+
system=REVIEW_SYSTEM,
|
|
494
|
+
temperature=0.3,
|
|
495
|
+
)
|
|
496
|
+
add_message(
|
|
497
|
+
self.session_id, "assistant", review,
|
|
498
|
+
model=self.reviewer_model.id, step=step.value,
|
|
499
|
+
)
|
|
500
|
+
result.review_notes.append(review)
|
|
501
|
+
self.on_step(step, review)
|
|
502
|
+
|
|
503
|
+
if "VERDICT: APPROVED" in review and "NEEDS_WORK" not in review:
|
|
504
|
+
break
|
|
505
|
+
|
|
506
|
+
# Apply corrections
|
|
507
|
+
self.on_step(Step.CORRECTIONS, "")
|
|
508
|
+
correction_prompt = (
|
|
509
|
+
f"Your previous response:\n{result.output}\n\n"
|
|
510
|
+
f"Reviewer feedback:\n{review}\n\n"
|
|
511
|
+
"Apply all corrections and output the complete revised response."
|
|
512
|
+
)
|
|
513
|
+
corrected = await self._call_primary(
|
|
514
|
+
extra_user=correction_prompt,
|
|
515
|
+
system=CORRECTIONS_SYSTEM,
|
|
516
|
+
step=Step.CORRECTIONS,
|
|
517
|
+
stream=True,
|
|
518
|
+
)
|
|
519
|
+
if isinstance(corrected, str):
|
|
520
|
+
result.output = corrected
|
|
521
|
+
else:
|
|
522
|
+
result.output = corrected.content
|
|
523
|
+
|
|
524
|
+
result.steps_run.append(step.value)
|
|
525
|
+
result.steps_run.append(Step.CORRECTIONS.value)
|
|
526
|
+
|
|
527
|
+
# ── Helpers ───────────────────────────────────────────────────────────────
|
|
528
|
+
|
|
529
|
+
async def _call_primary(
|
|
530
|
+
self,
|
|
531
|
+
extra_user: str,
|
|
532
|
+
system: str,
|
|
533
|
+
step: Step,
|
|
534
|
+
stream: bool = False,
|
|
535
|
+
) -> str | Message:
|
|
536
|
+
context = await self._get_context()
|
|
537
|
+
context.append(Message(role="user", content=extra_user))
|
|
538
|
+
|
|
539
|
+
if stream:
|
|
540
|
+
output = ""
|
|
541
|
+
async for token in self.primary.stream(
|
|
542
|
+
messages=context, model_id=self.primary_model.id, system=system
|
|
543
|
+
):
|
|
544
|
+
if isinstance(token, str):
|
|
545
|
+
output += token
|
|
546
|
+
self.on_token(token)
|
|
547
|
+
else:
|
|
548
|
+
# Tool calls not expected in simple calls, but just in case
|
|
549
|
+
pass
|
|
550
|
+
add_message(
|
|
551
|
+
self.session_id, "assistant", output,
|
|
552
|
+
model=self.primary_model.id, step=step.value,
|
|
553
|
+
)
|
|
554
|
+
return output
|
|
555
|
+
|
|
556
|
+
resp, winner_model = await self._race_complete(
|
|
557
|
+
context=context, system=system, temperature=0.5,
|
|
558
|
+
)
|
|
559
|
+
content = resp if isinstance(resp, str) else resp.content
|
|
560
|
+
add_message(
|
|
561
|
+
self.session_id, "assistant", content,
|
|
562
|
+
model=winner_model.id, step=step.value,
|
|
563
|
+
)
|
|
564
|
+
return resp
|
|
565
|
+
|
|
566
|
+
async def _race_complete(
|
|
567
|
+
self,
|
|
568
|
+
context: list[Message],
|
|
569
|
+
system: str,
|
|
570
|
+
temperature: float = 0.5,
|
|
571
|
+
) -> tuple[Any, ModelInfo]:
|
|
572
|
+
"""If a failover adapter is configured, run primary + failover in parallel
|
|
573
|
+
with delay; first success wins. Returns (response, winning_model_info)."""
|
|
574
|
+
delay = float(memory_setting(self.cfg, "race_delay_seconds"))
|
|
575
|
+
|
|
576
|
+
async def call(adapter: BaseAdapter, model_id: str):
|
|
577
|
+
return await adapter.complete(
|
|
578
|
+
messages=context, model_id=model_id, system=system, temperature=temperature,
|
|
579
|
+
)
|
|
580
|
+
|
|
581
|
+
primary = RaceCandidate(
|
|
582
|
+
id=self.primary_model.id,
|
|
583
|
+
factory=lambda: call(self.primary, self.primary_model.id),
|
|
584
|
+
)
|
|
585
|
+
if self.failover_adapter is None or self.failover_model is None:
|
|
586
|
+
return await primary.factory(), self.primary_model
|
|
587
|
+
|
|
588
|
+
fallback = RaceCandidate(
|
|
589
|
+
id=self.failover_model.id,
|
|
590
|
+
factory=lambda: call(self.failover_adapter, self.failover_model.id),
|
|
591
|
+
)
|
|
592
|
+
try:
|
|
593
|
+
outcome = await with_failover(primary, fallback, delay_seconds=delay)
|
|
594
|
+
except AllCandidatesFailed as e:
|
|
595
|
+
primary_exc = e.failures.get(primary.id)
|
|
596
|
+
if primary_exc:
|
|
597
|
+
raise primary_exc
|
|
598
|
+
raise
|
|
599
|
+
|
|
600
|
+
winner_model = self.primary_model if outcome.winner_id == primary.id else self.failover_model
|
|
601
|
+
return outcome.result, winner_model
|
|
602
|
+
|
|
603
|
+
async def _get_context(self) -> list[Message]:
|
|
604
|
+
summarizer_adapter = self.summarizer
|
|
605
|
+
summarizer_model_id = self.summarizer_model
|
|
606
|
+
return await build_context(
|
|
607
|
+
self.session_id,
|
|
608
|
+
summarizer=summarizer_adapter,
|
|
609
|
+
summarizer_model=summarizer_model_id,
|
|
610
|
+
)
|
|
611
|
+
|
|
612
|
+
|
|
613
|
+
async def quick_answer(
|
|
614
|
+
session_id: str,
|
|
615
|
+
question: str,
|
|
616
|
+
primary_adapter: BaseAdapter,
|
|
617
|
+
primary_model: ModelInfo,
|
|
618
|
+
active_project: str = "",
|
|
619
|
+
task_type: TaskType = TaskType.GENERAL,
|
|
620
|
+
summarizer_adapter: BaseAdapter | None = None,
|
|
621
|
+
summarizer_model: str = "",
|
|
622
|
+
on_token: Callable[[str], None] | None = None,
|
|
623
|
+
cfg: dict | None = None,
|
|
624
|
+
failover_adapter: BaseAdapter | None = None,
|
|
625
|
+
failover_model: ModelInfo | None = None,
|
|
626
|
+
) -> str:
|
|
627
|
+
"""
|
|
628
|
+
Lightweight path — no pipeline, no review, no clarify.
|
|
629
|
+
Injects typed project context + recent history and answers directly.
|
|
630
|
+
Used by --ask flag and `polymath ask` CLI.
|
|
631
|
+
|
|
632
|
+
If a failover adapter is provided, races primary + failover at the stream
|
|
633
|
+
level: first to emit a token wins the terminal, loser is silently cancelled.
|
|
634
|
+
"""
|
|
635
|
+
on_token = on_token or (lambda t: None)
|
|
636
|
+
cfg = cfg or {}
|
|
637
|
+
|
|
638
|
+
context = await build_context(
|
|
639
|
+
session_id,
|
|
640
|
+
summarizer=summarizer_adapter,
|
|
641
|
+
summarizer_model=summarizer_model,
|
|
642
|
+
)
|
|
643
|
+
context.append(Message(role="user", content=question))
|
|
644
|
+
|
|
645
|
+
project_ctx = (
|
|
646
|
+
await retrieve_or_fallback(active_project, question, task_type, cfg)
|
|
647
|
+
if active_project else ""
|
|
648
|
+
)
|
|
649
|
+
system = "Answer concisely and directly."
|
|
650
|
+
if project_ctx:
|
|
651
|
+
system = f"{project_ctx}\n\n{system}"
|
|
652
|
+
|
|
653
|
+
if failover_adapter is not None and failover_model is not None:
|
|
654
|
+
def make_stream(adapter: BaseAdapter, model_id: str):
|
|
655
|
+
def factory():
|
|
656
|
+
return adapter.stream(
|
|
657
|
+
messages=context,
|
|
658
|
+
model_id=model_id,
|
|
659
|
+
system=system,
|
|
660
|
+
)
|
|
661
|
+
return factory
|
|
662
|
+
|
|
663
|
+
outcome = await race_streams(
|
|
664
|
+
candidates=[
|
|
665
|
+
StreamCandidate(id=primary_model.id, factory=make_stream(primary_adapter, primary_model.id)),
|
|
666
|
+
StreamCandidate(id=failover_model.id, factory=make_stream(failover_adapter, failover_model.id)),
|
|
667
|
+
],
|
|
668
|
+
on_token=on_token,
|
|
669
|
+
)
|
|
670
|
+
output = outcome.output
|
|
671
|
+
winner_id = outcome.winner_id
|
|
672
|
+
else:
|
|
673
|
+
output = ""
|
|
674
|
+
async for token in primary_adapter.stream(
|
|
675
|
+
messages=context,
|
|
676
|
+
model_id=primary_model.id,
|
|
677
|
+
system=system,
|
|
678
|
+
):
|
|
679
|
+
if isinstance(token, str):
|
|
680
|
+
output += token
|
|
681
|
+
on_token(token)
|
|
682
|
+
winner_id = primary_model.id
|
|
683
|
+
|
|
684
|
+
add_message(session_id, "user", question, step="ask")
|
|
685
|
+
add_message(session_id, "assistant", output, model=winner_id, step="ask")
|
|
686
|
+
export_session_md(session_id)
|
|
687
|
+
return output
|
|
688
|
+
|
|
689
|
+
|
|
690
|
+
def _extract_questions(text: str) -> list[str]:
|
|
691
|
+
lines = text.strip().splitlines()
|
|
692
|
+
questions = []
|
|
693
|
+
for line in lines:
|
|
694
|
+
line = line.strip()
|
|
695
|
+
match = re.match(r"^[1-3][.)]\s*(.+)", line)
|
|
696
|
+
if match:
|
|
697
|
+
questions.append(match.group(1))
|
|
698
|
+
elif line.endswith("?") and len(line) > 10:
|
|
699
|
+
questions.append(line)
|
|
700
|
+
return questions[:3]
|