polymath-agent 0.4.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- polymath/__init__.py +2 -0
- polymath/adapters/__init__.py +7 -0
- polymath/adapters/base.py +175 -0
- polymath/adapters/claude.py +280 -0
- polymath/adapters/gemini.py +186 -0
- polymath/adapters/ollama.py +117 -0
- polymath/adapters/openai_adapter.py +168 -0
- polymath/bootstrap.py +159 -0
- polymath/command_registry.py +41 -0
- polymath/command_service.py +572 -0
- polymath/compressor.py +90 -0
- polymath/config.py +293 -0
- polymath/context_manager.py +76 -0
- polymath/context_store.py +336 -0
- polymath/detector.py +442 -0
- polymath/domain.py +78 -0
- polymath/execution_service.py +325 -0
- polymath/main.py +1293 -0
- polymath/memory/__init__.py +15 -0
- polymath/memory/chunker.py +6 -0
- polymath/memory/embedder.py +179 -0
- polymath/memory/migrate.py +2 -0
- polymath/memory/retriever.py +2 -0
- polymath/memory/store.py +9 -0
- polymath/memory/sync.py +2 -0
- polymath/memory/writer.py +9 -0
- polymath/model_policy.py +172 -0
- polymath/orchestrator/__init__.py +68 -0
- polymath/orchestrator/attempt_ledger.py +34 -0
- polymath/orchestrator/ensemble.py +229 -0
- polymath/orchestrator/fanout.py +322 -0
- polymath/orchestrator/output_policy.py +61 -0
- polymath/orchestrator/race.py +311 -0
- polymath/orchestrator/run_controller.py +91 -0
- polymath/orchestrator/speculative_review.py +120 -0
- polymath/orchestrator/state_responder.py +184 -0
- polymath/orchestrator/worker_pool.py +37 -0
- polymath/permissions.py +82 -0
- polymath/pipeline.py +700 -0
- polymath/project_config.py +229 -0
- polymath/project_runtime.py +109 -0
- polymath/router.py +127 -0
- polymath/setup_wizard.py +106 -0
- polymath/slash_commands.py +566 -0
- polymath/subagents.py +486 -0
- polymath/tools.py +333 -0
- polymath/ui_state.py +84 -0
- polymath/workspace.py +66 -0
- polymath_agent-0.4.0.dist-info/METADATA +693 -0
- polymath_agent-0.4.0.dist-info/RECORD +54 -0
- polymath_agent-0.4.0.dist-info/WHEEL +5 -0
- polymath_agent-0.4.0.dist-info/entry_points.txt +2 -0
- polymath_agent-0.4.0.dist-info/licenses/LICENSE +21 -0
- polymath_agent-0.4.0.dist-info/top_level.txt +1 -0
polymath/subagents.py
ADDED
|
@@ -0,0 +1,486 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Named post-processing subagents for polymath.
|
|
3
|
+
|
|
4
|
+
A subagent is a lightweight mini-pipeline that runs *after* the main pipeline
|
|
5
|
+
output is ready. Each subagent has a focused system prompt and optional
|
|
6
|
+
iterative loop (e.g. verify runs tests, fixes, re-runs).
|
|
7
|
+
|
|
8
|
+
Built-in subagents
|
|
9
|
+
──────────────────
|
|
10
|
+
code-simplifier Simplify / refactor after main work
|
|
11
|
+
verify Detect test runner, run tests, iterate on failures (max 3×)
|
|
12
|
+
security-scan Flag security issues (injection, hardcoded creds, etc.)
|
|
13
|
+
test-writer Generate test cases for the produced code
|
|
14
|
+
docs-writer Generate or update inline documentation
|
|
15
|
+
brainstorm-critic Challenge assumptions, surface edge cases
|
|
16
|
+
perf-reviewer Spot performance hotspots in code output
|
|
17
|
+
diff-explainer Plain-language explanation of what changed and why
|
|
18
|
+
|
|
19
|
+
Custom subagents
|
|
20
|
+
────────────────
|
|
21
|
+
Place .md files in .polymath/subagents/ (project) or ~/.polymath/subagents/ (global).
|
|
22
|
+
Frontmatter fields: name, description, max_iterations, requires_tool, trigger_on (comma list)
|
|
23
|
+
Body: the system prompt.
|
|
24
|
+
"""
|
|
25
|
+
from __future__ import annotations
|
|
26
|
+
|
|
27
|
+
import json
|
|
28
|
+
import re
|
|
29
|
+
from dataclasses import dataclass, field
|
|
30
|
+
from pathlib import Path
|
|
31
|
+
from typing import Any, Callable
|
|
32
|
+
|
|
33
|
+
from polymath.config import CONFIG_DIR, TaskType
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass
|
|
37
|
+
class SubagentConfig:
|
|
38
|
+
name: str
|
|
39
|
+
description: str
|
|
40
|
+
system_prompt: str
|
|
41
|
+
max_iterations: int = 1
|
|
42
|
+
requires_tool: str = "" # skip subagent if this tool is unavailable
|
|
43
|
+
trigger_on: list[TaskType] = field(default_factory=list)
|
|
44
|
+
source: str = "builtin"
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass
|
|
48
|
+
class SubagentResult:
|
|
49
|
+
name: str
|
|
50
|
+
output: str
|
|
51
|
+
iterations: int
|
|
52
|
+
success: bool
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
# ── Built-in subagent definitions ─────────────────────────────────────────────
|
|
56
|
+
|
|
57
|
+
BUILTIN_SUBAGENTS: dict[str, SubagentConfig] = {
|
|
58
|
+
|
|
59
|
+
"code-simplifier": SubagentConfig(
|
|
60
|
+
name="code-simplifier",
|
|
61
|
+
description="Simplify and clean code output after the main pipeline.",
|
|
62
|
+
system_prompt=(
|
|
63
|
+
"You are a code simplification expert. "
|
|
64
|
+
"Review the provided code output and:\n"
|
|
65
|
+
"1. Remove unnecessary complexity and dead code\n"
|
|
66
|
+
"2. Simplify logic without changing behaviour\n"
|
|
67
|
+
"3. Improve naming clarity\n"
|
|
68
|
+
"4. Output only the simplified code — no commentary."
|
|
69
|
+
),
|
|
70
|
+
trigger_on=[TaskType.CODE],
|
|
71
|
+
),
|
|
72
|
+
|
|
73
|
+
"verify": SubagentConfig(
|
|
74
|
+
name="verify",
|
|
75
|
+
description="Run the project's test suite and fix failures (up to 3 iterations).",
|
|
76
|
+
system_prompt=(
|
|
77
|
+
"You are a debugging agent. Tests have failed. "
|
|
78
|
+
"Analyse the test output, identify root causes, and provide a corrected "
|
|
79
|
+
"implementation that fixes all failures. "
|
|
80
|
+
"Output only the corrected code — no explanation."
|
|
81
|
+
),
|
|
82
|
+
max_iterations=3,
|
|
83
|
+
requires_tool="run_shell_command",
|
|
84
|
+
trigger_on=[TaskType.CODE],
|
|
85
|
+
),
|
|
86
|
+
|
|
87
|
+
"security-scan": SubagentConfig(
|
|
88
|
+
name="security-scan",
|
|
89
|
+
description="Scan output for security issues: injection, hardcoded secrets, OWASP top-10.",
|
|
90
|
+
system_prompt=(
|
|
91
|
+
"You are a security reviewer. Scan the following code for:\n"
|
|
92
|
+
"- Injection vulnerabilities (SQL, command, XSS)\n"
|
|
93
|
+
"- Hardcoded secrets or credentials\n"
|
|
94
|
+
"- Insecure deserialization\n"
|
|
95
|
+
"- Missing input validation\n"
|
|
96
|
+
"- Broken authentication patterns\n\n"
|
|
97
|
+
"Format: VERDICT: CLEAN | ISSUES_FOUND\n"
|
|
98
|
+
"ISSUES:\n- <issue> at <location>\n"
|
|
99
|
+
"SEVERITY: LOW | MEDIUM | HIGH for each issue."
|
|
100
|
+
),
|
|
101
|
+
trigger_on=[TaskType.CODE],
|
|
102
|
+
),
|
|
103
|
+
|
|
104
|
+
"test-writer": SubagentConfig(
|
|
105
|
+
name="test-writer",
|
|
106
|
+
description="Generate comprehensive test cases for the produced code.",
|
|
107
|
+
system_prompt=(
|
|
108
|
+
"You are a test engineering expert. "
|
|
109
|
+
"Given the following code, write comprehensive tests covering:\n"
|
|
110
|
+
"1. Happy path\n"
|
|
111
|
+
"2. Edge cases and boundary values\n"
|
|
112
|
+
"3. Error conditions\n\n"
|
|
113
|
+
"Match the test framework already used in the project if detectable. "
|
|
114
|
+
"Output only the test code."
|
|
115
|
+
),
|
|
116
|
+
trigger_on=[TaskType.CODE],
|
|
117
|
+
),
|
|
118
|
+
|
|
119
|
+
"docs-writer": SubagentConfig(
|
|
120
|
+
name="docs-writer",
|
|
121
|
+
description="Generate or update inline documentation for the output.",
|
|
122
|
+
system_prompt=(
|
|
123
|
+
"You are a technical writer. "
|
|
124
|
+
"Add clear, concise docstrings/comments to the following code. "
|
|
125
|
+
"Document: purpose, parameters, return values, side effects, and "
|
|
126
|
+
"any non-obvious logic. Do not change the code itself. "
|
|
127
|
+
"Output the fully-documented version."
|
|
128
|
+
),
|
|
129
|
+
trigger_on=[TaskType.CODE],
|
|
130
|
+
),
|
|
131
|
+
|
|
132
|
+
"brainstorm-critic": SubagentConfig(
|
|
133
|
+
name="brainstorm-critic",
|
|
134
|
+
description="Challenge assumptions and surface edge cases in any output.",
|
|
135
|
+
system_prompt=(
|
|
136
|
+
"You are a rigorous critic. Review the following output and:\n"
|
|
137
|
+
"1. Identify unstated assumptions that may not hold\n"
|
|
138
|
+
"2. Surface edge cases or failure modes not addressed\n"
|
|
139
|
+
"3. Point out logical inconsistencies\n"
|
|
140
|
+
"4. Suggest alternative approaches worth considering\n\n"
|
|
141
|
+
"Be constructive but direct. Format as a bulleted list."
|
|
142
|
+
),
|
|
143
|
+
trigger_on=[TaskType.GENERAL, TaskType.ANALYSIS, TaskType.RESEARCH],
|
|
144
|
+
),
|
|
145
|
+
|
|
146
|
+
"perf-reviewer": SubagentConfig(
|
|
147
|
+
name="perf-reviewer",
|
|
148
|
+
description="Identify performance hotspots and inefficiencies in code output.",
|
|
149
|
+
system_prompt=(
|
|
150
|
+
"You are a performance engineering expert. "
|
|
151
|
+
"Review the following code for:\n"
|
|
152
|
+
"1. O(n²) or worse loops that could be optimised\n"
|
|
153
|
+
"2. Unnecessary memory allocations\n"
|
|
154
|
+
"3. N+1 query patterns\n"
|
|
155
|
+
"4. Blocking I/O in async contexts\n"
|
|
156
|
+
"5. Missing caching opportunities\n\n"
|
|
157
|
+
"Format: list each issue with location and suggested fix."
|
|
158
|
+
),
|
|
159
|
+
trigger_on=[TaskType.CODE],
|
|
160
|
+
),
|
|
161
|
+
|
|
162
|
+
"diff-explainer": SubagentConfig(
|
|
163
|
+
name="diff-explainer",
|
|
164
|
+
description="Explain what changed and why in plain language (great for PRs).",
|
|
165
|
+
system_prompt=(
|
|
166
|
+
"You are a senior developer writing a PR description. "
|
|
167
|
+
"Given the following code or diff, explain in plain language:\n"
|
|
168
|
+
"1. What changed\n"
|
|
169
|
+
"2. Why it changed (the motivation)\n"
|
|
170
|
+
"3. Any risks or things reviewers should pay attention to\n\n"
|
|
171
|
+
"Keep it concise — 3–5 bullet points."
|
|
172
|
+
),
|
|
173
|
+
trigger_on=[TaskType.CODE],
|
|
174
|
+
),
|
|
175
|
+
|
|
176
|
+
"brain-gap": SubagentConfig(
|
|
177
|
+
name="brain-gap",
|
|
178
|
+
description=(
|
|
179
|
+
"After a session, propose what should be added to the project "
|
|
180
|
+
"brain (rules.md / decisions.md) so the same mistake doesn't "
|
|
181
|
+
"repeat. Boris-style failure-driven brain growth."
|
|
182
|
+
),
|
|
183
|
+
system_prompt=(
|
|
184
|
+
"You are inspecting a finished session to find what the project "
|
|
185
|
+
"brain was MISSING that would have prevented mistakes or saved "
|
|
186
|
+
"the model time. Look for:\n"
|
|
187
|
+
"1. Corrections the user made to the model (e.g. 'no, do X instead')\n"
|
|
188
|
+
"2. Repeated clarification rounds that suggest missing context\n"
|
|
189
|
+
"3. Wrong assumptions the model made that a rule would fix\n"
|
|
190
|
+
"4. Decisions the user made that future sessions should respect\n\n"
|
|
191
|
+
"Output format — propose 0-5 entries, each tagged with the bucket:\n"
|
|
192
|
+
" [rules] one-sentence imperative rule (e.g. 'never commit to main')\n"
|
|
193
|
+
" [decisions] one-line decision + brief why\n"
|
|
194
|
+
" [glossary] term — definition\n\n"
|
|
195
|
+
"If nothing should be added, output the single word: NONE.\n"
|
|
196
|
+
"Be conservative — only propose entries that are durably useful "
|
|
197
|
+
"across sessions, not session-specific facts."
|
|
198
|
+
),
|
|
199
|
+
trigger_on=[TaskType.GENERAL, TaskType.CODE, TaskType.ANALYSIS, TaskType.RESEARCH],
|
|
200
|
+
),
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
# ── File parser ───────────────────────────────────────────────────────────────
|
|
205
|
+
|
|
206
|
+
def _parse_subagent_file(path: Path, source: str) -> SubagentConfig | None:
|
|
207
|
+
try:
|
|
208
|
+
text = path.read_text()
|
|
209
|
+
except Exception:
|
|
210
|
+
return None
|
|
211
|
+
|
|
212
|
+
# JSON format
|
|
213
|
+
if path.suffix == ".json":
|
|
214
|
+
try:
|
|
215
|
+
data = json.loads(text)
|
|
216
|
+
trigger_raw = data.get("trigger_on", [])
|
|
217
|
+
trigger_on = _parse_trigger_on(trigger_raw)
|
|
218
|
+
return SubagentConfig(
|
|
219
|
+
name=data.get("name", path.stem),
|
|
220
|
+
description=data.get("description", ""),
|
|
221
|
+
system_prompt=data.get("system_prompt", ""),
|
|
222
|
+
max_iterations=int(data.get("max_iterations", 1)),
|
|
223
|
+
requires_tool=data.get("requires_tool", ""),
|
|
224
|
+
trigger_on=trigger_on,
|
|
225
|
+
source=source,
|
|
226
|
+
)
|
|
227
|
+
except Exception:
|
|
228
|
+
return None
|
|
229
|
+
|
|
230
|
+
# Markdown with frontmatter
|
|
231
|
+
parts = re.split(r"^---\s*$", text, maxsplit=2, flags=re.MULTILINE)
|
|
232
|
+
if len(parts) < 3:
|
|
233
|
+
return SubagentConfig(
|
|
234
|
+
name=path.stem, description="", system_prompt=text.strip(), source=source
|
|
235
|
+
)
|
|
236
|
+
|
|
237
|
+
_, front_raw, body = parts
|
|
238
|
+
meta: dict[str, str] = {}
|
|
239
|
+
for line in front_raw.splitlines():
|
|
240
|
+
if ":" in line:
|
|
241
|
+
k, _, v = line.partition(":")
|
|
242
|
+
meta[k.strip()] = v.strip()
|
|
243
|
+
|
|
244
|
+
trigger_on = _parse_trigger_on(
|
|
245
|
+
[t.strip() for t in meta.get("trigger_on", "").split(",") if t.strip()]
|
|
246
|
+
)
|
|
247
|
+
return SubagentConfig(
|
|
248
|
+
name=meta.get("name", path.stem),
|
|
249
|
+
description=meta.get("description", ""),
|
|
250
|
+
system_prompt=body.strip(),
|
|
251
|
+
max_iterations=int(meta.get("max_iterations", 1)),
|
|
252
|
+
requires_tool=meta.get("requires_tool", ""),
|
|
253
|
+
trigger_on=trigger_on,
|
|
254
|
+
source=source,
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
|
|
258
|
+
def _parse_trigger_on(raw: list[str]) -> list[TaskType]:
|
|
259
|
+
mapping = {t.value: t for t in TaskType}
|
|
260
|
+
result = []
|
|
261
|
+
for item in raw:
|
|
262
|
+
tt = mapping.get(item.lower())
|
|
263
|
+
if tt:
|
|
264
|
+
result.append(tt)
|
|
265
|
+
return result
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
# ── Loaders ───────────────────────────────────────────────────────────────────
|
|
269
|
+
|
|
270
|
+
def _load_from_dir(directory: Path | None, source: str) -> dict[str, SubagentConfig]:
|
|
271
|
+
if directory is None or not directory.exists():
|
|
272
|
+
return {}
|
|
273
|
+
result = {}
|
|
274
|
+
for f in sorted(directory.glob("*.md")) + sorted(directory.glob("*.json")):
|
|
275
|
+
cfg = _parse_subagent_file(f, source)
|
|
276
|
+
if cfg:
|
|
277
|
+
result[cfg.name] = cfg
|
|
278
|
+
return result
|
|
279
|
+
|
|
280
|
+
|
|
281
|
+
def get_subagents(project_subagents_dir: Path | None = None) -> dict[str, SubagentConfig]:
|
|
282
|
+
"""Merge: builtins < global < project (project wins on name conflicts)."""
|
|
283
|
+
merged: dict[str, SubagentConfig] = {}
|
|
284
|
+
merged.update(BUILTIN_SUBAGENTS)
|
|
285
|
+
merged.update(_load_from_dir(CONFIG_DIR / "subagents", "global"))
|
|
286
|
+
merged.update(_load_from_dir(project_subagents_dir, "project"))
|
|
287
|
+
return merged
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
# ── Test runner detection ─────────────────────────────────────────────────────
|
|
291
|
+
|
|
292
|
+
def _detect_test_runner(cwd: str = ".") -> tuple[str, str] | None:
|
|
293
|
+
"""
|
|
294
|
+
Inspect workspace for test runner indicators.
|
|
295
|
+
Returns (runner_name, command) or None.
|
|
296
|
+
"""
|
|
297
|
+
base = Path(cwd)
|
|
298
|
+
|
|
299
|
+
# pytest
|
|
300
|
+
for indicator in ("pytest.ini", "conftest.py", "setup.cfg"):
|
|
301
|
+
if (base / indicator).exists():
|
|
302
|
+
return ("pytest", "pytest -x --tb=short")
|
|
303
|
+
pyproject = base / "pyproject.toml"
|
|
304
|
+
if pyproject.exists() and "[tool.pytest" in pyproject.read_text():
|
|
305
|
+
return ("pytest", "pytest -x --tb=short")
|
|
306
|
+
|
|
307
|
+
# jest / vitest
|
|
308
|
+
pkg = base / "package.json"
|
|
309
|
+
if pkg.exists():
|
|
310
|
+
try:
|
|
311
|
+
data = json.loads(pkg.read_text())
|
|
312
|
+
scripts = data.get("scripts", {})
|
|
313
|
+
if "test" in scripts:
|
|
314
|
+
runner = "vitest" if "vitest" in scripts["test"] else "jest"
|
|
315
|
+
return (runner, "npm test -- --run")
|
|
316
|
+
except Exception:
|
|
317
|
+
pass
|
|
318
|
+
|
|
319
|
+
# cargo
|
|
320
|
+
if (base / "Cargo.toml").exists():
|
|
321
|
+
return ("cargo", "cargo test")
|
|
322
|
+
|
|
323
|
+
# go test
|
|
324
|
+
if list(base.glob("*.go")):
|
|
325
|
+
return ("go test", "go test ./...")
|
|
326
|
+
|
|
327
|
+
return None
|
|
328
|
+
|
|
329
|
+
|
|
330
|
+
def _run_tests(command: str) -> tuple[str, int]:
|
|
331
|
+
"""Run the detected test command and return its output and exit status.
|
|
332
|
+
|
|
333
|
+
Every test runner reports failure through its exit status, so that is
|
|
334
|
+
the signal to use. run_shell_command discards it, which is why this
|
|
335
|
+
keeps its own runner rather than calling the tool.
|
|
336
|
+
|
|
337
|
+
No shell, like everywhere else: the command is split into an argument
|
|
338
|
+
vector. That is also why the detected commands carry no redirection —
|
|
339
|
+
a "2>&1" would reach the program as a literal argument, and pytest
|
|
340
|
+
would report it as a missing test path on every run.
|
|
341
|
+
"""
|
|
342
|
+
import shlex
|
|
343
|
+
import subprocess
|
|
344
|
+
|
|
345
|
+
try:
|
|
346
|
+
argv = shlex.split(command)
|
|
347
|
+
except ValueError as e:
|
|
348
|
+
return (f"Error: invalid test command: {e}", 1)
|
|
349
|
+
try:
|
|
350
|
+
result = subprocess.run(
|
|
351
|
+
argv, shell=False, capture_output=True, text=True, timeout=600,
|
|
352
|
+
)
|
|
353
|
+
except FileNotFoundError:
|
|
354
|
+
return (f"Error: test runner not found: {argv[0]}", 127)
|
|
355
|
+
except subprocess.TimeoutExpired:
|
|
356
|
+
return ("Error: the test run timed out after 600s.", 1)
|
|
357
|
+
except Exception as e: # pragma: no cover - defensive
|
|
358
|
+
return (f"Error: {e}", 1)
|
|
359
|
+
|
|
360
|
+
output = result.stdout
|
|
361
|
+
if result.stderr:
|
|
362
|
+
output += ("\n" if output else "") + result.stderr
|
|
363
|
+
return (output, result.returncode)
|
|
364
|
+
|
|
365
|
+
|
|
366
|
+
def _has_test_failures(output: str) -> bool:
|
|
367
|
+
"""Fallback for callers holding only the text. Prefer the exit status:
|
|
368
|
+
every test runner reports failure through it, and these patterns match
|
|
369
|
+
any output mentioning a failure, including a test named for one."""
|
|
370
|
+
failure_patterns = [
|
|
371
|
+
r"FAILED",
|
|
372
|
+
r"failed",
|
|
373
|
+
r"ERROR",
|
|
374
|
+
r"\berror\[E", # Rust
|
|
375
|
+
r"FAIL\b",
|
|
376
|
+
r"Test failed",
|
|
377
|
+
r"AssertionError",
|
|
378
|
+
r"not ok", # TAP
|
|
379
|
+
]
|
|
380
|
+
for pat in failure_patterns:
|
|
381
|
+
if re.search(pat, output):
|
|
382
|
+
return True
|
|
383
|
+
return False
|
|
384
|
+
|
|
385
|
+
|
|
386
|
+
# ── Runner ────────────────────────────────────────────────────────────────────
|
|
387
|
+
|
|
388
|
+
async def run_subagent(
|
|
389
|
+
config: SubagentConfig,
|
|
390
|
+
pipeline_output: str,
|
|
391
|
+
primary_adapter: Any,
|
|
392
|
+
primary_model: Any,
|
|
393
|
+
session_id: str,
|
|
394
|
+
on_token: Callable[[str], None] | None = None,
|
|
395
|
+
) -> SubagentResult:
|
|
396
|
+
"""
|
|
397
|
+
Execute a subagent against the pipeline output.
|
|
398
|
+
Single-pass for max_iterations=1; iterative test-fix loop for verify.
|
|
399
|
+
"""
|
|
400
|
+
from polymath.adapters.base import Message
|
|
401
|
+
from polymath.tools import run_shell_command
|
|
402
|
+
|
|
403
|
+
_emit = on_token or (lambda t: None)
|
|
404
|
+
|
|
405
|
+
# ── Verify: iterative test-fix loop ───────────────────────────────────────
|
|
406
|
+
if config.name == "verify" or config.max_iterations > 1:
|
|
407
|
+
runner = _detect_test_runner()
|
|
408
|
+
if runner is None:
|
|
409
|
+
return SubagentResult(
|
|
410
|
+
name=config.name, output="No test runner detected — skipping verify.",
|
|
411
|
+
iterations=0, success=True,
|
|
412
|
+
)
|
|
413
|
+
runner_name, run_cmd = runner
|
|
414
|
+
current_code = pipeline_output
|
|
415
|
+
for iteration in range(1, config.max_iterations + 1):
|
|
416
|
+
test_output, exit_code = _run_tests(run_cmd)
|
|
417
|
+
if exit_code == 0:
|
|
418
|
+
return SubagentResult(
|
|
419
|
+
name=config.name,
|
|
420
|
+
output=f"All tests passed ({runner_name}). {iteration} iteration(s).",
|
|
421
|
+
iterations=iteration,
|
|
422
|
+
success=True,
|
|
423
|
+
)
|
|
424
|
+
# Ask model to fix
|
|
425
|
+
fix_prompt = (
|
|
426
|
+
f"Tests failed ({runner_name}):\n\n{test_output}\n\n"
|
|
427
|
+
f"Here is the current code:\n\n{current_code}\n\n"
|
|
428
|
+
f"Fix all test failures. Output only the corrected code."
|
|
429
|
+
)
|
|
430
|
+
resp = await primary_adapter.complete(
|
|
431
|
+
messages=[Message(role="user", content=fix_prompt)],
|
|
432
|
+
model_id=primary_model.id,
|
|
433
|
+
system=config.system_prompt,
|
|
434
|
+
temperature=0.3,
|
|
435
|
+
)
|
|
436
|
+
current_code = resp if isinstance(resp, str) else resp.content
|
|
437
|
+
|
|
438
|
+
# After max iterations — tests still failing
|
|
439
|
+
_, final_exit = _run_tests(run_cmd)
|
|
440
|
+
return SubagentResult(
|
|
441
|
+
name=config.name,
|
|
442
|
+
output=current_code,
|
|
443
|
+
iterations=config.max_iterations,
|
|
444
|
+
success=final_exit == 0,
|
|
445
|
+
)
|
|
446
|
+
|
|
447
|
+
# ── Single-pass subagent ──────────────────────────────────────────────────
|
|
448
|
+
prompt = f"{pipeline_output}"
|
|
449
|
+
output = ""
|
|
450
|
+
async for token in primary_adapter.stream(
|
|
451
|
+
messages=[Message(role="user", content=prompt)],
|
|
452
|
+
model_id=primary_model.id,
|
|
453
|
+
system=config.system_prompt,
|
|
454
|
+
):
|
|
455
|
+
if isinstance(token, str):
|
|
456
|
+
output += token
|
|
457
|
+
_emit(token)
|
|
458
|
+
|
|
459
|
+
return SubagentResult(name=config.name, output=output, iterations=1, success=True)
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
# ── Auto-selection ────────────────────────────────────────────────────────────
|
|
463
|
+
|
|
464
|
+
def select_auto_subagents(
|
|
465
|
+
task_type: TaskType,
|
|
466
|
+
flags: dict[str, bool],
|
|
467
|
+
available: dict[str, SubagentConfig],
|
|
468
|
+
) -> list[SubagentConfig]:
|
|
469
|
+
"""
|
|
470
|
+
Decide which subagents to run.
|
|
471
|
+
flags: {"verify": bool, "simplify": bool, ...}
|
|
472
|
+
Explicit flags take priority; trigger_on handles auto-selection.
|
|
473
|
+
"""
|
|
474
|
+
selected: list[SubagentConfig] = []
|
|
475
|
+
seen: set[str] = set()
|
|
476
|
+
|
|
477
|
+
# Explicit flags first (in a stable order)
|
|
478
|
+
explicit_order = ["verify", "simplify", "security-scan", "test-writer",
|
|
479
|
+
"docs-writer", "brainstorm-critic", "perf-reviewer",
|
|
480
|
+
"diff-explainer", "brain-gap"]
|
|
481
|
+
for name in explicit_order:
|
|
482
|
+
if flags.get(name) and name in available:
|
|
483
|
+
selected.append(available[name])
|
|
484
|
+
seen.add(name)
|
|
485
|
+
|
|
486
|
+
return selected
|