polymath-agent 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. polymath/__init__.py +2 -0
  2. polymath/adapters/__init__.py +7 -0
  3. polymath/adapters/base.py +175 -0
  4. polymath/adapters/claude.py +280 -0
  5. polymath/adapters/gemini.py +186 -0
  6. polymath/adapters/ollama.py +117 -0
  7. polymath/adapters/openai_adapter.py +168 -0
  8. polymath/bootstrap.py +159 -0
  9. polymath/command_registry.py +41 -0
  10. polymath/command_service.py +572 -0
  11. polymath/compressor.py +90 -0
  12. polymath/config.py +293 -0
  13. polymath/context_manager.py +76 -0
  14. polymath/context_store.py +336 -0
  15. polymath/detector.py +442 -0
  16. polymath/domain.py +78 -0
  17. polymath/execution_service.py +325 -0
  18. polymath/main.py +1293 -0
  19. polymath/memory/__init__.py +15 -0
  20. polymath/memory/chunker.py +6 -0
  21. polymath/memory/embedder.py +179 -0
  22. polymath/memory/migrate.py +2 -0
  23. polymath/memory/retriever.py +2 -0
  24. polymath/memory/store.py +9 -0
  25. polymath/memory/sync.py +2 -0
  26. polymath/memory/writer.py +9 -0
  27. polymath/model_policy.py +172 -0
  28. polymath/orchestrator/__init__.py +68 -0
  29. polymath/orchestrator/attempt_ledger.py +34 -0
  30. polymath/orchestrator/ensemble.py +229 -0
  31. polymath/orchestrator/fanout.py +322 -0
  32. polymath/orchestrator/output_policy.py +61 -0
  33. polymath/orchestrator/race.py +311 -0
  34. polymath/orchestrator/run_controller.py +91 -0
  35. polymath/orchestrator/speculative_review.py +120 -0
  36. polymath/orchestrator/state_responder.py +184 -0
  37. polymath/orchestrator/worker_pool.py +37 -0
  38. polymath/permissions.py +82 -0
  39. polymath/pipeline.py +700 -0
  40. polymath/project_config.py +229 -0
  41. polymath/project_runtime.py +109 -0
  42. polymath/router.py +127 -0
  43. polymath/setup_wizard.py +106 -0
  44. polymath/slash_commands.py +566 -0
  45. polymath/subagents.py +486 -0
  46. polymath/tools.py +333 -0
  47. polymath/ui_state.py +84 -0
  48. polymath/workspace.py +66 -0
  49. polymath_agent-0.4.0.dist-info/METADATA +693 -0
  50. polymath_agent-0.4.0.dist-info/RECORD +54 -0
  51. polymath_agent-0.4.0.dist-info/WHEEL +5 -0
  52. polymath_agent-0.4.0.dist-info/entry_points.txt +2 -0
  53. polymath_agent-0.4.0.dist-info/licenses/LICENSE +21 -0
  54. polymath_agent-0.4.0.dist-info/top_level.txt +1 -0
polymath/subagents.py ADDED
@@ -0,0 +1,486 @@
1
+ """
2
+ Named post-processing subagents for polymath.
3
+
4
+ A subagent is a lightweight mini-pipeline that runs *after* the main pipeline
5
+ output is ready. Each subagent has a focused system prompt and optional
6
+ iterative loop (e.g. verify runs tests, fixes, re-runs).
7
+
8
+ Built-in subagents
9
+ ──────────────────
10
+ code-simplifier Simplify / refactor after main work
11
+ verify Detect test runner, run tests, iterate on failures (max 3×)
12
+ security-scan Flag security issues (injection, hardcoded creds, etc.)
13
+ test-writer Generate test cases for the produced code
14
+ docs-writer Generate or update inline documentation
15
+ brainstorm-critic Challenge assumptions, surface edge cases
16
+ perf-reviewer Spot performance hotspots in code output
17
+ diff-explainer Plain-language explanation of what changed and why
18
+
19
+ Custom subagents
20
+ ────────────────
21
+ Place .md files in .polymath/subagents/ (project) or ~/.polymath/subagents/ (global).
22
+ Frontmatter fields: name, description, max_iterations, requires_tool, trigger_on (comma list)
23
+ Body: the system prompt.
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import json
28
+ import re
29
+ from dataclasses import dataclass, field
30
+ from pathlib import Path
31
+ from typing import Any, Callable
32
+
33
+ from polymath.config import CONFIG_DIR, TaskType
34
+
35
+
36
+ @dataclass
37
+ class SubagentConfig:
38
+ name: str
39
+ description: str
40
+ system_prompt: str
41
+ max_iterations: int = 1
42
+ requires_tool: str = "" # skip subagent if this tool is unavailable
43
+ trigger_on: list[TaskType] = field(default_factory=list)
44
+ source: str = "builtin"
45
+
46
+
47
+ @dataclass
48
+ class SubagentResult:
49
+ name: str
50
+ output: str
51
+ iterations: int
52
+ success: bool
53
+
54
+
55
+ # ── Built-in subagent definitions ─────────────────────────────────────────────
56
+
57
+ BUILTIN_SUBAGENTS: dict[str, SubagentConfig] = {
58
+
59
+ "code-simplifier": SubagentConfig(
60
+ name="code-simplifier",
61
+ description="Simplify and clean code output after the main pipeline.",
62
+ system_prompt=(
63
+ "You are a code simplification expert. "
64
+ "Review the provided code output and:\n"
65
+ "1. Remove unnecessary complexity and dead code\n"
66
+ "2. Simplify logic without changing behaviour\n"
67
+ "3. Improve naming clarity\n"
68
+ "4. Output only the simplified code — no commentary."
69
+ ),
70
+ trigger_on=[TaskType.CODE],
71
+ ),
72
+
73
+ "verify": SubagentConfig(
74
+ name="verify",
75
+ description="Run the project's test suite and fix failures (up to 3 iterations).",
76
+ system_prompt=(
77
+ "You are a debugging agent. Tests have failed. "
78
+ "Analyse the test output, identify root causes, and provide a corrected "
79
+ "implementation that fixes all failures. "
80
+ "Output only the corrected code — no explanation."
81
+ ),
82
+ max_iterations=3,
83
+ requires_tool="run_shell_command",
84
+ trigger_on=[TaskType.CODE],
85
+ ),
86
+
87
+ "security-scan": SubagentConfig(
88
+ name="security-scan",
89
+ description="Scan output for security issues: injection, hardcoded secrets, OWASP top-10.",
90
+ system_prompt=(
91
+ "You are a security reviewer. Scan the following code for:\n"
92
+ "- Injection vulnerabilities (SQL, command, XSS)\n"
93
+ "- Hardcoded secrets or credentials\n"
94
+ "- Insecure deserialization\n"
95
+ "- Missing input validation\n"
96
+ "- Broken authentication patterns\n\n"
97
+ "Format: VERDICT: CLEAN | ISSUES_FOUND\n"
98
+ "ISSUES:\n- <issue> at <location>\n"
99
+ "SEVERITY: LOW | MEDIUM | HIGH for each issue."
100
+ ),
101
+ trigger_on=[TaskType.CODE],
102
+ ),
103
+
104
+ "test-writer": SubagentConfig(
105
+ name="test-writer",
106
+ description="Generate comprehensive test cases for the produced code.",
107
+ system_prompt=(
108
+ "You are a test engineering expert. "
109
+ "Given the following code, write comprehensive tests covering:\n"
110
+ "1. Happy path\n"
111
+ "2. Edge cases and boundary values\n"
112
+ "3. Error conditions\n\n"
113
+ "Match the test framework already used in the project if detectable. "
114
+ "Output only the test code."
115
+ ),
116
+ trigger_on=[TaskType.CODE],
117
+ ),
118
+
119
+ "docs-writer": SubagentConfig(
120
+ name="docs-writer",
121
+ description="Generate or update inline documentation for the output.",
122
+ system_prompt=(
123
+ "You are a technical writer. "
124
+ "Add clear, concise docstrings/comments to the following code. "
125
+ "Document: purpose, parameters, return values, side effects, and "
126
+ "any non-obvious logic. Do not change the code itself. "
127
+ "Output the fully-documented version."
128
+ ),
129
+ trigger_on=[TaskType.CODE],
130
+ ),
131
+
132
+ "brainstorm-critic": SubagentConfig(
133
+ name="brainstorm-critic",
134
+ description="Challenge assumptions and surface edge cases in any output.",
135
+ system_prompt=(
136
+ "You are a rigorous critic. Review the following output and:\n"
137
+ "1. Identify unstated assumptions that may not hold\n"
138
+ "2. Surface edge cases or failure modes not addressed\n"
139
+ "3. Point out logical inconsistencies\n"
140
+ "4. Suggest alternative approaches worth considering\n\n"
141
+ "Be constructive but direct. Format as a bulleted list."
142
+ ),
143
+ trigger_on=[TaskType.GENERAL, TaskType.ANALYSIS, TaskType.RESEARCH],
144
+ ),
145
+
146
+ "perf-reviewer": SubagentConfig(
147
+ name="perf-reviewer",
148
+ description="Identify performance hotspots and inefficiencies in code output.",
149
+ system_prompt=(
150
+ "You are a performance engineering expert. "
151
+ "Review the following code for:\n"
152
+ "1. O(n²) or worse loops that could be optimised\n"
153
+ "2. Unnecessary memory allocations\n"
154
+ "3. N+1 query patterns\n"
155
+ "4. Blocking I/O in async contexts\n"
156
+ "5. Missing caching opportunities\n\n"
157
+ "Format: list each issue with location and suggested fix."
158
+ ),
159
+ trigger_on=[TaskType.CODE],
160
+ ),
161
+
162
+ "diff-explainer": SubagentConfig(
163
+ name="diff-explainer",
164
+ description="Explain what changed and why in plain language (great for PRs).",
165
+ system_prompt=(
166
+ "You are a senior developer writing a PR description. "
167
+ "Given the following code or diff, explain in plain language:\n"
168
+ "1. What changed\n"
169
+ "2. Why it changed (the motivation)\n"
170
+ "3. Any risks or things reviewers should pay attention to\n\n"
171
+ "Keep it concise — 3–5 bullet points."
172
+ ),
173
+ trigger_on=[TaskType.CODE],
174
+ ),
175
+
176
+ "brain-gap": SubagentConfig(
177
+ name="brain-gap",
178
+ description=(
179
+ "After a session, propose what should be added to the project "
180
+ "brain (rules.md / decisions.md) so the same mistake doesn't "
181
+ "repeat. Boris-style failure-driven brain growth."
182
+ ),
183
+ system_prompt=(
184
+ "You are inspecting a finished session to find what the project "
185
+ "brain was MISSING that would have prevented mistakes or saved "
186
+ "the model time. Look for:\n"
187
+ "1. Corrections the user made to the model (e.g. 'no, do X instead')\n"
188
+ "2. Repeated clarification rounds that suggest missing context\n"
189
+ "3. Wrong assumptions the model made that a rule would fix\n"
190
+ "4. Decisions the user made that future sessions should respect\n\n"
191
+ "Output format — propose 0-5 entries, each tagged with the bucket:\n"
192
+ " [rules] one-sentence imperative rule (e.g. 'never commit to main')\n"
193
+ " [decisions] one-line decision + brief why\n"
194
+ " [glossary] term — definition\n\n"
195
+ "If nothing should be added, output the single word: NONE.\n"
196
+ "Be conservative — only propose entries that are durably useful "
197
+ "across sessions, not session-specific facts."
198
+ ),
199
+ trigger_on=[TaskType.GENERAL, TaskType.CODE, TaskType.ANALYSIS, TaskType.RESEARCH],
200
+ ),
201
+ }
202
+
203
+
204
+ # ── File parser ───────────────────────────────────────────────────────────────
205
+
206
+ def _parse_subagent_file(path: Path, source: str) -> SubagentConfig | None:
207
+ try:
208
+ text = path.read_text()
209
+ except Exception:
210
+ return None
211
+
212
+ # JSON format
213
+ if path.suffix == ".json":
214
+ try:
215
+ data = json.loads(text)
216
+ trigger_raw = data.get("trigger_on", [])
217
+ trigger_on = _parse_trigger_on(trigger_raw)
218
+ return SubagentConfig(
219
+ name=data.get("name", path.stem),
220
+ description=data.get("description", ""),
221
+ system_prompt=data.get("system_prompt", ""),
222
+ max_iterations=int(data.get("max_iterations", 1)),
223
+ requires_tool=data.get("requires_tool", ""),
224
+ trigger_on=trigger_on,
225
+ source=source,
226
+ )
227
+ except Exception:
228
+ return None
229
+
230
+ # Markdown with frontmatter
231
+ parts = re.split(r"^---\s*$", text, maxsplit=2, flags=re.MULTILINE)
232
+ if len(parts) < 3:
233
+ return SubagentConfig(
234
+ name=path.stem, description="", system_prompt=text.strip(), source=source
235
+ )
236
+
237
+ _, front_raw, body = parts
238
+ meta: dict[str, str] = {}
239
+ for line in front_raw.splitlines():
240
+ if ":" in line:
241
+ k, _, v = line.partition(":")
242
+ meta[k.strip()] = v.strip()
243
+
244
+ trigger_on = _parse_trigger_on(
245
+ [t.strip() for t in meta.get("trigger_on", "").split(",") if t.strip()]
246
+ )
247
+ return SubagentConfig(
248
+ name=meta.get("name", path.stem),
249
+ description=meta.get("description", ""),
250
+ system_prompt=body.strip(),
251
+ max_iterations=int(meta.get("max_iterations", 1)),
252
+ requires_tool=meta.get("requires_tool", ""),
253
+ trigger_on=trigger_on,
254
+ source=source,
255
+ )
256
+
257
+
258
+ def _parse_trigger_on(raw: list[str]) -> list[TaskType]:
259
+ mapping = {t.value: t for t in TaskType}
260
+ result = []
261
+ for item in raw:
262
+ tt = mapping.get(item.lower())
263
+ if tt:
264
+ result.append(tt)
265
+ return result
266
+
267
+
268
+ # ── Loaders ───────────────────────────────────────────────────────────────────
269
+
270
+ def _load_from_dir(directory: Path | None, source: str) -> dict[str, SubagentConfig]:
271
+ if directory is None or not directory.exists():
272
+ return {}
273
+ result = {}
274
+ for f in sorted(directory.glob("*.md")) + sorted(directory.glob("*.json")):
275
+ cfg = _parse_subagent_file(f, source)
276
+ if cfg:
277
+ result[cfg.name] = cfg
278
+ return result
279
+
280
+
281
+ def get_subagents(project_subagents_dir: Path | None = None) -> dict[str, SubagentConfig]:
282
+ """Merge: builtins < global < project (project wins on name conflicts)."""
283
+ merged: dict[str, SubagentConfig] = {}
284
+ merged.update(BUILTIN_SUBAGENTS)
285
+ merged.update(_load_from_dir(CONFIG_DIR / "subagents", "global"))
286
+ merged.update(_load_from_dir(project_subagents_dir, "project"))
287
+ return merged
288
+
289
+
290
+ # ── Test runner detection ─────────────────────────────────────────────────────
291
+
292
+ def _detect_test_runner(cwd: str = ".") -> tuple[str, str] | None:
293
+ """
294
+ Inspect workspace for test runner indicators.
295
+ Returns (runner_name, command) or None.
296
+ """
297
+ base = Path(cwd)
298
+
299
+ # pytest
300
+ for indicator in ("pytest.ini", "conftest.py", "setup.cfg"):
301
+ if (base / indicator).exists():
302
+ return ("pytest", "pytest -x --tb=short")
303
+ pyproject = base / "pyproject.toml"
304
+ if pyproject.exists() and "[tool.pytest" in pyproject.read_text():
305
+ return ("pytest", "pytest -x --tb=short")
306
+
307
+ # jest / vitest
308
+ pkg = base / "package.json"
309
+ if pkg.exists():
310
+ try:
311
+ data = json.loads(pkg.read_text())
312
+ scripts = data.get("scripts", {})
313
+ if "test" in scripts:
314
+ runner = "vitest" if "vitest" in scripts["test"] else "jest"
315
+ return (runner, "npm test -- --run")
316
+ except Exception:
317
+ pass
318
+
319
+ # cargo
320
+ if (base / "Cargo.toml").exists():
321
+ return ("cargo", "cargo test")
322
+
323
+ # go test
324
+ if list(base.glob("*.go")):
325
+ return ("go test", "go test ./...")
326
+
327
+ return None
328
+
329
+
330
+ def _run_tests(command: str) -> tuple[str, int]:
331
+ """Run the detected test command and return its output and exit status.
332
+
333
+ Every test runner reports failure through its exit status, so that is
334
+ the signal to use. run_shell_command discards it, which is why this
335
+ keeps its own runner rather than calling the tool.
336
+
337
+ No shell, like everywhere else: the command is split into an argument
338
+ vector. That is also why the detected commands carry no redirection —
339
+ a "2>&1" would reach the program as a literal argument, and pytest
340
+ would report it as a missing test path on every run.
341
+ """
342
+ import shlex
343
+ import subprocess
344
+
345
+ try:
346
+ argv = shlex.split(command)
347
+ except ValueError as e:
348
+ return (f"Error: invalid test command: {e}", 1)
349
+ try:
350
+ result = subprocess.run(
351
+ argv, shell=False, capture_output=True, text=True, timeout=600,
352
+ )
353
+ except FileNotFoundError:
354
+ return (f"Error: test runner not found: {argv[0]}", 127)
355
+ except subprocess.TimeoutExpired:
356
+ return ("Error: the test run timed out after 600s.", 1)
357
+ except Exception as e: # pragma: no cover - defensive
358
+ return (f"Error: {e}", 1)
359
+
360
+ output = result.stdout
361
+ if result.stderr:
362
+ output += ("\n" if output else "") + result.stderr
363
+ return (output, result.returncode)
364
+
365
+
366
+ def _has_test_failures(output: str) -> bool:
367
+ """Fallback for callers holding only the text. Prefer the exit status:
368
+ every test runner reports failure through it, and these patterns match
369
+ any output mentioning a failure, including a test named for one."""
370
+ failure_patterns = [
371
+ r"FAILED",
372
+ r"failed",
373
+ r"ERROR",
374
+ r"\berror\[E", # Rust
375
+ r"FAIL\b",
376
+ r"Test failed",
377
+ r"AssertionError",
378
+ r"not ok", # TAP
379
+ ]
380
+ for pat in failure_patterns:
381
+ if re.search(pat, output):
382
+ return True
383
+ return False
384
+
385
+
386
+ # ── Runner ────────────────────────────────────────────────────────────────────
387
+
388
+ async def run_subagent(
389
+ config: SubagentConfig,
390
+ pipeline_output: str,
391
+ primary_adapter: Any,
392
+ primary_model: Any,
393
+ session_id: str,
394
+ on_token: Callable[[str], None] | None = None,
395
+ ) -> SubagentResult:
396
+ """
397
+ Execute a subagent against the pipeline output.
398
+ Single-pass for max_iterations=1; iterative test-fix loop for verify.
399
+ """
400
+ from polymath.adapters.base import Message
401
+ from polymath.tools import run_shell_command
402
+
403
+ _emit = on_token or (lambda t: None)
404
+
405
+ # ── Verify: iterative test-fix loop ───────────────────────────────────────
406
+ if config.name == "verify" or config.max_iterations > 1:
407
+ runner = _detect_test_runner()
408
+ if runner is None:
409
+ return SubagentResult(
410
+ name=config.name, output="No test runner detected — skipping verify.",
411
+ iterations=0, success=True,
412
+ )
413
+ runner_name, run_cmd = runner
414
+ current_code = pipeline_output
415
+ for iteration in range(1, config.max_iterations + 1):
416
+ test_output, exit_code = _run_tests(run_cmd)
417
+ if exit_code == 0:
418
+ return SubagentResult(
419
+ name=config.name,
420
+ output=f"All tests passed ({runner_name}). {iteration} iteration(s).",
421
+ iterations=iteration,
422
+ success=True,
423
+ )
424
+ # Ask model to fix
425
+ fix_prompt = (
426
+ f"Tests failed ({runner_name}):\n\n{test_output}\n\n"
427
+ f"Here is the current code:\n\n{current_code}\n\n"
428
+ f"Fix all test failures. Output only the corrected code."
429
+ )
430
+ resp = await primary_adapter.complete(
431
+ messages=[Message(role="user", content=fix_prompt)],
432
+ model_id=primary_model.id,
433
+ system=config.system_prompt,
434
+ temperature=0.3,
435
+ )
436
+ current_code = resp if isinstance(resp, str) else resp.content
437
+
438
+ # After max iterations — tests still failing
439
+ _, final_exit = _run_tests(run_cmd)
440
+ return SubagentResult(
441
+ name=config.name,
442
+ output=current_code,
443
+ iterations=config.max_iterations,
444
+ success=final_exit == 0,
445
+ )
446
+
447
+ # ── Single-pass subagent ──────────────────────────────────────────────────
448
+ prompt = f"{pipeline_output}"
449
+ output = ""
450
+ async for token in primary_adapter.stream(
451
+ messages=[Message(role="user", content=prompt)],
452
+ model_id=primary_model.id,
453
+ system=config.system_prompt,
454
+ ):
455
+ if isinstance(token, str):
456
+ output += token
457
+ _emit(token)
458
+
459
+ return SubagentResult(name=config.name, output=output, iterations=1, success=True)
460
+
461
+
462
+ # ── Auto-selection ────────────────────────────────────────────────────────────
463
+
464
+ def select_auto_subagents(
465
+ task_type: TaskType,
466
+ flags: dict[str, bool],
467
+ available: dict[str, SubagentConfig],
468
+ ) -> list[SubagentConfig]:
469
+ """
470
+ Decide which subagents to run.
471
+ flags: {"verify": bool, "simplify": bool, ...}
472
+ Explicit flags take priority; trigger_on handles auto-selection.
473
+ """
474
+ selected: list[SubagentConfig] = []
475
+ seen: set[str] = set()
476
+
477
+ # Explicit flags first (in a stable order)
478
+ explicit_order = ["verify", "simplify", "security-scan", "test-writer",
479
+ "docs-writer", "brainstorm-critic", "perf-reviewer",
480
+ "diff-explainer", "brain-gap"]
481
+ for name in explicit_order:
482
+ if flags.get(name) and name in available:
483
+ selected.append(available[name])
484
+ seen.add(name)
485
+
486
+ return selected