polymath-agent 0.4.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (54) hide show
  1. polymath/__init__.py +2 -0
  2. polymath/adapters/__init__.py +7 -0
  3. polymath/adapters/base.py +175 -0
  4. polymath/adapters/claude.py +280 -0
  5. polymath/adapters/gemini.py +186 -0
  6. polymath/adapters/ollama.py +117 -0
  7. polymath/adapters/openai_adapter.py +168 -0
  8. polymath/bootstrap.py +159 -0
  9. polymath/command_registry.py +41 -0
  10. polymath/command_service.py +572 -0
  11. polymath/compressor.py +90 -0
  12. polymath/config.py +293 -0
  13. polymath/context_manager.py +76 -0
  14. polymath/context_store.py +336 -0
  15. polymath/detector.py +442 -0
  16. polymath/domain.py +78 -0
  17. polymath/execution_service.py +325 -0
  18. polymath/main.py +1293 -0
  19. polymath/memory/__init__.py +15 -0
  20. polymath/memory/chunker.py +6 -0
  21. polymath/memory/embedder.py +179 -0
  22. polymath/memory/migrate.py +2 -0
  23. polymath/memory/retriever.py +2 -0
  24. polymath/memory/store.py +9 -0
  25. polymath/memory/sync.py +2 -0
  26. polymath/memory/writer.py +9 -0
  27. polymath/model_policy.py +172 -0
  28. polymath/orchestrator/__init__.py +68 -0
  29. polymath/orchestrator/attempt_ledger.py +34 -0
  30. polymath/orchestrator/ensemble.py +229 -0
  31. polymath/orchestrator/fanout.py +322 -0
  32. polymath/orchestrator/output_policy.py +61 -0
  33. polymath/orchestrator/race.py +311 -0
  34. polymath/orchestrator/run_controller.py +91 -0
  35. polymath/orchestrator/speculative_review.py +120 -0
  36. polymath/orchestrator/state_responder.py +184 -0
  37. polymath/orchestrator/worker_pool.py +37 -0
  38. polymath/permissions.py +82 -0
  39. polymath/pipeline.py +700 -0
  40. polymath/project_config.py +229 -0
  41. polymath/project_runtime.py +109 -0
  42. polymath/router.py +127 -0
  43. polymath/setup_wizard.py +106 -0
  44. polymath/slash_commands.py +566 -0
  45. polymath/subagents.py +486 -0
  46. polymath/tools.py +333 -0
  47. polymath/ui_state.py +84 -0
  48. polymath/workspace.py +66 -0
  49. polymath_agent-0.4.0.dist-info/METADATA +693 -0
  50. polymath_agent-0.4.0.dist-info/RECORD +54 -0
  51. polymath_agent-0.4.0.dist-info/WHEEL +5 -0
  52. polymath_agent-0.4.0.dist-info/entry_points.txt +2 -0
  53. polymath_agent-0.4.0.dist-info/licenses/LICENSE +21 -0
  54. polymath_agent-0.4.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,693 @@
1
+ Metadata-Version: 2.4
2
+ Name: polymath-agent
3
+ Version: 0.4.0
4
+ Summary: Multi-model AI orchestrator built on the nexus brain. Multi-model routing, ReAct pipeline, plan mode, custom slash commands, subagents.
5
+ Author: Ayushi Gupta
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/ayushigupta-29/polymath
8
+ Project-URL: Repository, https://github.com/ayushigupta-29/polymath
9
+ Project-URL: Brain, https://github.com/ayushigupta-29/nexus
10
+ Keywords: ai,llm,orchestrator,multi-model,claude,gemini,openai,ollama,agent
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Programming Language :: Python :: 3
15
+ Classifier: Programming Language :: Python :: 3.10
16
+ Classifier: Programming Language :: Python :: 3.11
17
+ Classifier: Programming Language :: Python :: 3.12
18
+ Classifier: Programming Language :: Python :: 3.13
19
+ Classifier: Topic :: Software Development :: Libraries :: Python Modules
20
+ Requires-Python: >=3.10
21
+ Description-Content-Type: text/markdown
22
+ License-File: LICENSE
23
+ Requires-Dist: polymath-nexus>=0.2.0
24
+ Requires-Dist: httpx>=0.27.0
25
+ Provides-Extra: orchestrator
26
+ Requires-Dist: anthropic>=0.40.0; extra == "orchestrator"
27
+ Requires-Dist: google-genai>=1.0.0; extra == "orchestrator"
28
+ Requires-Dist: openai>=1.50.0; extra == "orchestrator"
29
+ Requires-Dist: rich>=13.0.0; extra == "orchestrator"
30
+ Requires-Dist: prompt_toolkit>=3.0.0; extra == "orchestrator"
31
+ Requires-Dist: tiktoken>=0.7.0; extra == "orchestrator"
32
+ Requires-Dist: sqlite-vec>=0.1.6; extra == "orchestrator"
33
+ Provides-Extra: embed
34
+ Requires-Dist: google-genai>=1.0.0; extra == "embed"
35
+ Requires-Dist: openai>=1.50.0; extra == "embed"
36
+ Provides-Extra: dev
37
+ Requires-Dist: polymath-agent[orchestrator]; extra == "dev"
38
+ Requires-Dist: pytest>=7.0; extra == "dev"
39
+ Requires-Dist: pytest-asyncio>=0.21; extra == "dev"
40
+ Dynamic: license-file
41
+
42
+ # polymath
43
+
44
+ A multi-model AI orchestrator built on top of [**nexus**](https://github.com/ayushigupta-29/nexus), the long-term project brain.
45
+
46
+ Polymath routes tasks across every model available on your device — Claude, Gemini, GPT, Ollama local models, LM Studio — through a single CLI with unified context, an autonomous agentic loop, cross-model review, custom slash commands, subagents, and shared team project folders. The team's knowledge (rules, decisions, glossary, code patterns) lives in nexus and is shared with every AI tool, not just polymath.
47
+
48
+ ## Two-layer architecture
49
+
50
+ ```
51
+ ┌─────────────────────────────────────────────────────────────┐
52
+ │ polymath (this repo) — multi-model orchestrator │
53
+ │ • Routing across Claude / Gemini / GPT / Ollama │
54
+ │ • Agentic ReAct pipeline + plan mode │
55
+ │ • Slash commands, subagents, cross-model review │
56
+ │ • Ensemble + fanout for concurrent multi-model execution │
57
+ └─────────────────────────────────────────────────────────────┘
58
+ ↓ depends on
59
+ ┌─────────────────────────────────────────────────────────────┐
60
+ │ nexus — long-term team brain │
61
+ │ • 9-typed buckets, per-owner sub-files, date stamps │
62
+ │ • Auto-generated CLAUDE.md / AGENTS.md / .cursor / Copilot│
63
+ │ • Append-only, schema-versioned, forward-compatible │
64
+ │ • Lives at github.com/ayushigupta-29/nexus │
65
+ └─────────────────────────────────────────────────────────────┘
66
+ ```
67
+
68
+ The brain is its own package because **every** AI tool benefits from a shared team brain — not just polymath. You can `pip install polymath-nexus` and use the `nexus` CLI with Cursor / Claude Code / Codex without ever installing polymath.
69
+
70
+ ---
71
+
72
+ ## Install
73
+
74
+ **Brain only (just the team memory, works with any AI tool):**
75
+ ```bash
76
+ pip install polymath-nexus
77
+ cd your-repo
78
+ nexus init
79
+ ```
80
+
81
+ **Full orchestrator (brain + multi-model routing):**
82
+ ```bash
83
+ pip install "polymath-agent[orchestrator]"
84
+ # the command is still `polymath`; only the PyPI name differs
85
+ # or from source:
86
+ git clone https://github.com/ayushigupta-29/polymath
87
+ cd polymath
88
+ bash install.sh
89
+ ```
90
+
91
+ Requires Python 3.10+.
92
+
93
+ On startup polymath reports any missing dependency and tells you what to
94
+ install. It does not install anything itself. Set `POLYMATH_AUTO_INSTALL=1`
95
+ to get the old self-repairing behaviour, which is meant for local
96
+ development in a virtualenv you own.
97
+
98
+ ---
99
+
100
+ ## Getting API Keys (three ways — pick one)
101
+
102
+ ### Option 1 — CLI auto-detection (zero config)
103
+
104
+ If you already use any of these CLIs, polymath picks up their credentials automatically at startup. No setup needed.
105
+
106
+ | CLI | Install | What polymath reads |
107
+ |-----|---------|-------------------|
108
+ | [Claude Code](https://claude.ai/code) | `npm i -g @anthropic-ai/claude-code` | macOS Keychain / `~/.claude/config.json` (Linux) / `%APPDATA%\Claude\` (Windows) |
109
+ | [Gemini CLI](https://github.com/google-gemini/gemini-cli) | `npm i -g @google/gemini-cli` | `~/.gemini/oauth_creds.json` — auto-refreshed via `refresh_token` |
110
+ | [Codex CLI](https://github.com/openai/codex) | `npm i -g @openai/codex` | `~/.codex/auth.json` — ChatGPT OAuth mode |
111
+
112
+ ### Option 2 — Environment variables
113
+
114
+ ```bash
115
+ export ANTHROPIC_API_KEY=sk-ant-...
116
+ export GOOGLE_API_KEY=AIza...
117
+ export OPENAI_API_KEY=sk-...
118
+ ```
119
+
120
+ ### Option 3 — Setup wizard
121
+
122
+ ```bash
123
+ polymath --setup
124
+ ```
125
+
126
+ Walks you through entering API keys. Saved to `~/.polymath/config.json`.
127
+
128
+ ---
129
+
130
+ ## Platform support
131
+
132
+ | OS | Claude | Gemini | OpenAI |
133
+ |----|--------|--------|--------|
134
+ | macOS | Keychain auto-detect ✓ | `~/.gemini/` ✓ | `~/.codex/` ✓ |
135
+ | Linux | `~/.claude/config.json` ✓ | `~/.gemini/` ✓ | `~/.codex/` ✓ |
136
+ | Windows | `%APPDATA%\Claude\` ✓ | `~/.gemini/` ✓ | `~/.codex/` ✓ |
137
+
138
+ All providers also fall back to environment variables on every platform.
139
+
140
+ ---
141
+
142
+ ## Start
143
+
144
+ ```bash
145
+ polymath # interactive REPL
146
+ polymath --project <name> # start with a detached local project
147
+ polymath --models # list all detected models + availability
148
+ polymath --sessions # list past sessions
149
+ polymath --session <id> # resume a session
150
+ polymath --setup # re-run setup wizard
151
+ polymath ask "question" # one-shot, no REPL
152
+ ```
153
+
154
+ ---
155
+
156
+ ## How polymath behaves
157
+
158
+ polymath is designed to feel like one terminal assistant, not a manual model switcher.
159
+
160
+ - simple runtime-state questions are answered from local polymath state
161
+ - simple chat prompts use a direct answer path
162
+ - multi-step work uses the full planning/execution/review pipeline
163
+ - if a model fails because of auth, quota, or context pressure, polymath walks to the next viable model
164
+ - some internal work can run concurrently, but the user still sees one shared conversation
165
+
166
+ This means the user-facing experience stays unified even when multiple providers and model roles are involved behind the scenes.
167
+
168
+ ---
169
+
170
+ ## Task Pipeline
171
+
172
+ Multi-step tasks flow through a structured pipeline. Complexity is classified automatically.
173
+
174
+ ```
175
+ SIMPLE plan → [plan approval] → clarify → execute → review → output
176
+
177
+ MEDIUM clarify → plan → [plan approval] → clarify → execute → review → output
178
+
179
+ COMPLEX understand → clarify → plan → [plan approval] → clarify
180
+ → execute loop → review → corrections → re-review → output
181
+ ```
182
+
183
+ **Plan approval** pauses before execution on every task. The model's plan is shown; type:
184
+ - `y` / Enter — proceed
185
+ - `refine: <feedback>` — revise the plan (up to 3 rounds)
186
+ - `cancel` — abort
187
+
188
+ The **execute loop** is a full ReAct (Reason / Act / Observe) agentic loop — up to 10 tool-use iterations. Models can read files, list directories, and run shell commands until the task is complete.
189
+
190
+ On COMPLEX tasks, up to 5 key decisions are automatically extracted from the output and appended to `decisions.md` with a UTC timestamp.
191
+
192
+ Simple chat and runtime-state questions bypass this full pipeline when that would produce worse UX.
193
+
194
+ ---
195
+
196
+ ## REPL — what you see
197
+
198
+ ```
199
+ polymath > write a fastapi server for user auth
200
+
201
+ ⋯ code · complex ← processing indicator, immediate
202
+ ← prompt stays at bottom while streaming
203
+
204
+ polymath ⋯ [1↑] > explain last output ← next command already queued
205
+ ← ⋯ = processing, [1↑] = 1 queued
206
+
207
+ claude-sonnet-3-5: 4.2k↑ 1.1k↓ · gemini-2.0-flash: 3.1k↑ 0.4k↓
208
+ ← per-model token usage, right-aligned
209
+ steps: plan → clarify → execute → review | primary: Claude Sonnet 3.5
210
+ ```
211
+
212
+ **Command queue**: type and submit a new command any time — it queues and runs after the current one finishes.
213
+ - Up arrow on an empty prompt pulls the most recent queued command back into the draft for editing
214
+ - `/queue` shows queued commands
215
+ - `/queue drop <n>` removes one
216
+ - `/queue move <from> <to>` reorders them
217
+
218
+ **Inline interaction modes**: clarification, plan approval, and permission prompts happen in the same REPL surface with `clarify >`, `plan >`, and `permission >` prompts instead of nested modal sessions.
219
+
220
+ **Persistent state**: the footer shows repo/project provenance, queue preview, and git state while the REPL is running.
221
+
222
+ **Single-agent behavior**: polymath should feel like one assistant even when multiple provider/model workers are involved internally. Runtime-state questions are answered locally, normal chat uses a direct-answer path, and provider failures are reduced to short orchestration messages instead of raw tracebacks.
223
+
224
+ ---
225
+
226
+ ## Routing
227
+
228
+ ### Force a provider
229
+
230
+ ```
231
+ @claude <task> use Claude
232
+ @gemini <task> use Gemini
233
+ @openai <task> use OpenAI/GPT
234
+ @ollama <task> use local Ollama
235
+ ```
236
+
237
+ ### Flags
238
+
239
+ ```
240
+ --cheap <task> cost-first routing
241
+ --fast <task> speed-first routing
242
+ --ask <question> skip pipeline — direct Q&A
243
+ --verify run verify subagent after pipeline
244
+ --simplify run code-simplifier subagent after pipeline
245
+ ```
246
+
247
+ ### Priority profiles
248
+
249
+ Set during `/setup`.
250
+
251
+ | Profile | Order |
252
+ |---------|-------|
253
+ | `quality-first` | best quality → cheaper fallbacks |
254
+ | `cost-first` | free → cheap → premium |
255
+ | `speed-first` | fastest available |
256
+ | `balanced` | quality + speed averaged |
257
+
258
+ ---
259
+
260
+ ## Supported Models
261
+
262
+ | Provider | Models |
263
+ |----------|--------|
264
+ | Anthropic | Claude Sonnet 3.5, Claude Haiku 3.5, Claude Opus 3 |
265
+ | Google | Gemini 2.0 Flash, Gemini 1.5 Pro, Gemini 1.5 Flash |
266
+ | OpenAI | GPT-4o, GPT-4o Mini, o1 Mini |
267
+ | Ollama | Any locally pulled model (auto-detected at runtime) |
268
+ | LM Studio | Any loaded model (auto-detected at runtime) |
269
+
270
+ ---
271
+
272
+ ## Built-in Tools
273
+
274
+ Models use these autonomously during the execute loop:
275
+
276
+ | Tool | What it does | Requires approval |
277
+ |------|-------------|-------------------|
278
+ | `list_directory` | List files and folders | No |
279
+ | `read_file` | Read any file | No |
280
+ | `write_file` | Create or overwrite a file | **Yes** (unless pre-allowed) |
281
+ | `run_shell_command` | Run any shell command | **Yes** (unless pre-allowed) |
282
+
283
+ Pre-allow tools and shell patterns via `.polymath/settings.json` (see [Project Folders](#shared-project-folders)).
284
+
285
+ ---
286
+
287
+ ## Cross-Model Review
288
+
289
+ After every execution, a **different model** automatically reviews the output — verdict, issues, and suggestions — then corrections are applied and re-reviewed. Claude writes → Gemini reviews → Claude corrects → Gemini re-reviews.
290
+
291
+ For some phases, polymath can also run internal workers concurrently:
292
+
293
+ - post-run subagents can execute in parallel
294
+ - backup/fallback model attempts are tracked so the orchestrator does not loop
295
+ - the visible output still stays on one shared terminal thread
296
+
297
+ ---
298
+
299
+ ## Typed Project Context
300
+
301
+ 9 context files per project. Only the relevant ones are injected per task type — no token bloat.
302
+
303
+ | File | Purpose | Injected for |
304
+ |------|---------|-------------|
305
+ | `rules.md` | Golden rules, hard constraints | **Always** |
306
+ | `goals.md` | Objectives, success criteria | **Always** |
307
+ | `code.md` | Patterns, conventions, architecture | `code` tasks |
308
+ | `stack.md` | Tech stack, dependencies, versions | `code` tasks |
309
+ | `logic.md` | Business / domain logic | `analysis`, `research` |
310
+ | `data.md` | Schemas, field definitions, samples | `analysis` tasks |
311
+ | `glossary.md` | Domain-specific terms | `analysis`, `research` |
312
+ | `decisions.md` | Past decisions + rationale (ADR-style) | `planning` |
313
+ | `personas.md` | Tone, behavior, communication style | `creative` tasks |
314
+
315
+ ### `@context` mentions
316
+
317
+ ```
318
+ polymath > @rules what are the constraints on this project?
319
+ polymath > @stack which framework handles auth?
320
+ ```
321
+
322
+ ---
323
+
324
+ ## Chunked Memory
325
+
326
+ Polymath maintains a **chunked, embedded, team-shareable memory** for each project — semantic retrieval over your typed context plus durable facts auto-extracted from sessions.
327
+
328
+ ### How it differs from typed context alone
329
+
330
+ The 9 typed context files are still the **source of truth** (and what humans edit). Memory is the **derived, retrieval-friendly layer** built on top:
331
+
332
+ - Each `.md` file is split into heading-aware chunks
333
+ - Each chunk is embedded (Gemini `text-embedding-004` by default, OpenAI `text-embedding-3-small` fallback — Claude has no embeddings API)
334
+ - At injection time, only the **top-K chunks relevant to your prompt** are sent to the model — not the whole file
335
+ - After every pipeline run, a **writeback** step extracts up to 8 durable facts from the conversation as new candidate chunks
336
+ - `confirmed` chunks rank higher than `candidate` chunks; teammates promote candidates via `/memory confirm <id>`
337
+
338
+ ### Storage layout (committed to git)
339
+
340
+ ```
341
+ .polymath/memory/
342
+ ├── chunks/
343
+ │ ├── rules.jsonl ← one chunk per line, per type
344
+ │ ├── code.jsonl
345
+ │ ├── decisions.jsonl
346
+ │ ├── extracted.jsonl ← writeback output lands here
347
+ │ └── ... (one file per typed context concern)
348
+ ├── manifest.json ← embed_provider/model/dim, last writeback
349
+ └── .gitignore ← ignores cache.db
350
+ ```
351
+
352
+ Each chunk: `id`, `type`, `source_file`, `heading_path`, `content`, base64-encoded float32 `embedding`, `embed_provider`/`model`/`dim`, `created_by_model`/`user`, `created_at`, `last_validated_at`, `scope` (session/project/global), `status` (candidate/confirmed).
353
+
354
+ A local SQLite + sqlite-vec cache lives at `~/.polymath/cache/<project_id>/chunks.db` and is **derived** from the JSONL files. New devs auto-build it on first run; embeddings are committed so re-embed cost is zero.
355
+
356
+ ### `/memory` commands
357
+
358
+ | Command | What it does |
359
+ |---------|-------------|
360
+ | `/memory show` | Counts by type + status, embedder info, last writeback |
361
+ | `/memory rebuild` | Re-chunk + re-embed all context files for the active project |
362
+ | `/memory search <query>` | Debug what the retriever returns for a query |
363
+ | `/memory confirm <id>` | Promote a `candidate` chunk → `confirmed` (boosts ranking) |
364
+ | `/memory diff` | Preview what writeback would extract from the current session |
365
+ | `/memory test` | Verify your embedder config end-to-end |
366
+
367
+ ### Setup
368
+
369
+ Set one embedding key (Claude has no embeddings API, so this is required):
370
+
371
+ ```bash
372
+ # Option A — Gemini (free tier, recommended)
373
+ export GOOGLE_API_KEY=AIza... # https://aistudio.google.com/apikey
374
+
375
+ # Option B — OpenAI ($0.02 / 1M tokens, paid)
376
+ export OPENAI_API_KEY=sk-...
377
+ ```
378
+
379
+ Codex CLI tokens are chat-only and **cannot embed** — set a real `OPENAI_API_KEY` if going that route.
380
+
381
+ On first launch in a project with existing context, polymath asks once:
382
+ ```
383
+ Found 9 context files (~340 chunks) not yet indexed for memory.
384
+ Migrate to chunked memory now? Free Gemini embedding tier. (y / n)
385
+ ```
386
+
387
+ ### Curation workflow (team)
388
+
389
+ 1. Daily use — writeback auto-creates `candidate` chunks in `.polymath/memory/chunks/extracted.jsonl`
390
+ 2. Reviewer (you, on PR) sees the diff — line-by-line in git
391
+ 3. `/memory confirm <id>` promotes good candidates to `confirmed`; bad ones get pruned automatically after 30 days
392
+
393
+ ### Settings (`~/.polymath/config.json` or `.polymath/settings.json`)
394
+
395
+ ```json
396
+ {
397
+ "memory": {
398
+ "embedding_provider": "gemini",
399
+ "embedding_model": "text-embedding-004",
400
+ "retrieval_k": 12,
401
+ "dedup_threshold": 0.92,
402
+ "writeback_min_chars": 200,
403
+ "stale_candidate_days": 30
404
+ }
405
+ }
406
+ ```
407
+
408
+ ---
409
+
410
+ ## Shared Project Folders
411
+
412
+ `.polymath/` is a directory you commit to git — shared across your whole team.
413
+
414
+ ```
415
+ .polymath/
416
+ ├── context/ ← 9 typed context .md files
417
+ ├── commands/ ← custom slash command .md files
418
+ ├── subagents/ ← custom subagent configs (.md or .json)
419
+ ├── handoffs/ ← session briefings from /handoff
420
+ └── settings.json ← allowed tools, allowed shell patterns, model preferences
421
+ ```
422
+
423
+ polymath walks up from `cwd` to find `.polymath/`, like git.
424
+
425
+ ### Repo-first project model
426
+
427
+ polymath is repo-first:
428
+
429
+ - If the current git repo contains `.polymath/`, that repo is treated as the primary shared project
430
+ - Shared context, commands, subagents, and settings live in that repo and are meant to be committed
431
+ - Sessions, history, caches, and exports remain local under `~/.polymath/`
432
+ - Detached local projects under `~/.polymath/projects/` still work, but they are secondary to repo-backed projects
433
+
434
+ Preferred workflow:
435
+
436
+ ```bash
437
+ cd your-app-repo
438
+ polymath /project init
439
+ git add .polymath
440
+ git commit -m "Add polymath project metadata"
441
+ ```
442
+
443
+ ### Working as a team
444
+
445
+ For concurrent human collaboration, keep the repo as the shared source of truth and use git normally:
446
+
447
+ - commit `.polymath/` so context, commands, subagents, and handoffs are shared
448
+ - keep session transcripts and caches local under `~/.polymath/`
449
+ - use separate branches or worktrees for parallel implementation work
450
+ - use `/handoff` to drop a brief into `.polymath/handoffs/` when another person needs to continue the thread
451
+ - avoid editing the same code paths from multiple terminals unless you intend to resolve the merge at git level
452
+
453
+ Recommended team model:
454
+
455
+ - `.polymath/` is committed and shared
456
+ - `~/.polymath/` remains private and local
457
+ - coordination happens through git, not a shared backend
458
+ - handoffs are explicit, reviewable files in the repo
459
+
460
+ ### `settings.json`
461
+
462
+ ```json
463
+ {
464
+ "allowed_tools": ["list_directory", "read_file", "write_file"],
465
+ "allowed_shell_commands": ["git *", "pytest *", "npm *"],
466
+ "model_preferences": {
467
+ "primary": "claude",
468
+ "reviewer": "gemini"
469
+ }
470
+ }
471
+ ```
472
+
473
+ ---
474
+
475
+ ## Slash Commands
476
+
477
+ | Command | What it does |
478
+ |---------|-------------|
479
+ | `/ship` | AI-writes commit message → `git add -A && git commit` → `git push` → `gh pr create` |
480
+ | `/learn [text]` | Save a lesson to `decisions.md`. No text = AI extracts key takeaway from last output |
481
+ | `/handoff` | AI writes a structured briefing (Context / Current state / Next steps / Watch out for), saved to `.polymath/handoffs/` |
482
+ | `/simplify` | Simplify and tighten last output |
483
+ | `/explain` | Explain last output in plain language |
484
+
485
+ ### Custom slash commands
486
+
487
+ Drop a `.md` file in `.polymath/commands/` (project) or `~/.polymath/commands/` (global):
488
+
489
+ ```markdown
490
+ ---
491
+ name: deploy
492
+ description: Deploy to staging
493
+ shell_before: npm run build
494
+ shell_after: echo "{{ai_output}}" | pbcopy
495
+ ---
496
+ Deploy the following changes. Session context: {{session_output}}
497
+ ```
498
+
499
+ ---
500
+
501
+ ## Subagents
502
+
503
+ Post-pipeline focused passes on the output.
504
+
505
+ | Subagent | Trigger | What it does |
506
+ |----------|---------|-------------|
507
+ | `code-simplifier` | `--simplify` | Removes complexity, improves naming |
508
+ | `verify` | `--verify` | Runs tests, fixes failures — up to 3 iterations |
509
+ | `security-scan` | security tasks | Flags injections, hardcoded creds |
510
+ | `test-writer` | code tasks | Generates test cases |
511
+ | `docs-writer` | code tasks | Generates inline documentation |
512
+ | `brainstorm-critic` | planning tasks | Challenges assumptions |
513
+ | `perf-reviewer` | code tasks | Spots performance hotspots |
514
+ | `diff-explainer` | code tasks | Plain-language diff explanation |
515
+
516
+ Custom subagents: `.md` files in `.polymath/subagents/` or `~/.polymath/subagents/`.
517
+
518
+ ---
519
+
520
+ ## Auth & Accounts
521
+
522
+ ```
523
+ /accounts show all authenticated accounts per provider
524
+ /accounts switch gemini <email> switch active Gemini account
525
+ /accounts switch claude instructions to switch Claude account
526
+ /accounts switch openai instructions to switch OpenAI/Codex account
527
+ ```
528
+
529
+ When a token expires mid-session, polymath catches the 401, prints the exact re-login command, and re-probes automatically. For Gemini, a silent token refresh is attempted first via `refresh_token` — no interruption for routine expiry.
530
+
531
+ ---
532
+
533
+ ## Session Management
534
+
535
+ ```
536
+ /sessions list all past sessions
537
+ /session <id> resume a session
538
+ /session delete <id> delete a session
539
+ /new start a new session
540
+ ```
541
+
542
+ All sessions stored locally in `~/.polymath/context.db` (SQLite). Exported to Markdown in `~/.polymath/sessions/`. Nothing sent anywhere beyond the model API calls.
543
+
544
+ Use `/state` in the REPL to inspect:
545
+ - active session
546
+ - active project source (`repo` vs `local`)
547
+ - git repo / branch / dirty state
548
+ - loaded context files
549
+ - command and subagent provenance
550
+ - allowed tools from repo settings
551
+
552
+ ---
553
+
554
+ ## Parallel Sessions
555
+
556
+ Run polymath in multiple terminal tabs — one per workstream. OS notifications fire on pipeline completion so you can switch tabs without watching the screen.
557
+
558
+ ```
559
+ /parallel show how many other polymath sessions are running
560
+ ```
561
+
562
+ ---
563
+
564
+ ## All REPL Commands
565
+
566
+ ```
567
+ /models list detected models
568
+ /state show active session, project, repo, and permissions state
569
+ /queue list queued commands
570
+ /queue drop <n> remove queued command
571
+ /queue move <a> <b> reorder queued commands
572
+ /sessions list past sessions
573
+ /session <id> resume a session
574
+ /session delete <id> delete a session
575
+ /new new session
576
+ /project new <name> create a new project
577
+ /project use <name> switch to a project
578
+ /project list list all projects
579
+ /project init create .polymath/ in current directory
580
+ /context show <type> display a context file
581
+ /context add <t> <e> append entry to context
582
+ /context edit <type> open in $EDITOR
583
+ /ship commit + push + PR
584
+ /learn [text] save a lesson to decisions.md
585
+ /handoff write a teammate handoff brief
586
+ /simplify simplify last output
587
+ /explain explain last output
588
+ /<command> run any custom slash command
589
+ /commands list all slash commands
590
+ /subagents list all subagents
591
+ /accounts list authenticated accounts
592
+ /accounts switch ... switch account (see above)
593
+ /parallel show other active polymath sessions
594
+ /memory show chunk counts, embedder, last writeback
595
+ /memory rebuild re-chunk + re-embed all context
596
+ /memory search <q> debug retrieval
597
+ /memory confirm <id> promote candidate chunk → confirmed
598
+ /memory diff preview writeback for current session
599
+ /memory test verify embedder config
600
+ /setup re-run setup wizard
601
+ /exit exit
602
+ ```
603
+
604
+ ---
605
+
606
+ ## Global Config (`~/.polymath/`)
607
+
608
+ ```
609
+ ~/.polymath/
610
+ ├── config.json # settings, API keys, priority profile
611
+ ├── history.txt # REPL command history
612
+ ├── context.db # SQLite: all sessions + messages
613
+ ├── sessions/ # Markdown session exports
614
+ ├── commands/ # global custom slash commands
615
+ ├── subagents/ # global custom subagents
616
+ ├── cache/ # local-only, derived from .polymath/memory/
617
+ │ └── <project_id>/
618
+ │ └── chunks.db # sqlite + sqlite-vec retrieval index
619
+ └── projects/
620
+ └── <project-name>/
621
+ ├── context/
622
+ │ ├── rules.md
623
+ │ ├── logic.md
624
+ │ ├── code.md
625
+ │ ├── stack.md
626
+ │ ├── data.md
627
+ │ ├── goals.md
628
+ │ ├── decisions.md
629
+ │ ├── glossary.md
630
+ │ └── personas.md
631
+ └── memory/ # detached fallback when no .polymath/ exists
632
+ ├── chunks/*.jsonl
633
+ └── manifest.json
634
+ ```
635
+
636
+ ---
637
+
638
+ ## Architecture
639
+
640
+ ```
641
+ polymath/
642
+ ├── main.py startup wiring + REPL bootstrap
643
+ ├── orchestrator/
644
+ │ ├── run_controller.py request orchestration, mode selection, shared agent flow
645
+ │ ├── attempt_ledger.py model-attempt tracking to prevent fallback loops
646
+ │ ├── output_policy.py concise user-facing orchestration messages
647
+ │ ├── state_responder.py answer runtime/state questions from local polymath state
648
+ │ ├── race.py with_failover / race_first_success / gather_all helpers
649
+ │ └── worker_pool.py manages concurrent model/subagent workers
650
+ ├── memory/
651
+ │ ├── store.py per-type JSONL chunks + sqlite-vec cache
652
+ │ ├── embedder.py Gemini primary, OpenAI fallback
653
+ │ ├── chunker.py heading-aware markdown splitter
654
+ │ ├── retriever.py top-K hybrid retrieval, drop-in for build_context_injection
655
+ │ ├── writer.py end-of-turn extraction → candidate chunks
656
+ │ ├── migrate.py one-shot bootstrap from existing .md context
657
+ │ └── sync.py re-index a single source after edit
658
+ ├── command_service.py built-in + custom REPL command handling
659
+ ├── execution_service.py ask/pipeline execution on top of orchestrator policy
660
+ ├── command_registry.py REPL completion + command registry helpers
661
+ ├── ui_state.py footer, queue, and status rendering helpers
662
+ ├── domain.py Project, SessionState, ModelSelection, RunContext
663
+ ├── pipeline.py task pipeline + ReAct agentic loop
664
+ ├── router.py task classification heuristics
665
+ ├── model_policy.py model selection, adapter creation, fallback policy
666
+ ├── permissions.py permission parsing + repo allow-list checks
667
+ ├── detector.py model auto-detection, token refresh, account management
668
+ ├── project_runtime.py repo-first project resolution + git provenance
669
+ ├── config.py MODEL_REGISTRY, ModelInfo, TaskType, CostTier
670
+ ├── tools.py tool registry: list_dir, read_file, write_file, run_shell
671
+ ├── context_store.py SQLite session storage, message log, cost tracking
672
+ ├── context_manager.py typed project context: read/write/inject
673
+ ├── project_config.py .polymath/ discovery, settings, tool allow-lists
674
+ ├── slash_commands.py slash command registry + /ship, /learn, /handoff
675
+ ├── subagents.py subagent registry, verify loop, auto-selection
676
+ ├── compressor.py context window management + summarization
677
+ ├── workspace.py project structure scan
678
+ ├── setup_wizard.py first-run setup + model configuration
679
+ ├── bootstrap.py report missing dependencies on startup
680
+ └── adapters/
681
+ ├── base.py BaseAdapter, Message, ToolCall, AuthExpiredError
682
+ ├── claude.py Anthropic async adapter
683
+ ├── gemini.py Google Gemini async adapter (OAuth + API key)
684
+ ├── openai_adapter.py OpenAI-compatible async adapter (OpenAI + LM Studio)
685
+ └── ollama.py Ollama local async adapter
686
+ ```
687
+
688
+ ### Orchestrator responsibilities
689
+
690
+ - `run_controller.py` decides whether a request is local state, direct chat, or pipeline work
691
+ - `attempt_ledger.py` records which models already failed for a phase so fallback does not loop
692
+ - `output_policy.py` turns provider failures into short user-facing orchestration messages
693
+ - `worker_pool.py` manages concurrent internal workers without changing the single-threaded terminal UX