potetos-for-everyone 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (144) hide show
  1. package/.agents/plugins/marketplace.json +21 -0
  2. package/.claude-plugin/marketplace.json +32 -0
  3. package/.claude-plugin/plugin.json +36 -0
  4. package/.codex-plugin/plugin.json +13 -0
  5. package/.codex-plugin/prompts/architect.md +6 -0
  6. package/.codex-plugin/prompts/arena.md +6 -0
  7. package/.codex-plugin/prompts/automate-me.md +6 -0
  8. package/.codex-plugin/prompts/blast-radius.md +6 -0
  9. package/.codex-plugin/prompts/bro.md +6 -0
  10. package/.codex-plugin/prompts/create-verification-skill.md +6 -0
  11. package/.codex-plugin/prompts/figure-it-out.md +6 -0
  12. package/.codex-plugin/prompts/how.md +6 -0
  13. package/.codex-plugin/prompts/interrogate.md +6 -0
  14. package/.codex-plugin/prompts/maintain-verification-skill.md +6 -0
  15. package/.codex-plugin/prompts/make-bot-ui.md +6 -0
  16. package/.codex-plugin/prompts/no-comments.md +6 -0
  17. package/.codex-plugin/prompts/poteto-mode.md +6 -0
  18. package/.codex-plugin/prompts/recall.md +6 -0
  19. package/.codex-plugin/prompts/reflect.md +6 -0
  20. package/.codex-plugin/prompts/setup-pstack.md +6 -0
  21. package/.codex-plugin/prompts/show-me-your-work.md +6 -0
  22. package/.codex-plugin/prompts/swarm.md +6 -0
  23. package/.codex-plugin/prompts/tdd.md +6 -0
  24. package/.codex-plugin/prompts/teach.md +6 -0
  25. package/.codex-plugin/prompts/technical-writing.md +6 -0
  26. package/.codex-plugin/prompts/typescript-best-practices.md +6 -0
  27. package/.codex-plugin/prompts/unslop.md +6 -0
  28. package/.codex-plugin/prompts/why.md +6 -0
  29. package/AGENTS.md +11 -0
  30. package/LICENSE +22 -0
  31. package/NOTICE.md +17 -0
  32. package/PORTABILITY.md +29 -0
  33. package/README.md +483 -0
  34. package/UPSTREAM.json +9 -0
  35. package/UPSTREAM_INVENTORY.json +13 -0
  36. package/VERSION +1 -0
  37. package/adapters/README.md +9 -0
  38. package/adapters/registry.json +46 -0
  39. package/agents/comment-reviewer.md +3 -0
  40. package/agents/comment-sicko.md +31 -0
  41. package/agents/poteto-agent.md +14 -0
  42. package/assets/logo.png +0 -0
  43. package/automations/benny/FOR_AGENTS.md +5 -0
  44. package/automations/benny/README.md +10 -0
  45. package/automations/benny/skills/reproduce-and-fix-issues/SKILL.md +12 -0
  46. package/automations/benny/skills/triage-issue-reports/SKILL.md +11 -0
  47. package/automations/benny/templates/configuration.example.yaml +11 -0
  48. package/automations/benny/templates/reproduce-automation-prompt.md +1 -0
  49. package/automations/benny/templates/triage-automation-prompt.md +1 -0
  50. package/bin/potetos +547 -0
  51. package/docs/.nojekyll +0 -0
  52. package/docs/guide/01-setup.md +25 -0
  53. package/docs/guide/02-poteto-mode.md +11 -0
  54. package/docs/guide/03-understand.md +5 -0
  55. package/docs/guide/04-design.md +5 -0
  56. package/docs/guide/05-build-and-clean.md +5 -0
  57. package/docs/guide/06-verify-and-ship.md +5 -0
  58. package/docs/guide/07-overnight.md +5 -0
  59. package/docs/guide/08-principles.md +5 -0
  60. package/docs/guide/09-make-it-yours.md +5 -0
  61. package/docs/guide/10-recipes-and-pitfalls.md +8 -0
  62. package/docs/guide/README.md +16 -0
  63. package/docs/index.html +422 -0
  64. package/hooks/hooks.json +17 -0
  65. package/hooks/session-start-context.md +11 -0
  66. package/npm/core.mjs +318 -0
  67. package/npm/postinstall.mjs +22 -0
  68. package/npm/potetos.mjs +145 -0
  69. package/package.json +98 -0
  70. package/runtime/README.md +21 -0
  71. package/runtime/__init__.py +1 -0
  72. package/runtime/capabilities.json +36 -0
  73. package/runtime/config.example.json +17 -0
  74. package/runtime/runner.py +240 -0
  75. package/skills/architect/SKILL.md +14 -0
  76. package/skills/arena/SKILL.md +14 -0
  77. package/skills/automate-me/SKILL.md +11 -0
  78. package/skills/blast-radius/SKILL.md +11 -0
  79. package/skills/bro/SKILL.md +7 -0
  80. package/skills/create-verification-skill/SKILL.md +11 -0
  81. package/skills/figure-it-out/SKILL.md +13 -0
  82. package/skills/how/SKILL.md +14 -0
  83. package/skills/interrogate/SKILL.md +14 -0
  84. package/skills/maintain-verification-skill/SKILL.md +11 -0
  85. package/skills/make-bot-ui/SKILL.md +11 -0
  86. package/skills/no-comments/SKILL.md +11 -0
  87. package/skills/poteto-mode/SKILL.md +93 -0
  88. package/skills/poteto-mode/playbooks/authoring-a-skill.md +7 -0
  89. package/skills/poteto-mode/playbooks/autonomous-run.md +7 -0
  90. package/skills/poteto-mode/playbooks/autopilot-full.md +7 -0
  91. package/skills/poteto-mode/playbooks/autopilot-stack.md +7 -0
  92. package/skills/poteto-mode/playbooks/babysit.md +7 -0
  93. package/skills/poteto-mode/playbooks/bug-fix.md +8 -0
  94. package/skills/poteto-mode/playbooks/eval.md +7 -0
  95. package/skills/poteto-mode/playbooks/feature.md +8 -0
  96. package/skills/poteto-mode/playbooks/hillclimb.md +8 -0
  97. package/skills/poteto-mode/playbooks/investigation.md +7 -0
  98. package/skills/poteto-mode/playbooks/multi-phase-plan.md +7 -0
  99. package/skills/poteto-mode/playbooks/opening-a-pr.md +7 -0
  100. package/skills/poteto-mode/playbooks/orchestrate.md +8 -0
  101. package/skills/poteto-mode/playbooks/pause-safely.md +7 -0
  102. package/skills/poteto-mode/playbooks/perf-issue.md +8 -0
  103. package/skills/poteto-mode/playbooks/prototype.md +7 -0
  104. package/skills/poteto-mode/playbooks/refactoring.md +7 -0
  105. package/skills/poteto-mode/playbooks/runtime-forensics.md +7 -0
  106. package/skills/poteto-mode/playbooks/session-pickup.md +7 -0
  107. package/skills/poteto-mode/playbooks/shipping.md +7 -0
  108. package/skills/poteto-mode/playbooks/trace-forensics.md +7 -0
  109. package/skills/poteto-mode/playbooks/visual-parity.md +7 -0
  110. package/skills/poteto-mode/playbooks/worktree-cleanup.md +7 -0
  111. package/skills/principle-attack-the-premise/SKILL.md +20 -0
  112. package/skills/principle-boundary-discipline/SKILL.md +20 -0
  113. package/skills/principle-build-the-lever/SKILL.md +20 -0
  114. package/skills/principle-encode-lessons-in-structure/SKILL.md +20 -0
  115. package/skills/principle-exhaust-the-design-space/SKILL.md +20 -0
  116. package/skills/principle-experience-first/SKILL.md +20 -0
  117. package/skills/principle-fix-root-causes/SKILL.md +20 -0
  118. package/skills/principle-foundational-thinking/SKILL.md +20 -0
  119. package/skills/principle-guard-the-context-window/SKILL.md +20 -0
  120. package/skills/principle-laziness-protocol/SKILL.md +20 -0
  121. package/skills/principle-make-operations-idempotent/SKILL.md +20 -0
  122. package/skills/principle-migrate-callers-then-delete-legacy-apis/SKILL.md +20 -0
  123. package/skills/principle-minimize-reader-load/SKILL.md +20 -0
  124. package/skills/principle-model-the-domain/SKILL.md +20 -0
  125. package/skills/principle-never-block-on-the-human/SKILL.md +20 -0
  126. package/skills/principle-outcome-oriented-execution/SKILL.md +20 -0
  127. package/skills/principle-prove-it-works/SKILL.md +20 -0
  128. package/skills/principle-redesign-from-first-principles/SKILL.md +20 -0
  129. package/skills/principle-separate-before-serializing-shared-state/SKILL.md +20 -0
  130. package/skills/principle-sequence-verifiable-units/SKILL.md +20 -0
  131. package/skills/principle-subtract-before-you-add/SKILL.md +20 -0
  132. package/skills/principle-test-behavior-not-implementation/SKILL.md +20 -0
  133. package/skills/principle-type-system-discipline/SKILL.md +20 -0
  134. package/skills/recall/SKILL.md +13 -0
  135. package/skills/reflect/SKILL.md +11 -0
  136. package/skills/setup-pstack/SKILL.md +12 -0
  137. package/skills/show-me-your-work/SKILL.md +17 -0
  138. package/skills/swarm/SKILL.md +12 -0
  139. package/skills/tdd/SKILL.md +12 -0
  140. package/skills/teach/SKILL.md +11 -0
  141. package/skills/technical-writing/SKILL.md +12 -0
  142. package/skills/typescript-best-practices/SKILL.md +14 -0
  143. package/skills/unslop/SKILL.md +11 -0
  144. package/skills/why/SKILL.md +14 -0
@@ -0,0 +1,240 @@
1
+ from __future__ import annotations
2
+
3
+ import concurrent.futures
4
+ import datetime as dt
5
+ import json
6
+ import os
7
+ import re
8
+ import secrets
9
+ import subprocess
10
+ import tempfile
11
+ from dataclasses import dataclass, asdict
12
+ from pathlib import Path
13
+ from typing import Any
14
+
15
+
16
+ @dataclass
17
+ class RunResult:
18
+ runner: str
19
+ returncode: int
20
+ stdout_file: str
21
+ stderr_file: str
22
+ workspace: str
23
+ branch: str | None
24
+ command: list[str]
25
+ timed_out: bool = False
26
+
27
+
28
+ def _slug(value: str) -> str:
29
+ value = re.sub(r"[^A-Za-z0-9._-]+", "-", value).strip("-")
30
+ return value[:48] or "worker"
31
+
32
+
33
+ def _run_id() -> str:
34
+ stamp = dt.datetime.now(dt.timezone.utc).strftime("%Y%m%dT%H%M%SZ")
35
+ return f"{stamp}-{secrets.token_hex(3)}"
36
+
37
+
38
+ def load_config(path: Path) -> dict[str, Any]:
39
+ if not path.exists():
40
+ raise FileNotFoundError(
41
+ f"runner config not found: {path}. Run setup-pstack or create .potetos/config.json."
42
+ )
43
+ data = json.loads(path.read_text(encoding="utf-8"))
44
+ runners = data.get("runners")
45
+ if not isinstance(runners, dict) or not runners:
46
+ raise ValueError("config must contain a non-empty 'runners' object")
47
+ for name, spec in runners.items():
48
+ if not isinstance(spec, dict):
49
+ raise ValueError(f"runner {name!r} must be an object")
50
+ cmd = spec.get("command")
51
+ if not isinstance(cmd, list) or not cmd or not all(isinstance(x, str) and x for x in cmd):
52
+ raise ValueError(f"runner {name!r}.command must be a non-empty string array")
53
+ if "environment" in spec and not isinstance(spec["environment"], dict):
54
+ raise ValueError(f"runner {name!r}.environment must be an object")
55
+ if "env_passthrough" in spec and (not isinstance(spec["env_passthrough"], list) or not all(isinstance(x, str) for x in spec["env_passthrough"])):
56
+ raise ValueError(f"runner {name!r}.env_passthrough must be a string array")
57
+ return data
58
+
59
+
60
+ def _git_root(workspace: Path) -> Path:
61
+ p = subprocess.run(
62
+ ["git", "-C", str(workspace), "rev-parse", "--show-toplevel"],
63
+ text=True,
64
+ capture_output=True,
65
+ )
66
+ if p.returncode:
67
+ raise RuntimeError(f"--isolate requires a git worktree: {p.stderr.strip()}")
68
+ return Path(p.stdout.strip()).resolve()
69
+
70
+
71
+ def _isolated_worktree(workspace: Path, run_id: str, runner: str) -> tuple[Path, str]:
72
+ root = _git_root(workspace)
73
+ status = subprocess.run(
74
+ ["git", "-C", str(root), "status", "--porcelain"],
75
+ text=True,
76
+ capture_output=True,
77
+ check=True,
78
+ ).stdout.splitlines()
79
+ dirty = []
80
+ for line in status:
81
+ path = line[3:] if len(line) > 3 else line
82
+ if path == ".potetos" or path.startswith(".potetos/"):
83
+ continue
84
+ dirty.append(line)
85
+ if dirty:
86
+ raise RuntimeError(
87
+ "--isolate refuses a dirty base because a new worktree would silently omit uncommitted changes: "
88
+ + "; ".join(dirty[:8])
89
+ )
90
+ parent = Path(tempfile.gettempdir()) / "potetos-worktrees" / _slug(root.name)
91
+ parent.mkdir(parents=True, exist_ok=True)
92
+ dest = parent / f"{run_id}-{_slug(runner)}"
93
+ branch = f"potetos/{run_id}/{_slug(runner)}"
94
+ p = subprocess.run(
95
+ ["git", "-C", str(root), "worktree", "add", "-b", branch, str(dest), "HEAD"],
96
+ text=True,
97
+ capture_output=True,
98
+ )
99
+ if p.returncode:
100
+ raise RuntimeError(f"failed to create isolated worktree: {p.stderr.strip()}")
101
+ return dest, branch
102
+
103
+
104
+ def _render_command(parts: list[str], *, prompt: str, prompt_file: Path, workspace: Path, output_file: Path) -> list[str]:
105
+ values = {
106
+ "prompt": prompt,
107
+ "prompt_file": str(prompt_file),
108
+ "workspace": str(workspace),
109
+ "output_file": str(output_file),
110
+ }
111
+ rendered = []
112
+ for part in parts:
113
+ try:
114
+ rendered.append(part.format_map(values))
115
+ except KeyError as e:
116
+ raise ValueError(f"unknown runner command placeholder: {e.args[0]}") from e
117
+ return rendered
118
+
119
+
120
+ def run_worker(
121
+ *,
122
+ config: dict[str, Any],
123
+ runner: str,
124
+ prompt: str,
125
+ workspace: Path,
126
+ runs_dir: Path,
127
+ run_id: str | None = None,
128
+ isolate: bool = False,
129
+ ) -> RunResult:
130
+ run_id = run_id or _run_id()
131
+ runners = config["runners"]
132
+ if runner not in runners:
133
+ raise KeyError(f"runner {runner!r} not configured; available: {', '.join(sorted(runners))}")
134
+ spec = runners[runner]
135
+
136
+ branch = None
137
+ actual_workspace = workspace.resolve()
138
+ if isolate:
139
+ actual_workspace, branch = _isolated_worktree(actual_workspace, run_id, runner)
140
+
141
+ outdir = runs_dir.resolve() / run_id / _slug(runner)
142
+ outdir.mkdir(parents=True, exist_ok=True)
143
+ prompt_file = outdir / "prompt.txt"
144
+ stdout_file = outdir / "stdout.txt"
145
+ stderr_file = outdir / "stderr.txt"
146
+ result_file = outdir / "result.json"
147
+ prompt_file.write_text(prompt, encoding="utf-8")
148
+
149
+ command = _render_command(
150
+ spec["command"],
151
+ prompt=prompt,
152
+ prompt_file=prompt_file,
153
+ workspace=actual_workspace,
154
+ output_file=stdout_file,
155
+ )
156
+ env = os.environ.copy()
157
+ env.update({str(k): str(v) for k, v in spec.get("environment", {}).items()})
158
+ for key in spec.get("env_passthrough", []):
159
+ if key not in os.environ:
160
+ raise ValueError(f"runner {runner!r} requires missing environment variable {key}")
161
+ env[key] = os.environ[key]
162
+ timeout = int(spec.get("timeout_seconds", 1800))
163
+ stdin_text = prompt if spec.get("stdin", False) else None
164
+
165
+ timed_out = False
166
+ try:
167
+ p = subprocess.run(
168
+ command,
169
+ cwd=actual_workspace,
170
+ input=stdin_text,
171
+ text=True,
172
+ capture_output=True,
173
+ timeout=timeout,
174
+ env=env,
175
+ )
176
+ returncode = p.returncode
177
+ stdout = p.stdout
178
+ stderr = p.stderr
179
+ except subprocess.TimeoutExpired as e:
180
+ timed_out = True
181
+ returncode = 124
182
+ stdout = e.stdout.decode() if isinstance(e.stdout, bytes) else (e.stdout or "")
183
+ stderr = e.stderr.decode() if isinstance(e.stderr, bytes) else (e.stderr or "")
184
+ stderr += f"\npotetos: runner timed out after {timeout}s\n"
185
+
186
+ stdout_file.write_text(stdout, encoding="utf-8")
187
+ stderr_file.write_text(stderr, encoding="utf-8")
188
+ result = RunResult(
189
+ runner=runner,
190
+ returncode=returncode,
191
+ stdout_file=str(stdout_file),
192
+ stderr_file=str(stderr_file),
193
+ workspace=str(actual_workspace),
194
+ branch=branch,
195
+ command=command,
196
+ timed_out=timed_out,
197
+ )
198
+ result_file.write_text(json.dumps(asdict(result), indent=2) + "\n", encoding="utf-8")
199
+ return result
200
+
201
+
202
+ def run_panel(
203
+ *,
204
+ config: dict[str, Any],
205
+ runner_names: list[str],
206
+ prompt: str,
207
+ workspace: Path,
208
+ runs_dir: Path,
209
+ isolate: bool = False,
210
+ max_parallel: int | None = None,
211
+ ) -> tuple[str, list[RunResult]]:
212
+ run_id = _run_id()
213
+ names = runner_names or list(config["runners"])
214
+ if not names:
215
+ raise ValueError("panel requires at least one runner")
216
+ if len(set(names)) != len(names):
217
+ raise ValueError("panel runner names must be unique so evidence remains attributable")
218
+ parallel = max_parallel or min(len(names), int(config.get("max_parallel", 4)))
219
+ results: list[RunResult] = []
220
+ with concurrent.futures.ThreadPoolExecutor(max_workers=max(1, parallel)) as pool:
221
+ futures = {
222
+ pool.submit(
223
+ run_worker,
224
+ config=config,
225
+ runner=name,
226
+ prompt=prompt,
227
+ workspace=workspace,
228
+ runs_dir=runs_dir,
229
+ run_id=run_id,
230
+ isolate=isolate,
231
+ ): name
232
+ for name in names
233
+ }
234
+ for future in concurrent.futures.as_completed(futures):
235
+ results.append(future.result())
236
+ results.sort(key=lambda r: names.index(r.runner))
237
+ summary = runs_dir.resolve() / run_id / "panel.json"
238
+ summary.parent.mkdir(parents=True, exist_ok=True)
239
+ summary.write_text(json.dumps([asdict(r) for r in results], indent=2) + "\n", encoding="utf-8")
240
+ return run_id, results
@@ -0,0 +1,14 @@
1
+ ---
2
+ name: architect
3
+ description: Settle data shapes, ownership, boundaries, and interfaces before code crosses a function/module boundary. Use for non-trivial design work.
4
+ ---
5
+ # Architect
6
+
7
+ Read the caller first. Architecture starts from use, not from an isolated abstraction.
8
+
9
+ 1. Name the domain data shape and invariants.
10
+ 2. Map ownership and lifecycle. For concurrency, identify what is truly shared before proposing synchronization.
11
+ 3. Draft 2-3 boundary shapes when the decision is contested. If `delegate` is available, assign independent designs; otherwise keep alternatives explicit and sequential.
12
+ 4. Compare on reader load, invalid states, migration cost, idempotence, testability, and fit with existing architecture.
13
+ 5. Select the smallest coherent shape and write the caller-facing signature/contract before implementation.
14
+ 6. Use `interrogate` for high-risk or contested decisions.
@@ -0,0 +1,14 @@
1
+ ---
2
+ name: arena
3
+ description: Spawn parallel candidate implementations for the same task, cross-judge, and graft the best parts into the winning base. Use when exploring competing designs or critical artifacts.
4
+ ---
5
+ # Arena
6
+
7
+ Fan out N parallel attempts at the same task, score them against concrete criteria, pick a base, and graft the best ideas from the others.
8
+
9
+ 1. Frame: Define the artifact contract, 3-6 concrete rubric criteria, and isolated candidate output paths.
10
+ 2. Fan out: Run parallel candidate attempts on isolated workers (via `delegate` when supported, or labeled sequential passes). Require a concrete artifact and short rationale from each.
11
+ 3. Cross-judge: Score each candidate against the rubric criteria, compare trade-offs, and recommend the strongest base.
12
+ 4. Pick a base: Select the implementation that is cleanest, easiest to maintain, and safest under domain invariants.
13
+ 5. Graft: Port the best independent ideas or edge-case handling from losing candidates into the base by hand.
14
+ 6. Verify: Validate the synthesized result against the full test and verification suite.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: automate-me
3
+ description: Create a personal mode skill from recurring workflow patterns in available history. Use when the user wants their own agent style encoded.
4
+ ---
5
+ # Automate Me
6
+
7
+ 1. Gather a representative sample of the user's actual completed workflows from available, authorized history. Do not infer private history you cannot access.
8
+ 2. Extract repeated triggers, decisions, quality gates, preferred artifacts, and successful repair loops. Ignore one-off stylistic quirks unless explicitly requested.
9
+ 3. Separate universal engineering principles from user-specific routing preferences.
10
+ 4. Draft a `<name>-mode` Agent Skill that routes through existing skills rather than duplicating them.
11
+ 5. Validate triggers against positive and negative examples, then present the generated skill and evidence behind each rule.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: blast-radius
3
+ description: Prove what a small-looking change can affect before or after editing. Use for compatibility, caller, state, schema, or behavior impact analysis.
4
+ ---
5
+ # Blast Radius
6
+
7
+ 1. Name the changed contract: type, value, state transition, API, storage shape, timing, style token, or side effect.
8
+ 2. Enumerate direct readers/writers/callers and any serialized/public boundaries.
9
+ 3. Trace transitive assumptions by search, type references, tests, runtime wiring, and history as available.
10
+ 4. Identify the strongest fact that would make the change safe and prove that fact by running or inspecting code when possible.
11
+ 5. Return affected surfaces grouped as definite, plausible, and ruled out with evidence. Never write "safe" without the proof behind it.
@@ -0,0 +1,7 @@
1
+ ---
2
+ name: bro
3
+ description: Restate the last technical explanation in plain human language with minimal jargon. Use when the user asks for a simpler version.
4
+ ---
5
+ # Bro
6
+
7
+ Restate the answer in ordinary language. Keep the causal chain and the important caveat. Replace jargon with a concrete example where possible. Do not dumb down the actual conclusion or add new claims.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: create-verification-skill
3
+ description: Generate a project-local verification skill that drives the real app and captures evidence. Use when setting up automated verification for a repo.
4
+ ---
5
+ # Create Verification Skill
6
+
7
+ 1. Interview the repository to discover primary user-facing surfaces (web UI, CLI, API, app).
8
+ 2. Identify the canonical local start command, ports, environment variables, and required seed data.
9
+ 3. Determine how an agent can programmatically drive the system (browser/CDP, PTY/CLI, HTTP requests) and capture evidence (screenshots, transcripts, logs, exit codes).
10
+ 4. Verify isolation requirements: identify if instances can run concurrently or need dedicated ports/profiles.
11
+ 5. Create `.agents/skills/verify-<app>/SKILL.md` with explicit launch, drive, observe, and cleanup instructions.
@@ -0,0 +1,13 @@
1
+ ---
2
+ name: figure-it-out
3
+ description: Design a rigorous custom playbook when no bundled playbook safely fits a large or unusual task. Use for cross-cutting ambiguous execution.
4
+ ---
5
+ # Figure It Out
6
+
7
+ 1. Define the completion predicate and non-goals.
8
+ 2. Inventory uncertainty. Split it into factual questions, empirical forks, product preferences, and irreversible decisions.
9
+ 3. Resolve facts by investigation and empirical forks by prototype instead of asking the human when tools can decide.
10
+ 4. Build a dependency graph of verifiable work units with a proof attached to every node.
11
+ 5. Identify which units can safely fan out and which share mutable state.
12
+ 6. Write the bespoke playbook, then execute it using existing skills rather than improvising around its gates.
13
+ 7. Maintain a decision trail for work that can outlive the current session.
@@ -0,0 +1,14 @@
1
+ ---
2
+ name: how
3
+ description: Trace how a subsystem works using code and runtime evidence. Use for architecture walkthroughs, data flow, call paths, and are-we-sure investigations.
4
+ ---
5
+ # How
6
+
7
+ Build an evidence-backed explanation of how the requested subsystem works.
8
+
9
+ 1. Find the public/consumer entry point, then trace data and control flow toward the effect. Do not start from filenames guessed by name alone.
10
+ 2. Read the core types and state ownership before narrating individual functions.
11
+ 3. Use `vcs`, tests, or `observe` when they resolve an ambiguity. If `delegate` exists, split independent exploration areas and reconcile them yourself.
12
+ 4. Draw a compact text or Mermaid diagram when it reduces reader load.
13
+ 5. Test at least one claim that could easily be wrong by running or inspecting the real path.
14
+ 6. Explain in consumer order: trigger -> boundary -> state/transform -> side effect -> observable result. Name file/symbol evidence and uncertainty.
@@ -0,0 +1,14 @@
1
+ ---
2
+ name: interrogate
3
+ description: Adversarially review a diff or design with independent lenses and evidence. Use before shipping contested or high-impact changes.
4
+ ---
5
+ # Interrogate
6
+
7
+ Reviewers are attackers, not style voters.
8
+
9
+ 1. Freeze the exact artifact/revision and acceptance criteria.
10
+ 2. Cover at least these lenses: correctness/invariants, boundary/API compatibility, concurrency/state, tests/verification quality, and maintainability/reader load. Add security/performance when relevant.
11
+ 3. Use independent workers/models when `delegate` exists. Otherwise run separate sequential lenses and disclose they share context.
12
+ 4. Require each finding to include a concrete failure mode and evidence or a reproducible check. Reject unsupported nits.
13
+ 5. The owning agent adjudicates every finding as fix, dismiss, or investigate. Never pass reviewer prose through unexamined.
14
+ 6. Re-run targeted proof after accepted fixes.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: maintain-verification-skill
3
+ description: Repair a project verification skill whose feature map or commands drifted. Use when app behavior and verification docs no longer agree.
4
+ ---
5
+ # Maintain Verification Skill
6
+
7
+ 1. Diff the current product surfaces against the verification feature map.
8
+ 2. Run one representative live pass before changing the skill so drift is observed, not guessed.
9
+ 3. Classify differences as product regression, intentional product change, environment drift, or verifier bug.
10
+ 4. Update only proven verifier drift. Do not rewrite expectations to make a product regression green.
11
+ 5. Re-run affected features plus one untouched control feature.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: make-bot-ui
3
+ description: Build a small operator UI over an agent/webhook backend. Use when buttons or controls should trigger an agent workflow.
4
+ ---
5
+ # Make Bot UI
6
+
7
+ 1. Define the operator actions, payload schema, authentication boundary, and observable completion state before UI code.
8
+ 2. Keep secrets server-side. Treat webhook URLs/tokens as credentials.
9
+ 3. Implement the thinnest UI that exposes state, errors, retries, and idempotent request IDs.
10
+ 4. Make every action safe under duplicate delivery or refresh.
11
+ 5. Verify one full round trip through the real backend when `observe`/network access exists. Otherwise ship a deterministic local mock plus an explicit integration gap.
@@ -0,0 +1,11 @@
1
+ ---
2
+ name: no-comments
3
+ description: Remove redundant comments, dead workaround explanations, and AI tells from code while preserving necessary constraints. Use when cleaning up code or diffs.
4
+ ---
5
+ # No Comments
6
+
7
+ 1. Identify comments that explain obvious code, restate syntax, or explain temporary workarounds that should be deleted.
8
+ 2. Distinguish legitimate constraints (hard external requirements, regulatory/spec references, non-obvious invariants) from redundant explanations.
9
+ 3. Convert valid constraint comments into types, tests, assertions, or lints whenever possible.
10
+ 4. Delete redundant commentary, commented-out dead code, and speculative notes.
11
+ 5. Verify that code remains readable and self-documenting through clear naming and small, focused functions.
@@ -0,0 +1,93 @@
1
+ ---
2
+ name: poteto-mode
3
+ description: Lauren Tan-inspired rigorous engineering mode for concise communication, deliberate decomposition, simple code, independent review when available, and runtime verification. Use for non-trivial engineering tasks or when the user asks for Poteto Mode/pstack-style rigor.
4
+ ---
5
+ # Poteto Mode, portable edition
6
+
7
+ This skill adapts the core workflow architecture of Lauren Tan's pstack to any capable AI agent. Credit and provenance live in the repository `NOTICE.md`.
8
+
9
+ ## Non-negotiables
10
+
11
+ 1. **Classify before acting.** Match the request to one playbook below. For large cross-cutting work with no clean match, use `figure-it-out`. For program-scale standing work, use `orchestrate`.
12
+ 2. **Name the data shape before non-trivial code.** Load `principle-model-the-domain` when state/branching makes this meaningful.
13
+ 3. **Resolve empirical forks empirically.** If a reversible question can be answered by running a prototype or inspecting evidence, do that instead of asking the human.
14
+ 4. **Cross a boundary deliberately.** Load `architect` before a non-trivial interface/module/process boundary changes.
15
+ 5. **Use parallelism only when it is real.** `arena`, `swarm`, and `interrogate` may use isolated `delegate` workers. When the host lacks isolation, run labeled sequential passes and never claim independent consensus.
16
+ 6. **Own delegated work.** Review artifacts and evidence yourself. A worker's "done" is not proof.
17
+ 7. **Finish with the real proof.** Load `principle-prove-it-works`. Verify through the closest consumer-facing artifact available. Name any missing runtime surface.
18
+ 8. **Keep the change small.** Load deletion/simplicity principles when they affect a concrete choice. Do not apply principles as decorative slogans.
19
+ 9. **Write plainly.** For durable docs load `technical-writing`; for any prose, apply `unslop`.
20
+ 10. **Respect host safety and permissions.** Reversible workspace work can proceed. Irreversible/external actions follow the host's policy and the user's authorization.
21
+
22
+ ## Capability behavior
23
+
24
+ Read the repository `PORTABILITY.md` when present. Feature-detect `run`, `vcs`, `delegate`, `search`, `observe`, and `track`. Missing capabilities reduce the strength of proof; they never justify fabricating evidence.
25
+
26
+ ## Routing
27
+
28
+ - `investigation`: Read-only investigation
29
+ - `bug-fix`: Bug fix
30
+ - `perf-issue`: Performance issue
31
+ - `hillclimb`: Metric hillclimb
32
+ - `runtime-forensics`: Runtime forensics
33
+ - `trace-forensics`: Captured trace forensics
34
+ - `feature`: Feature
35
+ - `refactoring`: Refactoring
36
+ - `prototype`: Prototype
37
+ - `visual-parity`: Visual parity
38
+ - `authoring-a-skill`: Authoring a skill
39
+ - `eval`: Skill or prompt evaluation
40
+ - `babysit`: Drive a PR to merge-ready
41
+ - `shipping`: Verify and ship
42
+ - `autonomous-run`: Autonomous run
43
+ - `orchestrate`: Project orchestration
44
+ - `autopilot-full`: Independent PR autopilot
45
+ - `autopilot-stack`: Linear stack autopilot
46
+ - `session-pickup`: Session pickup
47
+ - `pause-safely`: Pause safely
48
+ - `multi-phase-plan`: Multi-phase work
49
+ - `worktree-cleanup`: Worktree and local-state cleanup
50
+ - `opening-a-pr`: Open a pull request
51
+
52
+ The internal `opening-a-pr` playbook is normally a final step when the task includes creating a PR and the host can do so.
53
+
54
+ ## Execution
55
+
56
+ 1. Read the selected file in `playbooks/` completely.
57
+ 2. If the host offers a todo/task tracker, copy the playbook steps into it. Otherwise keep a compact checklist in working notes. Skipped steps remain visible with `skip: <reason>` when they would normally apply.
58
+ 3. Load only supporting skills/principles that a concrete step triggers. Avoid loading the entire stack into context.
59
+ 4. Execute in verifiable units. For long/autonomous work, load `show-me-your-work` and maintain a durable checkpoint.
60
+ 5. Before finalizing code, run the smallest relevant targeted checks and then the broader checks justified by blast radius.
61
+ 6. Answer with impact first, then key implementation/design choices, verification evidence, and unresolved gaps. Do not forward raw worker summaries.
62
+
63
+ ## Principle index
64
+
65
+ Load a leaf skill only when its rule changes a decision in this task.
66
+
67
+ - `principle-laziness-protocol`: Bias toward deletion and the smallest change that fully solves the problem.
68
+ - `principle-foundational-thinking`: Choose the core data structures, ownership, and sequencing first so downstream logic becomes simpler.
69
+ - `principle-redesign-from-first-principles`: Design the shape you would choose if the requirement had existed from day one, then migrate toward that shape.
70
+ - `principle-attack-the-premise`: Inventory who actually owns or exhibits the imbalance, then challenge the shared premise before writing another patch.
71
+ - `principle-subtract-before-you-add`: Remove dead weight and obsolete paths first; build on the simpler remaining system.
72
+ - `principle-minimize-reader-load`: Reduce indirection, mutable scope, and one-caller abstractions until the causal path is easy to hold in one mind.
73
+ - `principle-outcome-oriented-execution`: Converge on the target architecture rather than preserving temporary compatibility states as permanent complexity.
74
+ - `principle-experience-first`: Optimize for the end-user experience unless the cost violates an explicit constraint.
75
+ - `principle-exhaust-the-design-space`: Build a few materially different cheap prototypes and compare evidence before committing to one design.
76
+ - `principle-build-the-lever`: Prefer a script, codemod, generator, benchmark, or verifier that performs or proves the work repeatably over hand edits.
77
+ - `principle-model-the-domain`: Encode the domain in the right structure: state machine, typed model, table, registry, reducer, boundary, or collection instead of scattered conditionals.
78
+ - `principle-boundary-discipline`: Validate and normalize at boundaries, trust internal invariants, and keep business logic free of adapter noise.
79
+ - `principle-type-system-discipline`: Make invalid states hard or impossible to represent and parse external primitives into meaningful internal types at the edge.
80
+ - `principle-make-operations-idempotent`: Design repeated execution to converge on the same correct end state rather than multiplying side effects.
81
+ - `principle-migrate-callers-then-delete-legacy-apis`: Move callers to the new shape and delete the old path in the same migration wave instead of carrying a compatibility layer.
82
+ - `principle-separate-before-serializing-shared-state`: Eliminate unnecessary sharing or partition ownership before adding locks, queues, or coordination.
83
+ - `principle-prove-it-works`: Verify the requested behavior against the real artifact or surface whenever possible; a proxy check only proves the proxy.
84
+ - `principle-fix-root-causes`: Reproduce the symptom, trace causality until one mechanism explains it, and fix that mechanism rather than compensating for downstream effects.
85
+ - `principle-sequence-verifiable-units`: Break work into small units that each end with a proof and order delivery so every next unit builds on verified state.
86
+ - `principle-test-behavior-not-implementation`: Call the system the way its consumer does and assert meaningful observable output; avoid mocks or assertions that only mirror internal calls.
87
+ - `principle-guard-the-context-window`: Route bulk exploration to isolated workers or files and bring back compact evidence summaries, not raw floods.
88
+ - `principle-never-block-on-the-human`: Run the experiment or make the reversible best-effort change, then show the result. Ask only for genuine preference, authority, or irreversible decisions.
89
+ - `principle-encode-lessons-in-structure`: Turn recurring guidance into a type, lint, test, metadata flag, generator, runtime check, or skill so the system carries the lesson.
90
+
91
+ ## Autonomy
92
+
93
+ Prefer action over clarification for reversible work whose correct answer can be observed. Ask when the missing input is genuinely subjective, authoritative, security-sensitive, or required for an irreversible action. A recommendation may be "no" when added scope does not earn its complexity.
@@ -0,0 +1,7 @@
1
+ # Authoring a skill
2
+
3
+ 1. Define precise triggers, non-triggers, outcome, and required capabilities.
4
+ 2. Keep the SKILL.md focused. Move heavy examples, references, and scripts into progressive-disclosure files.
5
+ 3. Use portable capability language. Add a fallback for optional capabilities instead of hard-coding one host tool.
6
+ 4. Validate Agent Skills frontmatter and directory/name parity.
7
+ 5. Exercise the skill on representative positive and negative cases. Revise based on behavior, not aesthetics alone.
@@ -0,0 +1,7 @@
1
+ # Autonomous run
2
+
3
+ 1. Translate the request into an explicit completion predicate, constraints, and irreversible-action boundaries.
4
+ 2. Create a decision trail and a small queue of verifiable units.
5
+ 3. Loop: choose the highest-leverage unit, execute it, verify it, checkpoint the result, then recompute the queue from evidence.
6
+ 4. Do not idle waiting for a human on reversible work. Do not fabricate progress when a capability is unavailable.
7
+ 5. Stop only when the predicate is satisfied, an explicit safety boundary requires the human, or a blocker cannot be removed with available capabilities.
@@ -0,0 +1,7 @@
1
+ # Independent PR autopilot
2
+
3
+ 1. Inventory the independent PR/change queue and define merge criteria for each item.
4
+ 2. Assign one owner per item. Owners build, verify, address review/CI, and maintain a clean head.
5
+ 3. Before merge, obtain an independent verification verdict for that exact head.
6
+ 4. Allow the owner to land only after the independent verdict and current PR state agree.
7
+ 5. Reconcile the base after each merge and re-evaluate queued items for new conflicts or invalidated assumptions.
@@ -0,0 +1,7 @@
1
+ # Linear stack autopilot
2
+
3
+ 1. Define the ordered change stack and acceptance gate for every layer.
4
+ 2. Build each layer on the verified previous layer. Keep one coherent linear base and avoid parallel edits to shared stack state.
5
+ 3. Verify each layer before adding the next; repair at the lowest layer that explains a failure.
6
+ 4. Deliver the full reviewed stack without landing it unless the operator explicitly requested shipping.
7
+ 5. Report stack order, head revisions, per-layer proof, and any assumptions the operator must check before landing.
@@ -0,0 +1,7 @@
1
+ # Drive a PR to merge-ready
2
+
3
+ 1. Inspect current branch/PR state, review threads, conflicts, and CI before changing anything.
4
+ 2. Classify blockers into code defect, flaky/infrastructure failure, conflict, review request, or non-actionable noise.
5
+ 3. Resolve one concrete blocker at a time. Reproduce code failures when possible and keep fixes scoped.
6
+ 4. Re-run only the checks needed to establish a new state, then refresh PR status.
7
+ 5. Stop at merge-ready or a real external blocker. Report outstanding items exactly; do not equate green CI with safe-to-merge.
@@ -0,0 +1,8 @@
1
+ # Bug fix
2
+
3
+ 1. Reproduce the defect on the closest real surface available. Record exact observed and expected behavior.
4
+ 2. Trace the failing behavior to a root cause before editing. If the symptom cannot be reproduced, gather evidence rather than guessing.
5
+ 3. When a cheap automated path exists, add or identify a behavior-level check that fails for the same reason.
6
+ 4. Make the smallest change that fixes the root cause. Avoid nearby cleanup unless it reduces the fix itself.
7
+ 5. Re-run the reproduction and targeted checks, then the smallest broader suite that can catch collateral damage.
8
+ 6. Report root cause, change, proof, and any verification surface that was unavailable.
@@ -0,0 +1,7 @@
1
+ # Skill or prompt evaluation
2
+
3
+ 1. Define the behavior under test, score rubric, guardrails, and representative task set before seeing candidate outputs.
4
+ 2. Freeze the baseline and candidate. Randomize or blind judging when practical.
5
+ 3. Run both under comparable tools, models, context, and budgets.
6
+ 4. Score outcomes and failure modes, not writing similarity. Inspect regressions individually.
7
+ 5. Promote only when the candidate clears the primary metric without unacceptable guardrail regressions. Preserve the eval artifacts.
@@ -0,0 +1,8 @@
1
+ # Feature
2
+
3
+ 1. Name the user-visible outcome and the core data shape/state transition before writing logic.
4
+ 2. Inspect the existing caller, boundary, and test/verification path. Reuse the native shape when it already fits.
5
+ 3. Write a throughput checkpoint: what can be implemented and verified as one coherent unit, and what is explicitly out of scope.
6
+ 4. Settle cross-boundary types/contracts before implementation. Prototype only genuine empirical forks.
7
+ 5. Implement the smallest end-to-end slice that produces the outcome.
8
+ 6. Verify through the consumer-facing path, plus targeted automated checks. Report the outcome before implementation detail.
@@ -0,0 +1,8 @@
1
+ # Metric hillclimb
2
+
3
+ 1. Define one primary metric, guardrail metrics, target, and reproducible benchmark command.
4
+ 2. Capture and save the baseline.
5
+ 3. Create a ranked hypothesis queue. Change one meaningful variable per iteration.
6
+ 4. For each iteration, benchmark before accepting it. Keep only wins that survive the guardrails.
7
+ 5. Checkpoint each accepted win independently so the sequence is bisectable and reversible.
8
+ 6. Stop at the target, a clearly exhausted hypothesis space, or a real blocker. Summarize accepted and rejected hypotheses with evidence.
@@ -0,0 +1,7 @@
1
+ # Read-only investigation
2
+
3
+ 1. Restate the question as a falsifiable claim or a small set of claims.
4
+ 2. Map the evidence that can answer it: code, tests, history, docs, tickets, runtime data, or connected sources. Read the highest-signal source first.
5
+ 3. Trace the relevant call/data path end to end. Separate observed facts from inference.
6
+ 4. Try to disprove the leading explanation with one targeted counter-check.
7
+ 5. Answer the question directly. Cite concrete files, revisions, commands, or external evidence that were actually inspected. Name unresolved gaps.
@@ -0,0 +1,7 @@
1
+ # Multi-phase work
2
+
3
+ 1. Define the target end state and phase boundaries in terms of observable acceptance criteria.
4
+ 2. Order phases so each one simplifies or unlocks the next and can be verified independently.
5
+ 3. Name temporary states explicitly. Do not preserve a transition API as permanent architecture unless it earns a long-term role.
6
+ 4. For each phase, list input state, changes, proof, rollback/checkpoint, and dependencies.
7
+ 5. Execute and verify one phase at a time. Re-plan downstream phases when evidence invalidates the original sequence.
@@ -0,0 +1,7 @@
1
+ # Open a pull request
2
+
3
+ 1. Review the complete diff against the task and remove unrelated changes.
4
+ 2. Run targeted and required repository checks. Record exact results.
5
+ 3. Write a concise title and description centered on user/maintainer impact, root cause or design choice, and verification.
6
+ 4. Open the PR through the available VCS host only when permissions allow. Return the real URL or identifier; never invent one.
7
+ 5. Refresh status once and report any immediate blocker.
@@ -0,0 +1,8 @@
1
+ # Project orchestration
2
+
3
+ 1. Define program outcome, milestone graph, shared interfaces, acceptance gates, and integration branch/state.
4
+ 2. Partition work so each slice has one owner and minimal shared mutable state. Separate before serializing.
5
+ 3. Delegate independent slices when isolated workers exist. Otherwise sequence them and label the loss of independent verification.
6
+ 4. Require each slice to return artifacts and evidence, not a done-summary. Integrate only after owner-independent review.
7
+ 5. Continuously recompute dependencies and risk. Keep a durable decision/integration log for session pickup.
8
+ 6. Close milestones only when the integrated artifact passes its gate, then verify the program-level outcome.
@@ -0,0 +1,7 @@
1
+ # Pause safely
2
+
3
+ 1. Stop creating new work and finish or revert any half-applied local operation.
4
+ 2. Record branch/worktree, dirty files, running jobs, external state, decisions, and the last verified checkpoint.
5
+ 3. Run the cheapest sanity check that proves the paused state is coherent.
6
+ 4. Write a pickup note with exact next action and any action that must not be repeated.
7
+ 5. Do not leave irreversible external actions armed or ambiguous.
@@ -0,0 +1,8 @@
1
+ # Performance issue
2
+
3
+ 1. Define the user-visible metric and capture a baseline under controlled conditions.
4
+ 2. Profile or instrument the actual slow path. Do not optimize from source inspection alone when measurement is possible.
5
+ 3. Name the dominant cost and form one hypothesis that predicts a measurable improvement.
6
+ 4. Apply the smallest change that tests the hypothesis.
7
+ 5. Re-measure against the same baseline. Reject changes that merely move cost or regress another required metric.
8
+ 6. Report before/after numbers, environment, traces used, and remaining bottlenecks.