@jiroamato/pstack 0.15.15 → 0.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +1 -0
- package/lib/install.js +20 -2
- package/package.json +1 -1
- package/pstack/.claude-plugin/plugin.json +1 -1
- package/pstack/.codex-plugin/plugin.json +1 -1
- package/pstack/README.md +7 -0
- package/pstack/skills/deslop/SKILL.md +1 -1
- package/pstack/skills/deslop/agents/openai.yaml +1 -1
- package/pstack/skills/kiss/SKILL.md +90 -0
- package/pstack/skills/kiss/agents/openai.yaml +2 -0
- package/pstack/skills/kiss/references/assess.md +110 -0
- package/pstack/skills/kiss/references/principles.md +138 -0
package/README.md
CHANGED
|
@@ -51,6 +51,7 @@ pstack's skills are written in the Agent Skills format, so they load on both har
|
|
|
51
51
|
- Every skill that had `disable-model-invocation: true` also has an `agents/openai.yaml` with `allow_implicit_invocation: false`, so Codex treats it as user-invoked too.
|
|
52
52
|
- [`deslop`](./pstack/skills/deslop/SKILL.md), [`control-cli`](./pstack/skills/control-cli/SKILL.md), and [`control-ui`](./pstack/skills/control-ui/SKILL.md) are bundled from Cursor's team kit and adapted for both harnesses. Code-writing workflows load `deslop`; web/IDE/Electron workflows load `control-ui`, and CLI/TUI workflows load `control-cli`. Project `verify-<app>` skills supply app-specific commands and selectors alongside those drivers. There is no separate `control` skill. They use available terminal/browser tools; no separate team-kit plugin is needed.
|
|
53
53
|
- `make-bot-ui` lost its Cursor Grok Bot webhook. It now wakes a Claude Code session over a channel, Claude Code with the reply on Telegram, or a ChatGPT Dot through Slack. Cloud subagents became git worktrees.
|
|
54
|
+
- [`kiss`](./pstack/skills/kiss/SKILL.md) is the port's own addition. It assesses a repo against the principles behind Dune, Lauren Tan's agent-friendly framework, and reports what to improve, in what order, and what to leave alone. It then lands the findings through `correct` and the Refactoring playbook. See the [plugin README](./pstack/README.md#kiss).
|
|
54
55
|
|
|
55
56
|
Upstream version at the time of the port: 0.15.15.
|
|
56
57
|
|
package/lib/install.js
CHANGED
|
@@ -24,6 +24,24 @@ export function listAgents(target, pluginRoot = PLUGIN_ROOT) {
|
|
|
24
24
|
.sort();
|
|
25
25
|
}
|
|
26
26
|
|
|
27
|
+
/**
|
|
28
|
+
* mkdir -p that explains itself when a file sits where the directory should be,
|
|
29
|
+
* such as a git symlink checked out on Windows without core.symlinks.
|
|
30
|
+
*/
|
|
31
|
+
function ensureDir(dir) {
|
|
32
|
+
let stat;
|
|
33
|
+
try {
|
|
34
|
+
stat = fs.statSync(dir);
|
|
35
|
+
} catch {
|
|
36
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
37
|
+
return;
|
|
38
|
+
}
|
|
39
|
+
if (stat.isDirectory()) return;
|
|
40
|
+
const text = fs.readFileSync(dir, "utf8").trim();
|
|
41
|
+
const hint = /^[^\n]{1,200}$/.test(text) && !text.includes("\0") ? ` It is a file containing "${text}".` : "";
|
|
42
|
+
throw new Error(`${dir} exists but is not a directory.${hint} Move it aside, or make it a real directory or symlink, then re-run.`);
|
|
43
|
+
}
|
|
44
|
+
|
|
27
45
|
export function readManifest(target) {
|
|
28
46
|
if (!fs.existsSync(target.manifest)) return null;
|
|
29
47
|
return JSON.parse(fs.readFileSync(target.manifest, "utf8"));
|
|
@@ -69,14 +87,14 @@ export function install(options) {
|
|
|
69
87
|
if (!agents.includes(name)) fs.rmSync(path.join(target.agentsRoot, name + target.agentExt), { force: true });
|
|
70
88
|
}
|
|
71
89
|
|
|
72
|
-
|
|
90
|
+
ensureDir(target.skillsRoot);
|
|
91
|
+
ensureDir(target.agentsRoot);
|
|
73
92
|
for (const name of skills) {
|
|
74
93
|
const dest = path.join(target.skillsRoot, name);
|
|
75
94
|
fs.rmSync(dest, { recursive: true, force: true });
|
|
76
95
|
fs.cpSync(path.join(pluginRoot, "skills", name), dest, { recursive: true });
|
|
77
96
|
}
|
|
78
97
|
|
|
79
|
-
fs.mkdirSync(target.agentsRoot, { recursive: true });
|
|
80
98
|
for (const name of agents) {
|
|
81
99
|
const file = name + target.agentExt;
|
|
82
100
|
const text = fs.readFileSync(path.join(pluginRoot, target.agentSource, file), "utf8");
|
package/package.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "pstack",
|
|
3
3
|
"displayName": "pstack",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.16.0",
|
|
5
5
|
"description": "if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Lauren Tan"
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
{
|
|
2
2
|
"$schema": "https://agent-plugins.org/schemas/1.0.0/plugin.schema.json",
|
|
3
3
|
"name": "pstack",
|
|
4
|
-
"version": "0.
|
|
4
|
+
"version": "0.16.0",
|
|
5
5
|
"description": "if you want to go fast, go deep first. pstack helps you write less, but higher quality code. rigorous agent workflows you can parallelize with confidence.",
|
|
6
6
|
"author": {
|
|
7
7
|
"name": "Lauren Tan"
|
package/pstack/README.md
CHANGED
|
@@ -149,6 +149,7 @@ to keep [`/poteto-mode`](./skills/poteto-mode/SKILL.md) on across turns, follow
|
|
|
149
149
|
| [`/setup-pstack`](./skills/setup-pstack/SKILL.md) | you want to pick which models pstack uses per role. detects your models and writes `~/.pstack/models.md`. on codex it also installs the two pstack agents. |
|
|
150
150
|
| [`/reflect`](./skills/reflect/SKILL.md) | a long task landed and you want the recipe captured as a skill edit. |
|
|
151
151
|
| [`/correct`](./skills/correct/SKILL.md) | you keep correcting agents for the same mistakes. mines history for mistake classes, fixes each at the highest level that works (architecture, then types, lint, and ci, then tests, with docs last), and keeps a table pairing each rule with what enforces it. |
|
|
152
|
+
| [`/kiss`](./skills/kiss/SKILL.md) | you want to know how agent-friendly a repo is. assesses it against the principles behind dune (one writer per value, discovery over registries, forbidden imports fail mechanically), writes `.kiss/assessment.md` with what to improve, in what order, and what to leave alone, then lands the findings one pr at a time through `correct` and the refactoring playbook. |
|
|
152
153
|
| [`/teach`](./skills/teach/SKILL.md) | you want to actually understand a change or subsystem, not just have it summarized. runs how + why and weaves one plain explanation, built up diagram by diagram. |
|
|
153
154
|
| [`/tdd`](./skills/tdd/SKILL.md) | you're fixing a bug and there's a cheap local test path. write the failing test first, then the fix. |
|
|
154
155
|
| [`/benchmark-checklist`](./skills/benchmark-checklist/SKILL.md) | you ran a benchmark or measured a speedup or regression. vets the number (limiter, tuning, errors, repeat runs, end-to-end relevance) before you report or act on it. |
|
|
@@ -266,6 +267,12 @@ twenty-four short skills, one principle each. `poteto-mode` indexes them inline
|
|
|
266
267
|
|
|
267
268
|
`poteto-mode` and the relevant standalone skills explicitly load them: `deslop` for kept code, `control-ui` for web/IDE/Electron, and `control-cli` for CLI/TUI. there is no separate `control` skill. a project's `verify-<app>` skill supplies app-specific commands, selectors, and feature coverage alongside the bundled driver. the drivers use terminal and browser tools already available in the session; installing pstack does not install those tools.
|
|
268
269
|
|
|
270
|
+
## kiss
|
|
271
|
+
|
|
272
|
+
[`/kiss`](./skills/kiss/SKILL.md) is the port's own addition, not from upstream. it is an assessment skill built on the principles behind dune, the framework lauren tan built so agents could ship thousands of prs a month to grok bot: a correct open-file edit should preserve the whole app's invariants. `assess` reads a repo and writes `.kiss/assessment.md`: the repo's shape as a noun table, what is already right, what to improve in the order it will be landed, and what to leave alone and why. `apply` lands the findings one pr at a time, gate first.
|
|
273
|
+
|
|
274
|
+
kiss adds no machinery of its own. it links the pstack skills as siblings: `how` and `why` while assessing, `correct` for a repeated mistake class, the refactoring and opening a pr playbooks for each change, `create-verification-skill` for a repo with no scripted way to prove itself. its ladder is numbered the way `correct` numbers its levels, so a finding and a fix name the same number. on claude code, invoke `/pstack:kiss`; on codex, `$kiss`.
|
|
275
|
+
|
|
269
276
|
## harness differences
|
|
270
277
|
|
|
271
278
|
a few things upstream `poteto-mode` leaned on were cursor-only. the [`pstack-harness`](./skills/pstack-harness/SKILL.md) skill says what the port does instead:
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
---
|
|
2
2
|
name: deslop
|
|
3
3
|
description: Remove AI-generated code slop from a branch or working diff before committing. Use for /deslop, code cleanup, unnecessary comments, redundant guards, type escape hatches, or patterns that do not fit the surrounding code.
|
|
4
|
-
disable-model-invocation:
|
|
4
|
+
disable-model-invocation: false
|
|
5
5
|
---
|
|
6
6
|
|
|
7
7
|
# Remove AI code slop
|
|
@@ -1,2 +1,2 @@
|
|
|
1
1
|
policy:
|
|
2
|
-
allow_implicit_invocation:
|
|
2
|
+
allow_implicit_invocation: true
|
|
@@ -0,0 +1,90 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: kiss
|
|
3
|
+
description: "Assess a codebase against the principles behind Dune, the framework agents extend correctly by default, and report what to improve, in what order, and what to leave alone. Then apply the findings one PR at a time through pstack. Use for /kiss, 'assess this repo', 'is this repo agent-friendly', 'refactor toward KISS', or when agents keep making the same mistakes in a repo."
|
|
4
|
+
disable-model-invocation: true
|
|
5
|
+
---
|
|
6
|
+
|
|
7
|
+
# KISS
|
|
8
|
+
|
|
9
|
+
Keep It Simple, Stupid. A codebase is agent-friendly when a correct open-file edit preserves the whole app's invariants. KISS tells you how far a repo is from that and what to change first.
|
|
10
|
+
|
|
11
|
+
KISS assumes every contributor is an agent that sees only the files it opened, copies the nearest pattern, and takes the shortest path that compiles. The question behind every finding is the same: does this repo turn that behavior into good code, or into a mistake?
|
|
12
|
+
|
|
13
|
+
KISS lives in the same skills directory as pstack and links its skills as siblings (`../correct/SKILL.md`). Read [`pstack-harness`](../pstack-harness/SKILL.md) first for how to load a sibling skill (the **skill** row) and how to spawn a subagent. Where this file names a pstack skill, load its `SKILL.md` in full and follow it, with the one exception the `correct` row names.
|
|
14
|
+
|
|
15
|
+
## Modes
|
|
16
|
+
|
|
17
|
+
| Mode | When | Reads | Produces |
|
|
18
|
+
| --- | --- | --- | --- |
|
|
19
|
+
| `assess` | Any repo. The default. | `pstack-harness`, [`references/principles.md`](references/principles.md), [`references/assess.md`](references/assess.md) | `.kiss/assessment.md` |
|
|
20
|
+
| `apply` | After an assessment, on request | `pstack-harness`, the assessment, `references/principles.md` | One PR per finding, in the report's order |
|
|
21
|
+
|
|
22
|
+
`assess` is read-only. It never fixes. Re-running it regenerates the report and marks each finding new, unchanged, or resolved. The previous report is in git.
|
|
23
|
+
|
|
24
|
+
## Assess
|
|
25
|
+
|
|
26
|
+
Follow `references/assess.md`. In short: derive the repo's shape from four questions (what runs where, what outlives a run and who writes it, how two sides talk, what grows per contribution), answer the question list with file evidence, sort every answer, and write the report in the shape it gives. Every finding names the rule or anti-pattern it breaks, the contributor behavior that produces it, and the fix at the strongest level of the ladder that works.
|
|
27
|
+
|
|
28
|
+
The report has a **Leave alone** section. A principle that presumes a shape the repo does not have, something a stronger mechanism already enforces, a cost that exceeds the benefit, and a constraint the repo does not own all go there, with the reason. That section matters as much as the findings. It stops the next agent from churning. A redesign the operator has not asked for is not a Leave alone reason. It stays in Improve, marked "needs a decision", with its evidence.
|
|
29
|
+
|
|
30
|
+
pstack skills used while assessing:
|
|
31
|
+
|
|
32
|
+
| Need | Skill |
|
|
33
|
+
| --- | --- |
|
|
34
|
+
| Trace how a subsystem works, where a feature lives, who writes a value | [`how`](../how/SKILL.md) |
|
|
35
|
+
| Know why a shape exists before calling it debt | [`why`](../why/SKILL.md) |
|
|
36
|
+
| More than one package, or more than about twenty thousand lines | [`swarm`](../swarm/SKILL.md): one read-only worker per package or side, the same question list, one report. Explain the cost and ask the operator before spawning. |
|
|
37
|
+
| Write the report | [`technical-writing`](../technical-writing/SKILL.md), every layer except Diátaxis, then [`unslop`](../unslop/SKILL.md) |
|
|
38
|
+
|
|
39
|
+
## Apply
|
|
40
|
+
|
|
41
|
+
Only on request, only from an existing assessment. Small PRs on the main branch, the app shippable after each one. No big-bang branch. The report's Improve list is the plan and its order is the PR order.
|
|
42
|
+
|
|
43
|
+
0. **Confirm the list.** Show the operator the Improve list and ask which findings to land. Findings marked "needs a decision" wait for a yes. The rest proceed.
|
|
44
|
+
1. **Gate first, report-only.** One command, checked in, that runs what the repo already has in report mode plus one named check per Improve finding whose fix is level 2 or 3. Each check prints its count and nothing fails yet. The counts match the report's Count fields. This PR adds no cleanup.
|
|
45
|
+
2. **Delete debt.** The Improve findings that are deletions: copy-and-modify files, dead code, legacy paths. Agents copy what exists; remove it before they do (the **subtract-before-you-add** and **migrate-callers-then-delete-legacy-apis** principle skills). A deletion that migrates callers is a reshape and goes through the Refactoring playbook, pin first.
|
|
46
|
+
3. **Mechanical changes.** A formatter and one package manager. No behavior change. Strict compiler flags are not mechanical on a real repo. They are a ratcheted finding in step 4.
|
|
47
|
+
4. **One finding per PR, in report order.** Each PR does the cleanup and then flips that finding's check to blocking when its count is zero. If zero is far off, ratchet: fail on net-new instances above the baseline. Never flip with open violations and no exceptions. Never weaken a check to land a PR. Each finding's check was written in step 1, so no PR here adds a check.
|
|
48
|
+
5. **Re-assess.** Run `assess` again. The new report marks each finding new, unchanged, or resolved. New findings go through step 0.
|
|
49
|
+
|
|
50
|
+
After the last PR the repo needs a gardener. Delete debt as it appears. Keep one paved path per pattern. Lint against an anti-pattern the moment it shows up. Re-assess on a schedule and read the whole batch before fixing anything, because a single fix hides the pattern. Sample landed PRs. When a sample shows a shortcut, fix the environment, not the PR.
|
|
51
|
+
|
|
52
|
+
pstack skills used while applying:
|
|
53
|
+
|
|
54
|
+
| Step | Skill |
|
|
55
|
+
| --- | --- |
|
|
56
|
+
| A finding that is a repeated mistake class | [`correct`](../correct/SKILL.md). Skip its "Find the mistake classes" step. The finding is the class; pass it the evidence lines and the report's level. Follow "Fix each class at the highest level", "Fix and prove", and "Keep the rule table" for that one class. One class per PR. `correct` owns the rule table in the agent instruction file and drops a row once the mistake cannot happen. |
|
|
57
|
+
| A reshape: move, extract, split a god file, change ownership, migrate callers | the Refactoring playbook in [`poteto-mode`](../poteto-mode/SKILL.md). Pin the behavior, subtract, move in steps, prove equivalence. |
|
|
58
|
+
| A fix that crosses a module boundary | [`architect`](../architect/SKILL.md) before code |
|
|
59
|
+
| Before each PR | [`blast-radius`](../blast-radius/SKILL.md) for the one fact the change is safe because of |
|
|
60
|
+
| Each PR | the Opening a PR playbook in `poteto-mode`, which runs [`deslop`](../deslop/SKILL.md) and [`no-comments`](../no-comments/SKILL.md). Run [`interrogate`](../interrogate/SKILL.md) on the diff yourself; the playbook only runs it for subagents. |
|
|
61
|
+
| A program that spans days or many stacked PRs | the Multi-phase plan playbook in `poteto-mode`, seeded with the Improve list in report order, or [`figure-it-out`](../figure-it-out/SKILL.md) for a migration |
|
|
62
|
+
| No verification skill in the repo | [`create-verification-skill`](../create-verification-skill/SKILL.md) |
|
|
63
|
+
| A verification map that drifted | [`maintain-verification-skill`](../maintain-verification-skill/SKILL.md) |
|
|
64
|
+
| A run longer than one sitting | [`show-me-your-work`](../show-me-your-work/SKILL.md) for the decision trail |
|
|
65
|
+
| After the last PR | [`reflect`](../reflect/SKILL.md), to turn what the refactor taught into skill edits |
|
|
66
|
+
|
|
67
|
+
## Principle skills by rule
|
|
68
|
+
|
|
69
|
+
Read the leaf skill in full when a finding falls in its class. `references/principles.md` holds the rules. These hold the reasoning an agent applies while fixing. An anti-pattern id without a row uses the row of the rule it breaks.
|
|
70
|
+
|
|
71
|
+
| Rule or anti-pattern | Principle skills |
|
|
72
|
+
| --- | --- |
|
|
73
|
+
| Rule 1, fewest decisions | [**laziness-protocol**](../principle-laziness-protocol/SKILL.md), [**minimize-reader-load**](../principle-minimize-reader-load/SKILL.md) |
|
|
74
|
+
| Rule 2, forbidden dependencies fail | [**encode-lessons-in-structure**](../principle-encode-lessons-in-structure/SKILL.md) |
|
|
75
|
+
| Rule 3, one writer | [**separate-before-serializing-shared-state**](../principle-separate-before-serializing-shared-state/SKILL.md), [**model-the-domain**](../principle-model-the-domain/SKILL.md) |
|
|
76
|
+
| Rule 4, isolated files, not shared roots | [**separate-before-serializing-shared-state**](../principle-separate-before-serializing-shared-state/SKILL.md), [**model-the-domain**](../principle-model-the-domain/SKILL.md), [**type-system-discipline**](../principle-type-system-discipline/SKILL.md) for closed sets |
|
|
77
|
+
| Rule 5, narrow exceptions | [**encode-lessons-in-structure**](../principle-encode-lessons-in-structure/SKILL.md) |
|
|
78
|
+
| `raw-transport` | [**boundary-discipline**](../principle-boundary-discipline/SKILL.md) |
|
|
79
|
+
| `god-file` | [**minimize-reader-load**](../principle-minimize-reader-load/SKILL.md) |
|
|
80
|
+
| `type-escape-hatch` | [**type-system-discipline**](../principle-type-system-discipline/SKILL.md) |
|
|
81
|
+
| `tautological-test` | [**test-behavior-not-implementation**](../principle-test-behavior-not-implementation/SKILL.md) |
|
|
82
|
+
| `comments-as-bandaids` | [**fix-root-causes**](../principle-fix-root-causes/SKILL.md) |
|
|
83
|
+
| `copy-and-modify` | [**migrate-callers-then-delete-legacy-apis**](../principle-migrate-callers-then-delete-legacy-apis/SKILL.md) |
|
|
84
|
+
| Ordering the PRs | [**sequence-verifiable-units**](../principle-sequence-verifiable-units/SKILL.md), [**prove-it-works**](../principle-prove-it-works/SKILL.md) |
|
|
85
|
+
|
|
86
|
+
## Rules that hold in both modes
|
|
87
|
+
|
|
88
|
+
- Prose is the last resort. If a rule can be a type, a build failure, a lint, or a test, it is not a sentence in the agent instruction file. A rule that lives only in a doc is unenforced and the report says so.
|
|
89
|
+
- An exception lives on the offending line as `kiss-allow(<finding id>): <why>. <issue link>. expires <date>. approved: <who>`. The issue link is what lets it survive `no-comments`, whose reviewer keeps issue links that explain a constraint. If the reviewer still kills it, follow `no-comments` step 5: encode the constraint or report it open. Never restore the comment. An agent never approves its own exception. A third live exception on one rule means the rule or the design is wrong.
|
|
90
|
+
- Every operator correction that names a repo mistake is a signal. In `apply`, route it through `correct`. In `assess`, record it as a finding with evidence and a proposed level. A correction about scope goes under Not checked; one that names a deliberate design goes under Leave alone.
|
|
@@ -0,0 +1,110 @@
|
|
|
1
|
+
# Assess
|
|
2
|
+
|
|
3
|
+
Produce `.kiss/assessment.md`. Read-only. Every finding has `path:line` evidence, or for an absence, the location searched. No fixes.
|
|
4
|
+
|
|
5
|
+
When the target is a folder inside a larger repository, say which one you assessed, put `.kiss/` at that folder's root, and scope git history to that path. When that history is a bulk import, widen to the git root and say so.
|
|
6
|
+
|
|
7
|
+
If `.kiss/assessment.md` already exists, read it first. The new report replaces it; git keeps the old one. Each finding carries a stable id, `<rule or anti-pattern id>:<path>`, so the new report can mark it new, unchanged, or resolved.
|
|
8
|
+
|
|
9
|
+
## Before the questions
|
|
10
|
+
|
|
11
|
+
Read `principles.md`. Then build the model the questions need:
|
|
12
|
+
|
|
13
|
+
- For a repo you can hold in context, read the tree, the package manifests, the CI config, and the agent instruction files. Pick the most recent unit of product work from git history and read its diff.
|
|
14
|
+
- For a subsystem you cannot judge from the tree, run the **how** skill on it. Its Where Things Live section is the start of the noun table.
|
|
15
|
+
- When a shape looks wrong but deliberate, run the **why** skill before calling it debt. A constraint the repo does not own goes under Leave alone, not Improve.
|
|
16
|
+
- For more than one package, or more than about twenty thousand lines, run the **swarm** skill: one read-only worker per package or side, each answering the full question list for its slice, one report back. Explain the cost and ask the operator first. Merge the noun tables yourself.
|
|
17
|
+
|
|
18
|
+
## Questions
|
|
19
|
+
|
|
20
|
+
Answer in order. Status is `pass`, `fail`, `partial`, `n/a (<the shape the question presumes>)`, or `unknown (<what you could not see>)`. One answer may yield several entries in the report.
|
|
21
|
+
|
|
22
|
+
**Shape** (principles.md section 3)
|
|
23
|
+
|
|
24
|
+
1. Where does code run, and can an agent tell from the open file which boundary it is in and what it may import? Does a wrong import fail?
|
|
25
|
+
2. Which values outlive a request or a run? For each, name every writer. Where a value lives on two sides, which side is authoritative?
|
|
26
|
+
3. Where two boundaries talk, is there one typed contract both sides import or generate from, so a change on one side fails the build on the other?
|
|
27
|
+
4. Which sets grow when someone contributes? For each, open or closed, and how it is populated today. For a closed set, does every runtime switch over it fail the build on a missing variant?
|
|
28
|
+
|
|
29
|
+
Write the noun table from these answers before going on. The table is the target. Each row where today differs becomes a finding.
|
|
30
|
+
|
|
31
|
+
**Rules** (principles.md section 2)
|
|
32
|
+
|
|
33
|
+
5. Rule 1. From git history, list the files the most recent unit of product work changed. Which were shared files? Is the blessed path the shortest one?
|
|
34
|
+
6. Rule 2. Which boundaries does the repo claim, in docs, folder names, or comments? Which are enforced mechanically, by what, and does the error name the owner?
|
|
35
|
+
7. Rule 3. From question 2, which values have more than one writer? Include specs restated in code.
|
|
36
|
+
8. Rule 4. From question 4, which open sets are hand-edited lists?
|
|
37
|
+
9. Rule 5. Every suppression, disabled rule, `TODO`, and workaround comment. Which carry a reason, an issue link, an expiry, and a human's approval? Expect-error directives that assert rejection are tests, not suppressions.
|
|
38
|
+
|
|
39
|
+
**Guardrails** (principles.md section 4)
|
|
40
|
+
|
|
41
|
+
10. What runs before merge, and is it one command or several? Formatter, linter, type checker, import guard, tests. What is missing, and what would each missing one catch in this repo? If CI lives outside the checkout, say so and answer `unknown`.
|
|
42
|
+
11. Type strictness: compiler flags, checker mode, which files they cover, and what they ignore.
|
|
43
|
+
12. For each anti-pattern id in principles.md section 5: present or absent, count, sanctioned hits excluded, enforced today or not.
|
|
44
|
+
|
|
45
|
+
**Verification** (principles.md section 6)
|
|
46
|
+
|
|
47
|
+
13. Is there a project verification skill? Look for `verify-*` under `.claude/skills/` or `.agents/skills/` at the project root. Does its feature map cover every entrypoint the repo has: routes, commands, jobs? Name any drift.
|
|
48
|
+
|
|
49
|
+
**Debt**
|
|
50
|
+
|
|
51
|
+
14. Files over the size budget, with line counts and, from git log, how many distinct units of work edited each. Use the repo's budget if it has one, else 400. Tests count.
|
|
52
|
+
15. Legacy paths, dead code, and copy-and-modify files an agent would copy. For dead code use an unused-export check or the compiler's unused flags; a grep for callers is the fallback and say so.
|
|
53
|
+
|
|
54
|
+
## Sorting the answers
|
|
55
|
+
|
|
56
|
+
Sort every answer. A passing answer may go under Already right. Any other answer goes under Improve or Leave alone, and one answer may produce entries in both.
|
|
57
|
+
|
|
58
|
+
**Improve** holds three kinds of finding.
|
|
59
|
+
|
|
60
|
+
- A rule or anti-pattern finding: one of the five contributor behaviors produces the mistake, the repo owns the code, and a fix exists at level 1, 2, or 3 of the ladder. One label per finding. When a rule and an id describe the same evidence, use the id and name the rule in the title.
|
|
61
|
+
- A guardrail gap from questions 10 or 11: a missing formatter, linter, type checker, import guard, or strict mode. The fix is level 2. No behavior needed.
|
|
62
|
+
- A verification gap from question 13: no skill, or drift. The fix is level 3. No behavior needed.
|
|
63
|
+
|
|
64
|
+
A redesign the operator has not asked for stays here with all fields and "needs a decision" after the title. `apply` skips it until the operator says yes.
|
|
65
|
+
|
|
66
|
+
Order Improve the way `apply` will land it: deletions first, then mechanical changes, then everything else by risk, then by effort. Risk is how likely an agent is to copy the pattern: `high` when it sits on the path every new unit of work takes, `medium` when it is in a module agents touch sometimes, `low` when it is isolated. Effort is `S` for one PR under an hour, `M` for one PR, `L` for several PRs.
|
|
67
|
+
|
|
68
|
+
**Leave alone** when any of these holds. Say which one.
|
|
69
|
+
|
|
70
|
+
- The question presumes a shape the repo does not have. The `n/a` answers land here with the presumed shape as the reason.
|
|
71
|
+
- Something stronger already enforces it. A doc rule that a type also makes impossible is fine as a doc.
|
|
72
|
+
- The cost exceeds the benefit. A single-writer module over the size budget whose every section is about one thing. A three-entry registry that has not grown in a year.
|
|
73
|
+
- An external constraint owns it: a vendor API, a platform, a file the framework requires, polling a service that offers no events.
|
|
74
|
+
|
|
75
|
+
**Already right** when the answer passes and agents are likely to copy it. Name it so the next agent protects it.
|
|
76
|
+
|
|
77
|
+
## Report shape
|
|
78
|
+
|
|
79
|
+
```markdown
|
|
80
|
+
# KISS assessment: <repo> <date>
|
|
81
|
+
|
|
82
|
+
## Shape
|
|
83
|
+
<Three to six lines: what runs where, which durable values exist and who writes them, how the sides talk, which sets grow per contribution.>
|
|
84
|
+
|
|
85
|
+
| Noun | Job | Lives at | May import | Found by |
|
|
86
|
+
| --- | --- | --- | --- | --- |
|
|
87
|
+
|
|
88
|
+
## Already right
|
|
89
|
+
- <what> (<path>). <Why it matters that agents copy it.>
|
|
90
|
+
|
|
91
|
+
## Improve
|
|
92
|
+
### <n>. <title> (<id>, <new | unchanged | resolved>)
|
|
93
|
+
Rule or anti-pattern: <rule 1 to 5, an id, guardrail gap, or verification gap>
|
|
94
|
+
Behavior: <1 to 5, or none for a gap>
|
|
95
|
+
Evidence:
|
|
96
|
+
- <path:line> <what>
|
|
97
|
+
Count: <number of hits, or 1>
|
|
98
|
+
Enforcement today: <none, or the tool>
|
|
99
|
+
Fix: <what> at level <1 to 3>. <Why a stronger level does not work.>
|
|
100
|
+
Risk: <high | medium | low>
|
|
101
|
+
Effort: <S | M | L>
|
|
102
|
+
|
|
103
|
+
## Leave alone
|
|
104
|
+
- <what> (<path>): <which reason above>
|
|
105
|
+
|
|
106
|
+
## Not checked
|
|
107
|
+
- Q<n>: <what you could not answer, and why>
|
|
108
|
+
```
|
|
109
|
+
|
|
110
|
+
Write it with every technical-writing layer except Diátaxis, then **unslop**. Keep it under 300 lines. A finding that needs a paragraph is two findings.
|
|
@@ -0,0 +1,138 @@
|
|
|
1
|
+
# Principles
|
|
2
|
+
|
|
3
|
+
The ideas behind Dune, the framework Lauren Tan built so agents could ship thousands of PRs a month to Grok Bot. Stack-agnostic. Nothing here names a tool. The assessment names the tool a given repo should use.
|
|
4
|
+
|
|
5
|
+
## 1. The contributor model
|
|
6
|
+
|
|
7
|
+
A coding agent optimizes for what fits in its context. It will, predictably:
|
|
8
|
+
|
|
9
|
+
1. copy the nearest working pattern;
|
|
10
|
+
2. edit the file already open;
|
|
11
|
+
3. choose the shortest path that compiles;
|
|
12
|
+
4. avoid deleting code whose callers are not visible;
|
|
13
|
+
5. follow the requested implementation even when it conflicts with a system invariant.
|
|
14
|
+
|
|
15
|
+
These are inputs to the design, not faults to prompt away. The codebase is the agent's memory. Whatever pattern exists will be copied, including workarounds, and each copy makes the next copy likelier. Keep the repo in a state where you would be happy for any file to be copied. Human friction is not a cost here. A repo this constrained is annoying for humans to write in, and agents absorb the annoyance.
|
|
16
|
+
|
|
17
|
+
Every rule and anti-pattern finding names which of the five behaviors produces the mistake. A guardrail or verification gap (sections 4 and 6) does not need one.
|
|
18
|
+
|
|
19
|
+
## 2. The five rules
|
|
20
|
+
|
|
21
|
+
Dune's rules, verbatim. Each has the behavior it addresses, the test question the assessment asks, and one example.
|
|
22
|
+
|
|
23
|
+
### Rule 1. The conventional path requires fewer decisions than a shortcut
|
|
24
|
+
|
|
25
|
+
Behavior 3. If the shortcut is easier, the shortcut wins.
|
|
26
|
+
|
|
27
|
+
Test: how many files must change to add one unit of product work, and how many of them are shared? A unit of product work is one merged change that adds user-visible behavior: a page, a command, an endpoint, a job. A shared file is one that two or more such units changed.
|
|
28
|
+
|
|
29
|
+
Violating: to add a page, create the view, then add a route to `routes.ts`, a key to `keys.ts`, a type to `types.ts`, a nav item to `nav.ts`, and a permission to `permissions.ts`. Six decisions, five shared files. An agent skips two and the page half-works.
|
|
30
|
+
|
|
31
|
+
Conforming: create `features/reports/entrypoint.*` and `features/reports/view.*`. The build discovers both. Two files, zero shared edits.
|
|
32
|
+
|
|
33
|
+
### Rule 2. Forbidden dependencies fail mechanically
|
|
34
|
+
|
|
35
|
+
Behaviors 1 and 3. A rule that lives only in a doc is unenforced.
|
|
36
|
+
|
|
37
|
+
Test: for each boundary the repo claims, what rejects a crossing, and does the error name what to use instead?
|
|
38
|
+
|
|
39
|
+
Violating: the agent instruction file says views do not call the API. `view.tsx` imports `api/client` and compiles.
|
|
40
|
+
|
|
41
|
+
Conforming: the same import fails with `features/* may not import api/*. Read data through client/reports.`
|
|
42
|
+
|
|
43
|
+
### Rule 3. Every durable value has one obvious writer
|
|
44
|
+
|
|
45
|
+
Behaviors 1 and 2. Three writers mean three sets of rules that drift apart.
|
|
46
|
+
|
|
47
|
+
Test: for each value that outlives a request or a run, name every writer. A spec restated in code (a template, rule text, a name list that another file also holds) is a durable value too, and one copy is the writer.
|
|
48
|
+
|
|
49
|
+
Violating: `view.tsx`, `nav.tsx`, and `url-sync.ts` each write `selectedReportId`.
|
|
50
|
+
|
|
51
|
+
Conforming: one module owns it and exposes `useSelectedReport()` and `selectReport(id)`. Everyone else calls the command.
|
|
52
|
+
|
|
53
|
+
### Rule 4. New product work adds isolated files rather than branches in shared roots
|
|
54
|
+
|
|
55
|
+
Behaviors 1 and 2. Two agents adding two features must never touch the same file.
|
|
56
|
+
|
|
57
|
+
Test: does any shared file grow when a unit of work is added?
|
|
58
|
+
|
|
59
|
+
Violating: `registry.ts` with one line per feature, which every feature edits.
|
|
60
|
+
|
|
61
|
+
Conforming: one folder per feature with a reserved filename. A build step collects them into a catalog that is derived, never hand-edited.
|
|
62
|
+
|
|
63
|
+
### Rule 5. Exceptions are narrow, explicit, and reviewed as architecture changes
|
|
64
|
+
|
|
65
|
+
Behavior 5. Every guard has an escape hatch, and the hatch has a cost.
|
|
66
|
+
|
|
67
|
+
Test: does every suppression, disabled rule, and workaround carry a reason, an issue link, an expiry, and a human's approval, on the offending line? An approval an agent wrote is a fail. An expect-error directive whose purpose is to assert that a type rejects a value is a test, not a suppression.
|
|
68
|
+
|
|
69
|
+
Format, in whatever comment syntax the language has: `kiss-allow(<finding id>): <why>. <issue link>. expires <date>. approved: <who>`. A third live exception on one rule means the rule or the design is wrong. Fix one of them.
|
|
70
|
+
|
|
71
|
+
## 3. Deriving the shape
|
|
72
|
+
|
|
73
|
+
Dune names its nouns for an Electron app: Feature, Entrypoint, Client, Host, and a typed edge between processes. Those names fit a UI with durable local state and an always-on backend. They do not fit a CLI or a library, and forcing them onto one is invention. What is universal is the four questions the nouns answer. Ask them of any repo:
|
|
74
|
+
|
|
75
|
+
1. **Where does code run, and can an agent tell from the open file?** Each process, deployable, package, or layer is a boundary. The folder, the filename, or a directive in the file (`"use client"`, `server-only`) must tell an agent which boundary the code is in and what it may import, and a wrong import must fail. Shared code is a leaf: it imports nothing above it.
|
|
76
|
+
2. **Which values outlive a request or a run, and who writes each one?** Stores, caches, tables, files on disk, config. One writer per value, behind a named interface. Where a value lives on two sides (a server row and a client cache), one side is authoritative and the other writes only through one named command.
|
|
77
|
+
3. **Where two boundaries talk, is there one typed contract both sides import or generate from?** A change to the contract on one side must fail the build on the other. Not applicable to a single-process repo.
|
|
78
|
+
4. **Which sets grow when someone contributes?** Routes, commands, jobs, registries, type mirrors, test lists. See open and closed sets below.
|
|
79
|
+
|
|
80
|
+
The answers are the repo's own noun table: noun, its one job at runtime, where it lives, what it may import, how it is found. A web app's table looks like Dune's. A CLI's table has commands, one state store, and one adapter per external tool. Write the table the repo needs, not the one Dune had. The table describes the target. Each row where today's repo differs is a finding that cites the row.
|
|
81
|
+
|
|
82
|
+
Two properties every table keeps:
|
|
83
|
+
|
|
84
|
+
- **Reserved filenames make a folder a complete contribution.** Dropping a folder with the right filenames into the tree is the whole change. The build discovers it, startup validates it, nothing shared is edited. For a CLI: one file per command group that exports a register function, collected by a glob, or by a hand list plus a check that every file in the folder is in it.
|
|
85
|
+
- **The side that owns durable state imports no presentation code.** Whatever accepts a write knows nothing about views or output formatting.
|
|
86
|
+
|
|
87
|
+
### Open sets are discovered, closed sets are unions
|
|
88
|
+
|
|
89
|
+
Two kinds of "list of things" exist.
|
|
90
|
+
|
|
91
|
+
- An **open set** has independent contributors who should never collide: features, routes, commands, jobs. Discover them from reserved filenames. No shared edit, validated at build or start.
|
|
92
|
+
- A **closed set** belongs to one owner and must be handled exhaustively: a feature's states, its commands, its event kinds. A discriminated union in one file, so the compiler forces every consumer to handle every variant. The shared edit is the point. Every runtime list or switch over a closed set fails the build when a variant is missing, through a `never` check or a `satisfies Record` check, never a default branch.
|
|
93
|
+
|
|
94
|
+
Ask: can two agents add to this set in parallel without talking? If yes, discover it. If adding a variant must break every consumer that forgot it, union it.
|
|
95
|
+
|
|
96
|
+
## 4. The ladder
|
|
97
|
+
|
|
98
|
+
When an agent makes a mistake, encode the correction at the strongest level that works. These are the four levels pstack's `correct` skill climbs, numbered the same way, so a KISS finding and a `correct` fix name the same number.
|
|
99
|
+
|
|
100
|
+
| Level | Mechanism | What happens when an agent skips it |
|
|
101
|
+
| --- | --- | --- |
|
|
102
|
+
| 1 | Architecture and ownership. One owner per value, one supported way per task, internals hidden so the wrong import fails, one source of truth, old ways deleted | It cannot be written |
|
|
103
|
+
| 2 | Types that make the bad state unwritable. If bad code still compiles, a compiler flag, lint, or build check whose error names the file, type, or function to use instead | The check goes red |
|
|
104
|
+
| 3 | A test of the behavior that fails when the behavior breaks. A project verification skill that drives the real app counts here | CI goes red |
|
|
105
|
+
| 4 | Docs, agent rules, skills, review | Nothing. Agents forget them and busy humans skip them |
|
|
106
|
+
|
|
107
|
+
Rules of use:
|
|
108
|
+
|
|
109
|
+
- A human catching the same thing in review is the signal that a level is missing. The fix is a check or a redesign, not another comment on the PR.
|
|
110
|
+
- Write the check before the cleanup. It stops the bleeding while the debt stays. If the pattern is already common, fail only on net-new instances.
|
|
111
|
+
- Prefer the stack's own strictness first. Strict compiler flags are level 2 for free, but turning them on in a live repo is a ratchet, not a mechanical change.
|
|
112
|
+
- Operator corrections during `apply`: a mistake class counts once it has happened twice. Record the first, act on the second. Assessment findings are already evidence of a pattern and do not wait for a second instance.
|
|
113
|
+
|
|
114
|
+
## 5. Anti-patterns
|
|
115
|
+
|
|
116
|
+
Things agents are reliably bad at. Each has a stable id that findings cite. The replacement is a shape, not a tool. The assessment names the tool for the repo in hand. When counting, a hit that is the replacement pattern itself (a cast right after a runtime check, an expect-error that asserts rejection, polling a vendor with backoff) is sanctioned; say how you decided.
|
|
117
|
+
|
|
118
|
+
| Id | Pattern | Behavior | Replacement |
|
|
119
|
+
| --- | --- | --- | --- |
|
|
120
|
+
| `comments-as-bandaids` | A comment that justifies a workaround, or cites review feedback as the reason the code is shaped this way. A constraint the code cannot show may stay. A reference to a past PR or bug alone moves to the commit message | 5 | A better name, a type, a test, or an issue link in the commit. Enforced by a lint that allows license headers, public API docs, and `kiss-allow` lines, with `no-comments` on top for judgment |
|
|
121
|
+
| `hand-edited-registry` | A list, switch, or registry file that grows per feature | 1 | Discovery for open sets, a union for closed sets |
|
|
122
|
+
| `cross-boundary-import` | An import across a process or layer boundary the repo claims | 3 | Shared code, or the typed contract between sides |
|
|
123
|
+
| `raw-transport` | HTTP, RPC, or subprocess calls scattered outside one adapter | 3 | One adapter per external system, behind named functions |
|
|
124
|
+
| `second-writer` | A durable value written from more than one module | 2 | The owner's named command |
|
|
125
|
+
| `god-file` | A file past the size budget (400 lines unless the repo sets its own; tests count) that several units of work edit or that no agent reads fully | 2 | A split along the repo's nouns, behind one writer if the file was one |
|
|
126
|
+
| `timer-as-sync` | A sleep or timer used to wait for state the code could observe | 3 | Explicit events, readiness checks, request keys |
|
|
127
|
+
| `type-escape-hatch` | `any`, double casts, `type: ignore`, untyped contracts | 3 | A real type, a branded id, a parse at the boundary |
|
|
128
|
+
| `suppression-without-reason` | A disabled rule or suppression with no reason, issue link, expiry, or approval | 3 | Fix the code, or a Rule 5 exception |
|
|
129
|
+
| `copy-and-modify` | `thing_v2`, `thing-new`, a parallel old and new path, a helper pasted into several files | 4 | Migrate callers, then delete the old copy in the same wave |
|
|
130
|
+
| `tautological-test` | A test that would still pass if every function it calls returned nothing: no assertion on a literal result, only mock-call assertions, a constant restated, a fixture asserting itself | 3 | A test that calls the public surface and asserts a literal result |
|
|
131
|
+
|
|
132
|
+
Dune also bans React's `useEffect` and fails CI on it. That ban belongs in a web repo's rule table, not in this list.
|
|
133
|
+
|
|
134
|
+
## 6. Verification
|
|
135
|
+
|
|
136
|
+
Guardrails prove the code is shaped right. Verification proves it works. Every serious project ships a verification skill: a CLI that launches the real app, drives it the way a user would, and captures evidence, plus a feature map that says what exists and how a user reaches it. Deterministic work lives in the CLI so no agent rebuilds it per session. The map is kept current by an automation, not by memory.
|
|
137
|
+
|
|
138
|
+
Drift is an entrypoint (a route, a command, a job) with no map entry, or a map entry whose route, selector, or command no longer exists. Drift is a finding. The fix is pstack's `create-verification-skill` when no skill exists and `maintain-verification-skill` when one does. Both count as level 3.
|